// Copyright (c) 未来飞马 // // This Source Code Form is subject to the terms of the Mozilla Public // License, v. 2.0. If a copy of the MPL was not distributed with this // file, You can obtain one at https://mozilla.org/MPL/2.0/. // // Trademark Notice: // The MPL-2.0 license grants copyright permissions for source code only. // It does NOT grant any rights to use trademarks including "未来飞马", // "Harness Loop", "RSI", and associated slogan "让AI进化提前发生,让AI落地快人一步". // Any use of these trademarks requires separate written permission. /** * 图片理解桥(复用兄弟技能 skill-vision) * * 职责:把一张图交给 skill-vision 的 analyze(),产出结构化结果: * - ocrText 图里出现的文字(尽量逐字) * - description 画面/内容语义描述 * - usageSuggestion 能不能作为素材、适合发给谁 * - label 给素材起一个短标签 * - canBeMaterial 是否可作为可发送素材(false → role=description) * - piiHints 疑似 PII 片段(姓名/手机/微信/邮箱/头像),仅记录不外发 * * analyze() 有两种返回: * 1) 宿主多模态:{ provider:'host', instruction, imagePath } —— 由本脚本打印指令, * 由宿主 Agent 用自己的 Read 工具读图后按提示词产出 JSON(不调 Fmode API); * 2) 网关:{ provider:'fmode', raw, parsed } —— 直接拿到结构化 JSON。 * 两条路径的输出契约一致,调用方(SKILL.md 工作流)无感知。 * * 绝不打印 token、绝不把图片内容写进仓库。 */ import fs from 'node:fs'; import path from 'node:path'; import { parseArgs } from 'node:util'; import { readJson, writeJson, ensureDir, resolveSiblingScript, truncate, shortHash, } from './lib.mjs'; const VISION_REL = path.join('skill-vision', 'scripts', 'vision-client.mjs'); const SYSTEM_PROMPT = [ '你是案例库采集助手。你看到的图片来自留学咨询/课程辅导机构的真实沟通素材(聊天截图、成绩单、反馈截图、海报、笔记等)。', '你要做两件事:① 如实 OCR 出图里的文字;② 判断这张图能不能作为「可发给客户的案例素材」,并给出使用建议。', '严格输出 JSON,不要输出多余文字。字段:', '{"ocrText":"图内文字,逐字,保留换行","description":"画面与内容说明,2-4 句",', ' "label":"不超过 16 字的短标签","usageSuggestion":"什么时候发给什么客户,一句话",', ' "canBeMaterial":true, "role":"material|description",', ' "piiHints":[{"field":"姓名|手机号|邮箱|微信号|头像|其它","snippet":"疑似片段"}]}', '判据:canBeMaterial=false 的典型情况——纯说明性配图、logo、无信息量的装饰图、与课程辅导无关。', 'piiHints 只记录疑似片段,不要脑补补全。', ].join('\n'); export function buildUserPrompt(context = {}) { const lines = [ '请分析这张图片,按系统提示词输出 JSON。', context.index ? `这是同一批素材中的第 ${context.index} 张${context.total ? `(共 ${context.total} 张)` : ''}。` : '', context.batchHint ? `同批其它图的初步内容:${truncate(context.batchHint, 300)}。请据此判断这张图在整组里的作用。` : '', context.extra ? String(context.extra) : '', ].filter(Boolean); return lines.join('\n'); } /** 从 vision 返回里取出结构化对象(宿主路径拿不到时回落为一个空壳)。 */ export function normalizeVisionResult(result, fallbackLabel = '') { const parsed = (result && result.parsed) || null; if (!parsed) { const needsHostRead = Boolean(result && result.instruction); return { ok: false, provider: result && result.provider, model: result && result.model, ocrText: '', description: '', label: fallbackLabel, usageSuggestion: '', // 还没真正分析过:不下结论,role 留空交给调用方保持默认(material) canBeMaterial: null, role: '', piiHints: [], needsHostRead, instruction: result && result.instruction, raw: (result && result.raw) || '', error: (result && result.error) || (needsHostRead ? 'HOST_READ_PENDING' : 'NO_PARSED_OUTPUT'), }; } const role = String(parsed.role || (parsed.canBeMaterial === false ? 'description' : 'material')).toLowerCase(); return { ok: true, provider: result && result.provider, model: result && result.model, ocrText: String(parsed.ocrText || ''), description: String(parsed.description || ''), label: String(parsed.label || fallbackLabel || ''), usageSuggestion: String(parsed.usageSuggestion || ''), canBeMaterial: parsed.canBeMaterial !== false && role === 'material', role: role === 'description' ? 'description' : 'material', piiHints: Array.isArray(parsed.piiHints) ? parsed.piiHints.filter((h) => h && h.field) : [], needsHostRead: false, instruction: '', raw: result && result.raw ? String(result.raw) : '', error: null, }; } /** * 分析一张图(或一组图)。 * @returns {Promise<{available:boolean, source:string, skillPath:string|null, results:object[], error:string|null}>} */ export async function analyzeImages(items, options = {}) { const visionPath = resolveSiblingScript('CASE_VISION_SCRIPT', VISION_REL); if (!visionPath) { return { available: false, source: 'missing', skillPath: null, results: [], error: `未找到 skill-vision(期望 ${VISION_REL})。请先安装:npx skill-vision@latest install`, }; } const analyze = (await import(pathToFileUrl(visionPath))).analyze; const results = []; for (let i = 0; i < items.length; i++) { const item = items[i]; const context = { index: i + 1, total: items.length, batchHint: options.batchHint || '' }; let result; try { result = await analyze({ imagePath: item.localPath || undefined, imageUrl: !item.localPath ? item.url || undefined : undefined, systemPrompt: SYSTEM_PROMPT, userPrompt: buildUserPrompt(context), model: options.model || undefined, maxTokens: options.maxTokens || 1800, }); } catch (error) { result = { provider: 'error', parsed: null, error: error.message }; } const normalized = normalizeVisionResult(result, item.label || `素材${i + 1}`); results.push({ file: item.localPath || item.url || '', index: i + 1, ...normalized }); } const provider = results.find((r) => r.ok)?.provider || results[0]?.provider || 'unknown'; const needsHost = results.some((r) => r.needsHostRead); return { available: true, source: provider, skillPath: visionPath, needsHostRead: needsHost, results, error: results.every((r) => !r.ok) ? '全部图片分析未产出结构化结果' : null, }; } function pathToFileUrl(p) { return new URL(`file://${p.split(path.sep).join('/')}`).href; } // --------------------------------------------------------------------------- // CLI // --------------------------------------------------------------------------- async function main() { const { values } = parseArgs({ options: { image: { type: 'string', multiple: true, default: [] }, 'image-list': { type: 'string' }, out: { type: 'string' }, model: { type: 'string' }, 'batch-hint': { type: 'string' }, help: { type: 'boolean', default: false }, }, allowPositionals: true, }); if (values.help) { process.stdout.write([ 'skill-case-get / vision-bridge — 逐图 OCR + 语义理解(复用 skill-vision)', '', ' node vision-bridge.mjs --image a.jpg --image b.jpg [--batch-hint "同批都是考前冲刺截图"] [--out vision.json]', ' node vision-bridge.mjs --image-list batch.json [--out vision.json] # batch.json: {items:[{localPath|url,label}], batchHint}', '', '说明:若宿主(FmodeCode / Claude Code)自带多模态模型,analyze() 会返回读图指令,', ' 此时输出里的 needsHostRead=true,由宿主 Agent 用自己的 Read 工具读图后补全。', ].join('\n')); return 0; } const items = values.image.map((p) => ({ localPath: p })); if (values['image-list']) { const bundle = readJson(values['image-list'], {}); for (const item of bundle.items || []) items.push(item); if (bundle.batchHint && !values['batch-hint']) values['batch-hint'] = bundle.batchHint; } if (!items.length) { process.stderr.write('至少需要一个 --image 或 --image-list\n'); return 2; } const payload = await analyzeImages(items, { model: values.model, batchHint: values['batch-hint'] }); const text = JSON.stringify(payload, null, 2); if (values.out) { ensureDir(path.dirname(path.resolve(values.out))); writeJson(path.resolve(values.out), payload); } else { process.stdout.write(`${text}\n`); } if (!payload.available) return 3; return 0; } const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]).endsWith(path.join('scripts', 'vision-bridge.mjs')); if (invokedDirectly) { main().then((code) => process.exit(code)).catch((error) => { process.stderr.write(`vision-bridge 失败:${error.message}\n`); process.exit(1); }); } export { SYSTEM_PROMPT };