commit b0699f2f0786c11090fc8ff089396c018bf42e25 Author: Maksim Totmin Date: Tue Aug 4 20:25:11 2026 +0700 feat: Initial commit of vision-bridge plugin and agent diff --git a/agents/vision.md b/agents/vision.md new file mode 100644 index 0000000..6273e76 --- /dev/null +++ b/agents/vision.md @@ -0,0 +1,18 @@ +--- +description: Анализ изображений и скриншотов (OCR текста или разбор UI) +mode: subagent +model: ludmila-ai/gemini/gemini-2.5-flash +temperature: 0.1 +permission: + read: allow + glob: allow + external_directory: allow + edit: deny + bash: deny +--- +Ты анализируешь изображение. Сначала определи тип контента: +1) Преимущественно ТЕКСТ/документ → извлеки текст ДОСЛОВНО, сохрани структуру + (заголовки, списки, колонки, код). Не пересказывай и не исправляй. +2) ЭЛЕМЕНТЫ ИНТЕРФЕЙСА (веб/моб/десктоп) → опиши: тип экрана, компоненты и + расположение, тексты кнопок/полей, состояния, интерактивность, стили. +3) Смешанное → сделай оба пункта раздельно. diff --git a/src/vision-bridge.ts b/src/vision-bridge.ts new file mode 100644 index 0000000..84cd1a1 --- /dev/null +++ b/src/vision-bridge.ts @@ -0,0 +1,280 @@ +import type { Plugin, Hooks } from "@opencode-ai/plugin" + +const VISION_PROMPT = `You are a vision assistant. Analyze the image in detail and produce a thorough text representation. + +Rules: +1. If the image is predominantly TEXT (code, error message, document, chat, log, terminal output) — transcribe the text VERBATIM. Preserve structure (headings, lists, columns, indentation). Do not paraphrase or "fix" typos. +2. If the image shows a UI / INTERFACE (web, mobile, desktop) — describe: type of screen, components and their layout, exact button/label/field texts, states, interactivity, visual style. +3. If mixed — do BOTH, in separate labeled sections. +4. If the image is a graph/diagram/chart — describe axes, labels, values, and the overall message. + +Be precise and exhaustive. This text will be given to another model that cannot see the image.` + +type ImagePart = { + id: string + sessionID: string + messageID: string + type: "file" + mime: string + filename?: string + url: string +} + +type TextPart = { + id: string + sessionID: string + messageID: string + type: "text" + text: string + synthetic?: boolean +} + +type AnyPart = { type: string; [key: string]: any } + +interface ProviderInfo { + baseURL: string + apiKey: string + visionModel: string +} + +const IMAGE_MIME_RE = /^image\// + +function isVisionModel(modelID: string | undefined, visionModels: Set): boolean { + if (!modelID) return false + // Normalize: strip provider prefix if present, and also handle "provider/model" form + for (const m of visionModels) { + if (modelID === m) return true + if (modelID.endsWith("/" + m)) return true + if (m.endsWith("/" + modelID)) return true + } + return false +} + +function mimeToDataUrl(url: string, mime: string): string { + // url is already data:image/png;base64,... + if (url.startsWith("data:")) return url + return `data:${mime};base64,${url}` +} + +async function recognizeImage( + provider: ProviderInfo, + modelID: string, + dataUrl: string, + timeoutMs = 60000, +): Promise { + const controller = new AbortController() + const timer = setTimeout(() => controller.abort(), timeoutMs) + try { + const res = await fetch(`${provider.baseURL}/chat/completions`, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${provider.apiKey}`, + }, + body: JSON.stringify({ + model: modelID, + messages: [ + { + role: "user", + content: [ + { type: "text", text: VISION_PROMPT }, + { type: "image_url", image_url: { url: dataUrl } }, + ], + }, + ], + max_tokens: 4096, + temperature: 0.1, + }), + signal: controller.signal, + }) + if (!res.ok) { + const body = await res.text().catch(() => "") + throw new Error(`Vision API ${res.status}: ${body.slice(0, 300)}`) + } + const data = await res.json() + const text = data?.choices?.[0]?.message?.content + if (typeof text !== "string" || !text.trim()) { + throw new Error("Vision API returned empty content") + } + return text.trim() + } finally { + clearTimeout(timer) + } +} + +export const VisionBridge: Plugin = async () => { + let config: any = null + let providerInfo: ProviderInfo | null = null + let visionModels = new Set() + + // Image content cache keyed by image data hash to avoid re-analyzing + // the same image on every turn. + const cache = new Map() + + // Track which sessions are running a vision-capable model. + const visionSessions = new Map() + + const providerID = "ludmila-ai" + + function loadProvider(configAny: any) { + const p = configAny?.provider?.[providerID] + const baseURL = p?.options?.baseURL + const apiKey = p?.options?.apiKey + if (!baseURL || !apiKey) { + providerInfo = null + return + } + providerInfo = { baseURL: baseURL.replace(/\/$/, ""), apiKey, visionModel: "" } + const models = p?.models ?? {} + visionModels = new Set() + let visionApiName = "" + for (const [id, m] of Object.entries(models)) { + if (m?.attachment === true) { + visionModels.add(id) + if (!visionApiName) visionApiName = typeof m?.id === "string" && m.id ? m.id : id + } + } + // The vision model used for recognition = the model flagged attachment:true. + // Change it in the config by moving the `attachment: true` flag. + providerInfo.visionModel = visionApiName + } + + function hashImage(dataUrl: string): string { + // Use a small fixed part of the base64 as a cheap cache key + const i = dataUrl.indexOf(",") + const b64 = i >= 0 ? dataUrl.slice(i + 1) : dataUrl + let h = 0 + const step = Math.max(1, Math.floor(b64.length / 512)) + for (let k = 0; k < b64.length; k += step) { + h = (h * 31 + b64.charCodeAt(k)) | 0 + } + return h.toString(16) + ":" + b64.length + } + + async function processImagePart( + part: ImagePart, + modelID: string | undefined, + isVision: boolean, + ): Promise { + if (isVision || !providerInfo) { + // Vision model sees the image natively, or no provider config → keep original part + return part + } + const base: TextPart = { + id: part.id, + sessionID: part.sessionID, + messageID: part.messageID, + type: "text", + synthetic: true, + } + const dataUrl = mimeToDataUrl(part.url, part.mime) + const key = hashImage(dataUrl) + let text = cache.get(key) + if (!text) { + try { + if (!providerInfo.visionModel) { + throw new Error('no model with "attachment": true configured for ' + providerID) + } + text = await recognizeImage(providerInfo, providerInfo.visionModel, dataUrl) + cache.set(key, text) + if (cache.size > 200) { + const first = cache.keys().next().value + if (first !== undefined) cache.delete(first) + } + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + text = `[Image analysis failed: ${msg}]` + } + } + return { ...base, text: `[Image: ${text}]` } + } + + const hooks: Hooks = { + config: async (cfg: any) => { + config = cfg + loadProvider(cfg) + }, + + "chat.message": async (input: any, output: any) => { + if (!config) return + const parts: AnyPart[] = output.parts ?? [] + const modelID = input.model?.modelID + const isVision = isVisionModel(modelID, visionModels) + if (input.sessionID) visionSessions.set(input.sessionID, isVision) + let changed = false + const newParts: AnyPart[] = [] + for (const part of parts) { + if ( + part.type === "file" && + typeof part.mime === "string" && + IMAGE_MIME_RE.test(part.mime) && + typeof part.url === "string" + ) { + const result = await processImagePart( + { + id: part.id, + sessionID: part.sessionID, + messageID: part.messageID, + type: "file", + mime: part.mime, + filename: part.filename, + url: part.url, + }, + modelID, + isVision, + ) + newParts.push(result) + changed = true + } else { + newParts.push(part) + } + } + if (changed) output.parts = newParts + }, + + "experimental.chat.messages.transform": async (_input: any, output: any) => { + if (!config || !providerInfo) return + const msgs: { info: any; parts: AnyPart[] }[] = output.messages ?? [] + for (const msg of msgs) { + if (!msg?.parts?.length) continue + // If this message belongs to a session known to use a vision model, + // keep its media intact (native multimodal experience). + const sessionID = msg.info?.sessionID ?? msg.parts[0]?.sessionID + if (sessionID && visionSessions.get(sessionID)) continue + let changed = false + const newParts: AnyPart[] = [] + for (const part of msg.parts) { + if ( + part.type === "file" && + typeof part.mime === "string" && + IMAGE_MIME_RE.test(part.mime) && + typeof part.url === "string" + ) { + const result = await processImagePart( + { + id: part.id, + sessionID: part.sessionID, + messageID: part.messageID, + type: "file", + mime: part.mime, + filename: part.filename, + url: part.url, + }, + undefined, + false, + ) + newParts.push(result) + changed = true + } else { + newParts.push(part) + } + } + if (changed) msg.parts = newParts + } + }, + } + + return hooks +} + +export default VisionBridge