feat: Initial commit of vision-bridge plugin and agent

This commit is contained in:
Maksim Totmin
2026-08-04 20:25:11 +07:00
commit b0699f2f07
2 changed files with 298 additions and 0 deletions
+18
View File
@@ -0,0 +1,18 @@
---
description: Анализ изображений и скриншотов (OCR текста или разбор UI)
mode: subagent
model: ludmila-ai/gemini/gemini-2.5-flash
temperature: 0.1
permission:
read: allow
glob: allow
external_directory: allow
edit: deny
bash: deny
---
Ты анализируешь изображение. Сначала определи тип контента:
1) Преимущественно ТЕКСТ/документ → извлеки текст ДОСЛОВНО, сохрани структуру
(заголовки, списки, колонки, код). Не пересказывай и не исправляй.
2) ЭЛЕМЕНТЫ ИНТЕРФЕЙСА (веб/моб/десктоп) → опиши: тип экрана, компоненты и
расположение, тексты кнопок/полей, состояния, интерактивность, стили.
3) Смешанное → сделай оба пункта раздельно.
+280
View File
@@ -0,0 +1,280 @@
import type { Plugin, Hooks } from "@opencode-ai/plugin"
const VISION_PROMPT = `You are a vision assistant. Analyze the image in detail and produce a thorough text representation.
Rules:
1. If the image is predominantly TEXT (code, error message, document, chat, log, terminal output) — transcribe the text VERBATIM. Preserve structure (headings, lists, columns, indentation). Do not paraphrase or "fix" typos.
2. If the image shows a UI / INTERFACE (web, mobile, desktop) — describe: type of screen, components and their layout, exact button/label/field texts, states, interactivity, visual style.
3. If mixed — do BOTH, in separate labeled sections.
4. If the image is a graph/diagram/chart — describe axes, labels, values, and the overall message.
Be precise and exhaustive. This text will be given to another model that cannot see the image.`
type ImagePart = {
id: string
sessionID: string
messageID: string
type: "file"
mime: string
filename?: string
url: string
}
type TextPart = {
id: string
sessionID: string
messageID: string
type: "text"
text: string
synthetic?: boolean
}
type AnyPart = { type: string; [key: string]: any }
interface ProviderInfo {
baseURL: string
apiKey: string
visionModel: string
}
const IMAGE_MIME_RE = /^image\//
function isVisionModel(modelID: string | undefined, visionModels: Set<string>): boolean {
if (!modelID) return false
// Normalize: strip provider prefix if present, and also handle "provider/model" form
for (const m of visionModels) {
if (modelID === m) return true
if (modelID.endsWith("/" + m)) return true
if (m.endsWith("/" + modelID)) return true
}
return false
}
function mimeToDataUrl(url: string, mime: string): string {
// url is already data:image/png;base64,...
if (url.startsWith("data:")) return url
return `data:${mime};base64,${url}`
}
async function recognizeImage(
provider: ProviderInfo,
modelID: string,
dataUrl: string,
timeoutMs = 60000,
): Promise<string> {
const controller = new AbortController()
const timer = setTimeout(() => controller.abort(), timeoutMs)
try {
const res = await fetch(`${provider.baseURL}/chat/completions`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${provider.apiKey}`,
},
body: JSON.stringify({
model: modelID,
messages: [
{
role: "user",
content: [
{ type: "text", text: VISION_PROMPT },
{ type: "image_url", image_url: { url: dataUrl } },
],
},
],
max_tokens: 4096,
temperature: 0.1,
}),
signal: controller.signal,
})
if (!res.ok) {
const body = await res.text().catch(() => "")
throw new Error(`Vision API ${res.status}: ${body.slice(0, 300)}`)
}
const data = await res.json()
const text = data?.choices?.[0]?.message?.content
if (typeof text !== "string" || !text.trim()) {
throw new Error("Vision API returned empty content")
}
return text.trim()
} finally {
clearTimeout(timer)
}
}
export const VisionBridge: Plugin = async () => {
let config: any = null
let providerInfo: ProviderInfo | null = null
let visionModels = new Set<string>()
// Image content cache keyed by image data hash to avoid re-analyzing
// the same image on every turn.
const cache = new Map<string, string>()
// Track which sessions are running a vision-capable model.
const visionSessions = new Map<string, boolean>()
const providerID = "ludmila-ai"
function loadProvider(configAny: any) {
const p = configAny?.provider?.[providerID]
const baseURL = p?.options?.baseURL
const apiKey = p?.options?.apiKey
if (!baseURL || !apiKey) {
providerInfo = null
return
}
providerInfo = { baseURL: baseURL.replace(/\/$/, ""), apiKey, visionModel: "" }
const models = p?.models ?? {}
visionModels = new Set<string>()
let visionApiName = ""
for (const [id, m] of Object.entries<any>(models)) {
if (m?.attachment === true) {
visionModels.add(id)
if (!visionApiName) visionApiName = typeof m?.id === "string" && m.id ? m.id : id
}
}
// The vision model used for recognition = the model flagged attachment:true.
// Change it in the config by moving the `attachment: true` flag.
providerInfo.visionModel = visionApiName
}
function hashImage(dataUrl: string): string {
// Use a small fixed part of the base64 as a cheap cache key
const i = dataUrl.indexOf(",")
const b64 = i >= 0 ? dataUrl.slice(i + 1) : dataUrl
let h = 0
const step = Math.max(1, Math.floor(b64.length / 512))
for (let k = 0; k < b64.length; k += step) {
h = (h * 31 + b64.charCodeAt(k)) | 0
}
return h.toString(16) + ":" + b64.length
}
async function processImagePart(
part: ImagePart,
modelID: string | undefined,
isVision: boolean,
): Promise<AnyPart> {
if (isVision || !providerInfo) {
// Vision model sees the image natively, or no provider config → keep original part
return part
}
const base: TextPart = {
id: part.id,
sessionID: part.sessionID,
messageID: part.messageID,
type: "text",
synthetic: true,
}
const dataUrl = mimeToDataUrl(part.url, part.mime)
const key = hashImage(dataUrl)
let text = cache.get(key)
if (!text) {
try {
if (!providerInfo.visionModel) {
throw new Error('no model with "attachment": true configured for ' + providerID)
}
text = await recognizeImage(providerInfo, providerInfo.visionModel, dataUrl)
cache.set(key, text)
if (cache.size > 200) {
const first = cache.keys().next().value
if (first !== undefined) cache.delete(first)
}
} catch (err) {
const msg = err instanceof Error ? err.message : String(err)
text = `[Image analysis failed: ${msg}]`
}
}
return { ...base, text: `[Image: ${text}]` }
}
const hooks: Hooks = {
config: async (cfg: any) => {
config = cfg
loadProvider(cfg)
},
"chat.message": async (input: any, output: any) => {
if (!config) return
const parts: AnyPart[] = output.parts ?? []
const modelID = input.model?.modelID
const isVision = isVisionModel(modelID, visionModels)
if (input.sessionID) visionSessions.set(input.sessionID, isVision)
let changed = false
const newParts: AnyPart[] = []
for (const part of parts) {
if (
part.type === "file" &&
typeof part.mime === "string" &&
IMAGE_MIME_RE.test(part.mime) &&
typeof part.url === "string"
) {
const result = await processImagePart(
{
id: part.id,
sessionID: part.sessionID,
messageID: part.messageID,
type: "file",
mime: part.mime,
filename: part.filename,
url: part.url,
},
modelID,
isVision,
)
newParts.push(result)
changed = true
} else {
newParts.push(part)
}
}
if (changed) output.parts = newParts
},
"experimental.chat.messages.transform": async (_input: any, output: any) => {
if (!config || !providerInfo) return
const msgs: { info: any; parts: AnyPart[] }[] = output.messages ?? []
for (const msg of msgs) {
if (!msg?.parts?.length) continue
// If this message belongs to a session known to use a vision model,
// keep its media intact (native multimodal experience).
const sessionID = msg.info?.sessionID ?? msg.parts[0]?.sessionID
if (sessionID && visionSessions.get(sessionID)) continue
let changed = false
const newParts: AnyPart[] = []
for (const part of msg.parts) {
if (
part.type === "file" &&
typeof part.mime === "string" &&
IMAGE_MIME_RE.test(part.mime) &&
typeof part.url === "string"
) {
const result = await processImagePart(
{
id: part.id,
sessionID: part.sessionID,
messageID: part.messageID,
type: "file",
mime: part.mime,
filename: part.filename,
url: part.url,
},
undefined,
false,
)
newParts.push(result)
changed = true
} else {
newParts.push(part)
}
}
if (changed) msg.parts = newParts
}
},
}
return hooks
}
export default VisionBridge