vision: images are the primary channel for multimodal models (digest off by default); digest is the text-only fallback; --digest/--no-digest to override

This commit is contained in:
opencode
2026-08-08 19:44:18 +03:00
parent ef02e5f9ea
commit d8373a50fc
5 changed files with 164 additions and 19 deletions
+21 -10
View File
@@ -229,6 +229,7 @@ class LLMController:
system_prompt: str = MISSION,
log: LogFn | None = None,
multimodal: bool | None = None,
digest: bool | None = None,
):
from openai import AsyncOpenAI
@@ -237,6 +238,9 @@ class LLMController:
if multimodal is None:
multimodal = detect_multimodal(base_url, model)
self.multimodal = multimodal
# The color-grid digest is the fallback channel for text-only models.
# For multimodal models it is off unless explicitly requested.
self.include_digest = (not multimodal) if digest is None else digest
self.ac = AsyncOpenAI(base_url=base_url, api_key=api_key)
self.tools = to_openai_tools(manifest.tools)
self.messages: list[dict[str, Any]] = [
@@ -344,9 +348,10 @@ class LLMController:
digest = None
extra = ""
if name == "vision" and isinstance(output.get("png_b64"), str):
digest = vision_digest(output["png_b64"])
extra = "\n" + digest
output = dict(output, png_b64="<binary, decoded for digest>")
if self.include_digest:
digest = vision_digest(outcome.output["png_b64"])
extra = "\n" + digest
self.log("bridge", f"ok {json.dumps(output)[:500]}{extra}")
if name == "world_query":
b = output.get("beacon")
@@ -358,13 +363,14 @@ class LLMController:
if name == "vision" and isinstance(outcome.output.get("png_b64"), str):
frame_b64 = outcome.output["png_b64"]
if self.multimodal and frame_b64 is not None:
# Real vision: hand the model the actual frame as an image
# (data URI), not as base64 text. Keep the digest too —
# it is a cheap textual anchor for the model.
content = (hint + "\n" if hint else "") + (
# Primary channel: hand the model the actual frame as an
# image (data URI). The digest is optional and off by
# default for multimodal models.
frame_note = (
f"Camera frame ({output['width']}x{output['height']}, "
f"tick {output.get('tick')}) attached as an image.{extra}"
f"tick {output.get('tick')}) attached as an image."
)
content = (hint + "\n" if hint else "") + frame_note + extra
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)
@@ -386,9 +392,14 @@ class LLMController:
}
)
else:
content = json.dumps(outcome.output) + extra
if hint:
content = hint + "\n" + content
# Fallback channel: the digest (plus frame metadata) for
# text-only models.
content = (
(hint + "\n" if hint else "")
+ f"Camera frame ({output['width']}x{output['height']}, "
+ f"tick {output.get('tick')})."
+ extra
)
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)