vision: images are the primary channel for multimodal models (digest off by default); digest is the text-only fallback; --digest/--no-digest to override
This commit is contained in:
+21
-10
@@ -229,6 +229,7 @@ class LLMController:
|
||||
system_prompt: str = MISSION,
|
||||
log: LogFn | None = None,
|
||||
multimodal: bool | None = None,
|
||||
digest: bool | None = None,
|
||||
):
|
||||
from openai import AsyncOpenAI
|
||||
|
||||
@@ -237,6 +238,9 @@ class LLMController:
|
||||
if multimodal is None:
|
||||
multimodal = detect_multimodal(base_url, model)
|
||||
self.multimodal = multimodal
|
||||
# The color-grid digest is the fallback channel for text-only models.
|
||||
# For multimodal models it is off unless explicitly requested.
|
||||
self.include_digest = (not multimodal) if digest is None else digest
|
||||
self.ac = AsyncOpenAI(base_url=base_url, api_key=api_key)
|
||||
self.tools = to_openai_tools(manifest.tools)
|
||||
self.messages: list[dict[str, Any]] = [
|
||||
@@ -344,9 +348,10 @@ class LLMController:
|
||||
digest = None
|
||||
extra = ""
|
||||
if name == "vision" and isinstance(output.get("png_b64"), str):
|
||||
digest = vision_digest(output["png_b64"])
|
||||
extra = "\n" + digest
|
||||
output = dict(output, png_b64="<binary, decoded for digest>")
|
||||
if self.include_digest:
|
||||
digest = vision_digest(outcome.output["png_b64"])
|
||||
extra = "\n" + digest
|
||||
self.log("bridge", f"ok {json.dumps(output)[:500]}{extra}")
|
||||
if name == "world_query":
|
||||
b = output.get("beacon")
|
||||
@@ -358,13 +363,14 @@ class LLMController:
|
||||
if name == "vision" and isinstance(outcome.output.get("png_b64"), str):
|
||||
frame_b64 = outcome.output["png_b64"]
|
||||
if self.multimodal and frame_b64 is not None:
|
||||
# Real vision: hand the model the actual frame as an image
|
||||
# (data URI), not as base64 text. Keep the digest too —
|
||||
# it is a cheap textual anchor for the model.
|
||||
content = (hint + "\n" if hint else "") + (
|
||||
# Primary channel: hand the model the actual frame as an
|
||||
# image (data URI). The digest is optional and off by
|
||||
# default for multimodal models.
|
||||
frame_note = (
|
||||
f"Camera frame ({output['width']}x{output['height']}, "
|
||||
f"tick {output.get('tick')}) attached as an image.{extra}"
|
||||
f"tick {output.get('tick')}) attached as an image."
|
||||
)
|
||||
content = (hint + "\n" if hint else "") + frame_note + extra
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
@@ -386,9 +392,14 @@ class LLMController:
|
||||
}
|
||||
)
|
||||
else:
|
||||
content = json.dumps(outcome.output) + extra
|
||||
if hint:
|
||||
content = hint + "\n" + content
|
||||
# Fallback channel: the digest (plus frame metadata) for
|
||||
# text-only models.
|
||||
content = (
|
||||
(hint + "\n" if hint else "")
|
||||
+ f"Camera frame ({output['width']}x{output['height']}, "
|
||||
+ f"tick {output.get('tick')})."
|
||||
+ extra
|
||||
)
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user