vision: real camera frames to the model (image_url), auto-detected for multimodal ollama models; --vision/--no-vision flags; verified gemma4:12b sees frames and completed the beacon mission

This commit is contained in:
opencode
2026-08-08 19:38:21 +03:00
parent 6a509aacc6
commit ef02e5f9ea
6 changed files with 5747 additions and 6 deletions
+68 -6
View File
@@ -184,6 +184,32 @@ class TurnResult:
message: str | None = None
def detect_multimodal(base_url: str, model: str) -> bool:
"""Best-effort check whether the model can actually see images.
For a local ollama we ask /api/show (capabilities include 'vision').
For other OpenAI-compatible endpoints we cannot introspect — the caller
may force it with --vision.
"""
host = base_url.replace("http://", "").replace("https://", "").split("/")[0]
if host not in ("localhost:11434", "127.0.0.1:11434"):
return False
import urllib.request
endpoint = base_url.rsplit("/v1", 1)[0] + "/api/show"
try:
req = urllib.request.Request(
endpoint,
data=json.dumps({"model": model}).encode(),
headers={"Content-Type": "application/json"},
)
with urllib.request.urlopen(req, timeout=10) as resp:
caps = json.loads(resp.read()).get("capabilities", [])
return "vision" in caps
except Exception: # noqa: BLE001 - detection is best-effort
return False
class LLMController:
"""Drives a tool-calling LLM over an OpenAI-compatible endpoint.
@@ -202,11 +228,15 @@ class LLMController:
model: str,
system_prompt: str = MISSION,
log: LogFn | None = None,
multimodal: bool | None = None,
):
from openai import AsyncOpenAI
self.client = client
self.model = model
if multimodal is None:
multimodal = detect_multimodal(base_url, model)
self.multimodal = multimodal
self.ac = AsyncOpenAI(base_url=base_url, api_key=api_key)
self.tools = to_openai_tools(manifest.tools)
self.messages: list[dict[str, Any]] = [
@@ -324,12 +354,44 @@ class LLMController:
self.beacon_pos = (b["x"], b["z"])
self.observe(output)
hint = self.state_hint()
content = json.dumps(outcome.output) + extra
if hint:
content = hint + "\n" + content
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)
frame_b64 = None
if name == "vision" and isinstance(outcome.output.get("png_b64"), str):
frame_b64 = outcome.output["png_b64"]
if self.multimodal and frame_b64 is not None:
# Real vision: hand the model the actual frame as an image
# (data URI), not as base64 text. Keep the digest too —
# it is a cheap textual anchor for the model.
content = (hint + "\n" if hint else "") + (
f"Camera frame ({output['width']}x{output['height']}, "
f"tick {output.get('tick')}) attached as an image.{extra}"
)
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)
self.messages.append(
{
"role": "user",
"content": [
{
"type": "text",
"text": "This is what your camera sees right now. Use it to orient yourself.",
},
{
"type": "image_url",
"image_url": {
"url": f"data:image/png;base64,{frame_b64}"
},
},
],
}
)
else:
content = json.dumps(outcome.output) + extra
if hint:
content = hint + "\n" + content
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)
call_result = ToolCallResult(
name=name, args=args, ok=True, output=outcome.output, digest=digest
)