vision: real camera frames to the model (image_url), auto-detected for multimodal ollama models; --vision/--no-vision flags; verified gemma4:12b sees frames and completed the beacon mission
This commit is contained in:
+68
-6
@@ -184,6 +184,32 @@ class TurnResult:
|
||||
message: str | None = None
|
||||
|
||||
|
||||
def detect_multimodal(base_url: str, model: str) -> bool:
|
||||
"""Best-effort check whether the model can actually see images.
|
||||
|
||||
For a local ollama we ask /api/show (capabilities include 'vision').
|
||||
For other OpenAI-compatible endpoints we cannot introspect — the caller
|
||||
may force it with --vision.
|
||||
"""
|
||||
host = base_url.replace("http://", "").replace("https://", "").split("/")[0]
|
||||
if host not in ("localhost:11434", "127.0.0.1:11434"):
|
||||
return False
|
||||
import urllib.request
|
||||
|
||||
endpoint = base_url.rsplit("/v1", 1)[0] + "/api/show"
|
||||
try:
|
||||
req = urllib.request.Request(
|
||||
endpoint,
|
||||
data=json.dumps({"model": model}).encode(),
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=10) as resp:
|
||||
caps = json.loads(resp.read()).get("capabilities", [])
|
||||
return "vision" in caps
|
||||
except Exception: # noqa: BLE001 - detection is best-effort
|
||||
return False
|
||||
|
||||
|
||||
class LLMController:
|
||||
"""Drives a tool-calling LLM over an OpenAI-compatible endpoint.
|
||||
|
||||
@@ -202,11 +228,15 @@ class LLMController:
|
||||
model: str,
|
||||
system_prompt: str = MISSION,
|
||||
log: LogFn | None = None,
|
||||
multimodal: bool | None = None,
|
||||
):
|
||||
from openai import AsyncOpenAI
|
||||
|
||||
self.client = client
|
||||
self.model = model
|
||||
if multimodal is None:
|
||||
multimodal = detect_multimodal(base_url, model)
|
||||
self.multimodal = multimodal
|
||||
self.ac = AsyncOpenAI(base_url=base_url, api_key=api_key)
|
||||
self.tools = to_openai_tools(manifest.tools)
|
||||
self.messages: list[dict[str, Any]] = [
|
||||
@@ -324,12 +354,44 @@ class LLMController:
|
||||
self.beacon_pos = (b["x"], b["z"])
|
||||
self.observe(output)
|
||||
hint = self.state_hint()
|
||||
content = json.dumps(outcome.output) + extra
|
||||
if hint:
|
||||
content = hint + "\n" + content
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
frame_b64 = None
|
||||
if name == "vision" and isinstance(outcome.output.get("png_b64"), str):
|
||||
frame_b64 = outcome.output["png_b64"]
|
||||
if self.multimodal and frame_b64 is not None:
|
||||
# Real vision: hand the model the actual frame as an image
|
||||
# (data URI), not as base64 text. Keep the digest too —
|
||||
# it is a cheap textual anchor for the model.
|
||||
content = (hint + "\n" if hint else "") + (
|
||||
f"Camera frame ({output['width']}x{output['height']}, "
|
||||
f"tick {output.get('tick')}) attached as an image.{extra}"
|
||||
)
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
self.messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "This is what your camera sees right now. Use it to orient yourself.",
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": f"data:image/png;base64,{frame_b64}"
|
||||
},
|
||||
},
|
||||
],
|
||||
}
|
||||
)
|
||||
else:
|
||||
content = json.dumps(outcome.output) + extra
|
||||
if hint:
|
||||
content = hint + "\n" + content
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
call_result = ToolCallResult(
|
||||
name=name, args=args, ok=True, output=outcome.output, digest=digest
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user