vision: real-time perception — auto frame attach every N turns (--look-every) and on collision, frame context cap (6), digest fallback for text models

This commit is contained in:
opencode
2026-08-08 20:06:22 +03:00
parent d8373a50fc
commit 1d7dbb7b79
5 changed files with 449 additions and 18 deletions
+65 -18
View File
@@ -256,6 +256,55 @@ class LLMController:
self.yaw: float = 0.0
self.path: list[tuple[float, float]] = []
# -- real-time vision ---------------------------------------------------
def _frame_message(self, png_b64: str, note: str) -> dict[str, Any]:
"""A user message carrying the camera frame: as an image for
multimodal models, as the color-grid digest for text-only ones."""
if self.multimodal:
return {
"role": "user",
"content": [
{"type": "text", "text": note},
{
"type": "image_url",
"image_url": {"url": f"data:image/png;base64,{png_b64}"},
},
],
}
text = note + (f"\n{vision_digest(png_b64)}" if self.include_digest else "")
return {"role": "user", "content": text}
def _trim_frames(self, max_frames: int = 6) -> None:
"""Keep the vision context bounded: drop the oldest attached frames."""
idxs = [
i
for i, m in enumerate(self.messages)
if isinstance(m.get("content"), list)
and any(
isinstance(p, dict) and p.get("type") == "image_url"
for p in m["content"]
)
]
while len(idxs) > max_frames:
del self.messages[idxs.pop(0)]
async def auto_frame(
self, note: str = "Fresh camera frame — look around and react if needed."
) -> None:
"""Capture a frame proactively (the model did not ask) and attach it,
so the agent always has recent visual context."""
try:
out = (await self.client.call_tool("vision", {})).output
except Exception as exc: # noqa: BLE001 - auto vision is best-effort
self.log("vision", f"auto frame failed: {type(exc).__name__}: {exc}")
return
tick = out.get("tick")
note = f"{note} (tick {tick})."
self.messages.append(self._frame_message(out["png_b64"], note))
self._trim_frames()
self.log("vision", f"auto frame attached (tick {tick})")
# -- agent-side working memory -----------------------------------------
def state_hint(self) -> str:
@@ -374,23 +423,8 @@ class LLMController:
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)
self.messages.append(
{
"role": "user",
"content": [
{
"type": "text",
"text": "This is what your camera sees right now. Use it to orient yourself.",
},
{
"type": "image_url",
"image_url": {
"url": f"data:image/png;base64,{frame_b64}"
},
},
],
}
)
self.messages.append(self._frame_message(frame_b64, frame_note))
self._trim_frames()
else:
# Fallback channel: the digest (plus frame metadata) for
# text-only models.
@@ -437,8 +471,13 @@ async def run_llm_agent_loop(
log: LogFn,
recorder: Any | None = None,
nudge_limit: int = 1,
look_every: int = 0,
) -> dict[str, Any]:
"""Drive the controller until the mission is done or steps run out."""
"""Drive the controller until the mission is done or steps run out.
``look_every``: attach a fresh camera frame every N steps so the model
always sees recent visual context without asking (0 disables).
"""
summary: dict[str, Any] = {
"steps": 0,
"tool_calls": 0,
@@ -493,6 +532,10 @@ async def run_llm_agent_loop(
summary["steps"] = step + 1
if recorder is not None and controller.pos is not None:
await recorder.snap(controller.client, controller.pos, controller.yaw)
# Real-time perception: attach a fresh frame on a cadence so the model
# sees what is happening without having to ask.
if look_every and step % look_every == 0:
await controller.auto_frame()
turn = await controller.invoke()
if turn.text:
log("agent", turn.text[:400])
@@ -516,6 +559,10 @@ async def run_llm_agent_loop(
):
controller.messages.append({"role": "user", "content": hint_msg})
log("agent", "(collision: go around)")
# Show the model what it just bumped into.
await controller.auto_frame(
"You just bumped into something. Look at what is in front of you."
)
maybe_correct(step)
if turn.interacted:
summary["interacted"] = True