vision: real-time perception — auto frame attach every N turns (--look-every) and on collision, frame context cap (6), digest fallback for text models
This commit is contained in:
+65
-18
@@ -256,6 +256,55 @@ class LLMController:
|
||||
self.yaw: float = 0.0
|
||||
self.path: list[tuple[float, float]] = []
|
||||
|
||||
# -- real-time vision ---------------------------------------------------
|
||||
|
||||
def _frame_message(self, png_b64: str, note: str) -> dict[str, Any]:
|
||||
"""A user message carrying the camera frame: as an image for
|
||||
multimodal models, as the color-grid digest for text-only ones."""
|
||||
if self.multimodal:
|
||||
return {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": note},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {"url": f"data:image/png;base64,{png_b64}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
text = note + (f"\n{vision_digest(png_b64)}" if self.include_digest else "")
|
||||
return {"role": "user", "content": text}
|
||||
|
||||
def _trim_frames(self, max_frames: int = 6) -> None:
|
||||
"""Keep the vision context bounded: drop the oldest attached frames."""
|
||||
idxs = [
|
||||
i
|
||||
for i, m in enumerate(self.messages)
|
||||
if isinstance(m.get("content"), list)
|
||||
and any(
|
||||
isinstance(p, dict) and p.get("type") == "image_url"
|
||||
for p in m["content"]
|
||||
)
|
||||
]
|
||||
while len(idxs) > max_frames:
|
||||
del self.messages[idxs.pop(0)]
|
||||
|
||||
async def auto_frame(
|
||||
self, note: str = "Fresh camera frame — look around and react if needed."
|
||||
) -> None:
|
||||
"""Capture a frame proactively (the model did not ask) and attach it,
|
||||
so the agent always has recent visual context."""
|
||||
try:
|
||||
out = (await self.client.call_tool("vision", {})).output
|
||||
except Exception as exc: # noqa: BLE001 - auto vision is best-effort
|
||||
self.log("vision", f"auto frame failed: {type(exc).__name__}: {exc}")
|
||||
return
|
||||
tick = out.get("tick")
|
||||
note = f"{note} (tick {tick})."
|
||||
self.messages.append(self._frame_message(out["png_b64"], note))
|
||||
self._trim_frames()
|
||||
self.log("vision", f"auto frame attached (tick {tick})")
|
||||
|
||||
# -- agent-side working memory -----------------------------------------
|
||||
|
||||
def state_hint(self) -> str:
|
||||
@@ -374,23 +423,8 @@ class LLMController:
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
self.messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "This is what your camera sees right now. Use it to orient yourself.",
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": f"data:image/png;base64,{frame_b64}"
|
||||
},
|
||||
},
|
||||
],
|
||||
}
|
||||
)
|
||||
self.messages.append(self._frame_message(frame_b64, frame_note))
|
||||
self._trim_frames()
|
||||
else:
|
||||
# Fallback channel: the digest (plus frame metadata) for
|
||||
# text-only models.
|
||||
@@ -437,8 +471,13 @@ async def run_llm_agent_loop(
|
||||
log: LogFn,
|
||||
recorder: Any | None = None,
|
||||
nudge_limit: int = 1,
|
||||
look_every: int = 0,
|
||||
) -> dict[str, Any]:
|
||||
"""Drive the controller until the mission is done or steps run out."""
|
||||
"""Drive the controller until the mission is done or steps run out.
|
||||
|
||||
``look_every``: attach a fresh camera frame every N steps so the model
|
||||
always sees recent visual context without asking (0 disables).
|
||||
"""
|
||||
summary: dict[str, Any] = {
|
||||
"steps": 0,
|
||||
"tool_calls": 0,
|
||||
@@ -493,6 +532,10 @@ async def run_llm_agent_loop(
|
||||
summary["steps"] = step + 1
|
||||
if recorder is not None and controller.pos is not None:
|
||||
await recorder.snap(controller.client, controller.pos, controller.yaw)
|
||||
# Real-time perception: attach a fresh frame on a cadence so the model
|
||||
# sees what is happening without having to ask.
|
||||
if look_every and step % look_every == 0:
|
||||
await controller.auto_frame()
|
||||
turn = await controller.invoke()
|
||||
if turn.text:
|
||||
log("agent", turn.text[:400])
|
||||
@@ -516,6 +559,10 @@ async def run_llm_agent_loop(
|
||||
):
|
||||
controller.messages.append({"role": "user", "content": hint_msg})
|
||||
log("agent", "(collision: go around)")
|
||||
# Show the model what it just bumped into.
|
||||
await controller.auto_frame(
|
||||
"You just bumped into something. Look at what is in front of you."
|
||||
)
|
||||
maybe_correct(step)
|
||||
if turn.interacted:
|
||||
summary["interacted"] = True
|
||||
|
||||
Reference in New Issue
Block a user