vision: real-time perception — auto frame attach every N turns (--look-every) and on collision, frame context cap (6), digest fallback for text models

This commit is contained in:
opencode
2026-08-08 20:06:22 +03:00
parent d8373a50fc
commit 1d7dbb7b79
5 changed files with 449 additions and 18 deletions
+12
View File
@@ -101,6 +101,18 @@ works: `python -m testbed.chat --base-url https://api.openai.com/v1
--model gpt-4o-mini --api-key $OPENAI_API_KEY`. `--auto-steps N` controls how
many tool steps the model may chain per request (0 = one action per turn).
### Real-time perception
The model sees continuously, not only when it asks:
- `--look-every N` attaches a fresh camera frame every N steps/turns (default 3,
`0` disables). Frames go in as images for multimodal models, as digests for
text-only ones, and old frames are trimmed (max 6) to keep the context
bounded.
- On a collision the current frame is attached immediately, so the model sees
what it bumped into.
- The live viewer (http://127.0.0.1:8000) shows the same frames in the browser.
### Real vision
Vision is the **primary channel**: when the model is multimodal, `vision`
+15
View File
@@ -122,6 +122,7 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
# The REPL stays responsive: /status to check progress, /stop to cancel.
mission: asyncio.Task | None = None
mission_goal = "reach the beacon and activate it"
user_turns = 0
def mission_log(role: str, msg: str) -> None:
print(f" [{role}] {msg}")
@@ -159,6 +160,7 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
args.mission_steps,
log=mission_log,
nudge_limit=args.mission_nudges,
look_every=args.look_every,
)
except RuntimeError as exc:
print(f" [mission] LLM error: {exc}")
@@ -330,6 +332,13 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
if not auto_steps:
break
# The model acted without commenting; let it keep going for 'go to X'
# Real-time perception in manual chat: refresh the model's view of the
# world on a cadence, so it reacts to what it sees without being asked.
user_turns += 1
if args.look_every and user_turns % args.look_every == 0:
await controller.auto_frame(
"Fresh camera frame for your reference — react if the world changed."
)
if controller.pos is not None:
save_map()
@@ -431,6 +440,12 @@ def main() -> int:
default=None,
help="always include the color-grid digest alongside images (off by default for multimodal models)",
)
parser.add_argument(
"--look-every",
type=int,
default=3,
help="attach a fresh camera frame every N turns (0 disables; default 3)",
)
args = parser.parse_args()
try:
return asyncio.run(run(args))
+9
View File
@@ -141,6 +141,7 @@ async def run_llm_agent(
recorder: FrameRecorder | None = None,
vision: bool | None = None,
digest: bool | None = None,
look_every: int = 0,
) -> dict[str, Any]:
"""Autonomous LLM run: controller + nudge/correct loop (see llm_agent)."""
from testbed.llm_agent import LLMController, run_llm_agent_loop
@@ -313,6 +314,7 @@ async def run_demo(args: argparse.Namespace) -> dict[str, Any]:
recorder=recorder,
vision=args.vision,
digest=args.digest,
look_every=args.look_every,
)
elif args.agent == "scripted":
summary = await run_scripted_agent(
@@ -335,6 +337,7 @@ async def run_demo(args: argparse.Namespace) -> dict[str, Any]:
recorder=recorder,
vision=args.vision,
digest=args.digest,
look_every=args.look_every,
)
if summary.get("interacted"):
return summary
@@ -408,6 +411,12 @@ def main() -> int:
default=None,
help="always include the color-grid digest alongside images (off by default for multimodal models)",
)
parser.add_argument(
"--look-every",
type=int,
default=3,
help="attach a fresh camera frame every N agent steps (0 disables; default 3)",
)
parser.add_argument(
"--frame",
default="demo_final_frame.png",
+65 -18
View File
@@ -256,6 +256,55 @@ class LLMController:
self.yaw: float = 0.0
self.path: list[tuple[float, float]] = []
# -- real-time vision ---------------------------------------------------
def _frame_message(self, png_b64: str, note: str) -> dict[str, Any]:
"""A user message carrying the camera frame: as an image for
multimodal models, as the color-grid digest for text-only ones."""
if self.multimodal:
return {
"role": "user",
"content": [
{"type": "text", "text": note},
{
"type": "image_url",
"image_url": {"url": f"data:image/png;base64,{png_b64}"},
},
],
}
text = note + (f"\n{vision_digest(png_b64)}" if self.include_digest else "")
return {"role": "user", "content": text}
def _trim_frames(self, max_frames: int = 6) -> None:
"""Keep the vision context bounded: drop the oldest attached frames."""
idxs = [
i
for i, m in enumerate(self.messages)
if isinstance(m.get("content"), list)
and any(
isinstance(p, dict) and p.get("type") == "image_url"
for p in m["content"]
)
]
while len(idxs) > max_frames:
del self.messages[idxs.pop(0)]
async def auto_frame(
self, note: str = "Fresh camera frame — look around and react if needed."
) -> None:
"""Capture a frame proactively (the model did not ask) and attach it,
so the agent always has recent visual context."""
try:
out = (await self.client.call_tool("vision", {})).output
except Exception as exc: # noqa: BLE001 - auto vision is best-effort
self.log("vision", f"auto frame failed: {type(exc).__name__}: {exc}")
return
tick = out.get("tick")
note = f"{note} (tick {tick})."
self.messages.append(self._frame_message(out["png_b64"], note))
self._trim_frames()
self.log("vision", f"auto frame attached (tick {tick})")
# -- agent-side working memory -----------------------------------------
def state_hint(self) -> str:
@@ -374,23 +423,8 @@ class LLMController:
self.messages.append(
{"role": "tool", "tool_call_id": tc.id, "content": content}
)
self.messages.append(
{
"role": "user",
"content": [
{
"type": "text",
"text": "This is what your camera sees right now. Use it to orient yourself.",
},
{
"type": "image_url",
"image_url": {
"url": f"data:image/png;base64,{frame_b64}"
},
},
],
}
)
self.messages.append(self._frame_message(frame_b64, frame_note))
self._trim_frames()
else:
# Fallback channel: the digest (plus frame metadata) for
# text-only models.
@@ -437,8 +471,13 @@ async def run_llm_agent_loop(
log: LogFn,
recorder: Any | None = None,
nudge_limit: int = 1,
look_every: int = 0,
) -> dict[str, Any]:
"""Drive the controller until the mission is done or steps run out."""
"""Drive the controller until the mission is done or steps run out.
``look_every``: attach a fresh camera frame every N steps so the model
always sees recent visual context without asking (0 disables).
"""
summary: dict[str, Any] = {
"steps": 0,
"tool_calls": 0,
@@ -493,6 +532,10 @@ async def run_llm_agent_loop(
summary["steps"] = step + 1
if recorder is not None and controller.pos is not None:
await recorder.snap(controller.client, controller.pos, controller.yaw)
# Real-time perception: attach a fresh frame on a cadence so the model
# sees what is happening without having to ask.
if look_every and step % look_every == 0:
await controller.auto_frame()
turn = await controller.invoke()
if turn.text:
log("agent", turn.text[:400])
@@ -516,6 +559,10 @@ async def run_llm_agent_loop(
):
controller.messages.append({"role": "user", "content": hint_msg})
log("agent", "(collision: go around)")
# Show the model what it just bumped into.
await controller.auto_frame(
"You just bumped into something. Look at what is in front of you."
)
maybe_correct(step)
if turn.interacted:
summary["interacted"] = True