vision: real-time perception — auto frame attach every N turns (--look-every) and on collision, frame context cap (6), digest fallback for text models
This commit is contained in:
@@ -101,6 +101,18 @@ works: `python -m testbed.chat --base-url https://api.openai.com/v1
|
||||
--model gpt-4o-mini --api-key $OPENAI_API_KEY`. `--auto-steps N` controls how
|
||||
many tool steps the model may chain per request (0 = one action per turn).
|
||||
|
||||
### Real-time perception
|
||||
|
||||
The model sees continuously, not only when it asks:
|
||||
|
||||
- `--look-every N` attaches a fresh camera frame every N steps/turns (default 3,
|
||||
`0` disables). Frames go in as images for multimodal models, as digests for
|
||||
text-only ones, and old frames are trimmed (max 6) to keep the context
|
||||
bounded.
|
||||
- On a collision the current frame is attached immediately, so the model sees
|
||||
what it bumped into.
|
||||
- The live viewer (http://127.0.0.1:8000) shows the same frames in the browser.
|
||||
|
||||
### Real vision
|
||||
|
||||
Vision is the **primary channel**: when the model is multimodal, `vision`
|
||||
|
||||
@@ -122,6 +122,7 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
|
||||
# The REPL stays responsive: /status to check progress, /stop to cancel.
|
||||
mission: asyncio.Task | None = None
|
||||
mission_goal = "reach the beacon and activate it"
|
||||
user_turns = 0
|
||||
|
||||
def mission_log(role: str, msg: str) -> None:
|
||||
print(f" [{role}] {msg}")
|
||||
@@ -159,6 +160,7 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
|
||||
args.mission_steps,
|
||||
log=mission_log,
|
||||
nudge_limit=args.mission_nudges,
|
||||
look_every=args.look_every,
|
||||
)
|
||||
except RuntimeError as exc:
|
||||
print(f" [mission] LLM error: {exc}")
|
||||
@@ -330,6 +332,13 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
|
||||
if not auto_steps:
|
||||
break
|
||||
# The model acted without commenting; let it keep going for 'go to X'
|
||||
# Real-time perception in manual chat: refresh the model's view of the
|
||||
# world on a cadence, so it reacts to what it sees without being asked.
|
||||
user_turns += 1
|
||||
if args.look_every and user_turns % args.look_every == 0:
|
||||
await controller.auto_frame(
|
||||
"Fresh camera frame for your reference — react if the world changed."
|
||||
)
|
||||
if controller.pos is not None:
|
||||
save_map()
|
||||
|
||||
@@ -431,6 +440,12 @@ def main() -> int:
|
||||
default=None,
|
||||
help="always include the color-grid digest alongside images (off by default for multimodal models)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--look-every",
|
||||
type=int,
|
||||
default=3,
|
||||
help="attach a fresh camera frame every N turns (0 disables; default 3)",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
return asyncio.run(run(args))
|
||||
|
||||
@@ -141,6 +141,7 @@ async def run_llm_agent(
|
||||
recorder: FrameRecorder | None = None,
|
||||
vision: bool | None = None,
|
||||
digest: bool | None = None,
|
||||
look_every: int = 0,
|
||||
) -> dict[str, Any]:
|
||||
"""Autonomous LLM run: controller + nudge/correct loop (see llm_agent)."""
|
||||
from testbed.llm_agent import LLMController, run_llm_agent_loop
|
||||
@@ -313,6 +314,7 @@ async def run_demo(args: argparse.Namespace) -> dict[str, Any]:
|
||||
recorder=recorder,
|
||||
vision=args.vision,
|
||||
digest=args.digest,
|
||||
look_every=args.look_every,
|
||||
)
|
||||
elif args.agent == "scripted":
|
||||
summary = await run_scripted_agent(
|
||||
@@ -335,6 +337,7 @@ async def run_demo(args: argparse.Namespace) -> dict[str, Any]:
|
||||
recorder=recorder,
|
||||
vision=args.vision,
|
||||
digest=args.digest,
|
||||
look_every=args.look_every,
|
||||
)
|
||||
if summary.get("interacted"):
|
||||
return summary
|
||||
@@ -408,6 +411,12 @@ def main() -> int:
|
||||
default=None,
|
||||
help="always include the color-grid digest alongside images (off by default for multimodal models)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--look-every",
|
||||
type=int,
|
||||
default=3,
|
||||
help="attach a fresh camera frame every N agent steps (0 disables; default 3)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--frame",
|
||||
default="demo_final_frame.png",
|
||||
|
||||
+65
-18
@@ -256,6 +256,55 @@ class LLMController:
|
||||
self.yaw: float = 0.0
|
||||
self.path: list[tuple[float, float]] = []
|
||||
|
||||
# -- real-time vision ---------------------------------------------------
|
||||
|
||||
def _frame_message(self, png_b64: str, note: str) -> dict[str, Any]:
|
||||
"""A user message carrying the camera frame: as an image for
|
||||
multimodal models, as the color-grid digest for text-only ones."""
|
||||
if self.multimodal:
|
||||
return {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": note},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {"url": f"data:image/png;base64,{png_b64}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
text = note + (f"\n{vision_digest(png_b64)}" if self.include_digest else "")
|
||||
return {"role": "user", "content": text}
|
||||
|
||||
def _trim_frames(self, max_frames: int = 6) -> None:
|
||||
"""Keep the vision context bounded: drop the oldest attached frames."""
|
||||
idxs = [
|
||||
i
|
||||
for i, m in enumerate(self.messages)
|
||||
if isinstance(m.get("content"), list)
|
||||
and any(
|
||||
isinstance(p, dict) and p.get("type") == "image_url"
|
||||
for p in m["content"]
|
||||
)
|
||||
]
|
||||
while len(idxs) > max_frames:
|
||||
del self.messages[idxs.pop(0)]
|
||||
|
||||
async def auto_frame(
|
||||
self, note: str = "Fresh camera frame — look around and react if needed."
|
||||
) -> None:
|
||||
"""Capture a frame proactively (the model did not ask) and attach it,
|
||||
so the agent always has recent visual context."""
|
||||
try:
|
||||
out = (await self.client.call_tool("vision", {})).output
|
||||
except Exception as exc: # noqa: BLE001 - auto vision is best-effort
|
||||
self.log("vision", f"auto frame failed: {type(exc).__name__}: {exc}")
|
||||
return
|
||||
tick = out.get("tick")
|
||||
note = f"{note} (tick {tick})."
|
||||
self.messages.append(self._frame_message(out["png_b64"], note))
|
||||
self._trim_frames()
|
||||
self.log("vision", f"auto frame attached (tick {tick})")
|
||||
|
||||
# -- agent-side working memory -----------------------------------------
|
||||
|
||||
def state_hint(self) -> str:
|
||||
@@ -374,23 +423,8 @@ class LLMController:
|
||||
self.messages.append(
|
||||
{"role": "tool", "tool_call_id": tc.id, "content": content}
|
||||
)
|
||||
self.messages.append(
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "This is what your camera sees right now. Use it to orient yourself.",
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": f"data:image/png;base64,{frame_b64}"
|
||||
},
|
||||
},
|
||||
],
|
||||
}
|
||||
)
|
||||
self.messages.append(self._frame_message(frame_b64, frame_note))
|
||||
self._trim_frames()
|
||||
else:
|
||||
# Fallback channel: the digest (plus frame metadata) for
|
||||
# text-only models.
|
||||
@@ -437,8 +471,13 @@ async def run_llm_agent_loop(
|
||||
log: LogFn,
|
||||
recorder: Any | None = None,
|
||||
nudge_limit: int = 1,
|
||||
look_every: int = 0,
|
||||
) -> dict[str, Any]:
|
||||
"""Drive the controller until the mission is done or steps run out."""
|
||||
"""Drive the controller until the mission is done or steps run out.
|
||||
|
||||
``look_every``: attach a fresh camera frame every N steps so the model
|
||||
always sees recent visual context without asking (0 disables).
|
||||
"""
|
||||
summary: dict[str, Any] = {
|
||||
"steps": 0,
|
||||
"tool_calls": 0,
|
||||
@@ -493,6 +532,10 @@ async def run_llm_agent_loop(
|
||||
summary["steps"] = step + 1
|
||||
if recorder is not None and controller.pos is not None:
|
||||
await recorder.snap(controller.client, controller.pos, controller.yaw)
|
||||
# Real-time perception: attach a fresh frame on a cadence so the model
|
||||
# sees what is happening without having to ask.
|
||||
if look_every and step % look_every == 0:
|
||||
await controller.auto_frame()
|
||||
turn = await controller.invoke()
|
||||
if turn.text:
|
||||
log("agent", turn.text[:400])
|
||||
@@ -516,6 +559,10 @@ async def run_llm_agent_loop(
|
||||
):
|
||||
controller.messages.append({"role": "user", "content": hint_msg})
|
||||
log("agent", "(collision: go around)")
|
||||
# Show the model what it just bumped into.
|
||||
await controller.auto_frame(
|
||||
"You just bumped into something. Look at what is in front of you."
|
||||
)
|
||||
maybe_correct(step)
|
||||
if turn.interacted:
|
||||
summary["interacted"] = True
|
||||
|
||||
Reference in New Issue
Block a user