record_search: three-view GIF recorder (map | first person | chat with model reasoning); capture reasoning_content in TurnResult; search_mission.gif artifact

This commit is contained in:
opencode
2026-08-09 17:25:08 +03:00
parent daa0371519
commit a9e45fb427
4 changed files with 289 additions and 0 deletions
Binary file not shown.

After

Width:  |  Height:  |  Size: 760 KiB

+2
View File
@@ -342,6 +342,8 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
steps_used += 1
if turn.text:
print(f" model> {turn.text}")
if turn.thinking:
print(f" [thinking] {turn.thinking[:300]}")
if turn.interacted:
print(f"\n[chat] BEACON ACTIVATED: {turn.message}")
save_map()
+12
View File
@@ -314,6 +314,7 @@ class TurnResult:
calls: list[ToolCallResult] = field(default_factory=list)
interacted: bool = False
message: str | None = None
thinking: str = ""
def detect_multimodal(base_url: str, model: str) -> bool:
@@ -502,6 +503,15 @@ class LLMController:
choice = resp.choices[0]
text = choice.message.content or ""
result = TurnResult(text=text)
# Some providers return reasoning tokens separately (e.g. deepseek-r1
# style models); surface them when present.
msg = choice.message
thinking = (
getattr(msg, "reasoning_content", None)
or getattr(msg, "reasoning", None)
or ""
)
result.thinking = thinking if isinstance(thinking, str) else ""
calls = choice.message.tool_calls
if not calls:
self.messages.append({"role": "assistant", "content": text})
@@ -741,6 +751,8 @@ async def run_llm_agent_loop(
turn = await controller.invoke()
if turn.text:
log("agent", turn.text[:400])
if turn.thinking:
log("think", turn.thinking[:400])
summary["tool_calls"] += len(turn.calls)
for c in turn.calls:
if (
+275
View File
@@ -0,0 +1,275 @@
"""Record a search mission and render a three-view GIF for the README.
Panes (left -> right):
top-down debug map (world state: crates, marker, path, capsule)
first person (the actual frames the model received)
chat (the model's messages, reasoning, and tool calls)
Usage:
python -m testbed.record_search --out search_mission.gif
"""
from __future__ import annotations
import argparse
import asyncio
import base64
import io
import sys
from pathlib import Path
from aicc.client import AICCClient
from aicc.transport.websocket import WebSocketClientTransport, WebSocketServer
from PIL import Image, ImageDraw, ImageFont
from testbed.bridge import build_bridge
from testbed.llm_agent import (
SEARCH_MISSION,
LLMController,
resolve_provider,
run_llm_agent_loop,
)
PANEL = 400
ROOM = 16.0
MAX_CHAT_LINES = 21
CHAT_LINE_MAX = 46 # chars per chat line
ROLE_COLORS = {
"agent": (220, 220, 230),
"think": (150, 165, 200),
"tool": (255, 184, 96),
"bridge": (150, 210, 150),
"mission": (255, 220, 120),
"vision": (130, 210, 250),
}
def _font(size: int = 13) -> ImageFont.FreeTypeFont | ImageFont.ImageFont:
try:
return ImageFont.load_default(size=size)
except TypeError: # older Pillow
return ImageFont.load_default()
def _wrap(text: str, width: int = CHAT_LINE_MAX) -> list[str]:
words = text.split()
lines: list[str] = []
cur = ""
for w in words:
if len(cur) + len(w) + 1 > width:
if cur:
lines.append(cur)
cur = w
else:
cur = (cur + " " + w).strip()
if cur:
lines.append(cur)
return lines[:3] # keep the pane readable
class ChatPane:
"""Accumulates the mission chat and renders the last lines."""
def __init__(self) -> None:
self.lines: list[tuple[str, str]] = []
def add(self, role: str, msg: str) -> None:
for piece in _wrap(msg):
self.lines.append((role, piece))
if len(self.lines) > MAX_CHAT_LINES * 2:
self.lines = self.lines[-MAX_CHAT_LINES:]
def render(self) -> Image.Image:
img = Image.new("RGB", (PANEL, PANEL), (12, 12, 18))
draw = ImageDraw.Draw(img)
font = _font(13)
draw.text((8, 6), "chat — what the model says", fill=(160, 170, 190), font=font)
y = 26
for role, piece in self.lines[-MAX_CHAT_LINES:]:
color = ROLE_COLORS.get(role, (200, 200, 200))
draw.text((8, y), piece[:CHAT_LINE_MAX], fill=color, font=font)
y += 17
return img
def _topdown_frame(
bridge, path: list[tuple[float, float]], step: int, marker_found: bool
) -> Image.Image:
from testbed.room.render import render_topdown
img = render_topdown(bridge.world, PANEL)
draw = ImageDraw.Draw(img)
scale = PANEL / ROOM
if len(path) > 1:
pts = [(x * scale, PANEL - z * scale) for x, z in path]
draw.line(pts, fill=(255, 170, 60), width=3)
m = bridge.world.marker_world_pos()
if m is not None:
mx, mz = m[0] * scale, PANEL - m[2] * scale
r = 9
draw.polygon(
[(mx, mz - r), (mx - r, mz + r * 0.8), (mx + r, mz + r * 0.8)],
outline=(255, 255, 255),
width=2,
)
status = f"step {step} {'FOUND' if marker_found else 'searching...'}"
draw.text((10, PANEL - 24), status, fill=(255, 220, 120), font=_font(13))
return img
def _compose(map_img, view_img, chat_img) -> Image.Image:
canvas = Image.new("RGB", (PANEL * 3, PANEL), (18, 18, 26))
canvas.paste(map_img, (0, 0))
canvas.paste(view_img, (PANEL, 0))
canvas.paste(chat_img, (PANEL * 2, 0))
draw = ImageDraw.Draw(canvas)
draw.text((8, 6), "top-down", fill=(160, 170, 190), font=_font(12))
draw.text((PANEL + 8, 6), "first person", fill=(160, 170, 190), font=_font(12))
draw.text((PANEL * 2 + 8, 6), "chat", fill=(160, 170, 190), font=_font(12))
for x in (PANEL, PANEL * 2):
draw.line([(x, 0), (x, PANEL)], fill=(60, 60, 80))
return canvas
async def run(args: argparse.Namespace) -> int:
base_url, api_key, model = resolve_provider(
provider=args.provider,
base_url=args.base_url,
api_key=args.api_key,
model=args.model,
)
bridge = build_bridge()
# Fixed marker so the GIF tells a clear story (crate_blue, east face).
bridge.world.marker = {
"box": "crate_blue",
"face": "x1",
"u": 0.5,
"v": 0.55,
"size": 0.42,
"color": (255, 150, 40),
}
m = bridge.world.marker_world_pos()
print(
f"[record] triangle marker on crate_blue east face at {tuple(round(v, 2) for v in m)}"
)
frames: list[Image.Image] = []
chat = ChatPane()
marker_seen = False
def log(role: str, msg: str) -> None:
print(f" [{role}] {msg}")
if role in ("agent", "think", "tool", "mission", "vision"):
chat.add(role, msg)
async with WebSocketServer(bridge, port=8765):
transport = WebSocketClientTransport("ws://127.0.0.1:8765")
async with AICCClient(transport) as client:
manifest = await client.handshake()
ctl = LLMController(
client,
manifest,
base_url=base_url,
api_key=api_key,
model=model,
system_prompt=SEARCH_MISSION,
log=log,
)
ctl.messages = [
ctl.messages[0],
{
"role": "user",
"content": (
"MISSION: explore the room, find the orange triangle on the "
"back of a crate, and report it. Keep exploring until found."
),
},
]
chat.add("mission", "MISSION: find the orange triangle")
orig_frame = ctl._frame_message
def record_frame(png: str, note: str):
nonlocal marker_seen
view = (
Image.open(io.BytesIO(base64.b64decode(png)))
.convert("RGB")
.resize((PANEL, PANEL))
)
top = _topdown_frame(bridge, ctl.path, len(frames), marker_seen)
frames.append(_compose(top, view, chat.render()))
return orig_frame(png, note)
ctl._frame_message = record_frame # type: ignore[method-assign]
def is_success(turn):
nonlocal marker_seen
for c in turn.calls:
if (
c.name == "report"
and c.ok
and c.output
and c.output.get("verified")
):
marker_seen = True
return True
return False
summary = await run_llm_agent_loop(
ctl,
args.steps,
log=log,
look_every=1,
cruise=args.cruise,
autonomous=True,
is_success=is_success,
)
# Final frame: end state of the map.
if marker_seen:
chat.add("mission", f"DONE: {summary.get('result', 'triangle found')}")
top = _topdown_frame(bridge, ctl.path, len(frames), marker_seen)
last_view = (
frames[-1].crop((PANEL, 0, PANEL * 2, PANEL))
if frames
else Image.new("RGB", (PANEL, PANEL), (0, 0, 0))
)
frames.append(_compose(top, last_view, chat.render()))
out = Path(args.out)
if frames:
frames[0].save(
out,
save_all=True,
append_images=frames[1:],
duration=args.duration,
loop=0,
)
frames[-1].save(out.with_suffix(".png"))
print(f"[record] wrote {out} ({len(frames)} frames, {args.duration} ms/frame)")
print(f"[record] mission result: {summary.get('result')}")
return 0 if marker_seen else 1
def main() -> int:
parser = argparse.ArgumentParser(
description="Record a search mission into a three-view GIF."
)
parser.add_argument("--out", default="search_mission.gif")
parser.add_argument("--provider", default=None)
parser.add_argument("--base-url", default=None)
parser.add_argument("--api-key", default=None)
parser.add_argument("--model", default=None)
parser.add_argument("--steps", type=int, default=45)
parser.add_argument("--cruise", type=float, default=0.6)
parser.add_argument("--duration", type=int, default=350, help="ms per GIF frame")
args = parser.parse_args()
try:
return asyncio.run(run(args))
except KeyboardInterrupt:
print("\n[record] interrupted — the partial GIF was NOT written")
return 130
if __name__ == "__main__":
sys.exit(main())