record_search: three-view GIF recorder (map | first person | chat with model reasoning); capture reasoning_content in TurnResult; search_mission.gif artifact
This commit is contained in:
Binary file not shown.
|
After Width: | Height: | Size: 760 KiB |
@@ -342,6 +342,8 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
|
|||||||
steps_used += 1
|
steps_used += 1
|
||||||
if turn.text:
|
if turn.text:
|
||||||
print(f" model> {turn.text}")
|
print(f" model> {turn.text}")
|
||||||
|
if turn.thinking:
|
||||||
|
print(f" [thinking] {turn.thinking[:300]}")
|
||||||
if turn.interacted:
|
if turn.interacted:
|
||||||
print(f"\n[chat] BEACON ACTIVATED: {turn.message}")
|
print(f"\n[chat] BEACON ACTIVATED: {turn.message}")
|
||||||
save_map()
|
save_map()
|
||||||
|
|||||||
@@ -314,6 +314,7 @@ class TurnResult:
|
|||||||
calls: list[ToolCallResult] = field(default_factory=list)
|
calls: list[ToolCallResult] = field(default_factory=list)
|
||||||
interacted: bool = False
|
interacted: bool = False
|
||||||
message: str | None = None
|
message: str | None = None
|
||||||
|
thinking: str = ""
|
||||||
|
|
||||||
|
|
||||||
def detect_multimodal(base_url: str, model: str) -> bool:
|
def detect_multimodal(base_url: str, model: str) -> bool:
|
||||||
@@ -502,6 +503,15 @@ class LLMController:
|
|||||||
choice = resp.choices[0]
|
choice = resp.choices[0]
|
||||||
text = choice.message.content or ""
|
text = choice.message.content or ""
|
||||||
result = TurnResult(text=text)
|
result = TurnResult(text=text)
|
||||||
|
# Some providers return reasoning tokens separately (e.g. deepseek-r1
|
||||||
|
# style models); surface them when present.
|
||||||
|
msg = choice.message
|
||||||
|
thinking = (
|
||||||
|
getattr(msg, "reasoning_content", None)
|
||||||
|
or getattr(msg, "reasoning", None)
|
||||||
|
or ""
|
||||||
|
)
|
||||||
|
result.thinking = thinking if isinstance(thinking, str) else ""
|
||||||
calls = choice.message.tool_calls
|
calls = choice.message.tool_calls
|
||||||
if not calls:
|
if not calls:
|
||||||
self.messages.append({"role": "assistant", "content": text})
|
self.messages.append({"role": "assistant", "content": text})
|
||||||
@@ -741,6 +751,8 @@ async def run_llm_agent_loop(
|
|||||||
turn = await controller.invoke()
|
turn = await controller.invoke()
|
||||||
if turn.text:
|
if turn.text:
|
||||||
log("agent", turn.text[:400])
|
log("agent", turn.text[:400])
|
||||||
|
if turn.thinking:
|
||||||
|
log("think", turn.thinking[:400])
|
||||||
summary["tool_calls"] += len(turn.calls)
|
summary["tool_calls"] += len(turn.calls)
|
||||||
for c in turn.calls:
|
for c in turn.calls:
|
||||||
if (
|
if (
|
||||||
|
|||||||
@@ -0,0 +1,275 @@
|
|||||||
|
"""Record a search mission and render a three-view GIF for the README.
|
||||||
|
|
||||||
|
Panes (left -> right):
|
||||||
|
top-down debug map (world state: crates, marker, path, capsule)
|
||||||
|
first person (the actual frames the model received)
|
||||||
|
chat (the model's messages, reasoning, and tool calls)
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python -m testbed.record_search --out search_mission.gif
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import base64
|
||||||
|
import io
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from aicc.client import AICCClient
|
||||||
|
from aicc.transport.websocket import WebSocketClientTransport, WebSocketServer
|
||||||
|
from PIL import Image, ImageDraw, ImageFont
|
||||||
|
|
||||||
|
from testbed.bridge import build_bridge
|
||||||
|
from testbed.llm_agent import (
|
||||||
|
SEARCH_MISSION,
|
||||||
|
LLMController,
|
||||||
|
resolve_provider,
|
||||||
|
run_llm_agent_loop,
|
||||||
|
)
|
||||||
|
|
||||||
|
PANEL = 400
|
||||||
|
ROOM = 16.0
|
||||||
|
MAX_CHAT_LINES = 21
|
||||||
|
CHAT_LINE_MAX = 46 # chars per chat line
|
||||||
|
|
||||||
|
ROLE_COLORS = {
|
||||||
|
"agent": (220, 220, 230),
|
||||||
|
"think": (150, 165, 200),
|
||||||
|
"tool": (255, 184, 96),
|
||||||
|
"bridge": (150, 210, 150),
|
||||||
|
"mission": (255, 220, 120),
|
||||||
|
"vision": (130, 210, 250),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _font(size: int = 13) -> ImageFont.FreeTypeFont | ImageFont.ImageFont:
|
||||||
|
try:
|
||||||
|
return ImageFont.load_default(size=size)
|
||||||
|
except TypeError: # older Pillow
|
||||||
|
return ImageFont.load_default()
|
||||||
|
|
||||||
|
|
||||||
|
def _wrap(text: str, width: int = CHAT_LINE_MAX) -> list[str]:
|
||||||
|
words = text.split()
|
||||||
|
lines: list[str] = []
|
||||||
|
cur = ""
|
||||||
|
for w in words:
|
||||||
|
if len(cur) + len(w) + 1 > width:
|
||||||
|
if cur:
|
||||||
|
lines.append(cur)
|
||||||
|
cur = w
|
||||||
|
else:
|
||||||
|
cur = (cur + " " + w).strip()
|
||||||
|
if cur:
|
||||||
|
lines.append(cur)
|
||||||
|
return lines[:3] # keep the pane readable
|
||||||
|
|
||||||
|
|
||||||
|
class ChatPane:
|
||||||
|
"""Accumulates the mission chat and renders the last lines."""
|
||||||
|
|
||||||
|
def __init__(self) -> None:
|
||||||
|
self.lines: list[tuple[str, str]] = []
|
||||||
|
|
||||||
|
def add(self, role: str, msg: str) -> None:
|
||||||
|
for piece in _wrap(msg):
|
||||||
|
self.lines.append((role, piece))
|
||||||
|
if len(self.lines) > MAX_CHAT_LINES * 2:
|
||||||
|
self.lines = self.lines[-MAX_CHAT_LINES:]
|
||||||
|
|
||||||
|
def render(self) -> Image.Image:
|
||||||
|
img = Image.new("RGB", (PANEL, PANEL), (12, 12, 18))
|
||||||
|
draw = ImageDraw.Draw(img)
|
||||||
|
font = _font(13)
|
||||||
|
draw.text((8, 6), "chat — what the model says", fill=(160, 170, 190), font=font)
|
||||||
|
y = 26
|
||||||
|
for role, piece in self.lines[-MAX_CHAT_LINES:]:
|
||||||
|
color = ROLE_COLORS.get(role, (200, 200, 200))
|
||||||
|
draw.text((8, y), piece[:CHAT_LINE_MAX], fill=color, font=font)
|
||||||
|
y += 17
|
||||||
|
return img
|
||||||
|
|
||||||
|
|
||||||
|
def _topdown_frame(
|
||||||
|
bridge, path: list[tuple[float, float]], step: int, marker_found: bool
|
||||||
|
) -> Image.Image:
|
||||||
|
from testbed.room.render import render_topdown
|
||||||
|
|
||||||
|
img = render_topdown(bridge.world, PANEL)
|
||||||
|
draw = ImageDraw.Draw(img)
|
||||||
|
scale = PANEL / ROOM
|
||||||
|
if len(path) > 1:
|
||||||
|
pts = [(x * scale, PANEL - z * scale) for x, z in path]
|
||||||
|
draw.line(pts, fill=(255, 170, 60), width=3)
|
||||||
|
m = bridge.world.marker_world_pos()
|
||||||
|
if m is not None:
|
||||||
|
mx, mz = m[0] * scale, PANEL - m[2] * scale
|
||||||
|
r = 9
|
||||||
|
draw.polygon(
|
||||||
|
[(mx, mz - r), (mx - r, mz + r * 0.8), (mx + r, mz + r * 0.8)],
|
||||||
|
outline=(255, 255, 255),
|
||||||
|
width=2,
|
||||||
|
)
|
||||||
|
status = f"step {step} {'FOUND' if marker_found else 'searching...'}"
|
||||||
|
draw.text((10, PANEL - 24), status, fill=(255, 220, 120), font=_font(13))
|
||||||
|
return img
|
||||||
|
|
||||||
|
|
||||||
|
def _compose(map_img, view_img, chat_img) -> Image.Image:
|
||||||
|
canvas = Image.new("RGB", (PANEL * 3, PANEL), (18, 18, 26))
|
||||||
|
canvas.paste(map_img, (0, 0))
|
||||||
|
canvas.paste(view_img, (PANEL, 0))
|
||||||
|
canvas.paste(chat_img, (PANEL * 2, 0))
|
||||||
|
draw = ImageDraw.Draw(canvas)
|
||||||
|
draw.text((8, 6), "top-down", fill=(160, 170, 190), font=_font(12))
|
||||||
|
draw.text((PANEL + 8, 6), "first person", fill=(160, 170, 190), font=_font(12))
|
||||||
|
draw.text((PANEL * 2 + 8, 6), "chat", fill=(160, 170, 190), font=_font(12))
|
||||||
|
for x in (PANEL, PANEL * 2):
|
||||||
|
draw.line([(x, 0), (x, PANEL)], fill=(60, 60, 80))
|
||||||
|
return canvas
|
||||||
|
|
||||||
|
|
||||||
|
async def run(args: argparse.Namespace) -> int:
|
||||||
|
base_url, api_key, model = resolve_provider(
|
||||||
|
provider=args.provider,
|
||||||
|
base_url=args.base_url,
|
||||||
|
api_key=args.api_key,
|
||||||
|
model=args.model,
|
||||||
|
)
|
||||||
|
bridge = build_bridge()
|
||||||
|
# Fixed marker so the GIF tells a clear story (crate_blue, east face).
|
||||||
|
bridge.world.marker = {
|
||||||
|
"box": "crate_blue",
|
||||||
|
"face": "x1",
|
||||||
|
"u": 0.5,
|
||||||
|
"v": 0.55,
|
||||||
|
"size": 0.42,
|
||||||
|
"color": (255, 150, 40),
|
||||||
|
}
|
||||||
|
m = bridge.world.marker_world_pos()
|
||||||
|
print(
|
||||||
|
f"[record] triangle marker on crate_blue east face at {tuple(round(v, 2) for v in m)}"
|
||||||
|
)
|
||||||
|
|
||||||
|
frames: list[Image.Image] = []
|
||||||
|
chat = ChatPane()
|
||||||
|
marker_seen = False
|
||||||
|
|
||||||
|
def log(role: str, msg: str) -> None:
|
||||||
|
print(f" [{role}] {msg}")
|
||||||
|
if role in ("agent", "think", "tool", "mission", "vision"):
|
||||||
|
chat.add(role, msg)
|
||||||
|
|
||||||
|
async with WebSocketServer(bridge, port=8765):
|
||||||
|
transport = WebSocketClientTransport("ws://127.0.0.1:8765")
|
||||||
|
async with AICCClient(transport) as client:
|
||||||
|
manifest = await client.handshake()
|
||||||
|
ctl = LLMController(
|
||||||
|
client,
|
||||||
|
manifest,
|
||||||
|
base_url=base_url,
|
||||||
|
api_key=api_key,
|
||||||
|
model=model,
|
||||||
|
system_prompt=SEARCH_MISSION,
|
||||||
|
log=log,
|
||||||
|
)
|
||||||
|
ctl.messages = [
|
||||||
|
ctl.messages[0],
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": (
|
||||||
|
"MISSION: explore the room, find the orange triangle on the "
|
||||||
|
"back of a crate, and report it. Keep exploring until found."
|
||||||
|
),
|
||||||
|
},
|
||||||
|
]
|
||||||
|
chat.add("mission", "MISSION: find the orange triangle")
|
||||||
|
orig_frame = ctl._frame_message
|
||||||
|
|
||||||
|
def record_frame(png: str, note: str):
|
||||||
|
nonlocal marker_seen
|
||||||
|
view = (
|
||||||
|
Image.open(io.BytesIO(base64.b64decode(png)))
|
||||||
|
.convert("RGB")
|
||||||
|
.resize((PANEL, PANEL))
|
||||||
|
)
|
||||||
|
top = _topdown_frame(bridge, ctl.path, len(frames), marker_seen)
|
||||||
|
frames.append(_compose(top, view, chat.render()))
|
||||||
|
return orig_frame(png, note)
|
||||||
|
|
||||||
|
ctl._frame_message = record_frame # type: ignore[method-assign]
|
||||||
|
|
||||||
|
def is_success(turn):
|
||||||
|
nonlocal marker_seen
|
||||||
|
for c in turn.calls:
|
||||||
|
if (
|
||||||
|
c.name == "report"
|
||||||
|
and c.ok
|
||||||
|
and c.output
|
||||||
|
and c.output.get("verified")
|
||||||
|
):
|
||||||
|
marker_seen = True
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
summary = await run_llm_agent_loop(
|
||||||
|
ctl,
|
||||||
|
args.steps,
|
||||||
|
log=log,
|
||||||
|
look_every=1,
|
||||||
|
cruise=args.cruise,
|
||||||
|
autonomous=True,
|
||||||
|
is_success=is_success,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Final frame: end state of the map.
|
||||||
|
if marker_seen:
|
||||||
|
chat.add("mission", f"DONE: {summary.get('result', 'triangle found')}")
|
||||||
|
top = _topdown_frame(bridge, ctl.path, len(frames), marker_seen)
|
||||||
|
last_view = (
|
||||||
|
frames[-1].crop((PANEL, 0, PANEL * 2, PANEL))
|
||||||
|
if frames
|
||||||
|
else Image.new("RGB", (PANEL, PANEL), (0, 0, 0))
|
||||||
|
)
|
||||||
|
frames.append(_compose(top, last_view, chat.render()))
|
||||||
|
|
||||||
|
out = Path(args.out)
|
||||||
|
if frames:
|
||||||
|
frames[0].save(
|
||||||
|
out,
|
||||||
|
save_all=True,
|
||||||
|
append_images=frames[1:],
|
||||||
|
duration=args.duration,
|
||||||
|
loop=0,
|
||||||
|
)
|
||||||
|
frames[-1].save(out.with_suffix(".png"))
|
||||||
|
print(f"[record] wrote {out} ({len(frames)} frames, {args.duration} ms/frame)")
|
||||||
|
print(f"[record] mission result: {summary.get('result')}")
|
||||||
|
return 0 if marker_seen else 1
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Record a search mission into a three-view GIF."
|
||||||
|
)
|
||||||
|
parser.add_argument("--out", default="search_mission.gif")
|
||||||
|
parser.add_argument("--provider", default=None)
|
||||||
|
parser.add_argument("--base-url", default=None)
|
||||||
|
parser.add_argument("--api-key", default=None)
|
||||||
|
parser.add_argument("--model", default=None)
|
||||||
|
parser.add_argument("--steps", type=int, default=45)
|
||||||
|
parser.add_argument("--cruise", type=float, default=0.6)
|
||||||
|
parser.add_argument("--duration", type=int, default=350, help="ms per GIF frame")
|
||||||
|
args = parser.parse_args()
|
||||||
|
try:
|
||||||
|
return asyncio.run(run(args))
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
print("\n[record] interrupted — the partial GIF was NOT written")
|
||||||
|
return 130
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Reference in New Issue
Block a user