vision: real camera frames to the model (image_url), auto-detected for multimodal ollama models; --vision/--no-vision flags; verified gemma4:12b sees frames and completed the beacon mission

This commit is contained in:
opencode
2026-08-08 19:38:21 +03:00
parent 6a509aacc6
commit ef02e5f9ea
6 changed files with 5747 additions and 6 deletions
+9
View File
@@ -62,9 +62,12 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
model=model,
system_prompt=CHAT_MISSION,
log=lambda role, msg: print(f" [{role}] {msg}"),
multimodal=args.vision,
)
auto_steps = args.auto_steps
print(f"[chat] model: {model} (endpoint {args.base_url})")
if controller.multimodal:
print("[chat] vision: ON — the model sees the actual camera frames")
print("[chat] type your commands; /help for the command list; /exit to quit\n")
async def cmd_state() -> None:
@@ -413,6 +416,12 @@ def main() -> int:
default=3,
help="how many consecutive nudges a mission may use before giving up (default 3)",
)
parser.add_argument(
"--vision",
action=argparse.BooleanOptionalAction,
default=None,
help="pass real camera frames to the model as images (auto-detected for local ollama)",
)
args = parser.parse_args()
try:
return asyncio.run(run(args))