vision: real camera frames to the model (image_url), auto-detected for multimodal ollama models; --vision/--no-vision flags; verified gemma4:12b sees frames and completed the beacon mission
This commit is contained in:
@@ -62,9 +62,12 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
|
||||
model=model,
|
||||
system_prompt=CHAT_MISSION,
|
||||
log=lambda role, msg: print(f" [{role}] {msg}"),
|
||||
multimodal=args.vision,
|
||||
)
|
||||
auto_steps = args.auto_steps
|
||||
print(f"[chat] model: {model} (endpoint {args.base_url})")
|
||||
if controller.multimodal:
|
||||
print("[chat] vision: ON — the model sees the actual camera frames")
|
||||
print("[chat] type your commands; /help for the command list; /exit to quit\n")
|
||||
|
||||
async def cmd_state() -> None:
|
||||
@@ -413,6 +416,12 @@ def main() -> int:
|
||||
default=3,
|
||||
help="how many consecutive nudges a mission may use before giving up (default 3)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--vision",
|
||||
action=argparse.BooleanOptionalAction,
|
||||
default=None,
|
||||
help="pass real camera frames to the model as images (auto-detected for local ollama)",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
return asyncio.run(run(args))
|
||||
|
||||
Reference in New Issue
Block a user