vision: images are the primary channel for multimodal models (digest off by default); digest is the text-only fallback; --digest/--no-digest to override

This commit is contained in:
opencode
2026-08-08 19:44:18 +03:00
parent ef02e5f9ea
commit d8373a50fc
5 changed files with 164 additions and 19 deletions
+9
View File
@@ -63,11 +63,14 @@ async def chat_loop(client: AICCClient, manifest, args: argparse.Namespace) -> i
system_prompt=CHAT_MISSION,
log=lambda role, msg: print(f" [{role}] {msg}"),
multimodal=args.vision,
digest=args.digest,
)
auto_steps = args.auto_steps
print(f"[chat] model: {model} (endpoint {args.base_url})")
if controller.multimodal:
print("[chat] vision: ON — the model sees the actual camera frames")
else:
print("[chat] vision: text-only model — frames are sent as a color-grid digest")
print("[chat] type your commands; /help for the command list; /exit to quit\n")
async def cmd_state() -> None:
@@ -422,6 +425,12 @@ def main() -> int:
default=None,
help="pass real camera frames to the model as images (auto-detected for local ollama)",
)
parser.add_argument(
"--digest",
action=argparse.BooleanOptionalAction,
default=None,
help="always include the color-grid digest alongside images (off by default for multimodal models)",
)
args = parser.parse_args()
try:
return asyncio.run(run(args))