feat: keep the main chat responsive while subagents run
The main session and subagents share the same model backend; when that backend serializes requests (cloud rate limits or a local server), subagent streams queue the main chat. Adds two config knobs: - agent_concurrency (default 3): a semaphore in AgentTree.start caps how many subagent sessions stream simultaneously (slot is released on completion, timeout, or setup error so failures cannot deadlock the queue). - subagent_model (optional): routes spawned agents to a different model or backend, e.g. ollama/gemma4:e4b, so subagents never contend with the main session at all. Wired through the spawn tool, the main-session tools, and the per-agent runtime tools. Documents both in .hypothesis-machine.example.yaml and adds a concurrency-cap test (43 tests passing, tsc clean).
This commit is contained in:
+1
-1
@@ -39,7 +39,7 @@ export class SupervisorIntegration {
|
||||
runtimeFactory.attachTree(this.tree); if (!runId) this.pi.appendEntry(RUN_ENTRY, { runId: this.tree.runId });
|
||||
this.loop = new ResearchLoop(stateDir, this.tree.runId, this.tree.inspect(this.tree.rootId).task, this.config);
|
||||
this.lastScheduledIteration = this.loop.snapshot().iteration;
|
||||
const tools = createResearchTools({ tree: this.tree, parentId: this.tree.rootId, memory: this.memory, web: this.web, experiments: this.experiments, cwd: ctx.cwd });
|
||||
const tools = createResearchTools({ tree: this.tree, parentId: this.tree.rootId, memory: this.memory, web: this.web, experiments: this.experiments, cwd: ctx.cwd, ...(this.config.subagent_model ? { subagentModel: this.config.subagent_model } : {}) });
|
||||
for (const tool of [...tools, this.researchControlTool()]) this.pi.registerTool(this.withCompactRenderer(tool));
|
||||
if (ctx.hasUI) { ctx.ui.setStatus("hypothesis-machine", `HM ${this.tree.runId} · ready`); this.installAgentWidget(ctx); }
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user