feat: keep the main chat responsive while subagents run
The main session and subagents share the same model backend; when that backend serializes requests (cloud rate limits or a local server), subagent streams queue the main chat. Adds two config knobs: - agent_concurrency (default 3): a semaphore in AgentTree.start caps how many subagent sessions stream simultaneously (slot is released on completion, timeout, or setup error so failures cannot deadlock the queue). - subagent_model (optional): routes spawned agents to a different model or backend, e.g. ollama/gemma4:e4b, so subagents never contend with the main session at all. Wired through the spawn tool, the main-session tools, and the per-agent runtime tools. Documents both in .hypothesis-machine.example.yaml and adds a concurrency-cap test (43 tests passing, tsc clean).
This commit is contained in:
@@ -22,4 +22,5 @@ describe("AgentTree", () => {
|
||||
it("does not count cancelled or failed children toward the child limit", async () => { const { tree } = setup({ ...DEFAULT_CONFIG, max_children_per_agent: 2 }); const first = await tree.spawn(request(tree.rootId, "Alpha", "First concrete assignment in a limited tree")); await tree.cancel(first.id); const second = await tree.spawn(request(tree.rootId, "Beta", "Second concrete assignment in a limited tree")); await tree.cancel(second.id); await expect(tree.spawn(request(tree.rootId, "Gamma", "Third concrete assignment in a limited tree"))).resolves.toBeTruthy(); });
|
||||
it("cancel does not clobber an already recorded result", async () => { const { tree } = setup(); const child = await tree.spawn(request(tree.rootId, "Done", "Complete a concrete assignment and return")); const result = await tree.start(child.id); expect(result.status).toBe("completed"); await tree.cancel(child.id); expect(tree.inspect(child.id).status).toBe("cancelled"); expect(tree.inspect(child.id).result?.status).toBe("completed"); });
|
||||
it("fails agents that exceed the configured timeout", async () => { const { tree } = setup({ ...DEFAULT_CONFIG, agent_timeout_seconds: 0.05 }, 200); const child = await tree.spawn(request(tree.rootId, "Slow", "Run a deliberately slow concrete assignment")); const result = await tree.start(child.id); expect(result.status).toBe("failed"); expect(result.summary).toMatch(/timed out/); expect(tree.inspect(child.id).error).toMatch(/timed out/); });
|
||||
it("caps how many agent sessions stream at the same time", async () => { const { tree } = setup({ ...DEFAULT_CONFIG, agent_concurrency: 1 }, 40); const a = await tree.spawn(request(tree.rootId, "First", "Sequential streaming assignment number one")); const b = await tree.spawn(request(tree.rootId, "Second", "Sequential streaming assignment number two")); const [ra, rb] = await Promise.all([tree.start(a.id), tree.start(b.id)]); expect(ra.status).toBe("completed"); expect(rb.status).toBe("completed"); const startedA = Date.parse(tree.inspect(a.id).startedAt!); const startedB = Date.parse(tree.inspect(b.id).startedAt!); const finishedA = Date.parse(tree.inspect(a.id).finishedAt!); expect(startedB).toBeGreaterThanOrEqual(finishedA - 5); });
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user