From 9409b06db836e9d897569266d17747bc6f9fadef Mon Sep 17 00:00:00 2001 From: Alem Tuzlak Date: Mon, 28 Sep 2026 11:23:13 +0200 Subject: [PATCH 01/52] feat(ai-sandbox, ai-harness): codingAgents plugin to delegate to Claude Code, Codex, and other coding agents; show child agent work in the CLI, ACP, and dashboard --- .changeset/harness-p9-coding-agents.md | 13 + docs/config.json | 11 +- docs/harness/coding-agent.md | 4 + docs/harness/coding-agents.md | 122 ++++++ docs/harness/dashboard.md | 2 +- docs/harness/deploy.md | 2 +- docs/harness/plugins.md | 1 + docs/harness/subagents.md | 2 +- examples/harness-cli/.env.example | 4 + examples/harness-cli/.gitignore | 2 + examples/harness-cli/README.md | 10 + examples/harness-cli/package.json | 4 + examples/harness-cli/src/harness.ts | 55 +++ packages/ai-acp/src/agent/index.ts | 7 +- packages/ai-acp/tests/child-updates.test.ts | 62 +++ packages/ai-dashboard/src/ui.ts | 31 +- .../ai-dashboard/tests/child-agents.test.ts | 143 +++++++ packages/ai-harness-cli/src/interactive.tsx | 7 + packages/ai-harness-cli/src/lines.ts | 4 +- packages/ai-harness-cli/src/session-view.ts | 85 +++- .../ai-harness-cli/tests/child-view.test.ts | 226 +++++++++++ .../ai-harness-cli/tests/commands.test.ts | 11 +- packages/ai-harness/src/plugins.ts | 14 + packages/ai-harness/src/session.ts | 12 +- .../ai-harness/tests/plugin-subagents.test.ts | 74 ++++ packages/ai-sandbox/package.json | 9 + packages/ai-sandbox/src/harness.ts | 210 ++++++++++ .../ai-sandbox/tests/coding-agents.test.ts | 377 ++++++++++++++++++ packages/ai-sandbox/vite.config.ts | 7 +- pnpm-lock.yaml | 15 + 30 files changed, 1502 insertions(+), 24 deletions(-) create mode 100644 .changeset/harness-p9-coding-agents.md create mode 100644 docs/harness/coding-agents.md create mode 100644 packages/ai-acp/tests/child-updates.test.ts create mode 100644 packages/ai-dashboard/tests/child-agents.test.ts create mode 100644 packages/ai-harness-cli/tests/child-view.test.ts create mode 100644 packages/ai-harness/tests/plugin-subagents.test.ts create mode 100644 packages/ai-sandbox/src/harness.ts create mode 100644 packages/ai-sandbox/tests/coding-agents.test.ts diff --git a/.changeset/harness-p9-coding-agents.md b/.changeset/harness-p9-coding-agents.md new file mode 100644 index 0000000000..62a57e4b40 --- /dev/null +++ b/.changeset/harness-p9-coding-agents.md @@ -0,0 +1,13 @@ +--- +'@tanstack/ai-sandbox': minor +'@tanstack/ai-harness': minor +'@tanstack/ai-harness-cli': minor +'@tanstack/ai-acp': minor +'@tanstack/ai-dashboard': minor +--- + +`@tanstack/ai-sandbox/harness` adds `codingAgents({ sandbox, agents, workspace? })`: a harness plugin that gives the lead model one tool per coding agent (Claude Code, Codex, Grok Build, or any ACP agent). Each agent runs in the sandbox, keeps its own session per thread (also after a restart), and starts read-only in the harness `plan` mode. `workspace: 'shared'` (default) runs one agent at a time in one sandbox per thread. `'per-agent'` gives each agent its own sandbox. `/fresh [agent]` starts new sessions. + +`@tanstack/ai-harness` plugins can contribute `subagents`: agents the model can call as tools. + +Child agent work is now visible: the CLI shows each child's tool calls and a finish line with the start of its answer, ACP editors get the child's tool calls, and the dashboard shows a block per child. diff --git a/docs/config.json b/docs/config.json index 71a5f53d3a..db8c43e1bc 100644 --- a/docs/config.json +++ b/docs/config.json @@ -827,12 +827,14 @@ { "label": "Write a plugin", "to": "harness/plugins", - "addedAt": "2026-09-26" + "addedAt": "2026-09-26", + "updatedAt": "2026-09-28" }, { "label": "Build a coding agent", "to": "harness/coding-agent", - "addedAt": "2026-09-26" + "addedAt": "2026-09-26", + "updatedAt": "2026-09-28" }, { "label": "Auth and connectors", @@ -849,6 +851,11 @@ "to": "harness/code-mode", "addedAt": "2026-09-26" }, + { + "label": "Delegate to coding agents", + "to": "harness/coding-agents", + "addedAt": "2026-09-28" + }, { "label": "Run agents from a harness", "to": "harness/subagents", diff --git a/docs/harness/coding-agent.md b/docs/harness/coding-agent.md index 8a45100343..3e1c2663ec 100644 --- a/docs/harness/coding-agent.md +++ b/docs/harness/coding-agent.md @@ -90,6 +90,10 @@ A trailing `*` matches every tool that starts with the text. The last matching r The workspace tools run on your machine with your permissions. Run code you do not trust in a sandbox. +## Hand work to Claude Code or Codex + +Your agent can also give tasks to coding agents you already use. [Delegate to coding agents](./coding-agents) shows how. + ## What you have now - A terminal coding agent with file tools, approvals, modes, a todo list, and a model picker. diff --git a/docs/harness/coding-agents.md b/docs/harness/coding-agents.md new file mode 100644 index 0000000000..7f61e47424 --- /dev/null +++ b/docs/harness/coding-agents.md @@ -0,0 +1,122 @@ +--- +title: Delegate to coding agents +id: harness-coding-agents +order: 10 +description: "Let a harness hand coding work to Claude Code, Codex, Grok Build, or any ACP agent. Each one works in a sandbox and keeps its own session." +keywords: + - tanstack ai + - harness + - claude code + - codex + - grok build + - sandbox + - subagents +--- + +Your lead agent plans the work, but you want Claude Code or Codex to do the edits, because they are good at code and you already use them. `codingAgents` gives the lead model one tool per coding agent. Each agent works in a sandbox, keeps its own session between calls, and its tool calls show up in your UI. + +## 1. Add the plugin + + + +react: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +vue: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +solid: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +svelte: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +preact: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +angular: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +octane: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex +vanilla: @tanstack/ai-harness @tanstack/ai-sandbox @tanstack/ai-sandbox-local-process @tanstack/ai-claude-code @tanstack/ai-codex + + + +```ts group=harness-coding-agents +import { defineHarness } from '@tanstack/ai-harness' +import { permissions } from '@tanstack/ai-harness/plugins' +import { claudeCodeText } from '@tanstack/ai-claude-code' +import { codexText } from '@tanstack/ai-codex' +import { openaiText } from '@tanstack/ai-openai' +import { defineSandbox, defineWorkspace, localSource } from '@tanstack/ai-sandbox' +import { codingAgents } from '@tanstack/ai-sandbox/harness' +import { localProcessSandbox } from '@tanstack/ai-sandbox-local-process' + +const repo = '/path/to/your/repo' + +export const lead = defineHarness({ + name: 'acme/lead', + adapter: openaiText('gpt-5.6'), + plugins: () => [ + permissions(), + codingAgents({ + sandbox: defineSandbox({ + id: 'repo', + provider: localProcessSandbox({ dir: repo }), + workspace: defineWorkspace({ source: localSource(repo) }), + }), + agents: { + claude_code: { + adapter: claudeCodeText('claude-opus-4-8', { permissionMode: 'acceptEdits' }), + description: 'Larger changes, refactors, and reviews', + }, + codex: { + adapter: codexText('gpt-5.3-codex', { sandboxMode: 'workspace-write' }), + description: 'Quick fixes and tests', + }, + }, + }), + ], +}) +``` + +The lead model now has a `claude_code` tool and a `codex` tool. Each tool takes one `task`: the whole job, in words. The agent sees only that text and the files in its sandbox. + +## 2. Ask for work + +1. Start the harness, for example with the CLI. +2. Ask the lead: `have claude_code add a test for the date parser, then have codex fix what fails`. +3. Watch the child work. The CLI shows each agent's tool calls, then one line with the start of its answer: + +```text +[agent claude_code started] +[claude_code: tool Write] +[agent claude_code finished: Added parse-date.test.ts with three cases.] +``` + +The dashboard shows the same work in a block under the lead's message. An ACP editor shows the child's tool calls next to the lead's own. + +## Sessions + +Each agent keeps its own session per harness thread. The next task for `claude_code` resumes the same Claude Code session, so it still knows the files it read. The session ids live in the plugin state, so they also survive a restart of the host. + +To start over, run `/fresh claude_code`, or `/fresh` for every agent. + +## Workspaces + +`workspace` chooses where the agents work: + +| Value | Where each agent works | At the same time | +| --- | --- | --- | +| `'shared'` (default) | One sandbox per harness thread | One agent at a time | +| `'per-agent'` | A sandbox per agent | Yes | + +Use `'shared'` when the agents build on each other's changes. Use `'per-agent'` when they work on separate copies and you merge the results. + +## Plan mode + +When the `permissions()` plugin is in `plan` mode, the agents start read-only: + +- Claude Code gets `permissionMode: 'plan'`. +- Codex gets `sandboxMode: 'read-only'`. +- Any other agent gets its `planModelOptions`, for example `{ permissionMode: 'default' }` for an ACP agent that asks before each edit. + +## Other agents and sandboxes + +- `adapter` takes any coding-agent adapter: `grokBuildText` from `@tanstack/ai-grok-build`, or `acpCompatibleText` from `@tanstack/ai-acp` for any ACP agent. +- `sandbox` takes any sandbox provider. [Sandbox providers](../sandbox/providers) lists them. +- `modelOptions` on an agent is added to every call, for example a fixed `permissionMode`. + +## What you have now + +- A lead model that hands tasks to Claude Code and Codex. +- One sandbox per thread, and one saved session per agent. +- Child tool calls in the CLI, the dashboard, and ACP editors. diff --git a/docs/harness/dashboard.md b/docs/harness/dashboard.md index afbd511cd7..7ed93a5de4 100644 --- a/docs/harness/dashboard.md +++ b/docs/harness/dashboard.md @@ -1,7 +1,7 @@ --- title: Self-host the dashboard id: harness-dashboard -order: 12 +order: 13 description: "Watch and steer your harness sessions from a browser or a phone. Agents dial out to your dashboard server, so they need no open port." keywords: - tanstack ai diff --git a/docs/harness/deploy.md b/docs/harness/deploy.md index ea6cf77423..04f94332f8 100644 --- a/docs/harness/deploy.md +++ b/docs/harness/deploy.md @@ -1,7 +1,7 @@ --- title: Deploy a harness id: harness-deploy -order: 11 +order: 12 description: "Run a harness in your server, as a worker process, on another machine, or as a single executable." keywords: - tanstack ai diff --git a/docs/harness/plugins.md b/docs/harness/plugins.md index 4e01346a91..f05cd1d3bc 100644 --- a/docs/harness/plugins.md +++ b/docs/harness/plugins.md @@ -33,6 +33,7 @@ Add it with `plugins: () => [today]` in `defineHarness`. `setup` runs once per s - `middleware`: chat middleware, the same type as `chat({ middleware })`. - `generationMiddleware`: middleware for the activities agents call. - `agents`: agents added to `session.agents`. +- `subagents`: agents the model can call as tools. They are also added to `session.agents`. [Delegate to coding agents](./coding-agents) uses them. - `commands`: user actions, see below. - `config`: session settings, see below. - `contribute`: items for another plugin's extension point. diff --git a/docs/harness/subagents.md b/docs/harness/subagents.md index 78de061c97..9e17ba8306 100644 --- a/docs/harness/subagents.md +++ b/docs/harness/subagents.md @@ -1,7 +1,7 @@ --- title: Run agents from a harness id: harness-subagents -order: 10 +order: 11 description: "Start typed agents from commands and plugins, run them in groups, call a whole harness as a child, and keep the tree within limits." keywords: - tanstack ai diff --git a/examples/harness-cli/.env.example b/examples/harness-cli/.env.example index bb89717ef3..38405cac10 100644 --- a/examples/harness-cli/.env.example +++ b/examples/harness-cli/.env.example @@ -5,3 +5,7 @@ OPENAI_API_KEY= ANTHROPIC_API_KEY= # Optional: make videos with Grok Imagine instead of OpenAI Sora. XAI_API_KEY= +# Optional: hand coding work to Claude Code and Codex (uses claude login and codex login). +CODING_AGENTS= +CODEX_MODEL= +CODEX_SANDBOX_MODE= diff --git a/examples/harness-cli/.gitignore b/examples/harness-cli/.gitignore index ec9309579e..d3d132daa7 100644 --- a/examples/harness-cli/.gitignore +++ b/examples/harness-cli/.gitignore @@ -1,2 +1,4 @@ .env playground/media/ +# Marker the sandbox writes into the workspace. +playground/.tanstack-projected-* diff --git a/examples/harness-cli/README.md b/examples/harness-cli/README.md index b87e8aa82f..65345557bb 100644 --- a/examples/harness-cli/README.md +++ b/examples/harness-cli/README.md @@ -29,6 +29,16 @@ Try these: - Images use `OPENAI_API_KEY`. Videos use Grok Imagine when `XAI_API_KEY` is set, and OpenAI Sora when it is not. - Code mode is on: read-only tools (file reads, read-only Notion and Linear tools) are `external_*` functions in one `execute_typescript` program, which runs in a QuickJS isolate. Ask: `in one program, list my Linear issues and search Notion for them`. +## Hand work to Claude Code and Codex + +1. Sign in to the CLIs once: `claude login` and `codex login`. +2. Start with `CODING_AGENTS=1`. If you use Codex with a ChatGPT login, also set `CODEX_MODEL` to the model in `~/.codex/config.toml`. +3. Ask: `have claude_code create notes.md with one line, then have codex add a second line`. + +- Both agents work in `./playground` with your own logins. The API keys are removed from their processes. +- The CLI shows each agent's tool calls and a finish line. `/fresh` starts new agent sessions. +- On Windows, the Codex sandbox can block the folder (Access is denied). Then set `CODEX_SANDBOX_MODE=danger-full-access`, only for a folder you trust. + ## Other modes - One prompt for scripts and CI: `pnpm --filter harness-cli-example start -p "list the files"` diff --git a/examples/harness-cli/package.json b/examples/harness-cli/package.json index a9d609b2b8..f85a495f63 100644 --- a/examples/harness-cli/package.json +++ b/examples/harness-cli/package.json @@ -12,7 +12,9 @@ "@tanstack/ai": "workspace:*", "@tanstack/ai-acp": "workspace:*", "@tanstack/ai-anthropic": "workspace:*", + "@tanstack/ai-claude-code": "workspace:*", "@tanstack/ai-code-mode": "workspace:*", + "@tanstack/ai-codex": "workspace:*", "@tanstack/ai-dashboard": "workspace:*", "@tanstack/ai-grok": "workspace:*", "@tanstack/ai-harness": "workspace:*", @@ -21,6 +23,8 @@ "@tanstack/ai-mcp": "workspace:*", "@tanstack/ai-openai": "workspace:*", "@tanstack/ai-persistence": "workspace:*", + "@tanstack/ai-sandbox": "workspace:*", + "@tanstack/ai-sandbox-local-process": "workspace:*", "zod": "^4.2.0" }, "devDependencies": { diff --git a/examples/harness-cli/src/harness.ts b/examples/harness-cli/src/harness.ts index bc0563abda..ce06031439 100644 --- a/examples/harness-cli/src/harness.ts +++ b/examples/harness-cli/src/harness.ts @@ -11,11 +11,20 @@ import { workspaceTools, } from '@tanstack/ai-harness/plugins' import { anthropicText } from '@tanstack/ai-anthropic' +import { claudeCodeText } from '@tanstack/ai-claude-code' import { codeMode } from '@tanstack/ai-code-mode/harness' +import { codexText } from '@tanstack/ai-codex' import { mcpConnector } from '@tanstack/ai-mcp/connector' import { grokVideo } from '@tanstack/ai-grok' import { createQuickJSIsolateDriver } from '@tanstack/ai-isolate-quickjs' import { openaiText, openaiVideo } from '@tanstack/ai-openai' +import { + defineSandbox, + defineWorkspace, + localSource, +} from '@tanstack/ai-sandbox' +import { codingAgents } from '@tanstack/ai-sandbox/harness' +import { localProcessSandbox } from '@tanstack/ai-sandbox-local-process' import { z } from 'zod' import { imageAgent, videoAgent } from './media' import type { AnyTextAdapter } from '@tanstack/ai' @@ -149,6 +158,51 @@ const linear = mcpConnector({ url: 'https://mcp.linear.app/mcp', }) +// With CODING_AGENTS=1, the lead model can hand coding work to Claude Code and +// Codex. They work in ./playground with your own `claude login` and +// `codex login`, so the API keys are removed from their processes. +const coding = + process.env.CODING_AGENTS === '1' + ? [ + codingAgents({ + sandbox: defineSandbox({ + id: 'playground', + provider: localProcessSandbox({ + dir: root, + scrubEnv: ['ANTHROPIC_API_KEY', 'OPENAI_API_KEY'], + }), + workspace: defineWorkspace({ source: localSource(root) }), + }), + agents: { + claude_code: { + adapter: claudeCodeText('claude-opus-4-8', { + authMode: 'host', + permissionMode: 'acceptEdits', + }), + description: + 'Claude Code. Larger changes, refactors, and reviews in ./playground.', + }, + codex: { + // A ChatGPT login supports only some models. Set CODEX_MODEL to the + // model in ~/.codex/config.toml. + adapter: codexText(process.env.CODEX_MODEL || 'gpt-5.3-codex', { + authMode: 'host', + // On Windows, the Codex sandbox can block the folder (Access is + // denied). Then set CODEX_SANDBOX_MODE=danger-full-access, only for + // a folder you trust. + sandboxMode: + process.env.CODEX_SANDBOX_MODE === 'danger-full-access' + ? 'danger-full-access' + : 'workspace-write', + approvalPolicy: 'never', + }), + description: 'Codex. Quick fixes and tests in ./playground.', + }, + }, + }), + ] + : [] + export const assistant = defineHarness({ name: 'example/coder', description: 'A small coding agent that works in ./playground', @@ -174,5 +228,6 @@ export const assistant = defineHarness({ // The program runs in a QuickJS isolate. Any @tanstack/ai-isolate-* driver // works here. codeMode({ driver: createQuickJSIsolateDriver() }), + ...coding, ], }) diff --git a/packages/ai-acp/src/agent/index.ts b/packages/ai-acp/src/agent/index.ts index a9d3fc1af0..c36cb9ac8e 100644 --- a/packages/ai-acp/src/agent/index.ts +++ b/packages/ai-acp/src/agent/index.ts @@ -40,9 +40,11 @@ function promptText( /** One AG-UI chunk as an ACP session update, or `undefined` to skip it. */ export function toSessionUpdate(chunk: StreamChunk): SessionUpdate | undefined { - // Child agent work stays inside the harness. ACP shows the main turn. - if ('subagentRunId' in chunk && chunk.subagentRunId) return undefined + // A child agent's text and thoughts stay out of the main message. Its tool + // calls (the edits and commands it runs) show up like the lead's own. + const fromChild = 'subagentRunId' in chunk && Boolean(chunk.subagentRunId) if (chunk.type === EventType.TEXT_MESSAGE_CONTENT) { + if (fromChild) return undefined return { sessionUpdate: 'agent_message_chunk', messageId: chunk.messageId, @@ -50,6 +52,7 @@ export function toSessionUpdate(chunk: StreamChunk): SessionUpdate | undefined { } } if (chunk.type === EventType.REASONING_MESSAGE_CONTENT) { + if (fromChild) return undefined return { sessionUpdate: 'agent_thought_chunk', messageId: chunk.messageId, diff --git a/packages/ai-acp/tests/child-updates.test.ts b/packages/ai-acp/tests/child-updates.test.ts new file mode 100644 index 0000000000..fd67ee7526 --- /dev/null +++ b/packages/ai-acp/tests/child-updates.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from 'vitest' +import { EventType } from '@tanstack/ai' +import { toSessionUpdate } from '../src/agent' +import type { StreamChunk } from '@tanstack/ai' + +const child = (event: Record): StreamChunk => + ({ subagentRunId: 'child-1', timestamp: 1, ...event }) as StreamChunk + +describe('toSessionUpdate for child agents', () => { + it('shows child tool calls like the lead tool calls', () => { + expect( + toSessionUpdate( + child({ + type: EventType.TOOL_CALL_START, + toolCallId: 'edit-1', + toolCallName: 'Edit', + }), + ), + ).toEqual({ + sessionUpdate: 'tool_call_update', + toolCallId: 'edit-1', + name: 'Edit', + title: 'Edit', + status: 'in_progress', + }) + expect( + toSessionUpdate( + child({ + type: EventType.TOOL_CALL_RESULT, + messageId: 'm', + toolCallId: 'edit-1', + content: 'ok', + }), + ), + ).toEqual({ + sessionUpdate: 'tool_call_update', + toolCallId: 'edit-1', + status: 'completed', + }) + }) + + it('keeps child text and thoughts out of the main message', () => { + expect( + toSessionUpdate( + child({ + type: EventType.TEXT_MESSAGE_CONTENT, + messageId: 'm', + delta: 'hi', + }), + ), + ).toBeUndefined() + expect( + toSessionUpdate( + child({ + type: EventType.REASONING_MESSAGE_CONTENT, + messageId: 'r', + delta: 'hmm', + }), + ), + ).toBeUndefined() + }) +}) diff --git a/packages/ai-dashboard/src/ui.ts b/packages/ai-dashboard/src/ui.ts index 4f0b2fd6ae..76d2e0c840 100644 --- a/packages/ai-dashboard/src/ui.ts +++ b/packages/ai-dashboard/src/ui.ts @@ -89,7 +89,7 @@ function open(session) { function apply(frame) { if (frame.type !== 'harness.event' || !current) return const event = frame.event - if (event.subagentRunId) return + if (event.subagentRunId) { applyChild(event); return } if (event.type === 'TEXT_MESSAGE_CONTENT') { const last = current.messages[current.messages.length - 1] if (last && last.kind === 'assistant' && last.op === frame.operationId) last.text += event.delta @@ -109,6 +109,32 @@ function apply(frame) { } } +// A child agent shows as one block under the message it came from: its name, +// status, tool calls, and text. +function applyChild(event) { + const children = current.children || (current.children = new Map()) + if (event.type === 'SUBAGENT_STARTED') { + const node = { kind: 'agent', name: event.name || 'agent', status: 'working', tools: [], text: '' } + children.set(event.subagentRunId, node) + current.messages.push(node) + return + } + const node = children.get(event.subagentRunId) + if (!node) return + if (event.type === 'TEXT_MESSAGE_CONTENT') node.text += event.delta + else if (event.type === 'TOOL_CALL_START') node.tools.push(event.toolCallName) + else if (event.type === 'SUBAGENT_FINISHED') node.status = 'done' + else if (event.type === 'SUBAGENT_ERROR') { node.status = 'failed'; node.text += (node.text ? '\\n' : '') + event.message } +} + +function messageNode(message) { + if (message.kind !== 'agent') return $('div', { class: 'msg ' + message.kind }, message.text) + return $('div', { class: 'msg agent ' + message.status }, + $('strong', {}, 'agent ' + message.name + ' (' + message.status + ')'), + ...(message.tools.length ? [$('div', { class: 'tools' }, message.tools.map((tool) => 'tool ' + tool).join(', '))] : []), + ...(message.text ? [$('div', {}, message.text)] : [])) +} + async function send(input) { if (input.op === 'prompt' || input.op === 'steer') current.messages.push({ kind: 'user', text: input.message }) const receipt = await api(current.path + '/input', { method: 'POST', body: JSON.stringify({ input }) }) @@ -119,7 +145,7 @@ async function send(input) { function draw() { const main = document.getElementById('main') if (!current) { main.replaceChildren($('p', { class: 'muted' }, 'Pick a session.')); return } - const log = $('div', { class: 'log' }, ...current.messages.map((message) => $('div', { class: 'msg ' + message.kind }, message.text))) + const log = $('div', { class: 'log' }, ...current.messages.map(messageNode)) const actions = [] if (current.interrupts.length) { const decide = (approved) => { const resume = current.interrupts.map((interrupt) => ({ interruptId: interrupt.id, status: 'resolved', payload: approved })); current.interrupts = []; send({ op: 'resolve', resume }) } @@ -183,6 +209,7 @@ export const DASHBOARD_HTML = ` .msg { white-space: pre-wrap; padding: 10px 12px; border-radius: 10px; background: var(--panel); max-width: 80ch; } .msg.user { align-self: flex-end; background: #1d3557; } .msg.tool, .msg.notice { color: var(--muted); background: transparent; padding: 2px 12px; } .msg.error { border: 1px solid #e5484d; } + .msg.agent { border-left: 3px solid #7c5cff; display: flex; flex-direction: column; gap: 4px; } .msg.agent.failed { border-left-color: #e5484d; } .msg.agent .tools { color: var(--muted); font-size: 0.9em; } code { background: var(--panel); padding: 2px 6px; border-radius: 6px; } @media (max-width: 720px) { #app { grid-template-columns: 1fr; } aside { border-right: 0; border-bottom: 1px solid var(--line); } } diff --git a/packages/ai-dashboard/tests/child-agents.test.ts b/packages/ai-dashboard/tests/child-agents.test.ts new file mode 100644 index 0000000000..c5fcb36485 --- /dev/null +++ b/packages/ai-dashboard/tests/child-agents.test.ts @@ -0,0 +1,143 @@ +import { runInContext, createContext } from 'node:vm' +import { describe, expect, it } from 'vitest' +import { DASHBOARD_HTML } from '../src/ui' + +/** Just enough DOM for the page script: nodes with children and text. */ +class FakeNode { + children: Array = [] + className = '' + attributes: Record = {} + constructor( + readonly tag: string, + readonly text = '', + ) {} + append(...children: Array) { + this.children.push(...children) + } + replaceChildren(...children: Array) { + this.children = children + } + setAttribute(key: string, value: string) { + this.attributes[key] = value + } + get textContent(): string { + return this.text + this.children.map((child) => child.textContent).join('') + } +} + +function loadPage() { + const script = /