subharness 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +158 -0
- package/dist/adapters/claude-auth.d.ts +7 -0
- package/dist/adapters/claude-auth.js +87 -0
- package/dist/adapters/claude-auth.js.map +1 -0
- package/dist/adapters/claude-config.d.ts +6 -0
- package/dist/adapters/claude-config.js +37 -0
- package/dist/adapters/claude-config.js.map +1 -0
- package/dist/adapters/claude-input.d.ts +9 -0
- package/dist/adapters/claude-input.js +32 -0
- package/dist/adapters/claude-input.js.map +1 -0
- package/dist/adapters/claude-permissions.d.ts +3 -0
- package/dist/adapters/claude-permissions.js +24 -0
- package/dist/adapters/claude-permissions.js.map +1 -0
- package/dist/adapters/claude-settings.d.ts +14 -0
- package/dist/adapters/claude-settings.js +43 -0
- package/dist/adapters/claude-settings.js.map +1 -0
- package/dist/adapters/claude-tools.d.ts +9 -0
- package/dist/adapters/claude-tools.js +47 -0
- package/dist/adapters/claude-tools.js.map +1 -0
- package/dist/adapters/claude.d.ts +22 -0
- package/dist/adapters/claude.js +295 -0
- package/dist/adapters/claude.js.map +1 -0
- package/dist/adapters/codex-config.d.ts +25 -0
- package/dist/adapters/codex-config.js +66 -0
- package/dist/adapters/codex-config.js.map +1 -0
- package/dist/adapters/codex.d.ts +2 -0
- package/dist/adapters/codex.js +238 -0
- package/dist/adapters/codex.js.map +1 -0
- package/dist/adapters/fx-auth.d.ts +4 -0
- package/dist/adapters/fx-auth.js +116 -0
- package/dist/adapters/fx-auth.js.map +1 -0
- package/dist/adapters/fx-rpc.d.ts +29 -0
- package/dist/adapters/fx-rpc.js +124 -0
- package/dist/adapters/fx-rpc.js.map +1 -0
- package/dist/adapters/fx-tools.d.ts +19 -0
- package/dist/adapters/fx-tools.js +119 -0
- package/dist/adapters/fx-tools.js.map +1 -0
- package/dist/adapters/fx.d.ts +3 -0
- package/dist/adapters/fx.js +302 -0
- package/dist/adapters/fx.js.map +1 -0
- package/dist/adapters/mcp-tool-content.d.ts +12 -0
- package/dist/adapters/mcp-tool-content.js +8 -0
- package/dist/adapters/mcp-tool-content.js.map +1 -0
- package/dist/adapters/rpc.d.ts +28 -0
- package/dist/adapters/rpc.js +101 -0
- package/dist/adapters/rpc.js.map +1 -0
- package/dist/adapters/types.d.ts +28 -0
- package/dist/adapters/types.js +2 -0
- package/dist/adapters/types.js.map +1 -0
- package/dist/cli/args.d.ts +18 -0
- package/dist/cli/args.js +118 -0
- package/dist/cli/args.js.map +1 -0
- package/dist/cli/catalog-worker.d.ts +1 -0
- package/dist/cli/catalog-worker.js +22 -0
- package/dist/cli/catalog-worker.js.map +1 -0
- package/dist/cli/catalog.d.ts +26 -0
- package/dist/cli/catalog.js +41 -0
- package/dist/cli/catalog.js.map +1 -0
- package/dist/cli/help.d.ts +3 -0
- package/dist/cli/help.js +68 -0
- package/dist/cli/help.js.map +1 -0
- package/dist/cli/input.d.ts +5 -0
- package/dist/cli/input.js +69 -0
- package/dist/cli/input.js.map +1 -0
- package/dist/cli/main.d.ts +2 -0
- package/dist/cli/main.js +105 -0
- package/dist/cli/main.js.map +1 -0
- package/dist/cli/output.d.ts +7 -0
- package/dist/cli/output.js +43 -0
- package/dist/cli/output.js.map +1 -0
- package/dist/config/access.d.ts +21 -0
- package/dist/config/access.js +97 -0
- package/dist/config/access.js.map +1 -0
- package/dist/config/loader.d.ts +13 -0
- package/dist/config/loader.js +80 -0
- package/dist/config/loader.js.map +1 -0
- package/dist/config/oidc.d.ts +1 -0
- package/dist/config/oidc.js +53 -0
- package/dist/config/oidc.js.map +1 -0
- package/dist/config/project.d.ts +5 -0
- package/dist/config/project.js +56 -0
- package/dist/config/project.js.map +1 -0
- package/dist/config/resolve-access.d.ts +4 -0
- package/dist/config/resolve-access.js +42 -0
- package/dist/config/resolve-access.js.map +1 -0
- package/dist/errors.d.ts +9 -0
- package/dist/errors.js +14 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -0
- package/dist/runtime/client.d.ts +9 -0
- package/dist/runtime/client.js +91 -0
- package/dist/runtime/client.js.map +1 -0
- package/dist/runtime/coordinator.d.ts +61 -0
- package/dist/runtime/coordinator.js +296 -0
- package/dist/runtime/coordinator.js.map +1 -0
- package/dist/runtime/daemon.d.ts +1 -0
- package/dist/runtime/daemon.js +28 -0
- package/dist/runtime/daemon.js.map +1 -0
- package/dist/runtime/definition.d.ts +6 -0
- package/dist/runtime/definition.js +19 -0
- package/dist/runtime/definition.js.map +1 -0
- package/dist/runtime/native-owner.d.ts +8 -0
- package/dist/runtime/native-owner.js +25 -0
- package/dist/runtime/native-owner.js.map +1 -0
- package/dist/runtime/observation.d.ts +1 -0
- package/dist/runtime/observation.js +15 -0
- package/dist/runtime/observation.js.map +1 -0
- package/dist/runtime/select-native.d.ts +3 -0
- package/dist/runtime/select-native.js +34 -0
- package/dist/runtime/select-native.js.map +1 -0
- package/dist/runtime/service.d.ts +5 -0
- package/dist/runtime/service.js +90 -0
- package/dist/runtime/service.js.map +1 -0
- package/dist/runtime/session-launcher.d.ts +6 -0
- package/dist/runtime/session-launcher.js +29 -0
- package/dist/runtime/session-launcher.js.map +1 -0
- package/dist/runtime/state.d.ts +14 -0
- package/dist/runtime/state.js +69 -0
- package/dist/runtime/state.js.map +1 -0
- package/dist/runtime/types.d.ts +81 -0
- package/dist/runtime/types.js +2 -0
- package/dist/runtime/types.js.map +1 -0
- package/dist/runtime/worker-client.d.ts +3 -0
- package/dist/runtime/worker-client.js +81 -0
- package/dist/runtime/worker-client.js.map +1 -0
- package/dist/runtime/worker.d.ts +1 -0
- package/dist/runtime/worker.js +67 -0
- package/dist/runtime/worker.js.map +1 -0
- package/dist/sdk/definitions.d.ts +6 -0
- package/dist/sdk/definitions.js +72 -0
- package/dist/sdk/definitions.js.map +1 -0
- package/dist/sdk/tool-result.d.ts +23 -0
- package/dist/sdk/tool-result.js +95 -0
- package/dist/sdk/tool-result.js.map +1 -0
- package/dist/sdk/tools.d.ts +9 -0
- package/dist/sdk/tools.js +73 -0
- package/dist/sdk/tools.js.map +1 -0
- package/dist/sdk/types.d.ts +49 -0
- package/dist/sdk/types.js +2 -0
- package/dist/sdk/types.js.map +1 -0
- package/dist/sdk/validation.d.ts +5 -0
- package/dist/sdk/validation.js +20 -0
- package/dist/sdk/validation.js.map +1 -0
- package/dist/shell.d.ts +2 -0
- package/dist/shell.js +5 -0
- package/dist/shell.js.map +1 -0
- package/package.json +61 -0
- package/sdk/access-config.md +52 -0
- package/sdk/adapter-contract.md +33 -0
- package/sdk/agent.md +52 -0
- package/sdk/authentication.md +11 -0
- package/sdk/cli/index.md +67 -0
- package/sdk/cli/output.md +28 -0
- package/sdk/completion-notifications.md +29 -0
- package/sdk/config.md +41 -0
- package/sdk/distribution.md +31 -0
- package/sdk/fx.md +71 -0
- package/sdk/harnesses.md +32 -0
- package/sdk/index.md +22 -0
- package/sdk/message-delivery.md +39 -0
- package/sdk/plugins/sub-agents.md +57 -0
- package/sdk/project-team.md +57 -0
- package/sdk/sessions.md +31 -0
- package/sdk/tools.md +73 -0
- package/sdk/v1-runtime.md +63 -0
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Message Delivery
|
|
2
|
+
|
|
3
|
+
`subharness send <session-id> --delivery <queue|steer|interrupt> --prompt <text>` controls delivery. `queue` is the default. Each session has one active library task and a FIFO queue; a library task can contain multiple native turns while coordinating descendants.
|
|
4
|
+
|
|
5
|
+
| Mode | Effect |
|
|
6
|
+
| --- | --- |
|
|
7
|
+
| `queue` | Creates a task behind previously queued work |
|
|
8
|
+
| `steer` | Adds guidance to the current task, or uses interrupt when active native steering is unavailable |
|
|
9
|
+
| `interrupt` | Stops the active task and its descendants, then starts a replacement before independently queued work |
|
|
10
|
+
|
|
11
|
+
## Queue advancement
|
|
12
|
+
|
|
13
|
+
Queue dispatch follows task completion, not the return of a CLI command or every native turn. If A returns a response while a reviewer is running, A remains active and B waits. Once descendants and result processing finish, the parent's complete response releases B. The library uses native completion and task relationships, not text classification.
|
|
14
|
+
|
|
15
|
+
A complete response asking a question finishes a task when no descendant work remains. B can then start. The coordinator decides whether its answer should be queued or applied to the current task with another delivery mode. A native approval/input request is different: it has not completed a turn. In v1, input the library cannot answer under native policy fails with `INPUT_REQUIRED`; it is never automatically approved.
|
|
16
|
+
|
|
17
|
+
Execution failure pauses pending work. A successful explicit native recovery must finish the failed task before pending tasks proceed. Unsupported recovery leaves work paused. Queued command observers may remain waiting until their task starts or is explicitly cancelled.
|
|
18
|
+
|
|
19
|
+
## Steering
|
|
20
|
+
|
|
21
|
+
Native steering adds guidance without cancelling the active task or creating another task. The command acknowledges acceptance; it does not return a new agent response. Steering a task waiting between native turns starts a continuation of that task while its children keep running. No active task is an error.
|
|
22
|
+
|
|
23
|
+
If active native steering is unavailable, the operation uses the full interrupt behavior. It cancels the active task and descendants, creates a replacement task with the correction, and preserves independently queued tasks. The output identifies both requested `steer` and effective `interrupt`. It returns the replacement's response under interrupt semantics. Other native errors do not trigger this fallback.
|
|
24
|
+
|
|
25
|
+
There is no expected-task guard: if B has already become active, the correction applies to B. Unsupported steering is detected before native mutation so fallback does not deliver the correction twice.
|
|
26
|
+
|
|
27
|
+
## Interruption and cancellation
|
|
28
|
+
|
|
29
|
+
If A is active and B/C are queued, interrupting with X stops A and its delegated descendants, runs X, and then resumes B/C in their original order. A and its children are not automatically requeued. X cannot start while affected native execution might still be running.
|
|
30
|
+
|
|
31
|
+
Cancellation applies recursively to recorded delegation relationships. Pending descendants are removed without execution; active descendants receive native interruption. Other tasks in the same queue, directory, or agent definition are not descendants solely for that reason. Completed results and tool effects remain. Already-admitted custom callbacks must settle before stop confirmation.
|
|
32
|
+
|
|
33
|
+
`cancel` returns after affected work has stopped. Repeated cancellation of already-settled work returns its state. An unconfirmed stop reports a cancellation error and keeps dispatch paused; it does not authorize starting replacement or queued work. Invalid replacement input is rejected before existing work is cancelled.
|
|
34
|
+
|
|
35
|
+
## Resume after an execution error
|
|
36
|
+
|
|
37
|
+
`subharness resume <session-id>` addresses the existing failed task and requires no new prompt. It does not replay the original task, migrate to another harness, or skip the failure to release the queue. An unavailable native recovery operation returns `RECOVERY_UNSUPPORTED`. A session without failed work returns an error.
|
|
38
|
+
|
|
39
|
+
V1 native adapters do not currently provide prompt-free recovery of failed work, so the command reports that limit. The caller can inspect results and explicitly cancel or create new work. Cancelling settled failed work releases its queue when no execution remains uncertain.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# Subagents
|
|
2
|
+
|
|
3
|
+
An agent's optional `subagents` map declares and authorizes the agents it can invoke. Values are ordinary `AgentDefinition` objects; children can have their own subagents. Declaration alone does not execute them or register them in a repository/global catalog.
|
|
4
|
+
|
|
5
|
+
```ts
|
|
6
|
+
import { agent, codex } from "subharness";
|
|
7
|
+
|
|
8
|
+
const reviewer = agent({
|
|
9
|
+
name: "reviewer",
|
|
10
|
+
description: "Reviews changes for correctness.",
|
|
11
|
+
instructions: "Inspect the diff and report actionable findings.",
|
|
12
|
+
harness: codex({ model: "CODEX_MODEL_ID" }),
|
|
13
|
+
});
|
|
14
|
+
|
|
15
|
+
export default agent({
|
|
16
|
+
name: "developer",
|
|
17
|
+
description: "Implements and verifies features.",
|
|
18
|
+
instructions: "Implement the requested change, then ask reviewer to review it.",
|
|
19
|
+
harness: codex({ model: "CODEX_MODEL_ID" }),
|
|
20
|
+
subagents: { reviewer },
|
|
21
|
+
});
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## Invocation
|
|
25
|
+
|
|
26
|
+
The adapter provides direct child names, descriptions, and CLI instructions to the parent. Managed agents receive the absolute path of a launcher for their session and use that executable for delegation and follow-ups. This avoids invoking another installed program named `agent` when a native shell changes PATH.
|
|
27
|
+
|
|
28
|
+
The launcher restores its coordinator location and parent context before invoking the CLI, so delegation does not depend on the shell preserving integration environment variables. The launcher is stored with owner-only permissions and removed when its session worker closes. It contains only local orchestration context, never provider credentials. Commands and agent instructions do not contain parent tokens or credentials.
|
|
29
|
+
|
|
30
|
+
For Claude Code, declaring at least one subagent adds native permission rules for the current session's exact launcher. Equivalent quoted or unquoted spellings are allowed only when they identify the same literal executable. These rules authorize `run subagent:<name>` for the declared names, plus the CLI's `list`, `--help`, `send`, `wait`, `status`, `queue`, `cancel`, and `resume` operations. They do not authorize launching repository or global agents through `run`, another executable, or arbitrary shell commands. Follow-up and observation commands retain their documented identifier-based behavior; the permission rules do not introduce a new task-access policy.
|
|
31
|
+
|
|
32
|
+
The adapter combines these rules with declared custom-tool permissions. It does not write user/project permission settings or change the native permission mode. Native deny rules, explicit approval requirements, managed policy, and sandbox restrictions remain authoritative. A request that still requires approval returns `INPUT_REQUIRED`. Each child starts with its own native tool permissions; permission to launch it does not grant permission for its edits or shell commands. A session without declared subagents receives no delegation rules. If a launcher path cannot be represented as a literal command in the native permission syntax, startup fails with `INVALID_CONFIG` rather than adding a broader rule.
|
|
33
|
+
|
|
34
|
+
The native execution environment must allow the CLI to read its private launcher and contact the coordinator over loopback HTTP. A native sandbox that blocks local networking also blocks CLI delegation. The library does not relax that policy; the caller configures the native harness or external execution environment.
|
|
35
|
+
|
|
36
|
+
The command has the same arguments as an ordinary CLI invocation:
|
|
37
|
+
|
|
38
|
+
```sh
|
|
39
|
+
subharness run subagent:reviewer --cwd /repo/worktree \
|
|
40
|
+
--prompt "Review the implementation against the documented requirements."
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
The suffix is the local map key. It resolves only against the current parent's direct declarations, with no ancestor/global/repository fallback. Missing parent context or an undeclared name is an error. Parent context is passed privately by the integration; no public parent-token flag is required.
|
|
44
|
+
|
|
45
|
+
Each invocation creates a child session and task. The same `send`, `wait`, `status`, `queue`, `cancel`, and `resume` commands apply. A caller can instead coordinate independently discovered agents directly; that does not implicitly create a nested pipeline.
|
|
46
|
+
|
|
47
|
+
## Context and results
|
|
48
|
+
|
|
49
|
+
A child receives its own instructions and explicit task input, without copying the parent's transcript or forking its native session. The parent supplies requirements, decisions, and relevant file references, including any skill files the child needs to read. Previously loaded skill contents are not inherited. Native repository instructions and skill discovery follow the selected harness's rules; Subharness does not install native subagent definitions or skills. The caller supplies an existing directory; no sandbox or worktree is created.
|
|
50
|
+
|
|
51
|
+
Results return to the immediate caller. A child may itself return a pending response while its descendants run. A parent's task remains active until its descendant work and result processing finish, as defined in [response delivery](../completion-notifications.md). The library does not forward every child transcript to every ancestor.
|
|
52
|
+
|
|
53
|
+
Cancellation or interruption of a task propagates through its recorded descendant tree, waits for affected execution to stop, and preserves independently queued tasks. Parent execution failure also cancels outstanding descendants; uncertain cancellation keeps dispatch paused. Child failure is delivered as an outcome the parent can interpret rather than automatically failing the parent.
|
|
54
|
+
|
|
55
|
+
A follow-up sent from an active managed parent attaches the new child task to that current parent task. Reusing a child session does not attach new work to a previously completed parent. External follow-ups have no implicit parent. Delegation into the caller's own session or an ancestor session is rejected because it would deadlock completion.
|
|
56
|
+
|
|
57
|
+
Nesting is limited to eight levels. Existing tools and file effects are not rolled back by cancellation. These relationships coordinate work; they do not isolate concurrent edits in the same directory.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# Repository Agent Team
|
|
2
|
+
|
|
3
|
+
This repository keeps five reusable agents in `.agents/agents/`. An agent defines a stable responsibility, model, and tool access. A skill supplies task-specific instructions that the agent reads when relevant. Adding a technique or library does not require another agent definition.
|
|
4
|
+
|
|
5
|
+
## Roles
|
|
6
|
+
|
|
7
|
+
| Agent | Harness and model | Responsibility |
|
|
8
|
+
| --- | --- | --- |
|
|
9
|
+
| `architect` | Codex, `gpt-6-astra`, high effort | Evaluate architecture and API decisions; develop requested design concepts and assets; perform independent visual review when assigned. |
|
|
10
|
+
| `developer` | Codex, `gpt-5.6-sol`, high effort | Implement documented behavior and application structure with focused TDD and independent review. |
|
|
11
|
+
| `visual-engineer` | fx, `anthropic/claude-fable-5.1`, native default effort | Implement approved interface styling and native vgpu/WGSL effects using the relevant skill. |
|
|
12
|
+
| `researcher` | fx, `google/gemini-3.8-flash`, native default effort | Answer a bounded technical or design question with primary-source evidence. |
|
|
13
|
+
| `reviewer` | Claude Code, `claude-opus-5[1m]`, high effort | Independently review correctness, lifecycle, credential routing, and contract compliance without editing files. |
|
|
14
|
+
|
|
15
|
+
Model identities are explicit. There is no automatic model downgrade or change to a paid API route. Fast mode is disabled for Codex and Claude Code; fx retains its native preference because its ACP interface has no selector. Codex and Claude Code use eligible native subscriptions unless personal access settings select another route. The fx roles require explicit personal `access.fx` configuration as described in [fx](fx.md); Gateway team policies can restrict model access.
|
|
16
|
+
|
|
17
|
+
Skills are reusable across roles. The table records this repository's preferred allocation of work, not an SDK restriction on which agent may read a skill. A developer can use an interface skill for one task and a graphics skill for another without changing its definition or model. An omitted custom `tools` map does not disable the harness's native tools.
|
|
18
|
+
|
|
19
|
+
Roles do not grant native capabilities or permissions. Browsing, image generation, image viewing, Blender, file writes, and shell access depend on the selected harness and execution environment. A missing capability must be reported before claiming a result.
|
|
20
|
+
|
|
21
|
+
## Skills and task context
|
|
22
|
+
|
|
23
|
+
Repository skills live in `.agents/skills/<name>/SKILL.md`. The short index in `AGENTS.md` explains when each skill applies. Agents read only the relevant skill and supporting references for their current task. Skill contents are not concatenated into every agent's instructions.
|
|
24
|
+
|
|
25
|
+
These files are ordinary instructions consumed through the native harness. Subharness does not install skills, discover them on the caller's behalf, provide a `skills` definition field, or guarantee that every harness exposes the same skill command. Native automatic discovery depends on the harness and its settings. This repository's shared instructions explicitly require reading `AGENTS.md`, so the index remains usable through ordinary file reading even when a harness does not automatically discover `.agents/skills/`.
|
|
26
|
+
|
|
27
|
+
The caller provides one bounded deliverable, governing contracts, input artifacts, working directory, and relevant skill paths. Each child receives its own task and instructions; it does not inherit the parent's transcript or previously loaded skills. A handoff therefore names required skills again.
|
|
28
|
+
|
|
29
|
+
```sh
|
|
30
|
+
subharness list
|
|
31
|
+
subharness run repo:architect --prompt "Read website/design.md and .agents/skills/design-exploration/SKILL.md. Evaluate the supplied hero reference and report a composition recommendation."
|
|
32
|
+
subharness run repo:visual-engineer --prompt "Use .agents/skills/web-interface/SKILL.md. Implement the hero layout documented in website/design.md. Limit changes to the hero styles."
|
|
33
|
+
subharness run repo:developer --prompt "Use .agents/skills/web-interface/SKILL.md. Read website/design.md. Check the hero's semantic HTML and keyboard behavior without changing styles."
|
|
34
|
+
subharness run repo:developer --prompt "Use .agents/skills/webgpu-graphics/SKILL.md. Read website/graphics.md. Verify the mascot camera's documented framing with focused tests."
|
|
35
|
+
subharness run repo:researcher --prompt "Use .agents/skills/source-research/SKILL.md. Read sdk/fx.md and src/adapters/fx.ts. Explain cancellation behavior with file references."
|
|
36
|
+
subharness run repo:reviewer --prompt "Review the mascot camera changes against website/graphics.md. Report actionable findings without editing files."
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
The website's three specialist labels, QA Tester, Developer, and Designer, demonstrate possible role configurations. They are not an inventory of this repository's agent definitions.
|
|
40
|
+
|
|
41
|
+
## Delegation and review
|
|
42
|
+
|
|
43
|
+
CLI and runtime work uses the existing architect, developer, and reviewer roles. Separate developer sessions can own independent modules; scope and governing contracts distinguish their assignments without adding permanent roles. The architect evaluates native protocol constraints, the developer implements and integrates behavior with TDD, and the reviewer independently checks user-facing behavior, lifecycle, and credential routing. The lead coordinates contracts, assignments, and final verification. Implementation and fixes run through repository agents whenever supported, so follow-ups, queues, and review cycles exercise Subharness itself. Reproducible coordination failures are evidence for library improvements under the same contract and review rules.
|
|
44
|
+
|
|
45
|
+
The lead agent coordinates bounded tasks and resolves missing decisions under `AGENTS.md`. Skills do not authorize new features, public API shapes, permission changes, or undeclared children. Research findings and generated concepts are evidence, not implementation contracts. Source research and explorations belong in the ignored `.context/research/` directory.
|
|
46
|
+
|
|
47
|
+
`developer` retains a declared `reviewer` child for independent code review after code changes. It supplies the scope, governing documents, and relevant skill paths, fixes actionable findings, and requests re-review. The lead can explicitly own this review cycle, including when native permissions prevent the developer from reaching the coordinator. In that case the developer returns its changes and verification evidence without starting a duplicate review or retrying a known blocked launcher. The lead must still obtain independent review and return findings to the developer. A read-only verification task returns its evidence without starting an unnecessary review cycle. The lead separately assigns visual review when needed. If delegation unexpectedly fails, the developer reports the exact operation; returning evidence alone does not establish a passed review. The other roles return work to the lead for independently assigned review and do not declare children. A task-specific specialization can use a fresh session of an existing role with an explicit skill; it does not require a new globally discoverable role.
|
|
48
|
+
|
|
49
|
+
Delegation through either harness requires the native environment to read and execute the private session launcher and contact the local coordinator over loopback HTTP. Declaring a child does not override a Codex sandbox or native permission policy. The caller configures that access in the harness or external environment; Subharness does not change it. See [subagents](plugins/sub-agents.md) for the launcher and permission boundaries.
|
|
50
|
+
|
|
51
|
+
A `subagents` declaration authorizes Subharness child sessions as described in [subagents](plugins/sub-agents.md). It does not install native harness subagent definitions or grant the child extra tools. Independent review uses a different session from the implementation or design author. Visual review requires actual rendered captures; code review and generated mockups do not establish visual correctness.
|
|
52
|
+
|
|
53
|
+
## Visual evidence and native limits
|
|
54
|
+
|
|
55
|
+
The fx researcher and visual engineer expose a `view_image` custom tool. It returns native image content through `toolResult`, rather than a path or base64 text. It reads supported image files up to 3 MiB only inside the execution repository's `.context/research/landing/`, `.context/site-verification/`, and `apps/docs/public/` directories. Canonical paths, including symlink targets, must stay within those roots. Symlinked allowed roots cannot redirect access elsewhere. The tool does not fetch URLs or alter permissions.
|
|
56
|
+
|
|
57
|
+
Image tasks respect the native version limits in [fx](fx.md). When native fx is affected by the documented ACP image-result crash, the lead can supply images through native `fx ask --image` using the same role, model, and explicit connection. This is an external native-harness invocation, not a Subharness session or proof of CLI queue and coordinator behavior. Supported text tasks continue through Subharness. Missing capabilities never justify a silent model substitution or custom inference loop.
|
package/sdk/sessions.md
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Sessions and Tasks
|
|
2
|
+
|
|
3
|
+
An agent definition describes reusable behavior. A session is one native conversation in a fixed execution directory. A task is a unit of work within that session. Tasks and sessions have separate opaque identifiers and are not shell process IDs.
|
|
4
|
+
|
|
5
|
+
Each `run` creates a new session; `send` creates follow-up tasks or steers the active task. A session retains native conversation state and its loaded instructions, tools, and harness configuration between tasks. The external harness owns its context and compaction.
|
|
6
|
+
|
|
7
|
+
## State
|
|
8
|
+
|
|
9
|
+
| Task state | Meaning |
|
|
10
|
+
| --- | --- |
|
|
11
|
+
| `queued` | Admitted but not yet dispatched |
|
|
12
|
+
| `running` | Native startup or generation is active |
|
|
13
|
+
| `waiting` | Between native turns while delegated work or result processing remains |
|
|
14
|
+
| `completed` | Native response and descendant completion boundary are satisfied |
|
|
15
|
+
| `failed` | Execution or cancellation failed; queue dispatch is paused |
|
|
16
|
+
| `cancelled` | Explicit cancellation confirmed |
|
|
17
|
+
| `interrupted` | Replacement cancellation confirmed |
|
|
18
|
+
|
|
19
|
+
Failure is a terminal observation, not proof that uncertain native cancellation succeeded. Such execution remains owned and its queue stays paused until stop can be confirmed.
|
|
20
|
+
|
|
21
|
+
Every complete response receives a response identifier and returns control to its caller. A response may ask a question or report failing tests; neither its text nor command exit `0` proves objective success. The task can remain pending after that response if descendants or their results remain.
|
|
22
|
+
|
|
23
|
+
`wait --after` observes later responses without creating another task. `status` reports current state independently of earlier response snapshots. Completing a task does not delete its conversation. Returning a command does not cancel the task or descendants.
|
|
24
|
+
|
|
25
|
+
## Lifetime and recovery
|
|
26
|
+
|
|
27
|
+
The on-demand coordinator retains task records and native session workers beyond individual command exits. Task IDs resolve from other working directories using that coordinator. Records remain available for its lifetime; they are not silently evicted.
|
|
28
|
+
|
|
29
|
+
Restarting the coordinator does not restore queues or automatically replay tasks. Native harness history can exist separately, but this version does not reconstruct library sessions from it. Unavailable identifiers return an error. `resume` is limited to supported native recovery of a retained failed task; it does not imply recovery after coordinator loss.
|
|
30
|
+
|
|
31
|
+
Cancellation, interrupt propagation, and queue behavior are defined in [Message Delivery](message-delivery.md). The detailed ownership and bounds are defined in the [runtime contract](v1-runtime.md).
|
package/sdk/tools.md
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# Custom Tools
|
|
2
|
+
|
|
3
|
+
`tool({ description, inputSchema, execute })` defines a function that a native harness can invoke. The map key in `agent.tools` is its public name. Declaring a tool does not execute it.
|
|
4
|
+
|
|
5
|
+
```ts
|
|
6
|
+
import { agent, codex, tool } from "subharness";
|
|
7
|
+
import { z } from "zod";
|
|
8
|
+
|
|
9
|
+
const uppercase = tool({
|
|
10
|
+
description: "Convert text to uppercase.",
|
|
11
|
+
inputSchema: z.object({ text: z.string() }),
|
|
12
|
+
execute: ({ text }) => text.toUpperCase(),
|
|
13
|
+
});
|
|
14
|
+
|
|
15
|
+
export default agent({
|
|
16
|
+
name: "editor",
|
|
17
|
+
description: "Edits text using the supplied tools.",
|
|
18
|
+
instructions: "Use uppercase when the task requests uppercase text.",
|
|
19
|
+
harness: codex({ model: "CODEX_MODEL_ID" }),
|
|
20
|
+
tools: { uppercase },
|
|
21
|
+
});
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
All three fields are required. `description` is a nonempty string. `inputSchema` is a Zod 4 object schema representable as JSON Schema. `execute(input)` receives the parsed value, with TypeScript inference, and returns text, JSON-serializable data, or an explicit rich tool result synchronously or asynchronously. No second execution-context argument is supplied.
|
|
25
|
+
|
|
26
|
+
Validation runs before execution; invalid arguments never reach the function. Unsupported schemas fail during definition validation. Non-serializable or oversized outputs and thrown exceptions become tool errors. Tool results are limited to 1 MiB. Errors do not masquerade as successful outputs.
|
|
27
|
+
|
|
28
|
+
Tools run in their session worker's execution directory with that worker's environment. The adapter registers them in a native namespace without replacing the harness's native tools or model loop. Declaring a custom tool authorizes that agent to invoke it. Native tool permission policies remain with the harness.
|
|
29
|
+
|
|
30
|
+
## Image results
|
|
31
|
+
|
|
32
|
+
`toolResult({ content })` creates an explicit result containing text and image blocks. Plain objects remain JSON data even when they have a `content` property. This preserves existing tools and avoids guessing whether a JSON value is an image response.
|
|
33
|
+
|
|
34
|
+
```ts
|
|
35
|
+
import { readFile } from "node:fs/promises";
|
|
36
|
+
import { tool, toolResult } from "subharness";
|
|
37
|
+
import { z } from "zod";
|
|
38
|
+
|
|
39
|
+
const inspectReference = tool({
|
|
40
|
+
description: "View the approved design reference.",
|
|
41
|
+
inputSchema: z.object({}),
|
|
42
|
+
execute: async () => toolResult({
|
|
43
|
+
content: [
|
|
44
|
+
{ type: "text", text: "Approved hero reference." },
|
|
45
|
+
{
|
|
46
|
+
type: "image",
|
|
47
|
+
mimeType: "image/png",
|
|
48
|
+
data: (await readFile("references/hero.png")).toString("base64"),
|
|
49
|
+
},
|
|
50
|
+
],
|
|
51
|
+
}),
|
|
52
|
+
});
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
The public signatures are:
|
|
56
|
+
|
|
57
|
+
```ts
|
|
58
|
+
type ToolContent =
|
|
59
|
+
| { readonly type: "text"; readonly text: string }
|
|
60
|
+
| {
|
|
61
|
+
readonly type: "image";
|
|
62
|
+
readonly data: string;
|
|
63
|
+
readonly mimeType: "image/png" | "image/jpeg" | "image/webp" | "image/gif";
|
|
64
|
+
};
|
|
65
|
+
|
|
66
|
+
function toolResult(result: { readonly content: readonly ToolContent[] }): ToolResult;
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
`ToolResult` is an opaque value created by `toolResult`; callers must return it directly from `execute`. It is not a JSON transport format. Content must be nonempty, contain only the supported block fields, and include at most eight images. Image data is canonical padded standard base64 without a data-URL prefix. The decoded file signature must match its declared MIME type. Each encoded image is limited to 5 MiB; the complete rich result's JSON content is limited to 8 MiB, and combined text is limited to 1 MiB. Ordinary text and JSON results retain their 1 MiB limit.
|
|
70
|
+
|
|
71
|
+
The SDK validates rich results at the tool execution boundary. Malformed blocks report `TOOL_RESULT_INVALID`; exceeded limits report `TOOL_RESULT_TOO_LARGE`. Native adapters pass images as image content, never as base64 inside a text result: MCP image blocks for Claude Code and fx, and image content items for Codex. Model-specific image support and lower native limits still apply; a text-only model cannot perform visual inspection. The [fx contract](fx.md) records a native image-tool crash affecting fx 0.0.9. The tool author controls which files are read. This helper does not grant filesystem access or fetch remote images.
|
|
72
|
+
|
|
73
|
+
Cancellation waits for already-admitted callbacks to settle. There is no cancellation signal in `execute(input)`, and completed effects are not rolled back. Streaming tool results and arbitrary external tool-definition objects are outside this API.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# V1 Runtime Contract
|
|
2
|
+
|
|
3
|
+
The first usable version supplies a TypeScript definition SDK and the `subharness` CLI with Codex, Claude Code, and fx adapters. Other harnesses remain compatibility requirements outside this version's adapter set. The library coordinates existing harnesses; it does not implement a model loop, sandbox, worktree manager, or native permission system.
|
|
4
|
+
|
|
5
|
+
## Package and definitions
|
|
6
|
+
|
|
7
|
+
The package is `subharness`, with the primary executable `subharness` and identical `agent` compatibility alias. Node.js 22.18 or newer is required. ESM exports include `agent`, `tool`, `toolResult`, `codex`, `claudeCode`, `fx`, and their definition/configuration types. The public SDK defines reusable specialists; direct CLI harness targets also execute without a definition. Execution in v1 uses the CLI. Definition objects have readonly configuration fields and no execution methods. They contain functions and schemas and are not a serialization format.
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
import { agent, codex, claudeCode, tool } from "subharness";
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Names and tool/subagent map keys match `[a-zA-Z][a-zA-Z0-9_-]{0,63}`. Descriptions, instructions, and model identifiers must be nonempty strings. Unknown configuration fields are errors. Definitions cannot contain recursive subagent references. Harness arrays are nonempty and retain declaration order. `fast` defaults to `false`.
|
|
14
|
+
|
|
15
|
+
Custom tools accept Zod 4 object schemas that can be represented as JSON Schema. Input validation runs before execution. Functions return strings, JSON-serializable values, or explicit rich results created with `toolResult`. Rich results preserve image blocks through the native harness. Unsupported schemas and name collisions are rejected before a native turn starts. Tool exceptions and invalid arguments return tool errors; they do not masquerade as successful results. Declaring a tool authorizes its invocation by that agent. Native harness tools retain their native permission rules.
|
|
16
|
+
|
|
17
|
+
## Discovery and loading
|
|
18
|
+
|
|
19
|
+
Global definitions live in `~/.agents/agents/`; repository definitions live in `.agents/agents/` at the execution worktree root. Outside Git, the supplied directory is the project root. Bare repositories are unsupported execution directories. Nested repositories use the nearest Git worktree root. Symlink directories are not traversed during discovery.
|
|
20
|
+
|
|
21
|
+
Definitions are trusted executable TypeScript. Loading a catalog evaluates its modules, so callers must trust the selected project and global definition files. Imports resolve relative to each definition through normal Node package resolution, with TypeScript loading supplied by the CLI. Missing imports, invalid exports, and duplicate names fail with the source filename. A session retains its loaded instructions, tools, and harness configuration for follow-ups; new sessions load current definition sources.
|
|
22
|
+
|
|
23
|
+
An omitted `--cwd` uses the command's working directory. `--prompt-file` paths resolve against the calling command's directory, independently of `--cwd`; files use UTF-8. Prompts must contain non-whitespace text and cannot exceed 1 MiB. One quoted positional prompt, `--prompt`, and `--prompt-file` are mutually exclusive. `--prompt-file -` reads bounded UTF-8 stdin to EOF and rejects a terminal input stream. Direct harness targets accept `--model` and `--effort`; specialist targets reject those overrides.
|
|
24
|
+
|
|
25
|
+
## Execution ownership
|
|
26
|
+
|
|
27
|
+
An on-demand local coordinator owns sessions, queues, task relationships, and response records independently of individual CLI commands. It starts automatically; users do not install or manage a service. Each native session runs in a separate worker process whose working directory is the selected execution directory. Definition loading and custom tool execution in that worker use the same directory. Direct harness sessions carry a built-in target descriptor instead of a definition source reference and do not load TypeScript modules. They supply no specialist instructions, custom tools, or declared children; native instructions and project settings remain effective. Workers inherit the submitting environment, subject to the adapter's explicit authentication selection.
|
|
28
|
+
|
|
29
|
+
The coordinator accepts local connections only. Its endpoint and access token are stored with owner-only permissions under `~/.agents/state/`. Task and session identifiers resolve through this coordinator from any working directory. Child invocation context is supplied privately by a session launcher that restores the coordinator location and parent context even when native shells change the inherited environment. Launchers use owner-only permissions, contain no provider credentials, and are removed when the session worker closes. Users do not supply parent identifiers or tokens in command flags. A context is accepted only while its parent task allows delegation.
|
|
30
|
+
|
|
31
|
+
Closing a response-reading command does not cancel work. The coordinator retains responses and session records for its lifetime. A coordinator restart does not replay tasks or reconstruct queues; unavailable identifiers return an explicit error. Native harness history may still exist, but this version does not promise recovery of library sessions after coordinator or environment exit. Responses are not consumed by readers; each reader supplies its own position. Records are not silently evicted while the coordinator is running.
|
|
32
|
+
|
|
33
|
+
## Task states and responses
|
|
34
|
+
|
|
35
|
+
Task states are `queued`, `running`, `waiting`, `completed`, `failed`, `cancelled`, and `interrupted`. `waiting` means the library task remains active between native turns while delegated work or child-result processing is pending. A native permission request is not a completed response. Terminal outcomes are `completed`, `failed`, `cancelled`, and `interrupted`.
|
|
36
|
+
|
|
37
|
+
Each complete native response has an opaque response identifier, text, and the task state at that response. `wait --after` returns the next retained response in order, including one that arrived before the command started. If none remains and the task is terminal, it returns the terminal outcome. Unknown response identifiers, including identifiers from another task, are errors. Multiple observers can independently read the same response.
|
|
38
|
+
|
|
39
|
+
## Queue and recovery rules
|
|
40
|
+
|
|
41
|
+
Each session admits task-creating requests in the order the coordinator receives them. It executes one native turn at a time. Normal task completion dispatches the next queued task. A task failure preserves and pauses the queue; a successful explicit `resume` of the failed task releases it only after that task completes. Native recovery that cannot continue the failed work without replay is reported as unsupported. Calling `resume` without a failed task is an error.
|
|
42
|
+
|
|
43
|
+
Steering an actively generating native turn uses native steering when available; otherwise it follows the interrupt contract: cancel the active task and its descendants, then create a replacement task with the supplied message while preserving independently queued work. The output reports the effective delivery mode. Steering a `waiting` task starts a native continuation of that same task with the new input, leaving descendants running. With no active task, steering fails; callers can submit a queued task. Acceptance means the native integration accepted the input, not that the model has acted on it.
|
|
44
|
+
|
|
45
|
+
Cancellation closes admission of new descendants, removes queued descendants, and requests native interruption for running descendants. `cancel` returns after affected executions have stopped. Repeating cancellation of a terminal task is a harmless status read. Interruption uses the same boundary before the replacement starts; independently queued tasks keep their order. A failed cancellation leaves dispatch paused and returns an error rather than claiming execution stopped. Concurrent interrupts are serialized in admission order.
|
|
46
|
+
|
|
47
|
+
On parent execution failure, affected descendants are cancelled before reporting the settled failure. A failed or cancelled child is delivered to its parent as an outcome the parent can interpret; it does not automatically fail the parent. Parent continuation uses the existing native session and original task. Child results received while the parent is busy are retained until they can be delivered without interrupting its turn. Results available together may be delivered in one continuation, once each.
|
|
48
|
+
|
|
49
|
+
When a child CLI command returns its terminal response to a still-running parent, that response already reaches the parent's native tool interaction. The coordinator marks it delivered and does not duplicate it in another turn. A child that finishes after its earlier pending response was returned requires explicit result delivery and continuation. The coordinator tracks delivery receipts, not the text's meaning.
|
|
50
|
+
|
|
51
|
+
New follow-up tasks in a child session belong to the currently invoking parent task when invoked from an active managed parent. They do not remain attached to a completed task merely because the child session was originally created there. External follow-ups have no implicit parent. Cancellation affects recorded task lineage, not every task in a session.
|
|
52
|
+
|
|
53
|
+
## Permissions and bounds
|
|
54
|
+
|
|
55
|
+
Adapters preserve native permission restrictions. Declaring subagents authorizes their invocation: Claude Code receives session-only native allow rules for the exact launcher and documented delegation operations, combined with permissions for declared custom tools. This does not change the native permission mode, user/project settings, explicit deny or approval rules, managed policy, or sandbox. This version does not provide an interactive permission-answer CLI: a native approval or input request that still cannot be handled under that policy fails the task with `INPUT_REQUIRED` and actionable guidance. The permission callback never automatically approves the request. The caller can configure its harness and explicitly recover where supported.
|
|
56
|
+
|
|
57
|
+
Ordinary text/JSON tool results and complete responses are limited to 1 MiB. Rich tool results have the image and aggregate limits defined in [custom tools](tools.md). Oversized results produce explicit errors rather than silent truncation. `status` includes at most 4,000 characters of the latest response and marks truncation; `status --full` retrieves the complete latest response and `wait` observes subsequent complete responses. Delegation depth is limited to 8, and each session admits at most 100 pending tasks. Exceeding a bound fails admission without dropping existing work. No automatic task-duration deadline is imposed.
|
|
58
|
+
|
|
59
|
+
A managed task cannot delegate follow-up work into its own session or an ancestor's session. Such an admission returns `INVALID_PARENT` before changing any execution: that queued work would otherwise wait for the very task whose completion depends on it. Invalid delegation and replacement requests never cancel existing work.
|
|
60
|
+
|
|
61
|
+
Coordinator startup recovers stale local startup locks whose recorded owner is absent or invalid. It never removes a lock owned by a live process and rechecks the file before removing it. An interrupted coordinator response is reported as unavailable without exposing native diagnostics or automatically repeating the request. Disconnecting an observer detaches that observation; it does not cancel execution.
|
|
62
|
+
|
|
63
|
+
Catalog discovery evaluates TypeScript in a short-lived process with the selected execution directory as its working directory. Module stdout and stderr are not mixed into CLI output. Session workers retain their loaded instructions, tools, and harness configuration. New sessions, including delegated child sessions, load the current definition source; the library does not serialize or freeze arbitrary TypeScript closures across sessions.
|