subharness 0.0.4 → 0.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -9
- package/dist/adapters/claude-process.d.ts +3 -0
- package/dist/adapters/claude-process.js +19 -2
- package/dist/adapters/claude-process.js.map +1 -1
- package/dist/adapters/claude-result.d.ts +2 -0
- package/dist/adapters/claude-result.js +22 -0
- package/dist/adapters/claude-result.js.map +1 -0
- package/dist/adapters/claude-tools.d.ts +6 -2
- package/dist/adapters/claude-tools.js +14 -12
- package/dist/adapters/claude-tools.js.map +1 -1
- package/dist/adapters/claude-worker-client.d.ts +49 -0
- package/dist/adapters/claude-worker-client.js +359 -0
- package/dist/adapters/claude-worker-client.js.map +1 -0
- package/dist/adapters/claude-worker-process.d.ts +20 -0
- package/dist/adapters/claude-worker-process.js +76 -0
- package/dist/adapters/claude-worker-process.js.map +1 -0
- package/dist/adapters/claude-worker-protocol.d.ts +38 -0
- package/dist/adapters/claude-worker-protocol.js +2 -0
- package/dist/adapters/claude-worker-protocol.js.map +1 -0
- package/dist/adapters/claude-worker.d.ts +1 -0
- package/dist/adapters/claude-worker.js +126 -0
- package/dist/adapters/claude-worker.js.map +1 -0
- package/dist/adapters/claude.d.ts +16 -6
- package/dist/adapters/claude.js +164 -75
- package/dist/adapters/claude.js.map +1 -1
- package/dist/adapters/codex.js +4 -1
- package/dist/adapters/codex.js.map +1 -1
- package/dist/adapters/copilot-permissions.d.ts +26 -0
- package/dist/adapters/copilot-permissions.js +121 -0
- package/dist/adapters/copilot-permissions.js.map +1 -0
- package/dist/adapters/copilot-tools.d.ts +12 -0
- package/dist/adapters/copilot-tools.js +61 -0
- package/dist/adapters/copilot-tools.js.map +1 -0
- package/dist/adapters/copilot.d.ts +108 -0
- package/dist/adapters/copilot.js +819 -0
- package/dist/adapters/copilot.js.map +1 -0
- package/dist/adapters/cursor-cli-approvals.d.ts +2 -0
- package/dist/adapters/cursor-cli-approvals.js +83 -0
- package/dist/adapters/cursor-cli-approvals.js.map +1 -0
- package/dist/adapters/cursor-cli-model.d.ts +4 -0
- package/dist/adapters/cursor-cli-model.js +64 -0
- package/dist/adapters/cursor-cli-model.js.map +1 -0
- package/dist/adapters/cursor-cli-session.d.ts +32 -0
- package/dist/adapters/cursor-cli-session.js +316 -0
- package/dist/adapters/cursor-cli-session.js.map +1 -0
- package/dist/adapters/cursor-cli-tools.d.ts +20 -0
- package/dist/adapters/cursor-cli-tools.js +181 -0
- package/dist/adapters/cursor-cli-tools.js.map +1 -0
- package/dist/adapters/cursor-cli.d.ts +2 -0
- package/dist/adapters/cursor-cli.js +306 -0
- package/dist/adapters/cursor-cli.js.map +1 -0
- package/dist/adapters/cursor-model.d.ts +7 -0
- package/dist/adapters/cursor-model.js +67 -0
- package/dist/adapters/cursor-model.js.map +1 -0
- package/dist/adapters/cursor-rpc.d.ts +42 -0
- package/dist/adapters/cursor-rpc.js +271 -0
- package/dist/adapters/cursor-rpc.js.map +1 -0
- package/dist/adapters/cursor-session.d.ts +4 -0
- package/dist/adapters/cursor-session.js +327 -0
- package/dist/adapters/cursor-session.js.map +1 -0
- package/dist/adapters/cursor-startup.d.ts +16 -0
- package/dist/adapters/cursor-startup.js +74 -0
- package/dist/adapters/cursor-startup.js.map +1 -0
- package/dist/adapters/cursor-tools.d.ts +11 -0
- package/dist/adapters/cursor-tools.js +49 -0
- package/dist/adapters/cursor-tools.js.map +1 -0
- package/dist/adapters/cursor.d.ts +6 -0
- package/dist/adapters/cursor.js +17 -0
- package/dist/adapters/cursor.js.map +1 -0
- package/dist/adapters/fx-auth.d.ts +12 -2
- package/dist/adapters/fx-auth.js +51 -61
- package/dist/adapters/fx-auth.js.map +1 -1
- package/dist/adapters/fx-profile.d.ts +8 -0
- package/dist/adapters/fx-profile.js +101 -0
- package/dist/adapters/fx-profile.js.map +1 -0
- package/dist/adapters/fx-rpc.d.ts +2 -0
- package/dist/adapters/fx-rpc.js +30 -4
- package/dist/adapters/fx-rpc.js.map +1 -1
- package/dist/adapters/fx-status.d.ts +2 -0
- package/dist/adapters/fx-status.js +96 -0
- package/dist/adapters/fx-status.js.map +1 -0
- package/dist/adapters/fx.js +28 -12
- package/dist/adapters/fx.js.map +1 -1
- package/dist/adapters/opencode-access.d.ts +13 -0
- package/dist/adapters/opencode-access.js +76 -0
- package/dist/adapters/opencode-access.js.map +1 -0
- package/dist/adapters/opencode-config.d.ts +11 -0
- package/dist/adapters/opencode-config.js +238 -0
- package/dist/adapters/opencode-config.js.map +1 -0
- package/dist/adapters/opencode-http.d.ts +28 -0
- package/dist/adapters/opencode-http.js +297 -0
- package/dist/adapters/opencode-http.js.map +1 -0
- package/dist/adapters/opencode-tools.d.ts +19 -0
- package/dist/adapters/opencode-tools.js +127 -0
- package/dist/adapters/opencode-tools.js.map +1 -0
- package/dist/adapters/opencode.d.ts +2 -0
- package/dist/adapters/opencode.js +569 -0
- package/dist/adapters/opencode.js.map +1 -0
- package/dist/adapters/rpc.d.ts +3 -0
- package/dist/adapters/rpc.js +45 -4
- package/dist/adapters/rpc.js.map +1 -1
- package/dist/adapters/types.d.ts +5 -0
- package/dist/adapters/types.js.map +1 -1
- package/dist/approvals/types.d.ts +1 -1
- package/dist/approvals/types.js.map +1 -1
- package/dist/cli/args.d.ts +1 -1
- package/dist/cli/args.js +4 -1
- package/dist/cli/args.js.map +1 -1
- package/dist/cli/catalog-worker.js.map +1 -1
- package/dist/cli/dashboard-client.js +20 -31
- package/dist/cli/dashboard-client.js.map +1 -1
- package/dist/cli/dashboard-controller.d.ts +7 -0
- package/dist/cli/dashboard-controller.js +27 -0
- package/dist/cli/dashboard-controller.js.map +1 -0
- package/dist/cli/dashboard-detail-view.d.ts +1 -1
- package/dist/cli/dashboard-detail-view.js +3 -4
- package/dist/cli/dashboard-detail-view.js.map +1 -1
- package/dist/cli/dashboard-history-view.js.map +1 -1
- package/dist/cli/dashboard-input.d.ts +1 -1
- package/dist/cli/dashboard-input.js +11 -1
- package/dist/cli/dashboard-input.js.map +1 -1
- package/dist/cli/dashboard-layout.js +3 -3
- package/dist/cli/dashboard-layout.js.map +1 -1
- package/dist/cli/dashboard-renderer.js.map +1 -1
- package/dist/cli/dashboard-style.js +2 -2
- package/dist/cli/dashboard-style.js.map +1 -1
- package/dist/cli/dashboard.d.ts +2 -2
- package/dist/cli/dashboard.js +7 -32
- package/dist/cli/dashboard.js.map +1 -1
- package/dist/cli/help.d.ts +1 -1
- package/dist/cli/help.js +30 -12
- package/dist/cli/help.js.map +1 -1
- package/dist/cli/main.js +55 -43
- package/dist/cli/main.js.map +1 -1
- package/dist/config/access-provenance.d.ts +10 -0
- package/dist/config/access-provenance.js +31 -0
- package/dist/config/access-provenance.js.map +1 -0
- package/dist/config/access.d.ts +12 -2
- package/dist/config/access.js +125 -27
- package/dist/config/access.js.map +1 -1
- package/dist/config/loader.js +1 -0
- package/dist/config/loader.js.map +1 -1
- package/dist/config/oidc.js +4 -2
- package/dist/config/oidc.js.map +1 -1
- package/dist/config/project.js +2 -1
- package/dist/config/project.js.map +1 -1
- package/dist/config/resolve-access.js +24 -10
- package/dist/config/resolve-access.js.map +1 -1
- package/dist/errors.d.ts +2 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +8 -2
- package/dist/index.js +4 -1
- package/dist/index.js.map +1 -1
- package/dist/process-diagnostics.d.ts +2 -0
- package/dist/process-diagnostics.js +11 -0
- package/dist/process-diagnostics.js.map +1 -0
- package/dist/runtime/access-errors.d.ts +4 -0
- package/dist/runtime/access-errors.js +25 -0
- package/dist/runtime/access-errors.js.map +1 -0
- package/dist/runtime/approval-registry.d.ts +9 -0
- package/dist/runtime/approval-registry.js +144 -5
- package/dist/runtime/approval-registry.js.map +1 -1
- package/dist/runtime/capture.d.ts +22 -0
- package/dist/runtime/capture.js +320 -0
- package/dist/runtime/capture.js.map +1 -0
- package/dist/runtime/catalog.d.ts +39 -0
- package/dist/runtime/catalog.js +84 -0
- package/dist/runtime/catalog.js.map +1 -0
- package/dist/runtime/client.d.ts +2 -1
- package/dist/runtime/client.js +74 -47
- package/dist/runtime/client.js.map +1 -1
- package/dist/runtime/coordinator.d.ts +44 -7
- package/dist/runtime/coordinator.js +265 -31
- package/dist/runtime/coordinator.js.map +1 -1
- package/dist/runtime/daemon.js +12 -6
- package/dist/runtime/daemon.js.map +1 -1
- package/dist/runtime/dashboard-sanitize.d.ts +2 -0
- package/dist/runtime/dashboard-sanitize.js +7 -0
- package/dist/runtime/dashboard-sanitize.js.map +1 -0
- package/dist/runtime/dashboard-workspace.js +5 -1
- package/dist/runtime/dashboard-workspace.js.map +1 -1
- package/dist/runtime/dashboard.d.ts +1 -1
- package/dist/runtime/dashboard.js +2 -3
- package/dist/runtime/dashboard.js.map +1 -1
- package/dist/runtime/definition.d.ts +6 -1
- package/dist/runtime/definition.js +42 -2
- package/dist/runtime/definition.js.map +1 -1
- package/dist/runtime/native-owner.js +7 -1
- package/dist/runtime/native-owner.js.map +1 -1
- package/dist/runtime/pagination.d.ts +9 -0
- package/dist/runtime/pagination.js +36 -0
- package/dist/runtime/pagination.js.map +1 -0
- package/dist/runtime/sdk-observation.d.ts +8 -0
- package/dist/runtime/sdk-observation.js +75 -0
- package/dist/runtime/sdk-observation.js.map +1 -0
- package/dist/runtime/sdk-projection.d.ts +8 -0
- package/dist/runtime/sdk-projection.js +19 -0
- package/dist/runtime/sdk-projection.js.map +1 -0
- package/dist/runtime/sdk-service.d.ts +11 -0
- package/dist/runtime/sdk-service.js +109 -0
- package/dist/runtime/sdk-service.js.map +1 -0
- package/dist/runtime/select-native.d.ts +4 -1
- package/dist/runtime/select-native.js +28 -13
- package/dist/runtime/select-native.js.map +1 -1
- package/dist/runtime/service.d.ts +11 -1
- package/dist/runtime/service.js +178 -3
- package/dist/runtime/service.js.map +1 -1
- package/dist/runtime/session-launcher.js +1 -1
- package/dist/runtime/session-launcher.js.map +1 -1
- package/dist/runtime/snapshots.d.ts +103 -0
- package/dist/runtime/snapshots.js +124 -0
- package/dist/runtime/snapshots.js.map +1 -0
- package/dist/runtime/state.d.ts +26 -0
- package/dist/runtime/state.js +86 -17
- package/dist/runtime/state.js.map +1 -1
- package/dist/runtime/task-data.d.ts +38 -0
- package/dist/runtime/task-data.js +2 -0
- package/dist/runtime/task-data.js.map +1 -0
- package/dist/runtime/transport.d.ts +13 -0
- package/dist/runtime/transport.js +166 -0
- package/dist/runtime/transport.js.map +1 -0
- package/dist/runtime/types.d.ts +28 -9
- package/dist/runtime/types.js.map +1 -1
- package/dist/runtime/worker-client.js +40 -13
- package/dist/runtime/worker-client.js.map +1 -1
- package/dist/runtime/worker-server.js +105 -24
- package/dist/runtime/worker-server.js.map +1 -1
- package/dist/sdk/connected-types.d.ts +42 -0
- package/dist/sdk/connected-types.js +2 -0
- package/dist/sdk/connected-types.js.map +1 -0
- package/dist/sdk/connected.d.ts +3 -0
- package/dist/sdk/connected.js +232 -0
- package/dist/sdk/connected.js.map +1 -0
- package/dist/sdk/definitions.d.ts +4 -1
- package/dist/sdk/definitions.js +16 -2
- package/dist/sdk/definitions.js.map +1 -1
- package/dist/sdk/execution-definition.d.ts +3 -0
- package/dist/sdk/execution-definition.js +59 -0
- package/dist/sdk/execution-definition.js.map +1 -0
- package/dist/sdk/execution-driver.d.ts +6 -0
- package/dist/sdk/execution-driver.js +86 -0
- package/dist/sdk/execution-driver.js.map +1 -0
- package/dist/sdk/execution-observation.d.ts +3 -0
- package/dist/sdk/execution-observation.js +35 -0
- package/dist/sdk/execution-observation.js.map +1 -0
- package/dist/sdk/execution-options.d.ts +17 -0
- package/dist/sdk/execution-options.js +115 -0
- package/dist/sdk/execution-options.js.map +1 -0
- package/dist/sdk/execution-types.d.ts +72 -0
- package/dist/sdk/execution-types.js +2 -0
- package/dist/sdk/execution-types.js.map +1 -0
- package/dist/sdk/execution.d.ts +3 -0
- package/dist/sdk/execution.js +3 -0
- package/dist/sdk/execution.js.map +1 -0
- package/dist/sdk/hosted-error.d.ts +3 -0
- package/dist/sdk/hosted-error.js +14 -0
- package/dist/sdk/hosted-error.js.map +1 -0
- package/dist/sdk/hosted.d.ts +23 -0
- package/dist/sdk/hosted.js +204 -0
- package/dist/sdk/hosted.js.map +1 -0
- package/dist/sdk/permission-validation.js +10 -1
- package/dist/sdk/permission-validation.js.map +1 -1
- package/dist/sdk/runner.d.ts +5 -0
- package/dist/sdk/runner.js +107 -0
- package/dist/sdk/runner.js.map +1 -0
- package/dist/sdk/tools.js +4 -1
- package/dist/sdk/tools.js.map +1 -1
- package/dist/sdk/types.d.ts +29 -1
- package/dist/sdk/types.js.map +1 -1
- package/package.json +6 -3
- package/sdk/access-config.md +60 -4
- package/sdk/adapter-contract.md +13 -3
- package/sdk/additional-harnesses.md +43 -0
- package/sdk/agent-skill.md +2 -2
- package/sdk/agent.md +6 -4
- package/sdk/approvals.md +5 -1
- package/sdk/authentication.md +1 -1
- package/sdk/cli/dashboard-design.md +6 -0
- package/sdk/cli/index.md +9 -6
- package/sdk/cli/output.md +3 -1
- package/sdk/completion-notifications.md +2 -0
- package/sdk/config.md +1 -1
- package/sdk/copilot.md +53 -0
- package/sdk/cursor.md +68 -0
- package/sdk/diagnostics.md +29 -0
- package/sdk/distribution.md +5 -3
- package/sdk/evals.md +1 -1
- package/sdk/examples/chat-tool.ts +126 -0
- package/sdk/execution.md +743 -0
- package/sdk/fx.md +50 -5
- package/sdk/harnesses.md +13 -13
- package/sdk/index.md +45 -20
- package/sdk/message-delivery.md +2 -0
- package/sdk/opencode.md +57 -0
- package/sdk/permissions.md +6 -2
- package/sdk/plugins/sub-agents.md +4 -2
- package/sdk/project-team.md +25 -1
- package/sdk/sessions.md +2 -0
- package/sdk/tools.md +4 -2
- package/sdk/v1-runtime.md +8 -6
- package/dist/cli/catalog.d.ts +0 -26
- package/dist/cli/catalog.js +0 -41
- package/dist/cli/catalog.js.map +0 -1
package/sdk/cli/index.md
CHANGED
|
@@ -46,7 +46,7 @@ subharness / trenton
|
|
|
46
46
|
◌ claude Review the current diff waiting 00:38
|
|
47
47
|
```
|
|
48
48
|
|
|
49
|
-
The harness label is `codex`, `claude`, or `
|
|
49
|
+
The harness label is `codex`, `claude`, `fx`, `opencode`, `copilot`, or `cursor`. A direct target is known during startup; a specialist displays `pending` until its actual native harness is selected. Selection reports the successfully opened harness, including fallback selection. The title is the first nonempty line of the current task's original prompt, with whitespace normalized and terminal control sequences removed, bounded to 120 Unicode code points before fitting it to the display. An empty sanitized title is `(untitled)`. No model call generates titles. Steering retains the task title; a follow-up uses its own prompt.
|
|
50
50
|
|
|
51
51
|
Time is elapsed wall time since that task first dispatched, including waiting and recovery. A queued task shows `00:00`. Completed, cancelled, and interrupted tasks freeze their elapsed time when that outcome is reached. Failed tasks freeze their elapsed time at failure and resume counting from the original start if explicitly recovered. Tasks cancelled before first dispatch show `00:00`. Repeated observation or cancellation that leaves a settled terminal outcome unchanged does not change its recorded finish time. If stopping previously unconfirmed execution reaches a new outcome, timing records that new outcome. Durations use `MM:SS` below one hour and `H:MM:SS` thereafter. State and elapsed time occupy separate aligned columns. Finished rows include completion age when space permits, as defined in [Dashboard presentation](dashboard-design.md). Narrow terminals truncate titles by display width without splitting grapheme clusters; when the fixed fields alone do not fit, the row is clipped to the available columns.
|
|
52
52
|
|
|
@@ -58,25 +58,28 @@ The command requires terminal stdout and a terminal supporting cursor control; n
|
|
|
58
58
|
|
|
59
59
|
## Direct harnesses and specialists
|
|
60
60
|
|
|
61
|
-
`codex`, `claude`, and `
|
|
61
|
+
`codex`, `claude`, `fx`, `opencode`, `copilot`, and `cursor` are reserved, case-sensitive target names for direct harness execution. `claude` selects the Claude Code adapter, whose SDK and access configuration key remains `claudeCode`. Direct execution needs its native runtime and eligible access, but no TypeScript definition, project-local SDK import, or agent catalog evaluation. An invalid specialist file does not block a direct harness run.
|
|
62
62
|
|
|
63
63
|
```sh
|
|
64
64
|
subharness run claude "Review the current diff."
|
|
65
65
|
subharness run codex --cwd ../feature-worktree "Implement the documented validation."
|
|
66
66
|
subharness run fx --model "provider/model" "Compare the proposed implementations."
|
|
67
|
+
subharness run opencode --model "creator/model" "Inspect the current implementation."
|
|
68
|
+
subharness run copilot --model "creator/model" "Review the current diff."
|
|
69
|
+
subharness run cursor --model "CURSOR_MODEL_ID" "Implement the documented change."
|
|
67
70
|
subharness run repo:reviewer "Review the current diff."
|
|
68
71
|
subharness check repo:reviewer --format jsonl
|
|
69
72
|
```
|
|
70
73
|
|
|
71
|
-
Replace `provider/model` with an available Gateway model identifier. Nonreserved bare names retain the existing unambiguous specialist lookup. Qualified `repo:`, `global:`, and `subagent:` names retain their existing meanings. A specialist named
|
|
74
|
+
Replace `provider/model` or `creator/model` with an available Gateway model identifier. Nonreserved bare names retain the existing unambiguous specialist lookup. Qualified `repo:`, `global:`, and `subagent:` names retain their existing meanings. A specialist named for any reserved harness target requires its scope qualifier; it never shadows the built-in target. Unknown targets fail rather than being executed as arbitrary commands. Omitting the target is an argument error; the CLI never chooses the first installed harness.
|
|
72
75
|
|
|
73
76
|
Direct harness sessions use native instructions and project context without adding a specialist role, custom tools, or declared children. They preserve the existing authentication, native permission, queue, cancellation, follow-up, and response contracts. They do not broaden the caller's native permissions or automatically authorize additional delegation.
|
|
74
77
|
|
|
75
|
-
`--model`
|
|
78
|
+
`--model` configures direct harness targets only for `run` and `check`, and can be combined with `--cwd` and, for `run`, any supported prompt source. `--effort` is additionally available for Codex, Claude Code, and fx. Empty values are errors. Specialist targets reject these overrides and retain their declared harness configurations. `send` retains the session configuration and does not accept model or effort overrides.
|
|
76
79
|
|
|
77
|
-
|
|
80
|
+
For Codex, Claude Code, and fx, omitting `--model` requests the native default for the authorized access route at session creation. OpenCode, Copilot, and Cursor require an explicit `--model`; omission fails with `INVALID_CONFIG` and guidance before native startup. The adapter retains the selected model for that conversation. It does not inherit the calling agent's model, rank models, choose a substitute, or change billing routes. An unavailable or unverifiable native default fails with actionable guidance to supply `--model`; it does not submit a prompt merely to discover a default. Explicit model identifiers retain native validation and substitution checks. TypeScript harness constructors continue to require a model.
|
|
78
81
|
|
|
79
|
-
Effort is harness-specific: Codex and fx accept native effort identifiers, while Claude Code accepts `low`, `medium`, `high`, `xhigh`, or `max`. Explicit effort must be compatible with the selected model where native capabilities expose that validation. An omitted effort retains the native default at session creation. Detected changes to an explicit request are errors. Codex and Claude Code retain standard-speed execution; this interface adds no fast-mode flag. fx retains its documented native preference and Gateway access contract.
|
|
82
|
+
Effort is harness-specific: Codex and fx accept native effort identifiers, while Claude Code accepts `low`, `medium`, `high`, `xhigh`, or `max`. OpenCode, Copilot, and Cursor reject `--effort`. Explicit effort must be compatible with the selected model where native capabilities expose that validation. An omitted effort retains the native default at session creation. Detected changes to an explicit request are errors. Codex and Claude Code retain standard-speed execution; this interface adds no fast-mode flag. fx retains its documented native preference and Gateway access contract.
|
|
80
83
|
|
|
81
84
|
Direct-harness help for `run` and `check` describes the relevant options and access requirements without loading definitions, starting a coordinator, or invoking a harness. Native CLI flags are not forwarded. Unsupported flags are errors.
|
|
82
85
|
|
package/sdk/cli/output.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# CLI Output
|
|
2
2
|
|
|
3
|
+
[Error diagnostics](../diagnostics.md) defines safe process and coordinator communication details. These details enrich message text without changing the error record shape or command exit-code rules.
|
|
4
|
+
|
|
3
5
|
`dashboard` is a terminal-only live view, not an execution-record command. It accepts no `--format` option and emits no text/JSONL records on success. Its rows, empty state, and terminal lifecycle are defined in the [CLI contract](index.md#live-dashboard).
|
|
4
6
|
|
|
5
7
|
Execution commands accept `--format text|jsonl`; `text` is the default. Both formats report the same operations. JSONL contains complete records separated by newlines, never native token streams or tool transcripts. All records include `version: 1` and a `type` discriminator. Identifiers are opaque strings prefixed with `ses_`, `tsk_`, `rsp_`, or `req_` and are not paths or process identifiers.
|
|
@@ -33,7 +35,7 @@ The identifiers above are illustrative. A `response` contains `sessionId`, `task
|
|
|
33
35
|
|
|
34
36
|
`status` emits a `status` record with `sessionId`, `taskId`, `state`, and optional `response` and `error`. The response preview contains `responseId`, `text`, and `truncated`; at most 4,000 text characters are included. `subharness status <task-id> --full` returns the complete latest response instead. `subharness wait <task-id>` returns or awaits the first retained response; adding `--after <response-id>` returns or awaits the next response after that cursor.
|
|
35
37
|
|
|
36
|
-
`list` emits an `agents` record with an `agents` array of `{ id, name, description, scope }`, where scope is `harness`, `repo`, `global`, or `subagent`. The built-in entries have IDs and names `codex`, `claude`, and `
|
|
38
|
+
`list` emits an `agents` record with an `agents` array of `{ id, name, description, scope }`, where scope is `harness`, `repo`, `global`, or `subagent`. The built-in entries have IDs and names `codex`, `claude`, `fx`, `opencode`, `copilot`, and `cursor`, and scope `harness`; they appear in that order before discovered specialists. They describe supported targets, not verified executable, authentication, or model availability. A managed parent's catalog includes its declared `subagent:` entries; generic parents have no declared children. `queue` emits a `queue` record with `sessionId`, `paused`, optional `active`, and a `tasks` array in pending order. Task summaries contain `taskId`, `state`, and a prompt `description` limited to 120 characters.
|
|
37
39
|
|
|
38
40
|
An `accepted` record contains `sessionId`, `taskId`, `delivery: "steer"`. A successful `cancel` emits a `cancelled` record with the targeted task identity and final state, after affected native execution has stopped. Already-terminal tasks retain their existing state. `resume` emits `started` for the existing task with `resumed: true`, then its next response or terminal outcome.
|
|
39
41
|
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Responses and Task Completion
|
|
2
2
|
|
|
3
|
+
The [execution SDK](execution.md) provides complete-response cursors and task snapshot observation directly to application code. It does not provide token deltas or native tool-progress events. UI disconnects detach observers. Embedded runner lifetime is application-owned; shared coordinator work survives connected-client disconnect.
|
|
4
|
+
|
|
3
5
|
Returning a response, completing a task, and starting another caller model turn are separate operations. The CLI returns control after each complete native response from the requested agent. It includes task/session identity, response identity, and the task state. It does not continuously forward token streams, native tool events, or descendants' transcripts.
|
|
4
6
|
|
|
5
7
|
The adapter captures complete responses automatically. Agents do not need a reporting tool, special JSON format, or a classifier that recognizes questions. A question is delivered through the same response mechanism as any other text.
|
package/sdk/config.md
CHANGED
|
@@ -39,7 +39,7 @@ The former `.agents/agents/` directory is not read. Definitions moved to `.subha
|
|
|
39
39
|
|
|
40
40
|
Discovery is required for listing and repository/global specialist lookup. Direct harness execution skips it. Declared-child execution loads only the parent's source and direct child map; unrelated catalog entries are not dependencies of a `subagent:` invocation.
|
|
41
41
|
|
|
42
|
-
The declared name determines identity. Duplicate names in the same scope are errors. `repo:reviewer` and `global:reviewer` remain distinct; bare `reviewer` is accepted only when unambiguous. The reserved CLI targets `codex`, `claude`, and `
|
|
42
|
+
The declared name determines identity. Duplicate names in the same scope are errors. `repo:reviewer` and `global:reviewer` remain distinct; bare `reviewer` is accepted only when unambiguous. The reserved CLI targets `codex`, `claude`, `fx`, `opencode`, `copilot`, and `cursor` always select native harnesses. Specialists with those names require a `repo:` or `global:` qualifier. Direct harness runs bypass definition discovery entirely, including invalid definition files. Global definitions can execute in any caller-supplied directory. Global and repository files resolve their own imports through normal Node package resolution.
|
|
43
43
|
|
|
44
44
|
```sh
|
|
45
45
|
subharness list --cwd /repo/worktree
|
package/sdk/copilot.md
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# GitHub Copilot adapter
|
|
2
|
+
|
|
3
|
+
The Copilot adapter uses the native GitHub Copilot SDK over a dedicated stdio runtime process. The pinned SDK is `@github/copilot-sdk` 1.0.14, with native protocol version 3 as exposed by Copilot runtime 1.0.85. It does not implement a model loop. The public model-only configuration and access routes are defined in [additional harnesses](additional-harnesses.md).
|
|
4
|
+
|
|
5
|
+
## Native GitHub access
|
|
6
|
+
|
|
7
|
+
An explicit `subscription` connection uses the native Copilot or GitHub CLI login. An `api-key` connection with `provider: "github"` instead reads the selected GitHub token, defaulting to `COPILOT_GITHUB_TOKEN`. The token must be eligible for Copilot; its presence alone does not prove entitlement. Neither route uses BYOK or Vercel Gateway.
|
|
8
|
+
|
|
9
|
+
The adapter removes competing ambient tokens and provider overrides. Saved-login access enables native logged-in-user discovery without passing a token; explicit GitHub access supplies the selected `gitHubToken` to the SDK client and session and disables logged-in-user fallback. Startup checks native authentication status and exact native model-catalog membership without generation. Saved-user or GitHub CLI authentication must not be confused with an environment token. Unknown authentication metadata and route changes fail explicitly. Auto-routing models are not accepted. Native login and refresh remain the harness's responsibility. Saved-login access preserves the native Copilot home so last-user metadata, keychain credentials, file-backed credentials, and GitHub CLI discovery remain native. Each session still uses a private `configDirectory` with configuration discovery disabled. Explicit GitHub-token, BYOK, and Gateway routes use a private runtime base directory as well. Subharness never reads or copies native credentials or imports personal provider/model configuration. Cleanup removes only adapter-owned private state. This preserves the existing manual approval and managed-policy behavior; it does not import personal permission configuration.
|
|
10
|
+
|
|
11
|
+
## Direct provider API keys
|
|
12
|
+
|
|
13
|
+
Copilot BYOK connections use `type: "api-key"` with `provider: "openai"`, `"anthropic"`, or `"azure"`. The selected variable defaults to `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, or `AZURE_OPENAI_API_KEY`, respectively; `env` and `envFile` can select another source. These routes disable GitHub login and configure one singular native provider, so requests use the chosen provider's billing.
|
|
14
|
+
|
|
15
|
+
```json
|
|
16
|
+
{
|
|
17
|
+
"access": {
|
|
18
|
+
"copilot": [{ "type": "api-key", "provider": "openai", "wireApi": "responses" }]
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
OpenAI defaults to `https://api.openai.com/v1`, and Anthropic defaults to `https://api.anthropic.com`. `baseUrl` can select another compatible endpoint; Azure requires it and expects its native resource/project URL. OpenAI-compatible services use `provider: "openai"` and their full API prefix. OpenAI/Azure accept `wireApi: "completions" | "responses"`, defaulting to `completions`; Anthropic uses Messages and rejects `wireApi`. Azure alone accepts `apiVersion`, passed to the SDK's native Azure option. HTTP transport is used.
|
|
24
|
+
|
|
25
|
+
The agent's `model` is the native behavior model. Optional `wireModel` supplies a different remote model or Azure deployment name; omission uses the agent model on the wire too. The adapter passes these explicit values without inventing aliases. Readiness verifies the native model and configured provider/endpoint/credential representation. A BYOK assistant response can label its model with the remote name or deployment; native model RPC readback remains the authority for the pinned behavior model. BYOK has no universal remote catalog check: readiness does not prove a remote model exists or that a key has quota. Neither configured readback nor a user-supplied endpoint proves the remotely executed model.
|
|
26
|
+
|
|
27
|
+
Custom headers, bearer-token callbacks, experimental mixed-provider sessions, and provider options outside the documented access shape are not exposed. Gateway connections remain a separate supported route with their existing fixed endpoint and `creator/model` names.
|
|
28
|
+
|
|
29
|
+
## Gateway access
|
|
30
|
+
|
|
31
|
+
The adapter accepts explicit Gateway API-key or project OIDC connections. It configures a single native OpenAI-compatible provider with the fixed base URL `https://ai-gateway.vercel.sh/coding-agent/v1`, HTTP Chat Completions, and the exact requested `creator/model` identifier. API-key access supplies the selected API key; OIDC uses bearer-token authentication. GitHub login and unrelated provider credentials are disabled for this inference route. Ambient `COPILOT_PROVIDER_*` settings, model preferences, alternate endpoints, and native logged-in-user discovery must not override it.
|
|
32
|
+
|
|
33
|
+
A dedicated private native configuration/state directory prevents stored provider settings or plugins from changing billing. Native repository instructions and the caller's working directory remain available. Credentials are passed through the private SDK/protocol configuration, never process arguments or user-visible diagnostics. Startup uses the native protocol handshake, session creation, current-model and provider-endpoint introspection to verify the selection without submitting a prompt. The native allowed model list is restricted to the selected identifier. Unverifiable or changed selection is an explicit configuration error, not an access fallback.
|
|
34
|
+
|
|
35
|
+
The SDK's bundled runtime may be used; a separate global Copilot installation is not required when that runtime is available. Missing or incompatible runtime support is `HARNESS_UNAVAILABLE`. `check copilot --model creator/model` checks startup and cleanup only, not remote allowance.
|
|
36
|
+
|
|
37
|
+
## Tools and approvals
|
|
38
|
+
|
|
39
|
+
Declared tools use native SDK tool callbacks and preserve validated text/image results. Copilot's result envelope groups text and images separately: text blocks are joined in order, and images retain their relative order as binary image results. Interleaving between those groups is not representable. Specialist instructions supplement the native system instructions; they never replace the harness's tool loop. Tool names retain their definition keys; a collision with a native built-in tool is an explicit configuration error, never an override. Callbacks are admitted only while a task is active, and interruption/close drain already-admitted callbacks.
|
|
40
|
+
|
|
41
|
+
The adapter uses the native manual permission mode and forwards supported active permission requests through the shared approval flow. Only the specific offered once/deny decisions are exposed initially; it does not invent persistent permission grants. The action context is bounded and excludes credential-bearing configuration. Unsupported input or startup requests fail with `INPUT_REQUIRED` and stop affected work. No callback automatically approves native commands or expands native permissions for declared children.
|
|
42
|
+
|
|
43
|
+
Enterprise managed-policy self-fetch is enabled for explicit GitHub-token access, which supplies the session identity required by the SDK. Saved-login, BYOK, and Gateway routes omit that opt-in because they do not supply a session GitHub token; the adapter does not extract one or promise enterprise policy self-fetch for those routes. Native manual approval requirements remain enabled on every route. A permission request that specifically requires human-only provenance or an unrepresentable sandbox bypass is unsupported and fails with `INPUT_REQUIRED`; a Subharness caller response must not be relabeled as a verified human decision. Requests already resolved by native hooks are not presented as new approval requests. Native withdrawal retires the matching request and late answers are ignored.
|
|
44
|
+
|
|
45
|
+
## Lifecycle
|
|
46
|
+
|
|
47
|
+
Follow-ups reuse the native session. Events are correlated to the active turn; a final assistant message alone is not successful completion. A session-idle event is accepted in any native session mode after evidence that the submitted turn started, and only when no native work remains. A stale idle event cannot complete a newly submitted task. The adapter reports native shutdown or transport loss as an execution failure and limits final response text to 1 MiB. Model or endpoint substitutions fail without replay. Turn completion withdraws remaining approval requests.
|
|
48
|
+
|
|
49
|
+
Interruption stops tool admission, withdraws pending approvals, requests native abort, confirms no native work remains, and waits for admitted callbacks. An abort acknowledgement alone does not prove the native process stopped. Native cancellation requests and status checks are bounded; a cancellation that cannot establish idle/termination returns `CANCELLATION_FAILED`. Already-admitted host callbacks are drained without imposing that native cancellation deadline on their execution. Steering uses the coordinator's interrupt behavior, and prompt-free recovery returns `RECOVERY_UNSUPPORTED`.
|
|
50
|
+
|
|
51
|
+
An unconfirmed interruption fails the active turn and prevents further tasks in that session. Tool admission remains closed, and close must still attempt native cleanup. The turn cannot later report success after continuing with disabled tool callbacks. This rule also applies while native prompt admission is pending.
|
|
52
|
+
|
|
53
|
+
Close retires the session, shuts down the dedicated native runtime, confirms termination, drains tools, and removes private state. Cleanup RPCs are bounded so an unresponsive session cannot prevent process shutdown attempts. Escalated termination without confirmed exit produces `CANCELLATION_FAILED`, never successful cleanup. Startup failure also cleans up. A successful readiness record requires completed runtime cleanup. Raw SDK errors and process output are replaced with fixed diagnostics so credentials do not enter CLI records.
|
package/sdk/cursor.md
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# Cursor adapter
|
|
2
|
+
|
|
3
|
+
The Cursor adapter runs the native Cursor CLI for subscription access and the native Cursor TypeScript SDK for explicit API-key access against the caller's local working directory. Cursor owns inference, tools, conversation state, and context management. Cloud agents are not part of this adapter.
|
|
4
|
+
|
|
5
|
+
## Configuration
|
|
6
|
+
|
|
7
|
+
```ts
|
|
8
|
+
import { cursor } from "subharness";
|
|
9
|
+
|
|
10
|
+
cursor({ model: "CURSOR_MODEL_ID", sandboxMode: "enabled" });
|
|
11
|
+
|
|
12
|
+
interface CursorOptions {
|
|
13
|
+
readonly model: string;
|
|
14
|
+
readonly sandboxMode?: "enabled" | "disabled";
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
interface CursorConfig extends CursorOptions {
|
|
18
|
+
readonly kind: "cursor";
|
|
19
|
+
}
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
The constructor returns an immutable configuration. Model is required and must be nonempty. Unknown options, including `fast` and `effort`, are rejected. For API-key access, `sandboxMode` maps to the native SDK's local sandbox option. Omission preserves the SDK default, which runs local tools without interactive approval and without a sandbox. The native sandbox, when enabled, restricts shell writes and network access according to Cursor's native policy; it is not a sandbox for host-side custom tool callbacks. The adapter never enables broader permissions to make declared-child launches succeed.
|
|
23
|
+
|
|
24
|
+
The CLI target is `cursor`. `run cursor` and `check cursor` require `--model`. The following catalog behavior applies to API-key access; subscription model selection is defined below. The native routing IDs `auto` and `auto-smart`, including catalog aliases that resolve to either ID, are rejected because they do not retain a concrete model. The catalog has no generic router flag; the adapter does not infer routing from display names or invent additional reserved IDs. `--effort` is unsupported for this adapter. Native model aliases are accepted only when the account catalog supplies their mapping. Startup verifies the account and selected model through the native SDK without a generation, creates an empty local agent, and pins the resolved selection for follow-ups. When terminal results report a model, a detected replacement fails explicitly without replay. The agent handle's configured model is not evidence of the model that executed a run.
|
|
25
|
+
|
|
26
|
+
## Access
|
|
27
|
+
|
|
28
|
+
Personal access uses the `cursor` key and requires an explicit connection. A `subscription` connection uses the current native Cursor CLI login; the user completes `cursor-agent login` outside Subharness. An `api-key` connection uses the SDK and defaults to `CURSOR_API_KEY`. Both environment and explicit dotenv references follow the shared [access contract](access-config.md). Omission does not discover or enable a connection, and an empty list disables Cursor. Native login and API-key access are separate billing routes; neither is an implicit fallback for the other. Vercel API-key and OIDC connections are unsupported and fail with `UNSUPPORTED_OPTION` before startup. A Cursor API key is a Cursor billing route, not a Gateway credential.
|
|
29
|
+
|
|
30
|
+
```json
|
|
31
|
+
{
|
|
32
|
+
"access": {
|
|
33
|
+
"cursor": [{ "type": "subscription" }]
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
For API-key access, the selected key is passed explicitly to the SDK. Subscription access delegates credential discovery to the native CLI; Subharness never reads, copies, or transfers its saved tokens. A missing or expired native login fails with `ACCESS_UNAVAILABLE` and login guidance before submission. Subharness does not open a browser or perform an interactive login. Explicit API-key/auth-token environment variables must not silently override a subscription connection. Backend or website endpoint overrides through `CURSOR_BACKEND_URL` or `CURSOR_WEBSITE_URL` are rejected before the key is used. This also applies to the SDK host environment, because the SDK reads process-level settings. The adapter does not transfer subscription tokens, change account-synced provider settings, or treat the editor's custom-provider configuration as verified Gateway routing. Credentials and raw native errors must not appear in diagnostics.
|
|
39
|
+
|
|
40
|
+
## API-key session behavior
|
|
41
|
+
|
|
42
|
+
Successive tasks use the same native agent. Specialist instructions supplement the native harness instructions in the first user turn; they do not replace Cursor's system prompt. Declared tools use native SDK custom tools, validate inputs, preserve text and image results, and stop admitting calls when the task stops. Cancellation and close wait for admitted callbacks to settle.
|
|
43
|
+
|
|
44
|
+
Active steering returns `UNSUPPORTED_DELIVERY`; the coordinator applies its documented interrupt-and-replace behavior. Interruption cancels the active native run and waits for terminal completion. Failure to confirm native stop returns `CANCELLATION_FAILED`, fails the active turn, and prevents further tasks in that session. Tool admission remains closed; close must still dispose the native agent and drain admitted callbacks. Prompt-free recovery returns `RECOVERY_UNSUPPORTED`. A turn succeeds only with a successful terminal result and returns its final text. Responses are limited to 1 MiB.
|
|
45
|
+
|
|
46
|
+
The SDK's headless permission decisions remain native. This SDK version exposes no supported interactive approval or input channel to the adapter. Its `request` message records a backend request identifier and is not an approval request. Native run failures produce a sanitized execution error; the adapter does not invent an `INPUT_REQUIRED` mapping from undocumented error codes. Close disposes the native agent and drains custom tools. `check cursor` verifies credential/model discovery, local agent creation, and disposal without submitting a prompt; it does not prove quota or that a later shell command will be permitted.
|
|
47
|
+
|
|
48
|
+
## Subscription protocol and lifetime
|
|
49
|
+
|
|
50
|
+
Subscription access uses the installed `cursor-agent` executable with its native ACP protocol over standard input and output. The supported CLI release is `2026.09.23-86fc751`, using ACP version 1. Other releases fail before input with `HARNESS_UNAVAILABLE` and compatibility guidance. The executable name is explicit because `agent` is also a Subharness compatibility alias. A session owns one native process and one native conversation in the requested working directory. Subscription access currently supports macOS and Linux, where the adapter can own and terminate a native process group. Windows subscription startup fails with `UNSUPPORTED_OPTION`; the SDK API-key route is unchanged.
|
|
51
|
+
|
|
52
|
+
The transport accepts newline-delimited JSON-RPC 2.0, correlates responses by request ID, and rejects malformed envelopes or oversized protocol frames with `PROTOCOL_ERROR`. Individual frames are limited to 16 MiB; returned assistant text remains limited to 1 MiB. Native standard error and raw error messages are not forwarded to callers. Startup and configuration requests are bounded; model execution itself has no arbitrary completion deadline. Transport loss fails pending operations and retires the native process rather than leaving a task pending indefinitely.
|
|
53
|
+
|
|
54
|
+
Native cancellation sends `session/cancel` and requires the active `session/prompt` to finish before confirming interruption. A cancellation timeout retires and poisons the session with `CANCELLATION_FAILED`; it does not replay input or silently open a replacement conversation. Closing stops admitting tool calls, retires the native process, and drains admitted custom-tool callbacks. Process retirement uses bounded termination and confirms process exit; failure to establish termination is an error. Tool callbacks have their own drain lifetime and are not falsely marked cancelled by a native timeout. Queued follow-ups reuse the same conversation after ordinary completion or confirmed native cancellation. Active steering and prompt-free recovery remain unsupported.
|
|
55
|
+
|
|
56
|
+
## Subscription startup and capabilities
|
|
57
|
+
|
|
58
|
+
Subscription access preserves native credential ownership. It removes `CURSOR_API_KEY` and `CURSOR_AUTH_TOKEN` from the child environment and does not call ACP `authenticate`, `login`, or `logout`. Native credential discovery, refresh, and account eligibility remain Cursor's responsibility. The adapter checks bounded `status --format json` output for authenticated access/refresh availability, then requires authenticated ACP session creation and model discovery. Status alone is not proof of readiness, quota, subscription-plan eligibility, or credential history. The `subscription` connection selects Cursor's saved native CLI account route; it does not attest how those native credentials were originally created or impose an account spending cap.
|
|
59
|
+
|
|
60
|
+
A private owner-only configuration and data directory isolates model selection and session state. Native non-credential permission and network settings are preserved in the private configuration; project permission rules continue to apply. Shared user/project configuration is never rewritten. Alternate Cursor endpoints, authless/local-provider settings, and Bedrock activation are unsupported and rejected before native startup. Ordinary AWS credentials are not rejected merely because unrelated tools may use them. Native API keys, custom endpoints, or API-key helpers cannot silently select a different inference route. Subharness-created configuration files use mode `0600`; native-generated files remain inside the owner-only directory. Private state is removed after confirmed process shutdown. The native credential store is neither copied nor exposed through diagnostics.
|
|
61
|
+
|
|
62
|
+
Startup negotiates ACP version 1 with the native parameterized model picker. The required model must match a concrete base ID in the returned native catalog; CLI display aliases and bracketed variant strings are not inferred. Native model parameters retain their defaults for that model. Auto routing IDs `auto`, `auto-smart`, `default`, and `default[]` are rejected. The selected model is applied with `session/set_config_option` and verified from its returned configuration before each prompt. A reported change fails explicitly. The adapter does not claim configured model readback proves the remotely executed model; this native protocol does not expose executed-model metadata in its terminal result. For example, `cursor({ model: "gpt-5-mini" })` selects that base model when present in the account catalog. A CLI display variant such as a model name with an effort suffix is not automatically treated as the same base ID.
|
|
63
|
+
|
|
64
|
+
For subscription access, either explicit `sandboxMode` value fails with `UNSUPPORTED_OPTION` before submission because ACP does not expose a verified sandbox control. Omitting it uses Cursor's native ACP execution and permission behavior; it does not enable a Subharness sandbox. The API-key route retains its SDK sandbox option.
|
|
65
|
+
|
|
66
|
+
Declared tools use a uniquely named private MCP server supplied in `session/new`. The adapter verifies that the native client initializes the server before admitting prompts; a skipped or failed native MCP connection is a startup error. MCP request bodies are limited to 16 MiB and headers to 16 KiB. Text and image results retain their MCP content types and shared tool-result limits. Native approval rules still apply. Active `session/request_permission` requests expose only the offered allow-once and reject-once choices through the shared approval flow with `harness: "cursor"`; persistent allow-always/reject-always choices are not exposed. Original native option IDs are retained. Unsupported or malformed interactive requests, including questions and plan approval, fail with `INPUT_REQUIRED` and cancel the native turn. Startup-time interactive requests are unsupported. Withdrawal, terminal completion, cancellation, and close retire pending requests; late answers never grant permission.
|
|
67
|
+
|
|
68
|
+
A terminal ACP `end_turn` returns the accumulated assistant text. `cancelled` is interruption; refusal or exhausted-limit terminal reasons and RPC/transport failures reject with a sanitized execution error. This CLI release can render some backend failures as ordinary assistant text followed by `end_turn`, without a separate failure signal. Such text is returned as native output; a completed Subharness task on this route means the native turn ended, not that every backend operation succeeded. The adapter does not guess error status from prose. Raw transport diagnostics remain private. This capability limit is specific to the subscription ACP route; the API-key SDK route retains its typed native results.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Error Diagnostics
|
|
2
|
+
|
|
3
|
+
Error records retain their existing `{ code, message }` shape. Messages describe the operation that failed and include safe, observed context that helps callers identify the cause. The diagnostics described here never include raw native stderr, SDK exception text, HTTP response bodies, RPC error messages or data, prompts, credential values, or configuration contents. Unknown native values receive a fixed generic explanation rather than being echoed. Paths included in these diagnostic cases are quoted with C0, C1, DEL, and Unicode line separators escaped.
|
|
4
|
+
|
|
5
|
+
## Claude startup and turn failures
|
|
6
|
+
|
|
7
|
+
Claude startup distinguishes native account inspection, model catalog retrieval, and model settings validation. An unexpected exception from account inspection or model catalog retrieval is `HARNESS_FAILED`, with the failed phase identified. Unexpected exceptions after successful access verification are also harness failures identified by the current startup phase. They do not establish unavailable access and do not permit connection or harness fallback. A completed account inspection that demonstrates unavailable or incompatible access retains `ACCESS_UNAVAILABLE` and its existing eligible fallback behavior. Existing explicit configuration, settings capability, model, and effort errors retain their codes and guidance, including `--model` guidance for an unverifiable native default. Cleanup and no-prompt-before-verification rules remain unchanged.
|
|
8
|
+
|
|
9
|
+
Claude terminal result failures retain `HARNESS_ERROR`. Recognized result subtypes distinguish a native turn limit, a native budget limit, exhausted structured-output retries, and an error during execution. Messages describe the reported category without claiming an unobserved provider cause, quoting native error text, or promising that Subharness exposes configuration for every native limit. Unknown subtypes and unsuccessful results without a recognized category receive a fixed generic turn-failure message. A reported execution failure is never automatically retried or replayed.
|
|
10
|
+
|
|
11
|
+
## Native protocol and process failures
|
|
12
|
+
|
|
13
|
+
Codex and fx RPC rejection messages identify the requested native method and include the native error code only when it is a safe integer. Known JSON-RPC codes use fixed categories: `-32700` is a parse error, `-32600` an invalid request, `-32601` an unavailable method, `-32602` invalid parameters, and `-32603` an internal error. Other safe integer codes remain numeric without an invented interpretation. Missing or malformed codes are not printed. A generic rejection does not claim that authentication or native configuration caused the error. Existing steering and session-option capability classifications retain their Subharness error codes and useful recovery guidance. Freeform native messages and error data are never forwarded.
|
|
14
|
+
|
|
15
|
+
When an observed process termination causes an error, diagnostics identify the process and include its observed exit code or recognized termination signal when available. This applies to Codex and fx transports, the Claude subprocess owned by the adapter, session workers, and coordinator startup. A signal does not establish why it was sent; for example, `SIGKILL` does not by itself prove an out-of-memory condition. Missing termination metadata is omitted. Cleanup-induced termination must not replace or be described as the cause of an earlier failure. Adding diagnostics does not wait indefinitely for process exit or weaken bounded shutdown.
|
|
16
|
+
|
|
17
|
+
Worker failures identify the pending operation when available, such as startup, turn submission, steering, interruption, recovery, or closing. Coordinator failures distinguish failure to spawn, early exit, and startup timeout, retaining existing retirement and endpoint-publication safeguards. Only fixed operation labels and recognized system error categories may be added; raw operating-system exception text is not forwarded.
|
|
18
|
+
|
|
19
|
+
## Coordinator communication
|
|
20
|
+
|
|
21
|
+
Coordinator communication failures retain `COORDINATOR_UNAVAILABLE`. A non-success HTTP response identifies its numeric status and a recognized requested operation, when available. A successful response with no body is distinguished from HTTP rejection. Invalid JSON records, an incomplete final record, and an interrupted response stream have distinct fixed explanations. Transport failures retain the statement that the task was not automatically retried. These diagnostics do not expose endpoint tokens, raw request or response contents, or arbitrary operation names. They do not infer whether a task completed from an HTTP or transport failure and do not automatically replay it.
|
|
22
|
+
|
|
23
|
+
## Access configuration and OIDC
|
|
24
|
+
|
|
25
|
+
Personal-access schema errors retain `INVALID_CONFIG` and identify their structural location: the root object, `access`, a recognized harness entry such as `access.claudeCode`, or a zero-based connection location such as `access.claudeCode[0]`. A known invalid connection property may extend that location, for example `access.claudeCode[0].env`. Unknown property names or harness names are not echoed because arbitrary keys can contain sensitive values. An unknown connection property identifies its containing connection. Errors raised while loading a file also identify its absolute, safely quoted main-checkout path, or execution-project path outside Git. Invalid JSON and file-read errors identify the same safely quoted path. Validation happens before any connection is attempted; it does not fabricate attempted-access context.
|
|
26
|
+
|
|
27
|
+
An inaccessible execution directory is identified by its safely quoted caller-supplied path in both the CLI and project resolver. A missing credential file identifies its safely quoted configured `envFile` reference; it retains `ACCESS_UNAVAILABLE` and eligible fallback behavior. These messages preserve the existing path resolution rules and do not read credential contents to enrich an error.
|
|
28
|
+
|
|
29
|
+
OIDC expiration and not-yet-valid claims have separate diagnostics, both retaining `INVALID_CONFIG`. Expiration directs the caller to refresh the configured credential source. A future not-before claim explains that the token is not yet valid and suggests checking the system clock or waiting until validity begins; it does not assert that the clock is wrong. If both conditions apply, expiration is reported first. No token values or raw claims are printed. Claim-validation rules, project matching, and credential-source selection remain unchanged.
|
package/sdk/distribution.md
CHANGED
|
@@ -4,13 +4,15 @@ The npm package name is `subharness`. One package provides the TypeScript SDK an
|
|
|
4
4
|
|
|
5
5
|
The source repository is [vercel-labs/subharness](https://github.com/vercel-labs/subharness). Its GitHub visibility is internal, so cloning requires repository access. The product name is subharness. The primary CLI command is `subharness`; `agent` remains an identical compatibility alias. Agent discovery and personal configuration live under `.subharness/`.
|
|
6
6
|
|
|
7
|
-
Node.js 22.18 or newer is required. The SDK is ESM and includes TypeScript declarations.
|
|
7
|
+
Node.js 22.18 or newer is required. The SDK is ESM and includes TypeScript declarations. Installing this package supplies the pinned Copilot and Cursor SDK dependencies. It does not configure native harness credentials or install the Codex, Claude Code, fx, OpenCode, or Cursor executables. Cursor subscription access requires the separately installed `cursor-agent` CLI; Cursor API-key access uses the bundled SDK dependency. The Copilot SDK may supply its compatible runtime as defined in the [Copilot contract](copilot.md).
|
|
8
8
|
|
|
9
9
|
## Installation and resolution
|
|
10
10
|
|
|
11
11
|
The public npm package is `subharness`, with an initial release version of `0.0.1`. Install the CLI globally with `npm install --global subharness`, or run it without a global installation using `npx subharness`. Registry installation does not require access to the internal source repository. The root package is publishable; the documentation workspace remains private.
|
|
12
12
|
|
|
13
|
-
|
|
13
|
+
Application backends install `subharness` with `npm install subharness` and import the [execution SDK](execution.md). Native executables remain external dependencies; the package does not require a globally installed Subharness CLI for SDK execution.
|
|
14
|
+
|
|
15
|
+
For project-local CLI use, install `subharness` with `npm install --save-dev subharness` and invoke its CLI with `npx subharness`. This also makes SDK imports resolve from repository agent definitions. A global CLI installation alone does not make SDK imports resolve from project-local definitions. Global definitions need an SDK dependency reachable from their own directory under normal Node package resolution.
|
|
14
16
|
|
|
15
17
|
Source development uses pnpm 11.20.0, pinned in the root `packageManager` field. `pnpm-workspace.yaml` includes the root SDK and `apps/docs`, with one committed `pnpm-lock.yaml`. Run `pnpm install --frozen-lockfile` from the repository root, then `pnpm run build`. Run the local CLI with `node dist/cli/main.js`; this requires no global link. Direct harness targets do not need a consumer SDK dependency. For use in another project, build a tarball with `pnpm pack` and install that tarball in the consumer so its agent definitions can resolve SDK imports.
|
|
16
18
|
|
|
@@ -22,7 +24,7 @@ A tarball built from a source checkout can differ from the registry release with
|
|
|
22
24
|
|
|
23
25
|
Packages built from this source contain the built JavaScript, declaration files and source maps under `dist/`, SDK Markdown documentation under `sdk/`, the README, the Apache License 2.0, and package metadata. Source maps embed the original TypeScript so debuggers can display it without a separate source checkout. Website code, development agents, skills, source assets, tests, `.context`, personal settings, and environment files are excluded.
|
|
24
26
|
|
|
25
|
-
Packing builds the SDK from the current source before assembling its files. The CLI's `--version` output matches the package version. A clean consumer must be able to import the SDK and discover a TypeScript agent definition using the installed executable without access to the source checkout or a paid model call. `pnpm run check:package` packs with pnpm and verifies this clean-consumer behavior and the packaged file boundary without running a coding model. It requires network access to install dependencies from the public npm registry and disables install lifecycle scripts in the temporary consumer.
|
|
27
|
+
Packing builds the SDK from the current source before assembling its files. The CLI's `--version` output matches the package version. A clean consumer must be able to import the execution SDK, run an inline definition against a controlled native protocol fixture, and discover a TypeScript agent definition using the installed executable without access to the source checkout or a paid model call. `pnpm run check:package` packs with pnpm and verifies this clean-consumer behavior and the packaged file boundary without running a coding model. It requires network access to install dependencies from the public npm registry and disables install lifecycle scripts in the temporary consumer.
|
|
26
28
|
|
|
27
29
|
The committed lockfile pins dependency content with integrity hashes and does not embed company registry URLs. pnpm resolves packages through the configured registry, which defaults to the public npm registry. Installation and package verification do not change the developer's global npm registry configuration. Registry authentication remains with npm.
|
|
28
30
|
|
package/sdk/evals.md
CHANGED
|
@@ -41,7 +41,7 @@ resolveEvalAccess(options: {
|
|
|
41
41
|
}): Promise<EvalAccess | EvalAccessError>
|
|
42
42
|
```
|
|
43
43
|
|
|
44
|
-
|
|
44
|
+
The evaluation harness discriminator remains limited to the original `codex`, `claudeCode`, and `fx` identifiers. The harness list is nonempty and has no duplicates. IDs are nonempty strings. `minimumValidityMs` is a positive finite safe integer. The default environment is `process.env`, and the default clock is `Date.now` in milliseconds. Input errors are returned before native startup; this function never starts a harness or makes a network request.
|
|
45
45
|
|
|
46
46
|
`EvalAccessError` is a typed returned error, not an exception-based domain result. Its `reason` is one of `invalid-input`, `policy`, `credentials-unavailable`, `validation`, `identity`, or `insufficient-lifetime`. Its message is exactly `Evaluation access is unavailable.` for every reason; it never retains raw caught exceptions, credentials, token claims, or native output. A policy error includes absent keys, a disabled participating harness, non-OIDC routes, or fallback lists. Credential reading failures are credentials-unavailable; malformed/expired/not-yet-valid credentials or linkage validation failures are validation. A valid linked token whose project or organization differs from the independent expectation is identity. Insufficient remaining validity uses insufficient-lifetime. Unclassified access-layer failures are validation, not guessed native authentication causes.
|
|
47
47
|
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import {
|
|
2
|
+
createRunner,
|
|
3
|
+
type AgentDefinition,
|
|
4
|
+
type ApprovalResponse,
|
|
5
|
+
type CompleteResponse,
|
|
6
|
+
type PromptReceipt,
|
|
7
|
+
type RunnerOptions,
|
|
8
|
+
type SessionHandle,
|
|
9
|
+
type TaskHandle,
|
|
10
|
+
type TaskResult,
|
|
11
|
+
type TaskSnapshot,
|
|
12
|
+
} from "subharness";
|
|
13
|
+
|
|
14
|
+
type StringField = Readonly<{
|
|
15
|
+
type: "string";
|
|
16
|
+
description: string;
|
|
17
|
+
minLength: number;
|
|
18
|
+
}>;
|
|
19
|
+
|
|
20
|
+
interface HostTool<Input, Output> {
|
|
21
|
+
readonly description: string;
|
|
22
|
+
readonly inputSchema: Readonly<{
|
|
23
|
+
type: "object";
|
|
24
|
+
properties: Readonly<Record<keyof Input, StringField>>;
|
|
25
|
+
required: readonly (keyof Input)[];
|
|
26
|
+
additionalProperties: false;
|
|
27
|
+
}>;
|
|
28
|
+
execute(input: Input): Promise<Output>;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export interface ChatToolApplication {
|
|
32
|
+
readonly consultation: HostTool<{ prompt: string }, ObservedTask>;
|
|
33
|
+
readonly followup: HostTool<{ prompt: string }, ObservedTask>;
|
|
34
|
+
responses(taskId: string, options?: { readonly after?: string; readonly signal?: AbortSignal }): AsyncIterable<CompleteResponse>;
|
|
35
|
+
watch(taskId: string, options?: { readonly signal?: AbortSignal }): AsyncIterable<TaskSnapshot>;
|
|
36
|
+
answerApproval(requestId: string, response: ApprovalResponse): Promise<void>;
|
|
37
|
+
cancel(taskId: string): Promise<TaskSnapshot>;
|
|
38
|
+
dispose(): Promise<void>;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export interface ObservedTask {
|
|
42
|
+
readonly taskId: string;
|
|
43
|
+
readonly sessionId: string;
|
|
44
|
+
readonly result: TaskResult;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export function createChatToolApplication(options: {
|
|
48
|
+
readonly definition: AgentDefinition;
|
|
49
|
+
/** An existing absolute directory, validated by the runner when consultation starts. */
|
|
50
|
+
readonly cwd: string;
|
|
51
|
+
readonly runner?: RunnerOptions;
|
|
52
|
+
/** Called at task admission so the host can expose approval, observation, and stop controls immediately. */
|
|
53
|
+
readonly onTask?: (task: TaskHandle) => void;
|
|
54
|
+
}): ChatToolApplication {
|
|
55
|
+
const runner = createRunner(options.runner);
|
|
56
|
+
const tasks = new Map<string, TaskHandle>();
|
|
57
|
+
let session: SessionHandle | undefined;
|
|
58
|
+
let sessionStarting = false;
|
|
59
|
+
|
|
60
|
+
const ownedTask = (taskId: string): TaskHandle => {
|
|
61
|
+
const task = tasks.get(taskId);
|
|
62
|
+
if (!task) throw new Error("The task does not belong to this chat execution scope.");
|
|
63
|
+
return task;
|
|
64
|
+
};
|
|
65
|
+
const remember = (task: TaskHandle): TaskHandle => {
|
|
66
|
+
tasks.set(task.id, task);
|
|
67
|
+
return task;
|
|
68
|
+
};
|
|
69
|
+
const observeTask = async (task: TaskHandle): Promise<ObservedTask> => {
|
|
70
|
+
const owned = remember(task);
|
|
71
|
+
options.onTask?.(owned);
|
|
72
|
+
return {taskId: owned.id, sessionId: owned.sessionId, result: await owned.result};
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
return {
|
|
76
|
+
consultation: {
|
|
77
|
+
description: "Consult the configured coding agent.",
|
|
78
|
+
inputSchema: promptSchema("The self-contained task and context for the specialist."),
|
|
79
|
+
execute: async ({ prompt }) => {
|
|
80
|
+
if (session || sessionStarting) throw new Error("This chat scope already has a specialist session; use followup.");
|
|
81
|
+
sessionStarting = true;
|
|
82
|
+
try {
|
|
83
|
+
const created = await runner.createSession(options.definition, { cwd: options.cwd });
|
|
84
|
+
let admitted: PromptReceipt;
|
|
85
|
+
try { admitted = await created.prompt(prompt); }
|
|
86
|
+
catch (error) { await created.close(); throw error; }
|
|
87
|
+
session = created;
|
|
88
|
+
return observeTask(admitted.task);
|
|
89
|
+
} finally {
|
|
90
|
+
sessionStarting = false;
|
|
91
|
+
}
|
|
92
|
+
},
|
|
93
|
+
},
|
|
94
|
+
followup: {
|
|
95
|
+
description: "Send a follow-up to the retained specialist session and observe its task.",
|
|
96
|
+
inputSchema: promptSchema("Additional context or a follow-up request for the specialist."),
|
|
97
|
+
execute: async ({ prompt }) => {
|
|
98
|
+
if (!session) throw new Error("Start a consultation before sending a follow-up.");
|
|
99
|
+
const sent = await session.prompt(prompt);
|
|
100
|
+
return observeTask(sent.task);
|
|
101
|
+
},
|
|
102
|
+
},
|
|
103
|
+
responses: (taskId, responseOptions) => ownedTask(taskId).responses(responseOptions),
|
|
104
|
+
watch: (taskId, watchOptions) => ownedTask(taskId).watch(watchOptions),
|
|
105
|
+
// The host must authenticate this route and submit an action offered by the request.
|
|
106
|
+
// Resolution only acknowledges the answer; keep observing to learn the outcome.
|
|
107
|
+
answerApproval: (requestId, content) => runner.respond(requestId, content),
|
|
108
|
+
cancel: taskId => ownedTask(taskId).cancel(),
|
|
109
|
+
dispose: () => runner.close(),
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function promptSchema(description: string) {
|
|
114
|
+
return {
|
|
115
|
+
type: "object" as const,
|
|
116
|
+
properties: { prompt: { type: "string" as const, description, minLength: 1 } },
|
|
117
|
+
required: ["prompt"] as const,
|
|
118
|
+
additionalProperties: false as const,
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// Keep one factory instance per authenticated, bounded chat/job scope. The host schedules
|
|
123
|
+
// its next chat turn after receiving the result. Complete response replay can include `waiting`
|
|
124
|
+
// and task completion alone does not prove the caller's objective succeeded.
|
|
125
|
+
// Passing an AbortSignal to responses/watch disconnects only that reader. Use cancel for an
|
|
126
|
+
// explicit Stop action, and dispose when the whole scope ends (not on browser disconnect).
|