subharness 0.0.4 → 0.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -9
- package/dist/adapters/claude-process.d.ts +3 -0
- package/dist/adapters/claude-process.js +19 -2
- package/dist/adapters/claude-process.js.map +1 -1
- package/dist/adapters/claude-result.d.ts +2 -0
- package/dist/adapters/claude-result.js +22 -0
- package/dist/adapters/claude-result.js.map +1 -0
- package/dist/adapters/claude-tools.d.ts +6 -2
- package/dist/adapters/claude-tools.js +14 -12
- package/dist/adapters/claude-tools.js.map +1 -1
- package/dist/adapters/claude-worker-client.d.ts +49 -0
- package/dist/adapters/claude-worker-client.js +359 -0
- package/dist/adapters/claude-worker-client.js.map +1 -0
- package/dist/adapters/claude-worker-process.d.ts +20 -0
- package/dist/adapters/claude-worker-process.js +76 -0
- package/dist/adapters/claude-worker-process.js.map +1 -0
- package/dist/adapters/claude-worker-protocol.d.ts +38 -0
- package/dist/adapters/claude-worker-protocol.js +2 -0
- package/dist/adapters/claude-worker-protocol.js.map +1 -0
- package/dist/adapters/claude-worker.d.ts +1 -0
- package/dist/adapters/claude-worker.js +126 -0
- package/dist/adapters/claude-worker.js.map +1 -0
- package/dist/adapters/claude.d.ts +16 -6
- package/dist/adapters/claude.js +164 -75
- package/dist/adapters/claude.js.map +1 -1
- package/dist/adapters/codex.js +4 -1
- package/dist/adapters/codex.js.map +1 -1
- package/dist/adapters/copilot-permissions.d.ts +26 -0
- package/dist/adapters/copilot-permissions.js +121 -0
- package/dist/adapters/copilot-permissions.js.map +1 -0
- package/dist/adapters/copilot-tools.d.ts +12 -0
- package/dist/adapters/copilot-tools.js +61 -0
- package/dist/adapters/copilot-tools.js.map +1 -0
- package/dist/adapters/copilot.d.ts +108 -0
- package/dist/adapters/copilot.js +819 -0
- package/dist/adapters/copilot.js.map +1 -0
- package/dist/adapters/cursor-cli-approvals.d.ts +2 -0
- package/dist/adapters/cursor-cli-approvals.js +83 -0
- package/dist/adapters/cursor-cli-approvals.js.map +1 -0
- package/dist/adapters/cursor-cli-model.d.ts +4 -0
- package/dist/adapters/cursor-cli-model.js +64 -0
- package/dist/adapters/cursor-cli-model.js.map +1 -0
- package/dist/adapters/cursor-cli-session.d.ts +32 -0
- package/dist/adapters/cursor-cli-session.js +316 -0
- package/dist/adapters/cursor-cli-session.js.map +1 -0
- package/dist/adapters/cursor-cli-tools.d.ts +20 -0
- package/dist/adapters/cursor-cli-tools.js +181 -0
- package/dist/adapters/cursor-cli-tools.js.map +1 -0
- package/dist/adapters/cursor-cli.d.ts +2 -0
- package/dist/adapters/cursor-cli.js +306 -0
- package/dist/adapters/cursor-cli.js.map +1 -0
- package/dist/adapters/cursor-model.d.ts +7 -0
- package/dist/adapters/cursor-model.js +67 -0
- package/dist/adapters/cursor-model.js.map +1 -0
- package/dist/adapters/cursor-rpc.d.ts +42 -0
- package/dist/adapters/cursor-rpc.js +271 -0
- package/dist/adapters/cursor-rpc.js.map +1 -0
- package/dist/adapters/cursor-session.d.ts +4 -0
- package/dist/adapters/cursor-session.js +327 -0
- package/dist/adapters/cursor-session.js.map +1 -0
- package/dist/adapters/cursor-startup.d.ts +16 -0
- package/dist/adapters/cursor-startup.js +74 -0
- package/dist/adapters/cursor-startup.js.map +1 -0
- package/dist/adapters/cursor-tools.d.ts +11 -0
- package/dist/adapters/cursor-tools.js +49 -0
- package/dist/adapters/cursor-tools.js.map +1 -0
- package/dist/adapters/cursor.d.ts +6 -0
- package/dist/adapters/cursor.js +17 -0
- package/dist/adapters/cursor.js.map +1 -0
- package/dist/adapters/fx-auth.d.ts +12 -2
- package/dist/adapters/fx-auth.js +51 -61
- package/dist/adapters/fx-auth.js.map +1 -1
- package/dist/adapters/fx-profile.d.ts +8 -0
- package/dist/adapters/fx-profile.js +101 -0
- package/dist/adapters/fx-profile.js.map +1 -0
- package/dist/adapters/fx-rpc.d.ts +2 -0
- package/dist/adapters/fx-rpc.js +30 -4
- package/dist/adapters/fx-rpc.js.map +1 -1
- package/dist/adapters/fx-status.d.ts +2 -0
- package/dist/adapters/fx-status.js +96 -0
- package/dist/adapters/fx-status.js.map +1 -0
- package/dist/adapters/fx.js +28 -12
- package/dist/adapters/fx.js.map +1 -1
- package/dist/adapters/opencode-access.d.ts +13 -0
- package/dist/adapters/opencode-access.js +76 -0
- package/dist/adapters/opencode-access.js.map +1 -0
- package/dist/adapters/opencode-config.d.ts +11 -0
- package/dist/adapters/opencode-config.js +238 -0
- package/dist/adapters/opencode-config.js.map +1 -0
- package/dist/adapters/opencode-http.d.ts +28 -0
- package/dist/adapters/opencode-http.js +297 -0
- package/dist/adapters/opencode-http.js.map +1 -0
- package/dist/adapters/opencode-tools.d.ts +19 -0
- package/dist/adapters/opencode-tools.js +127 -0
- package/dist/adapters/opencode-tools.js.map +1 -0
- package/dist/adapters/opencode.d.ts +2 -0
- package/dist/adapters/opencode.js +569 -0
- package/dist/adapters/opencode.js.map +1 -0
- package/dist/adapters/rpc.d.ts +3 -0
- package/dist/adapters/rpc.js +45 -4
- package/dist/adapters/rpc.js.map +1 -1
- package/dist/adapters/types.d.ts +5 -0
- package/dist/adapters/types.js.map +1 -1
- package/dist/approvals/types.d.ts +1 -1
- package/dist/approvals/types.js.map +1 -1
- package/dist/cli/args.d.ts +1 -1
- package/dist/cli/args.js +4 -1
- package/dist/cli/args.js.map +1 -1
- package/dist/cli/catalog-worker.js.map +1 -1
- package/dist/cli/dashboard-client.js +20 -31
- package/dist/cli/dashboard-client.js.map +1 -1
- package/dist/cli/dashboard-controller.d.ts +7 -0
- package/dist/cli/dashboard-controller.js +27 -0
- package/dist/cli/dashboard-controller.js.map +1 -0
- package/dist/cli/dashboard-detail-view.d.ts +1 -1
- package/dist/cli/dashboard-detail-view.js +3 -4
- package/dist/cli/dashboard-detail-view.js.map +1 -1
- package/dist/cli/dashboard-history-view.js.map +1 -1
- package/dist/cli/dashboard-input.d.ts +1 -1
- package/dist/cli/dashboard-input.js +11 -1
- package/dist/cli/dashboard-input.js.map +1 -1
- package/dist/cli/dashboard-layout.js +3 -3
- package/dist/cli/dashboard-layout.js.map +1 -1
- package/dist/cli/dashboard-renderer.js.map +1 -1
- package/dist/cli/dashboard-style.js +2 -2
- package/dist/cli/dashboard-style.js.map +1 -1
- package/dist/cli/dashboard.d.ts +2 -2
- package/dist/cli/dashboard.js +7 -32
- package/dist/cli/dashboard.js.map +1 -1
- package/dist/cli/help.d.ts +1 -1
- package/dist/cli/help.js +30 -12
- package/dist/cli/help.js.map +1 -1
- package/dist/cli/main.js +55 -43
- package/dist/cli/main.js.map +1 -1
- package/dist/config/access-provenance.d.ts +10 -0
- package/dist/config/access-provenance.js +31 -0
- package/dist/config/access-provenance.js.map +1 -0
- package/dist/config/access.d.ts +12 -2
- package/dist/config/access.js +125 -27
- package/dist/config/access.js.map +1 -1
- package/dist/config/loader.js +1 -0
- package/dist/config/loader.js.map +1 -1
- package/dist/config/oidc.js +4 -2
- package/dist/config/oidc.js.map +1 -1
- package/dist/config/project.js +2 -1
- package/dist/config/project.js.map +1 -1
- package/dist/config/resolve-access.js +24 -10
- package/dist/config/resolve-access.js.map +1 -1
- package/dist/errors.d.ts +2 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +8 -2
- package/dist/index.js +4 -1
- package/dist/index.js.map +1 -1
- package/dist/process-diagnostics.d.ts +2 -0
- package/dist/process-diagnostics.js +11 -0
- package/dist/process-diagnostics.js.map +1 -0
- package/dist/runtime/access-errors.d.ts +4 -0
- package/dist/runtime/access-errors.js +25 -0
- package/dist/runtime/access-errors.js.map +1 -0
- package/dist/runtime/approval-registry.d.ts +9 -0
- package/dist/runtime/approval-registry.js +144 -5
- package/dist/runtime/approval-registry.js.map +1 -1
- package/dist/runtime/capture.d.ts +22 -0
- package/dist/runtime/capture.js +320 -0
- package/dist/runtime/capture.js.map +1 -0
- package/dist/runtime/catalog.d.ts +39 -0
- package/dist/runtime/catalog.js +84 -0
- package/dist/runtime/catalog.js.map +1 -0
- package/dist/runtime/client.d.ts +2 -1
- package/dist/runtime/client.js +74 -47
- package/dist/runtime/client.js.map +1 -1
- package/dist/runtime/coordinator.d.ts +44 -7
- package/dist/runtime/coordinator.js +265 -31
- package/dist/runtime/coordinator.js.map +1 -1
- package/dist/runtime/daemon.js +12 -6
- package/dist/runtime/daemon.js.map +1 -1
- package/dist/runtime/dashboard-sanitize.d.ts +2 -0
- package/dist/runtime/dashboard-sanitize.js +7 -0
- package/dist/runtime/dashboard-sanitize.js.map +1 -0
- package/dist/runtime/dashboard-workspace.js +5 -1
- package/dist/runtime/dashboard-workspace.js.map +1 -1
- package/dist/runtime/dashboard.d.ts +1 -1
- package/dist/runtime/dashboard.js +2 -3
- package/dist/runtime/dashboard.js.map +1 -1
- package/dist/runtime/definition.d.ts +6 -1
- package/dist/runtime/definition.js +42 -2
- package/dist/runtime/definition.js.map +1 -1
- package/dist/runtime/native-owner.js +7 -1
- package/dist/runtime/native-owner.js.map +1 -1
- package/dist/runtime/pagination.d.ts +9 -0
- package/dist/runtime/pagination.js +36 -0
- package/dist/runtime/pagination.js.map +1 -0
- package/dist/runtime/sdk-observation.d.ts +8 -0
- package/dist/runtime/sdk-observation.js +75 -0
- package/dist/runtime/sdk-observation.js.map +1 -0
- package/dist/runtime/sdk-projection.d.ts +8 -0
- package/dist/runtime/sdk-projection.js +19 -0
- package/dist/runtime/sdk-projection.js.map +1 -0
- package/dist/runtime/sdk-service.d.ts +11 -0
- package/dist/runtime/sdk-service.js +109 -0
- package/dist/runtime/sdk-service.js.map +1 -0
- package/dist/runtime/select-native.d.ts +4 -1
- package/dist/runtime/select-native.js +28 -13
- package/dist/runtime/select-native.js.map +1 -1
- package/dist/runtime/service.d.ts +11 -1
- package/dist/runtime/service.js +178 -3
- package/dist/runtime/service.js.map +1 -1
- package/dist/runtime/session-launcher.js +1 -1
- package/dist/runtime/session-launcher.js.map +1 -1
- package/dist/runtime/snapshots.d.ts +103 -0
- package/dist/runtime/snapshots.js +124 -0
- package/dist/runtime/snapshots.js.map +1 -0
- package/dist/runtime/state.d.ts +26 -0
- package/dist/runtime/state.js +86 -17
- package/dist/runtime/state.js.map +1 -1
- package/dist/runtime/task-data.d.ts +38 -0
- package/dist/runtime/task-data.js +2 -0
- package/dist/runtime/task-data.js.map +1 -0
- package/dist/runtime/transport.d.ts +13 -0
- package/dist/runtime/transport.js +166 -0
- package/dist/runtime/transport.js.map +1 -0
- package/dist/runtime/types.d.ts +28 -9
- package/dist/runtime/types.js.map +1 -1
- package/dist/runtime/worker-client.js +40 -13
- package/dist/runtime/worker-client.js.map +1 -1
- package/dist/runtime/worker-server.js +105 -24
- package/dist/runtime/worker-server.js.map +1 -1
- package/dist/sdk/connected-types.d.ts +42 -0
- package/dist/sdk/connected-types.js +2 -0
- package/dist/sdk/connected-types.js.map +1 -0
- package/dist/sdk/connected.d.ts +3 -0
- package/dist/sdk/connected.js +232 -0
- package/dist/sdk/connected.js.map +1 -0
- package/dist/sdk/definitions.d.ts +4 -1
- package/dist/sdk/definitions.js +16 -2
- package/dist/sdk/definitions.js.map +1 -1
- package/dist/sdk/execution-definition.d.ts +3 -0
- package/dist/sdk/execution-definition.js +59 -0
- package/dist/sdk/execution-definition.js.map +1 -0
- package/dist/sdk/execution-driver.d.ts +6 -0
- package/dist/sdk/execution-driver.js +86 -0
- package/dist/sdk/execution-driver.js.map +1 -0
- package/dist/sdk/execution-observation.d.ts +3 -0
- package/dist/sdk/execution-observation.js +35 -0
- package/dist/sdk/execution-observation.js.map +1 -0
- package/dist/sdk/execution-options.d.ts +17 -0
- package/dist/sdk/execution-options.js +115 -0
- package/dist/sdk/execution-options.js.map +1 -0
- package/dist/sdk/execution-types.d.ts +72 -0
- package/dist/sdk/execution-types.js +2 -0
- package/dist/sdk/execution-types.js.map +1 -0
- package/dist/sdk/execution.d.ts +3 -0
- package/dist/sdk/execution.js +3 -0
- package/dist/sdk/execution.js.map +1 -0
- package/dist/sdk/hosted-error.d.ts +3 -0
- package/dist/sdk/hosted-error.js +14 -0
- package/dist/sdk/hosted-error.js.map +1 -0
- package/dist/sdk/hosted.d.ts +23 -0
- package/dist/sdk/hosted.js +204 -0
- package/dist/sdk/hosted.js.map +1 -0
- package/dist/sdk/permission-validation.js +10 -1
- package/dist/sdk/permission-validation.js.map +1 -1
- package/dist/sdk/runner.d.ts +5 -0
- package/dist/sdk/runner.js +107 -0
- package/dist/sdk/runner.js.map +1 -0
- package/dist/sdk/tools.js +4 -1
- package/dist/sdk/tools.js.map +1 -1
- package/dist/sdk/types.d.ts +29 -1
- package/dist/sdk/types.js.map +1 -1
- package/package.json +6 -3
- package/sdk/access-config.md +60 -4
- package/sdk/adapter-contract.md +13 -3
- package/sdk/additional-harnesses.md +43 -0
- package/sdk/agent-skill.md +2 -2
- package/sdk/agent.md +6 -4
- package/sdk/approvals.md +5 -1
- package/sdk/authentication.md +1 -1
- package/sdk/cli/dashboard-design.md +6 -0
- package/sdk/cli/index.md +9 -6
- package/sdk/cli/output.md +3 -1
- package/sdk/completion-notifications.md +2 -0
- package/sdk/config.md +1 -1
- package/sdk/copilot.md +53 -0
- package/sdk/cursor.md +68 -0
- package/sdk/diagnostics.md +29 -0
- package/sdk/distribution.md +5 -3
- package/sdk/evals.md +1 -1
- package/sdk/examples/chat-tool.ts +126 -0
- package/sdk/execution.md +743 -0
- package/sdk/fx.md +50 -5
- package/sdk/harnesses.md +13 -13
- package/sdk/index.md +45 -20
- package/sdk/message-delivery.md +2 -0
- package/sdk/opencode.md +57 -0
- package/sdk/permissions.md +6 -2
- package/sdk/plugins/sub-agents.md +4 -2
- package/sdk/project-team.md +25 -1
- package/sdk/sessions.md +2 -0
- package/sdk/tools.md +4 -2
- package/sdk/v1-runtime.md +8 -6
- package/dist/cli/catalog.d.ts +0 -26
- package/dist/cli/catalog.js +0 -41
- package/dist/cli/catalog.js.map +0 -1
package/sdk/fx.md
CHANGED
|
@@ -27,14 +27,13 @@ function fx(options: FxOptions): FxConfig;
|
|
|
27
27
|
|
|
28
28
|
`FxConfig` is a readonly harness configuration with `kind: "fx"`, the required `model`, optional `effort`, and optional `permissionMode`. It is a member of `HarnessConfig` and can appear alone or in an ordered `harness` array. The constructor only validates and declares configuration; it does not start a process or access credentials. Unknown fields, including `fast`, are rejected. Native fx does not expose a compatible fast-mode selector in the supported ACP interface; its native fast-mode preference remains in effect.
|
|
29
29
|
|
|
30
|
-
|
|
30
|
+
Gateway models use the exact AI Gateway `provider/model` identifier. Other routes use their native model identifier without inferring the authentication provider from the model name. An explicit effort must be supported by the native session's advertised configuration for that model. Unsupported effort fails with `UNSUPPORTED_OPTION` before a task is submitted. Omitting effort preserves the native model's default. The adapter verifies the selected model and explicit effort; it does not silently substitute a model. A model appearing in the catalog does not guarantee access through a particular Gateway team or key.
|
|
31
31
|
|
|
32
|
-
Direct CLI invocation does not require a definition: `subharness run fx "Review the current diff."` uses the native
|
|
32
|
+
Direct CLI invocation does not require a definition: `subharness run fx "Review the current diff."` uses the selected native provider's default model when it can be verified before submission. `--model <provider/model>` selects it explicitly. Both forms retain the access and permission requirements below; neither discovers a subscription or enables ambient paid credentials. The TypeScript constructor still requires `model`.
|
|
33
33
|
|
|
34
34
|
## Access
|
|
35
35
|
|
|
36
|
-
fx
|
|
37
|
-
|
|
36
|
+
fx accepts explicitly selected Gateway API-key/OIDC access, native saved logins, and API keys for native custom connections. Omitting `access.fx` leaves fx without an enabled connection; an empty array explicitly disables it. Ambient credentials do not enable a route. Gateway connections retain their existing shape and behavior; the following example selects one explicitly.
|
|
38
37
|
```json
|
|
39
38
|
{
|
|
40
39
|
"access": {
|
|
@@ -45,7 +44,53 @@ fx supports explicit `vercel-api-key` and `vercel-oidc` connections through the
|
|
|
45
44
|
}
|
|
46
45
|
```
|
|
47
46
|
|
|
48
|
-
|
|
47
|
+
## Native saved logins and custom API keys
|
|
48
|
+
|
|
49
|
+
Expanded access is verified against fx 0.0.11 and requires the described native status and ACP provider capabilities. Existing Gateway compatibility with fx 0.0.9 is retained.
|
|
50
|
+
|
|
51
|
+
A `subscription` connection requires `provider: "gateway"`, `"codex"`, or `"grok"`. The corresponding native setup is `fx login`, `fx login codex`, or `fx login grok`, performed outside Subharness. These are fx-owned logins; the adapter never imports another harness's tokens. `gateway` selects the saved Vercel login and still bills AI Gateway. The connection type identifies native saved-login access, not proof of a subscription plan or free usage.
|
|
52
|
+
|
|
53
|
+
```json
|
|
54
|
+
{ "access": { "fx": [{ "type": "subscription", "provider": "codex" }] } }
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
For saved Gateway login, native settings must explicitly select `credential_source: "fx_login"`. An automatic preference or a stored-key preference is insufficient because refresh must not fall through to another billing source. The adapter does not change that preference. Native status must report the exact selected login source; startup then initializes native ACP with the same environment. Native discovery, refresh, credential migration, and persistence remain fx's responsibility and may update its own credential stores. Subharness does not read or copy those stores. Local status and successful initialization do not prove remote entitlement or quota.
|
|
58
|
+
|
|
59
|
+
For direct `api-key`, `provider` names a preexisting connection in `~/.fx/settings.json`, and `env` explicitly selects the input credential variable. The native connection must use `protocol: "openai-chat-completions"` and `auth: { "type": "bearer", "env": "NATIVE_KEY_SLOT" }`. Built-in provider names cannot be used for this route. The adapter reads bounded connection metadata, resolves the selected input key, and supplies it only to the native connection's declared environment slot in the child. Input and output variable names can differ. The adapter never writes the key or endpoint into native settings.
|
|
60
|
+
|
|
61
|
+
```json
|
|
62
|
+
{
|
|
63
|
+
"access": {
|
|
64
|
+
"fx": [{ "type": "api-key", "provider": "openrouter", "env": "MY_ROUTER_KEY", "envFile": ".env.local" }]
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
A corresponding native connection, configured through fx, can be:
|
|
70
|
+
|
|
71
|
+
```json
|
|
72
|
+
{
|
|
73
|
+
"providers": {
|
|
74
|
+
"openrouter": {
|
|
75
|
+
"protocol": "openai-chat-completions",
|
|
76
|
+
"base_url": "https://openrouter.ai/api/v1",
|
|
77
|
+
"auth": { "type": "bearer", "env": "OPENROUTER_API_KEY" }
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Connection names follow native fx syntax: `[A-Za-z][A-Za-z0-9_-]*`, at most 64 characters, with built-in names reserved case-insensitively. Native endpoints require HTTPS, or HTTP on `localhost`, `127.0.0.1`, or `[::1]`, without user information, query, or fragment. The native connection retains its API path prefix.
|
|
84
|
+
|
|
85
|
+
Model metadata and defaults for custom connections remain native fx settings. The adapter does not invent context limits, protocols, or provider-specific capabilities. Anonymous custom connections are not represented by `api-key`; Anthropic Messages and generic Responses endpoints cannot be treated as Chat Completions. Native custom-connection vision limitations still apply even when metadata advertises image support.
|
|
86
|
+
|
|
87
|
+
Credential output slots cannot name process-control or routing variables such as `HOME`, `PATH`, `FX_PROVIDER`, `FX_MODEL`, `FX_AUTH_MODE`, loader/proxy controls, or native diagnostic/endpoint controls. Inapplicable access fields, malformed profiles, unknown connections, unsupported protocols, and unsafe slots fail before native startup. Both status and ACP use the same selected provider, model, and environment. Custom readiness checks the native provider name, endpoint, bearer slot metadata, and ACP provider/model selection; the generic status label `configured provider` alone is insufficient.
|
|
88
|
+
|
|
89
|
+
For all routes, inherited host-managed authentication, test endpoint overrides, competing Gateway credentials, and recording controls cannot replace the explicitly selected route. Native settings are fingerprinted around startup and before each task; a detected change requires a new session. This is not atomic protection against changes during a native request. Native permissions, approvals, same-session follow-ups, cancellation, and callback drain keep their existing behavior.
|
|
90
|
+
|
|
91
|
+
## Gateway credential selection
|
|
92
|
+
|
|
93
|
+
The connection reads only the named variable from the selected file. Existing main-checkout and worktree path rules apply. For Gateway connections, the adapter supplies the selected credential through the fx subprocess environment, fixes the native provider to AI Gateway, and removes competing credential variables and endpoint overrides. It does not save the key in native settings or place it in process arguments. Native shell processes can inherit that environment, so the credential is available within the native process tree. ACP does not provide an isolated credential channel for this release. Credential isolation from native tools belongs to the external harness or execution environment; this adapter does not provide it.
|
|
49
94
|
|
|
50
95
|
Native authentication preferences can override environment credentials. Startup runs native credential-source introspection with the same environment and requires the expected environment source. A known conflicting native login fails with `ACCESS_UNAVAILABLE` and guidance to select the environment source in fx. An unrecognized or malformed introspection result fails with `PROTOCOL_ERROR`, without trying another route. The adapter never changes native authentication preferences itself. It checks native settings for changes during startup and before each new task; detected changes fail with `INVALID_CONFIG` and require a new session. ACP does not provide atomic, in-process credential-source attestation, so native settings must remain stable while the session is active. OIDC project validation and token-lifetime rules are the same as for other Gateway connections.
|
|
51
96
|
|
package/sdk/harnesses.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Harnesses
|
|
2
2
|
|
|
3
|
-
V1 integrates Codex, Claude Code,
|
|
3
|
+
V1 integrates Codex, Claude Code, fx, OpenCode, GitHub Copilot, and Cursor. Compatibility means connecting the external harness, not merely calling its model through an API.
|
|
4
4
|
|
|
5
5
|
The library provides coordination and definitions. Harnesses own inference, native tools, conversation state, and context management. The execution environment provides any sandboxing or worktree isolation.
|
|
6
6
|
|
|
@@ -16,23 +16,23 @@ Automatic fallback ends at task submission. Execution failure, allowance exhaust
|
|
|
16
16
|
|
|
17
17
|
## Capabilities
|
|
18
18
|
|
|
19
|
-
| Operation | Codex | Claude Code | fx |
|
|
20
|
-
| --- | --- | --- | --- |
|
|
21
|
-
| Native session follow-ups | Supported | Supported | Supported |
|
|
22
|
-
| Queue between library tasks | Supported | Supported | Supported |
|
|
23
|
-
| Active steering | Native steering | Uses interrupt semantics | Uses interrupt semantics |
|
|
24
|
-
| Interruption | Native
|
|
25
|
-
| Custom tools | Native dynamic tools | Native SDK MCP tools | Private MCP HTTP tools |
|
|
26
|
-
| Declared-child launcher permissions |
|
|
27
|
-
| Prompt-free recovery of failed work | Explicit unsupported result | Explicit unsupported result | Explicit unsupported result |
|
|
28
|
-
| Startup readiness check without a turn | Supported | Supported | Supported |
|
|
19
|
+
| Operation | Codex | Claude Code | fx | OpenCode | Copilot | Cursor |
|
|
20
|
+
| --- | --- | --- | --- | --- | --- | --- |
|
|
21
|
+
| Native session follow-ups | Supported | Supported | Supported | Supported | Supported | Supported |
|
|
22
|
+
| Queue between library tasks | Supported | Supported | Supported | Supported | Supported | Supported |
|
|
23
|
+
| Active steering | Native steering | Uses interrupt semantics | Uses interrupt semantics | Uses interrupt semantics | Uses interrupt semantics | Uses interrupt semantics |
|
|
24
|
+
| Interruption | Native stop confirmation | Native stop confirmation | Native ACP stop confirmation | Native abort stop confirmation | Native abort stop confirmation | Native cancellation stop confirmation |
|
|
25
|
+
| Custom tools | Native dynamic tools | Native SDK MCP tools | Private MCP HTTP tools | Private MCP tools | Native SDK callbacks | Native SDK custom tools or subscription MCP tools |
|
|
26
|
+
| Declared-child launcher permissions | Effective native policy | Exact session-only launcher rules | Effective native policy | Effective native policy | Effective native policy | Effective native policy |
|
|
27
|
+
| Prompt-free recovery of failed work | Explicit unsupported result | Explicit unsupported result | Explicit unsupported result | Explicit unsupported result | Explicit unsupported result | Explicit unsupported result |
|
|
28
|
+
| Startup readiness check without a turn | Supported | Supported | Supported | Supported | Supported | Supported |
|
|
29
29
|
|
|
30
30
|
`resume` remains a stable command; these adapters return `RECOVERY_UNSUPPORTED` when they cannot resume failed work without replay. The queue remains paused. Unsupported native steering follows the [interrupt contract](message-delivery.md), including cancellation propagation and a new replacement task; output reports the effective mode.
|
|
31
31
|
|
|
32
|
-
All TypeScript constructors require a model.
|
|
32
|
+
All TypeScript constructors require a model. The Codex, Claude Code, and fx direct [CLI harness targets](cli/index.md) may omit it to select and retain a verifiable native default within the authorized access route. OpenCode, Copilot, and Cursor require an explicit CLI model and reject effort. Codex and Claude Code default fast mode to false. fx exposes model and optional native effort. Cursor additionally exposes its optional SDK sandbox mode for API-key access; explicit sandbox settings are unsupported for subscription ACP access. Explicit options are validated where the native interface exposes compatibility. A provider can reject a request after submission; that is an execution failure, not permission to choose another model. Model/provider substitutions detected by an adapter are rejected.
|
|
33
33
|
|
|
34
34
|
Claude fast mode with subscription access is unavailable in v1 because shared agent configuration does not authorize additional subscription spending. Explicit paid API or Gateway access can request it where supported. Native subscription eligibility and provider distribution terms still apply; subscription login is not an account-wide spending cap.
|
|
35
35
|
|
|
36
|
-
See [native adapter behavior](adapter-contract.md) for authentication isolation, native permissions, tool transport, and lifecycle details.
|
|
36
|
+
See [native adapter behavior](adapter-contract.md), [Additional native harnesses](additional-harnesses.md), [OpenCode](opencode.md), [Copilot](copilot.md), and [Cursor](cursor.md) for authentication isolation, native permissions, tool transport, and lifecycle details.
|
|
37
37
|
|
|
38
38
|
Codex permission behavior is version- and environment-dependent. The adapter does not assume that an installed Codex version can accept session-scoped launcher rules. Explicit [permission options](permissions.md) select native sandbox and network settings; the adapter does not broaden them automatically. A native approval request therefore remains possible for declared-child delegation even though the child is authorized by the Subharness definition.
|
package/sdk/index.md
CHANGED
|
@@ -1,25 +1,50 @@
|
|
|
1
|
-
# SDK and CLI
|
|
1
|
+
# SDK and CLI reference
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
This index is the entry point to the authoritative subharness contracts. The TypeScript SDK supports application-owned execution and connections to the shared local coordinator. The CLI lets a coding agent or developer run native harnesses directly, reuse specialists, and coordinate sessions, queues, responses, and approvals.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
## SDK API
|
|
6
|
+
|
|
7
|
+
- [Connected and embedded execution](execution.md): client and runner ownership, sessions, tasks, complete responses, approvals, and hosted delegation.
|
|
8
|
+
- [Agent definitions](agent.md): package exports, fields, harness options, and definition examples.
|
|
9
|
+
- [Custom tools](tools.md): validated application functions exposed through native harness integrations.
|
|
10
|
+
- [Subagents](plugins/sub-agents.md): nested delegation, lineage, context, and result-delivery boundaries.
|
|
11
|
+
|
|
12
|
+
## CLI
|
|
13
|
+
|
|
14
|
+
- [CLI](cli/index.md): commands, input, admission, waiting, follow-ups, control, and native permission responses.
|
|
15
|
+
- [CLI output](cli/output.md): compact text, typed JSONL records, states, and exit codes.
|
|
16
|
+
- [Dashboard presentation](cli/dashboard-design.md): the read-only live terminal hierarchy and interaction contract.
|
|
17
|
+
- [Agent discovery](config.md): repository and global definitions, naming, and personal worktree settings.
|
|
6
18
|
- [Agent skill](agent-skill.md): the installable skill that teaches a coding agent to delegate with the CLI.
|
|
7
|
-
- [Agent definitions](agent.md): package exports, fields, harness options, and examples.
|
|
8
|
-
- [Custom tools](tools.md): validated tool functions exposed through native harness integrations.
|
|
9
|
-
- [Discovery](config.md): repository/global definitions and personal worktree settings.
|
|
10
|
-
- [Personal access](access-config.md): subscription discovery, explicit API keys, and project OIDC.
|
|
11
|
-
- [Native permissions](permissions.md): session permission options, native limits, and background delegation.
|
|
12
|
-
- [Permission requests](approvals.md): request-specific schemas, structured answers, and approval lifecycle.
|
|
13
|
-
- [Harnesses](harnesses.md): selection, capabilities, and fallback boundaries.
|
|
14
|
-
- [CLI](cli/index.md): commands and response waiting.
|
|
15
|
-
- [Output](cli/output.md): compact text and typed JSONL records.
|
|
16
|
-
- [Sessions](sessions.md): task identity, state, and recovery.
|
|
17
|
-
- [Message delivery](message-delivery.md): queue, steer, interrupt, and cancellation.
|
|
18
|
-
- [Subagents](plugins/sub-agents.md): nested delegation and context boundaries.
|
|
19
|
-
- [Response delivery](completion-notifications.md): return points and parent continuation.
|
|
20
|
-
- [Runtime](v1-runtime.md): execution ownership, loading, limits, and failure handling.
|
|
21
|
-
- [Native adapters](adapter-contract.md): integration behavior and capability limits.
|
|
22
19
|
|
|
23
|
-
|
|
20
|
+
## Lifecycle and access
|
|
21
|
+
|
|
22
|
+
- [Sessions](sessions.md): task identity, state, queue ownership, and recovery boundaries.
|
|
23
|
+
- [Message delivery](message-delivery.md): queue, steer, interrupt, cancellation, and failure handling.
|
|
24
|
+
- [Response delivery](completion-notifications.md): complete responses, return points, and parent continuation.
|
|
25
|
+
- [Permission requests](approvals.md): request-specific CLI schemas, structured SDK actions, ownership, and lifecycle.
|
|
26
|
+
- [Authentication](authentication.md): native credentials, subscriptions, billing, and trust boundaries.
|
|
27
|
+
- [Personal access configuration](access-config.md): subscription discovery, explicit API keys, project OIDC, and local renewal.
|
|
28
|
+
- [Native permissions](permissions.md): session policies, native limits, and background delegation.
|
|
29
|
+
- [Error diagnostics](diagnostics.md): safe failure context and recovery guidance.
|
|
30
|
+
- [Distribution](distribution.md): package identity, installation, verification, and release boundaries.
|
|
31
|
+
- [Harness selection](harnesses.md): supported targets, capabilities, and fallback boundaries.
|
|
24
32
|
|
|
25
|
-
|
|
33
|
+
## Harness details
|
|
34
|
+
|
|
35
|
+
- [Additional native harnesses](additional-harnesses.md): shared configuration for OpenCode, GitHub Copilot, and Cursor.
|
|
36
|
+
- [fx](fx.md): fx ACP behavior, tools, access, permissions, and lifecycle.
|
|
37
|
+
- [OpenCode](opencode.md): OpenCode integration and supported access routes.
|
|
38
|
+
- [GitHub Copilot](copilot.md): Copilot integration and supported access routes.
|
|
39
|
+
- [Cursor](cursor.md): Cursor subscription and API-key integration boundaries.
|
|
40
|
+
|
|
41
|
+
## Maintainer internals
|
|
42
|
+
|
|
43
|
+
- [Native adapter contract](adapter-contract.md): integration behavior and capability limits.
|
|
44
|
+
- [v1 runtime](v1-runtime.md): execution ownership, loading, limits, and failure handling.
|
|
45
|
+
- [Dashboard inspection](cli/dashboard-inspection.md): coordinator projection used by the terminal dashboard.
|
|
46
|
+
- [Evaluations](evals.md): maintained evaluation inputs and scoring boundaries.
|
|
47
|
+
- [Repository agent team](project-team.md): reusable roles used to develop this project.
|
|
48
|
+
- [PR integration](pr-integration.md): reviewed-head and integration requirements.
|
|
49
|
+
|
|
50
|
+
These documents describe the v1 contract. They do not authorize automatic conversation migration, a custom model harness, sandbox provisioning, or a separate pipeline-definition API.
|
package/sdk/message-delivery.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Message Delivery
|
|
2
2
|
|
|
3
|
+
The [execution SDK](execution.md) binds these queue, steer, interrupt, and cancellation semantics to session/task handles. Observer abortion detaches observation; it does not cancel work. Session or embedded-runner close stops owned execution explicitly; connected-client disconnect does not.
|
|
4
|
+
|
|
3
5
|
`subharness send <session-id> --delivery <queue|steer|interrupt> [--detach] --prompt <text>` controls delivery. `queue` is the default. Each session has one active library task and a FIFO queue; a library task can contain multiple native turns while coordinating descendants.
|
|
4
6
|
|
|
5
7
|
| Mode | Effect |
|
package/sdk/opencode.md
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# OpenCode adapter
|
|
2
|
+
|
|
3
|
+
The OpenCode adapter runs the installed `opencode` executable in native HTTP server mode on a private loopback endpoint. This adapter supports OpenCode 1.18.32. It checks the executable version before starting the server and rejects other or unverifiable versions with `INVALID_CONFIG`, because configuration isolation depends on that version's native switches. OpenCode owns inference, native tools, conversation state, and compaction. The public configuration and access routes are defined in [additional harnesses](additional-harnesses.md).
|
|
4
|
+
|
|
5
|
+
## Direct API keys
|
|
6
|
+
|
|
7
|
+
An explicit `api-key` connection requires `provider` and `env`; `envFile` and `baseUrl` are optional. Supported single-key providers are `anthropic`, `openai`, `google`, `groq`, `openrouter`, `xai`, `mistral`, `cohere`, `opencode` (Zen), and `opencode-go` (Go). Provider IDs that require additional cloud identity, resource, or region settings are not represented by this single-key connection. Unknown providers fail explicitly instead of trying ambient credentials.
|
|
8
|
+
|
|
9
|
+
The agent model uses the exact native `provider/model` ID and its prefix must equal the connection's provider. OpenRouter model names may contain additional slashes. The installed native catalog must contain the exact model and compatible provider metadata; the adapter does not invent custom model metadata. An optional `baseUrl` overrides the selected catalog provider's endpoint. Without it, the pinned native SDK supplies that provider's endpoint. Startup verifies the explicit provider, token configuration, model, and any endpoint override without generation. Native default endpoints and SDK transformations are not remote executed-model attestation.
|
|
10
|
+
|
|
11
|
+
```json
|
|
12
|
+
{
|
|
13
|
+
"access": {
|
|
14
|
+
"opencode": [{ "type": "api-key", "provider": "anthropic", "env": "ANTHROPIC_API_KEY" }]
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Direct keys retain private configuration, data, state, and cache directories. Only the chosen provider is enabled; its environment-key discovery is disabled and the selected key is supplied explicitly. Saved native auth and default auth plugins are excluded. Both the main and auxiliary model are pinned to the exact native model. Existing Gateway behavior remains available independently.
|
|
20
|
+
|
|
21
|
+
## Native saved logins
|
|
22
|
+
|
|
23
|
+
A `subscription` connection requires `provider: "openai"`, `"github-copilot"`, or `"xai"`, selecting the corresponding built-in native OAuth integration. The user completes OpenCode's native login first. The model uses the same exact `provider/model` syntax as direct access, with a matching provider prefix. Saved API-key and well-known credentials are not treated as OAuth subscriptions. Plan names alone do not establish credential type or eligibility.
|
|
24
|
+
|
|
25
|
+
```json
|
|
26
|
+
{ "access": { "opencode": [{ "type": "subscription", "provider": "openai" }] } }
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
This route intentionally uses OpenCode's native persistent XDG data directory because OpenCode stores and refreshes its login there. The native process can access its complete authentication store and persist normal native data, including session history and token refreshes. Subharness does not read, copy, synchronize, or rewrite the credential file. It does not perform login or logout. Config, state, and cache remain private; project/provider configuration discovery remains disabled. Only built-in authentication plugins are enabled, with the explicitly selected inference provider and pinned main/auxiliary model. The subscription route does not offer an endpoint override.
|
|
30
|
+
|
|
31
|
+
Readiness requires the selected connected provider's OAuth plugin configuration and the exact native model. A missing login is unavailable; ambiguous, malformed, or conflicting auth metadata is an error. Detection relies on this pinned release's native provider source and OAuth markers, without extracting tokens or pretending that every native fetch transformation is exposed as an endpoint. A known saved API-key route cannot silently replace the requested OAuth route. Closing removes only adapter-owned private state; it never removes or restores the native persistent data directory.
|
|
32
|
+
|
|
33
|
+
## Gateway isolation
|
|
34
|
+
|
|
35
|
+
For Gateway connections, the public model is a Gateway `creator/model` identifier. The native provider is `vercel`; the adapter supplies the corresponding `vercel/creator/model` native selection. Startup uses a private configuration and private native state directories, enables only the Gateway provider, and pins the main and auxiliary model to the requested identifier. It disables native model fallback, project/provider configuration discovery, default plugins, automatic updates, and sharing so inherited provider configuration cannot change the selected billing route.
|
|
36
|
+
|
|
37
|
+
Repository instruction discovery is restored through the native `instructions` setting without importing repository provider configuration. Starting at the execution directory and stopping at the nearest Git worktree root, the adapter finds `AGENTS.md`, otherwise `CLAUDE.md`, otherwise `CONTEXT.md`, using the first filename category with matches. Matching ancestor files are passed as absolute native instruction paths. Outside Git, only the execution directory is searched. Native per-file instruction handling remains native. Subharness specialist instructions supplement the first task. Native configuration, plugins, and custom native agent definitions excluded by this isolation profile are not copied into the temporary state.
|
|
38
|
+
|
|
39
|
+
Residual home or machine-managed configuration that would still load outside the private directories must be checked before native startup. If it cannot be established compatible with the isolated profile, startup fails with `INVALID_CONFIG`; organizational policy is never bypassed using test-only environment overrides. Missing model metadata is also an error; the adapter does not invent context limits or image capabilities for unknown models.
|
|
40
|
+
|
|
41
|
+
Only the selected Gateway credential is supplied for inference. An API key and an OIDC token retain their distinct Gateway authentication modes; the adapter must not pass an OIDC token as an API key. Configuration files containing credentials have mode `0600` in an owner-only temporary directory. Credentials never appear in command arguments, native diagnostic output, or shared project files. Startup checks the effective provider, endpoint, model, and native configuration before admitting prompts, and rejects incompatible or unverifiable settings. Private files are removed on startup failure and normal close.
|
|
42
|
+
|
|
43
|
+
The server is bound to `127.0.0.1` with a random per-process password, and requests are scoped to the caller's directory. Startup requires health, provider/configuration discovery, and creation of an empty native session. No generation is used to verify readiness or quota. A missing executable is `HARNESS_UNAVAILABLE`; malformed native protocol data is `PROTOCOL_ERROR`.
|
|
44
|
+
|
|
45
|
+
## Tools and permissions
|
|
46
|
+
|
|
47
|
+
Declared tools are exposed through a private authenticated loopback MCP endpoint. Arguments are validated before callbacks run. Text and image content are preserved through MCP. The adapter verifies that the native MCP connection succeeds before submitting work. Tools stop admitting calls during interruption and close, and admitted callbacks must settle before stop is reported.
|
|
48
|
+
|
|
49
|
+
Native permissions are configured to ask. Active native permission requests use the shared approval flow, exposing bounded action context and the native `once`/`reject` decisions. Reusable grants are not exposed because native instance scope can include native child sessions. Startup-time requests and other interactive questions fail with `INPUT_REQUIRED`; no permission is automatically granted. The adapter does not add launcher allow rules. Native denial is passed back to OpenCode without authorizing a replacement operation. OpenCode can retire other pending requests in the same session after a rejection; the adapter reconciles native pending requests and drops late answers to retired operations.
|
|
50
|
+
|
|
51
|
+
## Lifecycle
|
|
52
|
+
|
|
53
|
+
Each Subharness session owns one native conversation and one native server. Follow-up tasks retain that conversation. Every prompt pins the requested provider/model; detectable changes are errors without replay. A task returns the final assistant text only after successful native completion. Native errors, unfinished responses, and malformed terminal data cannot count as success. The final text limit is 1 MiB.
|
|
54
|
+
|
|
55
|
+
Interruption requests native abort and confirms the active work stopped; an accepted HTTP abort request alone is insufficient. It also drains admitted tools and retires outstanding approvals, including when native work ended before interruption began. Native stop confirmation is bounded; that deadline does not limit already-admitted host callbacks, which must settle before interruption completes. Failure to confirm native termination produces `CANCELLATION_FAILED`, and no replacement prompt is admitted to still-running work. Steering uses the shared interrupt behavior; prompt-free recovery is unsupported.
|
|
56
|
+
|
|
57
|
+
Close aborts active work, closes event streams and the native session, terminates the owned server with bounded escalation when needed, and removes private configuration/state and tool endpoints. Startup failures use the same cleanup discipline. Normal cleanup never deletes project files or user-owned OpenCode settings.
|
package/sdk/permissions.md
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# Native Permissions
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
The [execution SDK](execution.md) uses the same native permission options. Embedded SDK-hosted declared-child tools receive native custom-tool permissions; they do not install CLI launcher shell rules. Host callbacks are trusted application code and are not confined by a native sandbox.
|
|
4
|
+
|
|
5
|
+
The Codex, Claude Code, and fx constructors accept optional native permission settings directly alongside model options. These settings configure the native session without modifying persistent native settings or creating a Subharness sandbox. OpenCode instead uses its isolated native ask policy, Copilot uses native manual permission mode, and neither exposes a public permission selector. Cursor exposes its documented SDK `sandboxMode` for API-key access; subscription ACP access rejects explicit sandbox settings and supports the documented once-only native approvals. Supported native tool approvals are surfaced to the caller through the [structured approval flow](approvals.md); an explicit valid caller decision is required before authorization. The installed harness evaluates the selected policy, and the external execution environment enforces its remaining restrictions.
|
|
4
6
|
|
|
5
7
|
Omitted settings preserve native configuration. Explicit settings remain fixed for session follow-ups. A child uses its own definition and native configuration; it does not inherit the parent's explicit permission options. Invalid definitions fail before execution. Unsupported or observably rejected explicit settings fail without fallback to another harness or permission policy.
|
|
6
8
|
|
|
@@ -21,6 +23,8 @@ codex({
|
|
|
21
23
|
|
|
22
24
|
`approvalPolicy: "never"` suppresses approval prompts; it does not grant operations blocked by the selected sandbox. `danger-full-access` removes Codex's native sandbox boundary, subject to external restrictions.
|
|
23
25
|
|
|
26
|
+
By default, `workspace-write` keeps `.git`, `.agents`, and `.codex` read-only inside the writable workspace. It also protects the Git directory named by a `.git` pointer file, even when that directory is inside the workspace; for a linked worktree, that directory is in the main checkout's Git directory and holds the worktree's index. Git index updates such as `git add` can therefore fail in ordinary checkouts and linked worktrees. subharness does not add writable roots or change the native policy to permit these operations.
|
|
27
|
+
|
|
24
28
|
## Claude Code
|
|
25
29
|
|
|
26
30
|
```ts
|
|
@@ -62,4 +66,4 @@ For a surfaced permission request, the caller inspects the action and responds t
|
|
|
62
66
|
|
|
63
67
|
The caller reads the Subharness skill and prefers one ordinary `subharness run <target>` through known host background-command controls. It continues independent work and later collects that hosted command's output, which already contains the first response or terminal outcome. When those controls are unavailable or uncertain, it uses `run --detach`, retains the returned task and session identifiers, continues other work, and later collects the result with `wait`. `status` is optional when a snapshot or additional state is needed. Shell `&` and an unobserved process do not replace managed task records. Both paths require collecting the result and confirming its terminal outcome before reporting completion; detached admission alone is not completion.
|
|
64
68
|
|
|
65
|
-
The native caller needs permission to execute the private launcher and reach the local coordinator. Each child needs permission for its own task. This remains true for
|
|
69
|
+
The native caller needs permission to execute the private launcher and reach the local coordinator. Each child needs permission for its own task. This remains true for all six native harnesses; no constructor option promises native desktop notifications or automatic chat reactivation. OpenCode and Copilot can surface supported active permission requests through the structured approval flow. Cursor subscription sessions also expose supported once-only ACP permissions. Other Cursor interactive requests fail with `INPUT_REQUIRED`.
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Subagents
|
|
2
2
|
|
|
3
|
+
The invocation rules below describe CLI-hosted sessions. The [execution SDK](../execution.md) also supports in-memory definitions through explicitly hosted delegation tools, with private parent binding and the same task lineage/completion semantics. Embedded SDK sessions do not create or require the CLI launcher, its shell grants, or a coordinator service. Native adapter transport and permission limits still apply.
|
|
4
|
+
|
|
3
5
|
An agent's optional `subagents` map declares and authorizes the agents it can invoke. Values are ordinary `AgentDefinition` objects; children can have their own subagents. Declaration alone does not execute them or register them in a repository/global catalog.
|
|
4
6
|
|
|
5
7
|
```ts
|
|
@@ -29,9 +31,9 @@ The launcher restores its coordinator location and parent context before invokin
|
|
|
29
31
|
|
|
30
32
|
For Claude Code, declaring at least one subagent adds native permission rules for the current session's exact launcher. Equivalent quoted or unquoted spellings are allowed only when they identify the same literal executable. These rules authorize `run subagent:<name>` for the declared names, plus the CLI's `list`, `--help`, `send`, `wait`, `status`, `queue`, `cancel`, and `resume` operations. They do not authorize launching repository or global agents through `run`, another executable, or arbitrary shell commands. Follow-up and observation commands retain their documented identifier-based behavior; the permission rules do not introduce a new task-access policy.
|
|
31
33
|
|
|
32
|
-
The adapter combines these rules with declared custom-tool permissions. Those rules do not write user/project permission settings or change the native permission mode; explicit constructor options independently select session permission settings as defined in [Native Permissions](../permissions.md). Native deny rules, explicit approval requirements, managed policy, and sandbox restrictions remain authoritative.
|
|
34
|
+
The adapter combines these rules with declared custom-tool permissions. Those rules do not write user/project permission settings or change the native permission mode; explicit constructor options independently select session permission settings as defined in [Native Permissions](../permissions.md). Native deny rules, explicit approval requirements, managed policy, and sandbox restrictions remain authoritative. Supported permission requests during an active turn use the [structured approval flow](../approvals.md); unsupported interactive input and startup-time requests fail with `INPUT_REQUIRED`. Each child starts with its own native tool permissions; permission to launch it does not grant permission for its edits or shell commands. A session without declared subagents receives no delegation rules. If a launcher path cannot be represented as a literal command in the native permission syntax, startup fails with `INVALID_CONFIG` rather than adding a broader rule.
|
|
33
35
|
|
|
34
|
-
Codex receives the same declared-child instructions and private launcher, but the adapter does not add native allow rules for them. Codex must already have effective permission to execute the launcher, read the files needed by that execution, and contact the coordinator. Native behavior differs across Codex versions, managed policies, sandbox configurations, and host environments. In some combinations, commands that access protected locations such as `.subharness/`, execute the private launcher, or use loopback networking can require approval or remain unavailable; this is an environment-dependent limitation, not a universal Codex rule.
|
|
36
|
+
Codex receives the same declared-child instructions and private launcher, but the adapter does not add native allow rules for them. Codex must already have effective permission to execute the launcher, read the files needed by that execution, and contact the coordinator. Native behavior differs across Codex versions, managed policies, sandbox configurations, and host environments. In some combinations, commands that access protected locations such as `.subharness/`, execute the private launcher, or use loopback networking can require approval or remain unavailable; this is an environment-dependent limitation, not a universal Codex rule. Supported permission requests during an active turn use the [structured approval flow](../approvals.md), including requests to launch an authorized declared child; unsupported interactive input and startup-time requests fail with `INPUT_REQUIRED`. The library does not automatically approve the request, disable sandboxing, or replace shell delegation with another transport.
|
|
35
37
|
|
|
36
38
|
For every harness, the native execution environment must allow the CLI to read its private launcher and contact the coordinator over loopback HTTP. A native sandbox that blocks local networking also blocks CLI delegation. The library does not relax that policy automatically; the caller supplies explicit native permission options, native settings, or an external execution environment. A lead can instead own independent review outside a managed agent's native delegation flow when the environment cannot grant these prerequisites.
|
|
37
39
|
|
package/sdk/project-team.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Repository Agent Team
|
|
2
2
|
|
|
3
|
-
This repository keeps
|
|
3
|
+
This repository keeps nine reusable agents in `.subharness/agents/`. An agent defines a stable responsibility, model, and tool access. A skill supplies task-specific instructions that the agent reads when relevant. Adding a technique or library does not require another agent definition.
|
|
4
4
|
|
|
5
5
|
## Roles
|
|
6
6
|
|
|
@@ -9,6 +9,9 @@ This repository keeps six reusable agents in `.subharness/agents/`. An agent def
|
|
|
9
9
|
| `architect` | Codex, `gpt-6-astra`, high effort | Evaluate architecture and API decisions; develop requested design concepts and assets; perform independent visual review when assigned. |
|
|
10
10
|
| `developer` | Codex, `gpt-5.6-sol`, high effort | Implement documented behavior and application structure with focused TDD and independent review. |
|
|
11
11
|
| `visual-engineer` | fx, `anthropic/claude-fable-5.1`, native default effort | Implement approved interface styling and native vgpu/WGSL effects using the relevant skill. |
|
|
12
|
+
| `api-researcher` | Codex, `gpt-5.6-luna`, low effort | Collect primary-source API evidence: signatures, defaults, events, errors, and lifecycle rules. |
|
|
13
|
+
| `systems-researcher` | Codex, `gpt-5.6-luna`, low effort | Collect evidence from related systems, including job queues, process runners, event streams, actors, and UI state synchronization. |
|
|
14
|
+
| `planner` | Codex, `gpt-6-astra`, high effort | Turn settled contracts into bounded implementation assignments and verification criteria in ignored scratch files. |
|
|
12
15
|
| `researcher` | fx, `google/gemini-3.8-flash`, native default effort | Answer a bounded technical or design question with primary-source evidence. |
|
|
13
16
|
| `reviewer` | Claude Code, `claude-opus-5[1m]`, high effort | Independently review correctness, lifecycle, credential routing, and contract compliance without editing files. |
|
|
14
17
|
| `integrator` | Codex, `gpt-5.6-sol`, high effort | Merge only the exact PRs and reviewed heads explicitly authorized by the lead for the current assignment, following [PR integration](pr-integration.md). |
|
|
@@ -21,6 +24,20 @@ Role instructions alone do not grant native capabilities or permissions. Browsin
|
|
|
21
24
|
|
|
22
25
|
The repository's Codex roles explicitly select `approvalPolicy: "never"`, `sandboxMode: "workspace-write"`, and `networkAccessEnabled: true` for local development and coordinator access. The reviewer selects Claude `permissionMode: "dontAsk"` with `Read`, `Glob`, `Grep`, and `Bash` allowed. Its read-only review responsibility remains an instruction; allowing Bash is not a filesystem read-only boundary. The fx roles explicitly select `permissionMode: "auto"`. These policies use [native permission options](permissions.md), preserve native restrictions, and do not guarantee that every operation is permitted. Protected paths and rejected automatic reviews can still require a lead-managed execution path.
|
|
23
26
|
|
|
27
|
+
Implementation roles return unstaged working-tree changes and verification evidence. Local Git staging and commits belong to the lead unless the assignment explicitly authorizes them. When an implementation task requires moving or deleting ordinary files, roles use filesystem operations; subsequent Git staging records the renames. By default, Codex workspace-write protects `.git`, `.agents`, and `.codex` inside the writable workspace. It also protects the Git directory named by a `.git` pointer file, even when that directory is inside the workspace; for a linked worktree, that directory is in the main checkout's Git directory and holds the worktree's index. A required change under those protected paths is reported with the exact operation, affected paths, and partial effects for lead-managed execution. A copied replacement does not complete a move while the original remains.
|
|
28
|
+
|
|
29
|
+
## Assigning work by complexity
|
|
30
|
+
|
|
31
|
+
Use `api-researcher` for bounded source searches and API inventories. Use `systems-researcher` for related problems outside agent orchestration. Both use the explicitly selected Codex model, read `.agents/skills/source-research/SKILL.md`, and write only the caller-named files under `.context/research/`. Their output records source URLs or paths, retrieval dates, revisions, observed behavior, and uncertainty. They distinguish facts from inference, do not rank architectural alternatives or settle contracts, and return missing evidence to the lead. The existing fx `researcher` remains available for general technical and visual evidence tasks. The two research roles have different standing evidence responsibilities: `api-researcher` inventories caller-visible interfaces, while `systems-researcher` investigates ownership, failure handling, delivery, and retention across related domains. Keeping those lenses explicit helps a design survey cover both interface conventions and operational behavior. Individual libraries and technologies remain task inputs, not additional permanent roles.
|
|
32
|
+
|
|
33
|
+
Use `architect` for reasoning across that evidence, including API alternatives with concrete signatures and usage examples, lifecycle ownership, cancellation, errors, compatibility, and misuse cases. Recommendations do not authorize implementation. The lead settles clear choices under `AGENTS.md` and asks the human only for unresolved product tradeoffs or signatures outside delegated authority.
|
|
34
|
+
|
|
35
|
+
Use `planner` only after the relevant decisions are settled in repository documentation. It reads the actual affected implementation and writes only the caller-named files under `.context/research/`. Its assignments identify the governing contracts, owned files, dependencies, behavior-focused failing tests, validation commands, and acceptance criteria. It reports missing decisions instead of choosing public behavior. Plans and progress records are scratch artifacts; they never replace repository contracts, authorize implementation, or become public documentation. The lead validates the assignments and owns task boundaries and implementation authorization.
|
|
36
|
+
|
|
37
|
+
Use `developer` for implementation and bounded review fixes, and `reviewer` for independent review. Documentation work uses the existing developer or architect with an explicit file scope; it does not need another permanent role. Graphics-specific research uses the existing researcher with the graphics skill when relevant. The `api-researcher`, `systems-researcher`, and `planner` return results to the lead without delegation. The lead supplies bounded questions, sources or search angles, output paths, and stopping criteria so economical evidence collection does not turn into unbounded design work.
|
|
38
|
+
|
|
39
|
+
These assignments express task complexity, not a guaranteed price or model benchmark. Models are fixed in definitions; there is no automatic routing based on a guessed task difficulty, silent fallback, or credential change. When the selected model or capability is unavailable, report that limitation to the lead. A follow-up can narrow an unanswered research question in the same session; a new independent angle gets a separate session.
|
|
40
|
+
|
|
24
41
|
## Skills and task context
|
|
25
42
|
|
|
26
43
|
Repository skills live in `.agents/skills/<name>/SKILL.md`. The short index in `AGENTS.md` explains when each skill applies. Agents read only the relevant skill and supporting references for their current task. Skill contents are not concatenated into every agent's instructions.
|
|
@@ -35,6 +52,9 @@ subharness run repo:architect --prompt "Read website/design.md and .agents/skill
|
|
|
35
52
|
subharness run repo:visual-engineer --prompt "Use .agents/skills/web-interface/SKILL.md. Implement the hero layout documented in website/design.md. Limit changes to the hero styles."
|
|
36
53
|
subharness run repo:developer --prompt "Use .agents/skills/web-interface/SKILL.md. Read website/design.md. Check the hero's semantic HTML and keyboard behavior without changing styles."
|
|
37
54
|
subharness run repo:developer --prompt "Use .agents/skills/webgpu-graphics/SKILL.md. Read website/graphics.md. Verify the mascot camera's documented framing with focused tests."
|
|
55
|
+
subharness run repo:api-researcher --prompt "Use .agents/skills/source-research/SKILL.md. Read sdk/sessions.md. Record official SDK cancellation signatures in .context/research/sdk-cancellation.md."
|
|
56
|
+
subharness run repo:systems-researcher --prompt "Use .agents/skills/source-research/SKILL.md. Read sdk/message-delivery.md. Record queue and observer lifecycle evidence in .context/research/queue-lifecycle.md."
|
|
57
|
+
subharness run repo:planner --prompt "Read sdk/sessions.md and sdk/message-delivery.md. Inspect src/runtime/coordinator.ts. Write bounded verification assignments for those settled contracts to .context/research/session-verification-plan.md."
|
|
38
58
|
subharness run repo:researcher --prompt "Use .agents/skills/source-research/SKILL.md. Read sdk/fx.md and src/adapters/fx.ts. Explain cancellation behavior with file references."
|
|
39
59
|
subharness run repo:reviewer --prompt "Review the mascot camera changes against website/graphics.md. Report actionable findings without editing files."
|
|
40
60
|
```
|
|
@@ -43,6 +63,10 @@ The website's illustrative conversation shows tasks delegated to Claude Code and
|
|
|
43
63
|
|
|
44
64
|
## Delegation and review
|
|
45
65
|
|
|
66
|
+
Every implementation handoff and final PR review includes a file-purpose audit against the target branch. The planner includes this in acceptance criteria, the developer removes temporary artifacts from its changes, and the reviewer inspects the complete added and changed file inventory, including documentation and supporting files. Findings identify the file and why it has no lasting repository purpose. The lead verifies the final inventory before declaring the PR ready. A bounded review states its coverage; it does not establish that the whole PR has passed this audit.
|
|
67
|
+
|
|
68
|
+
Temporary proposals, plans, progress logs, research notes, one-off diagnostics, and review output belong in ignored `.context/` files. Settled decisions belong in canonical documentation; superseded proposals and purposeless redirect stubs are removed. Intentional regression fixtures, maintained examples, evaluation tools, and product documentation remain repository assets. File purpose, rather than its name alone, determines whether it belongs in the final changes.
|
|
69
|
+
|
|
46
70
|
CLI and runtime work uses the existing architect, developer, and reviewer roles. Separate developer sessions can own independent modules; scope and governing contracts distinguish their assignments without adding permanent roles. The architect evaluates native protocol constraints, the developer implements and integrates behavior with TDD, and the reviewer independently checks user-facing behavior, lifecycle, and credential routing. The lead coordinates contracts, assignments, and final verification. Implementation and fixes run through repository agents whenever supported, so follow-ups, queues, and review cycles exercise subharness itself. Reproducible coordination failures are evidence for library improvements under the same contract and review rules.
|
|
47
71
|
|
|
48
72
|
Repository role instructions follow the same delegation preference as the public skill and managed child instructions: use one attached `run` through known host background-command controls, continue independent work, and collect its output. Use `--detach` followed later by `wait` when those controls are unavailable or uncertain. A hosted `run` already observes the first response and does not require a redundant `wait` or unconditional `status`. This preference does not authorize undeclared children or override a role's delegation restrictions.
|
package/sdk/sessions.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Sessions and Tasks
|
|
2
2
|
|
|
3
|
+
Applications can use the [execution SDK](execution.md) through an embedded runner or a connected client. Embedded records belong to the runner; connected records belong to the same shared coordinator as CLI records. Session close stops owned execution, while client disconnect releases observation without cancelling work.
|
|
4
|
+
|
|
3
5
|
An agent definition describes reusable behavior. A session is one native conversation in a fixed execution directory. A task is a unit of work within that session. Tasks and sessions have separate opaque identifiers and are not shell process IDs.
|
|
4
6
|
|
|
5
7
|
Each `run` creates a new session; `send` creates follow-up tasks or steers the active task. A session retains native conversation state and its loaded instructions, tools, and harness configuration between tasks. The external harness owns its context and compaction.
|
package/sdk/tools.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Custom Tools
|
|
2
2
|
|
|
3
|
+
With the embedded [execution SDK](execution.md), inline tool closures execute in the application process with its unchanged cwd and environment, rather than the CLI session worker. Captured application state remains live; callbacks may be shared across sessions. SDK cancellation drains admitted callbacks and does not force-stop arbitrary JavaScript.
|
|
4
|
+
|
|
3
5
|
`tool({ description, inputSchema, execute })` defines a function that a native harness can invoke. The map key in `agent.tools` is its public name. Declaring a tool does not execute it.
|
|
4
6
|
|
|
5
7
|
```ts
|
|
@@ -27,7 +29,7 @@ All three fields are required. `description` is a nonempty string. `inputSchema`
|
|
|
27
29
|
|
|
28
30
|
Validation runs before execution; invalid arguments never reach the function. Unsupported schemas fail during definition validation. Non-serializable or oversized outputs and thrown exceptions become tool errors. Tool results are limited to 1 MiB. Errors do not masquerade as successful outputs.
|
|
29
31
|
|
|
30
|
-
|
|
32
|
+
Connected and CLI-hosted tools run in the retained definition graph owner's creation directory and environment. Managed children share that captured callback context even when their native harness uses a different requested directory; use explicit paths or arguments for child-specific operations. Embedded SDK-hosted closures use the application context described above. The adapter registers them in a native namespace without replacing the harness's native tools or model loop. Declaring a custom tool authorizes that agent to invoke it. Native tool permission policies remain with the harness.
|
|
31
33
|
|
|
32
34
|
## Image results
|
|
33
35
|
|
|
@@ -70,6 +72,6 @@ function toolResult(result: { readonly content: readonly ToolContent[] }): ToolR
|
|
|
70
72
|
|
|
71
73
|
`ToolResult` is an opaque value created by `toolResult`; callers must return it directly from `execute`. It is not a JSON transport format. Content must be nonempty, contain only the supported block fields, and include at most eight images. Image data is canonical padded standard base64 without a data-URL prefix. The decoded file signature must match its declared MIME type. Each encoded image is limited to 5 MiB; the complete rich result's JSON content is limited to 8 MiB, and combined text is limited to 1 MiB. Ordinary text and JSON results retain their 1 MiB limit.
|
|
72
74
|
|
|
73
|
-
The SDK validates rich results at the tool execution boundary. Malformed blocks report `TOOL_RESULT_INVALID`; exceeded limits report `TOOL_RESULT_TOO_LARGE`. Native adapters pass images as image content, never as base64 inside a text result
|
|
75
|
+
The SDK validates rich results at the tool execution boundary. Malformed blocks report `TOOL_RESULT_INVALID`; exceeded limits report `TOOL_RESULT_TOO_LARGE`. Native adapters pass images as image content, never as base64 inside a text result. Claude Code, fx, and OpenCode use MCP image blocks; Codex and Cursor SDK sessions use native image content items; Cursor subscription sessions use MCP image blocks. [Copilot](copilot.md) groups content by type in its native result envelope: text blocks are joined in their original text order, images retain their original image order as binary image results, and interleaving between the text and image groups is unavailable. Model-specific image support and lower native limits still apply; a text-only model cannot perform visual inspection. The [fx contract](fx.md) records a native image-tool crash affecting fx 0.0.9. The tool author controls which files are read. This helper does not grant filesystem access or fetch remote images.
|
|
74
76
|
|
|
75
77
|
Cancellation waits for already-admitted callbacks to settle. There is no cancellation signal in `execute(input)`, and completed effects are not rolled back. Streaming tool results and arbitrary external tool-definition objects are outside this API.
|
package/sdk/v1-runtime.md
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
# V1 Runtime Contract
|
|
2
2
|
|
|
3
|
-
The first usable version supplies a TypeScript definition SDK and the `subharness` CLI with Codex, Claude Code,
|
|
3
|
+
The first usable version supplies a TypeScript definition and execution SDK and the `subharness` CLI with Codex, Claude Code, fx, OpenCode, GitHub Copilot, and Cursor adapters. The library coordinates existing harnesses; it does not implement a model loop, sandbox, worktree manager, or native permission system.
|
|
4
4
|
|
|
5
5
|
## Package and definitions
|
|
6
6
|
|
|
7
|
-
The package is `subharness`, with the primary executable `subharness` and identical `agent` compatibility alias. Node.js 22.18 or newer is required. ESM exports include `agent`, `tool`, `toolResult`, `codex`, `claudeCode`, `fx`, and their definition/configuration types. The public SDK defines reusable specialists; direct CLI harness targets also execute without a definition. Execution in
|
|
7
|
+
The package is `subharness`, with the primary executable `subharness` and identical `agent` compatibility alias. Node.js 22.18 or newer is required. ESM exports include `agent`, `tool`, `toolResult`, `codex`, `claudeCode`, `fx`, `opencode`, `copilot`, `cursor`, and their definition/configuration types. The public SDK defines reusable specialists; direct CLI harness targets also execute without a definition. Execution uses either the CLI or the connected or embedded [execution SDK](execution.md). The SDK additionally exports the client, runner, handle, result, access, approval, and error types defined there. The daemon, worker, catalog, and launcher rules below describe CLI execution; an embedded runner owns its coordinator and inline tool closures in the host process, while a connected client shares the CLI coordinator. Definition objects have readonly configuration fields and no execution methods. They contain functions and schemas and are not a serialization format.
|
|
8
8
|
|
|
9
9
|
```ts
|
|
10
10
|
import { agent, codex, claudeCode, tool } from "subharness";
|
|
@@ -20,7 +20,7 @@ Global definitions live in `~/.subharness/agents/`; repository definitions live
|
|
|
20
20
|
|
|
21
21
|
Definitions are trusted executable TypeScript. Loading a catalog evaluates its modules, so callers must trust the selected project and global definition files. Imports resolve relative to each definition through normal Node package resolution, with TypeScript loading supplied by the CLI. Missing imports, invalid exports, and duplicate names fail with the source filename. A session retains its loaded instructions, tools, and harness configuration for follow-ups; new sessions load current definition sources.
|
|
22
22
|
|
|
23
|
-
An omitted `--cwd` uses the command's working directory. `--prompt-file` paths resolve against the calling command's directory, independently of `--cwd`; files use UTF-8. Prompts must contain non-whitespace text and cannot exceed 1 MiB. One quoted positional prompt, `--prompt`, and `--prompt-file` are mutually exclusive. `--prompt-file -` reads bounded UTF-8 stdin to EOF and rejects a terminal input stream. Direct harness targets accept `--model`
|
|
23
|
+
An omitted `--cwd` uses the command's working directory. `--prompt-file` paths resolve against the calling command's directory, independently of `--cwd`; files use UTF-8. Prompts must contain non-whitespace text and cannot exceed 1 MiB. One quoted positional prompt, `--prompt`, and `--prompt-file` are mutually exclusive. `--prompt-file -` reads bounded UTF-8 stdin to EOF and rejects a terminal input stream. Direct harness targets accept `--model` for `run` and `check`. Codex, Claude Code, and fx also accept `--effort`; OpenCode, Copilot, and Cursor reject it. Specialist targets reject both overrides.
|
|
24
24
|
|
|
25
25
|
`run` and task-creating `send` commands accept `--detach` with every prompt source. The option ends the requesting command's observation immediately after successful admission, so the command returns exactly the existing `started` record and exits `0`. The service explicitly completes that observation response; it does not rely on a caller abandoning a response body. Admission errors keep their records and exit codes, while later failures remain available through `wait` and `status`. Interrupt delivery confirms affected execution has stopped before replacement admission even when detached. Steering rejects `--detach` with `INVALID_ARGUMENT`. The option is an observation request and is never retained as session configuration.
|
|
26
26
|
|
|
@@ -28,7 +28,7 @@ An omitted `--cwd` uses the command's working directory. `--prompt-file` paths r
|
|
|
28
28
|
|
|
29
29
|
An on-demand local coordinator owns sessions, queues, task relationships, and response records independently of individual CLI commands. It starts automatically; users do not install or manage a service. Each native session runs in a separate worker process whose working directory is the selected execution directory. Definition loading and custom tool execution in that worker use the same directory. Direct harness sessions carry a built-in target descriptor instead of a definition source reference and do not load TypeScript modules. They supply no specialist instructions, custom tools, or declared children; native instructions and project settings remain effective. Workers inherit the submitting environment, subject to the adapter's explicit authentication selection.
|
|
30
30
|
|
|
31
|
-
`check` is a CLI startup probe rather than an execution operation. It resolves the target and access configuration as `run` does, opens the native session in an isolated worker, and closes it without admitting a task or calling the adapter's turn interface. Success is withheld until the isolated worker exits and adapter-owned native processes confirm termination. Cleanup failure or uncertain native termination produces an error; after a failed graceful close, worker termination remains bounded and retains the cleanup diagnostic.
|
|
31
|
+
`check` is a CLI startup probe rather than an execution operation. It resolves the target and access configuration as `run` does, opens the native session in an isolated worker, and closes it without admitting a task or calling the adapter's turn interface. Success is withheld until the isolated worker exits and adapter-owned native processes confirm termination. Cleanup failure or uncertain native termination produces an error; after a failed graceful close, worker termination remains bounded and retains the cleanup diagnostic. The connected SDK exposes its own startup probe as `client.agents.check` under the [execution contract](execution.md); neither probe submits a task. The CLI probe does not require a coordinator for ordinary direct or specialist checks. Managed declared-child resolution still requires the active parent context used by `run`.
|
|
32
32
|
|
|
33
33
|
Readiness means only that the selected target's startup checks passed at that moment. It does not guarantee quota, successful task execution, shell or child-launch permissions, or continued availability, and it does not recursively check descendants. The probe may create an empty native conversation, a temporary session launcher, and adapter-owned temporary files. Loading personal access settings may update Git's local exclude file. Normal worker close removes library-owned temporary files. A removal failure is reported even though files can remain, and abrupt environment termination still cannot guarantee file cleanup.
|
|
34
34
|
|
|
@@ -66,13 +66,15 @@ Ordinary text/JSON tool results and complete responses are limited to 1 MiB. Ric
|
|
|
66
66
|
|
|
67
67
|
A managed task cannot delegate follow-up work into its own session or an ancestor's session. Such an admission returns `INVALID_PARENT` before changing any execution: that queued work would otherwise wait for the very task whose completion depends on it. Invalid delegation and replacement requests never cancel existing work.
|
|
68
68
|
|
|
69
|
-
|
|
69
|
+
The CLI and connected SDK select the same coordinator directory: an explicit SDK `stateDirectory`, otherwise `AGENT_STATE_DIR`, otherwise `~/.subharness/state/shared`. Connection authenticates the instance and requires protocol version 1 with the applicable `sdk-v1` or `cli-v1` capability.
|
|
70
|
+
|
|
71
|
+
Each candidate coordinator, rather than the CLI command that spawned it, holds an operating-system-backed startup lock from its final endpoint check through service startup and atomic endpoint publication. The lock is released automatically if the candidate exits. After acquiring it, a candidate retires without publishing when the retained endpoint gives a compatible authenticated health response for the recorded coordinator instance. It starts only when metadata is absent, or a well-formed stale record has a proven-dead owner and an unresponsive endpoint. Malformed, unauthorized, incompatible, or uncertain endpoints fail explicitly without replacement. This prevents a caller exit or a later candidate from exposing competing endpoint publication.
|
|
70
72
|
|
|
71
73
|
Startup callers do not infer whether their candidate published from an endpoint observation and do not signal a candidate based on such an observation. When its startup deadline expires, a caller sends its candidate a private retirement request. The candidate serializes that request with its publication transition while it still owns the startup lock. If retirement wins, the candidate closes any service that was never published, releases startup ownership, and confirms that it cannot publish later. If publication wins, including when another caller has already admitted work, the candidate reports its published endpoint and remains running. The original caller then uses that endpoint instead of terminating the coordinator.
|
|
72
74
|
|
|
73
75
|
Lock contention is bounded by the spawning command's startup deadline. Waiting for the candidate's publication-or-retirement response uses an additional fixed confirmation window, so command completion remains bounded. If the candidate does not respond, startup fails with `STARTUP_FAILED` stating that retirement could not be confirmed, and the caller leaves the candidate untouched. A later authenticated endpoint discovery may therefore find that candidate. An unconfirmed result never claims that the process stopped and never authorizes killing a potentially published coordinator.
|
|
74
76
|
|
|
75
|
-
Coordinators leave endpoint records in place during shutdown. Legacy pathname lock artifacts do not grant startup ownership and are left untouched. An unavailable endpoint
|
|
77
|
+
Coordinators leave endpoint records in place during shutdown. Legacy pathname lock artifacts do not grant startup ownership and are left untouched. An unavailable endpoint alone does not prove absence. A live or unknown owner, authentication failure, incompatible protocol, or instance mismatch forbids replacement startup. A permitted replacement atomically overwrites only a proven-stale record; tasks owned by the previous coordinator are not reconstructed. An interrupted coordinator response is reported as unavailable without exposing native diagnostics or automatically repeating the request. Disconnecting an observer detaches that observation; it does not cancel execution.
|
|
76
78
|
A successful detached task-creation request ends its service observation explicitly after the admission record while the coordinator continues to own the task.
|
|
77
79
|
|
|
78
80
|
Catalog discovery evaluates TypeScript in a short-lived process with the selected execution directory as its working directory. Module stdout and stderr are not mixed into CLI output. Session workers retain their loaded instructions, tools, and harness configuration. New sessions, including delegated child sessions, load the current definition source; the library does not serialize or freeze arbitrary TypeScript closures across sessions.
|
package/dist/cli/catalog.d.ts
DELETED
|
@@ -1,26 +0,0 @@
|
|
|
1
|
-
import { type ErrorInfo } from "../errors.js";
|
|
2
|
-
import type { AgentReference, ExecutionReference } from "../runtime/types.js";
|
|
3
|
-
export interface CatalogSnapshot {
|
|
4
|
-
entries: {
|
|
5
|
-
id: string;
|
|
6
|
-
name: string;
|
|
7
|
-
description: string;
|
|
8
|
-
scope: "repo" | "global";
|
|
9
|
-
file: string;
|
|
10
|
-
}[];
|
|
11
|
-
children: {
|
|
12
|
-
name: string;
|
|
13
|
-
description: string;
|
|
14
|
-
}[];
|
|
15
|
-
}
|
|
16
|
-
export type CatalogReply = {
|
|
17
|
-
result: CatalogSnapshot;
|
|
18
|
-
error?: never;
|
|
19
|
-
} | {
|
|
20
|
-
result?: never;
|
|
21
|
-
error: ErrorInfo;
|
|
22
|
-
};
|
|
23
|
-
/** Definition modules run in their selected directory without sharing the CLI's output streams. */
|
|
24
|
-
export declare function loadCatalog(cwd: string, parent?: ExecutionReference): Promise<CatalogSnapshot>;
|
|
25
|
-
/** Declared-child lookup evaluates only the active parent's definition source. */
|
|
26
|
-
export declare function loadDeclaredChildren(cwd: string, parent: AgentReference): Promise<CatalogSnapshot['children']>;
|
package/dist/cli/catalog.js
DELETED
|
@@ -1,41 +0,0 @@
|
|
|
1
|
-
import { fork } from "node:child_process";
|
|
2
|
-
import { fileURLToPath } from "node:url";
|
|
3
|
-
import { AgentError } from "../errors.js";
|
|
4
|
-
/** Definition modules run in their selected directory without sharing the CLI's output streams. */
|
|
5
|
-
export function loadCatalog(cwd, parent) {
|
|
6
|
-
return loadCatalogSnapshot(cwd, parent, true);
|
|
7
|
-
}
|
|
8
|
-
/** Declared-child lookup evaluates only the active parent's definition source. */
|
|
9
|
-
export async function loadDeclaredChildren(cwd, parent) {
|
|
10
|
-
return (await loadCatalogSnapshot(cwd, parent, false)).children;
|
|
11
|
-
}
|
|
12
|
-
function loadCatalogSnapshot(cwd, parent, discover) {
|
|
13
|
-
return new Promise((resolve, reject) => {
|
|
14
|
-
const extension = import.meta.url.endsWith(".ts") ? "ts" : "js";
|
|
15
|
-
const child = fork(fileURLToPath(new URL(`./catalog-worker.${extension}`, import.meta.url)), [], {
|
|
16
|
-
cwd, env: process.env, stdio: ["ignore", "ignore", "ignore", "ipc"],
|
|
17
|
-
execArgv: extension === "ts" ? ["--import", import.meta.resolve("tsx")] : [],
|
|
18
|
-
});
|
|
19
|
-
let reply;
|
|
20
|
-
const failed = () => reject(new AgentError("INVALID_DEFINITION", "The agent catalog worker exited before returning its catalog."));
|
|
21
|
-
child.once("error", failed);
|
|
22
|
-
child.once("exit", () => {
|
|
23
|
-
if (!reply)
|
|
24
|
-
failed();
|
|
25
|
-
else if (reply.error)
|
|
26
|
-
reject(new AgentError(reply.error.code, reply.error.message));
|
|
27
|
-
else
|
|
28
|
-
resolve(reply.result);
|
|
29
|
-
});
|
|
30
|
-
child.once("message", (message) => {
|
|
31
|
-
reply = message;
|
|
32
|
-
child.send({ received: true }, (error) => { if (error)
|
|
33
|
-
child.kill(); });
|
|
34
|
-
});
|
|
35
|
-
child.send({ cwd, parent, discover }, (error) => { if (error) {
|
|
36
|
-
child.kill();
|
|
37
|
-
failed();
|
|
38
|
-
} });
|
|
39
|
-
});
|
|
40
|
-
}
|
|
41
|
-
//# sourceMappingURL=catalog.js.map
|