@namzu/sdk 46.0.0 → 48.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +351 -8
- package/README.md +7 -8
- package/dist/advisory/history.d.ts.map +1 -1
- package/dist/advisory/history.js +21 -1
- package/dist/advisory/history.js.map +1 -1
- package/dist/advisory/index.d.ts +1 -1
- package/dist/advisory/index.d.ts.map +1 -1
- package/dist/advisory/index.js +1 -1
- package/dist/advisory/index.js.map +1 -1
- package/dist/advisory/registry.d.ts +17 -0
- package/dist/advisory/registry.d.ts.map +1 -1
- package/dist/advisory/registry.js +26 -0
- package/dist/advisory/registry.js.map +1 -1
- package/dist/agents/PipelineAgent.d.ts +2 -46
- package/dist/agents/PipelineAgent.d.ts.map +1 -1
- package/dist/agents/PipelineAgent.js +2 -242
- package/dist/agents/PipelineAgent.js.map +1 -1
- package/dist/agents/QueryAgent.d.ts +21 -0
- package/dist/agents/QueryAgent.d.ts.map +1 -0
- package/dist/agents/QueryAgent.js +214 -0
- package/dist/agents/QueryAgent.js.map +1 -0
- package/dist/agents/ReactiveAgent.d.ts +4 -16
- package/dist/agents/ReactiveAgent.d.ts.map +1 -1
- package/dist/agents/ReactiveAgent.js +4 -183
- package/dist/agents/ReactiveAgent.js.map +1 -1
- package/dist/agents/RouterAgent.d.ts +2 -20
- package/dist/agents/RouterAgent.d.ts.map +1 -1
- package/dist/agents/RouterAgent.js +2 -266
- package/dist/agents/RouterAgent.js.map +1 -1
- package/dist/agents/SupervisorAgent.d.ts +3 -43
- package/dist/agents/SupervisorAgent.d.ts.map +1 -1
- package/dist/agents/SupervisorAgent.js +3 -418
- package/dist/agents/SupervisorAgent.js.map +1 -1
- package/dist/agents/defineAgent.d.ts +2 -1
- package/dist/agents/defineAgent.d.ts.map +1 -1
- package/dist/agents/defineAgent.js +6 -1
- package/dist/agents/defineAgent.js.map +1 -1
- package/dist/agents/examples/PipelineAgent.d.ts +46 -0
- package/dist/agents/examples/PipelineAgent.d.ts.map +1 -0
- package/dist/agents/examples/PipelineAgent.js +242 -0
- package/dist/agents/examples/PipelineAgent.js.map +1 -0
- package/dist/agents/examples/RouterAgent.d.ts +22 -0
- package/dist/agents/examples/RouterAgent.d.ts.map +1 -0
- package/dist/agents/examples/RouterAgent.js +268 -0
- package/dist/agents/examples/RouterAgent.js.map +1 -0
- package/dist/agents/examples/SupervisorAgent.d.ts +45 -0
- package/dist/agents/examples/SupervisorAgent.d.ts.map +1 -0
- package/dist/agents/examples/SupervisorAgent.js +406 -0
- package/dist/agents/examples/SupervisorAgent.js.map +1 -0
- package/dist/agents/explore.d.ts +6 -3
- package/dist/agents/explore.d.ts.map +1 -1
- package/dist/agents/explore.js +6 -3
- package/dist/agents/explore.js.map +1 -1
- package/dist/agents/forward-options.d.ts +11 -0
- package/dist/agents/forward-options.d.ts.map +1 -0
- package/dist/agents/forward-options.js +15 -0
- package/dist/agents/forward-options.js.map +1 -0
- package/dist/agents/index.d.ts +1 -0
- package/dist/agents/index.d.ts.map +1 -1
- package/dist/agents/index.js +2 -0
- package/dist/agents/index.js.map +1 -1
- package/dist/agents/runAgent.d.ts +19 -8
- package/dist/agents/runAgent.d.ts.map +1 -1
- package/dist/agents/runAgent.js +99 -29
- package/dist/agents/runAgent.js.map +1 -1
- package/dist/authorization/command-line.d.ts +2 -0
- package/dist/authorization/command-line.d.ts.map +1 -1
- package/dist/authorization/command-line.js +2 -2
- package/dist/authorization/command-line.js.map +1 -1
- package/dist/authorization/gate.d.ts +8 -0
- package/dist/authorization/gate.d.ts.map +1 -1
- package/dist/authorization/gate.js +12 -1
- package/dist/authorization/gate.js.map +1 -1
- package/dist/authorization/program.d.ts +151 -0
- package/dist/authorization/program.d.ts.map +1 -0
- package/dist/authorization/program.js +713 -0
- package/dist/authorization/program.js.map +1 -0
- package/dist/authorization/reexec-wrapper.d.ts +106 -0
- package/dist/authorization/reexec-wrapper.d.ts.map +1 -0
- package/dist/authorization/reexec-wrapper.js +532 -0
- package/dist/authorization/reexec-wrapper.js.map +1 -0
- package/dist/authorization/rules.d.ts +9 -0
- package/dist/authorization/rules.d.ts.map +1 -1
- package/dist/authorization/rules.js +8 -1
- package/dist/authorization/rules.js.map +1 -1
- package/dist/authorization/shell-lexer.d.ts +19 -4
- package/dist/authorization/shell-lexer.d.ts.map +1 -1
- package/dist/authorization/shell-lexer.js +91 -65
- package/dist/authorization/shell-lexer.js.map +1 -1
- package/dist/bridge/sse/mapper.d.ts.map +1 -1
- package/dist/bridge/sse/mapper.js +6 -0
- package/dist/bridge/sse/mapper.js.map +1 -1
- package/dist/capabilities/index.d.ts +54 -0
- package/dist/capabilities/index.d.ts.map +1 -0
- package/dist/capabilities/index.js +92 -0
- package/dist/capabilities/index.js.map +1 -0
- package/dist/config/registry.d.ts +2 -1
- package/dist/config/registry.d.ts.map +1 -1
- package/dist/config/registry.js +3 -2
- package/dist/config/registry.js.map +1 -1
- package/dist/connector/index.d.ts +4 -4
- package/dist/connector/index.d.ts.map +1 -1
- package/dist/connector/index.js +3 -3
- package/dist/connector/index.js.map +1 -1
- package/dist/connector/mcp/adapter.d.ts +16 -1
- package/dist/connector/mcp/adapter.d.ts.map +1 -1
- package/dist/connector/mcp/adapter.js +71 -4
- package/dist/connector/mcp/adapter.js.map +1 -1
- package/dist/connector/mcp/client.d.ts +32 -4
- package/dist/connector/mcp/client.d.ts.map +1 -1
- package/dist/connector/mcp/client.js +383 -5
- package/dist/connector/mcp/client.js.map +1 -1
- package/dist/connector/mcp/discovery.d.ts +24 -1
- package/dist/connector/mcp/discovery.d.ts.map +1 -1
- package/dist/connector/mcp/discovery.js +44 -0
- package/dist/connector/mcp/discovery.js.map +1 -1
- package/dist/connector/mcp/era.d.ts.map +1 -1
- package/dist/connector/mcp/era.js +7 -0
- package/dist/connector/mcp/era.js.map +1 -1
- package/dist/connector/mcp/index.d.ts +2 -0
- package/dist/connector/mcp/index.d.ts.map +1 -1
- package/dist/connector/mcp/index.js +11 -0
- package/dist/connector/mcp/index.js.map +1 -1
- package/dist/connector/mcp/mcp-toolset.d.ts +133 -0
- package/dist/connector/mcp/mcp-toolset.d.ts.map +1 -0
- package/dist/connector/mcp/mcp-toolset.js +399 -0
- package/dist/connector/mcp/mcp-toolset.js.map +1 -0
- package/dist/connector/mcp/streamable-http.d.ts +15 -0
- package/dist/connector/mcp/streamable-http.d.ts.map +1 -1
- package/dist/connector/mcp/streamable-http.js +281 -6
- package/dist/connector/mcp/streamable-http.js.map +1 -1
- package/dist/connector/tools/index.d.ts +2 -2
- package/dist/connector/tools/index.d.ts.map +1 -1
- package/dist/connector/tools/index.js +1 -1
- package/dist/connector/tools/index.js.map +1 -1
- package/dist/connector/tools/router.d.ts +15 -13
- package/dist/connector/tools/router.d.ts.map +1 -1
- package/dist/connector/tools/router.js +30 -51
- package/dist/connector/tools/router.js.map +1 -1
- package/dist/contracts/a2a.d.ts +2 -2
- package/dist/directory/derive-supervisor.d.ts +2 -2
- package/dist/directory/derive-supervisor.d.ts.map +1 -1
- package/dist/directory/derive-supervisor.js +6 -7
- package/dist/directory/derive-supervisor.js.map +1 -1
- package/dist/directory/derive.d.ts.map +1 -1
- package/dist/directory/derive.js +5 -5
- package/dist/directory/derive.js.map +1 -1
- package/dist/execution/code-runtime/types.d.ts +2 -2
- package/dist/execution/code-runtime/types.js +1 -1
- package/dist/invariants/index.d.ts +2 -1
- package/dist/invariants/index.d.ts.map +1 -1
- package/dist/invariants/index.js +3 -2
- package/dist/invariants/index.js.map +1 -1
- package/dist/manager/agent/lifecycle.d.ts +1 -0
- package/dist/manager/agent/lifecycle.d.ts.map +1 -1
- package/dist/manager/agent/lifecycle.js +146 -18
- package/dist/manager/agent/lifecycle.js.map +1 -1
- package/dist/manager/connector/environment.d.ts +6 -0
- package/dist/manager/connector/environment.d.ts.map +1 -1
- package/dist/manager/connector/environment.js +11 -4
- package/dist/manager/connector/environment.js.map +1 -1
- package/dist/manager/connector/index.d.ts +2 -2
- package/dist/manager/connector/index.d.ts.map +1 -1
- package/dist/manager/connector/index.js +2 -2
- package/dist/manager/connector/index.js.map +1 -1
- package/dist/manager/connector/tenant.d.ts +6 -0
- package/dist/manager/connector/tenant.d.ts.map +1 -1
- package/dist/manager/connector/tenant.js +11 -4
- package/dist/manager/connector/tenant.js.map +1 -1
- package/dist/manager/index.d.ts +2 -2
- package/dist/manager/index.d.ts.map +1 -1
- package/dist/manager/index.js +2 -2
- package/dist/manager/index.js.map +1 -1
- package/dist/manager/resident/initiative.d.ts +6 -6
- package/dist/manager/resident/learning-observation.d.ts +4 -4
- package/dist/manager/resident/learning.d.ts +36 -36
- package/dist/manager/resident/outbox.d.ts +4 -4
- package/dist/manager/session/attribution.d.ts +22 -0
- package/dist/manager/session/attribution.d.ts.map +1 -0
- package/dist/manager/session/attribution.js +64 -0
- package/dist/manager/session/attribution.js.map +1 -0
- package/dist/manager/session/turn-recorder.d.ts +14 -0
- package/dist/manager/session/turn-recorder.d.ts.map +1 -1
- package/dist/manager/session/turn-recorder.js +208 -55
- package/dist/manager/session/turn-recorder.js.map +1 -1
- package/dist/peers/address.d.ts +36 -0
- package/dist/peers/address.d.ts.map +1 -0
- package/dist/peers/address.js +52 -0
- package/dist/peers/address.js.map +1 -0
- package/dist/peers/client.d.ts +47 -0
- package/dist/peers/client.d.ts.map +1 -0
- package/dist/peers/client.js +132 -0
- package/dist/peers/client.js.map +1 -0
- package/dist/peers/dir.d.ts +88 -0
- package/dist/peers/dir.d.ts.map +1 -0
- package/dist/peers/dir.js +139 -0
- package/dist/peers/dir.js.map +1 -0
- package/dist/peers/endpoint.d.ts +144 -0
- package/dist/peers/endpoint.d.ts.map +1 -0
- package/dist/peers/endpoint.js +420 -0
- package/dist/peers/endpoint.js.map +1 -0
- package/dist/peers/envelope.d.ts +42 -0
- package/dist/peers/envelope.d.ts.map +1 -0
- package/dist/peers/envelope.js +110 -0
- package/dist/peers/envelope.js.map +1 -0
- package/dist/peers/index.d.ts +30 -0
- package/dist/peers/index.d.ts.map +1 -0
- package/dist/peers/index.js +22 -0
- package/dist/peers/index.js.map +1 -0
- package/dist/peers/protocol.d.ts +645 -0
- package/dist/peers/protocol.d.ts.map +1 -0
- package/dist/peers/protocol.js +151 -0
- package/dist/peers/protocol.js.map +1 -0
- package/dist/peers/record.d.ts +71 -0
- package/dist/peers/record.d.ts.map +1 -0
- package/dist/peers/record.js +45 -0
- package/dist/peers/record.js.map +1 -0
- package/dist/peers/registry.d.ts +54 -0
- package/dist/peers/registry.d.ts.map +1 -0
- package/dist/peers/registry.js +213 -0
- package/dist/peers/registry.js.map +1 -0
- package/dist/plugin/define.d.ts +24 -0
- package/dist/plugin/define.d.ts.map +1 -0
- package/dist/plugin/define.js +42 -0
- package/dist/plugin/define.js.map +1 -0
- package/dist/plugin/index.d.ts +2 -0
- package/dist/plugin/index.d.ts.map +1 -1
- package/dist/plugin/index.js +1 -0
- package/dist/plugin/index.js.map +1 -1
- package/dist/plugin/lifecycle.d.ts +27 -3
- package/dist/plugin/lifecycle.d.ts.map +1 -1
- package/dist/plugin/lifecycle.js +242 -92
- package/dist/plugin/lifecycle.js.map +1 -1
- package/dist/plugin/resolver.d.ts +2 -2
- package/dist/plugin/resolver.d.ts.map +1 -1
- package/dist/plugin/resolver.js.map +1 -1
- package/dist/plugin/shell-hook.d.ts.map +1 -1
- package/dist/plugin/shell-hook.js +17 -3
- package/dist/plugin/shell-hook.js.map +1 -1
- package/dist/probe/errors.d.ts +2 -1
- package/dist/probe/errors.d.ts.map +1 -1
- package/dist/probe/errors.js +3 -2
- package/dist/probe/errors.js.map +1 -1
- package/dist/prompt/coding-agent-doctrine.d.ts +18 -7
- package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
- package/dist/prompt/coding-agent-doctrine.js +11 -6
- package/dist/prompt/coding-agent-doctrine.js.map +1 -1
- package/dist/prompt/contributions.d.ts +4 -1
- package/dist/prompt/contributions.d.ts.map +1 -1
- package/dist/prompt/contributions.js +7 -2
- package/dist/prompt/contributions.js.map +1 -1
- package/dist/prompt/index.d.ts +1 -1
- package/dist/prompt/index.d.ts.map +1 -1
- package/dist/prompt/index.js +1 -1
- package/dist/prompt/index.js.map +1 -1
- package/dist/provider/collect-chat-completion.d.ts +6 -1
- package/dist/provider/collect-chat-completion.d.ts.map +1 -1
- package/dist/provider/collect-chat-completion.js +88 -10
- package/dist/provider/collect-chat-completion.js.map +1 -1
- package/dist/provider/mock.d.ts.map +1 -1
- package/dist/provider/mock.js +23 -6
- package/dist/provider/mock.js.map +1 -1
- package/dist/provider/registry.d.ts +2 -1
- package/dist/provider/registry.d.ts.map +1 -1
- package/dist/provider/registry.js +3 -2
- package/dist/provider/registry.js.map +1 -1
- package/dist/provider/tool-call-framing.d.ts +164 -0
- package/dist/provider/tool-call-framing.d.ts.map +1 -0
- package/dist/provider/tool-call-framing.js +163 -0
- package/dist/provider/tool-call-framing.js.map +1 -0
- package/dist/public-runtime.d.ts +24 -11
- package/dist/public-runtime.d.ts.map +1 -1
- package/dist/public-runtime.js +45 -9
- package/dist/public-runtime.js.map +1 -1
- package/dist/public-tools.d.ts +6 -4
- package/dist/public-tools.d.ts.map +1 -1
- package/dist/public-tools.js +8 -3
- package/dist/public-tools.js.map +1 -1
- package/dist/public-types.d.ts +13 -2
- package/dist/public-types.d.ts.map +1 -1
- package/dist/read-model/registry.d.ts +2 -1
- package/dist/read-model/registry.d.ts.map +1 -1
- package/dist/read-model/registry.js +3 -2
- package/dist/read-model/registry.js.map +1 -1
- package/dist/registry/ManagedRegistry.d.ts +30 -0
- package/dist/registry/ManagedRegistry.d.ts.map +1 -1
- package/dist/registry/ManagedRegistry.js +45 -6
- package/dist/registry/ManagedRegistry.js.map +1 -1
- package/dist/registry/collision.d.ts +57 -0
- package/dist/registry/collision.d.ts.map +1 -0
- package/dist/registry/collision.js +41 -0
- package/dist/registry/collision.js.map +1 -0
- package/dist/registry/command/index.d.ts +5 -3
- package/dist/registry/command/index.d.ts.map +1 -1
- package/dist/registry/command/index.js +6 -4
- package/dist/registry/command/index.js.map +1 -1
- package/dist/registry/index.d.ts +3 -4
- package/dist/registry/index.d.ts.map +1 -1
- package/dist/registry/index.js +2 -2
- package/dist/registry/index.js.map +1 -1
- package/dist/registry/plugin/index.d.ts.map +1 -1
- package/dist/registry/plugin/index.js +14 -1
- package/dist/registry/plugin/index.js.map +1 -1
- package/dist/registry/tool/callable.d.ts +7 -7
- package/dist/registry/tool/callable.d.ts.map +1 -1
- package/dist/registry/tool/callable.js +6 -7
- package/dist/registry/tool/callable.js.map +1 -1
- package/dist/registry/tool/execute.d.ts +14 -77
- package/dist/registry/tool/execute.d.ts.map +1 -1
- package/dist/registry/tool/execute.js +15 -727
- package/dist/registry/tool/execute.js.map +1 -1
- package/dist/registry/tool/portable.d.ts +25 -0
- package/dist/registry/tool/portable.d.ts.map +1 -1
- package/dist/registry/tool/portable.js +50 -0
- package/dist/registry/tool/portable.js.map +1 -1
- package/dist/registry/tool/presentation.d.ts +6 -5
- package/dist/registry/tool/presentation.d.ts.map +1 -1
- package/dist/registry/tool/presentation.js +3 -3
- package/dist/registry/tool/presentation.js.map +1 -1
- package/dist/runtime/bidi/session.d.ts +5 -4
- package/dist/runtime/bidi/session.d.ts.map +1 -1
- package/dist/runtime/bidi/session.js +18 -1
- package/dist/runtime/bidi/session.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +45 -2
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.js +189 -5
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -1
- package/dist/runtime/query/executor.d.ts +33 -3
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +86 -10
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/file-evidence-replay.d.ts +1 -1
- package/dist/runtime/query/file-evidence-replay.d.ts.map +1 -1
- package/dist/runtime/query/file-evidence-replay.js +8 -7
- package/dist/runtime/query/file-evidence-replay.js.map +1 -1
- package/dist/runtime/query/fork/prepare.d.ts +4 -1
- package/dist/runtime/query/fork/prepare.d.ts.map +1 -1
- package/dist/runtime/query/fork/prepare.js +4 -0
- package/dist/runtime/query/fork/prepare.js.map +1 -1
- package/dist/runtime/query/guardrail-presets.d.ts +4 -3
- package/dist/runtime/query/guardrail-presets.d.ts.map +1 -1
- package/dist/runtime/query/guardrail-presets.js +4 -3
- package/dist/runtime/query/guardrail-presets.js.map +1 -1
- package/dist/runtime/query/index.d.ts +23 -10
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +149 -70
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +8 -1
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/phases/compaction.js +1 -1
- package/dist/runtime/query/iteration/phases/compaction.js.map +1 -1
- package/dist/runtime/query/iteration/phases/context.d.ts +2 -2
- package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/tool-review.js +55 -2
- package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
- package/dist/runtime/query/iteration/stream-turn.d.ts.map +1 -1
- package/dist/runtime/query/iteration/stream-turn.js +342 -116
- package/dist/runtime/query/iteration/stream-turn.js.map +1 -1
- package/dist/runtime/query/iteration/tool-input.d.ts +79 -0
- package/dist/runtime/query/iteration/tool-input.d.ts.map +1 -0
- package/dist/runtime/query/iteration/tool-input.js +291 -0
- package/dist/runtime/query/iteration/tool-input.js.map +1 -0
- package/dist/runtime/query/observation-context.d.ts +2 -2
- package/dist/runtime/query/observation-context.d.ts.map +1 -1
- package/dist/runtime/query/observation-context.js.map +1 -1
- package/dist/runtime/query/prelude-lease.d.ts +18 -0
- package/dist/runtime/query/prelude-lease.d.ts.map +1 -0
- package/dist/runtime/query/prelude-lease.js +121 -0
- package/dist/runtime/query/prelude-lease.js.map +1 -0
- package/dist/runtime/query/prepare-turn.d.ts.map +1 -1
- package/dist/runtime/query/prepare-turn.js +386 -187
- package/dist/runtime/query/prepare-turn.js.map +1 -1
- package/dist/runtime/query/prompt-cache.d.ts +2 -2
- package/dist/runtime/query/prompt-cache.d.ts.map +1 -1
- package/dist/runtime/query/prompt.d.ts +2 -2
- package/dist/runtime/query/prompt.d.ts.map +1 -1
- package/dist/runtime/query/prompt.js +1 -1
- package/dist/runtime/query/prompt.js.map +1 -1
- package/dist/runtime/query/resume-session.d.ts.map +1 -1
- package/dist/runtime/query/resume-session.js +37 -2
- package/dist/runtime/query/resume-session.js.map +1 -1
- package/dist/runtime/query/review-policy.d.ts +22 -3
- package/dist/runtime/query/review-policy.d.ts.map +1 -1
- package/dist/runtime/query/review-policy.js +84 -6
- package/dist/runtime/query/review-policy.js.map +1 -1
- package/dist/runtime/query/tooling.d.ts +3 -2
- package/dist/runtime/query/tooling.d.ts.map +1 -1
- package/dist/runtime/query/tooling.js.map +1 -1
- package/dist/runtime/query/turn-state.d.ts.map +1 -1
- package/dist/runtime/query/turn-state.js +5 -0
- package/dist/runtime/query/turn-state.js.map +1 -1
- package/dist/runtime/system-events.d.ts +112 -0
- package/dist/runtime/system-events.d.ts.map +1 -0
- package/dist/runtime/system-events.js +183 -0
- package/dist/runtime/system-events.js.map +1 -0
- package/dist/scheduler/delegating.d.ts +2 -1
- package/dist/scheduler/delegating.d.ts.map +1 -1
- package/dist/scheduler/delegating.js +3 -2
- package/dist/scheduler/delegating.js.map +1 -1
- package/dist/scheduler/local.d.ts.map +1 -1
- package/dist/scheduler/local.js +2 -0
- package/dist/scheduler/local.js.map +1 -1
- package/dist/session/retention/archive.js +1 -1
- package/dist/session/retention/archive.js.map +1 -1
- package/dist/skills/index.d.ts +1 -1
- package/dist/skills/index.d.ts.map +1 -1
- package/dist/skills/index.js +1 -1
- package/dist/skills/index.js.map +1 -1
- package/dist/skills/registry.d.ts +12 -0
- package/dist/skills/registry.d.ts.map +1 -1
- package/dist/skills/registry.js +19 -0
- package/dist/skills/registry.js.map +1 -1
- package/dist/store/evidence/index-page.d.ts +2 -2
- package/dist/store/session/disk.d.ts.map +1 -1
- package/dist/store/session/disk.js +2 -0
- package/dist/store/session/disk.js.map +1 -1
- package/dist/store/session-log/conformance.d.ts +1 -1
- package/dist/store/session-log/conformance.d.ts.map +1 -1
- package/dist/store/session-log/conformance.js +49 -8
- package/dist/store/session-log/conformance.js.map +1 -1
- package/dist/store/session-log/core.d.ts +8 -3
- package/dist/store/session-log/core.d.ts.map +1 -1
- package/dist/store/session-log/core.js +2 -1
- package/dist/store/session-log/core.js.map +1 -1
- package/dist/store/session-log/fold.d.ts +12 -0
- package/dist/store/session-log/fold.d.ts.map +1 -1
- package/dist/store/session-log/fold.js +57 -7
- package/dist/store/session-log/fold.js.map +1 -1
- package/dist/store/session-log/index.d.ts +2 -2
- package/dist/store/session-log/index.d.ts.map +1 -1
- package/dist/store/session-log/index.js +1 -1
- package/dist/store/session-log/index.js.map +1 -1
- package/dist/test-support/toolset.d.ts +19 -0
- package/dist/test-support/toolset.d.ts.map +1 -0
- package/dist/test-support/toolset.js +20 -0
- package/dist/test-support/toolset.js.map +1 -0
- package/dist/tools/builtins/bash.d.ts +29 -0
- package/dist/tools/builtins/bash.d.ts.map +1 -1
- package/dist/tools/builtins/bash.js +19 -9
- package/dist/tools/builtins/bash.js.map +1 -1
- package/dist/tools/builtins/browser-url.d.ts +1 -1
- package/dist/tools/builtins/browser-url.js +1 -1
- package/dist/tools/builtins/browser.d.ts +6 -6
- package/dist/tools/builtins/browser.js +2 -2
- package/dist/tools/builtins/browser.js.map +1 -1
- package/dist/tools/builtins/computer-use.d.ts.map +1 -1
- package/dist/tools/builtins/computer-use.js +19 -11
- package/dist/tools/builtins/computer-use.js.map +1 -1
- package/dist/tools/builtins/edit.d.ts.map +1 -1
- package/dist/tools/builtins/edit.js +9 -1
- package/dist/tools/builtins/edit.js.map +1 -1
- package/dist/tools/builtins/json-string-hint.d.ts +15 -0
- package/dist/tools/builtins/json-string-hint.d.ts.map +1 -0
- package/dist/tools/builtins/json-string-hint.js +17 -0
- package/dist/tools/builtins/json-string-hint.js.map +1 -0
- package/dist/tools/builtins/search-tools.d.ts.map +1 -1
- package/dist/tools/builtins/search-tools.js +7 -15
- package/dist/tools/builtins/search-tools.js.map +1 -1
- package/dist/tools/builtins/write-file.d.ts.map +1 -1
- package/dist/tools/builtins/write-file.js +11 -0
- package/dist/tools/builtins/write-file.js.map +1 -1
- package/dist/tools/command-shell.d.ts +10 -0
- package/dist/tools/command-shell.d.ts.map +1 -1
- package/dist/tools/command-shell.js +45 -1
- package/dist/tools/command-shell.js.map +1 -1
- package/dist/tools/coordinator/agent.d.ts.map +1 -1
- package/dist/tools/coordinator/agent.js +5 -0
- package/dist/tools/coordinator/agent.js.map +1 -1
- package/dist/tools/coordinator/ask-user-question.d.ts.map +1 -1
- package/dist/tools/coordinator/ask-user-question.js +26 -13
- package/dist/tools/coordinator/ask-user-question.js.map +1 -1
- package/dist/tools/coordinator/index.d.ts.map +1 -1
- package/dist/tools/coordinator/index.js +5 -0
- package/dist/tools/coordinator/index.js.map +1 -1
- package/dist/tools/coordinator/question-options.d.ts +78 -0
- package/dist/tools/coordinator/question-options.d.ts.map +1 -0
- package/dist/tools/coordinator/question-options.js +153 -0
- package/dist/tools/coordinator/question-options.js.map +1 -0
- package/dist/tools/defineTool.d.ts +10 -0
- package/dist/tools/defineTool.d.ts.map +1 -1
- package/dist/tools/defineTool.js +17 -0
- package/dist/tools/defineTool.js.map +1 -1
- package/dist/tools/memory/search.d.ts.map +1 -1
- package/dist/tools/memory/search.js +4 -1
- package/dist/tools/memory/search.js.map +1 -1
- package/dist/tools/memory/update.d.ts.map +1 -1
- package/dist/tools/memory/update.js +8 -5
- package/dist/tools/memory/update.js.map +1 -1
- package/dist/tools/render-nonce.d.ts +47 -0
- package/dist/tools/render-nonce.d.ts.map +1 -0
- package/dist/tools/render-nonce.js +63 -0
- package/dist/tools/render-nonce.js.map +1 -0
- package/dist/tools/roster.d.ts +46 -19
- package/dist/tools/roster.d.ts.map +1 -1
- package/dist/tools/roster.js +63 -33
- package/dist/tools/roster.js.map +1 -1
- package/dist/tools/schedules/schedule-tool.d.ts.map +1 -1
- package/dist/tools/schedules/schedule-tool.js +94 -14
- package/dist/tools/schedules/schedule-tool.js.map +1 -1
- package/dist/tools/schedules/types.d.ts +52 -6
- package/dist/tools/schedules/types.d.ts.map +1 -1
- package/dist/tools/trusted-read-only.d.ts +9 -5
- package/dist/tools/trusted-read-only.d.ts.map +1 -1
- package/dist/tools/trusted-read-only.js +14 -10
- package/dist/tools/trusted-read-only.js.map +1 -1
- package/dist/tools/untrusted-envelope.d.ts +67 -28
- package/dist/tools/untrusted-envelope.d.ts.map +1 -1
- package/dist/tools/untrusted-envelope.js +108 -59
- package/dist/tools/untrusted-envelope.js.map +1 -1
- package/dist/toolsets/combine.d.ts +44 -0
- package/dist/toolsets/combine.d.ts.map +1 -0
- package/dist/toolsets/combine.js +153 -0
- package/dist/toolsets/combine.js.map +1 -0
- package/dist/toolsets/manager.d.ts +201 -0
- package/dist/toolsets/manager.d.ts.map +1 -0
- package/dist/toolsets/manager.js +837 -0
- package/dist/toolsets/manager.js.map +1 -0
- package/dist/toolsets/source-glob.d.ts +15 -0
- package/dist/toolsets/source-glob.d.ts.map +1 -0
- package/dist/toolsets/source-glob.js +23 -0
- package/dist/toolsets/source-glob.js.map +1 -0
- package/dist/toolsets/toolset.d.ts +20 -0
- package/dist/toolsets/toolset.d.ts.map +1 -0
- package/dist/toolsets/toolset.js +28 -0
- package/dist/toolsets/toolset.js.map +1 -0
- package/dist/toolsets/types.d.ts +128 -0
- package/dist/toolsets/types.d.ts.map +1 -0
- package/dist/toolsets/types.js +33 -0
- package/dist/toolsets/types.js.map +1 -0
- package/dist/toolsets/wrappers.d.ts +59 -0
- package/dist/toolsets/wrappers.d.ts.map +1 -0
- package/dist/toolsets/wrappers.js +165 -0
- package/dist/toolsets/wrappers.js.map +1 -0
- package/dist/types/agent/base.d.ts +27 -16
- package/dist/types/agent/base.d.ts.map +1 -1
- package/dist/types/agent/index.d.ts +1 -0
- package/dist/types/agent/index.d.ts.map +1 -1
- package/dist/types/agent/index.js +1 -0
- package/dist/types/agent/index.js.map +1 -1
- package/dist/types/agent/pipeline.d.ts +2 -0
- package/dist/types/agent/pipeline.d.ts.map +1 -1
- package/dist/types/agent/query.d.ts +117 -0
- package/dist/types/agent/query.d.ts.map +1 -0
- package/dist/types/agent/query.js +2 -0
- package/dist/types/agent/query.js.map +1 -0
- package/dist/types/agent/reactive.d.ts +2 -129
- package/dist/types/agent/reactive.d.ts.map +1 -1
- package/dist/types/agent/router.d.ts +2 -0
- package/dist/types/agent/router.d.ts.map +1 -1
- package/dist/types/agent/scheduler.d.ts +6 -1
- package/dist/types/agent/scheduler.d.ts.map +1 -1
- package/dist/types/agent/supervisor.d.ts +5 -2
- package/dist/types/agent/supervisor.d.ts.map +1 -1
- package/dist/types/agent/task.d.ts +19 -1
- package/dist/types/agent/task.d.ts.map +1 -1
- package/dist/types/agent/task.js.map +1 -1
- package/dist/types/authorization/index.d.ts +49 -12
- package/dist/types/authorization/index.d.ts.map +1 -1
- package/dist/types/authorization/index.js +6 -0
- package/dist/types/authorization/index.js.map +1 -1
- package/dist/types/connector/mcp.d.ts +45 -0
- package/dist/types/connector/mcp.d.ts.map +1 -1
- package/dist/types/errors/index.d.ts +15 -0
- package/dist/types/errors/index.d.ts.map +1 -1
- package/dist/types/errors/index.js.map +1 -1
- package/dist/types/hitl/index.d.ts +57 -0
- package/dist/types/hitl/index.d.ts.map +1 -1
- package/dist/types/hitl/index.js.map +1 -1
- package/dist/types/message/index.d.ts +130 -9
- package/dist/types/message/index.d.ts.map +1 -1
- package/dist/types/message/index.js +4 -1
- package/dist/types/message/index.js.map +1 -1
- package/dist/types/plugin/index.d.ts +22 -4
- package/dist/types/plugin/index.d.ts.map +1 -1
- package/dist/types/plugin/index.js +1 -0
- package/dist/types/plugin/index.js.map +1 -1
- package/dist/types/provider/chat.d.ts +5 -0
- package/dist/types/provider/chat.d.ts.map +1 -1
- package/dist/types/provider/config.d.ts +8 -3
- package/dist/types/provider/config.d.ts.map +1 -1
- package/dist/types/provider/stream.d.ts +39 -0
- package/dist/types/provider/stream.d.ts.map +1 -1
- package/dist/types/sandbox/index.d.ts +1 -1
- package/dist/types/session/events.d.ts +14 -3
- package/dist/types/session/events.d.ts.map +1 -1
- package/dist/types/session/events.js.map +1 -1
- package/dist/types/session/sub-session.d.ts +2 -0
- package/dist/types/session/sub-session.d.ts.map +1 -1
- package/dist/types/tool/index.d.ts +152 -91
- package/dist/types/tool/index.d.ts.map +1 -1
- package/dist/types/tool/index.js.map +1 -1
- package/dist/types/toolset/index.d.ts +10 -40
- package/dist/types/toolset/index.d.ts.map +1 -1
- package/package.json +1 -1
- package/src/advisory/history.ts +27 -2
- package/src/advisory/index.ts +1 -1
- package/src/advisory/registry.ts +33 -0
- package/src/agents/AGENTS.md +14 -7
- package/src/agents/PipelineAgent.ts +2 -303
- package/src/agents/QueryAgent.ts +269 -0
- package/src/agents/ReactiveAgent.ts +6 -216
- package/src/agents/RouterAgent.ts +2 -337
- package/src/agents/SupervisorAgent.ts +3 -476
- package/src/agents/defineAgent.ts +13 -2
- package/src/agents/examples/PipelineAgent.ts +302 -0
- package/src/agents/examples/RouterAgent.ts +338 -0
- package/src/agents/examples/SupervisorAgent.ts +472 -0
- package/src/agents/explore.ts +6 -3
- package/src/agents/forward-options.ts +37 -0
- package/src/agents/index.ts +2 -0
- package/src/agents/runAgent.ts +144 -37
- package/src/authorization/command-line.ts +2 -2
- package/src/authorization/gate.ts +21 -1
- package/src/authorization/program.ts +773 -0
- package/src/authorization/reexec-wrapper.ts +589 -0
- package/src/authorization/rules.ts +18 -1
- package/src/authorization/shell-lexer.ts +104 -59
- package/src/bridge/sse/mapper.ts +6 -0
- package/src/capabilities/index.ts +157 -0
- package/src/config/registry.ts +4 -1
- package/src/connector/index.ts +9 -2
- package/src/connector/mcp/adapter.ts +73 -3
- package/src/connector/mcp/client.ts +425 -6
- package/src/connector/mcp/discovery.ts +60 -0
- package/src/connector/mcp/era.ts +7 -0
- package/src/connector/mcp/index.ts +13 -0
- package/src/connector/mcp/mcp-toolset.ts +515 -0
- package/src/connector/mcp/streamable-http.ts +302 -5
- package/src/connector/tools/index.ts +2 -2
- package/src/connector/tools/router.ts +35 -58
- package/src/directory/derive-supervisor.ts +12 -8
- package/src/directory/derive.ts +8 -5
- package/src/execution/code-runtime/types.ts +2 -2
- package/src/invariants/index.ts +4 -1
- package/src/manager/agent/lifecycle.ts +167 -19
- package/src/manager/connector/environment.ts +17 -4
- package/src/manager/connector/index.ts +2 -2
- package/src/manager/connector/tenant.ts +17 -4
- package/src/manager/index.ts +2 -2
- package/src/manager/session/attribution.ts +79 -0
- package/src/manager/session/turn-recorder.ts +211 -55
- package/src/peers/address.ts +58 -0
- package/src/peers/client.ts +202 -0
- package/src/peers/dir.ts +176 -0
- package/src/peers/endpoint.ts +564 -0
- package/src/peers/envelope.ts +121 -0
- package/src/peers/index.ts +96 -0
- package/src/peers/protocol.ts +196 -0
- package/src/peers/record.ts +56 -0
- package/src/peers/registry.ts +241 -0
- package/src/plugin/define.ts +68 -0
- package/src/plugin/index.ts +2 -0
- package/src/plugin/lifecycle.ts +286 -106
- package/src/plugin/resolver.ts +6 -3
- package/src/plugin/shell-hook.ts +21 -5
- package/src/probe/errors.ts +4 -1
- package/src/prompt/coding-agent-doctrine.ts +22 -10
- package/src/prompt/contributions.ts +12 -2
- package/src/prompt/index.ts +1 -0
- package/src/provider/collect-chat-completion.ts +96 -11
- package/src/provider/mock.ts +24 -6
- package/src/provider/registry.ts +4 -1
- package/src/provider/tool-call-framing.ts +226 -0
- package/src/public-runtime.ts +136 -7
- package/src/public-tools.ts +24 -4
- package/src/public-types.ts +60 -6
- package/src/read-model/registry.ts +7 -2
- package/src/registry/ManagedRegistry.ts +57 -6
- package/src/registry/collision.ts +64 -0
- package/src/registry/command/index.ts +7 -3
- package/src/registry/index.ts +3 -9
- package/src/registry/plugin/index.ts +14 -1
- package/src/registry/tool/callable.ts +8 -9
- package/src/registry/tool/execute.ts +16 -846
- package/src/registry/tool/portable.ts +59 -0
- package/src/registry/tool/presentation.ts +6 -5
- package/src/runtime/bidi/session.ts +24 -5
- package/src/runtime/query/executor/tool-call-admission.ts +228 -6
- package/src/runtime/query/executor.ts +110 -12
- package/src/runtime/query/file-evidence-replay.ts +8 -7
- package/src/runtime/query/fork/prepare.ts +6 -1
- package/src/runtime/query/guardrail-presets.ts +4 -3
- package/src/runtime/query/index.ts +199 -101
- package/src/runtime/query/iteration/index.ts +13 -3
- package/src/runtime/query/iteration/phases/compaction.ts +1 -1
- package/src/runtime/query/iteration/phases/context.ts +2 -2
- package/src/runtime/query/iteration/phases/tool-review.ts +59 -2
- package/src/runtime/query/iteration/stream-turn.ts +390 -136
- package/src/runtime/query/iteration/tool-input.ts +284 -0
- package/src/runtime/query/observation-context.ts +2 -2
- package/src/runtime/query/prelude-lease.ts +132 -0
- package/src/runtime/query/prepare-turn.ts +477 -198
- package/src/runtime/query/prompt-cache.ts +2 -2
- package/src/runtime/query/prompt.ts +4 -4
- package/src/runtime/query/resume-session.ts +42 -3
- package/src/runtime/query/review-policy.ts +87 -9
- package/src/runtime/query/tooling.ts +3 -6
- package/src/runtime/query/turn-state.ts +4 -0
- package/src/runtime/system-events.ts +246 -0
- package/src/scheduler/delegating.ts +7 -2
- package/src/scheduler/local.ts +2 -0
- package/src/session/retention/archive.ts +1 -1
- package/src/skills/index.ts +1 -1
- package/src/skills/registry.ts +25 -0
- package/src/store/session/disk.ts +3 -0
- package/src/store/session-log/conformance.ts +51 -8
- package/src/store/session-log/core.ts +10 -4
- package/src/store/session-log/fold.ts +33 -7
- package/src/store/session-log/index.ts +2 -0
- package/src/test-support/toolset.ts +23 -0
- package/src/tools/builtins/bash.ts +27 -10
- package/src/tools/builtins/browser-url.ts +1 -1
- package/src/tools/builtins/browser.ts +2 -2
- package/src/tools/builtins/computer-use.ts +21 -11
- package/src/tools/builtins/edit.ts +12 -2
- package/src/tools/builtins/json-string-hint.ts +16 -0
- package/src/tools/builtins/search-tools.ts +7 -20
- package/src/tools/builtins/write-file.ts +12 -0
- package/src/tools/command-shell.ts +46 -1
- package/src/tools/coordinator/agent.ts +6 -0
- package/src/tools/coordinator/ask-user-question.ts +29 -15
- package/src/tools/coordinator/index.ts +6 -0
- package/src/tools/coordinator/question-options.ts +195 -0
- package/src/tools/defineTool.ts +28 -0
- package/src/tools/memory/search.ts +4 -1
- package/src/tools/memory/update.ts +8 -5
- package/src/tools/render-nonce.ts +69 -0
- package/src/tools/roster.ts +84 -33
- package/src/tools/schedules/schedule-tool.ts +107 -14
- package/src/tools/schedules/types.ts +52 -6
- package/src/tools/trusted-read-only.ts +19 -10
- package/src/tools/untrusted-envelope.ts +121 -57
- package/src/toolsets/combine.ts +181 -0
- package/src/toolsets/manager.ts +996 -0
- package/src/toolsets/source-glob.ts +22 -0
- package/src/toolsets/toolset.ts +31 -0
- package/src/toolsets/types.ts +154 -0
- package/src/toolsets/wrappers.ts +191 -0
- package/src/types/agent/base.ts +29 -16
- package/src/types/agent/index.ts +1 -0
- package/src/types/agent/pipeline.ts +2 -0
- package/src/types/agent/query.ts +129 -0
- package/src/types/agent/reactive.ts +5 -142
- package/src/types/agent/router.ts +2 -0
- package/src/types/agent/scheduler.ts +6 -1
- package/src/types/agent/supervisor.ts +5 -2
- package/src/types/agent/task.ts +21 -1
- package/src/types/authorization/index.ts +12 -0
- package/src/types/connector/mcp.ts +46 -0
- package/src/types/errors/index.ts +15 -0
- package/src/types/hitl/index.ts +57 -0
- package/src/types/message/index.ts +135 -7
- package/src/types/plugin/index.ts +44 -21
- package/src/types/provider/chat.ts +5 -0
- package/src/types/provider/config.ts +8 -3
- package/src/types/provider/stream.ts +39 -0
- package/src/types/sandbox/index.ts +1 -1
- package/src/types/session/events.ts +14 -3
- package/src/types/session/sub-session.ts +2 -0
- package/src/types/tool/index.ts +154 -105
- package/src/types/toolset/index.ts +10 -47
- package/dist/registry/toolset/catalog.d.ts +0 -42
- package/dist/registry/toolset/catalog.d.ts.map +0 -1
- package/dist/registry/toolset/catalog.js +0 -234
- package/dist/registry/toolset/catalog.js.map +0 -1
- package/src/registry/toolset/catalog.ts +0 -308
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,349 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 48.1.0
|
|
4
|
+
|
|
5
|
+
### Minor Changes
|
|
6
|
+
|
|
7
|
+
- 3a0daac: Allow delegated CLI agents to opt into a separate managed Git worktree with
|
|
8
|
+
`workspace: "worktree"`. The child runs from the selected checkout's committed
|
|
9
|
+
HEAD, preserving the parent's directory beneath the Git root. The worktree
|
|
10
|
+
remains available for review after any outcome. SDK
|
|
11
|
+
callers can choose a per-child shared or isolated workspace and explicitly
|
|
12
|
+
retain it; isolated requests accept a checked relative `subdirectory`. A
|
|
13
|
+
manager can make omitted choices shared with
|
|
14
|
+
`workspaceDefault: 'shared'`; existing SDK backend provisioning and cleanup
|
|
15
|
+
remain the defaults.
|
|
16
|
+
- 3a0daac: `MCPClient.callTool` accepts `onProgress` and requests a unique MCP progress token per tool call. Concurrent calls receive only their own validated, bounded progress updates; callbacks stop on completion or cancellation. Streamable HTTP dispatches progress while the SSE response remains open and releases the reader as soon as the matching final reply arrives. MCP tools forward updates through `ToolContext.report` for live host and CLI status.
|
|
17
|
+
- 3a0daac: Modern MCP connections now follow advertised tool, prompt, and resource catalogue changes through `subscriptions/listen`. `mcpToolset` refreshes its live definitions after an acknowledged change, and Streamable HTTP reads the subscription incrementally instead of waiting for the long-lived response to end. Hosts using `StreamableHttpTransport` directly can use `sendSubscription` for a bounded SSE stream; existing legacy notification behavior is unchanged.
|
|
18
|
+
|
|
19
|
+
### Patch Changes
|
|
20
|
+
|
|
21
|
+
- c8bb87a: Describe a browser `http-auth` handoff as an HTTP authentication challenge. The old "password prompt" wording was inaccurate for responses using challenge schemes such as Bearer.
|
|
22
|
+
- 3a0daac: Reject an explicitly isolated child workspace when a registered Git worktree
|
|
23
|
+
driver returns a missing path or a path that resolves to the caller's directory.
|
|
24
|
+
SDK hosts using a custom driver must return an existing, separate checkout.
|
|
25
|
+
- 3a0daac: MCP Streamable HTTP connections now surface a 401 or 403 response to the modern protocol probe as an access failure, without attempting a legacy initialize handshake. Correct the server credentials or access policy and reconnect.
|
|
26
|
+
- 3a0daac: Modern MCP subscriptions now stop retrying a server that permanently refuses `subscriptions/listen`, including HTTP 404 with JSON-RPC method-not-found. Temporary server errors and rate limits still retry with backoff. Operators get a warning and can reconnect after changing the server configuration.
|
|
27
|
+
An HTTP 200 JSON-RPC method-not-found refusal also stops retries, while a capacity refusal still retries. The response body is read with size and time bounds and released on completion, timeout or disconnect, preventing held responses from accumulating.
|
|
28
|
+
|
|
29
|
+
## 48.0.0
|
|
30
|
+
|
|
31
|
+
### Major Changes
|
|
32
|
+
|
|
33
|
+
- 567ada8: The SDK now accepts application-defined agent type strings and provides `QueryAgent` as the generic managed query adapter. `defineAgent` gives its `run` callback a required fourth `AbortSignal` parameter and creates fresh instances for delegated turns. Code that invokes a `DefineAgentOptions.run` callback directly must supply that fourth argument; existing three-argument callback implementations remain assignable. `ReactiveAgent` and the supervisor, pipeline and router exports remain available as deprecated compatibility names. CLI delegated turns use a CLI-owned `NamzuCliAgent` while keeping their existing configuration behavior.
|
|
34
|
+
- 567ada8: `PluginHookDefinition` now checks each hook's returned action against its event. A TypeScript plugin whose handler is declared to return the full `PluginHookResult` union, or returns an action the runtime cannot use for that event, may stop compiling. Narrow the handler return type to `PluginHookResultFor<'event_name'>` or let TypeScript infer its specific result. For example, `retry` belongs to `post_tool_use`, `modify` to `pre_tool_use`, and `annotate` to `user_prompt_submit`. Session start/end and turn interrupt hooks can only declare `continue` as a result. JavaScript runtime behavior is unchanged.
|
|
35
|
+
- 567ada8: Plugin tools now carry a source for their owning plugin (`plugin:<name>`) or MCP server (`plugin:<name>/mcp:<server>`), so source authorization can distinguish them. Plugin MCP servers use `mcpToolset`, gaining live tool and prompt discovery, policy-filtered deferred resources and held changed definitions. Plugin MCP prompt tool names change from `<plugin>__mcp_prompt_<server>_<name>` to `<plugin>__mcp__<server>__prompt__<name>`; update saved tool names and permission rules that refer to them. `MCPToolsetOptions.reconnect` also accepts a function that returns the current reconnect policy.
|
|
36
|
+
- 567ada8: Every registry now throws on a duplicate id by default instead of silently overwriting or skipping. `ManagedRegistry`'s own default flips from warn-and-overwrite to throw (`RegistryCollisionError`), and the same convergence applies to registries that were not built on it: `AdvisorRegistry.register` now throws `AdvisorCollisionError` instead of overwriting in silence; `SkillRegistry.add` now throws `SkillCollisionError` instead of overwriting in silence; `EnvironmentConnectorManager.registerEnvironment` and `TenantConnectorManager.registerTenant` now throw `EnvironmentCollisionError`/`TenantCollisionError` instead of logging a warning and silently keeping the original. `AgentRegistry`, `ConnectorRegistry` and `HostCommandRegistry` (built on `ManagedRegistry` with no registry-specific override) throw by the same default flip.
|
|
37
|
+
|
|
38
|
+
`PluginRegistry` is the named exception and keeps its overwrite behaviour. `ToolRegistry` is removed in this release; toolset name collisions throw `ToolsetConflictError`. Pass `onCollision: 'warn-overwrite'` explicitly if you build your own `ManagedRegistry` subclass and need the old default. A single call that needs to replace one entry on a registry that now throws calls the new `ManagedRegistry.replace(id, item)` instead — see `docs/sdk/registries.md`.
|
|
39
|
+
|
|
40
|
+
Every SDK `*CollisionError` (`ToolNameCollisionError`, `HostCommandNameCollisionError`, `ConfigNamespaceCollisionError`, `PromptContributionCollisionError`, `ReadModelCollisionError`, `InvariantNameCollisionError`, `ProbeNameCollisionError`, `DuplicateProviderError`, plus the four new ones above, plus `DelegateIdCollisionError` (`scheduler/delegating.ts`)) now extends the new `RegistryCollisionError` base class; names, messages and fields are unchanged, and a catch block naming the concrete class still works. A catch that wants "some registry collided" without naming every class can catch `RegistryCollisionError`. See [docs/sdk/registries.md](../docs/sdk/registries.md).
|
|
41
|
+
|
|
42
|
+
If you called `register`/`registerEnvironment`/`registerTenant`/`SkillRegistry.add` a second time under the same id expecting a silent overwrite or a silent skip, that call now throws. Call `unregister` first, or (on a registry built on `ManagedRegistry`) call the new `replace(id, item)` instead.
|
|
43
|
+
|
|
44
|
+
- 567ada8: Removed `ToolCatalog`, `createToolCatalogFromRegistry`, `loadingFromAvailability`, `toolDefinitionToCatalogEntry`, `ToolsetDefinition`, `ToolsetPolicy`, `ToolCatalogEntry`, `ToolCatalogSearchResult`, `ToolCatalogSnapshot` and `ToolLoadingMode`. This was a parallel, unwired model of sources/toolsets/tools with a per-tool policy: no query run ever called `toLLMTools`/`getToolsByLoading`/`searchTools` on it. This release also removes `ToolRegistry` and moves live discovery to `ToolManager` and `ToolResult.reveals`; see the `tool-registry-removed` changeset. `ToolSource` and `ToolSourceKind` stay.
|
|
45
|
+
|
|
46
|
+
If you constructed a `ToolCatalog` directly or called `createToolCatalogFromRegistry`, pass `toolsets` to the runtime and use `ToolManager`'s `listNames()` / `availability(name)` for an advanced host's roster view. Nothing else in the kernel ever populated the catalog from live state.
|
|
47
|
+
|
|
48
|
+
- fbeac55: A scheduled job can now be a fixed shell script instead of a model prompt, or a cheap "wake-gate" script that decides whether to run the model at all — `namzu schedule add --kind script|script+agent`, or the model's `schedule` tool with `kind`/`script`. A `script` job spends zero tokens and opens no session; a `script+agent` job runs a gate script first and calls the model only when it prints `{"wake": true, "context": "..."}`. The script body is checked against the scheduled-run floor (the whole text, exactly as a live `bash` call is read) and then every `deny` rule — the operator's own and any config file's — per lexed command; **`allow`/`ask` rules and `unmatched` are never consulted for a script**, since it is fixed, human-confirmed text, not a call a model improvises. A `script+agent` job's `allow`/`ask`/`unmatched` rules still govern its AGENT phase only, unaffected by the script's own (deny-only) check.
|
|
49
|
+
|
|
50
|
+
**`@namzu/sdk` (major)**
|
|
51
|
+
|
|
52
|
+
- **Breaking type correction:** `ScheduleJobPreview.model` is optional because a pure `script` job has no model. Hosts that display it must handle its absence for `runKind: 'script'`; previews for `agent` and `script+agent` still carry the pinned model.
|
|
53
|
+
- `ScheduleJobPreview.networkAccess` now includes a host script or an allowed host shell that could reach the network, even without web/browser tool grants. Its new optional `networkGrantAccess` separates those grants from shell capability; a host implementing `previewUpdate` should set it so a kind-only conversion can be checked against the network-beside-host-shell rule without refusing every script. An older host that omits it is treated conservatively using `networkAccess`.
|
|
54
|
+
- An unattended sandbox-escape opt-in no longer waives review of a runtime-chosen program name or another call's outside-root path in the same batch.
|
|
55
|
+
- **Security fix:** a recognised wrapper such as `env`, `nice`, `sudo`, `busybox` or `toybox` can no longer hide a shell's `-c` payload from review. Literal payloads are read for permission checks; an expanding payload requires review and is refused in unattended runs. `script -t -c` now recognises `-c` as the command option; `-T` consumes its required timing-file value.
|
|
56
|
+
- **Security fix:** `find`'s `-exec` family now follows wrappers and shell `-c` payloads inside each clause, including when another wrapper launches `find`. A dynamically selected program or a `find` placeholder in a shell payload requires live review and is refused in an unattended scheduled script. The scheduled-run floor also sees service-changing tools behind wrappers in these clauses.
|
|
57
|
+
- **Breaking:** `ScheduleJobDraft.prompt` is now optional (`prompt?: string`), since a pure `script` draft has none. Any `ScheduleToolHost` implementation that reads `draft.prompt` as a bare `string` needs a null check (or to keep reading it only when `draft.runKind` is `'agent'`/`'script+agent'`, where it is still guaranteed present).
|
|
58
|
+
- `ScheduleJobDraft` gains `runKind?: 'agent' | 'script' | 'script+agent'` (absent = `'agent'`, unchanged from before) and `script?: { body, shell, timeoutMs? }`. `ScheduleJobPreview` gains the same `runKind`/`script` fields so a host's confirmation screen can show the exact text.
|
|
59
|
+
- The `schedule` tool's input schema gains `kind` and `script`; `create` requires `script` unless `kind` is `'agent'` and requires `prompt` unless `kind` is `'script'`, and refuses an agent draft that also carries a script. The existing network-beside-a-host-shell refusal now also covers a `script`/`script+agent` job on the host: the script IS shell code, not a rule a live call might reach later.
|
|
60
|
+
- **Security/UX fix:** `update` accepted `kind`/`script` in its input schema but silently dropped both from the change it actually applied, reporting success with no error or warning that the model's new script was ignored. `ScheduleJobChanges` gains `runKind`/`script` (additive, so this alone would be `minor`) and `update` now forwards both to `ScheduleToolHost.previewUpdate`, re-verified fresh against the scheduled-run floor and every `deny` rule exactly like a new job's, and shown in full — with what changed — on the same confirmation screen `create` uses. A host that does not yet read the new fields off `ScheduleJobChanges` silently keeps a job's existing kind/script on every update, the same as before this release; only `@namzu/cli`'s own host (below) newly acts on them.
|
|
61
|
+
- **Security fix:** `update` checks the effective preview's network grants after applying every change, so a kind-only conversion cannot inherit web/browser access beside a newly introduced host script. A call that changes the kind and leaves permissions unset is checked too.
|
|
62
|
+
- New exports from the root entry: `execHostShell`, `ExecHostShellProgress`, `CommandShell`, `CommandShellProbe`, `findCommandShell`, `hostCommandShell`, `hostShellSpawn`, `withoutBashStartup` — the shell-resolution and spawn machinery the `bash` tool already used internally, now available to a host that runs a command line outside a live tool call.
|
|
63
|
+
- `findCommandShellForDialect` and `installedCommandShellForDialect` select an installed bash or sh explicitly, returning no shell if the requested interpreter is absent. `execHostShell` accepts that `CommandShell` through its optional `shell` option, keeping its existing process supervision and output capture for hosts that choose an interpreter themselves.
|
|
64
|
+
- **Security fix, in the scheduled-run floor, so a live `bash` call benefits too:** `trap 'ACTION' SIGNAL` now reads `ACTION` as a command line (it runs when the signal fires); `find … -exec|-execdir|-ok|-okdir … ;|+` is refused when its clause contains `{}` and the search root is unknown or could reach NAMZU_HOME, a `-name`/`-path` filter could match NAMZU_HOME's own name, or the clause's own program can stop or remove a service; running a file the SAME command line wrote earlier (`>`, `>>`, `tee`, `cp`/`mv`, then `./x`, `sh x`, `bash x`, `. x`, `source x`) is refused, since its content was never confirmed.
|
|
65
|
+
- **`lexShellCommandLine` now reads `$(…)` and backtick bodies as nested command lines** (the same treatment `bash -c` payloads already got) instead of calling the whole containing line opaque outright: each command inside a substitution is in `ShellLexResult.commands`, marked `origin: 'substitution'`, checked by the scheduled-run floor and by every `deny` rule exactly like the rest of the line. Two shapes stay opaque because what actually runs cannot be pinned down from the text alone, measured against real bash (5.2.21 and 5.3.15): a command substitution beside brace expansion in the same word (`$(cmd){a,b}` runs `cmd` once per alternative, not once) and an unquoted substitution used as a `<`/`>` redirection target (a `${var:-…}`-style default value can run its substitution twice when the target turns out ambiguous) — both are refused the same way an unmodeled construct always was. `${ list; }` (bash 5.3's brace-form substitution) and process substitution (`<(…)`, `>(…)`) are unchanged, still opaque; out of this scope. `ShellWord` gains `substitutes: boolean` (additive). Proven differentially against a real bash on the host (`command_not_found_handle` argv capture, disabled builtins, empty `PATH`): every existing corpus line, the exhaustive up-to-three-token sweep, a 6,000-line seeded random sample (manually run to 150,000 with zero mismatches before landing), and new hand-written cases for nesting, mixed quoting, and the two constructs above. `decomposeCommandLine`'s `segments`/`opaque` change the same way for these two constructs.
|
|
66
|
+
- **Security fix, follow-up to the above:** a second review found that the scheduled-run floor's own `pkill`/`killall`/`systemctl`/`launchctl`/`schtasks`/D-Bus detection read a wild (expanding) program-name word as NOT being the tool in question unless its raw, unevaluated text happened to contain the tool's name as one contiguous run — `$(echo pk)ill -f node` and `$(echo system)ctl stop $(echo namzu)-scheduler.service` both passed. A wild word in the program-name position (the head, or right after a `sudo`/`env`-style one-level prefix) now counts as being every tool this checks for; a wild word elsewhere (an ordinary argument, a path) is unaffected. `find ~/.namzu -exec rm {} \;`-style commands are unaffected: the fix is scoped to the program-name position, not to "any wild word anywhere."
|
|
67
|
+
- **Security fix, live tool authorization behaviour changes (`packages/sdk/src/runtime/query/`):** the same review found that a `deny` rule written against a command's real name (`"git push*": deny`) never matches one produced by a substitution or a variable (`$(echo git) push`, `` `echo rm` -rf x``, `$X push`, or a variable in the program's path, `"$HOME"/bin/tool`), since the name never appears as such anywhere in the call's text — for the scheduled-run floor's script check (`verifyScheduledScript`, which now refuses such a command outright, naming it, the same posture as an opaque script) and, pre-existing and more broadly, for the live `bash` tool's own permission rules. **An operator now sees more review prompts:** a bash call whose lexed command's own program-name word expands is escalated (`ToolCallSummary.escalation.unknownProgram`, new field) the same unconditional way a sandbox escape already is — never approved by an allow rule, a remembered grant, `accept-edits` or `auto`/an unattended turn, asked about in every mode that gets that far (naming why: "the program this runs is decided at runtime: `<text>`"), and refused outright with no prompt (`UNKNOWN_PROGRAM_UNATTENDED_REFUSAL`, new export) — there is no unattended opt-in for this the way `unattendedSandboxEscape: 'allow'` is for a sandbox escape. A `deny` rule that matches some other way (the tool's own name, an unrelated word) still refuses the call outright, unaffected. Argument-level expansion alone does not trigger this — only the program-name word — so an ordinary `echo $(date)` is unaffected. New audit actions `unknown_program` (`approved`/`refused`). See `docs/sdk/escalations.md`.
|
|
68
|
+
- **Security fix, follow-up to both of the above — one extra word defeated every one of them:** a third review found that all three consumers (the escalation, the scheduled-run floor, `verifyScheduledScript`) only ever looked at a command's literal head word, so `env $(echo git) push`, `command $(echo git) push`, `exec $(echo git) push`, and a re-exec wrapper with a mandatory argument of its own before the program (`timeout 5 …`, `stdbuf -oL …`, `chrt 0 …`, `env VAR=value …`) all ran, or passed the script check, unverified — the floor's one-hop `REEXEC_PREFIX` guess put the program at the wrong word entirely for these, and `exec`/`command`/`taskset`/`time`/`builtin`/`xargs` were not in it at all. `eval`/`source`/`.` were not read as unverifiable either, though `command-line.ts` already treats them as opaque for `deny`/`allow` rules. **New SDK module, `packages/sdk/src/authorization/program.ts`, exported from `@namzu/sdk`:** `programPositions(command, dialect)` is the one place "where does this command actually exec a program" now lives — it unwraps a chain of re-exec wrappers (`sudo`, `doas`, `pkexec`, `env`, `nice`, `ionice`, `nohup`, `setsid`, `timeout`, `stdbuf`, `chrt`, `taskset`, `time`, `command`, `builtin`, `exec`, `xargs`) with each one's own real option grammar, failing closed on an option it does not recognise, and reads each `find -exec`-family clause as its own position. `source path`/`. path` and `eval word…` are read the same way a nested `bash -c '<literal>'` payload already is: a LITERAL path is known, exactly as `bash path` is (nothing inspects either one's contents), and a LITERAL `eval` payload is joined and lexed as a command line of its own, with every position inside it folded in (so `eval 'env $(echo git) push'` is still unknown, transitively); an expanding path/argument word, or a payload that does not read cleanly, is unknown. `hasPoisoningPrefix`/`poisonsLaterCommands`/`poisonedProgramPosition`/`resolveScriptPrograms`/`DYNAMIC_RESOLUTION_VARIABLES` cover the other half: a command-prefix or `export`/bare assignment to `PATH`, `LD_PRELOAD`, `LD_LIBRARY_PATH`, `BASH_ENV`, `ENV` or `IFS` makes a LATER literal program name unverifiable too. **Behaviour change:** `unknownProgram`'s text is now the command plus why its program's position could not be resolved, not a bare word — a host that pattern-matches the exact string should match on substring/prefix, not equality. `ToolCallEscalation.unknownProgram`'s doc comment, `escalationReason`'s wording, and the floor's own `heads`/`REEXEC_PREFIX` mechanism (removed, replaced by `programPositions`) all changed accordingly. `packages/sdk/src/authorization/__tests__/program.test.ts` (new). See `docs/sdk/escalations.md`, `docs/sdk/command-lines.md`, `docs/cli/scheduled-tasks.md`.
|
|
69
|
+
- **Security fix, two CRITICAL bugs from a final review of `program.ts`, plus wrapper coverage:** (1) `sudo`'s own case never skipped a `NAME=value` pair the way `env`'s already did, so `sudo VAR=value $(echo ls)` reported the ASSIGNMENT WORD as the program and never examined the real, wild one after it; both `sudo`'s and `env`'s assignment-skipping now also feed a poisoning name (`PATH=`, `LD_PRELOAD=`, …) into the poisoning check instead of silently passing it through. (2) `executor.ts`'s `unknownProgramOf` ignored the reading's own `opaque`/`!complete` flag, so `bash -c "$X"` — whose payload the lexer cannot follow at all — was never escalated; a new exported function, `unknownProgramInLine(value, dialect)`, checks that first, and `executor.ts` now delegates to it rather than re-implementing the walk. New wrapper coverage: `busybox`/`toybox` (transparent applet dispatch); `declare -x`/`typeset -x` (any option cluster with it, never `+x`) now poison like `export`, and `export -n` correctly does not; `su -c`/`runuser -c`/`script -c`/`flock -c` read a literal payload recursively like a nested `bash -c '<literal>'` (`su`/`runuser` with no `-c` are unknown: an interactive login shell); `flock`'s positional form; `unshare`, `nsenter`, `chroot` (unknown with no command), `setpriv`, `prlimit`, `numactl`; `watch` (default payload joined and lexed like `eval`'s, `-x` execs the argv directly); `sudo -i`/`-s`/`-e`/`--login`/`--shell`/`--edit` and bare `sudoedit` are explicit unknowns. `ssh host cmd` is deliberately not modelled (the command is remote); `docs/sdk/escalations.md` now documents this residual boundary — the full modelled-wrapper list, and that an unmodelled one is treated as the program itself, with `deny` rules, the floor and ordinary review still applying to it. `packages/sdk/src/authorization/program.ts`, `runtime/query/executor.ts`, `packages/sdk/src/authorization/__tests__/program.test.ts`.
|
|
70
|
+
|
|
71
|
+
**`@namzu/cli` (major)**
|
|
72
|
+
|
|
73
|
+
- A pure `script` job now creates and runs without a configured provider or `--model`. New script jobs omit the unused `model` field; older confirmed script jobs that already store a real model still read with their confirmation digest intact and keep it on a script edit. Explicit `--model`, `--effort`, prompt, agent budget, provider wait, approval TTL, session retention and browser grant inputs are refused for a pure script instead of silently ignored; older stored fields remain readable but inert. `agent` and `script+agent` still require a model; converting a model-free script to either kind requires one. Text previews, the SDK `ScheduleJobPreview.budget`, and `schedule show` report zero model tokens and the script timeout for pure scripts. The schedule tool's success message reports pure script history without claiming a session.
|
|
74
|
+
- **Breaking, on disk:** the job/run-result/history file format moves to `v: 2` for a job that actually uses `runKind`/`script`, a `check-failed` status, or `gateResult`/`scriptOutput`. An older namzu that reads such a file refuses it loudly ("was written by a newer namzu...") rather than misreading a script job as a malformed agent job. A plain `agent` job stays `v: 1`, byte-for-byte as before, and every existing job keeps reading and confirming exactly as it did — downgrade only affects a job you have actually converted to `script`/`script+agent`.
|
|
75
|
+
- **Fix:** a job file an older namzu (or a hand edit) left in a format the current one refuses to fully parse was previously invisible to `createJob`'s name-uniqueness check and to `findJob` (by name or id prefix) — a name collision with it went undetected, and looking it up said "No scheduled job is named…" instead of the real reason. Both now read a refused file's `id`/`name` leniently, off the raw JSON, without validating anything else about it: a name collision with such a file is now refused, naming the file and the reason; looking it up by its own name or a matching id prefix now reports the real reason (e.g. "written by a newer namzu…"); a name or id prefix matching more than one job — readable, unreadable, or a mix — is an error listing every match and how to address each. `schedule list` says `N job files could not be read` instead of `No scheduled jobs` when every job file is such a file.
|
|
76
|
+
- New `schedule add`/`edit` flags: `--kind agent|script|script+agent`, `--script`/`--script-file`, `--shell bash|sh`, `--script-timeout`. A `script` job needs `--shell` and a non-empty script and has no `--prompt`; `--unmatched park` is refused for a pure `script` job (nothing can wait for the operator mid-script); `--execution sandbox` is refused for `script`/`script+agent` in this release; `script`/`script+agent` job creation is refused outright on native (non-WSL) Windows, where the only available shell reading is a loose `cmd.exe` approximation the floor cannot fully verify.
|
|
77
|
+
- **Fix:** `--shell sh` previously only labeled a script for verification while fire always used the live bash tool's automatically chosen shell. On a bash host, a newly confirmed `sh` job was guaranteed to stop at first fire. A script now executes under its chosen installed `bash` or `sh`; creation and every confirmation refuse an absent interpreter, and fire blocks if it has since disappeared. Native Windows remains refused.
|
|
78
|
+
- `ScheduleRunStatus` gains `check-failed` (a script or wake-gate malfunctioned: non-zero exit, its own timeout, or stdout that is not the wake-gate's JSON contract) — CLI-internal, not part of the SDK's public surface. It joins the failure bucket for the failure streak and automatic pause, and gets its own notification and `noticeText` case.
|
|
79
|
+
- `schedule show`/`history --json` gain `gateResult` (`{ wake, contextChars }`) and `scriptOutput` (`{ stdout, stderr }`, capped with an explicit truncation marker) on a fired `script`/`script+agent` run; `show`'s text view gains a "Script"/"Wake-gate script" section beside "Prompt".
|
|
80
|
+
- `schedule list` (text and `--json`) now shows each job's kind: `[script, 0 tokens]` and `[script+agent]` in the text view, `kind` (absent for `agent`) in `--json`. The TUI's `/schedule` list marks the same two kinds. Found missing in a UX review — nothing distinguished a zero-token `script` job from an `agent` job in either listing.
|
|
81
|
+
- **Security fix:** a `script`/`script+agent` run's own output — a wake-gate parse failure's quoted stdout, a script-check refusal's quoted command, a `check-failed` notification's reason, `scriptSummaryOf`'s extracted summary line, and `schedule show`/`history`'s text-view `reason`/`summary` — is untrusted runtime text, not anything an operator confirmed, and is now run through the same `sanitizeLine()` a model's own answer already is before it reaches a terminal or a desktop notification, so it cannot plant control characters, terminal escape sequences or invisible/bidi characters on the operator's screen. `--json` output is unaffected and stays raw.
|
|
82
|
+
- **Security fix:** CLI and TUI schedule add/edit/confirm/list/show text views now show terminal controls and bidi characters in script text, prompts, changed lines and folder paths as visible code point escapes, so a script cannot erase or reorder its own confirmation. The model's schedule-tool confirmation uses the same projection. Persisted jobs and JSON output preserve the original source text.
|
|
83
|
+
- **Security/UX fix:** the TUI's `schedule` tool host now acts on an `update`'s `runKind`/`script`: previously it always carried a job's existing kind and script forward unchanged, silently, whatever the model asked to change. Moving to `kind: 'agent'` drops the script (an agent job cannot carry one); a change that touches `kind` or `script` shows up in "Changed since it was last confirmed" the same as every other field, alongside the new script's full text in its own section (`packages/cli/src/tui/schedule/tool-host.ts` `updateRequest`, `packages/cli/src/schedule/changes.ts` `confirmationView`).
|
|
84
|
+
- **Security fix:** `namzu schedule __fire` now refuses a `script`/`script+agent` run outright on native (non-WSL) Windows, the same way job creation already does. Job creation alone was not enough: a job confirmed on WSL or Linux, whose daemon later finds itself running on native Windows (a moved `NAMZU_HOME`, a machine re-imaged from WSL to a native install), previously ran anyway whenever the job's confirmed shell dialect happened to still read as `sh` — the label native Windows's `cmd.exe` approximation and POSIX sh both report, so the existing dialect-mismatch check alone could not tell them apart.
|
|
85
|
+
- A `script`/`script+agent` script (and the scheduled-run floor generally, so a live `bash` call benefits too) may now use `$(…)`/backtick command substitution: it is checked the same as the rest of the script instead of refusing the whole thing outright (see the SDK's `lexShellCommandLine` entry above). The `schedule-task` skill's guidance on this is updated to match; the two narrow shapes still refused (a substitution beside brace expansion, or unquoted as a `<`/`>` redirection target) are named there with why.
|
|
86
|
+
|
|
87
|
+
Upgrade step: nothing changes for an existing `agent` job. Adopting a `script`/`script+agent` job on a `NAMZU_HOME` you may downgrade the CLI on later means that job's files become unreadable by the older version until you upgrade again; keep that in mind before converting a job you might need to roll back.
|
|
88
|
+
|
|
89
|
+
- 567ada8: Continuing a session with a different project, tenant or topic now fails with `invalid_config` before its history, budget or queued topic messages are used. A supplied session log must also name the requested session. Pass the complete identity returned by the first `runAgent` call on later turns, or use the same recorded scope with `query` and `QueryAgent`. `resumeSession`, `loadTurnState`, `loadSelectedTurnState` and `prepareForkState` verify the source log before reading checkpoints. `PrepareForkInput.scope.topicId` is now required: change `scope: checkpointScope` to `scope: { ...checkpointScope, topicId: sourceTopicId }`, using the source session's recorded topic.
|
|
90
|
+
|
|
91
|
+
Older session logs whose `session_started` record lacks `tenantId` or `topicId` remain readable but can no longer be continued, resumed or forked because their owner cannot be established. To migrate, verify the owner outside the log, start a new session under that scope and seed it with trusted conversation messages. Keep the old hash-chained log unchanged.
|
|
92
|
+
|
|
93
|
+
Custom `SessionLog.claim` implementations must honor the new `repairTornTail: false` option used during owner admission. Defer torn-tail truncation and repair records until a later authorized append when that option is false.
|
|
94
|
+
The session-log conformance contract is now version 2; update a custom backend's declared `contractVersion` after it passes the new deferred-repair case.
|
|
95
|
+
|
|
96
|
+
- 567ada8: `ToolMessage.revealedTools` now stores `{ name, sourceId, sourceKind }` receipts instead of name strings. Code that constructs these messages directly must include the source of each revealed tool; old name-only receipts in stored sessions no longer load deferred tools and must be rediscovered. A failed tool call no longer reveals tools. Hosts whose tools need a live connection should wrap their toolset with `readyWhen(toolset, () => connectionIsReady)`; `deferred(...)` only delays schema loading and `search_tools` can load it whenever the host reports ready. A caller-provided `search_tools` must be active and ready when deferred tools are present, and a runtime override cannot defer or suspend it.
|
|
97
|
+
- 567ada8: `ToolRegistry` is removed. The runtime now resolves tools from `Toolset`s
|
|
98
|
+
(plan.md v3 §2) through a `ToolManager` (`toolsets/manager.ts`) that
|
|
99
|
+
`query()`/`drainQuery` builds for itself, once per turn,
|
|
100
|
+
from the `toolsets` you pass — it is never mutated by the runtime, and its
|
|
101
|
+
own generated tools (task tools, `search_tools`, the structured-output tool,
|
|
102
|
+
advisory tools) are combined in as a `runtime` toolset rather than injected
|
|
103
|
+
into your input.
|
|
104
|
+
|
|
105
|
+
**What breaks, and what to do about it:**
|
|
106
|
+
|
|
107
|
+
| Removed / changed | Replace with |
|
|
108
|
+
| ---------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------- |
|
|
109
|
+
| `tools: ToolRegistryContract` on `QueryParams`, `ReactiveAgentConfig`, `RunAgentOptions`, `BidiTurnParams` | `toolsets: readonly Toolset[]`. Wrap a plain list with `toolset('name', [...])`; `deferred(toolset(...))` for a toolset whose tools start deferred. |
|
|
110
|
+
| `tools?: ToolRegistryContract` on `SupervisorAgentConfig` | `toolsets?: readonly Toolset[]` (same shape). |
|
|
111
|
+
| `ToolRegistry` class, `ToolRegistryContract`, `ToolRegistryRef`, `ToolRegistryForkOptions` | `ToolManager` (`new ToolManager({ toolsets, messages: () => turnMessages, resultGuardrails?, tierConfig? })`), exported as an advanced API. `ToolContext.toolRegistry` is now a `ToolsView` (`has`/`availability`/`searchDeferred` — no `activate`, no `searchActive`). |
|
|
112
|
+
| `registry.fork({ deferExcept })` | No replacement: build the `toolsets` array you want for that turn/send instead of forking a shared registry. There is no live shared registry to fork from any more. |
|
|
113
|
+
| `registry.activate(names)` / `.defer(names)` / `.suspendAll()` / `.hasSuspended()` / `.assignTiers()` | Gone. `.defer`/`.suspendAll`/`.hasSuspended`/`.assignTiers` had no production caller; `.activate()` did (`runtime/query/executor.ts`'s reveal handling, `tools/builtins/search-tools.ts`'s `search_tools`), and both are now the same derived mechanism. Availability is now DERIVED, never mutated: a tool is `'active'` unless its toolset declared `'deferred'` and no tool message in the turn's post-compaction history has revealed it (`ToolResult.reveals`, persisted as `ToolMessage.revealedTools`). A tool's own result reveals names for the rest of the turn; nothing calls `activate` any more, including `search_tools`. |
|
|
114
|
+
| `registry.searchActive(query)` | Gone. `ToolManager.searchDeferred(query, limit?)` remains, with an optional result cap. |
|
|
115
|
+
| `ToolDefinition.provenance` / `ToolProvenance` on a tool | `ToolManager.sourceOf(name)` — a lean `ToolSourceRef` (`id`, `kind`, and for `mcp_server`, `server` + `readOnlyHintTrusted`) that names the OWNING TOOLSET, not the tool. A definition can no longer claim its own source. `tools/trusted-read-only.ts`'s `isTrustedReadOnly` now takes that source as a third argument. `ToolProvenance` itself stays, as the shape `screenToolResult` reads. |
|
|
116
|
+
| `filterReadOnlyTools(registry)` / `filterToolsNamed(registry, names)` (`tools/roster.ts`) | `filtered(toolset, (tool, source) => isTrustedReadOnly(tool, undefined, source))` / `filtered(toolset, names)` (`toolsets/wrappers.ts`) — keeps the inner toolset's own `availability` and stays live over a live source, instead of freezing an always-`'active'` snapshot. `source` is the ONE toolset `filtered` runs on: filter each of a wider roster's contributing toolsets this way before combining them, never a toolset `combineToolsets` already merged from several sources (`tools/roster.ts` says why). |
|
|
117
|
+
| `ConnectorToolRouter` class (`registerTools`/`unregisterTools`/`refreshTools` mutating a registry) | `connectorTools(manager, { strategy? })` (`connector/tools/router.js`) — a plain function returning `ToolDefinition[]`; wrap it in `toolset(...)` yourself. Never had a production caller. |
|
|
118
|
+
| `mcp_<server>_<tool>` naming from the CLI's own MCP path | `mcp__<server>__<tool>`; update saved prompts, permissions and integrations that name these tools. The CLI now mounts `mcpToolset` entries for each server. `mcpToolToToolDefinition` no longer sets `.provenance`; source identity and read-only trust belong to the owning toolset. |
|
|
119
|
+
| `PluginLifecycleManagerConfig.toolRegistry` | Removed. `PluginLifecycleManager` owns its tool contributions and exposes `.toolsets: readonly Toolset[]` for a host to fold into its own `toolsets` array. Each plugin has its own live file-tool source and each plugin MCP server has its own source; see the `plugin-source-toolsets` changeset. |
|
|
120
|
+
| `PluginResolver`'s second constructor argument | Was `ToolRegistryContract`; now `Pick<ToolManager, 'listNames' | 'has'>`. |
|
|
121
|
+
| `ToolRegistryConfig`, `ToolCatalog` and companions | `ToolManagerConfig` (`toolsets/manager.ts`) is the manager's construction config. See the `remove-tool-catalog` changeset for the catalog removal. |
|
|
122
|
+
|
|
123
|
+
The CLI, AG-UI adapter and other affected workspace packages migrate in this
|
|
124
|
+
same release; see their package changesets. `@namzu/files` and `@namzu/lsp`
|
|
125
|
+
do not use the removed registry API.
|
|
126
|
+
|
|
127
|
+
A caller toolset that contributes a name `query()` also generates internally
|
|
128
|
+
(a task-tool name, `search_tools`, the structured-output tool's name, or an
|
|
129
|
+
enabled advisory tool's name) is refused at construction with
|
|
130
|
+
`ToolsetConflictError`, naming both sources — this generalises
|
|
131
|
+
`SupervisorAgent`'s old hand-written refusal of a caller tool shadowing one
|
|
132
|
+
of its six coordinator names, which is now the same mechanism (its
|
|
133
|
+
coordinator tools are just another toolset).
|
|
134
|
+
|
|
135
|
+
`search_tools` no longer activates anything: it returns `reveals` on its own
|
|
136
|
+
result, like any other tool now can. Its receipt for "no deferred match"
|
|
137
|
+
no longer echoes already-active matching tools, since the tool-body-facing
|
|
138
|
+
`ToolsView` has no active-tool search — only `searchDeferred`.
|
|
139
|
+
|
|
140
|
+
**Security-relevant, additive:** `ToolCallContext` (`authorization/gate.ts`)
|
|
141
|
+
gains an optional `toolSource?: ToolSourceRef`, and `AuthorizationGate`'s
|
|
142
|
+
`allow_read_only` rule now reads it the same way `isTrustedReadOnly` always
|
|
143
|
+
has. Without `ToolDefinition.provenance` travelling with the tool object,
|
|
144
|
+
the live gate had no way to tell an operator-trusted MCP server's own
|
|
145
|
+
`readOnlyHint: true` from an untrusted one's — both call sites in
|
|
146
|
+
`runtime/query/iteration/phases/tool-review.ts` and the two in
|
|
147
|
+
`runtime/query/executor.ts` (nested tool calls) now pass
|
|
148
|
+
`ToolManager.sourceOf(name)`. A host that builds its own `AuthorizationGate`
|
|
149
|
+
calls and does not thread `toolSource` through gets the same "no untrusted
|
|
150
|
+
party known" default as before (host-defined tools were never affected),
|
|
151
|
+
but should pass it wherever it has a `ToolManager` to ask. Separately,
|
|
152
|
+
`ToolPredicate` (`toolsets/types.ts`) widens to
|
|
153
|
+
`(tool, source: ToolSourceRef) => boolean`; `filtered`/`requireApproval` now
|
|
154
|
+
supply the toolset's own source as the second argument, existing single-arg
|
|
155
|
+
predicates are unaffected. See the `filterReadOnlyTools` row above for the
|
|
156
|
+
one-source-at-a-time rule this predicate needs to stay correct across a
|
|
157
|
+
combined, multi-source roster.
|
|
158
|
+
|
|
159
|
+
### Minor Changes
|
|
160
|
+
|
|
161
|
+
- 567ada8: Releases live toolset change listeners when a turn or duplex session ends, including abandoned streams and failed duplex connections, so repeated runs using one MCP toolset do not retain earlier tool managers. `ToolManager.dispose()` unsubscribes without closing caller-owned toolsets; direct `ToolManager` users should call it when finished. Agent front doors now check every exposed config field against its runtime destination, and the deprecated `runAgent.verificationGate` option again applies the authorization policy. Supplying distinct policies under both gate names fails before the model runs.
|
|
162
|
+
- 567ada8: Managed hosts can now pass tenant, project, topic and session attribution per invocation through `ManagedAgentInput.managedScope` when calling `QueryAgent`. `AgentManager` supplies the admitted child scope itself, and `QueryAgent` accepts that scope while continuing to accept the four flat config fields. If both are supplied with different IDs, the run fails before a model call. The general `AgentInput` and other agent implementations keep their existing contracts. Idempotent `QueryAgent` calls are now deduplicated only within the same managed scope.
|
|
163
|
+
- 567ada8: Authorization rules can now match the owning toolset's source id with `{ type: 'by_source', sources: ['mcp:github'], decision: 'review' }`. This lets a host ask, allow or deny every tool from one or more MCP servers without listing names individually. A rule with no matching host-supplied source leaves the call to later rules.
|
|
164
|
+
|
|
165
|
+
The CLI accepts `[permissions.sources]` with source ids or globs mapped to `allow`, `ask` or `deny`, and its permissions display shows those rules and their decisions.
|
|
166
|
+
|
|
167
|
+
- 567ada8: Added `findUndescribedProperties(schema)`, exported alongside `findPortableSchemaViolations` from the package root. It walks a rendered tool schema and returns every named object property, at any depth, whose schema carries no non-empty `description` — the same non-throwing, test-time sweep shape as the portability check, for a different defect: a Zod field with no `.describe()` reaches the model with no account of what it is for. Purely additive; nothing existing changes behavior.
|
|
168
|
+
- 567ada8: Add `defineCapability` and `dynamicCapability` for reusable host-authored behavior passed to `runAgent({ capabilities })`. A capability can contribute toolsets, instructions, prompt contributions, input and output guardrails, and per-run model settings. Existing `runAgent` calls need no migration.
|
|
169
|
+
- 567ada8: Two additive changes to the MCP connector, both opt-in and default-preserving:
|
|
170
|
+
|
|
171
|
+
- `MCPInitializeResult.instructions` and `MCPClientState.serverInstructions` capture the server's `initialize` instructions string (legacy handshake only — the modern era has no `initialize` round trip and never populates this field). This is observability only: nothing folds it into an agent's instruction set automatically, matching namzu's existing rule that server-authored text never reaches instruction/system position unframed. A host that wants to display or log a server's instructions can read `client.getState().serverInstructions`.
|
|
172
|
+
- `mcpToolToToolDefinition` takes a new optional 5th parameter, `maxRetries?: number`. Left unset, behaviour is unchanged: no `maxRetries` on the returned `ToolDefinition`. When set, the adapter also now marks `ToolResult.retryable: true` on the two failures it already classifies as not having reached the server's side effect (`mcp_tool_input_required`, `mcp_tool_missing_client_capability`) — the field the executor's retry gate actually reads. An HTTP-redirect outcome-unknown failure and any raw transport error stay non-retryable, since whether either one reached the server is exactly what is not known.
|
|
173
|
+
|
|
174
|
+
Nothing to change for an existing caller: both are new optional fields/parameters with no effect until set.
|
|
175
|
+
|
|
176
|
+
- 567ada8: `MCPToolDiscoveryOptions` and `MCPToolsetOptions` gain `onRefused`. The callback reports the current policy refusals for each tool, prompt and resource listing, including an empty list when a later listing clears them. Hosts can show which server names were excluded and why.
|
|
177
|
+
- 567ada8: New `mcpToolset(client, options)` (`packages/sdk/src/connector/mcp/`): the path from a connected `MCPClient` to two live toolsets mounted together — tools and prompts under the configured availability, plus resources as two always-`deferred` tools (`mcp__<server>__list_resources`, `mcp__<server>__read_resource`, admitting only a `uri` the server has actually listed). `options.allow`/`options.deny` govern all three surfaces — tools, prompts and resources — matched on the server's own name for each. Every name is `mcp__<server>__<rest>` (tools), `mcp__<server>__prompt__<rest>` (prompts), shortened deterministically rather than refused when it would exceed the wire's 64-character limit. Reacts to `list_changed` notifications (only for a capability the server actually advertised) and to reconnection by re-discovering and reporting through `onChange` — including whether the server supports resources at all, re-checked on every reconnect rather than fixed at construction. Owns and stops an `MCPReconnectSupervisor` (`reconnect: { enabled: false }` opts out), and releases its `onNotification` subscription on `close()`. Discovery goes through the existing `MCPToolDiscovery` allow/deny policy and rug-pull drift detection, now including a `discoverResourcesFrom` alongside the existing `discoverFrom`/`discoverPromptsFrom`. Also new: `mcpToolsetName(serverName, ...segments)` and the `MCPToolsetOptions` type. See [docs/sdk/mcp-toolset.md](../docs/sdk/mcp-toolset.md).
|
|
178
|
+
|
|
179
|
+
`MCPClient.onNotification` now returns an unsubscribe function, mirroring `onLifecycle` — additive; any existing caller ignoring the return value is unaffected.
|
|
180
|
+
|
|
181
|
+
Purely additive: no existing export changed or was removed. `mcpToolToToolDefinition`/`mcpPromptToToolDefinition` (the pieces `mcpToolset` wraps) remain exported for direct SDK callers. The CLI and plugin lifecycle now use `mcpToolset`, with their naming changes covered by separate changesets. A caller building an MCP server's tools by hand can move to `mcpToolset` at its own pace.
|
|
182
|
+
|
|
183
|
+
- 567ada8: Plugin manifests can declare `instructions`. An enabled plugin contributes the text as labelled, untrusted request context; disabling it revokes the contribution, including for an already-running turn. SDK hosts can also call `definePlugin({ name, tools, hooks, instructions, mcpServers })` and `PluginLifecycleManager.installDefined(plugin)` to install a host-authored plugin without a manifest file or dynamic module import.
|
|
184
|
+
- 567ada8: `PromptContributionRegistry.unregister(id)` removes a prompt contribution and returns whether it existed. Hosts can revoke instructions when a plugin or other contribution owner is disabled, then register the id again when it is re-enabled.
|
|
185
|
+
- 567ada8: `runAgent` accepts a caller-owned `pluginManager` and mounts its toolsets, enabled plugin instructions, and hooks for the invocation. Hosts using the high-level agent entry point can now pass a configured `PluginLifecycleManager` without manually reassembling its contributions. The host still owns plugin admission, enablement, and cleanup.
|
|
186
|
+
- 567ada8: Added `ToolManager` (`packages/sdk/src/toolsets/manager.ts`), the runtime-owned resolver of `Toolset`s that plan.md v3 §2 describes: `new ToolManager({ toolsets, resultGuardrails?, tierConfig?, messages })` resolves toolsets in order, then tools in order, throwing `ToolsetConflictError` on a name conflict at construction. `refresh()` adopts a live toolset's change only when called, reporting `added`/`removed`/`drifted`/`refused` names. `availability(name)` is derived — `'active'` or `'deferred'`, never stored — from the owning toolset's own declaration and whether a tool message in the turn's post-compaction history has revealed the name (`ToolMessage.revealedTools`, also added, the persisted form of `ToolResult.reveals`). `sourceOf(name)`, `toLLMTools`/`toPromptSection`/`toTierGuidance`/`searchDeferred(query, limit?)`, a narrow `view()` for `ToolContext`, and the `prepareExecution`/`executePrepared`/`execute` pipeline (copied from `ToolRegistry`'s, reading availability and source from the derivation instead of a stored map and `ToolDefinition.provenance`) round it out.
|
|
187
|
+
|
|
188
|
+
`ToolManager` is exported as an advanced API. `query()` and the other runtime
|
|
189
|
+
entry points use it in this release; see the `tool-registry-removed` changeset
|
|
190
|
+
for the breaking input and export migration.
|
|
191
|
+
|
|
192
|
+
- 567ada8: **A tool can now declare that it always needs a person's approval, and carry free-form metadata a host can filter on.** Both are additive; nothing existing changes behaviour.
|
|
193
|
+
|
|
194
|
+
- `ToolDefinition.requiresApproval?(input): boolean` and `defineTool({ requiresApproval })` (a boolean or a function of the input) mark a call as needing a person's approval no matter what the host's authorization rules say or what mode the turn is running in. It is stronger than `isDestructive`: a gate `allow` rule and `auto`/unattended mode both still let a destructive call through, but neither can let a `requiresApproval` call through — it survives a gate `allow` rule the way an escalated call does, is asked about in every mode that reaches it, refused when nobody can be asked, refused outright under `strict`, and covered by no grant, skill grant or `accept-edits` exemption. A `deny` rule still wins. Reach for this on a call that is sensitive for a reason other than being destructive (it costs money, it leaves its own audit trail, it is policy-sensitive) instead of misdeclaring `isDestructive` to get a review the host might not otherwise configure. See [The review policy](../docs/sdk/review-policy.md#a-call-the-tool-itself-declares-always-needs-approval).
|
|
195
|
+
- An unattended sandbox-escape opt-in confirms only the escape. It no longer waives a tool's `requiresApproval` declaration or another call's outside-root review in the same batch.
|
|
196
|
+
- `ToolDefinition.metadata?: Readonly<Record<string, unknown>>` and `defineTool({ metadata })` carry free-form, tool-author-declared data for filtering and behaviour customization. Never sent to the model and never read by the runtime itself. The new `matchesToolSelector(selector, tool)` matches a tool against a list of names, a partial deep-equal match against `metadata`, or a predicate — for a capability, toolset wrapper or host to target a group of tools without maintaining a parallel name list.
|
|
197
|
+
- `mcpToolToToolDefinition` now carries a connected server's `title`, `idempotentHint`, `openWorldHint` and any tool-listing `_meta` into the new `metadata` field instead of dropping them (`readOnlyHint`/`destructiveHint` keep their existing typed home on `isReadOnly`/`isDestructive`). It never populates `requiresApproval`: a connected server cannot demand, or waive, its own review requirement. `MCPToolDefinition` gained a matching `_meta?: Record<string, unknown>` field for the raw listing value.
|
|
198
|
+
|
|
199
|
+
Nothing to do to keep existing behaviour: every new field is optional and absent unless a tool author sets it.
|
|
200
|
+
|
|
201
|
+
- 567ada8: Added `ToolResult.reveals?: readonly string[]`. A tool's own result can name further deferred tools to make callable — the same mechanism `search_tools` uses. The runtime persists admitted names on the tool message, so discovery survives later sends and resume until compaction. Unknown, already-active and out-of-allowlist names are ignored. There is no suspended state or mutable registry activation. See `docs/sdk/tool-discovery.md` for the migration from `ToolRegistry.activate()`.
|
|
202
|
+
- 567ada8: New `packages/sdk/src/toolsets/` module: `Toolset` (the unit every tool comes from), `toolset(source, tools)`, the composable wrappers `prefixed`, `renamed`, `filtered`, `deferred`, `requireApproval`, `withMetadata` and `mapTools`, and `combineToolsets(source, toolsets)` with atomic name-conflict detection (`ToolsetConflictError`, extending the shared `RegistryCollisionError` — see [docs/sdk/registries.md](../docs/sdk/registries.md) — with `combineToolsets` as its registry name and the tool name as the colliding id; `name`, message and `firstSource`/`secondSource` are unchanged). Also `matchesSourceIdGlob` and `toToolSourceRef`. See [docs/sdk/toolsets.md](../docs/sdk/toolsets.md).
|
|
203
|
+
|
|
204
|
+
`filtered`'s `{ metadata }` selector and `requireApproval` compose with the `ToolDefinition.metadata` and `requiresApproval` fields (see the `tool-requires-approval-and-metadata` changeset). Existing callers must migrate from `ToolRegistry` to `Toolset` as described in the `tool-registry-removed` changeset.
|
|
205
|
+
|
|
206
|
+
### Patch Changes
|
|
207
|
+
|
|
208
|
+
- 567ada8: The AG-UI adapter now reads tools and review metadata from the query's
|
|
209
|
+
`toolsets`, preserving review exemptions and per-tool timeouts after the
|
|
210
|
+
`ToolRegistry` removal. Hosts must return `QueryParams.toolsets` from
|
|
211
|
+
`createQuery` instead of passing a `ToolRegistry` as `tools`. Wrap definitions
|
|
212
|
+
with `toolset(source, definitions)` and require `@namzu/sdk >=48.0.0`; use
|
|
213
|
+
`@namzu/ag-ui` 2.x with SDK 45.1–47. MCP resource tools likewise rely on their
|
|
214
|
+
owning toolset for source trust instead of a removed per-tool provenance field.
|
|
215
|
+
- 567ada8: CLI MCP tool names change from `mcp_<server>_<tool>` to `mcp__<server>__<tool>`. Update any permissions, saved prompts or integrations that refer to those names; the CLI warns once for each configured permission rule still using the old form. MCP tools now update after server notifications and reconnects. Configure `allow`, `deny`, `maxRetries`, `requireApproval`, `readOnlyHintTrusted` and `instructions` per server under `mcpServers`; `instructions: true` includes the server's text as labelled untrusted request context. `/mcp tools` shows the server's current tool names and instructions.
|
|
216
|
+
|
|
217
|
+
The SDK now skips prompt discovery when a server did not advertise prompt support and keeps approval-wrapped tool definitions stable across unchanged toolset snapshots.
|
|
218
|
+
|
|
219
|
+
- 567ada8: Fixed: `combineToolsets` dropped every input toolset's own `availability` (`'active'`/`'deferred'`) — the merged `Toolset` it returned had no `availability` field, which `ToolManager` and everything else that reads it defaults to `'active'`, regardless of whether some of the merged inputs were wrapped with `deferred(...)`. `query()` already avoided this by keeping its own eager and deferred halves as two separate array entries instead of combining them (see its own comment in `runtime/query/index.ts`), but `combineToolsets` itself stayed exported with this defect for any other caller.
|
|
220
|
+
|
|
221
|
+
`combineToolsets` now refuses to merge toolsets that disagree on availability, naming both sources, instead of silently reporting the deferred side's tools as active. Toolsets that agree — all active, or all deferred — combine as before, and the combined toolset now carries `availability: 'deferred'` when every input does. No current caller passes a mixed set (this was latent, not user-visible); a caller with a genuine eager/deferred split keeps them as two separate entries in its own `toolsets` array, the way `query()` does.
|
|
222
|
+
|
|
223
|
+
- 567ada8: `combineToolsets` now releases earlier live-toolset listeners if a later source fails to subscribe. Its unsubscribe also attempts every inner cleanup when one throws, so failed setup and teardown do not leave other sources observing a finished manager.
|
|
224
|
+
- 567ada8: MCP tool definitions that change after initial admission keep serving their earlier definition across turns; new and removed names still update. `/mcp` and `exec --json` now report server discovery changes and current policy refusals, including changed definitions held for review.
|
|
225
|
+
- 567ada8: Former MCP permission-name warnings now use the CLI's structured logger instead of writing plain text to stderr. Consumers of `exec --json` stderr can continue parsing log records when a legacy permission rule is configured. Plugin MCP disconnect warnings use a fixed message and put the operation in a log attribute.
|
|
226
|
+
- 567ada8: Shipped prose no longer names other named products. `matchesToolSelector`'s doc comment (`packages/sdk/src/tools/roster.ts`) and [Tool metadata and selectors](../docs/sdk/tool-metadata-and-selectors.md) describe its synchronous-only design against "some other agent frameworks' equivalent selector" instead of one by name; [Tool servers](../docs/cli/mcp-servers.md) describes the wider `${VAR}`/`${VAR:-default}` interpolation some MCP clients accept as "several desktop and editor MCP clients" instead of naming two. Wording only: no type, default or behaviour changed.
|
|
227
|
+
- 567ada8: Invalid JavaScript plugin tools now fail when a plugin is enabled, with the
|
|
228
|
+
plugin and module named, if they omit a usable `inputSchema`. Before, a
|
|
229
|
+
deferred plugin tool could load successfully and then crash the turn with a
|
|
230
|
+
Zod `_def` error after `search_tools` revealed it. Supply a Zod
|
|
231
|
+
`inputSchema`, or a compatible parser and explicit `modelInputSchema`.
|
|
232
|
+
- 567ada8: **`@namzu/cli`**: the session's tool roster is now composed from named `Toolset`s (`@namzu/sdk`'s toolsets, plan.md v3 §8) instead of a `ToolRegistry`. This is internal composition, not a config or CLI-flag change — nothing in `namzu.config.json` or the command line changes — but it is a breaking change for anything that imported `@namzu/cli`'s internals directly:
|
|
233
|
+
|
|
234
|
+
| Removed / changed | Replace with |
|
|
235
|
+
| ---------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
236
|
+
| `createCliPluginRuntime(config, tools: ToolRegistry, cwd, hooks?, skillTool?)` | `createCliPluginRuntime(config, cwd, hooks?)` — it no longer takes or mutates a tool registry. Read `runtime.skills.size` and fold `runtime.manager.toolsets` into your own toolsets list; the `skill` tool is your own toolset now, not something this runtime registers. |
|
|
237
|
+
| `SubagentRuntimeOptions.buildTools: () => ToolRegistryContract` | `buildTools: () => readonly Toolset[]`. |
|
|
238
|
+
| `SubagentRuntimeOptions.configureWebSearch?: (provider, model, tools: ToolRegistryContract) => WebSearchConfig \| undefined` | `configureWebSearch?: (provider, model, toolsets: readonly Toolset[]) => WebSearchConfig \| undefined` — a pure decision now; it no longer removes `web_search` itself, so a caller building the child's final toolsets filters it out when this returns truthy (`filtered(ts, (t) => t.name !== 'web_search')`). |
|
|
239
|
+
| `McpConnection.tools: readonly ToolDefinition[]` (still present) | `McpConnection.toolsets: readonly Toolset[]`, two per connected server (main tools and deferred resources, both kind `mcp_server`, id `mcp:<server>`) — prefer these; `.tools` is now just the flattened union across all of them. |
|
|
240
|
+
|
|
241
|
+
Additive: `AgentSession` gained a `presenter: ToolPresenter` field — the session's real tool-call/result presenter, for a host that renders events outside `send()`'s own stream. The ACP bridge (`namzu acp`) now uses it instead of a presenter built over a permanently empty registry, so tool-specific `presentCall`/`presentResult` views (a richer diff, a custom label) render over ACP for the first time; every ACP tool view had silently fallen back to the generic label before.
|
|
242
|
+
|
|
243
|
+
Behaviour: because tool availability is now derived from the turn's own revealed-tool history (`@namzu/sdk`'s `ToolManager`) rather than a per-send registry fork, a tool `search_tools` reveals in one send now stays active in a later send of the same session, and after `namzu resume` — it no longer resets every send.
|
|
244
|
+
|
|
245
|
+
**`@namzu/sdk`**: two `combineToolsets` usage bugs, found while wiring the CLI onto toolsets, are fixed — `query()`'s own generated tools and `SupervisorAgent`'s coordinator tools are now two separate array entries (an eager one and a deferred one) rather than one `combineToolsets`'d together, so `runtimeToolOverrides: { name: 'deferred' }` (task tools, coordinator tools) is honoured again instead of silently falling back to `'active'` for both halves. `query()` also no longer adds its own `search_tools` when a caller's `toolsets` already contribute a tool under that exact name (a connector genuinely named `search_tools`), matching the pre-toolsets registry's behaviour and avoiding a spurious `ToolsetConflictError`. Pure bug fixes, no API change.
|
|
246
|
+
|
|
247
|
+
## 47.0.0
|
|
248
|
+
|
|
249
|
+
### Major Changes
|
|
250
|
+
|
|
251
|
+
- 49491b9: **Breaking, and what to do about each:**
|
|
252
|
+
|
|
253
|
+
- **`MockLLMProvider`'s `truncateArguments`.** A scripted call with it now sends only the first half of its arguments, ends the response there (calls scripted after it in the same turn are not streamed), and the turn finishes with `length` unless the script sets `finishReason`. It used to send the whole JSON, go on to later calls, and finish with `tool_calls`, so the call ran. A test that relied on the call running: drop `truncateArguments`. One that scripts calls after it: move them to the next turn. One that relied on `tool_calls`: set `finishReason: 'tool_calls'` on the turn (the cut call is then `malformed`).
|
|
254
|
+
- **A reused tool-call index is refused.** A stream that puts a second call id on an `index` another call holds now fails: the turn loop throws `ProviderRequestError` (`kind: 'server'`, so the turn pauses), and `collectChatCompletion` throws an `Error` naming the violation. Before, the second call's arguments were appended to the first call's and the stream went on. A custom driver that sends parallel calls must give each its own `index`, or leave `index` out entirely: a fragment with no index is placed by its id.
|
|
255
|
+
- **The message for an unreadable tool call.** The tool result a model gets for a call whose arguments did not parse is no longer one fixed string ("…was cut off while the model was streaming JSON arguments… Retry with a much shorter input…"); it depends on the reason and the tool (below). A host or test that matches the old text must match the new one, or read `ToolCall.metadata.inputError` instead.
|
|
256
|
+
- **Event order for a tool call.** `tool_input_delta` and `tool_input_completed` no longer come before the call's `tool_input_started`, and `tool_input_completed` carries the id the call was announced with, not the id on the driver's block close. A host that handled either early event must expect them after the start.
|
|
257
|
+
- **Advisor records.** `argumentsIncomplete: true` now marks only a call the response was cut off inside (see below).
|
|
258
|
+
- **One log record's text.** "Repaired a tool call whose input stream was truncated" is now "Repaired a tool call whose arguments could not be read" (see below).
|
|
259
|
+
|
|
260
|
+
A tool call whose streamed arguments do not parse is now reported as **truncated** or **malformed**, and the model is told which, instead of always hearing that its call was cut off and it should send less.
|
|
261
|
+
|
|
262
|
+
**Which is which.** Only the call the response was streaming when it stopped can be truncated: the last call, with nothing the model streamed after it, when the stream reported an output limit (`length`), a content filter (`content_filter`) or no finish reason at all. Every other call that does not parse is malformed, whatever the finish reason: the response finished normally, or the model went on to more text, reasoning or another call after it, so it had finished writing the call.
|
|
263
|
+
|
|
264
|
+
**What changes for the model.** The error for an unreadable call is built from the reason and from the tool. A malformed call gets the JSON parser's error and the character where the JSON went wrong, found by scanning the arguments, so it is there even for a bare `True` or `None`, which the parser's message names no position for. A truncated call is told what stopped the response (the output limit, a content filter or a dropped stream). After the output limit, a call is not blamed for output it did not produce: when the provider's usage shows that most of the response's output tokens went to reasoning (or to anything else the stream did not carry, such as encrypted or summarised thinking), the model is told so and to reason less or split the work. Otherwise a call that was less than half of what the response streamed is told how much came before it and to send less there, and any other is told to keep its arguments under half of what arrived, always less than what arrived, per declared argument if the tool declares large string arguments. A content filter's stop gets no advice. The fixed text about `content`/`new_string`, a 12000-character budget and write/edit markers is gone from every other tool. If you match on the old text ("call was cut off while the model was streaming JSON arguments"), update the match.
|
|
265
|
+
|
|
266
|
+
**New, all optional:**
|
|
267
|
+
|
|
268
|
+
- `ToolInputError` (`reason: 'truncated' | 'malformed'`, `finishReason`, `parseError`, `offset`, `length`, `precedingLength`, and on a truncated call `outputTokens` and `reasoningTokens` when the provider reported them) on `ToolCall.metadata.inputError`. `precedingLength` counts the characters the response streamed before the call began; output the provider did not stream is not in it.
|
|
269
|
+
- `StreamChunk.finishDetail` and `ChatCompletionResponse.finishDetail`: `'context_window'` with a `'length'` finish that was the model's context window rather than its output token limit. The turn loop no longer auto-continues such a reply, which would send a longer prompt into the window it had just filled; a reply the output limit cut off is still continued. A custom driver that reports a context-window stop as `length` should set it. `ToolInputError.finishDetail` carries it, and a call the window cut off is told so.
|
|
270
|
+
- `StreamChunk.delta.contentOrigin: 'driver'`, for text a driver adds of its own after the model's output, such as a list of sources. The turn loop does not count it as the model moving on from a call. A custom driver that appends text of its own should set it, or a call the output limit cut off before that text is reported as malformed.
|
|
271
|
+
- `inputError` and `partialArguments` (first 16 384 characters) on the `tool_input_completed` event. The SSE bridge sends them as `input_error` and `partial_arguments`, with `input_truncated`.
|
|
272
|
+
- `ToolDefinition.truncatedInputHint`, `ToolDefinition.malformedInputHint` and `ToolDefinition.largeStringArguments` (for example `{ content: 12_000 }`), also accepted by `defineTool`. The truncated hint is appended only when a cut-off call itself has to carry less: the stream ended, or the output limit cut a call that was at least half of the response. Never after a content filter. The malformed hint is appended only to a malformed call; a tool that declares none gets its `validationErrorHint` there instead, so a malformed `ask_user_question` or `approve_plan` call is shown the shape those tools require (`options` or `steps` as a JSON array, never a string). `write`, `edit`, `create_task` and the coordinator `Agent` tool declare large text and a truncated hint; they, and `bash`, also declare a malformed hint that says how to write a newline, tab, quote or backslash inside a JSON string, so a malformed call to them gets no size or file advice. `bash`'s truncated hint says to build a long file with `write` and `edit` instead of a heredoc.
|
|
273
|
+
|
|
274
|
+
`inputTruncated` is still set on every unreadable call, cut off or malformed, as before. Code that reads it keeps working. Read `inputError.reason` to tell the two apart.
|
|
275
|
+
|
|
276
|
+
**Logs.** The info record for a call a `repairToolCall` hook repaired read "Repaired a tool call whose input stream was truncated" for a malformed call too. It now reads "Repaired a tool call whose arguments could not be read" and carries `namzu.runtime.input_error_reason` (`truncated`, `malformed`, or `unknown` for a call recorded without a reason); update a filter that matches the old text.
|
|
277
|
+
|
|
278
|
+
**Advisor records.** The conversation records an SDK advisor is given marked every unreadable call `argumentsIncomplete: true`, a malformed one included. That flag now marks only a call the response was cut off inside; a malformed call carries `argumentsMalformed: true`, and a call recorded without a reason `argumentsUnreadable: true`.
|
|
279
|
+
|
|
280
|
+
**Tool-call framing.** A stream that puts a second call id on a tool-call `index` now fails the turn's model call with a `ProviderRequestError` (`kind: 'server'`, which pauses the turn), and `collectChatCompletion` throws an `Error`. Both name the violation. Before, the second call's arguments were appended to the first's, which left one call no tool could run. A custom driver that sends parallel calls on one index must give each call its own index. A fragment that arrives with no index at all, as some OpenAI-compatible servers send them, is placed by its id instead (an id seen before continues its call, a new one starts the next), so parallel calls from such a server are not refused.
|
|
281
|
+
|
|
282
|
+
A fragment with neither an index nor an id is placed only when exactly one open call could still be receiving more of its arguments right now: its buffer empty, or not yet a complete JSON value. An empty buffer is NOT read as already complete — that is what a call opens with, on the ordinary wire shape, before any of its argument text streams — so two calls opening back to back, both still empty, are both still candidates. This is evaluated fresh for every id-less fragment, not decided once when some call opened, since whether a call can still accept more changes over the stream. When more than one open call could still accept the fragment, or when none could, there is no way to tell which one it was for (or none was); every call that could have is reported unreadable (`reason: 'malformed'`) rather than guessed at, and the fragment itself is attributed to none of them. Guessing used to be able to splice one call's JSON into the other's buffer and, when the splice still happened to parse, run the wrong call with no error at all. The turn is not refused: every such call is still answered, as unreadable, and any other call keeps running. Accepted cost: a zero-argument call immediately followed by another call whose own arguments arrive in id-less fragments, from a server that also omits `index`, now comes out unreadable — the model is asked to send it again — rather than guessed at.
|
|
283
|
+
|
|
284
|
+
Arguments sent before a call's id are kept for the call at their index, by the turn loop as well as by `collectChatCompletion`, which always kept them. The turn loop used to drop them, so the call failed to parse and was reported as cut off. A call whose id never arrives is given one by the turn loop; it used to reach the executor with an empty id. `tool_input_delta` no longer comes before its call's `tool_input_started`: arguments that arrive before the call's id and name are sent in one delta right after it. `tool_input_completed` no longer comes before it either, and always carries the id the call was announced with and runs under. It used to carry the id on the block close (`toolCallEnd.id`), so a custom driver that closed a block with an empty id got a completion before the call, under an id the call did not run with, and `@namzu/ag-ui` failed the run with `NAMZU_TOOL_LIFECYCLE`.
|
|
285
|
+
|
|
286
|
+
**`MockLLMProvider`.** A `truncateArguments` call now does what its documentation said: it sends only the first half of the arguments, the response ends there, and the turn finishes with `length` unless the script sets `finishReason`. It used to send the whole JSON, so nothing was truncated, and then went on to any call scripted after it. A test that relied on that call running must drop `truncateArguments`; one that scripts calls after it must move them to a later turn, since they are no longer sent.
|
|
287
|
+
|
|
288
|
+
See `docs/sdk/unreadable-tool-input.md`.
|
|
289
|
+
|
|
290
|
+
- 7c810bc: **`wrapUntrusted`'s rendered frame changes shape, unconditionally, for every caller.** The opening and closing tags now carry a per-render nonce — `<namzu-untrusted-<nonce> kind="…">` … `</namzu-untrusted-<nonce>>` — instead of the fixed `<namzu-untrusted kind="…">` / `</namzu-untrusted>` literal this has always rendered. A caller that reads `wrapUntrusted`'s output and matches those tags by a literal string, or a fixed-form regex anchored to the exact old spelling (`^<namzu-untrusted `, `</namzu-untrusted>$`), stops matching. Read the body back with the already-exported `untrustedEnvelopeBody(text)` instead — it already understands the nonce and returns the same content it always did — or match the tolerant form `<namzu-untrusted-[0-9a-f]+`. One caller inside this monorepo (`packages/cli/src/integrations/web/search.ts`, a display-only presenter) did exactly the literal match and needed this fix; a consumer of the published package doing the same needs it too.
|
|
291
|
+
|
|
292
|
+
**Why.** The previous defense — matching the bare `namzu-untrusted` keyword, case-insensitively, to stop untrusted content from forging a fake close of its own frame — missed a whole class of same-script Unicode homoglyph: Cyrillic `ѕ`/`е` for Latin `s`/`e` (hundreds of such cross-script pairs exist) are, canonically, different letters that merely look the same, so no amount of Unicode normalization folds one to the other. `<nаmzu-untrusted…>` (Cyrillic `а`) still forged a fake frame. A per-render nonce closes the actual hole: the real boundary is a value nothing in the content could have predicted, in any script, so a lookalike does not need to be recognized as "the keyword" at all — it is just quoted content, whatever it looks like.
|
|
293
|
+
|
|
294
|
+
**The upside for every existing caller: untrusted content is no longer altered to close this hole.** An intermediate attempt (which never shipped on its own, only alongside this same release) folded Unicode confusables — NFKC, dropped zero-width/bidi-control characters, every dash and space mapped to ASCII — across the WHOLE untrusted string, for every piece of text `wrapUntrusted` ever framed. Non-English web pages and computer-use window text were hit hardest: an em dash became `-`, an ideographic space became an ASCII space, fullwidth digits and parentheses became ASCII, a ligature split into its letters. That fold is gone. Content is now emitted byte-for-byte, except a literal, case-insensitive occurrence of the bare `namzu-untrusted` keyword (the same narrow substitution this envelope has always made, kept as cheap defense in depth alongside the nonce).
|
|
295
|
+
|
|
296
|
+
**`wrapUntrusted` also gains an optional third parameter**, `options: { generateNonce?: () => string }`, for a caller that wants a deterministic nonce (tests, mainly); omit it for the default (`crypto.randomBytes`, 16 lowercase hex characters, redrawn automatically if it collides with the content). This addition alone would be a minor change; it ships in the same release as the breaking one above, so the release as a whole is major.
|
|
297
|
+
|
|
298
|
+
### Minor Changes
|
|
299
|
+
|
|
300
|
+
- 2d4ff9a: **`ask_user_question` says which option it recommends in a field, not in the label.** An option takes `recommended: true`, and `UserQuestionOption.recommended` is set on each recommended option of the question your `ResumeHandler` receives; it is absent on the others. The answer's `data.selected[]` carries `recommended: true` on a chosen option the model recommended. The tool no longer tells the model to append " (Recommended)" to a label.
|
|
301
|
+
|
|
302
|
+
Labels now reach your handler without a recommendation marker. An option is recommended when the model set `recommended: true` on it, or left the flag out and ended its label in "(Recommended)". That English marker was already removed from the answer; it is now also removed from the label you are shown, on every option that carries it, and an option the model set `recommended: false` on stays unrecommended. What was left before it still reaches you and the answer whole: "Cloud (AWS) (Recommended)" reads "Cloud (AWS)", as the answer did before. On a recommended option, a trailing parenthesised group of one to three words — "(Önerilen)", "(Empfohlen)", "(推荐)", which used to reach both your screen and the answer the model read back — is taken for the marker and removed from the label you are shown and from the answer. That includes a qualifier the model put there against the tool's instructions: a flagged "Use cache (Redis)" arrives as "Use cache". It is removed only when every option whose label ends in a group is recommended, did not end in "(Recommended)", and ends in that same group (letter case aside, compared the same way whatever your host's locale), so a group that differs between options is kept, flagged or not: "Postgres (managed)" / "Postgres (self-hosted)", or a multi-select's "Tests (unit)" / "Tests (e2e)". A marker that is all that tells two options apart ("Redis (Empfohlen)" next to "Redis") and a group with digits or symbols are kept too, and the order of the options changes none of this. Nothing else changes a label or makes an option recommended: apart from surrounding spaces and a trailing "(Recommended)", an option that is not recommended reaches you, and the answer, exactly as written, and a localised marker on it stays.
|
|
303
|
+
|
|
304
|
+
What to do: if your question UI showed the recommendation only because the label said "(Recommended)", show it from `option.recommended` instead; nothing else changes. In the CLI the question card draws `[recommended]` next to that option.
|
|
305
|
+
|
|
306
|
+
- 166fe10: The coding-agent doctrine's multi-agent text is renamed to match the mode's new name, hypermode. New: `CODING_AGENT_HYPERMODE_DOCTRINE`, and `codingAgentDoctrineContribution({ hypermode: true })`. Deprecated, still working until the next major: `CODING_AGENT_ORCHESTRATE_DOCTRINE` (the same string) and the `orchestrate` option (either flag set to `true` appends the text). Replace both names now; no other change is needed.
|
|
307
|
+
|
|
308
|
+
The text itself now names the mode as the operator sees it: it starts `### Hypermode` and `This session has hypermode on:` instead of `### Orchestrate mode` / `This session has orchestrate mode on:`. The rest of the text is unchanged. If you compare the rendered prompt byte for byte, or cache on it, expect one change when the flag is on; with the flag off the output is byte-identical.
|
|
309
|
+
|
|
310
|
+
- 7c810bc: Cross-session peer messaging, SDK half (`@experimental`): two namzu sessions run by the same OS user can now discover each other and exchange messages over a local socket.
|
|
311
|
+
|
|
312
|
+
New `packages/sdk/src/peers/` module: `resolvePeerRuntimeDir` (a hardened per-user runtime directory; among `$XDG_RUNTIME_DIR/namzu`, `$NAMZU_HOME/run` and `$TMPDIR/namzu-<uid>`, the first candidate that both fits the platform's Unix-domain-socket path limit AND can actually be hardened is used — a candidate that fits but cannot be hardened, e.g. one owned by another uid, falls through to the next rather than failing the whole call), the live-session registry (`PeerRecordSchema`, `writePeerRecord`/`readPeerRecord`/`readPeerRecords`/`removePeerRecord`, `derivePeerRef`, `listLivePeers`/`isPeerRecordLive`), the `namzu-peer/1` wire protocol (`PeerRequestSchema` and friends: `ping`/`deliver`/`subscribe_idle`/`notice`, a closed op set, newline-delimited JSON, a 64 KiB request cap, a 2-second read deadline, at most 16 concurrent connections, `timingSafeEqual` token comparison), `createPeerEndpoint` (a UDS server on POSIX; a named-pipe server is written for win32 but not exercised by this package's tests) and `PeerClient`/`pingPeer`.
|
|
313
|
+
|
|
314
|
+
Every op but `ping` authenticates its sender the same way: `deliver`, `subscribe_idle` and `notice` all carry a required `from`, checked by `verifySender` against the sender's own live registry record (address, `permissionMode`, `kind` must all match — a session cannot claim a different mode or kind than the one it registered, closing a hold-gate bypass this had before its first use). On success, the identity handed to `onDeliver`/`onSubscribeIdle`/`onNotice` is rebuilt entirely from that record — `name` is the record's own display name (`title`, or the last path segment of `cwd`), never the wire's claim — so a display name shown to an operator cannot be spoofed per-message. A `notice` additionally has to correlate to a relationship the recipient's endpoint actually has with that sender (a `deliver` it sent, via the new `PeerEndpoint.registerOutstandingDelivery`, or a subscription it made, via `registerOutstandingSubscription`) and must be about the sender's own identity (`from.sessionId === about.sessionId`); an uncorrelated or misattributed notice is refused. Both outstanding tables are bounded and expiring (`MAX_OUTSTANDING_PEERS` distinct peers, `MAX_OUTSTANDING_PER_PEER` per peer, `OUTSTANDING_EXPIRY_MS` of inactivity, all overridable `createPeerEndpoint` options), evicting by least-recently-used — a peer touched again (a fresh registration, or a notice consumed against it) is not the one forgotten first — and warning once per peer per minute, through the new `logger` option, when a registration is dropped at the per-peer cap or a peer is evicted to stay under the peer-count cap (no message content in the warning, only which table and which peer). All of this closes gaps an internal review found before this module's first release — nothing here has ever shipped in a published version.
|
|
315
|
+
|
|
316
|
+
Two new SDK-visible pieces:
|
|
317
|
+
|
|
318
|
+
- `RUNTIME_CONTEXT_MESSAGE_KINDS` gains `peer-message` and `peer-notice`; `isOperatorUserMessage` (internal) stays `false` for both, so a message from another session is never treated as operator intent.
|
|
319
|
+
- `formatSystemEvent` (`packages/sdk/src/runtime/system-events.ts`): one shared envelope for an asynchronous event a running turn reads as context — `<system-event-<nonce> kind="…" id="…" status="…">`, a fixed sentence that the event is not the operator and not consent, a second sentence naming the exact nonce-bound closing tag as the sole real boundary, `summary`/`source`/`usage`/`more` metadata, then the untrusted body. `formatPeerMessage`/`formatPeerNotice` render through it. This does not migrate the existing `formatCompletionNotification` (`<task-notification>`) text or the `job-exit` path — that is a separate change. Its delimiter defense against a value it did not author (a peer's chosen name, an agent's label) binds the real tag to a fresh, unpredictable per-render nonce (`tools/render-nonce.ts`, new and internal) rather than recognizing the keyword by pattern — a lookalike character (including a same-script homoglyph no amount of Unicode normalization folds to the ASCII original) cannot forge a fake close of the frame plus a fake second event, because it cannot predict the nonce, and untrusted text is no longer altered to achieve this. `options.generateNonce` now governs the body's own nested `wrapUntrusted` call too, not only the outer frame, so an event with a body renders fully deterministically when a caller injects one.
|
|
320
|
+
|
|
321
|
+
Everything here is transport and storage only. CLI wiring — registry lifecycle across TUI/exec/resident sessions, the mode-mismatch hold/approve flow, `/peers`, `list_sessions`, a widened `send_message` — is a later, separate change; nothing in this release queues a message, wakes an idle session, or injects anything into a running turn on its own.
|
|
322
|
+
|
|
323
|
+
- 443094a: A session whose last turn completed after using a tool could not take a third message: `namzu resume <id>` (or a still-running `exec`/TUI process) refused it with `Message history repeats tool-call id '…'; a signed assistant turn cannot be rewritten safely.` and the turn never reached a provider. Three rounds of independent review then found the fixes attempted along the way were themselves partial patches: a host that edited an earlier cached message, cached only a recent suffix, or never carried an id at all (a stateless `exec --json` host, or an AG-UI adapter) could still crash, silently duplicate a tool call, or silently drop or swallow a caller-added message instead of reaching the provider.
|
|
324
|
+
|
|
325
|
+
The root cause of all of these was the same: `query()` reconciled a caller's cached prior messages against a fresh fold of the session log by comparing the two arrays **positionally**, or, in two successive no-id fallbacks, by walking a cursor that never re-anchored after a mismatch, or by searching for some aligned position by value. None of these can tell a host echoing old content apart from a host writing new content that happens to match it — depending on which way the coincidence ran, that either duplicated shared history (including tool calls, which `validateToolCallIds` then correctly refused as unsafe) or silently dropped something that was never durable.
|
|
326
|
+
|
|
327
|
+
Reconciliation is now by **id** where a message carries one, and by an **anchored, whole-cache rule** against the log's own fold otherwise — never by position, a cursor, or a search for alignment. Every message a host is handed back — `Turn.messages`, the messages `onConversationMessages` reports, a checkpoint's restored messages — now carries the id of the durable `message` record it came from (`BaseMessage.id`, a new optional field the kernel alone sets; never send one yourself).
|
|
328
|
+
|
|
329
|
+
- A message carrying an id the log recorded, **unedited**, is already durable and is dropped — including one from before a compaction that has since summarized it away, since the log remembers every id it ever gave out, not only the current fold.
|
|
330
|
+
- A message carrying an id, **edited** before being resent, fails the turn with the new `stale_cached_history` error (`details.kind: 'edited'`) instead of crashing or silently corrupting history.
|
|
331
|
+
- A message carrying an id **this log never recorded** (a host that minted its own, or copied one over from a different session — including a fork's seeded history, now correctly stripped of the source session's ids) fails the same way (`details.kind: 'foreign'`).
|
|
332
|
+
- Every message with no id — new input, or one from a host that never adopted `.id` at all, or one a protocol adapter could only ever attach to a message it produced itself — reconciles against what remains of the fold by exactly one recognized shape: the cache's own no-id messages start with an EXACT, in-order match of every remaining fold message. That matched block is dropped as already durable; anything after it is new, unconditionally, never compared to anything — so a message that repeats older text as a host's newest addition (a second "hello") is never mistaken for an echo, in either direction. Everything else — a cache with no more messages than the fold has left, or one diverging from the fold's very first message — is new in full, with no value matching attempted at all. **A host without ids must therefore send either its whole cache (recognized and trimmed) or only new messages (all kept) to reconcile by value; a genuinely partial cache is not reconciled by value and may duplicate what the fold already holds — such a host needs `.id` instead.**
|
|
333
|
+
- The one case still refused rather than guessed at: a cache that looks like an attempted full resend (its first message matches the fold's first) but diverges further in — refused as `stale_cached_history` (`details.kind: 'unaligned'`), naming a fold this log does not actually hold, rather than duplicating or dropping content nobody can prove is redundant.
|
|
334
|
+
|
|
335
|
+
Every `stale_cached_history` refusal fails closed before any provider call — never concatenates a caller's stale cache after the log's own fold. The way out for a host that hits one is to pass only its new messages, or to re-read history from the session instead of reusing a cached copy. A host that only ever resends exactly what it was given back needs no change and sees no new error.
|
|
336
|
+
|
|
337
|
+
`resumeSession` also refuses a checkpoint whose turn the session log already shows completed or failed, before claiming a lease under its id, instead of only inside `query()`'s own deeper (and still-correct) refusal.
|
|
338
|
+
|
|
339
|
+
### Patch Changes
|
|
340
|
+
|
|
341
|
+
- 2d4ff9a: Shipped text no longer names one particular application built on namzu. Nothing to do to upgrade: no type, default or behaviour changes.
|
|
342
|
+
|
|
343
|
+
- `@namzu/sandbox`: when a self-hosted Firecracker orchestrator returns a network-mode (`mtls`) agent handle and the backend was given no client certificate, the error now tells you to pass `mtls: { ca, cert, key }` in the backend config. It used to name environment variables that only one host defines, which no other installer has. The message still starts with `firecracker: orchestrator returned an mtls agent handle but no client cert material was injected`, so code matching on that prefix keeps working.
|
|
344
|
+
- Doc comments in the published `.d.ts` files and sources (`ContainerBackendConfig.labels`, the ACI and Azure Blob name options, the sandbox mount-source types) say "the host" or "the consumer", and label examples use the placeholder `acme.` namespace. The `microvm` tier's (`MicroVMBackendConfig`, `AgentSnapshotRef`, `OrchestratorNetworkPolicy`) describe the orchestrator as a self-hosted one the host runs, not as namzu's own: namzu ships only the client.
|
|
345
|
+
- Earlier entries in the `@namzu/sandbox`, `@namzu/sdk` and `@namzu/anthropic` CHANGELOGs are reworded the same way; the `@namzu/files` CHANGELOG named no one and is unchanged. Versions already on npm keep their old text.
|
|
346
|
+
|
|
3
347
|
## 46.0.0
|
|
4
348
|
|
|
5
349
|
### Major Changes
|
|
@@ -6877,7 +7221,7 @@ v2'` instead of a scatter of assertion failures whose common cause is not
|
|
|
6877
7221
|
What a consumer sees change:
|
|
6878
7222
|
|
|
6879
7223
|
- `@namzu/sandbox` raised `Sandbox backend 'x' is not implemented yet. Track
|
|
6880
|
-
progress in
|
|
7224
|
+
progress in <a local notes directory>/...` — a runtime error
|
|
6881
7225
|
instructing the reader to open a path that is not in the package, not in the
|
|
6882
7226
|
repository, and not on the internet. It now names what does ship instead.
|
|
6883
7227
|
- `@namzu/computer-use`'s README linked to an adapter-pattern document under a
|
|
@@ -12622,7 +12966,7 @@ ProviderCapabilities` (with a new `supportsVision?` flag on the type)
|
|
|
12622
12966
|
|
|
12623
12967
|
### Patch Changes
|
|
12624
12968
|
|
|
12625
|
-
- 999e4be: Context-management correctness fixes (
|
|
12969
|
+
- 999e4be: Context-management correctness fixes (from a third-round architecture audit).
|
|
12626
12970
|
|
|
12627
12971
|
- **Compaction no longer orphans tool pairs.** `runCompactionCheck` now snaps
|
|
12628
12972
|
the recent-window boundary through `findSafeTrimIndex` (previously wired only
|
|
@@ -12771,13 +13115,13 @@ ProviderCapabilities` (with a new `supportsVision?` flag on the type)
|
|
|
12771
13115
|
outputs: {
|
|
12772
13116
|
source: {
|
|
12773
13117
|
type: "hostDir",
|
|
12774
|
-
hostPath: "/var/lib
|
|
13118
|
+
hostPath: "/var/lib/<host>/sessions/<task>/outputs",
|
|
12775
13119
|
},
|
|
12776
13120
|
},
|
|
12777
13121
|
uploads: {
|
|
12778
13122
|
source: {
|
|
12779
13123
|
type: "hostDir",
|
|
12780
|
-
hostPath: "/var/lib
|
|
13124
|
+
hostPath: "/var/lib/<host>/sessions/<task>/uploads",
|
|
12781
13125
|
},
|
|
12782
13126
|
},
|
|
12783
13127
|
skills: [
|
|
@@ -12864,8 +13208,7 @@ b.cause = a`), and longer loops, replacing the offending node with
|
|
|
12864
13208
|
- The docker backend no longer allocates host directories
|
|
12865
13209
|
(`mkdtemp`) or removes them on `destroy()`. Every bind source is
|
|
12866
13210
|
consumer-owned. This also fixes an `EACCES: permission denied,
|
|
12867
|
-
mkdir '/Users'` crash that hit sibling-container deployments
|
|
12868
|
-
(Vandal Cowork).
|
|
13211
|
+
mkdir '/Users'` crash that hit sibling-container deployments.
|
|
12869
13212
|
- The worker no longer reads `NAMZU_SANDBOX_LAYOUT` (it never
|
|
12870
13213
|
branched on the env, only logged it; size grew with the skill
|
|
12871
13214
|
list). Only `NAMZU_SANDBOX_WORKSPACE` is forwarded today.
|
|
@@ -13093,8 +13436,8 @@ test:smoke`) runs an opt-in docker integration test exercising the
|
|
|
13093
13436
|
The kernel now emits a per-message and per-tool-input lifecycle on the
|
|
13094
13437
|
event bus, and the provider contract collapses to a single streaming
|
|
13095
13438
|
entry point. Together these unlock live tool-call rendering (Calling →
|
|
13096
|
-
Running → Done with incremental input) for SSE consumers —
|
|
13097
|
-
workspace surface
|
|
13439
|
+
Running → Done with incremental input) for SSE consumers — a live
|
|
13440
|
+
workspace surface motivated the work in the first place.
|
|
13098
13441
|
|
|
13099
13442
|
## Breaking changes
|
|
13100
13443
|
|
package/README.md
CHANGED
|
@@ -93,11 +93,10 @@ This complete example scripts two model turns and executes a real local tool.
|
|
|
93
93
|
The mock requests `add`, then supplies the final answer; it does no inference.
|
|
94
94
|
|
|
95
95
|
```ts
|
|
96
|
-
import { defineTool, MockLLMProvider, runAgent,
|
|
96
|
+
import { defineTool, MockLLMProvider, runAgent, toolset } from '@namzu/sdk'
|
|
97
97
|
import { z } from 'zod'
|
|
98
98
|
|
|
99
|
-
const
|
|
100
|
-
tools.register(defineTool({
|
|
99
|
+
const toolsets = [toolset('math', [defineTool({
|
|
101
100
|
name: 'add',
|
|
102
101
|
description: 'Add two numbers.',
|
|
103
102
|
inputSchema: z.object({ a: z.number().finite(), b: z.number().finite() }),
|
|
@@ -107,7 +106,7 @@ tools.register(defineTool({
|
|
|
107
106
|
destructive: false,
|
|
108
107
|
concurrencySafe: true,
|
|
109
108
|
execute: async ({ a, b }) => ({ success: true, output: String(a + b) }),
|
|
110
|
-
}))
|
|
109
|
+
})])]
|
|
111
110
|
|
|
112
111
|
const provider = new MockLLMProvider({
|
|
113
112
|
turns: [
|
|
@@ -119,7 +118,7 @@ const provider = new MockLLMProvider({
|
|
|
119
118
|
const { output, turn } = await runAgent({
|
|
120
119
|
provider,
|
|
121
120
|
model: 'mock-model',
|
|
122
|
-
|
|
121
|
+
toolsets,
|
|
123
122
|
prompt: 'Add 20 and 22.',
|
|
124
123
|
maxIterations: 4,
|
|
125
124
|
tokenBudget: 8192,
|
|
@@ -140,7 +139,7 @@ checkpoints. The session is recorded in one append-only log under
|
|
|
140
139
|
and nothing is written under the working directory; see the
|
|
141
140
|
[session log](https://github.com/cogitave/namzu/blob/main/docs/sdk/session-log.md).
|
|
142
141
|
A session has at most one active turn: starting another while one is running
|
|
143
|
-
or paused throws `TurnInProgressError`. `
|
|
142
|
+
or paused throws `TurnInProgressError`. `QueryAgent` exposes additional
|
|
144
143
|
configuration such as compaction and where the session is stored; the config
|
|
145
144
|
passed to its `run` method requires `sessionId`, `topicId`, `projectId` and
|
|
146
145
|
`tenantId`. OS isolation is explicit rather than ambient: supply a
|
|
@@ -185,7 +184,7 @@ carries its durable trace parent into the cancelled turn, preserving one
|
|
|
185
184
|
cross-process timeline without a second checkpoint read.
|
|
186
185
|
|
|
187
186
|
Hosts that discover scoped repository policy can supply a
|
|
188
|
-
`ProjectInstructionContext` to `query`, `runAgent`, `
|
|
187
|
+
`ProjectInstructionContext` to `query`, `runAgent`, `QueryAgent`, or
|
|
189
188
|
`SupervisorAgent`. Its first-request snapshot is structurally tagged and
|
|
190
189
|
retained; completed registry calls, including nested dispatch, can publish a
|
|
191
190
|
replacement immediately after the complete tool-result batch. Each callback
|
|
@@ -197,7 +196,7 @@ predicate cannot strand the update. Canonical project-relative `AGENTS.md`
|
|
|
197
196
|
provenance survives compaction and lets a reconstructed host re-read disk
|
|
198
197
|
authority rather than trusting persisted policy text.
|
|
199
198
|
|
|
200
|
-
|
|
199
|
+
The `QueryAgent` configuration and the compatibility `SupervisorAgent` configuration also accept
|
|
201
200
|
`paths`, a `SessionPaths`. Supplying one puts the session log, its child
|
|
202
201
|
sessions, checkpoints, token ledger and task state under that root instead of
|
|
203
202
|
`resolveNamzuHome()`. A project id is minted once per working directory into
|