@namzu/sdk 40.0.0 → 42.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/CHANGELOG.md +236 -0
  2. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  3. package/dist/agents/ReactiveAgent.js +3 -0
  4. package/dist/agents/ReactiveAgent.js.map +1 -1
  5. package/dist/agents/SupervisorAgent.d.ts.map +1 -1
  6. package/dist/agents/SupervisorAgent.js +11 -0
  7. package/dist/agents/SupervisorAgent.js.map +1 -1
  8. package/dist/agents/runAgent.d.ts +14 -0
  9. package/dist/agents/runAgent.d.ts.map +1 -1
  10. package/dist/agents/runAgent.js +3 -0
  11. package/dist/agents/runAgent.js.map +1 -1
  12. package/dist/bridge/a2a/mapper.d.ts.map +1 -1
  13. package/dist/bridge/a2a/mapper.js +8 -0
  14. package/dist/bridge/a2a/mapper.js.map +1 -1
  15. package/dist/bridge/sse/mapper.d.ts.map +1 -1
  16. package/dist/bridge/sse/mapper.js +11 -0
  17. package/dist/bridge/sse/mapper.js.map +1 -1
  18. package/dist/connector/index.d.ts +2 -2
  19. package/dist/connector/index.d.ts.map +1 -1
  20. package/dist/connector/index.js +1 -1
  21. package/dist/connector/index.js.map +1 -1
  22. package/dist/connector/mcp/adapter.d.ts.map +1 -1
  23. package/dist/connector/mcp/adapter.js +92 -4
  24. package/dist/connector/mcp/adapter.js.map +1 -1
  25. package/dist/connector/mcp/audio-admission.d.ts +17 -0
  26. package/dist/connector/mcp/audio-admission.d.ts.map +1 -0
  27. package/dist/connector/mcp/audio-admission.js +171 -0
  28. package/dist/connector/mcp/audio-admission.js.map +1 -0
  29. package/dist/connector/mcp/client.d.ts +252 -1
  30. package/dist/connector/mcp/client.d.ts.map +1 -1
  31. package/dist/connector/mcp/client.js +611 -39
  32. package/dist/connector/mcp/client.js.map +1 -1
  33. package/dist/connector/mcp/envelope.d.ts +91 -0
  34. package/dist/connector/mcp/envelope.d.ts.map +1 -0
  35. package/dist/connector/mcp/envelope.js +173 -0
  36. package/dist/connector/mcp/envelope.js.map +1 -0
  37. package/dist/connector/mcp/era.d.ts +130 -0
  38. package/dist/connector/mcp/era.d.ts.map +1 -0
  39. package/dist/connector/mcp/era.js +304 -0
  40. package/dist/connector/mcp/era.js.map +1 -0
  41. package/dist/connector/mcp/errors.d.ts +106 -0
  42. package/dist/connector/mcp/errors.d.ts.map +1 -0
  43. package/dist/connector/mcp/errors.js +154 -0
  44. package/dist/connector/mcp/errors.js.map +1 -0
  45. package/dist/connector/mcp/http-sse.d.ts +11 -0
  46. package/dist/connector/mcp/http-sse.d.ts.map +1 -1
  47. package/dist/connector/mcp/http-sse.js +21 -6
  48. package/dist/connector/mcp/http-sse.js.map +1 -1
  49. package/dist/connector/mcp/index.d.ts +7 -0
  50. package/dist/connector/mcp/index.d.ts.map +1 -1
  51. package/dist/connector/mcp/index.js +10 -0
  52. package/dist/connector/mcp/index.js.map +1 -1
  53. package/dist/connector/mcp/streamable-http.d.ts +83 -0
  54. package/dist/connector/mcp/streamable-http.d.ts.map +1 -1
  55. package/dist/connector/mcp/streamable-http.js +177 -11
  56. package/dist/connector/mcp/streamable-http.js.map +1 -1
  57. package/dist/connector/mcp/x-mcp-header.d.ts +56 -0
  58. package/dist/connector/mcp/x-mcp-header.d.ts.map +1 -0
  59. package/dist/connector/mcp/x-mcp-header.js +254 -0
  60. package/dist/connector/mcp/x-mcp-header.js.map +1 -0
  61. package/dist/constants/mcp/index.d.ts +123 -15
  62. package/dist/constants/mcp/index.d.ts.map +1 -1
  63. package/dist/constants/mcp/index.js +135 -16
  64. package/dist/constants/mcp/index.js.map +1 -1
  65. package/dist/manager/agent/lifecycle.d.ts.map +1 -1
  66. package/dist/manager/agent/lifecycle.js +43 -0
  67. package/dist/manager/agent/lifecycle.js.map +1 -1
  68. package/dist/prompt/coding-agent-doctrine.d.ts +20 -0
  69. package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
  70. package/dist/prompt/coding-agent-doctrine.js +19 -3
  71. package/dist/prompt/coding-agent-doctrine.js.map +1 -1
  72. package/dist/prompt/index.d.ts +1 -1
  73. package/dist/prompt/index.d.ts.map +1 -1
  74. package/dist/prompt/index.js +1 -1
  75. package/dist/prompt/index.js.map +1 -1
  76. package/dist/public-runtime.d.ts +7 -4
  77. package/dist/public-runtime.d.ts.map +1 -1
  78. package/dist/public-runtime.js +16 -4
  79. package/dist/public-runtime.js.map +1 -1
  80. package/dist/public-tools.d.ts +1 -1
  81. package/dist/public-tools.d.ts.map +1 -1
  82. package/dist/public-tools.js +4 -2
  83. package/dist/public-tools.js.map +1 -1
  84. package/dist/registry/tool/execute.d.ts.map +1 -1
  85. package/dist/registry/tool/execute.js +10 -1
  86. package/dist/registry/tool/execute.js.map +1 -1
  87. package/dist/runtime/bidi/session.d.ts +11 -0
  88. package/dist/runtime/bidi/session.d.ts.map +1 -1
  89. package/dist/runtime/bidi/session.js +2 -0
  90. package/dist/runtime/bidi/session.js.map +1 -1
  91. package/dist/runtime/query/executor.d.ts +6 -0
  92. package/dist/runtime/query/executor.d.ts.map +1 -1
  93. package/dist/runtime/query/executor.js +6 -0
  94. package/dist/runtime/query/executor.js.map +1 -1
  95. package/dist/runtime/query/guardrail-presets.d.ts +187 -1
  96. package/dist/runtime/query/guardrail-presets.d.ts.map +1 -1
  97. package/dist/runtime/query/guardrail-presets.js +298 -0
  98. package/dist/runtime/query/guardrail-presets.js.map +1 -1
  99. package/dist/runtime/query/index.d.ts +14 -0
  100. package/dist/runtime/query/index.d.ts.map +1 -1
  101. package/dist/runtime/query/index.js +3 -0
  102. package/dist/runtime/query/index.js.map +1 -1
  103. package/dist/runtime/query/tooling.d.ts +2 -0
  104. package/dist/runtime/query/tooling.d.ts.map +1 -1
  105. package/dist/runtime/query/tooling.js +3 -0
  106. package/dist/runtime/query/tooling.js.map +1 -1
  107. package/dist/sandbox/provider/local.d.ts.map +1 -1
  108. package/dist/sandbox/provider/local.js +46 -3
  109. package/dist/sandbox/provider/local.js.map +1 -1
  110. package/dist/scheduler/local.d.ts.map +1 -1
  111. package/dist/scheduler/local.js +8 -0
  112. package/dist/scheduler/local.js.map +1 -1
  113. package/dist/store/run/disk.d.ts +35 -1
  114. package/dist/store/run/disk.d.ts.map +1 -1
  115. package/dist/store/run/disk.js +100 -0
  116. package/dist/store/run/disk.js.map +1 -1
  117. package/dist/tools/coordinator/agent.d.ts.map +1 -1
  118. package/dist/tools/coordinator/agent.js +17 -2
  119. package/dist/tools/coordinator/agent.js.map +1 -1
  120. package/dist/tools/coordinator/index.d.ts.map +1 -1
  121. package/dist/tools/coordinator/index.js +17 -3
  122. package/dist/tools/coordinator/index.js.map +1 -1
  123. package/dist/tools/untrusted-envelope.d.ts +35 -0
  124. package/dist/tools/untrusted-envelope.d.ts.map +1 -1
  125. package/dist/tools/untrusted-envelope.js +91 -3
  126. package/dist/tools/untrusted-envelope.js.map +1 -1
  127. package/dist/types/agent/base.d.ts +23 -0
  128. package/dist/types/agent/base.d.ts.map +1 -1
  129. package/dist/types/agent/scheduler.d.ts +20 -0
  130. package/dist/types/agent/scheduler.d.ts.map +1 -1
  131. package/dist/types/agent/task.d.ts +39 -0
  132. package/dist/types/agent/task.d.ts.map +1 -1
  133. package/dist/types/connector/mcp.d.ts +205 -0
  134. package/dist/types/connector/mcp.d.ts.map +1 -1
  135. package/dist/types/run/events.d.ts +56 -0
  136. package/dist/types/run/events.d.ts.map +1 -1
  137. package/dist/types/run/events.js.map +1 -1
  138. package/dist/types/run/store.d.ts +41 -0
  139. package/dist/types/run/store.d.ts.map +1 -1
  140. package/dist/types/sandbox/index.d.ts +68 -1
  141. package/dist/types/sandbox/index.d.ts.map +1 -1
  142. package/dist/types/sandbox/index.js.map +1 -1
  143. package/dist/types/tool/index.d.ts +19 -0
  144. package/dist/types/tool/index.d.ts.map +1 -1
  145. package/dist/types/tool/index.js.map +1 -1
  146. package/package.json +1 -1
  147. package/src/agents/ReactiveAgent.ts +3 -0
  148. package/src/agents/SupervisorAgent.ts +11 -0
  149. package/src/agents/runAgent.ts +18 -0
  150. package/src/bridge/a2a/mapper.ts +8 -0
  151. package/src/bridge/sse/mapper.ts +11 -0
  152. package/src/connector/index.ts +27 -0
  153. package/src/connector/mcp/adapter.ts +103 -4
  154. package/src/connector/mcp/audio-admission.ts +173 -0
  155. package/src/connector/mcp/client.ts +694 -45
  156. package/src/connector/mcp/envelope.ts +235 -0
  157. package/src/connector/mcp/era.ts +400 -0
  158. package/src/connector/mcp/errors.ts +171 -0
  159. package/src/connector/mcp/http-sse.ts +23 -6
  160. package/src/connector/mcp/index.ts +37 -0
  161. package/src/connector/mcp/streamable-http.ts +199 -11
  162. package/src/connector/mcp/x-mcp-header.ts +322 -0
  163. package/src/constants/mcp/index.ts +145 -16
  164. package/src/manager/agent/lifecycle.ts +51 -0
  165. package/src/prompt/coding-agent-doctrine.ts +31 -4
  166. package/src/prompt/index.ts +1 -0
  167. package/src/public-runtime.ts +42 -0
  168. package/src/public-tools.ts +8 -2
  169. package/src/registry/tool/execute.ts +9 -1
  170. package/src/runtime/bidi/session.ts +13 -0
  171. package/src/runtime/query/executor.ts +13 -0
  172. package/src/runtime/query/guardrail-presets.ts +356 -0
  173. package/src/runtime/query/index.ts +17 -0
  174. package/src/runtime/query/tooling.ts +5 -0
  175. package/src/sandbox/provider/local.ts +45 -2
  176. package/src/scheduler/local.ts +8 -0
  177. package/src/store/run/disk.ts +108 -0
  178. package/src/tools/coordinator/agent.ts +17 -2
  179. package/src/tools/coordinator/index.ts +17 -3
  180. package/src/tools/untrusted-envelope.ts +94 -3
  181. package/src/types/agent/base.ts +24 -0
  182. package/src/types/agent/scheduler.ts +21 -0
  183. package/src/types/agent/task.ts +41 -0
  184. package/src/types/connector/mcp.ts +205 -1
  185. package/src/types/run/events.ts +56 -0
  186. package/src/types/run/store.ts +42 -0
  187. package/src/types/sandbox/index.ts +69 -1
  188. package/src/types/tool/index.ts +20 -0
@@ -725,12 +725,29 @@ export {
725
725
  hasDrift,
726
726
  HttpSseTransport,
727
727
  HybridExecutionContext,
728
+ buildEnvelope,
729
+ classifyModernHttpFailure,
730
+ createMcpEraCache,
731
+ decodeResult,
732
+ defaultMcpEraCache,
733
+ encodeMcpHeaderValue,
734
+ isHeaderMismatchError,
735
+ isMissingRequiredClientCapabilityError,
736
+ isRecognizedModernError,
737
+ isResourceNotFoundError,
738
+ isUnsupportedProtocolVersionError,
728
739
  LocalExecutionContext,
740
+ mcpEraCacheKey,
741
+ MCPHttpStatusError,
742
+ MCPInputRequiredError,
743
+ MCPInvalidResultTypeError,
729
744
  MCPClient,
730
745
  MCPMethodNotFound,
731
746
  MCPConnectorBridge,
747
+ MCPProtocolError,
732
748
  MCPServer,
733
749
  MCPToolDiscovery,
750
+ resolveMcpEra,
734
751
  mcpPromptToToolDefinition,
735
752
  renderPromptMessages,
736
753
  mcpJsonSchemaToZod,
@@ -749,10 +766,20 @@ export {
749
766
  toolDefinitionToMCPTool,
750
767
  toolResultToMCPToolResult,
751
768
  toolsHash,
769
+ validateMcpHeaderAnnotations,
752
770
  WebhookConnector,
753
771
  zodToMCPJsonSchema,
754
772
  } from './connector/index.js'
755
773
  export type {
774
+ McpEnvelope,
775
+ McpEnvelopeInput,
776
+ McpEraProbe,
777
+ McpEraProbeAnswer,
778
+ McpEraResolution,
779
+ McpEraResolutionInput,
780
+ McpHeaderAnnotationVerdict,
781
+ McpParamHeaderBinding,
782
+ MCPDecodedResult,
756
783
  MCPToolDiscoveryOptions,
757
784
  MCPToolDrift,
758
785
  MCPToolPolicy,
@@ -1165,8 +1192,21 @@ export type { TokenUsageSample } from './telemetry/metrics.js'
1165
1192
  export {
1166
1193
  promptInjectionGuardrail,
1167
1194
  secretRedactionGuardrail,
1195
+ toolResultCorrespondenceGuardrail,
1168
1196
  toolResultInjectionGuardrail,
1169
1197
  } from './runtime/query/guardrail-presets.js'
1198
+ // Which names a `passthroughTools` list has to contain for a given tool. A
1199
+ // caller that REPORTS on such a list — the CLI checks an operator's names
1200
+ // against the tools its registry actually holds, so a name that matches
1201
+ // nothing is said out loud rather than silently ignored — has to derive them
1202
+ // the way the screen does rather than keep a second copy of the rule: a name
1203
+ // the reporter accepts and the screen does not is an exemption the operator
1204
+ // believes is in force.
1205
+ export { passthroughToolNames } from './runtime/query/guardrail-presets.js'
1206
+ // What a run installs when its host configured no screens. Exported because a
1207
+ // caller who wants to keep the default AND add to it has to be able to name
1208
+ // it: `[...DEFAULT_TOOL_RESULT_GUARDRAILS, myScreen()]`.
1209
+ export { DEFAULT_TOOL_RESULT_GUARDRAILS } from './runtime/query/guardrail-presets.js'
1170
1210
  // Thrown by a tool-result screen that returned `halt`. Exported because a
1171
1211
  // host has to be able to tell it from an ordinary failure — that is the
1172
1212
  // entire difference between the two refusal outcomes.
@@ -1229,6 +1269,7 @@ export {
1229
1269
  export {
1230
1270
  CODING_AGENT_DELEGATION_DOCTRINE,
1231
1271
  CODING_AGENT_DOCTRINE_CONTRIBUTION_ID,
1272
+ CODING_AGENT_ORCHESTRATE_DOCTRINE,
1232
1273
  CODING_AGENT_WORKING_DOCTRINE,
1233
1274
  PLAN_MODE_DOCTRINE,
1234
1275
  PromptContributionCollisionError,
@@ -1334,6 +1375,7 @@ export type { ToolCatalogFromRegistryOptions } from './registry/toolset/catalog.
1334
1375
  export type { MockBidiScript, MockBidiSession } from './runtime/bidi/mock.js'
1335
1376
  export type { BidiRun, BidiRunParams } from './runtime/bidi/session.js'
1336
1377
  export type { SecretRedactionOptions } from './runtime/query/guardrail-presets.js'
1378
+ export type { ToolResultCorrespondenceOptions } from './runtime/query/guardrail-presets.js'
1337
1379
  export type { ListCheckpointsInput } from './runtime/query/replay/list.js'
1338
1380
  export type {
1339
1381
  PrepareReplayInput,
@@ -17,8 +17,14 @@ export { defineTool } from './tools/defineTool.js'
17
17
  // this file needed them.
18
18
  export { isWithin, resolveWithin, resolveWithinReal } from './tools/paths.js'
19
19
  // A host that surfaces its own untrusted content to a model needs the same
20
- // framing the kernel applies to connector prompts and delegated results.
21
- export { neutralizeEnvelopeDelimiter, wrapUntrusted } from './tools/untrusted-envelope.js'
20
+ // framing the kernel applies to connector prompts and delegated results — and
21
+ // the reader for it, because a screen that judges a result has to reach past
22
+ // the frame without re-spelling the tag.
23
+ export {
24
+ neutralizeEnvelopeDelimiter,
25
+ untrustedEnvelopeBody,
26
+ wrapUntrusted,
27
+ } from './tools/untrusted-envelope.js'
22
28
  export { filterReadOnlyTools, filterToolsNamed } from './tools/roster.js'
23
29
  export type { UntrustedEnvelope } from './tools/untrusted-envelope.js'
24
30
 
@@ -679,6 +679,7 @@ Executable tool names, descriptions, and JSON input schemas are attached through
679
679
  try {
680
680
  this.log.debug('Executing tool', { 'namzu.tool.name': toolName })
681
681
  const startedAt = Date.now()
682
+ const runResultGuardrails = context.toolResultGuardrails
682
683
  const produced = await tool.execute(finalInput, context)
683
684
  // Screened here, which is the only place a result can be
684
685
  // examined before anything acts on it: the executor applies
@@ -689,7 +690,14 @@ Executable tool names, descriptions, and JSON input schemas are attached through
689
690
  // the server's name, and a screen reading only the value
690
691
  // cannot use that.
691
692
  const result = await screenToolResult(
692
- this.resultGuardrails,
693
+ // Explicit configuration wins, at whichever boundary it
694
+ // was made. A registry built WITH `resultGuardrails`
695
+ // has stated its policy — including `[]`, which means
696
+ // none — and a run must not overrule it. A registry
697
+ // built without one declared none, so the run's apply;
698
+ // that is the ordinary case, since a host assembles a
699
+ // registry and hands it to a run it does not own.
700
+ this.resultGuardrails ?? runResultGuardrails,
693
701
  produced,
694
702
  {
695
703
  toolName,
@@ -5,12 +5,14 @@ import type {
5
5
  BidiRunEvent,
6
6
  BidiSession,
7
7
  } from '../../types/bidi/index.js'
8
+ import type { ToolResultGuardrailSpec } from '../../types/guardrail/index.js'
8
9
  import type { RunId } from '../../types/ids/index.js'
9
10
  import type { ToolContext, ToolRegistryContract } from '../../types/tool/index.js'
10
11
  import { toErrorMessage } from '../../utils/error.js'
11
12
  import { generateRunId } from '../../utils/id.js'
12
13
  import { SCOPE_ATTRIBUTE } from '../../utils/log/types.js'
13
14
  import { type Logger, resolveLogger } from '../../utils/logger.js'
15
+ import { DEFAULT_TOOL_RESULT_GUARDRAILS } from '../query/guardrail-presets.js'
14
16
 
15
17
  const DEFAULT_CLOSE_TIMEOUT_MS = 5_000
16
18
  const MAX_TIMER_DELAY_MS = 2_147_483_647
@@ -78,6 +80,16 @@ export interface BidiRunParams {
78
80
  readonly log?: Logger
79
81
  /** Overrides the generated id, so a host can correlate its own. */
80
82
  readonly runId?: RunId
83
+ /**
84
+ * Screens for the results this session's tools produce. Absent installs
85
+ * {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array installs none.
86
+ *
87
+ * Here rather than nowhere because this path builds its OWN tool context:
88
+ * a duplex session executes the tools the model asks for, and its results
89
+ * reach a model just as a turn's do. A registry built with
90
+ * `resultGuardrails` still wins, as it does on the query path.
91
+ */
92
+ readonly toolResultGuardrails?: readonly ToolResultGuardrailSpec[]
81
93
  }
82
94
 
83
95
  export interface BidiRun {
@@ -230,6 +242,7 @@ export async function startBidiRun(params: BidiRunParams): Promise<BidiRun> {
230
242
  env: params.env ?? {},
231
243
  log: (level, message) => log[level](message),
232
244
  toolUseId: call.id,
245
+ toolResultGuardrails: params.toolResultGuardrails ?? DEFAULT_TOOL_RESULT_GUARDRAILS,
233
246
  }
234
247
  const result = await params.tools.execute(call.name, input, context)
235
248
  output = result.success ? result.output : (result.error ?? 'the tool failed')
@@ -13,6 +13,7 @@ import { renderToolSchema } from '../../registry/tool/schema.js'
13
13
  import type { ActivityStore } from '../../store/activity/memory.js'
14
14
  import { SKILL_TOOL_NAME } from '../../tools/builtins/skill.js'
15
15
  import { createFileReadTracker } from '../../tools/file-read-tracker.js'
16
+ import type { ToolResultGuardrailSpec } from '../../types/guardrail/index.js'
16
17
  import type { RunId, ToolUseId } from '../../types/ids/index.js'
17
18
  import type { InvocationState } from '../../types/invocation/index.js'
18
19
  import {
@@ -52,6 +53,7 @@ import { compressShellOutput } from '../../utils/shell-compress.js'
52
53
  import { type BackgroundJobRegistry, type JobProcess, bindOwner } from '../jobs/registry.js'
53
54
  import { describeVisibleFileEvidence } from './file-evidence-context.js'
54
55
  import { seedObservationLedger } from './file-evidence-seed.js'
56
+ import { DEFAULT_TOOL_RESULT_GUARDRAILS } from './guardrail-presets.js'
55
57
  import { skippedToolResultText } from './plugin-hooks.js'
56
58
  import type { ToolResultObservation } from './project-instructions.js'
57
59
  import { ToolCallBudget, assertMaxToolCalls } from './tool-call-budget.js'
@@ -393,6 +395,12 @@ export interface ToolExecutorConfig {
393
395
  /** See QueryParams.retainedToolPreviewChars; applies to the recorded host output. */
394
396
  retainedToolPreviewChars?: number
395
397
 
398
+ /**
399
+ * See QueryParams.toolResultGuardrails. Absent installs
400
+ * {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array installs none.
401
+ */
402
+ toolResultGuardrails?: readonly ToolResultGuardrailSpec[]
403
+
396
404
  /**
397
405
  * Cap on the RICH channel of a single tool result, in base64
398
406
  * characters. `0` or absent disables it.
@@ -1335,6 +1343,11 @@ export class ToolExecutor {
1335
1343
  this.skillScope = { ...scope, adoptedInBatch: this.batchCounter }
1336
1344
  },
1337
1345
  maxToolOutputChars: this.config.maxToolOutputChars ?? DEFAULT_MAX_TOOL_OUTPUT_CHARS,
1346
+ // The run's screens, defaulted HERE rather than on the registry: a
1347
+ // host builds the registry and hands it over, so a registry-side
1348
+ // default is the host's to write and the shipped one reaches
1349
+ // nobody. `[]` survives the `??` and is how a run says "none".
1350
+ toolResultGuardrails: this.config.toolResultGuardrails ?? DEFAULT_TOOL_RESULT_GUARDRAILS,
1338
1351
  ...(this.config.skills ? { skills: this.config.skills } : {}),
1339
1352
  ...(this.config.web ? { web: this.config.web } : {}),
1340
1353
  // The SAME registry and the SAME context a model-issued call
@@ -1,9 +1,12 @@
1
+ import { PLUGIN_NAMESPACE_SEPARATOR } from '../../constants/plugin/index.js'
1
2
  import { OUTPUT_SECRET_PATTERNS } from '../../constants/secret-patterns.js'
3
+ import { untrustedEnvelopeBody } from '../../tools/untrusted-envelope.js'
2
4
  import type {
3
5
  GuardrailVerdict,
4
6
  NamedGuardrail,
5
7
  OutputGuardrail,
6
8
  ToolResultGuardrail,
9
+ ToolResultGuardrailSpec,
7
10
  ToolResultVerdict,
8
11
  } from '../../types/guardrail/index.js'
9
12
 
@@ -169,3 +172,356 @@ export function toolResultInjectionGuardrail(): NamedGuardrail<ToolResultGuardra
169
172
  },
170
173
  }
171
174
  }
175
+
176
+ /**
177
+ * Requests shorter than this are not compared against the result.
178
+ *
179
+ * A one-word echo cannot be told from a one-word answer, and the shorter the
180
+ * value the likelier it is to recur in a legitimate result by coincidence:
181
+ * `ls` of a directory holding one entry called `src`, called with
182
+ * `{ path: "src" }`, returns `src`. The value of catching a five-character
183
+ * restatement is nil — nothing rides on it — so the comparison starts where
184
+ * a restatement actually means something.
185
+ *
186
+ * Counted in CODE POINTS, not UTF-16 units. Eight emoji are eight characters
187
+ * to every reader of this file and sixteen to `String.length`, and a floor
188
+ * that lets them through while the docblock says it excludes short values is
189
+ * a floor that is not doing what it says.
190
+ */
191
+ const MIN_RESTATEMENT_LENGTH = 16
192
+
193
+ /** Whether `text` is at least `minimum` code points long, without counting all of them. */
194
+ function isAtLeastCodePoints(text: string, minimum: number): boolean {
195
+ let count = 0
196
+ for (const _character of text) {
197
+ count += 1
198
+ if (count >= minimum) return true
199
+ }
200
+ return false
201
+ }
202
+
203
+ export interface ToolResultCorrespondenceOptions {
204
+ /**
205
+ * Which results this screen judges.
206
+ *
207
+ * `'framed'` — the default — is the results that carry the untrusted
208
+ * envelope: content this process did not author, marked as such by
209
+ * `wrapUntrusted`. For a connected server's tool that is every result the
210
+ * adapter returns (see `frameServerResult`), and it is also a host tool
211
+ * that framed its own result the same way — which is the point, and what
212
+ * an earlier `provenance`-based predicate got wrong: the CLI's own remote
213
+ * Exa search frames its answer and was left unscreened by accident of
214
+ * registration, while a fetch was screened by the same accident.
215
+ *
216
+ * `'all'` adds this process's own unframed tools. That is the host's call
217
+ * and not a safe default: a host that asks for it owns the consequences
218
+ * and names its own exceptions in {@link passthroughTools}.
219
+ */
220
+ readonly scope?: 'framed' | 'all'
221
+
222
+ /**
223
+ * Tools whose answer IS the request, and must not be refused for saying
224
+ * so: a validator returning what it validated, a normaliser returning the
225
+ * normalised form, a search that repeats its query when it found nothing,
226
+ * a fetch whose page body is its own URL.
227
+ *
228
+ * This is the escape hatch the check needs rather than a tuning knob. The
229
+ * screen is right about a tool that was asked a question and handed the
230
+ * question back; it is wrong about a tool whose purpose is to hand
231
+ * something back, and nothing on the context distinguishes the two. The
232
+ * false positive is the way this control dies — the host switches the
233
+ * screen off and it then protects nothing — so the exemption has to be
234
+ * cheap enough to be the first thing reached for.
235
+ *
236
+ * A tool answers to several names and each REGISTRATION SHAPE yields its
237
+ * own set; {@link passthroughToolNames} is the one implementation and
238
+ * spells them out, rather than this option and the check each carrying
239
+ * half of the rule.
240
+ */
241
+ readonly passthroughTools?: readonly string[]
242
+ }
243
+
244
+ interface RequestText {
245
+ /** Where in the request the text sits, e.g. `city` or `filters.city`. */
246
+ readonly path: string
247
+ readonly text: string
248
+ }
249
+
250
+ function normalizeText(text: string): string {
251
+ return text.trim().replace(/\s+/g, ' ')
252
+ }
253
+
254
+ /**
255
+ * Every string the call was made with, and where it sat.
256
+ *
257
+ * Strings only. A number is a plausible answer — a count of 10, a port of
258
+ * 8080 — and comparing one would fire on `{ maxResults: 10 }` answered with
259
+ * `10`, which is a legitimate result from a legitimate tool.
260
+ */
261
+ function collectRequestText(value: unknown, path: string, into: RequestText[]): void {
262
+ if (typeof value === 'string') {
263
+ const text = normalizeText(value)
264
+ if (isAtLeastCodePoints(text, MIN_RESTATEMENT_LENGTH)) into.push({ path, text })
265
+ return
266
+ }
267
+ if (Array.isArray(value)) {
268
+ for (const [index, item] of value.entries()) {
269
+ collectRequestText(item, `${path}[${index}]`, into)
270
+ }
271
+ return
272
+ }
273
+ if (value === null || typeof value !== 'object') return
274
+ for (const [key, nested] of Object.entries(value)) {
275
+ collectRequestText(nested, path === '' ? key : `${path}.${key}`, into)
276
+ }
277
+ }
278
+
279
+ /** How the refusal names the argument that came back, for a caller that cannot see the path syntax. */
280
+ function describeSubject(path: string): string {
281
+ if (path === '') return 'what this call sent'
282
+ const item = /^\[(\d+)\]$/.exec(path)
283
+ if (item) return `item ${item[1]} of the request`
284
+ return `the "${path}" argument this call sent`
285
+ }
286
+
287
+ /**
288
+ * Every name a tool answers to, so `passthroughTools` does not require the
289
+ * host to write the connector's own internal spelling.
290
+ *
291
+ * Each registration shape yields its own set, and the sets are NOT
292
+ * interchangeable — an exemption written for one shape does not exempt
293
+ * another:
294
+ *
295
+ * - A server connected directly registers `mcp_<server>_<tool>` (the
296
+ * adapter concatenates `mcp_${serverName}_${toolName}` verbatim). For
297
+ * `mcp_weather-co_lookup` with `server: 'weather-co'`, the accepted names
298
+ * are `mcp_weather-co_lookup`, `lookup`, and `weather-co:lookup`.
299
+ * - A server a plugin contributes registers
300
+ * `<plugin>__mcp__<server>__<tool>`. For
301
+ * `myplugin__mcp__weather__lookup` the accepted names are
302
+ * `myplugin__mcp__weather__lookup`, the bare tail after the last `__`
303
+ * (`lookup`), and `weather:lookup` — NOT `mcp_weather_lookup`, which is
304
+ * the direct shape's spelling of a name this tool does not have.
305
+ * - A tool of this process's own has no server and no namespace, so the
306
+ * registered name is the whole set.
307
+ *
308
+ * The bare name is what the server's manifest calls the tool — the name the
309
+ * operator has in front of them when they read its manifest — and
310
+ * `server:tool` is what a human writes. Both are derived from the registered
311
+ * name and the tool's provenance, so a caller that has only one of the two
312
+ * gets the registered name and nothing else.
313
+ */
314
+ export function passthroughToolNames(toolName: string, server?: string): readonly string[] {
315
+ const names = new Set<string>([toolName])
316
+
317
+ const separator = toolName.lastIndexOf(PLUGIN_NAMESPACE_SEPARATOR)
318
+ const pluginBare =
319
+ separator < 0 ? undefined : toolName.slice(separator + PLUGIN_NAMESPACE_SEPARATOR.length)
320
+ if (pluginBare !== undefined) names.add(pluginBare)
321
+
322
+ const directPrefix = server === undefined ? undefined : `mcp_${server}_`
323
+ const connectedBare =
324
+ directPrefix !== undefined && toolName.startsWith(directPrefix)
325
+ ? toolName.slice(directPrefix.length)
326
+ : undefined
327
+ if (connectedBare !== undefined) names.add(connectedBare)
328
+
329
+ const bare = connectedBare ?? pluginBare
330
+ if (bare !== undefined && server !== undefined) names.add(`${server}:${bare}`)
331
+
332
+ return [...names]
333
+ }
334
+
335
+ /**
336
+ * Refuse a result that is the request rather than an answer to it.
337
+ *
338
+ * #399 recorded that a screen at this boundary cannot know what a tool SHOULD
339
+ * have answered, and the issue that asked for this screen named three
340
+ * mismatches — a weather lookup for one city answering about another, a read
341
+ * of one path returning another's contents, a search returning instructions
342
+ * rather than results. **This screen decides none of the three.** The first
343
+ * two need a declaration of which argument names the subject, and the
344
+ * framework cannot infer one: for a file read the answer is the file's
345
+ * contents, which do not name the path, so the natural rule — an answer
346
+ * mentions its subject — would refuse every ordinary read. The third is what
347
+ * {@link toolResultInjectionGuardrail} already screens for at this same
348
+ * boundary, and re-implementing it here would give the two the same blind spot
349
+ * rather than the different one this adds.
350
+ *
351
+ * What is left is the one correspondence failure that needs no domain
352
+ * knowledge and cannot be an answer: **the result restates the request**. A
353
+ * tool handed a question and returning that question has answered nothing,
354
+ * whatever it is a tool for. The comparison is exact equality after
355
+ * whitespace normalisation against each string the call carried, so a result
356
+ * that says anything extra — the answer plus anything at all — passes.
357
+ *
358
+ * ## Why it judges framed results by default
359
+ *
360
+ * Because the envelope is the only marker that means *this process did not
361
+ * write this*, and the judgment it supports is the one this screen can make.
362
+ * Content arrives framed — by `wrapUntrusted`, through `frameServerResult`
363
+ * for a connector's tool result — when it came from outside: a connected
364
+ * server answered, or a host tool decided its own answer was not this
365
+ * process's to vouch for. A result that IS the request is a signal there.
366
+ *
367
+ * An earlier version judged results whose tool definition carried
368
+ * `provenance`, which is set by the MCP adapter and by nothing else. That
369
+ * predicate and this one agree on every connector result the comparison can
370
+ * act on, and they differ exactly where registration is an accident of the
371
+ * code rather than a fact about the content: the CLI's own remote Exa search
372
+ * frames its answer with this envelope and registers as a host tool, so a
373
+ * search that restated its query was out of scope while a fetch that did was
374
+ * in it. `scope` is a statement about what the screen is willing to judge, so
375
+ * it is expressed in terms of the content rather than of who mounted the
376
+ * tool.
377
+ *
378
+ * What the default leaves alone is this process's own UNFRAMED tools.
379
+ * `web_fetch` returns the page body, and a page whose body IS the URL it was
380
+ * fetched from — a redirect stub, a "you are here" placeholder, an echo
381
+ * endpoint — is a true result from a working tool that frames nothing.
382
+ * `test/…/a-result-that-does-not-answer-its-request` runs the shipped tool
383
+ * set through this screen with real calls and asserts none is refused, which
384
+ * is where that claim is checked rather than argued.
385
+ *
386
+ * `scope: 'all'` adds those too. That is the host's call, and it is the scope
387
+ * that costs something: under it, a fetch-like tool whose answer is its own
388
+ * argument is refused until the host names it in
389
+ * {@link ToolResultCorrespondenceOptions.passthroughTools}.
390
+ *
391
+ * The two other ways out were both worse. An exemption list of builtin NAMES
392
+ * inside this preset is a list that drifts — the next fetch-like tool has to
393
+ * remember to add itself, and a host tool that happens to be called
394
+ * `web_fetch` would be exempted by accident — and a subject declaration on
395
+ * every tool would be a new field on `ToolDefinition` that exists for this one
396
+ * screen.
397
+ *
398
+ * ## Three things it deliberately does not touch
399
+ *
400
+ * - **An empty result for a non-empty request.** Real, and not a signal:
401
+ * a search that matched nothing and a file with nothing in it both return
402
+ * nothing, and refusing either tells the model to stop looking when there
403
+ * was nothing to find.
404
+ * - **A shape contradicting `ToolDefinition.outputSchema`.** The schema is
405
+ * not on {@link ToolResultGuardrailContext} and is documented as shown to
406
+ * the model, never validated. Carrying it and enforcing it are changes to
407
+ * that contract, not this screen's to make.
408
+ * - **A failed call.** On `success: false` the text is a diagnostic, and
409
+ * `screenToolResult` replaces the output when it refuses — so refusing
410
+ * here would trade an echo nobody needs to catch for the error message
411
+ * the model needs to read.
412
+ *
413
+ * The comparison runs on the frame as well as on the text: a connector's
414
+ * result reaches a screen already wrapped by `wrapUntrusted`, so comparing
415
+ * the raw output would never match the one case this is most worth having —
416
+ * a connected server returning the request. Both the whole output and, when
417
+ * it is one wrapped block, the body inside it are compared.
418
+ *
419
+ * Detection is partial and this says so rather than implying coverage: an
420
+ * answer about the wrong subject, an answer that is plausible prose, and a
421
+ * restatement shorter than {@link MIN_RESTATEMENT_LENGTH} all pass.
422
+ */
423
+ export function toolResultCorrespondenceGuardrail(
424
+ options: ToolResultCorrespondenceOptions = {},
425
+ ): NamedGuardrail<ToolResultGuardrail> {
426
+ const passthrough = new Set(options.passthroughTools ?? [])
427
+ const scope = options.scope ?? 'framed'
428
+
429
+ return {
430
+ name: 'tool-result-correspondence',
431
+ check: ({ toolName, input, output, success, provenance }): ToolResultVerdict => {
432
+ if (!success) return { action: 'pass' }
433
+ // `output` is typed as a string and a tool that ignores its own
434
+ // contract can hand back something else. Nothing here can compare
435
+ // a number against a request, and a screen that throws fails
436
+ // CLOSED — so the one thing this must not do is refuse a result
437
+ // because it could not read it.
438
+ if (typeof output !== 'string') return { action: 'pass' }
439
+
440
+ // The scope, and the same reader the comparison below uses rather
441
+ // than a second "looks framed" test that could disagree with it.
442
+ //
443
+ // This is a superset of the `provenance`-based predicate it
444
+ // replaced, for every result the comparison can act on. The only
445
+ // tool definition carrying `provenance` is the MCP adapter's, and
446
+ // `frameServerResult` frames every non-empty text it returns — so
447
+ // a connected result that reaches the comparison at all carries a
448
+ // legible frame. The one case the new predicate drops is a
449
+ // connected result with an EMPTY output, which the comparison
450
+ // skipped anyway (`normalizeText('')` is `''`, and an empty string
451
+ // is never a request of the minimum length).
452
+ const body = untrustedEnvelopeBody(output)
453
+ if (scope !== 'all' && body === undefined) return { action: 'pass' }
454
+
455
+ if (
456
+ passthrough.size > 0 &&
457
+ passthroughToolNames(toolName, provenance?.server).some((name) => passthrough.has(name))
458
+ ) {
459
+ return { action: 'pass' }
460
+ }
461
+
462
+ const request: RequestText[] = []
463
+ collectRequestText(input, '', request)
464
+ if (request.length === 0) return { action: 'pass' }
465
+
466
+ // The whole output, and the body when the output is one wrapped
467
+ // block. The two are never the same string — the frame carries the
468
+ // tags — so this is two comparisons, not the same one twice, and
469
+ // the refusal says which one matched.
470
+ for (const [text, insideFrame] of [
471
+ [normalizeText(output), false],
472
+ [normalizeText(body ?? ''), true],
473
+ ] as const) {
474
+ if (text === '') continue
475
+ const restated = request.find((entry) => entry.text === text)
476
+ if (!restated) continue
477
+ // The reason names the tool, and says what the comparison
478
+ // actually did. Both were wrong before: a host reading the
479
+ // transcript is the person who can exempt this tool, and
480
+ // "verbatim" was a false claim about a comparison that ran
481
+ // after whitespace normalisation.
482
+ const matched = insideFrame
483
+ ? `the text inside the untrusted frame from "${toolName}"`
484
+ : `the result from "${toolName}"`
485
+ return {
486
+ action: 'refuse',
487
+ reason: `${matched} is ${describeSubject(restated.path)}, whitespace-normalised, so it is the request rather than an answer to it`,
488
+ }
489
+ }
490
+
491
+ return { action: 'pass' }
492
+ },
493
+ }
494
+ }
495
+
496
+ /**
497
+ * The screens a run installs on itself when its host configured none.
498
+ *
499
+ * A DEFAULT, and the reason it is here rather than on
500
+ * {@link ToolRegistryConfig.resultGuardrails}: a host assembles a registry
501
+ * and hands it to `runAgent`, so a registry-constructor default would be the
502
+ * host's to write and this repository's default would reach nobody. The run
503
+ * is the thing that has to carry it, and a run that wants none says so with
504
+ * an empty array — which is the escape hatch, and it exists precisely because
505
+ * the screen below can refuse a result.
506
+ *
507
+ * One screen, scoped to results framed as untrusted. See the preset for why
508
+ * the scope is what it is: an unframed host tool whose answer IS the request —
509
+ * `web_fetch` returning a page whose body is its own URL — is an ordinary
510
+ * result, and a default that refuses those is a default that breaks a working
511
+ * tool.
512
+ *
513
+ * **A default that refuses a legitimate result is a default that gets turned
514
+ * off, so the exemption has to be reachable from wherever the screen is
515
+ * installed.** `toolResultCorrespondenceGuardrail({ passthroughTools })` is
516
+ * how a host writing SDK code says so; a run's `toolResultGuardrails` array
517
+ * is where it substitutes its own configured instance. An application that
518
+ * ships the default without also shipping a way to name an exception has
519
+ * shipped a switch with one position.
520
+ *
521
+ * Frozen, because it is handed to callers who may want to extend it:
522
+ * `[...DEFAULT_TOOL_RESULT_GUARDRAILS, myScreen()]`. A caller that could
523
+ * mutate it would be mutating the default for every other run in the process.
524
+ */
525
+ export const DEFAULT_TOOL_RESULT_GUARDRAILS: readonly ToolResultGuardrailSpec[] = Object.freeze([
526
+ toolResultCorrespondenceGuardrail(),
527
+ ])
@@ -363,6 +363,20 @@ export interface QueryParams {
363
363
  * agent decides the rest is worth re-reading. Set `0` to disable.
364
364
  */
365
365
  maxToolOutputChars?: number
366
+ /**
367
+ * Screens to run against every tool result, where the registry was not
368
+ * built with its own.
369
+ *
370
+ * This is the run's half of a boundary whose only other door is the
371
+ * registry constructor — and a registry is usually the HOST's, assembled
372
+ * before the run exists, so a run-config option is the only way a run
373
+ * screens a registry it did not build. A registry built WITH
374
+ * `resultGuardrails` states its own policy and wins, `[]` included.
375
+ *
376
+ * Absent installs {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array
377
+ * installs none, which is how a caller turns the default off.
378
+ */
379
+ toolResultGuardrails?: readonly import('../../types/guardrail/index.js').ToolResultGuardrailSpec[]
366
380
  /**
367
381
  * Smaller preview for text that exceeded maxToolOutputChars, after its full
368
382
  * host output and integrity manifest have been saved. Unset/0 keeps the old
@@ -1841,6 +1855,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
1841
1855
  ...(params.maxToolOutputChars !== undefined
1842
1856
  ? { maxToolOutputChars: params.maxToolOutputChars }
1843
1857
  : {}),
1858
+ ...(params.toolResultGuardrails !== undefined
1859
+ ? { toolResultGuardrails: params.toolResultGuardrails }
1860
+ : {}),
1844
1861
  ...(params.retainedToolPreviewChars !== undefined
1845
1862
  ? { retainedToolPreviewChars: params.retainedToolPreviewChars }
1846
1863
  : {}),
@@ -49,6 +49,8 @@ export interface ToolingBootstrapConfig {
49
49
  maxToolCalls?: number
50
50
  readToolCallBudgetEvents?: () => Promise<readonly RunEvent[]>
51
51
  maxToolOutputChars?: number
52
+ /** See `QueryParams.toolResultGuardrails`. Absent installs the shipped default; a registry's own win. */
53
+ toolResultGuardrails?: readonly import('../../types/guardrail/index.js').ToolResultGuardrailSpec[]
52
54
  retainedToolPreviewChars?: number
53
55
  maxToolContentBytes?: number
54
56
  captureRunEvidence?: import('../../types/tool/index.js').ToolContext['captureRunEvidence']
@@ -104,6 +106,9 @@ export class ToolingBootstrap {
104
106
  ...(config.maxToolOutputChars !== undefined
105
107
  ? { maxToolOutputChars: config.maxToolOutputChars }
106
108
  : {}),
109
+ ...(config.toolResultGuardrails !== undefined
110
+ ? { toolResultGuardrails: config.toolResultGuardrails }
111
+ : {}),
107
112
  ...(config.retainedToolPreviewChars !== undefined
108
113
  ? { retainedToolPreviewChars: config.retainedToolPreviewChars }
109
114
  : {}),