@namzu/sdk 40.0.0 → 42.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +236 -0
- package/dist/agents/ReactiveAgent.d.ts.map +1 -1
- package/dist/agents/ReactiveAgent.js +3 -0
- package/dist/agents/ReactiveAgent.js.map +1 -1
- package/dist/agents/SupervisorAgent.d.ts.map +1 -1
- package/dist/agents/SupervisorAgent.js +11 -0
- package/dist/agents/SupervisorAgent.js.map +1 -1
- package/dist/agents/runAgent.d.ts +14 -0
- package/dist/agents/runAgent.d.ts.map +1 -1
- package/dist/agents/runAgent.js +3 -0
- package/dist/agents/runAgent.js.map +1 -1
- package/dist/bridge/a2a/mapper.d.ts.map +1 -1
- package/dist/bridge/a2a/mapper.js +8 -0
- package/dist/bridge/a2a/mapper.js.map +1 -1
- package/dist/bridge/sse/mapper.d.ts.map +1 -1
- package/dist/bridge/sse/mapper.js +11 -0
- package/dist/bridge/sse/mapper.js.map +1 -1
- package/dist/connector/index.d.ts +2 -2
- package/dist/connector/index.d.ts.map +1 -1
- package/dist/connector/index.js +1 -1
- package/dist/connector/index.js.map +1 -1
- package/dist/connector/mcp/adapter.d.ts.map +1 -1
- package/dist/connector/mcp/adapter.js +92 -4
- package/dist/connector/mcp/adapter.js.map +1 -1
- package/dist/connector/mcp/audio-admission.d.ts +17 -0
- package/dist/connector/mcp/audio-admission.d.ts.map +1 -0
- package/dist/connector/mcp/audio-admission.js +171 -0
- package/dist/connector/mcp/audio-admission.js.map +1 -0
- package/dist/connector/mcp/client.d.ts +252 -1
- package/dist/connector/mcp/client.d.ts.map +1 -1
- package/dist/connector/mcp/client.js +611 -39
- package/dist/connector/mcp/client.js.map +1 -1
- package/dist/connector/mcp/envelope.d.ts +91 -0
- package/dist/connector/mcp/envelope.d.ts.map +1 -0
- package/dist/connector/mcp/envelope.js +173 -0
- package/dist/connector/mcp/envelope.js.map +1 -0
- package/dist/connector/mcp/era.d.ts +130 -0
- package/dist/connector/mcp/era.d.ts.map +1 -0
- package/dist/connector/mcp/era.js +304 -0
- package/dist/connector/mcp/era.js.map +1 -0
- package/dist/connector/mcp/errors.d.ts +106 -0
- package/dist/connector/mcp/errors.d.ts.map +1 -0
- package/dist/connector/mcp/errors.js +154 -0
- package/dist/connector/mcp/errors.js.map +1 -0
- package/dist/connector/mcp/http-sse.d.ts +11 -0
- package/dist/connector/mcp/http-sse.d.ts.map +1 -1
- package/dist/connector/mcp/http-sse.js +21 -6
- package/dist/connector/mcp/http-sse.js.map +1 -1
- package/dist/connector/mcp/index.d.ts +7 -0
- package/dist/connector/mcp/index.d.ts.map +1 -1
- package/dist/connector/mcp/index.js +10 -0
- package/dist/connector/mcp/index.js.map +1 -1
- package/dist/connector/mcp/streamable-http.d.ts +83 -0
- package/dist/connector/mcp/streamable-http.d.ts.map +1 -1
- package/dist/connector/mcp/streamable-http.js +177 -11
- package/dist/connector/mcp/streamable-http.js.map +1 -1
- package/dist/connector/mcp/x-mcp-header.d.ts +56 -0
- package/dist/connector/mcp/x-mcp-header.d.ts.map +1 -0
- package/dist/connector/mcp/x-mcp-header.js +254 -0
- package/dist/connector/mcp/x-mcp-header.js.map +1 -0
- package/dist/constants/mcp/index.d.ts +123 -15
- package/dist/constants/mcp/index.d.ts.map +1 -1
- package/dist/constants/mcp/index.js +135 -16
- package/dist/constants/mcp/index.js.map +1 -1
- package/dist/manager/agent/lifecycle.d.ts.map +1 -1
- package/dist/manager/agent/lifecycle.js +43 -0
- package/dist/manager/agent/lifecycle.js.map +1 -1
- package/dist/prompt/coding-agent-doctrine.d.ts +20 -0
- package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
- package/dist/prompt/coding-agent-doctrine.js +19 -3
- package/dist/prompt/coding-agent-doctrine.js.map +1 -1
- package/dist/prompt/index.d.ts +1 -1
- package/dist/prompt/index.d.ts.map +1 -1
- package/dist/prompt/index.js +1 -1
- package/dist/prompt/index.js.map +1 -1
- package/dist/public-runtime.d.ts +7 -4
- package/dist/public-runtime.d.ts.map +1 -1
- package/dist/public-runtime.js +16 -4
- package/dist/public-runtime.js.map +1 -1
- package/dist/public-tools.d.ts +1 -1
- package/dist/public-tools.d.ts.map +1 -1
- package/dist/public-tools.js +4 -2
- package/dist/public-tools.js.map +1 -1
- package/dist/registry/tool/execute.d.ts.map +1 -1
- package/dist/registry/tool/execute.js +10 -1
- package/dist/registry/tool/execute.js.map +1 -1
- package/dist/runtime/bidi/session.d.ts +11 -0
- package/dist/runtime/bidi/session.d.ts.map +1 -1
- package/dist/runtime/bidi/session.js +2 -0
- package/dist/runtime/bidi/session.js.map +1 -1
- package/dist/runtime/query/executor.d.ts +6 -0
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +6 -0
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/guardrail-presets.d.ts +187 -1
- package/dist/runtime/query/guardrail-presets.d.ts.map +1 -1
- package/dist/runtime/query/guardrail-presets.js +298 -0
- package/dist/runtime/query/guardrail-presets.js.map +1 -1
- package/dist/runtime/query/index.d.ts +14 -0
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +3 -0
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/tooling.d.ts +2 -0
- package/dist/runtime/query/tooling.d.ts.map +1 -1
- package/dist/runtime/query/tooling.js +3 -0
- package/dist/runtime/query/tooling.js.map +1 -1
- package/dist/sandbox/provider/local.d.ts.map +1 -1
- package/dist/sandbox/provider/local.js +46 -3
- package/dist/sandbox/provider/local.js.map +1 -1
- package/dist/scheduler/local.d.ts.map +1 -1
- package/dist/scheduler/local.js +8 -0
- package/dist/scheduler/local.js.map +1 -1
- package/dist/store/run/disk.d.ts +35 -1
- package/dist/store/run/disk.d.ts.map +1 -1
- package/dist/store/run/disk.js +100 -0
- package/dist/store/run/disk.js.map +1 -1
- package/dist/tools/coordinator/agent.d.ts.map +1 -1
- package/dist/tools/coordinator/agent.js +17 -2
- package/dist/tools/coordinator/agent.js.map +1 -1
- package/dist/tools/coordinator/index.d.ts.map +1 -1
- package/dist/tools/coordinator/index.js +17 -3
- package/dist/tools/coordinator/index.js.map +1 -1
- package/dist/tools/untrusted-envelope.d.ts +35 -0
- package/dist/tools/untrusted-envelope.d.ts.map +1 -1
- package/dist/tools/untrusted-envelope.js +91 -3
- package/dist/tools/untrusted-envelope.js.map +1 -1
- package/dist/types/agent/base.d.ts +23 -0
- package/dist/types/agent/base.d.ts.map +1 -1
- package/dist/types/agent/scheduler.d.ts +20 -0
- package/dist/types/agent/scheduler.d.ts.map +1 -1
- package/dist/types/agent/task.d.ts +39 -0
- package/dist/types/agent/task.d.ts.map +1 -1
- package/dist/types/connector/mcp.d.ts +205 -0
- package/dist/types/connector/mcp.d.ts.map +1 -1
- package/dist/types/run/events.d.ts +56 -0
- package/dist/types/run/events.d.ts.map +1 -1
- package/dist/types/run/events.js.map +1 -1
- package/dist/types/run/store.d.ts +41 -0
- package/dist/types/run/store.d.ts.map +1 -1
- package/dist/types/sandbox/index.d.ts +68 -1
- package/dist/types/sandbox/index.d.ts.map +1 -1
- package/dist/types/sandbox/index.js.map +1 -1
- package/dist/types/tool/index.d.ts +19 -0
- package/dist/types/tool/index.d.ts.map +1 -1
- package/dist/types/tool/index.js.map +1 -1
- package/package.json +1 -1
- package/src/agents/ReactiveAgent.ts +3 -0
- package/src/agents/SupervisorAgent.ts +11 -0
- package/src/agents/runAgent.ts +18 -0
- package/src/bridge/a2a/mapper.ts +8 -0
- package/src/bridge/sse/mapper.ts +11 -0
- package/src/connector/index.ts +27 -0
- package/src/connector/mcp/adapter.ts +103 -4
- package/src/connector/mcp/audio-admission.ts +173 -0
- package/src/connector/mcp/client.ts +694 -45
- package/src/connector/mcp/envelope.ts +235 -0
- package/src/connector/mcp/era.ts +400 -0
- package/src/connector/mcp/errors.ts +171 -0
- package/src/connector/mcp/http-sse.ts +23 -6
- package/src/connector/mcp/index.ts +37 -0
- package/src/connector/mcp/streamable-http.ts +199 -11
- package/src/connector/mcp/x-mcp-header.ts +322 -0
- package/src/constants/mcp/index.ts +145 -16
- package/src/manager/agent/lifecycle.ts +51 -0
- package/src/prompt/coding-agent-doctrine.ts +31 -4
- package/src/prompt/index.ts +1 -0
- package/src/public-runtime.ts +42 -0
- package/src/public-tools.ts +8 -2
- package/src/registry/tool/execute.ts +9 -1
- package/src/runtime/bidi/session.ts +13 -0
- package/src/runtime/query/executor.ts +13 -0
- package/src/runtime/query/guardrail-presets.ts +356 -0
- package/src/runtime/query/index.ts +17 -0
- package/src/runtime/query/tooling.ts +5 -0
- package/src/sandbox/provider/local.ts +45 -2
- package/src/scheduler/local.ts +8 -0
- package/src/store/run/disk.ts +108 -0
- package/src/tools/coordinator/agent.ts +17 -2
- package/src/tools/coordinator/index.ts +17 -3
- package/src/tools/untrusted-envelope.ts +94 -3
- package/src/types/agent/base.ts +24 -0
- package/src/types/agent/scheduler.ts +21 -0
- package/src/types/agent/task.ts +41 -0
- package/src/types/connector/mcp.ts +205 -1
- package/src/types/run/events.ts +56 -0
- package/src/types/run/store.ts +42 -0
- package/src/types/sandbox/index.ts +69 -1
- package/src/types/tool/index.ts +20 -0
package/src/public-runtime.ts
CHANGED
|
@@ -725,12 +725,29 @@ export {
|
|
|
725
725
|
hasDrift,
|
|
726
726
|
HttpSseTransport,
|
|
727
727
|
HybridExecutionContext,
|
|
728
|
+
buildEnvelope,
|
|
729
|
+
classifyModernHttpFailure,
|
|
730
|
+
createMcpEraCache,
|
|
731
|
+
decodeResult,
|
|
732
|
+
defaultMcpEraCache,
|
|
733
|
+
encodeMcpHeaderValue,
|
|
734
|
+
isHeaderMismatchError,
|
|
735
|
+
isMissingRequiredClientCapabilityError,
|
|
736
|
+
isRecognizedModernError,
|
|
737
|
+
isResourceNotFoundError,
|
|
738
|
+
isUnsupportedProtocolVersionError,
|
|
728
739
|
LocalExecutionContext,
|
|
740
|
+
mcpEraCacheKey,
|
|
741
|
+
MCPHttpStatusError,
|
|
742
|
+
MCPInputRequiredError,
|
|
743
|
+
MCPInvalidResultTypeError,
|
|
729
744
|
MCPClient,
|
|
730
745
|
MCPMethodNotFound,
|
|
731
746
|
MCPConnectorBridge,
|
|
747
|
+
MCPProtocolError,
|
|
732
748
|
MCPServer,
|
|
733
749
|
MCPToolDiscovery,
|
|
750
|
+
resolveMcpEra,
|
|
734
751
|
mcpPromptToToolDefinition,
|
|
735
752
|
renderPromptMessages,
|
|
736
753
|
mcpJsonSchemaToZod,
|
|
@@ -749,10 +766,20 @@ export {
|
|
|
749
766
|
toolDefinitionToMCPTool,
|
|
750
767
|
toolResultToMCPToolResult,
|
|
751
768
|
toolsHash,
|
|
769
|
+
validateMcpHeaderAnnotations,
|
|
752
770
|
WebhookConnector,
|
|
753
771
|
zodToMCPJsonSchema,
|
|
754
772
|
} from './connector/index.js'
|
|
755
773
|
export type {
|
|
774
|
+
McpEnvelope,
|
|
775
|
+
McpEnvelopeInput,
|
|
776
|
+
McpEraProbe,
|
|
777
|
+
McpEraProbeAnswer,
|
|
778
|
+
McpEraResolution,
|
|
779
|
+
McpEraResolutionInput,
|
|
780
|
+
McpHeaderAnnotationVerdict,
|
|
781
|
+
McpParamHeaderBinding,
|
|
782
|
+
MCPDecodedResult,
|
|
756
783
|
MCPToolDiscoveryOptions,
|
|
757
784
|
MCPToolDrift,
|
|
758
785
|
MCPToolPolicy,
|
|
@@ -1165,8 +1192,21 @@ export type { TokenUsageSample } from './telemetry/metrics.js'
|
|
|
1165
1192
|
export {
|
|
1166
1193
|
promptInjectionGuardrail,
|
|
1167
1194
|
secretRedactionGuardrail,
|
|
1195
|
+
toolResultCorrespondenceGuardrail,
|
|
1168
1196
|
toolResultInjectionGuardrail,
|
|
1169
1197
|
} from './runtime/query/guardrail-presets.js'
|
|
1198
|
+
// Which names a `passthroughTools` list has to contain for a given tool. A
|
|
1199
|
+
// caller that REPORTS on such a list — the CLI checks an operator's names
|
|
1200
|
+
// against the tools its registry actually holds, so a name that matches
|
|
1201
|
+
// nothing is said out loud rather than silently ignored — has to derive them
|
|
1202
|
+
// the way the screen does rather than keep a second copy of the rule: a name
|
|
1203
|
+
// the reporter accepts and the screen does not is an exemption the operator
|
|
1204
|
+
// believes is in force.
|
|
1205
|
+
export { passthroughToolNames } from './runtime/query/guardrail-presets.js'
|
|
1206
|
+
// What a run installs when its host configured no screens. Exported because a
|
|
1207
|
+
// caller who wants to keep the default AND add to it has to be able to name
|
|
1208
|
+
// it: `[...DEFAULT_TOOL_RESULT_GUARDRAILS, myScreen()]`.
|
|
1209
|
+
export { DEFAULT_TOOL_RESULT_GUARDRAILS } from './runtime/query/guardrail-presets.js'
|
|
1170
1210
|
// Thrown by a tool-result screen that returned `halt`. Exported because a
|
|
1171
1211
|
// host has to be able to tell it from an ordinary failure — that is the
|
|
1172
1212
|
// entire difference between the two refusal outcomes.
|
|
@@ -1229,6 +1269,7 @@ export {
|
|
|
1229
1269
|
export {
|
|
1230
1270
|
CODING_AGENT_DELEGATION_DOCTRINE,
|
|
1231
1271
|
CODING_AGENT_DOCTRINE_CONTRIBUTION_ID,
|
|
1272
|
+
CODING_AGENT_ORCHESTRATE_DOCTRINE,
|
|
1232
1273
|
CODING_AGENT_WORKING_DOCTRINE,
|
|
1233
1274
|
PLAN_MODE_DOCTRINE,
|
|
1234
1275
|
PromptContributionCollisionError,
|
|
@@ -1334,6 +1375,7 @@ export type { ToolCatalogFromRegistryOptions } from './registry/toolset/catalog.
|
|
|
1334
1375
|
export type { MockBidiScript, MockBidiSession } from './runtime/bidi/mock.js'
|
|
1335
1376
|
export type { BidiRun, BidiRunParams } from './runtime/bidi/session.js'
|
|
1336
1377
|
export type { SecretRedactionOptions } from './runtime/query/guardrail-presets.js'
|
|
1378
|
+
export type { ToolResultCorrespondenceOptions } from './runtime/query/guardrail-presets.js'
|
|
1337
1379
|
export type { ListCheckpointsInput } from './runtime/query/replay/list.js'
|
|
1338
1380
|
export type {
|
|
1339
1381
|
PrepareReplayInput,
|
package/src/public-tools.ts
CHANGED
|
@@ -17,8 +17,14 @@ export { defineTool } from './tools/defineTool.js'
|
|
|
17
17
|
// this file needed them.
|
|
18
18
|
export { isWithin, resolveWithin, resolveWithinReal } from './tools/paths.js'
|
|
19
19
|
// A host that surfaces its own untrusted content to a model needs the same
|
|
20
|
-
// framing the kernel applies to connector prompts and delegated results
|
|
21
|
-
|
|
20
|
+
// framing the kernel applies to connector prompts and delegated results — and
|
|
21
|
+
// the reader for it, because a screen that judges a result has to reach past
|
|
22
|
+
// the frame without re-spelling the tag.
|
|
23
|
+
export {
|
|
24
|
+
neutralizeEnvelopeDelimiter,
|
|
25
|
+
untrustedEnvelopeBody,
|
|
26
|
+
wrapUntrusted,
|
|
27
|
+
} from './tools/untrusted-envelope.js'
|
|
22
28
|
export { filterReadOnlyTools, filterToolsNamed } from './tools/roster.js'
|
|
23
29
|
export type { UntrustedEnvelope } from './tools/untrusted-envelope.js'
|
|
24
30
|
|
|
@@ -679,6 +679,7 @@ Executable tool names, descriptions, and JSON input schemas are attached through
|
|
|
679
679
|
try {
|
|
680
680
|
this.log.debug('Executing tool', { 'namzu.tool.name': toolName })
|
|
681
681
|
const startedAt = Date.now()
|
|
682
|
+
const runResultGuardrails = context.toolResultGuardrails
|
|
682
683
|
const produced = await tool.execute(finalInput, context)
|
|
683
684
|
// Screened here, which is the only place a result can be
|
|
684
685
|
// examined before anything acts on it: the executor applies
|
|
@@ -689,7 +690,14 @@ Executable tool names, descriptions, and JSON input schemas are attached through
|
|
|
689
690
|
// the server's name, and a screen reading only the value
|
|
690
691
|
// cannot use that.
|
|
691
692
|
const result = await screenToolResult(
|
|
692
|
-
|
|
693
|
+
// Explicit configuration wins, at whichever boundary it
|
|
694
|
+
// was made. A registry built WITH `resultGuardrails`
|
|
695
|
+
// has stated its policy — including `[]`, which means
|
|
696
|
+
// none — and a run must not overrule it. A registry
|
|
697
|
+
// built without one declared none, so the run's apply;
|
|
698
|
+
// that is the ordinary case, since a host assembles a
|
|
699
|
+
// registry and hands it to a run it does not own.
|
|
700
|
+
this.resultGuardrails ?? runResultGuardrails,
|
|
693
701
|
produced,
|
|
694
702
|
{
|
|
695
703
|
toolName,
|
|
@@ -5,12 +5,14 @@ import type {
|
|
|
5
5
|
BidiRunEvent,
|
|
6
6
|
BidiSession,
|
|
7
7
|
} from '../../types/bidi/index.js'
|
|
8
|
+
import type { ToolResultGuardrailSpec } from '../../types/guardrail/index.js'
|
|
8
9
|
import type { RunId } from '../../types/ids/index.js'
|
|
9
10
|
import type { ToolContext, ToolRegistryContract } from '../../types/tool/index.js'
|
|
10
11
|
import { toErrorMessage } from '../../utils/error.js'
|
|
11
12
|
import { generateRunId } from '../../utils/id.js'
|
|
12
13
|
import { SCOPE_ATTRIBUTE } from '../../utils/log/types.js'
|
|
13
14
|
import { type Logger, resolveLogger } from '../../utils/logger.js'
|
|
15
|
+
import { DEFAULT_TOOL_RESULT_GUARDRAILS } from '../query/guardrail-presets.js'
|
|
14
16
|
|
|
15
17
|
const DEFAULT_CLOSE_TIMEOUT_MS = 5_000
|
|
16
18
|
const MAX_TIMER_DELAY_MS = 2_147_483_647
|
|
@@ -78,6 +80,16 @@ export interface BidiRunParams {
|
|
|
78
80
|
readonly log?: Logger
|
|
79
81
|
/** Overrides the generated id, so a host can correlate its own. */
|
|
80
82
|
readonly runId?: RunId
|
|
83
|
+
/**
|
|
84
|
+
* Screens for the results this session's tools produce. Absent installs
|
|
85
|
+
* {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array installs none.
|
|
86
|
+
*
|
|
87
|
+
* Here rather than nowhere because this path builds its OWN tool context:
|
|
88
|
+
* a duplex session executes the tools the model asks for, and its results
|
|
89
|
+
* reach a model just as a turn's do. A registry built with
|
|
90
|
+
* `resultGuardrails` still wins, as it does on the query path.
|
|
91
|
+
*/
|
|
92
|
+
readonly toolResultGuardrails?: readonly ToolResultGuardrailSpec[]
|
|
81
93
|
}
|
|
82
94
|
|
|
83
95
|
export interface BidiRun {
|
|
@@ -230,6 +242,7 @@ export async function startBidiRun(params: BidiRunParams): Promise<BidiRun> {
|
|
|
230
242
|
env: params.env ?? {},
|
|
231
243
|
log: (level, message) => log[level](message),
|
|
232
244
|
toolUseId: call.id,
|
|
245
|
+
toolResultGuardrails: params.toolResultGuardrails ?? DEFAULT_TOOL_RESULT_GUARDRAILS,
|
|
233
246
|
}
|
|
234
247
|
const result = await params.tools.execute(call.name, input, context)
|
|
235
248
|
output = result.success ? result.output : (result.error ?? 'the tool failed')
|
|
@@ -13,6 +13,7 @@ import { renderToolSchema } from '../../registry/tool/schema.js'
|
|
|
13
13
|
import type { ActivityStore } from '../../store/activity/memory.js'
|
|
14
14
|
import { SKILL_TOOL_NAME } from '../../tools/builtins/skill.js'
|
|
15
15
|
import { createFileReadTracker } from '../../tools/file-read-tracker.js'
|
|
16
|
+
import type { ToolResultGuardrailSpec } from '../../types/guardrail/index.js'
|
|
16
17
|
import type { RunId, ToolUseId } from '../../types/ids/index.js'
|
|
17
18
|
import type { InvocationState } from '../../types/invocation/index.js'
|
|
18
19
|
import {
|
|
@@ -52,6 +53,7 @@ import { compressShellOutput } from '../../utils/shell-compress.js'
|
|
|
52
53
|
import { type BackgroundJobRegistry, type JobProcess, bindOwner } from '../jobs/registry.js'
|
|
53
54
|
import { describeVisibleFileEvidence } from './file-evidence-context.js'
|
|
54
55
|
import { seedObservationLedger } from './file-evidence-seed.js'
|
|
56
|
+
import { DEFAULT_TOOL_RESULT_GUARDRAILS } from './guardrail-presets.js'
|
|
55
57
|
import { skippedToolResultText } from './plugin-hooks.js'
|
|
56
58
|
import type { ToolResultObservation } from './project-instructions.js'
|
|
57
59
|
import { ToolCallBudget, assertMaxToolCalls } from './tool-call-budget.js'
|
|
@@ -393,6 +395,12 @@ export interface ToolExecutorConfig {
|
|
|
393
395
|
/** See QueryParams.retainedToolPreviewChars; applies to the recorded host output. */
|
|
394
396
|
retainedToolPreviewChars?: number
|
|
395
397
|
|
|
398
|
+
/**
|
|
399
|
+
* See QueryParams.toolResultGuardrails. Absent installs
|
|
400
|
+
* {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array installs none.
|
|
401
|
+
*/
|
|
402
|
+
toolResultGuardrails?: readonly ToolResultGuardrailSpec[]
|
|
403
|
+
|
|
396
404
|
/**
|
|
397
405
|
* Cap on the RICH channel of a single tool result, in base64
|
|
398
406
|
* characters. `0` or absent disables it.
|
|
@@ -1335,6 +1343,11 @@ export class ToolExecutor {
|
|
|
1335
1343
|
this.skillScope = { ...scope, adoptedInBatch: this.batchCounter }
|
|
1336
1344
|
},
|
|
1337
1345
|
maxToolOutputChars: this.config.maxToolOutputChars ?? DEFAULT_MAX_TOOL_OUTPUT_CHARS,
|
|
1346
|
+
// The run's screens, defaulted HERE rather than on the registry: a
|
|
1347
|
+
// host builds the registry and hands it over, so a registry-side
|
|
1348
|
+
// default is the host's to write and the shipped one reaches
|
|
1349
|
+
// nobody. `[]` survives the `??` and is how a run says "none".
|
|
1350
|
+
toolResultGuardrails: this.config.toolResultGuardrails ?? DEFAULT_TOOL_RESULT_GUARDRAILS,
|
|
1338
1351
|
...(this.config.skills ? { skills: this.config.skills } : {}),
|
|
1339
1352
|
...(this.config.web ? { web: this.config.web } : {}),
|
|
1340
1353
|
// The SAME registry and the SAME context a model-issued call
|
|
@@ -1,9 +1,12 @@
|
|
|
1
|
+
import { PLUGIN_NAMESPACE_SEPARATOR } from '../../constants/plugin/index.js'
|
|
1
2
|
import { OUTPUT_SECRET_PATTERNS } from '../../constants/secret-patterns.js'
|
|
3
|
+
import { untrustedEnvelopeBody } from '../../tools/untrusted-envelope.js'
|
|
2
4
|
import type {
|
|
3
5
|
GuardrailVerdict,
|
|
4
6
|
NamedGuardrail,
|
|
5
7
|
OutputGuardrail,
|
|
6
8
|
ToolResultGuardrail,
|
|
9
|
+
ToolResultGuardrailSpec,
|
|
7
10
|
ToolResultVerdict,
|
|
8
11
|
} from '../../types/guardrail/index.js'
|
|
9
12
|
|
|
@@ -169,3 +172,356 @@ export function toolResultInjectionGuardrail(): NamedGuardrail<ToolResultGuardra
|
|
|
169
172
|
},
|
|
170
173
|
}
|
|
171
174
|
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* Requests shorter than this are not compared against the result.
|
|
178
|
+
*
|
|
179
|
+
* A one-word echo cannot be told from a one-word answer, and the shorter the
|
|
180
|
+
* value the likelier it is to recur in a legitimate result by coincidence:
|
|
181
|
+
* `ls` of a directory holding one entry called `src`, called with
|
|
182
|
+
* `{ path: "src" }`, returns `src`. The value of catching a five-character
|
|
183
|
+
* restatement is nil — nothing rides on it — so the comparison starts where
|
|
184
|
+
* a restatement actually means something.
|
|
185
|
+
*
|
|
186
|
+
* Counted in CODE POINTS, not UTF-16 units. Eight emoji are eight characters
|
|
187
|
+
* to every reader of this file and sixteen to `String.length`, and a floor
|
|
188
|
+
* that lets them through while the docblock says it excludes short values is
|
|
189
|
+
* a floor that is not doing what it says.
|
|
190
|
+
*/
|
|
191
|
+
const MIN_RESTATEMENT_LENGTH = 16
|
|
192
|
+
|
|
193
|
+
/** Whether `text` is at least `minimum` code points long, without counting all of them. */
|
|
194
|
+
function isAtLeastCodePoints(text: string, minimum: number): boolean {
|
|
195
|
+
let count = 0
|
|
196
|
+
for (const _character of text) {
|
|
197
|
+
count += 1
|
|
198
|
+
if (count >= minimum) return true
|
|
199
|
+
}
|
|
200
|
+
return false
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
export interface ToolResultCorrespondenceOptions {
|
|
204
|
+
/**
|
|
205
|
+
* Which results this screen judges.
|
|
206
|
+
*
|
|
207
|
+
* `'framed'` — the default — is the results that carry the untrusted
|
|
208
|
+
* envelope: content this process did not author, marked as such by
|
|
209
|
+
* `wrapUntrusted`. For a connected server's tool that is every result the
|
|
210
|
+
* adapter returns (see `frameServerResult`), and it is also a host tool
|
|
211
|
+
* that framed its own result the same way — which is the point, and what
|
|
212
|
+
* an earlier `provenance`-based predicate got wrong: the CLI's own remote
|
|
213
|
+
* Exa search frames its answer and was left unscreened by accident of
|
|
214
|
+
* registration, while a fetch was screened by the same accident.
|
|
215
|
+
*
|
|
216
|
+
* `'all'` adds this process's own unframed tools. That is the host's call
|
|
217
|
+
* and not a safe default: a host that asks for it owns the consequences
|
|
218
|
+
* and names its own exceptions in {@link passthroughTools}.
|
|
219
|
+
*/
|
|
220
|
+
readonly scope?: 'framed' | 'all'
|
|
221
|
+
|
|
222
|
+
/**
|
|
223
|
+
* Tools whose answer IS the request, and must not be refused for saying
|
|
224
|
+
* so: a validator returning what it validated, a normaliser returning the
|
|
225
|
+
* normalised form, a search that repeats its query when it found nothing,
|
|
226
|
+
* a fetch whose page body is its own URL.
|
|
227
|
+
*
|
|
228
|
+
* This is the escape hatch the check needs rather than a tuning knob. The
|
|
229
|
+
* screen is right about a tool that was asked a question and handed the
|
|
230
|
+
* question back; it is wrong about a tool whose purpose is to hand
|
|
231
|
+
* something back, and nothing on the context distinguishes the two. The
|
|
232
|
+
* false positive is the way this control dies — the host switches the
|
|
233
|
+
* screen off and it then protects nothing — so the exemption has to be
|
|
234
|
+
* cheap enough to be the first thing reached for.
|
|
235
|
+
*
|
|
236
|
+
* A tool answers to several names and each REGISTRATION SHAPE yields its
|
|
237
|
+
* own set; {@link passthroughToolNames} is the one implementation and
|
|
238
|
+
* spells them out, rather than this option and the check each carrying
|
|
239
|
+
* half of the rule.
|
|
240
|
+
*/
|
|
241
|
+
readonly passthroughTools?: readonly string[]
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
interface RequestText {
|
|
245
|
+
/** Where in the request the text sits, e.g. `city` or `filters.city`. */
|
|
246
|
+
readonly path: string
|
|
247
|
+
readonly text: string
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function normalizeText(text: string): string {
|
|
251
|
+
return text.trim().replace(/\s+/g, ' ')
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Every string the call was made with, and where it sat.
|
|
256
|
+
*
|
|
257
|
+
* Strings only. A number is a plausible answer — a count of 10, a port of
|
|
258
|
+
* 8080 — and comparing one would fire on `{ maxResults: 10 }` answered with
|
|
259
|
+
* `10`, which is a legitimate result from a legitimate tool.
|
|
260
|
+
*/
|
|
261
|
+
function collectRequestText(value: unknown, path: string, into: RequestText[]): void {
|
|
262
|
+
if (typeof value === 'string') {
|
|
263
|
+
const text = normalizeText(value)
|
|
264
|
+
if (isAtLeastCodePoints(text, MIN_RESTATEMENT_LENGTH)) into.push({ path, text })
|
|
265
|
+
return
|
|
266
|
+
}
|
|
267
|
+
if (Array.isArray(value)) {
|
|
268
|
+
for (const [index, item] of value.entries()) {
|
|
269
|
+
collectRequestText(item, `${path}[${index}]`, into)
|
|
270
|
+
}
|
|
271
|
+
return
|
|
272
|
+
}
|
|
273
|
+
if (value === null || typeof value !== 'object') return
|
|
274
|
+
for (const [key, nested] of Object.entries(value)) {
|
|
275
|
+
collectRequestText(nested, path === '' ? key : `${path}.${key}`, into)
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** How the refusal names the argument that came back, for a caller that cannot see the path syntax. */
|
|
280
|
+
function describeSubject(path: string): string {
|
|
281
|
+
if (path === '') return 'what this call sent'
|
|
282
|
+
const item = /^\[(\d+)\]$/.exec(path)
|
|
283
|
+
if (item) return `item ${item[1]} of the request`
|
|
284
|
+
return `the "${path}" argument this call sent`
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Every name a tool answers to, so `passthroughTools` does not require the
|
|
289
|
+
* host to write the connector's own internal spelling.
|
|
290
|
+
*
|
|
291
|
+
* Each registration shape yields its own set, and the sets are NOT
|
|
292
|
+
* interchangeable — an exemption written for one shape does not exempt
|
|
293
|
+
* another:
|
|
294
|
+
*
|
|
295
|
+
* - A server connected directly registers `mcp_<server>_<tool>` (the
|
|
296
|
+
* adapter concatenates `mcp_${serverName}_${toolName}` verbatim). For
|
|
297
|
+
* `mcp_weather-co_lookup` with `server: 'weather-co'`, the accepted names
|
|
298
|
+
* are `mcp_weather-co_lookup`, `lookup`, and `weather-co:lookup`.
|
|
299
|
+
* - A server a plugin contributes registers
|
|
300
|
+
* `<plugin>__mcp__<server>__<tool>`. For
|
|
301
|
+
* `myplugin__mcp__weather__lookup` the accepted names are
|
|
302
|
+
* `myplugin__mcp__weather__lookup`, the bare tail after the last `__`
|
|
303
|
+
* (`lookup`), and `weather:lookup` — NOT `mcp_weather_lookup`, which is
|
|
304
|
+
* the direct shape's spelling of a name this tool does not have.
|
|
305
|
+
* - A tool of this process's own has no server and no namespace, so the
|
|
306
|
+
* registered name is the whole set.
|
|
307
|
+
*
|
|
308
|
+
* The bare name is what the server's manifest calls the tool — the name the
|
|
309
|
+
* operator has in front of them when they read its manifest — and
|
|
310
|
+
* `server:tool` is what a human writes. Both are derived from the registered
|
|
311
|
+
* name and the tool's provenance, so a caller that has only one of the two
|
|
312
|
+
* gets the registered name and nothing else.
|
|
313
|
+
*/
|
|
314
|
+
export function passthroughToolNames(toolName: string, server?: string): readonly string[] {
|
|
315
|
+
const names = new Set<string>([toolName])
|
|
316
|
+
|
|
317
|
+
const separator = toolName.lastIndexOf(PLUGIN_NAMESPACE_SEPARATOR)
|
|
318
|
+
const pluginBare =
|
|
319
|
+
separator < 0 ? undefined : toolName.slice(separator + PLUGIN_NAMESPACE_SEPARATOR.length)
|
|
320
|
+
if (pluginBare !== undefined) names.add(pluginBare)
|
|
321
|
+
|
|
322
|
+
const directPrefix = server === undefined ? undefined : `mcp_${server}_`
|
|
323
|
+
const connectedBare =
|
|
324
|
+
directPrefix !== undefined && toolName.startsWith(directPrefix)
|
|
325
|
+
? toolName.slice(directPrefix.length)
|
|
326
|
+
: undefined
|
|
327
|
+
if (connectedBare !== undefined) names.add(connectedBare)
|
|
328
|
+
|
|
329
|
+
const bare = connectedBare ?? pluginBare
|
|
330
|
+
if (bare !== undefined && server !== undefined) names.add(`${server}:${bare}`)
|
|
331
|
+
|
|
332
|
+
return [...names]
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/**
|
|
336
|
+
* Refuse a result that is the request rather than an answer to it.
|
|
337
|
+
*
|
|
338
|
+
* #399 recorded that a screen at this boundary cannot know what a tool SHOULD
|
|
339
|
+
* have answered, and the issue that asked for this screen named three
|
|
340
|
+
* mismatches — a weather lookup for one city answering about another, a read
|
|
341
|
+
* of one path returning another's contents, a search returning instructions
|
|
342
|
+
* rather than results. **This screen decides none of the three.** The first
|
|
343
|
+
* two need a declaration of which argument names the subject, and the
|
|
344
|
+
* framework cannot infer one: for a file read the answer is the file's
|
|
345
|
+
* contents, which do not name the path, so the natural rule — an answer
|
|
346
|
+
* mentions its subject — would refuse every ordinary read. The third is what
|
|
347
|
+
* {@link toolResultInjectionGuardrail} already screens for at this same
|
|
348
|
+
* boundary, and re-implementing it here would give the two the same blind spot
|
|
349
|
+
* rather than the different one this adds.
|
|
350
|
+
*
|
|
351
|
+
* What is left is the one correspondence failure that needs no domain
|
|
352
|
+
* knowledge and cannot be an answer: **the result restates the request**. A
|
|
353
|
+
* tool handed a question and returning that question has answered nothing,
|
|
354
|
+
* whatever it is a tool for. The comparison is exact equality after
|
|
355
|
+
* whitespace normalisation against each string the call carried, so a result
|
|
356
|
+
* that says anything extra — the answer plus anything at all — passes.
|
|
357
|
+
*
|
|
358
|
+
* ## Why it judges framed results by default
|
|
359
|
+
*
|
|
360
|
+
* Because the envelope is the only marker that means *this process did not
|
|
361
|
+
* write this*, and the judgment it supports is the one this screen can make.
|
|
362
|
+
* Content arrives framed — by `wrapUntrusted`, through `frameServerResult`
|
|
363
|
+
* for a connector's tool result — when it came from outside: a connected
|
|
364
|
+
* server answered, or a host tool decided its own answer was not this
|
|
365
|
+
* process's to vouch for. A result that IS the request is a signal there.
|
|
366
|
+
*
|
|
367
|
+
* An earlier version judged results whose tool definition carried
|
|
368
|
+
* `provenance`, which is set by the MCP adapter and by nothing else. That
|
|
369
|
+
* predicate and this one agree on every connector result the comparison can
|
|
370
|
+
* act on, and they differ exactly where registration is an accident of the
|
|
371
|
+
* code rather than a fact about the content: the CLI's own remote Exa search
|
|
372
|
+
* frames its answer with this envelope and registers as a host tool, so a
|
|
373
|
+
* search that restated its query was out of scope while a fetch that did was
|
|
374
|
+
* in it. `scope` is a statement about what the screen is willing to judge, so
|
|
375
|
+
* it is expressed in terms of the content rather than of who mounted the
|
|
376
|
+
* tool.
|
|
377
|
+
*
|
|
378
|
+
* What the default leaves alone is this process's own UNFRAMED tools.
|
|
379
|
+
* `web_fetch` returns the page body, and a page whose body IS the URL it was
|
|
380
|
+
* fetched from — a redirect stub, a "you are here" placeholder, an echo
|
|
381
|
+
* endpoint — is a true result from a working tool that frames nothing.
|
|
382
|
+
* `test/…/a-result-that-does-not-answer-its-request` runs the shipped tool
|
|
383
|
+
* set through this screen with real calls and asserts none is refused, which
|
|
384
|
+
* is where that claim is checked rather than argued.
|
|
385
|
+
*
|
|
386
|
+
* `scope: 'all'` adds those too. That is the host's call, and it is the scope
|
|
387
|
+
* that costs something: under it, a fetch-like tool whose answer is its own
|
|
388
|
+
* argument is refused until the host names it in
|
|
389
|
+
* {@link ToolResultCorrespondenceOptions.passthroughTools}.
|
|
390
|
+
*
|
|
391
|
+
* The two other ways out were both worse. An exemption list of builtin NAMES
|
|
392
|
+
* inside this preset is a list that drifts — the next fetch-like tool has to
|
|
393
|
+
* remember to add itself, and a host tool that happens to be called
|
|
394
|
+
* `web_fetch` would be exempted by accident — and a subject declaration on
|
|
395
|
+
* every tool would be a new field on `ToolDefinition` that exists for this one
|
|
396
|
+
* screen.
|
|
397
|
+
*
|
|
398
|
+
* ## Three things it deliberately does not touch
|
|
399
|
+
*
|
|
400
|
+
* - **An empty result for a non-empty request.** Real, and not a signal:
|
|
401
|
+
* a search that matched nothing and a file with nothing in it both return
|
|
402
|
+
* nothing, and refusing either tells the model to stop looking when there
|
|
403
|
+
* was nothing to find.
|
|
404
|
+
* - **A shape contradicting `ToolDefinition.outputSchema`.** The schema is
|
|
405
|
+
* not on {@link ToolResultGuardrailContext} and is documented as shown to
|
|
406
|
+
* the model, never validated. Carrying it and enforcing it are changes to
|
|
407
|
+
* that contract, not this screen's to make.
|
|
408
|
+
* - **A failed call.** On `success: false` the text is a diagnostic, and
|
|
409
|
+
* `screenToolResult` replaces the output when it refuses — so refusing
|
|
410
|
+
* here would trade an echo nobody needs to catch for the error message
|
|
411
|
+
* the model needs to read.
|
|
412
|
+
*
|
|
413
|
+
* The comparison runs on the frame as well as on the text: a connector's
|
|
414
|
+
* result reaches a screen already wrapped by `wrapUntrusted`, so comparing
|
|
415
|
+
* the raw output would never match the one case this is most worth having —
|
|
416
|
+
* a connected server returning the request. Both the whole output and, when
|
|
417
|
+
* it is one wrapped block, the body inside it are compared.
|
|
418
|
+
*
|
|
419
|
+
* Detection is partial and this says so rather than implying coverage: an
|
|
420
|
+
* answer about the wrong subject, an answer that is plausible prose, and a
|
|
421
|
+
* restatement shorter than {@link MIN_RESTATEMENT_LENGTH} all pass.
|
|
422
|
+
*/
|
|
423
|
+
export function toolResultCorrespondenceGuardrail(
|
|
424
|
+
options: ToolResultCorrespondenceOptions = {},
|
|
425
|
+
): NamedGuardrail<ToolResultGuardrail> {
|
|
426
|
+
const passthrough = new Set(options.passthroughTools ?? [])
|
|
427
|
+
const scope = options.scope ?? 'framed'
|
|
428
|
+
|
|
429
|
+
return {
|
|
430
|
+
name: 'tool-result-correspondence',
|
|
431
|
+
check: ({ toolName, input, output, success, provenance }): ToolResultVerdict => {
|
|
432
|
+
if (!success) return { action: 'pass' }
|
|
433
|
+
// `output` is typed as a string and a tool that ignores its own
|
|
434
|
+
// contract can hand back something else. Nothing here can compare
|
|
435
|
+
// a number against a request, and a screen that throws fails
|
|
436
|
+
// CLOSED — so the one thing this must not do is refuse a result
|
|
437
|
+
// because it could not read it.
|
|
438
|
+
if (typeof output !== 'string') return { action: 'pass' }
|
|
439
|
+
|
|
440
|
+
// The scope, and the same reader the comparison below uses rather
|
|
441
|
+
// than a second "looks framed" test that could disagree with it.
|
|
442
|
+
//
|
|
443
|
+
// This is a superset of the `provenance`-based predicate it
|
|
444
|
+
// replaced, for every result the comparison can act on. The only
|
|
445
|
+
// tool definition carrying `provenance` is the MCP adapter's, and
|
|
446
|
+
// `frameServerResult` frames every non-empty text it returns — so
|
|
447
|
+
// a connected result that reaches the comparison at all carries a
|
|
448
|
+
// legible frame. The one case the new predicate drops is a
|
|
449
|
+
// connected result with an EMPTY output, which the comparison
|
|
450
|
+
// skipped anyway (`normalizeText('')` is `''`, and an empty string
|
|
451
|
+
// is never a request of the minimum length).
|
|
452
|
+
const body = untrustedEnvelopeBody(output)
|
|
453
|
+
if (scope !== 'all' && body === undefined) return { action: 'pass' }
|
|
454
|
+
|
|
455
|
+
if (
|
|
456
|
+
passthrough.size > 0 &&
|
|
457
|
+
passthroughToolNames(toolName, provenance?.server).some((name) => passthrough.has(name))
|
|
458
|
+
) {
|
|
459
|
+
return { action: 'pass' }
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
const request: RequestText[] = []
|
|
463
|
+
collectRequestText(input, '', request)
|
|
464
|
+
if (request.length === 0) return { action: 'pass' }
|
|
465
|
+
|
|
466
|
+
// The whole output, and the body when the output is one wrapped
|
|
467
|
+
// block. The two are never the same string — the frame carries the
|
|
468
|
+
// tags — so this is two comparisons, not the same one twice, and
|
|
469
|
+
// the refusal says which one matched.
|
|
470
|
+
for (const [text, insideFrame] of [
|
|
471
|
+
[normalizeText(output), false],
|
|
472
|
+
[normalizeText(body ?? ''), true],
|
|
473
|
+
] as const) {
|
|
474
|
+
if (text === '') continue
|
|
475
|
+
const restated = request.find((entry) => entry.text === text)
|
|
476
|
+
if (!restated) continue
|
|
477
|
+
// The reason names the tool, and says what the comparison
|
|
478
|
+
// actually did. Both were wrong before: a host reading the
|
|
479
|
+
// transcript is the person who can exempt this tool, and
|
|
480
|
+
// "verbatim" was a false claim about a comparison that ran
|
|
481
|
+
// after whitespace normalisation.
|
|
482
|
+
const matched = insideFrame
|
|
483
|
+
? `the text inside the untrusted frame from "${toolName}"`
|
|
484
|
+
: `the result from "${toolName}"`
|
|
485
|
+
return {
|
|
486
|
+
action: 'refuse',
|
|
487
|
+
reason: `${matched} is ${describeSubject(restated.path)}, whitespace-normalised, so it is the request rather than an answer to it`,
|
|
488
|
+
}
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
return { action: 'pass' }
|
|
492
|
+
},
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
/**
|
|
497
|
+
* The screens a run installs on itself when its host configured none.
|
|
498
|
+
*
|
|
499
|
+
* A DEFAULT, and the reason it is here rather than on
|
|
500
|
+
* {@link ToolRegistryConfig.resultGuardrails}: a host assembles a registry
|
|
501
|
+
* and hands it to `runAgent`, so a registry-constructor default would be the
|
|
502
|
+
* host's to write and this repository's default would reach nobody. The run
|
|
503
|
+
* is the thing that has to carry it, and a run that wants none says so with
|
|
504
|
+
* an empty array — which is the escape hatch, and it exists precisely because
|
|
505
|
+
* the screen below can refuse a result.
|
|
506
|
+
*
|
|
507
|
+
* One screen, scoped to results framed as untrusted. See the preset for why
|
|
508
|
+
* the scope is what it is: an unframed host tool whose answer IS the request —
|
|
509
|
+
* `web_fetch` returning a page whose body is its own URL — is an ordinary
|
|
510
|
+
* result, and a default that refuses those is a default that breaks a working
|
|
511
|
+
* tool.
|
|
512
|
+
*
|
|
513
|
+
* **A default that refuses a legitimate result is a default that gets turned
|
|
514
|
+
* off, so the exemption has to be reachable from wherever the screen is
|
|
515
|
+
* installed.** `toolResultCorrespondenceGuardrail({ passthroughTools })` is
|
|
516
|
+
* how a host writing SDK code says so; a run's `toolResultGuardrails` array
|
|
517
|
+
* is where it substitutes its own configured instance. An application that
|
|
518
|
+
* ships the default without also shipping a way to name an exception has
|
|
519
|
+
* shipped a switch with one position.
|
|
520
|
+
*
|
|
521
|
+
* Frozen, because it is handed to callers who may want to extend it:
|
|
522
|
+
* `[...DEFAULT_TOOL_RESULT_GUARDRAILS, myScreen()]`. A caller that could
|
|
523
|
+
* mutate it would be mutating the default for every other run in the process.
|
|
524
|
+
*/
|
|
525
|
+
export const DEFAULT_TOOL_RESULT_GUARDRAILS: readonly ToolResultGuardrailSpec[] = Object.freeze([
|
|
526
|
+
toolResultCorrespondenceGuardrail(),
|
|
527
|
+
])
|
|
@@ -363,6 +363,20 @@ export interface QueryParams {
|
|
|
363
363
|
* agent decides the rest is worth re-reading. Set `0` to disable.
|
|
364
364
|
*/
|
|
365
365
|
maxToolOutputChars?: number
|
|
366
|
+
/**
|
|
367
|
+
* Screens to run against every tool result, where the registry was not
|
|
368
|
+
* built with its own.
|
|
369
|
+
*
|
|
370
|
+
* This is the run's half of a boundary whose only other door is the
|
|
371
|
+
* registry constructor — and a registry is usually the HOST's, assembled
|
|
372
|
+
* before the run exists, so a run-config option is the only way a run
|
|
373
|
+
* screens a registry it did not build. A registry built WITH
|
|
374
|
+
* `resultGuardrails` states its own policy and wins, `[]` included.
|
|
375
|
+
*
|
|
376
|
+
* Absent installs {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array
|
|
377
|
+
* installs none, which is how a caller turns the default off.
|
|
378
|
+
*/
|
|
379
|
+
toolResultGuardrails?: readonly import('../../types/guardrail/index.js').ToolResultGuardrailSpec[]
|
|
366
380
|
/**
|
|
367
381
|
* Smaller preview for text that exceeded maxToolOutputChars, after its full
|
|
368
382
|
* host output and integrity manifest have been saved. Unset/0 keeps the old
|
|
@@ -1841,6 +1855,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
1841
1855
|
...(params.maxToolOutputChars !== undefined
|
|
1842
1856
|
? { maxToolOutputChars: params.maxToolOutputChars }
|
|
1843
1857
|
: {}),
|
|
1858
|
+
...(params.toolResultGuardrails !== undefined
|
|
1859
|
+
? { toolResultGuardrails: params.toolResultGuardrails }
|
|
1860
|
+
: {}),
|
|
1844
1861
|
...(params.retainedToolPreviewChars !== undefined
|
|
1845
1862
|
? { retainedToolPreviewChars: params.retainedToolPreviewChars }
|
|
1846
1863
|
: {}),
|
|
@@ -49,6 +49,8 @@ export interface ToolingBootstrapConfig {
|
|
|
49
49
|
maxToolCalls?: number
|
|
50
50
|
readToolCallBudgetEvents?: () => Promise<readonly RunEvent[]>
|
|
51
51
|
maxToolOutputChars?: number
|
|
52
|
+
/** See `QueryParams.toolResultGuardrails`. Absent installs the shipped default; a registry's own win. */
|
|
53
|
+
toolResultGuardrails?: readonly import('../../types/guardrail/index.js').ToolResultGuardrailSpec[]
|
|
52
54
|
retainedToolPreviewChars?: number
|
|
53
55
|
maxToolContentBytes?: number
|
|
54
56
|
captureRunEvidence?: import('../../types/tool/index.js').ToolContext['captureRunEvidence']
|
|
@@ -104,6 +106,9 @@ export class ToolingBootstrap {
|
|
|
104
106
|
...(config.maxToolOutputChars !== undefined
|
|
105
107
|
? { maxToolOutputChars: config.maxToolOutputChars }
|
|
106
108
|
: {}),
|
|
109
|
+
...(config.toolResultGuardrails !== undefined
|
|
110
|
+
? { toolResultGuardrails: config.toolResultGuardrails }
|
|
111
|
+
: {}),
|
|
107
112
|
...(config.retainedToolPreviewChars !== undefined
|
|
108
113
|
? { retainedToolPreviewChars: config.retainedToolPreviewChars }
|
|
109
114
|
: {}),
|