theorum 0.1.15 → 1.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +241 -98
- package/esm/mod.d.ts +57 -28
- package/esm/mod.js +43 -23
- package/esm/src/cli/commands/bench.js +18 -18
- package/esm/src/cli/commands/fuzz-canary.d.ts +13 -0
- package/esm/src/cli/commands/fuzz-canary.js +191 -0
- package/esm/src/cli/commands/fuzz-guardrails.d.ts +3 -5
- package/esm/src/cli/commands/fuzz-guardrails.js +4 -581
- package/esm/src/cli/commands/guardrails-eval.d.ts +14 -0
- package/esm/src/cli/commands/guardrails-eval.js +15 -0
- package/esm/src/cli/commands/profile.js +35 -15
- package/esm/src/cli/commands/run.d.ts +3 -0
- package/esm/src/cli/commands/run.js +23 -32
- package/esm/src/cli/commands/test.d.ts +10 -1
- package/esm/src/cli/commands/test.js +34 -34
- package/esm/src/cli/event-log.d.ts +19 -0
- package/esm/src/cli/event-log.js +147 -0
- package/esm/src/cli/index.js +57 -11
- package/esm/src/cli/matrix/synthesizer.d.ts +10 -12
- package/esm/src/cli/matrix/synthesizer.js +45 -118
- package/esm/src/guardrails/canary-gate.d.ts +21 -0
- package/esm/src/guardrails/canary-gate.js +32 -0
- package/esm/src/guardrails/canary.d.ts +34 -0
- package/esm/src/guardrails/canary.js +150 -0
- package/esm/src/guardrails/corpus/canary-egress-attacks.d.ts +17 -0
- package/esm/src/guardrails/corpus/canary-egress-attacks.js +151 -0
- package/esm/src/guardrails/corpus/fuzz-inbound.d.ts +11 -0
- package/esm/src/guardrails/corpus/fuzz-inbound.js +213 -0
- package/esm/src/guardrails/corpus/inbound-payloads.d.ts +10 -0
- package/esm/src/guardrails/corpus/inbound-payloads.js +125 -0
- package/esm/src/guardrails/corpus/live-attacks.d.ts +20 -0
- package/esm/src/guardrails/corpus/live-attacks.js +231 -0
- package/esm/src/guardrails/corpus/mod.d.ts +14 -0
- package/esm/src/guardrails/corpus/mod.js +11 -0
- package/esm/src/guardrails/corpus/secrets.d.ts +17 -0
- package/esm/src/guardrails/corpus/secrets.js +17 -0
- package/esm/src/guardrails/corpus/strings.d.ts +28 -0
- package/esm/src/guardrails/corpus/strings.js +34 -0
- package/esm/src/guardrails/corpus/types.d.ts +38 -0
- package/esm/src/guardrails/corpus/types.js +6 -0
- package/esm/src/guardrails/egress.d.ts +32 -0
- package/esm/src/guardrails/egress.js +87 -0
- package/esm/src/guardrails/error.d.ts +14 -23
- package/esm/src/guardrails/error.js +87 -76
- package/esm/src/guardrails/eval/corpus.d.ts +108 -0
- package/esm/src/guardrails/eval/corpus.js +978 -0
- package/esm/src/guardrails/eval/mod.d.ts +51 -0
- package/esm/src/guardrails/eval/mod.js +133 -0
- package/esm/src/guardrails/eval/score.d.ts +66 -0
- package/esm/src/guardrails/eval/score.js +114 -0
- package/esm/src/guardrails/events.d.ts +25 -0
- package/esm/src/guardrails/events.js +56 -0
- package/esm/src/guardrails/hits.d.ts +24 -0
- package/esm/src/guardrails/hits.js +45 -0
- package/esm/src/guardrails/injection.js +28 -5
- package/esm/src/guardrails/lexicon.d.ts +39 -0
- package/esm/src/guardrails/lexicon.js +200 -0
- package/esm/src/guardrails/live-outbound-gate.d.ts +41 -0
- package/esm/src/guardrails/live-outbound-gate.js +222 -0
- package/esm/src/guardrails/mod.d.ts +30 -6
- package/esm/src/guardrails/mod.js +20 -5
- package/esm/src/guardrails/network.d.ts +19 -0
- package/esm/src/guardrails/network.js +234 -0
- package/esm/src/guardrails/policy.d.ts +35 -0
- package/esm/src/guardrails/policy.js +50 -0
- package/esm/src/guardrails/progressive-yield.d.ts +51 -0
- package/esm/src/guardrails/progressive-yield.js +98 -0
- package/esm/src/guardrails/quota.d.ts +17 -3
- package/esm/src/guardrails/quota.js +18 -4
- package/esm/src/guardrails/sanitize.d.ts +45 -19
- package/esm/src/guardrails/sanitize.js +177 -94
- package/esm/src/guardrails/sensitive.js +2 -1
- package/esm/src/guardrails/serialize.d.ts +35 -0
- package/esm/src/guardrails/serialize.js +58 -0
- package/esm/src/guardrails/testing.d.ts +17 -0
- package/esm/src/guardrails/testing.js +13 -0
- package/esm/src/guardrails/theorum-error.d.ts +12 -0
- package/esm/src/guardrails/theorum-error.js +15 -0
- package/esm/src/guardrails/tool-directives.d.ts +48 -0
- package/esm/src/guardrails/tool-directives.js +124 -0
- package/esm/src/guardrails/tool-result.d.ts +93 -0
- package/esm/src/guardrails/tool-result.js +276 -0
- package/esm/src/guardrails/types.d.ts +291 -0
- package/esm/src/guardrails/types.js +72 -0
- package/esm/src/host/client-turn.d.ts +19 -0
- package/esm/src/host/client-turn.js +36 -0
- package/esm/src/host/mint-trace.d.ts +1 -1
- package/esm/src/host/mod.d.ts +5 -3
- package/esm/src/host/mod.js +4 -3
- package/esm/src/kernel/auth/crypto.d.ts +42 -0
- package/esm/src/kernel/auth/crypto.js +106 -0
- package/esm/src/kernel/auth/mod.d.ts +11 -0
- package/esm/src/kernel/auth/mod.js +11 -0
- package/esm/src/kernel/auth/oauth.d.ts +47 -0
- package/esm/src/kernel/auth/oauth.js +278 -0
- package/esm/src/kernel/auth/types.d.ts +133 -0
- package/esm/src/kernel/auth/types.js +13 -0
- package/esm/src/kernel/engine/delta.d.ts +24 -2
- package/esm/src/kernel/engine/delta.js +478 -39
- package/esm/src/kernel/engine/live-inbound.d.ts +21 -0
- package/esm/src/kernel/engine/live-inbound.js +31 -0
- package/esm/src/kernel/engine/live-ingress.d.ts +19 -0
- package/esm/src/kernel/engine/live-ingress.js +47 -0
- package/esm/src/kernel/engine/repair.js +13 -12
- package/esm/src/kernel/engine/runner/gates.d.ts +1 -1
- package/esm/src/kernel/engine/runner/gates.js +130 -43
- package/esm/src/kernel/engine/runner/mod.d.ts +6 -4
- package/esm/src/kernel/engine/runner/mod.js +192 -53
- package/esm/src/kernel/engine/runner/schema-validation.js +3 -3
- package/esm/src/kernel/engine/runner/stages.d.ts +39 -0
- package/esm/src/kernel/engine/runner/stages.js +89 -0
- package/esm/src/kernel/engine/runner/state.d.ts +31 -0
- package/esm/src/kernel/engine/runner/steps.d.ts +1 -1
- package/esm/src/kernel/engine/runner/steps.js +244 -43
- package/esm/src/kernel/engine/runner/stream.d.ts +9 -3
- package/esm/src/kernel/engine/runner/stream.js +140 -44
- package/esm/src/kernel/engine/session/mod.d.ts +25 -0
- package/esm/src/kernel/engine/session/mod.js +557 -0
- package/esm/src/kernel/interaction-parts.d.ts +14 -0
- package/esm/src/kernel/interaction-parts.js +23 -0
- package/esm/src/kernel/mod.d.ts +21 -10
- package/esm/src/kernel/mod.js +11 -8
- package/esm/src/kernel/profile-graph.d.ts +159 -0
- package/esm/src/kernel/profile-graph.js +156 -0
- package/esm/src/kernel/registry/attachments.d.ts +12 -10
- package/esm/src/kernel/registry/attachments.js +33 -27
- package/esm/src/kernel/registry/catalog.d.ts +25 -24
- package/esm/src/kernel/registry/catalog.js +60 -101
- package/esm/src/kernel/registry/ingress.d.ts +9 -4
- package/esm/src/kernel/registry/ingress.js +97 -75
- package/esm/src/kernel/registry/profile-outputs.d.ts +4 -0
- package/esm/src/kernel/registry/profile-outputs.js +8 -0
- package/esm/src/kernel/registry/profiles.d.ts +55 -12
- package/esm/src/kernel/registry/profiles.js +413 -73
- package/esm/src/kernel/registry/provider-request.js +13 -7
- package/esm/src/kernel/registry/resolve.d.ts +8 -8
- package/esm/src/kernel/registry/resolve.js +169 -154
- package/esm/src/kernel/registry/schemas.js +1 -1
- package/esm/src/kernel/registry/sole-model.d.ts +8 -0
- package/esm/src/kernel/registry/sole-model.js +10 -0
- package/esm/src/kernel/registry/system-prompt.d.ts +10 -0
- package/esm/src/kernel/registry/system-prompt.js +40 -0
- package/esm/src/kernel/registry/system-role.d.ts +8 -0
- package/esm/src/kernel/registry/system-role.js +14 -0
- package/esm/src/kernel/registry/vault.d.ts +12 -7
- package/esm/src/kernel/registry/vault.js +32 -10
- package/esm/src/kernel/schema.d.ts +231 -0
- package/esm/src/kernel/schema.js +607 -0
- package/esm/src/kernel/stages.d.ts +175 -0
- package/esm/src/kernel/stages.js +476 -0
- package/esm/src/kernel/stop.d.ts +78 -19
- package/esm/src/kernel/stop.js +51 -16
- package/esm/src/kernel/tools/events.d.ts +41 -0
- package/esm/src/kernel/tools/events.js +71 -0
- package/esm/src/kernel/tools/execute.d.ts +84 -0
- package/esm/src/kernel/tools/execute.js +614 -0
- package/esm/src/kernel/tools/harness.d.ts +8 -0
- package/esm/src/kernel/tools/harness.js +46 -0
- package/esm/src/kernel/tools/invoke.d.ts +10 -0
- package/esm/src/kernel/tools/invoke.js +101 -0
- package/esm/src/kernel/tools/mod.d.ts +13 -0
- package/esm/src/kernel/tools/mod.js +11 -0
- package/esm/src/kernel/tools/permission.d.ts +15 -0
- package/esm/src/kernel/tools/permission.js +47 -0
- package/esm/src/kernel/tools/project.d.ts +12 -0
- package/esm/src/kernel/tools/project.js +36 -0
- package/esm/src/kernel/tools/registry.d.ts +23 -0
- package/esm/src/kernel/tools/registry.js +81 -0
- package/esm/src/kernel/tools/remote.d.ts +94 -0
- package/esm/src/kernel/tools/remote.js +577 -0
- package/esm/src/kernel/tools/resolve.d.ts +39 -0
- package/esm/src/kernel/tools/resolve.js +283 -0
- package/esm/src/kernel/tools/schema.d.ts +15 -0
- package/esm/src/kernel/tools/schema.js +176 -0
- package/esm/src/kernel/tools/stage-run.d.ts +105 -0
- package/esm/src/kernel/tools/stage-run.js +155 -0
- package/esm/src/kernel/tools/types.d.ts +394 -0
- package/esm/src/kernel/tools/types.js +9 -0
- package/esm/src/kernel/types.d.ts +540 -256
- package/esm/src/kernel/util/find-last.d.ts +2 -0
- package/esm/src/kernel/util/find-last.js +10 -0
- package/esm/src/observability/destinations.d.ts +31 -0
- package/esm/src/observability/destinations.js +67 -0
- package/esm/src/observability/mod.d.ts +10 -3
- package/esm/src/observability/mod.js +6 -2
- package/esm/src/observability/policy.d.ts +27 -0
- package/esm/src/observability/policy.js +80 -0
- package/esm/src/observability/resolve-policy.d.ts +16 -0
- package/esm/src/observability/resolve-policy.js +64 -0
- package/esm/src/observability/trace-attach.d.ts +8 -4
- package/esm/src/observability/trace-attach.js +50 -29
- package/esm/src/observability/trace-record.d.ts +23 -13
- package/esm/src/observability/trace-record.js +96 -39
- package/esm/src/observability/trace-sink.d.ts +19 -0
- package/esm/src/observability/trace-sink.js +10 -0
- package/esm/src/observability/trace-usage.d.ts +10 -3
- package/esm/src/observability/trace-usage.js +70 -17
- package/esm/src/observability/trace.d.ts +18 -7
- package/esm/src/observability/trace.js +34 -17
- package/esm/src/observability/types.d.ts +113 -0
- package/esm/src/observability/types.js +11 -0
- package/esm/src/presets/google/speech-voices.d.ts +11 -0
- package/esm/src/presets/google/speech-voices.js +41 -0
- package/esm/src/presets/google.d.ts +36 -24
- package/esm/src/presets/google.js +50 -63
- package/esm/src/presets/mod.d.ts +2 -2
- package/esm/src/presets/mod.js +1 -1
- package/esm/src/providers/create-provider.d.ts +20 -17
- package/esm/src/providers/create-provider.js +72 -26
- package/esm/src/providers/google/interactions/framing.d.ts +23 -0
- package/esm/src/providers/google/interactions/framing.js +269 -0
- package/esm/src/providers/google/interactions/mod.d.ts +7 -0
- package/esm/src/providers/google/interactions/mod.js +7 -0
- package/esm/src/providers/google/interactions/stream.d.ts +83 -0
- package/esm/src/providers/google/interactions/stream.js +588 -0
- package/esm/src/providers/google/keys.d.ts +26 -0
- package/esm/src/providers/{keys.js → google/keys.js} +19 -31
- package/esm/src/providers/google/live/framing.d.ts +49 -0
- package/esm/src/providers/google/live/framing.js +552 -0
- package/esm/src/providers/google/live/openapi-schema.d.ts +6 -0
- package/esm/src/providers/google/live/openapi-schema.js +46 -0
- package/esm/src/providers/google/live/session.d.ts +25 -0
- package/esm/src/providers/google/live/session.js +134 -0
- package/esm/src/providers/google/live/stream.d.ts +45 -0
- package/esm/src/providers/google/live/stream.js +214 -0
- package/esm/src/providers/google/urls.d.ts +6 -0
- package/esm/src/providers/google/urls.js +6 -0
- package/esm/src/providers/local/local.d.ts +30 -0
- package/esm/src/providers/{local.js → local/local.js} +66 -126
- package/esm/src/providers/local/mod.d.ts +9 -0
- package/esm/src/providers/local/mod.js +9 -0
- package/esm/src/providers/mod.d.ts +6 -3
- package/esm/src/providers/mod.js +3 -1
- package/esm/src/providers/openrouter/cache-control.d.ts +24 -0
- package/esm/src/providers/openrouter/cache-control.js +23 -0
- package/esm/src/providers/openrouter/chat.d.ts +107 -0
- package/esm/src/providers/{openrouter.js → openrouter/chat.js} +117 -231
- package/esm/src/providers/openrouter/image.d.ts +34 -0
- package/esm/src/providers/openrouter/image.js +275 -0
- package/esm/src/providers/openrouter/openai/chat-payload.d.ts +24 -0
- package/esm/src/providers/openrouter/openai/chat-payload.js +82 -0
- package/esm/src/providers/openrouter/openai/compat.d.ts +53 -0
- package/esm/src/providers/openrouter/openai/compat.js +213 -0
- package/esm/src/providers/openrouter/openai/image-payload.d.ts +18 -0
- package/esm/src/providers/openrouter/openai/image-payload.js +90 -0
- package/esm/src/providers/openrouter/openai/sdk-messages.d.ts +22 -0
- package/esm/src/providers/openrouter/openai/sdk-messages.js +122 -0
- package/esm/src/providers/openrouter/resolve-api-key.d.ts +9 -0
- package/esm/src/providers/openrouter/resolve-api-key.js +24 -0
- package/esm/src/providers/openrouter/speech.d.ts +23 -0
- package/esm/src/providers/{speech.js → openrouter/speech.js} +32 -55
- package/esm/src/providers/probe.d.ts +1 -0
- package/esm/src/providers/probe.js +22 -0
- package/esm/src/providers/shared/pcm.d.ts +12 -0
- package/esm/src/providers/{pcm.js → shared/pcm.js} +16 -3
- package/esm/src/providers/shared/sse.d.ts +18 -0
- package/esm/src/providers/shared/sse.js +87 -0
- package/esm/src/providers/shared/tool-args.d.ts +17 -0
- package/esm/src/providers/shared/tool-args.js +45 -0
- package/esm/src/providers/shared/upstream-tap.d.ts +5 -0
- package/esm/src/providers/{google-tap.js → shared/upstream-tap.js} +4 -7
- package/esm/src/providers/shared/upstream-tape.d.ts +6 -0
- package/esm/src/providers/{gemini-tape.js → shared/upstream-tape.js} +12 -22
- package/esm/src/providers/types.d.ts +27 -0
- package/esm/src/providers/types.js +1 -0
- package/package.json +11 -7
- package/docs/cli.md +0 -97
- package/docs/guardrails.md +0 -178
- package/docs/host.md +0 -97
- package/docs/kernel.md +0 -404
- package/docs/observability.md +0 -105
- package/docs/openrouter.md +0 -125
- package/docs/presets-google.md +0 -91
- package/docs/presets.md +0 -88
- package/docs/providers.md +0 -202
- package/docs/streaming.md +0 -96
- package/esm/src/kernel/engine/boundary.d.ts +0 -10
- package/esm/src/kernel/engine/boundary.js +0 -55
- package/esm/src/kernel/engine/runner/tools.d.ts +0 -13
- package/esm/src/kernel/engine/runner/tools.js +0 -198
- package/esm/src/kernel/registry/tools.d.ts +0 -12
- package/esm/src/kernel/registry/tools.js +0 -36
- package/esm/src/providers/expose-for-tests.d.ts +0 -1
- package/esm/src/providers/expose-for-tests.js +0 -25
- package/esm/src/providers/gemini-tape.d.ts +0 -2
- package/esm/src/providers/google-tap.d.ts +0 -3
- package/esm/src/providers/interactions.d.ts +0 -5
- package/esm/src/providers/interactions.js +0 -169
- package/esm/src/providers/keys.d.ts +0 -19
- package/esm/src/providers/local.d.ts +0 -29
- package/esm/src/providers/openrouter-mod.d.ts +0 -13
- package/esm/src/providers/openrouter-mod.js +0 -12
- package/esm/src/providers/openrouter-payload.d.ts +0 -39
- package/esm/src/providers/openrouter-payload.js +0 -195
- package/esm/src/providers/openrouter.d.ts +0 -15
- package/esm/src/providers/pcm.d.ts +0 -7
- package/esm/src/providers/provider.d.ts +0 -15
- package/esm/src/providers/provider.js +0 -202
- package/esm/src/providers/speech.d.ts +0 -23
- package/esm/src/providers/sse.d.ts +0 -7
- package/esm/src/providers/sse.js +0 -55
- package/esm/src/streaming/mod.d.ts +0 -9
- package/esm/src/streaming/mod.js +0 -8
- /package/esm/src/{streaming → host}/readStreamingJsonStringField.d.ts +0 -0
- /package/esm/src/{streaming → host}/readStreamingJsonStringField.js +0 -0
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Directive detection for tool ingress.
|
|
3
|
+
*
|
|
4
|
+
* Tool results carry a different threat than user text. The jailbreak phrasings in
|
|
5
|
+
* `injection.ts` name the thing they attack — "ignore previous instructions",
|
|
6
|
+
* "reveal your system prompt" — and real indirect injection rarely does. It reads
|
|
7
|
+
* like a status update or a helpful next step, and pattern-matching for the word
|
|
8
|
+
* "instructions" misses all of it.
|
|
9
|
+
*
|
|
10
|
+
* What is anomalous in *data* is content that behaves like an instruction: naming
|
|
11
|
+
* a tool the agent can call, issuing an imperative at the agent, or claiming an
|
|
12
|
+
* authority the content does not have.
|
|
13
|
+
*
|
|
14
|
+
* A signal only counts when it co-occurs with a concrete external destination —
|
|
15
|
+
* an address or URL. Directive language alone is far too common in legitimate
|
|
16
|
+
* output to act on. These signals raise the turn's taint rather than rewriting the text. A page
|
|
17
|
+
* documenting an email API legitimately says "call send_email"; redacting that
|
|
18
|
+
* would corrupt content the model needs. Being wrong here should cost a refused
|
|
19
|
+
* write — recoverable and visible — not silently damaged input.
|
|
20
|
+
*
|
|
21
|
+
* @module
|
|
22
|
+
*/
|
|
23
|
+
import { normalizeForDetection } from './normalize.js';
|
|
24
|
+
/** Rule ids emitted by tool-ingress directive detection. */
|
|
25
|
+
export const DIRECTIVE_RULES = {
|
|
26
|
+
toolName: 'tool_result.names-callable-tool',
|
|
27
|
+
imperative: 'tool_result.imperative',
|
|
28
|
+
authority: 'tool_result.authority-claim',
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* Imperatives aimed at an agent rather than a reader.
|
|
32
|
+
*
|
|
33
|
+
* Bounded quantifiers throughout: tool results can be large, and an unbounded
|
|
34
|
+
* scanner on attacker-controlled bytes is a denial-of-service surface.
|
|
35
|
+
*/
|
|
36
|
+
const IMPERATIVES = [
|
|
37
|
+
/\b(?:next|final|remaining)\s+steps?\s*[:—-]/gi,
|
|
38
|
+
/\byou\s+(?:must|should|need\s+to|are\s+required\s+to)\s+(?:now\s+)?\w{2,20}/gi,
|
|
39
|
+
/\bto\s+(?:complete|finish|fulfil|fulfill)\s+(?:this|the)\s+\w{2,20}/gi,
|
|
40
|
+
/\b(?:call|invoke|run|execute)\s+(?:the\s+)?[\w.-]{2,40}\s+(?:tool|function)\b/gi,
|
|
41
|
+
/\b(?:send|forward|email|transfer|upload|post)\s+(?:the\s+|all\s+|your\s+)?[\w\s]{2,40}\s+to\s+\S{3,80}/gi,
|
|
42
|
+
/\b(?:delete|remove|drop|purge)\s+(?:the\s+|all\s+)?[\w\s]{2,40}\b/gi,
|
|
43
|
+
];
|
|
44
|
+
/**
|
|
45
|
+
* A concrete external destination — the thing exfiltration needs and ordinary
|
|
46
|
+
* process prose almost never carries.
|
|
47
|
+
*
|
|
48
|
+
* This is the discriminator. Directive language on its own is everywhere in
|
|
49
|
+
* legitimate tool output: documentation says "you must be an admin", support
|
|
50
|
+
* articles say "to remove a user", status reports say "the user has approved".
|
|
51
|
+
* Measured on a benign sample, directive signals alone produced false positives
|
|
52
|
+
* on most of it. Pairing a signal with a destination removed all of them.
|
|
53
|
+
*/
|
|
54
|
+
const EXFIL_TARGET = /\b[\w.+-]{1,64}@[\w-]{1,63}(?:\.[\w-]{1,63}){1,4}\b|\bhttps?:\/\/\S{4,120}/i;
|
|
55
|
+
/** Claims of permission or provenance the content cannot actually hold. */
|
|
56
|
+
const AUTHORITY = [
|
|
57
|
+
/\b(?:the\s+)?user\s+has\s+(?:already\s+)?(?:approved|authorised|authorized|confirmed|requested)\b/gi,
|
|
58
|
+
/\b(?:system|admin|operator)\s+(?:note|notice|message|override|instruction)s?\s*[:—-]/gi,
|
|
59
|
+
/\bon\s+behalf\s+of\s+the\s+(?:user|operator|admin)\b/gi,
|
|
60
|
+
/\bthis\s+(?:is|was)\s+(?:pre-?)?(?:approved|authorised|authorized)\b/gi,
|
|
61
|
+
];
|
|
62
|
+
function matches(patterns, text) {
|
|
63
|
+
return patterns.some((pattern) => {
|
|
64
|
+
pattern.lastIndex = 0;
|
|
65
|
+
return pattern.test(text);
|
|
66
|
+
});
|
|
67
|
+
}
|
|
68
|
+
/** Word-boundary match for a tool name, escaped so registry names cannot inject. */
|
|
69
|
+
function mentionsTool(text, tool) {
|
|
70
|
+
if (tool.length < 3) {
|
|
71
|
+
return false;
|
|
72
|
+
}
|
|
73
|
+
const escaped = tool.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
74
|
+
return new RegExp(`(?:^|[^\\w-])${escaped}(?:$|[^\\w-])`, 'i').test(text);
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* Detect instruction-shaped content in a tool result.
|
|
78
|
+
*
|
|
79
|
+
* `callableTools` is the set the model can actually invoke this turn. A result
|
|
80
|
+
* naming one is the highest-precision signal available — ordinary data has no
|
|
81
|
+
* reason to name the agent's tools, and no generic content filter can check it
|
|
82
|
+
* because it requires the turn's registry.
|
|
83
|
+
*/
|
|
84
|
+
function directiveHits(text, callableTools = []) {
|
|
85
|
+
if (!text || !EXFIL_TARGET.test(text)) {
|
|
86
|
+
// No destination, no exfiltration. Action-shaped attacks that carry no target
|
|
87
|
+
// are left to the taint gate, which does not depend on reading the content.
|
|
88
|
+
return [];
|
|
89
|
+
}
|
|
90
|
+
const normalized = normalizeForDetection(text);
|
|
91
|
+
const hits = [];
|
|
92
|
+
// One hit per named tool — several names is a stronger signal than one.
|
|
93
|
+
for (const _tool of callableTools.filter((tool) => mentionsTool(normalized, tool))) {
|
|
94
|
+
hits.push({ rule: DIRECTIVE_RULES.toolName, severity: 'high' });
|
|
95
|
+
}
|
|
96
|
+
if (matches(IMPERATIVES, normalized)) {
|
|
97
|
+
hits.push({ rule: DIRECTIVE_RULES.imperative, severity: 'medium' });
|
|
98
|
+
}
|
|
99
|
+
if (matches(AUTHORITY, normalized)) {
|
|
100
|
+
hits.push({ rule: DIRECTIVE_RULES.authority, severity: 'medium' });
|
|
101
|
+
}
|
|
102
|
+
return hits;
|
|
103
|
+
}
|
|
104
|
+
/** True when a result looked like it was trying to steer the agent. */
|
|
105
|
+
function looksDirective(hits) {
|
|
106
|
+
return hits.length > 0;
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* Strength of the signals, read off the hits rather than invented.
|
|
110
|
+
*
|
|
111
|
+
* Naming a tool the model can call is the sharpest signal available, so it alone
|
|
112
|
+
* reaches `high`; so does agreement between two different signal kinds.
|
|
113
|
+
*/
|
|
114
|
+
function advisoryLevel(hits) {
|
|
115
|
+
if (hits.length === 0) {
|
|
116
|
+
return 'none';
|
|
117
|
+
}
|
|
118
|
+
const kinds = new Set(hits.map((hit) => hit.rule));
|
|
119
|
+
if (kinds.has(DIRECTIVE_RULES.toolName) || kinds.size > 1) {
|
|
120
|
+
return 'high';
|
|
121
|
+
}
|
|
122
|
+
return 'elevated';
|
|
123
|
+
}
|
|
124
|
+
export { advisoryLevel, directiveHits, looksDirective };
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tool boundary guardrails — the surface where untrusted bytes re-enter the
|
|
3
|
+
* model's context carrying the model's own authority.
|
|
4
|
+
*
|
|
5
|
+
* A tool result is not user text: the model asked for it, so it arrives looking
|
|
6
|
+
* like something the turn already trusts. Remote HTTP and MCP servers control
|
|
7
|
+
* their own response bodies (including their error strings), and a delegated
|
|
8
|
+
* agent answers in prose that reads as authoritative. Everything crossing this
|
|
9
|
+
* boundary is therefore fenced, detected, and labelled with where it came from.
|
|
10
|
+
*
|
|
11
|
+
* @module
|
|
12
|
+
*/
|
|
13
|
+
import type { AdvisoryLevel, GuardrailEvent, GuardrailHit, Provenance, ResolvedGuardrailPolicy, ToolOrigin, TurnTaint, Verdict } from './types.js';
|
|
14
|
+
declare const TOOL_CLOSE = "</tool_data>";
|
|
15
|
+
/** True when a result's bytes came from outside the host's own code. */
|
|
16
|
+
declare function isRemoteOrigin(origin: ToolOrigin): boolean;
|
|
17
|
+
/**
|
|
18
|
+
* Wrap tool output so the model reads it as data and can see where it came from.
|
|
19
|
+
*
|
|
20
|
+
* The origin is on the tag rather than in prose so a result cannot claim a
|
|
21
|
+
* friendlier provenance than it has by writing one into its own body.
|
|
22
|
+
*/
|
|
23
|
+
declare function wrapToolData(text: string, provenance: Provenance, advisory?: AdvisoryLevel, guidance?: string): string;
|
|
24
|
+
export interface GuardedToolText {
|
|
25
|
+
/** Text to hand the model, fenced and redacted. */
|
|
26
|
+
text: string;
|
|
27
|
+
/** Emitted when the guard did anything worth recording. */
|
|
28
|
+
event?: GuardrailEvent;
|
|
29
|
+
/**
|
|
30
|
+
* Directive signals found in the content.
|
|
31
|
+
*
|
|
32
|
+
* Reported, never redacted: legitimate tool output is frequently
|
|
33
|
+
* instruction-shaped, so rewriting on this signal would corrupt real data.
|
|
34
|
+
* These raise the turn's taint instead.
|
|
35
|
+
*/
|
|
36
|
+
suspicious?: GuardrailHit[];
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Compose the model-facing text for a tool result.
|
|
40
|
+
*
|
|
41
|
+
* `finding` is the tool's summary and `data` its structured payload; both reach
|
|
42
|
+
* the model, so both are guarded together rather than only the prose half.
|
|
43
|
+
*/
|
|
44
|
+
declare function composeToolText(finding: string, data: unknown): string;
|
|
45
|
+
/**
|
|
46
|
+
* Guard one tool result on its way into the model's context.
|
|
47
|
+
*
|
|
48
|
+
* Local host tools are still detected — a host tool reading a database returns
|
|
49
|
+
* data the host did not write — but only remote origins are fenced, because
|
|
50
|
+
* fencing a local tool's output would change prompts hosts have already tuned.
|
|
51
|
+
*/
|
|
52
|
+
declare function guardToolResult(finding: string, data: unknown, provenance: Provenance, policy: ResolvedGuardrailPolicy, callableTools?: readonly string[]): GuardedToolText;
|
|
53
|
+
/**
|
|
54
|
+
* Guard a tool failure message.
|
|
55
|
+
*
|
|
56
|
+
* A remote server authors its own error strings, so an unguarded failure message
|
|
57
|
+
* is the cleanest injection path across this boundary: it reaches the model
|
|
58
|
+
* verbatim and is framed by the kernel as a system report.
|
|
59
|
+
*/
|
|
60
|
+
declare function guardToolFailureText(message: string, provenance: Provenance, policy: ResolvedGuardrailPolicy): GuardedToolText;
|
|
61
|
+
/**
|
|
62
|
+
* Inspect model-supplied tool arguments before the call runs.
|
|
63
|
+
*
|
|
64
|
+
* Arguments are model-authored, so the risk is not instruction smuggling but
|
|
65
|
+
* exfiltration: a credential lifted from context and posted outward as a
|
|
66
|
+
* parameter. Detection reports rather than rewrites — silently altering a tool
|
|
67
|
+
* argument would make the call succeed against something the model did not ask
|
|
68
|
+
* for.
|
|
69
|
+
*/
|
|
70
|
+
declare function inspectToolArguments(args: unknown, policy: ResolvedGuardrailPolicy): Verdict;
|
|
71
|
+
/** Guardrail event for a flagged tool call, for the runner to emit. */
|
|
72
|
+
declare function toolCallEvent(verdict: Verdict, provenance: Provenance): GuardrailEvent | undefined;
|
|
73
|
+
/**
|
|
74
|
+
* Record a tool result against the turn's taint.
|
|
75
|
+
*
|
|
76
|
+
* Only remote origins taint: a local host tool returns bytes the host's own code
|
|
77
|
+
* produced, and treating those as attacker-influenceable would make the gate
|
|
78
|
+
* useless in practice.
|
|
79
|
+
*/
|
|
80
|
+
declare function recordTaint(taint: TurnTaint | undefined, provenance: Provenance, suspicious?: GuardrailHit[]): TurnTaint;
|
|
81
|
+
/** True when the turn has already read attacker-influenceable content. */
|
|
82
|
+
declare function isTainted(taint: TurnTaint | undefined): boolean;
|
|
83
|
+
/** True when remote content this turn read looked like it was steering the agent. */
|
|
84
|
+
declare function isSuspicious(taint: TurnTaint | undefined): boolean;
|
|
85
|
+
/**
|
|
86
|
+
* Decide whether a tool call may proceed given what the turn has already read.
|
|
87
|
+
*
|
|
88
|
+
* A `flag` says the call is happening on a tainted turn and is worth recording; a
|
|
89
|
+
* `block` says the profile asked for it to be refused. Reporting happens whether
|
|
90
|
+
* or not enforcement is configured, so the risk is visible before a host opts in.
|
|
91
|
+
*/
|
|
92
|
+
declare function checkTaintGate(taint: TurnTaint | undefined, access: string, policy: ResolvedGuardrailPolicy): Verdict;
|
|
93
|
+
export { checkTaintGate, composeToolText, guardToolFailureText, guardToolResult, inspectToolArguments, isRemoteOrigin, isSuspicious, isTainted, recordTaint, TOOL_CLOSE, toolCallEvent, wrapToolData, };
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tool boundary guardrails — the surface where untrusted bytes re-enter the
|
|
3
|
+
* model's context carrying the model's own authority.
|
|
4
|
+
*
|
|
5
|
+
* A tool result is not user text: the model asked for it, so it arrives looking
|
|
6
|
+
* like something the turn already trusts. Remote HTTP and MCP servers control
|
|
7
|
+
* their own response bodies (including their error strings), and a delegated
|
|
8
|
+
* agent answers in prose that reads as authoritative. Everything crossing this
|
|
9
|
+
* boundary is therefore fenced, detected, and labelled with where it came from.
|
|
10
|
+
*
|
|
11
|
+
* @module
|
|
12
|
+
*/
|
|
13
|
+
import { lexiconText } from './lexicon.js';
|
|
14
|
+
import { detectionForTrust } from './policy.js';
|
|
15
|
+
import { sanitizeText } from './sanitize.js';
|
|
16
|
+
import { textForScan } from './serialize.js';
|
|
17
|
+
import { advisoryLevel, directiveHits } from './tool-directives.js';
|
|
18
|
+
const TOOL_CLOSE = '</tool_data>';
|
|
19
|
+
const TOOL_OPEN = '<tool_data';
|
|
20
|
+
/** Origins whose bytes the host does not author and cannot vouch for. */
|
|
21
|
+
const REMOTE_ORIGINS = new Set(['http', 'mcp', 'delegated']);
|
|
22
|
+
/** True when a result's bytes came from outside the host's own code. */
|
|
23
|
+
function isRemoteOrigin(origin) {
|
|
24
|
+
return REMOTE_ORIGINS.has(origin);
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Strip fence markers a tool result tried to forge before wrapping it.
|
|
28
|
+
* Linear scan — avoids polynomial regex on forged `<tool_data…>` runs.
|
|
29
|
+
*/
|
|
30
|
+
function stripToolFences(text) {
|
|
31
|
+
const lower = text.toLowerCase();
|
|
32
|
+
let out = '';
|
|
33
|
+
let i = 0;
|
|
34
|
+
while (i < text.length) {
|
|
35
|
+
const openAt = lower.indexOf(TOOL_OPEN, i);
|
|
36
|
+
const closeAt = lower.indexOf(TOOL_CLOSE, i);
|
|
37
|
+
let next = -1;
|
|
38
|
+
let kind = null;
|
|
39
|
+
if (openAt >= 0 && (closeAt < 0 || openAt <= closeAt)) {
|
|
40
|
+
next = openAt;
|
|
41
|
+
kind = 'open';
|
|
42
|
+
}
|
|
43
|
+
else if (closeAt >= 0) {
|
|
44
|
+
next = closeAt;
|
|
45
|
+
kind = 'close';
|
|
46
|
+
}
|
|
47
|
+
if (next < 0 || kind === null) {
|
|
48
|
+
out += text.slice(i);
|
|
49
|
+
break;
|
|
50
|
+
}
|
|
51
|
+
out += text.slice(i, next);
|
|
52
|
+
if (kind === 'close') {
|
|
53
|
+
i = next + TOOL_CLOSE.length;
|
|
54
|
+
continue;
|
|
55
|
+
}
|
|
56
|
+
const gt = text.indexOf('>', next + TOOL_OPEN.length);
|
|
57
|
+
if (gt < 0) {
|
|
58
|
+
out += text.slice(next);
|
|
59
|
+
break;
|
|
60
|
+
}
|
|
61
|
+
i = gt + 1;
|
|
62
|
+
}
|
|
63
|
+
return out;
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* Wrap tool output so the model reads it as data and can see where it came from.
|
|
67
|
+
*
|
|
68
|
+
* The origin is on the tag rather than in prose so a result cannot claim a
|
|
69
|
+
* friendlier provenance than it has by writing one into its own body.
|
|
70
|
+
*/
|
|
71
|
+
function wrapToolData(text, provenance, advisory = 'none', guidance) {
|
|
72
|
+
const attrs = `tool="${provenance.tool}" origin="${provenance.origin}"` +
|
|
73
|
+
(advisory === 'none' ? '' : ` advisory="${advisory}"`);
|
|
74
|
+
const notice = advisory === 'none' ? '' : `${advisoryNotice(advisory)}${guidance ? ` ${guidance}` : ''}\n`;
|
|
75
|
+
return `<${'tool_data'} ${attrs}>\n${notice}${stripToolFences(text)}\n${TOOL_CLOSE}`;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* The kernel's own statement of what it observed.
|
|
79
|
+
*
|
|
80
|
+
* Deliberately an observation, not an instruction: what the agent should do about
|
|
81
|
+
* it is product behaviour, supplied by the host as `advisoryGuidance`. Emitted
|
|
82
|
+
* only when signals fired, so it stays rare enough to carry weight — a warning on
|
|
83
|
+
* every fetch is one the model learns to skip.
|
|
84
|
+
*/
|
|
85
|
+
function advisoryNotice(advisory) {
|
|
86
|
+
return lexiconText(advisory === 'high' ? 'advisory.notice_high' : 'advisory.notice_elevated');
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Compose the model-facing text for a tool result.
|
|
90
|
+
*
|
|
91
|
+
* `finding` is the tool's summary and `data` its structured payload; both reach
|
|
92
|
+
* the model, so both are guarded together rather than only the prose half.
|
|
93
|
+
*/
|
|
94
|
+
function composeToolText(finding, data) {
|
|
95
|
+
if (data === undefined) {
|
|
96
|
+
return finding;
|
|
97
|
+
}
|
|
98
|
+
const rendered = textForScan(data);
|
|
99
|
+
if (rendered.unscannable) {
|
|
100
|
+
return finding;
|
|
101
|
+
}
|
|
102
|
+
return `${finding}\n${rendered.text}`;
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* Guard one tool result on its way into the model's context.
|
|
106
|
+
*
|
|
107
|
+
* Local host tools are still detected — a host tool reading a database returns
|
|
108
|
+
* data the host did not write — but only remote origins are fenced, because
|
|
109
|
+
* fencing a local tool's output would change prompts hosts have already tuned.
|
|
110
|
+
*/
|
|
111
|
+
function guardToolResult(finding, data, provenance, policy, callableTools = []) {
|
|
112
|
+
const composed = composeToolText(finding, data);
|
|
113
|
+
const options = detectionForTrust(policy, 'untrusted');
|
|
114
|
+
const redacted = sanitizeText(composed, options);
|
|
115
|
+
const changed = redacted !== composed;
|
|
116
|
+
// Directive detection runs on remote content only: a local tool's output is
|
|
117
|
+
// bytes the host's own code produced.
|
|
118
|
+
const remote = isRemoteOrigin(provenance.origin);
|
|
119
|
+
const suspicious = remote ? directiveHits(composed, callableTools) : [];
|
|
120
|
+
const advisory = advisoryLevel(suspicious);
|
|
121
|
+
const fenced = remote
|
|
122
|
+
? wrapToolData(redacted, provenance, advisory, policy.taint?.advisoryGuidance)
|
|
123
|
+
: redacted;
|
|
124
|
+
const hits = [
|
|
125
|
+
...(changed ? [{ rule: 'tool_result.redacted', severity: 'medium' }] : []),
|
|
126
|
+
...suspicious,
|
|
127
|
+
];
|
|
128
|
+
if (hits.length === 0) {
|
|
129
|
+
return { text: fenced };
|
|
130
|
+
}
|
|
131
|
+
return {
|
|
132
|
+
text: fenced,
|
|
133
|
+
...(suspicious.length > 0 ? { suspicious } : {}),
|
|
134
|
+
event: {
|
|
135
|
+
stage: 'tool_result',
|
|
136
|
+
trust: 'untrusted',
|
|
137
|
+
// Redaction changed the text; directive signals only annotate it.
|
|
138
|
+
action: changed ? 'redact' : 'flag',
|
|
139
|
+
hits,
|
|
140
|
+
provenance,
|
|
141
|
+
},
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Guard a tool failure message.
|
|
146
|
+
*
|
|
147
|
+
* A remote server authors its own error strings, so an unguarded failure message
|
|
148
|
+
* is the cleanest injection path across this boundary: it reaches the model
|
|
149
|
+
* verbatim and is framed by the kernel as a system report.
|
|
150
|
+
*/
|
|
151
|
+
function guardToolFailureText(message, provenance, policy) {
|
|
152
|
+
const options = detectionForTrust(policy, 'untrusted');
|
|
153
|
+
const redacted = sanitizeText(stripToolFences(message), options);
|
|
154
|
+
if (redacted === message) {
|
|
155
|
+
return { text: redacted };
|
|
156
|
+
}
|
|
157
|
+
return {
|
|
158
|
+
text: redacted,
|
|
159
|
+
event: {
|
|
160
|
+
stage: 'tool_result',
|
|
161
|
+
trust: 'untrusted',
|
|
162
|
+
action: 'redact',
|
|
163
|
+
hits: [{ rule: 'tool_failure.redacted', severity: 'medium' }],
|
|
164
|
+
provenance,
|
|
165
|
+
},
|
|
166
|
+
};
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Inspect model-supplied tool arguments before the call runs.
|
|
170
|
+
*
|
|
171
|
+
* Arguments are model-authored, so the risk is not instruction smuggling but
|
|
172
|
+
* exfiltration: a credential lifted from context and posted outward as a
|
|
173
|
+
* parameter. Detection reports rather than rewrites — silently altering a tool
|
|
174
|
+
* argument would make the call succeed against something the model did not ask
|
|
175
|
+
* for.
|
|
176
|
+
*/
|
|
177
|
+
function inspectToolArguments(args, policy) {
|
|
178
|
+
if (!policy.redactSensitive) {
|
|
179
|
+
return { action: 'allow' };
|
|
180
|
+
}
|
|
181
|
+
const rendered = textForScan(args);
|
|
182
|
+
if (rendered.unscannable) {
|
|
183
|
+
return { action: 'allow' };
|
|
184
|
+
}
|
|
185
|
+
const options = { sanitizeInput: false, redactSensitive: true };
|
|
186
|
+
if (sanitizeText(rendered.text, options) === rendered.text) {
|
|
187
|
+
return { action: 'allow' };
|
|
188
|
+
}
|
|
189
|
+
return {
|
|
190
|
+
action: 'flag',
|
|
191
|
+
hits: [{ rule: 'tool_call.sensitive-argument', severity: 'high' }],
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
/** Guardrail event for a flagged tool call, for the runner to emit. */
|
|
195
|
+
function toolCallEvent(verdict, provenance) {
|
|
196
|
+
if (verdict.action === 'allow') {
|
|
197
|
+
return undefined;
|
|
198
|
+
}
|
|
199
|
+
return {
|
|
200
|
+
stage: 'tool_call',
|
|
201
|
+
trust: 'untrusted',
|
|
202
|
+
action: verdict.action,
|
|
203
|
+
hits: verdict.hits,
|
|
204
|
+
provenance,
|
|
205
|
+
};
|
|
206
|
+
}
|
|
207
|
+
/**
|
|
208
|
+
* Record a tool result against the turn's taint.
|
|
209
|
+
*
|
|
210
|
+
* Only remote origins taint: a local host tool returns bytes the host's own code
|
|
211
|
+
* produced, and treating those as attacker-influenceable would make the gate
|
|
212
|
+
* useless in practice.
|
|
213
|
+
*/
|
|
214
|
+
function recordTaint(taint, provenance, suspicious = []) {
|
|
215
|
+
const sources = taint?.sources ?? [];
|
|
216
|
+
const prior = taint?.suspicious ?? [];
|
|
217
|
+
if (!isRemoteOrigin(provenance.origin)) {
|
|
218
|
+
return { sources, suspicious: prior };
|
|
219
|
+
}
|
|
220
|
+
return { sources: [...sources, provenance], suspicious: [...prior, ...suspicious] };
|
|
221
|
+
}
|
|
222
|
+
/** True when the turn has already read attacker-influenceable content. */
|
|
223
|
+
function isTainted(taint) {
|
|
224
|
+
return (taint?.sources.length ?? 0) > 0;
|
|
225
|
+
}
|
|
226
|
+
/** True when remote content this turn read looked like it was steering the agent. */
|
|
227
|
+
function isSuspicious(taint) {
|
|
228
|
+
return (taint?.suspicious.length ?? 0) > 0;
|
|
229
|
+
}
|
|
230
|
+
/** Capability rank, so a single threshold can express "this and anything worse". */
|
|
231
|
+
const CAPABILITY_RANK = {
|
|
232
|
+
'read-only': 0,
|
|
233
|
+
'read-write': 1,
|
|
234
|
+
destructive: 2,
|
|
235
|
+
};
|
|
236
|
+
const GATE_RANK = {
|
|
237
|
+
off: Number.POSITIVE_INFINITY,
|
|
238
|
+
destructive: 2,
|
|
239
|
+
write: 1,
|
|
240
|
+
};
|
|
241
|
+
/**
|
|
242
|
+
* Decide whether a tool call may proceed given what the turn has already read.
|
|
243
|
+
*
|
|
244
|
+
* A `flag` says the call is happening on a tainted turn and is worth recording; a
|
|
245
|
+
* `block` says the profile asked for it to be refused. Reporting happens whether
|
|
246
|
+
* or not enforcement is configured, so the risk is visible before a host opts in.
|
|
247
|
+
*/
|
|
248
|
+
function checkTaintGate(taint, access, policy) {
|
|
249
|
+
if (!isTainted(taint)) {
|
|
250
|
+
return { action: 'allow' };
|
|
251
|
+
}
|
|
252
|
+
const rank = CAPABILITY_RANK[access] ?? 0;
|
|
253
|
+
if (rank === 0) {
|
|
254
|
+
return { action: 'allow' };
|
|
255
|
+
}
|
|
256
|
+
const suspicious = isSuspicious(taint);
|
|
257
|
+
const hits = [
|
|
258
|
+
{
|
|
259
|
+
rule: suspicious ? 'tool_call.steered-turn' : 'tool_call.tainted-turn',
|
|
260
|
+
severity: 'high',
|
|
261
|
+
},
|
|
262
|
+
];
|
|
263
|
+
// Gated on origin alone. Whether the content looked directive changes the rule
|
|
264
|
+
// reported, never whether the call is refused.
|
|
265
|
+
if (rank < GATE_RANK[policy.taint?.afterRemoteRead ?? 'off']) {
|
|
266
|
+
return { action: 'flag', hits };
|
|
267
|
+
}
|
|
268
|
+
const read = taint?.sources.map((s) => s.tool).join(', ') ?? '';
|
|
269
|
+
const reason = lexiconText(suspicious ? 'taint.reason_steered' : 'taint.reason_tainted');
|
|
270
|
+
return {
|
|
271
|
+
action: 'block',
|
|
272
|
+
hits,
|
|
273
|
+
rejection: lexiconText('taint.blocked', { access, sources: read, reason }),
|
|
274
|
+
};
|
|
275
|
+
}
|
|
276
|
+
export { checkTaintGate, composeToolText, guardToolFailureText, guardToolResult, inspectToolArguments, isRemoteOrigin, isSuspicious, isTainted, recordTaint, TOOL_CLOSE, toolCallEvent, wrapToolData, };
|