theorum 0.1.15 → 1.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (305) hide show
  1. package/README.md +241 -98
  2. package/esm/mod.d.ts +57 -28
  3. package/esm/mod.js +43 -23
  4. package/esm/src/cli/commands/bench.js +18 -18
  5. package/esm/src/cli/commands/fuzz-canary.d.ts +13 -0
  6. package/esm/src/cli/commands/fuzz-canary.js +191 -0
  7. package/esm/src/cli/commands/fuzz-guardrails.d.ts +3 -5
  8. package/esm/src/cli/commands/fuzz-guardrails.js +4 -581
  9. package/esm/src/cli/commands/guardrails-eval.d.ts +14 -0
  10. package/esm/src/cli/commands/guardrails-eval.js +15 -0
  11. package/esm/src/cli/commands/profile.js +35 -15
  12. package/esm/src/cli/commands/run.d.ts +3 -0
  13. package/esm/src/cli/commands/run.js +23 -32
  14. package/esm/src/cli/commands/test.d.ts +10 -1
  15. package/esm/src/cli/commands/test.js +34 -34
  16. package/esm/src/cli/event-log.d.ts +19 -0
  17. package/esm/src/cli/event-log.js +147 -0
  18. package/esm/src/cli/index.js +57 -11
  19. package/esm/src/cli/matrix/synthesizer.d.ts +10 -12
  20. package/esm/src/cli/matrix/synthesizer.js +45 -118
  21. package/esm/src/guardrails/canary-gate.d.ts +21 -0
  22. package/esm/src/guardrails/canary-gate.js +32 -0
  23. package/esm/src/guardrails/canary.d.ts +34 -0
  24. package/esm/src/guardrails/canary.js +150 -0
  25. package/esm/src/guardrails/corpus/canary-egress-attacks.d.ts +17 -0
  26. package/esm/src/guardrails/corpus/canary-egress-attacks.js +151 -0
  27. package/esm/src/guardrails/corpus/fuzz-inbound.d.ts +11 -0
  28. package/esm/src/guardrails/corpus/fuzz-inbound.js +213 -0
  29. package/esm/src/guardrails/corpus/inbound-payloads.d.ts +10 -0
  30. package/esm/src/guardrails/corpus/inbound-payloads.js +125 -0
  31. package/esm/src/guardrails/corpus/live-attacks.d.ts +20 -0
  32. package/esm/src/guardrails/corpus/live-attacks.js +231 -0
  33. package/esm/src/guardrails/corpus/mod.d.ts +14 -0
  34. package/esm/src/guardrails/corpus/mod.js +11 -0
  35. package/esm/src/guardrails/corpus/secrets.d.ts +17 -0
  36. package/esm/src/guardrails/corpus/secrets.js +17 -0
  37. package/esm/src/guardrails/corpus/strings.d.ts +28 -0
  38. package/esm/src/guardrails/corpus/strings.js +34 -0
  39. package/esm/src/guardrails/corpus/types.d.ts +38 -0
  40. package/esm/src/guardrails/corpus/types.js +6 -0
  41. package/esm/src/guardrails/egress.d.ts +32 -0
  42. package/esm/src/guardrails/egress.js +87 -0
  43. package/esm/src/guardrails/error.d.ts +14 -23
  44. package/esm/src/guardrails/error.js +87 -76
  45. package/esm/src/guardrails/eval/corpus.d.ts +108 -0
  46. package/esm/src/guardrails/eval/corpus.js +978 -0
  47. package/esm/src/guardrails/eval/mod.d.ts +51 -0
  48. package/esm/src/guardrails/eval/mod.js +133 -0
  49. package/esm/src/guardrails/eval/score.d.ts +66 -0
  50. package/esm/src/guardrails/eval/score.js +114 -0
  51. package/esm/src/guardrails/events.d.ts +25 -0
  52. package/esm/src/guardrails/events.js +56 -0
  53. package/esm/src/guardrails/hits.d.ts +24 -0
  54. package/esm/src/guardrails/hits.js +45 -0
  55. package/esm/src/guardrails/injection.js +28 -5
  56. package/esm/src/guardrails/lexicon.d.ts +39 -0
  57. package/esm/src/guardrails/lexicon.js +200 -0
  58. package/esm/src/guardrails/live-outbound-gate.d.ts +41 -0
  59. package/esm/src/guardrails/live-outbound-gate.js +222 -0
  60. package/esm/src/guardrails/mod.d.ts +30 -6
  61. package/esm/src/guardrails/mod.js +20 -5
  62. package/esm/src/guardrails/network.d.ts +19 -0
  63. package/esm/src/guardrails/network.js +234 -0
  64. package/esm/src/guardrails/policy.d.ts +35 -0
  65. package/esm/src/guardrails/policy.js +50 -0
  66. package/esm/src/guardrails/progressive-yield.d.ts +51 -0
  67. package/esm/src/guardrails/progressive-yield.js +98 -0
  68. package/esm/src/guardrails/quota.d.ts +17 -3
  69. package/esm/src/guardrails/quota.js +18 -4
  70. package/esm/src/guardrails/sanitize.d.ts +45 -19
  71. package/esm/src/guardrails/sanitize.js +177 -94
  72. package/esm/src/guardrails/sensitive.js +2 -1
  73. package/esm/src/guardrails/serialize.d.ts +35 -0
  74. package/esm/src/guardrails/serialize.js +58 -0
  75. package/esm/src/guardrails/testing.d.ts +17 -0
  76. package/esm/src/guardrails/testing.js +13 -0
  77. package/esm/src/guardrails/theorum-error.d.ts +12 -0
  78. package/esm/src/guardrails/theorum-error.js +15 -0
  79. package/esm/src/guardrails/tool-directives.d.ts +48 -0
  80. package/esm/src/guardrails/tool-directives.js +124 -0
  81. package/esm/src/guardrails/tool-result.d.ts +93 -0
  82. package/esm/src/guardrails/tool-result.js +276 -0
  83. package/esm/src/guardrails/types.d.ts +291 -0
  84. package/esm/src/guardrails/types.js +72 -0
  85. package/esm/src/host/client-turn.d.ts +19 -0
  86. package/esm/src/host/client-turn.js +36 -0
  87. package/esm/src/host/mint-trace.d.ts +1 -1
  88. package/esm/src/host/mod.d.ts +5 -3
  89. package/esm/src/host/mod.js +4 -3
  90. package/esm/src/kernel/auth/crypto.d.ts +42 -0
  91. package/esm/src/kernel/auth/crypto.js +106 -0
  92. package/esm/src/kernel/auth/mod.d.ts +11 -0
  93. package/esm/src/kernel/auth/mod.js +11 -0
  94. package/esm/src/kernel/auth/oauth.d.ts +47 -0
  95. package/esm/src/kernel/auth/oauth.js +278 -0
  96. package/esm/src/kernel/auth/types.d.ts +133 -0
  97. package/esm/src/kernel/auth/types.js +13 -0
  98. package/esm/src/kernel/engine/delta.d.ts +24 -2
  99. package/esm/src/kernel/engine/delta.js +478 -39
  100. package/esm/src/kernel/engine/live-inbound.d.ts +21 -0
  101. package/esm/src/kernel/engine/live-inbound.js +31 -0
  102. package/esm/src/kernel/engine/live-ingress.d.ts +19 -0
  103. package/esm/src/kernel/engine/live-ingress.js +47 -0
  104. package/esm/src/kernel/engine/repair.js +13 -12
  105. package/esm/src/kernel/engine/runner/gates.d.ts +1 -1
  106. package/esm/src/kernel/engine/runner/gates.js +130 -43
  107. package/esm/src/kernel/engine/runner/mod.d.ts +6 -4
  108. package/esm/src/kernel/engine/runner/mod.js +192 -53
  109. package/esm/src/kernel/engine/runner/schema-validation.js +3 -3
  110. package/esm/src/kernel/engine/runner/stages.d.ts +39 -0
  111. package/esm/src/kernel/engine/runner/stages.js +89 -0
  112. package/esm/src/kernel/engine/runner/state.d.ts +31 -0
  113. package/esm/src/kernel/engine/runner/steps.d.ts +1 -1
  114. package/esm/src/kernel/engine/runner/steps.js +244 -43
  115. package/esm/src/kernel/engine/runner/stream.d.ts +9 -3
  116. package/esm/src/kernel/engine/runner/stream.js +140 -44
  117. package/esm/src/kernel/engine/session/mod.d.ts +25 -0
  118. package/esm/src/kernel/engine/session/mod.js +557 -0
  119. package/esm/src/kernel/interaction-parts.d.ts +14 -0
  120. package/esm/src/kernel/interaction-parts.js +23 -0
  121. package/esm/src/kernel/mod.d.ts +21 -10
  122. package/esm/src/kernel/mod.js +11 -8
  123. package/esm/src/kernel/profile-graph.d.ts +159 -0
  124. package/esm/src/kernel/profile-graph.js +156 -0
  125. package/esm/src/kernel/registry/attachments.d.ts +12 -10
  126. package/esm/src/kernel/registry/attachments.js +33 -27
  127. package/esm/src/kernel/registry/catalog.d.ts +25 -24
  128. package/esm/src/kernel/registry/catalog.js +60 -101
  129. package/esm/src/kernel/registry/ingress.d.ts +9 -4
  130. package/esm/src/kernel/registry/ingress.js +97 -75
  131. package/esm/src/kernel/registry/profile-outputs.d.ts +4 -0
  132. package/esm/src/kernel/registry/profile-outputs.js +8 -0
  133. package/esm/src/kernel/registry/profiles.d.ts +55 -12
  134. package/esm/src/kernel/registry/profiles.js +413 -73
  135. package/esm/src/kernel/registry/provider-request.js +13 -7
  136. package/esm/src/kernel/registry/resolve.d.ts +8 -8
  137. package/esm/src/kernel/registry/resolve.js +169 -154
  138. package/esm/src/kernel/registry/schemas.js +1 -1
  139. package/esm/src/kernel/registry/sole-model.d.ts +8 -0
  140. package/esm/src/kernel/registry/sole-model.js +10 -0
  141. package/esm/src/kernel/registry/system-prompt.d.ts +10 -0
  142. package/esm/src/kernel/registry/system-prompt.js +40 -0
  143. package/esm/src/kernel/registry/system-role.d.ts +8 -0
  144. package/esm/src/kernel/registry/system-role.js +14 -0
  145. package/esm/src/kernel/registry/vault.d.ts +12 -7
  146. package/esm/src/kernel/registry/vault.js +32 -10
  147. package/esm/src/kernel/schema.d.ts +231 -0
  148. package/esm/src/kernel/schema.js +607 -0
  149. package/esm/src/kernel/stages.d.ts +175 -0
  150. package/esm/src/kernel/stages.js +476 -0
  151. package/esm/src/kernel/stop.d.ts +78 -19
  152. package/esm/src/kernel/stop.js +51 -16
  153. package/esm/src/kernel/tools/events.d.ts +41 -0
  154. package/esm/src/kernel/tools/events.js +71 -0
  155. package/esm/src/kernel/tools/execute.d.ts +84 -0
  156. package/esm/src/kernel/tools/execute.js +614 -0
  157. package/esm/src/kernel/tools/harness.d.ts +8 -0
  158. package/esm/src/kernel/tools/harness.js +46 -0
  159. package/esm/src/kernel/tools/invoke.d.ts +10 -0
  160. package/esm/src/kernel/tools/invoke.js +101 -0
  161. package/esm/src/kernel/tools/mod.d.ts +13 -0
  162. package/esm/src/kernel/tools/mod.js +11 -0
  163. package/esm/src/kernel/tools/permission.d.ts +15 -0
  164. package/esm/src/kernel/tools/permission.js +47 -0
  165. package/esm/src/kernel/tools/project.d.ts +12 -0
  166. package/esm/src/kernel/tools/project.js +36 -0
  167. package/esm/src/kernel/tools/registry.d.ts +23 -0
  168. package/esm/src/kernel/tools/registry.js +81 -0
  169. package/esm/src/kernel/tools/remote.d.ts +94 -0
  170. package/esm/src/kernel/tools/remote.js +577 -0
  171. package/esm/src/kernel/tools/resolve.d.ts +39 -0
  172. package/esm/src/kernel/tools/resolve.js +283 -0
  173. package/esm/src/kernel/tools/schema.d.ts +15 -0
  174. package/esm/src/kernel/tools/schema.js +176 -0
  175. package/esm/src/kernel/tools/stage-run.d.ts +105 -0
  176. package/esm/src/kernel/tools/stage-run.js +155 -0
  177. package/esm/src/kernel/tools/types.d.ts +394 -0
  178. package/esm/src/kernel/tools/types.js +9 -0
  179. package/esm/src/kernel/types.d.ts +540 -256
  180. package/esm/src/kernel/util/find-last.d.ts +2 -0
  181. package/esm/src/kernel/util/find-last.js +10 -0
  182. package/esm/src/observability/destinations.d.ts +31 -0
  183. package/esm/src/observability/destinations.js +67 -0
  184. package/esm/src/observability/mod.d.ts +10 -3
  185. package/esm/src/observability/mod.js +6 -2
  186. package/esm/src/observability/policy.d.ts +27 -0
  187. package/esm/src/observability/policy.js +80 -0
  188. package/esm/src/observability/resolve-policy.d.ts +16 -0
  189. package/esm/src/observability/resolve-policy.js +64 -0
  190. package/esm/src/observability/trace-attach.d.ts +8 -4
  191. package/esm/src/observability/trace-attach.js +50 -29
  192. package/esm/src/observability/trace-record.d.ts +23 -13
  193. package/esm/src/observability/trace-record.js +96 -39
  194. package/esm/src/observability/trace-sink.d.ts +19 -0
  195. package/esm/src/observability/trace-sink.js +10 -0
  196. package/esm/src/observability/trace-usage.d.ts +10 -3
  197. package/esm/src/observability/trace-usage.js +70 -17
  198. package/esm/src/observability/trace.d.ts +18 -7
  199. package/esm/src/observability/trace.js +34 -17
  200. package/esm/src/observability/types.d.ts +113 -0
  201. package/esm/src/observability/types.js +11 -0
  202. package/esm/src/presets/google/speech-voices.d.ts +11 -0
  203. package/esm/src/presets/google/speech-voices.js +41 -0
  204. package/esm/src/presets/google.d.ts +36 -24
  205. package/esm/src/presets/google.js +50 -63
  206. package/esm/src/presets/mod.d.ts +2 -2
  207. package/esm/src/presets/mod.js +1 -1
  208. package/esm/src/providers/create-provider.d.ts +20 -17
  209. package/esm/src/providers/create-provider.js +72 -26
  210. package/esm/src/providers/google/interactions/framing.d.ts +23 -0
  211. package/esm/src/providers/google/interactions/framing.js +269 -0
  212. package/esm/src/providers/google/interactions/mod.d.ts +7 -0
  213. package/esm/src/providers/google/interactions/mod.js +7 -0
  214. package/esm/src/providers/google/interactions/stream.d.ts +83 -0
  215. package/esm/src/providers/google/interactions/stream.js +588 -0
  216. package/esm/src/providers/google/keys.d.ts +26 -0
  217. package/esm/src/providers/{keys.js → google/keys.js} +19 -31
  218. package/esm/src/providers/google/live/framing.d.ts +49 -0
  219. package/esm/src/providers/google/live/framing.js +552 -0
  220. package/esm/src/providers/google/live/openapi-schema.d.ts +6 -0
  221. package/esm/src/providers/google/live/openapi-schema.js +46 -0
  222. package/esm/src/providers/google/live/session.d.ts +25 -0
  223. package/esm/src/providers/google/live/session.js +134 -0
  224. package/esm/src/providers/google/live/stream.d.ts +45 -0
  225. package/esm/src/providers/google/live/stream.js +214 -0
  226. package/esm/src/providers/google/urls.d.ts +6 -0
  227. package/esm/src/providers/google/urls.js +6 -0
  228. package/esm/src/providers/local/local.d.ts +30 -0
  229. package/esm/src/providers/{local.js → local/local.js} +66 -126
  230. package/esm/src/providers/local/mod.d.ts +9 -0
  231. package/esm/src/providers/local/mod.js +9 -0
  232. package/esm/src/providers/mod.d.ts +6 -3
  233. package/esm/src/providers/mod.js +3 -1
  234. package/esm/src/providers/openrouter/cache-control.d.ts +24 -0
  235. package/esm/src/providers/openrouter/cache-control.js +23 -0
  236. package/esm/src/providers/openrouter/chat.d.ts +107 -0
  237. package/esm/src/providers/{openrouter.js → openrouter/chat.js} +117 -231
  238. package/esm/src/providers/openrouter/image.d.ts +34 -0
  239. package/esm/src/providers/openrouter/image.js +275 -0
  240. package/esm/src/providers/openrouter/openai/chat-payload.d.ts +24 -0
  241. package/esm/src/providers/openrouter/openai/chat-payload.js +82 -0
  242. package/esm/src/providers/openrouter/openai/compat.d.ts +53 -0
  243. package/esm/src/providers/openrouter/openai/compat.js +213 -0
  244. package/esm/src/providers/openrouter/openai/image-payload.d.ts +18 -0
  245. package/esm/src/providers/openrouter/openai/image-payload.js +90 -0
  246. package/esm/src/providers/openrouter/openai/sdk-messages.d.ts +22 -0
  247. package/esm/src/providers/openrouter/openai/sdk-messages.js +122 -0
  248. package/esm/src/providers/openrouter/resolve-api-key.d.ts +9 -0
  249. package/esm/src/providers/openrouter/resolve-api-key.js +24 -0
  250. package/esm/src/providers/openrouter/speech.d.ts +23 -0
  251. package/esm/src/providers/{speech.js → openrouter/speech.js} +32 -55
  252. package/esm/src/providers/probe.d.ts +1 -0
  253. package/esm/src/providers/probe.js +22 -0
  254. package/esm/src/providers/shared/pcm.d.ts +12 -0
  255. package/esm/src/providers/{pcm.js → shared/pcm.js} +16 -3
  256. package/esm/src/providers/shared/sse.d.ts +18 -0
  257. package/esm/src/providers/shared/sse.js +87 -0
  258. package/esm/src/providers/shared/tool-args.d.ts +17 -0
  259. package/esm/src/providers/shared/tool-args.js +45 -0
  260. package/esm/src/providers/shared/upstream-tap.d.ts +5 -0
  261. package/esm/src/providers/{google-tap.js → shared/upstream-tap.js} +4 -7
  262. package/esm/src/providers/shared/upstream-tape.d.ts +6 -0
  263. package/esm/src/providers/{gemini-tape.js → shared/upstream-tape.js} +12 -22
  264. package/esm/src/providers/types.d.ts +27 -0
  265. package/esm/src/providers/types.js +1 -0
  266. package/package.json +11 -7
  267. package/docs/cli.md +0 -97
  268. package/docs/guardrails.md +0 -178
  269. package/docs/host.md +0 -97
  270. package/docs/kernel.md +0 -404
  271. package/docs/observability.md +0 -105
  272. package/docs/openrouter.md +0 -125
  273. package/docs/presets-google.md +0 -91
  274. package/docs/presets.md +0 -88
  275. package/docs/providers.md +0 -202
  276. package/docs/streaming.md +0 -96
  277. package/esm/src/kernel/engine/boundary.d.ts +0 -10
  278. package/esm/src/kernel/engine/boundary.js +0 -55
  279. package/esm/src/kernel/engine/runner/tools.d.ts +0 -13
  280. package/esm/src/kernel/engine/runner/tools.js +0 -198
  281. package/esm/src/kernel/registry/tools.d.ts +0 -12
  282. package/esm/src/kernel/registry/tools.js +0 -36
  283. package/esm/src/providers/expose-for-tests.d.ts +0 -1
  284. package/esm/src/providers/expose-for-tests.js +0 -25
  285. package/esm/src/providers/gemini-tape.d.ts +0 -2
  286. package/esm/src/providers/google-tap.d.ts +0 -3
  287. package/esm/src/providers/interactions.d.ts +0 -5
  288. package/esm/src/providers/interactions.js +0 -169
  289. package/esm/src/providers/keys.d.ts +0 -19
  290. package/esm/src/providers/local.d.ts +0 -29
  291. package/esm/src/providers/openrouter-mod.d.ts +0 -13
  292. package/esm/src/providers/openrouter-mod.js +0 -12
  293. package/esm/src/providers/openrouter-payload.d.ts +0 -39
  294. package/esm/src/providers/openrouter-payload.js +0 -195
  295. package/esm/src/providers/openrouter.d.ts +0 -15
  296. package/esm/src/providers/pcm.d.ts +0 -7
  297. package/esm/src/providers/provider.d.ts +0 -15
  298. package/esm/src/providers/provider.js +0 -202
  299. package/esm/src/providers/speech.d.ts +0 -23
  300. package/esm/src/providers/sse.d.ts +0 -7
  301. package/esm/src/providers/sse.js +0 -55
  302. package/esm/src/streaming/mod.d.ts +0 -9
  303. package/esm/src/streaming/mod.js +0 -8
  304. /package/esm/src/{streaming → host}/readStreamingJsonStringField.d.ts +0 -0
  305. /package/esm/src/{streaming → host}/readStreamingJsonStringField.js +0 -0
@@ -0,0 +1,124 @@
1
+ /**
2
+ * Directive detection for tool ingress.
3
+ *
4
+ * Tool results carry a different threat than user text. The jailbreak phrasings in
5
+ * `injection.ts` name the thing they attack — "ignore previous instructions",
6
+ * "reveal your system prompt" — and real indirect injection rarely does. It reads
7
+ * like a status update or a helpful next step, and pattern-matching for the word
8
+ * "instructions" misses all of it.
9
+ *
10
+ * What is anomalous in *data* is content that behaves like an instruction: naming
11
+ * a tool the agent can call, issuing an imperative at the agent, or claiming an
12
+ * authority the content does not have.
13
+ *
14
+ * A signal only counts when it co-occurs with a concrete external destination —
15
+ * an address or URL. Directive language alone is far too common in legitimate
16
+ * output to act on. These signals raise the turn's taint rather than rewriting the text. A page
17
+ * documenting an email API legitimately says "call send_email"; redacting that
18
+ * would corrupt content the model needs. Being wrong here should cost a refused
19
+ * write — recoverable and visible — not silently damaged input.
20
+ *
21
+ * @module
22
+ */
23
+ import { normalizeForDetection } from './normalize.js';
24
+ /** Rule ids emitted by tool-ingress directive detection. */
25
+ export const DIRECTIVE_RULES = {
26
+ toolName: 'tool_result.names-callable-tool',
27
+ imperative: 'tool_result.imperative',
28
+ authority: 'tool_result.authority-claim',
29
+ };
30
+ /**
31
+ * Imperatives aimed at an agent rather than a reader.
32
+ *
33
+ * Bounded quantifiers throughout: tool results can be large, and an unbounded
34
+ * scanner on attacker-controlled bytes is a denial-of-service surface.
35
+ */
36
+ const IMPERATIVES = [
37
+ /\b(?:next|final|remaining)\s+steps?\s*[:—-]/gi,
38
+ /\byou\s+(?:must|should|need\s+to|are\s+required\s+to)\s+(?:now\s+)?\w{2,20}/gi,
39
+ /\bto\s+(?:complete|finish|fulfil|fulfill)\s+(?:this|the)\s+\w{2,20}/gi,
40
+ /\b(?:call|invoke|run|execute)\s+(?:the\s+)?[\w.-]{2,40}\s+(?:tool|function)\b/gi,
41
+ /\b(?:send|forward|email|transfer|upload|post)\s+(?:the\s+|all\s+|your\s+)?[\w\s]{2,40}\s+to\s+\S{3,80}/gi,
42
+ /\b(?:delete|remove|drop|purge)\s+(?:the\s+|all\s+)?[\w\s]{2,40}\b/gi,
43
+ ];
44
+ /**
45
+ * A concrete external destination — the thing exfiltration needs and ordinary
46
+ * process prose almost never carries.
47
+ *
48
+ * This is the discriminator. Directive language on its own is everywhere in
49
+ * legitimate tool output: documentation says "you must be an admin", support
50
+ * articles say "to remove a user", status reports say "the user has approved".
51
+ * Measured on a benign sample, directive signals alone produced false positives
52
+ * on most of it. Pairing a signal with a destination removed all of them.
53
+ */
54
+ const EXFIL_TARGET = /\b[\w.+-]{1,64}@[\w-]{1,63}(?:\.[\w-]{1,63}){1,4}\b|\bhttps?:\/\/\S{4,120}/i;
55
+ /** Claims of permission or provenance the content cannot actually hold. */
56
+ const AUTHORITY = [
57
+ /\b(?:the\s+)?user\s+has\s+(?:already\s+)?(?:approved|authorised|authorized|confirmed|requested)\b/gi,
58
+ /\b(?:system|admin|operator)\s+(?:note|notice|message|override|instruction)s?\s*[:—-]/gi,
59
+ /\bon\s+behalf\s+of\s+the\s+(?:user|operator|admin)\b/gi,
60
+ /\bthis\s+(?:is|was)\s+(?:pre-?)?(?:approved|authorised|authorized)\b/gi,
61
+ ];
62
+ function matches(patterns, text) {
63
+ return patterns.some((pattern) => {
64
+ pattern.lastIndex = 0;
65
+ return pattern.test(text);
66
+ });
67
+ }
68
+ /** Word-boundary match for a tool name, escaped so registry names cannot inject. */
69
+ function mentionsTool(text, tool) {
70
+ if (tool.length < 3) {
71
+ return false;
72
+ }
73
+ const escaped = tool.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
74
+ return new RegExp(`(?:^|[^\\w-])${escaped}(?:$|[^\\w-])`, 'i').test(text);
75
+ }
76
+ /**
77
+ * Detect instruction-shaped content in a tool result.
78
+ *
79
+ * `callableTools` is the set the model can actually invoke this turn. A result
80
+ * naming one is the highest-precision signal available — ordinary data has no
81
+ * reason to name the agent's tools, and no generic content filter can check it
82
+ * because it requires the turn's registry.
83
+ */
84
+ function directiveHits(text, callableTools = []) {
85
+ if (!text || !EXFIL_TARGET.test(text)) {
86
+ // No destination, no exfiltration. Action-shaped attacks that carry no target
87
+ // are left to the taint gate, which does not depend on reading the content.
88
+ return [];
89
+ }
90
+ const normalized = normalizeForDetection(text);
91
+ const hits = [];
92
+ // One hit per named tool — several names is a stronger signal than one.
93
+ for (const _tool of callableTools.filter((tool) => mentionsTool(normalized, tool))) {
94
+ hits.push({ rule: DIRECTIVE_RULES.toolName, severity: 'high' });
95
+ }
96
+ if (matches(IMPERATIVES, normalized)) {
97
+ hits.push({ rule: DIRECTIVE_RULES.imperative, severity: 'medium' });
98
+ }
99
+ if (matches(AUTHORITY, normalized)) {
100
+ hits.push({ rule: DIRECTIVE_RULES.authority, severity: 'medium' });
101
+ }
102
+ return hits;
103
+ }
104
+ /** True when a result looked like it was trying to steer the agent. */
105
+ function looksDirective(hits) {
106
+ return hits.length > 0;
107
+ }
108
+ /**
109
+ * Strength of the signals, read off the hits rather than invented.
110
+ *
111
+ * Naming a tool the model can call is the sharpest signal available, so it alone
112
+ * reaches `high`; so does agreement between two different signal kinds.
113
+ */
114
+ function advisoryLevel(hits) {
115
+ if (hits.length === 0) {
116
+ return 'none';
117
+ }
118
+ const kinds = new Set(hits.map((hit) => hit.rule));
119
+ if (kinds.has(DIRECTIVE_RULES.toolName) || kinds.size > 1) {
120
+ return 'high';
121
+ }
122
+ return 'elevated';
123
+ }
124
+ export { advisoryLevel, directiveHits, looksDirective };
@@ -0,0 +1,93 @@
1
+ /**
2
+ * Tool boundary guardrails — the surface where untrusted bytes re-enter the
3
+ * model's context carrying the model's own authority.
4
+ *
5
+ * A tool result is not user text: the model asked for it, so it arrives looking
6
+ * like something the turn already trusts. Remote HTTP and MCP servers control
7
+ * their own response bodies (including their error strings), and a delegated
8
+ * agent answers in prose that reads as authoritative. Everything crossing this
9
+ * boundary is therefore fenced, detected, and labelled with where it came from.
10
+ *
11
+ * @module
12
+ */
13
+ import type { AdvisoryLevel, GuardrailEvent, GuardrailHit, Provenance, ResolvedGuardrailPolicy, ToolOrigin, TurnTaint, Verdict } from './types.js';
14
+ declare const TOOL_CLOSE = "</tool_data>";
15
+ /** True when a result's bytes came from outside the host's own code. */
16
+ declare function isRemoteOrigin(origin: ToolOrigin): boolean;
17
+ /**
18
+ * Wrap tool output so the model reads it as data and can see where it came from.
19
+ *
20
+ * The origin is on the tag rather than in prose so a result cannot claim a
21
+ * friendlier provenance than it has by writing one into its own body.
22
+ */
23
+ declare function wrapToolData(text: string, provenance: Provenance, advisory?: AdvisoryLevel, guidance?: string): string;
24
+ export interface GuardedToolText {
25
+ /** Text to hand the model, fenced and redacted. */
26
+ text: string;
27
+ /** Emitted when the guard did anything worth recording. */
28
+ event?: GuardrailEvent;
29
+ /**
30
+ * Directive signals found in the content.
31
+ *
32
+ * Reported, never redacted: legitimate tool output is frequently
33
+ * instruction-shaped, so rewriting on this signal would corrupt real data.
34
+ * These raise the turn's taint instead.
35
+ */
36
+ suspicious?: GuardrailHit[];
37
+ }
38
+ /**
39
+ * Compose the model-facing text for a tool result.
40
+ *
41
+ * `finding` is the tool's summary and `data` its structured payload; both reach
42
+ * the model, so both are guarded together rather than only the prose half.
43
+ */
44
+ declare function composeToolText(finding: string, data: unknown): string;
45
+ /**
46
+ * Guard one tool result on its way into the model's context.
47
+ *
48
+ * Local host tools are still detected — a host tool reading a database returns
49
+ * data the host did not write — but only remote origins are fenced, because
50
+ * fencing a local tool's output would change prompts hosts have already tuned.
51
+ */
52
+ declare function guardToolResult(finding: string, data: unknown, provenance: Provenance, policy: ResolvedGuardrailPolicy, callableTools?: readonly string[]): GuardedToolText;
53
+ /**
54
+ * Guard a tool failure message.
55
+ *
56
+ * A remote server authors its own error strings, so an unguarded failure message
57
+ * is the cleanest injection path across this boundary: it reaches the model
58
+ * verbatim and is framed by the kernel as a system report.
59
+ */
60
+ declare function guardToolFailureText(message: string, provenance: Provenance, policy: ResolvedGuardrailPolicy): GuardedToolText;
61
+ /**
62
+ * Inspect model-supplied tool arguments before the call runs.
63
+ *
64
+ * Arguments are model-authored, so the risk is not instruction smuggling but
65
+ * exfiltration: a credential lifted from context and posted outward as a
66
+ * parameter. Detection reports rather than rewrites — silently altering a tool
67
+ * argument would make the call succeed against something the model did not ask
68
+ * for.
69
+ */
70
+ declare function inspectToolArguments(args: unknown, policy: ResolvedGuardrailPolicy): Verdict;
71
+ /** Guardrail event for a flagged tool call, for the runner to emit. */
72
+ declare function toolCallEvent(verdict: Verdict, provenance: Provenance): GuardrailEvent | undefined;
73
+ /**
74
+ * Record a tool result against the turn's taint.
75
+ *
76
+ * Only remote origins taint: a local host tool returns bytes the host's own code
77
+ * produced, and treating those as attacker-influenceable would make the gate
78
+ * useless in practice.
79
+ */
80
+ declare function recordTaint(taint: TurnTaint | undefined, provenance: Provenance, suspicious?: GuardrailHit[]): TurnTaint;
81
+ /** True when the turn has already read attacker-influenceable content. */
82
+ declare function isTainted(taint: TurnTaint | undefined): boolean;
83
+ /** True when remote content this turn read looked like it was steering the agent. */
84
+ declare function isSuspicious(taint: TurnTaint | undefined): boolean;
85
+ /**
86
+ * Decide whether a tool call may proceed given what the turn has already read.
87
+ *
88
+ * A `flag` says the call is happening on a tainted turn and is worth recording; a
89
+ * `block` says the profile asked for it to be refused. Reporting happens whether
90
+ * or not enforcement is configured, so the risk is visible before a host opts in.
91
+ */
92
+ declare function checkTaintGate(taint: TurnTaint | undefined, access: string, policy: ResolvedGuardrailPolicy): Verdict;
93
+ export { checkTaintGate, composeToolText, guardToolFailureText, guardToolResult, inspectToolArguments, isRemoteOrigin, isSuspicious, isTainted, recordTaint, TOOL_CLOSE, toolCallEvent, wrapToolData, };
@@ -0,0 +1,276 @@
1
+ /**
2
+ * Tool boundary guardrails — the surface where untrusted bytes re-enter the
3
+ * model's context carrying the model's own authority.
4
+ *
5
+ * A tool result is not user text: the model asked for it, so it arrives looking
6
+ * like something the turn already trusts. Remote HTTP and MCP servers control
7
+ * their own response bodies (including their error strings), and a delegated
8
+ * agent answers in prose that reads as authoritative. Everything crossing this
9
+ * boundary is therefore fenced, detected, and labelled with where it came from.
10
+ *
11
+ * @module
12
+ */
13
+ import { lexiconText } from './lexicon.js';
14
+ import { detectionForTrust } from './policy.js';
15
+ import { sanitizeText } from './sanitize.js';
16
+ import { textForScan } from './serialize.js';
17
+ import { advisoryLevel, directiveHits } from './tool-directives.js';
18
+ const TOOL_CLOSE = '</tool_data>';
19
+ const TOOL_OPEN = '<tool_data';
20
+ /** Origins whose bytes the host does not author and cannot vouch for. */
21
+ const REMOTE_ORIGINS = new Set(['http', 'mcp', 'delegated']);
22
+ /** True when a result's bytes came from outside the host's own code. */
23
+ function isRemoteOrigin(origin) {
24
+ return REMOTE_ORIGINS.has(origin);
25
+ }
26
+ /**
27
+ * Strip fence markers a tool result tried to forge before wrapping it.
28
+ * Linear scan — avoids polynomial regex on forged `<tool_data…>` runs.
29
+ */
30
+ function stripToolFences(text) {
31
+ const lower = text.toLowerCase();
32
+ let out = '';
33
+ let i = 0;
34
+ while (i < text.length) {
35
+ const openAt = lower.indexOf(TOOL_OPEN, i);
36
+ const closeAt = lower.indexOf(TOOL_CLOSE, i);
37
+ let next = -1;
38
+ let kind = null;
39
+ if (openAt >= 0 && (closeAt < 0 || openAt <= closeAt)) {
40
+ next = openAt;
41
+ kind = 'open';
42
+ }
43
+ else if (closeAt >= 0) {
44
+ next = closeAt;
45
+ kind = 'close';
46
+ }
47
+ if (next < 0 || kind === null) {
48
+ out += text.slice(i);
49
+ break;
50
+ }
51
+ out += text.slice(i, next);
52
+ if (kind === 'close') {
53
+ i = next + TOOL_CLOSE.length;
54
+ continue;
55
+ }
56
+ const gt = text.indexOf('>', next + TOOL_OPEN.length);
57
+ if (gt < 0) {
58
+ out += text.slice(next);
59
+ break;
60
+ }
61
+ i = gt + 1;
62
+ }
63
+ return out;
64
+ }
65
+ /**
66
+ * Wrap tool output so the model reads it as data and can see where it came from.
67
+ *
68
+ * The origin is on the tag rather than in prose so a result cannot claim a
69
+ * friendlier provenance than it has by writing one into its own body.
70
+ */
71
+ function wrapToolData(text, provenance, advisory = 'none', guidance) {
72
+ const attrs = `tool="${provenance.tool}" origin="${provenance.origin}"` +
73
+ (advisory === 'none' ? '' : ` advisory="${advisory}"`);
74
+ const notice = advisory === 'none' ? '' : `${advisoryNotice(advisory)}${guidance ? ` ${guidance}` : ''}\n`;
75
+ return `<${'tool_data'} ${attrs}>\n${notice}${stripToolFences(text)}\n${TOOL_CLOSE}`;
76
+ }
77
+ /**
78
+ * The kernel's own statement of what it observed.
79
+ *
80
+ * Deliberately an observation, not an instruction: what the agent should do about
81
+ * it is product behaviour, supplied by the host as `advisoryGuidance`. Emitted
82
+ * only when signals fired, so it stays rare enough to carry weight — a warning on
83
+ * every fetch is one the model learns to skip.
84
+ */
85
+ function advisoryNotice(advisory) {
86
+ return lexiconText(advisory === 'high' ? 'advisory.notice_high' : 'advisory.notice_elevated');
87
+ }
88
+ /**
89
+ * Compose the model-facing text for a tool result.
90
+ *
91
+ * `finding` is the tool's summary and `data` its structured payload; both reach
92
+ * the model, so both are guarded together rather than only the prose half.
93
+ */
94
+ function composeToolText(finding, data) {
95
+ if (data === undefined) {
96
+ return finding;
97
+ }
98
+ const rendered = textForScan(data);
99
+ if (rendered.unscannable) {
100
+ return finding;
101
+ }
102
+ return `${finding}\n${rendered.text}`;
103
+ }
104
+ /**
105
+ * Guard one tool result on its way into the model's context.
106
+ *
107
+ * Local host tools are still detected — a host tool reading a database returns
108
+ * data the host did not write — but only remote origins are fenced, because
109
+ * fencing a local tool's output would change prompts hosts have already tuned.
110
+ */
111
+ function guardToolResult(finding, data, provenance, policy, callableTools = []) {
112
+ const composed = composeToolText(finding, data);
113
+ const options = detectionForTrust(policy, 'untrusted');
114
+ const redacted = sanitizeText(composed, options);
115
+ const changed = redacted !== composed;
116
+ // Directive detection runs on remote content only: a local tool's output is
117
+ // bytes the host's own code produced.
118
+ const remote = isRemoteOrigin(provenance.origin);
119
+ const suspicious = remote ? directiveHits(composed, callableTools) : [];
120
+ const advisory = advisoryLevel(suspicious);
121
+ const fenced = remote
122
+ ? wrapToolData(redacted, provenance, advisory, policy.taint?.advisoryGuidance)
123
+ : redacted;
124
+ const hits = [
125
+ ...(changed ? [{ rule: 'tool_result.redacted', severity: 'medium' }] : []),
126
+ ...suspicious,
127
+ ];
128
+ if (hits.length === 0) {
129
+ return { text: fenced };
130
+ }
131
+ return {
132
+ text: fenced,
133
+ ...(suspicious.length > 0 ? { suspicious } : {}),
134
+ event: {
135
+ stage: 'tool_result',
136
+ trust: 'untrusted',
137
+ // Redaction changed the text; directive signals only annotate it.
138
+ action: changed ? 'redact' : 'flag',
139
+ hits,
140
+ provenance,
141
+ },
142
+ };
143
+ }
144
+ /**
145
+ * Guard a tool failure message.
146
+ *
147
+ * A remote server authors its own error strings, so an unguarded failure message
148
+ * is the cleanest injection path across this boundary: it reaches the model
149
+ * verbatim and is framed by the kernel as a system report.
150
+ */
151
+ function guardToolFailureText(message, provenance, policy) {
152
+ const options = detectionForTrust(policy, 'untrusted');
153
+ const redacted = sanitizeText(stripToolFences(message), options);
154
+ if (redacted === message) {
155
+ return { text: redacted };
156
+ }
157
+ return {
158
+ text: redacted,
159
+ event: {
160
+ stage: 'tool_result',
161
+ trust: 'untrusted',
162
+ action: 'redact',
163
+ hits: [{ rule: 'tool_failure.redacted', severity: 'medium' }],
164
+ provenance,
165
+ },
166
+ };
167
+ }
168
+ /**
169
+ * Inspect model-supplied tool arguments before the call runs.
170
+ *
171
+ * Arguments are model-authored, so the risk is not instruction smuggling but
172
+ * exfiltration: a credential lifted from context and posted outward as a
173
+ * parameter. Detection reports rather than rewrites — silently altering a tool
174
+ * argument would make the call succeed against something the model did not ask
175
+ * for.
176
+ */
177
+ function inspectToolArguments(args, policy) {
178
+ if (!policy.redactSensitive) {
179
+ return { action: 'allow' };
180
+ }
181
+ const rendered = textForScan(args);
182
+ if (rendered.unscannable) {
183
+ return { action: 'allow' };
184
+ }
185
+ const options = { sanitizeInput: false, redactSensitive: true };
186
+ if (sanitizeText(rendered.text, options) === rendered.text) {
187
+ return { action: 'allow' };
188
+ }
189
+ return {
190
+ action: 'flag',
191
+ hits: [{ rule: 'tool_call.sensitive-argument', severity: 'high' }],
192
+ };
193
+ }
194
+ /** Guardrail event for a flagged tool call, for the runner to emit. */
195
+ function toolCallEvent(verdict, provenance) {
196
+ if (verdict.action === 'allow') {
197
+ return undefined;
198
+ }
199
+ return {
200
+ stage: 'tool_call',
201
+ trust: 'untrusted',
202
+ action: verdict.action,
203
+ hits: verdict.hits,
204
+ provenance,
205
+ };
206
+ }
207
+ /**
208
+ * Record a tool result against the turn's taint.
209
+ *
210
+ * Only remote origins taint: a local host tool returns bytes the host's own code
211
+ * produced, and treating those as attacker-influenceable would make the gate
212
+ * useless in practice.
213
+ */
214
+ function recordTaint(taint, provenance, suspicious = []) {
215
+ const sources = taint?.sources ?? [];
216
+ const prior = taint?.suspicious ?? [];
217
+ if (!isRemoteOrigin(provenance.origin)) {
218
+ return { sources, suspicious: prior };
219
+ }
220
+ return { sources: [...sources, provenance], suspicious: [...prior, ...suspicious] };
221
+ }
222
+ /** True when the turn has already read attacker-influenceable content. */
223
+ function isTainted(taint) {
224
+ return (taint?.sources.length ?? 0) > 0;
225
+ }
226
+ /** True when remote content this turn read looked like it was steering the agent. */
227
+ function isSuspicious(taint) {
228
+ return (taint?.suspicious.length ?? 0) > 0;
229
+ }
230
+ /** Capability rank, so a single threshold can express "this and anything worse". */
231
+ const CAPABILITY_RANK = {
232
+ 'read-only': 0,
233
+ 'read-write': 1,
234
+ destructive: 2,
235
+ };
236
+ const GATE_RANK = {
237
+ off: Number.POSITIVE_INFINITY,
238
+ destructive: 2,
239
+ write: 1,
240
+ };
241
+ /**
242
+ * Decide whether a tool call may proceed given what the turn has already read.
243
+ *
244
+ * A `flag` says the call is happening on a tainted turn and is worth recording; a
245
+ * `block` says the profile asked for it to be refused. Reporting happens whether
246
+ * or not enforcement is configured, so the risk is visible before a host opts in.
247
+ */
248
+ function checkTaintGate(taint, access, policy) {
249
+ if (!isTainted(taint)) {
250
+ return { action: 'allow' };
251
+ }
252
+ const rank = CAPABILITY_RANK[access] ?? 0;
253
+ if (rank === 0) {
254
+ return { action: 'allow' };
255
+ }
256
+ const suspicious = isSuspicious(taint);
257
+ const hits = [
258
+ {
259
+ rule: suspicious ? 'tool_call.steered-turn' : 'tool_call.tainted-turn',
260
+ severity: 'high',
261
+ },
262
+ ];
263
+ // Gated on origin alone. Whether the content looked directive changes the rule
264
+ // reported, never whether the call is refused.
265
+ if (rank < GATE_RANK[policy.taint?.afterRemoteRead ?? 'off']) {
266
+ return { action: 'flag', hits };
267
+ }
268
+ const read = taint?.sources.map((s) => s.tool).join(', ') ?? '';
269
+ const reason = lexiconText(suspicious ? 'taint.reason_steered' : 'taint.reason_tainted');
270
+ return {
271
+ action: 'block',
272
+ hits,
273
+ rejection: lexiconText('taint.blocked', { access, sources: read, reason }),
274
+ };
275
+ }
276
+ export { checkTaintGate, composeToolText, guardToolFailureText, guardToolResult, inspectToolArguments, isRemoteOrigin, isSuspicious, isTainted, recordTaint, TOOL_CLOSE, toolCallEvent, wrapToolData, };