@dudousxd/nestjs-agent-core 0.15.5 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -0
- package/dist/guardrails/index.cjs +2519 -0
- package/dist/guardrails/index.cjs.map +1 -0
- package/dist/guardrails/index.d.cts +660 -0
- package/dist/guardrails/index.d.ts +660 -0
- package/dist/guardrails/index.js +2445 -0
- package/dist/guardrails/index.js.map +1 -0
- package/dist/index.cjs +27 -27
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +81 -1013
- package/dist/index.d.ts +81 -1013
- package/dist/index.js +25 -25
- package/dist/index.js.map +1 -1
- package/dist/tool-CL9oEytW.d.cts +1014 -0
- package/dist/tool-CL9oEytW.d.ts +1014 -0
- package/package.json +14 -3
|
@@ -0,0 +1,660 @@
|
|
|
1
|
+
import { A as Actor, I as InputProcessor, O as OutputProcessor, T as ToolHandler } from '../tool-CL9oEytW.cjs';
|
|
2
|
+
import '@standard-schema/spec';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Guardrails: detectors that look at text on its way to or from a model — prompts, answers, tool
|
|
6
|
+
* arguments, tool results, tool descriptions — and rules that decide what to do when they fire.
|
|
7
|
+
*
|
|
8
|
+
* Nothing in this module depends on the agent loop, on NestJS or on a runtime API beyond the
|
|
9
|
+
* language itself. A process that proxies raw provider traffic uses the detectors, `scan`, `Vault`
|
|
10
|
+
* and `StreamGuard` directly; `createGuardrails` is the adapter onto the loop's processor seams.
|
|
11
|
+
*/
|
|
12
|
+
/** Where in the traffic a rule applies. */
|
|
13
|
+
type GuardrailStage =
|
|
14
|
+
/** Prompt + attached context sent to a model (system, user, assistant history). */
|
|
15
|
+
'llm_request'
|
|
16
|
+
/** What the model answered (text and tool-call arguments), streaming included. */
|
|
17
|
+
| 'llm_response'
|
|
18
|
+
/** Arguments of a tool call, before the tool runs. */
|
|
19
|
+
| 'tool_args'
|
|
20
|
+
/** What a tool returned, before it flows back into the model (indirect prompt injection). */
|
|
21
|
+
| 'tool_result'
|
|
22
|
+
/** Tool descriptions and schemas of a remote tool server (tool poisoning). */
|
|
23
|
+
| 'tool_description';
|
|
24
|
+
declare const GUARDRAIL_STAGES: readonly GuardrailStage[];
|
|
25
|
+
/**
|
|
26
|
+
* - `allow`: exemption. When it matches (its detectors fire, or it has none), later rules are skipped.
|
|
27
|
+
* - `log`: record the hit only (a "flag").
|
|
28
|
+
* - `redact`: replace what was found by placeholders (reversible for values sent to a model).
|
|
29
|
+
* - `approve`: human-in-the-loop — refuse until someone approves this exact content. A caller that
|
|
30
|
+
* has no approval flow of its own treats it as `block`.
|
|
31
|
+
* - `block`: refuse with the rule's user-facing message.
|
|
32
|
+
*/
|
|
33
|
+
type GuardrailAction = 'allow' | 'log' | 'redact' | 'approve' | 'block';
|
|
34
|
+
/** Severity order used to combine several rules: the most severe wins. */
|
|
35
|
+
declare const ACTION_SEVERITY: Record<GuardrailAction, number>;
|
|
36
|
+
type PiiType = 'email' | 'phone' | 'cpf' | 'cnpj' | 'ssn' | 'credit_card' | 'iban' | 'ip_address';
|
|
37
|
+
declare const PII_TYPES: readonly PiiType[];
|
|
38
|
+
type SecretType = 'openai_key' | 'anthropic_key' | 'aws_access_key' | 'aws_secret_key' | 'github_token' | 'gitlab_token' | 'slack_token' | 'google_api_key' | 'stripe_key' | 'jwt' | 'bearer_token' | 'private_key' | 'generic_secret';
|
|
39
|
+
declare const SECRET_TYPES: readonly SecretType[];
|
|
40
|
+
/** Something a detector found. `value` is the matched text; keep it in-process. */
|
|
41
|
+
interface Finding {
|
|
42
|
+
/** The detector kind (or a custom detector's name). */
|
|
43
|
+
detector: string;
|
|
44
|
+
/** e.g. `pii.credit_card`, `secret.aws_access_key`, `injection.ignore_previous`. */
|
|
45
|
+
category: string;
|
|
46
|
+
start: number;
|
|
47
|
+
end: number;
|
|
48
|
+
score: number;
|
|
49
|
+
value: string;
|
|
50
|
+
/**
|
|
51
|
+
* Whether the finding can be redacted by replacing its span. `false` for whole-text verdicts
|
|
52
|
+
* (a moderation or classifier call), which replace the whole segment when redacting.
|
|
53
|
+
*/
|
|
54
|
+
spanned?: boolean;
|
|
55
|
+
/**
|
|
56
|
+
* What a redaction puts in place of this span. Undefined → a vault placeholder (`[EMAIL_1]`).
|
|
57
|
+
* Detectors whose finding is an instruction rather than a value (injection, tool poisoning) set
|
|
58
|
+
* this so the text is removed rather than tokenized.
|
|
59
|
+
*/
|
|
60
|
+
replacement?: string;
|
|
61
|
+
}
|
|
62
|
+
/** Who wrote a piece of text (model request stage). */
|
|
63
|
+
type SegmentSource = 'system' | 'user' | 'assistant' | 'tool';
|
|
64
|
+
/** A piece of text to scan, with where it came from. */
|
|
65
|
+
interface Segment {
|
|
66
|
+
text: string;
|
|
67
|
+
source?: SegmentSource;
|
|
68
|
+
/**
|
|
69
|
+
* Part of the newest turn. Content that is only in re-sent history is acted upon (redacted
|
|
70
|
+
* one-way instead of blocked) but its hits are reported as not fresh. Undefined → fresh.
|
|
71
|
+
*/
|
|
72
|
+
fresh?: boolean;
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* What the traffic is about. `stage` is the only field the engine reads itself; `tool` feeds the
|
|
76
|
+
* `match.tools` globs. Extend it with whatever your rules' `when` predicates need (tenant, actor,
|
|
77
|
+
* model, destination…).
|
|
78
|
+
*/
|
|
79
|
+
interface GuardContext {
|
|
80
|
+
stage: GuardrailStage;
|
|
81
|
+
/** Tool name, on tool stages. */
|
|
82
|
+
tool?: string;
|
|
83
|
+
}
|
|
84
|
+
/** A detector that runs your own code — NER, a classifier, a moderation endpoint, an LLM judge. */
|
|
85
|
+
interface CustomDetector<C extends GuardContext = GuardContext> {
|
|
86
|
+
kind: 'custom';
|
|
87
|
+
/** Identifies the detector in findings and hits (`detector`). Also its memoization key. */
|
|
88
|
+
name: string;
|
|
89
|
+
detect(text: string, ctx: C): Finding[] | Promise<Finding[]>;
|
|
90
|
+
}
|
|
91
|
+
/** A detector as configured on a rule. */
|
|
92
|
+
type DetectorSpec<C extends GuardContext = GuardContext> =
|
|
93
|
+
/** Regex + checksum validators (Luhn, CPF/CNPJ mod 11, IBAN mod 97…). */
|
|
94
|
+
{
|
|
95
|
+
kind: 'pii';
|
|
96
|
+
types?: readonly PiiType[];
|
|
97
|
+
}
|
|
98
|
+
/** API keys, tokens and private keys (patterns + entropy). */
|
|
99
|
+
| {
|
|
100
|
+
kind: 'secrets';
|
|
101
|
+
types?: readonly SecretType[];
|
|
102
|
+
}
|
|
103
|
+
/** Prompt-injection / jailbreak heuristics (EN, PT-BR, ES; hidden Unicode; encoded payloads). */
|
|
104
|
+
| {
|
|
105
|
+
kind: 'injection';
|
|
106
|
+
threshold?: number;
|
|
107
|
+
}
|
|
108
|
+
/** Suspicious instructions hidden in tool descriptions (tool poisoning). */
|
|
109
|
+
| {
|
|
110
|
+
kind: 'tool_poisoning';
|
|
111
|
+
threshold?: number;
|
|
112
|
+
}
|
|
113
|
+
/** Your own regular expressions (`(?i)` prefix = case-insensitive). */
|
|
114
|
+
| {
|
|
115
|
+
kind: 'regex';
|
|
116
|
+
patterns: readonly string[];
|
|
117
|
+
label?: string;
|
|
118
|
+
}
|
|
119
|
+
/** Words or phrases (case- and accent-insensitive, whole words). */
|
|
120
|
+
| {
|
|
121
|
+
kind: 'keywords';
|
|
122
|
+
words: readonly string[];
|
|
123
|
+
label?: string;
|
|
124
|
+
} | CustomDetector<C>;
|
|
125
|
+
type DetectorKind = DetectorSpec['kind'];
|
|
126
|
+
/** Scope a rule to part of the traffic. Every matcher that is set must match. */
|
|
127
|
+
interface GuardrailMatch {
|
|
128
|
+
/** Model request stage: which messages are scanned (default all). */
|
|
129
|
+
sources?: readonly SegmentSource[];
|
|
130
|
+
/** Tool stages: tool-name globs (`*` any run, `?` one character). Never matches model traffic. */
|
|
131
|
+
tools?: readonly string[];
|
|
132
|
+
}
|
|
133
|
+
interface GuardrailOptions {
|
|
134
|
+
/** Shown to the person when the rule blocks or asks for approval. */
|
|
135
|
+
message?: string;
|
|
136
|
+
/** What happens when a detector throws: `open` = ignore (default), `closed` = block. */
|
|
137
|
+
failMode?: 'open' | 'closed';
|
|
138
|
+
/**
|
|
139
|
+
* `redact` on `llm_request`: put the original values back into the answer the caller receives
|
|
140
|
+
* (the model only ever sees placeholders). Default true.
|
|
141
|
+
*/
|
|
142
|
+
restore?: boolean;
|
|
143
|
+
}
|
|
144
|
+
interface GuardrailRule<C extends GuardContext = GuardContext> {
|
|
145
|
+
id: string;
|
|
146
|
+
/** Shown in hits and decisions. Default: `id`. */
|
|
147
|
+
name?: string;
|
|
148
|
+
/** Ascending order of evaluation; ties by name. Default 100. */
|
|
149
|
+
priority?: number;
|
|
150
|
+
/** Default true. */
|
|
151
|
+
enabled?: boolean;
|
|
152
|
+
stages: readonly GuardrailStage[];
|
|
153
|
+
/** Empty → the rule fires on scope alone (an exemption, or a blanket block). */
|
|
154
|
+
detectors: readonly DetectorSpec<C>[];
|
|
155
|
+
action: GuardrailAction;
|
|
156
|
+
match?: GuardrailMatch;
|
|
157
|
+
/**
|
|
158
|
+
* Anything scope-like the engine does not know about — the tenant, the actor's roles, the model
|
|
159
|
+
* or its destination. Returning false leaves the rule out of this scan.
|
|
160
|
+
*/
|
|
161
|
+
when?: (ctx: C) => boolean;
|
|
162
|
+
options?: GuardrailOptions;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/** Dotted-quad IPv4 with every octet in 0-255 and no leading zeros. */
|
|
166
|
+
declare function isIPv4(raw: string): boolean;
|
|
167
|
+
/** Luhn (mod 10) checksum over a digit string. */
|
|
168
|
+
declare function luhnValid(digits: string): boolean;
|
|
169
|
+
/** Card brands by IIN prefix (digits only). Unknown prefixes are not treated as cards. */
|
|
170
|
+
declare function cardBrand(digits: string): string | undefined;
|
|
171
|
+
/** CPF (Brazilian individual taxpayer id): 11 digits, two mod-11 check digits. */
|
|
172
|
+
declare function cpfValid(raw: string): boolean;
|
|
173
|
+
/**
|
|
174
|
+
* CNPJ (Brazilian company id), numeric or the alphanumeric format in use since July 2026: 12
|
|
175
|
+
* characters [0-9A-Z] + 2 check digits, each character valued at its ASCII code minus 48.
|
|
176
|
+
*/
|
|
177
|
+
declare function cnpjValid(raw: string): boolean;
|
|
178
|
+
/** US Social Security Number structure (no 000/666/9xx area, no 00 group, no 0000 serial). */
|
|
179
|
+
declare function ssnValid(raw: string): boolean;
|
|
180
|
+
/** IBAN: known country length and ISO 7064 mod 97-10 checksum. */
|
|
181
|
+
declare function ibanValid(raw: string): boolean;
|
|
182
|
+
/** Phone numbers: international (+CC), Brazilian and North American formats with separators. */
|
|
183
|
+
declare function phoneValid(raw: string): boolean;
|
|
184
|
+
/** PII with format + checksum validation. Overlaps are resolved by the caller (longest, earliest). */
|
|
185
|
+
declare function detectPii(text: string, types?: readonly PiiType[]): Finding[];
|
|
186
|
+
|
|
187
|
+
/** Shannon entropy in bits per character: tells random tokens from words ("password=changeme"). */
|
|
188
|
+
declare function entropy(s: string): number;
|
|
189
|
+
declare function detectSecrets(text: string, types?: readonly SecretType[]): Finding[];
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Prompt-injection / jailbreak heuristics. Each signal has a weight; the text's score is the
|
|
193
|
+
* noisy-OR of the distinct signals that fired (1 - Π(1 - w)), so one strong signal or several weak
|
|
194
|
+
* ones cross the threshold. English, Portuguese (PT-BR) and Spanish phrasings, role/delimiter
|
|
195
|
+
* spoofing, hidden Unicode (tag characters, bidi overrides, zero-width runs) and exfiltration
|
|
196
|
+
* through rendered links/images.
|
|
197
|
+
*/
|
|
198
|
+
interface Signal {
|
|
199
|
+
id: string;
|
|
200
|
+
re: RegExp;
|
|
201
|
+
weight: number;
|
|
202
|
+
}
|
|
203
|
+
declare const INJECTION_SIGNALS: Signal[];
|
|
204
|
+
/** Decodes Unicode tag characters ("ASCII smuggling") back to the ASCII they encode. */
|
|
205
|
+
declare function decodeTagChars(s: string): string;
|
|
206
|
+
interface InjectionResult {
|
|
207
|
+
score: number;
|
|
208
|
+
findings: Finding[];
|
|
209
|
+
}
|
|
210
|
+
/** Scores a text; findings are returned only when the score reaches `threshold`. */
|
|
211
|
+
declare function scoreInjection(text: string, threshold?: number): InjectionResult;
|
|
212
|
+
|
|
213
|
+
interface PoisoningResult {
|
|
214
|
+
score: number;
|
|
215
|
+
findings: Finding[];
|
|
216
|
+
}
|
|
217
|
+
declare function scoreToolText(text: string, threshold?: number): PoisoningResult;
|
|
218
|
+
/** The parts of a tool definition a model reads — MCP's `Tool` shape satisfies it. */
|
|
219
|
+
interface ToolDefinitionText {
|
|
220
|
+
name: string;
|
|
221
|
+
title?: string | null | undefined;
|
|
222
|
+
description?: string | null | undefined;
|
|
223
|
+
inputSchema?: unknown;
|
|
224
|
+
}
|
|
225
|
+
/** Every description the model reads for a tool: its own and its parameters' (nested). */
|
|
226
|
+
declare function toolText(tool: ToolDefinitionText): string;
|
|
227
|
+
|
|
228
|
+
/** Compiled once per pattern (global flag added), with a bounded cache. */
|
|
229
|
+
declare function compileGuardPattern(pattern: string): RegExp;
|
|
230
|
+
declare function detectRegex(text: string, patterns: readonly string[], label?: string): Finding[];
|
|
231
|
+
/** Lower case without accents, keeping string length stable (one output char per input char). */
|
|
232
|
+
declare function fold(s: string): string;
|
|
233
|
+
/** Whole-word, case- and accent-insensitive keyword/phrase matching. */
|
|
234
|
+
declare function detectKeywords(text: string, words: readonly string[], label?: string): Finding[];
|
|
235
|
+
|
|
236
|
+
/** Minimal glob: `*` matches any run of characters, `?` a single one. Case-sensitive. */
|
|
237
|
+
declare function globToRegExp(pattern: string): RegExp;
|
|
238
|
+
declare function matchesAny(value: string, patterns: readonly string[] | undefined): boolean;
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Reversible tokenization for one request: each distinct sensitive value gets a stable placeholder
|
|
242
|
+
* (`[EMAIL_1]`, `[CREDIT_CARD_2]`…) in order of first appearance, so a conversation re-sent turn
|
|
243
|
+
* after turn maps to the same placeholders. The model only sees placeholders; `restore` puts the
|
|
244
|
+
* values back into what the caller receives.
|
|
245
|
+
*
|
|
246
|
+
* Values stay in this object. {@link toJSON} / {@link fromJSON} exist for a caller that must carry a
|
|
247
|
+
* vault across processes (a durable run resumed elsewhere); wherever that snapshot is stored holds
|
|
248
|
+
* the raw values, so treat it like the data it came from.
|
|
249
|
+
*/
|
|
250
|
+
declare class Vault {
|
|
251
|
+
private readonly byValue;
|
|
252
|
+
private readonly byToken;
|
|
253
|
+
private readonly counters;
|
|
254
|
+
/**
|
|
255
|
+
* Placeholder for `value` under `label` (e.g. `CREDIT_CARD`). Reversible placeholders look like
|
|
256
|
+
* `[CREDIT_CARD_1]`; one-way ones (tool results, answers, rules with `restore: false`) like
|
|
257
|
+
* `[REDACTED_CREDIT_CARD_1]`, so they can never be mistaken for a reversible one.
|
|
258
|
+
*/
|
|
259
|
+
tokenFor(label: string, value: string, restorable: boolean): string;
|
|
260
|
+
get size(): number;
|
|
261
|
+
get restorableCount(): number;
|
|
262
|
+
/** Longest placeholder, for stream hold-back windows. */
|
|
263
|
+
get maxTokenLength(): number;
|
|
264
|
+
/** Puts restorable values back. `jsonString`: the text is inside a JSON string literal. */
|
|
265
|
+
restore(text: string, jsonString?: boolean): string;
|
|
266
|
+
/** Whether `text` ends with what could be the beginning of a placeholder (streaming). */
|
|
267
|
+
pendingPrefix(text: string): number;
|
|
268
|
+
/** Every entry, in the order the placeholders were minted. */
|
|
269
|
+
toJSON(): VaultSnapshot;
|
|
270
|
+
/** Rebuilds a vault from {@link toJSON}; placeholders keep their numbers and counters resume. */
|
|
271
|
+
static fromJSON(snapshot: VaultSnapshot): Vault;
|
|
272
|
+
}
|
|
273
|
+
/** A serializable {@link Vault}. */
|
|
274
|
+
interface VaultSnapshot {
|
|
275
|
+
entries: Array<{
|
|
276
|
+
token: string;
|
|
277
|
+
value: string;
|
|
278
|
+
restorable: boolean;
|
|
279
|
+
}>;
|
|
280
|
+
}
|
|
281
|
+
/** `pii.credit_card` -> `CREDIT_CARD`, `secret.openai_key` -> `OPENAI_KEY`, `ner.person` -> `PERSON`. */
|
|
282
|
+
declare function labelFor(category: string): string;
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* One reported hit. `values` holds the distinct matched values so a caller can fingerprint them
|
|
286
|
+
* (a keyed hash correlates repeats without storing the value) — never persist them as they are.
|
|
287
|
+
*/
|
|
288
|
+
interface GuardHit {
|
|
289
|
+
ruleId: string;
|
|
290
|
+
ruleName: string;
|
|
291
|
+
action: GuardrailAction;
|
|
292
|
+
detector: string;
|
|
293
|
+
category: string;
|
|
294
|
+
score: number;
|
|
295
|
+
count: number;
|
|
296
|
+
/** Distinct values found (at most 20). In-memory only. */
|
|
297
|
+
values: string[];
|
|
298
|
+
/** Found in the newest turn (false: only in re-sent history, already reported once). */
|
|
299
|
+
fresh: boolean;
|
|
300
|
+
/** `detector_error` hits: what failed, and what the rule's fail mode made of it. */
|
|
301
|
+
error?: string;
|
|
302
|
+
failMode?: 'open' | 'closed';
|
|
303
|
+
}
|
|
304
|
+
interface ScanResult {
|
|
305
|
+
stage: GuardrailStage;
|
|
306
|
+
/** Most severe action among the rules that fired. */
|
|
307
|
+
action: GuardrailAction;
|
|
308
|
+
/** The rule that decided a block / approval, and the message for the person. */
|
|
309
|
+
decisive?: {
|
|
310
|
+
id: string;
|
|
311
|
+
name: string;
|
|
312
|
+
message: string;
|
|
313
|
+
};
|
|
314
|
+
/** Segment texts after redaction (same order as the input). */
|
|
315
|
+
segments: string[];
|
|
316
|
+
/** Whether any segment changed. */
|
|
317
|
+
changed: boolean;
|
|
318
|
+
hits: GuardHit[];
|
|
319
|
+
vault: Vault;
|
|
320
|
+
/** Redacted spans. */
|
|
321
|
+
redactions: number;
|
|
322
|
+
latencyMs: number;
|
|
323
|
+
}
|
|
324
|
+
interface ScanOptions {
|
|
325
|
+
/** Per-stage refusal messages, for rules that set none. */
|
|
326
|
+
messages?: Partial<Record<GuardrailStage, string>>;
|
|
327
|
+
/** Refusal message for `approve`, for rules that set none. */
|
|
328
|
+
approvalMessage?: string;
|
|
329
|
+
/**
|
|
330
|
+
* Stages whose redactions are reversible (unless the rule sets `restore: false`). Default
|
|
331
|
+
* `['llm_request']`: what the caller sent is theirs to get back; what a tool or the model produced
|
|
332
|
+
* is removed one-way. A caller whose tool results are its own data may add `tool_result`.
|
|
333
|
+
*/
|
|
334
|
+
restorableStages?: readonly GuardrailStage[];
|
|
335
|
+
}
|
|
336
|
+
declare const DEFAULT_MESSAGES: Record<GuardrailStage, string>;
|
|
337
|
+
declare const DEFAULT_APPROVAL_MESSAGE = "This needs approval under your organization's AI guardrails before it can go ahead.";
|
|
338
|
+
/** What replaces an instruction-like finding: it is removed, not tokenized. */
|
|
339
|
+
declare const INJECTION_REPLACEMENT = "[removed: suspected prompt injection]";
|
|
340
|
+
/** Whether a rule's scope covers the context (stage, tool, `when`). Segments are filtered apart. */
|
|
341
|
+
declare function ruleApplies<C extends GuardContext>(rule: GuardrailRule<C>, ctx: C): boolean;
|
|
342
|
+
declare function orderRules<C extends GuardContext>(rules: readonly GuardrailRule<C>[]): GuardrailRule<C>[];
|
|
343
|
+
/** Runs one detector on one text. A custom detector's throw is the rule's fail mode to decide. */
|
|
344
|
+
declare function runDetector<C extends GuardContext>(spec: DetectorSpec<C>, text: string, ctx: C): Promise<Finding[]>;
|
|
345
|
+
/**
|
|
346
|
+
* The in-process detectors of a rule set, as one span locator — what a stream window must never
|
|
347
|
+
* cut through (see `StreamGuard`'s `locate`). Custom detectors are left out: they may be remote.
|
|
348
|
+
*/
|
|
349
|
+
declare function spanLocator<C extends GuardContext>(rules: readonly GuardrailRule<C>[]): ((text: string) => Finding[]) | undefined;
|
|
350
|
+
/**
|
|
351
|
+
* Non-overlapping spans. Overlapping findings are merged into their union (nothing that any
|
|
352
|
+
* detector found is left uncovered), labelled by the longer finding (ties: the higher score) —
|
|
353
|
+
* an injected HTML comment that contains an email address is removed as a whole. With `text`, a
|
|
354
|
+
* merged span's value is re-read from it.
|
|
355
|
+
*/
|
|
356
|
+
declare function resolveOverlaps(findings: readonly Finding[], text?: string): Finding[];
|
|
357
|
+
declare function applyRedactions(text: string, findings: readonly Finding[], vault: Vault, restorable: boolean): {
|
|
358
|
+
text: string;
|
|
359
|
+
count: number;
|
|
360
|
+
};
|
|
361
|
+
/**
|
|
362
|
+
* Evaluates the rules for one stage over a set of segments. Rules run by ascending priority; an
|
|
363
|
+
* `allow` rule that fires stops evaluation; otherwise every firing rule contributes and the most
|
|
364
|
+
* severe action wins (block > approve > redact > log). Detector results are shared between rules.
|
|
365
|
+
*
|
|
366
|
+
* Blocking applies to the NEWEST turn: content found only in re-sent history (`fresh: false`) was
|
|
367
|
+
* already refused once, so it is removed one-way instead — the conversation can go on and the value
|
|
368
|
+
* still never reaches the model.
|
|
369
|
+
*/
|
|
370
|
+
declare function scan<C extends GuardContext>(rules: readonly GuardrailRule<C>[], ctx: C, segments: readonly Segment[], vault?: Vault, options?: ScanOptions): Promise<ScanResult>;
|
|
371
|
+
|
|
372
|
+
/**
|
|
373
|
+
* Text slots of provider and MCP payloads, for a process that proxies them: each slot is a piece of
|
|
374
|
+
* text to scan plus a setter that writes the (redacted / restored) text back in place.
|
|
375
|
+
*/
|
|
376
|
+
interface Slot {
|
|
377
|
+
segment: Segment;
|
|
378
|
+
set(text: string): void;
|
|
379
|
+
/** The text lives inside a JSON document (tool-call arguments): restored values are escaped. */
|
|
380
|
+
json?: boolean;
|
|
381
|
+
}
|
|
382
|
+
/** String leaves of a JSON value (tool arguments, structured results). */
|
|
383
|
+
declare function jsonSlots(root: unknown, source: SegmentSource, fresh: boolean, replaceRoot?: (v: string) => void): Slot[];
|
|
384
|
+
/** Slots of an LLM request body: OpenAI chat completions, Anthropic Messages or embeddings. */
|
|
385
|
+
declare function requestSlots(api: 'openai' | 'anthropic', body: Record<string, unknown>): Slot[];
|
|
386
|
+
/** Slots of a non-streaming answer in the caller's format (OpenAI chat completion or Anthropic message). */
|
|
387
|
+
declare function responseSlots(api: 'openai' | 'anthropic', body: unknown): Slot[];
|
|
388
|
+
/** Replaces the answer of a non-streaming response with a refusal (finish reason content_filter). */
|
|
389
|
+
declare function refuseResponse(api: 'openai' | 'anthropic', body: unknown, message: string): unknown;
|
|
390
|
+
/** Slots of an MCP tool result: text content, embedded text resources and structured content. */
|
|
391
|
+
declare function toolResultSlots(result: Record<string, unknown>): Slot[];
|
|
392
|
+
|
|
393
|
+
/**
|
|
394
|
+
* Guards a streamed answer in the caller's format (OpenAI chat-completion chunks or Anthropic
|
|
395
|
+
* Messages events). Text deltas are buffered per channel (the answer text, each tool call's
|
|
396
|
+
* arguments, each content block) and released in windows: the newest `holdback` characters stay
|
|
397
|
+
* buffered so a value split across deltas ("4111 1111" … "1111 1111") is seen whole, and a window
|
|
398
|
+
* never ends inside something a detector found or inside a placeholder being restored. Each
|
|
399
|
+
* released window is scanned (redaction, moderation…) and has placeholders restored. When a
|
|
400
|
+
* window is blocked the stream ends with the rule's message and a `content_filter` / `refusal`
|
|
401
|
+
* finish; the rest of the upstream is consumed (for usage) but not forwarded.
|
|
402
|
+
*/
|
|
403
|
+
interface WindowVerdict {
|
|
404
|
+
/** Text to release (redacted). */
|
|
405
|
+
text: string;
|
|
406
|
+
block?: {
|
|
407
|
+
message: string;
|
|
408
|
+
};
|
|
409
|
+
}
|
|
410
|
+
interface StreamGuardOptions {
|
|
411
|
+
api: 'openai' | 'anthropic';
|
|
412
|
+
vault: Vault;
|
|
413
|
+
/** Restore placeholders from the request's redactions into what the caller receives. */
|
|
414
|
+
restore: boolean;
|
|
415
|
+
/** Response-stage scan of one window; undefined when no response rule applies. */
|
|
416
|
+
scan?: (text: string, channel: string) => Promise<WindowVerdict>;
|
|
417
|
+
/** Spans (local detectors) a window must not cut through. */
|
|
418
|
+
locate?: (text: string) => Finding[];
|
|
419
|
+
/** Characters released per window (default 256). */
|
|
420
|
+
window?: number;
|
|
421
|
+
/** Characters kept back at the end of the buffer (default 64). */
|
|
422
|
+
holdback?: number;
|
|
423
|
+
}
|
|
424
|
+
declare class StreamGuard {
|
|
425
|
+
private readonly opts;
|
|
426
|
+
private buffer;
|
|
427
|
+
private readonly decoder;
|
|
428
|
+
private readonly channels;
|
|
429
|
+
private readonly window;
|
|
430
|
+
private readonly holdback;
|
|
431
|
+
blocked?: {
|
|
432
|
+
message: string;
|
|
433
|
+
};
|
|
434
|
+
private template;
|
|
435
|
+
private readonly openBlocks;
|
|
436
|
+
private maxBlock;
|
|
437
|
+
private done;
|
|
438
|
+
constructor(opts: StreamGuardOptions);
|
|
439
|
+
/** Whether the guard changes anything (else the caller can pass bytes straight through). */
|
|
440
|
+
get active(): boolean;
|
|
441
|
+
push(chunk: string | Uint8Array): Promise<string>;
|
|
442
|
+
end(): Promise<string>;
|
|
443
|
+
private frame;
|
|
444
|
+
private openaiFrame;
|
|
445
|
+
private openaiChunk;
|
|
446
|
+
private openaiBlock;
|
|
447
|
+
private flushChoice;
|
|
448
|
+
private anthropicFrame;
|
|
449
|
+
private flushBlock;
|
|
450
|
+
private anthropicBlock;
|
|
451
|
+
private flushAll;
|
|
452
|
+
private feed;
|
|
453
|
+
/** Releases a window (everything when `all`), scanned and restored. */
|
|
454
|
+
private release;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/**
|
|
458
|
+
* Guardrails on the agent loop's own seams: an {@link InputProcessor} that scans (and redacts) every
|
|
459
|
+
* prompt — including the tool results riding in it — an {@link OutputProcessor} that scans every
|
|
460
|
+
* answer and the tool calls it asks for, a tool-handler wrapper, and a tool-definition screen.
|
|
461
|
+
*
|
|
462
|
+
* Built only on the processor SPI, so it is a consumer of the loop rather than a part of it: the
|
|
463
|
+
* same rules and detectors run standalone through `scan` for traffic that never meets the loop.
|
|
464
|
+
*/
|
|
465
|
+
|
|
466
|
+
/** What a rule's `when` and a custom detector see when the guardrails run inside the loop. */
|
|
467
|
+
interface AgentGuardContext extends GuardContext {
|
|
468
|
+
/** Undefined only for a tool definition screened outside any turn. */
|
|
469
|
+
threadId?: string;
|
|
470
|
+
actor?: Actor;
|
|
471
|
+
agentName?: string;
|
|
472
|
+
/** Model step within the run; undefined outside a processor. */
|
|
473
|
+
step?: number;
|
|
474
|
+
/** The tool server a screened definition came from (e.g. the MCP server's name). */
|
|
475
|
+
server?: string;
|
|
476
|
+
}
|
|
477
|
+
type Stages = readonly GuardrailStage[];
|
|
478
|
+
interface PiiShorthand {
|
|
479
|
+
action: GuardrailAction;
|
|
480
|
+
types?: readonly PiiType[];
|
|
481
|
+
/**
|
|
482
|
+
* Put the original values back into the answer and into tool arguments (the model only sees
|
|
483
|
+
* placeholders). Covers values from the prompt and from tool results. Default true.
|
|
484
|
+
*/
|
|
485
|
+
restore?: boolean;
|
|
486
|
+
stages?: Stages;
|
|
487
|
+
}
|
|
488
|
+
interface SecretsShorthand {
|
|
489
|
+
action: GuardrailAction;
|
|
490
|
+
types?: readonly SecretType[];
|
|
491
|
+
stages?: Stages;
|
|
492
|
+
}
|
|
493
|
+
interface InjectionShorthand {
|
|
494
|
+
/** Default `block`. */
|
|
495
|
+
action?: GuardrailAction;
|
|
496
|
+
/** Noisy-OR score (0-1) at which the heuristics fire. Default 0.5. */
|
|
497
|
+
threshold?: number;
|
|
498
|
+
stages?: Stages;
|
|
499
|
+
}
|
|
500
|
+
interface ToolPoisoningShorthand {
|
|
501
|
+
/** `block` (default) refuses the tool in {@link Guardrails.screenTool}; `log` only reports it. */
|
|
502
|
+
action?: 'block' | 'log';
|
|
503
|
+
threshold?: number;
|
|
504
|
+
}
|
|
505
|
+
/** One decision worth auditing. Never carries the matched values themselves. */
|
|
506
|
+
interface GuardrailEvent {
|
|
507
|
+
stage: GuardrailStage;
|
|
508
|
+
/** The effective action (`approve` is reported as it was configured, and enforced as a block). */
|
|
509
|
+
action: GuardrailAction;
|
|
510
|
+
decisive?: {
|
|
511
|
+
id: string;
|
|
512
|
+
name: string;
|
|
513
|
+
message: string;
|
|
514
|
+
};
|
|
515
|
+
/** Hits in the newest content only — history re-sent every step is not reported again. */
|
|
516
|
+
hits: GuardrailEventHit[];
|
|
517
|
+
redactions: number;
|
|
518
|
+
latencyMs: number;
|
|
519
|
+
context: AgentGuardContext;
|
|
520
|
+
}
|
|
521
|
+
type GuardrailEventHit = Omit<GuardHit, 'values'> & {
|
|
522
|
+
/** {@link GuardrailsOptions.fingerprint} of each distinct value; empty without one. */
|
|
523
|
+
fingerprints: string[];
|
|
524
|
+
};
|
|
525
|
+
/** Where a thread's reversible placeholders live between the prompt and the answer. */
|
|
526
|
+
interface VaultStore {
|
|
527
|
+
load(threadId: string): Vault | undefined | Promise<Vault | undefined>;
|
|
528
|
+
save(threadId: string, vault: Vault): void | Promise<void>;
|
|
529
|
+
}
|
|
530
|
+
/**
|
|
531
|
+
* The default {@link VaultStore}: this process's memory, bounded to the most recent threads. A run
|
|
532
|
+
* resumed in ANOTHER process finds no vault, so its placeholders reach the reader unrestored — the
|
|
533
|
+
* values still never reach the model. Supply a shared store (with {@link Vault.toJSON}) to restore
|
|
534
|
+
* across processes; it then holds the raw values.
|
|
535
|
+
*/
|
|
536
|
+
declare class InMemoryVaultStore implements VaultStore {
|
|
537
|
+
private readonly maxThreads;
|
|
538
|
+
private readonly vaults;
|
|
539
|
+
constructor(maxThreads?: number);
|
|
540
|
+
load(threadId: string): Vault | undefined;
|
|
541
|
+
save(threadId: string, vault: Vault): void;
|
|
542
|
+
}
|
|
543
|
+
type GuardrailRulesSource = readonly GuardrailRule<AgentGuardContext>[] | ((ctx: AgentGuardContext) => readonly GuardrailRule<AgentGuardContext>[] | Promise<readonly GuardrailRule<AgentGuardContext>[]>);
|
|
544
|
+
interface GuardrailsOptions {
|
|
545
|
+
/** PII (email, phone, CPF/CNPJ, SSN, cards with Luhn, IBAN, IPv4). */
|
|
546
|
+
pii?: GuardrailAction | PiiShorthand;
|
|
547
|
+
/** API keys, tokens and private keys. */
|
|
548
|
+
secrets?: GuardrailAction | SecretsShorthand;
|
|
549
|
+
/** Prompt-injection / jailbreak heuristics. */
|
|
550
|
+
injection?: GuardrailAction | InjectionShorthand;
|
|
551
|
+
/** Screen tool definitions for hidden instructions — see {@link Guardrails.screenTool}. */
|
|
552
|
+
toolPoisoning?: boolean | ToolPoisoningShorthand;
|
|
553
|
+
/**
|
|
554
|
+
* More rules, evaluated alongside the shorthands. A function is called per scan with the turn's
|
|
555
|
+
* context, which is the seam for per-tenant policy: look the tenant's rules up by `ctx.actor`.
|
|
556
|
+
*/
|
|
557
|
+
rules?: GuardrailRulesSource;
|
|
558
|
+
/** Audit hook. Awaited; a throw is swallowed so an audit sink can never change a decision. */
|
|
559
|
+
onEvent?: (event: GuardrailEvent) => void | Promise<void>;
|
|
560
|
+
/** Keyed hash (or any stable digest) of a matched value, reported on events instead of it. */
|
|
561
|
+
fingerprint?: (value: string) => string;
|
|
562
|
+
/** Per-stage refusal messages for rules that set none. */
|
|
563
|
+
messages?: ScanOptions['messages'];
|
|
564
|
+
/** Where reversible placeholders live between prompt and answer. Default {@link InMemoryVaultStore}. */
|
|
565
|
+
vaults?: VaultStore;
|
|
566
|
+
/**
|
|
567
|
+
* Keep the turn streaming: the output processor declares `incremental` with this lookback, so
|
|
568
|
+
* the loop releases a prefix while holding the last `lookbackChars` characters. `false` gates the
|
|
569
|
+
* whole answer instead. Default `{ lookbackChars: 256 }` — wide enough for every built-in
|
|
570
|
+
* pattern and placeholder; widen it for custom patterns that match longer text.
|
|
571
|
+
*/
|
|
572
|
+
incremental?: false | {
|
|
573
|
+
lookbackChars?: number;
|
|
574
|
+
};
|
|
575
|
+
/** Processor name (user-visible on a refusal). Default `guardrails`. */
|
|
576
|
+
name?: string;
|
|
577
|
+
}
|
|
578
|
+
/** A guardrail refused content the loop cannot rewrite around: a prompt, or a tool call. */
|
|
579
|
+
declare class GuardrailBlockedError extends Error {
|
|
580
|
+
readonly stage: GuardrailStage;
|
|
581
|
+
/** The message for the person — the rule's own, or the stage default. */
|
|
582
|
+
readonly userMessage: string;
|
|
583
|
+
readonly rule?: {
|
|
584
|
+
id: string;
|
|
585
|
+
name: string;
|
|
586
|
+
} | undefined;
|
|
587
|
+
constructor(stage: GuardrailStage,
|
|
588
|
+
/** The message for the person — the rule's own, or the stage default. */
|
|
589
|
+
userMessage: string, rule?: {
|
|
590
|
+
id: string;
|
|
591
|
+
name: string;
|
|
592
|
+
} | undefined);
|
|
593
|
+
}
|
|
594
|
+
/** The rules the shorthand options stand for. Exported so a caller can inspect or extend them. */
|
|
595
|
+
declare function shorthandRules(options: GuardrailsOptions): GuardrailRule<AgentGuardContext>[];
|
|
596
|
+
/**
|
|
597
|
+
* Guardrails for the agent loop. Register the two processors, optionally wrap tool handlers, and
|
|
598
|
+
* hand {@link screenTool} to whatever imports remote tool definitions:
|
|
599
|
+
*
|
|
600
|
+
* ```ts
|
|
601
|
+
* const guardrails = createGuardrails({ pii: 'redact', secrets: 'block', injection: { threshold: 0.6 }, toolPoisoning: true });
|
|
602
|
+
* AgentModule.forRoot({ inputProcessors: [guardrails.input], outputProcessors: [guardrails.output], … });
|
|
603
|
+
* ```
|
|
604
|
+
*/
|
|
605
|
+
declare class Guardrails {
|
|
606
|
+
private readonly options;
|
|
607
|
+
/** Scans every prompt; redacts in place, withholds tool results, throws on a blocked prompt. */
|
|
608
|
+
readonly input: InputProcessor;
|
|
609
|
+
/** Scans every answer and the tool calls it asks for; redacts, restores placeholders, or rejects. */
|
|
610
|
+
readonly output: OutputProcessor;
|
|
611
|
+
private readonly shorthand;
|
|
612
|
+
private readonly vaults;
|
|
613
|
+
/** Per thread: hits already reported this turn, so re-scans of the same content stay quiet. */
|
|
614
|
+
private readonly reported;
|
|
615
|
+
constructor(options?: GuardrailsOptions);
|
|
616
|
+
/** Every rule in force for `ctx`: the shorthands, then {@link GuardrailsOptions.rules}. */
|
|
617
|
+
rulesFor(ctx: AgentGuardContext): Promise<GuardrailRule<AgentGuardContext>[]>;
|
|
618
|
+
/**
|
|
619
|
+
* Scans arbitrary segments at one stage with the configured rules, reporting through `onEvent`.
|
|
620
|
+
* What the processors use; also the entry point for traffic that never meets the loop.
|
|
621
|
+
*/
|
|
622
|
+
scan(ctx: AgentGuardContext, segments: readonly Segment[], vault?: Vault): Promise<ScanResult>;
|
|
623
|
+
/**
|
|
624
|
+
* Screens one tool definition (name, title, description, and every description in its input
|
|
625
|
+
* schema) for hidden instructions. `allowed: false` means a `block`/`approve` rule fired — leave
|
|
626
|
+
* the tool out of the catalog. Shaped to be handed straight to an MCP importer's screen hook.
|
|
627
|
+
*/
|
|
628
|
+
screenTool(tool: ToolDefinitionText & {
|
|
629
|
+
server?: string;
|
|
630
|
+
}): Promise<{
|
|
631
|
+
allowed: true;
|
|
632
|
+
} | {
|
|
633
|
+
allowed: false;
|
|
634
|
+
reason: string;
|
|
635
|
+
}>;
|
|
636
|
+
/**
|
|
637
|
+
* Wraps a tool handler so its arguments are guarded where the loop cannot rewrite them: the
|
|
638
|
+
* thread's placeholders are restored (the model only ever saw `[EMAIL_1]`, the tool needs the
|
|
639
|
+
* address), then the `tool_args` rules run — `redact` rewrites what the tool receives, `block`
|
|
640
|
+
* throws a {@link GuardrailBlockedError}, which the model reads as the tool's failure.
|
|
641
|
+
*/
|
|
642
|
+
wrapTool<I>(toolName: string, handler: ToolHandler<I>): ToolHandler<I>;
|
|
643
|
+
private contextOf;
|
|
644
|
+
private processInput;
|
|
645
|
+
private guardToolResult;
|
|
646
|
+
private processOutput;
|
|
647
|
+
private report;
|
|
648
|
+
}
|
|
649
|
+
/** Shorthand for `new Guardrails(options)`. */
|
|
650
|
+
declare function createGuardrails(options?: GuardrailsOptions): Guardrails;
|
|
651
|
+
/**
|
|
652
|
+
* The tool-poisoning screen alone — no rules, no events. `score` is the noisy-OR of the signals
|
|
653
|
+
* that fired; `findings` is empty below `threshold` (default 0.5).
|
|
654
|
+
*/
|
|
655
|
+
declare function screenToolDefinition(tool: ToolDefinitionText, threshold?: number): {
|
|
656
|
+
score: number;
|
|
657
|
+
findings: Finding[];
|
|
658
|
+
};
|
|
659
|
+
|
|
660
|
+
export { ACTION_SEVERITY, type AgentGuardContext, type CustomDetector, DEFAULT_APPROVAL_MESSAGE, DEFAULT_MESSAGES, type DetectorKind, type DetectorSpec, type Finding, GUARDRAIL_STAGES, type GuardContext, type GuardHit, type GuardrailAction, GuardrailBlockedError, type GuardrailEvent, type GuardrailEventHit, type GuardrailMatch, type GuardrailOptions, type GuardrailRule, type GuardrailRulesSource, type GuardrailStage, Guardrails, type GuardrailsOptions, INJECTION_REPLACEMENT, INJECTION_SIGNALS, InMemoryVaultStore, type InjectionResult, type InjectionShorthand, PII_TYPES, type PiiShorthand, type PiiType, type PoisoningResult, SECRET_TYPES, type ScanOptions, type ScanResult, type SecretType, type SecretsShorthand, type Segment, type SegmentSource, type Signal, type Slot, StreamGuard, type StreamGuardOptions, type ToolDefinitionText, type ToolPoisoningShorthand, Vault, type VaultSnapshot, type VaultStore, type WindowVerdict, applyRedactions, cardBrand, cnpjValid, compileGuardPattern, cpfValid, createGuardrails, decodeTagChars, detectKeywords, detectPii, detectRegex, detectSecrets, entropy, fold, globToRegExp, ibanValid, isIPv4, jsonSlots, labelFor, luhnValid, matchesAny, orderRules, phoneValid, refuseResponse, requestSlots, resolveOverlaps, responseSlots, ruleApplies, runDetector, scan, scoreInjection, scoreToolText, screenToolDefinition, shorthandRules, spanLocator, ssnValid, toolResultSlots, toolText };
|