@kindgi/guardrails 0.0.0-bootstrap.0 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +78 -1
- package/dist/action-handler.d.ts +82 -0
- package/dist/action-handler.d.ts.map +1 -0
- package/dist/action-handler.js +120 -0
- package/dist/action-handler.js.map +1 -0
- package/dist/checks.d.ts +9 -0
- package/dist/checks.d.ts.map +1 -0
- package/dist/checks.js +235 -0
- package/dist/checks.js.map +1 -0
- package/dist/define-check.d.ts +93 -0
- package/dist/define-check.d.ts.map +1 -0
- package/dist/define-check.js +110 -0
- package/dist/define-check.js.map +1 -0
- package/dist/define.d.ts +27 -0
- package/dist/define.d.ts.map +1 -0
- package/dist/define.js +126 -0
- package/dist/define.js.map +1 -0
- package/dist/engine.d.ts +49 -0
- package/dist/engine.d.ts.map +1 -0
- package/dist/engine.js +198 -0
- package/dist/engine.js.map +1 -0
- package/dist/errors.d.ts +91 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +4 -0
- package/dist/errors.js.map +1 -0
- package/dist/execution-strategy.d.ts +80 -0
- package/dist/execution-strategy.d.ts.map +1 -0
- package/dist/execution-strategy.js +96 -0
- package/dist/execution-strategy.js.map +1 -0
- package/dist/guardrail.schema.json +261 -0
- package/dist/index.d.ts +16 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11 -0
- package/dist/index.js.map +1 -0
- package/dist/judge.d.ts +31 -0
- package/dist/judge.d.ts.map +1 -0
- package/dist/judge.js +171 -0
- package/dist/judge.js.map +1 -0
- package/dist/types.d.ts +407 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +11 -0
- package/dist/types.js.map +1 -0
- package/package.json +64 -4
- package/src/action-handler.ts +179 -0
- package/src/checks.ts +236 -0
- package/src/define-check.ts +207 -0
- package/src/define.ts +146 -0
- package/src/engine.ts +271 -0
- package/src/errors.ts +107 -0
- package/src/execution-strategy.ts +184 -0
- package/src/guardrail.schema.json +261 -0
- package/src/index.ts +79 -0
- package/src/judge.ts +221 -0
- package/src/types.ts +455 -0
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://kindgi.com/schemas/v1/guardrail.schema.json",
|
|
4
|
+
"$comment": "schema-version: 1.1.0",
|
|
5
|
+
"title": "Guardrail",
|
|
6
|
+
"description": "An assertion about agent behavior. Enforced at runtime (halt / retry / escalate / log-only / compensate per guardrail) AND in CI (blocks builds). Same definition, two enforcement paths — no drift between what tests check and what prod enforces. Zero-LLM checks are default (fast, deterministic); LLM-judge scorers are an opt-in class with explicit cost declaration. Schema-version 1.1.0 matches the `Guardrail` type: `check` is required, a `retry` action needs `maxAttempts`, `kind` and `on-violation` accept any non-empty string (the built-ins are listed as `examples`), and `sandbox` / `limits` / `network` / `needsSpec` are allowed.",
|
|
7
|
+
"type": "object",
|
|
8
|
+
"required": ["id", "kind", "check", "action"],
|
|
9
|
+
"additionalProperties": false,
|
|
10
|
+
"properties": {
|
|
11
|
+
"id": {
|
|
12
|
+
"type": "string",
|
|
13
|
+
"minLength": 1,
|
|
14
|
+
"description": "Unique guardrail identifier."
|
|
15
|
+
},
|
|
16
|
+
"name": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"description": "Human-readable name."
|
|
19
|
+
},
|
|
20
|
+
"description": {
|
|
21
|
+
"type": "string",
|
|
22
|
+
"description": "What this guardrail guarantees, and what happens if violated."
|
|
23
|
+
},
|
|
24
|
+
"kind": {
|
|
25
|
+
"type": "string",
|
|
26
|
+
"minLength": 1,
|
|
27
|
+
"examples": ["zero-llm", "llm-judge", "external"],
|
|
28
|
+
"description": "How the guardrail is checked. Open string: the engine dispatches it to the execution strategy registered for that kind, and `defineGuardrail` requires a registered check of the same kind. Built-in kinds: 'zero-llm' = pure function over run trace (default, fast, deterministic). 'llm-judge' = uses a model to score (opt-in, costs money). 'external' = evaluated outside the engine by a caller-registered strategy (the built-in 'external' strategy returns an error). Adapters can register strategies for their own kinds."
|
|
29
|
+
},
|
|
30
|
+
"check": {
|
|
31
|
+
"type": "string",
|
|
32
|
+
"minLength": 1,
|
|
33
|
+
"description": "Id of the check in the CheckRegistry. For zero-llm: a built-in check ('must-cite', 'never-call-tool', 'max-tool-calls', 'output-matches', 'tool-order', 'required-substring', 'forbidden-substring') or a custom check id. For llm-judge: the id of a registered check of kind 'llm-judge' (the judge is configured by config and judgeCapabilities). For external and adapter kinds: an id the kind's strategy understands. Required since schema-version 1.1.0 (`defineGuardrail` already failed without a registered check)."
|
|
34
|
+
},
|
|
35
|
+
"config": {
|
|
36
|
+
"type": "object",
|
|
37
|
+
"additionalProperties": true,
|
|
38
|
+
"description": "Check-specific configuration. Interpreted by the check implementation."
|
|
39
|
+
},
|
|
40
|
+
"action": {
|
|
41
|
+
"$ref": "#/$defs/Action"
|
|
42
|
+
},
|
|
43
|
+
"severity": {
|
|
44
|
+
"type": "string",
|
|
45
|
+
"enum": ["info", "warn", "error", "critical"],
|
|
46
|
+
"default": "error"
|
|
47
|
+
},
|
|
48
|
+
"scope": {
|
|
49
|
+
"$ref": "#/$defs/Scope",
|
|
50
|
+
"description": "When this guardrail applies: 'always' (every run), 'ci-only' (blocks CI, not runtime), 'runtime-only' (runtime enforcement, not CI), or a per-agent/per-flow selector."
|
|
51
|
+
},
|
|
52
|
+
"budget": {
|
|
53
|
+
"type": "object",
|
|
54
|
+
"additionalProperties": false,
|
|
55
|
+
"description": "For llm-judge guardrails: cost budget per invocation. Declarative — not enforced by the runtime.",
|
|
56
|
+
"properties": {
|
|
57
|
+
"maxCostUsd": {
|
|
58
|
+
"type": "number",
|
|
59
|
+
"minimum": 0
|
|
60
|
+
},
|
|
61
|
+
"maxLatencyMs": {
|
|
62
|
+
"type": "integer",
|
|
63
|
+
"minimum": 0
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
"sandbox": {
|
|
68
|
+
"type": "string",
|
|
69
|
+
"enum": ["none", "context-isolated", "strict"],
|
|
70
|
+
"description": "Isolation posture the runtime enforces around the check's handler. Same values as a tool's `sandbox`. Added in schema-version 1.1.0."
|
|
71
|
+
},
|
|
72
|
+
"limits": {
|
|
73
|
+
"type": "object",
|
|
74
|
+
"additionalProperties": false,
|
|
75
|
+
"description": "Sandbox-enforced resource caps while the check runs. Same shape as a tool's `limits`. Added in schema-version 1.1.0.",
|
|
76
|
+
"properties": {
|
|
77
|
+
"memMB": {
|
|
78
|
+
"type": "integer",
|
|
79
|
+
"minimum": 1
|
|
80
|
+
},
|
|
81
|
+
"cpuMs": {
|
|
82
|
+
"type": "integer",
|
|
83
|
+
"minimum": 1
|
|
84
|
+
}
|
|
85
|
+
},
|
|
86
|
+
"required": ["memMB", "cpuMs"]
|
|
87
|
+
},
|
|
88
|
+
"network": {
|
|
89
|
+
"type": "object",
|
|
90
|
+
"description": "Network egress policy the sandbox honors while the check runs. Discriminated on `kind`; same shape as a tool's `network`. Added in schema-version 1.1.0.",
|
|
91
|
+
"oneOf": [
|
|
92
|
+
{
|
|
93
|
+
"type": "object",
|
|
94
|
+
"additionalProperties": false,
|
|
95
|
+
"properties": {
|
|
96
|
+
"kind": {
|
|
97
|
+
"type": "string",
|
|
98
|
+
"enum": ["none", "unrestricted"]
|
|
99
|
+
}
|
|
100
|
+
},
|
|
101
|
+
"required": ["kind"]
|
|
102
|
+
},
|
|
103
|
+
{
|
|
104
|
+
"type": "object",
|
|
105
|
+
"additionalProperties": false,
|
|
106
|
+
"properties": {
|
|
107
|
+
"kind": {
|
|
108
|
+
"type": "string",
|
|
109
|
+
"const": "allowlist"
|
|
110
|
+
},
|
|
111
|
+
"hosts": {
|
|
112
|
+
"type": "array",
|
|
113
|
+
"items": {
|
|
114
|
+
"type": "string",
|
|
115
|
+
"minLength": 1
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
},
|
|
119
|
+
"required": ["kind", "hosts"]
|
|
120
|
+
}
|
|
121
|
+
]
|
|
122
|
+
},
|
|
123
|
+
"needsSpec": {
|
|
124
|
+
"type": "object",
|
|
125
|
+
"additionalProperties": false,
|
|
126
|
+
"description": "Typed discriminated needs of the check (env / secrets / config / capabilities / bindings). Same shape as a tool's `needsSpec`. Added in schema-version 1.1.0.",
|
|
127
|
+
"properties": {
|
|
128
|
+
"env": {
|
|
129
|
+
"type": "object",
|
|
130
|
+
"additionalProperties": {
|
|
131
|
+
"type": "object"
|
|
132
|
+
}
|
|
133
|
+
},
|
|
134
|
+
"secrets": {
|
|
135
|
+
"type": "object",
|
|
136
|
+
"additionalProperties": {
|
|
137
|
+
"type": "object"
|
|
138
|
+
}
|
|
139
|
+
},
|
|
140
|
+
"config": {
|
|
141
|
+
"type": "object",
|
|
142
|
+
"additionalProperties": {
|
|
143
|
+
"type": "object"
|
|
144
|
+
}
|
|
145
|
+
},
|
|
146
|
+
"capabilities": {
|
|
147
|
+
"type": "array",
|
|
148
|
+
"items": {
|
|
149
|
+
"type": "string",
|
|
150
|
+
"minLength": 1
|
|
151
|
+
}
|
|
152
|
+
},
|
|
153
|
+
"bindings": {
|
|
154
|
+
"type": "array",
|
|
155
|
+
"items": {
|
|
156
|
+
"type": "string",
|
|
157
|
+
"minLength": 1
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
},
|
|
162
|
+
"codeArtifactRef": {
|
|
163
|
+
"description": "Handler-artifact pointer for the guardrail's check implementation. Discriminated on `kind`: `oci` is the deploy-pipeline shape (image + module path + artifactVersion); `filesystem` is the local-development shape — absolute host path at the check module. Production servers SHOULD reject `filesystem`.",
|
|
164
|
+
"oneOf": [
|
|
165
|
+
{
|
|
166
|
+
"type": "object",
|
|
167
|
+
"additionalProperties": false,
|
|
168
|
+
"properties": {
|
|
169
|
+
"kind": { "const": "oci" },
|
|
170
|
+
"imageRef": { "type": "string", "minLength": 1 },
|
|
171
|
+
"modulePath": { "type": "string", "minLength": 1 },
|
|
172
|
+
"artifactVersion": { "type": "string", "minLength": 1 }
|
|
173
|
+
},
|
|
174
|
+
"required": ["kind", "imageRef", "modulePath", "artifactVersion"]
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
"type": "object",
|
|
178
|
+
"additionalProperties": false,
|
|
179
|
+
"properties": {
|
|
180
|
+
"kind": { "const": "filesystem" },
|
|
181
|
+
"modulePath": { "type": "string", "minLength": 1 }
|
|
182
|
+
},
|
|
183
|
+
"required": ["kind", "modulePath"]
|
|
184
|
+
}
|
|
185
|
+
]
|
|
186
|
+
},
|
|
187
|
+
"judgeCapabilities": {
|
|
188
|
+
"type": "object",
|
|
189
|
+
"description": "For llm-judge guardrails: capability declaration for the judge model. Routed through `@kindgi/capabilities` over the providers bound for evaluation; when the caller passes a tenant policy (an agent turn passes its own), the judge is routed under it. BYO judges supported.",
|
|
190
|
+
"additionalProperties": true
|
|
191
|
+
}
|
|
192
|
+
},
|
|
193
|
+
"$defs": {
|
|
194
|
+
"Action": {
|
|
195
|
+
"type": "object",
|
|
196
|
+
"required": ["on-violation"],
|
|
197
|
+
"additionalProperties": false,
|
|
198
|
+
"properties": {
|
|
199
|
+
"on-violation": {
|
|
200
|
+
"type": "string",
|
|
201
|
+
"minLength": 1,
|
|
202
|
+
"examples": ["halt", "retry", "escalate", "log-only", "compensate"],
|
|
203
|
+
"description": "What happens when the guardrail fires. Open string: any non-empty name is accepted here, and an action handler registered under that name applies it; when the caller evaluates with an action-handler registry, a name with no handler fails with `unknown-action`. Built-in actions: 'halt' = fail the run. 'retry' = re-execute the step (needs `retry.maxAttempts`). 'escalate' = route to HITL review. 'log-only' = record but don't block. 'compensate' = invoke a compensation action."
|
|
204
|
+
},
|
|
205
|
+
"retry": {
|
|
206
|
+
"type": "object",
|
|
207
|
+
"additionalProperties": false,
|
|
208
|
+
"description": "For on-violation = 'retry'.",
|
|
209
|
+
"required": ["maxAttempts"],
|
|
210
|
+
"properties": {
|
|
211
|
+
"maxAttempts": {
|
|
212
|
+
"type": "integer",
|
|
213
|
+
"minimum": 1,
|
|
214
|
+
"maximum": 10
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
},
|
|
218
|
+
"escalateTo": {
|
|
219
|
+
"type": "string",
|
|
220
|
+
"description": "For on-violation = 'escalate': reviewer role or queue id."
|
|
221
|
+
},
|
|
222
|
+
"compensateWith": {
|
|
223
|
+
"type": "string",
|
|
224
|
+
"description": "For on-violation = 'compensate': tool id to invoke as compensation."
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
},
|
|
228
|
+
"Scope": {
|
|
229
|
+
"type": "object",
|
|
230
|
+
"additionalProperties": false,
|
|
231
|
+
"properties": {
|
|
232
|
+
"when": {
|
|
233
|
+
"type": "string",
|
|
234
|
+
"enum": ["always", "ci-only", "runtime-only"],
|
|
235
|
+
"default": "always"
|
|
236
|
+
},
|
|
237
|
+
"agents": {
|
|
238
|
+
"type": "array",
|
|
239
|
+
"items": {
|
|
240
|
+
"type": "string"
|
|
241
|
+
},
|
|
242
|
+
"description": "Agent ids this guardrail applies to. Empty = all agents."
|
|
243
|
+
},
|
|
244
|
+
"flows": {
|
|
245
|
+
"type": "array",
|
|
246
|
+
"items": {
|
|
247
|
+
"type": "string"
|
|
248
|
+
},
|
|
249
|
+
"description": "Flow ids this guardrail applies to. Empty = all flows."
|
|
250
|
+
},
|
|
251
|
+
"tenants": {
|
|
252
|
+
"type": "array",
|
|
253
|
+
"items": {
|
|
254
|
+
"type": "string"
|
|
255
|
+
},
|
|
256
|
+
"description": "Tenant ids this guardrail applies to. Empty = all tenants."
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
export { GUARDRAIL_SCHEMA_URI, defineGuardrail, validateGuardrailSpec } from './define.js';
|
|
5
|
+
export { defineCheck } from './define-check.js';
|
|
6
|
+
export type { DefineCheckSpec, DefinedCheck, InferCheckConfig } from './define-check.js';
|
|
7
|
+
export { BUILT_IN_CHECK_IDS, createCheckRegistry } from './checks.js';
|
|
8
|
+
export {
|
|
9
|
+
BUILT_IN_ACTION_HANDLERS,
|
|
10
|
+
compensateHandler,
|
|
11
|
+
createActionHandlerRegistry,
|
|
12
|
+
escalateHandler,
|
|
13
|
+
haltHandler,
|
|
14
|
+
logOnlyHandler,
|
|
15
|
+
noopHandler,
|
|
16
|
+
retryHandler,
|
|
17
|
+
} from './action-handler.js';
|
|
18
|
+
export type {
|
|
19
|
+
ActionContext,
|
|
20
|
+
ActionHandler,
|
|
21
|
+
ActionHandlerRegistry,
|
|
22
|
+
ActionResult,
|
|
23
|
+
} from './action-handler.js';
|
|
24
|
+
export { builtInStrategies, evaluateAll, evaluateGuardrail, violations } from './engine.js';
|
|
25
|
+
export type { EvaluationOutcome } from './engine.js';
|
|
26
|
+
export {
|
|
27
|
+
createExecutionStrategyRegistry,
|
|
28
|
+
externalStrategy,
|
|
29
|
+
makeLlmJudgeStrategy,
|
|
30
|
+
zeroLlmStrategy,
|
|
31
|
+
} from './execution-strategy.js';
|
|
32
|
+
export type {
|
|
33
|
+
ExecutionStrategy,
|
|
34
|
+
ExecutionStrategyRegistry,
|
|
35
|
+
StrategyError,
|
|
36
|
+
StrategyResult,
|
|
37
|
+
} from './execution-strategy.js';
|
|
38
|
+
export { invokeJudge } from './judge.js';
|
|
39
|
+
export type { LlmJudgeConfig } from './judge.js';
|
|
40
|
+
export type {
|
|
41
|
+
Action,
|
|
42
|
+
Budget,
|
|
43
|
+
BuiltInGuardrailKind,
|
|
44
|
+
BuiltInOnViolation,
|
|
45
|
+
CheckFunction,
|
|
46
|
+
CheckRegistry,
|
|
47
|
+
CheckResult,
|
|
48
|
+
CodeArtifactRef,
|
|
49
|
+
EvaluationBindings,
|
|
50
|
+
EvaluationResult,
|
|
51
|
+
Guardrail,
|
|
52
|
+
GuardrailKind,
|
|
53
|
+
GuardrailSeverity,
|
|
54
|
+
JsonSchema,
|
|
55
|
+
ModelCallRecord,
|
|
56
|
+
NetworkPolicy,
|
|
57
|
+
OnViolation,
|
|
58
|
+
RegisteredCheck,
|
|
59
|
+
RunTrace,
|
|
60
|
+
RuntimeLimits,
|
|
61
|
+
SandboxMode,
|
|
62
|
+
Scope,
|
|
63
|
+
ScopeWhen,
|
|
64
|
+
ToolCallRecord,
|
|
65
|
+
ToolResultRecord,
|
|
66
|
+
TypedNeeds,
|
|
67
|
+
} from './types.js';
|
|
68
|
+
export { BUILT_IN_GUARDRAIL_KINDS, BUILT_IN_ON_VIOLATIONS } from './types.js';
|
|
69
|
+
export type {
|
|
70
|
+
InvalidCheckConfigError,
|
|
71
|
+
InvalidCheckDefinitionError,
|
|
72
|
+
InvalidGuardrailError,
|
|
73
|
+
GuardrailError,
|
|
74
|
+
JudgeMissingError,
|
|
75
|
+
JudgeRoutingError,
|
|
76
|
+
ScopeMismatchError,
|
|
77
|
+
UnknownActionError,
|
|
78
|
+
UnknownCheckError,
|
|
79
|
+
} from './errors.js';
|
package/src/judge.ts
ADDED
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
import { route } from '@kindgi/capabilities';
|
|
5
|
+
import type { Capability, ModelInfo, ModelProvider } from '@kindgi/capabilities';
|
|
6
|
+
|
|
7
|
+
import type { JudgeMissingError, JudgeRoutingError } from './errors.js';
|
|
8
|
+
import type { CheckResult, EvaluationBindings, RunTrace } from './types.js';
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Configuration for an llm-judge guardrail, read from the guardrail's
|
|
12
|
+
* `config`. `invokeJudge` builds the prompt around `rubric`.
|
|
13
|
+
*/
|
|
14
|
+
export interface LlmJudgeConfig {
|
|
15
|
+
/** The rubric prompt — what the judge model evaluates against. */
|
|
16
|
+
readonly rubric: string;
|
|
17
|
+
/**
|
|
18
|
+
* How the judge should respond. `'pass-fail'` (default) expects a
|
|
19
|
+
* response starting with PASS or FAIL and a reason. `'score'` expects
|
|
20
|
+
* a numeric score in [0, 1] on the first line, compared to `threshold`.
|
|
21
|
+
*/
|
|
22
|
+
readonly responseFormat?: 'pass-fail' | 'score';
|
|
23
|
+
/** For `'score'` format — pass threshold (inclusive). Default `0.5`. */
|
|
24
|
+
readonly threshold?: number;
|
|
25
|
+
/** Optional override for the model's temperature. */
|
|
26
|
+
readonly temperature?: number;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Invoke an LLM judge to evaluate a guardrail. Resolves a model via
|
|
31
|
+
* the capability router (or uses `bindings.judgeProvider` if set), sends
|
|
32
|
+
* a structured judgment prompt including the run trace + rubric, parses
|
|
33
|
+
* the response.
|
|
34
|
+
*/
|
|
35
|
+
export async function invokeJudge(
|
|
36
|
+
config: LlmJudgeConfig,
|
|
37
|
+
capability: Capability,
|
|
38
|
+
trace: RunTrace,
|
|
39
|
+
bindings: EvaluationBindings,
|
|
40
|
+
): Promise<CheckResult | { readonly error: JudgeMissingError | JudgeRoutingError }> {
|
|
41
|
+
const provider = await resolveJudgeProvider(capability, trace, bindings);
|
|
42
|
+
if ('error' in provider) return provider;
|
|
43
|
+
|
|
44
|
+
const responseFormat = config.responseFormat ?? 'pass-fail';
|
|
45
|
+
const rubric = config.rubric;
|
|
46
|
+
const prompt = buildJudgePrompt(rubric, responseFormat, trace);
|
|
47
|
+
|
|
48
|
+
const result = await provider.provider.invoke({
|
|
49
|
+
model: provider.model.name,
|
|
50
|
+
messages: [
|
|
51
|
+
{
|
|
52
|
+
role: 'system',
|
|
53
|
+
content:
|
|
54
|
+
responseFormat === 'pass-fail'
|
|
55
|
+
? 'You are an evaluator. Respond starting with PASS or FAIL on the first line, then a brief reason on the next line. No preamble.'
|
|
56
|
+
: 'You are an evaluator. Respond with a numeric score from 0 to 1 on the first line, then a brief reason on the next line. No preamble.',
|
|
57
|
+
},
|
|
58
|
+
{ role: 'user', content: prompt },
|
|
59
|
+
],
|
|
60
|
+
...(config.temperature !== undefined && { temperature: config.temperature }),
|
|
61
|
+
maxOutputTokens: 256,
|
|
62
|
+
...(bindings.abortSignal !== undefined && { abortSignal: bindings.abortSignal }),
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
const text = result.message.content.trim();
|
|
66
|
+
return parseJudgeResponse(text, responseFormat, config.threshold);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
interface ResolvedProvider {
|
|
70
|
+
readonly provider: ModelProvider;
|
|
71
|
+
readonly model: ModelInfo;
|
|
72
|
+
readonly reason: string;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
async function resolveJudgeProvider(
|
|
76
|
+
capability: Capability,
|
|
77
|
+
trace: RunTrace,
|
|
78
|
+
bindings: EvaluationBindings,
|
|
79
|
+
): Promise<ResolvedProvider | { readonly error: JudgeMissingError | JudgeRoutingError }> {
|
|
80
|
+
if (bindings.judgeProvider !== undefined && bindings.tenantPolicy !== undefined) {
|
|
81
|
+
// Override under a tenant policy: route over the override alone, so
|
|
82
|
+
// the policy (and the judge capability) still apply.
|
|
83
|
+
const decision = route({
|
|
84
|
+
capability,
|
|
85
|
+
providers: [bindings.judgeProvider],
|
|
86
|
+
tenantPolicy: bindings.tenantPolicy,
|
|
87
|
+
});
|
|
88
|
+
if (decision.kind === 'err') {
|
|
89
|
+
const err: JudgeRoutingError = {
|
|
90
|
+
code: 'judge-routing-failed',
|
|
91
|
+
message: `Explicit judgeProvider "${bindings.judgeProvider.metadata.id}" is not allowed for this tenant: ${decision.error.message}`,
|
|
92
|
+
guardrailId: '<unknown>' as never,
|
|
93
|
+
cause: decision.error,
|
|
94
|
+
};
|
|
95
|
+
return { error: err };
|
|
96
|
+
}
|
|
97
|
+
return {
|
|
98
|
+
provider: decision.value.provider,
|
|
99
|
+
model: decision.value.model,
|
|
100
|
+
reason: `explicit judgeProvider override; ${decision.value.reason}`,
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
if (bindings.judgeProvider !== undefined) {
|
|
104
|
+
// Direct provider override without a tenant policy — no routing
|
|
105
|
+
// decision, so pick the provider's first advertised model. Callers
|
|
106
|
+
// who need a specific model within the override should go through
|
|
107
|
+
// `providerRegistry` + the router.
|
|
108
|
+
const provider = bindings.judgeProvider;
|
|
109
|
+
const first = provider.metadata.models[0];
|
|
110
|
+
if (first === undefined) {
|
|
111
|
+
const err: JudgeRoutingError = {
|
|
112
|
+
code: 'judge-routing-failed',
|
|
113
|
+
message: `Explicit judgeProvider "${provider.metadata.id}" exposes zero models`,
|
|
114
|
+
guardrailId: '<unknown>' as never,
|
|
115
|
+
cause: {
|
|
116
|
+
code: 'capability-unsatisfiable',
|
|
117
|
+
message: 'judgeProvider has no models',
|
|
118
|
+
reasons: [],
|
|
119
|
+
},
|
|
120
|
+
};
|
|
121
|
+
return { error: err };
|
|
122
|
+
}
|
|
123
|
+
return { provider, model: first, reason: 'explicit judgeProvider override' };
|
|
124
|
+
}
|
|
125
|
+
if (bindings.providerRegistry === undefined) {
|
|
126
|
+
const err: JudgeMissingError = {
|
|
127
|
+
code: 'judge-missing',
|
|
128
|
+
message:
|
|
129
|
+
'llm-judge guardrail evaluated without providerRegistry or judgeProvider in bindings',
|
|
130
|
+
guardrailId: '<unknown>' as never,
|
|
131
|
+
};
|
|
132
|
+
return { error: err };
|
|
133
|
+
}
|
|
134
|
+
const decision = route({
|
|
135
|
+
capability,
|
|
136
|
+
providers: bindings.providerRegistry.list(trace.tenantId),
|
|
137
|
+
...(bindings.tenantPolicy !== undefined && { tenantPolicy: bindings.tenantPolicy }),
|
|
138
|
+
});
|
|
139
|
+
if (decision.kind === 'err') {
|
|
140
|
+
const err: JudgeRoutingError = {
|
|
141
|
+
code: 'judge-routing-failed',
|
|
142
|
+
message: `Judge routing failed: ${decision.error.message}`,
|
|
143
|
+
guardrailId: '<unknown>' as never,
|
|
144
|
+
cause: decision.error,
|
|
145
|
+
};
|
|
146
|
+
return { error: err };
|
|
147
|
+
}
|
|
148
|
+
return {
|
|
149
|
+
provider: decision.value.provider,
|
|
150
|
+
model: decision.value.model,
|
|
151
|
+
reason: decision.value.reason,
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
function buildJudgePrompt(rubric: string, format: 'pass-fail' | 'score', trace: RunTrace): string {
|
|
156
|
+
const lines = [
|
|
157
|
+
'Evaluate the following agent run against the rubric.',
|
|
158
|
+
'',
|
|
159
|
+
'=== Rubric ===',
|
|
160
|
+
rubric,
|
|
161
|
+
'',
|
|
162
|
+
'=== Run output ===',
|
|
163
|
+
trace.output ?? '<empty>',
|
|
164
|
+
'',
|
|
165
|
+
'=== Tool calls ===',
|
|
166
|
+
trace.toolCalls.length === 0
|
|
167
|
+
? '<none>'
|
|
168
|
+
: trace.toolCalls.map((c) => `- ${c.toolName}(${JSON.stringify(c.arguments)})`).join('\n'),
|
|
169
|
+
'',
|
|
170
|
+
format === 'pass-fail'
|
|
171
|
+
? 'Respond PASS or FAIL then reason.'
|
|
172
|
+
: 'Respond with a score in [0, 1] then reason.',
|
|
173
|
+
];
|
|
174
|
+
return lines.join('\n');
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
function parseJudgeResponse(
|
|
178
|
+
raw: string,
|
|
179
|
+
format: 'pass-fail' | 'score',
|
|
180
|
+
threshold: number | undefined,
|
|
181
|
+
): CheckResult {
|
|
182
|
+
if (format === 'pass-fail') {
|
|
183
|
+
const upper = raw.toUpperCase();
|
|
184
|
+
if (upper.startsWith('PASS')) {
|
|
185
|
+
const reasonLine = raw.split(/\r?\n/)[1]?.trim() ?? '';
|
|
186
|
+
const result: CheckResult = { passed: true, judgeResponse: raw };
|
|
187
|
+
return reasonLine.length > 0 ? { ...result, reason: reasonLine } : result;
|
|
188
|
+
}
|
|
189
|
+
if (upper.startsWith('FAIL')) {
|
|
190
|
+
const reasonLine = raw.split(/\r?\n/)[1]?.trim() ?? '';
|
|
191
|
+
return {
|
|
192
|
+
passed: false,
|
|
193
|
+
reason: reasonLine.length > 0 ? reasonLine : 'judge returned FAIL',
|
|
194
|
+
judgeResponse: raw,
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
return {
|
|
198
|
+
passed: false,
|
|
199
|
+
reason: `judge response did not start with PASS/FAIL: ${raw.slice(0, 80)}`,
|
|
200
|
+
judgeResponse: raw,
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
// score format
|
|
204
|
+
const firstLine = raw.split(/\r?\n/)[0]?.trim() ?? '';
|
|
205
|
+
const score = Number.parseFloat(firstLine);
|
|
206
|
+
if (!Number.isFinite(score)) {
|
|
207
|
+
return {
|
|
208
|
+
passed: false,
|
|
209
|
+
reason: `judge did not return a numeric score: ${firstLine}`,
|
|
210
|
+
judgeResponse: raw,
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
const cutoff = threshold ?? 0.5;
|
|
214
|
+
const reasonLine = raw.split(/\r?\n/)[1]?.trim() ?? '';
|
|
215
|
+
return {
|
|
216
|
+
passed: score >= cutoff,
|
|
217
|
+
reason: reasonLine.length > 0 ? reasonLine : `score=${score}, threshold=${cutoff}`,
|
|
218
|
+
judgeResponse: raw,
|
|
219
|
+
attributes: { score, threshold: cutoff },
|
|
220
|
+
};
|
|
221
|
+
}
|