@tangle-network/agent-eval 0.117.0 → 0.118.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4276 -74
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-DP3BBMYJ.js} +100 -143
- package/dist/chunk-DP3BBMYJ.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-ZUXV7UWZ.js → chunk-Q442S5AS.js} +73 -19
- package/dist/{chunk-ZUXV7UWZ.js.map → chunk-Q442S5AS.js.map} +1 -1
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10982 -1286
- package/dist/index.js +97 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1970 -697
- package/dist/traces.js +49 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
|
@@ -1,171 +0,0 @@
|
|
|
1
|
-
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
|
-
import { z } from 'zod';
|
|
4
|
-
import { g as AnalystCost, a as AnalystContext, A as Analyst } from './policy-edit-wG9uFEFm.js';
|
|
5
|
-
|
|
6
|
-
/**
|
|
7
|
-
* Typed Ax output for analyst findings.
|
|
8
|
-
*
|
|
9
|
-
* Replaces the legacy `findings:string[]` pattern (where every bullet
|
|
10
|
-
* became a flat-severity `AnalystFinding`) with a structured object
|
|
11
|
-
* array. Ax binds the field as `findings:json[]` so the provider emits
|
|
12
|
-
* native structured output; at the kind-factory boundary we Zod-validate
|
|
13
|
-
* each emitted finding so malformed rows fail loud instead of being
|
|
14
|
-
* silently lifted with default severity.
|
|
15
|
-
*
|
|
16
|
-
* Why not `f.object().array()` directly in the signature? The Ax
|
|
17
|
-
* signature string `question:string -> findings:json[]` already lets
|
|
18
|
-
* the provider emit JSON arrays. A Zod boundary is required either
|
|
19
|
-
* way (the provider can return any JSON), and Zod gives us a single
|
|
20
|
-
* validation surface independent of which Ax version is installed.
|
|
21
|
-
*/
|
|
22
|
-
|
|
23
|
-
declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
|
|
24
|
-
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
25
|
-
severity: z.ZodEnum<{
|
|
26
|
-
info: "info";
|
|
27
|
-
critical: "critical";
|
|
28
|
-
medium: "medium";
|
|
29
|
-
low: "low";
|
|
30
|
-
high: "high";
|
|
31
|
-
}>;
|
|
32
|
-
claim: z.ZodString;
|
|
33
|
-
subject: z.ZodOptional<z.ZodString>;
|
|
34
|
-
evidence_uri: z.ZodString;
|
|
35
|
-
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
36
|
-
confidence: z.ZodNumber;
|
|
37
|
-
rationale: z.ZodOptional<z.ZodString>;
|
|
38
|
-
recommended_action: z.ZodOptional<z.ZodString>;
|
|
39
|
-
}, z.core.$strict>;
|
|
40
|
-
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
41
|
-
/**
|
|
42
|
-
* Description embedded into the actor prompt so the LLM knows what
|
|
43
|
-
* shape to emit. Kept here so kinds share one source of truth rather
|
|
44
|
-
* than restating the schema in every prompt.
|
|
45
|
-
*/
|
|
46
|
-
declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a JSON object with these fields:\n - severity: one of \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: the routing locus this finding is about. It MUST be one of the exact subject forms listed in this kind's instructions above (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`, `tool-doc:<tool>`). A free phrase, a bare noun, or any form not in that list is REJECTED at parse time and the finding is discarded \u2014 omit subject entirely rather than guess a form.\n - evidence_uri: REQUIRED, never blank. Exactly one of \"span://<trace_id>/<span_id>\" (trace evidence), \"artifact://<relative-path>\" (files), \"metric://<name>\" (named scalars) \u2014 ALWAYS cite a real id surfaced by the tools. If you have no citable id, do not emit the finding.\n - evidence_excerpt?: short quote (<=2000 chars) from the cited span/artifact\n - confidence: number 0..1 \u2014 0.9+ when backed by exact quotes, 0.6-0.8 for inferred patterns, <0.5 for speculative\n - rationale?: one or two sentences explaining the reasoning\n - recommended_action?: concrete change phrased as an imperative (\"Add ...\", \"Replace ...\", \"Stop ...\") \u2014 omit when the finding is purely descriptive\n\nEmit an empty array when the question has no findings to report. Do not fabricate evidence.";
|
|
47
|
-
/**
|
|
48
|
-
* Validate one row emitted by the LLM. Returns the typed finding on
|
|
49
|
-
* success; returns `null` and logs the reason on failure so the kind
|
|
50
|
-
* factory can skip-and-count rather than abort the whole analyst run.
|
|
51
|
-
*/
|
|
52
|
-
declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
|
|
53
|
-
|
|
54
|
-
/**
|
|
55
|
-
* Analyst-kind factory — the typed way to define trace analysts.
|
|
56
|
-
*
|
|
57
|
-
* A "kind" is a specialized analyst whose actor prompt, tool subset,
|
|
58
|
-
* and Ax recursion config target one failure-mode lens (failure-mode
|
|
59
|
-
* classification, knowledge gap discovery, knowledge poisoning, recursive
|
|
60
|
-
* self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
|
|
61
|
-
* shape via a JSON-array Ax output; the factory validates each row with
|
|
62
|
-
* Zod and lifts it into `AnalystFinding[]` with no shape guessing.
|
|
63
|
-
*
|
|
64
|
-
* Composition rules:
|
|
65
|
-
* - Each kind owns its actor description. No generic "answer this
|
|
66
|
-
* question" prompt — the prompt names the failure lens.
|
|
67
|
-
* - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
|
|
68
|
-
* A kind that never needs full-trace dumps can drop `viewTrace` /
|
|
69
|
-
* `viewSpans` and stay cheap.
|
|
70
|
-
* - Each kind declares its recursion + parallelism budget. Discovery-
|
|
71
|
-
* heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
|
|
72
|
-
* (poisoning) usually stay at 0 since they have a tighter brief.
|
|
73
|
-
*
|
|
74
|
-
* Optimizer hook: kinds may declare `goldens` — labeled examples used
|
|
75
|
-
* by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
|
|
76
|
-
* description programmatically. Stored on the kind, not the registry,
|
|
77
|
-
* because the right metric is kind-specific.
|
|
78
|
-
*/
|
|
79
|
-
|
|
80
|
-
/**
|
|
81
|
-
* Per-kind specification. The factory turns this into a regular
|
|
82
|
-
* `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
|
|
83
|
-
*/
|
|
84
|
-
interface TraceAnalystKindSpec {
|
|
85
|
-
/** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
|
|
86
|
-
id: string;
|
|
87
|
-
/** One-sentence description shown in `registry.list()`. */
|
|
88
|
-
description: string;
|
|
89
|
-
/** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
|
|
90
|
-
area: string;
|
|
91
|
-
/** Bump on any breaking change to the actor prompt or output schema. */
|
|
92
|
-
version: string;
|
|
93
|
-
/** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
|
|
94
|
-
actorDescription: string;
|
|
95
|
-
/** Responder system prompt; falls back to a minimal "format the findings" instruction. */
|
|
96
|
-
responderDescription?: string;
|
|
97
|
-
/** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
|
|
98
|
-
buildTools: (store: TraceAnalysisStore) => AxFunction[];
|
|
99
|
-
/** Recursion budget. `maxDepth: 0` disables subagents. */
|
|
100
|
-
recursion?: {
|
|
101
|
-
maxDepth: number;
|
|
102
|
-
maxParallelSubagents?: number;
|
|
103
|
-
};
|
|
104
|
-
/** Actor turn cap. Default 12. */
|
|
105
|
-
maxTurns?: number;
|
|
106
|
-
/** Runtime char cap. Default 6000. */
|
|
107
|
-
maxRuntimeChars?: number;
|
|
108
|
-
/** Cost classification surfaced in `registry.list()` and budget enforcement. */
|
|
109
|
-
cost: AnalystCost;
|
|
110
|
-
/** Per-finding-row hook — kinds may reject / rewrite before lifting. */
|
|
111
|
-
postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
|
|
112
|
-
/** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
|
|
113
|
-
goldens?: TraceAnalystGolden[];
|
|
114
|
-
}
|
|
115
|
-
/**
|
|
116
|
-
* One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
|
|
117
|
-
* Each input is the same `{question}` an analyst would receive; `expected`
|
|
118
|
-
* is the ground-truth finding set a fitted prompt should produce on this
|
|
119
|
-
* input. Metric: kind-specific (default: F1 on `finding_id` overlap).
|
|
120
|
-
*/
|
|
121
|
-
interface TraceAnalystGolden {
|
|
122
|
-
question: string;
|
|
123
|
-
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
124
|
-
}
|
|
125
|
-
interface CreateTraceAnalystKindOpts {
|
|
126
|
-
/** AxAIService bound at registration time. */
|
|
127
|
-
ai: AxAIService;
|
|
128
|
-
/** Optional model override; falls back to the AI service's default. */
|
|
129
|
-
model?: string;
|
|
130
|
-
/** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
|
|
131
|
-
versionSuffix?: string;
|
|
132
|
-
/**
|
|
133
|
-
* Optional two-phase recovery: when the agentic harvest is empty but the
|
|
134
|
-
* actor produced a substantive free-form `report`, extract findings from that
|
|
135
|
-
* prose via a tolerant chat-completions pass (`structureFindings`) — no
|
|
136
|
-
* strict-emission contract, so it works on weak models. Omit to leave the
|
|
137
|
-
* actor's harvest as-is (the report is still surfaced fail-loud either way).
|
|
138
|
-
*/
|
|
139
|
-
recovery?: {
|
|
140
|
-
baseUrl: string;
|
|
141
|
-
apiKey?: string;
|
|
142
|
-
model?: string;
|
|
143
|
-
fetchImpl?: typeof fetch;
|
|
144
|
-
};
|
|
145
|
-
}
|
|
146
|
-
/**
|
|
147
|
-
* Build an `Analyst<TraceAnalysisStore>` from a kind spec.
|
|
148
|
-
*
|
|
149
|
-
* Lifts the Ax pipeline once at registration time so the registry
|
|
150
|
-
* gets a stateless analyst. The Ax agent is freshly constructed per
|
|
151
|
-
* `analyze()` call (the agent carries chat-log + usage state we don't
|
|
152
|
-
* want shared across analyst runs).
|
|
153
|
-
*/
|
|
154
|
-
declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
|
|
155
|
-
/**
|
|
156
|
-
* Render a compact prior-findings block the actor reads alongside its
|
|
157
|
-
* brief. Each row is one line so the actor can scan dozens cheaply.
|
|
158
|
-
* The kind's prompt instructs the actor to (a) check whether a new
|
|
159
|
-
* cluster matches a prior `finding_id` (carry the id forward via
|
|
160
|
-
* `id_basis` to keep diffs stable) and (b) raise severity / confidence
|
|
161
|
-
* when a prior finding has reappeared without remediation.
|
|
162
|
-
*
|
|
163
|
-
* Returns the empty string when there are no prior findings — most
|
|
164
|
-
* runs are "first-of-its-kind" and the prompt stays unchanged.
|
|
165
|
-
*
|
|
166
|
-
* Exported for tests + for consumers that build their own actor
|
|
167
|
-
* prompts (e.g. specialized analysts living outside the default kinds).
|
|
168
|
-
*/
|
|
169
|
-
declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
|
|
170
|
-
|
|
171
|
-
export { ANALYST_SEVERITIES as A, type CreateTraceAnalystKindOpts as C, RAW_FINDING_SCHEMA_PROMPT as R, type TraceAnalystKindSpec as T, type RawAnalystFinding as a, RawAnalystFindingSchema as b, type TraceAnalystGolden as c, createTraceAnalystKind as d, parseRawFinding as p, renderPriorFindings as r };
|
|
@@ -1,289 +0,0 @@
|
|
|
1
|
-
import { c as CostReceiptInput, M as MaximumCharge } from './cost-ledger-DWy3XdJc.js';
|
|
2
|
-
import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
3
|
-
import { R as RawProviderSink, P as ProviderRedactor } from './raw-provider-sink-C46HDghv.js';
|
|
4
|
-
|
|
5
|
-
/**
|
|
6
|
-
* LLM client with graceful degrade.
|
|
7
|
-
*
|
|
8
|
-
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
9
|
-
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
10
|
-
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
11
|
-
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
12
|
-
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
13
|
-
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
14
|
-
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
15
|
-
*
|
|
16
|
-
* Usage:
|
|
17
|
-
* const { value, result } = await callLlmJson<MyType>(
|
|
18
|
-
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
19
|
-
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
20
|
-
* )
|
|
21
|
-
*
|
|
22
|
-
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
23
|
-
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
24
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
25
|
-
*/
|
|
26
|
-
|
|
27
|
-
interface LlmMessage {
|
|
28
|
-
role: 'system' | 'user' | 'assistant';
|
|
29
|
-
/**
|
|
30
|
-
* Either a plain text content string OR a multimodal content array
|
|
31
|
-
* (text + image_url parts) for vision-capable models.
|
|
32
|
-
*/
|
|
33
|
-
content: string | Array<{
|
|
34
|
-
type: 'text';
|
|
35
|
-
text: string;
|
|
36
|
-
} | {
|
|
37
|
-
type: 'image_url';
|
|
38
|
-
image_url: {
|
|
39
|
-
url: string;
|
|
40
|
-
detail?: 'auto' | 'low' | 'high';
|
|
41
|
-
};
|
|
42
|
-
}>;
|
|
43
|
-
}
|
|
44
|
-
interface LlmCallRequest {
|
|
45
|
-
model: string;
|
|
46
|
-
messages: LlmMessage[];
|
|
47
|
-
/** Optional JSON-mode response format (response_format: json_object). */
|
|
48
|
-
jsonMode?: boolean;
|
|
49
|
-
/** Optional structured output via JSON Schema. Falls back to json_object on 400. */
|
|
50
|
-
jsonSchema?: {
|
|
51
|
-
name: string;
|
|
52
|
-
schema: Record<string, unknown>;
|
|
53
|
-
};
|
|
54
|
-
temperature?: number;
|
|
55
|
-
maxTokens?: number;
|
|
56
|
-
/** Per-call timeout, default 300s. */
|
|
57
|
-
timeoutMs?: number;
|
|
58
|
-
}
|
|
59
|
-
/** Conservative priced bound for the exact text request sent to a provider.
|
|
60
|
-
* Returns undefined when output or multimodal input is not bounded, causing a
|
|
61
|
-
* capped CostLedger to reject the call before execution. */
|
|
62
|
-
declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens'>, options?: LlmClientOptions): MaximumCharge | undefined;
|
|
63
|
-
interface LlmUsage {
|
|
64
|
-
promptTokens: number;
|
|
65
|
-
completionTokens: number;
|
|
66
|
-
totalTokens: number;
|
|
67
|
-
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
68
|
-
captured?: boolean;
|
|
69
|
-
/** Proxies populate this when prompt caching is on. */
|
|
70
|
-
cachedPromptTokens?: number;
|
|
71
|
-
}
|
|
72
|
-
interface LlmCallResult {
|
|
73
|
-
/** The text content of the first choice. Empty string if none. */
|
|
74
|
-
content: string;
|
|
75
|
-
usage: LlmUsage;
|
|
76
|
-
/**
|
|
77
|
-
* Cost in USD. Pulled from proxy's `_response_cost` field when present;
|
|
78
|
-
* `null` when neither the proxy nor the caller can derive it.
|
|
79
|
-
*/
|
|
80
|
-
costUsd: number | null;
|
|
81
|
-
/** Model name actually used (echoed from response). */
|
|
82
|
-
model: string;
|
|
83
|
-
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
84
|
-
durationMs: number;
|
|
85
|
-
/**
|
|
86
|
-
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
87
|
-
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
88
|
-
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
89
|
-
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
90
|
-
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
91
|
-
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
92
|
-
*/
|
|
93
|
-
finishReason?: string | null;
|
|
94
|
-
/**
|
|
95
|
-
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
96
|
-
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
97
|
-
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
98
|
-
* surfaces it but does not throw on it.
|
|
99
|
-
*/
|
|
100
|
-
contentEmpty?: boolean;
|
|
101
|
-
/** Raw response body. */
|
|
102
|
-
raw: Record<string, unknown>;
|
|
103
|
-
}
|
|
104
|
-
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
105
|
-
/** Convert a provider result into the canonical paid-call receipt input. */
|
|
106
|
-
declare function costReceiptFromLlm(result: LlmCallResult): CostReceiptInput;
|
|
107
|
-
/** Structured-response failures retain their completed provider receipt. */
|
|
108
|
-
declare function costReceiptFromLlmError(error: Error): CostReceiptInput | undefined;
|
|
109
|
-
declare class LlmCallError extends AgentEvalError {
|
|
110
|
-
readonly status: number;
|
|
111
|
-
readonly body: string;
|
|
112
|
-
readonly model: string;
|
|
113
|
-
constructor(message: string, status: number, body: string, model: string);
|
|
114
|
-
}
|
|
115
|
-
/** A provider response completed and incurred measurable usage, but its content
|
|
116
|
-
* could not satisfy the caller's response contract. The response envelope is
|
|
117
|
-
* retained so accounting can commit the receipt before the error propagates. */
|
|
118
|
-
declare class LlmResponseError extends AgentEvalError {
|
|
119
|
-
readonly result: LlmCallResult;
|
|
120
|
-
constructor(message: string, result: LlmCallResult, options?: {
|
|
121
|
-
cause?: unknown;
|
|
122
|
-
});
|
|
123
|
-
}
|
|
124
|
-
interface LlmClientOptions {
|
|
125
|
-
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
126
|
-
baseUrl?: string;
|
|
127
|
-
/** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
|
|
128
|
-
apiKey?: string;
|
|
129
|
-
bearer?: string;
|
|
130
|
-
/** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
|
|
131
|
-
authHeader?: {
|
|
132
|
-
name: string;
|
|
133
|
-
value: string;
|
|
134
|
-
};
|
|
135
|
-
/** Stable provider idempotency key, reused across retries of this logical call. */
|
|
136
|
-
idempotencyKey?: string;
|
|
137
|
-
/** Default timeout in ms. Per-call can override. */
|
|
138
|
-
defaultTimeoutMs?: number;
|
|
139
|
-
/**
|
|
140
|
-
* Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
|
|
141
|
-
* each attempt's per-attempt timeout controller, so aborting it cancels
|
|
142
|
-
* the in-flight fetch. A caller abort is FATAL: it is not retried even
|
|
143
|
-
* though an AbortError otherwise matches the transient patterns.
|
|
144
|
-
*/
|
|
145
|
-
signal?: AbortSignal;
|
|
146
|
-
/**
|
|
147
|
-
* Cross-attempt wall-clock budget in ms, measured from the first attempt.
|
|
148
|
-
* Before launching each attempt the loop checks the remaining budget and
|
|
149
|
-
* stops retrying once it is exhausted, rather than waiting the full
|
|
150
|
-
* per-attempt timeout on every retry. Bounds total time independent of
|
|
151
|
-
* total attempts × `timeoutMs`.
|
|
152
|
-
*/
|
|
153
|
-
deadlineMs?: number;
|
|
154
|
-
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
155
|
-
maxRetries?: number;
|
|
156
|
-
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
157
|
-
fetch?: typeof fetch;
|
|
158
|
-
/**
|
|
159
|
-
* Optional raw HTTP capture sink. When provided, every request, response,
|
|
160
|
-
* and error (across all retry attempts) is recorded to the sink, with auth
|
|
161
|
-
* headers and credential-shaped body fields redacted by default. This is
|
|
162
|
-
* the layer-1 forensics primitive: structured `LlmSpan`s record intent,
|
|
163
|
-
* raw events record what actually crossed the wire.
|
|
164
|
-
*/
|
|
165
|
-
rawSink?: RawProviderSink;
|
|
166
|
-
/**
|
|
167
|
-
* Logical provider id attached to raw events. When omitted, derived from
|
|
168
|
-
* `baseUrl` via `providerFromBaseUrl`.
|
|
169
|
-
*/
|
|
170
|
-
provider?: string;
|
|
171
|
-
/** Trace context attached to raw events; populated by emitter-aware callers. */
|
|
172
|
-
traceContext?: {
|
|
173
|
-
runId?: string;
|
|
174
|
-
spanId?: string;
|
|
175
|
-
};
|
|
176
|
-
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
177
|
-
redactor?: ProviderRedactor;
|
|
178
|
-
}
|
|
179
|
-
/**
|
|
180
|
-
* True when an error is a transient transport/network fault worth retrying,
|
|
181
|
-
* as opposed to a deterministic failure (4xx schema reject, JSON parse) that
|
|
182
|
-
* a retry cannot fix. Inspects `LlmCallError.status`, then the error's
|
|
183
|
-
* name/message/code, then recurses into `error.cause` — undici nests the
|
|
184
|
-
* real socket fault one or more levels under `.cause`.
|
|
185
|
-
*
|
|
186
|
-
* This is THE retry classifier for the package: `callLlm` and
|
|
187
|
-
* `withJudgeRetry` both route through it, so a connection-class error is
|
|
188
|
-
* treated identically whether it surfaces in the HTTP client or a
|
|
189
|
-
* TCloud-backed judge.
|
|
190
|
-
*/
|
|
191
|
-
declare function isTransientLlmError(err: unknown): boolean;
|
|
192
|
-
/** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
|
|
193
|
-
declare function backoffMs(attempt: number): number;
|
|
194
|
-
/**
|
|
195
|
-
* Strip a ```json / ``` code fence if the model emitted one.
|
|
196
|
-
* Idempotent for naked JSON. Some models (claude-code via router, certain
|
|
197
|
-
* deepseek models) wrap output even under json_object.
|
|
198
|
-
*/
|
|
199
|
-
declare function stripFencedJson(raw: string): string;
|
|
200
|
-
/**
|
|
201
|
-
* Low-level call. Returns raw content + usage + cost. Retries on transient
|
|
202
|
-
* failures; does NOT degrade schema here — callers that want graceful
|
|
203
|
-
* degrade use `callLlmJson`.
|
|
204
|
-
*/
|
|
205
|
-
declare function callLlm(req: LlmCallRequest, opts?: LlmClientOptions): Promise<LlmCallResult>;
|
|
206
|
-
/**
|
|
207
|
-
* Structured-output call. Returns parsed JSON plus the raw result envelope.
|
|
208
|
-
* Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —
|
|
209
|
-
* critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept
|
|
210
|
-
* the `response_format.json_schema` shape but DO accept `json_object`.
|
|
211
|
-
*/
|
|
212
|
-
declare function callLlmJson<T = unknown>(req: LlmCallRequest, opts?: LlmClientOptions): Promise<{
|
|
213
|
-
value: T;
|
|
214
|
-
result: LlmCallResult;
|
|
215
|
-
}>;
|
|
216
|
-
type LlmRouteAssertionReason = 'no_explicit_base_url' | 'base_url_blocked' | 'base_url_not_allowed' | 'no_auth' | 'wrong_provider';
|
|
217
|
-
declare class LlmRouteAssertionError extends CaptureIntegrityError {
|
|
218
|
-
readonly reason: LlmRouteAssertionReason;
|
|
219
|
-
readonly baseUrl: string;
|
|
220
|
-
constructor(message: string, reason: LlmRouteAssertionReason, baseUrl: string);
|
|
221
|
-
}
|
|
222
|
-
interface LlmRouteRequirements {
|
|
223
|
-
/**
|
|
224
|
-
* Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
|
|
225
|
-
* `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
|
|
226
|
-
* the public/free-tier router is a defect — the launch reviewer needs to
|
|
227
|
-
* know exactly which provider answered.
|
|
228
|
-
*/
|
|
229
|
-
requireExplicitBaseUrl?: boolean;
|
|
230
|
-
/**
|
|
231
|
-
* Allowlist of acceptable base URLs. Strings match by prefix
|
|
232
|
-
* (case-insensitive); RegExps test against the full base URL.
|
|
233
|
-
*/
|
|
234
|
-
allowedBaseUrls?: Array<string | RegExp>;
|
|
235
|
-
/** Blocklist that takes precedence over `allowedBaseUrls`. */
|
|
236
|
-
blockedBaseUrls?: Array<string | RegExp>;
|
|
237
|
-
/** Throw if no auth header / api key is configured. */
|
|
238
|
-
requireAuth?: boolean;
|
|
239
|
-
/**
|
|
240
|
-
* Logical provider id the configured `baseUrl` is expected to match (via
|
|
241
|
-
* `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
|
|
242
|
-
*/
|
|
243
|
-
expectedProvider?: string;
|
|
244
|
-
}
|
|
245
|
-
/**
|
|
246
|
-
* Fail-loud assertion that the configured LLM client points at the route
|
|
247
|
-
* the caller intends. Designed for the matrix-runner preflight: invoke
|
|
248
|
-
* once before any LLM call to catch misconfiguration before a sweep burns
|
|
249
|
-
* dollars on the wrong provider.
|
|
250
|
-
*
|
|
251
|
-
* Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call
|
|
252
|
-
* from constructors and CI gates.
|
|
253
|
-
*/
|
|
254
|
-
declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequirements): void;
|
|
255
|
-
/**
|
|
256
|
-
* Probe whether a model is reachable. Returns latency + null error on
|
|
257
|
-
* success; `ok=false` + error message on any failure (HTTP, timeout,
|
|
258
|
-
* network, parse). Designed for sweep preflights — fail loud at the
|
|
259
|
-
* boundary before burning a 30-leaf run on a misconfigured router.
|
|
260
|
-
*
|
|
261
|
-
* Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
|
|
262
|
-
* (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
|
|
263
|
-
* for short prompts, so don't tighten this further. We don't validate
|
|
264
|
-
* content; HTTP 200 means reachable.
|
|
265
|
-
*/
|
|
266
|
-
declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
267
|
-
timeoutMs?: number;
|
|
268
|
-
}): Promise<{
|
|
269
|
-
ok: boolean;
|
|
270
|
-
latencyMs: number;
|
|
271
|
-
error: string | null;
|
|
272
|
-
}>;
|
|
273
|
-
/**
|
|
274
|
-
* Stateful client — construct once with defaults, call many times.
|
|
275
|
-
* Thin wrapper around the free functions; exists for callers that want
|
|
276
|
-
* to inject a single configured instance into multiple primitives.
|
|
277
|
-
*/
|
|
278
|
-
declare class LlmClient {
|
|
279
|
-
readonly maximumAttempts: number;
|
|
280
|
-
private readonly opts;
|
|
281
|
-
constructor(opts?: LlmClientOptions);
|
|
282
|
-
call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult>;
|
|
283
|
-
callJson<T = unknown>(req: LlmCallRequest, per?: LlmClientOptions): Promise<{
|
|
284
|
-
value: T;
|
|
285
|
-
result: LlmCallResult;
|
|
286
|
-
}>;
|
|
287
|
-
}
|
|
288
|
-
|
|
289
|
-
export { type LlmCallMetadata as L, type LlmClientOptions as a, type LlmRouteRequirements as b, type LlmCallRequest as c, type LlmCallResult as d, type LlmUsage as e, LlmCallError as f, LlmClient as g, type LlmMessage as h, LlmResponseError as i, LlmRouteAssertionError as j, assertLlmRoute as k, backoffMs as l, callLlm as m, callLlmJson as n, costReceiptFromLlm as o, costReceiptFromLlmError as p, isTransientLlmError as q, maximumChargeForLlmRequest as r, probeLlm as s, stripFencedJson as t };
|
|
@@ -1,150 +0,0 @@
|
|
|
1
|
-
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* Multi-layer verifier — ordered pipeline of verification layers.
|
|
5
|
-
*
|
|
6
|
-
* Different contract from {@link JudgeRunner} (which runs parallel
|
|
7
|
-
* specs against a sandbox). MultiLayerVerifier is a DAG of layers
|
|
8
|
-
* (install → typecheck → build → lint → serve → semantic → …) with
|
|
9
|
-
* dependency-based skip, per-layer findings, soft-fail semantics, and
|
|
10
|
-
* an aggregated `blendedScore` across all passed layers.
|
|
11
|
-
*
|
|
12
|
-
* Use when you want:
|
|
13
|
-
* - ordered stages where a failing upstream stage skips downstream ones
|
|
14
|
-
* - each stage produces rich `findings` (severity + message + evidence)
|
|
15
|
-
* - a single composite score across stages with per-stage weights
|
|
16
|
-
* - soft-fail stages whose failure doesn't abort the pipeline
|
|
17
|
-
*
|
|
18
|
-
* Use {@link JudgeRunner} when you want:
|
|
19
|
-
* - N independent judges running in parallel against the same artifact
|
|
20
|
-
* - no inter-judge dependencies
|
|
21
|
-
* - boolean `passed` per judge + overall
|
|
22
|
-
*
|
|
23
|
-
* Both primitives compose — JudgeRunner can be invoked as a single
|
|
24
|
-
* layer inside a MultiLayerVerifier if that suits the caller.
|
|
25
|
-
*/
|
|
26
|
-
|
|
27
|
-
type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
|
|
28
|
-
type Severity = 'critical' | 'major' | 'minor' | 'info';
|
|
29
|
-
interface Finding {
|
|
30
|
-
severity: Severity;
|
|
31
|
-
message: string;
|
|
32
|
-
evidence?: string;
|
|
33
|
-
/** Optional layer name the finding belongs to (set by the verifier if omitted). */
|
|
34
|
-
layer?: string;
|
|
35
|
-
/**
|
|
36
|
-
* Free-form structured payload — used by `multiToolchainLayer` to attach
|
|
37
|
-
* `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
|
|
38
|
-
* Renderers MAY interrogate; agent-eval primitives never assume shape.
|
|
39
|
-
*/
|
|
40
|
-
detail?: Record<string, unknown>;
|
|
41
|
-
}
|
|
42
|
-
interface LayerResult {
|
|
43
|
-
layer: string;
|
|
44
|
-
status: LayerStatus;
|
|
45
|
-
/** 0..1 score, optional — layers that don't produce a numeric score omit. */
|
|
46
|
-
score?: number;
|
|
47
|
-
durationMs: number;
|
|
48
|
-
findings: Finding[];
|
|
49
|
-
/** Short human-readable summary (one line). */
|
|
50
|
-
reason?: string;
|
|
51
|
-
/**
|
|
52
|
-
* Numeric layer-level diagnostics: error counts, warning counts,
|
|
53
|
-
* cyclomatic complexity, total adapter wall-time, etc. Keyed by
|
|
54
|
-
* diagnostic name; null = "diagnostic not applicable / not measured."
|
|
55
|
-
* Renderers that know the keys can display them; ones that don't,
|
|
56
|
-
* ignore. Free-form on purpose — consumers type the value shape in
|
|
57
|
-
* their own namespace.
|
|
58
|
-
*/
|
|
59
|
-
diagnostics?: Record<string, number | null>;
|
|
60
|
-
/** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
|
|
61
|
-
detail?: Record<string, unknown>;
|
|
62
|
-
}
|
|
63
|
-
interface VerifyContext<Env = unknown> {
|
|
64
|
-
/** Per-run opaque context the caller provides. Layers destructure what they need. */
|
|
65
|
-
env: Env;
|
|
66
|
-
/** Previously-computed results from layers that already ran. */
|
|
67
|
-
prior: Record<string, LayerResult>;
|
|
68
|
-
/** Signal — if aborted, layers MUST bail within reasonable wall. */
|
|
69
|
-
signal: AbortSignal;
|
|
70
|
-
}
|
|
71
|
-
interface Layer<Env = unknown> {
|
|
72
|
-
name: string;
|
|
73
|
-
/** Stages that must have `status: 'pass'` before this layer runs. */
|
|
74
|
-
dependsOn?: string[];
|
|
75
|
-
/**
|
|
76
|
-
* Weight in the composite `blendedScore`. Default 1.0. Layers with weight 0
|
|
77
|
-
* contribute findings but not score.
|
|
78
|
-
*/
|
|
79
|
-
weight?: number;
|
|
80
|
-
/**
|
|
81
|
-
* If true, a `fail` status contributes to `blendedScore` (as 0) instead of
|
|
82
|
-
* being dropped — use for layers whose failure is a real signal. Default:
|
|
83
|
-
* fail drops from numerator + denominator, matching VB's existing semantics.
|
|
84
|
-
*/
|
|
85
|
-
failContributesToScore?: boolean;
|
|
86
|
-
/** Optional per-layer wall-cap in ms. Honored by the verifier (AbortSignal). */
|
|
87
|
-
capMs?: number;
|
|
88
|
-
run: (ctx: VerifyContext<Env>) => Promise<LayerResult> | LayerResult;
|
|
89
|
-
}
|
|
90
|
-
interface VerifyOptions<Env = unknown> {
|
|
91
|
-
env: Env;
|
|
92
|
-
/**
|
|
93
|
-
* Overall wall cap. Default: sum of layer capMs, or Infinity if any layer
|
|
94
|
-
* omits a cap. The verifier short-circuits remaining layers on overall cap.
|
|
95
|
-
*/
|
|
96
|
-
overallCapMs?: number;
|
|
97
|
-
/** Called with each layer result as it completes. */
|
|
98
|
-
onLayer?: (result: LayerResult) => void;
|
|
99
|
-
}
|
|
100
|
-
/** Extends the substrate verdict spine: `valid` = `allPass` and `score` =
|
|
101
|
-
* `blendedScore` — derived where the report is aggregated, so spine
|
|
102
|
-
* consumers (drivers, gates) read this report without an adapter. */
|
|
103
|
-
interface VerificationReport extends DefaultVerdict {
|
|
104
|
-
layers: LayerResult[];
|
|
105
|
-
passCount: number;
|
|
106
|
-
failCount: number;
|
|
107
|
-
skippedCount: number;
|
|
108
|
-
errorCount: number;
|
|
109
|
-
/** True iff at least one scored layer ran AND every scored layer passed. */
|
|
110
|
-
allPass: boolean;
|
|
111
|
-
/**
|
|
112
|
-
* Weighted mean of `score` across contributing layers. 0 when no layers
|
|
113
|
-
* contributed. See {@link Layer.failContributesToScore} for fail semantics.
|
|
114
|
-
*/
|
|
115
|
-
blendedScore: number;
|
|
116
|
-
durationMs: number;
|
|
117
|
-
startedAt: string;
|
|
118
|
-
finishedAt: string;
|
|
119
|
-
}
|
|
120
|
-
/**
|
|
121
|
-
* Grade a semantic-concept-style judge result into a single layer status.
|
|
122
|
-
*
|
|
123
|
-
* Pass when overall score >= threshold AND no critical-severity concept gap.
|
|
124
|
-
* Fail otherwise. Use inside a `Layer.run` when wrapping a concept judge.
|
|
125
|
-
*
|
|
126
|
-
* Generalized from VerticalBench H3 fix: `failingConcepts.length === 0` was
|
|
127
|
-
* too strict — a single concept at 6/10 failed the entire layer despite
|
|
128
|
-
* overall score being >= 0.7. Now we trust the judge's own `severity` field:
|
|
129
|
-
* `critical` findings veto; `major`/`minor` reduce the score but don't veto.
|
|
130
|
-
*/
|
|
131
|
-
declare function gradeSemanticStatus(input: {
|
|
132
|
-
score: number;
|
|
133
|
-
findings: Array<{
|
|
134
|
-
severity: Severity;
|
|
135
|
-
present?: boolean;
|
|
136
|
-
score?: number;
|
|
137
|
-
}>;
|
|
138
|
-
available: boolean;
|
|
139
|
-
threshold?: number;
|
|
140
|
-
}): LayerStatus;
|
|
141
|
-
/**
|
|
142
|
-
* Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
|
|
143
|
-
*/
|
|
144
|
-
declare class MultiLayerVerifier<Env = unknown> {
|
|
145
|
-
private readonly layers;
|
|
146
|
-
constructor(layers: Layer<Env>[]);
|
|
147
|
-
run(opts: VerifyOptions<Env>): Promise<VerificationReport>;
|
|
148
|
-
}
|
|
149
|
-
|
|
150
|
-
export { type Finding as F, type Layer as L, MultiLayerVerifier as M, type Severity as S, type VerifyOptions as V, type VerificationReport as a, type LayerResult as b, type VerifyContext as c, type LayerStatus as d, gradeSemanticStatus as g };
|