@tangle-network/agent-eval 0.144.4 → 0.144.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +366 -19
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-CYtcIF2V.js → benchmark-B181aMF9.js} +2 -2
- package/dist/{benchmark-CYtcIF2V.js.map → benchmark-B181aMF9.js.map} +1 -1
- package/dist/{benchmark-CP6kWfj8.d.ts → benchmark-J9Qe6j2_.d.ts} +3 -3
- package/dist/{benchmark-CP6kWfj8.d.ts.map → benchmark-J9Qe6j2_.d.ts.map} +1 -1
- package/dist/{benchmark-command-9FTgq6Fg.js → benchmark-command-CQd78YHt.js} +922 -36
- package/dist/benchmark-command-CQd78YHt.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CkG1bWFa.js → benchmarks-BEOkuvIg.js} +3 -3
- package/dist/{benchmarks-CkG1bWFa.js.map → benchmarks-BEOkuvIg.js.map} +1 -1
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-DjGFyPxH.js → campaign-CXsdyym7.js} +4 -4
- package/dist/{campaign-DjGFyPxH.js.map → campaign-CXsdyym7.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-Bbht4xxl.d.ts → client-0JI64ovJ.d.ts} +4 -4
- package/dist/{client-Bbht4xxl.d.ts.map → client-0JI64ovJ.d.ts.map} +1 -1
- package/dist/{completion-verifier-EJERfFwF.d.ts → completion-verifier-CBiee74w.d.ts} +5 -5
- package/dist/{completion-verifier-EJERfFwF.d.ts.map → completion-verifier-CBiee74w.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +5 -5
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-SOyHB6qG.js → default-registry-Dta70shL.js} +4 -4
- package/dist/{default-registry-SOyHB6qG.js.map → default-registry-Dta70shL.js.map} +1 -1
- package/dist/{default-registry-iXfu2trt.d.ts → default-registry-J9m-_tya.d.ts} +5 -5
- package/dist/{default-registry-iXfu2trt.d.ts.map → default-registry-J9m-_tya.d.ts.map} +1 -1
- package/dist/{dspy-rlm-engine-IRCG8kdi.js → dspy-rlm-engine-19FQEMBK.js} +2 -2
- package/dist/{dspy-rlm-engine-IRCG8kdi.js.map → dspy-rlm-engine-19FQEMBK.js.map} +1 -1
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-CfLQQs9B.js} +2 -2
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-CfLQQs9B.js.map} +1 -1
- package/dist/{exact-types-BygCBR4L.d.ts → exact-types-CBYF5MGd.d.ts} +2 -2
- package/dist/{exact-types-BygCBR4L.d.ts.map → exact-types-CBYF5MGd.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-Q9c0L3eY.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-Q9c0L3eY.d.ts.map} +1 -1
- package/dist/{extract-usage-7l1Xq5ti.js → extract-usage-BW27f3XW.js} +2 -2
- package/dist/{extract-usage-7l1Xq5ti.js.map → extract-usage-BW27f3XW.js.map} +1 -1
- package/dist/{feedback-trajectory-CSIkRLQX.d.ts → feedback-trajectory-GgoS0-MK.d.ts} +4 -4
- package/dist/{feedback-trajectory-CSIkRLQX.d.ts.map → feedback-trajectory-GgoS0-MK.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-D_qTihaQ.d.ts → index-4XwggC10.d.ts} +5 -5
- package/dist/{index-D_qTihaQ.d.ts.map → index-4XwggC10.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-B6-B0zTB.d.ts} +2 -2
- package/dist/{index-CNOCxBLh.d.ts.map → index-B6-B0zTB.d.ts.map} +1 -1
- package/dist/{index-BuHs_OnD.d.ts → index-BIL5vxxt.d.ts} +12 -12
- package/dist/{index-BuHs_OnD.d.ts.map → index-BIL5vxxt.d.ts.map} +1 -1
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-Dx1kF3Ez.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-Dx1kF3Ez.d.ts.map} +1 -1
- package/dist/index.d.ts +68 -77
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +62 -126
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-DqEsugpr.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-DqEsugpr.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-CNGUaGBY.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-CNGUaGBY.d.ts.map} +1 -1
- package/dist/{kind-factory-Bvwe3pup.js → kind-factory-BHIgPmzS.js} +2 -2
- package/dist/{kind-factory-Bvwe3pup.js.map → kind-factory-BHIgPmzS.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-Dv5BiKLE.js} +289 -7
- package/dist/llm-client-Dv5BiKLE.js.map +1 -0
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/multishot/index.js +1 -1
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-ChOgpIoQ.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-ChOgpIoQ.d.ts.map} +1 -1
- package/dist/{replay-CqOsGjzU.js → replay-GW61ezMW.js} +4 -4
- package/dist/{replay-CqOsGjzU.js.map → replay-GW61ezMW.js.map} +1 -1
- package/dist/{replay-DQ-55DC_.d.ts → replay-Krvb114g.d.ts} +7 -7
- package/dist/{replay-DQ-55DC_.d.ts.map → replay-Krvb114g.d.ts.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-xLeNcpKX.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-xLeNcpKX.d.ts.map} +1 -1
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-RZgnGWlx.d.ts} +2 -2
- package/dist/{reward-hacking-DFgkEY4p.d.ts.map → reward-hacking-RZgnGWlx.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-H1vRpIdT.d.ts → run-evidence-C6G41MSI.d.ts} +3 -3
- package/dist/{run-evidence-H1vRpIdT.d.ts.map → run-evidence-C6G41MSI.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-Do5aM9wP.js → semantic-concept-judge-DwF6n05O.js} +3 -3
- package/dist/{semantic-concept-judge-Do5aM9wP.js.map → semantic-concept-judge-DwF6n05O.js.map} +1 -1
- package/dist/{server-Df00sdwz.js → server-D6XJQHw7.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-D6XJQHw7.js.map} +1 -1
- package/dist/{skill-usage-DtpLou9L.d.ts → skill-usage-GlOphAhX.d.ts} +9 -9
- package/dist/{skill-usage-DtpLou9L.d.ts.map → skill-usage-GlOphAhX.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts → skillopt-optimization-method-7S43rbDB.d.ts} +12 -12
- package/dist/{skillopt-optimization-method-CvSJGdm3.d.ts.map → skillopt-optimization-method-7S43rbDB.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-Bfb-vBKe.js} +2 -2
- package/dist/{skillopt-optimization-method-C4FX42dy.js.map → skillopt-optimization-method-Bfb-vBKe.js.map} +1 -1
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-C-dm-J6H.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-C-dm-J6H.d.ts.map} +1 -1
- package/dist/{store-otlp-D4I90_vR.js → store-otlp-CKtTpRhv.js} +2 -2
- package/dist/{store-otlp-D4I90_vR.js.map → store-otlp-CKtTpRhv.js.map} +1 -1
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-B0cAyA7N.d.ts} +3 -3
- package/dist/{summary-report-BOM6dfP7.d.ts.map → summary-report-B0cAyA7N.d.ts.map} +1 -1
- package/dist/{tool-groups-CMmsgTzj.d.ts → tool-groups-CK0JCkqO.d.ts} +9 -9
- package/dist/{tool-groups-CMmsgTzj.d.ts.map → tool-groups-CK0JCkqO.d.ts.map} +1 -1
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +4 -4
- package/dist/{types-y8jrxXWd.d.ts → types-BhP9q0Fq.d.ts} +22 -4
- package/dist/{types-y8jrxXWd.d.ts.map → types-BhP9q0Fq.d.ts.map} +1 -1
- package/dist/{types-DcJxgsLy.d.ts → types-DOZyvsFU.d.ts} +5 -5
- package/dist/{types-DcJxgsLy.d.ts.map → types-DOZyvsFU.d.ts.map} +1 -1
- package/dist/{types-BjMFz88h.d.ts → types-XMVEdrE_.d.ts} +207 -6
- package/dist/types-XMVEdrE_.d.ts.map +1 -0
- package/dist/{usage-receipt-CgxMEBZq.js → usage-receipt-EVI8B8Xu.js} +8 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/prime-analyst.md +120 -0
- package/docs/trace-analysis.md +1 -1
- package/package.json +3 -3
- package/dist/benchmark-command-9FTgq6Fg.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/usage-receipt-CgxMEBZq.js.map +0 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { r as CaptureIntegrityError, t as AgentEvalError } from "./errors-
|
|
2
|
-
import { b as CustomTokenPricing, c as CostLedgerHandle, f as CostLedgerSummary, g as CostReceiptInput, x as MaximumCharge } from "./cost-ledger-
|
|
1
|
+
import { r as CaptureIntegrityError, t as AgentEvalError } from "./errors-CKPfb2aH.js";
|
|
2
|
+
import { b as CustomTokenPricing, c as CostLedgerHandle, f as CostLedgerSummary, g as CostReceiptInput, x as MaximumCharge } from "./cost-ledger-Bv_e8XHY.js";
|
|
3
3
|
//#region src/trace/raw-provider-sink.d.ts
|
|
4
4
|
/**
|
|
5
5
|
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
@@ -132,6 +132,169 @@ declare class FileSystemRawProviderSink implements RawProviderSink {
|
|
|
132
132
|
*/
|
|
133
133
|
declare function providerFromBaseUrl(baseUrl: string): string;
|
|
134
134
|
//#endregion
|
|
135
|
+
//#region src/judge-families.d.ts
|
|
136
|
+
/**
|
|
137
|
+
* Judge model-family classification + cross-family enforcement.
|
|
138
|
+
*
|
|
139
|
+
* A judge ensemble built entirely from one provider family shares that
|
|
140
|
+
* family's blind spots and self-preference — its "agreement" is correlated
|
|
141
|
+
* bias, not independent signal. `assertCrossFamily` makes the consumer prove
|
|
142
|
+
* the ensemble spans ≥2 families; `judgeFamily` is the single regex map that
|
|
143
|
+
* replaces the per-consumer copies (tax/legal/creative/gtm each ship one).
|
|
144
|
+
*/
|
|
145
|
+
/** Provider family a model belongs to. `unknown` when no rule matches. */
|
|
146
|
+
type JudgeFamily = 'anthropic' | 'openai' | 'google' | 'meta' | 'mistral' | 'deepseek' | 'xai' | 'qwen' | 'cohere' | 'amazon' | 'moonshot' | 'zhipu' | 'unknown';
|
|
147
|
+
/**
|
|
148
|
+
* Classify a model id into its provider family. Strips a `@snapshot` suffix
|
|
149
|
+
* and prefers an explicit `provider/...` prefix; otherwise matches the model
|
|
150
|
+
* name. Returns `unknown` when nothing matches (callers decide whether that's
|
|
151
|
+
* acceptable — `assertCrossFamily` counts it as its own family).
|
|
152
|
+
*/
|
|
153
|
+
declare function judgeFamily(modelId: string): JudgeFamily;
|
|
154
|
+
interface AssertCrossFamilyOptions {
|
|
155
|
+
/** Minimum number of distinct families the ensemble must span. Default 2. */
|
|
156
|
+
minFamilies?: number;
|
|
157
|
+
/** When false (default), `unknown`-family models do NOT count toward the
|
|
158
|
+
* family total — an ensemble of all-unclassifiable models is not provably
|
|
159
|
+
* cross-family. Set true to count `unknown` as one shared family. */
|
|
160
|
+
allowUnknown?: boolean;
|
|
161
|
+
}
|
|
162
|
+
declare class CrossFamilyError extends Error {
|
|
163
|
+
readonly families: JudgeFamily[];
|
|
164
|
+
readonly models: string[];
|
|
165
|
+
constructor(message: string, families: JudgeFamily[], models: string[]);
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* Throw unless the judge models span at least `minFamilies` distinct provider
|
|
169
|
+
* families. Pass the model ids backing your judge ensemble. Fail-loud by
|
|
170
|
+
* design — a correlated single-family ensemble silently inflates agreement.
|
|
171
|
+
*
|
|
172
|
+
* Scope: this reads the ids you REQUEST. It proves the panel was configured
|
|
173
|
+
* across families; it cannot prove the panel RAN across families, because a
|
|
174
|
+
* routing gateway may answer several different ids from one provider. Where
|
|
175
|
+
* the diversity claim is load-bearing (a published leaderboard, a
|
|
176
|
+
* certification, a non-self-judging exclusion), assert on the ids the
|
|
177
|
+
* provider echoed instead: `assertCrossFamilyServed` in
|
|
178
|
+
* ./integrity/served-model.
|
|
179
|
+
*/
|
|
180
|
+
declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
|
|
181
|
+
//#endregion
|
|
182
|
+
//#region src/integrity/served-model.d.ts
|
|
183
|
+
/** How a served id relates to the id that was requested. */
|
|
184
|
+
type ServedModelVerdict =
|
|
185
|
+
/** Byte-identical after normalisation — the requested model answered. */
|
|
186
|
+
'exact' |
|
|
187
|
+
/** Same model, different spelling (provider prefix, snapshot, tier suffix). */
|
|
188
|
+
'alias' |
|
|
189
|
+
/** A different model of the SAME provider family answered. */
|
|
190
|
+
'substituted-within-family' |
|
|
191
|
+
/** A different provider's model answered. */
|
|
192
|
+
'substituted-cross-family' |
|
|
193
|
+
/** The response carried no model id — identity is unproven either way. */
|
|
194
|
+
'unreported';
|
|
195
|
+
interface ServedModelCheck {
|
|
196
|
+
/** The id the caller asked for. */
|
|
197
|
+
requested: string;
|
|
198
|
+
/** The id echoed on the response; `null` when the response omitted it. */
|
|
199
|
+
served: string | null;
|
|
200
|
+
requestedFamily: JudgeFamily;
|
|
201
|
+
/** `null` when `served` is null. */
|
|
202
|
+
servedFamily: JudgeFamily | null;
|
|
203
|
+
verdict: ServedModelVerdict;
|
|
204
|
+
/** True for every verdict except `exact` and `alias`. */
|
|
205
|
+
substituted: boolean;
|
|
206
|
+
}
|
|
207
|
+
/**
|
|
208
|
+
* Reduce a model id to its comparable core: lowercase, no surrounding space,
|
|
209
|
+
* no `provider/` prefix, no `@snapshot` / `:batch` / `:free` tier suffix, no
|
|
210
|
+
* trailing build date, and `.`/`_` folded to `-` so one version is spelled one
|
|
211
|
+
* way.
|
|
212
|
+
*
|
|
213
|
+
* Dropping the build date is what makes snapshot resolution legible as the
|
|
214
|
+
* non-event it is: a router answering `gpt-4o-mini` with
|
|
215
|
+
* `gpt-4o-mini-2024-07-18` pinned a floating alias to a reproducible build —
|
|
216
|
+
* the same model, which is the behaviour we want. Only routing decoration is
|
|
217
|
+
* stripped; version digits are load-bearing, so `deepseek-v3.2` and
|
|
218
|
+
* `deepseek-v4-flash` stay distinct, and comparison is EXACT equality rather
|
|
219
|
+
* than a prefix test (a prefix rule would accept `gpt-5` → `gpt-5-mini`, a
|
|
220
|
+
* silent downgrade wearing the right vendor name).
|
|
221
|
+
*/
|
|
222
|
+
declare function normalizeModelId(modelId: string): string;
|
|
223
|
+
/**
|
|
224
|
+
* Classify one requested/served pair. Pure — no I/O — so it is safe inside
|
|
225
|
+
* response handlers, reducers, and CI gates.
|
|
226
|
+
*
|
|
227
|
+
* `served` is the id echoed by the provider (OpenAI-compatible bodies put it
|
|
228
|
+
* at `model`). `null`/`undefined` means the body omitted it; that is
|
|
229
|
+
* `unreported`, NOT a pass — a provider that does not name what answered has
|
|
230
|
+
* not proven identity, and a transport that drops the field must not read as
|
|
231
|
+
* agreement.
|
|
232
|
+
*/
|
|
233
|
+
declare function checkServedModel(requested: string, served: string | null | undefined): ServedModelCheck;
|
|
234
|
+
declare class ModelSubstitutionError extends AgentEvalError {
|
|
235
|
+
readonly checks: ReadonlyArray<ServedModelCheck>;
|
|
236
|
+
constructor(message: string, checks: ReadonlyArray<ServedModelCheck>);
|
|
237
|
+
}
|
|
238
|
+
interface AssertServedModelOptions {
|
|
239
|
+
/**
|
|
240
|
+
* Accept a different model of the same provider family (e.g. requested
|
|
241
|
+
* `deepseek-v3.2`, served `deepseek-v4-flash`). Default false. Setting this
|
|
242
|
+
* keeps family-level claims valid and forfeits per-model claims.
|
|
243
|
+
*/
|
|
244
|
+
allowWithinFamily?: boolean;
|
|
245
|
+
/**
|
|
246
|
+
* Accept a response that carried no model id. Default false — an
|
|
247
|
+
* unidentified response cannot support a per-model or per-family claim.
|
|
248
|
+
*/
|
|
249
|
+
allowUnreported?: boolean;
|
|
250
|
+
/** Prefixed to the thrown message, e.g. the judge or campaign cell name. */
|
|
251
|
+
context?: string;
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* The one place the accept/reject policy lives, so a caller that reports
|
|
255
|
+
* substitution (a preflight table, a run record) and a caller that throws on it
|
|
256
|
+
* can never drift apart. A cross-family substitution is never acceptable.
|
|
257
|
+
*/
|
|
258
|
+
declare function servedModelAcceptable(check: ServedModelCheck, opts?: AssertServedModelOptions): boolean;
|
|
259
|
+
/**
|
|
260
|
+
* Throw `ModelSubstitutionError` unless the served id is the requested model.
|
|
261
|
+
* Returns the check on success so callers can record the served id alongside
|
|
262
|
+
* the result.
|
|
263
|
+
*/
|
|
264
|
+
declare function assertServedModel(requested: string, served: string | null | undefined, opts?: AssertServedModelOptions): ServedModelCheck;
|
|
265
|
+
/**
|
|
266
|
+
* Batch form: check every pair and throw naming EVERY substitution, so one
|
|
267
|
+
* failure does not hide the rest. Returns all checks on success.
|
|
268
|
+
*/
|
|
269
|
+
declare function assertServedModels(pairs: ReadonlyArray<{
|
|
270
|
+
requested: string;
|
|
271
|
+
served: string | null | undefined;
|
|
272
|
+
}>, opts?: AssertServedModelOptions): ServedModelCheck[];
|
|
273
|
+
interface AssertCrossFamilyServedOptions extends AssertServedModelOptions {
|
|
274
|
+
/** Minimum distinct SERVED families required. Default 2. */
|
|
275
|
+
minFamilies?: number;
|
|
276
|
+
/** Count `unknown`-family served ids toward the total. Default false. */
|
|
277
|
+
allowUnknown?: boolean;
|
|
278
|
+
}
|
|
279
|
+
declare class ServedCrossFamilyError extends AgentEvalError {
|
|
280
|
+
readonly families: JudgeFamily[];
|
|
281
|
+
readonly checks: ReadonlyArray<ServedModelCheck>;
|
|
282
|
+
constructor(message: string, families: JudgeFamily[], checks: ReadonlyArray<ServedModelCheck>);
|
|
283
|
+
}
|
|
284
|
+
/**
|
|
285
|
+
* Family-diversity rule over the models that actually ANSWERED.
|
|
286
|
+
*
|
|
287
|
+
* `assertCrossFamily` (../judge-families) reads the requested ids and so
|
|
288
|
+
* cannot see a gateway that answers three "different" requests from one
|
|
289
|
+
* provider. This one asserts no substitution first, then counts families from
|
|
290
|
+
* the served ids — a panel that collapsed to one family under the hood fails
|
|
291
|
+
* here even though its request list looked diverse.
|
|
292
|
+
*/
|
|
293
|
+
declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
|
|
294
|
+
requested: string;
|
|
295
|
+
served: string | null | undefined;
|
|
296
|
+
}>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
|
|
297
|
+
//#endregion
|
|
135
298
|
//#region src/llm-client.d.ts
|
|
136
299
|
interface LlmMessage {
|
|
137
300
|
role: 'system' | 'user' | 'assistant';
|
|
@@ -193,8 +356,27 @@ interface LlmCallResult {
|
|
|
193
356
|
* caller-supplied token pricing. `null` when neither is available.
|
|
194
357
|
*/
|
|
195
358
|
costUsd: number | null;
|
|
196
|
-
/**
|
|
359
|
+
/**
|
|
360
|
+
* Model id used for attribution (cost, pricing, log lines). The response's
|
|
361
|
+
* echoed id when the provider sent one, else the requested id.
|
|
362
|
+
*
|
|
363
|
+
* NOT evidence of which model answered — read `servedModel` for that. A
|
|
364
|
+
* provider that omits `model` makes this equal to the request, which is
|
|
365
|
+
* exactly the case an identity check must be able to distinguish.
|
|
366
|
+
*/
|
|
197
367
|
model: string;
|
|
368
|
+
/**
|
|
369
|
+
* The model id the provider echoed on the response, verbatim; `null` when
|
|
370
|
+
* the body carried none. This is the only field that can witness a gateway
|
|
371
|
+
* substituting a different model for the one requested — compare it with
|
|
372
|
+
* `assertServedModel` / `checkServedModel` (src/integrity/served-model.ts).
|
|
373
|
+
*
|
|
374
|
+
* Optional so hand-built results (mock/custom transports) still typecheck,
|
|
375
|
+
* but omitting it is not a pass: the identity checks read `undefined` as
|
|
376
|
+
* `unreported` and reject it by default. A transport that knows which model
|
|
377
|
+
* answered should say so.
|
|
378
|
+
*/
|
|
379
|
+
servedModel?: string | null;
|
|
198
380
|
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
199
381
|
durationMs: number;
|
|
200
382
|
/**
|
|
@@ -307,6 +489,16 @@ interface LlmClientOptions {
|
|
|
307
489
|
};
|
|
308
490
|
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
309
491
|
redactor?: ProviderRedactor;
|
|
492
|
+
/**
|
|
493
|
+
* Reject a response whose echoed model is not the model that was requested.
|
|
494
|
+
* A routing gateway can accept one id and answer from another, which
|
|
495
|
+
* silently invalidates every per-model and per-family claim downstream.
|
|
496
|
+
* `true` uses the strict default (aliases pass, substitutions and
|
|
497
|
+
* unidentified responses throw `ModelSubstitutionError`); pass an options
|
|
498
|
+
* object to relax a specific case. Off by default — turning it on for a
|
|
499
|
+
* measurement run is the point.
|
|
500
|
+
*/
|
|
501
|
+
assertServedModel?: boolean | AssertServedModelOptions;
|
|
310
502
|
}
|
|
311
503
|
/**
|
|
312
504
|
* True when an error is a transient transport/network fault worth retrying,
|
|
@@ -392,7 +584,12 @@ declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequiremen
|
|
|
392
584
|
* Sends a tiny `ping` message with `maxTokens=64`. Reasoning models
|
|
393
585
|
* (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning
|
|
394
586
|
* for short prompts, so don't tighten this further. We don't validate
|
|
395
|
-
* content
|
|
587
|
+
* content.
|
|
588
|
+
*
|
|
589
|
+
* Reachability and identity are separate answers: `ok` means the route
|
|
590
|
+
* answered, `servedModel` / `substituted` say WHICH model answered. A gateway
|
|
591
|
+
* that serves another provider's model returns `ok: true` with
|
|
592
|
+
* `substituted: true` — inspect both before treating the id as measured.
|
|
396
593
|
*/
|
|
397
594
|
declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
398
595
|
timeoutMs?: number;
|
|
@@ -400,6 +597,10 @@ declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
|
400
597
|
ok: boolean;
|
|
401
598
|
latencyMs: number;
|
|
402
599
|
error: string | null;
|
|
600
|
+
/** Id echoed by the provider; `null` when it sent none or the probe failed. */
|
|
601
|
+
servedModel: string | null;
|
|
602
|
+
/** True when the echoed id is a different model than `model` (or absent). */
|
|
603
|
+
substituted: boolean;
|
|
403
604
|
}>;
|
|
404
605
|
/**
|
|
405
606
|
* Stateful client — construct once with defaults, call many times.
|
|
@@ -800,5 +1001,5 @@ interface EvalResult {
|
|
|
800
1001
|
artifact?: string;
|
|
801
1002
|
}
|
|
802
1003
|
//#endregion
|
|
803
|
-
export { assertLlmRoute as $, ChatClient as A, SandboxSdkTransportOpts as B, ScenarioFile as C, TurnMetrics as D, Turn as E, CreateChatClientOpts as F, LlmCallResult as G, LlmCallError as H, CustomTransportOpts as I, LlmMessage as J, LlmClient as K, DirectProviderTransportOpts as L, ChatResponse as M, ChatTransport as N, TurnResult as O, CliBridgeTransportOpts as P, LlmUsage as Q, MockTransportOpts as R, Scenario as S, TestResult as T, LlmCallMetadata as U, createChatClient as V, LlmCallRequest as W, LlmRouteAssertionError as X, LlmResponseError as Y, LlmRouteRequirements as Z, PersonaConfig as _,
|
|
804
|
-
//# sourceMappingURL=types-
|
|
1004
|
+
export { assertLlmRoute as $, ChatClient as A, NoopRawProviderSink as At, SandboxSdkTransportOpts as B, ScenarioFile as C, JudgeFamily as Ct, TurnMetrics as D, FileSystemRawProviderSinkOptions as Dt, Turn as E, FileSystemRawProviderSink as Et, CreateChatClientOpts as F, RawProviderSinkFilter as Ft, LlmCallResult as G, LlmCallError as H, CustomTransportOpts as I, defaultProviderRedactor as It, LlmMessage as J, LlmClient as K, DirectProviderTransportOpts as L, providerFromBaseUrl as Lt, ChatResponse as M, RawProviderDirection as Mt, ChatTransport as N, RawProviderEvent as Nt, TurnResult as O, InMemoryRawProviderSink as Ot, CliBridgeTransportOpts as P, RawProviderSink as Pt, LlmUsage as Q, MockTransportOpts as R, Scenario as S, CrossFamilyError as St, TestResult as T, judgeFamily as Tt, LlmCallMetadata as U, createChatClient as V, LlmCallRequest as W, LlmRouteAssertionError as X, LlmResponseError as Y, LlmRouteRequirements as Z, PersonaConfig as _, assertServedModels as _t, CheckResult as a, isTransientLlmError as at, RouteMap as b, servedModelAcceptable as bt, DriverResult as c, stripFencedJson as ct, FeedbackPattern as d, ModelSubstitutionError as dt, backoffMs as et, JudgeConfig as f, ServedCrossFamilyError as ft, JudgeScore as g, assertServedModel as gt, JudgeRubric as h, assertCrossFamilyServed as ht, BenchmarkRunnerConfig as i, costReceiptFromLlmError as it, ChatRequest as j, ProviderRedactor as jt, ChatCallOpts as k, InMemoryRawProviderSinkOptions as kt, DriverState as l, AssertCrossFamilyServedOptions as lt, JudgeInput as m, ServedModelVerdict as mt, ArtifactResult as n, callLlmJson as nt, CollectedArtifacts as o, maximumChargeForLlmRequest as ot, JudgeFn as p, ServedModelCheck as pt, LlmClientOptions as q, BenchmarkReport as r, costReceiptFromLlm as rt, CompletionCriterion as s, probeLlm as st, ArtifactCheck as t, callLlm as tt, EvalResult as u, AssertServedModelOptions as ut, PersonaRigor as v, checkServedModel as vt, ScenarioResult as w, assertCrossFamily as wt, RubricDimension as x, AssertCrossFamilyOptions as xt, ProductClientConfig as y, normalizeModelId as yt, RouterTransportOpts as z };
|
|
1005
|
+
//# sourceMappingURL=types-XMVEdrE_.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types-XMVEdrE_.d.ts","names":[],"sources":["../src/trace/raw-provider-sink.ts","../src/judge-families.ts","../src/integrity/served-model.ts","../src/llm-client.ts","../src/analyst/chat-client.ts","../src/types.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;KA8BY;UAEK;;EAEf;;EAEA;EACA;;;;;;EAMA;EACA;;EAEA;;EAEA;;EAEA;EACA,WAAW;;EAEX;;EAEA;EACA;EACA,iBAAiB;EACjB;EACA,kBAAkB;EAClB;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA,YAAY;EACZ;;UAGe;EACf,OAAO,OAAO,mBAAmB;;EAEjC,MAAM,SAAS,wBAAwB,QAAQ;;EAE/C,UAAU;;KAGA,oBAAoB,OAAO,qBAAqB;;;;;;iBAmB5C,wBAAwB,OAAO,mBAAmB;UA8CjD;EACf,WAAW;;cAGA,mCAAmC;UACtC;UACA;EAER,YAAY,OAAM;EAIZ,OAAO,OAAO,mBAAmB;EAIjC,KAAK,SAAQ,wBAA6B,QAAQ;EAUxD;;cAKW,+BAA+B;EACpC,UAAU;;;;;;;EASV,QAAQ,QAAQ;;UAOP;;EAEf;;EAEA;;EAEA;EACA,WAAW;;cAGA,qCAAqC;UACxC;UACA;UACA;UACA;UACA;UACA;UACA;EAER,YAAY,MAAM;UAOJ;UAON;EAKF,OAAO,OAAO,mBAAmB;EAYjC,KAAK,SAAQ,wBAA6B,QAAQ;;;;;;iBAkC1C,oBAAoB;;;;;;;;;;;;;KC5QxB;;;;;;;iBAkEI,YAAY,kBAAkB;UAc7B;;EAEf;;;;EAIA;;cAGW,yBAAyB;WAGlB,UAAU;WACV;EAHlB,YACE,iBACgB,UAAU,eACV;;;;;;;;;;;;;;;iBAoBJ,kBACd,kBACA,OAAM,2BACL;;;;KCrGS;;;;;;;;;;;UAYK;;EAEf;;EAEA;EACA,iBAAiB;;EAEjB,cAAc;EACd,SAAS;;EAET;;;;;;;;;;;;;;;;;iBAkBc,iBAAiB;;;;;;;;;;;iBA0BjB,iBACd,mBACA,oCACC;cA4CU,+BAA+B;WAGxB,QAAQ,cAAc;EAFxC,YACE,iBACgB,QAAQ,cAAc;;UAOzB;;;;;;EAMf;;;;;EAKA;;EAEA;;;;;;;iBAQc,sBACd,OAAO,kBACP,OAAM;;;;;;iBA6BQ,kBACd,mBACA,mCACA,OAAM,2BACL;;;;;iBAea,mBACd,OAAO;EAAgB;EAAmB;IAC1C,OAAM,2BACL;UAec,uCAAuC;;EAEtD;;EAEA;;cAGW,+BAA+B;WAGxB,UAAU;WACV,QAAQ,cAAc;EAHxC,YACE,iBACgB,UAAU,eACV,QAAQ,cAAc;;;;;;;;;;;iBAgB1B,wBACd,OAAO;EAAgB;EAAmB;IAC1C,OAAM,iCACL;;;UChOc;EACf;;;;;EAKA,kBAEI;IACM;IAAc;;IACd;IAAmB;MAAa;MAAa;;;;KAI7C;UAEK;EACf;EACA,UAAU;;EAEV;;EAEA;IAAe;IAAc,QAAQ;;EACrC;EACA;;EAEA,WAAW;;EAEX;;;;;;iBAOc,2BACd,SAAS,KAAK,iFACd,UAAS,mBACR;UAgCc;EACf;EACA;EACA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;EACA,OAAO;;;;;EAKP;;;;;;;;;EASA;;;;;;;;;;;;EAYA;;EAEA;;;;;;;;;EASA;;;;;;;EAOA;;EAEA,KAAK;;KAGK,kBAAkB,KAAK;;iBAGnB,mBACd,QAAQ,eACR,qBAAqB,qBACpB;;iBA2Ba,wBACd,OAAO,OACP,qBAAqB,qBACpB;cAMU,qBAAqB;WAGd;WACA;WACA;EAJlB,YACE,iBACgB,gBACA,cACA;;;;;cASP,yBAAyB;WAGlB,QAAQ;EAF1B,YACE,iBACgB,QAAQ,eACxB;IAAY;;;UAMC;;EAEf;;EAEA;EACA;;EAEA;IAAe;IAAc;;;EAE7B;;EAEA;;;;;;;EAOA,SAAS;;;;;;;;EAQT;;EAEA;;EAEA,qBAAqB;;;;;;;EAOrB;;;;;;EAMA;;EAEA,WAAW;;EAEX,eAAe;;;;;;;;EAQf,UAAU;;;;;EAKV;;EAEA;IAAiB;IAAgB;;;EAEjC,WAAW;;;;;;;;;;EAUX,8BAA8B;;;;;;;;;;;;;iBAsEhB,oBAAoB;;iBAgCpB,UAAU;;;;;;iBA6GV,gBAAgB;;;;;;iBA6EV,QACpB,KAAK,gBACL,OAAM,mBACL,QAAQ;;;;;;;iBAoVW,YAAY,aAChC,KAAK,gBACL,OAAM,mBACL;EAAU,OAAO;EAAG,QAAQ;;KA4DnB;cAOC,+BAA+B;WAGxB,QAAQ;WACR;EAHlB,YACE,iBACgB,QAAQ,yBACR;;UAMH;;;;;;;EAOf;;;;;EAKA,kBAAkB,eAAe;;EAEjC,kBAAkB,eAAe;;EAEjC;;;;;EAKA;;;;;;;;;;;iBAYc,eAAe,MAAM,kBAAkB,MAAK;;;;;;;;;;;;;;;;;iBA4EtC,SACpB,eACA,OAAM;EAAqB;IAC1B;EACD;EACA;EACA;;EAEA;;EAEA;;;;;;;cAoCW;WACF;mBACQ;EAEjB,YAAY,OAAM;EAKlB,KAAK,KAAK,gBAAgB,MAAM,mBAAmB,QAAQ;EAK3D,SAAS,aACP,KAAK,gBACL,MAAM,mBACL;IAAU,OAAO;IAAG,QAAQ;;;;;;;;UC/pChB;;WAEN,WAAW;;WAEX;;WAEA;;EAGT,KAAK,KAAK,aAAa,OAAO,eAAe,QAAQ;;KAG3C;UAQK,oBAAoB,KAAK;;EAExC;;KAGU,eAAe;UAEV;;EAEf,SAAS;;EAET;;EAEA;;EAEA;;KAKU,uBACR,sBACA,yBACA,8BACA,0BACA,sBACA;UAEM;EACR;;EAEA;;UAGe,4BAA4B;EAC3C;EACA;EACA;;UAGe,+BAA+B;EAC9C;EACA;EACA;;UAGe,oCAAoC;EACnD;EACA;EACA;;;;;;UAOe,gCAAgC;EAC/C;EACA,OAAO,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;UAI1C,4BAA4B;EAC3C;EACA,OAAO,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;;;;UAO1C,0BAA0B;EACzC;EACA,UAAU,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;;;;iBAO9C,iBAAiB,MAAM,uBAAuB;;;UCjH7C;EACf;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,gBAAgB;EAChB;;UAGe;EACf;EACA;EACA;EACA;;UAKe;EACf;EAQA;EACA;EACA;EACA;;UAKe;EACf;EACA;EACA,QAAQ;;UAGO;EACf;EACA;EACA,YAAY;;UAGG;EACf;EACA;EACA;EACA;EACA;;UAKe;EACf;EACA;EACA,OAAO;EACP,iBAAiB;EACjB,aAAa;EACb;EACA;EACA;EACA,WAAW;;EAEX,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA;IAAmB;IAAc;;EACjC;EACA;;UAGe;EACf,OAAO;EACP;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;IAAc;IAAc;;EAC5B;IAAmB;IAAc,QAAQ;;EACzC;IAAc;IAAkB;;EAChC;;UAKe;EACf;EACA;EACA;EACA;EACA,SAAS;EACT,OAAO;EACP;IACE;IACA,WAAW;MAAiB;MAAa;MAAgB;;IACzD,aAAa;MAAiB;MAAa;;IAC3C;MAAW;MAAkB;MAAe;;IAC5C;MAAa;MAAkB;MAAe;;;;UAMjC;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;GACC;;UAGc;EACf;EACA,QAAQ;;EAER;;UAKe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;IACE;MACE;MACA;MACA;;;EAGJ,OAAO;EACP,gBAAgB;;UAKD;EACf;EACA,QAAQ,OAAO;EACf,YAAY,OAAO;;UAGJ;EACf;EACA;;;;;;;;;;KAWU;UAEK;EACf;EACA;EACA;EACA,oBAAoB;EACpB,mBAAmB;EACnB;EACA;;EAEA,QAAQ;;;;;;;EAOR;;;;;;;EAOA;;;;;;EAMA;;UAGe;EACf;EACA;EACA;IAAa;IAAiB;IAAkB;;EAChD;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;IAAa;IAAiB;IAAkB;;EAChD;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;;EAEA;;;;;;EAMA;EACA;EACA,SAAS;EACT,YAAY;EACZ;EACA;EACA;;UAKe;EACf,WAAW;EACX,QAAQ;EACR;EACA;EACA;EACA;EACA;EACA;;EAEA,aAAa;;UAGE;EACf,UAAU;EACV,OAAO;EACP,WAAW;;EAEX,aAAa;EACb;EACA,WAAW;EACX,SAAS;;KAGC,WAAW,MAAM,YAAY,OAAO,eAAe,QAAQ;UAItD;EACf;EACA;EACA;EACA;EACA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA"}
|
|
@@ -121,6 +121,13 @@ function assertValidAnalystUsageReceipt(receipt, context = "AnalystContext.recor
|
|
|
121
121
|
if (receipt.cost.kind !== "uncaptured") assertNonNegativeFinite(receipt.cost.usd, "cost.usd", context);
|
|
122
122
|
else if (receipt.cost.usd !== null) throw new Error(`${context}: uncaptured cost.usd must be null`);
|
|
123
123
|
if (receipt.knownCostUsd !== void 0) assertNonNegativeFinite(receipt.knownCostUsd, "knownCostUsd", context);
|
|
124
|
+
if (receipt.partialTokens) {
|
|
125
|
+
const { input, output } = receipt.partialTokens;
|
|
126
|
+
if (receipt.tokens) throw new Error(`${context}: partialTokens must be absent when tokens is complete`);
|
|
127
|
+
if (input === null && output === null) throw new Error(`${context}: partialTokens must carry at least one reported side`);
|
|
128
|
+
if (input !== null) assertNonNegativeSafeInteger(input, "partialTokens.input", context);
|
|
129
|
+
if (output !== null) assertNonNegativeSafeInteger(output, "partialTokens.output", context);
|
|
130
|
+
}
|
|
124
131
|
}
|
|
125
132
|
function assertNonNegativeSafeInteger(value, field, context) {
|
|
126
133
|
if (!Number.isSafeInteger(value) || value < 0) throw new Error(`${context}: ${field} must be a non-negative safe integer`);
|
|
@@ -131,4 +138,4 @@ function assertNonNegativeFinite(value, field, context) {
|
|
|
131
138
|
//#endregion
|
|
132
139
|
export { computeFindingId as a, validateUsageSettlementTimeout as i, settleUsageReceiptFromCostLedger as n, makeFinding as o, usageReceiptFromCostLedger as r, makeProposalFinding as s, assertValidAnalystUsageReceipt as t };
|
|
133
140
|
|
|
134
|
-
//# sourceMappingURL=usage-receipt-
|
|
141
|
+
//# sourceMappingURL=usage-receipt-EVI8B8Xu.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"usage-receipt-EVI8B8Xu.js","names":[],"sources":["../src/analyst/types.ts","../src/analyst/usage-receipt.ts"],"sourcesContent":["/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\nexport type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info'\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAiPA,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD;;ACxSA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E"}
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { c as CostLedgerHandle } from "../cost-ledger-
|
|
2
|
-
import { Z as LlmRouteRequirements, q as LlmClientOptions } from "../types-
|
|
1
|
+
import { c as CostLedgerHandle } from "../cost-ledger-Bv_e8XHY.js";
|
|
2
|
+
import { Z as LlmRouteRequirements, q as LlmClientOptions } from "../types-XMVEdrE_.js";
|
|
3
3
|
import { s as TraceStore } from "../store-CT9YIIve.js";
|
|
4
|
-
import { _ as FeedbackTrajectoryStore } from "../feedback-trajectory-
|
|
4
|
+
import { _ as FeedbackTrajectoryStore } from "../feedback-trajectory-GgoS0-MK.js";
|
|
5
5
|
import { z } from "zod";
|
|
6
6
|
import { ServerType } from "@hono/node-server";
|
|
7
7
|
import { Hono } from "hono";
|
package/dist/wire/index.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import { A as TraceEventSchema, C as HealthResponseSchema, D as RubricDimensionSchema, E as ListRubricsResponseSchema, F as hashRubric, M as TracesIngestResponseSchema, N as VersionResponseSchema, O as RubricInfoSchema, P as WIRE_VERSION, S as FeedbackTrajectorySchema, T as JudgeResultSchema, _ as ErrorResponseSchema, a as runRpcBatch, b as FeedbackIngestResponseSchema, c as WireError, d as handleListRubrics, f as handleTracesIngest, g as listBuiltinRubrics, h as getBuiltinRubric, i as dispatchRpc, j as TracesIngestRequestSchema, k as RubricSchema, l as handleFeedbackIngest, m as BUILTIN_RUBRICS, n as startServer, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi, t as createApp, u as handleJudge, v as FailureModeSchema, w as JudgeRequestSchema, x as FeedbackLabelSchema, y as FeedbackAttemptSchema } from "../server-
|
|
1
|
+
import { A as TraceEventSchema, C as HealthResponseSchema, D as RubricDimensionSchema, E as ListRubricsResponseSchema, F as hashRubric, M as TracesIngestResponseSchema, N as VersionResponseSchema, O as RubricInfoSchema, P as WIRE_VERSION, S as FeedbackTrajectorySchema, T as JudgeResultSchema, _ as ErrorResponseSchema, a as runRpcBatch, b as FeedbackIngestResponseSchema, c as WireError, d as handleListRubrics, f as handleTracesIngest, g as listBuiltinRubrics, h as getBuiltinRubric, i as dispatchRpc, j as TracesIngestRequestSchema, k as RubricSchema, l as handleFeedbackIngest, m as BUILTIN_RUBRICS, n as startServer, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi, t as createApp, u as handleJudge, v as FailureModeSchema, w as JudgeRequestSchema, x as FeedbackLabelSchema, y as FeedbackAttemptSchema } from "../server-D6XJQHw7.js";
|
|
2
2
|
export { BUILTIN_RUBRICS, ErrorResponseSchema, FailureModeSchema, FeedbackAttemptSchema, FeedbackIngestResponseSchema, FeedbackLabelSchema, FeedbackTrajectorySchema, HealthResponseSchema, JudgeRequestSchema, JudgeResultSchema, ListRubricsResponseSchema, RubricDimensionSchema, RubricInfoSchema, RubricSchema, TraceEventSchema, TracesIngestRequestSchema, TracesIngestResponseSchema, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
|
|
@@ -8,6 +8,21 @@ Every hard-coded model id or endpoint default is verifiable against the live rou
|
|
|
8
8
|
|
|
9
9
|
Enforced by: `preflightModels` (membership + optional probe) and `assertModelsServed` (gate that names every unreachable id with status + detail).
|
|
10
10
|
|
|
11
|
+
## 1a. Reachable is not the same as identified
|
|
12
|
+
|
|
13
|
+
A 200 proves something answered, never that the requested model answered.
|
|
14
|
+
A routing gateway can accept one id and reply from another, and the swap is silent: same status, same shape, only the response's `model` field betrays it.
|
|
15
|
+
Every claim that names a model — a leaderboard row, a per-model cost, a cross-family judge panel, a non-self-judging exclusion — must be built from the id the provider echoed, not the id we sent.
|
|
16
|
+
Requested is intent; served is evidence.
|
|
17
|
+
|
|
18
|
+
Two failure shapes, and only one is a defect.
|
|
19
|
+
Snapshot resolution (`gpt-4o-mini` → `gpt-4o-mini-2024-07-18`) pins a floating alias to a reproducible build and is the behaviour we want.
|
|
20
|
+
Substitution (`gpt-4.1-mini` → `gemini-2.5-flash-lite`) mislabels every number the call produces.
|
|
21
|
+
A response that echoes no model id at all is unproven, which fails closed rather than defaulting to agreement.
|
|
22
|
+
|
|
23
|
+
Enforced by: `assertServedModel` / `assertServedModels` per call, `assertCrossFamilyServed` for panel diversity computed over the ids that answered, `LlmClientOptions.assertServedModel` to enforce it at the transport, and `assertModelsServed` (probe mode), which now fails a substituted id exactly as it fails a dead one.
|
|
24
|
+
`assertCrossFamily` reads requested ids and therefore proves configuration only — reach for the served-side check wherever the diversity claim is load-bearing.
|
|
25
|
+
|
|
11
26
|
## 2. Probe the platform before peeling client layers
|
|
12
27
|
|
|
13
28
|
When a request fails, one direct call against the live endpoint bisects platform-versus-client before any code-level debugging begins. A 401 from the router on a `model_not_found` is the platform telling you the default is dead; a connection refused is the platform being unreachable. Establish which side is at fault with a probe first, then debug only the side that is actually broken.
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
# Prime Analyst Runner
|
|
2
|
+
|
|
3
|
+
The `prime` analyst runs an RLM coding agent as a trace analyst through an OpenAI-compatible cli-bridge.
|
|
4
|
+
It is the third scored arm of `agent-eval analyst-benchmark`, beside the recursive `dspy-rlm` engine and the one-shot `direct` baseline, and it exists so the prime-vs-dspy comparison is reproducible from this repository alone.
|
|
5
|
+
It speaks the CodeTraceBench failure-block contract only; `--analyst prime` with `--dataset agentrx` is rejected.
|
|
6
|
+
|
|
7
|
+
Implementation: `src/analyst/benchmark-runner-prime.ts` (`createPrimeBenchmarkRunner`) binds the CodeTraceBench block grammar to the shared protocol in `src/analyst/prime-protocol.ts` and `src/analyst/prime-bridge-transport.ts`.
|
|
8
|
+
Wiring: `--analyst prime` in `src/analyst/benchmark-command.ts`.
|
|
9
|
+
|
|
10
|
+
## What the runner does
|
|
11
|
+
|
|
12
|
+
The benchmark command prepares every case identically for every runner: it loads the label-free trajectory, appends the row's final-verification artifacts as benchmark-verification spans, and hands each runner the same trace store.
|
|
13
|
+
The prime runner adds nothing to that input.
|
|
14
|
+
|
|
15
|
+
Per case it:
|
|
16
|
+
|
|
17
|
+
1. Projects the full span set with `viewTrace` and serializes it as inline JSON in the prompt.
|
|
18
|
+
Prime has no REPL and no trace tools, so the same projection the dspy typed path binds as a REPL variable is delivered as text.
|
|
19
|
+
2. Sends one user message to `<bridge>/v1/chat/completions`: the CodeTraceBench task definition (`CODE_TRACE_BENCH_ANALYST_PROMPT`), a strict block-JSON output contract, the trajectory JSON, and the final-verification spans.
|
|
20
|
+
3. Parses the reply's fenced JSON object (`{ "answer", "blocks": [...] }`), validates each block row, and expands accepted blocks into one scored finding per member step with the published `expandCodeTraceFailureBlocks` — the exact conversion every benchmark runner uses.
|
|
21
|
+
|
|
22
|
+
Findings carry `analyst_id: 'prime'` and score through the same evidence resolution, comparison, and calibration paths as every other arm.
|
|
23
|
+
|
|
24
|
+
## Reproduce: prime vs dspy-rlm on the same rows
|
|
25
|
+
|
|
26
|
+
Case selection is deterministic in `--labels`, `--limit`, and `--seed`, so two runs that share those flags (and the same `--trace-dir`, `--artifact-dir`, `--revision`, `--split`) score exactly the same rows.
|
|
27
|
+
Run the two arms into two output directories and compare:
|
|
28
|
+
|
|
29
|
+
```sh
|
|
30
|
+
# Arm 1: prime through the cli-bridge (no model-owner module; the bridge owns execution)
|
|
31
|
+
agent-eval analyst-benchmark \
|
|
32
|
+
--dataset codetracebench \
|
|
33
|
+
--analyst prime \
|
|
34
|
+
--bridge-url http://localhost:4181 \
|
|
35
|
+
--labels .artifacts/manifest.jsonl \
|
|
36
|
+
--trace-dir .artifacts/traces \
|
|
37
|
+
--artifact-dir .artifacts/results \
|
|
38
|
+
--out .artifacts/prime-run \
|
|
39
|
+
--revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
|
|
40
|
+
--split verified \
|
|
41
|
+
--model prime/zai/glm-5.2 \
|
|
42
|
+
--timeout-ms 1200000 \
|
|
43
|
+
--limit 20 \
|
|
44
|
+
--seed 7 \
|
|
45
|
+
--concurrency 1
|
|
46
|
+
|
|
47
|
+
# Arm 2: the recursive DSPy engine on the SAME rows (same labels/limit/seed)
|
|
48
|
+
agent-eval analyst-benchmark \
|
|
49
|
+
--dataset codetracebench \
|
|
50
|
+
--analyst dspy-rlm \
|
|
51
|
+
--labels .artifacts/manifest.jsonl \
|
|
52
|
+
--trace-dir .artifacts/traces \
|
|
53
|
+
--artifact-dir .artifacts/results \
|
|
54
|
+
--out .artifacts/dspy-run \
|
|
55
|
+
--revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
|
|
56
|
+
--split verified \
|
|
57
|
+
--model-owner-module ./dist/runtime-model-owner.mjs \
|
|
58
|
+
--model opencode/zai-coding-plan/glm-5.2 \
|
|
59
|
+
--python .venv/bin/python \
|
|
60
|
+
--timeout-ms 1200000 \
|
|
61
|
+
--limit 20 \
|
|
62
|
+
--seed 7 \
|
|
63
|
+
--concurrency 1
|
|
64
|
+
|
|
65
|
+
node benchmarks/trace-analysis/tools/compare-analyst-runs.mjs \
|
|
66
|
+
.artifacts/prime-run/result.json .artifacts/dspy-run/result.json
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
`--model` keeps its normal semantics; for prime it is the bridge model id in `<backend>/<provider>/<model>` form (`prime/zai/glm-5.2`), which the bridge maps to its configured backend model.
|
|
70
|
+
Prime analyses routinely exceed the 300-second default deadline, so set `--timeout-ms` explicitly (the proven external rig used 1200000).
|
|
71
|
+
`--no-repair` disables the bounded repair turn described below.
|
|
72
|
+
|
|
73
|
+
## Bridge prerequisites
|
|
74
|
+
|
|
75
|
+
The runner needs a running cli-bridge whose `prime` backend is enabled:
|
|
76
|
+
|
|
77
|
+
- `BRIDGE_BACKENDS=prime` — enable the prime backend in the bridge.
|
|
78
|
+
- `PRIME_BIN` — absolute path to the prime binary (on nix installs, the nix store path of the `prime` executable).
|
|
79
|
+
- `PRIME_MODELS_JSON` — the bridge's model table mapping bridge model ids such as `prime/zai/glm-5.2` to the backend provider and model the prime agent runs.
|
|
80
|
+
|
|
81
|
+
The bridge listens on `http://localhost:4181` by default; pass `--bridge-url` when it listens elsewhere.
|
|
82
|
+
Provider credentials live in the bridge process, never in this command: for `--analyst prime` there is no `--model-owner-module`, and passing one is an error.
|
|
83
|
+
|
|
84
|
+
## Protocol notes
|
|
85
|
+
|
|
86
|
+
- **Short-strings contract, no rationale.**
|
|
87
|
+
The output contract caps every string the model must emit and forbids a `rationale` field.
|
|
88
|
+
Why: stream-splice corruption on long strings was measured on the live bridge path — the bridge splices its backend's streamed output into one reply, and long strings arrive corrupted often enough to void otherwise-correct work.
|
|
89
|
+
Short claims survive the splice; block coordinates carry the signal.
|
|
90
|
+
- **One bounded repair turn.**
|
|
91
|
+
A structurally malformed reply (no parseable JSON object, or no `blocks` array) gets exactly one stateless follow-up call carrying the malformed reply plus the output contract — never the trajectory — mirroring the dspy arm's typed-adapter repair so both arms face the same structured-output affordance.
|
|
92
|
+
Still malformed after repair = failed observation with a typed error (`PrimeMalformedReplyError`), recorded exactly as a dspy-rlm failure is.
|
|
93
|
+
Zero valid blocks from a well-formed reply is an honest null, not a failure.
|
|
94
|
+
- **Oversized traces fall back to chunked projection.**
|
|
95
|
+
When the full `viewTrace` response is oversized, or the rendered JSON exceeds the 360k-char inline budget, the runner re-projects every span through chunked `viewSpans` at a 1200-byte per-attribute cap, in store order, and fails loud if any span drops or the result is still oversized.
|
|
96
|
+
- **Usage receipts.**
|
|
97
|
+
Token counts are the bridge's exact reported counts; USD is a rate-based estimate from the model's catalog rates (for `prime/zai/glm-5.2`, the z.ai coding-plan list rates: 0.6/2.2 USD per million input/output tokens).
|
|
98
|
+
A reply without usage stays uncaptured — never a silent zero — and a repair turn's usage merges into the case's receipt, poisoning each side independently so a measured count survives a missing partner.
|
|
99
|
+
A reply that reports only one side lands in `AnalystUsageReceipt.partialTokens` with `tokens: null`, `cost` uncaptured, and the reported side priced into `knownCostUsd` as a lower bound: `RunTokenUsage` has no nullable side, so carrying a one-sided count in `tokens` would mean writing a zero nobody measured.
|
|
100
|
+
When the bridge reports `estimated: true` — it derived the counts from character lengths because the backend CLI reported none — the receipt carries `tokensEstimated: true`, which is what separates a rate estimate over exact tokens from one over derived tokens.
|
|
101
|
+
- **Per-observation protocol digest.**
|
|
102
|
+
Every prime observation records `primeAnalystProtocolSha256()` in its runner metadata, hashing the question, task prompt, output contract, repair contract, and projection limits that actually ran.
|
|
103
|
+
|
|
104
|
+
## Reusing the protocol outside this benchmark
|
|
105
|
+
|
|
106
|
+
`src/analyst/prime-protocol.ts` is the consumer-agnostic core, exported from `@tangle-network/agent-eval/analyst`.
|
|
107
|
+
It speaks raw rows and names no finding type, so an analyzer with a different row grammar — span-grounded findings against its own artifact, say — binds it without importing CodeTraceBench types:
|
|
108
|
+
|
|
109
|
+
- `PrimeReplyContract<TRow>` supplies the rows field name, the contract lines spliced into both prompts, a single-pass `decodeRow`, and an optional `maxRows` cap applied to ACCEPTED rows so malformed rows never consume a slot.
|
|
110
|
+
- `buildPrimePrompt` / `buildPrimeRepairPrompt` compose the prompts; the repair prompt never carries the trajectory.
|
|
111
|
+
- `runPrimeExchange` runs the call, the bounded repair turn, and row decoding, returning one typed outcome whose `PrimeFailure.kind` separates `transport`, `http-status`, `unparseable-json`, `no-content`, `deadline`, `malformed-reply`, and `aborted`.
|
|
112
|
+
A cancelled run is never recorded as an analyzer verdict.
|
|
113
|
+
- `projectPrimeTrajectory` runs the render → measure → fall back → re-measure → fail-loud ladder over a caller-supplied `PrimeProjectionSource`, so the source of the projection stays the consumer's choice.
|
|
114
|
+
- `normalizePrimeUsage` / `mergePrimeRawUsage` keep the bridge's report lossless; `analystUsageReceiptFromPrimeUsage` is the agent-eval-only binding to the typed receipt, so a consumer with no pricing table simply does not call it.
|
|
115
|
+
- `primeProtocolSha256` hashes the ACTUALLY composed contract, so two consumers that both stamp `analyst_id: 'prime'` while asking different questions get different digests by construction.
|
|
116
|
+
|
|
117
|
+
## Status
|
|
118
|
+
|
|
119
|
+
The first prime-vs-dspy comparison batch (20+ live CodeTraceBench cases through cli-bridge on `prime/zai/glm-5.2`) is in flight on the proven external rig this runner was ported from.
|
|
120
|
+
Numbers land in `benchmarks/trace-analysis/` when the batch completes; until then this document makes no accuracy claim for the prime arm.
|
package/docs/trace-analysis.md
CHANGED
|
@@ -227,7 +227,7 @@ Cross-run and pooled comparisons use `benchmarks/trace-analysis/tools/compare-an
|
|
|
227
227
|
`agent-eval analyst-benchmark` runs the public AgentRx or CodeTraceBench adapters with:
|
|
228
228
|
|
|
229
229
|
1. an empty-finding baseline,
|
|
230
|
-
2. the
|
|
230
|
+
2. the scored analyst `--analyst` selects: the recursive DSPy RLM engine (`dspy-rlm`, default), the one-shot `direct` baseline, or the `prime` RLM coding agent behind an OpenAI-compatible cli-bridge (CodeTraceBench only; see [prime-analyst.md](./prime-analyst.md) for bridge prerequisites and the prime-vs-dspy reproduce commands).
|
|
231
231
|
|
|
232
232
|
```sh
|
|
233
233
|
agent-eval analyst-benchmark \
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.144.
|
|
3
|
+
"version": "0.144.6",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -168,8 +168,8 @@
|
|
|
168
168
|
"dependencies": {
|
|
169
169
|
"@asteasolutions/zod-to-openapi": "^9.1.0",
|
|
170
170
|
"@hono/node-server": "^2.0.12",
|
|
171
|
-
"@tangle-network/agent-core": "0.5.
|
|
172
|
-
"@tangle-network/agent-interface": "0.
|
|
171
|
+
"@tangle-network/agent-core": "0.5.4",
|
|
172
|
+
"@tangle-network/agent-interface": "0.46.1",
|
|
173
173
|
"@tangle-network/agent-trace-contract": "^1.0.2",
|
|
174
174
|
"hono": "^4.12.32",
|
|
175
175
|
"linear-sum-assignment": "1.0.9",
|