@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,360 +0,0 @@
1
- import { AgentProfile } from '@tangle-network/agent-interface';
2
- import { V as ValidationError } from './errors-oeQrLqXC.js';
3
- import { F as FailureClass } from './schema-B3Q3l9Z_.js';
4
-
5
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
6
- type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
7
- [key: string]: AgentProfileJson;
8
- };
9
- type AgentProfileDimensionValue = string | number | boolean | null;
10
- interface AgentProfileSource {
11
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
12
- kind: string;
13
- /** sha256 over the canonical source profile object. */
14
- hash: string;
15
- }
16
- interface AgentProfileSourceInput {
17
- kind: string;
18
- /** Precomputed sha256 for callers that already sign their profile artifact. */
19
- hash?: string;
20
- /** Full canonical runtime profile; hashed and then discarded from the cell. */
21
- profile?: AgentProfileJson;
22
- }
23
- interface AgentProfileHarness {
24
- id: string;
25
- version?: string;
26
- hash?: string;
27
- }
28
- interface AgentProfileCellInput {
29
- profileId: string;
30
- sourceProfile: AgentProfileSourceInput;
31
- harness?: AgentProfileHarness;
32
- model?: string;
33
- promptHash?: string;
34
- dimensions?: Record<string, AgentProfileDimensionValue>;
35
- }
36
- interface AgentProfileCell {
37
- schemaVersion: AgentProfileCellSchemaVersion;
38
- cellId: string;
39
- profileId: string;
40
- sourceProfile: AgentProfileSource;
41
- harness?: AgentProfileHarness;
42
- model?: string;
43
- promptHash?: string;
44
- dimensions?: Record<string, AgentProfileDimensionValue>;
45
- }
46
- declare class AgentProfileCellValidationError extends ValidationError {
47
- readonly path: string;
48
- constructor(message: string, path?: string);
49
- }
50
- declare function buildAgentProfileCell(input: AgentProfileCellInput): Promise<AgentProfileCell>;
51
- declare function agentProfileCellHashMaterial(cell: AgentProfileCell): Omit<AgentProfileCell, 'cellId'>;
52
- /**
53
- * Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material fields, confirming the record has not been tampered with.
54
- */
55
- declare function verifyAgentProfileCell(cell: AgentProfileCell): Promise<boolean>;
56
- declare function validateAgentProfileCell(input: unknown): AgentProfileCell;
57
- declare function requireAgentProfileCell(record: {
58
- runId: string;
59
- agentProfile?: AgentProfileCell;
60
- }): AgentProfileCell;
61
- declare function agentProfileCellKey(record: {
62
- runId: string;
63
- agentProfile?: AgentProfileCell;
64
- }): string;
65
- declare function assertRunAgentProfileCell(record: {
66
- runId: string;
67
- model: string;
68
- promptHash: string;
69
- agentProfile?: AgentProfileCell;
70
- }): Promise<AgentProfileCell>;
71
- declare function groupRunsByAgentProfileCell<T extends {
72
- runId: string;
73
- agentProfile?: AgentProfileCell;
74
- }>(records: readonly T[]): Map<string, T[]>;
75
- /** Canonical `sourceProfile.kind` values. Two products fingerprinting the
76
- * same canonical profile MUST use the same kind for their cells to share
77
- * `sourceProfile.hash`. Extend rather than create new strings — adding a
78
- * new kind is a deliberate cross-product schema change. */
79
- declare const AGENT_PROFILE_KINDS: {
80
- /** A profile declared via `defineAgentProfile(...)` from
81
- * `@tangle-network/agent-interface`. The default kind for router-backed
82
- * and sandbox-backed products. */
83
- readonly AGENT_INTERFACE_PROFILE: "agent-interface-profile";
84
- };
85
- type AgentProfileKind = (typeof AGENT_PROFILE_KINDS)[keyof typeof AGENT_PROFILE_KINDS];
86
- /** Canonicalize an arbitrary value into `AgentProfileJson` by JSON
87
- * round-trip. Throws when the value contains anything not representable
88
- * as JSON (functions, BigInt, cycles) — non-portable profiles fail loud
89
- * rather than silently dropping fields. */
90
- declare function toAgentProfileJson(value: unknown): AgentProfileJson;
91
- /** Canonical AgentProfile shape required when deriving a stable cell id. */
92
- type AgentInterfaceProfileLike = AgentProfile & {
93
- name: string;
94
- version: string;
95
- };
96
- /** Higher-level helper that hard-codes the canonical
97
- * `agent-interface-profile` kind plus the JSON canonicalization. Equivalent
98
- * to calling `buildAgentProfileCell` with `profileId = \`${name}@${version}\``
99
- * and `sourceProfile = { kind: AGENT_INTERFACE_PROFILE, profile: <round-tripped> }`.
100
- *
101
- * Use this from any product consuming an agent-interface `AgentProfile`; the
102
- * manual `buildAgentProfileCell` call is reserved for advanced cases
103
- * (custom kinds, pre-computed source hashes, alternate profileId
104
- * conventions). */
105
- declare function buildAgentInterfaceProfileCell(profile: AgentInterfaceProfileLike, input: Omit<AgentProfileCellInput, 'profileId' | 'sourceProfile'>): Promise<AgentProfileCell>;
106
-
107
- /**
108
- * Paper-grade RunRecord schema + runtime validator.
109
- *
110
- * Every run that participates in a promotion gate, paper table, or
111
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
112
- * fields are exactly those the paper "Two Loops, Three Roles" requires
113
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
114
- * holdout split tag and either a `searchScore` or a `holdoutScore`.
115
- *
116
- * This is intentionally NOT a replacement for the rich `Run` /
117
- * `ProposeReviewReport` / `ScenarioResult` types already in the
118
- * package. Those are runtime structures with full provenance. A
119
- * `RunRecord` is the analysis-time projection — the JSON-friendly
120
- * row you'd put in a parquet file or paste into a notebook.
121
- *
122
- * Validate at the boundary:
123
- *
124
- * const rec = validateRunRecord(rawJson) // throws on missing
125
- * const ok = isRunRecord(rawJson) // boolean check
126
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
127
- *
128
- * The validator runs in pure TS — zod is intentionally NOT a
129
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
130
- */
131
-
132
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
133
- * combined train+test pool that the optimizer is allowed to read. */
134
- type RunSplitTag = 'search' | 'dev' | 'holdout';
135
- interface RunTokenUsage {
136
- input: number;
137
- output: number;
138
- cached?: number;
139
- }
140
- /**
141
- * How a run's USD amount was obtained.
142
- *
143
- * `costUsd` remains mandatory for wire compatibility. New producers should
144
- * always populate this discriminated union so a missing bill is never
145
- * mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
146
- * the legacy `0` sentinel while this field carries the truthful null.
147
- */
148
- type RunCostProvenance = {
149
- kind: 'observed';
150
- usd: number;
151
- } | {
152
- kind: 'estimated';
153
- usd: number;
154
- } | {
155
- kind: 'uncaptured';
156
- usd: null;
157
- };
158
- interface RunJudgeMetadata {
159
- model: string;
160
- promptVersion: string;
161
- /** [0,1] confidence the judge declared. Constant judge confidence
162
- * across many runs is a fallback signal (see `canary.ts`). */
163
- confidence: number;
164
- /** True if the judge degraded to a fallback path (rules-only,
165
- * prior-call cache, etc.). The canary uses this to alert. */
166
- fallback: boolean;
167
- }
168
- /**
169
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
170
- * judges over a multi-dimensional rubric.
171
- *
172
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
173
- * composite the gate uses. The full breakdown belongs here so consumers
174
- * can answer "which judge disagreed?", "which dimension dragged the
175
- * composite down?", and "did half the panel fail?" without re-running.
176
- *
177
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
178
- * `composite` are convenience projections — derivable but precomputed so
179
- * downstream IRR primitives (`interRaterReliability`,
180
- * `corpusInterRaterAgreement`) and reporters don't pay the same
181
- * aggregation twice.
182
- *
183
- * Fail-loud discipline: judges that errored out land in `failedJudges`
184
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
185
- * run); the explicit list makes a partial-failure recorded as such.
186
- */
187
- interface JudgeScoresRecord {
188
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
189
- perJudge: Record<string, Record<string, number>>;
190
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
191
- perDimMean: Record<string, number>;
192
- /** Composite mean across all dims and judges. Mirrors the score
193
- * the gate sees on `outcome.searchScore` / `holdoutScore`. */
194
- composite: number;
195
- /** Judges that errored or returned an unparseable verdict. Recorded
196
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
197
- * not inferred from missing keys in `perJudge`. */
198
- failedJudges?: string[];
199
- /** Free-form notes the judges emitted (joined across judges or
200
- * first-judge only — consumer's choice). */
201
- notes?: string;
202
- }
203
- interface RunOutcome {
204
- /** Score on the search/optimization split. Optional because a
205
- * holdout-only evaluation only fills `holdoutScore`. */
206
- searchScore?: number;
207
- /** Score on the held-out split. Optional because a search-only run
208
- * only fills `searchScore`. At least one must be present. */
209
- holdoutScore?: number;
210
- /** Bag of any other metric the run produced — judge dimensions,
211
- * pass/fail counters, latency stats, etc. Numeric only — keeps
212
- * reporters honest. */
213
- raw: Record<string, number>;
214
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
215
- * judgements populate this; substrate primitives like
216
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
217
- * these records as input. Optional — single-judge or scalar-only
218
- * runs leave it unset. */
219
- judgeScores?: JudgeScoresRecord;
220
- /** Authenticity / realness verdict — did the run build the REAL thing on the
221
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
222
- * with an authenticity config populate it. Carried in the corpus so the
223
- * flywheel / off-policy learning can optimize for real completion, not gamed
224
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
225
- * must not count as a real success regardless of `score`. */
226
- realness?: {
227
- score: number;
228
- gated: boolean;
229
- reason?: string;
230
- };
231
- }
232
- /**
233
- * Mandatory paper-grade fields for a single evaluation run. Optional
234
- * fields are extension points; mandatory fields throw if missing.
235
- *
236
- * Hash discipline:
237
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
238
- * model (after any steering bundle merge).
239
- * - `configHash` is the sha256 of the effective run config (model,
240
- * temperature, tools, judges, splits). The pair (promptHash,
241
- * configHash) uniquely identifies an experiment cell.
242
- *
243
- * Model snapshot discipline:
244
- * - `model` MUST encode a snapshot version. Bare aliases like
245
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
246
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
247
- */
248
- interface RunRecord {
249
- /** UUID for the run. */
250
- runId: string;
251
- /** Logical experiment grouping (a treatment vs a baseline within
252
- * the same sweep should share `experimentId`). */
253
- experimentId: string;
254
- /** Stable identifier for the candidate (variant) being run. The
255
- * promotion gate compares two `candidateId`s on matched items. */
256
- candidateId: string;
257
- /** RNG seed for the run. Always recorded — silent re-seeding is
258
- * the most common cause of non-reproducible numbers. */
259
- seed: number;
260
- /** Model identifier WITH snapshot version. */
261
- model: string;
262
- /** sha256 of the effective prompt (post-steering). */
263
- promptHash: string;
264
- /** sha256 of the effective config. */
265
- configHash: string;
266
- /** Git SHA the harness was run from. */
267
- commitSha: string;
268
- /** End-to-end wall-clock duration in milliseconds. */
269
- wallMs: number;
270
- /** Time spent queued before execution started, if known. */
271
- queueMs?: number;
272
- /** Total USD cost. Mandatory — runs without a cost number are
273
- * unbounded by definition and must not be admitted into the gate.
274
- * `0` is retained as the compatibility sentinel for an uncaptured amount;
275
- * inspect `costProvenance` before treating it as observed. */
276
- costUsd: number;
277
- /** Observed, model-priced estimate, or genuinely uncaptured USD amount.
278
- * Optional only so existing serialized RunRecords remain valid. */
279
- costProvenance?: RunCostProvenance;
280
- /** Token usage breakdown. */
281
- tokenUsage: RunTokenUsage;
282
- /** Judge-side metadata, if a judge was used. */
283
- judgeMetadata?: RunJudgeMetadata;
284
- /** Per-split scores + raw bag. */
285
- outcome: RunOutcome;
286
- /** Canonical, cross-agent failure class drawn from the shared
287
- * `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
288
- * "which failure dominates across the whole fleet" answerable in ONE
289
- * vocabulary — every agent classifies against the same enum. Producers
290
- * set it via the substrate classifier; leave unset only when the failure
291
- * genuinely can't be classified. */
292
- failureClass?: FailureClass;
293
- /** Free-form domain-specific failure detail, scoped UNDER `failureClass`
294
- * (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
295
- * The within-agent drill-down; `failureClass` is the cross-agent key. */
296
- failureMode?: string;
297
- /** Which split this run was drawn from. */
298
- splitTag: RunSplitTag;
299
- /**
300
- * Stable scenario identifier the run was scored against. Optional for
301
- * backwards compatibility, but **strongly recommended**: every primitive
302
- * that pairs runs by scenario (preferences, paired stats, BT tournament)
303
- * keys on this. The campaign artifact populates it canonically; legacy
304
- * runs without it fall back to inference from `outcome.raw.scenario_id`
305
- * or `experimentId`.
306
- */
307
- scenarioId?: string;
308
- /**
309
- * Canonical identity for the agent profile cell that produced this row:
310
- * profile artifact hash plus optional harness/model/prompt/reporting
311
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
312
- * longitudinal reports by the complete source profile, not by a loose
313
- * candidate label or opaque config hash.
314
- */
315
- agentProfile?: AgentProfileCell;
316
- }
317
- declare class RunRecordValidationError extends ValidationError {
318
- readonly path: string;
319
- constructor(message: string, path?: string);
320
- }
321
- /**
322
- * Strict validator. Throws `RunRecordValidationError` on the first
323
- * missing or wrongly-typed field. Returns the input cast to
324
- * `RunRecord` on success — the validator does not coerce.
325
- */
326
- declare function validateRunRecord(input: unknown): RunRecord;
327
- /**
328
- * Resolve provenance for both new and legacy records.
329
- *
330
- * Legacy producers sometimes set `outcome.raw.cost_estimated = 1`. A positive
331
- * unlabeled amount is treated as observed, matching the historical contract.
332
- * Zero without an explicit label is conservatively uncaptured: claiming an
333
- * observed $0 would be stronger than the serialized evidence supports.
334
- */
335
- declare function resolveRunCostProvenance(run: Pick<RunRecord, 'costUsd' | 'costProvenance' | 'outcome'>): RunCostProvenance;
336
- /** Boolean validator — convenience for filtering arrays. */
337
- declare function isRunRecord(input: unknown): input is RunRecord;
338
- /** Non-throwing validator — returns a discriminated union. */
339
- declare function parseRunRecordSafe(input: unknown): {
340
- ok: true;
341
- value: RunRecord;
342
- } | {
343
- ok: false;
344
- error: RunRecordValidationError;
345
- };
346
- /** Round-trip helper — `JSON.parse(JSON.stringify(record))` then validate. */
347
- declare function roundTripRunRecord(record: RunRecord): RunRecord;
348
- /**
349
- * Heuristic snapshot check. Accepts:
350
- * - `name@YYYY-MM-DD` (Anthropic style: `claude-sonnet-4-6@2025-04-15`)
351
- * - `name-YYYYMMDD` (OpenAI style: `gpt-4o-2024-11-20`)
352
- * - `name@<arbitrary-token>` (allow opaque snapshots like `@v3`)
353
- * - explicit `:date-...` Vertex-style tags
354
- *
355
- * Rejects bare aliases like `claude-sonnet-4` or `gpt-4o` that remap
356
- * silently as providers ship new snapshots.
357
- */
358
- declare function modelHasSnapshot(model: string): boolean;
359
-
360
- export { type AgentProfileCell as A, requireAgentProfileCell as B, resolveRunCostProvenance as C, roundTripRunRecord as D, toAgentProfileJson as E, validateAgentProfileCell as F, validateRunRecord as G, verifyAgentProfileCell as H, type JudgeScoresRecord as J, type RunRecord as R, type RunSplitTag as a, type RunCostProvenance as b, type RunTokenUsage as c, type RunJudgeMetadata as d, type AgentProfileCellInput as e, type AgentProfileJson as f, AGENT_PROFILE_KINDS as g, type AgentInterfaceProfileLike as h, type AgentProfileCellSchemaVersion as i, AgentProfileCellValidationError as j, type AgentProfileDimensionValue as k, type AgentProfileHarness as l, type AgentProfileKind as m, type AgentProfileSource as n, type AgentProfileSourceInput as o, type RunOutcome as p, RunRecordValidationError as q, agentProfileCellHashMaterial as r, agentProfileCellKey as s, assertRunAgentProfileCell as t, buildAgentInterfaceProfileCell as u, buildAgentProfileCell as v, groupRunsByAgentProfileCell as w, isRunRecord as x, modelHasSnapshot as y, parseRunRecordSafe as z };
@@ -1,49 +0,0 @@
1
- import { a as RunSplitTag } from './run-record-BDH49H2E.js';
2
-
3
- interface RuntimeTrajectoryHookEvent {
4
- id: string;
5
- runId: string;
6
- scenarioId?: string;
7
- target: string;
8
- phase: string;
9
- timestamp: number;
10
- stepIndex?: number;
11
- parentId?: string;
12
- payload?: unknown;
13
- metadata?: Record<string, unknown>;
14
- }
15
- interface RuntimeTrajectoryRecord {
16
- id?: string;
17
- scenarioId?: string;
18
- splitTag?: RunSplitTag;
19
- runtimeEvents?: unknown;
20
- [key: string]: unknown;
21
- }
22
- interface RuntimeTrajectoryRunRecord {
23
- runId: string;
24
- scenarioId?: string;
25
- splitTag: RunSplitTag;
26
- }
27
- interface RuntimeTrajectoryEvidenceSummary {
28
- recordCount: number;
29
- recordWithRuntimeEventsCount: number;
30
- runtimeRunCount: number;
31
- lifecycleEventCount: number;
32
- defaultedSplitCount: number;
33
- }
34
- interface RuntimeTrajectoryEvidenceProjection {
35
- runs: RuntimeTrajectoryRunRecord[];
36
- events: RuntimeTrajectoryHookEvent[];
37
- summary: RuntimeTrajectoryEvidenceSummary;
38
- diagnostics: string[];
39
- }
40
- interface ProjectRuntimeTrajectoryEvidenceOptions<TRecord extends RuntimeTrajectoryRecord = RuntimeTrajectoryRecord> {
41
- records: TRecord[];
42
- defaultSplitTag?: RunSplitTag;
43
- recordIdOf?: (record: TRecord, index: number) => string | undefined;
44
- scenarioIdOf?: (record: TRecord, index: number) => string | undefined;
45
- }
46
- declare function projectRuntimeTrajectoryEvidence<TRecord extends RuntimeTrajectoryRecord>(options: ProjectRuntimeTrajectoryEvidenceOptions<TRecord>): RuntimeTrajectoryEvidenceProjection;
47
- declare function parseRuntimeTrajectoryHookEvent(input: unknown): RuntimeTrajectoryHookEvent | null;
48
-
49
- export { type ProjectRuntimeTrajectoryEvidenceOptions as P, type RuntimeTrajectoryRecord as R, type RuntimeTrajectoryEvidenceProjection as a, type RuntimeTrajectoryEvidenceSummary as b, type RuntimeTrajectoryHookEvent as c, type RuntimeTrajectoryRunRecord as d, projectRuntimeTrajectoryEvidence as e, parseRuntimeTrajectoryHookEvent as p };
@@ -1,201 +0,0 @@
1
- /**
2
- * TraceSchema v1 — the canonical data model for agent-eval.
3
- *
4
- * Every score, every failure class, every pipeline in the framework is
5
- * a view over this data. Shape it once, live with it.
6
- *
7
- * Wire-compatible with OpenTelemetry span semantics (see trace/otel.ts)
8
- * but extended with agent-specific span kinds (llm, tool, retrieval,
9
- * judge, sandbox) and first-class BudgetLedger / Artifact / JudgeVerdict
10
- * entities that OTEL leaves as free-form attributes.
11
- */
12
- declare const TRACE_SCHEMA_VERSION = "1.0.0";
13
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
14
- interface BudgetSpec {
15
- tokens?: number;
16
- wallMs?: number;
17
- calls?: number;
18
- usd?: number;
19
- }
20
- interface RunOutcome {
21
- score?: number;
22
- pass?: boolean;
23
- failureClass?: FailureClass;
24
- notes?: string;
25
- }
26
- /**
27
- * Layer — optional classification in a nested build workflow.
28
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
29
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
30
- * `app-runtime`: a run of the generated agent against a domain scenario.
31
- * `meta`: any meta-eval (judge replay, correlation analysis).
32
- */
33
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
34
- interface Run {
35
- runId: string;
36
- /**
37
- * Stable identifier of the scenario being executed.
38
- *
39
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
40
- * input WITHOUT this field, substituting a sensible default
41
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
42
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
43
- * keeps the persisted shape unambiguous for downstream filters + aggregations
44
- * while removing the boilerplate of inventing placeholder ids at the call site.
45
- */
46
- scenarioId: string;
47
- variantId?: string;
48
- datasetVersion?: string;
49
- /** Git SHA of agent code at run time. */
50
- codeSha?: string;
51
- /** Hash of the prompt template + any system prompt. */
52
- promptSha?: string;
53
- /** Model id + date + system-prompt hash, concatenated. */
54
- modelFingerprint?: string;
55
- seed?: number;
56
- /** Arbitrary environment markers (shell, docker version, tz). */
57
- envFingerprint?: Record<string, string>;
58
- /** Version of the redaction rules applied to this run. */
59
- redactionVersion?: string;
60
- /** Parent run in a nested build workflow. A builder run's children are
61
- * app-build runs; those children are app-runtime runs. */
62
- parentRunId?: string;
63
- /** Stable project identifier — groups runs across chats + sessions. */
64
- projectId?: string;
65
- /** Chat/conversation identifier within a project. */
66
- chatId?: string;
67
- /** Layer classification — hint for aggregation; not enforced. */
68
- layer?: RunLayer;
69
- startedAt: number;
70
- endedAt?: number;
71
- status: RunStatus;
72
- outcome?: RunOutcome;
73
- budget?: BudgetSpec;
74
- /** Free-form labels for downstream grouping. */
75
- tags?: Record<string, string>;
76
- }
77
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
78
- type SpanStatus = 'ok' | 'error';
79
- interface SpanBase {
80
- spanId: string;
81
- parentSpanId?: string;
82
- runId: string;
83
- kind: SpanKind;
84
- name: string;
85
- startedAt: number;
86
- endedAt?: number;
87
- status?: SpanStatus;
88
- error?: string;
89
- /** Anything not covered by typed fields. Kept deliberately free-form. */
90
- attributes?: Record<string, unknown>;
91
- }
92
- interface Message {
93
- role: 'system' | 'user' | 'assistant' | 'tool';
94
- content: string;
95
- tokens?: number;
96
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
97
- images?: Array<{
98
- artifactId?: string;
99
- url?: string;
100
- mime?: string;
101
- }>;
102
- }
103
- interface LlmSpan extends SpanBase {
104
- kind: 'llm';
105
- model: string;
106
- messages: Message[];
107
- output?: string;
108
- inputTokens?: number;
109
- outputTokens?: number;
110
- cachedTokens?: number;
111
- reasoningTokens?: number;
112
- costUsd?: number;
113
- finishReason?: string;
114
- }
115
- interface ToolSpan extends SpanBase {
116
- kind: 'tool';
117
- toolName: string;
118
- args: unknown;
119
- /** False when the source observed the call but did not capture its arguments. */
120
- argsCaptured?: boolean;
121
- result?: unknown;
122
- latencyMs?: number;
123
- }
124
- interface RetrievalSpan extends SpanBase {
125
- kind: 'retrieval';
126
- query: string;
127
- hits: Array<{
128
- docId: string;
129
- score: number;
130
- content?: string;
131
- }>;
132
- }
133
- interface JudgeSpan extends SpanBase {
134
- kind: 'judge';
135
- judgeId: string;
136
- /** Span this judgment applies to. */
137
- targetSpanId: string;
138
- dimension: string;
139
- /** Numeric score (free-range; interpretation up to the judge). */
140
- score: number;
141
- rationale?: string;
142
- evidence?: string;
143
- }
144
- interface SandboxSpan extends SpanBase {
145
- kind: 'sandbox';
146
- image?: string;
147
- command?: string;
148
- exitCode?: number;
149
- testsTotal?: number;
150
- testsPassed?: number;
151
- stdoutHash?: string;
152
- stderrHash?: string;
153
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
154
- wallMs?: number;
155
- }
156
- interface GenericSpan extends SpanBase {
157
- kind: 'agent' | 'custom';
158
- }
159
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
160
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
161
- interface TraceEvent {
162
- eventId: string;
163
- runId: string;
164
- spanId?: string;
165
- kind: EventKind;
166
- timestamp: number;
167
- payload: Record<string, unknown>;
168
- }
169
- interface BudgetLedgerEntry {
170
- runId: string;
171
- dimension: keyof BudgetSpec;
172
- limit: number;
173
- consumed: number;
174
- remaining: number;
175
- timestamp: number;
176
- breached: boolean;
177
- /** Span that triggered this entry, if any. */
178
- spanId?: string;
179
- }
180
- interface Artifact {
181
- artifactId: string;
182
- runId: string;
183
- spanId?: string;
184
- contentType: string;
185
- sizeBytes: number;
186
- /** sha256 in hex. */
187
- hash: string;
188
- /** External storage URL (R2, S3, filesystem path). */
189
- storageUrl?: string;
190
- /** Inline content for small blobs — keep under ~64KB. */
191
- inlineContent?: string;
192
- }
193
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
194
- declare const FAILURE_CLASSES: readonly FailureClass[];
195
- declare function isLlmSpan(s: Span): s is LlmSpan;
196
- declare function isToolSpan(s: Span): s is ToolSpan;
197
- declare function isRetrievalSpan(s: Span): s is RetrievalSpan;
198
- declare function isJudgeSpan(s: Span): s is JudgeSpan;
199
- declare function isSandboxSpan(s: Span): s is SandboxSpan;
200
-
201
- export { type Artifact as A, type BudgetLedgerEntry as B, type EventKind as E, type FailureClass as F, type GenericSpan as G, type JudgeSpan as J, type LlmSpan as L, type Message as M, type Run as R, type Span as S, type ToolSpan as T, type RunOutcome as a, type SpanKind as b, type RetrievalSpan as c, type SandboxSpan as d, type TraceEvent as e, type RunStatus as f, type RunLayer as g, type BudgetSpec as h, FAILURE_CLASSES as i, type SpanBase as j, type SpanStatus as k, TRACE_SCHEMA_VERSION as l, isJudgeSpan as m, isLlmSpan as n, isRetrievalSpan as o, isSandboxSpan as p, isToolSpan as q };