@tangle-network/agent-eval 0.137.0 → 0.139.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (172) hide show
  1. package/CHANGELOG.md +78 -0
  2. package/README.md +34 -0
  3. package/dist/analyst/index.d.ts +485 -104
  4. package/dist/analyst/index.d.ts.map +1 -1
  5. package/dist/analyst/index.js +10 -607
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{benchmark-YDrpumqB.js → benchmark-CYtcIF2V.js} +299 -159
  8. package/dist/benchmark-CYtcIF2V.js.map +1 -0
  9. package/dist/{benchmark-CHX4orG7.d.ts → benchmark-DDVdWcwA.d.ts} +67 -15
  10. package/dist/benchmark-DDVdWcwA.d.ts.map +1 -0
  11. package/dist/benchmark-command-BKfjOBJ5.js +4537 -0
  12. package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
  13. package/dist/benchmarks/index.d.ts +1 -1
  14. package/dist/benchmarks/index.js +1 -1
  15. package/dist/{benchmarks-DCLkQOmc.js → benchmarks-zxhy1QV3.js} +4 -3
  16. package/dist/{benchmarks-DCLkQOmc.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
  17. package/dist/campaign/index.d.ts +5 -5
  18. package/dist/campaign/index.js +4 -3
  19. package/dist/{campaign-lgObcHFC.js → campaign-DrS6_hLd.js} +19 -11
  20. package/dist/campaign-DrS6_hLd.js.map +1 -0
  21. package/dist/canonical-D011XM8r.js +86 -0
  22. package/dist/canonical-D011XM8r.js.map +1 -0
  23. package/dist/cli.js +10 -3
  24. package/dist/cli.js.map +1 -1
  25. package/dist/{client-C8L6h6Wf.d.ts → client-BohnDFBq.d.ts} +4 -4
  26. package/dist/{client-C8L6h6Wf.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
  27. package/dist/{completion-verifier-DSyRNVzU.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
  28. package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
  29. package/dist/contract/index.d.ts +10 -10
  30. package/dist/contract/index.js +9 -8
  31. package/dist/contract/index.js.map +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/{cost-ledger-D-5_-dhi.js → cost-ledger-CZ9diLxY.js} +90 -45
  34. package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
  35. package/dist/{cost-ledger-D2o6JOrL.d.ts → cost-ledger-DKgyIWRj.d.ts} +6 -2
  36. package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
  37. package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
  38. package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
  39. package/dist/default-registry-BgJJItGr.js +2364 -0
  40. package/dist/default-registry-BgJJItGr.js.map +1 -0
  41. package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
  42. package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
  43. package/dist/{eval-campaign-CHqfLnff.js → eval-campaign-BmptJj50.js} +2 -2
  44. package/dist/{eval-campaign-CHqfLnff.js.map → eval-campaign-BmptJj50.js.map} +1 -1
  45. package/dist/exact-types-MaaFcllV.d.ts +234 -0
  46. package/dist/exact-types-MaaFcllV.d.ts.map +1 -0
  47. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
  48. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
  49. package/dist/{extract-usage-p-56bh8q.js → extract-usage-DZs601Va.js} +2 -2
  50. package/dist/{extract-usage-p-56bh8q.js.map → extract-usage-DZs601Va.js.map} +1 -1
  51. package/dist/{feedback-trajectory-N_F0PwHz.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -2
  52. package/dist/feedback-trajectory-BJUWOkJM.d.ts.map +1 -0
  53. package/dist/fuzz.d.ts +1 -1
  54. package/dist/fuzz.js +1 -1
  55. package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
  56. package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
  57. package/dist/hosted/index.d.ts +3 -3
  58. package/dist/{index-C-Pr4OWg.d.ts → index-BTm_P9aC.d.ts} +12 -11
  59. package/dist/index-BTm_P9aC.d.ts.map +1 -0
  60. package/dist/{index-U3RHOShi.d.ts → index-CWOPCJiw.d.ts} +2 -2
  61. package/dist/{index-U3RHOShi.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
  62. package/dist/{index-DRNl6g_N.d.ts → index-CtR1xh4V.d.ts} +3 -3
  63. package/dist/{index-DRNl6g_N.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
  64. package/dist/index-DEb46kc6.d.ts.map +1 -1
  65. package/dist/{index-BnP1QJUv.d.ts → index-_66rVpwN.d.ts} +5 -5
  66. package/dist/{index-BnP1QJUv.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
  67. package/dist/index.d.ts +35 -55
  68. package/dist/index.d.ts.map +1 -1
  69. package/dist/index.js +55 -514
  70. package/dist/index.js.map +1 -1
  71. package/dist/{insight-report-B9ooYH_g.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
  72. package/dist/{insight-report-B9ooYH_g.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
  73. package/dist/{integrity-CKxosZ5Z.d.ts → integrity-COTh3DTH.d.ts} +2 -2
  74. package/dist/{integrity-CKxosZ5Z.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
  75. package/dist/kind-factory-CFxA0JQX.js +2133 -0
  76. package/dist/kind-factory-CFxA0JQX.js.map +1 -0
  77. package/dist/ledger-core/index.js +2 -1
  78. package/dist/{ledger-core-t6sItivm.js → ledger-core-Dxz0Rkwa.js} +210 -99
  79. package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
  80. package/dist/{llm-client-DKB25jV8.js → llm-client-bkztEfIx.js} +5 -5
  81. package/dist/llm-client-bkztEfIx.js.map +1 -0
  82. package/dist/meta-eval/index.d.ts +2 -2
  83. package/dist/multishot/index.d.ts +2 -2
  84. package/dist/openapi.json +1 -1
  85. package/dist/{proposal-findings-DCawte-y.js → proposal-findings-2GIUo1et.js} +2 -68
  86. package/dist/proposal-findings-2GIUo1et.js.map +1 -0
  87. package/dist/{release-report-CofgVNZt.d.ts → release-report-fZarvIm-.d.ts} +3 -3
  88. package/dist/{release-report-CofgVNZt.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
  89. package/dist/{replay-K8FaC0CB.d.ts → replay-DjG4IG60.d.ts} +34 -143
  90. package/dist/replay-DjG4IG60.d.ts.map +1 -0
  91. package/dist/{replay-Bju0T8Ls.js → replay-SA4OB7O7.js} +48 -136
  92. package/dist/replay-SA4OB7O7.js.map +1 -0
  93. package/dist/reporting.d.ts +4 -4
  94. package/dist/{researcher-Da0Wj-bt.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
  95. package/dist/{researcher-Da0Wj-bt.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
  96. package/dist/{reward-hacking-CQ3hTCO3.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
  97. package/dist/{reward-hacking-CQ3hTCO3.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
  98. package/dist/rl.d.ts +5 -5
  99. package/dist/rl.js +1 -1
  100. package/dist/rollout/index.d.ts +1 -1
  101. package/dist/rollout/index.js +2 -2
  102. package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
  103. package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
  104. package/dist/{rubric-predictive-validity-C4sztLR3.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
  105. package/dist/{rubric-predictive-validity-C4sztLR3.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
  106. package/dist/{run-evidence-BDIircdA.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
  107. package/dist/{run-evidence-BDIircdA.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
  108. package/dist/{run-record-BPCa2rQ8.d.ts → run-record-CztDMXVF.d.ts} +2 -2
  109. package/dist/{run-record-BPCa2rQ8.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
  110. package/dist/{semantic-concept-judge-Bz64IckK.js → semantic-concept-judge-BuIJ9IfB.js} +49 -6
  111. package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
  112. package/dist/{server-KjXZZUDX.js → server-DaCpLfi0.js} +3 -3
  113. package/dist/{server-KjXZZUDX.js.map → server-DaCpLfi0.js.map} +1 -1
  114. package/dist/single-run-lock-BTTtPZ9N.js +989 -0
  115. package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
  116. package/dist/{skill-usage-CFDLLlhF.d.ts → skill-usage-B-BFS8M2.d.ts} +65 -40
  117. package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
  118. package/dist/{skillopt-optimization-method-f4o9sUT4.js → skillopt-optimization-method-BbGnCC53.js} +20 -979
  119. package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
  120. package/dist/{skillopt-optimization-method-BpbnlvAZ.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
  121. package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
  122. package/dist/{statistics-_7P642CN.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
  123. package/dist/{statistics-_7P642CN.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
  124. package/dist/store-otlp-DX4fGIcf.js +757 -0
  125. package/dist/store-otlp-DX4fGIcf.js.map +1 -0
  126. package/dist/{summary-report-DHipz9Kx.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
  127. package/dist/{summary-report-DHipz9Kx.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
  128. package/dist/tool-groups-CdYq22lX.d.ts +258 -0
  129. package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
  130. package/dist/traces.d.ts +7 -6
  131. package/dist/traces.js +5 -4
  132. package/dist/{types-CTGbIm57.d.ts → types-BBFNHxSK.d.ts} +5 -5
  133. package/dist/{types-CTGbIm57.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
  134. package/dist/{types-CKswbJGO.d.ts → types-DoEYskCd.d.ts} +5 -5
  135. package/dist/{types-CKswbJGO.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
  136. package/dist/{types-CTvKfr5F.d.ts → types-uPrS6mD-.d.ts} +2 -2
  137. package/dist/{types-CTvKfr5F.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
  138. package/dist/usage-receipt-CgxMEBZq.js +134 -0
  139. package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
  140. package/dist/wire/index.d.ts +3 -3
  141. package/dist/wire/index.js +1 -1
  142. package/docs/trace-analysis.md +191 -385
  143. package/package.json +5 -4
  144. package/dist/analyze-runs-PVtnfjvA.d.ts +0 -72
  145. package/dist/analyze-runs-PVtnfjvA.d.ts.map +0 -1
  146. package/dist/benchmark-CHX4orG7.d.ts.map +0 -1
  147. package/dist/benchmark-YDrpumqB.js.map +0 -1
  148. package/dist/campaign-lgObcHFC.js.map +0 -1
  149. package/dist/completion-verifier-DSyRNVzU.d.ts.map +0 -1
  150. package/dist/concurrency-MUjT7VjM.js +0 -109
  151. package/dist/concurrency-MUjT7VjM.js.map +0 -1
  152. package/dist/cost-ledger-D-5_-dhi.js.map +0 -1
  153. package/dist/cost-ledger-D2o6JOrL.d.ts.map +0 -1
  154. package/dist/default-registry-CLXbRt0f.js +0 -2594
  155. package/dist/default-registry-CLXbRt0f.js.map +0 -1
  156. package/dist/default-registry-Dc5D_Loc.d.ts +0 -202
  157. package/dist/default-registry-Dc5D_Loc.d.ts.map +0 -1
  158. package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +0 -1
  159. package/dist/index-C-Pr4OWg.d.ts.map +0 -1
  160. package/dist/ledger-core-t6sItivm.js.map +0 -1
  161. package/dist/llm-client-DKB25jV8.js.map +0 -1
  162. package/dist/proposal-findings-DCawte-y.js.map +0 -1
  163. package/dist/registry-BdM7SuTr.d.ts +0 -124
  164. package/dist/registry-BdM7SuTr.d.ts.map +0 -1
  165. package/dist/replay-Bju0T8Ls.js.map +0 -1
  166. package/dist/replay-K8FaC0CB.d.ts.map +0 -1
  167. package/dist/semantic-concept-judge-Bz64IckK.js.map +0 -1
  168. package/dist/skill-usage-CFDLLlhF.d.ts.map +0 -1
  169. package/dist/skillopt-optimization-method-BpbnlvAZ.d.ts.map +0 -1
  170. package/dist/skillopt-optimization-method-f4o9sUT4.js.map +0 -1
  171. package/dist/tools-DZk2Jn64.js +0 -1876
  172. package/dist/tools-DZk2Jn64.js.map +0 -1
@@ -1,9 +1,12 @@
1
- import { A as FindingSubjectStringSchema, C as parseRawFinding, D as FINDING_SUBJECT_GRAMMAR_PROMPT, E as stripCodeFences, F as behavioralAnalyst, I as deriveEfficiencyFindings, M as findingSubjectGrammarPromptFor, N as parseFindingSubject, O as FINDING_SUBJECT_KINDS, P as renderFindingSubject, R as createChatClient, S as evidenceRefsFromRawFinding, T as coerceToFindingRows, _ as structureFindings, a as KNOWLEDGE_GAP_KIND_SPEC, b as RawAnalystEvidenceSchema, c as buildTraceToolsForGroup, d as emitControlIntegrityFindings, f as createTraceAnalystKind, g as validateUsageSettlementTimeout, h as settleUsageReceiptFromCostLedger, i as KNOWLEDGE_POISONING_KIND_SPEC, j as KIND_EXPECTED_SUBJECTS, k as FINDING_SUBJECT_SYNTAX, l as CONTROL_INTEGRITY_ANALYST, m as renderUpstreamFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as renderPriorFindings, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as ControlIntegrityAnalyst, v as ANALYST_SEVERITIES, w as coerceJson, x as RawAnalystFindingSchema, y as RAW_FINDING_SCHEMA_PROMPT, z as createAnalystAi } from "../default-registry-CLXbRt0f.js";
2
- import { i as CostLedger } from "../cost-ledger-D-5_-dhi.js";
3
- import { c as makeFinding, l as makeProposalFinding, n as isProposalFinding, s as computeFindingId, t as assertProposalFindings } from "../proposal-findings-DCawte-y.js";
4
- import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-Bz64IckK.js";
5
- import { E as pairedBootstrap } from "../statistics-ByxzSiOM.js";
6
- import { i as traceStoreEvidenceResolver, n as runAnalystBenchmark, r as scoreAnalystFindings, t as registryBenchmarkRunner } from "../benchmark-YDrpumqB.js";
1
+ import { i as CostLedger } from "../cost-ledger-CZ9diLxY.js";
2
+ import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-BgJJItGr.js";
3
+ import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, G as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, W as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-CFxA0JQX.js";
4
+ import { a as computeFindingId, i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as makeFinding, s as makeProposalFinding } from "../usage-receipt-CgxMEBZq.js";
5
+ import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-2GIUo1et.js";
6
+ import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-BuIJ9IfB.js";
7
+ import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-DTkVyDX-.js";
8
+ import { a as scoreAnalystFindings, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-CYtcIF2V.js";
9
+ import { A as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, B as summarizeAgentRxCalibration, C as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, D as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, E as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, F as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, G as normalizeAgentRxCategory, H as codeTracerPredictionsToFindings, I as ANALYST_BENCHMARK_MANIFEST_FILE, K as roundAgentRxStep, L as ANALYST_BENCHMARK_OBSERVATIONS_FILE, M as analystBenchmarkDependencyLockDigest, N as analystBenchmarkImplementationDigest, O as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, P as ANALYST_BENCHMARK_COST_LEDGER_FILE, R as AGENT_RX_UPSTREAM_REVISION, S as emptyPublicBenchmarkRunner, T as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, U as agentRxBenchmarkCase, V as codeTraceBenchCase, W as agentRxPredictionsToFindings, _ as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, a as renderCodeTraceCalibrationMarkdown, b as parseVerificationOutcome, c as createPublicBenchmarkRlmRunner, d as publicBenchmarkProtocolSha256, f as loadPublicBenchmarkRows, g as selectPublicBenchmarkRows, h as publicBenchmarkSelectionReport, i as readAnalystBenchmarkArtifact, j as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, k as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, l as CODE_TRACE_BENCH_ANALYST_PROMPT, m as publicBenchmarkDistributions, n as runAnalystBenchmarkCommand, o as summarizeCodeTraceCalibration, p as preparePublicAnalystBenchmark, q as normalizeBenchmarkLabel, r as renderAnalystBenchmarkMarkdown, s as compareAnalystRunners, t as ANALYST_BENCHMARK_HELP, u as createPublicBenchmarkDirectRunner, v as appendVerificationArtifactsToOtlp, w as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, x as adaptPublicBenchmarkFindings, y as loadCodeTraceVerificationArtifacts, z as renderAgentRxCalibrationMarkdown } from "../benchmark-command-BKfjOBJ5.js";
7
10
  //#region src/analyst/adapters.ts
8
11
  /**
9
12
  * Adapter factories — lift each existing agent-eval primitive into the
@@ -291,606 +294,6 @@ function createSemanticConceptJudgeAdapter(opts = {}) {
291
294
  };
292
295
  }
293
296
  //#endregion
294
- //#region src/analyst/benchmark-comparison.ts
295
- function compareAnalystRunners(result, options) {
296
- const confidence = options.confidence ?? .95;
297
- const resamples = options.resamples ?? 2e3;
298
- assertComparisonControls(confidence, resamples);
299
- const runnerIds = new Set(result.summaries.map((summary) => summary.runnerId));
300
- if (!runnerIds.has(options.baselineRunnerId)) throw new TypeError(`unknown baseline analyst runner '${options.baselineRunnerId}'`);
301
- if (!runnerIds.has(options.candidateRunnerId)) throw new TypeError(`unknown candidate analyst runner '${options.candidateRunnerId}'`);
302
- if (options.baselineRunnerId === options.candidateRunnerId) throw new TypeError("baseline and candidate analyst runners must be different");
303
- const baseline = observationsByCase(result.observations, options.baselineRunnerId);
304
- const candidate = observationsByCase(result.observations, options.candidateRunnerId);
305
- const metrics = [];
306
- for (const metric of METRICS) {
307
- const before = [];
308
- const after = [];
309
- let pairedObservations = 0;
310
- for (const [caseId, baselineObservations] of baseline) {
311
- const candidateObservations = candidate.get(caseId);
312
- if (!candidateObservations) continue;
313
- const candidateByRepetition = new Map(candidateObservations.map((observation) => [observation.repetition, observation]));
314
- const caseBefore = [];
315
- const caseAfter = [];
316
- for (const baselineObservation of baselineObservations) {
317
- const candidateObservation = candidateByRepetition.get(baselineObservation.repetition);
318
- if (!candidateObservation) continue;
319
- const baselineValue = metricValue(baselineObservation, metric);
320
- const candidateValue = metricValue(candidateObservation, metric);
321
- if (baselineValue === null || candidateValue === null) continue;
322
- caseBefore.push(baselineValue);
323
- caseAfter.push(candidateValue);
324
- }
325
- if (caseBefore.length === 0) continue;
326
- before.push(mean(caseBefore));
327
- after.push(mean(caseAfter));
328
- pairedObservations += caseBefore.length;
329
- }
330
- if (before.length === 0) continue;
331
- const interval = pairedBootstrap(before, after, {
332
- confidence,
333
- resamples,
334
- statistic: "mean",
335
- seed: options.seed
336
- });
337
- const comparison = {
338
- metric,
339
- direction: LOWER_IS_BETTER.has(metric) ? "lower" : "higher",
340
- pairedCases: interval.n,
341
- pairedObservations,
342
- baselineMean: mean(before),
343
- candidateMean: mean(after),
344
- meanDelta: interval.mean,
345
- intervalLow: interval.low,
346
- intervalHigh: interval.high,
347
- confidence: interval.confidence,
348
- resamples: interval.resamples,
349
- enoughCasesForInference: interval.gateEligible
350
- };
351
- assertValidComparison(comparison);
352
- metrics.push(comparison);
353
- }
354
- return {
355
- baselineRunnerId: options.baselineRunnerId,
356
- candidateRunnerId: options.candidateRunnerId,
357
- metrics
358
- };
359
- }
360
- const METRICS = [
361
- "completion",
362
- "issueRecall",
363
- "findingPrecision",
364
- "f1",
365
- "criticalStepAccuracy",
366
- "citationCoverage",
367
- "citationLabelAgreement",
368
- "citationResolution",
369
- "cleanAccuracy",
370
- "latencyMs",
371
- "calls",
372
- "inputTokens",
373
- "outputTokens",
374
- "reasoningTokens",
375
- "cachedTokens",
376
- "cacheWriteTokens",
377
- "costUsd"
378
- ];
379
- const LOWER_IS_BETTER = /* @__PURE__ */ new Set([
380
- "latencyMs",
381
- "calls",
382
- "inputTokens",
383
- "outputTokens",
384
- "reasoningTokens",
385
- "cachedTokens",
386
- "cacheWriteTokens",
387
- "costUsd"
388
- ]);
389
- function observationsByCase(observations, runnerId) {
390
- const byCase = /* @__PURE__ */ new Map();
391
- for (const observation of observations) {
392
- if (observation.runnerId !== runnerId) continue;
393
- const rows = byCase.get(observation.caseId) ?? [];
394
- rows.push(observation);
395
- byCase.set(observation.caseId, rows);
396
- }
397
- return byCase;
398
- }
399
- function metricValue(observation, metric) {
400
- if (metric === "completion") return observation.error ? 0 : 1;
401
- if (metric === "latencyMs") return observation.latencyMs;
402
- if (metric === "cleanAccuracy") {
403
- if (observation.score.expectedIssueCount !== 0) return null;
404
- if (observation.error) return 0;
405
- return observation.score.cleanFalsePositive ? 0 : 1;
406
- }
407
- if ((metric === "issueRecall" || metric === "findingPrecision" || metric === "f1") && observation.score.expectedIssueCount === 0) return null;
408
- if (observation.error && (metric === "issueRecall" || metric === "findingPrecision" || metric === "f1" || metric === "criticalStepAccuracy")) {
409
- if (metric === "criticalStepAccuracy" && observation.score.criticalStepAccuracy === null) return null;
410
- return 0;
411
- }
412
- if (observation.error && (metric === "citationCoverage" || metric === "citationLabelAgreement" || metric === "citationResolution")) return null;
413
- if (metric === "issueRecall") return observation.score.issueRecall;
414
- if (metric === "findingPrecision") return observation.score.findingPrecision;
415
- if (metric === "f1") return observation.score.f1;
416
- if (metric === "criticalStepAccuracy") return observation.score.criticalStepAccuracy;
417
- if (metric === "citationCoverage") return observation.score.citationCoverage;
418
- if (metric === "citationLabelAgreement") return observation.score.citationLabelAgreement;
419
- if (metric === "citationResolution") return observation.evidenceResolution?.validity ?? null;
420
- if (metric === "calls") return observation.usage?.calls ?? null;
421
- if (metric === "inputTokens") return observation.usage?.tokens?.input ?? null;
422
- if (metric === "outputTokens") return observation.usage?.tokens?.output ?? null;
423
- if (metric === "reasoningTokens") return observation.usage?.tokens?.reasoning ?? null;
424
- if (metric === "cachedTokens") return observation.usage?.tokens?.cached ?? null;
425
- if (metric === "cacheWriteTokens") return observation.usage?.tokens?.cacheWrite ?? null;
426
- if (observation.usage?.cost.kind === "uncaptured") return null;
427
- return observation.usage?.cost.usd ?? null;
428
- }
429
- function mean(values) {
430
- return values.reduce((sum, value) => sum + value, 0) / values.length;
431
- }
432
- function assertComparisonControls(confidence, resamples) {
433
- if (!Number.isSafeInteger(resamples) || resamples <= 0 || resamples > 1e6) throw new Error(`compareAnalystRunners: resamples must be a positive safe integer no greater than 1000000, got ${String(resamples)}`);
434
- if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`compareAnalystRunners: confidence must be a finite number in (0,1), got ${String(confidence)}`);
435
- }
436
- function assertValidComparison(comparison) {
437
- if ([
438
- "pairedCases",
439
- "pairedObservations",
440
- "baselineMean",
441
- "candidateMean",
442
- "meanDelta",
443
- "intervalLow",
444
- "intervalHigh",
445
- "confidence",
446
- "resamples"
447
- ].some((field) => !Number.isFinite(comparison[field]))) throw new Error(`compareAnalystRunners: ${comparison.metric} produced non-finite comparison output`);
448
- if (comparison.intervalLow > comparison.intervalHigh) throw new Error(`compareAnalystRunners: ${comparison.metric} produced an invalid confidence interval`);
449
- }
450
- //#endregion
451
- //#region src/analyst/benchmark-datasets.ts
452
- function agentRxBenchmarkCase(row, input, options = {}) {
453
- const trajectoryId = externalId(row.trajectory_id, "AgentRx trajectory_id");
454
- if (!Array.isArray(row.failures) || row.failures.length === 0) throw new TypeError(`AgentRx trajectory '${trajectoryId}' must contain failures`);
455
- if (row.num_failures !== void 0 && row.num_failures !== row.failures.length) throw new TypeError(`AgentRx trajectory '${trajectoryId}' declares ${row.num_failures} failures but contains ${row.failures.length}`);
456
- const rootCauseId = externalId(row.root_cause_failure_id ?? row.root_cause?.failure_id, `AgentRx trajectory '${trajectoryId}' root cause failure id`);
457
- const failureIds = /* @__PURE__ */ new Set();
458
- const evidenceKind = options.evidenceKind ?? "span";
459
- const uri = options.stepUri ?? defaultStepUri;
460
- const allIssues = row.failures.map((failure) => {
461
- const failureId = externalId(failure.failure_id, `AgentRx trajectory '${trajectoryId}' failure id`);
462
- if (failureIds.has(failureId)) throw new TypeError(`AgentRx trajectory '${trajectoryId}' repeats failure id '${failureId}'`);
463
- failureIds.add(failureId);
464
- const step = positiveStep(failure.step_number, `AgentRx trajectory '${trajectoryId}'`);
465
- const evidence = [{
466
- kind: evidenceKind,
467
- uri: uri(trajectoryId, step)
468
- }];
469
- return {
470
- id: failureId,
471
- areas: [normalizeAgentRxCategory(failure.failure_category)],
472
- ...failureId === rootCauseId && (options.target ?? "root-cause") === "root-cause" ? {} : { evidence },
473
- criticalEvidence: failureId === rootCauseId ? evidence : void 0
474
- };
475
- });
476
- if (!failureIds.has(rootCauseId)) throw new TypeError(`AgentRx trajectory '${trajectoryId}' root cause '${rootCauseId}' is not in failures`);
477
- if (options.stepCount !== void 0 && row.failures.some((failure) => failure.step_number > options.stepCount)) throw new RangeError(`AgentRx trajectory '${trajectoryId}' contains a failure beyond stepCount ${options.stepCount}`);
478
- const expectedIssues = (options.target ?? "root-cause") === "root-cause" ? allIssues.filter((issue) => issue.id === rootCauseId) : allIssues;
479
- return {
480
- id: `agentrx:${trajectoryId}`,
481
- input,
482
- expectedIssues,
483
- labeledEvidence: expectedIssues.flatMap((issue) => issue.evidence ?? issue.criticalEvidence ?? []),
484
- tags: ["agentrx"],
485
- metadata: {
486
- benchmark: "AgentRx",
487
- trajectoryId,
488
- failureSummary: row.failure_summary,
489
- rootCauseReason: row.root_cause_reason ?? row.root_cause?.reason_for_root_cause,
490
- annotatedFailures: row.failures.length,
491
- target: options.target ?? "root-cause"
492
- }
493
- };
494
- }
495
- function codeTraceBenchCase(row, input, options = {}) {
496
- const trajectoryId = nonEmpty(row.traj_id, "CodeTraceBench traj_id");
497
- const stepCount = positiveStep(row.step_count, `CodeTraceBench '${trajectoryId}' step_count`);
498
- const stages = parseCodeTraceStages(row.incorrect_stages, trajectoryId);
499
- const evidenceKind = options.evidenceKind ?? "span";
500
- const uri = options.stepUri ?? defaultStepUri;
501
- const labelSet = codeTraceLabelSet(options.labelSet);
502
- const labels = /* @__PURE__ */ new Set();
503
- const expectedIssues = stages.flatMap((stage) => {
504
- const incorrect = stepIssues("incorrect", stage.incorrect_step_ids ?? []);
505
- const unuseful = stepIssues("unuseful", stage.unuseful_step_ids ?? []);
506
- return labelSet === "incorrect-only" ? incorrect : [...incorrect, ...unuseful];
507
- });
508
- return {
509
- id: `codetrace:${trajectoryId}`,
510
- input,
511
- expectedIssues,
512
- labeledEvidence: expectedIssues.flatMap((issue) => issue.evidence ?? []),
513
- tags: [
514
- "codetracebench",
515
- row.agent,
516
- row.model,
517
- ...row.difficulty ? [row.difficulty] : [],
518
- ...row.category ? [row.category] : [],
519
- ...parseTags(row.tags, trajectoryId)
520
- ],
521
- metadata: {
522
- benchmark: "CodeTraceBench",
523
- trajectoryId,
524
- taskName: row.task_name,
525
- agent: row.agent,
526
- model: row.model,
527
- solved: row.solved,
528
- stepCount,
529
- labelSet
530
- }
531
- };
532
- function stepIssues(label, steps) {
533
- return steps.map((rawStep) => {
534
- const step = positiveStep(rawStep, `CodeTraceBench '${trajectoryId}' ${label} step`);
535
- if (step > stepCount) throw new RangeError(`CodeTraceBench '${trajectoryId}' ${label} step ${step} exceeds step_count ${stepCount}`);
536
- const id = `${label}:${step}`;
537
- if (labels.has(id)) throw new TypeError(`CodeTraceBench '${trajectoryId}' repeats label '${id}'`);
538
- labels.add(id);
539
- return {
540
- id,
541
- areas: [label],
542
- evidence: [{
543
- kind: evidenceKind,
544
- uri: uri(trajectoryId, step)
545
- }]
546
- };
547
- });
548
- }
549
- }
550
- /** Translate AgentRx `Report.to_dict()` output or its `failures` array into findings. */
551
- function agentRxPredictionsToFindings(trajectoryIdValue, output, options = {}) {
552
- const trajectoryId = externalId(trajectoryIdValue, "AgentRx prediction trajectory id");
553
- const parsed = parseAgentRxPredictions(output, trajectoryId);
554
- for (const prediction of parsed.predictions) {
555
- assertStepWithinRange(prediction.step_number, parsed.report?.trajectory_length, `AgentRx prediction '${trajectoryId}' report`);
556
- assertStepWithinRange(prediction.step_number, options.stepCount, `AgentRx prediction '${trajectoryId}'`);
557
- }
558
- const consensus = agentRxConsensus(parsed, trajectoryId);
559
- if (consensus.failureCase === 0) return [];
560
- const confidence = predictionConfidence(options.confidence);
561
- const uri = options.stepUri ?? defaultStepUri;
562
- assertStepWithinRange(consensus.step, options.stepCount, `AgentRx prediction '${trajectoryId}'`);
563
- const area = AGENT_RX_TAXONOMY.get(consensus.failureCase);
564
- return [makeFinding({
565
- analyst_id: options.analystId ?? "agentrx",
566
- produced_at: options.producedAt,
567
- area,
568
- subject: "root-cause",
569
- claim: `AgentRx classified step ${consensus.step} as ${area}.`,
570
- id_basis: `${area}:${consensus.step}`,
571
- rationale: consensus.representative.description,
572
- severity: "high",
573
- confidence,
574
- evidence_refs: [{
575
- kind: options.evidenceKind ?? "span",
576
- uri: uri(trajectoryId, consensus.step)
577
- }],
578
- metadata: {
579
- upstream: "AgentRx",
580
- failure_case: consensus.failureCase,
581
- step: consensus.step,
582
- step_mean: consensus.stepMean,
583
- judge_votes: parsed.predictions.length,
584
- consensus_votes: consensus.votes,
585
- category_agreement: consensus.votes / parsed.predictions.length,
586
- ...consensus.representative.checklist_reasoning === void 0 || consensus.representative.checklist_reasoning === null ? {} : { checklist_reasoning: consensus.representative.checklist_reasoning }
587
- }
588
- })];
589
- }
590
- /** Translate CodeTracer's `codetracer_labels.json` into shared findings. */
591
- function codeTracerPredictionsToFindings(trajectoryIdValue, predictions, options = {}) {
592
- const trajectoryId = nonEmpty(trajectoryIdValue, "CodeTracer prediction trajectory id");
593
- const stages = parseCodeTraceStages(predictions, trajectoryId);
594
- const confidence = predictionConfidence(options.confidence);
595
- const uri = options.stepUri ?? defaultStepUri;
596
- const labelSet = codeTraceLabelSet(options.labelSet);
597
- const seen = /* @__PURE__ */ new Set();
598
- const findings = [];
599
- for (const stage of stages) for (const [area, steps] of [["incorrect", stage.incorrect_step_ids ?? []], ["unuseful", stage.unuseful_step_ids ?? []]]) for (const rawStep of steps) {
600
- const step = positiveStep(rawStep, `CodeTracer prediction '${trajectoryId}' ${area} step`);
601
- assertStepWithinRange(step, options.stepCount, `CodeTracer prediction '${trajectoryId}'`);
602
- const key = `${area}:${step}`;
603
- if (seen.has(key)) throw new TypeError(`CodeTracer prediction '${trajectoryId}' repeats label '${key}'`);
604
- seen.add(key);
605
- if (area === "unuseful" && labelSet === "incorrect-only") continue;
606
- findings.push(makeFinding({
607
- analyst_id: options.analystId ?? "codetracer",
608
- produced_at: options.producedAt,
609
- area,
610
- subject: `step-${step}`,
611
- claim: `CodeTracer labeled step ${step} as ${area}.`,
612
- id_basis: key,
613
- rationale: stage.reasoning,
614
- severity: "medium",
615
- confidence,
616
- evidence_refs: [{
617
- kind: options.evidenceKind ?? "span",
618
- uri: uri(trajectoryId, step)
619
- }],
620
- metadata: {
621
- upstream: "CodeTracer",
622
- stage_id: stage.stage_id,
623
- step
624
- }
625
- }));
626
- }
627
- return findings;
628
- }
629
- function normalizeBenchmarkLabel(value) {
630
- const normalized = nonEmpty(value, "benchmark label").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
631
- if (!normalized) throw new TypeError("benchmark label must contain letters or digits");
632
- return normalized;
633
- }
634
- function parseCodeTraceStages(value, trajectoryId) {
635
- let parsed = value;
636
- if (typeof value === "string") try {
637
- parsed = JSON.parse(value);
638
- } catch (error) {
639
- throw new TypeError(`CodeTraceBench '${trajectoryId}' incorrect_stages is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
640
- }
641
- if (!Array.isArray(parsed)) throw new TypeError(`CodeTraceBench '${trajectoryId}' incorrect_stages must be an array`);
642
- for (const stage of parsed) if (!stage || typeof stage !== "object" || !Number.isSafeInteger(stage.stage_id) || stage.stage_id < 1 || !optionalStepArray(stage.incorrect_step_ids) || !optionalStepArray(stage.unuseful_step_ids) || stage.reasoning !== void 0 && typeof stage.reasoning !== "string") throw new TypeError(`CodeTraceBench '${trajectoryId}' contains an invalid stage annotation`);
643
- return parsed;
644
- }
645
- function parseTags(value, trajectoryId) {
646
- if (value === void 0) return [];
647
- let parsed = value;
648
- if (typeof value === "string") try {
649
- parsed = JSON.parse(value);
650
- } catch (error) {
651
- throw new TypeError(`CodeTraceBench '${trajectoryId}' tags is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
652
- }
653
- if (!Array.isArray(parsed) || !parsed.every((tag) => typeof tag === "string")) throw new TypeError(`CodeTraceBench '${trajectoryId}' tags must be strings`);
654
- return [...parsed];
655
- }
656
- const AGENT_RX_TAXONOMY = /* @__PURE__ */ new Map([
657
- [1, "instruction-plan-adherence-failure"],
658
- [2, "invention-of-new-information"],
659
- [3, "invalid-invocation"],
660
- [4, "misinterpretation-of-tool-output-handoff-failure"],
661
- [5, "intent-plan-misalignment"],
662
- [6, "underspecified-user-intent"],
663
- [7, "intent-not-supported"],
664
- [8, "guardrails-triggered"],
665
- [9, "system-failure"],
666
- [10, "inconclusive"]
667
- ]);
668
- const AGENT_RX_CATEGORY_ALIASES = /* @__PURE__ */ new Map([["instruction-adherence-failure", "instruction-plan-adherence-failure"], ["misinterpretation-of-tool-output", "misinterpretation-of-tool-output-handoff-failure"]]);
669
- const AGENT_RX_TAXONOMY_BY_LABEL = new Map([...AGENT_RX_TAXONOMY].map(([failureCase, label]) => [label, failureCase]));
670
- function normalizeAgentRxCategory(value) {
671
- const normalized = normalizeBenchmarkLabel(value);
672
- return AGENT_RX_CATEGORY_ALIASES.get(normalized) ?? normalized;
673
- }
674
- function parseAgentRxFailureCase(value, field) {
675
- if (typeof value !== "number" && typeof value !== "string") throw new TypeError(`${field} must be a taxonomy number or label`);
676
- if (typeof value === "string" && !/^\d+$/.test(value.trim())) {
677
- const normalized = normalizeAgentRxCategory(value);
678
- const failureCase = AGENT_RX_TAXONOMY_BY_LABEL.get(normalized);
679
- if (failureCase === void 0) throw new RangeError(`${field} '${value}' is not an AgentRx taxonomy label`);
680
- return failureCase;
681
- }
682
- const numeric = typeof value === "number" ? value : Number(value);
683
- if (!Number.isSafeInteger(numeric)) throw new TypeError(`${field} must be a taxonomy number or label`);
684
- if (numeric < 0 || numeric > 10) throw new RangeError(`${field} ${numeric} is outside 0-10`);
685
- return numeric;
686
- }
687
- function predictionConfidence(value) {
688
- const confidence = value ?? .5;
689
- if (!Number.isFinite(confidence) || confidence < 0 || confidence > 1) throw new RangeError("upstream prediction confidence must be between 0 and 1");
690
- return confidence;
691
- }
692
- function codeTraceLabelSet(value) {
693
- if (value === void 0 || value === "incorrect-only") return "incorrect-only";
694
- if (value === "incorrect-and-unuseful") return value;
695
- throw new TypeError("CodeTraceBench labelSet must be 'incorrect-only' or 'incorrect-and-unuseful'");
696
- }
697
- function assertStepWithinRange(step, stepCount, field) {
698
- if (stepCount === void 0) return;
699
- const count = positiveStep(stepCount, `${field} stepCount`);
700
- if (step > count) throw new RangeError(`${field} step ${step} exceeds stepCount ${count}`);
701
- }
702
- function parseAgentRxPredictions(output, trajectoryId) {
703
- let failures;
704
- let report;
705
- if (Array.isArray(output)) failures = output;
706
- else if (isRecord(output)) {
707
- assertMatchingAgentRxTaskId(output.task_id, trajectoryId, "report.task_id");
708
- if (!Object.hasOwn(output, "failures")) throw new TypeError(`AgentRx prediction '${trajectoryId}' report must contain failures`);
709
- failures = output.failures;
710
- if (output.num_judges !== void 0) {
711
- if (!Number.isSafeInteger(output.num_judges) || output.num_judges < 0) throw new RangeError(`AgentRx prediction '${trajectoryId}' report.num_judges must be a non-negative safe integer`);
712
- }
713
- if (output.trajectory_length !== void 0) positiveStep(output.trajectory_length, `AgentRx prediction '${trajectoryId}' report.trajectory_length`);
714
- if (output.step_mean !== void 0) {
715
- if (typeof output.step_mean !== "number" || !Number.isFinite(output.step_mean)) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.step_mean must be finite`);
716
- }
717
- if (output.modes !== void 0 && !Array.isArray(output.modes)) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.modes must be an array`);
718
- report = output;
719
- } else throw new TypeError(`AgentRx prediction '${trajectoryId}' must be a report or failures array`);
720
- if (!Array.isArray(failures)) throw new TypeError(`AgentRx prediction '${trajectoryId}' failures must be an array`);
721
- if (failures.length === 0) throw new TypeError(`AgentRx prediction '${trajectoryId}' failures must contain a judge prediction`);
722
- if (isRecord(output) && output.num_judges !== void 0 && output.num_judges !== failures.length) throw new TypeError(`AgentRx prediction '${trajectoryId}' declares ${output.num_judges} judges but contains ${failures.length} failures`);
723
- return {
724
- predictions: failures.map((value, index) => {
725
- const field = `AgentRx prediction '${trajectoryId}' failures[${index}]`;
726
- if (!isRecord(value)) throw new TypeError(`${field} must be an object`);
727
- assertMatchingAgentRxTaskId(value.task_id, trajectoryId, `${field}.task_id`);
728
- const failureCase = parseAgentRxFailureCase(value.failure_case, `${field}.failure_case`);
729
- if (!Number.isSafeInteger(value.step_number)) throw new TypeError(`${field}.step_number must be a safe integer`);
730
- const stepNumber = value.step_number;
731
- if (failureCase === 0 ? stepNumber !== 0 : stepNumber < 1) throw new RangeError(failureCase === 0 ? `${field}.step_number must be 0 when failure_case is 0` : `${field}.step_number must be positive when failure_case is 1-10`);
732
- if (value.description !== void 0 && typeof value.description !== "string") throw new TypeError(`${field}.description must be a string`);
733
- if (value.checklist_reasoning !== void 0 && value.checklist_reasoning !== null && typeof value.checklist_reasoning !== "string") throw new TypeError(`${field}.checklist_reasoning must be a string or null`);
734
- return {
735
- ...value.task_id === void 0 ? {} : { task_id: value.task_id },
736
- failure_case: failureCase,
737
- step_number: stepNumber,
738
- ...value.description === void 0 ? {} : { description: value.description },
739
- ...value.checklist_reasoning === void 0 ? {} : { checklist_reasoning: value.checklist_reasoning }
740
- };
741
- }),
742
- report
743
- };
744
- }
745
- function agentRxConsensus(parsed, trajectoryId) {
746
- const counts = /* @__PURE__ */ new Map();
747
- for (const prediction of parsed.predictions) counts.set(prediction.failure_case, (counts.get(prediction.failure_case) ?? 0) + 1);
748
- const maxVotes = Math.max(...counts.values());
749
- let failureCase = [...counts].find(([, count]) => count === maxVotes)[0];
750
- if (parsed.report?.most_common_failure !== void 0) {
751
- const declared = parseAgentRxFailureCase(parsed.report.most_common_failure, `AgentRx prediction '${trajectoryId}' report.most_common_failure`);
752
- if ((counts.get(declared) ?? 0) !== maxVotes) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.most_common_failure disagrees with failures`);
753
- failureCase = declared;
754
- }
755
- if (parsed.report?.modes !== void 0) {
756
- const declaredModes = parsed.report.modes.map((value, index) => parseAgentRxFailureCase(value, `AgentRx prediction '${trajectoryId}' report.modes[${index}]`));
757
- if (new Set(declaredModes).size !== declaredModes.length) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.modes contains duplicates`);
758
- const expectedModes = [...counts].filter(([, count]) => count === maxVotes).map(([value]) => value).sort((left, right) => left - right);
759
- if ([...new Set(declaredModes)].sort((left, right) => left - right).join(",") !== expectedModes.join(",")) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.modes disagrees with failures`);
760
- }
761
- const computedStepMean = parsed.predictions.reduce((sum, prediction) => sum + prediction.step_number, 0) / parsed.predictions.length;
762
- if (parsed.report?.step_mean !== void 0 && Math.abs(parsed.report.step_mean - computedStepMean) > 1e-12) throw new TypeError(`AgentRx prediction '${trajectoryId}' report.step_mean disagrees with failures`);
763
- const stepMean = parsed.report?.step_mean ?? computedStepMean;
764
- const step = failureCase === 0 ? 0 : positiveStep(roundHalfToEven(stepMean), `AgentRx prediction '${trajectoryId}' consensus step`);
765
- const representative = parsed.predictions.filter((prediction) => prediction.failure_case === failureCase).sort((left, right) => Math.abs(left.step_number - step) - Math.abs(right.step_number - step))[0] ?? parsed.predictions[0];
766
- return {
767
- failureCase,
768
- step,
769
- stepMean,
770
- votes: counts.get(failureCase),
771
- representative
772
- };
773
- }
774
- function roundHalfToEven(value) {
775
- const lower = Math.floor(value);
776
- const fraction = value - lower;
777
- if (Math.abs(fraction - .5) <= Number.EPSILON * Math.max(1, Math.abs(value))) return lower % 2 === 0 ? lower : lower + 1;
778
- return Math.round(value);
779
- }
780
- function assertMatchingAgentRxTaskId(value, trajectoryId, field) {
781
- if (value === void 0) return;
782
- const taskId = externalId(value, `AgentRx prediction '${trajectoryId}' ${field}`);
783
- if (taskId !== trajectoryId) throw new TypeError(`AgentRx prediction '${trajectoryId}' ${field} '${taskId}' does not match trajectory id`);
784
- }
785
- function defaultStepUri(trajectoryId, step) {
786
- return `trace://${encodeURIComponent(trajectoryId)}/span/step-${step}`;
787
- }
788
- function optionalStepArray(value) {
789
- return value === void 0 || Array.isArray(value) && value.every(Number.isSafeInteger);
790
- }
791
- function isRecord(value) {
792
- return typeof value === "object" && value !== null && !Array.isArray(value);
793
- }
794
- function positiveStep(value, field) {
795
- if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${field} must be a positive safe integer`);
796
- return value;
797
- }
798
- function externalId(value, field) {
799
- if (typeof value !== "string" && typeof value !== "number") throw new TypeError(`${field} must be a string or number`);
800
- if (typeof value === "number" && !Number.isSafeInteger(value)) throw new TypeError(`${field} must be a safe integer when numeric`);
801
- return nonEmpty(String(value), field);
802
- }
803
- function nonEmpty(value, field) {
804
- if (!value.trim()) throw new TypeError(`${field} must not be empty`);
805
- return value;
806
- }
807
- //#endregion
808
- //#region src/analyst/benchmark-report.ts
809
- function renderAnalystBenchmarkMarkdown(result, comparisons = []) {
810
- const { provenance } = result;
811
- const lines = [
812
- "# Trace analyst benchmark",
813
- "",
814
- "## Run",
815
- "",
816
- "| Field | Value |",
817
- "| --- | --- |",
818
- `| Benchmark | ${escapeCell(provenance.id ?? "unspecified")} |`,
819
- `| Dataset | ${escapeCell(provenance.dataset?.id ?? "unspecified")} |`,
820
- `| Dataset revision | ${escapeCell(provenance.dataset?.revision ?? "unspecified")} |`,
821
- `| Dataset split | ${escapeCell(provenance.dataset?.split ?? "unspecified")} |`,
822
- `| Started | ${escapeCell(provenance.startedAt)} |`,
823
- `| Ended | ${escapeCell(provenance.endedAt)} |`,
824
- `| Cases | ${provenance.caseCount} |`,
825
- `| Runners | ${escapeCell(provenance.runnerIds.join(", "))} |`,
826
- `| Repetitions | ${provenance.repetitions} |`,
827
- `| Maximum concurrency | ${provenance.maxConcurrency} |`,
828
- `| Runner-order seed | ${provenance.runnerOrderSeed} |`,
829
- `| Command | ${escapeCell(provenance.command ?? "uncaptured")} |`,
830
- `| Environment | ${escapeCell(json(provenance.environment))} |`,
831
- `| Metadata | ${escapeCell(json(provenance.metadata))} |`,
832
- "",
833
- "## Summary",
834
- ""
835
- ];
836
- lines.push("| Runner | Runs | Failed | Recall | Precision | F1 | Critical step | Citation coverage | Label-location agreement | Citation resolution | Resolution unknown runs | Unresolved citations | Resolution errors | Clean false positives | Clean failures | Repeat agreement | Latency min/mean/p50/p95/max ms | Calls | Input tokens | Output tokens | Reasoning tokens | Cached tokens | Cache-write tokens | Known cost USD | Unknown calls | Unknown input/output | Unknown reasoning | Unknown cached | Unknown cache-write | Unknown cost |", "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
837
- for (const summary of result.summaries) lines.push(`| ${escapeCell(summary.runnerId)} | ${summary.completedRuns}/${summary.plannedRuns} | ${summary.failedRuns} | ${optionalRate(summary.issueRecall)} | ${optionalRate(summary.findingPrecision)} | ${optionalRate(summary.f1)} | ${optionalRate(summary.criticalStepAccuracy)} | ${optionalRate(summary.citationCoverage)} | ${optionalRate(summary.citationLabelAgreement)} | ${optionalRate(summary.citationResolution)} | ${summary.citationResolutionUnknownRuns} | ${summary.unresolvedCitations} | ${summary.citationResolutionErrors} | ${optionalRate(summary.cleanCaseFalsePositiveRate)} | ${optionalRate(summary.cleanCaseFailureRate)} | ${optionalRate(summary.runAgreement)} | ${latency(summary.latencyMs)} | ${summary.calls} | ${summary.inputTokens} | ${summary.outputTokens} | ${summary.reasoningTokens} | ${summary.cachedTokens} | ${summary.cacheWriteTokens} | ${summary.knownCostUsd.toFixed(6)} | ${summary.callsUnknownRuns} | ${summary.tokenUsageUnknownRuns} | ${summary.reasoningTokenUsageUnknownRuns} | ${summary.cachedTokenUsageUnknownRuns} | ${summary.cacheWriteTokenUsageUnknownRuns} | ${summary.costUnknownRuns} |`);
838
- for (const comparison of comparisons) {
839
- lines.push("", `## ${escapeCell(comparison.candidateRunnerId)} compared with ${escapeCell(comparison.baselineRunnerId)}`, "", "| Metric | Better direction | Paired cases | Paired observations | Baseline mean | Candidate mean | Delta | Interval | At least 20 independent cases |", "| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | --- |");
840
- for (const metric of comparison.metrics) lines.push(`| ${metric.metric} | ${metric.direction} | ${metric.pairedCases} | ${metric.pairedObservations} | ${number(metric.baselineMean)} | ${number(metric.candidateMean)} | ${signed(metric.meanDelta)} | [${number(metric.intervalLow)}, ${number(metric.intervalHigh)}] | ${metric.enoughCasesForInference ? "yes" : "no"} |`);
841
- }
842
- lines.push("", "## Runs", "", "| Runner | Case | Tags | Case metadata | Runner metadata | Rep | Execution index | Completed | Recall | Precision | F1 | Critical step | Citation coverage | Label-location agreement | Citation resolution | Clean false positive | Findings | Unlabeled citations | Unresolved citations | Resolution errors | Latency ms | Calls | Input tokens | Output tokens | Reasoning tokens | Cached tokens | Cache-write tokens | Cost USD | Known cost USD | Cost source | Error class | Error |", "| --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | --- | --- | --- |");
843
- for (const observation of result.observations) {
844
- const usage = observation.usage;
845
- const cost = usage?.cost.kind === "uncaptured" ? null : usage?.cost.usd;
846
- lines.push(`| ${escapeCell(observation.runnerId)} | ${escapeCell(observation.caseId)} | ${escapeCell(observation.caseTags.join(", "))} | ${escapeCell(json(observation.caseMetadata))} | ${escapeCell(json(observation.runnerMetadata))} | ${observation.repetition} | ${observation.executionIndex} | ${observation.error ? "no" : "yes"} | ${rate(observation.score.issueRecall)} | ${rate(observation.score.findingPrecision)} | ${rate(observation.score.f1)} | ${optionalRate(observation.score.criticalStepAccuracy)} | ${optionalRate(observation.score.citationCoverage)} | ${optionalRate(observation.score.citationLabelAgreement)} | ${optionalRate(observation.evidenceResolution?.validity ?? null)} | ${observation.score.expectedIssueCount === 0 ? observation.score.cleanFalsePositive ? "yes" : "no" : "n/a"} | ${observation.score.supportedFindingIndexes.length}/${observation.findings.length} | ${observation.score.unlabeledEvidence.length} | ${observation.evidenceResolution?.unresolvedEvidence.length ?? "unknown"} | ${observation.evidenceResolution?.errors.length ?? "unknown"} | ${number(observation.latencyMs)} | ${usage?.calls ?? "unknown"} | ${usage?.tokens?.input ?? "unknown"} | ${usage?.tokens?.output ?? "unknown"} | ${usage?.tokens?.reasoning ?? "unknown"} | ${usage?.tokens?.cached ?? "unknown"} | ${usage?.tokens?.cacheWrite ?? "unknown"} | ${cost === null || cost === void 0 ? "unknown" : cost.toFixed(6)} | ${usage?.knownCostUsd?.toFixed(6) ?? (cost === null || cost === void 0 ? "unknown" : cost.toFixed(6))} | ${usage?.cost.kind ?? "unknown"} | ${escapeCell(observation.error?.class ?? "")} | ${escapeCell(observation.error?.message ?? "")} |`);
847
- }
848
- return `${lines.join("\n")}\n`;
849
- }
850
- function rate(value) {
851
- return `${(value * 100).toFixed(1)}%`;
852
- }
853
- function optionalRate(value) {
854
- return value === null ? "n/a" : rate(value);
855
- }
856
- function number(value) {
857
- return Number.isInteger(value) ? String(value) : value.toFixed(3);
858
- }
859
- function signed(value) {
860
- return `${value >= 0 ? "+" : ""}${number(value)}`;
861
- }
862
- function latency(value) {
863
- return [
864
- value.min,
865
- value.mean,
866
- value.p50,
867
- value.p95,
868
- value.max
869
- ].map(number).join("/");
870
- }
871
- function json(value) {
872
- return value === void 0 ? "uncaptured" : JSON.stringify(value);
873
- }
874
- function escapeCell(value) {
875
- return value.replaceAll("|", "\\|").replaceAll("\n", " ");
876
- }
877
- //#endregion
878
- //#region src/analyst/define.ts
879
- /** Define a custom trace analyst without repeating fixed registry fields. */
880
- function defineTraceAnalyst(options) {
881
- if (!options.id.trim()) throw new TypeError("defineTraceAnalyst: id must not be empty");
882
- if (!options.description.trim()) throw new TypeError("defineTraceAnalyst: description must not be empty");
883
- if (options.cost === void 0) throw new TypeError("defineTraceAnalyst: cost must be declared");
884
- return {
885
- id: options.id,
886
- description: options.description,
887
- version: options.version ?? "1.0.0",
888
- inputKind: "trace-store",
889
- cost: options.cost,
890
- analyze: options.analyze
891
- };
892
- }
893
- //#endregion
894
- export { ANALYST_SEVERITIES, AnalystRegistry, CONTROL_INTEGRITY_ANALYST, ControlIntegrityAnalyst, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, agentRxBenchmarkCase, agentRxPredictionsToFindings, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, emitControlIntegrityFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, registryBenchmarkRunner, renderAnalystBenchmarkMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, runAnalystBenchmark, scoreAnalystFindings, stripCodeFences, structureFindings, traceStoreEvidenceResolver };
297
+ export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, AnalystRegistry, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, ControlIntegrityAnalyst, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, ExactAnalystRunExecutionError, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkSelectionReport, readAnalystBenchmarkArtifact, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
895
298
 
896
299
  //# sourceMappingURL=index.js.map