@tangle-network/agent-eval 0.144.9 → 0.144.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/CHANGELOG.md +18 -0
  2. package/README.md +6 -0
  3. package/dist/analyst/index.d.ts +1 -1
  4. package/dist/analyst/index.js +7 -7
  5. package/dist/{benchmark-B181aMF9.js → benchmark-CWeqGl7x.js} +2 -2
  6. package/dist/{benchmark-B181aMF9.js.map → benchmark-CWeqGl7x.js.map} +1 -1
  7. package/dist/{benchmark-command-Dtym4pA-.js → benchmark-command-BteMFN62.js} +8 -8
  8. package/dist/{benchmark-command-Dtym4pA-.js.map → benchmark-command-BteMFN62.js.map} +1 -1
  9. package/dist/benchmarks/index.d.ts +1 -1
  10. package/dist/benchmarks/index.js +1 -1
  11. package/dist/{benchmarks-Ce4Ae5fX.js → benchmarks-Dzs8CKb1.js} +4 -4
  12. package/dist/{benchmarks-Ce4Ae5fX.js.map → benchmarks-Dzs8CKb1.js.map} +1 -1
  13. package/dist/campaign/index.d.ts +4 -4
  14. package/dist/campaign/index.js +5 -5
  15. package/dist/{campaign-FG3thH2m.js → campaign-C2TTzQII.js} +4 -4
  16. package/dist/{campaign-FG3thH2m.js.map → campaign-C2TTzQII.js.map} +1 -1
  17. package/dist/cli.js +1 -1
  18. package/dist/contract/index.d.ts +1 -1
  19. package/dist/contract/index.js +6 -6
  20. package/dist/{default-registry-BaQXW1Ow.js → default-registry-BmktKy8r.js} +3 -3
  21. package/dist/{default-registry-BaQXW1Ow.js.map → default-registry-BmktKy8r.js.map} +1 -1
  22. package/dist/{dspy-rlm-engine-BiN49gK6.js → dspy-rlm-engine-DbTk4JdR.js} +3 -3
  23. package/dist/{dspy-rlm-engine-BiN49gK6.js.map → dspy-rlm-engine-DbTk4JdR.js.map} +1 -1
  24. package/dist/{extract-usage-BW27f3XW.js → extract-usage-CdZdoj1s.js} +2 -2
  25. package/dist/{extract-usage-BW27f3XW.js.map → extract-usage-CdZdoj1s.js.map} +1 -1
  26. package/dist/{index-CBEMKO8d.d.ts → index-DPPGNJ_R.d.ts} +2 -2
  27. package/dist/{index-CBEMKO8d.d.ts.map → index-DPPGNJ_R.d.ts.map} +1 -1
  28. package/dist/{index-CLKMJTJF2.d.ts → index-YE4KdKbO2.d.ts} +3 -3
  29. package/dist/{index-CLKMJTJF2.d.ts.map → index-YE4KdKbO2.d.ts.map} +1 -1
  30. package/dist/index.d.ts +3 -3
  31. package/dist/index.js +11 -11
  32. package/dist/{kind-factory-BHIgPmzS.js → kind-factory-B8-r8-y8.js} +3 -16
  33. package/dist/kind-factory-B8-r8-y8.js.map +1 -0
  34. package/dist/openapi.json +1 -1
  35. package/dist/{replay-B3gACG_H.js → replay-CohS93nE.js} +4 -4
  36. package/dist/{replay-B3gACG_H.js.map → replay-CohS93nE.js.map} +1 -1
  37. package/dist/{semantic-concept-judge-Bmrq6yqU.js → semantic-concept-judge-D1z-KepS.js} +2 -2
  38. package/dist/{semantic-concept-judge-Bmrq6yqU.js.map → semantic-concept-judge-D1z-KepS.js.map} +1 -1
  39. package/dist/{single-run-lock-BMQEv1wG.js → single-run-lock-DFWHEB09.js} +123 -123
  40. package/dist/{single-run-lock-BMQEv1wG.js.map → single-run-lock-DFWHEB09.js.map} +1 -1
  41. package/dist/{skillopt-optimization-method-Dtzp7QpP.js → skillopt-optimization-method-CQdVeM8k.js} +513 -134
  42. package/dist/skillopt-optimization-method-CQdVeM8k.js.map +1 -0
  43. package/dist/{skillopt-optimization-method-D8wysvU1.d.ts → skillopt-optimization-method-USDKhxSA.d.ts} +87 -3
  44. package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +1 -0
  45. package/dist/{store-otlp-CKtTpRhv.js → store-otlp-Dw8PPIlL.js} +2 -2
  46. package/dist/{store-otlp-CKtTpRhv.js.map → store-otlp-Dw8PPIlL.js.map} +1 -1
  47. package/dist/traces.js +4 -4
  48. package/dist/{usage-receipt-EVI8B8Xu.js → usage-receipt-t7vAzCRQ.js} +15 -2
  49. package/dist/usage-receipt-t7vAzCRQ.js.map +1 -0
  50. package/docs/campaign-proposers.md +4 -0
  51. package/package.json +1 -1
  52. package/dist/kind-factory-BHIgPmzS.js.map +0 -1
  53. package/dist/skillopt-optimization-method-D8wysvU1.d.ts.map +0 -1
  54. package/dist/skillopt-optimization-method-Dtzp7QpP.js.map +0 -1
  55. package/dist/usage-receipt-EVI8B8Xu.js.map +0 -1
package/CHANGELOG.md CHANGED
@@ -20,6 +20,24 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
20
20
  - Blind statement-equivalence protocol (`defineEquivalenceCheck`, `buildEquivalenceRecord`, `runEquivalenceCheck`): the two-arm design as a typed primitive with fail-loud refusals (`EquivalenceProtocolError`) — a non-blind arm, a wrong arm count, a refutation without its separating witness, or a mismatched checker strategy throws instead of recording.
21
21
  - `docs/verification-strategies.md`: the family, each member's failure mode, and the BCWW (4.6) formalization pilot as the worked example.
22
22
 
23
+ ## [0.144.11] - 2026-08-10 - GEPA candidate graph
24
+
25
+ ### Added
26
+
27
+ - `readGepaCandidatePopulationArtifact()` returns GEPA's exact accepted candidates, parent indices, selection scores, and discovery counts from a verified artifact.
28
+ - Direct GEPA method provenance addresses the candidate graph and its configured population bounds, selection identities, and candidate surface kind.
29
+
30
+ ### Fixed
31
+
32
+ - The Python bridge preserves the official GEPA result graph from both published GEPA 0.1.4 and the pinned source API wrapper.
33
+
34
+ ## [0.144.10] - 2026-08-10 - Optimizer candidate population
35
+
36
+ ### Added
37
+
38
+ - `readExternalOptimizerObservationArtifact()` returns every distinct callback-submitted optimizer candidate after it verifies the addressed artifact's digest, canonical rows, sequence, candidate identities, and counts.
39
+ - `decodeExternalTextCandidate()` exposes Eval's existing canonical conversion from external text or named components to a mutable optimization surface.
40
+
23
41
  ## [0.144.9] - 2026-08-10 - Token usage completeness
24
42
 
25
43
  ### Fixed
package/README.md CHANGED
@@ -337,6 +337,12 @@ Read `scores` for final-case lift and intervals.
337
337
  Read `pairwise` before claiming one method beat another.
338
338
  Read `totalCost.accountingComplete` before using the reported dollars as a complete total.
339
339
  Each official method score records the optimizer and bridge package versions, source revisions and source-tree hashes, Python runtime, configured optimizer model when present, custom engine module hashes, compatible run ID, exact attempt ID, resume status, evaluation count, artifact directory, and available optimizer token usage in `provenance`.
340
+ Call `readExternalOptimizerObservationArtifact()` with `provenance.observations` to read every distinct callback-submitted candidate.
341
+ The reader verifies the artifact digest, canonical rows, sequence, candidate identities, and summary counts before returning candidates.
342
+ This verification proves that the bytes match the supplied summary; use a summary from trusted method provenance when authenticity matters.
343
+ For a direct standard GEPA run, call `readGepaCandidatePopulationArtifact()` with `provenance.gepaCandidatePopulation`.
344
+ It returns GEPA's accepted candidates with exact parent indices, aggregate scores, per-case selection scores, and discovery evaluation counts.
345
+ The callback artifact remains the complete source for rejected or refused proposals that GEPA did not add to its accepted population.
340
346
 
341
347
  The [optimizer guide](./docs/campaign-proposers.md) covers recipes, budgets, resuming, and data separation.
342
348
  The [runnable comparison](./examples/compare-optimization-methods/) can run GEPA, SkillOpt, or both.
@@ -1274,7 +1274,7 @@ declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the
1274
1274
  declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
1275
1275
  declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
1276
1276
  declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
1277
- declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "3768273281457e480bf76aadc6001978be31ad7856e27a38ea8b856a21f33607";
1277
+ declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "36d8f26723c19516348db0647de861f0ab1e672cb2da5aac8b3e34a1d6b49bdc";
1278
1278
  /** The published benchmark evidence was produced at this package version, by
1279
1279
  * the retired one-shot direct runner, before trace analysts moved to the
1280
1280
  * recursive DSPy RLM engine. Both evidence digests below are historical facts
@@ -1,12 +1,12 @@
1
1
  import { i as CostLedger } from "../cost-ledger-DMFxsLKr.js";
2
- import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-BaQXW1Ow.js";
3
- import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, G as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, W as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-BHIgPmzS.js";
4
- import { a as computeFindingId, i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as makeFinding, s as makeProposalFinding } from "../usage-receipt-EVI8B8Xu.js";
2
+ import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-BmktKy8r.js";
3
+ import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, U as DEFAULT_TRACE_ANALYST_LIMITS, W as resolveTraceAnalystLimits, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-B8-r8-y8.js";
4
+ import { c as makeProposalFinding, i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as computeFindingId, s as makeFinding } from "../usage-receipt-t7vAzCRQ.js";
5
5
  import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-2GIUo1et.js";
6
- import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-Bmrq6yqU.js";
7
- import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-BiN49gK6.js";
8
- import { a as scoreAnalystFindings, i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-B181aMF9.js";
9
- import { $ as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, A as summarizeCodeTraceCalibration, B as publicBenchmarkSystemPrompt, C as analystDefinitionAsymmetries, D as expandCodeTraceFailureBlocks, E as emptyPublicBenchmarkRunner, F as CODE_TRACE_BENCH_ANALYST_PROMPT, G as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, H as appendVerificationArtifactsToOtlp, I as MAX_INCORRECT_BLOCKS, J as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, K as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, L as MAX_INCORRECT_BLOCK_STEPS, M as analystInstructionsOverrideFromText, N as effectiveAnalystProtocolSha256, O as readAnalystBenchmarkArtifact, P as readAnalystInstructionsOverride, Q as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, R as publicBenchmarkProtocolSha256, S as AnalystExpressivenessError, T as adaptPublicBenchmarkFindings, U as loadCodeTraceVerificationArtifacts, V as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, W as parseVerificationOutcome, X as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, Y as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, Z as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, _ as runReplVariableAnalystDefinition, a as primeAnalystProtocolSha256, at as ANALYST_BENCHMARK_OBSERVATIONS_FILE, b as runChunkedAnalystDefinition, c as nodeHttpPrimeBridgeTransport, ct as summarizeAgentRxCalibration, d as publicBenchmarkDistributions, dt as agentRxBenchmarkCase, et as analystBenchmarkDependencyLockDigest, f as publicBenchmarkSelectionReport, ft as agentRxPredictionsToFindings, g as rlmEngineLimits, h as publicRlmAnalystDefinition, ht as normalizeBenchmarkLabel, i as createPrimeBenchmarkRunner, it as ANALYST_BENCHMARK_MANIFEST_FILE, j as compareAnalystRunners, k as renderCodeTraceCalibrationMarkdown, l as loadPublicBenchmarkRows, lt as codeTraceBenchCase, m as createPublicBenchmarkRlmRunner, mt as roundAgentRxStep, n as runAnalystBenchmarkCommand, nt as ANALYST_BENCHMARK_COST_LEDGER_FILE, o as primeCodeTraceAnalystDefinition, ot as AGENT_RX_UPSTREAM_REVISION, p as selectPublicBenchmarkRows, pt as normalizeAgentRxCategory, q as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, r as renderAnalystBenchmarkMarkdown, rt as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, s as runInlineAnalystDefinition, st as renderAgentRxCalibrationMarkdown, t as ANALYST_BENCHMARK_HELP, tt as analystBenchmarkImplementationDigest, u as preparePublicAnalystBenchmark, ut as codeTracerPredictionsToFindings, v as createPublicBenchmarkDirectRunner, w as analystDefinitionProtocolSha256, x as decodeReplyRows, y as publicDirectAnalystDefinition, z as publicBenchmarkRlmInstructions } from "../benchmark-command-Dtym4pA-.js";
6
+ import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-D1z-KepS.js";
7
+ import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-DbTk4JdR.js";
8
+ import { a as scoreAnalystFindings, i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-CWeqGl7x.js";
9
+ import { $ as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, A as summarizeCodeTraceCalibration, B as publicBenchmarkSystemPrompt, C as analystDefinitionAsymmetries, D as expandCodeTraceFailureBlocks, E as emptyPublicBenchmarkRunner, F as CODE_TRACE_BENCH_ANALYST_PROMPT, G as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, H as appendVerificationArtifactsToOtlp, I as MAX_INCORRECT_BLOCKS, J as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, K as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, L as MAX_INCORRECT_BLOCK_STEPS, M as analystInstructionsOverrideFromText, N as effectiveAnalystProtocolSha256, O as readAnalystBenchmarkArtifact, P as readAnalystInstructionsOverride, Q as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, R as publicBenchmarkProtocolSha256, S as AnalystExpressivenessError, T as adaptPublicBenchmarkFindings, U as loadCodeTraceVerificationArtifacts, V as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, W as parseVerificationOutcome, X as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, Y as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, Z as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, _ as runReplVariableAnalystDefinition, a as primeAnalystProtocolSha256, at as ANALYST_BENCHMARK_OBSERVATIONS_FILE, b as runChunkedAnalystDefinition, c as nodeHttpPrimeBridgeTransport, ct as summarizeAgentRxCalibration, d as publicBenchmarkDistributions, dt as agentRxBenchmarkCase, et as analystBenchmarkDependencyLockDigest, f as publicBenchmarkSelectionReport, ft as agentRxPredictionsToFindings, g as rlmEngineLimits, h as publicRlmAnalystDefinition, ht as normalizeBenchmarkLabel, i as createPrimeBenchmarkRunner, it as ANALYST_BENCHMARK_MANIFEST_FILE, j as compareAnalystRunners, k as renderCodeTraceCalibrationMarkdown, l as loadPublicBenchmarkRows, lt as codeTraceBenchCase, m as createPublicBenchmarkRlmRunner, mt as roundAgentRxStep, n as runAnalystBenchmarkCommand, nt as ANALYST_BENCHMARK_COST_LEDGER_FILE, o as primeCodeTraceAnalystDefinition, ot as AGENT_RX_UPSTREAM_REVISION, p as selectPublicBenchmarkRows, pt as normalizeAgentRxCategory, q as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, r as renderAnalystBenchmarkMarkdown, rt as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, s as runInlineAnalystDefinition, st as renderAgentRxCalibrationMarkdown, t as ANALYST_BENCHMARK_HELP, tt as analystBenchmarkImplementationDigest, u as preparePublicAnalystBenchmark, ut as codeTracerPredictionsToFindings, v as createPublicBenchmarkDirectRunner, w as analystDefinitionProtocolSha256, x as decodeReplyRows, y as publicDirectAnalystDefinition, z as publicBenchmarkRlmInstructions } from "../benchmark-command-BteMFN62.js";
10
10
  import { a as extractPrimeJsonObject, c as primeProtocolSha256, d as runPrimeExchange, i as emptyPrimeRawUsage, l as primeReplyDefect, n as buildPrimePrompt, o as mergePrimeRawUsage, r as buildPrimeRepairPrompt, s as normalizePrimeUsage, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "../prime-protocol-BfSalTfR.js";
11
11
  //#region src/analyst/adapters.ts
12
12
  /**
@@ -1,4 +1,4 @@
1
- import { t as assertValidAnalystUsageReceipt } from "./usage-receipt-EVI8B8Xu.js";
1
+ import { t as assertValidAnalystUsageReceipt } from "./usage-receipt-t7vAzCRQ.js";
2
2
  import { performance } from "node:perf_hooks";
3
3
  import { linearSumAssignment } from "linear-sum-assignment";
4
4
  //#region src/analyst/benchmark-scoring.ts
@@ -551,4 +551,4 @@ function mergeRegistryUsage(result) {
551
551
  //#endregion
552
552
  export { scoreAnalystFindings as a, summarizeAnalystBenchmarkRunner as i, runAnalystBenchmark as n, traceStoreEvidenceResolver as r, registryBenchmarkRunner as t };
553
553
 
554
- //# sourceMappingURL=benchmark-B181aMF9.js.map
554
+ //# sourceMappingURL=benchmark-CWeqGl7x.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"benchmark-B181aMF9.js","names":[],"sources":["../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark-summary.ts","../src/analyst/benchmark.ts"],"sourcesContent":["import { linearSumAssignment } from 'linear-sum-assignment'\nimport type {\n AnalystBenchmarkCase,\n AnalystEvidenceExpectation,\n AnalystFindingScore,\n AnalystIssueExpectation,\n} from './benchmark'\nimport type { AnalystFinding, EvidenceRef } from './types'\n\nexport function scoreAnalystFindings(\n testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>,\n findings: readonly AnalystFinding[],\n): AnalystFindingScore {\n assertValidAnalystScoringCase(testCase)\n const matchedFindingByIssue = matchFindingsToIssues(testCase.expectedIssues, findings)\n const matchedIssueIds = testCase.expectedIssues\n .filter((_, index) => matchedFindingByIssue.has(index))\n .map((issue) => issue.id)\n const missedIssueIds = testCase.expectedIssues\n .filter((_, index) => !matchedFindingByIssue.has(index))\n .map((issue) => issue.id)\n const supportedFindingIndexes = new Set(matchedFindingByIssue.values())\n\n const unsupportedFindingIndexes = findings\n .map((_, index) => index)\n .filter((index) => !supportedFindingIndexes.has(index))\n const expectedIssueCount = testCase.expectedIssues.length\n const issueRecall = expectedIssueCount === 0 ? 1 : matchedIssueIds.length / expectedIssueCount\n const findingPrecision =\n findings.length === 0\n ? expectedIssueCount === 0\n ? 1\n : 0\n : supportedFindingIndexes.size / findings.length\n const f1 = harmonicMeanScore(findingPrecision, issueRecall)\n const allEvidence = findings.flatMap((finding) => finding.evidence_refs)\n\n const criticalIssues = testCase.expectedIssues.filter(\n (issue) => (issue.criticalEvidence?.length ?? 0) > 0,\n )\n const criticalHits = testCase.expectedIssues.filter((issue) => {\n if ((issue.criticalEvidence?.length ?? 0) === 0) return false\n return matchesEvidence(allEvidence, issue.criticalEvidence ?? [], 'any')\n }).length\n\n const findingsWithEvidence = findings.filter((finding) => finding.evidence_refs.length > 0).length\n const unlabeledEvidence = testCase.labeledEvidence\n ? allEvidence.filter(\n (ref) => !testCase.labeledEvidence!.some((expected) => evidenceMatches(ref, expected)),\n )\n : []\n\n return {\n expectedIssueCount,\n matchedIssueIds,\n missedIssueIds,\n supportedFindingIndexes: [...supportedFindingIndexes].sort((a, b) => a - b),\n unsupportedFindingIndexes,\n unlabeledEvidence,\n issueRecall,\n findingPrecision,\n f1,\n criticalStepAccuracy: criticalIssues.length === 0 ? null : criticalHits / criticalIssues.length,\n citationCoverage: findings.length === 0 ? null : findingsWithEvidence / findings.length,\n citationExcerptCoverage:\n allEvidence.length === 0\n ? null\n : allEvidence.filter((evidence) => Boolean(evidence.excerpt?.trim())).length /\n allEvidence.length,\n citationLabelAgreement:\n testCase.labeledEvidence === undefined\n ? null\n : allEvidence.length === 0\n ? findings.length === 0\n ? null\n : 0\n : (allEvidence.length - unlabeledEvidence.length) / allEvidence.length,\n predictionOnLabelEmptyCase: expectedIssueCount === 0 && findings.length > 0,\n }\n}\n\nexport function assertValidAnalystScoringCase(\n testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>,\n): void {\n if (!testCase.id.trim()) throw new TypeError('analyst benchmark case id must not be empty')\n const ids = new Set<string>()\n for (const issue of testCase.expectedIssues) {\n if (!issue.id.trim()) throw new TypeError(`${testCase.id}: expected issue id must not be empty`)\n if (ids.has(issue.id)) {\n throw new TypeError(`${testCase.id}: duplicate expected issue id '${issue.id}'`)\n }\n ids.add(issue.id)\n if (\n !issue.findingIds?.length &&\n !issue.areas?.length &&\n !issue.subjects?.length &&\n !issue.evidence?.length\n ) {\n throw new TypeError(\n `${testCase.id}/${issue.id}: expected issue must identify a finding by id, area, subject, or evidence`,\n )\n }\n }\n for (const ref of testCase.labeledEvidence ?? []) {\n if (!ref.uri.trim()) {\n throw new TypeError(`${testCase.id}: labeled evidence URI must not be empty`)\n }\n }\n}\n\nexport function harmonicMeanScore(a: number, b: number): number {\n return a + b === 0 ? 0 : (2 * a * b) / (a + b)\n}\n\nfunction matchFindingsToIssues(\n issues: readonly AnalystIssueExpectation[],\n findings: readonly AnalystFinding[],\n): Map<number, number> {\n if (issues.length === 0) return new Map()\n const cardinalityWeight = issues.length + 1\n const scores = issues.map((issue) => [\n ...findings.map((finding) => {\n if (!findingMatchesIssue(finding, issue)) return -1\n const criticalHit =\n (issue.criticalEvidence?.length ?? 0) > 0 &&\n matchesEvidence(finding.evidence_refs, issue.criticalEvidence ?? [], 'any')\n return cardinalityWeight + Number(criticalHit)\n }),\n ...Array.from({ length: issues.length }, () => 0),\n ])\n const assignment = linearSumAssignment(scores, { maximaze: true }).rowAssignments\n const matches = new Map<number, number>()\n for (const [issueIndex, column] of assignment.entries()) {\n if (column < 0 || column >= findings.length) continue\n if (!findingMatchesIssue(findings[column]!, issues[issueIndex]!)) continue\n matches.set(issueIndex, column)\n }\n return matches\n}\n\nfunction findingMatchesIssue(finding: AnalystFinding, issue: AnalystIssueExpectation): boolean {\n if (issue.findingIds && !issue.findingIds.includes(finding.finding_id)) return false\n if (issue.areas && !issue.areas.includes(finding.area)) return false\n if (issue.subjects && (!finding.subject || !issue.subjects.includes(finding.subject))) {\n return false\n }\n if (\n issue.evidence &&\n !matchesEvidence(finding.evidence_refs, issue.evidence, issue.evidenceMode ?? 'any')\n ) {\n return false\n }\n return true\n}\n\nfunction matchesEvidence(\n actual: readonly EvidenceRef[],\n expected: readonly AnalystEvidenceExpectation[],\n mode: 'any' | 'all',\n): boolean {\n if (expected.length === 0) return true\n const match = (target: AnalystEvidenceExpectation) =>\n actual.some((ref) => evidenceMatches(ref, target))\n return mode === 'all' ? expected.every(match) : expected.some(match)\n}\n\nfunction evidenceMatches(actual: EvidenceRef, expected: AnalystEvidenceExpectation): boolean {\n return (\n actual.uri === expected.uri && (expected.kind === undefined || actual.kind === expected.kind)\n )\n}\n","import type {\n AnalystBenchmarkObservation,\n AnalystBenchmarkSummary,\n AnalystLatencyDistribution,\n} from './benchmark'\nimport { harmonicMeanScore } from './benchmark-scoring'\n\nexport function summarizeAnalystBenchmarkRunner(\n runnerId: string,\n observations: readonly AnalystBenchmarkObservation[],\n): AnalystBenchmarkSummary {\n const issueBearing = observations.filter((observation) => observation.labelState === 'positive')\n const completed = observations.filter((observation) => !observation.error)\n const expectedIssues = issueBearing.reduce(\n (sum, observation) => sum + observation.score.expectedIssueCount,\n 0,\n )\n const matchedIssues = issueBearing.reduce(\n (sum, observation) => sum + observation.score.matchedIssueIds.length,\n 0,\n )\n const issueFindings = issueBearing.reduce(\n (sum, observation) => sum + (observation.error ? 0 : observation.findings.length),\n 0,\n )\n const supportedFindings = issueBearing.reduce(\n (sum, observation) => sum + observation.score.supportedFindingIndexes.length,\n 0,\n )\n const issueRecall = expectedIssues === 0 ? null : matchedIssues / expectedIssues\n const findingPrecision =\n expectedIssues === 0 ? null : issueFindings === 0 ? 0 : supportedFindings / issueFindings\n const macroIssueRecall =\n issueBearing.length === 0\n ? null\n : mean(issueBearing.map((observation) => observation.score.issueRecall))\n const macroFindingPrecision =\n issueBearing.length === 0\n ? null\n : mean(issueBearing.map((observation) => observation.score.findingPrecision))\n const macroF1 =\n issueBearing.length === 0 ? null : mean(issueBearing.map((observation) => observation.score.f1))\n const critical = observations\n .map((observation) => observation.score.criticalStepAccuracy)\n .filter((value): value is number => value !== null)\n const findingsWithEvidence = completed.reduce(\n (sum, observation) =>\n sum + observation.findings.filter((finding) => finding.evidence_refs.length > 0).length,\n 0,\n )\n const allFindings = completed.reduce((sum, observation) => sum + observation.findings.length, 0)\n const allCitations = completed.flatMap((observation) =>\n observation.findings.flatMap((finding) => finding.evidence_refs),\n )\n const citationObservations = completed.filter(\n (observation) => observation.score.citationLabelAgreement !== null,\n )\n const citationCount = citationObservations.reduce(\n (sum, observation) =>\n sum +\n observation.findings.reduce((count, finding) => count + finding.evidence_refs.length, 0),\n 0,\n )\n const invalidCitationCount = citationObservations.reduce(\n (sum, observation) => sum + observation.score.unlabeledEvidence.length,\n 0,\n )\n const trustedNegative = observations.filter(\n (observation) => observation.labelState === 'trusted-negative',\n )\n const completedTrustedNegative = trustedNegative.filter((observation) => !observation.error)\n const unlabeled = observations.filter((observation) => observation.labelState === 'unlabeled')\n const completedUnlabeled = unlabeled.filter((observation) => !observation.error)\n const resolvedCitations = completed.reduce(\n (sum, observation) => sum + (observation.evidenceResolution?.resolved ?? 0),\n 0,\n )\n const unresolvedCitations = completed.reduce(\n (sum, observation) => sum + (observation.evidenceResolution?.unresolvedEvidence.length ?? 0),\n 0,\n )\n const citationResolutionErrors = completed.reduce(\n (sum, observation) => sum + (observation.evidenceResolution?.errors.length ?? 0),\n 0,\n )\n const resolutionAttempts = completed.filter((observation) =>\n observation.findings.some((finding) => finding.evidence_refs.length > 0),\n )\n const citationResolutionUnknownRuns = resolutionAttempts.filter(\n (observation) =>\n !observation.evidenceResolution || observation.evidenceResolution.errors.length > 0,\n ).length\n const usages = observations.map((observation) => observation.usage)\n const knownCostUsd = stableSum(\n usages.map((usage) => {\n if (!usage) return 0\n return usage.cost.kind === 'uncaptured' ? (usage.knownCostUsd ?? 0) : usage.cost.usd\n }),\n )\n const predictionAgreement = repeatedObservationAgreement(observations, predictionSignature)\n const matchedLabelAgreement = repeatedObservationAgreement(\n observations.filter((observation) => observation.labelState === 'positive'),\n matchedLabelSignature,\n )\n return {\n runnerId,\n plannedRuns: observations.length,\n completedRuns: observations.filter((observation) => !observation.error).length,\n failedRuns: observations.filter((observation) => Boolean(observation.error)).length,\n issueBearingRuns: issueBearing.length,\n trustedNegativeRuns: trustedNegative.length,\n unlabeledRuns: unlabeled.length,\n issueRecall,\n findingPrecision,\n f1:\n findingPrecision === null || issueRecall === null\n ? null\n : harmonicMeanScore(findingPrecision, issueRecall),\n macroIssueRecall,\n macroFindingPrecision,\n macroF1,\n criticalStepAccuracy: critical.length === 0 ? null : mean(critical),\n citationCoverage: allFindings === 0 ? null : findingsWithEvidence / allFindings,\n citationExcerptCoverage:\n allCitations.length === 0\n ? null\n : allCitations.filter((evidence) => Boolean(evidence.excerpt?.trim())).length /\n allCitations.length,\n citationLabelAgreement:\n citationObservations.length === 0\n ? null\n : citationCount === 0\n ? 0\n : (citationCount - invalidCitationCount) / citationCount,\n citationResolution:\n resolutionAttempts.length === 0 ||\n citationResolutionUnknownRuns > 0 ||\n resolvedCitations + unresolvedCitations === 0\n ? null\n : resolvedCitations / (resolvedCitations + unresolvedCitations),\n citationResolutionUnknownRuns,\n unresolvedCitations,\n citationResolutionErrors,\n trustedNegativeFalsePositiveRate:\n completedTrustedNegative.length === 0\n ? null\n : completedTrustedNegative.filter(\n (observation) => observation.score.predictionOnLabelEmptyCase,\n ).length / completedTrustedNegative.length,\n trustedNegativeFailureRate:\n trustedNegative.length === 0\n ? null\n : trustedNegative.filter((observation) => Boolean(observation.error)).length /\n trustedNegative.length,\n unlabeledPredictionRate:\n completedUnlabeled.length === 0\n ? null\n : completedUnlabeled.filter((observation) => observation.findings.length > 0).length /\n completedUnlabeled.length,\n unlabeledFailureRate:\n unlabeled.length === 0\n ? null\n : unlabeled.filter((observation) => Boolean(observation.error)).length / unlabeled.length,\n predictionAgreement: predictionAgreement.value,\n predictionAgreementCases: predictionAgreement.cases,\n matchedLabelAgreement: matchedLabelAgreement.value,\n matchedLabelAgreementCases: matchedLabelAgreement.cases,\n latencyMs: latencyDistribution(\n observations\n .map((observation) => observation.latencyMs)\n .filter((value): value is number => value !== null),\n ),\n benchmarkClockLatencyRuns: observations.filter(\n (observation) => observation.latencySource === 'benchmark-clock',\n ).length,\n runnerReportedLatencyRuns: observations.filter(\n (observation) => observation.latencySource === 'runner-reported',\n ).length,\n latencyUnknownRuns: observations.filter(\n (observation) => observation.latencySource === 'uncaptured',\n ).length,\n calls: usages.reduce((sum, usage) => sum + (usage?.calls ?? 0), 0),\n callsUnknownRuns: usages.filter((usage) => !usage || usage.calls === null).length,\n inputTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.input ?? 0), 0),\n outputTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.output ?? 0), 0),\n reasoningTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.reasoning ?? 0), 0),\n cachedTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.cached ?? 0), 0),\n cacheWriteTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.cacheWrite ?? 0), 0),\n tokenUsageUnknownRuns: usages.filter((usage) => !usage?.tokens).length,\n reasoningTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.reasoning === undefined)\n .length,\n cachedTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.cached === undefined)\n .length,\n cacheWriteTokenUsageUnknownRuns: usages.filter(\n (usage) => usage?.tokens?.cacheWrite === undefined,\n ).length,\n knownCostUsd,\n costUnknownRuns: usages.filter((usage) => !usage || usage.cost.kind === 'uncaptured').length,\n }\n}\n\nfunction mean(values: readonly number[]): number {\n return values.length === 0 ? 0 : stableSum(values) / values.length\n}\n\nfunction stableSum(values: readonly number[]): number {\n const ordered = [...values].sort(\n (left, right) => Math.abs(left) - Math.abs(right) || left - right,\n )\n let sum = 0\n let correction = 0\n for (const value of ordered) {\n const next = sum + value\n correction += Math.abs(sum) >= Math.abs(value) ? sum - next + value : value - next + sum\n sum = next\n }\n return sum + correction\n}\n\nfunction latencyDistribution(values: readonly number[]): AnalystLatencyDistribution | null {\n if (values.length === 0) return null\n const sorted = [...values].sort((a, b) => a - b)\n return {\n min: sorted[0]!,\n mean: mean(sorted),\n p50: percentile(sorted, 0.5),\n p95: percentile(sorted, 0.95),\n max: sorted.at(-1)!,\n }\n}\n\nfunction percentile(sorted: readonly number[], quantile: number): number {\n if (sorted.length === 0) return 0\n return sorted[Math.ceil(quantile * sorted.length) - 1] ?? sorted.at(-1) ?? 0\n}\n\nfunction repeatedObservationAgreement(\n observations: readonly AnalystBenchmarkObservation[],\n signature: (observation: AnalystBenchmarkObservation) => readonly string[],\n): { value: number | null; cases: number } {\n const byCase = new Map<string, AnalystBenchmarkObservation[]>()\n for (const observation of observations) {\n const rows = byCase.get(observation.caseId) ?? []\n rows.push(observation)\n byCase.set(observation.caseId, rows)\n }\n const caseAgreements: number[] = []\n for (const rows of byCase.values()) {\n const agreements: number[] = []\n for (let left = 0; left < rows.length; left++) {\n for (let right = left + 1; right < rows.length; right++) {\n agreements.push(jaccard(signature(rows[left]!), signature(rows[right]!)))\n }\n }\n if (agreements.length > 0) caseAgreements.push(mean(agreements))\n }\n return {\n value: caseAgreements.length === 0 ? null : mean(caseAgreements),\n cases: caseAgreements.length,\n }\n}\n\nfunction matchedLabelSignature(observation: AnalystBenchmarkObservation): readonly string[] {\n if (observation.error) return [`error:${observation.error.class}`]\n return observation.score.matchedIssueIds\n}\n\nfunction predictionSignature(observation: AnalystBenchmarkObservation): readonly string[] {\n if (observation.error) return [`error:${observation.error.class}`]\n return observation.findings\n .map((finding) =>\n JSON.stringify([\n finding.finding_id,\n finding.evidence_refs\n .map((evidence) => [evidence.kind, evidence.uri, evidence.excerpt ?? null])\n .sort((left, right) => JSON.stringify(left).localeCompare(JSON.stringify(right))),\n ]),\n )\n .sort()\n}\n\nfunction jaccard(left: readonly string[], right: readonly string[]): number {\n const a = new Set(left)\n const b = new Set(right)\n const union = new Set([...a, ...b])\n if (union.size === 0) return 1\n let intersection = 0\n for (const value of a) if (b.has(value)) intersection += 1\n return intersection / union.size\n}\n","import { performance } from 'node:perf_hooks'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport { assertValidAnalystScoringCase, scoreAnalystFindings } from './benchmark-scoring'\nimport { summarizeAnalystBenchmarkRunner } from './benchmark-summary'\nimport type { AnalystRegistry, RegistryRunOpts } from './registry'\nimport type {\n AnalystFinding,\n AnalystRunInputs,\n AnalystRunResult,\n AnalystUsageReceipt,\n EvidenceRef,\n} from './types'\nimport { assertValidAnalystUsageReceipt } from './usage-receipt'\n\nexport { scoreAnalystFindings } from './benchmark-scoring'\n\nexport interface AnalystEvidenceExpectation {\n uri: string\n kind?: EvidenceRef['kind']\n}\n\nexport interface AnalystIssueExpectation {\n id: string\n findingIds?: readonly string[]\n areas?: readonly string[]\n subjects?: readonly string[]\n evidence?: readonly AnalystEvidenceExpectation[]\n evidenceMode?: 'any' | 'all'\n /** Exact evidence location for the first unrecoverable or causal step. */\n criticalEvidence?: readonly AnalystEvidenceExpectation[]\n}\n\nexport type AnalystBenchmarkLabelState = 'positive' | 'trusted-negative' | 'unlabeled'\n\nexport interface AnalystBenchmarkCase<TInput = unknown> {\n id: string\n /** Independent source unit used for resampling, such as a task or incident. */\n clusterId: string\n /** Whether labels prove an issue, prove no issue, or leave the outcome unknown. */\n labelState: AnalystBenchmarkLabelState\n input: TInput\n expectedIssues: readonly AnalystIssueExpectation[]\n /** Complete set of labeled locations used to measure label-location agreement. */\n labeledEvidence?: readonly AnalystEvidenceExpectation[]\n tags?: readonly string[]\n metadata?: Record<string, unknown>\n}\n\nexport interface AnalystFindingScore {\n expectedIssueCount: number\n matchedIssueIds: string[]\n missedIssueIds: string[]\n supportedFindingIndexes: number[]\n unsupportedFindingIndexes: number[]\n unlabeledEvidence: EvidenceRef[]\n issueRecall: number\n findingPrecision: number\n f1: number\n criticalStepAccuracy: number | null\n /** Share of findings that cite at least one evidence location. */\n citationCoverage: number | null\n /** Share of citations that include a non-empty source excerpt. */\n citationExcerptCoverage: number | null\n /** Share of citations that agree with a labeled case location. */\n citationLabelAgreement: number | null\n predictionOnLabelEmptyCase: boolean\n}\n\nexport interface AnalystEvidenceResolutionError {\n evidence: EvidenceRef\n class: string\n message: string\n}\n\nexport interface AnalystEvidenceResolution {\n checked: number\n resolved: number\n unresolvedEvidence: EvidenceRef[]\n errors: AnalystEvidenceResolutionError[]\n /** Null when no citations were checked or any resolution attempt failed. */\n validity: number | null\n}\n\nexport type AnalystEvidenceResolver<TInput = unknown> = (input: {\n caseId: string\n caseInput: TInput\n evidence: EvidenceRef\n signal?: AbortSignal\n}) => boolean | Promise<boolean>\n\n/**\n * Resolve canonical `trace://<trace>/span/<span>` evidence against a trace store.\n * Other evidence kinds and URI schemes require a caller-supplied resolver.\n */\nexport function traceStoreEvidenceResolver<TInput>(\n getStore: (input: TInput) => TraceAnalysisStore,\n): AnalystEvidenceResolver<TInput> {\n return async ({ caseInput, evidence, signal }) => {\n if (evidence.kind !== 'span') return false\n const location = parseTraceSpanUri(evidence.uri)\n if (!location) return false\n const result = await getStore(caseInput).viewSpans(\n {\n trace_id: location.traceId,\n span_ids: [location.spanId],\n },\n signal ? { signal } : undefined,\n )\n return (\n result.trace_id === location.traceId &&\n result.missing_span_ids.length === 0 &&\n result.spans.some((span) => span.span_id === location.spanId)\n )\n }\n}\n\nexport interface AnalystBenchmarkOutput {\n findings: readonly AnalystFinding[]\n usage?: AnalystUsageReceipt\n metadata?: Record<string, unknown>\n /**\n * End-to-end duration measured by an external runner before import.\n * Use null when the source explicitly did not capture duration.\n */\n observedLatencyMs?: number | null\n /** Marks a completed transport as a failed analyst run while retaining usage and metadata. */\n error?: AnalystBenchmarkError\n}\n\nexport interface AnalystBenchmarkError {\n class: string\n message: string\n code?: string\n status?: number\n}\n\nexport interface AnalystBenchmarkRunner<TInput = unknown> {\n id: string\n analyze(\n input: TInput,\n context: { caseId: string; repetition: number; signal?: AbortSignal },\n ): AnalystBenchmarkOutput | Promise<AnalystBenchmarkOutput>\n}\n\nexport interface AnalystBenchmarkObservation {\n runnerId: string\n caseId: string\n clusterId: string\n labelState: AnalystBenchmarkLabelState\n repetition: number\n executionIndex: number\n latencyMs: number | null\n latencySource: 'benchmark-clock' | 'runner-reported' | 'uncaptured'\n findings: readonly AnalystFinding[]\n score: AnalystFindingScore\n evidenceResolution?: AnalystEvidenceResolution\n caseTags: readonly string[]\n caseMetadata?: Record<string, unknown>\n usage?: AnalystUsageReceipt\n runnerMetadata?: Record<string, unknown>\n error?: AnalystBenchmarkError\n}\n\nexport interface AnalystLatencyDistribution {\n min: number\n mean: number\n p50: number\n p95: number\n max: number\n}\n\nexport interface AnalystBenchmarkSummary {\n runnerId: string\n plannedRuns: number\n completedRuns: number\n failedRuns: number\n issueBearingRuns: number\n trustedNegativeRuns: number\n unlabeledRuns: number\n /** Pooled across all labeled issues and findings. */\n issueRecall: number | null\n /** Pooled across all labeled issues and findings. */\n findingPrecision: number | null\n /** Harmonic mean of the pooled precision and recall. */\n f1: number | null\n /** Mean of per-case recall over issue-bearing runs. */\n macroIssueRecall: number | null\n /** Mean of per-case precision over issue-bearing runs. */\n macroFindingPrecision: number | null\n /** Mean of per-case F1 over issue-bearing runs. */\n macroF1: number | null\n criticalStepAccuracy: number | null\n citationCoverage: number | null\n citationExcerptCoverage: number | null\n citationLabelAgreement: number | null\n citationResolution: number | null\n citationResolutionUnknownRuns: number\n unresolvedCitations: number\n citationResolutionErrors: number\n trustedNegativeFalsePositiveRate: number | null\n trustedNegativeFailureRate: number | null\n unlabeledPredictionRate: number | null\n unlabeledFailureRate: number | null\n /** Primary repeatability measure over complete finding identity and evidence. */\n predictionAgreement: number | null\n /** Repeated cases contributing equally to predictionAgreement. */\n predictionAgreementCases: number\n /** Secondary repeatability detail over matched expected labels. */\n matchedLabelAgreement: number | null\n /** Positive repeated cases contributing equally to matchedLabelAgreement. */\n matchedLabelAgreementCases: number\n latencyMs: AnalystLatencyDistribution | null\n benchmarkClockLatencyRuns: number\n runnerReportedLatencyRuns: number\n latencyUnknownRuns: number\n calls: number\n callsUnknownRuns: number\n inputTokens: number\n outputTokens: number\n reasoningTokens: number\n cachedTokens: number\n cacheWriteTokens: number\n tokenUsageUnknownRuns: number\n reasoningTokenUsageUnknownRuns: number\n cachedTokenUsageUnknownRuns: number\n cacheWriteTokenUsageUnknownRuns: number\n knownCostUsd: number\n costUnknownRuns: number\n}\n\nexport interface AnalystBenchmarkDatasetRef {\n id: string\n revision: string\n split?: string\n}\n\nexport interface AnalystBenchmarkDescriptor {\n id?: string\n dataset?: AnalystBenchmarkDatasetRef\n command?: string\n environment?: Record<string, string>\n metadata?: Record<string, unknown>\n}\n\nexport interface AnalystBenchmarkProvenance extends AnalystBenchmarkDescriptor {\n startedAt: string\n endedAt: string\n caseCount: number\n runnerIds: string[]\n repetitions: number\n maxConcurrency: number\n runnerOrderSeed: number\n}\n\nexport interface AnalystBenchmarkResult {\n provenance: AnalystBenchmarkProvenance\n observations: AnalystBenchmarkObservation[]\n summaries: AnalystBenchmarkSummary[]\n}\n\nexport interface RunAnalystBenchmarkOptions<TInput> {\n cases: readonly AnalystBenchmarkCase<TInput>[]\n runners: readonly AnalystBenchmarkRunner<TInput>[]\n repetitions?: number\n maxConcurrency?: number\n runnerOrderSeed?: number\n resolveEvidence?: AnalystEvidenceResolver<TInput>\n benchmark?: AnalystBenchmarkDescriptor\n /** Previously persisted rows. Exact case, runner, repetition, and execution identities are required. */\n initialObservations?: readonly AnalystBenchmarkObservation[]\n onObservation?: (observation: AnalystBenchmarkObservation) => void | Promise<void>\n signal?: AbortSignal\n}\n\nexport async function runAnalystBenchmark<TInput>(\n options: RunAnalystBenchmarkOptions<TInput>,\n): Promise<AnalystBenchmarkResult> {\n validateBenchmarkOptions(options)\n const startedAt = new Date().toISOString()\n const repetitions = options.repetitions ?? 1\n const runnerOrderSeed = options.runnerOrderSeed ?? 0\n const allJobs = benchmarkJobs(options.cases, options.runners, repetitions, runnerOrderSeed)\n const maxConcurrency = Math.min(options.maxConcurrency ?? 1, allJobs.length)\n const initialObservations = validateInitialObservations(\n options.initialObservations ?? [],\n allJobs,\n )\n const completed = new Set(initialObservations.map(observationKey))\n const jobs = allJobs.filter((job) => !completed.has(jobKey(job)))\n const observations: AnalystBenchmarkObservation[] = [...initialObservations]\n let cursor = 0\n const worker = async (): Promise<void> => {\n while (cursor < jobs.length) {\n options.signal?.throwIfAborted()\n const job = jobs[cursor++]!\n const observation = await runBenchmarkJob(job, options.signal, options.resolveEvidence)\n options.signal?.throwIfAborted()\n await options.onObservation?.(observation)\n options.signal?.throwIfAborted()\n observations.push(observation)\n }\n }\n await Promise.all(Array.from({ length: Math.min(maxConcurrency, jobs.length) }, worker))\n options.signal?.throwIfAborted()\n const runnerOrder = new Map(options.runners.map((runner, index) => [runner.id, index]))\n const caseOrder = new Map(options.cases.map((testCase, index) => [testCase.id, index]))\n observations.sort(\n (a, b) =>\n (runnerOrder.get(a.runnerId) ?? 0) - (runnerOrder.get(b.runnerId) ?? 0) ||\n (caseOrder.get(a.caseId) ?? 0) - (caseOrder.get(b.caseId) ?? 0) ||\n a.repetition - b.repetition,\n )\n return {\n provenance: {\n ...options.benchmark,\n startedAt,\n endedAt: new Date().toISOString(),\n caseCount: options.cases.length,\n runnerIds: options.runners.map((runner) => runner.id),\n repetitions,\n maxConcurrency,\n runnerOrderSeed,\n },\n observations,\n summaries: options.runners.map((runner) =>\n summarizeAnalystBenchmarkRunner(\n runner.id,\n observations.filter((observation) => observation.runnerId === runner.id),\n ),\n ),\n }\n}\n\nexport function registryBenchmarkRunner(options: {\n id: string\n registry: AnalystRegistry\n runOptions?: Omit<RegistryRunOpts, 'signal'>\n /** Count any selected analyst failure as a failed benchmark run. */\n failOnAnalystFailure?: boolean\n}): AnalystBenchmarkRunner<AnalystRunInputs> {\n return {\n id: options.id,\n async analyze(input, context) {\n const result = await options.registry.run(\n `${options.id}:${context.caseId}:${context.repetition}`,\n input,\n { ...options.runOptions, signal: context.signal },\n )\n return {\n findings: result.findings,\n usage: mergeRegistryUsage(result),\n metadata: { analystRun: result },\n ...(options.failOnAnalystFailure ? { error: registryRunFailure(result) } : {}),\n }\n },\n }\n}\n\nasync function runBenchmarkJob<TInput>(\n job: {\n runner: AnalystBenchmarkRunner<TInput>\n testCase: AnalystBenchmarkCase<TInput>\n repetition: number\n executionIndex: number\n },\n signal?: AbortSignal,\n resolveEvidence?: AnalystEvidenceResolver<TInput>,\n): Promise<AnalystBenchmarkObservation> {\n const started = performance.now()\n try {\n const output = await job.runner.analyze(job.testCase.input, {\n caseId: job.testCase.id,\n repetition: job.repetition,\n signal,\n })\n if (output.usage) {\n assertValidAnalystUsageReceipt(output.usage, 'analyst benchmark usage')\n }\n const benchmarkLatencyMs = performance.now() - started\n const latency = resolveBenchmarkLatency(output.observedLatencyMs, benchmarkLatencyMs)\n const scoredFindings = output.error ? [] : output.findings\n return {\n runnerId: job.runner.id,\n caseId: job.testCase.id,\n clusterId: job.testCase.clusterId,\n labelState: job.testCase.labelState,\n repetition: job.repetition,\n executionIndex: job.executionIndex,\n latencyMs: latency.value,\n latencySource: latency.source,\n findings: output.findings,\n score: scoreAnalystFindings(job.testCase, scoredFindings),\n evidenceResolution: resolveEvidence\n ? await resolveFindingEvidence(job.testCase, output.findings, resolveEvidence, signal)\n : undefined,\n caseTags: [...(job.testCase.tags ?? [])],\n caseMetadata: job.testCase.metadata,\n usage: output.usage,\n runnerMetadata: output.metadata,\n ...(output.error ? { error: output.error } : {}),\n }\n } catch (error) {\n if (signal?.aborted) throw error\n const findings: AnalystFinding[] = []\n return {\n runnerId: job.runner.id,\n caseId: job.testCase.id,\n clusterId: job.testCase.clusterId,\n labelState: job.testCase.labelState,\n repetition: job.repetition,\n executionIndex: job.executionIndex,\n latencyMs: performance.now() - started,\n latencySource: 'benchmark-clock',\n findings,\n score: scoreAnalystFindings(job.testCase, findings),\n caseTags: [...(job.testCase.tags ?? [])],\n caseMetadata: job.testCase.metadata,\n error: {\n class: error instanceof Error ? error.constructor.name : 'Error',\n message: error instanceof Error ? error.message : String(error),\n },\n }\n }\n}\n\nfunction resolveBenchmarkLatency(\n observedLatencyMs: number | null | undefined,\n fallbackMs: number,\n): {\n value: number | null\n source: AnalystBenchmarkObservation['latencySource']\n} {\n if (observedLatencyMs === undefined) {\n return { value: fallbackMs, source: 'benchmark-clock' }\n }\n if (observedLatencyMs === null) return { value: null, source: 'uncaptured' }\n if (!Number.isFinite(observedLatencyMs) || observedLatencyMs < 0) {\n throw new RangeError('analyst benchmark observedLatencyMs must be finite and non-negative')\n }\n return { value: observedLatencyMs, source: 'runner-reported' }\n}\n\nfunction registryRunFailure(result: AnalystRunResult): AnalystBenchmarkOutput['error'] | undefined {\n const failed = result.per_analyst.filter((summary) => summary.status === 'failed')\n if (failed.length === 0) return undefined\n return {\n class: 'AnalystRunFailure',\n message: failed\n .map(\n (summary) =>\n `${summary.analyst_id}: ${summary.error?.class ?? 'Error'}: ${summary.error?.message ?? 'analyst failed'}`,\n )\n .join('; '),\n }\n}\n\nfunction benchmarkJobs<TInput>(\n cases: readonly AnalystBenchmarkCase<TInput>[],\n runners: readonly AnalystBenchmarkRunner<TInput>[],\n repetitions: number,\n runnerOrderSeed: number,\n): Array<{\n runner: AnalystBenchmarkRunner<TInput>\n testCase: AnalystBenchmarkCase<TInput>\n repetition: number\n executionIndex: number\n}> {\n const seededRunners = [...runners].sort(\n (left, right) =>\n stableHash(`${runnerOrderSeed}\\u0000${left.id}`) -\n stableHash(`${runnerOrderSeed}\\u0000${right.id}`) || left.id.localeCompare(right.id),\n )\n let executionIndex = 0\n return cases.flatMap((testCase, caseIndex) =>\n Array.from({ length: repetitions }, (_, repetition) => {\n const blockIndex = caseIndex * repetitions + repetition\n const rotation = blockIndex % seededRunners.length\n const ordered = [...seededRunners.slice(rotation), ...seededRunners.slice(0, rotation)]\n return ordered.map((runner) => ({\n runner,\n testCase,\n repetition,\n executionIndex: executionIndex++,\n }))\n }).flat(),\n )\n}\n\nfunction validateInitialObservations<TInput>(\n observations: readonly AnalystBenchmarkObservation[],\n jobs: readonly {\n runner: AnalystBenchmarkRunner<TInput>\n testCase: AnalystBenchmarkCase<TInput>\n repetition: number\n executionIndex: number\n }[],\n): AnalystBenchmarkObservation[] {\n const expected = new Map(jobs.map((job) => [jobKey(job), job]))\n const seen = new Set<string>()\n return observations.map((observation) => {\n const key = observationKey(observation)\n if (seen.has(key)) {\n throw new TypeError(\n `duplicate initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`,\n )\n }\n seen.add(key)\n const job = expected.get(key)\n if (!job) {\n throw new TypeError(\n `initial analyst benchmark observation does not match a planned job: '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`,\n )\n }\n if (observation.executionIndex !== job.executionIndex) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has executionIndex ${observation.executionIndex}; expected ${job.executionIndex}`,\n )\n }\n if (\n observation.clusterId !== job.testCase.clusterId ||\n observation.labelState !== job.testCase.labelState\n ) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' does not match the current case labels`,\n )\n }\n if (\n JSON.stringify(observation.caseTags) !== JSON.stringify(job.testCase.tags ?? []) ||\n JSON.stringify(observation.caseMetadata) !== JSON.stringify(job.testCase.metadata)\n ) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' does not match the current case metadata`,\n )\n }\n const expectedScore = scoreAnalystFindings(\n job.testCase,\n observation.error ? [] : observation.findings,\n )\n if (JSON.stringify(observation.score) !== JSON.stringify(expectedScore)) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has stale or invalid scores`,\n )\n }\n if (observation.usage) {\n assertValidAnalystUsageReceipt(\n observation.usage,\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' usage`,\n )\n }\n if (\n !['benchmark-clock', 'runner-reported', 'uncaptured'].includes(observation.latencySource) ||\n (observation.latencySource === 'uncaptured' && observation.latencyMs !== null) ||\n (observation.latencySource !== 'uncaptured' &&\n (observation.latencyMs === null ||\n !Number.isFinite(observation.latencyMs) ||\n observation.latencyMs < 0))\n ) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has invalid latency`,\n )\n }\n return { ...observation }\n })\n}\n\nfunction jobKey(job: {\n runner: { id: string }\n testCase: { id: string }\n repetition: number\n}): string {\n return `${job.runner.id}\\u0000${job.testCase.id}\\u0000${job.repetition}`\n}\n\nfunction observationKey(observation: {\n runnerId: string\n caseId: string\n repetition: number\n}): string {\n return `${observation.runnerId}\\u0000${observation.caseId}\\u0000${observation.repetition}`\n}\n\nasync function resolveFindingEvidence<TInput>(\n testCase: AnalystBenchmarkCase<TInput>,\n findings: readonly AnalystFinding[],\n resolver: AnalystEvidenceResolver<TInput>,\n signal?: AbortSignal,\n): Promise<AnalystEvidenceResolution> {\n const evidence = findings.flatMap((finding) => finding.evidence_refs)\n const resolved: EvidenceRef[] = []\n const unresolvedEvidence: EvidenceRef[] = []\n const errors: AnalystEvidenceResolutionError[] = []\n for (const ref of evidence) {\n signal?.throwIfAborted()\n try {\n if (\n await resolver({\n caseId: testCase.id,\n caseInput: testCase.input,\n evidence: ref,\n signal,\n })\n ) {\n resolved.push(ref)\n } else {\n unresolvedEvidence.push(ref)\n }\n } catch (error) {\n if (signal?.aborted) throw error\n errors.push({\n evidence: ref,\n class: error instanceof Error ? error.constructor.name : 'Error',\n message: error instanceof Error ? error.message : String(error),\n })\n }\n }\n return {\n checked: evidence.length,\n resolved: resolved.length,\n unresolvedEvidence,\n errors,\n validity: evidence.length === 0 || errors.length > 0 ? null : resolved.length / evidence.length,\n }\n}\n\nfunction parseTraceSpanUri(uri: string): { traceId: string; spanId: string } | null {\n const match = /^trace:\\/\\/([^/]+)\\/span\\/([^/]+)$/.exec(uri)\n if (!match) return null\n try {\n const traceId = decodeURIComponent(match[1]!)\n const spanId = decodeURIComponent(match[2]!)\n return traceId && spanId ? { traceId, spanId } : null\n } catch {\n return null\n }\n}\n\nfunction validateBenchmarkOptions<TInput>(options: RunAnalystBenchmarkOptions<TInput>): void {\n if (options.cases.length === 0) throw new TypeError('runAnalystBenchmark requires cases')\n if (options.runners.length === 0) throw new TypeError('runAnalystBenchmark requires runners')\n const repetitions = options.repetitions ?? 1\n const maxConcurrency = options.maxConcurrency ?? 1\n if (!Number.isSafeInteger(repetitions) || repetitions < 1) {\n throw new RangeError('runAnalystBenchmark repetitions must be a positive safe integer')\n }\n if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) {\n throw new RangeError('runAnalystBenchmark maxConcurrency must be a positive safe integer')\n }\n if (!Number.isSafeInteger(options.runnerOrderSeed ?? 0)) {\n throw new RangeError('runAnalystBenchmark runnerOrderSeed must be a safe integer')\n }\n assertUniqueNonEmpty(\n options.cases.map((testCase) => testCase.id),\n 'case',\n )\n assertUniqueNonEmpty(\n options.runners.map((runner) => runner.id),\n 'runner',\n )\n for (const testCase of options.cases) validateBenchmarkCase(testCase)\n}\n\nfunction validateBenchmarkCase(testCase: AnalystBenchmarkCase): void {\n assertValidAnalystScoringCase(testCase)\n if (!testCase.clusterId.trim()) {\n throw new TypeError(`${testCase.id}: analyst benchmark clusterId must not be empty`)\n }\n if (\n testCase.labelState !== 'positive' &&\n testCase.labelState !== 'trusted-negative' &&\n testCase.labelState !== 'unlabeled'\n ) {\n throw new TypeError(`${testCase.id}: analyst benchmark labelState is invalid`)\n }\n if (testCase.labelState === 'positive' && testCase.expectedIssues.length === 0) {\n throw new TypeError(`${testCase.id}: positive case requires at least one expected issue`)\n }\n if (testCase.labelState !== 'positive' && testCase.expectedIssues.length > 0) {\n throw new TypeError(\n `${testCase.id}: ${testCase.labelState} case cannot contain expected issues`,\n )\n }\n}\n\nfunction assertUniqueNonEmpty(values: readonly string[], label: string): void {\n const seen = new Set<string>()\n for (const value of values) {\n if (!value.trim()) throw new TypeError(`analyst benchmark ${label} id must not be empty`)\n if (seen.has(value)) throw new TypeError(`duplicate analyst benchmark ${label} id '${value}'`)\n seen.add(value)\n }\n}\n\nfunction stableHash(value: string): number {\n let hash = 2166136261\n for (let index = 0; index < value.length; index += 1) {\n hash ^= value.charCodeAt(index)\n hash = Math.imul(hash, 16777619)\n }\n return hash >>> 0\n}\n\nfunction mergeRegistryUsage(result: AnalystRunResult): AnalystUsageReceipt {\n const usages = result.per_analyst.map((summary) => summary.usage)\n const calls = usages.every((usage) => usage.calls !== null)\n ? usages.reduce((sum, usage) => sum + (usage.calls ?? 0), 0)\n : null\n const tokens = usages.every((usage) => usage.tokens !== null)\n ? usages.reduce(\n (sum, usage) => ({\n input: sum.input + (usage.tokens?.input ?? 0),\n output: sum.output + (usage.tokens?.output ?? 0),\n reasoning: sum.reasoning + (usage.tokens?.reasoning ?? 0),\n cached: sum.cached + (usage.tokens?.cached ?? 0),\n cacheWrite: sum.cacheWrite + (usage.tokens?.cacheWrite ?? 0),\n }),\n { input: 0, output: 0, reasoning: 0, cached: 0, cacheWrite: 0 },\n )\n : null\n const knownCostUsd = usages.reduce(\n (sum, usage) =>\n sum + (usage.cost.kind === 'uncaptured' ? (usage.knownCostUsd ?? 0) : usage.cost.usd),\n 0,\n )\n const cost = usages.some((usage) => usage.cost.kind === 'uncaptured')\n ? ({ kind: 'uncaptured', usd: null } as const)\n : usages.some((usage) => usage.cost.kind === 'estimated')\n ? ({ kind: 'estimated', usd: knownCostUsd } as const)\n : ({ kind: 'observed', usd: knownCostUsd } as const)\n return {\n calls,\n tokens,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd } : {}),\n }\n}\n"],"mappings":";;;;AASA,SAAgB,qBACd,UACA,UACqB;CACrB,8BAA8B,QAAQ;CACtC,MAAM,wBAAwB,sBAAsB,SAAS,gBAAgB,QAAQ;CACrF,MAAM,kBAAkB,SAAS,eAC9B,QAAQ,GAAG,UAAU,sBAAsB,IAAI,KAAK,CAAC,CAAC,CACtD,KAAK,UAAU,MAAM,EAAE;CAC1B,MAAM,iBAAiB,SAAS,eAC7B,QAAQ,GAAG,UAAU,CAAC,sBAAsB,IAAI,KAAK,CAAC,CAAC,CACvD,KAAK,UAAU,MAAM,EAAE;CAC1B,MAAM,0BAA0B,IAAI,IAAI,sBAAsB,OAAO,CAAC;CAEtE,MAAM,4BAA4B,SAC/B,KAAK,GAAG,UAAU,KAAK,CAAC,CACxB,QAAQ,UAAU,CAAC,wBAAwB,IAAI,KAAK,CAAC;CACxD,MAAM,qBAAqB,SAAS,eAAe;CACnD,MAAM,cAAc,uBAAuB,IAAI,IAAI,gBAAgB,SAAS;CAC5E,MAAM,mBACJ,SAAS,WAAW,IAChB,uBAAuB,IACrB,IACA,IACF,wBAAwB,OAAO,SAAS;CAC9C,MAAM,KAAK,kBAAkB,kBAAkB,WAAW;CAC1D,MAAM,cAAc,SAAS,SAAS,YAAY,QAAQ,aAAa;CAEvE,MAAM,iBAAiB,SAAS,eAAe,QAC5C,WAAW,MAAM,kBAAkB,UAAU,KAAK,CACrD;CACA,MAAM,eAAe,SAAS,eAAe,QAAQ,UAAU;EAC7D,KAAK,MAAM,kBAAkB,UAAU,OAAO,GAAG,OAAO;EACxD,OAAO,gBAAgB,aAAa,MAAM,oBAAoB,CAAC,GAAG,KAAK;CACzE,CAAC,CAAC,CAAC;CAEH,MAAM,uBAAuB,SAAS,QAAQ,YAAY,QAAQ,cAAc,SAAS,CAAC,CAAC,CAAC;CAC5F,MAAM,oBAAoB,SAAS,kBAC/B,YAAY,QACT,QAAQ,CAAC,SAAS,gBAAiB,MAAM,aAAa,gBAAgB,KAAK,QAAQ,CAAC,CACvF,IACA,CAAC;CAEL,OAAO;EACL;EACA;EACA;EACA,yBAAyB,CAAC,GAAG,uBAAuB,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EAC1E;EACA;EACA;EACA;EACA;EACA,sBAAsB,eAAe,WAAW,IAAI,OAAO,eAAe,eAAe;EACzF,kBAAkB,SAAS,WAAW,IAAI,OAAO,uBAAuB,SAAS;EACjF,yBACE,YAAY,WAAW,IACnB,OACA,YAAY,QAAQ,aAAa,QAAQ,SAAS,SAAS,KAAK,CAAC,CAAC,CAAC,CAAC,SACpE,YAAY;EAClB,wBACE,SAAS,oBAAoB,KAAA,IACzB,OACA,YAAY,WAAW,IACrB,SAAS,WAAW,IAClB,OACA,KACD,YAAY,SAAS,kBAAkB,UAAU,YAAY;EACtE,4BAA4B,uBAAuB,KAAK,SAAS,SAAS;CAC5E;AACF;AAEA,SAAgB,8BACd,UACM;CACN,IAAI,CAAC,SAAS,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,6CAA6C;CAC1F,MAAM,sBAAM,IAAI,IAAY;CAC5B,KAAK,MAAM,SAAS,SAAS,gBAAgB;EAC3C,IAAI,CAAC,MAAM,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,sCAAsC;EAC/F,IAAI,IAAI,IAAI,MAAM,EAAE,GAClB,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,iCAAiC,MAAM,GAAG,EAAE;EAEjF,IAAI,IAAI,MAAM,EAAE;EAChB,IACE,CAAC,MAAM,YAAY,UACnB,CAAC,MAAM,OAAO,UACd,CAAC,MAAM,UAAU,UACjB,CAAC,MAAM,UAAU,QAEjB,MAAM,IAAI,UACR,GAAG,SAAS,GAAG,GAAG,MAAM,GAAG,2EAC7B;CAEJ;CACA,KAAK,MAAM,OAAO,SAAS,mBAAmB,CAAC,GAC7C,IAAI,CAAC,IAAI,IAAI,KAAK,GAChB,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,yCAAyC;AAGlF;AAEA,SAAgB,kBAAkB,GAAW,GAAmB;CAC9D,OAAO,IAAI,MAAM,IAAI,IAAK,IAAI,IAAI,KAAM,IAAI;AAC9C;AAEA,SAAS,sBACP,QACA,UACqB;CACrB,IAAI,OAAO,WAAW,GAAG,uBAAO,IAAI,IAAI;CACxC,MAAM,oBAAoB,OAAO,SAAS;CAW1C,MAAM,aAAa,oBAVJ,OAAO,KAAK,UAAU,CACnC,GAAG,SAAS,KAAK,YAAY;EAC3B,IAAI,CAAC,oBAAoB,SAAS,KAAK,GAAG,OAAO;EACjD,MAAM,eACH,MAAM,kBAAkB,UAAU,KAAK,KACxC,gBAAgB,QAAQ,eAAe,MAAM,oBAAoB,CAAC,GAAG,KAAK;EAC5E,OAAO,oBAAoB,OAAO,WAAW;CAC/C,CAAC,GACD,GAAG,MAAM,KAAK,EAAE,QAAQ,OAAO,OAAO,SAAS,CAAC,CAClD,CAC4C,GAAG,EAAE,UAAU,KAAK,CAAC,CAAC,CAAC;CACnE,MAAM,0BAAU,IAAI,IAAoB;CACxC,KAAK,MAAM,CAAC,YAAY,WAAW,WAAW,QAAQ,GAAG;EACvD,IAAI,SAAS,KAAK,UAAU,SAAS,QAAQ;EAC7C,IAAI,CAAC,oBAAoB,SAAS,SAAU,OAAO,WAAY,GAAG;EAClE,QAAQ,IAAI,YAAY,MAAM;CAChC;CACA,OAAO;AACT;AAEA,SAAS,oBAAoB,SAAyB,OAAyC;CAC7F,IAAI,MAAM,cAAc,CAAC,MAAM,WAAW,SAAS,QAAQ,UAAU,GAAG,OAAO;CAC/E,IAAI,MAAM,SAAS,CAAC,MAAM,MAAM,SAAS,QAAQ,IAAI,GAAG,OAAO;CAC/D,IAAI,MAAM,aAAa,CAAC,QAAQ,WAAW,CAAC,MAAM,SAAS,SAAS,QAAQ,OAAO,IACjF,OAAO;CAET,IACE,MAAM,YACN,CAAC,gBAAgB,QAAQ,eAAe,MAAM,UAAU,MAAM,gBAAgB,KAAK,GAEnF,OAAO;CAET,OAAO;AACT;AAEA,SAAS,gBACP,QACA,UACA,MACS;CACT,IAAI,SAAS,WAAW,GAAG,OAAO;CAClC,MAAM,SAAS,WACb,OAAO,MAAM,QAAQ,gBAAgB,KAAK,MAAM,CAAC;CACnD,OAAO,SAAS,QAAQ,SAAS,MAAM,KAAK,IAAI,SAAS,KAAK,KAAK;AACrE;AAEA,SAAS,gBAAgB,QAAqB,UAA+C;CAC3F,OACE,OAAO,QAAQ,SAAS,QAAQ,SAAS,SAAS,KAAA,KAAa,OAAO,SAAS,SAAS;AAE5F;;;ACnKA,SAAgB,gCACd,UACA,cACyB;CACzB,MAAM,eAAe,aAAa,QAAQ,gBAAgB,YAAY,eAAe,UAAU;CAC/F,MAAM,YAAY,aAAa,QAAQ,gBAAgB,CAAC,YAAY,KAAK;CACzE,MAAM,iBAAiB,aAAa,QACjC,KAAK,gBAAgB,MAAM,YAAY,MAAM,oBAC9C,CACF;CACA,MAAM,gBAAgB,aAAa,QAChC,KAAK,gBAAgB,MAAM,YAAY,MAAM,gBAAgB,QAC9D,CACF;CACA,MAAM,gBAAgB,aAAa,QAChC,KAAK,gBAAgB,OAAO,YAAY,QAAQ,IAAI,YAAY,SAAS,SAC1E,CACF;CACA,MAAM,oBAAoB,aAAa,QACpC,KAAK,gBAAgB,MAAM,YAAY,MAAM,wBAAwB,QACtE,CACF;CACA,MAAM,cAAc,mBAAmB,IAAI,OAAO,gBAAgB;CAClE,MAAM,mBACJ,mBAAmB,IAAI,OAAO,kBAAkB,IAAI,IAAI,oBAAoB;CAC9E,MAAM,mBACJ,aAAa,WAAW,IACpB,OACA,KAAK,aAAa,KAAK,gBAAgB,YAAY,MAAM,WAAW,CAAC;CAC3E,MAAM,wBACJ,aAAa,WAAW,IACpB,OACA,KAAK,aAAa,KAAK,gBAAgB,YAAY,MAAM,gBAAgB,CAAC;CAChF,MAAM,UACJ,aAAa,WAAW,IAAI,OAAO,KAAK,aAAa,KAAK,gBAAgB,YAAY,MAAM,EAAE,CAAC;CACjG,MAAM,WAAW,aACd,KAAK,gBAAgB,YAAY,MAAM,oBAAoB,CAAC,CAC5D,QAAQ,UAA2B,UAAU,IAAI;CACpD,MAAM,uBAAuB,UAAU,QACpC,KAAK,gBACJ,MAAM,YAAY,SAAS,QAAQ,YAAY,QAAQ,cAAc,SAAS,CAAC,CAAC,CAAC,QACnF,CACF;CACA,MAAM,cAAc,UAAU,QAAQ,KAAK,gBAAgB,MAAM,YAAY,SAAS,QAAQ,CAAC;CAC/F,MAAM,eAAe,UAAU,SAAS,gBACtC,YAAY,SAAS,SAAS,YAAY,QAAQ,aAAa,CACjE;CACA,MAAM,uBAAuB,UAAU,QACpC,gBAAgB,YAAY,MAAM,2BAA2B,IAChE;CACA,MAAM,gBAAgB,qBAAqB,QACxC,KAAK,gBACJ,MACA,YAAY,SAAS,QAAQ,OAAO,YAAY,QAAQ,QAAQ,cAAc,QAAQ,CAAC,GACzF,CACF;CACA,MAAM,uBAAuB,qBAAqB,QAC/C,KAAK,gBAAgB,MAAM,YAAY,MAAM,kBAAkB,QAChE,CACF;CACA,MAAM,kBAAkB,aAAa,QAClC,gBAAgB,YAAY,eAAe,kBAC9C;CACA,MAAM,2BAA2B,gBAAgB,QAAQ,gBAAgB,CAAC,YAAY,KAAK;CAC3F,MAAM,YAAY,aAAa,QAAQ,gBAAgB,YAAY,eAAe,WAAW;CAC7F,MAAM,qBAAqB,UAAU,QAAQ,gBAAgB,CAAC,YAAY,KAAK;CAC/E,MAAM,oBAAoB,UAAU,QACjC,KAAK,gBAAgB,OAAO,YAAY,oBAAoB,YAAY,IACzE,CACF;CACA,MAAM,sBAAsB,UAAU,QACnC,KAAK,gBAAgB,OAAO,YAAY,oBAAoB,mBAAmB,UAAU,IAC1F,CACF;CACA,MAAM,2BAA2B,UAAU,QACxC,KAAK,gBAAgB,OAAO,YAAY,oBAAoB,OAAO,UAAU,IAC9E,CACF;CACA,MAAM,qBAAqB,UAAU,QAAQ,gBAC3C,YAAY,SAAS,MAAM,YAAY,QAAQ,cAAc,SAAS,CAAC,CACzE;CACA,MAAM,gCAAgC,mBAAmB,QACtD,gBACC,CAAC,YAAY,sBAAsB,YAAY,mBAAmB,OAAO,SAAS,CACtF,CAAC,CAAC;CACF,MAAM,SAAS,aAAa,KAAK,gBAAgB,YAAY,KAAK;CAClE,MAAM,eAAe,UACnB,OAAO,KAAK,UAAU;EACpB,IAAI,CAAC,OAAO,OAAO;EACnB,OAAO,MAAM,KAAK,SAAS,eAAgB,MAAM,gBAAgB,IAAK,MAAM,KAAK;CACnF,CAAC,CACH;CACA,MAAM,sBAAsB,6BAA6B,cAAc,mBAAmB;CAC1F,MAAM,wBAAwB,6BAC5B,aAAa,QAAQ,gBAAgB,YAAY,eAAe,UAAU,GAC1E,qBACF;CACA,OAAO;EACL;EACA,aAAa,aAAa;EAC1B,eAAe,aAAa,QAAQ,gBAAgB,CAAC,YAAY,KAAK,CAAC,CAAC;EACxE,YAAY,aAAa,QAAQ,gBAAgB,QAAQ,YAAY,KAAK,CAAC,CAAC,CAAC;EAC7E,kBAAkB,aAAa;EAC/B,qBAAqB,gBAAgB;EACrC,eAAe,UAAU;EACzB;EACA;EACA,IACE,qBAAqB,QAAQ,gBAAgB,OACzC,OACA,kBAAkB,kBAAkB,WAAW;EACrD;EACA;EACA;EACA,sBAAsB,SAAS,WAAW,IAAI,OAAO,KAAK,QAAQ;EAClE,kBAAkB,gBAAgB,IAAI,OAAO,uBAAuB;EACpE,yBACE,aAAa,WAAW,IACpB,OACA,aAAa,QAAQ,aAAa,QAAQ,SAAS,SAAS,KAAK,CAAC,CAAC,CAAC,CAAC,SACrE,aAAa;EACnB,wBACE,qBAAqB,WAAW,IAC5B,OACA,kBAAkB,IAChB,KACC,gBAAgB,wBAAwB;EACjD,oBACE,mBAAmB,WAAW,KAC9B,gCAAgC,KAChC,oBAAoB,wBAAwB,IACxC,OACA,qBAAqB,oBAAoB;EAC/C;EACA;EACA;EACA,kCACE,yBAAyB,WAAW,IAChC,OACA,yBAAyB,QACtB,gBAAgB,YAAY,MAAM,0BACrC,CAAC,CAAC,SAAS,yBAAyB;EAC1C,4BACE,gBAAgB,WAAW,IACvB,OACA,gBAAgB,QAAQ,gBAAgB,QAAQ,YAAY,KAAK,CAAC,CAAC,CAAC,SACpE,gBAAgB;EACtB,yBACE,mBAAmB,WAAW,IAC1B,OACA,mBAAmB,QAAQ,gBAAgB,YAAY,SAAS,SAAS,CAAC,CAAC,CAAC,SAC5E,mBAAmB;EACzB,sBACE,UAAU,WAAW,IACjB,OACA,UAAU,QAAQ,gBAAgB,QAAQ,YAAY,KAAK,CAAC,CAAC,CAAC,SAAS,UAAU;EACvF,qBAAqB,oBAAoB;EACzC,0BAA0B,oBAAoB;EAC9C,uBAAuB,sBAAsB;EAC7C,4BAA4B,sBAAsB;EAClD,WAAW,oBACT,aACG,KAAK,gBAAgB,YAAY,SAAS,CAAC,CAC3C,QAAQ,UAA2B,UAAU,IAAI,CACtD;EACA,2BAA2B,aAAa,QACrC,gBAAgB,YAAY,kBAAkB,iBACjD,CAAC,CAAC;EACF,2BAA2B,aAAa,QACrC,gBAAgB,YAAY,kBAAkB,iBACjD,CAAC,CAAC;EACF,oBAAoB,aAAa,QAC9B,gBAAgB,YAAY,kBAAkB,YACjD,CAAC,CAAC;EACF,OAAO,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,SAAS,IAAI,CAAC;EACjE,kBAAkB,OAAO,QAAQ,UAAU,CAAC,SAAS,MAAM,UAAU,IAAI,CAAC,CAAC;EAC3E,aAAa,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,SAAS,IAAI,CAAC;EAC/E,cAAc,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,UAAU,IAAI,CAAC;EACjF,iBAAiB,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,aAAa,IAAI,CAAC;EACvF,cAAc,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,UAAU,IAAI,CAAC;EACjF,kBAAkB,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,cAAc,IAAI,CAAC;EACzF,uBAAuB,OAAO,QAAQ,UAAU,CAAC,OAAO,MAAM,CAAC,CAAC;EAChE,gCAAgC,OAAO,QAAQ,UAAU,OAAO,QAAQ,cAAc,KAAA,CAAS,CAAC,CAC7F;EACH,6BAA6B,OAAO,QAAQ,UAAU,OAAO,QAAQ,WAAW,KAAA,CAAS,CAAC,CACvF;EACH,iCAAiC,OAAO,QACrC,UAAU,OAAO,QAAQ,eAAe,KAAA,CAC3C,CAAC,CAAC;EACF;EACA,iBAAiB,OAAO,QAAQ,UAAU,CAAC,SAAS,MAAM,KAAK,SAAS,YAAY,CAAC,CAAC;CACxF;AACF;AAEA,SAAS,KAAK,QAAmC;CAC/C,OAAO,OAAO,WAAW,IAAI,IAAI,UAAU,MAAM,IAAI,OAAO;AAC9D;AAEA,SAAS,UAAU,QAAmC;CACpD,MAAM,UAAU,CAAC,GAAG,MAAM,CAAC,CAAC,MACzB,MAAM,UAAU,KAAK,IAAI,IAAI,IAAI,KAAK,IAAI,KAAK,KAAK,OAAO,KAC9D;CACA,IAAI,MAAM;CACV,IAAI,aAAa;CACjB,KAAK,MAAM,SAAS,SAAS;EAC3B,MAAM,OAAO,MAAM;EACnB,cAAc,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,KAAK,IAAI,MAAM,OAAO,QAAQ,QAAQ,OAAO;EACrF,MAAM;CACR;CACA,OAAO,MAAM;AACf;AAEA,SAAS,oBAAoB,QAA8D;CACzF,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,OAAO;EACL,KAAK,OAAO;EACZ,MAAM,KAAK,MAAM;EACjB,KAAK,WAAW,QAAQ,EAAG;EAC3B,KAAK,WAAW,QAAQ,GAAI;EAC5B,KAAK,OAAO,GAAG,EAAE;CACnB;AACF;AAEA,SAAS,WAAW,QAA2B,UAA0B;CACvE,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,OAAO,OAAO,KAAK,KAAK,WAAW,OAAO,MAAM,IAAI,MAAM,OAAO,GAAG,EAAE,KAAK;AAC7E;AAEA,SAAS,6BACP,cACA,WACyC;CACzC,MAAM,yBAAS,IAAI,IAA2C;CAC9D,KAAK,MAAM,eAAe,cAAc;EACtC,MAAM,OAAO,OAAO,IAAI,YAAY,MAAM,KAAK,CAAC;EAChD,KAAK,KAAK,WAAW;EACrB,OAAO,IAAI,YAAY,QAAQ,IAAI;CACrC;CACA,MAAM,iBAA2B,CAAC;CAClC,KAAK,MAAM,QAAQ,OAAO,OAAO,GAAG;EAClC,MAAM,aAAuB,CAAC;EAC9B,KAAK,IAAI,OAAO,GAAG,OAAO,KAAK,QAAQ,QACrC,KAAK,IAAI,QAAQ,OAAO,GAAG,QAAQ,KAAK,QAAQ,SAC9C,WAAW,KAAK,QAAQ,UAAU,KAAK,KAAM,GAAG,UAAU,KAAK,MAAO,CAAC,CAAC;EAG5E,IAAI,WAAW,SAAS,GAAG,eAAe,KAAK,KAAK,UAAU,CAAC;CACjE;CACA,OAAO;EACL,OAAO,eAAe,WAAW,IAAI,OAAO,KAAK,cAAc;EAC/D,OAAO,eAAe;CACxB;AACF;AAEA,SAAS,sBAAsB,aAA6D;CAC1F,IAAI,YAAY,OAAO,OAAO,CAAC,SAAS,YAAY,MAAM,OAAO;CACjE,OAAO,YAAY,MAAM;AAC3B;AAEA,SAAS,oBAAoB,aAA6D;CACxF,IAAI,YAAY,OAAO,OAAO,CAAC,SAAS,YAAY,MAAM,OAAO;CACjE,OAAO,YAAY,SAChB,KAAK,YACJ,KAAK,UAAU,CACb,QAAQ,YACR,QAAQ,cACL,KAAK,aAAa;EAAC,SAAS;EAAM,SAAS;EAAK,SAAS,WAAW;CAAI,CAAC,CAAC,CAC1E,MAAM,MAAM,UAAU,KAAK,UAAU,IAAI,CAAC,CAAC,cAAc,KAAK,UAAU,KAAK,CAAC,CAAC,CACpF,CAAC,CACH,CAAC,CACA,KAAK;AACV;AAEA,SAAS,QAAQ,MAAyB,OAAkC;CAC1E,MAAM,IAAI,IAAI,IAAI,IAAI;CACtB,MAAM,IAAI,IAAI,IAAI,KAAK;CACvB,MAAM,wBAAQ,IAAI,IAAI,CAAC,GAAG,GAAG,GAAG,CAAC,CAAC;CAClC,IAAI,MAAM,SAAS,GAAG,OAAO;CAC7B,IAAI,eAAe;CACnB,KAAK,MAAM,SAAS,GAAG,IAAI,EAAE,IAAI,KAAK,GAAG,gBAAgB;CACzD,OAAO,eAAe,MAAM;AAC9B;;;;;;;ACnMA,SAAgB,2BACd,UACiC;CACjC,OAAO,OAAO,EAAE,WAAW,UAAU,aAAa;EAChD,IAAI,SAAS,SAAS,QAAQ,OAAO;EACrC,MAAM,WAAW,kBAAkB,SAAS,GAAG;EAC/C,IAAI,CAAC,UAAU,OAAO;EACtB,MAAM,SAAS,MAAM,SAAS,SAAS,CAAC,CAAC,UACvC;GACE,UAAU,SAAS;GACnB,UAAU,CAAC,SAAS,MAAM;EAC5B,GACA,SAAS,EAAE,OAAO,IAAI,KAAA,CACxB;EACA,OACE,OAAO,aAAa,SAAS,WAC7B,OAAO,iBAAiB,WAAW,KACnC,OAAO,MAAM,MAAM,SAAS,KAAK,YAAY,SAAS,MAAM;CAEhE;AACF;AAgKA,eAAsB,oBACpB,SACiC;CACjC,yBAAyB,OAAO;CAChC,MAAM,6BAAY,IAAI,KAAK,EAAA,CAAE,YAAY;CACzC,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,MAAM,UAAU,cAAc,QAAQ,OAAO,QAAQ,SAAS,aAAa,eAAe;CAC1F,MAAM,iBAAiB,KAAK,IAAI,QAAQ,kBAAkB,GAAG,QAAQ,MAAM;CAC3E,MAAM,sBAAsB,4BAC1B,QAAQ,uBAAuB,CAAC,GAChC,OACF;CACA,MAAM,YAAY,IAAI,IAAI,oBAAoB,IAAI,cAAc,CAAC;CACjE,MAAM,OAAO,QAAQ,QAAQ,QAAQ,CAAC,UAAU,IAAI,OAAO,GAAG,CAAC,CAAC;CAChE,MAAM,eAA8C,CAAC,GAAG,mBAAmB;CAC3E,IAAI,SAAS;CACb,MAAM,SAAS,YAA2B;EACxC,OAAO,SAAS,KAAK,QAAQ;GAC3B,QAAQ,QAAQ,eAAe;GAC/B,MAAM,MAAM,KAAK;GACjB,MAAM,cAAc,MAAM,gBAAgB,KAAK,QAAQ,QAAQ,QAAQ,eAAe;GACtF,QAAQ,QAAQ,eAAe;GAC/B,MAAM,QAAQ,gBAAgB,WAAW;GACzC,QAAQ,QAAQ,eAAe;GAC/B,aAAa,KAAK,WAAW;EAC/B;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,gBAAgB,KAAK,MAAM,EAAE,GAAG,MAAM,CAAC;CACvF,QAAQ,QAAQ,eAAe;CAC/B,MAAM,cAAc,IAAI,IAAI,QAAQ,QAAQ,KAAK,QAAQ,UAAU,CAAC,OAAO,IAAI,KAAK,CAAC,CAAC;CACtF,MAAM,YAAY,IAAI,IAAI,QAAQ,MAAM,KAAK,UAAU,UAAU,CAAC,SAAS,IAAI,KAAK,CAAC,CAAC;CACtF,aAAa,MACV,GAAG,OACD,YAAY,IAAI,EAAE,QAAQ,KAAK,MAAM,YAAY,IAAI,EAAE,QAAQ,KAAK,OACpE,UAAU,IAAI,EAAE,MAAM,KAAK,MAAM,UAAU,IAAI,EAAE,MAAM,KAAK,MAC7D,EAAE,aAAa,EAAE,UACrB;CACA,OAAO;EACL,YAAY;GACV,GAAG,QAAQ;GACX;GACA,0BAAS,IAAI,KAAK,EAAA,CAAE,YAAY;GAChC,WAAW,QAAQ,MAAM;GACzB,WAAW,QAAQ,QAAQ,KAAK,WAAW,OAAO,EAAE;GACpD;GACA;GACA;EACF;EACA;EACA,WAAW,QAAQ,QAAQ,KAAK,WAC9B,gCACE,OAAO,IACP,aAAa,QAAQ,gBAAgB,YAAY,aAAa,OAAO,EAAE,CACzE,CACF;CACF;AACF;AAEA,SAAgB,wBAAwB,SAMK;CAC3C,OAAO;EACL,IAAI,QAAQ;EACZ,MAAM,QAAQ,OAAO,SAAS;GAC5B,MAAM,SAAS,MAAM,QAAQ,SAAS,IACpC,GAAG,QAAQ,GAAG,GAAG,QAAQ,OAAO,GAAG,QAAQ,cAC3C,OACA;IAAE,GAAG,QAAQ;IAAY,QAAQ,QAAQ;GAAO,CAClD;GACA,OAAO;IACL,UAAU,OAAO;IACjB,OAAO,mBAAmB,MAAM;IAChC,UAAU,EAAE,YAAY,OAAO;IAC/B,GAAI,QAAQ,uBAAuB,EAAE,OAAO,mBAAmB,MAAM,EAAE,IAAI,CAAC;GAC9E;EACF;CACF;AACF;AAEA,eAAe,gBACb,KAMA,QACA,iBACsC;CACtC,MAAM,UAAU,YAAY,IAAI;CAChC,IAAI;EACF,MAAM,SAAS,MAAM,IAAI,OAAO,QAAQ,IAAI,SAAS,OAAO;GAC1D,QAAQ,IAAI,SAAS;GACrB,YAAY,IAAI;GAChB;EACF,CAAC;EACD,IAAI,OAAO,OACT,+BAA+B,OAAO,OAAO,yBAAyB;EAExE,MAAM,qBAAqB,YAAY,IAAI,IAAI;EAC/C,MAAM,UAAU,wBAAwB,OAAO,mBAAmB,kBAAkB;EACpF,MAAM,iBAAiB,OAAO,QAAQ,CAAC,IAAI,OAAO;EAClD,OAAO;GACL,UAAU,IAAI,OAAO;GACrB,QAAQ,IAAI,SAAS;GACrB,WAAW,IAAI,SAAS;GACxB,YAAY,IAAI,SAAS;GACzB,YAAY,IAAI;GAChB,gBAAgB,IAAI;GACpB,WAAW,QAAQ;GACnB,eAAe,QAAQ;GACvB,UAAU,OAAO;GACjB,OAAO,qBAAqB,IAAI,UAAU,cAAc;GACxD,oBAAoB,kBAChB,MAAM,uBAAuB,IAAI,UAAU,OAAO,UAAU,iBAAiB,MAAM,IACnF,KAAA;GACJ,UAAU,CAAC,GAAI,IAAI,SAAS,QAAQ,CAAC,CAAE;GACvC,cAAc,IAAI,SAAS;GAC3B,OAAO,OAAO;GACd,gBAAgB,OAAO;GACvB,GAAI,OAAO,QAAQ,EAAE,OAAO,OAAO,MAAM,IAAI,CAAC;EAChD;CACF,SAAS,OAAO;EACd,IAAI,QAAQ,SAAS,MAAM;EAC3B,MAAM,WAA6B,CAAC;EACpC,OAAO;GACL,UAAU,IAAI,OAAO;GACrB,QAAQ,IAAI,SAAS;GACrB,WAAW,IAAI,SAAS;GACxB,YAAY,IAAI,SAAS;GACzB,YAAY,IAAI;GAChB,gBAAgB,IAAI;GACpB,WAAW,YAAY,IAAI,IAAI;GAC/B,eAAe;GACf;GACA,OAAO,qBAAqB,IAAI,UAAU,QAAQ;GAClD,UAAU,CAAC,GAAI,IAAI,SAAS,QAAQ,CAAC,CAAE;GACvC,cAAc,IAAI,SAAS;GAC3B,OAAO;IACL,OAAO,iBAAiB,QAAQ,MAAM,YAAY,OAAO;IACzD,SAAS,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;GAChE;EACF;CACF;AACF;AAEA,SAAS,wBACP,mBACA,YAIA;CACA,IAAI,sBAAsB,KAAA,GACxB,OAAO;EAAE,OAAO;EAAY,QAAQ;CAAkB;CAExD,IAAI,sBAAsB,MAAM,OAAO;EAAE,OAAO;EAAM,QAAQ;CAAa;CAC3E,IAAI,CAAC,OAAO,SAAS,iBAAiB,KAAK,oBAAoB,GAC7D,MAAM,IAAI,WAAW,qEAAqE;CAE5F,OAAO;EAAE,OAAO;EAAmB,QAAQ;CAAkB;AAC/D;AAEA,SAAS,mBAAmB,QAAuE;CACjG,MAAM,SAAS,OAAO,YAAY,QAAQ,YAAY,QAAQ,WAAW,QAAQ;CACjF,IAAI,OAAO,WAAW,GAAG,OAAO,KAAA;CAChC,OAAO;EACL,OAAO;EACP,SAAS,OACN,KACE,YACC,GAAG,QAAQ,WAAW,IAAI,QAAQ,OAAO,SAAS,QAAQ,IAAI,QAAQ,OAAO,WAAW,kBAC5F,CAAC,CACA,KAAK,IAAI;CACd;AACF;AAEA,SAAS,cACP,OACA,SACA,aACA,iBAMC;CACD,MAAM,gBAAgB,CAAC,GAAG,OAAO,CAAC,CAAC,MAChC,MAAM,UACL,WAAW,GAAG,gBAAgB,QAAQ,KAAK,IAAI,IAC7C,WAAW,GAAG,gBAAgB,QAAQ,MAAM,IAAI,KAAK,KAAK,GAAG,cAAc,MAAM,EAAE,CACzF;CACA,IAAI,iBAAiB;CACrB,OAAO,MAAM,SAAS,UAAU,cAC9B,MAAM,KAAK,EAAE,QAAQ,YAAY,IAAI,GAAG,eAAe;EAErD,MAAM,YADa,YAAY,cAAc,cACf,cAAc;EAE5C,OAAO,CADU,GAAG,cAAc,MAAM,QAAQ,GAAG,GAAG,cAAc,MAAM,GAAG,QAAQ,CACxE,CAAC,CAAC,KAAK,YAAY;GAC9B;GACA;GACA;GACA,gBAAgB;EAClB,EAAE;CACJ,CAAC,CAAC,CAAC,KAAK,CACV;AACF;AAEA,SAAS,4BACP,cACA,MAM+B;CAC/B,MAAM,WAAW,IAAI,IAAI,KAAK,KAAK,QAAQ,CAAC,OAAO,GAAG,GAAG,GAAG,CAAC,CAAC;CAC9D,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,aAAa,KAAK,gBAAgB;EACvC,MAAM,MAAM,eAAe,WAAW;EACtC,IAAI,KAAK,IAAI,GAAG,GACd,MAAM,IAAI,UACR,oDAAoD,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,EAC3H;EAEF,KAAK,IAAI,GAAG;EACZ,MAAM,MAAM,SAAS,IAAI,GAAG;EAC5B,IAAI,CAAC,KACH,MAAM,IAAI,UACR,wEAAwE,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,EAC/I;EAEF,IAAI,YAAY,mBAAmB,IAAI,gBACrC,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,uBAAuB,YAAY,eAAe,aAAa,IAAI,gBACpL;EAEF,IACE,YAAY,cAAc,IAAI,SAAS,aACvC,YAAY,eAAe,IAAI,SAAS,YAExC,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,yCACjH;EAEF,IACE,KAAK,UAAU,YAAY,QAAQ,MAAM,KAAK,UAAU,IAAI,SAAS,QAAQ,CAAC,CAAC,KAC/E,KAAK,UAAU,YAAY,YAAY,MAAM,KAAK,UAAU,IAAI,SAAS,QAAQ,GAEjF,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,2CACjH;EAEF,MAAM,gBAAgB,qBACpB,IAAI,UACJ,YAAY,QAAQ,CAAC,IAAI,YAAY,QACvC;EACA,IAAI,KAAK,UAAU,YAAY,KAAK,MAAM,KAAK,UAAU,aAAa,GACpE,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,8BACjH;EAEF,IAAI,YAAY,OACd,+BACE,YAAY,OACZ,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,QACjH;EAEF,IACE,CAAC;GAAC;GAAmB;GAAmB;EAAY,CAAC,CAAC,SAAS,YAAY,aAAa,KACvF,YAAY,kBAAkB,gBAAgB,YAAY,cAAc,QACxE,YAAY,kBAAkB,iBAC5B,YAAY,cAAc,QACzB,CAAC,OAAO,SAAS,YAAY,SAAS,KACtC,YAAY,YAAY,IAE5B,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,sBACjH;EAEF,OAAO,EAAE,GAAG,YAAY;CAC1B,CAAC;AACH;AAEA,SAAS,OAAO,KAIL;CACT,OAAO,GAAG,IAAI,OAAO,GAAG,QAAQ,IAAI,SAAS,GAAG,QAAQ,IAAI;AAC9D;AAEA,SAAS,eAAe,aAIb;CACT,OAAO,GAAG,YAAY,SAAS,QAAQ,YAAY,OAAO,QAAQ,YAAY;AAChF;AAEA,eAAe,uBACb,UACA,UACA,UACA,QACoC;CACpC,MAAM,WAAW,SAAS,SAAS,YAAY,QAAQ,aAAa;CACpE,MAAM,WAA0B,CAAC;CACjC,MAAM,qBAAoC,CAAC;CAC3C,MAAM,SAA2C,CAAC;CAClD,KAAK,MAAM,OAAO,UAAU;EAC1B,QAAQ,eAAe;EACvB,IAAI;GACF,IACE,MAAM,SAAS;IACb,QAAQ,SAAS;IACjB,WAAW,SAAS;IACpB,UAAU;IACV;GACF,CAAC,GAED,SAAS,KAAK,GAAG;QAEjB,mBAAmB,KAAK,GAAG;EAE/B,SAAS,OAAO;GACd,IAAI,QAAQ,SAAS,MAAM;GAC3B,OAAO,KAAK;IACV,UAAU;IACV,OAAO,iBAAiB,QAAQ,MAAM,YAAY,OAAO;IACzD,SAAS,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;GAChE,CAAC;EACH;CACF;CACA,OAAO;EACL,SAAS,SAAS;EAClB,UAAU,SAAS;EACnB;EACA;EACA,UAAU,SAAS,WAAW,KAAK,OAAO,SAAS,IAAI,OAAO,SAAS,SAAS,SAAS;CAC3F;AACF;AAEA,SAAS,kBAAkB,KAAyD;CAClF,MAAM,QAAQ,qCAAqC,KAAK,GAAG;CAC3D,IAAI,CAAC,OAAO,OAAO;CACnB,IAAI;EACF,MAAM,UAAU,mBAAmB,MAAM,EAAG;EAC5C,MAAM,SAAS,mBAAmB,MAAM,EAAG;EAC3C,OAAO,WAAW,SAAS;GAAE;GAAS;EAAO,IAAI;CACnD,QAAQ;EACN,OAAO;CACT;AACF;AAEA,SAAS,yBAAiC,SAAmD;CAC3F,IAAI,QAAQ,MAAM,WAAW,GAAG,MAAM,IAAI,UAAU,oCAAoC;CACxF,IAAI,QAAQ,QAAQ,WAAW,GAAG,MAAM,IAAI,UAAU,sCAAsC;CAC5F,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,IAAI,CAAC,OAAO,cAAc,WAAW,KAAK,cAAc,GACtD,MAAM,IAAI,WAAW,iEAAiE;CAExF,IAAI,CAAC,OAAO,cAAc,cAAc,KAAK,iBAAiB,GAC5D,MAAM,IAAI,WAAW,oEAAoE;CAE3F,IAAI,CAAC,OAAO,cAAc,QAAQ,mBAAmB,CAAC,GACpD,MAAM,IAAI,WAAW,4DAA4D;CAEnF,qBACE,QAAQ,MAAM,KAAK,aAAa,SAAS,EAAE,GAC3C,MACF;CACA,qBACE,QAAQ,QAAQ,KAAK,WAAW,OAAO,EAAE,GACzC,QACF;CACA,KAAK,MAAM,YAAY,QAAQ,OAAO,sBAAsB,QAAQ;AACtE;AAEA,SAAS,sBAAsB,UAAsC;CACnE,8BAA8B,QAAQ;CACtC,IAAI,CAAC,SAAS,UAAU,KAAK,GAC3B,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,gDAAgD;CAErF,IACE,SAAS,eAAe,cACxB,SAAS,eAAe,sBACxB,SAAS,eAAe,aAExB,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,0CAA0C;CAE/E,IAAI,SAAS,eAAe,cAAc,SAAS,eAAe,WAAW,GAC3E,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,qDAAqD;CAE1F,IAAI,SAAS,eAAe,cAAc,SAAS,eAAe,SAAS,GACzE,MAAM,IAAI,UACR,GAAG,SAAS,GAAG,IAAI,SAAS,WAAW,qCACzC;AAEJ;AAEA,SAAS,qBAAqB,QAA2B,OAAqB;CAC5E,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,SAAS,QAAQ;EAC1B,IAAI,CAAC,MAAM,KAAK,GAAG,MAAM,IAAI,UAAU,qBAAqB,MAAM,sBAAsB;EACxF,IAAI,KAAK,IAAI,KAAK,GAAG,MAAM,IAAI,UAAU,+BAA+B,MAAM,OAAO,MAAM,EAAE;EAC7F,KAAK,IAAI,KAAK;CAChB;AACF;AAEA,SAAS,WAAW,OAAuB;CACzC,IAAI,OAAO;CACX,KAAK,IAAI,QAAQ,GAAG,QAAQ,MAAM,QAAQ,SAAS,GAAG;EACpD,QAAQ,MAAM,WAAW,KAAK;EAC9B,OAAO,KAAK,KAAK,MAAM,QAAQ;CACjC;CACA,OAAO,SAAS;AAClB;AAEA,SAAS,mBAAmB,QAA+C;CACzE,MAAM,SAAS,OAAO,YAAY,KAAK,YAAY,QAAQ,KAAK;CAChE,MAAM,QAAQ,OAAO,OAAO,UAAU,MAAM,UAAU,IAAI,IACtD,OAAO,QAAQ,KAAK,UAAU,OAAO,MAAM,SAAS,IAAI,CAAC,IACzD;CACJ,MAAM,SAAS,OAAO,OAAO,UAAU,MAAM,WAAW,IAAI,IACxD,OAAO,QACJ,KAAK,WAAW;EACf,OAAO,IAAI,SAAS,MAAM,QAAQ,SAAS;EAC3C,QAAQ,IAAI,UAAU,MAAM,QAAQ,UAAU;EAC9C,WAAW,IAAI,aAAa,MAAM,QAAQ,aAAa;EACvD,QAAQ,IAAI,UAAU,MAAM,QAAQ,UAAU;EAC9C,YAAY,IAAI,cAAc,MAAM,QAAQ,cAAc;CAC5D,IACA;EAAE,OAAO;EAAG,QAAQ;EAAG,WAAW;EAAG,QAAQ;EAAG,YAAY;CAAE,CAChE,IACA;CACJ,MAAM,eAAe,OAAO,QACzB,KAAK,UACJ,OAAO,MAAM,KAAK,SAAS,eAAgB,MAAM,gBAAgB,IAAK,MAAM,KAAK,MACnF,CACF;CACA,MAAM,OAAO,OAAO,MAAM,UAAU,MAAM,KAAK,SAAS,YAAY,IAC/D;EAAE,MAAM;EAAc,KAAK;CAAK,IACjC,OAAO,MAAM,UAAU,MAAM,KAAK,SAAS,WAAW,IACnD;EAAE,MAAM;EAAa,KAAK;CAAa,IACvC;EAAE,MAAM;EAAY,KAAK;CAAa;CAC7C,OAAO;EACL;EACA;EACA;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,aAAa,IAAI,CAAC;CACvD;AACF"}
1
+ {"version":3,"file":"benchmark-CWeqGl7x.js","names":[],"sources":["../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark-summary.ts","../src/analyst/benchmark.ts"],"sourcesContent":["import { linearSumAssignment } from 'linear-sum-assignment'\nimport type {\n AnalystBenchmarkCase,\n AnalystEvidenceExpectation,\n AnalystFindingScore,\n AnalystIssueExpectation,\n} from './benchmark'\nimport type { AnalystFinding, EvidenceRef } from './types'\n\nexport function scoreAnalystFindings(\n testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>,\n findings: readonly AnalystFinding[],\n): AnalystFindingScore {\n assertValidAnalystScoringCase(testCase)\n const matchedFindingByIssue = matchFindingsToIssues(testCase.expectedIssues, findings)\n const matchedIssueIds = testCase.expectedIssues\n .filter((_, index) => matchedFindingByIssue.has(index))\n .map((issue) => issue.id)\n const missedIssueIds = testCase.expectedIssues\n .filter((_, index) => !matchedFindingByIssue.has(index))\n .map((issue) => issue.id)\n const supportedFindingIndexes = new Set(matchedFindingByIssue.values())\n\n const unsupportedFindingIndexes = findings\n .map((_, index) => index)\n .filter((index) => !supportedFindingIndexes.has(index))\n const expectedIssueCount = testCase.expectedIssues.length\n const issueRecall = expectedIssueCount === 0 ? 1 : matchedIssueIds.length / expectedIssueCount\n const findingPrecision =\n findings.length === 0\n ? expectedIssueCount === 0\n ? 1\n : 0\n : supportedFindingIndexes.size / findings.length\n const f1 = harmonicMeanScore(findingPrecision, issueRecall)\n const allEvidence = findings.flatMap((finding) => finding.evidence_refs)\n\n const criticalIssues = testCase.expectedIssues.filter(\n (issue) => (issue.criticalEvidence?.length ?? 0) > 0,\n )\n const criticalHits = testCase.expectedIssues.filter((issue) => {\n if ((issue.criticalEvidence?.length ?? 0) === 0) return false\n return matchesEvidence(allEvidence, issue.criticalEvidence ?? [], 'any')\n }).length\n\n const findingsWithEvidence = findings.filter((finding) => finding.evidence_refs.length > 0).length\n const unlabeledEvidence = testCase.labeledEvidence\n ? allEvidence.filter(\n (ref) => !testCase.labeledEvidence!.some((expected) => evidenceMatches(ref, expected)),\n )\n : []\n\n return {\n expectedIssueCount,\n matchedIssueIds,\n missedIssueIds,\n supportedFindingIndexes: [...supportedFindingIndexes].sort((a, b) => a - b),\n unsupportedFindingIndexes,\n unlabeledEvidence,\n issueRecall,\n findingPrecision,\n f1,\n criticalStepAccuracy: criticalIssues.length === 0 ? null : criticalHits / criticalIssues.length,\n citationCoverage: findings.length === 0 ? null : findingsWithEvidence / findings.length,\n citationExcerptCoverage:\n allEvidence.length === 0\n ? null\n : allEvidence.filter((evidence) => Boolean(evidence.excerpt?.trim())).length /\n allEvidence.length,\n citationLabelAgreement:\n testCase.labeledEvidence === undefined\n ? null\n : allEvidence.length === 0\n ? findings.length === 0\n ? null\n : 0\n : (allEvidence.length - unlabeledEvidence.length) / allEvidence.length,\n predictionOnLabelEmptyCase: expectedIssueCount === 0 && findings.length > 0,\n }\n}\n\nexport function assertValidAnalystScoringCase(\n testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>,\n): void {\n if (!testCase.id.trim()) throw new TypeError('analyst benchmark case id must not be empty')\n const ids = new Set<string>()\n for (const issue of testCase.expectedIssues) {\n if (!issue.id.trim()) throw new TypeError(`${testCase.id}: expected issue id must not be empty`)\n if (ids.has(issue.id)) {\n throw new TypeError(`${testCase.id}: duplicate expected issue id '${issue.id}'`)\n }\n ids.add(issue.id)\n if (\n !issue.findingIds?.length &&\n !issue.areas?.length &&\n !issue.subjects?.length &&\n !issue.evidence?.length\n ) {\n throw new TypeError(\n `${testCase.id}/${issue.id}: expected issue must identify a finding by id, area, subject, or evidence`,\n )\n }\n }\n for (const ref of testCase.labeledEvidence ?? []) {\n if (!ref.uri.trim()) {\n throw new TypeError(`${testCase.id}: labeled evidence URI must not be empty`)\n }\n }\n}\n\nexport function harmonicMeanScore(a: number, b: number): number {\n return a + b === 0 ? 0 : (2 * a * b) / (a + b)\n}\n\nfunction matchFindingsToIssues(\n issues: readonly AnalystIssueExpectation[],\n findings: readonly AnalystFinding[],\n): Map<number, number> {\n if (issues.length === 0) return new Map()\n const cardinalityWeight = issues.length + 1\n const scores = issues.map((issue) => [\n ...findings.map((finding) => {\n if (!findingMatchesIssue(finding, issue)) return -1\n const criticalHit =\n (issue.criticalEvidence?.length ?? 0) > 0 &&\n matchesEvidence(finding.evidence_refs, issue.criticalEvidence ?? [], 'any')\n return cardinalityWeight + Number(criticalHit)\n }),\n ...Array.from({ length: issues.length }, () => 0),\n ])\n const assignment = linearSumAssignment(scores, { maximaze: true }).rowAssignments\n const matches = new Map<number, number>()\n for (const [issueIndex, column] of assignment.entries()) {\n if (column < 0 || column >= findings.length) continue\n if (!findingMatchesIssue(findings[column]!, issues[issueIndex]!)) continue\n matches.set(issueIndex, column)\n }\n return matches\n}\n\nfunction findingMatchesIssue(finding: AnalystFinding, issue: AnalystIssueExpectation): boolean {\n if (issue.findingIds && !issue.findingIds.includes(finding.finding_id)) return false\n if (issue.areas && !issue.areas.includes(finding.area)) return false\n if (issue.subjects && (!finding.subject || !issue.subjects.includes(finding.subject))) {\n return false\n }\n if (\n issue.evidence &&\n !matchesEvidence(finding.evidence_refs, issue.evidence, issue.evidenceMode ?? 'any')\n ) {\n return false\n }\n return true\n}\n\nfunction matchesEvidence(\n actual: readonly EvidenceRef[],\n expected: readonly AnalystEvidenceExpectation[],\n mode: 'any' | 'all',\n): boolean {\n if (expected.length === 0) return true\n const match = (target: AnalystEvidenceExpectation) =>\n actual.some((ref) => evidenceMatches(ref, target))\n return mode === 'all' ? expected.every(match) : expected.some(match)\n}\n\nfunction evidenceMatches(actual: EvidenceRef, expected: AnalystEvidenceExpectation): boolean {\n return (\n actual.uri === expected.uri && (expected.kind === undefined || actual.kind === expected.kind)\n )\n}\n","import type {\n AnalystBenchmarkObservation,\n AnalystBenchmarkSummary,\n AnalystLatencyDistribution,\n} from './benchmark'\nimport { harmonicMeanScore } from './benchmark-scoring'\n\nexport function summarizeAnalystBenchmarkRunner(\n runnerId: string,\n observations: readonly AnalystBenchmarkObservation[],\n): AnalystBenchmarkSummary {\n const issueBearing = observations.filter((observation) => observation.labelState === 'positive')\n const completed = observations.filter((observation) => !observation.error)\n const expectedIssues = issueBearing.reduce(\n (sum, observation) => sum + observation.score.expectedIssueCount,\n 0,\n )\n const matchedIssues = issueBearing.reduce(\n (sum, observation) => sum + observation.score.matchedIssueIds.length,\n 0,\n )\n const issueFindings = issueBearing.reduce(\n (sum, observation) => sum + (observation.error ? 0 : observation.findings.length),\n 0,\n )\n const supportedFindings = issueBearing.reduce(\n (sum, observation) => sum + observation.score.supportedFindingIndexes.length,\n 0,\n )\n const issueRecall = expectedIssues === 0 ? null : matchedIssues / expectedIssues\n const findingPrecision =\n expectedIssues === 0 ? null : issueFindings === 0 ? 0 : supportedFindings / issueFindings\n const macroIssueRecall =\n issueBearing.length === 0\n ? null\n : mean(issueBearing.map((observation) => observation.score.issueRecall))\n const macroFindingPrecision =\n issueBearing.length === 0\n ? null\n : mean(issueBearing.map((observation) => observation.score.findingPrecision))\n const macroF1 =\n issueBearing.length === 0 ? null : mean(issueBearing.map((observation) => observation.score.f1))\n const critical = observations\n .map((observation) => observation.score.criticalStepAccuracy)\n .filter((value): value is number => value !== null)\n const findingsWithEvidence = completed.reduce(\n (sum, observation) =>\n sum + observation.findings.filter((finding) => finding.evidence_refs.length > 0).length,\n 0,\n )\n const allFindings = completed.reduce((sum, observation) => sum + observation.findings.length, 0)\n const allCitations = completed.flatMap((observation) =>\n observation.findings.flatMap((finding) => finding.evidence_refs),\n )\n const citationObservations = completed.filter(\n (observation) => observation.score.citationLabelAgreement !== null,\n )\n const citationCount = citationObservations.reduce(\n (sum, observation) =>\n sum +\n observation.findings.reduce((count, finding) => count + finding.evidence_refs.length, 0),\n 0,\n )\n const invalidCitationCount = citationObservations.reduce(\n (sum, observation) => sum + observation.score.unlabeledEvidence.length,\n 0,\n )\n const trustedNegative = observations.filter(\n (observation) => observation.labelState === 'trusted-negative',\n )\n const completedTrustedNegative = trustedNegative.filter((observation) => !observation.error)\n const unlabeled = observations.filter((observation) => observation.labelState === 'unlabeled')\n const completedUnlabeled = unlabeled.filter((observation) => !observation.error)\n const resolvedCitations = completed.reduce(\n (sum, observation) => sum + (observation.evidenceResolution?.resolved ?? 0),\n 0,\n )\n const unresolvedCitations = completed.reduce(\n (sum, observation) => sum + (observation.evidenceResolution?.unresolvedEvidence.length ?? 0),\n 0,\n )\n const citationResolutionErrors = completed.reduce(\n (sum, observation) => sum + (observation.evidenceResolution?.errors.length ?? 0),\n 0,\n )\n const resolutionAttempts = completed.filter((observation) =>\n observation.findings.some((finding) => finding.evidence_refs.length > 0),\n )\n const citationResolutionUnknownRuns = resolutionAttempts.filter(\n (observation) =>\n !observation.evidenceResolution || observation.evidenceResolution.errors.length > 0,\n ).length\n const usages = observations.map((observation) => observation.usage)\n const knownCostUsd = stableSum(\n usages.map((usage) => {\n if (!usage) return 0\n return usage.cost.kind === 'uncaptured' ? (usage.knownCostUsd ?? 0) : usage.cost.usd\n }),\n )\n const predictionAgreement = repeatedObservationAgreement(observations, predictionSignature)\n const matchedLabelAgreement = repeatedObservationAgreement(\n observations.filter((observation) => observation.labelState === 'positive'),\n matchedLabelSignature,\n )\n return {\n runnerId,\n plannedRuns: observations.length,\n completedRuns: observations.filter((observation) => !observation.error).length,\n failedRuns: observations.filter((observation) => Boolean(observation.error)).length,\n issueBearingRuns: issueBearing.length,\n trustedNegativeRuns: trustedNegative.length,\n unlabeledRuns: unlabeled.length,\n issueRecall,\n findingPrecision,\n f1:\n findingPrecision === null || issueRecall === null\n ? null\n : harmonicMeanScore(findingPrecision, issueRecall),\n macroIssueRecall,\n macroFindingPrecision,\n macroF1,\n criticalStepAccuracy: critical.length === 0 ? null : mean(critical),\n citationCoverage: allFindings === 0 ? null : findingsWithEvidence / allFindings,\n citationExcerptCoverage:\n allCitations.length === 0\n ? null\n : allCitations.filter((evidence) => Boolean(evidence.excerpt?.trim())).length /\n allCitations.length,\n citationLabelAgreement:\n citationObservations.length === 0\n ? null\n : citationCount === 0\n ? 0\n : (citationCount - invalidCitationCount) / citationCount,\n citationResolution:\n resolutionAttempts.length === 0 ||\n citationResolutionUnknownRuns > 0 ||\n resolvedCitations + unresolvedCitations === 0\n ? null\n : resolvedCitations / (resolvedCitations + unresolvedCitations),\n citationResolutionUnknownRuns,\n unresolvedCitations,\n citationResolutionErrors,\n trustedNegativeFalsePositiveRate:\n completedTrustedNegative.length === 0\n ? null\n : completedTrustedNegative.filter(\n (observation) => observation.score.predictionOnLabelEmptyCase,\n ).length / completedTrustedNegative.length,\n trustedNegativeFailureRate:\n trustedNegative.length === 0\n ? null\n : trustedNegative.filter((observation) => Boolean(observation.error)).length /\n trustedNegative.length,\n unlabeledPredictionRate:\n completedUnlabeled.length === 0\n ? null\n : completedUnlabeled.filter((observation) => observation.findings.length > 0).length /\n completedUnlabeled.length,\n unlabeledFailureRate:\n unlabeled.length === 0\n ? null\n : unlabeled.filter((observation) => Boolean(observation.error)).length / unlabeled.length,\n predictionAgreement: predictionAgreement.value,\n predictionAgreementCases: predictionAgreement.cases,\n matchedLabelAgreement: matchedLabelAgreement.value,\n matchedLabelAgreementCases: matchedLabelAgreement.cases,\n latencyMs: latencyDistribution(\n observations\n .map((observation) => observation.latencyMs)\n .filter((value): value is number => value !== null),\n ),\n benchmarkClockLatencyRuns: observations.filter(\n (observation) => observation.latencySource === 'benchmark-clock',\n ).length,\n runnerReportedLatencyRuns: observations.filter(\n (observation) => observation.latencySource === 'runner-reported',\n ).length,\n latencyUnknownRuns: observations.filter(\n (observation) => observation.latencySource === 'uncaptured',\n ).length,\n calls: usages.reduce((sum, usage) => sum + (usage?.calls ?? 0), 0),\n callsUnknownRuns: usages.filter((usage) => !usage || usage.calls === null).length,\n inputTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.input ?? 0), 0),\n outputTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.output ?? 0), 0),\n reasoningTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.reasoning ?? 0), 0),\n cachedTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.cached ?? 0), 0),\n cacheWriteTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.cacheWrite ?? 0), 0),\n tokenUsageUnknownRuns: usages.filter((usage) => !usage?.tokens).length,\n reasoningTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.reasoning === undefined)\n .length,\n cachedTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.cached === undefined)\n .length,\n cacheWriteTokenUsageUnknownRuns: usages.filter(\n (usage) => usage?.tokens?.cacheWrite === undefined,\n ).length,\n knownCostUsd,\n costUnknownRuns: usages.filter((usage) => !usage || usage.cost.kind === 'uncaptured').length,\n }\n}\n\nfunction mean(values: readonly number[]): number {\n return values.length === 0 ? 0 : stableSum(values) / values.length\n}\n\nfunction stableSum(values: readonly number[]): number {\n const ordered = [...values].sort(\n (left, right) => Math.abs(left) - Math.abs(right) || left - right,\n )\n let sum = 0\n let correction = 0\n for (const value of ordered) {\n const next = sum + value\n correction += Math.abs(sum) >= Math.abs(value) ? sum - next + value : value - next + sum\n sum = next\n }\n return sum + correction\n}\n\nfunction latencyDistribution(values: readonly number[]): AnalystLatencyDistribution | null {\n if (values.length === 0) return null\n const sorted = [...values].sort((a, b) => a - b)\n return {\n min: sorted[0]!,\n mean: mean(sorted),\n p50: percentile(sorted, 0.5),\n p95: percentile(sorted, 0.95),\n max: sorted.at(-1)!,\n }\n}\n\nfunction percentile(sorted: readonly number[], quantile: number): number {\n if (sorted.length === 0) return 0\n return sorted[Math.ceil(quantile * sorted.length) - 1] ?? sorted.at(-1) ?? 0\n}\n\nfunction repeatedObservationAgreement(\n observations: readonly AnalystBenchmarkObservation[],\n signature: (observation: AnalystBenchmarkObservation) => readonly string[],\n): { value: number | null; cases: number } {\n const byCase = new Map<string, AnalystBenchmarkObservation[]>()\n for (const observation of observations) {\n const rows = byCase.get(observation.caseId) ?? []\n rows.push(observation)\n byCase.set(observation.caseId, rows)\n }\n const caseAgreements: number[] = []\n for (const rows of byCase.values()) {\n const agreements: number[] = []\n for (let left = 0; left < rows.length; left++) {\n for (let right = left + 1; right < rows.length; right++) {\n agreements.push(jaccard(signature(rows[left]!), signature(rows[right]!)))\n }\n }\n if (agreements.length > 0) caseAgreements.push(mean(agreements))\n }\n return {\n value: caseAgreements.length === 0 ? null : mean(caseAgreements),\n cases: caseAgreements.length,\n }\n}\n\nfunction matchedLabelSignature(observation: AnalystBenchmarkObservation): readonly string[] {\n if (observation.error) return [`error:${observation.error.class}`]\n return observation.score.matchedIssueIds\n}\n\nfunction predictionSignature(observation: AnalystBenchmarkObservation): readonly string[] {\n if (observation.error) return [`error:${observation.error.class}`]\n return observation.findings\n .map((finding) =>\n JSON.stringify([\n finding.finding_id,\n finding.evidence_refs\n .map((evidence) => [evidence.kind, evidence.uri, evidence.excerpt ?? null])\n .sort((left, right) => JSON.stringify(left).localeCompare(JSON.stringify(right))),\n ]),\n )\n .sort()\n}\n\nfunction jaccard(left: readonly string[], right: readonly string[]): number {\n const a = new Set(left)\n const b = new Set(right)\n const union = new Set([...a, ...b])\n if (union.size === 0) return 1\n let intersection = 0\n for (const value of a) if (b.has(value)) intersection += 1\n return intersection / union.size\n}\n","import { performance } from 'node:perf_hooks'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport { assertValidAnalystScoringCase, scoreAnalystFindings } from './benchmark-scoring'\nimport { summarizeAnalystBenchmarkRunner } from './benchmark-summary'\nimport type { AnalystRegistry, RegistryRunOpts } from './registry'\nimport type {\n AnalystFinding,\n AnalystRunInputs,\n AnalystRunResult,\n AnalystUsageReceipt,\n EvidenceRef,\n} from './types'\nimport { assertValidAnalystUsageReceipt } from './usage-receipt'\n\nexport { scoreAnalystFindings } from './benchmark-scoring'\n\nexport interface AnalystEvidenceExpectation {\n uri: string\n kind?: EvidenceRef['kind']\n}\n\nexport interface AnalystIssueExpectation {\n id: string\n findingIds?: readonly string[]\n areas?: readonly string[]\n subjects?: readonly string[]\n evidence?: readonly AnalystEvidenceExpectation[]\n evidenceMode?: 'any' | 'all'\n /** Exact evidence location for the first unrecoverable or causal step. */\n criticalEvidence?: readonly AnalystEvidenceExpectation[]\n}\n\nexport type AnalystBenchmarkLabelState = 'positive' | 'trusted-negative' | 'unlabeled'\n\nexport interface AnalystBenchmarkCase<TInput = unknown> {\n id: string\n /** Independent source unit used for resampling, such as a task or incident. */\n clusterId: string\n /** Whether labels prove an issue, prove no issue, or leave the outcome unknown. */\n labelState: AnalystBenchmarkLabelState\n input: TInput\n expectedIssues: readonly AnalystIssueExpectation[]\n /** Complete set of labeled locations used to measure label-location agreement. */\n labeledEvidence?: readonly AnalystEvidenceExpectation[]\n tags?: readonly string[]\n metadata?: Record<string, unknown>\n}\n\nexport interface AnalystFindingScore {\n expectedIssueCount: number\n matchedIssueIds: string[]\n missedIssueIds: string[]\n supportedFindingIndexes: number[]\n unsupportedFindingIndexes: number[]\n unlabeledEvidence: EvidenceRef[]\n issueRecall: number\n findingPrecision: number\n f1: number\n criticalStepAccuracy: number | null\n /** Share of findings that cite at least one evidence location. */\n citationCoverage: number | null\n /** Share of citations that include a non-empty source excerpt. */\n citationExcerptCoverage: number | null\n /** Share of citations that agree with a labeled case location. */\n citationLabelAgreement: number | null\n predictionOnLabelEmptyCase: boolean\n}\n\nexport interface AnalystEvidenceResolutionError {\n evidence: EvidenceRef\n class: string\n message: string\n}\n\nexport interface AnalystEvidenceResolution {\n checked: number\n resolved: number\n unresolvedEvidence: EvidenceRef[]\n errors: AnalystEvidenceResolutionError[]\n /** Null when no citations were checked or any resolution attempt failed. */\n validity: number | null\n}\n\nexport type AnalystEvidenceResolver<TInput = unknown> = (input: {\n caseId: string\n caseInput: TInput\n evidence: EvidenceRef\n signal?: AbortSignal\n}) => boolean | Promise<boolean>\n\n/**\n * Resolve canonical `trace://<trace>/span/<span>` evidence against a trace store.\n * Other evidence kinds and URI schemes require a caller-supplied resolver.\n */\nexport function traceStoreEvidenceResolver<TInput>(\n getStore: (input: TInput) => TraceAnalysisStore,\n): AnalystEvidenceResolver<TInput> {\n return async ({ caseInput, evidence, signal }) => {\n if (evidence.kind !== 'span') return false\n const location = parseTraceSpanUri(evidence.uri)\n if (!location) return false\n const result = await getStore(caseInput).viewSpans(\n {\n trace_id: location.traceId,\n span_ids: [location.spanId],\n },\n signal ? { signal } : undefined,\n )\n return (\n result.trace_id === location.traceId &&\n result.missing_span_ids.length === 0 &&\n result.spans.some((span) => span.span_id === location.spanId)\n )\n }\n}\n\nexport interface AnalystBenchmarkOutput {\n findings: readonly AnalystFinding[]\n usage?: AnalystUsageReceipt\n metadata?: Record<string, unknown>\n /**\n * End-to-end duration measured by an external runner before import.\n * Use null when the source explicitly did not capture duration.\n */\n observedLatencyMs?: number | null\n /** Marks a completed transport as a failed analyst run while retaining usage and metadata. */\n error?: AnalystBenchmarkError\n}\n\nexport interface AnalystBenchmarkError {\n class: string\n message: string\n code?: string\n status?: number\n}\n\nexport interface AnalystBenchmarkRunner<TInput = unknown> {\n id: string\n analyze(\n input: TInput,\n context: { caseId: string; repetition: number; signal?: AbortSignal },\n ): AnalystBenchmarkOutput | Promise<AnalystBenchmarkOutput>\n}\n\nexport interface AnalystBenchmarkObservation {\n runnerId: string\n caseId: string\n clusterId: string\n labelState: AnalystBenchmarkLabelState\n repetition: number\n executionIndex: number\n latencyMs: number | null\n latencySource: 'benchmark-clock' | 'runner-reported' | 'uncaptured'\n findings: readonly AnalystFinding[]\n score: AnalystFindingScore\n evidenceResolution?: AnalystEvidenceResolution\n caseTags: readonly string[]\n caseMetadata?: Record<string, unknown>\n usage?: AnalystUsageReceipt\n runnerMetadata?: Record<string, unknown>\n error?: AnalystBenchmarkError\n}\n\nexport interface AnalystLatencyDistribution {\n min: number\n mean: number\n p50: number\n p95: number\n max: number\n}\n\nexport interface AnalystBenchmarkSummary {\n runnerId: string\n plannedRuns: number\n completedRuns: number\n failedRuns: number\n issueBearingRuns: number\n trustedNegativeRuns: number\n unlabeledRuns: number\n /** Pooled across all labeled issues and findings. */\n issueRecall: number | null\n /** Pooled across all labeled issues and findings. */\n findingPrecision: number | null\n /** Harmonic mean of the pooled precision and recall. */\n f1: number | null\n /** Mean of per-case recall over issue-bearing runs. */\n macroIssueRecall: number | null\n /** Mean of per-case precision over issue-bearing runs. */\n macroFindingPrecision: number | null\n /** Mean of per-case F1 over issue-bearing runs. */\n macroF1: number | null\n criticalStepAccuracy: number | null\n citationCoverage: number | null\n citationExcerptCoverage: number | null\n citationLabelAgreement: number | null\n citationResolution: number | null\n citationResolutionUnknownRuns: number\n unresolvedCitations: number\n citationResolutionErrors: number\n trustedNegativeFalsePositiveRate: number | null\n trustedNegativeFailureRate: number | null\n unlabeledPredictionRate: number | null\n unlabeledFailureRate: number | null\n /** Primary repeatability measure over complete finding identity and evidence. */\n predictionAgreement: number | null\n /** Repeated cases contributing equally to predictionAgreement. */\n predictionAgreementCases: number\n /** Secondary repeatability detail over matched expected labels. */\n matchedLabelAgreement: number | null\n /** Positive repeated cases contributing equally to matchedLabelAgreement. */\n matchedLabelAgreementCases: number\n latencyMs: AnalystLatencyDistribution | null\n benchmarkClockLatencyRuns: number\n runnerReportedLatencyRuns: number\n latencyUnknownRuns: number\n calls: number\n callsUnknownRuns: number\n inputTokens: number\n outputTokens: number\n reasoningTokens: number\n cachedTokens: number\n cacheWriteTokens: number\n tokenUsageUnknownRuns: number\n reasoningTokenUsageUnknownRuns: number\n cachedTokenUsageUnknownRuns: number\n cacheWriteTokenUsageUnknownRuns: number\n knownCostUsd: number\n costUnknownRuns: number\n}\n\nexport interface AnalystBenchmarkDatasetRef {\n id: string\n revision: string\n split?: string\n}\n\nexport interface AnalystBenchmarkDescriptor {\n id?: string\n dataset?: AnalystBenchmarkDatasetRef\n command?: string\n environment?: Record<string, string>\n metadata?: Record<string, unknown>\n}\n\nexport interface AnalystBenchmarkProvenance extends AnalystBenchmarkDescriptor {\n startedAt: string\n endedAt: string\n caseCount: number\n runnerIds: string[]\n repetitions: number\n maxConcurrency: number\n runnerOrderSeed: number\n}\n\nexport interface AnalystBenchmarkResult {\n provenance: AnalystBenchmarkProvenance\n observations: AnalystBenchmarkObservation[]\n summaries: AnalystBenchmarkSummary[]\n}\n\nexport interface RunAnalystBenchmarkOptions<TInput> {\n cases: readonly AnalystBenchmarkCase<TInput>[]\n runners: readonly AnalystBenchmarkRunner<TInput>[]\n repetitions?: number\n maxConcurrency?: number\n runnerOrderSeed?: number\n resolveEvidence?: AnalystEvidenceResolver<TInput>\n benchmark?: AnalystBenchmarkDescriptor\n /** Previously persisted rows. Exact case, runner, repetition, and execution identities are required. */\n initialObservations?: readonly AnalystBenchmarkObservation[]\n onObservation?: (observation: AnalystBenchmarkObservation) => void | Promise<void>\n signal?: AbortSignal\n}\n\nexport async function runAnalystBenchmark<TInput>(\n options: RunAnalystBenchmarkOptions<TInput>,\n): Promise<AnalystBenchmarkResult> {\n validateBenchmarkOptions(options)\n const startedAt = new Date().toISOString()\n const repetitions = options.repetitions ?? 1\n const runnerOrderSeed = options.runnerOrderSeed ?? 0\n const allJobs = benchmarkJobs(options.cases, options.runners, repetitions, runnerOrderSeed)\n const maxConcurrency = Math.min(options.maxConcurrency ?? 1, allJobs.length)\n const initialObservations = validateInitialObservations(\n options.initialObservations ?? [],\n allJobs,\n )\n const completed = new Set(initialObservations.map(observationKey))\n const jobs = allJobs.filter((job) => !completed.has(jobKey(job)))\n const observations: AnalystBenchmarkObservation[] = [...initialObservations]\n let cursor = 0\n const worker = async (): Promise<void> => {\n while (cursor < jobs.length) {\n options.signal?.throwIfAborted()\n const job = jobs[cursor++]!\n const observation = await runBenchmarkJob(job, options.signal, options.resolveEvidence)\n options.signal?.throwIfAborted()\n await options.onObservation?.(observation)\n options.signal?.throwIfAborted()\n observations.push(observation)\n }\n }\n await Promise.all(Array.from({ length: Math.min(maxConcurrency, jobs.length) }, worker))\n options.signal?.throwIfAborted()\n const runnerOrder = new Map(options.runners.map((runner, index) => [runner.id, index]))\n const caseOrder = new Map(options.cases.map((testCase, index) => [testCase.id, index]))\n observations.sort(\n (a, b) =>\n (runnerOrder.get(a.runnerId) ?? 0) - (runnerOrder.get(b.runnerId) ?? 0) ||\n (caseOrder.get(a.caseId) ?? 0) - (caseOrder.get(b.caseId) ?? 0) ||\n a.repetition - b.repetition,\n )\n return {\n provenance: {\n ...options.benchmark,\n startedAt,\n endedAt: new Date().toISOString(),\n caseCount: options.cases.length,\n runnerIds: options.runners.map((runner) => runner.id),\n repetitions,\n maxConcurrency,\n runnerOrderSeed,\n },\n observations,\n summaries: options.runners.map((runner) =>\n summarizeAnalystBenchmarkRunner(\n runner.id,\n observations.filter((observation) => observation.runnerId === runner.id),\n ),\n ),\n }\n}\n\nexport function registryBenchmarkRunner(options: {\n id: string\n registry: AnalystRegistry\n runOptions?: Omit<RegistryRunOpts, 'signal'>\n /** Count any selected analyst failure as a failed benchmark run. */\n failOnAnalystFailure?: boolean\n}): AnalystBenchmarkRunner<AnalystRunInputs> {\n return {\n id: options.id,\n async analyze(input, context) {\n const result = await options.registry.run(\n `${options.id}:${context.caseId}:${context.repetition}`,\n input,\n { ...options.runOptions, signal: context.signal },\n )\n return {\n findings: result.findings,\n usage: mergeRegistryUsage(result),\n metadata: { analystRun: result },\n ...(options.failOnAnalystFailure ? { error: registryRunFailure(result) } : {}),\n }\n },\n }\n}\n\nasync function runBenchmarkJob<TInput>(\n job: {\n runner: AnalystBenchmarkRunner<TInput>\n testCase: AnalystBenchmarkCase<TInput>\n repetition: number\n executionIndex: number\n },\n signal?: AbortSignal,\n resolveEvidence?: AnalystEvidenceResolver<TInput>,\n): Promise<AnalystBenchmarkObservation> {\n const started = performance.now()\n try {\n const output = await job.runner.analyze(job.testCase.input, {\n caseId: job.testCase.id,\n repetition: job.repetition,\n signal,\n })\n if (output.usage) {\n assertValidAnalystUsageReceipt(output.usage, 'analyst benchmark usage')\n }\n const benchmarkLatencyMs = performance.now() - started\n const latency = resolveBenchmarkLatency(output.observedLatencyMs, benchmarkLatencyMs)\n const scoredFindings = output.error ? [] : output.findings\n return {\n runnerId: job.runner.id,\n caseId: job.testCase.id,\n clusterId: job.testCase.clusterId,\n labelState: job.testCase.labelState,\n repetition: job.repetition,\n executionIndex: job.executionIndex,\n latencyMs: latency.value,\n latencySource: latency.source,\n findings: output.findings,\n score: scoreAnalystFindings(job.testCase, scoredFindings),\n evidenceResolution: resolveEvidence\n ? await resolveFindingEvidence(job.testCase, output.findings, resolveEvidence, signal)\n : undefined,\n caseTags: [...(job.testCase.tags ?? [])],\n caseMetadata: job.testCase.metadata,\n usage: output.usage,\n runnerMetadata: output.metadata,\n ...(output.error ? { error: output.error } : {}),\n }\n } catch (error) {\n if (signal?.aborted) throw error\n const findings: AnalystFinding[] = []\n return {\n runnerId: job.runner.id,\n caseId: job.testCase.id,\n clusterId: job.testCase.clusterId,\n labelState: job.testCase.labelState,\n repetition: job.repetition,\n executionIndex: job.executionIndex,\n latencyMs: performance.now() - started,\n latencySource: 'benchmark-clock',\n findings,\n score: scoreAnalystFindings(job.testCase, findings),\n caseTags: [...(job.testCase.tags ?? [])],\n caseMetadata: job.testCase.metadata,\n error: {\n class: error instanceof Error ? error.constructor.name : 'Error',\n message: error instanceof Error ? error.message : String(error),\n },\n }\n }\n}\n\nfunction resolveBenchmarkLatency(\n observedLatencyMs: number | null | undefined,\n fallbackMs: number,\n): {\n value: number | null\n source: AnalystBenchmarkObservation['latencySource']\n} {\n if (observedLatencyMs === undefined) {\n return { value: fallbackMs, source: 'benchmark-clock' }\n }\n if (observedLatencyMs === null) return { value: null, source: 'uncaptured' }\n if (!Number.isFinite(observedLatencyMs) || observedLatencyMs < 0) {\n throw new RangeError('analyst benchmark observedLatencyMs must be finite and non-negative')\n }\n return { value: observedLatencyMs, source: 'runner-reported' }\n}\n\nfunction registryRunFailure(result: AnalystRunResult): AnalystBenchmarkOutput['error'] | undefined {\n const failed = result.per_analyst.filter((summary) => summary.status === 'failed')\n if (failed.length === 0) return undefined\n return {\n class: 'AnalystRunFailure',\n message: failed\n .map(\n (summary) =>\n `${summary.analyst_id}: ${summary.error?.class ?? 'Error'}: ${summary.error?.message ?? 'analyst failed'}`,\n )\n .join('; '),\n }\n}\n\nfunction benchmarkJobs<TInput>(\n cases: readonly AnalystBenchmarkCase<TInput>[],\n runners: readonly AnalystBenchmarkRunner<TInput>[],\n repetitions: number,\n runnerOrderSeed: number,\n): Array<{\n runner: AnalystBenchmarkRunner<TInput>\n testCase: AnalystBenchmarkCase<TInput>\n repetition: number\n executionIndex: number\n}> {\n const seededRunners = [...runners].sort(\n (left, right) =>\n stableHash(`${runnerOrderSeed}\\u0000${left.id}`) -\n stableHash(`${runnerOrderSeed}\\u0000${right.id}`) || left.id.localeCompare(right.id),\n )\n let executionIndex = 0\n return cases.flatMap((testCase, caseIndex) =>\n Array.from({ length: repetitions }, (_, repetition) => {\n const blockIndex = caseIndex * repetitions + repetition\n const rotation = blockIndex % seededRunners.length\n const ordered = [...seededRunners.slice(rotation), ...seededRunners.slice(0, rotation)]\n return ordered.map((runner) => ({\n runner,\n testCase,\n repetition,\n executionIndex: executionIndex++,\n }))\n }).flat(),\n )\n}\n\nfunction validateInitialObservations<TInput>(\n observations: readonly AnalystBenchmarkObservation[],\n jobs: readonly {\n runner: AnalystBenchmarkRunner<TInput>\n testCase: AnalystBenchmarkCase<TInput>\n repetition: number\n executionIndex: number\n }[],\n): AnalystBenchmarkObservation[] {\n const expected = new Map(jobs.map((job) => [jobKey(job), job]))\n const seen = new Set<string>()\n return observations.map((observation) => {\n const key = observationKey(observation)\n if (seen.has(key)) {\n throw new TypeError(\n `duplicate initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`,\n )\n }\n seen.add(key)\n const job = expected.get(key)\n if (!job) {\n throw new TypeError(\n `initial analyst benchmark observation does not match a planned job: '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`,\n )\n }\n if (observation.executionIndex !== job.executionIndex) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has executionIndex ${observation.executionIndex}; expected ${job.executionIndex}`,\n )\n }\n if (\n observation.clusterId !== job.testCase.clusterId ||\n observation.labelState !== job.testCase.labelState\n ) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' does not match the current case labels`,\n )\n }\n if (\n JSON.stringify(observation.caseTags) !== JSON.stringify(job.testCase.tags ?? []) ||\n JSON.stringify(observation.caseMetadata) !== JSON.stringify(job.testCase.metadata)\n ) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' does not match the current case metadata`,\n )\n }\n const expectedScore = scoreAnalystFindings(\n job.testCase,\n observation.error ? [] : observation.findings,\n )\n if (JSON.stringify(observation.score) !== JSON.stringify(expectedScore)) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has stale or invalid scores`,\n )\n }\n if (observation.usage) {\n assertValidAnalystUsageReceipt(\n observation.usage,\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' usage`,\n )\n }\n if (\n !['benchmark-clock', 'runner-reported', 'uncaptured'].includes(observation.latencySource) ||\n (observation.latencySource === 'uncaptured' && observation.latencyMs !== null) ||\n (observation.latencySource !== 'uncaptured' &&\n (observation.latencyMs === null ||\n !Number.isFinite(observation.latencyMs) ||\n observation.latencyMs < 0))\n ) {\n throw new TypeError(\n `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has invalid latency`,\n )\n }\n return { ...observation }\n })\n}\n\nfunction jobKey(job: {\n runner: { id: string }\n testCase: { id: string }\n repetition: number\n}): string {\n return `${job.runner.id}\\u0000${job.testCase.id}\\u0000${job.repetition}`\n}\n\nfunction observationKey(observation: {\n runnerId: string\n caseId: string\n repetition: number\n}): string {\n return `${observation.runnerId}\\u0000${observation.caseId}\\u0000${observation.repetition}`\n}\n\nasync function resolveFindingEvidence<TInput>(\n testCase: AnalystBenchmarkCase<TInput>,\n findings: readonly AnalystFinding[],\n resolver: AnalystEvidenceResolver<TInput>,\n signal?: AbortSignal,\n): Promise<AnalystEvidenceResolution> {\n const evidence = findings.flatMap((finding) => finding.evidence_refs)\n const resolved: EvidenceRef[] = []\n const unresolvedEvidence: EvidenceRef[] = []\n const errors: AnalystEvidenceResolutionError[] = []\n for (const ref of evidence) {\n signal?.throwIfAborted()\n try {\n if (\n await resolver({\n caseId: testCase.id,\n caseInput: testCase.input,\n evidence: ref,\n signal,\n })\n ) {\n resolved.push(ref)\n } else {\n unresolvedEvidence.push(ref)\n }\n } catch (error) {\n if (signal?.aborted) throw error\n errors.push({\n evidence: ref,\n class: error instanceof Error ? error.constructor.name : 'Error',\n message: error instanceof Error ? error.message : String(error),\n })\n }\n }\n return {\n checked: evidence.length,\n resolved: resolved.length,\n unresolvedEvidence,\n errors,\n validity: evidence.length === 0 || errors.length > 0 ? null : resolved.length / evidence.length,\n }\n}\n\nfunction parseTraceSpanUri(uri: string): { traceId: string; spanId: string } | null {\n const match = /^trace:\\/\\/([^/]+)\\/span\\/([^/]+)$/.exec(uri)\n if (!match) return null\n try {\n const traceId = decodeURIComponent(match[1]!)\n const spanId = decodeURIComponent(match[2]!)\n return traceId && spanId ? { traceId, spanId } : null\n } catch {\n return null\n }\n}\n\nfunction validateBenchmarkOptions<TInput>(options: RunAnalystBenchmarkOptions<TInput>): void {\n if (options.cases.length === 0) throw new TypeError('runAnalystBenchmark requires cases')\n if (options.runners.length === 0) throw new TypeError('runAnalystBenchmark requires runners')\n const repetitions = options.repetitions ?? 1\n const maxConcurrency = options.maxConcurrency ?? 1\n if (!Number.isSafeInteger(repetitions) || repetitions < 1) {\n throw new RangeError('runAnalystBenchmark repetitions must be a positive safe integer')\n }\n if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) {\n throw new RangeError('runAnalystBenchmark maxConcurrency must be a positive safe integer')\n }\n if (!Number.isSafeInteger(options.runnerOrderSeed ?? 0)) {\n throw new RangeError('runAnalystBenchmark runnerOrderSeed must be a safe integer')\n }\n assertUniqueNonEmpty(\n options.cases.map((testCase) => testCase.id),\n 'case',\n )\n assertUniqueNonEmpty(\n options.runners.map((runner) => runner.id),\n 'runner',\n )\n for (const testCase of options.cases) validateBenchmarkCase(testCase)\n}\n\nfunction validateBenchmarkCase(testCase: AnalystBenchmarkCase): void {\n assertValidAnalystScoringCase(testCase)\n if (!testCase.clusterId.trim()) {\n throw new TypeError(`${testCase.id}: analyst benchmark clusterId must not be empty`)\n }\n if (\n testCase.labelState !== 'positive' &&\n testCase.labelState !== 'trusted-negative' &&\n testCase.labelState !== 'unlabeled'\n ) {\n throw new TypeError(`${testCase.id}: analyst benchmark labelState is invalid`)\n }\n if (testCase.labelState === 'positive' && testCase.expectedIssues.length === 0) {\n throw new TypeError(`${testCase.id}: positive case requires at least one expected issue`)\n }\n if (testCase.labelState !== 'positive' && testCase.expectedIssues.length > 0) {\n throw new TypeError(\n `${testCase.id}: ${testCase.labelState} case cannot contain expected issues`,\n )\n }\n}\n\nfunction assertUniqueNonEmpty(values: readonly string[], label: string): void {\n const seen = new Set<string>()\n for (const value of values) {\n if (!value.trim()) throw new TypeError(`analyst benchmark ${label} id must not be empty`)\n if (seen.has(value)) throw new TypeError(`duplicate analyst benchmark ${label} id '${value}'`)\n seen.add(value)\n }\n}\n\nfunction stableHash(value: string): number {\n let hash = 2166136261\n for (let index = 0; index < value.length; index += 1) {\n hash ^= value.charCodeAt(index)\n hash = Math.imul(hash, 16777619)\n }\n return hash >>> 0\n}\n\nfunction mergeRegistryUsage(result: AnalystRunResult): AnalystUsageReceipt {\n const usages = result.per_analyst.map((summary) => summary.usage)\n const calls = usages.every((usage) => usage.calls !== null)\n ? usages.reduce((sum, usage) => sum + (usage.calls ?? 0), 0)\n : null\n const tokens = usages.every((usage) => usage.tokens !== null)\n ? usages.reduce(\n (sum, usage) => ({\n input: sum.input + (usage.tokens?.input ?? 0),\n output: sum.output + (usage.tokens?.output ?? 0),\n reasoning: sum.reasoning + (usage.tokens?.reasoning ?? 0),\n cached: sum.cached + (usage.tokens?.cached ?? 0),\n cacheWrite: sum.cacheWrite + (usage.tokens?.cacheWrite ?? 0),\n }),\n { input: 0, output: 0, reasoning: 0, cached: 0, cacheWrite: 0 },\n )\n : null\n const knownCostUsd = usages.reduce(\n (sum, usage) =>\n sum + (usage.cost.kind === 'uncaptured' ? (usage.knownCostUsd ?? 0) : usage.cost.usd),\n 0,\n )\n const cost = usages.some((usage) => usage.cost.kind === 'uncaptured')\n ? ({ kind: 'uncaptured', usd: null } as const)\n : usages.some((usage) => usage.cost.kind === 'estimated')\n ? ({ kind: 'estimated', usd: knownCostUsd } as const)\n : ({ kind: 'observed', usd: knownCostUsd } as const)\n return {\n calls,\n tokens,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd } : {}),\n }\n}\n"],"mappings":";;;;AASA,SAAgB,qBACd,UACA,UACqB;CACrB,8BAA8B,QAAQ;CACtC,MAAM,wBAAwB,sBAAsB,SAAS,gBAAgB,QAAQ;CACrF,MAAM,kBAAkB,SAAS,eAC9B,QAAQ,GAAG,UAAU,sBAAsB,IAAI,KAAK,CAAC,CAAC,CACtD,KAAK,UAAU,MAAM,EAAE;CAC1B,MAAM,iBAAiB,SAAS,eAC7B,QAAQ,GAAG,UAAU,CAAC,sBAAsB,IAAI,KAAK,CAAC,CAAC,CACvD,KAAK,UAAU,MAAM,EAAE;CAC1B,MAAM,0BAA0B,IAAI,IAAI,sBAAsB,OAAO,CAAC;CAEtE,MAAM,4BAA4B,SAC/B,KAAK,GAAG,UAAU,KAAK,CAAC,CACxB,QAAQ,UAAU,CAAC,wBAAwB,IAAI,KAAK,CAAC;CACxD,MAAM,qBAAqB,SAAS,eAAe;CACnD,MAAM,cAAc,uBAAuB,IAAI,IAAI,gBAAgB,SAAS;CAC5E,MAAM,mBACJ,SAAS,WAAW,IAChB,uBAAuB,IACrB,IACA,IACF,wBAAwB,OAAO,SAAS;CAC9C,MAAM,KAAK,kBAAkB,kBAAkB,WAAW;CAC1D,MAAM,cAAc,SAAS,SAAS,YAAY,QAAQ,aAAa;CAEvE,MAAM,iBAAiB,SAAS,eAAe,QAC5C,WAAW,MAAM,kBAAkB,UAAU,KAAK,CACrD;CACA,MAAM,eAAe,SAAS,eAAe,QAAQ,UAAU;EAC7D,KAAK,MAAM,kBAAkB,UAAU,OAAO,GAAG,OAAO;EACxD,OAAO,gBAAgB,aAAa,MAAM,oBAAoB,CAAC,GAAG,KAAK;CACzE,CAAC,CAAC,CAAC;CAEH,MAAM,uBAAuB,SAAS,QAAQ,YAAY,QAAQ,cAAc,SAAS,CAAC,CAAC,CAAC;CAC5F,MAAM,oBAAoB,SAAS,kBAC/B,YAAY,QACT,QAAQ,CAAC,SAAS,gBAAiB,MAAM,aAAa,gBAAgB,KAAK,QAAQ,CAAC,CACvF,IACA,CAAC;CAEL,OAAO;EACL;EACA;EACA;EACA,yBAAyB,CAAC,GAAG,uBAAuB,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EAC1E;EACA;EACA;EACA;EACA;EACA,sBAAsB,eAAe,WAAW,IAAI,OAAO,eAAe,eAAe;EACzF,kBAAkB,SAAS,WAAW,IAAI,OAAO,uBAAuB,SAAS;EACjF,yBACE,YAAY,WAAW,IACnB,OACA,YAAY,QAAQ,aAAa,QAAQ,SAAS,SAAS,KAAK,CAAC,CAAC,CAAC,CAAC,SACpE,YAAY;EAClB,wBACE,SAAS,oBAAoB,KAAA,IACzB,OACA,YAAY,WAAW,IACrB,SAAS,WAAW,IAClB,OACA,KACD,YAAY,SAAS,kBAAkB,UAAU,YAAY;EACtE,4BAA4B,uBAAuB,KAAK,SAAS,SAAS;CAC5E;AACF;AAEA,SAAgB,8BACd,UACM;CACN,IAAI,CAAC,SAAS,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,6CAA6C;CAC1F,MAAM,sBAAM,IAAI,IAAY;CAC5B,KAAK,MAAM,SAAS,SAAS,gBAAgB;EAC3C,IAAI,CAAC,MAAM,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,sCAAsC;EAC/F,IAAI,IAAI,IAAI,MAAM,EAAE,GAClB,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,iCAAiC,MAAM,GAAG,EAAE;EAEjF,IAAI,IAAI,MAAM,EAAE;EAChB,IACE,CAAC,MAAM,YAAY,UACnB,CAAC,MAAM,OAAO,UACd,CAAC,MAAM,UAAU,UACjB,CAAC,MAAM,UAAU,QAEjB,MAAM,IAAI,UACR,GAAG,SAAS,GAAG,GAAG,MAAM,GAAG,2EAC7B;CAEJ;CACA,KAAK,MAAM,OAAO,SAAS,mBAAmB,CAAC,GAC7C,IAAI,CAAC,IAAI,IAAI,KAAK,GAChB,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,yCAAyC;AAGlF;AAEA,SAAgB,kBAAkB,GAAW,GAAmB;CAC9D,OAAO,IAAI,MAAM,IAAI,IAAK,IAAI,IAAI,KAAM,IAAI;AAC9C;AAEA,SAAS,sBACP,QACA,UACqB;CACrB,IAAI,OAAO,WAAW,GAAG,uBAAO,IAAI,IAAI;CACxC,MAAM,oBAAoB,OAAO,SAAS;CAW1C,MAAM,aAAa,oBAVJ,OAAO,KAAK,UAAU,CACnC,GAAG,SAAS,KAAK,YAAY;EAC3B,IAAI,CAAC,oBAAoB,SAAS,KAAK,GAAG,OAAO;EACjD,MAAM,eACH,MAAM,kBAAkB,UAAU,KAAK,KACxC,gBAAgB,QAAQ,eAAe,MAAM,oBAAoB,CAAC,GAAG,KAAK;EAC5E,OAAO,oBAAoB,OAAO,WAAW;CAC/C,CAAC,GACD,GAAG,MAAM,KAAK,EAAE,QAAQ,OAAO,OAAO,SAAS,CAAC,CAClD,CAC4C,GAAG,EAAE,UAAU,KAAK,CAAC,CAAC,CAAC;CACnE,MAAM,0BAAU,IAAI,IAAoB;CACxC,KAAK,MAAM,CAAC,YAAY,WAAW,WAAW,QAAQ,GAAG;EACvD,IAAI,SAAS,KAAK,UAAU,SAAS,QAAQ;EAC7C,IAAI,CAAC,oBAAoB,SAAS,SAAU,OAAO,WAAY,GAAG;EAClE,QAAQ,IAAI,YAAY,MAAM;CAChC;CACA,OAAO;AACT;AAEA,SAAS,oBAAoB,SAAyB,OAAyC;CAC7F,IAAI,MAAM,cAAc,CAAC,MAAM,WAAW,SAAS,QAAQ,UAAU,GAAG,OAAO;CAC/E,IAAI,MAAM,SAAS,CAAC,MAAM,MAAM,SAAS,QAAQ,IAAI,GAAG,OAAO;CAC/D,IAAI,MAAM,aAAa,CAAC,QAAQ,WAAW,CAAC,MAAM,SAAS,SAAS,QAAQ,OAAO,IACjF,OAAO;CAET,IACE,MAAM,YACN,CAAC,gBAAgB,QAAQ,eAAe,MAAM,UAAU,MAAM,gBAAgB,KAAK,GAEnF,OAAO;CAET,OAAO;AACT;AAEA,SAAS,gBACP,QACA,UACA,MACS;CACT,IAAI,SAAS,WAAW,GAAG,OAAO;CAClC,MAAM,SAAS,WACb,OAAO,MAAM,QAAQ,gBAAgB,KAAK,MAAM,CAAC;CACnD,OAAO,SAAS,QAAQ,SAAS,MAAM,KAAK,IAAI,SAAS,KAAK,KAAK;AACrE;AAEA,SAAS,gBAAgB,QAAqB,UAA+C;CAC3F,OACE,OAAO,QAAQ,SAAS,QAAQ,SAAS,SAAS,KAAA,KAAa,OAAO,SAAS,SAAS;AAE5F;;;ACnKA,SAAgB,gCACd,UACA,cACyB;CACzB,MAAM,eAAe,aAAa,QAAQ,gBAAgB,YAAY,eAAe,UAAU;CAC/F,MAAM,YAAY,aAAa,QAAQ,gBAAgB,CAAC,YAAY,KAAK;CACzE,MAAM,iBAAiB,aAAa,QACjC,KAAK,gBAAgB,MAAM,YAAY,MAAM,oBAC9C,CACF;CACA,MAAM,gBAAgB,aAAa,QAChC,KAAK,gBAAgB,MAAM,YAAY,MAAM,gBAAgB,QAC9D,CACF;CACA,MAAM,gBAAgB,aAAa,QAChC,KAAK,gBAAgB,OAAO,YAAY,QAAQ,IAAI,YAAY,SAAS,SAC1E,CACF;CACA,MAAM,oBAAoB,aAAa,QACpC,KAAK,gBAAgB,MAAM,YAAY,MAAM,wBAAwB,QACtE,CACF;CACA,MAAM,cAAc,mBAAmB,IAAI,OAAO,gBAAgB;CAClE,MAAM,mBACJ,mBAAmB,IAAI,OAAO,kBAAkB,IAAI,IAAI,oBAAoB;CAC9E,MAAM,mBACJ,aAAa,WAAW,IACpB,OACA,KAAK,aAAa,KAAK,gBAAgB,YAAY,MAAM,WAAW,CAAC;CAC3E,MAAM,wBACJ,aAAa,WAAW,IACpB,OACA,KAAK,aAAa,KAAK,gBAAgB,YAAY,MAAM,gBAAgB,CAAC;CAChF,MAAM,UACJ,aAAa,WAAW,IAAI,OAAO,KAAK,aAAa,KAAK,gBAAgB,YAAY,MAAM,EAAE,CAAC;CACjG,MAAM,WAAW,aACd,KAAK,gBAAgB,YAAY,MAAM,oBAAoB,CAAC,CAC5D,QAAQ,UAA2B,UAAU,IAAI;CACpD,MAAM,uBAAuB,UAAU,QACpC,KAAK,gBACJ,MAAM,YAAY,SAAS,QAAQ,YAAY,QAAQ,cAAc,SAAS,CAAC,CAAC,CAAC,QACnF,CACF;CACA,MAAM,cAAc,UAAU,QAAQ,KAAK,gBAAgB,MAAM,YAAY,SAAS,QAAQ,CAAC;CAC/F,MAAM,eAAe,UAAU,SAAS,gBACtC,YAAY,SAAS,SAAS,YAAY,QAAQ,aAAa,CACjE;CACA,MAAM,uBAAuB,UAAU,QACpC,gBAAgB,YAAY,MAAM,2BAA2B,IAChE;CACA,MAAM,gBAAgB,qBAAqB,QACxC,KAAK,gBACJ,MACA,YAAY,SAAS,QAAQ,OAAO,YAAY,QAAQ,QAAQ,cAAc,QAAQ,CAAC,GACzF,CACF;CACA,MAAM,uBAAuB,qBAAqB,QAC/C,KAAK,gBAAgB,MAAM,YAAY,MAAM,kBAAkB,QAChE,CACF;CACA,MAAM,kBAAkB,aAAa,QAClC,gBAAgB,YAAY,eAAe,kBAC9C;CACA,MAAM,2BAA2B,gBAAgB,QAAQ,gBAAgB,CAAC,YAAY,KAAK;CAC3F,MAAM,YAAY,aAAa,QAAQ,gBAAgB,YAAY,eAAe,WAAW;CAC7F,MAAM,qBAAqB,UAAU,QAAQ,gBAAgB,CAAC,YAAY,KAAK;CAC/E,MAAM,oBAAoB,UAAU,QACjC,KAAK,gBAAgB,OAAO,YAAY,oBAAoB,YAAY,IACzE,CACF;CACA,MAAM,sBAAsB,UAAU,QACnC,KAAK,gBAAgB,OAAO,YAAY,oBAAoB,mBAAmB,UAAU,IAC1F,CACF;CACA,MAAM,2BAA2B,UAAU,QACxC,KAAK,gBAAgB,OAAO,YAAY,oBAAoB,OAAO,UAAU,IAC9E,CACF;CACA,MAAM,qBAAqB,UAAU,QAAQ,gBAC3C,YAAY,SAAS,MAAM,YAAY,QAAQ,cAAc,SAAS,CAAC,CACzE;CACA,MAAM,gCAAgC,mBAAmB,QACtD,gBACC,CAAC,YAAY,sBAAsB,YAAY,mBAAmB,OAAO,SAAS,CACtF,CAAC,CAAC;CACF,MAAM,SAAS,aAAa,KAAK,gBAAgB,YAAY,KAAK;CAClE,MAAM,eAAe,UACnB,OAAO,KAAK,UAAU;EACpB,IAAI,CAAC,OAAO,OAAO;EACnB,OAAO,MAAM,KAAK,SAAS,eAAgB,MAAM,gBAAgB,IAAK,MAAM,KAAK;CACnF,CAAC,CACH;CACA,MAAM,sBAAsB,6BAA6B,cAAc,mBAAmB;CAC1F,MAAM,wBAAwB,6BAC5B,aAAa,QAAQ,gBAAgB,YAAY,eAAe,UAAU,GAC1E,qBACF;CACA,OAAO;EACL;EACA,aAAa,aAAa;EAC1B,eAAe,aAAa,QAAQ,gBAAgB,CAAC,YAAY,KAAK,CAAC,CAAC;EACxE,YAAY,aAAa,QAAQ,gBAAgB,QAAQ,YAAY,KAAK,CAAC,CAAC,CAAC;EAC7E,kBAAkB,aAAa;EAC/B,qBAAqB,gBAAgB;EACrC,eAAe,UAAU;EACzB;EACA;EACA,IACE,qBAAqB,QAAQ,gBAAgB,OACzC,OACA,kBAAkB,kBAAkB,WAAW;EACrD;EACA;EACA;EACA,sBAAsB,SAAS,WAAW,IAAI,OAAO,KAAK,QAAQ;EAClE,kBAAkB,gBAAgB,IAAI,OAAO,uBAAuB;EACpE,yBACE,aAAa,WAAW,IACpB,OACA,aAAa,QAAQ,aAAa,QAAQ,SAAS,SAAS,KAAK,CAAC,CAAC,CAAC,CAAC,SACrE,aAAa;EACnB,wBACE,qBAAqB,WAAW,IAC5B,OACA,kBAAkB,IAChB,KACC,gBAAgB,wBAAwB;EACjD,oBACE,mBAAmB,WAAW,KAC9B,gCAAgC,KAChC,oBAAoB,wBAAwB,IACxC,OACA,qBAAqB,oBAAoB;EAC/C;EACA;EACA;EACA,kCACE,yBAAyB,WAAW,IAChC,OACA,yBAAyB,QACtB,gBAAgB,YAAY,MAAM,0BACrC,CAAC,CAAC,SAAS,yBAAyB;EAC1C,4BACE,gBAAgB,WAAW,IACvB,OACA,gBAAgB,QAAQ,gBAAgB,QAAQ,YAAY,KAAK,CAAC,CAAC,CAAC,SACpE,gBAAgB;EACtB,yBACE,mBAAmB,WAAW,IAC1B,OACA,mBAAmB,QAAQ,gBAAgB,YAAY,SAAS,SAAS,CAAC,CAAC,CAAC,SAC5E,mBAAmB;EACzB,sBACE,UAAU,WAAW,IACjB,OACA,UAAU,QAAQ,gBAAgB,QAAQ,YAAY,KAAK,CAAC,CAAC,CAAC,SAAS,UAAU;EACvF,qBAAqB,oBAAoB;EACzC,0BAA0B,oBAAoB;EAC9C,uBAAuB,sBAAsB;EAC7C,4BAA4B,sBAAsB;EAClD,WAAW,oBACT,aACG,KAAK,gBAAgB,YAAY,SAAS,CAAC,CAC3C,QAAQ,UAA2B,UAAU,IAAI,CACtD;EACA,2BAA2B,aAAa,QACrC,gBAAgB,YAAY,kBAAkB,iBACjD,CAAC,CAAC;EACF,2BAA2B,aAAa,QACrC,gBAAgB,YAAY,kBAAkB,iBACjD,CAAC,CAAC;EACF,oBAAoB,aAAa,QAC9B,gBAAgB,YAAY,kBAAkB,YACjD,CAAC,CAAC;EACF,OAAO,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,SAAS,IAAI,CAAC;EACjE,kBAAkB,OAAO,QAAQ,UAAU,CAAC,SAAS,MAAM,UAAU,IAAI,CAAC,CAAC;EAC3E,aAAa,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,SAAS,IAAI,CAAC;EAC/E,cAAc,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,UAAU,IAAI,CAAC;EACjF,iBAAiB,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,aAAa,IAAI,CAAC;EACvF,cAAc,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,UAAU,IAAI,CAAC;EACjF,kBAAkB,OAAO,QAAQ,KAAK,UAAU,OAAO,OAAO,QAAQ,cAAc,IAAI,CAAC;EACzF,uBAAuB,OAAO,QAAQ,UAAU,CAAC,OAAO,MAAM,CAAC,CAAC;EAChE,gCAAgC,OAAO,QAAQ,UAAU,OAAO,QAAQ,cAAc,KAAA,CAAS,CAAC,CAC7F;EACH,6BAA6B,OAAO,QAAQ,UAAU,OAAO,QAAQ,WAAW,KAAA,CAAS,CAAC,CACvF;EACH,iCAAiC,OAAO,QACrC,UAAU,OAAO,QAAQ,eAAe,KAAA,CAC3C,CAAC,CAAC;EACF;EACA,iBAAiB,OAAO,QAAQ,UAAU,CAAC,SAAS,MAAM,KAAK,SAAS,YAAY,CAAC,CAAC;CACxF;AACF;AAEA,SAAS,KAAK,QAAmC;CAC/C,OAAO,OAAO,WAAW,IAAI,IAAI,UAAU,MAAM,IAAI,OAAO;AAC9D;AAEA,SAAS,UAAU,QAAmC;CACpD,MAAM,UAAU,CAAC,GAAG,MAAM,CAAC,CAAC,MACzB,MAAM,UAAU,KAAK,IAAI,IAAI,IAAI,KAAK,IAAI,KAAK,KAAK,OAAO,KAC9D;CACA,IAAI,MAAM;CACV,IAAI,aAAa;CACjB,KAAK,MAAM,SAAS,SAAS;EAC3B,MAAM,OAAO,MAAM;EACnB,cAAc,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,KAAK,IAAI,MAAM,OAAO,QAAQ,QAAQ,OAAO;EACrF,MAAM;CACR;CACA,OAAO,MAAM;AACf;AAEA,SAAS,oBAAoB,QAA8D;CACzF,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,OAAO;EACL,KAAK,OAAO;EACZ,MAAM,KAAK,MAAM;EACjB,KAAK,WAAW,QAAQ,EAAG;EAC3B,KAAK,WAAW,QAAQ,GAAI;EAC5B,KAAK,OAAO,GAAG,EAAE;CACnB;AACF;AAEA,SAAS,WAAW,QAA2B,UAA0B;CACvE,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,OAAO,OAAO,KAAK,KAAK,WAAW,OAAO,MAAM,IAAI,MAAM,OAAO,GAAG,EAAE,KAAK;AAC7E;AAEA,SAAS,6BACP,cACA,WACyC;CACzC,MAAM,yBAAS,IAAI,IAA2C;CAC9D,KAAK,MAAM,eAAe,cAAc;EACtC,MAAM,OAAO,OAAO,IAAI,YAAY,MAAM,KAAK,CAAC;EAChD,KAAK,KAAK,WAAW;EACrB,OAAO,IAAI,YAAY,QAAQ,IAAI;CACrC;CACA,MAAM,iBAA2B,CAAC;CAClC,KAAK,MAAM,QAAQ,OAAO,OAAO,GAAG;EAClC,MAAM,aAAuB,CAAC;EAC9B,KAAK,IAAI,OAAO,GAAG,OAAO,KAAK,QAAQ,QACrC,KAAK,IAAI,QAAQ,OAAO,GAAG,QAAQ,KAAK,QAAQ,SAC9C,WAAW,KAAK,QAAQ,UAAU,KAAK,KAAM,GAAG,UAAU,KAAK,MAAO,CAAC,CAAC;EAG5E,IAAI,WAAW,SAAS,GAAG,eAAe,KAAK,KAAK,UAAU,CAAC;CACjE;CACA,OAAO;EACL,OAAO,eAAe,WAAW,IAAI,OAAO,KAAK,cAAc;EAC/D,OAAO,eAAe;CACxB;AACF;AAEA,SAAS,sBAAsB,aAA6D;CAC1F,IAAI,YAAY,OAAO,OAAO,CAAC,SAAS,YAAY,MAAM,OAAO;CACjE,OAAO,YAAY,MAAM;AAC3B;AAEA,SAAS,oBAAoB,aAA6D;CACxF,IAAI,YAAY,OAAO,OAAO,CAAC,SAAS,YAAY,MAAM,OAAO;CACjE,OAAO,YAAY,SAChB,KAAK,YACJ,KAAK,UAAU,CACb,QAAQ,YACR,QAAQ,cACL,KAAK,aAAa;EAAC,SAAS;EAAM,SAAS;EAAK,SAAS,WAAW;CAAI,CAAC,CAAC,CAC1E,MAAM,MAAM,UAAU,KAAK,UAAU,IAAI,CAAC,CAAC,cAAc,KAAK,UAAU,KAAK,CAAC,CAAC,CACpF,CAAC,CACH,CAAC,CACA,KAAK;AACV;AAEA,SAAS,QAAQ,MAAyB,OAAkC;CAC1E,MAAM,IAAI,IAAI,IAAI,IAAI;CACtB,MAAM,IAAI,IAAI,IAAI,KAAK;CACvB,MAAM,wBAAQ,IAAI,IAAI,CAAC,GAAG,GAAG,GAAG,CAAC,CAAC;CAClC,IAAI,MAAM,SAAS,GAAG,OAAO;CAC7B,IAAI,eAAe;CACnB,KAAK,MAAM,SAAS,GAAG,IAAI,EAAE,IAAI,KAAK,GAAG,gBAAgB;CACzD,OAAO,eAAe,MAAM;AAC9B;;;;;;;ACnMA,SAAgB,2BACd,UACiC;CACjC,OAAO,OAAO,EAAE,WAAW,UAAU,aAAa;EAChD,IAAI,SAAS,SAAS,QAAQ,OAAO;EACrC,MAAM,WAAW,kBAAkB,SAAS,GAAG;EAC/C,IAAI,CAAC,UAAU,OAAO;EACtB,MAAM,SAAS,MAAM,SAAS,SAAS,CAAC,CAAC,UACvC;GACE,UAAU,SAAS;GACnB,UAAU,CAAC,SAAS,MAAM;EAC5B,GACA,SAAS,EAAE,OAAO,IAAI,KAAA,CACxB;EACA,OACE,OAAO,aAAa,SAAS,WAC7B,OAAO,iBAAiB,WAAW,KACnC,OAAO,MAAM,MAAM,SAAS,KAAK,YAAY,SAAS,MAAM;CAEhE;AACF;AAgKA,eAAsB,oBACpB,SACiC;CACjC,yBAAyB,OAAO;CAChC,MAAM,6BAAY,IAAI,KAAK,EAAA,CAAE,YAAY;CACzC,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,MAAM,UAAU,cAAc,QAAQ,OAAO,QAAQ,SAAS,aAAa,eAAe;CAC1F,MAAM,iBAAiB,KAAK,IAAI,QAAQ,kBAAkB,GAAG,QAAQ,MAAM;CAC3E,MAAM,sBAAsB,4BAC1B,QAAQ,uBAAuB,CAAC,GAChC,OACF;CACA,MAAM,YAAY,IAAI,IAAI,oBAAoB,IAAI,cAAc,CAAC;CACjE,MAAM,OAAO,QAAQ,QAAQ,QAAQ,CAAC,UAAU,IAAI,OAAO,GAAG,CAAC,CAAC;CAChE,MAAM,eAA8C,CAAC,GAAG,mBAAmB;CAC3E,IAAI,SAAS;CACb,MAAM,SAAS,YAA2B;EACxC,OAAO,SAAS,KAAK,QAAQ;GAC3B,QAAQ,QAAQ,eAAe;GAC/B,MAAM,MAAM,KAAK;GACjB,MAAM,cAAc,MAAM,gBAAgB,KAAK,QAAQ,QAAQ,QAAQ,eAAe;GACtF,QAAQ,QAAQ,eAAe;GAC/B,MAAM,QAAQ,gBAAgB,WAAW;GACzC,QAAQ,QAAQ,eAAe;GAC/B,aAAa,KAAK,WAAW;EAC/B;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,gBAAgB,KAAK,MAAM,EAAE,GAAG,MAAM,CAAC;CACvF,QAAQ,QAAQ,eAAe;CAC/B,MAAM,cAAc,IAAI,IAAI,QAAQ,QAAQ,KAAK,QAAQ,UAAU,CAAC,OAAO,IAAI,KAAK,CAAC,CAAC;CACtF,MAAM,YAAY,IAAI,IAAI,QAAQ,MAAM,KAAK,UAAU,UAAU,CAAC,SAAS,IAAI,KAAK,CAAC,CAAC;CACtF,aAAa,MACV,GAAG,OACD,YAAY,IAAI,EAAE,QAAQ,KAAK,MAAM,YAAY,IAAI,EAAE,QAAQ,KAAK,OACpE,UAAU,IAAI,EAAE,MAAM,KAAK,MAAM,UAAU,IAAI,EAAE,MAAM,KAAK,MAC7D,EAAE,aAAa,EAAE,UACrB;CACA,OAAO;EACL,YAAY;GACV,GAAG,QAAQ;GACX;GACA,0BAAS,IAAI,KAAK,EAAA,CAAE,YAAY;GAChC,WAAW,QAAQ,MAAM;GACzB,WAAW,QAAQ,QAAQ,KAAK,WAAW,OAAO,EAAE;GACpD;GACA;GACA;EACF;EACA;EACA,WAAW,QAAQ,QAAQ,KAAK,WAC9B,gCACE,OAAO,IACP,aAAa,QAAQ,gBAAgB,YAAY,aAAa,OAAO,EAAE,CACzE,CACF;CACF;AACF;AAEA,SAAgB,wBAAwB,SAMK;CAC3C,OAAO;EACL,IAAI,QAAQ;EACZ,MAAM,QAAQ,OAAO,SAAS;GAC5B,MAAM,SAAS,MAAM,QAAQ,SAAS,IACpC,GAAG,QAAQ,GAAG,GAAG,QAAQ,OAAO,GAAG,QAAQ,cAC3C,OACA;IAAE,GAAG,QAAQ;IAAY,QAAQ,QAAQ;GAAO,CAClD;GACA,OAAO;IACL,UAAU,OAAO;IACjB,OAAO,mBAAmB,MAAM;IAChC,UAAU,EAAE,YAAY,OAAO;IAC/B,GAAI,QAAQ,uBAAuB,EAAE,OAAO,mBAAmB,MAAM,EAAE,IAAI,CAAC;GAC9E;EACF;CACF;AACF;AAEA,eAAe,gBACb,KAMA,QACA,iBACsC;CACtC,MAAM,UAAU,YAAY,IAAI;CAChC,IAAI;EACF,MAAM,SAAS,MAAM,IAAI,OAAO,QAAQ,IAAI,SAAS,OAAO;GAC1D,QAAQ,IAAI,SAAS;GACrB,YAAY,IAAI;GAChB;EACF,CAAC;EACD,IAAI,OAAO,OACT,+BAA+B,OAAO,OAAO,yBAAyB;EAExE,MAAM,qBAAqB,YAAY,IAAI,IAAI;EAC/C,MAAM,UAAU,wBAAwB,OAAO,mBAAmB,kBAAkB;EACpF,MAAM,iBAAiB,OAAO,QAAQ,CAAC,IAAI,OAAO;EAClD,OAAO;GACL,UAAU,IAAI,OAAO;GACrB,QAAQ,IAAI,SAAS;GACrB,WAAW,IAAI,SAAS;GACxB,YAAY,IAAI,SAAS;GACzB,YAAY,IAAI;GAChB,gBAAgB,IAAI;GACpB,WAAW,QAAQ;GACnB,eAAe,QAAQ;GACvB,UAAU,OAAO;GACjB,OAAO,qBAAqB,IAAI,UAAU,cAAc;GACxD,oBAAoB,kBAChB,MAAM,uBAAuB,IAAI,UAAU,OAAO,UAAU,iBAAiB,MAAM,IACnF,KAAA;GACJ,UAAU,CAAC,GAAI,IAAI,SAAS,QAAQ,CAAC,CAAE;GACvC,cAAc,IAAI,SAAS;GAC3B,OAAO,OAAO;GACd,gBAAgB,OAAO;GACvB,GAAI,OAAO,QAAQ,EAAE,OAAO,OAAO,MAAM,IAAI,CAAC;EAChD;CACF,SAAS,OAAO;EACd,IAAI,QAAQ,SAAS,MAAM;EAC3B,MAAM,WAA6B,CAAC;EACpC,OAAO;GACL,UAAU,IAAI,OAAO;GACrB,QAAQ,IAAI,SAAS;GACrB,WAAW,IAAI,SAAS;GACxB,YAAY,IAAI,SAAS;GACzB,YAAY,IAAI;GAChB,gBAAgB,IAAI;GACpB,WAAW,YAAY,IAAI,IAAI;GAC/B,eAAe;GACf;GACA,OAAO,qBAAqB,IAAI,UAAU,QAAQ;GAClD,UAAU,CAAC,GAAI,IAAI,SAAS,QAAQ,CAAC,CAAE;GACvC,cAAc,IAAI,SAAS;GAC3B,OAAO;IACL,OAAO,iBAAiB,QAAQ,MAAM,YAAY,OAAO;IACzD,SAAS,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;GAChE;EACF;CACF;AACF;AAEA,SAAS,wBACP,mBACA,YAIA;CACA,IAAI,sBAAsB,KAAA,GACxB,OAAO;EAAE,OAAO;EAAY,QAAQ;CAAkB;CAExD,IAAI,sBAAsB,MAAM,OAAO;EAAE,OAAO;EAAM,QAAQ;CAAa;CAC3E,IAAI,CAAC,OAAO,SAAS,iBAAiB,KAAK,oBAAoB,GAC7D,MAAM,IAAI,WAAW,qEAAqE;CAE5F,OAAO;EAAE,OAAO;EAAmB,QAAQ;CAAkB;AAC/D;AAEA,SAAS,mBAAmB,QAAuE;CACjG,MAAM,SAAS,OAAO,YAAY,QAAQ,YAAY,QAAQ,WAAW,QAAQ;CACjF,IAAI,OAAO,WAAW,GAAG,OAAO,KAAA;CAChC,OAAO;EACL,OAAO;EACP,SAAS,OACN,KACE,YACC,GAAG,QAAQ,WAAW,IAAI,QAAQ,OAAO,SAAS,QAAQ,IAAI,QAAQ,OAAO,WAAW,kBAC5F,CAAC,CACA,KAAK,IAAI;CACd;AACF;AAEA,SAAS,cACP,OACA,SACA,aACA,iBAMC;CACD,MAAM,gBAAgB,CAAC,GAAG,OAAO,CAAC,CAAC,MAChC,MAAM,UACL,WAAW,GAAG,gBAAgB,QAAQ,KAAK,IAAI,IAC7C,WAAW,GAAG,gBAAgB,QAAQ,MAAM,IAAI,KAAK,KAAK,GAAG,cAAc,MAAM,EAAE,CACzF;CACA,IAAI,iBAAiB;CACrB,OAAO,MAAM,SAAS,UAAU,cAC9B,MAAM,KAAK,EAAE,QAAQ,YAAY,IAAI,GAAG,eAAe;EAErD,MAAM,YADa,YAAY,cAAc,cACf,cAAc;EAE5C,OAAO,CADU,GAAG,cAAc,MAAM,QAAQ,GAAG,GAAG,cAAc,MAAM,GAAG,QAAQ,CACxE,CAAC,CAAC,KAAK,YAAY;GAC9B;GACA;GACA;GACA,gBAAgB;EAClB,EAAE;CACJ,CAAC,CAAC,CAAC,KAAK,CACV;AACF;AAEA,SAAS,4BACP,cACA,MAM+B;CAC/B,MAAM,WAAW,IAAI,IAAI,KAAK,KAAK,QAAQ,CAAC,OAAO,GAAG,GAAG,GAAG,CAAC,CAAC;CAC9D,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,aAAa,KAAK,gBAAgB;EACvC,MAAM,MAAM,eAAe,WAAW;EACtC,IAAI,KAAK,IAAI,GAAG,GACd,MAAM,IAAI,UACR,oDAAoD,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,EAC3H;EAEF,KAAK,IAAI,GAAG;EACZ,MAAM,MAAM,SAAS,IAAI,GAAG;EAC5B,IAAI,CAAC,KACH,MAAM,IAAI,UACR,wEAAwE,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,EAC/I;EAEF,IAAI,YAAY,mBAAmB,IAAI,gBACrC,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,uBAAuB,YAAY,eAAe,aAAa,IAAI,gBACpL;EAEF,IACE,YAAY,cAAc,IAAI,SAAS,aACvC,YAAY,eAAe,IAAI,SAAS,YAExC,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,yCACjH;EAEF,IACE,KAAK,UAAU,YAAY,QAAQ,MAAM,KAAK,UAAU,IAAI,SAAS,QAAQ,CAAC,CAAC,KAC/E,KAAK,UAAU,YAAY,YAAY,MAAM,KAAK,UAAU,IAAI,SAAS,QAAQ,GAEjF,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,2CACjH;EAEF,MAAM,gBAAgB,qBACpB,IAAI,UACJ,YAAY,QAAQ,CAAC,IAAI,YAAY,QACvC;EACA,IAAI,KAAK,UAAU,YAAY,KAAK,MAAM,KAAK,UAAU,aAAa,GACpE,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,8BACjH;EAEF,IAAI,YAAY,OACd,+BACE,YAAY,OACZ,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,QACjH;EAEF,IACE,CAAC;GAAC;GAAmB;GAAmB;EAAY,CAAC,CAAC,SAAS,YAAY,aAAa,KACvF,YAAY,kBAAkB,gBAAgB,YAAY,cAAc,QACxE,YAAY,kBAAkB,iBAC5B,YAAY,cAAc,QACzB,CAAC,OAAO,SAAS,YAAY,SAAS,KACtC,YAAY,YAAY,IAE5B,MAAM,IAAI,UACR,0CAA0C,YAAY,SAAS,GAAG,YAAY,OAAO,GAAG,YAAY,WAAW,sBACjH;EAEF,OAAO,EAAE,GAAG,YAAY;CAC1B,CAAC;AACH;AAEA,SAAS,OAAO,KAIL;CACT,OAAO,GAAG,IAAI,OAAO,GAAG,QAAQ,IAAI,SAAS,GAAG,QAAQ,IAAI;AAC9D;AAEA,SAAS,eAAe,aAIb;CACT,OAAO,GAAG,YAAY,SAAS,QAAQ,YAAY,OAAO,QAAQ,YAAY;AAChF;AAEA,eAAe,uBACb,UACA,UACA,UACA,QACoC;CACpC,MAAM,WAAW,SAAS,SAAS,YAAY,QAAQ,aAAa;CACpE,MAAM,WAA0B,CAAC;CACjC,MAAM,qBAAoC,CAAC;CAC3C,MAAM,SAA2C,CAAC;CAClD,KAAK,MAAM,OAAO,UAAU;EAC1B,QAAQ,eAAe;EACvB,IAAI;GACF,IACE,MAAM,SAAS;IACb,QAAQ,SAAS;IACjB,WAAW,SAAS;IACpB,UAAU;IACV;GACF,CAAC,GAED,SAAS,KAAK,GAAG;QAEjB,mBAAmB,KAAK,GAAG;EAE/B,SAAS,OAAO;GACd,IAAI,QAAQ,SAAS,MAAM;GAC3B,OAAO,KAAK;IACV,UAAU;IACV,OAAO,iBAAiB,QAAQ,MAAM,YAAY,OAAO;IACzD,SAAS,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK;GAChE,CAAC;EACH;CACF;CACA,OAAO;EACL,SAAS,SAAS;EAClB,UAAU,SAAS;EACnB;EACA;EACA,UAAU,SAAS,WAAW,KAAK,OAAO,SAAS,IAAI,OAAO,SAAS,SAAS,SAAS;CAC3F;AACF;AAEA,SAAS,kBAAkB,KAAyD;CAClF,MAAM,QAAQ,qCAAqC,KAAK,GAAG;CAC3D,IAAI,CAAC,OAAO,OAAO;CACnB,IAAI;EACF,MAAM,UAAU,mBAAmB,MAAM,EAAG;EAC5C,MAAM,SAAS,mBAAmB,MAAM,EAAG;EAC3C,OAAO,WAAW,SAAS;GAAE;GAAS;EAAO,IAAI;CACnD,QAAQ;EACN,OAAO;CACT;AACF;AAEA,SAAS,yBAAiC,SAAmD;CAC3F,IAAI,QAAQ,MAAM,WAAW,GAAG,MAAM,IAAI,UAAU,oCAAoC;CACxF,IAAI,QAAQ,QAAQ,WAAW,GAAG,MAAM,IAAI,UAAU,sCAAsC;CAC5F,MAAM,cAAc,QAAQ,eAAe;CAC3C,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,IAAI,CAAC,OAAO,cAAc,WAAW,KAAK,cAAc,GACtD,MAAM,IAAI,WAAW,iEAAiE;CAExF,IAAI,CAAC,OAAO,cAAc,cAAc,KAAK,iBAAiB,GAC5D,MAAM,IAAI,WAAW,oEAAoE;CAE3F,IAAI,CAAC,OAAO,cAAc,QAAQ,mBAAmB,CAAC,GACpD,MAAM,IAAI,WAAW,4DAA4D;CAEnF,qBACE,QAAQ,MAAM,KAAK,aAAa,SAAS,EAAE,GAC3C,MACF;CACA,qBACE,QAAQ,QAAQ,KAAK,WAAW,OAAO,EAAE,GACzC,QACF;CACA,KAAK,MAAM,YAAY,QAAQ,OAAO,sBAAsB,QAAQ;AACtE;AAEA,SAAS,sBAAsB,UAAsC;CACnE,8BAA8B,QAAQ;CACtC,IAAI,CAAC,SAAS,UAAU,KAAK,GAC3B,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,gDAAgD;CAErF,IACE,SAAS,eAAe,cACxB,SAAS,eAAe,sBACxB,SAAS,eAAe,aAExB,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,0CAA0C;CAE/E,IAAI,SAAS,eAAe,cAAc,SAAS,eAAe,WAAW,GAC3E,MAAM,IAAI,UAAU,GAAG,SAAS,GAAG,qDAAqD;CAE1F,IAAI,SAAS,eAAe,cAAc,SAAS,eAAe,SAAS,GACzE,MAAM,IAAI,UACR,GAAG,SAAS,GAAG,IAAI,SAAS,WAAW,qCACzC;AAEJ;AAEA,SAAS,qBAAqB,QAA2B,OAAqB;CAC5E,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,SAAS,QAAQ;EAC1B,IAAI,CAAC,MAAM,KAAK,GAAG,MAAM,IAAI,UAAU,qBAAqB,MAAM,sBAAsB;EACxF,IAAI,KAAK,IAAI,KAAK,GAAG,MAAM,IAAI,UAAU,+BAA+B,MAAM,OAAO,MAAM,EAAE;EAC7F,KAAK,IAAI,KAAK;CAChB;AACF;AAEA,SAAS,WAAW,OAAuB;CACzC,IAAI,OAAO;CACX,KAAK,IAAI,QAAQ,GAAG,QAAQ,MAAM,QAAQ,SAAS,GAAG;EACpD,QAAQ,MAAM,WAAW,KAAK;EAC9B,OAAO,KAAK,KAAK,MAAM,QAAQ;CACjC;CACA,OAAO,SAAS;AAClB;AAEA,SAAS,mBAAmB,QAA+C;CACzE,MAAM,SAAS,OAAO,YAAY,KAAK,YAAY,QAAQ,KAAK;CAChE,MAAM,QAAQ,OAAO,OAAO,UAAU,MAAM,UAAU,IAAI,IACtD,OAAO,QAAQ,KAAK,UAAU,OAAO,MAAM,SAAS,IAAI,CAAC,IACzD;CACJ,MAAM,SAAS,OAAO,OAAO,UAAU,MAAM,WAAW,IAAI,IACxD,OAAO,QACJ,KAAK,WAAW;EACf,OAAO,IAAI,SAAS,MAAM,QAAQ,SAAS;EAC3C,QAAQ,IAAI,UAAU,MAAM,QAAQ,UAAU;EAC9C,WAAW,IAAI,aAAa,MAAM,QAAQ,aAAa;EACvD,QAAQ,IAAI,UAAU,MAAM,QAAQ,UAAU;EAC9C,YAAY,IAAI,cAAc,MAAM,QAAQ,cAAc;CAC5D,IACA;EAAE,OAAO;EAAG,QAAQ;EAAG,WAAW;EAAG,QAAQ;EAAG,YAAY;CAAE,CAChE,IACA;CACJ,MAAM,eAAe,OAAO,QACzB,KAAK,UACJ,OAAO,MAAM,KAAK,SAAS,eAAgB,MAAM,gBAAgB,IAAK,MAAM,KAAK,MACnF,CACF;CACA,MAAM,OAAO,OAAO,MAAM,UAAU,MAAM,KAAK,SAAS,YAAY,IAC/D;EAAE,MAAM;EAAc,KAAK;CAAK,IACjC,OAAO,MAAM,UAAU,MAAM,KAAK,SAAS,WAAW,IACnD;EAAE,MAAM;EAAa,KAAK;CAAa,IACvC;EAAE,MAAM;EAAY,KAAK;CAAa;CAC7C,OAAO;EACL;EACA;EACA;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,aAAa,IAAI,CAAC;CACvD;AACF"}
@@ -2,15 +2,15 @@ import { c as ValidationError, t as AgentEvalError } from "./errors-D-LKuDhb.js"
2
2
  import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
3
3
  import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-DMFxsLKr.js";
4
4
  import { c as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-DzvMUsS_.js";
5
- import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-BHIgPmzS.js";
6
- import { o as makeFinding, r as usageReceiptFromCostLedger } from "./usage-receipt-EVI8B8Xu.js";
5
+ import { D as RawAnalystFindingSchema, O as evidenceRefsFromRawFinding, T as RAW_FINDING_SCHEMA_PROMPT, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst } from "./kind-factory-B8-r8-y8.js";
6
+ import { r as usageReceiptFromCostLedger, s as makeFinding } from "./usage-receipt-t7vAzCRQ.js";
7
7
  import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
8
- import { D as resolveExternalOptimizerProcessLimits, f as runWithCleanup, n as createRunCostLedger, p as startExternalOptimizerModelProxy, r as fsCampaignStorage, t as acquireSingleRunLock } from "./single-run-lock-BMQEv1wG.js";
9
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
8
+ import { D as resolveExternalOptimizerProcessLimits, f as runWithCleanup, i as fsCampaignStorage, p as startExternalOptimizerModelProxy, r as createRunCostLedger, t as acquireSingleRunLock } from "./single-run-lock-DFWHEB09.js";
9
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-DbTk4JdR.js";
10
10
  import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-DXZIqu17.js";
11
11
  import { E as pairedBootstrap } from "./statistics-ByxzSiOM.js";
12
- import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-CKtTpRhv.js";
13
- import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-B181aMF9.js";
12
+ import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-Dw8PPIlL.js";
13
+ import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-CWeqGl7x.js";
14
14
  import { c as primeProtocolSha256, d as runPrimeExchange, f as assertEqualDeclarativeTerms, n as buildPrimePrompt, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-BfSalTfR.js";
15
15
  import { z } from "zod";
16
16
  import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
@@ -1208,7 +1208,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
1208
1208
  "package.json",
1209
1209
  "pnpm-lock.yaml"
1210
1210
  ]);
1211
- const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "3768273281457e480bf76aadc6001978be31ad7856e27a38ea8b856a21f33607";
1211
+ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "36d8f26723c19516348db0647de861f0ab1e672cb2da5aac8b3e34a1d6b49bdc";
1212
1212
  /** The published benchmark evidence was produced at this package version, by
1213
1213
  * the retired one-shot direct runner, before trace analysts moved to the
1214
1214
  * recursive DSPy RLM engine. Both evidence digests below are historical facts
@@ -6412,4 +6412,4 @@ function shellQuote(value) {
6412
6412
  //#endregion
6413
6413
  export { ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as $, summarizeCodeTraceCalibration as A, publicBenchmarkSystemPrompt as B, analystDefinitionAsymmetries as C, expandCodeTraceFailureBlocks as D, emptyPublicBenchmarkRunner as E, CODE_TRACE_BENCH_ANALYST_PROMPT as F, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as G, appendVerificationArtifactsToOtlp as H, MAX_INCORRECT_BLOCKS as I, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as J, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as K, MAX_INCORRECT_BLOCK_STEPS as L, analystInstructionsOverrideFromText as M, effectiveAnalystProtocolSha256 as N, readAnalystBenchmarkArtifact as O, readAnalystInstructionsOverride as P, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as Q, publicBenchmarkProtocolSha256 as R, AnalystExpressivenessError as S, adaptPublicBenchmarkFindings as T, loadCodeTraceVerificationArtifacts as U, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as V, parseVerificationOutcome as W, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION as X, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as Y, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as Z, runReplVariableAnalystDefinition as _, primeAnalystProtocolSha256 as a, ANALYST_BENCHMARK_OBSERVATIONS_FILE as at, runChunkedAnalystDefinition as b, nodeHttpPrimeBridgeTransport as c, summarizeAgentRxCalibration as ct, publicBenchmarkDistributions as d, agentRxBenchmarkCase as dt, analystBenchmarkDependencyLockDigest as et, publicBenchmarkSelectionReport as f, agentRxPredictionsToFindings as ft, rlmEngineLimits as g, publicRlmAnalystDefinition as h, normalizeBenchmarkLabel as ht, createPrimeBenchmarkRunner as i, ANALYST_BENCHMARK_MANIFEST_FILE as it, compareAnalystRunners as j, renderCodeTraceCalibrationMarkdown as k, loadPublicBenchmarkRows as l, codeTraceBenchCase as lt, createPublicBenchmarkRlmRunner as m, roundAgentRxStep as mt, runAnalystBenchmarkCommand as n, ANALYST_BENCHMARK_COST_LEDGER_FILE as nt, primeCodeTraceAnalystDefinition as o, AGENT_RX_UPSTREAM_REVISION as ot, selectPublicBenchmarkRows as p, normalizeAgentRxCategory as pt, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as q, renderAnalystBenchmarkMarkdown as r, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as rt, runInlineAnalystDefinition as s, renderAgentRxCalibrationMarkdown as st, ANALYST_BENCHMARK_HELP as t, analystBenchmarkImplementationDigest as tt, preparePublicAnalystBenchmark as u, codeTracerPredictionsToFindings as ut, createPublicBenchmarkDirectRunner as v, analystDefinitionProtocolSha256 as w, decodeReplyRows as x, publicDirectAnalystDefinition as y, publicBenchmarkRlmInstructions as z };
6414
6414
 
6415
- //# sourceMappingURL=benchmark-command-Dtym4pA-.js.map
6415
+ //# sourceMappingURL=benchmark-command-BteMFN62.js.map