@tangle-network/agent-eval 0.150.1 → 0.150.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/CHANGELOG.md +29 -0
  2. package/dist/analyst/index.d.ts +3 -3
  3. package/dist/analyst/index.js +3 -3
  4. package/dist/{benchmark-command-CAFwbH0L.js → benchmark-command-BU1Las59.js} +6 -6
  5. package/dist/{benchmark-command-CAFwbH0L.js.map → benchmark-command-BU1Las59.js.map} +1 -1
  6. package/dist/benchmarks/index.js +3 -3
  7. package/dist/builder-eval/index.js +1 -1
  8. package/dist/campaign/index.js +5 -5
  9. package/dist/{campaign-CN_7xJdV.js → campaign-la-gEYNz.js} +7 -7
  10. package/dist/{campaign-CN_7xJdV.js.map → campaign-la-gEYNz.js.map} +1 -1
  11. package/dist/{chat-client-2bVfrzhN.js → chat-client-Bvmxedyv.js} +2 -2
  12. package/dist/{chat-client-2bVfrzhN.js.map → chat-client-Bvmxedyv.js.map} +1 -1
  13. package/dist/cli.js +1 -1
  14. package/dist/contract/index.js +5 -5
  15. package/dist/{define-agent-eval-rqNyVhVV.js → define-agent-eval-C8V8sMqP.js} +7 -7
  16. package/dist/{define-agent-eval-rqNyVhVV.js.map → define-agent-eval-C8V8sMqP.js.map} +1 -1
  17. package/dist/{descriptive-B5MwKfbf.js → descriptive-jDOuI6mz.js} +22 -2
  18. package/dist/descriptive-jDOuI6mz.js.map +1 -0
  19. package/dist/{dspy-rlm-engine-DHI0WrUU.js → dspy-rlm-engine-CBYlPvNy.js} +2 -2
  20. package/dist/{dspy-rlm-engine-DHI0WrUU.js.map → dspy-rlm-engine-CBYlPvNy.js.map} +1 -1
  21. package/dist/{eval-campaign-C4jmuM-b.js → eval-campaign-CQuZrLR_.js} +2 -2
  22. package/dist/{eval-campaign-C4jmuM-b.js.map → eval-campaign-CQuZrLR_.js.map} +1 -1
  23. package/dist/{external-optimizer-process-Bhmzngf-.js → external-optimizer-process-x9oEXKsU.js} +2 -2
  24. package/dist/{external-optimizer-process-Bhmzngf-.js.map → external-optimizer-process-x9oEXKsU.js.map} +1 -1
  25. package/dist/{external-optimizer-subprocess-Dn90UqN2.js → external-optimizer-subprocess-CKNb42oM.js} +7 -3
  26. package/dist/external-optimizer-subprocess-CKNb42oM.js.map +1 -0
  27. package/dist/{index-CTKpu9ry.d.ts → index-Aj3WO3_a.d.ts} +2 -2
  28. package/dist/{index-CTKpu9ry.d.ts.map → index-Aj3WO3_a.d.ts.map} +1 -1
  29. package/dist/index.d.ts +3 -72
  30. package/dist/index.d.ts.map +1 -1
  31. package/dist/index.js +11 -11
  32. package/dist/{integrity-DysDBWDu.js → integrity-DL91tucI.js} +16 -3
  33. package/dist/integrity-DL91tucI.js.map +1 -0
  34. package/dist/{judge-calibration-DZkWrm5H.js → judge-calibration-zZjLz8hr.js} +2 -2
  35. package/dist/{judge-calibration-DZkWrm5H.js.map → judge-calibration-zZjLz8hr.js.map} +1 -1
  36. package/dist/{llm-judge-CVq33oz1.js → llm-judge-BqqMS8t7.js} +4 -4
  37. package/dist/{llm-judge-CVq33oz1.js.map → llm-judge-BqqMS8t7.js.map} +1 -1
  38. package/dist/meta-eval/index.js +2 -2
  39. package/dist/openapi.json +1 -1
  40. package/dist/pipelines/index.js +1 -1
  41. package/dist/{produced-state-jfk8Du3b.js → produced-state-Bnq4FaDO.js} +2 -2
  42. package/dist/{produced-state-jfk8Du3b.js.map → produced-state-Bnq4FaDO.js.map} +1 -1
  43. package/dist/reporting.js +2 -2
  44. package/dist/{reward-hacking-DKI9T52l.js → reward-hacking-C0x0xihA.js} +2 -2
  45. package/dist/{reward-hacking-DKI9T52l.js.map → reward-hacking-C0x0xihA.js.map} +1 -1
  46. package/dist/rl.js +3 -3
  47. package/dist/{rubric-predictive-validity-Cwwyd7ah.js → rubric-predictive-validity-CzxLoZge.js} +2 -2
  48. package/dist/{rubric-predictive-validity-Cwwyd7ah.js.map → rubric-predictive-validity-CzxLoZge.js.map} +1 -1
  49. package/dist/{skillopt-optimization-method-DLeUcK-K.js → skillopt-optimization-method-CPBlTcj5.js} +4 -4
  50. package/dist/{skillopt-optimization-method-DLeUcK-K.js.map → skillopt-optimization-method-CPBlTcj5.js.map} +1 -1
  51. package/dist/{summary-report-Blysd6Z2.js → summary-report-DW2bEpdB.js} +2 -2
  52. package/dist/{summary-report-Blysd6Z2.js.map → summary-report-DW2bEpdB.js.map} +1 -1
  53. package/dist/supervisor-run/index.d.ts +26 -4
  54. package/dist/supervisor-run/index.d.ts.map +1 -1
  55. package/dist/supervisor-run/index.js +102 -24
  56. package/dist/supervisor-run/index.js.map +1 -1
  57. package/dist/{tool-waste-BDdBZG1F.js → tool-waste-C-VHSRwF.js} +3 -3
  58. package/dist/{tool-waste-BDdBZG1F.js.map → tool-waste-C-VHSRwF.js.map} +1 -1
  59. package/dist/{types-yLK8gXE9.d.ts → types-I5WwVzQ7.d.ts} +159 -9
  60. package/dist/types-I5WwVzQ7.d.ts.map +1 -0
  61. package/docs/adapters-observability.md +9 -23
  62. package/docs/campaign-proposers.md +13 -5
  63. package/docs/concepts.md +3 -4
  64. package/docs/wire-protocol.md +1 -1
  65. package/package.json +1 -1
  66. package/dist/descriptive-B5MwKfbf.js.map +0 -1
  67. package/dist/external-optimizer-subprocess-Dn90UqN2.js.map +0 -1
  68. package/dist/integrity-DysDBWDu.js.map +0 -1
  69. package/dist/types-yLK8gXE9.d.ts.map +0 -1
@@ -1,6 +1,6 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { r as pearsonR } from "./descriptive-B5MwKfbf.js";
3
- import { r as continuousAgreement } from "./judge-calibration-DZkWrm5H.js";
2
+ import { r as pearsonR } from "./descriptive-jDOuI6mz.js";
3
+ import { r as continuousAgreement } from "./judge-calibration-zZjLz8hr.js";
4
4
  import { i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy } from "./query-Di7eEQ79.js";
5
5
  //#region src/statistics/agreement-irr.ts
6
6
  /**
@@ -800,4 +800,4 @@ function stringify(v) {
800
800
  //#endregion
801
801
  export { budgetBreachView as a, corpusInterRaterAgreementFromJudgeScores as c, failureClusterView as i, interRaterReliability as l, computeToolUseMetrics as n, classifyFailure as o, judgeAgreementView as r, corpusInterRaterAgreement as s, toolWasteView as t };
802
802
 
803
- //# sourceMappingURL=tool-waste-BDdBZG1F.js.map
803
+ //# sourceMappingURL=tool-waste-C-VHSRwF.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"tool-waste-BDdBZG1F.js","names":["prev"],"sources":["../src/statistics/agreement-irr.ts","../src/failure-taxonomy.ts","../src/pipelines/budget-breach.ts","../src/pipelines/failure-cluster.ts","../src/pipelines/judge-agreement.ts","../src/tool-use-metrics.ts","../src/pipelines/tool-waste.ts"],"sourcesContent":["import { ValidationError } from '../errors'\nimport {\n type ContinuousAgreement,\n type ContinuousAgreementOptions,\n continuousAgreement,\n} from '../judge-calibration'\nimport type { JudgeScore } from '../types'\n\n/**\n * Inter-rater reliability — Krippendorff's α under the squared-difference\n * metric, pooled across dimensions.\n *\n * Each inner array is one judge's scores. Items are matched by position\n * WITHIN a dimension: the k-th score a judge supplies carrying dimension\n * `d` is item k of `d`, and the ratings compared against each other are\n * the ones different judges gave to the same item. Every judge that scores\n * a dimension at all must supply the same number of scores for it —\n * ragged input cannot be aligned into items and throws rather than\n * comparing mismatched items.\n *\n * α = 1 − D_observed / D_expected: D_observed averages the squared\n * difference over within-item judge pairs, D_expected over every pair of\n * ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,\n * negative is systematic disagreement.\n */\nexport function interRaterReliability(judgeScores: JudgeScore[][]): number {\n if (judgeScores.length < 2) return 1\n\n // dimension → one score sequence per judge, in that judge's supplied order.\n const perDimension = new Map<string, number[][]>()\n for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) {\n for (const s of judgeScores[judgeIndex]!) {\n let byJudge = perDimension.get(s.dimension)\n if (byJudge === undefined) {\n byJudge = Array.from({ length: judgeScores.length }, () => [] as number[])\n perDimension.set(s.dimension, byJudge)\n }\n byJudge[judgeIndex]!.push(s.score)\n }\n }\n\n const allValues: number[] = []\n const pairDiffs: number[] = []\n\n for (const [dimension, byJudge] of perDimension) {\n const scoring = byJudge.filter((scores) => scores.length > 0)\n if (scoring.length < 2) continue\n const itemCount = scoring[0]!.length\n if (scoring.some((scores) => scores.length !== itemCount)) {\n throw new ValidationError(\n `interRaterReliability: dimension '${dimension}' has judges supplying ` +\n `${scoring.map((scores) => scores.length).join('/')} scores — items cannot be aligned`,\n )\n }\n for (let item = 0; item < itemCount; item++) {\n const ratings = scoring.map((scores) => scores[item]!)\n for (const v of ratings) allValues.push(v)\n for (let i = 0; i < ratings.length; i++) {\n for (let j = i + 1; j < ratings.length; j++) {\n pairDiffs.push((ratings[i]! - ratings[j]!) ** 2)\n }\n }\n }\n }\n\n if (pairDiffs.length === 0 || allValues.length < 2) return 1\n\n const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length\n\n // Expected disagreement from all possible pairings of values\n let expectedDisagreement = 0\n let expectedCount = 0\n for (let i = 0; i < allValues.length; i++) {\n for (let j = i + 1; j < allValues.length; j++) {\n expectedDisagreement += (allValues[i]! - allValues[j]!) ** 2\n expectedCount++\n }\n }\n expectedDisagreement = expectedCount > 0 ? expectedDisagreement / expectedCount : 0\n\n if (expectedDisagreement === 0) return 1\n return 1 - observedDisagreement / expectedDisagreement\n}\n\n// ── Corpus-wide inter-rater agreement ──────────────────────────────\n//\n// `interRaterReliability(judgeScores)` computes a within-item\n// Krippendorff α — multiple judges score *the same item* and we ask\n// \"how much do their scores agree on that item?\" Useful for a single\n// scenario, but it cannot answer \"how reliable are these judges across\n// the whole evaluation corpus?\"\n//\n// `corpusInterRaterAgreement` does the corpus-wide question properly.\n// Inputs are flat per-(item, judge, dimension) score records. For each\n// dimension we pivot to a complete [n_items × n_judges] matrix and feed\n// it to the ICC(2,1) + κ_w machinery already validated in\n// `judge-calibration.ts`. An overall pooled metric averages the\n// per-dimension ICC/κ across dimensions.\n\nexport interface CorpusScoreRecord {\n /** Stable identifier for the rated item (scenario, span, turn, …). */\n itemId: string\n /** Identifier for the judge that produced this score. */\n judgeName: string\n /** Dimension name (matches `JudgeScore.dimension`). */\n dimension: string\n /** Numeric score; must be finite. */\n score: number\n}\n\nexport interface CorpusAgreementPerDimension extends ContinuousAgreement {\n dimension: string\n /** Item IDs that contributed to this dimension's matrix (every judge scored them). */\n itemIds: string[]\n /** Judge IDs that contributed to this dimension's matrix. */\n judgeIds: string[]\n}\n\nexport interface CorpusAgreementReport {\n /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */\n perDimension: CorpusAgreementPerDimension[]\n /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */\n overallIcc: number\n /** Mean weighted κ across dimensions (NaN if none finite). */\n overallWeightedKappa: number\n /** Dimensions evaluated (sorted). */\n dimensions: string[]\n /** Judges seen across the corpus (sorted). */\n judgeIds: string[]\n}\n\nexport interface CorpusAgreementOptions extends ContinuousAgreementOptions {\n /**\n * Restrict the audit to these dimensions. Default = every dimension\n * that appears in the input. A dimension named here but absent from\n * the input throws — silent omission would corrupt the overall metric.\n */\n dimensions?: string[]\n /**\n * Restrict the audit to these judges. Default = every judge that\n * appears in the input. A judge named here but absent from a\n * dimension throws (see \"fail loud\" below).\n */\n judges?: string[]\n}\n\n/**\n * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.\n *\n * For each dimension, builds the [n_items][n_judges] matrix of scores\n * (keeping only items every judge rated on that dimension), then runs\n * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and\n * bootstrap CIs. Reports a pooled mean across dimensions as a single\n * \"is this judge panel reliable on this corpus?\" number.\n *\n * Fail-loud contract:\n * - Empty input throws.\n * - Fewer than 2 judges or fewer than 2 items per dimension throws.\n * - A judge present in some dimensions but with zero scored items on\n * another dimension throws (would silently shrink the matrix).\n * - Duplicate (itemId, judgeName, dimension) records throw.\n */\nexport function corpusInterRaterAgreement(\n records: CorpusScoreRecord[],\n opts: CorpusAgreementOptions = {},\n): CorpusAgreementReport {\n if (records.length === 0) {\n throw new ValidationError('corpusInterRaterAgreement: no score records supplied')\n }\n\n const judgesSeen = new Set<string>()\n const dimsSeen = new Set<string>()\n // dimension → judge → itemId → score\n const grid = new Map<string, Map<string, Map<string, number>>>()\n\n for (const r of records) {\n if (!Number.isFinite(r.score)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: non-finite score for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`,\n )\n }\n judgesSeen.add(r.judgeName)\n dimsSeen.add(r.dimension)\n const byJudge = grid.get(r.dimension) ?? new Map<string, Map<string, number>>()\n const byItem = byJudge.get(r.judgeName) ?? new Map<string, number>()\n if (byItem.has(r.itemId)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: duplicate record for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`,\n )\n }\n byItem.set(r.itemId, r.score)\n byJudge.set(r.judgeName, byItem)\n grid.set(r.dimension, byJudge)\n }\n\n const targetDims = opts.dimensions ?? [...dimsSeen].sort()\n for (const d of targetDims) {\n if (!dimsSeen.has(d)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${d}' was requested but no records carry it`,\n )\n }\n }\n const targetJudges = opts.judges ? [...opts.judges] : [...judgesSeen].sort()\n for (const j of targetJudges) {\n if (!judgesSeen.has(j)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: judge '${j}' was requested but produced no records`,\n )\n }\n }\n if (targetJudges.length < 2) {\n throw new ValidationError(\n `corpusInterRaterAgreement: need ≥2 judges, got ${targetJudges.length}`,\n )\n }\n\n const perDimension: CorpusAgreementPerDimension[] = []\n const iccs: number[] = []\n const kappas: number[] = []\n\n for (const dim of targetDims) {\n const byJudge = grid.get(dim)!\n // Fail loud: every requested judge must have scored ≥1 item on this dim.\n const judgeItemCounts: Record<string, number> = {}\n for (const j of targetJudges) {\n const m = byJudge.get(j)\n judgeItemCounts[j] = m?.size ?? 0\n }\n const emptyJudges = targetJudges.filter((j) => judgeItemCounts[j] === 0)\n if (emptyJudges.length > 0) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${dim}' has no scores from judge(s) ${emptyJudges.join(', ')} (counts: ${JSON.stringify(judgeItemCounts)})`,\n )\n }\n\n // Items rated by *every* requested judge on this dim.\n let commonItems: Set<string> | null = null\n for (const j of targetJudges) {\n const ids = new Set(byJudge.get(j)!.keys())\n if (commonItems === null) {\n commonItems = ids\n } else {\n const prev: Set<string> = commonItems\n commonItems = new Set([...prev].filter((x) => ids.has(x)))\n }\n }\n const sortedItems = [...(commonItems ?? new Set<string>())].sort()\n if (sortedItems.length < 2) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${dim}' has ${sortedItems.length} item(s) rated by all ${targetJudges.length} judges (need ≥2)`,\n )\n }\n\n const matrix: number[][] = sortedItems.map((itemId) =>\n targetJudges.map((j) => byJudge.get(j)!.get(itemId)!),\n )\n const agreement = continuousAgreement(matrix, opts)\n perDimension.push({\n ...agreement,\n dimension: dim,\n itemIds: sortedItems,\n judgeIds: [...targetJudges],\n })\n if (Number.isFinite(agreement.icc)) iccs.push(agreement.icc)\n if (Number.isFinite(agreement.weightedKappa)) kappas.push(agreement.weightedKappa)\n }\n\n const mean = (xs: number[]) =>\n xs.length === 0 ? Number.NaN : xs.reduce((a, b) => a + b, 0) / xs.length\n return {\n perDimension,\n overallIcc: mean(iccs),\n overallWeightedKappa: mean(kappas),\n dimensions: targetDims,\n judgeIds: targetJudges,\n }\n}\n\n/**\n * Convenience adapter for `JudgeScore[]` data keyed externally by item.\n *\n * Use when you have per-item arrays of `JudgeScore[]` (e.g. one\n * `ScenarioResult.judgeScores` per scenario) and want corpus-wide\n * agreement without manually flattening. `itemId` must be unique per\n * row of `itemsScores`.\n */\nexport function corpusInterRaterAgreementFromJudgeScores(\n itemsScores: Array<{ itemId: string; scores: JudgeScore[] }>,\n opts: CorpusAgreementOptions = {},\n): CorpusAgreementReport {\n const records: CorpusScoreRecord[] = []\n const seen = new Set<string>()\n for (const { itemId, scores } of itemsScores) {\n if (seen.has(itemId)) {\n throw new ValidationError(\n `corpusInterRaterAgreementFromJudgeScores: duplicate itemId '${itemId}'`,\n )\n }\n seen.add(itemId)\n for (const s of scores) {\n records.push({\n itemId,\n judgeName: s.judgeName,\n dimension: s.dimension,\n score: s.score,\n })\n }\n }\n return corpusInterRaterAgreement(records, opts)\n}\n","/**\n * Failure taxonomy — canonical classes + a default classifier.\n *\n * Every failed run should end up in a named class. The classifier here\n * is rule-based (fast, deterministic); an LLM fallback can be added by\n * the consumer for novel cases and trained into the rule base over time.\n *\n * Consumers call `classifyFailure(run, spans, events)` and persist the\n * returned class as `Run.outcome.failureClass`.\n */\n\nimport type { FailureClass, Run, Span, TraceEvent } from './trace/schema'\nimport { FAILURE_CLASSES } from './trace/schema'\n\nexport { FAILURE_CLASSES, type FailureClass }\n\nexport interface FailureContext {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n}\n\nexport interface FailureClassification {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n}\n\n/** Ordered rules — first match wins. */\nexport interface FailureRule {\n id: string\n match: (ctx: FailureContext) => {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n } | null\n}\n\nexport const DEFAULT_RULES: FailureRule[] = [\n // Outcome already named? Respect it.\n {\n id: 'explicit-outcome',\n match: ({ run }) => {\n const fc = run.outcome?.failureClass\n if (fc && fc !== 'unknown')\n return { failureClass: fc, reason: 'outcome.failureClass set explicitly' }\n return null\n },\n },\n {\n id: 'knowledge-readiness-blocked',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'readiness_scored' &&\n e.payload.passed === false,\n )\n return event\n ? {\n failureClass: 'knowledge_readiness_blocked',\n reason: 'knowledge readiness report blocked execution',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-integration-manifest',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_validated' && e.payload.valid === false) ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'manifest_invalid')),\n )\n return event\n ? {\n failureClass: 'bad_integration_manifest',\n reason: 'integration manifest validation failed before launch',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-connection',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_manifest_resolved' &&\n hasResolutionStatus(e.payload, 'missing_connection'),\n )\n return event\n ? {\n failureClass: 'missing_integration_connection',\n reason: 'required integration connection was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-scope',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_resolved' && hasMissingScopes(e.payload)) ||\n (e.payload.kind === 'integration_invoke_failed' && e.payload.code === 'scope_denied')),\n )\n return event\n ? {\n failureClass: 'missing_integration_scope',\n reason: 'integration grant or connection lacks required scopes',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-approval-required',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_invoke' && e.payload.status === 'approval_required') ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'approval_required') ||\n e.payload.kind === 'integration_approval_required'),\n )\n return event\n ? {\n failureClass: 'integration_approval_required',\n reason: 'integration write paused for user approval',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-auth-expired',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'auth_expired' ||\n e.payload.code === 'connection_not_active' ||\n e.payload.code === 'capability_expired' ||\n e.payload.status === 'expired'),\n )\n return event\n ? {\n failureClass: 'integration_auth_expired',\n reason: 'integration connection or capability expired',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'unsafe-integration-write-denied',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'unsafe_write_denied' ||\n e.payload.code === 'policy_denied' ||\n e.payload.code === 'action_denied'),\n )\n return event\n ? {\n failureClass: 'unsafe_integration_write_denied',\n reason: 'integration write was denied by policy or capability scope',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-provider-failure',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n ![\n 'scope_denied',\n 'approval_required',\n 'auth_expired',\n 'connection_not_active',\n 'capability_expired',\n 'unsafe_write_denied',\n 'policy_denied',\n 'action_denied',\n 'manifest_invalid',\n ].includes(String(e.payload.code)),\n )\n return event\n ? {\n failureClass: 'integration_provider_failure',\n reason: 'integration provider invocation failed',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-credentials',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.category === 'credential_or_secret',\n )\n return event\n ? {\n failureClass: 'missing_credentials',\n reason: 'required credential or secret was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-retrieval',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const retrieval = spans.find(\n (s) =>\n s.kind === 'retrieval' && (s.hits.length === 0 || s.hits.every((hit) => hit.score <= 0)),\n )\n return retrieval\n ? {\n failureClass: 'bad_retrieval',\n reason: 'retrieval returned no useful hits for a failed run',\n triggerSpanId: retrieval.spanId,\n }\n : null\n },\n },\n {\n id: 'insufficient-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'insufficient_evidence',\n )\n return event\n ? {\n failureClass: 'insufficient_evidence',\n reason: 'task proceeded with insufficient supporting evidence',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'contradictory-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'contradictory_evidence',\n )\n return event\n ? {\n failureClass: 'contradictory_evidence',\n reason: 'supporting evidence contradicted itself',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n // Budget breach events\n {\n id: 'budget-breach',\n match: ({ events }) => {\n const breach = events.find((e) => e.kind === 'budget_breach')\n return breach\n ? {\n failureClass: 'budget_exceeded',\n reason: `budget breached on ${breach.payload.dimension ?? 'unknown dimension'}`,\n triggerEventId: breach.eventId,\n }\n : null\n },\n },\n // Policy violations\n {\n id: 'policy-violation',\n match: ({ events }) => {\n const e = events.find((x) => x.kind === 'policy_violation')\n return e\n ? {\n failureClass: 'policy_violation',\n reason: 'policy_violation event emitted',\n triggerEventId: e.eventId,\n }\n : null\n },\n },\n // Sandbox non-zero exit code\n {\n id: 'sandbox-failure',\n match: ({ spans }) => {\n const s = spans.find(\n (x) => x.kind === 'sandbox' && typeof x.exitCode === 'number' && x.exitCode !== 0,\n )\n if (!s) return null\n return {\n failureClass: 'sandbox_failure',\n reason: `sandbox exited ${(s as Extract<Span, { kind: 'sandbox' }>).exitCode}`,\n triggerSpanId: s.spanId,\n }\n },\n },\n // Timeout: run aborted by external signal\n {\n id: 'timeout',\n match: ({ run, events }) => {\n if (run.status !== 'aborted') return null\n const hasTimeout = events.some(\n (e) =>\n e.kind === 'error' &&\n String(e.payload.reason ?? '')\n .toLowerCase()\n .includes('timeout'),\n )\n const note = (run.outcome?.notes ?? '').toLowerCase()\n if (hasTimeout || note.includes('timeout') || note.includes('deadline')) {\n return { failureClass: 'timeout', reason: 'timeout signal observed' }\n }\n return null\n },\n },\n // Tool recovery failure: many consecutive tool errors on the same tool\n {\n id: 'tool-recovery-failure',\n match: ({ spans }) => {\n const tools = spans.filter((s) => s.kind === 'tool')\n const byTool = new Map<string, Span[]>()\n for (const t of tools) {\n const name = (t as Extract<Span, { kind: 'tool' }>).toolName\n const arr = byTool.get(name) ?? []\n arr.push(t)\n byTool.set(name, arr)\n }\n for (const [name, arr] of byTool) {\n const errs = arr.filter((s) => s.status === 'error')\n if (errs.length >= 3 && errs.length === arr.length) {\n return {\n failureClass: 'tool_recovery_failure',\n reason: `${errs.length} consecutive errors on tool \"${name}\"`,\n triggerSpanId: errs[errs.length - 1]!.spanId,\n }\n }\n }\n return null\n },\n },\n // Tool selection error: the run failed and agent called zero tools despite having them\n {\n id: 'tool-selection-error',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const hasToolsAvailable = spans.some(\n (s) =>\n s.kind === 'agent' &&\n (s.attributes?.toolsAvailable as number | undefined) !== undefined &&\n (s.attributes?.toolsAvailable as number) > 0,\n )\n const tools = spans.filter((s) => s.kind === 'tool')\n if (hasToolsAvailable && tools.length === 0) {\n return {\n failureClass: 'tool_selection_error',\n reason: 'tools were available but none were called',\n }\n }\n return null\n },\n },\n // Format drift: scored by a judge with dimension='format' below threshold\n {\n id: 'format-drift',\n match: ({ spans }) => {\n const judge = spans.find(\n (s) =>\n s.kind === 'judge' &&\n (s as Extract<Span, { kind: 'judge' }>).dimension === 'format' &&\n (s as Extract<Span, { kind: 'judge' }>).score < 0.5,\n )\n return judge\n ? {\n failureClass: 'format_drift',\n reason: 'format judge scored below 0.5',\n triggerSpanId: judge.spanId,\n }\n : null\n },\n },\n]\n\nfunction hasResolutionStatus(payload: Record<string, unknown>, status: string): boolean {\n if (status === 'missing_connection' && stringArray(payload.missingConnections).length > 0)\n return true\n return resolutionItems(payload).some((item) => item.status === status)\n}\n\nfunction hasMissingScopes(payload: Record<string, unknown>): boolean {\n if (stringArray(payload.missingScopes).length > 0) return true\n return resolutionItems(payload).some(\n (item) => Array.isArray(item.missingScopes) && item.missingScopes.length > 0,\n )\n}\n\nfunction resolutionItems(payload: Record<string, unknown>): Array<Record<string, unknown>> {\n return [\n ...records(payload.missing),\n ...records(payload.optionalMissing),\n ...records(payload.ready),\n ]\n}\n\nfunction records(value: unknown): Array<Record<string, unknown>> {\n if (!Array.isArray(value)) return []\n return value.filter(\n (item): item is Record<string, unknown> =>\n Boolean(item) && typeof item === 'object' && !Array.isArray(item),\n )\n}\n\nfunction stringArray(value: unknown): string[] {\n return Array.isArray(value)\n ? value.filter((item): item is string => typeof item === 'string')\n : []\n}\n\n/** Classify the failure mode of a run using an ordered rule list. */\nexport function classifyFailure(\n ctx: FailureContext,\n rules: FailureRule[] = DEFAULT_RULES,\n): FailureClassification {\n if (ctx.run.outcome?.pass !== false && ctx.run.status === 'completed') {\n return { failureClass: 'success', reason: 'run completed with pass=true (or no explicit fail)' }\n }\n for (const rule of rules) {\n const hit = rule.match(ctx)\n if (hit) return hit\n }\n return { failureClass: 'unknown', reason: 'no rule matched; run failed for unclassified reason' }\n}\n","/**\n * BudgetBreachView — aggregates breach events across the corpus.\n *\n * Answers: which dimensions get hit most often? Which scenarios are\n * underbudgeted? Which variants trigger the most breaches?\n */\n\nimport type { BudgetSpec } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface BudgetBreachFinding {\n runId: string\n scenarioId: string\n variantId?: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n excessRatio: number\n timestamp: number\n}\n\nexport interface BudgetBreachReport {\n findings: BudgetBreachFinding[]\n byDimension: Record<string, number>\n byScenario: Record<string, number>\n byVariant: Record<string, number>\n totalRuns: number\n breachedRunRatio: number\n}\n\nexport async function budgetBreachView(\n store: TraceStore,\n options: { scenarioId?: string; variantId?: string } = {},\n): Promise<BudgetBreachReport> {\n const runs = await store.listRuns({\n scenarioId: options.scenarioId,\n variantId: options.variantId,\n })\n const findings: BudgetBreachFinding[] = []\n const byDimension: Record<string, number> = {}\n const byScenario: Record<string, number> = {}\n const byVariant: Record<string, number> = {}\n\n for (const run of runs) {\n const entries = await store.budget(run.runId)\n for (const e of entries) {\n if (!e.breached) continue\n const excessRatio = e.limit > 0 ? e.consumed / e.limit : Infinity\n findings.push({\n runId: run.runId,\n scenarioId: run.scenarioId,\n variantId: run.variantId,\n dimension: e.dimension,\n limit: e.limit,\n consumed: e.consumed,\n excessRatio,\n timestamp: e.timestamp,\n })\n byDimension[e.dimension] = (byDimension[e.dimension] ?? 0) + 1\n byScenario[run.scenarioId] = (byScenario[run.scenarioId] ?? 0) + 1\n if (run.variantId) byVariant[run.variantId] = (byVariant[run.variantId] ?? 0) + 1\n }\n }\n\n const breachedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n byDimension,\n byScenario,\n byVariant,\n totalRuns: runs.length,\n breachedRunRatio: runs.length > 0 ? breachedRuns.size / runs.length : 0,\n }\n}\n","/**\n * FailureClusterView — groups failed runs by (failureClass, triggerTool,\n * argHash-prefix) so weekly reviews can prioritize the top-N clusters.\n *\n * Each cluster includes: N runs, scenarios affected, representative\n * error message, a proposed mitigation hint (rule → action table).\n */\n\nimport { classifyFailure, DEFAULT_RULES, type FailureRule } from '../failure-taxonomy'\nimport { argHash, hasCapturedToolArgs, toolSpans } from '../trace/query'\nimport type { FailureClass, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface FailureCluster {\n failureClass: FailureClass\n /** Tool name when the trigger was a tool span, else undefined. */\n toolName?: string\n /** First 16 chars of argHash — clusters similar args. */\n argPrefix?: string\n /**\n * Source dimension when the trigger was a judge span (e.g. `'format'`,\n * `'safety'`, `'correctness'`). Lets cross-template aggregators\n * group failures by the dimension that fired without overloading\n * `argPrefix`. Optional — clusters without this field deserialize cleanly.\n */\n dimension?: string\n runCount: number\n scenarioIds: string[]\n exampleError?: string\n exampleRunId: string\n}\n\nexport interface FailureClusterReport {\n clusters: FailureCluster[]\n totalFailures: number\n totalRuns: number\n}\n\nexport async function failureClusterView(\n store: TraceStore,\n options: { rules?: FailureRule[]; minClusterSize?: number } = {},\n): Promise<FailureClusterReport> {\n const rules = options.rules ?? DEFAULT_RULES\n const minSize = options.minClusterSize ?? 1\n const runs = await store.listRuns()\n\n type Key = string\n const clusters = new Map<Key, FailureCluster>()\n let totalFailures = 0\n\n for (const run of runs) {\n if (run.status === 'completed' && run.outcome?.pass !== false) continue\n totalFailures++\n const spans = await store.spans({ runId: run.runId })\n const events = await store.events({ runId: run.runId })\n const cls = classifyFailure({ run, spans, events }, rules)\n\n let toolName: string | undefined\n let argPrefix: string | undefined\n let dimension: string | undefined\n if (cls.triggerSpanId) {\n const trig = spans.find((s) => s.spanId === cls.triggerSpanId)\n if (trig?.kind === 'tool') {\n toolName = trig.toolName\n if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16)\n } else if (trig?.kind === 'judge') {\n dimension = trig.dimension\n }\n }\n // Fallback: look at the last errored tool span\n if (!toolName) {\n const ts = await toolSpans(store, run.runId)\n const errored = ts.filter((t) => t.status === 'error').pop()\n if (errored) {\n toolName = errored.toolName\n if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16)\n }\n }\n // Secondary signal: any judge span on the failed run carries a\n // dimension. Useful when the rule classified by judge score but\n // didn't surface the trigger span (or surfaced a non-judge span).\n if (!dimension) {\n const judge = spans.find((s) => s.kind === 'judge' && typeof s.dimension === 'string')\n if (judge?.kind === 'judge') dimension = judge.dimension\n }\n\n const key = `${cls.failureClass}|${toolName ?? ''}|${argPrefix ?? ''}|${dimension ?? ''}`\n let cluster = clusters.get(key)\n if (!cluster) {\n cluster = {\n failureClass: cls.failureClass,\n toolName,\n argPrefix,\n dimension,\n runCount: 0,\n scenarioIds: [],\n exampleRunId: run.runId,\n exampleError: firstErrorMessage(spans) ?? cls.reason,\n }\n clusters.set(key, cluster)\n }\n cluster.runCount++\n if (!cluster.scenarioIds.includes(run.scenarioId)) cluster.scenarioIds.push(run.scenarioId)\n }\n\n const arr = [...clusters.values()]\n .filter((c) => c.runCount >= minSize)\n .sort((a, b) => b.runCount - a.runCount)\n\n return { clusters: arr, totalFailures, totalRuns: runs.length }\n}\n\nfunction firstErrorMessage(spans: Span[]): string | undefined {\n const errored = spans.find((s) => s.status === 'error')\n return errored?.error\n}\n","/**\n * JudgeAgreementView — pairwise agreement between judges across the\n * corpus, grouped by dimension.\n *\n * Output drives two workflows:\n * - Judge robustness audit: \"does Claude agree with GPT at κ ≥ 0.6?\"\n * - Calibration tracking: κ vs golden human labels over time (by\n * providing a `humanGoldenJudgeId`).\n */\n\nimport { interRaterReliability, pearsonR } from '../statistics'\nimport type { JudgeSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface JudgePair {\n judgeA: string\n judgeB: string\n dimension: string\n /** Number of (targetSpanId, dimension) tuples both judges scored. */\n commonItems: number\n pearson: number\n krippendorff: number\n}\n\nexport interface JudgeAgreementReport {\n pairs: JudgePair[]\n dimensions: string[]\n judgeIds: string[]\n}\n\nexport async function judgeAgreementView(store: TraceStore): Promise<JudgeAgreementReport> {\n const all = (await store.spans({ kind: 'judge' })).filter(\n (s): s is JudgeSpan => s.kind === 'judge',\n )\n if (all.length === 0) return { pairs: [], dimensions: [], judgeIds: [] }\n\n const byDimension = new Map<string, JudgeSpan[]>()\n for (const s of all) {\n const arr = byDimension.get(s.dimension) ?? []\n arr.push(s)\n byDimension.set(s.dimension, arr)\n }\n\n const judgeIds = [...new Set(all.map((s) => s.judgeId))].sort()\n const pairs: JudgePair[] = []\n for (const [dim, spans] of byDimension) {\n const byJudge = new Map<string, Map<string, number>>()\n for (const s of spans) {\n const m = byJudge.get(s.judgeId) ?? new Map<string, number>()\n m.set(s.targetSpanId, s.score)\n byJudge.set(s.judgeId, m)\n }\n const judgesHere = [...byJudge.keys()]\n for (let i = 0; i < judgesHere.length; i++) {\n for (let j = i + 1; j < judgesHere.length; j++) {\n const judgeI = judgesHere[i]!\n const judgeJ = judgesHere[j]!\n const a = byJudge.get(judgeI)!\n const b = byJudge.get(judgeJ)!\n const common: Array<[number, number]> = []\n for (const [target, scoreA] of a) {\n const scoreB = b.get(target)\n if (scoreB !== undefined) common.push([scoreA, scoreB])\n }\n if (common.length < 2) continue\n const judgeScores = common.map(\n ([scoreA, scoreB]) =>\n [\n { judgeName: judgeI, dimension: dim, score: scoreA, reasoning: '' },\n { judgeName: judgeJ, dimension: dim, score: scoreB, reasoning: '' },\n ] as const,\n )\n const k = interRaterReliability(\n judgeScores[0]!.map((_, k2) => judgeScores.map((pair) => pair[k2]!)),\n )\n pairs.push({\n judgeA: judgeI,\n judgeB: judgeJ,\n dimension: dim,\n commonItems: common.length,\n pearson: pearsonR(\n common.map((c) => c[0]),\n common.map((c) => c[1]),\n ),\n krippendorff: k,\n })\n }\n }\n }\n\n return {\n pairs: pairs.sort((a, b) => b.commonItems - a.commonItems),\n dimensions: [...byDimension.keys()].sort(),\n judgeIds,\n }\n}\n","/**\n * Tool-use metrics — derived purely from trace data.\n *\n * No scoring assumptions: consumers supply optional ground-truth tool\n * selections per turn + optional \"information used downstream\" signals.\n * Without those, we still compute descriptive metrics (error rate,\n * retry rate, duplicate-call rate) that are useful on their own.\n */\n\nimport { argHash, groupBy, hasCapturedToolArgs, toolSpans } from './trace/query'\nimport type { Span } from './trace/schema'\nimport type { TraceStore } from './trace/store'\n\nexport interface ToolUseMetrics {\n runId: string\n totalCalls: number\n /** Calls whose arguments were captured and can be compared for duplication. */\n callsWithCapturedArgs: number\n byTool: Record<string, ToolStats>\n errorRate: number\n /** Ratio of captured-argument calls already seen with the same tool name and arguments. */\n duplicateRate: number\n /** Ratio of error calls followed by ≥1 retry on same tool. */\n retryRate: number\n /** Optional: of the calls agent made, fraction the evaluator marked as \"correct selection\". */\n selectionAccuracy?: number\n}\n\nexport interface ToolStats {\n calls: number\n callsWithCapturedArgs: number\n errors: number\n avgLatencyMs: number\n duplicates: number\n}\n\nexport interface ToolUseOptions {\n /** Map of spanId → whether the evaluator judged the tool selection correct. Optional. */\n selectionLabels?: Record<string, boolean>\n}\n\nexport async function computeToolUseMetrics(\n store: TraceStore,\n runId: string,\n options: ToolUseOptions = {},\n): Promise<ToolUseMetrics> {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n return {\n runId,\n totalCalls: 0,\n callsWithCapturedArgs: 0,\n byTool: {},\n errorRate: 0,\n duplicateRate: 0,\n retryRate: 0,\n }\n }\n\n const byTool: Record<string, ToolStats> = {}\n let totalErrors = 0\n let totalDuplicates = 0\n let callsWithCapturedArgs = 0\n const sortedTools = [...tools].sort((a, b) => a.startedAt - b.startedAt)\n const seenSignatures = new Set<string>()\n\n // duplicate detection + per-tool aggregation\n for (const t of sortedTools) {\n byTool[t.toolName] ??= {\n calls: 0,\n callsWithCapturedArgs: 0,\n errors: 0,\n avgLatencyMs: 0,\n duplicates: 0,\n }\n const stat = byTool[t.toolName]!\n stat.calls += 1\n if (t.status === 'error') {\n stat.errors += 1\n totalErrors += 1\n }\n if (typeof t.latencyMs === 'number') stat.avgLatencyMs += t.latencyMs\n if (hasCapturedToolArgs(t)) {\n callsWithCapturedArgs += 1\n stat.callsWithCapturedArgs += 1\n const sig = `${t.toolName}|${argHash(t.args)}`\n if (seenSignatures.has(sig)) {\n stat.duplicates += 1\n totalDuplicates += 1\n }\n seenSignatures.add(sig)\n }\n }\n\n for (const stat of Object.values(byTool)) {\n stat.avgLatencyMs = stat.calls > 0 ? stat.avgLatencyMs / stat.calls : 0\n }\n\n // retry detection: per-tool chronological adjacency where error → next same-tool call\n let retryOpportunities = 0\n let retriesFollowed = 0\n for (const [, arr] of groupBy(sortedTools, (t) => t.toolName)) {\n for (let i = 0; i < arr.length; i++) {\n if (arr[i]!.status !== 'error') continue\n retryOpportunities += 1\n if (arr[i + 1]) retriesFollowed += 1\n }\n }\n const retryRate = retryOpportunities > 0 ? retriesFollowed / retryOpportunities : 0\n\n let selectionAccuracy: number | undefined\n if (options.selectionLabels) {\n const labeled = sortedTools.filter((t) => t.spanId in options.selectionLabels!)\n if (labeled.length > 0) {\n selectionAccuracy =\n labeled.filter((t) => options.selectionLabels![t.spanId]).length / labeled.length\n }\n }\n\n return {\n runId,\n totalCalls: sortedTools.length,\n callsWithCapturedArgs,\n byTool,\n errorRate: totalErrors / sortedTools.length,\n duplicateRate: callsWithCapturedArgs > 0 ? totalDuplicates / callsWithCapturedArgs : 0,\n retryRate,\n selectionAccuracy,\n }\n}\n\nexport type { Span }\n","/**\n * ToolWasteView — fraction of tool calls whose results weren't used\n * downstream. Without a \"used\" signal we fall back to structural\n * proxies: error calls, duplicate calls, and tool calls followed by\n * zero subsequent LLM spans are all considered waste.\n *\n * Consumers can pass a `usageOracle` that inspects a tool span and\n * returns true iff the tool's result appears in a later LLM message,\n * artifact, or state mutation — that's the canonical definition; the\n * default heuristic is a reasonable fallback.\n */\n\nimport { computeToolUseMetrics } from '../tool-use-metrics'\nimport { llmSpans, toolSpans } from '../trace/query'\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface ToolWasteFinding {\n runId: string\n wastedCalls: number\n totalCalls: number\n wasteRate: number\n}\n\nexport interface ToolWasteReport {\n byRun: ToolWasteFinding[]\n overallWasteRate: number\n}\n\nexport interface ToolWasteOptions {\n runId?: string\n usageOracle?: (tool: ToolSpan, later: { llm: Awaited<ReturnType<typeof llmSpans>> }) => boolean\n}\n\nexport async function toolWasteView(\n store: TraceStore,\n options: ToolWasteOptions = {},\n): Promise<ToolWasteReport> {\n const runs = options.runId ? [options.runId] : (await store.listRuns()).map((r) => r.runId)\n\n const byRun: ToolWasteFinding[] = []\n let totalCalls = 0\n let totalWasted = 0\n for (const runId of runs) {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n byRun.push({ runId, wastedCalls: 0, totalCalls: 0, wasteRate: 0 })\n continue\n }\n const llms = await llmSpans(store, runId)\n // Sort LLM spans once by start time, then build a suffix index of the\n // concatenated message text. `suffixText[i]` is the haystack of every\n // string message content in spans[i..]. Per tool we binary-search the\n // first span started strictly after the tool, then test that one suffix\n // — turning the per-tool O(llms × messages × content) scan into a single\n // O(log llms) lookup over precomputed text.\n const sortedLlm = [...llms].sort((a, b) => a.startedAt - b.startedAt)\n const startTimes = sortedLlm.map((l) => l.startedAt)\n const suffixText = buildSuffixText(sortedLlm)\n let wasted = 0\n for (const t of tools) {\n if (t.status === 'error') {\n wasted++\n continue\n }\n // First LLM span started strictly after this tool (upper-bound search).\n const cutoff = upperBound(startTimes, t.startedAt)\n if (options.usageOracle) {\n if (!options.usageOracle(t, { llm: sortedLlm.slice(cutoff) })) wasted++\n } else {\n // Default heuristic: a tool whose result is NOT mentioned in any\n // later LLM input message is likely wasted. An empty/null result has\n // no payload to propagate downstream — there is nothing to find in a\n // later message, so it is not evidence of waste; skip it.\n const resultStr = stringify(t.result)\n if (resultStr === '') continue\n const haystack = suffixText[cutoff] ?? ''\n const used = haystack.includes(resultStr.slice(0, 120))\n if (!used) wasted++\n }\n }\n const wasteRate = wasted / tools.length\n byRun.push({ runId, wastedCalls: wasted, totalCalls: tools.length, wasteRate })\n totalCalls += tools.length\n totalWasted += wasted\n }\n return { byRun, overallWasteRate: totalCalls > 0 ? totalWasted / totalCalls : 0 }\n}\n\n/**\n * Build per-position suffix haystacks: result[i] is the concatenation of every\n * string message content in spans[i..end]. Built back-to-front so each entry\n * reuses the next one — O(total message text) rather than O(spans²).\n */\nfunction buildSuffixText(spans: LlmSpan[]): string[] {\n const result = new Array<string>(spans.length + 1)\n result[spans.length] = ''\n for (let i = spans.length - 1; i >= 0; i--) {\n const own = spans[i]!.messages.map((m) =>\n typeof m.content === 'string' ? m.content : '',\n ).join('\\n')\n result[i] = `${own}\\n${result[i + 1]}`\n }\n return result\n}\n\n/** Index of the first element strictly greater than `target` in a sorted array. */\nfunction upperBound(sorted: number[], target: number): number {\n let lo = 0\n let hi = sorted.length\n while (lo < hi) {\n const mid = (lo + hi) >>> 1\n if (sorted[mid]! <= target) lo = mid + 1\n else hi = mid\n }\n return lo\n}\n\nfunction stringify(v: unknown): string {\n if (v === null || v === undefined) return ''\n if (typeof v === 'string') return v\n try {\n return JSON.stringify(v)\n } catch {\n return String(v)\n }\n}\n\n// Re-export for convenience in consumers that want both descriptive and usage metrics.\nexport { computeToolUseMetrics }\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAyBA,SAAgB,sBAAsB,aAAqC;CACzE,IAAI,YAAY,SAAS,GAAG,OAAO;CAGnC,MAAM,+BAAe,IAAI,IAAwB;CACjD,KAAK,IAAI,aAAa,GAAG,aAAa,YAAY,QAAQ,cACxD,KAAK,MAAM,KAAK,YAAY,aAAc;EACxC,IAAI,UAAU,aAAa,IAAI,EAAE,SAAS;EAC1C,IAAI,YAAY,KAAA,GAAW;GACzB,UAAU,MAAM,KAAK,EAAE,QAAQ,YAAY,OAAO,SAAS,CAAC,CAAa;GACzE,aAAa,IAAI,EAAE,WAAW,OAAO;EACvC;EACA,QAAQ,WAAW,CAAE,KAAK,EAAE,KAAK;CACnC;CAGF,MAAM,YAAsB,CAAC;CAC7B,MAAM,YAAsB,CAAC;CAE7B,KAAK,MAAM,CAAC,WAAW,YAAY,cAAc;EAC/C,MAAM,UAAU,QAAQ,QAAQ,WAAW,OAAO,SAAS,CAAC;EAC5D,IAAI,QAAQ,SAAS,GAAG;EACxB,MAAM,YAAY,QAAQ,EAAE,CAAE;EAC9B,IAAI,QAAQ,MAAM,WAAW,OAAO,WAAW,SAAS,GACtD,MAAM,IAAI,gBACR,qCAAqC,UAAU,yBAC1C,QAAQ,KAAK,WAAW,OAAO,MAAM,CAAC,CAAC,KAAK,GAAG,EAAE,kCACxD;EAEF,KAAK,IAAI,OAAO,GAAG,OAAO,WAAW,QAAQ;GAC3C,MAAM,UAAU,QAAQ,KAAK,WAAW,OAAO,KAAM;GACrD,KAAK,MAAM,KAAK,SAAS,UAAU,KAAK,CAAC;GACzC,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAClC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KACtC,UAAU,MAAM,QAAQ,KAAM,QAAQ,OAAQ,CAAC;EAGrD;CACF;CAEA,IAAI,UAAU,WAAW,KAAK,UAAU,SAAS,GAAG,OAAO;CAE3D,MAAM,uBAAuB,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,UAAU;CAG9E,IAAI,uBAAuB;CAC3B,IAAI,gBAAgB;CACpB,KAAK,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KACpC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KAAK;EAC7C,yBAAyB,UAAU,KAAM,UAAU,OAAQ;EAC3D;CACF;CAEF,uBAAuB,gBAAgB,IAAI,uBAAuB,gBAAgB;CAElF,IAAI,yBAAyB,GAAG,OAAO;CACvC,OAAO,IAAI,uBAAuB;AACpC;;;;;;;;;;;;;;;;;AAgFA,SAAgB,0BACd,SACA,OAA+B,CAAC,GACT;CACvB,IAAI,QAAQ,WAAW,GACrB,MAAM,IAAI,gBAAgB,sDAAsD;CAGlF,MAAM,6BAAa,IAAI,IAAY;CACnC,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,uBAAO,IAAI,IAA8C;CAE/D,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,yDAAyD,EAAE,OAAO,UAAU,EAAE,UAAU,QAAQ,EAAE,UAAU,EAC9G;EAEF,WAAW,IAAI,EAAE,SAAS;EAC1B,SAAS,IAAI,EAAE,SAAS;EACxB,MAAM,UAAU,KAAK,IAAI,EAAE,SAAS,qBAAK,IAAI,IAAiC;EAC9E,MAAM,SAAS,QAAQ,IAAI,EAAE,SAAS,qBAAK,IAAI,IAAoB;EACnE,IAAI,OAAO,IAAI,EAAE,MAAM,GACrB,MAAM,IAAI,gBACR,yDAAyD,EAAE,OAAO,UAAU,EAAE,UAAU,QAAQ,EAAE,UAAU,EAC9G;EAEF,OAAO,IAAI,EAAE,QAAQ,EAAE,KAAK;EAC5B,QAAQ,IAAI,EAAE,WAAW,MAAM;EAC/B,KAAK,IAAI,EAAE,WAAW,OAAO;CAC/B;CAEA,MAAM,aAAa,KAAK,cAAc,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK;CACzD,KAAK,MAAM,KAAK,YACd,IAAI,CAAC,SAAS,IAAI,CAAC,GACjB,MAAM,IAAI,gBACR,yCAAyC,EAAE,wCAC7C;CAGJ,MAAM,eAAe,KAAK,SAAS,CAAC,GAAG,KAAK,MAAM,IAAI,CAAC,GAAG,UAAU,CAAC,CAAC,KAAK;CAC3E,KAAK,MAAM,KAAK,cACd,IAAI,CAAC,WAAW,IAAI,CAAC,GACnB,MAAM,IAAI,gBACR,qCAAqC,EAAE,wCACzC;CAGJ,IAAI,aAAa,SAAS,GACxB,MAAM,IAAI,gBACR,kDAAkD,aAAa,QACjE;CAGF,MAAM,eAA8C,CAAC;CACrD,MAAM,OAAiB,CAAC;CACxB,MAAM,SAAmB,CAAC;CAE1B,KAAK,MAAM,OAAO,YAAY;EAC5B,MAAM,UAAU,KAAK,IAAI,GAAG;EAE5B,MAAM,kBAA0C,CAAC;EACjD,KAAK,MAAM,KAAK,cAEd,gBAAgB,KADN,QAAQ,IAAI,CACD,CAAC,EAAE,QAAQ;EAElC,MAAM,cAAc,aAAa,QAAQ,MAAM,gBAAgB,OAAO,CAAC;EACvE,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,gBACR,yCAAyC,IAAI,gCAAgC,YAAY,KAAK,IAAI,EAAE,YAAY,KAAK,UAAU,eAAe,EAAE,EAClJ;EAIF,IAAI,cAAkC;EACtC,KAAK,MAAM,KAAK,cAAc;GAC5B,MAAM,MAAM,IAAI,IAAI,QAAQ,IAAI,CAAC,CAAC,CAAE,KAAK,CAAC;GAC1C,IAAI,gBAAgB,MAClB,cAAc;QAGd,cAAc,IAAI,IAAI,CAAC,GAAGA,WAAI,CAAC,CAAC,QAAQ,MAAM,IAAI,IAAI,CAAC,CAAC,CAAC;EAE7D;EACA,MAAM,cAAc,CAAC,GAAI,+BAAe,IAAI,IAAY,CAAE,CAAC,CAAC,KAAK;EACjE,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,gBACR,yCAAyC,IAAI,QAAQ,YAAY,OAAO,wBAAwB,aAAa,OAAO,kBACtH;EAMF,MAAM,YAAY,oBAHS,YAAY,KAAK,WAC1C,aAAa,KAAK,MAAM,QAAQ,IAAI,CAAC,CAAC,CAAE,IAAI,MAAM,CAAE,CAEX,GAAG,IAAI;EAClD,aAAa,KAAK;GAChB,GAAG;GACH,WAAW;GACX,SAAS;GACT,UAAU,CAAC,GAAG,YAAY;EAC5B,CAAC;EACD,IAAI,OAAO,SAAS,UAAU,GAAG,GAAG,KAAK,KAAK,UAAU,GAAG;EAC3D,IAAI,OAAO,SAAS,UAAU,aAAa,GAAG,OAAO,KAAK,UAAU,aAAa;CACnF;CAEA,MAAM,QAAQ,OACZ,GAAG,WAAW,IAAI,MAAa,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;CACpE,OAAO;EACL;EACA,YAAY,KAAK,IAAI;EACrB,sBAAsB,KAAK,MAAM;EACjC,YAAY;EACZ,UAAU;CACZ;AACF;;;;;;;;;AAUA,SAAgB,yCACd,aACA,OAA+B,CAAC,GACT;CACvB,MAAM,UAA+B,CAAC;CACtC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,EAAE,QAAQ,YAAY,aAAa;EAC5C,IAAI,KAAK,IAAI,MAAM,GACjB,MAAM,IAAI,gBACR,+DAA+D,OAAO,EACxE;EAEF,KAAK,IAAI,MAAM;EACf,KAAK,MAAM,KAAK,QACd,QAAQ,KAAK;GACX;GACA,WAAW,EAAE;GACb,WAAW,EAAE;GACb,OAAO,EAAE;EACX,CAAC;CAEL;CACA,OAAO,0BAA0B,SAAS,IAAI;AAChD;;;AC9QA,MAAa,gBAA+B;CAE1C;EACE,IAAI;EACJ,QAAQ,EAAE,UAAU;GAClB,MAAM,KAAK,IAAI,SAAS;GACxB,IAAI,MAAM,OAAO,WACf,OAAO;IAAE,cAAc;IAAI,QAAQ;GAAsC;GAC3E,OAAO;EACT;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,sBACnB,EAAE,QAAQ,WAAW,KACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,oCAAoC,EAAE,QAAQ,UAAU,SAC1E,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,mBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mCACnB,oBAAoB,EAAE,SAAS,oBAAoB,CACvD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,mCAAmC,iBAAiB,EAAE,OAAO,KAC/E,EAAE,QAAQ,SAAS,+BAA+B,EAAE,QAAQ,SAAS,eAC5E;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,wBAAwB,EAAE,QAAQ,WAAW,uBAC/D,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,uBACrB,EAAE,QAAQ,SAAS,gCACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,kBAClB,EAAE,QAAQ,SAAS,2BACnB,EAAE,QAAQ,SAAS,wBACnB,EAAE,QAAQ,WAAW,UAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,yBAClB,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,SAAS,gBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,+BACnB,CAAC;IACC;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;GACF,CAAC,CAAC,SAAS,OAAO,EAAE,QAAQ,IAAI,CAAC,CACrC;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,aAAa,sBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,YAAY,MAAM,MACrB,MACC,EAAE,SAAS,gBAAgB,EAAE,KAAK,WAAW,KAAK,EAAE,KAAK,OAAO,QAAQ,IAAI,SAAS,CAAC,EAC1F;GACA,OAAO,YACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,UAAU;GAC3B,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,uBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,wBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,SAAS,eAAe;GAC5D,OAAO,SACH;IACE,cAAc;IACd,QAAQ,sBAAsB,OAAO,QAAQ,aAAa;IAC1D,gBAAgB,OAAO;GACzB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,IAAI,OAAO,MAAM,MAAM,EAAE,SAAS,kBAAkB;GAC1D,OAAO,IACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,EAAE;GACpB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,IAAI,MAAM,MACb,MAAM,EAAE,SAAS,aAAa,OAAO,EAAE,aAAa,YAAY,EAAE,aAAa,CAClF;GACA,IAAI,CAAC,GAAG,OAAO;GACf,OAAO;IACL,cAAc;IACd,QAAQ,kBAAmB,EAAyC;IACpE,eAAe,EAAE;GACnB;EACF;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,aAAa;GAC1B,IAAI,IAAI,WAAW,WAAW,OAAO;GACrC,MAAM,aAAa,OAAO,MACvB,MACC,EAAE,SAAS,WACX,OAAO,EAAE,QAAQ,UAAU,EAAE,CAAC,CAC3B,YAAY,CAAC,CACb,SAAS,SAAS,CACzB;GACA,MAAM,QAAQ,IAAI,SAAS,SAAS,GAAA,CAAI,YAAY;GACpD,IAAI,cAAc,KAAK,SAAS,SAAS,KAAK,KAAK,SAAS,UAAU,GACpE,OAAO;IAAE,cAAc;IAAW,QAAQ;GAA0B;GAEtE,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,MAAM,yBAAS,IAAI,IAAoB;GACvC,KAAK,MAAM,KAAK,OAAO;IACrB,MAAM,OAAQ,EAAsC;IACpD,MAAM,MAAM,OAAO,IAAI,IAAI,KAAK,CAAC;IACjC,IAAI,KAAK,CAAC;IACV,OAAO,IAAI,MAAM,GAAG;GACtB;GACA,KAAK,MAAM,CAAC,MAAM,QAAQ,QAAQ;IAChC,MAAM,OAAO,IAAI,QAAQ,MAAM,EAAE,WAAW,OAAO;IACnD,IAAI,KAAK,UAAU,KAAK,KAAK,WAAW,IAAI,QAC1C,OAAO;KACL,cAAc;KACd,QAAQ,GAAG,KAAK,OAAO,+BAA+B,KAAK;KAC3D,eAAe,KAAK,KAAK,SAAS,EAAE,CAAE;IACxC;GAEJ;GACA,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,oBAAoB,MAAM,MAC7B,MACC,EAAE,SAAS,WACV,EAAE,YAAY,mBAA0C,KAAA,KACxD,EAAE,YAAY,iBAA4B,CAC/C;GACA,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,IAAI,qBAAqB,MAAM,WAAW,GACxC,OAAO;IACL,cAAc;IACd,QAAQ;GACV;GAEF,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,MACjB,MACC,EAAE,SAAS,WACV,EAAuC,cAAc,YACrD,EAAuC,QAAQ,EACpD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,MAAM;GACvB,IACA;EACN;CACF;AACF;AAEA,SAAS,oBAAoB,SAAkC,QAAyB;CACtF,IAAI,WAAW,wBAAwB,YAAY,QAAQ,kBAAkB,CAAC,CAAC,SAAS,GACtF,OAAO;CACT,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,MAAM;AACvE;AAEA,SAAS,iBAAiB,SAA2C;CACnE,IAAI,YAAY,QAAQ,aAAa,CAAC,CAAC,SAAS,GAAG,OAAO;CAC1D,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAC7B,SAAS,MAAM,QAAQ,KAAK,aAAa,KAAK,KAAK,cAAc,SAAS,CAC7E;AACF;AAEA,SAAS,gBAAgB,SAAkE;CACzF,OAAO;EACL,GAAG,QAAQ,QAAQ,OAAO;EAC1B,GAAG,QAAQ,QAAQ,eAAe;EAClC,GAAG,QAAQ,QAAQ,KAAK;CAC1B;AACF;AAEA,SAAS,QAAQ,OAAgD;CAC/D,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO,CAAC;CACnC,OAAO,MAAM,QACV,SACC,QAAQ,IAAI,KAAK,OAAO,SAAS,YAAY,CAAC,MAAM,QAAQ,IAAI,CACpE;AACF;AAEA,SAAS,YAAY,OAA0B;CAC7C,OAAO,MAAM,QAAQ,KAAK,IACtB,MAAM,QAAQ,SAAyB,OAAO,SAAS,QAAQ,IAC/D,CAAC;AACP;;AAGA,SAAgB,gBACd,KACA,QAAuB,eACA;CACvB,IAAI,IAAI,IAAI,SAAS,SAAS,SAAS,IAAI,IAAI,WAAW,aACxD,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAqD;CAEjG,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,MAAM,KAAK,MAAM,GAAG;EAC1B,IAAI,KAAK,OAAO;CAClB;CACA,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAsD;AAClG;;;AC/aA,eAAsB,iBACpB,OACA,UAAuD,CAAC,GAC3B;CAC7B,MAAM,OAAO,MAAM,MAAM,SAAS;EAChC,YAAY,QAAQ;EACpB,WAAW,QAAQ;CACrB,CAAC;CACD,MAAM,WAAkC,CAAC;CACzC,MAAM,cAAsC,CAAC;CAC7C,MAAM,aAAqC,CAAC;CAC5C,MAAM,YAAoC,CAAC;CAE3C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,UAAU,MAAM,MAAM,OAAO,IAAI,KAAK;EAC5C,KAAK,MAAM,KAAK,SAAS;GACvB,IAAI,CAAC,EAAE,UAAU;GACjB,MAAM,cAAc,EAAE,QAAQ,IAAI,EAAE,WAAW,EAAE,QAAQ;GACzD,SAAS,KAAK;IACZ,OAAO,IAAI;IACX,YAAY,IAAI;IAChB,WAAW,IAAI;IACf,WAAW,EAAE;IACb,OAAO,EAAE;IACT,UAAU,EAAE;IACZ;IACA,WAAW,EAAE;GACf,CAAC;GACD,YAAY,EAAE,cAAc,YAAY,EAAE,cAAc,KAAK;GAC7D,WAAW,IAAI,eAAe,WAAW,IAAI,eAAe,KAAK;GACjE,IAAI,IAAI,WAAW,UAAU,IAAI,cAAc,UAAU,IAAI,cAAc,KAAK;EAClF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA;EACA;EACA;EACA,WAAW,KAAK;EAChB,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;CACxE;AACF;;;;;;;;;;ACnCA,eAAsB,mBACpB,OACA,UAA8D,CAAC,GAChC;CAC/B,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,OAAO,MAAM,MAAM,SAAS;CAGlC,MAAM,2BAAW,IAAI,IAAyB;CAC9C,IAAI,gBAAgB;CAEpB,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,WAAW,eAAe,IAAI,SAAS,SAAS,OAAO;EAC/D;EACA,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,OAAO,IAAI,MAAM,CAAC;EAEpD,MAAM,MAAM,gBAAgB;GAAE;GAAK;GAAO,QAAA,MADrB,MAAM,OAAO,EAAE,OAAO,IAAI,MAAM,CAAC;EACL,GAAG,KAAK;EAEzD,IAAI;EACJ,IAAI;EACJ,IAAI;EACJ,IAAI,IAAI,eAAe;GACrB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,WAAW,IAAI,aAAa;GAC7D,IAAI,MAAM,SAAS,QAAQ;IACzB,WAAW,KAAK;IAChB,IAAI,oBAAoB,IAAI,GAAG,YAAY,QAAQ,KAAK,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GAC3E,OAAO,IAAI,MAAM,SAAS,SACxB,YAAY,KAAK;EAErB;EAEA,IAAI,CAAC,UAAU;GAEb,MAAM,WAAU,MADC,UAAU,OAAO,IAAI,KAAK,EAAA,CACxB,QAAQ,MAAM,EAAE,WAAW,OAAO,CAAC,CAAC,IAAI;GAC3D,IAAI,SAAS;IACX,WAAW,QAAQ;IACnB,IAAI,oBAAoB,OAAO,GAAG,YAAY,QAAQ,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GACjF;EACF;EAIA,IAAI,CAAC,WAAW;GACd,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,SAAS,WAAW,OAAO,EAAE,cAAc,QAAQ;GACrF,IAAI,OAAO,SAAS,SAAS,YAAY,MAAM;EACjD;EAEA,MAAM,MAAM,GAAG,IAAI,aAAa,GAAG,YAAY,GAAG,GAAG,aAAa,GAAG,GAAG,aAAa;EACrF,IAAI,UAAU,SAAS,IAAI,GAAG;EAC9B,IAAI,CAAC,SAAS;GACZ,UAAU;IACR,cAAc,IAAI;IAClB;IACA;IACA;IACA,UAAU;IACV,aAAa,CAAC;IACd,cAAc,IAAI;IAClB,cAAc,kBAAkB,KAAK,KAAK,IAAI;GAChD;GACA,SAAS,IAAI,KAAK,OAAO;EAC3B;EACA,QAAQ;EACR,IAAI,CAAC,QAAQ,YAAY,SAAS,IAAI,UAAU,GAAG,QAAQ,YAAY,KAAK,IAAI,UAAU;CAC5F;CAMA,OAAO;EAAE,UAJG,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAC/B,QAAQ,MAAM,EAAE,YAAY,OAAO,CAAC,CACpC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAEZ;EAAG;EAAe,WAAW,KAAK;CAAO;AAChE;AAEA,SAAS,kBAAkB,OAAmC;CAE5D,OADgB,MAAM,MAAM,MAAM,EAAE,WAAW,OAClC,CAAC,EAAE;AAClB;;;;;;;;;;;;ACrFA,eAAsB,mBAAmB,OAAkD;CACzF,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,QAAQ,CAAC,EAAA,CAAG,QAChD,MAAsB,EAAE,SAAS,OACpC;CACA,IAAI,IAAI,WAAW,GAAG,OAAO;EAAE,OAAO,CAAC;EAAG,YAAY,CAAC;EAAG,UAAU,CAAC;CAAE;CAEvE,MAAM,8BAAc,IAAI,IAAyB;CACjD,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,MAAM,YAAY,IAAI,EAAE,SAAS,KAAK,CAAC;EAC7C,IAAI,KAAK,CAAC;EACV,YAAY,IAAI,EAAE,WAAW,GAAG;CAClC;CAEA,MAAM,WAAW,CAAC,GAAG,IAAI,IAAI,IAAI,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,KAAK;CAC9D,MAAM,QAAqB,CAAC;CAC5B,KAAK,MAAM,CAAC,KAAK,UAAU,aAAa;EACtC,MAAM,0BAAU,IAAI,IAAiC;EACrD,KAAK,MAAM,KAAK,OAAO;GACrB,MAAM,IAAI,QAAQ,IAAI,EAAE,OAAO,qBAAK,IAAI,IAAoB;GAC5D,EAAE,IAAI,EAAE,cAAc,EAAE,KAAK;GAC7B,QAAQ,IAAI,EAAE,SAAS,CAAC;EAC1B;EACA,MAAM,aAAa,CAAC,GAAG,QAAQ,KAAK,CAAC;EACrC,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KACrC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KAAK;GAC9C,MAAM,SAAS,WAAW;GAC1B,MAAM,SAAS,WAAW;GAC1B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,SAAkC,CAAC;GACzC,KAAK,MAAM,CAAC,QAAQ,WAAW,GAAG;IAChC,MAAM,SAAS,EAAE,IAAI,MAAM;IAC3B,IAAI,WAAW,KAAA,GAAW,OAAO,KAAK,CAAC,QAAQ,MAAM,CAAC;GACxD;GACA,IAAI,OAAO,SAAS,GAAG;GACvB,MAAM,cAAc,OAAO,KACxB,CAAC,QAAQ,YACR,CACE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,GAClE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,CACpE,CACJ;GACA,MAAM,IAAI,sBACR,YAAY,EAAE,CAAE,KAAK,GAAG,OAAO,YAAY,KAAK,SAAS,KAAK,GAAI,CAAC,CACrE;GACA,MAAM,KAAK;IACT,QAAQ;IACR,QAAQ;IACR,WAAW;IACX,aAAa,OAAO;IACpB,SAAS,SACP,OAAO,KAAK,MAAM,EAAE,EAAE,GACtB,OAAO,KAAK,MAAM,EAAE,EAAE,CACxB;IACA,cAAc;GAChB,CAAC;EACH;CAEJ;CAEA,OAAO;EACL,OAAO,MAAM,MAAM,GAAG,MAAM,EAAE,cAAc,EAAE,WAAW;EACzD,YAAY,CAAC,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK;EACzC;CACF;AACF;;;;;;;;;;;ACtDA,eAAsB,sBACpB,OACA,OACA,UAA0B,CAAC,GACF;CACzB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;CAC1C,IAAI,MAAM,WAAW,GACnB,OAAO;EACL;EACA,YAAY;EACZ,uBAAuB;EACvB,QAAQ,CAAC;EACT,WAAW;EACX,eAAe;EACf,WAAW;CACb;CAGF,MAAM,SAAoC,CAAC;CAC3C,IAAI,cAAc;CAClB,IAAI,kBAAkB;CACtB,IAAI,wBAAwB;CAC5B,MAAM,cAAc,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACvE,MAAM,iCAAiB,IAAI,IAAY;CAGvC,KAAK,MAAM,KAAK,aAAa;EAC3B,OAAO,EAAE,cAAc;GACrB,OAAO;GACP,uBAAuB;GACvB,QAAQ;GACR,cAAc;GACd,YAAY;EACd;EACA,MAAM,OAAO,OAAO,EAAE;EACtB,KAAK,SAAS;EACd,IAAI,EAAE,WAAW,SAAS;GACxB,KAAK,UAAU;GACf,eAAe;EACjB;EACA,IAAI,OAAO,EAAE,cAAc,UAAU,KAAK,gBAAgB,EAAE;EAC5D,IAAI,oBAAoB,CAAC,GAAG;GAC1B,yBAAyB;GACzB,KAAK,yBAAyB;GAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,GAAG,QAAQ,EAAE,IAAI;GAC3C,IAAI,eAAe,IAAI,GAAG,GAAG;IAC3B,KAAK,cAAc;IACnB,mBAAmB;GACrB;GACA,eAAe,IAAI,GAAG;EACxB;CACF;CAEA,KAAK,MAAM,QAAQ,OAAO,OAAO,MAAM,GACrC,KAAK,eAAe,KAAK,QAAQ,IAAI,KAAK,eAAe,KAAK,QAAQ;CAIxE,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,KAAK,MAAM,GAAG,QAAQ,QAAQ,cAAc,MAAM,EAAE,QAAQ,GAC1D,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,IAAI,IAAI,EAAE,CAAE,WAAW,SAAS;EAChC,sBAAsB;EACtB,IAAI,IAAI,IAAI,IAAI,mBAAmB;CACrC;CAEF,MAAM,YAAY,qBAAqB,IAAI,kBAAkB,qBAAqB;CAElF,IAAI;CACJ,IAAI,QAAQ,iBAAiB;EAC3B,MAAM,UAAU,YAAY,QAAQ,MAAM,EAAE,UAAU,QAAQ,eAAgB;EAC9E,IAAI,QAAQ,SAAS,GACnB,oBACE,QAAQ,QAAQ,MAAM,QAAQ,gBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS,QAAQ;CAEjF;CAEA,OAAO;EACL;EACA,YAAY,YAAY;EACxB;EACA;EACA,WAAW,cAAc,YAAY;EACrC,eAAe,wBAAwB,IAAI,kBAAkB,wBAAwB;EACrF;EACA;CACF;AACF;;;;;;;;;;;;;;AC/FA,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,OAAO,QAAQ,QAAQ,CAAC,QAAQ,KAAK,KAAK,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,MAAM,EAAE,KAAK;CAE1F,MAAM,QAA4B,CAAC;CACnC,IAAI,aAAa;CACjB,IAAI,cAAc;CAClB,KAAK,MAAM,SAAS,MAAM;EACxB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;EAC1C,IAAI,MAAM,WAAW,GAAG;GACtB,MAAM,KAAK;IAAE;IAAO,aAAa;IAAG,YAAY;IAAG,WAAW;GAAE,CAAC;GACjE;EACF;EAQA,MAAM,YAAY,CAAC,GAAG,MAPH,SAAS,OAAO,KAAK,CAOd,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EACpE,MAAM,aAAa,UAAU,KAAK,MAAM,EAAE,SAAS;EACnD,MAAM,aAAa,gBAAgB,SAAS;EAC5C,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,OAAO;GACrB,IAAI,EAAE,WAAW,SAAS;IACxB;IACA;GACF;GAEA,MAAM,SAAS,WAAW,YAAY,EAAE,SAAS;GACjD,IAAI,QAAQ,aACN;QAAA,CAAC,QAAQ,YAAY,GAAG,EAAE,KAAK,UAAU,MAAM,MAAM,EAAE,CAAC,GAAG;GAAA,OAC1D;IAKL,MAAM,YAAY,UAAU,EAAE,MAAM;IACpC,IAAI,cAAc,IAAI;IAGtB,IAAI,EAFa,WAAW,WAAW,GAAA,CACjB,SAAS,UAAU,MAAM,GAAG,GAAG,CAC7C,GAAG;GACb;EACF;EACA,MAAM,YAAY,SAAS,MAAM;EACjC,MAAM,KAAK;GAAE;GAAO,aAAa;GAAQ,YAAY,MAAM;GAAQ;EAAU,CAAC;EAC9E,cAAc,MAAM;EACpB,eAAe;CACjB;CACA,OAAO;EAAE;EAAO,kBAAkB,aAAa,IAAI,cAAc,aAAa;CAAE;AAClF;;;;;;AAOA,SAAS,gBAAgB,OAA4B;CACnD,MAAM,SAAS,IAAI,MAAc,MAAM,SAAS,CAAC;CACjD,OAAO,MAAM,UAAU;CACvB,KAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAIrC,OAAO,KAAK,GAHA,MAAM,EAAE,CAAE,SAAS,KAAK,MAClC,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU,EAC9C,CAAC,CAAC,KAAK,IACU,EAAE,IAAI,OAAO,IAAI;CAEpC,OAAO;AACT;;AAGA,SAAS,WAAW,QAAkB,QAAwB;CAC5D,IAAI,KAAK;CACT,IAAI,KAAK,OAAO;CAChB,OAAO,KAAK,IAAI;EACd,MAAM,MAAO,KAAK,OAAQ;EAC1B,IAAI,OAAO,QAAS,QAAQ,KAAK,MAAM;OAClC,KAAK;CACZ;CACA,OAAO;AACT;AAEA,SAAS,UAAU,GAAoB;CACrC,IAAI,MAAM,QAAQ,MAAM,KAAA,GAAW,OAAO;CAC1C,IAAI,OAAO,MAAM,UAAU,OAAO;CAClC,IAAI;EACF,OAAO,KAAK,UAAU,CAAC;CACzB,QAAQ;EACN,OAAO,OAAO,CAAC;CACjB;AACF"}
1
+ {"version":3,"file":"tool-waste-C-VHSRwF.js","names":["prev"],"sources":["../src/statistics/agreement-irr.ts","../src/failure-taxonomy.ts","../src/pipelines/budget-breach.ts","../src/pipelines/failure-cluster.ts","../src/pipelines/judge-agreement.ts","../src/tool-use-metrics.ts","../src/pipelines/tool-waste.ts"],"sourcesContent":["import { ValidationError } from '../errors'\nimport {\n type ContinuousAgreement,\n type ContinuousAgreementOptions,\n continuousAgreement,\n} from '../judge-calibration'\nimport type { JudgeScore } from '../types'\n\n/**\n * Inter-rater reliability — Krippendorff's α under the squared-difference\n * metric, pooled across dimensions.\n *\n * Each inner array is one judge's scores. Items are matched by position\n * WITHIN a dimension: the k-th score a judge supplies carrying dimension\n * `d` is item k of `d`, and the ratings compared against each other are\n * the ones different judges gave to the same item. Every judge that scores\n * a dimension at all must supply the same number of scores for it —\n * ragged input cannot be aligned into items and throws rather than\n * comparing mismatched items.\n *\n * α = 1 − D_observed / D_expected: D_observed averages the squared\n * difference over within-item judge pairs, D_expected over every pair of\n * ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,\n * negative is systematic disagreement.\n */\nexport function interRaterReliability(judgeScores: JudgeScore[][]): number {\n if (judgeScores.length < 2) return 1\n\n // dimension → one score sequence per judge, in that judge's supplied order.\n const perDimension = new Map<string, number[][]>()\n for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) {\n for (const s of judgeScores[judgeIndex]!) {\n let byJudge = perDimension.get(s.dimension)\n if (byJudge === undefined) {\n byJudge = Array.from({ length: judgeScores.length }, () => [] as number[])\n perDimension.set(s.dimension, byJudge)\n }\n byJudge[judgeIndex]!.push(s.score)\n }\n }\n\n const allValues: number[] = []\n const pairDiffs: number[] = []\n\n for (const [dimension, byJudge] of perDimension) {\n const scoring = byJudge.filter((scores) => scores.length > 0)\n if (scoring.length < 2) continue\n const itemCount = scoring[0]!.length\n if (scoring.some((scores) => scores.length !== itemCount)) {\n throw new ValidationError(\n `interRaterReliability: dimension '${dimension}' has judges supplying ` +\n `${scoring.map((scores) => scores.length).join('/')} scores — items cannot be aligned`,\n )\n }\n for (let item = 0; item < itemCount; item++) {\n const ratings = scoring.map((scores) => scores[item]!)\n for (const v of ratings) allValues.push(v)\n for (let i = 0; i < ratings.length; i++) {\n for (let j = i + 1; j < ratings.length; j++) {\n pairDiffs.push((ratings[i]! - ratings[j]!) ** 2)\n }\n }\n }\n }\n\n if (pairDiffs.length === 0 || allValues.length < 2) return 1\n\n const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length\n\n // Expected disagreement from all possible pairings of values\n let expectedDisagreement = 0\n let expectedCount = 0\n for (let i = 0; i < allValues.length; i++) {\n for (let j = i + 1; j < allValues.length; j++) {\n expectedDisagreement += (allValues[i]! - allValues[j]!) ** 2\n expectedCount++\n }\n }\n expectedDisagreement = expectedCount > 0 ? expectedDisagreement / expectedCount : 0\n\n if (expectedDisagreement === 0) return 1\n return 1 - observedDisagreement / expectedDisagreement\n}\n\n// ── Corpus-wide inter-rater agreement ──────────────────────────────\n//\n// `interRaterReliability(judgeScores)` computes a within-item\n// Krippendorff α — multiple judges score *the same item* and we ask\n// \"how much do their scores agree on that item?\" Useful for a single\n// scenario, but it cannot answer \"how reliable are these judges across\n// the whole evaluation corpus?\"\n//\n// `corpusInterRaterAgreement` does the corpus-wide question properly.\n// Inputs are flat per-(item, judge, dimension) score records. For each\n// dimension we pivot to a complete [n_items × n_judges] matrix and feed\n// it to the ICC(2,1) + κ_w machinery already validated in\n// `judge-calibration.ts`. An overall pooled metric averages the\n// per-dimension ICC/κ across dimensions.\n\nexport interface CorpusScoreRecord {\n /** Stable identifier for the rated item (scenario, span, turn, …). */\n itemId: string\n /** Identifier for the judge that produced this score. */\n judgeName: string\n /** Dimension name (matches `JudgeScore.dimension`). */\n dimension: string\n /** Numeric score; must be finite. */\n score: number\n}\n\nexport interface CorpusAgreementPerDimension extends ContinuousAgreement {\n dimension: string\n /** Item IDs that contributed to this dimension's matrix (every judge scored them). */\n itemIds: string[]\n /** Judge IDs that contributed to this dimension's matrix. */\n judgeIds: string[]\n}\n\nexport interface CorpusAgreementReport {\n /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */\n perDimension: CorpusAgreementPerDimension[]\n /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */\n overallIcc: number\n /** Mean weighted κ across dimensions (NaN if none finite). */\n overallWeightedKappa: number\n /** Dimensions evaluated (sorted). */\n dimensions: string[]\n /** Judges seen across the corpus (sorted). */\n judgeIds: string[]\n}\n\nexport interface CorpusAgreementOptions extends ContinuousAgreementOptions {\n /**\n * Restrict the audit to these dimensions. Default = every dimension\n * that appears in the input. A dimension named here but absent from\n * the input throws — silent omission would corrupt the overall metric.\n */\n dimensions?: string[]\n /**\n * Restrict the audit to these judges. Default = every judge that\n * appears in the input. A judge named here but absent from a\n * dimension throws (see \"fail loud\" below).\n */\n judges?: string[]\n}\n\n/**\n * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.\n *\n * For each dimension, builds the [n_items][n_judges] matrix of scores\n * (keeping only items every judge rated on that dimension), then runs\n * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and\n * bootstrap CIs. Reports a pooled mean across dimensions as a single\n * \"is this judge panel reliable on this corpus?\" number.\n *\n * Fail-loud contract:\n * - Empty input throws.\n * - Fewer than 2 judges or fewer than 2 items per dimension throws.\n * - A judge present in some dimensions but with zero scored items on\n * another dimension throws (would silently shrink the matrix).\n * - Duplicate (itemId, judgeName, dimension) records throw.\n */\nexport function corpusInterRaterAgreement(\n records: CorpusScoreRecord[],\n opts: CorpusAgreementOptions = {},\n): CorpusAgreementReport {\n if (records.length === 0) {\n throw new ValidationError('corpusInterRaterAgreement: no score records supplied')\n }\n\n const judgesSeen = new Set<string>()\n const dimsSeen = new Set<string>()\n // dimension → judge → itemId → score\n const grid = new Map<string, Map<string, Map<string, number>>>()\n\n for (const r of records) {\n if (!Number.isFinite(r.score)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: non-finite score for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`,\n )\n }\n judgesSeen.add(r.judgeName)\n dimsSeen.add(r.dimension)\n const byJudge = grid.get(r.dimension) ?? new Map<string, Map<string, number>>()\n const byItem = byJudge.get(r.judgeName) ?? new Map<string, number>()\n if (byItem.has(r.itemId)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: duplicate record for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`,\n )\n }\n byItem.set(r.itemId, r.score)\n byJudge.set(r.judgeName, byItem)\n grid.set(r.dimension, byJudge)\n }\n\n const targetDims = opts.dimensions ?? [...dimsSeen].sort()\n for (const d of targetDims) {\n if (!dimsSeen.has(d)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${d}' was requested but no records carry it`,\n )\n }\n }\n const targetJudges = opts.judges ? [...opts.judges] : [...judgesSeen].sort()\n for (const j of targetJudges) {\n if (!judgesSeen.has(j)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: judge '${j}' was requested but produced no records`,\n )\n }\n }\n if (targetJudges.length < 2) {\n throw new ValidationError(\n `corpusInterRaterAgreement: need ≥2 judges, got ${targetJudges.length}`,\n )\n }\n\n const perDimension: CorpusAgreementPerDimension[] = []\n const iccs: number[] = []\n const kappas: number[] = []\n\n for (const dim of targetDims) {\n const byJudge = grid.get(dim)!\n // Fail loud: every requested judge must have scored ≥1 item on this dim.\n const judgeItemCounts: Record<string, number> = {}\n for (const j of targetJudges) {\n const m = byJudge.get(j)\n judgeItemCounts[j] = m?.size ?? 0\n }\n const emptyJudges = targetJudges.filter((j) => judgeItemCounts[j] === 0)\n if (emptyJudges.length > 0) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${dim}' has no scores from judge(s) ${emptyJudges.join(', ')} (counts: ${JSON.stringify(judgeItemCounts)})`,\n )\n }\n\n // Items rated by *every* requested judge on this dim.\n let commonItems: Set<string> | null = null\n for (const j of targetJudges) {\n const ids = new Set(byJudge.get(j)!.keys())\n if (commonItems === null) {\n commonItems = ids\n } else {\n const prev: Set<string> = commonItems\n commonItems = new Set([...prev].filter((x) => ids.has(x)))\n }\n }\n const sortedItems = [...(commonItems ?? new Set<string>())].sort()\n if (sortedItems.length < 2) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${dim}' has ${sortedItems.length} item(s) rated by all ${targetJudges.length} judges (need ≥2)`,\n )\n }\n\n const matrix: number[][] = sortedItems.map((itemId) =>\n targetJudges.map((j) => byJudge.get(j)!.get(itemId)!),\n )\n const agreement = continuousAgreement(matrix, opts)\n perDimension.push({\n ...agreement,\n dimension: dim,\n itemIds: sortedItems,\n judgeIds: [...targetJudges],\n })\n if (Number.isFinite(agreement.icc)) iccs.push(agreement.icc)\n if (Number.isFinite(agreement.weightedKappa)) kappas.push(agreement.weightedKappa)\n }\n\n const mean = (xs: number[]) =>\n xs.length === 0 ? Number.NaN : xs.reduce((a, b) => a + b, 0) / xs.length\n return {\n perDimension,\n overallIcc: mean(iccs),\n overallWeightedKappa: mean(kappas),\n dimensions: targetDims,\n judgeIds: targetJudges,\n }\n}\n\n/**\n * Convenience adapter for `JudgeScore[]` data keyed externally by item.\n *\n * Use when you have per-item arrays of `JudgeScore[]` (e.g. one\n * `ScenarioResult.judgeScores` per scenario) and want corpus-wide\n * agreement without manually flattening. `itemId` must be unique per\n * row of `itemsScores`.\n */\nexport function corpusInterRaterAgreementFromJudgeScores(\n itemsScores: Array<{ itemId: string; scores: JudgeScore[] }>,\n opts: CorpusAgreementOptions = {},\n): CorpusAgreementReport {\n const records: CorpusScoreRecord[] = []\n const seen = new Set<string>()\n for (const { itemId, scores } of itemsScores) {\n if (seen.has(itemId)) {\n throw new ValidationError(\n `corpusInterRaterAgreementFromJudgeScores: duplicate itemId '${itemId}'`,\n )\n }\n seen.add(itemId)\n for (const s of scores) {\n records.push({\n itemId,\n judgeName: s.judgeName,\n dimension: s.dimension,\n score: s.score,\n })\n }\n }\n return corpusInterRaterAgreement(records, opts)\n}\n","/**\n * Failure taxonomy — canonical classes + a default classifier.\n *\n * Every failed run should end up in a named class. The classifier here\n * is rule-based (fast, deterministic); an LLM fallback can be added by\n * the consumer for novel cases and trained into the rule base over time.\n *\n * Consumers call `classifyFailure(run, spans, events)` and persist the\n * returned class as `Run.outcome.failureClass`.\n */\n\nimport type { FailureClass, Run, Span, TraceEvent } from './trace/schema'\nimport { FAILURE_CLASSES } from './trace/schema'\n\nexport { FAILURE_CLASSES, type FailureClass }\n\nexport interface FailureContext {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n}\n\nexport interface FailureClassification {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n}\n\n/** Ordered rules — first match wins. */\nexport interface FailureRule {\n id: string\n match: (ctx: FailureContext) => {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n } | null\n}\n\nexport const DEFAULT_RULES: FailureRule[] = [\n // Outcome already named? Respect it.\n {\n id: 'explicit-outcome',\n match: ({ run }) => {\n const fc = run.outcome?.failureClass\n if (fc && fc !== 'unknown')\n return { failureClass: fc, reason: 'outcome.failureClass set explicitly' }\n return null\n },\n },\n {\n id: 'knowledge-readiness-blocked',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'readiness_scored' &&\n e.payload.passed === false,\n )\n return event\n ? {\n failureClass: 'knowledge_readiness_blocked',\n reason: 'knowledge readiness report blocked execution',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-integration-manifest',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_validated' && e.payload.valid === false) ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'manifest_invalid')),\n )\n return event\n ? {\n failureClass: 'bad_integration_manifest',\n reason: 'integration manifest validation failed before launch',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-connection',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_manifest_resolved' &&\n hasResolutionStatus(e.payload, 'missing_connection'),\n )\n return event\n ? {\n failureClass: 'missing_integration_connection',\n reason: 'required integration connection was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-scope',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_resolved' && hasMissingScopes(e.payload)) ||\n (e.payload.kind === 'integration_invoke_failed' && e.payload.code === 'scope_denied')),\n )\n return event\n ? {\n failureClass: 'missing_integration_scope',\n reason: 'integration grant or connection lacks required scopes',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-approval-required',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_invoke' && e.payload.status === 'approval_required') ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'approval_required') ||\n e.payload.kind === 'integration_approval_required'),\n )\n return event\n ? {\n failureClass: 'integration_approval_required',\n reason: 'integration write paused for user approval',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-auth-expired',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'auth_expired' ||\n e.payload.code === 'connection_not_active' ||\n e.payload.code === 'capability_expired' ||\n e.payload.status === 'expired'),\n )\n return event\n ? {\n failureClass: 'integration_auth_expired',\n reason: 'integration connection or capability expired',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'unsafe-integration-write-denied',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'unsafe_write_denied' ||\n e.payload.code === 'policy_denied' ||\n e.payload.code === 'action_denied'),\n )\n return event\n ? {\n failureClass: 'unsafe_integration_write_denied',\n reason: 'integration write was denied by policy or capability scope',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-provider-failure',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n ![\n 'scope_denied',\n 'approval_required',\n 'auth_expired',\n 'connection_not_active',\n 'capability_expired',\n 'unsafe_write_denied',\n 'policy_denied',\n 'action_denied',\n 'manifest_invalid',\n ].includes(String(e.payload.code)),\n )\n return event\n ? {\n failureClass: 'integration_provider_failure',\n reason: 'integration provider invocation failed',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-credentials',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.category === 'credential_or_secret',\n )\n return event\n ? {\n failureClass: 'missing_credentials',\n reason: 'required credential or secret was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-retrieval',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const retrieval = spans.find(\n (s) =>\n s.kind === 'retrieval' && (s.hits.length === 0 || s.hits.every((hit) => hit.score <= 0)),\n )\n return retrieval\n ? {\n failureClass: 'bad_retrieval',\n reason: 'retrieval returned no useful hits for a failed run',\n triggerSpanId: retrieval.spanId,\n }\n : null\n },\n },\n {\n id: 'insufficient-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'insufficient_evidence',\n )\n return event\n ? {\n failureClass: 'insufficient_evidence',\n reason: 'task proceeded with insufficient supporting evidence',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'contradictory-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'contradictory_evidence',\n )\n return event\n ? {\n failureClass: 'contradictory_evidence',\n reason: 'supporting evidence contradicted itself',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n // Budget breach events\n {\n id: 'budget-breach',\n match: ({ events }) => {\n const breach = events.find((e) => e.kind === 'budget_breach')\n return breach\n ? {\n failureClass: 'budget_exceeded',\n reason: `budget breached on ${breach.payload.dimension ?? 'unknown dimension'}`,\n triggerEventId: breach.eventId,\n }\n : null\n },\n },\n // Policy violations\n {\n id: 'policy-violation',\n match: ({ events }) => {\n const e = events.find((x) => x.kind === 'policy_violation')\n return e\n ? {\n failureClass: 'policy_violation',\n reason: 'policy_violation event emitted',\n triggerEventId: e.eventId,\n }\n : null\n },\n },\n // Sandbox non-zero exit code\n {\n id: 'sandbox-failure',\n match: ({ spans }) => {\n const s = spans.find(\n (x) => x.kind === 'sandbox' && typeof x.exitCode === 'number' && x.exitCode !== 0,\n )\n if (!s) return null\n return {\n failureClass: 'sandbox_failure',\n reason: `sandbox exited ${(s as Extract<Span, { kind: 'sandbox' }>).exitCode}`,\n triggerSpanId: s.spanId,\n }\n },\n },\n // Timeout: run aborted by external signal\n {\n id: 'timeout',\n match: ({ run, events }) => {\n if (run.status !== 'aborted') return null\n const hasTimeout = events.some(\n (e) =>\n e.kind === 'error' &&\n String(e.payload.reason ?? '')\n .toLowerCase()\n .includes('timeout'),\n )\n const note = (run.outcome?.notes ?? '').toLowerCase()\n if (hasTimeout || note.includes('timeout') || note.includes('deadline')) {\n return { failureClass: 'timeout', reason: 'timeout signal observed' }\n }\n return null\n },\n },\n // Tool recovery failure: many consecutive tool errors on the same tool\n {\n id: 'tool-recovery-failure',\n match: ({ spans }) => {\n const tools = spans.filter((s) => s.kind === 'tool')\n const byTool = new Map<string, Span[]>()\n for (const t of tools) {\n const name = (t as Extract<Span, { kind: 'tool' }>).toolName\n const arr = byTool.get(name) ?? []\n arr.push(t)\n byTool.set(name, arr)\n }\n for (const [name, arr] of byTool) {\n const errs = arr.filter((s) => s.status === 'error')\n if (errs.length >= 3 && errs.length === arr.length) {\n return {\n failureClass: 'tool_recovery_failure',\n reason: `${errs.length} consecutive errors on tool \"${name}\"`,\n triggerSpanId: errs[errs.length - 1]!.spanId,\n }\n }\n }\n return null\n },\n },\n // Tool selection error: the run failed and agent called zero tools despite having them\n {\n id: 'tool-selection-error',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const hasToolsAvailable = spans.some(\n (s) =>\n s.kind === 'agent' &&\n (s.attributes?.toolsAvailable as number | undefined) !== undefined &&\n (s.attributes?.toolsAvailable as number) > 0,\n )\n const tools = spans.filter((s) => s.kind === 'tool')\n if (hasToolsAvailable && tools.length === 0) {\n return {\n failureClass: 'tool_selection_error',\n reason: 'tools were available but none were called',\n }\n }\n return null\n },\n },\n // Format drift: scored by a judge with dimension='format' below threshold\n {\n id: 'format-drift',\n match: ({ spans }) => {\n const judge = spans.find(\n (s) =>\n s.kind === 'judge' &&\n (s as Extract<Span, { kind: 'judge' }>).dimension === 'format' &&\n (s as Extract<Span, { kind: 'judge' }>).score < 0.5,\n )\n return judge\n ? {\n failureClass: 'format_drift',\n reason: 'format judge scored below 0.5',\n triggerSpanId: judge.spanId,\n }\n : null\n },\n },\n]\n\nfunction hasResolutionStatus(payload: Record<string, unknown>, status: string): boolean {\n if (status === 'missing_connection' && stringArray(payload.missingConnections).length > 0)\n return true\n return resolutionItems(payload).some((item) => item.status === status)\n}\n\nfunction hasMissingScopes(payload: Record<string, unknown>): boolean {\n if (stringArray(payload.missingScopes).length > 0) return true\n return resolutionItems(payload).some(\n (item) => Array.isArray(item.missingScopes) && item.missingScopes.length > 0,\n )\n}\n\nfunction resolutionItems(payload: Record<string, unknown>): Array<Record<string, unknown>> {\n return [\n ...records(payload.missing),\n ...records(payload.optionalMissing),\n ...records(payload.ready),\n ]\n}\n\nfunction records(value: unknown): Array<Record<string, unknown>> {\n if (!Array.isArray(value)) return []\n return value.filter(\n (item): item is Record<string, unknown> =>\n Boolean(item) && typeof item === 'object' && !Array.isArray(item),\n )\n}\n\nfunction stringArray(value: unknown): string[] {\n return Array.isArray(value)\n ? value.filter((item): item is string => typeof item === 'string')\n : []\n}\n\n/** Classify the failure mode of a run using an ordered rule list. */\nexport function classifyFailure(\n ctx: FailureContext,\n rules: FailureRule[] = DEFAULT_RULES,\n): FailureClassification {\n if (ctx.run.outcome?.pass !== false && ctx.run.status === 'completed') {\n return { failureClass: 'success', reason: 'run completed with pass=true (or no explicit fail)' }\n }\n for (const rule of rules) {\n const hit = rule.match(ctx)\n if (hit) return hit\n }\n return { failureClass: 'unknown', reason: 'no rule matched; run failed for unclassified reason' }\n}\n","/**\n * BudgetBreachView — aggregates breach events across the corpus.\n *\n * Answers: which dimensions get hit most often? Which scenarios are\n * underbudgeted? Which variants trigger the most breaches?\n */\n\nimport type { BudgetSpec } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface BudgetBreachFinding {\n runId: string\n scenarioId: string\n variantId?: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n excessRatio: number\n timestamp: number\n}\n\nexport interface BudgetBreachReport {\n findings: BudgetBreachFinding[]\n byDimension: Record<string, number>\n byScenario: Record<string, number>\n byVariant: Record<string, number>\n totalRuns: number\n breachedRunRatio: number\n}\n\nexport async function budgetBreachView(\n store: TraceStore,\n options: { scenarioId?: string; variantId?: string } = {},\n): Promise<BudgetBreachReport> {\n const runs = await store.listRuns({\n scenarioId: options.scenarioId,\n variantId: options.variantId,\n })\n const findings: BudgetBreachFinding[] = []\n const byDimension: Record<string, number> = {}\n const byScenario: Record<string, number> = {}\n const byVariant: Record<string, number> = {}\n\n for (const run of runs) {\n const entries = await store.budget(run.runId)\n for (const e of entries) {\n if (!e.breached) continue\n const excessRatio = e.limit > 0 ? e.consumed / e.limit : Infinity\n findings.push({\n runId: run.runId,\n scenarioId: run.scenarioId,\n variantId: run.variantId,\n dimension: e.dimension,\n limit: e.limit,\n consumed: e.consumed,\n excessRatio,\n timestamp: e.timestamp,\n })\n byDimension[e.dimension] = (byDimension[e.dimension] ?? 0) + 1\n byScenario[run.scenarioId] = (byScenario[run.scenarioId] ?? 0) + 1\n if (run.variantId) byVariant[run.variantId] = (byVariant[run.variantId] ?? 0) + 1\n }\n }\n\n const breachedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n byDimension,\n byScenario,\n byVariant,\n totalRuns: runs.length,\n breachedRunRatio: runs.length > 0 ? breachedRuns.size / runs.length : 0,\n }\n}\n","/**\n * FailureClusterView — groups failed runs by (failureClass, triggerTool,\n * argHash-prefix) so weekly reviews can prioritize the top-N clusters.\n *\n * Each cluster includes: N runs, scenarios affected, representative\n * error message, a proposed mitigation hint (rule → action table).\n */\n\nimport { classifyFailure, DEFAULT_RULES, type FailureRule } from '../failure-taxonomy'\nimport { argHash, hasCapturedToolArgs, toolSpans } from '../trace/query'\nimport type { FailureClass, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface FailureCluster {\n failureClass: FailureClass\n /** Tool name when the trigger was a tool span, else undefined. */\n toolName?: string\n /** First 16 chars of argHash — clusters similar args. */\n argPrefix?: string\n /**\n * Source dimension when the trigger was a judge span (e.g. `'format'`,\n * `'safety'`, `'correctness'`). Lets cross-template aggregators\n * group failures by the dimension that fired without overloading\n * `argPrefix`. Optional — clusters without this field deserialize cleanly.\n */\n dimension?: string\n runCount: number\n scenarioIds: string[]\n exampleError?: string\n exampleRunId: string\n}\n\nexport interface FailureClusterReport {\n clusters: FailureCluster[]\n totalFailures: number\n totalRuns: number\n}\n\nexport async function failureClusterView(\n store: TraceStore,\n options: { rules?: FailureRule[]; minClusterSize?: number } = {},\n): Promise<FailureClusterReport> {\n const rules = options.rules ?? DEFAULT_RULES\n const minSize = options.minClusterSize ?? 1\n const runs = await store.listRuns()\n\n type Key = string\n const clusters = new Map<Key, FailureCluster>()\n let totalFailures = 0\n\n for (const run of runs) {\n if (run.status === 'completed' && run.outcome?.pass !== false) continue\n totalFailures++\n const spans = await store.spans({ runId: run.runId })\n const events = await store.events({ runId: run.runId })\n const cls = classifyFailure({ run, spans, events }, rules)\n\n let toolName: string | undefined\n let argPrefix: string | undefined\n let dimension: string | undefined\n if (cls.triggerSpanId) {\n const trig = spans.find((s) => s.spanId === cls.triggerSpanId)\n if (trig?.kind === 'tool') {\n toolName = trig.toolName\n if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16)\n } else if (trig?.kind === 'judge') {\n dimension = trig.dimension\n }\n }\n // Fallback: look at the last errored tool span\n if (!toolName) {\n const ts = await toolSpans(store, run.runId)\n const errored = ts.filter((t) => t.status === 'error').pop()\n if (errored) {\n toolName = errored.toolName\n if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16)\n }\n }\n // Secondary signal: any judge span on the failed run carries a\n // dimension. Useful when the rule classified by judge score but\n // didn't surface the trigger span (or surfaced a non-judge span).\n if (!dimension) {\n const judge = spans.find((s) => s.kind === 'judge' && typeof s.dimension === 'string')\n if (judge?.kind === 'judge') dimension = judge.dimension\n }\n\n const key = `${cls.failureClass}|${toolName ?? ''}|${argPrefix ?? ''}|${dimension ?? ''}`\n let cluster = clusters.get(key)\n if (!cluster) {\n cluster = {\n failureClass: cls.failureClass,\n toolName,\n argPrefix,\n dimension,\n runCount: 0,\n scenarioIds: [],\n exampleRunId: run.runId,\n exampleError: firstErrorMessage(spans) ?? cls.reason,\n }\n clusters.set(key, cluster)\n }\n cluster.runCount++\n if (!cluster.scenarioIds.includes(run.scenarioId)) cluster.scenarioIds.push(run.scenarioId)\n }\n\n const arr = [...clusters.values()]\n .filter((c) => c.runCount >= minSize)\n .sort((a, b) => b.runCount - a.runCount)\n\n return { clusters: arr, totalFailures, totalRuns: runs.length }\n}\n\nfunction firstErrorMessage(spans: Span[]): string | undefined {\n const errored = spans.find((s) => s.status === 'error')\n return errored?.error\n}\n","/**\n * JudgeAgreementView — pairwise agreement between judges across the\n * corpus, grouped by dimension.\n *\n * Output drives two workflows:\n * - Judge robustness audit: \"does Claude agree with GPT at κ ≥ 0.6?\"\n * - Calibration tracking: κ vs golden human labels over time (by\n * providing a `humanGoldenJudgeId`).\n */\n\nimport { interRaterReliability, pearsonR } from '../statistics'\nimport type { JudgeSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface JudgePair {\n judgeA: string\n judgeB: string\n dimension: string\n /** Number of (targetSpanId, dimension) tuples both judges scored. */\n commonItems: number\n pearson: number\n krippendorff: number\n}\n\nexport interface JudgeAgreementReport {\n pairs: JudgePair[]\n dimensions: string[]\n judgeIds: string[]\n}\n\nexport async function judgeAgreementView(store: TraceStore): Promise<JudgeAgreementReport> {\n const all = (await store.spans({ kind: 'judge' })).filter(\n (s): s is JudgeSpan => s.kind === 'judge',\n )\n if (all.length === 0) return { pairs: [], dimensions: [], judgeIds: [] }\n\n const byDimension = new Map<string, JudgeSpan[]>()\n for (const s of all) {\n const arr = byDimension.get(s.dimension) ?? []\n arr.push(s)\n byDimension.set(s.dimension, arr)\n }\n\n const judgeIds = [...new Set(all.map((s) => s.judgeId))].sort()\n const pairs: JudgePair[] = []\n for (const [dim, spans] of byDimension) {\n const byJudge = new Map<string, Map<string, number>>()\n for (const s of spans) {\n const m = byJudge.get(s.judgeId) ?? new Map<string, number>()\n m.set(s.targetSpanId, s.score)\n byJudge.set(s.judgeId, m)\n }\n const judgesHere = [...byJudge.keys()]\n for (let i = 0; i < judgesHere.length; i++) {\n for (let j = i + 1; j < judgesHere.length; j++) {\n const judgeI = judgesHere[i]!\n const judgeJ = judgesHere[j]!\n const a = byJudge.get(judgeI)!\n const b = byJudge.get(judgeJ)!\n const common: Array<[number, number]> = []\n for (const [target, scoreA] of a) {\n const scoreB = b.get(target)\n if (scoreB !== undefined) common.push([scoreA, scoreB])\n }\n if (common.length < 2) continue\n const judgeScores = common.map(\n ([scoreA, scoreB]) =>\n [\n { judgeName: judgeI, dimension: dim, score: scoreA, reasoning: '' },\n { judgeName: judgeJ, dimension: dim, score: scoreB, reasoning: '' },\n ] as const,\n )\n const k = interRaterReliability(\n judgeScores[0]!.map((_, k2) => judgeScores.map((pair) => pair[k2]!)),\n )\n pairs.push({\n judgeA: judgeI,\n judgeB: judgeJ,\n dimension: dim,\n commonItems: common.length,\n pearson: pearsonR(\n common.map((c) => c[0]),\n common.map((c) => c[1]),\n ),\n krippendorff: k,\n })\n }\n }\n }\n\n return {\n pairs: pairs.sort((a, b) => b.commonItems - a.commonItems),\n dimensions: [...byDimension.keys()].sort(),\n judgeIds,\n }\n}\n","/**\n * Tool-use metrics — derived purely from trace data.\n *\n * No scoring assumptions: consumers supply optional ground-truth tool\n * selections per turn + optional \"information used downstream\" signals.\n * Without those, we still compute descriptive metrics (error rate,\n * retry rate, duplicate-call rate) that are useful on their own.\n */\n\nimport { argHash, groupBy, hasCapturedToolArgs, toolSpans } from './trace/query'\nimport type { Span } from './trace/schema'\nimport type { TraceStore } from './trace/store'\n\nexport interface ToolUseMetrics {\n runId: string\n totalCalls: number\n /** Calls whose arguments were captured and can be compared for duplication. */\n callsWithCapturedArgs: number\n byTool: Record<string, ToolStats>\n errorRate: number\n /** Ratio of captured-argument calls already seen with the same tool name and arguments. */\n duplicateRate: number\n /** Ratio of error calls followed by ≥1 retry on same tool. */\n retryRate: number\n /** Optional: of the calls agent made, fraction the evaluator marked as \"correct selection\". */\n selectionAccuracy?: number\n}\n\nexport interface ToolStats {\n calls: number\n callsWithCapturedArgs: number\n errors: number\n avgLatencyMs: number\n duplicates: number\n}\n\nexport interface ToolUseOptions {\n /** Map of spanId → whether the evaluator judged the tool selection correct. Optional. */\n selectionLabels?: Record<string, boolean>\n}\n\nexport async function computeToolUseMetrics(\n store: TraceStore,\n runId: string,\n options: ToolUseOptions = {},\n): Promise<ToolUseMetrics> {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n return {\n runId,\n totalCalls: 0,\n callsWithCapturedArgs: 0,\n byTool: {},\n errorRate: 0,\n duplicateRate: 0,\n retryRate: 0,\n }\n }\n\n const byTool: Record<string, ToolStats> = {}\n let totalErrors = 0\n let totalDuplicates = 0\n let callsWithCapturedArgs = 0\n const sortedTools = [...tools].sort((a, b) => a.startedAt - b.startedAt)\n const seenSignatures = new Set<string>()\n\n // duplicate detection + per-tool aggregation\n for (const t of sortedTools) {\n byTool[t.toolName] ??= {\n calls: 0,\n callsWithCapturedArgs: 0,\n errors: 0,\n avgLatencyMs: 0,\n duplicates: 0,\n }\n const stat = byTool[t.toolName]!\n stat.calls += 1\n if (t.status === 'error') {\n stat.errors += 1\n totalErrors += 1\n }\n if (typeof t.latencyMs === 'number') stat.avgLatencyMs += t.latencyMs\n if (hasCapturedToolArgs(t)) {\n callsWithCapturedArgs += 1\n stat.callsWithCapturedArgs += 1\n const sig = `${t.toolName}|${argHash(t.args)}`\n if (seenSignatures.has(sig)) {\n stat.duplicates += 1\n totalDuplicates += 1\n }\n seenSignatures.add(sig)\n }\n }\n\n for (const stat of Object.values(byTool)) {\n stat.avgLatencyMs = stat.calls > 0 ? stat.avgLatencyMs / stat.calls : 0\n }\n\n // retry detection: per-tool chronological adjacency where error → next same-tool call\n let retryOpportunities = 0\n let retriesFollowed = 0\n for (const [, arr] of groupBy(sortedTools, (t) => t.toolName)) {\n for (let i = 0; i < arr.length; i++) {\n if (arr[i]!.status !== 'error') continue\n retryOpportunities += 1\n if (arr[i + 1]) retriesFollowed += 1\n }\n }\n const retryRate = retryOpportunities > 0 ? retriesFollowed / retryOpportunities : 0\n\n let selectionAccuracy: number | undefined\n if (options.selectionLabels) {\n const labeled = sortedTools.filter((t) => t.spanId in options.selectionLabels!)\n if (labeled.length > 0) {\n selectionAccuracy =\n labeled.filter((t) => options.selectionLabels![t.spanId]).length / labeled.length\n }\n }\n\n return {\n runId,\n totalCalls: sortedTools.length,\n callsWithCapturedArgs,\n byTool,\n errorRate: totalErrors / sortedTools.length,\n duplicateRate: callsWithCapturedArgs > 0 ? totalDuplicates / callsWithCapturedArgs : 0,\n retryRate,\n selectionAccuracy,\n }\n}\n\nexport type { Span }\n","/**\n * ToolWasteView — fraction of tool calls whose results weren't used\n * downstream. Without a \"used\" signal we fall back to structural\n * proxies: error calls, duplicate calls, and tool calls followed by\n * zero subsequent LLM spans are all considered waste.\n *\n * Consumers can pass a `usageOracle` that inspects a tool span and\n * returns true iff the tool's result appears in a later LLM message,\n * artifact, or state mutation — that's the canonical definition; the\n * default heuristic is a reasonable fallback.\n */\n\nimport { computeToolUseMetrics } from '../tool-use-metrics'\nimport { llmSpans, toolSpans } from '../trace/query'\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface ToolWasteFinding {\n runId: string\n wastedCalls: number\n totalCalls: number\n wasteRate: number\n}\n\nexport interface ToolWasteReport {\n byRun: ToolWasteFinding[]\n overallWasteRate: number\n}\n\nexport interface ToolWasteOptions {\n runId?: string\n usageOracle?: (tool: ToolSpan, later: { llm: Awaited<ReturnType<typeof llmSpans>> }) => boolean\n}\n\nexport async function toolWasteView(\n store: TraceStore,\n options: ToolWasteOptions = {},\n): Promise<ToolWasteReport> {\n const runs = options.runId ? [options.runId] : (await store.listRuns()).map((r) => r.runId)\n\n const byRun: ToolWasteFinding[] = []\n let totalCalls = 0\n let totalWasted = 0\n for (const runId of runs) {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n byRun.push({ runId, wastedCalls: 0, totalCalls: 0, wasteRate: 0 })\n continue\n }\n const llms = await llmSpans(store, runId)\n // Sort LLM spans once by start time, then build a suffix index of the\n // concatenated message text. `suffixText[i]` is the haystack of every\n // string message content in spans[i..]. Per tool we binary-search the\n // first span started strictly after the tool, then test that one suffix\n // — turning the per-tool O(llms × messages × content) scan into a single\n // O(log llms) lookup over precomputed text.\n const sortedLlm = [...llms].sort((a, b) => a.startedAt - b.startedAt)\n const startTimes = sortedLlm.map((l) => l.startedAt)\n const suffixText = buildSuffixText(sortedLlm)\n let wasted = 0\n for (const t of tools) {\n if (t.status === 'error') {\n wasted++\n continue\n }\n // First LLM span started strictly after this tool (upper-bound search).\n const cutoff = upperBound(startTimes, t.startedAt)\n if (options.usageOracle) {\n if (!options.usageOracle(t, { llm: sortedLlm.slice(cutoff) })) wasted++\n } else {\n // Default heuristic: a tool whose result is NOT mentioned in any\n // later LLM input message is likely wasted. An empty/null result has\n // no payload to propagate downstream — there is nothing to find in a\n // later message, so it is not evidence of waste; skip it.\n const resultStr = stringify(t.result)\n if (resultStr === '') continue\n const haystack = suffixText[cutoff] ?? ''\n const used = haystack.includes(resultStr.slice(0, 120))\n if (!used) wasted++\n }\n }\n const wasteRate = wasted / tools.length\n byRun.push({ runId, wastedCalls: wasted, totalCalls: tools.length, wasteRate })\n totalCalls += tools.length\n totalWasted += wasted\n }\n return { byRun, overallWasteRate: totalCalls > 0 ? totalWasted / totalCalls : 0 }\n}\n\n/**\n * Build per-position suffix haystacks: result[i] is the concatenation of every\n * string message content in spans[i..end]. Built back-to-front so each entry\n * reuses the next one — O(total message text) rather than O(spans²).\n */\nfunction buildSuffixText(spans: LlmSpan[]): string[] {\n const result = new Array<string>(spans.length + 1)\n result[spans.length] = ''\n for (let i = spans.length - 1; i >= 0; i--) {\n const own = spans[i]!.messages.map((m) =>\n typeof m.content === 'string' ? m.content : '',\n ).join('\\n')\n result[i] = `${own}\\n${result[i + 1]}`\n }\n return result\n}\n\n/** Index of the first element strictly greater than `target` in a sorted array. */\nfunction upperBound(sorted: number[], target: number): number {\n let lo = 0\n let hi = sorted.length\n while (lo < hi) {\n const mid = (lo + hi) >>> 1\n if (sorted[mid]! <= target) lo = mid + 1\n else hi = mid\n }\n return lo\n}\n\nfunction stringify(v: unknown): string {\n if (v === null || v === undefined) return ''\n if (typeof v === 'string') return v\n try {\n return JSON.stringify(v)\n } catch {\n return String(v)\n }\n}\n\n// Re-export for convenience in consumers that want both descriptive and usage metrics.\nexport { computeToolUseMetrics }\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAyBA,SAAgB,sBAAsB,aAAqC;CACzE,IAAI,YAAY,SAAS,GAAG,OAAO;CAGnC,MAAM,+BAAe,IAAI,IAAwB;CACjD,KAAK,IAAI,aAAa,GAAG,aAAa,YAAY,QAAQ,cACxD,KAAK,MAAM,KAAK,YAAY,aAAc;EACxC,IAAI,UAAU,aAAa,IAAI,EAAE,SAAS;EAC1C,IAAI,YAAY,KAAA,GAAW;GACzB,UAAU,MAAM,KAAK,EAAE,QAAQ,YAAY,OAAO,SAAS,CAAC,CAAa;GACzE,aAAa,IAAI,EAAE,WAAW,OAAO;EACvC;EACA,QAAQ,WAAW,CAAE,KAAK,EAAE,KAAK;CACnC;CAGF,MAAM,YAAsB,CAAC;CAC7B,MAAM,YAAsB,CAAC;CAE7B,KAAK,MAAM,CAAC,WAAW,YAAY,cAAc;EAC/C,MAAM,UAAU,QAAQ,QAAQ,WAAW,OAAO,SAAS,CAAC;EAC5D,IAAI,QAAQ,SAAS,GAAG;EACxB,MAAM,YAAY,QAAQ,EAAE,CAAE;EAC9B,IAAI,QAAQ,MAAM,WAAW,OAAO,WAAW,SAAS,GACtD,MAAM,IAAI,gBACR,qCAAqC,UAAU,yBAC1C,QAAQ,KAAK,WAAW,OAAO,MAAM,CAAC,CAAC,KAAK,GAAG,EAAE,kCACxD;EAEF,KAAK,IAAI,OAAO,GAAG,OAAO,WAAW,QAAQ;GAC3C,MAAM,UAAU,QAAQ,KAAK,WAAW,OAAO,KAAM;GACrD,KAAK,MAAM,KAAK,SAAS,UAAU,KAAK,CAAC;GACzC,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAClC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KACtC,UAAU,MAAM,QAAQ,KAAM,QAAQ,OAAQ,CAAC;EAGrD;CACF;CAEA,IAAI,UAAU,WAAW,KAAK,UAAU,SAAS,GAAG,OAAO;CAE3D,MAAM,uBAAuB,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,UAAU;CAG9E,IAAI,uBAAuB;CAC3B,IAAI,gBAAgB;CACpB,KAAK,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KACpC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KAAK;EAC7C,yBAAyB,UAAU,KAAM,UAAU,OAAQ;EAC3D;CACF;CAEF,uBAAuB,gBAAgB,IAAI,uBAAuB,gBAAgB;CAElF,IAAI,yBAAyB,GAAG,OAAO;CACvC,OAAO,IAAI,uBAAuB;AACpC;;;;;;;;;;;;;;;;;AAgFA,SAAgB,0BACd,SACA,OAA+B,CAAC,GACT;CACvB,IAAI,QAAQ,WAAW,GACrB,MAAM,IAAI,gBAAgB,sDAAsD;CAGlF,MAAM,6BAAa,IAAI,IAAY;CACnC,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,uBAAO,IAAI,IAA8C;CAE/D,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,yDAAyD,EAAE,OAAO,UAAU,EAAE,UAAU,QAAQ,EAAE,UAAU,EAC9G;EAEF,WAAW,IAAI,EAAE,SAAS;EAC1B,SAAS,IAAI,EAAE,SAAS;EACxB,MAAM,UAAU,KAAK,IAAI,EAAE,SAAS,qBAAK,IAAI,IAAiC;EAC9E,MAAM,SAAS,QAAQ,IAAI,EAAE,SAAS,qBAAK,IAAI,IAAoB;EACnE,IAAI,OAAO,IAAI,EAAE,MAAM,GACrB,MAAM,IAAI,gBACR,yDAAyD,EAAE,OAAO,UAAU,EAAE,UAAU,QAAQ,EAAE,UAAU,EAC9G;EAEF,OAAO,IAAI,EAAE,QAAQ,EAAE,KAAK;EAC5B,QAAQ,IAAI,EAAE,WAAW,MAAM;EAC/B,KAAK,IAAI,EAAE,WAAW,OAAO;CAC/B;CAEA,MAAM,aAAa,KAAK,cAAc,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK;CACzD,KAAK,MAAM,KAAK,YACd,IAAI,CAAC,SAAS,IAAI,CAAC,GACjB,MAAM,IAAI,gBACR,yCAAyC,EAAE,wCAC7C;CAGJ,MAAM,eAAe,KAAK,SAAS,CAAC,GAAG,KAAK,MAAM,IAAI,CAAC,GAAG,UAAU,CAAC,CAAC,KAAK;CAC3E,KAAK,MAAM,KAAK,cACd,IAAI,CAAC,WAAW,IAAI,CAAC,GACnB,MAAM,IAAI,gBACR,qCAAqC,EAAE,wCACzC;CAGJ,IAAI,aAAa,SAAS,GACxB,MAAM,IAAI,gBACR,kDAAkD,aAAa,QACjE;CAGF,MAAM,eAA8C,CAAC;CACrD,MAAM,OAAiB,CAAC;CACxB,MAAM,SAAmB,CAAC;CAE1B,KAAK,MAAM,OAAO,YAAY;EAC5B,MAAM,UAAU,KAAK,IAAI,GAAG;EAE5B,MAAM,kBAA0C,CAAC;EACjD,KAAK,MAAM,KAAK,cAEd,gBAAgB,KADN,QAAQ,IAAI,CACD,CAAC,EAAE,QAAQ;EAElC,MAAM,cAAc,aAAa,QAAQ,MAAM,gBAAgB,OAAO,CAAC;EACvE,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,gBACR,yCAAyC,IAAI,gCAAgC,YAAY,KAAK,IAAI,EAAE,YAAY,KAAK,UAAU,eAAe,EAAE,EAClJ;EAIF,IAAI,cAAkC;EACtC,KAAK,MAAM,KAAK,cAAc;GAC5B,MAAM,MAAM,IAAI,IAAI,QAAQ,IAAI,CAAC,CAAC,CAAE,KAAK,CAAC;GAC1C,IAAI,gBAAgB,MAClB,cAAc;QAGd,cAAc,IAAI,IAAI,CAAC,GAAGA,WAAI,CAAC,CAAC,QAAQ,MAAM,IAAI,IAAI,CAAC,CAAC,CAAC;EAE7D;EACA,MAAM,cAAc,CAAC,GAAI,+BAAe,IAAI,IAAY,CAAE,CAAC,CAAC,KAAK;EACjE,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,gBACR,yCAAyC,IAAI,QAAQ,YAAY,OAAO,wBAAwB,aAAa,OAAO,kBACtH;EAMF,MAAM,YAAY,oBAHS,YAAY,KAAK,WAC1C,aAAa,KAAK,MAAM,QAAQ,IAAI,CAAC,CAAC,CAAE,IAAI,MAAM,CAAE,CAEX,GAAG,IAAI;EAClD,aAAa,KAAK;GAChB,GAAG;GACH,WAAW;GACX,SAAS;GACT,UAAU,CAAC,GAAG,YAAY;EAC5B,CAAC;EACD,IAAI,OAAO,SAAS,UAAU,GAAG,GAAG,KAAK,KAAK,UAAU,GAAG;EAC3D,IAAI,OAAO,SAAS,UAAU,aAAa,GAAG,OAAO,KAAK,UAAU,aAAa;CACnF;CAEA,MAAM,QAAQ,OACZ,GAAG,WAAW,IAAI,MAAa,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;CACpE,OAAO;EACL;EACA,YAAY,KAAK,IAAI;EACrB,sBAAsB,KAAK,MAAM;EACjC,YAAY;EACZ,UAAU;CACZ;AACF;;;;;;;;;AAUA,SAAgB,yCACd,aACA,OAA+B,CAAC,GACT;CACvB,MAAM,UAA+B,CAAC;CACtC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,EAAE,QAAQ,YAAY,aAAa;EAC5C,IAAI,KAAK,IAAI,MAAM,GACjB,MAAM,IAAI,gBACR,+DAA+D,OAAO,EACxE;EAEF,KAAK,IAAI,MAAM;EACf,KAAK,MAAM,KAAK,QACd,QAAQ,KAAK;GACX;GACA,WAAW,EAAE;GACb,WAAW,EAAE;GACb,OAAO,EAAE;EACX,CAAC;CAEL;CACA,OAAO,0BAA0B,SAAS,IAAI;AAChD;;;AC9QA,MAAa,gBAA+B;CAE1C;EACE,IAAI;EACJ,QAAQ,EAAE,UAAU;GAClB,MAAM,KAAK,IAAI,SAAS;GACxB,IAAI,MAAM,OAAO,WACf,OAAO;IAAE,cAAc;IAAI,QAAQ;GAAsC;GAC3E,OAAO;EACT;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,sBACnB,EAAE,QAAQ,WAAW,KACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,oCAAoC,EAAE,QAAQ,UAAU,SAC1E,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,mBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mCACnB,oBAAoB,EAAE,SAAS,oBAAoB,CACvD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,mCAAmC,iBAAiB,EAAE,OAAO,KAC/E,EAAE,QAAQ,SAAS,+BAA+B,EAAE,QAAQ,SAAS,eAC5E;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,wBAAwB,EAAE,QAAQ,WAAW,uBAC/D,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,uBACrB,EAAE,QAAQ,SAAS,gCACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,kBAClB,EAAE,QAAQ,SAAS,2BACnB,EAAE,QAAQ,SAAS,wBACnB,EAAE,QAAQ,WAAW,UAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,yBAClB,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,SAAS,gBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,+BACnB,CAAC;IACC;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;GACF,CAAC,CAAC,SAAS,OAAO,EAAE,QAAQ,IAAI,CAAC,CACrC;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,aAAa,sBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,YAAY,MAAM,MACrB,MACC,EAAE,SAAS,gBAAgB,EAAE,KAAK,WAAW,KAAK,EAAE,KAAK,OAAO,QAAQ,IAAI,SAAS,CAAC,EAC1F;GACA,OAAO,YACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,UAAU;GAC3B,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,uBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,wBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,SAAS,eAAe;GAC5D,OAAO,SACH;IACE,cAAc;IACd,QAAQ,sBAAsB,OAAO,QAAQ,aAAa;IAC1D,gBAAgB,OAAO;GACzB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,IAAI,OAAO,MAAM,MAAM,EAAE,SAAS,kBAAkB;GAC1D,OAAO,IACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,EAAE;GACpB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,IAAI,MAAM,MACb,MAAM,EAAE,SAAS,aAAa,OAAO,EAAE,aAAa,YAAY,EAAE,aAAa,CAClF;GACA,IAAI,CAAC,GAAG,OAAO;GACf,OAAO;IACL,cAAc;IACd,QAAQ,kBAAmB,EAAyC;IACpE,eAAe,EAAE;GACnB;EACF;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,aAAa;GAC1B,IAAI,IAAI,WAAW,WAAW,OAAO;GACrC,MAAM,aAAa,OAAO,MACvB,MACC,EAAE,SAAS,WACX,OAAO,EAAE,QAAQ,UAAU,EAAE,CAAC,CAC3B,YAAY,CAAC,CACb,SAAS,SAAS,CACzB;GACA,MAAM,QAAQ,IAAI,SAAS,SAAS,GAAA,CAAI,YAAY;GACpD,IAAI,cAAc,KAAK,SAAS,SAAS,KAAK,KAAK,SAAS,UAAU,GACpE,OAAO;IAAE,cAAc;IAAW,QAAQ;GAA0B;GAEtE,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,MAAM,yBAAS,IAAI,IAAoB;GACvC,KAAK,MAAM,KAAK,OAAO;IACrB,MAAM,OAAQ,EAAsC;IACpD,MAAM,MAAM,OAAO,IAAI,IAAI,KAAK,CAAC;IACjC,IAAI,KAAK,CAAC;IACV,OAAO,IAAI,MAAM,GAAG;GACtB;GACA,KAAK,MAAM,CAAC,MAAM,QAAQ,QAAQ;IAChC,MAAM,OAAO,IAAI,QAAQ,MAAM,EAAE,WAAW,OAAO;IACnD,IAAI,KAAK,UAAU,KAAK,KAAK,WAAW,IAAI,QAC1C,OAAO;KACL,cAAc;KACd,QAAQ,GAAG,KAAK,OAAO,+BAA+B,KAAK;KAC3D,eAAe,KAAK,KAAK,SAAS,EAAE,CAAE;IACxC;GAEJ;GACA,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,oBAAoB,MAAM,MAC7B,MACC,EAAE,SAAS,WACV,EAAE,YAAY,mBAA0C,KAAA,KACxD,EAAE,YAAY,iBAA4B,CAC/C;GACA,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,IAAI,qBAAqB,MAAM,WAAW,GACxC,OAAO;IACL,cAAc;IACd,QAAQ;GACV;GAEF,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,MACjB,MACC,EAAE,SAAS,WACV,EAAuC,cAAc,YACrD,EAAuC,QAAQ,EACpD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,MAAM;GACvB,IACA;EACN;CACF;AACF;AAEA,SAAS,oBAAoB,SAAkC,QAAyB;CACtF,IAAI,WAAW,wBAAwB,YAAY,QAAQ,kBAAkB,CAAC,CAAC,SAAS,GACtF,OAAO;CACT,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,MAAM;AACvE;AAEA,SAAS,iBAAiB,SAA2C;CACnE,IAAI,YAAY,QAAQ,aAAa,CAAC,CAAC,SAAS,GAAG,OAAO;CAC1D,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAC7B,SAAS,MAAM,QAAQ,KAAK,aAAa,KAAK,KAAK,cAAc,SAAS,CAC7E;AACF;AAEA,SAAS,gBAAgB,SAAkE;CACzF,OAAO;EACL,GAAG,QAAQ,QAAQ,OAAO;EAC1B,GAAG,QAAQ,QAAQ,eAAe;EAClC,GAAG,QAAQ,QAAQ,KAAK;CAC1B;AACF;AAEA,SAAS,QAAQ,OAAgD;CAC/D,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO,CAAC;CACnC,OAAO,MAAM,QACV,SACC,QAAQ,IAAI,KAAK,OAAO,SAAS,YAAY,CAAC,MAAM,QAAQ,IAAI,CACpE;AACF;AAEA,SAAS,YAAY,OAA0B;CAC7C,OAAO,MAAM,QAAQ,KAAK,IACtB,MAAM,QAAQ,SAAyB,OAAO,SAAS,QAAQ,IAC/D,CAAC;AACP;;AAGA,SAAgB,gBACd,KACA,QAAuB,eACA;CACvB,IAAI,IAAI,IAAI,SAAS,SAAS,SAAS,IAAI,IAAI,WAAW,aACxD,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAqD;CAEjG,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,MAAM,KAAK,MAAM,GAAG;EAC1B,IAAI,KAAK,OAAO;CAClB;CACA,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAsD;AAClG;;;AC/aA,eAAsB,iBACpB,OACA,UAAuD,CAAC,GAC3B;CAC7B,MAAM,OAAO,MAAM,MAAM,SAAS;EAChC,YAAY,QAAQ;EACpB,WAAW,QAAQ;CACrB,CAAC;CACD,MAAM,WAAkC,CAAC;CACzC,MAAM,cAAsC,CAAC;CAC7C,MAAM,aAAqC,CAAC;CAC5C,MAAM,YAAoC,CAAC;CAE3C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,UAAU,MAAM,MAAM,OAAO,IAAI,KAAK;EAC5C,KAAK,MAAM,KAAK,SAAS;GACvB,IAAI,CAAC,EAAE,UAAU;GACjB,MAAM,cAAc,EAAE,QAAQ,IAAI,EAAE,WAAW,EAAE,QAAQ;GACzD,SAAS,KAAK;IACZ,OAAO,IAAI;IACX,YAAY,IAAI;IAChB,WAAW,IAAI;IACf,WAAW,EAAE;IACb,OAAO,EAAE;IACT,UAAU,EAAE;IACZ;IACA,WAAW,EAAE;GACf,CAAC;GACD,YAAY,EAAE,cAAc,YAAY,EAAE,cAAc,KAAK;GAC7D,WAAW,IAAI,eAAe,WAAW,IAAI,eAAe,KAAK;GACjE,IAAI,IAAI,WAAW,UAAU,IAAI,cAAc,UAAU,IAAI,cAAc,KAAK;EAClF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA;EACA;EACA;EACA,WAAW,KAAK;EAChB,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;CACxE;AACF;;;;;;;;;;ACnCA,eAAsB,mBACpB,OACA,UAA8D,CAAC,GAChC;CAC/B,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,OAAO,MAAM,MAAM,SAAS;CAGlC,MAAM,2BAAW,IAAI,IAAyB;CAC9C,IAAI,gBAAgB;CAEpB,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,WAAW,eAAe,IAAI,SAAS,SAAS,OAAO;EAC/D;EACA,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,OAAO,IAAI,MAAM,CAAC;EAEpD,MAAM,MAAM,gBAAgB;GAAE;GAAK;GAAO,QAAA,MADrB,MAAM,OAAO,EAAE,OAAO,IAAI,MAAM,CAAC;EACL,GAAG,KAAK;EAEzD,IAAI;EACJ,IAAI;EACJ,IAAI;EACJ,IAAI,IAAI,eAAe;GACrB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,WAAW,IAAI,aAAa;GAC7D,IAAI,MAAM,SAAS,QAAQ;IACzB,WAAW,KAAK;IAChB,IAAI,oBAAoB,IAAI,GAAG,YAAY,QAAQ,KAAK,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GAC3E,OAAO,IAAI,MAAM,SAAS,SACxB,YAAY,KAAK;EAErB;EAEA,IAAI,CAAC,UAAU;GAEb,MAAM,WAAU,MADC,UAAU,OAAO,IAAI,KAAK,EAAA,CACxB,QAAQ,MAAM,EAAE,WAAW,OAAO,CAAC,CAAC,IAAI;GAC3D,IAAI,SAAS;IACX,WAAW,QAAQ;IACnB,IAAI,oBAAoB,OAAO,GAAG,YAAY,QAAQ,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GACjF;EACF;EAIA,IAAI,CAAC,WAAW;GACd,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,SAAS,WAAW,OAAO,EAAE,cAAc,QAAQ;GACrF,IAAI,OAAO,SAAS,SAAS,YAAY,MAAM;EACjD;EAEA,MAAM,MAAM,GAAG,IAAI,aAAa,GAAG,YAAY,GAAG,GAAG,aAAa,GAAG,GAAG,aAAa;EACrF,IAAI,UAAU,SAAS,IAAI,GAAG;EAC9B,IAAI,CAAC,SAAS;GACZ,UAAU;IACR,cAAc,IAAI;IAClB;IACA;IACA;IACA,UAAU;IACV,aAAa,CAAC;IACd,cAAc,IAAI;IAClB,cAAc,kBAAkB,KAAK,KAAK,IAAI;GAChD;GACA,SAAS,IAAI,KAAK,OAAO;EAC3B;EACA,QAAQ;EACR,IAAI,CAAC,QAAQ,YAAY,SAAS,IAAI,UAAU,GAAG,QAAQ,YAAY,KAAK,IAAI,UAAU;CAC5F;CAMA,OAAO;EAAE,UAJG,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAC/B,QAAQ,MAAM,EAAE,YAAY,OAAO,CAAC,CACpC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAEZ;EAAG;EAAe,WAAW,KAAK;CAAO;AAChE;AAEA,SAAS,kBAAkB,OAAmC;CAE5D,OADgB,MAAM,MAAM,MAAM,EAAE,WAAW,OAClC,CAAC,EAAE;AAClB;;;;;;;;;;;;ACrFA,eAAsB,mBAAmB,OAAkD;CACzF,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,QAAQ,CAAC,EAAA,CAAG,QAChD,MAAsB,EAAE,SAAS,OACpC;CACA,IAAI,IAAI,WAAW,GAAG,OAAO;EAAE,OAAO,CAAC;EAAG,YAAY,CAAC;EAAG,UAAU,CAAC;CAAE;CAEvE,MAAM,8BAAc,IAAI,IAAyB;CACjD,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,MAAM,YAAY,IAAI,EAAE,SAAS,KAAK,CAAC;EAC7C,IAAI,KAAK,CAAC;EACV,YAAY,IAAI,EAAE,WAAW,GAAG;CAClC;CAEA,MAAM,WAAW,CAAC,GAAG,IAAI,IAAI,IAAI,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,KAAK;CAC9D,MAAM,QAAqB,CAAC;CAC5B,KAAK,MAAM,CAAC,KAAK,UAAU,aAAa;EACtC,MAAM,0BAAU,IAAI,IAAiC;EACrD,KAAK,MAAM,KAAK,OAAO;GACrB,MAAM,IAAI,QAAQ,IAAI,EAAE,OAAO,qBAAK,IAAI,IAAoB;GAC5D,EAAE,IAAI,EAAE,cAAc,EAAE,KAAK;GAC7B,QAAQ,IAAI,EAAE,SAAS,CAAC;EAC1B;EACA,MAAM,aAAa,CAAC,GAAG,QAAQ,KAAK,CAAC;EACrC,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KACrC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KAAK;GAC9C,MAAM,SAAS,WAAW;GAC1B,MAAM,SAAS,WAAW;GAC1B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,SAAkC,CAAC;GACzC,KAAK,MAAM,CAAC,QAAQ,WAAW,GAAG;IAChC,MAAM,SAAS,EAAE,IAAI,MAAM;IAC3B,IAAI,WAAW,KAAA,GAAW,OAAO,KAAK,CAAC,QAAQ,MAAM,CAAC;GACxD;GACA,IAAI,OAAO,SAAS,GAAG;GACvB,MAAM,cAAc,OAAO,KACxB,CAAC,QAAQ,YACR,CACE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,GAClE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,CACpE,CACJ;GACA,MAAM,IAAI,sBACR,YAAY,EAAE,CAAE,KAAK,GAAG,OAAO,YAAY,KAAK,SAAS,KAAK,GAAI,CAAC,CACrE;GACA,MAAM,KAAK;IACT,QAAQ;IACR,QAAQ;IACR,WAAW;IACX,aAAa,OAAO;IACpB,SAAS,SACP,OAAO,KAAK,MAAM,EAAE,EAAE,GACtB,OAAO,KAAK,MAAM,EAAE,EAAE,CACxB;IACA,cAAc;GAChB,CAAC;EACH;CAEJ;CAEA,OAAO;EACL,OAAO,MAAM,MAAM,GAAG,MAAM,EAAE,cAAc,EAAE,WAAW;EACzD,YAAY,CAAC,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK;EACzC;CACF;AACF;;;;;;;;;;;ACtDA,eAAsB,sBACpB,OACA,OACA,UAA0B,CAAC,GACF;CACzB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;CAC1C,IAAI,MAAM,WAAW,GACnB,OAAO;EACL;EACA,YAAY;EACZ,uBAAuB;EACvB,QAAQ,CAAC;EACT,WAAW;EACX,eAAe;EACf,WAAW;CACb;CAGF,MAAM,SAAoC,CAAC;CAC3C,IAAI,cAAc;CAClB,IAAI,kBAAkB;CACtB,IAAI,wBAAwB;CAC5B,MAAM,cAAc,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACvE,MAAM,iCAAiB,IAAI,IAAY;CAGvC,KAAK,MAAM,KAAK,aAAa;EAC3B,OAAO,EAAE,cAAc;GACrB,OAAO;GACP,uBAAuB;GACvB,QAAQ;GACR,cAAc;GACd,YAAY;EACd;EACA,MAAM,OAAO,OAAO,EAAE;EACtB,KAAK,SAAS;EACd,IAAI,EAAE,WAAW,SAAS;GACxB,KAAK,UAAU;GACf,eAAe;EACjB;EACA,IAAI,OAAO,EAAE,cAAc,UAAU,KAAK,gBAAgB,EAAE;EAC5D,IAAI,oBAAoB,CAAC,GAAG;GAC1B,yBAAyB;GACzB,KAAK,yBAAyB;GAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,GAAG,QAAQ,EAAE,IAAI;GAC3C,IAAI,eAAe,IAAI,GAAG,GAAG;IAC3B,KAAK,cAAc;IACnB,mBAAmB;GACrB;GACA,eAAe,IAAI,GAAG;EACxB;CACF;CAEA,KAAK,MAAM,QAAQ,OAAO,OAAO,MAAM,GACrC,KAAK,eAAe,KAAK,QAAQ,IAAI,KAAK,eAAe,KAAK,QAAQ;CAIxE,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,KAAK,MAAM,GAAG,QAAQ,QAAQ,cAAc,MAAM,EAAE,QAAQ,GAC1D,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,IAAI,IAAI,EAAE,CAAE,WAAW,SAAS;EAChC,sBAAsB;EACtB,IAAI,IAAI,IAAI,IAAI,mBAAmB;CACrC;CAEF,MAAM,YAAY,qBAAqB,IAAI,kBAAkB,qBAAqB;CAElF,IAAI;CACJ,IAAI,QAAQ,iBAAiB;EAC3B,MAAM,UAAU,YAAY,QAAQ,MAAM,EAAE,UAAU,QAAQ,eAAgB;EAC9E,IAAI,QAAQ,SAAS,GACnB,oBACE,QAAQ,QAAQ,MAAM,QAAQ,gBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS,QAAQ;CAEjF;CAEA,OAAO;EACL;EACA,YAAY,YAAY;EACxB;EACA;EACA,WAAW,cAAc,YAAY;EACrC,eAAe,wBAAwB,IAAI,kBAAkB,wBAAwB;EACrF;EACA;CACF;AACF;;;;;;;;;;;;;;AC/FA,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,OAAO,QAAQ,QAAQ,CAAC,QAAQ,KAAK,KAAK,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,MAAM,EAAE,KAAK;CAE1F,MAAM,QAA4B,CAAC;CACnC,IAAI,aAAa;CACjB,IAAI,cAAc;CAClB,KAAK,MAAM,SAAS,MAAM;EACxB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;EAC1C,IAAI,MAAM,WAAW,GAAG;GACtB,MAAM,KAAK;IAAE;IAAO,aAAa;IAAG,YAAY;IAAG,WAAW;GAAE,CAAC;GACjE;EACF;EAQA,MAAM,YAAY,CAAC,GAAG,MAPH,SAAS,OAAO,KAAK,CAOd,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EACpE,MAAM,aAAa,UAAU,KAAK,MAAM,EAAE,SAAS;EACnD,MAAM,aAAa,gBAAgB,SAAS;EAC5C,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,OAAO;GACrB,IAAI,EAAE,WAAW,SAAS;IACxB;IACA;GACF;GAEA,MAAM,SAAS,WAAW,YAAY,EAAE,SAAS;GACjD,IAAI,QAAQ,aACN;QAAA,CAAC,QAAQ,YAAY,GAAG,EAAE,KAAK,UAAU,MAAM,MAAM,EAAE,CAAC,GAAG;GAAA,OAC1D;IAKL,MAAM,YAAY,UAAU,EAAE,MAAM;IACpC,IAAI,cAAc,IAAI;IAGtB,IAAI,EAFa,WAAW,WAAW,GAAA,CACjB,SAAS,UAAU,MAAM,GAAG,GAAG,CAC7C,GAAG;GACb;EACF;EACA,MAAM,YAAY,SAAS,MAAM;EACjC,MAAM,KAAK;GAAE;GAAO,aAAa;GAAQ,YAAY,MAAM;GAAQ;EAAU,CAAC;EAC9E,cAAc,MAAM;EACpB,eAAe;CACjB;CACA,OAAO;EAAE;EAAO,kBAAkB,aAAa,IAAI,cAAc,aAAa;CAAE;AAClF;;;;;;AAOA,SAAS,gBAAgB,OAA4B;CACnD,MAAM,SAAS,IAAI,MAAc,MAAM,SAAS,CAAC;CACjD,OAAO,MAAM,UAAU;CACvB,KAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAIrC,OAAO,KAAK,GAHA,MAAM,EAAE,CAAE,SAAS,KAAK,MAClC,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU,EAC9C,CAAC,CAAC,KAAK,IACU,EAAE,IAAI,OAAO,IAAI;CAEpC,OAAO;AACT;;AAGA,SAAS,WAAW,QAAkB,QAAwB;CAC5D,IAAI,KAAK;CACT,IAAI,KAAK,OAAO;CAChB,OAAO,KAAK,IAAI;EACd,MAAM,MAAO,KAAK,OAAQ;EAC1B,IAAI,OAAO,QAAS,QAAQ,KAAK,MAAM;OAClC,KAAK;CACZ;CACA,OAAO;AACT;AAEA,SAAS,UAAU,GAAoB;CACrC,IAAI,MAAM,QAAQ,MAAM,KAAA,GAAW,OAAO;CAC1C,IAAI,OAAO,MAAM,UAAU,OAAO;CAClC,IAAI;EACF,OAAO,KAAK,UAAU,CAAC;CACzB,QAAQ;EACN,OAAO,OAAO,CAAC;CACjB;AACF"}
@@ -1,4 +1,91 @@
1
1
  import { h as RolloutLine } from "./schema-Cef2cFmb.js";
2
+ //#region src/statistics/descriptive.d.ts
3
+ /**
4
+ * Descriptive statistics: means, bootstrap spread, correlation, and the
5
+ * weighted judge-dimension composite. Nothing here is a significance test.
6
+ */
7
+ /** Weighted mean — falls back to uniform weights when omitted */
8
+ declare function weightedMean(scores: {
9
+ score: number;
10
+ weight?: number;
11
+ }[]): number;
12
+ /**
13
+ * Percentile bootstrap confidence interval on the mean of `scores`.
14
+ *
15
+ * Descriptive spread. It is not a significance test, and at small n its bounds
16
+ * are anti-conservative in the same way {@link pairedBootstrap}'s are — see
17
+ * {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
18
+ * the scores themselves, so the interval is reproducible either way.
19
+ */
20
+ declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
21
+ seed?: number;
22
+ resamples?: number;
23
+ }): {
24
+ mean: number;
25
+ lower: number;
26
+ upper: number;
27
+ };
28
+ /** Partial credit: returns 0-1 ratio of current toward target */
29
+ declare function partialCredit(current: number, target: number): number;
30
+ /** Distribution summary of a number series: count, extremes, quantiles, sum. */
31
+ interface SeriesDistribution {
32
+ readonly n: number;
33
+ readonly min: number;
34
+ readonly p50: number;
35
+ readonly p90: number;
36
+ readonly max: number;
37
+ readonly sum: number;
38
+ }
39
+ /**
40
+ * Fold a number series into its distribution summary. Quantiles use the
41
+ * nearest-rank definition — the `ceil(q·n)`-th order statistic — so every
42
+ * reported quantile is a value from the series. Returns `null` for an empty
43
+ * series: an empty series has no distribution, and a zero-filled summary
44
+ * would read as a measured all-zero series.
45
+ */
46
+ declare function summarizeNumberSeries(values: readonly number[]): SeriesDistribution | null;
47
+ /**
48
+ * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
49
+ * of the ranks they span, the standard correction for Spearman's ρ.
50
+ */
51
+ declare function ranks(xs: number[]): number[];
52
+ /**
53
+ * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
54
+ * equal-length series. See the edge-case contract above: NaN for n < 2 or
55
+ * unequal lengths, 1 when both series are constant, 0 when exactly one is.
56
+ */
57
+ declare function pearsonR(a: number[], b: number[]): number;
58
+ /**
59
+ * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
60
+ * transform of each series. Same edge-case contract as {@link pearsonR}.
61
+ */
62
+ declare function spearmanR(a: number[], b: number[]): number;
63
+ interface WeightedCompositeInput {
64
+ /** Per-dimension scores (typically 0..1). */
65
+ dims: Record<string, number>;
66
+ /** Weight per dimension. Every weighted dimension MUST be present in
67
+ * `dims` — a weight for an absent dimension is a config error and throws,
68
+ * because silently dropping it would renormalise the composite onto a
69
+ * different denominator than intended. */
70
+ weights: Record<string, number>;
71
+ /** Optional pass threshold; when set, the result reports `pass`. */
72
+ threshold?: number;
73
+ }
74
+ interface WeightedCompositeResult {
75
+ composite: number;
76
+ pass?: boolean;
77
+ }
78
+ /**
79
+ * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
80
+ * the weighted dimensions. The canonical replacement for the per-consumer
81
+ * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
82
+ *
83
+ * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
84
+ * weight is negative, or if the weights sum to 0 — none of which can produce
85
+ * a meaningful composite.
86
+ */
87
+ declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
88
+ //#endregion
2
89
  //#region src/supervisor-run/types.d.ts
3
90
  /** A metric that could not be computed, with the reason its artifact was missing. */
4
91
  interface Unavailable {
@@ -88,6 +175,12 @@ interface SupervisorRunSources {
88
175
  * settled `verdict` may be a legacy string or `{ valid, score, ... }`.
89
176
  */
90
177
  readonly journal: string | null;
178
+ /**
179
+ * Source-specific reason `journal` is null. The analyzer uses it verbatim as
180
+ * the `unavailable` reason on every journal-dependent metric, so a non-loops
181
+ * layout names its own journal file instead of inheriting the loops paths.
182
+ */
183
+ readonly journalMissingReason?: string;
91
184
  /** Per-brain-call tap (JSONL): finish_reason, completion tokens, requested max tokens. */
92
185
  readonly brainLog: string | null;
93
186
  /** Source-specific reason `brainLog` is absent. */
@@ -189,6 +282,15 @@ interface OrchestrationMetrics {
189
282
  readonly delegationDepth: Measured<number>;
190
283
  readonly timeToFirstSpawnMs: Measured<number>;
191
284
  readonly supervisorWallMs: Measured<number>;
285
+ /**
286
+ * Which measurement `supervisorWallMs` holds — never a silent substitution.
287
+ * `stamps`: explicit start and completion stamps. `journal-span`: the start
288
+ * stamp (or first stamped event) to the last stamped journal event — a
289
+ * lower bound, derived when the store wrote no completion stamp. `idleMs`,
290
+ * `idlePct`, and `workerUtilization` cover the same span. Unavailable
291
+ * exactly when `supervisorWallMs` is, with the same reason.
292
+ */
293
+ readonly supervisorWallSource: Measured<'stamps' | 'journal-span'>;
192
294
  /** Wall time inside the supervisor run with ZERO live workers. */
193
295
  readonly idleMs: Measured<number>;
194
296
  readonly idlePct: Measured<number>;
@@ -251,13 +353,33 @@ interface PerWorkerRow {
251
353
  /** Numeric verdict score exactly as recorded; null means no score was recorded. */
252
354
  readonly score: number | null;
253
355
  }
254
- interface WallDistribution {
255
- readonly n: number;
256
- readonly min: number;
257
- readonly p50: number;
258
- readonly p90: number;
259
- readonly max: number;
260
- readonly sum: number;
356
+ /**
357
+ * `SeriesDistribution` (from `../statistics`) over per-worker wall
358
+ * milliseconds. The fold itself is `summarizeNumberSeries`, exported for any
359
+ * series — fleet wall medians, tokens-per-claim spreads — not only wall.
360
+ */
361
+ type WallDistribution = SeriesDistribution;
362
+ /** One spend measurement and the number of source records behind it. */
363
+ interface SpendMeasurement {
364
+ readonly usd: Measured<number>;
365
+ /** Source records folded into `usd`; 0 when the measurement is unavailable. */
366
+ readonly records: number;
367
+ }
368
+ /**
369
+ * The run's total inference spend, measured two ways.
370
+ *
371
+ * `closeRecord` is the spend the store recorded as settled when the run
372
+ * closed (loops `state.json` `result.spentUsd`; Runtime `result.json`
373
+ * `spentTotal.usd`) — the billing-shaped answer. `journalDerived` is the
374
+ * spend execution observably consumed (journal `metered` + `settled` rows) —
375
+ * the execution-accounting answer. Neither is canonical for the other's
376
+ * question. The two cover different records at different moments, so
377
+ * divergence between them is itself a signal (a dropped settlement, a
378
+ * double meter, spend after the close) — read it, never average it away.
379
+ */
380
+ interface SpendMeasurements {
381
+ readonly journalDerived: SpendMeasurement;
382
+ readonly closeRecord: SpendMeasurement;
261
383
  }
262
384
  interface EconomicsMetrics {
263
385
  /** Driver/brain inference — journal `metered` events. */
@@ -272,6 +394,14 @@ interface EconomicsMetrics {
272
394
  readonly brainTruncations: Measured<number>;
273
395
  /** Worker inference — journal `settled` spend plus the harness session join. */
274
396
  readonly workers: RoleSpend;
397
+ /** Both total-spend measurements, each with its own record count. */
398
+ readonly spend: SpendMeasurements;
399
+ /**
400
+ * One collapsed number kept for existing consumers: the close record when
401
+ * the store wrote one, else the journal-derived sum. `totalUsdSource` names
402
+ * the pick. Prefer `spend` — the collapse hides which accounting question
403
+ * the number answers.
404
+ */
275
405
  readonly totalUsd: Measured<number>;
276
406
  /**
277
407
  * Where `totalUsd` came from. CLI-backend workers never price their own inference into
@@ -343,7 +473,27 @@ interface SupervisorRunRollup {
343
473
  readonly idlePctMean: Measured<number>;
344
474
  readonly workersSpawnedTotal: Measured<number>;
345
475
  readonly acceptedTotal: Measured<number>;
476
+ /**
477
+ * Sum of the per-run collapsed `totalUsd`. Prefer `spendUsd`: this total
478
+ * mixes close-record and journal-derived cells without saying which.
479
+ */
346
480
  readonly usdTotal: Measured<number>;
481
+ /**
482
+ * Fleet spend measured two ways. `runs` is each measurement's own
483
+ * denominator — the cells where that measurement was available. The two
484
+ * sums cover different run sets, so comparing the values without their
485
+ * denominators manufactures a phantom divergence.
486
+ */
487
+ readonly spendUsd: {
488
+ readonly journalDerived: {
489
+ readonly value: Measured<number>;
490
+ readonly runs: number;
491
+ };
492
+ readonly closeRecord: {
493
+ readonly value: Measured<number>;
494
+ readonly runs: number;
495
+ };
496
+ };
347
497
  readonly resolvedCount: Measured<number>;
348
498
  readonly perCell: readonly RollupCellRow[];
349
499
  }
@@ -368,5 +518,5 @@ interface SupervisorRunTreeGap {
368
518
  readonly count?: number;
369
519
  }
370
520
  //#endregion
371
- export { Unavailable as C, showMeasured as D, isUnavailable as E, unavailable as O, SupervisorRunTreeGapCode as S, WorkerLogSource as T, SupervisorRunReport as _, OrchestrationMetrics as a, SupervisorRunTree as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SupervisorRunReader as g, SupervisorRunNodeRole as h, NO_SOURCE_LIMITS as i, RoleSpend as l, SteerBreakdown as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunRollup as v, WallDistribution as w, SupervisorRunTreeGap as x, SupervisorRunSources as y };
372
- //# sourceMappingURL=types-yLK8gXE9.d.ts.map
521
+ export { unavailable as A, weightedComposite as B, SupervisorRunTreeGap as C, WorkerLogSource as D, WallDistribution as E, partialCredit as F, pearsonR as I, ranks as L, WeightedCompositeInput as M, WeightedCompositeResult as N, isUnavailable as O, confidenceInterval as P, spearmanR as R, SupervisorRunTree as S, Unavailable as T, weightedMean as V, SupervisorRunNodeRole as _, OrchestrationMetrics as a, SupervisorRunRollup as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SteerBreakdown as g, SpendMeasurements as h, NO_SOURCE_LIMITS as i, SeriesDistribution as j, showMeasured as k, RoleSpend as l, SpendMeasurement as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunReader as v, SupervisorRunTreeGapCode as w, SupervisorRunSources as x, SupervisorRunReport as y, summarizeNumberSeries as z };
522
+ //# sourceMappingURL=types-I5WwVzQ7.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-I5WwVzQ7.d.ts","names":[],"sources":["../src/statistics/descriptive.ts","../src/supervisor-run/types.ts"],"mappings":";;;;;;;iBAQgB,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;iBAiClB,cAAc,iBAAiB;;UAM9B;WACN;WACA;WACA;WACA;WACA;WACA;;;;;;;;;iBAUK,sBAAsB,4BAA4B;;;;;iBA8BlD,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;;;;UC5JjD;WACN;;;KAIC,SAAS,KAAK,IAAI;iBAEd,YAAY,iBAAiB;iBAI7B,cAAc,aAAa,KAAK;;iBAKhC,aAAa,GAAG;;KAWpB;;;;;UAMK;;;;;WAKN;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;WACA;WACA;WACA;;;;;;;;;;;;;UAcM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;;cAIE,kBAAkB;;;;;;;;;;UAiBd;;WAEN;WACA;;WAEA;;WAEA;;;;;;WAMA;;;;;;WAMA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA,kBAAkB;;WAElB;;WAEA;;;;;;;WAOA;;WAEA;;WAEA;;WAEA;;;;;WAKA;IACP;IACA;IACA;IACA;;IAEA;IACA;;WAEO;;WAEA,QAAQ;;;;;;WAMR;;;;;WAKA;;;;;;UAOM;;WAEN;EACT,QAAQ,QAAQ;;cAOL;cACA;UAEI;;WAEN;WACA;;WAEA;;WAEA;;UAGM;WACN,gBAAgB;WAChB,gBAAgB;WAChB,kBAAkB;;WAElB,QAAQ;WACR,iBAAiB;WACjB,gBAAgB,kBAAkB;;WAElC,kBAAkB;;;;;;WAMlB,OAAO;WACP,WAAW;WACX,gBAAgB;;WAEhB,UAAU;;WAEV,gBAAgB;;WAEhB,iBAAiB;WACjB,oBAAoB;WACpB,kBAAkB;;;;;;;;;WASlB,sBAAsB;;WAEtB,QAAQ;WACR,SAAS;;WAET,mBAAmB;;UAGb;WACN,iBAAiB,SAAS;WAC1B,iBAAiB,SAAS;;WAE1B,UAAU;;WAEV,UAAU;;WAEV,WAAW;;WAEX,oBAAoB;;WAEpB,wBAAwB;;WAExB,eAAe;WACf,qBAAqB;;UAGf;WACN,UAAU;WACV,WAAW;;;;;;WAMX,WAAW;WACX,YAAY;WACZ,KAAK;WACL;;UAGM;;WAEN;WACA;;WAEA,MAAM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;WACA;;WAEA;WACA;WACA;WACA;WACA;;WAEA;;;;;;;KAQC,mBAAmB;;UAGd;WACN,KAAK;;WAEL;;;;;;;;;;;;;;UAeM;WACN,gBAAgB;WAChB,aAAa;;UAGP;;WAEN,OAAO;;;;;;;;WAQP,kBAAkB;;WAElB,SAAS;;WAET,OAAO;;;;;;;WAOP,UAAU;;;;;;WAMV;WACA,yBAAyB;WACzB,0BAA0B,SAAS;WACnC,WAAW,kBAAkB;;UAGvB;WACN;WACA;WACA;WACA;;UAGM;WACN,WAAW;WACX,YAAY;WACZ,WAAW;WACX,eAAe;WACf,YAAY;WACZ,aAAa;WACb,YAAY;WACZ,YAAY;WACZ,UAAU;WACV,OAAO,SAAS;;WAEhB;;UAGM;WACN,eAAe;;WAEf;WACA;WACA;WACA,cAAc;WACd,yBAAyB;WACzB;WACA,eAAe;WACf,UAAU;WACV,WAAW;WACX,SAAS;;WAET;;WAEA;;UAGM;WACN;WACA;WACA,QAAQ;WACR,OAAO;WACP,aAAa;WACb,SAAS;WACT,UAAU;WACV,KAAK;;UAGC;WACN,eAAe;WACf;WACA,aAAa;WACb,iBAAiB;WACjB;WACA,WAAW;WACX,mBAAmB;WACnB,iBAAiB;WACjB,aAAa;WACb,qBAAqB;WACrB,eAAe;;;;;WAKf,UAAU;;;;;;;WAOV;aACE;eAA2B,OAAO;eAA2B;;aAC7D;eAAwB,OAAO;eAA2B;;;WAE5D,eAAe;WACf,kBAAkB;;;;;;;;UASZ;WACN;WACA,gBAAgB;;WAEhB,eAAe;;;KAId;UASK;WACN,MAAM;WACN;WACA;WACA"}
@@ -41,26 +41,13 @@ it*. Unified at the trace level, you see both as one timeline per cell.
41
41
  - Compose: register TraceAI's instrumentations on the global tracer
42
42
  provider, then either point both at your OTLP collector or at
43
43
  TraceAI's hosted backend if you want their UI.
44
- - **A bridge exists in source but is not published:** `createOtelBridge`
45
- (`src/adapters/otel.ts`) converts finished OTel spans (`ReadableSpan`
46
- shape) and forwards them into the hosted-tier ingest, lifting
47
- `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
48
- `tangle.generation` to first-class wire fields so the dashboard pivots
49
- correctly. As of 0.140.x there is no `./adapters/otel` entry in
50
- `package.json` `exports`, so it is not importable from the published
51
- package — the snippet below documents the design; it will throw
52
- `ERR_PACKAGE_PATH_NOT_EXPORTED` against `@tangle-network/agent-eval`
53
- installed from npm.
54
- ```ts
55
- // (not currently published — see status note above)
56
- import { createHostedClient } from '@tangle-network/agent-eval/hosted'
57
- import { createOtelBridge } from '@tangle-network/agent-eval/adapters/otel'
58
-
59
- const client = createHostedClient({ endpoint, apiKey, tenantId })
60
- const bridge = createOtelBridge({ client, defaultRunId: substrateRunId })
61
- processor.onEnd = (span) => { void bridge.ingest([span]) }
62
- // ...or call `bridge.ingest(batch)` from a SpanProcessor.onShutdown.
63
- ```
44
+ - **No OTel-span-to-hosted-ingest bridge ships.** To land finished OTel
45
+ spans in the hosted tier, write your own mapping from the span shape to
46
+ `TraceEvent` rows and post them through `createHostedClient` from
47
+ `@tangle-network/agent-eval/hosted` or the `/v1/traces/ingest` wire
48
+ route ([wire-protocol.md](./wire-protocol.md#tracesingest-batch-ingest-production-trace-events)).
49
+ For run records that already exist, `fromOtelSpans` from `/contract`
50
+ converts collector output into `RunRecord[]` for `analyzeRuns()`.
64
51
 
65
52
  ### Langfuse SDK
66
53
 
@@ -131,9 +118,8 @@ OTel-protocol composition:
131
118
  1. **Cost-aware judging.** Your observability tool's auto-instrumented
132
119
  spans carry token counts + cost. A custom `JudgeConfig` can read
133
120
  them via the OTel context and refuse to score artifacts that
134
- exceeded a per-call budget. Easy to write yourself; we'll ship a
135
- reference helper (`costAwareJudgeFromOtel`) when a partner pulls on
136
- this.
121
+ exceeded a per-call budget. Easy to write yourself; no reference
122
+ helper ships today.
137
123
  2. **Tool-aware judging.** Your instrumentation captures the tool-call
138
124
  sequence (`langchain.tool.invoked`, `openai.function.called`, etc.).
139
125
  A judge that scores "did the agent use the right tool" reads those