@tangle-network/agent-eval 0.132.0 → 0.133.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +15 -0
  2. package/README.md +1 -1
  3. package/dist/agent-profile-cell-OhuTee9n.js +335 -0
  4. package/dist/agent-profile-cell-OhuTee9n.js.map +1 -0
  5. package/dist/analyst/index.js +3 -3
  6. package/dist/{analyze-runs-BjPn_fOS.js → analyze-runs-B-afTpCv.js} +6 -19
  7. package/dist/analyze-runs-B-afTpCv.js.map +1 -0
  8. package/dist/{analyze-runs-AFDI5RI0.d.ts → analyze-runs-DZr7JW-m.d.ts} +2 -2
  9. package/dist/{analyze-runs-AFDI5RI0.d.ts.map → analyze-runs-DZr7JW-m.d.ts.map} +1 -1
  10. package/dist/{baseline-HsBvw_dk.js → baseline-DcX5hQDv.js} +2 -70
  11. package/dist/{baseline-HsBvw_dk.js.map → baseline-DcX5hQDv.js.map} +1 -1
  12. package/dist/baseline-hG3K85h4.d.ts.map +1 -1
  13. package/dist/benchmarks/index.d.ts +1 -1
  14. package/dist/benchmarks/index.js +1 -1
  15. package/dist/{benchmarks-DTrT3UH-.js → benchmarks-BU7P6PCW.js} +3 -3
  16. package/dist/{benchmarks-DTrT3UH-.js.map → benchmarks-BU7P6PCW.js.map} +1 -1
  17. package/dist/builder-eval/index.js +1 -1
  18. package/dist/campaign/index.d.ts +3 -3
  19. package/dist/campaign/index.js +2 -2
  20. package/dist/{campaign-Cx6CfMR4.js → campaign-CnzHQndg.js} +21 -15
  21. package/dist/{campaign-Cx6CfMR4.js.map → campaign-CnzHQndg.js.map} +1 -1
  22. package/dist/cli.js +1 -1
  23. package/dist/{client-aZDHJiKO.d.ts → client-D4F9hdzR.d.ts} +2 -2
  24. package/dist/{client-aZDHJiKO.d.ts.map → client-D4F9hdzR.d.ts.map} +1 -1
  25. package/dist/contract/index.d.ts +145 -6
  26. package/dist/contract/index.d.ts.map +1 -1
  27. package/dist/contract/index.js +1395 -11
  28. package/dist/contract/index.js.map +1 -1
  29. package/dist/control.js +1 -1
  30. package/dist/{cost-ledger-ZAa_P4r0.js → cost-ledger-BrJxbrMy.js} +238 -3
  31. package/dist/cost-ledger-BrJxbrMy.js.map +1 -0
  32. package/dist/{default-registry-B1JcpnRv.js → default-registry-D3T9XbuY.js} +3 -3
  33. package/dist/{default-registry-B1JcpnRv.js.map → default-registry-D3T9XbuY.js.map} +1 -1
  34. package/dist/{eval-campaign-mDKhkdUq.js → eval-campaign-DXhpZghy.js} +5 -6
  35. package/dist/{eval-campaign-mDKhkdUq.js.map → eval-campaign-DXhpZghy.js.map} +1 -1
  36. package/dist/{task-failure-attributes-CQZlB3et.js → extract-usage-2j25whHw.js} +154 -2
  37. package/dist/extract-usage-2j25whHw.js.map +1 -0
  38. package/dist/fuzz.js +1 -1
  39. package/dist/hosted/index.d.ts +1 -1
  40. package/dist/{index-3cdlURSk2.d.ts → index-3cdlURSk.d.ts} +1 -1
  41. package/dist/index-3cdlURSk.d.ts.map +1 -0
  42. package/dist/{index-FpfWFsKm.d.ts → index-Ba636PKl.d.ts} +29 -6
  43. package/dist/{index-FpfWFsKm.d.ts.map → index-Ba636PKl.d.ts.map} +1 -1
  44. package/dist/{index-p2TR_iWJ.d.ts → index-Wek5mU0y.d.ts} +3 -3
  45. package/dist/{index-p2TR_iWJ.d.ts.map → index-Wek5mU0y.d.ts.map} +1 -1
  46. package/dist/{index-BAvgST_9.d.ts → index-nhIYz9hn.d.ts} +67 -9
  47. package/dist/index-nhIYz9hn.d.ts.map +1 -0
  48. package/dist/index.d.ts +55 -9
  49. package/dist/index.d.ts.map +1 -1
  50. package/dist/index.js +111 -24
  51. package/dist/index.js.map +1 -1
  52. package/dist/ledger-core/index.d.ts +2 -2
  53. package/dist/ledger-core/index.js +2 -2
  54. package/dist/ledger-core-CPZfcrC2.js +620 -0
  55. package/dist/ledger-core-CPZfcrC2.js.map +1 -0
  56. package/dist/{llm-client-BNcP4v08.js → llm-client-ClPW-dWB.js} +2 -2
  57. package/dist/{llm-client-BNcP4v08.js.map → llm-client-ClPW-dWB.js.map} +1 -1
  58. package/dist/meta-eval/index.d.ts +215 -2
  59. package/dist/{index-CXs7QlR5.d.ts.map → meta-eval/index.d.ts.map} +1 -1
  60. package/dist/meta-eval/index.js +93 -3
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/{mint-D5_87M5L.js → mint-BvkwcYZU.js} +2 -2
  63. package/dist/{mint-D5_87M5L.js.map → mint-BvkwcYZU.js.map} +1 -1
  64. package/dist/openapi.json +1 -1
  65. package/dist/{paired-arms-D9D0wXj2.js → paired-arms-6XItKzd1.js} +2 -2
  66. package/dist/{paired-arms-D9D0wXj2.js.map → paired-arms-6XItKzd1.js.map} +1 -1
  67. package/dist/pipelines/index.js +2 -2
  68. package/dist/profile-cell.js +1 -242
  69. package/dist/{propose-review-control-Bqb7daEJ.js → propose-review-control-SQ-n9-We.js} +2 -2
  70. package/dist/{propose-review-control-Bqb7daEJ.js.map → propose-review-control-SQ-n9-We.js.map} +1 -1
  71. package/dist/{release-report-oWt9f2k-.js → release-report-wuilQkvK.js} +3 -3
  72. package/dist/{release-report-oWt9f2k-.js.map → release-report-wuilQkvK.js.map} +1 -1
  73. package/dist/{replay-D18-pBAA.js → replay-CJfGLdx4.js} +4 -5
  74. package/dist/{replay-D18-pBAA.js.map → replay-CJfGLdx4.js.map} +1 -1
  75. package/dist/reporting.d.ts +1 -1
  76. package/dist/reporting.js +4 -4
  77. package/dist/{reward-hacking-BEvjdUtD.js → reward-hacking-Dl2UBzej.js} +3 -3
  78. package/dist/{reward-hacking-BEvjdUtD.js.map → reward-hacking-Dl2UBzej.js.map} +1 -1
  79. package/dist/rl.d.ts +153 -3
  80. package/dist/rl.d.ts.map +1 -1
  81. package/dist/rl.js +222 -7
  82. package/dist/rl.js.map +1 -1
  83. package/dist/rollout/index.d.ts +1 -1
  84. package/dist/rollout/index.js +2 -2
  85. package/dist/{rollout-VEo41J0N.js → rollout-CeTlDrf6.js} +2 -2
  86. package/dist/{rollout-VEo41J0N.js.map → rollout-CeTlDrf6.js.map} +1 -1
  87. package/dist/{rubric-predictive-validity-B3xmbmS1.js → rubric-predictive-validity-QG7ydk0s.js} +2 -2
  88. package/dist/{rubric-predictive-validity-B3xmbmS1.js.map → rubric-predictive-validity-QG7ydk0s.js.map} +1 -1
  89. package/dist/{run-record-CN8Zd21B.js → run-record-BIwU2wdV.js} +2 -2
  90. package/dist/{run-record-CN8Zd21B.js.map → run-record-BIwU2wdV.js.map} +1 -1
  91. package/dist/{semantic-concept-judge-DKCtoOz8.js → semantic-concept-judge-BypLt6Fw.js} +4 -4
  92. package/dist/{semantic-concept-judge-DKCtoOz8.js.map → semantic-concept-judge-BypLt6Fw.js.map} +1 -1
  93. package/dist/{server-Dc_lsOYd.js → server-BPqlDBWK.js} +3 -3
  94. package/dist/{server-Dc_lsOYd.js.map → server-BPqlDBWK.js.map} +1 -1
  95. package/dist/{skillopt-optimization-method-CQlz8GQM.js → skillopt-optimization-method-BoIzh7Dl.js} +6 -6
  96. package/dist/{skillopt-optimization-method-CQlz8GQM.js.map → skillopt-optimization-method-BoIzh7Dl.js.map} +1 -1
  97. package/dist/{skillopt-optimization-method-C9M_lxdo.d.ts → skillopt-optimization-method-CAASpcS3.d.ts} +3 -3
  98. package/dist/{skillopt-optimization-method-C9M_lxdo.d.ts.map → skillopt-optimization-method-CAASpcS3.d.ts.map} +1 -1
  99. package/dist/{statistics-CnnxdpOg.js → statistics-DWM_AyLe.js} +90 -76
  100. package/dist/statistics-DWM_AyLe.js.map +1 -0
  101. package/dist/statistics-DbvkkDPa.d.ts.map +1 -1
  102. package/dist/{summary-report-BNs5nmXI.js → summary-report-Ci17nIdU.js} +4 -4
  103. package/dist/{summary-report-BNs5nmXI.js.map → summary-report-Ci17nIdU.js.map} +1 -1
  104. package/dist/traces.js +2 -2
  105. package/dist/wire/index.js +1 -1
  106. package/package.json +2 -7
  107. package/dist/analyze-runs-BjPn_fOS.js.map +0 -1
  108. package/dist/belief-state/index.d.ts +0 -622
  109. package/dist/belief-state/index.d.ts.map +0 -1
  110. package/dist/belief-state/index.js +0 -1819
  111. package/dist/belief-state/index.js.map +0 -1
  112. package/dist/calibration-CNWWA6K8.js +0 -94
  113. package/dist/calibration-CNWWA6K8.js.map +0 -1
  114. package/dist/code-agent-session-BjkMTQ7H.js +0 -1390
  115. package/dist/code-agent-session-BjkMTQ7H.js.map +0 -1
  116. package/dist/code-agent-session-D5URqc3_.d.ts +0 -143
  117. package/dist/code-agent-session-D5URqc3_.d.ts.map +0 -1
  118. package/dist/cost-ledger-ZAa_P4r0.js.map +0 -1
  119. package/dist/extract-usage-BrQ8mCLX.js +0 -155
  120. package/dist/extract-usage-BrQ8mCLX.js.map +0 -1
  121. package/dist/index-3cdlURSk2.d.ts.map +0 -1
  122. package/dist/index-BAvgST_9.d.ts.map +0 -1
  123. package/dist/index-CXs7QlR5.d.ts +0 -217
  124. package/dist/ledger-core-eqaI3PCD.js +0 -388
  125. package/dist/ledger-core-eqaI3PCD.js.map +0 -1
  126. package/dist/metrics-C9YY1OcL.js +0 -239
  127. package/dist/metrics-C9YY1OcL.js.map +0 -1
  128. package/dist/off-policy-DvgzvtIx.js +0 -220
  129. package/dist/off-policy-DvgzvtIx.js.map +0 -1
  130. package/dist/off-policy-mskQw8Mb.d.ts +0 -153
  131. package/dist/off-policy-mskQw8Mb.d.ts.map +0 -1
  132. package/dist/pre-registration-DakwTRXk.js +0 -96
  133. package/dist/pre-registration-DakwTRXk.js.map +0 -1
  134. package/dist/profile-cell.js.map +0 -1
  135. package/dist/runtime-trajectory-1gyaTOoC.js +0 -93
  136. package/dist/runtime-trajectory-1gyaTOoC.js.map +0 -1
  137. package/dist/runtime-trajectory-BW9Wszb-.d.ts +0 -50
  138. package/dist/runtime-trajectory-BW9Wszb-.d.ts.map +0 -1
  139. package/dist/statistics-CnnxdpOg.js.map +0 -1
  140. package/dist/task-failure-attributes-CQZlB3et.js.map +0 -1
@@ -1 +0,0 @@
1
- {"version":3,"file":"metrics-C9YY1OcL.js","names":[],"sources":["../src/metrics.ts"],"sourcesContent":["import type { ProductClient } from './client'\nimport type { DriverState, TurnMetrics } from './types'\n\ninterface TokenPrice {\n input: number\n output: number\n}\n\n/** Per-1K token pricing for exact model ids. */\nexport const MODEL_PRICING: Record<string, TokenPrice> = {\n 'gpt-4o': { input: 0.0025, output: 0.01 },\n 'gpt-4o-mini': { input: 0.00015, output: 0.0006 },\n 'gpt-4-turbo': { input: 0.01, output: 0.03 },\n 'claude-sonnet-4-20250514': { input: 0.003, output: 0.015 },\n 'claude-opus-4-20250514': { input: 0.015, output: 0.075 },\n 'claude-3-haiku-20240307': { input: 0.00025, output: 0.00125 },\n}\n\n/** Family-level pricing fallbacks (per-1K), matched against a normalized id\n * after exact lookup misses. Ordered — first match wins. Covers the model\n * ids actually used through the Tangle router + cli-bridge harnesses\n * (`claude-code/sonnet`, `opencode/zai-coding-plan/glm-5.1`,\n * `kimi-code/kimi-k2.6`, `deepseek-v4-pro`, `anthropic/claude-sonnet-4-6`, …),\n * none of which appear in the exact table above — without this they priced\n * to a silent $0, blanking every cost/Pareto axis downstream. */\nconst FAMILY_PRICING: Array<[RegExp, TokenPrice]> = [\n [/claude.*opus/, { input: 0.015, output: 0.075 }],\n [/claude.*haiku/, { input: 0.0008, output: 0.004 }],\n [/claude.*sonnet|claude-code|claude-sonnet/, { input: 0.003, output: 0.015 }],\n [/gpt-4o-mini/, { input: 0.00015, output: 0.0006 }],\n [/gpt-5|gpt-4\\.1|o[134]\\b/, { input: 0.00125, output: 0.01 }],\n [/gpt-4o|gpt-4/, { input: 0.0025, output: 0.01 }],\n [/deepseek/, { input: 0.0003, output: 0.0011 }],\n [/glm|zhipu|zai/, { input: 0.0006, output: 0.0022 }],\n [/kimi|moonshot/, { input: 0.0006, output: 0.0025 }],\n [/qwen/, { input: 0.0004, output: 0.0012 }],\n [/gemini.*flash/, { input: 0.0001, output: 0.0004 }],\n [/gemini/, { input: 0.00125, output: 0.005 }],\n [/llama/, { input: 0.0002, output: 0.0006 }],\n]\n\n/** Normalize a model id for pricing: drop a `@snapshot` suffix, lowercase,\n * and keep the final harness/provider-prefixed segment so family regexes\n * match (`opencode/zai-coding-plan/glm-5.1` → `glm-5.1`). */\nfunction normalizeModelId(model: string): string {\n return (model.split('@')[0] ?? model).trim().toLowerCase()\n}\n\n/** Resolve pricing for a model id: exact table, then family fallback.\n * Returns null when the id matches nothing (caller decides — never a\n * silent-zero masquerading as a real $0 cost). */\nexport function resolveModelPricing(model: string): TokenPrice | null {\n if (MODEL_PRICING[model]) return MODEL_PRICING[model]\n const id = normalizeModelId(model)\n if (MODEL_PRICING[id]) return MODEL_PRICING[id]\n for (const [pattern, price] of FAMILY_PRICING) {\n if (pattern.test(id)) return price\n }\n return null\n}\n\n/** True when `model` has known pricing (exact or family). Lets cost-aware\n * callers distinguish a real $0 from an unpriced model. */\nexport function isModelPriced(model: string): boolean {\n return resolveModelPricing(model) !== null\n}\n\nconst warnedUnpricedModels = new Set<string>()\n\n/** Estimate token count from string length (chars / 4 approximation) */\nexport function estimateTokens(text: string): number {\n return Math.ceil(text.length / 4)\n}\n\n/** Calculate cost in USD from token counts and model. Unknown models warn\n * once (not a silent zero) and return 0 so callers that ignore pricing keep\n * working; cost-sensitive callers should gate on {@link isModelPriced}. */\nexport function estimateCost(inputTokens: number, outputTokens: number, model: string): number {\n const pricing = resolveModelPricing(model)\n if (!pricing) {\n if (!warnedUnpricedModels.has(model)) {\n warnedUnpricedModels.add(model)\n console.warn(\n `estimateCost: no pricing for model \"${model}\" — returning 0; add it to ` +\n 'MODEL_PRICING/FAMILY_PRICING (cost/Pareto axes will be blank until then)',\n )\n }\n return 0\n }\n return (inputTokens / 1000) * pricing.input + (outputTokens / 1000) * pricing.output\n}\n\n/**\n * TokenCounter — accumulates token usage and cost across turns.\n */\nexport class TokenCounter {\n private totalInput = 0\n private totalOutput = 0\n private totalCost = 0\n private model: string\n\n constructor(model = 'gpt-4o') {\n this.model = model\n }\n\n /** Record tokens for a turn, returns per-turn cost */\n record(inputTokens: number, outputTokens: number): number {\n this.totalInput += inputTokens\n this.totalOutput += outputTokens\n const cost = estimateCost(inputTokens, outputTokens, this.model)\n this.totalCost += cost\n return cost\n }\n\n /** Estimate and record from raw text */\n recordFromText(\n inputText: string,\n outputText: string,\n ): { inputTokens: number; outputTokens: number; cost: number } {\n const inputTokens = estimateTokens(inputText)\n const outputTokens = estimateTokens(outputText)\n const cost = this.record(inputTokens, outputTokens)\n return { inputTokens, outputTokens, cost }\n }\n\n getTotalInput(): number {\n return this.totalInput\n }\n getTotalOutput(): number {\n return this.totalOutput\n }\n getTotalCost(): number {\n return this.totalCost\n }\n}\n\n/**\n * MetricsCollector — collects per-turn metrics from the product.\n *\n * After each turn, queries the product's APIs to measure state changes.\n */\nexport class MetricsCollector {\n private client: ProductClient\n private workspaceId: string\n private metrics: TurnMetrics[] = []\n constructor(client: ProductClient, workspaceId: string) {\n this.client = client\n this.workspaceId = workspaceId\n }\n\n /** Collect metrics after a turn completes */\n async collect(\n turn: number,\n responseLatencyMs: number,\n responseChars: number,\n codeBlocksProduced: number,\n blocksExtracted: number,\n completionCriteriaMet: number,\n completionCriteriaTotal: number,\n qualityScore?: number,\n inputTokens = 0,\n outputTokens = 0,\n estimatedCostUsd = 0,\n ): Promise<TurnMetrics> {\n const state = await this.getState()\n\n const m: TurnMetrics = {\n turn,\n timestamp: new Date().toISOString(),\n tasks: state.tasks,\n events: state.events,\n proposals: state.proposals,\n vaultFiles: state.vaultFiles.length,\n responseLatencyMs,\n responseChars,\n codeBlocksProduced,\n blocksExtracted,\n qualityScore,\n inputTokens,\n outputTokens,\n estimatedCostUsd,\n totalCostUsd: estimatedCostUsd,\n completionPercent:\n completionCriteriaTotal > 0 ? (completionCriteriaMet / completionCriteriaTotal) * 100 : 0,\n }\n\n this.metrics.push(m)\n return m\n }\n\n /** Get current product state */\n async getState(): Promise<DriverState> {\n const [tasks, events, approvals, vaultFiles] = await Promise.all([\n this.client.getTasks(this.workspaceId),\n this.client.getEvents(this.workspaceId),\n this.client.getApprovals(this.workspaceId),\n this.client.getVaultTree(this.workspaceId),\n ])\n\n return {\n tasks: tasks.length,\n events: events.length,\n proposals: {\n pending: approvals.filter((a) => a.status === 'pending').length,\n approved: approvals.filter((a) => a.status === 'approved').length,\n rejected: approvals.filter((a) => a.status === 'rejected').length,\n },\n vaultFiles,\n codeBlocks: 0,\n generations: 0,\n }\n }\n\n /** Get all collected metrics */\n getMetrics(): TurnMetrics[] {\n return [...this.metrics]\n }\n\n /** Get convergence curve (completion% over turns) */\n getConvergenceCurve(): number[] {\n return this.metrics.map((m) => m.completionPercent)\n }\n}\n"],"mappings":";;AASA,MAAa,gBAA4C;CACvD,UAAU;EAAE,OAAO;EAAQ,QAAQ;CAAK;CACxC,eAAe;EAAE,OAAO;EAAS,QAAQ;CAAO;CAChD,eAAe;EAAE,OAAO;EAAM,QAAQ;CAAK;CAC3C,4BAA4B;EAAE,OAAO;EAAO,QAAQ;CAAM;CAC1D,0BAA0B;EAAE,OAAO;EAAO,QAAQ;CAAM;CACxD,2BAA2B;EAAE,OAAO;EAAS,QAAQ;CAAQ;AAC/D;;;;;;;;AASA,MAAM,iBAA8C;CAClD,CAAC,gBAAgB;EAAE,OAAO;EAAO,QAAQ;CAAM,CAAC;CAChD,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAM,CAAC;CAClD,CAAC,4CAA4C;EAAE,OAAO;EAAO,QAAQ;CAAM,CAAC;CAC5E,CAAC,eAAe;EAAE,OAAO;EAAS,QAAQ;CAAO,CAAC;CAClD,CAAC,2BAA2B;EAAE,OAAO;EAAS,QAAQ;CAAK,CAAC;CAC5D,CAAC,gBAAgB;EAAE,OAAO;EAAQ,QAAQ;CAAK,CAAC;CAChD,CAAC,YAAY;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CAC9C,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CACnD,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CACnD,CAAC,QAAQ;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CAC1C,CAAC,iBAAiB;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;CACnD,CAAC,UAAU;EAAE,OAAO;EAAS,QAAQ;CAAM,CAAC;CAC5C,CAAC,SAAS;EAAE,OAAO;EAAQ,QAAQ;CAAO,CAAC;AAC7C;;;;AAKA,SAAS,iBAAiB,OAAuB;CAC/C,QAAQ,MAAM,MAAM,GAAG,CAAC,CAAC,MAAM,MAAA,CAAO,KAAK,CAAC,CAAC,YAAY;AAC3D;;;;AAKA,SAAgB,oBAAoB,OAAkC;CACpE,IAAI,cAAc,QAAQ,OAAO,cAAc;CAC/C,MAAM,KAAK,iBAAiB,KAAK;CACjC,IAAI,cAAc,KAAK,OAAO,cAAc;CAC5C,KAAK,MAAM,CAAC,SAAS,UAAU,gBAC7B,IAAI,QAAQ,KAAK,EAAE,GAAG,OAAO;CAE/B,OAAO;AACT;;;AAIA,SAAgB,cAAc,OAAwB;CACpD,OAAO,oBAAoB,KAAK,MAAM;AACxC;AAEA,MAAM,uCAAuB,IAAI,IAAY;;AAG7C,SAAgB,eAAe,MAAsB;CACnD,OAAO,KAAK,KAAK,KAAK,SAAS,CAAC;AAClC;;;;AAKA,SAAgB,aAAa,aAAqB,cAAsB,OAAuB;CAC7F,MAAM,UAAU,oBAAoB,KAAK;CACzC,IAAI,CAAC,SAAS;EACZ,IAAI,CAAC,qBAAqB,IAAI,KAAK,GAAG;GACpC,qBAAqB,IAAI,KAAK;GAC9B,QAAQ,KACN,uCAAuC,MAAM,oGAE/C;EACF;EACA,OAAO;CACT;CACA,OAAQ,cAAc,MAAQ,QAAQ,QAAS,eAAe,MAAQ,QAAQ;AAChF;;;;AAKA,IAAa,eAAb,MAA0B;CACxB,aAAqB;CACrB,cAAsB;CACtB,YAAoB;CACpB;CAEA,YAAY,QAAQ,UAAU;EAC5B,KAAK,QAAQ;CACf;;CAGA,OAAO,aAAqB,cAA8B;EACxD,KAAK,cAAc;EACnB,KAAK,eAAe;EACpB,MAAM,OAAO,aAAa,aAAa,cAAc,KAAK,KAAK;EAC/D,KAAK,aAAa;EAClB,OAAO;CACT;;CAGA,eACE,WACA,YAC6D;EAC7D,MAAM,cAAc,eAAe,SAAS;EAC5C,MAAM,eAAe,eAAe,UAAU;EAE9C,OAAO;GAAE;GAAa;GAAc,MADvB,KAAK,OAAO,aAAa,YACC;EAAE;CAC3C;CAEA,gBAAwB;EACtB,OAAO,KAAK;CACd;CACA,iBAAyB;EACvB,OAAO,KAAK;CACd;CACA,eAAuB;EACrB,OAAO,KAAK;CACd;AACF;;;;;;AAOA,IAAa,mBAAb,MAA8B;CAC5B;CACA;CACA,UAAiC,CAAC;CAClC,YAAY,QAAuB,aAAqB;EACtD,KAAK,SAAS;EACd,KAAK,cAAc;CACrB;;CAGA,MAAM,QACJ,MACA,mBACA,eACA,oBACA,iBACA,uBACA,yBACA,cACA,cAAc,GACd,eAAe,GACf,mBAAmB,GACG;EACtB,MAAM,QAAQ,MAAM,KAAK,SAAS;EAElC,MAAM,IAAiB;GACrB;GACA,4BAAW,IAAI,KAAK,EAAA,CAAE,YAAY;GAClC,OAAO,MAAM;GACb,QAAQ,MAAM;GACd,WAAW,MAAM;GACjB,YAAY,MAAM,WAAW;GAC7B;GACA;GACA;GACA;GACA;GACA;GACA;GACA;GACA,cAAc;GACd,mBACE,0BAA0B,IAAK,wBAAwB,0BAA2B,MAAM;EAC5F;EAEA,KAAK,QAAQ,KAAK,CAAC;EACnB,OAAO;CACT;;CAGA,MAAM,WAAiC;EACrC,MAAM,CAAC,OAAO,QAAQ,WAAW,cAAc,MAAM,QAAQ,IAAI;GAC/D,KAAK,OAAO,SAAS,KAAK,WAAW;GACrC,KAAK,OAAO,UAAU,KAAK,WAAW;GACtC,KAAK,OAAO,aAAa,KAAK,WAAW;GACzC,KAAK,OAAO,aAAa,KAAK,WAAW;EAC3C,CAAC;EAED,OAAO;GACL,OAAO,MAAM;GACb,QAAQ,OAAO;GACf,WAAW;IACT,SAAS,UAAU,QAAQ,MAAM,EAAE,WAAW,SAAS,CAAC,CAAC;IACzD,UAAU,UAAU,QAAQ,MAAM,EAAE,WAAW,UAAU,CAAC,CAAC;IAC3D,UAAU,UAAU,QAAQ,MAAM,EAAE,WAAW,UAAU,CAAC,CAAC;GAC7D;GACA;GACA,YAAY;GACZ,aAAa;EACf;CACF;;CAGA,aAA4B;EAC1B,OAAO,CAAC,GAAG,KAAK,OAAO;CACzB;;CAGA,sBAAgC;EAC9B,OAAO,KAAK,QAAQ,KAAK,MAAM,EAAE,iBAAiB;CACpD;AACF"}
@@ -1,220 +0,0 @@
1
- import { s as ValidationError } from "./errors-8YnH8WlF.js";
2
- //#region src/rl/off-policy.ts
3
- /**
4
- * Off-policy evaluation primitives.
5
- *
6
- * Standard inverse-probability-weighted (IPS), self-normalized
7
- * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
8
- * value of a *target* policy given trajectories collected under a
9
- * *behavior* policy. This is the canonical RL eval task: "we have last
10
- * week's runs, we changed the policy — how would the new one do without
11
- * re-running?"
12
- *
13
- * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
14
- * & Joachims 2015 for SNIPS) but the *application* to LLM-agent
15
- * evaluation needs care:
16
- *
17
- * - The "policy" is the (prompt, tool config, model snapshot) triple.
18
- * Two policies have the same probability over an action *iff* their
19
- * LLM call would emit the same token with the same probability —
20
- * which is generally unknowable without the model log-probs.
21
- * - For LLM agents, propensity scores must be supplied by the caller
22
- * (logged in the trace, recovered from token log-probs, or estimated
23
- * via a learned propensity model). We do NOT estimate propensity here.
24
- * - Doubly-robust requires two outputs from a Q-function: its prediction
25
- * for the logged action and its expectation under the target policy.
26
- * Consumers compute these with a tabular estimate, regression fit, or
27
- * learned reward model before constructing the trajectories.
28
- *
29
- * Bias / variance tradeoffs:
30
- * - IPS: unbiased; high variance for small overlap, infinite variance
31
- * when target has support outside behavior.
32
- * - SNIPS: lower variance, slight bias; usually preferred in practice.
33
- * - DR: doubly-robust — unbiased if either propensity OR Q-function is
34
- * correct. Lowest practical variance when Q is decent. Use this.
35
- *
36
- * Caveat the panel will land: on the LLM-agent setting, propensity scores
37
- * recovered from token log-probs are noisy, the action space is enormous,
38
- * and overlap is often poor. These estimators are useful but not magic;
39
- * complement with `replayCampaign` (exact replay where the request hashes
40
- * match) for high-confidence answers and OPE for the gap.
41
- */
42
- /**
43
- * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator
44
- * of E[reward under target policy]. Variance scales with the spread of
45
- * target/behavior ratios.
46
- */
47
- function inverseProbabilityWeighting(trajectories, opts = {}) {
48
- const cap = opts.weightCap ?? Infinity;
49
- const clip = opts.rewardClip ?? {
50
- low: 0,
51
- high: 1
52
- };
53
- if (trajectories.length === 0) return zeroEstimate();
54
- const weights = [];
55
- const weightedRewards = [];
56
- let maxW = 0;
57
- for (const t of trajectories) {
58
- if (t.behaviorProb <= 0) throw new ValidationError(`inverseProbabilityWeighting: behaviorProb must be > 0 (runId=${t.runId})`);
59
- const w = Math.min(cap, t.targetProb / t.behaviorProb);
60
- const r = clamp(t.reward, clip.low, clip.high);
61
- weights.push(w);
62
- weightedRewards.push(w * r);
63
- if (w > maxW) maxW = w;
64
- }
65
- const n = weights.length;
66
- const value = weightedRewards.reduce((s, x) => s + x, 0) / n;
67
- const variance = weightedRewards.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1);
68
- const sumW = weights.reduce((s, w) => s + w, 0);
69
- const sumW2 = weights.reduce((s, w) => s + w * w, 0);
70
- const effN = sumW === 0 ? 0 : sumW * sumW / sumW2;
71
- return {
72
- value,
73
- standardError: Math.sqrt(variance / n),
74
- effectiveSampleSize: effN,
75
- n,
76
- maxImportanceWeight: maxW
77
- };
78
- }
79
- /**
80
- * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at
81
- * the cost of small bias (vanishing as N grows). The right default for
82
- * LLM-agent evaluation where overlap is often poor.
83
- */
84
- function selfNormalizedImportanceWeighting(trajectories, opts = {}) {
85
- const cap = opts.weightCap ?? Infinity;
86
- const clip = opts.rewardClip ?? {
87
- low: 0,
88
- high: 1
89
- };
90
- if (trajectories.length === 0) return zeroEstimate();
91
- const weights = [];
92
- const rewards = [];
93
- let maxW = 0;
94
- for (const t of trajectories) {
95
- if (t.behaviorProb <= 0) throw new ValidationError(`selfNormalizedImportanceWeighting: behaviorProb must be > 0 (runId=${t.runId})`);
96
- const w = Math.min(cap, t.targetProb / t.behaviorProb);
97
- weights.push(w);
98
- rewards.push(clamp(t.reward, clip.low, clip.high));
99
- if (w > maxW) maxW = w;
100
- }
101
- const sumW = weights.reduce((s, w) => s + w, 0);
102
- const sumWR = weights.reduce((s, w, i) => s + w * rewards[i], 0);
103
- const value = sumW === 0 ? 0 : sumWR / sumW;
104
- const sumW2 = weights.reduce((s, w) => s + w * w, 0);
105
- const effN = sumW === 0 ? 0 : sumW * sumW / sumW2;
106
- const variance = weights.map((w, i) => w * (rewards[i] - value)).reduce((s, x) => s + x * x, 0) / Math.max(1, sumW * sumW);
107
- return {
108
- value,
109
- standardError: Math.sqrt(variance),
110
- effectiveSampleSize: effN,
111
- n: trajectories.length,
112
- maxImportanceWeight: maxW
113
- };
114
- }
115
- /**
116
- * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).
117
- *
118
- * V_DR = (1/N) * sum_i [ v_hat_target_i
119
- * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]
120
- *
121
- * Unbiased if EITHER:
122
- * - the importance ratios are correct (IPS-style validity), OR
123
- * - the Q-hat function is correct (model-based validity).
124
- *
125
- * In practice both are imperfect, but the residual bias is the *product*
126
- * of both errors — much smaller than either alone. This is why DR is the
127
- * default in production OPE pipelines.
128
- *
129
- * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither
130
- * use the exact IPS contribution. `contributionCounts` makes the mix explicit
131
- * in the result.
132
- * Callers must cross-fit the Q-function or train it on independent rows;
133
- * fitting and evaluating Q on the same outcomes leaks the answer.
134
- */
135
- function doublyRobust(trajectories, opts = {}) {
136
- const cap = opts.weightCap ?? Infinity;
137
- const clip = opts.rewardClip ?? {
138
- low: 0,
139
- high: 1
140
- };
141
- if (trajectories.length === 0) return {
142
- ...zeroEstimate(),
143
- contributionCounts: {
144
- dr: 0,
145
- ipsFallback: 0
146
- }
147
- };
148
- const contributions = [];
149
- const contributionCounts = {
150
- dr: 0,
151
- ipsFallback: 0
152
- };
153
- let maxW = 0;
154
- let sumW = 0;
155
- let sumW2 = 0;
156
- for (const t of trajectories) {
157
- if (t.behaviorProb <= 0) throw new ValidationError(`doublyRobust: behaviorProb must be > 0 (runId=${t.runId})`);
158
- const w = Math.min(cap, t.targetProb / t.behaviorProb);
159
- const r = clamp(t.reward, clip.low, clip.high);
160
- const rawQHatChosen = t.qHatChosen;
161
- const rawVHatTarget = t.vHatTarget;
162
- const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== void 0;
163
- const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== void 0;
164
- if (hasQHatChosen !== hasVHatTarget) throw new ValidationError(`doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`);
165
- if (hasQHatChosen && hasVHatTarget) {
166
- if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) throw new ValidationError(`doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`);
167
- const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high);
168
- const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high);
169
- contributions.push(vHatTarget + w * (r - qHatChosen));
170
- contributionCounts.dr += 1;
171
- } else {
172
- contributions.push(w * r);
173
- contributionCounts.ipsFallback += 1;
174
- }
175
- if (w > maxW) maxW = w;
176
- sumW += w;
177
- sumW2 += w * w;
178
- }
179
- const n = contributions.length;
180
- const value = contributions.reduce((s, x) => s + x, 0) / n;
181
- const variance = contributions.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1);
182
- const effN = sumW === 0 ? 0 : sumW * sumW / sumW2;
183
- return {
184
- value,
185
- standardError: Math.sqrt(variance / n),
186
- effectiveSampleSize: effN,
187
- n,
188
- maxImportanceWeight: maxW,
189
- contributionCounts
190
- };
191
- }
192
- /**
193
- * Convenience: run all three estimators and return them side-by-side.
194
- * The recommended diagnostic — agreement across estimators is a much
195
- * stronger signal than any single one.
196
- */
197
- function offPolicyEstimateAll(trajectories, opts = {}) {
198
- return {
199
- ips: inverseProbabilityWeighting(trajectories, opts),
200
- snips: selfNormalizedImportanceWeighting(trajectories, opts),
201
- dr: doublyRobust(trajectories, opts)
202
- };
203
- }
204
- function zeroEstimate() {
205
- return {
206
- value: 0,
207
- standardError: 0,
208
- effectiveSampleSize: 0,
209
- n: 0,
210
- maxImportanceWeight: 0
211
- };
212
- }
213
- function clamp(x, lo, hi) {
214
- if (!Number.isFinite(x)) return lo;
215
- return Math.max(lo, Math.min(hi, x));
216
- }
217
- //#endregion
218
- export { selfNormalizedImportanceWeighting as i, inverseProbabilityWeighting as n, offPolicyEstimateAll as r, doublyRobust as t };
219
-
220
- //# sourceMappingURL=off-policy-DvgzvtIx.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"off-policy-DvgzvtIx.js","names":[],"sources":["../src/rl/off-policy.ts"],"sourcesContent":["/**\n * Off-policy evaluation primitives.\n *\n * Standard inverse-probability-weighted (IPS), self-normalized\n * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the\n * value of a *target* policy given trajectories collected under a\n * *behavior* policy. This is the canonical RL eval task: \"we have last\n * week's runs, we changed the policy — how would the new one do without\n * re-running?\"\n *\n * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan\n * & Joachims 2015 for SNIPS) but the *application* to LLM-agent\n * evaluation needs care:\n *\n * - The \"policy\" is the (prompt, tool config, model snapshot) triple.\n * Two policies have the same probability over an action *iff* their\n * LLM call would emit the same token with the same probability —\n * which is generally unknowable without the model log-probs.\n * - For LLM agents, propensity scores must be supplied by the caller\n * (logged in the trace, recovered from token log-probs, or estimated\n * via a learned propensity model). We do NOT estimate propensity here.\n * - Doubly-robust requires two outputs from a Q-function: its prediction\n * for the logged action and its expectation under the target policy.\n * Consumers compute these with a tabular estimate, regression fit, or\n * learned reward model before constructing the trajectories.\n *\n * Bias / variance tradeoffs:\n * - IPS: unbiased; high variance for small overlap, infinite variance\n * when target has support outside behavior.\n * - SNIPS: lower variance, slight bias; usually preferred in practice.\n * - DR: doubly-robust — unbiased if either propensity OR Q-function is\n * correct. Lowest practical variance when Q is decent. Use this.\n *\n * Caveat the panel will land: on the LLM-agent setting, propensity scores\n * recovered from token log-probs are noisy, the action space is enormous,\n * and overlap is often poor. These estimators are useful but not magic;\n * complement with `replayCampaign` (exact replay where the request hashes\n * match) for high-confidence answers and OPE for the gap.\n */\n\nimport { ValidationError } from '../errors'\n\nexport interface OffPolicyTrajectory {\n /** Stable id, for traceability through the dataset. */\n runId: string\n /** Reward observed under the behavior policy (the realized outcome). */\n reward: number\n /**\n * Behavior-policy probability of the action that was taken. For LLM\n * agents this is typically `exp(sum(token_log_probs))` over the chosen\n * trajectory. Must be in (0, 1].\n */\n behaviorProb: number\n /**\n * Target-policy probability of the same action. For replay-style\n * counterfactual evaluation this is what the *new* policy would have\n * assigned to the *old* trajectory. Must be in [0, 1].\n */\n targetProb: number\n /**\n * Model-based reward prediction for the action selected by the behavior\n * policy: `Q_hat(context, loggedAction)`. Supply this together with\n * `vHatTarget` for contextual-bandit doubly-robust estimation.\n */\n qHatChosen?: number | null\n /**\n * Expected model-based reward under the target policy:\n * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.\n * Supply this together with `qHatChosen`. For an honest evaluation, both\n * values must come from a model cross-fitted or trained outside this row.\n */\n vHatTarget?: number | null\n}\n\nexport interface OffPolicyContributionCounts {\n /** Contributions using the contextual-bandit doubly-robust formula. */\n dr: number\n /** Contributions using exact IPS because no reward-model estimate was supplied. */\n ipsFallback: number\n}\n\nexport interface OffPolicyEstimate {\n /** Estimated value of the target policy. */\n value: number\n /** Standard error of the estimate. */\n standardError: number\n /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */\n effectiveSampleSize: number\n /** Number of trajectories used. */\n n: number\n /**\n * Diagnostic: maximum importance weight observed. Large values (>>10x\n * mean) are a red flag — variance is dominated by a few outliers.\n */\n maxImportanceWeight: number\n /** Populated by `doublyRobust` to expose which formula each row used. */\n contributionCounts?: OffPolicyContributionCounts\n}\n\nexport interface OffPolicyOptions {\n /**\n * Cap importance weights at this value (Ionides 2008 truncated IS) to\n * trade unbiasedness for variance reduction. Default `Infinity` (no cap).\n * Set e.g. `10` for stable estimates when the policies are close.\n */\n weightCap?: number\n /** Reward clipping range. Default `[0, 1]`. */\n rewardClip?: { low: number; high: number }\n}\n\n/**\n * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator\n * of E[reward under target policy]. Variance scales with the spread of\n * target/behavior ratios.\n */\nexport function inverseProbabilityWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n\n if (trajectories.length === 0) {\n return zeroEstimate()\n }\n\n const weights: number[] = []\n const weightedRewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `inverseProbabilityWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n weights.push(w)\n weightedRewards.push(w * r)\n if (w > maxW) maxW = w\n }\n const n = weights.length\n const value = weightedRewards.reduce((s, x) => s + x, 0) / n\n const variance = weightedRewards.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at\n * the cost of small bias (vanishing as N grows). The right default for\n * LLM-agent evaluation where overlap is often poor.\n */\nexport function selfNormalizedImportanceWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) return zeroEstimate()\n\n const weights: number[] = []\n const rewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `selfNormalizedImportanceWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n weights.push(w)\n rewards.push(clamp(t.reward, clip.low, clip.high))\n if (w > maxW) maxW = w\n }\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumWR = weights.reduce((s, w, i) => s + w * rewards[i]!, 0)\n const value = sumW === 0 ? 0 : sumWR / sumW\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n // Influence-function-based SE for SNIPS (Owen 2013, Ch. 9).\n const phi = weights.map((w, i) => w * (rewards[i]! - value))\n const variance = phi.reduce((s, x) => s + x * x, 0) / Math.max(1, sumW * sumW)\n return {\n value,\n standardError: Math.sqrt(variance),\n effectiveSampleSize: effN,\n n: trajectories.length,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).\n *\n * V_DR = (1/N) * sum_i [ v_hat_target_i\n * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]\n *\n * Unbiased if EITHER:\n * - the importance ratios are correct (IPS-style validity), OR\n * - the Q-hat function is correct (model-based validity).\n *\n * In practice both are imperfect, but the residual bias is the *product*\n * of both errors — much smaller than either alone. This is why DR is the\n * default in production OPE pipelines.\n *\n * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither\n * use the exact IPS contribution. `contributionCounts` makes the mix explicit\n * in the result.\n * Callers must cross-fit the Q-function or train it on independent rows;\n * fitting and evaluating Q on the same outcomes leaks the answer.\n */\nexport function doublyRobust(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) {\n return {\n ...zeroEstimate(),\n contributionCounts: { dr: 0, ipsFallback: 0 },\n }\n }\n\n const contributions: number[] = []\n const contributionCounts: OffPolicyContributionCounts = {\n dr: 0,\n ipsFallback: 0,\n }\n let maxW = 0\n let sumW = 0\n let sumW2 = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(`doublyRobust: behaviorProb must be > 0 (runId=${t.runId})`)\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n const rawQHatChosen = t.qHatChosen\n const rawVHatTarget = t.vHatTarget\n const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== undefined\n const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== undefined\n if (hasQHatChosen !== hasVHatTarget) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`,\n )\n }\n\n if (hasQHatChosen && hasVHatTarget) {\n if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`,\n )\n }\n const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high)\n const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high)\n contributions.push(vHatTarget + w * (r - qHatChosen))\n contributionCounts.dr += 1\n } else {\n contributions.push(w * r)\n contributionCounts.ipsFallback += 1\n }\n if (w > maxW) maxW = w\n sumW += w\n sumW2 += w * w\n }\n const n = contributions.length\n const value = contributions.reduce((s, x) => s + x, 0) / n\n const variance = contributions.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n contributionCounts,\n }\n}\n\n/**\n * Convenience: run all three estimators and return them side-by-side.\n * The recommended diagnostic — agreement across estimators is a much\n * stronger signal than any single one.\n */\nexport function offPolicyEstimateAll(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): { ips: OffPolicyEstimate; snips: OffPolicyEstimate; dr: OffPolicyEstimate } {\n return {\n ips: inverseProbabilityWeighting(trajectories, opts),\n snips: selfNormalizedImportanceWeighting(trajectories, opts),\n dr: doublyRobust(trajectories, opts),\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction zeroEstimate(): OffPolicyEstimate {\n return { value: 0, standardError: 0, effectiveSampleSize: 0, n: 0, maxImportanceWeight: 0 }\n}\n\nfunction clamp(x: number, lo: number, hi: number): number {\n if (!Number.isFinite(x)) return lo\n return Math.max(lo, Math.min(hi, x))\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmHA,SAAgB,4BACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAElD,IAAI,aAAa,WAAW,GAC1B,OAAO,aAAa;CAGtB,MAAM,UAAoB,CAAC;CAC3B,MAAM,kBAA4B,CAAC;CACnC,IAAI,OAAO;CACX,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,gEAAgE,EAAE,MAAM,EAC1E;EAEF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,MAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;EAC7C,QAAQ,KAAK,CAAC;EACd,gBAAgB,KAAK,IAAI,CAAC;EAC1B,IAAI,IAAI,MAAM,OAAO;CACvB;CACA,MAAM,IAAI,QAAQ;CAClB,MAAM,QAAQ,gBAAgB,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CAC3D,MAAM,WAAW,gBAAgB,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;CAC9F,MAAM,OAAO,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC;CAC9C,MAAM,QAAQ,QAAQ,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;CACnD,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAE9C,OAAO;EACL;EACA,eAAe,KAAK,KAAK,WAAW,CAAC;EACrC,qBAAqB;EACrB;EACA,qBAAqB;CACvB;AACF;;;;;;AAOA,SAAgB,kCACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAClD,IAAI,aAAa,WAAW,GAAG,OAAO,aAAa;CAEnD,MAAM,UAAoB,CAAC;CAC3B,MAAM,UAAoB,CAAC;CAC3B,IAAI,OAAO;CACX,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBACR,sEAAsE,EAAE,MAAM,EAChF;EAEF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,QAAQ,KAAK,CAAC;EACd,QAAQ,KAAK,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI,CAAC;EACjD,IAAI,IAAI,MAAM,OAAO;CACvB;CACA,MAAM,OAAO,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC;CAC9C,MAAM,QAAQ,QAAQ,QAAQ,GAAG,GAAG,MAAM,IAAI,IAAI,QAAQ,IAAK,CAAC;CAChE,MAAM,QAAQ,SAAS,IAAI,IAAI,QAAQ;CACvC,MAAM,QAAQ,QAAQ,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;CACnD,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAG9C,MAAM,WADM,QAAQ,KAAK,GAAG,MAAM,KAAK,QAAQ,KAAM,MAClC,CAAC,CAAC,QAAQ,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,OAAO,IAAI;CAC7E,OAAO;EACL;EACA,eAAe,KAAK,KAAK,QAAQ;EACjC,qBAAqB;EACrB,GAAG,aAAa;EAChB,qBAAqB;CACvB;AACF;;;;;;;;;;;;;;;;;;;;;AAsBA,SAAgB,aACd,cACA,OAAyB,CAAC,GACP;CACnB,MAAM,MAAM,KAAK,aAAa;CAC9B,MAAM,OAAO,KAAK,cAAc;EAAE,KAAK;EAAG,MAAM;CAAE;CAClD,IAAI,aAAa,WAAW,GAC1B,OAAO;EACL,GAAG,aAAa;EAChB,oBAAoB;GAAE,IAAI;GAAG,aAAa;EAAE;CAC9C;CAGF,MAAM,gBAA0B,CAAC;CACjC,MAAM,qBAAkD;EACtD,IAAI;EACJ,aAAa;CACf;CACA,IAAI,OAAO;CACX,IAAI,OAAO;CACX,IAAI,QAAQ;CACZ,KAAK,MAAM,KAAK,cAAc;EAC5B,IAAI,EAAE,gBAAgB,GACpB,MAAM,IAAI,gBAAgB,iDAAiD,EAAE,MAAM,EAAE;EAEvF,MAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;EACrD,MAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;EAC7C,MAAM,gBAAgB,EAAE;EACxB,MAAM,gBAAgB,EAAE;EACxB,MAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB,KAAA;EAClE,MAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB,KAAA;EAClE,IAAI,kBAAkB,eACpB,MAAM,IAAI,gBACR,4EAA4E,EAAE,MAAM,EACtF;EAGF,IAAI,iBAAiB,eAAe;GAClC,IAAI,CAAC,OAAO,SAAS,aAAa,KAAK,CAAC,OAAO,SAAS,aAAa,GACnE,MAAM,IAAI,gBACR,iEAAiE,EAAE,MAAM,EAC3E;GAEF,MAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;GAC3D,MAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;GAC3D,cAAc,KAAK,aAAa,KAAK,IAAI,WAAW;GACpD,mBAAmB,MAAM;EAC3B,OAAO;GACL,cAAc,KAAK,IAAI,CAAC;GACxB,mBAAmB,eAAe;EACpC;EACA,IAAI,IAAI,MAAM,OAAO;EACrB,QAAQ;EACR,SAAS,IAAI;CACf;CACA,MAAM,IAAI,cAAc;CACxB,MAAM,QAAQ,cAAc,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACzD,MAAM,WAAW,cAAc,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;CAC5F,MAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;CAC9C,OAAO;EACL;EACA,eAAe,KAAK,KAAK,WAAW,CAAC;EACrC,qBAAqB;EACrB;EACA,qBAAqB;EACrB;CACF;AACF;;;;;;AAOA,SAAgB,qBACd,cACA,OAAyB,CAAC,GACmD;CAC7E,OAAO;EACL,KAAK,4BAA4B,cAAc,IAAI;EACnD,OAAO,kCAAkC,cAAc,IAAI;EAC3D,IAAI,aAAa,cAAc,IAAI;CACrC;AACF;AAIA,SAAS,eAAkC;CACzC,OAAO;EAAE,OAAO;EAAG,eAAe;EAAG,qBAAqB;EAAG,GAAG;EAAG,qBAAqB;CAAE;AAC5F;AAEA,SAAS,MAAM,GAAW,IAAY,IAAoB;CACxD,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,CAAC,CAAC;AACrC"}
@@ -1,153 +0,0 @@
1
- //#region src/rl/off-policy.d.ts
2
- /**
3
- * Off-policy evaluation primitives.
4
- *
5
- * Standard inverse-probability-weighted (IPS), self-normalized
6
- * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
7
- * value of a *target* policy given trajectories collected under a
8
- * *behavior* policy. This is the canonical RL eval task: "we have last
9
- * week's runs, we changed the policy — how would the new one do without
10
- * re-running?"
11
- *
12
- * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
13
- * & Joachims 2015 for SNIPS) but the *application* to LLM-agent
14
- * evaluation needs care:
15
- *
16
- * - The "policy" is the (prompt, tool config, model snapshot) triple.
17
- * Two policies have the same probability over an action *iff* their
18
- * LLM call would emit the same token with the same probability —
19
- * which is generally unknowable without the model log-probs.
20
- * - For LLM agents, propensity scores must be supplied by the caller
21
- * (logged in the trace, recovered from token log-probs, or estimated
22
- * via a learned propensity model). We do NOT estimate propensity here.
23
- * - Doubly-robust requires two outputs from a Q-function: its prediction
24
- * for the logged action and its expectation under the target policy.
25
- * Consumers compute these with a tabular estimate, regression fit, or
26
- * learned reward model before constructing the trajectories.
27
- *
28
- * Bias / variance tradeoffs:
29
- * - IPS: unbiased; high variance for small overlap, infinite variance
30
- * when target has support outside behavior.
31
- * - SNIPS: lower variance, slight bias; usually preferred in practice.
32
- * - DR: doubly-robust — unbiased if either propensity OR Q-function is
33
- * correct. Lowest practical variance when Q is decent. Use this.
34
- *
35
- * Caveat the panel will land: on the LLM-agent setting, propensity scores
36
- * recovered from token log-probs are noisy, the action space is enormous,
37
- * and overlap is often poor. These estimators are useful but not magic;
38
- * complement with `replayCampaign` (exact replay where the request hashes
39
- * match) for high-confidence answers and OPE for the gap.
40
- */
41
- interface OffPolicyTrajectory {
42
- /** Stable id, for traceability through the dataset. */
43
- runId: string;
44
- /** Reward observed under the behavior policy (the realized outcome). */
45
- reward: number;
46
- /**
47
- * Behavior-policy probability of the action that was taken. For LLM
48
- * agents this is typically `exp(sum(token_log_probs))` over the chosen
49
- * trajectory. Must be in (0, 1].
50
- */
51
- behaviorProb: number;
52
- /**
53
- * Target-policy probability of the same action. For replay-style
54
- * counterfactual evaluation this is what the *new* policy would have
55
- * assigned to the *old* trajectory. Must be in [0, 1].
56
- */
57
- targetProb: number;
58
- /**
59
- * Model-based reward prediction for the action selected by the behavior
60
- * policy: `Q_hat(context, loggedAction)`. Supply this together with
61
- * `vHatTarget` for contextual-bandit doubly-robust estimation.
62
- */
63
- qHatChosen?: number | null;
64
- /**
65
- * Expected model-based reward under the target policy:
66
- * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.
67
- * Supply this together with `qHatChosen`. For an honest evaluation, both
68
- * values must come from a model cross-fitted or trained outside this row.
69
- */
70
- vHatTarget?: number | null;
71
- }
72
- interface OffPolicyContributionCounts {
73
- /** Contributions using the contextual-bandit doubly-robust formula. */
74
- dr: number;
75
- /** Contributions using exact IPS because no reward-model estimate was supplied. */
76
- ipsFallback: number;
77
- }
78
- interface OffPolicyEstimate {
79
- /** Estimated value of the target policy. */
80
- value: number;
81
- /** Standard error of the estimate. */
82
- standardError: number;
83
- /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
84
- effectiveSampleSize: number;
85
- /** Number of trajectories used. */
86
- n: number;
87
- /**
88
- * Diagnostic: maximum importance weight observed. Large values (>>10x
89
- * mean) are a red flag — variance is dominated by a few outliers.
90
- */
91
- maxImportanceWeight: number;
92
- /** Populated by `doublyRobust` to expose which formula each row used. */
93
- contributionCounts?: OffPolicyContributionCounts;
94
- }
95
- interface OffPolicyOptions {
96
- /**
97
- * Cap importance weights at this value (Ionides 2008 truncated IS) to
98
- * trade unbiasedness for variance reduction. Default `Infinity` (no cap).
99
- * Set e.g. `10` for stable estimates when the policies are close.
100
- */
101
- weightCap?: number;
102
- /** Reward clipping range. Default `[0, 1]`. */
103
- rewardClip?: {
104
- low: number;
105
- high: number;
106
- };
107
- }
108
- /**
109
- * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator
110
- * of E[reward under target policy]. Variance scales with the spread of
111
- * target/behavior ratios.
112
- */
113
- declare function inverseProbabilityWeighting(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
114
- /**
115
- * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at
116
- * the cost of small bias (vanishing as N grows). The right default for
117
- * LLM-agent evaluation where overlap is often poor.
118
- */
119
- declare function selfNormalizedImportanceWeighting(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
120
- /**
121
- * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).
122
- *
123
- * V_DR = (1/N) * sum_i [ v_hat_target_i
124
- * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]
125
- *
126
- * Unbiased if EITHER:
127
- * - the importance ratios are correct (IPS-style validity), OR
128
- * - the Q-hat function is correct (model-based validity).
129
- *
130
- * In practice both are imperfect, but the residual bias is the *product*
131
- * of both errors — much smaller than either alone. This is why DR is the
132
- * default in production OPE pipelines.
133
- *
134
- * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither
135
- * use the exact IPS contribution. `contributionCounts` makes the mix explicit
136
- * in the result.
137
- * Callers must cross-fit the Q-function or train it on independent rows;
138
- * fitting and evaluating Q on the same outcomes leaks the answer.
139
- */
140
- declare function doublyRobust(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
141
- /**
142
- * Convenience: run all three estimators and return them side-by-side.
143
- * The recommended diagnostic — agreement across estimators is a much
144
- * stronger signal than any single one.
145
- */
146
- declare function offPolicyEstimateAll(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): {
147
- ips: OffPolicyEstimate;
148
- snips: OffPolicyEstimate;
149
- dr: OffPolicyEstimate;
150
- };
151
- //#endregion
152
- export { doublyRobust as a, selfNormalizedImportanceWeighting as c, OffPolicyTrajectory as i, OffPolicyEstimate as n, inverseProbabilityWeighting as o, OffPolicyOptions as r, offPolicyEstimateAll as s, OffPolicyContributionCounts as t };
153
- //# sourceMappingURL=off-policy-mskQw8Mb.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"off-policy-mskQw8Mb.d.ts","names":[],"sources":["../src/rl/off-policy.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;UA0CiB;;EAEf;;EAEA;;;;;;EAMA;;;;;;EAMA;;;;;;EAMA;;;;;;;EAOA;;UAGe;;EAEf;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;EAKA;;EAEA,qBAAqB;;UAGN;;;;;;EAMf;;EAEA;IAAe;IAAa;;;;;;;;iBAQd,4BACd,cAAc,uBACd,OAAM,mBACL;;;;;;iBA4Ca,kCACd,cAAc,uBACd,OAAM,mBACL;;;;;;;;;;;;;;;;;;;;;iBAwDa,aACd,cAAc,uBACd,OAAM,mBACL;;;;;;iBAuEa,qBACd,cAAc,uBACd,OAAM;EACH,KAAK;EAAmB,OAAO;EAAmB,IAAI"}
@@ -1,96 +0,0 @@
1
- //#region src/pre-registration.ts
2
- /**
3
- * Deterministic JSON canonicalization — sort object keys recursively.
4
- *
5
- * Two semantically-equal objects produce byte-identical canonicalized output;
6
- * this is what makes a content-hash stable across encoders, key insertion
7
- * orders, and runtime versions. Exported for any consumer that needs the same
8
- * canonicalization guarantee outside the manifest-signing path (e.g., signing
9
- * an artifact bundle, hashing a dataset version, etc.).
10
- */
11
- function canonicalize(v) {
12
- if (v === null || typeof v !== "object") return v;
13
- if (Array.isArray(v)) return v.map(canonicalize);
14
- const keys = Object.keys(v).sort();
15
- const out = {};
16
- for (const k of keys) out[k] = canonicalize(v[k]);
17
- return out;
18
- }
19
- /**
20
- * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
21
- *
22
- * The same primitive `signManifest` and `verifyManifest` are built on, exposed
23
- * directly so consumers signing arbitrary structured content (artifact bundles,
24
- * production packets, dataset manifests, etc.) don't have to re-derive
25
- * canonicalize+sha256 from scratch.
26
- *
27
- * Stable across:
28
- * - object key insertion order (canonicalization sorts keys recursively)
29
- * - encoder choice (UTF-8 via TextEncoder, fixed)
30
- * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
31
- *
32
- * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
33
- * which takes a string input and returns a truncated 12-char prompt id.
34
- * Use `hashJson` when you mean "canonicalize then hash."
35
- *
36
- * @example
37
- * const hash = await hashJson({ id: '1', kind: 'spec' })
38
- * // 'a3f1...' (64 hex chars)
39
- */
40
- async function hashJson(obj) {
41
- const canonical = canonicalize(obj);
42
- const bytes = new TextEncoder().encode(JSON.stringify(canonical));
43
- const digest = await globalThis.crypto.subtle.digest("SHA-256", bytes);
44
- return Array.from(new Uint8Array(digest)).map((b) => b.toString(16).padStart(2, "0")).join("");
45
- }
46
- /**
47
- * Sign a manifest with a SHA-256 content hash.
48
- *
49
- * The hash covers the canonicalized manifest with the `contentHash`
50
- * and `algo` fields stripped; this lets verifiers re-sign the rest and
51
- * compare. Returned manifest always carries `algo: 'sha256-content'`
52
- * so downstream consumers can identify the scheme; manifests without
53
- * `algo` still verify because it is stripped before hashing on both sides.
54
- */
55
- async function signManifest(m) {
56
- const hash = await hashJson(m);
57
- return {
58
- ...m,
59
- contentHash: hash,
60
- algo: "sha256-content"
61
- };
62
- }
63
- /**
64
- * Verify that a signed manifest has not been tampered with.
65
- *
66
- * Strips `contentHash` and `algo` before re-signing so manifests without
67
- * `algo` verify identically to ones that carry it.
68
- */
69
- async function verifyManifest(m) {
70
- const { contentHash, algo: _algo, ...rest } = m;
71
- return (await signManifest(rest)).contentHash === contentHash;
72
- }
73
- /**
74
- * Evaluate a pre-registered hypothesis against observed results.
75
- * Mechanical — no re-interpretation permitted.
76
- */
77
- async function evaluateHypothesis(manifest, observed) {
78
- if (!await verifyManifest(manifest)) throw new Error("evaluateHypothesis: manifest content hash mismatch (tampered)");
79
- const reasons = [];
80
- if (!(manifest.direction === "increase" ? observed.effect > 0 : observed.effect < 0)) reasons.push("wrong_direction");
81
- if (Math.abs(observed.effect) < manifest.minEffect) reasons.push("effect_too_small");
82
- if (observed.pValue >= manifest.alpha) reasons.push("not_significant");
83
- if (observed.n < manifest.preRegisteredN) reasons.push("undersampled");
84
- return {
85
- manifest,
86
- observedN: observed.n,
87
- observedEffect: observed.effect,
88
- observedPValue: observed.pValue,
89
- confirmed: reasons.length === 0,
90
- rejectionReasons: reasons
91
- };
92
- }
93
- //#endregion
94
- export { verifyManifest as a, signManifest as i, evaluateHypothesis as n, hashJson as r, canonicalize as t };
95
-
96
- //# sourceMappingURL=pre-registration-DakwTRXk.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"pre-registration-DakwTRXk.js","names":[],"sources":["../src/pre-registration.ts"],"sourcesContent":["/**\n * Pre-registered hypotheses — declare what you're testing BEFORE the\n * run, check it AFTER. Prevents p-hacking, optional stopping, and the\n * \"we ran until it looked good\" failure mode.\n *\n * Manifest is a plain JSON-friendly object. Sign it with a content hash\n * + timestamp; the registered record becomes immutable. Post-run,\n * evaluate the manifest against observed results — the library refuses\n * to let you re-interpret a different metric as the declared one.\n */\n\nexport interface HypothesisManifest {\n id: string\n /** Human prose — goes into the audit trail. */\n hypothesis: string\n /** Metric the hypothesis claims to move. */\n metric: string\n /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */\n direction: 'increase' | 'decrease'\n /** Minimum effect size to count (same units as the metric). */\n minEffect: number\n /** Alpha threshold. */\n alpha: number\n /** Target statistical power at which sample size was pre-computed. */\n power: number\n /** Declared N per arm before running. */\n preRegisteredN: number\n /** ISO8601 timestamp the manifest was registered. */\n registeredAt: string\n /** Optional identifiers to tie into the trace corpus. */\n baselineLabel?: string\n candidateLabel?: string\n}\n\n/**\n * Identifier for the hashing scheme used to produce `contentHash`.\n *\n * `'sha256-content'` — sha256 hex over the canonicalized manifest with\n * the `contentHash` and `algo` fields stripped. Held as a string union\n * so future schemes can be added without breaking parsers; SignedManifest\n * values without `algo` deserialize cleanly because the field is optional.\n */\nexport type SignedManifestAlgo = 'sha256-content'\n\nexport interface SignedManifest extends HypothesisManifest {\n /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */\n contentHash: string\n /**\n * Algorithm string describing how `contentHash` was produced.\n *\n * Optional on the type so serialized manifests without it still parse,\n * but ALWAYS populated by {@link signManifest}. Consumers that want to\n * enforce a known algorithm should reject manifests where this field\n * is missing or unrecognized.\n */\n algo?: SignedManifestAlgo\n}\n\nexport interface HypothesisResult {\n manifest: SignedManifest\n observedN: number\n observedEffect: number\n observedPValue: number\n /** True iff the observed effect hits the pre-declared direction with\n * magnitude ≥ minEffect AND p < alpha. */\n confirmed: boolean\n /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */\n rejectionReasons: Array<\n 'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'\n >\n notes?: string\n}\n\n/**\n * Deterministic JSON canonicalization — sort object keys recursively.\n *\n * Two semantically-equal objects produce byte-identical canonicalized output;\n * this is what makes a content-hash stable across encoders, key insertion\n * orders, and runtime versions. Exported for any consumer that needs the same\n * canonicalization guarantee outside the manifest-signing path (e.g., signing\n * an artifact bundle, hashing a dataset version, etc.).\n */\nexport function canonicalize(v: unknown): unknown {\n if (v === null || typeof v !== 'object') return v\n if (Array.isArray(v)) return v.map(canonicalize)\n const keys = Object.keys(v as Record<string, unknown>).sort()\n const out: Record<string, unknown> = {}\n for (const k of keys) out[k] = canonicalize((v as Record<string, unknown>)[k])\n return out\n}\n\n/**\n * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.\n *\n * The same primitive `signManifest` and `verifyManifest` are built on, exposed\n * directly so consumers signing arbitrary structured content (artifact bundles,\n * production packets, dataset manifests, etc.) don't have to re-derive\n * canonicalize+sha256 from scratch.\n *\n * Stable across:\n * - object key insertion order (canonicalization sorts keys recursively)\n * - encoder choice (UTF-8 via TextEncoder, fixed)\n * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)\n *\n * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,\n * which takes a string input and returns a truncated 12-char prompt id.\n * Use `hashJson` when you mean \"canonicalize then hash.\"\n *\n * @example\n * const hash = await hashJson({ id: '1', kind: 'spec' })\n * // 'a3f1...' (64 hex chars)\n */\nexport async function hashJson<T>(obj: T): Promise<string> {\n const canonical = canonicalize(obj)\n const bytes = new TextEncoder().encode(JSON.stringify(canonical))\n const digest = await globalThis.crypto.subtle.digest('SHA-256', bytes)\n return Array.from(new Uint8Array(digest))\n .map((b) => b.toString(16).padStart(2, '0'))\n .join('')\n}\n\n/**\n * Sign a manifest with a SHA-256 content hash.\n *\n * The hash covers the canonicalized manifest with the `contentHash`\n * and `algo` fields stripped; this lets verifiers re-sign the rest and\n * compare. Returned manifest always carries `algo: 'sha256-content'`\n * so downstream consumers can identify the scheme; manifests without\n * `algo` still verify because it is stripped before hashing on both sides.\n */\nexport async function signManifest(m: HypothesisManifest): Promise<SignedManifest> {\n const hash = await hashJson(m)\n return { ...m, contentHash: hash, algo: 'sha256-content' }\n}\n\n/**\n * Verify that a signed manifest has not been tampered with.\n *\n * Strips `contentHash` and `algo` before re-signing so manifests without\n * `algo` verify identically to ones that carry it.\n */\nexport async function verifyManifest(m: SignedManifest): Promise<boolean> {\n const { contentHash, algo: _algo, ...rest } = m\n void _algo\n const resigned = await signManifest(rest)\n return resigned.contentHash === contentHash\n}\n\n/**\n * Evaluate a pre-registered hypothesis against observed results.\n * Mechanical — no re-interpretation permitted.\n */\nexport async function evaluateHypothesis(\n manifest: SignedManifest,\n observed: { n: number; effect: number; pValue: number },\n): Promise<HypothesisResult> {\n if (!(await verifyManifest(manifest))) {\n throw new Error('evaluateHypothesis: manifest content hash mismatch (tampered)')\n }\n const reasons: HypothesisResult['rejectionReasons'] = []\n const directionOk = manifest.direction === 'increase' ? observed.effect > 0 : observed.effect < 0\n if (!directionOk) reasons.push('wrong_direction')\n if (Math.abs(observed.effect) < manifest.minEffect) reasons.push('effect_too_small')\n if (observed.pValue >= manifest.alpha) reasons.push('not_significant')\n if (observed.n < manifest.preRegisteredN) reasons.push('undersampled')\n return {\n manifest,\n observedN: observed.n,\n observedEffect: observed.effect,\n observedPValue: observed.pValue,\n confirmed: reasons.length === 0,\n rejectionReasons: reasons,\n }\n}\n"],"mappings":";;;;;;;;;;AAkFA,SAAgB,aAAa,GAAqB;CAChD,IAAI,MAAM,QAAQ,OAAO,MAAM,UAAU,OAAO;CAChD,IAAI,MAAM,QAAQ,CAAC,GAAG,OAAO,EAAE,IAAI,YAAY;CAC/C,MAAM,OAAO,OAAO,KAAK,CAA4B,CAAC,CAAC,KAAK;CAC5D,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,KAAK,MAAM,IAAI,KAAK,aAAc,EAA8B,EAAE;CAC7E,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;AAuBA,eAAsB,SAAY,KAAyB;CACzD,MAAM,YAAY,aAAa,GAAG;CAClC,MAAM,QAAQ,IAAI,YAAY,CAAC,CAAC,OAAO,KAAK,UAAU,SAAS,CAAC;CAChE,MAAM,SAAS,MAAM,WAAW,OAAO,OAAO,OAAO,WAAW,KAAK;CACrE,OAAO,MAAM,KAAK,IAAI,WAAW,MAAM,CAAC,CAAC,CACtC,KAAK,MAAM,EAAE,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG,CAAC,CAAC,CAC3C,KAAK,EAAE;AACZ;;;;;;;;;;AAWA,eAAsB,aAAa,GAAgD;CACjF,MAAM,OAAO,MAAM,SAAS,CAAC;CAC7B,OAAO;EAAE,GAAG;EAAG,aAAa;EAAM,MAAM;CAAiB;AAC3D;;;;;;;AAQA,eAAsB,eAAe,GAAqC;CACxE,MAAM,EAAE,aAAa,MAAM,OAAO,GAAG,SAAS;CAG9C,QAAO,MADgB,aAAa,IAAI,EAAA,CACxB,gBAAgB;AAClC;;;;;AAMA,eAAsB,mBACpB,UACA,UAC2B;CAC3B,IAAI,CAAE,MAAM,eAAe,QAAQ,GACjC,MAAM,IAAI,MAAM,+DAA+D;CAEjF,MAAM,UAAgD,CAAC;CAEvD,IAAI,EADgB,SAAS,cAAc,aAAa,SAAS,SAAS,IAAI,SAAS,SAAS,IAC9E,QAAQ,KAAK,iBAAiB;CAChD,IAAI,KAAK,IAAI,SAAS,MAAM,IAAI,SAAS,WAAW,QAAQ,KAAK,kBAAkB;CACnF,IAAI,SAAS,UAAU,SAAS,OAAO,QAAQ,KAAK,iBAAiB;CACrE,IAAI,SAAS,IAAI,SAAS,gBAAgB,QAAQ,KAAK,cAAc;CACrE,OAAO;EACL;EACA,WAAW,SAAS;EACpB,gBAAgB,SAAS;EACzB,gBAAgB,SAAS;EACzB,WAAW,QAAQ,WAAW;EAC9B,kBAAkB;CACpB;AACF"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"profile-cell.js","names":[],"sources":["../src/agent-profile-cell.ts"],"sourcesContent":["import type { AgentProfile } from '@tangle-network/agent-interface'\nimport { ValidationError } from './errors'\nimport { hashJson } from './pre-registration'\n\nexport type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1'\n\nexport type AgentProfileJsonObject = { [key: string]: AgentProfileJson }\n\nexport type AgentProfileJson =\n | string\n | number\n | boolean\n | null\n | AgentProfileJson[]\n | AgentProfileJsonObject\n\nexport type AgentProfileDimensionValue = string | number | boolean | null\n\nexport interface AgentProfileSource {\n /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */\n kind: string\n /** sha256 over the canonical source profile object. */\n hash: string\n}\n\nexport interface AgentProfileSourceInput {\n kind: string\n /** Precomputed sha256 for callers that already sign their profile artifact. */\n hash?: string\n /** Full canonical runtime profile; hashed and then discarded from the cell. */\n profile?: AgentProfileJson\n}\n\nexport interface AgentProfileHarness {\n id: string\n version?: string\n hash?: string\n}\n\nexport interface AgentProfileCellInput {\n profileId: string\n sourceProfile: AgentProfileSourceInput\n harness?: AgentProfileHarness\n model?: string\n promptHash?: string\n dimensions?: Record<string, AgentProfileDimensionValue>\n}\n\nexport interface AgentProfileCell {\n schemaVersion: AgentProfileCellSchemaVersion\n cellId: string\n profileId: string\n sourceProfile: AgentProfileSource\n harness?: AgentProfileHarness\n model?: string\n promptHash?: string\n dimensions?: Record<string, AgentProfileDimensionValue>\n}\n\nexport class AgentProfileCellValidationError extends ValidationError {\n readonly path: string\n constructor(message: string, path = '') {\n super(path ? `${message} (at ${path})` : message)\n this.path = path\n }\n}\n\nconst SHA256_HEX = /^[0-9a-f]{64}$/\nconst CELL_ID = /^agent-profile-cell:sha256:[0-9a-f]{64}$/\n\nexport async function buildAgentProfileCell(\n input: AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n const material = await normalizeAgentProfileCellInput(input)\n const cellId = `agent-profile-cell:sha256:${await hashJson(material)}`\n return { ...material, cellId }\n}\n\nexport function agentProfileCellHashMaterial(\n cell: AgentProfileCell,\n): Omit<AgentProfileCell, 'cellId'> {\n const { cellId: _cellId, ...material } = cell\n void _cellId\n return normalizeAgentProfileCell(material)\n}\n\n/**\n * Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material fields, confirming the record has not been tampered with.\n */\nexport async function verifyAgentProfileCell(cell: AgentProfileCell): Promise<boolean> {\n validateAgentProfileCell(cell)\n return (\n cell.cellId ===\n `agent-profile-cell:sha256:${await hashJson(agentProfileCellHashMaterial(cell))}`\n )\n}\n\nexport function validateAgentProfileCell(input: unknown): AgentProfileCell {\n if (input === null || typeof input !== 'object') {\n throw new AgentProfileCellValidationError('expected object')\n }\n const obj = input as Record<string, unknown>\n expectLiteral(obj.schemaVersion, 'agent-profile-cell/v1', 'schemaVersion')\n if (typeof obj.cellId !== 'string' || !CELL_ID.test(obj.cellId)) {\n throw new AgentProfileCellValidationError(\n 'cellId must match agent-profile-cell:sha256:<64 lowercase hex chars>',\n 'cellId',\n )\n }\n expectString(obj.profileId, 'profileId')\n validateSource(obj.sourceProfile, 'sourceProfile')\n if (obj.harness !== undefined) validateHarness(obj.harness, 'harness')\n if (obj.model !== undefined) expectString(obj.model, 'model')\n if (obj.promptHash !== undefined) expectString(obj.promptHash, 'promptHash')\n if (obj.dimensions !== undefined) validateDimensions(obj.dimensions, 'dimensions')\n return input as AgentProfileCell\n}\n\nexport function requireAgentProfileCell(record: {\n runId: string\n agentProfile?: AgentProfileCell\n}): AgentProfileCell {\n if (!record.agentProfile) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" is missing agentProfile; profile-cell grouping requires explicit profile identity`,\n 'agentProfile',\n )\n }\n return validateAgentProfileCell(record.agentProfile)\n}\n\nexport function agentProfileCellKey(record: {\n runId: string\n agentProfile?: AgentProfileCell\n}): string {\n return requireAgentProfileCell(record).cellId\n}\n\nexport async function assertRunAgentProfileCell(record: {\n runId: string\n model: string\n promptHash: string\n agentProfile?: AgentProfileCell\n}): Promise<AgentProfileCell> {\n const profile = requireAgentProfileCell(record)\n if (!(await verifyAgentProfileCell(profile))) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" has an agentProfile.cellId that does not match its content`,\n 'agentProfile.cellId',\n )\n }\n if (profile.model !== undefined && profile.model !== record.model) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" agentProfile.model \"${profile.model}\" does not match model \"${record.model}\"`,\n 'agentProfile.model',\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== record.promptHash) {\n throw new AgentProfileCellValidationError(\n `run \"${record.runId}\" agentProfile.promptHash \"${profile.promptHash}\" does not match promptHash \"${record.promptHash}\"`,\n 'agentProfile.promptHash',\n )\n }\n return profile\n}\n\nexport function groupRunsByAgentProfileCell<\n T extends { runId: string; agentProfile?: AgentProfileCell },\n>(records: readonly T[]): Map<string, T[]> {\n const groups = new Map<string, T[]>()\n for (const record of records) {\n const key = agentProfileCellKey(record)\n const bucket = groups.get(key)\n if (bucket) bucket.push(record)\n else groups.set(key, [record])\n }\n return groups\n}\n\nasync function normalizeAgentProfileCellInput(\n input: AgentProfileCellInput,\n): Promise<Omit<AgentProfileCell, 'cellId'>> {\n return normalizeAgentProfileCell({\n schemaVersion: 'agent-profile-cell/v1',\n profileId: input.profileId,\n sourceProfile: await normalizeSourceInput(input.sourceProfile),\n harness: input.harness,\n model: input.model,\n promptHash: input.promptHash,\n dimensions: input.dimensions,\n })\n}\n\nfunction normalizeAgentProfileCell(\n input: Omit<AgentProfileCell, 'cellId'>,\n): Omit<AgentProfileCell, 'cellId'> {\n return compactObject({\n schemaVersion: 'agent-profile-cell/v1' as const,\n profileId: requireNonEmpty(input.profileId, 'profileId'),\n sourceProfile: normalizeSource(input.sourceProfile),\n harness: input.harness ? normalizeHarness(input.harness, 'harness') : undefined,\n model: optionalNonEmpty(input.model, 'model'),\n promptHash: optionalNonEmpty(input.promptHash, 'promptHash'),\n dimensions: input.dimensions\n ? nonEmptyRecord(normalizeDimensions(input.dimensions))\n : undefined,\n })\n}\n\nasync function normalizeSourceInput(input: AgentProfileSourceInput): Promise<AgentProfileSource> {\n const kind = requireNonEmpty(input.kind, 'sourceProfile.kind')\n if (input.hash !== undefined && input.profile !== undefined) {\n throw new AgentProfileCellValidationError(\n 'sourceProfile must provide either hash or profile, not both',\n 'sourceProfile',\n )\n }\n if (input.hash !== undefined) {\n return { kind, hash: requireSha256Hex(input.hash, 'sourceProfile.hash') }\n }\n if (input.profile === undefined) {\n throw new AgentProfileCellValidationError(\n 'sourceProfile must provide hash or profile',\n 'sourceProfile',\n )\n }\n assertJson(input.profile, 'sourceProfile.profile')\n return { kind, hash: await hashJson(input.profile) }\n}\n\nfunction normalizeSource(input: AgentProfileSource): AgentProfileSource {\n return {\n kind: requireNonEmpty(input.kind, 'sourceProfile.kind'),\n hash: requireSha256Hex(input.hash, 'sourceProfile.hash'),\n }\n}\n\nfunction normalizeHarness(input: AgentProfileHarness, path: string): AgentProfileHarness {\n return compactObject({\n id: requireNonEmpty(input.id, `${path}.id`),\n version: optionalNonEmpty(input.version, `${path}.version`),\n hash: optionalNonEmpty(input.hash, `${path}.hash`),\n })\n}\n\nfunction normalizeDimensions(\n input: Record<string, AgentProfileDimensionValue>,\n): Record<string, AgentProfileDimensionValue> {\n const out: Record<string, AgentProfileDimensionValue> = {}\n for (const key of Object.keys(input).sort()) {\n const value = input[key]\n requireNonEmpty(key, 'dimensions.<key>')\n if (\n value !== null &&\n typeof value !== 'string' &&\n typeof value !== 'number' &&\n typeof value !== 'boolean'\n ) {\n throw new AgentProfileCellValidationError(\n 'expected primitive dimension value',\n `dimensions.${key}`,\n )\n }\n if (typeof value === 'number' && !Number.isFinite(value)) {\n throw new AgentProfileCellValidationError('expected finite number', `dimensions.${key}`)\n }\n out[key] = value\n }\n return out\n}\n\nfunction compactObject<T extends Record<string, unknown>>(input: T): T {\n const out: Record<string, unknown> = {}\n for (const [key, value] of Object.entries(input)) {\n if (value !== undefined) out[key] = value\n }\n return out as T\n}\n\nfunction nonEmptyRecord<T extends Record<string, unknown>>(input: T): T | undefined {\n return Object.keys(input).length > 0 ? input : undefined\n}\n\nfunction validateSource(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) {\n throw new AgentProfileCellValidationError('expected object', path)\n }\n const rec = value as Record<string, unknown>\n expectString(rec.kind, `${path}.kind`)\n requireSha256Hex(rec.hash, `${path}.hash`)\n}\n\nfunction validateHarness(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) {\n throw new AgentProfileCellValidationError('expected object', path)\n }\n const rec = value as Record<string, unknown>\n expectString(rec.id, `${path}.id`)\n if (rec.version !== undefined) expectString(rec.version, `${path}.version`)\n if (rec.hash !== undefined) expectString(rec.hash, `${path}.hash`)\n}\n\nfunction validateDimensions(value: unknown, path: string): void {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) {\n throw new AgentProfileCellValidationError('expected object', path)\n }\n normalizeDimensions(value as Record<string, AgentProfileDimensionValue>)\n}\n\nfunction assertJson(value: AgentProfileJson, path: string): void {\n if (value === null) return\n const type = typeof value\n if (type === 'string' || type === 'boolean') return\n if (type === 'number') {\n if (!Number.isFinite(value)) {\n throw new AgentProfileCellValidationError('expected finite number', path)\n }\n return\n }\n if (Array.isArray(value)) {\n value.forEach((item, index) => {\n assertJson(item, `${path}[${index}]`)\n })\n return\n }\n if (type === 'object') {\n for (const [key, nested] of Object.entries(value)) {\n requireNonEmpty(key, `${path}.<key>`)\n assertJson(nested, `${path}.${key}`)\n }\n return\n }\n throw new AgentProfileCellValidationError('expected JSON-compatible value', path)\n}\n\nfunction expectLiteral(value: unknown, expected: string, path: string): void {\n if (value !== expected) {\n throw new AgentProfileCellValidationError(`expected ${expected}`, path)\n }\n}\n\nfunction expectString(value: unknown, path: string): void {\n if (typeof value !== 'string' || value.length === 0) {\n throw new AgentProfileCellValidationError('expected non-empty string', path)\n }\n}\n\nfunction requireNonEmpty(value: string, path: string): string {\n if (typeof value !== 'string' || value.length === 0) {\n throw new AgentProfileCellValidationError('expected non-empty string', path)\n }\n return value\n}\n\nfunction optionalNonEmpty(value: string | undefined, path: string): string | undefined {\n if (value === undefined) return undefined\n return requireNonEmpty(value, path)\n}\n\nfunction requireSha256Hex(value: unknown, path: string): string {\n if (typeof value !== 'string' || !SHA256_HEX.test(value)) {\n throw new AgentProfileCellValidationError('expected 64 lowercase sha256 hex chars', path)\n }\n return value\n}\n\n// ── Consumer helpers ─────────────────────────────────────────────────\n//\n// Boilerplate every product consuming `buildAgentProfileCell` used to duplicate:\n//\n// 1. A `JSON.parse(JSON.stringify(value))` helper that canonicalizes an\n// arbitrary `@tangle-network/agent-interface` `AgentProfile` into the recursive\n// `AgentProfileJson` shape, with a fail-loud error when the profile\n// is not JSON-serializable.\n//\n// 2. The magic string `'agent-interface-profile'` for `sourceProfile.kind`.\n//\n// Both belong here so the cross-product cell join (same canonical profile\n// hashes to the same `sourceProfile.hash` across products) is enforced by\n// the type system, not by every consumer remembering to do it right.\n// See blueprint-agent issue tangle-network/agent-eval#82.\n\n/** Canonical `sourceProfile.kind` values. Two products fingerprinting the\n * same canonical profile MUST use the same kind for their cells to share\n * `sourceProfile.hash`. Extend rather than create new strings — adding a\n * new kind is a deliberate cross-product schema change. */\nexport const AGENT_PROFILE_KINDS = {\n /** A profile declared via `defineAgentProfile(...)` from\n * `@tangle-network/agent-interface`. The default kind for router-backed\n * and sandbox-backed products. */\n AGENT_INTERFACE_PROFILE: 'agent-interface-profile',\n} as const\n\nexport type AgentProfileKind = (typeof AGENT_PROFILE_KINDS)[keyof typeof AGENT_PROFILE_KINDS]\n\n/** Canonicalize an arbitrary value into `AgentProfileJson` by JSON\n * round-trip. Throws when the value contains anything not representable\n * as JSON (functions, BigInt, cycles) — non-portable profiles fail loud\n * rather than silently dropping fields. */\nexport function toAgentProfileJson(value: unknown): AgentProfileJson {\n let serialized: string | undefined\n try {\n serialized = JSON.stringify(value)\n } catch (err) {\n throw new AgentProfileCellValidationError(\n `agent profile must be JSON-serializable: ${err instanceof Error ? err.message : String(err)}`,\n 'sourceProfile.profile',\n )\n }\n if (serialized === undefined) {\n throw new AgentProfileCellValidationError(\n 'agent profile must be JSON-serializable (got undefined after JSON.stringify)',\n 'sourceProfile.profile',\n )\n }\n return JSON.parse(serialized) as AgentProfileJson\n}\n\n/** Canonical AgentProfile shape required when deriving a stable cell id. */\nexport type AgentInterfaceProfileLike = AgentProfile & { name: string; version: string }\n\n/** Higher-level helper that hard-codes the canonical\n * `agent-interface-profile` kind plus the JSON canonicalization. Equivalent\n * to calling `buildAgentProfileCell` with `profileId = \\`${name}@${version}\\``\n * and `sourceProfile = { kind: AGENT_INTERFACE_PROFILE, profile: <round-tripped> }`.\n *\n * Use this from any product consuming an agent-interface `AgentProfile`; the\n * manual `buildAgentProfileCell` call is reserved for advanced cases\n * (custom kinds, pre-computed source hashes, alternate profileId\n * conventions). */\nexport async function buildAgentInterfaceProfileCell(\n profile: AgentInterfaceProfileLike,\n input: Omit<AgentProfileCellInput, 'profileId' | 'sourceProfile'>,\n): Promise<AgentProfileCell> {\n if (!profile || typeof profile !== 'object') {\n throw new AgentProfileCellValidationError('AgentProfile must be an object', 'profile')\n }\n if (typeof profile.name !== 'string' || profile.name.length === 0) {\n throw new AgentProfileCellValidationError(\n 'AgentProfile must have a non-empty `name`',\n 'profile.name',\n )\n }\n if (typeof profile.version !== 'string' || profile.version.length === 0) {\n throw new AgentProfileCellValidationError(\n 'AgentProfile must have a non-empty `version`',\n 'profile.version',\n )\n }\n return buildAgentProfileCell({\n ...input,\n profileId: `${profile.name}@${profile.version}`,\n sourceProfile: {\n kind: AGENT_PROFILE_KINDS.AGENT_INTERFACE_PROFILE,\n profile: toAgentProfileJson(profile),\n },\n })\n}\n"],"mappings":";;;AA2DA,IAAa,kCAAb,cAAqD,gBAAgB;CACnE;CACA,YAAY,SAAiB,OAAO,IAAI;EACtC,MAAM,OAAO,GAAG,QAAQ,OAAO,KAAK,KAAK,OAAO;EAChD,KAAK,OAAO;CACd;AACF;AAEA,MAAM,aAAa;AACnB,MAAM,UAAU;AAEhB,eAAsB,sBACpB,OAC2B;CAC3B,MAAM,WAAW,MAAM,+BAA+B,KAAK;CAC3D,MAAM,SAAS,6BAA6B,MAAM,SAAS,QAAQ;CACnE,OAAO;EAAE,GAAG;EAAU;CAAO;AAC/B;AAEA,SAAgB,6BACd,MACkC;CAClC,MAAM,EAAE,QAAQ,SAAS,GAAG,aAAa;CAEzC,OAAO,0BAA0B,QAAQ;AAC3C;;;;AAKA,eAAsB,uBAAuB,MAA0C;CACrF,yBAAyB,IAAI;CAC7B,OACE,KAAK,WACL,6BAA6B,MAAM,SAAS,6BAA6B,IAAI,CAAC;AAElF;AAEA,SAAgB,yBAAyB,OAAkC;CACzE,IAAI,UAAU,QAAQ,OAAO,UAAU,UACrC,MAAM,IAAI,gCAAgC,iBAAiB;CAE7D,MAAM,MAAM;CACZ,cAAc,IAAI,eAAe,yBAAyB,eAAe;CACzE,IAAI,OAAO,IAAI,WAAW,YAAY,CAAC,QAAQ,KAAK,IAAI,MAAM,GAC5D,MAAM,IAAI,gCACR,wEACA,QACF;CAEF,aAAa,IAAI,WAAW,WAAW;CACvC,eAAe,IAAI,eAAe,eAAe;CACjD,IAAI,IAAI,YAAY,KAAA,GAAW,gBAAgB,IAAI,SAAS,SAAS;CACrE,IAAI,IAAI,UAAU,KAAA,GAAW,aAAa,IAAI,OAAO,OAAO;CAC5D,IAAI,IAAI,eAAe,KAAA,GAAW,aAAa,IAAI,YAAY,YAAY;CAC3E,IAAI,IAAI,eAAe,KAAA,GAAW,mBAAmB,IAAI,YAAY,YAAY;CACjF,OAAO;AACT;AAEA,SAAgB,wBAAwB,QAGnB;CACnB,IAAI,CAAC,OAAO,cACV,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,sFACrB,cACF;CAEF,OAAO,yBAAyB,OAAO,YAAY;AACrD;AAEA,SAAgB,oBAAoB,QAGzB;CACT,OAAO,wBAAwB,MAAM,CAAC,CAAC;AACzC;AAEA,eAAsB,0BAA0B,QAKlB;CAC5B,MAAM,UAAU,wBAAwB,MAAM;CAC9C,IAAI,CAAE,MAAM,uBAAuB,OAAO,GACxC,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,+DACrB,qBACF;CAEF,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OAAO,OAC1D,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,wBAAwB,QAAQ,MAAM,0BAA0B,OAAO,MAAM,IAClG,oBACF;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,OAAO,YACpE,MAAM,IAAI,gCACR,QAAQ,OAAO,MAAM,6BAA6B,QAAQ,WAAW,+BAA+B,OAAO,WAAW,IACtH,yBACF;CAEF,OAAO;AACT;AAEA,SAAgB,4BAEd,SAAyC;CACzC,MAAM,yBAAS,IAAI,IAAiB;CACpC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,MAAM,oBAAoB,MAAM;EACtC,MAAM,SAAS,OAAO,IAAI,GAAG;EAC7B,IAAI,QAAQ,OAAO,KAAK,MAAM;OACzB,OAAO,IAAI,KAAK,CAAC,MAAM,CAAC;CAC/B;CACA,OAAO;AACT;AAEA,eAAe,+BACb,OAC2C;CAC3C,OAAO,0BAA0B;EAC/B,eAAe;EACf,WAAW,MAAM;EACjB,eAAe,MAAM,qBAAqB,MAAM,aAAa;EAC7D,SAAS,MAAM;EACf,OAAO,MAAM;EACb,YAAY,MAAM;EAClB,YAAY,MAAM;CACpB,CAAC;AACH;AAEA,SAAS,0BACP,OACkC;CAClC,OAAO,cAAc;EACnB,eAAe;EACf,WAAW,gBAAgB,MAAM,WAAW,WAAW;EACvD,eAAe,gBAAgB,MAAM,aAAa;EAClD,SAAS,MAAM,UAAU,iBAAiB,MAAM,SAAS,SAAS,IAAI,KAAA;EACtE,OAAO,iBAAiB,MAAM,OAAO,OAAO;EAC5C,YAAY,iBAAiB,MAAM,YAAY,YAAY;EAC3D,YAAY,MAAM,aACd,eAAe,oBAAoB,MAAM,UAAU,CAAC,IACpD,KAAA;CACN,CAAC;AACH;AAEA,eAAe,qBAAqB,OAA6D;CAC/F,MAAM,OAAO,gBAAgB,MAAM,MAAM,oBAAoB;CAC7D,IAAI,MAAM,SAAS,KAAA,KAAa,MAAM,YAAY,KAAA,GAChD,MAAM,IAAI,gCACR,+DACA,eACF;CAEF,IAAI,MAAM,SAAS,KAAA,GACjB,OAAO;EAAE;EAAM,MAAM,iBAAiB,MAAM,MAAM,oBAAoB;CAAE;CAE1E,IAAI,MAAM,YAAY,KAAA,GACpB,MAAM,IAAI,gCACR,8CACA,eACF;CAEF,WAAW,MAAM,SAAS,uBAAuB;CACjD,OAAO;EAAE;EAAM,MAAM,MAAM,SAAS,MAAM,OAAO;CAAE;AACrD;AAEA,SAAS,gBAAgB,OAA+C;CACtE,OAAO;EACL,MAAM,gBAAgB,MAAM,MAAM,oBAAoB;EACtD,MAAM,iBAAiB,MAAM,MAAM,oBAAoB;CACzD;AACF;AAEA,SAAS,iBAAiB,OAA4B,MAAmC;CACvF,OAAO,cAAc;EACnB,IAAI,gBAAgB,MAAM,IAAI,GAAG,KAAK,IAAI;EAC1C,SAAS,iBAAiB,MAAM,SAAS,GAAG,KAAK,SAAS;EAC1D,MAAM,iBAAiB,MAAM,MAAM,GAAG,KAAK,MAAM;CACnD,CAAC;AACH;AAEA,SAAS,oBACP,OAC4C;CAC5C,MAAM,MAAkD,CAAC;CACzD,KAAK,MAAM,OAAO,OAAO,KAAK,KAAK,CAAC,CAAC,KAAK,GAAG;EAC3C,MAAM,QAAQ,MAAM;EACpB,gBAAgB,KAAK,kBAAkB;EACvC,IACE,UAAU,QACV,OAAO,UAAU,YACjB,OAAO,UAAU,YACjB,OAAO,UAAU,WAEjB,MAAM,IAAI,gCACR,sCACA,cAAc,KAChB;EAEF,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GACrD,MAAM,IAAI,gCAAgC,0BAA0B,cAAc,KAAK;EAEzF,IAAI,OAAO;CACb;CACA,OAAO;AACT;AAEA,SAAS,cAAiD,OAAa;CACrE,MAAM,MAA+B,CAAC;CACtC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,KAAK,GAC7C,IAAI,UAAU,KAAA,GAAW,IAAI,OAAO;CAEtC,OAAO;AACT;AAEA,SAAS,eAAkD,OAAyB;CAClF,OAAO,OAAO,KAAK,KAAK,CAAC,CAAC,SAAS,IAAI,QAAQ,KAAA;AACjD;AAEA,SAAS,eAAe,OAAgB,MAAoB;CAC1D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GACpE,MAAM,IAAI,gCAAgC,mBAAmB,IAAI;CAEnE,MAAM,MAAM;CACZ,aAAa,IAAI,MAAM,GAAG,KAAK,MAAM;CACrC,iBAAiB,IAAI,MAAM,GAAG,KAAK,MAAM;AAC3C;AAEA,SAAS,gBAAgB,OAAgB,MAAoB;CAC3D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GACpE,MAAM,IAAI,gCAAgC,mBAAmB,IAAI;CAEnE,MAAM,MAAM;CACZ,aAAa,IAAI,IAAI,GAAG,KAAK,IAAI;CACjC,IAAI,IAAI,YAAY,KAAA,GAAW,aAAa,IAAI,SAAS,GAAG,KAAK,SAAS;CAC1E,IAAI,IAAI,SAAS,KAAA,GAAW,aAAa,IAAI,MAAM,GAAG,KAAK,MAAM;AACnE;AAEA,SAAS,mBAAmB,OAAgB,MAAoB;CAC9D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GACpE,MAAM,IAAI,gCAAgC,mBAAmB,IAAI;CAEnE,oBAAoB,KAAmD;AACzE;AAEA,SAAS,WAAW,OAAyB,MAAoB;CAC/D,IAAI,UAAU,MAAM;CACpB,MAAM,OAAO,OAAO;CACpB,IAAI,SAAS,YAAY,SAAS,WAAW;CAC7C,IAAI,SAAS,UAAU;EACrB,IAAI,CAAC,OAAO,SAAS,KAAK,GACxB,MAAM,IAAI,gCAAgC,0BAA0B,IAAI;EAE1E;CACF;CACA,IAAI,MAAM,QAAQ,KAAK,GAAG;EACxB,MAAM,SAAS,MAAM,UAAU;GAC7B,WAAW,MAAM,GAAG,KAAK,GAAG,MAAM,EAAE;EACtC,CAAC;EACD;CACF;CACA,IAAI,SAAS,UAAU;EACrB,KAAK,MAAM,CAAC,KAAK,WAAW,OAAO,QAAQ,KAAK,GAAG;GACjD,gBAAgB,KAAK,GAAG,KAAK,OAAO;GACpC,WAAW,QAAQ,GAAG,KAAK,GAAG,KAAK;EACrC;EACA;CACF;CACA,MAAM,IAAI,gCAAgC,kCAAkC,IAAI;AAClF;AAEA,SAAS,cAAc,OAAgB,UAAkB,MAAoB;CAC3E,IAAI,UAAU,UACZ,MAAM,IAAI,gCAAgC,YAAY,YAAY,IAAI;AAE1E;AAEA,SAAS,aAAa,OAAgB,MAAoB;CACxD,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,gCAAgC,6BAA6B,IAAI;AAE/E;AAEA,SAAS,gBAAgB,OAAe,MAAsB;CAC5D,IAAI,OAAO,UAAU,YAAY,MAAM,WAAW,GAChD,MAAM,IAAI,gCAAgC,6BAA6B,IAAI;CAE7E,OAAO;AACT;AAEA,SAAS,iBAAiB,OAA2B,MAAkC;CACrF,IAAI,UAAU,KAAA,GAAW,OAAO,KAAA;CAChC,OAAO,gBAAgB,OAAO,IAAI;AACpC;AAEA,SAAS,iBAAiB,OAAgB,MAAsB;CAC9D,IAAI,OAAO,UAAU,YAAY,CAAC,WAAW,KAAK,KAAK,GACrD,MAAM,IAAI,gCAAgC,0CAA0C,IAAI;CAE1F,OAAO;AACT;;;;;AAsBA,MAAa,sBAAsB;;;;AAIjC,yBAAyB,0BAC3B;;;;;AAQA,SAAgB,mBAAmB,OAAkC;CACnE,IAAI;CACJ,IAAI;EACF,aAAa,KAAK,UAAU,KAAK;CACnC,SAAS,KAAK;EACZ,MAAM,IAAI,gCACR,4CAA4C,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,KAC3F,uBACF;CACF;CACA,IAAI,eAAe,KAAA,GACjB,MAAM,IAAI,gCACR,gFACA,uBACF;CAEF,OAAO,KAAK,MAAM,UAAU;AAC9B;;;;;;;;;;AAcA,eAAsB,+BACpB,SACA,OAC2B;CAC3B,IAAI,CAAC,WAAW,OAAO,YAAY,UACjC,MAAM,IAAI,gCAAgC,kCAAkC,SAAS;CAEvF,IAAI,OAAO,QAAQ,SAAS,YAAY,QAAQ,KAAK,WAAW,GAC9D,MAAM,IAAI,gCACR,6CACA,cACF;CAEF,IAAI,OAAO,QAAQ,YAAY,YAAY,QAAQ,QAAQ,WAAW,GACpE,MAAM,IAAI,gCACR,gDACA,iBACF;CAEF,OAAO,sBAAsB;EAC3B,GAAG;EACH,WAAW,GAAG,QAAQ,KAAK,GAAG,QAAQ;EACtC,eAAe;GACb,MAAM,oBAAoB;GAC1B,SAAS,mBAAmB,OAAO;EACrC;CACF,CAAC;AACH"}