@tangle-network/agent-eval 0.161.0 → 0.163.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/dist/{active-curriculum-CD5TU2yW.js → active-curriculum-OjIWgrUJ.js} +2 -2
  3. package/dist/{active-curriculum-CD5TU2yW.js.map → active-curriculum-OjIWgrUJ.js.map} +1 -1
  4. package/dist/analyst/index.d.ts +2 -2
  5. package/dist/analyst/index.d.ts.map +1 -1
  6. package/dist/analyst/index.js +3 -3
  7. package/dist/{baseline-BhPRQBVn.js → baseline-BC-eBZ7U.js} +2 -2
  8. package/dist/{baseline-BhPRQBVn.js.map → baseline-BC-eBZ7U.js.map} +1 -1
  9. package/dist/{benchmark-command-BVtaq_ve.js → benchmark-command-CF-4GEWZ.js} +9 -8
  10. package/dist/benchmark-command-CF-4GEWZ.js.map +1 -0
  11. package/dist/benchmarks/index.d.ts +1 -1
  12. package/dist/benchmarks/index.js +3 -3
  13. package/dist/builder-eval/index.d.ts.map +1 -1
  14. package/dist/builder-eval/index.js +22 -8
  15. package/dist/builder-eval/index.js.map +1 -1
  16. package/dist/campaign/index.js +8 -8
  17. package/dist/{campaign-BSmOwskD.js → campaign-DQZmc2Dq.js} +12 -12
  18. package/dist/{campaign-BSmOwskD.js.map → campaign-DQZmc2Dq.js.map} +1 -1
  19. package/dist/{chat-json-call-6g5sJobJ.js → chat-json-call-5Jxna-aV.js} +2 -2
  20. package/dist/{chat-json-call-6g5sJobJ.js.map → chat-json-call-5Jxna-aV.js.map} +1 -1
  21. package/dist/cli.js +3 -3
  22. package/dist/contract/index.js +7 -7
  23. package/dist/{counterfactual-D_VWavVm.js → counterfactual-Bjq1mlUu.js} +2 -2
  24. package/dist/{counterfactual-D_VWavVm.js.map → counterfactual-Bjq1mlUu.js.map} +1 -1
  25. package/dist/{define-agent-eval-h-s-sI-v.js → define-agent-eval-D08pWIJb.js} +13 -13
  26. package/dist/{define-agent-eval-h-s-sI-v.js.map → define-agent-eval-D08pWIJb.js.map} +1 -1
  27. package/dist/{descriptive-jDOuI6mz.js → descriptive-1V17A-qa.js} +2 -2
  28. package/dist/{descriptive-jDOuI6mz.js.map → descriptive-1V17A-qa.js.map} +1 -1
  29. package/dist/{dspy-rlm-engine-DptEII26.js → dspy-rlm-engine-DhA9qKIm.js} +2 -2
  30. package/dist/{dspy-rlm-engine-DptEII26.js.map → dspy-rlm-engine-DhA9qKIm.js.map} +1 -1
  31. package/dist/emitter-D_jYSGRd.d.ts.map +1 -1
  32. package/dist/{emitter-BpYFQPj4.js → emitter-DeQHiDMm.js} +13 -6
  33. package/dist/emitter-DeQHiDMm.js.map +1 -0
  34. package/dist/{eval-campaign-BsXWL2-2.js → eval-campaign-BfohKmzx.js} +3 -3
  35. package/dist/{eval-campaign-BsXWL2-2.js.map → eval-campaign-BfohKmzx.js.map} +1 -1
  36. package/dist/experiment/index.js +8 -8
  37. package/dist/{experiment-tracker-Ym6rEQT1.js → experiment-tracker-BKEumQug.js} +2 -2
  38. package/dist/{experiment-tracker-Ym6rEQT1.js.map → experiment-tracker-BKEumQug.js.map} +1 -1
  39. package/dist/{external-optimizer-process-WosTBChy.js → external-optimizer-process-BFmh36vW.js} +2 -2
  40. package/dist/{external-optimizer-process-WosTBChy.js.map → external-optimizer-process-BFmh36vW.js.map} +1 -1
  41. package/dist/{external-optimizer-subprocess-BIWbHpgD.js → external-optimizer-subprocess-CqLMW3nh.js} +2 -2
  42. package/dist/{external-optimizer-subprocess-BIWbHpgD.js.map → external-optimizer-subprocess-CqLMW3nh.js.map} +1 -1
  43. package/dist/fuzz.d.ts +1 -2
  44. package/dist/fuzz.d.ts.map +1 -1
  45. package/dist/fuzz.js +4 -10
  46. package/dist/fuzz.js.map +1 -1
  47. package/dist/index.d.ts +3 -3
  48. package/dist/index.js +26 -26
  49. package/dist/index.js.map +1 -1
  50. package/dist/{internal-BDHPCnjk.js → internal-BMFSR8Ns.js} +3 -19
  51. package/dist/internal-BMFSR8Ns.js.map +1 -0
  52. package/dist/{judge-calibration-zZjLz8hr.js → judge-calibration-BnpVKtnb.js} +3 -13
  53. package/dist/judge-calibration-BnpVKtnb.js.map +1 -0
  54. package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -1
  55. package/dist/{llm-client-hgDieDNN.js → llm-client-BFMRpmqb.js} +8 -11
  56. package/dist/llm-client-BFMRpmqb.js.map +1 -0
  57. package/dist/{llm-judge-BhasIPFT.js → llm-judge-Du7WQPh7.js} +7 -7
  58. package/dist/{llm-judge-BhasIPFT.js.map → llm-judge-Du7WQPh7.js.map} +1 -1
  59. package/dist/meta-eval/index.d.ts +2 -1
  60. package/dist/meta-eval/index.d.ts.map +1 -1
  61. package/dist/meta-eval/index.js +14 -22
  62. package/dist/meta-eval/index.js.map +1 -1
  63. package/dist/openapi.json +1 -1
  64. package/dist/{paired-arms-D-XRF_fy.js → paired-arms-D4aeIHUy.js} +3 -3
  65. package/dist/{paired-arms-D-XRF_fy.js.map → paired-arms-D4aeIHUy.js.map} +1 -1
  66. package/dist/{paired-tests-BHIhYVdu.js → paired-tests-C8iCsioC.js} +3 -3
  67. package/dist/{paired-tests-BHIhYVdu.js.map → paired-tests-C8iCsioC.js.map} +1 -1
  68. package/dist/pipelines/index.d.ts +3 -2
  69. package/dist/pipelines/index.d.ts.map +1 -1
  70. package/dist/pipelines/index.js +4 -19
  71. package/dist/pipelines/index.js.map +1 -1
  72. package/dist/{power-and-mde-CHIrXJll.js → power-and-mde-B8F2RdcD.js} +3 -3
  73. package/dist/{power-and-mde-CHIrXJll.js.map → power-and-mde-B8F2RdcD.js.map} +1 -1
  74. package/dist/{power-preflight-DEw-uC7q.js → power-preflight-CFXm0Vjo.js} +3 -3
  75. package/dist/{power-preflight-DEw-uC7q.js.map → power-preflight-CFXm0Vjo.js.map} +1 -1
  76. package/dist/{produced-state-DZ89riy5.js → produced-state-Be0BK3RN.js} +3 -3
  77. package/dist/{produced-state-DZ89riy5.js.map → produced-state-Be0BK3RN.js.map} +1 -1
  78. package/dist/{promotion-policy-xzA40Evo.js → promotion-policy-LY9mVQ7W.js} +3 -3
  79. package/dist/{promotion-policy-xzA40Evo.js.map → promotion-policy-LY9mVQ7W.js.map} +1 -1
  80. package/dist/{query-DxPYqpmT.d.ts → query-CwnHlu5p.d.ts} +19 -2
  81. package/dist/query-CwnHlu5p.d.ts.map +1 -0
  82. package/dist/{query-CHmMP42p.js → query-_5g6re3_.js} +44 -2
  83. package/dist/query-_5g6re3_.js.map +1 -0
  84. package/dist/random-Dn5fPWkt.js +21 -0
  85. package/dist/random-Dn5fPWkt.js.map +1 -0
  86. package/dist/record-id-DUgsK5qp.js +17 -0
  87. package/dist/record-id-DUgsK5qp.js.map +1 -0
  88. package/dist/{release-confidence-DKfD2RYU.js → release-confidence-nGDJiiwc.js} +2 -2
  89. package/dist/{release-confidence-DKfD2RYU.js.map → release-confidence-nGDJiiwc.js.map} +1 -1
  90. package/dist/reporting.js +6 -6
  91. package/dist/{reward-hacking-t4lB1yt8.js → reward-hacking-O5zKDANP.js} +2 -2
  92. package/dist/{reward-hacking-t4lB1yt8.js.map → reward-hacking-O5zKDANP.js.map} +1 -1
  93. package/dist/rl.d.ts.map +1 -1
  94. package/dist/rl.js +9 -15
  95. package/dist/rl.js.map +1 -1
  96. package/dist/{rubric-predictive-validity-CK8SCOg-.js → rubric-predictive-validity-2D5Gw9z9.js} +3 -3
  97. package/dist/{rubric-predictive-validity-CK8SCOg-.js.map → rubric-predictive-validity-2D5Gw9z9.js.map} +1 -1
  98. package/dist/{semantic-concept-judge-BSkKKHeq.js → semantic-concept-judge-BsDMOwJr.js} +2 -2
  99. package/dist/{semantic-concept-judge-BSkKKHeq.js.map → semantic-concept-judge-BsDMOwJr.js.map} +1 -1
  100. package/dist/{sequential-rYW-Ophm.js → sequential-BLMbdrD7.js} +3 -3
  101. package/dist/{sequential-rYW-Ophm.js.map → sequential-BLMbdrD7.js.map} +1 -1
  102. package/dist/{server-BtFd4uzB.js → server-BjYiJHoJ.js} +2 -2
  103. package/dist/{server-BtFd4uzB.js.map → server-BjYiJHoJ.js.map} +1 -1
  104. package/dist/{skillopt-optimization-method-DbaekMcn.js → skillopt-optimization-method-UArRo-nr.js} +6 -6
  105. package/dist/{skillopt-optimization-method-DbaekMcn.js.map → skillopt-optimization-method-UArRo-nr.js.map} +1 -1
  106. package/dist/{store-tool-spans-Dlh9vkFK.js → store-tool-spans-BVga3c37.js} +21 -17
  107. package/dist/store-tool-spans-BVga3c37.js.map +1 -0
  108. package/dist/store-tool-spans-DPUG7UUY.d.ts.map +1 -1
  109. package/dist/{student-t-CvBq2mve.js → student-t-BA-Uy51p.js} +2 -2
  110. package/dist/{student-t-CvBq2mve.js.map → student-t-BA-Uy51p.js.map} +1 -1
  111. package/dist/{summary-report-BI5hUtvK.js → summary-report-BXeQ5Ues.js} +6 -6
  112. package/dist/{summary-report-BI5hUtvK.js.map → summary-report-BXeQ5Ues.js.map} +1 -1
  113. package/dist/supervisor-run/index.js +1 -1
  114. package/dist/{tool-waste-BqzmVdJk.js → tool-waste-8BQiUc8K.js} +4 -4
  115. package/dist/{tool-waste-BqzmVdJk.js.map → tool-waste-8BQiUc8K.js.map} +1 -1
  116. package/dist/{tool-waste-Dro0gJi3.d.ts → tool-waste-CKc7bYIg.d.ts} +2 -2
  117. package/dist/{tool-waste-Dro0gJi3.d.ts.map → tool-waste-CKc7bYIg.d.ts.map} +1 -1
  118. package/dist/trace-repair/index.js +1 -1
  119. package/dist/traces.d.ts +2 -2
  120. package/dist/traces.d.ts.map +1 -1
  121. package/dist/traces.js +8 -17
  122. package/dist/traces.js.map +1 -1
  123. package/dist/trajectory-replay/index.d.ts.map +1 -1
  124. package/dist/trajectory-replay/index.js +3 -13
  125. package/dist/trajectory-replay/index.js.map +1 -1
  126. package/dist/{types-BPb2Kf_C2.d.ts → types-BPb2Kf_C.d.ts} +1 -1
  127. package/dist/types-BPb2Kf_C.d.ts.map +1 -0
  128. package/dist/types-Bfk0uxRj.d.ts.map +1 -1
  129. package/dist/wire/index.js +1 -1
  130. package/package.json +2 -2
  131. package/dist/benchmark-command-BVtaq_ve.js.map +0 -1
  132. package/dist/emitter-BpYFQPj4.js.map +0 -1
  133. package/dist/internal-BDHPCnjk.js.map +0 -1
  134. package/dist/judge-calibration-zZjLz8hr.js.map +0 -1
  135. package/dist/llm-client-hgDieDNN.js.map +0 -1
  136. package/dist/query-CHmMP42p.js.map +0 -1
  137. package/dist/query-DxPYqpmT.d.ts.map +0 -1
  138. package/dist/store-tool-spans-Dlh9vkFK.js.map +0 -1
  139. package/dist/types-BPb2Kf_C2.d.ts.map +0 -1
@@ -1,4 +1,4 @@
1
- import { c as extractJsonPayload, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-hgDieDNN.js";
1
+ import { c as extractJsonPayload, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-BFMRpmqb.js";
2
2
  //#region src/chat-json-call.ts
3
3
  /** Parse a JSON answer out of a model response. The transport may fence it. */
4
4
  function parseJsonAnswer(response, actor) {
@@ -50,4 +50,4 @@ async function paidJsonChat(input) {
50
50
  //#endregion
51
51
  export { paidJsonChat as t };
52
52
 
53
- //# sourceMappingURL=chat-json-call-6g5sJobJ.js.map
53
+ //# sourceMappingURL=chat-json-call-5Jxna-aV.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"chat-json-call-6g5sJobJ.js","names":[],"sources":["../src/chat-json-call.ts"],"sourcesContent":["/**\n * One paid JSON model call through a caller-owned `ChatClient`, metered by the\n * cost ledger.\n *\n * agent-eval executes no paid model: the transport is supplied by the caller\n * and the credential never enters this package. What stays here is the\n * accounting around the call — the priced maximum reserved before it runs, the\n * stable call id forwarded as the provider idempotency key, the receipt\n * settled from the response, and an honest unknown-usage receipt when the\n * transport failed.\n *\n * The judges and the wire judge endpoint all make the same call in the same\n * order; this is the one copy of that sequence.\n */\n\nimport type { ChatClient, ChatResponse } from './analyst/chat-client'\nimport type { CostChannel, CostLedgerHandle, CostReceipt, CustomTokenPricing } from './cost-ledger'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n extractJsonPayload,\n type LlmCallRequest,\n maximumChargeForLlmRequest,\n} from './llm-client'\n\nexport interface PaidJsonChatInput {\n /** Caller-owned transport. One `chat()` call. */\n chat: ChatClient\n /** The exact canonical request, including its JSON-mode or schema fields. */\n request: LlmCallRequest\n ledger: CostLedgerHandle\n channel: CostChannel\n phase: string\n actor: string\n tags?: Record<string, string>\n signal?: AbortSignal\n /** Endpoint rates used when the transport reports no billed amount. */\n pricing?: CustomTokenPricing\n}\n\nexport type PaidJsonChatResult<T> =\n | { succeeded: true; value: T; response: ChatResponse; receipt: CostReceipt }\n | { succeeded: false; error: Error; receipt?: CostReceipt }\n\n/** Parse a JSON answer out of a model response. The transport may fence it. */\nfunction parseJsonAnswer<T>(response: ChatResponse, actor: string): T {\n try {\n return JSON.parse(extractJsonPayload(response.content)) as T\n } catch (error) {\n throw new Error(\n `${actor}: model answer was not JSON — ${error instanceof Error ? error.message : String(error)}`,\n )\n }\n}\n\nexport async function paidJsonChat<T>(input: PaidJsonChatInput): Promise<PaidJsonChatResult<T>> {\n const paid = await input.ledger.runPaidCall({\n channel: input.channel,\n phase: input.phase,\n actor: input.actor,\n model: input.request.model,\n ...(input.tags && Object.keys(input.tags).length > 0 ? { tags: input.tags } : {}),\n maximumCharge: maximumChargeForLlmRequest(input.request, {\n ...(input.chat.maximumAttempts === undefined\n ? {}\n : { maximumAttempts: input.chat.maximumAttempts }),\n ...(input.pricing ? { customTokenPricing: input.pricing } : {}),\n }),\n ...(input.signal ? { signal: input.signal } : {}),\n execute: (signal, callId) => input.chat.chat(input.request, { signal, idempotencyKey: callId }),\n receipt: (response) => costReceiptFromLlm(response, input.pricing),\n receiptFromError: (error) => costReceiptFromLlmError(error, input.pricing),\n })\n if (!paid.succeeded) {\n return {\n succeeded: false,\n error: paid.error,\n ...(paid.receipt ? { receipt: paid.receipt } : {}),\n }\n }\n // The call completed and was billed. A malformed answer is a contract\n // failure AFTER the money was spent, so it keeps the settled receipt instead\n // of reporting the spend as unknown.\n try {\n return {\n succeeded: true,\n value: parseJsonAnswer<T>(paid.value, input.actor),\n response: paid.value,\n receipt: paid.receipt,\n }\n } catch (error) {\n return {\n succeeded: false,\n error: error instanceof Error ? error : new Error(String(error)),\n receipt: paid.receipt,\n }\n }\n}\n"],"mappings":";;;AA6CA,SAAS,gBAAmB,UAAwB,OAAkB;CACpE,IAAI;EACF,OAAO,KAAK,MAAM,mBAAmB,SAAS,OAAO,CAAC;CACxD,SAAS,OAAO;EACd,MAAM,IAAI,MACR,GAAG,MAAM,gCAAgC,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,GAChG;CACF;AACF;AAEA,eAAsB,aAAgB,OAA0D;CAC9F,MAAM,OAAO,MAAM,MAAM,OAAO,YAAY;EAC1C,SAAS,MAAM;EACf,OAAO,MAAM;EACb,OAAO,MAAM;EACb,OAAO,MAAM,QAAQ;EACrB,GAAI,MAAM,QAAQ,OAAO,KAAK,MAAM,IAAI,CAAC,CAAC,SAAS,IAAI,EAAE,MAAM,MAAM,KAAK,IAAI,CAAC;EAC/E,eAAe,2BAA2B,MAAM,SAAS;GACvD,GAAI,MAAM,KAAK,oBAAoB,KAAA,IAC/B,CAAC,IACD,EAAE,iBAAiB,MAAM,KAAK,gBAAgB;GAClD,GAAI,MAAM,UAAU,EAAE,oBAAoB,MAAM,QAAQ,IAAI,CAAC;EAC/D,CAAC;EACD,GAAI,MAAM,SAAS,EAAE,QAAQ,MAAM,OAAO,IAAI,CAAC;EAC/C,UAAU,QAAQ,WAAW,MAAM,KAAK,KAAK,MAAM,SAAS;GAAE;GAAQ,gBAAgB;EAAO,CAAC;EAC9F,UAAU,aAAa,mBAAmB,UAAU,MAAM,OAAO;EACjE,mBAAmB,UAAU,wBAAwB,OAAO,MAAM,OAAO;CAC3E,CAAC;CACD,IAAI,CAAC,KAAK,WACR,OAAO;EACL,WAAW;EACX,OAAO,KAAK;EACZ,GAAI,KAAK,UAAU,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;CAClD;CAKF,IAAI;EACF,OAAO;GACL,WAAW;GACX,OAAO,gBAAmB,KAAK,OAAO,MAAM,KAAK;GACjD,UAAU,KAAK;GACf,SAAS,KAAK;EAChB;CACF,SAAS,OAAO;EACd,OAAO;GACL,WAAW;GACX,OAAO,iBAAiB,QAAQ,QAAQ,IAAI,MAAM,OAAO,KAAK,CAAC;GAC/D,SAAS,KAAK;EAChB;CACF;AACF"}
1
+ {"version":3,"file":"chat-json-call-5Jxna-aV.js","names":[],"sources":["../src/chat-json-call.ts"],"sourcesContent":["/**\n * One paid JSON model call through a caller-owned `ChatClient`, metered by the\n * cost ledger.\n *\n * agent-eval executes no paid model: the transport is supplied by the caller\n * and the credential never enters this package. What stays here is the\n * accounting around the call — the priced maximum reserved before it runs, the\n * stable call id forwarded as the provider idempotency key, the receipt\n * settled from the response, and an honest unknown-usage receipt when the\n * transport failed.\n *\n * The judges and the wire judge endpoint all make the same call in the same\n * order; this is the one copy of that sequence.\n */\n\nimport type { ChatClient, ChatResponse } from './analyst/chat-client'\nimport type { CostChannel, CostLedgerHandle, CostReceipt, CustomTokenPricing } from './cost-ledger'\nimport {\n costReceiptFromLlm,\n costReceiptFromLlmError,\n extractJsonPayload,\n type LlmCallRequest,\n maximumChargeForLlmRequest,\n} from './llm-client'\n\nexport interface PaidJsonChatInput {\n /** Caller-owned transport. One `chat()` call. */\n chat: ChatClient\n /** The exact canonical request, including its JSON-mode or schema fields. */\n request: LlmCallRequest\n ledger: CostLedgerHandle\n channel: CostChannel\n phase: string\n actor: string\n tags?: Record<string, string>\n signal?: AbortSignal\n /** Endpoint rates used when the transport reports no billed amount. */\n pricing?: CustomTokenPricing\n}\n\nexport type PaidJsonChatResult<T> =\n | { succeeded: true; value: T; response: ChatResponse; receipt: CostReceipt }\n | { succeeded: false; error: Error; receipt?: CostReceipt }\n\n/** Parse a JSON answer out of a model response. The transport may fence it. */\nfunction parseJsonAnswer<T>(response: ChatResponse, actor: string): T {\n try {\n return JSON.parse(extractJsonPayload(response.content)) as T\n } catch (error) {\n throw new Error(\n `${actor}: model answer was not JSON — ${error instanceof Error ? error.message : String(error)}`,\n )\n }\n}\n\nexport async function paidJsonChat<T>(input: PaidJsonChatInput): Promise<PaidJsonChatResult<T>> {\n const paid = await input.ledger.runPaidCall({\n channel: input.channel,\n phase: input.phase,\n actor: input.actor,\n model: input.request.model,\n ...(input.tags && Object.keys(input.tags).length > 0 ? { tags: input.tags } : {}),\n maximumCharge: maximumChargeForLlmRequest(input.request, {\n ...(input.chat.maximumAttempts === undefined\n ? {}\n : { maximumAttempts: input.chat.maximumAttempts }),\n ...(input.pricing ? { customTokenPricing: input.pricing } : {}),\n }),\n ...(input.signal ? { signal: input.signal } : {}),\n execute: (signal, callId) => input.chat.chat(input.request, { signal, idempotencyKey: callId }),\n receipt: (response) => costReceiptFromLlm(response, input.pricing),\n receiptFromError: (error) => costReceiptFromLlmError(error, input.pricing),\n })\n if (!paid.succeeded) {\n return {\n succeeded: false,\n error: paid.error,\n ...(paid.receipt ? { receipt: paid.receipt } : {}),\n }\n }\n // The call completed and was billed. A malformed answer is a contract\n // failure AFTER the money was spent, so it keeps the settled receipt instead\n // of reporting the spend as unknown.\n try {\n return {\n succeeded: true,\n value: parseJsonAnswer<T>(paid.value, input.actor),\n response: paid.value,\n receipt: paid.receipt,\n }\n } catch (error) {\n return {\n succeeded: false,\n error: error instanceof Error ? error : new Error(String(error)),\n receipt: paid.receipt,\n }\n }\n}\n"],"mappings":";;;AA6CA,SAAS,gBAAmB,UAAwB,OAAkB;CACpE,IAAI;EACF,OAAO,KAAK,MAAM,mBAAmB,SAAS,OAAO,CAAC;CACxD,SAAS,OAAO;EACd,MAAM,IAAI,MACR,GAAG,MAAM,gCAAgC,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,GAChG;CACF;AACF;AAEA,eAAsB,aAAgB,OAA0D;CAC9F,MAAM,OAAO,MAAM,MAAM,OAAO,YAAY;EAC1C,SAAS,MAAM;EACf,OAAO,MAAM;EACb,OAAO,MAAM;EACb,OAAO,MAAM,QAAQ;EACrB,GAAI,MAAM,QAAQ,OAAO,KAAK,MAAM,IAAI,CAAC,CAAC,SAAS,IAAI,EAAE,MAAM,MAAM,KAAK,IAAI,CAAC;EAC/E,eAAe,2BAA2B,MAAM,SAAS;GACvD,GAAI,MAAM,KAAK,oBAAoB,KAAA,IAC/B,CAAC,IACD,EAAE,iBAAiB,MAAM,KAAK,gBAAgB;GAClD,GAAI,MAAM,UAAU,EAAE,oBAAoB,MAAM,QAAQ,IAAI,CAAC;EAC/D,CAAC;EACD,GAAI,MAAM,SAAS,EAAE,QAAQ,MAAM,OAAO,IAAI,CAAC;EAC/C,UAAU,QAAQ,WAAW,MAAM,KAAK,KAAK,MAAM,SAAS;GAAE;GAAQ,gBAAgB;EAAO,CAAC;EAC9F,UAAU,aAAa,mBAAmB,UAAU,MAAM,OAAO;EACjE,mBAAmB,UAAU,wBAAwB,OAAO,MAAM,OAAO;CAC3E,CAAC;CACD,IAAI,CAAC,KAAK,WACR,OAAO;EACL,WAAW;EACX,OAAO,KAAK;EACZ,GAAI,KAAK,UAAU,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;CAClD;CAKF,IAAI;EACF,OAAO;GACL,WAAW;GACX,OAAO,gBAAmB,KAAK,OAAO,MAAM,KAAK;GACjD,UAAU,KAAK;GACf,SAAS,KAAK;EAChB;CACF,SAAS,OAAO;EACd,OAAO;GACL,WAAW;GACX,OAAO,iBAAiB,QAAQ,QAAQ,IAAI,MAAM,OAAO,KAAK,CAAC;GAC/D,SAAS,KAAK;EAChB;CACF;AACF"}
package/dist/cli.js CHANGED
@@ -1,8 +1,8 @@
1
1
  #!/usr/bin/env node
2
2
  import { i as runRolloutReleaseCli } from "./hf-dataset-D8_RNIis.js";
3
- import { n as LlmClient } from "./llm-client-hgDieDNN.js";
4
- import { t as runAnalystBenchmarkCommand } from "./benchmark-command-BVtaq_ve.js";
5
- import { a as runRpcBatch, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi } from "./server-BtFd4uzB.js";
3
+ import { n as LlmClient } from "./llm-client-BFMRpmqb.js";
4
+ import { t as runAnalystBenchmarkCommand } from "./benchmark-command-CF-4GEWZ.js";
5
+ import { a as runRpcBatch, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi } from "./server-BjYiJHoJ.js";
6
6
  import { writeFileSync } from "node:fs";
7
7
  //#region src/cli-config.ts
8
8
  /**
@@ -1,14 +1,14 @@
1
1
  import { s as ValidationError } from "../errors-Dngq5h35.js";
2
2
  import { i as hashCanonical } from "../canonical-IL-Bu-14.js";
3
- import { r as pairedBootstrap } from "../paired-tests-BHIhYVdu.js";
4
- import { a as summarizeExecution, i as analyzeRuns, n as SelfImproveRunError, r as selfImprove, t as defineAgentEval } from "../define-agent-eval-h-s-sI-v.js";
3
+ import { r as pairedBootstrap } from "../paired-tests-C8iCsioC.js";
4
+ import { a as summarizeExecution, i as analyzeRuns, n as SelfImproveRunError, r as selfImprove, t as defineAgentEval } from "../define-agent-eval-D08pWIJb.js";
5
5
  import { a as parseRunRecordSafe } from "../run-record-BC0ebuRP.js";
6
- import { E as compareOptimizationMethods, Y as defaultProductionGate, ft as campaignSplitDigest, i as transientDispatchFailure, m as runImprovementLoop, nt as runCampaign, t as llmJudge, tt as runEval } from "../llm-judge-BhasIPFT.js";
6
+ import { E as compareOptimizationMethods, Y as defaultProductionGate, ft as campaignSplitDigest, i as transientDispatchFailure, m as runImprovementLoop, nt as runCampaign, t as llmJudge, tt as runEval } from "../llm-judge-Du7WQPh7.js";
7
7
  import { i as isModelPriced, n as estimateCost } from "../metrics-Qv-cpptD.js";
8
8
  import { i as CostLedger } from "../cost-ledger-B1qx30B4.js";
9
9
  import { d as mapConcurrentRange } from "../ledger-core-BOzlRygb.js";
10
- import { S as inMemoryCampaignStorage, x as fsCampaignStorage } from "../external-optimizer-subprocess-BIWbHpgD.js";
11
- import { a as heldoutSignificance, s as decidePairedPromotion } from "../power-preflight-DEw-uC7q.js";
10
+ import { S as inMemoryCampaignStorage, x as fsCampaignStorage } from "../external-optimizer-subprocess-CqLMW3nh.js";
11
+ import { a as heldoutSignificance, s as decidePairedPromotion } from "../power-preflight-CFXm0Vjo.js";
12
12
  import { r as makeProposalFinding } from "../types-BI4fT3HN.js";
13
13
  import { G as isOtlpModelCall, W as classifyOtlpSpanRole } from "../kind-factory-DY8FdoXf.js";
14
14
  import { n as buildDefaultAnalystRegistry, t as createChatClient } from "../chat-client-DlMlAeYI.js";
@@ -16,8 +16,8 @@ import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js
16
16
  import { t as extractUsage } from "../extract-usage-BrQ8mCLX.js";
17
17
  import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
18
18
  import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-DTl-7-Kw.js";
19
- import { c as externalTextOptimizationMethod, d as REFERENCE_EQUIVALENCE_INPUT_LIMITS, f as REFERENCE_EQUIVALENCE_JUDGE_VERSION, m as runReferenceEquivalenceJudge, n as gepaOptimizationMethod, o as heldOutGate, p as createReferenceEquivalenceJudge, s as composeGate, t as skillOptOptimizationMethod } from "../skillopt-optimization-method-DbaekMcn.js";
20
- import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-xzA40Evo.js";
19
+ import { c as externalTextOptimizationMethod, d as REFERENCE_EQUIVALENCE_INPUT_LIMITS, f as REFERENCE_EQUIVALENCE_JUDGE_VERSION, m as runReferenceEquivalenceJudge, n as gepaOptimizationMethod, o as heldOutGate, p as createReferenceEquivalenceJudge, s as composeGate, t as skillOptOptimizationMethod } from "../skillopt-optimization-method-UArRo-nr.js";
20
+ import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-LY9mVQ7W.js";
21
21
  import { createHash } from "node:crypto";
22
22
  import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, agentProfileImprovementExperimentSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileImprovementRunCellSchema, agentProfileImprovementRunReceiptSchema, agentProfileImprovementSuiteInputsSchema, agentProfileImprovementSuiteSchema, agentProfileImprovementTaskSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, canonicalCandidateJson, numbersApproximatelyEqual, omitTopLevelDigest } from "@tangle-network/agent-interface";
23
23
  import { createReadStream } from "node:fs";
@@ -1,5 +1,5 @@
1
1
  import { o as NotFoundError, s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { t as TraceEmitter } from "./emitter-BpYFQPj4.js";
2
+ import { t as TraceEmitter } from "./emitter-DeQHiDMm.js";
3
3
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
4
4
  //#region src/counterfactual.ts
5
5
  /**
@@ -95,4 +95,4 @@ function applyMutation(step, mutation) {
95
95
  //#endregion
96
96
  export { runCounterfactual as t };
97
97
 
98
- //# sourceMappingURL=counterfactual-D_VWavVm.js.map
98
+ //# sourceMappingURL=counterfactual-Bjq1mlUu.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"counterfactual-D_VWavVm.js","names":[],"sources":["../src/counterfactual.ts"],"sourcesContent":["/**\n * Counterfactual replay — \"what would have happened if we'd changed\n * exactly one thing at turn N?\"\n *\n * The framework does NOT drive the agent — it sets up the replay\n * context (prior spans, prior state, mutation spec) and records the\n * resulting divergence. Consumers supply an `executeFrom(ctx)` callback\n * that runs their agent starting from turn N with the mutation applied.\n *\n * Counterfactual runs are recorded as a new Run with `layer='meta'` and\n * `parentRunId = originalRunId`, so downstream diff + correlation\n * pipelines see them natively.\n */\n\nimport { NotFoundError, ValidationError } from './errors'\nimport { TraceEmitter } from './trace/emitter'\nimport type { LlmSpan, Span, ToolSpan } from './trace/schema'\nimport type { TraceStore } from './trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from './trajectory'\n\nexport type CounterfactualMutation =\n | { kind: 'swap-model'; at: number; newModel: string }\n | { kind: 'swap-tool-result'; at: number; newResult: unknown }\n | { kind: 'truncate-after'; at: number }\n | { kind: 'inject-system-message'; at: number; content: string }\n | {\n kind: 'custom'\n at: number\n describe: string\n apply: (step: TrajectoryStep) => TrajectoryStep\n }\n\nexport interface CounterfactualContext {\n originalRunId: string\n originalTrajectory: Trajectory\n /** Steps up to (but not including) the mutation point — the prefix the\n * replayed agent inherits as its prior conversation/tool history. */\n prefix: TrajectoryStep[]\n mutation: CounterfactualMutation\n /** Pre-applied mutation on the step at `mutation.at`. Consumers use this\n * as the FIRST step the replayed agent emits (they decide whether to\n * re-emit it or continue from there). */\n mutatedStep: TrajectoryStep\n}\n\nexport interface CounterfactualResult {\n counterfactualRunId: string\n originalRunId: string\n mutation: CounterfactualMutation\n /** Structured delta summary — caller can extend via scoring. */\n delta: {\n originalOutcomeScore: number | null\n counterfactualOutcomeScore: number | null\n deltaScore: number | null\n }\n}\n\nexport interface CounterfactualRunner {\n /**\n * Execute the agent from `ctx.prefix` with the mutation applied.\n * MUST emit spans into the provided emitter so they become part of\n * the counterfactual run. MUST call emitter.endRun() with a verdict.\n */\n executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>\n}\n\nexport async function runCounterfactual(\n store: TraceStore,\n originalRunId: string,\n mutation: CounterfactualMutation,\n runner: CounterfactualRunner,\n): Promise<CounterfactualResult> {\n const originalRun = await store.getRun(originalRunId)\n if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`)\n const trajectory = await buildTrajectory(store, originalRunId)\n if (mutation.at < 0 || mutation.at >= trajectory.steps.length) {\n throw new ValidationError(\n `counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`,\n )\n }\n const targetStep = trajectory.steps[mutation.at]!\n const mutatedStep = applyMutation(targetStep, mutation)\n\n const cfEmitter = new TraceEmitter(store)\n await cfEmitter.startRun({\n scenarioId: originalRun.scenarioId,\n variantId: originalRun.variantId\n ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}`\n : `cf:${mutation.kind}@${mutation.at}`,\n projectId: originalRun.projectId,\n parentRunId: originalRunId,\n layer: 'meta',\n tags: { counterfactual: 'true', mutationKind: mutation.kind, mutationAt: String(mutation.at) },\n })\n\n await runner.executeFrom(\n {\n originalRunId,\n originalTrajectory: trajectory,\n prefix: trajectory.steps.slice(0, mutation.at),\n mutation,\n mutatedStep,\n },\n cfEmitter,\n )\n\n const counterfactual = await store.getRun(cfEmitter.runId)\n const delta = {\n originalOutcomeScore: originalRun.outcome?.score ?? null,\n counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,\n deltaScore:\n originalRun.outcome?.score !== undefined && counterfactual?.outcome?.score !== undefined\n ? counterfactual.outcome.score - originalRun.outcome.score\n : null,\n }\n return { counterfactualRunId: cfEmitter.runId, originalRunId, mutation, delta }\n}\n\nfunction applyMutation(step: TrajectoryStep, mutation: CounterfactualMutation): TrajectoryStep {\n if (mutation.kind === 'swap-model' && step.span.kind === 'llm') {\n const llm = step.span as LlmSpan\n return { ...step, span: { ...llm, model: mutation.newModel } }\n }\n if (mutation.kind === 'swap-tool-result' && step.span.kind === 'tool') {\n const tool = step.span as ToolSpan\n return { ...step, span: { ...tool, result: mutation.newResult } }\n }\n if (mutation.kind === 'inject-system-message' && step.span.kind === 'llm') {\n const llm = step.span as LlmSpan\n return {\n ...step,\n span: {\n ...llm,\n messages: [{ role: 'system', content: mutation.content }, ...llm.messages],\n },\n }\n }\n if (mutation.kind === 'custom') return mutation.apply(step)\n // swap-tool-result on non-tool span / swap-model on non-llm / truncate-after: no step-level change.\n return step\n}\n\n/**\n * Aggregate a batch of counterfactuals into a simple attribution table:\n * which mutation kinds move outcomes most? (Useful when you run a grid\n * over the same trajectory — swap-model at every llm span, swap-tool\n * at every tool span — and want a ranked summary.)\n */\nexport function attributeCounterfactuals(results: CounterfactualResult[]): Array<{\n mutationKind: CounterfactualMutation['kind']\n n: number\n meanAbsDelta: number\n meanSignedDelta: number\n}> {\n const grouped = new Map<string, CounterfactualResult[]>()\n for (const r of results) {\n const arr = grouped.get(r.mutation.kind) ?? []\n arr.push(r)\n grouped.set(r.mutation.kind, arr)\n }\n const out: Array<{\n mutationKind: CounterfactualMutation['kind']\n n: number\n meanAbsDelta: number\n meanSignedDelta: number\n }> = []\n for (const [kind, items] of grouped) {\n const deltas = items\n .map((i) => i.delta.deltaScore)\n .filter((d): d is number => typeof d === 'number')\n if (deltas.length === 0) continue\n const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length\n const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length\n out.push({\n mutationKind: kind as CounterfactualMutation['kind'],\n n: deltas.length,\n meanAbsDelta: meanAbs,\n meanSignedDelta: meanSigned,\n })\n }\n return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta)\n}\n\n// Re-export Span type for consumer ergonomics.\nexport type { Span }\n"],"mappings":";;;;;;;;;;;;;;;;;AAkEA,eAAsB,kBACpB,OACA,eACA,UACA,QAC+B;CAC/B,MAAM,cAAc,MAAM,MAAM,OAAO,aAAa;CACpD,IAAI,CAAC,aAAa,MAAM,IAAI,cAAc,uBAAuB,cAAc,WAAW;CAC1F,MAAM,aAAa,MAAM,gBAAgB,OAAO,aAAa;CAC7D,IAAI,SAAS,KAAK,KAAK,SAAS,MAAM,WAAW,MAAM,QACrD,MAAM,IAAI,gBACR,+BAA+B,SAAS,GAAG,oBAAoB,WAAW,MAAM,OAAO,EACzF;CAEF,MAAM,aAAa,WAAW,MAAM,SAAS;CAC7C,MAAM,cAAc,cAAc,YAAY,QAAQ;CAEtD,MAAM,YAAY,IAAI,aAAa,KAAK;CACxC,MAAM,UAAU,SAAS;EACvB,YAAY,YAAY;EACxB,WAAW,YAAY,YACnB,GAAG,YAAY,UAAU,MAAM,SAAS,KAAK,GAAG,SAAS,OACzD,MAAM,SAAS,KAAK,GAAG,SAAS;EACpC,WAAW,YAAY;EACvB,aAAa;EACb,OAAO;EACP,MAAM;GAAE,gBAAgB;GAAQ,cAAc,SAAS;GAAM,YAAY,OAAO,SAAS,EAAE;EAAE;CAC/F,CAAC;CAED,MAAM,OAAO,YACX;EACE;EACA,oBAAoB;EACpB,QAAQ,WAAW,MAAM,MAAM,GAAG,SAAS,EAAE;EAC7C;EACA;CACF,GACA,SACF;CAEA,MAAM,iBAAiB,MAAM,MAAM,OAAO,UAAU,KAAK;CACzD,MAAM,QAAQ;EACZ,sBAAsB,YAAY,SAAS,SAAS;EACpD,4BAA4B,gBAAgB,SAAS,SAAS;EAC9D,YACE,YAAY,SAAS,UAAU,KAAA,KAAa,gBAAgB,SAAS,UAAU,KAAA,IAC3E,eAAe,QAAQ,QAAQ,YAAY,QAAQ,QACnD;CACR;CACA,OAAO;EAAE,qBAAqB,UAAU;EAAO;EAAe;EAAU;CAAM;AAChF;AAEA,SAAS,cAAc,MAAsB,UAAkD;CAC7F,IAAI,SAAS,SAAS,gBAAgB,KAAK,KAAK,SAAS,OAAO;EAC9D,MAAM,MAAM,KAAK;EACjB,OAAO;GAAE,GAAG;GAAM,MAAM;IAAE,GAAG;IAAK,OAAO,SAAS;GAAS;EAAE;CAC/D;CACA,IAAI,SAAS,SAAS,sBAAsB,KAAK,KAAK,SAAS,QAAQ;EACrE,MAAM,OAAO,KAAK;EAClB,OAAO;GAAE,GAAG;GAAM,MAAM;IAAE,GAAG;IAAM,QAAQ,SAAS;GAAU;EAAE;CAClE;CACA,IAAI,SAAS,SAAS,2BAA2B,KAAK,KAAK,SAAS,OAAO;EACzE,MAAM,MAAM,KAAK;EACjB,OAAO;GACL,GAAG;GACH,MAAM;IACJ,GAAG;IACH,UAAU,CAAC;KAAE,MAAM;KAAU,SAAS,SAAS;IAAQ,GAAG,GAAG,IAAI,QAAQ;GAC3E;EACF;CACF;CACA,IAAI,SAAS,SAAS,UAAU,OAAO,SAAS,MAAM,IAAI;CAE1D,OAAO;AACT"}
1
+ {"version":3,"file":"counterfactual-Bjq1mlUu.js","names":[],"sources":["../src/counterfactual.ts"],"sourcesContent":["/**\n * Counterfactual replay — \"what would have happened if we'd changed\n * exactly one thing at turn N?\"\n *\n * The framework does NOT drive the agent — it sets up the replay\n * context (prior spans, prior state, mutation spec) and records the\n * resulting divergence. Consumers supply an `executeFrom(ctx)` callback\n * that runs their agent starting from turn N with the mutation applied.\n *\n * Counterfactual runs are recorded as a new Run with `layer='meta'` and\n * `parentRunId = originalRunId`, so downstream diff + correlation\n * pipelines see them natively.\n */\n\nimport { NotFoundError, ValidationError } from './errors'\nimport { TraceEmitter } from './trace/emitter'\nimport type { LlmSpan, Span, ToolSpan } from './trace/schema'\nimport type { TraceStore } from './trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from './trajectory'\n\nexport type CounterfactualMutation =\n | { kind: 'swap-model'; at: number; newModel: string }\n | { kind: 'swap-tool-result'; at: number; newResult: unknown }\n | { kind: 'truncate-after'; at: number }\n | { kind: 'inject-system-message'; at: number; content: string }\n | {\n kind: 'custom'\n at: number\n describe: string\n apply: (step: TrajectoryStep) => TrajectoryStep\n }\n\nexport interface CounterfactualContext {\n originalRunId: string\n originalTrajectory: Trajectory\n /** Steps up to (but not including) the mutation point — the prefix the\n * replayed agent inherits as its prior conversation/tool history. */\n prefix: TrajectoryStep[]\n mutation: CounterfactualMutation\n /** Pre-applied mutation on the step at `mutation.at`. Consumers use this\n * as the FIRST step the replayed agent emits (they decide whether to\n * re-emit it or continue from there). */\n mutatedStep: TrajectoryStep\n}\n\nexport interface CounterfactualResult {\n counterfactualRunId: string\n originalRunId: string\n mutation: CounterfactualMutation\n /** Structured delta summary — caller can extend via scoring. */\n delta: {\n originalOutcomeScore: number | null\n counterfactualOutcomeScore: number | null\n deltaScore: number | null\n }\n}\n\nexport interface CounterfactualRunner {\n /**\n * Execute the agent from `ctx.prefix` with the mutation applied.\n * MUST emit spans into the provided emitter so they become part of\n * the counterfactual run. MUST call emitter.endRun() with a verdict.\n */\n executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>\n}\n\nexport async function runCounterfactual(\n store: TraceStore,\n originalRunId: string,\n mutation: CounterfactualMutation,\n runner: CounterfactualRunner,\n): Promise<CounterfactualResult> {\n const originalRun = await store.getRun(originalRunId)\n if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`)\n const trajectory = await buildTrajectory(store, originalRunId)\n if (mutation.at < 0 || mutation.at >= trajectory.steps.length) {\n throw new ValidationError(\n `counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`,\n )\n }\n const targetStep = trajectory.steps[mutation.at]!\n const mutatedStep = applyMutation(targetStep, mutation)\n\n const cfEmitter = new TraceEmitter(store)\n await cfEmitter.startRun({\n scenarioId: originalRun.scenarioId,\n variantId: originalRun.variantId\n ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}`\n : `cf:${mutation.kind}@${mutation.at}`,\n projectId: originalRun.projectId,\n parentRunId: originalRunId,\n layer: 'meta',\n tags: { counterfactual: 'true', mutationKind: mutation.kind, mutationAt: String(mutation.at) },\n })\n\n await runner.executeFrom(\n {\n originalRunId,\n originalTrajectory: trajectory,\n prefix: trajectory.steps.slice(0, mutation.at),\n mutation,\n mutatedStep,\n },\n cfEmitter,\n )\n\n const counterfactual = await store.getRun(cfEmitter.runId)\n const delta = {\n originalOutcomeScore: originalRun.outcome?.score ?? null,\n counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,\n deltaScore:\n originalRun.outcome?.score !== undefined && counterfactual?.outcome?.score !== undefined\n ? counterfactual.outcome.score - originalRun.outcome.score\n : null,\n }\n return { counterfactualRunId: cfEmitter.runId, originalRunId, mutation, delta }\n}\n\nfunction applyMutation(step: TrajectoryStep, mutation: CounterfactualMutation): TrajectoryStep {\n if (mutation.kind === 'swap-model' && step.span.kind === 'llm') {\n const llm = step.span as LlmSpan\n return { ...step, span: { ...llm, model: mutation.newModel } }\n }\n if (mutation.kind === 'swap-tool-result' && step.span.kind === 'tool') {\n const tool = step.span as ToolSpan\n return { ...step, span: { ...tool, result: mutation.newResult } }\n }\n if (mutation.kind === 'inject-system-message' && step.span.kind === 'llm') {\n const llm = step.span as LlmSpan\n return {\n ...step,\n span: {\n ...llm,\n messages: [{ role: 'system', content: mutation.content }, ...llm.messages],\n },\n }\n }\n if (mutation.kind === 'custom') return mutation.apply(step)\n // swap-tool-result on non-tool span / swap-model on non-llm / truncate-after: no step-level change.\n return step\n}\n\n/**\n * Aggregate a batch of counterfactuals into a simple attribution table:\n * which mutation kinds move outcomes most? (Useful when you run a grid\n * over the same trajectory — swap-model at every llm span, swap-tool\n * at every tool span — and want a ranked summary.)\n */\nexport function attributeCounterfactuals(results: CounterfactualResult[]): Array<{\n mutationKind: CounterfactualMutation['kind']\n n: number\n meanAbsDelta: number\n meanSignedDelta: number\n}> {\n const grouped = new Map<string, CounterfactualResult[]>()\n for (const r of results) {\n const arr = grouped.get(r.mutation.kind) ?? []\n arr.push(r)\n grouped.set(r.mutation.kind, arr)\n }\n const out: Array<{\n mutationKind: CounterfactualMutation['kind']\n n: number\n meanAbsDelta: number\n meanSignedDelta: number\n }> = []\n for (const [kind, items] of grouped) {\n const deltas = items\n .map((i) => i.delta.deltaScore)\n .filter((d): d is number => typeof d === 'number')\n if (deltas.length === 0) continue\n const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length\n const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length\n out.push({\n mutationKind: kind as CounterfactualMutation['kind'],\n n: deltas.length,\n meanAbsDelta: meanAbs,\n meanSignedDelta: meanSigned,\n })\n }\n return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta)\n}\n\n// Re-export Span type for consumer ergonomics.\nexport type { Span }\n"],"mappings":";;;;;;;;;;;;;;;;;AAkEA,eAAsB,kBACpB,OACA,eACA,UACA,QAC+B;CAC/B,MAAM,cAAc,MAAM,MAAM,OAAO,aAAa;CACpD,IAAI,CAAC,aAAa,MAAM,IAAI,cAAc,uBAAuB,cAAc,WAAW;CAC1F,MAAM,aAAa,MAAM,gBAAgB,OAAO,aAAa;CAC7D,IAAI,SAAS,KAAK,KAAK,SAAS,MAAM,WAAW,MAAM,QACrD,MAAM,IAAI,gBACR,+BAA+B,SAAS,GAAG,oBAAoB,WAAW,MAAM,OAAO,EACzF;CAEF,MAAM,aAAa,WAAW,MAAM,SAAS;CAC7C,MAAM,cAAc,cAAc,YAAY,QAAQ;CAEtD,MAAM,YAAY,IAAI,aAAa,KAAK;CACxC,MAAM,UAAU,SAAS;EACvB,YAAY,YAAY;EACxB,WAAW,YAAY,YACnB,GAAG,YAAY,UAAU,MAAM,SAAS,KAAK,GAAG,SAAS,OACzD,MAAM,SAAS,KAAK,GAAG,SAAS;EACpC,WAAW,YAAY;EACvB,aAAa;EACb,OAAO;EACP,MAAM;GAAE,gBAAgB;GAAQ,cAAc,SAAS;GAAM,YAAY,OAAO,SAAS,EAAE;EAAE;CAC/F,CAAC;CAED,MAAM,OAAO,YACX;EACE;EACA,oBAAoB;EACpB,QAAQ,WAAW,MAAM,MAAM,GAAG,SAAS,EAAE;EAC7C;EACA;CACF,GACA,SACF;CAEA,MAAM,iBAAiB,MAAM,MAAM,OAAO,UAAU,KAAK;CACzD,MAAM,QAAQ;EACZ,sBAAsB,YAAY,SAAS,SAAS;EACpD,4BAA4B,gBAAgB,SAAS,SAAS;EAC9D,YACE,YAAY,SAAS,UAAU,KAAA,KAAa,gBAAgB,SAAS,UAAU,KAAA,IAC3E,eAAe,QAAQ,QAAQ,YAAY,QAAQ,QACnD;CACR;CACA,OAAO;EAAE,qBAAqB,UAAU;EAAO;EAAe;EAAU;CAAM;AAChF;AAEA,SAAS,cAAc,MAAsB,UAAkD;CAC7F,IAAI,SAAS,SAAS,gBAAgB,KAAK,KAAK,SAAS,OAAO;EAC9D,MAAM,MAAM,KAAK;EACjB,OAAO;GAAE,GAAG;GAAM,MAAM;IAAE,GAAG;IAAK,OAAO,SAAS;GAAS;EAAE;CAC/D;CACA,IAAI,SAAS,SAAS,sBAAsB,KAAK,KAAK,SAAS,QAAQ;EACrE,MAAM,OAAO,KAAK;EAClB,OAAO;GAAE,GAAG;GAAM,MAAM;IAAE,GAAG;IAAM,QAAQ,SAAS;GAAU;EAAE;CAClE;CACA,IAAI,SAAS,SAAS,2BAA2B,KAAK,KAAK,SAAS,OAAO;EACzE,MAAM,MAAM,KAAK;EACjB,OAAO;GACL,GAAG;GACH,MAAM;IACJ,GAAG;IACH,UAAU,CAAC;KAAE,MAAM;KAAU,SAAS,SAAS;IAAQ,GAAG,GAAG,IAAI,QAAQ;GAC3E;EACF;CACF;CACA,IAAI,SAAS,SAAS,UAAU,OAAO,SAAS,MAAM,IAAI;CAE1D,OAAO;AACT"}
@@ -1,19 +1,19 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { a as spearmanR, r as pearsonR } from "./descriptive-jDOuI6mz.js";
3
- import { r as continuousAgreement } from "./judge-calibration-zZjLz8hr.js";
2
+ import { a as spearmanR, r as pearsonR } from "./descriptive-1V17A-qa.js";
3
+ import { r as continuousAgreement } from "./judge-calibration-BnpVKtnb.js";
4
4
  import { i as pairedCohensDz } from "./effect-sizes-DiH8MGOH.js";
5
- import { i as requiredPairedSampleSize, r as pairedMde } from "./power-and-mde-CHIrXJll.js";
6
- import { r as pairRunRecords } from "./paired-arms-D-XRF_fy.js";
7
- import { o as pairedTTest, r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
8
- import { r as welchsTTest } from "./baseline-BhPRQBVn.js";
9
- import "./query-CHmMP42p.js";
5
+ import { i as requiredPairedSampleSize, r as pairedMde } from "./power-and-mde-B8F2RdcD.js";
6
+ import { r as pairRunRecords } from "./paired-arms-D4aeIHUy.js";
7
+ import { o as pairedTTest, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
8
+ import { r as welchsTTest } from "./baseline-BC-eBZ7U.js";
9
+ import "./query-_5g6re3_.js";
10
10
  import { r as observedSplitScore } from "./reward-nw2xZGZG.js";
11
11
  import { c as validateRunRecord, i as modelHasSnapshot } from "./run-record-BC0ebuRP.js";
12
- import { r as paretoChart } from "./summary-report-BI5hUtvK.js";
13
- import { N as surfaceContentHash, P as surfaceHash, Y as defaultProductionGate, c as emitLoopProvenance, it as resolveRunDir, l as loopProvenanceArgsFromResult, m as runImprovementLoop, tt as runEval, w as assertOptimizationResult } from "./llm-judge-BhasIPFT.js";
14
- import { c as campaignCellTaskScore, l as campaignCellToRunRecord, o as campaignCellExecutionEvidence, s as campaignCellJudgeDimensions } from "./reward-hacking-t4lB1yt8.js";
15
- import { S as inMemoryCampaignStorage, b as createRunCostLedger, x as fsCampaignStorage } from "./external-optimizer-subprocess-BIWbHpgD.js";
16
- import { t as powerPreflight } from "./power-preflight-DEw-uC7q.js";
12
+ import { r as paretoChart } from "./summary-report-BXeQ5Ues.js";
13
+ import { N as surfaceContentHash, P as surfaceHash, Y as defaultProductionGate, c as emitLoopProvenance, it as resolveRunDir, l as loopProvenanceArgsFromResult, m as runImprovementLoop, tt as runEval, w as assertOptimizationResult } from "./llm-judge-Du7WQPh7.js";
14
+ import { c as campaignCellTaskScore, l as campaignCellToRunRecord, o as campaignCellExecutionEvidence, s as campaignCellJudgeDimensions } from "./reward-hacking-O5zKDANP.js";
15
+ import { S as inMemoryCampaignStorage, b as createRunCostLedger, x as fsCampaignStorage } from "./external-optimizer-subprocess-CqLMW3nh.js";
16
+ import { t as powerPreflight } from "./power-preflight-CFXm0Vjo.js";
17
17
  import { t as createHostedClient } from "./client-CX7KqIdB.js";
18
18
  //#region src/contamination-guard.ts
19
19
  function checkCanaries(output, scenarios) {
@@ -1497,4 +1497,4 @@ function requirePositiveInteger(value, field) {
1497
1497
  //#endregion
1498
1498
  export { summarizeExecution as a, analyzeRuns as i, SelfImproveRunError as n, checkCanaries as o, selfImprove as r, defineAgentEval as t };
1499
1499
 
1500
- //# sourceMappingURL=define-agent-eval-h-s-sI-v.js.map
1500
+ //# sourceMappingURL=define-agent-eval-D08pWIJb.js.map