@tangle-network/agent-eval 0.149.0 → 0.150.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +42 -0
  2. package/README.md +5 -1
  3. package/dist/analyst/index.d.ts +13 -13
  4. package/dist/analyst/index.d.ts.map +1 -1
  5. package/dist/analyst/index.js +4 -4
  6. package/dist/{backend-integrity-Bz8nSrRE.d.ts → backend-integrity-DOCa_QrR.d.ts} +2 -2
  7. package/dist/{backend-integrity-Bz8nSrRE.d.ts.map → backend-integrity-DOCa_QrR.d.ts.map} +1 -1
  8. package/dist/{benchmark-Cn7ZMFNW.d.ts → benchmark-BtAWA8nT.d.ts} +3 -3
  9. package/dist/{benchmark-Cn7ZMFNW.d.ts.map → benchmark-BtAWA8nT.d.ts.map} +1 -1
  10. package/dist/{benchmark-command-gufocQ-N.js → benchmark-command-CAFwbH0L.js} +8 -7
  11. package/dist/benchmark-command-CAFwbH0L.js.map +1 -0
  12. package/dist/benchmarks/index.d.ts +4 -4
  13. package/dist/benchmarks/index.js +3 -3
  14. package/dist/campaign/index.d.ts +9 -9
  15. package/dist/campaign/index.js +6 -6
  16. package/dist/{campaign-BcfXzmPM.js → campaign-CN_7xJdV.js} +90 -8
  17. package/dist/campaign-CN_7xJdV.js.map +1 -0
  18. package/dist/{capture-fetch-BAx3Gntt.d.ts → capture-fetch-BBVFzhkk.d.ts} +2 -2
  19. package/dist/{capture-fetch-BAx3Gntt.d.ts.map → capture-fetch-BBVFzhkk.d.ts.map} +1 -1
  20. package/dist/{chat-client-BX8Wh3fn.js → chat-client-2bVfrzhN.js} +2 -2
  21. package/dist/{chat-client-BX8Wh3fn.js.map → chat-client-2bVfrzhN.js.map} +1 -1
  22. package/dist/cli.js +2 -2
  23. package/dist/{client-BV7icIfy.d.ts → client-kPQYT_56.d.ts} +2 -2
  24. package/dist/{client-BV7icIfy.d.ts.map → client-kPQYT_56.d.ts.map} +1 -1
  25. package/dist/contract/index.d.ts +10 -10
  26. package/dist/contract/index.js +5 -5
  27. package/dist/{default-registry-BUtRMRKk.d.ts → default-registry-Cw0Ohdoj.d.ts} +6 -6
  28. package/dist/{default-registry-BUtRMRKk.d.ts.map → default-registry-Cw0Ohdoj.d.ts.map} +1 -1
  29. package/dist/{define-agent-eval-BWuk2O_N.d.ts → define-agent-eval-CEQWL9Hy.d.ts} +5 -5
  30. package/dist/{define-agent-eval-BWuk2O_N.d.ts.map → define-agent-eval-CEQWL9Hy.d.ts.map} +1 -1
  31. package/dist/{define-agent-eval-CvZQW4u9.js → define-agent-eval-rqNyVhVV.js} +4 -4
  32. package/dist/{define-agent-eval-CvZQW4u9.js.map → define-agent-eval-rqNyVhVV.js.map} +1 -1
  33. package/dist/{dspy-rlm-engine-BmtOl_kP.js → dspy-rlm-engine-DHI0WrUU.js} +2 -2
  34. package/dist/{dspy-rlm-engine-BmtOl_kP.js.map → dspy-rlm-engine-DHI0WrUU.js.map} +1 -1
  35. package/dist/{engine-CIy18RTX.d.ts → engine-BLzhNzoY.d.ts} +4 -4
  36. package/dist/{engine-CIy18RTX.d.ts.map → engine-BLzhNzoY.d.ts.map} +1 -1
  37. package/dist/{eval-campaign-bmZ6NIIP.js → eval-campaign-C4jmuM-b.js} +2 -2
  38. package/dist/{eval-campaign-bmZ6NIIP.js.map → eval-campaign-C4jmuM-b.js.map} +1 -1
  39. package/dist/{exact-types-rdKFzEnK.d.ts → exact-types-ccQAyut1.d.ts} +2 -2
  40. package/dist/{exact-types-rdKFzEnK.d.ts.map → exact-types-ccQAyut1.d.ts.map} +1 -1
  41. package/dist/experiment/index.d.ts +73 -4
  42. package/dist/experiment/index.d.ts.map +1 -1
  43. package/dist/experiment/index.js +175 -1
  44. package/dist/experiment/index.js.map +1 -1
  45. package/dist/{experiment-tracker-D9VHwRm-.d.ts → experiment-tracker-0MhuPArU.d.ts} +2 -2
  46. package/dist/{experiment-tracker-D9VHwRm-.d.ts.map → experiment-tracker-0MhuPArU.d.ts.map} +1 -1
  47. package/dist/{external-optimizer-contracts-CSjDLLmr.d.ts → external-optimizer-contracts-DbLsm4Po.d.ts} +13 -6
  48. package/dist/{external-optimizer-contracts-CSjDLLmr.d.ts.map → external-optimizer-contracts-DbLsm4Po.d.ts.map} +1 -1
  49. package/dist/{external-optimizer-process-C4AyCcQd.js → external-optimizer-process-Bhmzngf-.js} +4 -4
  50. package/dist/{external-optimizer-process-C4AyCcQd.js.map → external-optimizer-process-Bhmzngf-.js.map} +1 -1
  51. package/dist/{external-optimizer-subprocess-DU1KM8yV.js → external-optimizer-subprocess-Dn90UqN2.js} +572 -25
  52. package/dist/external-optimizer-subprocess-Dn90UqN2.js.map +1 -0
  53. package/dist/{feedback-trajectory-niGeR6uJ.d.ts → feedback-trajectory-DpTTjo0q.d.ts} +3 -3
  54. package/dist/{feedback-trajectory-niGeR6uJ.d.ts.map → feedback-trajectory-DpTTjo0q.d.ts.map} +1 -1
  55. package/dist/hosted/index.d.ts +2 -2
  56. package/dist/index-C1ravkGA.d.ts +1 -0
  57. package/dist/{index-aHDnhBC7.d.ts → index-CTKpu9ry.d.ts} +7 -7
  58. package/dist/{index-aHDnhBC7.d.ts.map → index-CTKpu9ry.d.ts.map} +1 -1
  59. package/dist/{index-BCDyf_kT.d.ts → index-IQccV3Ou.d.ts} +47 -11
  60. package/dist/index-IQccV3Ou.d.ts.map +1 -0
  61. package/dist/index.d.ts +20 -20
  62. package/dist/index.js +8 -8
  63. package/dist/{integrity-CGfpTE5-.d.ts → integrity-B0dZ96EO.d.ts} +2 -2
  64. package/dist/{integrity-CGfpTE5-.d.ts.map → integrity-B0dZ96EO.d.ts.map} +1 -1
  65. package/dist/{llm-client-BMuxYoZy.js → llm-client-Bg32RW0j.js} +61 -7
  66. package/dist/llm-client-Bg32RW0j.js.map +1 -0
  67. package/dist/{llm-judge-DWq1Ptco.js → llm-judge-CVq33oz1.js} +5 -4
  68. package/dist/llm-judge-CVq33oz1.js.map +1 -0
  69. package/dist/{matrix-Bc111FKv.d.ts → matrix-DrVnRp4G.d.ts} +2 -2
  70. package/dist/{matrix-Bc111FKv.d.ts.map → matrix-DrVnRp4G.d.ts.map} +1 -1
  71. package/dist/meta-eval/index.d.ts +1 -1
  72. package/dist/multishot/golden/index.d.ts +1 -1
  73. package/dist/multishot/index.d.ts +2 -2
  74. package/dist/openapi.json +1 -1
  75. package/dist/{produced-state-C0oJ4vr-.js → produced-state-jfk8Du3b.js} +3 -3
  76. package/dist/{produced-state-C0oJ4vr-.js.map → produced-state-jfk8Du3b.js.map} +1 -1
  77. package/dist/{promotion-policy-CewdjfCJ.d.ts → promotion-policy-DLOUkYhI.d.ts} +2 -2
  78. package/dist/{promotion-policy-CewdjfCJ.d.ts.map → promotion-policy-DLOUkYhI.d.ts.map} +1 -1
  79. package/dist/{provenance-DA-Pmyfv.d.ts → provenance-oA4-zUqm.d.ts} +17 -6
  80. package/dist/{provenance-DA-Pmyfv.d.ts.map → provenance-oA4-zUqm.d.ts.map} +1 -1
  81. package/dist/{registry-CaqJIDdN.d.ts → registry-BQwrSYpC.d.ts} +4 -4
  82. package/dist/{registry-CaqJIDdN.d.ts.map → registry-BQwrSYpC.d.ts.map} +1 -1
  83. package/dist/{researcher-BySlpla_.d.ts → researcher-DJnoUE8c.d.ts} +3 -3
  84. package/dist/{researcher-BySlpla_.d.ts.map → researcher-DJnoUE8c.d.ts.map} +1 -1
  85. package/dist/{reward-hacking-DFo2FU5J.js → reward-hacking-DKI9T52l.js} +19 -1
  86. package/dist/reward-hacking-DKI9T52l.js.map +1 -0
  87. package/dist/rl.d.ts +2 -2
  88. package/dist/rl.js +2 -2
  89. package/dist/{semantic-concept-judge-BI7Rrl5-.js → semantic-concept-judge-laMCnTLn.js} +2 -2
  90. package/dist/{semantic-concept-judge-BI7Rrl5-.js.map → semantic-concept-judge-laMCnTLn.js.map} +1 -1
  91. package/dist/{series-convergence-D2fsoJ2w.d.ts → series-convergence-BxKEgBwA.d.ts} +2 -2
  92. package/dist/{series-convergence-D2fsoJ2w.d.ts.map → series-convergence-BxKEgBwA.d.ts.map} +1 -1
  93. package/dist/{server-D9wQclzG.js → server-dIWwF3j_.js} +2 -2
  94. package/dist/{server-D9wQclzG.js.map → server-dIWwF3j_.js.map} +1 -1
  95. package/dist/{skillopt-optimization-method-ChCTN4FD.d.ts → skillopt-optimization-method-BDD_o1xE.d.ts} +19 -13
  96. package/dist/{skillopt-optimization-method-ChCTN4FD.d.ts.map → skillopt-optimization-method-BDD_o1xE.d.ts.map} +1 -1
  97. package/dist/{skillopt-optimization-method-Donmq4sq.js → skillopt-optimization-method-DLeUcK-K.js} +92 -14
  98. package/dist/skillopt-optimization-method-DLeUcK-K.js.map +1 -0
  99. package/dist/{statistical-heldout-uxSpFEjm.d.ts → statistical-heldout-_woZ9q9j.d.ts} +2 -2
  100. package/dist/{statistical-heldout-uxSpFEjm.d.ts.map → statistical-heldout-_woZ9q9j.d.ts.map} +1 -1
  101. package/dist/{store-tool-spans-D5FhM0_A.d.ts → store-tool-spans-Br2_IUhm.d.ts} +4 -4
  102. package/dist/{store-tool-spans-D5FhM0_A.d.ts.map → store-tool-spans-Br2_IUhm.d.ts.map} +1 -1
  103. package/dist/{tool-groups-zpufabP8.d.ts → tool-groups-B4tqh8jB.d.ts} +3 -3
  104. package/dist/tool-groups-B4tqh8jB.d.ts.map +1 -0
  105. package/dist/trace-repair/index.d.ts +1 -1
  106. package/dist/traces.d.ts +7 -7
  107. package/dist/{types-BohHewKK.d.ts → types-B2NsbrNy.d.ts} +2 -2
  108. package/dist/{types-BohHewKK.d.ts.map → types-B2NsbrNy.d.ts.map} +1 -1
  109. package/dist/{types-ColZlZtc.d.ts → types-CLAwnY-L.d.ts} +2 -2
  110. package/dist/{types-ColZlZtc.d.ts.map → types-CLAwnY-L.d.ts.map} +1 -1
  111. package/dist/{types-C1Bmeb8X.d.ts → types-DdFNuyxQ.d.ts} +3 -3
  112. package/dist/{types-C1Bmeb8X.d.ts.map → types-DdFNuyxQ.d.ts.map} +1 -1
  113. package/dist/{types-DOhGq4S1.d.ts → types-jUBXJ7Iz.d.ts} +50 -5
  114. package/dist/types-jUBXJ7Iz.d.ts.map +1 -0
  115. package/dist/wire/index.d.ts +2 -2
  116. package/dist/wire/index.js +1 -1
  117. package/docs/campaign-proposers.md +67 -3
  118. package/docs/experiment.md +2 -0
  119. package/docs/multishot-golden-records.md +3 -0
  120. package/package.json +4 -2
  121. package/dist/benchmark-command-gufocQ-N.js.map +0 -1
  122. package/dist/campaign-BcfXzmPM.js.map +0 -1
  123. package/dist/external-optimizer-subprocess-DU1KM8yV.js.map +0 -1
  124. package/dist/index-BCDyf_kT.d.ts.map +0 -1
  125. package/dist/index-CKblBxtr.d.ts +0 -1
  126. package/dist/llm-client-BMuxYoZy.js.map +0 -1
  127. package/dist/llm-judge-DWq1Ptco.js.map +0 -1
  128. package/dist/reward-hacking-DFo2FU5J.js.map +0 -1
  129. package/dist/skillopt-optimization-method-Donmq4sq.js.map +0 -1
  130. package/dist/tool-groups-zpufabP8.d.ts.map +0 -1
  131. package/dist/types-DOhGq4S1.d.ts.map +0 -1
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index-IQccV3Ou.d.ts","names":[],"sources":["../src/campaign/analyst-surface.ts","../src/campaign/cross-surface-types.ts","../src/campaign/cross-surface-interaction.ts","../src/campaign/search-ledger-errors.ts","../src/campaign/fixtures.ts","../src/campaign/gates/neutralization-gate.ts","../src/campaign/grounded-reflection.ts","../src/campaign/labeled-store/fs-adapter.ts","../src/campaign/neutralize.ts","../src/campaign/openai-compatible-execution-owner.ts","../src/campaign/presets/run-profile-matrix.ts","../src/campaign/presets/playback.ts","../src/campaign/presets/segmented-profile-matrix.ts","../src/campaign/run-dir.ts","../src/campaign/scenario-selection.ts","../src/campaign/score-utils.ts","../src/campaign/single-run-lock.ts","../src/campaign/surface-identity.ts","../src/campaign/transient-failure.ts","../src/campaign/upstream-evaluators.ts","../src/campaign/worktree/index.ts"],"mappings":";;;;;;;;;;;;;;;;;UAWiB,6BAA6B;EAC5C;EACA,YAAY;EACZ,YAAY;EACZ,yBAAyB;EACzB,kBAAkB;;UAGH;EACf,mBAAmB;EACnB,QAAQ;EACR,MAAM;;UAGS;EACf,QAAQ;IACN;IACA,YAAY;IACZ;IACA,QAAQ;MACN,QAAQ;;iBAGE,iCACd,SAAS,2CAET,SAAS,gBACT,UAAU,sBACV,SAAS,oBACN,QAAQ;iBAgBG,4BAA4B,YAC1C,sBACA;;;;KCtDU;;UAGK;EACf;EACA;;EAEA;;;UAIe;EACf;EACA;EACA;EACA;;;UAIe;EACf;;EAEA;;EAEA;;;;;;;UAQe;EACf;EACA;;EAEA;EACA,cAAc;EACd;EACA;;;;;;EAMA,MAAM;EACN,mBAAmB;;EAEnB;;UAGe;EACf;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;;EAEA,kCAAkC;;EAElC;;UAGe,qCACf,aAAa,sBAAsB;EAEnC,qBAAqB;EACrB,qBAAqB;EACrB,eAAe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf,aAAa;EACb;EACA;EACA;EACA;;KAGU;UAWK;EACf;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,WAAW;EACX,SAAS;EACT,OAAO;EACP,OAAO,eAAe;EACtB,QAAQ;EACR,QAAQ;;EAER,sBAAsB;;EAEtB,aAAa;;UAGE;EACf;EACA;EACA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,cAAc,eAAe;;UAGd;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;UAGe;EACf,SAAS;EACT;EACA;EACA;EACA;EACA,eAAe;EACf,gBAAgB;;KAGN;UAYK;EACf;EACA,SAAS;EACT;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA,4BAA4B,iCAAiC;EAC7D,wBAAwB,eAAe;EACvC,QAAQ;EACR,QAAQ;EACR,aAAa;EACb,eAAe;;UAGA;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;UAGM;;EAEf;EACA;;KAGU;UAaK;EACf;EACA;EACA;EACA;EACA;EACA,uBAAuB;EACvB;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA,YAAY;EACZ;;;UAIe;EACf;EACA;EACA;EACA;EACA,OAAO;;UAGQ;;EAEf;;EAEA;EACA;;EAEA;EACA;;EAEA,gBAAgB;;EAEhB,OAAO;;UAGQ;EACf,YAAY;EACZ,YAAY;EACZ,kBAAkB;;UAGH,8BACf,aAAa,sBAAsB;EAEnC;EACA;EACA;EACA;;EAEA,MAAM;EACN,iBAAiB;EACjB,iBAAiB;EACjB,YAAY;EACZ,UAAU;EACV,YAAY;;;;;;;;;iBCnRE,gCAAgC,aAAa,qBAC3D,OAAO,qCAAqC,QAC3C,8BAA8B;;;;cClDpB,0BAA0B;;cAG1B,mCAAmC;;cAGnC,kCAAkC;;;KCDnC;UAEK;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;;UAGe,4BAA4B;EAC3C;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;;EAEf,aAAa;;EAEb;;UAGe,wCAAwC;EACvD;;UAGe,0BAA0B,6BACjC,KACN,uBAAuB,qBAAqB;EAG9C;EACA,aAAa;EACb;EACA;EACA,UAAU;;KAGA,qBAAqB;EAC/B,UAAU,MAAM,KAAK;;;iBAIP,qBAAqB;;;;;iBA4BrB,gBACd,kBACA,cACA,UAAS,yBACR;;iBA4Ca,yBACd,kBACA,UAAS,kCACR;;;;;iBAuBa,mBAAmB,qBACjC,SAAS,0BAA0B,aAClC;;;UChJc,0BAA0B,kBAAkB,WAAW;EACtE,WAAW;;;;;EAKX;;;;;;iBAuBc,mBAAmB,WAAW,kBAAkB,UAC9D,SAAS,0BAA0B,aAClC,KAAK,WAAW;;;;;;;;;;;;;;;;;;;;;;;;;;;;;UC9BF;WACN;WACA,MAAM,SAAS;;;UAIT;;WAEN;;WAEA;WACA,gBAAgB;;UAGV;;WAEN;;WAEA;;UAGM;;WAEN;;WAEA,eAAe;;WAEf,eAAe;;;;;;;;iBASV,oBACd,mBAAmB,iBACnB,OAAM,6BACL;UAsCc;;WAEN;;WAEA;;;;;;;;;iBAUK,2BACd,cACA,MAAM,KAAK,0DACV;;;UCjFc;;EAEf;;;EAGA;;EAEA;;;cAIW,kCAAkC;WAE3B;EADlB,YACkB,cAChB;;;;;;cAiBS,kCAAkC;mBAIhB;mBAHZ;mBACA;EAEjB,YAA6B,SAAS;EAKhC,QAAQ,OAAO,uBAAuB;EAYtC,OAAO,MAAM,4BAA4B,QAAQ;EA8CjD,QAAQ;IACZ;IACA;IACA,UAAU;IACV,SAAS,OAAO;;UAkCV;UAiCA;UAmBA;UAkBA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBCtNM,eAAe;;;UCpBd;;EAEf;;EAEA;;EAEA;;EAEA,UAAU;;EAEV;;EAEA;;EAEA,eAAe;;;;;;;;;;;;;;;;;;iBAmBD,qCACd,SAAS,wCACR;;;;;;cC+BU,2BAA2B;EACtC,YAAY;;;;;KAQF,kBAAkB,kBAAkB,UAAU,cACxD,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;UAEI,wBAAwB,kBAAkB,UAAU;;EAEnE,UAAU;;EAEV,WAAW;;EAEX,UAAU,kBAAkB,WAAW;;EAEvC,SAAS,YAAY,WAAW;;;EAGhC;;;EAGA;;;EAGA;;;;EAIA;;EAEA,WAAW;;EAEX;;EAEA;;;;;;;;EAQA;;;EAGA;;EAEA;;;EAGA;;EAEA;;EAEA,eAAe;EACf,gBAAgB;;;EAGhB,UAAU;;EAEV,YAAY;;;EAGZ,aAAa,UAAU;;;;EAIvB;;;;;;;EAOA,cACE,UAAU,WACV,UAAU;IACL;IAAgB;;;;;;;EAMvB,aAAa;IAAS,SAAS;IAAc,UAAU;IAAW;;;EAElE;;EAEA;;UAGe;EACf;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA,gBAAgB;;;EAGhB,WAAW;;UAGI;EACf;EACA;;UAGe,uBAAuB,WAAW,kBAAkB;EACnE;EACA;;;;EAIA,SAAS;EACT,WAAW,eAAe;EAC1B,YAAY,eAAe;;EAE3B,YAAY,eAAe;;EAE3B,WAAW;;EAEX,WAAW,eAAe,eAAe,WAAW;;;;;iBAoOhC,iBAAiB,kBAAkB,UAAU,WACjE,MAAM,wBAAwB,WAAW,aACxC,QAAQ,uBAAuB,WAAW;;;;;UC9Z5B;;EAEf;;EAEA,UAAU;;;;;;;UAQK,kBAAkB;;EAEjC;;EAEA,OAAO;;EAEP,cAAc;;;UAIC,wBAAwB;EACvC,SAAS;;;;;;;;;;;UAYM,eAAe,eAAe,YAAY;EACzD,IAAI,OAAO,QAAQ,KAAK,kBAAkB,iBAAiB;;;;;;;iBAQ7C,qBAAqB,eAAe,WAClD,QAAQ,eAAe,UACtB,kBAAkB,QAAQ;;UAQZ,yBAAyB;EACxC;;;;;;;;iBASoB,eACpB,OAAO,WACP,OAAO,eACP,kBAAkB,qBACjB,QAAQ;;UAUM;EACf;EACA;EACA;EACA;EACA;EACA;;;;;;;iBAQc,oBAAoB,mBAAmB,qBAAqB;;UAkB3D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;iBAIc,kBAAkB,eAAe,kBAAkB;UAwBlD;;EAEf;;EAEA,OAAO;;EAEP;;;;;;;;;iBAkBc,yBACd,eAAe,iBACf,OAAM;;;cC1KK,oCAAoC;EAC/C,YAAY;;UAKG;EACf;EACA;EACA;EACA;EACA;;UAGe,kBAAkB,kBAAkB,UAAU;WACpD;WACA;WACA;WACA;WACA,mBAAmB;WACnB,oBAAoB;WACpB,iBAAiB,YAAY,WAAW;WACxC;WACA;WACA,UAAU,YAAY,wBAAwB,WAAW;WACzD;;WAEA;WACA,WAAW,YAAY,wBAAwB,WAAW;WAC1D;WACA;WACA,aAAa,UAAU;WACvB,cACP,UAAU,WACV,UAAU;IACL;IAAgB;;WACd,eAAe;;UAGT,+BAA+B,kBAAkB,UAAU,mBAClE,KACN,wBAAwB,WAAW;;EAgBrC;;UAGe;EACf;EACA;EACA;EACA;EACA;;UAGe,+BAA+B,kBAAkB,UAAU;EAC1E,MAAM,kBAAkB,WAAW;;EAEnC;;EAEA,yBAAyB;EACzB,UAAU,kBAAkB,WAAW;EACvC;EACA,UAAU;EACV;EACA;EACA;EACA,eAAe,wBAAwB,WAAW;EAClD,gBAAgB,wBAAwB,WAAW;EACnD,YAAY;;UAGG,2BAA2B,WAAW,kBAAkB;EACvE;EACA;EACA,QAAQ,uBAAuB,WAAW;EAC1C,UAAU;;UAGK,6BAA6B,kBAAkB,UAAU;EACxE,MAAM,kBAAkB,WAAW;EACnC;EACA,UAAU;EACV;EACA;EACA,YAAY;;UAGG,6BAA6B,WAAW,kBAAkB,kBACjE,uBAAuB,WAAW;EAC1C,UAAU;;iBAGI,wBAAwB,kBAAkB,UAAU,WAClE,MAAM,+BAA+B,WAAW,aAC/C,kBAAkB,WAAW;iBA0HV,wBAAwB,kBAAkB,UAAU,WACxE,MAAM,+BAA+B,WAAW,aAC/C,QAAQ,2BAA2B,WAAW;iBAkD3B,sBAAsB,kBAAkB,UAAU,WACtE,MAAM,6BAA6B,WAAW,aAC7C,QAAQ,6BAA6B,WAAW;;;;;;;;iBCrTnC;;;;;iBAQA,cAAc,gBAAgB;;;;;;;;;;;;;;;;;UCD7B;EACf;;EAEA;;UAGe;EACf;;EAEA;;EAEA;EACA;;EAEA;;;;;;;;;;iBA4Cc,oBACd,SAAS,kBACT;EAAS;IACR;;;;;;;;;;;iBA0Ba,qBACd,SAAS,kBACT,WACA;EAAS;;;;;;;iBC3FK,sBAAsB,WAAW,kBAAkB,UACjE,UAAU,eAAe,WAAW;;;;iBA0CtB,gBAAgB,sBAAsB;UAWrC;;EAEf,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;;iBAK5D,kBAAkB,WAAW,kBAAkB,UAC7D,UAAU,eAAe,WAAW,aACnC;;;;;;;;;;;;;;;;;;UC5Dc;;WAEN;;WAEA;;WAEA;;WAEA;;UAGM;;EAEf;;;;;;;iBAwBc,qBAAqB,MAAM,uBAAuB;;;;iBCpDlD,0BAA0B,2BAA2B,WAAW;;iBAmChE,uBAAuB,2BAA2B,WAAW;;iBA8B7D,iCAAiC,SAAS;;;;iBAa1C,4BAA4B,SAAS;;iBAgBrC,mBAAmB,SAAS;;iBAW5B,YAAY,SAAS;;iBAKrB,kBACd,eAAe,gBACf,iBAAiB;;;;;;;;;;;;;;;;;;;;;UCpGF;;;;;;;WAON;;WAEA,yBAAyB;;;;;;iBAYpB,4BACd,oCACA,OAAM;;;UCnCS;EACf;EACA;EACA;;UAGe,qBAAqB,gBAAgB;EACpD;EACA;EACA;EACA,SACE,QAAQ,SACR,SAAS,4BACR,QAAQ;;UAGI;EACf;EACA;EACA,WAAW;;KAGD,oBAAoB,eAAe,4BAC7C,OAAO,QACP,SAAS,8BACN,qBAAqB,QAAQ;UAEjB;WACN,QAAQ;WACR;;KAGN,sBAAsB,WAAW,KACpC,iBAAiB;EAGjB;;UAGQ,qBAAqB,kBAAkB;EAC/C;EACA;EACA;EACA,aAAa,UAAU;;EAEvB,eAAe;;iBAGD,sBACd,gBAAgB,yBAChB,WACA,kBAAkB,WAAW,UAE7B,WAAW,qBAAqB,UAChC,SAAS,qBAAqB;EAC5B,SAAS;IAAS,UAAU;IAAW,UAAU;MAAc;EAC/D,WAAW,sBAAsB;IAElC,YAAY,WAAW;iBA8CV,qBACd,eAAe,yBACf,WACA,kBAAkB,WAAW,UAE7B,QAAQ,oBAAoB,SAC5B,SAAS,qBAAqB;EAC5B;EACA,SAAS;IAAS,UAAU;IAAW,UAAU;MAAc;;EAEzD;EAAc;;EACd;EAAa,UAAU,sBAAsB;KAEpD,YAAY,WAAW;;;KCxFrB,qBAAqB;KACrB,iBAAiB,SAAS;KAC1B,aAAa,gBAAgB,aAAa,MAAM,mBAAmB;UAcvD;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;UAGM;;EAEf,OAAO;IAAQ;IAAiB;MAAkB,QAAQ;;;EAG1D,SAAS,UAAU,UAAU,kBAAkB,QAAQ;;EAEvD,QAAQ,UAAU,WAAW;;;cAIlB,6BAA6B;WAG7B;EAFX,YACE,iBACS;;UAOI;;EAEf;;EAEA;;EAEA;;;;EAIA,MAAM;;UAobS;;EAEf;;EAEA;;EAEA;;;EAGA,YAAY;;;;;iBAoJE,mBAAmB,MAAM,4BAA4B;;;;iBA0FrD,kBACd,SAAS,aACT,uBACC;;;iBAOa,oBAAoB,SAAS,aAAa"}
package/dist/index.d.ts CHANGED
@@ -2,48 +2,48 @@ import { a as JudgeError, c as ValidationError, i as ConfigError, n as AgentEval
2
2
  import { C as verifyAgentProfileCell, h as agentProfileCellKey, i as AgentProfileCellInput, l as AgentProfileJson, m as agentProfileCellHashMaterial, p as AgentProfileSourceInput, r as AgentProfileCell, s as AgentProfileDimensionValue, t as AGENT_PROFILE_KINDS, v as buildAgentProfileCell, x as toAgentProfileJson, y as groupRunsByAgentProfileCell } from "./agent-profile-cell-BkcRDikH.js";
3
3
  import { C as defineEquivalenceCheck, S as buildEquivalenceRecord, _ as StrategyChecker, a as CheckerIdentity, b as VerificationStrategyProfile, c as EquivalenceCheckDefinition, d as EquivalenceCheckerInput, f as EquivalenceCheckerResult, g as EquivalenceRecord, h as EquivalenceProtocolError, i as equivalenceVerdict, l as EquivalenceCheckSpec, m as EquivalenceObligationStatus, n as VerdictCertification, o as CheckerOutcome, p as EquivalenceObligation, r as certificationEvidenceDigest, s as EquivalenceArm, t as DefaultVerdict, u as EquivalenceChecker, v as VERIFICATION_STRATEGIES, w as runEquivalenceCheck, x as VerificationStrategySource, y as VERIFICATION_STRATEGY_SOURCES } from "./verdict-E4eRNf7-.js";
4
4
  import { a as MultiLayerVerifier, c as VerifyOptions, i as LayerStatus, l as gradeSemanticStatus, n as Layer, o as Severity, r as LayerResult, s as VerificationReport, t as Finding } from "./multi-layer-verifier-BUaQ4C17.js";
5
- import { A as SemanticConceptJudgeInput, D as ConceptFinding, E as createDspyRlmTraceEngine, F as RunScoreWeights, I as aggregateRunScore, L as clamp01, M as SemanticConceptJudgeResult, N as runSemanticConceptJudge, O as ConceptSpec, P as RunScore, T as DspyRlmTraceEngineOptions, a as FAILURE_MODE_KIND_SPEC, d as FindingsStore, f as PersistedFinding, j as SemanticConceptJudgeOptions, k as SEMANTIC_CONCEPT_JUDGE_VERSION, l as DiffPolicy, m as diffFindings, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubject, y as FindingSubjectKind } from "./index-aHDnhBC7.js";
5
+ import { A as SemanticConceptJudgeInput, D as ConceptFinding, E as createDspyRlmTraceEngine, F as RunScoreWeights, I as aggregateRunScore, L as clamp01, M as SemanticConceptJudgeResult, N as runSemanticConceptJudge, O as ConceptSpec, P as RunScore, T as DspyRlmTraceEngineOptions, a as FAILURE_MODE_KIND_SPEC, d as FindingsStore, f as PersistedFinding, j as SemanticConceptJudgeOptions, k as SEMANTIC_CONCEPT_JUDGE_VERSION, l as DiffPolicy, m as diffFindings, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubject, y as FindingSubjectKind } from "./index-CTKpu9ry.js";
6
6
  import { C as PendingCostCall, D as costForUsage, E as costForTokenPricing, O as modelPriceKey, S as PaidCallResult, T as RunPaidCallInput, _ as CostReservationExceededError, a as CostChannel, b as CustomTokenPricing, c as CostLedgerHandle, d as CostLedgerPersistenceError, f as CostLedgerSummary, g as CostReceiptInput, h as CostReceiptCaptureError, i as CostCeilingReachedError, l as CostLedgerOptions, m as CostReceipt, n as CostAccountingIncompleteError, o as CostLedger, p as CostProvenance, r as CostCallConflictError, s as CostLedgerFilter, t as ChannelRollup, u as CostLedgerPersistence, v as CostResult, w as PendingCostCallView, x as MaximumCharge, y as CostUsage } from "./cost-ledger-DbQdN3nO.js";
7
7
  import { C as TraceEvent, O as isToolSpan, S as ToolSpan, T as isLlmSpan, _ as Span, a as FAILURE_CLASSES, c as JudgeSpan, f as Run, h as RunStatus, l as LlmSpan, n as BudgetLedgerEntry, o as FailureClass, r as BudgetSpec, s as GenericSpan, t as Artifact, w as isJudgeSpan } from "./schema-BtVldJ3T.js";
8
8
  import { _ as validateRunRecord, a as RunRecord, c as RunTaskFailure, d as UNKNOWN_MODEL, f as isRunRecord, g as runTaskScore, h as roundTripRunRecord, i as RunOutcome, l as RunTerminalOutcome, m as parseRunRecordSafe, n as RunCostProvenance, o as RunRecordValidationError, p as modelHasSnapshot, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage } from "./run-record-DVV82Gwh.js";
9
- import { a as ExtractedUsage, l as extractUsageFromSse, r as captureFetchToRawSink, s as extractUsage } from "./capture-fetch-BAx3Gntt.js";
10
- import { $ as checkServedModel, A as LlmClientOptions, B as isTransientLlmError, D as LlmCallRequest, E as LlmCallMetadata, F as assertLlmRoute, G as AssertServedModelOptions, H as probeLlm, I as callLlm, J as ServedModelCheck, K as ModelSubstitutionError, L as callLlmJson, M as LlmResponseError, N as LlmRouteRequirements, O as LlmCallResult, P as LlmUsage, Q as assertServedModels, R as costReceiptFromLlm, T as LlmCallError, U as stripFencedJson, V as maximumChargeForLlmRequest, W as AssertCrossFamilyServedOptions, X as assertCrossFamilyServed, Y as ServedModelVerdict, Z as assertServedModel, at as judgeFamily, c as PersonaConfig, ct as InMemoryRawProviderSink, d as Scenario, et as servedModelAcceptable, f as ChatCallOpts, h as ChatResponse, i as JudgeFn, it as assertCrossFamily, j as LlmMessage, k as LlmClient, l as ProductClientConfig, m as ChatRequest, mt as RawProviderSink, n as CompletionCriterion, nt as CrossFamilyError, o as JudgeRubric, ot as FileSystemRawProviderSink, p as ChatClient, pt as RawProviderEvent, q as ServedCrossFamilyError, r as DriverState, rt as JudgeFamily, s as JudgeScore, t as CheckResult, tt as AssertCrossFamilyOptions, u as RouteMap, ut as NoopRawProviderSink, v as CreateChatClientOpts, w as createChatClient, z as costReceiptFromLlmError } from "./types-DOhGq4S1.js";
9
+ import { a as ExtractedUsage, l as extractUsageFromSse, r as captureFetchToRawSink, s as extractUsage } from "./capture-fetch-BBVFzhkk.js";
10
+ import { $ as assertServedModels, A as LlmClientOptions, B as isTransientLlmError, D as LlmCallRequest, E as LlmCallMetadata, F as assertLlmRoute, G as AssertServedModelOptions, H as probeLlm, I as callLlm, J as ServedModelCheck, K as ModelSubstitutionError, L as callLlmJson, M as LlmResponseError, N as LlmRouteRequirements, O as LlmCallResult, P as LlmUsage, Q as assertServedModel, R as costReceiptFromLlm, T as LlmCallError, U as stripFencedJson, V as maximumChargeForLlmRequest, W as AssertCrossFamilyServedOptions, X as ServedModelVerdict, Y as ServedModelPolicy, Z as assertCrossFamilyServed, at as assertCrossFamily, c as PersonaConfig, d as Scenario, dt as NoopRawProviderSink, et as checkServedModel, f as ChatCallOpts, h as ChatResponse, ht as RawProviderSink, i as JudgeFn, it as JudgeFamily, j as LlmMessage, k as LlmClient, l as ProductClientConfig, lt as InMemoryRawProviderSink, m as ChatRequest, mt as RawProviderEvent, n as CompletionCriterion, nt as AssertCrossFamilyOptions, o as JudgeRubric, ot as judgeFamily, p as ChatClient, q as ServedCrossFamilyError, r as DriverState, rt as CrossFamilyError, s as JudgeScore, st as FileSystemRawProviderSink, t as CheckResult, tt as servedModelAcceptable, u as RouteMap, v as CreateChatClientOpts, w as createChatClient, z as costReceiptFromLlmError } from "./types-jUBXJ7Iz.js";
11
11
  import { a as RunFilter, i as InMemoryTraceStore, n as FileSystemTraceStore, o as SpanFilter, s as TraceStore, t as EventFilter } from "./store-CT9YIIve.js";
12
12
  import { i as TraceEmitter, r as SpanHandle } from "./emitter-DGQGoLyj.js";
13
- import { a as RunIntegrityReport, o as assertRunCaptured, t as RunIntegrityError } from "./integrity-CGfpTE5-.js";
14
- import { A as createBoundedTraceAnalysisStore, B as OtlpSpan, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, F as redactString, M as REDACTION_VERSION, O as analyzeTraces, P as RedactionRule, R as OtlpExport, V as exportRunAsOtlp, _ as buildTraceInsightPrompt, b as domainEvidencePattern, g as buildTraceInsightContext, j as DEFAULT_REDACTION_RULES, k as OtlpFlatLine, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, r as OtlpFileTraceStore, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope } from "./store-tool-spans-D5FhM0_A.js";
13
+ import { a as RunIntegrityReport, o as assertRunCaptured, t as RunIntegrityError } from "./integrity-B0dZ96EO.js";
14
+ import { A as createBoundedTraceAnalysisStore, B as OtlpSpan, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, F as redactString, M as REDACTION_VERSION, O as analyzeTraces, P as RedactionRule, R as OtlpExport, V as exportRunAsOtlp, _ as buildTraceInsightPrompt, b as domainEvidencePattern, g as buildTraceInsightContext, j as DEFAULT_REDACTION_RULES, k as OtlpFlatLine, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, r as OtlpFileTraceStore, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope } from "./store-tool-spans-Br2_IUhm.js";
15
15
  import { y as OUTPUT_VALUE } from "./attribute-vocabulary-DLJ6303h.js";
16
16
  import { a as judgeSpans, c as runsForScenario, n as argHash } from "./query-CJ_DX8vl.js";
17
- import { A as SearchSpanResult, B as ViewSpansResult, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, V as ViewTraceOversized, _ as ProposalFinding, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, i as AnalystFinding, j as SearchTraceResult, k as QueryTracesPage, l as AnalystRunResult, n as AnalystContext, p as EvidenceRef, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId, z as TraceAnalystTraceSummary } from "./types-BohHewKK.js";
18
- import { a as createTraceAnalyst, i as TraceAnalystDefinition, n as buildDefaultAnalystRegistry, r as CreateTraceAnalystOptions, t as DefaultAnalystRegistryOptions } from "./default-registry-BUtRMRKk.js";
19
- import { i as ExactAnalystRunEvent, l as ExactCapableAnalyst, o as ExactAnalystRunResult } from "./exact-types-rdKFzEnK.js";
20
- import { c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, r as AnalystRegistryOptions, s as ExactRegistryRunOpts } from "./registry-CaqJIDdN.js";
17
+ import { A as SearchSpanResult, B as ViewSpansResult, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, V as ViewTraceOversized, _ as ProposalFinding, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, i as AnalystFinding, j as SearchTraceResult, k as QueryTracesPage, l as AnalystRunResult, n as AnalystContext, p as EvidenceRef, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId, z as TraceAnalystTraceSummary } from "./types-B2NsbrNy.js";
18
+ import { a as createTraceAnalyst, i as TraceAnalystDefinition, n as buildDefaultAnalystRegistry, r as CreateTraceAnalystOptions, t as DefaultAnalystRegistryOptions } from "./default-registry-Cw0Ohdoj.js";
19
+ import { i as ExactAnalystRunEvent, l as ExactCapableAnalyst, o as ExactAnalystRunResult } from "./exact-types-ccQAyut1.js";
20
+ import { c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, r as AnalystRegistryOptions, s as ExactRegistryRunOpts } from "./registry-BQwrSYpC.js";
21
21
  import { a as ProfileAxisSpec, c as expandProfileAxes, i as HarnessType, l as harnessAxisOf, n as CODING_HARNESSES, o as agentProfileHash, r as HARNESS_NATIVE_MODEL, s as agentProfileId, t as AgentProfile } from "./agent-profile-B9_GGsG8.js";
22
- import { S as JudgeConfig, a as CampaignResult, w as JudgeScore$1 } from "./types-C1Bmeb8X.js";
23
- import { _ as SatisfiedBy, a as ArtifactEventLike, b as createLlmCorrectnessChecker, c as ToolCallEventLike, d as CompletionVerdict, f as CorrectnessChecker, g as RequirementCheck, h as ProducedState, i as summarizeBackendIntegrity, l as extractProducedState, m as ProducedProposal, n as BackendIntegrityReport, o as ProposalEventLike, p as LlmCorrectnessCheckerOpts, r as assertRealBackend, s as RuntimeEventLike, t as BackendIntegrityError, u as CompletionRequirement, v as TaskGold, x as verifyCompletion, y as completionVerdict } from "./backend-integrity-Bz8nSrRE.js";
22
+ import { S as JudgeConfig, a as CampaignResult, w as JudgeScore$1 } from "./types-DdFNuyxQ.js";
23
+ import { _ as SatisfiedBy, a as ArtifactEventLike, b as createLlmCorrectnessChecker, c as ToolCallEventLike, d as CompletionVerdict, f as CorrectnessChecker, g as RequirementCheck, h as ProducedState, i as summarizeBackendIntegrity, l as extractProducedState, m as ProducedProposal, n as BackendIntegrityReport, o as ProposalEventLike, p as LlmCorrectnessCheckerOpts, r as assertRealBackend, s as RuntimeEventLike, t as BackendIntegrityError, u as CompletionRequirement, v as TaskGold, x as verifyCompletion, y as completionVerdict } from "./backend-integrity-DOCa_QrR.js";
24
24
  import { i as DatasetSplit, n as DatasetManifest, r as DatasetScenario, t as Dataset } from "./dataset-CJjKqQfA.js";
25
25
  import { a as ContinuousCalibrationResult, c as calibrateJudge, d as verbosityBias, i as ContinuousAgreementOptions, l as calibrateJudgeContinuous, n as CandidateScore, o as GoldenItem, r as ContinuousAgreement, s as VerbosityBiasResult, t as CalibrationResult, u as continuousAgreement } from "./judge-calibration-C5CbMYce.js";
26
- import { a as CorpusAgreementPerDimension, c as corpusInterRaterAgreement, i as CorpusAgreementOptions, l as corpusInterRaterAgreementFromJudgeScores, n as SeriesConvergenceResult, o as CorpusAgreementReport, r as analyzeSeries, s as CorpusScoreRecord, t as SeriesConvergenceOptions, u as interRaterReliability } from "./series-convergence-D2fsoJ2w.js";
26
+ import { a as CorpusAgreementPerDimension, c as corpusInterRaterAgreement, i as CorpusAgreementOptions, l as corpusInterRaterAgreementFromJudgeScores, n as SeriesConvergenceResult, o as CorpusAgreementReport, r as analyzeSeries, s as CorpusScoreRecord, t as SeriesConvergenceOptions, u as interRaterReliability } from "./series-convergence-BxKEgBwA.js";
27
27
  import { n as bonferroni, r as holm, t as benjaminiHochberg } from "./multiplicity-DIWHvysC.js";
28
28
  import { A as RiskDifferenceResult, C as EProcessOptions, D as ExactRiskDifferenceResult, E as eProcess, F as pairedRiskDifference, I as pairedRiskDifferenceExact, L as pairedRiskDifferenceScore, M as isBinaryOutcomeVector, N as mcnemar, O as McNemarResult, P as pairedBinaryScale, R as passAtK, S as EProcess, T as EProcessStep, _ as PairedCorrectness, b as pairArms, d as MatchedRunRecordPair, f as PairArmsOptions, g as PairedArmsComparison, h as PairedArmRow, i as canonicalize, j as ScoreRiskDifferenceResult, k as ProportionInterval, l as ComparePairedArmsOptions, m as PairRunRecordsResult, o as hashJson, p as PairArmsResult, u as MatchedPair, v as PairedMetricDelta, w as EProcessState, x as pairRunRecords, y as comparePairedArms, z as wilson } from "./pre-registration-zFSLEiFU.js";
29
29
  import { _ as pairedSignTest, a as PairedPromotionDecision, c as BOOTSTRAP_GATE_MIN_N, d as PairedBootstrapResult, f as PairedSignTestResult, g as pairedDeltaTieFraction, h as pairedBootstrap, i as PairedMcNemarEvidence, l as DECISION_PAIRED_DELTA_STATISTIC, m as SignTestAlternative, n as PairedDecisionShape, o as PairedPromotionDecisionOptions, p as PairedTTestResult, r as PairedDecisionStatistic, s as decidePairedPromotion, t as PairedDecisionMethod, u as PairedBootstrapOptions, v as pairedTTest } from "./paired-promotion-decision-CGzg0cI_.js";
30
- import { _ as requiredSampleSize, c as computeExperimentStats, f as mulberry32, g as requiredPairedSampleSize, h as pairedMde, m as mcnemarRequiredN, n as ExperimentRep, o as ImprovementThresholds, p as mcnemarPower, r as ExperimentStats, s as ImprovementVerdictResult, u as improvementVerdict } from "./experiment-tracker-D9VHwRm-.js";
30
+ import { _ as requiredSampleSize, c as computeExperimentStats, f as mulberry32, g as requiredPairedSampleSize, h as pairedMde, m as mcnemarRequiredN, n as ExperimentRep, o as ImprovementThresholds, p as mcnemarPower, r as ExperimentStats, s as ImprovementVerdictResult, u as improvementVerdict } from "./experiment-tracker-0MhuPArU.js";
31
31
  import { C as RankTestMethod, D as WilcoxonSignedRankResult, E as WILCOXON_EXACT_MAX_N, O as mannWhitneyU, S as MannWhitneyResult, T as RankTestOptions, _ as bootstrapCi, a as ReleaseConfidenceIssue, b as MANN_WHITNEY_EXACT_MAX_STATES, f as evaluateReleaseConfidence, g as Verdict, i as ReleaseConfidenceInput, k as wilcoxonSignedRank, m as BootstrapResult, o as ReleaseConfidenceMetrics, p as BootstrapOptions, s as ReleaseConfidenceScorecard, t as ActionableSideInfo, u as ReleaseTraceEvidence, w as RankTestMethodRequest, x as MANN_WHITNEY_EXACT_MAX_WORK, y as DEFAULT_PERMUTATIONS } from "./release-confidence-4XrqlpFD.js";
32
32
  import { C as HeldOutGateConfig, S as HeldOutGate, T as SplitCoverage, _ as paretoChart, a as ParetoPoint, b as GateDecision, d as ResearchReportOptions, g as gainHistogram, h as SummaryTableRow, m as SummaryTableOptions, n as GainDistributionFigureSpec, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, w as HeldOutGateRejectionCode, x as GateEvidence, y as summaryTable } from "./summary-report-B__Y5ub3.js";
33
33
  import { a as FailureContext, i as FailureClassification, o as FailureRule, r as failureClusterView, s as classifyFailure } from "./failure-cluster-BLURuWG4.js";
34
34
  import { o as InsightReport } from "./insight-report-BeT8KCgI.js";
35
- import { C as analyzeRuns, b as AnalyzeRunsOptions, d as RawAnalystFinding, i as TraceAnalysisEngineResult, n as TraceAnalysisEngine } from "./engine-CIy18RTX.js";
35
+ import { C as analyzeRuns, b as AnalyzeRunsOptions, d as RawAnalystFinding, i as TraceAnalysisEngineResult, n as TraceAnalysisEngine } from "./engine-BLzhNzoY.js";
36
36
  import { o as MintedRolloutLine } from "./schema-Cef2cFmb.js";
37
- import { i as BenchmarkEvaluation } from "./types-ColZlZtc.js";
38
- import { A as RedTeamCategory, B as CanaryReport, F as scoreRedTeamOutput, I as CanaryAlert, L as CanaryEvaluation, M as RedTeamReport, Mn as llmJudge, N as redTeamDataset, O as DEFAULT_RED_TEAM_CORPUS, P as redTeamReport, R as CanaryKind, V as runCanaries, _n as RunCampaignOptions, gn as CampaignCellFailureReceipt, j as RedTeamFinding, jn as LlmJudgeOptions, k as RedTeamCase, vn as runCampaign, z as CanaryOptions } from "./provenance-DA-Pmyfv.js";
37
+ import { i as BenchmarkEvaluation } from "./types-CLAwnY-L.js";
38
+ import { A as RedTeamCategory, B as CanaryReport, F as scoreRedTeamOutput, I as CanaryAlert, L as CanaryEvaluation, M as RedTeamReport, Mn as llmJudge, N as redTeamDataset, O as DEFAULT_RED_TEAM_CORPUS, P as redTeamReport, R as CanaryKind, V as runCanaries, _n as RunCampaignOptions, gn as CampaignCellFailureReceipt, j as RedTeamFinding, jn as LlmJudgeOptions, k as RedTeamCase, vn as runCampaign, z as CanaryOptions } from "./provenance-oA4-zUqm.js";
39
39
  import { a as paretoFrontier, i as dominates, n as Objective, r as ParetoResult } from "./pareto-BqNW3LJR.js";
40
40
  import { n as SandboxDriver } from "./sandbox-harness-BlSOu4LX.js";
41
- import { a as DefinedAgentEval, c as SelfImproveOptions, f as selfImprove, i as DefineAgentEvalOptions, o as defineAgentEval, u as SelfImproveResult } from "./define-agent-eval-BWuk2O_N.js";
41
+ import { a as DefinedAgentEval, c as SelfImproveOptions, f as selfImprove, i as DefineAgentEvalOptions, o as defineAgentEval, u as SelfImproveResult } from "./define-agent-eval-CEQWL9Hy.js";
42
42
  import { a as PairedEvalueStep, c as SequentialDecision, i as PairedEvalueSequence, l as evaluateInterimReleaseConfidence, n as InterimReleaseConfidenceInput, r as PairedEvalueOptions, t as InterimReleaseConfidence, u as pairedEvalueSequence } from "./sequential-BhsrMupG.js";
43
43
  import { n as TrajectoryStep, r as buildTrajectory, t as Trajectory } from "./trajectory-YC15QDYQ.js";
44
44
  import { a as runCounterfactual, i as CounterfactualRunner, n as CounterfactualMutation, r as CounterfactualResult, t as CounterfactualContext } from "./counterfactual--bpysZF0.js";
45
- import { A as AnalystFindingDigest, B as ControlRunResult, C as createFeedbackTrajectory, D as renderPreferenceMemoryMarkdown, E as feedbackTrajectoryToOptimizerRow, F as ControlActionOutcome, G as StopDecision, H as ControlRuntimeError, I as ControlBudget, J as subjectiveEval, K as objectiveEval, L as ControlContext, M as AnalystRunDigest, N as analystFindingDigest, O as summarizePreferenceMemory, P as analystRunDigest, R as ControlDecision, S as controlRunToFeedbackTrajectory, T as feedbackTrajectoriesToOptimizerRows, U as ControlSeverity, V as ControlRuntimeConfig, W as ControlStep, _ as PreferenceMemoryEntry, a as FeedbackLabel, b as analystRunToReviewRequests, c as FeedbackOptimizerRow, d as FeedbackTask, f as FeedbackTrajectory, g as InMemoryFeedbackTrajectoryStore, h as FileSystemFeedbackTrajectoryStore, i as FeedbackAttempt, j as AnalystReviewDecision, k as withAssignedFeedbackSplit, l as FeedbackOutcome, m as FeedbackTrajectoryStore, n as AnalystReviewRequest, o as FeedbackLabelKind, p as FeedbackTrajectoryFilter, q as runAgentControlLoop, r as FeedbackArtifactType, s as FeedbackLabelSource, t as AnalystFeedbackTrajectoryOptions, u as FeedbackSplitPolicy, v as ProposedSideEffect, w as feedbackTrajectoriesToDatasetScenarios, x as assignFeedbackSplit, y as analystRunToFeedbackTrajectory, z as ControlEvalResult } from "./feedback-trajectory-niGeR6uJ.js";
46
- import { a as SteeringChange, c as CampaignRunContext, d as CampaignVariant, f as EvalCampaignOptions, h as runEvalCampaign, i as Researcher, l as CampaignRunOutcome, m as FailedRun, n as ExperimentResult, o as CampaignFactoryParams, p as EvalCampaignResult, r as FailureMode, s as CampaignIntegrityPolicy, t as ExperimentPlan, u as CampaignScenario } from "./researcher-BySlpla_.js";
45
+ import { A as AnalystFindingDigest, B as ControlRunResult, C as createFeedbackTrajectory, D as renderPreferenceMemoryMarkdown, E as feedbackTrajectoryToOptimizerRow, F as ControlActionOutcome, G as StopDecision, H as ControlRuntimeError, I as ControlBudget, J as subjectiveEval, K as objectiveEval, L as ControlContext, M as AnalystRunDigest, N as analystFindingDigest, O as summarizePreferenceMemory, P as analystRunDigest, R as ControlDecision, S as controlRunToFeedbackTrajectory, T as feedbackTrajectoriesToOptimizerRows, U as ControlSeverity, V as ControlRuntimeConfig, W as ControlStep, _ as PreferenceMemoryEntry, a as FeedbackLabel, b as analystRunToReviewRequests, c as FeedbackOptimizerRow, d as FeedbackTask, f as FeedbackTrajectory, g as InMemoryFeedbackTrajectoryStore, h as FileSystemFeedbackTrajectoryStore, i as FeedbackAttempt, j as AnalystReviewDecision, k as withAssignedFeedbackSplit, l as FeedbackOutcome, m as FeedbackTrajectoryStore, n as AnalystReviewRequest, o as FeedbackLabelKind, p as FeedbackTrajectoryFilter, q as runAgentControlLoop, r as FeedbackArtifactType, s as FeedbackLabelSource, t as AnalystFeedbackTrajectoryOptions, u as FeedbackSplitPolicy, v as ProposedSideEffect, w as feedbackTrajectoriesToDatasetScenarios, x as assignFeedbackSplit, y as analystRunToFeedbackTrajectory, z as ControlEvalResult } from "./feedback-trajectory-DpTTjo0q.js";
46
+ import { a as SteeringChange, c as CampaignRunContext, d as CampaignVariant, f as EvalCampaignOptions, h as runEvalCampaign, i as Researcher, l as CampaignRunOutcome, m as FailedRun, n as ExperimentResult, o as CampaignFactoryParams, p as EvalCampaignResult, r as FailureMode, s as CampaignIntegrityPolicy, t as ExperimentPlan, u as CampaignScenario } from "./researcher-DJnoUE8c.js";
47
47
  import { J as MintRolloutOptions, Y as MintRolloutResult, Z as mintRolloutRows, et as ScorePreference } from "./index-BNPtkBPf.js";
48
48
  import { _ as iqr, a as ToolStats, c as computeToolUseMetrics, d as judgeAgreementView, i as toolWasteView, m as budgetBreachView, o as ToolUseMetrics, s as ToolUseOptions } from "./tool-waste-DjRDEsuI.js";
49
49
  //#region src/statistics/descriptive.d.ts
@@ -2802,5 +2802,5 @@ declare class PairwiseSteeringOptimizer {
2802
2802
  optimize(rows: SteeringOptimizationRow[], config?: SteeringOptimizerConfig): SteeringOptimizationResult;
2803
2803
  }
2804
2804
  //#endregion
2805
- export { AGENT_PROFILE_KINDS, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, AgentEvalError, type AgentEvalErrorCode, type AgentProfile, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileDimensionValue, type AgentProfileJson, type AgentProfileSourceInput, type Analyst, type AnalystContext, type AnalystFeedbackTrajectoryOptions, type AnalystFinding, type AnalystFindingDigest, AnalystRegistry, type AnalystRegistryOptions, type AnalystReviewDecision, type AnalystReviewRequest, type AnalystRunDigest, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeRunsOptions, type AnalyzeTracesResult, type Artifact, type ArtifactEventLike, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertCrossFamilyServedOptions, type AssertServedModelOptions, type AssertSingleBackendOptions, BOOTSTRAP_GATE_MIN_N, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BenchmarkEvaluation, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CalibrationResult, type CampaignCellFailureReceipt, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignResult, type CampaignRunContext, type CampaignRunOutcome, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type CheckResult, type CheckerIdentity, type CheckerOutcome, type CliffsMagnitude, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptFinding, type ConceptSpec, ConfigError, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractSpan, type ContractVerdict, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRuntimeConfig, type ControlRuntimeError, type ControlStep, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostProvenance, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateChatClientOpts, type CreateTraceAnalystOptions, CrossFamilyError, type CustomTokenPricing, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, type DatasetManifest, type DatasetOverview, type DatasetScenario, type DatasetSplit, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DefineAgentEvalOptions, type DefinedAgentEval, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DetectorEvent, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, type DriverState, type DspyRlmTraceEngineOptions, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type EquivalenceArm, type EquivalenceCheckDefinition, type EquivalenceCheckSpec, type EquivalenceChecker, type EquivalenceCheckerInput, type EquivalenceCheckerResult, type EquivalenceObligation, type EquivalenceObligationStatus, EquivalenceProtocolError, type EquivalenceRecord, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EventFilter, type EvidenceRef, type ExactAnalystRunEvent, type ExactAnalystRunResult, type ExactCapableAnalyst, type ExactRegistryRunOpts, type ExactRiskDifferenceResult, type ExperimentPlan, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExtractOptions, type ExtractResult, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision, type GateEvidence, type GenericSpan, type GoldenItem, HARNESS_NATIVE_MODEL, type HarnessType, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, type InsightReport, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, JudgeError, type JudgeFamily, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, type JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeSensitivity, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, type LlmRouteRequirements, type LlmSpan, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, type MannWhitneyResult, type MatchedPair, type MatchedRunRecordPair, type MaximumCharge, type McNemarResult, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type ModelSeats, ModelSubstitutionError, MultiLayerVerifier, type NoLeakOptions, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OtlpExport, OtlpFileTraceStore, type OtlpFlatLine, type OtlpSpan, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedDecisionMethod, type PairedDecisionShape, type PairedDecisionStatistic, type PairedDeltaTestOptions, type PairedDeltaTestResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMcNemarEvidence, type PairedMetricDelta, type PairedPromotionDecision, type PairedPromotionDecisionOptions, type PairedSignTestResult, type PairedTTestResult, PairwiseSteeringOptimizer, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PreferenceMemoryEntry, type ProducedProposal, type ProducedState, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposalFinding, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposedSideEffect, type QueryTracesPage, REDACTION_VERSION, type RankTestMethod, type RankTestMethodRequest, type RankTestOptions, type RawAnalystFinding, type RawProviderEvent, type RawProviderSink, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamReport, type RedactionRule, type ReferenceReplayCaseRun, type ReferenceReplayRun, type ReferenceReplaySplit, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseTraceEvidence, type RepeatedActionOptions, type RequirementCheck, type ResearchReport, type ResearchReportOptions, type Researcher, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type RiskDifferenceResult, type RouteMap, type RoutedField, type Run, type RunCampaignOptions, type RunCommandInput, type RunCommandResult, type RunCostProvenance, type RunFilter, RunIntegrityError, type RunIntegrityReport, type RunJudgeMetadata, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RuntimeEventLike, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, type SandboxDriver, type SatisfiedBy, type Scenario, type ScenarioCost, type ScorePreference, type ScoreRiskDifferenceResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type SearchSpanResult, type SearchTraceResult, type SelfImproveOptions, type SelfImproveResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SeriesConvergenceOptions, type SeriesConvergenceResult, ServedCrossFamilyError, type ServedModelCheck, type ServedModelVerdict, type Severity, type SignTestAlternative, type SingleBackendDivergence, type SingleBackendReport, type Span, type SpanFilter, type SpanHandle, type SpanMatchRecord, type SplitCoverage, type SteeringBundle, type SteeringChange, type SteeringOptimizationResult, type SteeringOptimizationRow, type StopDecision, type StrategyChecker, type StreamingDetector, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SynthesisTarget, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, type TaskGold, type TaskHeadroom, type ToolCallEventLike, type ToolSpan, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAnalysisEngine, type TraceAnalysisEngineResult, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystDefinition, type TraceAnalystFilters, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceEmitter, type TraceEvent, type TraceInsightReadiness, type TraceInsightSuite, type TraceStore, type Trajectory, type TrajectoryStep, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, UNKNOWN_MODEL, type UserQuestion, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, type VerbosityBiasResult, type Verdict, type VerdictCacheStore, type VerdictCertification, type Verification, type VerificationReport, type VerificationStrategyProfile, type VerificationStrategySource, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type ViteDeployRunnerInput, WILCOXON_EXACT_MAX_N, type WeightedCompositeInput, type WeightedCompositeResult, type WilcoxonSignedRankResult, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, index_d_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
2805
+ export { AGENT_PROFILE_KINDS, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, AgentEvalError, type AgentEvalErrorCode, type AgentProfile, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileDimensionValue, type AgentProfileJson, type AgentProfileSourceInput, type Analyst, type AnalystContext, type AnalystFeedbackTrajectoryOptions, type AnalystFinding, type AnalystFindingDigest, AnalystRegistry, type AnalystRegistryOptions, type AnalystReviewDecision, type AnalystReviewRequest, type AnalystRunDigest, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeRunsOptions, type AnalyzeTracesResult, type Artifact, type ArtifactEventLike, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertCrossFamilyServedOptions, type AssertServedModelOptions, type AssertSingleBackendOptions, BOOTSTRAP_GATE_MIN_N, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BenchmarkEvaluation, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CalibrationResult, type CampaignCellFailureReceipt, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignResult, type CampaignRunContext, type CampaignRunOutcome, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type CheckResult, type CheckerIdentity, type CheckerOutcome, type CliffsMagnitude, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptFinding, type ConceptSpec, ConfigError, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractSpan, type ContractVerdict, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRuntimeConfig, type ControlRuntimeError, type ControlStep, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostProvenance, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateChatClientOpts, type CreateTraceAnalystOptions, CrossFamilyError, type CustomTokenPricing, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, type DatasetManifest, type DatasetOverview, type DatasetScenario, type DatasetSplit, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DefineAgentEvalOptions, type DefinedAgentEval, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DetectorEvent, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, type DriverState, type DspyRlmTraceEngineOptions, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type EquivalenceArm, type EquivalenceCheckDefinition, type EquivalenceCheckSpec, type EquivalenceChecker, type EquivalenceCheckerInput, type EquivalenceCheckerResult, type EquivalenceObligation, type EquivalenceObligationStatus, EquivalenceProtocolError, type EquivalenceRecord, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EventFilter, type EvidenceRef, type ExactAnalystRunEvent, type ExactAnalystRunResult, type ExactCapableAnalyst, type ExactRegistryRunOpts, type ExactRiskDifferenceResult, type ExperimentPlan, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExtractOptions, type ExtractResult, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision, type GateEvidence, type GenericSpan, type GoldenItem, HARNESS_NATIVE_MODEL, type HarnessType, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, type InsightReport, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, JudgeError, type JudgeFamily, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, type JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeSensitivity, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, type LlmRouteRequirements, type LlmSpan, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, type MannWhitneyResult, type MatchedPair, type MatchedRunRecordPair, type MaximumCharge, type McNemarResult, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type ModelSeats, ModelSubstitutionError, MultiLayerVerifier, type NoLeakOptions, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OtlpExport, OtlpFileTraceStore, type OtlpFlatLine, type OtlpSpan, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedDecisionMethod, type PairedDecisionShape, type PairedDecisionStatistic, type PairedDeltaTestOptions, type PairedDeltaTestResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMcNemarEvidence, type PairedMetricDelta, type PairedPromotionDecision, type PairedPromotionDecisionOptions, type PairedSignTestResult, type PairedTTestResult, PairwiseSteeringOptimizer, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PreferenceMemoryEntry, type ProducedProposal, type ProducedState, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposalFinding, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposedSideEffect, type QueryTracesPage, REDACTION_VERSION, type RankTestMethod, type RankTestMethodRequest, type RankTestOptions, type RawAnalystFinding, type RawProviderEvent, type RawProviderSink, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamReport, type RedactionRule, type ReferenceReplayCaseRun, type ReferenceReplayRun, type ReferenceReplaySplit, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseTraceEvidence, type RepeatedActionOptions, type RequirementCheck, type ResearchReport, type ResearchReportOptions, type Researcher, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type RiskDifferenceResult, type RouteMap, type RoutedField, type Run, type RunCampaignOptions, type RunCommandInput, type RunCommandResult, type RunCostProvenance, type RunFilter, RunIntegrityError, type RunIntegrityReport, type RunJudgeMetadata, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RuntimeEventLike, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, type SandboxDriver, type SatisfiedBy, type Scenario, type ScenarioCost, type ScorePreference, type ScoreRiskDifferenceResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type SearchSpanResult, type SearchTraceResult, type SelfImproveOptions, type SelfImproveResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SeriesConvergenceOptions, type SeriesConvergenceResult, ServedCrossFamilyError, type ServedModelCheck, type ServedModelPolicy, type ServedModelVerdict, type Severity, type SignTestAlternative, type SingleBackendDivergence, type SingleBackendReport, type Span, type SpanFilter, type SpanHandle, type SpanMatchRecord, type SplitCoverage, type SteeringBundle, type SteeringChange, type SteeringOptimizationResult, type SteeringOptimizationRow, type StopDecision, type StrategyChecker, type StreamingDetector, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SynthesisTarget, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, type TaskGold, type TaskHeadroom, type ToolCallEventLike, type ToolSpan, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAnalysisEngine, type TraceAnalysisEngineResult, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystDefinition, type TraceAnalystFilters, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceEmitter, type TraceEvent, type TraceInsightReadiness, type TraceInsightSuite, type TraceStore, type Trajectory, type TrajectoryStep, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, UNKNOWN_MODEL, type UserQuestion, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, type VerbosityBiasResult, type Verdict, type VerdictCacheStore, type VerdictCertification, type Verification, type VerificationReport, type VerificationStrategyProfile, type VerificationStrategySource, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type ViteDeployRunnerInput, WILCOXON_EXACT_MAX_N, type WeightedCompositeInput, type WeightedCompositeResult, type WilcoxonSignedRankResult, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, index_d_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
2806
2806
  //# sourceMappingURL=index.d.ts.map
package/dist/index.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
2
  import { i as JudgeError, o as NotFoundError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
3
3
  import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
4
- import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-C0oJ4vr-.js";
4
+ import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-jfk8Du3b.js";
5
5
  import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, verifyAgentProfileCell } from "./profile-cell.js";
6
6
  import { c as mulberry32 } from "./internal-BDHPCnjk.js";
7
7
  import { a as spearmanR, i as ranks, n as partialCredit, o as weightedComposite, r as pearsonR, s as weightedMean, t as confidenceInterval } from "./descriptive-B5MwKfbf.js";
@@ -15,12 +15,12 @@ import { t as eProcess } from "./sequential-eprocess-CbUt2htw.js";
15
15
  import { n as iqr, r as welchsTTest } from "./baseline-BhPRQBVn.js";
16
16
  import { i as isLlmSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
17
17
  import { a as judgeSpans, c as runsForScenario, n as argHash } from "./query-Di7eEQ79.js";
18
- import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-CvZQW4u9.js";
18
+ import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-rqNyVhVV.js";
19
19
  import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
20
20
  import { a as parseRunRecordSafe, c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, o as roundTripRunRecord, r as isRunRecord, s as runTaskScore, t as RunRecordValidationError } from "./run-record-D2lDdSAz.js";
21
21
  import { a as summaryTable, n as gainHistogram, r as paretoChart } from "./summary-report-Blysd6Z2.js";
22
22
  import { n as contentHash, r as fileVerdictCache, t as canonicalJson } from "./verdict-cache-mZf5FEiY.js";
23
- import { H as buildReflectionPrompt, J as DEFAULT_RED_TEAM_CORPUS, Q as runCanaries, U as parseReflectionResponse, X as redTeamReport, Y as redTeamDataset, Z as scoreRedTeamOutput, _ as paretoFrontier, at as assertRealBackend, et as runCampaign, g as dominates, it as BackendIntegrityError, k as surfaceContentHash, ot as summarizeBackendIntegrity, t as llmJudge } from "./llm-judge-DWq1Ptco.js";
23
+ import { H as buildReflectionPrompt, J as DEFAULT_RED_TEAM_CORPUS, Q as runCanaries, U as parseReflectionResponse, X as redTeamReport, Y as redTeamDataset, Z as scoreRedTeamOutput, _ as paretoFrontier, at as assertRealBackend, et as runCampaign, g as dominates, it as BackendIntegrityError, k as surfaceContentHash, ot as summarizeBackendIntegrity, t as llmJudge } from "./llm-judge-CVq33oz1.js";
24
24
  import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Qv-cpptD.js";
25
25
  import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-B1qx30B4.js";
26
26
  import { n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
@@ -32,7 +32,7 @@ import { t as TraceEmitter } from "./emitter-CPBAhxum.js";
32
32
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
33
33
  import { t as runCounterfactual } from "./counterfactual-lDfCx0Uz.js";
34
34
  import { l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, p as DEFAULT_TRACE_ANALYST_BUDGETS, t as createTraceAnalyst } from "./kind-factory-DmAa0h3K.js";
35
- import { S as validateAnalystReviewDecisions, _ as analystRunDigest, b as readAnalystReview, g as analystFindingDigest, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, r as AnalystRegistry, t as createChatClient, u as FAILURE_MODE_KIND_SPEC, v as assertUniqueFindingIds, x as snapshotAnalystRun, y as completedAnalystReviewQuality } from "./chat-client-BX8Wh3fn.js";
35
+ import { S as validateAnalystReviewDecisions, _ as analystRunDigest, b as readAnalystReview, g as analystFindingDigest, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, r as AnalystRegistry, t as createChatClient, u as FAILURE_MODE_KIND_SPEC, v as assertUniqueFindingIds, x as snapshotAnalystRun, y as completedAnalystReviewQuality } from "./chat-client-2bVfrzhN.js";
36
36
  import { OUTPUT_VALUE } from "./trace-attributes.js";
37
37
  import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
38
38
  import { n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
@@ -40,14 +40,14 @@ import { S as captureFetchToRawSink, d as inferDomainKeywords, h as analyzeTrace
40
40
  import { n as assertRunCaptured, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
41
41
  import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
42
42
  import { t as packageVersion$1 } from "./package-version-D7lQHt_-.js";
43
- import { C as assertCrossFamily, S as CrossFamilyError, _ as assertCrossFamilyServed, a as assertLlmRoute, b as checkServedModel, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as ServedCrossFamilyError, h as ModelSubstitutionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertServedModel, w as judgeFamily, x as servedModelAcceptable, y as assertServedModels } from "./llm-client-BMuxYoZy.js";
44
- import { t as runEvalCampaign } from "./eval-campaign-bmZ6NIIP.js";
43
+ import { C as CrossFamilyError, S as servedModelAcceptable, T as judgeFamily, _ as assertCrossFamilyServed, a as assertLlmRoute, b as assertServedModels, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as ServedCrossFamilyError, h as ModelSubstitutionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertServedModel, w as assertCrossFamily, x as checkServedModel } from "./llm-client-Bg32RW0j.js";
44
+ import { t as runEvalCampaign } from "./eval-campaign-C4jmuM-b.js";
45
45
  import { i as improvementVerdict, n as computeExperimentStats } from "./experiment-tracker-Ym6rEQT1.js";
46
46
  import "./rollout-ytVQ7WT8.js";
47
47
  import { t as mintRolloutRows } from "./mint-BV6tLVWl.js";
48
48
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
49
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BmtOl_kP.js";
50
- import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-BI7Rrl5-.js";
49
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-DHI0WrUU.js";
50
+ import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-laMCnTLn.js";
51
51
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
52
52
  import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CDYWW_8N.js";
53
53
  import { n as evaluateReleaseConfidence, r as bootstrapCi } from "./release-confidence-BknrpBnO.js";
@@ -1,5 +1,5 @@
1
1
  import { r as CaptureIntegrityError } from "./errors-DEE6u6ot.js";
2
- import { mt as RawProviderSink } from "./types-DOhGq4S1.js";
2
+ import { ht as RawProviderSink } from "./types-jUBXJ7Iz.js";
3
3
  import { s as TraceStore } from "./store-CT9YIIve.js";
4
4
  //#region src/trace/integrity.d.ts
5
5
  interface RunIntegrityExpectations {
@@ -58,4 +58,4 @@ declare function assertRunCaptured(store: TraceStore, runId: string, expectation
58
58
  declare function throwIfRunIncomplete(report: RunIntegrityReport): void;
59
59
  //#endregion
60
60
  export { RunIntegrityReport as a, RunIntegrityIssueCode as i, RunIntegrityExpectations as n, assertRunCaptured as o, RunIntegrityIssue as r, throwIfRunIncomplete as s, RunIntegrityError as t };
61
- //# sourceMappingURL=integrity-CGfpTE5-.d.ts.map
61
+ //# sourceMappingURL=integrity-B0dZ96EO.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"integrity-CGfpTE5-.d.ts","names":[],"sources":["../src/trace/integrity.ts"],"mappings":";;;;UAyBiB;;EAEf;;EAEA;;EAEA;;;;;EAKA,UAAU;;EAEV;;;;;;EAMA;;EAEA;;KAGU;UAUK;EACf,MAAM;EACN;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;;;;;;EAMA;IAAmB;IAAiB;;EACpC,QAAQ;;cAGG,0BAA0B;WACT,QAAQ;EAApC,YAA4B,QAAQ;;iBAOhB,kBACpB,OAAO,YACP,eACA,eAAc,2BACb,QAAQ;;iBA6GK,qBAAqB,QAAQ"}
1
+ {"version":3,"file":"integrity-B0dZ96EO.d.ts","names":[],"sources":["../src/trace/integrity.ts"],"mappings":";;;;UAyBiB;;EAEf;;EAEA;;EAEA;;;;;EAKA,UAAU;;EAEV;;;;;;EAMA;;EAEA;;KAGU;UAUK;EACf,MAAM;EACN;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;;;;;;EAMA;IAAmB;IAAiB;;EACpC,QAAQ;;cAGG,0BAA0B;WACT,QAAQ;EAApC,YAA4B,QAAQ;;iBAOhB,kBACpB,OAAO,YACP,eACA,eAAc,2BACb,QAAQ;;iBA6GK,qBAAqB,QAAQ"}
@@ -199,6 +199,9 @@ var ModelSubstitutionError = class extends AgentEvalError {
199
199
  this.name = "ModelSubstitutionError";
200
200
  }
201
201
  };
202
+ function assertServedModelPolicy(value, label) {
203
+ if (value !== void 0 && value !== "exact" && value !== "allow-within-family") throw new Error(`${label} must be 'exact' or 'allow-within-family'`);
204
+ }
202
205
  /**
203
206
  * The one place the accept/reject policy lives, so a caller that reports
204
207
  * substitution (a preflight table, a run record) and a caller that throws on it
@@ -317,7 +320,11 @@ function maximumChargeForLlmRequest(request, options = {}) {
317
320
  ...usage
318
321
  };
319
322
  }
320
- /** Convert a provider result into the canonical paid-call receipt input. */
323
+ /** Convert a provider result into the canonical paid-call receipt input.
324
+ * The receipt is JSON-clean: absent optional fields are omitted, never carried
325
+ * as explicit-undefined keys. Receipts cross JSON boundaries (the external
326
+ * optimizer's loopback proxy validates them with `assertJsonValue`), where an
327
+ * explicit-undefined value is rejected as non-serializable. */
321
328
  function costReceiptFromLlm(result, customTokenPricing) {
322
329
  const cachedTokens = result.usage.cachedPromptTokens ?? 0;
323
330
  const inputTokens = Math.max(0, result.usage.promptTokens - cachedTokens);
@@ -326,8 +333,8 @@ function costReceiptFromLlm(result, customTokenPricing) {
326
333
  model: result.model,
327
334
  inputTokens,
328
335
  outputTokens: result.usage.completionTokens,
329
- reasoningTokens: result.usage.reasoningTokens,
330
- cachedTokens: cachedTokens > 0 ? cachedTokens : void 0,
336
+ ...result.usage.reasoningTokens === void 0 ? {} : { reasoningTokens: result.usage.reasoningTokens },
337
+ ...cachedTokens > 0 ? { cachedTokens } : {},
331
338
  ...providerCostUsd === void 0 ? customTokenPricing && result.usage.captured !== false ? { customTokenPricing } : result.costUsd === null ? {} : { estimatedCostUsd: result.costUsd } : { actualCostUsd: providerCostUsd },
332
339
  usageUnknown: result.usage.captured === false
333
340
  };
@@ -376,7 +383,9 @@ const RETRYABLE_STATUS = /* @__PURE__ */ new Set([
376
383
  429,
377
384
  502,
378
385
  503,
379
- 504
386
+ 504,
387
+ 522,
388
+ 524
380
389
  ]);
381
390
  /**
382
391
  * Transient transport/network error signatures, matched against an error's
@@ -463,11 +472,13 @@ function isTemperatureOneRejection(status, body) {
463
472
  function buildBody(req, forceJsonObject, defaultThinking) {
464
473
  const body = {
465
474
  model: req.model,
466
- messages: req.messages,
475
+ messages: req.messages.map(encodeWireMessage),
467
476
  temperature: req.temperature ?? 0
468
477
  };
469
478
  if (req.maxTokens != null) if (usesMaxCompletionTokens(req.model)) body.max_completion_tokens = req.maxTokens;
470
479
  else body.max_tokens = req.maxTokens;
480
+ if (req.tools !== void 0) body.tools = req.tools;
481
+ if (req.toolChoice !== void 0) body.tool_choice = req.toolChoice;
471
482
  const thinking = req.thinking ?? defaultThinking;
472
483
  if (thinking !== void 0) body.thinking = { type: thinking };
473
484
  if (req.jsonSchema && !forceJsonObject) body.response_format = {
@@ -484,6 +495,47 @@ function buildBody(req, forceJsonObject, defaultThinking) {
484
495
  function usesMaxCompletionTokens(model) {
485
496
  return /^gpt-5(?:[.-]|$)/i.test(model);
486
497
  }
498
+ /** Encode one canonical message as its OpenAI chat-completions wire shape. */
499
+ function encodeWireMessage(message) {
500
+ if (message.role === "tool") return {
501
+ role: "tool",
502
+ tool_call_id: message.toolCallId,
503
+ content: message.content
504
+ };
505
+ if (message.toolCalls === void 0) return {
506
+ role: message.role,
507
+ content: message.content
508
+ };
509
+ return {
510
+ role: message.role,
511
+ content: message.content,
512
+ tool_calls: message.toolCalls.map((call) => ({
513
+ id: call.id,
514
+ type: "function",
515
+ function: {
516
+ name: call.name,
517
+ arguments: call.argumentsJson
518
+ }
519
+ }))
520
+ };
521
+ }
522
+ /** Parse provider tool_calls into the canonical shape. A malformed entry is a
523
+ * provider-contract violation and throws rather than degrading silently. */
524
+ function parseWireToolCalls(value, model) {
525
+ if (value === void 0 || value === null) return void 0;
526
+ if (!Array.isArray(value)) throw new Error(`LLM response tool_calls must be an array (model=${model})`);
527
+ if (value.length === 0) return void 0;
528
+ return value.map((entry, index) => {
529
+ const record = entry;
530
+ const fn = record && typeof record === "object" ? record.function : void 0;
531
+ if (!record || typeof record !== "object" || typeof record.id !== "string" || record.id.length === 0 || record.type !== void 0 && record.type !== "function" || !fn || typeof fn !== "object" || typeof fn.name !== "string" || fn.name.length === 0 || typeof fn.arguments !== "string") throw new Error(`LLM response tool_calls[${index}] is not a function call with string arguments (model=${model})`);
532
+ return {
533
+ id: record.id,
534
+ name: fn.name,
535
+ argumentsJson: fn.arguments
536
+ };
537
+ });
538
+ }
487
539
  async function sleep(ms) {
488
540
  return new Promise((resolve) => setTimeout(resolve, ms));
489
541
  }
@@ -709,6 +761,7 @@ async function callLlm(req, opts = {}) {
709
761
  redactedFields: []
710
762
  });
711
763
  const choice = json.choices?.[0];
764
+ const toolCalls = parseWireToolCalls(choice?.message?.tool_calls, req.model);
712
765
  const usageRaw = json.usage && typeof json.usage === "object" && !Array.isArray(json.usage) ? json.usage : void 0;
713
766
  const promptTokens = providerTokenCount(usageRaw?.prompt_tokens);
714
767
  const completionTokens = providerTokenCount(usageRaw?.completion_tokens);
@@ -729,6 +782,7 @@ async function callLlm(req, opts = {}) {
729
782
  if (opts.assertServedModel) assertServedModel(req.model, servedModel, opts.assertServedModel === true ? {} : opts.assertServedModel);
730
783
  return {
731
784
  content,
785
+ ...toolCalls === void 0 ? {} : { toolCalls },
732
786
  finishReason: choice?.finish_reason ?? null,
733
787
  contentEmpty: content.trim().length === 0,
734
788
  usage: {
@@ -965,6 +1019,6 @@ var LlmClient = class {
965
1019
  }
966
1020
  };
967
1021
  //#endregion
968
- export { assertCrossFamily as C, CrossFamilyError as S, assertCrossFamilyServed as _, assertLlmRoute as a, checkServedModel as b, callLlmJson as c, isTransientLlmError as d, maximumChargeForLlmRequest as f, ServedCrossFamilyError as g, ModelSubstitutionError as h, LlmRouteAssertionError as i, costReceiptFromLlm as l, stripFencedJson as m, LlmClient as n, backoffMs as o, probeLlm as p, LlmResponseError as r, callLlm as s, LlmCallError as t, costReceiptFromLlmError as u, assertServedModel as v, judgeFamily as w, servedModelAcceptable as x, assertServedModels as y };
1022
+ export { CrossFamilyError as C, servedModelAcceptable as S, judgeFamily as T, assertCrossFamilyServed as _, assertLlmRoute as a, assertServedModels as b, callLlmJson as c, isTransientLlmError as d, maximumChargeForLlmRequest as f, ServedCrossFamilyError as g, ModelSubstitutionError as h, LlmRouteAssertionError as i, costReceiptFromLlm as l, stripFencedJson as m, LlmClient as n, backoffMs as o, probeLlm as p, LlmResponseError as r, callLlm as s, LlmCallError as t, costReceiptFromLlmError as u, assertServedModel as v, assertCrossFamily as w, checkServedModel as x, assertServedModelPolicy as y };
969
1023
 
970
- //# sourceMappingURL=llm-client-BMuxYoZy.js.map
1024
+ //# sourceMappingURL=llm-client-Bg32RW0j.js.map