@tangle-network/agent-eval 0.146.0 → 0.147.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (145) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/dist/analyst/index.d.ts +13 -13
  3. package/dist/analyst/index.js +6 -6
  4. package/dist/{backend-integrity-HVmWyOaX.d.ts → backend-integrity-Bz8nSrRE.d.ts} +2 -2
  5. package/dist/{backend-integrity-HVmWyOaX.d.ts.map → backend-integrity-Bz8nSrRE.d.ts.map} +1 -1
  6. package/dist/{benchmark-DTQ1RApl.d.ts → benchmark-Cn7ZMFNW.d.ts} +3 -3
  7. package/dist/{benchmark-DTQ1RApl.d.ts.map → benchmark-Cn7ZMFNW.d.ts.map} +1 -1
  8. package/dist/{benchmark-command-CxLJljS1.js → benchmark-command-BTiysiYU.js} +11 -11
  9. package/dist/{benchmark-command-CxLJljS1.js.map → benchmark-command-BTiysiYU.js.map} +1 -1
  10. package/dist/benchmarks/index.d.ts +4 -4
  11. package/dist/benchmarks/index.js +3 -3
  12. package/dist/campaign/index.d.ts +8 -8
  13. package/dist/campaign/index.js +5 -5
  14. package/dist/{campaign-CbQXcARJ.js → campaign-DCDdhuv2.js} +7 -7
  15. package/dist/{campaign-CbQXcARJ.js.map → campaign-DCDdhuv2.js.map} +1 -1
  16. package/dist/{capture-fetch-BZiO2cEH.d.ts → capture-fetch-BAx3Gntt.d.ts} +2 -2
  17. package/dist/{capture-fetch-BZiO2cEH.d.ts.map → capture-fetch-BAx3Gntt.d.ts.map} +1 -1
  18. package/dist/{chat-client-BJmwfnjN.js → chat-client-BX8Wh3fn.js} +3 -3
  19. package/dist/{chat-client-BJmwfnjN.js.map → chat-client-BX8Wh3fn.js.map} +1 -1
  20. package/dist/cli.js +2 -2
  21. package/dist/{client-YDkU5QST.d.ts → client-BV7icIfy.d.ts} +2 -2
  22. package/dist/{client-YDkU5QST.d.ts.map → client-BV7icIfy.d.ts.map} +1 -1
  23. package/dist/contract/index.d.ts +10 -10
  24. package/dist/contract/index.js +9 -9
  25. package/dist/{cost-ledger-BSe92yAV.js → cost-ledger-B1qx30B4.js} +2 -2
  26. package/dist/{cost-ledger-BSe92yAV.js.map → cost-ledger-B1qx30B4.js.map} +1 -1
  27. package/dist/{default-registry-DONui3mA.d.ts → default-registry-BUtRMRKk.d.ts} +6 -6
  28. package/dist/{default-registry-DONui3mA.d.ts.map → default-registry-BUtRMRKk.d.ts.map} +1 -1
  29. package/dist/{define-agent-eval-BGxmy4W6.js → define-agent-eval-BqWFz3sK.js} +3 -3
  30. package/dist/{define-agent-eval-BGxmy4W6.js.map → define-agent-eval-BqWFz3sK.js.map} +1 -1
  31. package/dist/{define-agent-eval-DrpgTVPp.d.ts → define-agent-eval-D52ClbX2.d.ts} +5 -5
  32. package/dist/{define-agent-eval-DrpgTVPp.d.ts.map → define-agent-eval-D52ClbX2.d.ts.map} +1 -1
  33. package/dist/{dspy-rlm-engine-B2nst4NR.js → dspy-rlm-engine-BmtOl_kP.js} +4 -4
  34. package/dist/{dspy-rlm-engine-B2nst4NR.js.map → dspy-rlm-engine-BmtOl_kP.js.map} +1 -1
  35. package/dist/{engine-Bc3seRPe.d.ts → engine-CIy18RTX.d.ts} +4 -4
  36. package/dist/{engine-Bc3seRPe.d.ts.map → engine-CIy18RTX.d.ts.map} +1 -1
  37. package/dist/{eval-campaign-DeGLwACc.js → eval-campaign-bmZ6NIIP.js} +2 -2
  38. package/dist/{eval-campaign-DeGLwACc.js.map → eval-campaign-bmZ6NIIP.js.map} +1 -1
  39. package/dist/{exact-types-CYFbvA5U.d.ts → exact-types-rdKFzEnK.d.ts} +2 -2
  40. package/dist/{exact-types-CYFbvA5U.d.ts.map → exact-types-rdKFzEnK.d.ts.map} +1 -1
  41. package/dist/experiment/index.d.ts +5 -5
  42. package/dist/experiment/index.js +3 -3
  43. package/dist/{experiment-tracker-DWHZBAYL.d.ts → experiment-tracker-D9VHwRm-.d.ts} +6 -24
  44. package/dist/experiment-tracker-D9VHwRm-.d.ts.map +1 -0
  45. package/dist/{experiment-tracker-C29gXM4B.js → experiment-tracker-Ym6rEQT1.js} +15 -4
  46. package/dist/experiment-tracker-Ym6rEQT1.js.map +1 -0
  47. package/dist/{external-optimizer-contracts-CKEY98bK.d.ts → external-optimizer-contracts-CSjDLLmr.d.ts} +2 -2
  48. package/dist/{external-optimizer-contracts-CKEY98bK.d.ts.map → external-optimizer-contracts-CSjDLLmr.d.ts.map} +1 -1
  49. package/dist/{external-optimizer-process-DYNizz8W.js → external-optimizer-process-C4AyCcQd.js} +2 -2
  50. package/dist/{external-optimizer-process-DYNizz8W.js.map → external-optimizer-process-C4AyCcQd.js.map} +1 -1
  51. package/dist/{external-optimizer-subprocess-r6dP9DAj.js → external-optimizer-subprocess-DU1KM8yV.js} +3 -3
  52. package/dist/{external-optimizer-subprocess-r6dP9DAj.js.map → external-optimizer-subprocess-DU1KM8yV.js.map} +1 -1
  53. package/dist/{feedback-trajectory-Cg9EDxKB.d.ts → feedback-trajectory-niGeR6uJ.d.ts} +3 -3
  54. package/dist/{feedback-trajectory-Cg9EDxKB.d.ts.map → feedback-trajectory-niGeR6uJ.d.ts.map} +1 -1
  55. package/dist/fuzz.js +1 -1
  56. package/dist/hosted/index.d.ts +2 -2
  57. package/dist/{index-DGrcBMbC.d.ts → index-BjBjxiVv.d.ts} +9 -9
  58. package/dist/{index-DGrcBMbC.d.ts.map → index-BjBjxiVv.d.ts.map} +1 -1
  59. package/dist/index-CKblBxtr.d.ts +1 -0
  60. package/dist/{index-CevPFhDT.d.ts → index-gtjtfLEJ.d.ts} +7 -7
  61. package/dist/{index-CevPFhDT.d.ts.map → index-gtjtfLEJ.d.ts.map} +1 -1
  62. package/dist/index.d.ts +21 -51
  63. package/dist/index.d.ts.map +1 -1
  64. package/dist/index.js +16 -125
  65. package/dist/index.js.map +1 -1
  66. package/dist/{integrity-DDoeWibF.d.ts → integrity-CGfpTE5-.d.ts} +2 -2
  67. package/dist/{integrity-DDoeWibF.d.ts.map → integrity-CGfpTE5-.d.ts.map} +1 -1
  68. package/dist/{kind-factory-CPmSd58s.js → kind-factory-DmAa0h3K.js} +7 -7
  69. package/dist/kind-factory-DmAa0h3K.js.map +1 -0
  70. package/dist/{llm-client-_UE4fU7Z.js → llm-client-BMuxYoZy.js} +2 -2
  71. package/dist/{llm-client-_UE4fU7Z.js.map → llm-client-BMuxYoZy.js.map} +1 -1
  72. package/dist/{llm-judge-B7ffRF8w.js → llm-judge-DrzsVS5k.js} +4 -4
  73. package/dist/{llm-judge-B7ffRF8w.js.map → llm-judge-DrzsVS5k.js.map} +1 -1
  74. package/dist/{matrix-su7mIfbB.d.ts → matrix-C-2Qx1Zr.d.ts} +2 -2
  75. package/dist/{matrix-su7mIfbB.d.ts.map → matrix-C-2Qx1Zr.d.ts.map} +1 -1
  76. package/dist/meta-eval/index.d.ts +1 -1
  77. package/dist/{metrics-Cl0L1KUy.js → metrics-Qv-cpptD.js} +15 -8
  78. package/dist/metrics-Qv-cpptD.js.map +1 -0
  79. package/dist/multishot/golden/index.d.ts +1 -1
  80. package/dist/multishot/index.d.ts +2 -2
  81. package/dist/multishot/index.js +1 -1
  82. package/dist/openapi.json +1 -1
  83. package/dist/{produced-state-D3k45S9a.js → produced-state-Dtx60bUQ.js} +4 -4
  84. package/dist/{produced-state-D3k45S9a.js.map → produced-state-Dtx60bUQ.js.map} +1 -1
  85. package/dist/{promotion-policy-BKi5OK5l.d.ts → promotion-policy-CewdjfCJ.d.ts} +2 -2
  86. package/dist/{promotion-policy-BKi5OK5l.d.ts.map → promotion-policy-CewdjfCJ.d.ts.map} +1 -1
  87. package/dist/{provenance-Bb7Zmyj2.d.ts → provenance-LrOEOHQb.d.ts} +6 -6
  88. package/dist/{provenance-Bb7Zmyj2.d.ts.map → provenance-LrOEOHQb.d.ts.map} +1 -1
  89. package/dist/{registry-CZgEmVXb.d.ts → registry-CaqJIDdN.d.ts} +4 -4
  90. package/dist/{registry-CZgEmVXb.d.ts.map → registry-CaqJIDdN.d.ts.map} +1 -1
  91. package/dist/reporting.d.ts +1 -1
  92. package/dist/reporting.js +1 -1
  93. package/dist/{researcher-CC305Ed-.d.ts → researcher-BySlpla_.d.ts} +3 -3
  94. package/dist/{researcher-CC305Ed-.d.ts.map → researcher-BySlpla_.d.ts.map} +1 -1
  95. package/dist/rl.d.ts +3 -3
  96. package/dist/rl.js +2 -2
  97. package/dist/{semantic-concept-judge-BsY2Q0Oq.js → semantic-concept-judge-BI7Rrl5-.js} +3 -3
  98. package/dist/{semantic-concept-judge-BsY2Q0Oq.js.map → semantic-concept-judge-BI7Rrl5-.js.map} +1 -1
  99. package/dist/{sequential-CYwq6Ff_.d.ts → sequential-BhsrMupG.d.ts} +31 -2
  100. package/dist/sequential-BhsrMupG.d.ts.map +1 -0
  101. package/dist/{sequential-Br0mAPHA.js → sequential-CzK5DarL.js} +48 -6
  102. package/dist/sequential-CzK5DarL.js.map +1 -0
  103. package/dist/series-convergence-CjO2QdRW.js.map +1 -1
  104. package/dist/{series-convergence-C9G-GNYK.d.ts → series-convergence-D2fsoJ2w.d.ts} +4 -7
  105. package/dist/{series-convergence-C9G-GNYK.d.ts.map → series-convergence-D2fsoJ2w.d.ts.map} +1 -1
  106. package/dist/{server-cQXve8i4.js → server-D9wQclzG.js} +3 -3
  107. package/dist/{server-cQXve8i4.js.map → server-D9wQclzG.js.map} +1 -1
  108. package/dist/{skillopt-optimization-method-C9MrMgW9.d.ts → skillopt-optimization-method-BoC1Qccx.d.ts} +5 -5
  109. package/dist/{skillopt-optimization-method-C9MrMgW9.d.ts.map → skillopt-optimization-method-BoC1Qccx.d.ts.map} +1 -1
  110. package/dist/{skillopt-optimization-method-Cv3K4DaS.js → skillopt-optimization-method-C_UrqZs2.js} +4 -4
  111. package/dist/{skillopt-optimization-method-Cv3K4DaS.js.map → skillopt-optimization-method-C_UrqZs2.js.map} +1 -1
  112. package/dist/{statistical-heldout-fvxtHnKQ.d.ts → statistical-heldout-uxSpFEjm.d.ts} +2 -2
  113. package/dist/{statistical-heldout-fvxtHnKQ.d.ts.map → statistical-heldout-uxSpFEjm.d.ts.map} +1 -1
  114. package/dist/{store-otlp-CsptLYpN.js → store-otlp-CDYWW_8N.js} +2 -2
  115. package/dist/{store-otlp-CsptLYpN.js.map → store-otlp-CDYWW_8N.js.map} +1 -1
  116. package/dist/{store-tool-spans-Cq9mFd-q.js → store-tool-spans-CykkbOlv.js} +3 -3
  117. package/dist/{store-tool-spans-Cq9mFd-q.js.map → store-tool-spans-CykkbOlv.js.map} +1 -1
  118. package/dist/{store-tool-spans-CV0hRsUp.d.ts → store-tool-spans-D5FhM0_A.d.ts} +4 -4
  119. package/dist/{store-tool-spans-CV0hRsUp.d.ts.map → store-tool-spans-D5FhM0_A.d.ts.map} +1 -1
  120. package/dist/{task-failure-attributes-CpQ4y5RD.js → task-failure-attributes--ZTP3tYO.js} +2 -2
  121. package/dist/{task-failure-attributes-CpQ4y5RD.js.map → task-failure-attributes--ZTP3tYO.js.map} +1 -1
  122. package/dist/{tool-groups-CDDXNKhd.d.ts → tool-groups-zpufabP8.d.ts} +3 -3
  123. package/dist/tool-groups-zpufabP8.d.ts.map +1 -0
  124. package/dist/trace-repair/index.d.ts +1 -1
  125. package/dist/traces.d.ts +7 -7
  126. package/dist/traces.js +4 -4
  127. package/dist/{types-oAsa-fs4.d.ts → types-BohHewKK.d.ts} +2 -2
  128. package/dist/{types-oAsa-fs4.d.ts.map → types-BohHewKK.d.ts.map} +1 -1
  129. package/dist/{types-DABZgDGV.d.ts → types-C1Bmeb8X.d.ts} +3 -3
  130. package/dist/{types-DABZgDGV.d.ts.map → types-C1Bmeb8X.d.ts.map} +1 -1
  131. package/dist/{types-CWcPguwd.d.ts → types-ColZlZtc.d.ts} +2 -2
  132. package/dist/{types-CWcPguwd.d.ts.map → types-ColZlZtc.d.ts.map} +1 -1
  133. package/dist/{types-XBRdmbj8.d.ts → types-DOhGq4S1.d.ts} +1 -7
  134. package/dist/{types-XBRdmbj8.d.ts.map → types-DOhGq4S1.d.ts.map} +1 -1
  135. package/dist/wire/index.d.ts +2 -2
  136. package/dist/wire/index.js +1 -1
  137. package/package.json +2 -2
  138. package/dist/experiment-tracker-C29gXM4B.js.map +0 -1
  139. package/dist/experiment-tracker-DWHZBAYL.d.ts.map +0 -1
  140. package/dist/index-BgHYx1QG.d.ts +0 -1
  141. package/dist/kind-factory-CPmSd58s.js.map +0 -1
  142. package/dist/metrics-Cl0L1KUy.js.map +0 -1
  143. package/dist/sequential-Br0mAPHA.js.map +0 -1
  144. package/dist/sequential-CYwq6Ff_.d.ts.map +0 -1
  145. package/dist/tool-groups-CDDXNKhd.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
2
  import { i as JudgeError, o as NotFoundError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
3
3
  import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
4
- import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-D3k45S9a.js";
4
+ import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-Dtx60bUQ.js";
5
5
  import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, verifyAgentProfileCell } from "./profile-cell.js";
6
6
  import { c as mulberry32 } from "./internal-BDHPCnjk.js";
7
7
  import { a as spearmanR, i as ranks, n as partialCredit, o as weightedComposite, r as pearsonR, s as weightedMean, t as confidenceInterval } from "./descriptive-B5MwKfbf.js";
@@ -15,14 +15,14 @@ import { t as eProcess } from "./sequential-eprocess-CbUt2htw.js";
15
15
  import { n as iqr, r as welchsTTest } from "./baseline-BhPRQBVn.js";
16
16
  import { i as isLlmSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
17
17
  import { a as judgeSpans, c as runsForScenario, n as argHash } from "./query-Di7eEQ79.js";
18
- import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-BGxmy4W6.js";
18
+ import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-BqWFz3sK.js";
19
19
  import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
20
20
  import { a as parseRunRecordSafe, c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, o as roundTripRunRecord, r as isRunRecord, s as runTaskScore, t as RunRecordValidationError } from "./run-record-D2lDdSAz.js";
21
21
  import { a as summaryTable, n as gainHistogram, r as paretoChart } from "./summary-report-Blysd6Z2.js";
22
22
  import { n as contentHash, r as fileVerdictCache, t as canonicalJson } from "./verdict-cache-mZf5FEiY.js";
23
- import { B as DEFAULT_RED_TEAM_CORPUS, F as parseReflectionResponse, H as redTeamReport, K as runCampaign, P as buildReflectionPrompt, Q as summarizeBackendIntegrity, U as scoreRedTeamOutput, V as redTeamDataset, W as runCanaries, X as BackendIntegrityError, Z as assertRealBackend, _ as paretoFrontier, g as dominates, k as surfaceContentHash, t as llmJudge } from "./llm-judge-B7ffRF8w.js";
24
- import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Cl0L1KUy.js";
25
- import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-BSe92yAV.js";
23
+ import { B as DEFAULT_RED_TEAM_CORPUS, F as parseReflectionResponse, H as redTeamReport, K as runCampaign, P as buildReflectionPrompt, Q as summarizeBackendIntegrity, U as scoreRedTeamOutput, V as redTeamDataset, W as runCanaries, X as BackendIntegrityError, Z as assertRealBackend, _ as paretoFrontier, g as dominates, k as surfaceContentHash, t as llmJudge } from "./llm-judge-DrzsVS5k.js";
24
+ import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Qv-cpptD.js";
25
+ import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-B1qx30B4.js";
26
26
  import { n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
27
27
  import { c as pairedDecisionShape, l as minimumPairsForPairedDeltaTest, s as decidePairedPromotion, u as pairedDeltaTest } from "./power-preflight-DEw-uC7q.js";
28
28
  import { n as clamp01, t as aggregateRunScore } from "./run-score-lDzV0X8j.js";
@@ -31,25 +31,25 @@ import { n as equivalenceVerdict, t as certificationEvidenceDigest } from "./ver
31
31
  import { t as TraceEmitter } from "./emitter-CPBAhxum.js";
32
32
  import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
33
33
  import { t as runCounterfactual } from "./counterfactual-lDfCx0Uz.js";
34
- import { l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, p as DEFAULT_TRACE_ANALYST_BUDGETS, t as createTraceAnalyst } from "./kind-factory-CPmSd58s.js";
35
- import { S as validateAnalystReviewDecisions, _ as analystRunDigest, b as readAnalystReview, g as analystFindingDigest, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, r as AnalystRegistry, t as createChatClient, u as FAILURE_MODE_KIND_SPEC, v as assertUniqueFindingIds, x as snapshotAnalystRun, y as completedAnalystReviewQuality } from "./chat-client-BJmwfnjN.js";
34
+ import { l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, p as DEFAULT_TRACE_ANALYST_BUDGETS, t as createTraceAnalyst } from "./kind-factory-DmAa0h3K.js";
35
+ import { S as validateAnalystReviewDecisions, _ as analystRunDigest, b as readAnalystReview, g as analystFindingDigest, n as buildDefaultAnalystRegistry, o as DEFAULT_TRACE_ANALYST_KINDS, r as AnalystRegistry, t as createChatClient, u as FAILURE_MODE_KIND_SPEC, v as assertUniqueFindingIds, x as snapshotAnalystRun, y as completedAnalystReviewQuality } from "./chat-client-BX8Wh3fn.js";
36
36
  import { OUTPUT_VALUE } from "./trace-attributes.js";
37
37
  import { r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
38
38
  import { n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
39
- import { S as captureFetchToRawSink, d as inferDomainKeywords, h as analyzeTraces, l as describeTraceInsightScope, m as tokenizeDomainWords, n as toolSpansToTraceAnalysisStore, o as buildTraceInsightContext, p as scoreTraceInsightReadiness, s as buildTraceInsightPrompt, u as domainEvidencePattern, y as exportRunAsOtlp } from "./store-tool-spans-Cq9mFd-q.js";
39
+ import { S as captureFetchToRawSink, d as inferDomainKeywords, h as analyzeTraces, l as describeTraceInsightScope, m as tokenizeDomainWords, n as toolSpansToTraceAnalysisStore, o as buildTraceInsightContext, p as scoreTraceInsightReadiness, s as buildTraceInsightPrompt, u as domainEvidencePattern, y as exportRunAsOtlp } from "./store-tool-spans-CykkbOlv.js";
40
40
  import { n as assertRunCaptured, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
41
41
  import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
42
42
  import { t as packageVersion$1 } from "./package-version-D7lQHt_-.js";
43
- import { C as assertCrossFamily, S as CrossFamilyError, _ as assertCrossFamilyServed, a as assertLlmRoute, b as checkServedModel, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as ServedCrossFamilyError, h as ModelSubstitutionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertServedModel, w as judgeFamily, x as servedModelAcceptable, y as assertServedModels } from "./llm-client-_UE4fU7Z.js";
44
- import { t as runEvalCampaign } from "./eval-campaign-DeGLwACc.js";
45
- import { i as improvementVerdict, n as computeExperimentStats } from "./experiment-tracker-C29gXM4B.js";
43
+ import { C as assertCrossFamily, S as CrossFamilyError, _ as assertCrossFamilyServed, a as assertLlmRoute, b as checkServedModel, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as ServedCrossFamilyError, h as ModelSubstitutionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertServedModel, w as judgeFamily, x as servedModelAcceptable, y as assertServedModels } from "./llm-client-BMuxYoZy.js";
44
+ import { t as runEvalCampaign } from "./eval-campaign-bmZ6NIIP.js";
45
+ import { i as improvementVerdict, n as computeExperimentStats } from "./experiment-tracker-Ym6rEQT1.js";
46
46
  import "./rollout-ytVQ7WT8.js";
47
47
  import { t as mintRolloutRows } from "./mint-BV6tLVWl.js";
48
- import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
49
- import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-B2nst4NR.js";
50
- import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-BsY2Q0Oq.js";
48
+ import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
49
+ import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BmtOl_kP.js";
50
+ import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-BI7Rrl5-.js";
51
51
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
52
- import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CsptLYpN.js";
52
+ import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CDYWW_8N.js";
53
53
  import { n as evaluateReleaseConfidence, r as bootstrapCi } from "./release-confidence-BknrpBnO.js";
54
54
  import { createHash } from "node:crypto";
55
55
  import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
@@ -2816,115 +2816,6 @@ async function discoverPersonas(dir, opts = {}) {
2816
2816
  return results;
2817
2817
  }
2818
2818
  //#endregion
2819
- //#region src/driver.ts
2820
- /**
2821
- * Per-rigor stance the driver LLM adopts. Scales how hard the simulated
2822
- * user interrogates the agent — see `PersonaConfig.rigor`.
2823
- */
2824
- const RIGOR_STANCE = {
2825
- cooperative: "Your stance: a pragmatic early adopter. You accept reasonable answers and only push back on clear gaps or outright errors.",
2826
- demanding: "Your stance: an experienced professional with no time to waste. You do not accept vague, hedged, or generic answers — you expect specifics, and you say so plainly when you do not get them.",
2827
- relentless: "Your stance: a senior partner reviewing this work for a client who will litigate if it is wrong. You interrogate every claim. You accept nothing undefended. You find the single weakest point in every answer and attack it. Courteous, never satisfied."
2828
- };
2829
- /** Describe which nominal completion criteria are met, for the driver prompt. */
2830
- function describeCompletion(persona, state) {
2831
- const results = persona.completionCriteria.map((c) => {
2832
- const met = c.check(state);
2833
- return `${c.name}: ${met ? "MET" : "NOT MET"}`;
2834
- });
2835
- return `${results.filter((r) => r.includes("MET") && !r.includes("NOT")).length}/${persona.completionCriteria.length} — ${results.join(", ")}`;
2836
- }
2837
- /**
2838
- * Build the driver LLM's system prompt. The simulated user is an
2839
- * adversarial senior professional: it judges the agent's last response by a
2840
- * professional standard, refuses vague answers, challenges undefended
2841
- * claims, probes the persona's pressure points without revealing them, and
2842
- * signs off (DONE) only when a real practitioner would act on the work
2843
- * unmodified. Pure function of persona, product state, and product context
2844
- * — exported so harness authors can inspect and regression-test it.
2845
- *
2846
- * @deprecated A role expressed as a code function can never be optimized.
2847
- * Treat this output as SEED data for a registry-backed directive prompt.
2848
- * Removal tracked by tangle-network/agent-eval#618.
2849
- */
2850
- function buildDriverSystemPrompt(persona, state, productContext = "") {
2851
- warnDeprecatedOnce("buildDriverSystemPrompt", "buildDriverSystemPrompt is deprecated: roles belong in registry-backed prompt data, not code (tangle-network/agent-eval#618).");
2852
- const rigor = persona.rigor ?? "demanding";
2853
- const expertise = persona.expertise ? ` You are ${persona.expertise}.` : "";
2854
- const pressure = persona.pressurePoints && persona.pressurePoints.length > 0 ? `\nA competent ${persona.role} here MUST get the agent to address each of:\n${persona.pressurePoints.map((p) => ` - ${p}`).join("\n")}\nDo NOT hand these to the agent. Probe whether it surfaces them itself. If it misses one, press on exactly that gap until it delivers or demonstrably fails.\n` : "";
2855
- const curveballs = persona.curveballs && persona.curveballs.length > 0 ? `\nOnce the agent is coasting on easy answers, introduce ONE of these as a genuine new development — never as a quiz:\n${persona.curveballs.map((c) => ` - ${c}`).join("\n")}\n` : "";
2856
- return `You are role-playing a real ${persona.role} putting an AI agent through its paces.${expertise}
2857
- Your objective: ${persona.goal}
2858
- You are deciding whether this agent's work is good enough to stake your professional reputation on. Assume it is not — until it proves otherwise.
2859
-
2860
- ${RIGOR_STANCE[rigor]}
2861
- ${productContext ? `Product context:\n${productContext}\n` : ""}Current workspace state:
2862
- - Tasks: ${state.tasks} | Events: ${state.events}
2863
- - Proposals: pending=${state.proposals.pending}, approved=${state.proposals.approved}, rejected=${state.proposals.rejected}
2864
- - Vault files (${state.vaultFiles.length}): ${state.vaultFiles.slice(0, 10).join(", ")}${state.vaultFiles.length > 10 ? " …" : ""}
2865
- - Nominal task criteria: ${describeCompletion(persona, state)}
2866
- ${pressure}${curveballs}
2867
- How to choose your next message:
2868
- 1. Silently judge the agent's last response the way a ${persona.role} would. Is every claim defended with a specific authority, figure, or mechanism? Or is it vague, hedged, or generic?
2869
- 2. If it is vague or hand-waved — do NOT move on. Name the gap and demand the specific authority / figure / mechanism. "It depends" is not an answer; force the decision.
2870
- 3. If it makes a claim you can challenge — challenge it. Make the agent defend or correct it.
2871
- 4. If it missed something a ${persona.role} would catch — press on exactly that, without naming it for the agent.
2872
- 5. If it is genuinely solid — escalate: go a layer deeper, or introduce a curveball.
2873
- 6. First message — state your situation as you really would: realistic, specific, with the messy detail, but do not coach the agent.
2874
-
2875
- Sign-off: respond with exactly "DONE" only when a ${persona.role} would act on this work without redoing it. Nominal task completion is NOT sign-off — sloppy-but-complete still fails. If the agent never gets there, keep pushing; never sign off on weak work.
2876
-
2877
- Output ONLY your next message to the agent — in character, first person, no meta-commentary, no stage directions.`;
2878
- }
2879
- /**
2880
- * Decide the simulated user's next turn — the reactive, adversarial
2881
- * turn-generator an in-process eval harness uses to drive a multi-shot
2882
- * conversation. Returns the next user message, or the literal "DONE" when the
2883
- * simulated professional would sign off.
2884
- *
2885
- * @deprecated The persona-driver loop becomes a 2-node agent graph (driver
2886
- * profile + delegates edge). Removal tracked by
2887
- * tangle-network/agent-eval#618.
2888
- */
2889
- async function decideNextUserTurn(chat, opts) {
2890
- warnDeprecatedOnce("decideNextUserTurn", "decideNextUserTurn is deprecated: the persona-driver loop becomes a 2-node agent graph (tangle-network/agent-eval#618).");
2891
- const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
2892
- const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet — this is the first message)";
2893
- const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
2894
- const request = {
2895
- model,
2896
- messages: [{
2897
- role: "system",
2898
- content: buildDriverSystemPrompt(persona, state, productContext)
2899
- }, {
2900
- role: "user",
2901
- content: recentHistory ? `Recent conversation:\n${recentHistory}\n\nThe agent's latest response:\n${lastResponse}` : "No conversation yet. Send your opening message — in character, phrased as this person actually would."
2902
- }],
2903
- temperature: .5,
2904
- maxTokens: 700
2905
- };
2906
- const paid = await (opts.costLedger ?? new CostLedger()).runPaidCall({
2907
- channel: "driver",
2908
- phase: "driver-turn",
2909
- actor: "decideNextUserTurn",
2910
- model,
2911
- tags: opts.costTags,
2912
- maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
2913
- execute: (signal, callId) => chat.chat(request, {
2914
- signal,
2915
- idempotencyKey: callId
2916
- }),
2917
- receipt: costReceiptFromLlm,
2918
- receiptFromError: costReceiptFromLlmError
2919
- });
2920
- if (!paid.succeeded) throw paid.error;
2921
- assertServedModel(model, paid.value.servedModel, {
2922
- allowUnreported: true,
2923
- context: "decideNextUserTurn"
2924
- });
2925
- return paid.value.content.trim();
2926
- }
2927
- //#endregion
2928
2819
  //#region src/propose-review.ts
2929
2820
  /**
2930
2821
  * Propose / Verify / Review — the core multi-shot primitive.
@@ -7401,6 +7292,6 @@ function rankRows(rows, weights) {
7401
7292
  })).sort((a, b) => b.mean - a.mean);
7402
7293
  }
7403
7294
  //#endregion
7404
- export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmClient, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
7295
+ export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmClient, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, ModelSubstitutionError, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, ServedCrossFamilyError, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkServedModel, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, servedModelAcceptable, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
7405
7296
 
7406
7297
  //# sourceMappingURL=index.js.map