@tangle-network/agent-eval 0.102.1 → 0.103.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. package/dist/adapters/http.d.ts +2 -2
  2. package/dist/adapters/langchain.d.ts +2 -2
  3. package/dist/adapters/otel.d.ts +5 -5
  4. package/dist/analyst/index.d.ts +9 -9
  5. package/dist/analyst/index.js +3 -3
  6. package/dist/{analyze-runs-Cd-A_K4l.d.ts → analyze-runs-rF2eGxZ_.d.ts} +3 -3
  7. package/dist/belief-state/index.d.ts +3 -3
  8. package/dist/belief-state/index.js +1 -1
  9. package/dist/benchmarks/index.d.ts +2 -2
  10. package/dist/builder-eval/index.js +2 -2
  11. package/dist/campaign/index.d.ts +43 -16
  12. package/dist/campaign/index.js +9 -9
  13. package/dist/campaign/index.js.map +1 -1
  14. package/dist/{chunk-Z3FLN24V.js → chunk-2B4HPFXN.js} +104 -43
  15. package/dist/chunk-2B4HPFXN.js.map +1 -0
  16. package/dist/{chunk-7RBJANJD.js → chunk-3722PKPB.js} +2 -2
  17. package/dist/{chunk-YLKDN7JV.js → chunk-66T25VWE.js} +5 -5
  18. package/dist/chunk-66T25VWE.js.map +1 -0
  19. package/dist/{chunk-6FIAJHCU.js → chunk-6YVAN6R3.js} +4 -4
  20. package/dist/{chunk-ABOIVNXL.js → chunk-AGYDMORK.js} +1 -1
  21. package/dist/chunk-AGYDMORK.js.map +1 -0
  22. package/dist/{chunk-B2TMQM62.js → chunk-BHCFJGL4.js} +2 -2
  23. package/dist/{chunk-IH7LYRHL.js → chunk-BXZVBX3D.js} +3 -3
  24. package/dist/{chunk-6GT4NI4V.js → chunk-CLS3374R.js} +2 -2
  25. package/dist/{chunk-HRGUJTER.js → chunk-HYGRFL7C.js} +1 -1
  26. package/dist/{chunk-HRGUJTER.js.map → chunk-HYGRFL7C.js.map} +1 -1
  27. package/dist/{chunk-2NSLDY4B.js → chunk-J5MUFXEY.js} +2 -2
  28. package/dist/{chunk-UFSG7ACU.js → chunk-JZXGWLK5.js} +1 -1
  29. package/dist/chunk-JZXGWLK5.js.map +1 -0
  30. package/dist/{chunk-IN3SHQML.js → chunk-LMSQ6EFA.js} +2 -2
  31. package/dist/{chunk-AIGWQEME.js → chunk-LOZOZYHU.js} +2 -2
  32. package/dist/{chunk-NF7OZ4J7.js → chunk-QCBB6ZIU.js} +2 -2
  33. package/dist/chunk-RQNOLV3I.js +855 -0
  34. package/dist/chunk-RQNOLV3I.js.map +1 -0
  35. package/dist/{chunk-JU6ZX3CX.js → chunk-SMCACT4Z.js} +2 -2
  36. package/dist/{chunk-6Q2DYRWV.js → chunk-TCPP4Z5Q.js} +5 -5
  37. package/dist/{chunk-VDGPPGE3.js → chunk-UMEAR2FI.js} +2 -2
  38. package/dist/{chunk-VDGPPGE3.js.map → chunk-UMEAR2FI.js.map} +1 -1
  39. package/dist/{chunk-QIT2XZ4E.js → chunk-VNM52AGA.js} +3 -3
  40. package/dist/chunk-VNM52AGA.js.map +1 -0
  41. package/dist/{chunk-CMJSTXUR.js → chunk-YFYIOSNV.js} +107 -20
  42. package/dist/chunk-YFYIOSNV.js.map +1 -0
  43. package/dist/{code-agent-session-Ce-9u7YM.d.ts → code-agent-session-Cen-qD2y.d.ts} +1 -1
  44. package/dist/contract/index.d.ts +18 -18
  45. package/dist/contract/index.js +10 -10
  46. package/dist/{control-C8RmK9H4.d.ts → control-KofK3gfG.d.ts} +1 -1
  47. package/dist/control.d.ts +2 -2
  48. package/dist/control.js +3 -3
  49. package/dist/{corpus-CiSzzLa5.d.ts → corpus-CysLCiK4.d.ts} +1 -1
  50. package/dist/{default-registry-ZhqsTr4K.d.ts → default-registry-Ez4cxuZ4.d.ts} +2 -2
  51. package/dist/diagnose.d.ts +3 -3
  52. package/dist/diagnose.js +3 -3
  53. package/dist/{gepa-DeyPTlvx.d.ts → gepa-COlCAkHN.d.ts} +22 -1
  54. package/dist/governance/index.d.ts +1 -1
  55. package/dist/hosted/index.d.ts +5 -5
  56. package/dist/{index-B-bFgiAF.d.ts → index-unCSYRJJ.d.ts} +1 -1
  57. package/dist/index.d.ts +36 -28
  58. package/dist/index.js +32 -18
  59. package/dist/index.js.map +1 -1
  60. package/dist/{insight-report-k0sRTzKg.d.ts → insight-report-DumCfEur.d.ts} +2 -2
  61. package/dist/{judge-calibration-0p2QcWNE.d.ts → judge-calibration-7C-IDmKr.d.ts} +3 -0
  62. package/dist/{kind-factory-D0nk7AKV.d.ts → kind-factory-BLHxwX71.d.ts} +1 -1
  63. package/dist/meta-eval/index.d.ts +4 -4
  64. package/dist/meta-eval/index.js +3 -3
  65. package/dist/{multi-layer-verifier-DUZXrPDA.d.ts → multi-layer-verifier-CI4jdX-q.d.ts} +3 -0
  66. package/dist/multishot/index.d.ts +2 -2
  67. package/dist/openapi.json +1 -1
  68. package/dist/pipelines/index.d.ts +1 -1
  69. package/dist/pipelines/index.js +3 -3
  70. package/dist/{policy-edit-BDQzzsBU.d.ts → policy-edit-nUhuFtLF.d.ts} +2 -2
  71. package/dist/{pre-registration-Dzg61IQA.d.ts → pre-registration-CUOSGAZK.d.ts} +45 -12
  72. package/dist/product-benchmark/index.d.ts +104 -1
  73. package/dist/product-benchmark/index.js +15 -1
  74. package/dist/{provenance-BhJm32vN.d.ts → provenance-B2VsP0jP.d.ts} +20 -4
  75. package/dist/{query-B7GGjRox.d.ts → query-0aTmbmQe.d.ts} +1 -0
  76. package/dist/{release-report-BQ1Ziyu-.d.ts → release-report-DkaCZ9k4.d.ts} +5 -2
  77. package/dist/reporting.d.ts +6 -6
  78. package/dist/reporting.js +4 -4
  79. package/dist/{researcher-B_ODTAJs.d.ts → researcher-SDAezfML.d.ts} +2 -2
  80. package/dist/rl.d.ts +9 -9
  81. package/dist/rl.js +7 -7
  82. package/dist/{rubric-predictive-validity-0MdjTt8R.d.ts → rubric-predictive-validity-ZGIlJNce.d.ts} +1 -1
  83. package/dist/{run-campaign-3NWW5PLF.js → run-campaign-DAHKO5CT.js} +3 -3
  84. package/dist/{run-record-MRdJ-Kq2.d.ts → run-record-CPfd1ARZ.d.ts} +3 -0
  85. package/dist/{runtime-trajectory-8w0_jmtR.d.ts → runtime-trajectory-DZ8ei-Jo.d.ts} +1 -1
  86. package/dist/{semantic-concept-judge-D-IlH5v1.d.ts → semantic-concept-judge-C6mDEBIo.d.ts} +3 -3
  87. package/dist/{statistics-xP-cWc5k.d.ts → statistics-D88peojY.d.ts} +1 -1
  88. package/dist/{summary-report-C0nnxOD8.d.ts → summary-report-Fc_YFJat.d.ts} +1 -1
  89. package/dist/traces.d.ts +2 -2
  90. package/dist/traces.js +4 -4
  91. package/dist/{types-DFI_Z-ZL.d.ts → types-Bihq6-a3.d.ts} +1 -1
  92. package/dist/{types-Dz9cKF0g.d.ts → types-DA9yj-Jd.d.ts} +1 -1
  93. package/dist/workflow/index.d.ts +7 -7
  94. package/dist/workflow/index.js +3 -3
  95. package/package.json +1 -1
  96. package/dist/chunk-63MBSQTX.js +0 -350
  97. package/dist/chunk-63MBSQTX.js.map +0 -1
  98. package/dist/chunk-ABOIVNXL.js.map +0 -1
  99. package/dist/chunk-CMJSTXUR.js.map +0 -1
  100. package/dist/chunk-QIT2XZ4E.js.map +0 -1
  101. package/dist/chunk-UFSG7ACU.js.map +0 -1
  102. package/dist/chunk-YLKDN7JV.js.map +0 -1
  103. package/dist/chunk-Z3FLN24V.js.map +0 -1
  104. /package/dist/{chunk-7RBJANJD.js.map → chunk-3722PKPB.js.map} +0 -0
  105. /package/dist/{chunk-6FIAJHCU.js.map → chunk-6YVAN6R3.js.map} +0 -0
  106. /package/dist/{chunk-B2TMQM62.js.map → chunk-BHCFJGL4.js.map} +0 -0
  107. /package/dist/{chunk-IH7LYRHL.js.map → chunk-BXZVBX3D.js.map} +0 -0
  108. /package/dist/{chunk-6GT4NI4V.js.map → chunk-CLS3374R.js.map} +0 -0
  109. /package/dist/{chunk-2NSLDY4B.js.map → chunk-J5MUFXEY.js.map} +0 -0
  110. /package/dist/{chunk-IN3SHQML.js.map → chunk-LMSQ6EFA.js.map} +0 -0
  111. /package/dist/{chunk-AIGWQEME.js.map → chunk-LOZOZYHU.js.map} +0 -0
  112. /package/dist/{chunk-NF7OZ4J7.js.map → chunk-QCBB6ZIU.js.map} +0 -0
  113. /package/dist/{chunk-JU6ZX3CX.js.map → chunk-SMCACT4Z.js.map} +0 -0
  114. /package/dist/{chunk-6Q2DYRWV.js.map → chunk-TCPP4Z5Q.js.map} +0 -0
  115. /package/dist/{run-campaign-3NWW5PLF.js.map → run-campaign-DAHKO5CT.js.map} +0 -0
@@ -1,36 +1,36 @@
1
- import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, f as SurfaceProposer, g as Gate, L as LabeledScenarioStore, C as CampaignResult, j as GateDecision } from '../types-DFI_Z-ZL.js';
2
- export { k as CampaignAggregates, l as CampaignArtifactWriter, m as CampaignCellResult, n as CampaignCostMeter, d as CampaignTraceWriter, o as CodeSurface, D as Dispatch, h as GateContext, G as GateResult, p as GenerationCandidate, e as GenerationRecord, c as JudgeDimension, J as JudgeScore, i as Mutator, O as OptimizationProposer, q as OptimizerConfig, r as SessionScript } from '../types-DFI_Z-ZL.js';
3
- import { L as LoopProvenanceRecord, R as RunEvalOptions } from '../provenance-BhJm32vN.js';
4
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, e as buildEvidenceVector, f as composeGate, g as defaultProductionGate, h as evolutionaryProposer, i as heldOutGate, p as paretoPolicy, j as paretoSignificanceGate, r as runEval } from '../provenance-BhJm32vN.js';
5
- import { C as CampaignStorage, a as RunOptimizationOptions, b as RunImprovementLoopResult } from '../gepa-DeyPTlvx.js';
6
- export { G as GepaProposerOptions, R as RunCampaignOptions, c as RunImprovementLoopOptions, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, r as runCampaign, d as runImprovementLoop } from '../gepa-DeyPTlvx.js';
1
+ import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, f as SurfaceProposer, g as Gate, L as LabeledScenarioStore, C as CampaignResult, j as GateDecision } from '../types-Bihq6-a3.js';
2
+ export { k as CampaignAggregates, l as CampaignArtifactWriter, m as CampaignCellResult, n as CampaignCostMeter, d as CampaignTraceWriter, o as CodeSurface, D as Dispatch, h as GateContext, G as GateResult, p as GenerationCandidate, e as GenerationRecord, c as JudgeDimension, J as JudgeScore, i as Mutator, O as OptimizationProposer, q as OptimizerConfig, r as SessionScript } from '../types-Bihq6-a3.js';
3
+ import { L as LoopProvenanceRecord, R as RunEvalOptions } from '../provenance-B2VsP0jP.js';
4
+ export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, e as buildEvidenceVector, f as composeGate, g as defaultProductionGate, h as evolutionaryProposer, i as heldOutGate, p as paretoPolicy, j as paretoSignificanceGate, r as runEval } from '../provenance-B2VsP0jP.js';
5
+ import { C as CampaignStorage, a as RunOptimizationOptions, b as RunImprovementLoopResult } from '../gepa-COlCAkHN.js';
6
+ export { G as GepaProposerOptions, R as RunCampaignOptions, c as RunImprovementLoopOptions, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, r as runCampaign, d as runImprovementLoop } from '../gepa-COlCAkHN.js';
7
7
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
8
8
  import { HostedTenant, EvalRunCellScore, EvalRunGenerationSnapshot, EvalRunEvent, TraceSpanEvent } from '../hosted/index.js';
9
- import { R as RunRecord, b as RunSplitTag } from '../run-record-MRdJ-Kq2.js';
10
- import { I as InsightReport } from '../insight-report-k0sRTzKg.js';
11
- export { F as FailureClusterInsight, a as InterRaterInsight, J as JudgeInsight, L as LiftInsight, O as OutcomeCorrelationInsight, R as Recommendation, b as ReleaseSummary, S as ScalarDistribution } from '../insight-report-k0sRTzKg.js';
12
- export { D as DefaultAnalystRegistryOptions, c as buildDefaultAnalystRegistry } from '../default-registry-ZhqsTr4K.js';
13
- export { A as AnalystFinding } from '../types-Dz9cKF0g.js';
14
- import { A as AnalyzeRunsOptions } from '../analyze-runs-Cd-A_K4l.js';
15
- export { a as analyzeRuns } from '../analyze-runs-Cd-A_K4l.js';
16
- export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-Ce-9u7YM.js';
9
+ import { R as RunRecord, b as RunSplitTag } from '../run-record-CPfd1ARZ.js';
10
+ import { I as InsightReport } from '../insight-report-DumCfEur.js';
11
+ export { F as FailureClusterInsight, a as InterRaterInsight, J as JudgeInsight, L as LiftInsight, O as OutcomeCorrelationInsight, R as Recommendation, b as ReleaseSummary, S as ScalarDistribution } from '../insight-report-DumCfEur.js';
12
+ export { D as DefaultAnalystRegistryOptions, c as buildDefaultAnalystRegistry } from '../default-registry-Ez4cxuZ4.js';
13
+ export { A as AnalystFinding } from '../types-DA9yj-Jd.js';
14
+ import { A as AnalyzeRunsOptions } from '../analyze-runs-rF2eGxZ_.js';
15
+ export { a as analyzeRuns } from '../analyze-runs-rF2eGxZ_.js';
16
+ export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-Cen-qD2y.js';
17
17
  import '../red-team-BWdoyleI.js';
18
18
  import '../dataset-BbGkaN2I.js';
19
19
  import '../errors-CzMUYo7b.js';
20
20
  import '../store-BcFXE6LG.js';
21
21
  import '../schema-m0gsnbt3.js';
22
22
  import '../pareto-E-pembql.js';
23
- import '../statistics-xP-cWc5k.js';
24
- import '../judge-calibration-0p2QcWNE.js';
23
+ import '../statistics-D88peojY.js';
24
+ import '../judge-calibration-7C-IDmKr.js';
25
25
  import '../types-C7DGg5ex.js';
26
26
  import '@tangle-network/tcloud';
27
27
  import '../llm-client-Bj7g0rqu.js';
28
28
  import '../raw-provider-sink-C46HDghv.js';
29
29
  import '@tangle-network/agent-interface';
30
- import '../summary-report-C0nnxOD8.js';
30
+ import '../summary-report-Fc_YFJat.js';
31
31
  import '../failure-cluster-DH9Flgcf.js';
32
32
  import '@ax-llm/ax';
33
- import '../kind-factory-D0nk7AKV.js';
33
+ import '../kind-factory-BLHxwX71.js';
34
34
  import '../store-C1YxJDEK.js';
35
35
  import 'zod';
36
36
 
@@ -18,38 +18,38 @@ import {
18
18
  paretoPolicy,
19
19
  paretoSignificanceGate,
20
20
  runEval
21
- } from "../chunk-YLKDN7JV.js";
21
+ } from "../chunk-66T25VWE.js";
22
22
  import {
23
23
  analyzeRuns
24
- } from "../chunk-6Q2DYRWV.js";
24
+ } from "../chunk-TCPP4Z5Q.js";
25
25
  import {
26
26
  emitLoopProvenance,
27
27
  gepaProposer,
28
28
  heldOutGate,
29
29
  runImprovementLoop,
30
30
  surfaceContentHash
31
- } from "../chunk-Z3FLN24V.js";
31
+ } from "../chunk-2B4HPFXN.js";
32
32
  import {
33
33
  fsCampaignStorage,
34
34
  inMemoryCampaignStorage,
35
35
  runCampaign
36
- } from "../chunk-QIT2XZ4E.js";
36
+ } from "../chunk-VNM52AGA.js";
37
37
  import "../chunk-VI2UW6B6.js";
38
38
  import {
39
39
  buildDefaultAnalystRegistry
40
40
  } from "../chunk-L5TVEZFT.js";
41
41
  import "../chunk-FRI6RG3P.js";
42
- import "../chunk-AIGWQEME.js";
42
+ import "../chunk-LOZOZYHU.js";
43
43
  import {
44
44
  FileSystemOutcomeStore,
45
45
  InMemoryOutcomeStore
46
46
  } from "../chunk-3RF76KTD.js";
47
47
  import "../chunk-CWNP4DV4.js";
48
- import "../chunk-6GT4NI4V.js";
48
+ import "../chunk-CLS3374R.js";
49
49
  import "../chunk-45EEMHTC.js";
50
- import "../chunk-HRGUJTER.js";
50
+ import "../chunk-HYGRFL7C.js";
51
51
  import "../chunk-GGE4NNQT.js";
52
- import "../chunk-UFSG7ACU.js";
52
+ import "../chunk-JZXGWLK5.js";
53
53
  import "../chunk-5BKGXME7.js";
54
54
  import {
55
55
  LLM_INPUT_TOKEN_ATTR_KEYS,
@@ -59,8 +59,8 @@ import {
59
59
  import "../chunk-PC4UYEBM.js";
60
60
  import {
61
61
  parseRunRecordSafe
62
- } from "../chunk-2NSLDY4B.js";
63
- import "../chunk-ABOIVNXL.js";
62
+ } from "../chunk-J5MUFXEY.js";
63
+ import "../chunk-AGYDMORK.js";
64
64
  import "../chunk-VSMTAMNK.js";
65
65
  import "../chunk-3BFEG2F6.js";
66
66
  import "../chunk-PZ5AY32C.js";
@@ -3,7 +3,7 @@ import { C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfi
3
3
  import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
4
4
  import { F as FailureClass } from './schema-m0gsnbt3.js';
5
5
  import { T as TraceStore } from './store-BcFXE6LG.js';
6
- import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-MRdJ-Kq2.js';
6
+ import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-CPfd1ARZ.js';
7
7
 
8
8
  interface ActionExecutionPolicy {
9
9
  allowedTypes?: string[];
package/dist/control.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-C8RmK9H4.js';
1
+ export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-KofK3gfG.js';
2
2
  export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-Acf9CGhw.js';
3
3
  import './feedback-trajectory-BxY0cKfs.js';
4
4
  import './dataset-BbGkaN2I.js';
@@ -6,5 +6,5 @@ import './errors-CzMUYo7b.js';
6
6
  import './emitter-C2rqGH_l.js';
7
7
  import './schema-m0gsnbt3.js';
8
8
  import './store-BcFXE6LG.js';
9
- import './run-record-MRdJ-Kq2.js';
9
+ import './run-record-CPfd1ARZ.js';
10
10
  import '@tangle-network/agent-interface';
package/dist/control.js CHANGED
@@ -4,7 +4,7 @@ import {
4
4
  runProposeReview,
5
5
  runProposeReviewAsControlLoop,
6
6
  scoreFromEvals
7
- } from "./chunk-B2TMQM62.js";
7
+ } from "./chunk-BHCFJGL4.js";
8
8
  import {
9
9
  allCriticalPassed,
10
10
  objectiveEval,
@@ -13,9 +13,9 @@ import {
13
13
  stopOnRepeatedAction,
14
14
  subjectiveEval
15
15
  } from "./chunk-YEHAEDUD.js";
16
- import "./chunk-2NSLDY4B.js";
16
+ import "./chunk-J5MUFXEY.js";
17
17
  import "./chunk-TVVP3ZZQ.js";
18
- import "./chunk-ABOIVNXL.js";
18
+ import "./chunk-AGYDMORK.js";
19
19
  import "./chunk-VSMTAMNK.js";
20
20
  import "./chunk-3BFEG2F6.js";
21
21
  import "./chunk-PZ5AY32C.js";
@@ -1,4 +1,4 @@
1
- import { R as RunRecord, b as RunSplitTag } from './run-record-MRdJ-Kq2.js';
1
+ import { R as RunRecord, b as RunSplitTag } from './run-record-CPfd1ARZ.js';
2
2
  import { S as Span } from './schema-m0gsnbt3.js';
3
3
  import { T as TraceStore } from './store-BcFXE6LG.js';
4
4
 
@@ -1,6 +1,6 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { T as TraceAnalystKindSpec } from './kind-factory-D0nk7AKV.js';
3
- import { a as Analyst, b as AnalystContext, c as AnalystRunSummary, A as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-Dz9cKF0g.js';
2
+ import { T as TraceAnalystKindSpec } from './kind-factory-BLHxwX71.js';
3
+ import { a as Analyst, b as AnalystContext, c as AnalystRunSummary, A as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-DA9yj-Jd.js';
4
4
 
5
5
  /**
6
6
  * AnalystRegistry — orchestrate N analysts against one run.
@@ -3,9 +3,9 @@ export { c as CounterfactualContext, d as CounterfactualResult } from './counter
3
3
  import { S as Span } from './schema-m0gsnbt3.js';
4
4
  import { T as TraceStore } from './store-BcFXE6LG.js';
5
5
  import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
6
- import { h as AnalystSeverity, A as AnalystFinding } from './types-Dz9cKF0g.js';
7
- import { C as CorpusRecord } from './corpus-CiSzzLa5.js';
8
- import { R as RunRecord } from './run-record-MRdJ-Kq2.js';
6
+ import { h as AnalystSeverity, A as AnalystFinding } from './types-DA9yj-Jd.js';
7
+ import { C as CorpusRecord } from './corpus-CysLCiK4.js';
8
+ import { R as RunRecord } from './run-record-CPfd1ARZ.js';
9
9
  import './emitter-C2rqGH_l.js';
10
10
  import './store-C1YxJDEK.js';
11
11
  import './types-C7DGg5ex.js';
package/dist/diagnose.js CHANGED
@@ -10,12 +10,12 @@ import {
10
10
  } from "./chunk-45EEMHTC.js";
11
11
  import {
12
12
  confidenceInterval
13
- } from "./chunk-HRGUJTER.js";
13
+ } from "./chunk-HYGRFL7C.js";
14
14
  import {
15
15
  validateRunRecord
16
- } from "./chunk-2NSLDY4B.js";
16
+ } from "./chunk-J5MUFXEY.js";
17
17
  import "./chunk-TVVP3ZZQ.js";
18
- import "./chunk-ABOIVNXL.js";
18
+ import "./chunk-AGYDMORK.js";
19
19
  import "./chunk-VSMTAMNK.js";
20
20
  import {
21
21
  ValidationError
@@ -1,4 +1,4 @@
1
- import { S as Scenario, C as CampaignResult, G as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-DFI_Z-ZL.js';
1
+ import { S as Scenario, C as CampaignResult, G as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-Bihq6-a3.js';
2
2
  import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
3
3
 
4
4
  /**
@@ -45,6 +45,9 @@ interface OpenAutoPrResult {
45
45
  dryRun: boolean;
46
46
  reason: string;
47
47
  }
48
+ /**
49
+ * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
50
+ */
48
51
  declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
49
52
 
50
53
  /**
@@ -183,6 +186,9 @@ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
183
186
  generation?: number;
184
187
  }) => string | undefined;
185
188
  }
189
+ /**
190
+ * Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
191
+ */
186
192
  declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
187
193
  interface CampaignRunPlanCell {
188
194
  cellId: string;
@@ -303,7 +309,13 @@ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
303
309
  * NOT the winner is uniquely best on some scenario the winner loses on. */
304
310
  paretoFrontier: ParetoParent[];
305
311
  }
312
+ /**
313
+ * Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and promoting the top-scoring candidates to the next generation.
314
+ */
306
315
  declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
316
+ /**
317
+ * Short (16-char) sha256 fingerprint of a `MutableSurface`: hashes text content for prompt surfaces, or the worktree + base ref pair for code surfaces.
318
+ */
307
319
  declare function surfaceHash(surface: MutableSurface): string;
308
320
 
309
321
  /**
@@ -364,7 +376,13 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extend
364
376
  promotedDiff: string;
365
377
  prResult?: ReturnType<typeof openAutoPr>;
366
378
  }
379
+ /**
380
+ * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
381
+ */
367
382
  declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
383
+ /**
384
+ * Default surface diff renderer: produces a unified baseline/winner text diff for prompt surfaces or a worktree-ref summary for code surfaces.
385
+ */
368
386
  declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
369
387
 
370
388
  /**
@@ -443,6 +461,9 @@ interface GepaProposerOptions {
443
461
  * composite first), to bound prompt size. Default 4. */
444
462
  combineMaxParents?: number;
445
463
  }
464
+ /**
465
+ * GEPA reflective proposer: each generation reflects on the weakest scenarios and dimensions to produce targeted prompt rewrites, optionally combining Pareto-frontier parents.
466
+ */
446
467
  declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
447
468
  /** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
448
469
  * consumers building custom mutators that share the same invariant. */
@@ -1,5 +1,5 @@
1
1
  import { c as DatasetManifest } from '../dataset-BbGkaN2I.js';
2
- import { a as CalibrationResult } from '../judge-calibration-0p2QcWNE.js';
2
+ import { a as CalibrationResult } from '../judge-calibration-7C-IDmKr.js';
3
3
  import { b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
4
4
  import { d as RedTeamReport } from '../red-team-BWdoyleI.js';
5
5
  import { T as TraceStore } from '../store-BcFXE6LG.js';
@@ -1,13 +1,13 @@
1
- import { M as MutableSurface, j as GateDecision } from '../types-DFI_Z-ZL.js';
2
- import { I as InsightReport } from '../insight-report-k0sRTzKg.js';
3
- import '../run-record-MRdJ-Kq2.js';
1
+ import { M as MutableSurface, j as GateDecision } from '../types-Bihq6-a3.js';
2
+ import { I as InsightReport } from '../insight-report-DumCfEur.js';
3
+ import '../run-record-CPfd1ARZ.js';
4
4
  import '@tangle-network/agent-interface';
5
5
  import '../errors-CzMUYo7b.js';
6
6
  import '../schema-m0gsnbt3.js';
7
- import '../summary-report-C0nnxOD8.js';
7
+ import '../summary-report-Fc_YFJat.js';
8
8
  import '../failure-cluster-DH9Flgcf.js';
9
9
  import '../store-BcFXE6LG.js';
10
- import '../judge-calibration-0p2QcWNE.js';
10
+ import '../judge-calibration-7C-IDmKr.js';
11
11
 
12
12
  /**
13
13
  * # Hosted-tier wire format — the schema that EVERY orchestrator (ours,
@@ -1,4 +1,4 @@
1
- import { b as RunSplitTag } from './run-record-MRdJ-Kq2.js';
1
+ import { b as RunSplitTag } from './run-record-CPfd1ARZ.js';
2
2
 
3
3
  /**
4
4
  * Shared types for the reference benchmark wrappers under
package/dist/index.d.ts CHANGED
@@ -1,13 +1,13 @@
1
- export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-C8RmK9H4.js';
2
- import { R as RunRecord, b as RunSplitTag } from './run-record-MRdJ-Kq2.js';
3
- export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-MRdJ-Kq2.js';
4
- import { B as BehavioralMetrics } from './semantic-concept-judge-D-IlH5v1.js';
5
- export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-D-IlH5v1.js';
6
- import { l as ChatRequest, p as CreateChatClientOpts } from './types-Dz9cKF0g.js';
7
- export { a as Analyst, b as AnalystContext, g as AnalystCost, A as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-Dz9cKF0g.js';
8
- export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-ZhqsTr4K.js';
9
- export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-D0nk7AKV.js';
10
- export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-BDQzzsBU.js';
1
+ export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-KofK3gfG.js';
2
+ import { R as RunRecord, b as RunSplitTag } from './run-record-CPfd1ARZ.js';
3
+ export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-CPfd1ARZ.js';
4
+ import { B as BehavioralMetrics } from './semantic-concept-judge-C6mDEBIo.js';
5
+ export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-C6mDEBIo.js';
6
+ import { l as ChatRequest, p as CreateChatClientOpts } from './types-DA9yj-Jd.js';
7
+ export { a as Analyst, b as AnalystContext, g as AnalystCost, A as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-DA9yj-Jd.js';
8
+ export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-Ez4cxuZ4.js';
9
+ export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-BLHxwX71.js';
10
+ export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-nUhuFtLF.js';
11
11
  import { TCloud } from '@tangle-network/tcloud';
12
12
  import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
13
13
  export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
@@ -18,20 +18,20 @@ import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors
18
18
  export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-CzMUYo7b.js';
19
19
  import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-BxY0cKfs.js';
20
20
  export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-BxY0cKfs.js';
21
- import { b as CorrectnessChecker } from './pre-registration-Dzg61IQA.js';
22
- export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-Dzg61IQA.js';
21
+ import { b as CorrectnessChecker } from './pre-registration-CUOSGAZK.js';
22
+ export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-CUOSGAZK.js';
23
23
  export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
24
- import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-BQ1Ziyu-.js';
25
- export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-BQ1Ziyu-.js';
26
- export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as ProportionInterval, R as RiskDifferenceResult, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as pearsonR, K as ranks, L as requiredSampleSize, N as spearmanR, O as weightedComposite, Q as weightedMean, w as wilcoxonSignedRank, S as wilson } from './statistics-xP-cWc5k.js';
24
+ import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-DkaCZ9k4.js';
25
+ export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-DkaCZ9k4.js';
26
+ export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as ProportionInterval, R as RiskDifferenceResult, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as pearsonR, K as ranks, L as requiredSampleSize, N as spearmanR, O as weightedComposite, Q as weightedMean, w as wilcoxonSignedRank, S as wilson } from './statistics-D88peojY.js';
27
27
  import { OtelExporter, OtelExportConfig } from './traces.js';
28
28
  export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
29
29
  import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
30
30
  export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
31
31
  import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
32
32
  export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
33
- import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-DFI_Z-ZL.js';
34
- import { A as AnalyzeRunsOptions } from './analyze-runs-Cd-A_K4l.js';
33
+ import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-Bihq6-a3.js';
34
+ import { A as AnalyzeRunsOptions } from './analyze-runs-rF2eGxZ_.js';
35
35
  import { S as SteeringBundle } from './harness-optimizer-mOl9XX_O.js';
36
36
  export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-mOl9XX_O.js';
37
37
  import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-DeODGLra.js';
@@ -41,13 +41,13 @@ export { D as DEFAULT_RUN_SCORE_WEIGHTS, c as RunCritic, d as RunCriticOptions,
41
41
  import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
42
42
  export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-C2rqGH_l.js';
43
43
  export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-D2t12mMw.js';
44
- export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-B7GGjRox.js';
44
+ export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-0aTmbmQe.js';
45
45
  export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
46
46
  export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
47
47
  import { T as TraceStore, R as RunFilter } from './store-BcFXE6LG.js';
48
48
  export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BcFXE6LG.js';
49
49
  export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-DH9Flgcf.js';
50
- export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-8w0_jmtR.js';
50
+ export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-DZ8ei-Jo.js';
51
51
  import { a as BaselineReport } from './baseline-Bbid3WoO.js';
52
52
  export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-Bbid3WoO.js';
53
53
  import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
@@ -61,26 +61,26 @@ export { S as SeriesConvergenceOptions, a as SeriesConvergenceResult, b as analy
61
61
  import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
62
62
  import { a as DatasetScenario, b as Dataset } from './dataset-BbGkaN2I.js';
63
63
  export { d as DatasetDifficulty, c as DatasetManifest, e as DatasetProvenance, D as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-BbGkaN2I.js';
64
- export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-0p2QcWNE.js';
64
+ export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-7C-IDmKr.js';
65
65
  export { D as DEFAULT_RED_TEAM_CORPUS, R as RedTeamCase, a as RedTeamCategory, b as RedTeamFinding, c as RedTeamPayload, d as RedTeamReport, r as redTeamDataset, e as redTeamReport, s as scoreRedTeamOutput, t as toolNamesForRun } from './red-team-BWdoyleI.js';
66
66
  export { c as CounterfactualContext, C as CounterfactualMutation, d as CounterfactualResult, b as CounterfactualRunner, a as attributeCounterfactuals, r as runCounterfactual } from './counterfactual-DlOz8PBx.js';
67
67
  import { a as PrmGrader } from './rubric-Cc6UHvUb.js';
68
68
  export { EuRiskClass, GovernanceContext, GovernanceFinding, GovernanceReport, UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize } from './governance/index.js';
69
- import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-DUZXrPDA.js';
70
- export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DUZXrPDA.js';
69
+ import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-CI4jdX-q.js';
70
+ export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-CI4jdX-q.js';
71
71
  import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
72
72
  export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-Bj7g0rqu.js';
73
- export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-B-bFgiAF.js';
74
- export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-B_ODTAJs.js';
75
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-C0nnxOD8.js';
73
+ export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-unCSYRJJ.js';
74
+ export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-SDAezfML.js';
75
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Fc_YFJat.js';
76
76
  export { L as LockedJsonlAppender } from './testing-C21CHsq2.js';
77
77
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
78
- import { j as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-DeyPTlvx.js';
78
+ import { j as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-COlCAkHN.js';
79
79
  export { IntegrityResult, IntegrityViolation, JourneySpec, PerfBaseline, PerfGateResult, PerfRegression, PerfScenario, PerfStat, ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords } from './perf/index.js';
80
- export { AgentProfileRuntimeReceipt, ProductBenchmarkArm, ProductBenchmarkArtifactPaths, ProductBenchmarkBudgets, ProductBenchmarkManifest, ProductBenchmarkProfileRef, ProductBenchmarkRecord, ProductBenchmarkRepoRef, ProductBenchmarkRunInput, ProductBenchmarkScenario, ProductBenchmarkSplit, ProductBenchmarkSubstrateVersions, ProductBenchmarkValidationReport, RuntimeResolution, findProductBenchmarkArtifacts, productBenchmarkIntegrityFailures, productBenchmarkSplits, readProductBenchmarkManifest, readProductBenchmarkRecords, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun } from './product-benchmark/index.js';
80
+ export { AgentProfileRuntimeReceipt, ProductBenchmarkArm, ProductBenchmarkArtifactPaths, ProductBenchmarkBudgets, ProductBenchmarkExportOptions, ProductBenchmarkExportResult, ProductBenchmarkManifest, ProductBenchmarkProfileRef, ProductBenchmarkRecord, ProductBenchmarkRepoRef, ProductBenchmarkRunInput, ProductBenchmarkScenario, ProductBenchmarkSingleRunExportOptions, ProductBenchmarkSplit, ProductBenchmarkSubstrateVersions, ProductBenchmarkValidationReport, RuntimeResolution, assertProductBenchmarkRun, buildProductBenchmarkManifest, exportProductBenchmark, exportProductBenchmarkRuns, findProductBenchmarkArtifacts, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, readProductBenchmarkManifest, readProductBenchmarkRecords, runRecordToProductBenchmarkRecord, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun } from './product-benchmark/index.js';
81
81
  import '@ax-llm/ax';
82
82
  import 'zod';
83
- import './insight-report-k0sRTzKg.js';
83
+ import './insight-report-DumCfEur.js';
84
84
  import './outcome-store-rnXLEqSn.js';
85
85
 
86
86
  /**
@@ -1554,9 +1554,13 @@ declare class JudgeRunner {
1554
1554
  run(spec: SandboxJudgeSpec): Promise<SandboxJudgeResult>;
1555
1555
  }
1556
1556
  declare function runJudgeFleet(specs: SandboxJudgeSpec[], options?: JudgeFleetOptions): Promise<SandboxJudgeResult[]>;
1557
+ /** Build a `SandboxJudgeSpec` that scores whether the harness compiles without errors. */
1557
1558
  declare function compilerJudge(id: string, config: HarnessConfig): SandboxJudgeSpec;
1559
+ /** Build a `SandboxJudgeSpec` that scores the harness by its test-suite pass rate. */
1558
1560
  declare function testJudge(id: string, config: HarnessConfig): SandboxJudgeSpec;
1561
+ /** Build a `SandboxJudgeSpec` that scores the harness by linter rule violations. */
1559
1562
  declare function linterJudge(id: string, config: HarnessConfig): SandboxJudgeSpec;
1563
+ /** Build a `SandboxJudgeSpec` that scores the harness output for security issues via a security scanner. */
1560
1564
  declare function securityJudge(id: string, config: HarnessConfig): SandboxJudgeSpec;
1561
1565
 
1562
1566
  interface PlaybookEntry {
@@ -5271,6 +5275,10 @@ interface ReflectionProposal {
5271
5275
  rationale: string;
5272
5276
  payload: unknown;
5273
5277
  }
5278
+ /**
5279
+ * Parse the model's JSON response back into proposals. Tolerates markdown
5280
+ * fences and surrounding prose. Returns at most `maxProposals`.
5281
+ */
5274
5282
  declare function parseReflectionResponse(raw: string, maxProposals?: number): ReflectionProposal[];
5275
5283
 
5276
5284
  /**