@tangle-network/agent-eval 0.115.3 → 0.117.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/dist/analyst/index.d.ts +16 -11
  3. package/dist/analyst/index.js +33 -25
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  6. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/belief-state/index.js +1 -1
  9. package/dist/benchmarks/index.d.ts +12 -5
  10. package/dist/benchmarks/index.js +11 -10
  11. package/dist/builder-eval/index.d.ts +4 -4
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +247 -34
  15. package/dist/campaign/index.js +33 -13
  16. package/dist/chunk-3YYRZDON.js +45 -0
  17. package/dist/chunk-3YYRZDON.js.map +1 -0
  18. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  19. package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
  20. package/dist/chunk-CCZIVI3F.js.map +1 -0
  21. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  22. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  23. package/dist/chunk-HHWE3POT.js +94 -0
  24. package/dist/chunk-HHWE3POT.js.map +1 -0
  25. package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
  26. package/dist/chunk-HQPHZGL6.js.map +1 -0
  27. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  28. package/dist/chunk-IDZTTFRR.js.map +1 -0
  29. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  30. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  31. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  32. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  33. package/dist/chunk-LTVG32KX.js.map +1 -0
  34. package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
  35. package/dist/chunk-MGEHEHSN.js.map +1 -0
  36. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  37. package/dist/chunk-NJC7U437.js.map +1 -0
  38. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  39. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  40. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  41. package/dist/chunk-S2F4J57L.js.map +1 -0
  42. package/dist/chunk-VCTY3W6J.js +798 -0
  43. package/dist/chunk-VCTY3W6J.js.map +1 -0
  44. package/dist/chunk-VF3XSYTI.js +545 -0
  45. package/dist/chunk-VF3XSYTI.js.map +1 -0
  46. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  47. package/dist/chunk-YZPO4UHR.js.map +1 -0
  48. package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
  49. package/dist/chunk-ZUXV7UWZ.js.map +1 -0
  50. package/dist/cli.js +4 -2
  51. package/dist/cli.js.map +1 -1
  52. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  53. package/dist/contract/index.d.ts +45 -31
  54. package/dist/contract/index.js +58 -19
  55. package/dist/contract/index.js.map +1 -1
  56. package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
  57. package/dist/control.d.ts +6 -6
  58. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  59. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
  60. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  61. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  62. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  63. package/dist/fuzz.d.ts +8 -16
  64. package/dist/fuzz.js +72 -42
  65. package/dist/fuzz.js.map +1 -1
  66. package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
  67. package/dist/hosted/index.d.ts +14 -7
  68. package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
  69. package/dist/index.d.ts +97 -55
  70. package/dist/index.js +343 -244
  71. package/dist/index.js.map +1 -1
  72. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  73. package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
  74. package/dist/kind-factory-ClZmO25A.d.ts +171 -0
  75. package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
  76. package/dist/meta-eval/index.d.ts +8 -7
  77. package/dist/meta-eval/index.js +1 -1
  78. package/dist/multishot/index.d.ts +10 -3
  79. package/dist/openapi.json +1 -1
  80. package/dist/pipelines/index.d.ts +16 -6
  81. package/dist/pipelines/index.js +119 -23
  82. package/dist/pipelines/index.js.map +1 -1
  83. package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
  84. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
  85. package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
  86. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  87. package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  88. package/dist/reporting.d.ts +10 -9
  89. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
  90. package/dist/rl.d.ts +17 -12
  91. package/dist/rl.js +2 -2
  92. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  93. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  94. package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
  95. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  96. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  97. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
  98. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  99. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  100. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  101. package/dist/storyboard/index.d.ts +1 -1
  102. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  103. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  104. package/dist/traces.d.ts +19 -10
  105. package/dist/traces.js +16 -4
  106. package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
  107. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  108. package/dist/wire/index.d.ts +28 -19
  109. package/dist/wire/index.js +4 -2
  110. package/docs/design/loop-taxonomy.md +1 -2
  111. package/docs/distributed-driver.md +1 -1
  112. package/package.json +3 -3
  113. package/dist/chunk-4D5RVB3W.js.map +0 -1
  114. package/dist/chunk-5S5NJ63F.js.map +0 -1
  115. package/dist/chunk-ADYLPOSX.js.map +0 -1
  116. package/dist/chunk-FAOEFFRT.js.map +0 -1
  117. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  118. package/dist/chunk-I6LVHOV3.js +0 -205
  119. package/dist/chunk-I6LVHOV3.js.map +0 -1
  120. package/dist/chunk-KG4TD7EQ.js.map +0 -1
  121. package/dist/chunk-LNQEP766.js.map +0 -1
  122. package/dist/chunk-MHNQWM4I.js.map +0 -1
  123. package/dist/chunk-NYFUT3B3.js.map +0 -1
  124. package/dist/chunk-QMXXSNC4.js +0 -761
  125. package/dist/chunk-QMXXSNC4.js.map +0 -1
  126. package/dist/chunk-TLDB7WRY.js.map +0 -1
  127. package/dist/chunk-WSBUZMBU.js.map +0 -1
  128. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  129. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  130. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  131. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  132. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  133. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  134. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
@@ -1,35 +1,37 @@
1
- import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-D8h7ZxNL.js';
2
- export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-D8h7ZxNL.js';
1
+ import { v as PolicyEditAdmissionOptions, u as PolicyEditAdmission, t as PolicyEdit, c as AnalystFinding, F as FindingToPolicyEditOptions } from '../policy-edit-wG9uFEFm.js';
2
+ export { r as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, P as PolicyEditCandidateRecord, a2 as validatePolicyEditCandidateRecord } from '../policy-edit-wG9uFEFm.js';
3
+ import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-BWQhJ3vz.js';
4
+ export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-BWQhJ3vz.js';
5
+ import { C as CampaignRunPlan, P as PlanCampaignRunOptions, h as RunCampaignOptions, i as RunImprovementLoopOptions } from '../gepa-eESocoDi.js';
6
+ export { o as CampaignRunPlanCell, p as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, q as OpenAutoPrResult, e as ReferenceEquivalenceJudgeOptions, g as ReferenceEquivalenceScenario, a as RunImprovementLoopResult, R as RunOptimizationOptions, s as RunOptimizationResult, t as countSentenceEdits, j as createReferenceEquivalenceJudge, u as defaultRenderDiff, v as extractH2Sections, k as gepaProposer, w as openAutoPr, x as planCampaignRun, r as runCampaign, l as runImprovementLoop, y as runOptimization } from '../gepa-eESocoDi.js';
3
7
  import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
4
- import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, g as Gate, e as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, m as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-C5gJrOVT.js';
5
- export { i as CampaignAggregates, j as CampaignArtifactWriter, k as CampaignCellResult, l as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, c as DispatchFn, n as GateContext, h as GateDecision, G as GateResult, o as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-C5gJrOVT.js';
6
- import { C as CampaignRunPlan, P as PlanCampaignRunOptions, b as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-dne9JDPL.js';
7
- export { f as CampaignRunPlanCell, h as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, i as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, g as gepaProposer, o as openAutoPr, p as planCampaignRun, r as runCampaign, d as runImprovementLoop, n as runOptimization } from '../gepa-dne9JDPL.js';
8
- import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-oUbOJe-S.js';
9
- import { C as CampaignStorage } from '../storage-Dw_f7WMt.js';
10
- export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-Dw_f7WMt.js';
11
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-Bibyg1U9.js';
12
- import { L as LlmClientOptions } from '../llm-client-DyqEH4jH.js';
8
+ import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, G as Gate, o as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, c as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, j as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-BSw1rOUB.js';
9
+ export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, z as CampaignTokenUsage, i as CampaignTraceWriter, k as DispatchFn, l as GateContext, d as GateDecision, m as GateResult, n as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, E as ScoredSurfaceOutcome, r as SessionScript, T as TraceSpan, F as isProposedCandidate, H as labelTrustRank } from '../types-BSw1rOUB.js';
10
+ import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-KUnG73jH.js';
11
+ import { C as CampaignStorage } from '../storage-DrX3v_5B.js';
12
+ export { c as createRunCostLedger, f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-DrX3v_5B.js';
13
+ export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-DpjwyseI.js';
14
+ import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
13
15
  import { AgentProfile } from '@tangle-network/agent-interface';
14
16
  import { A as AgentEvalError, V as ValidationError } from '../errors-oeQrLqXC.js';
15
- import { b as RunSplitTag, R as RunRecord } from '../run-record-B7RTi_ix.js';
16
- import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-RLn8GWof.js';
17
- import { T as TraceAnalystKindSpec, A as AnalystFinding } from '../kind-factory-DcNg13sZ.js';
17
+ import { a as RunSplitTag, R as RunRecord } from '../run-record-BDH49H2E.js';
18
+ import { C as CostLedger, b as CostLedgerSummary, M as MaximumCharge, c as CostReceiptInput } from '../cost-ledger-DWy3XdJc.js';
19
+ import { T as TraceAnalystKindSpec } from '../kind-factory-ClZmO25A.js';
20
+ import '../store-C1YxJDEK.js';
21
+ import '../types-BkfcQnxV.js';
18
22
  import '@tangle-network/tcloud';
23
+ import 'zod';
19
24
  import '../raw-provider-sink-C46HDghv.js';
20
25
  import '../verdict-C9MlYujm.js';
21
- import '@ax-llm/ax';
22
- import '../store-C1YxJDEK.js';
23
26
  import '../dataset-NENEzRgk.js';
24
- import '../store-BsVi7ncX.js';
25
- import '../schema-SGWcK9wa.js';
27
+ import '../store-DGqD0Pyo.js';
28
+ import '../schema-B3Q3l9Z_.js';
29
+ import '@ax-llm/ax';
26
30
  import '../judge-calibration-7C-IDmKr.js';
27
- import '../types-C7DGg5ex.js';
28
31
  import '../hosted/index.js';
29
- import '../insight-report-IwwvqZZv.js';
30
- import '../summary-report-BJ5aNwZ1.js';
31
- import '../failure-cluster-C48PiReX.js';
32
- import 'zod';
32
+ import '../insight-report-DY4nDW9Q.js';
33
+ import '../summary-report-C5bKFfm-.js';
34
+ import '../failure-cluster-DOAcSJ87.js';
33
35
 
34
36
  /**
35
37
  * Lineage DAG — a git-graph of improvement candidates.
@@ -1469,12 +1471,12 @@ declare function fapoEscalationEntry<TScenario extends Scenario, TArtifact>(conf
1469
1471
  * It runs `runCampaign` once per profile (reusing its seeds, reps, bootstrap
1470
1472
  * CIs, resumability, and the `LabeledScenarioStore` capture flywheel), maps
1471
1473
  * every cell to a validated `RunRecord` carrying the real `tokenUsage` the
1472
- * dispatch reported via `ctx.cost.observeTokens`, and runs `assertRealBackend`
1474
+ * dispatch committed via `ctx.cost.runPaidCall`, and runs `assertRealBackend`
1473
1475
  * BY CONSTRUCTION before returning — so a stub-backend run fails loudly instead
1474
1476
  * of reporting a clean 0/N leaderboard.
1475
1477
  *
1476
1478
  * Dispatch contract: a dispatch that calls an LLM MUST report usage via
1477
- * `ctx.cost.observeTokens({ input, output })` (and cost via `ctx.cost.observe`).
1479
+ * `ctx.cost.runPaidCall({ execute, receipt })`.
1478
1480
  * A dispatch that reports zero tokens is indistinguishable from a stub and the
1479
1481
  * integrity guard treats it as one.
1480
1482
  */
@@ -1486,8 +1488,8 @@ declare class ProfileMatrixError extends AgentEvalError {
1486
1488
  constructor(message: string);
1487
1489
  }
1488
1490
  /** Dispatch for one cell: render `profile` against `scenario`, returning the
1489
- * artifact the judges score. Report LLM usage via `ctx.cost.observeTokens`
1490
- * and `ctx.cost.observe` — the integrity guard depends on it. */
1491
+ * artifact the judges score. Run LLM work through `ctx.cost.runPaidCall`
1492
+ * the integrity check depends on its receipt. */
1491
1493
  type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1492
1494
  interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
1493
1495
  /** Axis 3 — the agent-under-test configurations. Each is one column. */
@@ -1638,7 +1640,7 @@ interface PlaybackContext extends DispatchContext {
1638
1640
  * `SandboxPlaybackDriver` (real API / sandbox workspace) and
1639
1641
  * `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
1640
1642
  * browser infra the substrate must not import. The driver MUST report LLM
1641
- * usage via `ctx.cost.observeTokens` so the backend-integrity guard sees real
1643
+ * usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
1642
1644
  * tokens (a run that never reports tokens reads as a stub).
1643
1645
  */
1644
1646
  interface PlaybackDriver<TStory extends UserStory = UserStory> {
@@ -1951,10 +1953,14 @@ interface ProposePatchesArgs {
1951
1953
  /** How many candidate patches to propose. */
1952
1954
  count: number;
1953
1955
  signal: AbortSignal;
1956
+ costLedger?: CostLedger;
1957
+ costPhase?: string;
1954
1958
  }
1955
1959
  interface SkillOptProposerOptions {
1956
1960
  llm: LlmClientOptions;
1957
1961
  model: string;
1962
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
1963
+ costLedger?: CostLedger;
1958
1964
  /** What the skill document governs — orients the prompt. */
1959
1965
  target: string;
1960
1966
  /** Default ops-per-patch cap when used as a bare `SurfaceProposer`. The
@@ -1989,8 +1995,8 @@ declare function parseSkillPatchResponse(raw: string, maxPatches: number, editBu
1989
1995
 
1990
1996
  /**
1991
1997
  * `runSkillOpt` — the SkillOpt epoch hill-climb (Microsoft, arXiv:2605.23904).
1992
- * Unlike `runOptimization`'s population/promote-top-K search, SkillOpt is a
1993
- * sequential, held-out-gated hill-climb on ONE skill document:
1998
+ * Unlike `runOptimization`'s population search around one global incumbent,
1999
+ * SkillOpt is a sequential, held-out-gated hill-climb on ONE skill document:
1994
2000
  *
1995
2001
  * each epoch:
1996
2002
  * 1. reflect on the CURRENT surface's weakest TRAIN scenarios/dimensions
@@ -2078,6 +2084,8 @@ interface RunSkillOptResult {
2078
2084
  /** Total cost across every scoring campaign (train evidence + holdout
2079
2085
  * acceptance) the hill-climb ran. */
2080
2086
  totalCostUsd: number;
2087
+ /** Run-wide spend, including scoring, proposals, and judges. */
2088
+ cost: CostLedgerSummary;
2081
2089
  }
2082
2090
  /**
2083
2091
  * SkillOpt sequential hill-climb: each epoch reflects on train-scenario weaknesses, proposes bounded patches, accepts the first patch that strictly improves the held-out composite, and anneals the edit budget on consecutive rejections.
@@ -2192,6 +2200,11 @@ interface HaloProposerOptions {
2192
2200
  model?: string;
2193
2201
  /** Model used to APPLY halo's findings to the prompt surface. Default = `model`. */
2194
2202
  applyModel?: string;
2203
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2204
+ costLedger?: CostLedger;
2205
+ analysisMaximumCharge?: MaximumCharge;
2206
+ analysisReceipt?: (report: string) => CostReceiptInput;
2207
+ applyMaxTokens?: number;
2195
2208
  /** The real halo binary. Default 'halo' (from `pip install halo-engine`). */
2196
2209
  haloBin?: string;
2197
2210
  /** Resolve the OTLP traces (JSONL string) halo should analyze for THIS
@@ -2208,6 +2221,169 @@ interface HaloProposerOptions {
2208
2221
  /** Wrap the real halo-engine CLI as a SurfaceProposer (prompt-tier). */
2209
2222
  declare function haloProposer(opts: HaloProposerOptions): SurfaceProposer;
2210
2223
 
2224
+ declare const JSON_POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile"];
2225
+ type JsonPolicyEditTargetSurface = (typeof JSON_POLICY_EDIT_TARGET_SURFACES)[number];
2226
+ declare const DEFAULT_POLICY_EDIT_HISTORY_LIMITS: Readonly<{
2227
+ generations: 4;
2228
+ candidatesPerGeneration: 16;
2229
+ scenariosPerCandidate: 12;
2230
+ findings: 32;
2231
+ authorContextChars: 200000;
2232
+ }>;
2233
+ interface PolicyEditObjective {
2234
+ /** Stable objective key cited by forecasts, for example `search.composite`. */
2235
+ key: string;
2236
+ /** Steering objectives are search-only; fresh-task results never enter author context. */
2237
+ split: 'search';
2238
+ /** Search promotes only larger composite scores. */
2239
+ direction: 'increase';
2240
+ scale: {
2241
+ min: number;
2242
+ max: number;
2243
+ };
2244
+ /** Forecast amounts are absolute deltas on the declared score scale. */
2245
+ unit: 'score';
2246
+ }
2247
+ type PolicyEditFindingSource = {
2248
+ kind: 'surface';
2249
+ surfaceHash: string;
2250
+ generation: number;
2251
+ } | {
2252
+ kind: 'global';
2253
+ label: string;
2254
+ };
2255
+ /** Trace-derived findings must name the measured profile that produced them.
2256
+ * Cross-run doctrine can be explicitly global; an unwrapped finding is rejected. */
2257
+ interface PolicyEditFindingInput {
2258
+ finding: AnalystFinding;
2259
+ source: PolicyEditFindingSource;
2260
+ }
2261
+ interface PolicyEditHistoryProjectionOptions {
2262
+ /** Number of most recent generations retained. Default: 4. */
2263
+ maxGenerations?: number;
2264
+ /** Number of candidates retained per generation. Default: 16. */
2265
+ maxCandidatesPerGeneration?: number;
2266
+ /** Scored tasks retained per candidate/outcome after deterministic extreme selection. */
2267
+ maxScenariosPerCandidate?: number;
2268
+ /** Optional pseudonymizer applied before scenario IDs enter author text. */
2269
+ scenarioIdTransform?: (scenarioId: string) => string;
2270
+ /** Objectives used to compute forecast residuals from measured composite deltas. */
2271
+ objectives?: readonly PolicyEditObjective[];
2272
+ }
2273
+ interface PolicyEditCandidateSummary {
2274
+ editId: string;
2275
+ axis: PolicyEdit['axis'];
2276
+ target: PolicyEdit['target'];
2277
+ change: PolicyEdit['change'];
2278
+ claim: string;
2279
+ expectedGain: PolicyEdit['expectedGain'];
2280
+ confidence: number;
2281
+ risk: PolicyEdit['risk'];
2282
+ sourceFindingIds: string[];
2283
+ rationale: string | null;
2284
+ validationPlan: string | null;
2285
+ }
2286
+ interface PolicyEditHistoryCandidateContext {
2287
+ surfaceHash: string;
2288
+ parentSurfaceHash: string | null;
2289
+ parentComposite: number | null;
2290
+ label: string | null;
2291
+ rationale: string | null;
2292
+ composite: number;
2293
+ observedDeltaFromParent: number | null;
2294
+ eligibleForPromotion: boolean | null;
2295
+ coverage: {
2296
+ expectedCells: number;
2297
+ scorableCells: number;
2298
+ unscorableCells: Array<{
2299
+ reason: string;
2300
+ }>;
2301
+ } | null;
2302
+ dimensions: Record<string, number>;
2303
+ scenarios: Array<{
2304
+ scenarioId: string;
2305
+ composite: number;
2306
+ notes: string | null;
2307
+ }>;
2308
+ candidateEdit: PolicyEditCandidateSummary | null;
2309
+ forecastCalibration: {
2310
+ objectiveKey: string;
2311
+ predictedDelta: number;
2312
+ observedDelta: number;
2313
+ residual: number;
2314
+ } | null;
2315
+ }
2316
+ interface PolicyEditOutcomeContext {
2317
+ split: 'search';
2318
+ generation: number;
2319
+ surfaceHash: string;
2320
+ composite: number;
2321
+ dimensions: Record<string, number>;
2322
+ scenarios: Array<{
2323
+ scenarioId: string;
2324
+ composite: number;
2325
+ notes: string | null;
2326
+ }>;
2327
+ coverage: {
2328
+ expectedCells: number;
2329
+ scorableCells: number;
2330
+ };
2331
+ }
2332
+ interface PolicyEditHistoryGenerationContext {
2333
+ generationIndex: number;
2334
+ promoted: string[];
2335
+ candidates: PolicyEditHistoryCandidateContext[];
2336
+ }
2337
+ interface LlmPolicyEditProposerOptions {
2338
+ llm: LlmClientOptions;
2339
+ model: string;
2340
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2341
+ costLedger?: CostLedger;
2342
+ /** Plain-language description of the JSON surface being improved. */
2343
+ target: string;
2344
+ /** PolicyEdit target surface every authored edit must retain. */
2345
+ targetSurface: JsonPolicyEditTargetSurface;
2346
+ /** Exact JSON paths the author may change. Prefix or fuzzy matches are not accepted. */
2347
+ allowedJsonPaths: readonly string[];
2348
+ /** Exact search objectives forecasts may name. Unknown keys or mismatched directions fail. */
2349
+ objectives: readonly PolicyEditObjective[];
2350
+ /** Default: evidence-only, so uncertain edits are measured rather than
2351
+ * suppressed by their own model-authored predictions. */
2352
+ admissionMode?: 'evidence-only' | 'strict';
2353
+ /** Readiness thresholds used only when admissionMode is explicitly strict. */
2354
+ admission?: PolicyEditAdmissionOptions;
2355
+ maxCandidates?: number;
2356
+ temperature?: number;
2357
+ maxTokens?: number;
2358
+ timeoutMs?: number;
2359
+ /** Number of most recent scored generations sent to the author. Default: 4. */
2360
+ maxHistoryGenerations?: number;
2361
+ /** Candidates retained per admitted generation. Default: 16. */
2362
+ maxHistoryCandidatesPerGeneration?: number;
2363
+ /** Scored tasks retained per candidate/outcome after deterministic extreme selection. */
2364
+ maxScenariosPerCandidate?: number;
2365
+ /** Evidence-bearing findings retained after deterministic severity/confidence ordering. */
2366
+ maxFindings?: number;
2367
+ /** Hard character limit over system + schema + serialized author context. */
2368
+ maxAuthorContextChars?: number;
2369
+ /** Optional one-to-one pseudonymizer applied to every author-visible evidence field. */
2370
+ scenarioIdTransform?: (scenarioId: string) => string;
2371
+ onAdmission?: (admission: PolicyEditAdmission) => void;
2372
+ }
2373
+ /**
2374
+ * LLM-backed PolicyEdit author. It reads only the current JSON surface,
2375
+ * evidence-bearing analyst findings, and scored generation history. Model output
2376
+ * is validated and rebound to exact finding evidence before the deterministic
2377
+ * policyEditProposer applies, admits, and deduplicates candidates.
2378
+ */
2379
+ declare function llmPolicyEditProposer(opts: LlmPolicyEditProposerOptions): SurfaceProposer<PolicyEditFindingInput>;
2380
+ /**
2381
+ * Projects scored history into the only fields a policy author may consume.
2382
+ * It admits recent generations and a bounded candidate count, while retaining
2383
+ * every dimension and scenario score for each admitted candidate.
2384
+ */
2385
+ declare function projectPolicyEditHistory(history: readonly GenerationRecord[], options?: PolicyEditHistoryProjectionOptions): PolicyEditHistoryGenerationContext[];
2386
+
2211
2387
  /**
2212
2388
  * `memoryCurationProposer` — a CURATOR `SurfaceProposer`, the complement to the
2213
2389
  * OPTIMIZER proposers (`gepaProposer` rewrites the prompt; this one BUILDS a
@@ -2236,6 +2412,8 @@ declare function haloProposer(opts: HaloProposerOptions): SurfaceProposer;
2236
2412
  */
2237
2413
 
2238
2414
  interface MemoryCurationProposerOptions {
2415
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2416
+ costLedger?: CostLedger;
2239
2417
  /** Top-K lessons retained in the surface memory block. Default 12. */
2240
2418
  maxEntries?: number;
2241
2419
  /** Heading rendered above the lessons inside the block. Default below. */
@@ -2248,6 +2426,7 @@ interface MemoryCurationProposerOptions {
2248
2426
  baseUrl: string;
2249
2427
  apiKey?: string;
2250
2428
  model: string;
2429
+ maxTokens?: number;
2251
2430
  fetchImpl?: LlmClientOptions['fetch'];
2252
2431
  };
2253
2432
  }
@@ -2281,6 +2460,33 @@ interface PolicyEditProposerOptions {
2281
2460
  */
2282
2461
  declare function policyEditProposer(opts?: PolicyEditProposerOptions): SurfaceProposer;
2283
2462
 
2463
+ /** One measured scenario row eligible for PolicyEdit author context. */
2464
+ interface PolicyEditAuthorScenarioRow {
2465
+ scenarioId: string;
2466
+ composite: number;
2467
+ }
2468
+ interface SelectPolicyEditAuthorRowsOptions {
2469
+ /** Maximum returned rows. Must be a positive safe integer. */
2470
+ limit: number;
2471
+ /** Optional score to compare against, keyed by scenario ID. */
2472
+ referenceByScenario?: ReadonlyMap<string, number>;
2473
+ }
2474
+ interface SerializedJsonBudget {
2475
+ json: string;
2476
+ actualChars: number;
2477
+ maxChars: number;
2478
+ }
2479
+ /**
2480
+ * Select a bounded, deterministic evidence slice for a PolicyEdit author.
2481
+ *
2482
+ * Rows are deduplicated by scenario ID, keeping the first measured row. The
2483
+ * result then interleaves three ranked views: hardest score, largest regression,
2484
+ * and largest improvement. A row selected by multiple views appears once.
2485
+ */
2486
+ declare function selectPolicyEditAuthorRows<T extends PolicyEditAuthorScenarioRow>(rows: readonly T[], options: SelectPolicyEditAuthorRowsOptions): T[];
2487
+ /** Serialize once and fail before dispatch when author context exceeds its budget. */
2488
+ declare function assertPolicyEditAuthorContextBudget(value: unknown, maxChars: number): SerializedJsonBudget;
2489
+
2284
2490
  /**
2285
2491
  * `traceAnalystProposer` — wraps agent-eval's OWN trace-analyst engine
2286
2492
  * (`AnalystRegistry` over the agentic OTLP reader) as a `SurfaceProposer`.
@@ -2309,6 +2515,11 @@ interface TraceAnalystProposerOptions {
2309
2515
  /** Model used to APPLY findings to the prompt surface. Default = `model`.
2310
2516
  * Keep this EQUAL to haloProposer's `applyModel` for an apples-to-apples run. */
2311
2517
  applyModel?: string;
2518
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2519
+ costLedger?: CostLedger;
2520
+ analysisMaximumCharge?: MaximumCharge;
2521
+ analysisReceipt?: (report: string) => CostReceiptInput;
2522
+ applyMaxTokens?: number;
2312
2523
  /** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
2313
2524
  * via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
2314
2525
  provider?: string;
@@ -2399,9 +2610,11 @@ declare function selectDiscriminative(signals: ScenarioSignal[], k: number, opts
2399
2610
  * the optimizers cannot drift on how a surface's score is computed.
2400
2611
  */
2401
2612
 
2402
- /** Mean composite across a campaign: per cell, the mean of its judges'
2403
- * composites; then the mean across cells. Cells with no judge scores are
2404
- * skipped. Empty 0. */
2613
+ /** Mean composite across a campaign: per cell, the mean of its finite,
2614
+ * successful judge composites; then the mean across cells. Invalid scores
2615
+ * remain visible on raw cells and coverage receipts but never poison the
2616
+ * descriptive aggregate with NaN. Cells with no valid scores are skipped.
2617
+ * Empty ⇒ 0. */
2405
2618
  declare function campaignMeanComposite<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): number;
2406
2619
  interface CampaignBreakdown {
2407
2620
  /** Mean score per judge dimension across all cells. */
@@ -2899,4 +3112,4 @@ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string):
2899
3112
  * identity against the checkout at `worktreeRef`. */
2900
3113
  declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
2901
3114
 
2902
- export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type AnalyzeCrossSurfaceInteractionsInput, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CodeSurfaceVerification, type CompareProposersOptions, type CompositeProposerOptions, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FileSearchLedger, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type NeutralizationGateOptions, type OpenSearchLedgerOptions, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, SEARCH_LEDGER_SCHEMA, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SingleRunLock, type SingleRunLockOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, acquireSingleRunLock, analyzeCrossSurfaceInteractions, applySkillPatch, assertCodeSurfaceIdentity, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, compareProposers, compositeProposer, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, isTransientTransportFailure, lineageNodeId, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, neutralizationGate, neutralizeText, openSearchLedger, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface };
3115
+ export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type AnalyzeCrossSurfaceInteractionsInput, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignRunPlan, CampaignStorage, CodeSurface, type CodeSurfaceVerification, type CompareProposersOptions, type CompositeProposerOptions, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, DEFAULT_POLICY_EDIT_HISTORY_LIMITS, type DimensionRegression, type DiscriminationScore, DispatchContext, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FileSearchLedger, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type Governor, type GovernorContext, type GovernorOp, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type HeuristicGovernorOptions, type JsonPolicyEditTargetSurface, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, Lineage, type LineageEdge, type LineageGraph, type LineageNode, type LineageNodeInput, type LineageStore, type LlmPolicyEditProposerOptions, type LoadEvalFixtureScenariosOptions, type MemoryCurationProposerOptions, MutableSurface, type NeutralizationGateOptions, type OpenSearchLedgerOptions, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditAuthorScenarioRow, type PolicyEditCandidateSummary, type PolicyEditFindingInput, type PolicyEditFindingSource, type PolicyEditHistoryCandidateContext, type PolicyEditHistoryGenerationContext, type PolicyEditHistoryProjectionOptions, type PolicyEditObjective, type PolicyEditOutcomeContext, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, RunCampaignOptions, RunImprovementLoopOptions, type RunLineageLoopOptions, type RunLineageLoopResult, type RunLineageLoopSeed, type RunLineageOptions, type RunLineageResult, type RunLineageSeed, type RunLineageStepResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, SEARCH_LEDGER_SCHEMA, Scenario, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, type SelectPolicyEditAuthorRowsOptions, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SerializedJsonBudget, type SingleRunLock, type SingleRunLockOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type SurfaceScore, type TraceAnalystProposerOptions, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, acquireSingleRunLock, analyzeCrossSurfaceInteractions, applySkillPatch, assertCodeSurfaceIdentity, assertPolicyEditAuthorContextBudget, buildAnalystSurfaceDispatch, callbackGovernor, campaignBreakdown, campaignMeanComposite, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, compareProposers, compositeProposer, detectScale, dimensionRegressions, discoverEvalFixtures, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, fsLineageStore, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, heuristicGovernor, isTransientTransportFailure, lineageNodeId, llmPolicyEditProposer, loadEvalFixture, loadEvalFixtureScenarios, makePlaybackDispatch, memLineageStore, memoryCurationProposer, neutralizationGate, neutralizeText, openSearchLedger, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, planEvalFixtureRun, policyEditProposer, projectPolicyEditHistory, renderScoreboardMarkdown, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runLineage, runLineageLoop, runProfileMatrix, runSkillOpt, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, selectPolicyEditAuthorRows, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystProposer, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface };
@@ -1,19 +1,18 @@
1
1
  import {
2
+ DEFAULT_POLICY_EDIT_HISTORY_LIMITS,
2
3
  FileSearchLedger,
3
4
  FsLabeledScenarioStore,
4
5
  LabeledScenarioStoreError,
5
6
  Lineage,
6
7
  ProfileMatrixError,
7
8
  SEARCH_LEDGER_SCHEMA,
8
- SearchLedgerConflictError,
9
- SearchLedgerError,
10
- SearchLedgerIntegrityError,
11
9
  SkillPatchParseError,
12
10
  WorktreeAdapterError,
13
11
  aceProposer,
14
12
  acquireSingleRunLock,
15
13
  analyzeCrossSurfaceInteractions,
16
14
  applySkillPatch,
15
+ assertPolicyEditAuthorContextBudget,
17
16
  buildAnalystSurfaceDispatch,
18
17
  callbackGovernor,
19
18
  classifyUngroundedLiterals,
@@ -32,7 +31,7 @@ import {
32
31
  heuristicGovernor,
33
32
  isTransientTransportFailure,
34
33
  lineageNodeId,
35
- llmJudge,
34
+ llmPolicyEditProposer,
36
35
  loadEvalFixture,
37
36
  loadEvalFixtureScenarios,
38
37
  makePlaybackDispatch,
@@ -46,6 +45,7 @@ import {
46
45
  patchEditCount,
47
46
  planEvalFixtureRun,
48
47
  policyEditProposer,
48
+ projectPolicyEditHistory,
49
49
  renderScoreboardMarkdown,
50
50
  resolveWorktreePath,
51
51
  rolloutArgumentDiff,
@@ -57,6 +57,7 @@ import {
57
57
  scoreUserStory,
58
58
  scoreboardSummary,
59
59
  selectDiscriminative,
60
+ selectPolicyEditAuthorRows,
60
61
  sequentialDecide,
61
62
  sequentialPairedGate,
62
63
  skillOptEntry,
@@ -65,7 +66,7 @@ import {
65
66
  userStoryScoreboard,
66
67
  validateSearchLedgerEvent,
67
68
  verifyCodeSurface
68
- } from "../chunk-KG4TD7EQ.js";
69
+ } from "../chunk-ZUXV7UWZ.js";
69
70
  import {
70
71
  assertCodeSurfaceIdentity,
71
72
  buildEvidenceVector,
@@ -75,6 +76,7 @@ import {
75
76
  codeSurfaceIdentityMaterial,
76
77
  composeGate,
77
78
  countSentenceEdits,
79
+ createReferenceEquivalenceJudge,
78
80
  defaultProductionGate,
79
81
  defaultRenderDiff,
80
82
  detectScale,
@@ -87,6 +89,7 @@ import {
87
89
  heldoutSignificance,
88
90
  isProposedCandidate,
89
91
  labelTrustRank,
92
+ llmJudge,
90
93
  loopProvenanceSpans,
91
94
  openAutoPr,
92
95
  pairHoldout,
@@ -100,35 +103,45 @@ import {
100
103
  runOptimization,
101
104
  surfaceContentHash,
102
105
  surfaceHash
103
- } from "../chunk-ADYLPOSX.js";
104
- import "../chunk-VI2UW6B6.js";
106
+ } from "../chunk-HQPHZGL6.js";
105
107
  import {
108
+ SearchLedgerConflictError,
109
+ SearchLedgerError,
110
+ SearchLedgerIntegrityError,
111
+ createRunCostLedger,
106
112
  fsCampaignStorage,
107
113
  inMemoryCampaignStorage,
108
114
  planCampaignRun,
109
115
  resolveRunDir,
110
116
  runCampaign,
111
117
  tangleTracesRoot
112
- } from "../chunk-FAOEFFRT.js";
113
- import "../chunk-QMXXSNC4.js";
114
- import "../chunk-5S5NJ63F.js";
118
+ } from "../chunk-IDZTTFRR.js";
119
+ import "../chunk-3YYRZDON.js";
120
+ import {
121
+ POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
122
+ validatePolicyEditCandidateRecord
123
+ } from "../chunk-MGEHEHSN.js";
115
124
  import "../chunk-ARU2PZFM.js";
116
125
  import "../chunk-PJQFMIOX.js";
117
- import "../chunk-RPDDVKI7.js";
126
+ import "../chunk-4JLWXDYA.js";
118
127
  import "../chunk-GGE4NNQT.js";
119
- import "../chunk-LNQEP766.js";
128
+ import "../chunk-S2F4J57L.js";
120
129
  import "../chunk-5UF54T55.js";
121
130
  import "../chunk-XJYR7XFV.js";
122
131
  import "../chunk-VSMTAMNK.js";
123
- import "../chunk-GY4SYVPJ.js";
132
+ import "../chunk-NJC7U437.js";
133
+ import "../chunk-VCTY3W6J.js";
134
+ import "../chunk-VI2UW6B6.js";
124
135
  import "../chunk-PC4UYEBM.js";
125
136
  import "../chunk-ONWEPEDO.js";
126
137
  import "../chunk-PZ5AY32C.js";
127
138
  export {
139
+ DEFAULT_POLICY_EDIT_HISTORY_LIMITS,
128
140
  FileSearchLedger,
129
141
  FsLabeledScenarioStore,
130
142
  LabeledScenarioStoreError,
131
143
  Lineage,
144
+ POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
132
145
  ProfileMatrixError,
133
146
  SEARCH_LEDGER_SCHEMA,
134
147
  SearchLedgerConflictError,
@@ -141,6 +154,7 @@ export {
141
154
  analyzeCrossSurfaceInteractions,
142
155
  applySkillPatch,
143
156
  assertCodeSurfaceIdentity,
157
+ assertPolicyEditAuthorContextBudget,
144
158
  buildAnalystSurfaceDispatch,
145
159
  buildEvidenceVector,
146
160
  buildLoopProvenanceRecord,
@@ -153,6 +167,8 @@ export {
153
167
  composeGate,
154
168
  compositeProposer,
155
169
  countSentenceEdits,
170
+ createReferenceEquivalenceJudge,
171
+ createRunCostLedger,
156
172
  defaultProductionGate,
157
173
  defaultRenderDiff,
158
174
  detectScale,
@@ -181,6 +197,7 @@ export {
181
197
  labelTrustRank,
182
198
  lineageNodeId,
183
199
  llmJudge,
200
+ llmPolicyEditProposer,
184
201
  loadEvalFixture,
185
202
  loadEvalFixtureScenarios,
186
203
  loopProvenanceSpans,
@@ -201,6 +218,7 @@ export {
201
218
  planEvalFixtureRun,
202
219
  policyEditProposer,
203
220
  powerPreflight,
221
+ projectPolicyEditHistory,
204
222
  provenanceRecordPath,
205
223
  provenanceSpansPath,
206
224
  renderScoreboardMarkdown,
@@ -219,6 +237,7 @@ export {
219
237
  scoreUserStory,
220
238
  scoreboardSummary,
221
239
  selectDiscriminative,
240
+ selectPolicyEditAuthorRows,
222
241
  sequentialDecide,
223
242
  sequentialPairedGate,
224
243
  skillOptEntry,
@@ -228,6 +247,7 @@ export {
228
247
  tangleTracesRoot,
229
248
  traceAnalystProposer,
230
249
  userStoryScoreboard,
250
+ validatePolicyEditCandidateRecord,
231
251
  validateSearchLedgerEvent,
232
252
  verifyCodeSurface
233
253
  };
@@ -0,0 +1,45 @@
1
+ // src/concurrency.ts
2
+ var Mutex = class {
3
+ locked = false;
4
+ waiters = [];
5
+ async acquire() {
6
+ if (!this.locked) {
7
+ this.locked = true;
8
+ return () => this.release();
9
+ }
10
+ return new Promise((resolve) => {
11
+ this.waiters.push(() => {
12
+ resolve(() => this.release());
13
+ });
14
+ });
15
+ }
16
+ release() {
17
+ const next = this.waiters.shift();
18
+ if (next) {
19
+ next();
20
+ } else {
21
+ this.locked = false;
22
+ }
23
+ }
24
+ async runExclusive(fn) {
25
+ const release = await this.acquire();
26
+ try {
27
+ return await fn();
28
+ } finally {
29
+ release();
30
+ }
31
+ }
32
+ /** True iff someone holds the lock right now. Diagnostics only. */
33
+ get isLocked() {
34
+ return this.locked;
35
+ }
36
+ /** Pending waiter count. Diagnostics only. */
37
+ get pending() {
38
+ return this.waiters.length;
39
+ }
40
+ };
41
+
42
+ export {
43
+ Mutex
44
+ };
45
+ //# sourceMappingURL=chunk-3YYRZDON.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/concurrency.ts"],"sourcesContent":["/**\n * concurrency — small primitives the evolution loop needs.\n *\n * `Mutex` is a zero-dep async lock with FIFO fairness. The evolution loop\n * uses it to serialise checkout/build/commit sequences inside a single\n * pool slot, and to gate concurrent JSONL writers (see\n * `lockedJsonlReferenceReplayStore`).\n *\n * Deliberately minimal — no priority queue, no timeouts. If you need\n * those, swap to `async-mutex` at the call site.\n */\n\nexport class Mutex {\n private locked = false\n private readonly waiters: Array<() => void> = []\n\n async acquire(): Promise<() => void> {\n if (!this.locked) {\n this.locked = true\n return () => this.release()\n }\n return new Promise<() => void>((resolve) => {\n this.waiters.push(() => {\n resolve(() => this.release())\n })\n })\n }\n\n private release(): void {\n const next = this.waiters.shift()\n if (next) {\n next()\n } else {\n this.locked = false\n }\n }\n\n async runExclusive<T>(fn: () => Promise<T> | T): Promise<T> {\n const release = await this.acquire()\n try {\n return await fn()\n } finally {\n release()\n }\n }\n\n /** True iff someone holds the lock right now. Diagnostics only. */\n get isLocked(): boolean {\n return this.locked\n }\n\n /** Pending waiter count. Diagnostics only. */\n get pending(): number {\n return this.waiters.length\n }\n}\n"],"mappings":";AAYO,IAAM,QAAN,MAAY;AAAA,EACT,SAAS;AAAA,EACA,UAA6B,CAAC;AAAA,EAE/C,MAAM,UAA+B;AACnC,QAAI,CAAC,KAAK,QAAQ;AAChB,WAAK,SAAS;AACd,aAAO,MAAM,KAAK,QAAQ;AAAA,IAC5B;AACA,WAAO,IAAI,QAAoB,CAAC,YAAY;AAC1C,WAAK,QAAQ,KAAK,MAAM;AACtB,gBAAQ,MAAM,KAAK,QAAQ,CAAC;AAAA,MAC9B,CAAC;AAAA,IACH,CAAC;AAAA,EACH;AAAA,EAEQ,UAAgB;AACtB,UAAM,OAAO,KAAK,QAAQ,MAAM;AAChC,QAAI,MAAM;AACR,WAAK;AAAA,IACP,OAAO;AACL,WAAK,SAAS;AAAA,IAChB;AAAA,EACF;AAAA,EAEA,MAAM,aAAgB,IAAsC;AAC1D,UAAM,UAAU,MAAM,KAAK,QAAQ;AACnC,QAAI;AACF,aAAO,MAAM,GAAG;AAAA,IAClB,UAAE;AACA,cAAQ;AAAA,IACV;AAAA,EACF;AAAA;AAAA,EAGA,IAAI,WAAoB;AACtB,WAAO,KAAK;AAAA,EACd;AAAA;AAAA,EAGA,IAAI,UAAkB;AACpB,WAAO,KAAK,QAAQ;AAAA,EACtB;AACF;","names":[]}
@@ -2,7 +2,7 @@ import {
2
2
  OtlpFileTraceStore,
3
3
  TraceFileMissingError,
4
4
  buildTraceAnalystTools
5
- } from "./chunk-LNQEP766.js";
5
+ } from "./chunk-S2F4J57L.js";
6
6
 
7
7
  // src/trace-analyst/prompts.ts
8
8
  var TRACE_ANALYST_ACTOR_DESCRIPTION = `You answer questions about an OTLP-shaped JSONL trace dataset using the trace tools provided in the \`traces\` namespace.
@@ -198,4 +198,4 @@ export {
198
198
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
199
199
  analyzeTraces
200
200
  };
201
- //# sourceMappingURL=chunk-RPDDVKI7.js.map
201
+ //# sourceMappingURL=chunk-4JLWXDYA.js.map