@tangle-network/agent-eval 0.115.3 → 0.117.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/dist/analyst/index.d.ts +16 -11
  3. package/dist/analyst/index.js +33 -25
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  6. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/belief-state/index.js +1 -1
  9. package/dist/benchmarks/index.d.ts +12 -5
  10. package/dist/benchmarks/index.js +11 -10
  11. package/dist/builder-eval/index.d.ts +4 -4
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +247 -34
  15. package/dist/campaign/index.js +33 -13
  16. package/dist/chunk-3YYRZDON.js +45 -0
  17. package/dist/chunk-3YYRZDON.js.map +1 -0
  18. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  19. package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
  20. package/dist/chunk-CCZIVI3F.js.map +1 -0
  21. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  22. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  23. package/dist/chunk-HHWE3POT.js +94 -0
  24. package/dist/chunk-HHWE3POT.js.map +1 -0
  25. package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
  26. package/dist/chunk-HQPHZGL6.js.map +1 -0
  27. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  28. package/dist/chunk-IDZTTFRR.js.map +1 -0
  29. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  30. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  31. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  32. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  33. package/dist/chunk-LTVG32KX.js.map +1 -0
  34. package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
  35. package/dist/chunk-MGEHEHSN.js.map +1 -0
  36. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  37. package/dist/chunk-NJC7U437.js.map +1 -0
  38. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  39. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  40. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  41. package/dist/chunk-S2F4J57L.js.map +1 -0
  42. package/dist/chunk-VCTY3W6J.js +798 -0
  43. package/dist/chunk-VCTY3W6J.js.map +1 -0
  44. package/dist/chunk-VF3XSYTI.js +545 -0
  45. package/dist/chunk-VF3XSYTI.js.map +1 -0
  46. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  47. package/dist/chunk-YZPO4UHR.js.map +1 -0
  48. package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
  49. package/dist/chunk-ZUXV7UWZ.js.map +1 -0
  50. package/dist/cli.js +4 -2
  51. package/dist/cli.js.map +1 -1
  52. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  53. package/dist/contract/index.d.ts +45 -31
  54. package/dist/contract/index.js +58 -19
  55. package/dist/contract/index.js.map +1 -1
  56. package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
  57. package/dist/control.d.ts +6 -6
  58. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  59. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
  60. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  61. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  62. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  63. package/dist/fuzz.d.ts +8 -16
  64. package/dist/fuzz.js +72 -42
  65. package/dist/fuzz.js.map +1 -1
  66. package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
  67. package/dist/hosted/index.d.ts +14 -7
  68. package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
  69. package/dist/index.d.ts +97 -55
  70. package/dist/index.js +343 -244
  71. package/dist/index.js.map +1 -1
  72. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  73. package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
  74. package/dist/kind-factory-ClZmO25A.d.ts +171 -0
  75. package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
  76. package/dist/meta-eval/index.d.ts +8 -7
  77. package/dist/meta-eval/index.js +1 -1
  78. package/dist/multishot/index.d.ts +10 -3
  79. package/dist/openapi.json +1 -1
  80. package/dist/pipelines/index.d.ts +16 -6
  81. package/dist/pipelines/index.js +119 -23
  82. package/dist/pipelines/index.js.map +1 -1
  83. package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
  84. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
  85. package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
  86. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  87. package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  88. package/dist/reporting.d.ts +10 -9
  89. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
  90. package/dist/rl.d.ts +17 -12
  91. package/dist/rl.js +2 -2
  92. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  93. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  94. package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
  95. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  96. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  97. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
  98. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  99. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  100. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  101. package/dist/storyboard/index.d.ts +1 -1
  102. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  103. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  104. package/dist/traces.d.ts +19 -10
  105. package/dist/traces.js +16 -4
  106. package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
  107. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  108. package/dist/wire/index.d.ts +28 -19
  109. package/dist/wire/index.js +4 -2
  110. package/docs/design/loop-taxonomy.md +1 -2
  111. package/docs/distributed-driver.md +1 -1
  112. package/package.json +3 -3
  113. package/dist/chunk-4D5RVB3W.js.map +0 -1
  114. package/dist/chunk-5S5NJ63F.js.map +0 -1
  115. package/dist/chunk-ADYLPOSX.js.map +0 -1
  116. package/dist/chunk-FAOEFFRT.js.map +0 -1
  117. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  118. package/dist/chunk-I6LVHOV3.js +0 -205
  119. package/dist/chunk-I6LVHOV3.js.map +0 -1
  120. package/dist/chunk-KG4TD7EQ.js.map +0 -1
  121. package/dist/chunk-LNQEP766.js.map +0 -1
  122. package/dist/chunk-MHNQWM4I.js.map +0 -1
  123. package/dist/chunk-NYFUT3B3.js.map +0 -1
  124. package/dist/chunk-QMXXSNC4.js +0 -761
  125. package/dist/chunk-QMXXSNC4.js.map +0 -1
  126. package/dist/chunk-TLDB7WRY.js.map +0 -1
  127. package/dist/chunk-WSBUZMBU.js.map +0 -1
  128. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  129. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  130. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  131. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  132. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  133. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  134. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
@@ -1,57 +1,68 @@
1
1
  import {
2
+ JudgeParseError,
2
3
  assertCodeSurfaceIdentity,
3
4
  campaignBreakdown,
4
5
  campaignMeanComposite,
6
+ costReceiptFromTCloud,
5
7
  defaultProductionGate,
6
8
  gepaProposer,
7
9
  isProposedCandidate,
8
10
  labelTrustRank,
11
+ maximumChargeForTCloudRequest,
9
12
  pairHoldout,
10
13
  recoverTruncatedJson,
11
14
  renderAnalystEvidence,
12
15
  runImprovementLoop,
13
16
  surfaceContentHash,
14
17
  surfaceHash
15
- } from "./chunk-ADYLPOSX.js";
16
- import {
17
- estimateCost,
18
- isModelPriced
19
- } from "./chunk-VI2UW6B6.js";
18
+ } from "./chunk-HQPHZGL6.js";
20
19
  import {
20
+ SearchLedgerConflictError,
21
+ SearchLedgerError,
22
+ SearchLedgerIntegrityError,
23
+ appendSearchLedgerLine,
21
24
  assertRealBackend,
22
25
  contentHash,
26
+ createRunCostLedger,
27
+ fsCampaignStorage,
23
28
  planCampaignRun,
29
+ resolveRunDir,
24
30
  runCampaign,
25
- summarizeBackendIntegrity
26
- } from "./chunk-FAOEFFRT.js";
31
+ summarizeBackendIntegrity,
32
+ withSearchLedgerFileLock
33
+ } from "./chunk-IDZTTFRR.js";
27
34
  import {
28
- Mutex,
29
- admitPolicyEdit,
30
- applyPolicyEditToSurface,
31
- clamp01,
32
- isPolicyEdit,
33
- policyEditsFromFindings
34
- } from "./chunk-QMXXSNC4.js";
35
+ Mutex
36
+ } from "./chunk-3YYRZDON.js";
35
37
  import {
36
38
  AnalystRegistry,
37
39
  DEFAULT_TRACE_ANALYST_KINDS,
38
- createTraceAnalystKind
39
- } from "./chunk-5S5NJ63F.js";
40
+ POLICY_EDIT_AXES,
41
+ POLICY_EDIT_TARGET_SURFACES,
42
+ admitPolicyEdit,
43
+ applyPolicyEditToSurface,
44
+ assertNoJudgeVerdict,
45
+ createTraceAnalystKind,
46
+ isPolicyEdit,
47
+ makePolicyEdit,
48
+ makePolicyEditCandidateRecord,
49
+ policyEditsFromFindings,
50
+ validatePolicyEditCandidateRecord
51
+ } from "./chunk-MGEHEHSN.js";
40
52
  import {
41
53
  eProcess,
42
54
  mcnemar,
43
55
  mulberry32,
44
56
  pairedBootstrap,
45
57
  pairedRiskDifference,
46
- weightedComposite,
47
58
  wilcoxonSignedRank
48
59
  } from "./chunk-PJQFMIOX.js";
49
60
  import {
50
61
  analyzeTraces
51
- } from "./chunk-RPDDVKI7.js";
62
+ } from "./chunk-4JLWXDYA.js";
52
63
  import {
53
64
  OtlpFileTraceStore
54
- } from "./chunk-LNQEP766.js";
65
+ } from "./chunk-S2F4J57L.js";
55
66
  import {
56
67
  modelHasSnapshot,
57
68
  validateRunRecord
@@ -63,11 +74,18 @@ import {
63
74
  canonicalize
64
75
  } from "./chunk-VSMTAMNK.js";
65
76
  import {
66
- callLlm
67
- } from "./chunk-GY4SYVPJ.js";
77
+ callLlm,
78
+ callLlmJson,
79
+ costReceiptFromLlm,
80
+ costReceiptFromLlmError,
81
+ maximumChargeForLlmRequest
82
+ } from "./chunk-NJC7U437.js";
83
+ import {
84
+ CostAccountingIncompleteError,
85
+ CostLedger
86
+ } from "./chunk-VCTY3W6J.js";
68
87
  import {
69
88
  AgentEvalError,
70
- JudgeError,
71
89
  ValidationError
72
90
  } from "./chunk-ONWEPEDO.js";
73
91
 
@@ -489,339 +507,6 @@ async function runLineage(opts) {
489
507
  return { lineage, best: lineage.best(), steps };
490
508
  }
491
509
 
492
- // src/judges.ts
493
- var JudgeParseError = class extends JudgeError {
494
- /** Name of the judge whose response failed to parse. */
495
- judgeName;
496
- /** The raw (truncated) model response that failed to parse. */
497
- raw;
498
- constructor(judgeName, raw, options) {
499
- super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
500
- this.judgeName = judgeName;
501
- this.raw = raw;
502
- }
503
- };
504
- function createDomainExpertJudge(domain) {
505
- return async (tc, { scenario, turns }) => {
506
- const conversation = turns.map(
507
- (t, i) => `Turn ${i + 1}:
508
- User: ${t.userMessage}
509
- Agent: ${t.agentResponse.slice(0, 2e3)}`
510
- ).join("\n\n---\n\n");
511
- const resp = await tc.chat({
512
- model: "gpt-4o",
513
- messages: [
514
- {
515
- role: "system",
516
- content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
517
-
518
- Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
519
-
520
- Evaluate:
521
- 1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
522
- 2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
523
-
524
- Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
525
- },
526
- {
527
- role: "user",
528
- content: `Persona: ${scenario.persona} (${scenario.label})
529
- Scenario: ${scenario.thesis}
530
-
531
- ${conversation}`
532
- }
533
- ],
534
- temperature: 0.1,
535
- maxTokens: 800
536
- });
537
- return parseJudgeResponse("domain_expert", resp);
538
- };
539
- }
540
- var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
541
- const codeBlocks = artifacts.codeBlocks;
542
- if (codeBlocks.length === 0) {
543
- return [
544
- {
545
- judgeName: "code_execution",
546
- dimension: "code_execution",
547
- score: 0,
548
- reasoning: "No code blocks found in agent response."
549
- }
550
- ];
551
- }
552
- const codeText = codeBlocks.map(
553
- (b, i) => `Block ${i + 1} (${b.language}):
554
- \`\`\`${b.language}
555
- ${b.code.slice(0, 3e3)}
556
- \`\`\``
557
- ).join("\n\n");
558
- const resp = await tc.chat({
559
- model: "gpt-4o",
560
- messages: [
561
- {
562
- role: "system",
563
- content: `You are a principal software engineer reviewing code written by an AI agent.
564
-
565
- Score STRICTLY:
566
- 1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
567
- 2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
568
- 3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
569
-
570
- Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
571
- },
572
- {
573
- role: "user",
574
- content: `Task: ${scenario.thesis}
575
-
576
- ${codeText}`
577
- }
578
- ],
579
- temperature: 0.1,
580
- maxTokens: 1e3
581
- });
582
- return parseJudgeResponse("code_execution", resp);
583
- };
584
- var coherenceJudge = async (tc, { scenario, turns }) => {
585
- if (turns.length < 2) {
586
- return [];
587
- }
588
- const conversation = turns.map(
589
- (t, i) => `Turn ${i + 1}:
590
- User: ${t.userMessage}
591
- Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
592
- ).join("\n\n---\n\n");
593
- const resp = await tc.chat({
594
- model: "gpt-4o",
595
- messages: [
596
- {
597
- role: "system",
598
- content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
599
-
600
- Score STRICTLY:
601
- 1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
602
- 2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
603
- 3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
604
-
605
- Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
606
- },
607
- {
608
- role: "user",
609
- content: `Scenario: ${scenario.thesis}
610
-
611
- ${conversation}`
612
- }
613
- ],
614
- temperature: 0.1,
615
- maxTokens: 800
616
- });
617
- return parseJudgeResponse("coherence", resp);
618
- };
619
- var adversarialJudge = async (tc, { scenario, turns }) => {
620
- const conversation = turns.map(
621
- (t, i) => `Turn ${i + 1}:
622
- User: ${t.userMessage}
623
- Agent: ${t.agentResponse.slice(0, 1500)}`
624
- ).join("\n\n---\n\n");
625
- const resp = await tc.chat({
626
- model: "gpt-4o",
627
- messages: [
628
- {
629
- role: "system",
630
- content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
631
-
632
- 1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
633
- 2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
634
- 3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
635
-
636
- Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
637
-
638
- Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
639
- },
640
- {
641
- role: "user",
642
- content: `Persona: ${scenario.persona}
643
- Scenario: ${scenario.thesis}
644
-
645
- ${conversation}`
646
- }
647
- ],
648
- temperature: 0.2,
649
- maxTokens: 800
650
- });
651
- return parseJudgeResponse("adversarial", resp);
652
- };
653
- function createCustomJudge(name, systemPrompt, opts) {
654
- return async (tc, { scenario, turns }) => {
655
- const conversation = turns.map(
656
- (t, i) => `Turn ${i + 1}:
657
- User: ${t.userMessage}
658
- Agent: ${t.agentResponse.slice(0, 2e3)}`
659
- ).join("\n\n---\n\n");
660
- const resp = await tc.chat({
661
- model: opts?.model ?? "gpt-4o",
662
- messages: [
663
- {
664
- role: "system",
665
- content: systemPrompt
666
- },
667
- {
668
- role: "user",
669
- content: `Persona: ${scenario.persona} (${scenario.label})
670
- Scenario: ${scenario.thesis}
671
-
672
- ${conversation}`
673
- }
674
- ],
675
- temperature: opts?.temperature ?? 0.1,
676
- maxTokens: opts?.maxTokens ?? 1e3
677
- });
678
- return parseJudgeResponse(name, resp);
679
- };
680
- }
681
- function defaultJudges(domain) {
682
- return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
683
- }
684
- function parseJudgeResponse(judgeName, resp) {
685
- const content = resp.choices?.[0]?.message?.content ?? "";
686
- try {
687
- let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
688
- const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
689
- if (arrayMatch) cleaned = arrayMatch[0];
690
- const parsed = JSON.parse(cleaned);
691
- return parsed.map((p) => ({
692
- judgeName,
693
- dimension: p.dimension,
694
- score: Math.max(0, Math.min(10, p.score)),
695
- reasoning: p.reasoning ?? "",
696
- evidence: p.evidence
697
- }));
698
- } catch (err) {
699
- throw new JudgeParseError(judgeName, content, { cause: err });
700
- }
701
- }
702
-
703
- // src/llm-judge.ts
704
- function llmJudge(name, prompt, opts) {
705
- if (!name.trim()) {
706
- throw new Error("llmJudge: name must be non-empty");
707
- }
708
- if (!prompt.trim()) {
709
- throw new Error(`llmJudge '${name}': prompt must be non-empty`);
710
- }
711
- const model = opts.model ?? opts.chat.defaultModel;
712
- if (!model) {
713
- throw new Error(
714
- `llmJudge '${name}': no model on opts and no defaultModel on the ChatClient \u2014 pass opts.model or bind defaultModel at createChatClient().`
715
- );
716
- }
717
- const dimensions = normalizeDimensions(opts.dimensions, name);
718
- const scale = opts.scale ?? "unit";
719
- const divisor = scale === "ten" ? 10 : 1;
720
- const renderUser = opts.renderUser ?? ((input) => JSON.stringify({ scenario: input.scenario, artifact: input.artifact }, null, 2));
721
- if (opts.weights) {
722
- for (const key of Object.keys(opts.weights)) {
723
- if (!dimensions.some((d) => d.key === key)) {
724
- throw new Error(
725
- `llmJudge '${name}': weights names dimension '${key}' that is not declared in dimensions`
726
- );
727
- }
728
- }
729
- }
730
- const systemPrompt = `${prompt}
731
-
732
- ${renderContract(dimensions, scale)}`;
733
- return {
734
- name,
735
- dimensions,
736
- appliesTo: opts.appliesTo,
737
- async score({ artifact, scenario, signal }) {
738
- const response = await opts.chat.chat(
739
- {
740
- model,
741
- messages: [
742
- { role: "system", content: systemPrompt },
743
- { role: "user", content: renderUser({ artifact, scenario }) }
744
- ],
745
- jsonMode: true,
746
- temperature: opts.temperature ?? 0.1,
747
- maxTokens: opts.maxTokens ?? 800
748
- },
749
- { signal }
750
- );
751
- const parsed = parseResponse(name, response.content);
752
- const rawDims = parsed.dimensions ?? parsed.scores;
753
- if (!rawDims || typeof rawDims !== "object") {
754
- throw new JudgeParseError(name, response.content, {
755
- cause: new Error("response has no `dimensions` object")
756
- });
757
- }
758
- const dims = {};
759
- for (const { key } of dimensions) {
760
- const raw = rawDims[key];
761
- const value = Number(raw);
762
- if (raw === void 0 || raw === null || !Number.isFinite(value)) {
763
- throw new JudgeParseError(name, response.content, {
764
- cause: new Error(
765
- `dimension '${key}' missing or non-numeric (got ${JSON.stringify(raw)})`
766
- )
767
- });
768
- }
769
- dims[key] = clamp01(value / divisor);
770
- }
771
- const weights = opts.weights ?? Object.fromEntries(dimensions.map((d) => [d.key, 1 / dimensions.length]));
772
- const { composite } = weightedComposite({ dims, weights });
773
- const notes = firstString(parsed.notes) ?? firstString(parsed.rationale) ?? `${name}: composite ${composite.toFixed(3)} over ${dimensions.length} dimension(s)`;
774
- return { dimensions: dims, composite, notes };
775
- }
776
- };
777
- }
778
- function normalizeDimensions(input, name) {
779
- const raw = input && input.length > 0 ? input : ["quality"];
780
- const out = [];
781
- const seen = /* @__PURE__ */ new Set();
782
- for (const d of raw) {
783
- const dim = typeof d === "string" ? { key: d, description: d } : d;
784
- if (!dim.key.trim()) {
785
- throw new Error(`llmJudge '${name}': dimension key must be non-empty`);
786
- }
787
- if (seen.has(dim.key)) {
788
- throw new Error(`llmJudge '${name}': duplicate dimension key '${dim.key}'`);
789
- }
790
- seen.add(dim.key);
791
- out.push(dim);
792
- }
793
- return out;
794
- }
795
- function renderContract(dimensions, scale) {
796
- const range = scale === "ten" ? "0 to 10" : "0.0 to 1.0";
797
- const lines = dimensions.map((d) => ` - "${d.key}": ${d.description} (score ${range})`);
798
- const example = `{"dimensions": {${dimensions.map((d) => `"${d.key}": <number>`).join(", ")}}, "notes": "<one-line rationale>"}`;
799
- return [
800
- "Score the artifact on EACH of these dimensions:",
801
- ...lines,
802
- "",
803
- `Respond with JSON ONLY, no prose. Every dimension is a number in [${range}]:`,
804
- example
805
- ].join("\n");
806
- }
807
- function parseResponse(name, content) {
808
- const stripped = content.replace(/```json\n?|\n?```/g, "").trim();
809
- const objMatch = stripped.match(/\{[\s\S]*\}/);
810
- const payload = objMatch ? objMatch[0] : stripped;
811
- try {
812
- const parsed = JSON.parse(payload);
813
- if (typeof parsed !== "object" || parsed === null) {
814
- throw new Error("parsed value is not an object");
815
- }
816
- return parsed;
817
- } catch (err) {
818
- throw new JudgeParseError(name, content, { cause: err });
819
- }
820
- }
821
- function firstString(value) {
822
- return typeof value === "string" && value.trim() ? value : void 0;
823
- }
824
-
825
510
  // src/campaign/analyst-surface.ts
826
511
  function surfaceToText(surface) {
827
512
  if (typeof surface === "string") return surface;
@@ -3411,6 +3096,7 @@ var SKILLOPT_SYSTEM = 'You are a SkillOpt optimizer. You improve ONE skill docum
3411
3096
  function skillOptProposer(opts) {
3412
3097
  const evidenceK = opts.evidenceK ?? 3;
3413
3098
  const defaultBudget = opts.editBudget ?? 3;
3099
+ const directCostLedger = opts.costLedger ?? new CostLedger();
3414
3100
  async function proposePatches(args) {
3415
3101
  const userPrompt = buildPatchPrompt({
3416
3102
  target: opts.target,
@@ -3422,19 +3108,29 @@ function skillOptProposer(opts) {
3422
3108
  findingsNote: args.findingsNote,
3423
3109
  count: args.count
3424
3110
  });
3425
- const result = await callLlm(
3426
- {
3427
- model: opts.model,
3428
- messages: [
3429
- { role: "system", content: SKILLOPT_SYSTEM },
3430
- { role: "user", content: userPrompt }
3431
- ],
3432
- jsonMode: true,
3433
- temperature: opts.temperature ?? 0.6,
3434
- maxTokens: opts.maxTokens ?? 4e3
3435
- },
3436
- opts.llm
3437
- );
3111
+ const request = {
3112
+ model: opts.model,
3113
+ messages: [
3114
+ { role: "system", content: SKILLOPT_SYSTEM },
3115
+ { role: "user", content: userPrompt }
3116
+ ],
3117
+ jsonMode: true,
3118
+ temperature: opts.temperature ?? 0.6,
3119
+ maxTokens: opts.maxTokens ?? 4e3
3120
+ };
3121
+ const paid = await (args.costLedger ?? directCostLedger).runPaidCall({
3122
+ channel: "driver",
3123
+ phase: args.costPhase ?? "search.proposal",
3124
+ actor: "skill-opt.propose",
3125
+ model: opts.model,
3126
+ maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
3127
+ signal: args.signal,
3128
+ execute: (signal, callId) => callLlm(request, { ...opts.llm, signal, idempotencyKey: callId }),
3129
+ receipt: costReceiptFromLlm,
3130
+ receiptFromError: costReceiptFromLlmError
3131
+ });
3132
+ if (!paid.succeeded) throw paid.error;
3133
+ const result = paid.value;
3438
3134
  return parseSkillPatchResponse(result.content, args.count, args.editBudget);
3439
3135
  }
3440
3136
  return {
@@ -3454,7 +3150,9 @@ function skillOptProposer(opts) {
3454
3150
  rejectedBuffer: [],
3455
3151
  findingsNote: renderAnalystEvidence(ctx.findings, ctx.report) ?? void 0,
3456
3152
  count: ctx.populationSize,
3457
- signal: ctx.signal
3153
+ signal: ctx.signal,
3154
+ costLedger: ctx.costLedger ?? directCostLedger,
3155
+ costPhase: ctx.costPhase ?? "search.proposal"
3458
3156
  });
3459
3157
  const out = [];
3460
3158
  const seen = /* @__PURE__ */ new Set();
@@ -3618,16 +3316,20 @@ async function runSkillOpt(opts) {
3618
3316
  const budgetAnneal = opts.budgetAnneal ?? true;
3619
3317
  const rejectedBufferSize = opts.rejectedBufferSize ?? 12;
3620
3318
  const slowMetaEvery = opts.slowMetaEvery ?? 2;
3621
- let totalCostUsd = 0;
3319
+ opts.runDir = resolveRunDir(opts.runDir, opts.repo);
3320
+ const storage = opts.storage ?? fsCampaignStorage();
3321
+ const costLedger = opts.costLedger ?? createRunCostLedger({
3322
+ storage,
3323
+ runDir: opts.runDir,
3324
+ costCeilingUsd: opts.costCeiling
3325
+ });
3622
3326
  const scoreHoldout = async (surface, tag) => {
3623
- const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag);
3624
- totalCostUsd += campaign.aggregates.totalCostUsd;
3327
+ const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag, costLedger);
3625
3328
  return campaignMeanComposite(campaign);
3626
3329
  };
3627
3330
  const evidenceK = opts.evidenceK ?? 3;
3628
3331
  const trainEvidence = async (surface, tag) => {
3629
- const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag);
3630
- totalCostUsd += campaign.aggregates.totalCostUsd;
3332
+ const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag, costLedger);
3631
3333
  return toEvidence(campaign, evidenceK);
3632
3334
  };
3633
3335
  let current = opts.baselineSurface;
@@ -3651,7 +3353,9 @@ async function runSkillOpt(opts) {
3651
3353
  rejectedBuffer: buffer,
3652
3354
  metaNote,
3653
3355
  count: patchesPerEpoch,
3654
- signal: opts.signal ?? new AbortController().signal
3356
+ signal: opts.signal ?? new AbortController().signal,
3357
+ costLedger,
3358
+ costPhase: "skill-opt.proposal"
3655
3359
  });
3656
3360
  let accepted = null;
3657
3361
  const rejectedThisEpoch = [];
@@ -3710,6 +3414,7 @@ async function runSkillOpt(opts) {
3710
3414
  });
3711
3415
  if (sinceAccept >= patience) break;
3712
3416
  }
3417
+ const cost = costLedger.summary();
3713
3418
  return {
3714
3419
  winnerSurface: current,
3715
3420
  baselineHoldoutComposite: baselineHoldout,
@@ -3719,12 +3424,14 @@ async function runSkillOpt(opts) {
3719
3424
  rejectedEdits: rejectedAll,
3720
3425
  epochsRun,
3721
3426
  history,
3722
- totalCostUsd
3427
+ totalCostUsd: cost.totalCostUsd,
3428
+ cost
3723
3429
  };
3724
3430
  }
3725
- function runScoringCampaign(opts, scenarios, surface, tag) {
3431
+ function runScoringCampaign(opts, scenarios, surface, tag, costLedger) {
3726
3432
  return runCampaign({
3727
3433
  ...opts,
3434
+ costLedger,
3728
3435
  scenarios,
3729
3436
  dispatch: (scenario, ctx) => opts.dispatchWithSurface(surface, scenario, ctx),
3730
3437
  runDir: `${opts.runDir}/${tag}`
@@ -3896,11 +3603,11 @@ function gepaEntry(config, combineParents, name) {
3896
3603
  ...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
3897
3604
  ...config.report !== void 0 ? { report: config.report } : {}
3898
3605
  });
3899
- const costUsd = result.baselineCampaign.aggregates.totalCostUsd + result.generations.reduce(
3900
- (sum, g) => sum + g.surfaces.reduce((s, sf) => s + sf.campaign.aggregates.totalCostUsd, 0),
3901
- 0
3902
- );
3903
- return { winnerSurface: result.winnerSurface, costUsd, durationMs: Date.now() - started };
3606
+ return {
3607
+ winnerSurface: result.winnerSurface,
3608
+ costUsd: result.cost.totalCostUsd,
3609
+ durationMs: Date.now() - started
3610
+ };
3904
3611
  }
3905
3612
  };
3906
3613
  }
@@ -3973,11 +3680,11 @@ function fapoEscalationEntry(config, name = "fapo-escalation") {
3973
3680
  ...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
3974
3681
  ...config.report !== void 0 ? { report: config.report } : {}
3975
3682
  });
3976
- const costUsd = result.baselineCampaign.aggregates.totalCostUsd + result.generations.reduce(
3977
- (sum, g) => sum + g.surfaces.reduce((s, sf) => s + sf.campaign.aggregates.totalCostUsd, 0),
3978
- 0
3979
- );
3980
- return { winnerSurface: result.winnerSurface, costUsd, durationMs: Date.now() - started };
3683
+ return {
3684
+ winnerSurface: result.winnerSurface,
3685
+ costUsd: result.cost.totalCostUsd,
3686
+ durationMs: Date.now() - started
3687
+ };
3981
3688
  }
3982
3689
  };
3983
3690
  }
@@ -4211,6 +3918,7 @@ function createLlmCorrectnessChecker(tc, opts = {}) {
4211
3918
  const model = opts.model ?? "claude-sonnet-4-6";
4212
3919
  const maxContentChars = opts.maxContentChars ?? 8e3;
4213
3920
  const maxAttempts = opts.maxAttempts ?? 2;
3921
+ const costLedger = opts.costLedger ?? new CostLedger();
4214
3922
  const sink = opts.rawSink;
4215
3923
  const record = async (event) => {
4216
3924
  try {
@@ -4254,7 +3962,24 @@ ${content.slice(0, maxContentChars)}`
4254
3962
  redactedFields: []
4255
3963
  });
4256
3964
  try {
4257
- const resp = await tc.chat(request);
3965
+ const paid = await costLedger.runPaidCall({
3966
+ channel: "verifier",
3967
+ phase: opts.costPhase ?? "completion.correctness",
3968
+ actor: "correctness-checker",
3969
+ model,
3970
+ maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
3971
+ tags: {
3972
+ ...opts.costTags,
3973
+ requirementId: requirement.reqId,
3974
+ attempt: String(attempt)
3975
+ },
3976
+ signal: opts.signal,
3977
+ execute: () => tc.chat(request),
3978
+ receipt: (response) => costReceiptFromTCloud(response, model),
3979
+ receiptFromError: (error) => opts.receiptFromError?.(error, attempt)
3980
+ });
3981
+ if (!paid.succeeded) throw paid.error;
3982
+ const resp = paid.value;
4258
3983
  const raw = resp.choices?.[0]?.message?.content ?? "";
4259
3984
  await record({
4260
3985
  eventId: randomUUID(),
@@ -4770,12 +4495,12 @@ function requireResolvedModel(cell, profileId) {
4770
4495
  const resolved = cell.resolvedModel?.trim();
4771
4496
  if (!resolved) {
4772
4497
  throw new ProfileMatrixError(
4773
- `profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 report it via ctx.cost.observeModel(<id>) so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
4498
+ `profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 return it in the ctx.cost.runPaidCall receipt so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
4774
4499
  );
4775
4500
  }
4776
4501
  if (!modelHasSnapshot(resolved)) {
4777
4502
  throw new ProfileMatrixError(
4778
- `profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) before reporting it via ctx.cost.observeModel`
4503
+ `profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) in the paid-call receipt`
4779
4504
  );
4780
4505
  }
4781
4506
  return resolved;
@@ -4801,14 +4526,9 @@ function buildRunRecord(args) {
4801
4526
  }
4802
4527
  const perDimMean = {};
4803
4528
  for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean3(values);
4804
- let costUsd = cell.costUsd;
4805
- let costEstimated = false;
4806
- if (costUsd === 0 && cell.tokenUsage.output > 0 && isModelPriced(model)) {
4807
- costUsd = estimateCost(cell.tokenUsage.input, cell.tokenUsage.output, model);
4808
- costEstimated = costUsd > 0;
4809
- }
4529
+ const costUsd = cell.costUsd;
4810
4530
  raw.cost_usd = costUsd;
4811
- raw.cost_estimated = costEstimated ? 1 : 0;
4531
+ raw.cost_estimated = cell.costEstimated ? 1 : 0;
4812
4532
  raw.tokens_input = cell.tokenUsage.input;
4813
4533
  raw.tokens_output = cell.tokenUsage.output;
4814
4534
  if (typeof cell.tokenUsage.cached === "number") raw.tokens_cached = cell.tokenUsage.cached;
@@ -4951,11 +4671,8 @@ async function runProfileMatrix(opts) {
4951
4671
  profileRecords.push(record);
4952
4672
  records.push(record);
4953
4673
  }
4954
- const pricedTotalCostUsd = profileRecords.reduce((a, r) => a + r.costUsd, 0);
4955
- campaigns[profileId] = {
4956
- ...campaign,
4957
- aggregates: { ...campaign.aggregates, totalCostUsd: pricedTotalCostUsd }
4958
- };
4674
+ const totalCostUsd = campaign.aggregates.totalCostUsd;
4675
+ campaigns[profileId] = campaign;
4959
4676
  byProfile[profileId] = {
4960
4677
  profileId,
4961
4678
  profileHash,
@@ -4965,7 +4682,7 @@ async function runProfileMatrix(opts) {
4965
4682
  model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
4966
4683
  records: profileRecords.length,
4967
4684
  meanComposite: mean3(profileRecords.map(compositeOf)),
4968
- totalCostUsd: pricedTotalCostUsd,
4685
+ totalCostUsd,
4969
4686
  integrity: summarizeBackendIntegrity(profileRecords)
4970
4687
  };
4971
4688
  }
@@ -5130,10 +4847,16 @@ function compositeProposer(opts) {
5130
4847
  const surface = isCandidate ? proposal.surface : proposal;
5131
4848
  const label = isCandidate ? proposal.label : "candidate";
5132
4849
  const rationale = isCandidate ? proposal.rationale : "";
4850
+ const candidateRecord = isCandidate ? proposal.candidateRecord : void 0;
5133
4851
  const key = surfaceContentHash(surface);
5134
4852
  if (seen.has(key)) continue;
5135
4853
  seen.add(key);
5136
- pool.push({ surface, label: `${member.kind}:${label}`, rationale });
4854
+ pool.push({
4855
+ surface,
4856
+ label: `${member.kind}:${label}`,
4857
+ rationale,
4858
+ ...candidateRecord ? { candidateRecord } : {}
4859
+ });
5137
4860
  }
5138
4861
  } catch (err) {
5139
4862
  errors.push(`${member.kind}: ${err instanceof Error ? err.message : String(err)}`);
@@ -5176,36 +4899,78 @@ function surfaceToPromptText(surface) {
5176
4899
  return typeof surface === "string" ? surface : JSON.stringify(surface);
5177
4900
  }
5178
4901
  function analysisEditProposer(opts) {
4902
+ const directCostLedger = opts.costLedger ?? new CostLedger();
5179
4903
  return {
5180
4904
  kind: opts.kind,
5181
4905
  async propose(ctx) {
5182
4906
  const parent = surfaceToPromptText(ctx.currentSurface);
4907
+ const costLedger = ctx.costLedger ?? directCostLedger;
4908
+ const phase = ctx.costPhase ?? "search.proposal";
5183
4909
  const traces = await opts.resolveTraces(ctx) ?? "";
5184
4910
  if (!traces.trim()) throw new Error(opts.noTracesError);
5185
4911
  const dir = mkdtempSync(join4(tmpdir(), `${opts.kind}-proposer-`));
5186
4912
  const tracePath = join4(dir, "traces.jsonl");
5187
4913
  writeFileSync2(tracePath, traces.endsWith("\n") ? traces : `${traces}
5188
4914
  `);
5189
- const report = await opts.analyze(tracePath, ctx);
5190
- const applied = await callLlm(
5191
- {
5192
- model: opts.applyModel,
5193
- messages: [
5194
- { role: "system", content: APPLY_SYSTEM },
5195
- {
5196
- role: "user",
5197
- content: `CURRENT PROMPT:
4915
+ if (costLedger.costCeilingUsd !== void 0 && !opts.analysisReceipt) {
4916
+ throw new CostAccountingIncompleteError(
4917
+ `${opts.kind}: capped analysis requires analysisReceipt before external execution`
4918
+ );
4919
+ }
4920
+ const analysis = await costLedger.runPaidCall({
4921
+ channel: "analyst",
4922
+ phase,
4923
+ actor: `${opts.kind}.analyze`,
4924
+ model: opts.analysisModel,
4925
+ maximumCharge: opts.analysisMaximumCharge,
4926
+ tags: { generation: String(ctx.generation) },
4927
+ signal: ctx.signal,
4928
+ execute: (signal) => opts.analyze(tracePath, { ...ctx, signal }),
4929
+ receipt: (report2) => opts.analysisReceipt?.(report2) ?? {
4930
+ model: opts.analysisModel,
4931
+ inputTokens: 0,
4932
+ outputTokens: 0,
4933
+ costUnknown: true
4934
+ }
4935
+ });
4936
+ if (!analysis.succeeded) throw analysis.error;
4937
+ const report = analysis.value;
4938
+ const request = {
4939
+ model: opts.applyModel,
4940
+ messages: [
4941
+ { role: "system", content: APPLY_SYSTEM },
4942
+ {
4943
+ role: "user",
4944
+ content: `CURRENT PROMPT:
5198
4945
  ${parent}
5199
4946
 
5200
4947
  TRACE-ANALYSIS REPORT:
5201
4948
  ${report}
5202
4949
 
5203
4950
  Return the full revised prompt.`
5204
- }
5205
- ]
5206
- },
5207
- { baseUrl: opts.baseUrl, apiKey: opts.apiKey, fetch: opts.fetchImpl }
5208
- );
4951
+ }
4952
+ ],
4953
+ maxTokens: opts.applyMaxTokens ?? 6e3
4954
+ };
4955
+ const llm = {
4956
+ baseUrl: opts.baseUrl,
4957
+ apiKey: opts.apiKey,
4958
+ fetch: opts.fetchImpl
4959
+ };
4960
+ const apply = await costLedger.runPaidCall({
4961
+ channel: "driver",
4962
+ phase,
4963
+ actor: `${opts.kind}.apply`,
4964
+ model: opts.applyModel,
4965
+ maximumCharge: maximumChargeForLlmRequest(request, llm),
4966
+ tags: { generation: String(ctx.generation) },
4967
+ signal: ctx.signal,
4968
+ execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
4969
+ receipt: costReceiptFromLlm,
4970
+ receiptFromError: costReceiptFromLlmError
4971
+ });
4972
+ if (!apply.succeeded) throw apply.error;
4973
+ const applied = apply.value;
5209
4974
  const text = applied.content.trim();
5210
4975
  if (!text || text === parent) return [];
5211
4976
  return [{ surface: text, label: opts.label, rationale: opts.rationale(report) }];
@@ -5224,7 +4989,12 @@ function haloProposer(opts) {
5224
4989
  label: "halo",
5225
4990
  baseUrl: opts.baseUrl,
5226
4991
  apiKey: opts.apiKey,
4992
+ analysisModel: model,
5227
4993
  applyModel: opts.applyModel ?? model,
4994
+ costLedger: opts.costLedger,
4995
+ analysisMaximumCharge: opts.analysisMaximumCharge,
4996
+ analysisReceipt: opts.analysisReceipt,
4997
+ applyMaxTokens: opts.applyMaxTokens,
5228
4998
  fetchImpl: opts.fetchImpl,
5229
4999
  resolveTraces: opts.resolveTraces,
5230
5000
  noTracesError: "haloProposer: resolveTraces returned no OTLP traces \u2014 the halo engine has nothing to analyze",
@@ -5264,85 +5034,8 @@ ${findings.slice(0, 800)}`,
5264
5034
  });
5265
5035
  }
5266
5036
 
5267
- // src/campaign/proposers/memory.ts
5268
- var BLOCK_START2 = "<!-- BEGIN curated-memory (auto-managed by memoryCurationProposer) -->";
5269
- var BLOCK_END2 = "<!-- END curated-memory -->";
5270
- var DEFAULT_HEADING2 = "## Learned from prior runs (curated memory)";
5271
- var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, generalizable agent guidance. Output ONLY a JSON array of strings, each one imperative lesson the agent should follow (e.g. "Always fetch a resource before mutating it"). No prose outside the JSON. Deduplicate; keep the most actionable and general; drop case-specific noise.';
5272
- function extractExistingLessons(text) {
5273
- return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
5274
- }
5275
- async function distillLessons(raw, distill) {
5276
- const res = await callLlm(
5277
- {
5278
- model: distill.model,
5279
- messages: [
5280
- { role: "system", content: DISTILL_SYSTEM },
5281
- { role: "user", content: `Findings:
5282
- ${raw.map((r) => `- ${r}`).join("\n")}` }
5283
- ]
5284
- },
5285
- { baseUrl: distill.baseUrl, apiKey: distill.apiKey, fetch: distill.fetchImpl }
5286
- );
5287
- try {
5288
- const parsed = JSON.parse(res.content.trim());
5289
- if (Array.isArray(parsed)) {
5290
- const lessons = parsed.filter(
5291
- (x) => typeof x === "string" && x.trim().length > 0
5292
- );
5293
- if (lessons.length > 0) return lessons;
5294
- }
5295
- } catch {
5296
- }
5297
- return raw;
5298
- }
5299
- function memoryCurationProposer(opts = {}) {
5300
- const maxEntries = opts.maxEntries ?? 12;
5301
- const heading = opts.sectionHeading ?? DEFAULT_HEADING2;
5302
- return {
5303
- kind: "memory-curation",
5304
- async propose(ctx) {
5305
- const parent = surfaceToText2(ctx.currentSurface);
5306
- const fresh = [];
5307
- for (const f of ctx.findings ?? []) {
5308
- const l = findingToLesson(f);
5309
- if (l) fresh.push(l);
5310
- }
5311
- const carried = extractExistingLessons(parent);
5312
- if (fresh.length === 0 && carried.length === 0) return [];
5313
- const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill) : fresh;
5314
- const byKey = /* @__PURE__ */ new Map();
5315
- for (const l of carried) {
5316
- const k = normKey(l);
5317
- if (k) byKey.set(k, { text: l, count: 1 });
5318
- }
5319
- for (const l of distilled) {
5320
- const k = normKey(l);
5321
- if (!k) continue;
5322
- const e = byKey.get(k);
5323
- if (e) e.count += 1;
5324
- else byKey.set(k, { text: l, count: 1 });
5325
- }
5326
- const ranked = [...byKey.values()].sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).slice(0, maxEntries);
5327
- if (ranked.length === 0) return [];
5328
- const block = [BLOCK_START2, heading, ...ranked.map((e) => `- ${e.text}`), BLOCK_END2].join(
5329
- "\n"
5330
- );
5331
- const next = `${stripBlock(parent, BLOCK_START2, BLOCK_END2)}
5332
-
5333
- ${block}
5334
- `;
5335
- if (next === parent) return [];
5336
- return [
5337
- {
5338
- surface: next,
5339
- label: "memory-curation",
5340
- rationale: `curated ${ranked.length} lessons (from ${fresh.length} new finding(s) + ${carried.length} carried)`
5341
- }
5342
- ];
5343
- }
5344
- };
5345
- }
5037
+ // src/campaign/proposers/llm-policy-edit.ts
5038
+ import { z } from "zod";
5346
5039
 
5347
5040
  // src/campaign/proposers/policy-edit.ts
5348
5041
  function policyEditProposer(opts = {}) {
@@ -5355,17 +5048,21 @@ function policyEditProposer(opts = {}) {
5355
5048
  Math.min(opts.maxCandidates ?? ctx.populationSize, ctx.populationSize)
5356
5049
  );
5357
5050
  const out = [];
5051
+ const seen = /* @__PURE__ */ new Set([surfaceContentHash(ctx.currentSurface)]);
5358
5052
  if (limit === 0) return out;
5359
5053
  for (const edit of edits) {
5360
5054
  const admission = admitPolicyEdit(edit, opts.admission);
5361
5055
  opts.onAdmission?.(admission);
5362
5056
  if (admission.decision !== "admit") continue;
5363
5057
  const surface = coerceCandidateSurface(applyPolicyEditToSurface(ctx.currentSurface, edit));
5364
- if (sameSurface(ctx.currentSurface, surface)) continue;
5058
+ const hash = surfaceContentHash(surface);
5059
+ if (seen.has(hash)) continue;
5060
+ seen.add(hash);
5365
5061
  out.push({
5366
5062
  surface,
5367
5063
  label: `policy-edit:${edit.axis}`,
5368
- rationale: `${edit.editId} expected ${edit.expectedGain.direction} ${edit.expectedGain.metric} by ${edit.expectedGain.amount}; source findings [${edit.source.findingIds.join(", ")}]`
5064
+ rationale: `${edit.editId} expected ${edit.expectedGain.direction} ${edit.expectedGain.metric} by ${edit.expectedGain.amount}; source findings [${edit.source.findingIds.join(", ")}]`,
5065
+ candidateRecord: makePolicyEditCandidateRecord(edit)
5369
5066
  });
5370
5067
  if (out.length >= limit) break;
5371
5068
  }
@@ -5403,22 +5100,1206 @@ function coerceCandidateSurface(surface) {
5403
5100
  }
5404
5101
  throw new Error("policyEditProposer: policy edit produced an unsupported surface");
5405
5102
  }
5406
- function sameSurface(a, b) {
5407
- return surfaceContentHash(a) === surfaceContentHash(b);
5408
- }
5409
5103
 
5410
- // src/campaign/proposers/trace-analyst.ts
5411
- import { ai } from "@ax-llm/ax";
5412
- function renderFindings(findings) {
5413
- return findings.map((f, i) => {
5414
- const action = f.recommended_action ? `
5415
- FIX: ${f.recommended_action}` : "";
5416
- const subject = f.subject ? ` (${f.subject})` : "";
5417
- return `${i + 1}. [${f.severity}/${f.area}]${subject} ${f.claim}${action}`;
5418
- }).join("\n");
5419
- }
5420
- function traceAnalystProposer(opts) {
5421
- if (!opts.apiKey) throw new Error("traceAnalystProposer: apiKey is required");
5104
+ // src/campaign/proposers/policy-edit-author-context.ts
5105
+ function selectPolicyEditAuthorRows(rows, options) {
5106
+ assertPositiveSafeInteger(options.limit, "limit");
5107
+ const unique2 = /* @__PURE__ */ new Map();
5108
+ for (const row of rows) {
5109
+ if (!row.scenarioId || row.scenarioId.trim() !== row.scenarioId) {
5110
+ throw new Error("selectPolicyEditAuthorRows: scenarioId must be trimmed and non-empty");
5111
+ }
5112
+ if (!Number.isFinite(row.composite)) {
5113
+ throw new Error(
5114
+ `selectPolicyEditAuthorRows: composite must be finite for '${row.scenarioId}'`
5115
+ );
5116
+ }
5117
+ if (unique2.has(row.scenarioId)) continue;
5118
+ const reference = options.referenceByScenario?.get(row.scenarioId);
5119
+ if (reference !== void 0 && !Number.isFinite(reference)) {
5120
+ throw new Error(
5121
+ `selectPolicyEditAuthorRows: reference must be finite for '${row.scenarioId}'`
5122
+ );
5123
+ }
5124
+ unique2.set(row.scenarioId, {
5125
+ row,
5126
+ delta: reference === void 0 ? null : row.composite - reference
5127
+ });
5128
+ }
5129
+ const hardest = [...unique2.values()].sort(
5130
+ (a, b) => a.row.composite - b.row.composite || compareScenarioId(a.row, b.row)
5131
+ );
5132
+ const regressions = [...unique2.values()].filter((entry) => entry.delta !== null && entry.delta < 0).sort((a, b) => a.delta - b.delta || compareScenarioId(a.row, b.row));
5133
+ const improvements = [...unique2.values()].filter((entry) => entry.delta !== null && entry.delta > 0).sort((a, b) => b.delta - a.delta || compareScenarioId(a.row, b.row));
5134
+ const rankings = [hardest, regressions, improvements];
5135
+ const maxDepth = Math.max(...rankings.map((ranking) => ranking.length), 0);
5136
+ const selected = [];
5137
+ const selectedIds = /* @__PURE__ */ new Set();
5138
+ for (let depth = 0; depth < maxDepth && selected.length < options.limit; depth += 1) {
5139
+ for (const ranking of rankings) {
5140
+ const entry = ranking[depth];
5141
+ if (!entry || selectedIds.has(entry.row.scenarioId)) continue;
5142
+ selected.push(entry.row);
5143
+ selectedIds.add(entry.row.scenarioId);
5144
+ if (selected.length === options.limit) break;
5145
+ }
5146
+ }
5147
+ return selected;
5148
+ }
5149
+ function assertPolicyEditAuthorContextBudget(value, maxChars) {
5150
+ assertPositiveSafeInteger(maxChars, "maxChars");
5151
+ const json = JSON.stringify(value);
5152
+ if (json === void 0) {
5153
+ throw new Error("assertPolicyEditAuthorContextBudget: value must serialize to JSON");
5154
+ }
5155
+ const actualChars = json.length;
5156
+ if (actualChars > maxChars) {
5157
+ throw new Error(
5158
+ `assertPolicyEditAuthorContextBudget: serialized JSON exceeds budget (actualChars=${actualChars}, maxChars=${maxChars})`
5159
+ );
5160
+ }
5161
+ return { json, actualChars, maxChars };
5162
+ }
5163
+ function compareScenarioId(a, b) {
5164
+ return a.scenarioId < b.scenarioId ? -1 : a.scenarioId > b.scenarioId ? 1 : 0;
5165
+ }
5166
+ function assertPositiveSafeInteger(value, name) {
5167
+ if (!Number.isSafeInteger(value) || value <= 0) {
5168
+ throw new Error(`${name} must be a positive safe integer (got ${value})`);
5169
+ }
5170
+ }
5171
+
5172
+ // src/campaign/proposers/llm-policy-edit.ts
5173
+ var JSON_POLICY_EDIT_TARGET_SURFACES = [
5174
+ "prompt",
5175
+ "tool-contract",
5176
+ "runtime-config",
5177
+ "memory",
5178
+ "agent-profile"
5179
+ ];
5180
+ var NonEmptyStringSchema = z.string().trim().min(1);
5181
+ var JsonValueSchema = z.lazy(
5182
+ () => z.union([
5183
+ z.string(),
5184
+ z.number().finite(),
5185
+ z.boolean(),
5186
+ z.null(),
5187
+ z.array(JsonValueSchema),
5188
+ z.record(NonEmptyStringSchema, JsonValueSchema)
5189
+ ])
5190
+ );
5191
+ var AuthoredJsonChangeSchema = z.discriminatedUnion("mode", [
5192
+ z.object({
5193
+ kind: z.literal("json"),
5194
+ mode: z.literal("set"),
5195
+ path: NonEmptyStringSchema,
5196
+ value: JsonValueSchema
5197
+ }).strict(),
5198
+ z.object({
5199
+ kind: z.literal("json"),
5200
+ mode: z.literal("merge"),
5201
+ path: NonEmptyStringSchema,
5202
+ value: JsonValueSchema
5203
+ }).strict(),
5204
+ z.object({
5205
+ kind: z.literal("json"),
5206
+ mode: z.literal("remove"),
5207
+ path: NonEmptyStringSchema
5208
+ }).strict()
5209
+ ]);
5210
+ var AuthoredPolicyEditSchema = z.object({
5211
+ axis: z.enum(POLICY_EDIT_AXES),
5212
+ target: z.object({
5213
+ surface: z.enum(POLICY_EDIT_TARGET_SURFACES),
5214
+ path: NonEmptyStringSchema,
5215
+ label: NonEmptyStringSchema.max(200).nullable()
5216
+ }).strict(),
5217
+ change: AuthoredJsonChangeSchema,
5218
+ claim: NonEmptyStringSchema.max(2e3),
5219
+ expectedGain: z.object({
5220
+ metric: NonEmptyStringSchema.max(400),
5221
+ direction: z.enum(["increase", "decrease"]),
5222
+ amount: z.number().finite().positive(),
5223
+ unit: z.enum(["absolute", "relative", "percent", "score"]).nullable(),
5224
+ rationale: NonEmptyStringSchema.max(2e3).nullable()
5225
+ }).strict(),
5226
+ confidence: z.number().finite().min(0).max(1),
5227
+ risk: z.enum(["low", "medium", "high", "unknown"]),
5228
+ source: z.object({
5229
+ findingKeys: z.array(NonEmptyStringSchema).min(1).refine((ids) => new Set(ids).size === ids.length, "findingKeys must be unique")
5230
+ }).strict(),
5231
+ rationale: NonEmptyStringSchema.max(4e3).nullable(),
5232
+ validationPlan: NonEmptyStringSchema.max(2e3).nullable()
5233
+ }).strict();
5234
+ var PolicyEditAuthorResponseSchema = z.object({
5235
+ edits: z.array(AuthoredPolicyEditSchema)
5236
+ }).strict();
5237
+ var POLICY_EDIT_AUTHOR_JSON_SCHEMA = {
5238
+ type: "object",
5239
+ additionalProperties: false,
5240
+ required: ["edits"],
5241
+ properties: {
5242
+ edits: {
5243
+ type: "array",
5244
+ items: {
5245
+ type: "object",
5246
+ additionalProperties: false,
5247
+ required: [
5248
+ "axis",
5249
+ "target",
5250
+ "change",
5251
+ "claim",
5252
+ "expectedGain",
5253
+ "confidence",
5254
+ "risk",
5255
+ "source",
5256
+ "rationale",
5257
+ "validationPlan"
5258
+ ],
5259
+ properties: {
5260
+ axis: { type: "string", enum: [...POLICY_EDIT_AXES] },
5261
+ target: {
5262
+ type: "object",
5263
+ additionalProperties: false,
5264
+ required: ["surface", "path", "label"],
5265
+ properties: {
5266
+ surface: { type: "string", enum: [...POLICY_EDIT_TARGET_SURFACES] },
5267
+ path: { type: "string", minLength: 1 },
5268
+ label: { type: ["string", "null"], maxLength: 200 }
5269
+ }
5270
+ },
5271
+ change: {
5272
+ anyOf: [
5273
+ {
5274
+ type: "object",
5275
+ additionalProperties: false,
5276
+ required: ["kind", "mode", "path", "value"],
5277
+ properties: {
5278
+ kind: { const: "json" },
5279
+ mode: { const: "set" },
5280
+ path: { type: "string", minLength: 1 },
5281
+ value: {}
5282
+ }
5283
+ },
5284
+ {
5285
+ type: "object",
5286
+ additionalProperties: false,
5287
+ required: ["kind", "mode", "path", "value"],
5288
+ properties: {
5289
+ kind: { const: "json" },
5290
+ mode: { const: "merge" },
5291
+ path: { type: "string", minLength: 1 },
5292
+ value: {}
5293
+ }
5294
+ },
5295
+ {
5296
+ type: "object",
5297
+ additionalProperties: false,
5298
+ required: ["kind", "mode", "path"],
5299
+ properties: {
5300
+ kind: { const: "json" },
5301
+ mode: { const: "remove" },
5302
+ path: { type: "string", minLength: 1 }
5303
+ }
5304
+ }
5305
+ ]
5306
+ },
5307
+ claim: { type: "string", minLength: 1, maxLength: 2e3 },
5308
+ expectedGain: {
5309
+ type: "object",
5310
+ additionalProperties: false,
5311
+ required: ["metric", "direction", "amount", "unit", "rationale"],
5312
+ properties: {
5313
+ metric: { type: "string", minLength: 1, maxLength: 400 },
5314
+ direction: { type: "string", enum: ["increase", "decrease"] },
5315
+ amount: { type: "number", exclusiveMinimum: 0 },
5316
+ unit: {
5317
+ type: ["string", "null"],
5318
+ enum: ["absolute", "relative", "percent", "score", null]
5319
+ },
5320
+ rationale: { type: ["string", "null"], maxLength: 2e3 }
5321
+ }
5322
+ },
5323
+ confidence: { type: "number", minimum: 0, maximum: 1 },
5324
+ risk: { type: "string", enum: ["low", "medium", "high", "unknown"] },
5325
+ source: {
5326
+ type: "object",
5327
+ additionalProperties: false,
5328
+ required: ["findingKeys"],
5329
+ properties: {
5330
+ findingKeys: {
5331
+ type: "array",
5332
+ minItems: 1,
5333
+ uniqueItems: true,
5334
+ items: { type: "string", minLength: 1 }
5335
+ }
5336
+ }
5337
+ },
5338
+ rationale: { type: ["string", "null"], maxLength: 4e3 },
5339
+ validationPlan: { type: ["string", "null"], maxLength: 2e3 }
5340
+ }
5341
+ }
5342
+ }
5343
+ }
5344
+ };
5345
+ function policyEditAuthorJsonSchema(maxItems, targetSurface, allowedJsonPaths, objectives) {
5346
+ const schema = JSON.parse(JSON.stringify(POLICY_EDIT_AUTHOR_JSON_SCHEMA));
5347
+ const properties = schema.properties;
5348
+ const edits = properties.edits;
5349
+ edits.maxItems = maxItems;
5350
+ const item = edits.items;
5351
+ const itemProperties = item.properties;
5352
+ const target = itemProperties.target;
5353
+ const targetProperties = target.properties;
5354
+ targetProperties.surface = { type: "string", enum: [targetSurface] };
5355
+ targetProperties.path = { type: "string", enum: [...allowedJsonPaths] };
5356
+ const change = itemProperties.change;
5357
+ for (const variant of change.anyOf) {
5358
+ const variantProperties = variant.properties;
5359
+ variantProperties.path = { type: "string", enum: [...allowedJsonPaths] };
5360
+ }
5361
+ const expectedGain = itemProperties.expectedGain;
5362
+ const gainProperties = expectedGain.properties;
5363
+ gainProperties.metric = { type: "string", enum: objectives.map((objective) => objective.key) };
5364
+ gainProperties.direction = {
5365
+ type: "string",
5366
+ enum: [...new Set(objectives.map((objective) => objective.direction))]
5367
+ };
5368
+ gainProperties.unit = {
5369
+ type: "string",
5370
+ enum: [...new Set(objectives.map((objective) => objective.unit))]
5371
+ };
5372
+ return schema;
5373
+ }
5374
+ var DEFAULT_POLICY_EDIT_HISTORY_LIMITS = Object.freeze({
5375
+ generations: 4,
5376
+ candidatesPerGeneration: 16,
5377
+ scenariosPerCandidate: 12,
5378
+ findings: 32,
5379
+ authorContextChars: 2e5
5380
+ });
5381
+ var POLICY_EDIT_AUTHOR_SYSTEM = [
5382
+ "You author strictly typed PolicyEdit candidates over one JSON surface.",
5383
+ 'Return exactly one JSON object with shape {"edits":[...]}; emit an empty edits array when no evidence supports a change.',
5384
+ `axis must be one of: ${POLICY_EDIT_AXES.join(", ")}.`,
5385
+ `target.surface must be one of: ${POLICY_EDIT_TARGET_SURFACES.join(", ")}.`,
5386
+ "target.path and change.path must be the same caller-allowed JSON path.",
5387
+ 'change must be exactly one operation: {"kind":"json","mode":"set","path":string,"value":json}, {"kind":"json","mode":"merge","path":string,"value":json}, or {"kind":"json","mode":"remove","path":string}.',
5388
+ "Nullable fields required by the response schema must be null when they do not apply.",
5389
+ "Every edit must cite one or more supplied finding keys in source.findingKeys. Do not emit persistent finding IDs, analyst IDs, or evidence references; the caller binds those from the cited findings.",
5390
+ "Treat expectedGain and confidence as forecasts, never as measured evidence. Learn from baselineOutcome, incumbentOutcome, and observedDeltaFromParent.",
5391
+ "Do not invent a finding, path, field, score, or task fact. Do not include schemaVersion, editId, metadata, prose, or undeclared keys."
5392
+ ].join("\n");
5393
+ function llmPolicyEditProposer(opts) {
5394
+ const allowedJsonPaths = validateAllowedJsonPaths(opts.allowedJsonPaths);
5395
+ const allowedPathSet = new Set(allowedJsonPaths);
5396
+ const objectives = validateObjectives(opts.objectives);
5397
+ const objectiveByKey = new Map(objectives.map((objective) => [objective.key, objective]));
5398
+ requireNonEmpty(opts.model, "model");
5399
+ requireNonEmpty(opts.target, "target");
5400
+ if (!JSON_POLICY_EDIT_TARGET_SURFACES.includes(opts.targetSurface)) {
5401
+ throw new Error(
5402
+ `llmPolicyEditProposer: targetSurface '${opts.targetSurface}' is not a JSON-backed surface`
5403
+ );
5404
+ }
5405
+ if (opts.maxCandidates !== void 0 && (!Number.isSafeInteger(opts.maxCandidates) || opts.maxCandidates < 0)) {
5406
+ throw new Error("llmPolicyEditProposer: maxCandidates must be a non-negative safe integer");
5407
+ }
5408
+ const admissionMode = opts.admissionMode ?? "evidence-only";
5409
+ if (admissionMode !== "evidence-only" && admissionMode !== "strict") {
5410
+ throw new Error("llmPolicyEditProposer: admissionMode must be 'evidence-only' or 'strict'");
5411
+ }
5412
+ if (opts.admission && admissionMode !== "strict") {
5413
+ throw new Error("llmPolicyEditProposer: admission thresholds require admissionMode: 'strict'");
5414
+ }
5415
+ const historyLimits = validateHistoryLimits({
5416
+ ...opts.maxHistoryGenerations === void 0 ? {} : { maxGenerations: opts.maxHistoryGenerations },
5417
+ ...opts.maxHistoryCandidatesPerGeneration === void 0 ? {} : { maxCandidatesPerGeneration: opts.maxHistoryCandidatesPerGeneration },
5418
+ ...opts.maxScenariosPerCandidate === void 0 ? {} : { maxScenariosPerCandidate: opts.maxScenariosPerCandidate },
5419
+ ...opts.scenarioIdTransform === void 0 ? {} : { scenarioIdTransform: opts.scenarioIdTransform }
5420
+ });
5421
+ const maxFindings = positiveLimit(
5422
+ opts.maxFindings ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.findings,
5423
+ "maxFindings"
5424
+ );
5425
+ const maxAuthorContextChars = positiveLimit(
5426
+ opts.maxAuthorContextChars ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.authorContextChars,
5427
+ "maxAuthorContextChars"
5428
+ );
5429
+ const directCostLedger = opts.costLedger ?? new CostLedger();
5430
+ return {
5431
+ kind: "llm-policy-edit",
5432
+ async propose(ctx) {
5433
+ const limit = Math.min(ctx.populationSize, opts.maxCandidates ?? ctx.populationSize);
5434
+ if (limit <= 0) return [];
5435
+ const currentSurface = parseJsonSurface(ctx.currentSurface);
5436
+ assertSearchOutcome(ctx.baselineOutcome, "baselineOutcome");
5437
+ assertSearchOutcome(ctx.incumbentOutcome, "incumbentOutcome");
5438
+ assertMeasuredCompositesInScale(ctx, objectives[0]);
5439
+ const scenarioIds = createScenarioIdProjector(historyLimits.scenarioIdTransform);
5440
+ registerOutcomeScenarioIds(ctx.baselineOutcome, scenarioIds);
5441
+ registerOutcomeScenarioIds(ctx.incumbentOutcome, scenarioIds);
5442
+ registerHistoryScenarioIds(ctx.history.slice(-historyLimits.maxGenerations), scenarioIds);
5443
+ assertSurfaceIsTaskAgnostic(
5444
+ { currentSurface, allowedJsonPaths, objectives, targetSurface: opts.targetSurface },
5445
+ scenarioIds
5446
+ );
5447
+ const measuredSources = measuredSourceMeasurements(ctx);
5448
+ const findings = citableFindings(ctx.findings, measuredSources, maxFindings);
5449
+ const findingByKey = new Map(
5450
+ findings.map((finding, index) => [`finding-${index + 1}`, finding])
5451
+ );
5452
+ const authorContext = {
5453
+ target: scenarioIds.sanitize(opts.target),
5454
+ targetSurface: opts.targetSurface,
5455
+ allowedJsonPaths,
5456
+ objectives,
5457
+ candidateCount: limit,
5458
+ generation: ctx.generation,
5459
+ currentSurface,
5460
+ findings: findings.map(
5461
+ (finding, index) => renderFinding(finding, `finding-${index + 1}`, scenarioIds, measuredSources)
5462
+ ),
5463
+ baselineOutcome: projectOutcome(
5464
+ ctx.baselineOutcome,
5465
+ scenarioIds,
5466
+ historyLimits.maxScenariosPerCandidate
5467
+ ),
5468
+ incumbentOutcome: projectOutcome(
5469
+ ctx.incumbentOutcome,
5470
+ scenarioIds,
5471
+ historyLimits.maxScenariosPerCandidate,
5472
+ ctx.baselineOutcome
5473
+ ),
5474
+ history: projectPolicyEditHistoryWithProjector(
5475
+ ctx.history,
5476
+ historyLimits,
5477
+ scenarioIds,
5478
+ objectiveByKey
5479
+ )
5480
+ };
5481
+ const responseSchema = policyEditAuthorJsonSchema(
5482
+ limit,
5483
+ opts.targetSurface,
5484
+ allowedJsonPaths,
5485
+ objectives
5486
+ );
5487
+ assertPolicyEditAuthorContextBudget(
5488
+ { system: POLICY_EDIT_AUTHOR_SYSTEM, authorContext, responseSchema },
5489
+ maxAuthorContextChars
5490
+ );
5491
+ const userContent = JSON.stringify(authorContext);
5492
+ const request = {
5493
+ model: opts.model,
5494
+ messages: [
5495
+ { role: "system", content: POLICY_EDIT_AUTHOR_SYSTEM },
5496
+ { role: "user", content: userContent }
5497
+ ],
5498
+ jsonSchema: {
5499
+ name: "policy_edit_author",
5500
+ schema: responseSchema
5501
+ },
5502
+ temperature: opts.temperature ?? 0.2,
5503
+ maxTokens: opts.maxTokens ?? 6e3,
5504
+ timeoutMs: opts.timeoutMs
5505
+ };
5506
+ const paid = await (ctx.costLedger ?? directCostLedger).runPaidCall({
5507
+ channel: "driver",
5508
+ phase: ctx.costPhase ?? "search.proposal",
5509
+ actor: "llm-policy-edit.author",
5510
+ model: opts.model,
5511
+ maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
5512
+ tags: { generation: String(ctx.generation) },
5513
+ signal: ctx.signal,
5514
+ execute: (signal, callId) => callLlmJson(request, { ...opts.llm, signal, idempotencyKey: callId }),
5515
+ receipt: ({ result }) => costReceiptFromLlm(result),
5516
+ receiptFromError: costReceiptFromLlmError
5517
+ });
5518
+ if (!paid.succeeded) throw paid.error;
5519
+ const { value } = paid.value;
5520
+ const response = parseAuthorResponse(value);
5521
+ if (response.edits.length > limit) {
5522
+ throw new Error(
5523
+ `llmPolicyEditProposer: author returned ${response.edits.length} edits for ${limit} candidate slots`
5524
+ );
5525
+ }
5526
+ const edits = response.edits.map(
5527
+ (draft) => bindAuthoredEdit(
5528
+ draft,
5529
+ findingByKey,
5530
+ opts.targetSurface,
5531
+ allowedPathSet,
5532
+ objectiveByKey,
5533
+ ctx.incumbentOutcome?.composite ?? ctx.baselineOutcome?.composite
5534
+ )
5535
+ );
5536
+ return policyEditProposer({
5537
+ edits,
5538
+ admission: admissionMode === "strict" ? opts.admission : {
5539
+ minScore: 0,
5540
+ minExpectedGain: 0,
5541
+ allowHighRisk: true,
5542
+ requireEvidence: true
5543
+ },
5544
+ maxCandidates: limit,
5545
+ onAdmission: opts.onAdmission
5546
+ }).propose(ctx);
5547
+ }
5548
+ };
5549
+ }
5550
+ function projectPolicyEditHistory(history, options = {}) {
5551
+ const limits = validateHistoryLimits(options);
5552
+ const objectiveByKey = new Map(
5553
+ (options.objectives ? validateObjectives(options.objectives) : []).map((objective) => [
5554
+ objective.key,
5555
+ objective
5556
+ ])
5557
+ );
5558
+ const scenarioIds = createScenarioIdProjector(limits.scenarioIdTransform);
5559
+ registerHistoryScenarioIds(history.slice(-limits.maxGenerations), scenarioIds);
5560
+ return projectPolicyEditHistoryWithProjector(history, limits, scenarioIds, objectiveByKey);
5561
+ }
5562
+ function projectPolicyEditHistoryWithProjector(history, limits, scenarioIds, objectiveByKey) {
5563
+ const retainedHistory = history.slice(-limits.maxGenerations);
5564
+ const candidateByHash = new Map(
5565
+ retainedHistory.flatMap(
5566
+ (record) => record.candidates.map((candidate) => [candidate.surfaceHash, candidate])
5567
+ )
5568
+ );
5569
+ return retainedHistory.map((record) => {
5570
+ const candidates = selectHistoryCandidates(record, limits.maxCandidatesPerGeneration);
5571
+ const hashes = new Set(candidates.map((candidate) => candidate.surfaceHash));
5572
+ return {
5573
+ generationIndex: record.generationIndex,
5574
+ promoted: record.promoted.filter((hash) => hashes.has(hash)).map((hash) => scenarioIds.sanitize(hash)),
5575
+ candidates: candidates.map(
5576
+ (candidate) => projectHistoryCandidate(
5577
+ candidate,
5578
+ scenarioIds,
5579
+ limits.maxScenariosPerCandidate,
5580
+ candidate.parentSurfaceHash ? candidateByHash.get(candidate.parentSurfaceHash)?.scenarios : void 0,
5581
+ objectiveByKey
5582
+ )
5583
+ )
5584
+ };
5585
+ });
5586
+ }
5587
+ function selectHistoryCandidates(record, limit) {
5588
+ const promotionOrder = new Map(record.promoted.map((hash, index) => [hash, index]));
5589
+ const promoted = [...record.candidates].filter((candidate) => promotionOrder.has(candidate.surfaceHash)).sort((a, b) => promotionOrder.get(a.surfaceHash) - promotionOrder.get(b.surfaceHash));
5590
+ const selected = promoted.slice(0, limit);
5591
+ const selectedHashes = new Set(selected.map((candidate) => candidate.surfaceHash));
5592
+ const remaining = [...record.candidates].filter((candidate) => !promotionOrder.has(candidate.surfaceHash)).sort((a, b) => b.composite - a.composite || a.surfaceHash.localeCompare(b.surfaceHash));
5593
+ let high = 0;
5594
+ let low = remaining.length - 1;
5595
+ let takeLow = selected.length > 0;
5596
+ while (selected.length < limit && high <= low) {
5597
+ const candidate = takeLow ? remaining[low--] : remaining[high++];
5598
+ takeLow = !takeLow;
5599
+ if (!candidate || selectedHashes.has(candidate.surfaceHash)) continue;
5600
+ selected.push(candidate);
5601
+ selectedHashes.add(candidate.surfaceHash);
5602
+ }
5603
+ return selected;
5604
+ }
5605
+ function projectHistoryCandidate(candidate, scenarioIds, maxScenarios, parentScenarios, objectiveByKey) {
5606
+ const referenceByScenario = parentScenarios ? new Map(parentScenarios.map((scenario) => [scenario.scenarioId, scenario.composite])) : void 0;
5607
+ const selectedScenarios = selectPolicyEditAuthorRows(candidate.scenarios, {
5608
+ limit: maxScenarios,
5609
+ ...referenceByScenario ? { referenceByScenario } : {}
5610
+ });
5611
+ const validatedRecord = candidate.candidateRecord ? validatePolicyEditCandidateRecord(candidate.candidateRecord) : void 0;
5612
+ return {
5613
+ surfaceHash: scenarioIds.sanitize(candidate.surfaceHash),
5614
+ parentSurfaceHash: candidate.parentSurfaceHash ? scenarioIds.sanitize(candidate.parentSurfaceHash) : null,
5615
+ parentComposite: candidate.parentComposite ?? null,
5616
+ label: candidate.label ? scenarioIds.sanitize(candidate.label) : null,
5617
+ rationale: candidate.rationale ? scenarioIds.sanitize(candidate.rationale) : null,
5618
+ composite: candidate.composite,
5619
+ observedDeltaFromParent: candidate.observedDeltaFromParent ?? null,
5620
+ eligibleForPromotion: candidate.eligibleForPromotion ?? null,
5621
+ coverage: candidate.coverage ? {
5622
+ expectedCells: candidate.coverage.expectedCells,
5623
+ scorableCells: candidate.coverage.scorableCells,
5624
+ // `cellId` embeds the raw scenario ID. The aggregate reason is useful
5625
+ // for search, but the identifier must not bypass scenarioIdTransform.
5626
+ unscorableCells: [...candidate.coverage.unscorableCells].sort((a, b) => a.cellId.localeCompare(b.cellId)).slice(0, maxScenarios).map((cell) => ({ reason: scenarioIds.sanitize(cell.reason) }))
5627
+ } : null,
5628
+ // GenerationCandidate.ci95 is currently a placeholder [composite, composite],
5629
+ // not a measured interval. Keep it out of author context until it is real.
5630
+ dimensions: sanitizeDimensions(candidate.dimensions, scenarioIds),
5631
+ scenarios: selectedScenarios.map((scenario) => {
5632
+ return {
5633
+ scenarioId: scenarioIds.project(scenario.scenarioId),
5634
+ composite: scenario.composite,
5635
+ notes: scenario.notes ? scenarioIds.sanitize(scenario.notes) : null
5636
+ };
5637
+ }),
5638
+ candidateEdit: validatedRecord ? summarizeCandidateEdit(validatedRecord, scenarioIds) : null,
5639
+ forecastCalibration: forecastCalibration(candidate, validatedRecord, objectiveByKey)
5640
+ };
5641
+ }
5642
+ function projectOutcome(outcome, scenarioIds, maxScenarios, reference = void 0) {
5643
+ if (!outcome) return null;
5644
+ const referenceByScenario = reference ? new Map(reference.scenarios.map((scenario) => [scenario.scenarioId, scenario.composite])) : void 0;
5645
+ const selectedScenarios = selectPolicyEditAuthorRows(outcome.scenarios, {
5646
+ limit: maxScenarios,
5647
+ ...referenceByScenario ? { referenceByScenario } : {}
5648
+ });
5649
+ return {
5650
+ split: "search",
5651
+ generation: outcome.generation,
5652
+ surfaceHash: scenarioIds.sanitize(outcome.surfaceHash),
5653
+ composite: outcome.composite,
5654
+ dimensions: sanitizeDimensions(outcome.dimensions, scenarioIds),
5655
+ scenarios: selectedScenarios.map((scenario) => {
5656
+ return {
5657
+ scenarioId: scenarioIds.project(scenario.scenarioId),
5658
+ composite: scenario.composite,
5659
+ notes: scenario.notes ? scenarioIds.sanitize(scenario.notes) : null
5660
+ };
5661
+ }),
5662
+ coverage: { ...outcome.coverage }
5663
+ };
5664
+ }
5665
+ function createScenarioIdProjector(transform) {
5666
+ const aliases = /* @__PURE__ */ new Map();
5667
+ const originals = /* @__PURE__ */ new Map();
5668
+ return {
5669
+ project(scenarioId) {
5670
+ const known = aliases.get(scenarioId);
5671
+ if (known) return known;
5672
+ const alias = transform(scenarioId);
5673
+ if (!alias || alias.trim() !== alias) {
5674
+ throw new Error(
5675
+ "llmPolicyEditProposer: scenarioIdTransform must return a trimmed non-empty string"
5676
+ );
5677
+ }
5678
+ const collision = originals.get(alias);
5679
+ if (collision && collision !== scenarioId) {
5680
+ throw new Error(
5681
+ `llmPolicyEditProposer: scenarioIdTransform collision for '${collision}' and '${scenarioId}'`
5682
+ );
5683
+ }
5684
+ const aliasIsAnotherOriginal = aliases.has(alias) && alias !== scenarioId;
5685
+ const originalIsAnotherAlias = originals.has(scenarioId) && originals.get(scenarioId) !== scenarioId;
5686
+ if (aliasIsAnotherOriginal || originalIsAnotherAlias) {
5687
+ throw new Error(
5688
+ "llmPolicyEditProposer: scenarioIdTransform aliases overlap raw scenario IDs"
5689
+ );
5690
+ }
5691
+ aliases.set(scenarioId, alias);
5692
+ originals.set(alias, scenarioId);
5693
+ return alias;
5694
+ },
5695
+ sanitize(text) {
5696
+ const originalsByLength = [...aliases.keys()].sort((a, b) => b.length - a.length);
5697
+ if (originalsByLength.length === 0) return text;
5698
+ const pattern = new RegExp(
5699
+ `(^|[^A-Za-z0-9_-])(${originalsByLength.map(escapeRegExp).join("|")})(?=$|[^A-Za-z0-9_-])`,
5700
+ "g"
5701
+ );
5702
+ return text.replace(
5703
+ pattern,
5704
+ (_match, prefix, original) => `${prefix}${aliases.get(original) ?? original}`
5705
+ );
5706
+ }
5707
+ };
5708
+ }
5709
+ function escapeRegExp(value) {
5710
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
5711
+ }
5712
+ function sourceKey(surfaceHash2, generation) {
5713
+ return `${surfaceHash2}@${generation}`;
5714
+ }
5715
+ function measuredSourceMeasurements(ctx) {
5716
+ const measurements = /* @__PURE__ */ new Map();
5717
+ const add = (measurement) => {
5718
+ measurements.set(sourceKey(measurement.surfaceHash, measurement.generation), measurement);
5719
+ };
5720
+ if (ctx.baselineOutcome) {
5721
+ add({
5722
+ surfaceHash: ctx.baselineOutcome.surfaceHash,
5723
+ generation: ctx.baselineOutcome.generation,
5724
+ composite: ctx.baselineOutcome.composite,
5725
+ parentComposite: null,
5726
+ observedDeltaFromParent: null,
5727
+ eligibleForPromotion: true,
5728
+ coverage: { ...ctx.baselineOutcome.coverage }
5729
+ });
5730
+ }
5731
+ for (const record of ctx.history) {
5732
+ for (const candidate of record.candidates) {
5733
+ add({
5734
+ surfaceHash: candidate.surfaceHash,
5735
+ generation: record.generationIndex,
5736
+ composite: candidate.composite,
5737
+ parentComposite: candidate.parentComposite ?? null,
5738
+ observedDeltaFromParent: candidate.observedDeltaFromParent ?? null,
5739
+ eligibleForPromotion: candidate.eligibleForPromotion === true,
5740
+ coverage: candidate.coverage ? {
5741
+ expectedCells: candidate.coverage.expectedCells,
5742
+ scorableCells: candidate.coverage.scorableCells
5743
+ } : { expectedCells: 0, scorableCells: 0 }
5744
+ });
5745
+ }
5746
+ }
5747
+ if (ctx.incumbentOutcome) {
5748
+ const generation = ctx.incumbentOutcome.generation;
5749
+ const key = sourceKey(ctx.incumbentOutcome.surfaceHash, generation);
5750
+ if (!measurements.has(key)) {
5751
+ add({
5752
+ surfaceHash: ctx.incumbentOutcome.surfaceHash,
5753
+ generation,
5754
+ composite: ctx.incumbentOutcome.composite,
5755
+ parentComposite: null,
5756
+ observedDeltaFromParent: null,
5757
+ eligibleForPromotion: true,
5758
+ coverage: { ...ctx.incumbentOutcome.coverage }
5759
+ });
5760
+ }
5761
+ }
5762
+ return measurements;
5763
+ }
5764
+ function validateFindingSource(source, measured) {
5765
+ if (source.kind === "global") {
5766
+ requireNonEmpty(source.label, "global finding source label");
5767
+ return;
5768
+ }
5769
+ if (!measured.has(sourceKey(source.surfaceHash, source.generation))) {
5770
+ throw new Error(
5771
+ `llmPolicyEditProposer: finding source ${source.surfaceHash}@${source.generation} is not a measured surface`
5772
+ );
5773
+ }
5774
+ }
5775
+ function sameFindingSource(a, b) {
5776
+ if (a.kind !== b.kind) return false;
5777
+ return a.kind === "surface" && b.kind === "surface" ? a.surfaceHash === b.surfaceHash && a.generation === b.generation : a.kind === "global" && b.kind === "global" && a.label === b.label;
5778
+ }
5779
+ function summarizeCandidateEdit(record, scenarioIds) {
5780
+ const edit = record.policyEdit;
5781
+ return {
5782
+ editId: edit.editId,
5783
+ axis: edit.axis,
5784
+ target: sanitizeAuthorValue(edit.target, scenarioIds),
5785
+ change: sanitizeAuthorValue(edit.change, scenarioIds),
5786
+ claim: scenarioIds.sanitize(edit.claim),
5787
+ expectedGain: sanitizeAuthorValue(edit.expectedGain, scenarioIds),
5788
+ confidence: edit.confidence,
5789
+ risk: edit.risk,
5790
+ sourceFindingIds: edit.source.findingIds.map((id) => scenarioIds.sanitize(id)),
5791
+ rationale: edit.rationale ? scenarioIds.sanitize(edit.rationale) : null,
5792
+ validationPlan: edit.validationPlan ? scenarioIds.sanitize(edit.validationPlan) : null
5793
+ };
5794
+ }
5795
+ function sanitizeAuthorValue(value, scenarioIds) {
5796
+ if (typeof value === "string") return scenarioIds.sanitize(value);
5797
+ if (Array.isArray(value)) return value.map((item) => sanitizeAuthorValue(item, scenarioIds));
5798
+ if (value && typeof value === "object") {
5799
+ return Object.fromEntries(
5800
+ Object.entries(value).map(([key, child]) => [
5801
+ scenarioIds.sanitize(key),
5802
+ sanitizeAuthorValue(child, scenarioIds)
5803
+ ])
5804
+ );
5805
+ }
5806
+ return value;
5807
+ }
5808
+ function sanitizeDimensions(dimensions, scenarioIds) {
5809
+ const sanitized = /* @__PURE__ */ new Map();
5810
+ for (const [key, value] of Object.entries(dimensions)) {
5811
+ const projected = scenarioIds.sanitize(key);
5812
+ const prior = sanitized.get(projected);
5813
+ if (prior && prior.original !== key) {
5814
+ throw new Error(
5815
+ `llmPolicyEditProposer: pseudonymized dimension key collision for '${prior.original}' and '${key}'`
5816
+ );
5817
+ }
5818
+ sanitized.set(projected, { original: key, value });
5819
+ }
5820
+ return Object.fromEntries([...sanitized].map(([key, entry]) => [key, entry.value]));
5821
+ }
5822
+ function forecastCalibration(candidate, record, objectiveByKey) {
5823
+ if (!record || candidate.observedDeltaFromParent === void 0) return null;
5824
+ const forecast = record.policyEdit.expectedGain;
5825
+ const objective = objectiveByKey.get(forecast.metric);
5826
+ if (!objective || objective.key !== "search.composite") return null;
5827
+ if (forecast.direction !== objective.direction || forecast.unit !== objective.unit) return null;
5828
+ if (forecast.amount > objective.scale.max - objective.scale.min) return null;
5829
+ const predictedDelta = forecast.amount;
5830
+ return {
5831
+ objectiveKey: objective.key,
5832
+ predictedDelta,
5833
+ observedDelta: candidate.observedDeltaFromParent,
5834
+ residual: candidate.observedDeltaFromParent - predictedDelta
5835
+ };
5836
+ }
5837
+ function assertSearchOutcome(outcome, field) {
5838
+ if (outcome && outcome.split !== "search") {
5839
+ throw new Error(`llmPolicyEditProposer: ${field} must be a search-split outcome`);
5840
+ }
5841
+ }
5842
+ function assertMeasuredCompositesInScale(ctx, objective) {
5843
+ const assertInScale = (value, field) => {
5844
+ if (!Number.isFinite(value) || value < objective.scale.min || value > objective.scale.max) {
5845
+ throw new Error(
5846
+ `llmPolicyEditProposer: ${field} ${value} is outside objective '${objective.key}' scale [${objective.scale.min}, ${objective.scale.max}]`
5847
+ );
5848
+ }
5849
+ };
5850
+ const assertOutcome = (outcome, field) => {
5851
+ if (!outcome) return;
5852
+ assertInScale(outcome.composite, `${field}.composite`);
5853
+ for (const [index, scenario] of outcome.scenarios.entries()) {
5854
+ assertInScale(scenario.composite, `${field}.scenarios[${index}].composite`);
5855
+ }
5856
+ };
5857
+ assertOutcome(ctx.baselineOutcome, "baselineOutcome");
5858
+ assertOutcome(ctx.incumbentOutcome, "incumbentOutcome");
5859
+ for (const [generationIndex, generation] of ctx.history.entries()) {
5860
+ for (const [candidateIndex, candidate] of generation.candidates.entries()) {
5861
+ const field = `history[${generationIndex}].candidates[${candidateIndex}]`;
5862
+ assertInScale(candidate.composite, `${field}.composite`);
5863
+ if (candidate.parentComposite !== void 0) {
5864
+ assertInScale(candidate.parentComposite, `${field}.parentComposite`);
5865
+ }
5866
+ for (const [scenarioIndex, scenario] of candidate.scenarios.entries()) {
5867
+ assertInScale(scenario.composite, `${field}.scenarios[${scenarioIndex}].composite`);
5868
+ }
5869
+ }
5870
+ }
5871
+ }
5872
+ function assertSurfaceIsTaskAgnostic(value, scenarioIds) {
5873
+ const serialized = JSON.stringify(value);
5874
+ if (scenarioIds.sanitize(serialized) !== serialized) {
5875
+ throw new Error(
5876
+ "llmPolicyEditProposer: current JSON surface or signature contains a raw scenario identifier"
5877
+ );
5878
+ }
5879
+ }
5880
+ function registerOutcomeScenarioIds(outcome, scenarioIds) {
5881
+ for (const scenario of outcome?.scenarios ?? []) scenarioIds.project(scenario.scenarioId);
5882
+ }
5883
+ function registerHistoryScenarioIds(history, scenarioIds) {
5884
+ for (const record of history) {
5885
+ for (const candidate of record.candidates) {
5886
+ for (const scenario of candidate.scenarios) scenarioIds.project(scenario.scenarioId);
5887
+ for (const cell of candidate.coverage?.unscorableCells ?? []) {
5888
+ const separator = cell.cellId.lastIndexOf(":");
5889
+ if (separator <= 0 || !/^\d+$/.test(cell.cellId.slice(separator + 1))) continue;
5890
+ scenarioIds.project(cell.cellId.slice(0, separator));
5891
+ }
5892
+ }
5893
+ }
5894
+ }
5895
+ function validateHistoryLimits(options) {
5896
+ const maxGenerations = options.maxGenerations ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.generations;
5897
+ const maxCandidatesPerGeneration = options.maxCandidatesPerGeneration ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.candidatesPerGeneration;
5898
+ const maxScenariosPerCandidate = options.maxScenariosPerCandidate ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.scenariosPerCandidate;
5899
+ if (!Number.isSafeInteger(maxGenerations) || maxGenerations <= 0) {
5900
+ throw new Error("llmPolicyEditProposer: maxHistoryGenerations must be a positive safe integer");
5901
+ }
5902
+ if (!Number.isSafeInteger(maxCandidatesPerGeneration) || maxCandidatesPerGeneration <= 0) {
5903
+ throw new Error(
5904
+ "llmPolicyEditProposer: maxHistoryCandidatesPerGeneration must be a positive safe integer"
5905
+ );
5906
+ }
5907
+ if (!Number.isSafeInteger(maxScenariosPerCandidate) || maxScenariosPerCandidate <= 0) {
5908
+ throw new Error(
5909
+ "llmPolicyEditProposer: maxScenariosPerCandidate must be a positive safe integer"
5910
+ );
5911
+ }
5912
+ return {
5913
+ maxGenerations,
5914
+ maxCandidatesPerGeneration,
5915
+ maxScenariosPerCandidate,
5916
+ scenarioIdTransform: options.scenarioIdTransform ?? ((scenarioId) => scenarioId)
5917
+ };
5918
+ }
5919
+ function validateObjectives(inputs) {
5920
+ if (inputs.length === 0) {
5921
+ throw new Error("llmPolicyEditProposer: objectives must not be empty");
5922
+ }
5923
+ const seen = /* @__PURE__ */ new Set();
5924
+ return inputs.map((input) => {
5925
+ requireNonEmpty(input.key, "objective key");
5926
+ if (seen.has(input.key)) {
5927
+ throw new Error(`llmPolicyEditProposer: duplicate objective '${input.key}'`);
5928
+ }
5929
+ seen.add(input.key);
5930
+ if (input.split !== "search") {
5931
+ throw new Error(`llmPolicyEditProposer: objective '${input.key}' must use the search split`);
5932
+ }
5933
+ if (input.key !== "search.composite") {
5934
+ throw new Error(
5935
+ `llmPolicyEditProposer: objective '${input.key}' is not yet measurable; use 'search.composite'`
5936
+ );
5937
+ }
5938
+ if (input.direction !== "increase") {
5939
+ throw new Error(
5940
+ `llmPolicyEditProposer: objective '${input.key}' must increase because search promotes larger composite scores`
5941
+ );
5942
+ }
5943
+ if (!Number.isFinite(input.scale.min) || !Number.isFinite(input.scale.max) || input.scale.max <= input.scale.min) {
5944
+ throw new Error(`llmPolicyEditProposer: objective '${input.key}' has an invalid scale`);
5945
+ }
5946
+ if (input.unit !== "score") {
5947
+ throw new Error(`llmPolicyEditProposer: objective '${input.key}' must use raw score deltas`);
5948
+ }
5949
+ return {
5950
+ key: input.key,
5951
+ split: "search",
5952
+ direction: input.direction,
5953
+ scale: { ...input.scale },
5954
+ unit: input.unit
5955
+ };
5956
+ });
5957
+ }
5958
+ function positiveLimit(value, field) {
5959
+ if (!Number.isSafeInteger(value) || value <= 0) {
5960
+ throw new Error(`llmPolicyEditProposer: ${field} must be a positive safe integer`);
5961
+ }
5962
+ return value;
5963
+ }
5964
+ function parseAuthorResponse(value) {
5965
+ const parsed = PolicyEditAuthorResponseSchema.safeParse(value);
5966
+ if (parsed.success) return parsed.data;
5967
+ const issues = parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; ");
5968
+ throw new Error(`llmPolicyEditProposer: invalid PolicyEdit response: ${issues}`);
5969
+ }
5970
+ function bindAuthoredEdit(draft, findingByKey, targetSurface, allowedPaths, objectiveByKey, currentComposite) {
5971
+ if (draft.target.surface !== targetSurface) {
5972
+ throw new Error(
5973
+ `llmPolicyEditProposer: target surface '${draft.target.surface}' does not match '${targetSurface}'`
5974
+ );
5975
+ }
5976
+ if (draft.target.path !== draft.change.path) {
5977
+ throw new Error("llmPolicyEditProposer: target.path must equal change.path");
5978
+ }
5979
+ if (!allowedPaths.has(draft.change.path)) {
5980
+ throw new Error(
5981
+ `llmPolicyEditProposer: JSON path '${draft.change.path}' is outside allowedJsonPaths`
5982
+ );
5983
+ }
5984
+ const cited = draft.source.findingKeys.map((findingKey) => {
5985
+ const finding = findingByKey.get(findingKey);
5986
+ if (!finding) {
5987
+ throw new Error(
5988
+ `llmPolicyEditProposer: edit cites unknown or uncitable finding key '${findingKey}'`
5989
+ );
5990
+ }
5991
+ return finding;
5992
+ });
5993
+ const evidenceRefs = uniqueEvidenceRefs(cited.flatMap((finding) => finding.evidenceRefs));
5994
+ if (evidenceRefs.length === 0) {
5995
+ throw new Error("llmPolicyEditProposer: authored edit has no cited evidence");
5996
+ }
5997
+ const objective = objectiveByKey.get(draft.expectedGain.metric);
5998
+ if (!objective) {
5999
+ throw new Error(
6000
+ `llmPolicyEditProposer: unknown forecast objective '${draft.expectedGain.metric}'`
6001
+ );
6002
+ }
6003
+ if (draft.expectedGain.direction !== objective.direction) {
6004
+ throw new Error(
6005
+ `llmPolicyEditProposer: forecast direction for '${objective.key}' must be '${objective.direction}'`
6006
+ );
6007
+ }
6008
+ if (draft.expectedGain.unit !== objective.unit) {
6009
+ throw new Error(
6010
+ `llmPolicyEditProposer: forecast unit for '${objective.key}' must be '${objective.unit}'`
6011
+ );
6012
+ }
6013
+ const maxGain = currentComposite === void 0 ? objective.scale.max - objective.scale.min : objective.scale.max - currentComposite;
6014
+ if (draft.expectedGain.amount > maxGain) {
6015
+ throw new Error(
6016
+ `llmPolicyEditProposer: forecast amount for '${objective.key}' exceeds the available score headroom`
6017
+ );
6018
+ }
6019
+ const expectedGain = {
6020
+ metric: draft.expectedGain.metric,
6021
+ direction: draft.expectedGain.direction,
6022
+ amount: draft.expectedGain.amount,
6023
+ ...draft.expectedGain.unit ? { unit: draft.expectedGain.unit } : {},
6024
+ ...draft.expectedGain.rationale ? { rationale: draft.expectedGain.rationale } : {}
6025
+ };
6026
+ const init = {
6027
+ axis: draft.axis,
6028
+ target: {
6029
+ surface: draft.target.surface,
6030
+ path: draft.target.path,
6031
+ ...draft.target.label ? { label: draft.target.label } : {}
6032
+ },
6033
+ change: draft.change,
6034
+ claim: draft.claim,
6035
+ expectedGain,
6036
+ confidence: draft.confidence,
6037
+ risk: draft.risk,
6038
+ source: {
6039
+ findingIds: [...new Set(cited.map((finding) => finding.finding.finding_id))],
6040
+ analystIds: [...new Set(cited.map((finding) => finding.finding.analyst_id))],
6041
+ evidenceRefs
6042
+ },
6043
+ ...draft.rationale ? { rationale: draft.rationale } : {},
6044
+ ...draft.validationPlan ? { validationPlan: draft.validationPlan } : {}
6045
+ };
6046
+ return makePolicyEdit(init);
6047
+ }
6048
+ function parseJsonSurface(surface) {
6049
+ if (typeof surface !== "string") {
6050
+ throw new Error("llmPolicyEditProposer: currentSurface must be serialized JSON");
6051
+ }
6052
+ let parsed;
6053
+ try {
6054
+ parsed = JSON.parse(surface);
6055
+ } catch {
6056
+ throw new Error("llmPolicyEditProposer: currentSurface must be valid JSON");
6057
+ }
6058
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
6059
+ throw new Error("llmPolicyEditProposer: currentSurface JSON root must be an object");
6060
+ }
6061
+ return parsed;
6062
+ }
6063
+ function citableFindings(inputs, measuredSources, limit) {
6064
+ if (inputs.length === 0) {
6065
+ throw new Error("llmPolicyEditProposer: at least one analyst finding is required");
6066
+ }
6067
+ const findings = [];
6068
+ for (const input of inputs) {
6069
+ if (!isPolicyEditFindingInput(input)) {
6070
+ throw new Error(
6071
+ "llmPolicyEditProposer: ctx.findings must contain attributed PolicyEditFindingInput rows"
6072
+ );
6073
+ }
6074
+ findings.push(input.finding);
6075
+ }
6076
+ assertNoJudgeVerdict(findings, "llmPolicyEditProposer");
6077
+ const grouped = /* @__PURE__ */ new Map();
6078
+ for (const input of inputs) {
6079
+ validateFindingSource(input.source, measuredSources);
6080
+ if (input.finding.evidence_refs.length === 0) continue;
6081
+ const existing = grouped.get(input.finding.finding_id);
6082
+ if (existing) {
6083
+ if (existing.finding.analyst_id !== input.finding.analyst_id || existing.finding.area !== input.finding.area || existing.finding.claim !== input.finding.claim || existing.finding.subject !== input.finding.subject) {
6084
+ throw new Error(
6085
+ `llmPolicyEditProposer: finding '${input.finding.finding_id}' has conflicting content`
6086
+ );
6087
+ }
6088
+ if (!existing.sources.some((source) => sameFindingSource(source, input.source))) {
6089
+ existing.sources.push(input.source);
6090
+ }
6091
+ existing.evidenceRefs = uniqueEvidenceRefs([
6092
+ ...existing.evidenceRefs,
6093
+ ...input.finding.evidence_refs
6094
+ ]);
6095
+ continue;
6096
+ }
6097
+ grouped.set(input.finding.finding_id, {
6098
+ finding: input.finding,
6099
+ sources: [input.source],
6100
+ evidenceRefs: uniqueEvidenceRefs(input.finding.evidence_refs)
6101
+ });
6102
+ }
6103
+ if (grouped.size === 0) {
6104
+ throw new Error("llmPolicyEditProposer: no evidence-bearing findings are available");
6105
+ }
6106
+ const severityRank = {
6107
+ critical: 0,
6108
+ high: 1,
6109
+ medium: 2,
6110
+ low: 3,
6111
+ info: 4
6112
+ };
6113
+ return [...grouped.values()].sort(
6114
+ (a, b) => severityRank[a.finding.severity] - severityRank[b.finding.severity] || b.finding.confidence - a.finding.confidence || a.finding.finding_id.localeCompare(b.finding.finding_id)
6115
+ ).slice(0, limit);
6116
+ }
6117
+ function isAnalystFindingLike2(input) {
6118
+ if (!input || typeof input !== "object") return false;
6119
+ const value = input;
6120
+ return typeof value.finding_id === "string" && typeof value.analyst_id === "string" && typeof value.claim === "string" && Array.isArray(value.evidence_refs);
6121
+ }
6122
+ function isPolicyEditFindingInput(input) {
6123
+ if (!input || typeof input !== "object") return false;
6124
+ const value = input;
6125
+ return isAnalystFindingLike2(value.finding) && isFindingSource(value.source);
6126
+ }
6127
+ function isFindingSource(input) {
6128
+ if (!input || typeof input !== "object") return false;
6129
+ const value = input;
6130
+ if (value.kind === "surface") {
6131
+ return typeof value.surfaceHash === "string" && Number.isSafeInteger(value.generation);
6132
+ }
6133
+ return value.kind === "global" && typeof value.label === "string" && value.label.trim().length > 0;
6134
+ }
6135
+ function renderFinding(context, findingKey, scenarioIds, measuredSources) {
6136
+ const { finding } = context;
6137
+ return {
6138
+ findingKey,
6139
+ sources: context.sources.map(
6140
+ (source) => source.kind === "surface" ? {
6141
+ ...measuredSources.get(sourceKey(source.surfaceHash, source.generation)),
6142
+ surfaceHash: scenarioIds.sanitize(source.surfaceHash),
6143
+ kind: "surface"
6144
+ } : { kind: "global", label: scenarioIds.sanitize(source.label) }
6145
+ ),
6146
+ analystId: scenarioIds.sanitize(finding.analyst_id),
6147
+ area: scenarioIds.sanitize(finding.area),
6148
+ severity: finding.severity,
6149
+ subject: finding.subject ? scenarioIds.sanitize(finding.subject) : null,
6150
+ claim: scenarioIds.sanitize(finding.claim),
6151
+ rationale: finding.rationale ? scenarioIds.sanitize(finding.rationale) : null,
6152
+ recommendedAction: finding.recommended_action ? scenarioIds.sanitize(finding.recommended_action) : null,
6153
+ validationPlan: finding.validation_plan ? scenarioIds.sanitize(finding.validation_plan) : null,
6154
+ confidence: finding.confidence,
6155
+ evidenceRefs: context.evidenceRefs.map((ref) => ({
6156
+ kind: ref.kind,
6157
+ uri: scenarioIds.sanitize(ref.uri),
6158
+ excerpt: ref.excerpt ? scenarioIds.sanitize(ref.excerpt) : null
6159
+ }))
6160
+ };
6161
+ }
6162
+ function uniqueEvidenceRefs(refs) {
6163
+ const seen = /* @__PURE__ */ new Set();
6164
+ return refs.filter((ref) => {
6165
+ const key = JSON.stringify([ref.kind, ref.uri, ref.excerpt ?? null]);
6166
+ if (seen.has(key)) return false;
6167
+ seen.add(key);
6168
+ return true;
6169
+ });
6170
+ }
6171
+ function validateAllowedJsonPaths(paths) {
6172
+ if (paths.length === 0) {
6173
+ throw new Error("llmPolicyEditProposer: allowedJsonPaths must not be empty");
6174
+ }
6175
+ for (const path of paths) {
6176
+ if (!path || path.trim() !== path) {
6177
+ throw new Error(
6178
+ "llmPolicyEditProposer: allowedJsonPaths must contain trimmed non-empty paths"
6179
+ );
6180
+ }
6181
+ }
6182
+ if (new Set(paths).size !== paths.length) {
6183
+ throw new Error("llmPolicyEditProposer: allowedJsonPaths must be unique");
6184
+ }
6185
+ return [...paths];
6186
+ }
6187
+ function requireNonEmpty(value, field) {
6188
+ if (!value || value.trim() !== value) {
6189
+ throw new Error(`llmPolicyEditProposer: ${field} must be a trimmed non-empty string`);
6190
+ }
6191
+ }
6192
+
6193
+ // src/campaign/proposers/memory.ts
6194
+ var BLOCK_START2 = "<!-- BEGIN curated-memory (auto-managed by memoryCurationProposer) -->";
6195
+ var BLOCK_END2 = "<!-- END curated-memory -->";
6196
+ var DEFAULT_HEADING2 = "## Learned from prior runs (curated memory)";
6197
+ var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, generalizable agent guidance. Output ONLY a JSON array of strings, each one imperative lesson the agent should follow (e.g. "Always fetch a resource before mutating it"). No prose outside the JSON. Deduplicate; keep the most actionable and general; drop case-specific noise.';
6198
+ function extractExistingLessons(text) {
6199
+ return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
6200
+ }
6201
+ async function distillLessons(raw, distill, ctx, costLedger) {
6202
+ const request = {
6203
+ model: distill.model,
6204
+ messages: [
6205
+ { role: "system", content: DISTILL_SYSTEM },
6206
+ { role: "user", content: `Findings:
6207
+ ${raw.map((r) => `- ${r}`).join("\n")}` }
6208
+ ],
6209
+ maxTokens: distill.maxTokens ?? 2e3
6210
+ };
6211
+ const llm = {
6212
+ baseUrl: distill.baseUrl,
6213
+ apiKey: distill.apiKey,
6214
+ fetch: distill.fetchImpl
6215
+ };
6216
+ const paid = await costLedger.runPaidCall({
6217
+ channel: "driver",
6218
+ phase: ctx.costPhase ?? "search.proposal",
6219
+ actor: "memory-curation.distill",
6220
+ model: distill.model,
6221
+ maximumCharge: maximumChargeForLlmRequest(request, llm),
6222
+ tags: { generation: String(ctx.generation) },
6223
+ signal: ctx.signal,
6224
+ execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
6225
+ receipt: costReceiptFromLlm,
6226
+ receiptFromError: costReceiptFromLlmError
6227
+ });
6228
+ if (!paid.succeeded) throw paid.error;
6229
+ const res = paid.value;
6230
+ try {
6231
+ const parsed = JSON.parse(res.content.trim());
6232
+ if (Array.isArray(parsed)) {
6233
+ const lessons = parsed.filter(
6234
+ (x) => typeof x === "string" && x.trim().length > 0
6235
+ );
6236
+ if (lessons.length > 0) return lessons;
6237
+ }
6238
+ } catch {
6239
+ }
6240
+ return raw;
6241
+ }
6242
+ function memoryCurationProposer(opts = {}) {
6243
+ const maxEntries = opts.maxEntries ?? 12;
6244
+ const heading = opts.sectionHeading ?? DEFAULT_HEADING2;
6245
+ const directCostLedger = opts.costLedger ?? new CostLedger();
6246
+ return {
6247
+ kind: "memory-curation",
6248
+ async propose(ctx) {
6249
+ const parent = surfaceToText2(ctx.currentSurface);
6250
+ const fresh = [];
6251
+ for (const f of ctx.findings ?? []) {
6252
+ const l = findingToLesson(f);
6253
+ if (l) fresh.push(l);
6254
+ }
6255
+ const carried = extractExistingLessons(parent);
6256
+ if (fresh.length === 0 && carried.length === 0) return [];
6257
+ const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill, ctx, ctx.costLedger ?? directCostLedger) : fresh;
6258
+ const byKey = /* @__PURE__ */ new Map();
6259
+ for (const l of carried) {
6260
+ const k = normKey(l);
6261
+ if (k) byKey.set(k, { text: l, count: 1 });
6262
+ }
6263
+ for (const l of distilled) {
6264
+ const k = normKey(l);
6265
+ if (!k) continue;
6266
+ const e = byKey.get(k);
6267
+ if (e) e.count += 1;
6268
+ else byKey.set(k, { text: l, count: 1 });
6269
+ }
6270
+ const ranked = [...byKey.values()].sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).slice(0, maxEntries);
6271
+ if (ranked.length === 0) return [];
6272
+ const block = [BLOCK_START2, heading, ...ranked.map((e) => `- ${e.text}`), BLOCK_END2].join(
6273
+ "\n"
6274
+ );
6275
+ const next = `${stripBlock(parent, BLOCK_START2, BLOCK_END2)}
6276
+
6277
+ ${block}
6278
+ `;
6279
+ if (next === parent) return [];
6280
+ return [
6281
+ {
6282
+ surface: next,
6283
+ label: "memory-curation",
6284
+ rationale: `curated ${ranked.length} lessons (from ${fresh.length} new finding(s) + ${carried.length} carried)`
6285
+ }
6286
+ ];
6287
+ }
6288
+ };
6289
+ }
6290
+
6291
+ // src/campaign/proposers/trace-analyst.ts
6292
+ import { ai } from "@ax-llm/ax";
6293
+ function renderFindings(findings) {
6294
+ return findings.map((f, i) => {
6295
+ const action = f.recommended_action ? `
6296
+ FIX: ${f.recommended_action}` : "";
6297
+ const subject = f.subject ? ` (${f.subject})` : "";
6298
+ return `${i + 1}. [${f.severity}/${f.area}]${subject} ${f.claim}${action}`;
6299
+ }).join("\n");
6300
+ }
6301
+ function traceAnalystProposer(opts) {
6302
+ if (!opts.apiKey) throw new Error("traceAnalystProposer: apiKey is required");
5422
6303
  if (!opts.model) throw new Error("traceAnalystProposer: model is required");
5423
6304
  const kinds = opts.kinds ?? DEFAULT_TRACE_ANALYST_KINDS;
5424
6305
  const produceFindings = opts.analyze ?? (async (path, c) => {
@@ -5444,7 +6325,12 @@ function traceAnalystProposer(opts) {
5444
6325
  label: "trace-analyst",
5445
6326
  baseUrl: opts.baseUrl,
5446
6327
  apiKey: opts.apiKey,
6328
+ analysisModel: opts.model,
5447
6329
  applyModel: opts.applyModel ?? opts.model,
6330
+ costLedger: opts.costLedger,
6331
+ analysisMaximumCharge: opts.analysisMaximumCharge,
6332
+ analysisReceipt: opts.analysisReceipt,
6333
+ applyMaxTokens: opts.applyMaxTokens,
5448
6334
  fetchImpl: opts.fetchImpl,
5449
6335
  resolveTraces: opts.resolveTraces,
5450
6336
  noTracesError: "traceAnalystProposer: resolveTraces returned no OTLP traces \u2014 the analyst has nothing to read",
@@ -5516,161 +6402,9 @@ function selectDiscriminative(signals, k, opts) {
5516
6402
 
5517
6403
  // src/campaign/search-ledger.ts
5518
6404
  import { createHash as createHash7 } from "crypto";
5519
- import { existsSync as existsSync4, readFileSync as readFileSync4 } from "fs";
6405
+ import { existsSync as existsSync3, readFileSync as readFileSync3 } from "fs";
5520
6406
  import { resolve as resolve2 } from "path";
5521
6407
  import { z as z2 } from "zod";
5522
-
5523
- // src/campaign/search-ledger-errors.ts
5524
- var SearchLedgerError = class extends ValidationError {
5525
- };
5526
- var SearchLedgerIntegrityError = class extends SearchLedgerError {
5527
- };
5528
- var SearchLedgerConflictError = class extends SearchLedgerError {
5529
- };
5530
-
5531
- // src/campaign/search-ledger-file.ts
5532
- import { randomUUID as randomUUID2 } from "crypto";
5533
- import {
5534
- closeSync,
5535
- constants,
5536
- existsSync as existsSync3,
5537
- fsyncSync,
5538
- linkSync,
5539
- mkdirSync as mkdirSync2,
5540
- openSync,
5541
- readFileSync as readFileSync3,
5542
- renameSync,
5543
- unlinkSync,
5544
- writeSync
5545
- } from "fs";
5546
- import { hostname } from "os";
5547
- import { dirname as dirname2 } from "path";
5548
- import { z } from "zod";
5549
- function appendSearchLedgerLine(path, line) {
5550
- mkdirSync2(dirname2(path), { recursive: true });
5551
- const fd = openSync(path, constants.O_CREAT | constants.O_WRONLY | constants.O_APPEND, 384);
5552
- try {
5553
- writeAll(fd, Buffer.from(line, "utf8"));
5554
- fsyncSync(fd);
5555
- } finally {
5556
- closeSync(fd);
5557
- }
5558
- fsyncDirectory(dirname2(path));
5559
- }
5560
- function withSearchLedgerFileLock(ledgerPath, run) {
5561
- mkdirSync2(dirname2(ledgerPath), { recursive: true });
5562
- const lockPath = `${ledgerPath}.lock`;
5563
- const owner = acquireLock(lockPath);
5564
- try {
5565
- return run();
5566
- } finally {
5567
- releaseLock(lockPath, owner);
5568
- }
5569
- }
5570
- function writeAll(fd, bytes) {
5571
- let offset = 0;
5572
- while (offset < bytes.byteLength) {
5573
- const written = writeSync(fd, bytes, offset, bytes.byteLength - offset);
5574
- if (written <= 0) throw new SearchLedgerIntegrityError("filesystem wrote zero bytes");
5575
- offset += written;
5576
- }
5577
- }
5578
- function fsyncDirectory(path) {
5579
- const fd = openSync(path, constants.O_RDONLY);
5580
- try {
5581
- fsyncSync(fd);
5582
- } finally {
5583
- closeSync(fd);
5584
- }
5585
- }
5586
- function acquireLock(lockPath) {
5587
- const owner = { pid: process.pid, host: hostname(), nonce: randomUUID2() };
5588
- const ownerBytes = `${canonicalOwner(owner)}
5589
- `;
5590
- for (let attempt = 0; attempt < 8; attempt += 1) {
5591
- const ownerPath = `${lockPath}.${owner.pid}.${owner.nonce}.${attempt}.owner`;
5592
- const fd = openSync(ownerPath, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY, 384);
5593
- try {
5594
- writeAll(fd, Buffer.from(ownerBytes, "utf8"));
5595
- fsyncSync(fd);
5596
- } finally {
5597
- closeSync(fd);
5598
- }
5599
- try {
5600
- linkSync(ownerPath, lockPath);
5601
- unlinkSync(ownerPath);
5602
- return owner;
5603
- } catch (error) {
5604
- unlinkIfExists(ownerPath);
5605
- if (error.code !== "EEXIST") throw error;
5606
- }
5607
- const holder = readOwner(lockPath);
5608
- if (holder.host !== owner.host || isProcessAlive(holder.pid)) {
5609
- throw new SearchLedgerIntegrityError(
5610
- `search ledger lock is held by pid ${holder.pid} on ${holder.host}`
5611
- );
5612
- }
5613
- const tombstone = `${lockPath}.stale.${owner.nonce}.${attempt}`;
5614
- try {
5615
- renameSync(lockPath, tombstone);
5616
- unlinkSync(tombstone);
5617
- } catch (error) {
5618
- if (error.code !== "ENOENT") throw error;
5619
- }
5620
- }
5621
- throw new SearchLedgerIntegrityError(`could not acquire search ledger lock ${lockPath}`);
5622
- }
5623
- function releaseLock(lockPath, owner) {
5624
- if (!existsSync3(lockPath)) return;
5625
- const holder = readOwner(lockPath);
5626
- if (canonicalOwner(holder) !== canonicalOwner(owner)) {
5627
- throw new SearchLedgerIntegrityError(
5628
- `search ledger lock owner changed before release (${lockPath})`
5629
- );
5630
- }
5631
- unlinkSync(lockPath);
5632
- }
5633
- function readOwner(lockPath) {
5634
- let raw;
5635
- try {
5636
- raw = JSON.parse(readFileSync3(lockPath, "utf8"));
5637
- } catch (error) {
5638
- throw new SearchLedgerIntegrityError(`search ledger lock ${lockPath} is malformed`, {
5639
- cause: error
5640
- });
5641
- }
5642
- const parsed = z.object({
5643
- pid: z.number().int().positive().safe(),
5644
- host: z.string().trim().min(1),
5645
- nonce: z.string().trim().min(1)
5646
- }).strict().safeParse(raw);
5647
- if (!parsed.success) {
5648
- throw new SearchLedgerIntegrityError(
5649
- `search ledger lock ${lockPath} is malformed: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; ")}`
5650
- );
5651
- }
5652
- return parsed.data;
5653
- }
5654
- function canonicalOwner(owner) {
5655
- return JSON.stringify({ host: owner.host, nonce: owner.nonce, pid: owner.pid });
5656
- }
5657
- function isProcessAlive(pid) {
5658
- try {
5659
- process.kill(pid, 0);
5660
- return true;
5661
- } catch (error) {
5662
- return error.code !== "ESRCH";
5663
- }
5664
- }
5665
- function unlinkIfExists(path) {
5666
- try {
5667
- unlinkSync(path);
5668
- } catch (error) {
5669
- if (error.code !== "ENOENT") throw error;
5670
- }
5671
- }
5672
-
5673
- // src/campaign/search-ledger.ts
5674
6408
  var SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
5675
6409
  var NON_EMPTY = z2.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
5676
6410
  var HASH = z2.string().regex(/^sha256:[a-f0-9]{64}$/);
@@ -6036,8 +6770,8 @@ var FileSearchLedger = class {
6036
6770
  }
6037
6771
  };
6038
6772
  function replayFile(path, campaignId) {
6039
- if (!existsSync4(path)) return replayEntries([], campaignId);
6040
- const text = readFileSync4(path, "utf8");
6773
+ if (!existsSync3(path)) return replayEntries([], campaignId);
6774
+ const text = readFileSync3(path, "utf8");
6041
6775
  if (text.length === 0) return replayEntries([], campaignId);
6042
6776
  if (!text.endsWith("\n")) {
6043
6777
  throw new SearchLedgerIntegrityError(
@@ -6633,10 +7367,10 @@ function formatZodError(error) {
6633
7367
  }
6634
7368
 
6635
7369
  // src/campaign/single-run-lock.ts
6636
- import { existsSync as existsSync5, readFileSync as readFileSync5, unlinkSync as unlinkSync2, writeFileSync as writeFileSync3 } from "fs";
7370
+ import { existsSync as existsSync4, readFileSync as readFileSync4, unlinkSync, writeFileSync as writeFileSync3 } from "fs";
6637
7371
  function liveHolder(path) {
6638
- if (!existsSync5(path)) return null;
6639
- const holder = Number(readFileSync5(path, "utf8").trim());
7372
+ if (!existsSync4(path)) return null;
7373
+ const holder = Number(readFileSync4(path, "utf8").trim());
6640
7374
  if (!Number.isFinite(holder) || holder <= 0) return null;
6641
7375
  try {
6642
7376
  process.kill(holder, 0);
@@ -6658,8 +7392,8 @@ function acquireSingleRunLock(opts) {
6658
7392
  writeFileSync3(opts.lockPath, String(pid));
6659
7393
  const release = () => {
6660
7394
  try {
6661
- if (existsSync5(opts.lockPath) && readFileSync5(opts.lockPath, "utf8").trim() === String(pid)) {
6662
- unlinkSync2(opts.lockPath);
7395
+ if (existsSync4(opts.lockPath) && readFileSync4(opts.lockPath, "utf8").trim() === String(pid)) {
7396
+ unlinkSync(opts.lockPath);
6663
7397
  }
6664
7398
  } catch {
6665
7399
  }
@@ -6683,21 +7417,21 @@ function isTransientTransportFailure(message, opts = {}) {
6683
7417
  import { execFileSync } from "child_process";
6684
7418
  import { createHash as createHash8 } from "crypto";
6685
7419
  import {
6686
- closeSync as closeSync2,
6687
- existsSync as existsSync6,
7420
+ closeSync,
7421
+ existsSync as existsSync5,
6688
7422
  constants as fsConstants,
6689
7423
  fstatSync,
6690
7424
  lstatSync,
6691
- mkdirSync as mkdirSync3,
7425
+ mkdirSync as mkdirSync2,
6692
7426
  mkdtempSync as mkdtempSync2,
6693
- openSync as openSync2,
7427
+ openSync,
6694
7428
  readlinkSync,
6695
7429
  readSync,
6696
7430
  realpathSync,
6697
7431
  rmSync
6698
7432
  } from "fs";
6699
7433
  import { devNull, tmpdir as tmpdir2 } from "os";
6700
- import { basename, dirname as dirname3, isAbsolute as isAbsolute2, join as join5, relative as relative2, resolve as resolve3, sep } from "path";
7434
+ import { basename, dirname as dirname2, isAbsolute as isAbsolute2, join as join5, relative as relative2, resolve as resolve3, sep } from "path";
6701
7435
  var MAX_GIT_OUTPUT_BYTES = 256 * 1024 * 1024;
6702
7436
  var FILE_HASH_CHUNK_BYTES = 1024 * 1024;
6703
7437
  var GIT_REPOSITORY_ENV = /* @__PURE__ */ new Set([
@@ -6790,7 +7524,7 @@ function patchBytes(git, cwd, baseCommit, candidateCommit) {
6790
7524
  const scratch = mkdtempSync2(join5(tmpdir2(), "agent-eval-patch-"));
6791
7525
  const bareRepo = join5(scratch, "repo.git");
6792
7526
  const emptyTemplate = join5(scratch, "empty-template");
6793
- mkdirSync3(emptyTemplate);
7527
+ mkdirSync2(emptyTemplate);
6794
7528
  try {
6795
7529
  const objectFormat = gitObjectHashAlgorithm(candidateCommit);
6796
7530
  const sourceObjects = realpathSync(gitText(git, ["rev-parse", "--git-path", "objects"], cwd));
@@ -6928,7 +7662,7 @@ function hashGitBlobFile(path, objectId) {
6928
7662
  throw new WorktreeAdapterError(`CodeSurface expected a regular file at ${displayGitPath(path)}`);
6929
7663
  }
6930
7664
  const noFollow = process.platform === "win32" ? 0 : fsConstants.O_NOFOLLOW;
6931
- const fd = openSync2(path, fsConstants.O_RDONLY | noFollow);
7665
+ const fd = openSync(path, fsConstants.O_RDONLY | noFollow);
6932
7666
  try {
6933
7667
  const opened = fstatSync(fd);
6934
7668
  if (!opened.isFile() || opened.dev !== before.dev || opened.ino !== before.ino || opened.size !== before.size) {
@@ -6953,7 +7687,7 @@ function hashGitBlobFile(path, objectId) {
6953
7687
  }
6954
7688
  return { hash: hash.digest("hex"), executable: (opened.mode & 73) !== 0 };
6955
7689
  } finally {
6956
- closeSync2(fd);
7690
+ closeSync(fd);
6957
7691
  }
6958
7692
  }
6959
7693
  function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
@@ -6969,7 +7703,7 @@ function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
6969
7703
  `CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
6970
7704
  );
6971
7705
  }
6972
- const lexicalTarget = resolve3(dirname3(linkPath), target);
7706
+ const lexicalTarget = resolve3(dirname2(linkPath), target);
6973
7707
  if (!isWithinRoot(root, lexicalTarget)) {
6974
7708
  throw new WorktreeAdapterError(
6975
7709
  `CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
@@ -7046,7 +7780,7 @@ function assertRawTreeMatchesWorktree(git, root, candidateCommit) {
7046
7780
  }
7047
7781
  function verifyCodeSurfaceWithGit(surface, path, git) {
7048
7782
  assertCodeSurfaceIdentity(surface);
7049
- if (!existsSync6(path)) {
7783
+ if (!existsSync5(path)) {
7050
7784
  throw new WorktreeAdapterError(`CodeSurface worktree does not exist: ${path}`);
7051
7785
  }
7052
7786
  const lexicalRoot = resolve3(path);
@@ -7227,13 +7961,6 @@ function resolveWorktreePath(surface, worktreeDir) {
7227
7961
  }
7228
7962
 
7229
7963
  export {
7230
- JudgeParseError,
7231
- createDomainExpertJudge,
7232
- codeExecutionJudge,
7233
- coherenceJudge,
7234
- adversarialJudge,
7235
- createCustomJudge,
7236
- defaultJudges,
7237
7964
  pairArms,
7238
7965
  comparePairedArms,
7239
7966
  completionVerdict,
@@ -7241,7 +7968,6 @@ export {
7241
7968
  parseCorrectnessResponse,
7242
7969
  createLlmCorrectnessChecker,
7243
7970
  createTokenRecallChecker,
7244
- llmJudge,
7245
7971
  extractProducedState,
7246
7972
  CODING_HARNESSES,
7247
7973
  HARNESS_NATIVE_MODEL,
@@ -7297,14 +8023,16 @@ export {
7297
8023
  aceProposer,
7298
8024
  compositeProposer,
7299
8025
  haloProposer,
7300
- memoryCurationProposer,
7301
8026
  policyEditProposer,
8027
+ selectPolicyEditAuthorRows,
8028
+ assertPolicyEditAuthorContextBudget,
8029
+ DEFAULT_POLICY_EDIT_HISTORY_LIMITS,
8030
+ llmPolicyEditProposer,
8031
+ projectPolicyEditHistory,
8032
+ memoryCurationProposer,
7302
8033
  traceAnalystProposer,
7303
8034
  scoreDiscrimination,
7304
8035
  selectDiscriminative,
7305
- SearchLedgerError,
7306
- SearchLedgerIntegrityError,
7307
- SearchLedgerConflictError,
7308
8036
  SEARCH_LEDGER_SCHEMA,
7309
8037
  validateSearchLedgerEvent,
7310
8038
  openSearchLedger,
@@ -7316,4 +8044,4 @@ export {
7316
8044
  verifyCodeSurface,
7317
8045
  resolveWorktreePath
7318
8046
  };
7319
- //# sourceMappingURL=chunk-KG4TD7EQ.js.map
8047
+ //# sourceMappingURL=chunk-ZUXV7UWZ.js.map