@tangle-network/agent-eval 0.116.0 → 0.117.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/CHANGELOG.md +38 -0
  2. package/dist/analyst/index.d.ts +18 -11
  3. package/dist/analyst/index.js +10 -7
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  6. package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
  7. package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
  8. package/dist/belief-state/index.d.ts +6 -6
  9. package/dist/belief-state/index.js +1 -1
  10. package/dist/benchmarks/index.d.ts +11 -8
  11. package/dist/benchmarks/index.js +11 -10
  12. package/dist/builder-eval/index.d.ts +4 -4
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
  15. package/dist/campaign/index.d.ts +54 -30
  16. package/dist/campaign/index.js +18 -13
  17. package/dist/chunk-3YYRZDON.js +45 -0
  18. package/dist/chunk-3YYRZDON.js.map +1 -0
  19. package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
  20. package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
  21. package/dist/chunk-CCZIVI3F.js.map +1 -0
  22. package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
  23. package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
  24. package/dist/chunk-HHWE3POT.js +94 -0
  25. package/dist/chunk-HHWE3POT.js.map +1 -0
  26. package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
  27. package/dist/chunk-HQPHZGL6.js.map +1 -0
  28. package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
  29. package/dist/chunk-IDZTTFRR.js.map +1 -0
  30. package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
  31. package/dist/{chunk-GSW3OBHK.js → chunk-JSJZ4PJ6.js} +406 -726
  32. package/dist/chunk-JSJZ4PJ6.js.map +1 -0
  33. package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
  34. package/dist/chunk-LQUTGLOZ.js.map +1 -0
  35. package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
  36. package/dist/chunk-LTVG32KX.js.map +1 -0
  37. package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
  38. package/dist/chunk-MGEHEHSN.js.map +1 -0
  39. package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
  40. package/dist/chunk-NJC7U437.js.map +1 -0
  41. package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
  42. package/dist/chunk-ODVOOEWQ.js.map +1 -0
  43. package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
  44. package/dist/chunk-S2F4J57L.js.map +1 -0
  45. package/dist/chunk-VCTY3W6J.js +798 -0
  46. package/dist/chunk-VCTY3W6J.js.map +1 -0
  47. package/dist/chunk-VF3XSYTI.js +545 -0
  48. package/dist/chunk-VF3XSYTI.js.map +1 -0
  49. package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
  50. package/dist/chunk-YZPO4UHR.js.map +1 -0
  51. package/dist/cli.js +4 -2
  52. package/dist/cli.js.map +1 -1
  53. package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
  54. package/dist/contract/index.d.ts +43 -29
  55. package/dist/contract/index.js +56 -19
  56. package/dist/contract/index.js.map +1 -1
  57. package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
  58. package/dist/control.d.ts +6 -6
  59. package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
  60. package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
  61. package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
  62. package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
  63. package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
  64. package/dist/fuzz.d.ts +8 -16
  65. package/dist/fuzz.js +72 -42
  66. package/dist/fuzz.js.map +1 -1
  67. package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
  68. package/dist/hosted/index.d.ts +13 -10
  69. package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
  70. package/dist/index.d.ts +102 -57
  71. package/dist/index.js +328 -235
  72. package/dist/index.js.map +1 -1
  73. package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
  74. package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
  75. package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
  76. package/dist/llm-client-qoDd18Qz.d.ts +289 -0
  77. package/dist/meta-eval/index.d.ts +8 -7
  78. package/dist/meta-eval/index.js +1 -1
  79. package/dist/multishot/index.d.ts +9 -6
  80. package/dist/openapi.json +1 -1
  81. package/dist/pipelines/index.d.ts +16 -6
  82. package/dist/pipelines/index.js +119 -23
  83. package/dist/pipelines/index.js.map +1 -1
  84. package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
  85. package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
  86. package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
  87. package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
  88. package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
  89. package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
  90. package/dist/reporting.d.ts +10 -9
  91. package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
  92. package/dist/rl.d.ts +18 -15
  93. package/dist/rl.js +2 -2
  94. package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
  95. package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
  96. package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
  97. package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
  98. package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
  99. package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
  100. package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
  101. package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
  102. package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
  103. package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
  104. package/dist/storyboard/index.d.ts +1 -1
  105. package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
  106. package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
  107. package/dist/traces.d.ts +25 -14
  108. package/dist/traces.js +16 -4
  109. package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
  110. package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
  111. package/dist/wire/index.d.ts +28 -19
  112. package/dist/wire/index.js +4 -2
  113. package/docs/distributed-driver.md +1 -1
  114. package/package.json +3 -3
  115. package/dist/chunk-3274WNK7.js.map +0 -1
  116. package/dist/chunk-4D5RVB3W.js.map +0 -1
  117. package/dist/chunk-7GKEAIAD.js +0 -205
  118. package/dist/chunk-7GKEAIAD.js.map +0 -1
  119. package/dist/chunk-CIUOICJT.js.map +0 -1
  120. package/dist/chunk-FAOEFFRT.js.map +0 -1
  121. package/dist/chunk-GSW3OBHK.js.map +0 -1
  122. package/dist/chunk-GY4SYVPJ.js.map +0 -1
  123. package/dist/chunk-LNQEP766.js.map +0 -1
  124. package/dist/chunk-MHNQWM4I.js.map +0 -1
  125. package/dist/chunk-MPHTT5HE.js +0 -74
  126. package/dist/chunk-MPHTT5HE.js.map +0 -1
  127. package/dist/chunk-NBSS5NDZ.js.map +0 -1
  128. package/dist/chunk-NYFUT3B3.js.map +0 -1
  129. package/dist/chunk-TLDB7WRY.js.map +0 -1
  130. package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
  131. /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
  132. /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
  133. /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
  134. /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
  135. /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
@@ -1,33 +1,39 @@
1
1
  import {
2
+ JudgeParseError,
2
3
  assertCodeSurfaceIdentity,
3
4
  campaignBreakdown,
4
5
  campaignMeanComposite,
6
+ costReceiptFromTCloud,
5
7
  defaultProductionGate,
6
8
  gepaProposer,
7
9
  isProposedCandidate,
8
10
  labelTrustRank,
11
+ maximumChargeForTCloudRequest,
9
12
  pairHoldout,
10
13
  recoverTruncatedJson,
11
14
  renderAnalystEvidence,
12
15
  runImprovementLoop,
13
16
  surfaceContentHash,
14
17
  surfaceHash
15
- } from "./chunk-3274WNK7.js";
16
- import {
17
- estimateCost,
18
- isModelPriced
19
- } from "./chunk-VI2UW6B6.js";
18
+ } from "./chunk-HQPHZGL6.js";
20
19
  import {
20
+ SearchLedgerConflictError,
21
+ SearchLedgerError,
22
+ SearchLedgerIntegrityError,
23
+ appendSearchLedgerLine,
21
24
  assertRealBackend,
22
25
  contentHash,
26
+ createRunCostLedger,
27
+ fsCampaignStorage,
23
28
  planCampaignRun,
29
+ resolveRunDir,
24
30
  runCampaign,
25
- summarizeBackendIntegrity
26
- } from "./chunk-FAOEFFRT.js";
31
+ summarizeBackendIntegrity,
32
+ withSearchLedgerFileLock
33
+ } from "./chunk-IDZTTFRR.js";
27
34
  import {
28
- Mutex,
29
- clamp01
30
- } from "./chunk-MPHTT5HE.js";
35
+ Mutex
36
+ } from "./chunk-3YYRZDON.js";
31
37
  import {
32
38
  AnalystRegistry,
33
39
  DEFAULT_TRACE_ANALYST_KINDS,
@@ -42,22 +48,21 @@ import {
42
48
  makePolicyEditCandidateRecord,
43
49
  policyEditsFromFindings,
44
50
  validatePolicyEditCandidateRecord
45
- } from "./chunk-CIUOICJT.js";
51
+ } from "./chunk-MGEHEHSN.js";
46
52
  import {
47
53
  eProcess,
48
54
  mcnemar,
49
55
  mulberry32,
50
56
  pairedBootstrap,
51
57
  pairedRiskDifference,
52
- weightedComposite,
53
58
  wilcoxonSignedRank
54
59
  } from "./chunk-PJQFMIOX.js";
55
60
  import {
56
61
  analyzeTraces
57
- } from "./chunk-RPDDVKI7.js";
62
+ } from "./chunk-4JLWXDYA.js";
58
63
  import {
59
64
  OtlpFileTraceStore
60
- } from "./chunk-LNQEP766.js";
65
+ } from "./chunk-S2F4J57L.js";
61
66
  import {
62
67
  modelHasSnapshot,
63
68
  validateRunRecord
@@ -70,11 +75,17 @@ import {
70
75
  } from "./chunk-VSMTAMNK.js";
71
76
  import {
72
77
  callLlm,
73
- callLlmJson
74
- } from "./chunk-GY4SYVPJ.js";
78
+ callLlmJson,
79
+ costReceiptFromLlm,
80
+ costReceiptFromLlmError,
81
+ maximumChargeForLlmRequest
82
+ } from "./chunk-NJC7U437.js";
83
+ import {
84
+ CostAccountingIncompleteError,
85
+ CostLedger
86
+ } from "./chunk-VCTY3W6J.js";
75
87
  import {
76
88
  AgentEvalError,
77
- JudgeError,
78
89
  ValidationError
79
90
  } from "./chunk-ONWEPEDO.js";
80
91
 
@@ -496,339 +507,6 @@ async function runLineage(opts) {
496
507
  return { lineage, best: lineage.best(), steps };
497
508
  }
498
509
 
499
- // src/judges.ts
500
- var JudgeParseError = class extends JudgeError {
501
- /** Name of the judge whose response failed to parse. */
502
- judgeName;
503
- /** The raw (truncated) model response that failed to parse. */
504
- raw;
505
- constructor(judgeName, raw, options) {
506
- super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
507
- this.judgeName = judgeName;
508
- this.raw = raw;
509
- }
510
- };
511
- function createDomainExpertJudge(domain) {
512
- return async (tc, { scenario, turns }) => {
513
- const conversation = turns.map(
514
- (t, i) => `Turn ${i + 1}:
515
- User: ${t.userMessage}
516
- Agent: ${t.agentResponse.slice(0, 2e3)}`
517
- ).join("\n\n---\n\n");
518
- const resp = await tc.chat({
519
- model: "gpt-4o",
520
- messages: [
521
- {
522
- role: "system",
523
- content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
524
-
525
- Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
526
-
527
- Evaluate:
528
- 1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
529
- 2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
530
-
531
- Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
532
- },
533
- {
534
- role: "user",
535
- content: `Persona: ${scenario.persona} (${scenario.label})
536
- Scenario: ${scenario.thesis}
537
-
538
- ${conversation}`
539
- }
540
- ],
541
- temperature: 0.1,
542
- maxTokens: 800
543
- });
544
- return parseJudgeResponse("domain_expert", resp);
545
- };
546
- }
547
- var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
548
- const codeBlocks = artifacts.codeBlocks;
549
- if (codeBlocks.length === 0) {
550
- return [
551
- {
552
- judgeName: "code_execution",
553
- dimension: "code_execution",
554
- score: 0,
555
- reasoning: "No code blocks found in agent response."
556
- }
557
- ];
558
- }
559
- const codeText = codeBlocks.map(
560
- (b, i) => `Block ${i + 1} (${b.language}):
561
- \`\`\`${b.language}
562
- ${b.code.slice(0, 3e3)}
563
- \`\`\``
564
- ).join("\n\n");
565
- const resp = await tc.chat({
566
- model: "gpt-4o",
567
- messages: [
568
- {
569
- role: "system",
570
- content: `You are a principal software engineer reviewing code written by an AI agent.
571
-
572
- Score STRICTLY:
573
- 1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
574
- 2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
575
- 3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
576
-
577
- Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
578
- },
579
- {
580
- role: "user",
581
- content: `Task: ${scenario.thesis}
582
-
583
- ${codeText}`
584
- }
585
- ],
586
- temperature: 0.1,
587
- maxTokens: 1e3
588
- });
589
- return parseJudgeResponse("code_execution", resp);
590
- };
591
- var coherenceJudge = async (tc, { scenario, turns }) => {
592
- if (turns.length < 2) {
593
- return [];
594
- }
595
- const conversation = turns.map(
596
- (t, i) => `Turn ${i + 1}:
597
- User: ${t.userMessage}
598
- Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
599
- ).join("\n\n---\n\n");
600
- const resp = await tc.chat({
601
- model: "gpt-4o",
602
- messages: [
603
- {
604
- role: "system",
605
- content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
606
-
607
- Score STRICTLY:
608
- 1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
609
- 2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
610
- 3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
611
-
612
- Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
613
- },
614
- {
615
- role: "user",
616
- content: `Scenario: ${scenario.thesis}
617
-
618
- ${conversation}`
619
- }
620
- ],
621
- temperature: 0.1,
622
- maxTokens: 800
623
- });
624
- return parseJudgeResponse("coherence", resp);
625
- };
626
- var adversarialJudge = async (tc, { scenario, turns }) => {
627
- const conversation = turns.map(
628
- (t, i) => `Turn ${i + 1}:
629
- User: ${t.userMessage}
630
- Agent: ${t.agentResponse.slice(0, 1500)}`
631
- ).join("\n\n---\n\n");
632
- const resp = await tc.chat({
633
- model: "gpt-4o",
634
- messages: [
635
- {
636
- role: "system",
637
- content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
638
-
639
- 1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
640
- 2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
641
- 3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
642
-
643
- Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
644
-
645
- Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
646
- },
647
- {
648
- role: "user",
649
- content: `Persona: ${scenario.persona}
650
- Scenario: ${scenario.thesis}
651
-
652
- ${conversation}`
653
- }
654
- ],
655
- temperature: 0.2,
656
- maxTokens: 800
657
- });
658
- return parseJudgeResponse("adversarial", resp);
659
- };
660
- function createCustomJudge(name, systemPrompt, opts) {
661
- return async (tc, { scenario, turns }) => {
662
- const conversation = turns.map(
663
- (t, i) => `Turn ${i + 1}:
664
- User: ${t.userMessage}
665
- Agent: ${t.agentResponse.slice(0, 2e3)}`
666
- ).join("\n\n---\n\n");
667
- const resp = await tc.chat({
668
- model: opts?.model ?? "gpt-4o",
669
- messages: [
670
- {
671
- role: "system",
672
- content: systemPrompt
673
- },
674
- {
675
- role: "user",
676
- content: `Persona: ${scenario.persona} (${scenario.label})
677
- Scenario: ${scenario.thesis}
678
-
679
- ${conversation}`
680
- }
681
- ],
682
- temperature: opts?.temperature ?? 0.1,
683
- maxTokens: opts?.maxTokens ?? 1e3
684
- });
685
- return parseJudgeResponse(name, resp);
686
- };
687
- }
688
- function defaultJudges(domain) {
689
- return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
690
- }
691
- function parseJudgeResponse(judgeName, resp) {
692
- const content = resp.choices?.[0]?.message?.content ?? "";
693
- try {
694
- let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
695
- const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
696
- if (arrayMatch) cleaned = arrayMatch[0];
697
- const parsed = JSON.parse(cleaned);
698
- return parsed.map((p) => ({
699
- judgeName,
700
- dimension: p.dimension,
701
- score: Math.max(0, Math.min(10, p.score)),
702
- reasoning: p.reasoning ?? "",
703
- evidence: p.evidence
704
- }));
705
- } catch (err) {
706
- throw new JudgeParseError(judgeName, content, { cause: err });
707
- }
708
- }
709
-
710
- // src/llm-judge.ts
711
- function llmJudge(name, prompt, opts) {
712
- if (!name.trim()) {
713
- throw new Error("llmJudge: name must be non-empty");
714
- }
715
- if (!prompt.trim()) {
716
- throw new Error(`llmJudge '${name}': prompt must be non-empty`);
717
- }
718
- const model = opts.model ?? opts.chat.defaultModel;
719
- if (!model) {
720
- throw new Error(
721
- `llmJudge '${name}': no model on opts and no defaultModel on the ChatClient \u2014 pass opts.model or bind defaultModel at createChatClient().`
722
- );
723
- }
724
- const dimensions = normalizeDimensions(opts.dimensions, name);
725
- const scale = opts.scale ?? "unit";
726
- const divisor = scale === "ten" ? 10 : 1;
727
- const renderUser = opts.renderUser ?? ((input) => JSON.stringify({ scenario: input.scenario, artifact: input.artifact }, null, 2));
728
- if (opts.weights) {
729
- for (const key of Object.keys(opts.weights)) {
730
- if (!dimensions.some((d) => d.key === key)) {
731
- throw new Error(
732
- `llmJudge '${name}': weights names dimension '${key}' that is not declared in dimensions`
733
- );
734
- }
735
- }
736
- }
737
- const systemPrompt = `${prompt}
738
-
739
- ${renderContract(dimensions, scale)}`;
740
- return {
741
- name,
742
- dimensions,
743
- appliesTo: opts.appliesTo,
744
- async score({ artifact, scenario, signal }) {
745
- const response = await opts.chat.chat(
746
- {
747
- model,
748
- messages: [
749
- { role: "system", content: systemPrompt },
750
- { role: "user", content: renderUser({ artifact, scenario }) }
751
- ],
752
- jsonMode: true,
753
- temperature: opts.temperature ?? 0.1,
754
- maxTokens: opts.maxTokens ?? 800
755
- },
756
- { signal }
757
- );
758
- const parsed = parseResponse(name, response.content);
759
- const rawDims = parsed.dimensions ?? parsed.scores;
760
- if (!rawDims || typeof rawDims !== "object") {
761
- throw new JudgeParseError(name, response.content, {
762
- cause: new Error("response has no `dimensions` object")
763
- });
764
- }
765
- const dims = {};
766
- for (const { key } of dimensions) {
767
- const raw = rawDims[key];
768
- const value = Number(raw);
769
- if (raw === void 0 || raw === null || !Number.isFinite(value)) {
770
- throw new JudgeParseError(name, response.content, {
771
- cause: new Error(
772
- `dimension '${key}' missing or non-numeric (got ${JSON.stringify(raw)})`
773
- )
774
- });
775
- }
776
- dims[key] = clamp01(value / divisor);
777
- }
778
- const weights = opts.weights ?? Object.fromEntries(dimensions.map((d) => [d.key, 1 / dimensions.length]));
779
- const { composite } = weightedComposite({ dims, weights });
780
- const notes = firstString(parsed.notes) ?? firstString(parsed.rationale) ?? `${name}: composite ${composite.toFixed(3)} over ${dimensions.length} dimension(s)`;
781
- return { dimensions: dims, composite, notes };
782
- }
783
- };
784
- }
785
- function normalizeDimensions(input, name) {
786
- const raw = input && input.length > 0 ? input : ["quality"];
787
- const out = [];
788
- const seen = /* @__PURE__ */ new Set();
789
- for (const d of raw) {
790
- const dim = typeof d === "string" ? { key: d, description: d } : d;
791
- if (!dim.key.trim()) {
792
- throw new Error(`llmJudge '${name}': dimension key must be non-empty`);
793
- }
794
- if (seen.has(dim.key)) {
795
- throw new Error(`llmJudge '${name}': duplicate dimension key '${dim.key}'`);
796
- }
797
- seen.add(dim.key);
798
- out.push(dim);
799
- }
800
- return out;
801
- }
802
- function renderContract(dimensions, scale) {
803
- const range = scale === "ten" ? "0 to 10" : "0.0 to 1.0";
804
- const lines = dimensions.map((d) => ` - "${d.key}": ${d.description} (score ${range})`);
805
- const example = `{"dimensions": {${dimensions.map((d) => `"${d.key}": <number>`).join(", ")}}, "notes": "<one-line rationale>"}`;
806
- return [
807
- "Score the artifact on EACH of these dimensions:",
808
- ...lines,
809
- "",
810
- `Respond with JSON ONLY, no prose. Every dimension is a number in [${range}]:`,
811
- example
812
- ].join("\n");
813
- }
814
- function parseResponse(name, content) {
815
- const stripped = content.replace(/```json\n?|\n?```/g, "").trim();
816
- const objMatch = stripped.match(/\{[\s\S]*\}/);
817
- const payload = objMatch ? objMatch[0] : stripped;
818
- try {
819
- const parsed = JSON.parse(payload);
820
- if (typeof parsed !== "object" || parsed === null) {
821
- throw new Error("parsed value is not an object");
822
- }
823
- return parsed;
824
- } catch (err) {
825
- throw new JudgeParseError(name, content, { cause: err });
826
- }
827
- }
828
- function firstString(value) {
829
- return typeof value === "string" && value.trim() ? value : void 0;
830
- }
831
-
832
510
  // src/campaign/analyst-surface.ts
833
511
  function surfaceToText(surface) {
834
512
  if (typeof surface === "string") return surface;
@@ -3418,6 +3096,7 @@ var SKILLOPT_SYSTEM = 'You are a SkillOpt optimizer. You improve ONE skill docum
3418
3096
  function skillOptProposer(opts) {
3419
3097
  const evidenceK = opts.evidenceK ?? 3;
3420
3098
  const defaultBudget = opts.editBudget ?? 3;
3099
+ const directCostLedger = opts.costLedger ?? new CostLedger();
3421
3100
  async function proposePatches(args) {
3422
3101
  const userPrompt = buildPatchPrompt({
3423
3102
  target: opts.target,
@@ -3429,19 +3108,29 @@ function skillOptProposer(opts) {
3429
3108
  findingsNote: args.findingsNote,
3430
3109
  count: args.count
3431
3110
  });
3432
- const result = await callLlm(
3433
- {
3434
- model: opts.model,
3435
- messages: [
3436
- { role: "system", content: SKILLOPT_SYSTEM },
3437
- { role: "user", content: userPrompt }
3438
- ],
3439
- jsonMode: true,
3440
- temperature: opts.temperature ?? 0.6,
3441
- maxTokens: opts.maxTokens ?? 4e3
3442
- },
3443
- opts.llm
3444
- );
3111
+ const request = {
3112
+ model: opts.model,
3113
+ messages: [
3114
+ { role: "system", content: SKILLOPT_SYSTEM },
3115
+ { role: "user", content: userPrompt }
3116
+ ],
3117
+ jsonMode: true,
3118
+ temperature: opts.temperature ?? 0.6,
3119
+ maxTokens: opts.maxTokens ?? 4e3
3120
+ };
3121
+ const paid = await (args.costLedger ?? directCostLedger).runPaidCall({
3122
+ channel: "driver",
3123
+ phase: args.costPhase ?? "search.proposal",
3124
+ actor: "skill-opt.propose",
3125
+ model: opts.model,
3126
+ maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
3127
+ signal: args.signal,
3128
+ execute: (signal, callId) => callLlm(request, { ...opts.llm, signal, idempotencyKey: callId }),
3129
+ receipt: costReceiptFromLlm,
3130
+ receiptFromError: costReceiptFromLlmError
3131
+ });
3132
+ if (!paid.succeeded) throw paid.error;
3133
+ const result = paid.value;
3445
3134
  return parseSkillPatchResponse(result.content, args.count, args.editBudget);
3446
3135
  }
3447
3136
  return {
@@ -3461,7 +3150,9 @@ function skillOptProposer(opts) {
3461
3150
  rejectedBuffer: [],
3462
3151
  findingsNote: renderAnalystEvidence(ctx.findings, ctx.report) ?? void 0,
3463
3152
  count: ctx.populationSize,
3464
- signal: ctx.signal
3153
+ signal: ctx.signal,
3154
+ costLedger: ctx.costLedger ?? directCostLedger,
3155
+ costPhase: ctx.costPhase ?? "search.proposal"
3465
3156
  });
3466
3157
  const out = [];
3467
3158
  const seen = /* @__PURE__ */ new Set();
@@ -3625,16 +3316,20 @@ async function runSkillOpt(opts) {
3625
3316
  const budgetAnneal = opts.budgetAnneal ?? true;
3626
3317
  const rejectedBufferSize = opts.rejectedBufferSize ?? 12;
3627
3318
  const slowMetaEvery = opts.slowMetaEvery ?? 2;
3628
- let totalCostUsd = 0;
3319
+ opts.runDir = resolveRunDir(opts.runDir, opts.repo);
3320
+ const storage = opts.storage ?? fsCampaignStorage();
3321
+ const costLedger = opts.costLedger ?? createRunCostLedger({
3322
+ storage,
3323
+ runDir: opts.runDir,
3324
+ costCeilingUsd: opts.costCeiling
3325
+ });
3629
3326
  const scoreHoldout = async (surface, tag) => {
3630
- const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag);
3631
- totalCostUsd += campaign.aggregates.totalCostUsd;
3327
+ const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag, costLedger);
3632
3328
  return campaignMeanComposite(campaign);
3633
3329
  };
3634
3330
  const evidenceK = opts.evidenceK ?? 3;
3635
3331
  const trainEvidence = async (surface, tag) => {
3636
- const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag);
3637
- totalCostUsd += campaign.aggregates.totalCostUsd;
3332
+ const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag, costLedger);
3638
3333
  return toEvidence(campaign, evidenceK);
3639
3334
  };
3640
3335
  let current = opts.baselineSurface;
@@ -3658,7 +3353,9 @@ async function runSkillOpt(opts) {
3658
3353
  rejectedBuffer: buffer,
3659
3354
  metaNote,
3660
3355
  count: patchesPerEpoch,
3661
- signal: opts.signal ?? new AbortController().signal
3356
+ signal: opts.signal ?? new AbortController().signal,
3357
+ costLedger,
3358
+ costPhase: "skill-opt.proposal"
3662
3359
  });
3663
3360
  let accepted = null;
3664
3361
  const rejectedThisEpoch = [];
@@ -3717,6 +3414,7 @@ async function runSkillOpt(opts) {
3717
3414
  });
3718
3415
  if (sinceAccept >= patience) break;
3719
3416
  }
3417
+ const cost = costLedger.summary();
3720
3418
  return {
3721
3419
  winnerSurface: current,
3722
3420
  baselineHoldoutComposite: baselineHoldout,
@@ -3726,12 +3424,14 @@ async function runSkillOpt(opts) {
3726
3424
  rejectedEdits: rejectedAll,
3727
3425
  epochsRun,
3728
3426
  history,
3729
- totalCostUsd
3427
+ totalCostUsd: cost.totalCostUsd,
3428
+ cost
3730
3429
  };
3731
3430
  }
3732
- function runScoringCampaign(opts, scenarios, surface, tag) {
3431
+ function runScoringCampaign(opts, scenarios, surface, tag, costLedger) {
3733
3432
  return runCampaign({
3734
3433
  ...opts,
3434
+ costLedger,
3735
3435
  scenarios,
3736
3436
  dispatch: (scenario, ctx) => opts.dispatchWithSurface(surface, scenario, ctx),
3737
3437
  runDir: `${opts.runDir}/${tag}`
@@ -3903,11 +3603,11 @@ function gepaEntry(config, combineParents, name) {
3903
3603
  ...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
3904
3604
  ...config.report !== void 0 ? { report: config.report } : {}
3905
3605
  });
3906
- const costUsd = result.baselineCampaign.aggregates.totalCostUsd + result.generations.reduce(
3907
- (sum, g) => sum + g.surfaces.reduce((s, sf) => s + sf.campaign.aggregates.totalCostUsd, 0),
3908
- 0
3909
- );
3910
- return { winnerSurface: result.winnerSurface, costUsd, durationMs: Date.now() - started };
3606
+ return {
3607
+ winnerSurface: result.winnerSurface,
3608
+ costUsd: result.cost.totalCostUsd,
3609
+ durationMs: Date.now() - started
3610
+ };
3911
3611
  }
3912
3612
  };
3913
3613
  }
@@ -3980,11 +3680,11 @@ function fapoEscalationEntry(config, name = "fapo-escalation") {
3980
3680
  ...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
3981
3681
  ...config.report !== void 0 ? { report: config.report } : {}
3982
3682
  });
3983
- const costUsd = result.baselineCampaign.aggregates.totalCostUsd + result.generations.reduce(
3984
- (sum, g) => sum + g.surfaces.reduce((s, sf) => s + sf.campaign.aggregates.totalCostUsd, 0),
3985
- 0
3986
- );
3987
- return { winnerSurface: result.winnerSurface, costUsd, durationMs: Date.now() - started };
3683
+ return {
3684
+ winnerSurface: result.winnerSurface,
3685
+ costUsd: result.cost.totalCostUsd,
3686
+ durationMs: Date.now() - started
3687
+ };
3988
3688
  }
3989
3689
  };
3990
3690
  }
@@ -4218,6 +3918,7 @@ function createLlmCorrectnessChecker(tc, opts = {}) {
4218
3918
  const model = opts.model ?? "claude-sonnet-4-6";
4219
3919
  const maxContentChars = opts.maxContentChars ?? 8e3;
4220
3920
  const maxAttempts = opts.maxAttempts ?? 2;
3921
+ const costLedger = opts.costLedger ?? new CostLedger();
4221
3922
  const sink = opts.rawSink;
4222
3923
  const record = async (event) => {
4223
3924
  try {
@@ -4261,7 +3962,24 @@ ${content.slice(0, maxContentChars)}`
4261
3962
  redactedFields: []
4262
3963
  });
4263
3964
  try {
4264
- const resp = await tc.chat(request);
3965
+ const paid = await costLedger.runPaidCall({
3966
+ channel: "verifier",
3967
+ phase: opts.costPhase ?? "completion.correctness",
3968
+ actor: "correctness-checker",
3969
+ model,
3970
+ maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
3971
+ tags: {
3972
+ ...opts.costTags,
3973
+ requirementId: requirement.reqId,
3974
+ attempt: String(attempt)
3975
+ },
3976
+ signal: opts.signal,
3977
+ execute: () => tc.chat(request),
3978
+ receipt: (response) => costReceiptFromTCloud(response, model),
3979
+ receiptFromError: (error) => opts.receiptFromError?.(error, attempt)
3980
+ });
3981
+ if (!paid.succeeded) throw paid.error;
3982
+ const resp = paid.value;
4265
3983
  const raw = resp.choices?.[0]?.message?.content ?? "";
4266
3984
  await record({
4267
3985
  eventId: randomUUID(),
@@ -4777,12 +4495,12 @@ function requireResolvedModel(cell, profileId) {
4777
4495
  const resolved = cell.resolvedModel?.trim();
4778
4496
  if (!resolved) {
4779
4497
  throw new ProfileMatrixError(
4780
- `profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 report it via ctx.cost.observeModel(<id>) so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
4498
+ `profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 return it in the ctx.cost.runPaidCall receipt so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
4781
4499
  );
4782
4500
  }
4783
4501
  if (!modelHasSnapshot(resolved)) {
4784
4502
  throw new ProfileMatrixError(
4785
- `profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) before reporting it via ctx.cost.observeModel`
4503
+ `profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) in the paid-call receipt`
4786
4504
  );
4787
4505
  }
4788
4506
  return resolved;
@@ -4808,14 +4526,9 @@ function buildRunRecord(args) {
4808
4526
  }
4809
4527
  const perDimMean = {};
4810
4528
  for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean3(values);
4811
- let costUsd = cell.costUsd;
4812
- let costEstimated = false;
4813
- if (costUsd === 0 && cell.tokenUsage.output > 0 && isModelPriced(model)) {
4814
- costUsd = estimateCost(cell.tokenUsage.input, cell.tokenUsage.output, model);
4815
- costEstimated = costUsd > 0;
4816
- }
4529
+ const costUsd = cell.costUsd;
4817
4530
  raw.cost_usd = costUsd;
4818
- raw.cost_estimated = costEstimated ? 1 : 0;
4531
+ raw.cost_estimated = cell.costEstimated ? 1 : 0;
4819
4532
  raw.tokens_input = cell.tokenUsage.input;
4820
4533
  raw.tokens_output = cell.tokenUsage.output;
4821
4534
  if (typeof cell.tokenUsage.cached === "number") raw.tokens_cached = cell.tokenUsage.cached;
@@ -4958,11 +4671,8 @@ async function runProfileMatrix(opts) {
4958
4671
  profileRecords.push(record);
4959
4672
  records.push(record);
4960
4673
  }
4961
- const pricedTotalCostUsd = profileRecords.reduce((a, r) => a + r.costUsd, 0);
4962
- campaigns[profileId] = {
4963
- ...campaign,
4964
- aggregates: { ...campaign.aggregates, totalCostUsd: pricedTotalCostUsd }
4965
- };
4674
+ const totalCostUsd = campaign.aggregates.totalCostUsd;
4675
+ campaigns[profileId] = campaign;
4966
4676
  byProfile[profileId] = {
4967
4677
  profileId,
4968
4678
  profileHash,
@@ -4972,7 +4682,7 @@ async function runProfileMatrix(opts) {
4972
4682
  model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
4973
4683
  records: profileRecords.length,
4974
4684
  meanComposite: mean3(profileRecords.map(compositeOf)),
4975
- totalCostUsd: pricedTotalCostUsd,
4685
+ totalCostUsd,
4976
4686
  integrity: summarizeBackendIntegrity(profileRecords)
4977
4687
  };
4978
4688
  }
@@ -5189,36 +4899,78 @@ function surfaceToPromptText(surface) {
5189
4899
  return typeof surface === "string" ? surface : JSON.stringify(surface);
5190
4900
  }
5191
4901
  function analysisEditProposer(opts) {
4902
+ const directCostLedger = opts.costLedger ?? new CostLedger();
5192
4903
  return {
5193
4904
  kind: opts.kind,
5194
4905
  async propose(ctx) {
5195
4906
  const parent = surfaceToPromptText(ctx.currentSurface);
4907
+ const costLedger = ctx.costLedger ?? directCostLedger;
4908
+ const phase = ctx.costPhase ?? "search.proposal";
5196
4909
  const traces = await opts.resolveTraces(ctx) ?? "";
5197
4910
  if (!traces.trim()) throw new Error(opts.noTracesError);
5198
4911
  const dir = mkdtempSync(join4(tmpdir(), `${opts.kind}-proposer-`));
5199
4912
  const tracePath = join4(dir, "traces.jsonl");
5200
4913
  writeFileSync2(tracePath, traces.endsWith("\n") ? traces : `${traces}
5201
4914
  `);
5202
- const report = await opts.analyze(tracePath, ctx);
5203
- const applied = await callLlm(
5204
- {
5205
- model: opts.applyModel,
5206
- messages: [
5207
- { role: "system", content: APPLY_SYSTEM },
5208
- {
5209
- role: "user",
5210
- content: `CURRENT PROMPT:
4915
+ if (costLedger.costCeilingUsd !== void 0 && !opts.analysisReceipt) {
4916
+ throw new CostAccountingIncompleteError(
4917
+ `${opts.kind}: capped analysis requires analysisReceipt before external execution`
4918
+ );
4919
+ }
4920
+ const analysis = await costLedger.runPaidCall({
4921
+ channel: "analyst",
4922
+ phase,
4923
+ actor: `${opts.kind}.analyze`,
4924
+ model: opts.analysisModel,
4925
+ maximumCharge: opts.analysisMaximumCharge,
4926
+ tags: { generation: String(ctx.generation) },
4927
+ signal: ctx.signal,
4928
+ execute: (signal) => opts.analyze(tracePath, { ...ctx, signal }),
4929
+ receipt: (report2) => opts.analysisReceipt?.(report2) ?? {
4930
+ model: opts.analysisModel,
4931
+ inputTokens: 0,
4932
+ outputTokens: 0,
4933
+ costUnknown: true
4934
+ }
4935
+ });
4936
+ if (!analysis.succeeded) throw analysis.error;
4937
+ const report = analysis.value;
4938
+ const request = {
4939
+ model: opts.applyModel,
4940
+ messages: [
4941
+ { role: "system", content: APPLY_SYSTEM },
4942
+ {
4943
+ role: "user",
4944
+ content: `CURRENT PROMPT:
5211
4945
  ${parent}
5212
4946
 
5213
4947
  TRACE-ANALYSIS REPORT:
5214
4948
  ${report}
5215
4949
 
5216
4950
  Return the full revised prompt.`
5217
- }
5218
- ]
5219
- },
5220
- { baseUrl: opts.baseUrl, apiKey: opts.apiKey, fetch: opts.fetchImpl }
5221
- );
4951
+ }
4952
+ ],
4953
+ maxTokens: opts.applyMaxTokens ?? 6e3
4954
+ };
4955
+ const llm = {
4956
+ baseUrl: opts.baseUrl,
4957
+ apiKey: opts.apiKey,
4958
+ fetch: opts.fetchImpl
4959
+ };
4960
+ const apply = await costLedger.runPaidCall({
4961
+ channel: "driver",
4962
+ phase,
4963
+ actor: `${opts.kind}.apply`,
4964
+ model: opts.applyModel,
4965
+ maximumCharge: maximumChargeForLlmRequest(request, llm),
4966
+ tags: { generation: String(ctx.generation) },
4967
+ signal: ctx.signal,
4968
+ execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
4969
+ receipt: costReceiptFromLlm,
4970
+ receiptFromError: costReceiptFromLlmError
4971
+ });
4972
+ if (!apply.succeeded) throw apply.error;
4973
+ const applied = apply.value;
5222
4974
  const text = applied.content.trim();
5223
4975
  if (!text || text === parent) return [];
5224
4976
  return [{ surface: text, label: opts.label, rationale: opts.rationale(report) }];
@@ -5237,7 +4989,12 @@ function haloProposer(opts) {
5237
4989
  label: "halo",
5238
4990
  baseUrl: opts.baseUrl,
5239
4991
  apiKey: opts.apiKey,
4992
+ analysisModel: model,
5240
4993
  applyModel: opts.applyModel ?? model,
4994
+ costLedger: opts.costLedger,
4995
+ analysisMaximumCharge: opts.analysisMaximumCharge,
4996
+ analysisReceipt: opts.analysisReceipt,
4997
+ applyMaxTokens: opts.applyMaxTokens,
5241
4998
  fetchImpl: opts.fetchImpl,
5242
4999
  resolveTraces: opts.resolveTraces,
5243
5000
  noTracesError: "haloProposer: resolveTraces returned no OTLP traces \u2014 the halo engine has nothing to analyze",
@@ -5669,6 +5426,7 @@ function llmPolicyEditProposer(opts) {
5669
5426
  opts.maxAuthorContextChars ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.authorContextChars,
5670
5427
  "maxAuthorContextChars"
5671
5428
  );
5429
+ const directCostLedger = opts.costLedger ?? new CostLedger();
5672
5430
  return {
5673
5431
  kind: "llm-policy-edit",
5674
5432
  async propose(ctx) {
@@ -5731,26 +5489,34 @@ function llmPolicyEditProposer(opts) {
5731
5489
  maxAuthorContextChars
5732
5490
  );
5733
5491
  const userContent = JSON.stringify(authorContext);
5734
- const { value } = await callLlmJson(
5735
- {
5736
- model: opts.model,
5737
- messages: [
5738
- { role: "system", content: POLICY_EDIT_AUTHOR_SYSTEM },
5739
- {
5740
- role: "user",
5741
- content: userContent
5742
- }
5743
- ],
5744
- jsonSchema: {
5745
- name: "policy_edit_author",
5746
- schema: responseSchema
5747
- },
5748
- temperature: opts.temperature ?? 0.2,
5749
- maxTokens: opts.maxTokens ?? 6e3,
5750
- timeoutMs: opts.timeoutMs
5492
+ const request = {
5493
+ model: opts.model,
5494
+ messages: [
5495
+ { role: "system", content: POLICY_EDIT_AUTHOR_SYSTEM },
5496
+ { role: "user", content: userContent }
5497
+ ],
5498
+ jsonSchema: {
5499
+ name: "policy_edit_author",
5500
+ schema: responseSchema
5751
5501
  },
5752
- { ...opts.llm, signal: ctx.signal }
5753
- );
5502
+ temperature: opts.temperature ?? 0.2,
5503
+ maxTokens: opts.maxTokens ?? 6e3,
5504
+ timeoutMs: opts.timeoutMs
5505
+ };
5506
+ const paid = await (ctx.costLedger ?? directCostLedger).runPaidCall({
5507
+ channel: "driver",
5508
+ phase: ctx.costPhase ?? "search.proposal",
5509
+ actor: "llm-policy-edit.author",
5510
+ model: opts.model,
5511
+ maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
5512
+ tags: { generation: String(ctx.generation) },
5513
+ signal: ctx.signal,
5514
+ execute: (signal, callId) => callLlmJson(request, { ...opts.llm, signal, idempotencyKey: callId }),
5515
+ receipt: ({ result }) => costReceiptFromLlm(result),
5516
+ receiptFromError: costReceiptFromLlmError
5517
+ });
5518
+ if (!paid.succeeded) throw paid.error;
5519
+ const { value } = paid.value;
5754
5520
  const response = parseAuthorResponse(value);
5755
5521
  if (response.edits.length > limit) {
5756
5522
  throw new Error(
@@ -6432,18 +6198,35 @@ var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, gener
6432
6198
  function extractExistingLessons(text) {
6433
6199
  return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
6434
6200
  }
6435
- async function distillLessons(raw, distill) {
6436
- const res = await callLlm(
6437
- {
6438
- model: distill.model,
6439
- messages: [
6440
- { role: "system", content: DISTILL_SYSTEM },
6441
- { role: "user", content: `Findings:
6201
+ async function distillLessons(raw, distill, ctx, costLedger) {
6202
+ const request = {
6203
+ model: distill.model,
6204
+ messages: [
6205
+ { role: "system", content: DISTILL_SYSTEM },
6206
+ { role: "user", content: `Findings:
6442
6207
  ${raw.map((r) => `- ${r}`).join("\n")}` }
6443
- ]
6444
- },
6445
- { baseUrl: distill.baseUrl, apiKey: distill.apiKey, fetch: distill.fetchImpl }
6446
- );
6208
+ ],
6209
+ maxTokens: distill.maxTokens ?? 2e3
6210
+ };
6211
+ const llm = {
6212
+ baseUrl: distill.baseUrl,
6213
+ apiKey: distill.apiKey,
6214
+ fetch: distill.fetchImpl
6215
+ };
6216
+ const paid = await costLedger.runPaidCall({
6217
+ channel: "driver",
6218
+ phase: ctx.costPhase ?? "search.proposal",
6219
+ actor: "memory-curation.distill",
6220
+ model: distill.model,
6221
+ maximumCharge: maximumChargeForLlmRequest(request, llm),
6222
+ tags: { generation: String(ctx.generation) },
6223
+ signal: ctx.signal,
6224
+ execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
6225
+ receipt: costReceiptFromLlm,
6226
+ receiptFromError: costReceiptFromLlmError
6227
+ });
6228
+ if (!paid.succeeded) throw paid.error;
6229
+ const res = paid.value;
6447
6230
  try {
6448
6231
  const parsed = JSON.parse(res.content.trim());
6449
6232
  if (Array.isArray(parsed)) {
@@ -6459,6 +6242,7 @@ ${raw.map((r) => `- ${r}`).join("\n")}` }
6459
6242
  function memoryCurationProposer(opts = {}) {
6460
6243
  const maxEntries = opts.maxEntries ?? 12;
6461
6244
  const heading = opts.sectionHeading ?? DEFAULT_HEADING2;
6245
+ const directCostLedger = opts.costLedger ?? new CostLedger();
6462
6246
  return {
6463
6247
  kind: "memory-curation",
6464
6248
  async propose(ctx) {
@@ -6470,7 +6254,7 @@ function memoryCurationProposer(opts = {}) {
6470
6254
  }
6471
6255
  const carried = extractExistingLessons(parent);
6472
6256
  if (fresh.length === 0 && carried.length === 0) return [];
6473
- const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill) : fresh;
6257
+ const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill, ctx, ctx.costLedger ?? directCostLedger) : fresh;
6474
6258
  const byKey = /* @__PURE__ */ new Map();
6475
6259
  for (const l of carried) {
6476
6260
  const k = normKey(l);
@@ -6541,7 +6325,12 @@ function traceAnalystProposer(opts) {
6541
6325
  label: "trace-analyst",
6542
6326
  baseUrl: opts.baseUrl,
6543
6327
  apiKey: opts.apiKey,
6328
+ analysisModel: opts.model,
6544
6329
  applyModel: opts.applyModel ?? opts.model,
6330
+ costLedger: opts.costLedger,
6331
+ analysisMaximumCharge: opts.analysisMaximumCharge,
6332
+ analysisReceipt: opts.analysisReceipt,
6333
+ applyMaxTokens: opts.applyMaxTokens,
6545
6334
  fetchImpl: opts.fetchImpl,
6546
6335
  resolveTraces: opts.resolveTraces,
6547
6336
  noTracesError: "traceAnalystProposer: resolveTraces returned no OTLP traces \u2014 the analyst has nothing to read",
@@ -6613,238 +6402,86 @@ function selectDiscriminative(signals, k, opts) {
6613
6402
 
6614
6403
  // src/campaign/search-ledger.ts
6615
6404
  import { createHash as createHash7 } from "crypto";
6616
- import { existsSync as existsSync4, readFileSync as readFileSync4 } from "fs";
6405
+ import { existsSync as existsSync3, readFileSync as readFileSync3 } from "fs";
6617
6406
  import { resolve as resolve2 } from "path";
6618
- import { z as z3 } from "zod";
6619
-
6620
- // src/campaign/search-ledger-errors.ts
6621
- var SearchLedgerError = class extends ValidationError {
6622
- };
6623
- var SearchLedgerIntegrityError = class extends SearchLedgerError {
6624
- };
6625
- var SearchLedgerConflictError = class extends SearchLedgerError {
6626
- };
6627
-
6628
- // src/campaign/search-ledger-file.ts
6629
- import { randomUUID as randomUUID2 } from "crypto";
6630
- import {
6631
- closeSync,
6632
- constants,
6633
- existsSync as existsSync3,
6634
- fsyncSync,
6635
- linkSync,
6636
- mkdirSync as mkdirSync2,
6637
- openSync,
6638
- readFileSync as readFileSync3,
6639
- renameSync,
6640
- unlinkSync,
6641
- writeSync
6642
- } from "fs";
6643
- import { hostname } from "os";
6644
- import { dirname as dirname2 } from "path";
6645
6407
  import { z as z2 } from "zod";
6646
- function appendSearchLedgerLine(path, line) {
6647
- mkdirSync2(dirname2(path), { recursive: true });
6648
- const fd = openSync(path, constants.O_CREAT | constants.O_WRONLY | constants.O_APPEND, 384);
6649
- try {
6650
- writeAll(fd, Buffer.from(line, "utf8"));
6651
- fsyncSync(fd);
6652
- } finally {
6653
- closeSync(fd);
6654
- }
6655
- fsyncDirectory(dirname2(path));
6656
- }
6657
- function withSearchLedgerFileLock(ledgerPath, run) {
6658
- mkdirSync2(dirname2(ledgerPath), { recursive: true });
6659
- const lockPath = `${ledgerPath}.lock`;
6660
- const owner = acquireLock(lockPath);
6661
- try {
6662
- return run();
6663
- } finally {
6664
- releaseLock(lockPath, owner);
6665
- }
6666
- }
6667
- function writeAll(fd, bytes) {
6668
- let offset = 0;
6669
- while (offset < bytes.byteLength) {
6670
- const written = writeSync(fd, bytes, offset, bytes.byteLength - offset);
6671
- if (written <= 0) throw new SearchLedgerIntegrityError("filesystem wrote zero bytes");
6672
- offset += written;
6673
- }
6674
- }
6675
- function fsyncDirectory(path) {
6676
- const fd = openSync(path, constants.O_RDONLY);
6677
- try {
6678
- fsyncSync(fd);
6679
- } finally {
6680
- closeSync(fd);
6681
- }
6682
- }
6683
- function acquireLock(lockPath) {
6684
- const owner = { pid: process.pid, host: hostname(), nonce: randomUUID2() };
6685
- const ownerBytes = `${canonicalOwner(owner)}
6686
- `;
6687
- for (let attempt = 0; attempt < 8; attempt += 1) {
6688
- const ownerPath = `${lockPath}.${owner.pid}.${owner.nonce}.${attempt}.owner`;
6689
- const fd = openSync(ownerPath, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY, 384);
6690
- try {
6691
- writeAll(fd, Buffer.from(ownerBytes, "utf8"));
6692
- fsyncSync(fd);
6693
- } finally {
6694
- closeSync(fd);
6695
- }
6696
- try {
6697
- linkSync(ownerPath, lockPath);
6698
- unlinkSync(ownerPath);
6699
- return owner;
6700
- } catch (error) {
6701
- unlinkIfExists(ownerPath);
6702
- if (error.code !== "EEXIST") throw error;
6703
- }
6704
- const holder = readOwner(lockPath);
6705
- if (holder.host !== owner.host || isProcessAlive(holder.pid)) {
6706
- throw new SearchLedgerIntegrityError(
6707
- `search ledger lock is held by pid ${holder.pid} on ${holder.host}`
6708
- );
6709
- }
6710
- const tombstone = `${lockPath}.stale.${owner.nonce}.${attempt}`;
6711
- try {
6712
- renameSync(lockPath, tombstone);
6713
- unlinkSync(tombstone);
6714
- } catch (error) {
6715
- if (error.code !== "ENOENT") throw error;
6716
- }
6717
- }
6718
- throw new SearchLedgerIntegrityError(`could not acquire search ledger lock ${lockPath}`);
6719
- }
6720
- function releaseLock(lockPath, owner) {
6721
- if (!existsSync3(lockPath)) return;
6722
- const holder = readOwner(lockPath);
6723
- if (canonicalOwner(holder) !== canonicalOwner(owner)) {
6724
- throw new SearchLedgerIntegrityError(
6725
- `search ledger lock owner changed before release (${lockPath})`
6726
- );
6727
- }
6728
- unlinkSync(lockPath);
6729
- }
6730
- function readOwner(lockPath) {
6731
- let raw;
6732
- try {
6733
- raw = JSON.parse(readFileSync3(lockPath, "utf8"));
6734
- } catch (error) {
6735
- throw new SearchLedgerIntegrityError(`search ledger lock ${lockPath} is malformed`, {
6736
- cause: error
6737
- });
6738
- }
6739
- const parsed = z2.object({
6740
- pid: z2.number().int().positive().safe(),
6741
- host: z2.string().trim().min(1),
6742
- nonce: z2.string().trim().min(1)
6743
- }).strict().safeParse(raw);
6744
- if (!parsed.success) {
6745
- throw new SearchLedgerIntegrityError(
6746
- `search ledger lock ${lockPath} is malformed: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; ")}`
6747
- );
6748
- }
6749
- return parsed.data;
6750
- }
6751
- function canonicalOwner(owner) {
6752
- return JSON.stringify({ host: owner.host, nonce: owner.nonce, pid: owner.pid });
6753
- }
6754
- function isProcessAlive(pid) {
6755
- try {
6756
- process.kill(pid, 0);
6757
- return true;
6758
- } catch (error) {
6759
- return error.code !== "ESRCH";
6760
- }
6761
- }
6762
- function unlinkIfExists(path) {
6763
- try {
6764
- unlinkSync(path);
6765
- } catch (error) {
6766
- if (error.code !== "ENOENT") throw error;
6767
- }
6768
- }
6769
-
6770
- // src/campaign/search-ledger.ts
6771
6408
  var SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
6772
- var NON_EMPTY = z3.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
6773
- var HASH = z3.string().regex(/^sha256:[a-f0-9]{64}$/);
6774
- var LINEAGE_NODE_ID = z3.string().regex(/^[a-f0-9]{16}$/);
6775
- var IMMUTABLE_REVISION = z3.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
6776
- var ISO_TIMESTAMP = z3.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
6777
- var NON_NEGATIVE_INT = z3.number().int().nonnegative().safe();
6778
- var FINITE_NUMBER = z3.number().finite();
6779
- var ArtifactRefSchema = z3.object({
6409
+ var NON_EMPTY = z2.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
6410
+ var HASH = z2.string().regex(/^sha256:[a-f0-9]{64}$/);
6411
+ var LINEAGE_NODE_ID = z2.string().regex(/^[a-f0-9]{16}$/);
6412
+ var IMMUTABLE_REVISION = z2.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
6413
+ var ISO_TIMESTAMP = z2.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
6414
+ var NON_NEGATIVE_INT = z2.number().int().nonnegative().safe();
6415
+ var FINITE_NUMBER = z2.number().finite();
6416
+ var ArtifactRefSchema = z2.object({
6780
6417
  role: NON_EMPTY,
6781
6418
  uri: NON_EMPTY,
6782
6419
  sha256: HASH,
6783
6420
  byteLength: NON_NEGATIVE_INT
6784
6421
  }).strict();
6785
- var SourceRefSchema = z3.object({
6422
+ var SourceRefSchema = z2.object({
6786
6423
  uri: NON_EMPTY,
6787
6424
  revision: IMMUTABLE_REVISION
6788
6425
  }).strict();
6789
- var FailureReasonSchema = z3.object({
6426
+ var FailureReasonSchema = z2.object({
6790
6427
  code: NON_EMPTY,
6791
6428
  message: NON_EMPTY
6792
6429
  }).strict();
6793
6430
  var EventBaseShape = {
6794
6431
  eventId: NON_EMPTY,
6795
6432
  occurredAt: ISO_TIMESTAMP,
6796
- artifacts: z3.array(ArtifactRefSchema).min(1)
6433
+ artifacts: z2.array(ArtifactRefSchema).min(1)
6797
6434
  };
6798
- var OperationKindSchema = z3.enum([
6435
+ var OperationKindSchema = z2.enum([
6799
6436
  "candidate-generation",
6800
6437
  "analysis",
6801
6438
  "selection",
6802
6439
  "judge",
6803
6440
  "other"
6804
6441
  ]);
6805
- var SearchPlannedSchema = z3.object({
6442
+ var SearchPlannedSchema = z2.object({
6806
6443
  ...EventBaseShape,
6807
- kind: z3.literal("search-planned"),
6808
- plan: z3.object({
6809
- candidateSlots: z3.array(
6810
- z3.object({
6444
+ kind: z2.literal("search-planned"),
6445
+ plan: z2.object({
6446
+ candidateSlots: z2.array(
6447
+ z2.object({
6811
6448
  slotId: NON_EMPTY,
6812
6449
  generationOperationId: NON_EMPTY
6813
6450
  }).strict()
6814
6451
  ).min(1),
6815
- tasks: z3.array(
6816
- z3.object({
6452
+ tasks: z2.array(
6453
+ z2.object({
6817
6454
  taskId: NON_EMPTY,
6818
6455
  source: SourceRefSchema,
6819
6456
  benchmark: SourceRefSchema,
6820
- maxAttempts: z3.number().int().positive().safe()
6457
+ maxAttempts: z2.number().int().positive().safe()
6821
6458
  }).strict()
6822
6459
  ).min(1),
6823
- operations: z3.array(
6824
- z3.object({
6460
+ operations: z2.array(
6461
+ z2.object({
6825
6462
  operationId: NON_EMPTY,
6826
6463
  kind: OperationKindSchema
6827
6464
  }).strict()
6828
6465
  ).min(1)
6829
6466
  }).strict()
6830
6467
  }).strict();
6831
- var CandidateRegisteredSchema = z3.object({
6468
+ var CandidateRegisteredSchema = z2.object({
6832
6469
  ...EventBaseShape,
6833
- kind: z3.literal("candidate-registered"),
6470
+ kind: z2.literal("candidate-registered"),
6834
6471
  slotId: NON_EMPTY,
6835
6472
  generationOperationId: NON_EMPTY,
6836
6473
  candidateId: NON_EMPTY,
6837
- lineage: z3.object({
6474
+ lineage: z2.object({
6838
6475
  lineageNodeId: LINEAGE_NODE_ID,
6839
- parentCandidateIds: z3.array(NON_EMPTY),
6476
+ parentCandidateIds: z2.array(NON_EMPTY),
6840
6477
  generation: NON_NEGATIVE_INT,
6841
6478
  proposer: NON_EMPTY,
6842
6479
  proposerSource: SourceRefSchema
6843
6480
  }).strict(),
6844
- surfaces: z3.array(
6845
- z3.object({
6481
+ surfaces: z2.array(
6482
+ z2.object({
6846
6483
  surfaceId: NON_EMPTY,
6847
- kind: z3.enum([
6484
+ kind: z2.enum([
6848
6485
  "prompt",
6849
6486
  "tool-contract",
6850
6487
  "runtime-config",
@@ -6858,69 +6495,69 @@ var CandidateRegisteredSchema = z3.object({
6858
6495
  }).strict()
6859
6496
  ).min(1)
6860
6497
  }).strict();
6861
- var CandidateSlotClosedSchema = z3.object({
6498
+ var CandidateSlotClosedSchema = z2.object({
6862
6499
  ...EventBaseShape,
6863
- kind: z3.literal("candidate-slot-closed"),
6500
+ kind: z2.literal("candidate-slot-closed"),
6864
6501
  slotId: NON_EMPTY,
6865
6502
  generationOperationId: NON_EMPTY,
6866
6503
  reason: FailureReasonSchema
6867
6504
  }).strict();
6868
- var KnownTokensSchema = z3.object({
6869
- status: z3.literal("known"),
6505
+ var KnownTokensSchema = z2.object({
6506
+ status: z2.literal("known"),
6870
6507
  inputTokens: NON_NEGATIVE_INT,
6871
6508
  outputTokens: NON_NEGATIVE_INT,
6872
6509
  cachedTokens: NON_NEGATIVE_INT
6873
6510
  }).strict();
6874
- var UnknownSchema = z3.object({
6875
- status: z3.literal("unknown"),
6511
+ var UnknownSchema = z2.object({
6512
+ status: z2.literal("unknown"),
6876
6513
  reason: NON_EMPTY
6877
6514
  }).strict();
6878
- var KnownCostSchema = z3.object({
6879
- status: z3.literal("known"),
6880
- usd: z3.number().finite().nonnegative(),
6881
- source: z3.enum(["provider", "pricing-table", "free"])
6515
+ var KnownCostSchema = z2.object({
6516
+ status: z2.literal("known"),
6517
+ usd: z2.number().finite().nonnegative(),
6518
+ source: z2.enum(["provider", "pricing-table", "free"])
6882
6519
  }).strict().superRefine((cost, ctx) => {
6883
6520
  if (cost.source === "free" && cost.usd !== 0) {
6884
6521
  ctx.addIssue({ code: "custom", message: "free cost source must have usd 0" });
6885
6522
  }
6886
6523
  });
6887
- var UnknownCostSchema = z3.object({
6888
- status: z3.literal("unknown"),
6889
- knownLowerBoundUsd: z3.number().finite().nonnegative(),
6524
+ var UnknownCostSchema = z2.object({
6525
+ status: z2.literal("unknown"),
6526
+ knownLowerBoundUsd: z2.number().finite().nonnegative(),
6890
6527
  reason: NON_EMPTY
6891
6528
  }).strict();
6892
- var AccountingSchema = z3.object({
6893
- tokens: z3.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
6894
- cost: z3.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
6529
+ var AccountingSchema = z2.object({
6530
+ tokens: z2.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
6531
+ cost: z2.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
6895
6532
  }).strict();
6896
- var MetricsSchema = z3.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
6533
+ var MetricsSchema = z2.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
6897
6534
  for (const key of Object.keys(metrics)) {
6898
6535
  if (key === "__proto__" || key === "constructor" || key === "prototype") {
6899
6536
  ctx.addIssue({ code: "custom", message: `unsafe metric key ${key}` });
6900
6537
  }
6901
6538
  }
6902
6539
  });
6903
- var OutcomeSchema = z3.discriminatedUnion("status", [
6904
- z3.object({
6905
- status: z3.literal("passed"),
6540
+ var OutcomeSchema = z2.discriminatedUnion("status", [
6541
+ z2.object({
6542
+ status: z2.literal("passed"),
6906
6543
  score: FINITE_NUMBER,
6907
6544
  metrics: MetricsSchema
6908
6545
  }).strict(),
6909
- z3.object({
6910
- status: z3.literal("failed"),
6546
+ z2.object({
6547
+ status: z2.literal("failed"),
6911
6548
  score: FINITE_NUMBER,
6912
6549
  metrics: MetricsSchema,
6913
6550
  failure: FailureReasonSchema
6914
6551
  }).strict(),
6915
- z3.object({
6916
- status: z3.literal("errored"),
6552
+ z2.object({
6553
+ status: z2.literal("errored"),
6917
6554
  metrics: MetricsSchema,
6918
- error: FailureReasonSchema.extend({ retryable: z3.boolean() }).strict()
6555
+ error: FailureReasonSchema.extend({ retryable: z2.boolean() }).strict()
6919
6556
  }).strict()
6920
6557
  ]);
6921
- var EffectSchema = z3.discriminatedUnion("status", [
6922
- z3.object({
6923
- status: z3.literal("measured"),
6558
+ var EffectSchema = z2.discriminatedUnion("status", [
6559
+ z2.object({
6560
+ status: z2.literal("measured"),
6924
6561
  metric: NON_EMPTY,
6925
6562
  baselineValue: FINITE_NUMBER,
6926
6563
  candidateValue: FINITE_NUMBER,
@@ -6932,17 +6569,17 @@ var EffectSchema = z3.discriminatedUnion("status", [
6932
6569
  ctx.addIssue({ code: "custom", message: "delta must equal candidateValue - baselineValue" });
6933
6570
  }
6934
6571
  }),
6935
- z3.object({
6936
- status: z3.literal("not-measured"),
6572
+ z2.object({
6573
+ status: z2.literal("not-measured"),
6937
6574
  reason: NON_EMPTY
6938
6575
  }).strict()
6939
6576
  ]);
6940
- var SurfaceEvidenceSchema = z3.object({
6577
+ var SurfaceEvidenceSchema = z2.object({
6941
6578
  surfaceId: NON_EMPTY,
6942
- fired: z3.boolean(),
6579
+ fired: z2.boolean(),
6943
6580
  firingCount: NON_NEGATIVE_INT,
6944
6581
  effect: EffectSchema,
6945
- evidence: z3.array(ArtifactRefSchema).min(1)
6582
+ evidence: z2.array(ArtifactRefSchema).min(1)
6946
6583
  }).strict().superRefine((evidence, ctx) => {
6947
6584
  if (evidence.fired && evidence.firingCount === 0) {
6948
6585
  ctx.addIssue({ code: "custom", message: "a fired surface must have firingCount >= 1" });
@@ -6960,15 +6597,15 @@ var SurfaceEvidenceSchema = z3.object({
6960
6597
  });
6961
6598
  }
6962
6599
  });
6963
- var TaskAttemptedSchema = z3.object({
6600
+ var TaskAttemptedSchema = z2.object({
6964
6601
  ...EventBaseShape,
6965
- kind: z3.literal("task-attempted"),
6602
+ kind: z2.literal("task-attempted"),
6966
6603
  candidateId: NON_EMPTY,
6967
6604
  runId: NON_EMPTY,
6968
6605
  attemptIndex: NON_NEGATIVE_INT,
6969
- task: z3.object({ taskId: NON_EMPTY, source: SourceRefSchema }).strict(),
6970
- identity: z3.object({
6971
- model: z3.object({
6606
+ task: z2.object({ taskId: NON_EMPTY, source: SourceRefSchema }).strict(),
6607
+ identity: z2.object({
6608
+ model: z2.object({
6972
6609
  provider: NON_EMPTY,
6973
6610
  snapshot: NON_EMPTY.refine(
6974
6611
  modelHasSnapshot,
@@ -6980,17 +6617,17 @@ var TaskAttemptedSchema = z3.object({
6980
6617
  }).strict(),
6981
6618
  outcome: OutcomeSchema,
6982
6619
  accounting: AccountingSchema,
6983
- surfaceEvidence: z3.array(SurfaceEvidenceSchema).min(1)
6620
+ surfaceEvidence: z2.array(SurfaceEvidenceSchema).min(1)
6984
6621
  }).strict();
6985
- var SearchOperationRecordedSchema = z3.object({
6622
+ var SearchOperationRecordedSchema = z2.object({
6986
6623
  ...EventBaseShape,
6987
- kind: z3.literal("search-operation-recorded"),
6624
+ kind: z2.literal("search-operation-recorded"),
6988
6625
  operationId: NON_EMPTY,
6989
6626
  operationKind: OperationKindSchema,
6990
- execution: z3.discriminatedUnion("kind", [
6991
- z3.object({
6992
- kind: z3.literal("model"),
6993
- model: z3.object({
6627
+ execution: z2.discriminatedUnion("kind", [
6628
+ z2.object({
6629
+ kind: z2.literal("model"),
6630
+ model: z2.object({
6994
6631
  provider: NON_EMPTY,
6995
6632
  snapshot: NON_EMPTY.refine(
6996
6633
  modelHasSnapshot,
@@ -6999,51 +6636,51 @@ var SearchOperationRecordedSchema = z3.object({
6999
6636
  }).strict(),
7000
6637
  source: SourceRefSchema
7001
6638
  }).strict(),
7002
- z3.object({
7003
- kind: z3.literal("deterministic"),
6639
+ z2.object({
6640
+ kind: z2.literal("deterministic"),
7004
6641
  source: SourceRefSchema
7005
6642
  }).strict()
7006
6643
  ]),
7007
- outcome: z3.discriminatedUnion("status", [
7008
- z3.object({ status: z3.literal("completed") }).strict(),
7009
- z3.object({
7010
- status: z3.literal("partial"),
6644
+ outcome: z2.discriminatedUnion("status", [
6645
+ z2.object({ status: z2.literal("completed") }).strict(),
6646
+ z2.object({
6647
+ status: z2.literal("partial"),
7011
6648
  failure: FailureReasonSchema
7012
6649
  }).strict(),
7013
- z3.object({
7014
- status: z3.literal("failed"),
6650
+ z2.object({
6651
+ status: z2.literal("failed"),
7015
6652
  failure: FailureReasonSchema
7016
6653
  }).strict()
7017
6654
  ]),
7018
6655
  accounting: AccountingSchema
7019
6656
  }).strict();
7020
- var CandidateDecidedSchema = z3.object({
6657
+ var CandidateDecidedSchema = z2.object({
7021
6658
  ...EventBaseShape,
7022
- kind: z3.literal("candidate-decided"),
6659
+ kind: z2.literal("candidate-decided"),
7023
6660
  candidateId: NON_EMPTY,
7024
- decision: z3.discriminatedUnion("status", [
7025
- z3.object({ status: z3.literal("selected") }).strict(),
7026
- z3.object({
7027
- status: z3.literal("rejected"),
6661
+ decision: z2.discriminatedUnion("status", [
6662
+ z2.object({ status: z2.literal("selected") }).strict(),
6663
+ z2.object({
6664
+ status: z2.literal("rejected"),
7028
6665
  reason: FailureReasonSchema
7029
6666
  }).strict()
7030
6667
  ])
7031
6668
  }).strict();
7032
- var SearchCompletedSchema = z3.object({
6669
+ var SearchCompletedSchema = z2.object({
7033
6670
  ...EventBaseShape,
7034
- kind: z3.literal("search-completed"),
7035
- result: z3.discriminatedUnion("status", [
7036
- z3.object({
7037
- status: z3.literal("selected"),
6671
+ kind: z2.literal("search-completed"),
6672
+ result: z2.discriminatedUnion("status", [
6673
+ z2.object({
6674
+ status: z2.literal("selected"),
7038
6675
  candidateId: NON_EMPTY
7039
6676
  }).strict(),
7040
- z3.object({
7041
- status: z3.literal("all-rejected"),
6677
+ z2.object({
6678
+ status: z2.literal("all-rejected"),
7042
6679
  reason: FailureReasonSchema
7043
6680
  }).strict()
7044
6681
  ])
7045
6682
  }).strict();
7046
- var EventSchema = z3.discriminatedUnion("kind", [
6683
+ var EventSchema = z2.discriminatedUnion("kind", [
7047
6684
  SearchPlannedSchema,
7048
6685
  CandidateRegisteredSchema,
7049
6686
  CandidateSlotClosedSchema,
@@ -7052,11 +6689,11 @@ var EventSchema = z3.discriminatedUnion("kind", [
7052
6689
  CandidateDecidedSchema,
7053
6690
  SearchCompletedSchema
7054
6691
  ]);
7055
- var EntrySchema = z3.object({
7056
- schema: z3.literal(SEARCH_LEDGER_SCHEMA),
6692
+ var EntrySchema = z2.object({
6693
+ schema: z2.literal(SEARCH_LEDGER_SCHEMA),
7057
6694
  campaignId: NON_EMPTY,
7058
6695
  sequence: NON_NEGATIVE_INT,
7059
- previousHash: z3.union([HASH, z3.null()]),
6696
+ previousHash: z2.union([HASH, z2.null()]),
7060
6697
  event: EventSchema,
7061
6698
  entryHash: HASH
7062
6699
  }).strict();
@@ -7133,8 +6770,8 @@ var FileSearchLedger = class {
7133
6770
  }
7134
6771
  };
7135
6772
  function replayFile(path, campaignId) {
7136
- if (!existsSync4(path)) return replayEntries([], campaignId);
7137
- const text = readFileSync4(path, "utf8");
6773
+ if (!existsSync3(path)) return replayEntries([], campaignId);
6774
+ const text = readFileSync3(path, "utf8");
7138
6775
  if (text.length === 0) return replayEntries([], campaignId);
7139
6776
  if (!text.endsWith("\n")) {
7140
6777
  throw new SearchLedgerIntegrityError(
@@ -7730,10 +7367,10 @@ function formatZodError(error) {
7730
7367
  }
7731
7368
 
7732
7369
  // src/campaign/single-run-lock.ts
7733
- import { existsSync as existsSync5, readFileSync as readFileSync5, unlinkSync as unlinkSync2, writeFileSync as writeFileSync3 } from "fs";
7370
+ import { existsSync as existsSync4, readFileSync as readFileSync4, unlinkSync, writeFileSync as writeFileSync3 } from "fs";
7734
7371
  function liveHolder(path) {
7735
- if (!existsSync5(path)) return null;
7736
- const holder = Number(readFileSync5(path, "utf8").trim());
7372
+ if (!existsSync4(path)) return null;
7373
+ const holder = Number(readFileSync4(path, "utf8").trim());
7737
7374
  if (!Number.isFinite(holder) || holder <= 0) return null;
7738
7375
  try {
7739
7376
  process.kill(holder, 0);
@@ -7755,8 +7392,8 @@ function acquireSingleRunLock(opts) {
7755
7392
  writeFileSync3(opts.lockPath, String(pid));
7756
7393
  const release = () => {
7757
7394
  try {
7758
- if (existsSync5(opts.lockPath) && readFileSync5(opts.lockPath, "utf8").trim() === String(pid)) {
7759
- unlinkSync2(opts.lockPath);
7395
+ if (existsSync4(opts.lockPath) && readFileSync4(opts.lockPath, "utf8").trim() === String(pid)) {
7396
+ unlinkSync(opts.lockPath);
7760
7397
  }
7761
7398
  } catch {
7762
7399
  }
@@ -7780,21 +7417,21 @@ function isTransientTransportFailure(message, opts = {}) {
7780
7417
  import { execFileSync } from "child_process";
7781
7418
  import { createHash as createHash8 } from "crypto";
7782
7419
  import {
7783
- closeSync as closeSync2,
7784
- existsSync as existsSync6,
7420
+ closeSync,
7421
+ existsSync as existsSync5,
7785
7422
  constants as fsConstants,
7786
7423
  fstatSync,
7787
7424
  lstatSync,
7788
- mkdirSync as mkdirSync3,
7425
+ mkdirSync as mkdirSync2,
7789
7426
  mkdtempSync as mkdtempSync2,
7790
- openSync as openSync2,
7427
+ openSync,
7791
7428
  readlinkSync,
7792
7429
  readSync,
7793
7430
  realpathSync,
7794
7431
  rmSync
7795
7432
  } from "fs";
7796
7433
  import { devNull, tmpdir as tmpdir2 } from "os";
7797
- import { basename, dirname as dirname3, isAbsolute as isAbsolute2, join as join5, relative as relative2, resolve as resolve3, sep } from "path";
7434
+ import { basename, dirname as dirname2, isAbsolute as isAbsolute2, join as join5, relative as relative2, resolve as resolve3, sep } from "path";
7798
7435
  var MAX_GIT_OUTPUT_BYTES = 256 * 1024 * 1024;
7799
7436
  var FILE_HASH_CHUNK_BYTES = 1024 * 1024;
7800
7437
  var GIT_REPOSITORY_ENV = /* @__PURE__ */ new Set([
@@ -7846,6 +7483,45 @@ function gitBytes(git, args, cwd, env) {
7846
7483
  function gitText(git, args, cwd, env) {
7847
7484
  return gitBytes(git, args, cwd, env).toString("utf8").trim();
7848
7485
  }
7486
+ function hasRegisteredWorktree(git, repoRoot, path) {
7487
+ const expected = Buffer.from(`worktree ${resolve3(path)}`, "utf8");
7488
+ const records = gitBytes(git, ["worktree", "list", "--porcelain", "-z"], repoRoot);
7489
+ let start = 0;
7490
+ while (start < records.length) {
7491
+ const end = records.indexOf(0, start);
7492
+ if (end < 0) {
7493
+ throw new WorktreeAdapterError("Git worktree list output was not NUL-terminated");
7494
+ }
7495
+ if (records.subarray(start, end).equals(expected)) return true;
7496
+ start = end + 1;
7497
+ }
7498
+ return false;
7499
+ }
7500
+ function hasLocalBranch(git, repoRoot, branch) {
7501
+ const ref = `refs/heads/${branch}`;
7502
+ return gitText(git, ["for-each-ref", "--format=%(refname)", "--", ref], repoRoot).split("\n").some((candidate) => candidate === ref);
7503
+ }
7504
+ function reconcileAbsent(exists, remove) {
7505
+ try {
7506
+ if (!exists()) return void 0;
7507
+ } catch (err) {
7508
+ return err;
7509
+ }
7510
+ try {
7511
+ remove();
7512
+ return void 0;
7513
+ } catch (removeError) {
7514
+ try {
7515
+ if (!exists()) return void 0;
7516
+ } catch (recheckError) {
7517
+ return new AggregateError(
7518
+ [removeError, recheckError],
7519
+ "Removal failed and the resulting resource state could not be checked"
7520
+ );
7521
+ }
7522
+ return removeError;
7523
+ }
7524
+ }
7849
7525
  function sha2562(bytes) {
7850
7526
  return `sha256:${createHash8("sha256").update(bytes).digest("hex")}`;
7851
7527
  }
@@ -7887,7 +7563,7 @@ function patchBytes(git, cwd, baseCommit, candidateCommit) {
7887
7563
  const scratch = mkdtempSync2(join5(tmpdir2(), "agent-eval-patch-"));
7888
7564
  const bareRepo = join5(scratch, "repo.git");
7889
7565
  const emptyTemplate = join5(scratch, "empty-template");
7890
- mkdirSync3(emptyTemplate);
7566
+ mkdirSync2(emptyTemplate);
7891
7567
  try {
7892
7568
  const objectFormat = gitObjectHashAlgorithm(candidateCommit);
7893
7569
  const sourceObjects = realpathSync(gitText(git, ["rev-parse", "--git-path", "objects"], cwd));
@@ -8025,7 +7701,7 @@ function hashGitBlobFile(path, objectId) {
8025
7701
  throw new WorktreeAdapterError(`CodeSurface expected a regular file at ${displayGitPath(path)}`);
8026
7702
  }
8027
7703
  const noFollow = process.platform === "win32" ? 0 : fsConstants.O_NOFOLLOW;
8028
- const fd = openSync2(path, fsConstants.O_RDONLY | noFollow);
7704
+ const fd = openSync(path, fsConstants.O_RDONLY | noFollow);
8029
7705
  try {
8030
7706
  const opened = fstatSync(fd);
8031
7707
  if (!opened.isFile() || opened.dev !== before.dev || opened.ino !== before.ino || opened.size !== before.size) {
@@ -8050,7 +7726,7 @@ function hashGitBlobFile(path, objectId) {
8050
7726
  }
8051
7727
  return { hash: hash.digest("hex"), executable: (opened.mode & 73) !== 0 };
8052
7728
  } finally {
8053
- closeSync2(fd);
7729
+ closeSync(fd);
8054
7730
  }
8055
7731
  }
8056
7732
  function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
@@ -8066,7 +7742,7 @@ function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
8066
7742
  `CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
8067
7743
  );
8068
7744
  }
8069
- const lexicalTarget = resolve3(dirname3(linkPath), target);
7745
+ const lexicalTarget = resolve3(dirname2(linkPath), target);
8070
7746
  if (!isWithinRoot(root, lexicalTarget)) {
8071
7747
  throw new WorktreeAdapterError(
8072
7748
  `CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
@@ -8143,7 +7819,7 @@ function assertRawTreeMatchesWorktree(git, root, candidateCommit) {
8143
7819
  }
8144
7820
  function verifyCodeSurfaceWithGit(surface, path, git) {
8145
7821
  assertCodeSurfaceIdentity(surface);
8146
- if (!existsSync6(path)) {
7822
+ if (!existsSync5(path)) {
8147
7823
  throw new WorktreeAdapterError(`CodeSurface worktree does not exist: ${path}`);
8148
7824
  }
8149
7825
  const lexicalRoot = resolve3(path);
@@ -8310,8 +7986,23 @@ function gitWorktreeAdapter(opts) {
8310
7986
  return surface;
8311
7987
  },
8312
7988
  async discard(worktree) {
8313
- gitText(git, ["worktree", "remove", "--force", worktree.path], opts.repoRoot);
8314
- gitText(git, ["branch", "-D", worktree.branch], opts.repoRoot);
7989
+ const failures = [
7990
+ reconcileAbsent(
7991
+ () => hasRegisteredWorktree(git, opts.repoRoot, worktree.path),
7992
+ () => gitText(git, ["worktree", "remove", "--force", "--", worktree.path], opts.repoRoot)
7993
+ ),
7994
+ reconcileAbsent(
7995
+ () => hasLocalBranch(git, opts.repoRoot, worktree.branch),
7996
+ () => gitText(git, ["branch", "-D", "--", worktree.branch], opts.repoRoot)
7997
+ )
7998
+ ].filter((failure) => failure !== void 0);
7999
+ if (failures.length > 0) {
8000
+ const cause = failures.length === 1 ? failures[0] : new AggregateError(failures, "Multiple Git resources could not be removed");
8001
+ throw new WorktreeAdapterError(
8002
+ `Failed to discard worktree ${worktree.path} and branch ${worktree.branch}`,
8003
+ cause
8004
+ );
8005
+ }
8315
8006
  }
8316
8007
  };
8317
8008
  }
@@ -8324,13 +8015,6 @@ function resolveWorktreePath(surface, worktreeDir) {
8324
8015
  }
8325
8016
 
8326
8017
  export {
8327
- JudgeParseError,
8328
- createDomainExpertJudge,
8329
- codeExecutionJudge,
8330
- coherenceJudge,
8331
- adversarialJudge,
8332
- createCustomJudge,
8333
- defaultJudges,
8334
8018
  pairArms,
8335
8019
  comparePairedArms,
8336
8020
  completionVerdict,
@@ -8338,7 +8022,6 @@ export {
8338
8022
  parseCorrectnessResponse,
8339
8023
  createLlmCorrectnessChecker,
8340
8024
  createTokenRecallChecker,
8341
- llmJudge,
8342
8025
  extractProducedState,
8343
8026
  CODING_HARNESSES,
8344
8027
  HARNESS_NATIVE_MODEL,
@@ -8404,9 +8087,6 @@ export {
8404
8087
  traceAnalystProposer,
8405
8088
  scoreDiscrimination,
8406
8089
  selectDiscriminative,
8407
- SearchLedgerError,
8408
- SearchLedgerIntegrityError,
8409
- SearchLedgerConflictError,
8410
8090
  SEARCH_LEDGER_SCHEMA,
8411
8091
  validateSearchLedgerEvent,
8412
8092
  openSearchLedger,
@@ -8418,4 +8098,4 @@ export {
8418
8098
  verifyCodeSurface,
8419
8099
  resolveWorktreePath
8420
8100
  };
8421
- //# sourceMappingURL=chunk-GSW3OBHK.js.map
8101
+ //# sourceMappingURL=chunk-JSJZ4PJ6.js.map