@tangle-network/agent-eval 0.116.0 → 0.117.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +38 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +54 -30
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-GSW3OBHK.js → chunk-JSJZ4PJ6.js} +406 -726
- package/dist/chunk-JSJZ4PJ6.js.map +1 -0
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
|
@@ -1,33 +1,39 @@
|
|
|
1
1
|
import {
|
|
2
|
+
JudgeParseError,
|
|
2
3
|
assertCodeSurfaceIdentity,
|
|
3
4
|
campaignBreakdown,
|
|
4
5
|
campaignMeanComposite,
|
|
6
|
+
costReceiptFromTCloud,
|
|
5
7
|
defaultProductionGate,
|
|
6
8
|
gepaProposer,
|
|
7
9
|
isProposedCandidate,
|
|
8
10
|
labelTrustRank,
|
|
11
|
+
maximumChargeForTCloudRequest,
|
|
9
12
|
pairHoldout,
|
|
10
13
|
recoverTruncatedJson,
|
|
11
14
|
renderAnalystEvidence,
|
|
12
15
|
runImprovementLoop,
|
|
13
16
|
surfaceContentHash,
|
|
14
17
|
surfaceHash
|
|
15
|
-
} from "./chunk-
|
|
16
|
-
import {
|
|
17
|
-
estimateCost,
|
|
18
|
-
isModelPriced
|
|
19
|
-
} from "./chunk-VI2UW6B6.js";
|
|
18
|
+
} from "./chunk-HQPHZGL6.js";
|
|
20
19
|
import {
|
|
20
|
+
SearchLedgerConflictError,
|
|
21
|
+
SearchLedgerError,
|
|
22
|
+
SearchLedgerIntegrityError,
|
|
23
|
+
appendSearchLedgerLine,
|
|
21
24
|
assertRealBackend,
|
|
22
25
|
contentHash,
|
|
26
|
+
createRunCostLedger,
|
|
27
|
+
fsCampaignStorage,
|
|
23
28
|
planCampaignRun,
|
|
29
|
+
resolveRunDir,
|
|
24
30
|
runCampaign,
|
|
25
|
-
summarizeBackendIntegrity
|
|
26
|
-
|
|
31
|
+
summarizeBackendIntegrity,
|
|
32
|
+
withSearchLedgerFileLock
|
|
33
|
+
} from "./chunk-IDZTTFRR.js";
|
|
27
34
|
import {
|
|
28
|
-
Mutex
|
|
29
|
-
|
|
30
|
-
} from "./chunk-MPHTT5HE.js";
|
|
35
|
+
Mutex
|
|
36
|
+
} from "./chunk-3YYRZDON.js";
|
|
31
37
|
import {
|
|
32
38
|
AnalystRegistry,
|
|
33
39
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -42,22 +48,21 @@ import {
|
|
|
42
48
|
makePolicyEditCandidateRecord,
|
|
43
49
|
policyEditsFromFindings,
|
|
44
50
|
validatePolicyEditCandidateRecord
|
|
45
|
-
} from "./chunk-
|
|
51
|
+
} from "./chunk-MGEHEHSN.js";
|
|
46
52
|
import {
|
|
47
53
|
eProcess,
|
|
48
54
|
mcnemar,
|
|
49
55
|
mulberry32,
|
|
50
56
|
pairedBootstrap,
|
|
51
57
|
pairedRiskDifference,
|
|
52
|
-
weightedComposite,
|
|
53
58
|
wilcoxonSignedRank
|
|
54
59
|
} from "./chunk-PJQFMIOX.js";
|
|
55
60
|
import {
|
|
56
61
|
analyzeTraces
|
|
57
|
-
} from "./chunk-
|
|
62
|
+
} from "./chunk-4JLWXDYA.js";
|
|
58
63
|
import {
|
|
59
64
|
OtlpFileTraceStore
|
|
60
|
-
} from "./chunk-
|
|
65
|
+
} from "./chunk-S2F4J57L.js";
|
|
61
66
|
import {
|
|
62
67
|
modelHasSnapshot,
|
|
63
68
|
validateRunRecord
|
|
@@ -70,11 +75,17 @@ import {
|
|
|
70
75
|
} from "./chunk-VSMTAMNK.js";
|
|
71
76
|
import {
|
|
72
77
|
callLlm,
|
|
73
|
-
callLlmJson
|
|
74
|
-
|
|
78
|
+
callLlmJson,
|
|
79
|
+
costReceiptFromLlm,
|
|
80
|
+
costReceiptFromLlmError,
|
|
81
|
+
maximumChargeForLlmRequest
|
|
82
|
+
} from "./chunk-NJC7U437.js";
|
|
83
|
+
import {
|
|
84
|
+
CostAccountingIncompleteError,
|
|
85
|
+
CostLedger
|
|
86
|
+
} from "./chunk-VCTY3W6J.js";
|
|
75
87
|
import {
|
|
76
88
|
AgentEvalError,
|
|
77
|
-
JudgeError,
|
|
78
89
|
ValidationError
|
|
79
90
|
} from "./chunk-ONWEPEDO.js";
|
|
80
91
|
|
|
@@ -496,339 +507,6 @@ async function runLineage(opts) {
|
|
|
496
507
|
return { lineage, best: lineage.best(), steps };
|
|
497
508
|
}
|
|
498
509
|
|
|
499
|
-
// src/judges.ts
|
|
500
|
-
var JudgeParseError = class extends JudgeError {
|
|
501
|
-
/** Name of the judge whose response failed to parse. */
|
|
502
|
-
judgeName;
|
|
503
|
-
/** The raw (truncated) model response that failed to parse. */
|
|
504
|
-
raw;
|
|
505
|
-
constructor(judgeName, raw, options) {
|
|
506
|
-
super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
|
|
507
|
-
this.judgeName = judgeName;
|
|
508
|
-
this.raw = raw;
|
|
509
|
-
}
|
|
510
|
-
};
|
|
511
|
-
function createDomainExpertJudge(domain) {
|
|
512
|
-
return async (tc, { scenario, turns }) => {
|
|
513
|
-
const conversation = turns.map(
|
|
514
|
-
(t, i) => `Turn ${i + 1}:
|
|
515
|
-
User: ${t.userMessage}
|
|
516
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
517
|
-
).join("\n\n---\n\n");
|
|
518
|
-
const resp = await tc.chat({
|
|
519
|
-
model: "gpt-4o",
|
|
520
|
-
messages: [
|
|
521
|
-
{
|
|
522
|
-
role: "system",
|
|
523
|
-
content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
|
|
524
|
-
|
|
525
|
-
Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
|
|
526
|
-
|
|
527
|
-
Evaluate:
|
|
528
|
-
1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
|
|
529
|
-
2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
|
|
530
|
-
|
|
531
|
-
Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
|
|
532
|
-
},
|
|
533
|
-
{
|
|
534
|
-
role: "user",
|
|
535
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
536
|
-
Scenario: ${scenario.thesis}
|
|
537
|
-
|
|
538
|
-
${conversation}`
|
|
539
|
-
}
|
|
540
|
-
],
|
|
541
|
-
temperature: 0.1,
|
|
542
|
-
maxTokens: 800
|
|
543
|
-
});
|
|
544
|
-
return parseJudgeResponse("domain_expert", resp);
|
|
545
|
-
};
|
|
546
|
-
}
|
|
547
|
-
var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
|
|
548
|
-
const codeBlocks = artifacts.codeBlocks;
|
|
549
|
-
if (codeBlocks.length === 0) {
|
|
550
|
-
return [
|
|
551
|
-
{
|
|
552
|
-
judgeName: "code_execution",
|
|
553
|
-
dimension: "code_execution",
|
|
554
|
-
score: 0,
|
|
555
|
-
reasoning: "No code blocks found in agent response."
|
|
556
|
-
}
|
|
557
|
-
];
|
|
558
|
-
}
|
|
559
|
-
const codeText = codeBlocks.map(
|
|
560
|
-
(b, i) => `Block ${i + 1} (${b.language}):
|
|
561
|
-
\`\`\`${b.language}
|
|
562
|
-
${b.code.slice(0, 3e3)}
|
|
563
|
-
\`\`\``
|
|
564
|
-
).join("\n\n");
|
|
565
|
-
const resp = await tc.chat({
|
|
566
|
-
model: "gpt-4o",
|
|
567
|
-
messages: [
|
|
568
|
-
{
|
|
569
|
-
role: "system",
|
|
570
|
-
content: `You are a principal software engineer reviewing code written by an AI agent.
|
|
571
|
-
|
|
572
|
-
Score STRICTLY:
|
|
573
|
-
1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
|
|
574
|
-
2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
|
|
575
|
-
3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
|
|
576
|
-
|
|
577
|
-
Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
|
|
578
|
-
},
|
|
579
|
-
{
|
|
580
|
-
role: "user",
|
|
581
|
-
content: `Task: ${scenario.thesis}
|
|
582
|
-
|
|
583
|
-
${codeText}`
|
|
584
|
-
}
|
|
585
|
-
],
|
|
586
|
-
temperature: 0.1,
|
|
587
|
-
maxTokens: 1e3
|
|
588
|
-
});
|
|
589
|
-
return parseJudgeResponse("code_execution", resp);
|
|
590
|
-
};
|
|
591
|
-
var coherenceJudge = async (tc, { scenario, turns }) => {
|
|
592
|
-
if (turns.length < 2) {
|
|
593
|
-
return [];
|
|
594
|
-
}
|
|
595
|
-
const conversation = turns.map(
|
|
596
|
-
(t, i) => `Turn ${i + 1}:
|
|
597
|
-
User: ${t.userMessage}
|
|
598
|
-
Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
|
|
599
|
-
).join("\n\n---\n\n");
|
|
600
|
-
const resp = await tc.chat({
|
|
601
|
-
model: "gpt-4o",
|
|
602
|
-
messages: [
|
|
603
|
-
{
|
|
604
|
-
role: "system",
|
|
605
|
-
content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
|
|
606
|
-
|
|
607
|
-
Score STRICTLY:
|
|
608
|
-
1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
|
|
609
|
-
2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
|
|
610
|
-
3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
|
|
611
|
-
|
|
612
|
-
Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
|
|
613
|
-
},
|
|
614
|
-
{
|
|
615
|
-
role: "user",
|
|
616
|
-
content: `Scenario: ${scenario.thesis}
|
|
617
|
-
|
|
618
|
-
${conversation}`
|
|
619
|
-
}
|
|
620
|
-
],
|
|
621
|
-
temperature: 0.1,
|
|
622
|
-
maxTokens: 800
|
|
623
|
-
});
|
|
624
|
-
return parseJudgeResponse("coherence", resp);
|
|
625
|
-
};
|
|
626
|
-
var adversarialJudge = async (tc, { scenario, turns }) => {
|
|
627
|
-
const conversation = turns.map(
|
|
628
|
-
(t, i) => `Turn ${i + 1}:
|
|
629
|
-
User: ${t.userMessage}
|
|
630
|
-
Agent: ${t.agentResponse.slice(0, 1500)}`
|
|
631
|
-
).join("\n\n---\n\n");
|
|
632
|
-
const resp = await tc.chat({
|
|
633
|
-
model: "gpt-4o",
|
|
634
|
-
messages: [
|
|
635
|
-
{
|
|
636
|
-
role: "system",
|
|
637
|
-
content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
|
|
638
|
-
|
|
639
|
-
1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
|
|
640
|
-
2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
|
|
641
|
-
3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
|
|
642
|
-
|
|
643
|
-
Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
|
|
644
|
-
|
|
645
|
-
Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
|
|
646
|
-
},
|
|
647
|
-
{
|
|
648
|
-
role: "user",
|
|
649
|
-
content: `Persona: ${scenario.persona}
|
|
650
|
-
Scenario: ${scenario.thesis}
|
|
651
|
-
|
|
652
|
-
${conversation}`
|
|
653
|
-
}
|
|
654
|
-
],
|
|
655
|
-
temperature: 0.2,
|
|
656
|
-
maxTokens: 800
|
|
657
|
-
});
|
|
658
|
-
return parseJudgeResponse("adversarial", resp);
|
|
659
|
-
};
|
|
660
|
-
function createCustomJudge(name, systemPrompt, opts) {
|
|
661
|
-
return async (tc, { scenario, turns }) => {
|
|
662
|
-
const conversation = turns.map(
|
|
663
|
-
(t, i) => `Turn ${i + 1}:
|
|
664
|
-
User: ${t.userMessage}
|
|
665
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
666
|
-
).join("\n\n---\n\n");
|
|
667
|
-
const resp = await tc.chat({
|
|
668
|
-
model: opts?.model ?? "gpt-4o",
|
|
669
|
-
messages: [
|
|
670
|
-
{
|
|
671
|
-
role: "system",
|
|
672
|
-
content: systemPrompt
|
|
673
|
-
},
|
|
674
|
-
{
|
|
675
|
-
role: "user",
|
|
676
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
677
|
-
Scenario: ${scenario.thesis}
|
|
678
|
-
|
|
679
|
-
${conversation}`
|
|
680
|
-
}
|
|
681
|
-
],
|
|
682
|
-
temperature: opts?.temperature ?? 0.1,
|
|
683
|
-
maxTokens: opts?.maxTokens ?? 1e3
|
|
684
|
-
});
|
|
685
|
-
return parseJudgeResponse(name, resp);
|
|
686
|
-
};
|
|
687
|
-
}
|
|
688
|
-
function defaultJudges(domain) {
|
|
689
|
-
return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
|
|
690
|
-
}
|
|
691
|
-
function parseJudgeResponse(judgeName, resp) {
|
|
692
|
-
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
693
|
-
try {
|
|
694
|
-
let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
|
|
695
|
-
const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
|
|
696
|
-
if (arrayMatch) cleaned = arrayMatch[0];
|
|
697
|
-
const parsed = JSON.parse(cleaned);
|
|
698
|
-
return parsed.map((p) => ({
|
|
699
|
-
judgeName,
|
|
700
|
-
dimension: p.dimension,
|
|
701
|
-
score: Math.max(0, Math.min(10, p.score)),
|
|
702
|
-
reasoning: p.reasoning ?? "",
|
|
703
|
-
evidence: p.evidence
|
|
704
|
-
}));
|
|
705
|
-
} catch (err) {
|
|
706
|
-
throw new JudgeParseError(judgeName, content, { cause: err });
|
|
707
|
-
}
|
|
708
|
-
}
|
|
709
|
-
|
|
710
|
-
// src/llm-judge.ts
|
|
711
|
-
function llmJudge(name, prompt, opts) {
|
|
712
|
-
if (!name.trim()) {
|
|
713
|
-
throw new Error("llmJudge: name must be non-empty");
|
|
714
|
-
}
|
|
715
|
-
if (!prompt.trim()) {
|
|
716
|
-
throw new Error(`llmJudge '${name}': prompt must be non-empty`);
|
|
717
|
-
}
|
|
718
|
-
const model = opts.model ?? opts.chat.defaultModel;
|
|
719
|
-
if (!model) {
|
|
720
|
-
throw new Error(
|
|
721
|
-
`llmJudge '${name}': no model on opts and no defaultModel on the ChatClient \u2014 pass opts.model or bind defaultModel at createChatClient().`
|
|
722
|
-
);
|
|
723
|
-
}
|
|
724
|
-
const dimensions = normalizeDimensions(opts.dimensions, name);
|
|
725
|
-
const scale = opts.scale ?? "unit";
|
|
726
|
-
const divisor = scale === "ten" ? 10 : 1;
|
|
727
|
-
const renderUser = opts.renderUser ?? ((input) => JSON.stringify({ scenario: input.scenario, artifact: input.artifact }, null, 2));
|
|
728
|
-
if (opts.weights) {
|
|
729
|
-
for (const key of Object.keys(opts.weights)) {
|
|
730
|
-
if (!dimensions.some((d) => d.key === key)) {
|
|
731
|
-
throw new Error(
|
|
732
|
-
`llmJudge '${name}': weights names dimension '${key}' that is not declared in dimensions`
|
|
733
|
-
);
|
|
734
|
-
}
|
|
735
|
-
}
|
|
736
|
-
}
|
|
737
|
-
const systemPrompt = `${prompt}
|
|
738
|
-
|
|
739
|
-
${renderContract(dimensions, scale)}`;
|
|
740
|
-
return {
|
|
741
|
-
name,
|
|
742
|
-
dimensions,
|
|
743
|
-
appliesTo: opts.appliesTo,
|
|
744
|
-
async score({ artifact, scenario, signal }) {
|
|
745
|
-
const response = await opts.chat.chat(
|
|
746
|
-
{
|
|
747
|
-
model,
|
|
748
|
-
messages: [
|
|
749
|
-
{ role: "system", content: systemPrompt },
|
|
750
|
-
{ role: "user", content: renderUser({ artifact, scenario }) }
|
|
751
|
-
],
|
|
752
|
-
jsonMode: true,
|
|
753
|
-
temperature: opts.temperature ?? 0.1,
|
|
754
|
-
maxTokens: opts.maxTokens ?? 800
|
|
755
|
-
},
|
|
756
|
-
{ signal }
|
|
757
|
-
);
|
|
758
|
-
const parsed = parseResponse(name, response.content);
|
|
759
|
-
const rawDims = parsed.dimensions ?? parsed.scores;
|
|
760
|
-
if (!rawDims || typeof rawDims !== "object") {
|
|
761
|
-
throw new JudgeParseError(name, response.content, {
|
|
762
|
-
cause: new Error("response has no `dimensions` object")
|
|
763
|
-
});
|
|
764
|
-
}
|
|
765
|
-
const dims = {};
|
|
766
|
-
for (const { key } of dimensions) {
|
|
767
|
-
const raw = rawDims[key];
|
|
768
|
-
const value = Number(raw);
|
|
769
|
-
if (raw === void 0 || raw === null || !Number.isFinite(value)) {
|
|
770
|
-
throw new JudgeParseError(name, response.content, {
|
|
771
|
-
cause: new Error(
|
|
772
|
-
`dimension '${key}' missing or non-numeric (got ${JSON.stringify(raw)})`
|
|
773
|
-
)
|
|
774
|
-
});
|
|
775
|
-
}
|
|
776
|
-
dims[key] = clamp01(value / divisor);
|
|
777
|
-
}
|
|
778
|
-
const weights = opts.weights ?? Object.fromEntries(dimensions.map((d) => [d.key, 1 / dimensions.length]));
|
|
779
|
-
const { composite } = weightedComposite({ dims, weights });
|
|
780
|
-
const notes = firstString(parsed.notes) ?? firstString(parsed.rationale) ?? `${name}: composite ${composite.toFixed(3)} over ${dimensions.length} dimension(s)`;
|
|
781
|
-
return { dimensions: dims, composite, notes };
|
|
782
|
-
}
|
|
783
|
-
};
|
|
784
|
-
}
|
|
785
|
-
function normalizeDimensions(input, name) {
|
|
786
|
-
const raw = input && input.length > 0 ? input : ["quality"];
|
|
787
|
-
const out = [];
|
|
788
|
-
const seen = /* @__PURE__ */ new Set();
|
|
789
|
-
for (const d of raw) {
|
|
790
|
-
const dim = typeof d === "string" ? { key: d, description: d } : d;
|
|
791
|
-
if (!dim.key.trim()) {
|
|
792
|
-
throw new Error(`llmJudge '${name}': dimension key must be non-empty`);
|
|
793
|
-
}
|
|
794
|
-
if (seen.has(dim.key)) {
|
|
795
|
-
throw new Error(`llmJudge '${name}': duplicate dimension key '${dim.key}'`);
|
|
796
|
-
}
|
|
797
|
-
seen.add(dim.key);
|
|
798
|
-
out.push(dim);
|
|
799
|
-
}
|
|
800
|
-
return out;
|
|
801
|
-
}
|
|
802
|
-
function renderContract(dimensions, scale) {
|
|
803
|
-
const range = scale === "ten" ? "0 to 10" : "0.0 to 1.0";
|
|
804
|
-
const lines = dimensions.map((d) => ` - "${d.key}": ${d.description} (score ${range})`);
|
|
805
|
-
const example = `{"dimensions": {${dimensions.map((d) => `"${d.key}": <number>`).join(", ")}}, "notes": "<one-line rationale>"}`;
|
|
806
|
-
return [
|
|
807
|
-
"Score the artifact on EACH of these dimensions:",
|
|
808
|
-
...lines,
|
|
809
|
-
"",
|
|
810
|
-
`Respond with JSON ONLY, no prose. Every dimension is a number in [${range}]:`,
|
|
811
|
-
example
|
|
812
|
-
].join("\n");
|
|
813
|
-
}
|
|
814
|
-
function parseResponse(name, content) {
|
|
815
|
-
const stripped = content.replace(/```json\n?|\n?```/g, "").trim();
|
|
816
|
-
const objMatch = stripped.match(/\{[\s\S]*\}/);
|
|
817
|
-
const payload = objMatch ? objMatch[0] : stripped;
|
|
818
|
-
try {
|
|
819
|
-
const parsed = JSON.parse(payload);
|
|
820
|
-
if (typeof parsed !== "object" || parsed === null) {
|
|
821
|
-
throw new Error("parsed value is not an object");
|
|
822
|
-
}
|
|
823
|
-
return parsed;
|
|
824
|
-
} catch (err) {
|
|
825
|
-
throw new JudgeParseError(name, content, { cause: err });
|
|
826
|
-
}
|
|
827
|
-
}
|
|
828
|
-
function firstString(value) {
|
|
829
|
-
return typeof value === "string" && value.trim() ? value : void 0;
|
|
830
|
-
}
|
|
831
|
-
|
|
832
510
|
// src/campaign/analyst-surface.ts
|
|
833
511
|
function surfaceToText(surface) {
|
|
834
512
|
if (typeof surface === "string") return surface;
|
|
@@ -3418,6 +3096,7 @@ var SKILLOPT_SYSTEM = 'You are a SkillOpt optimizer. You improve ONE skill docum
|
|
|
3418
3096
|
function skillOptProposer(opts) {
|
|
3419
3097
|
const evidenceK = opts.evidenceK ?? 3;
|
|
3420
3098
|
const defaultBudget = opts.editBudget ?? 3;
|
|
3099
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
3421
3100
|
async function proposePatches(args) {
|
|
3422
3101
|
const userPrompt = buildPatchPrompt({
|
|
3423
3102
|
target: opts.target,
|
|
@@ -3429,19 +3108,29 @@ function skillOptProposer(opts) {
|
|
|
3429
3108
|
findingsNote: args.findingsNote,
|
|
3430
3109
|
count: args.count
|
|
3431
3110
|
});
|
|
3432
|
-
const
|
|
3433
|
-
|
|
3434
|
-
|
|
3435
|
-
|
|
3436
|
-
|
|
3437
|
-
|
|
3438
|
-
|
|
3439
|
-
|
|
3440
|
-
|
|
3441
|
-
|
|
3442
|
-
|
|
3443
|
-
|
|
3444
|
-
|
|
3111
|
+
const request = {
|
|
3112
|
+
model: opts.model,
|
|
3113
|
+
messages: [
|
|
3114
|
+
{ role: "system", content: SKILLOPT_SYSTEM },
|
|
3115
|
+
{ role: "user", content: userPrompt }
|
|
3116
|
+
],
|
|
3117
|
+
jsonMode: true,
|
|
3118
|
+
temperature: opts.temperature ?? 0.6,
|
|
3119
|
+
maxTokens: opts.maxTokens ?? 4e3
|
|
3120
|
+
};
|
|
3121
|
+
const paid = await (args.costLedger ?? directCostLedger).runPaidCall({
|
|
3122
|
+
channel: "driver",
|
|
3123
|
+
phase: args.costPhase ?? "search.proposal",
|
|
3124
|
+
actor: "skill-opt.propose",
|
|
3125
|
+
model: opts.model,
|
|
3126
|
+
maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
|
|
3127
|
+
signal: args.signal,
|
|
3128
|
+
execute: (signal, callId) => callLlm(request, { ...opts.llm, signal, idempotencyKey: callId }),
|
|
3129
|
+
receipt: costReceiptFromLlm,
|
|
3130
|
+
receiptFromError: costReceiptFromLlmError
|
|
3131
|
+
});
|
|
3132
|
+
if (!paid.succeeded) throw paid.error;
|
|
3133
|
+
const result = paid.value;
|
|
3445
3134
|
return parseSkillPatchResponse(result.content, args.count, args.editBudget);
|
|
3446
3135
|
}
|
|
3447
3136
|
return {
|
|
@@ -3461,7 +3150,9 @@ function skillOptProposer(opts) {
|
|
|
3461
3150
|
rejectedBuffer: [],
|
|
3462
3151
|
findingsNote: renderAnalystEvidence(ctx.findings, ctx.report) ?? void 0,
|
|
3463
3152
|
count: ctx.populationSize,
|
|
3464
|
-
signal: ctx.signal
|
|
3153
|
+
signal: ctx.signal,
|
|
3154
|
+
costLedger: ctx.costLedger ?? directCostLedger,
|
|
3155
|
+
costPhase: ctx.costPhase ?? "search.proposal"
|
|
3465
3156
|
});
|
|
3466
3157
|
const out = [];
|
|
3467
3158
|
const seen = /* @__PURE__ */ new Set();
|
|
@@ -3625,16 +3316,20 @@ async function runSkillOpt(opts) {
|
|
|
3625
3316
|
const budgetAnneal = opts.budgetAnneal ?? true;
|
|
3626
3317
|
const rejectedBufferSize = opts.rejectedBufferSize ?? 12;
|
|
3627
3318
|
const slowMetaEvery = opts.slowMetaEvery ?? 2;
|
|
3628
|
-
|
|
3319
|
+
opts.runDir = resolveRunDir(opts.runDir, opts.repo);
|
|
3320
|
+
const storage = opts.storage ?? fsCampaignStorage();
|
|
3321
|
+
const costLedger = opts.costLedger ?? createRunCostLedger({
|
|
3322
|
+
storage,
|
|
3323
|
+
runDir: opts.runDir,
|
|
3324
|
+
costCeilingUsd: opts.costCeiling
|
|
3325
|
+
});
|
|
3629
3326
|
const scoreHoldout = async (surface, tag) => {
|
|
3630
|
-
const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag);
|
|
3631
|
-
totalCostUsd += campaign.aggregates.totalCostUsd;
|
|
3327
|
+
const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag, costLedger);
|
|
3632
3328
|
return campaignMeanComposite(campaign);
|
|
3633
3329
|
};
|
|
3634
3330
|
const evidenceK = opts.evidenceK ?? 3;
|
|
3635
3331
|
const trainEvidence = async (surface, tag) => {
|
|
3636
|
-
const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag);
|
|
3637
|
-
totalCostUsd += campaign.aggregates.totalCostUsd;
|
|
3332
|
+
const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag, costLedger);
|
|
3638
3333
|
return toEvidence(campaign, evidenceK);
|
|
3639
3334
|
};
|
|
3640
3335
|
let current = opts.baselineSurface;
|
|
@@ -3658,7 +3353,9 @@ async function runSkillOpt(opts) {
|
|
|
3658
3353
|
rejectedBuffer: buffer,
|
|
3659
3354
|
metaNote,
|
|
3660
3355
|
count: patchesPerEpoch,
|
|
3661
|
-
signal: opts.signal ?? new AbortController().signal
|
|
3356
|
+
signal: opts.signal ?? new AbortController().signal,
|
|
3357
|
+
costLedger,
|
|
3358
|
+
costPhase: "skill-opt.proposal"
|
|
3662
3359
|
});
|
|
3663
3360
|
let accepted = null;
|
|
3664
3361
|
const rejectedThisEpoch = [];
|
|
@@ -3717,6 +3414,7 @@ async function runSkillOpt(opts) {
|
|
|
3717
3414
|
});
|
|
3718
3415
|
if (sinceAccept >= patience) break;
|
|
3719
3416
|
}
|
|
3417
|
+
const cost = costLedger.summary();
|
|
3720
3418
|
return {
|
|
3721
3419
|
winnerSurface: current,
|
|
3722
3420
|
baselineHoldoutComposite: baselineHoldout,
|
|
@@ -3726,12 +3424,14 @@ async function runSkillOpt(opts) {
|
|
|
3726
3424
|
rejectedEdits: rejectedAll,
|
|
3727
3425
|
epochsRun,
|
|
3728
3426
|
history,
|
|
3729
|
-
totalCostUsd
|
|
3427
|
+
totalCostUsd: cost.totalCostUsd,
|
|
3428
|
+
cost
|
|
3730
3429
|
};
|
|
3731
3430
|
}
|
|
3732
|
-
function runScoringCampaign(opts, scenarios, surface, tag) {
|
|
3431
|
+
function runScoringCampaign(opts, scenarios, surface, tag, costLedger) {
|
|
3733
3432
|
return runCampaign({
|
|
3734
3433
|
...opts,
|
|
3434
|
+
costLedger,
|
|
3735
3435
|
scenarios,
|
|
3736
3436
|
dispatch: (scenario, ctx) => opts.dispatchWithSurface(surface, scenario, ctx),
|
|
3737
3437
|
runDir: `${opts.runDir}/${tag}`
|
|
@@ -3903,11 +3603,11 @@ function gepaEntry(config, combineParents, name) {
|
|
|
3903
3603
|
...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
|
|
3904
3604
|
...config.report !== void 0 ? { report: config.report } : {}
|
|
3905
3605
|
});
|
|
3906
|
-
|
|
3907
|
-
|
|
3908
|
-
|
|
3909
|
-
|
|
3910
|
-
|
|
3606
|
+
return {
|
|
3607
|
+
winnerSurface: result.winnerSurface,
|
|
3608
|
+
costUsd: result.cost.totalCostUsd,
|
|
3609
|
+
durationMs: Date.now() - started
|
|
3610
|
+
};
|
|
3911
3611
|
}
|
|
3912
3612
|
};
|
|
3913
3613
|
}
|
|
@@ -3980,11 +3680,11 @@ function fapoEscalationEntry(config, name = "fapo-escalation") {
|
|
|
3980
3680
|
...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
|
|
3981
3681
|
...config.report !== void 0 ? { report: config.report } : {}
|
|
3982
3682
|
});
|
|
3983
|
-
|
|
3984
|
-
|
|
3985
|
-
|
|
3986
|
-
|
|
3987
|
-
|
|
3683
|
+
return {
|
|
3684
|
+
winnerSurface: result.winnerSurface,
|
|
3685
|
+
costUsd: result.cost.totalCostUsd,
|
|
3686
|
+
durationMs: Date.now() - started
|
|
3687
|
+
};
|
|
3988
3688
|
}
|
|
3989
3689
|
};
|
|
3990
3690
|
}
|
|
@@ -4218,6 +3918,7 @@ function createLlmCorrectnessChecker(tc, opts = {}) {
|
|
|
4218
3918
|
const model = opts.model ?? "claude-sonnet-4-6";
|
|
4219
3919
|
const maxContentChars = opts.maxContentChars ?? 8e3;
|
|
4220
3920
|
const maxAttempts = opts.maxAttempts ?? 2;
|
|
3921
|
+
const costLedger = opts.costLedger ?? new CostLedger();
|
|
4221
3922
|
const sink = opts.rawSink;
|
|
4222
3923
|
const record = async (event) => {
|
|
4223
3924
|
try {
|
|
@@ -4261,7 +3962,24 @@ ${content.slice(0, maxContentChars)}`
|
|
|
4261
3962
|
redactedFields: []
|
|
4262
3963
|
});
|
|
4263
3964
|
try {
|
|
4264
|
-
const
|
|
3965
|
+
const paid = await costLedger.runPaidCall({
|
|
3966
|
+
channel: "verifier",
|
|
3967
|
+
phase: opts.costPhase ?? "completion.correctness",
|
|
3968
|
+
actor: "correctness-checker",
|
|
3969
|
+
model,
|
|
3970
|
+
maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
|
|
3971
|
+
tags: {
|
|
3972
|
+
...opts.costTags,
|
|
3973
|
+
requirementId: requirement.reqId,
|
|
3974
|
+
attempt: String(attempt)
|
|
3975
|
+
},
|
|
3976
|
+
signal: opts.signal,
|
|
3977
|
+
execute: () => tc.chat(request),
|
|
3978
|
+
receipt: (response) => costReceiptFromTCloud(response, model),
|
|
3979
|
+
receiptFromError: (error) => opts.receiptFromError?.(error, attempt)
|
|
3980
|
+
});
|
|
3981
|
+
if (!paid.succeeded) throw paid.error;
|
|
3982
|
+
const resp = paid.value;
|
|
4265
3983
|
const raw = resp.choices?.[0]?.message?.content ?? "";
|
|
4266
3984
|
await record({
|
|
4267
3985
|
eventId: randomUUID(),
|
|
@@ -4777,12 +4495,12 @@ function requireResolvedModel(cell, profileId) {
|
|
|
4777
4495
|
const resolved = cell.resolvedModel?.trim();
|
|
4778
4496
|
if (!resolved) {
|
|
4779
4497
|
throw new ProfileMatrixError(
|
|
4780
|
-
`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014
|
|
4498
|
+
`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 return it in the ctx.cost.runPaidCall receipt so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
|
|
4781
4499
|
);
|
|
4782
4500
|
}
|
|
4783
4501
|
if (!modelHasSnapshot(resolved)) {
|
|
4784
4502
|
throw new ProfileMatrixError(
|
|
4785
|
-
`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD)
|
|
4503
|
+
`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) in the paid-call receipt`
|
|
4786
4504
|
);
|
|
4787
4505
|
}
|
|
4788
4506
|
return resolved;
|
|
@@ -4808,14 +4526,9 @@ function buildRunRecord(args) {
|
|
|
4808
4526
|
}
|
|
4809
4527
|
const perDimMean = {};
|
|
4810
4528
|
for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean3(values);
|
|
4811
|
-
|
|
4812
|
-
let costEstimated = false;
|
|
4813
|
-
if (costUsd === 0 && cell.tokenUsage.output > 0 && isModelPriced(model)) {
|
|
4814
|
-
costUsd = estimateCost(cell.tokenUsage.input, cell.tokenUsage.output, model);
|
|
4815
|
-
costEstimated = costUsd > 0;
|
|
4816
|
-
}
|
|
4529
|
+
const costUsd = cell.costUsd;
|
|
4817
4530
|
raw.cost_usd = costUsd;
|
|
4818
|
-
raw.cost_estimated = costEstimated ? 1 : 0;
|
|
4531
|
+
raw.cost_estimated = cell.costEstimated ? 1 : 0;
|
|
4819
4532
|
raw.tokens_input = cell.tokenUsage.input;
|
|
4820
4533
|
raw.tokens_output = cell.tokenUsage.output;
|
|
4821
4534
|
if (typeof cell.tokenUsage.cached === "number") raw.tokens_cached = cell.tokenUsage.cached;
|
|
@@ -4958,11 +4671,8 @@ async function runProfileMatrix(opts) {
|
|
|
4958
4671
|
profileRecords.push(record);
|
|
4959
4672
|
records.push(record);
|
|
4960
4673
|
}
|
|
4961
|
-
const
|
|
4962
|
-
campaigns[profileId] =
|
|
4963
|
-
...campaign,
|
|
4964
|
-
aggregates: { ...campaign.aggregates, totalCostUsd: pricedTotalCostUsd }
|
|
4965
|
-
};
|
|
4674
|
+
const totalCostUsd = campaign.aggregates.totalCostUsd;
|
|
4675
|
+
campaigns[profileId] = campaign;
|
|
4966
4676
|
byProfile[profileId] = {
|
|
4967
4677
|
profileId,
|
|
4968
4678
|
profileHash,
|
|
@@ -4972,7 +4682,7 @@ async function runProfileMatrix(opts) {
|
|
|
4972
4682
|
model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
|
|
4973
4683
|
records: profileRecords.length,
|
|
4974
4684
|
meanComposite: mean3(profileRecords.map(compositeOf)),
|
|
4975
|
-
totalCostUsd
|
|
4685
|
+
totalCostUsd,
|
|
4976
4686
|
integrity: summarizeBackendIntegrity(profileRecords)
|
|
4977
4687
|
};
|
|
4978
4688
|
}
|
|
@@ -5189,36 +4899,78 @@ function surfaceToPromptText(surface) {
|
|
|
5189
4899
|
return typeof surface === "string" ? surface : JSON.stringify(surface);
|
|
5190
4900
|
}
|
|
5191
4901
|
function analysisEditProposer(opts) {
|
|
4902
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
5192
4903
|
return {
|
|
5193
4904
|
kind: opts.kind,
|
|
5194
4905
|
async propose(ctx) {
|
|
5195
4906
|
const parent = surfaceToPromptText(ctx.currentSurface);
|
|
4907
|
+
const costLedger = ctx.costLedger ?? directCostLedger;
|
|
4908
|
+
const phase = ctx.costPhase ?? "search.proposal";
|
|
5196
4909
|
const traces = await opts.resolveTraces(ctx) ?? "";
|
|
5197
4910
|
if (!traces.trim()) throw new Error(opts.noTracesError);
|
|
5198
4911
|
const dir = mkdtempSync(join4(tmpdir(), `${opts.kind}-proposer-`));
|
|
5199
4912
|
const tracePath = join4(dir, "traces.jsonl");
|
|
5200
4913
|
writeFileSync2(tracePath, traces.endsWith("\n") ? traces : `${traces}
|
|
5201
4914
|
`);
|
|
5202
|
-
|
|
5203
|
-
|
|
5204
|
-
|
|
5205
|
-
|
|
5206
|
-
|
|
5207
|
-
|
|
5208
|
-
|
|
5209
|
-
|
|
5210
|
-
|
|
4915
|
+
if (costLedger.costCeilingUsd !== void 0 && !opts.analysisReceipt) {
|
|
4916
|
+
throw new CostAccountingIncompleteError(
|
|
4917
|
+
`${opts.kind}: capped analysis requires analysisReceipt before external execution`
|
|
4918
|
+
);
|
|
4919
|
+
}
|
|
4920
|
+
const analysis = await costLedger.runPaidCall({
|
|
4921
|
+
channel: "analyst",
|
|
4922
|
+
phase,
|
|
4923
|
+
actor: `${opts.kind}.analyze`,
|
|
4924
|
+
model: opts.analysisModel,
|
|
4925
|
+
maximumCharge: opts.analysisMaximumCharge,
|
|
4926
|
+
tags: { generation: String(ctx.generation) },
|
|
4927
|
+
signal: ctx.signal,
|
|
4928
|
+
execute: (signal) => opts.analyze(tracePath, { ...ctx, signal }),
|
|
4929
|
+
receipt: (report2) => opts.analysisReceipt?.(report2) ?? {
|
|
4930
|
+
model: opts.analysisModel,
|
|
4931
|
+
inputTokens: 0,
|
|
4932
|
+
outputTokens: 0,
|
|
4933
|
+
costUnknown: true
|
|
4934
|
+
}
|
|
4935
|
+
});
|
|
4936
|
+
if (!analysis.succeeded) throw analysis.error;
|
|
4937
|
+
const report = analysis.value;
|
|
4938
|
+
const request = {
|
|
4939
|
+
model: opts.applyModel,
|
|
4940
|
+
messages: [
|
|
4941
|
+
{ role: "system", content: APPLY_SYSTEM },
|
|
4942
|
+
{
|
|
4943
|
+
role: "user",
|
|
4944
|
+
content: `CURRENT PROMPT:
|
|
5211
4945
|
${parent}
|
|
5212
4946
|
|
|
5213
4947
|
TRACE-ANALYSIS REPORT:
|
|
5214
4948
|
${report}
|
|
5215
4949
|
|
|
5216
4950
|
Return the full revised prompt.`
|
|
5217
|
-
|
|
5218
|
-
|
|
5219
|
-
|
|
5220
|
-
|
|
5221
|
-
|
|
4951
|
+
}
|
|
4952
|
+
],
|
|
4953
|
+
maxTokens: opts.applyMaxTokens ?? 6e3
|
|
4954
|
+
};
|
|
4955
|
+
const llm = {
|
|
4956
|
+
baseUrl: opts.baseUrl,
|
|
4957
|
+
apiKey: opts.apiKey,
|
|
4958
|
+
fetch: opts.fetchImpl
|
|
4959
|
+
};
|
|
4960
|
+
const apply = await costLedger.runPaidCall({
|
|
4961
|
+
channel: "driver",
|
|
4962
|
+
phase,
|
|
4963
|
+
actor: `${opts.kind}.apply`,
|
|
4964
|
+
model: opts.applyModel,
|
|
4965
|
+
maximumCharge: maximumChargeForLlmRequest(request, llm),
|
|
4966
|
+
tags: { generation: String(ctx.generation) },
|
|
4967
|
+
signal: ctx.signal,
|
|
4968
|
+
execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
|
|
4969
|
+
receipt: costReceiptFromLlm,
|
|
4970
|
+
receiptFromError: costReceiptFromLlmError
|
|
4971
|
+
});
|
|
4972
|
+
if (!apply.succeeded) throw apply.error;
|
|
4973
|
+
const applied = apply.value;
|
|
5222
4974
|
const text = applied.content.trim();
|
|
5223
4975
|
if (!text || text === parent) return [];
|
|
5224
4976
|
return [{ surface: text, label: opts.label, rationale: opts.rationale(report) }];
|
|
@@ -5237,7 +4989,12 @@ function haloProposer(opts) {
|
|
|
5237
4989
|
label: "halo",
|
|
5238
4990
|
baseUrl: opts.baseUrl,
|
|
5239
4991
|
apiKey: opts.apiKey,
|
|
4992
|
+
analysisModel: model,
|
|
5240
4993
|
applyModel: opts.applyModel ?? model,
|
|
4994
|
+
costLedger: opts.costLedger,
|
|
4995
|
+
analysisMaximumCharge: opts.analysisMaximumCharge,
|
|
4996
|
+
analysisReceipt: opts.analysisReceipt,
|
|
4997
|
+
applyMaxTokens: opts.applyMaxTokens,
|
|
5241
4998
|
fetchImpl: opts.fetchImpl,
|
|
5242
4999
|
resolveTraces: opts.resolveTraces,
|
|
5243
5000
|
noTracesError: "haloProposer: resolveTraces returned no OTLP traces \u2014 the halo engine has nothing to analyze",
|
|
@@ -5669,6 +5426,7 @@ function llmPolicyEditProposer(opts) {
|
|
|
5669
5426
|
opts.maxAuthorContextChars ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.authorContextChars,
|
|
5670
5427
|
"maxAuthorContextChars"
|
|
5671
5428
|
);
|
|
5429
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
5672
5430
|
return {
|
|
5673
5431
|
kind: "llm-policy-edit",
|
|
5674
5432
|
async propose(ctx) {
|
|
@@ -5731,26 +5489,34 @@ function llmPolicyEditProposer(opts) {
|
|
|
5731
5489
|
maxAuthorContextChars
|
|
5732
5490
|
);
|
|
5733
5491
|
const userContent = JSON.stringify(authorContext);
|
|
5734
|
-
const
|
|
5735
|
-
|
|
5736
|
-
|
|
5737
|
-
|
|
5738
|
-
|
|
5739
|
-
|
|
5740
|
-
|
|
5741
|
-
|
|
5742
|
-
|
|
5743
|
-
],
|
|
5744
|
-
jsonSchema: {
|
|
5745
|
-
name: "policy_edit_author",
|
|
5746
|
-
schema: responseSchema
|
|
5747
|
-
},
|
|
5748
|
-
temperature: opts.temperature ?? 0.2,
|
|
5749
|
-
maxTokens: opts.maxTokens ?? 6e3,
|
|
5750
|
-
timeoutMs: opts.timeoutMs
|
|
5492
|
+
const request = {
|
|
5493
|
+
model: opts.model,
|
|
5494
|
+
messages: [
|
|
5495
|
+
{ role: "system", content: POLICY_EDIT_AUTHOR_SYSTEM },
|
|
5496
|
+
{ role: "user", content: userContent }
|
|
5497
|
+
],
|
|
5498
|
+
jsonSchema: {
|
|
5499
|
+
name: "policy_edit_author",
|
|
5500
|
+
schema: responseSchema
|
|
5751
5501
|
},
|
|
5752
|
-
|
|
5753
|
-
|
|
5502
|
+
temperature: opts.temperature ?? 0.2,
|
|
5503
|
+
maxTokens: opts.maxTokens ?? 6e3,
|
|
5504
|
+
timeoutMs: opts.timeoutMs
|
|
5505
|
+
};
|
|
5506
|
+
const paid = await (ctx.costLedger ?? directCostLedger).runPaidCall({
|
|
5507
|
+
channel: "driver",
|
|
5508
|
+
phase: ctx.costPhase ?? "search.proposal",
|
|
5509
|
+
actor: "llm-policy-edit.author",
|
|
5510
|
+
model: opts.model,
|
|
5511
|
+
maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
|
|
5512
|
+
tags: { generation: String(ctx.generation) },
|
|
5513
|
+
signal: ctx.signal,
|
|
5514
|
+
execute: (signal, callId) => callLlmJson(request, { ...opts.llm, signal, idempotencyKey: callId }),
|
|
5515
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
5516
|
+
receiptFromError: costReceiptFromLlmError
|
|
5517
|
+
});
|
|
5518
|
+
if (!paid.succeeded) throw paid.error;
|
|
5519
|
+
const { value } = paid.value;
|
|
5754
5520
|
const response = parseAuthorResponse(value);
|
|
5755
5521
|
if (response.edits.length > limit) {
|
|
5756
5522
|
throw new Error(
|
|
@@ -6432,18 +6198,35 @@ var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, gener
|
|
|
6432
6198
|
function extractExistingLessons(text) {
|
|
6433
6199
|
return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
|
|
6434
6200
|
}
|
|
6435
|
-
async function distillLessons(raw, distill) {
|
|
6436
|
-
const
|
|
6437
|
-
|
|
6438
|
-
|
|
6439
|
-
|
|
6440
|
-
|
|
6441
|
-
{ role: "user", content: `Findings:
|
|
6201
|
+
async function distillLessons(raw, distill, ctx, costLedger) {
|
|
6202
|
+
const request = {
|
|
6203
|
+
model: distill.model,
|
|
6204
|
+
messages: [
|
|
6205
|
+
{ role: "system", content: DISTILL_SYSTEM },
|
|
6206
|
+
{ role: "user", content: `Findings:
|
|
6442
6207
|
${raw.map((r) => `- ${r}`).join("\n")}` }
|
|
6443
|
-
|
|
6444
|
-
|
|
6445
|
-
|
|
6446
|
-
|
|
6208
|
+
],
|
|
6209
|
+
maxTokens: distill.maxTokens ?? 2e3
|
|
6210
|
+
};
|
|
6211
|
+
const llm = {
|
|
6212
|
+
baseUrl: distill.baseUrl,
|
|
6213
|
+
apiKey: distill.apiKey,
|
|
6214
|
+
fetch: distill.fetchImpl
|
|
6215
|
+
};
|
|
6216
|
+
const paid = await costLedger.runPaidCall({
|
|
6217
|
+
channel: "driver",
|
|
6218
|
+
phase: ctx.costPhase ?? "search.proposal",
|
|
6219
|
+
actor: "memory-curation.distill",
|
|
6220
|
+
model: distill.model,
|
|
6221
|
+
maximumCharge: maximumChargeForLlmRequest(request, llm),
|
|
6222
|
+
tags: { generation: String(ctx.generation) },
|
|
6223
|
+
signal: ctx.signal,
|
|
6224
|
+
execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
|
|
6225
|
+
receipt: costReceiptFromLlm,
|
|
6226
|
+
receiptFromError: costReceiptFromLlmError
|
|
6227
|
+
});
|
|
6228
|
+
if (!paid.succeeded) throw paid.error;
|
|
6229
|
+
const res = paid.value;
|
|
6447
6230
|
try {
|
|
6448
6231
|
const parsed = JSON.parse(res.content.trim());
|
|
6449
6232
|
if (Array.isArray(parsed)) {
|
|
@@ -6459,6 +6242,7 @@ ${raw.map((r) => `- ${r}`).join("\n")}` }
|
|
|
6459
6242
|
function memoryCurationProposer(opts = {}) {
|
|
6460
6243
|
const maxEntries = opts.maxEntries ?? 12;
|
|
6461
6244
|
const heading = opts.sectionHeading ?? DEFAULT_HEADING2;
|
|
6245
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
6462
6246
|
return {
|
|
6463
6247
|
kind: "memory-curation",
|
|
6464
6248
|
async propose(ctx) {
|
|
@@ -6470,7 +6254,7 @@ function memoryCurationProposer(opts = {}) {
|
|
|
6470
6254
|
}
|
|
6471
6255
|
const carried = extractExistingLessons(parent);
|
|
6472
6256
|
if (fresh.length === 0 && carried.length === 0) return [];
|
|
6473
|
-
const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill) : fresh;
|
|
6257
|
+
const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill, ctx, ctx.costLedger ?? directCostLedger) : fresh;
|
|
6474
6258
|
const byKey = /* @__PURE__ */ new Map();
|
|
6475
6259
|
for (const l of carried) {
|
|
6476
6260
|
const k = normKey(l);
|
|
@@ -6541,7 +6325,12 @@ function traceAnalystProposer(opts) {
|
|
|
6541
6325
|
label: "trace-analyst",
|
|
6542
6326
|
baseUrl: opts.baseUrl,
|
|
6543
6327
|
apiKey: opts.apiKey,
|
|
6328
|
+
analysisModel: opts.model,
|
|
6544
6329
|
applyModel: opts.applyModel ?? opts.model,
|
|
6330
|
+
costLedger: opts.costLedger,
|
|
6331
|
+
analysisMaximumCharge: opts.analysisMaximumCharge,
|
|
6332
|
+
analysisReceipt: opts.analysisReceipt,
|
|
6333
|
+
applyMaxTokens: opts.applyMaxTokens,
|
|
6545
6334
|
fetchImpl: opts.fetchImpl,
|
|
6546
6335
|
resolveTraces: opts.resolveTraces,
|
|
6547
6336
|
noTracesError: "traceAnalystProposer: resolveTraces returned no OTLP traces \u2014 the analyst has nothing to read",
|
|
@@ -6613,238 +6402,86 @@ function selectDiscriminative(signals, k, opts) {
|
|
|
6613
6402
|
|
|
6614
6403
|
// src/campaign/search-ledger.ts
|
|
6615
6404
|
import { createHash as createHash7 } from "crypto";
|
|
6616
|
-
import { existsSync as
|
|
6405
|
+
import { existsSync as existsSync3, readFileSync as readFileSync3 } from "fs";
|
|
6617
6406
|
import { resolve as resolve2 } from "path";
|
|
6618
|
-
import { z as z3 } from "zod";
|
|
6619
|
-
|
|
6620
|
-
// src/campaign/search-ledger-errors.ts
|
|
6621
|
-
var SearchLedgerError = class extends ValidationError {
|
|
6622
|
-
};
|
|
6623
|
-
var SearchLedgerIntegrityError = class extends SearchLedgerError {
|
|
6624
|
-
};
|
|
6625
|
-
var SearchLedgerConflictError = class extends SearchLedgerError {
|
|
6626
|
-
};
|
|
6627
|
-
|
|
6628
|
-
// src/campaign/search-ledger-file.ts
|
|
6629
|
-
import { randomUUID as randomUUID2 } from "crypto";
|
|
6630
|
-
import {
|
|
6631
|
-
closeSync,
|
|
6632
|
-
constants,
|
|
6633
|
-
existsSync as existsSync3,
|
|
6634
|
-
fsyncSync,
|
|
6635
|
-
linkSync,
|
|
6636
|
-
mkdirSync as mkdirSync2,
|
|
6637
|
-
openSync,
|
|
6638
|
-
readFileSync as readFileSync3,
|
|
6639
|
-
renameSync,
|
|
6640
|
-
unlinkSync,
|
|
6641
|
-
writeSync
|
|
6642
|
-
} from "fs";
|
|
6643
|
-
import { hostname } from "os";
|
|
6644
|
-
import { dirname as dirname2 } from "path";
|
|
6645
6407
|
import { z as z2 } from "zod";
|
|
6646
|
-
function appendSearchLedgerLine(path, line) {
|
|
6647
|
-
mkdirSync2(dirname2(path), { recursive: true });
|
|
6648
|
-
const fd = openSync(path, constants.O_CREAT | constants.O_WRONLY | constants.O_APPEND, 384);
|
|
6649
|
-
try {
|
|
6650
|
-
writeAll(fd, Buffer.from(line, "utf8"));
|
|
6651
|
-
fsyncSync(fd);
|
|
6652
|
-
} finally {
|
|
6653
|
-
closeSync(fd);
|
|
6654
|
-
}
|
|
6655
|
-
fsyncDirectory(dirname2(path));
|
|
6656
|
-
}
|
|
6657
|
-
function withSearchLedgerFileLock(ledgerPath, run) {
|
|
6658
|
-
mkdirSync2(dirname2(ledgerPath), { recursive: true });
|
|
6659
|
-
const lockPath = `${ledgerPath}.lock`;
|
|
6660
|
-
const owner = acquireLock(lockPath);
|
|
6661
|
-
try {
|
|
6662
|
-
return run();
|
|
6663
|
-
} finally {
|
|
6664
|
-
releaseLock(lockPath, owner);
|
|
6665
|
-
}
|
|
6666
|
-
}
|
|
6667
|
-
function writeAll(fd, bytes) {
|
|
6668
|
-
let offset = 0;
|
|
6669
|
-
while (offset < bytes.byteLength) {
|
|
6670
|
-
const written = writeSync(fd, bytes, offset, bytes.byteLength - offset);
|
|
6671
|
-
if (written <= 0) throw new SearchLedgerIntegrityError("filesystem wrote zero bytes");
|
|
6672
|
-
offset += written;
|
|
6673
|
-
}
|
|
6674
|
-
}
|
|
6675
|
-
function fsyncDirectory(path) {
|
|
6676
|
-
const fd = openSync(path, constants.O_RDONLY);
|
|
6677
|
-
try {
|
|
6678
|
-
fsyncSync(fd);
|
|
6679
|
-
} finally {
|
|
6680
|
-
closeSync(fd);
|
|
6681
|
-
}
|
|
6682
|
-
}
|
|
6683
|
-
function acquireLock(lockPath) {
|
|
6684
|
-
const owner = { pid: process.pid, host: hostname(), nonce: randomUUID2() };
|
|
6685
|
-
const ownerBytes = `${canonicalOwner(owner)}
|
|
6686
|
-
`;
|
|
6687
|
-
for (let attempt = 0; attempt < 8; attempt += 1) {
|
|
6688
|
-
const ownerPath = `${lockPath}.${owner.pid}.${owner.nonce}.${attempt}.owner`;
|
|
6689
|
-
const fd = openSync(ownerPath, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY, 384);
|
|
6690
|
-
try {
|
|
6691
|
-
writeAll(fd, Buffer.from(ownerBytes, "utf8"));
|
|
6692
|
-
fsyncSync(fd);
|
|
6693
|
-
} finally {
|
|
6694
|
-
closeSync(fd);
|
|
6695
|
-
}
|
|
6696
|
-
try {
|
|
6697
|
-
linkSync(ownerPath, lockPath);
|
|
6698
|
-
unlinkSync(ownerPath);
|
|
6699
|
-
return owner;
|
|
6700
|
-
} catch (error) {
|
|
6701
|
-
unlinkIfExists(ownerPath);
|
|
6702
|
-
if (error.code !== "EEXIST") throw error;
|
|
6703
|
-
}
|
|
6704
|
-
const holder = readOwner(lockPath);
|
|
6705
|
-
if (holder.host !== owner.host || isProcessAlive(holder.pid)) {
|
|
6706
|
-
throw new SearchLedgerIntegrityError(
|
|
6707
|
-
`search ledger lock is held by pid ${holder.pid} on ${holder.host}`
|
|
6708
|
-
);
|
|
6709
|
-
}
|
|
6710
|
-
const tombstone = `${lockPath}.stale.${owner.nonce}.${attempt}`;
|
|
6711
|
-
try {
|
|
6712
|
-
renameSync(lockPath, tombstone);
|
|
6713
|
-
unlinkSync(tombstone);
|
|
6714
|
-
} catch (error) {
|
|
6715
|
-
if (error.code !== "ENOENT") throw error;
|
|
6716
|
-
}
|
|
6717
|
-
}
|
|
6718
|
-
throw new SearchLedgerIntegrityError(`could not acquire search ledger lock ${lockPath}`);
|
|
6719
|
-
}
|
|
6720
|
-
function releaseLock(lockPath, owner) {
|
|
6721
|
-
if (!existsSync3(lockPath)) return;
|
|
6722
|
-
const holder = readOwner(lockPath);
|
|
6723
|
-
if (canonicalOwner(holder) !== canonicalOwner(owner)) {
|
|
6724
|
-
throw new SearchLedgerIntegrityError(
|
|
6725
|
-
`search ledger lock owner changed before release (${lockPath})`
|
|
6726
|
-
);
|
|
6727
|
-
}
|
|
6728
|
-
unlinkSync(lockPath);
|
|
6729
|
-
}
|
|
6730
|
-
function readOwner(lockPath) {
|
|
6731
|
-
let raw;
|
|
6732
|
-
try {
|
|
6733
|
-
raw = JSON.parse(readFileSync3(lockPath, "utf8"));
|
|
6734
|
-
} catch (error) {
|
|
6735
|
-
throw new SearchLedgerIntegrityError(`search ledger lock ${lockPath} is malformed`, {
|
|
6736
|
-
cause: error
|
|
6737
|
-
});
|
|
6738
|
-
}
|
|
6739
|
-
const parsed = z2.object({
|
|
6740
|
-
pid: z2.number().int().positive().safe(),
|
|
6741
|
-
host: z2.string().trim().min(1),
|
|
6742
|
-
nonce: z2.string().trim().min(1)
|
|
6743
|
-
}).strict().safeParse(raw);
|
|
6744
|
-
if (!parsed.success) {
|
|
6745
|
-
throw new SearchLedgerIntegrityError(
|
|
6746
|
-
`search ledger lock ${lockPath} is malformed: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; ")}`
|
|
6747
|
-
);
|
|
6748
|
-
}
|
|
6749
|
-
return parsed.data;
|
|
6750
|
-
}
|
|
6751
|
-
function canonicalOwner(owner) {
|
|
6752
|
-
return JSON.stringify({ host: owner.host, nonce: owner.nonce, pid: owner.pid });
|
|
6753
|
-
}
|
|
6754
|
-
function isProcessAlive(pid) {
|
|
6755
|
-
try {
|
|
6756
|
-
process.kill(pid, 0);
|
|
6757
|
-
return true;
|
|
6758
|
-
} catch (error) {
|
|
6759
|
-
return error.code !== "ESRCH";
|
|
6760
|
-
}
|
|
6761
|
-
}
|
|
6762
|
-
function unlinkIfExists(path) {
|
|
6763
|
-
try {
|
|
6764
|
-
unlinkSync(path);
|
|
6765
|
-
} catch (error) {
|
|
6766
|
-
if (error.code !== "ENOENT") throw error;
|
|
6767
|
-
}
|
|
6768
|
-
}
|
|
6769
|
-
|
|
6770
|
-
// src/campaign/search-ledger.ts
|
|
6771
6408
|
var SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
|
|
6772
|
-
var NON_EMPTY =
|
|
6773
|
-
var HASH =
|
|
6774
|
-
var LINEAGE_NODE_ID =
|
|
6775
|
-
var IMMUTABLE_REVISION =
|
|
6776
|
-
var ISO_TIMESTAMP =
|
|
6777
|
-
var NON_NEGATIVE_INT =
|
|
6778
|
-
var FINITE_NUMBER =
|
|
6779
|
-
var ArtifactRefSchema =
|
|
6409
|
+
var NON_EMPTY = z2.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
|
|
6410
|
+
var HASH = z2.string().regex(/^sha256:[a-f0-9]{64}$/);
|
|
6411
|
+
var LINEAGE_NODE_ID = z2.string().regex(/^[a-f0-9]{16}$/);
|
|
6412
|
+
var IMMUTABLE_REVISION = z2.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
|
|
6413
|
+
var ISO_TIMESTAMP = z2.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
|
|
6414
|
+
var NON_NEGATIVE_INT = z2.number().int().nonnegative().safe();
|
|
6415
|
+
var FINITE_NUMBER = z2.number().finite();
|
|
6416
|
+
var ArtifactRefSchema = z2.object({
|
|
6780
6417
|
role: NON_EMPTY,
|
|
6781
6418
|
uri: NON_EMPTY,
|
|
6782
6419
|
sha256: HASH,
|
|
6783
6420
|
byteLength: NON_NEGATIVE_INT
|
|
6784
6421
|
}).strict();
|
|
6785
|
-
var SourceRefSchema =
|
|
6422
|
+
var SourceRefSchema = z2.object({
|
|
6786
6423
|
uri: NON_EMPTY,
|
|
6787
6424
|
revision: IMMUTABLE_REVISION
|
|
6788
6425
|
}).strict();
|
|
6789
|
-
var FailureReasonSchema =
|
|
6426
|
+
var FailureReasonSchema = z2.object({
|
|
6790
6427
|
code: NON_EMPTY,
|
|
6791
6428
|
message: NON_EMPTY
|
|
6792
6429
|
}).strict();
|
|
6793
6430
|
var EventBaseShape = {
|
|
6794
6431
|
eventId: NON_EMPTY,
|
|
6795
6432
|
occurredAt: ISO_TIMESTAMP,
|
|
6796
|
-
artifacts:
|
|
6433
|
+
artifacts: z2.array(ArtifactRefSchema).min(1)
|
|
6797
6434
|
};
|
|
6798
|
-
var OperationKindSchema =
|
|
6435
|
+
var OperationKindSchema = z2.enum([
|
|
6799
6436
|
"candidate-generation",
|
|
6800
6437
|
"analysis",
|
|
6801
6438
|
"selection",
|
|
6802
6439
|
"judge",
|
|
6803
6440
|
"other"
|
|
6804
6441
|
]);
|
|
6805
|
-
var SearchPlannedSchema =
|
|
6442
|
+
var SearchPlannedSchema = z2.object({
|
|
6806
6443
|
...EventBaseShape,
|
|
6807
|
-
kind:
|
|
6808
|
-
plan:
|
|
6809
|
-
candidateSlots:
|
|
6810
|
-
|
|
6444
|
+
kind: z2.literal("search-planned"),
|
|
6445
|
+
plan: z2.object({
|
|
6446
|
+
candidateSlots: z2.array(
|
|
6447
|
+
z2.object({
|
|
6811
6448
|
slotId: NON_EMPTY,
|
|
6812
6449
|
generationOperationId: NON_EMPTY
|
|
6813
6450
|
}).strict()
|
|
6814
6451
|
).min(1),
|
|
6815
|
-
tasks:
|
|
6816
|
-
|
|
6452
|
+
tasks: z2.array(
|
|
6453
|
+
z2.object({
|
|
6817
6454
|
taskId: NON_EMPTY,
|
|
6818
6455
|
source: SourceRefSchema,
|
|
6819
6456
|
benchmark: SourceRefSchema,
|
|
6820
|
-
maxAttempts:
|
|
6457
|
+
maxAttempts: z2.number().int().positive().safe()
|
|
6821
6458
|
}).strict()
|
|
6822
6459
|
).min(1),
|
|
6823
|
-
operations:
|
|
6824
|
-
|
|
6460
|
+
operations: z2.array(
|
|
6461
|
+
z2.object({
|
|
6825
6462
|
operationId: NON_EMPTY,
|
|
6826
6463
|
kind: OperationKindSchema
|
|
6827
6464
|
}).strict()
|
|
6828
6465
|
).min(1)
|
|
6829
6466
|
}).strict()
|
|
6830
6467
|
}).strict();
|
|
6831
|
-
var CandidateRegisteredSchema =
|
|
6468
|
+
var CandidateRegisteredSchema = z2.object({
|
|
6832
6469
|
...EventBaseShape,
|
|
6833
|
-
kind:
|
|
6470
|
+
kind: z2.literal("candidate-registered"),
|
|
6834
6471
|
slotId: NON_EMPTY,
|
|
6835
6472
|
generationOperationId: NON_EMPTY,
|
|
6836
6473
|
candidateId: NON_EMPTY,
|
|
6837
|
-
lineage:
|
|
6474
|
+
lineage: z2.object({
|
|
6838
6475
|
lineageNodeId: LINEAGE_NODE_ID,
|
|
6839
|
-
parentCandidateIds:
|
|
6476
|
+
parentCandidateIds: z2.array(NON_EMPTY),
|
|
6840
6477
|
generation: NON_NEGATIVE_INT,
|
|
6841
6478
|
proposer: NON_EMPTY,
|
|
6842
6479
|
proposerSource: SourceRefSchema
|
|
6843
6480
|
}).strict(),
|
|
6844
|
-
surfaces:
|
|
6845
|
-
|
|
6481
|
+
surfaces: z2.array(
|
|
6482
|
+
z2.object({
|
|
6846
6483
|
surfaceId: NON_EMPTY,
|
|
6847
|
-
kind:
|
|
6484
|
+
kind: z2.enum([
|
|
6848
6485
|
"prompt",
|
|
6849
6486
|
"tool-contract",
|
|
6850
6487
|
"runtime-config",
|
|
@@ -6858,69 +6495,69 @@ var CandidateRegisteredSchema = z3.object({
|
|
|
6858
6495
|
}).strict()
|
|
6859
6496
|
).min(1)
|
|
6860
6497
|
}).strict();
|
|
6861
|
-
var CandidateSlotClosedSchema =
|
|
6498
|
+
var CandidateSlotClosedSchema = z2.object({
|
|
6862
6499
|
...EventBaseShape,
|
|
6863
|
-
kind:
|
|
6500
|
+
kind: z2.literal("candidate-slot-closed"),
|
|
6864
6501
|
slotId: NON_EMPTY,
|
|
6865
6502
|
generationOperationId: NON_EMPTY,
|
|
6866
6503
|
reason: FailureReasonSchema
|
|
6867
6504
|
}).strict();
|
|
6868
|
-
var KnownTokensSchema =
|
|
6869
|
-
status:
|
|
6505
|
+
var KnownTokensSchema = z2.object({
|
|
6506
|
+
status: z2.literal("known"),
|
|
6870
6507
|
inputTokens: NON_NEGATIVE_INT,
|
|
6871
6508
|
outputTokens: NON_NEGATIVE_INT,
|
|
6872
6509
|
cachedTokens: NON_NEGATIVE_INT
|
|
6873
6510
|
}).strict();
|
|
6874
|
-
var UnknownSchema =
|
|
6875
|
-
status:
|
|
6511
|
+
var UnknownSchema = z2.object({
|
|
6512
|
+
status: z2.literal("unknown"),
|
|
6876
6513
|
reason: NON_EMPTY
|
|
6877
6514
|
}).strict();
|
|
6878
|
-
var KnownCostSchema =
|
|
6879
|
-
status:
|
|
6880
|
-
usd:
|
|
6881
|
-
source:
|
|
6515
|
+
var KnownCostSchema = z2.object({
|
|
6516
|
+
status: z2.literal("known"),
|
|
6517
|
+
usd: z2.number().finite().nonnegative(),
|
|
6518
|
+
source: z2.enum(["provider", "pricing-table", "free"])
|
|
6882
6519
|
}).strict().superRefine((cost, ctx) => {
|
|
6883
6520
|
if (cost.source === "free" && cost.usd !== 0) {
|
|
6884
6521
|
ctx.addIssue({ code: "custom", message: "free cost source must have usd 0" });
|
|
6885
6522
|
}
|
|
6886
6523
|
});
|
|
6887
|
-
var UnknownCostSchema =
|
|
6888
|
-
status:
|
|
6889
|
-
knownLowerBoundUsd:
|
|
6524
|
+
var UnknownCostSchema = z2.object({
|
|
6525
|
+
status: z2.literal("unknown"),
|
|
6526
|
+
knownLowerBoundUsd: z2.number().finite().nonnegative(),
|
|
6890
6527
|
reason: NON_EMPTY
|
|
6891
6528
|
}).strict();
|
|
6892
|
-
var AccountingSchema =
|
|
6893
|
-
tokens:
|
|
6894
|
-
cost:
|
|
6529
|
+
var AccountingSchema = z2.object({
|
|
6530
|
+
tokens: z2.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
|
|
6531
|
+
cost: z2.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
|
|
6895
6532
|
}).strict();
|
|
6896
|
-
var MetricsSchema =
|
|
6533
|
+
var MetricsSchema = z2.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
|
|
6897
6534
|
for (const key of Object.keys(metrics)) {
|
|
6898
6535
|
if (key === "__proto__" || key === "constructor" || key === "prototype") {
|
|
6899
6536
|
ctx.addIssue({ code: "custom", message: `unsafe metric key ${key}` });
|
|
6900
6537
|
}
|
|
6901
6538
|
}
|
|
6902
6539
|
});
|
|
6903
|
-
var OutcomeSchema =
|
|
6904
|
-
|
|
6905
|
-
status:
|
|
6540
|
+
var OutcomeSchema = z2.discriminatedUnion("status", [
|
|
6541
|
+
z2.object({
|
|
6542
|
+
status: z2.literal("passed"),
|
|
6906
6543
|
score: FINITE_NUMBER,
|
|
6907
6544
|
metrics: MetricsSchema
|
|
6908
6545
|
}).strict(),
|
|
6909
|
-
|
|
6910
|
-
status:
|
|
6546
|
+
z2.object({
|
|
6547
|
+
status: z2.literal("failed"),
|
|
6911
6548
|
score: FINITE_NUMBER,
|
|
6912
6549
|
metrics: MetricsSchema,
|
|
6913
6550
|
failure: FailureReasonSchema
|
|
6914
6551
|
}).strict(),
|
|
6915
|
-
|
|
6916
|
-
status:
|
|
6552
|
+
z2.object({
|
|
6553
|
+
status: z2.literal("errored"),
|
|
6917
6554
|
metrics: MetricsSchema,
|
|
6918
|
-
error: FailureReasonSchema.extend({ retryable:
|
|
6555
|
+
error: FailureReasonSchema.extend({ retryable: z2.boolean() }).strict()
|
|
6919
6556
|
}).strict()
|
|
6920
6557
|
]);
|
|
6921
|
-
var EffectSchema =
|
|
6922
|
-
|
|
6923
|
-
status:
|
|
6558
|
+
var EffectSchema = z2.discriminatedUnion("status", [
|
|
6559
|
+
z2.object({
|
|
6560
|
+
status: z2.literal("measured"),
|
|
6924
6561
|
metric: NON_EMPTY,
|
|
6925
6562
|
baselineValue: FINITE_NUMBER,
|
|
6926
6563
|
candidateValue: FINITE_NUMBER,
|
|
@@ -6932,17 +6569,17 @@ var EffectSchema = z3.discriminatedUnion("status", [
|
|
|
6932
6569
|
ctx.addIssue({ code: "custom", message: "delta must equal candidateValue - baselineValue" });
|
|
6933
6570
|
}
|
|
6934
6571
|
}),
|
|
6935
|
-
|
|
6936
|
-
status:
|
|
6572
|
+
z2.object({
|
|
6573
|
+
status: z2.literal("not-measured"),
|
|
6937
6574
|
reason: NON_EMPTY
|
|
6938
6575
|
}).strict()
|
|
6939
6576
|
]);
|
|
6940
|
-
var SurfaceEvidenceSchema =
|
|
6577
|
+
var SurfaceEvidenceSchema = z2.object({
|
|
6941
6578
|
surfaceId: NON_EMPTY,
|
|
6942
|
-
fired:
|
|
6579
|
+
fired: z2.boolean(),
|
|
6943
6580
|
firingCount: NON_NEGATIVE_INT,
|
|
6944
6581
|
effect: EffectSchema,
|
|
6945
|
-
evidence:
|
|
6582
|
+
evidence: z2.array(ArtifactRefSchema).min(1)
|
|
6946
6583
|
}).strict().superRefine((evidence, ctx) => {
|
|
6947
6584
|
if (evidence.fired && evidence.firingCount === 0) {
|
|
6948
6585
|
ctx.addIssue({ code: "custom", message: "a fired surface must have firingCount >= 1" });
|
|
@@ -6960,15 +6597,15 @@ var SurfaceEvidenceSchema = z3.object({
|
|
|
6960
6597
|
});
|
|
6961
6598
|
}
|
|
6962
6599
|
});
|
|
6963
|
-
var TaskAttemptedSchema =
|
|
6600
|
+
var TaskAttemptedSchema = z2.object({
|
|
6964
6601
|
...EventBaseShape,
|
|
6965
|
-
kind:
|
|
6602
|
+
kind: z2.literal("task-attempted"),
|
|
6966
6603
|
candidateId: NON_EMPTY,
|
|
6967
6604
|
runId: NON_EMPTY,
|
|
6968
6605
|
attemptIndex: NON_NEGATIVE_INT,
|
|
6969
|
-
task:
|
|
6970
|
-
identity:
|
|
6971
|
-
model:
|
|
6606
|
+
task: z2.object({ taskId: NON_EMPTY, source: SourceRefSchema }).strict(),
|
|
6607
|
+
identity: z2.object({
|
|
6608
|
+
model: z2.object({
|
|
6972
6609
|
provider: NON_EMPTY,
|
|
6973
6610
|
snapshot: NON_EMPTY.refine(
|
|
6974
6611
|
modelHasSnapshot,
|
|
@@ -6980,17 +6617,17 @@ var TaskAttemptedSchema = z3.object({
|
|
|
6980
6617
|
}).strict(),
|
|
6981
6618
|
outcome: OutcomeSchema,
|
|
6982
6619
|
accounting: AccountingSchema,
|
|
6983
|
-
surfaceEvidence:
|
|
6620
|
+
surfaceEvidence: z2.array(SurfaceEvidenceSchema).min(1)
|
|
6984
6621
|
}).strict();
|
|
6985
|
-
var SearchOperationRecordedSchema =
|
|
6622
|
+
var SearchOperationRecordedSchema = z2.object({
|
|
6986
6623
|
...EventBaseShape,
|
|
6987
|
-
kind:
|
|
6624
|
+
kind: z2.literal("search-operation-recorded"),
|
|
6988
6625
|
operationId: NON_EMPTY,
|
|
6989
6626
|
operationKind: OperationKindSchema,
|
|
6990
|
-
execution:
|
|
6991
|
-
|
|
6992
|
-
kind:
|
|
6993
|
-
model:
|
|
6627
|
+
execution: z2.discriminatedUnion("kind", [
|
|
6628
|
+
z2.object({
|
|
6629
|
+
kind: z2.literal("model"),
|
|
6630
|
+
model: z2.object({
|
|
6994
6631
|
provider: NON_EMPTY,
|
|
6995
6632
|
snapshot: NON_EMPTY.refine(
|
|
6996
6633
|
modelHasSnapshot,
|
|
@@ -6999,51 +6636,51 @@ var SearchOperationRecordedSchema = z3.object({
|
|
|
6999
6636
|
}).strict(),
|
|
7000
6637
|
source: SourceRefSchema
|
|
7001
6638
|
}).strict(),
|
|
7002
|
-
|
|
7003
|
-
kind:
|
|
6639
|
+
z2.object({
|
|
6640
|
+
kind: z2.literal("deterministic"),
|
|
7004
6641
|
source: SourceRefSchema
|
|
7005
6642
|
}).strict()
|
|
7006
6643
|
]),
|
|
7007
|
-
outcome:
|
|
7008
|
-
|
|
7009
|
-
|
|
7010
|
-
status:
|
|
6644
|
+
outcome: z2.discriminatedUnion("status", [
|
|
6645
|
+
z2.object({ status: z2.literal("completed") }).strict(),
|
|
6646
|
+
z2.object({
|
|
6647
|
+
status: z2.literal("partial"),
|
|
7011
6648
|
failure: FailureReasonSchema
|
|
7012
6649
|
}).strict(),
|
|
7013
|
-
|
|
7014
|
-
status:
|
|
6650
|
+
z2.object({
|
|
6651
|
+
status: z2.literal("failed"),
|
|
7015
6652
|
failure: FailureReasonSchema
|
|
7016
6653
|
}).strict()
|
|
7017
6654
|
]),
|
|
7018
6655
|
accounting: AccountingSchema
|
|
7019
6656
|
}).strict();
|
|
7020
|
-
var CandidateDecidedSchema =
|
|
6657
|
+
var CandidateDecidedSchema = z2.object({
|
|
7021
6658
|
...EventBaseShape,
|
|
7022
|
-
kind:
|
|
6659
|
+
kind: z2.literal("candidate-decided"),
|
|
7023
6660
|
candidateId: NON_EMPTY,
|
|
7024
|
-
decision:
|
|
7025
|
-
|
|
7026
|
-
|
|
7027
|
-
status:
|
|
6661
|
+
decision: z2.discriminatedUnion("status", [
|
|
6662
|
+
z2.object({ status: z2.literal("selected") }).strict(),
|
|
6663
|
+
z2.object({
|
|
6664
|
+
status: z2.literal("rejected"),
|
|
7028
6665
|
reason: FailureReasonSchema
|
|
7029
6666
|
}).strict()
|
|
7030
6667
|
])
|
|
7031
6668
|
}).strict();
|
|
7032
|
-
var SearchCompletedSchema =
|
|
6669
|
+
var SearchCompletedSchema = z2.object({
|
|
7033
6670
|
...EventBaseShape,
|
|
7034
|
-
kind:
|
|
7035
|
-
result:
|
|
7036
|
-
|
|
7037
|
-
status:
|
|
6671
|
+
kind: z2.literal("search-completed"),
|
|
6672
|
+
result: z2.discriminatedUnion("status", [
|
|
6673
|
+
z2.object({
|
|
6674
|
+
status: z2.literal("selected"),
|
|
7038
6675
|
candidateId: NON_EMPTY
|
|
7039
6676
|
}).strict(),
|
|
7040
|
-
|
|
7041
|
-
status:
|
|
6677
|
+
z2.object({
|
|
6678
|
+
status: z2.literal("all-rejected"),
|
|
7042
6679
|
reason: FailureReasonSchema
|
|
7043
6680
|
}).strict()
|
|
7044
6681
|
])
|
|
7045
6682
|
}).strict();
|
|
7046
|
-
var EventSchema =
|
|
6683
|
+
var EventSchema = z2.discriminatedUnion("kind", [
|
|
7047
6684
|
SearchPlannedSchema,
|
|
7048
6685
|
CandidateRegisteredSchema,
|
|
7049
6686
|
CandidateSlotClosedSchema,
|
|
@@ -7052,11 +6689,11 @@ var EventSchema = z3.discriminatedUnion("kind", [
|
|
|
7052
6689
|
CandidateDecidedSchema,
|
|
7053
6690
|
SearchCompletedSchema
|
|
7054
6691
|
]);
|
|
7055
|
-
var EntrySchema =
|
|
7056
|
-
schema:
|
|
6692
|
+
var EntrySchema = z2.object({
|
|
6693
|
+
schema: z2.literal(SEARCH_LEDGER_SCHEMA),
|
|
7057
6694
|
campaignId: NON_EMPTY,
|
|
7058
6695
|
sequence: NON_NEGATIVE_INT,
|
|
7059
|
-
previousHash:
|
|
6696
|
+
previousHash: z2.union([HASH, z2.null()]),
|
|
7060
6697
|
event: EventSchema,
|
|
7061
6698
|
entryHash: HASH
|
|
7062
6699
|
}).strict();
|
|
@@ -7133,8 +6770,8 @@ var FileSearchLedger = class {
|
|
|
7133
6770
|
}
|
|
7134
6771
|
};
|
|
7135
6772
|
function replayFile(path, campaignId) {
|
|
7136
|
-
if (!
|
|
7137
|
-
const text =
|
|
6773
|
+
if (!existsSync3(path)) return replayEntries([], campaignId);
|
|
6774
|
+
const text = readFileSync3(path, "utf8");
|
|
7138
6775
|
if (text.length === 0) return replayEntries([], campaignId);
|
|
7139
6776
|
if (!text.endsWith("\n")) {
|
|
7140
6777
|
throw new SearchLedgerIntegrityError(
|
|
@@ -7730,10 +7367,10 @@ function formatZodError(error) {
|
|
|
7730
7367
|
}
|
|
7731
7368
|
|
|
7732
7369
|
// src/campaign/single-run-lock.ts
|
|
7733
|
-
import { existsSync as
|
|
7370
|
+
import { existsSync as existsSync4, readFileSync as readFileSync4, unlinkSync, writeFileSync as writeFileSync3 } from "fs";
|
|
7734
7371
|
function liveHolder(path) {
|
|
7735
|
-
if (!
|
|
7736
|
-
const holder = Number(
|
|
7372
|
+
if (!existsSync4(path)) return null;
|
|
7373
|
+
const holder = Number(readFileSync4(path, "utf8").trim());
|
|
7737
7374
|
if (!Number.isFinite(holder) || holder <= 0) return null;
|
|
7738
7375
|
try {
|
|
7739
7376
|
process.kill(holder, 0);
|
|
@@ -7755,8 +7392,8 @@ function acquireSingleRunLock(opts) {
|
|
|
7755
7392
|
writeFileSync3(opts.lockPath, String(pid));
|
|
7756
7393
|
const release = () => {
|
|
7757
7394
|
try {
|
|
7758
|
-
if (
|
|
7759
|
-
|
|
7395
|
+
if (existsSync4(opts.lockPath) && readFileSync4(opts.lockPath, "utf8").trim() === String(pid)) {
|
|
7396
|
+
unlinkSync(opts.lockPath);
|
|
7760
7397
|
}
|
|
7761
7398
|
} catch {
|
|
7762
7399
|
}
|
|
@@ -7780,21 +7417,21 @@ function isTransientTransportFailure(message, opts = {}) {
|
|
|
7780
7417
|
import { execFileSync } from "child_process";
|
|
7781
7418
|
import { createHash as createHash8 } from "crypto";
|
|
7782
7419
|
import {
|
|
7783
|
-
closeSync
|
|
7784
|
-
existsSync as
|
|
7420
|
+
closeSync,
|
|
7421
|
+
existsSync as existsSync5,
|
|
7785
7422
|
constants as fsConstants,
|
|
7786
7423
|
fstatSync,
|
|
7787
7424
|
lstatSync,
|
|
7788
|
-
mkdirSync as
|
|
7425
|
+
mkdirSync as mkdirSync2,
|
|
7789
7426
|
mkdtempSync as mkdtempSync2,
|
|
7790
|
-
openSync
|
|
7427
|
+
openSync,
|
|
7791
7428
|
readlinkSync,
|
|
7792
7429
|
readSync,
|
|
7793
7430
|
realpathSync,
|
|
7794
7431
|
rmSync
|
|
7795
7432
|
} from "fs";
|
|
7796
7433
|
import { devNull, tmpdir as tmpdir2 } from "os";
|
|
7797
|
-
import { basename, dirname as
|
|
7434
|
+
import { basename, dirname as dirname2, isAbsolute as isAbsolute2, join as join5, relative as relative2, resolve as resolve3, sep } from "path";
|
|
7798
7435
|
var MAX_GIT_OUTPUT_BYTES = 256 * 1024 * 1024;
|
|
7799
7436
|
var FILE_HASH_CHUNK_BYTES = 1024 * 1024;
|
|
7800
7437
|
var GIT_REPOSITORY_ENV = /* @__PURE__ */ new Set([
|
|
@@ -7846,6 +7483,45 @@ function gitBytes(git, args, cwd, env) {
|
|
|
7846
7483
|
function gitText(git, args, cwd, env) {
|
|
7847
7484
|
return gitBytes(git, args, cwd, env).toString("utf8").trim();
|
|
7848
7485
|
}
|
|
7486
|
+
function hasRegisteredWorktree(git, repoRoot, path) {
|
|
7487
|
+
const expected = Buffer.from(`worktree ${resolve3(path)}`, "utf8");
|
|
7488
|
+
const records = gitBytes(git, ["worktree", "list", "--porcelain", "-z"], repoRoot);
|
|
7489
|
+
let start = 0;
|
|
7490
|
+
while (start < records.length) {
|
|
7491
|
+
const end = records.indexOf(0, start);
|
|
7492
|
+
if (end < 0) {
|
|
7493
|
+
throw new WorktreeAdapterError("Git worktree list output was not NUL-terminated");
|
|
7494
|
+
}
|
|
7495
|
+
if (records.subarray(start, end).equals(expected)) return true;
|
|
7496
|
+
start = end + 1;
|
|
7497
|
+
}
|
|
7498
|
+
return false;
|
|
7499
|
+
}
|
|
7500
|
+
function hasLocalBranch(git, repoRoot, branch) {
|
|
7501
|
+
const ref = `refs/heads/${branch}`;
|
|
7502
|
+
return gitText(git, ["for-each-ref", "--format=%(refname)", "--", ref], repoRoot).split("\n").some((candidate) => candidate === ref);
|
|
7503
|
+
}
|
|
7504
|
+
function reconcileAbsent(exists, remove) {
|
|
7505
|
+
try {
|
|
7506
|
+
if (!exists()) return void 0;
|
|
7507
|
+
} catch (err) {
|
|
7508
|
+
return err;
|
|
7509
|
+
}
|
|
7510
|
+
try {
|
|
7511
|
+
remove();
|
|
7512
|
+
return void 0;
|
|
7513
|
+
} catch (removeError) {
|
|
7514
|
+
try {
|
|
7515
|
+
if (!exists()) return void 0;
|
|
7516
|
+
} catch (recheckError) {
|
|
7517
|
+
return new AggregateError(
|
|
7518
|
+
[removeError, recheckError],
|
|
7519
|
+
"Removal failed and the resulting resource state could not be checked"
|
|
7520
|
+
);
|
|
7521
|
+
}
|
|
7522
|
+
return removeError;
|
|
7523
|
+
}
|
|
7524
|
+
}
|
|
7849
7525
|
function sha2562(bytes) {
|
|
7850
7526
|
return `sha256:${createHash8("sha256").update(bytes).digest("hex")}`;
|
|
7851
7527
|
}
|
|
@@ -7887,7 +7563,7 @@ function patchBytes(git, cwd, baseCommit, candidateCommit) {
|
|
|
7887
7563
|
const scratch = mkdtempSync2(join5(tmpdir2(), "agent-eval-patch-"));
|
|
7888
7564
|
const bareRepo = join5(scratch, "repo.git");
|
|
7889
7565
|
const emptyTemplate = join5(scratch, "empty-template");
|
|
7890
|
-
|
|
7566
|
+
mkdirSync2(emptyTemplate);
|
|
7891
7567
|
try {
|
|
7892
7568
|
const objectFormat = gitObjectHashAlgorithm(candidateCommit);
|
|
7893
7569
|
const sourceObjects = realpathSync(gitText(git, ["rev-parse", "--git-path", "objects"], cwd));
|
|
@@ -8025,7 +7701,7 @@ function hashGitBlobFile(path, objectId) {
|
|
|
8025
7701
|
throw new WorktreeAdapterError(`CodeSurface expected a regular file at ${displayGitPath(path)}`);
|
|
8026
7702
|
}
|
|
8027
7703
|
const noFollow = process.platform === "win32" ? 0 : fsConstants.O_NOFOLLOW;
|
|
8028
|
-
const fd =
|
|
7704
|
+
const fd = openSync(path, fsConstants.O_RDONLY | noFollow);
|
|
8029
7705
|
try {
|
|
8030
7706
|
const opened = fstatSync(fd);
|
|
8031
7707
|
if (!opened.isFile() || opened.dev !== before.dev || opened.ino !== before.ino || opened.size !== before.size) {
|
|
@@ -8050,7 +7726,7 @@ function hashGitBlobFile(path, objectId) {
|
|
|
8050
7726
|
}
|
|
8051
7727
|
return { hash: hash.digest("hex"), executable: (opened.mode & 73) !== 0 };
|
|
8052
7728
|
} finally {
|
|
8053
|
-
|
|
7729
|
+
closeSync(fd);
|
|
8054
7730
|
}
|
|
8055
7731
|
}
|
|
8056
7732
|
function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
|
|
@@ -8066,7 +7742,7 @@ function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
|
|
|
8066
7742
|
`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
|
|
8067
7743
|
);
|
|
8068
7744
|
}
|
|
8069
|
-
const lexicalTarget = resolve3(
|
|
7745
|
+
const lexicalTarget = resolve3(dirname2(linkPath), target);
|
|
8070
7746
|
if (!isWithinRoot(root, lexicalTarget)) {
|
|
8071
7747
|
throw new WorktreeAdapterError(
|
|
8072
7748
|
`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
|
|
@@ -8143,7 +7819,7 @@ function assertRawTreeMatchesWorktree(git, root, candidateCommit) {
|
|
|
8143
7819
|
}
|
|
8144
7820
|
function verifyCodeSurfaceWithGit(surface, path, git) {
|
|
8145
7821
|
assertCodeSurfaceIdentity(surface);
|
|
8146
|
-
if (!
|
|
7822
|
+
if (!existsSync5(path)) {
|
|
8147
7823
|
throw new WorktreeAdapterError(`CodeSurface worktree does not exist: ${path}`);
|
|
8148
7824
|
}
|
|
8149
7825
|
const lexicalRoot = resolve3(path);
|
|
@@ -8310,8 +7986,23 @@ function gitWorktreeAdapter(opts) {
|
|
|
8310
7986
|
return surface;
|
|
8311
7987
|
},
|
|
8312
7988
|
async discard(worktree) {
|
|
8313
|
-
|
|
8314
|
-
|
|
7989
|
+
const failures = [
|
|
7990
|
+
reconcileAbsent(
|
|
7991
|
+
() => hasRegisteredWorktree(git, opts.repoRoot, worktree.path),
|
|
7992
|
+
() => gitText(git, ["worktree", "remove", "--force", "--", worktree.path], opts.repoRoot)
|
|
7993
|
+
),
|
|
7994
|
+
reconcileAbsent(
|
|
7995
|
+
() => hasLocalBranch(git, opts.repoRoot, worktree.branch),
|
|
7996
|
+
() => gitText(git, ["branch", "-D", "--", worktree.branch], opts.repoRoot)
|
|
7997
|
+
)
|
|
7998
|
+
].filter((failure) => failure !== void 0);
|
|
7999
|
+
if (failures.length > 0) {
|
|
8000
|
+
const cause = failures.length === 1 ? failures[0] : new AggregateError(failures, "Multiple Git resources could not be removed");
|
|
8001
|
+
throw new WorktreeAdapterError(
|
|
8002
|
+
`Failed to discard worktree ${worktree.path} and branch ${worktree.branch}`,
|
|
8003
|
+
cause
|
|
8004
|
+
);
|
|
8005
|
+
}
|
|
8315
8006
|
}
|
|
8316
8007
|
};
|
|
8317
8008
|
}
|
|
@@ -8324,13 +8015,6 @@ function resolveWorktreePath(surface, worktreeDir) {
|
|
|
8324
8015
|
}
|
|
8325
8016
|
|
|
8326
8017
|
export {
|
|
8327
|
-
JudgeParseError,
|
|
8328
|
-
createDomainExpertJudge,
|
|
8329
|
-
codeExecutionJudge,
|
|
8330
|
-
coherenceJudge,
|
|
8331
|
-
adversarialJudge,
|
|
8332
|
-
createCustomJudge,
|
|
8333
|
-
defaultJudges,
|
|
8334
8018
|
pairArms,
|
|
8335
8019
|
comparePairedArms,
|
|
8336
8020
|
completionVerdict,
|
|
@@ -8338,7 +8022,6 @@ export {
|
|
|
8338
8022
|
parseCorrectnessResponse,
|
|
8339
8023
|
createLlmCorrectnessChecker,
|
|
8340
8024
|
createTokenRecallChecker,
|
|
8341
|
-
llmJudge,
|
|
8342
8025
|
extractProducedState,
|
|
8343
8026
|
CODING_HARNESSES,
|
|
8344
8027
|
HARNESS_NATIVE_MODEL,
|
|
@@ -8404,9 +8087,6 @@ export {
|
|
|
8404
8087
|
traceAnalystProposer,
|
|
8405
8088
|
scoreDiscrimination,
|
|
8406
8089
|
selectDiscriminative,
|
|
8407
|
-
SearchLedgerError,
|
|
8408
|
-
SearchLedgerIntegrityError,
|
|
8409
|
-
SearchLedgerConflictError,
|
|
8410
8090
|
SEARCH_LEDGER_SCHEMA,
|
|
8411
8091
|
validateSearchLedgerEvent,
|
|
8412
8092
|
openSearchLedger,
|
|
@@ -8418,4 +8098,4 @@ export {
|
|
|
8418
8098
|
verifyCodeSurface,
|
|
8419
8099
|
resolveWorktreePath
|
|
8420
8100
|
};
|
|
8421
|
-
//# sourceMappingURL=chunk-
|
|
8101
|
+
//# sourceMappingURL=chunk-JSJZ4PJ6.js.map
|