@tangle-network/agent-eval 0.115.3 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/dist/analyst/index.d.ts +16 -11
- package/dist/analyst/index.js +33 -25
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +12 -5
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +247 -34
- package/dist/campaign/index.js +33 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +45 -31
- package/dist/contract/index.js +58 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
- package/dist/hosted/index.d.ts +14 -7
- package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +97 -55
- package/dist/index.js +343 -244
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
- package/dist/kind-factory-ClZmO25A.d.ts +171 -0
- package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +10 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
- package/dist/rl.d.ts +17 -12
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +19 -10
- package/dist/traces.js +16 -4
- package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-5S5NJ63F.js.map +0 -1
- package/dist/chunk-ADYLPOSX.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-I6LVHOV3.js +0 -205
- package/dist/chunk-I6LVHOV3.js.map +0 -1
- package/dist/chunk-KG4TD7EQ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-QMXXSNC4.js +0 -761
- package/dist/chunk-QMXXSNC4.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/chunk-WSBUZMBU.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
|
@@ -1,57 +1,68 @@
|
|
|
1
1
|
import {
|
|
2
|
+
JudgeParseError,
|
|
2
3
|
assertCodeSurfaceIdentity,
|
|
3
4
|
campaignBreakdown,
|
|
4
5
|
campaignMeanComposite,
|
|
6
|
+
costReceiptFromTCloud,
|
|
5
7
|
defaultProductionGate,
|
|
6
8
|
gepaProposer,
|
|
7
9
|
isProposedCandidate,
|
|
8
10
|
labelTrustRank,
|
|
11
|
+
maximumChargeForTCloudRequest,
|
|
9
12
|
pairHoldout,
|
|
10
13
|
recoverTruncatedJson,
|
|
11
14
|
renderAnalystEvidence,
|
|
12
15
|
runImprovementLoop,
|
|
13
16
|
surfaceContentHash,
|
|
14
17
|
surfaceHash
|
|
15
|
-
} from "./chunk-
|
|
16
|
-
import {
|
|
17
|
-
estimateCost,
|
|
18
|
-
isModelPriced
|
|
19
|
-
} from "./chunk-VI2UW6B6.js";
|
|
18
|
+
} from "./chunk-HQPHZGL6.js";
|
|
20
19
|
import {
|
|
20
|
+
SearchLedgerConflictError,
|
|
21
|
+
SearchLedgerError,
|
|
22
|
+
SearchLedgerIntegrityError,
|
|
23
|
+
appendSearchLedgerLine,
|
|
21
24
|
assertRealBackend,
|
|
22
25
|
contentHash,
|
|
26
|
+
createRunCostLedger,
|
|
27
|
+
fsCampaignStorage,
|
|
23
28
|
planCampaignRun,
|
|
29
|
+
resolveRunDir,
|
|
24
30
|
runCampaign,
|
|
25
|
-
summarizeBackendIntegrity
|
|
26
|
-
|
|
31
|
+
summarizeBackendIntegrity,
|
|
32
|
+
withSearchLedgerFileLock
|
|
33
|
+
} from "./chunk-IDZTTFRR.js";
|
|
27
34
|
import {
|
|
28
|
-
Mutex
|
|
29
|
-
|
|
30
|
-
applyPolicyEditToSurface,
|
|
31
|
-
clamp01,
|
|
32
|
-
isPolicyEdit,
|
|
33
|
-
policyEditsFromFindings
|
|
34
|
-
} from "./chunk-QMXXSNC4.js";
|
|
35
|
+
Mutex
|
|
36
|
+
} from "./chunk-3YYRZDON.js";
|
|
35
37
|
import {
|
|
36
38
|
AnalystRegistry,
|
|
37
39
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
38
|
-
|
|
39
|
-
|
|
40
|
+
POLICY_EDIT_AXES,
|
|
41
|
+
POLICY_EDIT_TARGET_SURFACES,
|
|
42
|
+
admitPolicyEdit,
|
|
43
|
+
applyPolicyEditToSurface,
|
|
44
|
+
assertNoJudgeVerdict,
|
|
45
|
+
createTraceAnalystKind,
|
|
46
|
+
isPolicyEdit,
|
|
47
|
+
makePolicyEdit,
|
|
48
|
+
makePolicyEditCandidateRecord,
|
|
49
|
+
policyEditsFromFindings,
|
|
50
|
+
validatePolicyEditCandidateRecord
|
|
51
|
+
} from "./chunk-MGEHEHSN.js";
|
|
40
52
|
import {
|
|
41
53
|
eProcess,
|
|
42
54
|
mcnemar,
|
|
43
55
|
mulberry32,
|
|
44
56
|
pairedBootstrap,
|
|
45
57
|
pairedRiskDifference,
|
|
46
|
-
weightedComposite,
|
|
47
58
|
wilcoxonSignedRank
|
|
48
59
|
} from "./chunk-PJQFMIOX.js";
|
|
49
60
|
import {
|
|
50
61
|
analyzeTraces
|
|
51
|
-
} from "./chunk-
|
|
62
|
+
} from "./chunk-4JLWXDYA.js";
|
|
52
63
|
import {
|
|
53
64
|
OtlpFileTraceStore
|
|
54
|
-
} from "./chunk-
|
|
65
|
+
} from "./chunk-S2F4J57L.js";
|
|
55
66
|
import {
|
|
56
67
|
modelHasSnapshot,
|
|
57
68
|
validateRunRecord
|
|
@@ -63,11 +74,18 @@ import {
|
|
|
63
74
|
canonicalize
|
|
64
75
|
} from "./chunk-VSMTAMNK.js";
|
|
65
76
|
import {
|
|
66
|
-
callLlm
|
|
67
|
-
|
|
77
|
+
callLlm,
|
|
78
|
+
callLlmJson,
|
|
79
|
+
costReceiptFromLlm,
|
|
80
|
+
costReceiptFromLlmError,
|
|
81
|
+
maximumChargeForLlmRequest
|
|
82
|
+
} from "./chunk-NJC7U437.js";
|
|
83
|
+
import {
|
|
84
|
+
CostAccountingIncompleteError,
|
|
85
|
+
CostLedger
|
|
86
|
+
} from "./chunk-VCTY3W6J.js";
|
|
68
87
|
import {
|
|
69
88
|
AgentEvalError,
|
|
70
|
-
JudgeError,
|
|
71
89
|
ValidationError
|
|
72
90
|
} from "./chunk-ONWEPEDO.js";
|
|
73
91
|
|
|
@@ -489,339 +507,6 @@ async function runLineage(opts) {
|
|
|
489
507
|
return { lineage, best: lineage.best(), steps };
|
|
490
508
|
}
|
|
491
509
|
|
|
492
|
-
// src/judges.ts
|
|
493
|
-
var JudgeParseError = class extends JudgeError {
|
|
494
|
-
/** Name of the judge whose response failed to parse. */
|
|
495
|
-
judgeName;
|
|
496
|
-
/** The raw (truncated) model response that failed to parse. */
|
|
497
|
-
raw;
|
|
498
|
-
constructor(judgeName, raw, options) {
|
|
499
|
-
super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
|
|
500
|
-
this.judgeName = judgeName;
|
|
501
|
-
this.raw = raw;
|
|
502
|
-
}
|
|
503
|
-
};
|
|
504
|
-
function createDomainExpertJudge(domain) {
|
|
505
|
-
return async (tc, { scenario, turns }) => {
|
|
506
|
-
const conversation = turns.map(
|
|
507
|
-
(t, i) => `Turn ${i + 1}:
|
|
508
|
-
User: ${t.userMessage}
|
|
509
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
510
|
-
).join("\n\n---\n\n");
|
|
511
|
-
const resp = await tc.chat({
|
|
512
|
-
model: "gpt-4o",
|
|
513
|
-
messages: [
|
|
514
|
-
{
|
|
515
|
-
role: "system",
|
|
516
|
-
content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
|
|
517
|
-
|
|
518
|
-
Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
|
|
519
|
-
|
|
520
|
-
Evaluate:
|
|
521
|
-
1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
|
|
522
|
-
2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
|
|
523
|
-
|
|
524
|
-
Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
|
|
525
|
-
},
|
|
526
|
-
{
|
|
527
|
-
role: "user",
|
|
528
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
529
|
-
Scenario: ${scenario.thesis}
|
|
530
|
-
|
|
531
|
-
${conversation}`
|
|
532
|
-
}
|
|
533
|
-
],
|
|
534
|
-
temperature: 0.1,
|
|
535
|
-
maxTokens: 800
|
|
536
|
-
});
|
|
537
|
-
return parseJudgeResponse("domain_expert", resp);
|
|
538
|
-
};
|
|
539
|
-
}
|
|
540
|
-
var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
|
|
541
|
-
const codeBlocks = artifacts.codeBlocks;
|
|
542
|
-
if (codeBlocks.length === 0) {
|
|
543
|
-
return [
|
|
544
|
-
{
|
|
545
|
-
judgeName: "code_execution",
|
|
546
|
-
dimension: "code_execution",
|
|
547
|
-
score: 0,
|
|
548
|
-
reasoning: "No code blocks found in agent response."
|
|
549
|
-
}
|
|
550
|
-
];
|
|
551
|
-
}
|
|
552
|
-
const codeText = codeBlocks.map(
|
|
553
|
-
(b, i) => `Block ${i + 1} (${b.language}):
|
|
554
|
-
\`\`\`${b.language}
|
|
555
|
-
${b.code.slice(0, 3e3)}
|
|
556
|
-
\`\`\``
|
|
557
|
-
).join("\n\n");
|
|
558
|
-
const resp = await tc.chat({
|
|
559
|
-
model: "gpt-4o",
|
|
560
|
-
messages: [
|
|
561
|
-
{
|
|
562
|
-
role: "system",
|
|
563
|
-
content: `You are a principal software engineer reviewing code written by an AI agent.
|
|
564
|
-
|
|
565
|
-
Score STRICTLY:
|
|
566
|
-
1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
|
|
567
|
-
2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
|
|
568
|
-
3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
|
|
569
|
-
|
|
570
|
-
Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
|
|
571
|
-
},
|
|
572
|
-
{
|
|
573
|
-
role: "user",
|
|
574
|
-
content: `Task: ${scenario.thesis}
|
|
575
|
-
|
|
576
|
-
${codeText}`
|
|
577
|
-
}
|
|
578
|
-
],
|
|
579
|
-
temperature: 0.1,
|
|
580
|
-
maxTokens: 1e3
|
|
581
|
-
});
|
|
582
|
-
return parseJudgeResponse("code_execution", resp);
|
|
583
|
-
};
|
|
584
|
-
var coherenceJudge = async (tc, { scenario, turns }) => {
|
|
585
|
-
if (turns.length < 2) {
|
|
586
|
-
return [];
|
|
587
|
-
}
|
|
588
|
-
const conversation = turns.map(
|
|
589
|
-
(t, i) => `Turn ${i + 1}:
|
|
590
|
-
User: ${t.userMessage}
|
|
591
|
-
Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
|
|
592
|
-
).join("\n\n---\n\n");
|
|
593
|
-
const resp = await tc.chat({
|
|
594
|
-
model: "gpt-4o",
|
|
595
|
-
messages: [
|
|
596
|
-
{
|
|
597
|
-
role: "system",
|
|
598
|
-
content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
|
|
599
|
-
|
|
600
|
-
Score STRICTLY:
|
|
601
|
-
1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
|
|
602
|
-
2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
|
|
603
|
-
3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
|
|
604
|
-
|
|
605
|
-
Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
|
|
606
|
-
},
|
|
607
|
-
{
|
|
608
|
-
role: "user",
|
|
609
|
-
content: `Scenario: ${scenario.thesis}
|
|
610
|
-
|
|
611
|
-
${conversation}`
|
|
612
|
-
}
|
|
613
|
-
],
|
|
614
|
-
temperature: 0.1,
|
|
615
|
-
maxTokens: 800
|
|
616
|
-
});
|
|
617
|
-
return parseJudgeResponse("coherence", resp);
|
|
618
|
-
};
|
|
619
|
-
var adversarialJudge = async (tc, { scenario, turns }) => {
|
|
620
|
-
const conversation = turns.map(
|
|
621
|
-
(t, i) => `Turn ${i + 1}:
|
|
622
|
-
User: ${t.userMessage}
|
|
623
|
-
Agent: ${t.agentResponse.slice(0, 1500)}`
|
|
624
|
-
).join("\n\n---\n\n");
|
|
625
|
-
const resp = await tc.chat({
|
|
626
|
-
model: "gpt-4o",
|
|
627
|
-
messages: [
|
|
628
|
-
{
|
|
629
|
-
role: "system",
|
|
630
|
-
content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
|
|
631
|
-
|
|
632
|
-
1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
|
|
633
|
-
2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
|
|
634
|
-
3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
|
|
635
|
-
|
|
636
|
-
Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
|
|
637
|
-
|
|
638
|
-
Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
|
|
639
|
-
},
|
|
640
|
-
{
|
|
641
|
-
role: "user",
|
|
642
|
-
content: `Persona: ${scenario.persona}
|
|
643
|
-
Scenario: ${scenario.thesis}
|
|
644
|
-
|
|
645
|
-
${conversation}`
|
|
646
|
-
}
|
|
647
|
-
],
|
|
648
|
-
temperature: 0.2,
|
|
649
|
-
maxTokens: 800
|
|
650
|
-
});
|
|
651
|
-
return parseJudgeResponse("adversarial", resp);
|
|
652
|
-
};
|
|
653
|
-
function createCustomJudge(name, systemPrompt, opts) {
|
|
654
|
-
return async (tc, { scenario, turns }) => {
|
|
655
|
-
const conversation = turns.map(
|
|
656
|
-
(t, i) => `Turn ${i + 1}:
|
|
657
|
-
User: ${t.userMessage}
|
|
658
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
659
|
-
).join("\n\n---\n\n");
|
|
660
|
-
const resp = await tc.chat({
|
|
661
|
-
model: opts?.model ?? "gpt-4o",
|
|
662
|
-
messages: [
|
|
663
|
-
{
|
|
664
|
-
role: "system",
|
|
665
|
-
content: systemPrompt
|
|
666
|
-
},
|
|
667
|
-
{
|
|
668
|
-
role: "user",
|
|
669
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
670
|
-
Scenario: ${scenario.thesis}
|
|
671
|
-
|
|
672
|
-
${conversation}`
|
|
673
|
-
}
|
|
674
|
-
],
|
|
675
|
-
temperature: opts?.temperature ?? 0.1,
|
|
676
|
-
maxTokens: opts?.maxTokens ?? 1e3
|
|
677
|
-
});
|
|
678
|
-
return parseJudgeResponse(name, resp);
|
|
679
|
-
};
|
|
680
|
-
}
|
|
681
|
-
function defaultJudges(domain) {
|
|
682
|
-
return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
|
|
683
|
-
}
|
|
684
|
-
function parseJudgeResponse(judgeName, resp) {
|
|
685
|
-
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
686
|
-
try {
|
|
687
|
-
let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
|
|
688
|
-
const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
|
|
689
|
-
if (arrayMatch) cleaned = arrayMatch[0];
|
|
690
|
-
const parsed = JSON.parse(cleaned);
|
|
691
|
-
return parsed.map((p) => ({
|
|
692
|
-
judgeName,
|
|
693
|
-
dimension: p.dimension,
|
|
694
|
-
score: Math.max(0, Math.min(10, p.score)),
|
|
695
|
-
reasoning: p.reasoning ?? "",
|
|
696
|
-
evidence: p.evidence
|
|
697
|
-
}));
|
|
698
|
-
} catch (err) {
|
|
699
|
-
throw new JudgeParseError(judgeName, content, { cause: err });
|
|
700
|
-
}
|
|
701
|
-
}
|
|
702
|
-
|
|
703
|
-
// src/llm-judge.ts
|
|
704
|
-
function llmJudge(name, prompt, opts) {
|
|
705
|
-
if (!name.trim()) {
|
|
706
|
-
throw new Error("llmJudge: name must be non-empty");
|
|
707
|
-
}
|
|
708
|
-
if (!prompt.trim()) {
|
|
709
|
-
throw new Error(`llmJudge '${name}': prompt must be non-empty`);
|
|
710
|
-
}
|
|
711
|
-
const model = opts.model ?? opts.chat.defaultModel;
|
|
712
|
-
if (!model) {
|
|
713
|
-
throw new Error(
|
|
714
|
-
`llmJudge '${name}': no model on opts and no defaultModel on the ChatClient \u2014 pass opts.model or bind defaultModel at createChatClient().`
|
|
715
|
-
);
|
|
716
|
-
}
|
|
717
|
-
const dimensions = normalizeDimensions(opts.dimensions, name);
|
|
718
|
-
const scale = opts.scale ?? "unit";
|
|
719
|
-
const divisor = scale === "ten" ? 10 : 1;
|
|
720
|
-
const renderUser = opts.renderUser ?? ((input) => JSON.stringify({ scenario: input.scenario, artifact: input.artifact }, null, 2));
|
|
721
|
-
if (opts.weights) {
|
|
722
|
-
for (const key of Object.keys(opts.weights)) {
|
|
723
|
-
if (!dimensions.some((d) => d.key === key)) {
|
|
724
|
-
throw new Error(
|
|
725
|
-
`llmJudge '${name}': weights names dimension '${key}' that is not declared in dimensions`
|
|
726
|
-
);
|
|
727
|
-
}
|
|
728
|
-
}
|
|
729
|
-
}
|
|
730
|
-
const systemPrompt = `${prompt}
|
|
731
|
-
|
|
732
|
-
${renderContract(dimensions, scale)}`;
|
|
733
|
-
return {
|
|
734
|
-
name,
|
|
735
|
-
dimensions,
|
|
736
|
-
appliesTo: opts.appliesTo,
|
|
737
|
-
async score({ artifact, scenario, signal }) {
|
|
738
|
-
const response = await opts.chat.chat(
|
|
739
|
-
{
|
|
740
|
-
model,
|
|
741
|
-
messages: [
|
|
742
|
-
{ role: "system", content: systemPrompt },
|
|
743
|
-
{ role: "user", content: renderUser({ artifact, scenario }) }
|
|
744
|
-
],
|
|
745
|
-
jsonMode: true,
|
|
746
|
-
temperature: opts.temperature ?? 0.1,
|
|
747
|
-
maxTokens: opts.maxTokens ?? 800
|
|
748
|
-
},
|
|
749
|
-
{ signal }
|
|
750
|
-
);
|
|
751
|
-
const parsed = parseResponse(name, response.content);
|
|
752
|
-
const rawDims = parsed.dimensions ?? parsed.scores;
|
|
753
|
-
if (!rawDims || typeof rawDims !== "object") {
|
|
754
|
-
throw new JudgeParseError(name, response.content, {
|
|
755
|
-
cause: new Error("response has no `dimensions` object")
|
|
756
|
-
});
|
|
757
|
-
}
|
|
758
|
-
const dims = {};
|
|
759
|
-
for (const { key } of dimensions) {
|
|
760
|
-
const raw = rawDims[key];
|
|
761
|
-
const value = Number(raw);
|
|
762
|
-
if (raw === void 0 || raw === null || !Number.isFinite(value)) {
|
|
763
|
-
throw new JudgeParseError(name, response.content, {
|
|
764
|
-
cause: new Error(
|
|
765
|
-
`dimension '${key}' missing or non-numeric (got ${JSON.stringify(raw)})`
|
|
766
|
-
)
|
|
767
|
-
});
|
|
768
|
-
}
|
|
769
|
-
dims[key] = clamp01(value / divisor);
|
|
770
|
-
}
|
|
771
|
-
const weights = opts.weights ?? Object.fromEntries(dimensions.map((d) => [d.key, 1 / dimensions.length]));
|
|
772
|
-
const { composite } = weightedComposite({ dims, weights });
|
|
773
|
-
const notes = firstString(parsed.notes) ?? firstString(parsed.rationale) ?? `${name}: composite ${composite.toFixed(3)} over ${dimensions.length} dimension(s)`;
|
|
774
|
-
return { dimensions: dims, composite, notes };
|
|
775
|
-
}
|
|
776
|
-
};
|
|
777
|
-
}
|
|
778
|
-
function normalizeDimensions(input, name) {
|
|
779
|
-
const raw = input && input.length > 0 ? input : ["quality"];
|
|
780
|
-
const out = [];
|
|
781
|
-
const seen = /* @__PURE__ */ new Set();
|
|
782
|
-
for (const d of raw) {
|
|
783
|
-
const dim = typeof d === "string" ? { key: d, description: d } : d;
|
|
784
|
-
if (!dim.key.trim()) {
|
|
785
|
-
throw new Error(`llmJudge '${name}': dimension key must be non-empty`);
|
|
786
|
-
}
|
|
787
|
-
if (seen.has(dim.key)) {
|
|
788
|
-
throw new Error(`llmJudge '${name}': duplicate dimension key '${dim.key}'`);
|
|
789
|
-
}
|
|
790
|
-
seen.add(dim.key);
|
|
791
|
-
out.push(dim);
|
|
792
|
-
}
|
|
793
|
-
return out;
|
|
794
|
-
}
|
|
795
|
-
function renderContract(dimensions, scale) {
|
|
796
|
-
const range = scale === "ten" ? "0 to 10" : "0.0 to 1.0";
|
|
797
|
-
const lines = dimensions.map((d) => ` - "${d.key}": ${d.description} (score ${range})`);
|
|
798
|
-
const example = `{"dimensions": {${dimensions.map((d) => `"${d.key}": <number>`).join(", ")}}, "notes": "<one-line rationale>"}`;
|
|
799
|
-
return [
|
|
800
|
-
"Score the artifact on EACH of these dimensions:",
|
|
801
|
-
...lines,
|
|
802
|
-
"",
|
|
803
|
-
`Respond with JSON ONLY, no prose. Every dimension is a number in [${range}]:`,
|
|
804
|
-
example
|
|
805
|
-
].join("\n");
|
|
806
|
-
}
|
|
807
|
-
function parseResponse(name, content) {
|
|
808
|
-
const stripped = content.replace(/```json\n?|\n?```/g, "").trim();
|
|
809
|
-
const objMatch = stripped.match(/\{[\s\S]*\}/);
|
|
810
|
-
const payload = objMatch ? objMatch[0] : stripped;
|
|
811
|
-
try {
|
|
812
|
-
const parsed = JSON.parse(payload);
|
|
813
|
-
if (typeof parsed !== "object" || parsed === null) {
|
|
814
|
-
throw new Error("parsed value is not an object");
|
|
815
|
-
}
|
|
816
|
-
return parsed;
|
|
817
|
-
} catch (err) {
|
|
818
|
-
throw new JudgeParseError(name, content, { cause: err });
|
|
819
|
-
}
|
|
820
|
-
}
|
|
821
|
-
function firstString(value) {
|
|
822
|
-
return typeof value === "string" && value.trim() ? value : void 0;
|
|
823
|
-
}
|
|
824
|
-
|
|
825
510
|
// src/campaign/analyst-surface.ts
|
|
826
511
|
function surfaceToText(surface) {
|
|
827
512
|
if (typeof surface === "string") return surface;
|
|
@@ -3411,6 +3096,7 @@ var SKILLOPT_SYSTEM = 'You are a SkillOpt optimizer. You improve ONE skill docum
|
|
|
3411
3096
|
function skillOptProposer(opts) {
|
|
3412
3097
|
const evidenceK = opts.evidenceK ?? 3;
|
|
3413
3098
|
const defaultBudget = opts.editBudget ?? 3;
|
|
3099
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
3414
3100
|
async function proposePatches(args) {
|
|
3415
3101
|
const userPrompt = buildPatchPrompt({
|
|
3416
3102
|
target: opts.target,
|
|
@@ -3422,19 +3108,29 @@ function skillOptProposer(opts) {
|
|
|
3422
3108
|
findingsNote: args.findingsNote,
|
|
3423
3109
|
count: args.count
|
|
3424
3110
|
});
|
|
3425
|
-
const
|
|
3426
|
-
|
|
3427
|
-
|
|
3428
|
-
|
|
3429
|
-
|
|
3430
|
-
|
|
3431
|
-
|
|
3432
|
-
|
|
3433
|
-
|
|
3434
|
-
|
|
3435
|
-
|
|
3436
|
-
|
|
3437
|
-
|
|
3111
|
+
const request = {
|
|
3112
|
+
model: opts.model,
|
|
3113
|
+
messages: [
|
|
3114
|
+
{ role: "system", content: SKILLOPT_SYSTEM },
|
|
3115
|
+
{ role: "user", content: userPrompt }
|
|
3116
|
+
],
|
|
3117
|
+
jsonMode: true,
|
|
3118
|
+
temperature: opts.temperature ?? 0.6,
|
|
3119
|
+
maxTokens: opts.maxTokens ?? 4e3
|
|
3120
|
+
};
|
|
3121
|
+
const paid = await (args.costLedger ?? directCostLedger).runPaidCall({
|
|
3122
|
+
channel: "driver",
|
|
3123
|
+
phase: args.costPhase ?? "search.proposal",
|
|
3124
|
+
actor: "skill-opt.propose",
|
|
3125
|
+
model: opts.model,
|
|
3126
|
+
maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
|
|
3127
|
+
signal: args.signal,
|
|
3128
|
+
execute: (signal, callId) => callLlm(request, { ...opts.llm, signal, idempotencyKey: callId }),
|
|
3129
|
+
receipt: costReceiptFromLlm,
|
|
3130
|
+
receiptFromError: costReceiptFromLlmError
|
|
3131
|
+
});
|
|
3132
|
+
if (!paid.succeeded) throw paid.error;
|
|
3133
|
+
const result = paid.value;
|
|
3438
3134
|
return parseSkillPatchResponse(result.content, args.count, args.editBudget);
|
|
3439
3135
|
}
|
|
3440
3136
|
return {
|
|
@@ -3454,7 +3150,9 @@ function skillOptProposer(opts) {
|
|
|
3454
3150
|
rejectedBuffer: [],
|
|
3455
3151
|
findingsNote: renderAnalystEvidence(ctx.findings, ctx.report) ?? void 0,
|
|
3456
3152
|
count: ctx.populationSize,
|
|
3457
|
-
signal: ctx.signal
|
|
3153
|
+
signal: ctx.signal,
|
|
3154
|
+
costLedger: ctx.costLedger ?? directCostLedger,
|
|
3155
|
+
costPhase: ctx.costPhase ?? "search.proposal"
|
|
3458
3156
|
});
|
|
3459
3157
|
const out = [];
|
|
3460
3158
|
const seen = /* @__PURE__ */ new Set();
|
|
@@ -3618,16 +3316,20 @@ async function runSkillOpt(opts) {
|
|
|
3618
3316
|
const budgetAnneal = opts.budgetAnneal ?? true;
|
|
3619
3317
|
const rejectedBufferSize = opts.rejectedBufferSize ?? 12;
|
|
3620
3318
|
const slowMetaEvery = opts.slowMetaEvery ?? 2;
|
|
3621
|
-
|
|
3319
|
+
opts.runDir = resolveRunDir(opts.runDir, opts.repo);
|
|
3320
|
+
const storage = opts.storage ?? fsCampaignStorage();
|
|
3321
|
+
const costLedger = opts.costLedger ?? createRunCostLedger({
|
|
3322
|
+
storage,
|
|
3323
|
+
runDir: opts.runDir,
|
|
3324
|
+
costCeilingUsd: opts.costCeiling
|
|
3325
|
+
});
|
|
3622
3326
|
const scoreHoldout = async (surface, tag) => {
|
|
3623
|
-
const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag);
|
|
3624
|
-
totalCostUsd += campaign.aggregates.totalCostUsd;
|
|
3327
|
+
const campaign = await runScoringCampaign(opts, opts.holdoutScenarios, surface, tag, costLedger);
|
|
3625
3328
|
return campaignMeanComposite(campaign);
|
|
3626
3329
|
};
|
|
3627
3330
|
const evidenceK = opts.evidenceK ?? 3;
|
|
3628
3331
|
const trainEvidence = async (surface, tag) => {
|
|
3629
|
-
const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag);
|
|
3630
|
-
totalCostUsd += campaign.aggregates.totalCostUsd;
|
|
3332
|
+
const campaign = await runScoringCampaign(opts, opts.trainScenarios, surface, tag, costLedger);
|
|
3631
3333
|
return toEvidence(campaign, evidenceK);
|
|
3632
3334
|
};
|
|
3633
3335
|
let current = opts.baselineSurface;
|
|
@@ -3651,7 +3353,9 @@ async function runSkillOpt(opts) {
|
|
|
3651
3353
|
rejectedBuffer: buffer,
|
|
3652
3354
|
metaNote,
|
|
3653
3355
|
count: patchesPerEpoch,
|
|
3654
|
-
signal: opts.signal ?? new AbortController().signal
|
|
3356
|
+
signal: opts.signal ?? new AbortController().signal,
|
|
3357
|
+
costLedger,
|
|
3358
|
+
costPhase: "skill-opt.proposal"
|
|
3655
3359
|
});
|
|
3656
3360
|
let accepted = null;
|
|
3657
3361
|
const rejectedThisEpoch = [];
|
|
@@ -3710,6 +3414,7 @@ async function runSkillOpt(opts) {
|
|
|
3710
3414
|
});
|
|
3711
3415
|
if (sinceAccept >= patience) break;
|
|
3712
3416
|
}
|
|
3417
|
+
const cost = costLedger.summary();
|
|
3713
3418
|
return {
|
|
3714
3419
|
winnerSurface: current,
|
|
3715
3420
|
baselineHoldoutComposite: baselineHoldout,
|
|
@@ -3719,12 +3424,14 @@ async function runSkillOpt(opts) {
|
|
|
3719
3424
|
rejectedEdits: rejectedAll,
|
|
3720
3425
|
epochsRun,
|
|
3721
3426
|
history,
|
|
3722
|
-
totalCostUsd
|
|
3427
|
+
totalCostUsd: cost.totalCostUsd,
|
|
3428
|
+
cost
|
|
3723
3429
|
};
|
|
3724
3430
|
}
|
|
3725
|
-
function runScoringCampaign(opts, scenarios, surface, tag) {
|
|
3431
|
+
function runScoringCampaign(opts, scenarios, surface, tag, costLedger) {
|
|
3726
3432
|
return runCampaign({
|
|
3727
3433
|
...opts,
|
|
3434
|
+
costLedger,
|
|
3728
3435
|
scenarios,
|
|
3729
3436
|
dispatch: (scenario, ctx) => opts.dispatchWithSurface(surface, scenario, ctx),
|
|
3730
3437
|
runDir: `${opts.runDir}/${tag}`
|
|
@@ -3896,11 +3603,11 @@ function gepaEntry(config, combineParents, name) {
|
|
|
3896
3603
|
...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
|
|
3897
3604
|
...config.report !== void 0 ? { report: config.report } : {}
|
|
3898
3605
|
});
|
|
3899
|
-
|
|
3900
|
-
|
|
3901
|
-
|
|
3902
|
-
|
|
3903
|
-
|
|
3606
|
+
return {
|
|
3607
|
+
winnerSurface: result.winnerSurface,
|
|
3608
|
+
costUsd: result.cost.totalCostUsd,
|
|
3609
|
+
durationMs: Date.now() - started
|
|
3610
|
+
};
|
|
3904
3611
|
}
|
|
3905
3612
|
};
|
|
3906
3613
|
}
|
|
@@ -3973,11 +3680,11 @@ function fapoEscalationEntry(config, name = "fapo-escalation") {
|
|
|
3973
3680
|
...config.analyzeGeneration ? { analyzeGeneration: config.analyzeGeneration } : {},
|
|
3974
3681
|
...config.report !== void 0 ? { report: config.report } : {}
|
|
3975
3682
|
});
|
|
3976
|
-
|
|
3977
|
-
|
|
3978
|
-
|
|
3979
|
-
|
|
3980
|
-
|
|
3683
|
+
return {
|
|
3684
|
+
winnerSurface: result.winnerSurface,
|
|
3685
|
+
costUsd: result.cost.totalCostUsd,
|
|
3686
|
+
durationMs: Date.now() - started
|
|
3687
|
+
};
|
|
3981
3688
|
}
|
|
3982
3689
|
};
|
|
3983
3690
|
}
|
|
@@ -4211,6 +3918,7 @@ function createLlmCorrectnessChecker(tc, opts = {}) {
|
|
|
4211
3918
|
const model = opts.model ?? "claude-sonnet-4-6";
|
|
4212
3919
|
const maxContentChars = opts.maxContentChars ?? 8e3;
|
|
4213
3920
|
const maxAttempts = opts.maxAttempts ?? 2;
|
|
3921
|
+
const costLedger = opts.costLedger ?? new CostLedger();
|
|
4214
3922
|
const sink = opts.rawSink;
|
|
4215
3923
|
const record = async (event) => {
|
|
4216
3924
|
try {
|
|
@@ -4254,7 +3962,24 @@ ${content.slice(0, maxContentChars)}`
|
|
|
4254
3962
|
redactedFields: []
|
|
4255
3963
|
});
|
|
4256
3964
|
try {
|
|
4257
|
-
const
|
|
3965
|
+
const paid = await costLedger.runPaidCall({
|
|
3966
|
+
channel: "verifier",
|
|
3967
|
+
phase: opts.costPhase ?? "completion.correctness",
|
|
3968
|
+
actor: "correctness-checker",
|
|
3969
|
+
model,
|
|
3970
|
+
maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
|
|
3971
|
+
tags: {
|
|
3972
|
+
...opts.costTags,
|
|
3973
|
+
requirementId: requirement.reqId,
|
|
3974
|
+
attempt: String(attempt)
|
|
3975
|
+
},
|
|
3976
|
+
signal: opts.signal,
|
|
3977
|
+
execute: () => tc.chat(request),
|
|
3978
|
+
receipt: (response) => costReceiptFromTCloud(response, model),
|
|
3979
|
+
receiptFromError: (error) => opts.receiptFromError?.(error, attempt)
|
|
3980
|
+
});
|
|
3981
|
+
if (!paid.succeeded) throw paid.error;
|
|
3982
|
+
const resp = paid.value;
|
|
4258
3983
|
const raw = resp.choices?.[0]?.message?.content ?? "";
|
|
4259
3984
|
await record({
|
|
4260
3985
|
eventId: randomUUID(),
|
|
@@ -4770,12 +4495,12 @@ function requireResolvedModel(cell, profileId) {
|
|
|
4770
4495
|
const resolved = cell.resolvedModel?.trim();
|
|
4771
4496
|
if (!resolved) {
|
|
4772
4497
|
throw new ProfileMatrixError(
|
|
4773
|
-
`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014
|
|
4498
|
+
`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 return it in the ctx.cost.runPaidCall receipt so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
|
|
4774
4499
|
);
|
|
4775
4500
|
}
|
|
4776
4501
|
if (!modelHasSnapshot(resolved)) {
|
|
4777
4502
|
throw new ProfileMatrixError(
|
|
4778
|
-
`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD)
|
|
4503
|
+
`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) in the paid-call receipt`
|
|
4779
4504
|
);
|
|
4780
4505
|
}
|
|
4781
4506
|
return resolved;
|
|
@@ -4801,14 +4526,9 @@ function buildRunRecord(args) {
|
|
|
4801
4526
|
}
|
|
4802
4527
|
const perDimMean = {};
|
|
4803
4528
|
for (const [dim, values] of Object.entries(dimAccum)) perDimMean[dim] = mean3(values);
|
|
4804
|
-
|
|
4805
|
-
let costEstimated = false;
|
|
4806
|
-
if (costUsd === 0 && cell.tokenUsage.output > 0 && isModelPriced(model)) {
|
|
4807
|
-
costUsd = estimateCost(cell.tokenUsage.input, cell.tokenUsage.output, model);
|
|
4808
|
-
costEstimated = costUsd > 0;
|
|
4809
|
-
}
|
|
4529
|
+
const costUsd = cell.costUsd;
|
|
4810
4530
|
raw.cost_usd = costUsd;
|
|
4811
|
-
raw.cost_estimated = costEstimated ? 1 : 0;
|
|
4531
|
+
raw.cost_estimated = cell.costEstimated ? 1 : 0;
|
|
4812
4532
|
raw.tokens_input = cell.tokenUsage.input;
|
|
4813
4533
|
raw.tokens_output = cell.tokenUsage.output;
|
|
4814
4534
|
if (typeof cell.tokenUsage.cached === "number") raw.tokens_cached = cell.tokenUsage.cached;
|
|
@@ -4951,11 +4671,8 @@ async function runProfileMatrix(opts) {
|
|
|
4951
4671
|
profileRecords.push(record);
|
|
4952
4672
|
records.push(record);
|
|
4953
4673
|
}
|
|
4954
|
-
const
|
|
4955
|
-
campaigns[profileId] =
|
|
4956
|
-
...campaign,
|
|
4957
|
-
aggregates: { ...campaign.aggregates, totalCostUsd: pricedTotalCostUsd }
|
|
4958
|
-
};
|
|
4674
|
+
const totalCostUsd = campaign.aggregates.totalCostUsd;
|
|
4675
|
+
campaigns[profileId] = campaign;
|
|
4959
4676
|
byProfile[profileId] = {
|
|
4960
4677
|
profileId,
|
|
4961
4678
|
profileHash,
|
|
@@ -4965,7 +4682,7 @@ async function runProfileMatrix(opts) {
|
|
|
4965
4682
|
model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
|
|
4966
4683
|
records: profileRecords.length,
|
|
4967
4684
|
meanComposite: mean3(profileRecords.map(compositeOf)),
|
|
4968
|
-
totalCostUsd
|
|
4685
|
+
totalCostUsd,
|
|
4969
4686
|
integrity: summarizeBackendIntegrity(profileRecords)
|
|
4970
4687
|
};
|
|
4971
4688
|
}
|
|
@@ -5130,10 +4847,16 @@ function compositeProposer(opts) {
|
|
|
5130
4847
|
const surface = isCandidate ? proposal.surface : proposal;
|
|
5131
4848
|
const label = isCandidate ? proposal.label : "candidate";
|
|
5132
4849
|
const rationale = isCandidate ? proposal.rationale : "";
|
|
4850
|
+
const candidateRecord = isCandidate ? proposal.candidateRecord : void 0;
|
|
5133
4851
|
const key = surfaceContentHash(surface);
|
|
5134
4852
|
if (seen.has(key)) continue;
|
|
5135
4853
|
seen.add(key);
|
|
5136
|
-
pool.push({
|
|
4854
|
+
pool.push({
|
|
4855
|
+
surface,
|
|
4856
|
+
label: `${member.kind}:${label}`,
|
|
4857
|
+
rationale,
|
|
4858
|
+
...candidateRecord ? { candidateRecord } : {}
|
|
4859
|
+
});
|
|
5137
4860
|
}
|
|
5138
4861
|
} catch (err) {
|
|
5139
4862
|
errors.push(`${member.kind}: ${err instanceof Error ? err.message : String(err)}`);
|
|
@@ -5176,36 +4899,78 @@ function surfaceToPromptText(surface) {
|
|
|
5176
4899
|
return typeof surface === "string" ? surface : JSON.stringify(surface);
|
|
5177
4900
|
}
|
|
5178
4901
|
function analysisEditProposer(opts) {
|
|
4902
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
5179
4903
|
return {
|
|
5180
4904
|
kind: opts.kind,
|
|
5181
4905
|
async propose(ctx) {
|
|
5182
4906
|
const parent = surfaceToPromptText(ctx.currentSurface);
|
|
4907
|
+
const costLedger = ctx.costLedger ?? directCostLedger;
|
|
4908
|
+
const phase = ctx.costPhase ?? "search.proposal";
|
|
5183
4909
|
const traces = await opts.resolveTraces(ctx) ?? "";
|
|
5184
4910
|
if (!traces.trim()) throw new Error(opts.noTracesError);
|
|
5185
4911
|
const dir = mkdtempSync(join4(tmpdir(), `${opts.kind}-proposer-`));
|
|
5186
4912
|
const tracePath = join4(dir, "traces.jsonl");
|
|
5187
4913
|
writeFileSync2(tracePath, traces.endsWith("\n") ? traces : `${traces}
|
|
5188
4914
|
`);
|
|
5189
|
-
|
|
5190
|
-
|
|
5191
|
-
|
|
5192
|
-
|
|
5193
|
-
|
|
5194
|
-
|
|
5195
|
-
|
|
5196
|
-
|
|
5197
|
-
|
|
4915
|
+
if (costLedger.costCeilingUsd !== void 0 && !opts.analysisReceipt) {
|
|
4916
|
+
throw new CostAccountingIncompleteError(
|
|
4917
|
+
`${opts.kind}: capped analysis requires analysisReceipt before external execution`
|
|
4918
|
+
);
|
|
4919
|
+
}
|
|
4920
|
+
const analysis = await costLedger.runPaidCall({
|
|
4921
|
+
channel: "analyst",
|
|
4922
|
+
phase,
|
|
4923
|
+
actor: `${opts.kind}.analyze`,
|
|
4924
|
+
model: opts.analysisModel,
|
|
4925
|
+
maximumCharge: opts.analysisMaximumCharge,
|
|
4926
|
+
tags: { generation: String(ctx.generation) },
|
|
4927
|
+
signal: ctx.signal,
|
|
4928
|
+
execute: (signal) => opts.analyze(tracePath, { ...ctx, signal }),
|
|
4929
|
+
receipt: (report2) => opts.analysisReceipt?.(report2) ?? {
|
|
4930
|
+
model: opts.analysisModel,
|
|
4931
|
+
inputTokens: 0,
|
|
4932
|
+
outputTokens: 0,
|
|
4933
|
+
costUnknown: true
|
|
4934
|
+
}
|
|
4935
|
+
});
|
|
4936
|
+
if (!analysis.succeeded) throw analysis.error;
|
|
4937
|
+
const report = analysis.value;
|
|
4938
|
+
const request = {
|
|
4939
|
+
model: opts.applyModel,
|
|
4940
|
+
messages: [
|
|
4941
|
+
{ role: "system", content: APPLY_SYSTEM },
|
|
4942
|
+
{
|
|
4943
|
+
role: "user",
|
|
4944
|
+
content: `CURRENT PROMPT:
|
|
5198
4945
|
${parent}
|
|
5199
4946
|
|
|
5200
4947
|
TRACE-ANALYSIS REPORT:
|
|
5201
4948
|
${report}
|
|
5202
4949
|
|
|
5203
4950
|
Return the full revised prompt.`
|
|
5204
|
-
|
|
5205
|
-
|
|
5206
|
-
|
|
5207
|
-
|
|
5208
|
-
|
|
4951
|
+
}
|
|
4952
|
+
],
|
|
4953
|
+
maxTokens: opts.applyMaxTokens ?? 6e3
|
|
4954
|
+
};
|
|
4955
|
+
const llm = {
|
|
4956
|
+
baseUrl: opts.baseUrl,
|
|
4957
|
+
apiKey: opts.apiKey,
|
|
4958
|
+
fetch: opts.fetchImpl
|
|
4959
|
+
};
|
|
4960
|
+
const apply = await costLedger.runPaidCall({
|
|
4961
|
+
channel: "driver",
|
|
4962
|
+
phase,
|
|
4963
|
+
actor: `${opts.kind}.apply`,
|
|
4964
|
+
model: opts.applyModel,
|
|
4965
|
+
maximumCharge: maximumChargeForLlmRequest(request, llm),
|
|
4966
|
+
tags: { generation: String(ctx.generation) },
|
|
4967
|
+
signal: ctx.signal,
|
|
4968
|
+
execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
|
|
4969
|
+
receipt: costReceiptFromLlm,
|
|
4970
|
+
receiptFromError: costReceiptFromLlmError
|
|
4971
|
+
});
|
|
4972
|
+
if (!apply.succeeded) throw apply.error;
|
|
4973
|
+
const applied = apply.value;
|
|
5209
4974
|
const text = applied.content.trim();
|
|
5210
4975
|
if (!text || text === parent) return [];
|
|
5211
4976
|
return [{ surface: text, label: opts.label, rationale: opts.rationale(report) }];
|
|
@@ -5224,7 +4989,12 @@ function haloProposer(opts) {
|
|
|
5224
4989
|
label: "halo",
|
|
5225
4990
|
baseUrl: opts.baseUrl,
|
|
5226
4991
|
apiKey: opts.apiKey,
|
|
4992
|
+
analysisModel: model,
|
|
5227
4993
|
applyModel: opts.applyModel ?? model,
|
|
4994
|
+
costLedger: opts.costLedger,
|
|
4995
|
+
analysisMaximumCharge: opts.analysisMaximumCharge,
|
|
4996
|
+
analysisReceipt: opts.analysisReceipt,
|
|
4997
|
+
applyMaxTokens: opts.applyMaxTokens,
|
|
5228
4998
|
fetchImpl: opts.fetchImpl,
|
|
5229
4999
|
resolveTraces: opts.resolveTraces,
|
|
5230
5000
|
noTracesError: "haloProposer: resolveTraces returned no OTLP traces \u2014 the halo engine has nothing to analyze",
|
|
@@ -5264,85 +5034,8 @@ ${findings.slice(0, 800)}`,
|
|
|
5264
5034
|
});
|
|
5265
5035
|
}
|
|
5266
5036
|
|
|
5267
|
-
// src/campaign/proposers/
|
|
5268
|
-
|
|
5269
|
-
var BLOCK_END2 = "<!-- END curated-memory -->";
|
|
5270
|
-
var DEFAULT_HEADING2 = "## Learned from prior runs (curated memory)";
|
|
5271
|
-
var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, generalizable agent guidance. Output ONLY a JSON array of strings, each one imperative lesson the agent should follow (e.g. "Always fetch a resource before mutating it"). No prose outside the JSON. Deduplicate; keep the most actionable and general; drop case-specific noise.';
|
|
5272
|
-
function extractExistingLessons(text) {
|
|
5273
|
-
return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
|
|
5274
|
-
}
|
|
5275
|
-
async function distillLessons(raw, distill) {
|
|
5276
|
-
const res = await callLlm(
|
|
5277
|
-
{
|
|
5278
|
-
model: distill.model,
|
|
5279
|
-
messages: [
|
|
5280
|
-
{ role: "system", content: DISTILL_SYSTEM },
|
|
5281
|
-
{ role: "user", content: `Findings:
|
|
5282
|
-
${raw.map((r) => `- ${r}`).join("\n")}` }
|
|
5283
|
-
]
|
|
5284
|
-
},
|
|
5285
|
-
{ baseUrl: distill.baseUrl, apiKey: distill.apiKey, fetch: distill.fetchImpl }
|
|
5286
|
-
);
|
|
5287
|
-
try {
|
|
5288
|
-
const parsed = JSON.parse(res.content.trim());
|
|
5289
|
-
if (Array.isArray(parsed)) {
|
|
5290
|
-
const lessons = parsed.filter(
|
|
5291
|
-
(x) => typeof x === "string" && x.trim().length > 0
|
|
5292
|
-
);
|
|
5293
|
-
if (lessons.length > 0) return lessons;
|
|
5294
|
-
}
|
|
5295
|
-
} catch {
|
|
5296
|
-
}
|
|
5297
|
-
return raw;
|
|
5298
|
-
}
|
|
5299
|
-
function memoryCurationProposer(opts = {}) {
|
|
5300
|
-
const maxEntries = opts.maxEntries ?? 12;
|
|
5301
|
-
const heading = opts.sectionHeading ?? DEFAULT_HEADING2;
|
|
5302
|
-
return {
|
|
5303
|
-
kind: "memory-curation",
|
|
5304
|
-
async propose(ctx) {
|
|
5305
|
-
const parent = surfaceToText2(ctx.currentSurface);
|
|
5306
|
-
const fresh = [];
|
|
5307
|
-
for (const f of ctx.findings ?? []) {
|
|
5308
|
-
const l = findingToLesson(f);
|
|
5309
|
-
if (l) fresh.push(l);
|
|
5310
|
-
}
|
|
5311
|
-
const carried = extractExistingLessons(parent);
|
|
5312
|
-
if (fresh.length === 0 && carried.length === 0) return [];
|
|
5313
|
-
const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill) : fresh;
|
|
5314
|
-
const byKey = /* @__PURE__ */ new Map();
|
|
5315
|
-
for (const l of carried) {
|
|
5316
|
-
const k = normKey(l);
|
|
5317
|
-
if (k) byKey.set(k, { text: l, count: 1 });
|
|
5318
|
-
}
|
|
5319
|
-
for (const l of distilled) {
|
|
5320
|
-
const k = normKey(l);
|
|
5321
|
-
if (!k) continue;
|
|
5322
|
-
const e = byKey.get(k);
|
|
5323
|
-
if (e) e.count += 1;
|
|
5324
|
-
else byKey.set(k, { text: l, count: 1 });
|
|
5325
|
-
}
|
|
5326
|
-
const ranked = [...byKey.values()].sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).slice(0, maxEntries);
|
|
5327
|
-
if (ranked.length === 0) return [];
|
|
5328
|
-
const block = [BLOCK_START2, heading, ...ranked.map((e) => `- ${e.text}`), BLOCK_END2].join(
|
|
5329
|
-
"\n"
|
|
5330
|
-
);
|
|
5331
|
-
const next = `${stripBlock(parent, BLOCK_START2, BLOCK_END2)}
|
|
5332
|
-
|
|
5333
|
-
${block}
|
|
5334
|
-
`;
|
|
5335
|
-
if (next === parent) return [];
|
|
5336
|
-
return [
|
|
5337
|
-
{
|
|
5338
|
-
surface: next,
|
|
5339
|
-
label: "memory-curation",
|
|
5340
|
-
rationale: `curated ${ranked.length} lessons (from ${fresh.length} new finding(s) + ${carried.length} carried)`
|
|
5341
|
-
}
|
|
5342
|
-
];
|
|
5343
|
-
}
|
|
5344
|
-
};
|
|
5345
|
-
}
|
|
5037
|
+
// src/campaign/proposers/llm-policy-edit.ts
|
|
5038
|
+
import { z } from "zod";
|
|
5346
5039
|
|
|
5347
5040
|
// src/campaign/proposers/policy-edit.ts
|
|
5348
5041
|
function policyEditProposer(opts = {}) {
|
|
@@ -5355,17 +5048,21 @@ function policyEditProposer(opts = {}) {
|
|
|
5355
5048
|
Math.min(opts.maxCandidates ?? ctx.populationSize, ctx.populationSize)
|
|
5356
5049
|
);
|
|
5357
5050
|
const out = [];
|
|
5051
|
+
const seen = /* @__PURE__ */ new Set([surfaceContentHash(ctx.currentSurface)]);
|
|
5358
5052
|
if (limit === 0) return out;
|
|
5359
5053
|
for (const edit of edits) {
|
|
5360
5054
|
const admission = admitPolicyEdit(edit, opts.admission);
|
|
5361
5055
|
opts.onAdmission?.(admission);
|
|
5362
5056
|
if (admission.decision !== "admit") continue;
|
|
5363
5057
|
const surface = coerceCandidateSurface(applyPolicyEditToSurface(ctx.currentSurface, edit));
|
|
5364
|
-
|
|
5058
|
+
const hash = surfaceContentHash(surface);
|
|
5059
|
+
if (seen.has(hash)) continue;
|
|
5060
|
+
seen.add(hash);
|
|
5365
5061
|
out.push({
|
|
5366
5062
|
surface,
|
|
5367
5063
|
label: `policy-edit:${edit.axis}`,
|
|
5368
|
-
rationale: `${edit.editId} expected ${edit.expectedGain.direction} ${edit.expectedGain.metric} by ${edit.expectedGain.amount}; source findings [${edit.source.findingIds.join(", ")}]
|
|
5064
|
+
rationale: `${edit.editId} expected ${edit.expectedGain.direction} ${edit.expectedGain.metric} by ${edit.expectedGain.amount}; source findings [${edit.source.findingIds.join(", ")}]`,
|
|
5065
|
+
candidateRecord: makePolicyEditCandidateRecord(edit)
|
|
5369
5066
|
});
|
|
5370
5067
|
if (out.length >= limit) break;
|
|
5371
5068
|
}
|
|
@@ -5403,22 +5100,1206 @@ function coerceCandidateSurface(surface) {
|
|
|
5403
5100
|
}
|
|
5404
5101
|
throw new Error("policyEditProposer: policy edit produced an unsupported surface");
|
|
5405
5102
|
}
|
|
5406
|
-
function sameSurface(a, b) {
|
|
5407
|
-
return surfaceContentHash(a) === surfaceContentHash(b);
|
|
5408
|
-
}
|
|
5409
5103
|
|
|
5410
|
-
// src/campaign/proposers/
|
|
5411
|
-
|
|
5412
|
-
|
|
5413
|
-
|
|
5414
|
-
|
|
5415
|
-
|
|
5416
|
-
|
|
5417
|
-
|
|
5418
|
-
|
|
5419
|
-
|
|
5420
|
-
|
|
5421
|
-
|
|
5104
|
+
// src/campaign/proposers/policy-edit-author-context.ts
|
|
5105
|
+
function selectPolicyEditAuthorRows(rows, options) {
|
|
5106
|
+
assertPositiveSafeInteger(options.limit, "limit");
|
|
5107
|
+
const unique2 = /* @__PURE__ */ new Map();
|
|
5108
|
+
for (const row of rows) {
|
|
5109
|
+
if (!row.scenarioId || row.scenarioId.trim() !== row.scenarioId) {
|
|
5110
|
+
throw new Error("selectPolicyEditAuthorRows: scenarioId must be trimmed and non-empty");
|
|
5111
|
+
}
|
|
5112
|
+
if (!Number.isFinite(row.composite)) {
|
|
5113
|
+
throw new Error(
|
|
5114
|
+
`selectPolicyEditAuthorRows: composite must be finite for '${row.scenarioId}'`
|
|
5115
|
+
);
|
|
5116
|
+
}
|
|
5117
|
+
if (unique2.has(row.scenarioId)) continue;
|
|
5118
|
+
const reference = options.referenceByScenario?.get(row.scenarioId);
|
|
5119
|
+
if (reference !== void 0 && !Number.isFinite(reference)) {
|
|
5120
|
+
throw new Error(
|
|
5121
|
+
`selectPolicyEditAuthorRows: reference must be finite for '${row.scenarioId}'`
|
|
5122
|
+
);
|
|
5123
|
+
}
|
|
5124
|
+
unique2.set(row.scenarioId, {
|
|
5125
|
+
row,
|
|
5126
|
+
delta: reference === void 0 ? null : row.composite - reference
|
|
5127
|
+
});
|
|
5128
|
+
}
|
|
5129
|
+
const hardest = [...unique2.values()].sort(
|
|
5130
|
+
(a, b) => a.row.composite - b.row.composite || compareScenarioId(a.row, b.row)
|
|
5131
|
+
);
|
|
5132
|
+
const regressions = [...unique2.values()].filter((entry) => entry.delta !== null && entry.delta < 0).sort((a, b) => a.delta - b.delta || compareScenarioId(a.row, b.row));
|
|
5133
|
+
const improvements = [...unique2.values()].filter((entry) => entry.delta !== null && entry.delta > 0).sort((a, b) => b.delta - a.delta || compareScenarioId(a.row, b.row));
|
|
5134
|
+
const rankings = [hardest, regressions, improvements];
|
|
5135
|
+
const maxDepth = Math.max(...rankings.map((ranking) => ranking.length), 0);
|
|
5136
|
+
const selected = [];
|
|
5137
|
+
const selectedIds = /* @__PURE__ */ new Set();
|
|
5138
|
+
for (let depth = 0; depth < maxDepth && selected.length < options.limit; depth += 1) {
|
|
5139
|
+
for (const ranking of rankings) {
|
|
5140
|
+
const entry = ranking[depth];
|
|
5141
|
+
if (!entry || selectedIds.has(entry.row.scenarioId)) continue;
|
|
5142
|
+
selected.push(entry.row);
|
|
5143
|
+
selectedIds.add(entry.row.scenarioId);
|
|
5144
|
+
if (selected.length === options.limit) break;
|
|
5145
|
+
}
|
|
5146
|
+
}
|
|
5147
|
+
return selected;
|
|
5148
|
+
}
|
|
5149
|
+
function assertPolicyEditAuthorContextBudget(value, maxChars) {
|
|
5150
|
+
assertPositiveSafeInteger(maxChars, "maxChars");
|
|
5151
|
+
const json = JSON.stringify(value);
|
|
5152
|
+
if (json === void 0) {
|
|
5153
|
+
throw new Error("assertPolicyEditAuthorContextBudget: value must serialize to JSON");
|
|
5154
|
+
}
|
|
5155
|
+
const actualChars = json.length;
|
|
5156
|
+
if (actualChars > maxChars) {
|
|
5157
|
+
throw new Error(
|
|
5158
|
+
`assertPolicyEditAuthorContextBudget: serialized JSON exceeds budget (actualChars=${actualChars}, maxChars=${maxChars})`
|
|
5159
|
+
);
|
|
5160
|
+
}
|
|
5161
|
+
return { json, actualChars, maxChars };
|
|
5162
|
+
}
|
|
5163
|
+
function compareScenarioId(a, b) {
|
|
5164
|
+
return a.scenarioId < b.scenarioId ? -1 : a.scenarioId > b.scenarioId ? 1 : 0;
|
|
5165
|
+
}
|
|
5166
|
+
function assertPositiveSafeInteger(value, name) {
|
|
5167
|
+
if (!Number.isSafeInteger(value) || value <= 0) {
|
|
5168
|
+
throw new Error(`${name} must be a positive safe integer (got ${value})`);
|
|
5169
|
+
}
|
|
5170
|
+
}
|
|
5171
|
+
|
|
5172
|
+
// src/campaign/proposers/llm-policy-edit.ts
|
|
5173
|
+
var JSON_POLICY_EDIT_TARGET_SURFACES = [
|
|
5174
|
+
"prompt",
|
|
5175
|
+
"tool-contract",
|
|
5176
|
+
"runtime-config",
|
|
5177
|
+
"memory",
|
|
5178
|
+
"agent-profile"
|
|
5179
|
+
];
|
|
5180
|
+
var NonEmptyStringSchema = z.string().trim().min(1);
|
|
5181
|
+
var JsonValueSchema = z.lazy(
|
|
5182
|
+
() => z.union([
|
|
5183
|
+
z.string(),
|
|
5184
|
+
z.number().finite(),
|
|
5185
|
+
z.boolean(),
|
|
5186
|
+
z.null(),
|
|
5187
|
+
z.array(JsonValueSchema),
|
|
5188
|
+
z.record(NonEmptyStringSchema, JsonValueSchema)
|
|
5189
|
+
])
|
|
5190
|
+
);
|
|
5191
|
+
var AuthoredJsonChangeSchema = z.discriminatedUnion("mode", [
|
|
5192
|
+
z.object({
|
|
5193
|
+
kind: z.literal("json"),
|
|
5194
|
+
mode: z.literal("set"),
|
|
5195
|
+
path: NonEmptyStringSchema,
|
|
5196
|
+
value: JsonValueSchema
|
|
5197
|
+
}).strict(),
|
|
5198
|
+
z.object({
|
|
5199
|
+
kind: z.literal("json"),
|
|
5200
|
+
mode: z.literal("merge"),
|
|
5201
|
+
path: NonEmptyStringSchema,
|
|
5202
|
+
value: JsonValueSchema
|
|
5203
|
+
}).strict(),
|
|
5204
|
+
z.object({
|
|
5205
|
+
kind: z.literal("json"),
|
|
5206
|
+
mode: z.literal("remove"),
|
|
5207
|
+
path: NonEmptyStringSchema
|
|
5208
|
+
}).strict()
|
|
5209
|
+
]);
|
|
5210
|
+
var AuthoredPolicyEditSchema = z.object({
|
|
5211
|
+
axis: z.enum(POLICY_EDIT_AXES),
|
|
5212
|
+
target: z.object({
|
|
5213
|
+
surface: z.enum(POLICY_EDIT_TARGET_SURFACES),
|
|
5214
|
+
path: NonEmptyStringSchema,
|
|
5215
|
+
label: NonEmptyStringSchema.max(200).nullable()
|
|
5216
|
+
}).strict(),
|
|
5217
|
+
change: AuthoredJsonChangeSchema,
|
|
5218
|
+
claim: NonEmptyStringSchema.max(2e3),
|
|
5219
|
+
expectedGain: z.object({
|
|
5220
|
+
metric: NonEmptyStringSchema.max(400),
|
|
5221
|
+
direction: z.enum(["increase", "decrease"]),
|
|
5222
|
+
amount: z.number().finite().positive(),
|
|
5223
|
+
unit: z.enum(["absolute", "relative", "percent", "score"]).nullable(),
|
|
5224
|
+
rationale: NonEmptyStringSchema.max(2e3).nullable()
|
|
5225
|
+
}).strict(),
|
|
5226
|
+
confidence: z.number().finite().min(0).max(1),
|
|
5227
|
+
risk: z.enum(["low", "medium", "high", "unknown"]),
|
|
5228
|
+
source: z.object({
|
|
5229
|
+
findingKeys: z.array(NonEmptyStringSchema).min(1).refine((ids) => new Set(ids).size === ids.length, "findingKeys must be unique")
|
|
5230
|
+
}).strict(),
|
|
5231
|
+
rationale: NonEmptyStringSchema.max(4e3).nullable(),
|
|
5232
|
+
validationPlan: NonEmptyStringSchema.max(2e3).nullable()
|
|
5233
|
+
}).strict();
|
|
5234
|
+
var PolicyEditAuthorResponseSchema = z.object({
|
|
5235
|
+
edits: z.array(AuthoredPolicyEditSchema)
|
|
5236
|
+
}).strict();
|
|
5237
|
+
var POLICY_EDIT_AUTHOR_JSON_SCHEMA = {
|
|
5238
|
+
type: "object",
|
|
5239
|
+
additionalProperties: false,
|
|
5240
|
+
required: ["edits"],
|
|
5241
|
+
properties: {
|
|
5242
|
+
edits: {
|
|
5243
|
+
type: "array",
|
|
5244
|
+
items: {
|
|
5245
|
+
type: "object",
|
|
5246
|
+
additionalProperties: false,
|
|
5247
|
+
required: [
|
|
5248
|
+
"axis",
|
|
5249
|
+
"target",
|
|
5250
|
+
"change",
|
|
5251
|
+
"claim",
|
|
5252
|
+
"expectedGain",
|
|
5253
|
+
"confidence",
|
|
5254
|
+
"risk",
|
|
5255
|
+
"source",
|
|
5256
|
+
"rationale",
|
|
5257
|
+
"validationPlan"
|
|
5258
|
+
],
|
|
5259
|
+
properties: {
|
|
5260
|
+
axis: { type: "string", enum: [...POLICY_EDIT_AXES] },
|
|
5261
|
+
target: {
|
|
5262
|
+
type: "object",
|
|
5263
|
+
additionalProperties: false,
|
|
5264
|
+
required: ["surface", "path", "label"],
|
|
5265
|
+
properties: {
|
|
5266
|
+
surface: { type: "string", enum: [...POLICY_EDIT_TARGET_SURFACES] },
|
|
5267
|
+
path: { type: "string", minLength: 1 },
|
|
5268
|
+
label: { type: ["string", "null"], maxLength: 200 }
|
|
5269
|
+
}
|
|
5270
|
+
},
|
|
5271
|
+
change: {
|
|
5272
|
+
anyOf: [
|
|
5273
|
+
{
|
|
5274
|
+
type: "object",
|
|
5275
|
+
additionalProperties: false,
|
|
5276
|
+
required: ["kind", "mode", "path", "value"],
|
|
5277
|
+
properties: {
|
|
5278
|
+
kind: { const: "json" },
|
|
5279
|
+
mode: { const: "set" },
|
|
5280
|
+
path: { type: "string", minLength: 1 },
|
|
5281
|
+
value: {}
|
|
5282
|
+
}
|
|
5283
|
+
},
|
|
5284
|
+
{
|
|
5285
|
+
type: "object",
|
|
5286
|
+
additionalProperties: false,
|
|
5287
|
+
required: ["kind", "mode", "path", "value"],
|
|
5288
|
+
properties: {
|
|
5289
|
+
kind: { const: "json" },
|
|
5290
|
+
mode: { const: "merge" },
|
|
5291
|
+
path: { type: "string", minLength: 1 },
|
|
5292
|
+
value: {}
|
|
5293
|
+
}
|
|
5294
|
+
},
|
|
5295
|
+
{
|
|
5296
|
+
type: "object",
|
|
5297
|
+
additionalProperties: false,
|
|
5298
|
+
required: ["kind", "mode", "path"],
|
|
5299
|
+
properties: {
|
|
5300
|
+
kind: { const: "json" },
|
|
5301
|
+
mode: { const: "remove" },
|
|
5302
|
+
path: { type: "string", minLength: 1 }
|
|
5303
|
+
}
|
|
5304
|
+
}
|
|
5305
|
+
]
|
|
5306
|
+
},
|
|
5307
|
+
claim: { type: "string", minLength: 1, maxLength: 2e3 },
|
|
5308
|
+
expectedGain: {
|
|
5309
|
+
type: "object",
|
|
5310
|
+
additionalProperties: false,
|
|
5311
|
+
required: ["metric", "direction", "amount", "unit", "rationale"],
|
|
5312
|
+
properties: {
|
|
5313
|
+
metric: { type: "string", minLength: 1, maxLength: 400 },
|
|
5314
|
+
direction: { type: "string", enum: ["increase", "decrease"] },
|
|
5315
|
+
amount: { type: "number", exclusiveMinimum: 0 },
|
|
5316
|
+
unit: {
|
|
5317
|
+
type: ["string", "null"],
|
|
5318
|
+
enum: ["absolute", "relative", "percent", "score", null]
|
|
5319
|
+
},
|
|
5320
|
+
rationale: { type: ["string", "null"], maxLength: 2e3 }
|
|
5321
|
+
}
|
|
5322
|
+
},
|
|
5323
|
+
confidence: { type: "number", minimum: 0, maximum: 1 },
|
|
5324
|
+
risk: { type: "string", enum: ["low", "medium", "high", "unknown"] },
|
|
5325
|
+
source: {
|
|
5326
|
+
type: "object",
|
|
5327
|
+
additionalProperties: false,
|
|
5328
|
+
required: ["findingKeys"],
|
|
5329
|
+
properties: {
|
|
5330
|
+
findingKeys: {
|
|
5331
|
+
type: "array",
|
|
5332
|
+
minItems: 1,
|
|
5333
|
+
uniqueItems: true,
|
|
5334
|
+
items: { type: "string", minLength: 1 }
|
|
5335
|
+
}
|
|
5336
|
+
}
|
|
5337
|
+
},
|
|
5338
|
+
rationale: { type: ["string", "null"], maxLength: 4e3 },
|
|
5339
|
+
validationPlan: { type: ["string", "null"], maxLength: 2e3 }
|
|
5340
|
+
}
|
|
5341
|
+
}
|
|
5342
|
+
}
|
|
5343
|
+
}
|
|
5344
|
+
};
|
|
5345
|
+
function policyEditAuthorJsonSchema(maxItems, targetSurface, allowedJsonPaths, objectives) {
|
|
5346
|
+
const schema = JSON.parse(JSON.stringify(POLICY_EDIT_AUTHOR_JSON_SCHEMA));
|
|
5347
|
+
const properties = schema.properties;
|
|
5348
|
+
const edits = properties.edits;
|
|
5349
|
+
edits.maxItems = maxItems;
|
|
5350
|
+
const item = edits.items;
|
|
5351
|
+
const itemProperties = item.properties;
|
|
5352
|
+
const target = itemProperties.target;
|
|
5353
|
+
const targetProperties = target.properties;
|
|
5354
|
+
targetProperties.surface = { type: "string", enum: [targetSurface] };
|
|
5355
|
+
targetProperties.path = { type: "string", enum: [...allowedJsonPaths] };
|
|
5356
|
+
const change = itemProperties.change;
|
|
5357
|
+
for (const variant of change.anyOf) {
|
|
5358
|
+
const variantProperties = variant.properties;
|
|
5359
|
+
variantProperties.path = { type: "string", enum: [...allowedJsonPaths] };
|
|
5360
|
+
}
|
|
5361
|
+
const expectedGain = itemProperties.expectedGain;
|
|
5362
|
+
const gainProperties = expectedGain.properties;
|
|
5363
|
+
gainProperties.metric = { type: "string", enum: objectives.map((objective) => objective.key) };
|
|
5364
|
+
gainProperties.direction = {
|
|
5365
|
+
type: "string",
|
|
5366
|
+
enum: [...new Set(objectives.map((objective) => objective.direction))]
|
|
5367
|
+
};
|
|
5368
|
+
gainProperties.unit = {
|
|
5369
|
+
type: "string",
|
|
5370
|
+
enum: [...new Set(objectives.map((objective) => objective.unit))]
|
|
5371
|
+
};
|
|
5372
|
+
return schema;
|
|
5373
|
+
}
|
|
5374
|
+
var DEFAULT_POLICY_EDIT_HISTORY_LIMITS = Object.freeze({
|
|
5375
|
+
generations: 4,
|
|
5376
|
+
candidatesPerGeneration: 16,
|
|
5377
|
+
scenariosPerCandidate: 12,
|
|
5378
|
+
findings: 32,
|
|
5379
|
+
authorContextChars: 2e5
|
|
5380
|
+
});
|
|
5381
|
+
var POLICY_EDIT_AUTHOR_SYSTEM = [
|
|
5382
|
+
"You author strictly typed PolicyEdit candidates over one JSON surface.",
|
|
5383
|
+
'Return exactly one JSON object with shape {"edits":[...]}; emit an empty edits array when no evidence supports a change.',
|
|
5384
|
+
`axis must be one of: ${POLICY_EDIT_AXES.join(", ")}.`,
|
|
5385
|
+
`target.surface must be one of: ${POLICY_EDIT_TARGET_SURFACES.join(", ")}.`,
|
|
5386
|
+
"target.path and change.path must be the same caller-allowed JSON path.",
|
|
5387
|
+
'change must be exactly one operation: {"kind":"json","mode":"set","path":string,"value":json}, {"kind":"json","mode":"merge","path":string,"value":json}, or {"kind":"json","mode":"remove","path":string}.',
|
|
5388
|
+
"Nullable fields required by the response schema must be null when they do not apply.",
|
|
5389
|
+
"Every edit must cite one or more supplied finding keys in source.findingKeys. Do not emit persistent finding IDs, analyst IDs, or evidence references; the caller binds those from the cited findings.",
|
|
5390
|
+
"Treat expectedGain and confidence as forecasts, never as measured evidence. Learn from baselineOutcome, incumbentOutcome, and observedDeltaFromParent.",
|
|
5391
|
+
"Do not invent a finding, path, field, score, or task fact. Do not include schemaVersion, editId, metadata, prose, or undeclared keys."
|
|
5392
|
+
].join("\n");
|
|
5393
|
+
function llmPolicyEditProposer(opts) {
|
|
5394
|
+
const allowedJsonPaths = validateAllowedJsonPaths(opts.allowedJsonPaths);
|
|
5395
|
+
const allowedPathSet = new Set(allowedJsonPaths);
|
|
5396
|
+
const objectives = validateObjectives(opts.objectives);
|
|
5397
|
+
const objectiveByKey = new Map(objectives.map((objective) => [objective.key, objective]));
|
|
5398
|
+
requireNonEmpty(opts.model, "model");
|
|
5399
|
+
requireNonEmpty(opts.target, "target");
|
|
5400
|
+
if (!JSON_POLICY_EDIT_TARGET_SURFACES.includes(opts.targetSurface)) {
|
|
5401
|
+
throw new Error(
|
|
5402
|
+
`llmPolicyEditProposer: targetSurface '${opts.targetSurface}' is not a JSON-backed surface`
|
|
5403
|
+
);
|
|
5404
|
+
}
|
|
5405
|
+
if (opts.maxCandidates !== void 0 && (!Number.isSafeInteger(opts.maxCandidates) || opts.maxCandidates < 0)) {
|
|
5406
|
+
throw new Error("llmPolicyEditProposer: maxCandidates must be a non-negative safe integer");
|
|
5407
|
+
}
|
|
5408
|
+
const admissionMode = opts.admissionMode ?? "evidence-only";
|
|
5409
|
+
if (admissionMode !== "evidence-only" && admissionMode !== "strict") {
|
|
5410
|
+
throw new Error("llmPolicyEditProposer: admissionMode must be 'evidence-only' or 'strict'");
|
|
5411
|
+
}
|
|
5412
|
+
if (opts.admission && admissionMode !== "strict") {
|
|
5413
|
+
throw new Error("llmPolicyEditProposer: admission thresholds require admissionMode: 'strict'");
|
|
5414
|
+
}
|
|
5415
|
+
const historyLimits = validateHistoryLimits({
|
|
5416
|
+
...opts.maxHistoryGenerations === void 0 ? {} : { maxGenerations: opts.maxHistoryGenerations },
|
|
5417
|
+
...opts.maxHistoryCandidatesPerGeneration === void 0 ? {} : { maxCandidatesPerGeneration: opts.maxHistoryCandidatesPerGeneration },
|
|
5418
|
+
...opts.maxScenariosPerCandidate === void 0 ? {} : { maxScenariosPerCandidate: opts.maxScenariosPerCandidate },
|
|
5419
|
+
...opts.scenarioIdTransform === void 0 ? {} : { scenarioIdTransform: opts.scenarioIdTransform }
|
|
5420
|
+
});
|
|
5421
|
+
const maxFindings = positiveLimit(
|
|
5422
|
+
opts.maxFindings ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.findings,
|
|
5423
|
+
"maxFindings"
|
|
5424
|
+
);
|
|
5425
|
+
const maxAuthorContextChars = positiveLimit(
|
|
5426
|
+
opts.maxAuthorContextChars ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.authorContextChars,
|
|
5427
|
+
"maxAuthorContextChars"
|
|
5428
|
+
);
|
|
5429
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
5430
|
+
return {
|
|
5431
|
+
kind: "llm-policy-edit",
|
|
5432
|
+
async propose(ctx) {
|
|
5433
|
+
const limit = Math.min(ctx.populationSize, opts.maxCandidates ?? ctx.populationSize);
|
|
5434
|
+
if (limit <= 0) return [];
|
|
5435
|
+
const currentSurface = parseJsonSurface(ctx.currentSurface);
|
|
5436
|
+
assertSearchOutcome(ctx.baselineOutcome, "baselineOutcome");
|
|
5437
|
+
assertSearchOutcome(ctx.incumbentOutcome, "incumbentOutcome");
|
|
5438
|
+
assertMeasuredCompositesInScale(ctx, objectives[0]);
|
|
5439
|
+
const scenarioIds = createScenarioIdProjector(historyLimits.scenarioIdTransform);
|
|
5440
|
+
registerOutcomeScenarioIds(ctx.baselineOutcome, scenarioIds);
|
|
5441
|
+
registerOutcomeScenarioIds(ctx.incumbentOutcome, scenarioIds);
|
|
5442
|
+
registerHistoryScenarioIds(ctx.history.slice(-historyLimits.maxGenerations), scenarioIds);
|
|
5443
|
+
assertSurfaceIsTaskAgnostic(
|
|
5444
|
+
{ currentSurface, allowedJsonPaths, objectives, targetSurface: opts.targetSurface },
|
|
5445
|
+
scenarioIds
|
|
5446
|
+
);
|
|
5447
|
+
const measuredSources = measuredSourceMeasurements(ctx);
|
|
5448
|
+
const findings = citableFindings(ctx.findings, measuredSources, maxFindings);
|
|
5449
|
+
const findingByKey = new Map(
|
|
5450
|
+
findings.map((finding, index) => [`finding-${index + 1}`, finding])
|
|
5451
|
+
);
|
|
5452
|
+
const authorContext = {
|
|
5453
|
+
target: scenarioIds.sanitize(opts.target),
|
|
5454
|
+
targetSurface: opts.targetSurface,
|
|
5455
|
+
allowedJsonPaths,
|
|
5456
|
+
objectives,
|
|
5457
|
+
candidateCount: limit,
|
|
5458
|
+
generation: ctx.generation,
|
|
5459
|
+
currentSurface,
|
|
5460
|
+
findings: findings.map(
|
|
5461
|
+
(finding, index) => renderFinding(finding, `finding-${index + 1}`, scenarioIds, measuredSources)
|
|
5462
|
+
),
|
|
5463
|
+
baselineOutcome: projectOutcome(
|
|
5464
|
+
ctx.baselineOutcome,
|
|
5465
|
+
scenarioIds,
|
|
5466
|
+
historyLimits.maxScenariosPerCandidate
|
|
5467
|
+
),
|
|
5468
|
+
incumbentOutcome: projectOutcome(
|
|
5469
|
+
ctx.incumbentOutcome,
|
|
5470
|
+
scenarioIds,
|
|
5471
|
+
historyLimits.maxScenariosPerCandidate,
|
|
5472
|
+
ctx.baselineOutcome
|
|
5473
|
+
),
|
|
5474
|
+
history: projectPolicyEditHistoryWithProjector(
|
|
5475
|
+
ctx.history,
|
|
5476
|
+
historyLimits,
|
|
5477
|
+
scenarioIds,
|
|
5478
|
+
objectiveByKey
|
|
5479
|
+
)
|
|
5480
|
+
};
|
|
5481
|
+
const responseSchema = policyEditAuthorJsonSchema(
|
|
5482
|
+
limit,
|
|
5483
|
+
opts.targetSurface,
|
|
5484
|
+
allowedJsonPaths,
|
|
5485
|
+
objectives
|
|
5486
|
+
);
|
|
5487
|
+
assertPolicyEditAuthorContextBudget(
|
|
5488
|
+
{ system: POLICY_EDIT_AUTHOR_SYSTEM, authorContext, responseSchema },
|
|
5489
|
+
maxAuthorContextChars
|
|
5490
|
+
);
|
|
5491
|
+
const userContent = JSON.stringify(authorContext);
|
|
5492
|
+
const request = {
|
|
5493
|
+
model: opts.model,
|
|
5494
|
+
messages: [
|
|
5495
|
+
{ role: "system", content: POLICY_EDIT_AUTHOR_SYSTEM },
|
|
5496
|
+
{ role: "user", content: userContent }
|
|
5497
|
+
],
|
|
5498
|
+
jsonSchema: {
|
|
5499
|
+
name: "policy_edit_author",
|
|
5500
|
+
schema: responseSchema
|
|
5501
|
+
},
|
|
5502
|
+
temperature: opts.temperature ?? 0.2,
|
|
5503
|
+
maxTokens: opts.maxTokens ?? 6e3,
|
|
5504
|
+
timeoutMs: opts.timeoutMs
|
|
5505
|
+
};
|
|
5506
|
+
const paid = await (ctx.costLedger ?? directCostLedger).runPaidCall({
|
|
5507
|
+
channel: "driver",
|
|
5508
|
+
phase: ctx.costPhase ?? "search.proposal",
|
|
5509
|
+
actor: "llm-policy-edit.author",
|
|
5510
|
+
model: opts.model,
|
|
5511
|
+
maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
|
|
5512
|
+
tags: { generation: String(ctx.generation) },
|
|
5513
|
+
signal: ctx.signal,
|
|
5514
|
+
execute: (signal, callId) => callLlmJson(request, { ...opts.llm, signal, idempotencyKey: callId }),
|
|
5515
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
5516
|
+
receiptFromError: costReceiptFromLlmError
|
|
5517
|
+
});
|
|
5518
|
+
if (!paid.succeeded) throw paid.error;
|
|
5519
|
+
const { value } = paid.value;
|
|
5520
|
+
const response = parseAuthorResponse(value);
|
|
5521
|
+
if (response.edits.length > limit) {
|
|
5522
|
+
throw new Error(
|
|
5523
|
+
`llmPolicyEditProposer: author returned ${response.edits.length} edits for ${limit} candidate slots`
|
|
5524
|
+
);
|
|
5525
|
+
}
|
|
5526
|
+
const edits = response.edits.map(
|
|
5527
|
+
(draft) => bindAuthoredEdit(
|
|
5528
|
+
draft,
|
|
5529
|
+
findingByKey,
|
|
5530
|
+
opts.targetSurface,
|
|
5531
|
+
allowedPathSet,
|
|
5532
|
+
objectiveByKey,
|
|
5533
|
+
ctx.incumbentOutcome?.composite ?? ctx.baselineOutcome?.composite
|
|
5534
|
+
)
|
|
5535
|
+
);
|
|
5536
|
+
return policyEditProposer({
|
|
5537
|
+
edits,
|
|
5538
|
+
admission: admissionMode === "strict" ? opts.admission : {
|
|
5539
|
+
minScore: 0,
|
|
5540
|
+
minExpectedGain: 0,
|
|
5541
|
+
allowHighRisk: true,
|
|
5542
|
+
requireEvidence: true
|
|
5543
|
+
},
|
|
5544
|
+
maxCandidates: limit,
|
|
5545
|
+
onAdmission: opts.onAdmission
|
|
5546
|
+
}).propose(ctx);
|
|
5547
|
+
}
|
|
5548
|
+
};
|
|
5549
|
+
}
|
|
5550
|
+
function projectPolicyEditHistory(history, options = {}) {
|
|
5551
|
+
const limits = validateHistoryLimits(options);
|
|
5552
|
+
const objectiveByKey = new Map(
|
|
5553
|
+
(options.objectives ? validateObjectives(options.objectives) : []).map((objective) => [
|
|
5554
|
+
objective.key,
|
|
5555
|
+
objective
|
|
5556
|
+
])
|
|
5557
|
+
);
|
|
5558
|
+
const scenarioIds = createScenarioIdProjector(limits.scenarioIdTransform);
|
|
5559
|
+
registerHistoryScenarioIds(history.slice(-limits.maxGenerations), scenarioIds);
|
|
5560
|
+
return projectPolicyEditHistoryWithProjector(history, limits, scenarioIds, objectiveByKey);
|
|
5561
|
+
}
|
|
5562
|
+
function projectPolicyEditHistoryWithProjector(history, limits, scenarioIds, objectiveByKey) {
|
|
5563
|
+
const retainedHistory = history.slice(-limits.maxGenerations);
|
|
5564
|
+
const candidateByHash = new Map(
|
|
5565
|
+
retainedHistory.flatMap(
|
|
5566
|
+
(record) => record.candidates.map((candidate) => [candidate.surfaceHash, candidate])
|
|
5567
|
+
)
|
|
5568
|
+
);
|
|
5569
|
+
return retainedHistory.map((record) => {
|
|
5570
|
+
const candidates = selectHistoryCandidates(record, limits.maxCandidatesPerGeneration);
|
|
5571
|
+
const hashes = new Set(candidates.map((candidate) => candidate.surfaceHash));
|
|
5572
|
+
return {
|
|
5573
|
+
generationIndex: record.generationIndex,
|
|
5574
|
+
promoted: record.promoted.filter((hash) => hashes.has(hash)).map((hash) => scenarioIds.sanitize(hash)),
|
|
5575
|
+
candidates: candidates.map(
|
|
5576
|
+
(candidate) => projectHistoryCandidate(
|
|
5577
|
+
candidate,
|
|
5578
|
+
scenarioIds,
|
|
5579
|
+
limits.maxScenariosPerCandidate,
|
|
5580
|
+
candidate.parentSurfaceHash ? candidateByHash.get(candidate.parentSurfaceHash)?.scenarios : void 0,
|
|
5581
|
+
objectiveByKey
|
|
5582
|
+
)
|
|
5583
|
+
)
|
|
5584
|
+
};
|
|
5585
|
+
});
|
|
5586
|
+
}
|
|
5587
|
+
function selectHistoryCandidates(record, limit) {
|
|
5588
|
+
const promotionOrder = new Map(record.promoted.map((hash, index) => [hash, index]));
|
|
5589
|
+
const promoted = [...record.candidates].filter((candidate) => promotionOrder.has(candidate.surfaceHash)).sort((a, b) => promotionOrder.get(a.surfaceHash) - promotionOrder.get(b.surfaceHash));
|
|
5590
|
+
const selected = promoted.slice(0, limit);
|
|
5591
|
+
const selectedHashes = new Set(selected.map((candidate) => candidate.surfaceHash));
|
|
5592
|
+
const remaining = [...record.candidates].filter((candidate) => !promotionOrder.has(candidate.surfaceHash)).sort((a, b) => b.composite - a.composite || a.surfaceHash.localeCompare(b.surfaceHash));
|
|
5593
|
+
let high = 0;
|
|
5594
|
+
let low = remaining.length - 1;
|
|
5595
|
+
let takeLow = selected.length > 0;
|
|
5596
|
+
while (selected.length < limit && high <= low) {
|
|
5597
|
+
const candidate = takeLow ? remaining[low--] : remaining[high++];
|
|
5598
|
+
takeLow = !takeLow;
|
|
5599
|
+
if (!candidate || selectedHashes.has(candidate.surfaceHash)) continue;
|
|
5600
|
+
selected.push(candidate);
|
|
5601
|
+
selectedHashes.add(candidate.surfaceHash);
|
|
5602
|
+
}
|
|
5603
|
+
return selected;
|
|
5604
|
+
}
|
|
5605
|
+
function projectHistoryCandidate(candidate, scenarioIds, maxScenarios, parentScenarios, objectiveByKey) {
|
|
5606
|
+
const referenceByScenario = parentScenarios ? new Map(parentScenarios.map((scenario) => [scenario.scenarioId, scenario.composite])) : void 0;
|
|
5607
|
+
const selectedScenarios = selectPolicyEditAuthorRows(candidate.scenarios, {
|
|
5608
|
+
limit: maxScenarios,
|
|
5609
|
+
...referenceByScenario ? { referenceByScenario } : {}
|
|
5610
|
+
});
|
|
5611
|
+
const validatedRecord = candidate.candidateRecord ? validatePolicyEditCandidateRecord(candidate.candidateRecord) : void 0;
|
|
5612
|
+
return {
|
|
5613
|
+
surfaceHash: scenarioIds.sanitize(candidate.surfaceHash),
|
|
5614
|
+
parentSurfaceHash: candidate.parentSurfaceHash ? scenarioIds.sanitize(candidate.parentSurfaceHash) : null,
|
|
5615
|
+
parentComposite: candidate.parentComposite ?? null,
|
|
5616
|
+
label: candidate.label ? scenarioIds.sanitize(candidate.label) : null,
|
|
5617
|
+
rationale: candidate.rationale ? scenarioIds.sanitize(candidate.rationale) : null,
|
|
5618
|
+
composite: candidate.composite,
|
|
5619
|
+
observedDeltaFromParent: candidate.observedDeltaFromParent ?? null,
|
|
5620
|
+
eligibleForPromotion: candidate.eligibleForPromotion ?? null,
|
|
5621
|
+
coverage: candidate.coverage ? {
|
|
5622
|
+
expectedCells: candidate.coverage.expectedCells,
|
|
5623
|
+
scorableCells: candidate.coverage.scorableCells,
|
|
5624
|
+
// `cellId` embeds the raw scenario ID. The aggregate reason is useful
|
|
5625
|
+
// for search, but the identifier must not bypass scenarioIdTransform.
|
|
5626
|
+
unscorableCells: [...candidate.coverage.unscorableCells].sort((a, b) => a.cellId.localeCompare(b.cellId)).slice(0, maxScenarios).map((cell) => ({ reason: scenarioIds.sanitize(cell.reason) }))
|
|
5627
|
+
} : null,
|
|
5628
|
+
// GenerationCandidate.ci95 is currently a placeholder [composite, composite],
|
|
5629
|
+
// not a measured interval. Keep it out of author context until it is real.
|
|
5630
|
+
dimensions: sanitizeDimensions(candidate.dimensions, scenarioIds),
|
|
5631
|
+
scenarios: selectedScenarios.map((scenario) => {
|
|
5632
|
+
return {
|
|
5633
|
+
scenarioId: scenarioIds.project(scenario.scenarioId),
|
|
5634
|
+
composite: scenario.composite,
|
|
5635
|
+
notes: scenario.notes ? scenarioIds.sanitize(scenario.notes) : null
|
|
5636
|
+
};
|
|
5637
|
+
}),
|
|
5638
|
+
candidateEdit: validatedRecord ? summarizeCandidateEdit(validatedRecord, scenarioIds) : null,
|
|
5639
|
+
forecastCalibration: forecastCalibration(candidate, validatedRecord, objectiveByKey)
|
|
5640
|
+
};
|
|
5641
|
+
}
|
|
5642
|
+
function projectOutcome(outcome, scenarioIds, maxScenarios, reference = void 0) {
|
|
5643
|
+
if (!outcome) return null;
|
|
5644
|
+
const referenceByScenario = reference ? new Map(reference.scenarios.map((scenario) => [scenario.scenarioId, scenario.composite])) : void 0;
|
|
5645
|
+
const selectedScenarios = selectPolicyEditAuthorRows(outcome.scenarios, {
|
|
5646
|
+
limit: maxScenarios,
|
|
5647
|
+
...referenceByScenario ? { referenceByScenario } : {}
|
|
5648
|
+
});
|
|
5649
|
+
return {
|
|
5650
|
+
split: "search",
|
|
5651
|
+
generation: outcome.generation,
|
|
5652
|
+
surfaceHash: scenarioIds.sanitize(outcome.surfaceHash),
|
|
5653
|
+
composite: outcome.composite,
|
|
5654
|
+
dimensions: sanitizeDimensions(outcome.dimensions, scenarioIds),
|
|
5655
|
+
scenarios: selectedScenarios.map((scenario) => {
|
|
5656
|
+
return {
|
|
5657
|
+
scenarioId: scenarioIds.project(scenario.scenarioId),
|
|
5658
|
+
composite: scenario.composite,
|
|
5659
|
+
notes: scenario.notes ? scenarioIds.sanitize(scenario.notes) : null
|
|
5660
|
+
};
|
|
5661
|
+
}),
|
|
5662
|
+
coverage: { ...outcome.coverage }
|
|
5663
|
+
};
|
|
5664
|
+
}
|
|
5665
|
+
function createScenarioIdProjector(transform) {
|
|
5666
|
+
const aliases = /* @__PURE__ */ new Map();
|
|
5667
|
+
const originals = /* @__PURE__ */ new Map();
|
|
5668
|
+
return {
|
|
5669
|
+
project(scenarioId) {
|
|
5670
|
+
const known = aliases.get(scenarioId);
|
|
5671
|
+
if (known) return known;
|
|
5672
|
+
const alias = transform(scenarioId);
|
|
5673
|
+
if (!alias || alias.trim() !== alias) {
|
|
5674
|
+
throw new Error(
|
|
5675
|
+
"llmPolicyEditProposer: scenarioIdTransform must return a trimmed non-empty string"
|
|
5676
|
+
);
|
|
5677
|
+
}
|
|
5678
|
+
const collision = originals.get(alias);
|
|
5679
|
+
if (collision && collision !== scenarioId) {
|
|
5680
|
+
throw new Error(
|
|
5681
|
+
`llmPolicyEditProposer: scenarioIdTransform collision for '${collision}' and '${scenarioId}'`
|
|
5682
|
+
);
|
|
5683
|
+
}
|
|
5684
|
+
const aliasIsAnotherOriginal = aliases.has(alias) && alias !== scenarioId;
|
|
5685
|
+
const originalIsAnotherAlias = originals.has(scenarioId) && originals.get(scenarioId) !== scenarioId;
|
|
5686
|
+
if (aliasIsAnotherOriginal || originalIsAnotherAlias) {
|
|
5687
|
+
throw new Error(
|
|
5688
|
+
"llmPolicyEditProposer: scenarioIdTransform aliases overlap raw scenario IDs"
|
|
5689
|
+
);
|
|
5690
|
+
}
|
|
5691
|
+
aliases.set(scenarioId, alias);
|
|
5692
|
+
originals.set(alias, scenarioId);
|
|
5693
|
+
return alias;
|
|
5694
|
+
},
|
|
5695
|
+
sanitize(text) {
|
|
5696
|
+
const originalsByLength = [...aliases.keys()].sort((a, b) => b.length - a.length);
|
|
5697
|
+
if (originalsByLength.length === 0) return text;
|
|
5698
|
+
const pattern = new RegExp(
|
|
5699
|
+
`(^|[^A-Za-z0-9_-])(${originalsByLength.map(escapeRegExp).join("|")})(?=$|[^A-Za-z0-9_-])`,
|
|
5700
|
+
"g"
|
|
5701
|
+
);
|
|
5702
|
+
return text.replace(
|
|
5703
|
+
pattern,
|
|
5704
|
+
(_match, prefix, original) => `${prefix}${aliases.get(original) ?? original}`
|
|
5705
|
+
);
|
|
5706
|
+
}
|
|
5707
|
+
};
|
|
5708
|
+
}
|
|
5709
|
+
function escapeRegExp(value) {
|
|
5710
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
5711
|
+
}
|
|
5712
|
+
function sourceKey(surfaceHash2, generation) {
|
|
5713
|
+
return `${surfaceHash2}@${generation}`;
|
|
5714
|
+
}
|
|
5715
|
+
function measuredSourceMeasurements(ctx) {
|
|
5716
|
+
const measurements = /* @__PURE__ */ new Map();
|
|
5717
|
+
const add = (measurement) => {
|
|
5718
|
+
measurements.set(sourceKey(measurement.surfaceHash, measurement.generation), measurement);
|
|
5719
|
+
};
|
|
5720
|
+
if (ctx.baselineOutcome) {
|
|
5721
|
+
add({
|
|
5722
|
+
surfaceHash: ctx.baselineOutcome.surfaceHash,
|
|
5723
|
+
generation: ctx.baselineOutcome.generation,
|
|
5724
|
+
composite: ctx.baselineOutcome.composite,
|
|
5725
|
+
parentComposite: null,
|
|
5726
|
+
observedDeltaFromParent: null,
|
|
5727
|
+
eligibleForPromotion: true,
|
|
5728
|
+
coverage: { ...ctx.baselineOutcome.coverage }
|
|
5729
|
+
});
|
|
5730
|
+
}
|
|
5731
|
+
for (const record of ctx.history) {
|
|
5732
|
+
for (const candidate of record.candidates) {
|
|
5733
|
+
add({
|
|
5734
|
+
surfaceHash: candidate.surfaceHash,
|
|
5735
|
+
generation: record.generationIndex,
|
|
5736
|
+
composite: candidate.composite,
|
|
5737
|
+
parentComposite: candidate.parentComposite ?? null,
|
|
5738
|
+
observedDeltaFromParent: candidate.observedDeltaFromParent ?? null,
|
|
5739
|
+
eligibleForPromotion: candidate.eligibleForPromotion === true,
|
|
5740
|
+
coverage: candidate.coverage ? {
|
|
5741
|
+
expectedCells: candidate.coverage.expectedCells,
|
|
5742
|
+
scorableCells: candidate.coverage.scorableCells
|
|
5743
|
+
} : { expectedCells: 0, scorableCells: 0 }
|
|
5744
|
+
});
|
|
5745
|
+
}
|
|
5746
|
+
}
|
|
5747
|
+
if (ctx.incumbentOutcome) {
|
|
5748
|
+
const generation = ctx.incumbentOutcome.generation;
|
|
5749
|
+
const key = sourceKey(ctx.incumbentOutcome.surfaceHash, generation);
|
|
5750
|
+
if (!measurements.has(key)) {
|
|
5751
|
+
add({
|
|
5752
|
+
surfaceHash: ctx.incumbentOutcome.surfaceHash,
|
|
5753
|
+
generation,
|
|
5754
|
+
composite: ctx.incumbentOutcome.composite,
|
|
5755
|
+
parentComposite: null,
|
|
5756
|
+
observedDeltaFromParent: null,
|
|
5757
|
+
eligibleForPromotion: true,
|
|
5758
|
+
coverage: { ...ctx.incumbentOutcome.coverage }
|
|
5759
|
+
});
|
|
5760
|
+
}
|
|
5761
|
+
}
|
|
5762
|
+
return measurements;
|
|
5763
|
+
}
|
|
5764
|
+
function validateFindingSource(source, measured) {
|
|
5765
|
+
if (source.kind === "global") {
|
|
5766
|
+
requireNonEmpty(source.label, "global finding source label");
|
|
5767
|
+
return;
|
|
5768
|
+
}
|
|
5769
|
+
if (!measured.has(sourceKey(source.surfaceHash, source.generation))) {
|
|
5770
|
+
throw new Error(
|
|
5771
|
+
`llmPolicyEditProposer: finding source ${source.surfaceHash}@${source.generation} is not a measured surface`
|
|
5772
|
+
);
|
|
5773
|
+
}
|
|
5774
|
+
}
|
|
5775
|
+
function sameFindingSource(a, b) {
|
|
5776
|
+
if (a.kind !== b.kind) return false;
|
|
5777
|
+
return a.kind === "surface" && b.kind === "surface" ? a.surfaceHash === b.surfaceHash && a.generation === b.generation : a.kind === "global" && b.kind === "global" && a.label === b.label;
|
|
5778
|
+
}
|
|
5779
|
+
function summarizeCandidateEdit(record, scenarioIds) {
|
|
5780
|
+
const edit = record.policyEdit;
|
|
5781
|
+
return {
|
|
5782
|
+
editId: edit.editId,
|
|
5783
|
+
axis: edit.axis,
|
|
5784
|
+
target: sanitizeAuthorValue(edit.target, scenarioIds),
|
|
5785
|
+
change: sanitizeAuthorValue(edit.change, scenarioIds),
|
|
5786
|
+
claim: scenarioIds.sanitize(edit.claim),
|
|
5787
|
+
expectedGain: sanitizeAuthorValue(edit.expectedGain, scenarioIds),
|
|
5788
|
+
confidence: edit.confidence,
|
|
5789
|
+
risk: edit.risk,
|
|
5790
|
+
sourceFindingIds: edit.source.findingIds.map((id) => scenarioIds.sanitize(id)),
|
|
5791
|
+
rationale: edit.rationale ? scenarioIds.sanitize(edit.rationale) : null,
|
|
5792
|
+
validationPlan: edit.validationPlan ? scenarioIds.sanitize(edit.validationPlan) : null
|
|
5793
|
+
};
|
|
5794
|
+
}
|
|
5795
|
+
function sanitizeAuthorValue(value, scenarioIds) {
|
|
5796
|
+
if (typeof value === "string") return scenarioIds.sanitize(value);
|
|
5797
|
+
if (Array.isArray(value)) return value.map((item) => sanitizeAuthorValue(item, scenarioIds));
|
|
5798
|
+
if (value && typeof value === "object") {
|
|
5799
|
+
return Object.fromEntries(
|
|
5800
|
+
Object.entries(value).map(([key, child]) => [
|
|
5801
|
+
scenarioIds.sanitize(key),
|
|
5802
|
+
sanitizeAuthorValue(child, scenarioIds)
|
|
5803
|
+
])
|
|
5804
|
+
);
|
|
5805
|
+
}
|
|
5806
|
+
return value;
|
|
5807
|
+
}
|
|
5808
|
+
function sanitizeDimensions(dimensions, scenarioIds) {
|
|
5809
|
+
const sanitized = /* @__PURE__ */ new Map();
|
|
5810
|
+
for (const [key, value] of Object.entries(dimensions)) {
|
|
5811
|
+
const projected = scenarioIds.sanitize(key);
|
|
5812
|
+
const prior = sanitized.get(projected);
|
|
5813
|
+
if (prior && prior.original !== key) {
|
|
5814
|
+
throw new Error(
|
|
5815
|
+
`llmPolicyEditProposer: pseudonymized dimension key collision for '${prior.original}' and '${key}'`
|
|
5816
|
+
);
|
|
5817
|
+
}
|
|
5818
|
+
sanitized.set(projected, { original: key, value });
|
|
5819
|
+
}
|
|
5820
|
+
return Object.fromEntries([...sanitized].map(([key, entry]) => [key, entry.value]));
|
|
5821
|
+
}
|
|
5822
|
+
function forecastCalibration(candidate, record, objectiveByKey) {
|
|
5823
|
+
if (!record || candidate.observedDeltaFromParent === void 0) return null;
|
|
5824
|
+
const forecast = record.policyEdit.expectedGain;
|
|
5825
|
+
const objective = objectiveByKey.get(forecast.metric);
|
|
5826
|
+
if (!objective || objective.key !== "search.composite") return null;
|
|
5827
|
+
if (forecast.direction !== objective.direction || forecast.unit !== objective.unit) return null;
|
|
5828
|
+
if (forecast.amount > objective.scale.max - objective.scale.min) return null;
|
|
5829
|
+
const predictedDelta = forecast.amount;
|
|
5830
|
+
return {
|
|
5831
|
+
objectiveKey: objective.key,
|
|
5832
|
+
predictedDelta,
|
|
5833
|
+
observedDelta: candidate.observedDeltaFromParent,
|
|
5834
|
+
residual: candidate.observedDeltaFromParent - predictedDelta
|
|
5835
|
+
};
|
|
5836
|
+
}
|
|
5837
|
+
function assertSearchOutcome(outcome, field) {
|
|
5838
|
+
if (outcome && outcome.split !== "search") {
|
|
5839
|
+
throw new Error(`llmPolicyEditProposer: ${field} must be a search-split outcome`);
|
|
5840
|
+
}
|
|
5841
|
+
}
|
|
5842
|
+
function assertMeasuredCompositesInScale(ctx, objective) {
|
|
5843
|
+
const assertInScale = (value, field) => {
|
|
5844
|
+
if (!Number.isFinite(value) || value < objective.scale.min || value > objective.scale.max) {
|
|
5845
|
+
throw new Error(
|
|
5846
|
+
`llmPolicyEditProposer: ${field} ${value} is outside objective '${objective.key}' scale [${objective.scale.min}, ${objective.scale.max}]`
|
|
5847
|
+
);
|
|
5848
|
+
}
|
|
5849
|
+
};
|
|
5850
|
+
const assertOutcome = (outcome, field) => {
|
|
5851
|
+
if (!outcome) return;
|
|
5852
|
+
assertInScale(outcome.composite, `${field}.composite`);
|
|
5853
|
+
for (const [index, scenario] of outcome.scenarios.entries()) {
|
|
5854
|
+
assertInScale(scenario.composite, `${field}.scenarios[${index}].composite`);
|
|
5855
|
+
}
|
|
5856
|
+
};
|
|
5857
|
+
assertOutcome(ctx.baselineOutcome, "baselineOutcome");
|
|
5858
|
+
assertOutcome(ctx.incumbentOutcome, "incumbentOutcome");
|
|
5859
|
+
for (const [generationIndex, generation] of ctx.history.entries()) {
|
|
5860
|
+
for (const [candidateIndex, candidate] of generation.candidates.entries()) {
|
|
5861
|
+
const field = `history[${generationIndex}].candidates[${candidateIndex}]`;
|
|
5862
|
+
assertInScale(candidate.composite, `${field}.composite`);
|
|
5863
|
+
if (candidate.parentComposite !== void 0) {
|
|
5864
|
+
assertInScale(candidate.parentComposite, `${field}.parentComposite`);
|
|
5865
|
+
}
|
|
5866
|
+
for (const [scenarioIndex, scenario] of candidate.scenarios.entries()) {
|
|
5867
|
+
assertInScale(scenario.composite, `${field}.scenarios[${scenarioIndex}].composite`);
|
|
5868
|
+
}
|
|
5869
|
+
}
|
|
5870
|
+
}
|
|
5871
|
+
}
|
|
5872
|
+
function assertSurfaceIsTaskAgnostic(value, scenarioIds) {
|
|
5873
|
+
const serialized = JSON.stringify(value);
|
|
5874
|
+
if (scenarioIds.sanitize(serialized) !== serialized) {
|
|
5875
|
+
throw new Error(
|
|
5876
|
+
"llmPolicyEditProposer: current JSON surface or signature contains a raw scenario identifier"
|
|
5877
|
+
);
|
|
5878
|
+
}
|
|
5879
|
+
}
|
|
5880
|
+
function registerOutcomeScenarioIds(outcome, scenarioIds) {
|
|
5881
|
+
for (const scenario of outcome?.scenarios ?? []) scenarioIds.project(scenario.scenarioId);
|
|
5882
|
+
}
|
|
5883
|
+
function registerHistoryScenarioIds(history, scenarioIds) {
|
|
5884
|
+
for (const record of history) {
|
|
5885
|
+
for (const candidate of record.candidates) {
|
|
5886
|
+
for (const scenario of candidate.scenarios) scenarioIds.project(scenario.scenarioId);
|
|
5887
|
+
for (const cell of candidate.coverage?.unscorableCells ?? []) {
|
|
5888
|
+
const separator = cell.cellId.lastIndexOf(":");
|
|
5889
|
+
if (separator <= 0 || !/^\d+$/.test(cell.cellId.slice(separator + 1))) continue;
|
|
5890
|
+
scenarioIds.project(cell.cellId.slice(0, separator));
|
|
5891
|
+
}
|
|
5892
|
+
}
|
|
5893
|
+
}
|
|
5894
|
+
}
|
|
5895
|
+
function validateHistoryLimits(options) {
|
|
5896
|
+
const maxGenerations = options.maxGenerations ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.generations;
|
|
5897
|
+
const maxCandidatesPerGeneration = options.maxCandidatesPerGeneration ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.candidatesPerGeneration;
|
|
5898
|
+
const maxScenariosPerCandidate = options.maxScenariosPerCandidate ?? DEFAULT_POLICY_EDIT_HISTORY_LIMITS.scenariosPerCandidate;
|
|
5899
|
+
if (!Number.isSafeInteger(maxGenerations) || maxGenerations <= 0) {
|
|
5900
|
+
throw new Error("llmPolicyEditProposer: maxHistoryGenerations must be a positive safe integer");
|
|
5901
|
+
}
|
|
5902
|
+
if (!Number.isSafeInteger(maxCandidatesPerGeneration) || maxCandidatesPerGeneration <= 0) {
|
|
5903
|
+
throw new Error(
|
|
5904
|
+
"llmPolicyEditProposer: maxHistoryCandidatesPerGeneration must be a positive safe integer"
|
|
5905
|
+
);
|
|
5906
|
+
}
|
|
5907
|
+
if (!Number.isSafeInteger(maxScenariosPerCandidate) || maxScenariosPerCandidate <= 0) {
|
|
5908
|
+
throw new Error(
|
|
5909
|
+
"llmPolicyEditProposer: maxScenariosPerCandidate must be a positive safe integer"
|
|
5910
|
+
);
|
|
5911
|
+
}
|
|
5912
|
+
return {
|
|
5913
|
+
maxGenerations,
|
|
5914
|
+
maxCandidatesPerGeneration,
|
|
5915
|
+
maxScenariosPerCandidate,
|
|
5916
|
+
scenarioIdTransform: options.scenarioIdTransform ?? ((scenarioId) => scenarioId)
|
|
5917
|
+
};
|
|
5918
|
+
}
|
|
5919
|
+
function validateObjectives(inputs) {
|
|
5920
|
+
if (inputs.length === 0) {
|
|
5921
|
+
throw new Error("llmPolicyEditProposer: objectives must not be empty");
|
|
5922
|
+
}
|
|
5923
|
+
const seen = /* @__PURE__ */ new Set();
|
|
5924
|
+
return inputs.map((input) => {
|
|
5925
|
+
requireNonEmpty(input.key, "objective key");
|
|
5926
|
+
if (seen.has(input.key)) {
|
|
5927
|
+
throw new Error(`llmPolicyEditProposer: duplicate objective '${input.key}'`);
|
|
5928
|
+
}
|
|
5929
|
+
seen.add(input.key);
|
|
5930
|
+
if (input.split !== "search") {
|
|
5931
|
+
throw new Error(`llmPolicyEditProposer: objective '${input.key}' must use the search split`);
|
|
5932
|
+
}
|
|
5933
|
+
if (input.key !== "search.composite") {
|
|
5934
|
+
throw new Error(
|
|
5935
|
+
`llmPolicyEditProposer: objective '${input.key}' is not yet measurable; use 'search.composite'`
|
|
5936
|
+
);
|
|
5937
|
+
}
|
|
5938
|
+
if (input.direction !== "increase") {
|
|
5939
|
+
throw new Error(
|
|
5940
|
+
`llmPolicyEditProposer: objective '${input.key}' must increase because search promotes larger composite scores`
|
|
5941
|
+
);
|
|
5942
|
+
}
|
|
5943
|
+
if (!Number.isFinite(input.scale.min) || !Number.isFinite(input.scale.max) || input.scale.max <= input.scale.min) {
|
|
5944
|
+
throw new Error(`llmPolicyEditProposer: objective '${input.key}' has an invalid scale`);
|
|
5945
|
+
}
|
|
5946
|
+
if (input.unit !== "score") {
|
|
5947
|
+
throw new Error(`llmPolicyEditProposer: objective '${input.key}' must use raw score deltas`);
|
|
5948
|
+
}
|
|
5949
|
+
return {
|
|
5950
|
+
key: input.key,
|
|
5951
|
+
split: "search",
|
|
5952
|
+
direction: input.direction,
|
|
5953
|
+
scale: { ...input.scale },
|
|
5954
|
+
unit: input.unit
|
|
5955
|
+
};
|
|
5956
|
+
});
|
|
5957
|
+
}
|
|
5958
|
+
function positiveLimit(value, field) {
|
|
5959
|
+
if (!Number.isSafeInteger(value) || value <= 0) {
|
|
5960
|
+
throw new Error(`llmPolicyEditProposer: ${field} must be a positive safe integer`);
|
|
5961
|
+
}
|
|
5962
|
+
return value;
|
|
5963
|
+
}
|
|
5964
|
+
function parseAuthorResponse(value) {
|
|
5965
|
+
const parsed = PolicyEditAuthorResponseSchema.safeParse(value);
|
|
5966
|
+
if (parsed.success) return parsed.data;
|
|
5967
|
+
const issues = parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; ");
|
|
5968
|
+
throw new Error(`llmPolicyEditProposer: invalid PolicyEdit response: ${issues}`);
|
|
5969
|
+
}
|
|
5970
|
+
function bindAuthoredEdit(draft, findingByKey, targetSurface, allowedPaths, objectiveByKey, currentComposite) {
|
|
5971
|
+
if (draft.target.surface !== targetSurface) {
|
|
5972
|
+
throw new Error(
|
|
5973
|
+
`llmPolicyEditProposer: target surface '${draft.target.surface}' does not match '${targetSurface}'`
|
|
5974
|
+
);
|
|
5975
|
+
}
|
|
5976
|
+
if (draft.target.path !== draft.change.path) {
|
|
5977
|
+
throw new Error("llmPolicyEditProposer: target.path must equal change.path");
|
|
5978
|
+
}
|
|
5979
|
+
if (!allowedPaths.has(draft.change.path)) {
|
|
5980
|
+
throw new Error(
|
|
5981
|
+
`llmPolicyEditProposer: JSON path '${draft.change.path}' is outside allowedJsonPaths`
|
|
5982
|
+
);
|
|
5983
|
+
}
|
|
5984
|
+
const cited = draft.source.findingKeys.map((findingKey) => {
|
|
5985
|
+
const finding = findingByKey.get(findingKey);
|
|
5986
|
+
if (!finding) {
|
|
5987
|
+
throw new Error(
|
|
5988
|
+
`llmPolicyEditProposer: edit cites unknown or uncitable finding key '${findingKey}'`
|
|
5989
|
+
);
|
|
5990
|
+
}
|
|
5991
|
+
return finding;
|
|
5992
|
+
});
|
|
5993
|
+
const evidenceRefs = uniqueEvidenceRefs(cited.flatMap((finding) => finding.evidenceRefs));
|
|
5994
|
+
if (evidenceRefs.length === 0) {
|
|
5995
|
+
throw new Error("llmPolicyEditProposer: authored edit has no cited evidence");
|
|
5996
|
+
}
|
|
5997
|
+
const objective = objectiveByKey.get(draft.expectedGain.metric);
|
|
5998
|
+
if (!objective) {
|
|
5999
|
+
throw new Error(
|
|
6000
|
+
`llmPolicyEditProposer: unknown forecast objective '${draft.expectedGain.metric}'`
|
|
6001
|
+
);
|
|
6002
|
+
}
|
|
6003
|
+
if (draft.expectedGain.direction !== objective.direction) {
|
|
6004
|
+
throw new Error(
|
|
6005
|
+
`llmPolicyEditProposer: forecast direction for '${objective.key}' must be '${objective.direction}'`
|
|
6006
|
+
);
|
|
6007
|
+
}
|
|
6008
|
+
if (draft.expectedGain.unit !== objective.unit) {
|
|
6009
|
+
throw new Error(
|
|
6010
|
+
`llmPolicyEditProposer: forecast unit for '${objective.key}' must be '${objective.unit}'`
|
|
6011
|
+
);
|
|
6012
|
+
}
|
|
6013
|
+
const maxGain = currentComposite === void 0 ? objective.scale.max - objective.scale.min : objective.scale.max - currentComposite;
|
|
6014
|
+
if (draft.expectedGain.amount > maxGain) {
|
|
6015
|
+
throw new Error(
|
|
6016
|
+
`llmPolicyEditProposer: forecast amount for '${objective.key}' exceeds the available score headroom`
|
|
6017
|
+
);
|
|
6018
|
+
}
|
|
6019
|
+
const expectedGain = {
|
|
6020
|
+
metric: draft.expectedGain.metric,
|
|
6021
|
+
direction: draft.expectedGain.direction,
|
|
6022
|
+
amount: draft.expectedGain.amount,
|
|
6023
|
+
...draft.expectedGain.unit ? { unit: draft.expectedGain.unit } : {},
|
|
6024
|
+
...draft.expectedGain.rationale ? { rationale: draft.expectedGain.rationale } : {}
|
|
6025
|
+
};
|
|
6026
|
+
const init = {
|
|
6027
|
+
axis: draft.axis,
|
|
6028
|
+
target: {
|
|
6029
|
+
surface: draft.target.surface,
|
|
6030
|
+
path: draft.target.path,
|
|
6031
|
+
...draft.target.label ? { label: draft.target.label } : {}
|
|
6032
|
+
},
|
|
6033
|
+
change: draft.change,
|
|
6034
|
+
claim: draft.claim,
|
|
6035
|
+
expectedGain,
|
|
6036
|
+
confidence: draft.confidence,
|
|
6037
|
+
risk: draft.risk,
|
|
6038
|
+
source: {
|
|
6039
|
+
findingIds: [...new Set(cited.map((finding) => finding.finding.finding_id))],
|
|
6040
|
+
analystIds: [...new Set(cited.map((finding) => finding.finding.analyst_id))],
|
|
6041
|
+
evidenceRefs
|
|
6042
|
+
},
|
|
6043
|
+
...draft.rationale ? { rationale: draft.rationale } : {},
|
|
6044
|
+
...draft.validationPlan ? { validationPlan: draft.validationPlan } : {}
|
|
6045
|
+
};
|
|
6046
|
+
return makePolicyEdit(init);
|
|
6047
|
+
}
|
|
6048
|
+
function parseJsonSurface(surface) {
|
|
6049
|
+
if (typeof surface !== "string") {
|
|
6050
|
+
throw new Error("llmPolicyEditProposer: currentSurface must be serialized JSON");
|
|
6051
|
+
}
|
|
6052
|
+
let parsed;
|
|
6053
|
+
try {
|
|
6054
|
+
parsed = JSON.parse(surface);
|
|
6055
|
+
} catch {
|
|
6056
|
+
throw new Error("llmPolicyEditProposer: currentSurface must be valid JSON");
|
|
6057
|
+
}
|
|
6058
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
|
|
6059
|
+
throw new Error("llmPolicyEditProposer: currentSurface JSON root must be an object");
|
|
6060
|
+
}
|
|
6061
|
+
return parsed;
|
|
6062
|
+
}
|
|
6063
|
+
function citableFindings(inputs, measuredSources, limit) {
|
|
6064
|
+
if (inputs.length === 0) {
|
|
6065
|
+
throw new Error("llmPolicyEditProposer: at least one analyst finding is required");
|
|
6066
|
+
}
|
|
6067
|
+
const findings = [];
|
|
6068
|
+
for (const input of inputs) {
|
|
6069
|
+
if (!isPolicyEditFindingInput(input)) {
|
|
6070
|
+
throw new Error(
|
|
6071
|
+
"llmPolicyEditProposer: ctx.findings must contain attributed PolicyEditFindingInput rows"
|
|
6072
|
+
);
|
|
6073
|
+
}
|
|
6074
|
+
findings.push(input.finding);
|
|
6075
|
+
}
|
|
6076
|
+
assertNoJudgeVerdict(findings, "llmPolicyEditProposer");
|
|
6077
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
6078
|
+
for (const input of inputs) {
|
|
6079
|
+
validateFindingSource(input.source, measuredSources);
|
|
6080
|
+
if (input.finding.evidence_refs.length === 0) continue;
|
|
6081
|
+
const existing = grouped.get(input.finding.finding_id);
|
|
6082
|
+
if (existing) {
|
|
6083
|
+
if (existing.finding.analyst_id !== input.finding.analyst_id || existing.finding.area !== input.finding.area || existing.finding.claim !== input.finding.claim || existing.finding.subject !== input.finding.subject) {
|
|
6084
|
+
throw new Error(
|
|
6085
|
+
`llmPolicyEditProposer: finding '${input.finding.finding_id}' has conflicting content`
|
|
6086
|
+
);
|
|
6087
|
+
}
|
|
6088
|
+
if (!existing.sources.some((source) => sameFindingSource(source, input.source))) {
|
|
6089
|
+
existing.sources.push(input.source);
|
|
6090
|
+
}
|
|
6091
|
+
existing.evidenceRefs = uniqueEvidenceRefs([
|
|
6092
|
+
...existing.evidenceRefs,
|
|
6093
|
+
...input.finding.evidence_refs
|
|
6094
|
+
]);
|
|
6095
|
+
continue;
|
|
6096
|
+
}
|
|
6097
|
+
grouped.set(input.finding.finding_id, {
|
|
6098
|
+
finding: input.finding,
|
|
6099
|
+
sources: [input.source],
|
|
6100
|
+
evidenceRefs: uniqueEvidenceRefs(input.finding.evidence_refs)
|
|
6101
|
+
});
|
|
6102
|
+
}
|
|
6103
|
+
if (grouped.size === 0) {
|
|
6104
|
+
throw new Error("llmPolicyEditProposer: no evidence-bearing findings are available");
|
|
6105
|
+
}
|
|
6106
|
+
const severityRank = {
|
|
6107
|
+
critical: 0,
|
|
6108
|
+
high: 1,
|
|
6109
|
+
medium: 2,
|
|
6110
|
+
low: 3,
|
|
6111
|
+
info: 4
|
|
6112
|
+
};
|
|
6113
|
+
return [...grouped.values()].sort(
|
|
6114
|
+
(a, b) => severityRank[a.finding.severity] - severityRank[b.finding.severity] || b.finding.confidence - a.finding.confidence || a.finding.finding_id.localeCompare(b.finding.finding_id)
|
|
6115
|
+
).slice(0, limit);
|
|
6116
|
+
}
|
|
6117
|
+
function isAnalystFindingLike2(input) {
|
|
6118
|
+
if (!input || typeof input !== "object") return false;
|
|
6119
|
+
const value = input;
|
|
6120
|
+
return typeof value.finding_id === "string" && typeof value.analyst_id === "string" && typeof value.claim === "string" && Array.isArray(value.evidence_refs);
|
|
6121
|
+
}
|
|
6122
|
+
function isPolicyEditFindingInput(input) {
|
|
6123
|
+
if (!input || typeof input !== "object") return false;
|
|
6124
|
+
const value = input;
|
|
6125
|
+
return isAnalystFindingLike2(value.finding) && isFindingSource(value.source);
|
|
6126
|
+
}
|
|
6127
|
+
function isFindingSource(input) {
|
|
6128
|
+
if (!input || typeof input !== "object") return false;
|
|
6129
|
+
const value = input;
|
|
6130
|
+
if (value.kind === "surface") {
|
|
6131
|
+
return typeof value.surfaceHash === "string" && Number.isSafeInteger(value.generation);
|
|
6132
|
+
}
|
|
6133
|
+
return value.kind === "global" && typeof value.label === "string" && value.label.trim().length > 0;
|
|
6134
|
+
}
|
|
6135
|
+
function renderFinding(context, findingKey, scenarioIds, measuredSources) {
|
|
6136
|
+
const { finding } = context;
|
|
6137
|
+
return {
|
|
6138
|
+
findingKey,
|
|
6139
|
+
sources: context.sources.map(
|
|
6140
|
+
(source) => source.kind === "surface" ? {
|
|
6141
|
+
...measuredSources.get(sourceKey(source.surfaceHash, source.generation)),
|
|
6142
|
+
surfaceHash: scenarioIds.sanitize(source.surfaceHash),
|
|
6143
|
+
kind: "surface"
|
|
6144
|
+
} : { kind: "global", label: scenarioIds.sanitize(source.label) }
|
|
6145
|
+
),
|
|
6146
|
+
analystId: scenarioIds.sanitize(finding.analyst_id),
|
|
6147
|
+
area: scenarioIds.sanitize(finding.area),
|
|
6148
|
+
severity: finding.severity,
|
|
6149
|
+
subject: finding.subject ? scenarioIds.sanitize(finding.subject) : null,
|
|
6150
|
+
claim: scenarioIds.sanitize(finding.claim),
|
|
6151
|
+
rationale: finding.rationale ? scenarioIds.sanitize(finding.rationale) : null,
|
|
6152
|
+
recommendedAction: finding.recommended_action ? scenarioIds.sanitize(finding.recommended_action) : null,
|
|
6153
|
+
validationPlan: finding.validation_plan ? scenarioIds.sanitize(finding.validation_plan) : null,
|
|
6154
|
+
confidence: finding.confidence,
|
|
6155
|
+
evidenceRefs: context.evidenceRefs.map((ref) => ({
|
|
6156
|
+
kind: ref.kind,
|
|
6157
|
+
uri: scenarioIds.sanitize(ref.uri),
|
|
6158
|
+
excerpt: ref.excerpt ? scenarioIds.sanitize(ref.excerpt) : null
|
|
6159
|
+
}))
|
|
6160
|
+
};
|
|
6161
|
+
}
|
|
6162
|
+
function uniqueEvidenceRefs(refs) {
|
|
6163
|
+
const seen = /* @__PURE__ */ new Set();
|
|
6164
|
+
return refs.filter((ref) => {
|
|
6165
|
+
const key = JSON.stringify([ref.kind, ref.uri, ref.excerpt ?? null]);
|
|
6166
|
+
if (seen.has(key)) return false;
|
|
6167
|
+
seen.add(key);
|
|
6168
|
+
return true;
|
|
6169
|
+
});
|
|
6170
|
+
}
|
|
6171
|
+
function validateAllowedJsonPaths(paths) {
|
|
6172
|
+
if (paths.length === 0) {
|
|
6173
|
+
throw new Error("llmPolicyEditProposer: allowedJsonPaths must not be empty");
|
|
6174
|
+
}
|
|
6175
|
+
for (const path of paths) {
|
|
6176
|
+
if (!path || path.trim() !== path) {
|
|
6177
|
+
throw new Error(
|
|
6178
|
+
"llmPolicyEditProposer: allowedJsonPaths must contain trimmed non-empty paths"
|
|
6179
|
+
);
|
|
6180
|
+
}
|
|
6181
|
+
}
|
|
6182
|
+
if (new Set(paths).size !== paths.length) {
|
|
6183
|
+
throw new Error("llmPolicyEditProposer: allowedJsonPaths must be unique");
|
|
6184
|
+
}
|
|
6185
|
+
return [...paths];
|
|
6186
|
+
}
|
|
6187
|
+
function requireNonEmpty(value, field) {
|
|
6188
|
+
if (!value || value.trim() !== value) {
|
|
6189
|
+
throw new Error(`llmPolicyEditProposer: ${field} must be a trimmed non-empty string`);
|
|
6190
|
+
}
|
|
6191
|
+
}
|
|
6192
|
+
|
|
6193
|
+
// src/campaign/proposers/memory.ts
|
|
6194
|
+
var BLOCK_START2 = "<!-- BEGIN curated-memory (auto-managed by memoryCurationProposer) -->";
|
|
6195
|
+
var BLOCK_END2 = "<!-- END curated-memory -->";
|
|
6196
|
+
var DEFAULT_HEADING2 = "## Learned from prior runs (curated memory)";
|
|
6197
|
+
var DISTILL_SYSTEM = 'You compress raw trace-analysis findings into crisp, generalizable agent guidance. Output ONLY a JSON array of strings, each one imperative lesson the agent should follow (e.g. "Always fetch a resource before mutating it"). No prose outside the JSON. Deduplicate; keep the most actionable and general; drop case-specific noise.';
|
|
6198
|
+
function extractExistingLessons(text) {
|
|
6199
|
+
return extractBlockBody(text, BLOCK_START2, BLOCK_END2).split("\n").map((l) => l.replace(/^\s*-\s+/, "").trim()).filter((l) => l && !l.startsWith("#"));
|
|
6200
|
+
}
|
|
6201
|
+
async function distillLessons(raw, distill, ctx, costLedger) {
|
|
6202
|
+
const request = {
|
|
6203
|
+
model: distill.model,
|
|
6204
|
+
messages: [
|
|
6205
|
+
{ role: "system", content: DISTILL_SYSTEM },
|
|
6206
|
+
{ role: "user", content: `Findings:
|
|
6207
|
+
${raw.map((r) => `- ${r}`).join("\n")}` }
|
|
6208
|
+
],
|
|
6209
|
+
maxTokens: distill.maxTokens ?? 2e3
|
|
6210
|
+
};
|
|
6211
|
+
const llm = {
|
|
6212
|
+
baseUrl: distill.baseUrl,
|
|
6213
|
+
apiKey: distill.apiKey,
|
|
6214
|
+
fetch: distill.fetchImpl
|
|
6215
|
+
};
|
|
6216
|
+
const paid = await costLedger.runPaidCall({
|
|
6217
|
+
channel: "driver",
|
|
6218
|
+
phase: ctx.costPhase ?? "search.proposal",
|
|
6219
|
+
actor: "memory-curation.distill",
|
|
6220
|
+
model: distill.model,
|
|
6221
|
+
maximumCharge: maximumChargeForLlmRequest(request, llm),
|
|
6222
|
+
tags: { generation: String(ctx.generation) },
|
|
6223
|
+
signal: ctx.signal,
|
|
6224
|
+
execute: (signal, callId) => callLlm(request, { ...llm, signal, idempotencyKey: callId }),
|
|
6225
|
+
receipt: costReceiptFromLlm,
|
|
6226
|
+
receiptFromError: costReceiptFromLlmError
|
|
6227
|
+
});
|
|
6228
|
+
if (!paid.succeeded) throw paid.error;
|
|
6229
|
+
const res = paid.value;
|
|
6230
|
+
try {
|
|
6231
|
+
const parsed = JSON.parse(res.content.trim());
|
|
6232
|
+
if (Array.isArray(parsed)) {
|
|
6233
|
+
const lessons = parsed.filter(
|
|
6234
|
+
(x) => typeof x === "string" && x.trim().length > 0
|
|
6235
|
+
);
|
|
6236
|
+
if (lessons.length > 0) return lessons;
|
|
6237
|
+
}
|
|
6238
|
+
} catch {
|
|
6239
|
+
}
|
|
6240
|
+
return raw;
|
|
6241
|
+
}
|
|
6242
|
+
function memoryCurationProposer(opts = {}) {
|
|
6243
|
+
const maxEntries = opts.maxEntries ?? 12;
|
|
6244
|
+
const heading = opts.sectionHeading ?? DEFAULT_HEADING2;
|
|
6245
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
6246
|
+
return {
|
|
6247
|
+
kind: "memory-curation",
|
|
6248
|
+
async propose(ctx) {
|
|
6249
|
+
const parent = surfaceToText2(ctx.currentSurface);
|
|
6250
|
+
const fresh = [];
|
|
6251
|
+
for (const f of ctx.findings ?? []) {
|
|
6252
|
+
const l = findingToLesson(f);
|
|
6253
|
+
if (l) fresh.push(l);
|
|
6254
|
+
}
|
|
6255
|
+
const carried = extractExistingLessons(parent);
|
|
6256
|
+
if (fresh.length === 0 && carried.length === 0) return [];
|
|
6257
|
+
const distilled = opts.distill && fresh.length > 0 ? await distillLessons(fresh, opts.distill, ctx, ctx.costLedger ?? directCostLedger) : fresh;
|
|
6258
|
+
const byKey = /* @__PURE__ */ new Map();
|
|
6259
|
+
for (const l of carried) {
|
|
6260
|
+
const k = normKey(l);
|
|
6261
|
+
if (k) byKey.set(k, { text: l, count: 1 });
|
|
6262
|
+
}
|
|
6263
|
+
for (const l of distilled) {
|
|
6264
|
+
const k = normKey(l);
|
|
6265
|
+
if (!k) continue;
|
|
6266
|
+
const e = byKey.get(k);
|
|
6267
|
+
if (e) e.count += 1;
|
|
6268
|
+
else byKey.set(k, { text: l, count: 1 });
|
|
6269
|
+
}
|
|
6270
|
+
const ranked = [...byKey.values()].sort((a, b) => b.count - a.count || a.text.localeCompare(b.text)).slice(0, maxEntries);
|
|
6271
|
+
if (ranked.length === 0) return [];
|
|
6272
|
+
const block = [BLOCK_START2, heading, ...ranked.map((e) => `- ${e.text}`), BLOCK_END2].join(
|
|
6273
|
+
"\n"
|
|
6274
|
+
);
|
|
6275
|
+
const next = `${stripBlock(parent, BLOCK_START2, BLOCK_END2)}
|
|
6276
|
+
|
|
6277
|
+
${block}
|
|
6278
|
+
`;
|
|
6279
|
+
if (next === parent) return [];
|
|
6280
|
+
return [
|
|
6281
|
+
{
|
|
6282
|
+
surface: next,
|
|
6283
|
+
label: "memory-curation",
|
|
6284
|
+
rationale: `curated ${ranked.length} lessons (from ${fresh.length} new finding(s) + ${carried.length} carried)`
|
|
6285
|
+
}
|
|
6286
|
+
];
|
|
6287
|
+
}
|
|
6288
|
+
};
|
|
6289
|
+
}
|
|
6290
|
+
|
|
6291
|
+
// src/campaign/proposers/trace-analyst.ts
|
|
6292
|
+
import { ai } from "@ax-llm/ax";
|
|
6293
|
+
function renderFindings(findings) {
|
|
6294
|
+
return findings.map((f, i) => {
|
|
6295
|
+
const action = f.recommended_action ? `
|
|
6296
|
+
FIX: ${f.recommended_action}` : "";
|
|
6297
|
+
const subject = f.subject ? ` (${f.subject})` : "";
|
|
6298
|
+
return `${i + 1}. [${f.severity}/${f.area}]${subject} ${f.claim}${action}`;
|
|
6299
|
+
}).join("\n");
|
|
6300
|
+
}
|
|
6301
|
+
function traceAnalystProposer(opts) {
|
|
6302
|
+
if (!opts.apiKey) throw new Error("traceAnalystProposer: apiKey is required");
|
|
5422
6303
|
if (!opts.model) throw new Error("traceAnalystProposer: model is required");
|
|
5423
6304
|
const kinds = opts.kinds ?? DEFAULT_TRACE_ANALYST_KINDS;
|
|
5424
6305
|
const produceFindings = opts.analyze ?? (async (path, c) => {
|
|
@@ -5444,7 +6325,12 @@ function traceAnalystProposer(opts) {
|
|
|
5444
6325
|
label: "trace-analyst",
|
|
5445
6326
|
baseUrl: opts.baseUrl,
|
|
5446
6327
|
apiKey: opts.apiKey,
|
|
6328
|
+
analysisModel: opts.model,
|
|
5447
6329
|
applyModel: opts.applyModel ?? opts.model,
|
|
6330
|
+
costLedger: opts.costLedger,
|
|
6331
|
+
analysisMaximumCharge: opts.analysisMaximumCharge,
|
|
6332
|
+
analysisReceipt: opts.analysisReceipt,
|
|
6333
|
+
applyMaxTokens: opts.applyMaxTokens,
|
|
5448
6334
|
fetchImpl: opts.fetchImpl,
|
|
5449
6335
|
resolveTraces: opts.resolveTraces,
|
|
5450
6336
|
noTracesError: "traceAnalystProposer: resolveTraces returned no OTLP traces \u2014 the analyst has nothing to read",
|
|
@@ -5516,161 +6402,9 @@ function selectDiscriminative(signals, k, opts) {
|
|
|
5516
6402
|
|
|
5517
6403
|
// src/campaign/search-ledger.ts
|
|
5518
6404
|
import { createHash as createHash7 } from "crypto";
|
|
5519
|
-
import { existsSync as
|
|
6405
|
+
import { existsSync as existsSync3, readFileSync as readFileSync3 } from "fs";
|
|
5520
6406
|
import { resolve as resolve2 } from "path";
|
|
5521
6407
|
import { z as z2 } from "zod";
|
|
5522
|
-
|
|
5523
|
-
// src/campaign/search-ledger-errors.ts
|
|
5524
|
-
var SearchLedgerError = class extends ValidationError {
|
|
5525
|
-
};
|
|
5526
|
-
var SearchLedgerIntegrityError = class extends SearchLedgerError {
|
|
5527
|
-
};
|
|
5528
|
-
var SearchLedgerConflictError = class extends SearchLedgerError {
|
|
5529
|
-
};
|
|
5530
|
-
|
|
5531
|
-
// src/campaign/search-ledger-file.ts
|
|
5532
|
-
import { randomUUID as randomUUID2 } from "crypto";
|
|
5533
|
-
import {
|
|
5534
|
-
closeSync,
|
|
5535
|
-
constants,
|
|
5536
|
-
existsSync as existsSync3,
|
|
5537
|
-
fsyncSync,
|
|
5538
|
-
linkSync,
|
|
5539
|
-
mkdirSync as mkdirSync2,
|
|
5540
|
-
openSync,
|
|
5541
|
-
readFileSync as readFileSync3,
|
|
5542
|
-
renameSync,
|
|
5543
|
-
unlinkSync,
|
|
5544
|
-
writeSync
|
|
5545
|
-
} from "fs";
|
|
5546
|
-
import { hostname } from "os";
|
|
5547
|
-
import { dirname as dirname2 } from "path";
|
|
5548
|
-
import { z } from "zod";
|
|
5549
|
-
function appendSearchLedgerLine(path, line) {
|
|
5550
|
-
mkdirSync2(dirname2(path), { recursive: true });
|
|
5551
|
-
const fd = openSync(path, constants.O_CREAT | constants.O_WRONLY | constants.O_APPEND, 384);
|
|
5552
|
-
try {
|
|
5553
|
-
writeAll(fd, Buffer.from(line, "utf8"));
|
|
5554
|
-
fsyncSync(fd);
|
|
5555
|
-
} finally {
|
|
5556
|
-
closeSync(fd);
|
|
5557
|
-
}
|
|
5558
|
-
fsyncDirectory(dirname2(path));
|
|
5559
|
-
}
|
|
5560
|
-
function withSearchLedgerFileLock(ledgerPath, run) {
|
|
5561
|
-
mkdirSync2(dirname2(ledgerPath), { recursive: true });
|
|
5562
|
-
const lockPath = `${ledgerPath}.lock`;
|
|
5563
|
-
const owner = acquireLock(lockPath);
|
|
5564
|
-
try {
|
|
5565
|
-
return run();
|
|
5566
|
-
} finally {
|
|
5567
|
-
releaseLock(lockPath, owner);
|
|
5568
|
-
}
|
|
5569
|
-
}
|
|
5570
|
-
function writeAll(fd, bytes) {
|
|
5571
|
-
let offset = 0;
|
|
5572
|
-
while (offset < bytes.byteLength) {
|
|
5573
|
-
const written = writeSync(fd, bytes, offset, bytes.byteLength - offset);
|
|
5574
|
-
if (written <= 0) throw new SearchLedgerIntegrityError("filesystem wrote zero bytes");
|
|
5575
|
-
offset += written;
|
|
5576
|
-
}
|
|
5577
|
-
}
|
|
5578
|
-
function fsyncDirectory(path) {
|
|
5579
|
-
const fd = openSync(path, constants.O_RDONLY);
|
|
5580
|
-
try {
|
|
5581
|
-
fsyncSync(fd);
|
|
5582
|
-
} finally {
|
|
5583
|
-
closeSync(fd);
|
|
5584
|
-
}
|
|
5585
|
-
}
|
|
5586
|
-
function acquireLock(lockPath) {
|
|
5587
|
-
const owner = { pid: process.pid, host: hostname(), nonce: randomUUID2() };
|
|
5588
|
-
const ownerBytes = `${canonicalOwner(owner)}
|
|
5589
|
-
`;
|
|
5590
|
-
for (let attempt = 0; attempt < 8; attempt += 1) {
|
|
5591
|
-
const ownerPath = `${lockPath}.${owner.pid}.${owner.nonce}.${attempt}.owner`;
|
|
5592
|
-
const fd = openSync(ownerPath, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY, 384);
|
|
5593
|
-
try {
|
|
5594
|
-
writeAll(fd, Buffer.from(ownerBytes, "utf8"));
|
|
5595
|
-
fsyncSync(fd);
|
|
5596
|
-
} finally {
|
|
5597
|
-
closeSync(fd);
|
|
5598
|
-
}
|
|
5599
|
-
try {
|
|
5600
|
-
linkSync(ownerPath, lockPath);
|
|
5601
|
-
unlinkSync(ownerPath);
|
|
5602
|
-
return owner;
|
|
5603
|
-
} catch (error) {
|
|
5604
|
-
unlinkIfExists(ownerPath);
|
|
5605
|
-
if (error.code !== "EEXIST") throw error;
|
|
5606
|
-
}
|
|
5607
|
-
const holder = readOwner(lockPath);
|
|
5608
|
-
if (holder.host !== owner.host || isProcessAlive(holder.pid)) {
|
|
5609
|
-
throw new SearchLedgerIntegrityError(
|
|
5610
|
-
`search ledger lock is held by pid ${holder.pid} on ${holder.host}`
|
|
5611
|
-
);
|
|
5612
|
-
}
|
|
5613
|
-
const tombstone = `${lockPath}.stale.${owner.nonce}.${attempt}`;
|
|
5614
|
-
try {
|
|
5615
|
-
renameSync(lockPath, tombstone);
|
|
5616
|
-
unlinkSync(tombstone);
|
|
5617
|
-
} catch (error) {
|
|
5618
|
-
if (error.code !== "ENOENT") throw error;
|
|
5619
|
-
}
|
|
5620
|
-
}
|
|
5621
|
-
throw new SearchLedgerIntegrityError(`could not acquire search ledger lock ${lockPath}`);
|
|
5622
|
-
}
|
|
5623
|
-
function releaseLock(lockPath, owner) {
|
|
5624
|
-
if (!existsSync3(lockPath)) return;
|
|
5625
|
-
const holder = readOwner(lockPath);
|
|
5626
|
-
if (canonicalOwner(holder) !== canonicalOwner(owner)) {
|
|
5627
|
-
throw new SearchLedgerIntegrityError(
|
|
5628
|
-
`search ledger lock owner changed before release (${lockPath})`
|
|
5629
|
-
);
|
|
5630
|
-
}
|
|
5631
|
-
unlinkSync(lockPath);
|
|
5632
|
-
}
|
|
5633
|
-
function readOwner(lockPath) {
|
|
5634
|
-
let raw;
|
|
5635
|
-
try {
|
|
5636
|
-
raw = JSON.parse(readFileSync3(lockPath, "utf8"));
|
|
5637
|
-
} catch (error) {
|
|
5638
|
-
throw new SearchLedgerIntegrityError(`search ledger lock ${lockPath} is malformed`, {
|
|
5639
|
-
cause: error
|
|
5640
|
-
});
|
|
5641
|
-
}
|
|
5642
|
-
const parsed = z.object({
|
|
5643
|
-
pid: z.number().int().positive().safe(),
|
|
5644
|
-
host: z.string().trim().min(1),
|
|
5645
|
-
nonce: z.string().trim().min(1)
|
|
5646
|
-
}).strict().safeParse(raw);
|
|
5647
|
-
if (!parsed.success) {
|
|
5648
|
-
throw new SearchLedgerIntegrityError(
|
|
5649
|
-
`search ledger lock ${lockPath} is malformed: ${parsed.error.issues.map((issue) => `${issue.path.join(".") || "<root>"}: ${issue.message}`).join("; ")}`
|
|
5650
|
-
);
|
|
5651
|
-
}
|
|
5652
|
-
return parsed.data;
|
|
5653
|
-
}
|
|
5654
|
-
function canonicalOwner(owner) {
|
|
5655
|
-
return JSON.stringify({ host: owner.host, nonce: owner.nonce, pid: owner.pid });
|
|
5656
|
-
}
|
|
5657
|
-
function isProcessAlive(pid) {
|
|
5658
|
-
try {
|
|
5659
|
-
process.kill(pid, 0);
|
|
5660
|
-
return true;
|
|
5661
|
-
} catch (error) {
|
|
5662
|
-
return error.code !== "ESRCH";
|
|
5663
|
-
}
|
|
5664
|
-
}
|
|
5665
|
-
function unlinkIfExists(path) {
|
|
5666
|
-
try {
|
|
5667
|
-
unlinkSync(path);
|
|
5668
|
-
} catch (error) {
|
|
5669
|
-
if (error.code !== "ENOENT") throw error;
|
|
5670
|
-
}
|
|
5671
|
-
}
|
|
5672
|
-
|
|
5673
|
-
// src/campaign/search-ledger.ts
|
|
5674
6408
|
var SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
|
|
5675
6409
|
var NON_EMPTY = z2.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
|
|
5676
6410
|
var HASH = z2.string().regex(/^sha256:[a-f0-9]{64}$/);
|
|
@@ -6036,8 +6770,8 @@ var FileSearchLedger = class {
|
|
|
6036
6770
|
}
|
|
6037
6771
|
};
|
|
6038
6772
|
function replayFile(path, campaignId) {
|
|
6039
|
-
if (!
|
|
6040
|
-
const text =
|
|
6773
|
+
if (!existsSync3(path)) return replayEntries([], campaignId);
|
|
6774
|
+
const text = readFileSync3(path, "utf8");
|
|
6041
6775
|
if (text.length === 0) return replayEntries([], campaignId);
|
|
6042
6776
|
if (!text.endsWith("\n")) {
|
|
6043
6777
|
throw new SearchLedgerIntegrityError(
|
|
@@ -6633,10 +7367,10 @@ function formatZodError(error) {
|
|
|
6633
7367
|
}
|
|
6634
7368
|
|
|
6635
7369
|
// src/campaign/single-run-lock.ts
|
|
6636
|
-
import { existsSync as
|
|
7370
|
+
import { existsSync as existsSync4, readFileSync as readFileSync4, unlinkSync, writeFileSync as writeFileSync3 } from "fs";
|
|
6637
7371
|
function liveHolder(path) {
|
|
6638
|
-
if (!
|
|
6639
|
-
const holder = Number(
|
|
7372
|
+
if (!existsSync4(path)) return null;
|
|
7373
|
+
const holder = Number(readFileSync4(path, "utf8").trim());
|
|
6640
7374
|
if (!Number.isFinite(holder) || holder <= 0) return null;
|
|
6641
7375
|
try {
|
|
6642
7376
|
process.kill(holder, 0);
|
|
@@ -6658,8 +7392,8 @@ function acquireSingleRunLock(opts) {
|
|
|
6658
7392
|
writeFileSync3(opts.lockPath, String(pid));
|
|
6659
7393
|
const release = () => {
|
|
6660
7394
|
try {
|
|
6661
|
-
if (
|
|
6662
|
-
|
|
7395
|
+
if (existsSync4(opts.lockPath) && readFileSync4(opts.lockPath, "utf8").trim() === String(pid)) {
|
|
7396
|
+
unlinkSync(opts.lockPath);
|
|
6663
7397
|
}
|
|
6664
7398
|
} catch {
|
|
6665
7399
|
}
|
|
@@ -6683,21 +7417,21 @@ function isTransientTransportFailure(message, opts = {}) {
|
|
|
6683
7417
|
import { execFileSync } from "child_process";
|
|
6684
7418
|
import { createHash as createHash8 } from "crypto";
|
|
6685
7419
|
import {
|
|
6686
|
-
closeSync
|
|
6687
|
-
existsSync as
|
|
7420
|
+
closeSync,
|
|
7421
|
+
existsSync as existsSync5,
|
|
6688
7422
|
constants as fsConstants,
|
|
6689
7423
|
fstatSync,
|
|
6690
7424
|
lstatSync,
|
|
6691
|
-
mkdirSync as
|
|
7425
|
+
mkdirSync as mkdirSync2,
|
|
6692
7426
|
mkdtempSync as mkdtempSync2,
|
|
6693
|
-
openSync
|
|
7427
|
+
openSync,
|
|
6694
7428
|
readlinkSync,
|
|
6695
7429
|
readSync,
|
|
6696
7430
|
realpathSync,
|
|
6697
7431
|
rmSync
|
|
6698
7432
|
} from "fs";
|
|
6699
7433
|
import { devNull, tmpdir as tmpdir2 } from "os";
|
|
6700
|
-
import { basename, dirname as
|
|
7434
|
+
import { basename, dirname as dirname2, isAbsolute as isAbsolute2, join as join5, relative as relative2, resolve as resolve3, sep } from "path";
|
|
6701
7435
|
var MAX_GIT_OUTPUT_BYTES = 256 * 1024 * 1024;
|
|
6702
7436
|
var FILE_HASH_CHUNK_BYTES = 1024 * 1024;
|
|
6703
7437
|
var GIT_REPOSITORY_ENV = /* @__PURE__ */ new Set([
|
|
@@ -6790,7 +7524,7 @@ function patchBytes(git, cwd, baseCommit, candidateCommit) {
|
|
|
6790
7524
|
const scratch = mkdtempSync2(join5(tmpdir2(), "agent-eval-patch-"));
|
|
6791
7525
|
const bareRepo = join5(scratch, "repo.git");
|
|
6792
7526
|
const emptyTemplate = join5(scratch, "empty-template");
|
|
6793
|
-
|
|
7527
|
+
mkdirSync2(emptyTemplate);
|
|
6794
7528
|
try {
|
|
6795
7529
|
const objectFormat = gitObjectHashAlgorithm(candidateCommit);
|
|
6796
7530
|
const sourceObjects = realpathSync(gitText(git, ["rev-parse", "--git-path", "objects"], cwd));
|
|
@@ -6928,7 +7662,7 @@ function hashGitBlobFile(path, objectId) {
|
|
|
6928
7662
|
throw new WorktreeAdapterError(`CodeSurface expected a regular file at ${displayGitPath(path)}`);
|
|
6929
7663
|
}
|
|
6930
7664
|
const noFollow = process.platform === "win32" ? 0 : fsConstants.O_NOFOLLOW;
|
|
6931
|
-
const fd =
|
|
7665
|
+
const fd = openSync(path, fsConstants.O_RDONLY | noFollow);
|
|
6932
7666
|
try {
|
|
6933
7667
|
const opened = fstatSync(fd);
|
|
6934
7668
|
if (!opened.isFile() || opened.dev !== before.dev || opened.ino !== before.ino || opened.size !== before.size) {
|
|
@@ -6953,7 +7687,7 @@ function hashGitBlobFile(path, objectId) {
|
|
|
6953
7687
|
}
|
|
6954
7688
|
return { hash: hash.digest("hex"), executable: (opened.mode & 73) !== 0 };
|
|
6955
7689
|
} finally {
|
|
6956
|
-
|
|
7690
|
+
closeSync(fd);
|
|
6957
7691
|
}
|
|
6958
7692
|
}
|
|
6959
7693
|
function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
|
|
@@ -6969,7 +7703,7 @@ function assertSymlinkTargetIsBound(root, linkPath, trackedPaths) {
|
|
|
6969
7703
|
`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
|
|
6970
7704
|
);
|
|
6971
7705
|
}
|
|
6972
|
-
const lexicalTarget = resolve3(
|
|
7706
|
+
const lexicalTarget = resolve3(dirname2(linkPath), target);
|
|
6973
7707
|
if (!isWithinRoot(root, lexicalTarget)) {
|
|
6974
7708
|
throw new WorktreeAdapterError(
|
|
6975
7709
|
`CodeSurface symbolic link escapes its worktree at ${displayGitPath(linkPath)}`
|
|
@@ -7046,7 +7780,7 @@ function assertRawTreeMatchesWorktree(git, root, candidateCommit) {
|
|
|
7046
7780
|
}
|
|
7047
7781
|
function verifyCodeSurfaceWithGit(surface, path, git) {
|
|
7048
7782
|
assertCodeSurfaceIdentity(surface);
|
|
7049
|
-
if (!
|
|
7783
|
+
if (!existsSync5(path)) {
|
|
7050
7784
|
throw new WorktreeAdapterError(`CodeSurface worktree does not exist: ${path}`);
|
|
7051
7785
|
}
|
|
7052
7786
|
const lexicalRoot = resolve3(path);
|
|
@@ -7227,13 +7961,6 @@ function resolveWorktreePath(surface, worktreeDir) {
|
|
|
7227
7961
|
}
|
|
7228
7962
|
|
|
7229
7963
|
export {
|
|
7230
|
-
JudgeParseError,
|
|
7231
|
-
createDomainExpertJudge,
|
|
7232
|
-
codeExecutionJudge,
|
|
7233
|
-
coherenceJudge,
|
|
7234
|
-
adversarialJudge,
|
|
7235
|
-
createCustomJudge,
|
|
7236
|
-
defaultJudges,
|
|
7237
7964
|
pairArms,
|
|
7238
7965
|
comparePairedArms,
|
|
7239
7966
|
completionVerdict,
|
|
@@ -7241,7 +7968,6 @@ export {
|
|
|
7241
7968
|
parseCorrectnessResponse,
|
|
7242
7969
|
createLlmCorrectnessChecker,
|
|
7243
7970
|
createTokenRecallChecker,
|
|
7244
|
-
llmJudge,
|
|
7245
7971
|
extractProducedState,
|
|
7246
7972
|
CODING_HARNESSES,
|
|
7247
7973
|
HARNESS_NATIVE_MODEL,
|
|
@@ -7297,14 +8023,16 @@ export {
|
|
|
7297
8023
|
aceProposer,
|
|
7298
8024
|
compositeProposer,
|
|
7299
8025
|
haloProposer,
|
|
7300
|
-
memoryCurationProposer,
|
|
7301
8026
|
policyEditProposer,
|
|
8027
|
+
selectPolicyEditAuthorRows,
|
|
8028
|
+
assertPolicyEditAuthorContextBudget,
|
|
8029
|
+
DEFAULT_POLICY_EDIT_HISTORY_LIMITS,
|
|
8030
|
+
llmPolicyEditProposer,
|
|
8031
|
+
projectPolicyEditHistory,
|
|
8032
|
+
memoryCurationProposer,
|
|
7302
8033
|
traceAnalystProposer,
|
|
7303
8034
|
scoreDiscrimination,
|
|
7304
8035
|
selectDiscriminative,
|
|
7305
|
-
SearchLedgerError,
|
|
7306
|
-
SearchLedgerIntegrityError,
|
|
7307
|
-
SearchLedgerConflictError,
|
|
7308
8036
|
SEARCH_LEDGER_SCHEMA,
|
|
7309
8037
|
validateSearchLedgerEvent,
|
|
7310
8038
|
openSearchLedger,
|
|
@@ -7316,4 +8044,4 @@ export {
|
|
|
7316
8044
|
verifyCodeSurface,
|
|
7317
8045
|
resolveWorktreePath
|
|
7318
8046
|
};
|
|
7319
|
-
//# sourceMappingURL=chunk-
|
|
8047
|
+
//# sourceMappingURL=chunk-ZUXV7UWZ.js.map
|