@tangle-network/agent-eval 0.128.1 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +271 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
|
@@ -13,7 +13,7 @@ import {
|
|
|
13
13
|
runCampaign,
|
|
14
14
|
summarizeAgentReceiptIntegrity,
|
|
15
15
|
tryAcquireAtomicFileLock
|
|
16
|
-
} from "./chunk-
|
|
16
|
+
} from "./chunk-C6LXANRU.js";
|
|
17
17
|
import {
|
|
18
18
|
clamp01,
|
|
19
19
|
combineAbortSignals,
|
|
@@ -21,25 +21,25 @@ import {
|
|
|
21
21
|
} from "./chunk-WGXIEX7P.js";
|
|
22
22
|
import {
|
|
23
23
|
detectRewardHacking
|
|
24
|
-
} from "./chunk-
|
|
24
|
+
} from "./chunk-EG66UGL4.js";
|
|
25
25
|
import {
|
|
26
26
|
campaignCellExecutionEvidence,
|
|
27
27
|
projectCampaignCellQuality
|
|
28
|
-
} from "./chunk-
|
|
28
|
+
} from "./chunk-E7QXT7SX.js";
|
|
29
29
|
import {
|
|
30
30
|
costReceiptFromLlm,
|
|
31
31
|
costReceiptFromLlmError,
|
|
32
32
|
maximumChargeForLlmRequest,
|
|
33
33
|
stripFencedJson
|
|
34
|
-
} from "./chunk-
|
|
34
|
+
} from "./chunk-SFLLL76A.js";
|
|
35
35
|
import {
|
|
36
36
|
pairedBootstrap,
|
|
37
37
|
weightedComposite
|
|
38
|
-
} from "./chunk-
|
|
38
|
+
} from "./chunk-ZHTZ4EYI.js";
|
|
39
39
|
import {
|
|
40
40
|
CostLedger,
|
|
41
41
|
costForTokenPricing
|
|
42
|
-
} from "./chunk-
|
|
42
|
+
} from "./chunk-VCZ5FQYW.js";
|
|
43
43
|
import {
|
|
44
44
|
DEFAULT_REDACTION_RULES
|
|
45
45
|
} from "./chunk-GGE4NNQT.js";
|
|
@@ -48,36 +48,10 @@ import {
|
|
|
48
48
|
ValidationError
|
|
49
49
|
} from "./chunk-ONWEPEDO.js";
|
|
50
50
|
|
|
51
|
-
// src/tcloud-cost.ts
|
|
52
|
-
function maximumChargeForTCloudRequest(request, maximumAttempts) {
|
|
53
|
-
if (maximumAttempts === void 0) return void 0;
|
|
54
|
-
return maximumChargeForLlmRequest(request, { maxRetries: maximumAttempts });
|
|
55
|
-
}
|
|
56
|
-
function costReceiptFromTCloud(response, requestedModel) {
|
|
57
|
-
const usage = response.usage;
|
|
58
|
-
const inputTokens = tokenCount(usage?.prompt_tokens);
|
|
59
|
-
const outputTokens = tokenCount(usage?.completion_tokens);
|
|
60
|
-
const totalTokens = tokenCount(usage?.total_tokens);
|
|
61
|
-
const usageUnknown = inputTokens === void 0 || outputTokens === void 0 || totalTokens !== void 0 && totalTokens !== inputTokens + outputTokens;
|
|
62
|
-
return {
|
|
63
|
-
model: response.model || requestedModel,
|
|
64
|
-
inputTokens: inputTokens ?? 0,
|
|
65
|
-
outputTokens: outputTokens ?? 0,
|
|
66
|
-
costUnknown: usageUnknown,
|
|
67
|
-
usageUnknown
|
|
68
|
-
};
|
|
69
|
-
}
|
|
70
|
-
function tokenCount(value) {
|
|
71
|
-
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : void 0;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
51
|
// src/judges.ts
|
|
75
52
|
var JudgeParseError = class extends JudgeError {
|
|
76
|
-
/** Name of the judge whose response failed to parse. */
|
|
77
53
|
judgeName;
|
|
78
|
-
/** The raw (truncated) model response that failed to parse. */
|
|
79
54
|
raw;
|
|
80
|
-
/** Paid-call metadata remains available even when the verdict is unusable. */
|
|
81
55
|
llmCall;
|
|
82
56
|
constructor(judgeName, raw, options) {
|
|
83
57
|
super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
|
|
@@ -86,224 +60,6 @@ var JudgeParseError = class extends JudgeError {
|
|
|
86
60
|
this.llmCall = options?.llmCall;
|
|
87
61
|
}
|
|
88
62
|
};
|
|
89
|
-
function createDomainExpertJudge(domain) {
|
|
90
|
-
return async (tc, input) => {
|
|
91
|
-
const { scenario, turns } = input;
|
|
92
|
-
const conversation = turns.map(
|
|
93
|
-
(t, i) => `Turn ${i + 1}:
|
|
94
|
-
User: ${t.userMessage}
|
|
95
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
96
|
-
).join("\n\n---\n\n");
|
|
97
|
-
const resp = await runJudgeChat(tc, input, "domain_expert", {
|
|
98
|
-
model: "gpt-4o",
|
|
99
|
-
messages: [
|
|
100
|
-
{
|
|
101
|
-
role: "system",
|
|
102
|
-
content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
|
|
103
|
-
|
|
104
|
-
Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
|
|
105
|
-
|
|
106
|
-
Evaluate:
|
|
107
|
-
1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
|
|
108
|
-
2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
|
|
109
|
-
|
|
110
|
-
Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
|
|
111
|
-
},
|
|
112
|
-
{
|
|
113
|
-
role: "user",
|
|
114
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
115
|
-
Scenario: ${scenario.thesis}
|
|
116
|
-
|
|
117
|
-
${conversation}`
|
|
118
|
-
}
|
|
119
|
-
],
|
|
120
|
-
temperature: 0.1,
|
|
121
|
-
maxTokens: 800
|
|
122
|
-
});
|
|
123
|
-
return parseJudgeResponse("domain_expert", resp);
|
|
124
|
-
};
|
|
125
|
-
}
|
|
126
|
-
var codeExecutionJudge = async (tc, input) => {
|
|
127
|
-
const { scenario, artifacts } = input;
|
|
128
|
-
const codeBlocks = artifacts.codeBlocks;
|
|
129
|
-
if (codeBlocks.length === 0) {
|
|
130
|
-
return [
|
|
131
|
-
{
|
|
132
|
-
judgeName: "code_execution",
|
|
133
|
-
dimension: "code_execution",
|
|
134
|
-
score: 0,
|
|
135
|
-
reasoning: "No code blocks found in agent response."
|
|
136
|
-
}
|
|
137
|
-
];
|
|
138
|
-
}
|
|
139
|
-
const codeText = codeBlocks.map(
|
|
140
|
-
(b, i) => `Block ${i + 1} (${b.language}):
|
|
141
|
-
\`\`\`${b.language}
|
|
142
|
-
${b.code.slice(0, 3e3)}
|
|
143
|
-
\`\`\``
|
|
144
|
-
).join("\n\n");
|
|
145
|
-
const resp = await runJudgeChat(tc, input, "code_execution", {
|
|
146
|
-
model: "gpt-4o",
|
|
147
|
-
messages: [
|
|
148
|
-
{
|
|
149
|
-
role: "system",
|
|
150
|
-
content: `You are a principal software engineer reviewing code written by an AI agent.
|
|
151
|
-
|
|
152
|
-
Score STRICTLY:
|
|
153
|
-
1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
|
|
154
|
-
2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
|
|
155
|
-
3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
|
|
156
|
-
|
|
157
|
-
Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
|
|
158
|
-
},
|
|
159
|
-
{
|
|
160
|
-
role: "user",
|
|
161
|
-
content: `Task: ${scenario.thesis}
|
|
162
|
-
|
|
163
|
-
${codeText}`
|
|
164
|
-
}
|
|
165
|
-
],
|
|
166
|
-
temperature: 0.1,
|
|
167
|
-
maxTokens: 1e3
|
|
168
|
-
});
|
|
169
|
-
return parseJudgeResponse("code_execution", resp);
|
|
170
|
-
};
|
|
171
|
-
var coherenceJudge = async (tc, input) => {
|
|
172
|
-
const { scenario, turns } = input;
|
|
173
|
-
if (turns.length < 2) {
|
|
174
|
-
return [];
|
|
175
|
-
}
|
|
176
|
-
const conversation = turns.map(
|
|
177
|
-
(t, i) => `Turn ${i + 1}:
|
|
178
|
-
User: ${t.userMessage}
|
|
179
|
-
Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
|
|
180
|
-
).join("\n\n---\n\n");
|
|
181
|
-
const resp = await runJudgeChat(tc, input, "coherence", {
|
|
182
|
-
model: "gpt-4o",
|
|
183
|
-
messages: [
|
|
184
|
-
{
|
|
185
|
-
role: "system",
|
|
186
|
-
content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
|
|
187
|
-
|
|
188
|
-
Score STRICTLY:
|
|
189
|
-
1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
|
|
190
|
-
2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
|
|
191
|
-
3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
|
|
192
|
-
|
|
193
|
-
Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
|
|
194
|
-
},
|
|
195
|
-
{
|
|
196
|
-
role: "user",
|
|
197
|
-
content: `Scenario: ${scenario.thesis}
|
|
198
|
-
|
|
199
|
-
${conversation}`
|
|
200
|
-
}
|
|
201
|
-
],
|
|
202
|
-
temperature: 0.1,
|
|
203
|
-
maxTokens: 800
|
|
204
|
-
});
|
|
205
|
-
return parseJudgeResponse("coherence", resp);
|
|
206
|
-
};
|
|
207
|
-
var adversarialJudge = async (tc, input) => {
|
|
208
|
-
const { scenario, turns } = input;
|
|
209
|
-
const conversation = turns.map(
|
|
210
|
-
(t, i) => `Turn ${i + 1}:
|
|
211
|
-
User: ${t.userMessage}
|
|
212
|
-
Agent: ${t.agentResponse.slice(0, 1500)}`
|
|
213
|
-
).join("\n\n---\n\n");
|
|
214
|
-
const resp = await runJudgeChat(tc, input, "adversarial", {
|
|
215
|
-
model: "gpt-4o",
|
|
216
|
-
messages: [
|
|
217
|
-
{
|
|
218
|
-
role: "system",
|
|
219
|
-
content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
|
|
220
|
-
|
|
221
|
-
1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
|
|
222
|
-
2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
|
|
223
|
-
3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
|
|
224
|
-
|
|
225
|
-
Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
|
|
226
|
-
|
|
227
|
-
Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
|
|
228
|
-
},
|
|
229
|
-
{
|
|
230
|
-
role: "user",
|
|
231
|
-
content: `Persona: ${scenario.persona}
|
|
232
|
-
Scenario: ${scenario.thesis}
|
|
233
|
-
|
|
234
|
-
${conversation}`
|
|
235
|
-
}
|
|
236
|
-
],
|
|
237
|
-
temperature: 0.2,
|
|
238
|
-
maxTokens: 800
|
|
239
|
-
});
|
|
240
|
-
return parseJudgeResponse("adversarial", resp);
|
|
241
|
-
};
|
|
242
|
-
function createCustomJudge(name, systemPrompt, opts) {
|
|
243
|
-
return async (tc, input) => {
|
|
244
|
-
const { scenario, turns } = input;
|
|
245
|
-
const conversation = turns.map(
|
|
246
|
-
(t, i) => `Turn ${i + 1}:
|
|
247
|
-
User: ${t.userMessage}
|
|
248
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
249
|
-
).join("\n\n---\n\n");
|
|
250
|
-
const resp = await runJudgeChat(tc, input, name, {
|
|
251
|
-
model: opts?.model ?? "gpt-4o",
|
|
252
|
-
messages: [
|
|
253
|
-
{
|
|
254
|
-
role: "system",
|
|
255
|
-
content: systemPrompt
|
|
256
|
-
},
|
|
257
|
-
{
|
|
258
|
-
role: "user",
|
|
259
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
260
|
-
Scenario: ${scenario.thesis}
|
|
261
|
-
|
|
262
|
-
${conversation}`
|
|
263
|
-
}
|
|
264
|
-
],
|
|
265
|
-
temperature: opts?.temperature ?? 0.1,
|
|
266
|
-
maxTokens: opts?.maxTokens ?? 1e3
|
|
267
|
-
});
|
|
268
|
-
return parseJudgeResponse(name, resp);
|
|
269
|
-
};
|
|
270
|
-
}
|
|
271
|
-
function defaultJudges(domain) {
|
|
272
|
-
return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
|
|
273
|
-
}
|
|
274
|
-
function parseJudgeResponse(judgeName, resp) {
|
|
275
|
-
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
276
|
-
try {
|
|
277
|
-
let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
|
|
278
|
-
const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
|
|
279
|
-
if (arrayMatch) cleaned = arrayMatch[0];
|
|
280
|
-
const parsed = JSON.parse(cleaned);
|
|
281
|
-
return parsed.map((p) => ({
|
|
282
|
-
judgeName,
|
|
283
|
-
dimension: p.dimension,
|
|
284
|
-
score: Math.max(0, Math.min(10, p.score)),
|
|
285
|
-
reasoning: p.reasoning ?? "",
|
|
286
|
-
evidence: p.evidence
|
|
287
|
-
}));
|
|
288
|
-
} catch (err) {
|
|
289
|
-
throw new JudgeParseError(judgeName, content, { cause: err });
|
|
290
|
-
}
|
|
291
|
-
}
|
|
292
|
-
async function runJudgeChat(tc, input, judgeName, request) {
|
|
293
|
-
const paid = await (input.costLedger ?? new CostLedger()).runPaidCall({
|
|
294
|
-
channel: "judge",
|
|
295
|
-
phase: input.costPhase ?? "judge",
|
|
296
|
-
actor: `legacy-judge.${judgeName}`,
|
|
297
|
-
model: request.model,
|
|
298
|
-
maximumCharge: maximumChargeForTCloudRequest(request, input.tcloudMaximumAttempts),
|
|
299
|
-
tags: input.costTags,
|
|
300
|
-
signal: input.signal,
|
|
301
|
-
execute: () => tc.chat(request),
|
|
302
|
-
receipt: (response) => costReceiptFromTCloud(response, request.model)
|
|
303
|
-
});
|
|
304
|
-
if (!paid.succeeded) throw paid.error;
|
|
305
|
-
return paid.value;
|
|
306
|
-
}
|
|
307
63
|
|
|
308
64
|
// src/pareto.ts
|
|
309
65
|
function dominates(a, b, objectives) {
|
|
@@ -473,7 +229,7 @@ ${renderContract(dimensions, scale)}`;
|
|
|
473
229
|
actor: name,
|
|
474
230
|
model,
|
|
475
231
|
maximumCharge: opts.chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, {
|
|
476
|
-
|
|
232
|
+
maximumAttempts: opts.chat.maximumAttempts
|
|
477
233
|
}),
|
|
478
234
|
tags: { ...costTags, scenarioId: scenario.id },
|
|
479
235
|
signal,
|
|
@@ -1194,7 +950,7 @@ function renderPrBody(result, gate, diff) {
|
|
|
1194
950
|
lines.push(
|
|
1195
951
|
`**Cells**: executed ${result.aggregates.cellsExecuted}, cached ${result.aggregates.cellsCached}, skipped ${result.aggregates.cellsSkipped}, failed ${result.aggregates.cellsFailed}`
|
|
1196
952
|
);
|
|
1197
|
-
lines.push(`**Total spend**: $${result.aggregates.totalCostUsd.toFixed(2)}`);
|
|
953
|
+
lines.push(`**Total spend**: $${result.aggregates.cost.totalCostUsd.toFixed(2)}`);
|
|
1198
954
|
lines.push("");
|
|
1199
955
|
lines.push(`### Gate verdict: \`${gate.decision}\``);
|
|
1200
956
|
lines.push("");
|
|
@@ -2142,12 +1898,6 @@ function assertComparisonControls(opts, seed, resamples, confidence) {
|
|
|
2142
1898
|
}
|
|
2143
1899
|
}
|
|
2144
1900
|
function assertComparisonPartitions(opts) {
|
|
2145
|
-
const legacy = opts;
|
|
2146
|
-
if (legacy.holdoutScenarios !== void 0) {
|
|
2147
|
-
throw new Error(
|
|
2148
|
-
"compareOptimizationMethods: holdoutScenarios is ambiguous and no longer accepted. Provide disjoint trainScenarios, selectionScenarios, and testScenarios; selection may be reused adaptively, test must remain untouched."
|
|
2149
|
-
);
|
|
2150
|
-
}
|
|
2151
1901
|
const partitions = [
|
|
2152
1902
|
{ name: "trainScenarios", scenarios: opts.trainScenarios },
|
|
2153
1903
|
{ name: "selectionScenarios", scenarios: opts.selectionScenarios },
|
|
@@ -2329,7 +2079,7 @@ function assertComparisonCost(cost, label) {
|
|
|
2329
2079
|
|
|
2330
2080
|
// src/campaign/single-run-lock.ts
|
|
2331
2081
|
function assertAvailable(path) {
|
|
2332
|
-
const unavailable = probeAtomicFileLock({ lockPath: path
|
|
2082
|
+
const unavailable = probeAtomicFileLock({ lockPath: path });
|
|
2333
2083
|
if (unavailable) throw unavailableError(path, unavailable);
|
|
2334
2084
|
}
|
|
2335
2085
|
function unavailableError(path, unavailable) {
|
|
@@ -2349,8 +2099,7 @@ function acquireSingleRunLock(opts) {
|
|
|
2349
2099
|
for (const path of opts.alsoCheck ?? []) assertAvailable(path);
|
|
2350
2100
|
const acquisition = tryAcquireAtomicFileLock({
|
|
2351
2101
|
lockPath: opts.lockPath,
|
|
2352
|
-
pid
|
|
2353
|
-
acceptLegacyPid: true
|
|
2102
|
+
pid
|
|
2354
2103
|
});
|
|
2355
2104
|
if (!acquisition.acquired) throw unavailableError(opts.lockPath, acquisition);
|
|
2356
2105
|
const release = () => {
|
|
@@ -6349,7 +6098,7 @@ async function runImprovementLoop(opts) {
|
|
|
6349
6098
|
const dispatchTimeoutMs = opts.dispatchTimeoutMs ?? DEFAULT_DISPATCH_TIMEOUT_MS;
|
|
6350
6099
|
const optimization = await runOptimization({ ...opts, dispatchTimeoutMs, costLedger });
|
|
6351
6100
|
const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface);
|
|
6352
|
-
const { runCampaign: runCampaign2 } = await import("./run-campaign-
|
|
6101
|
+
const { runCampaign: runCampaign2 } = await import("./run-campaign-OJJ7CZF4.js");
|
|
6353
6102
|
const holdoutDeferred = (opts.holdout ?? "measured") === "deferred";
|
|
6354
6103
|
const baselineOnHoldout = holdoutDeferred ? await runCampaign2({
|
|
6355
6104
|
...opts,
|
|
@@ -6464,8 +6213,8 @@ async function runImprovementLoop(opts) {
|
|
|
6464
6213
|
neutralizedJudgeScores,
|
|
6465
6214
|
scenarios: opts.holdoutScenarios,
|
|
6466
6215
|
cost: {
|
|
6467
|
-
candidate: winnerOnHoldout.aggregates.totalCostUsd,
|
|
6468
|
-
baseline: baselineOnHoldout.aggregates.totalCostUsd
|
|
6216
|
+
candidate: winnerOnHoldout.aggregates.cost.totalCostUsd,
|
|
6217
|
+
baseline: baselineOnHoldout.aggregates.cost.totalCostUsd
|
|
6469
6218
|
},
|
|
6470
6219
|
costLedger,
|
|
6471
6220
|
costPhase: "promotion.gate",
|
|
@@ -7094,7 +6843,7 @@ function snapshotFromHoldout(index, surfaceHash2, surface, campaign) {
|
|
|
7094
6843
|
surface,
|
|
7095
6844
|
cells,
|
|
7096
6845
|
compositeMean: campaignMeanCompositeOrNull(campaign),
|
|
7097
|
-
costUsd: campaign.aggregates.totalCostUsd,
|
|
6846
|
+
costUsd: campaign.aggregates.cost.totalCostUsd,
|
|
7098
6847
|
durationMs: campaign.durationMs
|
|
7099
6848
|
};
|
|
7100
6849
|
}
|
|
@@ -7553,15 +7302,7 @@ function skillOptOptimizationMethod(config) {
|
|
|
7553
7302
|
}
|
|
7554
7303
|
|
|
7555
7304
|
export {
|
|
7556
|
-
maximumChargeForTCloudRequest,
|
|
7557
|
-
costReceiptFromTCloud,
|
|
7558
7305
|
JudgeParseError,
|
|
7559
|
-
createDomainExpertJudge,
|
|
7560
|
-
codeExecutionJudge,
|
|
7561
|
-
coherenceJudge,
|
|
7562
|
-
adversarialJudge,
|
|
7563
|
-
createCustomJudge,
|
|
7564
|
-
defaultJudges,
|
|
7565
7306
|
recoverTruncatedJson,
|
|
7566
7307
|
dominates,
|
|
7567
7308
|
paretoFrontier,
|
|
@@ -7630,4 +7371,4 @@ export {
|
|
|
7630
7371
|
emitLoopProvenance,
|
|
7631
7372
|
skillOptOptimizationMethod
|
|
7632
7373
|
};
|
|
7633
|
-
//# sourceMappingURL=chunk-
|
|
7374
|
+
//# sourceMappingURL=chunk-2QU3YOPR.js.map
|