anpord 0.1.11 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -78
- package/dist/bin.cjs +211 -204
- package/dist/bin.mjs +211 -204
- package/dist/cli-runtime.cjs +2 -2
- package/dist/cli-runtime.d.cts +1 -1
- package/dist/cli-runtime.d.mts +1 -1
- package/dist/cli-runtime.mjs +2 -2
- package/dist/client-BfdjFtlV.mjs +305 -0
- package/dist/{client-C9m0uLCz.d.cts → client-BgRs8JvW.d.mts} +2150 -174
- package/dist/{client-TFCLuMv8.d.mts → client-BirFO3jz.d.cts} +2150 -174
- package/dist/client-C-P2I00z.cjs +346 -0
- package/dist/{compiler-DXlpU_In.mjs → compiler-D96wI7UP.mjs} +108 -75
- package/dist/{compiler-Ap8DM-JU.cjs → compiler-uwb1pkdK.cjs} +116 -83
- package/dist/config.cjs +1 -1
- package/dist/config.d.cts +1 -1
- package/dist/config.d.mts +1 -1
- package/dist/config.mjs +1 -1
- package/dist/{errors-C9bQUcA3.mjs → errors-B0YknR5V.mjs} +3 -19
- package/dist/{errors-Boum34f5.cjs → errors-BX1wry8K.cjs} +3 -19
- package/dist/{errors-BK3f7Rdr.d.cts → errors-aGrd21jy.d.cts} +0 -3
- package/dist/{errors-BK3f7Rdr.d.mts → errors-aGrd21jy.d.mts} +0 -3
- package/dist/eval-judges-BkmiVLLq.cjs +59 -0
- package/dist/eval-judges-CwTvTq7N.mjs +42 -0
- package/dist/eval-judges-DPPUftbh.d.cts +48 -0
- package/dist/eval-judges-DPPUftbh.d.mts +48 -0
- package/dist/eval-validations-BRdoZrTQ.mjs +113 -0
- package/dist/eval-validations-_MlnQ_uv.cjs +148 -0
- package/dist/eval.cjs +1 -1
- package/dist/eval.d.cts +60 -6
- package/dist/eval.d.mts +60 -6
- package/dist/eval.mjs +1 -1
- package/dist/evals-B6aT3xgH.d.cts +1364 -0
- package/dist/evals-B6aT3xgH.d.mts +1364 -0
- package/dist/evals-Dcc_x6jW.mjs +582 -0
- package/dist/evals-DouMsrsM.cjs +737 -0
- package/dist/index.cjs +23 -99
- package/dist/index.d.cts +138 -34
- package/dist/index.d.mts +138 -34
- package/dist/index.mjs +23 -99
- package/dist/source.d.cts +1 -1
- package/dist/source.d.mts +1 -1
- package/dist/{types-oaMF6-E1.d.mts → types-Codjaq-o.d.mts} +146 -45
- package/dist/{types-H9L450Iu.d.cts → types-ISsteq95.d.cts} +146 -45
- package/dist/validator-runtime.cjs +238 -0
- package/dist/validator-runtime.d.cts +9 -0
- package/dist/validator-runtime.d.mts +9 -0
- package/dist/validator-runtime.mjs +237 -0
- package/dist/validators.cjs +10 -0
- package/dist/validators.d.cts +8 -0
- package/dist/validators.d.mts +8 -0
- package/dist/validators.mjs +9 -0
- package/package.json +11 -1
- package/dist/client-5K3dMI5Q.mjs +0 -849
- package/dist/client-Ci0woWLW.cjs +0 -890
- package/dist/evals-BEgEmfiO.d.cts +0 -603
- package/dist/evals-BEgEmfiO.d.mts +0 -603
- package/dist/harness-profile-BMjsN460.mjs +0 -43
- package/dist/harness-profile-CDQpPLPd.cjs +0 -78
package/dist/bin.cjs
CHANGED
|
@@ -1,23 +1,20 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
const require_client = require("./client-
|
|
3
|
-
const
|
|
2
|
+
const require_client = require("./client-C-P2I00z.cjs");
|
|
3
|
+
const require_evals = require("./evals-DouMsrsM.cjs");
|
|
4
|
+
const require_errors = require("./errors-BX1wry8K.cjs");
|
|
4
5
|
const require_config = require("./config.cjs");
|
|
5
|
-
const require_compiler = require("./compiler-
|
|
6
|
+
const require_compiler = require("./compiler-uwb1pkdK.cjs");
|
|
6
7
|
let _effect_cli = require("@effect/cli");
|
|
7
8
|
let _effect_platform = require("@effect/platform");
|
|
8
9
|
let _effect_platform_node = require("@effect/platform-node");
|
|
9
10
|
let effect = require("effect");
|
|
10
11
|
let yaml = require("yaml");
|
|
11
12
|
//#region package.json
|
|
12
|
-
var version$1 = "0.1.
|
|
13
|
+
var version$1 = "0.1.13";
|
|
13
14
|
//#endregion
|
|
14
15
|
//#region ../template/src/extract.ts
|
|
15
|
-
/**
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
* Escaped braces are read here too, so a name the renderer will treat as
|
|
19
|
-
* literal text is never reported as a variable the editor should draw.
|
|
20
|
-
*/
|
|
16
|
+
/** Reads escapes too, so text the renderer leaves literal is never reported as
|
|
17
|
+
* a variable. */
|
|
21
18
|
function extractVariables(template) {
|
|
22
19
|
const names = [];
|
|
23
20
|
for (const [, open, close, name] of template.matchAll(require_errors.tokenMatcher())) if (open === void 0 && close === void 0 && name !== void 0) names.push(name);
|
|
@@ -25,13 +22,6 @@ function extractVariables(template) {
|
|
|
25
22
|
}
|
|
26
23
|
//#endregion
|
|
27
24
|
//#region src/cli/declarations.ts
|
|
28
|
-
/**
|
|
29
|
-
* The generated file has to open by importing the module it augments.
|
|
30
|
-
*
|
|
31
|
-
* Without it `declare module` shadows the module rather than adding to it, and
|
|
32
|
-
* every import of the SDK stops resolving with an error naming the missing
|
|
33
|
-
* export rather than the file that hid it.
|
|
34
|
-
*/
|
|
35
25
|
const PREAMBLE = [
|
|
36
26
|
"// Generated by `anpord generate`. Do not edit.",
|
|
37
27
|
"// Run it again after changing which variables a prompt uses.",
|
|
@@ -40,7 +30,6 @@ const PREAMBLE = [
|
|
|
40
30
|
"declare module \"anpord\" {",
|
|
41
31
|
" interface AnpordPromptVariables {"
|
|
42
32
|
].join("\n");
|
|
43
|
-
/** Quoted, because an id may carry characters an identifier cannot. */
|
|
44
33
|
const entry = (id, names) => {
|
|
45
34
|
return ` "${id}": ${names.length === 0 ? "Record<string, never>" : `{ ${names.map((name) => `"${name}": string`).join("; ")} }`};`;
|
|
46
35
|
};
|
|
@@ -70,10 +59,11 @@ const evalFilesIn = (directory) => effect.Effect.gen(function* () {
|
|
|
70
59
|
}).pipe(effect.Effect.withSpan("Cli.evalFilesIn"));
|
|
71
60
|
//#endregion
|
|
72
61
|
//#region src/cli/eval-gate.ts
|
|
62
|
+
const EvalGate = effect.Schema.Literal("strict", "never", "regressed", "unscored");
|
|
73
63
|
const regressions = (run) => run.cells.filter((cell) => cell.comparison?.verdict === "regressed");
|
|
74
64
|
const unscored = (run) => run.cells.filter((cell) => (cell.distribution?.scored ?? 0) === 0);
|
|
75
65
|
const rate = (value) => `${Math.round(value * 100) / 100}`;
|
|
76
|
-
const versionClause
|
|
66
|
+
const versionClause = (run, cell, found) => found.baselineHarnessVersion === found.candidateHarnessVersion ? "" : `${run.tasks[cell.taskIndex]?.harness ?? "harness"} ${found.baselineHarnessVersion} → ${found.candidateHarnessVersion}, `;
|
|
77
67
|
const profileClause = (run, cell, found) => {
|
|
78
68
|
const { baselineProfileVersion, candidateProfileVersion } = found;
|
|
79
69
|
if (baselineProfileVersion === null || candidateProfileVersion === null || baselineProfileVersion === candidateProfileVersion) return "";
|
|
@@ -82,10 +72,20 @@ const profileClause = (run, cell, found) => {
|
|
|
82
72
|
const regressionSentence = (run, cell) => {
|
|
83
73
|
const found = cell.comparison;
|
|
84
74
|
if (found === null) return `${cell.caseName} regressed against its baseline.`;
|
|
85
|
-
return `${cell.caseName} regressed against its baseline: ${versionClause
|
|
75
|
+
return `${cell.caseName} regressed against its baseline: ${versionClause(run, cell, found)}${profileClause(run, cell, found)}pass rate ${rate(found.baselinePassRate)} → ${rate(found.candidatePassRate)}.`;
|
|
86
76
|
};
|
|
87
|
-
const problemsWith = (run, failOn) => {
|
|
77
|
+
const problemsWith = (run, failOn, expected) => {
|
|
88
78
|
if (run.status === "failed") return [run.failure ?? "The run failed."];
|
|
79
|
+
if (run.status !== "finished") return ["The run has not finished."];
|
|
80
|
+
if (run.cells.length === 0) return ["The run recorded no cells."];
|
|
81
|
+
if (failOn === "strict") {
|
|
82
|
+
if (expected && run.cells.length !== expected.cells) return [`Expected ${expected.cells} cells, received ${run.cells.length}.`];
|
|
83
|
+
return run.cells.flatMap((cell) => {
|
|
84
|
+
if (cell.status !== "finished" || cell.trials.length === 0) return [`${cell.caseName} has no complete trial results.`];
|
|
85
|
+
if (expected && cell.trials.length !== expected.trials) return [`${cell.caseName}: expected ${expected.trials} trials, received ${cell.trials.length}.`];
|
|
86
|
+
return cell.trials.flatMap((trial) => trial.status === "passed" && trial.passed ? [] : [`${cell.caseName}, trial ${trial.ordinal}: ${trial.status}.`]);
|
|
87
|
+
});
|
|
88
|
+
}
|
|
89
89
|
if (failOn === "never") return [];
|
|
90
90
|
const found = regressions(run).map((cell) => regressionSentence(run, cell));
|
|
91
91
|
return failOn === "unscored" ? [...found, ...unscored(run).map((cell) => `${cell.caseName} produced no scored trials.`)] : found;
|
|
@@ -108,8 +108,6 @@ const promptContent = (prompt) => effect.Effect.sync(() => {
|
|
|
108
108
|
process.stdout.write(prompt.content);
|
|
109
109
|
if (!prompt.content.endsWith("\n")) process.stdout.write("\n");
|
|
110
110
|
});
|
|
111
|
-
/** A result the caller may pipe into another tool, so it belongs on stdout
|
|
112
|
-
* alongside the prompt content rather than beside the status messages. */
|
|
113
111
|
const row = (line) => effect.Effect.sync(() => {
|
|
114
112
|
process.stdout.write(`${line}\n`);
|
|
115
113
|
});
|
|
@@ -129,43 +127,43 @@ const DONE = "●";
|
|
|
129
127
|
const RUNNING = "◐";
|
|
130
128
|
const FILLED = "▰";
|
|
131
129
|
const HOLLOW = "▱";
|
|
132
|
-
const PERCENT
|
|
130
|
+
const PERCENT = 100;
|
|
133
131
|
const SECONDS = 1e3;
|
|
134
132
|
const MINUTE = 60;
|
|
135
133
|
const paint = (colour, text) => `${colour}${text}${RESET}`;
|
|
136
|
-
const
|
|
134
|
+
const formatElapsed = (ms) => {
|
|
137
135
|
const total = Math.floor(ms / SECONDS);
|
|
138
136
|
const minutes = Math.floor(total / MINUTE);
|
|
139
137
|
return minutes === 0 ? `${total}s` : `${minutes}m${String(total % MINUTE).padStart(2, "0")}s`;
|
|
140
138
|
};
|
|
141
|
-
const
|
|
139
|
+
const formatStatus = (cell) => {
|
|
142
140
|
if (cell.status === "finished") return paint(GREEN, DONE);
|
|
143
141
|
return cell.status === "failed" ? paint(RED, DONE) : paint(YELLOW, RUNNING);
|
|
144
142
|
};
|
|
145
|
-
const
|
|
143
|
+
const formatTrialProgress = (cell, trials) => {
|
|
146
144
|
const settled = cell.trials.filter((trial) => trial.status !== "queued" && trial.status !== "running").length;
|
|
147
145
|
return `${FILLED.repeat(settled)}${paint(DIM, HOLLOW.repeat(Math.max(0, trials - settled)))}`;
|
|
148
146
|
};
|
|
149
|
-
const
|
|
147
|
+
const formatPassRate = (cell) => {
|
|
150
148
|
const rate = cell.distribution?.passRate;
|
|
151
149
|
if (rate === void 0 || cell.distribution?.scored === 0) return paint(DIM, "—");
|
|
152
|
-
const shown = `${Math.round(rate * PERCENT
|
|
150
|
+
const shown = `${Math.round(rate * PERCENT)}%`;
|
|
153
151
|
return paint(rate === 1 ? GREEN : RED, shown);
|
|
154
152
|
};
|
|
155
|
-
const
|
|
153
|
+
const formatVariant = (run, cell) => {
|
|
156
154
|
const task = run.tasks[cell.taskIndex];
|
|
157
155
|
return task === void 0 ? "?" : `${task.harness}/${task.model}`;
|
|
158
156
|
};
|
|
159
|
-
const widest = (run) => run.cells.reduce((width, cell) => Math.max(width,
|
|
160
|
-
const
|
|
157
|
+
const widest = (run) => run.cells.reduce((width, cell) => Math.max(width, formatVariant(run, cell).length), 0);
|
|
158
|
+
const formatGrid = (run, trials, elapsedMs) => {
|
|
161
159
|
const width = widest(run);
|
|
162
160
|
const lines = [];
|
|
163
161
|
for (const caseName of run.cases) {
|
|
164
162
|
lines.push(` ${BOLD}${caseName}${RESET}`);
|
|
165
|
-
for (const cell of run.cells.filter((one) => one.caseName === caseName)) lines.push(` ${
|
|
163
|
+
for (const cell of run.cells.filter((one) => one.caseName === caseName)) lines.push(` ${formatStatus(cell)} ${formatVariant(run, cell).padEnd(width)} ${formatTrialProgress(cell, trials)} ${formatPassRate(cell)}`);
|
|
166
164
|
lines.push("");
|
|
167
165
|
}
|
|
168
|
-
lines.push(paint(DIM, ` ${
|
|
166
|
+
lines.push(paint(DIM, ` ${formatElapsed(elapsedMs)} elapsed`));
|
|
169
167
|
return lines;
|
|
170
168
|
};
|
|
171
169
|
const up = (rows) => `[${rows}A[0J`;
|
|
@@ -174,12 +172,12 @@ const liveGrid = (trials, interactive) => effect.Effect.gen(function* () {
|
|
|
174
172
|
return (run, elapsedMs) => effect.Effect.gen(function* () {
|
|
175
173
|
if (!interactive) return;
|
|
176
174
|
const rows = yield* effect.Ref.getAndSet(drawn, 0);
|
|
177
|
-
const lines =
|
|
175
|
+
const lines = formatGrid(run, trials, elapsedMs);
|
|
178
176
|
yield* note(`${rows === 0 ? "" : up(rows)}${lines.join("\n")}`);
|
|
179
177
|
yield* effect.Ref.set(drawn, lines.length);
|
|
180
178
|
});
|
|
181
179
|
});
|
|
182
|
-
const
|
|
180
|
+
const formatGridSummary = (run, trials, drawn) => drawn ? "" : formatGrid(run, trials, 0).join("\n");
|
|
183
181
|
//#endregion
|
|
184
182
|
//#region src/imports/evals-json-errors.ts
|
|
185
183
|
var CaseFileUnreadable = class extends effect.Data.TaggedError("CaseFileUnreadable") {
|
|
@@ -192,8 +190,6 @@ var CaseFileNotJson = class extends effect.Data.TaggedError("CaseFileNotJson") {
|
|
|
192
190
|
return `${this.path} is not JSON: ${this.reason}`;
|
|
193
191
|
}
|
|
194
192
|
};
|
|
195
|
-
/** Names the case rather than the byte offset, because the author fixing it
|
|
196
|
-
* reads the file by case, not by position. */
|
|
197
193
|
var CaseFileNotEvalsJson = class extends effect.Data.TaggedError("CaseFileNotEvalsJson") {
|
|
198
194
|
get message() {
|
|
199
195
|
return `${this.path} is not an evals-json file: ${this.reason}`;
|
|
@@ -206,29 +202,17 @@ var CaseFileEmpty = class extends effect.Data.TaggedError("CaseFileEmpty") {
|
|
|
206
202
|
};
|
|
207
203
|
//#endregion
|
|
208
204
|
//#region src/imports/typescript-literal.ts
|
|
209
|
-
/** A backslash-escaped double-quoted literal. Backslash first, so the escapes
|
|
210
|
-
* added after it are not themselves escaped. */
|
|
211
205
|
const quoted = (value) => `"${value.replaceAll("\\", "\\\\").replaceAll("\"", "\\\"").replaceAll("\n", "\\n").replaceAll("\r", "\\r").replaceAll(" ", "\\t")}"`;
|
|
212
|
-
/** A template literal, which keeps a multi-line prompt readable in the
|
|
213
|
-
* generated file. A backtick ends the literal and `${` opens a substitution,
|
|
214
|
-
* so both are escaped; a lone `$` is left alone because only the pair means
|
|
215
|
-
* anything. A trailing backslash would escape the closing backtick. */
|
|
216
206
|
const templated = (value) => `\`${value.replaceAll("\\", "\\\\").replaceAll("`", "\\`").replaceAll("${", "\\${")}\``;
|
|
217
|
-
/** A comment cannot carry the sequence that closes it, and a newline would end
|
|
218
|
-
* a line comment and let the rest of the text run as code. */
|
|
219
207
|
const commentSafe = (value) => value.replaceAll("*/", "*\\/").replaceAll(/\r?\n/g, " ");
|
|
220
208
|
//#endregion
|
|
221
209
|
//#region src/imports/unwritten-check.ts
|
|
222
210
|
const PLACEHOLDER = "unwritten";
|
|
223
|
-
/** The author's own words, kept whole, because they are the specification for
|
|
224
|
-
* the check that replaces the line beneath them. */
|
|
225
211
|
const proseLine = (text) => [
|
|
226
212
|
" /* Write this check, then delete the line under it: */",
|
|
227
213
|
` /* ${commentSafe(text)} */`,
|
|
228
214
|
` ${PLACEHOLDER}(${quoted(text)}),`
|
|
229
215
|
].join("\n");
|
|
230
|
-
/** Local rather than imported, so the generated file carries its own proof
|
|
231
|
-
* that an unconverted assertion fails. Deleting the last call deletes it. */
|
|
232
216
|
const placeholderBlock = [
|
|
233
217
|
"/* A check nobody has written yet. It is false, so the case stays red until",
|
|
234
218
|
" the sentence above it becomes a real check. The argument is that",
|
|
@@ -259,8 +243,6 @@ const nameOf = (subject) => `${slug$1(subject.name ?? "case", "case")}-${subject
|
|
|
259
243
|
const scorerLine = (assertion) => [` /* ${commentSafe(assertion.text)} */`, ` ${SCORER_OF[assertion.kind]}(answer, [${assertion.needles.map(quoted).join(", ")}]),`].join("\n");
|
|
260
244
|
const scorerLines = (subject) => subject.assertions.map((assertion) => isStructured(assertion) ? scorerLine(assertion) : proseLine(assertion)).join("\n");
|
|
261
245
|
const expectationComment = (subject) => subject.expected_output === void 0 ? [] : [` /* The file's expected output: ${commentSafe(subject.expected_output)} */`];
|
|
262
|
-
/** The JSON names fixture directories by convention and carries none of their
|
|
263
|
-
* contents, so the author supplies the files the name stood for. */
|
|
264
246
|
const sourceComment = (subject) => subject.files.length === 0 ? " /* This case named no fixture directory. Add the files it starts from. */" : ` /* This case named ${subject.files.map(commentSafe).join(", ")}. The JSON carries the directory names, not their contents, so add the files here. */`;
|
|
265
247
|
const caseBlock$1 = (subject) => [
|
|
266
248
|
" {",
|
|
@@ -312,23 +294,19 @@ const renderEvalSuite = (file) => {
|
|
|
312
294
|
" ],",
|
|
313
295
|
" tasks: [",
|
|
314
296
|
" /* Name the harness, model and sandbox this suite runs on. */",
|
|
315
|
-
" { harness: \"codex\", model: \"gpt-5.6-sol\"
|
|
297
|
+
" { harness: \"codex\", model: \"gpt-5.6-sol\" },",
|
|
316
298
|
" ],",
|
|
317
299
|
"});"
|
|
318
300
|
].join("\n")}\n`;
|
|
319
301
|
};
|
|
320
302
|
//#endregion
|
|
321
303
|
//#region src/imports/evals-json-schema.ts
|
|
322
|
-
/** The three checks that convert mechanically, because each is a needle list
|
|
323
|
-
* and a rule over it rather than a description of intent. */
|
|
324
304
|
const AssertionKind = effect.Schema.Literal("content_contains_any", "content_contains_all", "content_contains_none");
|
|
325
305
|
const StructuredAssertion = effect.Schema.Struct({
|
|
326
306
|
kind: AssertionKind,
|
|
327
307
|
needles: effect.Schema.Array(effect.Schema.String),
|
|
328
308
|
text: effect.Schema.String
|
|
329
309
|
});
|
|
330
|
-
/** Both dialects occur, sometimes in sibling files: a bare string is prose one
|
|
331
|
-
* person wrote for another, an object is a mechanical check. */
|
|
332
310
|
const Assertion = effect.Schema.Union(effect.Schema.String, StructuredAssertion);
|
|
333
311
|
const EvalsJsonCase = effect.Schema.Struct({
|
|
334
312
|
assertions: effect.Schema.Array(Assertion),
|
|
@@ -356,8 +334,6 @@ const idAt = (parsed, index) => {
|
|
|
356
334
|
const found = parsed?.evals?.[index];
|
|
357
335
|
return found?.id === void 0 ? `${index}` : String(found.id);
|
|
358
336
|
};
|
|
359
|
-
/** The case as its author sees it. A path into the decoded value tells them
|
|
360
|
-
* nothing; the id is what they search the file for. */
|
|
361
337
|
const located = (parsed, path) => {
|
|
362
338
|
const [head, index, ...rest] = path;
|
|
363
339
|
if (head !== "evals" || typeof index !== "number") return path.map(String).join(".");
|
|
@@ -393,8 +369,6 @@ var CaseFileNotYaml = class extends effect.Data.TaggedError("CaseFileNotYaml") {
|
|
|
393
369
|
return `${this.path} is not YAML: ${this.reason}`;
|
|
394
370
|
}
|
|
395
371
|
};
|
|
396
|
-
/** Names the file rather than a path into the decoded value, because one case
|
|
397
|
-
* is one file and the file name is what the author searches for. */
|
|
398
372
|
var CaseFileNotYamlCase = class extends effect.Data.TaggedError("CaseFileNotYamlCase") {
|
|
399
373
|
get message() {
|
|
400
374
|
return `${this.path} is not a yaml case: ${this.reason}`;
|
|
@@ -411,20 +385,12 @@ const slug = (value, fallback) => {
|
|
|
411
385
|
const cleaned = value.toLowerCase().replaceAll(/[^a-z0-9]+/g, "-").replaceAll(/^-+|-+$/g, "");
|
|
412
386
|
return cleaned === "" ? fallback : cleaned;
|
|
413
387
|
};
|
|
414
|
-
/** Every line is prose a person wrote for a model judge, so none of it
|
|
415
|
-
* converts and the whole list is what a human still owes. A case with no
|
|
416
|
-
* lines owes one too: something has to say what a good answer is. */
|
|
417
388
|
const tallyOf = (files) => ({
|
|
418
389
|
cases: files.length,
|
|
419
390
|
converted: 0,
|
|
420
391
|
needsAuthor: files.reduce((total, file) => total + Math.max(file.subject.judge_context.length, 1), 0)
|
|
421
392
|
});
|
|
422
|
-
/** Anpord has no step budget, so the number is kept as a note rather than
|
|
423
|
-
* dropped: the suite it came from ran under it, and a case that needed ten
|
|
424
|
-
* steps is a different case from one that needed a hundred. */
|
|
425
393
|
const budgetComment = (subject) => ` /* The file allowed ${subject.max_steps} steps. Anpord does not cap steps, so this is a note, not a limit. */`;
|
|
426
|
-
/** A file with no judge context says nothing about what a good answer is, so
|
|
427
|
-
* it is owed a check like any other rather than passing by default. */
|
|
428
394
|
const UNJUDGED = "This case named no judge context. Write what a good answer is.";
|
|
429
395
|
const judgeLines = (subject) => (subject.judge_context.length === 0 ? [UNJUDGED] : subject.judge_context).map((line) => proseLine(line)).join("\n");
|
|
430
396
|
const caseBlock = (file) => [
|
|
@@ -446,8 +412,6 @@ const caseBlock = (file) => [
|
|
|
446
412
|
" },",
|
|
447
413
|
" },"
|
|
448
414
|
].join("\n");
|
|
449
|
-
/** A directory has no name of its own in the files, so a suite built from
|
|
450
|
-
* several is named generically and the author renames it. */
|
|
451
415
|
const suiteName = (files) => files.length === 1 ? slug(files[0]?.subject.name ?? "", "imported-suite") : "imported-suite";
|
|
452
416
|
const renderYamlSuite = (files) => `${[
|
|
453
417
|
"import { defineEval, files } from \"anpord\";",
|
|
@@ -463,18 +427,13 @@ const renderYamlSuite = (files) => `${[
|
|
|
463
427
|
" ],",
|
|
464
428
|
" tasks: [",
|
|
465
429
|
" /* Name the harness, model and sandbox this suite runs on. */",
|
|
466
|
-
" { harness: \"codex\", model: \"gpt-5.6-sol\"
|
|
430
|
+
" { harness: \"codex\", model: \"gpt-5.6-sol\" },",
|
|
467
431
|
" ],",
|
|
468
432
|
"});"
|
|
469
433
|
].join("\n")}\n`;
|
|
470
434
|
//#endregion
|
|
471
435
|
//#region src/imports/yaml-cases-schema.ts
|
|
472
|
-
/** The runner's own default when a file omits the field, kept here so an
|
|
473
|
-
* imported case carries the budget it actually ran under. */
|
|
474
436
|
const DEFAULT_MAX_STEPS = 15;
|
|
475
|
-
/** A missing `judge_context` is a case whose author wrote nothing about how to
|
|
476
|
-
* judge it, which the runner reads as one generic line. It is not an error, so
|
|
477
|
-
* it decodes to an empty list and the generated case says so. */
|
|
478
437
|
const YamlCase = effect.Schema.Struct({
|
|
479
438
|
judge_context: effect.Schema.optionalWith(effect.Schema.Array(effect.Schema.String), { default: () => [] }),
|
|
480
439
|
max_steps: effect.Schema.optionalWith(effect.Schema.Int, { default: () => DEFAULT_MAX_STEPS }),
|
|
@@ -484,10 +443,6 @@ const YamlCase = effect.Schema.Struct({
|
|
|
484
443
|
const decodeYamlCase = effect.Schema.decodeUnknown(YamlCase, { errors: "all" });
|
|
485
444
|
//#endregion
|
|
486
445
|
//#region src/imports/yaml-document.ts
|
|
487
|
-
/** A YAML document can parse into a value while still carrying errors the
|
|
488
|
-
* author needs to see, so the errors are read rather than the throw relied on.
|
|
489
|
-
* The value is returned as `unknown` because the schema, not the parser,
|
|
490
|
-
* decides what shape it has. */
|
|
491
446
|
const parseYamlDocument = (body) => effect.Effect.try({
|
|
492
447
|
catch: (cause) => cause instanceof Error ? cause.message : String(cause),
|
|
493
448
|
try: () => {
|
|
@@ -519,8 +474,6 @@ const readCase = (path) => effect.Effect.gen(function* () {
|
|
|
519
474
|
subject: yield* decode(path, parsed)
|
|
520
475
|
};
|
|
521
476
|
});
|
|
522
|
-
/** Sorted, so the same directory always produces the same suite and a diff of
|
|
523
|
-
* two imports shows what changed rather than what moved. */
|
|
524
477
|
const caseFilesIn = (directory) => effect.Effect.gen(function* () {
|
|
525
478
|
return (yield* (yield* _effect_platform.FileSystem.FileSystem).readDirectory(directory).pipe(effect.Effect.mapError((cause) => new CaseDirectoryUnreadable({
|
|
526
479
|
cause,
|
|
@@ -547,20 +500,13 @@ const importYamlCases = (path) => effect.Effect.gen(function* () {
|
|
|
547
500
|
}).pipe(effect.Effect.withSpan("Imports.yamlCases"));
|
|
548
501
|
//#endregion
|
|
549
502
|
//#region src/cli/import-formats.ts
|
|
550
|
-
/** A format is an entry here, so adding one adds an entry rather than editing
|
|
551
|
-
* the command. Each name carries the importer it names, so the parsed option
|
|
552
|
-
* is the importer itself and no lookup can miss. */
|
|
553
503
|
const FORMATS = [["evals-json", importEvalsJson], ["yaml", importYamlCases]];
|
|
554
504
|
//#endregion
|
|
555
505
|
//#region src/cli/eval-import.ts
|
|
556
|
-
/** A path rather than a file, because a format may keep one case per file and
|
|
557
|
-
* import a directory of them as one suite. */
|
|
558
506
|
const caseFile = _effect_cli.Args.path({ name: "path" }).pipe(_effect_cli.Args.withDescription("The case file, or a directory of them, to read"));
|
|
559
507
|
const format = _effect_cli.Options.choiceWithValue("format", FORMATS).pipe(_effect_cli.Options.withDescription("The case file's format"));
|
|
560
508
|
const out$1 = _effect_cli.Options.file("out").pipe(_effect_cli.Options.withDescription("Where to write the suite; it goes to stdout when omitted"), _effect_cli.Options.optional);
|
|
561
509
|
const plural = (count, one) => `${count} ${count === 1 ? one : `${one}s`}`;
|
|
562
|
-
/** The count of unwritten checks leads, because a suite that reports a pass
|
|
563
|
-
* for a check nobody wrote is the failure this product exists to prevent. */
|
|
564
510
|
const summaryOf = (tally) => {
|
|
565
511
|
const read = `Read ${plural(tally.cases, "case")}, converted ${plural(tally.converted, "assertion")}.`;
|
|
566
512
|
return tally.needsAuthor === 0 ? read : `${read} ${plural(tally.needsAuthor, "assertion")} could not be converted and ${tally.needsAuthor === 1 ? "needs" : "need"} a human: each is written as prose, and the suite fails until you write the check it describes.`;
|
|
@@ -581,12 +527,102 @@ const importEval = _effect_cli.Command.make("import", {
|
|
|
581
527
|
return yield* note(summaryOf(suite.tally));
|
|
582
528
|
}).pipe(effect.Effect.withSpan("Cli.evalImport"))).pipe(_effect_cli.Command.withDescription("Turn a case file a team already wrote into a suite"));
|
|
583
529
|
//#endregion
|
|
530
|
+
//#region src/cli/eval-outcome.ts
|
|
531
|
+
const EvalOutcome = effect.Schema.Struct({
|
|
532
|
+
file: effect.Schema.String,
|
|
533
|
+
problems: effect.Schema.Array(effect.Schema.String),
|
|
534
|
+
run: effect.Schema.NullOr(require_evals.EvalRun),
|
|
535
|
+
runId: effect.Schema.NullOr(effect.Schema.String)
|
|
536
|
+
});
|
|
537
|
+
effect.Schema.Struct({
|
|
538
|
+
conclusion: effect.Schema.Literal("failure", "neutral", "success"),
|
|
539
|
+
details_url: effect.Schema.optional(effect.Schema.String),
|
|
540
|
+
name: effect.Schema.Literal("anpord"),
|
|
541
|
+
output: effect.Schema.Struct({
|
|
542
|
+
summary: effect.Schema.String,
|
|
543
|
+
title: effect.Schema.String
|
|
544
|
+
})
|
|
545
|
+
});
|
|
546
|
+
const TRUNCATED = "\n\n… truncated";
|
|
547
|
+
const TRAILING_SLASH = /\/$/;
|
|
548
|
+
const TITLES = {
|
|
549
|
+
failure: "Eval gate failed",
|
|
550
|
+
success: "Eval gate passed",
|
|
551
|
+
neutral: "Evals still running"
|
|
552
|
+
};
|
|
553
|
+
const percent = (rate) => rate === void 0 ? "-" : `${Math.round(rate * 100)}%`;
|
|
554
|
+
const escaped = (text) => text.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll("|", "\\|").replaceAll(/[\r\n]/g, " ");
|
|
555
|
+
const formatComparison = (run, cell) => {
|
|
556
|
+
const comparison = cell.comparison;
|
|
557
|
+
if (comparison === null) return "-";
|
|
558
|
+
const { baselineHarnessVersion: before, candidateHarnessVersion: after } = comparison;
|
|
559
|
+
const changed = before === after ? "" : ` (${run.tasks[cell.taskIndex]?.harness} ${before} → ${after})`;
|
|
560
|
+
return `${comparison.verdict}${changed}`;
|
|
561
|
+
};
|
|
562
|
+
const formatCellRow = (run, cell) => {
|
|
563
|
+
const rate = cell.distribution?.scored ? cell.distribution.passRate : void 0;
|
|
564
|
+
return `| ${escaped(cell.caseName)} | ${escaped(formatVariant(run, cell))} | ${percent(rate)} | ${percent(cell.comparison?.baselinePassRate)} | ${escaped(formatComparison(run, cell))} |`;
|
|
565
|
+
};
|
|
566
|
+
const runUrl = (webUrl, id) => `${webUrl.replace(TRAILING_SLASH, "")}/evals/${encodeURIComponent(id)}`;
|
|
567
|
+
const formatOutcome = ({ file, problems, run, runId }, webUrl) => [
|
|
568
|
+
`### ${escaped(file)}`,
|
|
569
|
+
"",
|
|
570
|
+
...runId === null ? [] : [`[View run](${runUrl(webUrl, runId)})`, ""],
|
|
571
|
+
...problems.map((problem) => `- ${escaped(problem)}`),
|
|
572
|
+
...run === null ? [] : [
|
|
573
|
+
"",
|
|
574
|
+
"| Case | Variant | Pass rate | Baseline | Verdict |",
|
|
575
|
+
"| --- | --- | --- | --- | --- |",
|
|
576
|
+
...run.cells.map((cell) => formatCellRow(run, cell))
|
|
577
|
+
]
|
|
578
|
+
].join("\n");
|
|
579
|
+
const buildGithubCheck = (outcomes, webUrl) => {
|
|
580
|
+
const failed = outcomes.some((outcome) => outcome.problems.length > 0);
|
|
581
|
+
const completed = outcomes.length > 0 && outcomes.every((outcome) => outcome.run !== null);
|
|
582
|
+
const first = outcomes.find((outcome) => outcome.runId !== null);
|
|
583
|
+
const conclusion = failed ? "failure" : completed ? "success" : "neutral";
|
|
584
|
+
const summary = outcomes.map((outcome) => formatOutcome(outcome, webUrl)).join("\n\n");
|
|
585
|
+
return {
|
|
586
|
+
conclusion,
|
|
587
|
+
details_url: first?.runId ? runUrl(webUrl, first.runId) : void 0,
|
|
588
|
+
name: "anpord",
|
|
589
|
+
output: {
|
|
590
|
+
title: TITLES[conclusion],
|
|
591
|
+
summary: summary.length < 65535 ? summary : `${summary.slice(0, 65521)}${TRUNCATED}`
|
|
592
|
+
}
|
|
593
|
+
};
|
|
594
|
+
};
|
|
595
|
+
//#endregion
|
|
596
|
+
//#region src/cli/eval-report.ts
|
|
597
|
+
const reportJson = effect.Schema.encodeSync(effect.Schema.parseJson(effect.Schema.Array(EvalOutcome)));
|
|
598
|
+
const appendSummary = (text) => effect.Effect.gen(function* () {
|
|
599
|
+
const path = yield* effect.Config.string("GITHUB_STEP_SUMMARY").pipe(effect.Config.option);
|
|
600
|
+
if (effect.Option.isSome(path) && path.value !== "") yield* (yield* _effect_platform.FileSystem.FileSystem).writeFileString(path.value, `${text}\n\n`, { flag: "a" });
|
|
601
|
+
});
|
|
602
|
+
const reportStarted = (file, id) => effect.Effect.gen(function* () {
|
|
603
|
+
const url = runUrl(yield* require_config.webUrlConfig, id);
|
|
604
|
+
yield* note(`${file}: ${url}`);
|
|
605
|
+
yield* appendSummary(`[Run ${id}](${url}) started.`);
|
|
606
|
+
});
|
|
607
|
+
const writeReport = (outcomes, path) => effect.Effect.gen(function* () {
|
|
608
|
+
if (effect.Option.isSome(path)) yield* (yield* _effect_platform.FileSystem.FileSystem).writeFileString(path.value, reportJson(outcomes));
|
|
609
|
+
});
|
|
610
|
+
const reportFinished = (outcomes) => effect.Effect.gen(function* () {
|
|
611
|
+
const report = buildGithubCheck(outcomes, yield* require_config.webUrlConfig);
|
|
612
|
+
yield* appendSummary(`## ${report.output.title}\n\n${report.output.summary}`);
|
|
613
|
+
});
|
|
614
|
+
//#endregion
|
|
584
615
|
//#region src/cli/eval-run.ts
|
|
585
616
|
const FIRST_POLL = 2e3;
|
|
586
617
|
const SLOWEST_POLL = 1e4;
|
|
587
618
|
const WIDENING = 1.5;
|
|
588
619
|
const running = (run) => run.status === "running";
|
|
589
|
-
|
|
620
|
+
var EvalWaitTimeout = class extends effect.Data.TaggedError("EvalWaitTimeout") {
|
|
621
|
+
get message() {
|
|
622
|
+
return `Timed out after ${this.seconds}s waiting for ${this.runId}. The remote run was not cancelled.`;
|
|
623
|
+
}
|
|
624
|
+
};
|
|
625
|
+
const waitForRun = (id, onProgress, timeoutSeconds) => effect.Effect.gen(function* () {
|
|
590
626
|
const api = yield* require_client.AnpordApi;
|
|
591
627
|
const startedAt = yield* effect.Clock.currentTimeMillis;
|
|
592
628
|
const gap = yield* effect.Ref.make(FIRST_POLL);
|
|
@@ -604,73 +640,28 @@ const waitForRun = (id, onProgress) => effect.Effect.gen(function* () {
|
|
|
604
640
|
body: () => waitThenPoll,
|
|
605
641
|
while: running
|
|
606
642
|
});
|
|
607
|
-
}).pipe(effect.Effect.
|
|
608
|
-
effect.
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
output: effect.Schema.Struct({
|
|
613
|
-
summary: effect.Schema.String,
|
|
614
|
-
title: effect.Schema.String
|
|
643
|
+
}).pipe(effect.Effect.timeoutFail({
|
|
644
|
+
duration: effect.Duration.seconds(timeoutSeconds),
|
|
645
|
+
onTimeout: () => new EvalWaitTimeout({
|
|
646
|
+
runId: id,
|
|
647
|
+
seconds: timeoutSeconds
|
|
615
648
|
})
|
|
649
|
+
}), effect.Effect.withSpan("Cli.waitForRun", { attributes: { runId: id } }));
|
|
650
|
+
//#endregion
|
|
651
|
+
//#region src/cli/eval-trigger.ts
|
|
652
|
+
const evalTrigger = effect.Effect.gen(function* () {
|
|
653
|
+
const github = yield* effect.Config.boolean("GITHUB_ACTIONS").pipe(effect.Config.withDefault(false));
|
|
654
|
+
const ci = yield* effect.Config.boolean("CI").pipe(effect.Config.withDefault(false));
|
|
655
|
+
if (!github) return { source: ci ? "ci" : "cli" };
|
|
656
|
+
const repository = yield* effect.Config.string("GITHUB_REPOSITORY");
|
|
657
|
+
const runId = yield* effect.Config.string("GITHUB_RUN_ID");
|
|
658
|
+
const attempt = yield* effect.Config.string("GITHUB_RUN_ATTEMPT").pipe(effect.Config.withDefault("1"));
|
|
659
|
+
const server = yield* effect.Config.string("GITHUB_SERVER_URL").pipe(effect.Config.withDefault("https://github.com"));
|
|
660
|
+
return yield* effect.Schema.decodeUnknown(require_evals.EvalTrigger)({
|
|
661
|
+
source: "ci",
|
|
662
|
+
url: `${server}/${repository}/actions/runs/${runId}/attempts/${attempt}`
|
|
663
|
+
});
|
|
616
664
|
});
|
|
617
|
-
const TRUNCATED = "\n\n… truncated";
|
|
618
|
-
const PERCENT = 100;
|
|
619
|
-
const ABSENT = "—";
|
|
620
|
-
const percent = (rate) => rate === void 0 ? ABSENT : `${Math.round(rate * PERCENT)}%`;
|
|
621
|
-
const rateOf = (cell) => cell.distribution === null || cell.distribution.scored === 0 ? ABSENT : percent(cell.distribution.passRate);
|
|
622
|
-
const versionsOf = (comparison) => ({
|
|
623
|
-
baseline: "baselineHarnessVersion" in comparison && typeof comparison.baselineHarnessVersion === "string" ? comparison.baselineHarnessVersion : void 0,
|
|
624
|
-
candidate: "candidateHarnessVersion" in comparison && typeof comparison.candidateHarnessVersion === "string" ? comparison.candidateHarnessVersion : void 0
|
|
625
|
-
});
|
|
626
|
-
const versionClause = (run, cell) => {
|
|
627
|
-
if (cell.comparison === null) return "";
|
|
628
|
-
const { baseline, candidate } = versionsOf(cell.comparison);
|
|
629
|
-
const harness = run.tasks[cell.taskIndex]?.harness ?? "harness";
|
|
630
|
-
return baseline === void 0 || candidate === void 0 || baseline === candidate ? "" : ` (${harness} ${baseline} → ${candidate})`;
|
|
631
|
-
};
|
|
632
|
-
const verdictOf = (run, cell) => cell.comparison === null ? ABSENT : `${cell.comparison.verdict}${versionClause(run, cell)}`;
|
|
633
|
-
const escaped = (text) => text.replaceAll("|", "\\|");
|
|
634
|
-
const rowOf = (run, cell) => `| ${escaped(cell.caseName)} | ${escaped(variantOf(run, cell))} | ${rateOf(cell)} | ${percent(cell.comparison?.baselinePassRate)} | ${verdictOf(run, cell)} |`;
|
|
635
|
-
const tableOf = (file, run) => [
|
|
636
|
-
`### ${escaped(file)}`,
|
|
637
|
-
"",
|
|
638
|
-
...run.failure === null ? [] : [`Run failed: ${run.failure}`, ""],
|
|
639
|
-
"| Case | Variant | Pass rate | Baseline | Verdict |",
|
|
640
|
-
"| --- | --- | --- | --- | --- |",
|
|
641
|
-
...run.cells.map((cell) => rowOf(run, cell))
|
|
642
|
-
].join("\n");
|
|
643
|
-
const truncated = (summary) => summary.length < 65535 ? summary : `${summary.slice(0, 65521)}${TRUNCATED}`;
|
|
644
|
-
const conclusionOf = (cells) => {
|
|
645
|
-
const verdicts = cells.flatMap((cell) => cell.comparison === null ? [] : [cell.comparison.verdict]);
|
|
646
|
-
if (verdicts.includes("regressed")) return "failure";
|
|
647
|
-
return verdicts.every((verdict) => verdict === "incomparable") ? "neutral" : "success";
|
|
648
|
-
};
|
|
649
|
-
const TITLES = {
|
|
650
|
-
failure: "A cell regressed against its baseline",
|
|
651
|
-
neutral: "Nothing to compare against a baseline",
|
|
652
|
-
success: "No cell regressed against its baseline"
|
|
653
|
-
};
|
|
654
|
-
const checkRunOf = (outcomes, webUrl) => {
|
|
655
|
-
const finished = outcomes.flatMap((outcome) => effect.Option.match(outcome.run, {
|
|
656
|
-
onNone: () => [],
|
|
657
|
-
onSome: (run) => [{
|
|
658
|
-
file: outcome.file,
|
|
659
|
-
run
|
|
660
|
-
}]
|
|
661
|
-
}));
|
|
662
|
-
const conclusion = conclusionOf(finished.flatMap(({ run }) => run.cells));
|
|
663
|
-
const first = finished[0];
|
|
664
|
-
return {
|
|
665
|
-
conclusion,
|
|
666
|
-
details_url: first === void 0 ? void 0 : `${webUrl}/evals/${first.run.id}`,
|
|
667
|
-
name: "anpord",
|
|
668
|
-
output: {
|
|
669
|
-
summary: truncated(finished.map(({ file, run }) => tableOf(file, run)).join("\n\n")),
|
|
670
|
-
title: TITLES[conclusion]
|
|
671
|
-
}
|
|
672
|
-
};
|
|
673
|
-
};
|
|
674
665
|
//#endregion
|
|
675
666
|
//#region src/cli/github-check-client.ts
|
|
676
667
|
const API = "https://api.github.com";
|
|
@@ -731,64 +722,90 @@ const githubContext = effect.Effect.gen(function* () {
|
|
|
731
722
|
});
|
|
732
723
|
//#endregion
|
|
733
724
|
//#region src/cli/eval-command.ts
|
|
734
|
-
const asJson$1 = _effect_cli.Options.boolean("json").pipe(_effect_cli.Options.withDescription("Print
|
|
735
|
-
const evalFile = _effect_cli.Args.text({ name: "file" }).pipe(_effect_cli.Args.withDescription("A TypeScript file
|
|
736
|
-
const noWait = _effect_cli.Options.boolean("no-wait").pipe(_effect_cli.Options.withDescription("Start
|
|
737
|
-
const failOn = _effect_cli.Options.choice("fail-on",
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
yield*
|
|
748
|
-
|
|
725
|
+
const asJson$1 = _effect_cli.Options.boolean("json").pipe(_effect_cli.Options.withDescription("Print each finished run as JSON"));
|
|
726
|
+
const evalFile = _effect_cli.Args.text({ name: "file" }).pipe(_effect_cli.Args.withDescription("A TypeScript eval file; discovers *.eval.ts when omitted"), _effect_cli.Args.optional);
|
|
727
|
+
const noWait = _effect_cli.Options.boolean("no-wait").pipe(_effect_cli.Options.withDescription("Start runs without waiting"));
|
|
728
|
+
const failOn = _effect_cli.Options.choice("fail-on", EvalGate.literals).pipe(_effect_cli.Options.withDescription("strict requires every trial to pass"), _effect_cli.Options.withDefault("strict"));
|
|
729
|
+
const timeout = _effect_cli.Options.integer("timeout").pipe(_effect_cli.Options.withDescription("Maximum seconds to wait per run"), _effect_cli.Options.withSchema(effect.Schema.Int.pipe(effect.Schema.positive())), _effect_cli.Options.withDefault(1200));
|
|
730
|
+
const output = _effect_cli.Options.text("output").pipe(_effect_cli.Options.withDescription("Write a JSON report to this file"), _effect_cli.Options.optional);
|
|
731
|
+
const describe$1 = (error) => error instanceof Error ? error.message : String(error);
|
|
732
|
+
const runOneEval = (file, options, save) => effect.Effect.gen(function* () {
|
|
733
|
+
let runId = null;
|
|
734
|
+
return yield* effect.Effect.gen(function* () {
|
|
735
|
+
const api = yield* require_client.AnpordApi;
|
|
736
|
+
const payload = yield* require_compiler.compileEvalEffect(file);
|
|
737
|
+
const trigger = yield* evalTrigger;
|
|
738
|
+
const started = yield* api.evals.start({ payload: {
|
|
739
|
+
...payload,
|
|
740
|
+
trigger
|
|
741
|
+
} });
|
|
742
|
+
runId = started.id;
|
|
743
|
+
yield* reportStarted(file, runId);
|
|
744
|
+
const pending = {
|
|
749
745
|
file,
|
|
746
|
+
runId,
|
|
750
747
|
problems: [],
|
|
751
|
-
run:
|
|
748
|
+
run: null
|
|
752
749
|
};
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
750
|
+
yield* save(pending);
|
|
751
|
+
if (options.skipWait) {
|
|
752
|
+
yield* json(started);
|
|
753
|
+
return pending;
|
|
754
|
+
}
|
|
755
|
+
const live = !options.wantsJson && (yield* attended);
|
|
756
|
+
const draw = yield* liveGrid(payload.trials, live);
|
|
757
|
+
const run = yield* waitForRun(runId, draw, options.timeoutSeconds);
|
|
758
|
+
yield* options.wantsJson ? json(run) : note(formatGridSummary(run, payload.trials, live));
|
|
759
|
+
return {
|
|
760
|
+
file,
|
|
761
|
+
runId,
|
|
762
|
+
run,
|
|
763
|
+
problems: problemsWith(run, options.gate, {
|
|
764
|
+
cells: payload.cases.length * payload.tasks.length,
|
|
765
|
+
trials: payload.trials
|
|
766
|
+
})
|
|
767
|
+
};
|
|
768
|
+
}).pipe(effect.Effect.catchAll((error) => effect.Effect.succeed({
|
|
760
769
|
file,
|
|
761
|
-
|
|
762
|
-
run:
|
|
763
|
-
|
|
770
|
+
runId,
|
|
771
|
+
run: null,
|
|
772
|
+
problems: [describe$1(error)]
|
|
773
|
+
})));
|
|
764
774
|
});
|
|
765
|
-
const describe$1 = (error) => error instanceof Error ? error.message : String(error);
|
|
766
775
|
const reportToGithub = (outcomes) => effect.Effect.gen(function* () {
|
|
767
|
-
if (outcomes.every((outcome) => effect.Option.isNone(outcome.run))) return;
|
|
768
776
|
const context = yield* githubContext;
|
|
769
777
|
if (effect.Option.isNone(context)) return;
|
|
770
778
|
const webUrl = yield* require_config.webUrlConfig;
|
|
771
|
-
yield* postCheckRun(context.value,
|
|
772
|
-
yield* note(`Posted the anpord check on ${context.value.sha}`);
|
|
779
|
+
yield* postCheckRun(context.value, buildGithubCheck(outcomes, webUrl));
|
|
773
780
|
}).pipe(effect.Effect.catchAll((error) => note(`The GitHub check was not posted. ${describe$1(error)}`)));
|
|
774
781
|
const runEval = _effect_cli.Command.make("eval", {
|
|
775
782
|
asJson: asJson$1,
|
|
776
783
|
evalFile,
|
|
777
784
|
failOn,
|
|
778
|
-
noWait
|
|
779
|
-
|
|
785
|
+
noWait,
|
|
786
|
+
output,
|
|
787
|
+
timeout
|
|
788
|
+
}, ({ asJson: wantsJson, evalFile: file, failOn: gate, noWait: skipWait, output: path, timeout: timeoutSeconds }) => effect.Effect.gen(function* () {
|
|
780
789
|
const files = yield* effect.Option.match(file, {
|
|
781
790
|
onNone: () => evalFilesIn("."),
|
|
782
791
|
onSome: (one) => effect.Effect.succeed([one])
|
|
783
792
|
});
|
|
784
793
|
if (files.length === 0) return yield* effect.Effect.fail(new NoEvalFiles());
|
|
785
|
-
const outcomes =
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
794
|
+
const outcomes = [];
|
|
795
|
+
yield* effect.Effect.forEach(files, (one, index) => effect.Effect.gen(function* () {
|
|
796
|
+
const save = (outcome) => effect.Effect.gen(function* () {
|
|
797
|
+
outcomes[index] = outcome;
|
|
798
|
+
yield* writeReport(outcomes, path);
|
|
799
|
+
});
|
|
800
|
+
yield* save(yield* runOneEval(one, {
|
|
801
|
+
gate,
|
|
802
|
+
skipWait,
|
|
803
|
+
timeoutSeconds,
|
|
804
|
+
wantsJson
|
|
805
|
+
}, save));
|
|
790
806
|
}));
|
|
791
|
-
yield*
|
|
807
|
+
yield* reportFinished(outcomes);
|
|
808
|
+
if (!skipWait) yield* reportToGithub(outcomes);
|
|
792
809
|
return yield* failWhen(outcomes.flatMap((outcome) => outcome.problems));
|
|
793
810
|
}).pipe(effect.Effect.provide(require_config.ClientLayer))).pipe(_effect_cli.Command.withDescription("Compile and run an eval from TypeScript"), _effect_cli.Command.withSubcommands([importEval]));
|
|
794
811
|
//#endregion
|
|
@@ -835,12 +852,6 @@ const promote = _effect_cli.Command.make("promote", {
|
|
|
835
852
|
} });
|
|
836
853
|
return yield* note(`${id} v${pin} is now ${to}`);
|
|
837
854
|
})).pipe(_effect_cli.Command.withDescription("Point a channel at a version"));
|
|
838
|
-
/**
|
|
839
|
-
* Read from the stream rather than by opening `/dev/stdin` as a file. The path
|
|
840
|
-
* only names the pipe, so reading it races whoever is writing: a body arriving
|
|
841
|
-
* in more than one chunk, which is what a pipe does under load, was read as
|
|
842
|
-
* whatever had landed by then.
|
|
843
|
-
*/
|
|
844
855
|
const readStdin = effect.Effect.async((resume) => {
|
|
845
856
|
let body = "";
|
|
846
857
|
process.stdin.setEncoding("utf8");
|
|
@@ -865,8 +876,6 @@ const push = _effect_cli.Command.make("push", {
|
|
|
865
876
|
return yield* note(`${id} is now v${prompt.version}`);
|
|
866
877
|
})).pipe(_effect_cli.Command.withDescription("Add a version to a prompt"));
|
|
867
878
|
const out = _effect_cli.Options.file("out").pipe(_effect_cli.Options.withDescription("Where to write the declarations"), _effect_cli.Options.withDefault("anpord-env.d.ts"));
|
|
868
|
-
/** Reading one prompt at a time because the list carries no content, bounded
|
|
869
|
-
* so a large organisation does not open a connection per prompt. */
|
|
870
879
|
const READ_AT_ONCE = 8;
|
|
871
880
|
const writeDeclarations = ({ out: path }) => effect.Effect.gen(function* () {
|
|
872
881
|
const api = yield* require_client.AnpordApi;
|
|
@@ -878,8 +887,6 @@ const writeDeclarations = ({ out: path }) => effect.Effect.gen(function* () {
|
|
|
878
887
|
});
|
|
879
888
|
const DESCRIPTION = "Write TypeScript declarations for prompt variables";
|
|
880
889
|
const generate = _effect_cli.Command.make("generate", { out }, writeDeclarations).pipe(_effect_cli.Command.withDescription(DESCRIPTION));
|
|
881
|
-
/** A second command rather than an alias, because a command carries one name
|
|
882
|
-
* and the shorter one is what anybody types twice. */
|
|
883
890
|
const gen = _effect_cli.Command.make("gen", { out }, writeDeclarations).pipe(_effect_cli.Command.withDescription(DESCRIPTION));
|
|
884
891
|
const withClient = (command) => _effect_cli.Command.provide(command, require_config.ClientLayer);
|
|
885
892
|
const commands = [
|