@tangle-network/agent-runtime 0.197.1 → 0.198.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -2
- package/dist/{improve-DZs0KXhK.d.ts → activation-DFRTurvU.d.ts} +99 -5
- package/dist/{activation-BxMZybuo.js → activation-IBEVN3VI.js} +2 -2
- package/dist/{activation-BxMZybuo.js.map → activation-IBEVN3VI.js.map} +1 -1
- package/dist/agent.d.ts +1 -1
- package/dist/agent.js +2 -2
- package/dist/candidate-execution/index.d.ts +370 -2
- package/dist/candidate-execution/index.js +828 -4
- package/dist/candidate-execution/index.js.map +1 -0
- package/dist/{coordination-driver-zkCrfEbS.js → coordination-driver-NFvZ5ofi.js} +57 -26
- package/dist/coordination-driver-NFvZ5ofi.js.map +1 -0
- package/dist/{delegate-DHeUYU5E.js → delegate-dN5yooEj.js} +2 -2
- package/dist/{delegate-DHeUYU5E.js.map → delegate-dN5yooEj.js.map} +1 -1
- package/dist/durable.d.ts +2 -2
- package/dist/durable.js +4 -5
- package/dist/durable.js.map +1 -1
- package/dist/{graph-51SJOEcw.js → graph-CgCVtMuz.js} +3 -3
- package/dist/{graph-51SJOEcw.js.map → graph-CgCVtMuz.js.map} +1 -1
- package/dist/{improvement-cycle-Tw5nYT5R.js → improvement-cycle-Cfu6kDOs.js} +8 -10
- package/dist/{improvement-cycle-Tw5nYT5R.js.map → improvement-cycle-Cfu6kDOs.js.map} +1 -1
- package/dist/{index-CYOJsxSg.d.ts → index-CMTUgh-T.d.ts} +9 -11
- package/dist/index.d.ts +694 -16
- package/dist/index.js +1780 -17
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +3 -7
- package/dist/intelligence.js +5 -6
- package/dist/intelligence.js.map +1 -1
- package/dist/kernel.d.ts +2 -5
- package/dist/kernel.js +8 -10
- package/dist/{loop-runner-bin-DwrSB04m.d.ts → loop-runner-bin-CAf1OQot.d.ts} +3 -3
- package/dist/{loop-runner-bin-2LQSjRTB.js → loop-runner-bin-WQniYJ8C.js} +3 -3
- package/dist/{loop-runner-bin-2LQSjRTB.js.map → loop-runner-bin-WQniYJ8C.js.map} +1 -1
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/mcp/bin.js +3 -3
- package/dist/mcp/index.d.ts +2 -3
- package/dist/mcp/index.js +4 -5
- package/dist/mcp/index.js.map +1 -1
- package/dist/{prepare-DDGp0-rW.js → prepare-CAO1yXov.js} +2407 -2407
- package/dist/prepare-CAO1yXov.js.map +1 -0
- package/dist/{protected-model-port-DxFN8DLS.js → protected-model-port-B5avcRiQ.js} +2 -2
- package/dist/{protected-model-port-DxFN8DLS.js.map → protected-model-port-B5avcRiQ.js.map} +1 -1
- package/dist/{provision-supervisor-CpMMShE_.js → provision-supervisor-BxsIaJ35.js} +3 -6
- package/dist/{provision-supervisor-CpMMShE_.js.map → provision-supervisor-BxsIaJ35.js.map} +1 -1
- package/dist/{redact-Cbl2O-4N.js → redact-DqfB7oB4.js} +5233 -2526
- package/dist/redact-DqfB7oB4.js.map +1 -0
- package/dist/{runtime-BSFz2z7h.js → runtime-CHEtvaTY.js} +427 -23
- package/dist/runtime-CHEtvaTY.js.map +1 -0
- package/dist/{server-COXa19sp.js → server-ccGua5tH.js} +4 -4
- package/dist/{server-COXa19sp.js.map → server-ccGua5tH.js.map} +1 -1
- package/dist/{types-DFLZMaeh.d.ts → stream-agent-turn-CLOQr497.d.ts} +1986 -6
- package/dist/{structural-rollout-DPbZWgEm.js → structural-rollout-BmDuXyR9.js} +1101 -8
- package/dist/structural-rollout-BmDuXyR9.js.map +1 -0
- package/dist/{supervise-b93YUaka.js → supervise-DVt8-TI-.js} +9 -5
- package/dist/supervise-DVt8-TI-.js.map +1 -0
- package/dist/testing.d.ts +2 -2
- package/dist/testing.js +13 -13
- package/dist/tui/index.d.ts +1 -1
- package/dist/tui/index.js +1 -1
- package/dist/{workspace-archive-Ybomp7AN.js → workspace-archive-BMOnloFf.js} +3 -3
- package/dist/{workspace-archive-Ybomp7AN.js.map → workspace-archive-BMOnloFf.js.map} +1 -1
- package/package.json +1 -28
- package/dist/activation-DyWB0K6E.d.ts +0 -98
- package/dist/authored-code-URmkdgjv.js +0 -37
- package/dist/authored-code-URmkdgjv.js.map +0 -1
- package/dist/candidate-execution-nvqVIMyS.js +0 -829
- package/dist/candidate-execution-nvqVIMyS.js.map +0 -1
- package/dist/conversation-BxJ0SIBM.js +0 -1363
- package/dist/conversation-BxJ0SIBM.js.map +0 -1
- package/dist/conversation.d.ts +0 -2
- package/dist/conversation.js +0 -2
- package/dist/coordination-driver-zkCrfEbS.js.map +0 -1
- package/dist/environment-provider-1fKZh2zl.js +0 -2281
- package/dist/environment-provider-1fKZh2zl.js.map +0 -1
- package/dist/environment-provider-B-I2jlQy.d.ts +0 -143
- package/dist/environment-provider.d.ts +0 -2
- package/dist/environment-provider.js +0 -2
- package/dist/graph.d.ts +0 -753
- package/dist/graph.js +0 -2111
- package/dist/graph.js.map +0 -1
- package/dist/index-CUosKU4N.d.ts +0 -372
- package/dist/index-D9mb6fn2.d.ts +0 -691
- package/dist/index-ZnxSe6iK.d.ts +0 -138
- package/dist/jsonl-file-BEpaEYjT.js +0 -141
- package/dist/jsonl-file-BEpaEYjT.js.map +0 -1
- package/dist/knowledge-B_MsOtDG.js +0 -428
- package/dist/knowledge-B_MsOtDG.js.map +0 -1
- package/dist/knowledge.d.ts +0 -2
- package/dist/knowledge.js +0 -2
- package/dist/materialization-Cy0oM8tb.js +0 -672
- package/dist/materialization-Cy0oM8tb.js.map +0 -1
- package/dist/prepare-DDGp0-rW.js.map +0 -1
- package/dist/primeintellect/index.d.ts +0 -218
- package/dist/primeintellect/index.js +0 -739
- package/dist/primeintellect/index.js.map +0 -1
- package/dist/redact-Cbl2O-4N.js.map +0 -1
- package/dist/runtime-0xNaV6TJ.d.ts +0 -1699
- package/dist/runtime-BSFz2z7h.js.map +0 -1
- package/dist/stream-agent-turn-Dt5mZpc3.js +0 -1103
- package/dist/stream-agent-turn-Dt5mZpc3.js.map +0 -1
- package/dist/stream-agent-turn-urHpmO_Z.d.ts +0 -160
- package/dist/structural-rollout-DPbZWgEm.js.map +0 -1
- package/dist/supervise-b93YUaka.js.map +0 -1
|
@@ -1,739 +0,0 @@
|
|
|
1
|
-
import { canonicalJson, validateRunRecord } from "@tangle-network/agent-eval";
|
|
2
|
-
import { createHash, randomUUID } from "node:crypto";
|
|
3
|
-
import { mkdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
|
|
4
|
-
import { basename, dirname, isAbsolute, join, normalize, relative, resolve, sep } from "node:path";
|
|
5
|
-
//#region src/primeintellect/validation.ts
|
|
6
|
-
/** Validate the PrimeIntellect prompt shape shared by package creation and runner input. */
|
|
7
|
-
function validatePrimeIntellectPrompt(value, path) {
|
|
8
|
-
if (typeof value === "string" && value.length > 0) return value;
|
|
9
|
-
if (!Array.isArray(value) || value.length === 0) throw new Error(`${path} must be a non-empty string or message array`);
|
|
10
|
-
for (const [index, message] of value.entries()) validateMessage(message, `${path}[${index}]`);
|
|
11
|
-
return value;
|
|
12
|
-
}
|
|
13
|
-
function validatePrimeIntellectJson(value, path) {
|
|
14
|
-
if (value === null || ["string", "boolean"].includes(typeof value)) return;
|
|
15
|
-
if (typeof value === "number") {
|
|
16
|
-
if (!Number.isFinite(value)) throw new Error(`${path} contains a non-finite number`);
|
|
17
|
-
return;
|
|
18
|
-
}
|
|
19
|
-
if (Array.isArray(value)) {
|
|
20
|
-
value.forEach((entry, index) => {
|
|
21
|
-
validatePrimeIntellectJson(entry, `${path}[${index}]`);
|
|
22
|
-
});
|
|
23
|
-
return;
|
|
24
|
-
}
|
|
25
|
-
if (typeof value === "object") {
|
|
26
|
-
for (const [key, entry] of Object.entries(value)) validatePrimeIntellectJson(entry, `${path}.${key}`);
|
|
27
|
-
return;
|
|
28
|
-
}
|
|
29
|
-
throw new Error(`${path} is not JSON serializable`);
|
|
30
|
-
}
|
|
31
|
-
function validatePrimeIntellectJsonObject(value, path) {
|
|
32
|
-
const output = record$1(value, path);
|
|
33
|
-
validatePrimeIntellectJson(output, path);
|
|
34
|
-
return output;
|
|
35
|
-
}
|
|
36
|
-
function validateMessage(value, path) {
|
|
37
|
-
const message = record$1(value, path);
|
|
38
|
-
const role = message.role;
|
|
39
|
-
if (![
|
|
40
|
-
"system",
|
|
41
|
-
"user",
|
|
42
|
-
"assistant",
|
|
43
|
-
"tool"
|
|
44
|
-
].includes(String(role))) throw new Error(`${path}.role is invalid`);
|
|
45
|
-
if (role === "system" || role === "user") {
|
|
46
|
-
assertOnlyKeys(message, ["role", "content"], path);
|
|
47
|
-
validateContent(message.content, `${path}.content`);
|
|
48
|
-
} else if (role === "assistant") {
|
|
49
|
-
assertOnlyKeys(message, [
|
|
50
|
-
"role",
|
|
51
|
-
"content",
|
|
52
|
-
"reasoning_content",
|
|
53
|
-
"tool_calls",
|
|
54
|
-
"provider_state"
|
|
55
|
-
], path);
|
|
56
|
-
optionalNullableString(message.content, `${path}.content`);
|
|
57
|
-
optionalNullableString(message.reasoning_content, `${path}.reasoning_content`);
|
|
58
|
-
if (message.tool_calls !== void 0) validateToolCalls(message.tool_calls, `${path}.tool_calls`);
|
|
59
|
-
if (message.provider_state !== void 0) validateProviderState(message.provider_state, `${path}.provider_state`);
|
|
60
|
-
} else {
|
|
61
|
-
assertOnlyKeys(message, [
|
|
62
|
-
"role",
|
|
63
|
-
"tool_call_id",
|
|
64
|
-
"content",
|
|
65
|
-
"name"
|
|
66
|
-
], path);
|
|
67
|
-
nonEmptyString$2(message.tool_call_id, `${path}.tool_call_id`);
|
|
68
|
-
if (message.name !== void 0 && typeof message.name !== "string") throw new Error(`${path}.name must be a string`);
|
|
69
|
-
validateContent(message.content, `${path}.content`);
|
|
70
|
-
}
|
|
71
|
-
validatePrimeIntellectJson(message, path);
|
|
72
|
-
}
|
|
73
|
-
function validateToolCalls(value, path) {
|
|
74
|
-
if (!Array.isArray(value)) throw new Error(`${path} must be an array`);
|
|
75
|
-
for (const [index, rawCall] of value.entries()) {
|
|
76
|
-
const call = record$1(rawCall, `${path}[${index}]`);
|
|
77
|
-
assertOnlyKeys(call, [
|
|
78
|
-
"id",
|
|
79
|
-
"name",
|
|
80
|
-
"arguments"
|
|
81
|
-
], `${path}[${index}]`);
|
|
82
|
-
for (const field of [
|
|
83
|
-
"id",
|
|
84
|
-
"name",
|
|
85
|
-
"arguments"
|
|
86
|
-
]) nonEmptyString$2(call[field], `${path}[${index}].${field}`);
|
|
87
|
-
}
|
|
88
|
-
}
|
|
89
|
-
function validateProviderState(value, path) {
|
|
90
|
-
if (!Array.isArray(value)) throw new Error(`${path} must be an array`);
|
|
91
|
-
for (const [index, rawState] of value.entries()) validatePrimeIntellectJson(record$1(rawState, `${path}[${index}]`), `${path}[${index}]`);
|
|
92
|
-
}
|
|
93
|
-
function validateContent(value, path) {
|
|
94
|
-
if (typeof value === "string") return;
|
|
95
|
-
if (!Array.isArray(value) || value.length === 0) throw new Error(`${path} must be a string or non-empty content array`);
|
|
96
|
-
for (const [index, rawPart] of value.entries()) {
|
|
97
|
-
const part = record$1(rawPart, `${path}[${index}]`);
|
|
98
|
-
if (part.type === "text" && typeof part.text === "string") {
|
|
99
|
-
assertOnlyKeys(part, ["type", "text"], `${path}[${index}]`);
|
|
100
|
-
continue;
|
|
101
|
-
}
|
|
102
|
-
const image = part.image_url;
|
|
103
|
-
if (part.type === "image_url" && image !== null && typeof image === "object" && !Array.isArray(image) && typeof image.url === "string") {
|
|
104
|
-
assertOnlyKeys(part, ["type", "image_url"], `${path}[${index}]`);
|
|
105
|
-
assertOnlyKeys(image, ["url"], `${path}[${index}].image_url`);
|
|
106
|
-
continue;
|
|
107
|
-
}
|
|
108
|
-
throw new Error(`${path}[${index}] is not a supported text or image_url content part`);
|
|
109
|
-
}
|
|
110
|
-
}
|
|
111
|
-
function optionalNullableString(value, path) {
|
|
112
|
-
if (value !== void 0 && value !== null && typeof value !== "string") throw new Error(`${path} must be a string or null`);
|
|
113
|
-
}
|
|
114
|
-
function nonEmptyString$2(value, path) {
|
|
115
|
-
if (typeof value !== "string" || value.length === 0) throw new Error(`${path} must be a non-empty string`);
|
|
116
|
-
return value;
|
|
117
|
-
}
|
|
118
|
-
function record$1(value, path) {
|
|
119
|
-
if (value === null || typeof value !== "object" || Array.isArray(value)) throw new Error(`${path} must be an object`);
|
|
120
|
-
return value;
|
|
121
|
-
}
|
|
122
|
-
function assertOnlyKeys(value, allowed, path) {
|
|
123
|
-
for (const key of Object.keys(value)) if (!allowed.includes(key)) throw new Error(`${path}.${key} is not supported`);
|
|
124
|
-
}
|
|
125
|
-
//#endregion
|
|
126
|
-
//#region src/primeintellect/package.ts
|
|
127
|
-
const VERIFIERS_RANGE = ">=0.2.0,<0.3.0";
|
|
128
|
-
const ENV_NAME = /^[A-Z_][A-Z0-9_]*$/;
|
|
129
|
-
const PACKAGE_NAME = /^[a-z][a-z0-9-]{0,62}$/;
|
|
130
|
-
const VERSION = /^\d+\.\d+\.\d+(?:[-+][a-zA-Z0-9.-]+)?$/;
|
|
131
|
-
const DEFAULT_MAX_TURNS = 16;
|
|
132
|
-
const DEFAULT_ROLLOUT_TIMEOUT = 3600;
|
|
133
|
-
const DEFAULT_SCORING_TIMEOUT = 300;
|
|
134
|
-
/** Build a complete PrimeIntellect Verifiers package without writing to disk. */
|
|
135
|
-
function createPrimeIntellectPackage(options) {
|
|
136
|
-
const validated = validateOptions(options);
|
|
137
|
-
const moduleName = validated.name.replaceAll("-", "_");
|
|
138
|
-
const rows = validated.tasks.map((task, idx) => taskRow(task, idx));
|
|
139
|
-
const runnerFiles = validated.runner.files ?? {};
|
|
140
|
-
const scoringFiles = validated.scoring.kind === "command" ? validated.scoring.files ?? {} : {};
|
|
141
|
-
const files = {
|
|
142
|
-
"pyproject.toml": renderPyproject(validated, moduleName),
|
|
143
|
-
"prime.eval.toml": renderPrimeConfig(validated, "eval"),
|
|
144
|
-
"prime.train.toml": renderPrimeConfig(validated, "train"),
|
|
145
|
-
"README.md": renderReadme(validated),
|
|
146
|
-
[`${moduleName}/__init__.py`]: renderInit(moduleName),
|
|
147
|
-
[`${moduleName}/taskset.py`]: renderTaskset(moduleName, validated.scoring),
|
|
148
|
-
[`${moduleName}/harness.py`]: renderHarness(moduleName),
|
|
149
|
-
[`${moduleName}/tasks.jsonl`]: `${rows.map((row) => JSON.stringify(row)).join("\n")}\n`,
|
|
150
|
-
[`${moduleName}/runner.json`]: `${JSON.stringify({
|
|
151
|
-
command: validated.runner.command,
|
|
152
|
-
files: runnerFiles,
|
|
153
|
-
setup: validated.runner.setup ?? [],
|
|
154
|
-
forwardEnv: validated.runner.forwardEnv ?? []
|
|
155
|
-
}, null, 2)}\n`
|
|
156
|
-
};
|
|
157
|
-
for (const [path, contents] of Object.entries(scoringFiles)) files[`${moduleName}/scoring/${path}`] = contents;
|
|
158
|
-
const filesSha256 = Object.fromEntries(Object.entries(files).sort(([left], [right]) => left.localeCompare(right)).map(([path, contents]) => [path, sha256(contents)]));
|
|
159
|
-
const manifest = {
|
|
160
|
-
kind: "tangle.primeintellect.package",
|
|
161
|
-
name: validated.name,
|
|
162
|
-
moduleName,
|
|
163
|
-
version: validated.version,
|
|
164
|
-
verifiers: VERIFIERS_RANGE,
|
|
165
|
-
taskCount: validated.tasks.length,
|
|
166
|
-
splits: {
|
|
167
|
-
train: validated.tasks.filter((task) => task.split === "train").length,
|
|
168
|
-
eval: validated.tasks.filter((task) => task.split === "eval").length
|
|
169
|
-
},
|
|
170
|
-
taskIdsSha256: sha256(validated.tasks.map((task) => `${task.split}:${task.id}`).sort().join("\n")),
|
|
171
|
-
filesSha256
|
|
172
|
-
};
|
|
173
|
-
files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}\n`;
|
|
174
|
-
return {
|
|
175
|
-
manifest,
|
|
176
|
-
files: Object.freeze(files)
|
|
177
|
-
};
|
|
178
|
-
}
|
|
179
|
-
/** Write a bundle through a sibling temporary directory, then rename it into place. */
|
|
180
|
-
async function writePrimeIntellectPackage(bundle, outputDirectory, options = {}) {
|
|
181
|
-
const output = resolve(outputDirectory);
|
|
182
|
-
const parent = dirname(output);
|
|
183
|
-
await mkdir(parent, { recursive: true });
|
|
184
|
-
const replacing = await pathExists(output);
|
|
185
|
-
if (replacing) {
|
|
186
|
-
if (!options.replace) throw new Error(`PrimeIntellect output already exists: ${output}`);
|
|
187
|
-
await assertGeneratedPackage(output);
|
|
188
|
-
}
|
|
189
|
-
const temporary = join(parent, `.${basename(output)}.${randomUUID()}.tmp`);
|
|
190
|
-
const backup = replacing ? join(parent, `.${basename(output)}.${randomUUID()}.backup`) : void 0;
|
|
191
|
-
try {
|
|
192
|
-
await mkdir(temporary);
|
|
193
|
-
for (const [path, contents] of Object.entries(bundle.files)) {
|
|
194
|
-
assertRelativePath(path, "bundle file");
|
|
195
|
-
const target = resolve(temporary, path);
|
|
196
|
-
if (target !== temporary && !target.startsWith(`${temporary}${sep}`)) throw new Error(`bundle file escapes output directory: ${path}`);
|
|
197
|
-
await mkdir(dirname(target), { recursive: true });
|
|
198
|
-
await writeFile(target, contents, "utf8");
|
|
199
|
-
}
|
|
200
|
-
if (backup) await rename(output, backup);
|
|
201
|
-
try {
|
|
202
|
-
await rename(temporary, output);
|
|
203
|
-
} catch (error) {
|
|
204
|
-
if (backup) try {
|
|
205
|
-
await rename(backup, output);
|
|
206
|
-
} catch (restoreError) {
|
|
207
|
-
throw new AggregateError([error, restoreError], `failed to install PrimeIntellect package and restore ${output}`);
|
|
208
|
-
}
|
|
209
|
-
throw error;
|
|
210
|
-
}
|
|
211
|
-
if (backup) await rm(backup, { recursive: true });
|
|
212
|
-
} catch (error) {
|
|
213
|
-
await rm(temporary, {
|
|
214
|
-
recursive: true,
|
|
215
|
-
force: true
|
|
216
|
-
});
|
|
217
|
-
throw error;
|
|
218
|
-
}
|
|
219
|
-
return output;
|
|
220
|
-
}
|
|
221
|
-
function validateOptions(options) {
|
|
222
|
-
if (!PACKAGE_NAME.test(options.name)) throw new Error("PrimeIntellect package name must match /^[a-z][a-z0-9-]{0,62}$/");
|
|
223
|
-
if (!VERSION.test(options.version)) throw new Error("PrimeIntellect package version must be a numeric semantic version");
|
|
224
|
-
if (!Array.isArray(options.tasks) || options.tasks.length === 0) throw new Error("PrimeIntellect package requires tasks");
|
|
225
|
-
const tasks = options.tasks;
|
|
226
|
-
const seen = /* @__PURE__ */ new Set();
|
|
227
|
-
const seenInputs = /* @__PURE__ */ new Map();
|
|
228
|
-
const splitCounts = {
|
|
229
|
-
train: 0,
|
|
230
|
-
eval: 0
|
|
231
|
-
};
|
|
232
|
-
for (const [index, task] of tasks.entries()) {
|
|
233
|
-
validateTask(task, index, options.scoring);
|
|
234
|
-
if (seen.has(task.id)) throw new Error(`duplicate PrimeIntellect task id: ${task.id}`);
|
|
235
|
-
seen.add(task.id);
|
|
236
|
-
const input = canonicalJson({
|
|
237
|
-
prompt: task.prompt,
|
|
238
|
-
systemPrompt: task.systemPrompt ?? null,
|
|
239
|
-
metadata: task.metadata ?? {}
|
|
240
|
-
});
|
|
241
|
-
const duplicate = seenInputs.get(input);
|
|
242
|
-
if (duplicate) throw new Error(`PrimeIntellect tasks ${duplicate.id} (${duplicate.split}) and ${task.id} (${task.split}) expose the same public input`);
|
|
243
|
-
seenInputs.set(input, {
|
|
244
|
-
id: task.id,
|
|
245
|
-
split: task.split
|
|
246
|
-
});
|
|
247
|
-
splitCounts[task.split] += 1;
|
|
248
|
-
}
|
|
249
|
-
if (splitCounts.train === 0 || splitCounts.eval === 0) throw new Error("PrimeIntellect package requires non-empty, disjoint train and eval splits");
|
|
250
|
-
validateScoring(options.scoring);
|
|
251
|
-
validateCommand(options.runner.command, "runner.command");
|
|
252
|
-
validateFiles(options.runner.files ?? {}, "runner.files");
|
|
253
|
-
for (const [index, command] of (options.runner.setup ?? []).entries()) validateCommand(command, `runner.setup[${index}]`);
|
|
254
|
-
validateEnvNames(options.runner.forwardEnv ?? [], "runner.forwardEnv");
|
|
255
|
-
if (typeof options.runner.image !== "string" || options.runner.image.trim().length === 0) throw new Error("runner.image must be a non-empty container image");
|
|
256
|
-
if (/(^|:)latest$/i.test(options.runner.image)) throw new Error("runner.image must not use the mutable latest tag");
|
|
257
|
-
positiveInteger(options.maxTurns ?? DEFAULT_MAX_TURNS, "maxTurns");
|
|
258
|
-
for (const [name, value] of [
|
|
259
|
-
["maxInputTokens", options.maxInputTokens],
|
|
260
|
-
["maxOutputTokens", options.maxOutputTokens],
|
|
261
|
-
["maxTotalTokens", options.maxTotalTokens]
|
|
262
|
-
]) if (value !== void 0) positiveInteger(value, name);
|
|
263
|
-
positiveNumber(options.rolloutTimeoutSeconds ?? DEFAULT_ROLLOUT_TIMEOUT, "rolloutTimeoutSeconds");
|
|
264
|
-
positiveNumber(options.scoringTimeoutSeconds ?? DEFAULT_SCORING_TIMEOUT, "scoringTimeoutSeconds");
|
|
265
|
-
return options;
|
|
266
|
-
}
|
|
267
|
-
function validateTask(task, index, scoring) {
|
|
268
|
-
const path = `tasks[${index}]`;
|
|
269
|
-
if (typeof task.id !== "string" || task.id.trim().length === 0) throw new Error(`${path}.id must be a non-empty string`);
|
|
270
|
-
if (task.split !== "train" && task.split !== "eval") throw new Error(`${path}.split must be train or eval`);
|
|
271
|
-
const prompt = validatePrimeIntellectPrompt(task.prompt, `${path}.prompt`);
|
|
272
|
-
if (Array.isArray(prompt)) {
|
|
273
|
-
if (task.systemPrompt !== void 0 && prompt.some((message) => message.role === "system")) throw new Error(`${path} must not set systemPrompt and include a system message`);
|
|
274
|
-
}
|
|
275
|
-
if (task.systemPrompt !== void 0 && typeof task.systemPrompt !== "string") throw new Error(`${path}.systemPrompt must be a string`);
|
|
276
|
-
if (task.metadata !== void 0) validatePrimeIntellectJson(task.metadata, `${path}.metadata`);
|
|
277
|
-
if (scoring.kind !== "command") {
|
|
278
|
-
const answers = Array.isArray(task.answer) ? task.answer : [task.answer];
|
|
279
|
-
if (answers.length === 0 || answers.some((answer) => typeof answer !== "string" || answer.length === 0)) throw new Error(`${path}.answer is required for ${scoring.kind} scoring`);
|
|
280
|
-
}
|
|
281
|
-
}
|
|
282
|
-
function validateScoring(scoring) {
|
|
283
|
-
if (scoring.kind === "exact") return;
|
|
284
|
-
if (scoring.kind === "reference-judge") {
|
|
285
|
-
if (typeof scoring.model !== "string" || scoring.model.length === 0) throw new Error("reference-judge scoring requires a model");
|
|
286
|
-
return;
|
|
287
|
-
}
|
|
288
|
-
validateCommand(scoring.command, "scoring.command");
|
|
289
|
-
validateFiles(scoring.files ?? {}, "scoring.files");
|
|
290
|
-
validateEnvNames(scoring.forwardEnv ?? [], "scoring.forwardEnv");
|
|
291
|
-
positiveNumber(scoring.timeoutSeconds ?? DEFAULT_SCORING_TIMEOUT, "scoring.timeoutSeconds");
|
|
292
|
-
}
|
|
293
|
-
function validateCommand(command, path) {
|
|
294
|
-
if (!Array.isArray(command) || command.length === 0) throw new Error(`${path} must be a non-empty argv array`);
|
|
295
|
-
for (const [index, argument] of command.entries()) if (typeof argument !== "string" || argument.length === 0 || argument.includes("\0")) throw new Error(`${path}[${index}] must be a non-empty string without NUL bytes`);
|
|
296
|
-
}
|
|
297
|
-
function validateFiles(files, path) {
|
|
298
|
-
for (const [file, contents] of Object.entries(files)) {
|
|
299
|
-
assertRelativePath(file, path);
|
|
300
|
-
if (typeof contents !== "string") throw new Error(`${path}.${file} must be a string`);
|
|
301
|
-
}
|
|
302
|
-
}
|
|
303
|
-
function validateEnvNames(names, path) {
|
|
304
|
-
const seen = /* @__PURE__ */ new Set();
|
|
305
|
-
for (const [index, name] of names.entries()) {
|
|
306
|
-
if (!ENV_NAME.test(name)) throw new Error(`${path}[${index}] is not a valid environment name`);
|
|
307
|
-
if (seen.has(name)) throw new Error(`${path} contains duplicate name ${name}`);
|
|
308
|
-
seen.add(name);
|
|
309
|
-
}
|
|
310
|
-
}
|
|
311
|
-
function assertRelativePath(path, label) {
|
|
312
|
-
const normalized = normalize(path);
|
|
313
|
-
if (path.length === 0 || path.includes("\0") || isAbsolute(path) || normalized === ".." || normalized.startsWith(`..${sep}`) || relative(".", normalized).startsWith("..")) throw new Error(`${label} contains unsafe path: ${path}`);
|
|
314
|
-
}
|
|
315
|
-
function taskRow(task, idx) {
|
|
316
|
-
return {
|
|
317
|
-
idx,
|
|
318
|
-
name: task.id,
|
|
319
|
-
prompt: task.prompt,
|
|
320
|
-
system_prompt: task.systemPrompt ?? null,
|
|
321
|
-
split: task.split,
|
|
322
|
-
answer: task.answer ?? null,
|
|
323
|
-
metadata: task.metadata ?? {}
|
|
324
|
-
};
|
|
325
|
-
}
|
|
326
|
-
function renderPyproject(options, moduleName) {
|
|
327
|
-
const description = options.description ?? `PrimeIntellect tasks for ${options.name}`;
|
|
328
|
-
return `[project]\nname = ${toml(options.name)}\nversion = ${toml(options.version)}\ndescription = ${toml(description)}\nrequires-python = ">=3.11,<3.14"\ndependencies = ["verifiers${VERIFIERS_RANGE}"]\n\n[build-system]\nrequires = ["hatchling"]\nbuild-backend = "hatchling.build"\n\n[tool.hatch.build.targets.wheel]\npackages = [${toml(moduleName)}]\n\n[tool.uv]\nprerelease = "allow"\n`;
|
|
329
|
-
}
|
|
330
|
-
function renderPrimeConfig(options, split) {
|
|
331
|
-
return `${[
|
|
332
|
-
`max_turns = ${options.maxTurns ?? DEFAULT_MAX_TURNS}`,
|
|
333
|
-
options.maxInputTokens === void 0 ? void 0 : `max_input_tokens = ${options.maxInputTokens}`,
|
|
334
|
-
options.maxOutputTokens === void 0 ? void 0 : `max_output_tokens = ${options.maxOutputTokens}`,
|
|
335
|
-
options.maxTotalTokens === void 0 ? void 0 : `max_total_tokens = ${options.maxTotalTokens}`
|
|
336
|
-
].filter((line) => line !== void 0).join("\n")}\npush = false\n\n[timeout]\nrollout = ${options.rolloutTimeoutSeconds ?? DEFAULT_ROLLOUT_TIMEOUT}\nscoring = ${options.scoringTimeoutSeconds ?? DEFAULT_SCORING_TIMEOUT}\n\n[taskset]\nid = ${toml(options.name)}\nsplit = ${toml(split)}\n\n[harness]\nid = ${toml(options.name)}\nprogram = ${tomlArray(options.runner.command)}\nforward_env = ${tomlArray(options.runner.forwardEnv ?? [])}\n\n[harness.runtime]\ntype = "docker"\nimage = ${toml(options.runner.image)}\n`;
|
|
337
|
-
}
|
|
338
|
-
function renderInit(moduleName) {
|
|
339
|
-
return `from ${moduleName}.harness import TangleRuntimeHarness\nfrom ${moduleName}.taskset import TangleTaskset\n\n__all__ = ["TangleRuntimeHarness", "TangleTaskset"]\n`;
|
|
340
|
-
}
|
|
341
|
-
function renderTaskset(moduleName, scoring) {
|
|
342
|
-
const config = scoringConfig(scoring);
|
|
343
|
-
return `import asyncio\nimport json\nimport math\nimport os\nfrom importlib.resources import files\nfrom pathlib import Path\nfrom typing import Any, Literal\n\nimport verifiers.v1 as vf\n\nSCORING = json.loads(${pythonString(JSON.stringify(config))})\nPACKAGE_ROOT = Path(__file__).resolve().parent\n\n\nclass TangleTaskData(vf.TaskData):\n split: Literal["train", "eval"]\n answer: str | list[str] | None = None\n metadata: dict[str, Any] = {}\n\n\nclass TangleTaskConfig(vf.TaskConfig):\n scoring: Literal["exact", "reference-judge", "command"] = SCORING["kind"]\n normalization: Literal["none", "trim", "trim-casefold"] = SCORING.get("normalization", "trim")\n judge_model: str = SCORING.get("model", "openai/gpt-5.4-nano")\n judge_prompt: str | None = SCORING.get("prompt")\n judge_view: Literal["last_reply", "full_trace"] = SCORING.get("view", "last_reply")\n score_program: list[str] = SCORING.get("command", [])\n score_forward_env: list[str] = SCORING.get("forwardEnv", [])\n score_timeout_seconds: float = SCORING.get("timeoutSeconds", 300)\n\n\ndef _normalize(value: str, mode: str) -> str:\n if mode == "none":\n return value\n value = value.strip()\n return value.casefold() if mode == "trim-casefold" else value\n\n\nasync def _run_score_command(config: TangleTaskConfig, data: TangleTaskData, trace: vf.Trace) -> float:\n if not config.score_program:\n raise ValueError("command scoring requires score_program")\n safe_env = {\n key: os.environ[key]\n for key in ("PATH", "HOME", "TMPDIR", "LANG")\n if key in os.environ\n }\n safe_env.update(\n {key: os.environ[key] for key in config.score_forward_env if key in os.environ}\n )\n request = {\n "kind": "tangle.primeintellect.score",\n "task": data.model_dump(mode="json", exclude_none=True),\n "trace": trace.model_dump(mode="json", exclude_none=True),\n }\n process = await asyncio.create_subprocess_exec(\n *config.score_program,\n cwd=PACKAGE_ROOT,\n env=safe_env,\n stdin=asyncio.subprocess.PIPE,\n stdout=asyncio.subprocess.PIPE,\n stderr=asyncio.subprocess.PIPE,\n )\n payload = json.dumps(request, separators=(",", ":")).encode()\n try:\n stdout, stderr = await asyncio.wait_for(\n process.communicate(payload), timeout=config.score_timeout_seconds\n )\n except TimeoutError:\n process.kill()\n await process.communicate()\n raise RuntimeError(\n f"score command timed out after {config.score_timeout_seconds}s"\n )\n if process.returncode != 0:\n detail = (stderr or stdout).decode(errors="replace").strip()[-2000:]\n raise RuntimeError(f"score command exited {process.returncode}: {detail}")\n try:\n result = json.loads(stdout)\n except json.JSONDecodeError as error:\n raise ValueError(f"score command returned invalid JSON: {error}") from error\n if not isinstance(result, dict):\n raise ValueError("score command must return an object")\n reward = result.get("reward")\n if isinstance(reward, bool) or not isinstance(reward, (int, float)) or not math.isfinite(reward):\n raise ValueError("score command reward must be a finite number")\n metrics = result.get("metrics", {})\n if not isinstance(metrics, dict):\n raise ValueError("score command metrics must be an object")\n for name, value in metrics.items():\n if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value):\n raise ValueError(f"score command metric {name!r} must be a finite number")\n trace.record_metrics(metrics)\n return float(reward)\n\n\nclass TangleTask(vf.Task[TangleTaskData, vf.State, TangleTaskConfig]):\n @vf.reward(weight=1.0)\n async def task_reward(self, trace: vf.Trace) -> float:\n if self.config.scoring == "command":\n return await _run_score_command(self.config, self.data, trace)\n if self.data.answer is None:\n raise ValueError(f"task {self.data.name!r} has no reference answer")\n if self.config.scoring == "reference-judge":\n judge = vf.ReferenceJudge(\n vf.ReferenceJudgeConfig(\n model=self.config.judge_model,\n prompt=self.config.judge_prompt,\n view=self.config.judge_view,\n )\n )\n return await judge.score(self.data, trace)\n expected = self.data.answer if isinstance(self.data.answer, list) else [self.data.answer]\n actual = _normalize(trace.last_reply, self.config.normalization)\n return float(any(actual == _normalize(answer, self.config.normalization) for answer in expected))\n\n\nclass TangleTasksetConfig(vf.TasksetConfig):\n split: Literal["train", "eval"] = "eval"\n task: TangleTaskConfig = TangleTaskConfig()\n\n\nclass TangleTaskset(vf.Taskset[TangleTask, TangleTasksetConfig]):\n def load(self) -> list[TangleTask]:\n resource = files(${pythonString(moduleName)}).joinpath("tasks.jsonl")\n tasks: list[TangleTask] = []\n with resource.open("r", encoding="utf-8") as handle:\n for line in handle:\n if not line.strip():\n continue\n row = json.loads(line)\n if row["split"] != self.config.split:\n continue\n tasks.append(TangleTask(TangleTaskData.model_validate(row), self.config.task))\n if not tasks:\n raise ValueError(f"task split {self.config.split!r} is empty")\n return tasks\n\n\n__all__ = ["TangleTaskset"]\n`;
|
|
344
|
-
}
|
|
345
|
-
function renderHarness(moduleName) {
|
|
346
|
-
return `import json\nfrom importlib.resources import files\n\nimport verifiers.v1 as vf\n\nRUNNER = json.loads(files(${pythonString(moduleName)}).joinpath("runner.json").read_text(encoding="utf-8"))\n\n\nclass TangleRuntimeHarnessConfig(vf.HarnessConfig):\n program: list[str] = RUNNER["command"]\n setup_commands: list[list[str]] = RUNNER["setup"]\n forward_env: list[str] = RUNNER["forwardEnv"]\n\n\nclass TangleRuntimeHarness(vf.Harness[TangleRuntimeHarnessConfig]):\n APPENDS_SYSTEM_PROMPT = True\n SUPPORTS_MCP = True\n SUPPORTS_MESSAGE_PROMPT = True\n\n async def setup(self, runtime: vf.Runtime) -> None:\n for path, contents in RUNNER["files"].items():\n await runtime.write(path, contents.encode())\n for command in self.config.setup_commands:\n result = await runtime.run(command, self.config.resolved_env)\n if result.exit_code != 0:\n detail = (result.stderr or result.stdout).strip()[-2000:]\n raise RuntimeError(\n f"runner setup command {command[0]!r} exited {result.exit_code}: {detail}"\n )\n\n async def launch(\n self,\n ctx: vf.ModelContext,\n trace: vf.Trace,\n runtime: vf.Runtime,\n endpoint: str,\n secret: str,\n mcp_urls: dict[str, str],\n ) -> vf.ProgramResult:\n data = trace.task.data\n public_task = {\n "id": data.name or str(data.idx),\n "split": data.split,\n "prompt": data.prompt,\n "metadata": data.metadata,\n }\n if data.system_prompt is not None:\n public_task["systemPrompt"] = data.system_prompt\n env = {\n **self.config.resolved_env,\n "OPENAI_BASE_URL": endpoint,\n "OPENAI_API_KEY": secret,\n "OPENAI_MODEL": ctx.model,\n "TANGLE_PRIME_TASK_JSON": json.dumps(public_task, separators=(",", ":")),\n "TANGLE_PRIME_MCP_SERVERS_JSON": json.dumps(mcp_urls, separators=(",", ":")),\n }\n if not self.config.program:\n raise ValueError("Tangle runtime harness requires a program argv")\n return await runtime.run_program(self.config.program, env)\n\n\n__all__ = ["TangleRuntimeHarness"]\n`;
|
|
347
|
-
}
|
|
348
|
-
function scoringConfig(scoring) {
|
|
349
|
-
if (scoring.kind === "exact") return {
|
|
350
|
-
kind: scoring.kind,
|
|
351
|
-
normalization: scoring.normalization ?? "trim"
|
|
352
|
-
};
|
|
353
|
-
if (scoring.kind === "reference-judge") return {
|
|
354
|
-
kind: scoring.kind,
|
|
355
|
-
model: scoring.model,
|
|
356
|
-
prompt: scoring.prompt ?? null,
|
|
357
|
-
view: scoring.view ?? "last_reply"
|
|
358
|
-
};
|
|
359
|
-
return {
|
|
360
|
-
kind: scoring.kind,
|
|
361
|
-
command: [...scoring.command],
|
|
362
|
-
forwardEnv: [...scoring.forwardEnv ?? []],
|
|
363
|
-
timeoutSeconds: scoring.timeoutSeconds ?? DEFAULT_SCORING_TIMEOUT
|
|
364
|
-
};
|
|
365
|
-
}
|
|
366
|
-
function renderReadme(options) {
|
|
367
|
-
return `# ${options.name}\n\nPrimeIntellect Verifiers tasks that run the caller's Tangle agent program.\n\n## Evaluate\n\n\`\`\`bash\nuv run eval @ prime.eval.toml --model <provider/model-snapshot>\n\`\`\`\n\n## Train\n\nUse \`prime.train.toml\` as the environment config for Prime RL. The train and eval rows are disjoint and selected by \`taskset.split\`.\n\nThe runner receives only the prompt, metadata, intercepted model endpoint, and MCP URLs. Reference answers remain in the task process and are never written into the runner workspace or environment.\n`;
|
|
368
|
-
}
|
|
369
|
-
function toml(value) {
|
|
370
|
-
return JSON.stringify(value);
|
|
371
|
-
}
|
|
372
|
-
function tomlArray(values) {
|
|
373
|
-
return `[${values.map(toml).join(", ")}]`;
|
|
374
|
-
}
|
|
375
|
-
function pythonString(value) {
|
|
376
|
-
return JSON.stringify(value);
|
|
377
|
-
}
|
|
378
|
-
function positiveInteger(value, path) {
|
|
379
|
-
if (!Number.isSafeInteger(value) || value <= 0) throw new Error(`${path} must be a positive integer`);
|
|
380
|
-
}
|
|
381
|
-
function positiveNumber(value, path) {
|
|
382
|
-
if (!Number.isFinite(value) || value <= 0) throw new Error(`${path} must be positive`);
|
|
383
|
-
}
|
|
384
|
-
async function pathExists(path) {
|
|
385
|
-
try {
|
|
386
|
-
await stat(path);
|
|
387
|
-
return true;
|
|
388
|
-
} catch (error) {
|
|
389
|
-
if (error.code === "ENOENT") return false;
|
|
390
|
-
throw error;
|
|
391
|
-
}
|
|
392
|
-
}
|
|
393
|
-
async function assertGeneratedPackage(output) {
|
|
394
|
-
let manifest;
|
|
395
|
-
try {
|
|
396
|
-
manifest = JSON.parse(await readFile(join(output, "manifest.json"), "utf8"));
|
|
397
|
-
} catch (error) {
|
|
398
|
-
throw new Error(`refusing to replace ${output}: it is not a generated PrimeIntellect package (${error instanceof Error ? error.message : String(error)})`);
|
|
399
|
-
}
|
|
400
|
-
if (manifest === null || typeof manifest !== "object" || manifest.kind !== "tangle.primeintellect.package") throw new Error(`refusing to replace ${output}: manifest kind does not match`);
|
|
401
|
-
}
|
|
402
|
-
function sha256(value) {
|
|
403
|
-
return createHash("sha256").update(value, "utf8").digest("hex");
|
|
404
|
-
}
|
|
405
|
-
//#endregion
|
|
406
|
-
//#region src/primeintellect/runner.ts
|
|
407
|
-
const ENV = {
|
|
408
|
-
task: "TANGLE_PRIME_TASK_JSON",
|
|
409
|
-
model: "OPENAI_MODEL",
|
|
410
|
-
baseUrl: "OPENAI_BASE_URL",
|
|
411
|
-
apiKey: "OPENAI_API_KEY",
|
|
412
|
-
mcpServers: "TANGLE_PRIME_MCP_SERVERS_JSON"
|
|
413
|
-
};
|
|
414
|
-
/** Read and validate the private process contract installed by the generated Prime harness. */
|
|
415
|
-
function readPrimeIntellectEpisodeContext(env = process.env) {
|
|
416
|
-
const rawTask = requiredEnv(env, ENV.task);
|
|
417
|
-
const rawMcp = env[ENV.mcpServers] ?? "{}";
|
|
418
|
-
const parsedTask = parseJson(rawTask, ENV.task);
|
|
419
|
-
const parsedMcp = parseJson(rawMcp, ENV.mcpServers);
|
|
420
|
-
const task = validatePublicTask(parsedTask);
|
|
421
|
-
const mcpServers = validateStringMap(parsedMcp, ENV.mcpServers);
|
|
422
|
-
return {
|
|
423
|
-
task,
|
|
424
|
-
model: {
|
|
425
|
-
name: requiredEnv(env, ENV.model),
|
|
426
|
-
baseUrl: requiredEnv(env, ENV.baseUrl),
|
|
427
|
-
apiKey: requiredEnv(env, ENV.apiKey)
|
|
428
|
-
},
|
|
429
|
-
mcpServers
|
|
430
|
-
};
|
|
431
|
-
}
|
|
432
|
-
/** Resolve Prime's intercepted endpoint as transport-only Runtime executor configuration.
|
|
433
|
-
* The caller's exact `AgentProfile` remains the sole owner of model and behavior. */
|
|
434
|
-
function primeIntellectExecutorConfig(context) {
|
|
435
|
-
return Object.freeze({
|
|
436
|
-
backend: "router",
|
|
437
|
-
routerBaseUrl: context.model.baseUrl,
|
|
438
|
-
routerKey: context.model.apiKey
|
|
439
|
-
});
|
|
440
|
-
}
|
|
441
|
-
/**
|
|
442
|
-
* Execute the caller's canonical runtime program inside a Prime rollout.
|
|
443
|
-
* The callback may call runPersonified, runAgentic, runAgentRounds, or any product wrapper.
|
|
444
|
-
*/
|
|
445
|
-
async function runPrimeIntellectProgram(run, options = {}) {
|
|
446
|
-
return run(readPrimeIntellectEpisodeContext(options.env));
|
|
447
|
-
}
|
|
448
|
-
function requiredEnv(env, name) {
|
|
449
|
-
const value = env[name];
|
|
450
|
-
if (typeof value !== "string" || value.length === 0) throw new Error(`PrimeIntellect runner requires ${name}`);
|
|
451
|
-
return value;
|
|
452
|
-
}
|
|
453
|
-
function parseJson(value, name) {
|
|
454
|
-
try {
|
|
455
|
-
return JSON.parse(value);
|
|
456
|
-
} catch (error) {
|
|
457
|
-
throw new Error(`${name} must contain valid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
458
|
-
}
|
|
459
|
-
}
|
|
460
|
-
function validatePublicTask(value) {
|
|
461
|
-
const task = record(value, ENV.task);
|
|
462
|
-
for (const privateField of [
|
|
463
|
-
"answer",
|
|
464
|
-
"reference",
|
|
465
|
-
"scoring",
|
|
466
|
-
"score"
|
|
467
|
-
]) if (privateField in task) throw new Error(`${ENV.task} exposed private field ${privateField}`);
|
|
468
|
-
for (const field of Object.keys(task)) if (![
|
|
469
|
-
"id",
|
|
470
|
-
"split",
|
|
471
|
-
"prompt",
|
|
472
|
-
"systemPrompt",
|
|
473
|
-
"metadata"
|
|
474
|
-
].includes(field)) throw new Error(`${ENV.task}.${field} is not supported`);
|
|
475
|
-
const id = nonEmptyString$1(task.id, `${ENV.task}.id`);
|
|
476
|
-
const split = validateSplit(task.split, `${ENV.task}.split`);
|
|
477
|
-
const prompt = validatePrimeIntellectPrompt(task.prompt, `${ENV.task}.prompt`);
|
|
478
|
-
const systemPrompt = optionalString(task.systemPrompt, `${ENV.task}.systemPrompt`);
|
|
479
|
-
if (systemPrompt !== void 0 && Array.isArray(prompt) && prompt.some((message) => message.role === "system")) throw new Error(`${ENV.task} must not set systemPrompt and include a system message`);
|
|
480
|
-
const metadata = task.metadata === void 0 ? void 0 : validatePrimeIntellectJsonObject(task.metadata, `${ENV.task}.metadata`);
|
|
481
|
-
return {
|
|
482
|
-
id,
|
|
483
|
-
split,
|
|
484
|
-
prompt,
|
|
485
|
-
...systemPrompt !== void 0 ? { systemPrompt } : {},
|
|
486
|
-
...metadata !== void 0 ? { metadata } : {}
|
|
487
|
-
};
|
|
488
|
-
}
|
|
489
|
-
function validateSplit(value, path) {
|
|
490
|
-
if (value !== "train" && value !== "eval") throw new Error(`${path} must be train or eval`);
|
|
491
|
-
return value;
|
|
492
|
-
}
|
|
493
|
-
function validateStringMap(value, path) {
|
|
494
|
-
const input = record(value, path);
|
|
495
|
-
const output = {};
|
|
496
|
-
for (const [key, entry] of Object.entries(input)) {
|
|
497
|
-
if (typeof entry !== "string" || entry.length === 0) throw new Error(`${path}.${key} must be a non-empty string`);
|
|
498
|
-
output[key] = entry;
|
|
499
|
-
}
|
|
500
|
-
return output;
|
|
501
|
-
}
|
|
502
|
-
function record(value, path) {
|
|
503
|
-
if (value === null || typeof value !== "object" || Array.isArray(value)) throw new Error(`${path} must be an object`);
|
|
504
|
-
return value;
|
|
505
|
-
}
|
|
506
|
-
function nonEmptyString$1(value, path) {
|
|
507
|
-
if (typeof value !== "string" || value.length === 0) throw new Error(`${path} must be a non-empty string`);
|
|
508
|
-
return value;
|
|
509
|
-
}
|
|
510
|
-
function optionalString(value, path) {
|
|
511
|
-
if (value === void 0) return void 0;
|
|
512
|
-
if (typeof value !== "string") throw new Error(`${path} must be a string`);
|
|
513
|
-
return value;
|
|
514
|
-
}
|
|
515
|
-
//#endregion
|
|
516
|
-
//#region src/primeintellect/traces.ts
|
|
517
|
-
/** Parse Prime's durable `traces.jsonl` and reject malformed rows with a line number. */
|
|
518
|
-
function parsePrimeIntellectTraces(jsonl) {
|
|
519
|
-
const traces = [];
|
|
520
|
-
for (const [lineIndex, line] of jsonl.split("\n").entries()) {
|
|
521
|
-
if (!line.trim()) continue;
|
|
522
|
-
let parsed;
|
|
523
|
-
try {
|
|
524
|
-
parsed = JSON.parse(line);
|
|
525
|
-
} catch (error) {
|
|
526
|
-
throw new Error(`PrimeIntellect trace line ${lineIndex + 1} is invalid JSON: ${error instanceof Error ? error.message : String(error)}`);
|
|
527
|
-
}
|
|
528
|
-
traces.push(validateTrace(parsed, lineIndex + 1));
|
|
529
|
-
}
|
|
530
|
-
if (traces.length === 0) throw new Error("PrimeIntellect traces.jsonl contains no traces");
|
|
531
|
-
return traces;
|
|
532
|
-
}
|
|
533
|
-
/** Convert all Prime traces to agent-eval RunRecords while retaining one shared run config. */
|
|
534
|
-
function importPrimeIntellectTraces(jsonl, defaults) {
|
|
535
|
-
return parsePrimeIntellectTraces(jsonl).map((trace) => primeIntellectTraceToRunRecord(trace, defaults));
|
|
536
|
-
}
|
|
537
|
-
/** Project one complete Prime trace into the common agent-eval analysis row. */
|
|
538
|
-
function primeIntellectTraceToRunRecord(trace, options) {
|
|
539
|
-
const split = trace.task.data.split;
|
|
540
|
-
if (split !== "train" && split !== "eval") throw new Error(`PrimeIntellect trace ${trace.id} has no train/eval split`);
|
|
541
|
-
const reward = sumFinite(Object.values(trace.rewards), `trace ${trace.id} rewards`);
|
|
542
|
-
const usage = aggregateUsage([...trace.nodes.map((node) => node.usage).filter((value) => value != null), ...trace.extra_usage ?? []]);
|
|
543
|
-
const errors = trace.errors ?? [];
|
|
544
|
-
const raw = {
|
|
545
|
-
reward,
|
|
546
|
-
"prime.turns": trace.nodes.filter((node) => node.sampled === true).length,
|
|
547
|
-
"prime.branches": countBranches(trace.nodes),
|
|
548
|
-
"prime.errors": errors.length,
|
|
549
|
-
execution_error_count: errors.length,
|
|
550
|
-
"prime.completed": trace.is_completed ? 1 : 0,
|
|
551
|
-
"prime.cost_complete": usage.costComplete ? 1 : 0,
|
|
552
|
-
"prime.reported_cost_usd": usage.reportedCostUsd
|
|
553
|
-
};
|
|
554
|
-
for (const [name, value] of Object.entries(trace.rewards)) raw[`reward.${name}`] = finite(value, `trace ${trace.id} reward ${name}`);
|
|
555
|
-
for (const [name, value] of Object.entries(trace.metrics)) raw[`metric.${name}`] = finite(value, `trace ${trace.id} metric ${name}`);
|
|
556
|
-
return validateRunRecord({
|
|
557
|
-
runId: trace.id,
|
|
558
|
-
experimentId: options.experimentId,
|
|
559
|
-
candidateId: options.candidateId,
|
|
560
|
-
seed: options.seed,
|
|
561
|
-
model: options.model,
|
|
562
|
-
promptHash: options.promptHash,
|
|
563
|
-
configHash: options.configHash,
|
|
564
|
-
commitSha: options.commitSha,
|
|
565
|
-
wallMs: traceWallMs(trace),
|
|
566
|
-
costUsd: usage.costComplete ? usage.reportedCostUsd : null,
|
|
567
|
-
costProvenance: usage.costComplete ? {
|
|
568
|
-
kind: "observed",
|
|
569
|
-
usd: usage.reportedCostUsd
|
|
570
|
-
} : {
|
|
571
|
-
kind: "uncaptured",
|
|
572
|
-
usd: null
|
|
573
|
-
},
|
|
574
|
-
terminalOutcome: errors.length > 0 ? "failed" : trace.is_completed ? "succeeded" : "incomplete",
|
|
575
|
-
...errors[0] ? { terminalFailureReason: `${errors[0].type}:${errors[0].message}` } : {},
|
|
576
|
-
tokenUsage: {
|
|
577
|
-
input: usage.input,
|
|
578
|
-
output: usage.output,
|
|
579
|
-
...usage.reasoning !== void 0 ? { reasoning: usage.reasoning } : {},
|
|
580
|
-
...usage.cached !== void 0 ? { cached: usage.cached } : {}
|
|
581
|
-
},
|
|
582
|
-
outcome: split === "eval" ? {
|
|
583
|
-
holdoutScore: reward,
|
|
584
|
-
raw
|
|
585
|
-
} : {
|
|
586
|
-
searchScore: reward,
|
|
587
|
-
raw
|
|
588
|
-
},
|
|
589
|
-
splitTag: split === "eval" ? "holdout" : "search",
|
|
590
|
-
scenarioId: trace.task.data.name ?? String(trace.task.data.idx)
|
|
591
|
-
});
|
|
592
|
-
}
|
|
593
|
-
function validateTrace(value, line) {
|
|
594
|
-
const trace = object(value, `PrimeIntellect trace line ${line}`);
|
|
595
|
-
nonEmptyString(trace.id, `PrimeIntellect trace line ${line}.id`);
|
|
596
|
-
const task = object(trace.task, `PrimeIntellect trace line ${line}.task`);
|
|
597
|
-
nonEmptyString(task.type, `PrimeIntellect trace line ${line}.task.type`);
|
|
598
|
-
const data = object(task.data, `PrimeIntellect trace line ${line}.task.data`);
|
|
599
|
-
if (!Number.isSafeInteger(data.idx) || data.idx < 0) throw new Error(`PrimeIntellect trace line ${line}.task.data.idx must be a non-negative integer`);
|
|
600
|
-
if (!Array.isArray(trace.nodes)) throw new Error(`PrimeIntellect trace line ${line}.nodes must be an array`);
|
|
601
|
-
for (const [index, rawNode] of trace.nodes.entries()) {
|
|
602
|
-
const node = object(rawNode, `PrimeIntellect trace line ${line}.nodes[${index}]`);
|
|
603
|
-
const parent = node.parent;
|
|
604
|
-
if (parent !== void 0 && parent !== null && (!Number.isSafeInteger(parent) || parent < 0 || parent >= index)) throw new Error(`PrimeIntellect trace line ${line}.nodes[${index}].parent must reference an earlier node`);
|
|
605
|
-
if (node.sampled !== void 0 && typeof node.sampled !== "boolean") throw new Error(`PrimeIntellect trace line ${line}.nodes[${index}].sampled must be boolean`);
|
|
606
|
-
if (node.usage !== void 0 && node.usage !== null) validateUsage(node.usage, `PrimeIntellect trace line ${line}.nodes[${index}].usage`);
|
|
607
|
-
}
|
|
608
|
-
validateNumberMap(trace.rewards, `PrimeIntellect trace line ${line}.rewards`);
|
|
609
|
-
validateNumberMap(trace.metrics, `PrimeIntellect trace line ${line}.metrics`);
|
|
610
|
-
if (trace.extra_usage !== void 0 && !Array.isArray(trace.extra_usage)) throw new Error(`PrimeIntellect trace line ${line}.extra_usage must be an array`);
|
|
611
|
-
for (const [index, usage] of (trace.extra_usage ?? []).entries()) validateUsage(usage, `PrimeIntellect trace line ${line}.extra_usage[${index}]`);
|
|
612
|
-
if (trace.errors !== void 0 && !Array.isArray(trace.errors)) throw new Error(`PrimeIntellect trace line ${line}.errors must be an array`);
|
|
613
|
-
for (const [index, rawError] of (trace.errors ?? []).entries()) {
|
|
614
|
-
const error = object(rawError, `PrimeIntellect trace line ${line}.errors[${index}]`);
|
|
615
|
-
nonEmptyString(error.type, `PrimeIntellect trace line ${line}.errors[${index}].type`);
|
|
616
|
-
nonEmptyString(error.message, `PrimeIntellect trace line ${line}.errors[${index}].message`);
|
|
617
|
-
if (error.traceback !== void 0 && error.traceback !== null && typeof error.traceback !== "string") throw new Error(`PrimeIntellect trace line ${line}.errors[${index}].traceback must be a string or null`);
|
|
618
|
-
}
|
|
619
|
-
if (trace.is_completed !== void 0 && typeof trace.is_completed !== "boolean") throw new Error(`PrimeIntellect trace line ${line}.is_completed must be boolean`);
|
|
620
|
-
if (trace.stop_condition !== void 0 && trace.stop_condition !== null && typeof trace.stop_condition !== "string") throw new Error(`PrimeIntellect trace line ${line}.stop_condition must be a string or null`);
|
|
621
|
-
if (trace.timing !== void 0) validateTiming(trace.timing, line);
|
|
622
|
-
return value;
|
|
623
|
-
}
|
|
624
|
-
function validateTiming(value, line) {
|
|
625
|
-
const timing = object(value, `PrimeIntellect trace line ${line}.timing`);
|
|
626
|
-
optionalFinite(timing.start, `PrimeIntellect trace line ${line}.timing.start`);
|
|
627
|
-
for (const phase of [
|
|
628
|
-
"setup",
|
|
629
|
-
"generation",
|
|
630
|
-
"finalize",
|
|
631
|
-
"scoring"
|
|
632
|
-
]) {
|
|
633
|
-
if (timing[phase] === void 0) continue;
|
|
634
|
-
const span = object(timing[phase], `PrimeIntellect trace line ${line}.timing.${phase}`);
|
|
635
|
-
optionalFinite(span.start, `PrimeIntellect trace line ${line}.timing.${phase}.start`);
|
|
636
|
-
optionalFinite(span.end, `PrimeIntellect trace line ${line}.timing.${phase}.end`);
|
|
637
|
-
}
|
|
638
|
-
}
|
|
639
|
-
function aggregateUsage(usages) {
|
|
640
|
-
let input = 0;
|
|
641
|
-
let output = 0;
|
|
642
|
-
let reasoning = 0;
|
|
643
|
-
let cached = 0;
|
|
644
|
-
let costUsd = 0;
|
|
645
|
-
let sawReasoning = false;
|
|
646
|
-
let sawCached = false;
|
|
647
|
-
let costsReported = 0;
|
|
648
|
-
for (const [index, usage] of usages.entries()) {
|
|
649
|
-
const prompt = nonNegative(usage.prompt_tokens, `usage[${index}].prompt_tokens`);
|
|
650
|
-
const completion = nonNegative(usage.completion_tokens, `usage[${index}].completion_tokens`);
|
|
651
|
-
const cachedInput = optionalNonNegative(usage.cached_input_tokens, `usage[${index}].cached_input_tokens`);
|
|
652
|
-
const reasoningTokens = optionalNonNegative(usage.reasoning_tokens, `usage[${index}].reasoning_tokens`);
|
|
653
|
-
const cost = optionalNonNegative(usage.cost, `usage[${index}].cost`);
|
|
654
|
-
input += prompt + (cachedInput ?? 0);
|
|
655
|
-
output += completion;
|
|
656
|
-
if (cachedInput !== void 0) {
|
|
657
|
-
cached += cachedInput;
|
|
658
|
-
sawCached = true;
|
|
659
|
-
}
|
|
660
|
-
if (reasoningTokens !== void 0) {
|
|
661
|
-
reasoning += reasoningTokens;
|
|
662
|
-
sawReasoning = true;
|
|
663
|
-
}
|
|
664
|
-
if (cost !== void 0) {
|
|
665
|
-
costUsd += cost;
|
|
666
|
-
costsReported += 1;
|
|
667
|
-
}
|
|
668
|
-
}
|
|
669
|
-
if (reasoning > output) throw new Error(`PrimeIntellect reasoning token total ${reasoning} exceeds output total ${output}`);
|
|
670
|
-
return {
|
|
671
|
-
input,
|
|
672
|
-
output,
|
|
673
|
-
...sawReasoning ? { reasoning } : {},
|
|
674
|
-
...sawCached ? { cached } : {},
|
|
675
|
-
reportedCostUsd: costUsd,
|
|
676
|
-
costComplete: usages.length > 0 && costsReported === usages.length
|
|
677
|
-
};
|
|
678
|
-
}
|
|
679
|
-
function validateUsage(value, path) {
|
|
680
|
-
const usage = object(value, path);
|
|
681
|
-
nonNegative(usage.prompt_tokens, `${path}.prompt_tokens`);
|
|
682
|
-
nonNegative(usage.completion_tokens, `${path}.completion_tokens`);
|
|
683
|
-
optionalNonNegative(usage.cached_input_tokens, `${path}.cached_input_tokens`);
|
|
684
|
-
optionalNonNegative(usage.reasoning_tokens, `${path}.reasoning_tokens`);
|
|
685
|
-
optionalNonNegative(usage.cost, `${path}.cost`);
|
|
686
|
-
}
|
|
687
|
-
function traceWallMs(trace) {
|
|
688
|
-
const timing = trace.timing;
|
|
689
|
-
if (!timing || typeof timing.start !== "number" || !Number.isFinite(timing.start)) return 0;
|
|
690
|
-
const ends = [
|
|
691
|
-
timing.setup?.end,
|
|
692
|
-
timing.generation?.end,
|
|
693
|
-
timing.finalize?.end,
|
|
694
|
-
timing.scoring?.end
|
|
695
|
-
].filter((value) => typeof value === "number" && Number.isFinite(value));
|
|
696
|
-
if (ends.length === 0) return 0;
|
|
697
|
-
return Math.max(0, (Math.max(...ends) - timing.start) * 1e3);
|
|
698
|
-
}
|
|
699
|
-
function countBranches(nodes) {
|
|
700
|
-
if (nodes.length === 0) return 0;
|
|
701
|
-
const parents = new Set(nodes.map((node) => node.parent).filter((parent) => Number.isSafeInteger(parent) && parent >= 0));
|
|
702
|
-
return nodes.reduce((count, _node, index) => count + (parents.has(index) ? 0 : 1), 0);
|
|
703
|
-
}
|
|
704
|
-
function validateNumberMap(value, path) {
|
|
705
|
-
const map = object(value, path);
|
|
706
|
-
for (const [key, entry] of Object.entries(map)) finite(entry, `${path}.${key}`);
|
|
707
|
-
}
|
|
708
|
-
function sumFinite(values, path) {
|
|
709
|
-
return values.reduce((sum, value, index) => sum + finite(value, `${path}[${index}]`), 0);
|
|
710
|
-
}
|
|
711
|
-
function finite(value, path) {
|
|
712
|
-
if (typeof value !== "number" || !Number.isFinite(value)) throw new Error(`${path} must be a finite number`);
|
|
713
|
-
return value;
|
|
714
|
-
}
|
|
715
|
-
function nonNegative(value, path) {
|
|
716
|
-
const parsed = finite(value, path);
|
|
717
|
-
if (parsed < 0) throw new Error(`${path} must be non-negative`);
|
|
718
|
-
return parsed;
|
|
719
|
-
}
|
|
720
|
-
function optionalNonNegative(value, path) {
|
|
721
|
-
if (value === void 0 || value === null) return void 0;
|
|
722
|
-
return nonNegative(value, path);
|
|
723
|
-
}
|
|
724
|
-
function optionalFinite(value, path) {
|
|
725
|
-
if (value === void 0 || value === null) return void 0;
|
|
726
|
-
return finite(value, path);
|
|
727
|
-
}
|
|
728
|
-
function object(value, path) {
|
|
729
|
-
if (value === null || typeof value !== "object" || Array.isArray(value)) throw new Error(`${path} must be an object`);
|
|
730
|
-
return value;
|
|
731
|
-
}
|
|
732
|
-
function nonEmptyString(value, path) {
|
|
733
|
-
if (typeof value !== "string" || value.length === 0) throw new Error(`${path} must be a non-empty string`);
|
|
734
|
-
return value;
|
|
735
|
-
}
|
|
736
|
-
//#endregion
|
|
737
|
-
export { createPrimeIntellectPackage, importPrimeIntellectTraces, parsePrimeIntellectTraces, primeIntellectExecutorConfig, primeIntellectTraceToRunRecord, readPrimeIntellectEpisodeContext, runPrimeIntellectProgram, writePrimeIntellectPackage };
|
|
738
|
-
|
|
739
|
-
//# sourceMappingURL=index.js.map
|