premanmcp 1.1.7 → 1.1.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/agent.js +315 -36
- package/bin/desktop.js +2 -2
- package/bin/detect.js +29 -0
- package/bin/eval.js +430 -22
- package/bin/eval_entry.js +762 -0
- package/bin/eval_harness.py +166 -5
- package/bin/eval_runtime.js +156 -0
- package/bin/eval_target.js +160 -113
- package/bin/eval_templates.js +684 -0
- package/bin/runner.js +42 -3
- package/bin/shared.js +2 -1
- package/bin/target.js +105 -3
- package/package.json +2 -2
package/bin/eval.js
CHANGED
|
@@ -44,7 +44,10 @@ import os from "node:os";
|
|
|
44
44
|
import path from "node:path";
|
|
45
45
|
import { fileURLToPath } from "node:url";
|
|
46
46
|
|
|
47
|
+
import { download, progressLine } from "./desktop.js";
|
|
48
|
+
import { repoEnv } from "./detect.js";
|
|
47
49
|
import { adapterStats, discoverTarget, startAdapter, stepFeed } from "./eval_target.js";
|
|
50
|
+
import { ensureRuntime } from "./eval_runtime.js";
|
|
48
51
|
import { formatSurfaceLink, resolveEvalSurface } from "./link.js";
|
|
49
52
|
import {
|
|
50
53
|
backendUrl,
|
|
@@ -278,6 +281,26 @@ export function materialize(job, root) {
|
|
|
278
281
|
* prefix has to be produced by executing the snippet, so echoing it back does
|
|
279
282
|
* not satisfy the check.
|
|
280
283
|
*/
|
|
284
|
+
/**
|
|
285
|
+
* The harness, named so that a resolver can fetch it alongside a customer's
|
|
286
|
+
* own pins.
|
|
287
|
+
*
|
|
288
|
+
* A git reference rather than a package name: PyPI serves upstream ASSERT, and
|
|
289
|
+
* PreMan's fork carries functional changes, so a customer resolving the public
|
|
290
|
+
* name would run differently-shaped code and nothing records which harness
|
|
291
|
+
* produced a result. Pinned to a commit for the same reason `pyproject.toml`
|
|
292
|
+
* pins one -- two machines must agree about what measured them.
|
|
293
|
+
*
|
|
294
|
+
* Overridable so a machine with no route to GitHub, or a contributor testing a
|
|
295
|
+
* local checkout, is not stuck.
|
|
296
|
+
*/
|
|
297
|
+
export const HARNESS_SPEC =
|
|
298
|
+
String(process.env.PREMAN_EVAL_HARNESS_SPEC || "").trim() ||
|
|
299
|
+
"git+https://github.com/AleWang16/ASSERT@14f4e2136ab753897ce4cb90613e1663730aeb70";
|
|
300
|
+
|
|
301
|
+
/** How long a first resolve may take before it counts as a failure. */
|
|
302
|
+
const COLD_RESOLVE_MS = 600_000;
|
|
303
|
+
|
|
281
304
|
const PROBE_MARKER = "preman-assert-ai=";
|
|
282
305
|
const PROBE_SNIPPET = `import assert_ai,importlib.metadata as m;print("${PROBE_MARKER}"+m.version("assert-ai"))`;
|
|
283
306
|
|
|
@@ -289,11 +312,11 @@ const PROBE_SNIPPET = `import assert_ai,importlib.metadata as m;print("${PROBE_M
|
|
|
289
312
|
* about to use" are different questions, and only the second one predicts
|
|
290
313
|
* whether the run works.
|
|
291
314
|
*/
|
|
292
|
-
function probeInterpreter(command, argv) {
|
|
315
|
+
function probeInterpreter(command, argv, { timeoutMs = 120_000 } = {}) {
|
|
293
316
|
const probe = spawnSync(command, [...argv, "-c", PROBE_SNIPPET], {
|
|
294
317
|
encoding: "utf8",
|
|
295
318
|
stdio: ["ignore", "pipe", "ignore"],
|
|
296
|
-
timeout:
|
|
319
|
+
timeout: timeoutMs,
|
|
297
320
|
});
|
|
298
321
|
if (probe.status !== 0) return "";
|
|
299
322
|
const line = String(probe.stdout || "")
|
|
@@ -317,11 +340,23 @@ function probeInterpreter(command, argv) {
|
|
|
317
340
|
* Then a bare `python3`, which is the case where somebody installed the harness
|
|
318
341
|
* globally.
|
|
319
342
|
*
|
|
320
|
-
*
|
|
321
|
-
*
|
|
322
|
-
*
|
|
343
|
+
* `runtime` is PreMan's own interpreter, already fetched, and it is ranked
|
|
344
|
+
* second rather than first for a reason worth stating: it is the right answer
|
|
345
|
+
* for the harness and the wrong one for the customer's agent. A `callable`
|
|
346
|
+
* target has the harness import their agent in-process, and a bundle that
|
|
347
|
+
* cannot import their LangGraph measures nothing. So the *caller* decides
|
|
348
|
+
* whether to offer it — `executeEvalRun` passes `""` for a callable target —
|
|
349
|
+
* and this function only ranks what it is given. See
|
|
350
|
+
* docs/EVAL_SELF_CONTAINED_RUNTIME.md.
|
|
351
|
+
*
|
|
352
|
+
* The harness is never added to a customer's declared dependencies. The runtime
|
|
353
|
+
* lives under `~/.preman` and touches nothing of theirs. The harness-alongside
|
|
354
|
+
* branch is the one exception worth stating plainly: on a project that has
|
|
355
|
+
* never been synced it creates `.venv/` and `uv.lock`, exactly as the
|
|
356
|
+
* customer's own `uv run` would, and their lock records their packages only --
|
|
357
|
+
* measured, not assumed. Nothing pins PreMan into their project.
|
|
323
358
|
*/
|
|
324
|
-
export function resolveInterpreter({ projectPath = process.cwd(), log = () => {} } = {}) {
|
|
359
|
+
export function resolveInterpreter({ projectPath = process.cwd(), log = () => {}, runtime = "" } = {}) {
|
|
325
360
|
const override = String(process.env.PREMAN_EVAL_PYTHON || "").trim();
|
|
326
361
|
if (override) {
|
|
327
362
|
const version = probeInterpreter(override, []);
|
|
@@ -333,6 +368,14 @@ export function resolveInterpreter({ projectPath = process.cwd(), log = () => {}
|
|
|
333
368
|
return { command: override, argv: [], version, how: "PREMAN_EVAL_PYTHON" };
|
|
334
369
|
}
|
|
335
370
|
|
|
371
|
+
// Above the project check on purpose: that check keys on whatever directory
|
|
372
|
+
// the command was run from, which is how a run in `~/codex` picked up a conda
|
|
373
|
+
// interpreter that imported assert_ai and then produced nothing.
|
|
374
|
+
if (runtime) {
|
|
375
|
+
const version = probeInterpreter(runtime, []);
|
|
376
|
+
if (version) return { command: runtime, argv: [], version, how: "preman runtime" };
|
|
377
|
+
}
|
|
378
|
+
|
|
336
379
|
if (existsSync(path.join(projectPath, "pyproject.toml"))) {
|
|
337
380
|
const version = probeInterpreter("uv", ["run", "--project", projectPath, "python"]);
|
|
338
381
|
if (version) {
|
|
@@ -343,6 +386,40 @@ export function resolveInterpreter({ projectPath = process.cwd(), log = () => {}
|
|
|
343
386
|
how: "uv run in this project",
|
|
344
387
|
};
|
|
345
388
|
}
|
|
389
|
+
|
|
390
|
+
// The same project, with the harness resolved alongside it into an
|
|
391
|
+
// environment uv owns rather than one the customer does. Ranked below the
|
|
392
|
+
// branch above because a repository that pins the harness itself should
|
|
393
|
+
// keep its own pin; this is for the repository that has never heard of
|
|
394
|
+
// PreMan, which is every customer's.
|
|
395
|
+
//
|
|
396
|
+
// This is the only branch that can serve a `callable` target in a stranger's
|
|
397
|
+
// repository: their agent is imported in-process, so their dependencies and
|
|
398
|
+
// the harness have to be reachable from one interpreter, and PreMan's own
|
|
399
|
+
// bundle has none of theirs.
|
|
400
|
+
//
|
|
401
|
+
// What this does leave behind, measured on a repository that had never been
|
|
402
|
+
// synced: `.venv/` and `uv.lock` in their project -- the same two files
|
|
403
|
+
// their own `uv run` would have created. What it does *not* leave behind is
|
|
404
|
+
// any trace of the harness in `uv.lock`: the lock carried their 18 packages
|
|
405
|
+
// and no mention of assert-ai, so `--with` does not pin PreMan into a
|
|
406
|
+
// customer's project. That is the property worth protecting if this branch
|
|
407
|
+
// is ever rewritten.
|
|
408
|
+
//
|
|
409
|
+
// Given a long timeout on purpose. The first resolve on a cold machine
|
|
410
|
+
// fetches the harness's whole dependency tree, and treating a slow network
|
|
411
|
+
// as "this interpreter cannot import assert_ai" would send somebody looking
|
|
412
|
+
// for a missing package.
|
|
413
|
+
const added = ["run", "--project", projectPath, "--with", HARNESS_SPEC, "python"];
|
|
414
|
+
const resolved = probeInterpreter("uv", added, { timeoutMs: COLD_RESOLVE_MS });
|
|
415
|
+
if (resolved) {
|
|
416
|
+
return {
|
|
417
|
+
command: "uv",
|
|
418
|
+
argv: added,
|
|
419
|
+
version: resolved,
|
|
420
|
+
how: "uv run in this project, with PreMan's harness alongside it",
|
|
421
|
+
};
|
|
422
|
+
}
|
|
346
423
|
}
|
|
347
424
|
|
|
348
425
|
const version = probeInterpreter("python3", []);
|
|
@@ -413,10 +490,20 @@ export function missingModelKey(job, env = process.env) {
|
|
|
413
490
|
if (!name.startsWith(prefix) || seen.has(variable)) continue;
|
|
414
491
|
seen.add(variable);
|
|
415
492
|
if (env[variable] || supplied[variable]) continue;
|
|
493
|
+
// Says whose key this is, because the old wording ("needs a model key
|
|
494
|
+
// for X and there isn't one") reads as a configuration slip the reader
|
|
495
|
+
// made. It is not: PreMan's own stages -- deriving the failure
|
|
496
|
+
// categories, writing the cases, judging the transcripts -- run on this
|
|
497
|
+
// provider's models whatever the agent under test is written against. A
|
|
498
|
+
// customer whose agent uses a different provider would otherwise go
|
|
499
|
+
// looking for a key they were never asked for and conclude the tool is
|
|
500
|
+
// broken.
|
|
416
501
|
return (
|
|
417
|
-
`
|
|
418
|
-
`
|
|
419
|
-
`
|
|
502
|
+
`PreMan's own evaluation stages run on ${PROVIDER_NAMES[prefix]} models, so a key ` +
|
|
503
|
+
`for ${PROVIDER_NAMES[prefix]} is needed even when your agent uses a different ` +
|
|
504
|
+
`provider. This is separate from whatever key your agent itself needs, which stays ` +
|
|
505
|
+
`on this machine. Save one under Settings → Model keys and every machine on this ` +
|
|
506
|
+
`account picks it up, or export ${variable} in this terminal for a one-off.`
|
|
420
507
|
);
|
|
421
508
|
}
|
|
422
509
|
}
|
|
@@ -532,6 +619,11 @@ export const RUN_ARTIFACTS = [
|
|
|
532
619
|
"scores.jsonl",
|
|
533
620
|
"metrics.json",
|
|
534
621
|
"artifacts.json",
|
|
622
|
+
// Which bank case each row answers, when this run was handed a subset. The
|
|
623
|
+
// server writes it for a run it executes itself; listed here so the two paths
|
|
624
|
+
// keep publishing the same names, which is what lets one dashboard read a run
|
|
625
|
+
// without caring where it ran.
|
|
626
|
+
"case_map.json",
|
|
535
627
|
];
|
|
536
628
|
|
|
537
629
|
/**
|
|
@@ -705,6 +797,138 @@ export function statsSince(baseline, stats) {
|
|
|
705
797
|
};
|
|
706
798
|
}
|
|
707
799
|
|
|
800
|
+
/**
|
|
801
|
+
* Did this run write anything down at all?
|
|
802
|
+
*
|
|
803
|
+
* The most dangerous shape a run can take, and the one that cost this project
|
|
804
|
+
* an afternoon: the harness exits 0, writes no metrics and no transcript, and
|
|
805
|
+
* every layer above reports success. On disk the suite directory was empty; in
|
|
806
|
+
* the database the row said `succeeded` with a `{}` summary after 5m38s. A
|
|
807
|
+
* broken interpreter and a working one are indistinguishable from the outside,
|
|
808
|
+
* which is the worst property a measurement tool can have.
|
|
809
|
+
*
|
|
810
|
+
* Deliberately not keyed on the adapter's counters, and that is the whole point
|
|
811
|
+
* of this function existing separately from the two guards above it. Those read
|
|
812
|
+
* `targetStats`, which nothing increments on a `callable` target -- the harness
|
|
813
|
+
* imports the agent in-process and no request ever arrives at an adapter -- so
|
|
814
|
+
* on the default target kind they are not unreliable, they are inert. A
|
|
815
|
+
* counter-based check would be correct over HTTP and dead where it is needed
|
|
816
|
+
* most. An empty summary and an empty artifact directory are facts about the
|
|
817
|
+
* run on every path, including a future one that splits the harness and the
|
|
818
|
+
* agent into two processes.
|
|
819
|
+
*
|
|
820
|
+
* Any one file with bytes in it is enough. A partial run is a finding worth
|
|
821
|
+
* keeping; only nothing at all is a failure to measure.
|
|
822
|
+
*/
|
|
823
|
+
/**
|
|
824
|
+
* The module name the generated wrapper is always written under.
|
|
825
|
+
*
|
|
826
|
+
* Fixed rather than derived: the server writes `DEVICE_CALLABLE_REF` itself and
|
|
827
|
+
* interpolates nothing a caller sent, which is what makes importing it by name
|
|
828
|
+
* on a customer's machine safe. If that constant ever moves, this has to move
|
|
829
|
+
* with it or the check below silently stops checking.
|
|
830
|
+
*/
|
|
831
|
+
const CALLABLE_MODULE = "preman_agent_entry";
|
|
832
|
+
|
|
833
|
+
/**
|
|
834
|
+
* Does the approved wrapper still import, in the interpreter this run will use?
|
|
835
|
+
*
|
|
836
|
+
* Returns "" when it does, a short reason when it does not. An approved entry
|
|
837
|
+
* point is trusted on a digest, which proves the file has not changed and says
|
|
838
|
+
* nothing about whether it still runs. The two come apart exactly where it
|
|
839
|
+
* matters -- a rebuilt virtual environment, a dependency that went away, a
|
|
840
|
+
* different interpreter resolved than the one it was approved against -- and
|
|
841
|
+
* the first time that happened the run reached `inference`, spent model calls,
|
|
842
|
+
* and reported a missing attribute on a function that plainly existed.
|
|
843
|
+
*
|
|
844
|
+
* Checked here rather than at discovery because the interpreter is already
|
|
845
|
+
* resolved by this point. Doing it earlier would mean resolving one twice, and
|
|
846
|
+
* on the harness-alongside branch a resolve is seconds warm and minutes cold.
|
|
847
|
+
*
|
|
848
|
+
* A load and not a turn. Making the agent answer is what the approval gate
|
|
849
|
+
* does, and it costs a real model call; paying that on every run would tax
|
|
850
|
+
* every suite for a check that only earns its keep when something drifted. It
|
|
851
|
+
* is not free either -- `preman_selfcheck` loads the customer's module, which
|
|
852
|
+
* for a framework repository means importing crewai or langchain -- but only
|
|
853
|
+
* this probe pays it, where the gate's version bills a model.
|
|
854
|
+
*
|
|
855
|
+
* Why a named function rather than importing the wrapper and letting its own
|
|
856
|
+
* side effects do the work: the template loads the agent lazily on purpose,
|
|
857
|
+
* because a customer's module can bind a port or open a connection at import
|
|
858
|
+
* and a probe should not be the thing that triggers it. The two designs also
|
|
859
|
+
* fail differently, which is the deciding reason. A template that stops
|
|
860
|
+
* defining `preman_selfcheck` makes this skip *and say so*; a template that
|
|
861
|
+
* stopped loading eagerly would make it import cleanly and pass -- a false
|
|
862
|
+
* negative with no signal, which is the shape of the bug this exists to catch.
|
|
863
|
+
*
|
|
864
|
+
* A wrapper generated before the contract existed has no `preman_selfcheck`,
|
|
865
|
+
* so it is skipped rather than failed. Those are only re-checked once they are
|
|
866
|
+
* regenerated; the hole is closed going forward, not retroactively.
|
|
867
|
+
*/
|
|
868
|
+
export function entryImportFailure(python, sysPath, { timeoutMs = 120_000, log = null } = {}) {
|
|
869
|
+
if (!python?.command || !sysPath) return "";
|
|
870
|
+
|
|
871
|
+
const snippet = [
|
|
872
|
+
`import sys`,
|
|
873
|
+
`sys.path.insert(0,${JSON.stringify(sysPath)})`,
|
|
874
|
+
`import ${CALLABLE_MODULE} as _w`,
|
|
875
|
+
`_c=getattr(_w,"preman_selfcheck",None)`,
|
|
876
|
+
`print("preman-selfcheck=absent") if _c is None else (_c(),print("preman-selfcheck=ok"))`,
|
|
877
|
+
].join(";");
|
|
878
|
+
|
|
879
|
+
const probe = spawnSync(python.command, [...(python.argv || []), "-c", snippet], {
|
|
880
|
+
encoding: "utf8",
|
|
881
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
882
|
+
timeout: timeoutMs,
|
|
883
|
+
});
|
|
884
|
+
|
|
885
|
+
// Killed, or never started. `spawnSync` reports both as `status: null` with no
|
|
886
|
+
// stderr, and the fallback below would turn that into "exit null" and block
|
|
887
|
+
// the run -- a check that could not finish, reported as proof of breakage.
|
|
888
|
+
// The timeout is the one that can really happen: `preman_selfcheck` imports
|
|
889
|
+
// the customer's agent, and importing a framework is slow.
|
|
890
|
+
if (probe.error || probe.status === null) {
|
|
891
|
+
const why = probe.error?.message || `it did not finish within ${Math.round(timeoutMs / 1000)}s`;
|
|
892
|
+
log?.(
|
|
893
|
+
`whether the approved entry point still loads could not be checked (${why}); ` +
|
|
894
|
+
`running anyway rather than refusing on a check that did not complete`
|
|
895
|
+
);
|
|
896
|
+
return "";
|
|
897
|
+
}
|
|
898
|
+
|
|
899
|
+
if (probe.status === 0) {
|
|
900
|
+
if (String(probe.stdout || "").includes("preman-selfcheck=absent")) {
|
|
901
|
+
log?.(
|
|
902
|
+
`the approved entry point predates PreMan's self-check, so whether it still loads ` +
|
|
903
|
+
`was not verified; regenerate it with \`${cliInvocation()} eval target\` to have it checked`
|
|
904
|
+
);
|
|
905
|
+
}
|
|
906
|
+
return "";
|
|
907
|
+
}
|
|
908
|
+
|
|
909
|
+
const lines = String(probe.stderr || "").trim().split("\n").filter(Boolean);
|
|
910
|
+
return lines[lines.length - 1]?.slice(0, 300) || `exit ${probe.status}`;
|
|
911
|
+
}
|
|
912
|
+
|
|
913
|
+
export function producedNothing(dir, summary) {
|
|
914
|
+
if (summary && typeof summary === "object" && Object.keys(summary).length) return "";
|
|
915
|
+
|
|
916
|
+
const wrote = ["metrics.json", "scores.jsonl", "inference_set.jsonl"].some((name) => {
|
|
917
|
+
try {
|
|
918
|
+
return statSync(path.join(dir, name)).size > 0;
|
|
919
|
+
} catch {
|
|
920
|
+
return false;
|
|
921
|
+
}
|
|
922
|
+
});
|
|
923
|
+
if (wrote) return "";
|
|
924
|
+
|
|
925
|
+
return (
|
|
926
|
+
"the harness exited cleanly having written no metrics and no transcript, so there is " +
|
|
927
|
+
"nothing here that was measured. The usual cause is an interpreter that can import " +
|
|
928
|
+
"assert_ai but cannot complete a run; `eval doctor` reports which one this machine picks."
|
|
929
|
+
);
|
|
930
|
+
}
|
|
931
|
+
|
|
708
932
|
export function missingToolEvidence(dir, stats) {
|
|
709
933
|
const forwarded = Number(stats?.toolCalls);
|
|
710
934
|
if (!Number.isFinite(forwarded) || forwarded <= 0) return "";
|
|
@@ -945,7 +1169,15 @@ export async function executeEvalRun(
|
|
|
945
1169
|
home = suiteHome(job);
|
|
946
1170
|
pruneRuns(home, job);
|
|
947
1171
|
laid = materialize(job, home);
|
|
948
|
-
|
|
1172
|
+
// Offered only when the harness is the only thing that has to import
|
|
1173
|
+
// anything. A `callable` target imports the customer's agent in-process, so
|
|
1174
|
+
// it must resolve against an environment that has *their* dependencies —
|
|
1175
|
+
// PreMan's bundle has assert-ai and nothing of theirs.
|
|
1176
|
+
const runtime =
|
|
1177
|
+
String(job.target_kind || "") === "callable"
|
|
1178
|
+
? ""
|
|
1179
|
+
: await ensureRuntime({ log, download, progress: progressLine });
|
|
1180
|
+
python = resolveInterpreter({ projectPath, log, runtime });
|
|
949
1181
|
log(`eval ${job.id}: assert-ai ${python.version} via ${python.how}`);
|
|
950
1182
|
} catch (error) {
|
|
951
1183
|
// Every one of these is a fact about this machine or this frame, and no
|
|
@@ -960,7 +1192,26 @@ export async function executeEvalRun(
|
|
|
960
1192
|
return finish(false, { error: `the eval harness is missing from this install (${HARNESS_PATH})` });
|
|
961
1193
|
}
|
|
962
1194
|
|
|
963
|
-
|
|
1195
|
+
// Before the harness, because everything after this point costs model calls.
|
|
1196
|
+
if (String(job.target_kind || "") === "callable") {
|
|
1197
|
+
const broken = entryImportFailure(python, process.env.PREMAN_EVAL_SYS_PATH || "", {
|
|
1198
|
+
log: (line) => log(`eval ${job.id}: ${line}`),
|
|
1199
|
+
});
|
|
1200
|
+
if (broken) {
|
|
1201
|
+
tidy(home);
|
|
1202
|
+
return finish(false, {
|
|
1203
|
+
error:
|
|
1204
|
+
`the approved entry point no longer loads in this environment: ${broken}. It was ` +
|
|
1205
|
+
`trusted on its digest, which only proves the file is unchanged -- re-run ` +
|
|
1206
|
+
`\`${cliInvocation()} eval target\` here to regenerate and re-approve it.`,
|
|
1207
|
+
});
|
|
1208
|
+
}
|
|
1209
|
+
}
|
|
1210
|
+
|
|
1211
|
+
// The environment the harness will actually get, so the check cannot pass
|
|
1212
|
+
// where the run fails or refuse where the key was in the repository all along.
|
|
1213
|
+
const runEnv = childEnv({}, { ...(job.env || {}), ...repoEnv(projectPath) });
|
|
1214
|
+
const missingKey = missingModelKey(job, runEnv);
|
|
964
1215
|
if (missingKey) {
|
|
965
1216
|
// Checked before the stages start rather than left to the provider client,
|
|
966
1217
|
// which reports it from inside `inference` as an authentication error
|
|
@@ -987,7 +1238,7 @@ export async function executeEvalRun(
|
|
|
987
1238
|
const timeout = Math.min(RUN_TIMEOUT_MS, RUN_TIMEOUT_CAP_MS);
|
|
988
1239
|
const child = spawn(python.command, [...python.argv, HARNESS_PATH, "run", "--config", laid.configPath], {
|
|
989
1240
|
cwd: laid.cwd,
|
|
990
|
-
env: childEnv({ PREMAN_EVAL_RUN: `${job.suite}/${job.run}` }, job.env || {}),
|
|
1241
|
+
env: childEnv({ PREMAN_EVAL_RUN: `${job.suite}/${job.run}` }, { ...(job.env || {}), ...repoEnv(projectPath) }),
|
|
991
1242
|
stdio: ["ignore", "ignore", "pipe"],
|
|
992
1243
|
});
|
|
993
1244
|
|
|
@@ -1135,15 +1386,45 @@ export async function executeEvalRun(
|
|
|
1135
1386
|
return { ...outcome, summary, artifacts };
|
|
1136
1387
|
}
|
|
1137
1388
|
|
|
1389
|
+
// Last, because the two guards above name a cause and this one only reports
|
|
1390
|
+
// that there is nothing to report. Still ahead of `finish(true, ...)`,
|
|
1391
|
+
// because a run with no results is not a passing run.
|
|
1392
|
+
const empty = producedNothing(artifacts, summary);
|
|
1393
|
+
if (empty) {
|
|
1394
|
+
const outcome = await finish(false, { summary, error: empty, exitCode });
|
|
1395
|
+
tidy(home);
|
|
1396
|
+
return { ...outcome, summary, artifacts };
|
|
1397
|
+
}
|
|
1398
|
+
|
|
1138
1399
|
const outcome = await finish(true, { summary, exitCode });
|
|
1139
1400
|
tidy(home);
|
|
1140
1401
|
return { ...outcome, summary, artifacts };
|
|
1141
1402
|
}
|
|
1142
1403
|
|
|
1404
|
+
// The harness's own account of the failure, kept on disk before it is reduced
|
|
1405
|
+
// to something that fits in a row.
|
|
1406
|
+
//
|
|
1407
|
+
// The tail is eight lines because that is what a terminal and a database
|
|
1408
|
+
// column can carry, and the first time an inference failure had to be
|
|
1409
|
+
// diagnosed the rest had already been thrown away: the tail named a symptom
|
|
1410
|
+
// -- a wrapper reporting `has no attribute` on a function that plainly
|
|
1411
|
+
// existed -- while the exception that said *why* was the one before it, and
|
|
1412
|
+
// gone. Written beside the artifacts, which `tidy` leaves in place, so the
|
|
1413
|
+
// account outlives the process that produced it.
|
|
1414
|
+
const failureLog = path.join(laid.cwd, "last-failure.log");
|
|
1415
|
+
let kept = "";
|
|
1416
|
+
try {
|
|
1417
|
+
writeFileSync(failureLog, stderr, { mode: 0o600 });
|
|
1418
|
+
kept = ` (full output: ${failureLog})`;
|
|
1419
|
+
} catch {
|
|
1420
|
+
// A run that cannot write its own post-mortem still has to report the
|
|
1421
|
+
// failure it was handed.
|
|
1422
|
+
}
|
|
1423
|
+
|
|
1143
1424
|
const tail = stderr.trim().split("\n").slice(-8).join(" / ");
|
|
1144
1425
|
const outcome = await finish(false, {
|
|
1145
1426
|
summary,
|
|
1146
|
-
error: `the eval harness exited ${exitCode}: ${tail}`,
|
|
1427
|
+
error: `the eval harness exited ${exitCode}: ${tail}${kept}`,
|
|
1147
1428
|
exitCode,
|
|
1148
1429
|
});
|
|
1149
1430
|
tidy(home);
|
|
@@ -1277,9 +1558,15 @@ Eval options:
|
|
|
1277
1558
|
token names
|
|
1278
1559
|
eval run Pair this project and run the eval PreMan has
|
|
1279
1560
|
queued for it, then exit
|
|
1561
|
+
eval target Find the code in this repository that runs your
|
|
1562
|
+
agent, write a wrapper the eval can call, and
|
|
1563
|
+
remember it once you approve it
|
|
1280
1564
|
eval doctor Report whether this machine can run an eval
|
|
1565
|
+
--entry <file>:<symbol> Skip the search; this is the agent. Still writes
|
|
1566
|
+
a wrapper and still asks you to approve it
|
|
1281
1567
|
--target <url> Skip discovery; this URL is the agent. Must take
|
|
1282
1568
|
POST {"message","history"} and answer {"response"}
|
|
1569
|
+
--preman-agent <url> Evaluate PreMan's own chat agent at this address
|
|
1283
1570
|
--agent <name> claude-code | cursor | codex (for pairing only)
|
|
1284
1571
|
--path <dir> Project to pair as. Defaults to cwd
|
|
1285
1572
|
--wait <seconds> How long to wait for a queued eval (default 600)
|
|
@@ -1322,6 +1609,66 @@ function pairingAgent(args) {
|
|
|
1322
1609
|
* numbers set on the settings page. Sending anything here -- even the defaults
|
|
1323
1610
|
* -- would override them and quietly make this command disagree with the page.
|
|
1324
1611
|
*/
|
|
1612
|
+
/**
|
|
1613
|
+
* A run that ended without this device ever being handed it.
|
|
1614
|
+
*
|
|
1615
|
+
* The server can refuse a run at claim time -- a target it will not dispatch, a
|
|
1616
|
+
* bundle it cannot build, a budget it will not spend -- and it records the
|
|
1617
|
+
* reason on the row without dispatching an event. The runner is left listening
|
|
1618
|
+
* on a stream that will never carry the job, which reads as a hang: the terminal
|
|
1619
|
+
* stops at "connected as runner" and the only account of what happened is a
|
|
1620
|
+
* column in Postgres. That is the same shape as every other bug found today, so
|
|
1621
|
+
* it is worth a poll.
|
|
1622
|
+
*
|
|
1623
|
+
* Only `failed` and `cancelled` end the wait. A successful run also reaches a
|
|
1624
|
+
* terminal status, but it does so *because this device finished it*, from inside
|
|
1625
|
+
* the job the loop is still tidying up after -- cutting the loop off there would
|
|
1626
|
+
* report `jobsRun: 0` for a run that worked.
|
|
1627
|
+
*/
|
|
1628
|
+
const TERMINAL_RUN_STATES = new Set(["failed", "cancelled"]);
|
|
1629
|
+
|
|
1630
|
+
export function watchRunTerminal(args, runId, { token, intervalMs = 3_000, fetchRun = null } = {}) {
|
|
1631
|
+
const controller = new AbortController();
|
|
1632
|
+
let over = null;
|
|
1633
|
+
let timer = null;
|
|
1634
|
+
let stopped = false;
|
|
1635
|
+
|
|
1636
|
+
const read =
|
|
1637
|
+
fetchRun ||
|
|
1638
|
+
((id) => callBackendJson(args, "GET", `/eval-runs/${encodeURIComponent(id)}`, { token }));
|
|
1639
|
+
|
|
1640
|
+
const tick = async () => {
|
|
1641
|
+
if (stopped) return;
|
|
1642
|
+
try {
|
|
1643
|
+
const seen = await read(runId);
|
|
1644
|
+
const status = String(seen?.status || "");
|
|
1645
|
+
if (seen?.ok !== false && TERMINAL_RUN_STATES.has(status)) {
|
|
1646
|
+
over = { status, error: String(seen.error || "") };
|
|
1647
|
+
stopped = true;
|
|
1648
|
+
// Wakes the loop: `stopWhen` is not read while the stream is idle.
|
|
1649
|
+
controller.abort();
|
|
1650
|
+
return;
|
|
1651
|
+
}
|
|
1652
|
+
} catch {
|
|
1653
|
+
// A poll that cannot reach the backend says nothing about the run, and a
|
|
1654
|
+
// run is not over because the network blinked.
|
|
1655
|
+
}
|
|
1656
|
+
if (!stopped) timer = setTimeout(tick, intervalMs);
|
|
1657
|
+
};
|
|
1658
|
+
|
|
1659
|
+
timer = setTimeout(tick, intervalMs);
|
|
1660
|
+
|
|
1661
|
+
return {
|
|
1662
|
+
signal: controller.signal,
|
|
1663
|
+
stopWhen: () => over !== null,
|
|
1664
|
+
ended: () => over,
|
|
1665
|
+
stop: () => {
|
|
1666
|
+
stopped = true;
|
|
1667
|
+
if (timer) clearTimeout(timer);
|
|
1668
|
+
},
|
|
1669
|
+
};
|
|
1670
|
+
}
|
|
1671
|
+
|
|
1325
1672
|
async function runForThisAccount(args, deps) {
|
|
1326
1673
|
const say = (line) => process.stdout.write(`${line}\n`);
|
|
1327
1674
|
|
|
@@ -1423,6 +1770,13 @@ async function runWithToken(
|
|
|
1423
1770
|
// This is for the ones it cannot: a signal, a closed tab, a crash in a
|
|
1424
1771
|
// callback -- after which the socket would otherwise outlive the run.
|
|
1425
1772
|
const forgetAdapter = registerEvalTeardown(() => adapter.close());
|
|
1773
|
+
if (target.kind === "callable") {
|
|
1774
|
+
// The import root for the customer's own wrapper. assert-ai resolves a
|
|
1775
|
+
// callable only as a dotted module name and has no spec key that adds an
|
|
1776
|
+
// import path, so the harness reads this and inserts it before handing off.
|
|
1777
|
+
process.env.PREMAN_EVAL_SYS_PATH = target.sysPath;
|
|
1778
|
+
}
|
|
1779
|
+
|
|
1426
1780
|
if (adapter.translated) {
|
|
1427
1781
|
// Set for this process so `childEnv` carries it to the harness. assert-ai
|
|
1428
1782
|
// refuses loopback endpoints by default and that default is right: it stops
|
|
@@ -1446,7 +1800,19 @@ async function runWithToken(
|
|
|
1446
1800
|
app_version: packageVersion() || undefined,
|
|
1447
1801
|
capabilities: { agent_eval_v1: true },
|
|
1448
1802
|
discovery: {
|
|
1449
|
-
|
|
1803
|
+
// Exactly one of these, which the backend enforces. A callable has
|
|
1804
|
+
// no address: the harness imports it in-process, so there is nothing
|
|
1805
|
+
// for the server to render a sandbox against.
|
|
1806
|
+
...(target.kind === "callable"
|
|
1807
|
+
? {
|
|
1808
|
+
callable: target.ref,
|
|
1809
|
+
sys_path: target.sysPath,
|
|
1810
|
+
entry_file: target.entryFile,
|
|
1811
|
+
entry_sha256: target.sha256,
|
|
1812
|
+
language: target.language,
|
|
1813
|
+
framework: target.framework,
|
|
1814
|
+
}
|
|
1815
|
+
: { url: adapter.url }),
|
|
1450
1816
|
label: target.label,
|
|
1451
1817
|
how: target.how,
|
|
1452
1818
|
// The addresses that did not answer, kept because "why did it test
|
|
@@ -1501,15 +1867,28 @@ async function runWithToken(
|
|
|
1501
1867
|
const headless = args.has("--headless") || args.has("--no-wait") || !process.stdout.isTTY;
|
|
1502
1868
|
let done = 0;
|
|
1503
1869
|
for (const run of runs) {
|
|
1504
|
-
const
|
|
1505
|
-
|
|
1506
|
-
|
|
1507
|
-
|
|
1508
|
-
|
|
1509
|
-
|
|
1870
|
+
const watch = watchRunTerminal(args, run.id, { token: resolveApiKey(args) });
|
|
1871
|
+
let result;
|
|
1872
|
+
try {
|
|
1873
|
+
result = await runnerLoop(args, state, {
|
|
1874
|
+
once: true,
|
|
1875
|
+
only: "eval",
|
|
1876
|
+
headless,
|
|
1877
|
+
log: (line) => say(` ${line}`),
|
|
1878
|
+
signal: watch.signal,
|
|
1879
|
+
stopWhen: watch.stopWhen,
|
|
1880
|
+
});
|
|
1881
|
+
} finally {
|
|
1882
|
+
watch.stop();
|
|
1883
|
+
}
|
|
1510
1884
|
done += result.jobsRun || 0;
|
|
1511
1885
|
if (!result.jobsRun) {
|
|
1512
|
-
|
|
1886
|
+
// The row's own reason, when there is one. `result.reason` describes why
|
|
1887
|
+
// the loop stopped, which on this path is "the watcher aborted it" --
|
|
1888
|
+
// true and useless. What somebody needs is why the run ended.
|
|
1889
|
+
const over = watch.ended();
|
|
1890
|
+
const why = over ? over.error || `the run ended as ${over.status}` : result.reason;
|
|
1891
|
+
say(` stopped after ${done} of ${runs.length} (${why})`);
|
|
1513
1892
|
break;
|
|
1514
1893
|
}
|
|
1515
1894
|
}
|
|
@@ -1579,11 +1958,40 @@ export async function evalCommand(
|
|
|
1579
1958
|
const sub = commandArgs.find((value) => !value.startsWith("-")) || "run";
|
|
1580
1959
|
const args = makeArgs(commandArgs);
|
|
1581
1960
|
|
|
1961
|
+
if (sub === "target") {
|
|
1962
|
+
// Its own command so the model call and the file write happen once, on
|
|
1963
|
+
// purpose, with somebody watching -- never as a side effect of a run.
|
|
1964
|
+
const projectPath = path.resolve(args.value("--path", process.cwd()));
|
|
1965
|
+
const say = (line = "") => process.stdout.write(`${line}\n`);
|
|
1966
|
+
const { searchForEntry, entryFromNamedSymbol } = await import("./eval_entry.js");
|
|
1967
|
+
const named = String(args.value("--entry", "") || "").trim();
|
|
1968
|
+
try {
|
|
1969
|
+
const found = named
|
|
1970
|
+
? await entryFromNamedSymbol(named, { projectPath, args, log: say })
|
|
1971
|
+
: await searchForEntry({ projectPath, args, log: say });
|
|
1972
|
+
say(` ${found.label}`);
|
|
1973
|
+
say(` ${found.how}`);
|
|
1974
|
+
return { ok: true, target: found };
|
|
1975
|
+
} catch (error) {
|
|
1976
|
+
say(error.message);
|
|
1977
|
+
return { ok: false, reason: "no_entry" };
|
|
1978
|
+
}
|
|
1979
|
+
}
|
|
1980
|
+
|
|
1582
1981
|
if (sub === "doctor") {
|
|
1583
1982
|
const projectPath = path.resolve(args.value("--path", process.cwd()));
|
|
1584
1983
|
let python;
|
|
1585
1984
|
try {
|
|
1586
|
-
|
|
1985
|
+
// Doctor answers "could a run happen here", so it has to resolve the same
|
|
1986
|
+
// way a run does -- including fetching the runtime if that is what a run
|
|
1987
|
+
// would do. Reporting an interpreter a real run would not pick is the one
|
|
1988
|
+
// thing a diagnostic must not do.
|
|
1989
|
+
const runtime = await ensureRuntime({
|
|
1990
|
+
log: (line) => process.stdout.write(`${line}\n`),
|
|
1991
|
+
download,
|
|
1992
|
+
progress: progressLine,
|
|
1993
|
+
});
|
|
1994
|
+
python = resolveInterpreter({ projectPath, runtime });
|
|
1587
1995
|
} catch (error) {
|
|
1588
1996
|
process.stdout.write(`Cannot run evals here.\n ${error.message}\n`);
|
|
1589
1997
|
return { ok: false, reason: "no_harness" };
|