premanmcp 1.1.7 → 1.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/eval.js CHANGED
@@ -44,7 +44,10 @@ import os from "node:os";
44
44
  import path from "node:path";
45
45
  import { fileURLToPath } from "node:url";
46
46
 
47
+ import { download, progressLine } from "./desktop.js";
48
+ import { repoEnv } from "./detect.js";
47
49
  import { adapterStats, discoverTarget, startAdapter, stepFeed } from "./eval_target.js";
50
+ import { ensureRuntime } from "./eval_runtime.js";
48
51
  import { formatSurfaceLink, resolveEvalSurface } from "./link.js";
49
52
  import {
50
53
  backendUrl,
@@ -278,6 +281,26 @@ export function materialize(job, root) {
278
281
  * prefix has to be produced by executing the snippet, so echoing it back does
279
282
  * not satisfy the check.
280
283
  */
284
+ /**
285
+ * The harness, named so that a resolver can fetch it alongside a customer's
286
+ * own pins.
287
+ *
288
+ * A git reference rather than a package name: PyPI serves upstream ASSERT, and
289
+ * PreMan's fork carries functional changes, so a customer resolving the public
290
+ * name would run differently-shaped code and nothing records which harness
291
+ * produced a result. Pinned to a commit for the same reason `pyproject.toml`
292
+ * pins one -- two machines must agree about what measured them.
293
+ *
294
+ * Overridable so a machine with no route to GitHub, or a contributor testing a
295
+ * local checkout, is not stuck.
296
+ */
297
+ export const HARNESS_SPEC =
298
+ String(process.env.PREMAN_EVAL_HARNESS_SPEC || "").trim() ||
299
+ "git+https://github.com/AleWang16/ASSERT@14f4e2136ab753897ce4cb90613e1663730aeb70";
300
+
301
+ /** How long a first resolve may take before it counts as a failure. */
302
+ const COLD_RESOLVE_MS = 600_000;
303
+
281
304
  const PROBE_MARKER = "preman-assert-ai=";
282
305
  const PROBE_SNIPPET = `import assert_ai,importlib.metadata as m;print("${PROBE_MARKER}"+m.version("assert-ai"))`;
283
306
 
@@ -289,11 +312,11 @@ const PROBE_SNIPPET = `import assert_ai,importlib.metadata as m;print("${PROBE_M
289
312
  * about to use" are different questions, and only the second one predicts
290
313
  * whether the run works.
291
314
  */
292
- function probeInterpreter(command, argv) {
315
+ function probeInterpreter(command, argv, { timeoutMs = 120_000 } = {}) {
293
316
  const probe = spawnSync(command, [...argv, "-c", PROBE_SNIPPET], {
294
317
  encoding: "utf8",
295
318
  stdio: ["ignore", "pipe", "ignore"],
296
- timeout: 120_000,
319
+ timeout: timeoutMs,
297
320
  });
298
321
  if (probe.status !== 0) return "";
299
322
  const line = String(probe.stdout || "")
@@ -317,11 +340,23 @@ function probeInterpreter(command, argv) {
317
340
  * Then a bare `python3`, which is the case where somebody installed the harness
318
341
  * globally.
319
342
  *
320
- * There is no fourth branch that installs anything. Downloading and executing a
321
- * Python package on a developer's machine, unattended, because a job frame
322
- * arrived, is not a thing a runner should decide to do on its own.
343
+ * `runtime` is PreMan's own interpreter, already fetched, and it is ranked
344
+ * second rather than first for a reason worth stating: it is the right answer
345
+ * for the harness and the wrong one for the customer's agent. A `callable`
346
+ * target has the harness import their agent in-process, and a bundle that
347
+ * cannot import their LangGraph measures nothing. So the *caller* decides
348
+ * whether to offer it — `executeEvalRun` passes `""` for a callable target —
349
+ * and this function only ranks what it is given. See
350
+ * docs/EVAL_SELF_CONTAINED_RUNTIME.md.
351
+ *
352
+ * The harness is never added to a customer's declared dependencies. The runtime
353
+ * lives under `~/.preman` and touches nothing of theirs. The harness-alongside
354
+ * branch is the one exception worth stating plainly: on a project that has
355
+ * never been synced it creates `.venv/` and `uv.lock`, exactly as the
356
+ * customer's own `uv run` would, and their lock records their packages only --
357
+ * measured, not assumed. Nothing pins PreMan into their project.
323
358
  */
324
- export function resolveInterpreter({ projectPath = process.cwd(), log = () => {} } = {}) {
359
+ export function resolveInterpreter({ projectPath = process.cwd(), log = () => {}, runtime = "" } = {}) {
325
360
  const override = String(process.env.PREMAN_EVAL_PYTHON || "").trim();
326
361
  if (override) {
327
362
  const version = probeInterpreter(override, []);
@@ -333,6 +368,14 @@ export function resolveInterpreter({ projectPath = process.cwd(), log = () => {}
333
368
  return { command: override, argv: [], version, how: "PREMAN_EVAL_PYTHON" };
334
369
  }
335
370
 
371
+ // Above the project check on purpose: that check keys on whatever directory
372
+ // the command was run from, which is how a run in `~/codex` picked up a conda
373
+ // interpreter that imported assert_ai and then produced nothing.
374
+ if (runtime) {
375
+ const version = probeInterpreter(runtime, []);
376
+ if (version) return { command: runtime, argv: [], version, how: "preman runtime" };
377
+ }
378
+
336
379
  if (existsSync(path.join(projectPath, "pyproject.toml"))) {
337
380
  const version = probeInterpreter("uv", ["run", "--project", projectPath, "python"]);
338
381
  if (version) {
@@ -343,6 +386,40 @@ export function resolveInterpreter({ projectPath = process.cwd(), log = () => {}
343
386
  how: "uv run in this project",
344
387
  };
345
388
  }
389
+
390
+ // The same project, with the harness resolved alongside it into an
391
+ // environment uv owns rather than one the customer does. Ranked below the
392
+ // branch above because a repository that pins the harness itself should
393
+ // keep its own pin; this is for the repository that has never heard of
394
+ // PreMan, which is every customer's.
395
+ //
396
+ // This is the only branch that can serve a `callable` target in a stranger's
397
+ // repository: their agent is imported in-process, so their dependencies and
398
+ // the harness have to be reachable from one interpreter, and PreMan's own
399
+ // bundle has none of theirs.
400
+ //
401
+ // What this does leave behind, measured on a repository that had never been
402
+ // synced: `.venv/` and `uv.lock` in their project -- the same two files
403
+ // their own `uv run` would have created. What it does *not* leave behind is
404
+ // any trace of the harness in `uv.lock`: the lock carried their 18 packages
405
+ // and no mention of assert-ai, so `--with` does not pin PreMan into a
406
+ // customer's project. That is the property worth protecting if this branch
407
+ // is ever rewritten.
408
+ //
409
+ // Given a long timeout on purpose. The first resolve on a cold machine
410
+ // fetches the harness's whole dependency tree, and treating a slow network
411
+ // as "this interpreter cannot import assert_ai" would send somebody looking
412
+ // for a missing package.
413
+ const added = ["run", "--project", projectPath, "--with", HARNESS_SPEC, "python"];
414
+ const resolved = probeInterpreter("uv", added, { timeoutMs: COLD_RESOLVE_MS });
415
+ if (resolved) {
416
+ return {
417
+ command: "uv",
418
+ argv: added,
419
+ version: resolved,
420
+ how: "uv run in this project, with PreMan's harness alongside it",
421
+ };
422
+ }
346
423
  }
347
424
 
348
425
  const version = probeInterpreter("python3", []);
@@ -413,10 +490,20 @@ export function missingModelKey(job, env = process.env) {
413
490
  if (!name.startsWith(prefix) || seen.has(variable)) continue;
414
491
  seen.add(variable);
415
492
  if (env[variable] || supplied[variable]) continue;
493
+ // Says whose key this is, because the old wording ("needs a model key
494
+ // for X and there isn't one") reads as a configuration slip the reader
495
+ // made. It is not: PreMan's own stages -- deriving the failure
496
+ // categories, writing the cases, judging the transcripts -- run on this
497
+ // provider's models whatever the agent under test is written against. A
498
+ // customer whose agent uses a different provider would otherwise go
499
+ // looking for a key they were never asked for and conclude the tool is
500
+ // broken.
416
501
  return (
417
- `this run needs a model key for ${PROVIDER_NAMES[prefix]} and there isn't one. Save ` +
418
- `yours once under Settings → Model keys and every machine on this account picks it ` +
419
- `up, or export ${variable} in this terminal for a one-off.`
502
+ `PreMan's own evaluation stages run on ${PROVIDER_NAMES[prefix]} models, so a key ` +
503
+ `for ${PROVIDER_NAMES[prefix]} is needed even when your agent uses a different ` +
504
+ `provider. This is separate from whatever key your agent itself needs, which stays ` +
505
+ `on this machine. Save one under Settings → Model keys and every machine on this ` +
506
+ `account picks it up, or export ${variable} in this terminal for a one-off.`
420
507
  );
421
508
  }
422
509
  }
@@ -532,6 +619,11 @@ export const RUN_ARTIFACTS = [
532
619
  "scores.jsonl",
533
620
  "metrics.json",
534
621
  "artifacts.json",
622
+ // Which bank case each row answers, when this run was handed a subset. The
623
+ // server writes it for a run it executes itself; listed here so the two paths
624
+ // keep publishing the same names, which is what lets one dashboard read a run
625
+ // without caring where it ran.
626
+ "case_map.json",
535
627
  ];
536
628
 
537
629
  /**
@@ -705,6 +797,138 @@ export function statsSince(baseline, stats) {
705
797
  };
706
798
  }
707
799
 
800
+ /**
801
+ * Did this run write anything down at all?
802
+ *
803
+ * The most dangerous shape a run can take, and the one that cost this project
804
+ * an afternoon: the harness exits 0, writes no metrics and no transcript, and
805
+ * every layer above reports success. On disk the suite directory was empty; in
806
+ * the database the row said `succeeded` with a `{}` summary after 5m38s. A
807
+ * broken interpreter and a working one are indistinguishable from the outside,
808
+ * which is the worst property a measurement tool can have.
809
+ *
810
+ * Deliberately not keyed on the adapter's counters, and that is the whole point
811
+ * of this function existing separately from the two guards above it. Those read
812
+ * `targetStats`, which nothing increments on a `callable` target -- the harness
813
+ * imports the agent in-process and no request ever arrives at an adapter -- so
814
+ * on the default target kind they are not unreliable, they are inert. A
815
+ * counter-based check would be correct over HTTP and dead where it is needed
816
+ * most. An empty summary and an empty artifact directory are facts about the
817
+ * run on every path, including a future one that splits the harness and the
818
+ * agent into two processes.
819
+ *
820
+ * Any one file with bytes in it is enough. A partial run is a finding worth
821
+ * keeping; only nothing at all is a failure to measure.
822
+ */
823
+ /**
824
+ * The module name the generated wrapper is always written under.
825
+ *
826
+ * Fixed rather than derived: the server writes `DEVICE_CALLABLE_REF` itself and
827
+ * interpolates nothing a caller sent, which is what makes importing it by name
828
+ * on a customer's machine safe. If that constant ever moves, this has to move
829
+ * with it or the check below silently stops checking.
830
+ */
831
+ const CALLABLE_MODULE = "preman_agent_entry";
832
+
833
+ /**
834
+ * Does the approved wrapper still import, in the interpreter this run will use?
835
+ *
836
+ * Returns "" when it does, a short reason when it does not. An approved entry
837
+ * point is trusted on a digest, which proves the file has not changed and says
838
+ * nothing about whether it still runs. The two come apart exactly where it
839
+ * matters -- a rebuilt virtual environment, a dependency that went away, a
840
+ * different interpreter resolved than the one it was approved against -- and
841
+ * the first time that happened the run reached `inference`, spent model calls,
842
+ * and reported a missing attribute on a function that plainly existed.
843
+ *
844
+ * Checked here rather than at discovery because the interpreter is already
845
+ * resolved by this point. Doing it earlier would mean resolving one twice, and
846
+ * on the harness-alongside branch a resolve is seconds warm and minutes cold.
847
+ *
848
+ * A load and not a turn. Making the agent answer is what the approval gate
849
+ * does, and it costs a real model call; paying that on every run would tax
850
+ * every suite for a check that only earns its keep when something drifted. It
851
+ * is not free either -- `preman_selfcheck` loads the customer's module, which
852
+ * for a framework repository means importing crewai or langchain -- but only
853
+ * this probe pays it, where the gate's version bills a model.
854
+ *
855
+ * Why a named function rather than importing the wrapper and letting its own
856
+ * side effects do the work: the template loads the agent lazily on purpose,
857
+ * because a customer's module can bind a port or open a connection at import
858
+ * and a probe should not be the thing that triggers it. The two designs also
859
+ * fail differently, which is the deciding reason. A template that stops
860
+ * defining `preman_selfcheck` makes this skip *and say so*; a template that
861
+ * stopped loading eagerly would make it import cleanly and pass -- a false
862
+ * negative with no signal, which is the shape of the bug this exists to catch.
863
+ *
864
+ * A wrapper generated before the contract existed has no `preman_selfcheck`,
865
+ * so it is skipped rather than failed. Those are only re-checked once they are
866
+ * regenerated; the hole is closed going forward, not retroactively.
867
+ */
868
+ export function entryImportFailure(python, sysPath, { timeoutMs = 120_000, log = null } = {}) {
869
+ if (!python?.command || !sysPath) return "";
870
+
871
+ const snippet = [
872
+ `import sys`,
873
+ `sys.path.insert(0,${JSON.stringify(sysPath)})`,
874
+ `import ${CALLABLE_MODULE} as _w`,
875
+ `_c=getattr(_w,"preman_selfcheck",None)`,
876
+ `print("preman-selfcheck=absent") if _c is None else (_c(),print("preman-selfcheck=ok"))`,
877
+ ].join(";");
878
+
879
+ const probe = spawnSync(python.command, [...(python.argv || []), "-c", snippet], {
880
+ encoding: "utf8",
881
+ stdio: ["ignore", "pipe", "pipe"],
882
+ timeout: timeoutMs,
883
+ });
884
+
885
+ // Killed, or never started. `spawnSync` reports both as `status: null` with no
886
+ // stderr, and the fallback below would turn that into "exit null" and block
887
+ // the run -- a check that could not finish, reported as proof of breakage.
888
+ // The timeout is the one that can really happen: `preman_selfcheck` imports
889
+ // the customer's agent, and importing a framework is slow.
890
+ if (probe.error || probe.status === null) {
891
+ const why = probe.error?.message || `it did not finish within ${Math.round(timeoutMs / 1000)}s`;
892
+ log?.(
893
+ `whether the approved entry point still loads could not be checked (${why}); ` +
894
+ `running anyway rather than refusing on a check that did not complete`
895
+ );
896
+ return "";
897
+ }
898
+
899
+ if (probe.status === 0) {
900
+ if (String(probe.stdout || "").includes("preman-selfcheck=absent")) {
901
+ log?.(
902
+ `the approved entry point predates PreMan's self-check, so whether it still loads ` +
903
+ `was not verified; regenerate it with \`${cliInvocation()} eval target\` to have it checked`
904
+ );
905
+ }
906
+ return "";
907
+ }
908
+
909
+ const lines = String(probe.stderr || "").trim().split("\n").filter(Boolean);
910
+ return lines[lines.length - 1]?.slice(0, 300) || `exit ${probe.status}`;
911
+ }
912
+
913
+ export function producedNothing(dir, summary) {
914
+ if (summary && typeof summary === "object" && Object.keys(summary).length) return "";
915
+
916
+ const wrote = ["metrics.json", "scores.jsonl", "inference_set.jsonl"].some((name) => {
917
+ try {
918
+ return statSync(path.join(dir, name)).size > 0;
919
+ } catch {
920
+ return false;
921
+ }
922
+ });
923
+ if (wrote) return "";
924
+
925
+ return (
926
+ "the harness exited cleanly having written no metrics and no transcript, so there is " +
927
+ "nothing here that was measured. The usual cause is an interpreter that can import " +
928
+ "assert_ai but cannot complete a run; `eval doctor` reports which one this machine picks."
929
+ );
930
+ }
931
+
708
932
  export function missingToolEvidence(dir, stats) {
709
933
  const forwarded = Number(stats?.toolCalls);
710
934
  if (!Number.isFinite(forwarded) || forwarded <= 0) return "";
@@ -945,7 +1169,15 @@ export async function executeEvalRun(
945
1169
  home = suiteHome(job);
946
1170
  pruneRuns(home, job);
947
1171
  laid = materialize(job, home);
948
- python = resolveInterpreter({ projectPath, log });
1172
+ // Offered only when the harness is the only thing that has to import
1173
+ // anything. A `callable` target imports the customer's agent in-process, so
1174
+ // it must resolve against an environment that has *their* dependencies —
1175
+ // PreMan's bundle has assert-ai and nothing of theirs.
1176
+ const runtime =
1177
+ String(job.target_kind || "") === "callable"
1178
+ ? ""
1179
+ : await ensureRuntime({ log, download, progress: progressLine });
1180
+ python = resolveInterpreter({ projectPath, log, runtime });
949
1181
  log(`eval ${job.id}: assert-ai ${python.version} via ${python.how}`);
950
1182
  } catch (error) {
951
1183
  // Every one of these is a fact about this machine or this frame, and no
@@ -960,7 +1192,26 @@ export async function executeEvalRun(
960
1192
  return finish(false, { error: `the eval harness is missing from this install (${HARNESS_PATH})` });
961
1193
  }
962
1194
 
963
- const missingKey = missingModelKey(job);
1195
+ // Before the harness, because everything after this point costs model calls.
1196
+ if (String(job.target_kind || "") === "callable") {
1197
+ const broken = entryImportFailure(python, process.env.PREMAN_EVAL_SYS_PATH || "", {
1198
+ log: (line) => log(`eval ${job.id}: ${line}`),
1199
+ });
1200
+ if (broken) {
1201
+ tidy(home);
1202
+ return finish(false, {
1203
+ error:
1204
+ `the approved entry point no longer loads in this environment: ${broken}. It was ` +
1205
+ `trusted on its digest, which only proves the file is unchanged -- re-run ` +
1206
+ `\`${cliInvocation()} eval target\` here to regenerate and re-approve it.`,
1207
+ });
1208
+ }
1209
+ }
1210
+
1211
+ // The environment the harness will actually get, so the check cannot pass
1212
+ // where the run fails or refuse where the key was in the repository all along.
1213
+ const runEnv = childEnv({}, { ...(job.env || {}), ...repoEnv(projectPath) });
1214
+ const missingKey = missingModelKey(job, runEnv);
964
1215
  if (missingKey) {
965
1216
  // Checked before the stages start rather than left to the provider client,
966
1217
  // which reports it from inside `inference` as an authentication error
@@ -987,7 +1238,7 @@ export async function executeEvalRun(
987
1238
  const timeout = Math.min(RUN_TIMEOUT_MS, RUN_TIMEOUT_CAP_MS);
988
1239
  const child = spawn(python.command, [...python.argv, HARNESS_PATH, "run", "--config", laid.configPath], {
989
1240
  cwd: laid.cwd,
990
- env: childEnv({ PREMAN_EVAL_RUN: `${job.suite}/${job.run}` }, job.env || {}),
1241
+ env: childEnv({ PREMAN_EVAL_RUN: `${job.suite}/${job.run}` }, { ...(job.env || {}), ...repoEnv(projectPath) }),
991
1242
  stdio: ["ignore", "ignore", "pipe"],
992
1243
  });
993
1244
 
@@ -1135,15 +1386,45 @@ export async function executeEvalRun(
1135
1386
  return { ...outcome, summary, artifacts };
1136
1387
  }
1137
1388
 
1389
+ // Last, because the two guards above name a cause and this one only reports
1390
+ // that there is nothing to report. Still ahead of `finish(true, ...)`,
1391
+ // because a run with no results is not a passing run.
1392
+ const empty = producedNothing(artifacts, summary);
1393
+ if (empty) {
1394
+ const outcome = await finish(false, { summary, error: empty, exitCode });
1395
+ tidy(home);
1396
+ return { ...outcome, summary, artifacts };
1397
+ }
1398
+
1138
1399
  const outcome = await finish(true, { summary, exitCode });
1139
1400
  tidy(home);
1140
1401
  return { ...outcome, summary, artifacts };
1141
1402
  }
1142
1403
 
1404
+ // The harness's own account of the failure, kept on disk before it is reduced
1405
+ // to something that fits in a row.
1406
+ //
1407
+ // The tail is eight lines because that is what a terminal and a database
1408
+ // column can carry, and the first time an inference failure had to be
1409
+ // diagnosed the rest had already been thrown away: the tail named a symptom
1410
+ // -- a wrapper reporting `has no attribute` on a function that plainly
1411
+ // existed -- while the exception that said *why* was the one before it, and
1412
+ // gone. Written beside the artifacts, which `tidy` leaves in place, so the
1413
+ // account outlives the process that produced it.
1414
+ const failureLog = path.join(laid.cwd, "last-failure.log");
1415
+ let kept = "";
1416
+ try {
1417
+ writeFileSync(failureLog, stderr, { mode: 0o600 });
1418
+ kept = ` (full output: ${failureLog})`;
1419
+ } catch {
1420
+ // A run that cannot write its own post-mortem still has to report the
1421
+ // failure it was handed.
1422
+ }
1423
+
1143
1424
  const tail = stderr.trim().split("\n").slice(-8).join(" / ");
1144
1425
  const outcome = await finish(false, {
1145
1426
  summary,
1146
- error: `the eval harness exited ${exitCode}: ${tail}`,
1427
+ error: `the eval harness exited ${exitCode}: ${tail}${kept}`,
1147
1428
  exitCode,
1148
1429
  });
1149
1430
  tidy(home);
@@ -1277,9 +1558,15 @@ Eval options:
1277
1558
  token names
1278
1559
  eval run Pair this project and run the eval PreMan has
1279
1560
  queued for it, then exit
1561
+ eval target Find the code in this repository that runs your
1562
+ agent, write a wrapper the eval can call, and
1563
+ remember it once you approve it
1280
1564
  eval doctor Report whether this machine can run an eval
1565
+ --entry <file>:<symbol> Skip the search; this is the agent. Still writes
1566
+ a wrapper and still asks you to approve it
1281
1567
  --target <url> Skip discovery; this URL is the agent. Must take
1282
1568
  POST {"message","history"} and answer {"response"}
1569
+ --preman-agent <url> Evaluate PreMan's own chat agent at this address
1283
1570
  --agent <name> claude-code | cursor | codex (for pairing only)
1284
1571
  --path <dir> Project to pair as. Defaults to cwd
1285
1572
  --wait <seconds> How long to wait for a queued eval (default 600)
@@ -1322,6 +1609,66 @@ function pairingAgent(args) {
1322
1609
  * numbers set on the settings page. Sending anything here -- even the defaults
1323
1610
  * -- would override them and quietly make this command disagree with the page.
1324
1611
  */
1612
+ /**
1613
+ * A run that ended without this device ever being handed it.
1614
+ *
1615
+ * The server can refuse a run at claim time -- a target it will not dispatch, a
1616
+ * bundle it cannot build, a budget it will not spend -- and it records the
1617
+ * reason on the row without dispatching an event. The runner is left listening
1618
+ * on a stream that will never carry the job, which reads as a hang: the terminal
1619
+ * stops at "connected as runner" and the only account of what happened is a
1620
+ * column in Postgres. That is the same shape as every other bug found today, so
1621
+ * it is worth a poll.
1622
+ *
1623
+ * Only `failed` and `cancelled` end the wait. A successful run also reaches a
1624
+ * terminal status, but it does so *because this device finished it*, from inside
1625
+ * the job the loop is still tidying up after -- cutting the loop off there would
1626
+ * report `jobsRun: 0` for a run that worked.
1627
+ */
1628
+ const TERMINAL_RUN_STATES = new Set(["failed", "cancelled"]);
1629
+
1630
+ export function watchRunTerminal(args, runId, { token, intervalMs = 3_000, fetchRun = null } = {}) {
1631
+ const controller = new AbortController();
1632
+ let over = null;
1633
+ let timer = null;
1634
+ let stopped = false;
1635
+
1636
+ const read =
1637
+ fetchRun ||
1638
+ ((id) => callBackendJson(args, "GET", `/eval-runs/${encodeURIComponent(id)}`, { token }));
1639
+
1640
+ const tick = async () => {
1641
+ if (stopped) return;
1642
+ try {
1643
+ const seen = await read(runId);
1644
+ const status = String(seen?.status || "");
1645
+ if (seen?.ok !== false && TERMINAL_RUN_STATES.has(status)) {
1646
+ over = { status, error: String(seen.error || "") };
1647
+ stopped = true;
1648
+ // Wakes the loop: `stopWhen` is not read while the stream is idle.
1649
+ controller.abort();
1650
+ return;
1651
+ }
1652
+ } catch {
1653
+ // A poll that cannot reach the backend says nothing about the run, and a
1654
+ // run is not over because the network blinked.
1655
+ }
1656
+ if (!stopped) timer = setTimeout(tick, intervalMs);
1657
+ };
1658
+
1659
+ timer = setTimeout(tick, intervalMs);
1660
+
1661
+ return {
1662
+ signal: controller.signal,
1663
+ stopWhen: () => over !== null,
1664
+ ended: () => over,
1665
+ stop: () => {
1666
+ stopped = true;
1667
+ if (timer) clearTimeout(timer);
1668
+ },
1669
+ };
1670
+ }
1671
+
1325
1672
  async function runForThisAccount(args, deps) {
1326
1673
  const say = (line) => process.stdout.write(`${line}\n`);
1327
1674
 
@@ -1423,6 +1770,13 @@ async function runWithToken(
1423
1770
  // This is for the ones it cannot: a signal, a closed tab, a crash in a
1424
1771
  // callback -- after which the socket would otherwise outlive the run.
1425
1772
  const forgetAdapter = registerEvalTeardown(() => adapter.close());
1773
+ if (target.kind === "callable") {
1774
+ // The import root for the customer's own wrapper. assert-ai resolves a
1775
+ // callable only as a dotted module name and has no spec key that adds an
1776
+ // import path, so the harness reads this and inserts it before handing off.
1777
+ process.env.PREMAN_EVAL_SYS_PATH = target.sysPath;
1778
+ }
1779
+
1426
1780
  if (adapter.translated) {
1427
1781
  // Set for this process so `childEnv` carries it to the harness. assert-ai
1428
1782
  // refuses loopback endpoints by default and that default is right: it stops
@@ -1446,7 +1800,19 @@ async function runWithToken(
1446
1800
  app_version: packageVersion() || undefined,
1447
1801
  capabilities: { agent_eval_v1: true },
1448
1802
  discovery: {
1449
- url: adapter.url,
1803
+ // Exactly one of these, which the backend enforces. A callable has
1804
+ // no address: the harness imports it in-process, so there is nothing
1805
+ // for the server to render a sandbox against.
1806
+ ...(target.kind === "callable"
1807
+ ? {
1808
+ callable: target.ref,
1809
+ sys_path: target.sysPath,
1810
+ entry_file: target.entryFile,
1811
+ entry_sha256: target.sha256,
1812
+ language: target.language,
1813
+ framework: target.framework,
1814
+ }
1815
+ : { url: adapter.url }),
1450
1816
  label: target.label,
1451
1817
  how: target.how,
1452
1818
  // The addresses that did not answer, kept because "why did it test
@@ -1501,15 +1867,28 @@ async function runWithToken(
1501
1867
  const headless = args.has("--headless") || args.has("--no-wait") || !process.stdout.isTTY;
1502
1868
  let done = 0;
1503
1869
  for (const run of runs) {
1504
- const result = await runnerLoop(args, state, {
1505
- once: true,
1506
- only: "eval",
1507
- headless,
1508
- log: (line) => say(` ${line}`),
1509
- });
1870
+ const watch = watchRunTerminal(args, run.id, { token: resolveApiKey(args) });
1871
+ let result;
1872
+ try {
1873
+ result = await runnerLoop(args, state, {
1874
+ once: true,
1875
+ only: "eval",
1876
+ headless,
1877
+ log: (line) => say(` ${line}`),
1878
+ signal: watch.signal,
1879
+ stopWhen: watch.stopWhen,
1880
+ });
1881
+ } finally {
1882
+ watch.stop();
1883
+ }
1510
1884
  done += result.jobsRun || 0;
1511
1885
  if (!result.jobsRun) {
1512
- say(` stopped after ${done} of ${runs.length} (${result.reason})`);
1886
+ // The row's own reason, when there is one. `result.reason` describes why
1887
+ // the loop stopped, which on this path is "the watcher aborted it" --
1888
+ // true and useless. What somebody needs is why the run ended.
1889
+ const over = watch.ended();
1890
+ const why = over ? over.error || `the run ended as ${over.status}` : result.reason;
1891
+ say(` stopped after ${done} of ${runs.length} (${why})`);
1513
1892
  break;
1514
1893
  }
1515
1894
  }
@@ -1579,11 +1958,40 @@ export async function evalCommand(
1579
1958
  const sub = commandArgs.find((value) => !value.startsWith("-")) || "run";
1580
1959
  const args = makeArgs(commandArgs);
1581
1960
 
1961
+ if (sub === "target") {
1962
+ // Its own command so the model call and the file write happen once, on
1963
+ // purpose, with somebody watching -- never as a side effect of a run.
1964
+ const projectPath = path.resolve(args.value("--path", process.cwd()));
1965
+ const say = (line = "") => process.stdout.write(`${line}\n`);
1966
+ const { searchForEntry, entryFromNamedSymbol } = await import("./eval_entry.js");
1967
+ const named = String(args.value("--entry", "") || "").trim();
1968
+ try {
1969
+ const found = named
1970
+ ? await entryFromNamedSymbol(named, { projectPath, args, log: say })
1971
+ : await searchForEntry({ projectPath, args, log: say });
1972
+ say(` ${found.label}`);
1973
+ say(` ${found.how}`);
1974
+ return { ok: true, target: found };
1975
+ } catch (error) {
1976
+ say(error.message);
1977
+ return { ok: false, reason: "no_entry" };
1978
+ }
1979
+ }
1980
+
1582
1981
  if (sub === "doctor") {
1583
1982
  const projectPath = path.resolve(args.value("--path", process.cwd()));
1584
1983
  let python;
1585
1984
  try {
1586
- python = resolveInterpreter({ projectPath });
1985
+ // Doctor answers "could a run happen here", so it has to resolve the same
1986
+ // way a run does -- including fetching the runtime if that is what a run
1987
+ // would do. Reporting an interpreter a real run would not pick is the one
1988
+ // thing a diagnostic must not do.
1989
+ const runtime = await ensureRuntime({
1990
+ log: (line) => process.stdout.write(`${line}\n`),
1991
+ download,
1992
+ progress: progressLine,
1993
+ });
1994
+ python = resolveInterpreter({ projectPath, runtime });
1587
1995
  } catch (error) {
1588
1996
  process.stdout.write(`Cannot run evals here.\n ${error.message}\n`);
1589
1997
  return { ok: false, reason: "no_harness" };