premanmcp 1.0.4 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/link.js CHANGED
@@ -15,7 +15,7 @@
15
15
  */
16
16
 
17
17
  import { desktopAppInstalled, installedDesktopVersion } from "./desktop.js";
18
- import { DEFAULT_FRONTEND, frontendUrl } from "./shared.js";
18
+ import { DEFAULT_FRONTEND, frontendUrl, pairedFrontendUrl } from "./shared.js";
19
19
 
20
20
  export { installedDesktopVersion };
21
21
 
@@ -98,18 +98,93 @@ export function desktopSupportsRouting(destination = "/Applications") {
98
98
  * handing it a localhost route would open the wrong build.
99
99
  */
100
100
  export function resolveRunSurface(args, { batchId = "", destination = "/Applications" } = {}) {
101
- const base = frontendUrl(args);
102
101
  const route = batchId ? `${RUN_ROUTE}?batch=${encodeURIComponent(batchId)}` : RUN_ROUTE;
103
- const webUrl = `${base}${route}`;
102
+ return resolveSurface(args, route, { destination, source: "push" });
103
+ }
104
104
 
105
- if (base !== DEFAULT_FRONTEND) return { href: webUrl, webUrl, surface: "web-local" };
106
- if (desktopAppInstalled(destination) && desktopSupportsRouting(destination)) {
105
+ /** The dashboard view an eval link points at. */
106
+ export const EVAL_ROUTE = "/workbench/agent-eval";
107
+
108
+ /**
109
+ * Where to send someone to watch the eval this CLI is about to run.
110
+ *
111
+ * Same three-branch resolution as a push link, and worth being explicit about
112
+ * what it does and does not promise, because the eval flow leans on it harder:
113
+ * a push link is a nicety after the work is done, whereas an eval is only worth
114
+ * watching *while* it runs.
115
+ *
116
+ * The desktop branch needs macOS, an installed app, and a build new enough to
117
+ * read `route`; older builds bring the window forward and land on whatever view
118
+ * was last open. So `webUrl` is printed in every branch and the caller must
119
+ * treat "opened" as best-effort. See docs/EVAL_REVEAL_CONTRACT.md.
120
+ */
121
+ export function resolveEvalSurface(args, { runId = "", destination = "/Applications" } = {}) {
122
+ const route = runId ? `${EVAL_ROUTE}?run=${encodeURIComponent(runId)}` : EVAL_ROUTE;
123
+ return resolveSurface(args, route, {
124
+ destination,
125
+ source: "eval",
126
+ requirePairing: true,
127
+ // An eval is watched while it happens, and the app is where somebody
128
+ // watching it is already sitting. A browser tab is the fallback, not the
129
+ // default, for the one link in this CLI that is time-sensitive.
130
+ preferApp: true,
131
+ });
132
+ }
133
+
134
+ /**
135
+ * `requirePairing` is the difference between a pointer and an appointment.
136
+ *
137
+ * A push link is read after the fact: it says "your results are in the
138
+ * dashboard", and naming the usual dashboard is a fair answer even when this
139
+ * particular invocation went somewhere unusual. Guessing costs a click.
140
+ *
141
+ * An eval link is an appointment to watch something happen in the next few
142
+ * seconds. Sending someone to a deployment the run does not exist on gives them
143
+ * a page that loads, renders, and shows nothing — which reads as a lost run
144
+ * rather than a wrong address, and the run is over by the time that is worked
145
+ * out. So the eval side refuses to guess and the caller drops the link.
146
+ */
147
+ /**
148
+ * `preferApp` is for the case where the app is the point.
149
+ *
150
+ * Routing into the app needs a build that reads `route`, and only the hosted
151
+ * dashboard is worth routing to — the app ships its own bundled UI, so handing
152
+ * it a localhost path opens the wrong build. Both conditions fail often, and
153
+ * until now failing either meant a browser.
154
+ *
155
+ * That is the wrong answer for somebody sitting in front of the app watching an
156
+ * eval. Raising is a strictly weaker action than routing — the URL carries no
157
+ * route, the window simply comes forward on whatever it was showing — so it is
158
+ * safe when routing is not: no build can misread it, and no dashboard can be
159
+ * the wrong one, because none is named. It cannot land you on the run, and the
160
+ * caller says so rather than implying otherwise.
161
+ */
162
+ function resolveSurface(
163
+ args,
164
+ route,
165
+ { destination, source, requirePairing = false, preferApp = false }
166
+ ) {
167
+ const base = requirePairing ? pairedFrontendUrl(args) : frontendUrl(args);
168
+ const webUrl = base ? `${base}${route}` : "";
169
+ const installed = desktopAppInstalled(destination);
170
+
171
+ if (base === DEFAULT_FRONTEND && installed && desktopSupportsRouting(destination)) {
107
172
  return {
108
- href: `preman://open?source=push&route=${encodeURIComponent(route)}`,
173
+ href: `preman://open?source=${source}&route=${encodeURIComponent(route)}`,
109
174
  webUrl,
110
175
  surface: "desktop",
111
176
  };
112
177
  }
178
+
179
+ // Deliberately ahead of the "no dashboard is known" return below: not knowing
180
+ // which browser page holds the run says nothing about whether the app is
181
+ // installed, and raising it is still the most useful thing available.
182
+ if (preferApp && installed) {
183
+ return { href: `preman://open?source=${source}`, webUrl, surface: "desktop-raise" };
184
+ }
185
+
186
+ if (!base) return null;
187
+ if (base !== DEFAULT_FRONTEND) return { href: webUrl, webUrl, surface: "web-local" };
113
188
  return { href: webUrl, webUrl, surface: "web" };
114
189
  }
115
190
 
package/bin/runner.js CHANGED
@@ -38,7 +38,10 @@ import os from "node:os";
38
38
  import path from "node:path";
39
39
  import { fileURLToPath } from "node:url";
40
40
 
41
- import { CREDENTIALS_DIR, backendUrl, callBackendJson, cliInvocation, makeArgs, nowMs, onPath, packageVersion, resolveApiKey } from "./shared.js";
41
+ import { executeEvalRun } from "./eval.js";
42
+ import { discoverTarget, startAdapter, stepFeed } from "./eval_target.js";
43
+ import { resolveEvalSurface } from "./link.js";
44
+ import { CREDENTIALS_DIR, backendUrl, callBackendJson, cliInvocation, makeArgs, nowMs, onPath, openUrl, packageVersion, resolveApiKey } from "./shared.js";
42
45
 
43
46
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
44
47
 
@@ -70,6 +73,11 @@ const ERROR_CAP = 4_000;
70
73
  export const RUNNER_CAPABILITIES = {
71
74
  tool_call_telemetry_v1: true,
72
75
  tool_fact_manifest_v1: true,
76
+ // `find_eval_runner` requires this before it will lease an eval here, and the
77
+ // requirement is not cosmetic: the stream parser below ignores an event name
78
+ // it does not know, so a build without eval support would take the lease,
79
+ // execute nothing, and look like a device that went offline.
80
+ agent_eval_v1: true,
73
81
  transport: "cli",
74
82
  client: "premanmcp",
75
83
  };
@@ -398,6 +406,74 @@ async function callback(args, state, route, body, log) {
398
406
  }
399
407
  }
400
408
 
409
+ /**
410
+ * Give a leased job straight back, because this session will not run it.
411
+ *
412
+ * Both protocols reach a terminal state through their own `complete`, and both
413
+ * accept "this device declines" as a legitimate outcome — which is why this is a
414
+ * completion with `ok: false` rather than an attempt to un-lease.
415
+ */
416
+ async function declineJob(args, state, kind, job, log) {
417
+ const reason =
418
+ kind === "eval"
419
+ ? "this session is running a coding-agent job, not evals"
420
+ : "this session is waiting on an eval and will not run coding-agent jobs";
421
+ log(`declining ${kind} ${job.id}: ${reason}`);
422
+ if (kind === "eval") {
423
+ return callback(
424
+ args,
425
+ state,
426
+ `/workbench/coding-agent/local-runner/evals/${job.id}/complete`,
427
+ { lease_token: String(job.lease_token || ""), ok: false, error: reason },
428
+ log
429
+ );
430
+ }
431
+ return callback(
432
+ args,
433
+ state,
434
+ `/workbench/coding-agent/local-runner/jobs/${job.id}/complete`,
435
+ {
436
+ lease_token: String(job.lease_token || ""),
437
+ ok: false,
438
+ error: reason,
439
+ attempt_count: Math.max(1, Number(job.attempt_count) || 1),
440
+ tool_events: [],
441
+ tool_call_count: 0,
442
+ metadata: { transport: "cli", declined: true },
443
+ },
444
+ log
445
+ );
446
+ }
447
+
448
+ /**
449
+ * Hand a leased eval to `eval.js`, with this runner's way of making a callback.
450
+ *
451
+ * The callback shape is injected rather than imported there so the eval executor
452
+ * has no opinion about tokens or base URLs — this file stays the single place
453
+ * that knows a runner token goes on a runner request.
454
+ */
455
+ export async function runLeasedEval(args, state, job, { log = () => {}, headless = true } = {}) {
456
+ // One callback seam for both kinds of traffic this run produces: JSON for the
457
+ // progress and completion reports, multipart for the artifact uploads. The
458
+ // executor should not have to know which transport a given route wants.
459
+ const call = (route, body) =>
460
+ callBackendJson(args, "POST", route, {
461
+ token: state.runner_token,
462
+ ...(body instanceof FormData ? { form: body } : { json: body }),
463
+ });
464
+ const surface = resolveEvalSurface(args, { runId: job.id });
465
+ const result = await executeEvalRun(job, {
466
+ call,
467
+ log,
468
+ headless,
469
+ projectPath: state.project_path || process.cwd(),
470
+ surface: { ...surface, open: () => openUrl(surface.href) },
471
+ });
472
+ if (result.reason === "lease_lost") log(`eval ${job.id} was taken over or cancelled`);
473
+ else log(`eval ${job.id} finished: ${result.ok ? "ok" : "failed"}`);
474
+ return result;
475
+ }
476
+
401
477
  export async function executeJob(args, state, job, { log = () => {}, fullAccess = false } = {}) {
402
478
  const lease = String(job.lease_token || "");
403
479
  const jobRoute = `/workbench/coding-agent/local-runner/jobs/${job.id}`;
@@ -600,6 +676,21 @@ function defaultLog(message) {
600
676
  * leave a machine that reads as paired and never runs anything again. A missed
601
677
  * heartbeat only costs one TTL window: the next one puts this runner back online.
602
678
  */
679
+ /**
680
+ * What this machine currently offers, refreshed on every heartbeat.
681
+ *
682
+ * Set by the daemon once it has an adapter listening, and deliberately *not*
683
+ * persisted in the runner state file. An adapter is a socket this process
684
+ * opened; a URL for one that outlived the process it belonged to is a URL that
685
+ * points at nothing, and writing it to disk is how it would come back after a
686
+ * restart looking authoritative.
687
+ */
688
+ let advertised = null;
689
+
690
+ export function advertise(capabilities) {
691
+ advertised = capabilities;
692
+ }
693
+
603
694
  export async function sendHeartbeat(args, state, { busy = false, log = () => {} } = {}) {
604
695
  try {
605
696
  // Reported honestly because the matcher prefers idle runners: a busy device
@@ -608,7 +699,17 @@ export async function sendHeartbeat(args, state, { busy = false, log = () => {}
608
699
  args,
609
700
  "POST",
610
701
  "/workbench/coding-agent/local-runner/heartbeat",
611
- { token: state.runner_token, json: { state: busy ? "busy" : "idle" } }
702
+ {
703
+ token: state.runner_token,
704
+ json: {
705
+ state: busy ? "busy" : "idle",
706
+ // On every beat rather than once at registration, because both facts
707
+ // it carries expire: an agent restarted on a new port makes the old
708
+ // URL wrong, not merely stale, and a dashboard offering a Run button
709
+ // for it would fail on the press.
710
+ ...(advertised ? { capabilities: advertised } : {}),
711
+ },
712
+ }
612
713
  );
613
714
  return { revoked: result.status_code === 401 };
614
715
  } catch (error) {
@@ -617,6 +718,52 @@ export async function sendHeartbeat(args, state, { busy = false, log = () => {}
617
718
  }
618
719
  }
619
720
 
721
+ /**
722
+ * Find this machine's agent and hold an adapter open for as long as the daemon runs.
723
+ *
724
+ * Why the daemon does this at startup rather than when a run arrives: the
725
+ * dashboard's Run button has to know, *before* anybody presses it, that there is
726
+ * something here to measure and where it is. That is what turns "copy this
727
+ * command into a terminal" into a button, and it is only knowable if the address
728
+ * exists ahead of the request rather than being created in response to one.
729
+ *
730
+ * Failure here is not fatal and must not be. A machine paired for fix tasks is
731
+ * perfectly useful with no agent to evaluate on it, and refusing to start the
732
+ * daemon because discovery came up empty would break the feature this one is
733
+ * bolted onto. The daemon says so once and carries on without eval capability.
734
+ */
735
+ async function serveEvalTarget(args, { log }) {
736
+ let target;
737
+ try {
738
+ target = await discoverTarget(args, { log: () => {} });
739
+ } catch (error) {
740
+ log(`no agent to evaluate on this machine (${error.message}); eval runs will not be offered`);
741
+ return null;
742
+ }
743
+
744
+ try {
745
+ const adapter = await startAdapter(target, args, {
746
+ log: () => {},
747
+ onStep: (step) => stepFeed.push(step),
748
+ });
749
+ if (adapter.translated) {
750
+ // Same reason as the interactive path: the address is a socket this
751
+ // process opened moments ago, and assert-ai's refusal to fetch private
752
+ // addresses is right in general and wrong for exactly this.
753
+ process.env.ASSERT_ALLOW_PRIVATE_ENDPOINTS = "1";
754
+ }
755
+ log(`offering ${target.label} for evals`);
756
+ advertise({
757
+ agent_eval_v1: true,
758
+ eval_target: { url: adapter.url, label: target.label, how: target.how },
759
+ });
760
+ return adapter;
761
+ } catch (error) {
762
+ log(`could not make this machine's agent answerable (${error.message}); eval runs will not be offered`);
763
+ return null;
764
+ }
765
+ }
766
+
620
767
  /**
621
768
  * Hold the stream and run what it leases, until stopped.
622
769
  *
@@ -627,7 +774,17 @@ export async function sendHeartbeat(args, state, { busy = false, log = () => {}
627
774
  export async function runnerLoop(
628
775
  args,
629
776
  state,
630
- { log = defaultLog, once = false, stopWhen = null, fullAccess = false } = {}
777
+ {
778
+ log = defaultLog,
779
+ once = false,
780
+ stopWhen = null,
781
+ fullAccess = false,
782
+ only = "",
783
+ // Headless unless a caller says otherwise. `preman runner` is a background
784
+ // daemon; only the interactive `preman eval run` has somebody in front of it
785
+ // to hand a run over to.
786
+ headless = true,
787
+ } = {}
631
788
  ) {
632
789
  let stopped = false;
633
790
  let busy = false;
@@ -677,7 +834,11 @@ export async function runnerLoop(
677
834
  let terminal = "";
678
835
  const push = createSseParser((event, data) => {
679
836
  if (event === "connected") log(`connected as runner ${data.runner_id || state.runner_id}`);
680
- else if (event === "job") queue.push(data);
837
+ else if (event === "job") queue.push({ kind: "fix", job: data });
838
+ // A distinct name rather than a discriminator inside `job`, so a runner
839
+ // older than eval support ignores the frame instead of reading an eval
840
+ // as a coding task and looking for a fix task that is not there.
841
+ else if (event === "eval_job") queue.push({ kind: "eval", job: data });
681
842
  else if (event === "revoked" || event === "replaced") terminal = event;
682
843
  else if (event === "reconnect") terminal = "reconnect";
683
844
  });
@@ -687,11 +848,24 @@ export async function runnerLoop(
687
848
  for await (const chunk of response.body) {
688
849
  push(decoder.decode(chunk, { stream: true }));
689
850
  while (queue.length) {
690
- const job = queue.shift();
691
- log(`leased job ${job.id}`);
851
+ const { kind, job } = queue.shift();
852
+ // `preman eval run` is somebody waiting on one eval, not somebody
853
+ // volunteering their machine as a coding-agent runner. Executing a
854
+ // queued fix task because it happened to arrive down the same
855
+ // stream would edit files they did not ask to have edited.
856
+ //
857
+ // Declined rather than dropped: a leased job nobody answers sits
858
+ // until its lease expires, which delays the fix and tells whoever
859
+ // queued it nothing. Declining hands it straight back.
860
+ if (only && only !== kind) {
861
+ await declineJob(args, state, kind, job, log);
862
+ continue;
863
+ }
864
+ log(`leased ${kind === "eval" ? "eval" : "job"} ${job.id}`);
692
865
  busy = true;
693
866
  try {
694
- await executeJob(args, state, job, { log, fullAccess });
867
+ if (kind === "eval") await runLeasedEval(args, state, job, { log, headless });
868
+ else await executeJob(args, state, job, { log, fullAccess });
695
869
  } finally {
696
870
  busy = false;
697
871
  }
@@ -876,6 +1050,14 @@ async function startForeground(args, commandArgs) {
876
1050
  process.on("uncaughtException", die("uncaught exception"));
877
1051
  process.on("unhandledRejection", die("unhandled rejection"));
878
1052
 
1053
+ const adapter = await serveEvalTarget(args, { log });
1054
+
1055
+ // Pushed immediately rather than waiting for the loop's first tick. The
1056
+ // capability only reaches the server on a heartbeat, so without this the
1057
+ // dashboard spends the first interval telling somebody their machine has
1058
+ // found no agent — while the terminal beside it says it has.
1059
+ if (adapter) await sendHeartbeat(args, state, { log });
1060
+
879
1061
  try {
880
1062
  const result = await runnerLoop(args, state, {
881
1063
  log,
@@ -884,6 +1066,10 @@ async function startForeground(args, commandArgs) {
884
1066
  });
885
1067
  log(`stopped after ${result.jobsRun} job(s): ${result.reason}`);
886
1068
  } finally {
1069
+ // Closed on every path. The adapter is an unauthenticated door to this
1070
+ // machine's agent, held open only for as long as something is listening for
1071
+ // work behind it.
1072
+ if (adapter) await adapter.close().catch(() => {});
887
1073
  if (!shuttingDown) {
888
1074
  await goOffline(args, state, log);
889
1075
  rmSync(RUNNER_PID_FILE, { force: true });
package/bin/shared.js CHANGED
@@ -263,6 +263,39 @@ export function frontendUrl(args) {
263
263
  .replace(/\/+$/, "");
264
264
  }
265
265
 
266
+ /**
267
+ * The dashboard that can actually show a thing living on *this* backend, or
268
+ * nothing when there is no way to know.
269
+ *
270
+ * `frontendUrl` above states the pairing rule and then does not enforce it: the
271
+ * stored `frontend_url` is used whatever backend this invocation was pointed
272
+ * at. For prose that is harmless -- "sign in at ..." wants the dashboard the
273
+ * customer normally uses, and always has one to name. For a *link to a specific
274
+ * run* it is a wrong answer dressed as a right one: point `--backend` at a
275
+ * local API and the run gets a production URL where it does not exist, so the
276
+ * link resolves, loads, and shows nothing. That is worse than no link, because
277
+ * an empty dashboard reads as a lost run rather than a bad address.
278
+ *
279
+ * So this one refuses to guess. An unrecognised backend with no `--frontend`
280
+ * and no matching login returns `""`, and callers omit the link. Saying nothing
281
+ * about where to look is honest; naming somewhere the run has never been is not.
282
+ */
283
+ export function pairedFrontendUrl(args) {
284
+ const explicit = args.value("--frontend", process.env.PREMAN_FRONTEND || "");
285
+ if (explicit) return explicit.replace(/\/+$/, "");
286
+
287
+ const backend = backendUrl(args);
288
+ const login = activeLogin(args);
289
+ const paired = String(login?.backend_url || "").replace(/\/+$/, "");
290
+ // Only while this invocation is talking to the deployment the key was minted
291
+ // against -- which is exactly the condition `frontendUrl`'s docstring names.
292
+ if (login?.frontend_url && paired && paired === backend) {
293
+ return String(login.frontend_url).replace(/\/+$/, "");
294
+ }
295
+ if (backend === DEFAULT_BACKEND) return DEFAULT_FRONTEND;
296
+ return "";
297
+ }
298
+
266
299
  /**
267
300
  * Open a URL in the user's default browser, best effort.
268
301
  *
@@ -428,7 +461,7 @@ export async function callBackendJson(
428
461
  args,
429
462
  method,
430
463
  routePath,
431
- { json, token, query, headers: extraHeaders } = {}
464
+ { json, form, token, query, headers: extraHeaders } = {}
432
465
  ) {
433
466
  const url = new URL(routePath.replace(/^\/+/, ""), `${backendUrl(args)}/`);
434
467
  if (query) {
@@ -439,6 +472,8 @@ export async function callBackendJson(
439
472
 
440
473
  const headers = { Accept: "application/json" };
441
474
  const hasBody = json !== undefined && json !== null;
475
+ // No Content-Type for a form: fetch sets it from the FormData, and the
476
+ // multipart boundary it appends is not something a caller can supply.
442
477
  if (hasBody) headers["Content-Type"] = "application/json";
443
478
  if (token) headers.Authorization = `Bearer ${token}`;
444
479
  if (extraHeaders) {
@@ -450,7 +485,7 @@ export async function callBackendJson(
450
485
  const resp = await fetch(url, {
451
486
  method,
452
487
  headers,
453
- body: hasBody ? JSON.stringify(json) : undefined,
488
+ body: form !== undefined && form !== null ? form : hasBody ? JSON.stringify(json) : undefined,
454
489
  });
455
490
  const text = await resp.text();
456
491
  let body = {};
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "premanmcp",
3
- "version": "1.0.4",
3
+ "version": "1.1.0",
4
4
  "description": "PreMan CLI and stdio proxy for PreMan's hosted MCP server",
5
5
  "type": "module",
6
6
  "bin": {
@@ -15,7 +15,7 @@
15
15
  "test": "npm run build && npm run test:proxy",
16
16
  "test:proxy": "node --test scripts/smoke-proxy.mjs",
17
17
  "test:connect": "node scripts/smoke-connect.mjs",
18
- "test:node": "node --test --test-timeout=90000 scripts/smoke-account.mjs scripts/smoke-cli-entrypoint.mjs scripts/smoke-launcher-config.mjs scripts/smoke-runner.mjs scripts/smoke-repo-config.mjs scripts/smoke-onboard.mjs scripts/smoke-onboard-opening.mjs scripts/smoke-local-detect.mjs scripts/smoke-prepush-hook.mjs scripts/smoke-cli-identity.mjs scripts/smoke-runner-heartbeat.mjs scripts/smoke-verify-prepush.mjs scripts/smoke-push-diff.mjs scripts/smoke-progress-reporter.mjs scripts/smoke-verify-plan.mjs scripts/smoke-install-desktop.mjs scripts/smoke-desktop-session.mjs scripts/smoke-api-tools.mjs scripts/smoke-tests-workbench.mjs scripts/smoke-bin-scope.mjs scripts/smoke-shared-errors.mjs",
18
+ "test:node": "node --test --test-timeout=90000 scripts/smoke-account.mjs scripts/smoke-cli-entrypoint.mjs scripts/smoke-launcher-config.mjs scripts/smoke-runner.mjs scripts/smoke-eval.mjs scripts/smoke-repo-config.mjs scripts/smoke-onboard.mjs scripts/smoke-onboard-opening.mjs scripts/smoke-local-detect.mjs scripts/smoke-prepush-hook.mjs scripts/smoke-cli-identity.mjs scripts/smoke-runner-heartbeat.mjs scripts/smoke-verify-prepush.mjs scripts/smoke-push-diff.mjs scripts/smoke-progress-reporter.mjs scripts/smoke-verify-plan.mjs scripts/smoke-install-desktop.mjs scripts/smoke-desktop-session.mjs scripts/smoke-api-tools.mjs scripts/smoke-tests-workbench.mjs scripts/smoke-bin-scope.mjs scripts/smoke-shared-errors.mjs",
19
19
  "test:dmg": "node --test --test-timeout=300000 scripts/smoke-install-desktop-volume.mjs"
20
20
  },
21
21
  "devDependencies": {