premanmcp 1.1.4 → 1.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -0
- package/bin/agent.js +1925 -0
- package/bin/agent_dialects.js +385 -0
- package/bin/cli.js +136 -17
- package/bin/eval.js +247 -15
- package/bin/eval_harness.py +6 -2
- package/bin/eval_target.js +102 -18
- package/bin/link.js +40 -13
- package/bin/predict.js +857 -0
- package/bin/runner.js +4 -1
- package/bin/target.js +505 -0
- package/dist/server.js +6 -0
- package/package.json +2 -2
|
@@ -0,0 +1,385 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Talking to an agent PreMan did not write.
|
|
3
|
+
*
|
|
4
|
+
* Everything else in the eval path assumes the agent under test speaks one
|
|
5
|
+
* shape: `POST {"message","history"}` answered with `{"response"}`. That is
|
|
6
|
+
* PreMan's own workbench, and no customer's agent speaks it. Their agent is a
|
|
7
|
+
* LangGraph app on `/invoke`, an OpenAI-compatible route on
|
|
8
|
+
* `/v1/chat/completions`, a Next.js handler on `/api/chat`, or a Flask function
|
|
9
|
+
* somebody wrote on a Tuesday — and asking a developer which of those they have
|
|
10
|
+
* is asking them to answer a question their own code already answers.
|
|
11
|
+
*
|
|
12
|
+
* So this module asks the agent instead. It is deliberately a *probe* rather
|
|
13
|
+
* than a framework detector: reading a repository tells you what was imported,
|
|
14
|
+
* while a request tells you what actually answers, and those differ the moment
|
|
15
|
+
* anything sits in front of the app. One harmless message, a small table of
|
|
16
|
+
* shapes, and the first one that produces text wins.
|
|
17
|
+
*
|
|
18
|
+
* Two things bound it, because this runs against a developer's machine and a
|
|
19
|
+
* discovery pass that hangs is worse than one that finds nothing: every attempt
|
|
20
|
+
* has its own timeout, and the whole sweep stops at `MAX_ATTEMPTS`.
|
|
21
|
+
*
|
|
22
|
+
* What it cannot do is invent visibility. If an agent answers with prose and
|
|
23
|
+
* nothing else, the transcript a judge reads has no tool calls in it, and the
|
|
24
|
+
* honest report is that the flow was measured without them -- see
|
|
25
|
+
* `toolEventsFrom` for the shapes that do carry them.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
/** Per request. Long enough for a cold model call, short enough to sweep. */
|
|
29
|
+
export const PROBE_TIMEOUT_MS = 12_000;
|
|
30
|
+
|
|
31
|
+
/** A turn during a real run, where a slow answer is the agent thinking. */
|
|
32
|
+
export const TURN_TIMEOUT_MS = 120_000;
|
|
33
|
+
|
|
34
|
+
/** The whole sweep, so an agent that 200s everything cannot run forever. */
|
|
35
|
+
export const MAX_ATTEMPTS = 40;
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* What the probe says.
|
|
39
|
+
*
|
|
40
|
+
* It names itself and asks for nothing, because it is a real message to
|
|
41
|
+
* somebody's real agent: it may be logged, it may be answered by a model that
|
|
42
|
+
* costs money, and it may end up in a customer-facing transcript if the agent
|
|
43
|
+
* under test is wired to one. "Hello" would be cheaper to type and worse to
|
|
44
|
+
* find in a support queue.
|
|
45
|
+
*/
|
|
46
|
+
export const PROBE_MESSAGE =
|
|
47
|
+
"This is PreMan checking that it can reach this agent. No action needed — please reply briefly.";
|
|
48
|
+
|
|
49
|
+
/** Where a chat agent usually listens, most specific first. */
|
|
50
|
+
export const PROBE_PATHS = [
|
|
51
|
+
"/chat",
|
|
52
|
+
"/api/chat",
|
|
53
|
+
"/v1/chat/completions",
|
|
54
|
+
"/chat/completions",
|
|
55
|
+
"/invoke",
|
|
56
|
+
"/agent",
|
|
57
|
+
"/api/agent",
|
|
58
|
+
"/message",
|
|
59
|
+
"/query",
|
|
60
|
+
"/",
|
|
61
|
+
];
|
|
62
|
+
|
|
63
|
+
const asText = (value) => (typeof value === "string" && value.trim() ? value : "");
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* The shortest thing the plain-text dialect will believe is an answer.
|
|
67
|
+
*
|
|
68
|
+
* An agent asked a full sentence answers with one. `OK`, `healthy` and `pong`
|
|
69
|
+
* are the other things that answer a POST with 200 and plain text, and every
|
|
70
|
+
* one of them is a health check.
|
|
71
|
+
*/
|
|
72
|
+
const MIN_PROSE = 24;
|
|
73
|
+
|
|
74
|
+
const looksLikeMarkup = (value) => /^\s*<(!doctype|html|\?xml|svg)/i.test(String(value || ""));
|
|
75
|
+
|
|
76
|
+
/** The first field of a set that carries text, whatever the shape called it. */
|
|
77
|
+
function firstText(body, keys) {
|
|
78
|
+
if (typeof body === "string") return asText(body);
|
|
79
|
+
if (!body || typeof body !== "object") return "";
|
|
80
|
+
for (const key of keys) {
|
|
81
|
+
const found = asText(body[key]);
|
|
82
|
+
if (found) return found;
|
|
83
|
+
}
|
|
84
|
+
return "";
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* The shapes, in the order they are tried.
|
|
89
|
+
*
|
|
90
|
+
* Ordered by how specific the answer is rather than by popularity: a route that
|
|
91
|
+
* answers an OpenAI-shaped request with an OpenAI-shaped body has told you what
|
|
92
|
+
* it is, while `{"message": …}` → `{"response": …}` is a shape half the
|
|
93
|
+
* internet uses for half a dozen different things. Being wrong about the
|
|
94
|
+
* dialect is cheap here (the turn produces no text and the sweep moves on) and
|
|
95
|
+
* expensive later (a run that measures an agent's error handler), so the
|
|
96
|
+
* specific ones go first.
|
|
97
|
+
*/
|
|
98
|
+
export const DIALECTS = [
|
|
99
|
+
{
|
|
100
|
+
name: "openai-chat",
|
|
101
|
+
summary: "OpenAI-compatible chat completions",
|
|
102
|
+
body: (message, history) => ({
|
|
103
|
+
model: "preman-eval",
|
|
104
|
+
stream: false,
|
|
105
|
+
messages: [
|
|
106
|
+
...history.map((turn) => ({
|
|
107
|
+
role: turn.role === "assistant" ? "assistant" : "user",
|
|
108
|
+
content: String(turn.content ?? ""),
|
|
109
|
+
})),
|
|
110
|
+
{ role: "user", content: message },
|
|
111
|
+
],
|
|
112
|
+
}),
|
|
113
|
+
read: (body) => {
|
|
114
|
+
const choice = body?.choices?.[0]?.message;
|
|
115
|
+
return { text: asText(choice?.content), calls: choice?.tool_calls };
|
|
116
|
+
},
|
|
117
|
+
},
|
|
118
|
+
{
|
|
119
|
+
name: "langserve",
|
|
120
|
+
summary: "LangServe / LangGraph invoke",
|
|
121
|
+
body: (message, history) => ({
|
|
122
|
+
input: { input: message, messages: history, chat_history: history },
|
|
123
|
+
}),
|
|
124
|
+
read: (body) => {
|
|
125
|
+
const output = body?.output ?? body?.result;
|
|
126
|
+
const text =
|
|
127
|
+
firstText(output, ["output", "content", "answer", "text", "response"]) ||
|
|
128
|
+
asText(output) ||
|
|
129
|
+
firstText(body, ["output", "content", "answer", "text"]);
|
|
130
|
+
return {
|
|
131
|
+
text,
|
|
132
|
+
calls: body?.intermediate_steps || output?.intermediate_steps || output?.tool_calls,
|
|
133
|
+
};
|
|
134
|
+
},
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
name: "messages",
|
|
138
|
+
summary: "a messages array, answered with a message",
|
|
139
|
+
body: (message, history) => ({
|
|
140
|
+
messages: [
|
|
141
|
+
...history.map((turn) => ({
|
|
142
|
+
role: turn.role === "assistant" ? "assistant" : "user",
|
|
143
|
+
content: String(turn.content ?? ""),
|
|
144
|
+
})),
|
|
145
|
+
{ role: "user", content: message },
|
|
146
|
+
],
|
|
147
|
+
}),
|
|
148
|
+
read: (body) => {
|
|
149
|
+
const last = Array.isArray(body?.messages) ? body.messages[body.messages.length - 1] : null;
|
|
150
|
+
return {
|
|
151
|
+
text:
|
|
152
|
+
firstText(body, ["response", "reply", "output", "answer", "content", "text", "message"]) ||
|
|
153
|
+
asText(last?.content),
|
|
154
|
+
calls: body?.tool_calls || last?.tool_calls || body?.steps || body?.events,
|
|
155
|
+
};
|
|
156
|
+
},
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
name: "message",
|
|
160
|
+
summary: "a message and a history, answered with text",
|
|
161
|
+
body: (message, history) => ({ message, history }),
|
|
162
|
+
read: readLoosely,
|
|
163
|
+
},
|
|
164
|
+
{
|
|
165
|
+
name: "input",
|
|
166
|
+
summary: "an input field, answered with text",
|
|
167
|
+
body: (message) => ({ input: message }),
|
|
168
|
+
read: readLoosely,
|
|
169
|
+
},
|
|
170
|
+
{
|
|
171
|
+
name: "query",
|
|
172
|
+
summary: "a query field, answered with text",
|
|
173
|
+
body: (message) => ({ query: message }),
|
|
174
|
+
read: readLoosely,
|
|
175
|
+
},
|
|
176
|
+
{
|
|
177
|
+
name: "prompt",
|
|
178
|
+
summary: "a prompt field, answered with text",
|
|
179
|
+
body: (message) => ({ prompt: message }),
|
|
180
|
+
read: readLoosely,
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
name: "text",
|
|
184
|
+
summary: "plain text in, plain text out",
|
|
185
|
+
contentType: "text/plain",
|
|
186
|
+
body: (message) => message,
|
|
187
|
+
// The greediest dialect, and therefore the most guarded. A body that parses
|
|
188
|
+
// is a structured answer this one cannot read; markup is a web page; and a
|
|
189
|
+
// two-word answer is a health check. All three were being adopted as the
|
|
190
|
+
// agent under test -- and since discovery tries port 3000 first, the most
|
|
191
|
+
// likely victim was a developer's own frontend, which would then have been
|
|
192
|
+
// measured for months for customer-service behaviour it does not have.
|
|
193
|
+
read: (body, raw, contentType = "") => {
|
|
194
|
+
if (body !== null) return { text: "", calls: null };
|
|
195
|
+
if (/html|xml/i.test(contentType)) return { text: "", calls: null };
|
|
196
|
+
const text = asText(raw);
|
|
197
|
+
if (looksLikeMarkup(text) || text.length < MIN_PROSE) return { text: "", calls: null };
|
|
198
|
+
return { text, calls: null };
|
|
199
|
+
},
|
|
200
|
+
},
|
|
201
|
+
];
|
|
202
|
+
|
|
203
|
+
/**
|
|
204
|
+
* The bodies that answer in prose rather than in a schema.
|
|
205
|
+
*
|
|
206
|
+
* One reader for all of them because the request is what distinguishes these
|
|
207
|
+
* dialects, not the response: an agent taking `{"message"}` and one taking
|
|
208
|
+
* `{"query"}` answer the same half-dozen field names.
|
|
209
|
+
*/
|
|
210
|
+
function readLoosely(body) {
|
|
211
|
+
return {
|
|
212
|
+
text: firstText(body, [
|
|
213
|
+
"response",
|
|
214
|
+
"reply",
|
|
215
|
+
"output",
|
|
216
|
+
"answer",
|
|
217
|
+
"content",
|
|
218
|
+
"text",
|
|
219
|
+
"message",
|
|
220
|
+
"result",
|
|
221
|
+
]),
|
|
222
|
+
calls: body?.tool_calls || body?.events || body?.steps || body?.intermediate_steps,
|
|
223
|
+
};
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/**
|
|
227
|
+
* Tool calls, from whatever the agent happened to call them.
|
|
228
|
+
*
|
|
229
|
+
* The output shape is assert-ai's `AdapterEvent` and the details are not
|
|
230
|
+
* negotiable -- `eval_target.js::eventsFrom` documents what was learned the
|
|
231
|
+
* hard way: its normalizer keys on `role`, silently drops anything whose role
|
|
232
|
+
* is not one of its three, and only `content`, `tool_name` and `tool_args`
|
|
233
|
+
* survive into the artifacts a judge reads. A plausible-looking event with the
|
|
234
|
+
* wrong role produces exactly the transcript of an agent that called nothing.
|
|
235
|
+
*
|
|
236
|
+
* Anything unrecognised returns `[]` rather than a guess. A judge asked whether
|
|
237
|
+
* the agent looked something up before answering is better served by a
|
|
238
|
+
* transcript that shows nothing than by one that shows the wrong thing.
|
|
239
|
+
*/
|
|
240
|
+
export function toolEventsFrom(raw) {
|
|
241
|
+
const list = Array.isArray(raw) ? raw : [];
|
|
242
|
+
const events = [];
|
|
243
|
+
for (const item of list) {
|
|
244
|
+
if (!item) continue;
|
|
245
|
+
|
|
246
|
+
// LangChain's `intermediate_steps`: [action, observation] pairs.
|
|
247
|
+
if (Array.isArray(item)) {
|
|
248
|
+
const [action] = item;
|
|
249
|
+
const name = String(action?.tool || action?.name || "");
|
|
250
|
+
if (!name) continue;
|
|
251
|
+
events.push(event(name, action?.tool_input ?? action?.args));
|
|
252
|
+
continue;
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// OpenAI: the name and the arguments live under `function`, and the
|
|
256
|
+
// arguments are a JSON *string* rather than an object.
|
|
257
|
+
const fn = item.function;
|
|
258
|
+
const name = String(item.tool_name || item.name || item.tool || fn?.name || "");
|
|
259
|
+
if (!name) continue;
|
|
260
|
+
let args = item.tool_args ?? item.arguments ?? item.args ?? item.input ?? fn?.arguments;
|
|
261
|
+
if (typeof args === "string") {
|
|
262
|
+
try {
|
|
263
|
+
args = JSON.parse(args);
|
|
264
|
+
} catch {
|
|
265
|
+
args = { arguments: args };
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
events.push(event(name, args, item.executed));
|
|
269
|
+
}
|
|
270
|
+
return events;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
function event(name, args, executed) {
|
|
274
|
+
return {
|
|
275
|
+
role: "tool_result",
|
|
276
|
+
content:
|
|
277
|
+
executed === false
|
|
278
|
+
? `executed: false — ${name} was proposed and did not run.`
|
|
279
|
+
: `executed: true — ${name} ran and returned its result to the agent.`,
|
|
280
|
+
tool_name: name,
|
|
281
|
+
tool_args: args && typeof args === "object" && !Array.isArray(args) ? args : {},
|
|
282
|
+
};
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/** One request, with its own timeout, returning the parsed body and the text. */
|
|
286
|
+
async function attempt(url, dialect, message, history, timeoutMs) {
|
|
287
|
+
const controller = new AbortController();
|
|
288
|
+
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
|
289
|
+
try {
|
|
290
|
+
const payload = dialect.body(message, history);
|
|
291
|
+
const resp = await fetch(url, {
|
|
292
|
+
method: "POST",
|
|
293
|
+
signal: controller.signal,
|
|
294
|
+
headers: { "content-type": dialect.contentType || "application/json" },
|
|
295
|
+
body: dialect.contentType === "text/plain" ? String(payload) : JSON.stringify(payload),
|
|
296
|
+
});
|
|
297
|
+
const raw = await resp.text();
|
|
298
|
+
if (!resp.ok) return { ok: false, status: resp.status, why: `${resp.status}` };
|
|
299
|
+
let parsed = null;
|
|
300
|
+
try {
|
|
301
|
+
parsed = JSON.parse(raw);
|
|
302
|
+
} catch {
|
|
303
|
+
parsed = null;
|
|
304
|
+
}
|
|
305
|
+
const { text, calls } = dialect.read(parsed, raw, resp.headers.get("content-type") || "");
|
|
306
|
+
if (!text) return { ok: false, why: "answered, but nothing in it reads as a reply" };
|
|
307
|
+
return { ok: true, text, events: toolEventsFrom(calls), body: parsed };
|
|
308
|
+
} catch (error) {
|
|
309
|
+
return {
|
|
310
|
+
ok: false,
|
|
311
|
+
status: 0,
|
|
312
|
+
why: error.name === "AbortError" ? "no answer in time" : error.message,
|
|
313
|
+
};
|
|
314
|
+
} finally {
|
|
315
|
+
clearTimeout(timer);
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Find the address and the shape that this agent answers to.
|
|
321
|
+
*
|
|
322
|
+
* Returns `{ url, path, dialect, sample }` for the first combination that
|
|
323
|
+
* produces text, or `{ tried }` when nothing did -- the list of what was tried
|
|
324
|
+
* and what each one said, because "we could not reach your agent" without it is
|
|
325
|
+
* the least actionable sentence in a CLI.
|
|
326
|
+
*/
|
|
327
|
+
export async function probeAgent(
|
|
328
|
+
baseUrl,
|
|
329
|
+
{ paths = PROBE_PATHS, dialects = DIALECTS, message = PROBE_MESSAGE, timeoutMs = PROBE_TIMEOUT_MS, log = () => {} } = {}
|
|
330
|
+
) {
|
|
331
|
+
const base = String(baseUrl || "").replace(/\/+$/, "");
|
|
332
|
+
const tried = [];
|
|
333
|
+
let attempts = 0;
|
|
334
|
+
|
|
335
|
+
for (const path of paths) {
|
|
336
|
+
let first = true;
|
|
337
|
+
for (const dialect of dialects) {
|
|
338
|
+
if (attempts >= MAX_ATTEMPTS) return { tried, exhausted: true };
|
|
339
|
+
attempts += 1;
|
|
340
|
+
const url = `${base}${path}`;
|
|
341
|
+
const result = await attempt(url, dialect, message, [], timeoutMs);
|
|
342
|
+
// A path that is not there is not there for any dialect. Without this the
|
|
343
|
+
// sweep spends eight requests proving one 404 eight times, which on ten
|
|
344
|
+
// candidate paths is most of its budget and all of the wait.
|
|
345
|
+
if (first && !result.ok && (result.status === 404 || result.status === 405)) {
|
|
346
|
+
tried.push({ path, dialect: dialect.name, why: `${result.status}, so this path is not the agent` });
|
|
347
|
+
break;
|
|
348
|
+
}
|
|
349
|
+
first = false;
|
|
350
|
+
if (result.ok) {
|
|
351
|
+
log(`${path} answers ${dialect.name}`);
|
|
352
|
+
return {
|
|
353
|
+
url: base,
|
|
354
|
+
path,
|
|
355
|
+
dialect: dialect.name,
|
|
356
|
+
sample: result.text.slice(0, 300),
|
|
357
|
+
events: result.events,
|
|
358
|
+
};
|
|
359
|
+
}
|
|
360
|
+
tried.push({ path, dialect: dialect.name, why: result.why });
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
return { tried };
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/** The dialect by name, or null. */
|
|
367
|
+
export function dialectNamed(name) {
|
|
368
|
+
return DIALECTS.find((entry) => entry.name === name) || null;
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
/**
|
|
372
|
+
* One turn against a target already probed, in the shape the eval expects.
|
|
373
|
+
*
|
|
374
|
+
* Throws rather than returning an error string: the adapter above this decides
|
|
375
|
+
* what an unreachable agent means for a case, and it is the only layer that
|
|
376
|
+
* knows whether the run can survive it.
|
|
377
|
+
*/
|
|
378
|
+
export async function askAgent(target, message, history = [], { timeoutMs = TURN_TIMEOUT_MS } = {}) {
|
|
379
|
+
const dialect = dialectNamed(target.dialect);
|
|
380
|
+
if (!dialect) throw new Error(`no dialect called ${target.dialect}`);
|
|
381
|
+
const url = `${String(target.url).replace(/\/+$/, "")}${target.path || ""}`;
|
|
382
|
+
const result = await attempt(url, dialect, message, history, timeoutMs);
|
|
383
|
+
if (!result.ok) throw new Error(result.why);
|
|
384
|
+
return { text: result.text, events: result.events };
|
|
385
|
+
}
|
package/bin/cli.js
CHANGED
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
* npm exec -y premanmcp@latest --
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
|
-
import { spawn } from "node:child_process";
|
|
13
|
+
import { spawn, spawnSync } from "node:child_process";
|
|
14
14
|
import { existsSync } from "node:fs";
|
|
15
15
|
import os from "node:os";
|
|
16
16
|
import path from "node:path";
|
|
@@ -34,9 +34,11 @@ import {
|
|
|
34
34
|
slackCommand,
|
|
35
35
|
} from "./integrations.js";
|
|
36
36
|
import { HOSTED_HELP, linkCommand, runCommand, toolsCommand } from "./hosted.js";
|
|
37
|
+
import { PREDICT_HELP, predictCommand } from "./predict.js";
|
|
37
38
|
import { STATUS_HELP, statusCommand } from "./status.js";
|
|
38
39
|
import { HOOK_HELP, hookCommand, installHook, scheduleHookRepair } from "./hook.js";
|
|
39
40
|
import { EVAL_HELP } from "./eval.js";
|
|
41
|
+
import { TARGET_HELP, targetCommand } from "./target.js";
|
|
40
42
|
import { RUNNER_HELP, runnerCommand } from "./runner.js";
|
|
41
43
|
import { VERIFY_HELP, verifyCommand } from "./verify.js";
|
|
42
44
|
import {
|
|
@@ -46,9 +48,11 @@ import {
|
|
|
46
48
|
openDesktopSignedIn,
|
|
47
49
|
} from "./desktop.js";
|
|
48
50
|
import { ACCOUNT_HELP, doctorCommand, loginBrowser, logoutCommand, watchCommand } from "./account.js";
|
|
51
|
+
import { AGENT_HELP, agentCommand } from "./agent.js";
|
|
49
52
|
import {
|
|
50
53
|
CREDENTIALS_FILE,
|
|
51
54
|
cliInvocation,
|
|
55
|
+
packageVersion,
|
|
52
56
|
DEFAULT_BACKEND,
|
|
53
57
|
DEFAULT_FRONTEND,
|
|
54
58
|
authenticateTerminal,
|
|
@@ -61,18 +65,34 @@ import {
|
|
|
61
65
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
62
66
|
const ROOT = path.join(__dirname, "..");
|
|
63
67
|
|
|
68
|
+
// What this process is called wherever processes are listed: `ps`, `top`, an
|
|
69
|
+
// editor's process picker, a "what is using this port" hunt. Without it every
|
|
70
|
+
// one of them says `node`, which is true of the interpreter and useless to
|
|
71
|
+
// somebody with three of them running -- the same reason the session sets an
|
|
72
|
+
// OSC title on the terminal tab. Set before anything else so a crash in
|
|
73
|
+
// argument parsing is still attributable.
|
|
74
|
+
process.title = "preman";
|
|
75
|
+
|
|
64
76
|
const args = process.argv.slice(2);
|
|
65
|
-
// A bare `preman` in a terminal
|
|
66
|
-
//
|
|
67
|
-
//
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
77
|
+
// A bare `preman` in a terminal opens the agent session: that is what the
|
|
78
|
+
// product is in a terminal, and it signs the caller in on the way if they are
|
|
79
|
+
// new. The same argv with piped or redirected stdin is an MCP host launching
|
|
80
|
+
// the stdio server, which must keep working -- so the tty, not the argv, is
|
|
81
|
+
// what separates a person from a host.
|
|
82
|
+
//
|
|
83
|
+
// A leading flag follows the same rule (`preman --continue`, `preman -p "..."`),
|
|
84
|
+
// because those flags belong to the session. `preman start ...` stays the
|
|
85
|
+
// explicit way to ask for the server from a terminal.
|
|
86
|
+
const bare = args.length === 0;
|
|
87
|
+
const leadingFlag = Boolean(args[0]) && args[0].startsWith("-");
|
|
88
|
+
const command = args[0] === "--help" || args[0] === "-h"
|
|
71
89
|
? "help"
|
|
72
|
-
:
|
|
73
|
-
?
|
|
74
|
-
:
|
|
75
|
-
|
|
90
|
+
: bare || leadingFlag
|
|
91
|
+
? (process.stdin.isTTY ? "agent" : "start")
|
|
92
|
+
: args[0];
|
|
93
|
+
// A session or the server started this way owns every argument; a named
|
|
94
|
+
// subcommand owns everything after its name.
|
|
95
|
+
const commandArgs = command === "start" || bare || leadingFlag ? args : args.slice(1);
|
|
76
96
|
const cliArgs = makeArgs(commandArgs);
|
|
77
97
|
|
|
78
98
|
function argValue(name, fallback = "") {
|
|
@@ -91,6 +111,8 @@ function printHelp() {
|
|
|
91
111
|
// for a global install and 38 for npm exec, so a fixed layout is wrong for
|
|
92
112
|
// one of them.
|
|
93
113
|
const usage = [
|
|
114
|
+
["", "Chat with the PreMan agent in this terminal"],
|
|
115
|
+
["chat [message] [--continue]", "The same session, named explicitly"],
|
|
94
116
|
["status", "Healthy / failing / recently fixed endpoints"],
|
|
95
117
|
["verify [options]", "Test endpoints against your local app"],
|
|
96
118
|
["hook install|uninstall|status", "Manage the git pre-push hook"],
|
|
@@ -98,6 +120,7 @@ function printHelp() {
|
|
|
98
120
|
["eval run|doctor", "Run PreMan's queued agent eval against your local agent"],
|
|
99
121
|
["doctor", "Diagnose credentials, backend, target, integrations"],
|
|
100
122
|
["install-desktop", "Download and install the PreMan desktop app"],
|
|
123
|
+
["predict [options]", "Local dev: bring the stack up and measure the forecaster"],
|
|
101
124
|
["onboard", "Create an account, install the app signed in, then integrations"],
|
|
102
125
|
["connect [options]", "Pick a coding agent and connect it (optional)"],
|
|
103
126
|
["dispatch [options]", "Let PreMan start agent runs for you"],
|
|
@@ -105,8 +128,10 @@ function printHelp() {
|
|
|
105
128
|
["login [--browser]", "Create/login to PreMan from the terminal"],
|
|
106
129
|
["logout", "Delete stored CLI credentials"],
|
|
107
130
|
["watch <run> <integration>", "Follow a push simulation live"],
|
|
108
|
-
["install
|
|
109
|
-
["", "
|
|
131
|
+
["install", "Set PreMan up: account, desktop app, integrations"],
|
|
132
|
+
["install --cursor", "Write Cursor's MCP config only (the old `install`)"],
|
|
133
|
+
["update", "Update this CLI to the newest published version"],
|
|
134
|
+
["start", "Start the PreMan MCP server (what an MCP host runs)"],
|
|
110
135
|
["link|tools|run ...", "Drive a published hosted MCP"],
|
|
111
136
|
["endpoints list|discover|setup ...", "Discover, list, and set up API endpoints"],
|
|
112
137
|
["test --agent [--only <id>]", "Measure your agent against the behaviours you turned on"],
|
|
@@ -122,7 +147,7 @@ function printHelp() {
|
|
|
122
147
|
|
|
123
148
|
Usage:
|
|
124
149
|
${usageLines}
|
|
125
|
-
${INTEGRATIONS_HELP}${CONNECT_HELP}${DISPATCH_HELP}${STATUS_HELP}${VERIFY_HELP}${HOOK_HELP}${RUNNER_HELP}${EVAL_HELP}${ACCOUNT_HELP}${DESKTOP_HELP}${ENDPOINTS_HELP}${TEST_HELP}${TESTS_HELP}
|
|
150
|
+
${AGENT_HELP}${INTEGRATIONS_HELP}${CONNECT_HELP}${DISPATCH_HELP}${STATUS_HELP}${VERIFY_HELP}${HOOK_HELP}${RUNNER_HELP}${EVAL_HELP}${TARGET_HELP}${ACCOUNT_HELP}${DESKTOP_HELP}${PREDICT_HELP}${ENDPOINTS_HELP}${TEST_HELP}${TESTS_HELP}
|
|
126
151
|
Login options:
|
|
127
152
|
--email <email> Pre-fill the email prompt
|
|
128
153
|
--backend <url> PreMan backend URL. Defaults to ${DEFAULT_BACKEND}
|
|
@@ -138,6 +163,8 @@ Install options (Cursor only — prefer '${cli} connect'):
|
|
|
138
163
|
--print Print the config instead of writing it
|
|
139
164
|
|
|
140
165
|
Examples:
|
|
166
|
+
${cli}
|
|
167
|
+
${cli} chat "which endpoints broke since Friday?"
|
|
141
168
|
npm exec -y premanmcp@latest -- connect
|
|
142
169
|
${cli} connect --agent claude-code
|
|
143
170
|
${cli} login
|
|
@@ -175,6 +202,60 @@ You can now run:
|
|
|
175
202
|
`);
|
|
176
203
|
}
|
|
177
204
|
|
|
205
|
+
// The published package. Not the same word as the command, which is exactly
|
|
206
|
+
// why `preman update` exists rather than a line of npm for people to memorise.
|
|
207
|
+
const PACKAGE_NAME = "premanmcp";
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Update this CLI in place.
|
|
211
|
+
*
|
|
212
|
+
* `npm install -g premanmcp@latest` is the whole mechanism, and the only
|
|
213
|
+
* reason to wrap it is that nobody should have to remember the package name
|
|
214
|
+
* is not the command name. Run through `npx`/`npm exec` there is nothing
|
|
215
|
+
* installed to update -- that form already fetches the newest version every
|
|
216
|
+
* time -- so say that rather than silently installing something global the
|
|
217
|
+
* person did not ask for.
|
|
218
|
+
*/
|
|
219
|
+
async function updateCommand(commandArgs = []) {
|
|
220
|
+
const args = makeArgs(commandArgs);
|
|
221
|
+
const wanted = args.value("--version", "latest");
|
|
222
|
+
const before = packageVersion();
|
|
223
|
+
const here = fileURLToPath(import.meta.url);
|
|
224
|
+
|
|
225
|
+
if (/[\\/]_npx[\\/]/.test(here)) {
|
|
226
|
+
process.stdout.write(
|
|
227
|
+
`Running from npm exec, which fetches the newest version each time — ` +
|
|
228
|
+
`there is nothing installed to update.\n` +
|
|
229
|
+
` To install it properly: npm install -g ${PACKAGE_NAME}\n`
|
|
230
|
+
);
|
|
231
|
+
return;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
process.stdout.write(`Updating ${PACKAGE_NAME} (have ${before})…\n`);
|
|
235
|
+
const done = spawnSync("npm", ["install", "-g", `${PACKAGE_NAME}@${wanted}`], {
|
|
236
|
+
stdio: "inherit",
|
|
237
|
+
});
|
|
238
|
+
if (done.error || done.status !== 0) {
|
|
239
|
+
process.stdout.write(
|
|
240
|
+
`\nCould not update. Try it directly:\n` +
|
|
241
|
+
` npm install -g ${PACKAGE_NAME}@${wanted}\n` +
|
|
242
|
+
`If that reports a permissions error, npm's global directory is not ` +
|
|
243
|
+
`writable by you — that is an npm setup problem, not a PreMan one.\n`
|
|
244
|
+
);
|
|
245
|
+
process.exitCode = 1;
|
|
246
|
+
return;
|
|
247
|
+
}
|
|
248
|
+
// Read from the freshly installed copy rather than this process, which is
|
|
249
|
+
// still running the version we just replaced.
|
|
250
|
+
const now = spawnSync("npm", ["view", PACKAGE_NAME, "version"], { encoding: "utf8" });
|
|
251
|
+
const after = String(now.stdout || "").trim();
|
|
252
|
+
process.stdout.write(
|
|
253
|
+
after && after !== before
|
|
254
|
+
? `\nUpdated ${before} → ${after}. Run \`preman\` to start a session.\n`
|
|
255
|
+
: `\nAlready on the newest version (${before}).\n`
|
|
256
|
+
);
|
|
257
|
+
}
|
|
258
|
+
|
|
178
259
|
/**
|
|
179
260
|
* Cursor-only installer, kept for the documented `install` flow. `connect` is
|
|
180
261
|
* the same write plus agent choice, pairing, and dispatch setup.
|
|
@@ -249,7 +330,11 @@ async function main() {
|
|
|
249
330
|
// would only race with the one the user is watching.
|
|
250
331
|
if (!["hook", "verify", "connect"].includes(command)) scheduleHookRepair();
|
|
251
332
|
|
|
252
|
-
if (command === "
|
|
333
|
+
if (command === "agent" || command === "chat") {
|
|
334
|
+
await agentCommand(commandArgs);
|
|
335
|
+
} else if (command === "target") {
|
|
336
|
+
await targetCommand(commandArgs, { makeArgs });
|
|
337
|
+
} else if (command === "login") {
|
|
253
338
|
await loginCommand();
|
|
254
339
|
} else if (command === "status") {
|
|
255
340
|
await statusCommand(commandArgs);
|
|
@@ -286,6 +371,11 @@ async function main() {
|
|
|
286
371
|
await watchCommand(commandArgs);
|
|
287
372
|
} else if (command === "install-desktop") {
|
|
288
373
|
await installDesktopCommand(commandArgs);
|
|
374
|
+
} else if (command === "predict") {
|
|
375
|
+
// Exits on its own code: bringing the stack up can fail in half a dozen
|
|
376
|
+
// ways that are worth distinguishing from "the forecaster scored badly",
|
|
377
|
+
// and a caller in a shell loop should be able to tell them apart.
|
|
378
|
+
process.exitCode = await predictCommand(commandArgs);
|
|
289
379
|
} else if (command === "connect") {
|
|
290
380
|
await connectCommand(commandArgs);
|
|
291
381
|
} else if (command === "dispatch") {
|
|
@@ -306,12 +396,35 @@ async function main() {
|
|
|
306
396
|
await githubCommand(makeArgs(commandArgs));
|
|
307
397
|
} else if (command === "slack") {
|
|
308
398
|
await slackCommand(makeArgs(commandArgs));
|
|
399
|
+
} else if (command === "update" || command === "upgrade") {
|
|
400
|
+
await updateCommand(commandArgs);
|
|
309
401
|
} else if (command === "install") {
|
|
310
|
-
|
|
402
|
+
// `install` meant "write Cursor's MCP config", and every documented form of
|
|
403
|
+
// it passes a flag (`--api-key`, `--project`, `--backend`). Bare `install`
|
|
404
|
+
// never had a use, so it is the one spelling free to become what people
|
|
405
|
+
// actually type it expecting: set me up. Flagged forms keep working
|
|
406
|
+
// untouched, and `--cursor` names the old behaviour explicitly.
|
|
407
|
+
if (commandArgs.length > 0) {
|
|
408
|
+
await installCursorMcp();
|
|
409
|
+
} else {
|
|
410
|
+
await onboardCommand(commandArgs, {
|
|
411
|
+
makeArgs,
|
|
412
|
+
authenticateTerminal,
|
|
413
|
+
installDesktop: installDesktopCommand,
|
|
414
|
+
openDesktopSignedIn,
|
|
415
|
+
installPushHook: installHook,
|
|
416
|
+
});
|
|
417
|
+
}
|
|
311
418
|
} else if (command === "link") {
|
|
312
419
|
await linkCommand(commandArgs);
|
|
313
420
|
} else if (command === "tools") {
|
|
314
421
|
await toolsCommand(commandArgs);
|
|
422
|
+
} else if (command === "run" && commandArgs.includes("--predict")) {
|
|
423
|
+
// `run --predict` is the spelling people reach for, and `run` was already
|
|
424
|
+
// the hosted-MCP driver by the time this existed. Both spellings work
|
|
425
|
+
// rather than one of them being corrected: a tool name never starts with a
|
|
426
|
+
// dash, so there is nothing here for the two to argue about.
|
|
427
|
+
process.exitCode = await predictCommand(commandArgs.filter((arg) => arg !== "--predict"));
|
|
315
428
|
} else if (command === "run") {
|
|
316
429
|
await runCommand(commandArgs);
|
|
317
430
|
} else if (command === "endpoints") {
|
|
@@ -341,7 +454,13 @@ async function main() {
|
|
|
341
454
|
} else if (command === "start") {
|
|
342
455
|
startServer();
|
|
343
456
|
} else {
|
|
344
|
-
|
|
457
|
+
// Not routed to the agent on purpose: with twenty subcommands, a typo is a
|
|
458
|
+
// likelier explanation than a question, and silently chatting about
|
|
459
|
+
// "statuss" is a worse answer than saying the word is unknown.
|
|
460
|
+
process.stderr.write(
|
|
461
|
+
`Unknown command: ${command}\n` +
|
|
462
|
+
`To ask PreMan instead, run \`${cliInvocation()}\` or \`${cliInvocation()} chat "${command} ..."\`.\n\n`
|
|
463
|
+
);
|
|
345
464
|
printHelp();
|
|
346
465
|
process.exit(1);
|
|
347
466
|
}
|