premanmcp 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/cli.js +36 -1
- package/bin/eval.js +1200 -0
- package/bin/eval_harness.py +231 -0
- package/bin/eval_target.js +530 -0
- package/bin/hook.js +61 -5
- package/bin/link.js +81 -6
- package/bin/runner.js +193 -7
- package/bin/shared.js +37 -2
- package/package.json +2 -2
|
@@ -0,0 +1,530 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Finding the agent to evaluate, and making it answerable.
|
|
3
|
+
*
|
|
4
|
+
* Two jobs that look like one, and separating them is the whole design.
|
|
5
|
+
*
|
|
6
|
+
* **Finding it.** Nobody should have to tell PreMan where their agent is. The
|
|
7
|
+
* answer is a port on this machine, and the machine already knows: the project
|
|
8
|
+
* is right here, the process is running, and the URL is in a config file or a
|
|
9
|
+
* listening socket. Asking the customer to paste it is asking them to look up
|
|
10
|
+
* something we could have read.
|
|
11
|
+
*
|
|
12
|
+
* **Making it answerable.** assert-ai drives a target by POSTing
|
|
13
|
+
* `{message, history}` and expecting `{response}` back. Almost no real agent
|
|
14
|
+
* speaks that, and the ones that do are the ones somebody already wrote an
|
|
15
|
+
* adapter for. So this module stands up the adapter: a local server that speaks
|
|
16
|
+
* assert-ai's contract on one side and the agent's own API on the other,
|
|
17
|
+
* carrying whatever credentials the agent needs.
|
|
18
|
+
*
|
|
19
|
+
* The adapter is the reason the credentials story works. It runs here, on the
|
|
20
|
+
* developer's machine, holding the developer's key. PreMan's servers render the
|
|
21
|
+
* spec and store the results, and at no point does a key for the agent under
|
|
22
|
+
* test leave this process.
|
|
23
|
+
*
|
|
24
|
+
* ## Why the adapter is not optional even for an agent that "already has an HTTP API"
|
|
25
|
+
*
|
|
26
|
+
* A chat API is not a turn API. PreMan's own is the worked example: a
|
|
27
|
+
* conversation is a server-side object you create once and then append to, so
|
|
28
|
+
* `history` is not a field you send — it is state the server already holds. An
|
|
29
|
+
* adapter that ignored that and replayed the whole history every turn would be
|
|
30
|
+
* measuring a different agent than the one customers talk to. So the mapping
|
|
31
|
+
* from "assert-ai's stateless turn" to "this agent's idea of a conversation"
|
|
32
|
+
* is per-agent work, and this is where it lives.
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
import { createServer } from "node:http";
|
|
36
|
+
import { createHash } from "node:crypto";
|
|
37
|
+
|
|
38
|
+
import { backendUrl, resolveApiKey } from "./shared.js";
|
|
39
|
+
|
|
40
|
+
export class TargetError extends Error {}
|
|
41
|
+
|
|
42
|
+
/** Ports a locally-running PreMan backend is plausibly on, in order of likelihood. */
|
|
43
|
+
const LOCAL_PORTS = [8000, 8001, 8010, 8011, 8002, 8080];
|
|
44
|
+
|
|
45
|
+
const PROBE_TIMEOUT_MS = 1_500;
|
|
46
|
+
|
|
47
|
+
/** The route that makes a service *the agent* rather than merely a PreMan service. */
|
|
48
|
+
const AGENT_ROUTE = "/workbench/conversations/{conversation_id}/messages";
|
|
49
|
+
|
|
50
|
+
async function get(url, { timeout = PROBE_TIMEOUT_MS } = {}) {
|
|
51
|
+
const controller = new AbortController();
|
|
52
|
+
const timer = setTimeout(() => controller.abort(), timeout);
|
|
53
|
+
try {
|
|
54
|
+
return await fetch(url, { signal: controller.signal });
|
|
55
|
+
} finally {
|
|
56
|
+
clearTimeout(timer);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Is the agent we know how to drive answering here?
|
|
62
|
+
*
|
|
63
|
+
* Two questions, and asking only the first is what makes discovery pick the
|
|
64
|
+
* wrong thing. `/health` says a PreMan-shaped service is on this port; it does
|
|
65
|
+
* not say the service has a chat agent. A developer's machine can easily have an
|
|
66
|
+
* old build, a cut-down demo API, or a differently-scoped deployment answering
|
|
67
|
+
* `/health` with the right name — and pointing an eval at one produces a run
|
|
68
|
+
* where every case fails with a 404 that looks like the agent misbehaving.
|
|
69
|
+
*
|
|
70
|
+
* So the second question is asked of the service's own OpenAPI document: does it
|
|
71
|
+
* publish the route the adapter is about to POST to. That is a capability check
|
|
72
|
+
* rather than a version check, which means it keeps working when the route moves
|
|
73
|
+
* behind a different service and stops working the moment it is renamed — both
|
|
74
|
+
* of which are the answers you want.
|
|
75
|
+
*/
|
|
76
|
+
async function probeAgent(url) {
|
|
77
|
+
try {
|
|
78
|
+
const health = await get(`${url}/health`);
|
|
79
|
+
if (!health.ok) return { url, ok: false, why: `health answered ${health.status}` };
|
|
80
|
+
const body = await health.json().catch(() => ({}));
|
|
81
|
+
const service = String(body?.service || "");
|
|
82
|
+
if (!/preman/i.test(service)) {
|
|
83
|
+
return { url, ok: false, why: `something else is on this port (${service || "no service name"})` };
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const spec = await get(`${url}/openapi.json`, { timeout: 6_000 });
|
|
87
|
+
if (!spec.ok) return { url, ok: false, why: `a PreMan service, but it publishes no API document` };
|
|
88
|
+
const document = await spec.json().catch(() => ({}));
|
|
89
|
+
if (!document?.paths?.[AGENT_ROUTE]) {
|
|
90
|
+
return { url, ok: false, why: `a PreMan service, but it has no chat agent on it` };
|
|
91
|
+
}
|
|
92
|
+
return { url, ok: true, service };
|
|
93
|
+
} catch (error) {
|
|
94
|
+
return { url, ok: false, why: error.name === "AbortError" ? "no answer" : error.message };
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Work out which agent this eval is about.
|
|
100
|
+
*
|
|
101
|
+
* The order is deliberate and it is "most explicit first". A `--target` the
|
|
102
|
+
* customer typed beats anything discovered, because they have already answered
|
|
103
|
+
* the question and a tool that overrode them would be unusable in the one
|
|
104
|
+
* environment they configured it for. Then a backend on this machine, because
|
|
105
|
+
* somebody running an eval from a project directory with a dev server up almost
|
|
106
|
+
* certainly means that one. Then the configured backend, which is the honest
|
|
107
|
+
* fallback and is also the answer for anyone who is not a PreMan developer.
|
|
108
|
+
*/
|
|
109
|
+
export async function discoverTarget(args, { log = () => {} } = {}) {
|
|
110
|
+
const explicit = String(args.value("--target", "") || "").trim();
|
|
111
|
+
if (explicit) {
|
|
112
|
+
return {
|
|
113
|
+
kind: "raw",
|
|
114
|
+
url: explicit,
|
|
115
|
+
label: explicit,
|
|
116
|
+
how: "you named it with --target, so nothing was probed",
|
|
117
|
+
probed: [],
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const probed = [];
|
|
122
|
+
for (const port of LOCAL_PORTS) {
|
|
123
|
+
const result = await probeAgent(`http://127.0.0.1:${port}`);
|
|
124
|
+
probed.push(result);
|
|
125
|
+
if (result.ok) {
|
|
126
|
+
log(`Found a PreMan backend on port ${port}.`);
|
|
127
|
+
return {
|
|
128
|
+
kind: "preman-chat",
|
|
129
|
+
baseUrl: `http://127.0.0.1:${port}`,
|
|
130
|
+
// The agent's own credential, held here and never sent to PreMan's
|
|
131
|
+
// control plane. The adapter is the only thing that ever presents it,
|
|
132
|
+
// which is the whole reason the adapter runs on this machine.
|
|
133
|
+
token: resolveApiKey(args),
|
|
134
|
+
label: `PreMan agent (local, port ${port})`,
|
|
135
|
+
how: `probed localhost and got a PreMan /health on port ${port}`,
|
|
136
|
+
probed,
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const configured = backendUrl(args);
|
|
142
|
+
const remote = await probeAgent(configured);
|
|
143
|
+
probed.push(remote);
|
|
144
|
+
if (remote.ok) {
|
|
145
|
+
log(`No local backend; using the one this CLI is configured against.`);
|
|
146
|
+
return {
|
|
147
|
+
kind: "preman-chat",
|
|
148
|
+
baseUrl: configured,
|
|
149
|
+
token: resolveApiKey(args),
|
|
150
|
+
label: `PreMan agent (${configured})`,
|
|
151
|
+
how: `nothing local answered, so this is the backend the CLI is configured against`,
|
|
152
|
+
probed,
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
throw new TargetError(
|
|
157
|
+
`Could not find an agent to evaluate. Tried ${probed.length} address(es): ` +
|
|
158
|
+
probed.map((p) => `${p.url} (${p.why})`).join(", ") +
|
|
159
|
+
`. Point at one directly with --target <url>, where that URL accepts ` +
|
|
160
|
+
`POST {"message","history"} and answers {"response"}.`
|
|
161
|
+
);
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
// ── What the agent said it was doing ────────────────────────────────────
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* How many recent steps travel with a progress tick.
|
|
168
|
+
*
|
|
169
|
+
* The wire allows twenty and four of those are the pipeline stages, so this is
|
|
170
|
+
* what is left. It is a window on purpose: somebody watching wants the last few
|
|
171
|
+
* seconds, and a longer tail would push the stage summary off a small panel to
|
|
172
|
+
* show steps that have already been superseded.
|
|
173
|
+
*/
|
|
174
|
+
const STEP_WINDOW = 16;
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* The steps the adapter has seen, waiting to be reported.
|
|
178
|
+
*
|
|
179
|
+
* A module-level singleton, which is worth defending. Two things need it and
|
|
180
|
+
* they sit on opposite sides of the CLI: the adapter fills it from inside a
|
|
181
|
+
* turn, and the heartbeat drains it from inside the job loop. Threading a
|
|
182
|
+
* buffer between them would mean a parameter on `runnerLoop` and on everything
|
|
183
|
+
* it calls, all to carry something that is per-process anyway — this CLI runs
|
|
184
|
+
* one adapter and one eval at a time, by construction.
|
|
185
|
+
*
|
|
186
|
+
* Bounded, and dropped rather than queued when it overflows. This is a view of
|
|
187
|
+
* what is happening now; a backlog of steps from a minute ago is not a thing
|
|
188
|
+
* anybody is waiting to see, and keeping it would grow with the run.
|
|
189
|
+
*/
|
|
190
|
+
function createStepFeed() {
|
|
191
|
+
let entries = [];
|
|
192
|
+
return {
|
|
193
|
+
push(entry) {
|
|
194
|
+
entries.push(entry);
|
|
195
|
+
if (entries.length > STEP_WINDOW) entries = entries.slice(-STEP_WINDOW);
|
|
196
|
+
},
|
|
197
|
+
recent() {
|
|
198
|
+
return entries.slice();
|
|
199
|
+
},
|
|
200
|
+
/** Between runs, so one eval's feed cannot appear under the next one's. */
|
|
201
|
+
reset() {
|
|
202
|
+
entries = [];
|
|
203
|
+
},
|
|
204
|
+
};
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
export const stepFeed = createStepFeed();
|
|
208
|
+
|
|
209
|
+
// ── The adapter ─────────────────────────────────────────────────────────
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* Which PreMan conversation continues this exchange.
|
|
213
|
+
*
|
|
214
|
+
* assert-ai sends the whole history every turn and holds no server-side handle;
|
|
215
|
+
* PreMan holds the conversation and expects only the new message. So the history
|
|
216
|
+
* has to be turned back into an identity, and the identity is a hash of it.
|
|
217
|
+
*
|
|
218
|
+
* Hashing the *prior* turns rather than all of them is what makes the lookup
|
|
219
|
+
* work: on turn three assert-ai sends [u1, a1, u2, a2, u3], and what we stored
|
|
220
|
+
* after turn two was keyed on [u1, a1, u2, a2]. Dropping the trailing user
|
|
221
|
+
* message is what turns "here is everything so far" into "here is the
|
|
222
|
+
* conversation you already have".
|
|
223
|
+
*/
|
|
224
|
+
function conversationKey(history) {
|
|
225
|
+
const prior = history.filter((m) => m && m.role && typeof m.content === "string");
|
|
226
|
+
while (prior.length && prior[prior.length - 1].role === "user") prior.pop();
|
|
227
|
+
if (!prior.length) return "";
|
|
228
|
+
return createHash("sha256")
|
|
229
|
+
.update(JSON.stringify(prior.map((m) => [m.role, m.content])))
|
|
230
|
+
.digest("hex");
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
function readBody(request) {
|
|
234
|
+
return new Promise((resolve, reject) => {
|
|
235
|
+
let raw = "";
|
|
236
|
+
request.on("data", (chunk) => {
|
|
237
|
+
raw += chunk;
|
|
238
|
+
// A turn is a sentence. Anything past this is a client that has lost
|
|
239
|
+
// track of what it is sending, and buffering it would be this process
|
|
240
|
+
// absorbing somebody else's bug as memory pressure.
|
|
241
|
+
if (raw.length > 2 * 1024 * 1024) reject(new Error("request body too large"));
|
|
242
|
+
});
|
|
243
|
+
request.on("end", () => {
|
|
244
|
+
try {
|
|
245
|
+
resolve(raw ? JSON.parse(raw) : {});
|
|
246
|
+
} catch (error) {
|
|
247
|
+
reject(error);
|
|
248
|
+
}
|
|
249
|
+
});
|
|
250
|
+
request.on("error", reject);
|
|
251
|
+
});
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* One turn against PreMan's own agent.
|
|
256
|
+
*
|
|
257
|
+
* Tool calls come back in `turn.artifacts` and are forwarded as adapter-shaped
|
|
258
|
+
* `events`, which is the difference between the judge seeing a transcript and
|
|
259
|
+
* the judge seeing what the agent *did*. `HTTPEndpointSession` promotes those to
|
|
260
|
+
* first-class interaction messages; without them a tool-using agent is scored on
|
|
261
|
+
* its prose alone, which is exactly how a confidently-worded wrong action passes.
|
|
262
|
+
*/
|
|
263
|
+
async function post(target, route, body) {
|
|
264
|
+
const response = await fetch(new URL(route, `${target.baseUrl}/`), {
|
|
265
|
+
method: "POST",
|
|
266
|
+
headers: {
|
|
267
|
+
"content-type": "application/json",
|
|
268
|
+
accept: "application/json",
|
|
269
|
+
authorization: `Bearer ${target.token}`,
|
|
270
|
+
},
|
|
271
|
+
body: JSON.stringify(body),
|
|
272
|
+
});
|
|
273
|
+
const text = await response.text();
|
|
274
|
+
let parsed = {};
|
|
275
|
+
try {
|
|
276
|
+
parsed = text ? JSON.parse(text) : {};
|
|
277
|
+
} catch {
|
|
278
|
+
parsed = { raw: text };
|
|
279
|
+
}
|
|
280
|
+
if (!response.ok) {
|
|
281
|
+
const detail = typeof parsed.detail === "string" ? parsed.detail : text.slice(0, 200);
|
|
282
|
+
throw new Error(`${route} answered ${response.status}: ${detail}`);
|
|
283
|
+
}
|
|
284
|
+
return parsed;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Read one SSE turn, announcing each step as it arrives.
|
|
289
|
+
*
|
|
290
|
+
* The streaming endpoint is used for the *live* half of the eval, not the
|
|
291
|
+
* scored half. Both arrive here: `status` frames are the agent narrating its own
|
|
292
|
+
* tool activity while the turn is still running, and the `done` frame carries
|
|
293
|
+
* the identical persisted turn the JSON endpoint would have returned. So
|
|
294
|
+
* streaming costs nothing in fidelity and buys the only view of a run that
|
|
295
|
+
* updates faster than a case row.
|
|
296
|
+
*
|
|
297
|
+
* What a `status` label is, and what it is not: it is prose the agent wrote for
|
|
298
|
+
* a human — "Running smoke tests…" — emitted immediately before it invokes a
|
|
299
|
+
* tool. It is a truthful signal that a tool call is happening now, and it is
|
|
300
|
+
* *not* the tool's name or its arguments. Nothing scored is derived from it;
|
|
301
|
+
* the judge reads `turn.artifacts`, same as before. Treating these as evidence
|
|
302
|
+
* would be scoring an agent on its own commentary.
|
|
303
|
+
*/
|
|
304
|
+
async function streamTurn(target, conversationId, message, state) {
|
|
305
|
+
const response = await fetch(
|
|
306
|
+
new URL(
|
|
307
|
+
`workbench/conversations/${encodeURIComponent(conversationId)}/messages/stream`,
|
|
308
|
+
`${target.baseUrl}/`
|
|
309
|
+
),
|
|
310
|
+
{
|
|
311
|
+
method: "POST",
|
|
312
|
+
headers: {
|
|
313
|
+
"content-type": "application/json",
|
|
314
|
+
accept: "text/event-stream",
|
|
315
|
+
authorization: `Bearer ${target.token}`,
|
|
316
|
+
},
|
|
317
|
+
body: JSON.stringify({ content: message }),
|
|
318
|
+
}
|
|
319
|
+
);
|
|
320
|
+
|
|
321
|
+
if (!response.ok || !response.body) {
|
|
322
|
+
throw new Error(`the stream answered ${response.status}`);
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
let buffered = "";
|
|
326
|
+
let turn = null;
|
|
327
|
+
let failure = "";
|
|
328
|
+
|
|
329
|
+
for await (const chunk of response.body) {
|
|
330
|
+
buffered += Buffer.from(chunk).toString("utf8");
|
|
331
|
+
// Frames are `data: {json}\n\n`. Splitting on the blank line rather than on
|
|
332
|
+
// every newline is what keeps a pretty-printed payload in one piece.
|
|
333
|
+
const frames = buffered.split("\n\n");
|
|
334
|
+
buffered = frames.pop() ?? "";
|
|
335
|
+
|
|
336
|
+
for (const frame of frames) {
|
|
337
|
+
const line = frame.split("\n").find((l) => l.startsWith("data:"));
|
|
338
|
+
if (!line) continue;
|
|
339
|
+
let event;
|
|
340
|
+
try {
|
|
341
|
+
event = JSON.parse(line.slice(5).trim());
|
|
342
|
+
} catch {
|
|
343
|
+
continue; // A frame we cannot read is not a reason to fail the turn.
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
if (event.type === "status" && event.label) {
|
|
347
|
+
state.step(String(event.label));
|
|
348
|
+
} else if (event.type === "done") {
|
|
349
|
+
turn = event.turn || {};
|
|
350
|
+
} else if (event.type === "error") {
|
|
351
|
+
failure = String(event.error || event.detail || "the turn failed");
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
if (turn) return turn;
|
|
357
|
+
throw new Error(failure || "the stream ended without finishing the turn");
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
async function premanTurn(target, args, { message, history }, state) {
|
|
361
|
+
const key = conversationKey(history || []);
|
|
362
|
+
let conversationId = key ? state.conversations.get(key) : "";
|
|
363
|
+
|
|
364
|
+
if (!conversationId) {
|
|
365
|
+
const created = await post(target, "workbench/conversations", {
|
|
366
|
+
title: "PreMan agent eval",
|
|
367
|
+
});
|
|
368
|
+
conversationId = String(created.id || created.conversation_id || "");
|
|
369
|
+
if (!conversationId) throw new Error("the agent created a conversation with no id");
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
let turn;
|
|
373
|
+
try {
|
|
374
|
+
turn = await streamTurn(target, conversationId, message, state);
|
|
375
|
+
} catch (error) {
|
|
376
|
+
// The live view is a nicety; the measurement is not. An agent whose build
|
|
377
|
+
// predates the streaming route, or a proxy that buffers SSE into
|
|
378
|
+
// uselessness, should produce a run with no step feed rather than no run.
|
|
379
|
+
state.streamFailures += 1;
|
|
380
|
+
if (state.streamFailures === 1) {
|
|
381
|
+
state.log(`no live steps from this agent (${error.message}); measuring it anyway`);
|
|
382
|
+
}
|
|
383
|
+
const sent = await post(
|
|
384
|
+
target,
|
|
385
|
+
`workbench/conversations/${encodeURIComponent(conversationId)}/messages`,
|
|
386
|
+
{ content: message }
|
|
387
|
+
);
|
|
388
|
+
turn = sent.turn || {};
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
const reply = String(turn.content || "");
|
|
392
|
+
|
|
393
|
+
// Keyed on the exchange that just happened, so the next turn's history hashes
|
|
394
|
+
// to this same conversation.
|
|
395
|
+
const nextKey = createHash("sha256")
|
|
396
|
+
.update(
|
|
397
|
+
JSON.stringify([
|
|
398
|
+
...(history || [])
|
|
399
|
+
.filter((m) => m && m.role && typeof m.content === "string")
|
|
400
|
+
.map((m) => [m.role, m.content]),
|
|
401
|
+
["assistant", reply],
|
|
402
|
+
])
|
|
403
|
+
)
|
|
404
|
+
.digest("hex");
|
|
405
|
+
state.conversations.set(nextKey, conversationId);
|
|
406
|
+
|
|
407
|
+
return { response: reply, events: eventsFrom(turn) };
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/**
|
|
411
|
+
* PreMan's `turn.artifacts` as the events assert-ai understands.
|
|
412
|
+
*
|
|
413
|
+
* Best-effort and deliberately lossy: only artifacts that clearly describe a
|
|
414
|
+
* tool invocation become events. Forwarding everything would put rendered
|
|
415
|
+
* markdown and UI intents in front of the judge as though they were actions the
|
|
416
|
+
* agent took, and a judge scoring "did it do something it should not have" would
|
|
417
|
+
* be reading the wrong evidence.
|
|
418
|
+
*/
|
|
419
|
+
function eventsFrom(turn) {
|
|
420
|
+
const artifacts = Array.isArray(turn.artifacts) ? turn.artifacts : [];
|
|
421
|
+
const events = [];
|
|
422
|
+
for (const artifact of artifacts) {
|
|
423
|
+
const name = artifact?.tool || artifact?.name || artifact?.tool_name;
|
|
424
|
+
if (!name) continue;
|
|
425
|
+
events.push({ type: "tool_call", name: String(name), arguments: artifact.arguments ?? artifact.input ?? {} });
|
|
426
|
+
if (artifact.result !== undefined || artifact.output !== undefined) {
|
|
427
|
+
events.push({
|
|
428
|
+
type: "tool_result",
|
|
429
|
+
name: String(name),
|
|
430
|
+
result: artifact.result ?? artifact.output,
|
|
431
|
+
});
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
return events;
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
/**
|
|
438
|
+
* Serve assert-ai's contract on loopback until the eval is done.
|
|
439
|
+
*
|
|
440
|
+
* Bound to 127.0.0.1 explicitly rather than to every interface. The only client
|
|
441
|
+
* is a subprocess on this machine, and a target endpoint reachable from the
|
|
442
|
+
* network is a small unauthenticated proxy to the customer's own agent, held
|
|
443
|
+
* open for the length of a run.
|
|
444
|
+
*
|
|
445
|
+
* Port zero, so the OS picks. A fixed port would collide with the agent, with a
|
|
446
|
+
* previous run that has not finished tearing down, or with whatever else the
|
|
447
|
+
* developer has open — and every one of those failures reads as "the eval
|
|
448
|
+
* couldn't reach your agent".
|
|
449
|
+
*/
|
|
450
|
+
export async function startAdapter(target, args, { log = () => {}, onStep = null } = {}) {
|
|
451
|
+
if (target.kind === "raw") {
|
|
452
|
+
// Nothing to translate: the customer named something that already speaks
|
|
453
|
+
// this. Standing a proxy in front of it would add a hop that can fail and
|
|
454
|
+
// buy nothing.
|
|
455
|
+
//
|
|
456
|
+
// No step feed either, and that is honest rather than a gap: a URL the
|
|
457
|
+
// customer named speaks assert-ai's turn contract and nothing else, so
|
|
458
|
+
// there is no channel on which it could narrate itself.
|
|
459
|
+
return { url: target.url, close: async () => {}, translated: false };
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
const state = {
|
|
463
|
+
conversations: new Map(),
|
|
464
|
+
turns: 0,
|
|
465
|
+
failures: 0,
|
|
466
|
+
streamFailures: 0,
|
|
467
|
+
log,
|
|
468
|
+
/**
|
|
469
|
+
* One thing the agent said it was doing, on its way to the dashboard.
|
|
470
|
+
*
|
|
471
|
+
* Ids are positional rather than derived from the label, because the same
|
|
472
|
+
* label recurs legitimately — an agent that runs smoke tests three times
|
|
473
|
+
* says so three times — and collapsing those would under-report the work.
|
|
474
|
+
*/
|
|
475
|
+
step: (label) => {
|
|
476
|
+
if (!onStep) return;
|
|
477
|
+
state.steps += 1;
|
|
478
|
+
onStep({ id: `step.${state.turns}.${state.steps}`, label, state: "complete" });
|
|
479
|
+
},
|
|
480
|
+
steps: 0,
|
|
481
|
+
};
|
|
482
|
+
|
|
483
|
+
const server = createServer((request, response) => {
|
|
484
|
+
const reply = (status, body) => {
|
|
485
|
+
const text = JSON.stringify(body);
|
|
486
|
+
response.writeHead(status, {
|
|
487
|
+
"content-type": "application/json",
|
|
488
|
+
"content-length": Buffer.byteLength(text),
|
|
489
|
+
});
|
|
490
|
+
response.end(text);
|
|
491
|
+
};
|
|
492
|
+
|
|
493
|
+
if (request.method !== "POST") return reply(405, { error: "post a turn here" });
|
|
494
|
+
|
|
495
|
+
void (async () => {
|
|
496
|
+
try {
|
|
497
|
+
const body = await readBody(request);
|
|
498
|
+
state.turns += 1;
|
|
499
|
+
const result = await premanTurn(target, args, body, state);
|
|
500
|
+
reply(200, result);
|
|
501
|
+
} catch (error) {
|
|
502
|
+
state.failures += 1;
|
|
503
|
+
// 200 with an error string in `response`, not a 5xx. A transport failure
|
|
504
|
+
// aborts the case and loses it; an agent that said something unhelpful is
|
|
505
|
+
// a transcript the judge can score. Which of those this is depends on
|
|
506
|
+
// whose fault it was, and from in here it usually is not assert-ai's.
|
|
507
|
+
log(`adapter turn failed: ${error.message}`);
|
|
508
|
+
reply(200, { response: `[the agent could not answer: ${error.message}]` });
|
|
509
|
+
}
|
|
510
|
+
})();
|
|
511
|
+
});
|
|
512
|
+
|
|
513
|
+
await new Promise((resolve, reject) => {
|
|
514
|
+
server.once("error", reject);
|
|
515
|
+
server.listen(0, "127.0.0.1", resolve);
|
|
516
|
+
});
|
|
517
|
+
|
|
518
|
+
const { port } = server.address();
|
|
519
|
+
const url = `http://127.0.0.1:${port}/chat`;
|
|
520
|
+
log(`Adapter listening on ${url} -> ${target.label}`);
|
|
521
|
+
|
|
522
|
+
return {
|
|
523
|
+
url,
|
|
524
|
+
translated: true,
|
|
525
|
+
stats: state,
|
|
526
|
+
close: async () => {
|
|
527
|
+
await new Promise((resolve) => server.close(resolve));
|
|
528
|
+
},
|
|
529
|
+
};
|
|
530
|
+
}
|
package/bin/hook.js
CHANGED
|
@@ -139,6 +139,57 @@ function isGeneratedInvocation(invocation) {
|
|
|
139
139
|
return invocation === "preman" || /^npm exec -y premanmcp@\S+ --$/.test(invocation);
|
|
140
140
|
}
|
|
141
141
|
|
|
142
|
+
/**
|
|
143
|
+
* The release a generated invocation pins, or "" when it does not pin one.
|
|
144
|
+
*
|
|
145
|
+
* The bare `preman` form resolves at push time, and `@latest` is whatever we
|
|
146
|
+
* ship next, so neither can be out of date. Only an explicit version can be.
|
|
147
|
+
*/
|
|
148
|
+
export function pinnedRelease(invocation) {
|
|
149
|
+
const match = /^npm exec -y premanmcp@(\S+) --$/.exec(String(invocation || "").trim());
|
|
150
|
+
const version = match ? match[1] : "";
|
|
151
|
+
return version === "latest" ? "" : version;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Is `left` an earlier release than `right`?
|
|
156
|
+
*
|
|
157
|
+
* Anything this cannot read as three numbers compares as not-older, so a
|
|
158
|
+
* prerelease tag or a mirror's own numbering is left alone rather than being
|
|
159
|
+
* rewritten on a guess.
|
|
160
|
+
*/
|
|
161
|
+
export function isOlderRelease(left, right) {
|
|
162
|
+
const parse = (value) => String(value).split(".").map((part) => Number.parseInt(part, 10));
|
|
163
|
+
const a = parse(left);
|
|
164
|
+
const b = parse(right);
|
|
165
|
+
if (a.length !== 3 || b.length !== 3) return false;
|
|
166
|
+
if ([...a, ...b].some((part) => !Number.isFinite(part))) return false;
|
|
167
|
+
for (let index = 0; index < 3; index += 1) {
|
|
168
|
+
if (a[index] !== b[index]) return a[index] < b[index];
|
|
169
|
+
}
|
|
170
|
+
return false;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* Does this hook run a PreMan older than the one asking?
|
|
175
|
+
*
|
|
176
|
+
* A pinned hook keeps working long after it stops being current -- it answers
|
|
177
|
+
* `help`, it runs `verify`, it just does all of it as the version that wrote
|
|
178
|
+
* it. Found in the wild: a hook still on 0.13.0 while 1.0.4 was installed,
|
|
179
|
+
* printing an error message whose formatting had been fixed several releases
|
|
180
|
+
* earlier, with nothing anywhere to say why.
|
|
181
|
+
*
|
|
182
|
+
* Only ever true when we would be moving forward. An older CLI run for some
|
|
183
|
+
* other reason must not drag a newer hook back with it.
|
|
184
|
+
*/
|
|
185
|
+
export function hookIsBehind(invocation) {
|
|
186
|
+
if (declaredInvocation()) return false;
|
|
187
|
+
const pinned = pinnedRelease(invocation);
|
|
188
|
+
const running = packageVersion();
|
|
189
|
+
if (!pinned || !running) return false;
|
|
190
|
+
return isOlderRelease(pinned, running);
|
|
191
|
+
}
|
|
192
|
+
|
|
142
193
|
function hookBody(invocation) {
|
|
143
194
|
// `exec` is deliberately absent: we want the wrapper to survive the CLI exiting
|
|
144
195
|
// non-zero and still exit 0 itself.
|
|
@@ -319,9 +370,11 @@ function rememberHookCheck(cwd, now, fields) {
|
|
|
319
370
|
* PreMan is running here -- any command, the MCP server starting -- is taken as
|
|
320
371
|
* the moment to check.
|
|
321
372
|
*
|
|
322
|
-
* Deliberately narrow. It only ever rewrites a hook we wrote
|
|
323
|
-
*
|
|
324
|
-
*
|
|
373
|
+
* Deliberately narrow. It only ever rewrites a hook we wrote: an absent hook is
|
|
374
|
+
* not installed behind the user, and a foreign hook is not touched. A working
|
|
375
|
+
* hook is rewritten only when it pins a release older than the one running --
|
|
376
|
+
* see `hookIsBehind` -- because "it answers" and "it is the CLI we ship" stopped
|
|
377
|
+
* being the same question once a pin could outlive several releases.
|
|
325
378
|
*/
|
|
326
379
|
export function repairDeadHook({ cwd = process.cwd(), now = Date.now(), force = false } = {}) {
|
|
327
380
|
const remember = (fields) => rememberHookCheck(cwd, now, fields);
|
|
@@ -337,11 +390,14 @@ export function repairDeadHook({ cwd = process.cwd(), now = Date.now(), force =
|
|
|
337
390
|
return remember({ action: "not-a-repository" });
|
|
338
391
|
}
|
|
339
392
|
if (status.state !== "installed") return remember({ action: status.state });
|
|
340
|
-
|
|
393
|
+
const behind = hookIsBehind(status.invocation);
|
|
394
|
+
if (status.works && !behind) {
|
|
395
|
+
return remember({ action: "healthy", invocation: status.invocation });
|
|
396
|
+
}
|
|
341
397
|
|
|
342
398
|
const result = installHook(makeArgs([]));
|
|
343
399
|
return remember({
|
|
344
|
-
action: result.action === "updated" ? "repaired" : result.action,
|
|
400
|
+
action: result.action === "updated" ? (behind ? "upgraded" : "repaired") : result.action,
|
|
345
401
|
invocation: result.invocation || "",
|
|
346
402
|
replaced: status.invocation,
|
|
347
403
|
});
|