hilos-agent 0.9.0 → 0.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +80 -16
- package/bin/hilos-agent.mjs +16 -6
- package/package.json +1 -1
- package/src/acp-session.mjs +69 -54
- package/src/agent-events.mjs +645 -45
- package/src/argv.mjs +61 -0
- package/src/attachments.mjs +310 -0
- package/src/claude-permissions.mjs +445 -0
- package/src/cli.mjs +56 -0
- package/src/codex-mcp-session.mjs +619 -0
- package/src/config.mjs +83 -7
- package/src/handler.mjs +914 -77
- package/src/hook.mjs +793 -108
- package/src/mcp-loopback.mjs +142 -0
- package/src/mcp.mjs +3 -2
- package/src/model-resolve.mjs +180 -11
- package/src/permission-gate.mjs +269 -0
- package/src/progress-emitter.mjs +100 -5
- package/src/queue.mjs +21 -5
- package/src/redact.mjs +11 -1
- package/src/reply-bridge.mjs +847 -0
- package/src/resume.mjs +48 -11
- package/src/run.mjs +130 -4
- package/src/transcript.mjs +153 -0
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
// Turn-scoped loopback transport for a resumed Codex session (0854).
|
|
2
|
+
//
|
|
3
|
+
// Codex's remote MCP auth is normally an environment variable. The daemon
|
|
4
|
+
// deliberately strips HILOS_* secrets from coding children, and handing a
|
|
5
|
+
// persistent workspace token to a model-visible shell would undo that safety.
|
|
6
|
+
// Instead, keep the token in the daemon process and expose a random loopback
|
|
7
|
+
// URL only for the lifetime of one continuation. Codex can use every tool its
|
|
8
|
+
// agent identity is authorized for; it never receives the bearer credential.
|
|
9
|
+
|
|
10
|
+
import { createServer } from "node:http";
|
|
11
|
+
import { randomBytes } from "node:crypto";
|
|
12
|
+
import { DAEMON_CLIENT } from "./mcp.mjs";
|
|
13
|
+
|
|
14
|
+
const MAX_BODY_BYTES = 2 * 1024 * 1024;
|
|
15
|
+
const FORWARDED_RESPONSE_HEADERS = [
|
|
16
|
+
"content-type",
|
|
17
|
+
"mcp-session-id",
|
|
18
|
+
"mcp-protocol-version",
|
|
19
|
+
"www-authenticate",
|
|
20
|
+
];
|
|
21
|
+
const FORWARDED_REQUEST_HEADERS = [
|
|
22
|
+
"content-type",
|
|
23
|
+
"accept",
|
|
24
|
+
"mcp-session-id",
|
|
25
|
+
"mcp-protocol-version",
|
|
26
|
+
"mcp-method",
|
|
27
|
+
"mcp-name",
|
|
28
|
+
];
|
|
29
|
+
const MAX_HEADER_LENGTH = 4096;
|
|
30
|
+
|
|
31
|
+
function readBody(request) {
|
|
32
|
+
return new Promise((resolve, reject) => {
|
|
33
|
+
let size = 0;
|
|
34
|
+
const chunks = [];
|
|
35
|
+
request.on("data", (chunk) => {
|
|
36
|
+
size += chunk.length;
|
|
37
|
+
if (size > MAX_BODY_BYTES) {
|
|
38
|
+
reject(new Error("request body too large"));
|
|
39
|
+
request.destroy();
|
|
40
|
+
return;
|
|
41
|
+
}
|
|
42
|
+
chunks.push(chunk);
|
|
43
|
+
});
|
|
44
|
+
request.on("end", () => resolve(Buffer.concat(chunks)));
|
|
45
|
+
request.on("error", reject);
|
|
46
|
+
});
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Start an unguessable, loopback-only HTTP proxy to one authenticated hilos MCP
|
|
51
|
+
* endpoint. The caller must close it in finally after the coding turn.
|
|
52
|
+
* @param {{
|
|
53
|
+
* url?: string,
|
|
54
|
+
* token?: string,
|
|
55
|
+
* channelId?: string,
|
|
56
|
+
* fetchImpl?: typeof fetch,
|
|
57
|
+
* }} [options]
|
|
58
|
+
* @returns {Promise<{url: string, close: () => Promise<void>} | null>}
|
|
59
|
+
*/
|
|
60
|
+
export async function startHilosMcpLoopback({ url, token, channelId = "", fetchImpl = fetch } = {}) {
|
|
61
|
+
if (!url || !token) return null;
|
|
62
|
+
const nonce = randomBytes(24).toString("hex");
|
|
63
|
+
const path = `/mcp/${nonce}`;
|
|
64
|
+
const upstream = new URL(url);
|
|
65
|
+
// Preserve the room that caused this turn as MCP's effective context. This
|
|
66
|
+
// does not widen or narrow token access, but it keeps guest-room memory gates
|
|
67
|
+
// and other context-sensitive server policy active even when the model calls
|
|
68
|
+
// a workspace-scoped tool without a channelId argument.
|
|
69
|
+
if (channelId) upstream.searchParams.set("channelId", channelId);
|
|
70
|
+
const upstreamRequests = new Set();
|
|
71
|
+
|
|
72
|
+
const server = createServer(async (request, response) => {
|
|
73
|
+
if (request.url !== path || !["POST", "GET", "DELETE"].includes(request.method || "")) {
|
|
74
|
+
response.writeHead(404).end();
|
|
75
|
+
return;
|
|
76
|
+
}
|
|
77
|
+
const controller = new AbortController();
|
|
78
|
+
upstreamRequests.add(controller);
|
|
79
|
+
const abortUpstream = () => controller.abort();
|
|
80
|
+
request.once("aborted", abortUpstream);
|
|
81
|
+
response.once("close", abortUpstream);
|
|
82
|
+
try {
|
|
83
|
+
const body = request.method === "POST" ? await readBody(request) : undefined;
|
|
84
|
+
const headers = {
|
|
85
|
+
authorization: `Bearer ${token}`,
|
|
86
|
+
// This short-lived proxy is part of the persistent daemon's return
|
|
87
|
+
// path. Marking it on-demand would temporarily relabel a live daemon
|
|
88
|
+
// and could suppress the offline/wakeup behavior after the turn.
|
|
89
|
+
"x-hilos-connection-mode": "daemon",
|
|
90
|
+
...(DAEMON_CLIENT ? { "x-hilos-client": DAEMON_CLIENT } : {}),
|
|
91
|
+
};
|
|
92
|
+
for (const name of FORWARDED_REQUEST_HEADERS) {
|
|
93
|
+
const value = request.headers[name];
|
|
94
|
+
if (typeof value === "string" && value.length <= MAX_HEADER_LENGTH) {
|
|
95
|
+
headers[name] = value;
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
const forwarded = await fetchImpl(upstream, {
|
|
99
|
+
method: request.method,
|
|
100
|
+
headers,
|
|
101
|
+
...(body ? { body } : {}),
|
|
102
|
+
signal: controller.signal,
|
|
103
|
+
});
|
|
104
|
+
for (const name of FORWARDED_RESPONSE_HEADERS) {
|
|
105
|
+
const value = forwarded.headers.get(name);
|
|
106
|
+
if (value) response.setHeader(name, value);
|
|
107
|
+
}
|
|
108
|
+
response.writeHead(forwarded.status);
|
|
109
|
+
response.end(Buffer.from(await forwarded.arrayBuffer()));
|
|
110
|
+
} catch {
|
|
111
|
+
if (!response.destroyed && !response.writableEnded) {
|
|
112
|
+
response.writeHead(502, { "content-type": "application/json" });
|
|
113
|
+
response.end(JSON.stringify({ error: "hilos MCP loopback unavailable" }));
|
|
114
|
+
}
|
|
115
|
+
} finally {
|
|
116
|
+
upstreamRequests.delete(controller);
|
|
117
|
+
request.removeListener("aborted", abortUpstream);
|
|
118
|
+
response.removeListener("close", abortUpstream);
|
|
119
|
+
}
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
await new Promise((resolve, reject) => {
|
|
123
|
+
server.once("error", reject);
|
|
124
|
+
server.listen(0, "127.0.0.1", resolve);
|
|
125
|
+
});
|
|
126
|
+
const address = server.address();
|
|
127
|
+
if (!address || typeof address === "string") {
|
|
128
|
+
server.close();
|
|
129
|
+
return null;
|
|
130
|
+
}
|
|
131
|
+
let closed = false;
|
|
132
|
+
return {
|
|
133
|
+
url: `http://127.0.0.1:${address.port}${path}`,
|
|
134
|
+
async close() {
|
|
135
|
+
if (closed) return;
|
|
136
|
+
closed = true;
|
|
137
|
+
for (const controller of upstreamRequests) controller.abort();
|
|
138
|
+
server.closeAllConnections?.();
|
|
139
|
+
await new Promise((resolve) => server.close(() => resolve()));
|
|
140
|
+
},
|
|
141
|
+
};
|
|
142
|
+
}
|
package/src/mcp.mjs
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
import os from "node:os";
|
|
4
4
|
|
|
5
5
|
// Best-effort host label so hilos can show "whose machine" this daemon runs on.
|
|
6
|
-
const
|
|
6
|
+
export const DAEMON_CLIENT = (() => {
|
|
7
7
|
try {
|
|
8
8
|
return `${os.userInfo().username}@${os.hostname()}`;
|
|
9
9
|
} catch {
|
|
@@ -20,7 +20,8 @@ export function makeClient({ url, token }) {
|
|
|
20
20
|
headers: {
|
|
21
21
|
"content-type": "application/json",
|
|
22
22
|
authorization: `Bearer ${token}`,
|
|
23
|
-
|
|
23
|
+
"x-hilos-connection-mode": "daemon",
|
|
24
|
+
...(DAEMON_CLIENT ? { "x-hilos-client": DAEMON_CLIENT } : {}),
|
|
24
25
|
},
|
|
25
26
|
body: JSON.stringify({ jsonrpc: "2.0", id: ++id, method, params }),
|
|
26
27
|
});
|
package/src/model-resolve.mjs
CHANGED
|
@@ -11,7 +11,13 @@
|
|
|
11
11
|
// PURE + node-builtins-only; the resolver takes an injected `run` (runCli) so
|
|
12
12
|
// tests drive it with no CLI; a resolution failure NEVER breaks a run ([]).
|
|
13
13
|
//
|
|
14
|
-
// codex
|
|
14
|
+
// codex joined in 0783 on the SAME terms: `codex debug models` ("Render the raw
|
|
15
|
+
// model catalog as JSON") is its `--list-models`, so its tier also resolves at
|
|
16
|
+
// run time against the account's own catalog and never against a baked slug.
|
|
17
|
+
// That matters more for codex than for anyone: its slugs are codenames that rev
|
|
18
|
+
// (`gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` in the live catalog) and are
|
|
19
|
+
// plan-gated — a run on a ChatGPT account rejects `gpt-5` outright with "The
|
|
20
|
+
// 'gpt-5' model is not supported when using Codex with a ChatGPT account".
|
|
15
21
|
|
|
16
22
|
/** Strip ANSI SGR color codes (`--list-models` output is colorized). */
|
|
17
23
|
export function stripAnsi(s) {
|
|
@@ -65,30 +71,193 @@ export function resolveCursorModel(tier, ids) {
|
|
|
65
71
|
return null;
|
|
66
72
|
}
|
|
67
73
|
|
|
74
|
+
/**
|
|
75
|
+
* Parse `codex debug models` stdout → the account's LISTED models, ranked by
|
|
76
|
+
* the catalog's own `priority` (lower first — it is Codex's own ordering of its
|
|
77
|
+
* lineup). Wire shape captured live against codex-cli 0.144.1:
|
|
78
|
+
*
|
|
79
|
+
* {"models":[{"slug":"gpt-5.6-sol","display_name":"GPT-5.6-Sol",
|
|
80
|
+
* "description":"Latest frontier agentic coding model.","priority":1,
|
|
81
|
+
* "visibility":"list", …}, …]}
|
|
82
|
+
*
|
|
83
|
+
* `visibility:"hide"` entries are internal (the live catalog hides
|
|
84
|
+
* `codex-auto-review`, Codex's own approval-review model) and are dropped — the
|
|
85
|
+
* user's preset must never select something the CLI doesn't offer. A drifted
|
|
86
|
+
* or non-JSON payload → [] and the preset silently falls back to the tool
|
|
87
|
+
* default, same as the cursor path.
|
|
88
|
+
*
|
|
89
|
+
* `supported_in_api` rides along rather than being filtered here, because the
|
|
90
|
+
* catalog is the RAW lineup — the same bytes whichever way the account is
|
|
91
|
+
* logged in (verified live) — and whether an api-false model is usable depends
|
|
92
|
+
* on that login. `gpt-5.3-codex-spark` is the live example: api-false, and
|
|
93
|
+
* perfectly runnable on a ChatGPT account. A row that doesn't state the field
|
|
94
|
+
* counts as supported; we never filter on something the catalog didn't say.
|
|
95
|
+
* @param {string} stdout
|
|
96
|
+
* @returns {{slug: string, text: string, supportedInApi: boolean}[]}
|
|
97
|
+
*/
|
|
98
|
+
export function parseCodexModels(stdout) {
|
|
99
|
+
let payload;
|
|
100
|
+
try {
|
|
101
|
+
payload = JSON.parse(String(stdout || ""));
|
|
102
|
+
} catch {
|
|
103
|
+
return [];
|
|
104
|
+
}
|
|
105
|
+
const models = payload && Array.isArray(payload.models) ? payload.models : [];
|
|
106
|
+
const rows = [];
|
|
107
|
+
for (const m of models) {
|
|
108
|
+
if (!m || typeof m !== "object") continue;
|
|
109
|
+
if (typeof m.slug !== "string" || !m.slug) continue;
|
|
110
|
+
if (typeof m.visibility === "string" && m.visibility !== "list") continue;
|
|
111
|
+
const priority = Number.isFinite(m.priority) ? m.priority : Number.MAX_SAFE_INTEGER;
|
|
112
|
+
const text = `${m.description || ""} ${m.display_name || ""} ${m.slug}`.toLowerCase();
|
|
113
|
+
rows.push({ slug: m.slug, text, priority, supportedInApi: m.supported_in_api !== false });
|
|
114
|
+
}
|
|
115
|
+
rows.sort((a, b) => a.priority - b.priority);
|
|
116
|
+
return rows.map(({ slug, text, supportedInApi }) => ({ slug, text, supportedInApi }));
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Which login `codex login status` is reporting. The CLI's own one-line answer
|
|
121
|
+
* ("Logged in using ChatGPT" on the live install, 0.4s, printed on STDERR) —
|
|
122
|
+
* asking it beats parsing `~/.codex/auth.json`, which is private and can move
|
|
123
|
+
* with CODEX_HOME. Feed it both streams.
|
|
124
|
+
*
|
|
125
|
+
* ONLY a positive "api key" reading counts as apikey; everything else,
|
|
126
|
+
* including an empty/failed/unrecognized answer, is "unknown" and changes
|
|
127
|
+
* nothing downstream. The ChatGPT wording is live-verified; the API-key wording
|
|
128
|
+
* is not (I had no API-key install to log into, and would not clobber a real
|
|
129
|
+
* login to make one), so the matcher is deliberately loose on that side and
|
|
130
|
+
* fail-safe on every other.
|
|
131
|
+
* @param {string} stdout
|
|
132
|
+
* @returns {'apikey'|'chatgpt'|'unknown'}
|
|
133
|
+
*/
|
|
134
|
+
export function codexAuthMode(stdout) {
|
|
135
|
+
const s = String(stdout || "").toLowerCase();
|
|
136
|
+
if (/api[\s_-]?key/.test(s)) return "apikey";
|
|
137
|
+
if (/chatgpt/.test(s)) return "chatgpt";
|
|
138
|
+
return "unknown";
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Drop models this login cannot actually run. Only bites for `apikey`: the
|
|
143
|
+
* catalog's `supported_in_api:false` rows are exactly the ones an API-key run
|
|
144
|
+
* rejects, and the "Fastest" tier would otherwise reach for one
|
|
145
|
+
* (`gpt-5.3-codex-spark` is api-false in the live catalog). chatgpt/unknown
|
|
146
|
+
* pass through untouched — under ChatGPT auth those models are fine, and an
|
|
147
|
+
* unreadable auth answer must never silently narrow a user's lineup.
|
|
148
|
+
* @param {{slug: string, text: string, supportedInApi: boolean}[]} models
|
|
149
|
+
* @param {'apikey'|'chatgpt'|'unknown'} mode
|
|
150
|
+
*/
|
|
151
|
+
export function filterCodexModels(models, mode) {
|
|
152
|
+
if (!Array.isArray(models)) return [];
|
|
153
|
+
if (mode !== "apikey") return models;
|
|
154
|
+
return models.filter((m) => m && m.supportedInApi !== false);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// Ranked preferences per tier for codex, matched against each model's own
|
|
158
|
+
// description + display name + slug. Codex's catalog SAYS which tier a model
|
|
159
|
+
// is ("Latest frontier agentic coding model." / "Balanced agentic coding model
|
|
160
|
+
// for everyday work." / "Fast and affordable…"), and unlike the slugs those
|
|
161
|
+
// words are stable — reading them beats pattern-matching codenames that rev
|
|
162
|
+
// every release. First pattern with any match wins, then the first model in
|
|
163
|
+
// the catalog's own priority order that matches it.
|
|
164
|
+
const CODEX_TIER_PREFS = {
|
|
165
|
+
opus: [/frontier/, /complex/],
|
|
166
|
+
sonnet: [/balanced/, /everyday/, /strong model/],
|
|
167
|
+
haiku: [/ultra-fast/, /fast and affordable/, /small,|cost-efficient|simpler/, /\b(mini|nano|spark|lite)\b/],
|
|
168
|
+
};
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* Pick the account's codex slug for a tier, or null when nothing fits.
|
|
172
|
+
*
|
|
173
|
+
* The positional fallback is the point of the priority sort: with no wording
|
|
174
|
+
* match, "Most capable" takes the catalog's own top-ranked model and "Fastest"
|
|
175
|
+
* takes its last — still only ever a slug the CLI just listed (the 0504 rule),
|
|
176
|
+
* so the worst case is a slightly-off tier, never a failed run. "Balanced" has
|
|
177
|
+
* no honest positional answer in a two-model catalog, so it falls back to the
|
|
178
|
+
* top-ranked one rather than inventing a middle.
|
|
179
|
+
* @param {'opus'|'sonnet'|'haiku'|string} tier
|
|
180
|
+
* @param {{slug: string, text: string}[]} models
|
|
181
|
+
* @returns {string|null}
|
|
182
|
+
*/
|
|
183
|
+
export function resolveCodexModel(tier, models) {
|
|
184
|
+
const prefs = CODEX_TIER_PREFS[tier];
|
|
185
|
+
if (!prefs || !Array.isArray(models) || models.length === 0) return null;
|
|
186
|
+
for (const re of prefs) {
|
|
187
|
+
const hit = models.find((m) => m && typeof m.text === "string" && re.test(m.text));
|
|
188
|
+
if (hit) return hit.slug;
|
|
189
|
+
}
|
|
190
|
+
if (tier === "haiku") return models[models.length - 1].slug;
|
|
191
|
+
return models[0].slug;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* How each runtime-resolved vendor asks its own CLI what it can run: the list
|
|
196
|
+
* command, how to read its output, and the fallback binary name.
|
|
197
|
+
*/
|
|
198
|
+
const LIST_MODELS = {
|
|
199
|
+
cursor: {
|
|
200
|
+
bin: "cursor-agent",
|
|
201
|
+
args: ["--list-models"],
|
|
202
|
+
read: (stdout) => parseCursorModels(stdout),
|
|
203
|
+
pick: (tier, models) => resolveCursorModel(tier, models),
|
|
204
|
+
},
|
|
205
|
+
codex: {
|
|
206
|
+
bin: "codex",
|
|
207
|
+
args: ["debug", "models"],
|
|
208
|
+
read: (stdout) => parseCodexModels(stdout),
|
|
209
|
+
pick: (tier, models) => resolveCodexModel(tier, models),
|
|
210
|
+
// codex hands back its RAW lineup, including models a given login can't
|
|
211
|
+
// run — so it gets a second, cheaper question before the tier is picked.
|
|
212
|
+
authArgs: ["login", "status"],
|
|
213
|
+
narrow: (models, authStdout) => filterCodexModels(models, codexAuthMode(authStdout)),
|
|
214
|
+
},
|
|
215
|
+
};
|
|
216
|
+
|
|
68
217
|
/**
|
|
69
218
|
* Build a memoized `modelArgsFor(cfg, vendor)` → `["--model", id]` or [].
|
|
70
|
-
* Emits [] (tool default) when: the vendor
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
* (binary, tier) for the
|
|
74
|
-
*
|
|
219
|
+
* Emits [] (tool default) when: the vendor resolves no tier at run time
|
|
220
|
+
* (anything outside LIST_MODELS), no/default tier is configured, the user
|
|
221
|
+
* already pinned a model by hand in codingCmd, the list command fails, or
|
|
222
|
+
* nothing matches. Successful lookups are cached per (binary, tier) for the
|
|
223
|
+
* process lifetime; failures are NOT cached so a transient hiccup (offline,
|
|
224
|
+
* auth) retries on the next run.
|
|
75
225
|
*
|
|
76
226
|
* @param {{ run: (opts: object) => Promise<{status: number|null, stdout: string}> }} o
|
|
77
227
|
*/
|
|
78
228
|
export function createModelArgsResolver({ run } = {}) {
|
|
79
229
|
const cache = new Map();
|
|
230
|
+
const authCache = new Map(); // binary → the CLI's own login answer, asked once
|
|
80
231
|
return async function modelArgsFor(cfg, vendor) {
|
|
81
232
|
try {
|
|
233
|
+
const spec = LIST_MODELS[vendor];
|
|
82
234
|
const tier = String(cfg?.codingModel || "").trim();
|
|
83
|
-
if (
|
|
235
|
+
if (!spec || !tier || tier === "default") return [];
|
|
84
236
|
const cmd = String(cfg?.codingCmd || "");
|
|
85
|
-
|
|
86
|
-
|
|
237
|
+
// Hand-pinned wins — including codex's short spelling (`-m gpt-5.6-sol`),
|
|
238
|
+
// which is what its own help leads with.
|
|
239
|
+
if (/(^|\s)(--model|-m)(\s|=)/.test(cmd + " ")) return [];
|
|
240
|
+
const bin = cmd.trim().split(/\s+/)[0] || spec.bin;
|
|
87
241
|
const key = `${bin} ${tier}`;
|
|
88
242
|
if (cache.has(key)) return cache.get(key);
|
|
89
|
-
const r = await run({ cmd: bin, args:
|
|
243
|
+
const r = await run({ cmd: bin, args: spec.args, timeoutMs: 30000, heartbeatMs: 0 });
|
|
90
244
|
if (!r || r.status !== 0) return [];
|
|
91
|
-
|
|
245
|
+
let models = spec.read(r.stdout);
|
|
246
|
+
if (spec.narrow) {
|
|
247
|
+
if (!authCache.has(bin)) {
|
|
248
|
+
const a = await run({ cmd: bin, args: spec.authArgs, timeoutMs: 15000, heartbeatMs: 0 });
|
|
249
|
+
// BOTH streams: `codex login status` prints its one line on STDERR
|
|
250
|
+
// (verified live — reading stdout alone returns nothing, which would
|
|
251
|
+
// silently mean "unknown" forever and quietly disable the narrowing).
|
|
252
|
+
// A failed/absent answer caches as "" → mode unknown → no narrowing.
|
|
253
|
+
authCache.set(
|
|
254
|
+
bin,
|
|
255
|
+
a && a.status === 0 ? `${String(a.stdout || "")}\n${String(a.stderr || "")}` : "",
|
|
256
|
+
);
|
|
257
|
+
}
|
|
258
|
+
models = spec.narrow(models, authCache.get(bin));
|
|
259
|
+
}
|
|
260
|
+
const id = spec.pick(tier, models);
|
|
92
261
|
const args = id ? ["--model", id] : [];
|
|
93
262
|
cache.set(key, args);
|
|
94
263
|
return args;
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
// The one block-on-a-human decision loop every vendor adapter shares (0777).
|
|
2
|
+
//
|
|
3
|
+
// opencode proved this flow first (0593 over HTTP/SSE, 0759 over ACP): raise the
|
|
4
|
+
// durable hilos card, poll it under a hard deadline, and answer the agent with
|
|
5
|
+
// the human's decision — with EVERY failure mode (callback error, transport
|
|
6
|
+
// death, timeout, cancellation) resolving to a rejection AND settling the
|
|
7
|
+
// durable card fail-closed, so a card can never be left pending while the tool
|
|
8
|
+
// call proceeds.
|
|
9
|
+
//
|
|
10
|
+
// It lived inline in acp-session.mjs until claude_code and codex needed the
|
|
11
|
+
// identical loop over completely different wires (an MCP permission-prompt tool
|
|
12
|
+
// and codex's elicitation channel). Extracting it is what makes "the same card,
|
|
13
|
+
// the same block, the same timeout" a mechanical fact instead of three
|
|
14
|
+
// hand-copied implementations that can drift apart.
|
|
15
|
+
|
|
16
|
+
import { mapOpenCodePermissionDecision } from "./opencode-permissions.mjs";
|
|
17
|
+
|
|
18
|
+
export const DEFAULT_PERMISSION_TIMEOUT_MS = 30 * 60_000;
|
|
19
|
+
export const DEFAULT_PERMISSION_POLL_MS = 1_000;
|
|
20
|
+
|
|
21
|
+
function abortError(reason = "cancelled") {
|
|
22
|
+
const error = new Error(String(reason || "cancelled"));
|
|
23
|
+
error.name = "AbortError";
|
|
24
|
+
return error;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* What the room is told when a gated run turns out to be ungateable (0785).
|
|
29
|
+
*
|
|
30
|
+
* A gate that quietly isn't there is worse than no gate, because the room
|
|
31
|
+
* believes it is. Two CLIs can land here: a codex without `mcp-server`, and a
|
|
32
|
+
* Claude Code that rejects `--permission-prompt-tool`. Both mean the same thing
|
|
33
|
+
* to a person watching the thread, so they say the same thing.
|
|
34
|
+
*
|
|
35
|
+
* @param {string} [tool] the binary that couldn't ask, when we know it
|
|
36
|
+
*/
|
|
37
|
+
export function ungatedRunNotice(tool) {
|
|
38
|
+
const name = typeof tool === "string" ? tool.trim() : "";
|
|
39
|
+
const who = name ? `\`${name}\` on this machine` : "the tool running this";
|
|
40
|
+
return `This run executes without asking first — ${who} can't raise approval requests. A newer version can.`;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* A once-per-RUN latch for that notice.
|
|
45
|
+
*
|
|
46
|
+
* The gate is a property of the transport, decided once when the CLI starts —
|
|
47
|
+
* not of each ask. A run that degrades and then makes forty tool calls owes the
|
|
48
|
+
* room one line, and a run that retries its CLI (a gated iterate runs it more
|
|
49
|
+
* than once) still owes exactly one.
|
|
50
|
+
*
|
|
51
|
+
* Callers await this BEFORE spawning the ungated child, so the room is warned
|
|
52
|
+
* while the run can still be stopped rather than told afterwards what it
|
|
53
|
+
* already did. That is also why the post is bounded: it must go out first, but
|
|
54
|
+
* a stalled network call must not hold a run hostage — past the deadline we
|
|
55
|
+
* stop waiting and let the send land whenever it lands. Failure is swallowed
|
|
56
|
+
* either way; a notice must never take the run down with it.
|
|
57
|
+
*
|
|
58
|
+
* @param {{ post: (body: string) => Promise<unknown>, timeoutMs?: number,
|
|
59
|
+
* log?: { error?: (m: string) => void } }} o
|
|
60
|
+
* @returns {(tool?: string) => Promise<boolean>} true only when the post landed
|
|
61
|
+
*/
|
|
62
|
+
export function createUngatedRunNotice({ post, timeoutMs = 10_000, log } = {}) {
|
|
63
|
+
let posted = false;
|
|
64
|
+
return async function noticeUngatedRun(tool) {
|
|
65
|
+
if (posted) return false;
|
|
66
|
+
posted = true;
|
|
67
|
+
let timer = null;
|
|
68
|
+
try {
|
|
69
|
+
const sent = Promise.resolve(post(ungatedRunNotice(tool)));
|
|
70
|
+
// Keep the rejection handled even if the deadline wins the race.
|
|
71
|
+
sent.catch(() => {});
|
|
72
|
+
const deadline = new Promise((resolve) => {
|
|
73
|
+
timer = setTimeout(() => resolve("timeout"), Math.max(1, timeoutMs));
|
|
74
|
+
timer?.unref?.();
|
|
75
|
+
});
|
|
76
|
+
const outcome = await Promise.race([sent.then(() => "sent"), deadline]);
|
|
77
|
+
if (outcome === "timeout") {
|
|
78
|
+
log?.error?.("ungated-run notice: still sending past its deadline; not holding the run");
|
|
79
|
+
return false;
|
|
80
|
+
}
|
|
81
|
+
return true;
|
|
82
|
+
} catch (error) {
|
|
83
|
+
log?.error?.(`ungated-run notice: ${error?.message ?? error}`);
|
|
84
|
+
return false;
|
|
85
|
+
} finally {
|
|
86
|
+
if (timer) clearTimeout(timer);
|
|
87
|
+
}
|
|
88
|
+
};
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function raceWithAbort(promise, signal) {
|
|
92
|
+
if (!signal) return Promise.resolve(promise);
|
|
93
|
+
if (signal.aborted) return Promise.reject(abortError(signal.reason));
|
|
94
|
+
return new Promise((resolve, reject) => {
|
|
95
|
+
const onAbort = () => reject(abortError(signal.reason));
|
|
96
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
97
|
+
Promise.resolve(promise).then(
|
|
98
|
+
(value) => {
|
|
99
|
+
signal.removeEventListener("abort", onAbort);
|
|
100
|
+
resolve(value);
|
|
101
|
+
},
|
|
102
|
+
(error) => {
|
|
103
|
+
signal.removeEventListener("abort", onAbort);
|
|
104
|
+
reject(error);
|
|
105
|
+
},
|
|
106
|
+
);
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
const defaultSleep = (ms) => new Promise((resolve) => setTimeout(resolve, Math.max(0, ms)));
|
|
111
|
+
|
|
112
|
+
/** How long a fail-closed settlement gets before we stop waiting on it. The
|
|
113
|
+
* card is already being rejected on the wire; a hung settlement call must not
|
|
114
|
+
* hold the agent's tool call open on top of it. */
|
|
115
|
+
const SETTLEMENT_TIMEOUT_MS = 10_000;
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Race a callback against BOTH the abort signal and a wall-clock deadline.
|
|
119
|
+
*
|
|
120
|
+
* The deadline check between polls is not enough on its own: these callbacks
|
|
121
|
+
* are network calls (the daemon's MCP fetch has no timeout of its own), so a
|
|
122
|
+
* single stalled request could sit past the deadline forever and the ask would
|
|
123
|
+
* neither resolve nor fail. Bounding each call is what makes the deadline real.
|
|
124
|
+
*/
|
|
125
|
+
function withDeadline(promise, { signal, deadlineAt, now, label }) {
|
|
126
|
+
const remaining = deadlineAt - now();
|
|
127
|
+
if (remaining <= 0) return Promise.reject(abortError("timeout"));
|
|
128
|
+
let timer = null;
|
|
129
|
+
const timeout = new Promise((_resolve, reject) => {
|
|
130
|
+
timer = setTimeout(() => reject(abortError("timeout")), remaining);
|
|
131
|
+
timer?.unref?.();
|
|
132
|
+
});
|
|
133
|
+
return Promise.race([raceWithAbort(promise, signal), timeout]).finally(() => {
|
|
134
|
+
if (timer) clearTimeout(timer);
|
|
135
|
+
void label;
|
|
136
|
+
});
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Raise one permission card and block until a human decides it.
|
|
141
|
+
*
|
|
142
|
+
* Never throws: an unresolvable ask is a rejection, because the alternative —
|
|
143
|
+
* letting the tool call through — is the one outcome that cannot be undone.
|
|
144
|
+
*
|
|
145
|
+
* @param {{
|
|
146
|
+
* request: object,
|
|
147
|
+
* requestPermission: (request: object, context: object) => Promise<unknown>,
|
|
148
|
+
* getPermissionDecision: (handle: unknown, context: object) => Promise<unknown>,
|
|
149
|
+
* signal?: AbortSignal,
|
|
150
|
+
* timeoutMs?: number,
|
|
151
|
+
* deadlineAt?: number,
|
|
152
|
+
* pollIntervalMs?: number,
|
|
153
|
+
* settlementTimeoutMs?: number,
|
|
154
|
+
* sleep?: (ms: number) => Promise<void>,
|
|
155
|
+
* now?: () => number,
|
|
156
|
+
* log?: { error?: (message: string) => void },
|
|
157
|
+
* label?: string,
|
|
158
|
+
* }} options
|
|
159
|
+
* @returns {Promise<{ reply: "once"|"always"|"reject", outcome: string, handle: unknown }>}
|
|
160
|
+
*/
|
|
161
|
+
export async function resolveHilosPermissionReply({
|
|
162
|
+
request,
|
|
163
|
+
requestPermission,
|
|
164
|
+
getPermissionDecision,
|
|
165
|
+
signal,
|
|
166
|
+
timeoutMs = DEFAULT_PERMISSION_TIMEOUT_MS,
|
|
167
|
+
deadlineAt,
|
|
168
|
+
pollIntervalMs = DEFAULT_PERMISSION_POLL_MS,
|
|
169
|
+
settlementTimeoutMs = SETTLEMENT_TIMEOUT_MS,
|
|
170
|
+
sleep = defaultSleep,
|
|
171
|
+
now = () => Date.now(),
|
|
172
|
+
log,
|
|
173
|
+
label = "permission",
|
|
174
|
+
}) {
|
|
175
|
+
if (typeof requestPermission !== "function" || typeof getPermissionDecision !== "function") {
|
|
176
|
+
throw new Error("permission gate requires requestPermission and getPermissionDecision");
|
|
177
|
+
}
|
|
178
|
+
const deadline = deadlineAt ?? now() + Math.max(1, timeoutMs || DEFAULT_PERMISSION_TIMEOUT_MS);
|
|
179
|
+
let handle = null;
|
|
180
|
+
let reply = "reject";
|
|
181
|
+
let outcome = "transport-error";
|
|
182
|
+
// 0813 — the envelope that actually SETTLED this ask. A workspace rule can
|
|
183
|
+
// refuse before any human sees the card, and its reason is the one thing a
|
|
184
|
+
// standing rule can hand the model that a one-off rejection cannot. Only the
|
|
185
|
+
// seams that carry a message use it; the rest ignore it.
|
|
186
|
+
let settled = null;
|
|
187
|
+
try {
|
|
188
|
+
if (signal?.aborted) throw abortError(signal.reason);
|
|
189
|
+
handle = await withDeadline(requestPermission(request, { signal, deadlineAt: deadline }), {
|
|
190
|
+
signal,
|
|
191
|
+
deadlineAt: deadline,
|
|
192
|
+
now,
|
|
193
|
+
label: "request",
|
|
194
|
+
});
|
|
195
|
+
let decision = handle;
|
|
196
|
+
settled = decision;
|
|
197
|
+
let mapped = mapOpenCodePermissionDecision(decision);
|
|
198
|
+
while (!mapped) {
|
|
199
|
+
if (now() >= deadline) throw abortError("timeout");
|
|
200
|
+
decision = await withDeadline(
|
|
201
|
+
getPermissionDecision(handle, { request, signal, deadlineAt: deadline }),
|
|
202
|
+
{ signal, deadlineAt: deadline, now, label: "poll" },
|
|
203
|
+
);
|
|
204
|
+
settled = decision;
|
|
205
|
+
mapped = mapOpenCodePermissionDecision(decision);
|
|
206
|
+
if (mapped) break;
|
|
207
|
+
const waitMs = Math.min(Math.max(1, pollIntervalMs), Math.max(1, deadline - now()));
|
|
208
|
+
await raceWithAbort(sleep(waitMs), signal);
|
|
209
|
+
}
|
|
210
|
+
reply = mapped;
|
|
211
|
+
outcome = "decided";
|
|
212
|
+
} catch (error) {
|
|
213
|
+
reply = "reject";
|
|
214
|
+
outcome = error?.message === "timeout" ? "timeout" : signal?.aborted ? "aborted" : "transport-error";
|
|
215
|
+
log?.error?.(`${label} decision: ${error?.message ?? error}`);
|
|
216
|
+
// The agent is about to be failed closed. Settle the durable card too, on a
|
|
217
|
+
// bounded one-shot call with NO run signal — the run's signal is necessarily
|
|
218
|
+
// aborted on this path and must not strand a card visibly pending.
|
|
219
|
+
if (handle) {
|
|
220
|
+
// Best-effort, on its OWN short clock: the run deadline may already be
|
|
221
|
+
// blown (that is often why we are here), and a settlement call that
|
|
222
|
+
// stalls must not hold the agent's tool call open behind a rejection it
|
|
223
|
+
// has already earned.
|
|
224
|
+
try {
|
|
225
|
+
await withDeadline(
|
|
226
|
+
getPermissionDecision(handle, { request, deadlineAt: deadline, failClosed: true }),
|
|
227
|
+
{ deadlineAt: now() + Math.max(1, settlementTimeoutMs), now, label: "settlement" },
|
|
228
|
+
);
|
|
229
|
+
} catch (settlementError) {
|
|
230
|
+
log?.error?.(`${label} settlement: ${settlementError?.message ?? settlementError}`);
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
return {
|
|
235
|
+
reply,
|
|
236
|
+
outcome,
|
|
237
|
+
handle,
|
|
238
|
+
byPolicy: isPolicyDecision(settled),
|
|
239
|
+
policyReason: policyReasonFrom(settled),
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/**
|
|
244
|
+
* Was a workspace RULE what settled this, rather than a person?
|
|
245
|
+
*
|
|
246
|
+
* Kept separate from the reason on purpose. The reason is optional — a rule may
|
|
247
|
+
* carry none — so deriving "policy" from "has a reason" would quietly send a
|
|
248
|
+
* reasonless rule's refusal back as "a person declined this", which is exactly
|
|
249
|
+
* the false statement this pair exists to prevent. Provenance is the fact; the
|
|
250
|
+
* reason is a bonus.
|
|
251
|
+
*/
|
|
252
|
+
export function isPolicyDecision(envelope) {
|
|
253
|
+
if (!envelope || typeof envelope !== "object") return false;
|
|
254
|
+
return (envelope.decisionSource ?? envelope.decision_source) === "workspace_policy";
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* The workspace rule's own words, when a rule is what refused this and it
|
|
259
|
+
* bothered to say why.
|
|
260
|
+
*
|
|
261
|
+
* Deliberately narrow: only a `workspace_policy` decision source yields one, so
|
|
262
|
+
* a human's rejection can never be dressed up as policy, and a server that
|
|
263
|
+
* sends nothing yields null rather than a guess.
|
|
264
|
+
*/
|
|
265
|
+
export function policyReasonFrom(envelope) {
|
|
266
|
+
if (!isPolicyDecision(envelope)) return null;
|
|
267
|
+
const reason = envelope.policyReason ?? envelope.policy_reason;
|
|
268
|
+
return typeof reason === "string" && reason.trim() ? reason.trim() : null;
|
|
269
|
+
}
|