hilos-agent 0.9.0 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,142 @@
1
+ // Turn-scoped loopback transport for a resumed Codex session (0854).
2
+ //
3
+ // Codex's remote MCP auth is normally an environment variable. The daemon
4
+ // deliberately strips HILOS_* secrets from coding children, and handing a
5
+ // persistent workspace token to a model-visible shell would undo that safety.
6
+ // Instead, keep the token in the daemon process and expose a random loopback
7
+ // URL only for the lifetime of one continuation. Codex can use every tool its
8
+ // agent identity is authorized for; it never receives the bearer credential.
9
+
10
+ import { createServer } from "node:http";
11
+ import { randomBytes } from "node:crypto";
12
+ import { DAEMON_CLIENT } from "./mcp.mjs";
13
+
14
+ const MAX_BODY_BYTES = 2 * 1024 * 1024;
15
+ const FORWARDED_RESPONSE_HEADERS = [
16
+ "content-type",
17
+ "mcp-session-id",
18
+ "mcp-protocol-version",
19
+ "www-authenticate",
20
+ ];
21
+ const FORWARDED_REQUEST_HEADERS = [
22
+ "content-type",
23
+ "accept",
24
+ "mcp-session-id",
25
+ "mcp-protocol-version",
26
+ "mcp-method",
27
+ "mcp-name",
28
+ ];
29
+ const MAX_HEADER_LENGTH = 4096;
30
+
31
+ function readBody(request) {
32
+ return new Promise((resolve, reject) => {
33
+ let size = 0;
34
+ const chunks = [];
35
+ request.on("data", (chunk) => {
36
+ size += chunk.length;
37
+ if (size > MAX_BODY_BYTES) {
38
+ reject(new Error("request body too large"));
39
+ request.destroy();
40
+ return;
41
+ }
42
+ chunks.push(chunk);
43
+ });
44
+ request.on("end", () => resolve(Buffer.concat(chunks)));
45
+ request.on("error", reject);
46
+ });
47
+ }
48
+
49
+ /**
50
+ * Start an unguessable, loopback-only HTTP proxy to one authenticated hilos MCP
51
+ * endpoint. The caller must close it in finally after the coding turn.
52
+ * @param {{
53
+ * url?: string,
54
+ * token?: string,
55
+ * channelId?: string,
56
+ * fetchImpl?: typeof fetch,
57
+ * }} [options]
58
+ * @returns {Promise<{url: string, close: () => Promise<void>} | null>}
59
+ */
60
+ export async function startHilosMcpLoopback({ url, token, channelId = "", fetchImpl = fetch } = {}) {
61
+ if (!url || !token) return null;
62
+ const nonce = randomBytes(24).toString("hex");
63
+ const path = `/mcp/${nonce}`;
64
+ const upstream = new URL(url);
65
+ // Preserve the room that caused this turn as MCP's effective context. This
66
+ // does not widen or narrow token access, but it keeps guest-room memory gates
67
+ // and other context-sensitive server policy active even when the model calls
68
+ // a workspace-scoped tool without a channelId argument.
69
+ if (channelId) upstream.searchParams.set("channelId", channelId);
70
+ const upstreamRequests = new Set();
71
+
72
+ const server = createServer(async (request, response) => {
73
+ if (request.url !== path || !["POST", "GET", "DELETE"].includes(request.method || "")) {
74
+ response.writeHead(404).end();
75
+ return;
76
+ }
77
+ const controller = new AbortController();
78
+ upstreamRequests.add(controller);
79
+ const abortUpstream = () => controller.abort();
80
+ request.once("aborted", abortUpstream);
81
+ response.once("close", abortUpstream);
82
+ try {
83
+ const body = request.method === "POST" ? await readBody(request) : undefined;
84
+ const headers = {
85
+ authorization: `Bearer ${token}`,
86
+ // This short-lived proxy is part of the persistent daemon's return
87
+ // path. Marking it on-demand would temporarily relabel a live daemon
88
+ // and could suppress the offline/wakeup behavior after the turn.
89
+ "x-hilos-connection-mode": "daemon",
90
+ ...(DAEMON_CLIENT ? { "x-hilos-client": DAEMON_CLIENT } : {}),
91
+ };
92
+ for (const name of FORWARDED_REQUEST_HEADERS) {
93
+ const value = request.headers[name];
94
+ if (typeof value === "string" && value.length <= MAX_HEADER_LENGTH) {
95
+ headers[name] = value;
96
+ }
97
+ }
98
+ const forwarded = await fetchImpl(upstream, {
99
+ method: request.method,
100
+ headers,
101
+ ...(body ? { body } : {}),
102
+ signal: controller.signal,
103
+ });
104
+ for (const name of FORWARDED_RESPONSE_HEADERS) {
105
+ const value = forwarded.headers.get(name);
106
+ if (value) response.setHeader(name, value);
107
+ }
108
+ response.writeHead(forwarded.status);
109
+ response.end(Buffer.from(await forwarded.arrayBuffer()));
110
+ } catch {
111
+ if (!response.destroyed && !response.writableEnded) {
112
+ response.writeHead(502, { "content-type": "application/json" });
113
+ response.end(JSON.stringify({ error: "hilos MCP loopback unavailable" }));
114
+ }
115
+ } finally {
116
+ upstreamRequests.delete(controller);
117
+ request.removeListener("aborted", abortUpstream);
118
+ response.removeListener("close", abortUpstream);
119
+ }
120
+ });
121
+
122
+ await new Promise((resolve, reject) => {
123
+ server.once("error", reject);
124
+ server.listen(0, "127.0.0.1", resolve);
125
+ });
126
+ const address = server.address();
127
+ if (!address || typeof address === "string") {
128
+ server.close();
129
+ return null;
130
+ }
131
+ let closed = false;
132
+ return {
133
+ url: `http://127.0.0.1:${address.port}${path}`,
134
+ async close() {
135
+ if (closed) return;
136
+ closed = true;
137
+ for (const controller of upstreamRequests) controller.abort();
138
+ server.closeAllConnections?.();
139
+ await new Promise((resolve) => server.close(() => resolve()));
140
+ },
141
+ };
142
+ }
package/src/mcp.mjs CHANGED
@@ -3,7 +3,7 @@
3
3
  import os from "node:os";
4
4
 
5
5
  // Best-effort host label so hilos can show "whose machine" this daemon runs on.
6
- const CLIENT = (() => {
6
+ export const DAEMON_CLIENT = (() => {
7
7
  try {
8
8
  return `${os.userInfo().username}@${os.hostname()}`;
9
9
  } catch {
@@ -20,7 +20,8 @@ export function makeClient({ url, token }) {
20
20
  headers: {
21
21
  "content-type": "application/json",
22
22
  authorization: `Bearer ${token}`,
23
- ...(CLIENT ? { "x-hilos-client": CLIENT } : {}),
23
+ "x-hilos-connection-mode": "daemon",
24
+ ...(DAEMON_CLIENT ? { "x-hilos-client": DAEMON_CLIENT } : {}),
24
25
  },
25
26
  body: JSON.stringify({ jsonrpc: "2.0", id: ++id, method, params }),
26
27
  });
@@ -11,7 +11,13 @@
11
11
  // PURE + node-builtins-only; the resolver takes an injected `run` (runCli) so
12
12
  // tests drive it with no CLI; a resolution failure NEVER breaks a run ([]).
13
13
  //
14
- // codex stays out: its CLI has no verified model-list command (0504 notes).
14
+ // codex joined in 0783 on the SAME terms: `codex debug models` ("Render the raw
15
+ // model catalog as JSON") is its `--list-models`, so its tier also resolves at
16
+ // run time against the account's own catalog and never against a baked slug.
17
+ // That matters more for codex than for anyone: its slugs are codenames that rev
18
+ // (`gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna` in the live catalog) and are
19
+ // plan-gated — a run on a ChatGPT account rejects `gpt-5` outright with "The
20
+ // 'gpt-5' model is not supported when using Codex with a ChatGPT account".
15
21
 
16
22
  /** Strip ANSI SGR color codes (`--list-models` output is colorized). */
17
23
  export function stripAnsi(s) {
@@ -65,30 +71,193 @@ export function resolveCursorModel(tier, ids) {
65
71
  return null;
66
72
  }
67
73
 
74
+ /**
75
+ * Parse `codex debug models` stdout → the account's LISTED models, ranked by
76
+ * the catalog's own `priority` (lower first — it is Codex's own ordering of its
77
+ * lineup). Wire shape captured live against codex-cli 0.144.1:
78
+ *
79
+ * {"models":[{"slug":"gpt-5.6-sol","display_name":"GPT-5.6-Sol",
80
+ * "description":"Latest frontier agentic coding model.","priority":1,
81
+ * "visibility":"list", …}, …]}
82
+ *
83
+ * `visibility:"hide"` entries are internal (the live catalog hides
84
+ * `codex-auto-review`, Codex's own approval-review model) and are dropped — the
85
+ * user's preset must never select something the CLI doesn't offer. A drifted
86
+ * or non-JSON payload → [] and the preset silently falls back to the tool
87
+ * default, same as the cursor path.
88
+ *
89
+ * `supported_in_api` rides along rather than being filtered here, because the
90
+ * catalog is the RAW lineup — the same bytes whichever way the account is
91
+ * logged in (verified live) — and whether an api-false model is usable depends
92
+ * on that login. `gpt-5.3-codex-spark` is the live example: api-false, and
93
+ * perfectly runnable on a ChatGPT account. A row that doesn't state the field
94
+ * counts as supported; we never filter on something the catalog didn't say.
95
+ * @param {string} stdout
96
+ * @returns {{slug: string, text: string, supportedInApi: boolean}[]}
97
+ */
98
+ export function parseCodexModels(stdout) {
99
+ let payload;
100
+ try {
101
+ payload = JSON.parse(String(stdout || ""));
102
+ } catch {
103
+ return [];
104
+ }
105
+ const models = payload && Array.isArray(payload.models) ? payload.models : [];
106
+ const rows = [];
107
+ for (const m of models) {
108
+ if (!m || typeof m !== "object") continue;
109
+ if (typeof m.slug !== "string" || !m.slug) continue;
110
+ if (typeof m.visibility === "string" && m.visibility !== "list") continue;
111
+ const priority = Number.isFinite(m.priority) ? m.priority : Number.MAX_SAFE_INTEGER;
112
+ const text = `${m.description || ""} ${m.display_name || ""} ${m.slug}`.toLowerCase();
113
+ rows.push({ slug: m.slug, text, priority, supportedInApi: m.supported_in_api !== false });
114
+ }
115
+ rows.sort((a, b) => a.priority - b.priority);
116
+ return rows.map(({ slug, text, supportedInApi }) => ({ slug, text, supportedInApi }));
117
+ }
118
+
119
+ /**
120
+ * Which login `codex login status` is reporting. The CLI's own one-line answer
121
+ * ("Logged in using ChatGPT" on the live install, 0.4s, printed on STDERR) —
122
+ * asking it beats parsing `~/.codex/auth.json`, which is private and can move
123
+ * with CODEX_HOME. Feed it both streams.
124
+ *
125
+ * ONLY a positive "api key" reading counts as apikey; everything else,
126
+ * including an empty/failed/unrecognized answer, is "unknown" and changes
127
+ * nothing downstream. The ChatGPT wording is live-verified; the API-key wording
128
+ * is not (I had no API-key install to log into, and would not clobber a real
129
+ * login to make one), so the matcher is deliberately loose on that side and
130
+ * fail-safe on every other.
131
+ * @param {string} stdout
132
+ * @returns {'apikey'|'chatgpt'|'unknown'}
133
+ */
134
+ export function codexAuthMode(stdout) {
135
+ const s = String(stdout || "").toLowerCase();
136
+ if (/api[\s_-]?key/.test(s)) return "apikey";
137
+ if (/chatgpt/.test(s)) return "chatgpt";
138
+ return "unknown";
139
+ }
140
+
141
+ /**
142
+ * Drop models this login cannot actually run. Only bites for `apikey`: the
143
+ * catalog's `supported_in_api:false` rows are exactly the ones an API-key run
144
+ * rejects, and the "Fastest" tier would otherwise reach for one
145
+ * (`gpt-5.3-codex-spark` is api-false in the live catalog). chatgpt/unknown
146
+ * pass through untouched — under ChatGPT auth those models are fine, and an
147
+ * unreadable auth answer must never silently narrow a user's lineup.
148
+ * @param {{slug: string, text: string, supportedInApi: boolean}[]} models
149
+ * @param {'apikey'|'chatgpt'|'unknown'} mode
150
+ */
151
+ export function filterCodexModels(models, mode) {
152
+ if (!Array.isArray(models)) return [];
153
+ if (mode !== "apikey") return models;
154
+ return models.filter((m) => m && m.supportedInApi !== false);
155
+ }
156
+
157
+ // Ranked preferences per tier for codex, matched against each model's own
158
+ // description + display name + slug. Codex's catalog SAYS which tier a model
159
+ // is ("Latest frontier agentic coding model." / "Balanced agentic coding model
160
+ // for everyday work." / "Fast and affordable…"), and unlike the slugs those
161
+ // words are stable — reading them beats pattern-matching codenames that rev
162
+ // every release. First pattern with any match wins, then the first model in
163
+ // the catalog's own priority order that matches it.
164
+ const CODEX_TIER_PREFS = {
165
+ opus: [/frontier/, /complex/],
166
+ sonnet: [/balanced/, /everyday/, /strong model/],
167
+ haiku: [/ultra-fast/, /fast and affordable/, /small,|cost-efficient|simpler/, /\b(mini|nano|spark|lite)\b/],
168
+ };
169
+
170
+ /**
171
+ * Pick the account's codex slug for a tier, or null when nothing fits.
172
+ *
173
+ * The positional fallback is the point of the priority sort: with no wording
174
+ * match, "Most capable" takes the catalog's own top-ranked model and "Fastest"
175
+ * takes its last — still only ever a slug the CLI just listed (the 0504 rule),
176
+ * so the worst case is a slightly-off tier, never a failed run. "Balanced" has
177
+ * no honest positional answer in a two-model catalog, so it falls back to the
178
+ * top-ranked one rather than inventing a middle.
179
+ * @param {'opus'|'sonnet'|'haiku'|string} tier
180
+ * @param {{slug: string, text: string}[]} models
181
+ * @returns {string|null}
182
+ */
183
+ export function resolveCodexModel(tier, models) {
184
+ const prefs = CODEX_TIER_PREFS[tier];
185
+ if (!prefs || !Array.isArray(models) || models.length === 0) return null;
186
+ for (const re of prefs) {
187
+ const hit = models.find((m) => m && typeof m.text === "string" && re.test(m.text));
188
+ if (hit) return hit.slug;
189
+ }
190
+ if (tier === "haiku") return models[models.length - 1].slug;
191
+ return models[0].slug;
192
+ }
193
+
194
+ /**
195
+ * How each runtime-resolved vendor asks its own CLI what it can run: the list
196
+ * command, how to read its output, and the fallback binary name.
197
+ */
198
+ const LIST_MODELS = {
199
+ cursor: {
200
+ bin: "cursor-agent",
201
+ args: ["--list-models"],
202
+ read: (stdout) => parseCursorModels(stdout),
203
+ pick: (tier, models) => resolveCursorModel(tier, models),
204
+ },
205
+ codex: {
206
+ bin: "codex",
207
+ args: ["debug", "models"],
208
+ read: (stdout) => parseCodexModels(stdout),
209
+ pick: (tier, models) => resolveCodexModel(tier, models),
210
+ // codex hands back its RAW lineup, including models a given login can't
211
+ // run — so it gets a second, cheaper question before the tier is picked.
212
+ authArgs: ["login", "status"],
213
+ narrow: (models, authStdout) => filterCodexModels(models, codexAuthMode(authStdout)),
214
+ },
215
+ };
216
+
68
217
  /**
69
218
  * Build a memoized `modelArgsFor(cfg, vendor)` → `["--model", id]` or [].
70
- * Emits [] (tool default) when: the vendor isn't cursor, no/default tier is
71
- * configured, the user already pinned `--model` by hand in codingCmd, the
72
- * list command fails, or nothing matches. Successful lookups are cached per
73
- * (binary, tier) for the process lifetime; failures are NOT cached so a
74
- * transient hiccup (offline, auth) retries on the next run.
219
+ * Emits [] (tool default) when: the vendor resolves no tier at run time
220
+ * (anything outside LIST_MODELS), no/default tier is configured, the user
221
+ * already pinned a model by hand in codingCmd, the list command fails, or
222
+ * nothing matches. Successful lookups are cached per (binary, tier) for the
223
+ * process lifetime; failures are NOT cached so a transient hiccup (offline,
224
+ * auth) retries on the next run.
75
225
  *
76
226
  * @param {{ run: (opts: object) => Promise<{status: number|null, stdout: string}> }} o
77
227
  */
78
228
  export function createModelArgsResolver({ run } = {}) {
79
229
  const cache = new Map();
230
+ const authCache = new Map(); // binary → the CLI's own login answer, asked once
80
231
  return async function modelArgsFor(cfg, vendor) {
81
232
  try {
233
+ const spec = LIST_MODELS[vendor];
82
234
  const tier = String(cfg?.codingModel || "").trim();
83
- if (vendor !== "cursor" || !tier || tier === "default") return [];
235
+ if (!spec || !tier || tier === "default") return [];
84
236
  const cmd = String(cfg?.codingCmd || "");
85
- if (/(^|\s)--model(\s|=)/.test(cmd + " ")) return []; // hand-pinned wins
86
- const bin = cmd.trim().split(/\s+/)[0] || "cursor-agent";
237
+ // Hand-pinned wins — including codex's short spelling (`-m gpt-5.6-sol`),
238
+ // which is what its own help leads with.
239
+ if (/(^|\s)(--model|-m)(\s|=)/.test(cmd + " ")) return [];
240
+ const bin = cmd.trim().split(/\s+/)[0] || spec.bin;
87
241
  const key = `${bin} ${tier}`;
88
242
  if (cache.has(key)) return cache.get(key);
89
- const r = await run({ cmd: bin, args: ["--list-models"], timeoutMs: 30000, heartbeatMs: 0 });
243
+ const r = await run({ cmd: bin, args: spec.args, timeoutMs: 30000, heartbeatMs: 0 });
90
244
  if (!r || r.status !== 0) return [];
91
- const id = resolveCursorModel(tier, parseCursorModels(r.stdout));
245
+ let models = spec.read(r.stdout);
246
+ if (spec.narrow) {
247
+ if (!authCache.has(bin)) {
248
+ const a = await run({ cmd: bin, args: spec.authArgs, timeoutMs: 15000, heartbeatMs: 0 });
249
+ // BOTH streams: `codex login status` prints its one line on STDERR
250
+ // (verified live — reading stdout alone returns nothing, which would
251
+ // silently mean "unknown" forever and quietly disable the narrowing).
252
+ // A failed/absent answer caches as "" → mode unknown → no narrowing.
253
+ authCache.set(
254
+ bin,
255
+ a && a.status === 0 ? `${String(a.stdout || "")}\n${String(a.stderr || "")}` : "",
256
+ );
257
+ }
258
+ models = spec.narrow(models, authCache.get(bin));
259
+ }
260
+ const id = spec.pick(tier, models);
92
261
  const args = id ? ["--model", id] : [];
93
262
  cache.set(key, args);
94
263
  return args;
@@ -0,0 +1,269 @@
1
+ // The one block-on-a-human decision loop every vendor adapter shares (0777).
2
+ //
3
+ // opencode proved this flow first (0593 over HTTP/SSE, 0759 over ACP): raise the
4
+ // durable hilos card, poll it under a hard deadline, and answer the agent with
5
+ // the human's decision — with EVERY failure mode (callback error, transport
6
+ // death, timeout, cancellation) resolving to a rejection AND settling the
7
+ // durable card fail-closed, so a card can never be left pending while the tool
8
+ // call proceeds.
9
+ //
10
+ // It lived inline in acp-session.mjs until claude_code and codex needed the
11
+ // identical loop over completely different wires (an MCP permission-prompt tool
12
+ // and codex's elicitation channel). Extracting it is what makes "the same card,
13
+ // the same block, the same timeout" a mechanical fact instead of three
14
+ // hand-copied implementations that can drift apart.
15
+
16
+ import { mapOpenCodePermissionDecision } from "./opencode-permissions.mjs";
17
+
18
+ export const DEFAULT_PERMISSION_TIMEOUT_MS = 30 * 60_000;
19
+ export const DEFAULT_PERMISSION_POLL_MS = 1_000;
20
+
21
+ function abortError(reason = "cancelled") {
22
+ const error = new Error(String(reason || "cancelled"));
23
+ error.name = "AbortError";
24
+ return error;
25
+ }
26
+
27
+ /**
28
+ * What the room is told when a gated run turns out to be ungateable (0785).
29
+ *
30
+ * A gate that quietly isn't there is worse than no gate, because the room
31
+ * believes it is. Two CLIs can land here: a codex without `mcp-server`, and a
32
+ * Claude Code that rejects `--permission-prompt-tool`. Both mean the same thing
33
+ * to a person watching the thread, so they say the same thing.
34
+ *
35
+ * @param {string} [tool] the binary that couldn't ask, when we know it
36
+ */
37
+ export function ungatedRunNotice(tool) {
38
+ const name = typeof tool === "string" ? tool.trim() : "";
39
+ const who = name ? `\`${name}\` on this machine` : "the tool running this";
40
+ return `This run executes without asking first — ${who} can't raise approval requests. A newer version can.`;
41
+ }
42
+
43
+ /**
44
+ * A once-per-RUN latch for that notice.
45
+ *
46
+ * The gate is a property of the transport, decided once when the CLI starts —
47
+ * not of each ask. A run that degrades and then makes forty tool calls owes the
48
+ * room one line, and a run that retries its CLI (a gated iterate runs it more
49
+ * than once) still owes exactly one.
50
+ *
51
+ * Callers await this BEFORE spawning the ungated child, so the room is warned
52
+ * while the run can still be stopped rather than told afterwards what it
53
+ * already did. That is also why the post is bounded: it must go out first, but
54
+ * a stalled network call must not hold a run hostage — past the deadline we
55
+ * stop waiting and let the send land whenever it lands. Failure is swallowed
56
+ * either way; a notice must never take the run down with it.
57
+ *
58
+ * @param {{ post: (body: string) => Promise<unknown>, timeoutMs?: number,
59
+ * log?: { error?: (m: string) => void } }} o
60
+ * @returns {(tool?: string) => Promise<boolean>} true only when the post landed
61
+ */
62
+ export function createUngatedRunNotice({ post, timeoutMs = 10_000, log } = {}) {
63
+ let posted = false;
64
+ return async function noticeUngatedRun(tool) {
65
+ if (posted) return false;
66
+ posted = true;
67
+ let timer = null;
68
+ try {
69
+ const sent = Promise.resolve(post(ungatedRunNotice(tool)));
70
+ // Keep the rejection handled even if the deadline wins the race.
71
+ sent.catch(() => {});
72
+ const deadline = new Promise((resolve) => {
73
+ timer = setTimeout(() => resolve("timeout"), Math.max(1, timeoutMs));
74
+ timer?.unref?.();
75
+ });
76
+ const outcome = await Promise.race([sent.then(() => "sent"), deadline]);
77
+ if (outcome === "timeout") {
78
+ log?.error?.("ungated-run notice: still sending past its deadline; not holding the run");
79
+ return false;
80
+ }
81
+ return true;
82
+ } catch (error) {
83
+ log?.error?.(`ungated-run notice: ${error?.message ?? error}`);
84
+ return false;
85
+ } finally {
86
+ if (timer) clearTimeout(timer);
87
+ }
88
+ };
89
+ }
90
+
91
+ function raceWithAbort(promise, signal) {
92
+ if (!signal) return Promise.resolve(promise);
93
+ if (signal.aborted) return Promise.reject(abortError(signal.reason));
94
+ return new Promise((resolve, reject) => {
95
+ const onAbort = () => reject(abortError(signal.reason));
96
+ signal.addEventListener("abort", onAbort, { once: true });
97
+ Promise.resolve(promise).then(
98
+ (value) => {
99
+ signal.removeEventListener("abort", onAbort);
100
+ resolve(value);
101
+ },
102
+ (error) => {
103
+ signal.removeEventListener("abort", onAbort);
104
+ reject(error);
105
+ },
106
+ );
107
+ });
108
+ }
109
+
110
+ const defaultSleep = (ms) => new Promise((resolve) => setTimeout(resolve, Math.max(0, ms)));
111
+
112
+ /** How long a fail-closed settlement gets before we stop waiting on it. The
113
+ * card is already being rejected on the wire; a hung settlement call must not
114
+ * hold the agent's tool call open on top of it. */
115
+ const SETTLEMENT_TIMEOUT_MS = 10_000;
116
+
117
+ /**
118
+ * Race a callback against BOTH the abort signal and a wall-clock deadline.
119
+ *
120
+ * The deadline check between polls is not enough on its own: these callbacks
121
+ * are network calls (the daemon's MCP fetch has no timeout of its own), so a
122
+ * single stalled request could sit past the deadline forever and the ask would
123
+ * neither resolve nor fail. Bounding each call is what makes the deadline real.
124
+ */
125
+ function withDeadline(promise, { signal, deadlineAt, now, label }) {
126
+ const remaining = deadlineAt - now();
127
+ if (remaining <= 0) return Promise.reject(abortError("timeout"));
128
+ let timer = null;
129
+ const timeout = new Promise((_resolve, reject) => {
130
+ timer = setTimeout(() => reject(abortError("timeout")), remaining);
131
+ timer?.unref?.();
132
+ });
133
+ return Promise.race([raceWithAbort(promise, signal), timeout]).finally(() => {
134
+ if (timer) clearTimeout(timer);
135
+ void label;
136
+ });
137
+ }
138
+
139
+ /**
140
+ * Raise one permission card and block until a human decides it.
141
+ *
142
+ * Never throws: an unresolvable ask is a rejection, because the alternative —
143
+ * letting the tool call through — is the one outcome that cannot be undone.
144
+ *
145
+ * @param {{
146
+ * request: object,
147
+ * requestPermission: (request: object, context: object) => Promise<unknown>,
148
+ * getPermissionDecision: (handle: unknown, context: object) => Promise<unknown>,
149
+ * signal?: AbortSignal,
150
+ * timeoutMs?: number,
151
+ * deadlineAt?: number,
152
+ * pollIntervalMs?: number,
153
+ * settlementTimeoutMs?: number,
154
+ * sleep?: (ms: number) => Promise<void>,
155
+ * now?: () => number,
156
+ * log?: { error?: (message: string) => void },
157
+ * label?: string,
158
+ * }} options
159
+ * @returns {Promise<{ reply: "once"|"always"|"reject", outcome: string, handle: unknown }>}
160
+ */
161
+ export async function resolveHilosPermissionReply({
162
+ request,
163
+ requestPermission,
164
+ getPermissionDecision,
165
+ signal,
166
+ timeoutMs = DEFAULT_PERMISSION_TIMEOUT_MS,
167
+ deadlineAt,
168
+ pollIntervalMs = DEFAULT_PERMISSION_POLL_MS,
169
+ settlementTimeoutMs = SETTLEMENT_TIMEOUT_MS,
170
+ sleep = defaultSleep,
171
+ now = () => Date.now(),
172
+ log,
173
+ label = "permission",
174
+ }) {
175
+ if (typeof requestPermission !== "function" || typeof getPermissionDecision !== "function") {
176
+ throw new Error("permission gate requires requestPermission and getPermissionDecision");
177
+ }
178
+ const deadline = deadlineAt ?? now() + Math.max(1, timeoutMs || DEFAULT_PERMISSION_TIMEOUT_MS);
179
+ let handle = null;
180
+ let reply = "reject";
181
+ let outcome = "transport-error";
182
+ // 0813 — the envelope that actually SETTLED this ask. A workspace rule can
183
+ // refuse before any human sees the card, and its reason is the one thing a
184
+ // standing rule can hand the model that a one-off rejection cannot. Only the
185
+ // seams that carry a message use it; the rest ignore it.
186
+ let settled = null;
187
+ try {
188
+ if (signal?.aborted) throw abortError(signal.reason);
189
+ handle = await withDeadline(requestPermission(request, { signal, deadlineAt: deadline }), {
190
+ signal,
191
+ deadlineAt: deadline,
192
+ now,
193
+ label: "request",
194
+ });
195
+ let decision = handle;
196
+ settled = decision;
197
+ let mapped = mapOpenCodePermissionDecision(decision);
198
+ while (!mapped) {
199
+ if (now() >= deadline) throw abortError("timeout");
200
+ decision = await withDeadline(
201
+ getPermissionDecision(handle, { request, signal, deadlineAt: deadline }),
202
+ { signal, deadlineAt: deadline, now, label: "poll" },
203
+ );
204
+ settled = decision;
205
+ mapped = mapOpenCodePermissionDecision(decision);
206
+ if (mapped) break;
207
+ const waitMs = Math.min(Math.max(1, pollIntervalMs), Math.max(1, deadline - now()));
208
+ await raceWithAbort(sleep(waitMs), signal);
209
+ }
210
+ reply = mapped;
211
+ outcome = "decided";
212
+ } catch (error) {
213
+ reply = "reject";
214
+ outcome = error?.message === "timeout" ? "timeout" : signal?.aborted ? "aborted" : "transport-error";
215
+ log?.error?.(`${label} decision: ${error?.message ?? error}`);
216
+ // The agent is about to be failed closed. Settle the durable card too, on a
217
+ // bounded one-shot call with NO run signal — the run's signal is necessarily
218
+ // aborted on this path and must not strand a card visibly pending.
219
+ if (handle) {
220
+ // Best-effort, on its OWN short clock: the run deadline may already be
221
+ // blown (that is often why we are here), and a settlement call that
222
+ // stalls must not hold the agent's tool call open behind a rejection it
223
+ // has already earned.
224
+ try {
225
+ await withDeadline(
226
+ getPermissionDecision(handle, { request, deadlineAt: deadline, failClosed: true }),
227
+ { deadlineAt: now() + Math.max(1, settlementTimeoutMs), now, label: "settlement" },
228
+ );
229
+ } catch (settlementError) {
230
+ log?.error?.(`${label} settlement: ${settlementError?.message ?? settlementError}`);
231
+ }
232
+ }
233
+ }
234
+ return {
235
+ reply,
236
+ outcome,
237
+ handle,
238
+ byPolicy: isPolicyDecision(settled),
239
+ policyReason: policyReasonFrom(settled),
240
+ };
241
+ }
242
+
243
+ /**
244
+ * Was a workspace RULE what settled this, rather than a person?
245
+ *
246
+ * Kept separate from the reason on purpose. The reason is optional — a rule may
247
+ * carry none — so deriving "policy" from "has a reason" would quietly send a
248
+ * reasonless rule's refusal back as "a person declined this", which is exactly
249
+ * the false statement this pair exists to prevent. Provenance is the fact; the
250
+ * reason is a bonus.
251
+ */
252
+ export function isPolicyDecision(envelope) {
253
+ if (!envelope || typeof envelope !== "object") return false;
254
+ return (envelope.decisionSource ?? envelope.decision_source) === "workspace_policy";
255
+ }
256
+
257
+ /**
258
+ * The workspace rule's own words, when a rule is what refused this and it
259
+ * bothered to say why.
260
+ *
261
+ * Deliberately narrow: only a `workspace_policy` decision source yields one, so
262
+ * a human's rejection can never be dressed up as policy, and a server that
263
+ * sends nothing yields null rather than a guess.
264
+ */
265
+ export function policyReasonFrom(envelope) {
266
+ if (!isPolicyDecision(envelope)) return null;
267
+ const reason = envelope.policyReason ?? envelope.policy_reason;
268
+ return typeof reason === "string" && reason.trim() ? reason.trim() : null;
269
+ }