hilos-agent 0.11.13 → 0.11.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -138,10 +138,23 @@ export function defaultProcessInstanceIdentity(
138
138
  }
139
139
  }
140
140
 
141
+ // A directory fsync only hardens the rename/link against power loss; it never
142
+ // decides who owns the seat. Windows refuses FlushFileBuffers on a directory
143
+ // handle (EPERM), and some filesystems answer EINVAL or EBADF. Before 1330 that
144
+ // throw was caught by acquireLock's publish step and reported as "another
145
+ // local daemon already owns this agent", so every Windows daemon refused to
146
+ // start against an empty lock directory.
141
147
  function syncDirectory(fs, path) {
142
- const fd = fs.openSync(path, "r");
148
+ let fd;
149
+ try {
150
+ fd = fs.openSync(path, "r");
151
+ } catch {
152
+ return;
153
+ }
143
154
  try {
144
155
  fs.fsyncSync(fd);
156
+ } catch {
157
+ /* durability nicety only; the link/unlink already happened */
145
158
  } finally {
146
159
  fs.closeSync(fd);
147
160
  }
@@ -178,6 +191,10 @@ export function createIterateClaimRecoveryStore({
178
191
  const processInstanceIdentity = getProcessInstanceIdentity(pid);
179
192
  let lockHeld = false;
180
193
  let heldElectionPath = null;
194
+ // Why the last acquireLock() returned false: "contended" when a live daemon
195
+ // holds the seat, otherwise the filesystem error. run.mjs turns this into
196
+ // an honest startup message (1330).
197
+ let lastAcquireFailure = null;
181
198
 
182
199
  function ensureScope() {
183
200
  fs.mkdirSync(scopeDir, { recursive: true, mode: 0o700 });
@@ -360,9 +377,11 @@ export function createIterateClaimRecoveryStore({
360
377
  acquireLock() {
361
378
  if (lockHeld) return true;
362
379
  if (!ownerKey || !processInstanceIdentity) return false;
380
+ lastAcquireFailure = null;
363
381
  try {
364
382
  ensureScope();
365
383
  } catch (error) {
384
+ lastAcquireFailure = `lock directory unavailable (${scopeDir}): ${error?.message || error}`;
366
385
  log?.error?.(`iterate claim scope unavailable: ${error?.message || error}`);
367
386
  return false;
368
387
  }
@@ -392,12 +411,14 @@ export function createIterateClaimRecoveryStore({
392
411
  }
393
412
  }
394
413
  removeOwnCandidate();
414
+ lastAcquireFailure = `lock file not written (${scopeDir}): ${error?.message || error}`;
395
415
  log?.error?.(`iterate claim scope not locked: ${error?.message || error}`);
396
416
  return false;
397
417
  }
398
418
 
399
419
  if (!acquireElection()) {
400
420
  removeOwnCandidate();
421
+ lastAcquireFailure = "contended";
401
422
  return false;
402
423
  }
403
424
 
@@ -408,6 +429,7 @@ export function createIterateClaimRecoveryStore({
408
429
  .filter((name) => name.startsWith(".owner.") && name.endsWith(".lock"));
409
430
  } catch (error) {
410
431
  removeOwnPublishedLock();
432
+ lastAcquireFailure = `lock directory not readable (${scopeDir}): ${error?.message || error}`;
411
433
  log?.error?.(`iterate claim scope not inspected: ${error?.message || error}`);
412
434
  return false;
413
435
  }
@@ -440,6 +462,7 @@ export function createIterateClaimRecoveryStore({
440
462
 
441
463
  if (!ownSeen || liveContender) {
442
464
  removeOwnPublishedLock();
465
+ lastAcquireFailure = "contended";
443
466
  return false;
444
467
  }
445
468
 
@@ -460,6 +483,11 @@ export function createIterateClaimRecoveryStore({
460
483
  return true;
461
484
  },
462
485
 
486
+ /** Why the last acquireLock() failed: "contended" or a filesystem reason. */
487
+ acquireFailure() {
488
+ return lastAcquireFailure;
489
+ },
490
+
463
491
  releaseLock() {
464
492
  if (!ownsPublishedLock()) return false;
465
493
  if (!removeOwnPublishedLock()) {
package/src/run.mjs CHANGED
@@ -15,6 +15,7 @@ import { startWake, createWakeGate } from "./wake.mjs";
15
15
  import { scanReplyBridge, handleReplyBridgeJob } from "./reply-bridge.mjs";
16
16
  import { detectVendor, fastChatCmd, webCapability } from "./progress-emitter.mjs";
17
17
  import { commandArgv } from "./argv.mjs";
18
+ import { resolveCommand } from "./cli.mjs";
18
19
  import {
19
20
  createIterateClaimRecoveryStore,
20
21
  reconcileDaemonIterateClaims,
@@ -22,6 +23,24 @@ import {
22
23
 
23
24
  const MAX_IMMEDIATE_SEQUENCE_PAGES = 8;
24
25
 
26
+ // 1290: the hilos.sh contract. Missing-tool paths exist only for injected tests;
27
+ // tools/list remains the token's prompt inventory and the recall scope check.
28
+ const HILOS_DAEMON_TOOLS = Object.freeze({
29
+ editMessage: true, postProgress: true, runs: true, review: true,
30
+ recall: true, agentIntent: true, prActions: true, runtimePermissions: true,
31
+ uploadTranscript: true, hepEvents: true, runInputs: true, react: true,
32
+ decisionWaitMs: 20_000,
33
+ });
34
+
35
+ /** Where to get the vendor CLI a daemon command names (1330). */
36
+ function installHint(vendor) {
37
+ if (vendor === "cursor") return "the Cursor CLI is separate from the Cursor app (macOS/Linux: curl https://cursor.com/install -fsS | bash; Windows: irm https://cursor.com/install -useb | iex)";
38
+ if (vendor === "claude_code") return "npm install -g @anthropic-ai/claude-code";
39
+ if (vendor === "codex") return "npm install -g @openai/codex";
40
+ if (vendor === "opencode") return "https://opencode.ai";
41
+ return "";
42
+ }
43
+
25
44
  /**
26
45
  * The poll loop. Embeddable: pass a `signal` to stop it cleanly (interrupts the
27
46
  * inter-poll sleep, cancels the active job, then resolves) and an `onEvent`
@@ -30,6 +49,7 @@ const MAX_IMMEDIATE_SEQUENCE_PAGES = 8;
30
49
  *
31
50
  * @param {any} cfg
32
51
  * @param {{
52
+ * testServerFixture?: any,
33
53
  * handler?: (ctx: any, cfg: any, deps?: any, opts?: any) => Promise<any>,
34
54
  * log?: any,
35
55
  * signal?: AbortSignal,
@@ -45,6 +65,7 @@ export async function run(
45
65
  cfg,
46
66
  {
47
67
  handler = handleTask,
68
+ testServerFixture,
48
69
  log = console,
49
70
  signal,
50
71
  onEvent,
@@ -143,8 +164,7 @@ export async function run(
143
164
  // Throw for the same reason as the token guard above — embed-safe, CLI-equal.
144
165
  throw new Error("This agent has no display name / handle — set one in hilos, then reconnect.");
145
166
  }
146
- // Initialized only after the server advertises the complete recovery
147
- // protocol below. Older servers keep their pre-recovery Iterate behavior.
167
+ // Initialized after discovery so the recovery lock is scoped to this agent.
148
168
  let iterateClaimRecoveryStore = null;
149
169
  let queue = null;
150
170
  let wake = null;
@@ -178,6 +198,19 @@ export async function run(
178
198
  enabled: cfg.webSearch !== false,
179
199
  args: commandArgv(cfg.codingCmd).slice(1),
180
200
  });
201
+ // 1330 — say at startup, not at the first mention, when the CLI this daemon
202
+ // is configured to drive is not installed. A person who connected through an
203
+ // IDE agent read "Hilo is connected and online" and then got "(my chat
204
+ // command `cursor-agent …` isn't installed or on PATH.)" in the room.
205
+ for (const [role, command] of [["chat", chatCommand], ["code", cfg.codingCmd]]) {
206
+ const bin = commandArgv(command)[0];
207
+ if (!bin || resolveCommand(bin)) continue;
208
+ const hint = installHint(detectVendor(command));
209
+ log.error(
210
+ `${role} command not found: \`${bin}\` is not installed or not on PATH${hint ? ` — ${hint}` : ""}. Mentions will fail until it is.`,
211
+ );
212
+ emit({ type: "status", text: `${role} command missing: ${bin}` });
213
+ }
181
214
  log.log(
182
215
  chatVendor === codingVendor && web.status === codingWeb.status
183
216
  ? `web: ${web.status} — ${web.source}`
@@ -186,29 +219,20 @@ export async function run(
186
219
 
187
220
  const since = cfg.backfill ? 0 : Date.now();
188
221
  // listTools (schemas included) is the primary read; a client exposing only
189
- // listToolNames (an embedder's minimal mock, or an older embedded client)
190
- // still names the tools — it just never advertises arg-level capabilities.
222
+ // listToolNames is a minimal embedded test client. Schemas no longer
223
+ // control production behavior.
191
224
  const toolList = typeof listTools === "function" ? await listTools({ signal: runSignal }) : [];
192
225
  const toolNames = toolList.length
193
226
  ? toolList.map((t) => t.name)
194
227
  : typeof listToolNames === "function"
195
228
  ? await listToolNames({ signal: runSignal })
196
229
  : [];
197
- // Arg-level capability sniffing (0866): a long-poll-capable server declares
198
- // `waitMs` in the tool's inputSchema. Absent (older server) → the daemon
199
- // falls back to plain interval polling, exactly as before.
200
- const toolArgProps = (name) =>
201
- toolList.find((t) => t.name === name)?.inputSchema?.properties ?? {};
202
- const serverMentionWait = "waitMs" in toolArgProps("list_mentions");
203
- const serverMentionSequence = "sinceMentionSeq" in toolArgProps("list_mentions");
204
- const serverMentionSequenceBootstrap = "initializeMentionSeq" in toolArgProps("list_mentions");
205
- const serverDecisionWait = "waitMs" in toolArgProps("get_permission_decision");
206
- const updateRunArgs = toolArgProps("update_run");
207
- const serverIterateClaimRecovery =
208
- "claimId" in updateRunArgs &&
209
- "confirmIterateClaim" in updateRunArgs &&
210
- "releaseIterateClaim" in updateRunArgs;
211
- if (me.agentId && serverIterateClaimRecovery) {
230
+ // Schema omissions are fixtures, never version negotiation with hilos.sh.
231
+ const mentionWaitEnabled = testServerFixture?.mentionWait ?? true;
232
+ const mentionSequenceEnabled = testServerFixture?.mentionSequence ?? true;
233
+ const mentionSequenceBootstrapEnabled = testServerFixture?.mentionSequenceBootstrap ?? true;
234
+ const iterateClaimRecoveryEnabled = testServerFixture?.iterateClaimRecovery ?? true;
235
+ if (me.agentId && iterateClaimRecoveryEnabled) {
212
236
  iterateClaimRecoveryStore = createIterateClaimRecoveryStore({
213
237
  agentId: me.agentId,
214
238
  url: cfg.url,
@@ -217,7 +241,12 @@ export async function run(
217
241
  log,
218
242
  });
219
243
  if (!iterateClaimRecoveryStore.acquireLock()) {
220
- throw new Error("Another local daemon already owns this agent and server connection.");
244
+ const why = iterateClaimRecoveryStore.acquireFailure?.();
245
+ throw new Error(
246
+ why && why !== "contended"
247
+ ? `The local run lock could not be taken: ${why}. Make that directory writable and start again.`
248
+ : "Another local daemon already owns this agent and server connection.",
249
+ );
221
250
  }
222
251
  reconcileClaimsOnce = async (context) => {
223
252
  try {
@@ -239,64 +268,13 @@ export async function run(
239
268
  }
240
269
  };
241
270
  }
242
- // Capabilities of THIS server, so the handler degrades gracefully on older
243
- // deploys (e.g. no edit_message → no live heartbeat, rather than erroring).
244
- const caps = {
245
- editMessage: toolNames.includes("edit_message"),
246
- postProgress: toolNames.includes("post_progress"),
247
- // The RUNS entity (0279/0280): when absent (older server) the handler skips
248
- // start_run/update_run silently and behaves exactly as before.
249
- runs: toolNames.includes("start_run"),
250
- // Cross-agent REVIEW execution (0288): the reviewer needs to read a PR's diff
251
- // AND post an advisory review. Absent on older servers → review-requests fall
252
- // through to today's chat/code behavior, exactly as before.
253
- review: toolNames.includes("post_review") && toolNames.includes("get_pr_diff"),
254
- // Agent memory (0297): when present the handler best-effort recalls team
255
- // learnings before each coding task and prepends them to the prompt so the
256
- // coding agent has the workspace's conventions and gotchas from the start.
257
- // Absent on older servers → silently skipped, no change in behavior.
258
- recall: toolNames.includes("recall"),
259
- // Semantic local-folder intent (0515): the server guarantees a forced-tool
260
- // ask/ship/deploy decision. Older servers fall back to the local router.
261
- agentIntent: toolNames.includes("classify_agent_intent"),
262
- // Relaying a human's merge/close instruction (0701/0704). Both tools are
263
- // required; on an older server the daemon simply never relays and behaves
264
- // exactly as before. The authority is the server's either way — the daemon
265
- // only passes along the id of the message that asked.
266
- prActions: toolNames.includes("merge_pr") && toolNames.includes("close_pr"),
267
- // Harness-enforced runtime permission bridge (0593). Both tools are
268
- // required: a request without a durable reply path must fail closed.
269
- runtimePermissions:
270
- toolNames.includes("request_permission") &&
271
- toolNames.includes("get_permission_decision"),
272
- // Opt-in run transcripts (0792). Two gates, and BOTH must say yes: the
273
- // server has to offer the tool, and the operator has to have turned
274
- // `uploadTranscripts` on. Older servers simply never see the call.
275
- uploadTranscript: toolNames.includes("upload_run_transcript"),
276
- // Runtime-neutral replay exhaust (1148). The daemon only projects local
277
- // coding streams when this exact ingest boundary exists; older servers
278
- // keep the current progress/transcript behavior byte-for-byte.
279
- hepEvents: toolNames.includes("append_run_events"),
280
- // Room directions for a running job (1181/1182). When the server can take
281
- // the receipt, the daemon declares `next_turn` at start_run, reads
282
- // `pendingInputs` off its heartbeat, and runs one more turn with them.
283
- // Older servers never list a direction, and the daemon never declares.
284
- runInputs: toolNames.includes("acknowledge_run_input"),
285
- // Emoji reactions (0860): the lightest answer to an untagged DM message
286
- // that needs no words. Absent on older servers → the agent stays quiet
287
- // instead, exactly as if it had no hand to wave.
288
- react: toolNames.includes("add_reaction"),
289
- // Long-poll hold for the permission gate (0866): each decision read blocks
290
- // server-side until a human decides, so an Allow reaches the paused tool in
291
- // under a second instead of on the gate's next poll. 0 on older servers.
292
- decisionWaitMs: serverDecisionWait ? 20000 : 0,
293
- };
271
+ const testCapabilities = testServerFixture?.capabilities ?? HILOS_DAEMON_TOOLS;
272
+ // 0513 is an authorization boundary, not an older-server fallback: a
273
+ // channel-bound token is never advertised workspace memory.
274
+ const canRecall = toolNames.includes("recall");
294
275
 
295
- // The wake doorbell (0824): capability-checked like every optional surface.
296
- // On an older server the tool is absent and the daemon polls exactly as
297
- // before; with it, a new message in the workspace rings a content-free
298
- // Realtime broadcast that cuts mention pickup from pollMs to instant.
299
- if (toolNames.includes("get_wake_channel")) {
276
+ // The doorbell is always registered. Polling still survives a network failure.
277
+ if (testServerFixture?.wake !== false) {
300
278
  try {
301
279
  const chan = await tool("get_wake_channel");
302
280
  if (chan?.url && chan?.topic) {
@@ -327,12 +305,11 @@ export async function run(
327
305
 
328
306
  // Register local folders (0324/0325): a folder-mode daemon announces each
329
307
  // channel→folder mapping to the server so the channel shows a folder chip
330
- // without any manual step. Capability-gated on the server advertising
331
- // link_folder (older deploys skip silently, exactly like recall). Best-effort:
308
+ // without any manual step. Best-effort:
332
309
  // each call is wrapped so a failure logs one line and NEVER blocks startup.
333
310
  // Done ONCE here at startup — config live-reload (reloadConfig, below) does NOT
334
311
  // re-register, so a folders entry added while running needs a restart to link.
335
- if (toolNames.includes("link_folder") && cfg.folders && Object.keys(cfg.folders).length) {
312
+ if (testServerFixture?.linkFolder !== false && cfg.folders && Object.keys(cfg.folders).length) {
336
313
  for (const [channelId, path] of Object.entries(cfg.folders)) {
337
314
  if (!channelId || !path) continue;
338
315
  const title = basename(path);
@@ -377,9 +354,9 @@ export async function run(
377
354
  if (persistedCursor) log.log(`resuming mention cursor from ${persistedCursor}`);
378
355
  const cursor = {
379
356
  value: persistedCursor ?? (since ? new Date(since).toISOString() : null),
380
- seq: serverMentionSequence ? persistedCursorState?.mentionSeq ?? null : null,
357
+ seq: mentionSequenceEnabled ? persistedCursorState?.mentionSeq ?? null : null,
381
358
  };
382
- let mentionSequenceActive = serverMentionSequence && cursor.seq != null;
359
+ let mentionSequenceActive = mentionSequenceEnabled && cursor.seq != null;
383
360
  // Fetch progress is deliberately volatile. It moves through pages as soon as
384
361
  // the server scans them (so an active job does not hot-poll one seen row),
385
362
  // while `cursor` moves only after queue/recovery settlement. A restart always
@@ -404,7 +381,7 @@ export async function run(
404
381
  // Deploy, dispatch, ambient, and DM projections are useful intake, but are
405
382
  // not rows in the explicit mention scan. They must never advance that
406
383
  // durable high-watermark.
407
- if (serverMentionSequence && mentionSeq == null) return;
384
+ if (mentionSequenceEnabled && mentionSeq == null) return;
408
385
  const key = batchKey(timestamp, mentionSeq);
409
386
  let batch = cursorBatches.get(key);
410
387
  if (!batch) {
@@ -573,7 +550,8 @@ export async function run(
573
550
  channelId,
574
551
  tool: taskTool,
575
552
  me,
576
- caps,
553
+ testCapabilities,
554
+ canRecall,
577
555
  iterateClaimRecoveryStore,
578
556
  },
579
557
  liveCfg,
@@ -700,7 +678,7 @@ export async function run(
700
678
  // moment a mention lands — pickup latency stops being pollMs. The server
701
679
  // caps a hold at 25s; asking for more just gets 25s.
702
680
  const mentionWaitMs = () =>
703
- serverMentionWait && liveCfg.longPollMs > 0 ? Math.floor(liveCfg.longPollMs) : 0;
681
+ mentionWaitEnabled && liveCfg.longPollMs > 0 ? Math.floor(liveCfg.longPollMs) : 0;
704
682
 
705
683
  async function passViaMentions(immediatePage = 1) {
706
684
  const waitMs = mentionWaitMs();
@@ -709,7 +687,7 @@ export async function run(
709
687
  ...(mentionSequenceActive && fetchCursor.seq != null
710
688
  ? { sinceMentionSeq: fetchCursor.seq }
711
689
  : {}),
712
- ...(serverMentionSequenceBootstrap && !mentionSequenceActive
690
+ ...(mentionSequenceBootstrapEnabled && !mentionSequenceActive
713
691
  ? { initializeMentionSeq: true }
714
692
  : {}),
715
693
  ...(cfg.channelId ? { channelId: cfg.channelId } : {}),
@@ -720,7 +698,7 @@ export async function run(
720
698
  // 1247 — the server may send a rendered roster per channel alongside the
721
699
  // mentions. Stamp it on the row so the CLI prompt can carry it, and cache
722
700
  // it for the reply bridge, which polls bound threads instead of mentions.
723
- // Absent on an older server; then nothing changes.
701
+ // Roster enrichment is optional; task delivery does not depend on it.
724
702
  const rosters = out?.rosters && typeof out.rosters === "object" ? out.rosters : null;
725
703
  if (rosters) {
726
704
  for (const m of mentions) {
@@ -768,7 +746,7 @@ export async function run(
768
746
  settledEmptySequencePage = true;
769
747
  }
770
748
  } else if (
771
- serverMentionSequenceBootstrap &&
749
+ mentionSequenceBootstrapEnabled &&
772
750
  !mentionSequenceActive &&
773
751
  Number.isSafeInteger(announcedSequence) &&
774
752
  announcedSequence >= 0 &&