hilos-agent 0.11.13 → 0.11.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/run.mjs CHANGED
@@ -22,6 +22,15 @@ import {
22
22
 
23
23
  const MAX_IMMEDIATE_SEQUENCE_PAGES = 8;
24
24
 
25
+ // 1290: the hilos.sh contract. Missing-tool paths exist only for injected tests;
26
+ // tools/list remains the token's prompt inventory and the recall scope check.
27
+ const HILOS_DAEMON_TOOLS = Object.freeze({
28
+ editMessage: true, postProgress: true, runs: true, review: true,
29
+ recall: true, agentIntent: true, prActions: true, runtimePermissions: true,
30
+ uploadTranscript: true, hepEvents: true, runInputs: true, react: true,
31
+ decisionWaitMs: 20_000,
32
+ });
33
+
25
34
  /**
26
35
  * The poll loop. Embeddable: pass a `signal` to stop it cleanly (interrupts the
27
36
  * inter-poll sleep, cancels the active job, then resolves) and an `onEvent`
@@ -30,6 +39,7 @@ const MAX_IMMEDIATE_SEQUENCE_PAGES = 8;
30
39
  *
31
40
  * @param {any} cfg
32
41
  * @param {{
42
+ * testServerFixture?: any,
33
43
  * handler?: (ctx: any, cfg: any, deps?: any, opts?: any) => Promise<any>,
34
44
  * log?: any,
35
45
  * signal?: AbortSignal,
@@ -45,6 +55,7 @@ export async function run(
45
55
  cfg,
46
56
  {
47
57
  handler = handleTask,
58
+ testServerFixture,
48
59
  log = console,
49
60
  signal,
50
61
  onEvent,
@@ -143,8 +154,7 @@ export async function run(
143
154
  // Throw for the same reason as the token guard above — embed-safe, CLI-equal.
144
155
  throw new Error("This agent has no display name / handle — set one in hilos, then reconnect.");
145
156
  }
146
- // Initialized only after the server advertises the complete recovery
147
- // protocol below. Older servers keep their pre-recovery Iterate behavior.
157
+ // Initialized after discovery so the recovery lock is scoped to this agent.
148
158
  let iterateClaimRecoveryStore = null;
149
159
  let queue = null;
150
160
  let wake = null;
@@ -186,29 +196,20 @@ export async function run(
186
196
 
187
197
  const since = cfg.backfill ? 0 : Date.now();
188
198
  // listTools (schemas included) is the primary read; a client exposing only
189
- // listToolNames (an embedder's minimal mock, or an older embedded client)
190
- // still names the tools — it just never advertises arg-level capabilities.
199
+ // listToolNames is a minimal embedded test client. Schemas no longer
200
+ // control production behavior.
191
201
  const toolList = typeof listTools === "function" ? await listTools({ signal: runSignal }) : [];
192
202
  const toolNames = toolList.length
193
203
  ? toolList.map((t) => t.name)
194
204
  : typeof listToolNames === "function"
195
205
  ? await listToolNames({ signal: runSignal })
196
206
  : [];
197
- // Arg-level capability sniffing (0866): a long-poll-capable server declares
198
- // `waitMs` in the tool's inputSchema. Absent (older server) → the daemon
199
- // falls back to plain interval polling, exactly as before.
200
- const toolArgProps = (name) =>
201
- toolList.find((t) => t.name === name)?.inputSchema?.properties ?? {};
202
- const serverMentionWait = "waitMs" in toolArgProps("list_mentions");
203
- const serverMentionSequence = "sinceMentionSeq" in toolArgProps("list_mentions");
204
- const serverMentionSequenceBootstrap = "initializeMentionSeq" in toolArgProps("list_mentions");
205
- const serverDecisionWait = "waitMs" in toolArgProps("get_permission_decision");
206
- const updateRunArgs = toolArgProps("update_run");
207
- const serverIterateClaimRecovery =
208
- "claimId" in updateRunArgs &&
209
- "confirmIterateClaim" in updateRunArgs &&
210
- "releaseIterateClaim" in updateRunArgs;
211
- if (me.agentId && serverIterateClaimRecovery) {
207
+ // Schema omissions are fixtures, never version negotiation with hilos.sh.
208
+ const mentionWaitEnabled = testServerFixture?.mentionWait ?? true;
209
+ const mentionSequenceEnabled = testServerFixture?.mentionSequence ?? true;
210
+ const mentionSequenceBootstrapEnabled = testServerFixture?.mentionSequenceBootstrap ?? true;
211
+ const iterateClaimRecoveryEnabled = testServerFixture?.iterateClaimRecovery ?? true;
212
+ if (me.agentId && iterateClaimRecoveryEnabled) {
212
213
  iterateClaimRecoveryStore = createIterateClaimRecoveryStore({
213
214
  agentId: me.agentId,
214
215
  url: cfg.url,
@@ -239,64 +240,13 @@ export async function run(
239
240
  }
240
241
  };
241
242
  }
242
- // Capabilities of THIS server, so the handler degrades gracefully on older
243
- // deploys (e.g. no edit_message → no live heartbeat, rather than erroring).
244
- const caps = {
245
- editMessage: toolNames.includes("edit_message"),
246
- postProgress: toolNames.includes("post_progress"),
247
- // The RUNS entity (0279/0280): when absent (older server) the handler skips
248
- // start_run/update_run silently and behaves exactly as before.
249
- runs: toolNames.includes("start_run"),
250
- // Cross-agent REVIEW execution (0288): the reviewer needs to read a PR's diff
251
- // AND post an advisory review. Absent on older servers → review-requests fall
252
- // through to today's chat/code behavior, exactly as before.
253
- review: toolNames.includes("post_review") && toolNames.includes("get_pr_diff"),
254
- // Agent memory (0297): when present the handler best-effort recalls team
255
- // learnings before each coding task and prepends them to the prompt so the
256
- // coding agent has the workspace's conventions and gotchas from the start.
257
- // Absent on older servers → silently skipped, no change in behavior.
258
- recall: toolNames.includes("recall"),
259
- // Semantic local-folder intent (0515): the server guarantees a forced-tool
260
- // ask/ship/deploy decision. Older servers fall back to the local router.
261
- agentIntent: toolNames.includes("classify_agent_intent"),
262
- // Relaying a human's merge/close instruction (0701/0704). Both tools are
263
- // required; on an older server the daemon simply never relays and behaves
264
- // exactly as before. The authority is the server's either way — the daemon
265
- // only passes along the id of the message that asked.
266
- prActions: toolNames.includes("merge_pr") && toolNames.includes("close_pr"),
267
- // Harness-enforced runtime permission bridge (0593). Both tools are
268
- // required: a request without a durable reply path must fail closed.
269
- runtimePermissions:
270
- toolNames.includes("request_permission") &&
271
- toolNames.includes("get_permission_decision"),
272
- // Opt-in run transcripts (0792). Two gates, and BOTH must say yes: the
273
- // server has to offer the tool, and the operator has to have turned
274
- // `uploadTranscripts` on. Older servers simply never see the call.
275
- uploadTranscript: toolNames.includes("upload_run_transcript"),
276
- // Runtime-neutral replay exhaust (1148). The daemon only projects local
277
- // coding streams when this exact ingest boundary exists; older servers
278
- // keep the current progress/transcript behavior byte-for-byte.
279
- hepEvents: toolNames.includes("append_run_events"),
280
- // Room directions for a running job (1181/1182). When the server can take
281
- // the receipt, the daemon declares `next_turn` at start_run, reads
282
- // `pendingInputs` off its heartbeat, and runs one more turn with them.
283
- // Older servers never list a direction, and the daemon never declares.
284
- runInputs: toolNames.includes("acknowledge_run_input"),
285
- // Emoji reactions (0860): the lightest answer to an untagged DM message
286
- // that needs no words. Absent on older servers → the agent stays quiet
287
- // instead, exactly as if it had no hand to wave.
288
- react: toolNames.includes("add_reaction"),
289
- // Long-poll hold for the permission gate (0866): each decision read blocks
290
- // server-side until a human decides, so an Allow reaches the paused tool in
291
- // under a second instead of on the gate's next poll. 0 on older servers.
292
- decisionWaitMs: serverDecisionWait ? 20000 : 0,
293
- };
243
+ const testCapabilities = testServerFixture?.capabilities ?? HILOS_DAEMON_TOOLS;
244
+ // 0513 is an authorization boundary, not an older-server fallback: a
245
+ // channel-bound token is never advertised workspace memory.
246
+ const canRecall = toolNames.includes("recall");
294
247
 
295
- // The wake doorbell (0824): capability-checked like every optional surface.
296
- // On an older server the tool is absent and the daemon polls exactly as
297
- // before; with it, a new message in the workspace rings a content-free
298
- // Realtime broadcast that cuts mention pickup from pollMs to instant.
299
- if (toolNames.includes("get_wake_channel")) {
248
+ // The doorbell is always registered. Polling still survives a network failure.
249
+ if (testServerFixture?.wake !== false) {
300
250
  try {
301
251
  const chan = await tool("get_wake_channel");
302
252
  if (chan?.url && chan?.topic) {
@@ -327,12 +277,11 @@ export async function run(
327
277
 
328
278
  // Register local folders (0324/0325): a folder-mode daemon announces each
329
279
  // channel→folder mapping to the server so the channel shows a folder chip
330
- // without any manual step. Capability-gated on the server advertising
331
- // link_folder (older deploys skip silently, exactly like recall). Best-effort:
280
+ // without any manual step. Best-effort:
332
281
  // each call is wrapped so a failure logs one line and NEVER blocks startup.
333
282
  // Done ONCE here at startup — config live-reload (reloadConfig, below) does NOT
334
283
  // re-register, so a folders entry added while running needs a restart to link.
335
- if (toolNames.includes("link_folder") && cfg.folders && Object.keys(cfg.folders).length) {
284
+ if (testServerFixture?.linkFolder !== false && cfg.folders && Object.keys(cfg.folders).length) {
336
285
  for (const [channelId, path] of Object.entries(cfg.folders)) {
337
286
  if (!channelId || !path) continue;
338
287
  const title = basename(path);
@@ -377,9 +326,9 @@ export async function run(
377
326
  if (persistedCursor) log.log(`resuming mention cursor from ${persistedCursor}`);
378
327
  const cursor = {
379
328
  value: persistedCursor ?? (since ? new Date(since).toISOString() : null),
380
- seq: serverMentionSequence ? persistedCursorState?.mentionSeq ?? null : null,
329
+ seq: mentionSequenceEnabled ? persistedCursorState?.mentionSeq ?? null : null,
381
330
  };
382
- let mentionSequenceActive = serverMentionSequence && cursor.seq != null;
331
+ let mentionSequenceActive = mentionSequenceEnabled && cursor.seq != null;
383
332
  // Fetch progress is deliberately volatile. It moves through pages as soon as
384
333
  // the server scans them (so an active job does not hot-poll one seen row),
385
334
  // while `cursor` moves only after queue/recovery settlement. A restart always
@@ -404,7 +353,7 @@ export async function run(
404
353
  // Deploy, dispatch, ambient, and DM projections are useful intake, but are
405
354
  // not rows in the explicit mention scan. They must never advance that
406
355
  // durable high-watermark.
407
- if (serverMentionSequence && mentionSeq == null) return;
356
+ if (mentionSequenceEnabled && mentionSeq == null) return;
408
357
  const key = batchKey(timestamp, mentionSeq);
409
358
  let batch = cursorBatches.get(key);
410
359
  if (!batch) {
@@ -573,7 +522,8 @@ export async function run(
573
522
  channelId,
574
523
  tool: taskTool,
575
524
  me,
576
- caps,
525
+ testCapabilities,
526
+ canRecall,
577
527
  iterateClaimRecoveryStore,
578
528
  },
579
529
  liveCfg,
@@ -700,7 +650,7 @@ export async function run(
700
650
  // moment a mention lands — pickup latency stops being pollMs. The server
701
651
  // caps a hold at 25s; asking for more just gets 25s.
702
652
  const mentionWaitMs = () =>
703
- serverMentionWait && liveCfg.longPollMs > 0 ? Math.floor(liveCfg.longPollMs) : 0;
653
+ mentionWaitEnabled && liveCfg.longPollMs > 0 ? Math.floor(liveCfg.longPollMs) : 0;
704
654
 
705
655
  async function passViaMentions(immediatePage = 1) {
706
656
  const waitMs = mentionWaitMs();
@@ -709,7 +659,7 @@ export async function run(
709
659
  ...(mentionSequenceActive && fetchCursor.seq != null
710
660
  ? { sinceMentionSeq: fetchCursor.seq }
711
661
  : {}),
712
- ...(serverMentionSequenceBootstrap && !mentionSequenceActive
662
+ ...(mentionSequenceBootstrapEnabled && !mentionSequenceActive
713
663
  ? { initializeMentionSeq: true }
714
664
  : {}),
715
665
  ...(cfg.channelId ? { channelId: cfg.channelId } : {}),
@@ -720,7 +670,7 @@ export async function run(
720
670
  // 1247 — the server may send a rendered roster per channel alongside the
721
671
  // mentions. Stamp it on the row so the CLI prompt can carry it, and cache
722
672
  // it for the reply bridge, which polls bound threads instead of mentions.
723
- // Absent on an older server; then nothing changes.
673
+ // Roster enrichment is optional; task delivery does not depend on it.
724
674
  const rosters = out?.rosters && typeof out.rosters === "object" ? out.rosters : null;
725
675
  if (rosters) {
726
676
  for (const m of mentions) {
@@ -768,7 +718,7 @@ export async function run(
768
718
  settledEmptySequencePage = true;
769
719
  }
770
720
  } else if (
771
- serverMentionSequenceBootstrap &&
721
+ mentionSequenceBootstrapEnabled &&
772
722
  !mentionSequenceActive &&
773
723
  Number.isSafeInteger(announcedSequence) &&
774
724
  announcedSequence >= 0 &&