@staix/agent-hub 0.8.1 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,104 @@
1
+ import { chmodSync, existsSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import type { JournalDelivery } from "./delivery-journal.ts";
4
+
5
+ /**
6
+ * Unplanned-crash recovery (issue #37). While the hub runs it keeps `sessions.json`: each attached peer's session
7
+ * identity (the adapters' recovery metadata: ids and launch options, never message text). A clean stop removes the
8
+ * file, so finding it at start means the previous run died.
9
+ */
10
+ export interface SessionRecord {
11
+ peer: string;
12
+ /** The adapter's recovery metadata: `launch`, and `sessionId` / `threadId` / `sessionFile` where it has one. */
13
+ meta: Record<string, unknown>;
14
+ }
15
+ export interface SessionsFile {
16
+ instanceId: string;
17
+ at: number;
18
+ peers: SessionRecord[];
19
+ }
20
+
21
+ const fileOf = (stateDir: string) => join(stateDir, "sessions.json");
22
+
23
+ export function writeSessions(stateDir: string, s: SessionsFile): void {
24
+ const tmp = `${fileOf(stateDir)}.tmp`;
25
+ writeFileSync(tmp, `${JSON.stringify(s)}\n`, { mode: 0o600 });
26
+ chmodSync(tmp, 0o600);
27
+ renameSync(tmp, fileOf(stateDir));
28
+ }
29
+
30
+ export function readSessions(stateDir: string): SessionsFile | undefined {
31
+ if (!existsSync(fileOf(stateDir))) return undefined;
32
+ try {
33
+ const s = JSON.parse(readFileSync(fileOf(stateDir), "utf8")) as SessionsFile;
34
+ return Array.isArray(s.peers) ? s : undefined;
35
+ } catch {
36
+ return undefined; // cut short by the crash: nothing to resume from
37
+ }
38
+ }
39
+
40
+ /**
41
+ * Only the run that wrote it removes it: a stop of an older instance must not erase a newer run's record. Without an
42
+ * instance id it goes whatever wrote it (a controlled restart, which has its own record of the peers).
43
+ */
44
+ export function removeSessions(stateDir: string, instanceId?: string): void {
45
+ if (instanceId === undefined || readSessions(stateDir)?.instanceId === instanceId) rmSync(fileOf(stateDir), { force: true });
46
+ }
47
+
48
+ const str = (v: unknown) => (typeof v === "string" && v ? v : undefined);
49
+
50
+ /**
51
+ * What can come back after a crash. `resume` holds the start arguments for peers the hub launches itself (Kimi through
52
+ * ACP `session/load`, Pi through its session file, the local worker afresh); the others say what the user has to do.
53
+ */
54
+ export function crashPlan(records: SessionRecord[]): { peer: string; resume?: Record<string, string>; how: string }[] {
55
+ return records.map(({ peer, meta }) => {
56
+ const launch = (meta.launch && typeof meta.launch === "object" ? meta.launch : {}) as Record<string, unknown>;
57
+ const model = str(launch.model);
58
+ switch (peer) {
59
+ case "claude":
60
+ return { peer, how: "claude: the Claude Code plugin reconnects by itself while that session is still open" };
61
+ case "codex":
62
+ return { peer, how: `codex: its app-server died with the hub; run ahub codex again${str(meta.threadId) ? ` (its conversation was thread ${meta.threadId})` : ""}` };
63
+ case "kimi": {
64
+ const sessionId = str(meta.sessionId);
65
+ if (!sessionId) return { peer, how: "kimi: no session id was recorded; start it again with ahub kimi" };
66
+ return { peer, resume: { sessionId, ...(model ? { model } : {}) }, how: `kimi: session ${sessionId} can be loaded again (ACP session/load)` };
67
+ }
68
+ case "pi": {
69
+ const sessionFile = str(meta.sessionFile) ?? str(launch.sessionFile);
70
+ if (!sessionFile) return { peer, how: "pi: no session file was recorded; start it again with ahub pi" };
71
+ // A terminal Pi is run by the CLI that launched it, not by the hub: nothing here could start it again.
72
+ if (str(launch.mode) === "tui") return { peer, how: `pi: it ran in a terminal; start it again with ahub pi --mode tui --session-file ${sessionFile}` };
73
+ const args: Record<string, string> = { sessionFile };
74
+ for (const k of ["mode", "backend", "model"] as const) if (str(launch[k])) args[k] = str(launch[k])!;
75
+ return { peer, resume: args, how: `pi: its session file can be resumed (${sessionFile})` };
76
+ }
77
+ case "local": {
78
+ // `model` is also recorded as the route's fallback, and a model given at start pins it: pass one or the other.
79
+ const route = str(launch.route);
80
+ const args: Record<string, string> = route ? { route } : model ? { model } : {};
81
+ return { peer, resume: args, how: "local: starts again without its history (the worker keeps none across a hub stop)" };
82
+ }
83
+ default:
84
+ return { peer, how: `${peer}: reconnects by itself if its client is still running` };
85
+ }
86
+ });
87
+ }
88
+
89
+ /**
90
+ * The loss notice for one peer: the deliveries to it that the crash left in needs_review, with their senders and the
91
+ * tasks they were about. `taskTitle` must return the public title (`#n [pii]` for a PII task). No message text.
92
+ */
93
+ export function lossNotice(lost: JournalDelivery[], taskTitle: (id: number) => string | undefined): string {
94
+ const lines = lost.map((d) => {
95
+ const from = [...new Set(d.originals.map((e) => e.from))].join(", ");
96
+ const tasks = [...new Set(d.originals.map((e) => Number(e.refs?.task)).filter((n) => Number.isInteger(n) && n > 0))].map((n) => taskTitle(n) ?? `#${n}`);
97
+ return `- delivery ${d.id} from ${from}${tasks.length ? `, about task ${tasks.join(", ")}` : ""}`;
98
+ });
99
+ return [
100
+ "The hub stopped unexpectedly and was started again. These deliveries to you were in flight when it stopped, so whether you acted on them is not known; the console has since marked each one completed, retried or discarded:",
101
+ ...lines,
102
+ "Check your work against them before you go on.",
103
+ ].join("\n");
104
+ }
package/src/hub/daemon.ts CHANGED
@@ -36,11 +36,15 @@ import { Bus } from "./bus.ts";
36
36
  import { DeliveryJournal } from "./delivery-journal.ts";
37
37
  import { startDashboard } from "./ui.ts";
38
38
  import { PROTOCOL, stateDirFor } from "./control-client.ts";
39
- import { newEnvelope, parseMarker, replyParent, sanitize, USER, type Envelope, type PeerId } from "./envelope.ts";
39
+ import { newEnvelope, parseMarker, replyParent, sanitize, USER, type Envelope, type PeerId, type Priority } from "./envelope.ts";
40
40
  import { BasePeer, DEFAULT_WATCHDOG_MS, type PeerAdapter } from "./peers.ts";
41
41
  import { MemoryClient, workerUrl } from "../memory/client.ts";
42
42
  import { VERSION } from "../version.ts";
43
43
  import { projectChain, recallFor } from "../memory/recall.ts";
44
+ import { conflictsOf } from "./conflicts.ts";
45
+ import { crashPlan, lossNotice, readSessions, removeSessions, writeSessions, type SessionsFile } from "./crash.ts";
46
+ import type { JournalDelivery } from "./delivery-journal.ts";
47
+ import { DEFAULT_LIMITS, Limiter, PROJECT_LIMITS, type LimitsConfig } from "./limits.ts";
44
48
  import { changedPaths, repoOf, snapshot, Turns } from "./snapshots.ts";
45
49
  import { archiveRestartSnapshot, readRestartSnapshot, removeRestartSnapshot, restartPath, writeRestartSnapshot, type RecoveryPhase, type RestartPeerSnapshot, type RestartSnapshot } from "./restart.ts";
46
50
 
@@ -58,7 +62,8 @@ export interface HubConfig {
58
62
  omniroute: OmniRouteConfig;
59
63
  pi: { enabled: boolean; auto_start: boolean; cmd: string[]; backend: "auto" | "dgx" | "mlx"; dgx_coding: string; dgx_fast: string; max_steps: number };
60
64
  mlx: Pick<MlxOptions, "provider" | "host" | "runtimeDir" | "modelPath" | "port" | "model" | "sourceModel" | "contextWindow" | "maxInputTokens" | "maxTokens" | "maxConcurrency">;
61
- local: { deny: string[]; bash_network: boolean; max_steps: number; read_allow: string[] };
65
+ /** `sandbox`: "deny-default" (issue #39), or "allow-default", the profile of 0.9 and earlier, kept for one release. */
66
+ local: { deny: string[]; bash_network: boolean; max_steps: number; read_allow: string[]; sandbox: "deny-default" | "allow-default" };
62
67
  /** Pending permission requests: how long they wait, and whether the desktop is told (issue #5). */
63
68
  approvals: { timeout_s: number; notify: boolean };
64
69
  /** An owner offline this long loses its open tasks back to routing; 0 turns it off (issue #6). */
@@ -67,6 +72,14 @@ export interface HubConfig {
67
72
  checks: { timeout_s: number; [cls: string]: string | number };
68
73
  /** A git tree at each turn boundary for `ahub turns` and `ahub undo`, the last `keep` per peer (issue #33). */
69
74
  snapshots: { enabled: boolean; keep: number };
75
+ /** Per-sender rate limits and repeat suppression for what agents send (issue #38). */
76
+ limits: LimitsConfig;
77
+ /** Reviewer choice from recorded review outcomes, once a reviewer has `min_reviews` of an implementer (issue #35). */
78
+ review: { adaptive: boolean; min_reviews: number };
79
+ /** After an unplanned stop, start Kimi, Pi and the local worker again with their recorded sessions (issue #37). */
80
+ recovery: { auto_resume_after_crash: boolean };
81
+ /** Per peer, the hub-tool capabilities it has (issue #39); a peer not listed has all of them. */
82
+ capabilities: Record<string, string[]>;
70
83
  /** Machine-local fields a config file set but git could not vouch for, and why (issue #17). */
71
84
  ignored?: string[];
72
85
  }
@@ -84,13 +97,17 @@ export const DEFAULT_CONFIG: HubConfig = {
84
97
  omniroute: DEFAULT_OMNIROUTE,
85
98
  pi: { enabled: false, auto_start: false, cmd: ["pi"], backend: "auto", dgx_coding: "coding", dgx_fast: "fast", max_steps: 30 },
86
99
  mlx: { provider: "ollama", model: "agenthub-fast-mlx:4b-8k", sourceModel: "qwen3.5:4b-mlx", contextWindow: 8192, maxInputTokens: 6000, maxTokens: 2048, maxConcurrency: 1 },
87
- local: { deny: [], bash_network: false, max_steps: 30, read_allow: [] },
100
+ local: { deny: [], bash_network: false, max_steps: 30, read_allow: [], sandbox: "deny-default" },
88
101
  // Off here, so tests and a hub without a config file stay silent; a project's config defaults it on for macOS.
89
102
  approvals: { timeout_s: 120, notify: false },
90
103
  tasks: { release_after_min: 30 },
91
104
  checks: { timeout_s: 600 },
92
105
  // Off here like approvals.notify, so tests (whose cwd is this repository) write no objects; a project's config turns it on.
93
106
  snapshots: { enabled: false, keep: 20 },
107
+ limits: DEFAULT_LIMITS,
108
+ review: { adaptive: false, min_reviews: 5 },
109
+ recovery: { auto_resume_after_crash: false },
110
+ capabilities: {},
94
111
  };
95
112
 
96
113
  export { stateDirFor };
@@ -101,7 +118,7 @@ const PEER_ID = /^[a-z][a-z0-9-]{0,31}$/;
101
118
 
102
119
  /** The shared project config, then the machine's own file, which overrides it block by block (issue #17). */
103
120
  const CONFIG_FILES = ["config.json", "config.local.json"] as const;
104
- const CONFIG_BLOCKS = ["memory", "roles", "budget", "inference", "omniroute", "local", "pi", "approvals", "tasks", "checks", "snapshots", "mlx"];
121
+ const CONFIG_BLOCKS = ["memory", "roles", "budget", "inference", "omniroute", "local", "pi", "approvals", "tasks", "checks", "snapshots", "limits", "review", "recovery", "capabilities", "mlx"];
105
122
 
106
123
  export function loadConfig(cwd: string): HubConfig {
107
124
  const ignored: string[] = [];
@@ -133,7 +150,7 @@ export function loadConfig(cwd: string): HubConfig {
133
150
  ...file,
134
151
  memory: { ...DEFAULT_CONFIG.memory, ...file.memory },
135
152
  roles: { ...DEFAULT_CONFIG.roles, ...file.roles },
136
- budget: { ...DEFAULT_CONFIG.budget, ...file.budget },
153
+ budget: { ...DEFAULT_CONFIG.budget, wait_max_min: 30, ...file.budget }, // on with any project config (issue #36)
137
154
  inference: { ...DEFAULT_CONFIG.inference, ...file.inference },
138
155
  omniroute: { ...DEFAULT_CONFIG.omniroute, ...file.omniroute },
139
156
  local: { ...DEFAULT_CONFIG.local, ...file.local },
@@ -145,6 +162,10 @@ export function loadConfig(cwd: string): HubConfig {
145
162
  tasks: { ...DEFAULT_CONFIG.tasks, ...file.tasks },
146
163
  checks: { ...DEFAULT_CONFIG.checks, ...file.checks },
147
164
  snapshots: { ...DEFAULT_CONFIG.snapshots, enabled: true, ...file.snapshots },
165
+ limits: { ...PROJECT_LIMITS, ...file.limits }, // on with any project config (issue #38)
166
+ review: { ...DEFAULT_CONFIG.review, ...file.review },
167
+ recovery: { ...DEFAULT_CONFIG.recovery, ...file.recovery },
168
+ capabilities: { ...file.capabilities },
148
169
  mlx,
149
170
  ...(ignored.length ? { ignored } : {}),
150
171
  };
@@ -274,10 +295,57 @@ export async function startDaemon(opts: DaemonOptions) {
274
295
  : undefined;
275
296
  if ((restartFilePresent && !restored) || (recoveryOperation && !restored)) throw new Error("restart state is unreadable, missing, or does not match this project and recovery operation");
276
297
 
298
+ // A session record left by a run that never shut down means it crashed (issue #37). A controlled restart has its own.
299
+ const crashed = !recoveryOperation && !restartFilePresent ? readSessions(opts.stateDir) : undefined;
300
+ // A controlled restart's source may have been cut short before it removed its record: this run is not a crash, and
301
+ // a record left now would make the next ordinary start look like one.
302
+ if (recoveryOperation || restartFilePresent) try { removeSessions(opts.stateDir); } catch { /* nothing to remove */ }
303
+ const autoResume = config.recovery.auto_resume_after_crash === true; // a string "false" is not a yes
304
+ const startedAt = Date.now();
305
+ /** What crash recovery did or asks the user to do, for `ahub status`. */
306
+ const crashReport: string[] = [];
307
+
277
308
  // The hub's own model calls (digest condensation, task triage) are wired below, once the gateway client exists.
278
309
  let inference: Inference | undefined;
279
310
  const journal = new DeliveryJournal({ file: join(opts.stateDir, "hub.db"), projectRoot: opts.cwd, projectId, instanceId, operationId: recoveryOperation });
280
- const bus = new Bus({ journal, batchMax: config.batch_max, batchMs: config.batch_ms, queueCap: config.queue_cap, condense: (envs) => inference?.condense(envs) ?? Promise.resolve(envs) });
311
+ // What the crash left in flight, per recipient: opening the journal just marked these needs_review.
312
+ const lost = new Map<PeerId, JournalDelivery[]>();
313
+ if (crashed) for (const d of journal.list()) if (d.state === "needs_review" && d.reason === "daemon stopped during delivery" && d.updatedAt >= startedAt) lost.set(d.peer, [...(lost.get(d.peer) ?? []), d]);
314
+ // Capabilities (issue #39): enforced here and in taskOp, never by role text alone. Unlisted peers keep everything.
315
+ // A listed peer whose value is not a list gets nothing: whoever listed it meant to narrow it.
316
+ const may = (peer: PeerId, cap: "propose" | "assign" | "remember" | "important"): boolean => {
317
+ if (peer === USER || peer === HUB || !Object.hasOwn(config.capabilities, peer)) return true;
318
+ const list = config.capabilities[peer];
319
+ return Array.isArray(list) && list.includes(cap);
320
+ };
321
+ const CAPABILITIES = ["propose", "assign", "remember", "important"];
322
+ for (const [peer, list] of Object.entries(config.capabilities)) {
323
+ if (!PEER_ID.test(peer)) log(`capabilities.${peer} is not a peer id; ignored (capabilities is an object of lists, one per peer)`);
324
+ else if (!Array.isArray(list)) log(`capabilities.${peer} is not a list: ${peer} gets no capabilities`);
325
+ else for (const c of list) if (!CAPABILITIES.includes(c)) log(`capabilities.${peer}: ${JSON.stringify(c)} is not a capability (${CAPABILITIES.join(", ")}); it grants nothing`);
326
+ }
327
+ // Agents only: the console user and the hub itself are never limited (issue #38).
328
+ // A typo such as "12/min" would read as 0, which turns a limit off without a word: the project default instead.
329
+ for (const k of Object.keys(config.limits)) if (!(k in PROJECT_LIMITS)) log(`limits.${k} is not a known limit; ignored`);
330
+ const limits = Object.fromEntries(Object.entries(PROJECT_LIMITS).map(([k, fallback]) => {
331
+ const v = config.limits[k as keyof LimitsConfig];
332
+ if (typeof v === "number" && Number.isFinite(v) && v >= 0) return [k, v];
333
+ log(`limits.${k}: ${JSON.stringify(v)} is not a number of 0 or more; using ${fallback}`);
334
+ return [k, fallback];
335
+ })) as unknown as LimitsConfig;
336
+ const limiter = new Limiter(limits);
337
+ const admit = (env: Envelope, parent?: string): string | undefined => {
338
+ // [FYI] is recorded and costs nobody a turn: nothing to limit.
339
+ if (env.from === USER || env.from === HUB || env.from === DIGEST || env.priority === "fyi") return undefined;
340
+ if (env.priority === "important" && !may(env.from, "important")) {
341
+ log(`capabilities: ${env.from} may not send important messages`);
342
+ return `${env.from} may not send important messages (no "important" in capabilities.${env.from}): send it without [IMPORTANT]`;
343
+ }
344
+ const refused = limiter.admit(env.from, env.to, env.priority, env.body, parent);
345
+ if (refused) log(`limits: ${env.from}: ${refused}`);
346
+ return refused;
347
+ };
348
+ const bus = new Bus({ journal, batchMax: config.batch_max, batchMs: config.batch_ms, queueCap: config.queue_cap, condense: (envs) => inference?.condense(envs) ?? Promise.resolve(envs), admit });
281
349
  startupCleanup.push(() => bus.closeJournal());
282
350
  const manualPaused = new Set<PeerId>(bus.manualPausedPeers()); // recovery never lifts an operator's pause
283
351
  let recoveryOperationId: string | undefined;
@@ -386,6 +454,8 @@ export async function startDaemon(opts: DaemonOptions) {
386
454
  }
387
455
  },
388
456
  triage: { classify: (title, detail) => inference?.triage(title, detail) ?? Promise.resolve(undefined), onCampus: () => onCampus() },
457
+ quota: (): ReturnType<Budget["headroom"]> => budget.headroom(), // budget is built below; this runs at assignment time
458
+ review: config.review,
389
459
  });
390
460
  board.onChange = (t, h) => event({ type: "task", id: t.id, event: h.event, by: h.by, state: t.state, owner: t.owner, reviewer: t.reviewer, class: t.class, pii: tasks.isPii(t) });
391
461
  // ---- budget relay -------------------------------------------------------------------------------------------
@@ -423,9 +493,9 @@ export async function startDaemon(opts: DaemonOptions) {
423
493
  attached: (peer) => bus.peers.has(peer),
424
494
  // Somebody other than the paused peer has to be there, or the handoff would only leave its tasks without an owner.
425
495
  canHandOff: (peer) => [...bus.peers.keys()].some((id) => id !== peer && ["idle", "busy"].includes(bus.stateOf(id))),
426
- handoff: (peer, context) => tasks.reassignForPause(peer, context),
427
496
  // A reading that arrived through a file carries the file's time; a stale one must not look fresh in the export.
428
497
  reading: (peer, windows, hard, at) => event({ type: "quota", peer, windows: windows.map((w) => ({ id: w.id, used: w.used, ...(w.resetsAt ? { resetsAt: w.resetsAt } : {}) })), hard, ...(Math.abs(Date.now() - at) > 1000 ? { measuredAt: new Date(at).toISOString() } : {}) }),
498
+ handoff: (peer, context, urgentOnly) => tasks.reassignForPause(peer, context, urgentOnly),
429
499
  resumed: (record) => {
430
500
  if (record.peer === "kimi") kimiTokens.length = 0; // a new window: the old counts would pause it again at once
431
501
  const moved = record.moved.length ? `While you were paused these moved: ${record.moved.map((m) => `${m.title} (${m.role} -> ${m.to ?? "nobody"})`).join("; ")}. They stay where they are; ask the user if you should take one back.` : "Nothing was moved while you were paused.";
@@ -519,7 +589,19 @@ export async function startDaemon(opts: DaemonOptions) {
519
589
  const toolEnv = (peer: PeerId) => ({ AGENTHUB_MODE: "tools", AGENTHUB_PEER_ID: peer, AGENTHUB_STATE_DIR: opts.stateDir, AGENTHUB_PROJECT_DIR: opts.cwd });
520
590
 
521
591
  /** One entry point for the task tools, whoever calls them: MCP clients, the local worker, the console. */
522
- async function taskOp(by: PeerId, op: string, a: Record<string, any>, inProcess = false, piiTurn = false): Promise<string> {
592
+ // A task op can write the board across awaits (triage, briefs, the dependents an approval releases): a recovery
593
+ // commit waits for those in flight, or its integrity digest misses their later writes. Completion checks outlive
594
+ // their op, so recoveryReady() also waits for `tasks.checksPending()`: a check the commit's stop kills would write.
595
+ let taskOpsInFlight = 0;
596
+ const taskOp = async (...args: Parameters<typeof taskOpBody>): Promise<string> => {
597
+ taskOpsInFlight++;
598
+ try {
599
+ return await taskOpBody(...args);
600
+ } finally {
601
+ taskOpsInFlight--;
602
+ }
603
+ };
604
+ async function taskOpBody(by: PeerId, op: string, a: Record<string, any>, inProcess = false, piiTurn = false): Promise<string> {
523
605
  // Inside a PII turn the worker's words may carry the PII whatever they are attached to: a note would go to
524
606
  // claude-mem (a cloud observer) and a new task could be routed to a cloud peer without matching any pattern.
525
607
  if (piiTurn && (op === "hub_remember" || op === "hub_task_propose")) throw new Error(`${op} is not available while working on a PII task: its text must not leave this machine`);
@@ -534,6 +616,16 @@ export async function startDaemon(opts: DaemonOptions) {
534
616
  // console reads a PII task's text deliberately, with `ahub task show <id>`.
535
617
  const onPrem = inProcess && by === "local";
536
618
  const line = (t: { id: number; state: string; owner: PeerId | null; reviewer: PeerId | null }) => `task #${t.id}: ${t.state}, owner ${t.owner ?? "none"}, reviewer ${t.reviewer ?? "none"}`;
619
+ const need = (cap: "propose" | "assign" | "remember", what: string) => {
620
+ if (may(by, cap)) return;
621
+ log(`capabilities: ${by} may not ${what} (${op})`);
622
+ throw new Error(`${by} may not ${what} (no "${cap}" in capabilities.${by} in .agenthub/config.json)`);
623
+ };
624
+ if (op === "hub_task_propose") {
625
+ need("propose", "propose tasks");
626
+ if (typeof a.owner === "string" && a.owner && a.owner !== by) need("assign", "hand tasks to other peers");
627
+ }
628
+ if (op === "hub_remember") need("remember", "save notes to shared memory");
537
629
  switch (op) {
538
630
  case "hub_task_propose": {
539
631
  const t = await tasks.propose(by, a);
@@ -552,7 +644,7 @@ export async function startDaemon(opts: DaemonOptions) {
552
644
  return tasks.isChecking(t.id) ? `${line(t)}; its check is queued or running, and the result comes as a task message` : line(t);
553
645
  }
554
646
  case "hub_review":
555
- return line(await tasks.review(by, a.id, a.verdict, a.note));
647
+ return line(await tasks.review(by, a.id, a.verdict, a.note, a.unmet));
556
648
  case "hub_remember":
557
649
  return tasks.remember(by, a);
558
650
  case "hub_checkpoint": {
@@ -563,12 +655,12 @@ export async function startDaemon(opts: DaemonOptions) {
563
655
  return "checkpoint received; you will be paused now and resumed when your window resets";
564
656
  }
565
657
  case "hub_task_list":
566
- return JSON.stringify(board.list(a.state).map((t) => (onPrem ? t : tasks.publicView(t))).map(({ history: _h, ...t }) => t));
658
+ return JSON.stringify(board.list(a.ready === true ? "proposed" : a.state).filter((t) => a.ready !== true || !tasks.waitsFor(t).length).map((t) => (onPrem ? t : tasks.publicView(t))).map(({ history: _h, ...t }) => t));
567
659
  }
568
660
  if (by !== USER) throw new Error(`${op} is a console command`);
569
661
  switch (op) {
570
662
  case "task_show":
571
- return JSON.stringify(board.get(Number(a.id)) ?? `no task #${a.id}`, null, 2);
663
+ return JSON.stringify(board.get(Number(a.id)) ? { ...board.get(Number(a.id)), reviews: board.reviews({ task: Number(a.id) }) } : `no task #${a.id}`, null, 2);
572
664
  case "task_assign":
573
665
  return line(await tasks.assignTo(a.id, String(a.peer)));
574
666
  case "task_escalate":
@@ -614,9 +706,9 @@ export async function startDaemon(opts: DaemonOptions) {
614
706
  const r = budget.record(id); // one read per peer: status.json is rewritten on every bus event
615
707
  return r ? { paused: `budget: ${r.reason}, resets ${new Date(r.resetsAt).toLocaleTimeString()}` } : manualPaused.has(id) && bus.stateOf(id) === "offline" ? { paused: "manual" } : {};
616
708
  };
617
- let releasing = false; // gone-owner release (#6): one run at a time, and a recovery commit waits for it
709
+ let releasing = false; // gone-owner release (#6) and the ready sweep (#34): one run at a time, and a recovery commit waits for it
618
710
  const recoveryReady = () => {
619
- if (!recoveryActive() || releasing || (piReceipts?.inFlight ?? 0) !== 0 || permissions.size !== 0 || starting.size !== 0 || !budget.recoverySettled || [...bus.peers.values()].some((peer) => peer.state === "busy" || (peer instanceof PiPeer && !peer.recoveryReady))) return false;
711
+ if (!recoveryActive() || releasing || taskOpsInFlight !== 0 || tasks.checksPending() !== 0 || (piReceipts?.inFlight ?? 0) !== 0 || permissions.size !== 0 || starting.size !== 0 || !budget.recoverySettled || [...bus.peers.values()].some((peer) => peer.state === "busy" || (peer instanceof PiPeer && !peer.recoveryReady))) return false;
620
712
  if (!recoveryPeerSnapshot) return true;
621
713
  const current = recoveryPeers();
622
714
  return recoveryPeerSnapshot.every((saved) => {
@@ -673,23 +765,33 @@ export async function startDaemon(opts: DaemonOptions) {
673
765
  return [id, row];
674
766
  }));
675
767
  const digest = (value: unknown) => createHash("sha256").update(JSON.stringify(value)).digest("hex");
676
- // `prePlan`: the board as a hub from before the plan column (#31, 0.8.0) digested it. An empty plan is left out, so an
677
- // upgrade from such a hub still verifies the board it was handed; a real plan never matches that shape.
678
- const integrity = (prePlan = false) => {
768
+ // Task columns added since a source may have recorded its digest, newest first, with their empty values. A hub from
769
+ // before them digested its rows without them (0.8.x: no deps, #34; 0.7.x: no plan either, #31). `older` leaves out
770
+ // the newest `older` of them while they are empty, so an upgrade from such a hub verifies the board it was handed; a
771
+ // filled one stays in the row and never matches.
772
+ const ADDED_COLUMNS = [["deps", "[]"], ["plan", "{}"]] as const;
773
+ const integrity = (older = 0) => {
679
774
  const queues = Object.fromEntries(Object.keys(bus.snapshot().queues).sort().map((id) => [id, bus.queueIds(id)]));
680
- const tasks = board.list().sort((a, b) => a.id - b.id);
681
- const boardState = prePlan ? tasks.map(({ plan, ...t }) => (plan && Object.keys(plan).length ? { ...t, plan } : t)) : tasks;
775
+ const boardState = board.list().sort((a, b) => a.id - b.id).map((t) => {
776
+ const row: Record<string, unknown> = { ...t };
777
+ for (const [col, empty] of ADDED_COLUMNS.slice(0, older)) if (JSON.stringify(row[col]) === empty) delete row[col];
778
+ return row;
779
+ });
682
780
  const budgetState = budget.persistedPauseDigestRows().sort((a, b) => a.peer.localeCompare(b.peer));
683
781
  return { queues, manualPaused: [...manualPaused].sort(), boardDigest: digest(boardState), budgetDigest: digest(budgetState) };
684
782
  };
685
- /** The integrity in the shape `expected` was recorded in: the current one unless only the pre-plan shape matches. */
783
+ /** The integrity in the shape `expected` was recorded in: the current one unless only an older shape matches. */
686
784
  const integrityAs = (expected: unknown) => {
687
785
  const now = integrity();
688
786
  if (!expected || JSON.stringify(expected) === JSON.stringify(now)) return now;
689
- const old = integrity(true);
690
- return JSON.stringify(expected) === JSON.stringify(old) ? old : now;
787
+ for (let older = 1; older <= ADDED_COLUMNS.length; older++) {
788
+ const then = integrity(older);
789
+ if (JSON.stringify(expected) === JSON.stringify(then)) return then;
790
+ }
791
+ return now;
691
792
  };
692
793
  const status = () => ({
794
+ ...(crashReport.length ? { crash: crashReport } : {}),
693
795
  projectId,
694
796
  instanceId,
695
797
  version: VERSION,
@@ -726,10 +828,11 @@ export async function startDaemon(opts: DaemonOptions) {
726
828
  const hubStartedAt = Date.now();
727
829
  const releaseGoneOwners = async () => {
728
830
  const limit = config.tasks.release_after_min;
729
- if (!(limit > 0) || releasing || stopping || recoveryActive()) return;
831
+ if (releasing || stopping || recoveryActive()) return;
730
832
  releasing = true;
731
833
  try {
732
- await releaseOwners(limit);
834
+ await tasks.releaseReady(); // #34: dependents a stop cut off between an approval and their assignment
835
+ if (limit > 0) await releaseOwners(limit);
733
836
  } finally {
734
837
  releasing = false;
735
838
  }
@@ -749,6 +852,74 @@ export async function startDaemon(opts: DaemonOptions) {
749
852
  releaseTimer.unref?.();
750
853
  intervals.push(releaseTimer);
751
854
 
855
+ // Early conflict detection (issue #32): the files a turn changed, against what other owners' open tasks changed
856
+ // before it. Warns both owners once per file and task; never blocks a write.
857
+ const conflictSeen = new Set<string>();
858
+ const detectConflicts = (peer: PeerId, turnId: string, since: number, changed: string[]) => {
859
+ const open = board.list().filter((t) => t.owner && ["proposed", "in_progress", "changes_requested"].includes(t.state));
860
+ const mine = open.filter((t) => t.owner === peer && t.state === "in_progress");
861
+ if (mine.some((t) => tasks.isPii(t))) return; // a PII turn's files are nobody else's business
862
+ const visible = open.filter((t) => !tasks.isPii(t));
863
+ const found = conflictsOf(peer, changed, turnLog!.touchesFor(visible.map((t) => t.id)), visible);
864
+ // A turn's diff holds whatever changed while it ran. Only what no overlapping turn of another peer changed is
865
+ // recorded as this peer's; while such a turn's changes are unknown (still running, say), nothing is (review of #49).
866
+ const record = turnLog!.get(turnId);
867
+ const overlap = record ? turnLog!.overlapping(record, Math.max(1, Number(config.snapshots.keep) || 20)) : { paths: [], unknown: [] };
868
+ if (!overlap.unknown.length) {
869
+ const theirs = new Set(overlap.paths);
870
+ const own = changed.filter((p) => !theirs.has(p));
871
+ for (const t of mine) turnLog!.touch(t.id, peer, own);
872
+ }
873
+ if (!found.length) return;
874
+ const others = turnLog!.busySince(peer, since);
875
+ const concurrent = others.length ? ` Concurrent: ${others.join(", ")} also worked during that turn, so some of these changes may be theirs.` : "";
876
+ const ours = mine.length ? ` (task ${mine.map((t) => `#${t.id}`).join(", ")})` : "";
877
+ for (const { task, paths: all } of found) {
878
+ const paths = all.filter((p) => !conflictSeen.has(`${peer}\0${task.id}\0${p}`));
879
+ if (!paths.length) continue;
880
+ for (const p of paths) conflictSeen.add(`${peer}\0${task.id}\0${p}`);
881
+ const owner = task.owner!;
882
+ // As with overlaps (#31): a file name that matches a PII pattern is never named to peers, the log or telemetry.
883
+ const named = paths.filter(tasks.nameable);
884
+ const hidden = paths.length - named.length;
885
+ const files = [...named, ...(hidden ? [`${hidden} file(s) whose names are withheld (they match a PII pattern)`] : [])].join(", ");
886
+ notify(`conflict: ${peer}${ours} changed ${files}, which #${task.id} (owner ${owner}) changed before${others.length ? ` (concurrent: ${others.join(", ")})` : ""}`);
887
+ event({ type: "conflict", peer, ...(mine[0] ? { task: mine[0].id } : {}), other: task.id, owner, paths: named, concurrent: others.length > 0 });
888
+ bus.publish(newEnvelope(HUB, `Your last turn${ours} changed ${files}, which ${owner}'s open task (${tasks.publicTitle(task)}) changed before it. Check that you did not overwrite that work, and settle it with ${owner} via hub_send.${concurrent}`, { to: [peer], kind: "task", ...(mine[0] ? { refs: { task: String(mine[0].id) } } : {}) }));
889
+ if (owner !== USER && owner !== HUB) bus.publish(newEnvelope(HUB, `${peer}'s last turn${ours} changed ${files}, which your open task #${task.id} changed before it. Check that your work there is intact.${concurrent}`, { to: [owner], kind: "task", refs: { task: String(task.id) } }));
890
+ }
891
+ };
892
+
893
+ // Session identities of the attached peers, kept current for crash recovery (issue #37); a clean stop removes them.
894
+ let sessionsWritten = "";
895
+ const recordSessions = () => {
896
+ if (stopping) return;
897
+ const peers = [...bus.peers.values()].filter((p) => p.state !== "offline").map((p) => {
898
+ let meta: Record<string, unknown> = {};
899
+ try { meta = (p as { recoveryMetadata?: () => Record<string, unknown> }).recoveryMetadata?.() ?? {}; } catch { /* not ready yet: the id alone */ }
900
+ return { peer: p.id, meta };
901
+ });
902
+ const text = JSON.stringify(peers);
903
+ if (text === sessionsWritten) return;
904
+ sessionsWritten = text;
905
+ try { writeSessions(opts.stateDir, { instanceId, at: Date.now(), peers }); } catch (error) { log(`session record not written: ${(error as Error).message}`); }
906
+ };
907
+ const recoverAfterCrash = async (prev: SessionsFile) => {
908
+ const report = (line: string) => (crashReport.push(line), notify(`crash recovery: ${line}`));
909
+ const lostCount = [...lost.values()].reduce((n, l) => n + l.length, 0);
910
+ report(`the previous hub run stopped without shutting down${lostCount ? `; ${lostCount} deliveries it had in flight are in needs_review (ahub queue list)` : ""}`);
911
+ for (const step of crashPlan(prev.peers)) {
912
+ if (!step.resume || !autoResume) {
913
+ const fresh = step.peer === "pi" && config.pi.enabled && config.pi.auto_start ? "; pi.auto_start starts a fresh session" : "";
914
+ report(`${step.how}${step.resume ? " (recovery.auto_resume_after_crash is off)" : ""}${fresh}`);
915
+ continue;
916
+ }
917
+ const r = await startPeer(step.peer, step.resume as Parameters<typeof startPeer>[1]).catch((e: Error) => ({ ok: false, error: e.message }));
918
+ report(r.ok ? `${step.peer} resumed: ${step.how}` : `${step.peer} not resumed (${String(r.error)}); ${step.how}`);
919
+ }
920
+ writeStatus();
921
+ };
922
+
752
923
  bus.tap((e) => {
753
924
  e = redact(e);
754
925
  uiEvents.push({ seq: ++uiSequence, event: e });
@@ -770,6 +941,7 @@ export async function startDaemon(opts: DaemonOptions) {
770
941
  event({ type: "turn_start", peer: e.peer, turn: id });
771
942
  } else if (!busy && open) {
772
943
  turns.delete(e.peer);
944
+ let afterTurn: (() => void) | undefined;
773
945
  let files: number | undefined;
774
946
  let snapshotMs = open.snapshotMs;
775
947
  if (turnLog && (open.private || holdsPii(e.peer))) {
@@ -780,11 +952,21 @@ export async function startDaemon(opts: DaemonOptions) {
780
952
  if (open.tree && end.tree) files = changed.length; // unknown, not zero, when a snapshot failed
781
953
  snapshotMs = (snapshotMs ?? 0) + end.ms;
782
954
  try { turnLog.end(open.id, end.tree, changed, Math.max(1, Number(config.snapshots.keep) || 20)); } catch (error) { log(`turn record ${open.id}: ${(error as Error).message}`); }
955
+ // After the turn_end event: a notice delivered at once starts the peer's next turn, which must come after it.
956
+ if (changed.length) afterTurn = () => detectConflicts(e.peer, open.id, open.start, changed);
783
957
  }
784
958
  event({ type: "turn_end", peer: e.peer, turn: open.id, ms: Date.now() - open.start, ...(open.tokens ? { tokens: open.tokens } : {}), ...(files !== undefined ? { files, snapshotMs } : {}) });
959
+ try { afterTurn?.(); } catch (error) { log(`conflict check after ${open.id}: ${(error as Error).message}`); }
785
960
  }
786
961
  if (e.state === "offline") offlineSince.set(e.peer, offlineSince.get(e.peer) ?? Date.now());
787
962
  else offlineSince.delete(e.peer);
963
+ // After a crash, a peer's first attach brings the loss notice: it leads its next delivery (issue #37).
964
+ if (e.state !== "offline" && lost.has(e.peer)) {
965
+ const still = lost.get(e.peer)!.filter((d) => { try { return journal.get(d.id)?.state === "needs_review"; } catch { return false; } });
966
+ lost.delete(e.peer);
967
+ if (still.length) bus.preface(e.peer, lossNotice(still, (id) => { const t = board.get(id); return t ? tasks.publicTitle(t) : undefined; }));
968
+ }
969
+ recordSessions();
788
970
  }
789
971
  else if (e.t === "undeliverable" || e.t === "overflow") {
790
972
  log(e.t === "undeliverable" ? `UNDELIVERABLE to ${e.peer} after retries: ${e.env.id} from ${e.env.from}` : `OVERFLOW ${e.peer}: dropped ${e.env.id} from ${e.env.from}`);
@@ -935,6 +1117,7 @@ export async function startDaemon(opts: DaemonOptions) {
935
1117
  const kimi = new AcpPeer("kimi", {
936
1118
  cmd,
937
1119
  ...(args.model ? { launchModel: args.model } : {}),
1120
+ ...(args.sessionId ? { resumeSessionId: args.sessionId } : {}),
938
1121
  cwd: opts.cwd,
939
1122
  watchdogMs: config.watchdog_ms,
940
1123
  onPermission,
@@ -994,11 +1177,12 @@ export async function startDaemon(opts: DaemonOptions) {
994
1177
  let piReply: Envelope | undefined;
995
1178
  const ctx: ToolContext = {
996
1179
  cwd: opts.cwd, deny: config.local.deny,
997
- sandboxProfile: profile(opts.cwd, config.local.bash_network, config.local.read_allow, config.local.deny),
1180
+ sandboxProfile: profile(opts.cwd, config.local.bash_network, config.local.read_allow, config.local.deny, config.local.sandbox === "allow-default" ? "allow" : "deny"),
998
1181
  permit: (title) => onPermission({ peer: "pi", title, options: [{ optionId: "allow", name: "Allow", kind: "allow_once" }, { optionId: "deny", name: "Deny", kind: "reject_once" }] }).then((picked) => picked === "allow" && pi.acceptingTools && bus.peers.get("pi") === pi),
999
1182
  send: (text, to) => {
1000
1183
  if (to?.some((id) => !bus.peers.has(id) && id !== USER)) return "error: unknown peer";
1001
- pi.onMessage?.(text, { inReplyTo: piReply, ...(to?.length ? { to } : {}) }); return "sent";
1184
+ const refused = pi.onMessage?.(text, { inReplyTo: piReply, ...(to?.length ? { to } : {}) });
1185
+ return typeof refused === "string" ? `not sent: ${refused}` : "sent";
1002
1186
  },
1003
1187
  };
1004
1188
  const routing = currentRouting(opts.cwd, log);
@@ -1084,7 +1268,7 @@ export async function startDaemon(opts: DaemonOptions) {
1084
1268
  omni,
1085
1269
  ...(sidecar && route ? { sidecar, route } : {}),
1086
1270
  fixedModel: args.model ?? routing.local.fixed_model,
1087
- tools: { deny: config.local.deny, bashNetwork: config.local.bash_network, readAllow: config.local.read_allow, permit },
1271
+ tools: { deny: config.local.deny, bashNetwork: config.local.bash_network, readAllow: config.local.read_allow, sandbox: config.local.sandbox, permit },
1088
1272
  ...(capture ? { capture } : {}),
1089
1273
  taskTool: (name, a, turn) => taskOp("local", name, a, true, turn.pii),
1090
1274
  turnPolicy: (envs) => tasks.turnPolicy(envs),
@@ -1270,7 +1454,7 @@ export async function startDaemon(opts: DaemonOptions) {
1270
1454
  return { t: "recovery", ok: true, aborted: true, recovery: recoveryView() };
1271
1455
  }
1272
1456
  if (msg.op === "commit") {
1273
- if ((recoveryPhase !== "prepared" && recoveryPhase !== "preparing") || !recoveryReady()) return recoveryError("recovery is not ready; inspect until peers are idle and approvals are complete");
1457
+ if ((recoveryPhase !== "prepared" && recoveryPhase !== "preparing") || !recoveryReady()) return recoveryError("recovery is not ready; inspect until peers are idle, approvals are complete and completion checks have finished");
1274
1458
  recoveryPhase = "prepared";
1275
1459
  recoveryPeerSnapshot ??= Object.values(recoveryPeers());
1276
1460
  const currentPeers = recoveryPeers();
@@ -1395,11 +1579,15 @@ export async function startDaemon(opts: DaemonOptions) {
1395
1579
  // A human at the console should not wait out the batch window; agents default to status.
1396
1580
  const { priority, body } = parseMarker(String(msg.body ?? ""), c.peer ? "status" : "important");
1397
1581
  if (!body) return void reply({ t: "sent", ok: false, error: "empty body" });
1398
- const to: PeerId[] | undefined = Array.isArray(msg.to) && msg.to.length ? msg.to.map(String) : undefined;
1582
+ const to: PeerId[] | undefined = Array.isArray(msg.to) && msg.to.length ? [...new Set<string>(msg.to.map(String))] : undefined;
1399
1583
  const unknown = to?.filter((id) => !bus.knownPeers().includes(id)) ?? [];
1400
1584
  if (unknown.length) return void reply({ t: "sent", ok: false, error: `unknown peer: ${unknown.join(", ")}` });
1401
1585
  const inReplyTo = msg.reply_to ? bus.get(String(msg.reply_to)) : undefined;
1402
- const targets = bus.publish(newEnvelope(c.peer ?? USER, body, { priority, ...(to ? { to } : {}), ...(inReplyTo ? { inReplyTo } : {}) }));
1586
+ // Built first: limits count the audience it really has (a reply goes to the parent's sender).
1587
+ const env = newEnvelope(c.peer ?? USER, body, { priority, ...(to ? { to } : {}), ...(inReplyTo ? { inReplyTo } : {}) });
1588
+ const refused = c.peer ? admit(env, inReplyTo?.id) : undefined;
1589
+ if (refused) return void reply({ t: "sent", ok: false, error: refused });
1590
+ const targets = bus.publish(env);
1403
1591
  if (c.peer && inReplyTo) bus.completeReply(c.peer, inReplyTo.id);
1404
1592
  return void reply({ t: "sent", ok: true, targets, recorded: priority === "fyi" });
1405
1593
  }
@@ -1526,6 +1714,8 @@ export async function startDaemon(opts: DaemonOptions) {
1526
1714
  }
1527
1715
  async function stopOnce(): Promise<void> {
1528
1716
  stopping = true;
1717
+ // A stop someone asked for is not a crash, even if it then runs past the shutdown deadline: forget the sessions now.
1718
+ try { removeSessions(opts.stateDir, instanceId); } catch { /* the state dir is gone */ }
1529
1719
  checksClosed = true;
1530
1720
  for (const kill of runningChecks) kill();
1531
1721
  log("hub stopping");
@@ -1581,7 +1771,14 @@ export async function startDaemon(opts: DaemonOptions) {
1581
1771
  writeStatus();
1582
1772
  log(`${RUN_START}${process.pid} control=127.0.0.1:${server.port} cwd=${opts.cwd}`);
1583
1773
  ready = true;
1584
- if (config.pi.enabled && config.pi.auto_start && !recoveryActive()) void startPeer("pi", {}).catch((error) => log(`Pi auto-start failed: ${error.message}`));
1774
+ if (crashed) {
1775
+ // This run owns the record now, so its clean stop removes it even if no peer attaches to rewrite it.
1776
+ try { if (readSessions(opts.stateDir)?.instanceId === crashed.instanceId) writeSessions(opts.stateDir, { ...crashed, instanceId, at: Date.now() }); } catch (error) { log(`session record not adopted: ${(error as Error).message}`); }
1777
+ void recoverAfterCrash(crashed).catch((error) => log(`crash recovery failed: ${(error as Error).message}`));
1778
+ }
1779
+ // Skipped only when crash recovery itself starts Pi again on its recorded session.
1780
+ const piResumes = !!crashed && autoResume && crashPlan(crashed.peers).some((s) => s.peer === "pi" && s.resume);
1781
+ if (config.pi.enabled && config.pi.auto_start && !recoveryActive() && !piResumes) void startPeer("pi", {}).catch((error) => log(`Pi auto-start failed: ${error.message}`));
1585
1782
  return { bus, token, port: server.port as number, stop, stopped: new Promise<void>((r) => (onStop = r)) };
1586
1783
  } finally {
1587
1784
  if (!ready) for (const cleanup of startupCleanup.reverse()) { try { cleanup(); } catch { /* preserve startup error */ } }
package/src/hub/events.ts CHANGED
@@ -16,6 +16,7 @@ export type HubEvent =
16
16
  | { type: "tokens"; peer: string; n: number }
17
17
  | { type: "task"; id: number; event: string; by: string; state: string; owner: string | null; reviewer: string | null; class: string; pii: boolean }
18
18
  | { type: "overlap"; task: number; owner: string; others: { task: number; owner: string; paths: string[]; symbols?: string[] }[] }
19
+ | { type: "conflict"; peer: string; task?: number; other: number; owner: string; paths: string[]; concurrent: boolean }
19
20
  | { type: "quota"; peer: string; windows: { id: string; used: number; resetsAt?: number }[]; hard: boolean; measuredAt?: string };
20
21
 
21
22
  export type StampedEvent = HubEvent & { v: number; at: string };