@staix/agent-hub 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/hub/ask.ts CHANGED
@@ -85,14 +85,19 @@ export async function gather(question: string, d: AskDeps): Promise<{ evidence:
85
85
 
86
86
  // Matching tasks first, then the most recently touched; PII rows keep their place with a stub, so counts stay right.
87
87
  const tasks = d.board.list().sort((a, b) => hit(b.title) - hit(a.title) || b.updated - a.updated).slice(0, MAX_TASKS);
88
- const anyPii = tasks.some((t) => d.isPii(t));
88
+ // An ordinary task's last note is model-written (a summary, a review note): one that matches a PII pattern is shown
89
+ // only on campus, like a PII task (#69).
90
+ const notePii = (t: (typeof tasks)[number]) => !d.isPii(t) && !!t.history.at(-1)?.note && (d.isPiiText?.(t.history.at(-1)!.note!) ?? false);
91
+ const anyPii = tasks.some((t) => d.isPii(t) || notePii(t));
89
92
  const showPii = anyPii ? await onCampus() : false;
90
93
  let piiShown = false;
91
94
  const taskRows: Evidence[] = tasks.map((t) => {
92
95
  const secret = d.isPii(t);
93
96
  if (secret && showPii) piiShown = true;
94
97
  const last = t.history.at(-1);
95
- const text = secret && !showPii ? "[pii] (its text is shown only when the model is reached on campus)" : `${t.title}${last ? ` | last: ${last.event} by ${last.by}${last.note ? ` (${last.note})` : ""}` : ""}`;
98
+ if (notePii(t) && showPii) piiShown = true;
99
+ const note = last?.note && notePii(t) && !showPii ? "its note is shown only when the model is reached on campus" : last?.note;
100
+ const text = secret && !showPii ? "[pii] (its text is shown only when the model is reached on campus)" : `${t.title}${last ? ` | last: ${last.event} by ${last.by}${note ? ` (${note})` : ""}` : ""}`;
96
101
  return { id: `task #${t.id}`, kind: "task", text: `[${t.class}] ${t.state}, owner ${t.owner ?? "none"}, reviewer ${t.reviewer ?? "none"}: ${text}`.slice(0, ITEM_CHARS) };
97
102
  });
98
103
 
package/src/hub/board.ts CHANGED
@@ -22,6 +22,8 @@ export interface HistoryEntry {
22
22
  by: PeerId;
23
23
  event: string; // proposed | assigned | accepted | declined | done | approved | changes_requested | escalated | reassigned
24
24
  note?: string;
25
+ /** The owner an event that set it left the task with (#67); absent in rows written before 0.11. */
26
+ owner?: PeerId | null;
25
27
  }
26
28
  export interface Task {
27
29
  id: number;
@@ -74,6 +76,11 @@ export class Board {
74
76
  // Who did well or badly at which class, for demotion (issue #36). Only peers, classes and times: no task text.
75
77
  this.db.run("CREATE TABLE IF NOT EXISTS outcomes (peer TEXT NOT NULL, class TEXT NOT NULL, ok INTEGER NOT NULL, at INTEGER NOT NULL)");
76
78
  this.db.run("CREATE INDEX IF NOT EXISTS outcomes_class_at ON outcomes (class, at)");
79
+ // How reviews turned out, per (implementer, reviewer, class) (issue #35): approved, contradicted later, caught, escalated.
80
+ this.db.run("CREATE TABLE IF NOT EXISTS reviews (implementer TEXT NOT NULL, reviewer TEXT NOT NULL, class TEXT NOT NULL, kind TEXT NOT NULL, task INTEGER NOT NULL, at INTEGER NOT NULL)");
81
+ // ponytail: kept for good (a record is the whole history); prune by age if it ever grows large.
82
+ this.db.run("CREATE INDEX IF NOT EXISTS reviews_class ON reviews (class)");
83
+ this.db.run("CREATE INDEX IF NOT EXISTS reviews_task ON reviews (task)");
77
84
  // Boards from before issues #31 and #34 lack these columns; existing rows get the defaults.
78
85
  const have = new Set((this.db.query("PRAGMA table_info(tasks)").all() as { name: string }[]).map((c) => c.name));
79
86
  for (const [col, empty] of [["plan", "{}"], ["deps", "[]"]] as const) {
@@ -117,7 +124,7 @@ export class Board {
117
124
  throw new Error(`task #${id} is ${task.state}: cannot move to ${patch.state}`);
118
125
  }
119
126
  const next = { ...task, ...patch, refs: { ...task.refs, ...patch.refs } };
120
- const history = [...task.history, { at: Date.now(), by, event, ...(note ? { note } : {}) }];
127
+ const history = [...task.history, { at: Date.now(), by, event, ...(note ? { note } : {}), ...("owner" in patch ? { owner: next.owner } : {}) }];
121
128
  this.db
122
129
  .query("UPDATE tasks SET state = ?, owner = ?, reviewer = ?, refs = ?, plan = ?, rejections = ?, history = ?, updated = ? WHERE id = ?")
123
130
  .run(next.state, next.owner, next.reviewer, JSON.stringify(next.refs), JSON.stringify(next.plan ?? {}), next.rejections, JSON.stringify(history), Date.now(), id);
@@ -135,11 +142,32 @@ export class Board {
135
142
  return this.db.query("SELECT peer, ok, at FROM outcomes WHERE class = ? AND at >= ?").all(cls, since) as { peer: PeerId; ok: number; at: number }[];
136
143
  }
137
144
 
145
+ recordReview(r: Omit<ReviewOutcome, "at">, at = Date.now()): void {
146
+ this.db.query("INSERT INTO reviews (implementer, reviewer, class, kind, task, at) VALUES (?, ?, ?, ?, ?, ?)").run(r.implementer, r.reviewer, r.class, r.kind, r.task, at);
147
+ }
148
+
149
+ /** One task's review outcomes, or a class's (for reviewer choice). */
150
+ reviews(where: { task: number } | { class: TaskClass }): ReviewOutcome[] {
151
+ return ("task" in where
152
+ ? this.db.query("SELECT * FROM reviews WHERE task = ? ORDER BY at").all(where.task)
153
+ : this.db.query("SELECT * FROM reviews WHERE class = ? ORDER BY at").all(where.class)) as ReviewOutcome[];
154
+ }
155
+
138
156
  close(): void {
139
157
  this.db.close();
140
158
  }
141
159
  }
142
160
 
161
+ export interface ReviewOutcome {
162
+ implementer: PeerId;
163
+ reviewer: PeerId;
164
+ class: TaskClass;
165
+ /** approved; contradicted (a later check failure or changes requested on the same places); caught (changes requested, then the redo passed); escalated */
166
+ kind: "approved" | "contradicted" | "caught" | "escalated";
167
+ task: number;
168
+ at: number;
169
+ }
170
+
143
171
  function parse(row: Record<string, unknown>): Task {
144
172
  const out = { ...row } as Record<string, unknown>;
145
173
  for (const col of JSON_COLS) out[col] = JSON.parse(String(row[col]));
@@ -24,6 +24,8 @@ export const MACHINE_LOCAL = [
24
24
  "memory.worker_url",
25
25
  "local.read_allow",
26
26
  "local.bash_network",
27
+ "local.network_allow", // the hosts the egress proxy opens (#65)
28
+ "local.sandbox", // "allow-default" widens what the worker's commands may do
27
29
  ] as const;
28
30
 
29
31
  /**
@@ -0,0 +1,109 @@
1
+ import { chmodSync, existsSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import type { JournalDelivery } from "./delivery-journal.ts";
4
+
5
+ /**
6
+ * Unplanned-crash recovery (issue #37). While the hub runs it keeps `sessions.json`: each attached peer's session
7
+ * identity (the adapters' recovery metadata: ids and launch options, never message text). A clean stop removes the
8
+ * file, so finding it at start means the previous run died.
9
+ */
10
+ export interface SessionRecord {
11
+ peer: string;
12
+ /** The adapter's recovery metadata: `launch`, and `sessionId` / `threadId` / `sessionFile` where it has one. */
13
+ meta: Record<string, unknown>;
14
+ }
15
+ export interface SessionsFile {
16
+ instanceId: string;
17
+ at: number;
18
+ peers: SessionRecord[];
19
+ }
20
+
21
+ const fileOf = (stateDir: string) => join(stateDir, "sessions.json");
22
+
23
+ export function writeSessions(stateDir: string, s: SessionsFile): void {
24
+ const tmp = `${fileOf(stateDir)}.tmp`;
25
+ writeFileSync(tmp, `${JSON.stringify(s)}\n`, { mode: 0o600 });
26
+ chmodSync(tmp, 0o600);
27
+ renameSync(tmp, fileOf(stateDir));
28
+ }
29
+
30
+ export function readSessions(stateDir: string): SessionsFile | undefined {
31
+ if (!existsSync(fileOf(stateDir))) return undefined;
32
+ try {
33
+ const s = JSON.parse(readFileSync(fileOf(stateDir), "utf8")) as SessionsFile;
34
+ if (!Array.isArray(s.peers)) return undefined;
35
+ // A record without a peer id or metadata says nothing to resume, and reading it would stop recovery for the rest.
36
+ return { ...s, peers: s.peers.filter((p) => !!p && typeof p.peer === "string" && !!p.meta && typeof p.meta === "object") };
37
+ } catch {
38
+ return undefined; // cut short by the crash: nothing to resume from
39
+ }
40
+ }
41
+
42
+ /**
43
+ * Only the run that wrote it removes it: a stop of an older instance must not erase a newer run's record. Without an
44
+ * instance id it goes whatever wrote it (a controlled restart, which has its own record of the peers).
45
+ */
46
+ export function removeSessions(stateDir: string, instanceId?: string): void {
47
+ if (instanceId === undefined || readSessions(stateDir)?.instanceId === instanceId) rmSync(fileOf(stateDir), { force: true });
48
+ }
49
+
50
+ const str = (v: unknown) => (typeof v === "string" && v ? v : undefined);
51
+
52
+ /**
53
+ * What can come back after a crash. `resume` holds the start arguments for peers the hub launches itself (Kimi through
54
+ * ACP `session/load`, Pi through its session file, the local worker afresh); the others say what the user has to do.
55
+ */
56
+ export function crashPlan(records: SessionRecord[]): { peer: string; resume?: Record<string, string>; how: string; fresh?: Record<string, string>; tui?: true }[] {
57
+ return records.map(({ peer, meta }) => {
58
+ const launch = (meta.launch && typeof meta.launch === "object" ? meta.launch : {}) as Record<string, unknown>;
59
+ const model = str(launch.model);
60
+ switch (peer) {
61
+ case "claude":
62
+ return { peer, how: "claude: the Claude Code plugin reconnects by itself while that session is still open" };
63
+ case "codex":
64
+ return { peer, how: `codex: its app-server died with the hub; run ahub codex again${str(meta.threadId) ? ` (its conversation was thread ${meta.threadId})` : ""}` };
65
+ case "kimi": {
66
+ const sessionId = str(meta.sessionId);
67
+ if (!sessionId) return { peer, how: "kimi: no session id was recorded; start it again with ahub kimi" };
68
+ return { peer, resume: { sessionId, ...(model ? { model } : {}) }, how: `kimi: session ${sessionId} can be loaded again (ACP session/load)` };
69
+ }
70
+ case "pi": {
71
+ // `fresh`: what a new headless session keeps of the recorded one (pi.auto_start's fallback, #66).
72
+ const fresh: Record<string, string> = {};
73
+ for (const k of ["backend", "model"] as const) if (str(launch[k])) fresh[k] = str(launch[k])!;
74
+ const sessionFile = str(meta.sessionFile) ?? str(launch.sessionFile);
75
+ if (!sessionFile) return { peer, fresh, how: "pi: no session file was recorded; start it again with ahub pi" };
76
+ // A terminal Pi is run by the CLI that launched it, not by the hub: nothing here could start it again.
77
+ if (str(launch.mode) === "tui") return { peer, fresh, tui: true, how: `pi: it ran in a terminal; start it again with ahub pi --mode tui --session-file ${sessionFile}` };
78
+ const args: Record<string, string> = { sessionFile };
79
+ for (const k of ["mode", "backend", "model"] as const) if (str(launch[k])) args[k] = str(launch[k])!;
80
+ return { peer, fresh, resume: args, how: `pi: its session file can be resumed (${sessionFile})` };
81
+ }
82
+ case "local": {
83
+ // `model` is also recorded as the route's fallback, and a model given at start pins it: pass one or the other.
84
+ const route = str(launch.route);
85
+ const args: Record<string, string> = route ? { route } : model ? { model } : {};
86
+ return { peer, resume: args, how: "local: starts again without its history (the worker keeps none across a hub stop)" };
87
+ }
88
+ default:
89
+ return { peer, how: `${peer}: reconnects by itself if its client is still running` };
90
+ }
91
+ });
92
+ }
93
+
94
+ /**
95
+ * The loss notice for one peer: the deliveries to it that the crash left in needs_review, with their senders and the
96
+ * tasks they were about. `taskTitle` must return the public title (`#n [pii]` for a PII task). No message text.
97
+ */
98
+ export function lossNotice(lost: JournalDelivery[], taskTitle: (id: number) => string | undefined): string {
99
+ const lines = lost.map((d) => {
100
+ const from = [...new Set(d.originals.map((e) => e.from))].join(", ");
101
+ const tasks = [...new Set(d.originals.map((e) => Number(e.refs?.task)).filter((n) => Number.isInteger(n) && n > 0))].map((n) => taskTitle(n) ?? `#${n}`);
102
+ return `- delivery ${d.id} from ${from}${tasks.length ? `, about task ${tasks.join(", ")}` : ""}`;
103
+ });
104
+ return [
105
+ "The hub stopped unexpectedly and was started again. These deliveries to you were in flight when it stopped, so whether you acted on them is not known; the console has since marked each one completed, retried or discarded:",
106
+ ...lines,
107
+ "Check your work against them before you go on.",
108
+ ].join("\n");
109
+ }
package/src/hub/daemon.ts CHANGED
@@ -14,7 +14,8 @@ import { PiPeer } from "../adapters/pi.ts";
14
14
  import { startModelRelay, type ModelRelay } from "../models/relay.ts";
15
15
  import type { MlxOptions } from "../models/mlx.ts";
16
16
  import { PiToolReceipts } from "../pi/tool-receipts.ts";
17
- import { profile } from "../local/sandbox.ts";
17
+ import { profile, proxyEnv, type SandboxNetwork } from "../local/sandbox.ts";
18
+ import { DEFAULT_NETWORK_ALLOW, startEgressProxy, type EgressProxy } from "../local/proxy.ts";
18
19
  import { runTool, TOOL_SCHEMAS, type ToolContext } from "../local/tools.ts";
19
20
  import { LocalPeer } from "../adapters/local-worker.ts";
20
21
  import { Capture, skipTools } from "../memory/capture.ts";
@@ -42,6 +43,8 @@ import { MemoryClient, workerUrl } from "../memory/client.ts";
42
43
  import { VERSION } from "../version.ts";
43
44
  import { projectChain, recallFor } from "../memory/recall.ts";
44
45
  import { conflictsOf } from "./conflicts.ts";
46
+ import { crashPlan, lossNotice, readSessions, removeSessions, writeSessions, type SessionsFile } from "./crash.ts";
47
+ import type { JournalDelivery } from "./delivery-journal.ts";
45
48
  import { DEFAULT_LIMITS, Limiter, PROJECT_LIMITS, type LimitsConfig } from "./limits.ts";
46
49
  import { changedPaths, repoOf, snapshot, Turns } from "./snapshots.ts";
47
50
  import { archiveRestartSnapshot, readRestartSnapshot, removeRestartSnapshot, restartPath, writeRestartSnapshot, type RecoveryPhase, type RestartPeerSnapshot, type RestartSnapshot } from "./restart.ts";
@@ -60,7 +63,9 @@ export interface HubConfig {
60
63
  omniroute: OmniRouteConfig;
61
64
  pi: { enabled: boolean; auto_start: boolean; cmd: string[]; backend: "auto" | "dgx" | "mlx"; dgx_coding: string; dgx_fast: string; max_steps: number };
62
65
  mlx: Pick<MlxOptions, "provider" | "host" | "runtimeDir" | "modelPath" | "port" | "model" | "sourceModel" | "contextWindow" | "maxInputTokens" | "maxTokens" | "maxConcurrency">;
63
- local: { deny: string[]; bash_network: boolean; max_steps: number; read_allow: string[] };
66
+ /** `sandbox`: "deny-default" (issue #39), or "allow-default", the profile of 0.9 and earlier, kept for one release. */
67
+ /** `bash_network`: true is network through the egress proxy to `network_allow` (#65); "direct" is everything, for one release. */
68
+ local: { deny: string[]; bash_network: boolean | "direct"; network_allow: string[]; max_steps: number; read_allow: string[]; sandbox: "deny-default" | "allow-default" };
64
69
  /** Pending permission requests: how long they wait, and whether the desktop is told (issue #5). */
65
70
  approvals: { timeout_s: number; notify: boolean };
66
71
  /** An owner offline this long loses its open tasks back to routing; 0 turns it off (issue #6). */
@@ -71,6 +76,12 @@ export interface HubConfig {
71
76
  snapshots: { enabled: boolean; keep: number };
72
77
  /** Per-sender rate limits and repeat suppression for what agents send (issue #38). */
73
78
  limits: LimitsConfig;
79
+ /** Reviewer choice from recorded review outcomes, once a reviewer has `min_reviews` of an implementer (issue #35). */
80
+ review: { adaptive: boolean; min_reviews: number };
81
+ /** After an unplanned stop, start Kimi, Pi and the local worker again with their recorded sessions (issue #37). */
82
+ recovery: { auto_resume_after_crash: boolean };
83
+ /** Per peer, the hub-tool capabilities it has (issue #39); a peer not listed has all of them. */
84
+ capabilities: Record<string, string[]>;
74
85
  /** Machine-local fields a config file set but git could not vouch for, and why (issue #17). */
75
86
  ignored?: string[];
76
87
  }
@@ -88,7 +99,7 @@ export const DEFAULT_CONFIG: HubConfig = {
88
99
  omniroute: DEFAULT_OMNIROUTE,
89
100
  pi: { enabled: false, auto_start: false, cmd: ["pi"], backend: "auto", dgx_coding: "coding", dgx_fast: "fast", max_steps: 30 },
90
101
  mlx: { provider: "ollama", model: "agenthub-fast-mlx:4b-8k", sourceModel: "qwen3.5:4b-mlx", contextWindow: 8192, maxInputTokens: 6000, maxTokens: 2048, maxConcurrency: 1 },
91
- local: { deny: [], bash_network: false, max_steps: 30, read_allow: [] },
102
+ local: { deny: [], bash_network: false, network_allow: DEFAULT_NETWORK_ALLOW, max_steps: 30, read_allow: [], sandbox: "deny-default" },
92
103
  // Off here, so tests and a hub without a config file stay silent; a project's config defaults it on for macOS.
93
104
  approvals: { timeout_s: 120, notify: false },
94
105
  tasks: { release_after_min: 30 },
@@ -96,6 +107,9 @@ export const DEFAULT_CONFIG: HubConfig = {
96
107
  // Off here like approvals.notify, so tests (whose cwd is this repository) write no objects; a project's config turns it on.
97
108
  snapshots: { enabled: false, keep: 20 },
98
109
  limits: DEFAULT_LIMITS,
110
+ review: { adaptive: false, min_reviews: 5 },
111
+ recovery: { auto_resume_after_crash: false },
112
+ capabilities: {},
99
113
  };
100
114
 
101
115
  export { stateDirFor };
@@ -106,7 +120,7 @@ const PEER_ID = /^[a-z][a-z0-9-]{0,31}$/;
106
120
 
107
121
  /** The shared project config, then the machine's own file, which overrides it block by block (issue #17). */
108
122
  const CONFIG_FILES = ["config.json", "config.local.json"] as const;
109
- const CONFIG_BLOCKS = ["memory", "roles", "budget", "inference", "omniroute", "local", "pi", "approvals", "tasks", "checks", "snapshots", "limits", "mlx"];
123
+ const CONFIG_BLOCKS = ["memory", "roles", "budget", "inference", "omniroute", "local", "pi", "approvals", "tasks", "checks", "snapshots", "limits", "review", "recovery", "capabilities", "mlx"];
110
124
 
111
125
  export function loadConfig(cwd: string): HubConfig {
112
126
  const ignored: string[] = [];
@@ -151,6 +165,9 @@ export function loadConfig(cwd: string): HubConfig {
151
165
  checks: { ...DEFAULT_CONFIG.checks, ...file.checks },
152
166
  snapshots: { ...DEFAULT_CONFIG.snapshots, enabled: true, ...file.snapshots },
153
167
  limits: { ...PROJECT_LIMITS, ...file.limits }, // on with any project config (issue #38)
168
+ review: { ...DEFAULT_CONFIG.review, ...file.review },
169
+ recovery: { ...DEFAULT_CONFIG.recovery, ...file.recovery },
170
+ capabilities: { ...file.capabilities },
154
171
  mlx,
155
172
  ...(ignored.length ? { ignored } : {}),
156
173
  };
@@ -280,9 +297,36 @@ export async function startDaemon(opts: DaemonOptions) {
280
297
  : undefined;
281
298
  if ((restartFilePresent && !restored) || (recoveryOperation && !restored)) throw new Error("restart state is unreadable, missing, or does not match this project and recovery operation");
282
299
 
300
+ // A session record left by a run that never shut down means it crashed (issue #37). A controlled restart has its own.
301
+ const crashed = !recoveryOperation && !restartFilePresent ? readSessions(opts.stateDir) : undefined;
302
+ // A controlled restart's source may have been cut short before it removed its record: this run is not a crash, and
303
+ // a record left now would make the next ordinary start look like one.
304
+ if (recoveryOperation || restartFilePresent) try { removeSessions(opts.stateDir); } catch { /* nothing to remove */ }
305
+ const autoResume = config.recovery.auto_resume_after_crash === true; // a string "false" is not a yes
306
+ const piAutoStart = config.pi.enabled && config.pi.auto_start;
307
+ const startedAt = Date.now();
308
+ /** What crash recovery did or asks the user to do, for `ahub status`. */
309
+ const crashReport: string[] = [];
310
+
283
311
  // The hub's own model calls (digest condensation, task triage) are wired below, once the gateway client exists.
284
312
  let inference: Inference | undefined;
285
313
  const journal = new DeliveryJournal({ file: join(opts.stateDir, "hub.db"), projectRoot: opts.cwd, projectId, instanceId, operationId: recoveryOperation });
314
+ // What the crash left in flight, per recipient: opening the journal just marked these needs_review.
315
+ const lost = new Map<PeerId, JournalDelivery[]>();
316
+ if (crashed) for (const d of journal.list()) if (d.state === "needs_review" && d.reason === "daemon stopped during delivery" && d.updatedAt >= startedAt) lost.set(d.peer, [...(lost.get(d.peer) ?? []), d]);
317
+ // Capabilities (issue #39): enforced here and in taskOp, never by role text alone. Unlisted peers keep everything.
318
+ // A listed peer whose value is not a list gets nothing: whoever listed it meant to narrow it.
319
+ const may = (peer: PeerId, cap: "propose" | "assign" | "remember" | "important"): boolean => {
320
+ if (peer === USER || peer === HUB || !Object.hasOwn(config.capabilities, peer)) return true;
321
+ const list = config.capabilities[peer];
322
+ return Array.isArray(list) && list.includes(cap);
323
+ };
324
+ const CAPABILITIES = ["propose", "assign", "remember", "important"];
325
+ for (const [peer, list] of Object.entries(config.capabilities)) {
326
+ if (!PEER_ID.test(peer)) log(`capabilities.${peer} is not a peer id; ignored (capabilities is an object of lists, one per peer)`);
327
+ else if (!Array.isArray(list)) log(`capabilities.${peer} is not a list: ${peer} gets no capabilities`);
328
+ else for (const c of list) if (!CAPABILITIES.includes(c)) log(`capabilities.${peer}: ${JSON.stringify(c)} is not a capability (${CAPABILITIES.join(", ")}); it grants nothing`);
329
+ }
286
330
  // Agents only: the console user and the hub itself are never limited (issue #38).
287
331
  // A typo such as "12/min" would read as 0, which turns a limit off without a word: the project default instead.
288
332
  for (const k of Object.keys(config.limits)) if (!(k in PROJECT_LIMITS)) log(`limits.${k} is not a known limit; ignored`);
@@ -293,9 +337,23 @@ export async function startDaemon(opts: DaemonOptions) {
293
337
  return [k, fallback];
294
338
  })) as unknown as LimitsConfig;
295
339
  const limiter = new Limiter(limits);
340
+ // The local worker's and Pi's commands reach the network only through this proxy (issue #65); "direct" keeps the
341
+ // open network of 0.10 and earlier for one release, anything else means none.
342
+ let egress: EgressProxy | undefined;
343
+ if (config.local.bash_network === true) {
344
+ const allow = Array.isArray(config.local.network_allow) ? config.local.network_allow.filter((h): h is string => typeof h === "string") : DEFAULT_NETWORK_ALLOW;
345
+ egress = await startEgressProxy({ allow, log });
346
+ startupCleanup.push(() => void egress?.close());
347
+ log(`network: egress proxy on 127.0.0.1:${egress.port}, ${allow.length} allowed host(s) (local.network_allow)`);
348
+ }
349
+ const sandboxNetwork: SandboxNetwork = config.local.bash_network === "direct" ? true : egress ? { proxyPort: egress.port } : false;
296
350
  const admit = (env: Envelope, parent?: string): string | undefined => {
297
351
  // [FYI] is recorded and costs nobody a turn: nothing to limit.
298
352
  if (env.from === USER || env.from === HUB || env.from === DIGEST || env.priority === "fyi") return undefined;
353
+ if (env.priority === "important" && !may(env.from, "important")) {
354
+ log(`capabilities: ${env.from} may not send important messages`);
355
+ return `${env.from} may not send important messages (no "important" in capabilities.${env.from}): send it without [IMPORTANT]`;
356
+ }
299
357
  const refused = limiter.admit(env.from, env.to, env.priority, env.body, parent);
300
358
  if (refused) log(`limits: ${env.from}: ${refused}`);
301
359
  return refused;
@@ -410,6 +468,7 @@ export async function startDaemon(opts: DaemonOptions) {
410
468
  },
411
469
  triage: { classify: (title, detail) => inference?.triage(title, detail) ?? Promise.resolve(undefined), onCampus: () => onCampus() },
412
470
  quota: (): ReturnType<Budget["headroom"]> => budget.headroom(), // budget is built below; this runs at assignment time
471
+ review: config.review,
413
472
  });
414
473
  board.onChange = (t, h) => event({ type: "task", id: t.id, event: h.event, by: h.by, state: t.state, owner: t.owner, reviewer: t.reviewer, class: t.class, pii: tasks.isPii(t) });
415
474
  // ---- budget relay -------------------------------------------------------------------------------------------
@@ -555,6 +614,13 @@ export async function startDaemon(opts: DaemonOptions) {
555
614
  taskOpsInFlight--;
556
615
  }
557
616
  };
617
+ /** A model-written peer argument: a peer id, or nothing for absent, null or "". Anything else is refused. */
618
+ const peerArg = (v: unknown, name: string): PeerId | undefined => {
619
+ if (v == null || v === "") return undefined;
620
+ const id = typeof v === "string" ? v.trim().toLowerCase() : undefined; // peer ids are lowercase; "Codex" means codex
621
+ if (!id || !PEER_ID.test(id)) throw new Error(`${name} must be a peer id, not ${JSON.stringify(v).slice(0, 60)}`);
622
+ return id;
623
+ };
558
624
  async function taskOpBody(by: PeerId, op: string, a: Record<string, any>, inProcess = false, piiTurn = false): Promise<string> {
559
625
  // Inside a PII turn the worker's words may carry the PII whatever they are attached to: a note would go to
560
626
  // claude-mem (a cloud observer) and a new task could be routed to a cloud peer without matching any pattern.
@@ -570,6 +636,18 @@ export async function startDaemon(opts: DaemonOptions) {
570
636
  // console reads a PII task's text deliberately, with `ahub task show <id>`.
571
637
  const onPrem = inProcess && by === "local";
572
638
  const line = (t: { id: number; state: string; owner: PeerId | null; reviewer: PeerId | null }) => `task #${t.id}: ${t.state}, owner ${t.owner ?? "none"}, reviewer ${t.reviewer ?? "none"}`;
639
+ const need = (cap: "propose" | "assign" | "remember", what: string) => {
640
+ if (may(by, cap)) return;
641
+ log(`capabilities: ${by} may not ${what} (${op})`);
642
+ throw new Error(`${by} may not ${what} (no "${cap}" in capabilities.${by} in .agenthub/config.json)`);
643
+ };
644
+ // Tool callers are models (#70): `owner` is a peer id or nothing, settled before anything reaches the board.
645
+ if (op === "hub_task_propose") {
646
+ a = { ...a, owner: peerArg(a.owner, "owner") };
647
+ need("propose", "propose tasks");
648
+ if (a.owner && a.owner !== by) need("assign", "hand tasks to other peers");
649
+ }
650
+ if (op === "hub_remember") need("remember", "save notes to shared memory");
573
651
  switch (op) {
574
652
  case "hub_task_propose": {
575
653
  const t = await tasks.propose(by, a);
@@ -588,7 +666,7 @@ export async function startDaemon(opts: DaemonOptions) {
588
666
  return tasks.isChecking(t.id) ? `${line(t)}; its check is queued or running, and the result comes as a task message` : line(t);
589
667
  }
590
668
  case "hub_review":
591
- return line(await tasks.review(by, a.id, a.verdict, a.note));
669
+ return line(await tasks.review(by, a.id, a.verdict, a.note, a.unmet));
592
670
  case "hub_remember":
593
671
  return tasks.remember(by, a);
594
672
  case "hub_checkpoint": {
@@ -604,9 +682,12 @@ export async function startDaemon(opts: DaemonOptions) {
604
682
  if (by !== USER) throw new Error(`${op} is a console command`);
605
683
  switch (op) {
606
684
  case "task_show":
607
- return JSON.stringify(board.get(Number(a.id)) ?? `no task #${a.id}`, null, 2);
608
- case "task_assign":
609
- return line(await tasks.assignTo(a.id, String(a.peer)));
685
+ return JSON.stringify(board.get(Number(a.id)) ? { ...board.get(Number(a.id)), reviews: board.reviews({ task: Number(a.id) }) } : `no task #${a.id}`, null, 2);
686
+ case "task_assign": {
687
+ const peer = peerArg(a.peer, "peer");
688
+ if (!peer) throw new Error("peer is required");
689
+ return line(await tasks.assignTo(a.id, peer));
690
+ }
610
691
  case "task_escalate":
611
692
  return line(await tasks.escalate(USER, a.id));
612
693
  case "route_explain":
@@ -735,6 +816,7 @@ export async function startDaemon(opts: DaemonOptions) {
735
816
  return now;
736
817
  };
737
818
  const status = () => ({
819
+ ...(crashReport.length ? { crash: crashReport } : {}),
738
820
  projectId,
739
821
  instanceId,
740
822
  version: VERSION,
@@ -833,6 +915,49 @@ export async function startDaemon(opts: DaemonOptions) {
833
915
  }
834
916
  };
835
917
 
918
+ // Session identities of the attached peers, kept current for crash recovery (issue #37); a clean stop removes them.
919
+ let sessionsWritten = "";
920
+ const recordSessions = () => {
921
+ if (stopping) return;
922
+ const peers = [...bus.peers.values()].filter((p) => p.state !== "offline").map((p) => {
923
+ let meta: Record<string, unknown> = {};
924
+ try { meta = (p as { recoveryMetadata?: () => Record<string, unknown> }).recoveryMetadata?.() ?? {}; } catch { /* not ready yet: the id alone */ }
925
+ return { peer: p.id, meta };
926
+ });
927
+ const text = JSON.stringify(peers);
928
+ if (text === sessionsWritten) return;
929
+ sessionsWritten = text;
930
+ try { writeSessions(opts.stateDir, { instanceId, at: Date.now(), peers }); } catch (error) { log(`session record not written: ${(error as Error).message}`); }
931
+ };
932
+ const recoverAfterCrash = async (prev: SessionsFile) => {
933
+ const report = (line: string) => (crashReport.push(line), notify(`crash recovery: ${line}`));
934
+ const lostCount = [...lost.values()].reduce((n, l) => n + l.length, 0);
935
+ report(`the previous hub run stopped without shutting down${lostCount ? `; ${lostCount} deliveries it had in flight are in needs_review (ahub queue list)` : ""}`);
936
+ const start = (peer: string, args: Parameters<typeof startPeer>[1]) => startPeer(peer, args).catch((e: Error) => ({ ok: false, error: e.message }));
937
+ for (const step of crashPlan(prev.peers)) {
938
+ // `pi.auto_start` already asks for Pi (#66): its recorded headless session comes back, auto-resume or not, and a
939
+ // fresh one starts when that fails or there is nothing headless to resume. Pi runs on-prem: no cloud quota.
940
+ if (step.peer === "pi" && piAutoStart) {
941
+ if (step.resume) {
942
+ const r = await start("pi", step.resume as Parameters<typeof startPeer>[1]);
943
+ if (r.ok) { report(`pi resumed (pi.auto_start): ${step.how}`); continue; }
944
+ report(`pi not resumed (${String(r.error)}); ${step.how}`);
945
+ } else report(step.how);
946
+ const fresh = await start("pi", { ...(step.fresh as Parameters<typeof startPeer>[1]), fresh: true });
947
+ const back = step.tui ? `; to go back to the recorded session, run ahub kill, start the hub without pi.auto_start, then the command above` : "";
948
+ report(fresh.ok ? `pi.auto_start started a fresh session${back}` : `pi.auto_start could not start Pi either (${String(fresh.error)})`);
949
+ continue;
950
+ }
951
+ if (!step.resume || !autoResume) {
952
+ report(`${step.how}${step.resume ? " (recovery.auto_resume_after_crash is off)" : ""}`);
953
+ continue;
954
+ }
955
+ const r = await start(step.peer, step.resume as Parameters<typeof startPeer>[1]);
956
+ report(r.ok ? `${step.peer} resumed: ${step.how}` : `${step.peer} not resumed (${String(r.error)}); ${step.how}`);
957
+ }
958
+ writeStatus();
959
+ };
960
+
836
961
  bus.tap((e) => {
837
962
  e = redact(e);
838
963
  uiEvents.push({ seq: ++uiSequence, event: e });
@@ -873,6 +998,13 @@ export async function startDaemon(opts: DaemonOptions) {
873
998
  }
874
999
  if (e.state === "offline") offlineSince.set(e.peer, offlineSince.get(e.peer) ?? Date.now());
875
1000
  else offlineSince.delete(e.peer);
1001
+ // After a crash, a peer's first attach brings the loss notice: it leads its next delivery (issue #37).
1002
+ if (e.state !== "offline" && lost.has(e.peer)) {
1003
+ const still = lost.get(e.peer)!.filter((d) => { try { return journal.get(d.id)?.state === "needs_review"; } catch { return false; } });
1004
+ lost.delete(e.peer);
1005
+ if (still.length) bus.preface(e.peer, lossNotice(still, (id) => { const t = board.get(id); return t ? tasks.publicTitle(t) : undefined; }));
1006
+ }
1007
+ recordSessions();
876
1008
  }
877
1009
  else if (e.t === "undeliverable" || e.t === "overflow") {
878
1010
  log(e.t === "undeliverable" ? `UNDELIVERABLE to ${e.peer} after retries: ${e.env.id} from ${e.env.from}` : `OVERFLOW ${e.peer}: dropped ${e.env.id} from ${e.env.from}`);
@@ -944,7 +1076,7 @@ export async function startDaemon(opts: DaemonOptions) {
944
1076
 
945
1077
  // One start per peer at a time: a second `ahub codex` must not tear down an adapter that is still coming up.
946
1078
  const starting = new Map<string, Promise<Record<string, unknown>>>();
947
- function startPeer(peer: string, args: { model?: string; route?: string; mode?: "headless" | "tui"; backend?: "auto" | "dgx" | "mlx"; sessionId?: string; sessionFile?: string }): Promise<Record<string, unknown>> {
1079
+ function startPeer(peer: string, args: { model?: string; route?: string; mode?: "headless" | "tui"; backend?: "auto" | "dgx" | "mlx"; sessionId?: string; sessionFile?: string; fresh?: boolean }): Promise<Record<string, unknown>> {
948
1080
  if (stopping) return Promise.resolve({ ok: false, error: "hub is stopping" });
949
1081
  if (peer === "pi" && starting.has(peer)) return Promise.resolve({ ok: false, error: "Pi start is in progress; inspect status before retrying" });
950
1082
  const running = starting.get(peer) ?? startPeerOnce(peer, args).finally(() => starting.delete(peer));
@@ -968,7 +1100,7 @@ export async function startDaemon(opts: DaemonOptions) {
968
1100
  };
969
1101
  }
970
1102
 
971
- async function startPeerOnce(peer: string, args: { model?: string; route?: string; mode?: "headless" | "tui"; backend?: "auto" | "dgx" | "mlx"; sessionId?: string; sessionFile?: string }): Promise<Record<string, unknown>> {
1103
+ async function startPeerOnce(peer: string, args: { model?: string; route?: string; mode?: "headless" | "tui"; backend?: "auto" | "dgx" | "mlx"; sessionId?: string; sessionFile?: string; fresh?: boolean }): Promise<Record<string, unknown>> {
972
1104
  let unmute: (() => void) | undefined;
973
1105
  try {
974
1106
  return await startPeerBody(peer, args, (p) => { unmute = muteState(p); });
@@ -977,7 +1109,7 @@ export async function startDaemon(opts: DaemonOptions) {
977
1109
  }
978
1110
  }
979
1111
 
980
- async function startPeerBody(peer: string, args: { model?: string; route?: string; mode?: "headless" | "tui"; backend?: "auto" | "dgx" | "mlx"; sessionId?: string; sessionFile?: string }, mute: (p: PeerAdapter) => void): Promise<Record<string, unknown>> {
1112
+ async function startPeerBody(peer: string, args: { model?: string; route?: string; mode?: "headless" | "tui"; backend?: "auto" | "dgx" | "mlx"; sessionId?: string; sessionFile?: string; fresh?: boolean }, mute: (p: PeerAdapter) => void): Promise<Record<string, unknown>> {
981
1113
  if (peer === "pi") {
982
1114
  if (args.mode !== undefined && !["headless", "tui"].includes(args.mode)) return { ok: false, error: "invalid Pi mode" };
983
1115
  if (args.backend !== undefined && !["auto", "dgx", "mlx"].includes(args.backend)) return { ok: false, error: "invalid Pi backend" };
@@ -993,7 +1125,8 @@ export async function startDaemon(opts: DaemonOptions) {
993
1125
  const unclaimed = existing.state === "offline" && !saved.sessionId && !saved.sessionFile;
994
1126
  if (changesOwner && (existing.state === "busy" || (existing.state !== "offline" && !existing.recoveryReady))) return { ok: false, error: "Pi is busy; wait for agent_settled before changing mode/backend" };
995
1127
  if (unclaimed) {
996
- if (!args.sessionId && !args.sessionFile) args = { ...args, ...existing.pendingResume };
1128
+ // `fresh` (crash recovery's fallback, #66): the pending session is the one that just failed to load.
1129
+ if (!args.sessionId && !args.sessionFile && !args.fresh) args = { ...args, ...existing.pendingResume };
997
1130
  // Revoke the previous launch bridge before issuing another launch. A late
998
1131
  // process from the abandoned CLI cannot claim the replacement owner.
999
1132
  // The replacement's start event is the only state change the console should see (issue #42).
@@ -1023,6 +1156,7 @@ export async function startDaemon(opts: DaemonOptions) {
1023
1156
  const kimi = new AcpPeer("kimi", {
1024
1157
  cmd,
1025
1158
  ...(args.model ? { launchModel: args.model } : {}),
1159
+ ...(args.sessionId ? { resumeSessionId: args.sessionId } : {}),
1026
1160
  cwd: opts.cwd,
1027
1161
  watchdogMs: config.watchdog_ms,
1028
1162
  onPermission,
@@ -1082,7 +1216,8 @@ export async function startDaemon(opts: DaemonOptions) {
1082
1216
  let piReply: Envelope | undefined;
1083
1217
  const ctx: ToolContext = {
1084
1218
  cwd: opts.cwd, deny: config.local.deny,
1085
- sandboxProfile: profile(opts.cwd, config.local.bash_network, config.local.read_allow, config.local.deny),
1219
+ sandboxProfile: profile(opts.cwd, sandboxNetwork, config.local.read_allow, config.local.deny, config.local.sandbox === "allow-default" ? "allow" : "deny"),
1220
+ sandboxEnv: proxyEnv(sandboxNetwork),
1086
1221
  permit: (title) => onPermission({ peer: "pi", title, options: [{ optionId: "allow", name: "Allow", kind: "allow_once" }, { optionId: "deny", name: "Deny", kind: "reject_once" }] }).then((picked) => picked === "allow" && pi.acceptingTools && bus.peers.get("pi") === pi),
1087
1222
  send: (text, to) => {
1088
1223
  if (to?.some((id) => !bus.peers.has(id) && id !== USER)) return "error: unknown peer";
@@ -1173,7 +1308,7 @@ export async function startDaemon(opts: DaemonOptions) {
1173
1308
  omni,
1174
1309
  ...(sidecar && route ? { sidecar, route } : {}),
1175
1310
  fixedModel: args.model ?? routing.local.fixed_model,
1176
- tools: { deny: config.local.deny, bashNetwork: config.local.bash_network, readAllow: config.local.read_allow, permit },
1311
+ tools: { deny: config.local.deny, bashNetwork: sandboxNetwork, readAllow: config.local.read_allow, sandbox: config.local.sandbox, permit },
1177
1312
  ...(capture ? { capture } : {}),
1178
1313
  taskTool: (name, a, turn) => taskOp("local", name, a, true, turn.pii),
1179
1314
  turnPolicy: (envs) => tasks.turnPolicy(envs),
@@ -1619,6 +1754,8 @@ export async function startDaemon(opts: DaemonOptions) {
1619
1754
  }
1620
1755
  async function stopOnce(): Promise<void> {
1621
1756
  stopping = true;
1757
+ // A stop someone asked for is not a crash, even if it then runs past the shutdown deadline: forget the sessions now.
1758
+ try { removeSessions(opts.stateDir, instanceId); } catch { /* the state dir is gone */ }
1622
1759
  checksClosed = true;
1623
1760
  for (const kill of runningChecks) kill();
1624
1761
  log("hub stopping");
@@ -1636,6 +1773,7 @@ export async function startDaemon(opts: DaemonOptions) {
1636
1773
  // logged and shutdown finishes anyway, or the process ignores SIGTERM forever.
1637
1774
  for (const exit of exits) if (exit.status === "rejected") log(`peer stop failed during shutdown: ${(exit.reason as Error)?.message ?? exit.reason}`);
1638
1775
  await modelRelay?.close();
1776
+ await egress?.close();
1639
1777
  await piReceipts?.close();
1640
1778
  budget.close();
1641
1779
  board.close();
@@ -1674,7 +1812,18 @@ export async function startDaemon(opts: DaemonOptions) {
1674
1812
  writeStatus();
1675
1813
  log(`${RUN_START}${process.pid} control=127.0.0.1:${server.port} cwd=${opts.cwd}`);
1676
1814
  ready = true;
1677
- if (config.pi.enabled && config.pi.auto_start && !recoveryActive()) void startPeer("pi", {}).catch((error) => log(`Pi auto-start failed: ${error.message}`));
1815
+ if (crashed) {
1816
+ // This run owns the record now, so its clean stop removes it even if no peer attaches to rewrite it.
1817
+ try { if (readSessions(opts.stateDir)?.instanceId === crashed.instanceId) writeSessions(opts.stateDir, { ...crashed, instanceId, at: Date.now() }); } catch (error) { log(`session record not adopted: ${(error as Error).message}`); }
1818
+ void recoverAfterCrash(crashed).catch((error) => {
1819
+ log(`crash recovery failed: ${(error as Error).message}`);
1820
+ // pi.auto_start still holds: a running Pi answers `already`, a starting one refuses the second start.
1821
+ if (piAutoStart) void startPeer("pi", {}).catch((e) => log(`Pi auto-start failed: ${e.message}`));
1822
+ });
1823
+ }
1824
+ // After a crash that recorded Pi, crash recovery starts it (#66): its recorded session first, a fresh one if that fails.
1825
+ const piRecovered = !!crashed?.peers.some((p) => p.peer === "pi");
1826
+ if (piAutoStart && !recoveryActive() && !piRecovered) void startPeer("pi", {}).catch((error) => log(`Pi auto-start failed: ${error.message}`));
1678
1827
  return { bus, token, port: server.port as number, stop, stopped: new Promise<void>((r) => (onStop = r)) };
1679
1828
  } finally {
1680
1829
  if (!ready) for (const cleanup of startupCleanup.reverse()) { try { cleanup(); } catch { /* preserve startup error */ } }
@@ -40,7 +40,7 @@ export const TASK_TOOLS: HubTool[] = [
40
40
  tool("hub_task_decline", "Pass on a task assigned to you; the hub offers it to the next peer.", { id, reason: str }, ["id"]),
41
41
  tool("hub_task_done", "Mark your task finished. It goes to its reviewer with your summary and refs. When the project configures a check for its class, the hub runs it first and the result comes as a task message: a failed check keeps the task with you.", { id, summary: { type: "string", description: "what changed, why, and the check you ran with its result" }, refs }, ["id", "summary"]),
42
42
  tool("hub_task_list", "The task board. PII tasks show as [pii].", { state: { type: "string", enum: ["proposed", "in_progress", "in_review", "approved", "changes_requested"] }, ready: { type: "boolean", description: "only proposed tasks with nothing left to wait for" } }),
43
- tool("hub_review", "Give your verdict on a task you were asked to review. Two changes_requested in a row move the task to another peer.", { id, verdict: { type: "string", enum: ["approved", "changes_requested"] }, note: str }, ["id", "verdict"]),
43
+ tool("hub_review", "Give your verdict on a task you were asked to review: map the changed signatures and call sites to the task's plan or detail, read the check result, and list what is unmet. Two changes_requested in a row move the task to another peer.", { id, verdict: { type: "string", enum: ["approved", "changes_requested"] }, note: str, unmet: { type: "array", items: str, description: "each requirement of the plan or detail that the change does not meet" } }, ["id", "verdict"]),
44
44
  tool("hub_checkpoint", "Answer a checkpoint request from the hub (your quota window is nearly used up): what you were doing, what is half done, what whoever continues must know. Write the same to .agenthub/checkpoint.md first if you can.", { summary: str }, ["summary"]),
45
45
  tool("hub_remember", "Save a decision, finding, contract or fail to the memory all agents share (claude-mem); the other agents also get it with their next message. A fail is an approach you tried that does not work, and why: the most useful note, it stops the others spending their quota on it. Do not retry what a fail note rules out without new evidence. Conclusions worth recalling, not chatter.", { text: str, title: str, kind: { type: "string", enum: [...NOTE_KINDS] }, task: id }, ["text"]),
46
46
  ];