viber-channel 0.8.15 → 0.8.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,8 +17,9 @@
17
17
  * MCP tool-result shape. It does NOT hold state.
18
18
  */
19
19
  import { postMessage, parseArtifact, type Artifact } from "./messages.js";
20
- import { listPeersAuto, openDm, type OpenDmResult } from "./peers.js";
20
+ import { listPeersAuto, openDm, type OpenDmResult, type PeerFilters } from "./peers.js";
21
21
  import { ConversationTokenExpiredError } from "./messages.js";
22
+ import type { CallToolResult } from "@modelcontextprotocol/sdk/types.js";
22
23
 
23
24
  /** MCP tool-result shape (text content + optional error flag). */
24
25
  export interface AgentToolResult {
@@ -26,6 +27,28 @@ export interface AgentToolResult {
26
27
  content: { type: "text"; text: string }[];
27
28
  }
28
29
 
30
+ /**
31
+ * #489 step-04 — adapt a tool result to what the MCP SDK's handler signature wants.
32
+ *
33
+ * The SDK's `CallToolResult` carries an index signature (the MCP result object is
34
+ * open: `_meta` and future fields pass through), so `AgentToolResult` was NOT
35
+ * assignable to the SDK handler type — the TS2345 at `lib/bridge_tool_host.ts` and
36
+ * `viber-channel.ts`. Verified against the SDK schema, not against our output:
37
+ * `{ content: [{ type: "text", text }], isError }` IS a valid `CallToolResult`. The
38
+ * missing openness was the ONLY incompatibility, so there is no protocol divergence.
39
+ *
40
+ * The openness lives HERE, at the SDK boundary, and NOT on `AgentToolResult` — review
41
+ * measured that opening our own interface silently disables excess-property checking
42
+ * on it, so `{ content: […], isErorr: true }` would compile. Keeping our type closed
43
+ * preserves that check at every construction site; the single unavoidable widening is
44
+ * named, commented, and confined to this function.
45
+ */
46
+ export function toMcpToolResult(result: AgentToolResult): CallToolResult {
47
+ // No cast: the spread is a FRESH object type, which TypeScript grants an implicit
48
+ // index signature when assigning to the SDK's open result type.
49
+ return { ...result };
50
+ }
51
+
29
52
  /**
30
53
  * Live, by-reference access to the host channel's runtime state. Implemented by
31
54
  * both hosts so the shared tool logic never reads a stale snapshot. Every getter
@@ -75,12 +98,27 @@ function errorText(s: string): AgentToolResult {
75
98
  * entry then carries a `project` field); a regular instance transparently
76
99
  * falls back to its own project (unchanged behaviour, no `project` field).
77
100
  */
78
- export async function listAgents(ctx: AgentToolsContext): Promise<AgentToolResult> {
101
+ export async function listAgents(
102
+ ctx: AgentToolsContext,
103
+ args: Record<string, unknown> = {},
104
+ ): Promise<AgentToolResult> {
79
105
  if (!ctx.instanceToken()) {
80
106
  return errorText("Channel not ready: no instance identity yet.");
81
107
  }
108
+ // #501 step-05 — optional filters, applied SERVER-side. Without them a
109
+ // dev-lead looking for its two reviewers pulls every agent the project ever
110
+ // created (30 rows for 5 live ones, measured), in a loop.
111
+ const filters: PeerFilters = {};
112
+ if (args.online === true) filters.online = true;
113
+ if (typeof args.label_prefix === "string" && args.label_prefix !== "") {
114
+ filters.labelPrefix = args.label_prefix;
115
+ }
82
116
  try {
83
- const { peers, scope } = await listPeersAuto(ctx.baseUrl(), ctx.instanceToken());
117
+ const { peers, scope, presenceStatus } = await listPeersAuto(
118
+ ctx.baseUrl(),
119
+ ctx.instanceToken(),
120
+ filters,
121
+ );
84
122
  // Project only what the model needs to pick a peer (drop last_seen /
85
123
  // active_conversation_id — available over the wire if a future tool needs them).
86
124
  const summary = peers.map((p) => ({
@@ -91,13 +129,22 @@ export async function listAgents(ctx: AgentToolsContext): Promise<AgentToolResul
91
129
  // Present only in the user-scoped (orchestrator) listing.
92
130
  ...(p.project_name !== undefined ? { project: p.project_name } : {}),
93
131
  }));
132
+ // An `online` filter under an unreadable presence source would come back
133
+ // empty and read as "nobody is alive" — the false green in listing form.
134
+ // The server drops the filter in that case; say so rather than let the
135
+ // caller believe it was applied.
136
+ const presenceWarning =
137
+ filters.online === true && presenceStatus === "unavailable"
138
+ ? "WARNING: the presence source is unavailable, so the `online` filter was NOT applied — " +
139
+ "`online` values below are unknown, not observed.\n\n"
140
+ : "";
94
141
  const body =
95
142
  summary.length === 0
96
143
  ? scope === "user"
97
144
  ? "No other agents are currently registered in any of your projects."
98
145
  : "No other agents are currently registered in this project."
99
146
  : JSON.stringify(summary, null, 2);
100
- return text(body);
147
+ return text(`${presenceWarning}${body}`);
101
148
  } catch (err) {
102
149
  return errorText(`list_agents failed: ${String(err)}`);
103
150
  }
@@ -16,6 +16,7 @@
16
16
  import { mkdirSync, readFileSync, unlinkSync, writeFileSync } from "node:fs";
17
17
  import { createTokenRefreshScheduler, type TokenRefreshScheduler } from "./token_refresh.js";
18
18
  import type { ApiBaseUrlResolver } from "./base_urls.js";
19
+ import { acquireLockFile, isProcessAlive } from "./bridge_lock.js";
19
20
  import { lockFilePath } from "./lockfile.js";
20
21
  import {
21
22
  type ConversationMintResponse,
@@ -159,24 +160,24 @@ export function planAwaitInviteFirstPass<T>(msgs: T[]): { forward: T[]; seed: T[
159
160
  return { forward: [msgs[msgs.length - 1]], seed: msgs.slice(0, -1) };
160
161
  }
161
162
 
162
- /** True if a PID names a live process (EPERM counts as alive). */
163
- export function isProcessAlive(pid: number): boolean {
164
- if (!Number.isInteger(pid) || pid <= 0) return false;
165
- try {
166
- process.kill(pid, 0);
167
- return true;
168
- } catch (err) {
169
- if (typeof err === "object" && err !== null && "code" in err && err.code === "EPERM") return true;
170
- return false;
171
- }
172
- }
163
+ /**
164
+ * #489 step-02: re-exported from `lib/bridge_lock.ts`, which is now the single
165
+ * implementation. Kept exported here because callers and tests already import it
166
+ * from this module.
167
+ */
168
+ export { isProcessAlive };
173
169
 
174
170
  /**
175
- * Acquire the per-agent bridge lock (a PID file). `sessionId` is the identity
176
- * axis (e.g. `codex-agent:<instanceId>` / `gemma-agent:<instanceId>`), so two
177
- * processes with the SAME identity cannot both run. A stale lock (dead PID) is
178
- * reclaimed; a live one throws a BridgeShutdownError. `logPrefix` keeps each
179
- * bridge's log lines under its own tag.
171
+ * Acquire the per-agent bridge lock (a PID file). `sessionId` is the identity axis
172
+ * (e.g. `codex-agent:<instanceId>` / `gemma-agent:<instanceId>`), so two processes
173
+ * with the SAME identity cannot both run. A stale lock (dead PID) is reclaimed; a
174
+ * live one throws a BridgeShutdownError. `logPrefix` keeps each bridge's log lines
175
+ * under its own tag.
176
+ *
177
+ * #489 step-02: a thin adapter now. The loop, the payload and the reclaim decision
178
+ * live in `lib/bridge_lock.ts`, shared with the MCP client — the duplication was one
179
+ * of the defects of #489. What stays HERE is this site's error protocol: throw, so
180
+ * the bridge dies rather than reporting a pid to a human.
180
181
  */
181
182
  export function acquireBridgeLock(opts: {
182
183
  /**
@@ -190,49 +191,37 @@ export function acquireBridgeLock(opts: {
190
191
  sessionId: string;
191
192
  lockDir: string;
192
193
  logPrefix: string;
194
+ /** Called if the lock is lost mid-run — the caller tears the bridge down. */
195
+ onLost?: (reason: string) => void;
193
196
  }): BridgeLock {
194
197
  const { identityBaseUrl, fingerprint, sessionId, lockDir, logPrefix } = opts;
195
- mkdirSync(lockDir, { recursive: true });
196
198
  const path = lockFilePath(identityBaseUrl, fingerprint, sessionId, lockDir);
197
- const release = (): void => {
198
- try {
199
- unlinkSync(path);
200
- } catch {
201
- // Best effort: the file may already be gone.
202
- }
203
- };
204
-
205
- for (let attempt = 0; attempt < 2; attempt++) {
206
- try {
207
- writeFileSync(path, `${process.pid}\n`, { flag: "wx" });
208
- process.stderr.write(`${logPrefix} lock acquired: ${path}\n`);
209
- return { path, release };
210
- } catch (err) {
211
- if (typeof err !== "object" || err === null || !("code" in err) || err.code !== "EEXIST") {
212
- throw err;
213
- }
214
-
215
- let existingPid = Number.NaN;
216
- try {
217
- existingPid = Number.parseInt(readFileSync(path, "utf-8").trim(), 10);
218
- } catch {
219
- release();
220
- continue;
221
- }
222
-
223
- if (!Number.isFinite(existingPid) || !isProcessAlive(existingPid)) {
224
- release();
225
- continue;
226
- }
227
-
228
- throw new BridgeShutdownError(
229
- `Another bridge already holds the lock ${sessionId} (PID ${existingPid})`,
230
- 1,
231
- );
232
- }
199
+ const result = acquireLockFile({
200
+ lockFile: path,
201
+ // #489 step-03 (review): losing the lock is a FENCING signal, not a log line. A
202
+ // dispossessed bridge that keeps running has no authority over its identity, which
203
+ // is the unbounded duplicate this design promises not to create. Withdraw the same
204
+ // way the parent watchdog does: request shutdown and let the normal teardown run.
205
+ onLost: (reason) => {
206
+ process.stderr.write(`${logPrefix} ${reason} — withdrawing
207
+ `);
208
+ opts.onLost?.(reason);
209
+ },
210
+ });
211
+ if (result.ok) {
212
+ process.stderr.write(`${logPrefix} lock acquired: ${path}
213
+ `);
214
+ // The release closure captures the path it actually took — it never
215
+ // recomputes it from the environment (#489 step-02, raised in review).
216
+ return { path, release: result.release };
233
217
  }
234
-
235
- throw new BridgeShutdownError(`Failed to acquire bridge lock ${sessionId}`, 1);
218
+ if (result.blockedBy === -1) {
219
+ throw new BridgeShutdownError(`Failed to acquire bridge lock ${sessionId}`, 1);
220
+ }
221
+ throw new BridgeShutdownError(
222
+ `Another bridge already holds the lock ${sessionId} (PID ${result.blockedBy})`,
223
+ 1,
224
+ );
236
225
  }
237
226
 
238
227
  /** Optional VIBER_REFRESH_LEAD_SECONDS override (clamped to ≥ 0). */
@@ -1159,6 +1148,12 @@ export async function acquireBridgeIdentity(opts: {
1159
1148
  sessionId: `${sessionTag}-agent:${acquired.instance_key}`,
1160
1149
  lockDir,
1161
1150
  logPrefix,
1151
+ // #489 step-03 (review A): fencing — a dispossessed bridge must not keep running.
1152
+ onLost: (reason) => {
1153
+ process.stderr.write(`${logPrefix} ${reason} — withdrawing to avoid a duplicate
1154
+ `);
1155
+ process.exit(1);
1156
+ },
1162
1157
  });
1163
1158
  return {
1164
1159
  instanceKey: acquired.instance_key,
@@ -0,0 +1,323 @@
1
+ /**
2
+ * bridge_lock.ts — the ONE implementation of the per-folder / per-agent lock
3
+ * (#489 step-02).
4
+ *
5
+ * Before this file, the same invariant was implemented TWICE, with real divergences:
6
+ *
7
+ * | | viber-channel.ts (MCP client) | lib/bridge_core.ts (bridges) |
8
+ * |----------------|-------------------------------|------------------------------|
9
+ * | liveness probe | local isProcessAlive | exported isProcessAlive |
10
+ * | `pid <= 0` | ABSENT | present |
11
+ * | on conflict | returns the blocking pid | throws BridgeShutdownError |
12
+ *
13
+ * Fixing only the bridge would have left the Claude MCP client broken, so both go
14
+ * through here. The two error protocols are PRESERVED on purpose and NOT unified:
15
+ * the MCP client prints the blocking pid for the user, the bridge dies. Only the
16
+ * decision — is this lock reclaimable? — is shared, and it is a PURE function so it
17
+ * can be exhausted by tests without touching a filesystem.
18
+ *
19
+ * Dependencies (liveness, clock, fs) are injected with real defaults: step-03 needs
20
+ * to drive them, and a probe hard-wired to `process.kill` is precisely what made the
21
+ * recycled-PID case untestable.
22
+ */
23
+ import { existsSync, mkdirSync, readFileSync, statSync, unlinkSync, writeFileSync } from "node:fs";
24
+ import { dirname } from "node:path";
25
+ import lockfile from "proper-lockfile";
26
+ import { classifyLegacyHolder, processStartTime } from "./process_start.js";
27
+ import { lockTimingOptions } from "./lock_timing.js";
28
+
29
+ /** True if a PID names a live process (EPERM counts as alive: it exists, it is not ours). */
30
+ export function isProcessAlive(pid: number): boolean {
31
+ // The `pid <= 0` guard came from the bridge side only; the MCP client lacked it.
32
+ // Negative/zero pids have platform-specific meanings in kill(2) (process groups),
33
+ // none of which is "the holder of this lock".
34
+ if (!Number.isInteger(pid) || pid <= 0) return false;
35
+ try {
36
+ process.kill(pid, 0);
37
+ return true;
38
+ } catch (err) {
39
+ if (typeof err === "object" && err !== null && "code" in err && err.code === "EPERM") return true;
40
+ return false;
41
+ }
42
+ }
43
+
44
+ /** What a reader concludes about an existing lock file. */
45
+ export type LockVerdict =
46
+ | { kind: "reclaimable"; why: string }
47
+ | { kind: "held"; pid: number };
48
+
49
+ /** Everything the decision needs, so it stays pure and fully testable. */
50
+ export interface LockDecisionInput {
51
+ /** Raw file contents, or `null` when unreadable/absent. */
52
+ payload: string | null;
53
+ isAlive: (pid: number) => boolean;
54
+ }
55
+
56
+ /**
57
+ * Decide whether an existing lock can be taken over. PURE — no I/O, no clock.
58
+ *
59
+ * Today's rule is exactly the historical one, so step-02 stays a refactor: a lock is
60
+ * reclaimable when its payload is unreadable or its PID is not alive. Step-03 changes
61
+ * THIS function, and the tests it already has become the regression net.
62
+ */
63
+ export function decideLock(input: LockDecisionInput): LockVerdict {
64
+ const { payload, isAlive } = input;
65
+ if (payload === null) return { kind: "reclaimable", why: "unreadable" };
66
+ const pid = Number.parseInt(payload.trim(), 10);
67
+ if (!Number.isFinite(pid)) return { kind: "reclaimable", why: "malformed payload" };
68
+ if (!isAlive(pid)) return { kind: "reclaimable", why: `pid ${pid} is not alive` };
69
+ return { kind: "held", pid };
70
+ }
71
+
72
+ export interface AcquireLockFileOptions {
73
+ /** Absolute path of the lock file. */
74
+ lockFile: string;
75
+ /**
76
+ * Liveness probe, used ONLY for the legacy-payload path below. The authoritative
77
+ * mechanism no longer looks at PIDs at all — see the module header.
78
+ */
79
+ isAlive?: (pid: number) => boolean;
80
+ /**
81
+ * Process start time for a PID, epoch ms, or `undefined` when unknowable. Resolves
82
+ * the legacy/recycled ambiguity documented below BY PROOF, not by resemblance.
83
+ */
84
+ startTimeOf?: (pid: number) => number | undefined;
85
+ /**
86
+ * Called when the library reports our lock COMPROMISED (its directory vanished under
87
+ * us — someone reclaimed it after the staleness window). The default logs; every real
88
+ * caller passes a controlled withdrawal, because continuing without authority is the
89
+ * duplicate this design promises to bound.
90
+ */
91
+ onLost?: (reason: string) => void;
92
+ /** PID written for diagnostics and for older clients to see. Defaults to ours. */
93
+ pid?: number;
94
+ /**
95
+ * Staleness regime. Defaults to the PINNED values of `lib/lock_timing.ts`; injected
96
+ * only by tests, which cannot wait 5 minutes to prove that a crashed holder's lock
97
+ * becomes reclaimable.
98
+ */
99
+ timing?: { stale: number; update: number };
100
+ }
101
+
102
+ export type AcquireLockFileResult =
103
+ | { ok: true; release: () => void }
104
+ | { ok: false; blockedBy: number };
105
+
106
+ /**
107
+ * Read the legacy PID payload, if any. `null` when absent or unusable.
108
+ */
109
+ function readLegacyPid(lockFile: string): number | null {
110
+ try {
111
+ const pid = Number.parseInt(readFileSync(lockFile, "utf-8").trim(), 10);
112
+ return Number.isFinite(pid) ? pid : null;
113
+ } catch {
114
+ return null;
115
+ }
116
+ }
117
+
118
+ /**
119
+ * Take the lock.
120
+ *
121
+ * TWO mechanisms, on purpose, and the order matters:
122
+ *
123
+ * 1. **Legacy PID file, checked FIRST.** A client from before #489 advertises itself
124
+ * only by writing its PID into `lockFile`; it knows nothing about the directory
125
+ * `proper-lockfile` uses. If we skipped this check, a new client would see no
126
+ * `.lock` directory, acquire, and run ALONGSIDE a live old holder — #311 re-opened
127
+ * for the whole duration of a mixed deployment (and mixed versions do exist here:
128
+ * the embedded plugin and vibe-master run the same source, npm is the only
129
+ * authority — #503). So: legacy payload + LIVE pid = held. Reviewers required
130
+ * exactly this, and only on a live pid — a dead one is reclaimable, which is what
131
+ * makes the 81 residual locks disappear.
132
+ *
133
+ * 2. **`proper-lockfile` for everything else** — the authority. Its lock carries NO
134
+ * PID: mutual exclusion is a `mkdir` CAS and liveness is the mtime it refreshes.
135
+ * That is why the #489 defect does not come back — a recycled PID cannot make a
136
+ * dead holder look alive, because no PID is consulted at all.
137
+ *
138
+ * After acquiring we still WRITE our pid into `lockFile`, for two reasons: an older
139
+ * client must be able to see that the folder is taken, and the message shown to a human
140
+ * ("another agent holds this folder, PID N") stays useful. That PID is diagnostic
141
+ * only — it is never read as a liveness signal by this code.
142
+ */
143
+ export function acquireLockFile(opts: AcquireLockFileOptions): AcquireLockFileResult {
144
+ const { lockFile } = opts;
145
+ const isAlive = opts.isAlive ?? isProcessAlive;
146
+ const pid = opts.pid ?? process.pid;
147
+ const onLost = opts.onLost ?? ((reason: string) => process.stderr.write(`[viber-lock] ${reason}
148
+ `));
149
+
150
+ mkdirSync(dirname(lockFile), { recursive: true });
151
+
152
+ const legacyPid = readLegacyPid(lockFile);
153
+ if (legacyPid !== null && !existsSync(`${lockFile}.lock`) && isAlive(legacyPid)) {
154
+ // ── The one genuinely ambiguous state, and how it is settled ──────────────
155
+ //
156
+ // On disk these are IDENTICAL: a pre-#489 client still running (pid file, no lock
157
+ // directory, pid alive) and a leftover pid file whose number the OS recycled. Both
158
+ // reviewer requirements land here and pull opposite ways — "respect a live legacy
159
+ // holder" vs "never block permanently on a recycled pid", the latter being why #489
160
+ // exists at all, since the reported case WAS a leftover file.
161
+ //
162
+ // Settled by PROOF, not resemblance. A first attempt compared process NAMES, and
163
+ // review demolished it with a measurement: 110 live bun/node processes on the dev
164
+ // machine, so a recycled pid landing on any other bun was read as a live holder and
165
+ // the lock stayed immortal — the #489 bug surviving in its likeliest case.
166
+ //
167
+ // The proof is a timestamp the same query already returns: a real holder existed
168
+ // BEFORE it wrote its lock file; a recycled pid started AFTER the file was written.
169
+ // See lib/process_start.ts.
170
+ const probe = opts.startTimeOf ?? processStartTime;
171
+ let fileMtimeMs = 0;
172
+ try {
173
+ fileMtimeMs = statSync(lockFile).mtimeMs;
174
+ } catch {
175
+ fileMtimeMs = 0;
176
+ }
177
+ const verdict = classifyLegacyHolder({ startedAt: probe(legacyPid), fileMtimeMs });
178
+ // "unknown" stays conservative: refusing to start is recoverable, two agents on one
179
+ // identity is not.
180
+ if (verdict !== "recycled") {
181
+ return { ok: false, blockedBy: legacyPid };
182
+ }
183
+ }
184
+
185
+ let release: () => void;
186
+ try {
187
+ release = lockfile.lockSync(lockFile, {
188
+ ...(opts.timing ?? lockTimingOptions()),
189
+ retries: 0,
190
+ /**
191
+ * MUST be provided. `proper-lockfile`'s default handler THROWS when its lock
192
+ * directory disappears under a live holder (another process reclaimed it after
193
+ * the staleness window, or something cleaned the directory). That throw happens
194
+ * inside the library's own refresh timer, i.e. outside any of our try/catch —
195
+ * an uncaught exception that would take the whole agent down. Hit for real while
196
+ * building the crash-reclaim test.
197
+ *
198
+ * But swallowing it is just as wrong: a dispossessed holder that keeps running has
199
+ * no authority over its identity while another process may already hold it — the
200
+ * unbounded duplicate this design promises not to create. So we hand it to
201
+ * `onLost`, and every entry point wires that to a controlled withdrawal (the shape
202
+ * `onParentGone` uses in viber-channel.ts).
203
+ */
204
+ onCompromised: (err: Error) => {
205
+ onLost(`lock compromised: ${err.message}`);
206
+ },
207
+ // The target need not exist: we lock a PATH, and the pid file below is written
208
+ // only once the lock is ours.
209
+ realpath: false,
210
+ });
211
+ } catch (err) {
212
+ if (typeof err === "object" && err !== null && "code" in err && err.code === "ELOCKED") {
213
+ // Held by a live holder. Report its pid when the diagnostic file has one.
214
+ return { ok: false, blockedBy: readLegacyPid(lockFile) ?? -1 };
215
+ }
216
+ throw err;
217
+ }
218
+
219
+ try {
220
+ writeFileSync(lockFile, `${pid}\n`);
221
+ } catch {
222
+ // Diagnostics only — never fail an acquisition over it.
223
+ }
224
+
225
+ // #489 step-03 (review C): REPAIR the diagnostic pid file periodically.
226
+ //
227
+ // Review measured the real severity of losing it: the library's directory still holds
228
+ // the authority, but a pre-#489 client reads ONLY this file — so if it disappears,
229
+ // that old client walks in and stays for B's whole lifetime, not for microseconds.
230
+ // Re-creating it on a beat bounds the exposure to one interval. Same cadence as the
231
+ // library's own refresh; `unref` so it can never keep a process alive.
232
+ const repair = setInterval(() => {
233
+ try {
234
+ if (readLegacyPid(lockFile) !== pid) writeFileSync(lockFile, `${pid}
235
+ `);
236
+ } catch {
237
+ // Diagnostic only.
238
+ }
239
+ }, (opts.timing ?? lockTimingOptions()).update);
240
+ repair.unref?.();
241
+
242
+ let released = false;
243
+ return {
244
+ ok: true,
245
+ release: () => {
246
+ // Idempotent: exit handlers can fire twice, and releasing twice must not throw.
247
+ if (released) return;
248
+ released = true;
249
+ clearInterval(repair);
250
+ try {
251
+ release();
252
+ } catch {
253
+ // Already gone / reclaimed elsewhere.
254
+ }
255
+ // #489 step-03 (review): remove the diagnostic pid file ONLY IF IT IS STILL OURS.
256
+ //
257
+ // Review asked for no unlink at all, to close this window: we release the
258
+ // library's lock, B acquires and writes ITS pid, and our unlink would delete B's
259
+ // file — after which an old client, which only reads that file, would miss B and
260
+ // start in parallel.
261
+ //
262
+ // Measured consequence of never unlinking, though: our own pid file survives our
263
+ // release, our process is still alive and STARTED BEFORE the file, so the
264
+ // start-time proof classifies us as a live legacy holder — and the folder is
265
+ // blocked for good. A DETERMINISTIC immortal lock, i.e. the #489 bug re-created by
266
+ // its own fix. Two tests caught it.
267
+ //
268
+ // So: compare, then unlink. The residual race is a read-then-unlink of a few
269
+ // microseconds whose worst outcome is a MISSING diagnostic file (bounded, and the
270
+ // library's directory still holds the authority), against a permanent block. The
271
+ // window is stated rather than hidden.
272
+ try {
273
+ if (readLegacyPid(lockFile) === pid) unlinkSync(lockFile);
274
+ } catch {
275
+ // Diagnostic only — never fail a release over it.
276
+ }
277
+ },
278
+ };
279
+ }
280
+
281
+ /**
282
+ * Hold the release of a lock we ACTUALLY took — nothing else (#489 step-02).
283
+ *
284
+ * This exists because of a real bug in the pre-#489 code, found in review and
285
+ * verified on `e67a2e0`:
286
+ *
287
+ * - `process.on("exit", … releaseLock())` was registered at MODULE scope
288
+ * (`viber-channel.ts:335`), therefore BEFORE `acquireLock()` ran (l.523);
289
+ * - the old `releaseLock()` did an UNCONDITIONAL `unlinkSync(LOCK_FILE)`;
290
+ * - a launch that LOST the race went on to `process.exit(1)`, firing that handler.
291
+ *
292
+ * So every accidental duplicate ended by DELETING THE WINNER'S LOCK: the folder went
293
+ * unprotected while the winner was still running, and a third launch acquired freely.
294
+ * The invariant this lock exists to hold (#166/#311) was broken by its own error path.
295
+ *
296
+ * The slot makes the guarantee structural instead of incidental: `releaseIfHeld()` is
297
+ * a no-op unless an acquisition succeeded and handed over its release. Kept HERE, in
298
+ * the tested module, rather than as a bare `let` in the entry point that no test can
299
+ * import.
300
+ */
301
+ export interface LockSlot {
302
+ /** Record the release of a successful acquisition. */
303
+ adopt: (release: () => void) => void;
304
+ /** Release only if we hold it. Idempotent — exit handlers can fire twice. */
305
+ releaseIfHeld: () => void;
306
+ /** True once an acquisition has been adopted and not yet released. */
307
+ isHeld: () => boolean;
308
+ }
309
+
310
+ export function createLockSlot(): LockSlot {
311
+ let release: (() => void) | null = null;
312
+ return {
313
+ adopt: (r) => {
314
+ release = r;
315
+ },
316
+ releaseIfHeld: () => {
317
+ const r = release;
318
+ release = null;
319
+ r?.();
320
+ },
321
+ isHeld: () => release !== null,
322
+ };
323
+ }
@@ -27,6 +27,7 @@ import {
27
27
  messageAgent,
28
28
  sendMessage,
29
29
  type AgentToolsContext,
30
+ toMcpToolResult,
30
31
  type AgentToolResult,
31
32
  } from "./agent_tools.js";
32
33
  import { CONTRACT_STYLE_NEUTRAL, CONTRACT_STYLE_TOOLS } from "./channel_instructions.js";
@@ -91,8 +92,19 @@ export const TOOL_DEFS = [
91
92
  "List the OTHER agents (instances) you can DM. Returns each agent's id, " +
92
93
  "label, runtime kind, and whether it is online. Use it to find an id before message_agent. " +
93
94
  "Normally project-scoped; an ORCHESTRATOR instance (#307) sees all the owner's projects, " +
94
- "each entry tagged with a `project` field.",
95
- inputSchema: { type: "object", properties: {}, additionalProperties: false },
95
+ "each entry tagged with a `project` field. " +
96
+ "FILTER instead of listing everything (#501): `online: true` for live agents only, " +
97
+ "`label_prefix` for one team (e.g. \"501-\"). Both are applied server-side. " +
98
+ "If the presence source is unreachable the `online` filter is DROPPED and the answer " +
99
+ "says so — an empty list would wrongly read as \"no agent is alive\".",
100
+ inputSchema: {
101
+ type: "object",
102
+ properties: {
103
+ online: { type: "boolean", description: "Only agents currently seen online." },
104
+ label_prefix: { type: "string", description: 'Only labels starting with this, e.g. "501-".' },
105
+ },
106
+ additionalProperties: false,
107
+ },
96
108
  },
97
109
  {
98
110
  name: "message_agent",
@@ -145,7 +157,7 @@ export async function dispatchBridgeTool(
145
157
  ): Promise<AgentToolResult> {
146
158
  let result: AgentToolResult;
147
159
  if (name === "list_agents") {
148
- result = await listAgents(opts.ctx);
160
+ result = await listAgents(opts.ctx, args);
149
161
  } else if (name === "message_agent") {
150
162
  result = await messageAgent(opts.ctx, args);
151
163
  } else if (name === "send_message") {
@@ -166,7 +178,7 @@ function buildServer(opts: BridgeToolHostOptions): Server {
166
178
  mcp.setRequestHandler(CallToolRequestSchema, async (request) => {
167
179
  const args = (request.params.arguments ?? {}) as Record<string, unknown>;
168
180
  try {
169
- return await dispatchBridgeTool(request.params.name, args, opts);
181
+ return toMcpToolResult(await dispatchBridgeTool(request.params.name, args, opts));
170
182
  } catch (err) {
171
183
  // The shared lib rethrows ConversationTokenExpiredError (host-specific
172
184
  // teardown). In the bridge we surface it as a tool error rather than
@@ -71,10 +71,24 @@ export const CLAUDE_TOOL_DEFS = [
71
71
  "Use this to find the id of an agent (e.g. 'Codex Review') before calling message_agent. " +
72
72
  "Normally scoped to this project; if the owner designated this instance an ORCHESTRATOR " +
73
73
  "(#307), the list covers ALL the owner's projects and each entry carries a `project` field. " +
74
- "You are never in the list.",
74
+ "You are never in the list. " +
75
+ "FILTER instead of listing everything (#501): `online: true` returns only agents the server " +
76
+ 'currently sees online, `label_prefix` only those whose label starts with it (e.g. "501-" ' +
77
+ "for one team). Both are applied server-side, so an unfiltered call in a loop is the " +
78
+ "expensive path. If the presence source is unreachable the `online` filter is DROPPED and " +
79
+ 'the answer says so — an empty list would wrongly read as "no agent is alive".',
75
80
  inputSchema: {
76
81
  type: "object" as const,
77
- properties: {},
82
+ properties: {
83
+ online: {
84
+ type: "boolean",
85
+ description: "Only agents the server currently sees online.",
86
+ },
87
+ label_prefix: {
88
+ type: "string",
89
+ description: 'Only agents whose label starts with this prefix, e.g. "501-".',
90
+ },
91
+ },
78
92
  required: [],
79
93
  additionalProperties: false,
80
94
  },
package/lib/instance.ts CHANGED
@@ -193,6 +193,16 @@ export interface AcquiredInstance {
193
193
  * 3. `auth.json` `instance_token` — reused (durable identity, D4 → same id).
194
194
  * 4. register a NEW instance with the stable project_token, persisted back to
195
195
  * auth.json (self-healing migration — no forced manual reconnect).
196
+ *
197
+ * ⚠ #501 DEPENDS ON THE PRIORITY ORDER ABOVE. vibe-master's readiness gate proves
198
+ * an agent came up by watching for an instance id that did NOT exist before the
199
+ * launch. Only route 2 mints one: routes 1 and 3 REUSE an id, so a launch that
200
+ * took either would be reported as "never registered" while being perfectly
201
+ * healthy — a false red. vibe-master's spawns take route 2 because
202
+ * `vibe-master/lib/terminal.ts` scrubs `VIBER_INSTANCE_TOKEN`/`_ID` from the
203
+ * child environment; the guarantee lives there, and this comment exists so a
204
+ * change here is not made unaware of it. Locked by the route-precedence tests in
205
+ * `test/instance.test.ts`.
196
206
  */
197
207
  export async function acquireInstance(
198
208
  baseUrl: string,