viber-channel 0.8.15 → 0.8.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/agent_tools.ts +51 -4
- package/lib/bridge_core.ts +50 -55
- package/lib/bridge_lock.ts +323 -0
- package/lib/bridge_tool_host.ts +16 -4
- package/lib/claude_tool_defs.ts +16 -2
- package/lib/instance.ts +10 -0
- package/lib/lock_dir.ts +62 -0
- package/lib/lock_timing.ts +39 -0
- package/lib/lockfile.ts +8 -5
- package/lib/peers.ts +62 -7
- package/lib/process_start.ts +156 -0
- package/lib/spawn_reason.ts +30 -1
- package/package.json +10 -3
- package/viber-channel.ts +71 -64
- package/viber-codex-bridge.ts +49 -15
- package/viber-gemma-bridge.ts +6 -5
package/lib/agent_tools.ts
CHANGED
|
@@ -17,8 +17,9 @@
|
|
|
17
17
|
* MCP tool-result shape. It does NOT hold state.
|
|
18
18
|
*/
|
|
19
19
|
import { postMessage, parseArtifact, type Artifact } from "./messages.js";
|
|
20
|
-
import { listPeersAuto, openDm, type OpenDmResult } from "./peers.js";
|
|
20
|
+
import { listPeersAuto, openDm, type OpenDmResult, type PeerFilters } from "./peers.js";
|
|
21
21
|
import { ConversationTokenExpiredError } from "./messages.js";
|
|
22
|
+
import type { CallToolResult } from "@modelcontextprotocol/sdk/types.js";
|
|
22
23
|
|
|
23
24
|
/** MCP tool-result shape (text content + optional error flag). */
|
|
24
25
|
export interface AgentToolResult {
|
|
@@ -26,6 +27,28 @@ export interface AgentToolResult {
|
|
|
26
27
|
content: { type: "text"; text: string }[];
|
|
27
28
|
}
|
|
28
29
|
|
|
30
|
+
/**
|
|
31
|
+
* #489 step-04 — adapt a tool result to what the MCP SDK's handler signature wants.
|
|
32
|
+
*
|
|
33
|
+
* The SDK's `CallToolResult` carries an index signature (the MCP result object is
|
|
34
|
+
* open: `_meta` and future fields pass through), so `AgentToolResult` was NOT
|
|
35
|
+
* assignable to the SDK handler type — the TS2345 at `lib/bridge_tool_host.ts` and
|
|
36
|
+
* `viber-channel.ts`. Verified against the SDK schema, not against our output:
|
|
37
|
+
* `{ content: [{ type: "text", text }], isError }` IS a valid `CallToolResult`. The
|
|
38
|
+
* missing openness was the ONLY incompatibility, so there is no protocol divergence.
|
|
39
|
+
*
|
|
40
|
+
* The openness lives HERE, at the SDK boundary, and NOT on `AgentToolResult` — review
|
|
41
|
+
* measured that opening our own interface silently disables excess-property checking
|
|
42
|
+
* on it, so `{ content: […], isErorr: true }` would compile. Keeping our type closed
|
|
43
|
+
* preserves that check at every construction site; the single unavoidable widening is
|
|
44
|
+
* named, commented, and confined to this function.
|
|
45
|
+
*/
|
|
46
|
+
export function toMcpToolResult(result: AgentToolResult): CallToolResult {
|
|
47
|
+
// No cast: the spread is a FRESH object type, which TypeScript grants an implicit
|
|
48
|
+
// index signature when assigning to the SDK's open result type.
|
|
49
|
+
return { ...result };
|
|
50
|
+
}
|
|
51
|
+
|
|
29
52
|
/**
|
|
30
53
|
* Live, by-reference access to the host channel's runtime state. Implemented by
|
|
31
54
|
* both hosts so the shared tool logic never reads a stale snapshot. Every getter
|
|
@@ -75,12 +98,27 @@ function errorText(s: string): AgentToolResult {
|
|
|
75
98
|
* entry then carries a `project` field); a regular instance transparently
|
|
76
99
|
* falls back to its own project (unchanged behaviour, no `project` field).
|
|
77
100
|
*/
|
|
78
|
-
export async function listAgents(
|
|
101
|
+
export async function listAgents(
|
|
102
|
+
ctx: AgentToolsContext,
|
|
103
|
+
args: Record<string, unknown> = {},
|
|
104
|
+
): Promise<AgentToolResult> {
|
|
79
105
|
if (!ctx.instanceToken()) {
|
|
80
106
|
return errorText("Channel not ready: no instance identity yet.");
|
|
81
107
|
}
|
|
108
|
+
// #501 step-05 — optional filters, applied SERVER-side. Without them a
|
|
109
|
+
// dev-lead looking for its two reviewers pulls every agent the project ever
|
|
110
|
+
// created (30 rows for 5 live ones, measured), in a loop.
|
|
111
|
+
const filters: PeerFilters = {};
|
|
112
|
+
if (args.online === true) filters.online = true;
|
|
113
|
+
if (typeof args.label_prefix === "string" && args.label_prefix !== "") {
|
|
114
|
+
filters.labelPrefix = args.label_prefix;
|
|
115
|
+
}
|
|
82
116
|
try {
|
|
83
|
-
const { peers, scope } = await listPeersAuto(
|
|
117
|
+
const { peers, scope, presenceStatus } = await listPeersAuto(
|
|
118
|
+
ctx.baseUrl(),
|
|
119
|
+
ctx.instanceToken(),
|
|
120
|
+
filters,
|
|
121
|
+
);
|
|
84
122
|
// Project only what the model needs to pick a peer (drop last_seen /
|
|
85
123
|
// active_conversation_id — available over the wire if a future tool needs them).
|
|
86
124
|
const summary = peers.map((p) => ({
|
|
@@ -91,13 +129,22 @@ export async function listAgents(ctx: AgentToolsContext): Promise<AgentToolResul
|
|
|
91
129
|
// Present only in the user-scoped (orchestrator) listing.
|
|
92
130
|
...(p.project_name !== undefined ? { project: p.project_name } : {}),
|
|
93
131
|
}));
|
|
132
|
+
// An `online` filter under an unreadable presence source would come back
|
|
133
|
+
// empty and read as "nobody is alive" — the false green in listing form.
|
|
134
|
+
// The server drops the filter in that case; say so rather than let the
|
|
135
|
+
// caller believe it was applied.
|
|
136
|
+
const presenceWarning =
|
|
137
|
+
filters.online === true && presenceStatus === "unavailable"
|
|
138
|
+
? "WARNING: the presence source is unavailable, so the `online` filter was NOT applied — " +
|
|
139
|
+
"`online` values below are unknown, not observed.\n\n"
|
|
140
|
+
: "";
|
|
94
141
|
const body =
|
|
95
142
|
summary.length === 0
|
|
96
143
|
? scope === "user"
|
|
97
144
|
? "No other agents are currently registered in any of your projects."
|
|
98
145
|
: "No other agents are currently registered in this project."
|
|
99
146
|
: JSON.stringify(summary, null, 2);
|
|
100
|
-
return text(body);
|
|
147
|
+
return text(`${presenceWarning}${body}`);
|
|
101
148
|
} catch (err) {
|
|
102
149
|
return errorText(`list_agents failed: ${String(err)}`);
|
|
103
150
|
}
|
package/lib/bridge_core.ts
CHANGED
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
import { mkdirSync, readFileSync, unlinkSync, writeFileSync } from "node:fs";
|
|
17
17
|
import { createTokenRefreshScheduler, type TokenRefreshScheduler } from "./token_refresh.js";
|
|
18
18
|
import type { ApiBaseUrlResolver } from "./base_urls.js";
|
|
19
|
+
import { acquireLockFile, isProcessAlive } from "./bridge_lock.js";
|
|
19
20
|
import { lockFilePath } from "./lockfile.js";
|
|
20
21
|
import {
|
|
21
22
|
type ConversationMintResponse,
|
|
@@ -159,24 +160,24 @@ export function planAwaitInviteFirstPass<T>(msgs: T[]): { forward: T[]; seed: T[
|
|
|
159
160
|
return { forward: [msgs[msgs.length - 1]], seed: msgs.slice(0, -1) };
|
|
160
161
|
}
|
|
161
162
|
|
|
162
|
-
/**
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
} catch (err) {
|
|
169
|
-
if (typeof err === "object" && err !== null && "code" in err && err.code === "EPERM") return true;
|
|
170
|
-
return false;
|
|
171
|
-
}
|
|
172
|
-
}
|
|
163
|
+
/**
|
|
164
|
+
* #489 step-02: re-exported from `lib/bridge_lock.ts`, which is now the single
|
|
165
|
+
* implementation. Kept exported here because callers and tests already import it
|
|
166
|
+
* from this module.
|
|
167
|
+
*/
|
|
168
|
+
export { isProcessAlive };
|
|
173
169
|
|
|
174
170
|
/**
|
|
175
|
-
* Acquire the per-agent bridge lock (a PID file). `sessionId` is the identity
|
|
176
|
-
*
|
|
177
|
-
*
|
|
178
|
-
*
|
|
179
|
-
*
|
|
171
|
+
* Acquire the per-agent bridge lock (a PID file). `sessionId` is the identity axis
|
|
172
|
+
* (e.g. `codex-agent:<instanceId>` / `gemma-agent:<instanceId>`), so two processes
|
|
173
|
+
* with the SAME identity cannot both run. A stale lock (dead PID) is reclaimed; a
|
|
174
|
+
* live one throws a BridgeShutdownError. `logPrefix` keeps each bridge's log lines
|
|
175
|
+
* under its own tag.
|
|
176
|
+
*
|
|
177
|
+
* #489 step-02: a thin adapter now. The loop, the payload and the reclaim decision
|
|
178
|
+
* live in `lib/bridge_lock.ts`, shared with the MCP client — the duplication was one
|
|
179
|
+
* of the defects of #489. What stays HERE is this site's error protocol: throw, so
|
|
180
|
+
* the bridge dies rather than reporting a pid to a human.
|
|
180
181
|
*/
|
|
181
182
|
export function acquireBridgeLock(opts: {
|
|
182
183
|
/**
|
|
@@ -190,49 +191,37 @@ export function acquireBridgeLock(opts: {
|
|
|
190
191
|
sessionId: string;
|
|
191
192
|
lockDir: string;
|
|
192
193
|
logPrefix: string;
|
|
194
|
+
/** Called if the lock is lost mid-run — the caller tears the bridge down. */
|
|
195
|
+
onLost?: (reason: string) => void;
|
|
193
196
|
}): BridgeLock {
|
|
194
197
|
const { identityBaseUrl, fingerprint, sessionId, lockDir, logPrefix } = opts;
|
|
195
|
-
mkdirSync(lockDir, { recursive: true });
|
|
196
198
|
const path = lockFilePath(identityBaseUrl, fingerprint, sessionId, lockDir);
|
|
197
|
-
const
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
}
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
let existingPid = Number.NaN;
|
|
216
|
-
try {
|
|
217
|
-
existingPid = Number.parseInt(readFileSync(path, "utf-8").trim(), 10);
|
|
218
|
-
} catch {
|
|
219
|
-
release();
|
|
220
|
-
continue;
|
|
221
|
-
}
|
|
222
|
-
|
|
223
|
-
if (!Number.isFinite(existingPid) || !isProcessAlive(existingPid)) {
|
|
224
|
-
release();
|
|
225
|
-
continue;
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
throw new BridgeShutdownError(
|
|
229
|
-
`Another bridge already holds the lock ${sessionId} (PID ${existingPid})`,
|
|
230
|
-
1,
|
|
231
|
-
);
|
|
232
|
-
}
|
|
199
|
+
const result = acquireLockFile({
|
|
200
|
+
lockFile: path,
|
|
201
|
+
// #489 step-03 (review): losing the lock is a FENCING signal, not a log line. A
|
|
202
|
+
// dispossessed bridge that keeps running has no authority over its identity, which
|
|
203
|
+
// is the unbounded duplicate this design promises not to create. Withdraw the same
|
|
204
|
+
// way the parent watchdog does: request shutdown and let the normal teardown run.
|
|
205
|
+
onLost: (reason) => {
|
|
206
|
+
process.stderr.write(`${logPrefix} ${reason} — withdrawing
|
|
207
|
+
`);
|
|
208
|
+
opts.onLost?.(reason);
|
|
209
|
+
},
|
|
210
|
+
});
|
|
211
|
+
if (result.ok) {
|
|
212
|
+
process.stderr.write(`${logPrefix} lock acquired: ${path}
|
|
213
|
+
`);
|
|
214
|
+
// The release closure captures the path it actually took — it never
|
|
215
|
+
// recomputes it from the environment (#489 step-02, raised in review).
|
|
216
|
+
return { path, release: result.release };
|
|
233
217
|
}
|
|
234
|
-
|
|
235
|
-
|
|
218
|
+
if (result.blockedBy === -1) {
|
|
219
|
+
throw new BridgeShutdownError(`Failed to acquire bridge lock ${sessionId}`, 1);
|
|
220
|
+
}
|
|
221
|
+
throw new BridgeShutdownError(
|
|
222
|
+
`Another bridge already holds the lock ${sessionId} (PID ${result.blockedBy})`,
|
|
223
|
+
1,
|
|
224
|
+
);
|
|
236
225
|
}
|
|
237
226
|
|
|
238
227
|
/** Optional VIBER_REFRESH_LEAD_SECONDS override (clamped to ≥ 0). */
|
|
@@ -1159,6 +1148,12 @@ export async function acquireBridgeIdentity(opts: {
|
|
|
1159
1148
|
sessionId: `${sessionTag}-agent:${acquired.instance_key}`,
|
|
1160
1149
|
lockDir,
|
|
1161
1150
|
logPrefix,
|
|
1151
|
+
// #489 step-03 (review A): fencing — a dispossessed bridge must not keep running.
|
|
1152
|
+
onLost: (reason) => {
|
|
1153
|
+
process.stderr.write(`${logPrefix} ${reason} — withdrawing to avoid a duplicate
|
|
1154
|
+
`);
|
|
1155
|
+
process.exit(1);
|
|
1156
|
+
},
|
|
1162
1157
|
});
|
|
1163
1158
|
return {
|
|
1164
1159
|
instanceKey: acquired.instance_key,
|
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* bridge_lock.ts — the ONE implementation of the per-folder / per-agent lock
|
|
3
|
+
* (#489 step-02).
|
|
4
|
+
*
|
|
5
|
+
* Before this file, the same invariant was implemented TWICE, with real divergences:
|
|
6
|
+
*
|
|
7
|
+
* | | viber-channel.ts (MCP client) | lib/bridge_core.ts (bridges) |
|
|
8
|
+
* |----------------|-------------------------------|------------------------------|
|
|
9
|
+
* | liveness probe | local isProcessAlive | exported isProcessAlive |
|
|
10
|
+
* | `pid <= 0` | ABSENT | present |
|
|
11
|
+
* | on conflict | returns the blocking pid | throws BridgeShutdownError |
|
|
12
|
+
*
|
|
13
|
+
* Fixing only the bridge would have left the Claude MCP client broken, so both go
|
|
14
|
+
* through here. The two error protocols are PRESERVED on purpose and NOT unified:
|
|
15
|
+
* the MCP client prints the blocking pid for the user, the bridge dies. Only the
|
|
16
|
+
* decision — is this lock reclaimable? — is shared, and it is a PURE function so it
|
|
17
|
+
* can be exhausted by tests without touching a filesystem.
|
|
18
|
+
*
|
|
19
|
+
* Dependencies (liveness, clock, fs) are injected with real defaults: step-03 needs
|
|
20
|
+
* to drive them, and a probe hard-wired to `process.kill` is precisely what made the
|
|
21
|
+
* recycled-PID case untestable.
|
|
22
|
+
*/
|
|
23
|
+
import { existsSync, mkdirSync, readFileSync, statSync, unlinkSync, writeFileSync } from "node:fs";
|
|
24
|
+
import { dirname } from "node:path";
|
|
25
|
+
import lockfile from "proper-lockfile";
|
|
26
|
+
import { classifyLegacyHolder, processStartTime } from "./process_start.js";
|
|
27
|
+
import { lockTimingOptions } from "./lock_timing.js";
|
|
28
|
+
|
|
29
|
+
/** True if a PID names a live process (EPERM counts as alive: it exists, it is not ours). */
|
|
30
|
+
export function isProcessAlive(pid: number): boolean {
|
|
31
|
+
// The `pid <= 0` guard came from the bridge side only; the MCP client lacked it.
|
|
32
|
+
// Negative/zero pids have platform-specific meanings in kill(2) (process groups),
|
|
33
|
+
// none of which is "the holder of this lock".
|
|
34
|
+
if (!Number.isInteger(pid) || pid <= 0) return false;
|
|
35
|
+
try {
|
|
36
|
+
process.kill(pid, 0);
|
|
37
|
+
return true;
|
|
38
|
+
} catch (err) {
|
|
39
|
+
if (typeof err === "object" && err !== null && "code" in err && err.code === "EPERM") return true;
|
|
40
|
+
return false;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** What a reader concludes about an existing lock file. */
|
|
45
|
+
export type LockVerdict =
|
|
46
|
+
| { kind: "reclaimable"; why: string }
|
|
47
|
+
| { kind: "held"; pid: number };
|
|
48
|
+
|
|
49
|
+
/** Everything the decision needs, so it stays pure and fully testable. */
|
|
50
|
+
export interface LockDecisionInput {
|
|
51
|
+
/** Raw file contents, or `null` when unreadable/absent. */
|
|
52
|
+
payload: string | null;
|
|
53
|
+
isAlive: (pid: number) => boolean;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Decide whether an existing lock can be taken over. PURE — no I/O, no clock.
|
|
58
|
+
*
|
|
59
|
+
* Today's rule is exactly the historical one, so step-02 stays a refactor: a lock is
|
|
60
|
+
* reclaimable when its payload is unreadable or its PID is not alive. Step-03 changes
|
|
61
|
+
* THIS function, and the tests it already has become the regression net.
|
|
62
|
+
*/
|
|
63
|
+
export function decideLock(input: LockDecisionInput): LockVerdict {
|
|
64
|
+
const { payload, isAlive } = input;
|
|
65
|
+
if (payload === null) return { kind: "reclaimable", why: "unreadable" };
|
|
66
|
+
const pid = Number.parseInt(payload.trim(), 10);
|
|
67
|
+
if (!Number.isFinite(pid)) return { kind: "reclaimable", why: "malformed payload" };
|
|
68
|
+
if (!isAlive(pid)) return { kind: "reclaimable", why: `pid ${pid} is not alive` };
|
|
69
|
+
return { kind: "held", pid };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export interface AcquireLockFileOptions {
|
|
73
|
+
/** Absolute path of the lock file. */
|
|
74
|
+
lockFile: string;
|
|
75
|
+
/**
|
|
76
|
+
* Liveness probe, used ONLY for the legacy-payload path below. The authoritative
|
|
77
|
+
* mechanism no longer looks at PIDs at all — see the module header.
|
|
78
|
+
*/
|
|
79
|
+
isAlive?: (pid: number) => boolean;
|
|
80
|
+
/**
|
|
81
|
+
* Process start time for a PID, epoch ms, or `undefined` when unknowable. Resolves
|
|
82
|
+
* the legacy/recycled ambiguity documented below BY PROOF, not by resemblance.
|
|
83
|
+
*/
|
|
84
|
+
startTimeOf?: (pid: number) => number | undefined;
|
|
85
|
+
/**
|
|
86
|
+
* Called when the library reports our lock COMPROMISED (its directory vanished under
|
|
87
|
+
* us — someone reclaimed it after the staleness window). The default logs; every real
|
|
88
|
+
* caller passes a controlled withdrawal, because continuing without authority is the
|
|
89
|
+
* duplicate this design promises to bound.
|
|
90
|
+
*/
|
|
91
|
+
onLost?: (reason: string) => void;
|
|
92
|
+
/** PID written for diagnostics and for older clients to see. Defaults to ours. */
|
|
93
|
+
pid?: number;
|
|
94
|
+
/**
|
|
95
|
+
* Staleness regime. Defaults to the PINNED values of `lib/lock_timing.ts`; injected
|
|
96
|
+
* only by tests, which cannot wait 5 minutes to prove that a crashed holder's lock
|
|
97
|
+
* becomes reclaimable.
|
|
98
|
+
*/
|
|
99
|
+
timing?: { stale: number; update: number };
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export type AcquireLockFileResult =
|
|
103
|
+
| { ok: true; release: () => void }
|
|
104
|
+
| { ok: false; blockedBy: number };
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Read the legacy PID payload, if any. `null` when absent or unusable.
|
|
108
|
+
*/
|
|
109
|
+
function readLegacyPid(lockFile: string): number | null {
|
|
110
|
+
try {
|
|
111
|
+
const pid = Number.parseInt(readFileSync(lockFile, "utf-8").trim(), 10);
|
|
112
|
+
return Number.isFinite(pid) ? pid : null;
|
|
113
|
+
} catch {
|
|
114
|
+
return null;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Take the lock.
|
|
120
|
+
*
|
|
121
|
+
* TWO mechanisms, on purpose, and the order matters:
|
|
122
|
+
*
|
|
123
|
+
* 1. **Legacy PID file, checked FIRST.** A client from before #489 advertises itself
|
|
124
|
+
* only by writing its PID into `lockFile`; it knows nothing about the directory
|
|
125
|
+
* `proper-lockfile` uses. If we skipped this check, a new client would see no
|
|
126
|
+
* `.lock` directory, acquire, and run ALONGSIDE a live old holder — #311 re-opened
|
|
127
|
+
* for the whole duration of a mixed deployment (and mixed versions do exist here:
|
|
128
|
+
* the embedded plugin and vibe-master run the same source, npm is the only
|
|
129
|
+
* authority — #503). So: legacy payload + LIVE pid = held. Reviewers required
|
|
130
|
+
* exactly this, and only on a live pid — a dead one is reclaimable, which is what
|
|
131
|
+
* makes the 81 residual locks disappear.
|
|
132
|
+
*
|
|
133
|
+
* 2. **`proper-lockfile` for everything else** — the authority. Its lock carries NO
|
|
134
|
+
* PID: mutual exclusion is a `mkdir` CAS and liveness is the mtime it refreshes.
|
|
135
|
+
* That is why the #489 defect does not come back — a recycled PID cannot make a
|
|
136
|
+
* dead holder look alive, because no PID is consulted at all.
|
|
137
|
+
*
|
|
138
|
+
* After acquiring we still WRITE our pid into `lockFile`, for two reasons: an older
|
|
139
|
+
* client must be able to see that the folder is taken, and the message shown to a human
|
|
140
|
+
* ("another agent holds this folder, PID N") stays useful. That PID is diagnostic
|
|
141
|
+
* only — it is never read as a liveness signal by this code.
|
|
142
|
+
*/
|
|
143
|
+
export function acquireLockFile(opts: AcquireLockFileOptions): AcquireLockFileResult {
|
|
144
|
+
const { lockFile } = opts;
|
|
145
|
+
const isAlive = opts.isAlive ?? isProcessAlive;
|
|
146
|
+
const pid = opts.pid ?? process.pid;
|
|
147
|
+
const onLost = opts.onLost ?? ((reason: string) => process.stderr.write(`[viber-lock] ${reason}
|
|
148
|
+
`));
|
|
149
|
+
|
|
150
|
+
mkdirSync(dirname(lockFile), { recursive: true });
|
|
151
|
+
|
|
152
|
+
const legacyPid = readLegacyPid(lockFile);
|
|
153
|
+
if (legacyPid !== null && !existsSync(`${lockFile}.lock`) && isAlive(legacyPid)) {
|
|
154
|
+
// ── The one genuinely ambiguous state, and how it is settled ──────────────
|
|
155
|
+
//
|
|
156
|
+
// On disk these are IDENTICAL: a pre-#489 client still running (pid file, no lock
|
|
157
|
+
// directory, pid alive) and a leftover pid file whose number the OS recycled. Both
|
|
158
|
+
// reviewer requirements land here and pull opposite ways — "respect a live legacy
|
|
159
|
+
// holder" vs "never block permanently on a recycled pid", the latter being why #489
|
|
160
|
+
// exists at all, since the reported case WAS a leftover file.
|
|
161
|
+
//
|
|
162
|
+
// Settled by PROOF, not resemblance. A first attempt compared process NAMES, and
|
|
163
|
+
// review demolished it with a measurement: 110 live bun/node processes on the dev
|
|
164
|
+
// machine, so a recycled pid landing on any other bun was read as a live holder and
|
|
165
|
+
// the lock stayed immortal — the #489 bug surviving in its likeliest case.
|
|
166
|
+
//
|
|
167
|
+
// The proof is a timestamp the same query already returns: a real holder existed
|
|
168
|
+
// BEFORE it wrote its lock file; a recycled pid started AFTER the file was written.
|
|
169
|
+
// See lib/process_start.ts.
|
|
170
|
+
const probe = opts.startTimeOf ?? processStartTime;
|
|
171
|
+
let fileMtimeMs = 0;
|
|
172
|
+
try {
|
|
173
|
+
fileMtimeMs = statSync(lockFile).mtimeMs;
|
|
174
|
+
} catch {
|
|
175
|
+
fileMtimeMs = 0;
|
|
176
|
+
}
|
|
177
|
+
const verdict = classifyLegacyHolder({ startedAt: probe(legacyPid), fileMtimeMs });
|
|
178
|
+
// "unknown" stays conservative: refusing to start is recoverable, two agents on one
|
|
179
|
+
// identity is not.
|
|
180
|
+
if (verdict !== "recycled") {
|
|
181
|
+
return { ok: false, blockedBy: legacyPid };
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
let release: () => void;
|
|
186
|
+
try {
|
|
187
|
+
release = lockfile.lockSync(lockFile, {
|
|
188
|
+
...(opts.timing ?? lockTimingOptions()),
|
|
189
|
+
retries: 0,
|
|
190
|
+
/**
|
|
191
|
+
* MUST be provided. `proper-lockfile`'s default handler THROWS when its lock
|
|
192
|
+
* directory disappears under a live holder (another process reclaimed it after
|
|
193
|
+
* the staleness window, or something cleaned the directory). That throw happens
|
|
194
|
+
* inside the library's own refresh timer, i.e. outside any of our try/catch —
|
|
195
|
+
* an uncaught exception that would take the whole agent down. Hit for real while
|
|
196
|
+
* building the crash-reclaim test.
|
|
197
|
+
*
|
|
198
|
+
* But swallowing it is just as wrong: a dispossessed holder that keeps running has
|
|
199
|
+
* no authority over its identity while another process may already hold it — the
|
|
200
|
+
* unbounded duplicate this design promises not to create. So we hand it to
|
|
201
|
+
* `onLost`, and every entry point wires that to a controlled withdrawal (the shape
|
|
202
|
+
* `onParentGone` uses in viber-channel.ts).
|
|
203
|
+
*/
|
|
204
|
+
onCompromised: (err: Error) => {
|
|
205
|
+
onLost(`lock compromised: ${err.message}`);
|
|
206
|
+
},
|
|
207
|
+
// The target need not exist: we lock a PATH, and the pid file below is written
|
|
208
|
+
// only once the lock is ours.
|
|
209
|
+
realpath: false,
|
|
210
|
+
});
|
|
211
|
+
} catch (err) {
|
|
212
|
+
if (typeof err === "object" && err !== null && "code" in err && err.code === "ELOCKED") {
|
|
213
|
+
// Held by a live holder. Report its pid when the diagnostic file has one.
|
|
214
|
+
return { ok: false, blockedBy: readLegacyPid(lockFile) ?? -1 };
|
|
215
|
+
}
|
|
216
|
+
throw err;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
try {
|
|
220
|
+
writeFileSync(lockFile, `${pid}\n`);
|
|
221
|
+
} catch {
|
|
222
|
+
// Diagnostics only — never fail an acquisition over it.
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// #489 step-03 (review C): REPAIR the diagnostic pid file periodically.
|
|
226
|
+
//
|
|
227
|
+
// Review measured the real severity of losing it: the library's directory still holds
|
|
228
|
+
// the authority, but a pre-#489 client reads ONLY this file — so if it disappears,
|
|
229
|
+
// that old client walks in and stays for B's whole lifetime, not for microseconds.
|
|
230
|
+
// Re-creating it on a beat bounds the exposure to one interval. Same cadence as the
|
|
231
|
+
// library's own refresh; `unref` so it can never keep a process alive.
|
|
232
|
+
const repair = setInterval(() => {
|
|
233
|
+
try {
|
|
234
|
+
if (readLegacyPid(lockFile) !== pid) writeFileSync(lockFile, `${pid}
|
|
235
|
+
`);
|
|
236
|
+
} catch {
|
|
237
|
+
// Diagnostic only.
|
|
238
|
+
}
|
|
239
|
+
}, (opts.timing ?? lockTimingOptions()).update);
|
|
240
|
+
repair.unref?.();
|
|
241
|
+
|
|
242
|
+
let released = false;
|
|
243
|
+
return {
|
|
244
|
+
ok: true,
|
|
245
|
+
release: () => {
|
|
246
|
+
// Idempotent: exit handlers can fire twice, and releasing twice must not throw.
|
|
247
|
+
if (released) return;
|
|
248
|
+
released = true;
|
|
249
|
+
clearInterval(repair);
|
|
250
|
+
try {
|
|
251
|
+
release();
|
|
252
|
+
} catch {
|
|
253
|
+
// Already gone / reclaimed elsewhere.
|
|
254
|
+
}
|
|
255
|
+
// #489 step-03 (review): remove the diagnostic pid file ONLY IF IT IS STILL OURS.
|
|
256
|
+
//
|
|
257
|
+
// Review asked for no unlink at all, to close this window: we release the
|
|
258
|
+
// library's lock, B acquires and writes ITS pid, and our unlink would delete B's
|
|
259
|
+
// file — after which an old client, which only reads that file, would miss B and
|
|
260
|
+
// start in parallel.
|
|
261
|
+
//
|
|
262
|
+
// Measured consequence of never unlinking, though: our own pid file survives our
|
|
263
|
+
// release, our process is still alive and STARTED BEFORE the file, so the
|
|
264
|
+
// start-time proof classifies us as a live legacy holder — and the folder is
|
|
265
|
+
// blocked for good. A DETERMINISTIC immortal lock, i.e. the #489 bug re-created by
|
|
266
|
+
// its own fix. Two tests caught it.
|
|
267
|
+
//
|
|
268
|
+
// So: compare, then unlink. The residual race is a read-then-unlink of a few
|
|
269
|
+
// microseconds whose worst outcome is a MISSING diagnostic file (bounded, and the
|
|
270
|
+
// library's directory still holds the authority), against a permanent block. The
|
|
271
|
+
// window is stated rather than hidden.
|
|
272
|
+
try {
|
|
273
|
+
if (readLegacyPid(lockFile) === pid) unlinkSync(lockFile);
|
|
274
|
+
} catch {
|
|
275
|
+
// Diagnostic only — never fail a release over it.
|
|
276
|
+
}
|
|
277
|
+
},
|
|
278
|
+
};
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* Hold the release of a lock we ACTUALLY took — nothing else (#489 step-02).
|
|
283
|
+
*
|
|
284
|
+
* This exists because of a real bug in the pre-#489 code, found in review and
|
|
285
|
+
* verified on `e67a2e0`:
|
|
286
|
+
*
|
|
287
|
+
* - `process.on("exit", … releaseLock())` was registered at MODULE scope
|
|
288
|
+
* (`viber-channel.ts:335`), therefore BEFORE `acquireLock()` ran (l.523);
|
|
289
|
+
* - the old `releaseLock()` did an UNCONDITIONAL `unlinkSync(LOCK_FILE)`;
|
|
290
|
+
* - a launch that LOST the race went on to `process.exit(1)`, firing that handler.
|
|
291
|
+
*
|
|
292
|
+
* So every accidental duplicate ended by DELETING THE WINNER'S LOCK: the folder went
|
|
293
|
+
* unprotected while the winner was still running, and a third launch acquired freely.
|
|
294
|
+
* The invariant this lock exists to hold (#166/#311) was broken by its own error path.
|
|
295
|
+
*
|
|
296
|
+
* The slot makes the guarantee structural instead of incidental: `releaseIfHeld()` is
|
|
297
|
+
* a no-op unless an acquisition succeeded and handed over its release. Kept HERE, in
|
|
298
|
+
* the tested module, rather than as a bare `let` in the entry point that no test can
|
|
299
|
+
* import.
|
|
300
|
+
*/
|
|
301
|
+
export interface LockSlot {
|
|
302
|
+
/** Record the release of a successful acquisition. */
|
|
303
|
+
adopt: (release: () => void) => void;
|
|
304
|
+
/** Release only if we hold it. Idempotent — exit handlers can fire twice. */
|
|
305
|
+
releaseIfHeld: () => void;
|
|
306
|
+
/** True once an acquisition has been adopted and not yet released. */
|
|
307
|
+
isHeld: () => boolean;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
export function createLockSlot(): LockSlot {
|
|
311
|
+
let release: (() => void) | null = null;
|
|
312
|
+
return {
|
|
313
|
+
adopt: (r) => {
|
|
314
|
+
release = r;
|
|
315
|
+
},
|
|
316
|
+
releaseIfHeld: () => {
|
|
317
|
+
const r = release;
|
|
318
|
+
release = null;
|
|
319
|
+
r?.();
|
|
320
|
+
},
|
|
321
|
+
isHeld: () => release !== null,
|
|
322
|
+
};
|
|
323
|
+
}
|
package/lib/bridge_tool_host.ts
CHANGED
|
@@ -27,6 +27,7 @@ import {
|
|
|
27
27
|
messageAgent,
|
|
28
28
|
sendMessage,
|
|
29
29
|
type AgentToolsContext,
|
|
30
|
+
toMcpToolResult,
|
|
30
31
|
type AgentToolResult,
|
|
31
32
|
} from "./agent_tools.js";
|
|
32
33
|
import { CONTRACT_STYLE_NEUTRAL, CONTRACT_STYLE_TOOLS } from "./channel_instructions.js";
|
|
@@ -91,8 +92,19 @@ export const TOOL_DEFS = [
|
|
|
91
92
|
"List the OTHER agents (instances) you can DM. Returns each agent's id, " +
|
|
92
93
|
"label, runtime kind, and whether it is online. Use it to find an id before message_agent. " +
|
|
93
94
|
"Normally project-scoped; an ORCHESTRATOR instance (#307) sees all the owner's projects, " +
|
|
94
|
-
"each entry tagged with a `project` field."
|
|
95
|
-
|
|
95
|
+
"each entry tagged with a `project` field. " +
|
|
96
|
+
"FILTER instead of listing everything (#501): `online: true` for live agents only, " +
|
|
97
|
+
"`label_prefix` for one team (e.g. \"501-\"). Both are applied server-side. " +
|
|
98
|
+
"If the presence source is unreachable the `online` filter is DROPPED and the answer " +
|
|
99
|
+
"says so — an empty list would wrongly read as \"no agent is alive\".",
|
|
100
|
+
inputSchema: {
|
|
101
|
+
type: "object",
|
|
102
|
+
properties: {
|
|
103
|
+
online: { type: "boolean", description: "Only agents currently seen online." },
|
|
104
|
+
label_prefix: { type: "string", description: 'Only labels starting with this, e.g. "501-".' },
|
|
105
|
+
},
|
|
106
|
+
additionalProperties: false,
|
|
107
|
+
},
|
|
96
108
|
},
|
|
97
109
|
{
|
|
98
110
|
name: "message_agent",
|
|
@@ -145,7 +157,7 @@ export async function dispatchBridgeTool(
|
|
|
145
157
|
): Promise<AgentToolResult> {
|
|
146
158
|
let result: AgentToolResult;
|
|
147
159
|
if (name === "list_agents") {
|
|
148
|
-
result = await listAgents(opts.ctx);
|
|
160
|
+
result = await listAgents(opts.ctx, args);
|
|
149
161
|
} else if (name === "message_agent") {
|
|
150
162
|
result = await messageAgent(opts.ctx, args);
|
|
151
163
|
} else if (name === "send_message") {
|
|
@@ -166,7 +178,7 @@ function buildServer(opts: BridgeToolHostOptions): Server {
|
|
|
166
178
|
mcp.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
167
179
|
const args = (request.params.arguments ?? {}) as Record<string, unknown>;
|
|
168
180
|
try {
|
|
169
|
-
return await dispatchBridgeTool(request.params.name, args, opts);
|
|
181
|
+
return toMcpToolResult(await dispatchBridgeTool(request.params.name, args, opts));
|
|
170
182
|
} catch (err) {
|
|
171
183
|
// The shared lib rethrows ConversationTokenExpiredError (host-specific
|
|
172
184
|
// teardown). In the bridge we surface it as a tool error rather than
|
package/lib/claude_tool_defs.ts
CHANGED
|
@@ -71,10 +71,24 @@ export const CLAUDE_TOOL_DEFS = [
|
|
|
71
71
|
"Use this to find the id of an agent (e.g. 'Codex Review') before calling message_agent. " +
|
|
72
72
|
"Normally scoped to this project; if the owner designated this instance an ORCHESTRATOR " +
|
|
73
73
|
"(#307), the list covers ALL the owner's projects and each entry carries a `project` field. " +
|
|
74
|
-
"You are never in the list."
|
|
74
|
+
"You are never in the list. " +
|
|
75
|
+
"FILTER instead of listing everything (#501): `online: true` returns only agents the server " +
|
|
76
|
+
'currently sees online, `label_prefix` only those whose label starts with it (e.g. "501-" ' +
|
|
77
|
+
"for one team). Both are applied server-side, so an unfiltered call in a loop is the " +
|
|
78
|
+
"expensive path. If the presence source is unreachable the `online` filter is DROPPED and " +
|
|
79
|
+
'the answer says so — an empty list would wrongly read as "no agent is alive".',
|
|
75
80
|
inputSchema: {
|
|
76
81
|
type: "object" as const,
|
|
77
|
-
properties: {
|
|
82
|
+
properties: {
|
|
83
|
+
online: {
|
|
84
|
+
type: "boolean",
|
|
85
|
+
description: "Only agents the server currently sees online.",
|
|
86
|
+
},
|
|
87
|
+
label_prefix: {
|
|
88
|
+
type: "string",
|
|
89
|
+
description: 'Only agents whose label starts with this prefix, e.g. "501-".',
|
|
90
|
+
},
|
|
91
|
+
},
|
|
78
92
|
required: [],
|
|
79
93
|
additionalProperties: false,
|
|
80
94
|
},
|
package/lib/instance.ts
CHANGED
|
@@ -193,6 +193,16 @@ export interface AcquiredInstance {
|
|
|
193
193
|
* 3. `auth.json` `instance_token` — reused (durable identity, D4 → same id).
|
|
194
194
|
* 4. register a NEW instance with the stable project_token, persisted back to
|
|
195
195
|
* auth.json (self-healing migration — no forced manual reconnect).
|
|
196
|
+
*
|
|
197
|
+
* ⚠ #501 DEPENDS ON THE PRIORITY ORDER ABOVE. vibe-master's readiness gate proves
|
|
198
|
+
* an agent came up by watching for an instance id that did NOT exist before the
|
|
199
|
+
* launch. Only route 2 mints one: routes 1 and 3 REUSE an id, so a launch that
|
|
200
|
+
* took either would be reported as "never registered" while being perfectly
|
|
201
|
+
* healthy — a false red. vibe-master's spawns take route 2 because
|
|
202
|
+
* `vibe-master/lib/terminal.ts` scrubs `VIBER_INSTANCE_TOKEN`/`_ID` from the
|
|
203
|
+
* child environment; the guarantee lives there, and this comment exists so a
|
|
204
|
+
* change here is not made unaware of it. Locked by the route-precedence tests in
|
|
205
|
+
* `test/instance.test.ts`.
|
|
196
206
|
*/
|
|
197
207
|
export async function acquireInstance(
|
|
198
208
|
baseUrl: string,
|