viber-channel 0.8.4 → 0.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -50,6 +50,89 @@ export function sessionFilePath(
50
50
  return join(dir, `channel-session-${suffix}.json`);
51
51
  }
52
52
 
53
+ /**
54
+ * Session file path for a **named** channel agent (#400), namespaced by
55
+ * `(baseUrl, projectId, name)` instead of the server `instance_id`.
56
+ *
57
+ * Why a separate key: an agent launched by vibe-master carries a stable
58
+ * `VIBER_CHANNEL_LABEL` (its name) but re-registers a FRESH, non-persisted
59
+ * `instance_id` on every start (`acquireInstance` path 2). Keying the handle by
60
+ * that ephemeral id means it is never found on the next start → the channel
61
+ * mints a new conversation every time (the #400 accumulation). The NAME is the
62
+ * only key that survives across re-spawns and MCP reconnects, so a named agent
63
+ * namespaces its handle by name and finds — and reattaches — its previous
64
+ * conversation.
65
+ *
66
+ * `projectId` is folded in so the same name in two different projects (or
67
+ * worktrees connected to different projects) never collide on one handle file.
68
+ * The plain (un-named) channel keeps `sessionFilePath` keyed by the persisted
69
+ * `instance_id` — do NOT merge the two paths (#269 isolation on plain channel).
70
+ */
71
+ export function sessionFilePathForName(
72
+ baseUrl: string,
73
+ projectId: number,
74
+ name: string,
75
+ dir: string,
76
+ ): string {
77
+ const suffix = createHash("sha256")
78
+ .update(`${baseUrl}\n${projectId}\n${name}`)
79
+ .digest("hex")
80
+ .slice(0, 8);
81
+ return join(dir, `channel-session-${suffix}.json`);
82
+ }
83
+
84
+ /**
85
+ * Pick the session-handle path for this channel process (#400), applying the
86
+ * named-vs-plain rule in ONE testable place (no `as` cast at the call site).
87
+ *
88
+ * A NAMED agent — a non-empty `channelLabel` (`VIBER_CHANNEL_LABEL`) AND no
89
+ * env-injected `instanceToken` (`VIBER_INSTANCE_TOKEN`) — keys its handle by the
90
+ * stable NAME (`sessionFilePathForName`), because `acquireInstance` path 2
91
+ * re-registers a fresh instance_id every start. Otherwise (plain channel, or an
92
+ * orchestrator-injected stable instance token = path 1) the handle stays keyed
93
+ * by the durable `instanceKey` (`sessionFilePath`) — #269 isolation preserved.
94
+ *
95
+ * Precedence mirrors `acquireInstance`: an env instance token (path 1) wins over
96
+ * a label (path 2), so `instanceToken` present ⇒ instance-keyed even if a label
97
+ * is also set.
98
+ */
99
+ /**
100
+ * The stable resume NAME for a named agent session (#400), or `null` for a plain
101
+ * / orchestrator-injected session.
102
+ *
103
+ * A NAMED agent = a non-empty `channelLabel` (VIBER_CHANNEL_LABEL) AND no
104
+ * env-injected `instanceToken` (VIBER_INSTANCE_TOKEN). Such an agent re-registers
105
+ * a fresh instance_id every start (acquireInstance path 2), so its stable
106
+ * identity is the NAME: it keys its handle by name AND resumes by name. An
107
+ * env-injected token (path 1, orchestrator) means a durable instance → NOT a
108
+ * named-resume session (returns null). Single source of truth for both the
109
+ * handle-namespacing and resume-by-name decisions; returns the trimmed name so
110
+ * callers avoid re-deriving it (and no `as` cast is needed).
111
+ */
112
+ export function namedAgentLabel(
113
+ channelLabel: string | undefined,
114
+ instanceToken: string | undefined,
115
+ ): string | null {
116
+ const name = channelLabel?.trim();
117
+ if (name === undefined || name === "") return null;
118
+ if ((instanceToken ?? "").trim() !== "") return null;
119
+ return name;
120
+ }
121
+
122
+ export function resolveSessionFilePath(opts: {
123
+ baseUrl: string;
124
+ projectId: number;
125
+ channelLabel: string | undefined;
126
+ instanceToken: string | undefined;
127
+ instanceKey: string;
128
+ dir: string;
129
+ }): string {
130
+ const name = namedAgentLabel(opts.channelLabel, opts.instanceToken);
131
+ return name !== null
132
+ ? sessionFilePathForName(opts.baseUrl, opts.projectId, name, opts.dir)
133
+ : sessionFilePath(opts.baseUrl, opts.instanceKey, opts.dir);
134
+ }
135
+
53
136
  /**
54
137
  * Read a persisted handle from `path`.
55
138
  *
@@ -36,17 +36,29 @@ export function buildControlStreamUrl(baseUrl: string, instanceId: string): stri
36
36
  *
37
37
  * The agent beats on a fixed interval so the server's #311 sweeper never purges a
38
38
  * live-but-idle agent: a server→client SSE keepalive can't prove this process is
39
- * still alive behind the Cloudflare tunnel, but this beat can. Best-effort —
40
- * `true` on 2xx, `false` on any non-2xx / network error. Bounded by a per-POST
41
- * timeout so a hung tunnel socket can't stack beats (mirrors #298).
39
+ * still alive behind the Cloudflare tunnel, but this beat can. Bounded by a
40
+ * per-POST timeout so a hung tunnel socket can't stack beats (mirrors #298).
41
+ *
42
+ * #400 (recadrage): when `conversationId` is given, the server RENEWS this
43
+ * instance's conversation lease and reports `lease_active` — `true` (still held),
44
+ * `false` (a new holder took over → the channel must close), or `null`
45
+ * (indeterminate: not checked, or a D1 error — never a false "still yours").
42
46
  */
47
+ export interface HeartbeatResult {
48
+ /** true on 2xx, false on any non-2xx / network error. */
49
+ ok: boolean;
50
+ /** Lease state from the server (#400): true=held, false=lost, null=indeterminate. */
51
+ leaseActive: boolean | null;
52
+ }
53
+
43
54
  export async function sendInstanceHeartbeat(
44
55
  baseUrl: string,
45
56
  instanceId: string,
46
57
  instanceToken: string,
58
+ conversationId: string | null = null,
47
59
  timeoutMs = 20_000,
48
60
  fetchImpl: typeof fetch = fetch,
49
- ): Promise<boolean> {
61
+ ): Promise<HeartbeatResult> {
50
62
  try {
51
63
  const resp = await fetchImpl(`${baseUrl}/api/instances/${instanceId}/heartbeat`, {
52
64
  method: "POST",
@@ -55,12 +67,22 @@ export async function sendInstanceHeartbeat(
55
67
  "Content-Type": "application/json",
56
68
  ...cfAccessHeaders(),
57
69
  },
58
- body: "{}",
70
+ body: JSON.stringify(conversationId !== null ? { conversation_id: conversationId } : {}),
59
71
  signal: AbortSignal.timeout(timeoutMs),
60
72
  });
61
- return resp.ok;
73
+ if (!resp.ok) return { ok: false, leaseActive: null };
74
+ let leaseActive: boolean | null = null;
75
+ try {
76
+ const body = (await resp.json()) as { lease_active?: boolean | null };
77
+ if (body.lease_active === true || body.lease_active === false) {
78
+ leaseActive = body.lease_active;
79
+ }
80
+ } catch {
81
+ // no/invalid body — leave indeterminate
82
+ }
83
+ return { ok: true, leaseActive };
62
84
  } catch {
63
- return false;
85
+ return { ok: false, leaseActive: null };
64
86
  }
65
87
  }
66
88
 
@@ -40,6 +40,25 @@ export class ConversationMintError extends Error {
40
40
  }
41
41
  }
42
42
 
43
+ /**
44
+ * #400 (recadrage) — an EXPLICIT `--conversation <id>` (VIBER_TARGET_CONVERSATION_ID)
45
+ * could not be loaded, and the operator asked for THAT conversation specifically:
46
+ * we must NOT invent a different one (that would recreate the very session clutter
47
+ * #400 removes). The channel catches this and exits cleanly (clear message,
48
+ * non-zero) — NEVER a fresh mint, never an uncaught stack trace. Distinct from
49
+ * name-resume, which is a heuristic allowed to fall back to a fresh mint.
50
+ */
51
+ export class ExplicitTargetUnavailableError extends Error {
52
+ conversationId: string;
53
+ reason: string;
54
+ constructor(conversationId: string, reason: string) {
55
+ super(`Requested conversation ${conversationId} ${reason}`);
56
+ this.name = "ExplicitTargetUnavailableError";
57
+ this.conversationId = conversationId;
58
+ this.reason = reason;
59
+ }
60
+ }
61
+
43
62
  export async function mintConversation(
44
63
  baseUrl: string,
45
64
  projectId: number,
@@ -121,13 +140,54 @@ export async function reattachConversation(
121
140
  }
122
141
  throw new ReattachFailedError(code);
123
142
  }
143
+ // Non-409 non-2xx (401 revoked, 503 lease-unavailable, other 5xx): throw a
144
+ // typed error carrying the status so the caller can react (401 → re-register;
145
+ // 5xx → explicit-target hard-fail / graceful fallback). #400 (recadrage).
124
146
  const detail = await resp.text();
125
- throw new Error(`Failed to reattach conversation (HTTP ${resp.status}): ${detail}`);
147
+ throw new ConversationMintError(resp.status, detail);
126
148
  }
127
149
 
128
150
  return (await resp.json()) as ConversationMintResponse;
129
151
  }
130
152
 
153
+ export interface ResumeCandidate {
154
+ conversation_id: string;
155
+ created_at: number;
156
+ }
157
+
158
+ /**
159
+ * GET /api/instances/resume-candidates?label=<name> (#400) — ask the server
160
+ * which channel-attached conversations carry `label` (the resume-by-name lookup
161
+ * for a named agent that lost its local handle). Auth: instance_token; the owner
162
+ * + project are derived server-side from the validated instance, so `label` is
163
+ * the only input. Returns the candidate list (may be empty). NEVER throws for an
164
+ * empty result — only for a real HTTP/network failure, which the caller treats
165
+ * as "no resume → mint".
166
+ */
167
+ export async function resumeCandidatesByLabel(
168
+ baseUrl: string,
169
+ instanceToken: string,
170
+ fingerprint: string,
171
+ label: string,
172
+ ): Promise<ResumeCandidate[]> {
173
+ const resp = await fetch(
174
+ `${baseUrl}/api/instances/resume-candidates?label=${encodeURIComponent(label)}`,
175
+ {
176
+ method: "GET",
177
+ headers: {
178
+ Authorization: `Bearer ${instanceToken}`,
179
+ "X-Client-Fingerprint": fingerprint,
180
+ ...cfAccessHeaders(),
181
+ },
182
+ },
183
+ );
184
+ if (!resp.ok) {
185
+ throw new Error(`resume-candidates lookup failed (HTTP ${resp.status})`);
186
+ }
187
+ const body = (await resp.json()) as { candidates?: ResumeCandidate[] };
188
+ return Array.isArray(body.candidates) ? body.candidates : [];
189
+ }
190
+
131
191
  export interface RefreshTokenResponse {
132
192
  conversation_token: string;
133
193
  expires_at: number;
@@ -232,8 +292,14 @@ export function defaultLabel(folderPath: string): string {
232
292
  }
233
293
 
234
294
  /**
235
- * Default reattach window in seconds — matches the server's CONVERSATION_TTL_SECONDS.
236
- * A handle older than this points at an expired conversation that cannot be reattached.
295
+ * Legacy reattach window in seconds (kept for tests / explicit opt-in).
296
+ *
297
+ * NOTE (#400): this is NO LONGER the channel default. The server `handleReattach`
298
+ * reattaches a conversation of any age (only the token expires; the row persists),
299
+ * so the channel now defaults to an UNBOUNDED window (always attempt reattach) and
300
+ * only caps it when `VIBER_REATTACH_WINDOW_SECONDS` is set to a finite value. The
301
+ * old comment ("a handle older than this points at an expired conversation") was
302
+ * incorrect and was the reason a channel minted a fresh conversation after 1h.
237
303
  */
238
304
  export const DEFAULT_REATTACH_WINDOW_SECONDS = 3600;
239
305
 
@@ -268,8 +334,68 @@ export async function acquireConversation(
268
334
  targetConversationId?: string | null,
269
335
  reattachWindowSeconds: number = DEFAULT_REATTACH_WINDOW_SECONDS,
270
336
  log: (msg: string) => void = (msg) => process.stderr.write(msg),
337
+ /**
338
+ * #400 resume-by-name key: the conversation label a NAMED agent minted with
339
+ * (VIBER_CONVERSATION_LABEL). When non-null AND no local handle reattaches, the
340
+ * server is asked which channel-attached conversation carries this name; a
341
+ * UNIQUE match is reattached, 0/collision → mint. null (plain channel /
342
+ * timestamped label) disables the lookup. Last param so existing positional
343
+ * callers/tests (…, window, log) are unaffected.
344
+ */
345
+ resumeLabel: string | null = null,
346
+ /**
347
+ * #400 (recadrage) — EXPLICIT per-agent target conversation id (env
348
+ * `VIBER_TARGET_CONVERSATION_ID`, injected by vibe-master `spawn --conversation`).
349
+ * HIGHEST priority: bind THIS conversation by id. On success → use it. On ANY
350
+ * failure (conversation_in_use / revoked-or-missing / 5xx server-unavailable) →
351
+ * HARD FAIL via ExplicitTargetUnavailableError (the channel exits cleanly): NO
352
+ * mint, NO heuristic fallback — the operator asked for a SPECIFIC conversation,
353
+ * so inventing a different one would recreate the clutter #400 removes. A 401
354
+ * (token revoked) still propagates so the outer loop re-registers. (Contrast:
355
+ * name-resume is a heuristic allowed to mint gracefully on failure.) Last param
356
+ * → existing positional callers unaffected.
357
+ */
358
+ envTargetConversationId: string | null = null,
271
359
  ): Promise<ConversationMintResponse> {
272
360
  const nowSeconds = Math.floor(Date.now() / 1000);
361
+
362
+ // #400 (recadrage) — EXPLICIT per-agent target wins over every heuristic AND
363
+ // never invents a different conversation. The operator asked for THIS id: bind
364
+ // it, or FAIL cleanly (no fresh mint — that would recreate the clutter #400
365
+ // removes). A 401 still propagates so the outer loop re-registers.
366
+ if (envTargetConversationId !== null) {
367
+ log(`[viber-channel] startup: explicit target ${envTargetConversationId} (VIBER_TARGET_CONVERSATION_ID), attempting reattach\n`);
368
+ try {
369
+ const result = await reattachConversation(baseUrl, projectId, instanceToken, fingerprint, envTargetConversationId);
370
+ log(`[viber-channel] startup: explicit-target reattach OK, conv_id=${result.conversation_id}\n`);
371
+ try {
372
+ writeHandle(sessionPath, result.conversation_id);
373
+ } catch (writeErr) {
374
+ log(`[viber-channel] Warning: failed to write session handle after explicit-target reattach: ${String(writeErr)}\n`);
375
+ }
376
+ return result;
377
+ } catch (err) {
378
+ // 401 (token revoked) → let the outer handler re-register and retry.
379
+ if (err instanceof ConversationMintError && err.status === 401) throw err;
380
+ let reason: string;
381
+ if (err instanceof ReattachFailedError) {
382
+ reason =
383
+ err.code === "conversation_in_use"
384
+ ? "is already in use by a live agent"
385
+ : err.code === "reattach_revoked" || err.code === "reattach_not_found"
386
+ ? "was revoked or no longer exists"
387
+ : `could not be reattached (${err.code})`;
388
+ } else if (err instanceof ConversationMintError && err.status >= 500) {
389
+ reason = "could not be verified — the server was unavailable (try again)";
390
+ } else {
391
+ reason = `could not be loaded (${String(err)})`;
392
+ }
393
+ // HARD FAIL — no mint, no heuristic fallback. The channel catches this and
394
+ // exits cleanly with a clear message (never a fresh conversation).
395
+ throw new ExplicitTargetUnavailableError(envTargetConversationId, reason);
396
+ }
397
+ }
398
+
273
399
  const handle = readHandle(sessionPath);
274
400
 
275
401
  // Branch on `handle` directly so TS narrows it to non-null inside — readHandle
@@ -292,10 +418,13 @@ export async function acquireConversation(
292
418
  } catch (err) {
293
419
  // Reattach failure is NEVER fatal. Everything (ReattachFailedError 409,
294
420
  // unexpected 5xx, network errors from a thrown fetch) falls through to
295
- // mint below, which surfaces a 401 to the caller if the token is dead.
421
+ // the resume-by-name lookup then mint below (which surfaces a 401 to the
422
+ // caller if the token is dead). A 409 (handle points at a dead/revoked
423
+ // conv) is the realistic case: resume-by-name then finds the LIVE conv of
424
+ // the same name instead of minting.
296
425
  const reason = err instanceof ReattachFailedError ? err.code : String(err);
297
- log(`[viber-channel] startup: reattach failed (${reason}), falling back to mint\n`);
298
- // Fall through to mint below
426
+ log(`[viber-channel] startup: reattach failed (${reason}), falling back to resume-by-name / mint\n`);
427
+ // Fall through to resume-by-name / mint below
299
428
  }
300
429
  } else {
301
430
  log(`[viber-channel] startup: session handle stale (conv_id=${handle.conversation_id}, age=${handleAge}s >= ${reattachWindowSeconds}s window), minting fresh\n`);
@@ -304,8 +433,72 @@ export async function acquireConversation(
304
433
  log(`[viber-channel] startup: no session handle found, minting fresh\n`);
305
434
  }
306
435
 
307
- // Mint a new conversation
308
- const result = await mintConversation(baseUrl, projectId, instanceToken, fingerprint, label, targetConversationId ?? null);
436
+ // #400 resume-by-name (handle-absent path): a named agent re-registers a fresh
437
+ // instance_id each start, so on a new machine / cleared tmp / MCP reconnect the
438
+ // handle above is missing. Ask the server which channel-attached conversation
439
+ // carries this name; a UNIQUE match is reattached (reuse instead of mint), a
440
+ // collision (≥2) or no match falls through to mint — never guess. Lookup
441
+ // failure is non-fatal → mint.
442
+ if (resumeLabel !== null) {
443
+ try {
444
+ const candidates = await resumeCandidatesByLabel(baseUrl, instanceToken, fingerprint, resumeLabel);
445
+ if (candidates.length === 1) {
446
+ const targetId = candidates[0].conversation_id;
447
+ log(`[viber-channel] startup: resume-by-name "${resumeLabel}" → unique match ${targetId}, attempting reattach\n`);
448
+ try {
449
+ const result = await reattachConversation(baseUrl, projectId, instanceToken, fingerprint, targetId);
450
+ log(`[viber-channel] startup: resume-by-name reattach OK, conv_id=${result.conversation_id}\n`);
451
+ try {
452
+ writeHandle(sessionPath, result.conversation_id);
453
+ } catch (writeErr) {
454
+ log(`[viber-channel] Warning: failed to write session handle after resume-by-name reattach: ${String(writeErr)}\n`);
455
+ }
456
+ return result;
457
+ } catch (err) {
458
+ const reason = err instanceof ReattachFailedError ? err.code : String(err);
459
+ log(`[viber-channel] startup: resume-by-name reattach failed (${reason}), minting fresh\n`);
460
+ }
461
+ } else if (candidates.length > 1) {
462
+ const ids = candidates.map((c) => c.conversation_id).join(", ");
463
+ log(`[viber-channel] startup: resume-by-name "${resumeLabel}" → label collision (${candidates.length} candidates: ${ids}), minting fresh\n`);
464
+ } else {
465
+ log(`[viber-channel] startup: resume-by-name "${resumeLabel}" → no match, minting fresh\n`);
466
+ }
467
+ } catch (err) {
468
+ log(`[viber-channel] startup: resume-by-name lookup failed (${String(err)}), minting fresh\n`);
469
+ }
470
+ }
471
+
472
+ // Mint a new conversation.
473
+ //
474
+ // #400 sticky-target safety: target_conversation_id in auth.json is STICKY
475
+ // (written once by the reconnect/attach flow, never cleared — connect.ts). When
476
+ // that target is later REVOKED (e.g. by the #400 batch cleanup, or unit
477
+ // revocation), the server routes target → handleReattach → 409 reattach_revoked
478
+ // / reattach_not_found, and mintConversation throws ConversationMintError. Since
479
+ // viber-channel.ts only catches 401 (re-register), an un-tolerated 409 here
480
+ // would CRASH startup — and the two #400 features (reconnect + cleanup) combine
481
+ // to make it reachable. So: on a 409 from the TARGET path, fall back to a fresh
482
+ // mint WITHOUT the (dead) target. A 401 still propagates (token revoked →
483
+ // caller re-registers); other errors still propagate.
484
+ let result: ConversationMintResponse;
485
+ try {
486
+ result = await mintConversation(baseUrl, projectId, instanceToken, fingerprint, label, targetConversationId ?? null);
487
+ } catch (err) {
488
+ if (
489
+ targetConversationId != null &&
490
+ err instanceof ConversationMintError &&
491
+ (err.status === 409 || err.status >= 500)
492
+ ) {
493
+ // 409 = dead/revoked/in-use target; 5xx = transient server/lease fault. The
494
+ // auth.json target is a STICKY folder-level hint (not an explicit operator
495
+ // request), so a graceful fresh mint is correct — never crash on it (Opus).
496
+ log(`[viber-channel] startup: sticky target ${targetConversationId} not reattachable (${err.message}), minting fresh without target\n`);
497
+ result = await mintConversation(baseUrl, projectId, instanceToken, fingerprint, label, null);
498
+ } else {
499
+ throw err;
500
+ }
501
+ }
309
502
  log(`[viber-channel] startup: mint OK, conv_id=${result.conversation_id}\n`);
310
503
  try {
311
504
  writeHandle(sessionPath, result.conversation_id);
package/lib/heartbeat.ts CHANGED
@@ -13,10 +13,18 @@
13
13
  * was removed in #315 once "Active" derived from instance liveness.)
14
14
  */
15
15
 
16
- import { sendInstanceHeartbeat } from "./control_stream.ts";
16
+ import { type HeartbeatResult, sendInstanceHeartbeat } from "./control_stream.ts";
17
17
 
18
18
  export const DEFAULT_HEARTBEAT_INTERVAL_MS = 25_000;
19
19
 
20
+ /**
21
+ * #400 (recadrage) — consecutive INDETERMINATE lease beats tolerated before the
22
+ * channel closes. A definite `lease_active:false` closes immediately; `null`
23
+ * (D1 error / no body) is tolerated up to this many beats so a transient fault
24
+ * doesn't kill a live agent, then we close (the margin Codex asked for).
25
+ */
26
+ export const DEFAULT_MAX_INDETERMINATE_LEASE_BEATS = 3;
27
+
20
28
  /**
21
29
  * Resolve the beat interval from the environment (seconds), defaulting to 25s.
22
30
  * Must stay comfortably under the server timeout (VIBER_CHANNEL_PRESENCE_TIMEOUT,
@@ -44,8 +52,26 @@ export interface InstanceHeartbeatOptions {
44
52
  log?: (line: string) => void;
45
53
  setTimer?: (fn: () => void, ms: number) => unknown;
46
54
  clearTimer?: (handle: unknown) => void;
55
+ /**
56
+ * #400 — the conversation this agent is attached to, read each tick so the beat
57
+ * renews its lease. null (plain channel / not yet bound) → no lease renewal.
58
+ */
59
+ getConversationId?: () => string | null;
60
+ /**
61
+ * #400 — called when the agent has DEFINITELY lost its conversation lease
62
+ * (a new holder took over, or too many consecutive indeterminate beats). The
63
+ * channel closes to avoid two live agents on one conversation.
64
+ */
65
+ onLeaseLost?: () => void;
66
+ /** Consecutive indeterminate beats tolerated before closing (default 3). */
67
+ maxIndeterminateLeaseBeats?: number;
47
68
  /** Injectable beat fn for tests. Defaults to sendInstanceHeartbeat. */
48
- beat?: (baseUrl: string, instanceId: string, instanceToken: string) => Promise<boolean>;
69
+ beat?: (
70
+ baseUrl: string,
71
+ instanceId: string,
72
+ instanceToken: string,
73
+ conversationId: string | null,
74
+ ) => Promise<HeartbeatResult>;
49
75
  }
50
76
 
51
77
  /**
@@ -64,15 +90,41 @@ export function startInstanceHeartbeat(opts: InstanceHeartbeatOptions): Heartbea
64
90
  opts.clearTimer ??
65
91
  ((h: unknown) => clearInterval(h as ReturnType<typeof setInterval>));
66
92
  const beat = opts.beat ?? sendInstanceHeartbeat;
93
+ const getConversationId = opts.getConversationId ?? (() => null);
94
+ const maxIndeterminate =
95
+ opts.maxIndeterminateLeaseBeats ?? DEFAULT_MAX_INDETERMINATE_LEASE_BEATS;
67
96
 
68
97
  let inFlight = false;
98
+ let indeterminateStreak = 0;
99
+ let leaseLostFired = false;
100
+ const fireLeaseLost = (why: string) => {
101
+ if (leaseLostFired) return;
102
+ leaseLostFired = true;
103
+ log(`[viber-channel] conversation lease lost (${why}) — closing channel\n`);
104
+ opts.onLeaseLost?.();
105
+ };
106
+
69
107
  const handle = setTimer(() => {
70
108
  if (inFlight) return;
71
109
  inFlight = true;
72
- void beat(opts.baseUrl, opts.instanceId, opts.getInstanceToken())
73
- .then((ok) => {
74
- if (!ok) {
110
+ const conversationId = getConversationId();
111
+ void beat(opts.baseUrl, opts.instanceId, opts.getInstanceToken(), conversationId)
112
+ .then((result) => {
113
+ if (!result.ok) {
75
114
  log("[viber-channel] instance heartbeat failed (will retry next interval)\n");
115
+ return; // a failed beat never reached the server — not a lease signal
116
+ }
117
+ // #400 lease state — only meaningful when we asked about a conversation.
118
+ if (conversationId === null) return;
119
+ if (result.leaseActive === false) {
120
+ fireLeaseLost("a new holder took over");
121
+ } else if (result.leaseActive === null) {
122
+ indeterminateStreak += 1;
123
+ if (indeterminateStreak >= maxIndeterminate) {
124
+ fireLeaseLost(`${indeterminateStreak} consecutive indeterminate beats`);
125
+ }
126
+ } else {
127
+ indeterminateStreak = 0; // true → still ours, reset the margin
76
128
  }
77
129
  })
78
130
  .finally(() => {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "viber-channel",
3
- "version": "0.8.4",
3
+ "version": "0.8.6",
4
4
  "description": "Voice + text MCP channel between a Claude Code session and the Viber UI (https://viber.dgypx.dev). Push transcripts to Claude; send_message tool delivers text back to the UI.",
5
5
  "type": "module",
6
6
  "bin": {
package/viber-channel.ts CHANGED
@@ -31,13 +31,13 @@ import { mkdirSync, writeFileSync, readFileSync, unlinkSync } from "node:fs";
31
31
  import { join } from "node:path";
32
32
  import { runConnect } from "./lib/connect.ts";
33
33
  import { lockFilePath, lockSessionId } from "./lib/lockfile.ts";
34
- import { sessionFilePath, clearHandle } from "./lib/channel_session.ts";
34
+ import { resolveSessionFilePath, namedAgentLabel, clearHandle } from "./lib/channel_session.ts";
35
35
  import { clientFingerprint } from "./lib/fingerprint.ts";
36
36
  import { MessageDedup } from "./lib/message_dedup.ts";
37
37
  import {
38
38
  acquireConversation,
39
39
  ConversationMintError,
40
- DEFAULT_REATTACH_WINDOW_SECONDS,
40
+ ExplicitTargetUnavailableError,
41
41
  type ConversationMintResponse,
42
42
  } from "./lib/conversation.ts";
43
43
  import { startInstanceHeartbeat } from "./lib/heartbeat.ts";
@@ -169,22 +169,28 @@ const LOCK_FILE = lockFilePath(LOCK_BASE_URL, LOCK_FINGERPRINT, LOCK_SESSION_ID,
169
169
  // acquired (#269) — it is namespaced by the server-issued instance, not the
170
170
  // local fingerprint/sessionId. Declared as a `let` near the mint below.
171
171
 
172
- // VIBER_REATTACH_WINDOW_SECONDS overrides the default TTL-based window.
173
- // A handle older than this cannot point at a live conversation — skip reattach.
174
- // Parsed once, mirroring the VIBER_REFRESH_LEAD_SECONDS pattern below: validate
175
- // finite && >= 0, log invalid→ignore, log override, fall back to the default.
172
+ // Reattach window (#400): NEUTRALISED BY DEFAULT. The server `handleReattach`
173
+ // is the contract of truth — it reattaches any non-revoked channel conversation
174
+ // of the right owner/project REGARDLESS OF AGE (only the token expires, the row
175
+ // persists). The old 1h default (DEFAULT_REATTACH_WINDOW_SECONDS) was a purely
176
+ // client-side guard whose "stale handle points at an expired conversation"
177
+ // rationale is FALSE per the server code, and it was the reason a channel minted
178
+ // a fresh conversation after 1h instead of reattaching. Default is now
179
+ // unbounded (Infinity) → always attempt reattach when a handle exists; the
180
+ // server rejects (409) if the conversation is genuinely gone → fall back to mint.
181
+ // VIBER_REATTACH_WINDOW_SECONDS still caps the window if a finite value is set.
176
182
  const REATTACH_WINDOW_SECONDS: number = (() => {
177
183
  const raw = process.env.VIBER_REATTACH_WINDOW_SECONDS;
178
- if (raw === undefined) return DEFAULT_REATTACH_WINDOW_SECONDS;
184
+ if (raw === undefined) return Number.POSITIVE_INFINITY;
179
185
  const parsed = Number.parseInt(raw, 10);
180
186
  if (!Number.isFinite(parsed) || parsed < 0) {
181
187
  process.stderr.write(
182
- `[viber-channel] Invalid VIBER_REATTACH_WINDOW_SECONDS=${raw}, ignoring.\n`
188
+ `[viber-channel] Invalid VIBER_REATTACH_WINDOW_SECONDS=${raw}, ignoring (window unbounded).\n`
183
189
  );
184
- return DEFAULT_REATTACH_WINDOW_SECONDS;
190
+ return Number.POSITIVE_INFINITY;
185
191
  }
186
192
  process.stderr.write(
187
- `[viber-channel] reattach window overridden to ${parsed}s via VIBER_REATTACH_WINDOW_SECONDS\n`
193
+ `[viber-channel] reattach window capped to ${parsed}s via VIBER_REATTACH_WINDOW_SECONDS\n`
188
194
  );
189
195
  return parsed;
190
196
  })();
@@ -659,14 +665,60 @@ try {
659
665
  // token that instance_key may fall back to (Codex review P1). Undefined → ""
660
666
  // → the guard is a no-op (delivers everything), which is the safe default.
661
667
  OWN_INSTANCE_ID = acquired.instance_id ?? "";
662
- SESSION_FILE = sessionFilePath(LOCK_BASE_URL, acquired.instance_key, LOCK_DIR);
663
668
 
664
- process.stderr.write(`[viber-channel] startup: acquiring conversation (project=${auth.project_id}, target=${auth.target_conversation_id ?? "null"}, session=${SESSION_FILE})\n`);
669
+ // #400: a NAMED agent (VIBER_CHANNEL_LABEL, no env-injected VIBER_INSTANCE_TOKEN)
670
+ // re-registers a FRESH instance_id on every start (acquireInstance path 2), so
671
+ // namespacing the handle by instance_id never finds it on the next start → mint
672
+ // loop. resolveSessionFilePath keys such an agent by the STABLE NAME (project-
673
+ // scoped) so a re-spawn / MCP reconnect reattaches its previous conversation;
674
+ // the plain channel (or a path-1 injected token) keeps the instance_id key —
675
+ // #269 isolation. Selection logic lives in the pure helper (unit-tested).
676
+ const computeSessionFile = (instanceKey: string): string =>
677
+ resolveSessionFilePath({
678
+ baseUrl: LOCK_BASE_URL,
679
+ projectId: auth.project_id,
680
+ channelLabel: process.env.VIBER_CHANNEL_LABEL,
681
+ instanceToken: process.env.VIBER_INSTANCE_TOKEN,
682
+ instanceKey,
683
+ dir: LOCK_DIR,
684
+ });
685
+ SESSION_FILE = computeSessionFile(acquired.instance_key);
686
+
687
+ // #400 resume-by-name lookup key: EXACTLY the label the mint writes into
688
+ // conversations.label, i.e. VIBER_CONVERSATION_LABEL (see `label` above:
689
+ // `VIBER_CONVERSATION_LABEL ?? defaultLabel(cwd)`). Only meaningful when it is
690
+ // explicitly set AND this is a named agent — otherwise the mint uses the
691
+ // timestamped defaultLabel, which is NOT resumable, so the lookup is disabled
692
+ // (null). No VIBER_CHANNEL_LABEL fallback: that would query a value the mint
693
+ // never wrote (Opus review B — query exactly what the mint writes).
694
+ const namedSession = namedAgentLabel(process.env.VIBER_CHANNEL_LABEL, process.env.VIBER_INSTANCE_TOKEN) !== null;
695
+ const explicitConvLabel = process.env.VIBER_CONVERSATION_LABEL?.trim();
696
+ const RESUME_LABEL: string | null = namedSession && explicitConvLabel ? explicitConvLabel : null;
697
+
698
+ // #400 (recadrage) — EXPLICIT per-agent target conversation, injected by
699
+ // vibe-master `spawn --conversation <id>` / the TUI "load" picker. Highest
700
+ // priority in acquireConversation: bind THIS conversation by id (never via the
701
+ // shared folder auth.json). null when not spawned onto a specific conversation.
702
+ const ENV_TARGET_CONVERSATION_ID = process.env.VIBER_TARGET_CONVERSATION_ID?.trim() || null;
703
+
704
+ process.stderr.write(`[viber-channel] startup: acquiring conversation (project=${auth.project_id}, target=${auth.target_conversation_id ?? "null"}, session=${SESSION_FILE}, resume_label=${RESUME_LABEL ?? "none"})\n`);
665
705
 
666
706
  // Join the conversation with the instance_token. A 401 means the instance was
667
707
  // revoked (the token is durable — no expiry case): re-register a fresh instance
668
708
  // ONCE (a NEW instance_id, the correct post-revocation outcome) and retry. A
669
709
  // second failure falls through to the catch below.
710
+ // #400 (recadrage): an explicit `--conversation <id>` that can't be loaded is a
711
+ // CLEAN, non-zero exit with a readable message — never a fresh conversation,
712
+ // never an uncaught stack trace.
713
+ const exitExplicitTargetUnavailable = (err: ExplicitTargetUnavailableError): never => {
714
+ process.stderr.write(
715
+ `[viber-channel] cannot load requested conversation ${err.conversationId}: ${err.reason}. Agent NOT started.\n`,
716
+ );
717
+ scheduler?.cancel();
718
+ releaseLock();
719
+ process.exit(1);
720
+ };
721
+
670
722
  let minted: ConversationMintResponse;
671
723
  try {
672
724
  minted = await acquireConversation(
@@ -678,8 +730,14 @@ try {
678
730
  label,
679
731
  auth.target_conversation_id ?? null,
680
732
  REATTACH_WINDOW_SECONDS,
733
+ undefined, // log — use the default stderr writer
734
+ RESUME_LABEL,
735
+ ENV_TARGET_CONVERSATION_ID,
681
736
  );
682
737
  } catch (err) {
738
+ if (err instanceof ExplicitTargetUnavailableError) {
739
+ exitExplicitTargetUnavailable(err);
740
+ }
683
741
  if (err instanceof ConversationMintError && err.status === 401) {
684
742
  process.stderr.write(
685
743
  `[viber-channel] instance token rejected (401 revoked) — re-registering a fresh instance once\n`,
@@ -695,17 +753,30 @@ try {
695
753
  OWN_INSTANCE_ID = reg.instance_id;
696
754
  // #398: re-attach the fresh (post-revocation) instance to its team.
697
755
  await maybeAttachTeam(BASE_URL, auth.project_id, auth.project_token, auth.client_fingerprint, reg.instance_id);
698
- SESSION_FILE = sessionFilePath(LOCK_BASE_URL, reg.instance_id, LOCK_DIR);
699
- minted = await acquireConversation(
700
- SESSION_FILE,
701
- BASE_URL,
702
- auth.project_id,
703
- instanceToken,
704
- auth.client_fingerprint,
705
- label,
706
- auth.target_conversation_id ?? null,
707
- REATTACH_WINDOW_SECONDS,
708
- );
756
+ // #400: keep the SAME namespacing rule after a post-revocation re-register.
757
+ // A named agent stays keyed by its stable name (not the new instance_id),
758
+ // so its handle survives; the plain channel re-keys to the fresh id.
759
+ SESSION_FILE = computeSessionFile(reg.instance_id);
760
+ try {
761
+ minted = await acquireConversation(
762
+ SESSION_FILE,
763
+ BASE_URL,
764
+ auth.project_id,
765
+ instanceToken,
766
+ auth.client_fingerprint,
767
+ label,
768
+ auth.target_conversation_id ?? null,
769
+ REATTACH_WINDOW_SECONDS,
770
+ undefined, // log — use the default stderr writer
771
+ RESUME_LABEL,
772
+ ENV_TARGET_CONVERSATION_ID,
773
+ );
774
+ } catch (retryErr) {
775
+ if (retryErr instanceof ExplicitTargetUnavailableError) {
776
+ exitExplicitTargetUnavailable(retryErr);
777
+ }
778
+ throw retryErr;
779
+ }
709
780
  } else {
710
781
  throw err;
711
782
  }
@@ -754,6 +825,16 @@ try {
754
825
  baseUrl: BASE_URL,
755
826
  instanceId: OWN_INSTANCE_ID,
756
827
  getInstanceToken: () => INSTANCE_TOKEN,
828
+ // #400 (recadrage): renew this agent's conversation lease each beat. If the
829
+ // server reports the lease was taken over (a new holder), or too many
830
+ // consecutive indeterminate beats, close the channel — never stay a second
831
+ // live agent on a conversation another agent now owns.
832
+ getConversationId: () => CONVERSATION_ID || null,
833
+ onLeaseLost: () => {
834
+ scheduler?.cancel();
835
+ releaseLock();
836
+ process.exit(0);
837
+ },
757
838
  });
758
839
  // Stop beating once the control stream ends (revoke / 401 / abort) so a
759
840
  // doomed instance doesn't keep beating (Opus review).
@@ -7,6 +7,7 @@
7
7
  * Codex's final assistant reply back into Viber.
8
8
  */
9
9
  import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process";
10
+ import { createHash } from "node:crypto";
10
11
  import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
11
12
  import { dirname, join } from "node:path";
12
13
  import { authFilePath, loadAuth } from "./lib/auth.ts";
@@ -242,33 +243,81 @@ function buildReply(receivedContent: string, msg: ConversationMessage | undefine
242
243
  // the shared runConversationStream now owns token refresh, so the bridge no
243
244
  // longer wires it directly.
244
245
 
245
- // #280 step-12: the state file is per AGENT. Several agents on one machine each own
246
- // their own file, so concurrent read-modify-write of a single shared JSON can't drop
247
- // another agent's thread entry (Codex review P2-A). An explicit
248
- // VIBER_CODEX_BRIDGE_STATE override still wins (single file, caller's responsibility).
249
- function stateFilePath(instanceId?: string, cwd: string = process.cwd()): string {
246
+ // #409: monotonic (perf) timing for diagnosing codex spawn / first-response
247
+ // latency. Structured, greppable `timing:` lines — the user's "codex is slow" is
248
+ // largely INHERENT (app-server initialize + first model turn), so we MEASURE where
249
+ // the time goes rather than guess. NOT a behaviour change.
250
+ const nowMs = (): number => performance.now();
251
+ // #409 (Codex): classify a conversation's turn as first vs subsequent and RESERVE
252
+ // the "first" label atomically (before any await), so two near-simultaneous
253
+ // messages can't both be logged `which=first`. Pure/testable.
254
+ export function classifyTurn(
255
+ seen: Set<string>,
256
+ convId: string,
257
+ ): "first" | "subsequent" {
258
+ if (seen.has(convId)) return "subsequent";
259
+ seen.add(convId);
260
+ return "first";
261
+ }
262
+ function logTiming(phase: string, ms: number, extra = ""): void {
263
+ process.stderr.write(
264
+ `[viber-codex-bridge] timing: ${phase} ms=${Math.round(ms)}${extra ? ` ${extra}` : ""}\n`,
265
+ );
266
+ }
267
+
268
+ // #414: the persistent thread identity. #280 keyed the thread by the server-issued
269
+ // instanceId so multiple agents sharing one conversation each get their own Codex
270
+ // thread — but the instanceId is FRESH on every register (#269), so a same-name
271
+ // re-spawn never found its predecessor's thread (thread/start instead of resume,
272
+ // runtime memory lost). Fix: when an EXPLICIT VIBER_CODEX_BRIDGE_LABEL is set (the
273
+ // stable logical name — vibe-master always sets it to the agent id), key by
274
+ // sha256(label) so a re-spawn under the same name resumes; distinct labels stay
275
+ // isolated (#280 preserved). No explicit label → fall back to the instanceId (no
276
+ // resume, but #280 isolation kept — the DEFAULT `Codex bridge • <cwd>` label is NOT
277
+ // a unique identity, and an empty label must never collapse everyone onto
278
+ // hash("")). The hash is filename-safe (hex) and doesn't leak the label.
279
+ export function threadIdentity(
280
+ instanceId: string,
281
+ label: string | undefined,
282
+ ): { key: string; fileSuffix: string } {
283
+ const trimmed = label?.trim();
284
+ if (trimmed) {
285
+ const h = createHash("sha256").update(trimmed, "utf8").digest("hex");
286
+ return { key: `label:${h}`, fileSuffix: `label-${h}` };
287
+ }
288
+ return { key: `instance:${instanceId}`, fileSuffix: `instance-${instanceId}` };
289
+ }
290
+
291
+ // #280 step-12 / #414: the state file is per AGENT IDENTITY (stable label when set,
292
+ // else instanceId). It lives in the workspace's `.viber` dir (per-folder), so two
293
+ // folders — even a cross-project #307 pair using the same explicit label — write
294
+ // DIFFERENT files; the fingerprint in stateKey is defence-in-depth. Concurrent RMW
295
+ // of one file across agents can't drop another's entry (distinct identities →
296
+ // distinct files). An explicit VIBER_CODEX_BRIDGE_STATE override still wins.
297
+ function stateFilePath(
298
+ instanceId: string,
299
+ label: string | undefined,
300
+ cwd: string = process.cwd(),
301
+ ): string {
250
302
  const override = process.env.VIBER_CODEX_BRIDGE_STATE;
251
303
  if (override !== undefined && override.trim() !== "") return override;
252
- // Full instance id in the filename (server-issued opaque hex): a truncated id
253
- // could collide and reintroduce the shared-file write race this split avoids.
254
- const name =
255
- instanceId && instanceId.trim() !== ""
256
- ? `codex-bridge-threads-${instanceId}.json`
257
- : "codex-bridge-threads.json";
258
- return join(dirname(authFilePath(cwd)), name);
304
+ const { fileSuffix } = threadIdentity(instanceId, label);
305
+ return join(dirname(authFilePath(cwd)), `codex-bridge-threads-${fileSuffix}.json`);
259
306
  }
260
307
 
261
- // #280 step-12: the key is per (conversation, AGENT), not per conversation. With
262
- // several agents sharing one conversation, a conversation-only key would make them
263
- // resume/clobber the same Codex thread. Including the instance id gives each agent
264
- // its own persisted thread.
308
+ // #280/#414: the key is per (project, folder, conversation, AGENT IDENTITY). The
309
+ // conversationId is KEPT (a same agent on TWO conversations must not share one
310
+ // runtime thread — cross-conversation contamination); the fingerprint scopes it
311
+ // per folder; the identity is the stable label (or instanceId fallback).
265
312
  export function stateKey(
266
313
  projectId: number,
267
314
  fingerprint: string,
268
315
  conversationId: string,
269
316
  instanceId: string,
317
+ label: string | undefined,
270
318
  ): string {
271
- return `${projectId}:${fingerprint}:${conversationId}:${instanceId}`;
319
+ const { key } = threadIdentity(instanceId, label);
320
+ return `${projectId}:${fingerprint}:${conversationId}:${key}`;
272
321
  }
273
322
 
274
323
  function readThreadState(path: string): ThreadState {
@@ -584,11 +633,13 @@ class CodexAppServer {
584
633
  }
585
634
 
586
635
  async initialize(): Promise<void> {
636
+ const t0 = nowMs(); // #409: measure the app-server init (a big share of spawn latency)
587
637
  await this.request("initialize", {
588
638
  clientInfo: { name: "viber-codex-bridge", title: "Viber Codex Bridge", version: "0.1.0" },
589
639
  capabilities: { experimentalApi: true },
590
640
  });
591
641
  this.notify("initialized");
642
+ logTiming("codex.initialize", nowMs() - t0);
592
643
  }
593
644
 
594
645
  async ensureThread(options: {
@@ -598,18 +649,25 @@ class CodexAppServer {
598
649
  instanceKey: string;
599
650
  newThread: boolean;
600
651
  }): Promise<string> {
601
- const path = stateFilePath(options.instanceKey);
652
+ // #414: prefer the STABLE explicit label over the fresh instanceId so a
653
+ // same-name re-spawn resumes its own thread (see threadIdentity).
654
+ const label = process.env.VIBER_CODEX_BRIDGE_LABEL;
655
+ const path = stateFilePath(options.instanceKey, label);
602
656
  const state = readThreadState(path);
603
657
  const key = stateKey(
604
658
  options.auth.project_id,
605
659
  options.fingerprint,
606
660
  options.conversationId,
607
661
  options.instanceKey,
662
+ label,
608
663
  );
609
664
  const existing = state.threads[key]?.thread_id;
665
+ const t0 = nowMs(); // #409: measure thread resume/start (the first-DM cost)
666
+ let resumeAttempted = false;
610
667
 
611
668
  if (existing && !options.newThread) {
612
669
  try {
670
+ resumeAttempted = true;
613
671
  process.stderr.write(`[viber-codex-bridge] codex: resuming persisted thread ${existing}\n`);
614
672
  const resumed = await this.request("thread/resume", {
615
673
  threadId: existing,
@@ -620,7 +678,11 @@ class CodexAppServer {
620
678
  developerInstructions: AGENT_INSTRUCTIONS,
621
679
  excludeTurns: true,
622
680
  });
623
- return extractThreadId(resumed);
681
+ // #409 (Codex): validate/extract the id BEFORE logging success — a malformed
682
+ // response must NOT be logged as resume then again as resume-failed→start.
683
+ const resumedId = extractThreadId(resumed);
684
+ logTiming("ensureThread", nowMs() - t0, "outcome=resume");
685
+ return resumedId;
624
686
  } catch (err) {
625
687
  process.stderr.write(
626
688
  `[viber-codex-bridge] codex: resume failed (${safeErrorMessage(err)}), starting a fresh thread\n`,
@@ -651,6 +713,11 @@ class CodexAppServer {
651
713
  };
652
714
  writeThreadState(path, state);
653
715
  process.stderr.write(`[viber-codex-bridge] codex: persisted thread ${threadId} in ${path}\n`);
716
+ logTiming(
717
+ "ensureThread",
718
+ nowMs() - t0,
719
+ `outcome=${resumeAttempted ? "resume-failed→start" : "start"}`,
720
+ );
654
721
  return threadId;
655
722
  }
656
723
 
@@ -1166,9 +1233,15 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<void
1166
1233
 
1167
1234
  // codex's per-conversation adapter for the shared runConversationStream:
1168
1235
  // ensureThread (keyed by conversation+agent) then the tracked, queued codex turn.
1236
+ // #409: which conversations have already run their FIRST model turn, so we can
1237
+ // report the (expensive) first turn distinctly from the fast subsequent ones.
1238
+ const firstTurnDone = new Set<string>();
1169
1239
  const makeCodexAdapter = (codex: CodexAppServer, instanceKey: string): MakeRunTurn =>
1170
1240
  async (convId, runtime) => {
1241
+ // #409: time join → thread ready (the eager ensureThread on a fresh join).
1242
+ const tJoin = nowMs();
1171
1243
  let threadPromise = threadCache.get(convId);
1244
+ const cached = threadPromise !== undefined;
1172
1245
  if (!threadPromise) {
1173
1246
  threadPromise = codex
1174
1247
  .ensureThread({ auth, fingerprint, conversationId: convId, instanceKey, newThread: options.newThread })
@@ -1179,8 +1252,22 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<void
1179
1252
  threadCache.set(convId, threadPromise);
1180
1253
  }
1181
1254
  const codexThreadId = await threadPromise;
1255
+ if (!cached) logTiming("join→thread-ready", nowMs() - tJoin, `conv=${convId.slice(0, 8)}`);
1182
1256
  process.stderr.write(`${LOG_PREFIX} codex thread ready for ${convId} (thread ${codexThreadId})\n`);
1183
- return makeCodexRunTurn(codex, codexThreadId, runtime, options, queue, tracker);
1257
+ const runTurn = makeCodexRunTurn(codex, codexThreadId, runtime, options, queue, tracker);
1258
+ // #409: wrap to measure the FIRST model turn (cold: the biggest perceived
1259
+ // latency) apart from subsequent turns. Pure instrumentation, same behaviour.
1260
+ return async (content, msg, signal) => {
1261
+ // #409 (Codex): reserve first/subsequent BEFORE the await so concurrent
1262
+ // turns can't both be classified `first`.
1263
+ const which = classifyTurn(firstTurnDone, convId);
1264
+ const tTurn = nowMs();
1265
+ try {
1266
+ return await runTurn(content, msg, signal);
1267
+ } finally {
1268
+ logTiming("turn", nowMs() - tTurn, `which=${which} conv=${convId.slice(0, 8)}`);
1269
+ }
1270
+ };
1184
1271
  };
1185
1272
 
1186
1273
  if (options.awaitInvite) {
@@ -1259,10 +1346,20 @@ export async function main(argv: string[] = process.argv.slice(2)): Promise<void
1259
1346
  logCodexSafety();
1260
1347
  process.stderr.write(`${LOG_PREFIX} ready: joined ${res.minted.conversation_id} as instance ${res.instanceKey.slice(0, 8)}\n`);
1261
1348
  // #315: no control stream here, so beat the instance liveness directly.
1349
+ // #400 (recadrage): this bridge holds a lease on its ONE joined conversation
1350
+ // (acquired via reattachConversation → handleReattach). RENEW it each beat, or
1351
+ // it expires after the TTL while the agent is alive and the conversation
1352
+ // falsely frees → another agent could load onto it. On a definite loss (a new
1353
+ // holder took over), shut down so we never double-attach.
1262
1354
  const ihb = startInstanceHeartbeat({
1263
1355
  baseUrl: BASE_URL,
1264
1356
  instanceId: res.instanceKey,
1265
1357
  getInstanceToken: () => res.instanceToken,
1358
+ getConversationId: () => res.minted.conversation_id,
1359
+ onLeaseLost: () =>
1360
+ shutdown.abort(
1361
+ new BridgeShutdownError("conversation lease lost (taken over)", 0),
1362
+ ),
1266
1363
  });
1267
1364
  shutdown.signal.addEventListener("abort", () => ihb.stop(), { once: true });
1268
1365
  try {