viber-channel 0.8.6 → 0.8.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -46,6 +46,21 @@ export interface ConversationMessage {
46
46
  sender_instance_id?: string | null;
47
47
  }
48
48
 
49
+ /**
50
+ * Persistent cross-run dedup watermark for ONE conversation (#429, mirror of
51
+ * `DmWatermark` in dm_stream.ts). Holds the highest message id already DELIVERED
52
+ * (replied to) in a prior run of this conversation's stream. The caller keeps one
53
+ * per `convId` and passes it back on every re-join, so it SURVIVES the
54
+ * abort+restart cycle `runConversationStream` does on a re-pushed join — unlike
55
+ * the run-scoped `handledIds` (fresh + empty each run). Without it the first
56
+ * await-invite pass re-forwards (re-replies to) the most recent trigger on every
57
+ * re-spawn/rejoin. Mutated in place. `0` = nothing delivered yet.
58
+ */
59
+ export interface BridgeWatermark {
60
+ /** Highest delivered (replied-to) message id (0 = nothing delivered yet). */
61
+ lastId: number;
62
+ }
63
+
49
64
  /** One live conversation stream's mutable runtime state (token rotates). */
50
65
  export interface ConversationRuntime {
51
66
  id: string;
@@ -333,6 +348,32 @@ export function decideTurnPostAction(opts: {
333
348
  return "auto-post";
334
349
  }
335
350
 
351
+ /**
352
+ * #429 cross-run dedup, mirror of dm_stream's `forwardMessage` guard (L146-147).
353
+ * True when this id was already DELIVERED in a prior run and must be skipped on a
354
+ * re-join. The D1 message id is `INTEGER PRIMARY KEY AUTOINCREMENT` (migration
355
+ * 0009) → strictly increasing, never reused, so anything at or below the watermark
356
+ * was already handled. `Number.isSafeInteger` (not `isFinite`) gate: an id beyond
357
+ * 2^53 — where `Number()` loses precision and could collide — falls through to
358
+ * normal handling instead of being mis-gated. No watermark / absent / NaN id →
359
+ * never skips. Pure + exported for unit tests (pattern `decideTurnPostAction`).
360
+ */
361
+ export function watermarkShouldSkip(numId: number, watermark?: BridgeWatermark): boolean {
362
+ return !!watermark && Number.isSafeInteger(numId) && numId <= watermark.lastId;
363
+ }
364
+
365
+ /**
366
+ * #429 advance the cross-run watermark after a SUCCESSFUL delivery (mirror of
367
+ * dm_stream L166-168). Monotone: only ever moves forward, so a brief overlap of
368
+ * two streams during supersession can't regress it. Same `Number.isSafeInteger`
369
+ * gate; a NaN/absent/unsafe id is never watermarked. Pure + exported.
370
+ */
371
+ export function watermarkAdvance(numId: number, watermark?: BridgeWatermark): void {
372
+ if (watermark && Number.isSafeInteger(numId) && numId > watermark.lastId) {
373
+ watermark.lastId = numId;
374
+ }
375
+ }
376
+
336
377
  /** Bearer + fingerprint + CF Access headers for a conversation request. */
337
378
  function conversationHeaders(conversationToken: string, fingerprint: string): Record<string, string> {
338
379
  return {
@@ -378,6 +419,13 @@ export interface ConversationStreamDeps {
378
419
  options: SseLoopOptions;
379
420
  /** Registry of live streams (keyed by conversation id) — re-join supersedes. */
380
421
  activeStreams: Map<string, AbortController>;
422
+ /**
423
+ * Registry of persistent per-conversation watermarks (#429). Keyed by
424
+ * conversation id. UNLIKE `activeStreams`, this MUST survive the abort+restart
425
+ * on re-join (never deleted in the stream's finally) — that persistence is the
426
+ * whole point of the cross-run dedup.
427
+ */
428
+ watermarks: Map<string, BridgeWatermark>;
381
429
  /** Global shutdown signal — aborting it stops every stream. */
382
430
  parentSignal: AbortSignal;
383
431
  baseUrl: string;
@@ -405,7 +453,7 @@ export async function runConversationStream(
405
453
  minted: ConversationMintResponse,
406
454
  deps: ConversationStreamDeps,
407
455
  ): Promise<void> {
408
- const { fingerprint, instanceKey, options, activeStreams, parentSignal, baseUrl, lockDir, logPrefix, sessionTag, makeRunTurn } = deps;
456
+ const { fingerprint, instanceKey, options, activeStreams, watermarks, parentSignal, baseUrl, lockDir, logPrefix, sessionTag, makeRunTurn } = deps;
409
457
  const convId = minted.conversation_id;
410
458
  if (parentSignal.aborted) return;
411
459
 
@@ -443,7 +491,15 @@ export async function runConversationStream(
443
491
  });
444
492
  const runTurn = await makeRunTurn(convId, runtime, ctrl.signal);
445
493
  process.stderr.write(`${logPrefix} stream ready for ${convId}\n`);
446
- await sseLoop({ minted, runtime, fingerprint, ownInstanceKey: instanceKey, options, signal: ctrl.signal, baseUrl, logPrefix, runTurn });
494
+ // Get-or-create the persistent watermark for this conversation. It lives in
495
+ // the caller-owned `watermarks` registry so it SURVIVES this stream's
496
+ // abort+restart on re-join (the finally below deliberately does NOT delete it).
497
+ let watermark = watermarks.get(convId);
498
+ if (!watermark) {
499
+ watermark = { lastId: 0 };
500
+ watermarks.set(convId, watermark);
501
+ }
502
+ await sseLoop({ minted, runtime, fingerprint, ownInstanceKey: instanceKey, options, signal: ctrl.signal, baseUrl, logPrefix, runTurn, watermark });
447
503
  } finally {
448
504
  parentSignal.removeEventListener("abort", onParentAbort);
449
505
  runtime.scheduler?.cancel();
@@ -484,8 +540,21 @@ export async function sseLoop(opts: {
484
540
  baseUrl: string;
485
541
  logPrefix: string;
486
542
  runTurn: RunTurn;
543
+ /**
544
+ * Persistent cross-run watermark for this conversation (#429). When present,
545
+ * a re-join skips re-replying to any message id already delivered in a prior
546
+ * run. Unused in step-01 (plumbing only); the guard/advance lands in step-02.
547
+ * `runConversationStream` always passes it; a direct test caller may omit it.
548
+ */
549
+ watermark?: BridgeWatermark;
550
+ /** Injectable for tests; defaults to global fetch (mirror dm_stream). */
551
+ fetchImpl?: typeof fetch;
552
+ /** Injectable catch-up fetch; defaults to messages.ts `fetchMessages`. */
553
+ fetchMessagesImpl?: typeof fetchMessages;
487
554
  }): Promise<void> {
488
- const { minted, runtime, fingerprint, ownInstanceKey, options, signal, baseUrl, logPrefix, runTurn } = opts;
555
+ const { minted, runtime, fingerprint, ownInstanceKey, options, signal, baseUrl, logPrefix, runTurn, watermark } = opts;
556
+ const fetchImpl = opts.fetchImpl ?? fetch;
557
+ const fetchMessagesImpl = opts.fetchMessagesImpl ?? fetchMessages;
489
558
  const sseUrl = buildSseUrl(minted.ws_url, minted.conversation_id);
490
559
  const ownPostedIds = new Set<string>();
491
560
 
@@ -563,7 +632,7 @@ export async function sseLoop(opts: {
563
632
  if (!idStr) return;
564
633
  if (ackedIds.has(idStr) || ackInFlightIds.has(idStr)) return;
565
634
  ackInFlightIds.add(idStr);
566
- void ackMessage(baseUrl, runtime.id, runtime.token, idStr, signal)
635
+ void ackMessage(baseUrl, runtime.id, runtime.token, idStr, signal, fetchImpl)
567
636
  .then((ok) => {
568
637
  if (ok) ackedIds.add(idStr);
569
638
  else process.stderr.write(`${logPrefix} ack failed for message ${idStr} (will retry on reconnect)\n`);
@@ -573,10 +642,18 @@ export async function sseLoop(opts: {
573
642
 
574
643
  async function handleInboundMessage(msg: ConversationMessage): Promise<boolean> {
575
644
  const idStr = msg.id !== undefined && msg.id !== null ? String(msg.id) : undefined;
645
+ // TRUTHY check (not `!== undefined`): idStr === "" must yield NaN, not
646
+ // Number("") === 0, so a blank id never satisfies the `<= lastId` guard.
647
+ const numId = idStr ? Number(idStr) : Number.NaN;
576
648
  // Confirm receipt to the web INDEPENDENTLY of handled/reply state, BEFORE the
577
649
  // handledIds early-return — so a previously-failed ACK is retried when this
578
650
  // message comes back through catch-up on reconnect (codex step-02 review).
579
651
  ackInboundReceipt(msg);
652
+ // #429 cross-run dedup: a re-join re-delivers the trigger; skip replying to
653
+ // anything already delivered in a prior run. Placed AFTER ackInboundReceipt
654
+ // so the #346 read-ACK still fires on a re-delivered trigger, only the reply
655
+ // is suppressed (mirror of dm_stream; handledIds still gates the intra-run dup).
656
+ if (watermarkShouldSkip(numId, watermark)) return false;
580
657
  if (idStr && handledIds.has(idStr)) return false;
581
658
  if (!shouldHandleMessage(msg, ownInstanceKey, ownPostedIds, options)) {
582
659
  if (idStr) handledIds.add(idStr); // skip decision is final — role won't change
@@ -590,6 +667,12 @@ export async function sseLoop(opts: {
590
667
  try {
591
668
  const aborted = await replyForTurn(msg.content!, msg);
592
669
  if (aborted) return false;
670
+ // #429 advance ONLY after a non-aborted reply (delivered — incl.
671
+ // suppress-outbound and skip-empty). INSIDE the try, after the abort
672
+ // check, never in catch/finally: a thrown reply (caught below) is a
673
+ // failed delivery → NO advance, so it retries on the next re-join
674
+ // (at-least-once, same asymmetry as dm_stream's watermark).
675
+ watermarkAdvance(numId, watermark);
593
676
  } catch (err) {
594
677
  if (err instanceof BridgeShutdownError) throw err;
595
678
  process.stderr.write(`${logPrefix} failed to handle message: ${safeErrorMessage(err)}\n`);
@@ -607,7 +690,7 @@ export async function sseLoop(opts: {
607
690
  // is the SSE/voice host (the CF tunnel in staging/prod). Using ws_url here
608
691
  // 404s whenever the SSE host differs from the API host (masked in dev where
609
692
  // they collapse to one origin). The SSE subscription still uses ws_url.
610
- msgs = await fetchMessages(baseUrl, minted.conversation_id, runtime.token);
693
+ msgs = await fetchMessagesImpl(baseUrl, minted.conversation_id, runtime.token);
611
694
  } catch (err) {
612
695
  if (err instanceof ConversationTokenExpiredError) throw err;
613
696
  process.stderr.write(`${logPrefix} catch-up fetch failed: ${safeErrorMessage(err)}\n`);
@@ -660,7 +743,7 @@ export async function sseLoop(opts: {
660
743
  requestHeaders.Accept = "text/event-stream";
661
744
  process.stderr.write(`${logPrefix} connecting to SSE: ${sseUrl}\n`);
662
745
 
663
- const resp = await fetch(sseUrl, { headers: requestHeaders, signal });
746
+ const resp = await fetchImpl(sseUrl, { headers: requestHeaders, signal });
664
747
  if (resp.status === 401) throw new ConversationTokenExpiredError();
665
748
  if (!resp.ok || !resp.body) throw new Error(`SSE connection failed: HTTP ${resp.status}`);
666
749
 
@@ -983,6 +1066,11 @@ export async function runAwaitInviteBridge(adapter: AwaitInviteAdapter): Promise
983
1066
  let bridgeLock: BridgeLock | null = null;
984
1067
  let controlStream: PersistentControlStream | null = null;
985
1068
  const activeStreams = new Map<string, AbortController>();
1069
+ // Persistent per-conversation watermarks (#429). Lives for the WHOLE bridge run
1070
+ // (never cleared per-stream) so a conversation's watermark survives the
1071
+ // abort+restart on re-join. Bounded growth: one `{lastId}` per convId seen —
1072
+ // acceptable (consensus reviewers, step-01 invariant).
1073
+ const watermarks = new Map<string, BridgeWatermark>();
986
1074
  const streamTasks = new Set<Promise<void>>();
987
1075
  let voiceBaseUrl = "";
988
1076
  let instanceKey = "";
@@ -998,6 +1086,7 @@ export async function runAwaitInviteBridge(adapter: AwaitInviteAdapter): Promise
998
1086
  instanceKey,
999
1087
  options,
1000
1088
  activeStreams,
1089
+ watermarks,
1001
1090
  parentSignal: shutdown.signal,
1002
1091
  baseUrl,
1003
1092
  lockDir,
package/lib/dm_stream.ts CHANGED
@@ -27,6 +27,21 @@ export interface DmMessage {
27
27
  artifact?: { content: string; format?: "markdown" | "code" | "json" | "html" } | null;
28
28
  }
29
29
 
30
+ /**
31
+ * Persistent cross-run dedup watermark for ONE DM conversation (#427). Holds the
32
+ * highest message id already DELIVERED to the agent in a prior `runDmStream`.
33
+ * The caller keeps one of these per conversation and passes it back on every
34
+ * (re-)join, so it SURVIVES the abort+restart cycle that `startDmStream` does on
35
+ * a re-pushed join. Mutated in place. Without it every re-join allocates a fresh
36
+ * `handledIds`, re-fetches the full history, and replays every already-delivered
37
+ * message (the "pong loop" the agent saw) — the run-scoped `handledIds` cannot
38
+ * catch that because it is empty at the start of each run.
39
+ */
40
+ export interface DmWatermark {
41
+ /** Highest delivered message id (0 = nothing delivered yet). */
42
+ lastId: number;
43
+ }
44
+
30
45
  /** Signature of the catch-up fetch (mirrors messages.ts `fetchMessages`). */
31
46
  export type FetchMessagesFn = (
32
47
  baseUrl: string,
@@ -66,6 +81,16 @@ export interface DmStreamOptions {
66
81
  * forward anything published while we were disconnected (#291 step-03).
67
82
  */
68
83
  fetchMessagesImpl?: FetchMessagesFn;
84
+ /**
85
+ * Cross-run dedup watermark (#427). The run-scoped `handledIds` only dedups
86
+ * WITHIN a single run; a re-join starts a fresh run with an empty set and the
87
+ * catch-up re-fetches the whole history. Supplying a caller-owned watermark
88
+ * (persisted across joins) suppresses ids already delivered in a prior run,
89
+ * while ids ABOVE the watermark still fire — so a message that arrived during
90
+ * detachment is preserved (#291 trigger-on-join). Advanced only after a
91
+ * successful delivery. Omitted → no cross-run dedup (legacy behaviour).
92
+ */
93
+ watermark?: DmWatermark;
69
94
  }
70
95
 
71
96
  /** Resolve after `ms`, or early when `signal` aborts. */
@@ -106,6 +131,20 @@ async function forwardMessage(
106
131
  ? String(msg.id)
107
132
  : undefined;
108
133
  if (idStr && handledIds.has(idStr)) return; // already forwarded (live or catch-up)
134
+ // Cross-run dedup (#427): the message id is `INTEGER PRIMARY KEY AUTOINCREMENT`
135
+ // in D1 (migration 0009), so the server assigns a STRICTLY INCREASING, never-
136
+ // reused id at insert — a genuinely new message always outranks everything
137
+ // already delivered. Hence any id at or below the watermark was already
138
+ // DELIVERED in a prior run and a re-join's full-history re-fetch must not replay
139
+ // it; ids ABOVE the watermark are new (incl. a trigger that arrived while
140
+ // detached) and still fire (#291). The AUTOINCREMENT guarantee is why a
141
+ // watermark is sufficient and a per-id Set is not needed (review: Opus/Codex).
142
+ // Use isSafeInteger (not isFinite) so an id beyond 2^53 — where Number() loses
143
+ // precision and could collide — falls through to normal handling + the in-run
144
+ // handledIds filter rather than silently mis-gating. The watermark only ever
145
+ // advances AFTER a successful delivery, so a NaN/absent id is never watermarked.
146
+ const numId = idStr !== undefined ? Number(idStr) : Number.NaN;
147
+ if (opts.watermark && Number.isSafeInteger(numId) && numId <= opts.watermark.lastId) return;
109
148
  // Content guard BEFORE consuming the id: a content-less payload must not burn
110
149
  // its id in handledIds, or a later (correctly-populated) catch-up of the same
111
150
  // id would be suppressed forever (review: Opus #4).
@@ -121,6 +160,12 @@ async function forwardMessage(
121
160
  try {
122
161
  await opts.onMessage(normalized);
123
162
  if (idStr) handledIds.add(idStr);
163
+ // Advance the cross-run watermark ONLY after a successful delivery (#427,
164
+ // Opus guard): a crash mid-handling leaves the id un-watermarked so the next
165
+ // catch-up retries it — at-least-once over at-most-once, same as handledIds.
166
+ if (opts.watermark && Number.isSafeInteger(numId) && numId > opts.watermark.lastId) {
167
+ opts.watermark.lastId = numId;
168
+ }
124
169
  } catch (handlerErr) {
125
170
  opts.log?.(
126
171
  `[dm-stream] onMessage handler error: ${handlerErr instanceof Error ? handlerErr.message : String(handlerErr)}\n`,
@@ -45,3 +45,14 @@ export function clientFingerprint(folderPath: string): string {
45
45
  const absPath = normalizePath(realpathSync(folderPath) as string);
46
46
  return createHash("sha256").update(`${machineId}:${absPath}`).digest("hex").slice(0, 32);
47
47
  }
48
+
49
+ /**
50
+ * Per-MACHINE fingerprint (#460 Phase 2, step-07) — the runner is a machine-level
51
+ * principal, so unlike clientFingerprint it is folder-INDEPENDENT. 32 hex chars,
52
+ * matching the server's bound format (`/^[0-9a-zA-Z_-]{8,128}$/`). Deterministic:
53
+ * re-enrolling the same machine yields the same fingerprint (→ atomic rotation).
54
+ */
55
+ export function machineFingerprint(): string {
56
+ const machineId = getMachineId();
57
+ return createHash("sha256").update(`runner:${machineId}`).digest("hex").slice(0, 32);
58
+ }
package/lib/heartbeat.ts CHANGED
@@ -15,7 +15,32 @@
15
15
 
16
16
  import { type HeartbeatResult, sendInstanceHeartbeat } from "./control_stream.ts";
17
17
 
18
- export const DEFAULT_HEARTBEAT_INTERVAL_MS = 25_000;
18
+ /**
19
+ * #478 — 25s → 60s to cut CF Worker requests (the instance heartbeat was ~59% of
20
+ * all Worker traffic, the single dominant source). Every beat is one
21
+ * `POST /api/instances/:id/heartbeat` = one Worker request per agent, forever.
22
+ *
23
+ * The three server-side windows this must stay under were raised in the same
24
+ * change so a 60s cadence keeps the SAME safety margins it had at 25s:
25
+ * - `VIBER_INSTANCE_PROBE_SUSPECT` (voice_pipeline.py) 8s → 75s. This is the
26
+ * one that actually caused the observed ~10-13s effective cadence: the
27
+ * sweeper probed (`beat_now`) every agent whose liveness was older than 8s,
28
+ * i.e. every 15s sweep, and each probe forces an EXTRA beat on top of the
29
+ * timer. With SUSPECT > the beat interval, a healthy agent is never probed
30
+ * and the timer is the only beat. Cost: a CRASHED agent is force-offlined in
31
+ * ~90-105s instead of ~13s — both probe stages are quantized to the 15s
32
+ * sweep, so arming AND deadline-enforcement each cost up to one sweep on top
33
+ * of SUSPECT + PROBE_TIMEOUT (Codex review of #478).
34
+ * - `VIBER_INSTANCE_CONTROL_PRESENCE_TIMEOUT` 90s → 150s, so a single missed
35
+ * beat still cannot force an agent offline (150 > 2 x 60).
36
+ * - `LEASE_TTL_SECONDS` (viber-api) 90s → 180s, same reason for the #400 lease.
37
+ *
38
+ * NB do NOT "de-dupe" the `beat_now` answer client-side: an unanswered probe
39
+ * force-offlines the agent at its deadline (`_enforce_instance_probes` requires
40
+ * a beat at-or-after `requested_at`), so skipping it would false-offline a live
41
+ * agent. The probe is fixed by not suspecting healthy agents, not by ignoring it.
42
+ */
43
+ export const DEFAULT_HEARTBEAT_INTERVAL_MS = 60_000;
19
44
 
20
45
  /**
21
46
  * #400 (recadrage) — consecutive INDETERMINATE lease beats tolerated before the
@@ -26,9 +51,11 @@ export const DEFAULT_HEARTBEAT_INTERVAL_MS = 25_000;
26
51
  export const DEFAULT_MAX_INDETERMINATE_LEASE_BEATS = 3;
27
52
 
28
53
  /**
29
- * Resolve the beat interval from the environment (seconds), defaulting to 25s.
30
- * Must stay comfortably under the server timeout (VIBER_CHANNEL_PRESENCE_TIMEOUT,
31
- * default 90s) so a single missed beat never trips the sweeper.
54
+ * Resolve the beat interval from the environment (seconds), defaulting to 60s.
55
+ * Must stay under HALF the server presence timeout
56
+ * (VIBER_INSTANCE_CONTROL_PRESENCE_TIMEOUT, default 150s since #478) so a single
57
+ * missed beat never trips the sweeper, and under VIBER_INSTANCE_PROBE_SUSPECT
58
+ * (default 75s) so a healthy agent is never `beat_now`-probed.
32
59
  */
33
60
  export function resolveHeartbeatIntervalMs(
34
61
  env: Record<string, string | undefined> = process.env,
package/lib/messages.ts CHANGED
@@ -112,10 +112,11 @@ export async function ackMessage(
112
112
  conversationToken: string,
113
113
  messageId: string | number,
114
114
  signal?: AbortSignal,
115
+ fetchImpl: typeof fetch = fetch,
115
116
  ): Promise<boolean> {
116
117
  const url = `${baseUrl}/api/conversations/${conversationId}/messages/${messageId}/ack`;
117
118
  try {
118
- const resp = await fetch(url, {
119
+ const resp = await fetchImpl(url, {
119
120
  method: "POST",
120
121
  headers: { Authorization: `Bearer ${conversationToken}`, ...cfAccessHeaders() },
121
122
  signal,
@@ -0,0 +1,210 @@
1
+ /**
2
+ * runner_enroll.ts — client side of the runner enrollment ceremony (#460 Phase 2,
3
+ * step-07/#472). Turns an enrollment URL into `.viber/runner.json` so the machine
4
+ * can run the resident spawn daemon.
5
+ *
6
+ * Invoked as:
7
+ * bunx viber-channel enroll-runner "https://viber.dgypx.dev/connect/runner/<claim_id>#<wait_token>"
8
+ *
9
+ * The owner opens the enrollment in the web (JWT-owner, C2), which yields the
10
+ * URL above carrying the claim_id (path) + the wait_token (fragment — a secret
11
+ * that gates the CAS bind so a leaked confirm-link can't be pre-bound). The
12
+ * client here:
13
+ * 1. generates its OWN runner_token (kept local, never sent — only its SHA-256);
14
+ * 2. binds (fingerprint + token hash) via the wait_token;
15
+ * 3. prints the confirm URL for the human to approve the shown fingerprint;
16
+ * 4. polls consume until the runner is materialized;
17
+ * 5. writes `.viber/runner.json` (0600) with the raw token.
18
+ */
19
+
20
+ import { appendFileSync, chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
21
+ import { createHash, randomBytes } from "node:crypto";
22
+ import { dirname, join } from "node:path";
23
+ import { machineFingerprint } from "./fingerprint.js";
24
+ import { cfAccessHeaders } from "./cfAccess.js";
25
+ import { isTrustedViberOrigin } from "./urls.js";
26
+
27
+ const POLL_INTERVAL_MS = 2000;
28
+ const MAX_POLL_DURATION_MS = 10 * 60 * 1000;
29
+
30
+ export interface ParsedEnrollUrl {
31
+ baseUrl: string;
32
+ claimId: string;
33
+ waitToken: string;
34
+ }
35
+
36
+ const CLAIM_ID_RE = /^[a-f0-9]{16,64}$/i; // server mints randomHex(16) = 32 hex
37
+ const WAIT_TOKEN_RE = /^[a-f0-9]{64}$/; // server mints randomHex(32) = 64 hex
38
+
39
+ /**
40
+ * Parse + HARDEN an enrollment URL (Codex review — adversarial inputs).
41
+ * Uses the WHATWG `URL` parser (canonicalizes `%2F`, `\`, `//`, ports) then:
42
+ * - rejects userinfo (`user:pass@host` — no credential smuggling);
43
+ * - requires https (or explicit localhost for dev);
44
+ * - accepts EXACTLY `/connect/runner/<claim_id>` (claim id format-checked);
45
+ * - takes the wait_token ONLY from the fragment, format-checked (64 hex).
46
+ * The fragment never reaches the server/logs (OAuth-implicit pattern); there is
47
+ * no query-param fallback on purpose (a `?wait_token=` would be logged).
48
+ */
49
+ export function parseEnrollUrl(input: string): ParsedEnrollUrl {
50
+ let url: URL;
51
+ try {
52
+ url = new URL(input);
53
+ } catch {
54
+ throw new Error(`Invalid URL: ${input}`);
55
+ }
56
+ if (url.username || url.password) {
57
+ throw new Error("Enrollment URL must not contain credentials (userinfo).");
58
+ }
59
+ // The origin must be a KNOWN Viber host (or configured VIBER_BASE_URL / dev
60
+ // localhost). Rejects a forged `https://evil.example/...` that would otherwise
61
+ // capture the runner bearer (Codex review — https alone is not enough).
62
+ if (!isTrustedViberOrigin(url.origin)) {
63
+ throw new Error(`Untrusted enrollment origin: ${url.origin}. Expected a Viber host.`);
64
+ }
65
+ // Canonical pathname match (URL already decoded %2F etc. — a smuggled
66
+ // `/connect/runner/x%2F..` would NOT match this exact shape).
67
+ const match = url.pathname.match(/^\/connect\/runner\/([^/]+)\/?$/);
68
+ if (!match || !CLAIM_ID_RE.test(match[1])) {
69
+ throw new Error(
70
+ "Not a runner enrollment URL. Expected: https://<host>/connect/runner/<claim_id>#<wait_token>",
71
+ );
72
+ }
73
+ const waitToken = url.hash.replace(/^#/, "");
74
+ if (!WAIT_TOKEN_RE.test(waitToken)) {
75
+ throw new Error("Enrollment URL is missing/invalid wait_token fragment (#<64-hex>).");
76
+ }
77
+ return { baseUrl: url.origin, claimId: match[1], waitToken };
78
+ }
79
+
80
+ function sha256Hex(input: string): string {
81
+ return createHash("sha256").update(input).digest("hex");
82
+ }
83
+
84
+ function sleep(ms: number): Promise<void> {
85
+ return new Promise((resolve) => setTimeout(resolve, ms));
86
+ }
87
+
88
+ function writeRunnerJson(
89
+ cwd: string,
90
+ data: { baseUrl: string; runnerId: string; runnerToken: string; machineFingerprint: string },
91
+ ): string {
92
+ const viberDir = join(cwd, ".viber");
93
+ mkdirSync(viberDir, { recursive: true });
94
+ const runnerPath = join(viberDir, "runner.json");
95
+ mkdirSync(dirname(runnerPath), { recursive: true });
96
+ const payload = {
97
+ schema_version: 1,
98
+ base_url: data.baseUrl,
99
+ runner_id: data.runnerId,
100
+ // The bearer — a local-code-execution capability. Stored client-side (the
101
+ // server keeps only its SHA-256). We request owner-only perms below, but note
102
+ // the real protection differs by OS (see chmod).
103
+ runner_token: data.runnerToken,
104
+ machine_fingerprint: data.machineFingerprint,
105
+ };
106
+ writeFileSync(runnerPath, JSON.stringify(payload, null, 2), { encoding: "utf-8", mode: 0o600 });
107
+ // Honesty (Opus review P1): `mode 0o600` + chmod are POSIX bits — effective on
108
+ // Linux/macOS (owner-read-only). On WINDOWS (JP's primary platform) NTFS uses
109
+ // ACLs, not POSIX bits, so this is essentially a no-op: the file inherits the
110
+ // user profile's ACL. That is the actual protection on Windows — not a claimed
111
+ // 0600. A stricter explicit ACL (icacls) is a documented follow-up.
112
+ try {
113
+ chmodSync(runnerPath, 0o600);
114
+ } catch {
115
+ /* best-effort — Windows / some filesystems reject chmod */
116
+ }
117
+
118
+ // Ensure `.viber/` is git-ignored so runner.json (the cleartext bearer) is not
119
+ // committed on a fresh external repo (mirrors connect.ts — Opus review P1).
120
+ const gitignorePath = join(cwd, ".gitignore");
121
+ let current = "";
122
+ if (existsSync(gitignorePath)) current = readFileSync(gitignorePath, "utf-8");
123
+ const alreadyIgnored = current
124
+ .split(/\r?\n/)
125
+ .some((line) => {
126
+ const t = line.trim();
127
+ return t === ".viber/" || t === ".viber";
128
+ });
129
+ if (!alreadyIgnored) {
130
+ const prefix = current === "" || current.endsWith("\n") ? "" : "\n";
131
+ appendFileSync(gitignorePath, `${prefix}.viber/\n`);
132
+ }
133
+ return runnerPath;
134
+ }
135
+
136
+ /**
137
+ * Run the enrollment ceremony. Writes `.viber/runner.json` on success; throws on
138
+ * any terminal outcome (rejected/superseded/expired/timeout).
139
+ */
140
+ export async function runRunnerEnroll(
141
+ enrollUrl: string,
142
+ cwd: string = process.cwd(),
143
+ label?: string,
144
+ ): Promise<void> {
145
+ const { baseUrl, claimId, waitToken } = parseEnrollUrl(enrollUrl);
146
+ const fingerprint = machineFingerprint();
147
+
148
+ // Client-generated bearer — kept here, only its hash leaves the machine.
149
+ const runnerToken = randomBytes(32).toString("hex");
150
+ const tokenHash = sha256Hex(runnerToken);
151
+
152
+ process.stderr.write(`[viber-channel] runner fingerprint=${fingerprint.slice(0, 8)}…\n`);
153
+
154
+ const bindResp = await fetch(`${baseUrl}/api/connect/runner/${claimId}/bind`, {
155
+ method: "POST",
156
+ redirect: "error", // parity with the daemon — never send the secret across a redirect
157
+ headers: { "Content-Type": "application/json", ...cfAccessHeaders() },
158
+ body: JSON.stringify({
159
+ wait_token: waitToken,
160
+ machine_fingerprint: fingerprint,
161
+ proposed_runner_token_hash: tokenHash,
162
+ ...(label ? { label } : {}),
163
+ }),
164
+ });
165
+ if (!bindResp.ok) {
166
+ const body = await bindResp.text().catch(() => "");
167
+ throw new Error(`Bind failed (HTTP ${bindResp.status}): ${body}`);
168
+ }
169
+
170
+ process.stdout.write(
171
+ `\nApprove this machine in your browser (confirm the shown fingerprint):\n` +
172
+ ` ${baseUrl}/runners/enroll/${claimId}\n\n` +
173
+ `Polling for confirmation (up to 10 minutes)…\n`,
174
+ );
175
+
176
+ const consumeUrl = `${baseUrl}/api/connect/runner/${claimId}/consume`;
177
+ const deadline = Date.now() + MAX_POLL_DURATION_MS;
178
+ while (Date.now() < deadline) {
179
+ await sleep(POLL_INTERVAL_MS);
180
+ const resp = await fetch(consumeUrl, {
181
+ method: "POST",
182
+ redirect: "error",
183
+ headers: { "Content-Type": "application/json", ...cfAccessHeaders() },
184
+ body: JSON.stringify({ wait_token: waitToken }),
185
+ });
186
+ if (resp.status === 200) {
187
+ const data = (await resp.json().catch(() => ({}))) as { runner_id?: string };
188
+ if (!data.runner_id || typeof data.runner_id !== "string") {
189
+ throw new Error("Consume succeeded but returned no runner_id — server response malformed.");
190
+ }
191
+ const path = writeRunnerJson(cwd, {
192
+ baseUrl,
193
+ runnerId: data.runner_id,
194
+ runnerToken,
195
+ machineFingerprint: fingerprint,
196
+ });
197
+ process.stdout.write(
198
+ `\n✓ Runner enrolled (id ${data.runner_id.slice(0, 8)}…). Wrote ${path}\n\n` +
199
+ `Start the runner daemon with:\n bunx viber-channel run-runner\n\n`,
200
+ );
201
+ return;
202
+ }
203
+ if (resp.status === 425) continue; // not confirmed yet — keep polling
204
+ if (resp.status === 409) throw new Error("Superseded by a newer enrollment — re-enroll this machine.");
205
+ if (resp.status === 401) throw new Error("Invalid wait token — the enrollment URL is wrong or stale.");
206
+ const body = await resp.text().catch(() => "");
207
+ throw new Error(`Consume failed (HTTP ${resp.status}): ${body}`);
208
+ }
209
+ throw new Error(`Enrollment timed out after ${MAX_POLL_DURATION_MS / 60_000} minutes.`);
210
+ }