@tokenfactory/acc-runner 0.40.26 → 0.40.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/dist/attachments/allowlist.d.ts +16 -0
  2. package/dist/attachments/allowlist.d.ts.map +1 -0
  3. package/dist/attachments/allowlist.js +34 -0
  4. package/dist/attachments/allowlist.js.map +1 -0
  5. package/dist/attachments/content-block.d.ts +16 -0
  6. package/dist/attachments/content-block.d.ts.map +1 -0
  7. package/dist/attachments/content-block.js +38 -0
  8. package/dist/attachments/content-block.js.map +1 -0
  9. package/dist/attachments/fetch-attachment.d.ts +24 -0
  10. package/dist/attachments/fetch-attachment.d.ts.map +1 -0
  11. package/dist/attachments/fetch-attachment.js +55 -0
  12. package/dist/attachments/fetch-attachment.js.map +1 -0
  13. package/dist/attachments/index.d.ts +21 -0
  14. package/dist/attachments/index.d.ts.map +1 -0
  15. package/dist/attachments/index.js +20 -0
  16. package/dist/attachments/index.js.map +1 -0
  17. package/dist/attachments/types.d.ts +39 -0
  18. package/dist/attachments/types.d.ts.map +1 -0
  19. package/dist/attachments/types.js +9 -0
  20. package/dist/attachments/types.js.map +1 -0
  21. package/dist/chat-cancel/cancel-controller.d.ts +77 -0
  22. package/dist/chat-cancel/cancel-controller.d.ts.map +1 -0
  23. package/dist/chat-cancel/cancel-controller.js +148 -0
  24. package/dist/chat-cancel/cancel-controller.js.map +1 -0
  25. package/dist/chat-cancel/cancel-source.d.ts +37 -0
  26. package/dist/chat-cancel/cancel-source.d.ts.map +1 -0
  27. package/dist/chat-cancel/cancel-source.js +26 -0
  28. package/dist/chat-cancel/cancel-source.js.map +1 -0
  29. package/dist/chat-cancel/child-terminator.d.ts +37 -0
  30. package/dist/chat-cancel/child-terminator.d.ts.map +1 -0
  31. package/dist/chat-cancel/child-terminator.js +81 -0
  32. package/dist/chat-cancel/child-terminator.js.map +1 -0
  33. package/dist/chat-cancel/index.d.ts +21 -0
  34. package/dist/chat-cancel/index.d.ts.map +1 -0
  35. package/dist/chat-cancel/index.js +20 -0
  36. package/dist/chat-cancel/index.js.map +1 -0
  37. package/dist/chat-cancel/types.d.ts +69 -0
  38. package/dist/chat-cancel/types.d.ts.map +1 -0
  39. package/dist/chat-cancel/types.js +7 -0
  40. package/dist/chat-cancel/types.js.map +1 -0
  41. package/dist/claim-deadman/deadman.d.ts +152 -0
  42. package/dist/claim-deadman/deadman.d.ts.map +1 -0
  43. package/dist/claim-deadman/deadman.js +176 -0
  44. package/dist/claim-deadman/deadman.js.map +1 -0
  45. package/dist/claim-deadman/heartbeat-signal.d.ts +48 -0
  46. package/dist/claim-deadman/heartbeat-signal.d.ts.map +1 -0
  47. package/dist/claim-deadman/heartbeat-signal.js +31 -0
  48. package/dist/claim-deadman/heartbeat-signal.js.map +1 -0
  49. package/dist/claim-deadman/index.d.ts +19 -0
  50. package/dist/claim-deadman/index.d.ts.map +1 -0
  51. package/dist/claim-deadman/index.js +17 -0
  52. package/dist/claim-deadman/index.js.map +1 -0
  53. package/dist/doctor.d.ts +15 -0
  54. package/dist/doctor.d.ts.map +1 -1
  55. package/dist/doctor.js +45 -0
  56. package/dist/doctor.js.map +1 -1
  57. package/dist/git-auth/classify.d.ts +16 -0
  58. package/dist/git-auth/classify.d.ts.map +1 -0
  59. package/dist/git-auth/classify.js +51 -0
  60. package/dist/git-auth/classify.js.map +1 -0
  61. package/dist/git-auth/controller.d.ts +70 -0
  62. package/dist/git-auth/controller.d.ts.map +1 -0
  63. package/dist/git-auth/controller.js +83 -0
  64. package/dist/git-auth/controller.js.map +1 -0
  65. package/dist/git-auth/index.d.ts +14 -0
  66. package/dist/git-auth/index.d.ts.map +1 -0
  67. package/dist/git-auth/index.js +14 -0
  68. package/dist/git-auth/index.js.map +1 -0
  69. package/dist/git-auth/probe.d.ts +34 -0
  70. package/dist/git-auth/probe.d.ts.map +1 -0
  71. package/dist/git-auth/probe.js +72 -0
  72. package/dist/git-auth/probe.js.map +1 -0
  73. package/dist/watch.d.ts +112 -0
  74. package/dist/watch.d.ts.map +1 -1
  75. package/dist/watch.js +421 -2
  76. package/dist/watch.js.map +1 -1
  77. package/package.json +1 -1
package/dist/watch.js CHANGED
@@ -43,6 +43,7 @@ import { acquireSingletonLock, singletonRunnerId, SingletonLockHeldError, } from
43
43
  import { getClaudeVersion, recordTaskClaudeVersion, } from "./runtime/version-drift.js";
44
44
  import { autoUpgradeDisabled, maybeSelfUpgrade, maybeSelfUpdateOn426, } from "./runtime/self-upgrade.js";
45
45
  import { postRunnerStateMessage } from "./messaging.js";
46
+ import { GitAuthController, makeGitAuthProbe, RUNNER_GIT_AUTH_FAILED_VERB, RUNNER_GIT_AUTH_RECOVERED_VERB, } from "./git-auth/index.js";
46
47
  // AS-3 (RESUME-WIRE): drive the live capacity pause/resume through the
47
48
  // ResumeController (pause-until reset OR account-change OR ping) instead of a
48
49
  // fixed pause-until-reset setTimeout, and off the AS-2 account-probe cadence.
@@ -54,6 +55,7 @@ import { getChatEngine } from "./engines/registry.js";
54
55
  // a `watch` runner — the runner serves its bound user's chat turns behind their
55
56
  // companion. Glue lives in watch-chat/wire.ts to keep this file the task loop.
56
57
  import { startFleetChatServing, } from "./watch-chat/wire.js";
58
+ import { ClaimDeadman, readDispatchSignal, DEFAULT_DEADMAN_CHECK_MS, DEFAULT_FALLBACK_POLL_MS, CLAIM_CHANNEL_WEDGED_VERB, CLAIM_CHANNEL_RECOVERED_VERB, CLAIM_CHANNEL_REBUILD_VERB, CLAIM_CHANNEL_FALLBACK_VERB, } from "./claim-deadman/index.js";
57
59
  import { unknownAccountIdentity } from "./engines/account-identity.js";
58
60
  import { checkVersion, compareSemver } from "./version-check.js";
59
61
  import { PACKAGE_VERSION } from "./pkg-version.js";
@@ -119,6 +121,29 @@ const DEFAULT_CLAIM_WATCHDOG_CHECK_MS = 30_000;
119
121
  // regressions of the same shape. Default 60s — prompt recovery without a third
120
122
  // frequent DB poll racing the watchdog. Knob: `claim_backstop_interval_ms`.
121
123
  const DEFAULT_CLAIM_BACKSTOP_INTERVAL_MS = 60_000;
124
+ // CLAIM-DEADMAN: dead-man switch for a WEDGED claim channel
125
+ // that the 90s watchdog resubscribe can't clear. Live incident 2026-07-22: a
126
+ // runner heartbeated fine (hb_age 0min) but its Realtime claim subscription was
127
+ // dead AND the watchdog resubscribe had itself stopped firing — 12 queued / 0
128
+ // running with two online runners, cleared only by a manual restart. The deadman
129
+ // tracks pending-dispatch (work the server is offering) vs the last processed
130
+ // claim; when work is pending and no claim lands for CLAIM_DEADMAN_WEDGE_MS
131
+ // (default 120s, > the 90s watchdog so the cheap resubscribe gets first crack)
132
+ // it (1) TEARS DOWN + FULLY REBUILDS the Realtime client up to
133
+ // CLAIM_DEADMAN_MAX_REBUILDS times, then (2) falls back to degraded DB-poll
134
+ // claiming every CLAIM_DEADMAN_FALLBACK_POLL_MS. CLAIM_DEADMAN_CHECK_MS is the
135
+ // evaluation cadence. Tests override all of these for speed. See
136
+ // packages/acc-runner/src/claim-deadman/deadman.ts + docs/acc/CLAIM_DEADMAN.md.
137
+ const DEFAULT_CLAIM_DEADMAN_WEDGE_MS = 120_000;
138
+ const DEFAULT_CLAIM_DEADMAN_REBUILD_GRACE_MS = 15_000;
139
+ const DEFAULT_CLAIM_DEADMAN_MAX_REBUILDS = 2;
140
+ // GIT-AUTH-ALERT: how often the runner probes its base clone's git credential
141
+ // (a masked `ls-remote origin` dry-run). The same 60s cadence detects the first
142
+ // failure AND polls for recovery — while degraded, task-lane claiming is paused
143
+ // and this probe is the only thing that can clear it. Tests set a large value so
144
+ // only the `gitAuthProbeOnce()` hook drives it deterministically. Knob:
145
+ // `git_auth_probe_interval_ms`.
146
+ const DEFAULT_GIT_AUTH_PROBE_INTERVAL_MS = 60_000;
122
147
  // v0.56 (T-56-1): capacity-aware pause. When claude reports a session/usage
123
148
  // limit (api_error_status 429) the runner pauses claiming tasks AND reviews
124
149
  // until the stated reset time. When no reset time is parseable, fall back to
@@ -200,6 +225,13 @@ function enqueue(state, taskId, factory) {
200
225
  state.seen.add(taskId);
201
226
  state.queue.push(taskId);
202
227
  touchActivity(state);
228
+ // CLAIM-DEADMAN: a claim was processed via SOME path
229
+ // (realtime / poll / scan / watchdog / the deadman's own DB-poll fallback).
230
+ // Reset the deadman's staleness clock so a runner that IS claiming — even
231
+ // slowly, even degraded — is never treated as wedged. Recovery itself is
232
+ // declared by the offered backlog draining (observeDispatchSignal), so this
233
+ // never prematurely clears a wedge episode.
234
+ state.claimDeadman?.observeClaimProcessed();
203
235
  void pump(state, factory);
204
236
  return true;
205
237
  }
@@ -648,6 +680,230 @@ async function claimBackstopScan(state, factory) {
648
680
  state.backstopScanning = false;
649
681
  }
650
682
  }
683
+ /**
684
+ * CLAIM-DEADMAN: a runner may have free capacity but be
685
+ * deliberately not claiming (quarantined, capacity-paused, review-claim-paused,
686
+ * or already at its effective task-claim ceiling). In those states pending work
687
+ * SHOULD sit queued, so it must NOT be read as a wedge. Mirrors the
688
+ * claimBackstopScan guard. Returns true only when the runner both wants to and
689
+ * can claim another task.
690
+ */
691
+ function deadmanCanClaim(state) {
692
+ if (state.quarantined || state.pausedCapacity || state.capacityClaimPaused) {
693
+ return false;
694
+ }
695
+ const limit = computeTaskClaimLimit({
696
+ concurrencyLimit: state.concurrencyLimit,
697
+ reservedReviewSlots: state.reservedReviewSlots,
698
+ capacityProbe: state.capacityProbe,
699
+ claimPaused: state.capacityClaimPaused,
700
+ });
701
+ return state.running.size < limit;
702
+ }
703
+ /**
704
+ * CLAIM-DEADMAN: parse the pending-dispatch fields off a `heartbeat_runner`
705
+ * response. Tolerant of a scalar row, a single-row array, or null (today's
706
+ * void return) — returns null unless a numeric `pending_dispatch` is present, so
707
+ * a pre-extension server leaves the deadman on its assigned-queued read.
708
+ */
709
+ export function parseHeartbeatDeadmanFields(data) {
710
+ const row = Array.isArray(data) ? data[0] : data;
711
+ if (!row || typeof row !== "object")
712
+ return null;
713
+ const r = row;
714
+ if (typeof r.pending_dispatch !== "number")
715
+ return null;
716
+ return {
717
+ pending_dispatch: Math.max(0, Math.floor(r.pending_dispatch)),
718
+ suspect: Boolean(r.suspect),
719
+ };
720
+ }
721
+ /**
722
+ * CLAIM-DEADMAN: emit the audit event + operator-facing alert for a claim-
723
+ * channel state change. `runner.claim_channel_wedged` / `..._recovered` are the
724
+ * UI alert breadcrumbs; the payload NAMES the runner (id + label) so the derived
725
+ * notifications read model can surface "runner X's claim channel is wedged"
726
+ * without a join. Best-effort by contract — a logging failure never disturbs the
727
+ * watch loop. Also mirrors to stderr so a locally-tailed runner shows it.
728
+ */
729
+ function emitDeadmanEvent(state, verb, payload, humanLine) {
730
+ process.stderr.write(`[acc-runner] ${humanLine}\n`);
731
+ void (async () => {
732
+ try {
733
+ await state.supabase.rpc("log_activity", {
734
+ p_verb: verb,
735
+ p_target_id: state.session.runner_id,
736
+ p_payload: {
737
+ // NAMES the runner so the derived notifications read model can surface
738
+ // "runner <id>'s claim channel is wedged" without a join (the UI
739
+ // resolves the display name from acc.runners by this id).
740
+ runner_id: state.session.runner_id,
741
+ version: PACKAGE_VERSION,
742
+ ...payload,
743
+ },
744
+ p_target_type: "runner",
745
+ });
746
+ }
747
+ catch {
748
+ /* best-effort breadcrumb */
749
+ }
750
+ })();
751
+ }
752
+ /**
753
+ * CLAIM-DEADMAN: tear down + FULLY REBUILD the Realtime client — not just a
754
+ * resubscribe. The v0.68 resubscribe / v0.73 watchdog both remove+re-add a
755
+ * channel on the SAME Supabase client, so a wedged underlying socket / Realtime
756
+ * instance survives them (the incident's failure mode). This discards the whole
757
+ * client and builds a fresh one around the same JWT + realtime config (mirrors
758
+ * doRefresh/doFailover's client rebuild, minus the token change), then recovers
759
+ * anything the dead channel missed. Single-flight via `deadmanRebuilding`.
760
+ */
761
+ export async function rebuildRealtimeClient(state, factory, attempt) {
762
+ if (state.stopped || state.deadmanRebuilding)
763
+ return;
764
+ state.deadmanRebuilding = true;
765
+ try {
766
+ // Tear down the old channel + socket as fully as the client allows.
767
+ try {
768
+ await state.supabase.removeChannel(state.channel);
769
+ }
770
+ catch {
771
+ /* channel may already be dead */
772
+ }
773
+ try {
774
+ await state.supabase.removeAllChannels?.();
775
+ }
776
+ catch {
777
+ /* best-effort */
778
+ }
779
+ try {
780
+ // any-allowed: `.realtime` is not on the narrowed RunnerSupabaseClient type
781
+ // but supabase-js exposes a RealtimeClient with disconnect() here — this is
782
+ // the "fully rebuild, not just resubscribe" teardown of the live socket.
783
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
784
+ state.supabase.realtime?.disconnect?.();
785
+ }
786
+ catch {
787
+ /* best-effort */
788
+ }
789
+ // Fresh client around the current (still-valid) credential.
790
+ state.supabase = createRunnerClient(state.cfg, state.session.access_token, state.session.realtime_config);
791
+ subscribeChannel(state, factory);
792
+ // FLEET-SERVE (wire): hot-swap the chat lane onto the fresh client so chat
793
+ // serving survives the rebuild (mirrors doRefresh/doFailover).
794
+ await state.fleetChat?.refreshClient(state.supabase, state.session.access_token);
795
+ emitDeadmanEvent(state, CLAIM_CHANNEL_REBUILD_VERB, { attempt }, `claim-deadman: rebuilding Realtime client (attempt ${attempt}) — ` +
796
+ `claim channel wedged with a live heartbeat`);
797
+ // Recover anything the wedged channel missed. The seen/seenReviews guards
798
+ // dedupe against a late re-delivery on the fresh channel.
799
+ void reassertRunningClaims(state);
800
+ void pollOnce(state, factory);
801
+ void claimScan(state, factory, "periodic");
802
+ void reviewScan(state, "periodic");
803
+ }
804
+ catch (err) {
805
+ process.stderr.write(`[acc-runner] claim-deadman rebuild failed: ${err.message}\n`);
806
+ }
807
+ finally {
808
+ state.deadmanRebuilding = false;
809
+ }
810
+ }
811
+ /**
812
+ * CLAIM-DEADMAN: the degraded DB-poll fallback loop. Once the rebuild budget is
813
+ * spent the runner claims by polling the claim path directly — a full epoch-
814
+ * watermark claim scan plus the `seen`-reconcile backstop — every
815
+ * `deadmanFallbackPollMs`. Degraded (no realtime latency) but FUNCTIONAL: tasks
816
+ * get claimed with zero human action while the WS stays dead. Idempotent —
817
+ * starting an already-running loop is a no-op.
818
+ */
819
+ function startDeadmanFallback(state, factory) {
820
+ if (state.stopped || state.deadmanFallbackTimer)
821
+ return;
822
+ emitDeadmanEvent(state, CLAIM_CHANNEL_FALLBACK_VERB, { poll_ms: state.deadmanFallbackPollMs }, `claim-deadman: Realtime rebuild budget spent — falling back to DB-poll ` +
823
+ `claiming every ${state.deadmanFallbackPollMs}ms (degraded but functional)`);
824
+ const tick = () => {
825
+ void (async () => {
826
+ if (state.stopped)
827
+ return;
828
+ await claimScan(state, factory, "periodic");
829
+ await claimBackstopScan(state, factory);
830
+ })();
831
+ };
832
+ tick(); // claim immediately, don't wait a full interval
833
+ state.deadmanFallbackTimer = setInterval(tick, state.deadmanFallbackPollMs);
834
+ state.deadmanFallbackTimer.unref?.();
835
+ }
836
+ /** CLAIM-DEADMAN: stop the degraded DB-poll fallback loop (on recovery / stop). */
837
+ function stopDeadmanFallback(state) {
838
+ if (state.deadmanFallbackTimer) {
839
+ clearInterval(state.deadmanFallbackTimer);
840
+ state.deadmanFallbackTimer = null;
841
+ }
842
+ }
843
+ /**
844
+ * CLAIM-DEADMAN: the dead-man evaluation. Reads how much
845
+ * work the server is offering this runner (pending-dispatch — from the heartbeat
846
+ * response when the server supplies it, else the assigned-queued read over the
847
+ * same HTTP path that survives a dead WS), feeds it + the last-claim clock to the
848
+ * pure ClaimDeadman, and acts on its decision:
849
+ *
850
+ * - recovered (offered backlog drained after a wedge episode) → stop the
851
+ * fallback loop + emit `runner.claim_channel_recovered`.
852
+ * - rebuild(N) → full Realtime-client rebuild (rebuildRealtimeClient).
853
+ * - fallback → engage/keep the degraded DB-poll claim loop.
854
+ *
855
+ * A heartbeat failure short-circuits to observeHeartbeatFailure (a network/auth
856
+ * outage is a different failure with its own handling — not a wedge). Bounded
857
+ * and side-effect-safe: never throws out of the loop. Single-flight via
858
+ * `claimDeadmanChecking`.
859
+ */
860
+ async function claimDeadmanScan(state, factory) {
861
+ if (state.stopped || state.claimDeadmanChecking)
862
+ return;
863
+ state.claimDeadmanChecking = true;
864
+ try {
865
+ if (state.heartbeatFailures > 0) {
866
+ state.claimDeadman.observeHeartbeatFailure();
867
+ return;
868
+ }
869
+ // Only pending work the runner WOULD claim counts as offered — a runner that
870
+ // is deliberately not claiming (paused/quarantined/at-capacity) is not wedged.
871
+ let signal;
872
+ if (deadmanCanClaim(state)) {
873
+ signal = await readDispatchSignal(state.supabase, state.session.runner_id, state.lastHeartbeatDeadmanFields);
874
+ }
875
+ else {
876
+ signal = { pendingDispatch: 0, suspect: false };
877
+ }
878
+ const { recovered } = state.claimDeadman.observeDispatchSignal(signal);
879
+ if (recovered) {
880
+ stopDeadmanFallback(state);
881
+ emitDeadmanEvent(state, CLAIM_CHANNEL_RECOVERED_VERB, {}, `claim-deadman: claim channel recovered — offered backlog drained, ` +
882
+ `runner is claiming normally again`);
883
+ }
884
+ const decision = state.claimDeadman.tick();
885
+ if (decision.wedged) {
886
+ emitDeadmanEvent(state, CLAIM_CHANNEL_WEDGED_VERB, {
887
+ pending_dispatch: decision.pendingDispatch,
888
+ ms_since_claim: decision.msSinceClaim,
889
+ }, `claim-deadman: claim channel WEDGED — ${decision.pendingDispatch} ` +
890
+ `task(s) offered but none claimed for ${Math.round(decision.msSinceClaim / 1000)}s ` +
891
+ `with a live heartbeat; self-healing`);
892
+ }
893
+ if (decision.action === "rebuild") {
894
+ await rebuildRealtimeClient(state, factory, decision.rebuildAttempt ?? 1);
895
+ }
896
+ else if (decision.action === "fallback") {
897
+ startDeadmanFallback(state, factory);
898
+ }
899
+ }
900
+ catch (err) {
901
+ process.stderr.write(`[acc-runner] claim deadman scan failed: ${err.message}\n`);
902
+ }
903
+ finally {
904
+ state.claimDeadmanChecking = false;
905
+ }
906
+ }
651
907
  /**
652
908
  * v0.65 (T-65-2): full review scan — the review twin of claimScan. Lists EVERY
653
909
  * pending review assigned to this runner with an epoch watermark (`1970-…Z`)
@@ -915,6 +1171,8 @@ export async function watchCommand(options = {}) {
915
1171
  reviewScanOnce: async () => { },
916
1172
  claimWatchdogOnce: async () => { },
917
1173
  claimBackstopOnce: async () => { },
1174
+ claimDeadmanOnce: async () => { },
1175
+ gitAuthProbeOnce: async () => false,
918
1176
  runningCount: () => 0,
919
1177
  };
920
1178
  }
@@ -1096,6 +1354,27 @@ export async function watchCommand(options = {}) {
1096
1354
  claimBackstopTimer: undefined,
1097
1355
  claimBackstopIntervalMs: options.claimBackstopIntervalMs ?? DEFAULT_CLAIM_BACKSTOP_INTERVAL_MS,
1098
1356
  backstopScanning: false,
1357
+ // CLAIM-DEADMAN: wedge dead-man switch. The pure state
1358
+ // machine carries its own clock (Date.now); tests inject tiny thresholds.
1359
+ claimDeadman: new ClaimDeadman({
1360
+ wedgeThresholdMs: options.claimDeadmanWedgeMs ?? DEFAULT_CLAIM_DEADMAN_WEDGE_MS,
1361
+ rebuildGraceMs: options.claimDeadmanRebuildGraceMs ?? DEFAULT_CLAIM_DEADMAN_REBUILD_GRACE_MS,
1362
+ maxRebuilds: options.claimDeadmanMaxRebuilds ?? DEFAULT_CLAIM_DEADMAN_MAX_REBUILDS,
1363
+ }),
1364
+ claimDeadmanTimer: undefined,
1365
+ claimDeadmanCheckMs: options.claimDeadmanCheckMs ?? DEFAULT_DEADMAN_CHECK_MS,
1366
+ claimDeadmanChecking: false,
1367
+ deadmanRebuilding: false,
1368
+ deadmanFallbackTimer: null,
1369
+ deadmanFallbackPollMs: options.claimDeadmanFallbackPollMs ?? DEFAULT_FALLBACK_POLL_MS,
1370
+ lastHeartbeatDeadmanFields: null,
1371
+ // GIT-AUTH-ALERT: controller wired below once `state` exists (its emit
1372
+ // callbacks close over `state`); timer armed in the timer-setup block.
1373
+ gitAuthController: null,
1374
+ gitAuthTimer: null,
1375
+ gitAuthProbeIntervalMs: options.gitAuthProbeIntervalMs ?? DEFAULT_GIT_AUTH_PROBE_INTERVAL_MS,
1376
+ gitAuthChecking: false,
1377
+ gitAuthFailed: false,
1099
1378
  // v0.68 (T-68-1): realtime channel-health recovery state.
1100
1379
  resubscribing: false,
1101
1380
  resubscribeAttempts: 0,
@@ -1167,6 +1446,10 @@ export async function watchCommand(options = {}) {
1167
1446
  // starting account is seeded before any claim can 429. Both timers unref so
1168
1447
  // the probe/poll cadence never holds the process open on its own.
1169
1448
  await wireCapacityResume(state, taskRunnerFactory, options);
1449
+ // GIT-AUTH-ALERT: build the git-credential health controller (no-op when no
1450
+ // base clone is resolvable). The probe timer is armed in the timer-setup
1451
+ // block below alongside the claim/review scans.
1452
+ wireGitAuthController(state, options);
1170
1453
  subscribeChannel(state, taskRunnerFactory);
1171
1454
  // v0.34-B / v0.62 (T-62-1): one-shot startup claim scan. The 5-minute
1172
1455
  // pollSince watermark (v0.30-B) can miss a task dispatched (runner_id set)
@@ -1243,6 +1526,29 @@ export async function watchCommand(options = {}) {
1243
1526
  state.claimBackstopTimer = setInterval(() => {
1244
1527
  void claimBackstopScan(state, taskRunnerFactory);
1245
1528
  }, state.claimBackstopIntervalMs);
1529
+ // CLAIM-DEADMAN: the wedge dead-man switch. Escalates
1530
+ // BEYOND the 90s watchdog resubscribe (which only re-adds a channel on the
1531
+ // same client and, per the incident, can itself stop firing): tracks pending-
1532
+ // dispatch vs the last processed claim, and when work is offered but nothing
1533
+ // is claimed for the wedge threshold it FULLY REBUILDS the Realtime client
1534
+ // (twice) then degrades to DB-poll claiming — self-healing the "heartbeat-
1535
+ // alive but claim-dead" runner with zero human action. Independent timer +
1536
+ // single-flight inside claimDeadmanScan().
1537
+ state.claimDeadmanTimer = setInterval(() => {
1538
+ void claimDeadmanScan(state, taskRunnerFactory);
1539
+ }, state.claimDeadmanCheckMs);
1540
+ state.claimDeadmanTimer.unref?.();
1541
+ // GIT-AUTH-ALERT: periodic git-credential health probe. Armed only when the
1542
+ // controller is wired (a base clone is resolvable). The same cadence detects
1543
+ // the first auth failure AND polls for recovery while degraded. Single-flight
1544
+ // inside gitAuthProbeScan(); unref'd so the probe never holds the process
1545
+ // open on its own.
1546
+ if (state.gitAuthController) {
1547
+ state.gitAuthTimer = setInterval(() => {
1548
+ void gitAuthProbeScan(state);
1549
+ }, state.gitAuthProbeIntervalMs);
1550
+ state.gitAuthTimer.unref?.();
1551
+ }
1246
1552
  // v0.13-MULTI-RUNNER: refresh the runner's capability tags so an
1247
1553
  // operator change to ACC_RUNNER_CAPABILITIES (or to ~/.config/acc-
1248
1554
  // runner/config.json) takes effect on the next restart without
@@ -1337,6 +1643,10 @@ export async function watchCommand(options = {}) {
1337
1643
  clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
1338
1644
  clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
1339
1645
  clearInterval(state.claimBackstopTimer); // FIX-D
1646
+ clearInterval(state.claimDeadmanTimer); // CLAIM-DEADMAN
1647
+ stopDeadmanFallback(state); // CLAIM-DEADMAN
1648
+ if (state.gitAuthTimer)
1649
+ clearInterval(state.gitAuthTimer); // GIT-AUTH-ALERT
1340
1650
  if (state.resubscribeTimer)
1341
1651
  clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
1342
1652
  if (state.idleTimer)
@@ -1447,6 +1757,8 @@ export async function watchCommand(options = {}) {
1447
1757
  reviewScanOnce: () => reviewScan(state, "periodic"), // v0.65 (T-65-2)
1448
1758
  claimWatchdogOnce: () => claimWatchdogScan(state, taskRunnerFactory), // v0.73 (T-73-2)
1449
1759
  claimBackstopOnce: () => claimBackstopScan(state, taskRunnerFactory), // FIX-D
1760
+ claimDeadmanOnce: () => claimDeadmanScan(state, taskRunnerFactory), // CLAIM-DEADMAN
1761
+ gitAuthProbeOnce: () => gitAuthProbeScan(state), // GIT-AUTH-ALERT
1450
1762
  runningCount: () => state.running.size,
1451
1763
  };
1452
1764
  }
@@ -1538,12 +1850,20 @@ function scheduleHeartbeat(state, taskRunnerFactory, baseMs) {
1538
1850
  const delay = nextHeartbeatDelayMs(state.heartbeatFailures, baseMs);
1539
1851
  state.heartbeatTimer = setTimeout(() => {
1540
1852
  void (async () => {
1541
- const { error } = await state.supabase.rpc("heartbeat_runner", {
1853
+ const { data, error } = await state.supabase.rpc("heartbeat_runner", {
1542
1854
  p_id: state.session.runner_id,
1543
1855
  p_version: PACKAGE_VERSION,
1544
1856
  });
1545
1857
  if (!error) {
1546
1858
  state.heartbeatFailures = 0;
1859
+ // CLAIM-DEADMAN: capture the pending-dispatch count
1860
+ // + suspect flag when the heartbeat RESPONSE carries them, so the runner
1861
+ // learns it is being offered work even with a dead WS. Forward-compatible:
1862
+ // today `heartbeat_runner` returns void and this is a no-op (the deadman
1863
+ // reads the assigned-queued count over HTTP instead). Once the RPC is
1864
+ // extended (Track A migration) this is the authoritative signal — the
1865
+ // deadman prefers it in readDispatchSignal.
1866
+ state.lastHeartbeatDeadmanFields = parseHeartbeatDeadmanFields(data);
1547
1867
  // RUNCFG-1: pull the operator's server-set concurrency and apply it to
1548
1868
  // the live ceiling within this heartbeat. Runs only on a healthy
1549
1869
  // heartbeat (a network/auth failure would fail this RPC too). Awaited so
@@ -1577,6 +1897,8 @@ function scheduleHeartbeat(state, taskRunnerFactory, baseMs) {
1577
1897
  clearInterval(state.reviewScanTimer); // v0.65 (T-65-2)
1578
1898
  clearInterval(state.claimWatchdogTimer); // v0.73 (T-73-2)
1579
1899
  clearInterval(state.claimBackstopTimer); // FIX-D
1900
+ clearInterval(state.claimDeadmanTimer); // CLAIM-DEADMAN
1901
+ stopDeadmanFallback(state); // CLAIM-DEADMAN
1580
1902
  if (state.resubscribeTimer)
1581
1903
  clearTimeout(state.resubscribeTimer); // v0.68 (T-68-1)
1582
1904
  process.exit(1);
@@ -1788,6 +2110,99 @@ async function pumpReviews(state) {
1788
2110
  * duplicate audit events. Sets the in-memory flag synchronously (pump()
1789
2111
  * reads it without an async hop) and best-effort persists + announces.
1790
2112
  */
2113
+ /**
2114
+ * GIT-AUTH-ALERT: build the git-credential health controller and stash it on
2115
+ * `state`. The probe defaults to a masked `ls-remote origin` against the base
2116
+ * clone (worktreeRepoPath ?? repoPath); tests inject `options.gitAuthProbe`.
2117
+ * When neither is available the controller stays null and the probe loop is
2118
+ * never armed — the feature is a no-op for chat-only / repo-less runners.
2119
+ *
2120
+ * The degraded/recovered callbacks flip `state.gitAuthFailed` (read
2121
+ * synchronously in pump() to pause task-lane claiming) and emit the
2122
+ * runner-scoped `runner.git_auth_failed` / `runner.git_auth_recovered` audit
2123
+ * events the alerts tick scans to page the operator. Both emits name the machine
2124
+ * and carry only masked, credential-safe detail.
2125
+ */
2126
+ function wireGitAuthController(state, options) {
2127
+ const repoPath = state.cfg.worktreeRepoPath ?? state.cfg.repoPath;
2128
+ const probe = options.gitAuthProbe ?? (repoPath ? makeGitAuthProbe({ repoPath }) : null);
2129
+ if (!probe)
2130
+ return;
2131
+ const machine = machineIdentity();
2132
+ const runnerId = state.session.runner_id;
2133
+ state.gitAuthController = new GitAuthController({
2134
+ probe,
2135
+ threshold: options.gitAuthFailureThreshold,
2136
+ onDegraded: async ({ consecutiveFailures, detail }) => {
2137
+ state.gitAuthFailed = true;
2138
+ process.stderr.write(chalk.red.bold(`[acc-runner] GIT AUTH FAILED on ${machine} (runner ${runnerId}): ${detail} ` +
2139
+ `— ${consecutiveFailures} consecutive auth-class fetch failures. Pausing ` +
2140
+ `task-lane claiming until a fetch succeeds (probe every ` +
2141
+ `${Math.round(state.gitAuthProbeIntervalMs / 1000)}s). ` +
2142
+ `Fix: run \`gh auth login\` or renew the PAT.\n`));
2143
+ try {
2144
+ await state.supabase.rpc("log_activity", {
2145
+ p_verb: RUNNER_GIT_AUTH_FAILED_VERB,
2146
+ p_target_id: runnerId,
2147
+ p_payload: {
2148
+ runner_id: runnerId,
2149
+ machine,
2150
+ consecutive_failures: consecutiveFailures,
2151
+ detail,
2152
+ remediation: "Run `gh auth login` on the runner host, or renew the PAT it uses.",
2153
+ version: PACKAGE_VERSION,
2154
+ },
2155
+ p_target_type: "runner",
2156
+ });
2157
+ }
2158
+ catch {
2159
+ /* best-effort audit — a failed emit must not wedge the probe loop. */
2160
+ }
2161
+ },
2162
+ onRecovered: async ({ failuresBeforeRecovery, detail }) => {
2163
+ state.gitAuthFailed = false;
2164
+ process.stderr.write(chalk.green(`[acc-runner] git auth RECOVERED on ${machine} (runner ${runnerId}): ${detail}. ` +
2165
+ `Resuming task-lane claiming.\n`));
2166
+ try {
2167
+ await state.supabase.rpc("log_activity", {
2168
+ p_verb: RUNNER_GIT_AUTH_RECOVERED_VERB,
2169
+ p_target_id: runnerId,
2170
+ p_payload: {
2171
+ runner_id: runnerId,
2172
+ machine,
2173
+ failures_before_recovery: failuresBeforeRecovery,
2174
+ version: PACKAGE_VERSION,
2175
+ },
2176
+ p_target_type: "runner",
2177
+ });
2178
+ }
2179
+ catch {
2180
+ /* best-effort audit. */
2181
+ }
2182
+ // Re-drive the pump so queued work is claimed immediately on recovery
2183
+ // instead of waiting for the next poll/scan tick.
2184
+ void pump(state, state.taskRunnerFactory);
2185
+ },
2186
+ });
2187
+ }
2188
+ /**
2189
+ * GIT-AUTH-ALERT: run exactly one git-credential probe. Single-flight via
2190
+ * `gitAuthChecking` so a slow probe can't overlap itself. Returns false (a
2191
+ * no-op) when the controller is disabled. Production drives this from the
2192
+ * probe timer; tests drive it via the `gitAuthProbeOnce()` handle hook.
2193
+ */
2194
+ async function gitAuthProbeScan(state) {
2195
+ if (state.stopped || !state.gitAuthController || state.gitAuthChecking)
2196
+ return false;
2197
+ state.gitAuthChecking = true;
2198
+ try {
2199
+ await state.gitAuthController.check();
2200
+ return true;
2201
+ }
2202
+ finally {
2203
+ state.gitAuthChecking = false;
2204
+ }
2205
+ }
1791
2206
  function enterQuarantine(state, taskId, cause, detail,
1792
2207
  // v0.53 T-53-4: how many consecutive instant-empty exits produced an
1793
2208
  // env_broken cause (1 = definitive, 2 = heuristic confirmed by the
@@ -2107,7 +2522,11 @@ async function pump(state, factory) {
2107
2522
  capacityProbe: state.capacityProbe,
2108
2523
  // RVU-2 (FU-RVU2R): pause NEW task claims under review capacity pressure so
2109
2524
  // reviews get the scarce subscription capacity first (cap-priority).
2110
- claimPaused: state.capacityClaimPaused,
2525
+ // GIT-AUTH-ALERT: also pause NEW task claims while the base git credential
2526
+ // is degraded — every claim would fail at phase=git and requeue, so backing
2527
+ // off (task-lane only; reviews keep draining) avoids the churn the live
2528
+ // 2026-07-22 incident spammed. Cleared by the recovery probe.
2529
+ claimPaused: state.capacityClaimPaused || state.gitAuthFailed,
2111
2530
  });
2112
2531
  // Fill available concurrency slots from the queue.
2113
2532
  while (!state.stopped &&