@lifeaitools/clauth 2.15.5 → 2.15.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -32,6 +32,15 @@
32
32
  .supervisor-pill.ok{border-color:rgba(74,222,128,.35);color:var(--green-light);background:rgba(34,197,94,.08)}
33
33
  .supervisor-pill.warn{border-color:rgba(250,204,21,.35);color:var(--gold-light);background:rgba(250,204,21,.08)}
34
34
  .supervisor-pill.bad{border-color:rgba(248,113,113,.35);color:var(--pink);background:rgba(248,113,113,.08)}
35
+ /* NEEDS HELP -- a surface that exhausted MAX_CONSECUTIVE_RECONCILE_FAILURES
36
+ and clauth has stopped auto-restarting. Distinct from .bad ("down right
37
+ now, clauth is still retrying"): this is "clauth gave up, a human/agent
38
+ has to look." Pulses so it reads as an alarm among a fleet that regularly
39
+ has a few surfaces transiently down in .bad. */
40
+ .supervisor-pill.critical{border-color:rgba(248,113,113,.6);color:#fff;background:rgba(220,38,38,.55);animation:supervisor-pulse 1.4s ease-in-out infinite}
41
+ .supervisor-row.needs-help{border-color:rgba(220,38,38,.6);box-shadow:0 0 0 1px rgba(220,38,38,.35) inset}
42
+ .supervisor-meta.critical{color:var(--pink);white-space:normal}
43
+ @keyframes supervisor-pulse{0%,100%{opacity:1}50%{opacity:.55}}
35
44
  /* Card GRID, not a stack. flex-direction:column gave one full-width row per
36
45
  surface, so ten surfaces meant ten rows and constant scrolling to see the
37
46
  fleet. auto-fill + minmax packs as many small cards per row as the panel
@@ -305,7 +305,15 @@ class ClauthSurfacesPanel extends window.ClauthPanelElement {
305
305
  renderSupervisorSurface(surface) {
306
306
  const compositeId = surface.plugin_id + ":" + surface.id;
307
307
  const ownerKind = surface.lifecycle_owner === "clauth" ? "ok" : (surface.lifecycle_owner === "plugin" ? "warn" : "");
308
- const stateKind = surface.state === "current" || surface.status === "healthy" ? "ok" : (surface.state === "unavailable" ? "bad" : "warn");
308
+ // "needs_help" is a DIFFERENT alarm than "bad": bad means "down right now,
309
+ // clauth is still retrying"; needs_help means "clauth gave up after
310
+ // MAX_CONSECUTIVE_RECONCILE_FAILURES — a human/agent has to look." It gets
311
+ // its own loud, pulsing pill (see .supervisor-pill.critical) rather than
312
+ // reusing "bad", which would render as just another red flicker in a fleet
313
+ // that regularly has a few surfaces transiently down.
314
+ const needsHelp = surface.state === "needs_help";
315
+ const stateKind = needsHelp ? "critical" : (surface.state === "current" || surface.status === "healthy" ? "ok" : (surface.state === "unavailable" ? "bad" : "warn"));
316
+ const stateLabel = needsHelp ? "NEEDS HELP" : (surface.state || surface.status || "unknown");
309
317
  const openUrl = surface.__pseudo ? null : openUiUrlForSurface(surface);
310
318
  const isSelected = this.selectedSupervisorSurface && this.selectedSupervisorSurface.id === compositeId;
311
319
  const rowClass = "supervisor-row" + (surface.__pseudo ? " readonly" : "") + (isSelected ? " selected" : "");
@@ -339,14 +347,22 @@ class ClauthSurfacesPanel extends window.ClauthPanelElement {
339
347
  : (surface.last_health_ok === false && surface.last_health_error
340
348
  ? surface.last_health_error + " · " + (surface.health || "no health")
341
349
  : (surface.health || surface.health_url || "no health"));
342
- return '<div class="' + rowClass + '" data-surface-id="' + htmlEscape(compositeId) + '"' + rowAction + '>' +
350
+ // The loud line: WHY it needs help and WHAT to do, spelled out on the
351
+ // card itself — not just a pill label a reader has to already know the
352
+ // meaning of. Mirrors the same wording reconcileSurfaceHealth() writes
353
+ // into the daemon log, so the card and the log never disagree.
354
+ const needsHelpLine = needsHelp
355
+ ? "Stopped auto-restarting after " + (surface.consecutive_reconcile_failures || "several") + " failed attempts — check the log, fix the cause, then Start/Restart to resume auto-repair."
356
+ : null;
357
+ return '<div class="' + rowClass + (needsHelp ? " needs-help" : "") + '" data-surface-id="' + htmlEscape(compositeId) + '"' + rowAction + '>' +
343
358
  '<div class="supervisor-row-top"><span class="supervisor-name">' + htmlEscape(surface.name || compositeId) + '</span>' +
344
359
  healthPill +
345
- supervisorBadge(surface.lifecycle_owner || "unknown", ownerKind) + supervisorBadge(surface.state || surface.status || "unknown", stateKind) +
360
+ supervisorBadge(surface.lifecycle_owner || "unknown", ownerKind) + supervisorBadge(stateLabel, stateKind) +
346
361
  openBtn +
347
362
  '</div>' +
348
363
  '<div class="supervisor-meta">' + htmlEscape(surface.destination || "—") + ' · port ' + htmlEscape(surface.port || "—") + '</div>' +
349
364
  '<div class="supervisor-meta">' + htmlEscape(healthLine) + '</div>' +
365
+ (needsHelpLine ? '<div class="supervisor-meta critical">' + htmlEscape(needsHelpLine) + '</div>' : '') +
350
366
  '</div>';
351
367
  }
352
368
 
@@ -15,7 +15,33 @@ import path from "path";
15
15
  // Cache path — avoids re-querying WMI/CIM on every daemon start.
16
16
  // This eliminates the spawnSync cmd.exe ETIMEDOUT crash that occurs
17
17
  // when PowerShell/WMI is slow on first call after boot.
18
- const CACHE_FILE = path.join(os.tmpdir(), "clauth-machine.cache");
18
+ //
19
+ // Durable as of 2026-09-06 (was os.tmpdir(), see LEGACY_CACHE_FILE below).
20
+ // getMachineId() derives (primary, secondary) from TWO independent execSync
21
+ // calls, each wrapped in its own try/catch that falls through silently on
22
+ // failure (see the comments at those calls). If either call transiently
23
+ // timed out, the resulting pairing was composed differently than a
24
+ // successful run, producing a DIFFERENT sha256 machine hash for the exact
25
+ // same physical machine — surfacing server-side as a false
26
+ // machine_not_found lockout, not a security event. A cache in system temp
27
+ // does not survive Windows Disk Cleanup or some machines' reboot behavior,
28
+ // which re-exposed the flaky first-computation path on any later restart
29
+ // where WMI happened to be slow. Matches the durable pattern boot.key
30
+ // already uses (getBootKeyPath() in cli/commands/serve.js) — same
31
+ // AppData/Roaming/clauth (win32) / ~/.config/clauth (else) directory.
32
+ function getCacheDir() {
33
+ if (os.platform() === "win32") {
34
+ return path.join(process.env.APPDATA || path.join(os.homedir(), "AppData", "Roaming"), "clauth");
35
+ }
36
+ return path.join(os.homedir(), ".config", "clauth");
37
+ }
38
+ const CACHE_FILE = path.join(getCacheDir(), "machine.cache");
39
+ // Pre-2026-09-06 location. An already-registered machine may still have a
40
+ // valid cache here; migrate it into the durable location on first read
41
+ // instead of recomputing via WMI — recomputing is exactly the flaky path
42
+ // this fix exists to avoid, and the durable write means this only ever
43
+ // happens once per machine.
44
+ const LEGACY_CACHE_FILE = path.join(os.tmpdir(), "clauth-machine.cache");
19
45
 
20
46
  function readCache() {
21
47
  try {
@@ -23,12 +49,24 @@ function readCache() {
23
49
  // Validate: must be two non-empty lines (primary:secondary)
24
50
  const [primary, secondary] = raw.split("\n");
25
51
  if (primary && secondary) return { primary: primary.trim(), secondary: secondary.trim() };
26
- } catch { /* cache miss */ }
52
+ } catch { /* durable cache miss */ }
53
+ try {
54
+ const raw = fs.readFileSync(LEGACY_CACHE_FILE, "utf8").trim();
55
+ const [primary, secondary] = raw.split("\n");
56
+ if (primary && secondary) {
57
+ const migrated = { primary: primary.trim(), secondary: secondary.trim() };
58
+ writeCache(migrated.primary, migrated.secondary);
59
+ return migrated;
60
+ }
61
+ } catch { /* no legacy cache either */ }
27
62
  return null;
28
63
  }
29
64
 
30
65
  function writeCache(primary, secondary) {
31
- try { fs.writeFileSync(CACHE_FILE, `${primary}\n${secondary}`, "utf8"); } catch { /* best effort */ }
66
+ try {
67
+ fs.mkdirSync(getCacheDir(), { recursive: true });
68
+ fs.writeFileSync(CACHE_FILE, `${primary}\n${secondary}`, "utf8");
69
+ } catch { /* best effort */ }
32
70
  }
33
71
 
34
72
  function getMachineId() {
@@ -140,4 +178,8 @@ export function deriveSeedHash(machineHash, password) {
140
178
  .digest("hex");
141
179
  }
142
180
 
181
+ // Exported for test path introspection only — not part of the public
182
+ // CLI/API surface.
183
+ export { CACHE_FILE, LEGACY_CACHE_FILE, getCacheDir };
184
+
143
185
  export default { getMachineHash, deriveToken, deriveSeedHash };
@@ -4,7 +4,12 @@ import fs from "node:fs";
4
4
  import os from "node:os";
5
5
  import path from "node:path";
6
6
 
7
- import { getMachineHash, deriveToken, deriveSeedHash } from "./fingerprint.js";
7
+ import { getMachineHash, deriveToken, deriveSeedHash, CACHE_FILE, LEGACY_CACHE_FILE } from "./fingerprint.js";
8
+
9
+ // CACHE_FILE/LEGACY_CACHE_FILE are the REAL production paths (durable
10
+ // AppData/Roaming/clauth on win32, formerly os.tmpdir()) — read-only in this
11
+ // suite. Every test below only ever reads mtimeMs, never asserts on or
12
+ // writes machine-identifying content into either file.
8
13
 
9
14
  // getMachineId() normally shells out to WMI/registry/ioreg, which this harness
10
15
  // must never do (slow, platform-dependent, pollutes nothing but still real
@@ -84,12 +89,31 @@ test("deriveSeedHash changes with the password", () => {
84
89
  assert.notEqual(deriveSeedHash(hash, "one"), deriveSeedHash(hash, "two"));
85
90
  });
86
91
 
87
- test("machine id cache: a fresh CLAUTH_MACHINE_ID call never touches the WMI cache file", () => {
92
+ test("machine id cache: a fresh CLAUTH_MACHINE_ID call never touches the WMI cache file (durable or legacy)", () => {
88
93
  // Positive control that the cache mechanism exists and this test can see it,
89
94
  // so "the cache file was untouched" below actually means something.
90
- const cacheFile = path.join(os.tmpdir(), "clauth-machine.cache");
91
- const before = fs.existsSync(cacheFile) ? fs.statSync(cacheFile).mtimeMs : null;
95
+ const before = {
96
+ durable: fs.existsSync(CACHE_FILE) ? fs.statSync(CACHE_FILE).mtimeMs : null,
97
+ legacy: fs.existsSync(LEGACY_CACHE_FILE) ? fs.statSync(LEGACY_CACHE_FILE).mtimeMs : null,
98
+ };
92
99
  withMachineId("container-fast-path", () => getMachineHash());
93
- const after = fs.existsSync(cacheFile) ? fs.statSync(cacheFile).mtimeMs : null;
94
- assert.equal(before, after, "the container-id fast path must not read or write the WMI cache file");
100
+ const after = {
101
+ durable: fs.existsSync(CACHE_FILE) ? fs.statSync(CACHE_FILE).mtimeMs : null,
102
+ legacy: fs.existsSync(LEGACY_CACHE_FILE) ? fs.statSync(LEGACY_CACHE_FILE).mtimeMs : null,
103
+ };
104
+ assert.deepEqual(before, after, "the container-id fast path must not read or write either cache file");
105
+ });
106
+
107
+ test("cache paths are durable (AppData/Roaming/clauth or ~/.config/clauth), never os.tmpdir()", () => {
108
+ // The whole point of this fix: CACHE_FILE must NOT live in system temp,
109
+ // which Windows Disk Cleanup (or equivalent) can wipe, re-exposing the
110
+ // flaky first-computation WMI path on the next restart.
111
+ assert.ok(!CACHE_FILE.startsWith(os.tmpdir()), `CACHE_FILE must be durable, got: ${CACHE_FILE}`);
112
+ assert.equal(path.basename(CACHE_FILE), "machine.cache");
113
+ const dir = path.dirname(CACHE_FILE);
114
+ assert.ok(dir.endsWith(path.join("clauth")), `cache dir must be a clauth-owned config directory, got: ${dir}`);
115
+ // LEGACY_CACHE_FILE is the pre-fix location, kept only so readCache() can
116
+ // migrate an already-registered machine's existing cache instead of
117
+ // recomputing via WMI (see fingerprint.js).
118
+ assert.equal(LEGACY_CACHE_FILE, path.join(os.tmpdir(), "clauth-machine.cache"));
95
119
  });
@@ -18,6 +18,22 @@ const ACTIONS = new Set(["start", "stop", "restart", "reconcile", "test", "promo
18
18
  const DEFAULT_HEALTH_RECONCILE_INTERVAL_MS = 10000;
19
19
  const DEFAULT_HEALTH_TIMEOUT_MS = 2500;
20
20
  const HEALTH_RECONCILE_COOLDOWN_MS = 15000;
21
+ // After this many CONSECUTIVE failed reconcile attempts (each gated by
22
+ // HEALTH_RECONCILE_COOLDOWN_MS, so >= 5 * 15s = 75s before this fires),
23
+ // reconcileSurfaceHealth() stops auto-restarting the surface and latches it
24
+ // into "needs_help" instead of retrying forever. Without a cap, a surface
25
+ // that can never come up (a missing dependency, a bad config) gets
26
+ // relaunched every cooldown window FOREVER: the restart command itself can
27
+ // exit 0 (pm2 accepted it) while the process crashes immediately after, so
28
+ // this loop never sees a hard failure to stop on, and pm2's OWN crash-loop
29
+ // backoff never engages either because clauth's periodic explicit restart
30
+ // keeps resetting it. Root cause of the regen-media-local window-flashing
31
+ // incident (2026-09-05): a missing `tsx` dependency meant every restart
32
+ // "succeeded" and immediately crashed, forever, because nothing was
33
+ // counting. A latched surface self-heals the moment its health probe
34
+ // succeeds (handled before this cap is ever consulted) or a human/agent
35
+ // issues an explicit non-reconcile start/stop/restart (see runSurfaceAction).
36
+ export const MAX_CONSECUTIVE_RECONCILE_FAILURES = 5;
21
37
  const DOCUMENTATION_FIELDS = ["architecture", "operator_guide", "install", "runbook", "tool_reference", "release", "agent_context"];
22
38
 
23
39
  function hasMcpToken(value) {
@@ -182,11 +198,25 @@ export async function reconcileSurfaceHealth({ fetchImpl = globalThis.fetch, tim
182
198
  const error = health.error;
183
199
 
184
200
  if (healthy) {
185
- updateSurfaceState(id, { state: "current", last_health_at: observedAt, last_health_ok: true, last_health_error: null });
201
+ updateSurfaceState(id, { state: "current", last_health_at: observedAt, last_health_ok: true, last_health_error: null, consecutive_reconcile_failures: 0 });
186
202
  inspected.push({ surface_id: id, state: "healthy", observed_at: observedAt });
187
203
  continue;
188
204
  }
189
205
 
206
+ // Latched: this surface already exhausted MAX_CONSECUTIVE_RECONCILE_FAILURES
207
+ // and clauth stopped auto-restarting it. Record the failed probe (so the
208
+ // UI's "last checked" freshness stays honest) WITHOUT calling
209
+ // runSurfaceAction again — that call is exactly the crash loop this latch
210
+ // exists to stop. The only ways out are a health probe that succeeds
211
+ // (the branch above, checked every tick regardless of latch state) or an
212
+ // explicit human/agent action (runSurfaceAction resets the counter and
213
+ // clears this latch on any successful non-reconcile command).
214
+ if (surface.state === "needs_help") {
215
+ updateSurfaceState(id, { last_health_at: observedAt, last_health_ok: false, last_health_error: error });
216
+ inspected.push({ surface_id: id, state: "needs_help", error, latched: true, observed_at: observedAt });
217
+ continue;
218
+ }
219
+
190
220
  const lastAttempt = Date.parse(surface.last_reconcile_at || "") || 0;
191
221
  if (Date.now() - lastAttempt < HEALTH_RECONCILE_COOLDOWN_MS) {
192
222
  updateSurfaceState(id, { state: "unavailable", last_health_at: observedAt, last_health_ok: false, last_health_error: error });
@@ -206,15 +236,52 @@ export async function reconcileSurfaceHealth({ fetchImpl = globalThis.fetch, tim
206
236
  const commandCompleted = receipt?.resulting_state?.ok === true;
207
237
  const postHealth = commandCompleted ? await probeSurfaceHealth(url, fetchImpl, timeoutMs) : { healthy: false, error: receipt?.resulting_state?.state || "reconcile_failed" };
208
238
  const repaired = commandCompleted && postHealth.healthy;
239
+
240
+ if (repaired) {
241
+ updateSurfaceState(id, {
242
+ state: "current",
243
+ last_health_at: now(),
244
+ last_health_ok: true,
245
+ last_health_error: null,
246
+ last_reconcile_operation_id: receipt?.operationId || null,
247
+ consecutive_reconcile_failures: 0,
248
+ });
249
+ appendSupervisorEvent({ kind: "surface_reconciled", surface_id: id, operation_id: receipt?.operationId || null, command_completed: commandCompleted, health_ok: true, error: null });
250
+ inspected.push({ surface_id: id, state: "reconciled", error: null, operation_id: receipt?.operationId || null, observed_at: observedAt });
251
+ continue;
252
+ }
253
+
254
+ // Not repaired: count it. MAX_CONSECUTIVE_RECONCILE_FAILURES is the
255
+ // circuit breaker — cross it and the surface latches to "needs_help"
256
+ // instead of feeding another attempt back into runSurfaceAction next tick.
257
+ const failures = (surface.consecutive_reconcile_failures || 0) + 1;
258
+ const exhausted = failures >= MAX_CONSECUTIVE_RECONCILE_FAILURES;
209
259
  updateSurfaceState(id, {
210
- state: repaired ? "current" : "unavailable",
260
+ state: exhausted ? "needs_help" : "unavailable",
211
261
  last_health_at: now(),
212
- last_health_ok: repaired,
213
- last_health_error: repaired ? null : postHealth.error,
262
+ last_health_ok: false,
263
+ last_health_error: postHealth.error || error,
214
264
  last_reconcile_operation_id: receipt?.operationId || null,
265
+ consecutive_reconcile_failures: failures,
266
+ });
267
+ appendSupervisorEvent({
268
+ kind: exhausted ? "surface_needs_help" : "surface_reconcile_failed",
269
+ surface_id: id,
270
+ operation_id: receipt?.operationId || null,
271
+ command_completed: commandCompleted,
272
+ health_ok: false,
273
+ error: postHealth.error || error,
274
+ consecutive_failures: failures,
275
+ ...(exhausted ? { message: `${id} failed to come up after ${failures} consecutive restart attempts — clauth has stopped auto-restarting it. Check its log, fix the underlying cause, then Start/Restart it manually (dashboard or \`clauth mcp restart ${id}\`) to resume auto-repair.` } : {}),
276
+ });
277
+ inspected.push({
278
+ surface_id: id,
279
+ state: exhausted ? "needs_help" : "reconcile_failed",
280
+ error: postHealth.error || error,
281
+ operation_id: receipt?.operationId || null,
282
+ consecutive_failures: failures,
283
+ observed_at: observedAt,
215
284
  });
216
- appendSupervisorEvent({ kind: repaired ? "surface_reconciled" : "surface_reconcile_failed", surface_id: id, operation_id: receipt?.operationId || null, command_completed: commandCompleted, health_ok: postHealth.healthy, error: postHealth.error || null });
217
- inspected.push({ surface_id: id, state: repaired ? "reconciled" : "reconcile_failed", error: postHealth.error || error, operation_id: receipt?.operationId || null, observed_at: observedAt });
218
285
  }
219
286
  return { inspected };
220
287
  }
@@ -1245,6 +1312,22 @@ export function runSurfaceAction(id, action, actor = "localhost") {
1245
1312
  fallbackUsed = true;
1246
1313
  result.stderr = `restart exited ${restartStatus}; start fallback attempted\n${result.stderr || ""}`;
1247
1314
  }
1315
+ // A deliberate, successful manual action — start/stop/restart, never
1316
+ // "reconcile" itself — is a human or agent saying "I looked at this."
1317
+ // Give the surface a fresh crash-loop budget instead of carrying forward a
1318
+ // counter (or a "needs_help" latch) from before the intervention. Only
1319
+ // reconcileSurfaceHealth's own automatic attempts count against
1320
+ // MAX_CONSECUTIVE_RECONCILE_FAILURES, so this is the other side of that
1321
+ // cap: the way OUT of "needs_help", not just the way IN. Reset does not
1322
+ // claim the surface is healthy again — it has done no health probe — it
1323
+ // only clears the latch so the next reconcile tick (or health probe) gets
1324
+ // to judge it fresh.
1325
+ if (action !== "reconcile" && result.status === 0) {
1326
+ updateSurfaceState(id, {
1327
+ consecutive_reconcile_failures: 0,
1328
+ state: surface.state === "needs_help" ? "unavailable" : surface.state,
1329
+ });
1330
+ }
1248
1331
  return operation(action, { surface_id: id }, surface, {
1249
1332
  ok: result.status === 0,
1250
1333
  state: result.status === 0 ? "operation_completed" : "operation_failed",
@@ -12,8 +12,10 @@ import {
12
12
  isMcpServerPlugin,
13
13
  listPlugins,
14
14
  listSurfaces,
15
+ MAX_CONSECUTIVE_RECONCILE_FAILURES,
15
16
  probeAllSurfaceHealth,
16
17
  surfaceOpenUrl,
18
+ readSupervisorEvents,
17
19
  reconcileSurfaceHealth,
18
20
  registerPlugin,
19
21
  runPluginAction,
@@ -60,6 +62,33 @@ async function withTempSupervisor(fn) {
60
62
  }
61
63
  }
62
64
 
65
+ // Fast-forwards Date.now()/`new Date()` without real sleeping.
66
+ // HEALTH_RECONCILE_COOLDOWN_MS is a real 15s per attempt, and proving the
67
+ // crash-loop cap needs MAX_CONSECUTIVE_RECONCILE_FAILURES of them -- a test
68
+ // cannot afford to actually sleep that long. Overriding ONLY Date.now() is
69
+ // not enough: now() (supervisor-registry.js) stamps last_reconcile_at via
70
+ // `new Date().toISOString()`, so the cooldown gate (Date.now() minus
71
+ // Date.parse(last_reconcile_at)) would compare a mocked "now" against a REAL
72
+ // timestamp and never agree. Subclassing keeps every other Date behavior
73
+ // (Date.parse, `new Date(iso)`) real.
74
+ async function withMockedClock(fn) {
75
+ const RealDate = global.Date;
76
+ let clock = RealDate.now();
77
+ class MockDate extends RealDate {
78
+ constructor(...args) {
79
+ if (args.length === 0) super(clock);
80
+ else super(...args);
81
+ }
82
+ static now() { return clock; }
83
+ }
84
+ global.Date = MockDate;
85
+ try {
86
+ return await fn({ advance: (ms) => { clock += ms; } });
87
+ } finally {
88
+ global.Date = RealDate;
89
+ }
90
+ }
91
+
63
92
  function writePlugin(root, source, id, manifest) {
64
93
  const dir = path.join(root, source, id);
65
94
  fs.mkdirSync(dir, { recursive: true });
@@ -453,6 +482,135 @@ test("health reconciliation does not claim current when a successful command lea
453
482
  }
454
483
  });
455
484
 
485
+ test("a crash-looping surface stops auto-restarting after MAX_CONSECUTIVE_RECONCILE_FAILURES and needs a human", async () => {
486
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "clauth-supervisor-crashloop-"));
487
+ const oldDir = process.env.CLAUTH_SUPERVISOR_DIR;
488
+ const oldManaged = process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
489
+ const oldUser = process.env.CLAUTH_USER_PLUGIN_ROOTS;
490
+ process.env.CLAUTH_SUPERVISOR_DIR = root;
491
+ process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = path.join(root, "managed");
492
+ process.env.CLAUTH_USER_PLUGIN_ROOTS = path.join(root, "user");
493
+ try {
494
+ // restart "succeeds" (exit 0 -- pm2 would accept it) but health NEVER
495
+ // recovers: exactly the regen-media-local shape -- a process that
496
+ // restarts cleanly and crashes immediately after, forever.
497
+ writePlugin(root, "managed", "crashloop-demo", baseManifest("crashloop-demo", {
498
+ core: true,
499
+ enable_default: true,
500
+ surfaces: [{
501
+ id: "primary",
502
+ destination: "local/clauth/pm2",
503
+ lifecycle_owner: "clauth",
504
+ port: 39121,
505
+ health: "/health",
506
+ restart: [process.execPath, "--version"],
507
+ start: [process.execPath, "--version"],
508
+ }],
509
+ }));
510
+ discoverPlugins();
511
+ const alwaysDown = { fetchImpl: async () => ({ ok: false, status: 503 }) };
512
+
513
+ await withMockedClock(async ({ advance }) => {
514
+ let last;
515
+ for (let i = 0; i < MAX_CONSECUTIVE_RECONCILE_FAILURES; i++) {
516
+ advance(20000); // past the 15s reconcile cooldown every tick
517
+ last = await reconcileSurfaceHealth(alwaysDown);
518
+ }
519
+ const surface = listSurfaces().find((item) => item.plugin_id === "crashloop-demo");
520
+ assert.equal(surface.state, "needs_help", "must latch once the cap is reached");
521
+ assert.equal(surface.consecutive_reconcile_failures, MAX_CONSECUTIVE_RECONCILE_FAILURES);
522
+ assert.equal(last.inspected[0].state, "needs_help");
523
+
524
+ // THE load-bearing assertion: one more tick past cooldown must NOT
525
+ // spawn another restart. Count "reconcile" operation receipts before
526
+ // and after -- if the count moves, the loop never actually stopped.
527
+ const before = readSupervisorEvents(1000).filter((e) => e.kind === "operation" && e.action === "reconcile").length;
528
+ advance(20000);
529
+ const again = await reconcileSurfaceHealth(alwaysDown);
530
+ const after = readSupervisorEvents(1000).filter((e) => e.kind === "operation" && e.action === "reconcile").length;
531
+ assert.equal(after, before, "a latched surface must not be re-restarted -- that is the crash loop this cap exists to stop");
532
+ assert.equal(again.inspected[0].state, "needs_help");
533
+ assert.equal(again.inspected[0].latched, true);
534
+
535
+ // Self-heal: health recovering on its own (independent of clauth's
536
+ // restart) clears the latch -- the cap is a circuit breaker, not a
537
+ // permanent ban.
538
+ advance(20000);
539
+ const healed = await reconcileSurfaceHealth({ fetchImpl: async () => ({ ok: true, status: 200 }) });
540
+ assert.equal(healed.inspected[0].state, "healthy");
541
+ const healedSurface = listSurfaces().find((item) => item.plugin_id === "crashloop-demo");
542
+ assert.equal(healedSurface.state, "current");
543
+ assert.equal(healedSurface.consecutive_reconcile_failures, 0);
544
+ });
545
+ } finally {
546
+ if (oldDir === undefined) delete process.env.CLAUTH_SUPERVISOR_DIR;
547
+ else process.env.CLAUTH_SUPERVISOR_DIR = oldDir;
548
+ if (oldManaged === undefined) delete process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
549
+ else process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = oldManaged;
550
+ if (oldUser === undefined) delete process.env.CLAUTH_USER_PLUGIN_ROOTS;
551
+ else process.env.CLAUTH_USER_PLUGIN_ROOTS = oldUser;
552
+ fs.rmSync(root, { recursive: true, force: true });
553
+ }
554
+ });
555
+
556
+ test("an explicit manual restart clears a needs_help latch and gives a fresh crash-loop budget", async () => {
557
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "clauth-supervisor-manual-reset-"));
558
+ const oldDir = process.env.CLAUTH_SUPERVISOR_DIR;
559
+ const oldManaged = process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
560
+ const oldUser = process.env.CLAUTH_USER_PLUGIN_ROOTS;
561
+ process.env.CLAUTH_SUPERVISOR_DIR = root;
562
+ process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = path.join(root, "managed");
563
+ process.env.CLAUTH_USER_PLUGIN_ROOTS = path.join(root, "user");
564
+ try {
565
+ writePlugin(root, "managed", "manual-reset-demo", baseManifest("manual-reset-demo", {
566
+ core: true,
567
+ enable_default: true,
568
+ surfaces: [{
569
+ id: "primary",
570
+ destination: "local/clauth/pm2",
571
+ lifecycle_owner: "clauth",
572
+ port: 39122,
573
+ health: "/health",
574
+ restart: [process.execPath, "--version"],
575
+ start: [process.execPath, "--version"],
576
+ }],
577
+ }));
578
+ discoverPlugins();
579
+ const alwaysDown = { fetchImpl: async () => ({ ok: false, status: 503 }) };
580
+
581
+ await withMockedClock(async ({ advance }) => {
582
+ for (let i = 0; i < MAX_CONSECUTIVE_RECONCILE_FAILURES; i++) {
583
+ advance(20000);
584
+ await reconcileSurfaceHealth(alwaysDown);
585
+ }
586
+ assert.equal(listSurfaces()[0].state, "needs_help");
587
+
588
+ // A human/agent looks at it and issues an explicit restart -- through
589
+ // runSurfaceAction directly, never through the "reconcile" action the
590
+ // background loop uses.
591
+ const receipt = runSurfaceAction("manual-reset-demo:primary", "restart", "human");
592
+ assert.equal(receipt.resulting_state.ok, true);
593
+ const reset = listSurfaces()[0];
594
+ assert.equal(reset.state, "unavailable", "cleared out of needs_help, but not falsely claimed healthy without a probe");
595
+ assert.equal(reset.consecutive_reconcile_failures, 0);
596
+
597
+ // And the surface gets its full budget back, not zero.
598
+ advance(20000);
599
+ const tick = await reconcileSurfaceHealth(alwaysDown);
600
+ assert.equal(tick.inspected[0].state, "reconcile_failed");
601
+ assert.equal(listSurfaces()[0].consecutive_reconcile_failures, 1);
602
+ });
603
+ } finally {
604
+ if (oldDir === undefined) delete process.env.CLAUTH_SUPERVISOR_DIR;
605
+ else process.env.CLAUTH_SUPERVISOR_DIR = oldDir;
606
+ if (oldManaged === undefined) delete process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
607
+ else process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = oldManaged;
608
+ if (oldUser === undefined) delete process.env.CLAUTH_USER_PLUGIN_ROOTS;
609
+ else process.env.CLAUTH_USER_PLUGIN_ROOTS = oldUser;
610
+ fs.rmSync(root, { recursive: true, force: true });
611
+ }
612
+ });
613
+
456
614
  test("health reconciliation never restarts external or plugin-owned surfaces", async () => {
457
615
  const root = fs.mkdtempSync(path.join(os.tmpdir(), "clauth-supervisor-observe-"));
458
616
  const oldDir = process.env.CLAUTH_SUPERVISOR_DIR;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lifeaitools/clauth",
3
- "version": "2.15.5",
3
+ "version": "2.15.7",
4
4
  "description": "Hardware-bound credential vault for the LIFEAI infrastructure stack",
5
5
  "type": "module",
6
6
  "bin": {
@@ -1,5 +1,12 @@
1
- // clauth — auth-vault Edge Function v2
2
- // Added: IP whitelist, rate limiting, machine lockout (fail_count + locked)
1
+ // clauth — auth-vault Edge Function v3
2
+ // IP whitelist + machine lockout (fail_count + locked) remain. Rate limiting
3
+ // and audit logging (clauth_audit) removed 2026-09-03 -- see the comments at
4
+ // validateHMAC and where auditLog used to be defined for why: the rate-limit
5
+ // check was an unindexed COUNT against clauth_audit run on EVERY request,
6
+ // pre-auth, and every rejection it produced also wrote an audit row -- a
7
+ // feedback loop that took the whole project down under sustained load. The
8
+ // security value it provided was already covered, tighter, by the
9
+ // per-machine 5-failed-attempts lockout below.
3
10
 
4
11
  import { createClient } from "https://esm.sh/@supabase/supabase-js@2";
5
12
 
@@ -11,8 +18,6 @@ const ADMIN_BOOTSTRAP_TOKEN = Deno.env.get("CLAUTH_ADMIN_BOOTSTRAP_TOKEN")!;
11
18
  const ALLOWED_IPS: string[] = (Deno.env.get("CLAUTH_ALLOWED_IPS") || "")
12
19
  .split(",").map(s => s.trim()).filter(Boolean);
13
20
 
14
- const RATE_LIMIT_MAX = 30;
15
- const RATE_LIMIT_WINDOW = 60;
16
21
  const REPLAY_WINDOW_MS = 5 * 60 * 1000;
17
22
  const MAX_FAIL_COUNT = 5;
18
23
  const DEFAULT_INSTALL_ID = "default";
@@ -55,20 +60,36 @@ function checkIP(ip: string): { allowed: boolean; reason?: string } {
55
60
  return { allowed: false, reason: `IP not whitelisted: ${ip}` };
56
61
  }
57
62
 
58
- async function checkRateLimit(sb: any, machine_hash: string): Promise<{ allowed: boolean; reason?: string }> {
59
- const windowStart = new Date(Date.now() - RATE_LIMIT_WINDOW * 1000).toISOString();
60
- const { count } = await sb.from("clauth_audit")
61
- .select("id", { count: "exact", head: true })
62
- .eq("machine_hash", machine_hash)
63
- .gte("created_at", windowStart);
64
- if ((count || 0) >= RATE_LIMIT_MAX) {
65
- return { allowed: false, reason: `Rate limit: ${count}/${RATE_LIMIT_MAX} per ${RATE_LIMIT_WINDOW}s` };
66
- }
67
- return { allowed: true };
68
- }
69
-
63
+ // rate-limiting removed 2026-09-03 -- see supervisor-registry.js's own
64
+ // state.json lock history for the shape of this bug: checkRateLimit() ran
65
+ // this exact query -- a COUNT against clauth_audit filtered by machine_hash
66
+ // and created_at -- on EVERY request, BEFORE authentication (validateHMAC
67
+ // below), against a table with no index on either column and no retention.
68
+ // As the table grew (174,385 rows, unbounded), the scan cost grew with it;
69
+ // under sustained traffic the queries started stacking, Postgres killed them
70
+ // at statement_timeout, and Cloudflare killed the stacked connections at its
71
+ // own 90s ceiling (522s) -- and every one of those rejections ALSO wrote an
72
+ // audit row (see auditLog's removal below), so the failure fed itself. The
73
+ // actual security protection this duplicated is already provided, tighter,
74
+ // by the per-machine lockout in validateHMAC below (5 failed attempts locks
75
+ // the machine -- well under the 30-per-60s this used to allow). Rate
76
+ // limiting for a genuine runaway client belongs at Cloudflare, in front of
77
+ // this function, not as a synchronous Postgres query on the hot path of
78
+ // every request.
70
79
  async function validateHMAC(sb: any, body: any): Promise<{ valid: boolean; reason?: string }> {
71
80
  const now = Date.now();
81
+
82
+ // A caller that omits machine_hash entirely used to fall straight through
83
+ // to the clauth_machines lookup below with body.machine_hash === undefined.
84
+ // PostgREST's .single() then answers "0 rows" for a query that can never
85
+ // match, which comes back as an HTTP 406 -- 1,438 of these in 24h
86
+ // (2026-09-06), every one a wasted round trip for a request already known
87
+ // to be malformed. Reject before the DB call: same auth_failed response
88
+ // the caller already gets, no query, no spurious 406 in the logs.
89
+ if (typeof body.machine_hash !== "string" || !body.machine_hash) {
90
+ return { valid: false, reason: "machine_hash_missing" };
91
+ }
92
+
72
93
  if (Math.abs(now - body.timestamp) > REPLAY_WINDOW_MS) return { valid: false, reason: "timestamp_expired" };
73
94
 
74
95
  const { data: machine, error } = await sb.from("clauth_machines")
@@ -99,20 +120,23 @@ async function validateHMAC(sb: any, body: any): Promise<{ valid: boolean; reaso
99
120
  return { valid: true };
100
121
  }
101
122
 
102
- async function auditLog(sb: any, machine_hash: string, service_name: string, action: string, result: string, detail?: string) {
103
- await sb.from("clauth_audit").insert({ machine_hash, service_name, action, result, detail });
104
- }
105
-
123
+ // auditLog() removed 2026-09-03 alongside checkRateLimit() -- it wrote a row
124
+ // to clauth_audit on every path through this function, including rejections
125
+ // (rate-limited, auth-denied, IP-blocked), which is what turned "the table
126
+ // got slow" into a feedback loop (more rejections -> more rows -> slower
127
+ // scans -> more rejections). No retention policy ever existed for this
128
+ // table. If durable audit logging is wanted back, it needs its own design --
129
+ // write-only, indexed for its actual read pattern (if any), with a real
130
+ // retention/partition policy -- not a bare insert-on-every-call with no cap.
106
131
  async function handleRetrieve(sb: any, body: any, mh: string) {
107
132
  const { service } = body;
108
133
  if (!service) return { error: "service required" };
109
134
  const { data: svc } = await sb.from("clauth_services").select("*").eq("name", service).single();
110
- if (!svc) { await auditLog(sb, mh, service, "retrieve", "fail", "service_not_found"); return { error: "service_not_found" }; }
111
- if (!svc.enabled) { await auditLog(sb, mh, service, "retrieve", "denied", "service_disabled"); return { error: "service_disabled" }; }
112
- if (!svc.vault_key) { await auditLog(sb, mh, service, "retrieve", "fail", "no_key_stored"); return { error: "no_key_stored" }; }
135
+ if (!svc) return { error: "service_not_found" };
136
+ if (!svc.enabled) return { error: "service_disabled" };
137
+ if (!svc.vault_key) return { error: "no_key_stored" };
113
138
  const { data: secret } = await sb.rpc("vault_decrypt_secret", { secret_name: svc.vault_key });
114
139
  await sb.from("clauth_services").update({ last_retrieved: new Date().toISOString() }).eq("name", service);
115
- await auditLog(sb, mh, service, "retrieve", "success");
116
140
  return { service, key_type: svc.key_type, value: secret };
117
141
  }
118
142
 
@@ -121,9 +145,8 @@ async function handleWrite(sb: any, body: any, mh: string) {
121
145
  if (!service || !value) return { error: "service and value required" };
122
146
  const vaultKey = `clauth.${service}`;
123
147
  const { error } = await sb.rpc("vault_upsert_secret", { secret_name: vaultKey, secret_value: typeof value === "string" ? value : JSON.stringify(value) });
124
- if (error) { await auditLog(sb, mh, service, "write", "fail", error.message); return { error: error.message }; }
148
+ if (error) return { error: error.message };
125
149
  await sb.from("clauth_services").update({ vault_key: vaultKey, last_rotated: new Date().toISOString() }).eq("name", service);
126
- await auditLog(sb, mh, service, "write", "success");
127
150
  return { success: true, service, vault_key: vaultKey };
128
151
  }
129
152
 
@@ -133,7 +156,6 @@ async function handleEnable(sb: any, body: any, mh: string) {
133
156
  q = service !== "all" ? q.eq("name", service) : q.not("vault_key", "is", null);
134
157
  const { error } = await q;
135
158
  if (error) return { error: error.message };
136
- await auditLog(sb, mh, service, enabled ? "enable" : "disable", "success");
137
159
  return { success: true, service, enabled };
138
160
  }
139
161
 
@@ -144,7 +166,6 @@ async function handleAdd(sb: any, body: any, mh: string) {
144
166
  if (project) row.project = project;
145
167
  const { error } = await sb.from("clauth_services").insert(row);
146
168
  if (error) return { error: error.message };
147
- await auditLog(sb, mh, name, "add", "success");
148
169
  return { success: true, name, label, key_type, project: project || null };
149
170
  }
150
171
 
@@ -157,7 +178,6 @@ async function handleUpdate(sb: any, body: any, mh: string) {
157
178
  if (description !== undefined) updates.description = description || null;
158
179
  const { error } = await sb.from("clauth_services").update(updates).eq("name", service);
159
180
  if (error) return { error: error.message };
160
- await auditLog(sb, mh, service, "update", "success", `fields: ${Object.keys(updates).join(", ")}`);
161
181
  return { success: true, service, ...updates };
162
182
  }
163
183
 
@@ -166,7 +186,6 @@ async function handleRemove(sb: any, body: any, mh: string) {
166
186
  if (confirm !== `CONFIRM REMOVE ${service.toUpperCase()}`) return { error: "confirm phrase mismatch" };
167
187
  await sb.rpc("vault_delete_secret", { secret_name: `clauth.${service}` });
168
188
  await sb.from("clauth_services").delete().eq("name", service);
169
- await auditLog(sb, mh, service, "remove", "success");
170
189
  return { success: true, service };
171
190
  }
172
191
 
@@ -182,7 +201,6 @@ async function handleRevoke(sb: any, body: any, mh: string) {
182
201
  await sb.rpc("vault_delete_secret", { secret_name: `clauth.${service}` });
183
202
  await sb.from("clauth_services").update({ vault_key: null, enabled: false }).eq("name", service);
184
203
  }
185
- await auditLog(sb, mh, service, "revoke", "success");
186
204
  return { success: true, service };
187
205
  }
188
206
 
@@ -193,7 +211,6 @@ async function handleStatus(sb: any, body: any, mh: string) {
193
211
  .order("name");
194
212
  if (body.project) q = q.eq("project", body.project);
195
213
  const { data: services } = await q;
196
- await auditLog(sb, mh, "all", "status", "success");
197
214
  return { services: services || [] };
198
215
  }
199
216
 
@@ -204,7 +221,6 @@ async function handleChangePassword(sb: any, body: any, mh: string) {
204
221
  .update({ hmac_seed_hash: new_hmac_seed_hash, fail_count: 0, locked: false })
205
222
  .eq("machine_hash", mh);
206
223
  if (error) return { error: error.message };
207
- await auditLog(sb, mh, "system", "change-password", "success");
208
224
  return { success: true };
209
225
  }
210
226
 
@@ -231,12 +247,8 @@ async function handleCreateEnrollment(sb: any, body: any, mh: string) {
231
247
  created_by_machine_hash: mh,
232
248
  expires_at,
233
249
  });
234
- if (error) {
235
- await auditLog(sb, mh, "system", "create-enrollment", "fail", error.message);
236
- return { error: error.message };
237
- }
250
+ if (error) return { error: error.message };
238
251
 
239
- await auditLog(sb, mh, "system", "create-enrollment", "success", `install_id=${install_id}`);
240
252
  return { success: true, enrollment_code: code, install_id, expires_at, label };
241
253
  }
242
254
 
@@ -274,7 +286,6 @@ async function handleRedeemEnrollment(sb: any, body: any) {
274
286
  if (consumeError) return { error: consumeError.message };
275
287
  if (!consumedRows || consumedRows.length !== 1) return { error: "enrollment_already_used" };
276
288
 
277
- await auditLog(sb, machine_hash, "system", "redeem-enrollment", "success", `install_id=${install_id}`);
278
289
  return { success: true, machine_hash, install_id };
279
290
  }
280
291
 
@@ -314,21 +325,11 @@ Deno.serve(async (req: Request) => {
314
325
 
315
326
  const ipCheck = checkIP(ip);
316
327
  if (!ipCheck.allowed) {
317
- await auditLog(sb, body.machine_hash || "unknown", "system", route, "blocked", ipCheck.reason);
318
328
  return Response.json({ error: "ip_blocked", reason: ipCheck.reason }, { status: 403 });
319
329
  }
320
330
 
321
- if (body.machine_hash) {
322
- const rateCheck = await checkRateLimit(sb, body.machine_hash);
323
- if (!rateCheck.allowed) {
324
- await auditLog(sb, body.machine_hash, "system", route, "rate_limited", rateCheck.reason);
325
- return Response.json({ error: "rate_limited", reason: rateCheck.reason }, { status: 429 });
326
- }
327
- }
328
-
329
331
  const authResult = await validateHMAC(sb, { machine_hash: body.machine_hash, token: body.token, timestamp: body.timestamp, password: body.password });
330
332
  if (!authResult.valid) {
331
- await auditLog(sb, body.machine_hash || "unknown", body.service || "unknown", route, "denied", authResult.reason);
332
333
  return Response.json({ error: "auth_failed", reason: authResult.reason }, { status: 401 });
333
334
  }
334
335