@lifeaitools/clauth 2.15.5 → 2.15.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/cli/dashboard/panels/surfaces-panel.css +9 -0
- package/cli/dashboard/panels/surfaces-panel.js +19 -3
- package/cli/fingerprint.js +45 -3
- package/cli/fingerprint.test.js +30 -6
- package/cli/supervisor-registry.js +89 -6
- package/cli/supervisor-registry.test.js +158 -0
- package/package.json +1 -1
- package/supabase/functions/auth-vault/index.ts +50 -49
|
@@ -32,6 +32,15 @@
|
|
|
32
32
|
.supervisor-pill.ok{border-color:rgba(74,222,128,.35);color:var(--green-light);background:rgba(34,197,94,.08)}
|
|
33
33
|
.supervisor-pill.warn{border-color:rgba(250,204,21,.35);color:var(--gold-light);background:rgba(250,204,21,.08)}
|
|
34
34
|
.supervisor-pill.bad{border-color:rgba(248,113,113,.35);color:var(--pink);background:rgba(248,113,113,.08)}
|
|
35
|
+
/* NEEDS HELP -- a surface that exhausted MAX_CONSECUTIVE_RECONCILE_FAILURES
|
|
36
|
+
and clauth has stopped auto-restarting. Distinct from .bad ("down right
|
|
37
|
+
now, clauth is still retrying"): this is "clauth gave up, a human/agent
|
|
38
|
+
has to look." Pulses so it reads as an alarm among a fleet that regularly
|
|
39
|
+
has a few surfaces transiently down in .bad. */
|
|
40
|
+
.supervisor-pill.critical{border-color:rgba(248,113,113,.6);color:#fff;background:rgba(220,38,38,.55);animation:supervisor-pulse 1.4s ease-in-out infinite}
|
|
41
|
+
.supervisor-row.needs-help{border-color:rgba(220,38,38,.6);box-shadow:0 0 0 1px rgba(220,38,38,.35) inset}
|
|
42
|
+
.supervisor-meta.critical{color:var(--pink);white-space:normal}
|
|
43
|
+
@keyframes supervisor-pulse{0%,100%{opacity:1}50%{opacity:.55}}
|
|
35
44
|
/* Card GRID, not a stack. flex-direction:column gave one full-width row per
|
|
36
45
|
surface, so ten surfaces meant ten rows and constant scrolling to see the
|
|
37
46
|
fleet. auto-fill + minmax packs as many small cards per row as the panel
|
|
@@ -305,7 +305,15 @@ class ClauthSurfacesPanel extends window.ClauthPanelElement {
|
|
|
305
305
|
renderSupervisorSurface(surface) {
|
|
306
306
|
const compositeId = surface.plugin_id + ":" + surface.id;
|
|
307
307
|
const ownerKind = surface.lifecycle_owner === "clauth" ? "ok" : (surface.lifecycle_owner === "plugin" ? "warn" : "");
|
|
308
|
-
|
|
308
|
+
// "needs_help" is a DIFFERENT alarm than "bad": bad means "down right now,
|
|
309
|
+
// clauth is still retrying"; needs_help means "clauth gave up after
|
|
310
|
+
// MAX_CONSECUTIVE_RECONCILE_FAILURES — a human/agent has to look." It gets
|
|
311
|
+
// its own loud, pulsing pill (see .supervisor-pill.critical) rather than
|
|
312
|
+
// reusing "bad", which would render as just another red flicker in a fleet
|
|
313
|
+
// that regularly has a few surfaces transiently down.
|
|
314
|
+
const needsHelp = surface.state === "needs_help";
|
|
315
|
+
const stateKind = needsHelp ? "critical" : (surface.state === "current" || surface.status === "healthy" ? "ok" : (surface.state === "unavailable" ? "bad" : "warn"));
|
|
316
|
+
const stateLabel = needsHelp ? "NEEDS HELP" : (surface.state || surface.status || "unknown");
|
|
309
317
|
const openUrl = surface.__pseudo ? null : openUiUrlForSurface(surface);
|
|
310
318
|
const isSelected = this.selectedSupervisorSurface && this.selectedSupervisorSurface.id === compositeId;
|
|
311
319
|
const rowClass = "supervisor-row" + (surface.__pseudo ? " readonly" : "") + (isSelected ? " selected" : "");
|
|
@@ -339,14 +347,22 @@ class ClauthSurfacesPanel extends window.ClauthPanelElement {
|
|
|
339
347
|
: (surface.last_health_ok === false && surface.last_health_error
|
|
340
348
|
? surface.last_health_error + " · " + (surface.health || "no health")
|
|
341
349
|
: (surface.health || surface.health_url || "no health"));
|
|
342
|
-
|
|
350
|
+
// The loud line: WHY it needs help and WHAT to do, spelled out on the
|
|
351
|
+
// card itself — not just a pill label a reader has to already know the
|
|
352
|
+
// meaning of. Mirrors the same wording reconcileSurfaceHealth() writes
|
|
353
|
+
// into the daemon log, so the card and the log never disagree.
|
|
354
|
+
const needsHelpLine = needsHelp
|
|
355
|
+
? "Stopped auto-restarting after " + (surface.consecutive_reconcile_failures || "several") + " failed attempts — check the log, fix the cause, then Start/Restart to resume auto-repair."
|
|
356
|
+
: null;
|
|
357
|
+
return '<div class="' + rowClass + (needsHelp ? " needs-help" : "") + '" data-surface-id="' + htmlEscape(compositeId) + '"' + rowAction + '>' +
|
|
343
358
|
'<div class="supervisor-row-top"><span class="supervisor-name">' + htmlEscape(surface.name || compositeId) + '</span>' +
|
|
344
359
|
healthPill +
|
|
345
|
-
supervisorBadge(surface.lifecycle_owner || "unknown", ownerKind) + supervisorBadge(
|
|
360
|
+
supervisorBadge(surface.lifecycle_owner || "unknown", ownerKind) + supervisorBadge(stateLabel, stateKind) +
|
|
346
361
|
openBtn +
|
|
347
362
|
'</div>' +
|
|
348
363
|
'<div class="supervisor-meta">' + htmlEscape(surface.destination || "—") + ' · port ' + htmlEscape(surface.port || "—") + '</div>' +
|
|
349
364
|
'<div class="supervisor-meta">' + htmlEscape(healthLine) + '</div>' +
|
|
365
|
+
(needsHelpLine ? '<div class="supervisor-meta critical">' + htmlEscape(needsHelpLine) + '</div>' : '') +
|
|
350
366
|
'</div>';
|
|
351
367
|
}
|
|
352
368
|
|
package/cli/fingerprint.js
CHANGED
|
@@ -15,7 +15,33 @@ import path from "path";
|
|
|
15
15
|
// Cache path — avoids re-querying WMI/CIM on every daemon start.
|
|
16
16
|
// This eliminates the spawnSync cmd.exe ETIMEDOUT crash that occurs
|
|
17
17
|
// when PowerShell/WMI is slow on first call after boot.
|
|
18
|
-
|
|
18
|
+
//
|
|
19
|
+
// Durable as of 2026-09-06 (was os.tmpdir(), see LEGACY_CACHE_FILE below).
|
|
20
|
+
// getMachineId() derives (primary, secondary) from TWO independent execSync
|
|
21
|
+
// calls, each wrapped in its own try/catch that falls through silently on
|
|
22
|
+
// failure (see the comments at those calls). If either call transiently
|
|
23
|
+
// timed out, the resulting pairing was composed differently than a
|
|
24
|
+
// successful run, producing a DIFFERENT sha256 machine hash for the exact
|
|
25
|
+
// same physical machine — surfacing server-side as a false
|
|
26
|
+
// machine_not_found lockout, not a security event. A cache in system temp
|
|
27
|
+
// does not survive Windows Disk Cleanup or some machines' reboot behavior,
|
|
28
|
+
// which re-exposed the flaky first-computation path on any later restart
|
|
29
|
+
// where WMI happened to be slow. Matches the durable pattern boot.key
|
|
30
|
+
// already uses (getBootKeyPath() in cli/commands/serve.js) — same
|
|
31
|
+
// AppData/Roaming/clauth (win32) / ~/.config/clauth (else) directory.
|
|
32
|
+
function getCacheDir() {
|
|
33
|
+
if (os.platform() === "win32") {
|
|
34
|
+
return path.join(process.env.APPDATA || path.join(os.homedir(), "AppData", "Roaming"), "clauth");
|
|
35
|
+
}
|
|
36
|
+
return path.join(os.homedir(), ".config", "clauth");
|
|
37
|
+
}
|
|
38
|
+
const CACHE_FILE = path.join(getCacheDir(), "machine.cache");
|
|
39
|
+
// Pre-2026-09-06 location. An already-registered machine may still have a
|
|
40
|
+
// valid cache here; migrate it into the durable location on first read
|
|
41
|
+
// instead of recomputing via WMI — recomputing is exactly the flaky path
|
|
42
|
+
// this fix exists to avoid, and the durable write means this only ever
|
|
43
|
+
// happens once per machine.
|
|
44
|
+
const LEGACY_CACHE_FILE = path.join(os.tmpdir(), "clauth-machine.cache");
|
|
19
45
|
|
|
20
46
|
function readCache() {
|
|
21
47
|
try {
|
|
@@ -23,12 +49,24 @@ function readCache() {
|
|
|
23
49
|
// Validate: must be two non-empty lines (primary:secondary)
|
|
24
50
|
const [primary, secondary] = raw.split("\n");
|
|
25
51
|
if (primary && secondary) return { primary: primary.trim(), secondary: secondary.trim() };
|
|
26
|
-
} catch { /* cache miss */ }
|
|
52
|
+
} catch { /* durable cache miss */ }
|
|
53
|
+
try {
|
|
54
|
+
const raw = fs.readFileSync(LEGACY_CACHE_FILE, "utf8").trim();
|
|
55
|
+
const [primary, secondary] = raw.split("\n");
|
|
56
|
+
if (primary && secondary) {
|
|
57
|
+
const migrated = { primary: primary.trim(), secondary: secondary.trim() };
|
|
58
|
+
writeCache(migrated.primary, migrated.secondary);
|
|
59
|
+
return migrated;
|
|
60
|
+
}
|
|
61
|
+
} catch { /* no legacy cache either */ }
|
|
27
62
|
return null;
|
|
28
63
|
}
|
|
29
64
|
|
|
30
65
|
function writeCache(primary, secondary) {
|
|
31
|
-
try {
|
|
66
|
+
try {
|
|
67
|
+
fs.mkdirSync(getCacheDir(), { recursive: true });
|
|
68
|
+
fs.writeFileSync(CACHE_FILE, `${primary}\n${secondary}`, "utf8");
|
|
69
|
+
} catch { /* best effort */ }
|
|
32
70
|
}
|
|
33
71
|
|
|
34
72
|
function getMachineId() {
|
|
@@ -140,4 +178,8 @@ export function deriveSeedHash(machineHash, password) {
|
|
|
140
178
|
.digest("hex");
|
|
141
179
|
}
|
|
142
180
|
|
|
181
|
+
// Exported for test path introspection only — not part of the public
|
|
182
|
+
// CLI/API surface.
|
|
183
|
+
export { CACHE_FILE, LEGACY_CACHE_FILE, getCacheDir };
|
|
184
|
+
|
|
143
185
|
export default { getMachineHash, deriveToken, deriveSeedHash };
|
package/cli/fingerprint.test.js
CHANGED
|
@@ -4,7 +4,12 @@ import fs from "node:fs";
|
|
|
4
4
|
import os from "node:os";
|
|
5
5
|
import path from "node:path";
|
|
6
6
|
|
|
7
|
-
import { getMachineHash, deriveToken, deriveSeedHash } from "./fingerprint.js";
|
|
7
|
+
import { getMachineHash, deriveToken, deriveSeedHash, CACHE_FILE, LEGACY_CACHE_FILE } from "./fingerprint.js";
|
|
8
|
+
|
|
9
|
+
// CACHE_FILE/LEGACY_CACHE_FILE are the REAL production paths (durable
|
|
10
|
+
// AppData/Roaming/clauth on win32, formerly os.tmpdir()) — read-only in this
|
|
11
|
+
// suite. Every test below only ever reads mtimeMs, never asserts on or
|
|
12
|
+
// writes machine-identifying content into either file.
|
|
8
13
|
|
|
9
14
|
// getMachineId() normally shells out to WMI/registry/ioreg, which this harness
|
|
10
15
|
// must never do (slow, platform-dependent, pollutes nothing but still real
|
|
@@ -84,12 +89,31 @@ test("deriveSeedHash changes with the password", () => {
|
|
|
84
89
|
assert.notEqual(deriveSeedHash(hash, "one"), deriveSeedHash(hash, "two"));
|
|
85
90
|
});
|
|
86
91
|
|
|
87
|
-
test("machine id cache: a fresh CLAUTH_MACHINE_ID call never touches the WMI cache file", () => {
|
|
92
|
+
test("machine id cache: a fresh CLAUTH_MACHINE_ID call never touches the WMI cache file (durable or legacy)", () => {
|
|
88
93
|
// Positive control that the cache mechanism exists and this test can see it,
|
|
89
94
|
// so "the cache file was untouched" below actually means something.
|
|
90
|
-
const
|
|
91
|
-
|
|
95
|
+
const before = {
|
|
96
|
+
durable: fs.existsSync(CACHE_FILE) ? fs.statSync(CACHE_FILE).mtimeMs : null,
|
|
97
|
+
legacy: fs.existsSync(LEGACY_CACHE_FILE) ? fs.statSync(LEGACY_CACHE_FILE).mtimeMs : null,
|
|
98
|
+
};
|
|
92
99
|
withMachineId("container-fast-path", () => getMachineHash());
|
|
93
|
-
const after =
|
|
94
|
-
|
|
100
|
+
const after = {
|
|
101
|
+
durable: fs.existsSync(CACHE_FILE) ? fs.statSync(CACHE_FILE).mtimeMs : null,
|
|
102
|
+
legacy: fs.existsSync(LEGACY_CACHE_FILE) ? fs.statSync(LEGACY_CACHE_FILE).mtimeMs : null,
|
|
103
|
+
};
|
|
104
|
+
assert.deepEqual(before, after, "the container-id fast path must not read or write either cache file");
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
test("cache paths are durable (AppData/Roaming/clauth or ~/.config/clauth), never os.tmpdir()", () => {
|
|
108
|
+
// The whole point of this fix: CACHE_FILE must NOT live in system temp,
|
|
109
|
+
// which Windows Disk Cleanup (or equivalent) can wipe, re-exposing the
|
|
110
|
+
// flaky first-computation WMI path on the next restart.
|
|
111
|
+
assert.ok(!CACHE_FILE.startsWith(os.tmpdir()), `CACHE_FILE must be durable, got: ${CACHE_FILE}`);
|
|
112
|
+
assert.equal(path.basename(CACHE_FILE), "machine.cache");
|
|
113
|
+
const dir = path.dirname(CACHE_FILE);
|
|
114
|
+
assert.ok(dir.endsWith(path.join("clauth")), `cache dir must be a clauth-owned config directory, got: ${dir}`);
|
|
115
|
+
// LEGACY_CACHE_FILE is the pre-fix location, kept only so readCache() can
|
|
116
|
+
// migrate an already-registered machine's existing cache instead of
|
|
117
|
+
// recomputing via WMI (see fingerprint.js).
|
|
118
|
+
assert.equal(LEGACY_CACHE_FILE, path.join(os.tmpdir(), "clauth-machine.cache"));
|
|
95
119
|
});
|
|
@@ -18,6 +18,22 @@ const ACTIONS = new Set(["start", "stop", "restart", "reconcile", "test", "promo
|
|
|
18
18
|
const DEFAULT_HEALTH_RECONCILE_INTERVAL_MS = 10000;
|
|
19
19
|
const DEFAULT_HEALTH_TIMEOUT_MS = 2500;
|
|
20
20
|
const HEALTH_RECONCILE_COOLDOWN_MS = 15000;
|
|
21
|
+
// After this many CONSECUTIVE failed reconcile attempts (each gated by
|
|
22
|
+
// HEALTH_RECONCILE_COOLDOWN_MS, so >= 5 * 15s = 75s before this fires),
|
|
23
|
+
// reconcileSurfaceHealth() stops auto-restarting the surface and latches it
|
|
24
|
+
// into "needs_help" instead of retrying forever. Without a cap, a surface
|
|
25
|
+
// that can never come up (a missing dependency, a bad config) gets
|
|
26
|
+
// relaunched every cooldown window FOREVER: the restart command itself can
|
|
27
|
+
// exit 0 (pm2 accepted it) while the process crashes immediately after, so
|
|
28
|
+
// this loop never sees a hard failure to stop on, and pm2's OWN crash-loop
|
|
29
|
+
// backoff never engages either because clauth's periodic explicit restart
|
|
30
|
+
// keeps resetting it. Root cause of the regen-media-local window-flashing
|
|
31
|
+
// incident (2026-09-05): a missing `tsx` dependency meant every restart
|
|
32
|
+
// "succeeded" and immediately crashed, forever, because nothing was
|
|
33
|
+
// counting. A latched surface self-heals the moment its health probe
|
|
34
|
+
// succeeds (handled before this cap is ever consulted) or a human/agent
|
|
35
|
+
// issues an explicit non-reconcile start/stop/restart (see runSurfaceAction).
|
|
36
|
+
export const MAX_CONSECUTIVE_RECONCILE_FAILURES = 5;
|
|
21
37
|
const DOCUMENTATION_FIELDS = ["architecture", "operator_guide", "install", "runbook", "tool_reference", "release", "agent_context"];
|
|
22
38
|
|
|
23
39
|
function hasMcpToken(value) {
|
|
@@ -182,11 +198,25 @@ export async function reconcileSurfaceHealth({ fetchImpl = globalThis.fetch, tim
|
|
|
182
198
|
const error = health.error;
|
|
183
199
|
|
|
184
200
|
if (healthy) {
|
|
185
|
-
updateSurfaceState(id, { state: "current", last_health_at: observedAt, last_health_ok: true, last_health_error: null });
|
|
201
|
+
updateSurfaceState(id, { state: "current", last_health_at: observedAt, last_health_ok: true, last_health_error: null, consecutive_reconcile_failures: 0 });
|
|
186
202
|
inspected.push({ surface_id: id, state: "healthy", observed_at: observedAt });
|
|
187
203
|
continue;
|
|
188
204
|
}
|
|
189
205
|
|
|
206
|
+
// Latched: this surface already exhausted MAX_CONSECUTIVE_RECONCILE_FAILURES
|
|
207
|
+
// and clauth stopped auto-restarting it. Record the failed probe (so the
|
|
208
|
+
// UI's "last checked" freshness stays honest) WITHOUT calling
|
|
209
|
+
// runSurfaceAction again — that call is exactly the crash loop this latch
|
|
210
|
+
// exists to stop. The only ways out are a health probe that succeeds
|
|
211
|
+
// (the branch above, checked every tick regardless of latch state) or an
|
|
212
|
+
// explicit human/agent action (runSurfaceAction resets the counter and
|
|
213
|
+
// clears this latch on any successful non-reconcile command).
|
|
214
|
+
if (surface.state === "needs_help") {
|
|
215
|
+
updateSurfaceState(id, { last_health_at: observedAt, last_health_ok: false, last_health_error: error });
|
|
216
|
+
inspected.push({ surface_id: id, state: "needs_help", error, latched: true, observed_at: observedAt });
|
|
217
|
+
continue;
|
|
218
|
+
}
|
|
219
|
+
|
|
190
220
|
const lastAttempt = Date.parse(surface.last_reconcile_at || "") || 0;
|
|
191
221
|
if (Date.now() - lastAttempt < HEALTH_RECONCILE_COOLDOWN_MS) {
|
|
192
222
|
updateSurfaceState(id, { state: "unavailable", last_health_at: observedAt, last_health_ok: false, last_health_error: error });
|
|
@@ -206,15 +236,52 @@ export async function reconcileSurfaceHealth({ fetchImpl = globalThis.fetch, tim
|
|
|
206
236
|
const commandCompleted = receipt?.resulting_state?.ok === true;
|
|
207
237
|
const postHealth = commandCompleted ? await probeSurfaceHealth(url, fetchImpl, timeoutMs) : { healthy: false, error: receipt?.resulting_state?.state || "reconcile_failed" };
|
|
208
238
|
const repaired = commandCompleted && postHealth.healthy;
|
|
239
|
+
|
|
240
|
+
if (repaired) {
|
|
241
|
+
updateSurfaceState(id, {
|
|
242
|
+
state: "current",
|
|
243
|
+
last_health_at: now(),
|
|
244
|
+
last_health_ok: true,
|
|
245
|
+
last_health_error: null,
|
|
246
|
+
last_reconcile_operation_id: receipt?.operationId || null,
|
|
247
|
+
consecutive_reconcile_failures: 0,
|
|
248
|
+
});
|
|
249
|
+
appendSupervisorEvent({ kind: "surface_reconciled", surface_id: id, operation_id: receipt?.operationId || null, command_completed: commandCompleted, health_ok: true, error: null });
|
|
250
|
+
inspected.push({ surface_id: id, state: "reconciled", error: null, operation_id: receipt?.operationId || null, observed_at: observedAt });
|
|
251
|
+
continue;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// Not repaired: count it. MAX_CONSECUTIVE_RECONCILE_FAILURES is the
|
|
255
|
+
// circuit breaker — cross it and the surface latches to "needs_help"
|
|
256
|
+
// instead of feeding another attempt back into runSurfaceAction next tick.
|
|
257
|
+
const failures = (surface.consecutive_reconcile_failures || 0) + 1;
|
|
258
|
+
const exhausted = failures >= MAX_CONSECUTIVE_RECONCILE_FAILURES;
|
|
209
259
|
updateSurfaceState(id, {
|
|
210
|
-
state:
|
|
260
|
+
state: exhausted ? "needs_help" : "unavailable",
|
|
211
261
|
last_health_at: now(),
|
|
212
|
-
last_health_ok:
|
|
213
|
-
last_health_error:
|
|
262
|
+
last_health_ok: false,
|
|
263
|
+
last_health_error: postHealth.error || error,
|
|
214
264
|
last_reconcile_operation_id: receipt?.operationId || null,
|
|
265
|
+
consecutive_reconcile_failures: failures,
|
|
266
|
+
});
|
|
267
|
+
appendSupervisorEvent({
|
|
268
|
+
kind: exhausted ? "surface_needs_help" : "surface_reconcile_failed",
|
|
269
|
+
surface_id: id,
|
|
270
|
+
operation_id: receipt?.operationId || null,
|
|
271
|
+
command_completed: commandCompleted,
|
|
272
|
+
health_ok: false,
|
|
273
|
+
error: postHealth.error || error,
|
|
274
|
+
consecutive_failures: failures,
|
|
275
|
+
...(exhausted ? { message: `${id} failed to come up after ${failures} consecutive restart attempts — clauth has stopped auto-restarting it. Check its log, fix the underlying cause, then Start/Restart it manually (dashboard or \`clauth mcp restart ${id}\`) to resume auto-repair.` } : {}),
|
|
276
|
+
});
|
|
277
|
+
inspected.push({
|
|
278
|
+
surface_id: id,
|
|
279
|
+
state: exhausted ? "needs_help" : "reconcile_failed",
|
|
280
|
+
error: postHealth.error || error,
|
|
281
|
+
operation_id: receipt?.operationId || null,
|
|
282
|
+
consecutive_failures: failures,
|
|
283
|
+
observed_at: observedAt,
|
|
215
284
|
});
|
|
216
|
-
appendSupervisorEvent({ kind: repaired ? "surface_reconciled" : "surface_reconcile_failed", surface_id: id, operation_id: receipt?.operationId || null, command_completed: commandCompleted, health_ok: postHealth.healthy, error: postHealth.error || null });
|
|
217
|
-
inspected.push({ surface_id: id, state: repaired ? "reconciled" : "reconcile_failed", error: postHealth.error || error, operation_id: receipt?.operationId || null, observed_at: observedAt });
|
|
218
285
|
}
|
|
219
286
|
return { inspected };
|
|
220
287
|
}
|
|
@@ -1245,6 +1312,22 @@ export function runSurfaceAction(id, action, actor = "localhost") {
|
|
|
1245
1312
|
fallbackUsed = true;
|
|
1246
1313
|
result.stderr = `restart exited ${restartStatus}; start fallback attempted\n${result.stderr || ""}`;
|
|
1247
1314
|
}
|
|
1315
|
+
// A deliberate, successful manual action — start/stop/restart, never
|
|
1316
|
+
// "reconcile" itself — is a human or agent saying "I looked at this."
|
|
1317
|
+
// Give the surface a fresh crash-loop budget instead of carrying forward a
|
|
1318
|
+
// counter (or a "needs_help" latch) from before the intervention. Only
|
|
1319
|
+
// reconcileSurfaceHealth's own automatic attempts count against
|
|
1320
|
+
// MAX_CONSECUTIVE_RECONCILE_FAILURES, so this is the other side of that
|
|
1321
|
+
// cap: the way OUT of "needs_help", not just the way IN. Reset does not
|
|
1322
|
+
// claim the surface is healthy again — it has done no health probe — it
|
|
1323
|
+
// only clears the latch so the next reconcile tick (or health probe) gets
|
|
1324
|
+
// to judge it fresh.
|
|
1325
|
+
if (action !== "reconcile" && result.status === 0) {
|
|
1326
|
+
updateSurfaceState(id, {
|
|
1327
|
+
consecutive_reconcile_failures: 0,
|
|
1328
|
+
state: surface.state === "needs_help" ? "unavailable" : surface.state,
|
|
1329
|
+
});
|
|
1330
|
+
}
|
|
1248
1331
|
return operation(action, { surface_id: id }, surface, {
|
|
1249
1332
|
ok: result.status === 0,
|
|
1250
1333
|
state: result.status === 0 ? "operation_completed" : "operation_failed",
|
|
@@ -12,8 +12,10 @@ import {
|
|
|
12
12
|
isMcpServerPlugin,
|
|
13
13
|
listPlugins,
|
|
14
14
|
listSurfaces,
|
|
15
|
+
MAX_CONSECUTIVE_RECONCILE_FAILURES,
|
|
15
16
|
probeAllSurfaceHealth,
|
|
16
17
|
surfaceOpenUrl,
|
|
18
|
+
readSupervisorEvents,
|
|
17
19
|
reconcileSurfaceHealth,
|
|
18
20
|
registerPlugin,
|
|
19
21
|
runPluginAction,
|
|
@@ -60,6 +62,33 @@ async function withTempSupervisor(fn) {
|
|
|
60
62
|
}
|
|
61
63
|
}
|
|
62
64
|
|
|
65
|
+
// Fast-forwards Date.now()/`new Date()` without real sleeping.
|
|
66
|
+
// HEALTH_RECONCILE_COOLDOWN_MS is a real 15s per attempt, and proving the
|
|
67
|
+
// crash-loop cap needs MAX_CONSECUTIVE_RECONCILE_FAILURES of them -- a test
|
|
68
|
+
// cannot afford to actually sleep that long. Overriding ONLY Date.now() is
|
|
69
|
+
// not enough: now() (supervisor-registry.js) stamps last_reconcile_at via
|
|
70
|
+
// `new Date().toISOString()`, so the cooldown gate (Date.now() minus
|
|
71
|
+
// Date.parse(last_reconcile_at)) would compare a mocked "now" against a REAL
|
|
72
|
+
// timestamp and never agree. Subclassing keeps every other Date behavior
|
|
73
|
+
// (Date.parse, `new Date(iso)`) real.
|
|
74
|
+
async function withMockedClock(fn) {
|
|
75
|
+
const RealDate = global.Date;
|
|
76
|
+
let clock = RealDate.now();
|
|
77
|
+
class MockDate extends RealDate {
|
|
78
|
+
constructor(...args) {
|
|
79
|
+
if (args.length === 0) super(clock);
|
|
80
|
+
else super(...args);
|
|
81
|
+
}
|
|
82
|
+
static now() { return clock; }
|
|
83
|
+
}
|
|
84
|
+
global.Date = MockDate;
|
|
85
|
+
try {
|
|
86
|
+
return await fn({ advance: (ms) => { clock += ms; } });
|
|
87
|
+
} finally {
|
|
88
|
+
global.Date = RealDate;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
63
92
|
function writePlugin(root, source, id, manifest) {
|
|
64
93
|
const dir = path.join(root, source, id);
|
|
65
94
|
fs.mkdirSync(dir, { recursive: true });
|
|
@@ -453,6 +482,135 @@ test("health reconciliation does not claim current when a successful command lea
|
|
|
453
482
|
}
|
|
454
483
|
});
|
|
455
484
|
|
|
485
|
+
test("a crash-looping surface stops auto-restarting after MAX_CONSECUTIVE_RECONCILE_FAILURES and needs a human", async () => {
|
|
486
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "clauth-supervisor-crashloop-"));
|
|
487
|
+
const oldDir = process.env.CLAUTH_SUPERVISOR_DIR;
|
|
488
|
+
const oldManaged = process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
|
|
489
|
+
const oldUser = process.env.CLAUTH_USER_PLUGIN_ROOTS;
|
|
490
|
+
process.env.CLAUTH_SUPERVISOR_DIR = root;
|
|
491
|
+
process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = path.join(root, "managed");
|
|
492
|
+
process.env.CLAUTH_USER_PLUGIN_ROOTS = path.join(root, "user");
|
|
493
|
+
try {
|
|
494
|
+
// restart "succeeds" (exit 0 -- pm2 would accept it) but health NEVER
|
|
495
|
+
// recovers: exactly the regen-media-local shape -- a process that
|
|
496
|
+
// restarts cleanly and crashes immediately after, forever.
|
|
497
|
+
writePlugin(root, "managed", "crashloop-demo", baseManifest("crashloop-demo", {
|
|
498
|
+
core: true,
|
|
499
|
+
enable_default: true,
|
|
500
|
+
surfaces: [{
|
|
501
|
+
id: "primary",
|
|
502
|
+
destination: "local/clauth/pm2",
|
|
503
|
+
lifecycle_owner: "clauth",
|
|
504
|
+
port: 39121,
|
|
505
|
+
health: "/health",
|
|
506
|
+
restart: [process.execPath, "--version"],
|
|
507
|
+
start: [process.execPath, "--version"],
|
|
508
|
+
}],
|
|
509
|
+
}));
|
|
510
|
+
discoverPlugins();
|
|
511
|
+
const alwaysDown = { fetchImpl: async () => ({ ok: false, status: 503 }) };
|
|
512
|
+
|
|
513
|
+
await withMockedClock(async ({ advance }) => {
|
|
514
|
+
let last;
|
|
515
|
+
for (let i = 0; i < MAX_CONSECUTIVE_RECONCILE_FAILURES; i++) {
|
|
516
|
+
advance(20000); // past the 15s reconcile cooldown every tick
|
|
517
|
+
last = await reconcileSurfaceHealth(alwaysDown);
|
|
518
|
+
}
|
|
519
|
+
const surface = listSurfaces().find((item) => item.plugin_id === "crashloop-demo");
|
|
520
|
+
assert.equal(surface.state, "needs_help", "must latch once the cap is reached");
|
|
521
|
+
assert.equal(surface.consecutive_reconcile_failures, MAX_CONSECUTIVE_RECONCILE_FAILURES);
|
|
522
|
+
assert.equal(last.inspected[0].state, "needs_help");
|
|
523
|
+
|
|
524
|
+
// THE load-bearing assertion: one more tick past cooldown must NOT
|
|
525
|
+
// spawn another restart. Count "reconcile" operation receipts before
|
|
526
|
+
// and after -- if the count moves, the loop never actually stopped.
|
|
527
|
+
const before = readSupervisorEvents(1000).filter((e) => e.kind === "operation" && e.action === "reconcile").length;
|
|
528
|
+
advance(20000);
|
|
529
|
+
const again = await reconcileSurfaceHealth(alwaysDown);
|
|
530
|
+
const after = readSupervisorEvents(1000).filter((e) => e.kind === "operation" && e.action === "reconcile").length;
|
|
531
|
+
assert.equal(after, before, "a latched surface must not be re-restarted -- that is the crash loop this cap exists to stop");
|
|
532
|
+
assert.equal(again.inspected[0].state, "needs_help");
|
|
533
|
+
assert.equal(again.inspected[0].latched, true);
|
|
534
|
+
|
|
535
|
+
// Self-heal: health recovering on its own (independent of clauth's
|
|
536
|
+
// restart) clears the latch -- the cap is a circuit breaker, not a
|
|
537
|
+
// permanent ban.
|
|
538
|
+
advance(20000);
|
|
539
|
+
const healed = await reconcileSurfaceHealth({ fetchImpl: async () => ({ ok: true, status: 200 }) });
|
|
540
|
+
assert.equal(healed.inspected[0].state, "healthy");
|
|
541
|
+
const healedSurface = listSurfaces().find((item) => item.plugin_id === "crashloop-demo");
|
|
542
|
+
assert.equal(healedSurface.state, "current");
|
|
543
|
+
assert.equal(healedSurface.consecutive_reconcile_failures, 0);
|
|
544
|
+
});
|
|
545
|
+
} finally {
|
|
546
|
+
if (oldDir === undefined) delete process.env.CLAUTH_SUPERVISOR_DIR;
|
|
547
|
+
else process.env.CLAUTH_SUPERVISOR_DIR = oldDir;
|
|
548
|
+
if (oldManaged === undefined) delete process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
|
|
549
|
+
else process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = oldManaged;
|
|
550
|
+
if (oldUser === undefined) delete process.env.CLAUTH_USER_PLUGIN_ROOTS;
|
|
551
|
+
else process.env.CLAUTH_USER_PLUGIN_ROOTS = oldUser;
|
|
552
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
553
|
+
}
|
|
554
|
+
});
|
|
555
|
+
|
|
556
|
+
test("an explicit manual restart clears a needs_help latch and gives a fresh crash-loop budget", async () => {
|
|
557
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "clauth-supervisor-manual-reset-"));
|
|
558
|
+
const oldDir = process.env.CLAUTH_SUPERVISOR_DIR;
|
|
559
|
+
const oldManaged = process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
|
|
560
|
+
const oldUser = process.env.CLAUTH_USER_PLUGIN_ROOTS;
|
|
561
|
+
process.env.CLAUTH_SUPERVISOR_DIR = root;
|
|
562
|
+
process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = path.join(root, "managed");
|
|
563
|
+
process.env.CLAUTH_USER_PLUGIN_ROOTS = path.join(root, "user");
|
|
564
|
+
try {
|
|
565
|
+
writePlugin(root, "managed", "manual-reset-demo", baseManifest("manual-reset-demo", {
|
|
566
|
+
core: true,
|
|
567
|
+
enable_default: true,
|
|
568
|
+
surfaces: [{
|
|
569
|
+
id: "primary",
|
|
570
|
+
destination: "local/clauth/pm2",
|
|
571
|
+
lifecycle_owner: "clauth",
|
|
572
|
+
port: 39122,
|
|
573
|
+
health: "/health",
|
|
574
|
+
restart: [process.execPath, "--version"],
|
|
575
|
+
start: [process.execPath, "--version"],
|
|
576
|
+
}],
|
|
577
|
+
}));
|
|
578
|
+
discoverPlugins();
|
|
579
|
+
const alwaysDown = { fetchImpl: async () => ({ ok: false, status: 503 }) };
|
|
580
|
+
|
|
581
|
+
await withMockedClock(async ({ advance }) => {
|
|
582
|
+
for (let i = 0; i < MAX_CONSECUTIVE_RECONCILE_FAILURES; i++) {
|
|
583
|
+
advance(20000);
|
|
584
|
+
await reconcileSurfaceHealth(alwaysDown);
|
|
585
|
+
}
|
|
586
|
+
assert.equal(listSurfaces()[0].state, "needs_help");
|
|
587
|
+
|
|
588
|
+
// A human/agent looks at it and issues an explicit restart -- through
|
|
589
|
+
// runSurfaceAction directly, never through the "reconcile" action the
|
|
590
|
+
// background loop uses.
|
|
591
|
+
const receipt = runSurfaceAction("manual-reset-demo:primary", "restart", "human");
|
|
592
|
+
assert.equal(receipt.resulting_state.ok, true);
|
|
593
|
+
const reset = listSurfaces()[0];
|
|
594
|
+
assert.equal(reset.state, "unavailable", "cleared out of needs_help, but not falsely claimed healthy without a probe");
|
|
595
|
+
assert.equal(reset.consecutive_reconcile_failures, 0);
|
|
596
|
+
|
|
597
|
+
// And the surface gets its full budget back, not zero.
|
|
598
|
+
advance(20000);
|
|
599
|
+
const tick = await reconcileSurfaceHealth(alwaysDown);
|
|
600
|
+
assert.equal(tick.inspected[0].state, "reconcile_failed");
|
|
601
|
+
assert.equal(listSurfaces()[0].consecutive_reconcile_failures, 1);
|
|
602
|
+
});
|
|
603
|
+
} finally {
|
|
604
|
+
if (oldDir === undefined) delete process.env.CLAUTH_SUPERVISOR_DIR;
|
|
605
|
+
else process.env.CLAUTH_SUPERVISOR_DIR = oldDir;
|
|
606
|
+
if (oldManaged === undefined) delete process.env.CLAUTH_MANAGED_PLUGIN_ROOTS;
|
|
607
|
+
else process.env.CLAUTH_MANAGED_PLUGIN_ROOTS = oldManaged;
|
|
608
|
+
if (oldUser === undefined) delete process.env.CLAUTH_USER_PLUGIN_ROOTS;
|
|
609
|
+
else process.env.CLAUTH_USER_PLUGIN_ROOTS = oldUser;
|
|
610
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
611
|
+
}
|
|
612
|
+
});
|
|
613
|
+
|
|
456
614
|
test("health reconciliation never restarts external or plugin-owned surfaces", async () => {
|
|
457
615
|
const root = fs.mkdtempSync(path.join(os.tmpdir(), "clauth-supervisor-observe-"));
|
|
458
616
|
const oldDir = process.env.CLAUTH_SUPERVISOR_DIR;
|
package/package.json
CHANGED
|
@@ -1,5 +1,12 @@
|
|
|
1
|
-
// clauth — auth-vault Edge Function
|
|
2
|
-
//
|
|
1
|
+
// clauth — auth-vault Edge Function v3
|
|
2
|
+
// IP whitelist + machine lockout (fail_count + locked) remain. Rate limiting
|
|
3
|
+
// and audit logging (clauth_audit) removed 2026-09-03 -- see the comments at
|
|
4
|
+
// validateHMAC and where auditLog used to be defined for why: the rate-limit
|
|
5
|
+
// check was an unindexed COUNT against clauth_audit run on EVERY request,
|
|
6
|
+
// pre-auth, and every rejection it produced also wrote an audit row -- a
|
|
7
|
+
// feedback loop that took the whole project down under sustained load. The
|
|
8
|
+
// security value it provided was already covered, tighter, by the
|
|
9
|
+
// per-machine 5-failed-attempts lockout below.
|
|
3
10
|
|
|
4
11
|
import { createClient } from "https://esm.sh/@supabase/supabase-js@2";
|
|
5
12
|
|
|
@@ -11,8 +18,6 @@ const ADMIN_BOOTSTRAP_TOKEN = Deno.env.get("CLAUTH_ADMIN_BOOTSTRAP_TOKEN")!;
|
|
|
11
18
|
const ALLOWED_IPS: string[] = (Deno.env.get("CLAUTH_ALLOWED_IPS") || "")
|
|
12
19
|
.split(",").map(s => s.trim()).filter(Boolean);
|
|
13
20
|
|
|
14
|
-
const RATE_LIMIT_MAX = 30;
|
|
15
|
-
const RATE_LIMIT_WINDOW = 60;
|
|
16
21
|
const REPLAY_WINDOW_MS = 5 * 60 * 1000;
|
|
17
22
|
const MAX_FAIL_COUNT = 5;
|
|
18
23
|
const DEFAULT_INSTALL_ID = "default";
|
|
@@ -55,20 +60,36 @@ function checkIP(ip: string): { allowed: boolean; reason?: string } {
|
|
|
55
60
|
return { allowed: false, reason: `IP not whitelisted: ${ip}` };
|
|
56
61
|
}
|
|
57
62
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
63
|
+
// rate-limiting removed 2026-09-03 -- see supervisor-registry.js's own
|
|
64
|
+
// state.json lock history for the shape of this bug: checkRateLimit() ran
|
|
65
|
+
// this exact query -- a COUNT against clauth_audit filtered by machine_hash
|
|
66
|
+
// and created_at -- on EVERY request, BEFORE authentication (validateHMAC
|
|
67
|
+
// below), against a table with no index on either column and no retention.
|
|
68
|
+
// As the table grew (174,385 rows, unbounded), the scan cost grew with it;
|
|
69
|
+
// under sustained traffic the queries started stacking, Postgres killed them
|
|
70
|
+
// at statement_timeout, and Cloudflare killed the stacked connections at its
|
|
71
|
+
// own 90s ceiling (522s) -- and every one of those rejections ALSO wrote an
|
|
72
|
+
// audit row (see auditLog's removal below), so the failure fed itself. The
|
|
73
|
+
// actual security protection this duplicated is already provided, tighter,
|
|
74
|
+
// by the per-machine lockout in validateHMAC below (5 failed attempts locks
|
|
75
|
+
// the machine -- well under the 30-per-60s this used to allow). Rate
|
|
76
|
+
// limiting for a genuine runaway client belongs at Cloudflare, in front of
|
|
77
|
+
// this function, not as a synchronous Postgres query on the hot path of
|
|
78
|
+
// every request.
|
|
70
79
|
async function validateHMAC(sb: any, body: any): Promise<{ valid: boolean; reason?: string }> {
|
|
71
80
|
const now = Date.now();
|
|
81
|
+
|
|
82
|
+
// A caller that omits machine_hash entirely used to fall straight through
|
|
83
|
+
// to the clauth_machines lookup below with body.machine_hash === undefined.
|
|
84
|
+
// PostgREST's .single() then answers "0 rows" for a query that can never
|
|
85
|
+
// match, which comes back as an HTTP 406 -- 1,438 of these in 24h
|
|
86
|
+
// (2026-09-06), every one a wasted round trip for a request already known
|
|
87
|
+
// to be malformed. Reject before the DB call: same auth_failed response
|
|
88
|
+
// the caller already gets, no query, no spurious 406 in the logs.
|
|
89
|
+
if (typeof body.machine_hash !== "string" || !body.machine_hash) {
|
|
90
|
+
return { valid: false, reason: "machine_hash_missing" };
|
|
91
|
+
}
|
|
92
|
+
|
|
72
93
|
if (Math.abs(now - body.timestamp) > REPLAY_WINDOW_MS) return { valid: false, reason: "timestamp_expired" };
|
|
73
94
|
|
|
74
95
|
const { data: machine, error } = await sb.from("clauth_machines")
|
|
@@ -99,20 +120,23 @@ async function validateHMAC(sb: any, body: any): Promise<{ valid: boolean; reaso
|
|
|
99
120
|
return { valid: true };
|
|
100
121
|
}
|
|
101
122
|
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
123
|
+
// auditLog() removed 2026-09-03 alongside checkRateLimit() -- it wrote a row
|
|
124
|
+
// to clauth_audit on every path through this function, including rejections
|
|
125
|
+
// (rate-limited, auth-denied, IP-blocked), which is what turned "the table
|
|
126
|
+
// got slow" into a feedback loop (more rejections -> more rows -> slower
|
|
127
|
+
// scans -> more rejections). No retention policy ever existed for this
|
|
128
|
+
// table. If durable audit logging is wanted back, it needs its own design --
|
|
129
|
+
// write-only, indexed for its actual read pattern (if any), with a real
|
|
130
|
+
// retention/partition policy -- not a bare insert-on-every-call with no cap.
|
|
106
131
|
async function handleRetrieve(sb: any, body: any, mh: string) {
|
|
107
132
|
const { service } = body;
|
|
108
133
|
if (!service) return { error: "service required" };
|
|
109
134
|
const { data: svc } = await sb.from("clauth_services").select("*").eq("name", service).single();
|
|
110
|
-
if (!svc)
|
|
111
|
-
if (!svc.enabled)
|
|
112
|
-
if (!svc.vault_key)
|
|
135
|
+
if (!svc) return { error: "service_not_found" };
|
|
136
|
+
if (!svc.enabled) return { error: "service_disabled" };
|
|
137
|
+
if (!svc.vault_key) return { error: "no_key_stored" };
|
|
113
138
|
const { data: secret } = await sb.rpc("vault_decrypt_secret", { secret_name: svc.vault_key });
|
|
114
139
|
await sb.from("clauth_services").update({ last_retrieved: new Date().toISOString() }).eq("name", service);
|
|
115
|
-
await auditLog(sb, mh, service, "retrieve", "success");
|
|
116
140
|
return { service, key_type: svc.key_type, value: secret };
|
|
117
141
|
}
|
|
118
142
|
|
|
@@ -121,9 +145,8 @@ async function handleWrite(sb: any, body: any, mh: string) {
|
|
|
121
145
|
if (!service || !value) return { error: "service and value required" };
|
|
122
146
|
const vaultKey = `clauth.${service}`;
|
|
123
147
|
const { error } = await sb.rpc("vault_upsert_secret", { secret_name: vaultKey, secret_value: typeof value === "string" ? value : JSON.stringify(value) });
|
|
124
|
-
if (error)
|
|
148
|
+
if (error) return { error: error.message };
|
|
125
149
|
await sb.from("clauth_services").update({ vault_key: vaultKey, last_rotated: new Date().toISOString() }).eq("name", service);
|
|
126
|
-
await auditLog(sb, mh, service, "write", "success");
|
|
127
150
|
return { success: true, service, vault_key: vaultKey };
|
|
128
151
|
}
|
|
129
152
|
|
|
@@ -133,7 +156,6 @@ async function handleEnable(sb: any, body: any, mh: string) {
|
|
|
133
156
|
q = service !== "all" ? q.eq("name", service) : q.not("vault_key", "is", null);
|
|
134
157
|
const { error } = await q;
|
|
135
158
|
if (error) return { error: error.message };
|
|
136
|
-
await auditLog(sb, mh, service, enabled ? "enable" : "disable", "success");
|
|
137
159
|
return { success: true, service, enabled };
|
|
138
160
|
}
|
|
139
161
|
|
|
@@ -144,7 +166,6 @@ async function handleAdd(sb: any, body: any, mh: string) {
|
|
|
144
166
|
if (project) row.project = project;
|
|
145
167
|
const { error } = await sb.from("clauth_services").insert(row);
|
|
146
168
|
if (error) return { error: error.message };
|
|
147
|
-
await auditLog(sb, mh, name, "add", "success");
|
|
148
169
|
return { success: true, name, label, key_type, project: project || null };
|
|
149
170
|
}
|
|
150
171
|
|
|
@@ -157,7 +178,6 @@ async function handleUpdate(sb: any, body: any, mh: string) {
|
|
|
157
178
|
if (description !== undefined) updates.description = description || null;
|
|
158
179
|
const { error } = await sb.from("clauth_services").update(updates).eq("name", service);
|
|
159
180
|
if (error) return { error: error.message };
|
|
160
|
-
await auditLog(sb, mh, service, "update", "success", `fields: ${Object.keys(updates).join(", ")}`);
|
|
161
181
|
return { success: true, service, ...updates };
|
|
162
182
|
}
|
|
163
183
|
|
|
@@ -166,7 +186,6 @@ async function handleRemove(sb: any, body: any, mh: string) {
|
|
|
166
186
|
if (confirm !== `CONFIRM REMOVE ${service.toUpperCase()}`) return { error: "confirm phrase mismatch" };
|
|
167
187
|
await sb.rpc("vault_delete_secret", { secret_name: `clauth.${service}` });
|
|
168
188
|
await sb.from("clauth_services").delete().eq("name", service);
|
|
169
|
-
await auditLog(sb, mh, service, "remove", "success");
|
|
170
189
|
return { success: true, service };
|
|
171
190
|
}
|
|
172
191
|
|
|
@@ -182,7 +201,6 @@ async function handleRevoke(sb: any, body: any, mh: string) {
|
|
|
182
201
|
await sb.rpc("vault_delete_secret", { secret_name: `clauth.${service}` });
|
|
183
202
|
await sb.from("clauth_services").update({ vault_key: null, enabled: false }).eq("name", service);
|
|
184
203
|
}
|
|
185
|
-
await auditLog(sb, mh, service, "revoke", "success");
|
|
186
204
|
return { success: true, service };
|
|
187
205
|
}
|
|
188
206
|
|
|
@@ -193,7 +211,6 @@ async function handleStatus(sb: any, body: any, mh: string) {
|
|
|
193
211
|
.order("name");
|
|
194
212
|
if (body.project) q = q.eq("project", body.project);
|
|
195
213
|
const { data: services } = await q;
|
|
196
|
-
await auditLog(sb, mh, "all", "status", "success");
|
|
197
214
|
return { services: services || [] };
|
|
198
215
|
}
|
|
199
216
|
|
|
@@ -204,7 +221,6 @@ async function handleChangePassword(sb: any, body: any, mh: string) {
|
|
|
204
221
|
.update({ hmac_seed_hash: new_hmac_seed_hash, fail_count: 0, locked: false })
|
|
205
222
|
.eq("machine_hash", mh);
|
|
206
223
|
if (error) return { error: error.message };
|
|
207
|
-
await auditLog(sb, mh, "system", "change-password", "success");
|
|
208
224
|
return { success: true };
|
|
209
225
|
}
|
|
210
226
|
|
|
@@ -231,12 +247,8 @@ async function handleCreateEnrollment(sb: any, body: any, mh: string) {
|
|
|
231
247
|
created_by_machine_hash: mh,
|
|
232
248
|
expires_at,
|
|
233
249
|
});
|
|
234
|
-
if (error) {
|
|
235
|
-
await auditLog(sb, mh, "system", "create-enrollment", "fail", error.message);
|
|
236
|
-
return { error: error.message };
|
|
237
|
-
}
|
|
250
|
+
if (error) return { error: error.message };
|
|
238
251
|
|
|
239
|
-
await auditLog(sb, mh, "system", "create-enrollment", "success", `install_id=${install_id}`);
|
|
240
252
|
return { success: true, enrollment_code: code, install_id, expires_at, label };
|
|
241
253
|
}
|
|
242
254
|
|
|
@@ -274,7 +286,6 @@ async function handleRedeemEnrollment(sb: any, body: any) {
|
|
|
274
286
|
if (consumeError) return { error: consumeError.message };
|
|
275
287
|
if (!consumedRows || consumedRows.length !== 1) return { error: "enrollment_already_used" };
|
|
276
288
|
|
|
277
|
-
await auditLog(sb, machine_hash, "system", "redeem-enrollment", "success", `install_id=${install_id}`);
|
|
278
289
|
return { success: true, machine_hash, install_id };
|
|
279
290
|
}
|
|
280
291
|
|
|
@@ -314,21 +325,11 @@ Deno.serve(async (req: Request) => {
|
|
|
314
325
|
|
|
315
326
|
const ipCheck = checkIP(ip);
|
|
316
327
|
if (!ipCheck.allowed) {
|
|
317
|
-
await auditLog(sb, body.machine_hash || "unknown", "system", route, "blocked", ipCheck.reason);
|
|
318
328
|
return Response.json({ error: "ip_blocked", reason: ipCheck.reason }, { status: 403 });
|
|
319
329
|
}
|
|
320
330
|
|
|
321
|
-
if (body.machine_hash) {
|
|
322
|
-
const rateCheck = await checkRateLimit(sb, body.machine_hash);
|
|
323
|
-
if (!rateCheck.allowed) {
|
|
324
|
-
await auditLog(sb, body.machine_hash, "system", route, "rate_limited", rateCheck.reason);
|
|
325
|
-
return Response.json({ error: "rate_limited", reason: rateCheck.reason }, { status: 429 });
|
|
326
|
-
}
|
|
327
|
-
}
|
|
328
|
-
|
|
329
331
|
const authResult = await validateHMAC(sb, { machine_hash: body.machine_hash, token: body.token, timestamp: body.timestamp, password: body.password });
|
|
330
332
|
if (!authResult.valid) {
|
|
331
|
-
await auditLog(sb, body.machine_hash || "unknown", body.service || "unknown", route, "denied", authResult.reason);
|
|
332
333
|
return Response.json({ error: "auth_failed", reason: authResult.reason }, { status: 401 });
|
|
333
334
|
}
|
|
334
335
|
|