shraga 0.1.112 → 0.1.114
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/defaults/mcps/README.md +6 -3
- package/defaults/scripts/agent-once.ts +2 -0
- package/defaults/security/policy.example.json +27 -0
- package/defaults/skills/mcp-server.md +15 -14
- package/defaults/skills/platform.md +3 -3
- package/defaults/system-prompt.md +1 -1
- package/dist/client/assets/index-CFxKc8dH.css +10 -0
- package/dist/client/assets/index-DwmlZAEF.js +2009 -0
- package/dist/client/index.html +2 -2
- package/package.json +3 -2
- package/src/client/App.tsx +33 -8
- package/src/client/components/BackendStatusBanner.tsx +62 -0
- package/src/client/components/ConfigPanel.tsx +83 -15
- package/src/client/components/ConversationHeader.tsx +5 -1
- package/src/client/components/McpManager.tsx +26 -9
- package/src/client/components/Sidebar.tsx +16 -1
- package/src/client/components/SkillsManager.tsx +48 -27
- package/src/client/components/owner/ApiKeysTab.tsx +87 -0
- package/src/client/components/owner/AuditTab.tsx +110 -0
- package/src/client/components/owner/BindingsTab.tsx +110 -0
- package/src/client/components/owner/BlocklistTab.tsx +96 -0
- package/src/client/components/owner/OwnerConsole.tsx +87 -0
- package/src/client/components/owner/PrincipalsTab.tsx +55 -0
- package/src/client/components/owner/RolesTab.tsx +132 -0
- package/src/client/components/owner/shared.tsx +84 -0
- package/src/client/hooks/useAuth.ts +11 -2
- package/src/client/hooks/useIsOwner.ts +24 -0
- package/src/client/hooks/useModules.ts +5 -1
- package/src/client/hooks/useOwner.ts +20 -0
- package/src/client/lib/api.ts +23 -7
- package/src/client/lib/backendHealth.ts +230 -0
- package/src/client/lib/debug.ts +48 -0
- package/src/client/lib/sessionApi.ts +24 -8
- package/src/client/lib/ws.ts +21 -13
- package/src/index.ts +10 -2
- package/src/scripts/harden-audit.sh +55 -0
- package/src/server/agent-config.ts +34 -0
- package/src/server/api-key-routes.ts +66 -0
- package/src/server/api-keys.ts +181 -43
- package/src/server/auth.ts +113 -47
- package/src/server/boot.ts +161 -104
- package/src/server/claude.ts +102 -26
- package/src/server/data-sync.ts +56 -6
- package/src/server/directives.ts +2 -4
- package/src/server/engine/claude-code.ts +78 -17
- package/src/server/engine/types.ts +7 -0
- package/src/server/hooks.ts +19 -0
- package/src/server/mcp-oauth.ts +24 -5
- package/src/server/mcp-server.ts +55 -25
- package/src/server/modules/routes.ts +2 -6
- package/src/server/notify-owners.ts +5 -17
- package/src/server/owners.ts +14 -0
- package/src/server/scheduler/builtins.ts +3 -1
- package/src/server/scheduler/runner.ts +3 -0
- package/src/server/security/audit.ts +505 -0
- package/src/server/security/enforce.ts +343 -0
- package/src/server/security/escalate.ts +194 -0
- package/src/server/security/guard.ts +337 -0
- package/src/server/security/owner-only.ts +34 -0
- package/src/server/security/owner-routes.ts +285 -0
- package/src/server/security/policy.ts +416 -0
- package/src/server/security/principal.ts +80 -0
- package/src/server/security/revocation.ts +50 -0
- package/src/server/security/runtime.ts +226 -0
- package/src/server/sessions.ts +38 -0
- package/src/server/shraga-config.ts +13 -0
- package/src/server/slack/bot.ts +44 -11
- package/src/server/slack/context-cache.ts +40 -7
- package/src/server/webhook-lane/feature.ts +17 -6
- package/src/shared/models.ts +11 -0
- package/dist/client/assets/index-DIMte_k6.css +0 -10
- package/dist/client/assets/index-Dc1ljSt3.js +0 -1949
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
/** Backend-health classifier + tiny store.
|
|
2
|
+
*
|
|
3
|
+
* WHY this exists: a broken backend used to be INVISIBLE in the UI. When something other than shraga
|
|
4
|
+
* answered the app's port (e.g. another project's dev server bound `127.0.0.1:3033` while shraga held
|
|
5
|
+
* `*:3033` — macOS routes the loopback name to the MORE SPECIFIC bind, so the proxy fed every request
|
|
6
|
+
* to the wrong process), the foreign server 404'd `/api/config`, `/api/workspace` and refused the `/ws`
|
|
7
|
+
* upgrade. The UI rendered a normal-looking EMPTY shell: no workspace, no terminals, no error. Every
|
|
8
|
+
* clue was in the browser console and nothing reached the screen.
|
|
9
|
+
*
|
|
10
|
+
* The load-bearing signal is {@link SHRAGA_VERSION_HEADER}: the shraga server stamps it on EVERY
|
|
11
|
+
* response including 404/401/5xx (src/server/boot.ts). So a response WITHOUT it did not come from
|
|
12
|
+
* shraga at all — which turns an ambiguous "some 404" into the exact diagnosis "something else is
|
|
13
|
+
* serving this port".
|
|
14
|
+
*
|
|
15
|
+
* Producers report here from the SHARED helpers (`apiFetch`, `api`, `AgentSocket`) so every existing
|
|
16
|
+
* call site benefits with no call-site change. The single consumer is <BackendStatusBanner />.
|
|
17
|
+
*/
|
|
18
|
+
import { logger } from '@/lib/debug';
|
|
19
|
+
|
|
20
|
+
const log = logger.forComponent('BackendHealth');
|
|
21
|
+
|
|
22
|
+
/** Identity header stamped by the shraga server on every response. Lowercase: `Headers` is
|
|
23
|
+
* case-insensitive, but keeping it lowercase avoids any doubt about what we're matching. */
|
|
24
|
+
export const SHRAGA_VERSION_HEADER = 'x-shraga-version';
|
|
25
|
+
|
|
26
|
+
export type BackendFaultKind = 'wrong-backend' | 'server-error' | 'offline' | 'ws-down';
|
|
27
|
+
|
|
28
|
+
export interface BackendFault {
|
|
29
|
+
kind: BackendFaultKind;
|
|
30
|
+
/** One-line headline for the banner. */
|
|
31
|
+
title: string;
|
|
32
|
+
/** Actionable next step — what the user should actually go do. */
|
|
33
|
+
hint: string;
|
|
34
|
+
/** The request that exposed the fault, when there is one. */
|
|
35
|
+
path?: string;
|
|
36
|
+
status?: number;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Fault sources are tracked SEPARATELY: HTTP recovering must not erase a still-broken WebSocket
|
|
40
|
+
* (that is exactly the "terminals stay blank while the page looks fine" half of the incident). */
|
|
41
|
+
type Source = 'http' | 'ws';
|
|
42
|
+
const faults: Partial<Record<Source, BackendFault>> = {};
|
|
43
|
+
const raisedAt: Partial<Record<Source, number>> = {};
|
|
44
|
+
|
|
45
|
+
/** A raised fault does NOT vanish on the very next good response. The app polls continuously (pty cwd,
|
|
46
|
+
* workspace, layout), so a single interleaved success would otherwise erase the banner milliseconds
|
|
47
|
+
* after it appeared — leaving a user staring at a flicker they can neither read nor act on. Recovery
|
|
48
|
+
* must therefore be CONFIRMED: several consecutive good responses AND a minimum time on screen. */
|
|
49
|
+
/** ...and the SAME reasoning governs the WebSocket. An ordinary reconnect blip closes and re-opens in
|
|
50
|
+
* a couple of seconds (backoff is ~1s * 2^attempt + jitter); painting and erasing an amber banner
|
|
51
|
+
* inside that window is the flicker described above, not information. So a close is given
|
|
52
|
+
* `wsGraceMs` to recover before it becomes a fault at all — which still catches the case that
|
|
53
|
+
* motivated the feature, a socket that never comes back (including one wedged in CONNECTING, where
|
|
54
|
+
* counting reconnect ATTEMPTS would never trip) — and a recovered socket serves the same
|
|
55
|
+
* `minDwellMs` before the banner is taken down. */
|
|
56
|
+
const DEFAULT_TIMING = { okStreakToClear: 3, minDwellMs: 5_000, wsGraceMs: 5_000 };
|
|
57
|
+
type Timing = typeof DEFAULT_TIMING;
|
|
58
|
+
let timing: Timing = { ...DEFAULT_TIMING };
|
|
59
|
+
let okStreak = 0;
|
|
60
|
+
/** The single pending ws timer: a delayed RAISE while no ws fault is shown, a delayed CLEAR while one
|
|
61
|
+
* is (`faults.ws` disambiguates, and every path that changes that state cancels it first). */
|
|
62
|
+
let wsTimer: ReturnType<typeof setTimeout> | null = null;
|
|
63
|
+
function cancelWsTimer() {
|
|
64
|
+
if (wsTimer) { clearTimeout(wsTimer); wsTimer = null; }
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Worst-first. `wrong-backend` outranks everything: when the port is hijacked the 404s and the dead
|
|
68
|
+
* socket are SYMPTOMS, and showing either of those instead would send the user down the wrong path. */
|
|
69
|
+
const PRECEDENCE: BackendFaultKind[] = ['wrong-backend', 'offline', 'server-error', 'ws-down'];
|
|
70
|
+
|
|
71
|
+
type Listener = (fault: BackendFault | null) => void;
|
|
72
|
+
const listeners = new Set<Listener>();
|
|
73
|
+
|
|
74
|
+
export function currentFault(): BackendFault | null {
|
|
75
|
+
const present = (Object.values(faults) as BackendFault[]).filter(Boolean);
|
|
76
|
+
if (present.length === 0) return null;
|
|
77
|
+
return present.sort((a, b) => PRECEDENCE.indexOf(a.kind) - PRECEDENCE.indexOf(b.kind))[0];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function emit() {
|
|
81
|
+
const f = currentFault();
|
|
82
|
+
listeners.forEach((l) => l(f));
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export function subscribeBackendHealth(listener: Listener): () => void {
|
|
86
|
+
listeners.add(listener);
|
|
87
|
+
listener(currentFault());
|
|
88
|
+
return () => { listeners.delete(listener); };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function set(source: Source, fault: BackendFault) {
|
|
92
|
+
const prev = faults[source];
|
|
93
|
+
faults[source] = fault;
|
|
94
|
+
// Restart the dwell when the fault CHANGES KIND too: a wrong-backend that supersedes an older
|
|
95
|
+
// server-error is a new message and needs its own time on screen, not the predecessor's leftovers.
|
|
96
|
+
if (prev?.kind !== fault.kind) raisedAt[source] = Date.now();
|
|
97
|
+
if (source === 'http') okStreak = 0;
|
|
98
|
+
// Log the transition only, not every repeat — a broken backend is polled continuously.
|
|
99
|
+
if (prev?.kind !== fault.kind || prev?.path !== fault.path || prev?.status !== fault.status) {
|
|
100
|
+
log.warn(`${fault.kind}: ${fault.title}`, { path: fault.path, status: fault.status });
|
|
101
|
+
}
|
|
102
|
+
emit();
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function clear(source: Source) {
|
|
106
|
+
if (!faults[source]) return;
|
|
107
|
+
log.info(`${source} recovered`);
|
|
108
|
+
delete faults[source];
|
|
109
|
+
delete raisedAt[source];
|
|
110
|
+
emit();
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Describe the origin the app believes it is talking to, for the wrong-backend message. */
|
|
114
|
+
function originLabel(): string {
|
|
115
|
+
try { return window.location.origin; } catch { return 'this origin'; }
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Classify a COMPLETED HTTP response from the shraga API. Call for ok and !ok alike — a wrong backend
|
|
120
|
+
* can answer `200` just as easily as `404`, so the identity header is checked first, unconditionally.
|
|
121
|
+
*
|
|
122
|
+
* `expect` lists statuses this CALL SITE treats as a normal, handled outcome (e.g. the pty-cwd poller,
|
|
123
|
+
* for which 404 means "that pane is gone" and is swallowed by design). An expected status is not a
|
|
124
|
+
* fault: raising a banner for a by-design 404 is the cry-wolf failure this whole module exists to
|
|
125
|
+
* avoid. It is checked AFTER the identity header, so a hijacker answering 404 is still caught.
|
|
126
|
+
*/
|
|
127
|
+
export function reportApiResponse(path: string, res: Response, expect?: readonly number[]): void {
|
|
128
|
+
if (!res.headers.has(SHRAGA_VERSION_HEADER)) {
|
|
129
|
+
set('http', {
|
|
130
|
+
kind: 'wrong-backend',
|
|
131
|
+
title: `${originLabel()} is being served by something that is not Shraga.`,
|
|
132
|
+
hint:
|
|
133
|
+
'Another process is almost certainly bound to the app port and shadowing the real server ' +
|
|
134
|
+
'(a more specific 127.0.0.1 bind wins over Shraga’s wildcard bind). Find it with ' +
|
|
135
|
+
'`lsof -nP -iTCP:<port> -sTCP:LISTEN`, stop it, then retry.',
|
|
136
|
+
path,
|
|
137
|
+
status: res.status,
|
|
138
|
+
});
|
|
139
|
+
return;
|
|
140
|
+
}
|
|
141
|
+
// Genuine shraga auth responses are the ORDINARY login/permission flow — never a banner. Checked
|
|
142
|
+
// after the header so a hijacker that answers 401 is still caught above.
|
|
143
|
+
if (res.status === 401 || res.status === 403) return;
|
|
144
|
+
// Handled by the caller — neither a fault nor evidence of recovery.
|
|
145
|
+
if (expect?.includes(res.status)) return;
|
|
146
|
+
|
|
147
|
+
if (!res.ok) {
|
|
148
|
+
set('http', {
|
|
149
|
+
kind: 'server-error',
|
|
150
|
+
title: `Shraga returned ${res.status} for ${path}.`,
|
|
151
|
+
hint: 'The server is reachable but this request failed. Retry; if it persists, check the server logs.',
|
|
152
|
+
path,
|
|
153
|
+
status: res.status,
|
|
154
|
+
});
|
|
155
|
+
return;
|
|
156
|
+
}
|
|
157
|
+
okStreak++;
|
|
158
|
+
if (faults.http && okStreak >= timing.okStreakToClear && Date.now() - (raisedAt.http ?? 0) >= timing.minDwellMs) {
|
|
159
|
+
clear('http');
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** Classify a request that never produced a response: network down, server unreachable, or aborted.
|
|
164
|
+
*
|
|
165
|
+
* `timedOut` says the HELPER's OWN deadline fired (`apiFetch`'s `timeoutMs` controller). It is the
|
|
166
|
+
* only thing that separates a genuine wedged server from a CALLER-initiated `abort()` — both surface
|
|
167
|
+
* as an indistinguishable `AbortError`. A caller abort is ordinary control flow, not a fault: the tab
|
|
168
|
+
* palette re-issues its search on every keystroke and aborts the in-flight one, so raising here
|
|
169
|
+
* painted a red "cannot reach the server" banner for plain typing. Raise NOTHING for it.
|
|
170
|
+
*/
|
|
171
|
+
export function reportApiFailure(path: string, err: unknown, opts?: { timedOut?: boolean }): void {
|
|
172
|
+
const aborted = (err as { name?: string } | null)?.name === 'AbortError';
|
|
173
|
+
if (aborted && !opts?.timedOut) return;
|
|
174
|
+
set('http', {
|
|
175
|
+
kind: 'offline',
|
|
176
|
+
title: opts?.timedOut ? `Request to ${path} timed out.` : `Cannot reach the Shraga server.`,
|
|
177
|
+
hint: opts?.timedOut
|
|
178
|
+
? 'The server accepted the connection but did not answer in time. It may be overloaded or wedged.'
|
|
179
|
+
: 'The network is down, or nothing is listening on the app port. Check your connection and that the server is running.',
|
|
180
|
+
path,
|
|
181
|
+
});
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/** WebSocket health, reported by {@link AgentSocket}. A dead socket is why terminals and live updates
|
|
185
|
+
* silently stop; without this the panes just stay blank.
|
|
186
|
+
*
|
|
187
|
+
* A close does NOT raise on its own — it starts the grace window (see {@link DEFAULT_TIMING}). Only a
|
|
188
|
+
* socket still down when the window expires is a fault the user can act on. */
|
|
189
|
+
export function reportWsDown(code?: number): void {
|
|
190
|
+
// Already on screen: a re-close during the recovery dwell just keeps the banner up.
|
|
191
|
+
if (faults.ws) { cancelWsTimer(); return; }
|
|
192
|
+
// Grace already running from an earlier close in the same outage — don't restart it, or a socket
|
|
193
|
+
// that flaps every few seconds would push the deadline out forever and never raise.
|
|
194
|
+
if (wsTimer) return;
|
|
195
|
+
const raise = () => {
|
|
196
|
+
wsTimer = null;
|
|
197
|
+
set('ws', {
|
|
198
|
+
kind: 'ws-down',
|
|
199
|
+
title: 'Live connection lost — terminals and streaming updates are offline.',
|
|
200
|
+
hint:
|
|
201
|
+
code === 1006
|
|
202
|
+
? 'The connection was refused or dropped without a close handshake, which usually means the ' +
|
|
203
|
+
'WebSocket upgrade never reached Shraga. Reconnecting…'
|
|
204
|
+
: 'Reconnecting…',
|
|
205
|
+
status: code,
|
|
206
|
+
});
|
|
207
|
+
};
|
|
208
|
+
if (timing.wsGraceMs <= 0) raise();
|
|
209
|
+
else wsTimer = setTimeout(raise, timing.wsGraceMs);
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
export function reportWsUp(): void {
|
|
213
|
+
cancelWsTimer(); // a blip that recovered inside the grace window never earned a banner
|
|
214
|
+
if (!faults.ws) return;
|
|
215
|
+
const remaining = timing.minDwellMs - (Date.now() - (raisedAt.ws ?? 0));
|
|
216
|
+
if (remaining <= 0) { clear('ws'); return; }
|
|
217
|
+
wsTimer = setTimeout(() => { wsTimer = null; clear('ws'); }, remaining);
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/** Test seam — drop all state, and optionally shrink the timings so a test needn't wait seconds. */
|
|
221
|
+
export function __resetBackendHealth(overrides?: Partial<Timing>): void {
|
|
222
|
+
cancelWsTimer();
|
|
223
|
+
timing = { ...DEFAULT_TIMING, ...overrides };
|
|
224
|
+
delete faults.http;
|
|
225
|
+
delete faults.ws;
|
|
226
|
+
delete raisedAt.http;
|
|
227
|
+
delete raisedAt.ws;
|
|
228
|
+
okStreak = 0;
|
|
229
|
+
emit();
|
|
230
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/** Component-scoped console logger.
|
|
2
|
+
*
|
|
3
|
+
* `debug`/`verbose` are development output and are GATED per component — they print only when the
|
|
4
|
+
* component's name (or `*`) appears in the comma-separated `DEBUG` localStorage key:
|
|
5
|
+
* localStorage.setItem('DEBUG', 'BackendHealth,ws')
|
|
6
|
+
* `info`/`warn`/`error` are operational and ALWAYS print. Use this instead of raw `console.*` so a
|
|
7
|
+
* noisy module can be silenced without deleting the logging that diagnoses a live incident.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
type LogFn = (...args: unknown[]) => void;
|
|
11
|
+
|
|
12
|
+
export interface ComponentLogger {
|
|
13
|
+
debug: LogFn;
|
|
14
|
+
verbose: LogFn;
|
|
15
|
+
info: LogFn;
|
|
16
|
+
warn: LogFn;
|
|
17
|
+
error: LogFn;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** Read once per call rather than caching: a dev flipping `DEBUG` in devtools expects it to take
|
|
21
|
+
* effect without a reload, and this only runs on a log call that is already about to hit console. */
|
|
22
|
+
function gateAllows(component: string): boolean {
|
|
23
|
+
let raw: string | null = null;
|
|
24
|
+
try {
|
|
25
|
+
raw = localStorage.getItem('DEBUG');
|
|
26
|
+
} catch {
|
|
27
|
+
return false; // storage blocked (private mode / sandboxed iframe) — dev output is not worth throwing over
|
|
28
|
+
}
|
|
29
|
+
if (!raw) return false;
|
|
30
|
+
return raw
|
|
31
|
+
.split(',')
|
|
32
|
+
.map((s) => s.trim())
|
|
33
|
+
.some((s) => s === '*' || s === component);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export const logger = {
|
|
37
|
+
forComponent(component: string): ComponentLogger {
|
|
38
|
+
const tag = `[${component}]`;
|
|
39
|
+
const gated = (fn: LogFn): LogFn => (...args) => { if (gateAllows(component)) fn(tag, ...args); };
|
|
40
|
+
return {
|
|
41
|
+
debug: gated((...a) => console.debug(...a)),
|
|
42
|
+
verbose: gated((...a) => console.debug(...a)),
|
|
43
|
+
info: (...a) => console.info(tag, ...a),
|
|
44
|
+
warn: (...a) => console.warn(tag, ...a),
|
|
45
|
+
error: (...a) => console.error(tag, ...a),
|
|
46
|
+
};
|
|
47
|
+
},
|
|
48
|
+
};
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { reportApiFailure, reportApiResponse } from '@/lib/backendHealth';
|
|
1
2
|
import { randomUUID } from '@/lib/utils';
|
|
2
3
|
import type { ChatMessage, MessageBlock } from '@/hooks/useConversation';
|
|
3
4
|
|
|
@@ -15,24 +16,39 @@ export class ApiError extends Error {
|
|
|
15
16
|
}
|
|
16
17
|
}
|
|
17
18
|
|
|
18
|
-
/** Authenticated fetch with bearer token + timeout. Throws `ApiError` (with `.status`) on !ok.
|
|
19
|
+
/** Authenticated fetch with bearer token + timeout. Throws `ApiError` (with `.status`) on !ok.
|
|
20
|
+
*
|
|
21
|
+
* `expect` lists statuses this call site HANDLES as a normal outcome (it still throws — see `ApiError`
|
|
22
|
+
* — the list only tells the backend-health classifier not to treat them as a fault). Use it wherever a
|
|
23
|
+
* non-2xx is by design, or the shared banner cries wolf on routine traffic. */
|
|
19
24
|
export async function apiFetch(
|
|
20
25
|
path: string,
|
|
21
26
|
getToken: () => Promise<string | null>,
|
|
22
|
-
init?: RequestInit & { timeoutMs?: number },
|
|
27
|
+
init?: RequestInit & { timeoutMs?: number; expect?: readonly number[] },
|
|
23
28
|
) {
|
|
24
29
|
const token = await getToken();
|
|
25
30
|
if (!token) throw new Error('No auth token');
|
|
26
31
|
const timeout = init?.timeoutMs ?? 15_000;
|
|
27
32
|
const controller = new AbortController();
|
|
28
33
|
if (init?.signal) init.signal.addEventListener('abort', () => controller.abort());
|
|
29
|
-
|
|
34
|
+
// `timedOut` is the ONLY thing that distinguishes our own deadline from a caller's abort() — both
|
|
35
|
+
// reject with an identical AbortError, and only the former is a real backend fault.
|
|
36
|
+
let timedOut = false;
|
|
37
|
+
const timer = setTimeout(() => { timedOut = true; controller.abort(); }, timeout);
|
|
30
38
|
try {
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
39
|
+
let res: Response;
|
|
40
|
+
try {
|
|
41
|
+
res = await fetch(path, {
|
|
42
|
+
...init,
|
|
43
|
+
signal: controller.signal,
|
|
44
|
+
headers: { Authorization: `Bearer ${token}`, ...init?.headers },
|
|
45
|
+
});
|
|
46
|
+
} catch (err) {
|
|
47
|
+
// Single choke point: every apiFetch call site gets backend-fault surfacing for free.
|
|
48
|
+
reportApiFailure(path, err, { timedOut });
|
|
49
|
+
throw err;
|
|
50
|
+
}
|
|
51
|
+
reportApiResponse(path, res, init?.expect);
|
|
36
52
|
if (!res.ok) throw new ApiError(res.status, res.statusText);
|
|
37
53
|
return res;
|
|
38
54
|
} finally {
|
package/src/client/lib/ws.ts
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
import { reportWsDown, reportWsUp } from '@/lib/backendHealth';
|
|
2
|
+
import { logger } from '@/lib/debug';
|
|
3
|
+
|
|
4
|
+
const log = logger.forComponent('ws');
|
|
5
|
+
|
|
1
6
|
/** Subprotocol marker used to carry a bearer token through a WebSocket handshake: open the socket as
|
|
2
7
|
* `new WebSocket(url, [WS_AUTH_PROTOCOL, token])`. Browsers can't set headers on a WS handshake, and the
|
|
3
8
|
* subprotocol list is the one field they can — used by the sidecar WS proxy (see authenticateWsUpgrade
|
|
@@ -72,7 +77,7 @@ export class AgentSocket {
|
|
|
72
77
|
this.intentionalClose = false;
|
|
73
78
|
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
|
|
74
79
|
const url = `${proto}://${location.host}/ws`;
|
|
75
|
-
|
|
80
|
+
log.debug('connecting…');
|
|
76
81
|
this.ws = new WebSocket(url);
|
|
77
82
|
|
|
78
83
|
this.ws.onopen = async () => {
|
|
@@ -81,11 +86,12 @@ export class AgentSocket {
|
|
|
81
86
|
try {
|
|
82
87
|
token = await this.tokenProvider();
|
|
83
88
|
} catch (err) {
|
|
84
|
-
|
|
89
|
+
log.warn('token fetch failed', err);
|
|
85
90
|
}
|
|
86
91
|
if (this.ws?.readyState !== WebSocket.OPEN) return;
|
|
87
|
-
|
|
92
|
+
log.debug('open, sending auth');
|
|
88
93
|
this.ws.send(JSON.stringify({ type: 'auth', token }));
|
|
94
|
+
reportWsUp();
|
|
89
95
|
if (this.reconnecting) {
|
|
90
96
|
this.reconnecting = false;
|
|
91
97
|
this.listeners.forEach((l) => l({ type: 'reconnected' }));
|
|
@@ -95,7 +101,7 @@ export class AgentSocket {
|
|
|
95
101
|
this.ws.onmessage = (e) => {
|
|
96
102
|
try {
|
|
97
103
|
const data = JSON.parse(e.data) as ServerEvent;
|
|
98
|
-
if (data.type !== 'text_delta')
|
|
104
|
+
if (data.type !== 'text_delta') log.verbose('←', data.type);
|
|
99
105
|
if (data.type === 'auth_ok') {
|
|
100
106
|
this.authRetries = 0;
|
|
101
107
|
this.flushPending();
|
|
@@ -113,37 +119,38 @@ export class AgentSocket {
|
|
|
113
119
|
this.intentionalClose = true;
|
|
114
120
|
} else {
|
|
115
121
|
this.authRetries++;
|
|
116
|
-
|
|
122
|
+
log.warn(`auth failed (attempt ${this.authRetries}), will retry with fresh token`);
|
|
117
123
|
}
|
|
118
124
|
}
|
|
119
125
|
this.listeners.forEach((l) => l(data));
|
|
120
126
|
} catch (err) {
|
|
121
|
-
|
|
127
|
+
log.warn('bad frame', err);
|
|
122
128
|
}
|
|
123
129
|
};
|
|
124
130
|
|
|
125
131
|
this.ws.onclose = (e) => {
|
|
126
|
-
|
|
132
|
+
log.debug(`closed code=${e.code} intentional=${this.intentionalClose}`);
|
|
127
133
|
if (!this.intentionalClose) {
|
|
128
134
|
this.reconnecting = true;
|
|
129
135
|
this.connectAttempts++;
|
|
136
|
+
reportWsDown(e.code);
|
|
130
137
|
this.listeners.forEach((l) => l({ type: 'disconnected' }));
|
|
131
138
|
const base = this.authRetries > 0 ? Math.min(2000 * this.authRetries, 10000) : Math.min(1000 * 2 ** this.connectAttempts, 30000);
|
|
132
139
|
const jitter = Math.random() * 1000;
|
|
133
140
|
const delay = base + jitter;
|
|
134
|
-
|
|
141
|
+
log.debug(`reconnecting in ${(delay / 1000).toFixed(1)}s (attempt ${this.connectAttempts})`);
|
|
135
142
|
setTimeout(() => this.connect(this.tokenProvider), delay);
|
|
136
143
|
}
|
|
137
144
|
};
|
|
138
145
|
|
|
139
|
-
this.ws.onerror = (e) =>
|
|
146
|
+
this.ws.onerror = (e) => log.warn('error', e);
|
|
140
147
|
}
|
|
141
148
|
|
|
142
149
|
private flushPending() {
|
|
143
150
|
if (this.pendingMessage) {
|
|
144
151
|
const msg = this.pendingMessage;
|
|
145
152
|
this.pendingMessage = null;
|
|
146
|
-
|
|
153
|
+
log.debug('flushing pending message after reconnect');
|
|
147
154
|
if (!this.send(msg)) {
|
|
148
155
|
this.pendingMessage = msg;
|
|
149
156
|
}
|
|
@@ -152,24 +159,25 @@ export class AgentSocket {
|
|
|
152
159
|
|
|
153
160
|
disconnect() {
|
|
154
161
|
this.intentionalClose = true;
|
|
162
|
+
reportWsUp(); // an intentional close is not a fault — don't leave a stale banner behind
|
|
155
163
|
this.ws?.close();
|
|
156
164
|
this.ws = null;
|
|
157
165
|
}
|
|
158
166
|
|
|
159
167
|
send(msg: object): boolean {
|
|
160
168
|
if (this.ws?.readyState === WebSocket.OPEN) {
|
|
161
|
-
|
|
169
|
+
log.verbose('→', (msg as any).type);
|
|
162
170
|
this.ws.send(JSON.stringify(msg));
|
|
163
171
|
return true;
|
|
164
172
|
}
|
|
165
|
-
|
|
173
|
+
log.warn('send failed, readyState=', this.ws?.readyState);
|
|
166
174
|
return false;
|
|
167
175
|
}
|
|
168
176
|
|
|
169
177
|
sendOrQueue(msg: object): boolean {
|
|
170
178
|
if (this.send(msg)) return true;
|
|
171
179
|
this.pendingMessage = msg;
|
|
172
|
-
|
|
180
|
+
log.debug('message queued for reconnect');
|
|
173
181
|
return false;
|
|
174
182
|
}
|
|
175
183
|
|
package/src/index.ts
CHANGED
|
@@ -25,8 +25,10 @@ import type { AgentEngine, EngineModel, EngineStreamOpts } from './server/engine
|
|
|
25
25
|
import type { ExtRegisterFn, ExtensionContext } from './server/extensions.ts';
|
|
26
26
|
import type { WebhookOptions } from './server/events/webhook.ts';
|
|
27
27
|
import type { ShragaEvent, ShragaEventMap, PayloadOf } from './server/events/types.ts';
|
|
28
|
+
import type { PolicyOptions, PolicyFile } from './server/security/policy.ts';
|
|
28
29
|
|
|
29
30
|
export type {
|
|
31
|
+
PolicyFile,
|
|
30
32
|
ServerHandle,
|
|
31
33
|
ServerFeature,
|
|
32
34
|
FeatureContext,
|
|
@@ -67,6 +69,12 @@ export class ShragaOptions {
|
|
|
67
69
|
runtimeRegistration?: boolean = false;
|
|
68
70
|
/** Arbitrary extra environment to apply before boot (e.g. ANTHROPIC_API_KEY, SHRAGA_FEAT_*). */
|
|
69
71
|
env?: Record<string, string>;
|
|
72
|
+
/** Security policy seams. `migrate` seeds `data/security/policy.json` ONCE, when it's first generated (no file, no
|
|
73
|
+
* `.migrated` marker): it receives the default draft (legacy whitelist already bound as operator) and returns the
|
|
74
|
+
* final draft. The result is validated and saved like any policy; an invalid draft or a throw fails closed (owners
|
|
75
|
+
* only). Never runs on a PASSIVE standby (it runs on promotion if the file is still missing). Later edits go through
|
|
76
|
+
* the Owner Console, not this hook. */
|
|
77
|
+
security?: { migrate?: PolicyOptions['migrate'] };
|
|
70
78
|
}
|
|
71
79
|
|
|
72
80
|
export interface ShragaInstance {
|
|
@@ -94,7 +102,7 @@ export interface ShragaInstance {
|
|
|
94
102
|
|
|
95
103
|
class Shraga implements ShragaInstance {
|
|
96
104
|
public options: ShragaOptions;
|
|
97
|
-
private reg: Required<BootRegistrations
|
|
105
|
+
private reg: Required<Omit<BootRegistrations, 'security'>> ={ features: [], engines: [], extensions: [], eventSubs: [] };
|
|
98
106
|
private handle: ServerHandle | null = null;
|
|
99
107
|
private starting: Promise<ServerHandle> | null = null;
|
|
100
108
|
|
|
@@ -150,7 +158,7 @@ class Shraga implements ShragaInstance {
|
|
|
150
158
|
if (this.starting) return this.starting;
|
|
151
159
|
this.starting = (async () => {
|
|
152
160
|
const { bootServer } = await import('./server/boot.ts');
|
|
153
|
-
this.handle = await bootServer(this.reg);
|
|
161
|
+
this.handle = await bootServer({ ...this.reg, security: this.options.security });
|
|
154
162
|
return this.handle;
|
|
155
163
|
})();
|
|
156
164
|
return this.starting;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# harden-audit.sh — make shraga's audit log append-only at the OS level. Linux + root. Idempotent: run it from cron.
|
|
3
|
+
#
|
|
4
|
+
# sudo harden-audit.sh <DATA_DIR> (or DATA_DIR=<dir> sudo -E harden-audit.sh)
|
|
5
|
+
# root crontab, hourly: 17 * * * * /app/node_modules/shraga/src/scripts/harden-audit.sh /app/data-prod
|
|
6
|
+
#
|
|
7
|
+
# `chattr +a` on <DATA_DIR>/audit/: the server user can still create month files and append, but can't delete or
|
|
8
|
+
# rename any entry. `chattr +a` on each YYYY-MM.jsonl: it opens for append only — no truncate, no rewrite.
|
|
9
|
+
# New files do NOT inherit +a (it is not in ext4's EXT4_FL_INHERITED nor XFS's inherited flags), so a new month's
|
|
10
|
+
# file stays truncatable until the next run — hence hourly cron. Retention/rotation is a root-only op (chattr -a).
|
|
11
|
+
# +a can't stop the server user from CREATING a month entry as a symlink/directory (planted). Those are never hardened:
|
|
12
|
+
# they're reported and the script exits non-zero — after +a is applied to every valid file. The server refuses to
|
|
13
|
+
# append through them and alerts owners. Remove one as root: chattr -a "$audit"; rm -r <entry>; re-run.
|
|
14
|
+
set -euo pipefail
|
|
15
|
+
|
|
16
|
+
[ "$(uname -s)" = Linux ] || { echo "harden-audit: Linux only (chattr) — nothing done" >&2; exit 1; }
|
|
17
|
+
[ "$(id -u)" -eq 0 ] || { echo "harden-audit: must run as root (chattr +a needs CAP_LINUX_IMMUTABLE)" >&2; exit 1; }
|
|
18
|
+
data="${1:-${DATA_DIR:-}}"
|
|
19
|
+
[ -n "$data" ] || { echo "usage: $0 <DATA_DIR>" >&2; exit 2; }
|
|
20
|
+
audit="$(realpath "$data")/audit"
|
|
21
|
+
[ -d "$audit" ] && [ ! -L "$audit" ] || { echo "harden-audit: $audit is not a directory (start the server once first)" >&2; exit 1; }
|
|
22
|
+
# Builds before tamper protection lock INSIDE the audit dir; in an append-only dir that lock can never be released.
|
|
23
|
+
if [ -e "$audit/.lock" ]; then
|
|
24
|
+
{
|
|
25
|
+
echo "harden-audit: $audit/.lock exists — left by a pre-tamper-protection shraga (it locks inside the audit dir)."
|
|
26
|
+
echo " 1. Verify no old build is running against $data (e.g. ps -eo pid,args | grep shraga); upgrade or stop it."
|
|
27
|
+
if lsattr -d "$audit" 2>/dev/null | awk '{print $1}' | grep -q a; then
|
|
28
|
+
echo " 2. $audit is already append-only, so first: chattr -a '$audit'"
|
|
29
|
+
echo " 3. As root: rmdir '$audit/.lock'"
|
|
30
|
+
else
|
|
31
|
+
echo " 2. As root: rmdir '$audit/.lock'"
|
|
32
|
+
fi
|
|
33
|
+
echo " Then re-run: $0 $data"
|
|
34
|
+
} >&2
|
|
35
|
+
exit 1
|
|
36
|
+
fi
|
|
37
|
+
|
|
38
|
+
chattr +a "$audit"
|
|
39
|
+
shopt -s nullglob
|
|
40
|
+
files=() bad=()
|
|
41
|
+
for f in "$audit"/*.jsonl; do
|
|
42
|
+
if [ -f "$f" ] && [ ! -L "$f" ]; then files+=("$f"); else bad+=("$f"); fi
|
|
43
|
+
done
|
|
44
|
+
for f in "${files[@]}"; do chattr +a "$f"; done
|
|
45
|
+
|
|
46
|
+
lsattr -d "$audit"
|
|
47
|
+
[ ${#files[@]} -eq 0 ] || lsattr "${files[@]}"
|
|
48
|
+
echo "harden-audit: $audit append-only (+${#files[@]} month files) — server can append, not truncate/delete/rename"
|
|
49
|
+
|
|
50
|
+
if [ ${#bad[@]} -gt 0 ]; then
|
|
51
|
+
for f in "${bad[@]}"; do
|
|
52
|
+
echo "harden-audit: ERROR — $f is not a regular file ($(stat -c %F "$f" 2>/dev/null || echo unknown)$([ -L "$f" ] && echo " -> $(readlink "$f")")); possibly PLANTED to divert or disable auditing. NOT hardened. Inspect, then as root: chattr -a '$audit' && rm -r '$f' && $0 $data" >&2
|
|
53
|
+
done
|
|
54
|
+
exit 1
|
|
55
|
+
fi
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
// Agent config READ path (shared across users), as its own leaf module.
|
|
2
|
+
//
|
|
3
|
+
// Why it isn't in claude.ts: the security runtime reads a flag from it (`allowUntrustedReplies`, see
|
|
4
|
+
// `mayReplyTo`), and claude.ts imports the runtime — plus the whole engine/skills/sessions graph. A leaf
|
|
5
|
+
// that only touches fs + paths keeps that dependency one-way and cheap. The WRITE path stays in claude.ts
|
|
6
|
+
// (it also has to tell data-sync, which pulls the Agent SDK).
|
|
7
|
+
import { existsSync, readFileSync } from 'node:fs';
|
|
8
|
+
import { dataPath } from './paths.ts';
|
|
9
|
+
import type { AgentSettings as AgentConfig } from './shraga-config.ts';
|
|
10
|
+
|
|
11
|
+
export const CONFIG_PATH = dataPath('agent-config.json');
|
|
12
|
+
|
|
13
|
+
export const DEFAULT_CONFIG: AgentConfig = {
|
|
14
|
+
/** ToolSearch loads deferred MCP tools; without it, permission prompts / tool graph can block Meta Ads tools. */
|
|
15
|
+
allowedTools: ['Read', 'Edit', 'Bash', 'WebSearch', 'Glob', 'LS', 'ToolSearch'],
|
|
16
|
+
permissionMode: 'acceptEdits',
|
|
17
|
+
maxTurns: 15,
|
|
18
|
+
// Defaults are what a fresh self-hosted install runs before anyone touches the UI, so they favour
|
|
19
|
+
// cost/latency over ceiling. Both are overridable per-deployment via agent-config.json and per-send
|
|
20
|
+
// via directives — an operator who wants a bigger model sets it once; every operator pays for a default.
|
|
21
|
+
model: 'claude-sonnet-5',
|
|
22
|
+
effort: 'low',
|
|
23
|
+
// `allowUntrustedReplies` is deliberately absent: undefined = off, so a fresh install never auto-replies
|
|
24
|
+
// to an unverified sender, and an owner has to turn it on explicitly.
|
|
25
|
+
};
|
|
26
|
+
|
|
27
|
+
export function getAgentConfig(): AgentConfig {
|
|
28
|
+
// agent-config.json (UI-writable, git-tracked) is the single source of truth for agent settings.
|
|
29
|
+
let config = { ...DEFAULT_CONFIG };
|
|
30
|
+
if (existsSync(CONFIG_PATH)) {
|
|
31
|
+
try { Object.assign(config, JSON.parse(readFileSync(CONFIG_PATH, 'utf-8'))); } catch (e) { console.warn('[claude] failed to parse agent-config.json:', e); }
|
|
32
|
+
}
|
|
33
|
+
return config;
|
|
34
|
+
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// API key list/create/delete, ONE implementation for both mounts: `/api/api-keys` (self) and `/api/owner/api-keys`
|
|
2
|
+
// (owner console). Gates run per route, so mounting the router never gates unrelated requests.
|
|
3
|
+
// Audit actor is always the caller's principal id. Store refusals (validation 400, PASSIVE 409) map to their status.
|
|
4
|
+
import { Router, type Request, type RequestHandler, type Response } from 'express';
|
|
5
|
+
import type { AuthUser } from './auth.ts';
|
|
6
|
+
import { apiKeyStore, ApiKeyStoreError } from './api-keys.ts';
|
|
7
|
+
import { security } from './security/runtime.ts';
|
|
8
|
+
|
|
9
|
+
export class ApiKeyRoutesOptions {
|
|
10
|
+
base: string = '/api/api-keys';
|
|
11
|
+
gate: RequestHandler[] = [];
|
|
12
|
+
/** Extra gates for create/delete (after `gate`). */
|
|
13
|
+
writeGate: RequestHandler[] = [];
|
|
14
|
+
/** Owner console: sees/deletes every key and may set `role`/`expiresAt`. Self: own keys (all when the caller is an
|
|
15
|
+
* owner, as before) and label only. */
|
|
16
|
+
asOwner: boolean = false;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
const userOf = (req: Request) => (req as any).user as AuthUser;
|
|
20
|
+
|
|
21
|
+
function fail(res: Response, e: any): void {
|
|
22
|
+
const status = e instanceof ApiKeyStoreError ? e.status : 400;
|
|
23
|
+
console.warn(`[api-keys] ${e?.message ?? e}`);
|
|
24
|
+
res.status(status).json({ error: e?.message ?? String(e) });
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export function apiKeyRouter(options?: Partial<ApiKeyRoutesOptions>): Router {
|
|
28
|
+
const { base, gate, writeGate, asOwner } = { ...new ApiKeyRoutesOptions(), ...options };
|
|
29
|
+
const router = Router();
|
|
30
|
+
const isOwner = (u: AuthUser) => asOwner || u.isOwner;
|
|
31
|
+
|
|
32
|
+
router.get(base, ...gate, (req, res) => {
|
|
33
|
+
const user = userOf(req);
|
|
34
|
+
res.json({ keys: apiKeyStore().list({ uid: user.uid, isOwner: isOwner(user) }) });
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
/** Body: { label?, role?, expiresAt? (epoch ms) } — role/expiresAt honored on the owner mount only. Plaintext returned once. */
|
|
38
|
+
router.post(base, ...gate, ...writeGate, (req, res) => {
|
|
39
|
+
const user = userOf(req);
|
|
40
|
+
// Minting a key is minting a credential AS this identity — only an interactive login may (same rule as OAuth
|
|
41
|
+
// consent). Otherwise a role-capped or expiring key could mint itself an uncapped, non-expiring one.
|
|
42
|
+
const kind = user.principal?.kind ?? 'unknown';
|
|
43
|
+
if (kind !== 'user') {
|
|
44
|
+
console.warn(`[api-keys] create refused: non-interactive credential (${kind})`);
|
|
45
|
+
security()?.authDeny(`http:${base}`, `non-interactive:${kind}`, req.ip);
|
|
46
|
+
return void res.status(403).json({ error: 'Creating an API key requires an interactive login' });
|
|
47
|
+
}
|
|
48
|
+
const { label, role, expiresAt } = (req.body ?? {}) as { label?: string; role?: string; expiresAt?: number };
|
|
49
|
+
try {
|
|
50
|
+
const opts = asOwner ? { role, expiresAt } : {};
|
|
51
|
+
res.json(apiKeyStore().create(user.uid, user.email, label || 'Unnamed', { ...opts, actor: user.principal.id }));
|
|
52
|
+
} catch (e: any) { fail(res, e); }
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
router.delete(`${base}/:id`, ...gate, ...writeGate, (req: Request<{ id: string }>, res) => {
|
|
56
|
+
const user = userOf(req);
|
|
57
|
+
try {
|
|
58
|
+
const r = apiKeyStore().delete(req.params.id, user.uid, isOwner(user), user.principal.id);
|
|
59
|
+
if (r === 'not_found') return void res.status(404).json({ error: 'Key not found' });
|
|
60
|
+
if (r === 'forbidden') return void res.status(403).json({ error: 'Cannot delete another user\'s key' });
|
|
61
|
+
res.json({ ok: true });
|
|
62
|
+
} catch (e: any) { fail(res, e); }
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
return router;
|
|
66
|
+
}
|