shraga 0.1.111 → 0.1.113

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +4 -1
  2. package/defaults/mcps/README.md +6 -3
  3. package/defaults/skills/mcp-server.md +12 -5
  4. package/defaults/skills/platform.md +4 -1
  5. package/dist/client/assets/index-DIDtPQb-.css +10 -0
  6. package/dist/client/assets/index-DJ0AgGIu.js +1969 -0
  7. package/dist/client/index.html +2 -2
  8. package/package.json +3 -2
  9. package/src/client/App.tsx +33 -8
  10. package/src/client/components/BackendStatusBanner.tsx +62 -0
  11. package/src/client/components/ConfigPanel.tsx +61 -15
  12. package/src/client/components/ConversationHeader.tsx +5 -1
  13. package/src/client/components/McpManager.tsx +26 -9
  14. package/src/client/components/SkillsManager.tsx +48 -27
  15. package/src/client/hooks/useAuth.ts +11 -2
  16. package/src/client/hooks/useIsOwner.ts +24 -0
  17. package/src/client/hooks/useModules.ts +5 -1
  18. package/src/client/lib/api.ts +21 -5
  19. package/src/client/lib/backendHealth.ts +230 -0
  20. package/src/client/lib/debug.ts +48 -0
  21. package/src/client/lib/sessionApi.ts +24 -8
  22. package/src/client/lib/ws.ts +21 -13
  23. package/src/scripts/harden-audit.sh +55 -0
  24. package/src/server/api-key-routes.ts +64 -0
  25. package/src/server/api-keys.ts +181 -43
  26. package/src/server/auth.ts +113 -47
  27. package/src/server/boot.ts +158 -104
  28. package/src/server/claude.ts +112 -4
  29. package/src/server/data-sync.ts +55 -6
  30. package/src/server/directives.ts +10 -5
  31. package/src/server/engine/claude-code.ts +190 -32
  32. package/src/server/engine/claude-resume.ts +205 -0
  33. package/src/server/engine/types.ts +10 -0
  34. package/src/server/hooks.ts +19 -0
  35. package/src/server/mcp-oauth.ts +24 -5
  36. package/src/server/mcp-server.ts +55 -25
  37. package/src/server/modules/routes.ts +2 -6
  38. package/src/server/notify-owners.ts +5 -17
  39. package/src/server/owners.ts +14 -0
  40. package/src/server/scheduler/builtins.ts +3 -1
  41. package/src/server/scheduler/runner.ts +3 -0
  42. package/src/server/security/audit.ts +498 -0
  43. package/src/server/security/enforce.ts +306 -0
  44. package/src/server/security/escalate.ts +194 -0
  45. package/src/server/security/guard.ts +329 -0
  46. package/src/server/security/owner-only.ts +15 -0
  47. package/src/server/security/owner-routes.ts +43 -0
  48. package/src/server/security/policy.ts +413 -0
  49. package/src/server/security/principal.ts +80 -0
  50. package/src/server/security/revocation.ts +50 -0
  51. package/src/server/security/runtime.ts +174 -0
  52. package/src/server/sessions.ts +51 -9
  53. package/src/server/shraga-config.ts +3 -0
  54. package/src/server/slack/bot.ts +44 -11
  55. package/src/server/slack/context-cache.ts +40 -7
  56. package/src/server/webhook-lane/feature.ts +17 -6
  57. package/src/shared/models.ts +11 -0
  58. package/dist/client/assets/index-BNAh4GUs.js +0 -1949
  59. package/dist/client/assets/index-DIMte_k6.css +0 -10
@@ -8,7 +8,12 @@ import { listSkills } from '../skills.ts';
8
8
  import { loadAgents } from '../agents.ts';
9
9
  import { registerProactiveMessage } from '../slack/sessions.ts';
10
10
  import { registerPoll } from '../polls.ts';
11
- import { getSession, setSessionModel, getSessionModel, type ConvMessage } from '../sessions.ts';
11
+ import { getSession, setSessionModel, getSessionModel, setClaudeResume, type ConvMessage } from '../sessions.ts';
12
+ import {
13
+ isResumeEnabled, decideClaudeTurn, claudeConfigDir, findClaudeTranscript, shortHash, sectionHashes, speakerKey,
14
+ sectionsAfterSubmit, isResumeFailure, conversationSummaryKey, buildContextDelta, buildResumePrompt, renderConvMessage,
15
+ type TurnPath, type ClaudeResumeState,
16
+ } from './claude-resume.ts';
12
17
  import { DEFAULT_MODEL } from '../directives.ts';
13
18
  import { resolveModelSwitch, MODEL_ALIASES } from '../model-aliases.ts';
14
19
  import type { WsEvent, AskQuestion, QuestionAnswers, QuestionHandler } from '../claude.ts';
@@ -18,6 +23,8 @@ import { APP_ROOT } from '../paths.ts';
18
23
  import { writeMcpConfigFile } from './mcp-config-file.ts';
19
24
  import { claudeUsageFor } from '../claude-usage.ts';
20
25
  import { claudeAccountDir, applyClaudeAccount, claudeAccountRef, type ClaudeAccountRef } from '../claude-account.ts';
26
+ import { buildAgentEnv, builtinTools, filterMcpServers, allowsEscalate, allowsInternalToken, SENSITIVE_PATH_PATTERNS } from '../security/enforce.ts';
27
+ import { escalateMcpServer } from '../security/escalate.ts';
21
28
  const IMMUTABLE_SYSTEM_PROMPT = readFileSync(path.resolve(import.meta.dirname, '../../../defaults/system-prompt.md'), 'utf-8');
22
29
  const DEFAULT_USER_PROMPT = `You are a helpful assistant with access to MCP tools.`;
23
30
  const DEFAULT_ALLOWED_TOOLS = ['Read', 'Edit', 'Bash', 'WebSearch', 'Glob', 'LS', 'ToolSearch'];
@@ -27,10 +34,6 @@ const HISTORY_LIMIT = 50;
27
34
 
28
35
  const NO_INTERACTIVE_ANSWER = 'No interactive channel is available to answer right now. Use your best judgement to proceed, and surface these options to the user in your reply so they can redirect if needed.';
29
36
 
30
- const SENSITIVE_PATTERNS = [
31
- /\.env($|\.)/i, /secrets?\//i, /credentials/i, /\.pem$/i, /\.key$/i,
32
- /service.account.*\.json/i, /\/\.claude\/credentials/i,
33
- ];
34
37
  const SENSITIVE_BASH_PATTERNS = [
35
38
  /\.env\b/i, /\bprintenv\b/i, /\b(env|set)\s*\|/i, /\bsecrets?\//i,
36
39
  /credentials/i, /service.account/i, /\.(pem|key)\b/i,
@@ -58,7 +61,7 @@ type DenyResult = { behavior: 'deny'; message: string };
58
61
  function checkSensitiveAccess(toolName: string, input: Record<string, unknown>): DenyResult | null {
59
62
  const filePath = (input.file_path ?? input.path ?? '') as string;
60
63
  if ((toolName === 'Read' || toolName === 'Edit' || toolName === 'Write') && filePath) {
61
- if (SENSITIVE_PATTERNS.some(p => p.test(filePath))) {
64
+ if (SENSITIVE_PATH_PATTERNS.some(p => p.test(filePath))) {
62
65
  console.log(`[security] Blocked ${toolName} on sensitive file: ${filePath}`);
63
66
  return { behavior: 'deny', message: 'Access to sensitive files (.env, secrets, credentials) is blocked.' };
64
67
  }
@@ -95,15 +98,8 @@ function buildHistoryPrompt(conv: ConvMessage[], contextBlock: string, userPromp
95
98
  const parts: string[] = [];
96
99
  if (summary) parts.push(`<conversation_summary>\n${summary}\n</conversation_summary>`);
97
100
  for (const m of recent) {
98
- const role = m.role === 'user' ? 'User' : 'Assistant';
99
- const texts = m.blocks
100
- .filter((b) => b.type === 'text' || b.type === 'context')
101
- .map((b) => {
102
- if (b.type === 'context') return `[${(b as any).label}]: ${(b as any).text}`;
103
- return (b as { type: 'text'; text: string }).text;
104
- })
105
- .filter(Boolean);
106
- if (texts.length) parts.push(`${role}: ${texts.join('\n')}`);
101
+ const line = renderConvMessage(m);
102
+ if (line) parts.push(line);
107
103
  }
108
104
 
109
105
  if (parts.length) {
@@ -117,10 +113,11 @@ function buildHistoryPrompt(conv: ConvMessage[], contextBlock: string, userPromp
117
113
  * Log prompt-cache effectiveness from the SDK result `usage`. The hit rate is
118
114
  * cache_read / (cache_read + cache_creation + uncached input) — a low rate over
119
115
  * many turns points to a silent prefix invalidator or sessions spread past the
120
- * 5-min cache TTL. Note: cross-turn history is re-sent uncached (single-shot
121
- * prompt per query, no SDK resume) — so hit rate tracks tool density per turn.
116
+ * 5-min cache TTL. `path` says how the prompt was built — `fresh` (history re-sent in one message),
117
+ * `resume` (SDK session resumed, only the new message sent) or `fallback:<reason>` (resume enabled but a
118
+ * fresh query ran) — so journal lines can be A/B-compared per path.
122
119
  */
123
- function logCacheUsage(usage: any, model: string): void {
120
+ function logCacheUsage(usage: any, model: string, path: TurnPath): void {
124
121
  if (!usage) return;
125
122
  const read = usage.cache_read_input_tokens ?? 0;
126
123
  const created = usage.cache_creation_input_tokens ?? 0;
@@ -128,9 +125,47 @@ function logCacheUsage(usage: any, model: string): void {
128
125
  const totalIn = read + created + fresh;
129
126
  if (totalIn === 0) return;
130
127
  const hitRate = ((read / totalIn) * 100).toFixed(1);
131
- console.log(`[claude] Cache: hit=${hitRate}% read=${read} write=${created} uncached=${fresh} out=${usage.output_tokens ?? 0} model=${model}`);
128
+ console.log(`[claude] Cache: hit=${hitRate}% read=${read} write=${created} uncached=${fresh} out=${usage.output_tokens ?? 0} model=${model} path=${path}`);
129
+ }
130
+
131
+ /** How one query attempt is built. `persist` is set when resume is enabled: after a query that produced
132
+ * output, the CC session id + these facts are saved so the next turn can resume. */
133
+ interface RunPlan {
134
+ path: TurnPath;
135
+ prompt: string;
136
+ accountDir: string | null;
137
+ resumeId?: string;
138
+ persist?: Omit<ClaudeResumeState, 'claudeSessionId' | 'model' | 'interruptedBy'>;
139
+ /** Resume only: section hashes to store once the prompt is submitted (see sectionsAfterSubmit). */
140
+ submitSections?: Record<string, string>;
141
+ }
142
+
143
+ /** Exit promises of CLI processes still running, per shraga session. A turn that took over a session (external
144
+ * steer) can start while the aborted run's CLI is still exiting; resuming then would put two writers on one
145
+ * transcript. */
146
+ const liveCli = new Map<string, Set<Promise<void>>>();
147
+
148
+ function trackCliExit(sessionId: string | undefined, child: { once: (event: string, fn: () => void) => unknown }): void {
149
+ if (!sessionId) return;
150
+ const set = liveCli.get(sessionId) ?? new Set();
151
+ liveCli.set(sessionId, set);
152
+ const exited = new Promise<void>((resolve) => { child.once('exit', resolve); child.once('error', resolve); });
153
+ set.add(exited);
154
+ void exited.then(() => { set.delete(exited); if (!set.size && liveCli.get(sessionId) === set) liveCli.delete(sessionId); });
155
+ }
156
+
157
+ /** true once every CLI process of the session has exited, false if one is still alive after `ms`. */
158
+ async function cliExited(sessionId: string, ms: number): Promise<boolean> {
159
+ const set = liveCli.get(sessionId);
160
+ if (!set?.size) return true;
161
+ let timer: ReturnType<typeof setTimeout> | undefined;
162
+ const timeout = new Promise<false>((r) => { timer = setTimeout(() => r(false), ms); });
163
+ try { return await Promise.race([Promise.all(set).then(() => true), timeout]); } finally { clearTimeout(timer); }
132
164
  }
133
165
 
166
+ /** Events that prove a resumed query actually runs; until one arrives the attempt can still be retried fresh unseen. */
167
+ const LIVE_EVENTS = new Set<WsEvent['type']>(['text_delta', 'thinking_delta', 'tool_use', 'tool_use_input', 'tool_result', 'tool_result_image', 'done']);
168
+
134
169
  /** SDK spawn hook: same call the SDK would make, plus `detached` (own process group). See the
135
170
  * call site for why. Shape mirrors the SDK's own spawnLocalProcess return. */
136
171
  function spawnDetached(cfg: { command: string; args: string[]; cwd?: string; env: Record<string, string | undefined>; signal?: AbortSignal }) {
@@ -205,6 +240,7 @@ async function* buildLegacyImagePrompt(text: string, images: string[], sessionId
205
240
 
206
241
  export class ClaudeCodeEngine implements AgentEngine {
207
242
  readonly name = 'claude-code';
243
+ readonly enforcesProfile = true;
208
244
 
209
245
  getModels(): EngineModel[] {
210
246
  return [
@@ -221,15 +257,82 @@ export class ClaudeCodeEngine implements AgentEngine {
221
257
  ];
222
258
  }
223
259
 
260
+ /** How long a resume-enabled turn waits for an earlier CLI process on the session to exit before going fresh. */
261
+ static cliExitWaitMs = 5_000;
262
+
224
263
  async *stream(opts: EngineStreamOpts): AsyncGenerator<WsEvent> {
264
+ let cliAlive = false;
265
+ if (opts.sessionId && isResumeEnabled(opts.directives, opts.config) && !(await cliExited(opts.sessionId, ClaudeCodeEngine.cliExitWaitMs))) {
266
+ cliAlive = true;
267
+ console.warn(`[claude] An earlier CLI process on session=${opts.sessionId} is still running after ${ClaudeCodeEngine.cliExitWaitMs}ms — not resuming its transcript`);
268
+ }
269
+ const plan = this.plan(opts, cliAlive);
270
+ if (plan.path !== 'resume') { yield* this.run(opts, plan); return; }
271
+ // Hold non-output events (init's model_resolved / switch notice) until the resumed query proves it
272
+ // runs, so a failed resume can be retried fresh with nothing duplicated on the user's side.
273
+ const attempt = this.run(opts, plan);
274
+ const held: WsEvent[] = [];
275
+ let live = false;
276
+ let r: IteratorResult<WsEvent, 'resume-failed' | void>;
277
+ try {
278
+ while (!(r = await attempt.next()).done) {
279
+ if (!live && !LIVE_EVENTS.has(r.value.type)) { held.push(r.value); continue; }
280
+ if (!live) { live = true; yield* held; }
281
+ yield r.value;
282
+ }
283
+ } finally {
284
+ await attempt.return(undefined);
285
+ }
286
+ if (r.value !== 'resume-failed') { if (!live) yield* held; return; }
287
+ console.warn(`[claude] Resume of ${plan.resumeId} failed before any output — retrying fresh in the same turn (session=${opts.sessionId})`);
288
+ setClaudeResume(opts.sessionId, undefined);
289
+ yield* this.run(opts, this.freshPlan(opts, plan.accountDir, 'fallback:resume-failed', plan.persist));
290
+ }
291
+
292
+ private freshPlan(opts: EngineStreamOpts, accountDir: string | null, path: TurnPath, persist?: RunPlan['persist']): RunPlan {
293
+ return {
294
+ path, accountDir,
295
+ prompt: buildHistoryPrompt(opts.conversation, opts.contextBlock, opts.prompt),
296
+ ...(persist ? { persist: { ...persist, sections: sectionHashes(opts.contextSections ?? { context: opts.contextBlock }), startedAt: Date.now() } } : {}),
297
+ };
298
+ }
299
+
300
+ /** Resume vs fresh, and why — see claude-resume.ts. */
301
+ private plan(opts: EngineStreamOpts, cliAlive: boolean): RunPlan {
302
+ const accountDir = claudeAccountDir(opts.userEmail);
303
+ if (!isResumeEnabled(opts.directives, opts.config)) return this.freshPlan(opts, accountDir, 'fresh');
304
+ const configDir = claudeConfigDir(accountDir);
305
+ const configDirHash = shortHash(configDir);
306
+ const state = opts.sessionId ? getSession(opts.sessionId)?.claudeResume : undefined;
307
+ const speaker = speakerKey(opts.uid, opts.userEmail);
308
+ const persist = { configDirHash, speaker, markId: opts.conversation.at(-1)?.id, summaryKey: conversationSummaryKey(opts.conversation), sections: {}, startedAt: Date.now() };
309
+ const decision = decideClaudeTurn({
310
+ enabled: true, state, conversation: opts.conversation, configDirHash, speaker, cliAlive, conversationReset: opts.conversationReset,
311
+ hasTranscript: (id) => !!findClaudeTranscript(id, configDir),
312
+ });
313
+ if (decision.path !== 'resume' || !state) return this.freshPlan(opts, accountDir, decision.path, persist);
314
+ const sections = opts.contextSections ?? { context: opts.contextBlock };
315
+ const hashes = sectionHashes(sections);
316
+ return {
317
+ path: 'resume', accountDir, resumeId: state.claudeSessionId,
318
+ prompt: buildResumePrompt(opts.prompt, decision.unseen, buildContextDelta(state.sections, sections)),
319
+ persist: { ...persist, sections: hashes, startedAt: state.startedAt },
320
+ submitSections: sectionsAfterSubmit(state.sections, hashes),
321
+ };
322
+ }
323
+
324
+ private async *run(opts: EngineStreamOpts, plan: RunPlan): AsyncGenerator<WsEvent, 'resume-failed' | void> {
225
325
  const { config, directives } = opts;
226
326
  const cwd = APP_ROOT;
227
327
 
228
- const fullPrompt = buildHistoryPrompt(opts.conversation, opts.contextBlock, opts.prompt);
328
+ const fullPrompt = plan.prompt;
229
329
  const permMode = opts.onPermissionRequest ? 'default' : (config.permissionMode ?? 'acceptEdits');
230
330
 
231
- const sdkEnv: Record<string, string> = {};
232
- for (const [k, v] of Object.entries(process.env)) {
331
+ // SECURITY_ENFORCE: the effective profile at spawn decides env/tools/MCPs; the per-call gate re-reads it (enforce.ts).
332
+ const guard = opts.security;
333
+ const eff = guard?.current();
334
+ const sdkEnv: Record<string, string> = eff ? buildAgentEnv(process.env, eff.profile) : {};
335
+ if (!eff) for (const [k, v] of Object.entries(process.env)) {
233
336
  if (v !== undefined) sdkEnv[k] = v;
234
337
  }
235
338
  // Injected for the agent's own tools/scripts. Each is written under both the canonical
@@ -251,7 +354,7 @@ export class ClaudeCodeEngine implements AgentEngine {
251
354
  // regardless of settingSources; headless they are auth stubs or self-recursion. Our MCPs come from mcp-config.
252
355
  setIfBlank('ENABLE_CLAUDEAI_MCP_SERVERS', 'false');
253
356
  // Per-user subscription (workspace/users/<contactId>/.claude): run on that login, never the box's credentials.
254
- const accountDir = claudeAccountDir(opts.userEmail);
357
+ const accountDir = plan.accountDir;
255
358
  if (accountDir) applyClaudeAccount(sdkEnv, accountDir);
256
359
  // Which login this run is on (email/plan, never a token) — read lazily from the login's local
257
360
  // files, only once init proves the run is on a subscription login (see init below).
@@ -259,14 +362,24 @@ export class ClaudeCodeEngine implements AgentEngine {
259
362
  const accountRef = () => (accountRefP ??= claudeAccountRef(accountDir));
260
363
  let runAccount: ClaudeAccountRef | undefined;
261
364
  let sawInit = false;
365
+ // Did THIS attempt produce model output? Gates both the resume-failure retry and saving the session id
366
+ // (a resume that died on "No conversation found" can report a session id it never wrote a turn to).
367
+ let sawOutput = false;
368
+ let initModel: string | undefined;
262
369
  sdkEnv.INTERNAL_API_TOKEN = signInternalToken(opts.uid, opts.userEmail || 'unknown');
263
370
 
371
+ if (eff && !allowsInternalToken(eff.profile)) delete sdkEnv.INTERNAL_API_TOKEN;
372
+
264
373
  const baseAllowed = config.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
265
- const allowedTools = baseAllowed.includes('ToolSearch') ? baseAllowed : [...baseAllowed, 'ToolSearch'];
374
+ const withToolSearch = baseAllowed.includes('ToolSearch') ? baseAllowed : [...baseAllowed, 'ToolSearch'];
375
+ // Enforced: built-in tool AVAILABILITY is the profile's (tools outside it are not in the model's context), and
376
+ // auto-approval never widens it.
377
+ const available = eff ? builtinTools(eff.profile) : 'all';
378
+ const allowedTools = available === 'all' ? withToolSearch : withToolSearch.filter((t) => available.includes(t));
266
379
  const maxTurns = directives.turns ?? config.maxTurns ?? 50;
267
380
 
268
381
  const options: Record<string, unknown> = {
269
- tools: { type: 'preset', preset: 'claude_code' },
382
+ tools: available === 'all' ? { type: 'preset', preset: 'claude_code' } : available,
270
383
  env: sdkEnv,
271
384
  allowedTools,
272
385
  cwd,
@@ -286,15 +399,24 @@ export class ClaudeCodeEngine implements AgentEngine {
286
399
  skills: listSkills(),
287
400
  disallowedTools: ['Skill'],
288
401
  agents: loadAgents(),
289
- hooks: buildHooks({ exemptBashCommand: opts.foregroundBashCommand }),
402
+ // Enforced: the profile gate runs FIRST on every tool call — hooks fire even for auto-approved `allowedTools`,
403
+ // which never reach canUseTool.
404
+ hooks: ((h) => (guard ? { ...h, PreToolUse: [{ hooks: [guard.hook()] }, ...(h.PreToolUse ?? [])] } : h))(buildHooks({ exemptBashCommand: opts.foregroundBashCommand })),
290
405
  };
291
406
 
292
407
  const userHandler = opts.onPermissionRequest;
293
408
  const destructiveHandler = opts.onDestructiveApproval;
294
409
  const questionHandler = opts.onUserQuestion;
295
410
  options['canUseTool'] = async (toolName: string, input: Record<string, unknown>) => {
296
- const denied = checkSensitiveAccess(toolName, input);
411
+ // Enforced: TurnGuard owns the file-path deny (tightened list); the legacy path list stays flag-OFF behavior only.
412
+ const denied = guard && toolName !== 'Bash' ? null : checkSensitiveAccess(toolName, input);
297
413
  if (denied) return denied;
414
+ // Enforced: the profile gate (re-reads the session floor) runs before ANY handler below, so an
415
+ // `onPermissionRequest: allow` call site can never override a profile deny.
416
+ if (guard) {
417
+ const gate = guard.check(toolName, input, cwd);
418
+ if (!gate.allow) return { behavior: 'deny' as const, message: gate.message };
419
+ }
298
420
  if (toolName === 'AskUserQuestion') {
299
421
  const questions = (input.questions ?? []) as AskQuestion[];
300
422
  const answers = questionHandler
@@ -357,24 +479,42 @@ export class ClaudeCodeEngine implements AgentEngine {
357
479
  const addonSuffix = getPromptSuffix(opts.turnHints);
358
480
  options['systemPrompt'] = `${IMMUTABLE_SYSTEM_PROMPT}\n\n${userPrompt}${addonSuffix ? `\n\n${addonSuffix}` : ''}`;
359
481
  if (opts.abortController) options['abortController'] = opts.abortController;
482
+ if (plan.resumeId) options['resume'] = plan.resumeId;
360
483
  // Spawn the CLI in its OWN process group. The service manager signals the whole JOB on restart
361
484
  // (`launchctl kickstart -k`, `systemctl restart`), so an inherited process group means the child
362
485
  // dies instantly with SIGTERM — surfacing mid-reply as `exited with code 143` and making the
363
486
  // server's 90s drain (gracefulShutdown) protect nothing: it only ever waited for a turn that was
364
487
  // already dead. Detached, the signal reaches the server alone and the drain can finish the turn.
365
488
  // Teardown is unaffected: the SDK still kills the child on abort/close and on process exit.
366
- options['spawnClaudeCodeProcess'] = spawnDetached;
489
+ options['spawnClaudeCodeProcess'] = (cfg: Parameters<typeof spawnDetached>[0]) => {
490
+ const child = spawnDetached(cfg);
491
+ trackCliExit(opts.sessionId, child);
492
+ return child;
493
+ };
367
494
 
368
495
  // Passed as a FILE, never as `options.mcpServers` — the SDK would put the whole config (every MCP
369
496
  // server's credentials) on the CLI's argv, where `ps` / `/proc` / journald expose it. See
370
497
  // writeMcpConfigFile. Setting both would re-add the argv copy, so it's one or the other.
371
498
  let mcpConfigFile: { path: string; cleanup: () => void } | undefined;
372
- if (opts.mcpServers && Object.keys(opts.mcpServers).length > 0) {
373
- mcpConfigFile = writeMcpConfigFile(opts.mcpServers);
499
+ // Enforced: servers outside profile.mcps never reach the file, so the CLI never starts them.
500
+ const mcpServers = eff ? filterMcpServers(opts.mcpServers, eff.profile) : opts.mcpServers;
501
+ if (mcpServers && Object.keys(mcpServers).length > 0) {
502
+ mcpConfigFile = writeMcpConfigFile(mcpServers);
374
503
  options['extraArgs'] = { ...(options['extraArgs'] as Record<string, string> | undefined), 'mcp-config': mcpConfigFile.path };
375
504
  }
505
+ // `escalate` is an in-process SDK server (no credentials): `options.mcpServers` hands the CLI only its name.
506
+ if (guard && eff && allowsEscalate(eff.profile)) {
507
+ const { principal } = guard.options;
508
+ const lastUser = opts.conversation.findLast((m) => m.role === 'user');
509
+ const excerpt = lastUser?.blocks.map((b) => (b.type === 'text' ? b.text : '')).filter(Boolean).join('\n') || opts.prompt;
510
+ const server = escalateMcpServer({
511
+ principal, role: () => guard.current().role, sessionId: opts.sessionId, excerpt,
512
+ channel: [principal.kind, principal.attrs.lane, opts.context?.source].filter(Boolean).join(':'),
513
+ });
514
+ options['mcpServers'] = { [server.name]: server };
515
+ }
376
516
 
377
- const mcpNames = opts.mcpServers ? Object.keys(opts.mcpServers) : [];
517
+ const mcpNames = mcpServers ? Object.keys(mcpServers) : [];
378
518
  const activeModel = (options['model'] as string) || 'default';
379
519
  const directivesTag = Object.keys(directives).length ? ` directives=${JSON.stringify(directives)}` : '';
380
520
  console.log(`[claude] Starting query user=${opts.uid} session=${opts.sessionId ?? 'new'} model=${activeModel} perms=${config.permissionMode} mcps=[${mcpNames.join(',')}]${directivesTag} cwd=${cwd}`);
@@ -391,6 +531,8 @@ export class ClaudeCodeEngine implements AgentEngine {
391
531
  : fullPrompt;
392
532
 
393
533
  const q = query({ prompt, options: options as any });
534
+ // The CLI writes this prompt to the transcript before any output: from here a cancel can't un-deliver it.
535
+ if (plan.submitSections && opts.sessionId) setClaudeResume(opts.sessionId, undefined, { sections: plan.submitSections });
394
536
 
395
537
  let lastSessionId = '';
396
538
  let messageCount = 0;
@@ -449,6 +591,7 @@ export class ClaudeCodeEngine implements AgentEngine {
449
591
  // default only counts when the SDK says it resolved a login, not an API key.
450
592
  if ((authSource ?? (accountDir ? 'subscription' : undefined)) === 'subscription') runAccount = await accountRef();
451
593
  if (m.model) {
594
+ initModel = m.model;
452
595
  console.log(`[claude] Init model=${m.model}${m.model !== options['model'] ? ` (requested ${options['model']})` : ''}`);
453
596
  // If the user explicitly asked to switch models via a [directive], announce the change
454
597
  // inline so the response confirms the switch took effect.
@@ -494,7 +637,7 @@ export class ClaudeCodeEngine implements AgentEngine {
494
637
  const raw = m.subtype ?? 'unknown';
495
638
  const sub = raw === 'error_max_turns' || (raw === 'end_turn' && sdkTurns >= maxTurns) ? 'max_turns_reached' : raw;
496
639
  console.log(`[claude] Result: subtype=${m.subtype}→${sub} session=${lastSessionId} turns=${sdkTurns}/${maxTurns} cost=$${m.total_cost_usd?.toFixed(4) ?? '?'} msgs=${messageCount} deltas=${textDeltaCount} (${elapsed()})`);
497
- logCacheUsage(m.usage, activeModel);
640
+ logCacheUsage(m.usage, activeModel, plan.path);
498
641
  if (outstandingTasks.size > 0) {
499
642
  if (!waitingForBg) {
500
643
  waitingForBg = true; clearBgTimer();
@@ -514,6 +657,12 @@ export class ClaudeCodeEngine implements AgentEngine {
514
657
  // `error_max_turns` also sets is_error but is a benign stop reason we already model.
515
658
  if ((m.is_error || m.api_error_status) && sub !== 'max_turns_reached') {
516
659
  const detail = typeof m.result === 'string' && m.result ? m.result : `api_error_status=${m.api_error_status ?? '?'}`;
660
+ // The CLI's reason ("No conversation found with session ID …") rides in `errors`, not `result`.
661
+ const reasons = Array.isArray(m.errors) && m.errors.length ? m.errors.join('; ') : detail;
662
+ if (plan.resumeId && !sawOutput && isResumeFailure(reasons)) {
663
+ console.error(`[claude] Resume error result (subtype=${raw}) — ${reasons}`);
664
+ return 'resume-failed';
665
+ }
517
666
  // A usage/rate-limit failure is about ONE login's quota — name it, or a routed user's own
518
667
  // exhausted Pro plan reads as the shared agent subscription being out.
519
668
  const limitHit = m.api_error_status === 429 || /\b(usage|rate.?limit(ed)?|hit your limit)\b/i.test(detail);
@@ -532,10 +681,12 @@ export class ClaudeCodeEngine implements AgentEngine {
532
681
  if (m.type === 'stream_event') {
533
682
  const event = m.event;
534
683
  if (event?.type === 'content_block_delta' && event.delta?.type === 'thinking_delta') {
684
+ sawOutput = true;
535
685
  yield { type: 'thinking_delta', text: event.delta.thinking };
536
686
  }
537
687
  if (event?.type === 'content_block_delta' && event.delta?.type === 'text_delta') {
538
688
  textDeltaCount++;
689
+ sawOutput = true;
539
690
  yield { type: 'text_delta', text: event.delta.text };
540
691
  }
541
692
  if (event?.type === 'content_block_start' && event.content_block?.type === 'tool_use') {
@@ -557,6 +708,7 @@ export class ClaudeCodeEngine implements AgentEngine {
557
708
 
558
709
  if (m.type === 'assistant' && Array.isArray(m.message?.content)) {
559
710
  turnCount++;
711
+ sawOutput = true;
560
712
  for (const block of m.message.content) {
561
713
  if (block.type === 'tool_use') {
562
714
  pendingToolUses.set(block.id, { tool: block.name, input: block.input });
@@ -645,6 +797,7 @@ export class ClaudeCodeEngine implements AgentEngine {
645
797
  return;
646
798
  }
647
799
  console.error(`[claude] Error after ${messageCount} msgs (${elapsed()}):`, err.message || err);
800
+ if (plan.resumeId && !sawOutput && isResumeFailure(err.message || String(err))) return 'resume-failed';
648
801
  yield { type: 'error', message: err.message || String(err) };
649
802
  return;
650
803
  } finally {
@@ -652,6 +805,11 @@ export class ClaudeCodeEngine implements AgentEngine {
652
805
  // The CLI has read the file by now (it loads MCP config at startup); holding it any longer just
653
806
  // widens the window in which the credentials sit on disk.
654
807
  mcpConfigFile?.cleanup();
808
+ // Save the mapping once this attempt really ran a turn (incl. an aborted one — the transcript holds it).
809
+ if (plan.persist && opts.sessionId && lastSessionId && sawOutput) {
810
+ setClaudeResume(opts.sessionId, { ...plan.persist, claudeSessionId: lastSessionId, ...(initModel ? { model: initModel } : {}) });
811
+ if (plan.resumeId && plan.resumeId !== lastSessionId) console.warn(`[claude] Resume returned a new session id ${lastSessionId} (was ${plan.resumeId})`);
812
+ }
655
813
  }
656
814
 
657
815
  const inferredReason = turnCount >= maxTurns ? 'max_turns_reached' : 'end_turn';
@@ -0,0 +1,205 @@
1
+ /**
2
+ * Opt-in SDK session resume for the claude-code engine — the pure half (decision + prompt building),
3
+ * kept free of SDK/IO wiring so every fallback reason is unit-testable.
4
+ *
5
+ * Fresh turn (today's path): ONE user message = contextBlock + <conversation_history> + new message, so
6
+ * the whole history is re-written to prompt cache every turn. Resume turn: `resume: <claudeSessionId>`
7
+ * and send only the new message; the stable context went in once on the session's fresh query, and
8
+ * whatever CHANGED since (context sections, messages other channels appended) plus the current speaker's
9
+ * `user` section rides AFTER the user text so the cached transcript prefix stays byte-identical. Only one
10
+ * speaker's turns ever share a CC session (speaker-change).
11
+ *
12
+ * Flag: `directives.resume` (per conversation: `[resume:on]` or PUT /api/sessions/:id/directives) over
13
+ * `agent-config.json` `sdkResume` (global, re-read every turn). Default OFF.
14
+ */
15
+ import { createHash } from 'node:crypto';
16
+ import { existsSync, readdirSync } from 'node:fs';
17
+ import { homedir } from 'node:os';
18
+ import path from 'node:path';
19
+ import type { ConvMessage } from '../sessions.ts';
20
+ import type { Directives } from '../directives.ts';
21
+ import type { AgentSettings } from '../shraga-config.ts';
22
+
23
+ /** Persisted on SessionMeta.claudeResume — the facts needed to decide whether the stored CC session
24
+ * may be resumed. Hashes, not content: this lives in the 9 MB sessions index. */
25
+ export interface ClaudeResumeState {
26
+ /** Claude Code session id (SDK init/result `session_id`); resume keeps it stable across turns. */
27
+ claudeSessionId: string;
28
+ /** Hash of the CLAUDE_CONFIG_DIR the transcript was written under (per-user login or box default). */
29
+ configDirHash: string;
30
+ /** speakerKey() of the person whose turns this CC session holds. Another speaker never resumes it: the
31
+ * transcript carries this speaker's private user context (learnings/corrections). */
32
+ speaker?: string;
33
+ /** Model the last turn resolved (informational — CC resumes across models; only the cache is per-model). */
34
+ model?: string;
35
+ /** When this CC session was started (first fresh query). */
36
+ startedAt: number;
37
+ /** Id of the last conversation message at the start of the last turn; messages after it are what CC hasn't seen. */
38
+ markId?: string;
39
+ /** Identity of the newest shraga summary / compact marker when the state was saved. */
40
+ summaryKey: string;
41
+ /** Section name → hash of the context CC has already been given. */
42
+ sections: Record<string, string>;
43
+ /** Set by the core when another engine ran a turn on this session (its turns aren't in the CC transcript). */
44
+ interruptedBy?: string;
45
+ }
46
+
47
+ export type TurnPath = 'resume' | 'fresh' | `fallback:${string}`;
48
+
49
+ /** Unseen out-of-band text above this size means resume would re-send a history anyway — go fresh. */
50
+ export const MAX_UNSEEN_CHARS = 30_000;
51
+
52
+ export function isResumeEnabled(directives: Directives, config: AgentSettings): boolean {
53
+ // The per-session directives endpoint stores passthrough values opaquely, so accept string forms too.
54
+ const v = (directives.resume ?? config.sdkResume) as unknown;
55
+ return v === true || v === 'on' || v === 'true';
56
+ }
57
+
58
+ export function shortHash(text: string): string {
59
+ return createHash('sha256').update(text).digest('hex').slice(0, 16);
60
+ }
61
+
62
+ /** Who is speaking, as a hash (the sessions index must not gain raw emails). */
63
+ export function speakerKey(uid: string, email?: string): string {
64
+ return shortHash(`${uid}\n${email?.trim().toLowerCase() ?? ''}`);
65
+ }
66
+
67
+ /** Errors that mean THIS resume can't run (the stored transcript is gone/unreadable) — worth one fresh retry.
68
+ * Anything else (quota, auth, API, process crash) would fail a fresh query the same way: surface it as-is. */
69
+ export function isResumeFailure(text: string): boolean {
70
+ return /No conversation found|\b(session|transcript)\b.{0,60}\b(not found|missing|corrupt|invalid)\b/i.test(text);
71
+ }
72
+
73
+ /** Where the CLI keeps transcripts for a run: the routed per-user login, else the process's config dir. */
74
+ export function claudeConfigDir(accountDir: string | null | undefined): string {
75
+ return accountDir || process.env.CLAUDE_CONFIG_DIR?.trim() || path.join(homedir(), '.claude');
76
+ }
77
+
78
+ /** `<configDir>/projects/<cwd-slug>/<id>.jsonl`, found by scan rather than by re-deriving the CLI's slug
79
+ * rule (a wrong guess would silently turn every resume into a fallback). */
80
+ export function findClaudeTranscript(sessionId: string, configDir: string): string | null {
81
+ const projects = path.join(configDir, 'projects');
82
+ if (!existsSync(projects)) return null;
83
+ for (const dir of readdirSync(projects, { withFileTypes: true })) {
84
+ if (!dir.isDirectory()) continue;
85
+ const file = path.join(projects, dir.name, `${sessionId}.jsonl`);
86
+ if (existsSync(file)) return file;
87
+ }
88
+ return null;
89
+ }
90
+
91
+ /** One conversation message as history text — shared by the fresh history prompt and the resume tail. */
92
+ export function renderConvMessage(m: ConvMessage): string | null {
93
+ const texts = m.blocks
94
+ .filter((b) => b.type === 'text' || b.type === 'context')
95
+ .map((b) => (b.type === 'context' ? `[${b.label}]: ${b.text}` : (b as { text: string }).text))
96
+ .filter(Boolean);
97
+ return texts.length ? `${m.role === 'user' ? 'User' : 'Assistant'}: ${texts.join('\n')}` : null;
98
+ }
99
+
100
+ /** Changes whenever maybeCompact writes a summary or /compact adds a marker (applyCompactMarkers
101
+ * turns the latter into a synthetic leading message). */
102
+ export function conversationSummaryKey(conv: ConvMessage[]): string {
103
+ for (let i = conv.length - 1; i >= 0; i--) {
104
+ const s = conv[i].blocks.find((b) => b.type === 'summary') as { compactedCount: number } | undefined;
105
+ if (s) return `summary:${s.compactedCount}`;
106
+ }
107
+ const first = conv[0];
108
+ if (first?.id === 'compact-summary') return `marker:${shortHash(renderConvMessage(first) ?? '')}`;
109
+ return '';
110
+ }
111
+
112
+ /**
113
+ * Messages CC has not seen: everything after the mark, minus the previous turn's own reply (the first
114
+ * message, when it is the assistant's) and this turn's prompt (the last, when it is a user message —
115
+ * every channel appends it before streaming). null = the mark is gone (history was rewritten).
116
+ */
117
+ export function unseenMessages(conv: ConvMessage[], markId: string | undefined): ConvMessage[] | null {
118
+ if (!markId) return null;
119
+ const idx = conv.findLastIndex((m) => m.id === markId);
120
+ if (idx < 0) return null;
121
+ const after = conv.slice(idx + 1);
122
+ if (after[0]?.role === 'assistant') after.shift();
123
+ if (after.at(-1)?.role === 'user') after.pop();
124
+ return after;
125
+ }
126
+
127
+ export interface TurnDecisionInput {
128
+ enabled: boolean;
129
+ state?: ClaudeResumeState;
130
+ conversation: ConvMessage[];
131
+ configDirHash: string;
132
+ speaker: string;
133
+ /** A CLI process from an earlier run on this session is still alive (e.g. the run a steer took over). */
134
+ cliAlive?: boolean;
135
+ conversationReset?: boolean;
136
+ hasTranscript: (claudeSessionId: string) => boolean;
137
+ }
138
+
139
+ export function decideClaudeTurn(i: TurnDecisionInput): { path: TurnPath; unseen: ConvMessage[] } {
140
+ const fallback = (reason: string) => ({ path: `fallback:${reason}` as TurnPath, unseen: [] });
141
+ if (!i.enabled) return { path: 'fresh', unseen: [] };
142
+ const s = i.state;
143
+ if (!s?.claudeSessionId) return fallback('no-session');
144
+ // Two CLI processes appending to one transcript can interleave it; the takeover turn starts its own.
145
+ if (i.cliAlive) return fallback('concurrent-run');
146
+ if (s.speaker !== i.speaker) return fallback('speaker-change');
147
+ if (s.interruptedBy) return fallback('engine-switch');
148
+ if (s.configDirHash !== i.configDirHash) return fallback('account-change');
149
+ if (i.conversationReset) return fallback('reset');
150
+ if (s.summaryKey !== conversationSummaryKey(i.conversation)) return fallback('summary');
151
+ const unseen = unseenMessages(i.conversation, s.markId);
152
+ if (!unseen) return fallback('history-diverged');
153
+ if (unseen.reduce((n, m) => n + (renderConvMessage(m)?.length ?? 0), 0) > MAX_UNSEEN_CHARS) return fallback('drift');
154
+ // Checked last (filesystem): the CLI's periodic cleanup (cleanupPeriodDays) or a moved config dir
155
+ // deletes transcripts under sessions we still hold — catch it before spawning a doomed resume.
156
+ if (!i.hasTranscript(s.claudeSessionId)) return fallback('transcript-missing');
157
+ return { path: 'resume', unseen };
158
+ }
159
+
160
+ export function sectionHashes(sections: Record<string, string>): Record<string, string> {
161
+ const out: Record<string, string> = {};
162
+ for (const [k, v] of Object.entries(sections)) if (v) out[k] = shortHash(v);
163
+ return out;
164
+ }
165
+
166
+ /** Re-sent on every resume turn regardless of hashes: who is speaking (identity + role) must never depend on
167
+ * bookkeeping about what an earlier, possibly cancelled, turn managed to deliver. Small. */
168
+ export const ALWAYS_SENT_SECTIONS = ['user'];
169
+
170
+ /** Hash marking a section whose delivery is unknown — never equals a real hash, so the next turn re-sends it. */
171
+ const UNKNOWN = '?';
172
+
173
+ /**
174
+ * Section hashes to store the moment a resume prompt is SUBMITTED. The CLI writes the prompt to the
175
+ * transcript before any output, so a turn cancelled early may or may not have delivered its delta. Unchanged
176
+ * sections are known either way; changed, new and removed ones are marked unknown and re-sent next turn.
177
+ */
178
+ export function sectionsAfterSubmit(prev: Record<string, string>, next: Record<string, string>): Record<string, string> {
179
+ const out: Record<string, string> = {};
180
+ for (const k of new Set([...Object.keys(prev), ...Object.keys(next)])) out[k] = prev[k] === next[k] ? prev[k] : UNKNOWN;
181
+ return out;
182
+ }
183
+
184
+ /** The context sections that changed since CC last saw them (new or edited) plus ALWAYS_SENT_SECTIONS, and a
185
+ * note for sections that no longer apply. '' when there is nothing to send. */
186
+ export function buildContextDelta(prev: Record<string, string>, sections: Record<string, string>): string {
187
+ const changed = Object.entries(sections).filter(([k, v]) => v && (ALWAYS_SENT_SECTIONS.includes(k) || prev[k] !== shortHash(v)));
188
+ const removed = Object.keys(prev).filter((k) => !sections[k]);
189
+ if (!changed.length && !removed.length) return '';
190
+ return [
191
+ '<context_update>',
192
+ 'Current context for this turn. Each section below replaces its earlier version.',
193
+ ...changed.map(([k, v]) => `<section name="${k}">\n${v}\n</section>`),
194
+ ...(removed.length ? [`No longer applicable: ${removed.join(', ')}`] : []),
195
+ '</context_update>',
196
+ ].join('\n');
197
+ }
198
+
199
+ export function buildResumePrompt(prompt: string, unseen: ConvMessage[], delta: string): string {
200
+ const lines = unseen.map(renderConvMessage).filter(Boolean);
201
+ const unseenBlock = lines.length
202
+ ? `<messages_since_last_turn>\nAdded to this conversation since your last reply (other participants, channel context, notices):\n${lines.join('\n\n')}\n</messages_since_last_turn>`
203
+ : '';
204
+ return [prompt, unseenBlock, delta].filter(Boolean).join('\n\n');
205
+ }
@@ -3,6 +3,7 @@ import type { WsEvent, AttachmentMeta, PermissionHandler, QuestionHandler } from
3
3
  import type { ConvMessage } from '../sessions.ts';
4
4
  import type { Directives } from '../directives.ts';
5
5
  import type { AgentSettings } from '../shraga-config.ts';
6
+ import type { TurnGuard } from '../security/enforce.ts';
6
7
 
7
8
  export interface EngineStreamOpts {
8
9
  /** The user's effective prompt (after directive stripping, slash command expansion, skill/workspace mentions) */
@@ -11,6 +12,9 @@ export interface EngineStreamOpts {
11
12
  conversation: ConvMessage[];
12
13
  /** Contextual blocks to prepend (user block, skills, workspace tree, etc.) */
13
14
  contextBlock: string;
15
+ /** The same context split into named sections (in contextBlock order), so an engine that keeps
16
+ * state across turns can send only what changed. Joined with '\n' they equal contextBlock. */
17
+ contextSections?: Record<string, string>;
14
18
 
15
19
  attachments?: AttachmentMeta[];
16
20
  images?: string[];
@@ -32,6 +36,9 @@ export interface EngineStreamOpts {
32
36
  /** True when the conversation was truncated (user replayed/edited a message) — engines with cached state should reset. */
33
37
  conversationReset?: boolean;
34
38
  context?: Record<string, string>;
39
+ /** SECURITY_ENFORCE only: the turn's security context. The engine derives tools/MCP/env from its effective
40
+ * profile and gates every tool call through it. Absent (flag off) ⇒ today's behavior, untouched. */
41
+ security?: TurnGuard;
35
42
 
36
43
  directives: Directives;
37
44
  config: AgentSettings;
@@ -39,6 +46,9 @@ export interface EngineStreamOpts {
39
46
 
40
47
  export interface AgentEngine {
41
48
  readonly name: string;
49
+ /** True if the engine applies `EngineStreamOpts.security` (profile tools/MCP/env + per-call gate). Under
50
+ * SECURITY_ENFORCE an engine without it may only run turns whose effective profile is unrestricted. */
51
+ readonly enforcesProfile?: boolean;
42
52
  stream(opts: EngineStreamOpts): AsyncGenerator<WsEvent>;
43
53
  /** Return model options for the UI picker (only called when engine is available) */
44
54
  getModels(): EngineModel[];