@drakon-systems/shieldcortex-realtime 4.47.38 → 4.47.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.ts CHANGED
@@ -2,26 +2,46 @@
2
2
  * ShieldCortex Real-time Scanning Plugin for OpenClaw v2026.3.22+
3
3
  *
4
4
  * Uses typed OpenClaw plugin hooks (`api.on`) for llm_input/llm_output
5
- * scanning and before_tool_call interception. `api.registerHook` registers
6
- * internal HOOK-style automation and does not participate in the agent-loop
7
- * block/approval semantics ShieldCortex needs.
8
- * All scanning operations are fire-and-forget.
5
+ * scanning and before_tool_call / before_agent_run interception.
6
+ * `api.registerHook` registers internal HOOK-style automation and does not
7
+ * participate in the agent-loop block/approval semantics ShieldCortex needs.
8
+ *
9
+ * NOT all scanning is fire-and-forget, and the distinction is the product:
10
+ *
11
+ * llm_input — OBSERVATION. Fire-and-forget; it has no blocking
12
+ * contract, so a detection here cannot stop the turn.
13
+ * before_agent_run — THE GATE (#225). Awaited by the gateway; its return
14
+ * value decides whether the run proceeds. Bounded end to
15
+ * end (CONVERSATION_SCAN_MAX_MS for the scan,
16
+ * CONVERSATION_NOTIFY_MAX_MS for the alert) because the
17
+ * user's turn waits on it, and failing OPEN on every
18
+ * internal error — as an EXPLICIT `{ outcome: 'pass' }`
19
+ * (#226), never as void, and never by throwing: the host
20
+ * registers this hook fail-CLOSED. See `gatePass`.
21
+ * before_tool_call — the Action Guard's gate, likewise awaited.
22
+ *
23
+ * Both conversation hooks honour `interceptor.conversation.posture`, including
24
+ * `off`, which is read before any scanner, audit write or cloud call.
9
25
  */
10
26
 
11
- import { createHash } from "node:crypto";
27
+ import { createHash, randomUUID } from "node:crypto";
12
28
  import fs from "node:fs/promises";
13
- import { existsSync, readFileSync, realpathSync } from "node:fs";
29
+ import { existsSync, readdirSync, readFileSync, realpathSync } from "node:fs";
14
30
  import path from "node:path";
15
- import { homedir } from "node:os";
31
+ import { homedir, hostname } from "node:os";
16
32
  import { fileURLToPath, pathToFileURL } from "node:url";
33
+ import { createRequire } from "node:module";
17
34
 
18
35
  import { readConversationAccess, describeRegisteredHooks } from './conversation-access.js';
19
36
  import { createSessionTaintStore } from './session-taint.js';
20
37
  import { classifyConversationOrigin } from './conversation-trust.js';
38
+ import type { ConversationTrustDecision } from './conversation-trust.js';
21
39
  import { createInterceptor, DEFAULT_CONFIG as DEFAULT_INTERCEPTOR_CONFIG } from './interceptor.js';
22
40
  import type { InterceptorConfig, BrokerRuntime } from './interceptor.js';
23
41
  import { syncInterceptEvent } from './intercept-ingest.js';
24
42
  import { cloudSync } from './cloud-sync.js';
43
+ import { createGatewayNotifyChannel } from './gateway-notify-channel.js';
44
+ import type { GatewayNotifyContext, NotifyChannelLike } from './gateway-notify-channel.js';
25
45
 
26
46
  // ==================== RESILIENT RUNTIME LOADER ====================
27
47
  // Resolves runtime.mjs from multiple locations so the plugin works both
@@ -46,6 +66,40 @@ type DefenceModule = {
46
66
  clean: boolean;
47
67
  injection: { clean: boolean; riskLevel: string; detections: unknown[] };
48
68
  };
69
+ /** #225 sink: the notify transport shared with the Action Guard (#143).
70
+ * Every member is optional — an older installed dist won't have them, and
71
+ * the guard must degrade to a loud log rather than fail. The field names
72
+ * MIRROR `NotifyConfig` in src/defence/iron-dome/notify-config.ts exactly
73
+ * (`webhookSecret`, not `secret`): a mirror that renames a field silently
74
+ * drops it, and the field this one would have dropped is the HMAC key —
75
+ * i.e. every POST would have gone out unsigned. */
76
+ normaliseNotifyConfig?: (raw: unknown) => {
77
+ enabled: boolean;
78
+ timeoutMs: number;
79
+ webhookUrl?: string;
80
+ webhookSecret?: string;
81
+ openclaw: boolean;
82
+ };
83
+ createWebhookNotifyChannel?: (opts: { url: string; secret?: string }) => NotifyChannelLike;
84
+ /** Builds the #225 notification. Used rather than an object literal so the
85
+ * bounding/truncation rules live in ONE place (see the module doc on
86
+ * operator-notify.ts) and a plugin cannot hand a channel a malformed shape. */
87
+ buildConversationThreatNotification?: (input: {
88
+ outcome: 'blocked' | 'observed' | 'unavailable';
89
+ posture: string;
90
+ summary: string;
91
+ reason: string;
92
+ sessionId?: string;
93
+ model?: string;
94
+ host?: string;
95
+ detectedAt: string;
96
+ }) => Record<string, unknown>;
97
+ /** The shared delivery core: bounded deadline, every failure normalised,
98
+ * nothing but the delivered boolean read back from a channel. */
99
+ deliverOperatorNotification?: (
100
+ notification: unknown,
101
+ deps: { channels: NotifyChannelLike[]; timeoutMs?: number },
102
+ ) => Promise<{ deliveredVia: string | null; attempts: Array<{ channel: string; result: { delivered: boolean; reason?: string } }> }>;
49
103
  };
50
104
 
51
105
  let runtimePromise: Promise<OpenClawRuntime> | null = null;
@@ -67,44 +121,110 @@ function addAncestorCandidates(candidates: Set<string>, startPath: string) {
67
121
  }
68
122
  }
69
123
 
70
- function collectRuntimeCandidates(): string[] {
124
+ /**
125
+ * Ask Node where the `shieldcortex` package actually is (#174).
126
+ *
127
+ * The plugin declares `shieldcortex` as a peer, so on ANY layout Node's own
128
+ * resolver can find it from here — no guessing at install prefixes. Resolving
129
+ * `shieldcortex/package.json` rather than the runtime file directly is
130
+ * deliberate: `./package.json` is the one subpath the main package's `exports`
131
+ * map always declares, whereas `hooks/openclaw/**` is in `files` but NOT in
132
+ * `exports`, so resolving it throws ERR_PACKAGE_PATH_NOT_EXPORTED.
133
+ *
134
+ * This is the strategy that fixes the reported `~/.local` host, and it works
135
+ * without widening the public `exports` surface.
136
+ */
137
+ function addResolvedPeerCandidate(
138
+ candidates: Set<string>,
139
+ fromUrl: string,
140
+ resolve: (spec: string, from: string) => string = (spec, from) => createRequire(from).resolve(spec),
141
+ ): void {
142
+ try {
143
+ addRuntimeCandidate(candidates, path.dirname(resolve("shieldcortex/package.json", fromUrl)));
144
+ } catch { /* not resolvable from here — later strategies still apply */ }
145
+ }
146
+
147
+ /**
148
+ * Every place the runtime might live, in the order we should try them.
149
+ *
150
+ * `home` is injected because Jest sandboxes SHIELDCORTEX_CONFIG_DIR but never
151
+ * HOME, so a test that did not inject it would probe the developer's real
152
+ * install and pass for the wrong reason.
153
+ */
154
+ function collectRuntimeCandidates(
155
+ home: string = homedir(),
156
+ resolveFrom: string = import.meta.url,
157
+ ): string[] {
71
158
  const candidates = new Set<string>();
72
159
 
73
- // 1. Relative path (works when running from within npm package tree)
74
- candidates.add(new URL("../../hooks/openclaw/cortex-memory/runtime.mjs", import.meta.url).href);
160
+ // 0. Operator escape hatch. `resolveOpenClawBinary` and the approval channel
161
+ // already honour an env override for the same class of "we guessed your
162
+ // install prefix wrong" problem; this list was the only copy without one,
163
+ // which is why every new prefix (bun, volta, asdf, ~/.local) has needed a
164
+ // code change. Accepts either the runtime file itself or a package root.
165
+ const envOverride = process.env.SHIELDCORTEX_RUNTIME_PATH?.trim();
166
+ if (envOverride) {
167
+ if (envOverride.endsWith(".mjs") && existsSync(envOverride)) {
168
+ candidates.add(pathToFileURL(envOverride).href);
169
+ } else {
170
+ addRuntimeCandidate(candidates, envOverride);
171
+ }
172
+ }
75
173
 
76
- // 2. Config file override (reads path from ~/.shieldcortex/config.json instead of env var)
174
+ // 1. Ask Node. Works on every layout including the ~/.local one this fixes.
175
+ addResolvedPeerCandidate(candidates, resolveFrom);
176
+
177
+ // 2. Relative path — the repo/source-tree layout, where ../../hooks/… is real.
178
+ // GUARDED, unlike before: on an installed layout this resolves to a
179
+ // non-existent scoped path (`@drakon-systems/hooks/…`), and because it was
180
+ // the only unguarded entry it became the SOLE list member and its
181
+ // ERR_MODULE_NOT_FOUND became the operator-visible failure — the exact
182
+ // message #174 reports. It is a real candidate in the source tree, so it
183
+ // is kept and screened rather than deleted.
184
+ const relative = fileURLToPath(new URL("../../hooks/openclaw/cortex-memory/runtime.mjs", resolveFrom));
185
+ if (existsSync(relative)) candidates.add(pathToFileURL(relative).href);
186
+
187
+ // 3. Config file override. Honours SHIELDCORTEX_CONFIG_DIR like the rest of
188
+ // the product — reading homedir() directly made this permanently blind on
189
+ // a host that relocates its config.
77
190
  try {
78
- const cfgPath = path.join(homedir(), ".shieldcortex", "config.json");
191
+ const configDir = process.env.SHIELDCORTEX_CONFIG_DIR?.trim() || path.join(home, ".shieldcortex");
192
+ const cfgPath = path.join(configDir, "config.json");
79
193
  if (existsSync(cfgPath)) {
80
194
  const cfg = JSON.parse(readFileSync(cfgPath, "utf-8"));
81
195
  if (cfg.installRoot) addRuntimeCandidate(candidates, cfg.installRoot);
82
196
  }
83
197
  } catch { /* no config */ }
84
198
 
85
- // 3. Walk up from current file location
86
- addAncestorCandidates(candidates, path.dirname(fileURLToPath(import.meta.url)));
199
+ // 4. Walk up from current file location
200
+ addAncestorCandidates(candidates, path.dirname(fileURLToPath(resolveFrom)));
87
201
 
88
- // 4. Resolve via common bin symlink paths (no child_process needed)
89
- for (const binDir of ["/usr/local/bin", "/opt/homebrew/bin", path.join(homedir(), ".npm-global", "bin")]) {
202
+ // 5. Resolve via common bin symlink paths (no child_process needed)
203
+ for (const binDir of [
204
+ "/usr/local/bin",
205
+ "/opt/homebrew/bin",
206
+ path.join(home, ".npm-global", "bin"),
207
+ path.join(home, ".local", "bin"), // #174: pip-style / npm --prefix ~/.local
208
+ ]) {
90
209
  const binPath = path.join(binDir, "shieldcortex");
91
210
  try {
92
211
  if (existsSync(binPath)) addAncestorCandidates(candidates, realpathSync(binPath));
93
212
  } catch { /* broken symlink */ }
94
213
  }
95
214
 
96
- // 5. Common global install paths (covers npm root -g results without spawning npm)
215
+ // 6. Common global install paths (covers npm root -g results without spawning npm)
97
216
  for (const root of [
98
217
  "/usr/lib/node_modules/shieldcortex",
99
218
  "/usr/local/lib/node_modules/shieldcortex",
100
219
  "/opt/homebrew/lib/node_modules/shieldcortex",
101
- path.join(homedir(), ".npm-global", "lib", "node_modules", "shieldcortex"),
102
- path.join(homedir(), ".nvm", "versions", "node"), // nvm users
220
+ path.join(home, ".npm-global", "lib", "node_modules", "shieldcortex"),
221
+ path.join(home, ".local", "lib", "node_modules", "shieldcortex"), // #174
222
+ path.join(home, ".nvm", "versions", "node"), // nvm users
103
223
  ]) {
104
224
  if (root.includes(".nvm")) {
105
225
  // For nvm, check the current symlink
106
226
  try {
107
- const currentNode = path.join(homedir(), ".nvm", "current", "lib", "node_modules", "shieldcortex");
227
+ const currentNode = path.join(home, ".nvm", "current", "lib", "node_modules", "shieldcortex");
108
228
  addRuntimeCandidate(candidates, currentNode);
109
229
  } catch { /* no nvm */ }
110
230
  } else {
@@ -137,6 +257,18 @@ async function getRuntime(): Promise<OpenClawRuntime> {
137
257
  }
138
258
  }
139
259
 
260
+ // #174: with every candidate screened by existsSync, "none found" is a
261
+ // real outcome and must not render as `Tried: . Last error: unknown
262
+ // error`. Name the escape hatch instead — this message is the only thing
263
+ // an operator on an unusual install prefix has to go on.
264
+ if (tried.length === 0) {
265
+ throw new Error(
266
+ "Could not load OpenClaw runtime: the shieldcortex package was not found from the plugin, " +
267
+ "and no known install prefix contained hooks/openclaw/cortex-memory/runtime.mjs. " +
268
+ "Point at it explicitly with SHIELDCORTEX_RUNTIME_PATH=/path/to/shieldcortex " +
269
+ "(or to the runtime.mjs itself), or reinstall so `shieldcortex` resolves as a peer of the plugin.",
270
+ );
271
+ }
140
272
  const detail = lastError instanceof Error ? lastError.message : String(lastError ?? "unknown error");
141
273
  throw new Error(`Could not load OpenClaw runtime. Tried: ${tried.join(", ")}. Last error: ${detail}`);
142
274
  })();
@@ -179,6 +311,13 @@ export function __getSessionTaintForTest(): typeof sessionTaint {
179
311
  return sessionTaint;
180
312
  }
181
313
 
314
+ /** Test seam for #174 runtime resolution: `home` and the resolving module URL
315
+ * are injectable because Jest sandboxes SHIELDCORTEX_CONFIG_DIR but never
316
+ * HOME, so an un-injected probe would read the developer's real install. */
317
+ export function __collectRuntimeCandidatesForTest(home?: string, from?: string): string[] {
318
+ return collectRuntimeCandidates(home, from);
319
+ }
320
+
182
321
  export function __setDefenceModuleForTest(mod: DefenceModule | null | undefined): void {
183
322
  _defenceModOverride = mod;
184
323
  _defenceModPromise = null;
@@ -191,9 +330,16 @@ export function __resetConfigStateForTest(): void {
191
330
  _config = null;
192
331
  _configOverride = null;
193
332
  _lastShieldConfigRef = null;
333
+ // Re-arm the once-per-load config-failure warning (#226).
334
+ _shieldConfigLoadFailureLogged = false;
194
335
  _registered = false;
195
336
  _beforeToolCallRegistered = false;
196
337
  _registrationError = null;
338
+ _beforeAgentRunRequested = false;
339
+ _conversationAccessGranted = false;
340
+ _gatewayNotifyContext = null;
341
+ _hostRuntimeVersion = null;
342
+ __resetScanUnavailableAlertState();
197
343
  }
198
344
 
199
345
  type LlmInputEvent = {
@@ -263,6 +409,616 @@ interface InterceptorUserConfig {
263
409
  /** Reviewed-script allowlist (#189). Passed through RAW; validated
264
410
  * entry-by-entry inside createReviewedScriptCheck. */
265
411
  reviewedScripts?: unknown[];
412
+ /** Operator-notify transport (#143), reused by the #225 conversation sink.
413
+ * Passed through RAW for the same reason as `broker`: `normaliseNotifyConfig`
414
+ * in the main package is the single boundary that knows which values arm a
415
+ * channel, and splitting that judgement across two files is how one of the
416
+ * halves ends up being the lenient one. Until #225 this key was silently
417
+ * DROPPED here, so the plugin could not reach an operator at all even on a
418
+ * box where the Claude Code hook could. */
419
+ notify?: Record<string, unknown>;
420
+ };
421
+ /** Conversation firewall posture (#225). See CONVERSATION_POSTURES. */
422
+ conversation?: { posture?: ConversationPosture };
423
+ }
424
+
425
+ /**
426
+ * What the conversation firewall is allowed to DO about a detection (#225).
427
+ *
428
+ * Before this existed, scanning ran on `llm_input` — an OpenClaw *observation*
429
+ * hook with no blocking contract — and a detection's entire effect was a
430
+ * console line. The product implied a firewall and shipped a logger. These
431
+ * three postures make the difference explicit and reportable:
432
+ *
433
+ * off — do not scan the conversation at all
434
+ * observe — scan, audit, notify the operator, NEVER block
435
+ * enforce — additionally block the run (via `before_agent_run`)
436
+ *
437
+ * Default is `observe`, deliberately. It is exactly the behaviour that shipped
438
+ * before — but named honestly instead of implied to be protection — and #182
439
+ * says the guard's false-positive rate is still unmeasured. An unmeasured
440
+ * blocker in front of every turn would be a worse incident than the one this
441
+ * fixes; `enforce` is opt-in until that number exists.
442
+ */
443
+ export type ConversationPosture = 'off' | 'observe' | 'enforce';
444
+ const CONVERSATION_POSTURES: readonly ConversationPosture[] = ['off', 'observe', 'enforce'];
445
+
446
+ /** Resolve the configured posture. Anything unrecognised resolves DOWN to
447
+ * `observe`, never up to `enforce`: a typo must not silently start blocking
448
+ * every turn on an operator's box. */
449
+ export function conversationPosture(raw: unknown): ConversationPosture {
450
+ if (!raw || typeof raw !== 'object') return 'observe';
451
+ const value = (raw as { posture?: unknown }).posture;
452
+ return typeof value === 'string' && (CONVERSATION_POSTURES as readonly string[]).includes(value)
453
+ ? (value as ConversationPosture)
454
+ : 'observe';
455
+ }
456
+
457
+ /**
458
+ * A scan outcome as the conversation guard sees it.
459
+ *
460
+ * `available: false` marks a scan that could not be RUN at all — no defence
461
+ * module, no MCP fallback, or a throw inside the scanner. It is deliberately a
462
+ * separate axis from `clean`, and `clean` is set FALSE alongside it, because
463
+ * the bug this replaces returned `{ clean: true, summary: 'scan unavailable' }`
464
+ * from the ordinary unavailable path: every caller that read only `clean` then
465
+ * treated an unscanned turn as a scanned-and-fine one, silently, forever.
466
+ *
467
+ * `errored` is kept as an alias of "not available" for the pure decision
468
+ * function's existing contract.
469
+ */
470
+ export interface ConversationScanResult {
471
+ /** Only meaningful when `available` is true. False when unavailable, so a
472
+ * caller that ignores `available` still cannot read "clean". */
473
+ clean: boolean;
474
+ summary: string;
475
+ /** The scan actually ran and produced a verdict. */
476
+ available: boolean;
477
+ /** Set when the scan could not be completed. Mirrors `!available`. */
478
+ errored?: boolean;
479
+ /** Failure detail, for the audit row and the operator alert. Never contains
480
+ * scanned content. */
481
+ error?: string;
482
+ }
483
+
484
+ export interface ConversationDecision {
485
+ block: boolean;
486
+ notify: boolean;
487
+ audit: boolean;
488
+ reason: string | null;
489
+ /** What to tell a human happened to this turn — the same vocabulary the
490
+ * notification carries, so the audit row and the alert cannot disagree. */
491
+ outcome: 'clean' | 'blocked' | 'observed' | 'unavailable' | 'not-scanned';
492
+ }
493
+
494
+ /**
495
+ * The whole decision, as a pure function — no I/O, no hooks, so the posture
496
+ * semantics are testable directly and cannot drift as the plumbing changes.
497
+ *
498
+ * The key line is `notify` on a non-blocking detection: logging is not a sink.
499
+ * The #225 finding was that a HIGH verdict reached a log file and nothing else,
500
+ * so a real threat on a real box was seen by nobody. Under `observe` we still
501
+ * do not stop the turn — but a human hears about it.
502
+ *
503
+ * `trust` is the second input because BLOCKING is a consequence, and #235's rule
504
+ * is that source trust gates consequences (see conversation-trust.ts). It is
505
+ * optional, and its absence means "origin not established", which resolves
506
+ * toward enforcement rather than away from it: a caller that does not know who
507
+ * spoke has not proved the owner did.
508
+ */
509
+ export function evaluateConversationRun(
510
+ posture: ConversationPosture,
511
+ scan: ConversationScanResult,
512
+ trust?: ConversationTrustDecision,
513
+ ): ConversationDecision {
514
+ if (posture === 'off') {
515
+ return { block: false, notify: false, audit: false, reason: null, outcome: 'not-scanned' };
516
+ }
517
+
518
+ // Scanner failure fails OPEN — a broken scanner must not wedge every turn,
519
+ // which is the outcome ShieldCortex exists to prevent — but it is reported,
520
+ // because an unprotected turn must never read as a protected one. Note the
521
+ // condition: `available === false` OR the legacy `errored` flag, so a caller
522
+ // still constructing the old shape cannot route an unscanned turn into the
523
+ // clean branch.
524
+ if (scan.available === false || scan.errored) {
525
+ return {
526
+ block: false,
527
+ notify: true,
528
+ audit: true,
529
+ reason: `conversation scan unavailable (${scan.error ?? scan.summary}) — turn allowed UNSCANNED`,
530
+ outcome: 'unavailable',
531
+ };
532
+ }
533
+
534
+ if (scan.clean) {
535
+ return { block: false, notify: false, audit: false, reason: null, outcome: 'clean' };
536
+ }
537
+
538
+ // #235: the owner's own words are an instruction, so `enforce` does not act on
539
+ // them. This is the branch the whole trust module exists for — a block here
540
+ // does not warn the operator, it DESTROYS their message: OpenClaw keeps only
541
+ // the replacement text. Everything above still happened: the content was
542
+ // scanned, and `notify`/`audit` below are true whoever sent it. Only the
543
+ // consequence is withheld, and the reason says so on the row rather than
544
+ // leaving an enforce-posture host that did not block looking like a bug.
545
+ const trusted = trust !== undefined && !trust.mayTaint;
546
+ const block = posture === 'enforce' && !trusted;
547
+ return {
548
+ block,
549
+ notify: true,
550
+ audit: true,
551
+ reason:
552
+ posture === 'enforce' && trusted
553
+ ? `conversation threat: ${scan.summary} — NOT blocked: ${trust!.reason}`
554
+ : `conversation threat: ${scan.summary}`,
555
+ outcome: block ? 'blocked' : 'observed',
556
+ };
557
+ }
558
+
559
+ // ==================== CONVERSATION PLANE: HOST SUPPORT + CONSENT ============
560
+
561
+ /**
562
+ * The first OpenClaw build whose plugin SDK declares the `before_agent_run`
563
+ * gate. Established by inspecting published npm artifacts, not by guessing:
564
+ *
565
+ * 2026.5.7 — `hook-types.d.ts` has no `before_agent_run` anywhere (0 hits);
566
+ * CONVERSATION_HOOK_NAMES = llm_input, llm_output,
567
+ * before_agent_finalize, agent_end
568
+ * 2026.5.9-beta.1 — FIRST published build declaring it: in `PLUGIN_HOOK_NAMES`,
569
+ * in `CONVERSATION_HOOK_NAMES`, in `PluginHookHandlerMap`, with
570
+ * `PluginHookBeforeAgentRunResult = InputGateDecision | void`
571
+ * 2026.5.12 — first STABLE (non-prerelease) build with it (2026.5.10 and
572
+ * 2026.5.12 published betas in between; there is no plain
573
+ * 2026.5.9 release)
574
+ *
575
+ * Below this floor `api.on('before_agent_run', …)` is accepted by the API and
576
+ * then DROPPED by the registry with an `unknown typed hook … ignored`
577
+ * diagnostic — it does not throw. So a version check is the only honest way to
578
+ * know, and claiming enforcement without one is exactly the class of false
579
+ * green #222 is about.
580
+ *
581
+ * ── ONE FLOOR, THREE FILES ────────────────────────────────────────────────
582
+ *
583
+ * The authoritative value is the STABLE release, and it is stated in three
584
+ * places that cannot import each other:
585
+ *
586
+ * plugins/openclaw/index.ts — this constant
587
+ * src/integrations/openclaw-conversation-capability.ts
588
+ * — CONVERSATION_ENFORCEMENT_MIN_OPENCLAW
589
+ * plugins/openclaw/openclaw.plugin.json — engines.conversationGate
590
+ *
591
+ * THE BOUNDARY IS REAL, not a preference. The plugin ships as its own dist,
592
+ * compiled by `tsconfig.openclaw-plugin.json` with `rootDir:
593
+ * ./plugins/openclaw` and an explicit `include` list; a `src/` import does not
594
+ * merely offend layering, it fails to emit — and the src module imports
595
+ * `semver`, which the plugin bundle does not carry (hence the hand-rolled
596
+ * `compareOpenClawVersions` below). The manifest is JSON read by the host and
597
+ * imports nothing at all.
598
+ *
599
+ * So the three are pinned EQUAL by test instead of shared by import:
600
+ * `src/__tests__/conversation-gate-floor-parity-226.test.ts` reads all three
601
+ * and fails on drift. Change one, that test tells you about the other two.
602
+ *
603
+ * `CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW` is deliberately SUBORDINATE: it
604
+ * decides nothing an operator sees. Its only job is to mark the band where a
605
+ * version number alone cannot answer the question — see
606
+ * `hostSupportsConversationGate`.
607
+ */
608
+ export const CONVERSATION_GATE_MIN_OPENCLAW = '2026.5.12';
609
+ /**
610
+ * Documentation of when the hook first appeared, NOT a second floor.
611
+ *
612
+ * The previous cut used this as the support threshold, which made the plugin's
613
+ * operator-facing verdict disagree with the CLI's on any 2026.5.9-beta.1 →
614
+ * 2026.5.11 host: `shieldcortex doctor` said enforcement was unavailable while
615
+ * the plugin's own status line said supported. Two answers to one question is
616
+ * how the next false green gets built.
617
+ */
618
+ export const CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW = '2026.5.9-beta.1';
619
+
620
+ /**
621
+ * Compare two OpenClaw CalVer strings (`2026.5.12`, `2026.5.9-beta.1`).
622
+ * Deliberately local and tiny: the plugin build cannot import semver, and the
623
+ * only question asked is "is this host at or above the floor".
624
+ * Returns null when either side cannot be parsed — "unknown", never "yes".
625
+ */
626
+ export function compareOpenClawVersions(a: string, b: string): number | null {
627
+ const parse = (v: string): { nums: number[]; pre: string[] } | null => {
628
+ // Exactly three numeric parts, and only `-` introduces a prerelease. A
629
+ // trailing `.4` is NOT a prerelease tail — it is a version shape we do not
630
+ // understand, and the safe answer to that is "unknown".
631
+ const m = String(v ?? '').trim().match(/^(\d+)\.(\d+)\.(\d+)(?:-([0-9A-Za-z.-]+))?$/);
632
+ if (!m) return null;
633
+ return { nums: [Number(m[1]), Number(m[2]), Number(m[3])], pre: m[4] ? m[4].split('.') : [] };
634
+ };
635
+ const pa = parse(a);
636
+ const pb = parse(b);
637
+ if (!pa || !pb) return null;
638
+ for (let i = 0; i < 3; i++) {
639
+ if (pa.nums[i] !== pb.nums[i]) return pa.nums[i] < pb.nums[i] ? -1 : 1;
640
+ }
641
+ // A prerelease sorts BELOW the same numeric release (2026.5.9-beta.1 < 2026.5.9).
642
+ if (pa.pre.length === 0 && pb.pre.length === 0) return 0;
643
+ if (pa.pre.length === 0) return 1;
644
+ if (pb.pre.length === 0) return -1;
645
+ return comparePrerelease(pa.pre, pb.pre);
646
+ }
647
+
648
+ /** Semver prerelease precedence, restricted to what a CalVer tail can hold:
649
+ * numeric identifiers compare numerically, a numeric identifier sorts below an
650
+ * alphanumeric one, and a shorter identifier list sorts below an
651
+ * otherwise-equal longer one (`beta` < `beta.1` < `beta.2` < `beta.10`).
652
+ *
653
+ * The string compare this replaces put `beta.10` BELOW `beta.1`, so the tenth
654
+ * beta of the gate build was classified as predating the first — an error in
655
+ * the one direction this file must never make, since it demotes a host that
656
+ * HAS the gate to 'unsupported'. */
657
+ function comparePrerelease(a: string[], b: string[]): number {
658
+ const len = Math.max(a.length, b.length);
659
+ for (let i = 0; i < len; i++) {
660
+ const x = a[i];
661
+ const y = b[i];
662
+ if (x === undefined) return -1;
663
+ if (y === undefined) return 1;
664
+ const xNum = /^\d+$/.test(x);
665
+ const yNum = /^\d+$/.test(y);
666
+ if (xNum && yNum) {
667
+ const nx = Number(x);
668
+ const ny = Number(y);
669
+ if (nx !== ny) return nx < ny ? -1 : 1;
670
+ continue;
671
+ }
672
+ if (xNum !== yNum) return xNum ? -1 : 1;
673
+ if (x !== y) return x < y ? -1 : 1;
674
+ }
675
+ return 0;
676
+ }
677
+
678
+ /** What we could establish about the OpenClaw build we are running inside. */
679
+ export interface HostOpenClawProbe {
680
+ /** The host runtime version, or null when it could not be established. */
681
+ version: string | null;
682
+ /**
683
+ * Where `version` came from, so a reader can weigh it.
684
+ *
685
+ * 'runtime' — `api.runtime.version`, the host's own declared runtime
686
+ * version (`PluginRuntimeCore.version`). First choice: the
687
+ * gateway states it about itself.
688
+ * 'package.json' — the openclaw package.json found by walking up from the
689
+ * entry path. A fallback, and wrong on an install whose
690
+ * on-disk package and running process differ.
691
+ * null — no version evidence at all.
692
+ */
693
+ versionSource?: 'runtime' | 'package.json' | null;
694
+ /** The host package root, or null. */
695
+ root: string | null;
696
+ /**
697
+ * Whether THIS host's shipped hook declarations name `before_agent_run`.
698
+ * true/false when the declarations were found and read; null when they were
699
+ * not (an install without type declarations, an unusual layout).
700
+ *
701
+ * This is the primary evidence, ahead of the version string, because it is a
702
+ * property of the build actually on this box: a fork, a backport or a patched
703
+ * install answers correctly here and would be mis-classified by a version
704
+ * comparison. NOTE: the runtime `openclaw/plugin-sdk` entrypoint exports
705
+ * exactly ten symbols (ContextEngine helpers, onDiagnosticEvent, stringEnum,
706
+ * …) and none of them is the hook-name list — read off 2026.7.1's shipped
707
+ * `dist/plugin-sdk/index.js` — so importing the SDK and asking it directly is
708
+ * not available.
709
+ */
710
+ declaresGate: boolean | null;
711
+ }
712
+
713
+ /** Test seam: pins the host probe without touching disk. */
714
+ let _hostProbeOverride: HostOpenClawProbe | null | undefined;
715
+ export function __setHostOpenClawProbeForTest(p: HostOpenClawProbe | null | undefined): void {
716
+ _hostProbeOverride = p;
717
+ _hostProbeCache = undefined;
718
+ }
719
+ let _hostProbeCache: HostOpenClawProbe | undefined;
720
+
721
+ /**
722
+ * The host runtime version the gateway told us about, captured at register().
723
+ *
724
+ * `api.runtime.version` is declared by the host SDK as
725
+ * `PluginRuntimeCore.version: string` — the version of the OpenClaw runtime
726
+ * this plugin is loaded into (verified against the installed host's
727
+ * `dist/plugin-sdk/src/plugins/runtime/types-core.d.ts`, and `api.runtime` is
728
+ * on `OpenClawPluginApi` in the same build). It is NOT `api.version`, which is
729
+ * the plugin's own version and would answer a completely different question:
730
+ * comparing OUR version against an OpenClaw floor would classify every host as
731
+ * unsupported.
732
+ *
733
+ * It is preferred over the filesystem walk because it is the running process
734
+ * describing itself, where the walk infers from whichever package.json happens
735
+ * to sit above the entry path. Null until a host actually supplies it — an
736
+ * older gateway, a CLI invocation or a test rig may not, and that is UNKNOWN.
737
+ */
738
+ let _hostRuntimeVersion: string | null = null;
739
+
740
+ /** Test seam for the runtime-supplied host version. */
741
+ export function __setHostRuntimeVersionForTest(v: string | null): void {
742
+ _hostRuntimeVersion = v;
743
+ }
744
+
745
+ /**
746
+ * Record `api.runtime.version` if this host exposes it. Returns what was
747
+ * recorded (null when nothing usable was offered), and never throws: a host
748
+ * with an exotic `runtime` getter must not take the plugin's registration down.
749
+ */
750
+ export function recordHostRuntimeVersion(api: unknown): string | null {
751
+ try {
752
+ const runtime = (api as { runtime?: { version?: unknown } } | null | undefined)?.runtime;
753
+ const version = runtime?.version;
754
+ _hostRuntimeVersion = typeof version === 'string' && version.trim() ? version.trim() : null;
755
+ } catch {
756
+ _hostRuntimeVersion = null;
757
+ }
758
+ return _hostRuntimeVersion;
759
+ }
760
+
761
+ /** Does this host's shipped SDK declare the gate? Bounded, best-effort, and
762
+ * null on anything unexpected — an unreadable install is UNKNOWN, never
763
+ * "supported". */
764
+ function probeGateDeclaration(root: string): boolean | null {
765
+ const candidates: string[] = [];
766
+ // 2026.5.2-era layout: the declarations live under the plugin-sdk tree.
767
+ candidates.push(path.join(root, 'dist', 'plugin-sdk', 'src', 'plugins', 'hook-types.d.ts'));
768
+ // 2026.6+/2026.7 layout: a single hashed `hook-types-<hash>.d.ts` at dist root.
769
+ try {
770
+ const distDir = path.join(root, 'dist');
771
+ if (existsSync(distDir)) {
772
+ for (const name of readdirSync(distDir)) {
773
+ if (/^hook-types.*\.d\.ts$/.test(name)) candidates.push(path.join(distDir, name));
774
+ }
775
+ }
776
+ } catch { /* fall through to whatever candidates we have */ }
777
+
778
+ let sawAny = false;
779
+ for (const file of candidates) {
780
+ try {
781
+ if (!existsSync(file)) continue;
782
+ sawAny = true;
783
+ if (/\bbefore_agent_run\b/.test(readFileSync(file, 'utf-8'))) return true;
784
+ } catch { /* unreadable candidate — try the next */ }
785
+ }
786
+ return sawAny ? false : null;
787
+ }
788
+
789
+ /**
790
+ * Everything we can learn about the host OpenClaw build from inside the plugin.
791
+ *
792
+ * Two independent sources, both read, neither invented:
793
+ *
794
+ * - `api.runtime.version` — the running gateway's own statement of its
795
+ * version, captured at register() (see `recordHostRuntimeVersion`). This is
796
+ * the primary VERSION evidence when the host offers it. Note it is not
797
+ * `api.version`, which is this plugin's version.
798
+ * - the filesystem — walk up from the gateway's entry path to the package.json
799
+ * that names openclaw, then read that install's own shipped hook
800
+ * declarations. This is the version FALLBACK, and it is the only source of
801
+ * `declaresGate`, which stays the strongest gate-support evidence of the two
802
+ * (a backport or a fork answers it correctly where a version comparison
803
+ * cannot — see `hostSupportsConversationGate`).
804
+ *
805
+ * No process is ever spawned. A null everywhere is a legitimate, frequently
806
+ * correct answer (a CLI invocation, an unusual install layout) and callers must
807
+ * treat it as UNKNOWN — never as "supported".
808
+ */
809
+ export function detectHostOpenClaw(): HostOpenClawProbe {
810
+ const disk = detectHostOpenClawFromDisk();
811
+ // The runtime's own version outranks whatever package.json the walk landed
812
+ // on — but only for the version; `declaresGate` and `root` are disk facts and
813
+ // are carried through untouched.
814
+ if (_hostRuntimeVersion) return { ...disk, version: _hostRuntimeVersion, versionSource: 'runtime' };
815
+ return disk;
816
+ }
817
+
818
+ function detectHostOpenClawFromDisk(): HostOpenClawProbe {
819
+ if (_hostProbeOverride !== undefined) return _hostProbeOverride ?? { version: null, root: null, declaresGate: null };
820
+ if (_hostProbeCache !== undefined) return _hostProbeCache;
821
+ _hostProbeCache = (() => {
822
+ const empty: HostOpenClawProbe = { version: null, root: null, declaresGate: null, versionSource: null };
823
+ const entry = process.argv?.[1];
824
+ if (!entry || typeof entry !== 'string') return empty;
825
+ let current: string;
826
+ try {
827
+ current = path.dirname(realpathSync(entry));
828
+ } catch {
829
+ current = path.dirname(entry);
830
+ }
831
+ let previous = '';
832
+ for (let i = 0; i < 8 && current !== previous; i++) {
833
+ try {
834
+ const pkgPath = path.join(current, 'package.json');
835
+ if (existsSync(pkgPath)) {
836
+ const hostPkg = JSON.parse(readFileSync(pkgPath, 'utf-8')) as { name?: unknown; version?: unknown };
837
+ if (hostPkg?.name === 'openclaw') {
838
+ const version = typeof hostPkg.version === 'string' ? hostPkg.version : null;
839
+ return {
840
+ version,
841
+ versionSource: version ? ('package.json' as const) : null,
842
+ root: current,
843
+ declaresGate: probeGateDeclaration(current),
844
+ };
845
+ }
846
+ }
847
+ } catch { /* keep walking up */ }
848
+ previous = current;
849
+ current = path.dirname(current);
850
+ }
851
+ return empty;
852
+ })();
853
+ return _hostProbeCache;
854
+ }
855
+
856
+ /** Convenience for callers that only want the version string. */
857
+ export function detectHostOpenClawVersion(): string | null {
858
+ return detectHostOpenClaw().version;
859
+ }
860
+
861
+ export type GateSupport = 'supported' | 'unsupported' | 'unknown';
862
+
863
+ /**
864
+ * Does this host have the `before_agent_run` gate at all?
865
+ *
866
+ * Order matters: what the installed build DECLARES outranks what its version
867
+ * number implies, and both outrank a guess. There is no branch here that
868
+ * returns 'supported' without evidence.
869
+ */
870
+ export function hostSupportsConversationGate(probe: HostOpenClawProbe | string | null): GateSupport {
871
+ const resolved: HostOpenClawProbe =
872
+ typeof probe === 'string' || probe === null
873
+ ? { version: probe, root: null, declaresGate: null }
874
+ : probe;
875
+ if (resolved.declaresGate === true) return 'supported';
876
+ if (resolved.declaresGate === false) return 'unsupported';
877
+ if (!resolved.version) return 'unknown';
878
+ const cmp = compareOpenClawVersions(resolved.version, CONVERSATION_GATE_MIN_OPENCLAW);
879
+ if (cmp === null) return 'unknown';
880
+ if (cmp >= 0) return 'supported';
881
+
882
+ // Below the STABLE floor. One band inside that is not honestly 'unsupported':
883
+ // 2026.5.9-beta.1 → 2026.5.11 ship the hook as a prerelease, so calling them
884
+ // unsupported would tell an operator "no posture can block a turn on this
885
+ // host" about a host that blocks. The opposite claim is worse still, so
886
+ // neither is made: this is the absence of a measurement, and
887
+ // `describeConversationPlane` renders it as UNPROVEN and active:false.
888
+ //
889
+ // In practice a real prerelease install lands on `declaresGate` above and
890
+ // never reaches here — this branch is what happens when the declarations
891
+ // could not be read either, i.e. when we genuinely do not know.
892
+ const pre = compareOpenClawVersions(resolved.version, CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW);
893
+ if (pre === null) return 'unknown';
894
+ return pre >= 0 ? 'unknown' : 'unsupported';
895
+ }
896
+
897
+ /**
898
+ * Read the operator's CONVERSATION-ACCESS consent for this plugin from the
899
+ * host config: `plugins.entries.<id>.hooks.allowConversationAccess === true`.
900
+ *
901
+ * OpenClaw refuses every conversation hook for a non-bundled plugin without
902
+ * this exact value (registry: `record.origin !== "bundled" &&
903
+ * explicitConversationAccess !== true`). `llm_input` and `llm_output` are on
904
+ * that list in every build; `before_agent_run` joins it in 2026.5.9-beta.1,
905
+ * the same build that first declares the gate at all — so from there on the
906
+ * grant governs the conversation firewall's enforcement point too.
907
+ * Strict `true` only, matching the host: `undefined` and `false` are the same
908
+ * refusal there, and reading them differently here would report protection the
909
+ * gateway is not providing.
910
+ *
911
+ * It is the operator's per-box CONSENT grant, and this plugin only ever READS
912
+ * it. Nothing on the plugin's own path — `register()`, a hook, a background
913
+ * refresh — may write it: a security product that silently grants itself the
914
+ * right to read every conversation is the behaviour this product exists to
915
+ * catch. The only thing that may set it is an explicit, operator-initiated
916
+ * install/repair that says so out loud (#225: "the installer must never set it
917
+ * silently"). Absence is therefore reported, loudly and by name, rather than
918
+ * fixed from in here.
919
+ */
920
+ export function readConversationAccessGrant(rootConfig: unknown): boolean {
921
+ if (!rootConfig || typeof rootConfig !== 'object' || Array.isArray(rootConfig)) return false;
922
+ const entries = (rootConfig as {
923
+ plugins?: { entries?: Record<string, { hooks?: { allowConversationAccess?: unknown } } | undefined> };
924
+ }).plugins?.entries;
925
+ const entry = entries?.[PLUGIN_ID] ?? entries?.[PLUGIN_PACKAGE_NAME];
926
+ return entry?.hooks?.allowConversationAccess === true;
927
+ }
928
+
929
+ /**
930
+ * Everything we can HONESTLY say about the conversation plane on this host.
931
+ *
932
+ * Note what is not here: any claim that the gateway accepted the hook.
933
+ * `api.on()` returns void, never throws on an unknown hook name, and never
934
+ * throws on a refused conversation hook — it records a diagnostic and returns —
935
+ * so "the host accepted our registration" is not knowable from inside the
936
+ * plugin, and a flag asserting it would be decoration. What IS knowable is the
937
+ * two facts that decide the outcome: whether the host build has the gate, and
938
+ * whether the operator granted conversation access.
939
+ */
940
+ export interface ConversationPlaneState {
941
+ posture: ConversationPosture;
942
+ /** We called `api.on('before_agent_run', …)` this session. */
943
+ hookRequested: boolean;
944
+ gateSupport: GateSupport;
945
+ hostOpenClawVersion: string | null;
946
+ consentGranted: boolean;
947
+ /** True only when the posture is on AND both preconditions hold. */
948
+ active: boolean;
949
+ /** One line, exact about the evidence, for status and doctor. */
950
+ summary: string;
951
+ }
952
+
953
+ export function describeConversationPlane(input: {
954
+ posture: ConversationPosture;
955
+ hookRequested: boolean;
956
+ gateSupport: GateSupport;
957
+ hostOpenClawVersion: string | null;
958
+ consentGranted: boolean;
959
+ }): ConversationPlaneState {
960
+ const { posture, hookRequested, gateSupport, hostOpenClawVersion, consentGranted } = input;
961
+ const hostText = hostOpenClawVersion ? `OpenClaw ${hostOpenClawVersion}` : 'OpenClaw version undetermined';
962
+
963
+ if (posture === 'off') {
964
+ return {
965
+ ...input,
966
+ active: false,
967
+ summary: 'off — conversation scanning disabled by config (interceptor.conversation.posture=off)',
968
+ };
969
+ }
970
+ if (!consentGranted) {
971
+ return {
972
+ ...input,
973
+ active: false,
974
+ summary:
975
+ `INACTIVE: conversation access NOT granted on this host — set plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess=true ` +
976
+ 'in openclaw.json (operator consent; the installer will never set it for you). Until then the gateway REFUSES llm_input and ' +
977
+ 'llm_output for this plugin — and, on builds that have it, before_agent_run too: nothing on the conversation path is scanned or blocked',
978
+ };
979
+ }
980
+ if (gateSupport === 'unsupported') {
981
+ return {
982
+ ...input,
983
+ active: false,
984
+ summary:
985
+ `INACTIVE for enforcement: ${hostText} predates the before_agent_run gate ` +
986
+ `(floor ${CONVERSATION_GATE_MIN_OPENCLAW}; first seen as a prerelease in ${CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW}) — ` +
987
+ 'observation only; no posture can block a turn on this host',
988
+ };
989
+ }
990
+ if (!hookRequested) {
991
+ return { ...input, active: false, summary: 'INACTIVE: the before_agent_run hook was not registered this session' };
992
+ }
993
+ if (gateSupport === 'unknown') {
994
+ // UNPROVEN IS NOT ACTIVE. Every other branch above is a fact we read — the
995
+ // posture, the grant, the host build. This one is the absence of a
996
+ // measurement: we could not establish that this host has the gate at all,
997
+ // and `api.on` does not acknowledge a registration, so nothing here knows
998
+ // whether the hook exists. Reporting `active: true` with a caveat glued to
999
+ // the summary string — which is what this did — means every caller that
1000
+ // reads the boolean instead of the prose (status renderers, doctor, any
1001
+ // future check) claims a live firewall on evidence nobody has. Under
1002
+ // `enforce` that is the worst version of it: the operator believes turns
1003
+ // are being blocked on a host where the gate may be silently dropped.
1004
+ return {
1005
+ ...input,
1006
+ active: false,
1007
+ summary:
1008
+ `UNPROVEN: could not verify that host ${hostText} provides the before_agent_run gate ` +
1009
+ `(no runtime version, no readable hook declarations), and the plugin API does not acknowledge a registration. ` +
1010
+ (posture === 'enforce'
1011
+ ? 'The posture is enforce, so a dirty verdict WOULD block the run where the gate exists — but that it exists here is not established. Treat this host as observation-only until it is.'
1012
+ : 'Detections are audited and sent to the operator where the hook runs at all; nothing is blocked in this posture regardless.'),
1013
+ };
1014
+ }
1015
+ return {
1016
+ ...input,
1017
+ active: true,
1018
+ summary:
1019
+ posture === 'enforce'
1020
+ ? 'enforce — a dirty verdict BLOCKS the run via before_agent_run'
1021
+ : 'observe — detections are audited and sent to the operator; turns are NOT blocked',
266
1022
  };
267
1023
  }
268
1024
 
@@ -276,6 +1032,20 @@ interface SCConfig {
276
1032
  openclawAutoMemoryNoveltyThreshold?: number;
277
1033
  openclawAutoMemoryMaxRecent?: number;
278
1034
  interceptor?: InterceptorUserConfig;
1035
+ /**
1036
+ * Source trust for conversation content (see conversation-trust.ts).
1037
+ *
1038
+ * Declared and PARSED here, not read off an untyped cast. `normaliseConfig`
1039
+ * is a strict allowlist — that is the whole point of #112/#115 — so a key it
1040
+ * does not name is dropped on both config paths (~/.shieldcortex/config.json
1041
+ * and the openclaw.json plugin entry). Landed as a cast against an SCConfig
1042
+ * that had no such field, `conversationTrust.trustOwnerInput` was therefore
1043
+ * read as `undefined` on every host, and the documented opt-out could not be
1044
+ * turned on by anyone. It matters more now that trust also gates BLOCKING:
1045
+ * an operator who wants owner input policed like everything else has to have
1046
+ * a working way to say so.
1047
+ */
1048
+ conversationTrust?: { trustOwnerInput?: boolean };
279
1049
  }
280
1050
 
281
1051
  const PLUGIN_ID = "shieldcortex-realtime";
@@ -330,6 +1100,21 @@ const PLUGIN_CONFIG_UI_HINTS = {
330
1100
  label: "Enable Tool Call Interceptor",
331
1101
  help: "Scan memory-write tool calls and gate suspicious content behind user approval.",
332
1102
  },
1103
+ // #226: these two exist in openclaw.plugin.json's uiHints and were missing
1104
+ // here, so the host UI and the plugin's own declared hints described
1105
+ // different sets of settings. The manifest parity test now pins the two key
1106
+ // sets EQUAL in both directions, because a hint present on only one side is
1107
+ // a setting one surface documents and the other silently omits.
1108
+ "interceptor.severityActions.high": {
1109
+ label: "High Severity Action",
1110
+ help: "Action for high-severity threats: log, warn, or require_approval.",
1111
+ advanced: true,
1112
+ },
1113
+ "interceptor.severityActions.critical": {
1114
+ label: "Critical Severity Action",
1115
+ help: "Action for critical-severity threats: log, warn, or require_approval.",
1116
+ advanced: true,
1117
+ },
333
1118
  "interceptor.actionGuard.enabled": {
334
1119
  label: "Action Guard",
335
1120
  help: "Gate dangerous shell/file/network/git tool calls before they execute. Catastrophic operations are always blocked while enabled.",
@@ -358,6 +1143,39 @@ const PLUGIN_CONFIG_UI_HINTS = {
358
1143
  help: "Off = every dangerous-tier action still waits for you; the broker can then only harden, never release.",
359
1144
  advanced: true,
360
1145
  },
1146
+ // #225. The posture is the whole product claim on the conversation path, so
1147
+ // it is NOT marked advanced: an operator must be able to see, in the UI that
1148
+ // configures this plugin, whether the firewall in front of their prompts can
1149
+ // stop anything.
1150
+ "interceptor.conversation.posture": {
1151
+ label: "Conversation Firewall",
1152
+ help:
1153
+ "What the conversation firewall does with a detection on the input path. " +
1154
+ "off = do not scan; observe = scan, audit and alert the operator but never stop the turn (default); " +
1155
+ "enforce = block the run via before_agent_run. Requires plugins.entries.shieldcortex-realtime.hooks.allowConversationAccess=true " +
1156
+ "on this host — OpenClaw refuses conversation hooks without that operator grant, and ShieldCortex will never set it for you.",
1157
+ },
1158
+ "interceptor.actionGuard.notify.enabled": {
1159
+ label: "Operator Notifications",
1160
+ help: "Reach a human when the guard holds an action, or when the conversation firewall detects a threat. Off by default.",
1161
+ advanced: true,
1162
+ },
1163
+ "interceptor.actionGuard.notify.webhookUrl": {
1164
+ label: "Notify Webhook URL",
1165
+ help: "http(s) endpoint the notification is POSTed to. Conversation-firewall alerts carry no approve/deny affordance — there is nothing to approve.",
1166
+ advanced: true,
1167
+ },
1168
+ "interceptor.actionGuard.notify.webhookSecret": {
1169
+ label: "Notify Webhook Secret",
1170
+ help: "HMAC-SHA256 key for X-ShieldCortex-Signature, so the receiver can reject spoofed POSTs.",
1171
+ sensitive: true,
1172
+ advanced: true,
1173
+ },
1174
+ "interceptor.actionGuard.notify.openclaw": {
1175
+ label: "Notify via OpenClaw",
1176
+ help: "Deliver through the gateway's own channel where the runtime provides that seam.",
1177
+ advanced: true,
1178
+ },
361
1179
  } as const;
362
1180
 
363
1181
  const SEVERITY_ACTION_SCHEMA = {
@@ -376,64 +1194,143 @@ const FAILURE_POLICY_SCHEMA = {
376
1194
  ),
377
1195
  };
378
1196
 
379
- const INTERCEPTOR_JSON_SCHEMA = {
1197
+ /** #225. Mirrored verbatim into openclaw.plugin.json's configSchema — the host
1198
+ * validates the on-disk config against THAT file, so a posture accepted here
1199
+ * and absent there is a config an operator writes from our own docs and the
1200
+ * gateway rejects. */
1201
+ const CONVERSATION_JSON_SCHEMA = {
1202
+ type: "object",
1203
+ additionalProperties: false,
1204
+ properties: {
1205
+ posture: {
1206
+ type: "string",
1207
+ enum: [...CONVERSATION_POSTURES],
1208
+ default: "observe",
1209
+ description:
1210
+ "off = do not scan the conversation; observe = scan, audit and alert but never block (default); " +
1211
+ "enforce = block the run on a dirty verdict via before_agent_run.",
1212
+ },
1213
+ },
1214
+ };
1215
+
1216
+ /** #235/#226. Mirrored verbatim into openclaw.plugin.json's configSchema, for
1217
+ * the same reason as the conversation posture above: the host validates the
1218
+ * on-disk config against THAT file, and a key our parser reads but neither
1219
+ * schema declares is one an operator cannot set at all. */
1220
+ const CONVERSATION_TRUST_JSON_SCHEMA = {
1221
+ type: "object",
1222
+ additionalProperties: false,
1223
+ properties: {
1224
+ trustOwnerInput: {
1225
+ type: "boolean",
1226
+ default: true,
1227
+ description:
1228
+ "Default true: a message the host attributes to the gateway OWNER is an instruction, so a detection in it " +
1229
+ "is audited and alerted but never taints the session or blocks the turn. Set false on a host where the owner " +
1230
+ "routinely pastes untrusted content and you would rather have the caution than the quiet. Content from " +
1231
+ "anyone else — including another agent on a trusted channel — is data regardless of this setting.",
1232
+ },
1233
+ },
1234
+ };
1235
+
1236
+ /**
1237
+ * The Action Guard block, declared ONCE and mounted in BOTH places the parser
1238
+ * accepts it (#226).
1239
+ *
1240
+ * `normaliseConfig` has read a TOP-LEVEL `actionGuard` since #209 — that is the
1241
+ * canonical location, and `interceptor.actionGuard` is the deprecated alias
1242
+ * kept for pre-#209 configs. The schemas said the opposite: only the nested
1243
+ * alias was declared, under `additionalProperties: false`, so a config written
1244
+ * from our own documentation — `actionGuard.notify` at the top level — was
1245
+ * rejected as an unknown key by any host that validates against the schema.
1246
+ * The parser would have kept it; the config never reached the parser. That is
1247
+ * the shape behind the original `parsedNotify: null` reproduction.
1248
+ *
1249
+ * One constant, two mount points, so the two can never drift. Mirrored by hand
1250
+ * into openclaw.plugin.json's configSchema (the host validates the on-disk
1251
+ * config against THAT file) and pinned equal by manifest-config-schema-226.test.ts.
1252
+ */
1253
+ const ACTION_GUARD_JSON_SCHEMA = {
380
1254
  type: "object",
381
1255
  additionalProperties: false,
382
1256
  properties: {
383
1257
  enabled: { type: "boolean" },
384
- severityActions: SEVERITY_ACTION_SCHEMA,
385
- failurePolicy: FAILURE_POLICY_SCHEMA,
386
- actionGuard: {
1258
+ enforce: { type: "boolean" },
1259
+ autoApprove: { type: "array", items: { type: "string" } },
1260
+ auditAllows: { type: "boolean" },
1261
+ // #143. Mirrors normaliseBrokerConfig's allowlist; that function still
1262
+ // has the last word, so a value that slips past the schema is still
1263
+ // range-checked (and dropped) before the broker sees it.
1264
+ broker: {
387
1265
  type: "object",
388
1266
  additionalProperties: false,
389
1267
  properties: {
390
1268
  enabled: { type: "boolean" },
391
- enforce: { type: "boolean" },
392
- autoApprove: { type: "array", items: { type: "string" } },
393
- auditAllows: { type: "boolean" },
394
- // #143. Mirrors normaliseBrokerConfig's allowlist; that function still
395
- // has the last word, so a value that slips past the schema is still
396
- // range-checked (and dropped) before the broker sees it.
397
- broker: {
1269
+ allowPreClear: { type: "boolean" },
1270
+ preClearConfidence: { type: "number", minimum: 0.9, maximum: 1 },
1271
+ judgeTimeoutMs: { type: "number", minimum: 500, maximum: 60000 },
1272
+ approvalTimeoutMs: {
398
1273
  type: "object",
399
1274
  additionalProperties: false,
400
1275
  properties: {
401
- enabled: { type: "boolean" },
402
- allowPreClear: { type: "boolean" },
403
- preClearConfidence: { type: "number", minimum: 0.9, maximum: 1 },
404
- judgeTimeoutMs: { type: "number", minimum: 500, maximum: 60000 },
405
- approvalTimeoutMs: {
406
- type: "object",
407
- additionalProperties: false,
408
- properties: {
409
- sensitive: { type: "number", minimum: 1000, maximum: 3600000 },
410
- dangerous: { type: "number", minimum: 1000, maximum: 3600000 },
411
- },
412
- },
413
- model: { type: "string" },
1276
+ sensitive: { type: "number", minimum: 1000, maximum: 3600000 },
1277
+ dangerous: { type: "number", minimum: 1000, maximum: 3600000 },
414
1278
  },
415
1279
  },
416
- // #189. Each entry pins one script by absolute path + content hash;
417
- // createReviewedScriptCheck has the last word on every field.
418
- reviewedScripts: {
419
- type: "array",
420
- items: {
421
- type: "object",
422
- additionalProperties: false,
423
- properties: {
424
- path: { type: "string" },
425
- sha256: { type: "string" },
426
- note: { type: "string" },
427
- addedAt: { type: "number" },
428
- },
429
- required: ["path", "sha256"],
430
- },
1280
+ model: { type: "string" },
1281
+ },
1282
+ },
1283
+ // #143/#225. Mirrors NotifyConfig in notify-config.ts, which still has
1284
+ // the last word (strict-true booleans, http(s)-only URL, bounded
1285
+ // timeout). Declared here because `additionalProperties: false` above
1286
+ // means an undeclared key makes the WHOLE containing block invalid on
1287
+ // a host that validates config against this schema — which is how the
1288
+ // Action Guard's notify transport came to be unusable from inside the
1289
+ // gateway plugin at all.
1290
+ notify: {
1291
+ type: "object",
1292
+ additionalProperties: false,
1293
+ properties: {
1294
+ enabled: { type: "boolean" },
1295
+ webhookUrl: { type: "string" },
1296
+ webhookSecret: { type: "string" },
1297
+ openclaw: { type: "boolean" },
1298
+ timeoutMs: { type: "number", minimum: 500, maximum: 60000 },
1299
+ },
1300
+ },
1301
+ // #189. Each entry pins one script by absolute path + content hash;
1302
+ // createReviewedScriptCheck has the last word on every field.
1303
+ reviewedScripts: {
1304
+ type: "array",
1305
+ items: {
1306
+ type: "object",
1307
+ additionalProperties: false,
1308
+ properties: {
1309
+ path: { type: "string" },
1310
+ sha256: { type: "string" },
1311
+ note: { type: "string" },
1312
+ addedAt: { type: "number" },
431
1313
  },
1314
+ required: ["path", "sha256"],
432
1315
  },
433
1316
  },
434
1317
  },
435
1318
  };
436
1319
 
1320
+ const INTERCEPTOR_JSON_SCHEMA = {
1321
+ type: "object",
1322
+ additionalProperties: false,
1323
+ properties: {
1324
+ enabled: { type: "boolean" },
1325
+ severityActions: SEVERITY_ACTION_SCHEMA,
1326
+ failurePolicy: FAILURE_POLICY_SCHEMA,
1327
+ conversation: CONVERSATION_JSON_SCHEMA,
1328
+ /** The DEPRECATED alias (#209). Still accepted, still parsed, still
1329
+ * gap-fills the canonical top-level block key by key. */
1330
+ actionGuard: ACTION_GUARD_JSON_SCHEMA,
1331
+ },
1332
+ };
1333
+
437
1334
  const PLUGIN_CONFIG_JSON_SCHEMA = {
438
1335
  type: "object",
439
1336
  additionalProperties: false,
@@ -456,6 +1353,15 @@ const PLUGIN_CONFIG_JSON_SCHEMA = {
456
1353
  // #112: without this, `additionalProperties: false` declared the whole
457
1354
  // interceptor block invalid — mirror openclaw.plugin.json's configSchema.
458
1355
  interceptor: INTERCEPTOR_JSON_SCHEMA,
1356
+ // #209/#226: the CANONICAL Action Guard location. normaliseConfig has read
1357
+ // it here since #209 and folds it over the nested alias; the schema did not
1358
+ // declare it, so `additionalProperties: false` rejected the documented
1359
+ // config shape before the parser ever saw it.
1360
+ actionGuard: ACTION_GUARD_JSON_SCHEMA,
1361
+ // #235/#226: source trust. Same reason as `actionGuard` above — the parser
1362
+ // reads it, so the schema must declare it or `additionalProperties: false`
1363
+ // rejects the whole config an operator writes from our documentation.
1364
+ conversationTrust: CONVERSATION_TRUST_JSON_SCHEMA,
459
1365
  },
460
1366
  };
461
1367
 
@@ -494,6 +1400,17 @@ let _registered = false;
494
1400
  // unattended Codex agents even a no-op registered hook changes how OpenClaw
495
1401
  // resolves approvals).
496
1402
  let _beforeToolCallRegistered = false;
1403
+ // #225: whether we CALLED api.on('before_agent_run', …) this session — nothing
1404
+ // more. It is deliberately not named "…Accepted": `api.on` returns void, never
1405
+ // throws on an unknown hook name, and never throws when the host refuses a
1406
+ // conversation hook (it records a diagnostic and returns), so acceptance is not
1407
+ // observable from in here. The two facts that decide whether the plane is live
1408
+ // — host build ≥ the gate's floor, and the operator's allowConversationAccess
1409
+ // grant — are read separately and reported by describeConversationPlane().
1410
+ let _beforeAgentRunRequested = false;
1411
+ // The operator's conversation-access grant as read from the host config at
1412
+ // register() time. Reported by /shieldcortex-status; never written by us.
1413
+ let _conversationAccessGranted = false;
497
1414
  // #134 §2: register() wraps its whole body in try/catch so a plugin failure
498
1415
  // never blocks channel startup — correct, but it used to report the failure
499
1416
  // with a bare console.warn (bypasses the gateway's structured log, so the
@@ -549,6 +1466,20 @@ function normaliseConfig(raw: unknown, dropped?: string[]): SCConfig {
549
1466
  const interceptor = normaliseInterceptorConfig(value.interceptor, dropped);
550
1467
  if (interceptor) config.interceptor = interceptor;
551
1468
 
1469
+ // Source trust (#235, wired to the gate in #226). Booleans only, and the
1470
+ // block is kept only when it holds a valid one: a `trustOwnerInput: "false"`
1471
+ // typo — the exact shape of the #112 incident — must not read as the opt-out
1472
+ // having been applied, because the operator who wrote it believes owner input
1473
+ // is being policed and it would not be.
1474
+ if (value.conversationTrust && typeof value.conversationTrust === "object" && !Array.isArray(value.conversationTrust)) {
1475
+ const trustRaw = value.conversationTrust as Record<string, unknown>;
1476
+ if (typeof trustRaw.trustOwnerInput === "boolean") {
1477
+ config.conversationTrust = { trustOwnerInput: trustRaw.trustOwnerInput };
1478
+ } else if (trustRaw.trustOwnerInput !== undefined) {
1479
+ dropped?.push("conversationTrust.trustOwnerInput");
1480
+ }
1481
+ }
1482
+
552
1483
  // #209: single source of truth for the Action Guard. A top-level
553
1484
  // `actionGuard` block governs every surface; `interceptor.actionGuard` is a
554
1485
  // deprecated alias kept as per-key gap-fill so pre-#209 configs keep their
@@ -650,6 +1581,19 @@ function normaliseActionGuardBlock(
650
1581
  if (Array.isArray(rawGuard.reviewedScripts)) {
651
1582
  guard.reviewedScripts = [...rawGuard.reviewedScripts];
652
1583
  }
1584
+ // #143/#225: the notify transport. Same passthrough discipline again —
1585
+ // normaliseNotifyConfig is the boundary. Shallow-copied rather than aliased
1586
+ // (the #115 reason: a later in-place mutation of the host config object must
1587
+ // not reach into the normalised one), and a non-object is DROPPED by name so
1588
+ // the #115 warn log can say which key was ignored, rather than silently
1589
+ // leaving the operator with a transport that never fires.
1590
+ if (rawGuard.notify !== undefined) {
1591
+ if (rawGuard.notify && typeof rawGuard.notify === 'object' && !Array.isArray(rawGuard.notify)) {
1592
+ guard.notify = { ...(rawGuard.notify as Record<string, unknown>) };
1593
+ } else {
1594
+ dropped?.push(`${pathPrefix}.notify`);
1595
+ }
1596
+ }
653
1597
  return Object.keys(guard).length > 0 ? guard : undefined;
654
1598
  }
655
1599
 
@@ -672,6 +1616,24 @@ function normaliseInterceptorConfig(raw: unknown, dropped?: string[]): Intercept
672
1616
  const actionGuard = normaliseActionGuardBlock(value.actionGuard, dropped, "interceptor.actionGuard");
673
1617
  if (actionGuard) out.actionGuard = actionGuard;
674
1618
 
1619
+ // #225: conversation posture. An invalid value is DROPPED (and named in the
1620
+ // warn log) rather than coerced, so `conversationPosture()` falls back to
1621
+ // `observe` — the safe direction. A typo must never start blocking turns.
1622
+ if (value.conversation !== undefined) {
1623
+ if (value.conversation && typeof value.conversation === "object" && !Array.isArray(value.conversation)) {
1624
+ const posture = (value.conversation as { posture?: unknown }).posture;
1625
+ if (posture !== undefined) {
1626
+ if (typeof posture === "string" && (CONVERSATION_POSTURES as readonly string[]).includes(posture)) {
1627
+ out.conversation = { posture: posture as ConversationPosture };
1628
+ } else {
1629
+ dropped?.push("interceptor.conversation.posture");
1630
+ }
1631
+ }
1632
+ } else {
1633
+ dropped?.push("interceptor.conversation");
1634
+ }
1635
+ }
1636
+
675
1637
  // #115: empty/all-invalid normalises to undefined, not {} — {} is truthy
676
1638
  // and made applyPluginConfigOverride treat a no-op interceptor block as a
677
1639
  // real override, inconsistent with normaliseSeverityMap's own contract.
@@ -684,6 +1646,19 @@ function normaliseInterceptorConfig(raw: unknown, dropped?: string[]): Intercept
684
1646
  * - `interceptor` deep-merges PER KEY — per severity entry, per guard flag —
685
1647
  * so an override that sets one nested key does not wholesale-discard the
686
1648
  * base's other interceptor settings.
1649
+ * - `actionGuard.notify` and `actionGuard.broker` deep-merge per key too
1650
+ * (#226). They are the two OBJECT-valued guard keys, and a shallow spread
1651
+ * replaced them wholesale: an openclaw.json entry that says only
1652
+ * `notify: { enabled: true }` — the shape the UI writes when an operator
1653
+ * ticks "Operator Notifications" — discarded the shield config's
1654
+ * `webhookUrl` and `webhookSecret`, leaving notify ARMED with no channel and
1655
+ * no signing key. Every alert then reported "enabled but no channel is
1656
+ * configured/buildable on this host", which is the #143 silent-no-sink
1657
+ * failure in a new place.
1658
+ * - ARRAY-valued keys (`autoApprove`, `reviewedScripts`) still REPLACE. They
1659
+ * are allowlists: merging two of them would union permissions an operator
1660
+ * removed back into the effective config, which is the wrong direction for a
1661
+ * security control.
687
1662
  * - Explicit values, including `false`, always win over base values; absent
688
1663
  * keys fall through to the base.
689
1664
  * - Defaults are NOT applied here: DEFAULT_INTERCEPTOR_CONFIG only fills the
@@ -692,6 +1667,13 @@ function normaliseInterceptorConfig(raw: unknown, dropped?: string[]): Intercept
692
1667
  */
693
1668
  function mergeConfigs(base: SCConfig, override: SCConfig): SCConfig {
694
1669
  const merged: SCConfig = { ...base, ...override };
1670
+ // Per-key, like every other nested block here. A plain spread would let an
1671
+ // openclaw.json entry that mentions `conversationTrust` at all replace the
1672
+ // shield-config block wholesale, silently reverting an opt-out set in the
1673
+ // file the operator considers authoritative.
1674
+ if (base.conversationTrust || override.conversationTrust) {
1675
+ merged.conversationTrust = { ...base.conversationTrust, ...override.conversationTrust };
1676
+ }
695
1677
  if (base.interceptor || override.interceptor) {
696
1678
  const b = base.interceptor ?? {};
697
1679
  const o = override.interceptor ?? {};
@@ -703,7 +1685,12 @@ function mergeConfigs(base: SCConfig, override: SCConfig): SCConfig {
703
1685
  merged.interceptor.failurePolicy = { ...b.failurePolicy, ...o.failurePolicy };
704
1686
  }
705
1687
  if (b.actionGuard || o.actionGuard) {
706
- merged.interceptor.actionGuard = { ...b.actionGuard, ...o.actionGuard };
1688
+ const bg = b.actionGuard ?? {};
1689
+ const og = o.actionGuard ?? {};
1690
+ const guard: NonNullable<InterceptorUserConfig['actionGuard']> = { ...bg, ...og };
1691
+ if (bg.notify || og.notify) guard.notify = { ...bg.notify, ...og.notify };
1692
+ if (bg.broker || og.broker) guard.broker = { ...bg.broker, ...og.broker };
1693
+ merged.interceptor.actionGuard = guard;
707
1694
  }
708
1695
  }
709
1696
  return merged;
@@ -750,8 +1737,60 @@ function applyPluginConfigOverride(api: PluginApi): void {
750
1737
  _lastShieldConfigRef = null;
751
1738
  }
752
1739
 
1740
+ /**
1741
+ * Load the effective config, DEGRADING rather than throwing (#226).
1742
+ *
1743
+ * `getRuntime()` resolves `shieldcortex/dist/…/runtime.mjs` by walking a list of
1744
+ * install locations, and `loadShieldConfig()` then reads a file. Both can fail
1745
+ * for ordinary reasons — the package was upgraded underneath a running gateway,
1746
+ * a global install moved, `~/.shieldcortex/config.json` is half-written.
1747
+ *
1748
+ * This used to propagate, and the propagation went somewhere bad: every caller
1749
+ * of `loadConfig` is a hook body, and in `handleBeforeAgentRun` the throw was
1750
+ * caught by the OUTER catch — the one that fails open. So a host that could not
1751
+ * load the runtime produced one console line per turn and NOTHING else: no
1752
+ * posture, no scan, no audit row, no alert. Precisely the "unprotected turn that
1753
+ * leaves no trace" the #225/#226 work exists to eliminate, reintroduced through
1754
+ * the config read rather than through the scanner.
1755
+ *
1756
+ * Now it degrades to the plugin config from openclaw.json (already normalised by
1757
+ * `applyPluginConfigOverride`), or to an empty config. The posture therefore
1758
+ * still resolves, the scan still runs, `scanRealtimeContent` reports UNAVAILABLE
1759
+ * on its own (the same runtime failure defeats the MCP fallback), and the gate
1760
+ * writes its normal audit row and raises its normal alert.
1761
+ *
1762
+ * It does NOT cache the degraded result — a later successful load must take
1763
+ * effect without a restart — and it never claims the shield config loaded: the
1764
+ * warning says exactly what is missing, and is bounded, redacted, and emitted
1765
+ * ONCE per plugin load (`__resetConfigStateForTest` re-arms it) so a per-turn
1766
+ * failure cannot become per-turn log spam.
1767
+ */
1768
+ let _shieldConfigLoadFailureLogged = false;
1769
+
753
1770
  async function loadConfig(): Promise<SCConfig> {
754
- const shieldConfigRaw = await (await getRuntime()).loadShieldConfig();
1771
+ let shieldConfigRaw: unknown;
1772
+ try {
1773
+ shieldConfigRaw = await (await getRuntime()).loadShieldConfig();
1774
+ } catch (err) {
1775
+ if (!_shieldConfigLoadFailureLogged) {
1776
+ _shieldConfigLoadFailureLogged = true;
1777
+ const detail = redactNotifyDetail(err instanceof Error ? err.message : String(err)).slice(0, 300);
1778
+ console.warn(
1779
+ '[shieldcortex] ⚠️ shield config could NOT be loaded — the ShieldCortex runtime did not resolve, ' +
1780
+ `or it could not read ~/.shieldcortex/config.json (${detail}). Continuing with the openclaw.json ` +
1781
+ 'plugin config only; anything configured in the shield config file is NOT in effect, and ' +
1782
+ 'conversation scanning will report UNAVAILABLE until this is fixed. (Logged once per plugin load.)',
1783
+ );
1784
+ }
1785
+ // A fresh object every time: `_configOverride` is module state and callers
1786
+ // must not be handed something they could mutate.
1787
+ return mergeConfigs({}, _configOverride ?? {});
1788
+ }
1789
+ // A load that succeeds after a failure re-arms the warning, so a SECOND
1790
+ // outage is reported rather than swallowed by the first one's flag. Set
1791
+ // before the cache check: a runtime that hands back the same object every
1792
+ // call would otherwise take the early return and leave the flag latched.
1793
+ _shieldConfigLoadFailureLogged = false;
755
1794
  if (_config && shieldConfigRaw === _lastShieldConfigRef) return _config;
756
1795
  _lastShieldConfigRef = shieldConfigRaw;
757
1796
  // Plugin config (openclaw.json) deep-merges over the shield config file —
@@ -790,36 +1829,130 @@ function parseScanResponse(response: string): { clean: boolean; summary: string
790
1829
  return { clean, summary };
791
1830
  }
792
1831
 
793
- export async function scanRealtimeContent(text: string): Promise<{ clean: boolean; summary: string }> {
1832
+ /**
1833
+ * Scan one piece of conversation content.
1834
+ *
1835
+ * The contract changed in #225 and the change is the point: an unavailable
1836
+ * scanner is now reported as `available: false, clean: false`, not as the old
1837
+ * `{ clean: true, summary: 'scan unavailable' }`. That old return was the
1838
+ * quietest bug in this file — the ORDINARY unavailable path (MCP fallback
1839
+ * returns nothing, e.g. no shieldcortex binary on PATH) manufactured a clean
1840
+ * verdict, so on any box where in-process defence failed to load, every message
1841
+ * was reported scanned-and-fine while nothing had been looked at.
1842
+ *
1843
+ * Fails OPEN — callers must not block on `available: false` — but LOUDLY: the
1844
+ * caller audits it, alerts on it, and doctor/status report the plane as
1845
+ * unavailable rather than protected.
1846
+ */
1847
+ export async function scanRealtimeContent(text: string): Promise<ConversationScanResult> {
794
1848
  // PRIMARY: scan in-process via the shared shieldcortex/defence module. The
795
1849
  // scan is pure (no DB handle required — scanToolResponse's audit write is
796
1850
  // guarded by isDatabaseInitialized()), so it is safe in the long-lived
797
1851
  // gateway and avoids booting a cold MCP server per message.
798
- const defenceMod = await getDefenceModule();
1852
+ let defenceMod: DefenceModule | null = null;
1853
+ try {
1854
+ defenceMod = await getDefenceModule();
1855
+ } catch (err) {
1856
+ defenceMod = null;
1857
+ void err;
1858
+ }
799
1859
  if (defenceMod && typeof defenceMod.scanToolResponse === "function") {
800
- const scan = defenceMod.scanToolResponse("openclaw-realtime", text, "advisory");
801
- // Reproduce the historical summary contract exactly: risk level + detection
802
- // count only when the injection scan flagged something.
803
- const risk = scan.injection.clean ? "unknown" : scan.injection.riskLevel;
804
- const summary = scan.injection.clean
805
- ? risk
806
- : `${risk} (${scan.injection.detections.length} detections)`;
807
- return { clean: scan.clean, summary };
1860
+ try {
1861
+ const scan = defenceMod.scanToolResponse("openclaw-realtime", text, "advisory");
1862
+ // Reproduce the historical summary contract exactly: risk level + detection
1863
+ // count only when the injection scan flagged something.
1864
+ const risk = scan.injection.clean ? "unknown" : scan.injection.riskLevel;
1865
+ const summary = scan.injection.clean
1866
+ ? risk
1867
+ : `${risk} (${scan.injection.detections.length} detections)`;
1868
+ return { clean: scan.clean, summary, available: true };
1869
+ } catch (err) {
1870
+ // A scanner that THROWS is not a clean verdict either. Same treatment as
1871
+ // an absent one: unavailable, reported, never silently allowed to read as
1872
+ // protected.
1873
+ const detail = err instanceof Error ? err.message : String(err);
1874
+ return { clean: false, available: false, errored: true, error: `in-process scanner threw: ${detail}`, summary: "scan unavailable" };
1875
+ }
808
1876
  }
809
1877
 
810
1878
  // FALLBACK: in-process defence unavailable (older install, import failed) —
811
1879
  // degrade to the MCP shell-out so scanning still happens rather than breaking.
812
- const response = await callCortex("scan_tool_response", {
813
- toolName: "openclaw-realtime",
814
- content: text,
815
- mode: "advisory",
816
- });
1880
+ let response: string | null = null;
1881
+ try {
1882
+ response = await callCortex("scan_tool_response", {
1883
+ toolName: "openclaw-realtime",
1884
+ content: text,
1885
+ mode: "advisory",
1886
+ });
1887
+ } catch (err) {
1888
+ const detail = err instanceof Error ? err.message : String(err);
1889
+ return { clean: false, available: false, errored: true, error: `scan fallback failed: ${detail}`, summary: "scan unavailable" };
1890
+ }
817
1891
 
818
1892
  if (!response) {
819
- return { clean: true, summary: "scan unavailable" };
1893
+ return {
1894
+ clean: false,
1895
+ available: false,
1896
+ errored: true,
1897
+ error: 'no in-process defence module and the MCP fallback returned nothing',
1898
+ summary: 'scan unavailable',
1899
+ };
820
1900
  }
821
1901
 
822
- return parseScanResponse(response);
1902
+ const parsed = parseScanResponse(response);
1903
+ return { ...parsed, available: true };
1904
+ }
1905
+
1906
+ /**
1907
+ * `scanRealtimeContent` with a hard deadline (#226).
1908
+ *
1909
+ * Used ONLY by the `before_agent_run` gate, which the gateway awaits: an
1910
+ * unbounded scan there is an unbounded pause in front of the user's prompt. The
1911
+ * MCP fallback boots a cold server through `npx` and has been measured at ~15s,
1912
+ * so "it usually returns quickly" is not a bound.
1913
+ *
1914
+ * On expiry the result is an ordinary UNAVAILABLE verdict — fail open, audited,
1915
+ * alerted — and the error string names the deadline and NOTHING ELSE. It must
1916
+ * never quote the prompt: a timeout message is the one error a developer is
1917
+ * most likely to paste into an issue.
1918
+ *
1919
+ * The losing promise is not abandoned silently: a `.catch` is attached before
1920
+ * the race so a scan that rejects AFTER the deadline settles into a no-op
1921
+ * instead of an unhandled rejection that could take the gateway down under
1922
+ * `--unhandled-rejections=strict`.
1923
+ */
1924
+ export async function scanWithDeadline(
1925
+ text: string,
1926
+ timeoutMs: number = CONVERSATION_SCAN_MAX_MS,
1927
+ ): Promise<ConversationScanResult> {
1928
+ const timedOut: ConversationScanResult = {
1929
+ clean: false,
1930
+ available: false,
1931
+ errored: true,
1932
+ error: `conversation scan exceeded its ${timeoutMs}ms deadline`,
1933
+ summary: 'scan unavailable',
1934
+ };
1935
+
1936
+ const scan = scanRealtimeContent(text);
1937
+ // Attached BEFORE the race, so a late rejection can never be unhandled.
1938
+ scan.catch(() => { /* the race already answered; nothing left to report */ });
1939
+
1940
+ let timer: ReturnType<typeof setTimeout> | undefined;
1941
+ try {
1942
+ return await Promise.race([
1943
+ scan,
1944
+ new Promise<ConversationScanResult>((resolve) => {
1945
+ timer = setTimeout(() => resolve(timedOut), timeoutMs);
1946
+ // Never hold the process open on the deadline timer alone.
1947
+ (timer as { unref?: () => void }).unref?.();
1948
+ }),
1949
+ ]);
1950
+ } catch (err) {
1951
+ const detail = err instanceof Error ? err.message : String(err);
1952
+ return { clean: false, available: false, errored: true, error: detail, summary: 'scan unavailable' };
1953
+ } finally {
1954
+ if (timer) clearTimeout(timer);
1955
+ }
823
1956
  }
824
1957
 
825
1958
  // ==================== CONTENT PATTERNS ====================
@@ -869,20 +2002,56 @@ function extractUserContent(msgs: unknown[]): string[] {
869
2002
  return out;
870
2003
  }
871
2004
 
872
- const AUDIT_DIR = path.join(homedir(), ".shieldcortex", "audit");
2005
+ /** Where the realtime audit jsonl lives.
2006
+ *
2007
+ * Resolved PER CALL, and honouring `SHIELDCORTEX_AUDIT_DIR`, so a test can
2008
+ * exercise the hook end-to-end without appending fabricated "threat" rows to
2009
+ * a real box's security audit trail. Default is unchanged
2010
+ * (`~/.shieldcortex/audit`), so nothing moves for an install that does not set
2011
+ * the variable. */
2012
+ function auditDir(): string {
2013
+ const override = process.env.SHIELDCORTEX_AUDIT_DIR;
2014
+ if (override && override.trim()) return override.trim();
2015
+ return path.join(homedir(), ".shieldcortex", "audit");
2016
+ }
873
2017
  const NOVELTY_CACHE_FILE = path.join(homedir(), ".shieldcortex", "openclaw-memory-cache.json");
874
2018
  const DEFAULT_NOVELTY_THRESHOLD = 0.88;
875
2019
  const DEFAULT_MAX_RECENT = 300;
876
2020
  const MIN_NOVELTY_CHARS = 40;
877
2021
 
878
- async function auditLog(entry: Record<string, unknown>) {
2022
+ /**
2023
+ * Append one row to the realtime audit jsonl. Returns WHETHER IT LANDED (#226).
2024
+ *
2025
+ * This used to swallow every failure and return void, so a caller could not
2026
+ * distinguish "the evidence is on disk" from "the disk is full / the audit dir
2027
+ * is not writable / it is a file where a directory should be". The
2028
+ * `before_agent_run` gate then wrote a decision row, got nothing back, and
2029
+ * proceeded to tell the operator — and the delivery row — that the decision was
2030
+ * recorded. A security control claiming evidence it does not have is worse than
2031
+ * one that admits the gap, because the gap is invisible in exactly the incident
2032
+ * where the log matters.
2033
+ *
2034
+ * Still never throws: a broken audit sink must not become a broken turn.
2035
+ */
2036
+ async function auditLog(entry: Record<string, unknown>): Promise<boolean> {
2037
+ const dir = auditDir();
879
2038
  try {
880
- await fs.mkdir(AUDIT_DIR, { recursive: true });
2039
+ await fs.mkdir(dir, { recursive: true });
881
2040
  await fs.appendFile(
882
- path.join(AUDIT_DIR, `realtime-${new Date().toISOString().slice(0, 10)}.jsonl`),
2041
+ path.join(dir, `realtime-${new Date().toISOString().slice(0, 10)}.jsonl`),
883
2042
  JSON.stringify(entry) + "\n",
884
2043
  );
885
- } catch {}
2044
+ return true;
2045
+ } catch (err) {
2046
+ // LOUD. The detail is the failure and the directory — never the row, which
2047
+ // may carry a verdict summary, and never a credential (nothing in this path
2048
+ // holds one). Bounded so a pathological error message cannot flood stderr.
2049
+ const detail = err instanceof Error ? err.message : String(err);
2050
+ console.error(
2051
+ `[shieldcortex] ⚠️ AUDIT WRITE FAILED (${dir}) — this event is NOT on disk: ${detail.slice(0, 300)}`,
2052
+ );
2053
+ return false;
2054
+ }
886
2055
  }
887
2056
 
888
2057
  // `cloudSync` lives in ./cloud-sync.ts (no fs imports there) so the plugin
@@ -1052,6 +2221,17 @@ function isInternalContent(text: string): boolean {
1052
2221
  // itself stays non-blocking.
1053
2222
  export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promise<void> {
1054
2223
  try {
2224
+ // #226: THE POSTURE GOVERNS THIS HOOK TOO. `off` means "do not scan the
2225
+ // conversation at all", and this observation hook used to ignore it
2226
+ // entirely: it scanned every prompt, wrote threat and scan_unavailable rows,
2227
+ // and forwarded detections to the cloud on a box whose operator had
2228
+ // explicitly turned conversation inspection OFF. The gate honoured the
2229
+ // setting, so a reader of `handleBeforeAgentRun` would conclude the product
2230
+ // did too. Read it FIRST, before any scanner, any audit row and any cloud
2231
+ // call, so `off` costs exactly one config read.
2232
+ const cfg = await loadConfig();
2233
+ if (conversationPosture(cfg.interceptor?.conversation) === 'off') return;
2234
+
1055
2235
  // Only scan user content, skip system/boot/heartbeat prompts
1056
2236
  // Trust is resolved per TURN, not per message: the host tells us who sent
1057
2237
  // this turn, but history messages carry no individual attribution, so there
@@ -1060,13 +2240,16 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1060
2240
  // Resolved LAZILY, on first detection only. Computing it up front would put
1061
2241
  // a config read on every single turn to answer a question that only matters
1062
2242
  // when something is actually found.
1063
- let trustMemo: ReturnType<typeof classifyConversationOrigin> | null = null;
2243
+ let trustMemo: ConversationTrustDecision | null = null;
1064
2244
  const resolveTrust = async () => {
1065
2245
  if (!trustMemo) {
2246
+ // `cfg` is already in hand from the posture read above, so this costs
2247
+ // no I/O. It also reads the PARSED field rather than casting the config
2248
+ // object — the cast this replaces asserted a key that normaliseConfig
2249
+ // drops, so it was always undefined. See SCConfig.conversationTrust.
1066
2250
  trustMemo = classifyConversationOrigin({
1067
2251
  senderIsOwner: event.senderIsOwner,
1068
- trustOwnerInput: (await loadConfig() as { conversationTrust?: { trustOwnerInput?: boolean } })
1069
- ?.conversationTrust?.trustOwnerInput,
2252
+ trustOwnerInput: cfg.conversationTrust?.trustOwnerInput,
1070
2253
  });
1071
2254
  }
1072
2255
  return trustMemo;
@@ -1076,6 +2259,35 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1076
2259
  for (const text of texts) {
1077
2260
  if (!text || text.length < 10) continue;
1078
2261
  const result = await scanRealtimeContent(text);
2262
+ // #225: "we could not look" is its own outcome. Before this branch the
2263
+ // unavailable path returned clean:true and this loop did nothing at all —
2264
+ // an unscanned message was indistinguishable from a scanned one, on the
2265
+ // observation hook as well as the gate.
2266
+ if (!result.available) {
2267
+ // #226: redacted on the CONSOLE too, not only in the row. The reason
2268
+ // comes from a transport/scanner failure string, which can name the
2269
+ // endpoint it failed to reach — and a gateway's stdout is routinely
2270
+ // shipped to a log aggregator, so "ephemeral" is not a property the
2271
+ // console actually has.
2272
+ const detail = redactNotifyDetail(result.error ?? result.summary);
2273
+ console.warn(
2274
+ `[shieldcortex] ⚠️ conversation scan UNAVAILABLE (${detail}) — this message was NOT scanned`,
2275
+ );
2276
+ // AWAITED. The whole function is already fire-and-forget from
2277
+ // `handleLlmInput`, so this blocks nothing the gateway is waiting on —
2278
+ // and it means the row is on disk before the loop moves to the next
2279
+ // message, and that `auditLog`'s new boolean (which logs loudly on
2280
+ // failure) is actually reached rather than discarded into a floating
2281
+ // promise.
2282
+ await auditLog({
2283
+ type: 'scan_unavailable', hook: 'llm_input', sessionId: event.sessionId,
2284
+ model: event.model, reason: detail,
2285
+ chars: text.length,
2286
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2287
+ ts: new Date().toISOString(),
2288
+ });
2289
+ continue;
2290
+ }
1079
2291
  if (!result.clean) {
1080
2292
  const trust = await resolveTrust();
1081
2293
  console.warn(`[shieldcortex] ⚠️ Threat in LLM input: ${result.summary} [${trust.origin}]`);
@@ -1085,7 +2297,7 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1085
2297
  // the Action Guard tighten by one notch for a bounded window.
1086
2298
  //
1087
2299
  // Source trust gates the CONSEQUENCE, never the detection: the warn and
1088
- // the audit row above happen whoever sent this. What trust decides is
2300
+ // the audit row below happen whoever sent this. What trust decides is
1089
2301
  // whether it may tighten the guard. The operator typing "delete the old
1090
2302
  // logs" is an instruction, and treating it as an attack is the false
1091
2303
  // alarm that gets a control switched off. Everything the agent was
@@ -1093,17 +2305,28 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1093
2305
  if (trust.mayTaint) {
1094
2306
  sessionTaint.mark(event.sessionId, { reason: `conversation scan: ${result.summary}` });
1095
2307
  }
2308
+ // #226: NO `preview`. This row carried the first 100 characters of the
2309
+ // prompt — the exact text that tripped an injection detector, i.e.
2310
+ // hostile by assumption — into an append-only file that syncs. The gate
2311
+ // on the very next hook has recorded only `chars` + `contentSha256`
2312
+ // since #225 and says in its own comment that the prompt is never
2313
+ // persisted; the observation hook quietly did the opposite, so the
2314
+ // claim was false on the path that runs on every single turn. Length
2315
+ // plus digest keeps the row correlatable with the gate's row for the
2316
+ // same text without storing the text.
1096
2317
  const entry = {
1097
2318
  type: "threat", hook: "llm_input", sessionId: event.sessionId,
1098
2319
  model: event.model, reason: result.summary,
1099
- preview: text.slice(0, 100), ts: new Date().toISOString(),
2320
+ chars: text.length,
2321
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2322
+ ts: new Date().toISOString(),
1100
2323
  };
1101
- auditLog(entry);
2324
+ await auditLog(entry);
1102
2325
  loadConfig()
1103
2326
  // Pass the local entry as-is; cloudSync rebuilds a canonical metadata-only
1104
- // entry from named fields and never reads preview/content. No raw LLM input
2327
+ // entry from named fields and never reads content. No raw LLM input
1105
2328
  // leaves here.
1106
- .then(cfg => cloudSync(entry, cfg))
2329
+ .then(cfg2 => cloudSync(entry, cfg2))
1107
2330
  .catch(() => {});
1108
2331
  }
1109
2332
  }
@@ -1113,10 +2336,749 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1113
2336
  }
1114
2337
 
1115
2338
  function handleLlmInput(event: LlmInputEvent, ctx: AgentCtx): void {
1116
- // Fire and forget
2339
+ // Fire and forget — OBSERVATION ONLY. This hook cannot block (#225); the
2340
+ // enforcement point is handleBeforeAgentRun below.
1117
2341
  void scanLlmInput(event, ctx);
1118
2342
  }
1119
2343
 
2344
+ /**
2345
+ * Route a conversation-threat detection to a HUMAN (#225).
2346
+ *
2347
+ * This is the "sink" the issue is named for. Before it existed, a HIGH verdict
2348
+ * produced a console line and an audit row, and a real detection on a live box
2349
+ * was seen by nobody. It reuses the Action Guard's notify transport (#143)
2350
+ * rather than inventing a second one — two notification paths would drift, and
2351
+ * the operator would learn which one to ignore.
2352
+ *
2353
+ * Returns whether a human was actually reached, so callers (and doctor) can
2354
+ * report "detected but undeliverable" instead of implying someone was told.
2355
+ * Never throws: a failed notification must not become a failed turn.
2356
+ */
2357
+ /** The longest the conversation gate will wait for an operator alert to be
2358
+ * handed to a transport. See the call site: the user's turn is blocked on this
2359
+ * hook, so the alert's deadline has to be a fraction of the hook's. */
2360
+ const CONVERSATION_NOTIFY_MAX_MS = 5_000;
2361
+
2362
+ /**
2363
+ * The longest the conversation gate will wait for a SCAN (#226).
2364
+ *
2365
+ * `before_agent_run` is awaited by the gateway — the user's turn is stopped
2366
+ * dead until this handler returns — and the scan's fallback path is an MCP
2367
+ * shell-out that boots a cold server via `npx`, which can take upwards of 15s.
2368
+ * That is half the hook's entire 30s budget before an alert and two audit
2369
+ * writes are added behind it, and it happens on exactly the hosts where the
2370
+ * in-process defence module failed to load: the ones already degraded.
2371
+ *
2372
+ * Past this deadline the scan is treated as UNAVAILABLE, which fails OPEN (the
2373
+ * turn proceeds) and is audited and alerted like any other unavailable scan. A
2374
+ * security control that silently adds fifteen seconds to every prompt is one an
2375
+ * operator uninstalls.
2376
+ */
2377
+ export const CONVERSATION_SCAN_MAX_MS = 5_000;
2378
+
2379
+ /**
2380
+ * Repeat-alert suppression for the scan-unavailable path (#226).
2381
+ *
2382
+ * An unavailable scanner is not a transient event: it is usually a missing
2383
+ * install, a broken defence build or an absent binary, and it recurs on EVERY
2384
+ * turn. Alerting per turn turns the operator's phone into a metronome and the
2385
+ * alert into something they mute — which is the same outcome as never sending
2386
+ * one, reached by a more expensive route. First occurrence goes immediately;
2387
+ * after that, at most one alert per window, with the suppressed count carried
2388
+ * on the next alert that does go out so nothing is lost.
2389
+ *
2390
+ * THE WINDOW IS PER SESSION, not per process — see `noteScanUnavailable`. A
2391
+ * gateway runs many sessions at once, and "we already told you about session A"
2392
+ * is not a reason to stay silent about session B.
2393
+ *
2394
+ * AUDITING IS NOT RATE LIMITED. Every occurrence still writes its row, and the
2395
+ * row records whether an alert was suppressed and how many have been seen —
2396
+ * the evidence trail must be complete even when the notification stream is not.
2397
+ */
2398
+ export const SCAN_UNAVAILABLE_ALERT_WINDOW_MS = 5 * 60_000;
2399
+
2400
+ interface ScanUnavailableAlertState {
2401
+ /** Total occurrences this session, across suppressed and alerted. */
2402
+ count: number;
2403
+ /** `Date.now()` of the last alert we actually sent, null until the first. */
2404
+ lastAlertAtMs: number | null;
2405
+ /** Occurrences suppressed since that alert. */
2406
+ suppressedSinceAlert: number;
2407
+ /** Last time this session touched the state — eviction key only, never
2408
+ * reported. */
2409
+ lastSeenAtMs: number;
2410
+ }
2411
+
2412
+ /**
2413
+ * The bucket used when the host hands us no session identity at all.
2414
+ *
2415
+ * `before_agent_run`'s context declares `sessionId` and `sessionKey` as
2416
+ * OPTIONAL, so both can be absent. Keying such an occurrence under a fixed
2417
+ * fallback keeps rate limiting working exactly as it did on a single-session
2418
+ * host, and — critically — keeps a nameless occurrence from sharing a bucket
2419
+ * with a NAMED one, which is what a `String(undefined)` key would have done.
2420
+ */
2421
+ const SCAN_UNAVAILABLE_FALLBACK_SESSION = '__unkeyed-session__';
2422
+
2423
+ /**
2424
+ * Cap on distinct sessions tracked at once.
2425
+ *
2426
+ * `session_end` is what normally frees an entry, and it is not guaranteed: an
2427
+ * older host may not emit it, and a crashed session never will. This map is
2428
+ * therefore bounded and evicts least-recently-seen first. Overshooting the cap
2429
+ * costs at most one extra alert for the evicted session — the safe direction,
2430
+ * since the failure mode of eviction is "tell the operator again", not "stay
2431
+ * quiet". Each entry is four numbers and a short key, so 512 of them is a few
2432
+ * kilobytes in a process that already holds a scanner.
2433
+ */
2434
+ export const SCAN_UNAVAILABLE_MAX_SESSIONS = 512;
2435
+
2436
+ const _scanUnavailable = new Map<string, ScanUnavailableAlertState>();
2437
+
2438
+ export interface ScanUnavailableAlertDecision {
2439
+ alert: boolean;
2440
+ /** How many occurrences this session, including this one. */
2441
+ count: number;
2442
+ /** Occurrences suppressed since the last alert. On an ALERT this is the
2443
+ * backlog being reported and then cleared; on a suppression it is the
2444
+ * running total. */
2445
+ suppressedSinceLastAlert: number;
2446
+ }
2447
+
2448
+ /** Normalise whatever the host gave us into a map key. */
2449
+ function scanUnavailableSessionKey(sessionKey?: string | null): string {
2450
+ const trimmed = typeof sessionKey === 'string' ? sessionKey.trim() : '';
2451
+ return trimmed === '' ? SCAN_UNAVAILABLE_FALLBACK_SESSION : trimmed;
2452
+ }
2453
+
2454
+ /**
2455
+ * Should this scan-unavailable occurrence raise an operator alert?
2456
+ *
2457
+ * PER SESSION (#226). The first cut kept one module-global counter, which on a
2458
+ * gateway — a process that multiplexes every channel and every concurrent
2459
+ * agent — meant one session's broken scanner silenced the FIRST failure of
2460
+ * every other session for the next five minutes. That is the same class of bug
2461
+ * the rate limit exists to avoid, inverted: instead of too many alerts, a real
2462
+ * new failure is never reported at all. Suppression is a property of one
2463
+ * session's repeating failure, so the state is keyed by one session.
2464
+ *
2465
+ * Pure apart from the per-session counter it advances, and driven by an
2466
+ * injectable `now` so the window is testable without sleeping. Exported for the
2467
+ * regression test; not part of the plugin's host-facing surface.
2468
+ *
2469
+ * The key is a session id, never logged: it reaches this function only to index
2470
+ * the map. The audit rows that carry `sessionId` are the deliberate place that
2471
+ * fact is recorded.
2472
+ */
2473
+ export function noteScanUnavailable(
2474
+ sessionKey?: string | null,
2475
+ nowMs: number = Date.now(),
2476
+ ): ScanUnavailableAlertDecision {
2477
+ const key = scanUnavailableSessionKey(sessionKey);
2478
+ let state = _scanUnavailable.get(key);
2479
+ if (!state) {
2480
+ evictScanUnavailableOverflow(nowMs);
2481
+ state = { count: 0, lastAlertAtMs: null, suppressedSinceAlert: 0, lastSeenAtMs: nowMs };
2482
+ _scanUnavailable.set(key, state);
2483
+ }
2484
+ state.count += 1;
2485
+ state.lastSeenAtMs = nowMs;
2486
+ const last = state.lastAlertAtMs;
2487
+ // A clock that jumped BACKWARDS (NTP step, suspend/resume) must not be able
2488
+ // to wedge alerting off forever: treat a negative elapsed as "window over".
2489
+ const elapsed = last === null ? Infinity : nowMs - last;
2490
+ if (last === null || elapsed >= SCAN_UNAVAILABLE_ALERT_WINDOW_MS || elapsed < 0) {
2491
+ const suppressed = state.suppressedSinceAlert;
2492
+ state.lastAlertAtMs = nowMs;
2493
+ state.suppressedSinceAlert = 0;
2494
+ return { alert: true, count: state.count, suppressedSinceLastAlert: suppressed };
2495
+ }
2496
+ state.suppressedSinceAlert += 1;
2497
+ return {
2498
+ alert: false,
2499
+ count: state.count,
2500
+ suppressedSinceLastAlert: state.suppressedSinceAlert,
2501
+ };
2502
+ }
2503
+
2504
+ /** Keep the session map bounded when `session_end` never arrives. Evicts the
2505
+ * least-recently-seen entries; an evicted session simply alerts once more. */
2506
+ function evictScanUnavailableOverflow(nowMs: number): void {
2507
+ if (_scanUnavailable.size < SCAN_UNAVAILABLE_MAX_SESSIONS) return;
2508
+ const oldestFirst = [..._scanUnavailable.entries()].sort(
2509
+ (a, b) => (a[1].lastSeenAtMs ?? nowMs) - (b[1].lastSeenAtMs ?? nowMs),
2510
+ );
2511
+ const drop = _scanUnavailable.size - SCAN_UNAVAILABLE_MAX_SESSIONS + 1;
2512
+ for (const [key] of oldestFirst.slice(0, drop)) _scanUnavailable.delete(key);
2513
+ }
2514
+
2515
+ /**
2516
+ * Forget ONE session's suppression window. Called from `session_end`, so a
2517
+ * long-lived gateway does not carry a finished session's suppression into a
2518
+ * reused id — and, just as importantly, does not clear anyone ELSE's.
2519
+ *
2520
+ * A `session_end` that names no session clears the fallback bucket only: on a
2521
+ * host that supplies no session identity every occurrence lands there, so that
2522
+ * is precisely the state that ended.
2523
+ */
2524
+ export function resetScanUnavailableAlertState(sessionKey?: string | null): void {
2525
+ _scanUnavailable.delete(scanUnavailableSessionKey(sessionKey));
2526
+ }
2527
+
2528
+ /** Test/reset seam: forget EVERY session. Production never wants this — one
2529
+ * session ending must not re-arm alerting for the others — so it is reachable
2530
+ * only from `__resetConfigStateForTest`. */
2531
+ export function __resetScanUnavailableAlertState(): void {
2532
+ _scanUnavailable.clear();
2533
+ }
2534
+
2535
+ /** What actually happened to an alert, so callers can report it truthfully.
2536
+ * `configured: false` is the honest "nobody opted in" case and is NOT a
2537
+ * failure — it is the #143 default-off contract. */
2538
+ export interface NotifyOutcome {
2539
+ configured: boolean;
2540
+ delivered: boolean;
2541
+ via: string | null;
2542
+ detail: string;
2543
+ }
2544
+
2545
+ /**
2546
+ * Make a failure detail safe to PERSIST, SEND or PRINT (#226).
2547
+ *
2548
+ * The detail is assembled from channel names, scanner errors and whatever a
2549
+ * transport said went wrong. Node's fetch failures do not name the URL, but a
2550
+ * transport is free to put one in its reason — and a notify webhook URL
2551
+ * routinely carries a token in its path or query
2552
+ * (`https://hooks.example/services/T0/B0/XXXXXXXX`). So any http(s) URL is
2553
+ * reduced to its origin: enough to tell WHICH endpoint failed, not enough to
2554
+ * replay a request to it. Bounded too, so a transport that returns a page of
2555
+ * HTML cannot bloat the log.
2556
+ *
2557
+ * EVERY sink gets the redacted string — not just the ones that obviously
2558
+ * outlive the process. The audit row is append-only and syncs; the notification
2559
+ * leaves the box; and the console is NOT the ephemeral thing an earlier version
2560
+ * of this comment claimed it was, because a gateway's stdout is routinely
2561
+ * shipped to a log aggregator and kept longer than the audit file. Redacting
2562
+ * for the row and not for the other two protected the least exposed of the
2563
+ * three.
2564
+ */
2565
+ export function redactNotifyDetail(detail: string): string {
2566
+ const withoutUrls = String(detail ?? '').replace(
2567
+ /https?:\/\/[^\s'"]+/gi,
2568
+ (url) => {
2569
+ try {
2570
+ return `${new URL(url).origin}/…`;
2571
+ } catch {
2572
+ return '<url>';
2573
+ }
2574
+ },
2575
+ );
2576
+ return withoutUrls.length > 500 ? `${withoutUrls.slice(0, 499)}…` : withoutUrls;
2577
+ }
2578
+
2579
+ /**
2580
+ * The seam a gateway MIGHT offer for sending an operator a message, captured at
2581
+ * register() time if the API exposes it.
2582
+ *
2583
+ * No OpenClaw build we have inspected exposes it — neither 2026.5.2 nor
2584
+ * 2026.7.1 has a `notifyOperator` anywhere in its plugin API — so in practice
2585
+ * the webhook is the load-bearing channel and this stays null. It is read
2586
+ * structurally rather than removed because #143's design intent was that on
2587
+ * OpenClaw the transport should use the gateway's own message capability, and
2588
+ * that only becomes true if the code is ready for the day it appears. Nothing
2589
+ * here should be read as "ShieldCortex delivers natively on OpenClaw today".
2590
+ */
2591
+ let _gatewayNotifyContext: GatewayNotifyContext | null = null;
2592
+ export function __setGatewayNotifyContextForTest(ctx: GatewayNotifyContext | null): void {
2593
+ _gatewayNotifyContext = ctx;
2594
+ }
2595
+
2596
+ /**
2597
+ * Route a conversation-firewall detection to a HUMAN (#225).
2598
+ *
2599
+ * This is the "sink" the issue is named for: before it existed, a HIGH verdict
2600
+ * produced a console line and an audit row, and a real detection on a live box
2601
+ * was seen by nobody.
2602
+ *
2603
+ * It reuses the Action Guard's #143 transport rather than inventing a second
2604
+ * one — but *correctly*, which the first cut did not:
2605
+ *
2606
+ * - the notification is built by the main package's
2607
+ * `buildConversationThreatNotification`, so it is a real, bounded
2608
+ * notification with its own event discriminator, NOT an ad-hoc
2609
+ * `{kind, severity, …}` literal cast through `NotifyChannel.send`. A
2610
+ * conversation alert therefore cannot render Approve/Deny controls or a
2611
+ * hash that does not exist — the fields simply are not on the type.
2612
+ * - delivery goes through `deliverOperatorNotification`, the same core the
2613
+ * approval path uses, so the deadline, the malformed-result handling and
2614
+ * the "nothing but the boolean is read back" rule are shared, not copied.
2615
+ * - the webhook secret is read from `webhookSecret` — the field
2616
+ * `normaliseNotifyConfig` actually returns. Mirroring it as `secret`
2617
+ * silently produced UNSIGNED POSTs.
2618
+ * - both channels are offered where the runtime provides them: the gateway's
2619
+ * own message seam first WHERE IT EXISTS (no build we have inspected
2620
+ * exposes one — see `_gatewayNotifyContext`), then the configured webhook,
2621
+ * which is what actually carries an alert off the box today.
2622
+ *
2623
+ * Returns what happened, and NEVER throws: a failed notification must not
2624
+ * become a failed turn.
2625
+ */
2626
+ export async function notifyOperatorOfConversationThreat(input: {
2627
+ outcome: 'blocked' | 'observed' | 'unavailable';
2628
+ posture: ConversationPosture;
2629
+ summary: string;
2630
+ reason: string;
2631
+ sessionId?: string;
2632
+ model?: string;
2633
+ }): Promise<NotifyOutcome> {
2634
+ try {
2635
+ const mod = await getDefenceModule();
2636
+ const cfg = await loadConfig();
2637
+ const raw = cfg.interceptor?.actionGuard?.notify;
2638
+ if (!raw) return { configured: false, delivered: false, via: null, detail: 'no notify config' };
2639
+ if (typeof mod?.normaliseNotifyConfig !== 'function') {
2640
+ return { configured: true, delivered: false, via: null, detail: 'installed shieldcortex build has no notify transport' };
2641
+ }
2642
+ const notify = mod.normaliseNotifyConfig(raw);
2643
+ if (!notify.enabled) return { configured: false, delivered: false, via: null, detail: 'notify disabled' };
2644
+
2645
+ const channels: NotifyChannelLike[] = [];
2646
+ // The gateway's own message seam, WHERE the runtime provides one. It would
2647
+ // go first, because it would reach the operator on a channel they already
2648
+ // read — but `_gatewayNotifyContext` is null on every build we have
2649
+ // inspected, so in practice this list starts at the webhook below.
2650
+ if (notify.openclaw === true && _gatewayNotifyContext) {
2651
+ const gatewayChannel = createGatewayNotifyChannel(_gatewayNotifyContext);
2652
+ if (gatewayChannel) channels.push(gatewayChannel);
2653
+ }
2654
+ if (notify.webhookUrl && typeof mod.createWebhookNotifyChannel === 'function') {
2655
+ channels.push(
2656
+ mod.createWebhookNotifyChannel({
2657
+ url: notify.webhookUrl,
2658
+ // The signing key. Passed straight through and never logged — see
2659
+ // notify-config.ts, which is the only place this value is parsed.
2660
+ secret: notify.webhookSecret,
2661
+ }),
2662
+ );
2663
+ }
2664
+ if (channels.length === 0) {
2665
+ return { configured: true, delivered: false, via: null, detail: 'notify enabled but no channel is configured/buildable on this host' };
2666
+ }
2667
+
2668
+ const notification =
2669
+ typeof mod.buildConversationThreatNotification === 'function'
2670
+ ? mod.buildConversationThreatNotification({
2671
+ outcome: input.outcome,
2672
+ posture: input.posture,
2673
+ summary: input.summary,
2674
+ reason: input.reason,
2675
+ sessionId: input.sessionId,
2676
+ model: input.model,
2677
+ host: hostname(),
2678
+ detectedAt: new Date().toISOString(),
2679
+ })
2680
+ : null;
2681
+ if (!notification) {
2682
+ // An older dist has the transport but not this event. Sending the
2683
+ // approval-shaped payload instead would put an Approve button on an alert
2684
+ // with nothing behind it — refuse, and say why.
2685
+ return {
2686
+ configured: true,
2687
+ delivered: false,
2688
+ via: null,
2689
+ detail: 'installed shieldcortex build predates the conversation-threat notification — refusing to send an approval-shaped alert',
2690
+ };
2691
+ }
2692
+
2693
+ if (typeof mod.deliverOperatorNotification !== 'function') {
2694
+ return { configured: true, delivered: false, via: null, detail: 'installed shieldcortex build has no notification delivery core' };
2695
+ }
2696
+ const result = await mod.deliverOperatorNotification(notification, {
2697
+ channels,
2698
+ // Bounded HARDER than the transport's own configured deadline, because
2699
+ // this call sits inside a gate the gateway awaits: the user's turn is
2700
+ // waiting on it. The hook is registered with a 30s timeout, and a gate
2701
+ // that exceeds its own timeout is a security control that fails in a way
2702
+ // nobody has reasoned about. The alert is already in the log and the
2703
+ // audit row by this point, so what a longer wait buys is one extra retry
2704
+ // window on a transport that is, by then, visibly unhealthy.
2705
+ timeoutMs: Math.min(notify.timeoutMs ?? CONVERSATION_NOTIFY_MAX_MS, CONVERSATION_NOTIFY_MAX_MS),
2706
+ });
2707
+ const failures = result.attempts
2708
+ .filter((a) => !a.result.delivered)
2709
+ .map((a) => `${a.channel}: ${a.result.reason ?? 'failed'}`)
2710
+ .join('; ');
2711
+ return {
2712
+ configured: true,
2713
+ delivered: result.deliveredVia !== null,
2714
+ via: result.deliveredVia,
2715
+ detail: result.deliveredVia ? `delivered via ${result.deliveredVia}` : `undeliverable — ${failures || 'no channel accepted it'}`,
2716
+ };
2717
+ } catch (err) {
2718
+ return {
2719
+ configured: true,
2720
+ delivered: false,
2721
+ via: null,
2722
+ detail: `notify error: ${err instanceof Error ? err.message : String(err)}`,
2723
+ };
2724
+ }
2725
+ }
2726
+
2727
+ /**
2728
+ * The event payload OpenClaw hands `before_agent_run`, as declared by the host
2729
+ * SDK (`PluginHookBeforeAgentRunEvent`, hook-types.d.ts). Note what is NOT on
2730
+ * it: `sessionId` and `model` live on the CONTEXT, not the event — reading them
2731
+ * off the event, as the first cut did, produced `undefined` in every audit row
2732
+ * and every alert.
2733
+ */
2734
+ type BeforeAgentRunEvent = {
2735
+ prompt?: string;
2736
+ messages?: unknown[];
2737
+ systemPrompt?: string;
2738
+ accountId?: string;
2739
+ channelId?: string;
2740
+ senderId?: string;
2741
+ senderIsOwner?: boolean;
2742
+ };
2743
+
2744
+ /**
2745
+ * The gate's return contract, verbatim from the host SDK:
2746
+ *
2747
+ * type HookDecisionPass = { outcome: "pass" }
2748
+ * type HookDecisionBlock = { outcome: "block"; reason: string; message?: string;
2749
+ * category?: string; metadata?: Record<string, unknown> }
2750
+ * type PluginHookBeforeAgentRunResult = InputGateDecision | void
2751
+ *
2752
+ * `{ block: true }` — what the first cut returned — is the `before_tool_call`
2753
+ * shape, and this gate does not understand it: it is neither "pass" nor
2754
+ * "block", so the run would have proceeded while our audit row said BLOCKED.
2755
+ * A firewall whose block is a no-op is worse than no firewall, because the
2756
+ * evidence says it worked.
2757
+ *
2758
+ * `reason` is documented as INTERNAL ("core must not log, persist, broadcast,
2759
+ * or expose it verbatim"); `message` is the user-facing half. We keep the
2760
+ * verdict summary in both — it names the risk level and the detection count,
2761
+ * never the offending text.
2762
+ *
2763
+ * SHAPE IS EXACT, and the host enforces it structurally (`isHookDecision`,
2764
+ * hook-runner-global, 2026.7.1-2): a pass must be `{ outcome: 'pass' }` and
2765
+ * NOTHING else — the guard is `keys.length === 1`. Adding so much as a
2766
+ * `metadata` field to the pass branch for debugging would make it "not a hook
2767
+ * decision", and the runner's answer to that is not to ignore it: it is
2768
+ * `{ outcome: 'block', reason: 'before_agent_run returned an invalid
2769
+ * decision' }`. The failure mode of a malformed ALLOW here is a BLOCKED turn.
2770
+ */
2771
+ export type InputGateDecision =
2772
+ | { outcome: 'pass' }
2773
+ | { outcome: 'block'; reason: string; message?: string; category?: string; metadata?: Record<string, unknown> };
2774
+
2775
+ /**
2776
+ * The gate's allow answer, stated explicitly (#226).
2777
+ *
2778
+ * A fresh literal per call, not a shared constant: the runner passes whatever
2779
+ * we return into its own merge/normalise chain, and a frozen singleton handed
2780
+ * to a host that decides to annotate it would fail in a way this plugin cannot
2781
+ * see. It costs one object per turn.
2782
+ *
2783
+ * WHY EXPLICIT, when the SDK types the result `InputGateDecision | void` and
2784
+ * this handler previously returned `undefined` on every allow path:
2785
+ *
2786
+ * The 2026.7.1-2 runner contradicts itself about void, one guard deep.
2787
+ * `runBeforeAgentRun`'s doc comment says "Handlers that return void are treated
2788
+ * as pass", and its `mergeResults` body opens with
2789
+ *
2790
+ * if (next === void 0 || next === null) → { outcome: "block",
2791
+ * reason: "…invalid decision" }
2792
+ *
2793
+ * i.e. the merge is written to BLOCK on void. What saves an `undefined` return
2794
+ * today is only that `runModifyingHook` never calls the merge for it —
2795
+ * `if (handlerResult !== void 0 && (handlerResult !== null || mergeNullResults))`
2796
+ * — so the void branch inside the merge is unreachable dead code, while `null`,
2797
+ * the sibling value that same line treats identically, reaches it and DOES
2798
+ * block. Verified by executing the real 2026.7.1-2 runner: `undefined` → pass,
2799
+ * `null` → block/invalid, `{outcome:'pass'}` → pass.
2800
+ *
2801
+ * So void is not broken here — it is correct by one guard, against a merge
2802
+ * function whose stated intent is to reject it. `{ outcome: 'pass' }` is
2803
+ * correct under BOTH readings, and is the shape the host validates rather than
2804
+ * the shape it happens to skip. That is the difference worth having in front of
2805
+ * every user turn.
2806
+ */
2807
+ function gatePass(): InputGateDecision {
2808
+ return { outcome: 'pass' };
2809
+ }
2810
+
2811
+ /**
2812
+ * The conversation firewall's enforcement point (#225).
2813
+ *
2814
+ * Unlike `llm_input`, this hook is awaited by the gateway and its return value
2815
+ * decides whether the run proceeds. It scans the prompt, applies the configured
2816
+ * posture, and — critically — routes a detection to a HUMAN rather than only to
2817
+ * a log file. The finding this fixes was that a HIGH verdict on a live box was
2818
+ * seen by nobody.
2819
+ *
2820
+ * Fails OPEN on any internal error: a security plugin that bricks the gateway
2821
+ * has caused a worse outage than the one it prevents. Every failure is reported.
2822
+ *
2823
+ * EVERY path returns a decision — `gatePass()` to allow, `{ outcome: 'block' }`
2824
+ * only for a dirty verdict under `enforce`. Nothing returns `undefined`; see
2825
+ * `gatePass` for the host-contract reason. "Fails open" therefore now means an
2826
+ * explicit pass, which is a stronger statement than the absence of an answer:
2827
+ * it is the same word said in the vocabulary the host validates.
2828
+ */
2829
+ export async function handleBeforeAgentRun(
2830
+ event: BeforeAgentRunEvent,
2831
+ ctx: AgentCtx,
2832
+ ): Promise<InputGateDecision> {
2833
+ let posture: ConversationPosture = 'observe';
2834
+ try {
2835
+ const cfg = await loadConfig();
2836
+ posture = conversationPosture(cfg.interceptor?.conversation);
2837
+ if (posture === 'off') return gatePass();
2838
+
2839
+ const text = String(event?.prompt ?? '');
2840
+ if (!text || text.length < 10 || isInternalContent(text)) return gatePass();
2841
+
2842
+ // sessionId/model come off the hook CONTEXT (PluginHookAgentContext); the
2843
+ // event carries neither. Both are optional there too, so both may be absent.
2844
+ const sessionId = ctx?.sessionId ?? ctx?.sessionKey;
2845
+ const model = (ctx as { modelId?: string } | undefined)?.modelId;
2846
+
2847
+ // scanRealtimeContent no longer throws on the paths that used to (it
2848
+ // reports `available:false` instead), but a defensive catch stays: this
2849
+ // function's contract is that nothing here can stop a turn by accident.
2850
+ // #226: BOUNDED. The gateway awaits this hook, so an unbounded scan is an
2851
+ // unbounded pause in front of the user's prompt — see scanWithDeadline.
2852
+ let scan: ConversationScanResult;
2853
+ try {
2854
+ scan = await scanWithDeadline(text);
2855
+ } catch (err) {
2856
+ const detail = err instanceof Error ? err.message : String(err);
2857
+ scan = { clean: false, available: false, errored: true, error: detail, summary: 'scan unavailable' };
2858
+ }
2859
+
2860
+ // #235: WHO sent this turn, resolved before the verdict is applied.
2861
+ // `senderIsOwner` was declared on the event and read by nothing, so the
2862
+ // enforce path could block the operator's own paste and destroy it. The
2863
+ // config is already loaded, so this is a pure call — no second read.
2864
+ const trust = classifyConversationOrigin({
2865
+ senderIsOwner: event?.senderIsOwner,
2866
+ trustOwnerInput: cfg.conversationTrust?.trustOwnerInput,
2867
+ });
2868
+
2869
+ const decision = evaluateConversationRun(posture, scan, trust);
2870
+
2871
+ // #226: REDACT ONCE, then use the redacted string everywhere the reason
2872
+ // goes — the persisted decision row, the outbound notification, the block
2873
+ // reason, the console line. On the unavailable path `decision.reason`
2874
+ // embeds the scanner's own failure string verbatim
2875
+ // (`conversation scan unavailable (${scan.error})`), and that string is
2876
+ // assembled from a transport error: a cold MCP start, a fetch, a defence
2877
+ // build download. Any of those can name the endpoint it failed to reach,
2878
+ // and such a URL routinely carries a credential in its path
2879
+ // (`https://hooks.example/services/T0/B0/XXXX`). The console line was
2880
+ // already redacted while the row and the alert — the two that PERSIST and
2881
+ // LEAVE THE BOX — were not, which had the guarantee exactly backwards.
2882
+ const safeReason = decision.reason === null ? null : redactNotifyDetail(decision.reason);
2883
+
2884
+ // #226: repeated unavailability alerts at most once per window. Called here
2885
+ // rather than at the notify site so the COUNTERS advance on every
2886
+ // occurrence, and the decision row can record what was suppressed even when
2887
+ // no alert goes out.
2888
+ // Keyed by SESSION: a broken scanner in one session must not silence the
2889
+ // first report of a broken scanner in another. `sessionId` may be absent —
2890
+ // noteScanUnavailable buckets that case separately rather than letting one
2891
+ // nameless session stand in for all of them.
2892
+ const unavailable = decision.outcome === 'unavailable';
2893
+ const alertGate = unavailable ? noteScanUnavailable(sessionId) : null;
2894
+ const suppressAlert = alertGate !== null && !alertGate.alert;
2895
+
2896
+ // ── EVIDENCE FIRST, SIDE EFFECT SECOND ────────────────────────────────
2897
+ //
2898
+ // The decision row is a LOCAL append and it goes to disk before anything
2899
+ // leaves this box. The previous order awaited an external notification and
2900
+ // then wrote the row — under a comment claiming the row already existed —
2901
+ // so every way that call can end badly took the evidence with it: a
2902
+ // notification channel that hangs until the gateway's 30s hook timeout
2903
+ // fires, a transport that throws past its own catch, an operator restarting
2904
+ // the gateway mid-alert, the process dying. In each case the block or the
2905
+ // detection HAPPENED and there is no record that it did. That inverts the
2906
+ // whole point: a security control's own log must not be contingent on an
2907
+ // unrelated network round trip succeeding.
2908
+ //
2909
+ // The row carries a stable `eventId`, so the delivery row appended after
2910
+ // the attempt below can be joined to it without either row having to
2911
+ // predict the other's outcome.
2912
+ const eventId = randomUUID();
2913
+ // #226: whether the decision row ACTUALLY LANDED. `auditLog` used to
2914
+ // swallow its failures and return void, so the code below could not tell an
2915
+ // append from a silent no-op and every downstream statement — the operator
2916
+ // alert, the delivery row, this function's own comments — asserted that
2917
+ // evidence existed. Now the boolean is carried, said out loud on stderr,
2918
+ // and attached to the alert as a bounded, secret-free fact.
2919
+ let decisionRowPersisted = true;
2920
+ if (decision.audit) {
2921
+ // AWAITED: the row that says what was decided must exist before the
2922
+ // decision is handed back, and its success or failure must be READ. The
2923
+ // write is a bounded local append wrapped in its own try/catch.
2924
+ decisionRowPersisted = await auditLog({
2925
+ type: decision.outcome === 'unavailable' ? 'scan_unavailable' : 'threat',
2926
+ hook: 'before_agent_run',
2927
+ eventId,
2928
+ sessionId,
2929
+ model,
2930
+ // The REDACTED reason. This row is appended to a file that syncs.
2931
+ reason: safeReason,
2932
+ posture,
2933
+ outcome: decision.outcome,
2934
+ // #235: the origin, on every conversation decision row. Without it an
2935
+ // operator auditing an `enforce` host cannot tell a turn that was not
2936
+ // blocked because it was clean from one that was not blocked because
2937
+ // the owner sent it — and "why did this not block?" is the question
2938
+ // this row exists to answer. A label ('owner'/'non-owner'/'unknown'),
2939
+ // never a sender id: the row syncs.
2940
+ origin: trust.origin,
2941
+ // The verdict summary, never the prompt. The input that trips an
2942
+ // injection detector is hostile text by assumption; copying it into an
2943
+ // audit row that syncs to the dashboard/cloud would carry the payload
2944
+ // one hop further. A length + digest keeps rows correlatable without
2945
+ // storing the content.
2946
+ verdict: scan.summary,
2947
+ chars: text.length,
2948
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2949
+ // Deliberately NOT `notified: false`. Nothing has been attempted yet,
2950
+ // and a false here would read as "we tried and failed". The attempt's
2951
+ // result is its own row, keyed by this eventId.
2952
+ notifyPending: decision.notify && !suppressAlert,
2953
+ // #226: the unavailability run-length, on EVERY occurrence. Alerting is
2954
+ // rate limited; auditing is not, so the row is where the true count
2955
+ // lives — and it says explicitly when an alert was withheld, so a gap in
2956
+ // the alert stream can never be mistaken for a gap in the failures.
2957
+ ...(alertGate
2958
+ ? {
2959
+ unavailableCount: alertGate.count,
2960
+ alertSuppressed: suppressAlert,
2961
+ alertSuppressedSinceLastAlert: alertGate.suppressedSinceLastAlert,
2962
+ }
2963
+ : {}),
2964
+ ts: new Date().toISOString(),
2965
+ });
2966
+ if (!decisionRowPersisted) {
2967
+ console.error(
2968
+ `[shieldcortex] ⚠️ conversation ${decision.outcome} decision could NOT be written to the audit log — ` +
2969
+ 'the decision itself still stands, but there is no local record of it. Check the audit directory ' +
2970
+ '(SHIELDCORTEX_AUDIT_DIR or ~/.shieldcortex/audit) for permissions or disk space.',
2971
+ );
2972
+ }
2973
+ }
2974
+
2975
+ // The sink. Awaited — the first cut fired this off with `void` and threw
2976
+ // the delivery boolean away, so the code could not tell "a human was told"
2977
+ // from "nothing left this box". It is bounded (CONVERSATION_NOTIFY_MAX_MS,
2978
+ // well under the hook's own 30s timeout) and never throws.
2979
+ let notifyResult: NotifyOutcome | null = null;
2980
+ if (decision.notify) {
2981
+ const label = decision.outcome === 'unavailable' ? 'unavailable' : decision.block ? 'blocked' : 'observed';
2982
+ // The same redacted string the row got. A gateway's stdout is routinely
2983
+ // shipped to a log aggregator, so "ephemeral" is not a property the
2984
+ // console actually has either.
2985
+ console.warn(
2986
+ `[shieldcortex] ⚠️ ${safeReason ?? 'conversation threat'} — posture=${posture}, outcome=${label}` +
2987
+ (suppressAlert
2988
+ ? ` (operator alert SUPPRESSED — ${alertGate?.suppressedSinceLastAlert} since the last one; ${alertGate?.count} this session)`
2989
+ : ''),
2990
+ );
2991
+ // Audited above, not alerted. Nothing further is written for a suppressed
2992
+ // occurrence: no attempt was made, and a delivery row saying
2993
+ // `delivered: false` would read as a transport failure that never
2994
+ // happened. The decision itself is unaffected — suppression governs who
2995
+ // is TOLD, never what is DECIDED.
2996
+ if (!suppressAlert) {
2997
+ // The audit-persistence fact rides ALONG with the alert when the local
2998
+ // record failed: bounded, no secrets, and it tells the operator that
2999
+ // this notification is the only trace of the event. Appended to the
3000
+ // reason rather than added as a field so it survives an older installed
3001
+ // dist whose notification builder does not know about it.
3002
+ const auditNote = decisionRowPersisted ? '' : ' [auditPersistence=failed: no local audit row for this event]';
3003
+ const suppressedNote =
3004
+ alertGate && alertGate.suppressedSinceLastAlert > 0
3005
+ ? ` [${alertGate.suppressedSinceLastAlert} further scan-unavailable event(s) suppressed since the last alert; ${alertGate.count} this session]`
3006
+ : '';
3007
+ notifyResult = await notifyOperatorOfConversationThreat({
3008
+ outcome: label,
3009
+ posture,
3010
+ summary: scan.summary,
3011
+ // REDACTED. This one leaves the box entirely — to a webhook, an
3012
+ // aggregator, a phone — so it is the last place a tokenised endpoint
3013
+ // URL lifted out of a scanner error may appear.
3014
+ reason: `${safeReason ?? 'conversation threat'}${suppressedNote}${auditNote}`,
3015
+ sessionId,
3016
+ model,
3017
+ });
3018
+ // Truthful reporting: never imply a human was reached unless a
3019
+ // transport said so. "Not configured" is not a failure — it is the #143
3020
+ // default.
3021
+ if (notifyResult.configured && !notifyResult.delivered) {
3022
+ // #226: redacted HERE too, not only on the row below. The same detail
3023
+ // string reaches both, and a gateway's stdout is routinely shipped
3024
+ // somewhere it outlives the process.
3025
+ console.warn(`[shieldcortex] ⚠️ conversation alert UNDELIVERED — ${redactNotifyDetail(notifyResult.detail)}`);
3026
+ }
3027
+ // Guarded on `audit` because `eventId` has to point at something:
3028
+ // `evaluateConversationRun` never sets notify without audit, and if that
3029
+ // ever changed, a delivery row keyed to a decision row that was never
3030
+ // written would be a dangling reference rather than evidence.
3031
+ if (decision.audit) {
3032
+ // A SECOND row, not a rewrite of the first. The audit sink is an
3033
+ // append-only JSONL file, so "what was decided" and "who was told" are
3034
+ // separate facts recorded when each became true, joined by eventId.
3035
+ // `via` is the channel NAME ('webhook', 'openclaw-gateway'), never a
3036
+ // URL; the detail is redacted before it is persisted.
3037
+ await auditLog({
3038
+ type: 'notification_delivery',
3039
+ hook: 'before_agent_run',
3040
+ eventId,
3041
+ sessionId,
3042
+ configured: notifyResult.configured,
3043
+ delivered: notifyResult.delivered,
3044
+ via: notifyResult.via,
3045
+ detail: redactNotifyDetail(notifyResult.detail),
3046
+ // The eventId this row joins on may point at a row that was never
3047
+ // written. Say so here rather than leave a dangling reference that
3048
+ // reads as a missing file rather than a failed write.
3049
+ ...(decisionRowPersisted ? {} : { auditPersistence: 'failed' }),
3050
+ ts: new Date().toISOString(),
3051
+ });
3052
+ }
3053
+ }
3054
+ }
3055
+
3056
+ // Clean, observed-not-blocked, and scan-unavailable all land here. The
3057
+ // audit row and the operator alert above have already recorded what
3058
+ // happened; the run itself proceeds, and says so.
3059
+ if (!decision.block) return gatePass();
3060
+ return {
3061
+ outcome: 'block',
3062
+ // Redacted for the same reason as the row above: the host SDK documents
3063
+ // `reason` as internal, but "internal" is a policy, not a guarantee.
3064
+ reason: safeReason ?? 'conversation threat',
3065
+ message: `ShieldCortex blocked this turn: ${scan.summary}. The prompt was not sent to the model.`,
3066
+ category: 'prompt_injection',
3067
+ };
3068
+ } catch (e) {
3069
+ // Fail open, loudly. Never let the guard's own failure stop the agent.
3070
+ //
3071
+ // This catch is also why the handler must not be allowed to THROW: the host
3072
+ // registers `before_agent_run` as fail-CLOSED
3073
+ // (`failurePolicyByHook: { before_agent_run: 'fail-closed' }`), so an
3074
+ // exception escaping here does not fail open at all — the gateway catches it
3075
+ // and blocks the run with "before_agent_run hook failed". An explicit pass
3076
+ // is the only way this function actually keeps its fail-open promise.
3077
+ console.error('[shieldcortex] before_agent_run error (failing open):', e instanceof Error ? e.message : String(e));
3078
+ return gatePass();
3079
+ }
3080
+ }
3081
+
1120
3082
  // Skip text blocks that are ShieldCortex/OpenClaw tool-result pass-throughs
1121
3083
  function isToolResultContent(text: string): boolean {
1122
3084
  // ShieldCortex recall returns "Found N memories:" header
@@ -1314,6 +3276,13 @@ export default {
1314
3276
  if (_registered) return;
1315
3277
  _registered = true;
1316
3278
 
3279
+ // #226: the host runtime's own version, before anything can throw. It is
3280
+ // the primary version evidence for the conversation-gate check — the
3281
+ // gateway stating its own version beats inferring one from whichever
3282
+ // package.json sits above the entry path. Absent on a host that does not
3283
+ // expose it, which stays UNKNOWN rather than becoming a guess.
3284
+ recordHostRuntimeVersion(api);
3285
+
1317
3286
  // --- Interceptor (lazy init) ---
1318
3287
  let interceptorReady: ReturnType<typeof createInterceptor> | null = null;
1319
3288
  let interceptorInitAttempted = false;
@@ -1362,12 +3331,44 @@ export default {
1362
3331
  : `${guardCfg.enforce ? "enforce" : "warn"}${autoApproved > 0 ? ` (${autoApproved} auto-approved)` : ""}${interceptorReady ? "" : " — not yet initialised this session"}`;
1363
3332
  const hooksLine = _beforeToolCallRegistered
1364
3333
  ? "llm_input (scan), llm_output (memory), before_tool_call (action guard), session_end (cache reset)"
1365
- : "llm_input (scan), llm_output (memory)";
3334
+ // #226: session_end is registered even with the interceptor off —
3335
+ // the conversation gate keeps per-session state that needs freeing.
3336
+ : "llm_input (scan), llm_output (memory), session_end (cache reset)";
3337
+ // #225: the conversation plane, stated as evidence rather than as a
3338
+ // tick. Every clause below is something this process actually knows:
3339
+ // the configured posture, that we asked for the hook, the host build,
3340
+ // and the operator's grant. Nothing here claims the gateway accepted
3341
+ // the registration, because the plugin API never says so.
3342
+ const hostProbe = detectHostOpenClaw();
3343
+ const plane = describeConversationPlane({
3344
+ posture: conversationPosture(cfg.interceptor?.conversation),
3345
+ hookRequested: _beforeAgentRunRequested,
3346
+ gateSupport: hostSupportsConversationGate(hostProbe),
3347
+ hostOpenClawVersion: hostProbe.version,
3348
+ consentGranted: _conversationAccessGranted,
3349
+ });
3350
+ const notifyRaw = cfg.interceptor?.actionGuard?.notify;
3351
+ const notifyState = notifyRaw && (notifyRaw as { enabled?: unknown }).enabled === true
3352
+ ? 'configured'
3353
+ : 'not configured — detections reach the audit log and this box only';
1366
3354
  return {
1367
3355
  text:
1368
3356
  `ShieldCortex v${_version}\n` +
1369
- ` Hooks: ${hooksLine}\n` +
3357
+ ` Hooks: ${hooksLine}${_beforeAgentRunRequested ? ', before_agent_run (conversation gate, requested)' : ''}\n` +
1370
3358
  ` Action guard: ${guardState}\n` +
3359
+ ` Conversation firewall: ${plane.summary}\n` +
3360
+ // #226: state the PROVENANCE, not just the value. This flag is a
3361
+ // SNAPSHOT taken once, when the plugin loaded — the host reads
3362
+ // the grant at hook-registration time and this process never
3363
+ // re-reads it. So an operator who has just edited openclaw.json
3364
+ // and re-run the command sees the old answer, correctly, and
3365
+ // would otherwise conclude the grant does not work. Nothing here
3366
+ // is live: changing it requires a gateway restart before either
3367
+ // the gateway or this line reflects it.
3368
+ ` Conversation access grant: ${_conversationAccessGranted ? 'granted' : 'NOT granted'} (plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess)\n` +
3369
+ ' — read from openclaw.json when this plugin LOADED; it is a snapshot, not a live read.\n' +
3370
+ ' Editing that key takes effect only after a gateway restart, for the gateway and for this line.\n' +
3371
+ ` Operator notify: ${notifyState}\n` +
1371
3372
  ` Auto memory: ${autoMemory} | Dedupe: ${dedupe}\n` +
1372
3373
  ` Cloud sync: ${cloud}`,
1373
3374
  };
@@ -1471,42 +3472,153 @@ export default {
1471
3472
  return handleTypedBeforeToolCall(event, interceptor, api.logger, ctx?.sessionId);
1472
3473
  }, { priority: 80, timeoutMs: 30_000 });
1473
3474
  _beforeToolCallRegistered = true;
1474
-
1475
- // Try to register session_end for cache cleanup (only meaningful while
1476
- // an interceptor can exist)
1477
- try {
1478
- api.on('session_end', (ev?: { sessionId?: string }) => {
1479
- interceptorReady?.resetSession();
1480
- // #233: a taint must not outlive the conversation that earned it.
1481
- if (ev?.sessionId) sessionTaint.clear(ev.sessionId);
1482
- });
1483
- } catch {
1484
- // session_end may not be a supported hook — TTL safety net handles this
1485
- }
3475
+ // NOTE: session_end is NOT registered here — it moved out of this guard
3476
+ // in #226 and is registered unconditionally below.
1486
3477
  } else {
1487
3478
  api.logger?.info?.('[shieldcortex] interceptor.enabled:false in plugin config — before_tool_call hook not registered');
1488
3479
  }
1489
3480
 
1490
- // These two are CONVERSATION hooks: OpenClaw drops them at registration
1491
- // for a non-bundled plugin unless the host grants
1492
- // plugins.entries.<id>.hooks.allowConversationAccess = true. We still
1493
- // attempt registration (the host decides, and the grant can be added
1494
- // without a code change), but we must not CLAIM them afterwards — see the
1495
- // honesty note on the log line below (#225).
3481
+ // session_end — registered UNCONDITIONALLY (#226).
3482
+ //
3483
+ // It used to live inside the `interceptorDisabledInHostConfig` guard above,
3484
+ // on the reasoning that it exists for the interceptor's session cache. That
3485
+ // stopped being true when `before_agent_run` landed: the gate is registered
3486
+ // regardless of `interceptor.enabled` (its posture, not the interceptor
3487
+ // flag, decides what it does), and it accumulates per-session
3488
+ // scan-unavailable suppression state. With the cleanup hook skipped, a host
3489
+ // that disabled the interceptor kept every session's window alive for the
3490
+ // life of the gateway process.
3491
+ //
3492
+ // Registering it does NOT reintroduce #112. That incident was specific to
3493
+ // `before_tool_call`: a registered approval hook changes how OpenClaw
3494
+ // resolves tool-call approvals for unattended Codex agents, so an
3495
+ // unattended turn waited 120s on a decision nobody could give. `session_end`
3496
+ // is a notification — it cannot block, approve, or delay anything — and its
3497
+ // handler here only frees local state.
3498
+ try {
3499
+ api.on('session_end', (event?: { sessionId?: string; sessionKey?: string }, ctx?: AgentCtx) => {
3500
+ interceptorReady?.resetSession();
3501
+ const endedSession = ctx?.sessionId ?? ctx?.sessionKey ?? event?.sessionId ?? event?.sessionKey ?? null;
3502
+ // #226: the scan-unavailable alert window is session state too, and it
3503
+ // is keyed per session — so clear THIS session's window and nobody
3504
+ // else's. Clearing them all would re-arm alerting for every live
3505
+ // session every time any one of them ended.
3506
+ resetScanUnavailableAlertState(endedSession);
3507
+ // #233: a taint must not outlive the conversation that earned it. Same
3508
+ // per-session rule, for the same reason.
3509
+ if (endedSession) sessionTaint.clear(endedSession);
3510
+ });
3511
+ } catch {
3512
+ // session_end may not be a supported hook — TTL safety net handles this
3513
+ }
3514
+
3515
+ // llm_input/llm_output are CONVERSATION hooks: OpenClaw drops them at
3516
+ // registration for a non-bundled plugin unless the host grants
3517
+ // plugins.entries.<id>.hooks.allowConversationAccess = true. Registration is
3518
+ // still attempted (the host decides, and the grant can be added without a
3519
+ // code change), but they must not be CLAIMED afterwards — see the startup
3520
+ // line below (#225/#230).
1496
3521
  api.on("llm_input", handleLlmInput, { timeoutMs: 30_000 });
1497
3522
  api.on("llm_output", handleLlmOutput, { timeoutMs: 30_000 });
1498
3523
 
1499
- // #225: this line used to announce `llm_input + llm_output` unconditionally.
1500
- // On any host without the conversation-access grant the gateway logged, on
1501
- // the very next two lines, that it had dropped both — so ShieldCortex was
1502
- // claiming conversation protection it did not have, in the one place an
1503
- // operator looks to confirm startup. Report only what is actually live, and
1504
- // name the missing grant when it is the reason.
1505
- const conversationAccess = readConversationAccess(homedir(), PLUGIN_ID);
3524
+ // #225: the conversation firewall's ENFORCEMENT point. `llm_input` above is
3525
+ // an OpenClaw *observation* hook — it cannot stop anything, which is why a
3526
+ // detected injection reached the model regardless and the only trace was a
3527
+ // console line. `before_agent_run` is the documented hook that can block a
3528
+ // run, so the verdict lands here where it can actually act.
3529
+ //
3530
+ // Registration is attempted unconditionally: the posture
3531
+ // (off/observe/enforce) decides what happens, and it is read per-call so a
3532
+ // config change takes effect without a restart. Registering conditionally
3533
+ // would make "is the guard wired?" depend on config read at boot — the
3534
+ // exact class of silent gap that #214/#222 were.
3535
+ //
3536
+ // The try/catch is for a host whose `api.on` throws on an unknown name. On
3537
+ // the hosts we have inspected it does NOT throw — an unsupported name is
3538
+ // dropped with a diagnostic, and a conversation hook without the operator's
3539
+ // grant is refused the same way — so a successful call proves only that we
3540
+ // ASKED. That is exactly what the flag is named after, and the honest
3541
+ // reporting comes from the version + consent evidence below.
3542
+ try {
3543
+ api.on("before_agent_run", handleBeforeAgentRun, { timeoutMs: 30_000 });
3544
+ _beforeAgentRunRequested = true;
3545
+ } catch (err) {
3546
+ _beforeAgentRunRequested = false;
3547
+ (api.logger as any)?.warn?.(
3548
+ `[shieldcortex] before_agent_run could not be registered on this host (${err instanceof Error ? err.message : String(err)}) — the conversation firewall cannot block on this gateway`,
3549
+ );
3550
+ }
3551
+
3552
+ // The gateway's own message seam, WHERE a host provides one (#143's design
3553
+ // intent: "on OpenClaw the transport should use the gateway's own message
3554
+ // capability"). Probed structurally, never required. No build we have
3555
+ // inspected exposes it — `notifyOperator` appears nowhere in the plugin API
3556
+ // of 2026.5.2 or 2026.7.1 — so on today's hosts this stays null and
3557
+ // conversation alerts go to the webhook.
3558
+ const notifyCtx = (api as { runtime?: { notifyOperator?: unknown }; notifyOperator?: unknown });
3559
+ if (typeof notifyCtx.notifyOperator === 'function') {
3560
+ _gatewayNotifyContext = notifyCtx as GatewayNotifyContext;
3561
+ } else if (typeof notifyCtx.runtime?.notifyOperator === 'function') {
3562
+ _gatewayNotifyContext = notifyCtx.runtime as GatewayNotifyContext;
3563
+ }
3564
+
3565
+ // The operator's conversation-access grant. Read, never written: OpenClaw
3566
+ // refuses every conversation hook for a non-bundled plugin without it, so a
3567
+ // box missing it runs with NO conversation plane at all — and on four of
3568
+ // five fleet hosts surveyed in #222 that was the normal outcome of a
3569
+ // documented install. Report it at boot rather than let the operator infer
3570
+ // protection from a registration line that only states intent.
3571
+ // The host's own in-memory config is the better source (it is what the
3572
+ // loader consulted), so it is preferred; the file the host reads is the
3573
+ // fallback for a runtime that does not expose it. `readConversationAccess`
3574
+ // is #225's shared reader — it also tells us whether the config could be
3575
+ // read at all, which is what keeps "not granted" apart from "cannot tell"
3576
+ // on the startup line below.
3577
+ const diskAccess = readConversationAccess(homedir(), PLUGIN_ID);
3578
+ let rootConfigSeen = false;
3579
+ try {
3580
+ const runtimeConfigApi = (api as PluginApi).runtime?.config;
3581
+ const rootConfig = typeof runtimeConfigApi?.current === 'function'
3582
+ ? runtimeConfigApi.current()
3583
+ : typeof runtimeConfigApi?.loadConfig === 'function'
3584
+ ? runtimeConfigApi.loadConfig()
3585
+ : (api as PluginApi).config;
3586
+ rootConfigSeen = Boolean(rootConfig) && typeof rootConfig === 'object';
3587
+ _conversationAccessGranted = rootConfigSeen
3588
+ ? readConversationAccessGrant(rootConfig)
3589
+ : diskAccess.granted;
3590
+ } catch {
3591
+ _conversationAccessGranted = diskAccess.granted;
3592
+ }
3593
+ if (!_conversationAccessGranted) {
3594
+ (api.logger as any)?.warn?.(
3595
+ `[shieldcortex] conversation firewall INACTIVE: plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess is not true in openclaw.json — ` +
3596
+ 'the gateway will refuse llm_input, llm_output and before_agent_run for this plugin. Nothing on the conversation path is scanned or blocked. ' +
3597
+ 'This is an operator consent grant and ShieldCortex will never set it for you.',
3598
+ );
3599
+ }
3600
+
3601
+ // #225/#230: this line used to announce `llm_input + llm_output`
3602
+ // unconditionally. On any host without the conversation-access grant the
3603
+ // gateway logged, on the very next two lines, that it had dropped both — so
3604
+ // ShieldCortex was claiming conversation protection it did not have, in the
3605
+ // one place an operator looks to confirm startup. Report only what is
3606
+ // actually live, and name the missing grant when it is the reason.
3607
+ //
3608
+ // `before_agent_run` (#226) is on the same list from 2026.5.9-beta.1, so it
3609
+ // is claimed only when the grant is present AND registration was attempted
3610
+ // this session.
1506
3611
  api.logger.info(
1507
3612
  `[shieldcortex] v${_version} registered (${describeRegisteredHooks({
1508
- access: conversationAccess,
3613
+ access: {
3614
+ granted: _conversationAccessGranted,
3615
+ // We could read SOMETHING (the host's config or the file) ⇒ the
3616
+ // ungranted state is a fact, not a failed measurement.
3617
+ readable: rootConfigSeen || diskAccess.readable,
3618
+ entryPresent: diskAccess.entryPresent,
3619
+ },
1509
3620
  beforeToolCallRegistered: _beforeToolCallRegistered,
3621
+ beforeAgentRunRequested: _beforeAgentRunRequested,
1510
3622
  })})`,
1511
3623
  );
1512
3624
  } catch (err) {