@drakon-systems/shieldcortex-realtime 4.47.38 → 4.47.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.ts CHANGED
@@ -2,26 +2,45 @@
2
2
  * ShieldCortex Real-time Scanning Plugin for OpenClaw v2026.3.22+
3
3
  *
4
4
  * Uses typed OpenClaw plugin hooks (`api.on`) for llm_input/llm_output
5
- * scanning and before_tool_call interception. `api.registerHook` registers
6
- * internal HOOK-style automation and does not participate in the agent-loop
7
- * block/approval semantics ShieldCortex needs.
8
- * All scanning operations are fire-and-forget.
5
+ * scanning and before_tool_call / before_agent_run interception.
6
+ * `api.registerHook` registers internal HOOK-style automation and does not
7
+ * participate in the agent-loop block/approval semantics ShieldCortex needs.
8
+ *
9
+ * NOT all scanning is fire-and-forget, and the distinction is the product:
10
+ *
11
+ * llm_input — OBSERVATION. Fire-and-forget; it has no blocking
12
+ * contract, so a detection here cannot stop the turn.
13
+ * before_agent_run — THE GATE (#225). Awaited by the gateway; its return
14
+ * value decides whether the run proceeds. Bounded end to
15
+ * end (CONVERSATION_SCAN_MAX_MS for the scan,
16
+ * CONVERSATION_NOTIFY_MAX_MS for the alert) because the
17
+ * user's turn waits on it, and failing OPEN on every
18
+ * internal error — as an EXPLICIT `{ outcome: 'pass' }`
19
+ * (#226), never as void, and never by throwing: the host
20
+ * registers this hook fail-CLOSED. See `gatePass`.
21
+ * before_tool_call — the Action Guard's gate, likewise awaited.
22
+ *
23
+ * Both conversation hooks honour `interceptor.conversation.posture`, including
24
+ * `off`, which is read before any scanner, audit write or cloud call.
9
25
  */
10
26
 
11
- import { createHash } from "node:crypto";
27
+ import { createHash, randomUUID } from "node:crypto";
12
28
  import fs from "node:fs/promises";
13
- import { existsSync, readFileSync, realpathSync } from "node:fs";
29
+ import { existsSync, readdirSync, readFileSync, realpathSync } from "node:fs";
14
30
  import path from "node:path";
15
- import { homedir } from "node:os";
31
+ import { homedir, hostname } from "node:os";
16
32
  import { fileURLToPath, pathToFileURL } from "node:url";
17
33
 
18
34
  import { readConversationAccess, describeRegisteredHooks } from './conversation-access.js';
19
35
  import { createSessionTaintStore } from './session-taint.js';
20
36
  import { classifyConversationOrigin } from './conversation-trust.js';
37
+ import type { ConversationTrustDecision } from './conversation-trust.js';
21
38
  import { createInterceptor, DEFAULT_CONFIG as DEFAULT_INTERCEPTOR_CONFIG } from './interceptor.js';
22
39
  import type { InterceptorConfig, BrokerRuntime } from './interceptor.js';
23
40
  import { syncInterceptEvent } from './intercept-ingest.js';
24
41
  import { cloudSync } from './cloud-sync.js';
42
+ import { createGatewayNotifyChannel } from './gateway-notify-channel.js';
43
+ import type { GatewayNotifyContext, NotifyChannelLike } from './gateway-notify-channel.js';
25
44
 
26
45
  // ==================== RESILIENT RUNTIME LOADER ====================
27
46
  // Resolves runtime.mjs from multiple locations so the plugin works both
@@ -46,6 +65,40 @@ type DefenceModule = {
46
65
  clean: boolean;
47
66
  injection: { clean: boolean; riskLevel: string; detections: unknown[] };
48
67
  };
68
+ /** #225 sink: the notify transport shared with the Action Guard (#143).
69
+ * Every member is optional — an older installed dist won't have them, and
70
+ * the guard must degrade to a loud log rather than fail. The field names
71
+ * MIRROR `NotifyConfig` in src/defence/iron-dome/notify-config.ts exactly
72
+ * (`webhookSecret`, not `secret`): a mirror that renames a field silently
73
+ * drops it, and the field this one would have dropped is the HMAC key —
74
+ * i.e. every POST would have gone out unsigned. */
75
+ normaliseNotifyConfig?: (raw: unknown) => {
76
+ enabled: boolean;
77
+ timeoutMs: number;
78
+ webhookUrl?: string;
79
+ webhookSecret?: string;
80
+ openclaw: boolean;
81
+ };
82
+ createWebhookNotifyChannel?: (opts: { url: string; secret?: string }) => NotifyChannelLike;
83
+ /** Builds the #225 notification. Used rather than an object literal so the
84
+ * bounding/truncation rules live in ONE place (see the module doc on
85
+ * operator-notify.ts) and a plugin cannot hand a channel a malformed shape. */
86
+ buildConversationThreatNotification?: (input: {
87
+ outcome: 'blocked' | 'observed' | 'unavailable';
88
+ posture: string;
89
+ summary: string;
90
+ reason: string;
91
+ sessionId?: string;
92
+ model?: string;
93
+ host?: string;
94
+ detectedAt: string;
95
+ }) => Record<string, unknown>;
96
+ /** The shared delivery core: bounded deadline, every failure normalised,
97
+ * nothing but the delivered boolean read back from a channel. */
98
+ deliverOperatorNotification?: (
99
+ notification: unknown,
100
+ deps: { channels: NotifyChannelLike[]; timeoutMs?: number },
101
+ ) => Promise<{ deliveredVia: string | null; attempts: Array<{ channel: string; result: { delivered: boolean; reason?: string } }> }>;
49
102
  };
50
103
 
51
104
  let runtimePromise: Promise<OpenClawRuntime> | null = null;
@@ -191,9 +244,16 @@ export function __resetConfigStateForTest(): void {
191
244
  _config = null;
192
245
  _configOverride = null;
193
246
  _lastShieldConfigRef = null;
247
+ // Re-arm the once-per-load config-failure warning (#226).
248
+ _shieldConfigLoadFailureLogged = false;
194
249
  _registered = false;
195
250
  _beforeToolCallRegistered = false;
196
251
  _registrationError = null;
252
+ _beforeAgentRunRequested = false;
253
+ _conversationAccessGranted = false;
254
+ _gatewayNotifyContext = null;
255
+ _hostRuntimeVersion = null;
256
+ __resetScanUnavailableAlertState();
197
257
  }
198
258
 
199
259
  type LlmInputEvent = {
@@ -263,6 +323,616 @@ interface InterceptorUserConfig {
263
323
  /** Reviewed-script allowlist (#189). Passed through RAW; validated
264
324
  * entry-by-entry inside createReviewedScriptCheck. */
265
325
  reviewedScripts?: unknown[];
326
+ /** Operator-notify transport (#143), reused by the #225 conversation sink.
327
+ * Passed through RAW for the same reason as `broker`: `normaliseNotifyConfig`
328
+ * in the main package is the single boundary that knows which values arm a
329
+ * channel, and splitting that judgement across two files is how one of the
330
+ * halves ends up being the lenient one. Until #225 this key was silently
331
+ * DROPPED here, so the plugin could not reach an operator at all even on a
332
+ * box where the Claude Code hook could. */
333
+ notify?: Record<string, unknown>;
334
+ };
335
+ /** Conversation firewall posture (#225). See CONVERSATION_POSTURES. */
336
+ conversation?: { posture?: ConversationPosture };
337
+ }
338
+
339
+ /**
340
+ * What the conversation firewall is allowed to DO about a detection (#225).
341
+ *
342
+ * Before this existed, scanning ran on `llm_input` — an OpenClaw *observation*
343
+ * hook with no blocking contract — and a detection's entire effect was a
344
+ * console line. The product implied a firewall and shipped a logger. These
345
+ * three postures make the difference explicit and reportable:
346
+ *
347
+ * off — do not scan the conversation at all
348
+ * observe — scan, audit, notify the operator, NEVER block
349
+ * enforce — additionally block the run (via `before_agent_run`)
350
+ *
351
+ * Default is `observe`, deliberately. It is exactly the behaviour that shipped
352
+ * before — but named honestly instead of implied to be protection — and #182
353
+ * says the guard's false-positive rate is still unmeasured. An unmeasured
354
+ * blocker in front of every turn would be a worse incident than the one this
355
+ * fixes; `enforce` is opt-in until that number exists.
356
+ */
357
+ export type ConversationPosture = 'off' | 'observe' | 'enforce';
358
+ const CONVERSATION_POSTURES: readonly ConversationPosture[] = ['off', 'observe', 'enforce'];
359
+
360
+ /** Resolve the configured posture. Anything unrecognised resolves DOWN to
361
+ * `observe`, never up to `enforce`: a typo must not silently start blocking
362
+ * every turn on an operator's box. */
363
+ export function conversationPosture(raw: unknown): ConversationPosture {
364
+ if (!raw || typeof raw !== 'object') return 'observe';
365
+ const value = (raw as { posture?: unknown }).posture;
366
+ return typeof value === 'string' && (CONVERSATION_POSTURES as readonly string[]).includes(value)
367
+ ? (value as ConversationPosture)
368
+ : 'observe';
369
+ }
370
+
371
+ /**
372
+ * A scan outcome as the conversation guard sees it.
373
+ *
374
+ * `available: false` marks a scan that could not be RUN at all — no defence
375
+ * module, no MCP fallback, or a throw inside the scanner. It is deliberately a
376
+ * separate axis from `clean`, and `clean` is set FALSE alongside it, because
377
+ * the bug this replaces returned `{ clean: true, summary: 'scan unavailable' }`
378
+ * from the ordinary unavailable path: every caller that read only `clean` then
379
+ * treated an unscanned turn as a scanned-and-fine one, silently, forever.
380
+ *
381
+ * `errored` is kept as an alias of "not available" for the pure decision
382
+ * function's existing contract.
383
+ */
384
+ export interface ConversationScanResult {
385
+ /** Only meaningful when `available` is true. False when unavailable, so a
386
+ * caller that ignores `available` still cannot read "clean". */
387
+ clean: boolean;
388
+ summary: string;
389
+ /** The scan actually ran and produced a verdict. */
390
+ available: boolean;
391
+ /** Set when the scan could not be completed. Mirrors `!available`. */
392
+ errored?: boolean;
393
+ /** Failure detail, for the audit row and the operator alert. Never contains
394
+ * scanned content. */
395
+ error?: string;
396
+ }
397
+
398
+ export interface ConversationDecision {
399
+ block: boolean;
400
+ notify: boolean;
401
+ audit: boolean;
402
+ reason: string | null;
403
+ /** What to tell a human happened to this turn — the same vocabulary the
404
+ * notification carries, so the audit row and the alert cannot disagree. */
405
+ outcome: 'clean' | 'blocked' | 'observed' | 'unavailable' | 'not-scanned';
406
+ }
407
+
408
+ /**
409
+ * The whole decision, as a pure function — no I/O, no hooks, so the posture
410
+ * semantics are testable directly and cannot drift as the plumbing changes.
411
+ *
412
+ * The key line is `notify` on a non-blocking detection: logging is not a sink.
413
+ * The #225 finding was that a HIGH verdict reached a log file and nothing else,
414
+ * so a real threat on a real box was seen by nobody. Under `observe` we still
415
+ * do not stop the turn — but a human hears about it.
416
+ *
417
+ * `trust` is the second input because BLOCKING is a consequence, and #235's rule
418
+ * is that source trust gates consequences (see conversation-trust.ts). It is
419
+ * optional, and its absence means "origin not established", which resolves
420
+ * toward enforcement rather than away from it: a caller that does not know who
421
+ * spoke has not proved the owner did.
422
+ */
423
+ export function evaluateConversationRun(
424
+ posture: ConversationPosture,
425
+ scan: ConversationScanResult,
426
+ trust?: ConversationTrustDecision,
427
+ ): ConversationDecision {
428
+ if (posture === 'off') {
429
+ return { block: false, notify: false, audit: false, reason: null, outcome: 'not-scanned' };
430
+ }
431
+
432
+ // Scanner failure fails OPEN — a broken scanner must not wedge every turn,
433
+ // which is the outcome ShieldCortex exists to prevent — but it is reported,
434
+ // because an unprotected turn must never read as a protected one. Note the
435
+ // condition: `available === false` OR the legacy `errored` flag, so a caller
436
+ // still constructing the old shape cannot route an unscanned turn into the
437
+ // clean branch.
438
+ if (scan.available === false || scan.errored) {
439
+ return {
440
+ block: false,
441
+ notify: true,
442
+ audit: true,
443
+ reason: `conversation scan unavailable (${scan.error ?? scan.summary}) — turn allowed UNSCANNED`,
444
+ outcome: 'unavailable',
445
+ };
446
+ }
447
+
448
+ if (scan.clean) {
449
+ return { block: false, notify: false, audit: false, reason: null, outcome: 'clean' };
450
+ }
451
+
452
+ // #235: the owner's own words are an instruction, so `enforce` does not act on
453
+ // them. This is the branch the whole trust module exists for — a block here
454
+ // does not warn the operator, it DESTROYS their message: OpenClaw keeps only
455
+ // the replacement text. Everything above still happened: the content was
456
+ // scanned, and `notify`/`audit` below are true whoever sent it. Only the
457
+ // consequence is withheld, and the reason says so on the row rather than
458
+ // leaving an enforce-posture host that did not block looking like a bug.
459
+ const trusted = trust !== undefined && !trust.mayTaint;
460
+ const block = posture === 'enforce' && !trusted;
461
+ return {
462
+ block,
463
+ notify: true,
464
+ audit: true,
465
+ reason:
466
+ posture === 'enforce' && trusted
467
+ ? `conversation threat: ${scan.summary} — NOT blocked: ${trust!.reason}`
468
+ : `conversation threat: ${scan.summary}`,
469
+ outcome: block ? 'blocked' : 'observed',
470
+ };
471
+ }
472
+
473
+ // ==================== CONVERSATION PLANE: HOST SUPPORT + CONSENT ============
474
+
475
+ /**
476
+ * The first OpenClaw build whose plugin SDK declares the `before_agent_run`
477
+ * gate. Established by inspecting published npm artifacts, not by guessing:
478
+ *
479
+ * 2026.5.7 — `hook-types.d.ts` has no `before_agent_run` anywhere (0 hits);
480
+ * CONVERSATION_HOOK_NAMES = llm_input, llm_output,
481
+ * before_agent_finalize, agent_end
482
+ * 2026.5.9-beta.1 — FIRST published build declaring it: in `PLUGIN_HOOK_NAMES`,
483
+ * in `CONVERSATION_HOOK_NAMES`, in `PluginHookHandlerMap`, with
484
+ * `PluginHookBeforeAgentRunResult = InputGateDecision | void`
485
+ * 2026.5.12 — first STABLE (non-prerelease) build with it (2026.5.10 and
486
+ * 2026.5.12 published betas in between; there is no plain
487
+ * 2026.5.9 release)
488
+ *
489
+ * Below this floor `api.on('before_agent_run', …)` is accepted by the API and
490
+ * then DROPPED by the registry with an `unknown typed hook … ignored`
491
+ * diagnostic — it does not throw. So a version check is the only honest way to
492
+ * know, and claiming enforcement without one is exactly the class of false
493
+ * green #222 is about.
494
+ *
495
+ * ── ONE FLOOR, THREE FILES ────────────────────────────────────────────────
496
+ *
497
+ * The authoritative value is the STABLE release, and it is stated in three
498
+ * places that cannot import each other:
499
+ *
500
+ * plugins/openclaw/index.ts — this constant
501
+ * src/integrations/openclaw-conversation-capability.ts
502
+ * — CONVERSATION_ENFORCEMENT_MIN_OPENCLAW
503
+ * plugins/openclaw/openclaw.plugin.json — engines.conversationGate
504
+ *
505
+ * THE BOUNDARY IS REAL, not a preference. The plugin ships as its own dist,
506
+ * compiled by `tsconfig.openclaw-plugin.json` with `rootDir:
507
+ * ./plugins/openclaw` and an explicit `include` list; a `src/` import does not
508
+ * merely offend layering, it fails to emit — and the src module imports
509
+ * `semver`, which the plugin bundle does not carry (hence the hand-rolled
510
+ * `compareOpenClawVersions` below). The manifest is JSON read by the host and
511
+ * imports nothing at all.
512
+ *
513
+ * So the three are pinned EQUAL by test instead of shared by import:
514
+ * `src/__tests__/conversation-gate-floor-parity-226.test.ts` reads all three
515
+ * and fails on drift. Change one, that test tells you about the other two.
516
+ *
517
+ * `CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW` is deliberately SUBORDINATE: it
518
+ * decides nothing an operator sees. Its only job is to mark the band where a
519
+ * version number alone cannot answer the question — see
520
+ * `hostSupportsConversationGate`.
521
+ */
522
+ export const CONVERSATION_GATE_MIN_OPENCLAW = '2026.5.12';
523
+ /**
524
+ * Documentation of when the hook first appeared, NOT a second floor.
525
+ *
526
+ * The previous cut used this as the support threshold, which made the plugin's
527
+ * operator-facing verdict disagree with the CLI's on any 2026.5.9-beta.1 →
528
+ * 2026.5.11 host: `shieldcortex doctor` said enforcement was unavailable while
529
+ * the plugin's own status line said supported. Two answers to one question is
530
+ * how the next false green gets built.
531
+ */
532
+ export const CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW = '2026.5.9-beta.1';
533
+
534
+ /**
535
+ * Compare two OpenClaw CalVer strings (`2026.5.12`, `2026.5.9-beta.1`).
536
+ * Deliberately local and tiny: the plugin build cannot import semver, and the
537
+ * only question asked is "is this host at or above the floor".
538
+ * Returns null when either side cannot be parsed — "unknown", never "yes".
539
+ */
540
+ export function compareOpenClawVersions(a: string, b: string): number | null {
541
+ const parse = (v: string): { nums: number[]; pre: string[] } | null => {
542
+ // Exactly three numeric parts, and only `-` introduces a prerelease. A
543
+ // trailing `.4` is NOT a prerelease tail — it is a version shape we do not
544
+ // understand, and the safe answer to that is "unknown".
545
+ const m = String(v ?? '').trim().match(/^(\d+)\.(\d+)\.(\d+)(?:-([0-9A-Za-z.-]+))?$/);
546
+ if (!m) return null;
547
+ return { nums: [Number(m[1]), Number(m[2]), Number(m[3])], pre: m[4] ? m[4].split('.') : [] };
548
+ };
549
+ const pa = parse(a);
550
+ const pb = parse(b);
551
+ if (!pa || !pb) return null;
552
+ for (let i = 0; i < 3; i++) {
553
+ if (pa.nums[i] !== pb.nums[i]) return pa.nums[i] < pb.nums[i] ? -1 : 1;
554
+ }
555
+ // A prerelease sorts BELOW the same numeric release (2026.5.9-beta.1 < 2026.5.9).
556
+ if (pa.pre.length === 0 && pb.pre.length === 0) return 0;
557
+ if (pa.pre.length === 0) return 1;
558
+ if (pb.pre.length === 0) return -1;
559
+ return comparePrerelease(pa.pre, pb.pre);
560
+ }
561
+
562
+ /** Semver prerelease precedence, restricted to what a CalVer tail can hold:
563
+ * numeric identifiers compare numerically, a numeric identifier sorts below an
564
+ * alphanumeric one, and a shorter identifier list sorts below an
565
+ * otherwise-equal longer one (`beta` < `beta.1` < `beta.2` < `beta.10`).
566
+ *
567
+ * The string compare this replaces put `beta.10` BELOW `beta.1`, so the tenth
568
+ * beta of the gate build was classified as predating the first — an error in
569
+ * the one direction this file must never make, since it demotes a host that
570
+ * HAS the gate to 'unsupported'. */
571
+ function comparePrerelease(a: string[], b: string[]): number {
572
+ const len = Math.max(a.length, b.length);
573
+ for (let i = 0; i < len; i++) {
574
+ const x = a[i];
575
+ const y = b[i];
576
+ if (x === undefined) return -1;
577
+ if (y === undefined) return 1;
578
+ const xNum = /^\d+$/.test(x);
579
+ const yNum = /^\d+$/.test(y);
580
+ if (xNum && yNum) {
581
+ const nx = Number(x);
582
+ const ny = Number(y);
583
+ if (nx !== ny) return nx < ny ? -1 : 1;
584
+ continue;
585
+ }
586
+ if (xNum !== yNum) return xNum ? -1 : 1;
587
+ if (x !== y) return x < y ? -1 : 1;
588
+ }
589
+ return 0;
590
+ }
591
+
592
+ /** What we could establish about the OpenClaw build we are running inside. */
593
+ export interface HostOpenClawProbe {
594
+ /** The host runtime version, or null when it could not be established. */
595
+ version: string | null;
596
+ /**
597
+ * Where `version` came from, so a reader can weigh it.
598
+ *
599
+ * 'runtime' — `api.runtime.version`, the host's own declared runtime
600
+ * version (`PluginRuntimeCore.version`). First choice: the
601
+ * gateway states it about itself.
602
+ * 'package.json' — the openclaw package.json found by walking up from the
603
+ * entry path. A fallback, and wrong on an install whose
604
+ * on-disk package and running process differ.
605
+ * null — no version evidence at all.
606
+ */
607
+ versionSource?: 'runtime' | 'package.json' | null;
608
+ /** The host package root, or null. */
609
+ root: string | null;
610
+ /**
611
+ * Whether THIS host's shipped hook declarations name `before_agent_run`.
612
+ * true/false when the declarations were found and read; null when they were
613
+ * not (an install without type declarations, an unusual layout).
614
+ *
615
+ * This is the primary evidence, ahead of the version string, because it is a
616
+ * property of the build actually on this box: a fork, a backport or a patched
617
+ * install answers correctly here and would be mis-classified by a version
618
+ * comparison. NOTE: the runtime `openclaw/plugin-sdk` entrypoint exports
619
+ * exactly ten symbols (ContextEngine helpers, onDiagnosticEvent, stringEnum,
620
+ * …) and none of them is the hook-name list — read off 2026.7.1's shipped
621
+ * `dist/plugin-sdk/index.js` — so importing the SDK and asking it directly is
622
+ * not available.
623
+ */
624
+ declaresGate: boolean | null;
625
+ }
626
+
627
+ /** Test seam: pins the host probe without touching disk. */
628
+ let _hostProbeOverride: HostOpenClawProbe | null | undefined;
629
+ export function __setHostOpenClawProbeForTest(p: HostOpenClawProbe | null | undefined): void {
630
+ _hostProbeOverride = p;
631
+ _hostProbeCache = undefined;
632
+ }
633
+ let _hostProbeCache: HostOpenClawProbe | undefined;
634
+
635
+ /**
636
+ * The host runtime version the gateway told us about, captured at register().
637
+ *
638
+ * `api.runtime.version` is declared by the host SDK as
639
+ * `PluginRuntimeCore.version: string` — the version of the OpenClaw runtime
640
+ * this plugin is loaded into (verified against the installed host's
641
+ * `dist/plugin-sdk/src/plugins/runtime/types-core.d.ts`, and `api.runtime` is
642
+ * on `OpenClawPluginApi` in the same build). It is NOT `api.version`, which is
643
+ * the plugin's own version and would answer a completely different question:
644
+ * comparing OUR version against an OpenClaw floor would classify every host as
645
+ * unsupported.
646
+ *
647
+ * It is preferred over the filesystem walk because it is the running process
648
+ * describing itself, where the walk infers from whichever package.json happens
649
+ * to sit above the entry path. Null until a host actually supplies it — an
650
+ * older gateway, a CLI invocation or a test rig may not, and that is UNKNOWN.
651
+ */
652
+ let _hostRuntimeVersion: string | null = null;
653
+
654
+ /** Test seam for the runtime-supplied host version. */
655
+ export function __setHostRuntimeVersionForTest(v: string | null): void {
656
+ _hostRuntimeVersion = v;
657
+ }
658
+
659
+ /**
660
+ * Record `api.runtime.version` if this host exposes it. Returns what was
661
+ * recorded (null when nothing usable was offered), and never throws: a host
662
+ * with an exotic `runtime` getter must not take the plugin's registration down.
663
+ */
664
+ export function recordHostRuntimeVersion(api: unknown): string | null {
665
+ try {
666
+ const runtime = (api as { runtime?: { version?: unknown } } | null | undefined)?.runtime;
667
+ const version = runtime?.version;
668
+ _hostRuntimeVersion = typeof version === 'string' && version.trim() ? version.trim() : null;
669
+ } catch {
670
+ _hostRuntimeVersion = null;
671
+ }
672
+ return _hostRuntimeVersion;
673
+ }
674
+
675
+ /** Does this host's shipped SDK declare the gate? Bounded, best-effort, and
676
+ * null on anything unexpected — an unreadable install is UNKNOWN, never
677
+ * "supported". */
678
+ function probeGateDeclaration(root: string): boolean | null {
679
+ const candidates: string[] = [];
680
+ // 2026.5.2-era layout: the declarations live under the plugin-sdk tree.
681
+ candidates.push(path.join(root, 'dist', 'plugin-sdk', 'src', 'plugins', 'hook-types.d.ts'));
682
+ // 2026.6+/2026.7 layout: a single hashed `hook-types-<hash>.d.ts` at dist root.
683
+ try {
684
+ const distDir = path.join(root, 'dist');
685
+ if (existsSync(distDir)) {
686
+ for (const name of readdirSync(distDir)) {
687
+ if (/^hook-types.*\.d\.ts$/.test(name)) candidates.push(path.join(distDir, name));
688
+ }
689
+ }
690
+ } catch { /* fall through to whatever candidates we have */ }
691
+
692
+ let sawAny = false;
693
+ for (const file of candidates) {
694
+ try {
695
+ if (!existsSync(file)) continue;
696
+ sawAny = true;
697
+ if (/\bbefore_agent_run\b/.test(readFileSync(file, 'utf-8'))) return true;
698
+ } catch { /* unreadable candidate — try the next */ }
699
+ }
700
+ return sawAny ? false : null;
701
+ }
702
+
703
+ /**
704
+ * Everything we can learn about the host OpenClaw build from inside the plugin.
705
+ *
706
+ * Two independent sources, both read, neither invented:
707
+ *
708
+ * - `api.runtime.version` — the running gateway's own statement of its
709
+ * version, captured at register() (see `recordHostRuntimeVersion`). This is
710
+ * the primary VERSION evidence when the host offers it. Note it is not
711
+ * `api.version`, which is this plugin's version.
712
+ * - the filesystem — walk up from the gateway's entry path to the package.json
713
+ * that names openclaw, then read that install's own shipped hook
714
+ * declarations. This is the version FALLBACK, and it is the only source of
715
+ * `declaresGate`, which stays the strongest gate-support evidence of the two
716
+ * (a backport or a fork answers it correctly where a version comparison
717
+ * cannot — see `hostSupportsConversationGate`).
718
+ *
719
+ * No process is ever spawned. A null everywhere is a legitimate, frequently
720
+ * correct answer (a CLI invocation, an unusual install layout) and callers must
721
+ * treat it as UNKNOWN — never as "supported".
722
+ */
723
+ export function detectHostOpenClaw(): HostOpenClawProbe {
724
+ const disk = detectHostOpenClawFromDisk();
725
+ // The runtime's own version outranks whatever package.json the walk landed
726
+ // on — but only for the version; `declaresGate` and `root` are disk facts and
727
+ // are carried through untouched.
728
+ if (_hostRuntimeVersion) return { ...disk, version: _hostRuntimeVersion, versionSource: 'runtime' };
729
+ return disk;
730
+ }
731
+
732
+ function detectHostOpenClawFromDisk(): HostOpenClawProbe {
733
+ if (_hostProbeOverride !== undefined) return _hostProbeOverride ?? { version: null, root: null, declaresGate: null };
734
+ if (_hostProbeCache !== undefined) return _hostProbeCache;
735
+ _hostProbeCache = (() => {
736
+ const empty: HostOpenClawProbe = { version: null, root: null, declaresGate: null, versionSource: null };
737
+ const entry = process.argv?.[1];
738
+ if (!entry || typeof entry !== 'string') return empty;
739
+ let current: string;
740
+ try {
741
+ current = path.dirname(realpathSync(entry));
742
+ } catch {
743
+ current = path.dirname(entry);
744
+ }
745
+ let previous = '';
746
+ for (let i = 0; i < 8 && current !== previous; i++) {
747
+ try {
748
+ const pkgPath = path.join(current, 'package.json');
749
+ if (existsSync(pkgPath)) {
750
+ const hostPkg = JSON.parse(readFileSync(pkgPath, 'utf-8')) as { name?: unknown; version?: unknown };
751
+ if (hostPkg?.name === 'openclaw') {
752
+ const version = typeof hostPkg.version === 'string' ? hostPkg.version : null;
753
+ return {
754
+ version,
755
+ versionSource: version ? ('package.json' as const) : null,
756
+ root: current,
757
+ declaresGate: probeGateDeclaration(current),
758
+ };
759
+ }
760
+ }
761
+ } catch { /* keep walking up */ }
762
+ previous = current;
763
+ current = path.dirname(current);
764
+ }
765
+ return empty;
766
+ })();
767
+ return _hostProbeCache;
768
+ }
769
+
770
+ /** Convenience for callers that only want the version string. */
771
+ export function detectHostOpenClawVersion(): string | null {
772
+ return detectHostOpenClaw().version;
773
+ }
774
+
775
+ export type GateSupport = 'supported' | 'unsupported' | 'unknown';
776
+
777
+ /**
778
+ * Does this host have the `before_agent_run` gate at all?
779
+ *
780
+ * Order matters: what the installed build DECLARES outranks what its version
781
+ * number implies, and both outrank a guess. There is no branch here that
782
+ * returns 'supported' without evidence.
783
+ */
784
+ export function hostSupportsConversationGate(probe: HostOpenClawProbe | string | null): GateSupport {
785
+ const resolved: HostOpenClawProbe =
786
+ typeof probe === 'string' || probe === null
787
+ ? { version: probe, root: null, declaresGate: null }
788
+ : probe;
789
+ if (resolved.declaresGate === true) return 'supported';
790
+ if (resolved.declaresGate === false) return 'unsupported';
791
+ if (!resolved.version) return 'unknown';
792
+ const cmp = compareOpenClawVersions(resolved.version, CONVERSATION_GATE_MIN_OPENCLAW);
793
+ if (cmp === null) return 'unknown';
794
+ if (cmp >= 0) return 'supported';
795
+
796
+ // Below the STABLE floor. One band inside that is not honestly 'unsupported':
797
+ // 2026.5.9-beta.1 → 2026.5.11 ship the hook as a prerelease, so calling them
798
+ // unsupported would tell an operator "no posture can block a turn on this
799
+ // host" about a host that blocks. The opposite claim is worse still, so
800
+ // neither is made: this is the absence of a measurement, and
801
+ // `describeConversationPlane` renders it as UNPROVEN and active:false.
802
+ //
803
+ // In practice a real prerelease install lands on `declaresGate` above and
804
+ // never reaches here — this branch is what happens when the declarations
805
+ // could not be read either, i.e. when we genuinely do not know.
806
+ const pre = compareOpenClawVersions(resolved.version, CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW);
807
+ if (pre === null) return 'unknown';
808
+ return pre >= 0 ? 'unknown' : 'unsupported';
809
+ }
810
+
811
+ /**
812
+ * Read the operator's CONVERSATION-ACCESS consent for this plugin from the
813
+ * host config: `plugins.entries.<id>.hooks.allowConversationAccess === true`.
814
+ *
815
+ * OpenClaw refuses every conversation hook for a non-bundled plugin without
816
+ * this exact value (registry: `record.origin !== "bundled" &&
817
+ * explicitConversationAccess !== true`). `llm_input` and `llm_output` are on
818
+ * that list in every build; `before_agent_run` joins it in 2026.5.9-beta.1,
819
+ * the same build that first declares the gate at all — so from there on the
820
+ * grant governs the conversation firewall's enforcement point too.
821
+ * Strict `true` only, matching the host: `undefined` and `false` are the same
822
+ * refusal there, and reading them differently here would report protection the
823
+ * gateway is not providing.
824
+ *
825
+ * It is the operator's per-box CONSENT grant, and this plugin only ever READS
826
+ * it. Nothing on the plugin's own path — `register()`, a hook, a background
827
+ * refresh — may write it: a security product that silently grants itself the
828
+ * right to read every conversation is the behaviour this product exists to
829
+ * catch. The only thing that may set it is an explicit, operator-initiated
830
+ * install/repair that says so out loud (#225: "the installer must never set it
831
+ * silently"). Absence is therefore reported, loudly and by name, rather than
832
+ * fixed from in here.
833
+ */
834
+ export function readConversationAccessGrant(rootConfig: unknown): boolean {
835
+ if (!rootConfig || typeof rootConfig !== 'object' || Array.isArray(rootConfig)) return false;
836
+ const entries = (rootConfig as {
837
+ plugins?: { entries?: Record<string, { hooks?: { allowConversationAccess?: unknown } } | undefined> };
838
+ }).plugins?.entries;
839
+ const entry = entries?.[PLUGIN_ID] ?? entries?.[PLUGIN_PACKAGE_NAME];
840
+ return entry?.hooks?.allowConversationAccess === true;
841
+ }
842
+
843
+ /**
844
+ * Everything we can HONESTLY say about the conversation plane on this host.
845
+ *
846
+ * Note what is not here: any claim that the gateway accepted the hook.
847
+ * `api.on()` returns void, never throws on an unknown hook name, and never
848
+ * throws on a refused conversation hook — it records a diagnostic and returns —
849
+ * so "the host accepted our registration" is not knowable from inside the
850
+ * plugin, and a flag asserting it would be decoration. What IS knowable is the
851
+ * two facts that decide the outcome: whether the host build has the gate, and
852
+ * whether the operator granted conversation access.
853
+ */
854
+ export interface ConversationPlaneState {
855
+ posture: ConversationPosture;
856
+ /** We called `api.on('before_agent_run', …)` this session. */
857
+ hookRequested: boolean;
858
+ gateSupport: GateSupport;
859
+ hostOpenClawVersion: string | null;
860
+ consentGranted: boolean;
861
+ /** True only when the posture is on AND both preconditions hold. */
862
+ active: boolean;
863
+ /** One line, exact about the evidence, for status and doctor. */
864
+ summary: string;
865
+ }
866
+
867
+ export function describeConversationPlane(input: {
868
+ posture: ConversationPosture;
869
+ hookRequested: boolean;
870
+ gateSupport: GateSupport;
871
+ hostOpenClawVersion: string | null;
872
+ consentGranted: boolean;
873
+ }): ConversationPlaneState {
874
+ const { posture, hookRequested, gateSupport, hostOpenClawVersion, consentGranted } = input;
875
+ const hostText = hostOpenClawVersion ? `OpenClaw ${hostOpenClawVersion}` : 'OpenClaw version undetermined';
876
+
877
+ if (posture === 'off') {
878
+ return {
879
+ ...input,
880
+ active: false,
881
+ summary: 'off — conversation scanning disabled by config (interceptor.conversation.posture=off)',
882
+ };
883
+ }
884
+ if (!consentGranted) {
885
+ return {
886
+ ...input,
887
+ active: false,
888
+ summary:
889
+ `INACTIVE: conversation access NOT granted on this host — set plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess=true ` +
890
+ 'in openclaw.json (operator consent; the installer will never set it for you). Until then the gateway REFUSES llm_input and ' +
891
+ 'llm_output for this plugin — and, on builds that have it, before_agent_run too: nothing on the conversation path is scanned or blocked',
892
+ };
893
+ }
894
+ if (gateSupport === 'unsupported') {
895
+ return {
896
+ ...input,
897
+ active: false,
898
+ summary:
899
+ `INACTIVE for enforcement: ${hostText} predates the before_agent_run gate ` +
900
+ `(floor ${CONVERSATION_GATE_MIN_OPENCLAW}; first seen as a prerelease in ${CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW}) — ` +
901
+ 'observation only; no posture can block a turn on this host',
902
+ };
903
+ }
904
+ if (!hookRequested) {
905
+ return { ...input, active: false, summary: 'INACTIVE: the before_agent_run hook was not registered this session' };
906
+ }
907
+ if (gateSupport === 'unknown') {
908
+ // UNPROVEN IS NOT ACTIVE. Every other branch above is a fact we read — the
909
+ // posture, the grant, the host build. This one is the absence of a
910
+ // measurement: we could not establish that this host has the gate at all,
911
+ // and `api.on` does not acknowledge a registration, so nothing here knows
912
+ // whether the hook exists. Reporting `active: true` with a caveat glued to
913
+ // the summary string — which is what this did — means every caller that
914
+ // reads the boolean instead of the prose (status renderers, doctor, any
915
+ // future check) claims a live firewall on evidence nobody has. Under
916
+ // `enforce` that is the worst version of it: the operator believes turns
917
+ // are being blocked on a host where the gate may be silently dropped.
918
+ return {
919
+ ...input,
920
+ active: false,
921
+ summary:
922
+ `UNPROVEN: could not verify that host ${hostText} provides the before_agent_run gate ` +
923
+ `(no runtime version, no readable hook declarations), and the plugin API does not acknowledge a registration. ` +
924
+ (posture === 'enforce'
925
+ ? 'The posture is enforce, so a dirty verdict WOULD block the run where the gate exists — but that it exists here is not established. Treat this host as observation-only until it is.'
926
+ : 'Detections are audited and sent to the operator where the hook runs at all; nothing is blocked in this posture regardless.'),
927
+ };
928
+ }
929
+ return {
930
+ ...input,
931
+ active: true,
932
+ summary:
933
+ posture === 'enforce'
934
+ ? 'enforce — a dirty verdict BLOCKS the run via before_agent_run'
935
+ : 'observe — detections are audited and sent to the operator; turns are NOT blocked',
266
936
  };
267
937
  }
268
938
 
@@ -276,6 +946,20 @@ interface SCConfig {
276
946
  openclawAutoMemoryNoveltyThreshold?: number;
277
947
  openclawAutoMemoryMaxRecent?: number;
278
948
  interceptor?: InterceptorUserConfig;
949
+ /**
950
+ * Source trust for conversation content (see conversation-trust.ts).
951
+ *
952
+ * Declared and PARSED here, not read off an untyped cast. `normaliseConfig`
953
+ * is a strict allowlist — that is the whole point of #112/#115 — so a key it
954
+ * does not name is dropped on both config paths (~/.shieldcortex/config.json
955
+ * and the openclaw.json plugin entry). Landed as a cast against an SCConfig
956
+ * that had no such field, `conversationTrust.trustOwnerInput` was therefore
957
+ * read as `undefined` on every host, and the documented opt-out could not be
958
+ * turned on by anyone. It matters more now that trust also gates BLOCKING:
959
+ * an operator who wants owner input policed like everything else has to have
960
+ * a working way to say so.
961
+ */
962
+ conversationTrust?: { trustOwnerInput?: boolean };
279
963
  }
280
964
 
281
965
  const PLUGIN_ID = "shieldcortex-realtime";
@@ -330,6 +1014,21 @@ const PLUGIN_CONFIG_UI_HINTS = {
330
1014
  label: "Enable Tool Call Interceptor",
331
1015
  help: "Scan memory-write tool calls and gate suspicious content behind user approval.",
332
1016
  },
1017
+ // #226: these two exist in openclaw.plugin.json's uiHints and were missing
1018
+ // here, so the host UI and the plugin's own declared hints described
1019
+ // different sets of settings. The manifest parity test now pins the two key
1020
+ // sets EQUAL in both directions, because a hint present on only one side is
1021
+ // a setting one surface documents and the other silently omits.
1022
+ "interceptor.severityActions.high": {
1023
+ label: "High Severity Action",
1024
+ help: "Action for high-severity threats: log, warn, or require_approval.",
1025
+ advanced: true,
1026
+ },
1027
+ "interceptor.severityActions.critical": {
1028
+ label: "Critical Severity Action",
1029
+ help: "Action for critical-severity threats: log, warn, or require_approval.",
1030
+ advanced: true,
1031
+ },
333
1032
  "interceptor.actionGuard.enabled": {
334
1033
  label: "Action Guard",
335
1034
  help: "Gate dangerous shell/file/network/git tool calls before they execute. Catastrophic operations are always blocked while enabled.",
@@ -358,6 +1057,39 @@ const PLUGIN_CONFIG_UI_HINTS = {
358
1057
  help: "Off = every dangerous-tier action still waits for you; the broker can then only harden, never release.",
359
1058
  advanced: true,
360
1059
  },
1060
+ // #225. The posture is the whole product claim on the conversation path, so
1061
+ // it is NOT marked advanced: an operator must be able to see, in the UI that
1062
+ // configures this plugin, whether the firewall in front of their prompts can
1063
+ // stop anything.
1064
+ "interceptor.conversation.posture": {
1065
+ label: "Conversation Firewall",
1066
+ help:
1067
+ "What the conversation firewall does with a detection on the input path. " +
1068
+ "off = do not scan; observe = scan, audit and alert the operator but never stop the turn (default); " +
1069
+ "enforce = block the run via before_agent_run. Requires plugins.entries.shieldcortex-realtime.hooks.allowConversationAccess=true " +
1070
+ "on this host — OpenClaw refuses conversation hooks without that operator grant, and ShieldCortex will never set it for you.",
1071
+ },
1072
+ "interceptor.actionGuard.notify.enabled": {
1073
+ label: "Operator Notifications",
1074
+ help: "Reach a human when the guard holds an action, or when the conversation firewall detects a threat. Off by default.",
1075
+ advanced: true,
1076
+ },
1077
+ "interceptor.actionGuard.notify.webhookUrl": {
1078
+ label: "Notify Webhook URL",
1079
+ help: "http(s) endpoint the notification is POSTed to. Conversation-firewall alerts carry no approve/deny affordance — there is nothing to approve.",
1080
+ advanced: true,
1081
+ },
1082
+ "interceptor.actionGuard.notify.webhookSecret": {
1083
+ label: "Notify Webhook Secret",
1084
+ help: "HMAC-SHA256 key for X-ShieldCortex-Signature, so the receiver can reject spoofed POSTs.",
1085
+ sensitive: true,
1086
+ advanced: true,
1087
+ },
1088
+ "interceptor.actionGuard.notify.openclaw": {
1089
+ label: "Notify via OpenClaw",
1090
+ help: "Deliver through the gateway's own channel where the runtime provides that seam.",
1091
+ advanced: true,
1092
+ },
361
1093
  } as const;
362
1094
 
363
1095
  const SEVERITY_ACTION_SCHEMA = {
@@ -376,64 +1108,143 @@ const FAILURE_POLICY_SCHEMA = {
376
1108
  ),
377
1109
  };
378
1110
 
379
- const INTERCEPTOR_JSON_SCHEMA = {
1111
+ /** #225. Mirrored verbatim into openclaw.plugin.json's configSchema — the host
1112
+ * validates the on-disk config against THAT file, so a posture accepted here
1113
+ * and absent there is a config an operator writes from our own docs and the
1114
+ * gateway rejects. */
1115
+ const CONVERSATION_JSON_SCHEMA = {
1116
+ type: "object",
1117
+ additionalProperties: false,
1118
+ properties: {
1119
+ posture: {
1120
+ type: "string",
1121
+ enum: [...CONVERSATION_POSTURES],
1122
+ default: "observe",
1123
+ description:
1124
+ "off = do not scan the conversation; observe = scan, audit and alert but never block (default); " +
1125
+ "enforce = block the run on a dirty verdict via before_agent_run.",
1126
+ },
1127
+ },
1128
+ };
1129
+
1130
+ /** #235/#226. Mirrored verbatim into openclaw.plugin.json's configSchema, for
1131
+ * the same reason as the conversation posture above: the host validates the
1132
+ * on-disk config against THAT file, and a key our parser reads but neither
1133
+ * schema declares is one an operator cannot set at all. */
1134
+ const CONVERSATION_TRUST_JSON_SCHEMA = {
1135
+ type: "object",
1136
+ additionalProperties: false,
1137
+ properties: {
1138
+ trustOwnerInput: {
1139
+ type: "boolean",
1140
+ default: true,
1141
+ description:
1142
+ "Default true: a message the host attributes to the gateway OWNER is an instruction, so a detection in it " +
1143
+ "is audited and alerted but never taints the session or blocks the turn. Set false on a host where the owner " +
1144
+ "routinely pastes untrusted content and you would rather have the caution than the quiet. Content from " +
1145
+ "anyone else — including another agent on a trusted channel — is data regardless of this setting.",
1146
+ },
1147
+ },
1148
+ };
1149
+
1150
+ /**
1151
+ * The Action Guard block, declared ONCE and mounted in BOTH places the parser
1152
+ * accepts it (#226).
1153
+ *
1154
+ * `normaliseConfig` has read a TOP-LEVEL `actionGuard` since #209 — that is the
1155
+ * canonical location, and `interceptor.actionGuard` is the deprecated alias
1156
+ * kept for pre-#209 configs. The schemas said the opposite: only the nested
1157
+ * alias was declared, under `additionalProperties: false`, so a config written
1158
+ * from our own documentation — `actionGuard.notify` at the top level — was
1159
+ * rejected as an unknown key by any host that validates against the schema.
1160
+ * The parser would have kept it; the config never reached the parser. That is
1161
+ * the shape behind the original `parsedNotify: null` reproduction.
1162
+ *
1163
+ * One constant, two mount points, so the two can never drift. Mirrored by hand
1164
+ * into openclaw.plugin.json's configSchema (the host validates the on-disk
1165
+ * config against THAT file) and pinned equal by manifest-config-schema-226.test.ts.
1166
+ */
1167
+ const ACTION_GUARD_JSON_SCHEMA = {
380
1168
  type: "object",
381
1169
  additionalProperties: false,
382
1170
  properties: {
383
1171
  enabled: { type: "boolean" },
384
- severityActions: SEVERITY_ACTION_SCHEMA,
385
- failurePolicy: FAILURE_POLICY_SCHEMA,
386
- actionGuard: {
1172
+ enforce: { type: "boolean" },
1173
+ autoApprove: { type: "array", items: { type: "string" } },
1174
+ auditAllows: { type: "boolean" },
1175
+ // #143. Mirrors normaliseBrokerConfig's allowlist; that function still
1176
+ // has the last word, so a value that slips past the schema is still
1177
+ // range-checked (and dropped) before the broker sees it.
1178
+ broker: {
387
1179
  type: "object",
388
1180
  additionalProperties: false,
389
1181
  properties: {
390
1182
  enabled: { type: "boolean" },
391
- enforce: { type: "boolean" },
392
- autoApprove: { type: "array", items: { type: "string" } },
393
- auditAllows: { type: "boolean" },
394
- // #143. Mirrors normaliseBrokerConfig's allowlist; that function still
395
- // has the last word, so a value that slips past the schema is still
396
- // range-checked (and dropped) before the broker sees it.
397
- broker: {
1183
+ allowPreClear: { type: "boolean" },
1184
+ preClearConfidence: { type: "number", minimum: 0.9, maximum: 1 },
1185
+ judgeTimeoutMs: { type: "number", minimum: 500, maximum: 60000 },
1186
+ approvalTimeoutMs: {
398
1187
  type: "object",
399
1188
  additionalProperties: false,
400
1189
  properties: {
401
- enabled: { type: "boolean" },
402
- allowPreClear: { type: "boolean" },
403
- preClearConfidence: { type: "number", minimum: 0.9, maximum: 1 },
404
- judgeTimeoutMs: { type: "number", minimum: 500, maximum: 60000 },
405
- approvalTimeoutMs: {
406
- type: "object",
407
- additionalProperties: false,
408
- properties: {
409
- sensitive: { type: "number", minimum: 1000, maximum: 3600000 },
410
- dangerous: { type: "number", minimum: 1000, maximum: 3600000 },
411
- },
412
- },
413
- model: { type: "string" },
1190
+ sensitive: { type: "number", minimum: 1000, maximum: 3600000 },
1191
+ dangerous: { type: "number", minimum: 1000, maximum: 3600000 },
414
1192
  },
415
1193
  },
416
- // #189. Each entry pins one script by absolute path + content hash;
417
- // createReviewedScriptCheck has the last word on every field.
418
- reviewedScripts: {
419
- type: "array",
420
- items: {
421
- type: "object",
422
- additionalProperties: false,
423
- properties: {
424
- path: { type: "string" },
425
- sha256: { type: "string" },
426
- note: { type: "string" },
427
- addedAt: { type: "number" },
428
- },
429
- required: ["path", "sha256"],
430
- },
1194
+ model: { type: "string" },
1195
+ },
1196
+ },
1197
+ // #143/#225. Mirrors NotifyConfig in notify-config.ts, which still has
1198
+ // the last word (strict-true booleans, http(s)-only URL, bounded
1199
+ // timeout). Declared here because `additionalProperties: false` above
1200
+ // means an undeclared key makes the WHOLE containing block invalid on
1201
+ // a host that validates config against this schema — which is how the
1202
+ // Action Guard's notify transport came to be unusable from inside the
1203
+ // gateway plugin at all.
1204
+ notify: {
1205
+ type: "object",
1206
+ additionalProperties: false,
1207
+ properties: {
1208
+ enabled: { type: "boolean" },
1209
+ webhookUrl: { type: "string" },
1210
+ webhookSecret: { type: "string" },
1211
+ openclaw: { type: "boolean" },
1212
+ timeoutMs: { type: "number", minimum: 500, maximum: 60000 },
1213
+ },
1214
+ },
1215
+ // #189. Each entry pins one script by absolute path + content hash;
1216
+ // createReviewedScriptCheck has the last word on every field.
1217
+ reviewedScripts: {
1218
+ type: "array",
1219
+ items: {
1220
+ type: "object",
1221
+ additionalProperties: false,
1222
+ properties: {
1223
+ path: { type: "string" },
1224
+ sha256: { type: "string" },
1225
+ note: { type: "string" },
1226
+ addedAt: { type: "number" },
431
1227
  },
1228
+ required: ["path", "sha256"],
432
1229
  },
433
1230
  },
434
1231
  },
435
1232
  };
436
1233
 
1234
+ const INTERCEPTOR_JSON_SCHEMA = {
1235
+ type: "object",
1236
+ additionalProperties: false,
1237
+ properties: {
1238
+ enabled: { type: "boolean" },
1239
+ severityActions: SEVERITY_ACTION_SCHEMA,
1240
+ failurePolicy: FAILURE_POLICY_SCHEMA,
1241
+ conversation: CONVERSATION_JSON_SCHEMA,
1242
+ /** The DEPRECATED alias (#209). Still accepted, still parsed, still
1243
+ * gap-fills the canonical top-level block key by key. */
1244
+ actionGuard: ACTION_GUARD_JSON_SCHEMA,
1245
+ },
1246
+ };
1247
+
437
1248
  const PLUGIN_CONFIG_JSON_SCHEMA = {
438
1249
  type: "object",
439
1250
  additionalProperties: false,
@@ -456,6 +1267,15 @@ const PLUGIN_CONFIG_JSON_SCHEMA = {
456
1267
  // #112: without this, `additionalProperties: false` declared the whole
457
1268
  // interceptor block invalid — mirror openclaw.plugin.json's configSchema.
458
1269
  interceptor: INTERCEPTOR_JSON_SCHEMA,
1270
+ // #209/#226: the CANONICAL Action Guard location. normaliseConfig has read
1271
+ // it here since #209 and folds it over the nested alias; the schema did not
1272
+ // declare it, so `additionalProperties: false` rejected the documented
1273
+ // config shape before the parser ever saw it.
1274
+ actionGuard: ACTION_GUARD_JSON_SCHEMA,
1275
+ // #235/#226: source trust. Same reason as `actionGuard` above — the parser
1276
+ // reads it, so the schema must declare it or `additionalProperties: false`
1277
+ // rejects the whole config an operator writes from our documentation.
1278
+ conversationTrust: CONVERSATION_TRUST_JSON_SCHEMA,
459
1279
  },
460
1280
  };
461
1281
 
@@ -494,6 +1314,17 @@ let _registered = false;
494
1314
  // unattended Codex agents even a no-op registered hook changes how OpenClaw
495
1315
  // resolves approvals).
496
1316
  let _beforeToolCallRegistered = false;
1317
+ // #225: whether we CALLED api.on('before_agent_run', …) this session — nothing
1318
+ // more. It is deliberately not named "…Accepted": `api.on` returns void, never
1319
+ // throws on an unknown hook name, and never throws when the host refuses a
1320
+ // conversation hook (it records a diagnostic and returns), so acceptance is not
1321
+ // observable from in here. The two facts that decide whether the plane is live
1322
+ // — host build ≥ the gate's floor, and the operator's allowConversationAccess
1323
+ // grant — are read separately and reported by describeConversationPlane().
1324
+ let _beforeAgentRunRequested = false;
1325
+ // The operator's conversation-access grant as read from the host config at
1326
+ // register() time. Reported by /shieldcortex-status; never written by us.
1327
+ let _conversationAccessGranted = false;
497
1328
  // #134 §2: register() wraps its whole body in try/catch so a plugin failure
498
1329
  // never blocks channel startup — correct, but it used to report the failure
499
1330
  // with a bare console.warn (bypasses the gateway's structured log, so the
@@ -549,6 +1380,20 @@ function normaliseConfig(raw: unknown, dropped?: string[]): SCConfig {
549
1380
  const interceptor = normaliseInterceptorConfig(value.interceptor, dropped);
550
1381
  if (interceptor) config.interceptor = interceptor;
551
1382
 
1383
+ // Source trust (#235, wired to the gate in #226). Booleans only, and the
1384
+ // block is kept only when it holds a valid one: a `trustOwnerInput: "false"`
1385
+ // typo — the exact shape of the #112 incident — must not read as the opt-out
1386
+ // having been applied, because the operator who wrote it believes owner input
1387
+ // is being policed and it would not be.
1388
+ if (value.conversationTrust && typeof value.conversationTrust === "object" && !Array.isArray(value.conversationTrust)) {
1389
+ const trustRaw = value.conversationTrust as Record<string, unknown>;
1390
+ if (typeof trustRaw.trustOwnerInput === "boolean") {
1391
+ config.conversationTrust = { trustOwnerInput: trustRaw.trustOwnerInput };
1392
+ } else if (trustRaw.trustOwnerInput !== undefined) {
1393
+ dropped?.push("conversationTrust.trustOwnerInput");
1394
+ }
1395
+ }
1396
+
552
1397
  // #209: single source of truth for the Action Guard. A top-level
553
1398
  // `actionGuard` block governs every surface; `interceptor.actionGuard` is a
554
1399
  // deprecated alias kept as per-key gap-fill so pre-#209 configs keep their
@@ -650,6 +1495,19 @@ function normaliseActionGuardBlock(
650
1495
  if (Array.isArray(rawGuard.reviewedScripts)) {
651
1496
  guard.reviewedScripts = [...rawGuard.reviewedScripts];
652
1497
  }
1498
+ // #143/#225: the notify transport. Same passthrough discipline again —
1499
+ // normaliseNotifyConfig is the boundary. Shallow-copied rather than aliased
1500
+ // (the #115 reason: a later in-place mutation of the host config object must
1501
+ // not reach into the normalised one), and a non-object is DROPPED by name so
1502
+ // the #115 warn log can say which key was ignored, rather than silently
1503
+ // leaving the operator with a transport that never fires.
1504
+ if (rawGuard.notify !== undefined) {
1505
+ if (rawGuard.notify && typeof rawGuard.notify === 'object' && !Array.isArray(rawGuard.notify)) {
1506
+ guard.notify = { ...(rawGuard.notify as Record<string, unknown>) };
1507
+ } else {
1508
+ dropped?.push(`${pathPrefix}.notify`);
1509
+ }
1510
+ }
653
1511
  return Object.keys(guard).length > 0 ? guard : undefined;
654
1512
  }
655
1513
 
@@ -672,6 +1530,24 @@ function normaliseInterceptorConfig(raw: unknown, dropped?: string[]): Intercept
672
1530
  const actionGuard = normaliseActionGuardBlock(value.actionGuard, dropped, "interceptor.actionGuard");
673
1531
  if (actionGuard) out.actionGuard = actionGuard;
674
1532
 
1533
+ // #225: conversation posture. An invalid value is DROPPED (and named in the
1534
+ // warn log) rather than coerced, so `conversationPosture()` falls back to
1535
+ // `observe` — the safe direction. A typo must never start blocking turns.
1536
+ if (value.conversation !== undefined) {
1537
+ if (value.conversation && typeof value.conversation === "object" && !Array.isArray(value.conversation)) {
1538
+ const posture = (value.conversation as { posture?: unknown }).posture;
1539
+ if (posture !== undefined) {
1540
+ if (typeof posture === "string" && (CONVERSATION_POSTURES as readonly string[]).includes(posture)) {
1541
+ out.conversation = { posture: posture as ConversationPosture };
1542
+ } else {
1543
+ dropped?.push("interceptor.conversation.posture");
1544
+ }
1545
+ }
1546
+ } else {
1547
+ dropped?.push("interceptor.conversation");
1548
+ }
1549
+ }
1550
+
675
1551
  // #115: empty/all-invalid normalises to undefined, not {} — {} is truthy
676
1552
  // and made applyPluginConfigOverride treat a no-op interceptor block as a
677
1553
  // real override, inconsistent with normaliseSeverityMap's own contract.
@@ -684,6 +1560,19 @@ function normaliseInterceptorConfig(raw: unknown, dropped?: string[]): Intercept
684
1560
  * - `interceptor` deep-merges PER KEY — per severity entry, per guard flag —
685
1561
  * so an override that sets one nested key does not wholesale-discard the
686
1562
  * base's other interceptor settings.
1563
+ * - `actionGuard.notify` and `actionGuard.broker` deep-merge per key too
1564
+ * (#226). They are the two OBJECT-valued guard keys, and a shallow spread
1565
+ * replaced them wholesale: an openclaw.json entry that says only
1566
+ * `notify: { enabled: true }` — the shape the UI writes when an operator
1567
+ * ticks "Operator Notifications" — discarded the shield config's
1568
+ * `webhookUrl` and `webhookSecret`, leaving notify ARMED with no channel and
1569
+ * no signing key. Every alert then reported "enabled but no channel is
1570
+ * configured/buildable on this host", which is the #143 silent-no-sink
1571
+ * failure in a new place.
1572
+ * - ARRAY-valued keys (`autoApprove`, `reviewedScripts`) still REPLACE. They
1573
+ * are allowlists: merging two of them would union permissions an operator
1574
+ * removed back into the effective config, which is the wrong direction for a
1575
+ * security control.
687
1576
  * - Explicit values, including `false`, always win over base values; absent
688
1577
  * keys fall through to the base.
689
1578
  * - Defaults are NOT applied here: DEFAULT_INTERCEPTOR_CONFIG only fills the
@@ -692,6 +1581,13 @@ function normaliseInterceptorConfig(raw: unknown, dropped?: string[]): Intercept
692
1581
  */
693
1582
  function mergeConfigs(base: SCConfig, override: SCConfig): SCConfig {
694
1583
  const merged: SCConfig = { ...base, ...override };
1584
+ // Per-key, like every other nested block here. A plain spread would let an
1585
+ // openclaw.json entry that mentions `conversationTrust` at all replace the
1586
+ // shield-config block wholesale, silently reverting an opt-out set in the
1587
+ // file the operator considers authoritative.
1588
+ if (base.conversationTrust || override.conversationTrust) {
1589
+ merged.conversationTrust = { ...base.conversationTrust, ...override.conversationTrust };
1590
+ }
695
1591
  if (base.interceptor || override.interceptor) {
696
1592
  const b = base.interceptor ?? {};
697
1593
  const o = override.interceptor ?? {};
@@ -703,7 +1599,12 @@ function mergeConfigs(base: SCConfig, override: SCConfig): SCConfig {
703
1599
  merged.interceptor.failurePolicy = { ...b.failurePolicy, ...o.failurePolicy };
704
1600
  }
705
1601
  if (b.actionGuard || o.actionGuard) {
706
- merged.interceptor.actionGuard = { ...b.actionGuard, ...o.actionGuard };
1602
+ const bg = b.actionGuard ?? {};
1603
+ const og = o.actionGuard ?? {};
1604
+ const guard: NonNullable<InterceptorUserConfig['actionGuard']> = { ...bg, ...og };
1605
+ if (bg.notify || og.notify) guard.notify = { ...bg.notify, ...og.notify };
1606
+ if (bg.broker || og.broker) guard.broker = { ...bg.broker, ...og.broker };
1607
+ merged.interceptor.actionGuard = guard;
707
1608
  }
708
1609
  }
709
1610
  return merged;
@@ -750,8 +1651,60 @@ function applyPluginConfigOverride(api: PluginApi): void {
750
1651
  _lastShieldConfigRef = null;
751
1652
  }
752
1653
 
1654
+ /**
1655
+ * Load the effective config, DEGRADING rather than throwing (#226).
1656
+ *
1657
+ * `getRuntime()` resolves `shieldcortex/dist/…/runtime.mjs` by walking a list of
1658
+ * install locations, and `loadShieldConfig()` then reads a file. Both can fail
1659
+ * for ordinary reasons — the package was upgraded underneath a running gateway,
1660
+ * a global install moved, `~/.shieldcortex/config.json` is half-written.
1661
+ *
1662
+ * This used to propagate, and the propagation went somewhere bad: every caller
1663
+ * of `loadConfig` is a hook body, and in `handleBeforeAgentRun` the throw was
1664
+ * caught by the OUTER catch — the one that fails open. So a host that could not
1665
+ * load the runtime produced one console line per turn and NOTHING else: no
1666
+ * posture, no scan, no audit row, no alert. Precisely the "unprotected turn that
1667
+ * leaves no trace" the #225/#226 work exists to eliminate, reintroduced through
1668
+ * the config read rather than through the scanner.
1669
+ *
1670
+ * Now it degrades to the plugin config from openclaw.json (already normalised by
1671
+ * `applyPluginConfigOverride`), or to an empty config. The posture therefore
1672
+ * still resolves, the scan still runs, `scanRealtimeContent` reports UNAVAILABLE
1673
+ * on its own (the same runtime failure defeats the MCP fallback), and the gate
1674
+ * writes its normal audit row and raises its normal alert.
1675
+ *
1676
+ * It does NOT cache the degraded result — a later successful load must take
1677
+ * effect without a restart — and it never claims the shield config loaded: the
1678
+ * warning says exactly what is missing, and is bounded, redacted, and emitted
1679
+ * ONCE per plugin load (`__resetConfigStateForTest` re-arms it) so a per-turn
1680
+ * failure cannot become per-turn log spam.
1681
+ */
1682
+ let _shieldConfigLoadFailureLogged = false;
1683
+
753
1684
  async function loadConfig(): Promise<SCConfig> {
754
- const shieldConfigRaw = await (await getRuntime()).loadShieldConfig();
1685
+ let shieldConfigRaw: unknown;
1686
+ try {
1687
+ shieldConfigRaw = await (await getRuntime()).loadShieldConfig();
1688
+ } catch (err) {
1689
+ if (!_shieldConfigLoadFailureLogged) {
1690
+ _shieldConfigLoadFailureLogged = true;
1691
+ const detail = redactNotifyDetail(err instanceof Error ? err.message : String(err)).slice(0, 300);
1692
+ console.warn(
1693
+ '[shieldcortex] ⚠️ shield config could NOT be loaded — the ShieldCortex runtime did not resolve, ' +
1694
+ `or it could not read ~/.shieldcortex/config.json (${detail}). Continuing with the openclaw.json ` +
1695
+ 'plugin config only; anything configured in the shield config file is NOT in effect, and ' +
1696
+ 'conversation scanning will report UNAVAILABLE until this is fixed. (Logged once per plugin load.)',
1697
+ );
1698
+ }
1699
+ // A fresh object every time: `_configOverride` is module state and callers
1700
+ // must not be handed something they could mutate.
1701
+ return mergeConfigs({}, _configOverride ?? {});
1702
+ }
1703
+ // A load that succeeds after a failure re-arms the warning, so a SECOND
1704
+ // outage is reported rather than swallowed by the first one's flag. Set
1705
+ // before the cache check: a runtime that hands back the same object every
1706
+ // call would otherwise take the early return and leave the flag latched.
1707
+ _shieldConfigLoadFailureLogged = false;
755
1708
  if (_config && shieldConfigRaw === _lastShieldConfigRef) return _config;
756
1709
  _lastShieldConfigRef = shieldConfigRaw;
757
1710
  // Plugin config (openclaw.json) deep-merges over the shield config file —
@@ -790,36 +1743,130 @@ function parseScanResponse(response: string): { clean: boolean; summary: string
790
1743
  return { clean, summary };
791
1744
  }
792
1745
 
793
- export async function scanRealtimeContent(text: string): Promise<{ clean: boolean; summary: string }> {
1746
+ /**
1747
+ * Scan one piece of conversation content.
1748
+ *
1749
+ * The contract changed in #225 and the change is the point: an unavailable
1750
+ * scanner is now reported as `available: false, clean: false`, not as the old
1751
+ * `{ clean: true, summary: 'scan unavailable' }`. That old return was the
1752
+ * quietest bug in this file — the ORDINARY unavailable path (MCP fallback
1753
+ * returns nothing, e.g. no shieldcortex binary on PATH) manufactured a clean
1754
+ * verdict, so on any box where in-process defence failed to load, every message
1755
+ * was reported scanned-and-fine while nothing had been looked at.
1756
+ *
1757
+ * Fails OPEN — callers must not block on `available: false` — but LOUDLY: the
1758
+ * caller audits it, alerts on it, and doctor/status report the plane as
1759
+ * unavailable rather than protected.
1760
+ */
1761
+ export async function scanRealtimeContent(text: string): Promise<ConversationScanResult> {
794
1762
  // PRIMARY: scan in-process via the shared shieldcortex/defence module. The
795
1763
  // scan is pure (no DB handle required — scanToolResponse's audit write is
796
1764
  // guarded by isDatabaseInitialized()), so it is safe in the long-lived
797
1765
  // gateway and avoids booting a cold MCP server per message.
798
- const defenceMod = await getDefenceModule();
1766
+ let defenceMod: DefenceModule | null = null;
1767
+ try {
1768
+ defenceMod = await getDefenceModule();
1769
+ } catch (err) {
1770
+ defenceMod = null;
1771
+ void err;
1772
+ }
799
1773
  if (defenceMod && typeof defenceMod.scanToolResponse === "function") {
800
- const scan = defenceMod.scanToolResponse("openclaw-realtime", text, "advisory");
801
- // Reproduce the historical summary contract exactly: risk level + detection
802
- // count only when the injection scan flagged something.
803
- const risk = scan.injection.clean ? "unknown" : scan.injection.riskLevel;
804
- const summary = scan.injection.clean
805
- ? risk
806
- : `${risk} (${scan.injection.detections.length} detections)`;
807
- return { clean: scan.clean, summary };
1774
+ try {
1775
+ const scan = defenceMod.scanToolResponse("openclaw-realtime", text, "advisory");
1776
+ // Reproduce the historical summary contract exactly: risk level + detection
1777
+ // count only when the injection scan flagged something.
1778
+ const risk = scan.injection.clean ? "unknown" : scan.injection.riskLevel;
1779
+ const summary = scan.injection.clean
1780
+ ? risk
1781
+ : `${risk} (${scan.injection.detections.length} detections)`;
1782
+ return { clean: scan.clean, summary, available: true };
1783
+ } catch (err) {
1784
+ // A scanner that THROWS is not a clean verdict either. Same treatment as
1785
+ // an absent one: unavailable, reported, never silently allowed to read as
1786
+ // protected.
1787
+ const detail = err instanceof Error ? err.message : String(err);
1788
+ return { clean: false, available: false, errored: true, error: `in-process scanner threw: ${detail}`, summary: "scan unavailable" };
1789
+ }
808
1790
  }
809
1791
 
810
1792
  // FALLBACK: in-process defence unavailable (older install, import failed) —
811
1793
  // degrade to the MCP shell-out so scanning still happens rather than breaking.
812
- const response = await callCortex("scan_tool_response", {
813
- toolName: "openclaw-realtime",
814
- content: text,
815
- mode: "advisory",
816
- });
1794
+ let response: string | null = null;
1795
+ try {
1796
+ response = await callCortex("scan_tool_response", {
1797
+ toolName: "openclaw-realtime",
1798
+ content: text,
1799
+ mode: "advisory",
1800
+ });
1801
+ } catch (err) {
1802
+ const detail = err instanceof Error ? err.message : String(err);
1803
+ return { clean: false, available: false, errored: true, error: `scan fallback failed: ${detail}`, summary: "scan unavailable" };
1804
+ }
817
1805
 
818
1806
  if (!response) {
819
- return { clean: true, summary: "scan unavailable" };
1807
+ return {
1808
+ clean: false,
1809
+ available: false,
1810
+ errored: true,
1811
+ error: 'no in-process defence module and the MCP fallback returned nothing',
1812
+ summary: 'scan unavailable',
1813
+ };
820
1814
  }
821
1815
 
822
- return parseScanResponse(response);
1816
+ const parsed = parseScanResponse(response);
1817
+ return { ...parsed, available: true };
1818
+ }
1819
+
1820
+ /**
1821
+ * `scanRealtimeContent` with a hard deadline (#226).
1822
+ *
1823
+ * Used ONLY by the `before_agent_run` gate, which the gateway awaits: an
1824
+ * unbounded scan there is an unbounded pause in front of the user's prompt. The
1825
+ * MCP fallback boots a cold server through `npx` and has been measured at ~15s,
1826
+ * so "it usually returns quickly" is not a bound.
1827
+ *
1828
+ * On expiry the result is an ordinary UNAVAILABLE verdict — fail open, audited,
1829
+ * alerted — and the error string names the deadline and NOTHING ELSE. It must
1830
+ * never quote the prompt: a timeout message is the one error a developer is
1831
+ * most likely to paste into an issue.
1832
+ *
1833
+ * The losing promise is not abandoned silently: a `.catch` is attached before
1834
+ * the race so a scan that rejects AFTER the deadline settles into a no-op
1835
+ * instead of an unhandled rejection that could take the gateway down under
1836
+ * `--unhandled-rejections=strict`.
1837
+ */
1838
+ export async function scanWithDeadline(
1839
+ text: string,
1840
+ timeoutMs: number = CONVERSATION_SCAN_MAX_MS,
1841
+ ): Promise<ConversationScanResult> {
1842
+ const timedOut: ConversationScanResult = {
1843
+ clean: false,
1844
+ available: false,
1845
+ errored: true,
1846
+ error: `conversation scan exceeded its ${timeoutMs}ms deadline`,
1847
+ summary: 'scan unavailable',
1848
+ };
1849
+
1850
+ const scan = scanRealtimeContent(text);
1851
+ // Attached BEFORE the race, so a late rejection can never be unhandled.
1852
+ scan.catch(() => { /* the race already answered; nothing left to report */ });
1853
+
1854
+ let timer: ReturnType<typeof setTimeout> | undefined;
1855
+ try {
1856
+ return await Promise.race([
1857
+ scan,
1858
+ new Promise<ConversationScanResult>((resolve) => {
1859
+ timer = setTimeout(() => resolve(timedOut), timeoutMs);
1860
+ // Never hold the process open on the deadline timer alone.
1861
+ (timer as { unref?: () => void }).unref?.();
1862
+ }),
1863
+ ]);
1864
+ } catch (err) {
1865
+ const detail = err instanceof Error ? err.message : String(err);
1866
+ return { clean: false, available: false, errored: true, error: detail, summary: 'scan unavailable' };
1867
+ } finally {
1868
+ if (timer) clearTimeout(timer);
1869
+ }
823
1870
  }
824
1871
 
825
1872
  // ==================== CONTENT PATTERNS ====================
@@ -869,20 +1916,56 @@ function extractUserContent(msgs: unknown[]): string[] {
869
1916
  return out;
870
1917
  }
871
1918
 
872
- const AUDIT_DIR = path.join(homedir(), ".shieldcortex", "audit");
1919
+ /** Where the realtime audit jsonl lives.
1920
+ *
1921
+ * Resolved PER CALL, and honouring `SHIELDCORTEX_AUDIT_DIR`, so a test can
1922
+ * exercise the hook end-to-end without appending fabricated "threat" rows to
1923
+ * a real box's security audit trail. Default is unchanged
1924
+ * (`~/.shieldcortex/audit`), so nothing moves for an install that does not set
1925
+ * the variable. */
1926
+ function auditDir(): string {
1927
+ const override = process.env.SHIELDCORTEX_AUDIT_DIR;
1928
+ if (override && override.trim()) return override.trim();
1929
+ return path.join(homedir(), ".shieldcortex", "audit");
1930
+ }
873
1931
  const NOVELTY_CACHE_FILE = path.join(homedir(), ".shieldcortex", "openclaw-memory-cache.json");
874
1932
  const DEFAULT_NOVELTY_THRESHOLD = 0.88;
875
1933
  const DEFAULT_MAX_RECENT = 300;
876
1934
  const MIN_NOVELTY_CHARS = 40;
877
1935
 
878
- async function auditLog(entry: Record<string, unknown>) {
1936
+ /**
1937
+ * Append one row to the realtime audit jsonl. Returns WHETHER IT LANDED (#226).
1938
+ *
1939
+ * This used to swallow every failure and return void, so a caller could not
1940
+ * distinguish "the evidence is on disk" from "the disk is full / the audit dir
1941
+ * is not writable / it is a file where a directory should be". The
1942
+ * `before_agent_run` gate then wrote a decision row, got nothing back, and
1943
+ * proceeded to tell the operator — and the delivery row — that the decision was
1944
+ * recorded. A security control claiming evidence it does not have is worse than
1945
+ * one that admits the gap, because the gap is invisible in exactly the incident
1946
+ * where the log matters.
1947
+ *
1948
+ * Still never throws: a broken audit sink must not become a broken turn.
1949
+ */
1950
+ async function auditLog(entry: Record<string, unknown>): Promise<boolean> {
1951
+ const dir = auditDir();
879
1952
  try {
880
- await fs.mkdir(AUDIT_DIR, { recursive: true });
1953
+ await fs.mkdir(dir, { recursive: true });
881
1954
  await fs.appendFile(
882
- path.join(AUDIT_DIR, `realtime-${new Date().toISOString().slice(0, 10)}.jsonl`),
1955
+ path.join(dir, `realtime-${new Date().toISOString().slice(0, 10)}.jsonl`),
883
1956
  JSON.stringify(entry) + "\n",
884
1957
  );
885
- } catch {}
1958
+ return true;
1959
+ } catch (err) {
1960
+ // LOUD. The detail is the failure and the directory — never the row, which
1961
+ // may carry a verdict summary, and never a credential (nothing in this path
1962
+ // holds one). Bounded so a pathological error message cannot flood stderr.
1963
+ const detail = err instanceof Error ? err.message : String(err);
1964
+ console.error(
1965
+ `[shieldcortex] ⚠️ AUDIT WRITE FAILED (${dir}) — this event is NOT on disk: ${detail.slice(0, 300)}`,
1966
+ );
1967
+ return false;
1968
+ }
886
1969
  }
887
1970
 
888
1971
  // `cloudSync` lives in ./cloud-sync.ts (no fs imports there) so the plugin
@@ -1052,6 +2135,17 @@ function isInternalContent(text: string): boolean {
1052
2135
  // itself stays non-blocking.
1053
2136
  export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promise<void> {
1054
2137
  try {
2138
+ // #226: THE POSTURE GOVERNS THIS HOOK TOO. `off` means "do not scan the
2139
+ // conversation at all", and this observation hook used to ignore it
2140
+ // entirely: it scanned every prompt, wrote threat and scan_unavailable rows,
2141
+ // and forwarded detections to the cloud on a box whose operator had
2142
+ // explicitly turned conversation inspection OFF. The gate honoured the
2143
+ // setting, so a reader of `handleBeforeAgentRun` would conclude the product
2144
+ // did too. Read it FIRST, before any scanner, any audit row and any cloud
2145
+ // call, so `off` costs exactly one config read.
2146
+ const cfg = await loadConfig();
2147
+ if (conversationPosture(cfg.interceptor?.conversation) === 'off') return;
2148
+
1055
2149
  // Only scan user content, skip system/boot/heartbeat prompts
1056
2150
  // Trust is resolved per TURN, not per message: the host tells us who sent
1057
2151
  // this turn, but history messages carry no individual attribution, so there
@@ -1060,13 +2154,16 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1060
2154
  // Resolved LAZILY, on first detection only. Computing it up front would put
1061
2155
  // a config read on every single turn to answer a question that only matters
1062
2156
  // when something is actually found.
1063
- let trustMemo: ReturnType<typeof classifyConversationOrigin> | null = null;
2157
+ let trustMemo: ConversationTrustDecision | null = null;
1064
2158
  const resolveTrust = async () => {
1065
2159
  if (!trustMemo) {
2160
+ // `cfg` is already in hand from the posture read above, so this costs
2161
+ // no I/O. It also reads the PARSED field rather than casting the config
2162
+ // object — the cast this replaces asserted a key that normaliseConfig
2163
+ // drops, so it was always undefined. See SCConfig.conversationTrust.
1066
2164
  trustMemo = classifyConversationOrigin({
1067
2165
  senderIsOwner: event.senderIsOwner,
1068
- trustOwnerInput: (await loadConfig() as { conversationTrust?: { trustOwnerInput?: boolean } })
1069
- ?.conversationTrust?.trustOwnerInput,
2166
+ trustOwnerInput: cfg.conversationTrust?.trustOwnerInput,
1070
2167
  });
1071
2168
  }
1072
2169
  return trustMemo;
@@ -1076,6 +2173,35 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1076
2173
  for (const text of texts) {
1077
2174
  if (!text || text.length < 10) continue;
1078
2175
  const result = await scanRealtimeContent(text);
2176
+ // #225: "we could not look" is its own outcome. Before this branch the
2177
+ // unavailable path returned clean:true and this loop did nothing at all —
2178
+ // an unscanned message was indistinguishable from a scanned one, on the
2179
+ // observation hook as well as the gate.
2180
+ if (!result.available) {
2181
+ // #226: redacted on the CONSOLE too, not only in the row. The reason
2182
+ // comes from a transport/scanner failure string, which can name the
2183
+ // endpoint it failed to reach — and a gateway's stdout is routinely
2184
+ // shipped to a log aggregator, so "ephemeral" is not a property the
2185
+ // console actually has.
2186
+ const detail = redactNotifyDetail(result.error ?? result.summary);
2187
+ console.warn(
2188
+ `[shieldcortex] ⚠️ conversation scan UNAVAILABLE (${detail}) — this message was NOT scanned`,
2189
+ );
2190
+ // AWAITED. The whole function is already fire-and-forget from
2191
+ // `handleLlmInput`, so this blocks nothing the gateway is waiting on —
2192
+ // and it means the row is on disk before the loop moves to the next
2193
+ // message, and that `auditLog`'s new boolean (which logs loudly on
2194
+ // failure) is actually reached rather than discarded into a floating
2195
+ // promise.
2196
+ await auditLog({
2197
+ type: 'scan_unavailable', hook: 'llm_input', sessionId: event.sessionId,
2198
+ model: event.model, reason: detail,
2199
+ chars: text.length,
2200
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2201
+ ts: new Date().toISOString(),
2202
+ });
2203
+ continue;
2204
+ }
1079
2205
  if (!result.clean) {
1080
2206
  const trust = await resolveTrust();
1081
2207
  console.warn(`[shieldcortex] ⚠️ Threat in LLM input: ${result.summary} [${trust.origin}]`);
@@ -1085,7 +2211,7 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1085
2211
  // the Action Guard tighten by one notch for a bounded window.
1086
2212
  //
1087
2213
  // Source trust gates the CONSEQUENCE, never the detection: the warn and
1088
- // the audit row above happen whoever sent this. What trust decides is
2214
+ // the audit row below happen whoever sent this. What trust decides is
1089
2215
  // whether it may tighten the guard. The operator typing "delete the old
1090
2216
  // logs" is an instruction, and treating it as an attack is the false
1091
2217
  // alarm that gets a control switched off. Everything the agent was
@@ -1093,17 +2219,28 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1093
2219
  if (trust.mayTaint) {
1094
2220
  sessionTaint.mark(event.sessionId, { reason: `conversation scan: ${result.summary}` });
1095
2221
  }
2222
+ // #226: NO `preview`. This row carried the first 100 characters of the
2223
+ // prompt — the exact text that tripped an injection detector, i.e.
2224
+ // hostile by assumption — into an append-only file that syncs. The gate
2225
+ // on the very next hook has recorded only `chars` + `contentSha256`
2226
+ // since #225 and says in its own comment that the prompt is never
2227
+ // persisted; the observation hook quietly did the opposite, so the
2228
+ // claim was false on the path that runs on every single turn. Length
2229
+ // plus digest keeps the row correlatable with the gate's row for the
2230
+ // same text without storing the text.
1096
2231
  const entry = {
1097
2232
  type: "threat", hook: "llm_input", sessionId: event.sessionId,
1098
2233
  model: event.model, reason: result.summary,
1099
- preview: text.slice(0, 100), ts: new Date().toISOString(),
2234
+ chars: text.length,
2235
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2236
+ ts: new Date().toISOString(),
1100
2237
  };
1101
- auditLog(entry);
2238
+ await auditLog(entry);
1102
2239
  loadConfig()
1103
2240
  // Pass the local entry as-is; cloudSync rebuilds a canonical metadata-only
1104
- // entry from named fields and never reads preview/content. No raw LLM input
2241
+ // entry from named fields and never reads content. No raw LLM input
1105
2242
  // leaves here.
1106
- .then(cfg => cloudSync(entry, cfg))
2243
+ .then(cfg2 => cloudSync(entry, cfg2))
1107
2244
  .catch(() => {});
1108
2245
  }
1109
2246
  }
@@ -1113,10 +2250,749 @@ export async function scanLlmInput(event: LlmInputEvent, _ctx: AgentCtx): Promis
1113
2250
  }
1114
2251
 
1115
2252
  function handleLlmInput(event: LlmInputEvent, ctx: AgentCtx): void {
1116
- // Fire and forget
2253
+ // Fire and forget — OBSERVATION ONLY. This hook cannot block (#225); the
2254
+ // enforcement point is handleBeforeAgentRun below.
1117
2255
  void scanLlmInput(event, ctx);
1118
2256
  }
1119
2257
 
2258
+ /**
2259
+ * Route a conversation-threat detection to a HUMAN (#225).
2260
+ *
2261
+ * This is the "sink" the issue is named for. Before it existed, a HIGH verdict
2262
+ * produced a console line and an audit row, and a real detection on a live box
2263
+ * was seen by nobody. It reuses the Action Guard's notify transport (#143)
2264
+ * rather than inventing a second one — two notification paths would drift, and
2265
+ * the operator would learn which one to ignore.
2266
+ *
2267
+ * Returns whether a human was actually reached, so callers (and doctor) can
2268
+ * report "detected but undeliverable" instead of implying someone was told.
2269
+ * Never throws: a failed notification must not become a failed turn.
2270
+ */
2271
+ /** The longest the conversation gate will wait for an operator alert to be
2272
+ * handed to a transport. See the call site: the user's turn is blocked on this
2273
+ * hook, so the alert's deadline has to be a fraction of the hook's. */
2274
+ const CONVERSATION_NOTIFY_MAX_MS = 5_000;
2275
+
2276
+ /**
2277
+ * The longest the conversation gate will wait for a SCAN (#226).
2278
+ *
2279
+ * `before_agent_run` is awaited by the gateway — the user's turn is stopped
2280
+ * dead until this handler returns — and the scan's fallback path is an MCP
2281
+ * shell-out that boots a cold server via `npx`, which can take upwards of 15s.
2282
+ * That is half the hook's entire 30s budget before an alert and two audit
2283
+ * writes are added behind it, and it happens on exactly the hosts where the
2284
+ * in-process defence module failed to load: the ones already degraded.
2285
+ *
2286
+ * Past this deadline the scan is treated as UNAVAILABLE, which fails OPEN (the
2287
+ * turn proceeds) and is audited and alerted like any other unavailable scan. A
2288
+ * security control that silently adds fifteen seconds to every prompt is one an
2289
+ * operator uninstalls.
2290
+ */
2291
+ export const CONVERSATION_SCAN_MAX_MS = 5_000;
2292
+
2293
+ /**
2294
+ * Repeat-alert suppression for the scan-unavailable path (#226).
2295
+ *
2296
+ * An unavailable scanner is not a transient event: it is usually a missing
2297
+ * install, a broken defence build or an absent binary, and it recurs on EVERY
2298
+ * turn. Alerting per turn turns the operator's phone into a metronome and the
2299
+ * alert into something they mute — which is the same outcome as never sending
2300
+ * one, reached by a more expensive route. First occurrence goes immediately;
2301
+ * after that, at most one alert per window, with the suppressed count carried
2302
+ * on the next alert that does go out so nothing is lost.
2303
+ *
2304
+ * THE WINDOW IS PER SESSION, not per process — see `noteScanUnavailable`. A
2305
+ * gateway runs many sessions at once, and "we already told you about session A"
2306
+ * is not a reason to stay silent about session B.
2307
+ *
2308
+ * AUDITING IS NOT RATE LIMITED. Every occurrence still writes its row, and the
2309
+ * row records whether an alert was suppressed and how many have been seen —
2310
+ * the evidence trail must be complete even when the notification stream is not.
2311
+ */
2312
+ export const SCAN_UNAVAILABLE_ALERT_WINDOW_MS = 5 * 60_000;
2313
+
2314
+ interface ScanUnavailableAlertState {
2315
+ /** Total occurrences this session, across suppressed and alerted. */
2316
+ count: number;
2317
+ /** `Date.now()` of the last alert we actually sent, null until the first. */
2318
+ lastAlertAtMs: number | null;
2319
+ /** Occurrences suppressed since that alert. */
2320
+ suppressedSinceAlert: number;
2321
+ /** Last time this session touched the state — eviction key only, never
2322
+ * reported. */
2323
+ lastSeenAtMs: number;
2324
+ }
2325
+
2326
+ /**
2327
+ * The bucket used when the host hands us no session identity at all.
2328
+ *
2329
+ * `before_agent_run`'s context declares `sessionId` and `sessionKey` as
2330
+ * OPTIONAL, so both can be absent. Keying such an occurrence under a fixed
2331
+ * fallback keeps rate limiting working exactly as it did on a single-session
2332
+ * host, and — critically — keeps a nameless occurrence from sharing a bucket
2333
+ * with a NAMED one, which is what a `String(undefined)` key would have done.
2334
+ */
2335
+ const SCAN_UNAVAILABLE_FALLBACK_SESSION = '__unkeyed-session__';
2336
+
2337
+ /**
2338
+ * Cap on distinct sessions tracked at once.
2339
+ *
2340
+ * `session_end` is what normally frees an entry, and it is not guaranteed: an
2341
+ * older host may not emit it, and a crashed session never will. This map is
2342
+ * therefore bounded and evicts least-recently-seen first. Overshooting the cap
2343
+ * costs at most one extra alert for the evicted session — the safe direction,
2344
+ * since the failure mode of eviction is "tell the operator again", not "stay
2345
+ * quiet". Each entry is four numbers and a short key, so 512 of them is a few
2346
+ * kilobytes in a process that already holds a scanner.
2347
+ */
2348
+ export const SCAN_UNAVAILABLE_MAX_SESSIONS = 512;
2349
+
2350
+ const _scanUnavailable = new Map<string, ScanUnavailableAlertState>();
2351
+
2352
+ export interface ScanUnavailableAlertDecision {
2353
+ alert: boolean;
2354
+ /** How many occurrences this session, including this one. */
2355
+ count: number;
2356
+ /** Occurrences suppressed since the last alert. On an ALERT this is the
2357
+ * backlog being reported and then cleared; on a suppression it is the
2358
+ * running total. */
2359
+ suppressedSinceLastAlert: number;
2360
+ }
2361
+
2362
+ /** Normalise whatever the host gave us into a map key. */
2363
+ function scanUnavailableSessionKey(sessionKey?: string | null): string {
2364
+ const trimmed = typeof sessionKey === 'string' ? sessionKey.trim() : '';
2365
+ return trimmed === '' ? SCAN_UNAVAILABLE_FALLBACK_SESSION : trimmed;
2366
+ }
2367
+
2368
+ /**
2369
+ * Should this scan-unavailable occurrence raise an operator alert?
2370
+ *
2371
+ * PER SESSION (#226). The first cut kept one module-global counter, which on a
2372
+ * gateway — a process that multiplexes every channel and every concurrent
2373
+ * agent — meant one session's broken scanner silenced the FIRST failure of
2374
+ * every other session for the next five minutes. That is the same class of bug
2375
+ * the rate limit exists to avoid, inverted: instead of too many alerts, a real
2376
+ * new failure is never reported at all. Suppression is a property of one
2377
+ * session's repeating failure, so the state is keyed by one session.
2378
+ *
2379
+ * Pure apart from the per-session counter it advances, and driven by an
2380
+ * injectable `now` so the window is testable without sleeping. Exported for the
2381
+ * regression test; not part of the plugin's host-facing surface.
2382
+ *
2383
+ * The key is a session id, never logged: it reaches this function only to index
2384
+ * the map. The audit rows that carry `sessionId` are the deliberate place that
2385
+ * fact is recorded.
2386
+ */
2387
+ export function noteScanUnavailable(
2388
+ sessionKey?: string | null,
2389
+ nowMs: number = Date.now(),
2390
+ ): ScanUnavailableAlertDecision {
2391
+ const key = scanUnavailableSessionKey(sessionKey);
2392
+ let state = _scanUnavailable.get(key);
2393
+ if (!state) {
2394
+ evictScanUnavailableOverflow(nowMs);
2395
+ state = { count: 0, lastAlertAtMs: null, suppressedSinceAlert: 0, lastSeenAtMs: nowMs };
2396
+ _scanUnavailable.set(key, state);
2397
+ }
2398
+ state.count += 1;
2399
+ state.lastSeenAtMs = nowMs;
2400
+ const last = state.lastAlertAtMs;
2401
+ // A clock that jumped BACKWARDS (NTP step, suspend/resume) must not be able
2402
+ // to wedge alerting off forever: treat a negative elapsed as "window over".
2403
+ const elapsed = last === null ? Infinity : nowMs - last;
2404
+ if (last === null || elapsed >= SCAN_UNAVAILABLE_ALERT_WINDOW_MS || elapsed < 0) {
2405
+ const suppressed = state.suppressedSinceAlert;
2406
+ state.lastAlertAtMs = nowMs;
2407
+ state.suppressedSinceAlert = 0;
2408
+ return { alert: true, count: state.count, suppressedSinceLastAlert: suppressed };
2409
+ }
2410
+ state.suppressedSinceAlert += 1;
2411
+ return {
2412
+ alert: false,
2413
+ count: state.count,
2414
+ suppressedSinceLastAlert: state.suppressedSinceAlert,
2415
+ };
2416
+ }
2417
+
2418
+ /** Keep the session map bounded when `session_end` never arrives. Evicts the
2419
+ * least-recently-seen entries; an evicted session simply alerts once more. */
2420
+ function evictScanUnavailableOverflow(nowMs: number): void {
2421
+ if (_scanUnavailable.size < SCAN_UNAVAILABLE_MAX_SESSIONS) return;
2422
+ const oldestFirst = [..._scanUnavailable.entries()].sort(
2423
+ (a, b) => (a[1].lastSeenAtMs ?? nowMs) - (b[1].lastSeenAtMs ?? nowMs),
2424
+ );
2425
+ const drop = _scanUnavailable.size - SCAN_UNAVAILABLE_MAX_SESSIONS + 1;
2426
+ for (const [key] of oldestFirst.slice(0, drop)) _scanUnavailable.delete(key);
2427
+ }
2428
+
2429
+ /**
2430
+ * Forget ONE session's suppression window. Called from `session_end`, so a
2431
+ * long-lived gateway does not carry a finished session's suppression into a
2432
+ * reused id — and, just as importantly, does not clear anyone ELSE's.
2433
+ *
2434
+ * A `session_end` that names no session clears the fallback bucket only: on a
2435
+ * host that supplies no session identity every occurrence lands there, so that
2436
+ * is precisely the state that ended.
2437
+ */
2438
+ export function resetScanUnavailableAlertState(sessionKey?: string | null): void {
2439
+ _scanUnavailable.delete(scanUnavailableSessionKey(sessionKey));
2440
+ }
2441
+
2442
+ /** Test/reset seam: forget EVERY session. Production never wants this — one
2443
+ * session ending must not re-arm alerting for the others — so it is reachable
2444
+ * only from `__resetConfigStateForTest`. */
2445
+ export function __resetScanUnavailableAlertState(): void {
2446
+ _scanUnavailable.clear();
2447
+ }
2448
+
2449
+ /** What actually happened to an alert, so callers can report it truthfully.
2450
+ * `configured: false` is the honest "nobody opted in" case and is NOT a
2451
+ * failure — it is the #143 default-off contract. */
2452
+ export interface NotifyOutcome {
2453
+ configured: boolean;
2454
+ delivered: boolean;
2455
+ via: string | null;
2456
+ detail: string;
2457
+ }
2458
+
2459
+ /**
2460
+ * Make a failure detail safe to PERSIST, SEND or PRINT (#226).
2461
+ *
2462
+ * The detail is assembled from channel names, scanner errors and whatever a
2463
+ * transport said went wrong. Node's fetch failures do not name the URL, but a
2464
+ * transport is free to put one in its reason — and a notify webhook URL
2465
+ * routinely carries a token in its path or query
2466
+ * (`https://hooks.example/services/T0/B0/XXXXXXXX`). So any http(s) URL is
2467
+ * reduced to its origin: enough to tell WHICH endpoint failed, not enough to
2468
+ * replay a request to it. Bounded too, so a transport that returns a page of
2469
+ * HTML cannot bloat the log.
2470
+ *
2471
+ * EVERY sink gets the redacted string — not just the ones that obviously
2472
+ * outlive the process. The audit row is append-only and syncs; the notification
2473
+ * leaves the box; and the console is NOT the ephemeral thing an earlier version
2474
+ * of this comment claimed it was, because a gateway's stdout is routinely
2475
+ * shipped to a log aggregator and kept longer than the audit file. Redacting
2476
+ * for the row and not for the other two protected the least exposed of the
2477
+ * three.
2478
+ */
2479
+ export function redactNotifyDetail(detail: string): string {
2480
+ const withoutUrls = String(detail ?? '').replace(
2481
+ /https?:\/\/[^\s'"]+/gi,
2482
+ (url) => {
2483
+ try {
2484
+ return `${new URL(url).origin}/…`;
2485
+ } catch {
2486
+ return '<url>';
2487
+ }
2488
+ },
2489
+ );
2490
+ return withoutUrls.length > 500 ? `${withoutUrls.slice(0, 499)}…` : withoutUrls;
2491
+ }
2492
+
2493
+ /**
2494
+ * The seam a gateway MIGHT offer for sending an operator a message, captured at
2495
+ * register() time if the API exposes it.
2496
+ *
2497
+ * No OpenClaw build we have inspected exposes it — neither 2026.5.2 nor
2498
+ * 2026.7.1 has a `notifyOperator` anywhere in its plugin API — so in practice
2499
+ * the webhook is the load-bearing channel and this stays null. It is read
2500
+ * structurally rather than removed because #143's design intent was that on
2501
+ * OpenClaw the transport should use the gateway's own message capability, and
2502
+ * that only becomes true if the code is ready for the day it appears. Nothing
2503
+ * here should be read as "ShieldCortex delivers natively on OpenClaw today".
2504
+ */
2505
+ let _gatewayNotifyContext: GatewayNotifyContext | null = null;
2506
+ export function __setGatewayNotifyContextForTest(ctx: GatewayNotifyContext | null): void {
2507
+ _gatewayNotifyContext = ctx;
2508
+ }
2509
+
2510
+ /**
2511
+ * Route a conversation-firewall detection to a HUMAN (#225).
2512
+ *
2513
+ * This is the "sink" the issue is named for: before it existed, a HIGH verdict
2514
+ * produced a console line and an audit row, and a real detection on a live box
2515
+ * was seen by nobody.
2516
+ *
2517
+ * It reuses the Action Guard's #143 transport rather than inventing a second
2518
+ * one — but *correctly*, which the first cut did not:
2519
+ *
2520
+ * - the notification is built by the main package's
2521
+ * `buildConversationThreatNotification`, so it is a real, bounded
2522
+ * notification with its own event discriminator, NOT an ad-hoc
2523
+ * `{kind, severity, …}` literal cast through `NotifyChannel.send`. A
2524
+ * conversation alert therefore cannot render Approve/Deny controls or a
2525
+ * hash that does not exist — the fields simply are not on the type.
2526
+ * - delivery goes through `deliverOperatorNotification`, the same core the
2527
+ * approval path uses, so the deadline, the malformed-result handling and
2528
+ * the "nothing but the boolean is read back" rule are shared, not copied.
2529
+ * - the webhook secret is read from `webhookSecret` — the field
2530
+ * `normaliseNotifyConfig` actually returns. Mirroring it as `secret`
2531
+ * silently produced UNSIGNED POSTs.
2532
+ * - both channels are offered where the runtime provides them: the gateway's
2533
+ * own message seam first WHERE IT EXISTS (no build we have inspected
2534
+ * exposes one — see `_gatewayNotifyContext`), then the configured webhook,
2535
+ * which is what actually carries an alert off the box today.
2536
+ *
2537
+ * Returns what happened, and NEVER throws: a failed notification must not
2538
+ * become a failed turn.
2539
+ */
2540
+ export async function notifyOperatorOfConversationThreat(input: {
2541
+ outcome: 'blocked' | 'observed' | 'unavailable';
2542
+ posture: ConversationPosture;
2543
+ summary: string;
2544
+ reason: string;
2545
+ sessionId?: string;
2546
+ model?: string;
2547
+ }): Promise<NotifyOutcome> {
2548
+ try {
2549
+ const mod = await getDefenceModule();
2550
+ const cfg = await loadConfig();
2551
+ const raw = cfg.interceptor?.actionGuard?.notify;
2552
+ if (!raw) return { configured: false, delivered: false, via: null, detail: 'no notify config' };
2553
+ if (typeof mod?.normaliseNotifyConfig !== 'function') {
2554
+ return { configured: true, delivered: false, via: null, detail: 'installed shieldcortex build has no notify transport' };
2555
+ }
2556
+ const notify = mod.normaliseNotifyConfig(raw);
2557
+ if (!notify.enabled) return { configured: false, delivered: false, via: null, detail: 'notify disabled' };
2558
+
2559
+ const channels: NotifyChannelLike[] = [];
2560
+ // The gateway's own message seam, WHERE the runtime provides one. It would
2561
+ // go first, because it would reach the operator on a channel they already
2562
+ // read — but `_gatewayNotifyContext` is null on every build we have
2563
+ // inspected, so in practice this list starts at the webhook below.
2564
+ if (notify.openclaw === true && _gatewayNotifyContext) {
2565
+ const gatewayChannel = createGatewayNotifyChannel(_gatewayNotifyContext);
2566
+ if (gatewayChannel) channels.push(gatewayChannel);
2567
+ }
2568
+ if (notify.webhookUrl && typeof mod.createWebhookNotifyChannel === 'function') {
2569
+ channels.push(
2570
+ mod.createWebhookNotifyChannel({
2571
+ url: notify.webhookUrl,
2572
+ // The signing key. Passed straight through and never logged — see
2573
+ // notify-config.ts, which is the only place this value is parsed.
2574
+ secret: notify.webhookSecret,
2575
+ }),
2576
+ );
2577
+ }
2578
+ if (channels.length === 0) {
2579
+ return { configured: true, delivered: false, via: null, detail: 'notify enabled but no channel is configured/buildable on this host' };
2580
+ }
2581
+
2582
+ const notification =
2583
+ typeof mod.buildConversationThreatNotification === 'function'
2584
+ ? mod.buildConversationThreatNotification({
2585
+ outcome: input.outcome,
2586
+ posture: input.posture,
2587
+ summary: input.summary,
2588
+ reason: input.reason,
2589
+ sessionId: input.sessionId,
2590
+ model: input.model,
2591
+ host: hostname(),
2592
+ detectedAt: new Date().toISOString(),
2593
+ })
2594
+ : null;
2595
+ if (!notification) {
2596
+ // An older dist has the transport but not this event. Sending the
2597
+ // approval-shaped payload instead would put an Approve button on an alert
2598
+ // with nothing behind it — refuse, and say why.
2599
+ return {
2600
+ configured: true,
2601
+ delivered: false,
2602
+ via: null,
2603
+ detail: 'installed shieldcortex build predates the conversation-threat notification — refusing to send an approval-shaped alert',
2604
+ };
2605
+ }
2606
+
2607
+ if (typeof mod.deliverOperatorNotification !== 'function') {
2608
+ return { configured: true, delivered: false, via: null, detail: 'installed shieldcortex build has no notification delivery core' };
2609
+ }
2610
+ const result = await mod.deliverOperatorNotification(notification, {
2611
+ channels,
2612
+ // Bounded HARDER than the transport's own configured deadline, because
2613
+ // this call sits inside a gate the gateway awaits: the user's turn is
2614
+ // waiting on it. The hook is registered with a 30s timeout, and a gate
2615
+ // that exceeds its own timeout is a security control that fails in a way
2616
+ // nobody has reasoned about. The alert is already in the log and the
2617
+ // audit row by this point, so what a longer wait buys is one extra retry
2618
+ // window on a transport that is, by then, visibly unhealthy.
2619
+ timeoutMs: Math.min(notify.timeoutMs ?? CONVERSATION_NOTIFY_MAX_MS, CONVERSATION_NOTIFY_MAX_MS),
2620
+ });
2621
+ const failures = result.attempts
2622
+ .filter((a) => !a.result.delivered)
2623
+ .map((a) => `${a.channel}: ${a.result.reason ?? 'failed'}`)
2624
+ .join('; ');
2625
+ return {
2626
+ configured: true,
2627
+ delivered: result.deliveredVia !== null,
2628
+ via: result.deliveredVia,
2629
+ detail: result.deliveredVia ? `delivered via ${result.deliveredVia}` : `undeliverable — ${failures || 'no channel accepted it'}`,
2630
+ };
2631
+ } catch (err) {
2632
+ return {
2633
+ configured: true,
2634
+ delivered: false,
2635
+ via: null,
2636
+ detail: `notify error: ${err instanceof Error ? err.message : String(err)}`,
2637
+ };
2638
+ }
2639
+ }
2640
+
2641
+ /**
2642
+ * The event payload OpenClaw hands `before_agent_run`, as declared by the host
2643
+ * SDK (`PluginHookBeforeAgentRunEvent`, hook-types.d.ts). Note what is NOT on
2644
+ * it: `sessionId` and `model` live on the CONTEXT, not the event — reading them
2645
+ * off the event, as the first cut did, produced `undefined` in every audit row
2646
+ * and every alert.
2647
+ */
2648
+ type BeforeAgentRunEvent = {
2649
+ prompt?: string;
2650
+ messages?: unknown[];
2651
+ systemPrompt?: string;
2652
+ accountId?: string;
2653
+ channelId?: string;
2654
+ senderId?: string;
2655
+ senderIsOwner?: boolean;
2656
+ };
2657
+
2658
+ /**
2659
+ * The gate's return contract, verbatim from the host SDK:
2660
+ *
2661
+ * type HookDecisionPass = { outcome: "pass" }
2662
+ * type HookDecisionBlock = { outcome: "block"; reason: string; message?: string;
2663
+ * category?: string; metadata?: Record<string, unknown> }
2664
+ * type PluginHookBeforeAgentRunResult = InputGateDecision | void
2665
+ *
2666
+ * `{ block: true }` — what the first cut returned — is the `before_tool_call`
2667
+ * shape, and this gate does not understand it: it is neither "pass" nor
2668
+ * "block", so the run would have proceeded while our audit row said BLOCKED.
2669
+ * A firewall whose block is a no-op is worse than no firewall, because the
2670
+ * evidence says it worked.
2671
+ *
2672
+ * `reason` is documented as INTERNAL ("core must not log, persist, broadcast,
2673
+ * or expose it verbatim"); `message` is the user-facing half. We keep the
2674
+ * verdict summary in both — it names the risk level and the detection count,
2675
+ * never the offending text.
2676
+ *
2677
+ * SHAPE IS EXACT, and the host enforces it structurally (`isHookDecision`,
2678
+ * hook-runner-global, 2026.7.1-2): a pass must be `{ outcome: 'pass' }` and
2679
+ * NOTHING else — the guard is `keys.length === 1`. Adding so much as a
2680
+ * `metadata` field to the pass branch for debugging would make it "not a hook
2681
+ * decision", and the runner's answer to that is not to ignore it: it is
2682
+ * `{ outcome: 'block', reason: 'before_agent_run returned an invalid
2683
+ * decision' }`. The failure mode of a malformed ALLOW here is a BLOCKED turn.
2684
+ */
2685
+ export type InputGateDecision =
2686
+ | { outcome: 'pass' }
2687
+ | { outcome: 'block'; reason: string; message?: string; category?: string; metadata?: Record<string, unknown> };
2688
+
2689
+ /**
2690
+ * The gate's allow answer, stated explicitly (#226).
2691
+ *
2692
+ * A fresh literal per call, not a shared constant: the runner passes whatever
2693
+ * we return into its own merge/normalise chain, and a frozen singleton handed
2694
+ * to a host that decides to annotate it would fail in a way this plugin cannot
2695
+ * see. It costs one object per turn.
2696
+ *
2697
+ * WHY EXPLICIT, when the SDK types the result `InputGateDecision | void` and
2698
+ * this handler previously returned `undefined` on every allow path:
2699
+ *
2700
+ * The 2026.7.1-2 runner contradicts itself about void, one guard deep.
2701
+ * `runBeforeAgentRun`'s doc comment says "Handlers that return void are treated
2702
+ * as pass", and its `mergeResults` body opens with
2703
+ *
2704
+ * if (next === void 0 || next === null) → { outcome: "block",
2705
+ * reason: "…invalid decision" }
2706
+ *
2707
+ * i.e. the merge is written to BLOCK on void. What saves an `undefined` return
2708
+ * today is only that `runModifyingHook` never calls the merge for it —
2709
+ * `if (handlerResult !== void 0 && (handlerResult !== null || mergeNullResults))`
2710
+ * — so the void branch inside the merge is unreachable dead code, while `null`,
2711
+ * the sibling value that same line treats identically, reaches it and DOES
2712
+ * block. Verified by executing the real 2026.7.1-2 runner: `undefined` → pass,
2713
+ * `null` → block/invalid, `{outcome:'pass'}` → pass.
2714
+ *
2715
+ * So void is not broken here — it is correct by one guard, against a merge
2716
+ * function whose stated intent is to reject it. `{ outcome: 'pass' }` is
2717
+ * correct under BOTH readings, and is the shape the host validates rather than
2718
+ * the shape it happens to skip. That is the difference worth having in front of
2719
+ * every user turn.
2720
+ */
2721
+ function gatePass(): InputGateDecision {
2722
+ return { outcome: 'pass' };
2723
+ }
2724
+
2725
+ /**
2726
+ * The conversation firewall's enforcement point (#225).
2727
+ *
2728
+ * Unlike `llm_input`, this hook is awaited by the gateway and its return value
2729
+ * decides whether the run proceeds. It scans the prompt, applies the configured
2730
+ * posture, and — critically — routes a detection to a HUMAN rather than only to
2731
+ * a log file. The finding this fixes was that a HIGH verdict on a live box was
2732
+ * seen by nobody.
2733
+ *
2734
+ * Fails OPEN on any internal error: a security plugin that bricks the gateway
2735
+ * has caused a worse outage than the one it prevents. Every failure is reported.
2736
+ *
2737
+ * EVERY path returns a decision — `gatePass()` to allow, `{ outcome: 'block' }`
2738
+ * only for a dirty verdict under `enforce`. Nothing returns `undefined`; see
2739
+ * `gatePass` for the host-contract reason. "Fails open" therefore now means an
2740
+ * explicit pass, which is a stronger statement than the absence of an answer:
2741
+ * it is the same word said in the vocabulary the host validates.
2742
+ */
2743
+ export async function handleBeforeAgentRun(
2744
+ event: BeforeAgentRunEvent,
2745
+ ctx: AgentCtx,
2746
+ ): Promise<InputGateDecision> {
2747
+ let posture: ConversationPosture = 'observe';
2748
+ try {
2749
+ const cfg = await loadConfig();
2750
+ posture = conversationPosture(cfg.interceptor?.conversation);
2751
+ if (posture === 'off') return gatePass();
2752
+
2753
+ const text = String(event?.prompt ?? '');
2754
+ if (!text || text.length < 10 || isInternalContent(text)) return gatePass();
2755
+
2756
+ // sessionId/model come off the hook CONTEXT (PluginHookAgentContext); the
2757
+ // event carries neither. Both are optional there too, so both may be absent.
2758
+ const sessionId = ctx?.sessionId ?? ctx?.sessionKey;
2759
+ const model = (ctx as { modelId?: string } | undefined)?.modelId;
2760
+
2761
+ // scanRealtimeContent no longer throws on the paths that used to (it
2762
+ // reports `available:false` instead), but a defensive catch stays: this
2763
+ // function's contract is that nothing here can stop a turn by accident.
2764
+ // #226: BOUNDED. The gateway awaits this hook, so an unbounded scan is an
2765
+ // unbounded pause in front of the user's prompt — see scanWithDeadline.
2766
+ let scan: ConversationScanResult;
2767
+ try {
2768
+ scan = await scanWithDeadline(text);
2769
+ } catch (err) {
2770
+ const detail = err instanceof Error ? err.message : String(err);
2771
+ scan = { clean: false, available: false, errored: true, error: detail, summary: 'scan unavailable' };
2772
+ }
2773
+
2774
+ // #235: WHO sent this turn, resolved before the verdict is applied.
2775
+ // `senderIsOwner` was declared on the event and read by nothing, so the
2776
+ // enforce path could block the operator's own paste and destroy it. The
2777
+ // config is already loaded, so this is a pure call — no second read.
2778
+ const trust = classifyConversationOrigin({
2779
+ senderIsOwner: event?.senderIsOwner,
2780
+ trustOwnerInput: cfg.conversationTrust?.trustOwnerInput,
2781
+ });
2782
+
2783
+ const decision = evaluateConversationRun(posture, scan, trust);
2784
+
2785
+ // #226: REDACT ONCE, then use the redacted string everywhere the reason
2786
+ // goes — the persisted decision row, the outbound notification, the block
2787
+ // reason, the console line. On the unavailable path `decision.reason`
2788
+ // embeds the scanner's own failure string verbatim
2789
+ // (`conversation scan unavailable (${scan.error})`), and that string is
2790
+ // assembled from a transport error: a cold MCP start, a fetch, a defence
2791
+ // build download. Any of those can name the endpoint it failed to reach,
2792
+ // and such a URL routinely carries a credential in its path
2793
+ // (`https://hooks.example/services/T0/B0/XXXX`). The console line was
2794
+ // already redacted while the row and the alert — the two that PERSIST and
2795
+ // LEAVE THE BOX — were not, which had the guarantee exactly backwards.
2796
+ const safeReason = decision.reason === null ? null : redactNotifyDetail(decision.reason);
2797
+
2798
+ // #226: repeated unavailability alerts at most once per window. Called here
2799
+ // rather than at the notify site so the COUNTERS advance on every
2800
+ // occurrence, and the decision row can record what was suppressed even when
2801
+ // no alert goes out.
2802
+ // Keyed by SESSION: a broken scanner in one session must not silence the
2803
+ // first report of a broken scanner in another. `sessionId` may be absent —
2804
+ // noteScanUnavailable buckets that case separately rather than letting one
2805
+ // nameless session stand in for all of them.
2806
+ const unavailable = decision.outcome === 'unavailable';
2807
+ const alertGate = unavailable ? noteScanUnavailable(sessionId) : null;
2808
+ const suppressAlert = alertGate !== null && !alertGate.alert;
2809
+
2810
+ // ── EVIDENCE FIRST, SIDE EFFECT SECOND ────────────────────────────────
2811
+ //
2812
+ // The decision row is a LOCAL append and it goes to disk before anything
2813
+ // leaves this box. The previous order awaited an external notification and
2814
+ // then wrote the row — under a comment claiming the row already existed —
2815
+ // so every way that call can end badly took the evidence with it: a
2816
+ // notification channel that hangs until the gateway's 30s hook timeout
2817
+ // fires, a transport that throws past its own catch, an operator restarting
2818
+ // the gateway mid-alert, the process dying. In each case the block or the
2819
+ // detection HAPPENED and there is no record that it did. That inverts the
2820
+ // whole point: a security control's own log must not be contingent on an
2821
+ // unrelated network round trip succeeding.
2822
+ //
2823
+ // The row carries a stable `eventId`, so the delivery row appended after
2824
+ // the attempt below can be joined to it without either row having to
2825
+ // predict the other's outcome.
2826
+ const eventId = randomUUID();
2827
+ // #226: whether the decision row ACTUALLY LANDED. `auditLog` used to
2828
+ // swallow its failures and return void, so the code below could not tell an
2829
+ // append from a silent no-op and every downstream statement — the operator
2830
+ // alert, the delivery row, this function's own comments — asserted that
2831
+ // evidence existed. Now the boolean is carried, said out loud on stderr,
2832
+ // and attached to the alert as a bounded, secret-free fact.
2833
+ let decisionRowPersisted = true;
2834
+ if (decision.audit) {
2835
+ // AWAITED: the row that says what was decided must exist before the
2836
+ // decision is handed back, and its success or failure must be READ. The
2837
+ // write is a bounded local append wrapped in its own try/catch.
2838
+ decisionRowPersisted = await auditLog({
2839
+ type: decision.outcome === 'unavailable' ? 'scan_unavailable' : 'threat',
2840
+ hook: 'before_agent_run',
2841
+ eventId,
2842
+ sessionId,
2843
+ model,
2844
+ // The REDACTED reason. This row is appended to a file that syncs.
2845
+ reason: safeReason,
2846
+ posture,
2847
+ outcome: decision.outcome,
2848
+ // #235: the origin, on every conversation decision row. Without it an
2849
+ // operator auditing an `enforce` host cannot tell a turn that was not
2850
+ // blocked because it was clean from one that was not blocked because
2851
+ // the owner sent it — and "why did this not block?" is the question
2852
+ // this row exists to answer. A label ('owner'/'non-owner'/'unknown'),
2853
+ // never a sender id: the row syncs.
2854
+ origin: trust.origin,
2855
+ // The verdict summary, never the prompt. The input that trips an
2856
+ // injection detector is hostile text by assumption; copying it into an
2857
+ // audit row that syncs to the dashboard/cloud would carry the payload
2858
+ // one hop further. A length + digest keeps rows correlatable without
2859
+ // storing the content.
2860
+ verdict: scan.summary,
2861
+ chars: text.length,
2862
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2863
+ // Deliberately NOT `notified: false`. Nothing has been attempted yet,
2864
+ // and a false here would read as "we tried and failed". The attempt's
2865
+ // result is its own row, keyed by this eventId.
2866
+ notifyPending: decision.notify && !suppressAlert,
2867
+ // #226: the unavailability run-length, on EVERY occurrence. Alerting is
2868
+ // rate limited; auditing is not, so the row is where the true count
2869
+ // lives — and it says explicitly when an alert was withheld, so a gap in
2870
+ // the alert stream can never be mistaken for a gap in the failures.
2871
+ ...(alertGate
2872
+ ? {
2873
+ unavailableCount: alertGate.count,
2874
+ alertSuppressed: suppressAlert,
2875
+ alertSuppressedSinceLastAlert: alertGate.suppressedSinceLastAlert,
2876
+ }
2877
+ : {}),
2878
+ ts: new Date().toISOString(),
2879
+ });
2880
+ if (!decisionRowPersisted) {
2881
+ console.error(
2882
+ `[shieldcortex] ⚠️ conversation ${decision.outcome} decision could NOT be written to the audit log — ` +
2883
+ 'the decision itself still stands, but there is no local record of it. Check the audit directory ' +
2884
+ '(SHIELDCORTEX_AUDIT_DIR or ~/.shieldcortex/audit) for permissions or disk space.',
2885
+ );
2886
+ }
2887
+ }
2888
+
2889
+ // The sink. Awaited — the first cut fired this off with `void` and threw
2890
+ // the delivery boolean away, so the code could not tell "a human was told"
2891
+ // from "nothing left this box". It is bounded (CONVERSATION_NOTIFY_MAX_MS,
2892
+ // well under the hook's own 30s timeout) and never throws.
2893
+ let notifyResult: NotifyOutcome | null = null;
2894
+ if (decision.notify) {
2895
+ const label = decision.outcome === 'unavailable' ? 'unavailable' : decision.block ? 'blocked' : 'observed';
2896
+ // The same redacted string the row got. A gateway's stdout is routinely
2897
+ // shipped to a log aggregator, so "ephemeral" is not a property the
2898
+ // console actually has either.
2899
+ console.warn(
2900
+ `[shieldcortex] ⚠️ ${safeReason ?? 'conversation threat'} — posture=${posture}, outcome=${label}` +
2901
+ (suppressAlert
2902
+ ? ` (operator alert SUPPRESSED — ${alertGate?.suppressedSinceLastAlert} since the last one; ${alertGate?.count} this session)`
2903
+ : ''),
2904
+ );
2905
+ // Audited above, not alerted. Nothing further is written for a suppressed
2906
+ // occurrence: no attempt was made, and a delivery row saying
2907
+ // `delivered: false` would read as a transport failure that never
2908
+ // happened. The decision itself is unaffected — suppression governs who
2909
+ // is TOLD, never what is DECIDED.
2910
+ if (!suppressAlert) {
2911
+ // The audit-persistence fact rides ALONG with the alert when the local
2912
+ // record failed: bounded, no secrets, and it tells the operator that
2913
+ // this notification is the only trace of the event. Appended to the
2914
+ // reason rather than added as a field so it survives an older installed
2915
+ // dist whose notification builder does not know about it.
2916
+ const auditNote = decisionRowPersisted ? '' : ' [auditPersistence=failed: no local audit row for this event]';
2917
+ const suppressedNote =
2918
+ alertGate && alertGate.suppressedSinceLastAlert > 0
2919
+ ? ` [${alertGate.suppressedSinceLastAlert} further scan-unavailable event(s) suppressed since the last alert; ${alertGate.count} this session]`
2920
+ : '';
2921
+ notifyResult = await notifyOperatorOfConversationThreat({
2922
+ outcome: label,
2923
+ posture,
2924
+ summary: scan.summary,
2925
+ // REDACTED. This one leaves the box entirely — to a webhook, an
2926
+ // aggregator, a phone — so it is the last place a tokenised endpoint
2927
+ // URL lifted out of a scanner error may appear.
2928
+ reason: `${safeReason ?? 'conversation threat'}${suppressedNote}${auditNote}`,
2929
+ sessionId,
2930
+ model,
2931
+ });
2932
+ // Truthful reporting: never imply a human was reached unless a
2933
+ // transport said so. "Not configured" is not a failure — it is the #143
2934
+ // default.
2935
+ if (notifyResult.configured && !notifyResult.delivered) {
2936
+ // #226: redacted HERE too, not only on the row below. The same detail
2937
+ // string reaches both, and a gateway's stdout is routinely shipped
2938
+ // somewhere it outlives the process.
2939
+ console.warn(`[shieldcortex] ⚠️ conversation alert UNDELIVERED — ${redactNotifyDetail(notifyResult.detail)}`);
2940
+ }
2941
+ // Guarded on `audit` because `eventId` has to point at something:
2942
+ // `evaluateConversationRun` never sets notify without audit, and if that
2943
+ // ever changed, a delivery row keyed to a decision row that was never
2944
+ // written would be a dangling reference rather than evidence.
2945
+ if (decision.audit) {
2946
+ // A SECOND row, not a rewrite of the first. The audit sink is an
2947
+ // append-only JSONL file, so "what was decided" and "who was told" are
2948
+ // separate facts recorded when each became true, joined by eventId.
2949
+ // `via` is the channel NAME ('webhook', 'openclaw-gateway'), never a
2950
+ // URL; the detail is redacted before it is persisted.
2951
+ await auditLog({
2952
+ type: 'notification_delivery',
2953
+ hook: 'before_agent_run',
2954
+ eventId,
2955
+ sessionId,
2956
+ configured: notifyResult.configured,
2957
+ delivered: notifyResult.delivered,
2958
+ via: notifyResult.via,
2959
+ detail: redactNotifyDetail(notifyResult.detail),
2960
+ // The eventId this row joins on may point at a row that was never
2961
+ // written. Say so here rather than leave a dangling reference that
2962
+ // reads as a missing file rather than a failed write.
2963
+ ...(decisionRowPersisted ? {} : { auditPersistence: 'failed' }),
2964
+ ts: new Date().toISOString(),
2965
+ });
2966
+ }
2967
+ }
2968
+ }
2969
+
2970
+ // Clean, observed-not-blocked, and scan-unavailable all land here. The
2971
+ // audit row and the operator alert above have already recorded what
2972
+ // happened; the run itself proceeds, and says so.
2973
+ if (!decision.block) return gatePass();
2974
+ return {
2975
+ outcome: 'block',
2976
+ // Redacted for the same reason as the row above: the host SDK documents
2977
+ // `reason` as internal, but "internal" is a policy, not a guarantee.
2978
+ reason: safeReason ?? 'conversation threat',
2979
+ message: `ShieldCortex blocked this turn: ${scan.summary}. The prompt was not sent to the model.`,
2980
+ category: 'prompt_injection',
2981
+ };
2982
+ } catch (e) {
2983
+ // Fail open, loudly. Never let the guard's own failure stop the agent.
2984
+ //
2985
+ // This catch is also why the handler must not be allowed to THROW: the host
2986
+ // registers `before_agent_run` as fail-CLOSED
2987
+ // (`failurePolicyByHook: { before_agent_run: 'fail-closed' }`), so an
2988
+ // exception escaping here does not fail open at all — the gateway catches it
2989
+ // and blocks the run with "before_agent_run hook failed". An explicit pass
2990
+ // is the only way this function actually keeps its fail-open promise.
2991
+ console.error('[shieldcortex] before_agent_run error (failing open):', e instanceof Error ? e.message : String(e));
2992
+ return gatePass();
2993
+ }
2994
+ }
2995
+
1120
2996
  // Skip text blocks that are ShieldCortex/OpenClaw tool-result pass-throughs
1121
2997
  function isToolResultContent(text: string): boolean {
1122
2998
  // ShieldCortex recall returns "Found N memories:" header
@@ -1314,6 +3190,13 @@ export default {
1314
3190
  if (_registered) return;
1315
3191
  _registered = true;
1316
3192
 
3193
+ // #226: the host runtime's own version, before anything can throw. It is
3194
+ // the primary version evidence for the conversation-gate check — the
3195
+ // gateway stating its own version beats inferring one from whichever
3196
+ // package.json sits above the entry path. Absent on a host that does not
3197
+ // expose it, which stays UNKNOWN rather than becoming a guess.
3198
+ recordHostRuntimeVersion(api);
3199
+
1317
3200
  // --- Interceptor (lazy init) ---
1318
3201
  let interceptorReady: ReturnType<typeof createInterceptor> | null = null;
1319
3202
  let interceptorInitAttempted = false;
@@ -1362,12 +3245,44 @@ export default {
1362
3245
  : `${guardCfg.enforce ? "enforce" : "warn"}${autoApproved > 0 ? ` (${autoApproved} auto-approved)` : ""}${interceptorReady ? "" : " — not yet initialised this session"}`;
1363
3246
  const hooksLine = _beforeToolCallRegistered
1364
3247
  ? "llm_input (scan), llm_output (memory), before_tool_call (action guard), session_end (cache reset)"
1365
- : "llm_input (scan), llm_output (memory)";
3248
+ // #226: session_end is registered even with the interceptor off —
3249
+ // the conversation gate keeps per-session state that needs freeing.
3250
+ : "llm_input (scan), llm_output (memory), session_end (cache reset)";
3251
+ // #225: the conversation plane, stated as evidence rather than as a
3252
+ // tick. Every clause below is something this process actually knows:
3253
+ // the configured posture, that we asked for the hook, the host build,
3254
+ // and the operator's grant. Nothing here claims the gateway accepted
3255
+ // the registration, because the plugin API never says so.
3256
+ const hostProbe = detectHostOpenClaw();
3257
+ const plane = describeConversationPlane({
3258
+ posture: conversationPosture(cfg.interceptor?.conversation),
3259
+ hookRequested: _beforeAgentRunRequested,
3260
+ gateSupport: hostSupportsConversationGate(hostProbe),
3261
+ hostOpenClawVersion: hostProbe.version,
3262
+ consentGranted: _conversationAccessGranted,
3263
+ });
3264
+ const notifyRaw = cfg.interceptor?.actionGuard?.notify;
3265
+ const notifyState = notifyRaw && (notifyRaw as { enabled?: unknown }).enabled === true
3266
+ ? 'configured'
3267
+ : 'not configured — detections reach the audit log and this box only';
1366
3268
  return {
1367
3269
  text:
1368
3270
  `ShieldCortex v${_version}\n` +
1369
- ` Hooks: ${hooksLine}\n` +
3271
+ ` Hooks: ${hooksLine}${_beforeAgentRunRequested ? ', before_agent_run (conversation gate, requested)' : ''}\n` +
1370
3272
  ` Action guard: ${guardState}\n` +
3273
+ ` Conversation firewall: ${plane.summary}\n` +
3274
+ // #226: state the PROVENANCE, not just the value. This flag is a
3275
+ // SNAPSHOT taken once, when the plugin loaded — the host reads
3276
+ // the grant at hook-registration time and this process never
3277
+ // re-reads it. So an operator who has just edited openclaw.json
3278
+ // and re-run the command sees the old answer, correctly, and
3279
+ // would otherwise conclude the grant does not work. Nothing here
3280
+ // is live: changing it requires a gateway restart before either
3281
+ // the gateway or this line reflects it.
3282
+ ` Conversation access grant: ${_conversationAccessGranted ? 'granted' : 'NOT granted'} (plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess)\n` +
3283
+ ' — read from openclaw.json when this plugin LOADED; it is a snapshot, not a live read.\n' +
3284
+ ' Editing that key takes effect only after a gateway restart, for the gateway and for this line.\n' +
3285
+ ` Operator notify: ${notifyState}\n` +
1371
3286
  ` Auto memory: ${autoMemory} | Dedupe: ${dedupe}\n` +
1372
3287
  ` Cloud sync: ${cloud}`,
1373
3288
  };
@@ -1471,42 +3386,153 @@ export default {
1471
3386
  return handleTypedBeforeToolCall(event, interceptor, api.logger, ctx?.sessionId);
1472
3387
  }, { priority: 80, timeoutMs: 30_000 });
1473
3388
  _beforeToolCallRegistered = true;
1474
-
1475
- // Try to register session_end for cache cleanup (only meaningful while
1476
- // an interceptor can exist)
1477
- try {
1478
- api.on('session_end', (ev?: { sessionId?: string }) => {
1479
- interceptorReady?.resetSession();
1480
- // #233: a taint must not outlive the conversation that earned it.
1481
- if (ev?.sessionId) sessionTaint.clear(ev.sessionId);
1482
- });
1483
- } catch {
1484
- // session_end may not be a supported hook — TTL safety net handles this
1485
- }
3389
+ // NOTE: session_end is NOT registered here — it moved out of this guard
3390
+ // in #226 and is registered unconditionally below.
1486
3391
  } else {
1487
3392
  api.logger?.info?.('[shieldcortex] interceptor.enabled:false in plugin config — before_tool_call hook not registered');
1488
3393
  }
1489
3394
 
1490
- // These two are CONVERSATION hooks: OpenClaw drops them at registration
1491
- // for a non-bundled plugin unless the host grants
1492
- // plugins.entries.<id>.hooks.allowConversationAccess = true. We still
1493
- // attempt registration (the host decides, and the grant can be added
1494
- // without a code change), but we must not CLAIM them afterwards — see the
1495
- // honesty note on the log line below (#225).
3395
+ // session_end — registered UNCONDITIONALLY (#226).
3396
+ //
3397
+ // It used to live inside the `interceptorDisabledInHostConfig` guard above,
3398
+ // on the reasoning that it exists for the interceptor's session cache. That
3399
+ // stopped being true when `before_agent_run` landed: the gate is registered
3400
+ // regardless of `interceptor.enabled` (its posture, not the interceptor
3401
+ // flag, decides what it does), and it accumulates per-session
3402
+ // scan-unavailable suppression state. With the cleanup hook skipped, a host
3403
+ // that disabled the interceptor kept every session's window alive for the
3404
+ // life of the gateway process.
3405
+ //
3406
+ // Registering it does NOT reintroduce #112. That incident was specific to
3407
+ // `before_tool_call`: a registered approval hook changes how OpenClaw
3408
+ // resolves tool-call approvals for unattended Codex agents, so an
3409
+ // unattended turn waited 120s on a decision nobody could give. `session_end`
3410
+ // is a notification — it cannot block, approve, or delay anything — and its
3411
+ // handler here only frees local state.
3412
+ try {
3413
+ api.on('session_end', (event?: { sessionId?: string; sessionKey?: string }, ctx?: AgentCtx) => {
3414
+ interceptorReady?.resetSession();
3415
+ const endedSession = ctx?.sessionId ?? ctx?.sessionKey ?? event?.sessionId ?? event?.sessionKey ?? null;
3416
+ // #226: the scan-unavailable alert window is session state too, and it
3417
+ // is keyed per session — so clear THIS session's window and nobody
3418
+ // else's. Clearing them all would re-arm alerting for every live
3419
+ // session every time any one of them ended.
3420
+ resetScanUnavailableAlertState(endedSession);
3421
+ // #233: a taint must not outlive the conversation that earned it. Same
3422
+ // per-session rule, for the same reason.
3423
+ if (endedSession) sessionTaint.clear(endedSession);
3424
+ });
3425
+ } catch {
3426
+ // session_end may not be a supported hook — TTL safety net handles this
3427
+ }
3428
+
3429
+ // llm_input/llm_output are CONVERSATION hooks: OpenClaw drops them at
3430
+ // registration for a non-bundled plugin unless the host grants
3431
+ // plugins.entries.<id>.hooks.allowConversationAccess = true. Registration is
3432
+ // still attempted (the host decides, and the grant can be added without a
3433
+ // code change), but they must not be CLAIMED afterwards — see the startup
3434
+ // line below (#225/#230).
1496
3435
  api.on("llm_input", handleLlmInput, { timeoutMs: 30_000 });
1497
3436
  api.on("llm_output", handleLlmOutput, { timeoutMs: 30_000 });
1498
3437
 
1499
- // #225: this line used to announce `llm_input + llm_output` unconditionally.
1500
- // On any host without the conversation-access grant the gateway logged, on
1501
- // the very next two lines, that it had dropped both — so ShieldCortex was
1502
- // claiming conversation protection it did not have, in the one place an
1503
- // operator looks to confirm startup. Report only what is actually live, and
1504
- // name the missing grant when it is the reason.
1505
- const conversationAccess = readConversationAccess(homedir(), PLUGIN_ID);
3438
+ // #225: the conversation firewall's ENFORCEMENT point. `llm_input` above is
3439
+ // an OpenClaw *observation* hook — it cannot stop anything, which is why a
3440
+ // detected injection reached the model regardless and the only trace was a
3441
+ // console line. `before_agent_run` is the documented hook that can block a
3442
+ // run, so the verdict lands here where it can actually act.
3443
+ //
3444
+ // Registration is attempted unconditionally: the posture
3445
+ // (off/observe/enforce) decides what happens, and it is read per-call so a
3446
+ // config change takes effect without a restart. Registering conditionally
3447
+ // would make "is the guard wired?" depend on config read at boot — the
3448
+ // exact class of silent gap that #214/#222 were.
3449
+ //
3450
+ // The try/catch is for a host whose `api.on` throws on an unknown name. On
3451
+ // the hosts we have inspected it does NOT throw — an unsupported name is
3452
+ // dropped with a diagnostic, and a conversation hook without the operator's
3453
+ // grant is refused the same way — so a successful call proves only that we
3454
+ // ASKED. That is exactly what the flag is named after, and the honest
3455
+ // reporting comes from the version + consent evidence below.
3456
+ try {
3457
+ api.on("before_agent_run", handleBeforeAgentRun, { timeoutMs: 30_000 });
3458
+ _beforeAgentRunRequested = true;
3459
+ } catch (err) {
3460
+ _beforeAgentRunRequested = false;
3461
+ (api.logger as any)?.warn?.(
3462
+ `[shieldcortex] before_agent_run could not be registered on this host (${err instanceof Error ? err.message : String(err)}) — the conversation firewall cannot block on this gateway`,
3463
+ );
3464
+ }
3465
+
3466
+ // The gateway's own message seam, WHERE a host provides one (#143's design
3467
+ // intent: "on OpenClaw the transport should use the gateway's own message
3468
+ // capability"). Probed structurally, never required. No build we have
3469
+ // inspected exposes it — `notifyOperator` appears nowhere in the plugin API
3470
+ // of 2026.5.2 or 2026.7.1 — so on today's hosts this stays null and
3471
+ // conversation alerts go to the webhook.
3472
+ const notifyCtx = (api as { runtime?: { notifyOperator?: unknown }; notifyOperator?: unknown });
3473
+ if (typeof notifyCtx.notifyOperator === 'function') {
3474
+ _gatewayNotifyContext = notifyCtx as GatewayNotifyContext;
3475
+ } else if (typeof notifyCtx.runtime?.notifyOperator === 'function') {
3476
+ _gatewayNotifyContext = notifyCtx.runtime as GatewayNotifyContext;
3477
+ }
3478
+
3479
+ // The operator's conversation-access grant. Read, never written: OpenClaw
3480
+ // refuses every conversation hook for a non-bundled plugin without it, so a
3481
+ // box missing it runs with NO conversation plane at all — and on four of
3482
+ // five fleet hosts surveyed in #222 that was the normal outcome of a
3483
+ // documented install. Report it at boot rather than let the operator infer
3484
+ // protection from a registration line that only states intent.
3485
+ // The host's own in-memory config is the better source (it is what the
3486
+ // loader consulted), so it is preferred; the file the host reads is the
3487
+ // fallback for a runtime that does not expose it. `readConversationAccess`
3488
+ // is #225's shared reader — it also tells us whether the config could be
3489
+ // read at all, which is what keeps "not granted" apart from "cannot tell"
3490
+ // on the startup line below.
3491
+ const diskAccess = readConversationAccess(homedir(), PLUGIN_ID);
3492
+ let rootConfigSeen = false;
3493
+ try {
3494
+ const runtimeConfigApi = (api as PluginApi).runtime?.config;
3495
+ const rootConfig = typeof runtimeConfigApi?.current === 'function'
3496
+ ? runtimeConfigApi.current()
3497
+ : typeof runtimeConfigApi?.loadConfig === 'function'
3498
+ ? runtimeConfigApi.loadConfig()
3499
+ : (api as PluginApi).config;
3500
+ rootConfigSeen = Boolean(rootConfig) && typeof rootConfig === 'object';
3501
+ _conversationAccessGranted = rootConfigSeen
3502
+ ? readConversationAccessGrant(rootConfig)
3503
+ : diskAccess.granted;
3504
+ } catch {
3505
+ _conversationAccessGranted = diskAccess.granted;
3506
+ }
3507
+ if (!_conversationAccessGranted) {
3508
+ (api.logger as any)?.warn?.(
3509
+ `[shieldcortex] conversation firewall INACTIVE: plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess is not true in openclaw.json — ` +
3510
+ 'the gateway will refuse llm_input, llm_output and before_agent_run for this plugin. Nothing on the conversation path is scanned or blocked. ' +
3511
+ 'This is an operator consent grant and ShieldCortex will never set it for you.',
3512
+ );
3513
+ }
3514
+
3515
+ // #225/#230: this line used to announce `llm_input + llm_output`
3516
+ // unconditionally. On any host without the conversation-access grant the
3517
+ // gateway logged, on the very next two lines, that it had dropped both — so
3518
+ // ShieldCortex was claiming conversation protection it did not have, in the
3519
+ // one place an operator looks to confirm startup. Report only what is
3520
+ // actually live, and name the missing grant when it is the reason.
3521
+ //
3522
+ // `before_agent_run` (#226) is on the same list from 2026.5.9-beta.1, so it
3523
+ // is claimed only when the grant is present AND registration was attempted
3524
+ // this session.
1506
3525
  api.logger.info(
1507
3526
  `[shieldcortex] v${_version} registered (${describeRegisteredHooks({
1508
- access: conversationAccess,
3527
+ access: {
3528
+ granted: _conversationAccessGranted,
3529
+ // We could read SOMETHING (the host's config or the file) ⇒ the
3530
+ // ungranted state is a fact, not a failed measurement.
3531
+ readable: rootConfigSeen || diskAccess.readable,
3532
+ entryPresent: diskAccess.entryPresent,
3533
+ },
1509
3534
  beforeToolCallRegistered: _beforeToolCallRegistered,
3535
+ beforeAgentRunRequested: _beforeAgentRunRequested,
1510
3536
  })})`,
1511
3537
  );
1512
3538
  } catch (err) {