@drakon-systems/shieldcortex-realtime 4.47.37 → 4.47.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -2,16 +2,32 @@
2
2
  * ShieldCortex Real-time Scanning Plugin for OpenClaw v2026.3.22+
3
3
  *
4
4
  * Uses typed OpenClaw plugin hooks (`api.on`) for llm_input/llm_output
5
- * scanning and before_tool_call interception. `api.registerHook` registers
6
- * internal HOOK-style automation and does not participate in the agent-loop
7
- * block/approval semantics ShieldCortex needs.
8
- * All scanning operations are fire-and-forget.
5
+ * scanning and before_tool_call / before_agent_run interception.
6
+ * `api.registerHook` registers internal HOOK-style automation and does not
7
+ * participate in the agent-loop block/approval semantics ShieldCortex needs.
8
+ *
9
+ * NOT all scanning is fire-and-forget, and the distinction is the product:
10
+ *
11
+ * llm_input — OBSERVATION. Fire-and-forget; it has no blocking
12
+ * contract, so a detection here cannot stop the turn.
13
+ * before_agent_run — THE GATE (#225). Awaited by the gateway; its return
14
+ * value decides whether the run proceeds. Bounded end to
15
+ * end (CONVERSATION_SCAN_MAX_MS for the scan,
16
+ * CONVERSATION_NOTIFY_MAX_MS for the alert) because the
17
+ * user's turn waits on it, and failing OPEN on every
18
+ * internal error — as an EXPLICIT `{ outcome: 'pass' }`
19
+ * (#226), never as void, and never by throwing: the host
20
+ * registers this hook fail-CLOSED. See `gatePass`.
21
+ * before_tool_call — the Action Guard's gate, likewise awaited.
22
+ *
23
+ * Both conversation hooks honour `interceptor.conversation.posture`, including
24
+ * `off`, which is read before any scanner, audit write or cloud call.
9
25
  */
10
- import { createHash } from "node:crypto";
26
+ import { createHash, randomUUID } from "node:crypto";
11
27
  import fs from "node:fs/promises";
12
- import { existsSync, readFileSync, realpathSync } from "node:fs";
28
+ import { existsSync, readdirSync, readFileSync, realpathSync } from "node:fs";
13
29
  import path from "node:path";
14
- import { homedir } from "node:os";
30
+ import { homedir, hostname } from "node:os";
15
31
  import { fileURLToPath, pathToFileURL } from "node:url";
16
32
  import { readConversationAccess, describeRegisteredHooks } from './conversation-access.js';
17
33
  import { createSessionTaintStore } from './session-taint.js';
@@ -19,6 +35,7 @@ import { classifyConversationOrigin } from './conversation-trust.js';
19
35
  import { createInterceptor, DEFAULT_CONFIG as DEFAULT_INTERCEPTOR_CONFIG } from './interceptor.js';
20
36
  import { syncInterceptEvent } from './intercept-ingest.js';
21
37
  import { cloudSync } from './cloud-sync.js';
38
+ import { createGatewayNotifyChannel } from './gateway-notify-channel.js';
22
39
  let runtimePromise = null;
23
40
  function addRuntimeCandidate(candidates, packageRoot) {
24
41
  const runtimePath = path.join(packageRoot, "hooks", "openclaw", "cortex-memory", "runtime.mjs");
@@ -155,13 +172,492 @@ export function __resetConfigStateForTest() {
155
172
  _config = null;
156
173
  _configOverride = null;
157
174
  _lastShieldConfigRef = null;
175
+ // Re-arm the once-per-load config-failure warning (#226).
176
+ _shieldConfigLoadFailureLogged = false;
158
177
  _registered = false;
159
178
  _beforeToolCallRegistered = false;
160
179
  _registrationError = null;
180
+ _beforeAgentRunRequested = false;
181
+ _conversationAccessGranted = false;
182
+ _gatewayNotifyContext = null;
183
+ _hostRuntimeVersion = null;
184
+ __resetScanUnavailableAlertState();
161
185
  }
162
186
  const INTERCEPT_SEVERITIES = ['low', 'medium', 'high', 'critical'];
163
187
  const INTERCEPT_ACTIONS = ['log', 'warn', 'require_approval'];
164
188
  const FAILURE_ACTIONS = ['allow', 'deny'];
189
+ const CONVERSATION_POSTURES = ['off', 'observe', 'enforce'];
190
+ /** Resolve the configured posture. Anything unrecognised resolves DOWN to
191
+ * `observe`, never up to `enforce`: a typo must not silently start blocking
192
+ * every turn on an operator's box. */
193
+ export function conversationPosture(raw) {
194
+ if (!raw || typeof raw !== 'object')
195
+ return 'observe';
196
+ const value = raw.posture;
197
+ return typeof value === 'string' && CONVERSATION_POSTURES.includes(value)
198
+ ? value
199
+ : 'observe';
200
+ }
201
+ /**
202
+ * The whole decision, as a pure function — no I/O, no hooks, so the posture
203
+ * semantics are testable directly and cannot drift as the plumbing changes.
204
+ *
205
+ * The key line is `notify` on a non-blocking detection: logging is not a sink.
206
+ * The #225 finding was that a HIGH verdict reached a log file and nothing else,
207
+ * so a real threat on a real box was seen by nobody. Under `observe` we still
208
+ * do not stop the turn — but a human hears about it.
209
+ *
210
+ * `trust` is the second input because BLOCKING is a consequence, and #235's rule
211
+ * is that source trust gates consequences (see conversation-trust.ts). It is
212
+ * optional, and its absence means "origin not established", which resolves
213
+ * toward enforcement rather than away from it: a caller that does not know who
214
+ * spoke has not proved the owner did.
215
+ */
216
+ export function evaluateConversationRun(posture, scan, trust) {
217
+ if (posture === 'off') {
218
+ return { block: false, notify: false, audit: false, reason: null, outcome: 'not-scanned' };
219
+ }
220
+ // Scanner failure fails OPEN — a broken scanner must not wedge every turn,
221
+ // which is the outcome ShieldCortex exists to prevent — but it is reported,
222
+ // because an unprotected turn must never read as a protected one. Note the
223
+ // condition: `available === false` OR the legacy `errored` flag, so a caller
224
+ // still constructing the old shape cannot route an unscanned turn into the
225
+ // clean branch.
226
+ if (scan.available === false || scan.errored) {
227
+ return {
228
+ block: false,
229
+ notify: true,
230
+ audit: true,
231
+ reason: `conversation scan unavailable (${scan.error ?? scan.summary}) — turn allowed UNSCANNED`,
232
+ outcome: 'unavailable',
233
+ };
234
+ }
235
+ if (scan.clean) {
236
+ return { block: false, notify: false, audit: false, reason: null, outcome: 'clean' };
237
+ }
238
+ // #235: the owner's own words are an instruction, so `enforce` does not act on
239
+ // them. This is the branch the whole trust module exists for — a block here
240
+ // does not warn the operator, it DESTROYS their message: OpenClaw keeps only
241
+ // the replacement text. Everything above still happened: the content was
242
+ // scanned, and `notify`/`audit` below are true whoever sent it. Only the
243
+ // consequence is withheld, and the reason says so on the row rather than
244
+ // leaving an enforce-posture host that did not block looking like a bug.
245
+ const trusted = trust !== undefined && !trust.mayTaint;
246
+ const block = posture === 'enforce' && !trusted;
247
+ return {
248
+ block,
249
+ notify: true,
250
+ audit: true,
251
+ reason: posture === 'enforce' && trusted
252
+ ? `conversation threat: ${scan.summary} — NOT blocked: ${trust.reason}`
253
+ : `conversation threat: ${scan.summary}`,
254
+ outcome: block ? 'blocked' : 'observed',
255
+ };
256
+ }
257
+ // ==================== CONVERSATION PLANE: HOST SUPPORT + CONSENT ============
258
+ /**
259
+ * The first OpenClaw build whose plugin SDK declares the `before_agent_run`
260
+ * gate. Established by inspecting published npm artifacts, not by guessing:
261
+ *
262
+ * 2026.5.7 — `hook-types.d.ts` has no `before_agent_run` anywhere (0 hits);
263
+ * CONVERSATION_HOOK_NAMES = llm_input, llm_output,
264
+ * before_agent_finalize, agent_end
265
+ * 2026.5.9-beta.1 — FIRST published build declaring it: in `PLUGIN_HOOK_NAMES`,
266
+ * in `CONVERSATION_HOOK_NAMES`, in `PluginHookHandlerMap`, with
267
+ * `PluginHookBeforeAgentRunResult = InputGateDecision | void`
268
+ * 2026.5.12 — first STABLE (non-prerelease) build with it (2026.5.10 and
269
+ * 2026.5.12 published betas in between; there is no plain
270
+ * 2026.5.9 release)
271
+ *
272
+ * Below this floor `api.on('before_agent_run', …)` is accepted by the API and
273
+ * then DROPPED by the registry with an `unknown typed hook … ignored`
274
+ * diagnostic — it does not throw. So a version check is the only honest way to
275
+ * know, and claiming enforcement without one is exactly the class of false
276
+ * green #222 is about.
277
+ *
278
+ * ── ONE FLOOR, THREE FILES ────────────────────────────────────────────────
279
+ *
280
+ * The authoritative value is the STABLE release, and it is stated in three
281
+ * places that cannot import each other:
282
+ *
283
+ * plugins/openclaw/index.ts — this constant
284
+ * src/integrations/openclaw-conversation-capability.ts
285
+ * — CONVERSATION_ENFORCEMENT_MIN_OPENCLAW
286
+ * plugins/openclaw/openclaw.plugin.json — engines.conversationGate
287
+ *
288
+ * THE BOUNDARY IS REAL, not a preference. The plugin ships as its own dist,
289
+ * compiled by `tsconfig.openclaw-plugin.json` with `rootDir:
290
+ * ./plugins/openclaw` and an explicit `include` list; a `src/` import does not
291
+ * merely offend layering, it fails to emit — and the src module imports
292
+ * `semver`, which the plugin bundle does not carry (hence the hand-rolled
293
+ * `compareOpenClawVersions` below). The manifest is JSON read by the host and
294
+ * imports nothing at all.
295
+ *
296
+ * So the three are pinned EQUAL by test instead of shared by import:
297
+ * `src/__tests__/conversation-gate-floor-parity-226.test.ts` reads all three
298
+ * and fails on drift. Change one, that test tells you about the other two.
299
+ *
300
+ * `CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW` is deliberately SUBORDINATE: it
301
+ * decides nothing an operator sees. Its only job is to mark the band where a
302
+ * version number alone cannot answer the question — see
303
+ * `hostSupportsConversationGate`.
304
+ */
305
+ export const CONVERSATION_GATE_MIN_OPENCLAW = '2026.5.12';
306
+ /**
307
+ * Documentation of when the hook first appeared, NOT a second floor.
308
+ *
309
+ * The previous cut used this as the support threshold, which made the plugin's
310
+ * operator-facing verdict disagree with the CLI's on any 2026.5.9-beta.1 →
311
+ * 2026.5.11 host: `shieldcortex doctor` said enforcement was unavailable while
312
+ * the plugin's own status line said supported. Two answers to one question is
313
+ * how the next false green gets built.
314
+ */
315
+ export const CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW = '2026.5.9-beta.1';
316
+ /**
317
+ * Compare two OpenClaw CalVer strings (`2026.5.12`, `2026.5.9-beta.1`).
318
+ * Deliberately local and tiny: the plugin build cannot import semver, and the
319
+ * only question asked is "is this host at or above the floor".
320
+ * Returns null when either side cannot be parsed — "unknown", never "yes".
321
+ */
322
+ export function compareOpenClawVersions(a, b) {
323
+ const parse = (v) => {
324
+ // Exactly three numeric parts, and only `-` introduces a prerelease. A
325
+ // trailing `.4` is NOT a prerelease tail — it is a version shape we do not
326
+ // understand, and the safe answer to that is "unknown".
327
+ const m = String(v ?? '').trim().match(/^(\d+)\.(\d+)\.(\d+)(?:-([0-9A-Za-z.-]+))?$/);
328
+ if (!m)
329
+ return null;
330
+ return { nums: [Number(m[1]), Number(m[2]), Number(m[3])], pre: m[4] ? m[4].split('.') : [] };
331
+ };
332
+ const pa = parse(a);
333
+ const pb = parse(b);
334
+ if (!pa || !pb)
335
+ return null;
336
+ for (let i = 0; i < 3; i++) {
337
+ if (pa.nums[i] !== pb.nums[i])
338
+ return pa.nums[i] < pb.nums[i] ? -1 : 1;
339
+ }
340
+ // A prerelease sorts BELOW the same numeric release (2026.5.9-beta.1 < 2026.5.9).
341
+ if (pa.pre.length === 0 && pb.pre.length === 0)
342
+ return 0;
343
+ if (pa.pre.length === 0)
344
+ return 1;
345
+ if (pb.pre.length === 0)
346
+ return -1;
347
+ return comparePrerelease(pa.pre, pb.pre);
348
+ }
349
+ /** Semver prerelease precedence, restricted to what a CalVer tail can hold:
350
+ * numeric identifiers compare numerically, a numeric identifier sorts below an
351
+ * alphanumeric one, and a shorter identifier list sorts below an
352
+ * otherwise-equal longer one (`beta` < `beta.1` < `beta.2` < `beta.10`).
353
+ *
354
+ * The string compare this replaces put `beta.10` BELOW `beta.1`, so the tenth
355
+ * beta of the gate build was classified as predating the first — an error in
356
+ * the one direction this file must never make, since it demotes a host that
357
+ * HAS the gate to 'unsupported'. */
358
+ function comparePrerelease(a, b) {
359
+ const len = Math.max(a.length, b.length);
360
+ for (let i = 0; i < len; i++) {
361
+ const x = a[i];
362
+ const y = b[i];
363
+ if (x === undefined)
364
+ return -1;
365
+ if (y === undefined)
366
+ return 1;
367
+ const xNum = /^\d+$/.test(x);
368
+ const yNum = /^\d+$/.test(y);
369
+ if (xNum && yNum) {
370
+ const nx = Number(x);
371
+ const ny = Number(y);
372
+ if (nx !== ny)
373
+ return nx < ny ? -1 : 1;
374
+ continue;
375
+ }
376
+ if (xNum !== yNum)
377
+ return xNum ? -1 : 1;
378
+ if (x !== y)
379
+ return x < y ? -1 : 1;
380
+ }
381
+ return 0;
382
+ }
383
+ /** Test seam: pins the host probe without touching disk. */
384
+ let _hostProbeOverride;
385
+ export function __setHostOpenClawProbeForTest(p) {
386
+ _hostProbeOverride = p;
387
+ _hostProbeCache = undefined;
388
+ }
389
+ let _hostProbeCache;
390
+ /**
391
+ * The host runtime version the gateway told us about, captured at register().
392
+ *
393
+ * `api.runtime.version` is declared by the host SDK as
394
+ * `PluginRuntimeCore.version: string` — the version of the OpenClaw runtime
395
+ * this plugin is loaded into (verified against the installed host's
396
+ * `dist/plugin-sdk/src/plugins/runtime/types-core.d.ts`, and `api.runtime` is
397
+ * on `OpenClawPluginApi` in the same build). It is NOT `api.version`, which is
398
+ * the plugin's own version and would answer a completely different question:
399
+ * comparing OUR version against an OpenClaw floor would classify every host as
400
+ * unsupported.
401
+ *
402
+ * It is preferred over the filesystem walk because it is the running process
403
+ * describing itself, where the walk infers from whichever package.json happens
404
+ * to sit above the entry path. Null until a host actually supplies it — an
405
+ * older gateway, a CLI invocation or a test rig may not, and that is UNKNOWN.
406
+ */
407
+ let _hostRuntimeVersion = null;
408
+ /** Test seam for the runtime-supplied host version. */
409
+ export function __setHostRuntimeVersionForTest(v) {
410
+ _hostRuntimeVersion = v;
411
+ }
412
+ /**
413
+ * Record `api.runtime.version` if this host exposes it. Returns what was
414
+ * recorded (null when nothing usable was offered), and never throws: a host
415
+ * with an exotic `runtime` getter must not take the plugin's registration down.
416
+ */
417
+ export function recordHostRuntimeVersion(api) {
418
+ try {
419
+ const runtime = api?.runtime;
420
+ const version = runtime?.version;
421
+ _hostRuntimeVersion = typeof version === 'string' && version.trim() ? version.trim() : null;
422
+ }
423
+ catch {
424
+ _hostRuntimeVersion = null;
425
+ }
426
+ return _hostRuntimeVersion;
427
+ }
428
+ /** Does this host's shipped SDK declare the gate? Bounded, best-effort, and
429
+ * null on anything unexpected — an unreadable install is UNKNOWN, never
430
+ * "supported". */
431
+ function probeGateDeclaration(root) {
432
+ const candidates = [];
433
+ // 2026.5.2-era layout: the declarations live under the plugin-sdk tree.
434
+ candidates.push(path.join(root, 'dist', 'plugin-sdk', 'src', 'plugins', 'hook-types.d.ts'));
435
+ // 2026.6+/2026.7 layout: a single hashed `hook-types-<hash>.d.ts` at dist root.
436
+ try {
437
+ const distDir = path.join(root, 'dist');
438
+ if (existsSync(distDir)) {
439
+ for (const name of readdirSync(distDir)) {
440
+ if (/^hook-types.*\.d\.ts$/.test(name))
441
+ candidates.push(path.join(distDir, name));
442
+ }
443
+ }
444
+ }
445
+ catch { /* fall through to whatever candidates we have */ }
446
+ let sawAny = false;
447
+ for (const file of candidates) {
448
+ try {
449
+ if (!existsSync(file))
450
+ continue;
451
+ sawAny = true;
452
+ if (/\bbefore_agent_run\b/.test(readFileSync(file, 'utf-8')))
453
+ return true;
454
+ }
455
+ catch { /* unreadable candidate — try the next */ }
456
+ }
457
+ return sawAny ? false : null;
458
+ }
459
+ /**
460
+ * Everything we can learn about the host OpenClaw build from inside the plugin.
461
+ *
462
+ * Two independent sources, both read, neither invented:
463
+ *
464
+ * - `api.runtime.version` — the running gateway's own statement of its
465
+ * version, captured at register() (see `recordHostRuntimeVersion`). This is
466
+ * the primary VERSION evidence when the host offers it. Note it is not
467
+ * `api.version`, which is this plugin's version.
468
+ * - the filesystem — walk up from the gateway's entry path to the package.json
469
+ * that names openclaw, then read that install's own shipped hook
470
+ * declarations. This is the version FALLBACK, and it is the only source of
471
+ * `declaresGate`, which stays the strongest gate-support evidence of the two
472
+ * (a backport or a fork answers it correctly where a version comparison
473
+ * cannot — see `hostSupportsConversationGate`).
474
+ *
475
+ * No process is ever spawned. A null everywhere is a legitimate, frequently
476
+ * correct answer (a CLI invocation, an unusual install layout) and callers must
477
+ * treat it as UNKNOWN — never as "supported".
478
+ */
479
+ export function detectHostOpenClaw() {
480
+ const disk = detectHostOpenClawFromDisk();
481
+ // The runtime's own version outranks whatever package.json the walk landed
482
+ // on — but only for the version; `declaresGate` and `root` are disk facts and
483
+ // are carried through untouched.
484
+ if (_hostRuntimeVersion)
485
+ return { ...disk, version: _hostRuntimeVersion, versionSource: 'runtime' };
486
+ return disk;
487
+ }
488
+ function detectHostOpenClawFromDisk() {
489
+ if (_hostProbeOverride !== undefined)
490
+ return _hostProbeOverride ?? { version: null, root: null, declaresGate: null };
491
+ if (_hostProbeCache !== undefined)
492
+ return _hostProbeCache;
493
+ _hostProbeCache = (() => {
494
+ const empty = { version: null, root: null, declaresGate: null, versionSource: null };
495
+ const entry = process.argv?.[1];
496
+ if (!entry || typeof entry !== 'string')
497
+ return empty;
498
+ let current;
499
+ try {
500
+ current = path.dirname(realpathSync(entry));
501
+ }
502
+ catch {
503
+ current = path.dirname(entry);
504
+ }
505
+ let previous = '';
506
+ for (let i = 0; i < 8 && current !== previous; i++) {
507
+ try {
508
+ const pkgPath = path.join(current, 'package.json');
509
+ if (existsSync(pkgPath)) {
510
+ const hostPkg = JSON.parse(readFileSync(pkgPath, 'utf-8'));
511
+ if (hostPkg?.name === 'openclaw') {
512
+ const version = typeof hostPkg.version === 'string' ? hostPkg.version : null;
513
+ return {
514
+ version,
515
+ versionSource: version ? 'package.json' : null,
516
+ root: current,
517
+ declaresGate: probeGateDeclaration(current),
518
+ };
519
+ }
520
+ }
521
+ }
522
+ catch { /* keep walking up */ }
523
+ previous = current;
524
+ current = path.dirname(current);
525
+ }
526
+ return empty;
527
+ })();
528
+ return _hostProbeCache;
529
+ }
530
+ /** Convenience for callers that only want the version string. */
531
+ export function detectHostOpenClawVersion() {
532
+ return detectHostOpenClaw().version;
533
+ }
534
+ /**
535
+ * Does this host have the `before_agent_run` gate at all?
536
+ *
537
+ * Order matters: what the installed build DECLARES outranks what its version
538
+ * number implies, and both outrank a guess. There is no branch here that
539
+ * returns 'supported' without evidence.
540
+ */
541
+ export function hostSupportsConversationGate(probe) {
542
+ const resolved = typeof probe === 'string' || probe === null
543
+ ? { version: probe, root: null, declaresGate: null }
544
+ : probe;
545
+ if (resolved.declaresGate === true)
546
+ return 'supported';
547
+ if (resolved.declaresGate === false)
548
+ return 'unsupported';
549
+ if (!resolved.version)
550
+ return 'unknown';
551
+ const cmp = compareOpenClawVersions(resolved.version, CONVERSATION_GATE_MIN_OPENCLAW);
552
+ if (cmp === null)
553
+ return 'unknown';
554
+ if (cmp >= 0)
555
+ return 'supported';
556
+ // Below the STABLE floor. One band inside that is not honestly 'unsupported':
557
+ // 2026.5.9-beta.1 → 2026.5.11 ship the hook as a prerelease, so calling them
558
+ // unsupported would tell an operator "no posture can block a turn on this
559
+ // host" about a host that blocks. The opposite claim is worse still, so
560
+ // neither is made: this is the absence of a measurement, and
561
+ // `describeConversationPlane` renders it as UNPROVEN and active:false.
562
+ //
563
+ // In practice a real prerelease install lands on `declaresGate` above and
564
+ // never reaches here — this branch is what happens when the declarations
565
+ // could not be read either, i.e. when we genuinely do not know.
566
+ const pre = compareOpenClawVersions(resolved.version, CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW);
567
+ if (pre === null)
568
+ return 'unknown';
569
+ return pre >= 0 ? 'unknown' : 'unsupported';
570
+ }
571
+ /**
572
+ * Read the operator's CONVERSATION-ACCESS consent for this plugin from the
573
+ * host config: `plugins.entries.<id>.hooks.allowConversationAccess === true`.
574
+ *
575
+ * OpenClaw refuses every conversation hook for a non-bundled plugin without
576
+ * this exact value (registry: `record.origin !== "bundled" &&
577
+ * explicitConversationAccess !== true`). `llm_input` and `llm_output` are on
578
+ * that list in every build; `before_agent_run` joins it in 2026.5.9-beta.1,
579
+ * the same build that first declares the gate at all — so from there on the
580
+ * grant governs the conversation firewall's enforcement point too.
581
+ * Strict `true` only, matching the host: `undefined` and `false` are the same
582
+ * refusal there, and reading them differently here would report protection the
583
+ * gateway is not providing.
584
+ *
585
+ * It is the operator's per-box CONSENT grant, and this plugin only ever READS
586
+ * it. Nothing on the plugin's own path — `register()`, a hook, a background
587
+ * refresh — may write it: a security product that silently grants itself the
588
+ * right to read every conversation is the behaviour this product exists to
589
+ * catch. The only thing that may set it is an explicit, operator-initiated
590
+ * install/repair that says so out loud (#225: "the installer must never set it
591
+ * silently"). Absence is therefore reported, loudly and by name, rather than
592
+ * fixed from in here.
593
+ */
594
+ export function readConversationAccessGrant(rootConfig) {
595
+ if (!rootConfig || typeof rootConfig !== 'object' || Array.isArray(rootConfig))
596
+ return false;
597
+ const entries = rootConfig.plugins?.entries;
598
+ const entry = entries?.[PLUGIN_ID] ?? entries?.[PLUGIN_PACKAGE_NAME];
599
+ return entry?.hooks?.allowConversationAccess === true;
600
+ }
601
+ export function describeConversationPlane(input) {
602
+ const { posture, hookRequested, gateSupport, hostOpenClawVersion, consentGranted } = input;
603
+ const hostText = hostOpenClawVersion ? `OpenClaw ${hostOpenClawVersion}` : 'OpenClaw version undetermined';
604
+ if (posture === 'off') {
605
+ return {
606
+ ...input,
607
+ active: false,
608
+ summary: 'off — conversation scanning disabled by config (interceptor.conversation.posture=off)',
609
+ };
610
+ }
611
+ if (!consentGranted) {
612
+ return {
613
+ ...input,
614
+ active: false,
615
+ summary: `INACTIVE: conversation access NOT granted on this host — set plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess=true ` +
616
+ 'in openclaw.json (operator consent; the installer will never set it for you). Until then the gateway REFUSES llm_input and ' +
617
+ 'llm_output for this plugin — and, on builds that have it, before_agent_run too: nothing on the conversation path is scanned or blocked',
618
+ };
619
+ }
620
+ if (gateSupport === 'unsupported') {
621
+ return {
622
+ ...input,
623
+ active: false,
624
+ summary: `INACTIVE for enforcement: ${hostText} predates the before_agent_run gate ` +
625
+ `(floor ${CONVERSATION_GATE_MIN_OPENCLAW}; first seen as a prerelease in ${CONVERSATION_GATE_FIRST_PRERELEASE_OPENCLAW}) — ` +
626
+ 'observation only; no posture can block a turn on this host',
627
+ };
628
+ }
629
+ if (!hookRequested) {
630
+ return { ...input, active: false, summary: 'INACTIVE: the before_agent_run hook was not registered this session' };
631
+ }
632
+ if (gateSupport === 'unknown') {
633
+ // UNPROVEN IS NOT ACTIVE. Every other branch above is a fact we read — the
634
+ // posture, the grant, the host build. This one is the absence of a
635
+ // measurement: we could not establish that this host has the gate at all,
636
+ // and `api.on` does not acknowledge a registration, so nothing here knows
637
+ // whether the hook exists. Reporting `active: true` with a caveat glued to
638
+ // the summary string — which is what this did — means every caller that
639
+ // reads the boolean instead of the prose (status renderers, doctor, any
640
+ // future check) claims a live firewall on evidence nobody has. Under
641
+ // `enforce` that is the worst version of it: the operator believes turns
642
+ // are being blocked on a host where the gate may be silently dropped.
643
+ return {
644
+ ...input,
645
+ active: false,
646
+ summary: `UNPROVEN: could not verify that host ${hostText} provides the before_agent_run gate ` +
647
+ `(no runtime version, no readable hook declarations), and the plugin API does not acknowledge a registration. ` +
648
+ (posture === 'enforce'
649
+ ? 'The posture is enforce, so a dirty verdict WOULD block the run where the gate exists — but that it exists here is not established. Treat this host as observation-only until it is.'
650
+ : 'Detections are audited and sent to the operator where the hook runs at all; nothing is blocked in this posture regardless.'),
651
+ };
652
+ }
653
+ return {
654
+ ...input,
655
+ active: true,
656
+ summary: posture === 'enforce'
657
+ ? 'enforce — a dirty verdict BLOCKS the run via before_agent_run'
658
+ : 'observe — detections are audited and sent to the operator; turns are NOT blocked',
659
+ };
660
+ }
165
661
  const PLUGIN_ID = "shieldcortex-realtime";
166
662
  /**
167
663
  * #233: conversation-level taint, shared between the conversation scan (which
@@ -213,6 +709,21 @@ const PLUGIN_CONFIG_UI_HINTS = {
213
709
  label: "Enable Tool Call Interceptor",
214
710
  help: "Scan memory-write tool calls and gate suspicious content behind user approval.",
215
711
  },
712
+ // #226: these two exist in openclaw.plugin.json's uiHints and were missing
713
+ // here, so the host UI and the plugin's own declared hints described
714
+ // different sets of settings. The manifest parity test now pins the two key
715
+ // sets EQUAL in both directions, because a hint present on only one side is
716
+ // a setting one surface documents and the other silently omits.
717
+ "interceptor.severityActions.high": {
718
+ label: "High Severity Action",
719
+ help: "Action for high-severity threats: log, warn, or require_approval.",
720
+ advanced: true,
721
+ },
722
+ "interceptor.severityActions.critical": {
723
+ label: "Critical Severity Action",
724
+ help: "Action for critical-severity threats: log, warn, or require_approval.",
725
+ advanced: true,
726
+ },
216
727
  "interceptor.actionGuard.enabled": {
217
728
  label: "Action Guard",
218
729
  help: "Gate dangerous shell/file/network/git tool calls before they execute. Catastrophic operations are always blocked while enabled.",
@@ -241,6 +752,38 @@ const PLUGIN_CONFIG_UI_HINTS = {
241
752
  help: "Off = every dangerous-tier action still waits for you; the broker can then only harden, never release.",
242
753
  advanced: true,
243
754
  },
755
+ // #225. The posture is the whole product claim on the conversation path, so
756
+ // it is NOT marked advanced: an operator must be able to see, in the UI that
757
+ // configures this plugin, whether the firewall in front of their prompts can
758
+ // stop anything.
759
+ "interceptor.conversation.posture": {
760
+ label: "Conversation Firewall",
761
+ help: "What the conversation firewall does with a detection on the input path. " +
762
+ "off = do not scan; observe = scan, audit and alert the operator but never stop the turn (default); " +
763
+ "enforce = block the run via before_agent_run. Requires plugins.entries.shieldcortex-realtime.hooks.allowConversationAccess=true " +
764
+ "on this host — OpenClaw refuses conversation hooks without that operator grant, and ShieldCortex will never set it for you.",
765
+ },
766
+ "interceptor.actionGuard.notify.enabled": {
767
+ label: "Operator Notifications",
768
+ help: "Reach a human when the guard holds an action, or when the conversation firewall detects a threat. Off by default.",
769
+ advanced: true,
770
+ },
771
+ "interceptor.actionGuard.notify.webhookUrl": {
772
+ label: "Notify Webhook URL",
773
+ help: "http(s) endpoint the notification is POSTed to. Conversation-firewall alerts carry no approve/deny affordance — there is nothing to approve.",
774
+ advanced: true,
775
+ },
776
+ "interceptor.actionGuard.notify.webhookSecret": {
777
+ label: "Notify Webhook Secret",
778
+ help: "HMAC-SHA256 key for X-ShieldCortex-Signature, so the receiver can reject spoofed POSTs.",
779
+ sensitive: true,
780
+ advanced: true,
781
+ },
782
+ "interceptor.actionGuard.notify.openclaw": {
783
+ label: "Notify via OpenClaw",
784
+ help: "Deliver through the gateway's own channel where the runtime provides that seam.",
785
+ advanced: true,
786
+ },
244
787
  };
245
788
  const SEVERITY_ACTION_SCHEMA = {
246
789
  type: "object",
@@ -252,63 +795,137 @@ const FAILURE_POLICY_SCHEMA = {
252
795
  additionalProperties: false,
253
796
  properties: Object.fromEntries(INTERCEPT_SEVERITIES.map((severity) => [severity, { type: "string", enum: [...FAILURE_ACTIONS] }])),
254
797
  };
255
- const INTERCEPTOR_JSON_SCHEMA = {
798
+ /** #225. Mirrored verbatim into openclaw.plugin.json's configSchema — the host
799
+ * validates the on-disk config against THAT file, so a posture accepted here
800
+ * and absent there is a config an operator writes from our own docs and the
801
+ * gateway rejects. */
802
+ const CONVERSATION_JSON_SCHEMA = {
803
+ type: "object",
804
+ additionalProperties: false,
805
+ properties: {
806
+ posture: {
807
+ type: "string",
808
+ enum: [...CONVERSATION_POSTURES],
809
+ default: "observe",
810
+ description: "off = do not scan the conversation; observe = scan, audit and alert but never block (default); " +
811
+ "enforce = block the run on a dirty verdict via before_agent_run.",
812
+ },
813
+ },
814
+ };
815
+ /** #235/#226. Mirrored verbatim into openclaw.plugin.json's configSchema, for
816
+ * the same reason as the conversation posture above: the host validates the
817
+ * on-disk config against THAT file, and a key our parser reads but neither
818
+ * schema declares is one an operator cannot set at all. */
819
+ const CONVERSATION_TRUST_JSON_SCHEMA = {
820
+ type: "object",
821
+ additionalProperties: false,
822
+ properties: {
823
+ trustOwnerInput: {
824
+ type: "boolean",
825
+ default: true,
826
+ description: "Default true: a message the host attributes to the gateway OWNER is an instruction, so a detection in it " +
827
+ "is audited and alerted but never taints the session or blocks the turn. Set false on a host where the owner " +
828
+ "routinely pastes untrusted content and you would rather have the caution than the quiet. Content from " +
829
+ "anyone else — including another agent on a trusted channel — is data regardless of this setting.",
830
+ },
831
+ },
832
+ };
833
+ /**
834
+ * The Action Guard block, declared ONCE and mounted in BOTH places the parser
835
+ * accepts it (#226).
836
+ *
837
+ * `normaliseConfig` has read a TOP-LEVEL `actionGuard` since #209 — that is the
838
+ * canonical location, and `interceptor.actionGuard` is the deprecated alias
839
+ * kept for pre-#209 configs. The schemas said the opposite: only the nested
840
+ * alias was declared, under `additionalProperties: false`, so a config written
841
+ * from our own documentation — `actionGuard.notify` at the top level — was
842
+ * rejected as an unknown key by any host that validates against the schema.
843
+ * The parser would have kept it; the config never reached the parser. That is
844
+ * the shape behind the original `parsedNotify: null` reproduction.
845
+ *
846
+ * One constant, two mount points, so the two can never drift. Mirrored by hand
847
+ * into openclaw.plugin.json's configSchema (the host validates the on-disk
848
+ * config against THAT file) and pinned equal by manifest-config-schema-226.test.ts.
849
+ */
850
+ const ACTION_GUARD_JSON_SCHEMA = {
256
851
  type: "object",
257
852
  additionalProperties: false,
258
853
  properties: {
259
854
  enabled: { type: "boolean" },
260
- severityActions: SEVERITY_ACTION_SCHEMA,
261
- failurePolicy: FAILURE_POLICY_SCHEMA,
262
- actionGuard: {
855
+ enforce: { type: "boolean" },
856
+ autoApprove: { type: "array", items: { type: "string" } },
857
+ auditAllows: { type: "boolean" },
858
+ // #143. Mirrors normaliseBrokerConfig's allowlist; that function still
859
+ // has the last word, so a value that slips past the schema is still
860
+ // range-checked (and dropped) before the broker sees it.
861
+ broker: {
263
862
  type: "object",
264
863
  additionalProperties: false,
265
864
  properties: {
266
865
  enabled: { type: "boolean" },
267
- enforce: { type: "boolean" },
268
- autoApprove: { type: "array", items: { type: "string" } },
269
- auditAllows: { type: "boolean" },
270
- // #143. Mirrors normaliseBrokerConfig's allowlist; that function still
271
- // has the last word, so a value that slips past the schema is still
272
- // range-checked (and dropped) before the broker sees it.
273
- broker: {
866
+ allowPreClear: { type: "boolean" },
867
+ preClearConfidence: { type: "number", minimum: 0.9, maximum: 1 },
868
+ judgeTimeoutMs: { type: "number", minimum: 500, maximum: 60000 },
869
+ approvalTimeoutMs: {
274
870
  type: "object",
275
871
  additionalProperties: false,
276
872
  properties: {
277
- enabled: { type: "boolean" },
278
- allowPreClear: { type: "boolean" },
279
- preClearConfidence: { type: "number", minimum: 0.9, maximum: 1 },
280
- judgeTimeoutMs: { type: "number", minimum: 500, maximum: 60000 },
281
- approvalTimeoutMs: {
282
- type: "object",
283
- additionalProperties: false,
284
- properties: {
285
- sensitive: { type: "number", minimum: 1000, maximum: 3600000 },
286
- dangerous: { type: "number", minimum: 1000, maximum: 3600000 },
287
- },
288
- },
289
- model: { type: "string" },
873
+ sensitive: { type: "number", minimum: 1000, maximum: 3600000 },
874
+ dangerous: { type: "number", minimum: 1000, maximum: 3600000 },
290
875
  },
291
876
  },
292
- // #189. Each entry pins one script by absolute path + content hash;
293
- // createReviewedScriptCheck has the last word on every field.
294
- reviewedScripts: {
295
- type: "array",
296
- items: {
297
- type: "object",
298
- additionalProperties: false,
299
- properties: {
300
- path: { type: "string" },
301
- sha256: { type: "string" },
302
- note: { type: "string" },
303
- addedAt: { type: "number" },
304
- },
305
- required: ["path", "sha256"],
306
- },
877
+ model: { type: "string" },
878
+ },
879
+ },
880
+ // #143/#225. Mirrors NotifyConfig in notify-config.ts, which still has
881
+ // the last word (strict-true booleans, http(s)-only URL, bounded
882
+ // timeout). Declared here because `additionalProperties: false` above
883
+ // means an undeclared key makes the WHOLE containing block invalid on
884
+ // a host that validates config against this schema — which is how the
885
+ // Action Guard's notify transport came to be unusable from inside the
886
+ // gateway plugin at all.
887
+ notify: {
888
+ type: "object",
889
+ additionalProperties: false,
890
+ properties: {
891
+ enabled: { type: "boolean" },
892
+ webhookUrl: { type: "string" },
893
+ webhookSecret: { type: "string" },
894
+ openclaw: { type: "boolean" },
895
+ timeoutMs: { type: "number", minimum: 500, maximum: 60000 },
896
+ },
897
+ },
898
+ // #189. Each entry pins one script by absolute path + content hash;
899
+ // createReviewedScriptCheck has the last word on every field.
900
+ reviewedScripts: {
901
+ type: "array",
902
+ items: {
903
+ type: "object",
904
+ additionalProperties: false,
905
+ properties: {
906
+ path: { type: "string" },
907
+ sha256: { type: "string" },
908
+ note: { type: "string" },
909
+ addedAt: { type: "number" },
307
910
  },
911
+ required: ["path", "sha256"],
308
912
  },
309
913
  },
310
914
  },
311
915
  };
916
+ const INTERCEPTOR_JSON_SCHEMA = {
917
+ type: "object",
918
+ additionalProperties: false,
919
+ properties: {
920
+ enabled: { type: "boolean" },
921
+ severityActions: SEVERITY_ACTION_SCHEMA,
922
+ failurePolicy: FAILURE_POLICY_SCHEMA,
923
+ conversation: CONVERSATION_JSON_SCHEMA,
924
+ /** The DEPRECATED alias (#209). Still accepted, still parsed, still
925
+ * gap-fills the canonical top-level block key by key. */
926
+ actionGuard: ACTION_GUARD_JSON_SCHEMA,
927
+ },
928
+ };
312
929
  const PLUGIN_CONFIG_JSON_SCHEMA = {
313
930
  type: "object",
314
931
  additionalProperties: false,
@@ -331,6 +948,15 @@ const PLUGIN_CONFIG_JSON_SCHEMA = {
331
948
  // #112: without this, `additionalProperties: false` declared the whole
332
949
  // interceptor block invalid — mirror openclaw.plugin.json's configSchema.
333
950
  interceptor: INTERCEPTOR_JSON_SCHEMA,
951
+ // #209/#226: the CANONICAL Action Guard location. normaliseConfig has read
952
+ // it here since #209 and folds it over the nested alias; the schema did not
953
+ // declare it, so `additionalProperties: false` rejected the documented
954
+ // config shape before the parser ever saw it.
955
+ actionGuard: ACTION_GUARD_JSON_SCHEMA,
956
+ // #235/#226: source trust. Same reason as `actionGuard` above — the parser
957
+ // reads it, so the schema must declare it or `additionalProperties: false`
958
+ // rejects the whole config an operator writes from our documentation.
959
+ conversationTrust: CONVERSATION_TRUST_JSON_SCHEMA,
334
960
  },
335
961
  };
336
962
  let _config = null;
@@ -369,6 +995,17 @@ let _registered = false;
369
995
  // unattended Codex agents even a no-op registered hook changes how OpenClaw
370
996
  // resolves approvals).
371
997
  let _beforeToolCallRegistered = false;
998
+ // #225: whether we CALLED api.on('before_agent_run', …) this session — nothing
999
+ // more. It is deliberately not named "…Accepted": `api.on` returns void, never
1000
+ // throws on an unknown hook name, and never throws when the host refuses a
1001
+ // conversation hook (it records a diagnostic and returns), so acceptance is not
1002
+ // observable from in here. The two facts that decide whether the plane is live
1003
+ // — host build ≥ the gate's floor, and the operator's allowConversationAccess
1004
+ // grant — are read separately and reported by describeConversationPlane().
1005
+ let _beforeAgentRunRequested = false;
1006
+ // The operator's conversation-access grant as read from the host config at
1007
+ // register() time. Reported by /shieldcortex-status; never written by us.
1008
+ let _conversationAccessGranted = false;
372
1009
  // #134 §2: register() wraps its whole body in try/catch so a plugin failure
373
1010
  // never blocks channel startup — correct, but it used to report the failure
374
1011
  // with a bare console.warn (bypasses the gateway's structured log, so the
@@ -421,6 +1058,20 @@ function normaliseConfig(raw, dropped) {
421
1058
  const interceptor = normaliseInterceptorConfig(value.interceptor, dropped);
422
1059
  if (interceptor)
423
1060
  config.interceptor = interceptor;
1061
+ // Source trust (#235, wired to the gate in #226). Booleans only, and the
1062
+ // block is kept only when it holds a valid one: a `trustOwnerInput: "false"`
1063
+ // typo — the exact shape of the #112 incident — must not read as the opt-out
1064
+ // having been applied, because the operator who wrote it believes owner input
1065
+ // is being policed and it would not be.
1066
+ if (value.conversationTrust && typeof value.conversationTrust === "object" && !Array.isArray(value.conversationTrust)) {
1067
+ const trustRaw = value.conversationTrust;
1068
+ if (typeof trustRaw.trustOwnerInput === "boolean") {
1069
+ config.conversationTrust = { trustOwnerInput: trustRaw.trustOwnerInput };
1070
+ }
1071
+ else if (trustRaw.trustOwnerInput !== undefined) {
1072
+ dropped?.push("conversationTrust.trustOwnerInput");
1073
+ }
1074
+ }
424
1075
  // #209: single source of truth for the Action Guard. A top-level
425
1076
  // `actionGuard` block governs every surface; `interceptor.actionGuard` is a
426
1077
  // deprecated alias kept as per-key gap-fill so pre-#209 configs keep their
@@ -521,6 +1172,20 @@ function normaliseActionGuardBlock(raw, dropped, pathPrefix) {
521
1172
  if (Array.isArray(rawGuard.reviewedScripts)) {
522
1173
  guard.reviewedScripts = [...rawGuard.reviewedScripts];
523
1174
  }
1175
+ // #143/#225: the notify transport. Same passthrough discipline again —
1176
+ // normaliseNotifyConfig is the boundary. Shallow-copied rather than aliased
1177
+ // (the #115 reason: a later in-place mutation of the host config object must
1178
+ // not reach into the normalised one), and a non-object is DROPPED by name so
1179
+ // the #115 warn log can say which key was ignored, rather than silently
1180
+ // leaving the operator with a transport that never fires.
1181
+ if (rawGuard.notify !== undefined) {
1182
+ if (rawGuard.notify && typeof rawGuard.notify === 'object' && !Array.isArray(rawGuard.notify)) {
1183
+ guard.notify = { ...rawGuard.notify };
1184
+ }
1185
+ else {
1186
+ dropped?.push(`${pathPrefix}.notify`);
1187
+ }
1188
+ }
524
1189
  return Object.keys(guard).length > 0 ? guard : undefined;
525
1190
  }
526
1191
  function normaliseInterceptorConfig(raw, dropped) {
@@ -543,6 +1208,25 @@ function normaliseInterceptorConfig(raw, dropped) {
543
1208
  const actionGuard = normaliseActionGuardBlock(value.actionGuard, dropped, "interceptor.actionGuard");
544
1209
  if (actionGuard)
545
1210
  out.actionGuard = actionGuard;
1211
+ // #225: conversation posture. An invalid value is DROPPED (and named in the
1212
+ // warn log) rather than coerced, so `conversationPosture()` falls back to
1213
+ // `observe` — the safe direction. A typo must never start blocking turns.
1214
+ if (value.conversation !== undefined) {
1215
+ if (value.conversation && typeof value.conversation === "object" && !Array.isArray(value.conversation)) {
1216
+ const posture = value.conversation.posture;
1217
+ if (posture !== undefined) {
1218
+ if (typeof posture === "string" && CONVERSATION_POSTURES.includes(posture)) {
1219
+ out.conversation = { posture: posture };
1220
+ }
1221
+ else {
1222
+ dropped?.push("interceptor.conversation.posture");
1223
+ }
1224
+ }
1225
+ }
1226
+ else {
1227
+ dropped?.push("interceptor.conversation");
1228
+ }
1229
+ }
546
1230
  // #115: empty/all-invalid normalises to undefined, not {} — {} is truthy
547
1231
  // and made applyPluginConfigOverride treat a no-op interceptor block as a
548
1232
  // real override, inconsistent with normaliseSeverityMap's own contract.
@@ -554,6 +1238,19 @@ function normaliseInterceptorConfig(raw, dropped) {
554
1238
  * - `interceptor` deep-merges PER KEY — per severity entry, per guard flag —
555
1239
  * so an override that sets one nested key does not wholesale-discard the
556
1240
  * base's other interceptor settings.
1241
+ * - `actionGuard.notify` and `actionGuard.broker` deep-merge per key too
1242
+ * (#226). They are the two OBJECT-valued guard keys, and a shallow spread
1243
+ * replaced them wholesale: an openclaw.json entry that says only
1244
+ * `notify: { enabled: true }` — the shape the UI writes when an operator
1245
+ * ticks "Operator Notifications" — discarded the shield config's
1246
+ * `webhookUrl` and `webhookSecret`, leaving notify ARMED with no channel and
1247
+ * no signing key. Every alert then reported "enabled but no channel is
1248
+ * configured/buildable on this host", which is the #143 silent-no-sink
1249
+ * failure in a new place.
1250
+ * - ARRAY-valued keys (`autoApprove`, `reviewedScripts`) still REPLACE. They
1251
+ * are allowlists: merging two of them would union permissions an operator
1252
+ * removed back into the effective config, which is the wrong direction for a
1253
+ * security control.
557
1254
  * - Explicit values, including `false`, always win over base values; absent
558
1255
  * keys fall through to the base.
559
1256
  * - Defaults are NOT applied here: DEFAULT_INTERCEPTOR_CONFIG only fills the
@@ -562,6 +1259,13 @@ function normaliseInterceptorConfig(raw, dropped) {
562
1259
  */
563
1260
  function mergeConfigs(base, override) {
564
1261
  const merged = { ...base, ...override };
1262
+ // Per-key, like every other nested block here. A plain spread would let an
1263
+ // openclaw.json entry that mentions `conversationTrust` at all replace the
1264
+ // shield-config block wholesale, silently reverting an opt-out set in the
1265
+ // file the operator considers authoritative.
1266
+ if (base.conversationTrust || override.conversationTrust) {
1267
+ merged.conversationTrust = { ...base.conversationTrust, ...override.conversationTrust };
1268
+ }
565
1269
  if (base.interceptor || override.interceptor) {
566
1270
  const b = base.interceptor ?? {};
567
1271
  const o = override.interceptor ?? {};
@@ -573,7 +1277,14 @@ function mergeConfigs(base, override) {
573
1277
  merged.interceptor.failurePolicy = { ...b.failurePolicy, ...o.failurePolicy };
574
1278
  }
575
1279
  if (b.actionGuard || o.actionGuard) {
576
- merged.interceptor.actionGuard = { ...b.actionGuard, ...o.actionGuard };
1280
+ const bg = b.actionGuard ?? {};
1281
+ const og = o.actionGuard ?? {};
1282
+ const guard = { ...bg, ...og };
1283
+ if (bg.notify || og.notify)
1284
+ guard.notify = { ...bg.notify, ...og.notify };
1285
+ if (bg.broker || og.broker)
1286
+ guard.broker = { ...bg.broker, ...og.broker };
1287
+ merged.interceptor.actionGuard = guard;
577
1288
  }
578
1289
  }
579
1290
  return merged;
@@ -610,8 +1321,58 @@ function applyPluginConfigOverride(api) {
610
1321
  _config = null;
611
1322
  _lastShieldConfigRef = null;
612
1323
  }
1324
+ /**
1325
+ * Load the effective config, DEGRADING rather than throwing (#226).
1326
+ *
1327
+ * `getRuntime()` resolves `shieldcortex/dist/…/runtime.mjs` by walking a list of
1328
+ * install locations, and `loadShieldConfig()` then reads a file. Both can fail
1329
+ * for ordinary reasons — the package was upgraded underneath a running gateway,
1330
+ * a global install moved, `~/.shieldcortex/config.json` is half-written.
1331
+ *
1332
+ * This used to propagate, and the propagation went somewhere bad: every caller
1333
+ * of `loadConfig` is a hook body, and in `handleBeforeAgentRun` the throw was
1334
+ * caught by the OUTER catch — the one that fails open. So a host that could not
1335
+ * load the runtime produced one console line per turn and NOTHING else: no
1336
+ * posture, no scan, no audit row, no alert. Precisely the "unprotected turn that
1337
+ * leaves no trace" the #225/#226 work exists to eliminate, reintroduced through
1338
+ * the config read rather than through the scanner.
1339
+ *
1340
+ * Now it degrades to the plugin config from openclaw.json (already normalised by
1341
+ * `applyPluginConfigOverride`), or to an empty config. The posture therefore
1342
+ * still resolves, the scan still runs, `scanRealtimeContent` reports UNAVAILABLE
1343
+ * on its own (the same runtime failure defeats the MCP fallback), and the gate
1344
+ * writes its normal audit row and raises its normal alert.
1345
+ *
1346
+ * It does NOT cache the degraded result — a later successful load must take
1347
+ * effect without a restart — and it never claims the shield config loaded: the
1348
+ * warning says exactly what is missing, and is bounded, redacted, and emitted
1349
+ * ONCE per plugin load (`__resetConfigStateForTest` re-arms it) so a per-turn
1350
+ * failure cannot become per-turn log spam.
1351
+ */
1352
+ let _shieldConfigLoadFailureLogged = false;
613
1353
  async function loadConfig() {
614
- const shieldConfigRaw = await (await getRuntime()).loadShieldConfig();
1354
+ let shieldConfigRaw;
1355
+ try {
1356
+ shieldConfigRaw = await (await getRuntime()).loadShieldConfig();
1357
+ }
1358
+ catch (err) {
1359
+ if (!_shieldConfigLoadFailureLogged) {
1360
+ _shieldConfigLoadFailureLogged = true;
1361
+ const detail = redactNotifyDetail(err instanceof Error ? err.message : String(err)).slice(0, 300);
1362
+ console.warn('[shieldcortex] ⚠️ shield config could NOT be loaded — the ShieldCortex runtime did not resolve, ' +
1363
+ `or it could not read ~/.shieldcortex/config.json (${detail}). Continuing with the openclaw.json ` +
1364
+ 'plugin config only; anything configured in the shield config file is NOT in effect, and ' +
1365
+ 'conversation scanning will report UNAVAILABLE until this is fixed. (Logged once per plugin load.)');
1366
+ }
1367
+ // A fresh object every time: `_configOverride` is module state and callers
1368
+ // must not be handed something they could mutate.
1369
+ return mergeConfigs({}, _configOverride ?? {});
1370
+ }
1371
+ // A load that succeeds after a failure re-arms the warning, so a SECOND
1372
+ // outage is reported rather than swallowed by the first one's flag. Set
1373
+ // before the cache check: a runtime that hands back the same object every
1374
+ // call would otherwise take the early return and leave the flag latched.
1375
+ _shieldConfigLoadFailureLogged = false;
615
1376
  if (_config && shieldConfigRaw === _lastShieldConfigRef)
616
1377
  return _config;
617
1378
  _lastShieldConfigRef = shieldConfigRaw;
@@ -643,33 +1404,127 @@ function parseScanResponse(response) {
643
1404
  const summary = detections ? `${risk} (${detections} detections)` : risk;
644
1405
  return { clean, summary };
645
1406
  }
1407
+ /**
1408
+ * Scan one piece of conversation content.
1409
+ *
1410
+ * The contract changed in #225 and the change is the point: an unavailable
1411
+ * scanner is now reported as `available: false, clean: false`, not as the old
1412
+ * `{ clean: true, summary: 'scan unavailable' }`. That old return was the
1413
+ * quietest bug in this file — the ORDINARY unavailable path (MCP fallback
1414
+ * returns nothing, e.g. no shieldcortex binary on PATH) manufactured a clean
1415
+ * verdict, so on any box where in-process defence failed to load, every message
1416
+ * was reported scanned-and-fine while nothing had been looked at.
1417
+ *
1418
+ * Fails OPEN — callers must not block on `available: false` — but LOUDLY: the
1419
+ * caller audits it, alerts on it, and doctor/status report the plane as
1420
+ * unavailable rather than protected.
1421
+ */
646
1422
  export async function scanRealtimeContent(text) {
647
1423
  // PRIMARY: scan in-process via the shared shieldcortex/defence module. The
648
1424
  // scan is pure (no DB handle required — scanToolResponse's audit write is
649
1425
  // guarded by isDatabaseInitialized()), so it is safe in the long-lived
650
1426
  // gateway and avoids booting a cold MCP server per message.
651
- const defenceMod = await getDefenceModule();
1427
+ let defenceMod = null;
1428
+ try {
1429
+ defenceMod = await getDefenceModule();
1430
+ }
1431
+ catch (err) {
1432
+ defenceMod = null;
1433
+ void err;
1434
+ }
652
1435
  if (defenceMod && typeof defenceMod.scanToolResponse === "function") {
653
- const scan = defenceMod.scanToolResponse("openclaw-realtime", text, "advisory");
654
- // Reproduce the historical summary contract exactly: risk level + detection
655
- // count only when the injection scan flagged something.
656
- const risk = scan.injection.clean ? "unknown" : scan.injection.riskLevel;
657
- const summary = scan.injection.clean
658
- ? risk
659
- : `${risk} (${scan.injection.detections.length} detections)`;
660
- return { clean: scan.clean, summary };
1436
+ try {
1437
+ const scan = defenceMod.scanToolResponse("openclaw-realtime", text, "advisory");
1438
+ // Reproduce the historical summary contract exactly: risk level + detection
1439
+ // count only when the injection scan flagged something.
1440
+ const risk = scan.injection.clean ? "unknown" : scan.injection.riskLevel;
1441
+ const summary = scan.injection.clean
1442
+ ? risk
1443
+ : `${risk} (${scan.injection.detections.length} detections)`;
1444
+ return { clean: scan.clean, summary, available: true };
1445
+ }
1446
+ catch (err) {
1447
+ // A scanner that THROWS is not a clean verdict either. Same treatment as
1448
+ // an absent one: unavailable, reported, never silently allowed to read as
1449
+ // protected.
1450
+ const detail = err instanceof Error ? err.message : String(err);
1451
+ return { clean: false, available: false, errored: true, error: `in-process scanner threw: ${detail}`, summary: "scan unavailable" };
1452
+ }
661
1453
  }
662
1454
  // FALLBACK: in-process defence unavailable (older install, import failed) —
663
1455
  // degrade to the MCP shell-out so scanning still happens rather than breaking.
664
- const response = await callCortex("scan_tool_response", {
665
- toolName: "openclaw-realtime",
666
- content: text,
667
- mode: "advisory",
668
- });
1456
+ let response = null;
1457
+ try {
1458
+ response = await callCortex("scan_tool_response", {
1459
+ toolName: "openclaw-realtime",
1460
+ content: text,
1461
+ mode: "advisory",
1462
+ });
1463
+ }
1464
+ catch (err) {
1465
+ const detail = err instanceof Error ? err.message : String(err);
1466
+ return { clean: false, available: false, errored: true, error: `scan fallback failed: ${detail}`, summary: "scan unavailable" };
1467
+ }
669
1468
  if (!response) {
670
- return { clean: true, summary: "scan unavailable" };
1469
+ return {
1470
+ clean: false,
1471
+ available: false,
1472
+ errored: true,
1473
+ error: 'no in-process defence module and the MCP fallback returned nothing',
1474
+ summary: 'scan unavailable',
1475
+ };
1476
+ }
1477
+ const parsed = parseScanResponse(response);
1478
+ return { ...parsed, available: true };
1479
+ }
1480
+ /**
1481
+ * `scanRealtimeContent` with a hard deadline (#226).
1482
+ *
1483
+ * Used ONLY by the `before_agent_run` gate, which the gateway awaits: an
1484
+ * unbounded scan there is an unbounded pause in front of the user's prompt. The
1485
+ * MCP fallback boots a cold server through `npx` and has been measured at ~15s,
1486
+ * so "it usually returns quickly" is not a bound.
1487
+ *
1488
+ * On expiry the result is an ordinary UNAVAILABLE verdict — fail open, audited,
1489
+ * alerted — and the error string names the deadline and NOTHING ELSE. It must
1490
+ * never quote the prompt: a timeout message is the one error a developer is
1491
+ * most likely to paste into an issue.
1492
+ *
1493
+ * The losing promise is not abandoned silently: a `.catch` is attached before
1494
+ * the race so a scan that rejects AFTER the deadline settles into a no-op
1495
+ * instead of an unhandled rejection that could take the gateway down under
1496
+ * `--unhandled-rejections=strict`.
1497
+ */
1498
+ export async function scanWithDeadline(text, timeoutMs = CONVERSATION_SCAN_MAX_MS) {
1499
+ const timedOut = {
1500
+ clean: false,
1501
+ available: false,
1502
+ errored: true,
1503
+ error: `conversation scan exceeded its ${timeoutMs}ms deadline`,
1504
+ summary: 'scan unavailable',
1505
+ };
1506
+ const scan = scanRealtimeContent(text);
1507
+ // Attached BEFORE the race, so a late rejection can never be unhandled.
1508
+ scan.catch(() => { });
1509
+ let timer;
1510
+ try {
1511
+ return await Promise.race([
1512
+ scan,
1513
+ new Promise((resolve) => {
1514
+ timer = setTimeout(() => resolve(timedOut), timeoutMs);
1515
+ // Never hold the process open on the deadline timer alone.
1516
+ timer.unref?.();
1517
+ }),
1518
+ ]);
1519
+ }
1520
+ catch (err) {
1521
+ const detail = err instanceof Error ? err.message : String(err);
1522
+ return { clean: false, available: false, errored: true, error: detail, summary: 'scan unavailable' };
1523
+ }
1524
+ finally {
1525
+ if (timer)
1526
+ clearTimeout(timer);
671
1527
  }
672
- return parseScanResponse(response);
673
1528
  }
674
1529
  // ==================== CONTENT PATTERNS ====================
675
1530
  const PATTERNS = {
@@ -725,17 +1580,52 @@ function extractUserContent(msgs) {
725
1580
  }
726
1581
  return out;
727
1582
  }
728
- const AUDIT_DIR = path.join(homedir(), ".shieldcortex", "audit");
1583
+ /** Where the realtime audit jsonl lives.
1584
+ *
1585
+ * Resolved PER CALL, and honouring `SHIELDCORTEX_AUDIT_DIR`, so a test can
1586
+ * exercise the hook end-to-end without appending fabricated "threat" rows to
1587
+ * a real box's security audit trail. Default is unchanged
1588
+ * (`~/.shieldcortex/audit`), so nothing moves for an install that does not set
1589
+ * the variable. */
1590
+ function auditDir() {
1591
+ const override = process.env.SHIELDCORTEX_AUDIT_DIR;
1592
+ if (override && override.trim())
1593
+ return override.trim();
1594
+ return path.join(homedir(), ".shieldcortex", "audit");
1595
+ }
729
1596
  const NOVELTY_CACHE_FILE = path.join(homedir(), ".shieldcortex", "openclaw-memory-cache.json");
730
1597
  const DEFAULT_NOVELTY_THRESHOLD = 0.88;
731
1598
  const DEFAULT_MAX_RECENT = 300;
732
1599
  const MIN_NOVELTY_CHARS = 40;
1600
+ /**
1601
+ * Append one row to the realtime audit jsonl. Returns WHETHER IT LANDED (#226).
1602
+ *
1603
+ * This used to swallow every failure and return void, so a caller could not
1604
+ * distinguish "the evidence is on disk" from "the disk is full / the audit dir
1605
+ * is not writable / it is a file where a directory should be". The
1606
+ * `before_agent_run` gate then wrote a decision row, got nothing back, and
1607
+ * proceeded to tell the operator — and the delivery row — that the decision was
1608
+ * recorded. A security control claiming evidence it does not have is worse than
1609
+ * one that admits the gap, because the gap is invisible in exactly the incident
1610
+ * where the log matters.
1611
+ *
1612
+ * Still never throws: a broken audit sink must not become a broken turn.
1613
+ */
733
1614
  async function auditLog(entry) {
1615
+ const dir = auditDir();
734
1616
  try {
735
- await fs.mkdir(AUDIT_DIR, { recursive: true });
736
- await fs.appendFile(path.join(AUDIT_DIR, `realtime-${new Date().toISOString().slice(0, 10)}.jsonl`), JSON.stringify(entry) + "\n");
1617
+ await fs.mkdir(dir, { recursive: true });
1618
+ await fs.appendFile(path.join(dir, `realtime-${new Date().toISOString().slice(0, 10)}.jsonl`), JSON.stringify(entry) + "\n");
1619
+ return true;
1620
+ }
1621
+ catch (err) {
1622
+ // LOUD. The detail is the failure and the directory — never the row, which
1623
+ // may carry a verdict summary, and never a credential (nothing in this path
1624
+ // holds one). Bounded so a pathological error message cannot flood stderr.
1625
+ const detail = err instanceof Error ? err.message : String(err);
1626
+ console.error(`[shieldcortex] ⚠️ AUDIT WRITE FAILED (${dir}) — this event is NOT on disk: ${detail.slice(0, 300)}`);
1627
+ return false;
737
1628
  }
738
- catch { }
739
1629
  }
740
1630
  function normalizeMemoryText(text) {
741
1631
  return String(text || "")
@@ -873,6 +1763,17 @@ function isInternalContent(text) {
873
1763
  // itself stays non-blocking.
874
1764
  export async function scanLlmInput(event, _ctx) {
875
1765
  try {
1766
+ // #226: THE POSTURE GOVERNS THIS HOOK TOO. `off` means "do not scan the
1767
+ // conversation at all", and this observation hook used to ignore it
1768
+ // entirely: it scanned every prompt, wrote threat and scan_unavailable rows,
1769
+ // and forwarded detections to the cloud on a box whose operator had
1770
+ // explicitly turned conversation inspection OFF. The gate honoured the
1771
+ // setting, so a reader of `handleBeforeAgentRun` would conclude the product
1772
+ // did too. Read it FIRST, before any scanner, any audit row and any cloud
1773
+ // call, so `off` costs exactly one config read.
1774
+ const cfg = await loadConfig();
1775
+ if (conversationPosture(cfg.interceptor?.conversation) === 'off')
1776
+ return;
876
1777
  // Only scan user content, skip system/boot/heartbeat prompts
877
1778
  // Trust is resolved per TURN, not per message: the host tells us who sent
878
1779
  // this turn, but history messages carry no individual attribution, so there
@@ -884,10 +1785,13 @@ export async function scanLlmInput(event, _ctx) {
884
1785
  let trustMemo = null;
885
1786
  const resolveTrust = async () => {
886
1787
  if (!trustMemo) {
1788
+ // `cfg` is already in hand from the posture read above, so this costs
1789
+ // no I/O. It also reads the PARSED field rather than casting the config
1790
+ // object — the cast this replaces asserted a key that normaliseConfig
1791
+ // drops, so it was always undefined. See SCConfig.conversationTrust.
887
1792
  trustMemo = classifyConversationOrigin({
888
1793
  senderIsOwner: event.senderIsOwner,
889
- trustOwnerInput: (await loadConfig())
890
- ?.conversationTrust?.trustOwnerInput,
1794
+ trustOwnerInput: cfg.conversationTrust?.trustOwnerInput,
891
1795
  });
892
1796
  }
893
1797
  return trustMemo;
@@ -898,6 +1802,33 @@ export async function scanLlmInput(event, _ctx) {
898
1802
  if (!text || text.length < 10)
899
1803
  continue;
900
1804
  const result = await scanRealtimeContent(text);
1805
+ // #225: "we could not look" is its own outcome. Before this branch the
1806
+ // unavailable path returned clean:true and this loop did nothing at all —
1807
+ // an unscanned message was indistinguishable from a scanned one, on the
1808
+ // observation hook as well as the gate.
1809
+ if (!result.available) {
1810
+ // #226: redacted on the CONSOLE too, not only in the row. The reason
1811
+ // comes from a transport/scanner failure string, which can name the
1812
+ // endpoint it failed to reach — and a gateway's stdout is routinely
1813
+ // shipped to a log aggregator, so "ephemeral" is not a property the
1814
+ // console actually has.
1815
+ const detail = redactNotifyDetail(result.error ?? result.summary);
1816
+ console.warn(`[shieldcortex] ⚠️ conversation scan UNAVAILABLE (${detail}) — this message was NOT scanned`);
1817
+ // AWAITED. The whole function is already fire-and-forget from
1818
+ // `handleLlmInput`, so this blocks nothing the gateway is waiting on —
1819
+ // and it means the row is on disk before the loop moves to the next
1820
+ // message, and that `auditLog`'s new boolean (which logs loudly on
1821
+ // failure) is actually reached rather than discarded into a floating
1822
+ // promise.
1823
+ await auditLog({
1824
+ type: 'scan_unavailable', hook: 'llm_input', sessionId: event.sessionId,
1825
+ model: event.model, reason: detail,
1826
+ chars: text.length,
1827
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
1828
+ ts: new Date().toISOString(),
1829
+ });
1830
+ continue;
1831
+ }
901
1832
  if (!result.clean) {
902
1833
  const trust = await resolveTrust();
903
1834
  console.warn(`[shieldcortex] ⚠️ Threat in LLM input: ${result.summary} [${trust.origin}]`);
@@ -907,7 +1838,7 @@ export async function scanLlmInput(event, _ctx) {
907
1838
  // the Action Guard tighten by one notch for a bounded window.
908
1839
  //
909
1840
  // Source trust gates the CONSEQUENCE, never the detection: the warn and
910
- // the audit row above happen whoever sent this. What trust decides is
1841
+ // the audit row below happen whoever sent this. What trust decides is
911
1842
  // whether it may tighten the guard. The operator typing "delete the old
912
1843
  // logs" is an instruction, and treating it as an attack is the false
913
1844
  // alarm that gets a control switched off. Everything the agent was
@@ -915,17 +1846,28 @@ export async function scanLlmInput(event, _ctx) {
915
1846
  if (trust.mayTaint) {
916
1847
  sessionTaint.mark(event.sessionId, { reason: `conversation scan: ${result.summary}` });
917
1848
  }
1849
+ // #226: NO `preview`. This row carried the first 100 characters of the
1850
+ // prompt — the exact text that tripped an injection detector, i.e.
1851
+ // hostile by assumption — into an append-only file that syncs. The gate
1852
+ // on the very next hook has recorded only `chars` + `contentSha256`
1853
+ // since #225 and says in its own comment that the prompt is never
1854
+ // persisted; the observation hook quietly did the opposite, so the
1855
+ // claim was false on the path that runs on every single turn. Length
1856
+ // plus digest keeps the row correlatable with the gate's row for the
1857
+ // same text without storing the text.
918
1858
  const entry = {
919
1859
  type: "threat", hook: "llm_input", sessionId: event.sessionId,
920
1860
  model: event.model, reason: result.summary,
921
- preview: text.slice(0, 100), ts: new Date().toISOString(),
1861
+ chars: text.length,
1862
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
1863
+ ts: new Date().toISOString(),
922
1864
  };
923
- auditLog(entry);
1865
+ await auditLog(entry);
924
1866
  loadConfig()
925
1867
  // Pass the local entry as-is; cloudSync rebuilds a canonical metadata-only
926
- // entry from named fields and never reads preview/content. No raw LLM input
1868
+ // entry from named fields and never reads content. No raw LLM input
927
1869
  // leaves here.
928
- .then(cfg => cloudSync(entry, cfg))
1870
+ .then(cfg2 => cloudSync(entry, cfg2))
929
1871
  .catch(() => { });
930
1872
  }
931
1873
  }
@@ -935,9 +1877,625 @@ export async function scanLlmInput(event, _ctx) {
935
1877
  }
936
1878
  }
937
1879
  function handleLlmInput(event, ctx) {
938
- // Fire and forget
1880
+ // Fire and forget — OBSERVATION ONLY. This hook cannot block (#225); the
1881
+ // enforcement point is handleBeforeAgentRun below.
939
1882
  void scanLlmInput(event, ctx);
940
1883
  }
1884
+ /**
1885
+ * Route a conversation-threat detection to a HUMAN (#225).
1886
+ *
1887
+ * This is the "sink" the issue is named for. Before it existed, a HIGH verdict
1888
+ * produced a console line and an audit row, and a real detection on a live box
1889
+ * was seen by nobody. It reuses the Action Guard's notify transport (#143)
1890
+ * rather than inventing a second one — two notification paths would drift, and
1891
+ * the operator would learn which one to ignore.
1892
+ *
1893
+ * Returns whether a human was actually reached, so callers (and doctor) can
1894
+ * report "detected but undeliverable" instead of implying someone was told.
1895
+ * Never throws: a failed notification must not become a failed turn.
1896
+ */
1897
+ /** The longest the conversation gate will wait for an operator alert to be
1898
+ * handed to a transport. See the call site: the user's turn is blocked on this
1899
+ * hook, so the alert's deadline has to be a fraction of the hook's. */
1900
+ const CONVERSATION_NOTIFY_MAX_MS = 5_000;
1901
+ /**
1902
+ * The longest the conversation gate will wait for a SCAN (#226).
1903
+ *
1904
+ * `before_agent_run` is awaited by the gateway — the user's turn is stopped
1905
+ * dead until this handler returns — and the scan's fallback path is an MCP
1906
+ * shell-out that boots a cold server via `npx`, which can take upwards of 15s.
1907
+ * That is half the hook's entire 30s budget before an alert and two audit
1908
+ * writes are added behind it, and it happens on exactly the hosts where the
1909
+ * in-process defence module failed to load: the ones already degraded.
1910
+ *
1911
+ * Past this deadline the scan is treated as UNAVAILABLE, which fails OPEN (the
1912
+ * turn proceeds) and is audited and alerted like any other unavailable scan. A
1913
+ * security control that silently adds fifteen seconds to every prompt is one an
1914
+ * operator uninstalls.
1915
+ */
1916
+ export const CONVERSATION_SCAN_MAX_MS = 5_000;
1917
+ /**
1918
+ * Repeat-alert suppression for the scan-unavailable path (#226).
1919
+ *
1920
+ * An unavailable scanner is not a transient event: it is usually a missing
1921
+ * install, a broken defence build or an absent binary, and it recurs on EVERY
1922
+ * turn. Alerting per turn turns the operator's phone into a metronome and the
1923
+ * alert into something they mute — which is the same outcome as never sending
1924
+ * one, reached by a more expensive route. First occurrence goes immediately;
1925
+ * after that, at most one alert per window, with the suppressed count carried
1926
+ * on the next alert that does go out so nothing is lost.
1927
+ *
1928
+ * THE WINDOW IS PER SESSION, not per process — see `noteScanUnavailable`. A
1929
+ * gateway runs many sessions at once, and "we already told you about session A"
1930
+ * is not a reason to stay silent about session B.
1931
+ *
1932
+ * AUDITING IS NOT RATE LIMITED. Every occurrence still writes its row, and the
1933
+ * row records whether an alert was suppressed and how many have been seen —
1934
+ * the evidence trail must be complete even when the notification stream is not.
1935
+ */
1936
+ export const SCAN_UNAVAILABLE_ALERT_WINDOW_MS = 5 * 60_000;
1937
+ /**
1938
+ * The bucket used when the host hands us no session identity at all.
1939
+ *
1940
+ * `before_agent_run`'s context declares `sessionId` and `sessionKey` as
1941
+ * OPTIONAL, so both can be absent. Keying such an occurrence under a fixed
1942
+ * fallback keeps rate limiting working exactly as it did on a single-session
1943
+ * host, and — critically — keeps a nameless occurrence from sharing a bucket
1944
+ * with a NAMED one, which is what a `String(undefined)` key would have done.
1945
+ */
1946
+ const SCAN_UNAVAILABLE_FALLBACK_SESSION = '__unkeyed-session__';
1947
+ /**
1948
+ * Cap on distinct sessions tracked at once.
1949
+ *
1950
+ * `session_end` is what normally frees an entry, and it is not guaranteed: an
1951
+ * older host may not emit it, and a crashed session never will. This map is
1952
+ * therefore bounded and evicts least-recently-seen first. Overshooting the cap
1953
+ * costs at most one extra alert for the evicted session — the safe direction,
1954
+ * since the failure mode of eviction is "tell the operator again", not "stay
1955
+ * quiet". Each entry is four numbers and a short key, so 512 of them is a few
1956
+ * kilobytes in a process that already holds a scanner.
1957
+ */
1958
+ export const SCAN_UNAVAILABLE_MAX_SESSIONS = 512;
1959
+ const _scanUnavailable = new Map();
1960
+ /** Normalise whatever the host gave us into a map key. */
1961
+ function scanUnavailableSessionKey(sessionKey) {
1962
+ const trimmed = typeof sessionKey === 'string' ? sessionKey.trim() : '';
1963
+ return trimmed === '' ? SCAN_UNAVAILABLE_FALLBACK_SESSION : trimmed;
1964
+ }
1965
+ /**
1966
+ * Should this scan-unavailable occurrence raise an operator alert?
1967
+ *
1968
+ * PER SESSION (#226). The first cut kept one module-global counter, which on a
1969
+ * gateway — a process that multiplexes every channel and every concurrent
1970
+ * agent — meant one session's broken scanner silenced the FIRST failure of
1971
+ * every other session for the next five minutes. That is the same class of bug
1972
+ * the rate limit exists to avoid, inverted: instead of too many alerts, a real
1973
+ * new failure is never reported at all. Suppression is a property of one
1974
+ * session's repeating failure, so the state is keyed by one session.
1975
+ *
1976
+ * Pure apart from the per-session counter it advances, and driven by an
1977
+ * injectable `now` so the window is testable without sleeping. Exported for the
1978
+ * regression test; not part of the plugin's host-facing surface.
1979
+ *
1980
+ * The key is a session id, never logged: it reaches this function only to index
1981
+ * the map. The audit rows that carry `sessionId` are the deliberate place that
1982
+ * fact is recorded.
1983
+ */
1984
+ export function noteScanUnavailable(sessionKey, nowMs = Date.now()) {
1985
+ const key = scanUnavailableSessionKey(sessionKey);
1986
+ let state = _scanUnavailable.get(key);
1987
+ if (!state) {
1988
+ evictScanUnavailableOverflow(nowMs);
1989
+ state = { count: 0, lastAlertAtMs: null, suppressedSinceAlert: 0, lastSeenAtMs: nowMs };
1990
+ _scanUnavailable.set(key, state);
1991
+ }
1992
+ state.count += 1;
1993
+ state.lastSeenAtMs = nowMs;
1994
+ const last = state.lastAlertAtMs;
1995
+ // A clock that jumped BACKWARDS (NTP step, suspend/resume) must not be able
1996
+ // to wedge alerting off forever: treat a negative elapsed as "window over".
1997
+ const elapsed = last === null ? Infinity : nowMs - last;
1998
+ if (last === null || elapsed >= SCAN_UNAVAILABLE_ALERT_WINDOW_MS || elapsed < 0) {
1999
+ const suppressed = state.suppressedSinceAlert;
2000
+ state.lastAlertAtMs = nowMs;
2001
+ state.suppressedSinceAlert = 0;
2002
+ return { alert: true, count: state.count, suppressedSinceLastAlert: suppressed };
2003
+ }
2004
+ state.suppressedSinceAlert += 1;
2005
+ return {
2006
+ alert: false,
2007
+ count: state.count,
2008
+ suppressedSinceLastAlert: state.suppressedSinceAlert,
2009
+ };
2010
+ }
2011
+ /** Keep the session map bounded when `session_end` never arrives. Evicts the
2012
+ * least-recently-seen entries; an evicted session simply alerts once more. */
2013
+ function evictScanUnavailableOverflow(nowMs) {
2014
+ if (_scanUnavailable.size < SCAN_UNAVAILABLE_MAX_SESSIONS)
2015
+ return;
2016
+ const oldestFirst = [..._scanUnavailable.entries()].sort((a, b) => (a[1].lastSeenAtMs ?? nowMs) - (b[1].lastSeenAtMs ?? nowMs));
2017
+ const drop = _scanUnavailable.size - SCAN_UNAVAILABLE_MAX_SESSIONS + 1;
2018
+ for (const [key] of oldestFirst.slice(0, drop))
2019
+ _scanUnavailable.delete(key);
2020
+ }
2021
+ /**
2022
+ * Forget ONE session's suppression window. Called from `session_end`, so a
2023
+ * long-lived gateway does not carry a finished session's suppression into a
2024
+ * reused id — and, just as importantly, does not clear anyone ELSE's.
2025
+ *
2026
+ * A `session_end` that names no session clears the fallback bucket only: on a
2027
+ * host that supplies no session identity every occurrence lands there, so that
2028
+ * is precisely the state that ended.
2029
+ */
2030
+ export function resetScanUnavailableAlertState(sessionKey) {
2031
+ _scanUnavailable.delete(scanUnavailableSessionKey(sessionKey));
2032
+ }
2033
+ /** Test/reset seam: forget EVERY session. Production never wants this — one
2034
+ * session ending must not re-arm alerting for the others — so it is reachable
2035
+ * only from `__resetConfigStateForTest`. */
2036
+ export function __resetScanUnavailableAlertState() {
2037
+ _scanUnavailable.clear();
2038
+ }
2039
+ /**
2040
+ * Make a failure detail safe to PERSIST, SEND or PRINT (#226).
2041
+ *
2042
+ * The detail is assembled from channel names, scanner errors and whatever a
2043
+ * transport said went wrong. Node's fetch failures do not name the URL, but a
2044
+ * transport is free to put one in its reason — and a notify webhook URL
2045
+ * routinely carries a token in its path or query
2046
+ * (`https://hooks.example/services/T0/B0/XXXXXXXX`). So any http(s) URL is
2047
+ * reduced to its origin: enough to tell WHICH endpoint failed, not enough to
2048
+ * replay a request to it. Bounded too, so a transport that returns a page of
2049
+ * HTML cannot bloat the log.
2050
+ *
2051
+ * EVERY sink gets the redacted string — not just the ones that obviously
2052
+ * outlive the process. The audit row is append-only and syncs; the notification
2053
+ * leaves the box; and the console is NOT the ephemeral thing an earlier version
2054
+ * of this comment claimed it was, because a gateway's stdout is routinely
2055
+ * shipped to a log aggregator and kept longer than the audit file. Redacting
2056
+ * for the row and not for the other two protected the least exposed of the
2057
+ * three.
2058
+ */
2059
+ export function redactNotifyDetail(detail) {
2060
+ const withoutUrls = String(detail ?? '').replace(/https?:\/\/[^\s'"]+/gi, (url) => {
2061
+ try {
2062
+ return `${new URL(url).origin}/…`;
2063
+ }
2064
+ catch {
2065
+ return '<url>';
2066
+ }
2067
+ });
2068
+ return withoutUrls.length > 500 ? `${withoutUrls.slice(0, 499)}…` : withoutUrls;
2069
+ }
2070
+ /**
2071
+ * The seam a gateway MIGHT offer for sending an operator a message, captured at
2072
+ * register() time if the API exposes it.
2073
+ *
2074
+ * No OpenClaw build we have inspected exposes it — neither 2026.5.2 nor
2075
+ * 2026.7.1 has a `notifyOperator` anywhere in its plugin API — so in practice
2076
+ * the webhook is the load-bearing channel and this stays null. It is read
2077
+ * structurally rather than removed because #143's design intent was that on
2078
+ * OpenClaw the transport should use the gateway's own message capability, and
2079
+ * that only becomes true if the code is ready for the day it appears. Nothing
2080
+ * here should be read as "ShieldCortex delivers natively on OpenClaw today".
2081
+ */
2082
+ let _gatewayNotifyContext = null;
2083
+ export function __setGatewayNotifyContextForTest(ctx) {
2084
+ _gatewayNotifyContext = ctx;
2085
+ }
2086
+ /**
2087
+ * Route a conversation-firewall detection to a HUMAN (#225).
2088
+ *
2089
+ * This is the "sink" the issue is named for: before it existed, a HIGH verdict
2090
+ * produced a console line and an audit row, and a real detection on a live box
2091
+ * was seen by nobody.
2092
+ *
2093
+ * It reuses the Action Guard's #143 transport rather than inventing a second
2094
+ * one — but *correctly*, which the first cut did not:
2095
+ *
2096
+ * - the notification is built by the main package's
2097
+ * `buildConversationThreatNotification`, so it is a real, bounded
2098
+ * notification with its own event discriminator, NOT an ad-hoc
2099
+ * `{kind, severity, …}` literal cast through `NotifyChannel.send`. A
2100
+ * conversation alert therefore cannot render Approve/Deny controls or a
2101
+ * hash that does not exist — the fields simply are not on the type.
2102
+ * - delivery goes through `deliverOperatorNotification`, the same core the
2103
+ * approval path uses, so the deadline, the malformed-result handling and
2104
+ * the "nothing but the boolean is read back" rule are shared, not copied.
2105
+ * - the webhook secret is read from `webhookSecret` — the field
2106
+ * `normaliseNotifyConfig` actually returns. Mirroring it as `secret`
2107
+ * silently produced UNSIGNED POSTs.
2108
+ * - both channels are offered where the runtime provides them: the gateway's
2109
+ * own message seam first WHERE IT EXISTS (no build we have inspected
2110
+ * exposes one — see `_gatewayNotifyContext`), then the configured webhook,
2111
+ * which is what actually carries an alert off the box today.
2112
+ *
2113
+ * Returns what happened, and NEVER throws: a failed notification must not
2114
+ * become a failed turn.
2115
+ */
2116
+ export async function notifyOperatorOfConversationThreat(input) {
2117
+ try {
2118
+ const mod = await getDefenceModule();
2119
+ const cfg = await loadConfig();
2120
+ const raw = cfg.interceptor?.actionGuard?.notify;
2121
+ if (!raw)
2122
+ return { configured: false, delivered: false, via: null, detail: 'no notify config' };
2123
+ if (typeof mod?.normaliseNotifyConfig !== 'function') {
2124
+ return { configured: true, delivered: false, via: null, detail: 'installed shieldcortex build has no notify transport' };
2125
+ }
2126
+ const notify = mod.normaliseNotifyConfig(raw);
2127
+ if (!notify.enabled)
2128
+ return { configured: false, delivered: false, via: null, detail: 'notify disabled' };
2129
+ const channels = [];
2130
+ // The gateway's own message seam, WHERE the runtime provides one. It would
2131
+ // go first, because it would reach the operator on a channel they already
2132
+ // read — but `_gatewayNotifyContext` is null on every build we have
2133
+ // inspected, so in practice this list starts at the webhook below.
2134
+ if (notify.openclaw === true && _gatewayNotifyContext) {
2135
+ const gatewayChannel = createGatewayNotifyChannel(_gatewayNotifyContext);
2136
+ if (gatewayChannel)
2137
+ channels.push(gatewayChannel);
2138
+ }
2139
+ if (notify.webhookUrl && typeof mod.createWebhookNotifyChannel === 'function') {
2140
+ channels.push(mod.createWebhookNotifyChannel({
2141
+ url: notify.webhookUrl,
2142
+ // The signing key. Passed straight through and never logged — see
2143
+ // notify-config.ts, which is the only place this value is parsed.
2144
+ secret: notify.webhookSecret,
2145
+ }));
2146
+ }
2147
+ if (channels.length === 0) {
2148
+ return { configured: true, delivered: false, via: null, detail: 'notify enabled but no channel is configured/buildable on this host' };
2149
+ }
2150
+ const notification = typeof mod.buildConversationThreatNotification === 'function'
2151
+ ? mod.buildConversationThreatNotification({
2152
+ outcome: input.outcome,
2153
+ posture: input.posture,
2154
+ summary: input.summary,
2155
+ reason: input.reason,
2156
+ sessionId: input.sessionId,
2157
+ model: input.model,
2158
+ host: hostname(),
2159
+ detectedAt: new Date().toISOString(),
2160
+ })
2161
+ : null;
2162
+ if (!notification) {
2163
+ // An older dist has the transport but not this event. Sending the
2164
+ // approval-shaped payload instead would put an Approve button on an alert
2165
+ // with nothing behind it — refuse, and say why.
2166
+ return {
2167
+ configured: true,
2168
+ delivered: false,
2169
+ via: null,
2170
+ detail: 'installed shieldcortex build predates the conversation-threat notification — refusing to send an approval-shaped alert',
2171
+ };
2172
+ }
2173
+ if (typeof mod.deliverOperatorNotification !== 'function') {
2174
+ return { configured: true, delivered: false, via: null, detail: 'installed shieldcortex build has no notification delivery core' };
2175
+ }
2176
+ const result = await mod.deliverOperatorNotification(notification, {
2177
+ channels,
2178
+ // Bounded HARDER than the transport's own configured deadline, because
2179
+ // this call sits inside a gate the gateway awaits: the user's turn is
2180
+ // waiting on it. The hook is registered with a 30s timeout, and a gate
2181
+ // that exceeds its own timeout is a security control that fails in a way
2182
+ // nobody has reasoned about. The alert is already in the log and the
2183
+ // audit row by this point, so what a longer wait buys is one extra retry
2184
+ // window on a transport that is, by then, visibly unhealthy.
2185
+ timeoutMs: Math.min(notify.timeoutMs ?? CONVERSATION_NOTIFY_MAX_MS, CONVERSATION_NOTIFY_MAX_MS),
2186
+ });
2187
+ const failures = result.attempts
2188
+ .filter((a) => !a.result.delivered)
2189
+ .map((a) => `${a.channel}: ${a.result.reason ?? 'failed'}`)
2190
+ .join('; ');
2191
+ return {
2192
+ configured: true,
2193
+ delivered: result.deliveredVia !== null,
2194
+ via: result.deliveredVia,
2195
+ detail: result.deliveredVia ? `delivered via ${result.deliveredVia}` : `undeliverable — ${failures || 'no channel accepted it'}`,
2196
+ };
2197
+ }
2198
+ catch (err) {
2199
+ return {
2200
+ configured: true,
2201
+ delivered: false,
2202
+ via: null,
2203
+ detail: `notify error: ${err instanceof Error ? err.message : String(err)}`,
2204
+ };
2205
+ }
2206
+ }
2207
+ /**
2208
+ * The gate's allow answer, stated explicitly (#226).
2209
+ *
2210
+ * A fresh literal per call, not a shared constant: the runner passes whatever
2211
+ * we return into its own merge/normalise chain, and a frozen singleton handed
2212
+ * to a host that decides to annotate it would fail in a way this plugin cannot
2213
+ * see. It costs one object per turn.
2214
+ *
2215
+ * WHY EXPLICIT, when the SDK types the result `InputGateDecision | void` and
2216
+ * this handler previously returned `undefined` on every allow path:
2217
+ *
2218
+ * The 2026.7.1-2 runner contradicts itself about void, one guard deep.
2219
+ * `runBeforeAgentRun`'s doc comment says "Handlers that return void are treated
2220
+ * as pass", and its `mergeResults` body opens with
2221
+ *
2222
+ * if (next === void 0 || next === null) → { outcome: "block",
2223
+ * reason: "…invalid decision" }
2224
+ *
2225
+ * i.e. the merge is written to BLOCK on void. What saves an `undefined` return
2226
+ * today is only that `runModifyingHook` never calls the merge for it —
2227
+ * `if (handlerResult !== void 0 && (handlerResult !== null || mergeNullResults))`
2228
+ * — so the void branch inside the merge is unreachable dead code, while `null`,
2229
+ * the sibling value that same line treats identically, reaches it and DOES
2230
+ * block. Verified by executing the real 2026.7.1-2 runner: `undefined` → pass,
2231
+ * `null` → block/invalid, `{outcome:'pass'}` → pass.
2232
+ *
2233
+ * So void is not broken here — it is correct by one guard, against a merge
2234
+ * function whose stated intent is to reject it. `{ outcome: 'pass' }` is
2235
+ * correct under BOTH readings, and is the shape the host validates rather than
2236
+ * the shape it happens to skip. That is the difference worth having in front of
2237
+ * every user turn.
2238
+ */
2239
+ function gatePass() {
2240
+ return { outcome: 'pass' };
2241
+ }
2242
+ /**
2243
+ * The conversation firewall's enforcement point (#225).
2244
+ *
2245
+ * Unlike `llm_input`, this hook is awaited by the gateway and its return value
2246
+ * decides whether the run proceeds. It scans the prompt, applies the configured
2247
+ * posture, and — critically — routes a detection to a HUMAN rather than only to
2248
+ * a log file. The finding this fixes was that a HIGH verdict on a live box was
2249
+ * seen by nobody.
2250
+ *
2251
+ * Fails OPEN on any internal error: a security plugin that bricks the gateway
2252
+ * has caused a worse outage than the one it prevents. Every failure is reported.
2253
+ *
2254
+ * EVERY path returns a decision — `gatePass()` to allow, `{ outcome: 'block' }`
2255
+ * only for a dirty verdict under `enforce`. Nothing returns `undefined`; see
2256
+ * `gatePass` for the host-contract reason. "Fails open" therefore now means an
2257
+ * explicit pass, which is a stronger statement than the absence of an answer:
2258
+ * it is the same word said in the vocabulary the host validates.
2259
+ */
2260
+ export async function handleBeforeAgentRun(event, ctx) {
2261
+ let posture = 'observe';
2262
+ try {
2263
+ const cfg = await loadConfig();
2264
+ posture = conversationPosture(cfg.interceptor?.conversation);
2265
+ if (posture === 'off')
2266
+ return gatePass();
2267
+ const text = String(event?.prompt ?? '');
2268
+ if (!text || text.length < 10 || isInternalContent(text))
2269
+ return gatePass();
2270
+ // sessionId/model come off the hook CONTEXT (PluginHookAgentContext); the
2271
+ // event carries neither. Both are optional there too, so both may be absent.
2272
+ const sessionId = ctx?.sessionId ?? ctx?.sessionKey;
2273
+ const model = ctx?.modelId;
2274
+ // scanRealtimeContent no longer throws on the paths that used to (it
2275
+ // reports `available:false` instead), but a defensive catch stays: this
2276
+ // function's contract is that nothing here can stop a turn by accident.
2277
+ // #226: BOUNDED. The gateway awaits this hook, so an unbounded scan is an
2278
+ // unbounded pause in front of the user's prompt — see scanWithDeadline.
2279
+ let scan;
2280
+ try {
2281
+ scan = await scanWithDeadline(text);
2282
+ }
2283
+ catch (err) {
2284
+ const detail = err instanceof Error ? err.message : String(err);
2285
+ scan = { clean: false, available: false, errored: true, error: detail, summary: 'scan unavailable' };
2286
+ }
2287
+ // #235: WHO sent this turn, resolved before the verdict is applied.
2288
+ // `senderIsOwner` was declared on the event and read by nothing, so the
2289
+ // enforce path could block the operator's own paste and destroy it. The
2290
+ // config is already loaded, so this is a pure call — no second read.
2291
+ const trust = classifyConversationOrigin({
2292
+ senderIsOwner: event?.senderIsOwner,
2293
+ trustOwnerInput: cfg.conversationTrust?.trustOwnerInput,
2294
+ });
2295
+ const decision = evaluateConversationRun(posture, scan, trust);
2296
+ // #226: REDACT ONCE, then use the redacted string everywhere the reason
2297
+ // goes — the persisted decision row, the outbound notification, the block
2298
+ // reason, the console line. On the unavailable path `decision.reason`
2299
+ // embeds the scanner's own failure string verbatim
2300
+ // (`conversation scan unavailable (${scan.error})`), and that string is
2301
+ // assembled from a transport error: a cold MCP start, a fetch, a defence
2302
+ // build download. Any of those can name the endpoint it failed to reach,
2303
+ // and such a URL routinely carries a credential in its path
2304
+ // (`https://hooks.example/services/T0/B0/XXXX`). The console line was
2305
+ // already redacted while the row and the alert — the two that PERSIST and
2306
+ // LEAVE THE BOX — were not, which had the guarantee exactly backwards.
2307
+ const safeReason = decision.reason === null ? null : redactNotifyDetail(decision.reason);
2308
+ // #226: repeated unavailability alerts at most once per window. Called here
2309
+ // rather than at the notify site so the COUNTERS advance on every
2310
+ // occurrence, and the decision row can record what was suppressed even when
2311
+ // no alert goes out.
2312
+ // Keyed by SESSION: a broken scanner in one session must not silence the
2313
+ // first report of a broken scanner in another. `sessionId` may be absent —
2314
+ // noteScanUnavailable buckets that case separately rather than letting one
2315
+ // nameless session stand in for all of them.
2316
+ const unavailable = decision.outcome === 'unavailable';
2317
+ const alertGate = unavailable ? noteScanUnavailable(sessionId) : null;
2318
+ const suppressAlert = alertGate !== null && !alertGate.alert;
2319
+ // ── EVIDENCE FIRST, SIDE EFFECT SECOND ────────────────────────────────
2320
+ //
2321
+ // The decision row is a LOCAL append and it goes to disk before anything
2322
+ // leaves this box. The previous order awaited an external notification and
2323
+ // then wrote the row — under a comment claiming the row already existed —
2324
+ // so every way that call can end badly took the evidence with it: a
2325
+ // notification channel that hangs until the gateway's 30s hook timeout
2326
+ // fires, a transport that throws past its own catch, an operator restarting
2327
+ // the gateway mid-alert, the process dying. In each case the block or the
2328
+ // detection HAPPENED and there is no record that it did. That inverts the
2329
+ // whole point: a security control's own log must not be contingent on an
2330
+ // unrelated network round trip succeeding.
2331
+ //
2332
+ // The row carries a stable `eventId`, so the delivery row appended after
2333
+ // the attempt below can be joined to it without either row having to
2334
+ // predict the other's outcome.
2335
+ const eventId = randomUUID();
2336
+ // #226: whether the decision row ACTUALLY LANDED. `auditLog` used to
2337
+ // swallow its failures and return void, so the code below could not tell an
2338
+ // append from a silent no-op and every downstream statement — the operator
2339
+ // alert, the delivery row, this function's own comments — asserted that
2340
+ // evidence existed. Now the boolean is carried, said out loud on stderr,
2341
+ // and attached to the alert as a bounded, secret-free fact.
2342
+ let decisionRowPersisted = true;
2343
+ if (decision.audit) {
2344
+ // AWAITED: the row that says what was decided must exist before the
2345
+ // decision is handed back, and its success or failure must be READ. The
2346
+ // write is a bounded local append wrapped in its own try/catch.
2347
+ decisionRowPersisted = await auditLog({
2348
+ type: decision.outcome === 'unavailable' ? 'scan_unavailable' : 'threat',
2349
+ hook: 'before_agent_run',
2350
+ eventId,
2351
+ sessionId,
2352
+ model,
2353
+ // The REDACTED reason. This row is appended to a file that syncs.
2354
+ reason: safeReason,
2355
+ posture,
2356
+ outcome: decision.outcome,
2357
+ // #235: the origin, on every conversation decision row. Without it an
2358
+ // operator auditing an `enforce` host cannot tell a turn that was not
2359
+ // blocked because it was clean from one that was not blocked because
2360
+ // the owner sent it — and "why did this not block?" is the question
2361
+ // this row exists to answer. A label ('owner'/'non-owner'/'unknown'),
2362
+ // never a sender id: the row syncs.
2363
+ origin: trust.origin,
2364
+ // The verdict summary, never the prompt. The input that trips an
2365
+ // injection detector is hostile text by assumption; copying it into an
2366
+ // audit row that syncs to the dashboard/cloud would carry the payload
2367
+ // one hop further. A length + digest keeps rows correlatable without
2368
+ // storing the content.
2369
+ verdict: scan.summary,
2370
+ chars: text.length,
2371
+ contentSha256: createHash('sha256').update(text).digest('hex').slice(0, 16),
2372
+ // Deliberately NOT `notified: false`. Nothing has been attempted yet,
2373
+ // and a false here would read as "we tried and failed". The attempt's
2374
+ // result is its own row, keyed by this eventId.
2375
+ notifyPending: decision.notify && !suppressAlert,
2376
+ // #226: the unavailability run-length, on EVERY occurrence. Alerting is
2377
+ // rate limited; auditing is not, so the row is where the true count
2378
+ // lives — and it says explicitly when an alert was withheld, so a gap in
2379
+ // the alert stream can never be mistaken for a gap in the failures.
2380
+ ...(alertGate
2381
+ ? {
2382
+ unavailableCount: alertGate.count,
2383
+ alertSuppressed: suppressAlert,
2384
+ alertSuppressedSinceLastAlert: alertGate.suppressedSinceLastAlert,
2385
+ }
2386
+ : {}),
2387
+ ts: new Date().toISOString(),
2388
+ });
2389
+ if (!decisionRowPersisted) {
2390
+ console.error(`[shieldcortex] ⚠️ conversation ${decision.outcome} decision could NOT be written to the audit log — ` +
2391
+ 'the decision itself still stands, but there is no local record of it. Check the audit directory ' +
2392
+ '(SHIELDCORTEX_AUDIT_DIR or ~/.shieldcortex/audit) for permissions or disk space.');
2393
+ }
2394
+ }
2395
+ // The sink. Awaited — the first cut fired this off with `void` and threw
2396
+ // the delivery boolean away, so the code could not tell "a human was told"
2397
+ // from "nothing left this box". It is bounded (CONVERSATION_NOTIFY_MAX_MS,
2398
+ // well under the hook's own 30s timeout) and never throws.
2399
+ let notifyResult = null;
2400
+ if (decision.notify) {
2401
+ const label = decision.outcome === 'unavailable' ? 'unavailable' : decision.block ? 'blocked' : 'observed';
2402
+ // The same redacted string the row got. A gateway's stdout is routinely
2403
+ // shipped to a log aggregator, so "ephemeral" is not a property the
2404
+ // console actually has either.
2405
+ console.warn(`[shieldcortex] ⚠️ ${safeReason ?? 'conversation threat'} — posture=${posture}, outcome=${label}` +
2406
+ (suppressAlert
2407
+ ? ` (operator alert SUPPRESSED — ${alertGate?.suppressedSinceLastAlert} since the last one; ${alertGate?.count} this session)`
2408
+ : ''));
2409
+ // Audited above, not alerted. Nothing further is written for a suppressed
2410
+ // occurrence: no attempt was made, and a delivery row saying
2411
+ // `delivered: false` would read as a transport failure that never
2412
+ // happened. The decision itself is unaffected — suppression governs who
2413
+ // is TOLD, never what is DECIDED.
2414
+ if (!suppressAlert) {
2415
+ // The audit-persistence fact rides ALONG with the alert when the local
2416
+ // record failed: bounded, no secrets, and it tells the operator that
2417
+ // this notification is the only trace of the event. Appended to the
2418
+ // reason rather than added as a field so it survives an older installed
2419
+ // dist whose notification builder does not know about it.
2420
+ const auditNote = decisionRowPersisted ? '' : ' [auditPersistence=failed: no local audit row for this event]';
2421
+ const suppressedNote = alertGate && alertGate.suppressedSinceLastAlert > 0
2422
+ ? ` [${alertGate.suppressedSinceLastAlert} further scan-unavailable event(s) suppressed since the last alert; ${alertGate.count} this session]`
2423
+ : '';
2424
+ notifyResult = await notifyOperatorOfConversationThreat({
2425
+ outcome: label,
2426
+ posture,
2427
+ summary: scan.summary,
2428
+ // REDACTED. This one leaves the box entirely — to a webhook, an
2429
+ // aggregator, a phone — so it is the last place a tokenised endpoint
2430
+ // URL lifted out of a scanner error may appear.
2431
+ reason: `${safeReason ?? 'conversation threat'}${suppressedNote}${auditNote}`,
2432
+ sessionId,
2433
+ model,
2434
+ });
2435
+ // Truthful reporting: never imply a human was reached unless a
2436
+ // transport said so. "Not configured" is not a failure — it is the #143
2437
+ // default.
2438
+ if (notifyResult.configured && !notifyResult.delivered) {
2439
+ // #226: redacted HERE too, not only on the row below. The same detail
2440
+ // string reaches both, and a gateway's stdout is routinely shipped
2441
+ // somewhere it outlives the process.
2442
+ console.warn(`[shieldcortex] ⚠️ conversation alert UNDELIVERED — ${redactNotifyDetail(notifyResult.detail)}`);
2443
+ }
2444
+ // Guarded on `audit` because `eventId` has to point at something:
2445
+ // `evaluateConversationRun` never sets notify without audit, and if that
2446
+ // ever changed, a delivery row keyed to a decision row that was never
2447
+ // written would be a dangling reference rather than evidence.
2448
+ if (decision.audit) {
2449
+ // A SECOND row, not a rewrite of the first. The audit sink is an
2450
+ // append-only JSONL file, so "what was decided" and "who was told" are
2451
+ // separate facts recorded when each became true, joined by eventId.
2452
+ // `via` is the channel NAME ('webhook', 'openclaw-gateway'), never a
2453
+ // URL; the detail is redacted before it is persisted.
2454
+ await auditLog({
2455
+ type: 'notification_delivery',
2456
+ hook: 'before_agent_run',
2457
+ eventId,
2458
+ sessionId,
2459
+ configured: notifyResult.configured,
2460
+ delivered: notifyResult.delivered,
2461
+ via: notifyResult.via,
2462
+ detail: redactNotifyDetail(notifyResult.detail),
2463
+ // The eventId this row joins on may point at a row that was never
2464
+ // written. Say so here rather than leave a dangling reference that
2465
+ // reads as a missing file rather than a failed write.
2466
+ ...(decisionRowPersisted ? {} : { auditPersistence: 'failed' }),
2467
+ ts: new Date().toISOString(),
2468
+ });
2469
+ }
2470
+ }
2471
+ }
2472
+ // Clean, observed-not-blocked, and scan-unavailable all land here. The
2473
+ // audit row and the operator alert above have already recorded what
2474
+ // happened; the run itself proceeds, and says so.
2475
+ if (!decision.block)
2476
+ return gatePass();
2477
+ return {
2478
+ outcome: 'block',
2479
+ // Redacted for the same reason as the row above: the host SDK documents
2480
+ // `reason` as internal, but "internal" is a policy, not a guarantee.
2481
+ reason: safeReason ?? 'conversation threat',
2482
+ message: `ShieldCortex blocked this turn: ${scan.summary}. The prompt was not sent to the model.`,
2483
+ category: 'prompt_injection',
2484
+ };
2485
+ }
2486
+ catch (e) {
2487
+ // Fail open, loudly. Never let the guard's own failure stop the agent.
2488
+ //
2489
+ // This catch is also why the handler must not be allowed to THROW: the host
2490
+ // registers `before_agent_run` as fail-CLOSED
2491
+ // (`failurePolicyByHook: { before_agent_run: 'fail-closed' }`), so an
2492
+ // exception escaping here does not fail open at all — the gateway catches it
2493
+ // and blocks the run with "before_agent_run hook failed". An explicit pass
2494
+ // is the only way this function actually keeps its fail-open promise.
2495
+ console.error('[shieldcortex] before_agent_run error (failing open):', e instanceof Error ? e.message : String(e));
2496
+ return gatePass();
2497
+ }
2498
+ }
941
2499
  // Skip text blocks that are ShieldCortex/OpenClaw tool-result pass-throughs
942
2500
  function isToolResultContent(text) {
943
2501
  // ShieldCortex recall returns "Found N memories:" header
@@ -1117,6 +2675,12 @@ export default {
1117
2675
  if (_registered)
1118
2676
  return;
1119
2677
  _registered = true;
2678
+ // #226: the host runtime's own version, before anything can throw. It is
2679
+ // the primary version evidence for the conversation-gate check — the
2680
+ // gateway stating its own version beats inferring one from whichever
2681
+ // package.json sits above the entry path. Absent on a host that does not
2682
+ // expose it, which stays UNKNOWN rather than becoming a guess.
2683
+ recordHostRuntimeVersion(api);
1120
2684
  // --- Interceptor (lazy init) ---
1121
2685
  let interceptorReady = null;
1122
2686
  let interceptorInitAttempted = false;
@@ -1163,11 +2727,43 @@ export default {
1163
2727
  : `${guardCfg.enforce ? "enforce" : "warn"}${autoApproved > 0 ? ` (${autoApproved} auto-approved)` : ""}${interceptorReady ? "" : " — not yet initialised this session"}`;
1164
2728
  const hooksLine = _beforeToolCallRegistered
1165
2729
  ? "llm_input (scan), llm_output (memory), before_tool_call (action guard), session_end (cache reset)"
1166
- : "llm_input (scan), llm_output (memory)";
2730
+ // #226: session_end is registered even with the interceptor off —
2731
+ // the conversation gate keeps per-session state that needs freeing.
2732
+ : "llm_input (scan), llm_output (memory), session_end (cache reset)";
2733
+ // #225: the conversation plane, stated as evidence rather than as a
2734
+ // tick. Every clause below is something this process actually knows:
2735
+ // the configured posture, that we asked for the hook, the host build,
2736
+ // and the operator's grant. Nothing here claims the gateway accepted
2737
+ // the registration, because the plugin API never says so.
2738
+ const hostProbe = detectHostOpenClaw();
2739
+ const plane = describeConversationPlane({
2740
+ posture: conversationPosture(cfg.interceptor?.conversation),
2741
+ hookRequested: _beforeAgentRunRequested,
2742
+ gateSupport: hostSupportsConversationGate(hostProbe),
2743
+ hostOpenClawVersion: hostProbe.version,
2744
+ consentGranted: _conversationAccessGranted,
2745
+ });
2746
+ const notifyRaw = cfg.interceptor?.actionGuard?.notify;
2747
+ const notifyState = notifyRaw && notifyRaw.enabled === true
2748
+ ? 'configured'
2749
+ : 'not configured — detections reach the audit log and this box only';
1167
2750
  return {
1168
2751
  text: `ShieldCortex v${_version}\n` +
1169
- ` Hooks: ${hooksLine}\n` +
2752
+ ` Hooks: ${hooksLine}${_beforeAgentRunRequested ? ', before_agent_run (conversation gate, requested)' : ''}\n` +
1170
2753
  ` Action guard: ${guardState}\n` +
2754
+ ` Conversation firewall: ${plane.summary}\n` +
2755
+ // #226: state the PROVENANCE, not just the value. This flag is a
2756
+ // SNAPSHOT taken once, when the plugin loaded — the host reads
2757
+ // the grant at hook-registration time and this process never
2758
+ // re-reads it. So an operator who has just edited openclaw.json
2759
+ // and re-run the command sees the old answer, correctly, and
2760
+ // would otherwise conclude the grant does not work. Nothing here
2761
+ // is live: changing it requires a gateway restart before either
2762
+ // the gateway or this line reflects it.
2763
+ ` Conversation access grant: ${_conversationAccessGranted ? 'granted' : 'NOT granted'} (plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess)\n` +
2764
+ ' — read from openclaw.json when this plugin LOADED; it is a snapshot, not a live read.\n' +
2765
+ ' Editing that key takes effect only after a gateway restart, for the gateway and for this line.\n' +
2766
+ ` Operator notify: ${notifyState}\n` +
1171
2767
  ` Auto memory: ${autoMemory} | Dedupe: ${dedupe}\n` +
1172
2768
  ` Cloud sync: ${cloud}`,
1173
2769
  };
@@ -1269,41 +2865,148 @@ export default {
1269
2865
  return handleTypedBeforeToolCall(event, interceptor, api.logger, ctx?.sessionId);
1270
2866
  }, { priority: 80, timeoutMs: 30_000 });
1271
2867
  _beforeToolCallRegistered = true;
1272
- // Try to register session_end for cache cleanup (only meaningful while
1273
- // an interceptor can exist)
1274
- try {
1275
- api.on('session_end', (ev) => {
1276
- interceptorReady?.resetSession();
1277
- // #233: a taint must not outlive the conversation that earned it.
1278
- if (ev?.sessionId)
1279
- sessionTaint.clear(ev.sessionId);
1280
- });
1281
- }
1282
- catch {
1283
- // session_end may not be a supported hook — TTL safety net handles this
1284
- }
2868
+ // NOTE: session_end is NOT registered here — it moved out of this guard
2869
+ // in #226 and is registered unconditionally below.
1285
2870
  }
1286
2871
  else {
1287
2872
  api.logger?.info?.('[shieldcortex] interceptor.enabled:false in plugin config — before_tool_call hook not registered');
1288
2873
  }
1289
- // These two are CONVERSATION hooks: OpenClaw drops them at registration
1290
- // for a non-bundled plugin unless the host grants
1291
- // plugins.entries.<id>.hooks.allowConversationAccess = true. We still
1292
- // attempt registration (the host decides, and the grant can be added
1293
- // without a code change), but we must not CLAIM them afterwards — see the
1294
- // honesty note on the log line below (#225).
2874
+ // session_end — registered UNCONDITIONALLY (#226).
2875
+ //
2876
+ // It used to live inside the `interceptorDisabledInHostConfig` guard above,
2877
+ // on the reasoning that it exists for the interceptor's session cache. That
2878
+ // stopped being true when `before_agent_run` landed: the gate is registered
2879
+ // regardless of `interceptor.enabled` (its posture, not the interceptor
2880
+ // flag, decides what it does), and it accumulates per-session
2881
+ // scan-unavailable suppression state. With the cleanup hook skipped, a host
2882
+ // that disabled the interceptor kept every session's window alive for the
2883
+ // life of the gateway process.
2884
+ //
2885
+ // Registering it does NOT reintroduce #112. That incident was specific to
2886
+ // `before_tool_call`: a registered approval hook changes how OpenClaw
2887
+ // resolves tool-call approvals for unattended Codex agents, so an
2888
+ // unattended turn waited 120s on a decision nobody could give. `session_end`
2889
+ // is a notification — it cannot block, approve, or delay anything — and its
2890
+ // handler here only frees local state.
2891
+ try {
2892
+ api.on('session_end', (event, ctx) => {
2893
+ interceptorReady?.resetSession();
2894
+ const endedSession = ctx?.sessionId ?? ctx?.sessionKey ?? event?.sessionId ?? event?.sessionKey ?? null;
2895
+ // #226: the scan-unavailable alert window is session state too, and it
2896
+ // is keyed per session — so clear THIS session's window and nobody
2897
+ // else's. Clearing them all would re-arm alerting for every live
2898
+ // session every time any one of them ended.
2899
+ resetScanUnavailableAlertState(endedSession);
2900
+ // #233: a taint must not outlive the conversation that earned it. Same
2901
+ // per-session rule, for the same reason.
2902
+ if (endedSession)
2903
+ sessionTaint.clear(endedSession);
2904
+ });
2905
+ }
2906
+ catch {
2907
+ // session_end may not be a supported hook — TTL safety net handles this
2908
+ }
2909
+ // llm_input/llm_output are CONVERSATION hooks: OpenClaw drops them at
2910
+ // registration for a non-bundled plugin unless the host grants
2911
+ // plugins.entries.<id>.hooks.allowConversationAccess = true. Registration is
2912
+ // still attempted (the host decides, and the grant can be added without a
2913
+ // code change), but they must not be CLAIMED afterwards — see the startup
2914
+ // line below (#225/#230).
1295
2915
  api.on("llm_input", handleLlmInput, { timeoutMs: 30_000 });
1296
2916
  api.on("llm_output", handleLlmOutput, { timeoutMs: 30_000 });
1297
- // #225: this line used to announce `llm_input + llm_output` unconditionally.
1298
- // On any host without the conversation-access grant the gateway logged, on
1299
- // the very next two lines, that it had dropped both — so ShieldCortex was
1300
- // claiming conversation protection it did not have, in the one place an
1301
- // operator looks to confirm startup. Report only what is actually live, and
1302
- // name the missing grant when it is the reason.
1303
- const conversationAccess = readConversationAccess(homedir(), PLUGIN_ID);
2917
+ // #225: the conversation firewall's ENFORCEMENT point. `llm_input` above is
2918
+ // an OpenClaw *observation* hook — it cannot stop anything, which is why a
2919
+ // detected injection reached the model regardless and the only trace was a
2920
+ // console line. `before_agent_run` is the documented hook that can block a
2921
+ // run, so the verdict lands here where it can actually act.
2922
+ //
2923
+ // Registration is attempted unconditionally: the posture
2924
+ // (off/observe/enforce) decides what happens, and it is read per-call so a
2925
+ // config change takes effect without a restart. Registering conditionally
2926
+ // would make "is the guard wired?" depend on config read at boot — the
2927
+ // exact class of silent gap that #214/#222 were.
2928
+ //
2929
+ // The try/catch is for a host whose `api.on` throws on an unknown name. On
2930
+ // the hosts we have inspected it does NOT throw — an unsupported name is
2931
+ // dropped with a diagnostic, and a conversation hook without the operator's
2932
+ // grant is refused the same way — so a successful call proves only that we
2933
+ // ASKED. That is exactly what the flag is named after, and the honest
2934
+ // reporting comes from the version + consent evidence below.
2935
+ try {
2936
+ api.on("before_agent_run", handleBeforeAgentRun, { timeoutMs: 30_000 });
2937
+ _beforeAgentRunRequested = true;
2938
+ }
2939
+ catch (err) {
2940
+ _beforeAgentRunRequested = false;
2941
+ api.logger?.warn?.(`[shieldcortex] before_agent_run could not be registered on this host (${err instanceof Error ? err.message : String(err)}) — the conversation firewall cannot block on this gateway`);
2942
+ }
2943
+ // The gateway's own message seam, WHERE a host provides one (#143's design
2944
+ // intent: "on OpenClaw the transport should use the gateway's own message
2945
+ // capability"). Probed structurally, never required. No build we have
2946
+ // inspected exposes it — `notifyOperator` appears nowhere in the plugin API
2947
+ // of 2026.5.2 or 2026.7.1 — so on today's hosts this stays null and
2948
+ // conversation alerts go to the webhook.
2949
+ const notifyCtx = api;
2950
+ if (typeof notifyCtx.notifyOperator === 'function') {
2951
+ _gatewayNotifyContext = notifyCtx;
2952
+ }
2953
+ else if (typeof notifyCtx.runtime?.notifyOperator === 'function') {
2954
+ _gatewayNotifyContext = notifyCtx.runtime;
2955
+ }
2956
+ // The operator's conversation-access grant. Read, never written: OpenClaw
2957
+ // refuses every conversation hook for a non-bundled plugin without it, so a
2958
+ // box missing it runs with NO conversation plane at all — and on four of
2959
+ // five fleet hosts surveyed in #222 that was the normal outcome of a
2960
+ // documented install. Report it at boot rather than let the operator infer
2961
+ // protection from a registration line that only states intent.
2962
+ // The host's own in-memory config is the better source (it is what the
2963
+ // loader consulted), so it is preferred; the file the host reads is the
2964
+ // fallback for a runtime that does not expose it. `readConversationAccess`
2965
+ // is #225's shared reader — it also tells us whether the config could be
2966
+ // read at all, which is what keeps "not granted" apart from "cannot tell"
2967
+ // on the startup line below.
2968
+ const diskAccess = readConversationAccess(homedir(), PLUGIN_ID);
2969
+ let rootConfigSeen = false;
2970
+ try {
2971
+ const runtimeConfigApi = api.runtime?.config;
2972
+ const rootConfig = typeof runtimeConfigApi?.current === 'function'
2973
+ ? runtimeConfigApi.current()
2974
+ : typeof runtimeConfigApi?.loadConfig === 'function'
2975
+ ? runtimeConfigApi.loadConfig()
2976
+ : api.config;
2977
+ rootConfigSeen = Boolean(rootConfig) && typeof rootConfig === 'object';
2978
+ _conversationAccessGranted = rootConfigSeen
2979
+ ? readConversationAccessGrant(rootConfig)
2980
+ : diskAccess.granted;
2981
+ }
2982
+ catch {
2983
+ _conversationAccessGranted = diskAccess.granted;
2984
+ }
2985
+ if (!_conversationAccessGranted) {
2986
+ api.logger?.warn?.(`[shieldcortex] conversation firewall INACTIVE: plugins.entries.${PLUGIN_ID}.hooks.allowConversationAccess is not true in openclaw.json — ` +
2987
+ 'the gateway will refuse llm_input, llm_output and before_agent_run for this plugin. Nothing on the conversation path is scanned or blocked. ' +
2988
+ 'This is an operator consent grant and ShieldCortex will never set it for you.');
2989
+ }
2990
+ // #225/#230: this line used to announce `llm_input + llm_output`
2991
+ // unconditionally. On any host without the conversation-access grant the
2992
+ // gateway logged, on the very next two lines, that it had dropped both — so
2993
+ // ShieldCortex was claiming conversation protection it did not have, in the
2994
+ // one place an operator looks to confirm startup. Report only what is
2995
+ // actually live, and name the missing grant when it is the reason.
2996
+ //
2997
+ // `before_agent_run` (#226) is on the same list from 2026.5.9-beta.1, so it
2998
+ // is claimed only when the grant is present AND registration was attempted
2999
+ // this session.
1304
3000
  api.logger.info(`[shieldcortex] v${_version} registered (${describeRegisteredHooks({
1305
- access: conversationAccess,
3001
+ access: {
3002
+ granted: _conversationAccessGranted,
3003
+ // We could read SOMETHING (the host's config or the file) ⇒ the
3004
+ // ungranted state is a fact, not a failed measurement.
3005
+ readable: rootConfigSeen || diskAccess.readable,
3006
+ entryPresent: diskAccess.entryPresent,
3007
+ },
1306
3008
  beforeToolCallRegistered: _beforeToolCallRegistered,
3009
+ beforeAgentRunRequested: _beforeAgentRunRequested,
1307
3010
  })})`);
1308
3011
  }
1309
3012
  catch (err) {