claude-flow 3.41.4 → 3.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/.claude/.proven-config-version +1 -0
  2. package/.claude/proven-config.json +42 -0
  3. package/node_modules/@claude-flow/codex/dist/cli.js +0 -0
  4. package/node_modules/@claude-flow/codex/dist/dual-mode/orchestrator.d.ts +4 -0
  5. package/node_modules/@claude-flow/codex/dist/dual-mode/orchestrator.d.ts.map +1 -1
  6. package/node_modules/@claude-flow/codex/dist/dual-mode/orchestrator.js +38 -1
  7. package/node_modules/@claude-flow/codex/dist/dual-mode/orchestrator.js.map +1 -1
  8. package/node_modules/@claude-flow/codex/package.json +2 -1
  9. package/node_modules/@claude-flow/plugin-agent-federation/dist/bin.js +0 -0
  10. package/node_modules/@claude-flow/security/dist/index.d.ts +1 -1
  11. package/node_modules/@claude-flow/security/dist/index.d.ts.map +1 -1
  12. package/node_modules/@claude-flow/security/dist/index.js +1 -1
  13. package/node_modules/@claude-flow/security/dist/index.js.map +1 -1
  14. package/node_modules/@claude-flow/security/dist/mcp-caller-identity.d.ts +11 -0
  15. package/node_modules/@claude-flow/security/dist/mcp-caller-identity.d.ts.map +1 -1
  16. package/node_modules/@claude-flow/security/dist/mcp-caller-identity.js +0 -0
  17. package/node_modules/@claude-flow/security/dist/mcp-caller-identity.js.map +1 -1
  18. package/node_modules/@claude-flow/security/package.json +1 -0
  19. package/package.json +1 -1
  20. package/v3/@claude-flow/cli/catalog-manifest.json +2 -2
  21. package/v3/@claude-flow/cli/dist/src/commands/hive-mind.js +4 -3
  22. package/v3/@claude-flow/cli/dist/src/init/claudemd-generator.js +2 -2
  23. package/v3/@claude-flow/cli/dist/src/mcp-server.js +12 -0
  24. package/v3/@claude-flow/cli/dist/src/mcp-tools/agentdb-tools.js +17 -1
  25. package/v3/@claude-flow/cli/dist/src/mcp-tools/hive-mind-tools.d.ts +8 -0
  26. package/v3/@claude-flow/cli/dist/src/mcp-tools/hive-mind-tools.js +80 -2
  27. package/v3/@claude-flow/cli/dist/src/mcp-tools/policy-enforcer.d.ts +126 -0
  28. package/v3/@claude-flow/cli/dist/src/mcp-tools/policy-enforcer.js +177 -0
  29. package/v3/@claude-flow/cli/dist/src/memory/intelligence.d.ts +14 -2
  30. package/v3/@claude-flow/cli/dist/src/memory/intelligence.js +33 -12
  31. package/v3/@claude-flow/cli/dist/src/memory/memory-bridge.js +10 -2
  32. package/v3/@claude-flow/cli/dist/src/services/policy-runtime.js +63 -5
  33. package/v3/@claude-flow/cli/package.json +1 -1
  34. package/v3/@claude-flow/cli/dist/src/ruvector/diskann-backend.d.ts +0 -78
  35. package/v3/@claude-flow/cli/dist/src/ruvector/diskann-backend.js +0 -310
@@ -5,6 +5,7 @@
5
5
  */
6
6
  import { existsSync, readFileSync, writeFileSync, mkdirSync } from 'node:fs';
7
7
  import { join } from 'node:path';
8
+ import { randomBytes, timingSafeEqual } from 'node:crypto';
8
9
  import { getProjectCwd } from './types.js';
9
10
  import { validateIdentifier, validateText } from './validate-input.js';
10
11
  // Storage paths
@@ -83,6 +84,36 @@ function tryResolveProposal(proposal, totalNodes) {
83
84
  }
84
85
  return null;
85
86
  }
87
+ /**
88
+ * Verify a caller-supplied hiveToken against the one minted by hive-mind_init,
89
+ * using a constant-time comparison (bearer-token capability check -- callers
90
+ * that never joined, and callers that guess/omit the token, get identical
91
+ * rejection). Returns an error string on failure, or null on success.
92
+ */
93
+ function requireHiveToken(state, suppliedToken) {
94
+ if (!state.hiveToken) {
95
+ return 'Hive-mind has no capability token minted (re-run hive-mind_init)';
96
+ }
97
+ if (typeof suppliedToken !== 'string' || !suppliedToken) {
98
+ return 'hiveToken is required';
99
+ }
100
+ const expected = Buffer.from(state.hiveToken, 'utf-8');
101
+ const actual = Buffer.from(suppliedToken, 'utf-8');
102
+ if (expected.length !== actual.length || !timingSafeEqual(expected, actual)) {
103
+ return 'Invalid hiveToken';
104
+ }
105
+ return null;
106
+ }
107
+ /**
108
+ * Read the current hiveToken directly off disk, for same-machine callers
109
+ * that already have filesystem access to hive state (the CLI's own
110
+ * `hive-mind join/leave/consensus` subcommands) -- NOT exposed over any MCP
111
+ * tool response (in particular, not hive-mind_status), since that's a
112
+ * remote-reachable surface a capability token must not leak through.
113
+ */
114
+ export function getHiveTokenForCli() {
115
+ return loadHiveState().hiveToken;
116
+ }
86
117
  function getHiveDir() {
87
118
  return join(getProjectCwd(), STORAGE_DIR, HIVE_DIR);
88
119
  }
@@ -248,6 +279,14 @@ export const hiveMindTools = [
248
279
  electedAt: new Date().toISOString(),
249
280
  term: 1,
250
281
  };
282
+ // Mint a capability token on first init; a re-init (topology/consensus
283
+ // change on an already-initialized hive) keeps the existing token and
284
+ // roster rather than silently invalidating workers who already hold
285
+ // it -- only re-generated if somehow absent (e.g. state predates this
286
+ // field).
287
+ if (!state.hiveToken) {
288
+ state.hiveToken = randomBytes(32).toString('hex');
289
+ }
251
290
  saveHiveState(state);
252
291
  return {
253
292
  success: true,
@@ -256,6 +295,7 @@ export const hiveMindTools = [
256
295
  consensus: state.consensusStrategy,
257
296
  queenId,
258
297
  status: 'initialized',
298
+ hiveToken: state.hiveToken,
259
299
  config: {
260
300
  topology: state.topology,
261
301
  consensus: state.consensusStrategy,
@@ -375,8 +415,9 @@ export const hiveMindTools = [
375
415
  properties: {
376
416
  agentId: { type: 'string', description: 'Agent ID to join' },
377
417
  role: { type: 'string', enum: ['worker', 'specialist', 'scout'], description: 'Agent role in hive' },
418
+ hiveToken: { type: 'string', description: 'Capability token minted by hive-mind_init' },
378
419
  },
379
- required: ['agentId'],
420
+ required: ['agentId', 'hiveToken'],
380
421
  },
381
422
  handler: async (input) => {
382
423
  const state = loadHiveState();
@@ -389,6 +430,13 @@ export const hiveMindTools = [
389
430
  if (!state.initialized) {
390
431
  return { success: false, error: 'Hive-mind not initialized' };
391
432
  }
433
+ // Fail-closed: an unrecognized/missing token makes no membership
434
+ // change at all -- state.workers is untouched, not just left
435
+ // unsaved (the write below never happens on this path).
436
+ const tokenError = requireHiveToken(state, input.hiveToken);
437
+ if (tokenError) {
438
+ return { success: false, agentId, error: tokenError };
439
+ }
392
440
  if (!state.workers.includes(agentId)) {
393
441
  state.workers.push(agentId);
394
442
  saveHiveState(state);
@@ -410,8 +458,9 @@ export const hiveMindTools = [
410
458
  type: 'object',
411
459
  properties: {
412
460
  agentId: { type: 'string', description: 'Agent ID to remove' },
461
+ hiveToken: { type: 'string', description: 'Capability token minted by hive-mind_init' },
413
462
  },
414
- required: ['agentId'],
463
+ required: ['agentId', 'hiveToken'],
415
464
  },
416
465
  handler: async (input) => {
417
466
  const state = loadHiveState();
@@ -421,6 +470,12 @@ export const hiveMindTools = [
421
470
  if (!v.valid)
422
471
  return { success: false, agentId, error: v.error };
423
472
  }
473
+ // Fail-closed: a denied caller makes no membership change -- the
474
+ // splice/save below is unreachable on this path.
475
+ const tokenError = requireHiveToken(state, input.hiveToken);
476
+ if (tokenError) {
477
+ return { success: false, agentId, error: tokenError };
478
+ }
424
479
  const index = state.workers.indexOf(agentId);
425
480
  if (index > -1) {
426
481
  state.workers.splice(index, 1);
@@ -448,6 +503,7 @@ export const hiveMindTools = [
448
503
  value: { description: 'Proposal value (for propose)' },
449
504
  vote: { type: 'boolean', description: 'Vote (true=for, false=against)' },
450
505
  voterId: { type: 'string', description: 'Voter agent ID' },
506
+ hiveToken: { type: 'string', description: 'Capability token minted by hive-mind_init (required to vote)' },
451
507
  strategy: { type: 'string', enum: ['bft', 'raft', 'quorum'], description: 'Consensus strategy (default: raft)' },
452
508
  quorumPreset: { type: 'string', enum: ['unanimous', 'majority', 'supermajority'], description: 'Quorum threshold preset (for quorum strategy, default: majority)' },
453
509
  term: { type: 'number', description: 'Term number (for raft strategy)' },
@@ -531,6 +587,28 @@ export const hiveMindTools = [
531
587
  if (!voterId) {
532
588
  return { action, error: 'voterId is required for voting' };
533
589
  }
590
+ // Fail-closed: a denied caller records no vote at all -- the
591
+ // votes[voterId] write below is unreachable on this path, and
592
+ // nothing about the proposal (vote tallies, status) changes.
593
+ const tokenError = requireHiveToken(state, input.hiveToken);
594
+ if (tokenError) {
595
+ return { action, error: tokenError, proposalId: proposal.proposalId };
596
+ }
597
+ // voterId was previously trusted as-is: any caller-supplied string
598
+ // was recorded into proposal.votes and counted toward
599
+ // calculateRequiredVotes()'s threshold (derived from
600
+ // state.workers.length), with no check that it named a worker who
601
+ // actually joined this hive-mind. That let a single caller cross
602
+ // any strategy's quorum (raft/bft/quorum alike) by voting under
603
+ // fabricated ids — a Sybil attack on consensus, not merely a
604
+ // double-vote. Require the voter to be a registered worker.
605
+ if (!state.workers.includes(voterId)) {
606
+ return {
607
+ action,
608
+ error: `Voter ${voterId} is not a registered hive-mind worker`,
609
+ proposalId: proposal.proposalId,
610
+ };
611
+ }
534
612
  const voteValue = input.vote;
535
613
  const proposalStrategy = proposal.strategy || 'raft';
536
614
  const required = calculateRequiredVotes(proposalStrategy, totalNodes, proposal.quorumPreset);
@@ -0,0 +1,126 @@
1
+ /**
2
+ * MCP Governance Policy Enforcer (opt-in).
3
+ *
4
+ * `.harness/mcp-policy.json` declares governance intent (defaultDeny,
5
+ * auditLog, maxToolCallsPerTurn, dangerousPatterns, ...) for the claude-flow
6
+ * MCP server, but until now nothing in the running server (mcp-server.ts)
7
+ * ever read it: `harness mcp-scan` grades the file's *posture* offline, the
8
+ * live `tools/call` dispatch never consulted it. Any connected MCP client
9
+ * could call every registered tool with no audit trail and no call budget.
10
+ *
11
+ * This module wires the two policy fields that are actually in this
12
+ * server's jurisdiction, per the policy file's own rationale comment
13
+ * (`dangerousPatterns` / `allowShell` / `allowNetwork` / `allowFileWrite`
14
+ * describe the native-Claude-Code-tool layer — Bash/Write/Edit/WebFetch —
15
+ * not this MCP server's memory_-, hooks_-, agentdb_-prefixed tool surface,
16
+ * so they are intentionally left unenforced here):
17
+ * - `auditLog`: append a JSONL record for every `tools/call`.
18
+ * - `maxToolCallsPerTurn`: bound calls per MCP *session* (one stdio
19
+ * process lifetime), deny once exceeded.
20
+ *
21
+ * Fully opt-in via `RUFLO_MCP_ENFORCE_POLICY=1` (or `true`). Unset/false
22
+ * means every function below is a no-op on the hot path — the pre-existing
23
+ * `tools/call` behavior is unchanged.
24
+ *
25
+ * FAIL-CLOSED once enforcement is enabled (PR #3139 review round 1):
26
+ * a missing/malformed policy file, or a failed mandatory audit-log write,
27
+ * denies the call rather than silently degrading to unrestricted execution.
28
+ * The whole point of opting in is a restriction that actually holds; an
29
+ * enforcement flag that quietly falls back to "no restriction" on its own
30
+ * misconfiguration defeats the feature. See `evaluateToolCall()`.
31
+ *
32
+ * Known scope limits (disclosed, not fixed here):
33
+ * - Only wired into the stdio `tools/call` dispatch
34
+ * (`MCPServerManager.handleMCPMessage`). The separate HTTP/websocket
35
+ * path (`startHttpServer()`, via `@claude-flow/mcp`) does not call
36
+ * this module and is unaffected even when this flag is set.
37
+ *
38
+ * `maxToolCallsPerTurn` reset semantics (dream-cycle 2026-09-01, follow-up
39
+ * to 2026-08-31 review round 1): despite the field's name, the original
40
+ * implementation enforced a *session-lifetime cumulative* cap that never
41
+ * reset — a long-lived stdio session could exhaust the budget under
42
+ * entirely legitimate use and stay locked out until the MCP server process
43
+ * restarted. Research that night (see the dream-cycle gist) found: (1) the
44
+ * MCP spec only mandates "rate limit tool invocations" with zero mechanism
45
+ * guidance, and its 2026-07-28 revision (SEP-2567) is actively removing the
46
+ * session concept from the protocol entirely; (2) every framework/product
47
+ * that gets this right (FastMCP's rate-limiting middleware, the PolicyLayer
48
+ * MCP firewall, Cloudflare's public rate limiter) anchors the reset to
49
+ * wall-clock time, not to a turn or session counter that never decays —
50
+ * a turn-count reset is gameable by a chatty loop re-arming its own budget,
51
+ * which wall-clock time is not. This module now enforces a *sliding
52
+ * wall-clock window*: `maxToolCallsPerTurn` calls are allowed per rolling
53
+ * `turnWindowMs` (default 60000) per session, keyed by call timestamp so
54
+ * calls fall out of the window as time passes rather than accumulating
55
+ * forever. `now` is an injectable parameter (defaults to `Date.now`) so
56
+ * production callers need no change and tests stay fully deterministic via
57
+ * `vi.useFakeTimers()`.
58
+ */
59
+ export interface McpPolicy {
60
+ schema?: number;
61
+ policyVersion?: number;
62
+ harnessId?: string;
63
+ defaultDeny?: boolean;
64
+ auditLog?: boolean;
65
+ requireApprovalForDangerous?: boolean;
66
+ toolTimeoutMs?: number;
67
+ /**
68
+ * Despite the name, this is a WALL-CLOCK rate limit, not a literal
69
+ * conversational-turn counter — MCP has no protocol-level concept of a
70
+ * "turn" to count against (confirmed: the spec is silent on it, and its
71
+ * 2026-07-28 revision removes the session concept entirely). Enforced as
72
+ * "at most this many calls in any rolling `turnWindowMs` window" per
73
+ * session. Treat it, and document it to callers, as rate limiting.
74
+ */
75
+ maxToolCallsPerTurn?: number;
76
+ /** Rolling window (ms) over which `maxToolCallsPerTurn` is counted. Default 60000. */
77
+ turnWindowMs?: number;
78
+ dangerousPatterns?: string[];
79
+ approvedServers?: string[];
80
+ [key: string]: unknown;
81
+ }
82
+ export declare function isPolicyEnforcementEnabled(env?: NodeJS.ProcessEnv): boolean;
83
+ export declare function loadMcpPolicy(policyPath?: string): McpPolicy | null;
84
+ /** Test-only: clear per-session call state between test cases. */
85
+ export declare function resetPolicyEnforcerState(): void;
86
+ export interface PolicyCheckResult {
87
+ allowed: boolean;
88
+ reason?: string;
89
+ }
90
+ /**
91
+ * Checks (and, if allowed, records) a tool call against
92
+ * `policy.maxToolCallsPerTurn`, counted over a sliding window of
93
+ * `policy.turnWindowMs` (default 60000ms) rather than the session's whole
94
+ * lifetime. Calls older than the window are pruned before comparing count
95
+ * to limit, so a session that pauses gets its budget back rather than
96
+ * staying denied until the process restarts. `now` defaults to `Date.now`
97
+ * for production callers; tests inject a controlled clock instead.
98
+ */
99
+ export declare function checkAndRecordCall(policy: McpPolicy, sessionId: string, now?: number): PolicyCheckResult;
100
+ export interface AuditLogEntry {
101
+ timestamp: string;
102
+ sessionId: string;
103
+ toolName: string;
104
+ allowed: boolean;
105
+ reason?: string;
106
+ }
107
+ /** Test-only: redirect the audit log to a temp file instead of the default path. */
108
+ export declare function setAuditLogPathForTesting(p: string | null): void;
109
+ export declare function getAuditLogPath(): string;
110
+ /**
111
+ * Appends one JSONL audit record. Returns `true` if `policy.auditLog` is not
112
+ * set (nothing was required) or the write succeeded; `false` only when
113
+ * `auditLog` is required and the write itself failed (disk full, unwritable
114
+ * path, etc). Never throws — the caller (`evaluateToolCall`) decides what a
115
+ * failed *mandatory* write means for the call (fail-closed: deny it).
116
+ */
117
+ export declare function appendAuditLog(policy: McpPolicy, entry: AuditLogEntry): boolean;
118
+ /**
119
+ * Single enforcement entry point for a `tools/call` dispatch. Combines, in
120
+ * order: fail-closed on a missing/malformed policy, the per-session call
121
+ * budget, and fail-closed on a failed mandatory audit-log write. `policy`
122
+ * is the result of `loadMcpPolicy()` — pass `null` straight through when it
123
+ * failed to load, rather than re-deciding that here.
124
+ */
125
+ export declare function evaluateToolCall(policy: McpPolicy | null, sessionId: string, toolName: string, now?: number): PolicyCheckResult;
126
+ //# sourceMappingURL=policy-enforcer.d.ts.map
@@ -0,0 +1,177 @@
1
+ /**
2
+ * MCP Governance Policy Enforcer (opt-in).
3
+ *
4
+ * `.harness/mcp-policy.json` declares governance intent (defaultDeny,
5
+ * auditLog, maxToolCallsPerTurn, dangerousPatterns, ...) for the claude-flow
6
+ * MCP server, but until now nothing in the running server (mcp-server.ts)
7
+ * ever read it: `harness mcp-scan` grades the file's *posture* offline, the
8
+ * live `tools/call` dispatch never consulted it. Any connected MCP client
9
+ * could call every registered tool with no audit trail and no call budget.
10
+ *
11
+ * This module wires the two policy fields that are actually in this
12
+ * server's jurisdiction, per the policy file's own rationale comment
13
+ * (`dangerousPatterns` / `allowShell` / `allowNetwork` / `allowFileWrite`
14
+ * describe the native-Claude-Code-tool layer — Bash/Write/Edit/WebFetch —
15
+ * not this MCP server's memory_-, hooks_-, agentdb_-prefixed tool surface,
16
+ * so they are intentionally left unenforced here):
17
+ * - `auditLog`: append a JSONL record for every `tools/call`.
18
+ * - `maxToolCallsPerTurn`: bound calls per MCP *session* (one stdio
19
+ * process lifetime), deny once exceeded.
20
+ *
21
+ * Fully opt-in via `RUFLO_MCP_ENFORCE_POLICY=1` (or `true`). Unset/false
22
+ * means every function below is a no-op on the hot path — the pre-existing
23
+ * `tools/call` behavior is unchanged.
24
+ *
25
+ * FAIL-CLOSED once enforcement is enabled (PR #3139 review round 1):
26
+ * a missing/malformed policy file, or a failed mandatory audit-log write,
27
+ * denies the call rather than silently degrading to unrestricted execution.
28
+ * The whole point of opting in is a restriction that actually holds; an
29
+ * enforcement flag that quietly falls back to "no restriction" on its own
30
+ * misconfiguration defeats the feature. See `evaluateToolCall()`.
31
+ *
32
+ * Known scope limits (disclosed, not fixed here):
33
+ * - Only wired into the stdio `tools/call` dispatch
34
+ * (`MCPServerManager.handleMCPMessage`). The separate HTTP/websocket
35
+ * path (`startHttpServer()`, via `@claude-flow/mcp`) does not call
36
+ * this module and is unaffected even when this flag is set.
37
+ *
38
+ * `maxToolCallsPerTurn` reset semantics (dream-cycle 2026-09-01, follow-up
39
+ * to 2026-08-31 review round 1): despite the field's name, the original
40
+ * implementation enforced a *session-lifetime cumulative* cap that never
41
+ * reset — a long-lived stdio session could exhaust the budget under
42
+ * entirely legitimate use and stay locked out until the MCP server process
43
+ * restarted. Research that night (see the dream-cycle gist) found: (1) the
44
+ * MCP spec only mandates "rate limit tool invocations" with zero mechanism
45
+ * guidance, and its 2026-07-28 revision (SEP-2567) is actively removing the
46
+ * session concept from the protocol entirely; (2) every framework/product
47
+ * that gets this right (FastMCP's rate-limiting middleware, the PolicyLayer
48
+ * MCP firewall, Cloudflare's public rate limiter) anchors the reset to
49
+ * wall-clock time, not to a turn or session counter that never decays —
50
+ * a turn-count reset is gameable by a chatty loop re-arming its own budget,
51
+ * which wall-clock time is not. This module now enforces a *sliding
52
+ * wall-clock window*: `maxToolCallsPerTurn` calls are allowed per rolling
53
+ * `turnWindowMs` (default 60000) per session, keyed by call timestamp so
54
+ * calls fall out of the window as time passes rather than accumulating
55
+ * forever. `now` is an injectable parameter (defaults to `Date.now`) so
56
+ * production callers need no change and tests stay fully deterministic via
57
+ * `vi.useFakeTimers()`.
58
+ */
59
+ import * as fs from 'fs';
60
+ import * as path from 'path';
61
+ import * as os from 'os';
62
+ export function isPolicyEnforcementEnabled(env = process.env) {
63
+ const v = env.RUFLO_MCP_ENFORCE_POLICY;
64
+ return v === '1' || (v ?? '').toLowerCase() === 'true';
65
+ }
66
+ export function loadMcpPolicy(policyPath = path.join(process.cwd(), '.harness', 'mcp-policy.json')) {
67
+ try {
68
+ const raw = fs.readFileSync(policyPath, 'utf-8');
69
+ const parsed = JSON.parse(raw);
70
+ if (typeof parsed !== 'object' || parsed === null)
71
+ return null;
72
+ return parsed;
73
+ }
74
+ catch {
75
+ return null;
76
+ }
77
+ }
78
+ const DEFAULT_TURN_WINDOW_MS = 60_000;
79
+ const sessionState = new Map();
80
+ /** Test-only: clear per-session call state between test cases. */
81
+ export function resetPolicyEnforcerState() {
82
+ sessionState.clear();
83
+ }
84
+ /**
85
+ * Checks (and, if allowed, records) a tool call against
86
+ * `policy.maxToolCallsPerTurn`, counted over a sliding window of
87
+ * `policy.turnWindowMs` (default 60000ms) rather than the session's whole
88
+ * lifetime. Calls older than the window are pruned before comparing count
89
+ * to limit, so a session that pauses gets its budget back rather than
90
+ * staying denied until the process restarts. `now` defaults to `Date.now`
91
+ * for production callers; tests inject a controlled clock instead.
92
+ */
93
+ export function checkAndRecordCall(policy, sessionId, now = Date.now()) {
94
+ const limit = policy.maxToolCallsPerTurn;
95
+ if (typeof limit !== 'number' || !Number.isFinite(limit) || limit <= 0) {
96
+ return { allowed: true };
97
+ }
98
+ const configuredWindow = policy.turnWindowMs;
99
+ const windowMs = typeof configuredWindow === 'number' && Number.isFinite(configuredWindow) && configuredWindow > 0
100
+ ? configuredWindow
101
+ : DEFAULT_TURN_WINDOW_MS;
102
+ const state = sessionState.get(sessionId) ?? { callTimes: [] };
103
+ const cutoff = now - windowMs;
104
+ state.callTimes = state.callTimes.filter((t) => t > cutoff);
105
+ if (state.callTimes.length >= limit) {
106
+ sessionState.set(sessionId, state);
107
+ return {
108
+ allowed: false,
109
+ reason: `maxToolCallsPerTurn (${limit}) exceeded within the last ${windowMs}ms for this session`,
110
+ };
111
+ }
112
+ state.callTimes.push(now);
113
+ sessionState.set(sessionId, state);
114
+ return { allowed: true };
115
+ }
116
+ let auditLogPathOverride = null;
117
+ /** Test-only: redirect the audit log to a temp file instead of the default path. */
118
+ export function setAuditLogPathForTesting(p) {
119
+ auditLogPathOverride = p;
120
+ }
121
+ function defaultAuditLogPath() {
122
+ return path.join(os.tmpdir(), 'ruflo-mcp-audit.jsonl');
123
+ }
124
+ export function getAuditLogPath() {
125
+ return auditLogPathOverride ?? defaultAuditLogPath();
126
+ }
127
+ /**
128
+ * Appends one JSONL audit record. Returns `true` if `policy.auditLog` is not
129
+ * set (nothing was required) or the write succeeded; `false` only when
130
+ * `auditLog` is required and the write itself failed (disk full, unwritable
131
+ * path, etc). Never throws — the caller (`evaluateToolCall`) decides what a
132
+ * failed *mandatory* write means for the call (fail-closed: deny it).
133
+ */
134
+ export function appendAuditLog(policy, entry) {
135
+ if (!policy.auditLog)
136
+ return true;
137
+ try {
138
+ fs.appendFileSync(getAuditLogPath(), `${JSON.stringify(entry)}\n`, 'utf-8');
139
+ return true;
140
+ }
141
+ catch {
142
+ return false;
143
+ }
144
+ }
145
+ /**
146
+ * Single enforcement entry point for a `tools/call` dispatch. Combines, in
147
+ * order: fail-closed on a missing/malformed policy, the per-session call
148
+ * budget, and fail-closed on a failed mandatory audit-log write. `policy`
149
+ * is the result of `loadMcpPolicy()` — pass `null` straight through when it
150
+ * failed to load, rather than re-deciding that here.
151
+ */
152
+ export function evaluateToolCall(policy, sessionId, toolName, now = Date.now()) {
153
+ if (policy === null) {
154
+ return {
155
+ allowed: false,
156
+ reason: 'RUFLO_MCP_ENFORCE_POLICY is set but .harness/mcp-policy.json is missing or invalid — failing closed',
157
+ };
158
+ }
159
+ const budget = checkAndRecordCall(policy, sessionId, now);
160
+ const auditOk = appendAuditLog(policy, {
161
+ timestamp: new Date(now).toISOString(),
162
+ sessionId,
163
+ toolName,
164
+ allowed: budget.allowed,
165
+ reason: budget.reason,
166
+ });
167
+ if (!budget.allowed)
168
+ return budget;
169
+ if (!auditOk) {
170
+ return {
171
+ allowed: false,
172
+ reason: 'audit log write failed and policy.auditLog is required — failing closed',
173
+ };
174
+ }
175
+ return { allowed: true };
176
+ }
177
+ //# sourceMappingURL=policy-enforcer.js.map
@@ -178,13 +178,25 @@ declare class LocalReasoningBank {
178
178
  */
179
179
  store(pattern: Omit<StoredPattern, 'usageCount' | 'createdAt' | 'lastUsedAt'> & Partial<StoredPattern>): void;
180
180
  /**
181
- * Find similar patterns by embedding
181
+ * Find similar patterns by embedding.
182
+ *
183
+ * `confidence` on each result is the pattern's own learned reliability
184
+ * (unchanged from storage) — NOT how well it matches this query. The
185
+ * per-query cosine score is returned separately as `similarity`. Callers
186
+ * that want "how good a semantic match is this" must read `.similarity`;
187
+ * callers that want "how reliable has this pattern proven to be" read
188
+ * `.confidence`. Prior to this fix both were conflated (confidence was
189
+ * overwritten with the cosine score), which silently broke any consumer
190
+ * that needed to tell them apart (found during the 2026-09-12 dream-cycle
191
+ * intelligence-surface review).
182
192
  */
183
193
  findSimilar(queryEmbedding: number[], options: {
184
194
  k?: number;
185
195
  threshold?: number;
186
196
  type?: string;
187
- }): StoredPattern[];
197
+ }): (StoredPattern & {
198
+ similarity: number;
199
+ })[];
188
200
  /**
189
201
  * Optimized cosine similarity
190
202
  */
@@ -250,10 +250,20 @@ class LocalSonaCoordinator {
250
250
  const oldConfidence = pattern.confidence;
251
251
  // Check EWC penalty before applying update
252
252
  if (ewcConsolidator) {
253
- const oldWeights = [oldConfidence];
254
253
  const proposedConfidence = Math.min(1.0, oldConfidence + this.config.loraLearningRate * reward);
255
- const newWeights = [proposedConfidence];
256
- const penalty = ewcConsolidator.getPenalty(oldWeights, newWeights);
254
+ // Use computeConfidencePenalty (averages the full Fisher diagonal),
255
+ // not getPenalty([oldConf],[newConf]) — that call shape collapses
256
+ // to fisherDiag[0] only (Math.min(1,1,384) === 1), an arbitrary
257
+ // single dimension instead of the full accumulated Fisher signal.
258
+ // computeConfidencePenalty exists precisely for this
259
+ // scalar-confidence case (see its docstring) but was unwired.
260
+ // Note: neither call shape differentiates between patterns — both
261
+ // take only a confidence delta, not a per-pattern embedding, so
262
+ // two patterns with the same delta under the same consolidator
263
+ // state get the same penalty either way. This fix corrects which
264
+ // shared Fisher signal informs that penalty; it does not add
265
+ // per-pattern discrimination (see ewc-distill-confidence-gate.test.ts).
266
+ const penalty = ewcConsolidator.computeConfidencePenalty(oldConfidence, proposedConfidence);
257
267
  totalEwcPenalty += penalty;
258
268
  // If penalty is too high, reduce the update magnitude
259
269
  if (penalty > this.config.ewcLambda) {
@@ -279,13 +289,14 @@ class LocalSonaCoordinator {
279
289
  }
280
290
  }
281
291
  }
282
- // Update EWC Fisher matrix with confidence changes
292
+ // Update EWC Fisher matrix with confidence changes. updateFisherFromConfidences
293
+ // takes the full per-pattern embedding + confidence-delta batch directly (it
294
+ // computes the same squared confidence-delta-scaled-embedding gradient proxy
295
+ // internally) — replaces the previous per-change recordGradient loop, which
296
+ // updated the full 384-dim globalFisher but fed a signal that getPenalty's
297
+ // 1-element call shape then read back only at index 0.
283
298
  if (ewcConsolidator && confidenceChanges.length > 0) {
284
- for (const change of confidenceChanges) {
285
- // Use confidence delta as gradient proxy
286
- const gradient = change.embedding.map(e => e * Math.abs(change.newConf - change.oldConf));
287
- ewcConsolidator.recordGradient(change.id, gradient, true);
288
- }
299
+ ewcConsolidator.updateFisherFromConfidences(confidenceChanges);
289
300
  }
290
301
  // Persist updated patterns
291
302
  bank.flushToDisk();
@@ -467,7 +478,17 @@ class LocalReasoningBank {
467
478
  this.saveToDisk();
468
479
  }
469
480
  /**
470
- * Find similar patterns by embedding
481
+ * Find similar patterns by embedding.
482
+ *
483
+ * `confidence` on each result is the pattern's own learned reliability
484
+ * (unchanged from storage) — NOT how well it matches this query. The
485
+ * per-query cosine score is returned separately as `similarity`. Callers
486
+ * that want "how good a semantic match is this" must read `.similarity`;
487
+ * callers that want "how reliable has this pattern proven to be" read
488
+ * `.confidence`. Prior to this fix both were conflated (confidence was
489
+ * overwritten with the cosine score), which silently broke any consumer
490
+ * that needed to tell them apart (found during the 2026-09-12 dream-cycle
491
+ * intelligence-surface review).
471
492
  */
472
493
  findSimilar(queryEmbedding, options) {
473
494
  const { k = 5, threshold = 0.5, type } = options;
@@ -489,7 +510,7 @@ class LocalReasoningBank {
489
510
  // Update usage
490
511
  s.pattern.usageCount++;
491
512
  s.pattern.lastUsedAt = Date.now();
492
- return { ...s.pattern, confidence: s.score };
513
+ return { ...s.pattern, similarity: s.score };
493
514
  });
494
515
  }
495
516
  /**
@@ -985,7 +1006,7 @@ export async function findSimilarPatterns(query, options) {
985
1006
  usageCount: r.usageCount,
986
1007
  createdAt: r.createdAt,
987
1008
  lastUsedAt: r.lastUsedAt,
988
- similarity: r.similarity ?? r.confidence ?? 0.5
1009
+ similarity: r.similarity
989
1010
  }));
990
1011
  }
991
1012
  catch {
@@ -1866,7 +1866,12 @@ export async function bridgeStorePattern(options) {
1866
1866
  }
1867
1867
  catch { /* HNSW is best-effort */ }
1868
1868
  }
1869
- return { success: true, patternId: result.id, controller: 'bridge-fallback' };
1869
+ // #3324: bridgeStoreEntry's `result.id` is its OWN internally generated
1870
+ // row id (generateId('entry')), a different value from the `key` the row
1871
+ // was actually stored under. getEntry/memory_retrieve look up by `key`,
1872
+ // so returning result.id here handed the caller a handle that can never
1873
+ // be read back — return `patternId` (the real key) instead.
1874
+ return { success: true, patternId, controller: 'bridge-fallback' };
1870
1875
  }
1871
1876
  catch {
1872
1877
  return null;
@@ -1915,10 +1920,13 @@ export async function bridgeSearchPatterns(options) {
1915
1920
  const qEmb = await generateEmbedding(options.query);
1916
1921
  if (qEmb && Array.isArray(qEmb.embedding) && qEmb.embedding.length > 0) {
1917
1922
  const hits = reasoningBank.findSimilar(qEmb.embedding, { k, threshold });
1923
+ // findSimilar() no longer overwrites confidence with the query-match
1924
+ // score — prefer similarity (the actual match strength) for search ranking,
1925
+ // falling back to confidence/score only for older/foreign result shapes.
1918
1926
  mapped = (Array.isArray(hits) ? hits : []).map((r) => ({
1919
1927
  id: r.id ?? '',
1920
1928
  content: r.content ?? '',
1921
- score: r.confidence ?? r.score ?? 0,
1929
+ score: r.similarity ?? r.confidence ?? r.score ?? 0,
1922
1930
  }));
1923
1931
  }
1924
1932
  }