@remnic/core 9.3.744 → 9.3.746

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (242) hide show
  1. package/dist/access-boundary.d.ts +17 -15
  2. package/dist/access-boundary.js +14 -15
  3. package/dist/access-cli.js +92 -33
  4. package/dist/access-cli.js.map +1 -1
  5. package/dist/access-http.d.ts +17 -15
  6. package/dist/access-http.js +20 -21
  7. package/dist/access-mcp.d.ts +17 -15
  8. package/dist/access-mcp.js +18 -19
  9. package/dist/access-operations-batch.js +16 -17
  10. package/dist/access-operations.d.ts +17 -15
  11. package/dist/access-operations.js +17 -18
  12. package/dist/access-schema.js +3 -3
  13. package/dist/{access-service-OT3Gk79W.d.ts → access-service-Cx16nmCv.d.ts} +3 -30
  14. package/dist/access-service.d.ts +7 -5
  15. package/dist/access-service.js +13 -14
  16. package/dist/access-surface-catalog.d.ts +17 -15
  17. package/dist/action-confidence.d.ts +1 -1
  18. package/dist/active-memory-bridge.d.ts +1 -1
  19. package/dist/active-recall.d.ts +1 -1
  20. package/dist/{auto-sync-XHSKKQAH.js → auto-sync-IEBKPGAM.js} +2 -2
  21. package/dist/behavior-learner.d.ts +1 -1
  22. package/dist/behavior-signals.d.ts +1 -1
  23. package/dist/bootstrap.d.ts +6 -4
  24. package/dist/briefing.d.ts +1 -1
  25. package/dist/briefing.js +3 -3
  26. package/dist/buffer-surprise-report.d.ts +1 -1
  27. package/dist/buffer.d.ts +1 -1
  28. package/dist/calibration.d.ts +1 -1
  29. package/dist/capabilities.d.ts +1 -1
  30. package/dist/{catalog-CX0CjOaf.d.ts → catalog-D_ZDxu8Y.d.ts} +1 -1
  31. package/dist/causal-behavior.d.ts +1 -1
  32. package/dist/causal-consolidation.d.ts +1 -1
  33. package/dist/causal-consolidation.js +4 -4
  34. package/dist/causal-trajectory-graph.d.ts +1 -1
  35. package/dist/{chunk-WPLRJ4XM.js → chunk-24YDB5QP.js} +2098 -230
  36. package/dist/chunk-24YDB5QP.js.map +1 -0
  37. package/dist/{chunk-6XG3NQRC.js → chunk-3IJGF6K2.js} +2 -2
  38. package/dist/{chunk-U5WVRMCA.js → chunk-5IN26V6W.js} +2 -2
  39. package/dist/{chunk-XEADGIAA.js → chunk-DIHCFJVV.js} +7 -7
  40. package/dist/{chunk-C7LDPZJJ.js → chunk-EHYB325U.js} +227 -227
  41. package/dist/chunk-EHYB325U.js.map +1 -0
  42. package/dist/{chunk-E7RGJQWP.js → chunk-HIHCB2S6.js} +2 -2
  43. package/dist/{chunk-JFCS6PQU.js → chunk-JHQUIH3W.js} +5 -5
  44. package/dist/{chunk-MXDG6SRH.js → chunk-JRDPT6DA.js} +2 -2
  45. package/dist/{chunk-BYLC4DKI.js → chunk-KMNYXLPL.js} +2 -2
  46. package/dist/{chunk-TQPSOTC7.js → chunk-L4N7SVNN.js} +8 -10
  47. package/dist/{chunk-TQPSOTC7.js.map → chunk-L4N7SVNN.js.map} +1 -1
  48. package/dist/{chunk-DYK27X7V.js → chunk-LMFS64GY.js} +3 -3
  49. package/dist/{chunk-FWXLKH2H.js → chunk-NVYF5OB6.js} +3 -3
  50. package/dist/{chunk-G5FP7VSQ.js → chunk-P4FW3ZU5.js} +58 -60
  51. package/dist/{chunk-G5FP7VSQ.js.map → chunk-P4FW3ZU5.js.map} +1 -1
  52. package/dist/{chunk-SODQQCAZ.js → chunk-PJE6D3NH.js} +10362 -4909
  53. package/dist/chunk-PJE6D3NH.js.map +1 -0
  54. package/dist/{chunk-56F6EAQ6.js → chunk-Q3X7GUAQ.js} +2 -2
  55. package/dist/{chunk-7N7S2RKF.js → chunk-R2OAN6IP.js} +2 -2
  56. package/dist/{chunk-CHELXDTO.js → chunk-RB6A7OUN.js} +2 -2
  57. package/dist/{chunk-3EA5LIFJ.js → chunk-SIV3JZPP.js} +2 -2
  58. package/dist/{chunk-IP5SANO2.js → chunk-UIKZZ242.js} +2 -2
  59. package/dist/{chunk-UGFPGWJB.js → chunk-URQLO7BX.js} +2 -2
  60. package/dist/{chunk-5UNSGRCR.js → chunk-WK3NK6SP.js} +162 -16
  61. package/dist/chunk-WK3NK6SP.js.map +1 -0
  62. package/dist/{chunk-VE54E6RP.js → chunk-WWVXXSLF.js} +4 -6
  63. package/dist/{chunk-VE54E6RP.js.map → chunk-WWVXXSLF.js.map} +1 -1
  64. package/dist/{chunk-WO22UEM3.js → chunk-XTEWVF3C.js} +2 -2
  65. package/dist/{chunk-FCQ7ZYCQ.js → chunk-ZGAJB74V.js} +2 -2
  66. package/dist/{chunk-NBP3YHHS.js → chunk-ZGZNLRKF.js} +2 -2
  67. package/dist/{chunk-C7C5WIIF.js → chunk-ZPZ4VVLT.js} +2 -2
  68. package/dist/{cli-BQiTZ7Zs.d.ts → cli-D1y3vooB.d.ts} +3 -3
  69. package/dist/cli.d.ts +9 -7
  70. package/dist/cli.js +44 -46
  71. package/dist/compounding/engine.d.ts +1 -1
  72. package/dist/compounding/engine.js +3 -3
  73. package/dist/compounding/preference-consolidator.d.ts +1 -1
  74. package/dist/compression-optimizer.d.ts +1 -1
  75. package/dist/config.d.ts +1 -1
  76. package/dist/connectors/codex-materialize-runner.d.ts +1 -1
  77. package/dist/connectors/codex-materialize-runner.js +3 -3
  78. package/dist/connectors/codex-materialize.d.ts +1 -1
  79. package/dist/connectors/index.d.ts +1 -1
  80. package/dist/connectors/index.js +3 -3
  81. package/dist/consolidation-provenance-check.d.ts +1 -1
  82. package/dist/consolidation-undo.d.ts +1 -1
  83. package/dist/contradiction/index.d.ts +1 -1
  84. package/dist/contradiction/index.js +4 -4
  85. package/dist/conversation-index/backend.d.ts +1 -1
  86. package/dist/conversation-index/chunker.d.ts +1 -1
  87. package/dist/conversation-index/faiss-adapter.d.ts +1 -1
  88. package/dist/conversation-index/indexer.d.ts +1 -1
  89. package/dist/conversation-index/search.d.ts +1 -1
  90. package/dist/day-summary.d.ts +1 -1
  91. package/dist/delinearize.d.ts +1 -1
  92. package/dist/direct-answer-wiring.d.ts +1 -1
  93. package/dist/direct-answer.d.ts +1 -1
  94. package/dist/embedding-fallback.d.ts +1 -1
  95. package/dist/enrichment/index.d.ts +1 -1
  96. package/dist/entity-retrieval.d.ts +1 -1
  97. package/dist/entity-retrieval.js +3 -3
  98. package/dist/entity-schema.d.ts +1 -1
  99. package/dist/explicit-capture.d.ts +6 -4
  100. package/dist/extraction-faithfulness.d.ts +1 -1
  101. package/dist/extraction-judge-telemetry.d.ts +1 -1
  102. package/dist/extraction-judge-training.d.ts +1 -1
  103. package/dist/extraction-judge.d.ts +1 -1
  104. package/dist/extraction.d.ts +1 -1
  105. package/dist/fallback-llm.d.ts +1 -1
  106. package/dist/graph-dashboard-diff.d.ts +1 -1
  107. package/dist/graph-dashboard-key.d.ts +1 -1
  108. package/dist/graph-dashboard-parser.d.ts +1 -1
  109. package/dist/graph-edge-reinforcement.d.ts +1 -1
  110. package/dist/graph-snapshot.d.ts +1 -1
  111. package/dist/graph.d.ts +1 -1
  112. package/dist/identity-continuity.d.ts +1 -1
  113. package/dist/importance.d.ts +1 -1
  114. package/dist/index.d.ts +12 -10
  115. package/dist/index.js +271 -5362
  116. package/dist/index.js.map +1 -1
  117. package/dist/intent.d.ts +1 -1
  118. package/dist/lcm/engine.d.ts +1 -1
  119. package/dist/lcm/index.d.ts +1 -1
  120. package/dist/lcm/tools.d.ts +1 -1
  121. package/dist/lifecycle.d.ts +1 -1
  122. package/dist/live-connectors-runner.d.ts +1 -1
  123. package/dist/local-llm.d.ts +1 -1
  124. package/dist/local-model-endpoint.d.ts +1 -1
  125. package/dist/maintenance/memory-governance.d.ts +1 -1
  126. package/dist/maintenance/memory-governance.js +3 -3
  127. package/dist/maintenance/rebuild-memory-lifecycle-ledger.js +3 -3
  128. package/dist/maintenance/rebuild-memory-projection.js +4 -4
  129. package/dist/mcp-memory-inspector-app.d.ts +16 -14
  130. package/dist/memory-action-policy.d.ts +1 -1
  131. package/dist/memory-cache.d.ts +1 -1
  132. package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
  133. package/dist/memory-projection-store.d.ts +1 -1
  134. package/dist/memory-provenance.d.ts +1 -1
  135. package/dist/memory-worth-outcomes.d.ts +1 -1
  136. package/dist/models-json.d.ts +1 -1
  137. package/dist/namespaces/migrate.d.ts +2 -2
  138. package/dist/namespaces/migrate.js +4 -4
  139. package/dist/namespaces/principal.d.ts +1 -1
  140. package/dist/namespaces/search.d.ts +1 -1
  141. package/dist/namespaces/storage.d.ts +2 -2
  142. package/dist/namespaces/storage.js +3 -3
  143. package/dist/native-knowledge.d.ts +1 -1
  144. package/dist/operator-toolkit.d.ts +1 -1
  145. package/dist/operator-toolkit.js +10 -11
  146. package/dist/orchestration/compression-guideline-coordinator.d.ts +1 -1
  147. package/dist/orchestration/maintenance.d.ts +2 -2
  148. package/dist/orchestration/maintenance.js +5 -5
  149. package/dist/{orchestrator-BPQGHpJz.d.ts → orchestrator-CnCE3Hvq.d.ts} +180 -12
  150. package/dist/orchestrator.d.ts +6 -4
  151. package/dist/orchestrator.js +119 -16
  152. package/dist/patterns-cli.d.ts +1 -1
  153. package/dist/policy-runtime.d.ts +1 -1
  154. package/dist/provenance.d.ts +1 -1
  155. package/dist/qmd-recall-cache.d.ts +1 -1
  156. package/dist/qmd.d.ts +1 -1
  157. package/dist/recall-disclosure-escalation.d.ts +1 -1
  158. package/dist/recall-explain-renderer.d.ts +1 -1
  159. package/dist/recall-planner-llm.d.ts +1 -1
  160. package/dist/recall-state.d.ts +1 -1
  161. package/dist/recall-tag-filter.d.ts +1 -1
  162. package/dist/recall-xray-cli.d.ts +1 -1
  163. package/dist/recall-xray-renderer.d.ts +1 -1
  164. package/dist/recall-xray.d.ts +1 -1
  165. package/dist/resolve-auth-token.d.ts +1 -1
  166. package/dist/retrieval-agents.d.ts +1 -1
  167. package/dist/retrieval-tiers.d.ts +1 -1
  168. package/dist/routing/engine.d.ts +1 -1
  169. package/dist/routing/store.d.ts +1 -1
  170. package/dist/search/embed-helper.d.ts +1 -1
  171. package/dist/search/factory.d.ts +1 -1
  172. package/dist/search/index.d.ts +1 -1
  173. package/dist/search/lancedb-backend.d.ts +1 -1
  174. package/dist/search/meilisearch-backend.d.ts +1 -1
  175. package/dist/search/noop-backend.d.ts +1 -1
  176. package/dist/search/orama-backend.d.ts +1 -1
  177. package/dist/search/port.d.ts +1 -1
  178. package/dist/search/remote-backend.d.ts +1 -1
  179. package/dist/{semantic-consolidation-ByzGwC9c.d.ts → semantic-consolidation-D3aSiu9t.d.ts} +1 -1
  180. package/dist/semantic-consolidation.d.ts +2 -2
  181. package/dist/semantic-consolidation.js +4 -4
  182. package/dist/semantic-rule-promotion.js +3 -3
  183. package/dist/semantic-rule-verifier.d.ts +1 -1
  184. package/dist/semantic-rule-verifier.js +3 -3
  185. package/dist/session-observer-bands.d.ts +1 -1
  186. package/dist/session-observer-state.d.ts +1 -1
  187. package/dist/shared-context/manager.d.ts +1 -1
  188. package/dist/signal.d.ts +1 -1
  189. package/dist/storage.d.ts +1 -1
  190. package/dist/storage.js +2 -2
  191. package/dist/summarizer.d.ts +1 -1
  192. package/dist/summary-snapshot.d.ts +1 -1
  193. package/dist/temporal-supersession.d.ts +1 -1
  194. package/dist/temporal-validity.d.ts +1 -1
  195. package/dist/threading.d.ts +1 -1
  196. package/dist/tier-migration.d.ts +1 -1
  197. package/dist/tier-routing.d.ts +1 -1
  198. package/dist/topics.d.ts +1 -1
  199. package/dist/transcript.d.ts +1 -1
  200. package/dist/transfer/backup.js +2 -2
  201. package/dist/transfer/capsule-export.js +2 -2
  202. package/dist/transfer/capsule-import.js +1 -1
  203. package/dist/trust-score-stage.d.ts +1 -1
  204. package/dist/trust-score.d.ts +1 -1
  205. package/dist/{types-5Jik3r4G.d.ts → types-CZm6rx9-.d.ts} +1 -1
  206. package/dist/types.d.ts +1 -1
  207. package/dist/utility-runtime.d.ts +1 -1
  208. package/dist/verified-recall.js +3 -3
  209. package/package.json +2 -2
  210. package/src/index.ts +2 -0
  211. package/src/orchestration/consolidation-run.ts +633 -0
  212. package/src/orchestration/extraction-persist.ts +2785 -0
  213. package/src/orchestration/extraction-run.ts +874 -0
  214. package/src/orchestrator.ts +206 -3815
  215. package/dist/chunk-5UNSGRCR.js.map +0 -1
  216. package/dist/chunk-A2KKLCOS.js +0 -1889
  217. package/dist/chunk-A2KKLCOS.js.map +0 -1
  218. package/dist/chunk-AVBT5WEW.js +0 -153
  219. package/dist/chunk-AVBT5WEW.js.map +0 -1
  220. package/dist/chunk-C7LDPZJJ.js.map +0 -1
  221. package/dist/chunk-SODQQCAZ.js.map +0 -1
  222. package/dist/chunk-WPLRJ4XM.js.map +0 -1
  223. /package/dist/{auto-sync-XHSKKQAH.js.map → auto-sync-IEBKPGAM.js.map} +0 -0
  224. /package/dist/{chunk-6XG3NQRC.js.map → chunk-3IJGF6K2.js.map} +0 -0
  225. /package/dist/{chunk-U5WVRMCA.js.map → chunk-5IN26V6W.js.map} +0 -0
  226. /package/dist/{chunk-XEADGIAA.js.map → chunk-DIHCFJVV.js.map} +0 -0
  227. /package/dist/{chunk-E7RGJQWP.js.map → chunk-HIHCB2S6.js.map} +0 -0
  228. /package/dist/{chunk-JFCS6PQU.js.map → chunk-JHQUIH3W.js.map} +0 -0
  229. /package/dist/{chunk-MXDG6SRH.js.map → chunk-JRDPT6DA.js.map} +0 -0
  230. /package/dist/{chunk-BYLC4DKI.js.map → chunk-KMNYXLPL.js.map} +0 -0
  231. /package/dist/{chunk-DYK27X7V.js.map → chunk-LMFS64GY.js.map} +0 -0
  232. /package/dist/{chunk-FWXLKH2H.js.map → chunk-NVYF5OB6.js.map} +0 -0
  233. /package/dist/{chunk-56F6EAQ6.js.map → chunk-Q3X7GUAQ.js.map} +0 -0
  234. /package/dist/{chunk-7N7S2RKF.js.map → chunk-R2OAN6IP.js.map} +0 -0
  235. /package/dist/{chunk-CHELXDTO.js.map → chunk-RB6A7OUN.js.map} +0 -0
  236. /package/dist/{chunk-3EA5LIFJ.js.map → chunk-SIV3JZPP.js.map} +0 -0
  237. /package/dist/{chunk-IP5SANO2.js.map → chunk-UIKZZ242.js.map} +0 -0
  238. /package/dist/{chunk-UGFPGWJB.js.map → chunk-URQLO7BX.js.map} +0 -0
  239. /package/dist/{chunk-WO22UEM3.js.map → chunk-XTEWVF3C.js.map} +0 -0
  240. /package/dist/{chunk-FCQ7ZYCQ.js.map → chunk-ZGAJB74V.js.map} +0 -0
  241. /package/dist/{chunk-NBP3YHHS.js.map → chunk-ZGZNLRKF.js.map} +0 -0
  242. /package/dist/{chunk-C7C5WIIF.js.map → chunk-ZPZ4VVLT.js.map} +0 -0
@@ -0,0 +1,2785 @@
1
+ /**
2
+ * Extraction-persist coordinator — extracted from the orchestrator
3
+ * (issue #1526, seam 16).
4
+ *
5
+ * Owns the `persistExtraction` pipeline: the ~2.5k-LOC method that turns
6
+ * an ExtractionResult into durable memory writes. Hosts:
7
+ * - pre-judge multi-namespace redaction gate
8
+ * - extraction-judge gating
9
+ * - scope routing
10
+ * - tombstone postWriteGuard
11
+ * - dedup + bitemporal backfill (3 paths)
12
+ * - contradiction auto-resolve deferral
13
+ * - promotion paths
14
+ * - post-persist content-hash indexing
15
+ * - catalog write-recording
16
+ *
17
+ * Behavior-preserving move from orchestrator.ts. No logic changes — the
18
+ * orchestrator keeps a thin delegating method so existing call sites and
19
+ * tests continue to work.
20
+ */
21
+
22
+ import path from "node:path";
23
+ import {
24
+ StorageManager,
25
+ ContentHashIndex,
26
+ normalizeAttributePairs,
27
+ } from "../index.js";
28
+ import { log } from "../logger.js";
29
+ import { chunkContent, type ChunkingConfig } from "../chunking.js";
30
+ import { semanticChunkContent, type SemanticChunkResult } from "../semantic-chunking.js";
31
+ import { isAboveImportanceThreshold, scoreImportance } from "../importance.js";
32
+ import { sanitizeMemoryContent } from "../sanitize.js";
33
+ import {
34
+ resolveGraphConstructionCapabilities,
35
+ resolveMemoryLifecycleCapabilities,
36
+ resolvePipelineProcessingCapabilities,
37
+ resolvePresentationCapabilities,
38
+ resolveNamespaceCapabilities,
39
+ resolveRecallEnhancementCapabilities,
40
+ resolveConversationContextCapabilities,
41
+ type GraphConstructionCapabilitySet,
42
+ type MemoryLifecycleCapabilitySet,
43
+ } from "../capabilities.js";
44
+ import {
45
+ applyTemporalSupersession,
46
+ normalizeSupersessionKey,
47
+ } from "../temporal-supersession.js";
48
+ import { pickFactEventTimeAnchor, resolveFactEventTime } from "../event-time.js";
49
+ import {
50
+ judgeFactDurability,
51
+ getVerdictKind,
52
+ validateProcedureExtraction,
53
+ type JudgeVerdict,
54
+ type JudgeCandidate,
55
+ } from "../extraction-judge.js";
56
+ import {
57
+ EXTRACTION_JUDGE_VERDICT_CATEGORY,
58
+ recordJudgeVerdict,
59
+ } from "../extraction-judge-telemetry.js";
60
+ import { recordJudgeTrainingPair } from "../extraction-judge-training.js";
61
+ import {
62
+ applyFaithfulnessVerdict,
63
+ runFaithfulnessGateBatch,
64
+ type FaithfulnessGateCounters,
65
+ } from "../extraction-faithfulness.js";
66
+ import {
67
+ contentMatchesRedactionRules,
68
+ loadRedactionRules,
69
+ type CompiledRedactionRule,
70
+ } from "../extraction-redaction-rules.js";
71
+ import {
72
+ attachCitation,
73
+ type CitationContext,
74
+ hasCitationForTemplate,
75
+ stripCitationForTemplate,
76
+ } from "../source-attribution.js";
77
+ import { classifyMemoryKind } from "../himem.js";
78
+ import {
79
+ buildBehaviorSignalsForMemory,
80
+ dedupeBehaviorSignalsByMemoryAndHash,
81
+ } from "../behavior-signals.js";
82
+ import { buildProcedurePersistBody } from "../procedural/procedure-types.js";
83
+ import { LocalLlmClient } from "../local-llm.js";
84
+ import {
85
+ FallbackLlmClient,
86
+ fallbackLlmRuntimeContextFromConfig,
87
+ } from "../fallback-llm.js";
88
+ import { EmbeddingFallback } from "../embedding-fallback.js";
89
+ import { decideSemanticDedup, type SemanticDedupDecision, type SemanticDedupHit } from "../dedup/semantic.js";
90
+ import { selectRouteRule, type RouteRule, type RoutingEngineOptions } from "../routing/engine.js";
91
+ import { ThreadingManager } from "../threading.js";
92
+ import { NamespaceStorageRouter } from "../namespaces/storage.js";
93
+ import type { SearchBackend } from "../search/port.js";
94
+ import { inferIntentFromText } from "../intent.js";
95
+ import type {
96
+ BehaviorSignalEvent,
97
+ ExtractionResult,
98
+ MemoryFile,
99
+ MemoryFrontmatter,
100
+ MemoryLink,
101
+ PluginConfig,
102
+ ProvenanceSource,
103
+ } from "../types.js";
104
+ import { confidenceTier } from "../types.js";
105
+ import type { ResolvedScopeProfilePlan } from "../namespaces/scope-profiles.js";
106
+ import {
107
+ buildMemoryPathById,
108
+ appendMemoryToGraphContext,
109
+ resolvePersistedMemoryRelativePath,
110
+ } from "../orchestrator.js";
111
+
112
+ export interface ExtractionPersistDeps {
113
+ config: PluginConfig;
114
+ getStorageRouter: () => NamespaceStorageRouter;
115
+ getThreading: () => ThreadingManager;
116
+ getLocalLlm: () => LocalLlmClient;
117
+ getQmd: () => SearchBackend;
118
+ getJudgeVerdictCache: () => Map<string, JudgeVerdict>;
119
+ getJudgeDeferCounts: () => Map<string, number>;
120
+ getFaithfulnessCounters: () => FaithfulnessGateCounters;
121
+ getEmbeddingFallback: () => EmbeddingFallback;
122
+ setLastPersistExtractionDeferredCount: (value: number) => void;
123
+ setLastPersistExtractionPendingReviewIds: (ids: string[]) => void;
124
+ addContentHashDedup: (targetStorage: StorageManager, content: string) => Promise<void>;
125
+ hasContentHashDedup: (targetStorage: StorageManager, content: string) => Promise<boolean>;
126
+ backfillTemporalBoundsOnDedupHit: (
127
+ targetStorage: StorageManager,
128
+ dedupContent: string,
129
+ bounds: {
130
+ invalidAt?: string;
131
+ validFrom?: string;
132
+ observedAt?: string;
133
+ eventTimeSource?: "extracted" | "assumed";
134
+ },
135
+ entityRef?: string,
136
+ ) => Promise<void>;
137
+ saveContentHashIndexes: () => Promise<void>;
138
+ artifactTypeForCategory: (
139
+ category: string,
140
+ ) =>
141
+ | "decision"
142
+ | "constraint"
143
+ | "todo"
144
+ | "definition"
145
+ | "commitment"
146
+ | "correction"
147
+ | "fact";
148
+ loadRoutingRules: () => Promise<RouteRule[]>;
149
+ routeEngineOptions: () => RoutingEngineOptions;
150
+ semanticDedupLookup: (
151
+ content: string,
152
+ limit: number,
153
+ targetStorage: StorageManager,
154
+ ) => Promise<SemanticDedupHit[]>;
155
+ checkForContradiction: (
156
+ content: string,
157
+ category: string,
158
+ namespaceScope: string,
159
+ ) => Promise<{
160
+ supersededId: string;
161
+ confidence: number;
162
+ reason: string;
163
+ supersededPath: string;
164
+ supersededCreated: string;
165
+ supersededTags: string[];
166
+ } | null>;
167
+ applyDeferredContradictionResolve: (
168
+ contradiction: {
169
+ supersededId: string;
170
+ reason: string;
171
+ supersededPath: string;
172
+ supersededCreated: string;
173
+ supersededTags: string[];
174
+ } | null | undefined,
175
+ storage: StorageManager,
176
+ newMemoryId: string,
177
+ postWriteGuard: boolean,
178
+ ) => Promise<void>;
179
+ suggestLinksForMemory: (
180
+ content: string,
181
+ category: string,
182
+ namespaceScope: string,
183
+ ) => Promise<MemoryLink[]>;
184
+ storageDirNamespace: (storageDir: string) => string;
185
+ indexPersistedMemory: (storage: StorageManager, memoryId: string) => Promise<void>;
186
+ buildGraphEdge: (
187
+ storage: StorageManager,
188
+ memoryRelPath: string,
189
+ entityRef: string | undefined,
190
+ memoryId: string,
191
+ factContent: string,
192
+ allMemsForGraph: MemoryFile[] | null | undefined,
193
+ memoryPathById: Map<string, string>,
194
+ threadIdForEdge: string | undefined,
195
+ threadEpisodeIdsForGraph: string[] | undefined,
196
+ fallbackCausalPredecessor: string | undefined,
197
+ graphCaps?: GraphConstructionCapabilitySet,
198
+ ) => Promise<void>;
199
+ updateTemporalTagIndexes: (
200
+ storage: StorageManager,
201
+ persistedIds: string[],
202
+ ) => Promise<void>;
203
+ }
204
+
205
+ export class ExtractionPersistCoordinator {
206
+ constructor(
207
+ private readonly deps: ExtractionPersistDeps,
208
+ ) {}
209
+
210
+ private get config(): PluginConfig {
211
+ return this.deps.config;
212
+ }
213
+
214
+ async persistExtraction(
215
+ result: ExtractionResult,
216
+ storage: StorageManager,
217
+ threadIdForExtraction?: string | null,
218
+ sourceContext?: { sessionKey?: string; principal?: string; validAt?: string },
219
+ baseNamespace?: string,
220
+ scopeProfileWritePlan?: ResolvedScopeProfilePlan | null,
221
+ /** Verbatim source turn text the facts were extracted from (faithfulness gate #1576). */
222
+ sourceText?: string,
223
+ graphCaps: GraphConstructionCapabilitySet = resolveGraphConstructionCapabilities(this.deps.config),
224
+ lifecycleCaps: MemoryLifecycleCapabilitySet = resolveMemoryLifecycleCapabilities(this.deps.config),
225
+ ): Promise<string[]> {
226
+ // Inline source attribution (issue #369). When enabled, every extracted
227
+ // fact is rewritten to carry a compact provenance tag inside its body so
228
+ // the citation survives hostile memory text, copy/paste, and LLM quoting.
229
+ // The helper is a no-op when the feature flag is off, so legacy pipelines
230
+ // see zero behavioral change.
231
+ const citationEnabled = resolvePipelineProcessingCapabilities(this.deps.config).inlineSourceAttribution === true;
232
+ const citationTemplate = this.deps.config.inlineSourceAttributionFormat;
233
+ // The stable fields (agent, session) are computed once; `ts` is intentionally
234
+ // omitted here and added fresh per invocation so each fact in a large batch
235
+ // gets its own insertion timestamp rather than sharing a single batch-start time.
236
+ const citationContextBase: Omit<CitationContext, "ts"> = citationEnabled
237
+ ? {
238
+ agent: sourceContext?.principal,
239
+ session: sourceContext?.sessionKey,
240
+ }
241
+ : {};
242
+ const applyInlineCitation = (content: string): string => {
243
+ if (!citationEnabled) return content;
244
+ if (typeof content !== "string" || content.length === 0) return content;
245
+ // Build a fresh CitationContext per call so `ts` reflects the actual
246
+ // insertion time of each individual fact rather than the batch-start time.
247
+ const citationContext: CitationContext = {
248
+ ...citationContextBase,
249
+ ts: new Date().toISOString(),
250
+ };
251
+ // `attachCitation` already calls `hasCitationForTemplate` internally and
252
+ // is a no-op when the content already carries a citation (default or
253
+ // custom template). The outer check was redundant and has been removed
254
+ // to avoid a maintenance hazard where the two guard paths could diverge.
255
+ return attachCitation(content, citationContext, citationTemplate);
256
+ };
257
+ const persistedIds: string[] = [];
258
+ const supersessionOrderingAt = (validAt?: string): string =>
259
+ validAt && validAt.length > 0 ? validAt : new Date().toISOString();
260
+ // #1635: pending_review persisted ids, excluded from the thread episode set below.
261
+ const pendingReviewPersistedIds: string[] = [];
262
+ const persistedIdsByStorage = new Map<
263
+ string,
264
+ { storage: StorageManager; ids: string[] }
265
+ >();
266
+ const trackPersistedId = (
267
+ targetStorage: StorageManager,
268
+ id: string,
269
+ options: {
270
+ includeReturnedIds?: boolean;
271
+ /** #1635: keep this id out of the persisted thread episode set. */
272
+ pendingReview?: boolean;
273
+ } = {},
274
+ ): void => {
275
+ if (options.includeReturnedIds !== false) {
276
+ persistedIds.push(id);
277
+ }
278
+ if (options.pendingReview) {
279
+ pendingReviewPersistedIds.push(id);
280
+ }
281
+ const key = targetStorage.dir;
282
+ const existing = persistedIdsByStorage.get(key);
283
+ if (existing) {
284
+ existing.ids.push(id);
285
+ return;
286
+ }
287
+ persistedIdsByStorage.set(key, { storage: targetStorage, ids: [id] });
288
+ };
289
+ let dedupedCount = 0;
290
+ // Counter for facts skipped by the importance write-gate (issue #372).
291
+ // Emitted via the `importance_gated` metric below and rolled into the
292
+ // final `persisted:` log line so operators can tune the threshold.
293
+ let importanceGatedCount = 0;
294
+ // UUI2: short-circuit semantic dedup after first backend-unavailable signal
295
+ // within this batch. Once any fact in the batch gets reason="backend_unavailable"
296
+ // (meaning the embedding backend is degraded), subsequent facts skip the
297
+ // lookup entirely and proceed directly to write. This prevents N-fact batches
298
+ // from paying N × timeout when the backend is down. The flag resets per-batch
299
+ // (declared here, inside persistExtraction) so a transient hiccup in one
300
+ // batch does not permanently disable dedup in future batches.
301
+ let batchBackendUnavailable = false;
302
+ // #1669: per-namespace redaction-rule cache for this persist pass. A
303
+ // `never_store` / redaction_rule correction persists patterns under each
304
+ // namespace's state/corrections/redaction-rules/ dir; we consult them
305
+ // before a fact reaches the storage write chokepoint so matching content
306
+ // is withheld entirely rather than landing as pending_review. Cached per
307
+ // dir so a multi-fact batch over one namespace reads the dir once.
308
+ let redactionGatedCount = 0;
309
+ const redactionRulesByDir = new Map<string, CompiledRedactionRule[]>();
310
+ const redactionRulesFor = async (...dirs: string[]): Promise<CompiledRedactionRule[]> => {
311
+ const out: CompiledRedactionRule[] = [];
312
+ for (const d of dirs) {
313
+ let r = redactionRulesByDir.get(d);
314
+ if (!r) { r = await loadRedactionRules(d); redactionRulesByDir.set(d, r); }
315
+ out.push(...r);
316
+ }
317
+ return out;
318
+ };
319
+ const behaviorSignalsByStorage = new Map<
320
+ string,
321
+ { storage: StorageManager; events: BehaviorSignalEvent[] }
322
+ >();
323
+ const trackBehaviorSignals = (
324
+ targetStorage: StorageManager,
325
+ events: BehaviorSignalEvent[],
326
+ ): void => {
327
+ if (events.length === 0) return;
328
+ const key = targetStorage.dir;
329
+ const existing = behaviorSignalsByStorage.get(key);
330
+ if (existing) {
331
+ existing.events.push(...events);
332
+ return;
333
+ }
334
+ behaviorSignalsByStorage.set(key, {
335
+ storage: targetStorage,
336
+ events: [...events],
337
+ });
338
+ };
339
+ const confidenceTierOrder = [
340
+ "explicit",
341
+ "implied",
342
+ "inferred",
343
+ "speculative",
344
+ ] as const;
345
+ const sharedProfileLayer = scopeProfileWritePlan?.layers.find(
346
+ (layer) =>
347
+ layer.id === "serverShared" &&
348
+ layer.namespace === this.deps.config.sharedNamespace,
349
+ );
350
+ const sharedPromotionTarget = scopeProfileWritePlan?.promotionTargets.find(
351
+ (target) =>
352
+ target.target === "serverShared" &&
353
+ target.namespace === this.deps.config.sharedNamespace,
354
+ );
355
+ const profileAllowsSharedWrites =
356
+ !scopeProfileWritePlan ||
357
+ Boolean(
358
+ scopeProfileWritePlan.profile.readOrder.includes("serverShared") &&
359
+ scopeProfileWritePlan.readNamespaces.includes(this.deps.config.sharedNamespace) &&
360
+ sharedProfileLayer?.readable &&
361
+ sharedProfileLayer.writable &&
362
+ sharedPromotionTarget?.authorized,
363
+ );
364
+ // #1713: hoist namespaces-enabled once for the scope-routing mirrors
365
+ // (pre-judge + write-loop) so no new scattered config.*Enabled read is
366
+ // introduced (ratchet scatteredConfigFlagReads; see #1523).
367
+ const namespacesEnabled = resolveNamespaceCapabilities(this.deps.config).namespaces;
368
+ const profileAutoPromotionAllows = (
369
+ category: string,
370
+ confidence: number,
371
+ ): boolean => {
372
+ if (!scopeProfileWritePlan) return false;
373
+ const actualTier = confidenceTier(confidence);
374
+ const actualRank = confidenceTierOrder.indexOf(actualTier);
375
+ if (actualRank === -1) return false;
376
+ const autoPromote = scopeProfileWritePlan.profile.autoPromote;
377
+ if (!autoPromote.enabled) return false;
378
+ if (!autoPromote.categories.includes(category as any)) return false;
379
+ const minimumRank = confidenceTierOrder.indexOf(autoPromote.minConfidenceTier);
380
+ return minimumRank !== -1 && actualRank <= minimumRank;
381
+ };
382
+ const sharedAutoPromotionAllows = (
383
+ category: string,
384
+ confidence: number,
385
+ ): boolean => {
386
+ if (!scopeProfileWritePlan) {
387
+ const actualTier = confidenceTier(confidence);
388
+ const actualRank = confidenceTierOrder.indexOf(actualTier);
389
+ if (actualRank === -1) return false;
390
+ if (!resolveRecallEnhancementCapabilities(this.deps.config).autoPromoteToShared) return false;
391
+ if (!this.deps.config.autoPromoteToSharedCategories.includes(category as any))
392
+ return false;
393
+ const minimumRank = confidenceTierOrder.indexOf(
394
+ this.deps.config.autoPromoteMinConfidenceTier,
395
+ );
396
+ return minimumRank !== -1 && actualRank <= minimumRank;
397
+ }
398
+ return (
399
+ scopeProfileWritePlan.profile.autoPromote.targets.includes("serverShared") &&
400
+ profileAutoPromotionAllows(category, confidence)
401
+ );
402
+ };
403
+ const shouldPromoteToShared = (
404
+ targetStorage: StorageManager,
405
+ category: string,
406
+ confidence: number,
407
+ ): boolean => {
408
+ if (
409
+ !resolveNamespaceCapabilities(this.deps.config).namespaces ||
410
+ !profileAllowsSharedWrites ||
411
+ !sharedAutoPromotionAllows(category, confidence)
412
+ )
413
+ return false;
414
+ if (
415
+ this.deps.storageDirNamespace(targetStorage.dir) ===
416
+ this.deps.config.sharedNamespace
417
+ )
418
+ return false;
419
+ return true;
420
+ };
421
+ const promoteMemoryToProfileTargets = async (options: {
422
+ sourceStorage: StorageManager;
423
+ category: string;
424
+ content: string;
425
+ confidence: number;
426
+ tags: string[];
427
+ entityRef?: string;
428
+ structuredAttributes?: Record<string, string>;
429
+ sourceMemoryId: string;
430
+ importance?: ReturnType<typeof scoreImportance>;
431
+ intentGoal?: string;
432
+ intentActionType?: string;
433
+ intentEntityTypes?: string[];
434
+ memoryKind?: MemoryFrontmatter["memoryKind"];
435
+ validAt?: string;
436
+ // #1578 — bi-temporal bounds + ingestion provenance forwarded to profile-
437
+ // target copies (same defect class as shared promotion; cursor bugbot).
438
+ invalidAt?: string;
439
+ observedAt?: string;
440
+ eventTimeSource?: "extracted" | "assumed";
441
+ source: string;
442
+ sources?: ProvenanceSource[];
443
+ provenance?: "verified" | "unverified" | "none";
444
+ }): Promise<void> => {
445
+ if (
446
+ !scopeProfileWritePlan ||
447
+ !profileAutoPromotionAllows(options.category, options.confidence)
448
+ )
449
+ return;
450
+ const autoTargets = new Set(scopeProfileWritePlan.profile.autoPromote.targets);
451
+ const targets = scopeProfileWritePlan.promotionTargets.filter(
452
+ (target) =>
453
+ target.target !== "serverShared" &&
454
+ autoTargets.has(target.target) &&
455
+ target.authorized &&
456
+ target.namespace,
457
+ );
458
+ if (targets.length === 0) return;
459
+ const rawContent =
460
+ citationEnabled && hasCitationForTemplate(options.content, citationTemplate)
461
+ ? stripCitationForTemplate(options.content, citationTemplate)
462
+ : options.content;
463
+ const citedContent = applyInlineCitation(rawContent);
464
+ const sanitizedBase = sanitizeMemoryContent(rawContent);
465
+ const dedupContent =
466
+ options.category === "fact" &&
467
+ options.structuredAttributes &&
468
+ Object.keys(options.structuredAttributes).length > 0
469
+ ? `${sanitizedBase.text}\n[Attributes: ${normalizeAttributePairs(options.structuredAttributes)}]`
470
+ : sanitizedBase.text;
471
+ for (const target of targets) {
472
+ if (!target.namespace) continue;
473
+ try {
474
+ const targetStorage = await this.deps.getStorageRouter().storageFor(target.namespace);
475
+ if (targetStorage.dir === options.sourceStorage.dir) continue;
476
+ if (
477
+ options.category === "fact" &&
478
+ (await targetStorage.hasFactContentHash(dedupContent))
479
+ ) {
480
+ // #1671 — backfill bi-temporal bounds the existing promoted copy
481
+ // lacks (re-extraction with a now-resolved invalidAt). Best-effort,
482
+ // fail-open; the helper gates on invalidAt to avoid I/O when no
483
+ // end bound is present.
484
+ if (options.invalidAt || options.validAt) {
485
+ await this.deps.backfillTemporalBoundsOnDedupHit(
486
+ targetStorage,
487
+ dedupContent,
488
+ {
489
+ invalidAt: options.invalidAt,
490
+ // #1707 thread 2 — carry the corrected start bound.
491
+ validFrom: options.validAt,
492
+ ...(options.observedAt ? { observedAt: options.observedAt } : {}),
493
+ ...(options.eventTimeSource ? { eventTimeSource: options.eventTimeSource } : {}),
494
+ },
495
+ options.entityRef,
496
+ );
497
+ }
498
+ continue;
499
+ }
500
+ const targetPromotion = await targetStorage.writeMemory(
501
+ options.category as any,
502
+ citedContent,
503
+ {
504
+ confidence: options.confidence,
505
+ tags: [...options.tags, `${target.target}-promotion`],
506
+ entityRef: options.entityRef,
507
+ structuredAttributes: options.structuredAttributes,
508
+ source: `${options.source}-${target.target}-promotion`,
509
+ importance: options.importance,
510
+ lineage: [options.sourceMemoryId],
511
+ sourceMemoryId: options.sourceMemoryId,
512
+ intentGoal: options.intentGoal,
513
+ intentActionType: options.intentActionType,
514
+ intentEntityTypes: options.intentEntityTypes,
515
+ memoryKind: options.memoryKind,
516
+ validAt: options.validAt,
517
+ // #1578 — forward bi-temporal bounds + ingestion provenance.
518
+ ...(options.invalidAt ? { invalidAt: options.invalidAt } : {}),
519
+ ...(options.observedAt ? { observedAt: options.observedAt } : {}),
520
+ ...(options.eventTimeSource ? { eventTimeSource: options.eventTimeSource } : {}),
521
+ contentHashSource: options.category === "fact" ? dedupContent : rawContent,
522
+ ...(options.sources && options.sources.length > 0 ? { sources: options.sources } : {}),
523
+ ...(options.provenance ? { provenance: options.provenance } : {}),
524
+ },
525
+ );
526
+ const promotedId = targetPromotion.id;
527
+ // #1645: if the TARGET namespace's own tombstone blocked this promotion,
528
+ // the row lands pending_review — do NOT supersede active target memories.
529
+ if (
530
+ !targetPromotion.tombstoneBlocked &&
531
+ lifecycleCaps.temporalSupersession &&
532
+ options.category === "fact" &&
533
+ options.entityRef &&
534
+ options.structuredAttributes &&
535
+ Object.keys(options.structuredAttributes).length > 0
536
+ ) {
537
+ try {
538
+ await applyTemporalSupersession({
539
+ storage: targetStorage,
540
+ newMemoryId: promotedId,
541
+ entityRef: options.entityRef,
542
+ structuredAttributes: options.structuredAttributes,
543
+ createdAt: supersessionOrderingAt(options.validAt),
544
+ enabled: !(options.eventTimeSource === "extracted" && !options.validAt),
545
+ });
546
+ } catch (profileSupersessionErr) {
547
+ log.warn(
548
+ `persistExtraction: ${target.target} promotion temporal supersession failed open for promoted ${promotedId}: ${profileSupersessionErr}`,
549
+ );
550
+ }
551
+ }
552
+ // #1645 TV6: a tombstone-blocked promotion is pending_review (no
553
+ // active copy) — skip catalog/index/behavior like postWriteGuard.
554
+ if (!targetPromotion.tombstoneBlocked) {
555
+ trackPersistedId(targetStorage, promotedId, { includeReturnedIds: false });
556
+ await this.deps.indexPersistedMemory(targetStorage, promotedId);
557
+ trackBehaviorSignals(
558
+ targetStorage,
559
+ buildBehaviorSignalsForMemory({
560
+ memoryId: promotedId,
561
+ category: options.category as any,
562
+ content: options.content,
563
+ namespace: target.namespace,
564
+ confidence: options.confidence,
565
+ source: "extraction",
566
+ }),
567
+ );
568
+ }
569
+ } catch (err) {
570
+ log.warn(
571
+ `persistExtraction: ${target.target} promotion failed open for ${options.sourceMemoryId}: ${err}`,
572
+ );
573
+ }
574
+ }
575
+ };
576
+ const promoteMemoryToShared = async (options: {
577
+ sourceStorage: StorageManager;
578
+ category: string;
579
+ content: string;
580
+ confidence: number;
581
+ tags: string[];
582
+ entityRef?: string;
583
+ structuredAttributes?: Record<string, string>;
584
+ sourceMemoryId: string;
585
+ importance?: ReturnType<typeof scoreImportance>;
586
+ intentGoal?: string;
587
+ intentActionType?: string;
588
+ intentEntityTypes?: string[];
589
+ memoryKind?: MemoryFrontmatter["memoryKind"];
590
+ validAt?: string;
591
+ // #1578 — bi-temporal bounds + ingestion provenance forwarded to the
592
+ // shared-namespace copy so shared recall honours the same invalid_at
593
+ // window as the source fact (cursor bugbot).
594
+ invalidAt?: string;
595
+ observedAt?: string;
596
+ eventTimeSource?: "extracted" | "assumed";
597
+ source: string;
598
+ /** Claim-level provenance spans (issue #1575 PR 2). */
599
+ sources?: ProvenanceSource[];
600
+ provenance?: "verified" | "unverified" | "none";
601
+ }): Promise<void> => {
602
+ await promoteMemoryToProfileTargets(options);
603
+ if (
604
+ !shouldPromoteToShared(
605
+ options.sourceStorage,
606
+ options.category,
607
+ options.confidence,
608
+ )
609
+ )
610
+ return;
611
+ try {
612
+ const sharedStorage = await this.deps.getStorageRouter().storageFor(
613
+ this.deps.config.sharedNamespace,
614
+ );
615
+ // Dedup gate: canonicalize content before hashing.
616
+ //
617
+ // Issue #369 (PR #401): When inline attribution is enabled,
618
+ // `applyInlineCitation` appends a timestamp-bearing marker (e.g.
619
+ // `[Source: ..., ts=2026-04-11T...]`). Because the timestamp changes
620
+ // on every call, hashing cited content produces a unique hash each
621
+ // time — defeating dedup entirely and allowing the same logical fact to
622
+ // be promoted repeatedly. Also, both promotion call sites pass
623
+ // `fact.content`, which can already carry an inline citation (e.g. a
624
+ // relayed or reprocessed fact). Strip any pre-existing citation so the
625
+ // dedup key matches the hash stored from the original un-cited write.
626
+ //
627
+ // PR #402 round-6 (Fix #2 / chatgpt-codex P1 PRRT_kwDORJXyws56U74n):
628
+ // Compute the enriched content before the hash-dedup check so the
629
+ // lookup uses the same content that writeMemory will actually store.
630
+ // When structuredAttributes are present, writeMemory appends an
631
+ // "[Attributes: ...]" suffix before hashing; hasFactContentHash must
632
+ // receive the same enriched body or the check is against a different
633
+ // hash and dedup fails to fire (letting duplicates through) or fires
634
+ // when it shouldn't (collapsing memories with different enrichments).
635
+ // Fix #1 (P2 PRRT_kwDORJXyws56VHZc): use normalizeAttributePairs so
636
+ // key order and casing are canonical — identical to the enrichment
637
+ // applied by storage.writeMemory — preventing spurious hash misses
638
+ // when attribute maps arrive with different insertion orders or casing.
639
+ //
640
+ // Fix #4 (Low PRRT_kwDORJXyws56VHth): sanitize the base content before
641
+ // building dedupContent. writeMemory runs sanitizeMemoryContent on the
642
+ // enriched body before hashing; if sanitization redacts the content to
643
+ // REDACTED_PLACEHOLDER the stored hash is for the redacted form, not
644
+ // the raw form. Computing dedupContent from sanitized.text here ensures
645
+ // the hash lookup and the normalizedIncoming comparison both use the
646
+ // same content that writeMemory will actually store.
647
+ //
648
+ // Combined fix: strip any pre-existing citation FIRST to obtain
649
+ // rawContent (the canonical body), then sanitize rawContent (not
650
+ // options.content) when building dedupContent, so that citation
651
+ // stripping and sanitization are applied in a consistent order.
652
+ const rawContent =
653
+ citationEnabled &&
654
+ hasCitationForTemplate(options.content, citationTemplate)
655
+ ? stripCitationForTemplate(options.content, citationTemplate)
656
+ : options.content;
657
+ const citedContent = applyInlineCitation(rawContent);
658
+ const sanitizedBase = sanitizeMemoryContent(rawContent);
659
+ const dedupContent =
660
+ options.category === "fact" &&
661
+ options.structuredAttributes &&
662
+ Object.keys(options.structuredAttributes).length > 0
663
+ ? `${sanitizedBase.text}\n[Attributes: ${normalizeAttributePairs(options.structuredAttributes)}]`
664
+ : sanitizedBase.text;
665
+ if (
666
+ options.category === "fact" &&
667
+ (await sharedStorage.hasFactContentHash(dedupContent))
668
+ ) {
669
+ // #1671 — backfill bi-temporal bounds onto the existing shared copy
670
+ // before the supersession short-circuit. Covers all return paths below
671
+ // (supersession-hit, catch-skip, and the no-supersession short-circuit)
672
+ // in one shot. Best-effort / fail-open; the helper gates on invalidAt.
673
+ if (options.invalidAt || options.validAt) {
674
+ await this.deps.backfillTemporalBoundsOnDedupHit(
675
+ sharedStorage,
676
+ dedupContent,
677
+ {
678
+ invalidAt: options.invalidAt,
679
+ // #1707 thread 2 — carry the corrected start bound.
680
+ validFrom: options.validAt,
681
+ ...(options.observedAt ? { observedAt: options.observedAt } : {}),
682
+ ...(options.eventTimeSource ? { eventTimeSource: options.eventTimeSource } : {}),
683
+ },
684
+ options.entityRef,
685
+ );
686
+ }
687
+ // Uj6H fix: shared-namespace temporal supersession must also run when
688
+ // the hash-dedup short-circuit fires. Without this, an existing shared
689
+ // fact whose structuredAttributes are stale (or an older conflicting
690
+ // shared fact that is still active) never gets retired — supersession
691
+ // only ran in the post-writeMemory block which is unreachable here.
692
+ //
693
+ // Strategy: scan the shared namespace for the existing fact whose
694
+ // normalized content matches the incoming content, then run
695
+ // applyTemporalSupersession against it using the same logic that
696
+ // would have run post-writeMemory. This is a best-effort / fail-open
697
+ // step — if the lookup fails we skip silently (same as the normal path).
698
+ if (
699
+ lifecycleCaps.temporalSupersession &&
700
+ options.entityRef &&
701
+ options.structuredAttributes &&
702
+ Object.keys(options.structuredAttributes).length > 0
703
+ ) {
704
+ // PR #402 round-7 (Fix #2 / Codex P1 PRRT_kwDORJXyws56VALC):
705
+ // Track whether matchingFact lookup completed before the try block
706
+ // so the catch block can distinguish an early-lookup failure (where
707
+ // we don't know if a duplicate exists) from a post-lookup supersession
708
+ // failure (where we confirmed a duplicate and must skip the write).
709
+ let hashDedupMatchingFact: MemoryFile | undefined;
710
+ let hashDedupLookupComplete = false;
711
+ try {
712
+ // Fix #2 (P2 PRRT_kwDORJXyws56VHZf): dedupContent is now built
713
+ // from sanitizedBase.text (see fix #4 above), so normalizedIncoming
714
+ // uses the same sanitized+normalized content that writeMemory hashes
715
+ // and that hasFactContentHash just matched. Previously this used the
716
+ // raw options.content, which diverged from the stored hash when
717
+ // sanitization redacted the content, causing the candidate lookup to
718
+ // return undefined and leaving stale facts active.
719
+ const normalizedIncoming = ContentHashIndex.normalizeContent(dedupContent);
720
+ const allShared = await sharedStorage.readAllMemories();
721
+ // PR #402 round-12 (Finding Uybg): restrict hash-dedup matching to
722
+ // the SAME entity. Content-hash equality alone can collide across
723
+ // entities when two entities share identical fact text. Using an
724
+ // unrelated entity's existing fact as `newMemoryId` would anchor
725
+ // supersession to that entity's record and corrupt its
726
+ // `supersededBy` links. Only consider facts whose normalized
727
+ // `entityRef` matches the incoming entity.
728
+ const incomingEntityNorm = normalizeSupersessionKey(options.entityRef);
729
+ hashDedupMatchingFact = allShared.find((m) => {
730
+ if (m.frontmatter.category !== "fact") return false;
731
+ if ((m.frontmatter.status ?? "active") !== "active") return false;
732
+ // Same-entity guard: skip if entity doesn't match.
733
+ if (!m.frontmatter.entityRef) return false;
734
+ if (normalizeSupersessionKey(m.frontmatter.entityRef) !== incomingEntityNorm) {
735
+ log.debug(
736
+ `persistExtraction: hash-dedup skipping cross-entity match (incoming="${incomingEntityNorm}" candidate="${normalizeSupersessionKey(m.frontmatter.entityRef)}")`,
737
+ );
738
+ return false;
739
+ }
740
+ // PR #402 round-7 (Fix #2): compare stored fact's full body
741
+ // (including any appended "[Attributes: ...]" suffix) against the
742
+ // enriched normalizedIncoming so the candidate selected is the one
743
+ // whose hash actually matched in hasFactContentHash.
744
+ return ContentHashIndex.normalizeContent(m.content ?? "") === normalizedIncoming;
745
+ });
746
+ hashDedupLookupComplete = true;
747
+ if (hashDedupMatchingFact) {
748
+ // Finding UvU1 (PR #402 round-11): anchor supersession to the
749
+ // incoming event's time, not the existing fact's persisted
750
+ // `created`. For source-dated replay/import, this is the
751
+ // source valid_at; otherwise it is the current wall-clock. The
752
+ // matching fact may be an old shared copy whose `created`
753
+ // predates the incoming promotion event — using it as
754
+ // `createdAt` would make the new memory appear older than the
755
+ // existing one, preventing supersession from firing.
756
+ // PR #402 round-12 (Finding Uyui): the matching fact is an
757
+ // existing OLD memory — its persisted `frontmatter.created` is
758
+ // stale relative to the incoming promotion event. Pass
759
+ // `useCallerTimestamp: true` so the function uses
760
+ // `createdAt` as the ordering anchor instead of the old fact's
761
+ // timestamp, ensuring supersession fires correctly even when
762
+ // the matching fact predates conflicting candidates.
763
+ const hashDedupSupersession = await applyTemporalSupersession({
764
+ storage: sharedStorage,
765
+ newMemoryId: hashDedupMatchingFact.frontmatter.id,
766
+ entityRef: options.entityRef,
767
+ structuredAttributes: options.structuredAttributes,
768
+ createdAt: supersessionOrderingAt(options.validAt),
769
+ enabled: !(options.eventTimeSource === "extracted" && !options.validAt),
770
+ useCallerTimestamp: true,
771
+ });
772
+ // Catalog touch (issue #1499 — codex P2 NElSf): this dedup branch
773
+ // returns WITHOUT the post-write catalog touch (now at the storage chokepoint #1522),
774
+ // but `applyTemporalSupersession` mutated the shared namespace
775
+ // (it rewrote frontmatter to retire stale shared facts). When any
776
+ // ids were actually superseded, the shared namespace changed, so we
777
+ // must record the write — otherwise the shared record's
778
+ // `lastWriteAt` stays stale and `writtenSince` maintenance / QMD
779
+ // fanout skips the namespace after a supersession-only update.
780
+ // Best-effort and failure-tolerant (the storage chokepoint swallows
781
+ // errors); only touch when work happened to avoid spurious writes.
782
+ if (hashDedupSupersession.supersededIds.length > 0) {
783
+ }
784
+ // Active matching fact exists — normal short-circuit is safe.
785
+ return;
786
+ }
787
+ // No active same-entity shared fact found with this content hash.
788
+ // This can happen when the previously-written shared fact has since
789
+ // been superseded (e.g. Austin → NYC → Austin reversion): the hash
790
+ // index still records the hash but the fact is no longer active.
791
+ // Fall through to the write path below so a new active shared
792
+ // memory is created, then supersession fires post-write as usual.
793
+ log.debug(
794
+ `persistExtraction: hash-dedup found no active same-entity shared fact for ${options.sourceMemoryId}; falling through to write`,
795
+ );
796
+ } catch (hashDedupSupersessionErr) {
797
+ log.warn(
798
+ `persistExtraction: shared-namespace supersession on hash-dedup path failed open for ${options.sourceMemoryId}: ${hashDedupSupersessionErr}`,
799
+ );
800
+ // PR #402 round-7 (Fix #1 / cursor Medium PRRT_kwDORJXyws56U_ig):
801
+ // Only skip the write if we CONFIRMED a matching active shared fact
802
+ // before the error occurred (hashDedupLookupComplete is true AND
803
+ // hashDedupMatchingFact is set). If the error was thrown before
804
+ // matchingFact was resolved — e.g. readAllMemories() threw — we
805
+ // cannot assume a duplicate exists, and unconditionally returning
806
+ // would permanently lose the shared promotion. Fall through to the
807
+ // write path so the fact is not silently dropped.
808
+ if (hashDedupLookupComplete && hashDedupMatchingFact) {
809
+ // A matching active shared fact was confirmed — skip the write to
810
+ // avoid duplicating content that is already present. The existing
811
+ // fact remains active and the supersession failure is logged above.
812
+ return;
813
+ }
814
+ // Lookup did not complete or no candidate was found — we cannot
815
+ // confirm a duplicate. Fall through to the write + post-write
816
+ // supersession path so the shared promotion is not lost.
817
+ log.debug(
818
+ `persistExtraction: hash-dedup catch: lookup incomplete or no candidate found for ${options.sourceMemoryId}; falling through to write`,
819
+ );
820
+ }
821
+ } else {
822
+ // temporalSupersessionEnabled is off or no entity/attributes — keep
823
+ // the original short-circuit behaviour.
824
+ return;
825
+ }
826
+ }
827
+ const sharedPromotion = await sharedStorage.writeMemory(
828
+ options.category as any,
829
+ citedContent,
830
+ {
831
+ confidence: options.confidence,
832
+ tags: [...options.tags, "shared-promotion"],
833
+ entityRef: options.entityRef,
834
+ structuredAttributes: options.structuredAttributes,
835
+ source: `${options.source}-shared-promotion`,
836
+ importance: options.importance,
837
+ lineage: [options.sourceMemoryId],
838
+ sourceMemoryId: options.sourceMemoryId,
839
+ intentGoal: options.intentGoal,
840
+ intentActionType: options.intentActionType,
841
+ intentEntityTypes: options.intentEntityTypes,
842
+ memoryKind: options.memoryKind,
843
+ validAt: options.validAt,
844
+ // #1578 — forward bi-temporal bounds + ingestion provenance.
845
+ ...(options.invalidAt ? { invalidAt: options.invalidAt } : {}),
846
+ ...(options.observedAt ? { observedAt: options.observedAt } : {}),
847
+ ...(options.eventTimeSource ? { eventTimeSource: options.eventTimeSource } : {}),
848
+ contentHashSource: options.category === "fact" ? dedupContent : rawContent,
849
+ // Claim-level provenance spans (issue #1575 PR 2).
850
+ ...(options.sources && options.sources.length > 0 ? { sources: options.sources } : {}),
851
+ ...(options.provenance ? { provenance: options.provenance } : {}),
852
+ },
853
+ );
854
+ const promotedId = sharedPromotion.id;
855
+ // #1645: if the shared namespace's own tombstone blocked this promotion,
856
+ // leave the row pending_review but do NOT supersede active shared memories.
857
+ // PR #402 Finding 3 fix: run temporal supersession against the shared
858
+ // namespace after the promoted write lands so stale shared-namespace
859
+ // copies of the same entity attribute are retired. Without this,
860
+ // source-namespace supersession leaves the shared copy active and
861
+ // shared recall continues returning the stale state. Reuses the same
862
+ // applyTemporalSupersession helper — no logic duplication.
863
+ if (
864
+ !sharedPromotion.tombstoneBlocked &&
865
+ lifecycleCaps.temporalSupersession &&
866
+ options.entityRef &&
867
+ options.structuredAttributes &&
868
+ Object.keys(options.structuredAttributes).length > 0
869
+ ) {
870
+ try {
871
+ await applyTemporalSupersession({
872
+ storage: sharedStorage,
873
+ newMemoryId: promotedId,
874
+ entityRef: options.entityRef,
875
+ structuredAttributes: options.structuredAttributes,
876
+ createdAt: supersessionOrderingAt(options.validAt),
877
+ enabled: !(options.eventTimeSource === "extracted" && !options.validAt),
878
+ });
879
+ } catch (sharedSupersessionErr) {
880
+ log.warn(
881
+ `persistExtraction: shared-namespace temporal supersession failed open for promoted ${promotedId}: ${sharedSupersessionErr}`,
882
+ );
883
+ }
884
+ }
885
+ // Catalog touch (issue #1499, Issue B + ordering sweep): a shared-
886
+ // namespace promotion is the ONLY write the shared namespace receives on
887
+ // this path, so without this the shared record's lastWriteAt stays stale
888
+ // and `writtenSince` filters / maintenance fanout skip it. Record AFTER
889
+ // the promoted write and the shared temporal-supersession attempt so the
890
+ // catalog timestamp never precedes a later durable frontmatter mutation in
891
+ // the same promotion pass. The hot-path source-namespace touch uses a
892
+ // different storage dir, so this does not double-count the source.
893
+ // Best-effort and failure-tolerant — it must never crash the promotion.
894
+ // #1645 TV6: same guard as the profile-target promotion above.
895
+ if (!sharedPromotion.tombstoneBlocked) {
896
+ trackPersistedId(sharedStorage, promotedId, {
897
+ includeReturnedIds: false,
898
+ });
899
+ await this.deps.indexPersistedMemory(sharedStorage, promotedId);
900
+ trackBehaviorSignals(
901
+ sharedStorage,
902
+ buildBehaviorSignalsForMemory({
903
+ memoryId: promotedId,
904
+ category: options.category as any,
905
+ content: options.content,
906
+ namespace: this.deps.config.sharedNamespace,
907
+ confidence: options.confidence,
908
+ source: "extraction",
909
+ }),
910
+ );
911
+ }
912
+ } catch (err) {
913
+ log.warn(
914
+ `persistExtraction: shared promotion failed open for ${options.sourceMemoryId}: ${err}`,
915
+ );
916
+ }
917
+ };
918
+ // #1707 thread 1 — backfill temporal bounds onto promotion copies when the
919
+ // SOURCE-namespace dedup short-circuit fires. That branch patches the
920
+ // source copy then `continue`s before the promotion dedup paths run, so
921
+ // promoted shared/profile copies written before the source fact carried
922
+ // resolved bounds stay stale (cross-namespace recall surfaces an expired
923
+ // fact). This mirrors the promotion-target resolution used by the two
924
+ // promote closures above and calls the same fail-open helper against each
925
+ // target storage. Backfill-only: never writes a new promoted copy.
926
+ const backfillTemporalBoundsOnPromotionCopies = async (args: {
927
+ sourceStorage: StorageManager;
928
+ content: string;
929
+ category: string;
930
+ confidence: number;
931
+ entityRef?: string;
932
+ structuredAttributes?: Record<string, string>;
933
+ bounds: {
934
+ invalidAt?: string;
935
+ validFrom?: string;
936
+ observedAt?: string;
937
+ eventTimeSource?: "extracted" | "assumed";
938
+ };
939
+ }): Promise<void> => {
940
+ // Mirror the helper's I/O gate: only resolve promotion targets when
941
+ // there is an end bound OR an EXTRACTED start bound. An "assumed"
942
+ // validFrom is just the ingestion anchor (no recall effect), so
943
+ // skipping it avoids resolving target storages on every bi-temporal
944
+ // duplicate (review cursor PRRT_OvHk).
945
+ const hasExtractedStart =
946
+ args.bounds.validFrom !== undefined &&
947
+ args.bounds.eventTimeSource === "extracted";
948
+ if (!args.bounds.invalidAt && !hasExtractedStart) return;
949
+ // Build the same dedupContent the promotion functions hash on so the
950
+ // content-hash lookup in the helper matches what was stored.
951
+ const rawContent =
952
+ citationEnabled && hasCitationForTemplate(args.content, citationTemplate)
953
+ ? stripCitationForTemplate(args.content, citationTemplate)
954
+ : args.content;
955
+ const sanitizedBase = sanitizeMemoryContent(rawContent);
956
+ const dedupContent =
957
+ args.category === "fact" &&
958
+ args.structuredAttributes &&
959
+ Object.keys(args.structuredAttributes).length > 0
960
+ ? `${sanitizedBase.text}\n[Attributes: ${normalizeAttributePairs(args.structuredAttributes)}]`
961
+ : sanitizedBase.text;
962
+ // Profile targets. NOTE: we do NOT gate on profileAutoPromotionAllows —
963
+ // a promoted profile copy may exist from an EARLIER extraction with
964
+ // higher confidence or older auto-promote settings, so the current
965
+ // duplicate's confidence must not skip backfilling an already-existing
966
+ // promoted copy (review codex PRRT_Ov7LKF). The helper no-ops when no
967
+ // matching copy is found, so resolving all configured auto-promote
968
+ // targets is safe.
969
+ if (scopeProfileWritePlan) {
970
+ const autoTargets = new Set(
971
+ scopeProfileWritePlan.profile.autoPromote.targets,
972
+ );
973
+ const profileTargets = scopeProfileWritePlan.promotionTargets.filter(
974
+ (target) =>
975
+ target.target !== "serverShared" &&
976
+ autoTargets.has(target.target) &&
977
+ target.authorized &&
978
+ target.namespace,
979
+ );
980
+ for (const target of profileTargets) {
981
+ if (!target.namespace) continue;
982
+ try {
983
+ const targetStorage = await this.deps.getStorageRouter().storageFor(
984
+ target.namespace,
985
+ );
986
+ if (targetStorage.dir === args.sourceStorage.dir) continue;
987
+ await this.deps.backfillTemporalBoundsOnDedupHit(
988
+ targetStorage,
989
+ dedupContent,
990
+ args.bounds,
991
+ args.entityRef,
992
+ );
993
+ } catch (err) {
994
+ log.warn(
995
+ `bitemporal-backfill: profile-target backfill failed open for ${target.target}: ${err}`,
996
+ );
997
+ }
998
+ }
999
+ }
1000
+ // Shared target. A shared copy may exist from an earlier extraction
1001
+ // regardless of the current confidence, so do not gate on
1002
+ // shouldPromoteToShared (review codex PRRT_Ov7LKF) — BUT still respect
1003
+ // shared-write AUTHORIZATION: a scoped profile that does not authorize
1004
+ // serverShared writes must not have its shared namespace backfilled
1005
+ // (review codex PRRT_Ov7dHR). profileAllowsSharedWrites is true for the
1006
+ // legacy no-scope case and encodes the readable/writable/authorized
1007
+ // checks under a scope profile.
1008
+ if (profileAllowsSharedWrites) {
1009
+ try {
1010
+ const sharedStorage = await this.deps.getStorageRouter().storageFor(
1011
+ this.deps.config.sharedNamespace,
1012
+ );
1013
+ if (sharedStorage.dir !== args.sourceStorage.dir) {
1014
+ await this.deps.backfillTemporalBoundsOnDedupHit(
1015
+ sharedStorage,
1016
+ dedupContent,
1017
+ args.bounds,
1018
+ args.entityRef,
1019
+ );
1020
+ }
1021
+ } catch (err) {
1022
+ log.warn(
1023
+ `bitemporal-backfill: shared-target backfill failed open: ${err}`,
1024
+ );
1025
+ }
1026
+ }
1027
+ };
1028
+
1029
+ // Defensive: validate result and facts array
1030
+ if (!result || !Array.isArray(result.facts)) {
1031
+ log.warn(
1032
+ "persistExtraction: result or result.facts is invalid, skipping",
1033
+ { resultType: typeof result, factsType: typeof result?.facts },
1034
+ );
1035
+ return persistedIds;
1036
+ }
1037
+
1038
+ // Chunking config from plugin settings
1039
+ const chunkingConfig: ChunkingConfig = {
1040
+ targetTokens: this.deps.config.chunkingTargetTokens,
1041
+ minTokens: this.deps.config.chunkingMinTokens,
1042
+ overlapSentences: this.deps.config.chunkingOverlapSentences,
1043
+ };
1044
+
1045
+ const rawEntities = Array.isArray((result as any).entities)
1046
+ ? (result as any).entities
1047
+ : [];
1048
+ const rawQuestions = Array.isArray((result as any).questions)
1049
+ ? (result as any).questions
1050
+ : [];
1051
+ const rawProfileUpdates = Array.isArray((result as any).profileUpdates)
1052
+ ? (result as any).profileUpdates
1053
+ : [];
1054
+
1055
+ const facts = result.facts.slice(0, this.deps.config.extractionMaxFactsPerRun);
1056
+ const entities = rawEntities.slice(
1057
+ 0,
1058
+ this.deps.config.extractionMaxEntitiesPerRun,
1059
+ );
1060
+ const questions = rawQuestions.slice(
1061
+ 0,
1062
+ this.deps.config.extractionMaxQuestionsPerRun,
1063
+ );
1064
+ const profileUpdates = rawProfileUpdates.slice(
1065
+ 0,
1066
+ this.deps.config.extractionMaxProfileUpdatesPerRun,
1067
+ );
1068
+
1069
+ if (
1070
+ facts.length < result.facts.length ||
1071
+ entities.length < result.entities.length ||
1072
+ questions.length < result.questions.length ||
1073
+ profileUpdates.length < result.profileUpdates.length
1074
+ ) {
1075
+ log.warn(
1076
+ "persistExtraction: capped extraction payload to guardrails " +
1077
+ `(facts ${facts.length}/${result.facts.length}, entities ${entities.length}/${result.entities.length}, ` +
1078
+ `questions ${questions.length}/${result.questions.length}, profile ${profileUpdates.length}/${result.profileUpdates.length})`,
1079
+ );
1080
+ }
1081
+
1082
+ // v8.2: pre-load all memories once for entity-sibling graph edges (avoids per-fact disk scan)
1083
+ type GraphStorageContext = {
1084
+ allMemsForGraph: Awaited<
1085
+ ReturnType<typeof storage.readAllMemories>
1086
+ > | null;
1087
+ memoryPathById: Map<string, string>;
1088
+ previousPersistedRelPath?: string;
1089
+ };
1090
+ const graphContextByStorageDir = new Map<string, GraphStorageContext>();
1091
+ const ensureGraphContext = async (
1092
+ targetStorage: StorageManager,
1093
+ ): Promise<GraphStorageContext> => {
1094
+ const existing = graphContextByStorageDir.get(targetStorage.dir);
1095
+ if (existing) return existing;
1096
+ const created: GraphStorageContext = {
1097
+ allMemsForGraph: null,
1098
+ memoryPathById: new Map<string, string>(),
1099
+ };
1100
+ if (graphCaps.multiGraphMemory) {
1101
+ try {
1102
+ created.allMemsForGraph = await targetStorage.readAllMemories();
1103
+ for (const [id, relPath] of buildMemoryPathById(
1104
+ created.allMemsForGraph,
1105
+ targetStorage.dir,
1106
+ )) {
1107
+ created.memoryPathById.set(id, relPath);
1108
+ }
1109
+ } catch {
1110
+ /* fail-open */
1111
+ }
1112
+ }
1113
+ graphContextByStorageDir.set(targetStorage.dir, created);
1114
+ return created;
1115
+ };
1116
+ let threadEpisodeIdsForGraph: string[] | undefined;
1117
+ if (graphCaps.multiGraphMemory && threadIdForExtraction) {
1118
+ try {
1119
+ const thread = await this.deps.getThreading().loadThread(threadIdForExtraction);
1120
+ threadEpisodeIdsForGraph = thread?.episodeIds
1121
+ ? [...thread.episodeIds]
1122
+ : [];
1123
+ } catch {
1124
+ /* fail-open */
1125
+ }
1126
+ }
1127
+ const routeRules = await this.deps.loadRoutingRules();
1128
+ const routeOptions = this.deps.routeEngineOptions();
1129
+
1130
+ // Pre-routing pass: compute the routed category for every fact BEFORE
1131
+ // building judge candidates. Route rules may override f.category (e.g.
1132
+ // via taxonomy remapping), and the judge must evaluate against the
1133
+ // *final* category that will actually be persisted — not the raw
1134
+ // extraction-time category. The per-fact write loop below reuses
1135
+ // these pre-computed results so routing is evaluated exactly once per
1136
+ // fact (no duplicated logic).
1137
+ const preRoutedCategories: Array<string | undefined> = new Array(facts.length);
1138
+ // #1713 Item 1: collect routed target namespaces so the pre-judge redaction
1139
+ // filter can consult rules from cross-namespace targets, not just the source.
1140
+ // #1713 P2 (cursor): track per-fact routed namespace for per-fact pre-judge rules
1141
+ const preRoutedNamespaceByFact: Array<string | undefined> = new Array(facts.length);
1142
+ if (routeRules.length > 0) {
1143
+ for (let fi = 0; fi < facts.length; fi++) {
1144
+ const f = facts[fi];
1145
+ if (
1146
+ !f ||
1147
+ typeof f.content !== "string" ||
1148
+ !f.content.trim() ||
1149
+ typeof f.category !== "string" ||
1150
+ !f.category.trim()
1151
+ ) {
1152
+ continue;
1153
+ }
1154
+ try {
1155
+ const tags = Array.isArray(f.tags) ? f.tags : [];
1156
+ const routeText = `${f.category} ${tags.join(" ")} ${f.content}`;
1157
+ const selected = selectRouteRule(routeText, routeRules, routeOptions);
1158
+ if (selected?.target.category) {
1159
+ preRoutedCategories[fi] = selected.target.category;
1160
+ }
1161
+ if (selected?.target.namespace) {
1162
+ preRoutedNamespaceByFact[fi] = selected.target.namespace;
1163
+ }
1164
+ } catch {
1165
+ // Fail-open: routing errors fall through to the extracted category.
1166
+ }
1167
+ }
1168
+ }
1169
+
1170
+ // Extraction judge gate (issue #376). When enabled, batch-evaluate all
1171
+ // candidate facts for durability before the per-fact write loop.
1172
+ // The verdicts map is keyed by candidate index — we maintain a
1173
+ // candidateIndexToFactIndex mapping so the write loop can look up
1174
+ // verdicts by original fact index.
1175
+ //
1176
+ // Candidates are built using the *routed* category (preRoutedCategories)
1177
+ // so the judge evaluates durability against the same category that will
1178
+ // be persisted, not the raw extraction-time category.
1179
+ let judgeVerdictsByFactIndex: Map<number, import("../extraction-judge.js").JudgeVerdict> | null = null;
1180
+ let judgeGatedCount = 0;
1181
+ // Reset the side-channel defer count at the start of every
1182
+ // persistExtraction call so stale state from a prior call cannot leak
1183
+ // into the caller's buffer-retention decision.
1184
+ this.deps.setLastPersistExtractionDeferredCount(0);
1185
+ if (lifecycleCaps.extractionJudge) {
1186
+ // #1669 P1 + #1713 Item 1: pre-filter redacted facts from judge candidates
1187
+ // so never-store content is not persisted as judge training data.
1188
+ // Consult source + shared + routed target namespace rules so a
1189
+ // never-store pattern registered under a cross-namespace target is
1190
+ // caught at the batch pre-filter point, not just at the persist gate.
1191
+ let preJudgeRedactionRules: CompiledRedactionRule[] = [];
1192
+ try {
1193
+ // #1713: base rules are source-only. Per-fact routed target rules
1194
+ // (including shared when a fact is routed there) are checked in the
1195
+ // fact loop, matching the write-time gate's per-fact scoping exactly.
1196
+ preJudgeRedactionRules = await redactionRulesFor(storage.dir);
1197
+ } catch { /* fail open */ }
1198
+ try {
1199
+ const judgeCandidates: JudgeCandidate[] = [];
1200
+ const candidateToFactIndex: number[] = [];
1201
+ for (let fi = 0; fi < facts.length; fi++) {
1202
+ const f = facts[fi];
1203
+ if (
1204
+ !f ||
1205
+ typeof f.content !== "string" ||
1206
+ !f.content.trim() ||
1207
+ typeof f.category !== "string" ||
1208
+ !f.category.trim()
1209
+ ) {
1210
+ continue;
1211
+ }
1212
+ // Use the routed category when available so the judge sees the
1213
+ // final persisted category, not the raw extraction-time value.
1214
+ // Cast to MemoryCategory — routing targets are always valid
1215
+ // category slugs defined in the taxonomy; the fallback is the
1216
+ // original ExtractedFact.category which is already typed.
1217
+ const judgeCategory = (preRoutedCategories[fi] ?? f.category) as import("../types.js").MemoryCategory;
1218
+ if (judgeCategory === "procedure") {
1219
+ continue;
1220
+ }
1221
+ const tags = Array.isArray(f.tags) ? f.tags : [];
1222
+ const imp = scoreImportance(
1223
+ f.content,
1224
+ judgeCategory,
1225
+ tags,
1226
+ );
1227
+ // Pre-filter: skip facts below importance threshold to avoid
1228
+ // wasting LLM calls on facts that will be filtered anyway in
1229
+ // the per-fact write loop (issue #376 review finding).
1230
+ if (
1231
+ !isAboveImportanceThreshold(
1232
+ imp.level,
1233
+ this.deps.config.extractionMinImportanceLevel,
1234
+ )
1235
+ ) {
1236
+ continue;
1237
+ }
1238
+ // #1713 P2 (cursor): per-fact rules — load this fact's routed target
1239
+ // namespace rules FIRST, then check. This catches target-only rules
1240
+ // even when the base (source+shared) rules are empty (threads acc58c42
1241
+ // + PBJEe/PBKAj).
1242
+ let factRedactionRules = preJudgeRedactionRules;
1243
+ let factNs = preRoutedNamespaceByFact[fi];
1244
+ // #1713 (codex PRRT_kwDORJXyws6PBj5X): scope classification can route
1245
+ // a scope=global fact to the shared namespace independent of routing
1246
+ // rules. If so, the pre-judge filter must consult the shared
1247
+ // namespace's rules too, or a never-store pattern under shared is
1248
+ // missed at pre-judge and only caught at the write gate — letting the
1249
+ // content reach the extraction judge/training path. Mirror the write
1250
+ // loop's scope-routing conditions exactly.
1251
+ if (
1252
+ !factNs &&
1253
+ lifecycleCaps.extractionScopeClassification &&
1254
+ namespacesEnabled &&
1255
+ f.scope === "global" &&
1256
+ profileAllowsSharedWrites &&
1257
+ this.deps.storageDirNamespace(storage.dir) !== this.deps.config.sharedNamespace
1258
+ ) {
1259
+ factNs = this.deps.config.sharedNamespace;
1260
+ }
1261
+ if (factNs) {
1262
+ try {
1263
+ const factDir = (await this.deps.getStorageRouter().storageFor(factNs)).dir;
1264
+ if (factDir !== storage.dir) {
1265
+ const targetRules = await redactionRulesFor(factDir);
1266
+ if (targetRules.length) factRedactionRules = [...preJudgeRedactionRules, ...targetRules];
1267
+ }
1268
+ } catch { /* fail open */ }
1269
+ }
1270
+ if (factRedactionRules.length > 0) {
1271
+ const rc = f.content + (f.structuredAttributes ? " " + JSON.stringify(f.structuredAttributes) : "")
1272
+ + (f.procedureSteps ? " " + f.procedureSteps.map((s) => `${s.intent} ${s.expectedOutcome ?? ""} ${s.toolCall ? `${s.toolCall.kind} ${s.toolCall.signature}` : ""}`.trim()).join(" ") : "");
1273
+ if (contentMatchesRedactionRules(rc, factRedactionRules)) continue;
1274
+ }
1275
+ judgeCandidates.push({
1276
+ text: f.content,
1277
+ category: judgeCategory,
1278
+ confidence: typeof f.confidence === "number" ? f.confidence : 0.7,
1279
+ tags,
1280
+ importanceLevel: imp.level,
1281
+ });
1282
+ candidateToFactIndex.push(fi);
1283
+ }
1284
+ // Telemetry + training-pair emit (issue #562 PR 3 + PR 4). The
1285
+ // orchestrator wires two fire-and-forget writers behind a single
1286
+ // callback so `judgeFactDurability` does not need to know about
1287
+ // either ledger. Both handlers are skipped when their flags are
1288
+ // off; the combined callback itself is undefined when both are
1289
+ // disabled so there is zero overhead in the default configuration.
1290
+ const judgeTelemetryOpts = {
1291
+ enabled: lifecycleCaps.extractionJudgeTelemetry,
1292
+ memoryDir: this.deps.config.memoryDir,
1293
+ };
1294
+ const judgeTrainingOpts = {
1295
+ enabled: this.deps.config.collectJudgeTrainingPairs === true,
1296
+ ...(this.deps.config.judgeTrainingDir
1297
+ ? { directory: this.deps.config.judgeTrainingDir }
1298
+ : {}),
1299
+ };
1300
+ const judgeTelemetryHandler =
1301
+ judgeTelemetryOpts.enabled || judgeTrainingOpts.enabled
1302
+ ? (obs: import("../extraction-judge.js").JudgeVerdictObservation) => {
1303
+ const ts = new Date().toISOString();
1304
+ const verdictKind = getVerdictKind(obs.verdict);
1305
+ if (judgeTelemetryOpts.enabled) {
1306
+ const event: import("../extraction-judge-telemetry.js").JudgeVerdictEvent = {
1307
+ version: 1,
1308
+ category: EXTRACTION_JUDGE_VERDICT_CATEGORY,
1309
+ ts,
1310
+ verdictKind,
1311
+ reason: obs.verdict.reason,
1312
+ deferrals: obs.priorDeferrals,
1313
+ elapsedMs: obs.elapsedMs,
1314
+ candidateCategory: obs.candidate.category,
1315
+ confidence: obs.candidate.confidence,
1316
+ contentHash: obs.contentHash,
1317
+ fromCache: obs.source === "cache",
1318
+ ...(obs.source === "llm-cap-rejected"
1319
+ ? { deferCapTriggered: true }
1320
+ : {}),
1321
+ };
1322
+ void recordJudgeVerdict(event, judgeTelemetryOpts);
1323
+ }
1324
+ if (judgeTrainingOpts.enabled) {
1325
+ const pair: import("../extraction-judge-training.js").JudgeTrainingPair = {
1326
+ version: 1,
1327
+ ts,
1328
+ candidateText: obs.candidate.text,
1329
+ candidateCategory: obs.candidate.category,
1330
+ ...(typeof obs.candidate.confidence === "number"
1331
+ ? { candidateConfidence: obs.candidate.confidence }
1332
+ : {}),
1333
+ verdictKind,
1334
+ reason: obs.verdict.reason,
1335
+ priorDeferrals: obs.priorDeferrals,
1336
+ };
1337
+ void recordJudgeTrainingPair(pair, judgeTrainingOpts);
1338
+ }
1339
+ }
1340
+ : undefined;
1341
+ const judgeResult = await judgeFactDurability(
1342
+ judgeCandidates,
1343
+ this.deps.config,
1344
+ this.deps.getLocalLlm(),
1345
+ new FallbackLlmClient(
1346
+ this.deps.config.gatewayConfig,
1347
+ fallbackLlmRuntimeContextFromConfig(this.deps.config),
1348
+ ),
1349
+ this.deps.getJudgeVerdictCache(),
1350
+ this.deps.getJudgeDeferCounts(),
1351
+ judgeTelemetryHandler,
1352
+ );
1353
+ // Remap candidate-indexed verdicts to original fact indexes
1354
+ judgeVerdictsByFactIndex = new Map();
1355
+ for (const [candidateIdx, verdict] of judgeResult.verdicts) {
1356
+ const factIdx = candidateToFactIndex[candidateIdx];
1357
+ if (factIdx !== undefined) {
1358
+ judgeVerdictsByFactIndex.set(factIdx, verdict);
1359
+ }
1360
+ }
1361
+ log.info(
1362
+ `extraction-judge: ${judgeResult.verdicts.size}/${judgeCandidates.length} facts evaluated, ` +
1363
+ `${judgeResult.cached} cached, ${judgeResult.judged} judged, ` +
1364
+ `${judgeResult.deferred} deferred` +
1365
+ (judgeResult.deferredCappedToReject > 0
1366
+ ? ` (${judgeResult.deferredCappedToReject} cap-rejected)`
1367
+ : "") +
1368
+ `, ${judgeResult.elapsed}ms`,
1369
+ );
1370
+ // Expose defer count to the caller (issue #562 PR 2) so it can decide
1371
+ // whether to retain buffer turns for the next extraction pass.
1372
+ this.deps.setLastPersistExtractionDeferredCount(judgeResult.deferred);
1373
+ } catch (err) {
1374
+ // Fail-open: if the entire judge pipeline errors, proceed without filtering
1375
+ log.warn(
1376
+ `extraction-judge: pipeline error, proceeding without filtering (fail-open): ${err instanceof Error ? err.message : String(err)}`,
1377
+ );
1378
+ }
1379
+ }
1380
+
1381
+ // Faithfulness gate (issue #1576). Entailment-verification of extracted
1382
+ // facts against their verified source spans from #1575. Placement: after
1383
+ // parse + provenance validation, BEFORE persist/index (rule 44). The
1384
+ // substantive batch logic lives in the pure module; this is thin
1385
+ // delegation (ground rule 4). off → null map (byte-identical pre-feature
1386
+ // pipeline, rule 39); shadow → record only; enforce → pending_review.
1387
+ const faithfulnessMode = this.deps.config.extractionFaithfulnessGate;
1388
+ const faithfulnessResultsByFactIndex =
1389
+ faithfulnessMode === "shadow" || faithfulnessMode === "enforce"
1390
+ ? await runFaithfulnessGateBatch(
1391
+ facts,
1392
+ faithfulnessMode,
1393
+ this.deps.config,
1394
+ this.deps.getLocalLlm(),
1395
+ new FallbackLlmClient(
1396
+ this.deps.config.gatewayConfig,
1397
+ fallbackLlmRuntimeContextFromConfig(this.deps.config),
1398
+ ),
1399
+ this.deps.getFaithfulnessCounters(),
1400
+ sourceText,
1401
+ )
1402
+ : null;
1403
+
1404
+ let factLoopIndex = -1;
1405
+ for (const fact of facts) {
1406
+ factLoopIndex++;
1407
+ if (
1408
+ !fact ||
1409
+ typeof (fact as any).content !== "string" ||
1410
+ !(fact as any).content.trim()
1411
+ ) {
1412
+ continue;
1413
+ }
1414
+ if (
1415
+ typeof (fact as any).category !== "string" ||
1416
+ !(fact as any).category.trim()
1417
+ ) {
1418
+ continue;
1419
+ }
1420
+ (fact as any).tags = Array.isArray((fact as any).tags)
1421
+ ? (fact as any).tags.filter((t: any) => typeof t === "string")
1422
+ : [];
1423
+ (fact as any).confidence =
1424
+ typeof (fact as any).confidence === "number"
1425
+ ? (fact as any).confidence
1426
+ : 0.7;
1427
+ // #1670 — anchor each fact's event-time to its SOURCE TURN timestamp,
1428
+ // not the batch-wide latest. When a buffered conversation spans a date
1429
+ // boundary, a relative expression ("yesterday") on an early-turn fact
1430
+ // must resolve against that early turn's date. Prefer the fact's
1431
+ // explicit sourceTurnTimestamp, then the earliest provenance span's
1432
+ // observedAt, then fall back to the batch anchor (legacy extractors).
1433
+ const factAnchor = pickFactEventTimeAnchor(fact, sourceContext?.validAt);
1434
+ const biTemporal = this.deps.config.temporalBiTemporal && factAnchor ? resolveFactEventTime(fact.eventTime, factAnchor) : undefined;
1435
+ // Content-hash dedup check (v6.0)
1436
+ //
1437
+ // Canonicalize pre-tagged facts before hashing (Codex P2 — issue #369).
1438
+ // When a fact already carries an inline citation (e.g. relayed or
1439
+ // reprocessed), hashing `fact.content` as-is would produce a different
1440
+ // hash than the one stored from the original write (which used the raw,
1441
+ // un-cited body as contentHashSource). Strip any citation first so the
1442
+ // dedup key matches what the hash index recorded.
1443
+ //
1444
+ // stripCitationForTemplate handles both the default and custom template
1445
+ // formats. For all-placeholder templates it cannot detect citations and
1446
+ // returns the text unchanged — dedup may miss in that edge case, which
1447
+ // is acceptable (no false-positive suppression).
1448
+ //
1449
+ // Routing runs before content-hash dedup and scoring so category overrides
1450
+ // affect both the dedup fingerprint and importance (issue #519 procedure routing).
1451
+ let writeCategory = fact.category;
1452
+ let targetStorage = storage;
1453
+ const sourceStorageDir = storage.dir; // #1669 thread #2: pre-routing source ns for redaction gate
1454
+ // Track the KNOWN target namespace NAME alongside targetStorage (round 6,
1455
+ // codex P2 — NCQI0). Re-deriving it from `targetStorage.dir` mangles a raw
1456
+ // namespace literally named like a canonical token (e.g. `ns-616c706861`
1457
+ // served from its legacy raw dir decodes to `alpha`). We seed it from the
1458
+ // EXPLICIT base namespace the caller used to obtain `storage` (NHIdx, codex
1459
+ // P2) — `selfNamespace`/`writeNamespaceOverride` — so the catalog write touch
1460
+ // records the real namespace, not a guess decoded from the directory. We only
1461
+ // fall back to decoding the dir when no base namespace was passed (legacy
1462
+ // callers). The EXPLICIT routed name (below) still overrides this verbatim.
1463
+ let targetNamespaceName =
1464
+ baseNamespace && baseNamespace.length > 0
1465
+ ? baseNamespace
1466
+ : this.deps.storageDirNamespace(targetStorage.dir);
1467
+ let routedRuleId: string | undefined;
1468
+ let routedNamespaceExplicit = false;
1469
+ if (routeRules.length > 0) {
1470
+ try {
1471
+ const routeText = `${fact.category} ${fact.tags.join(" ")} ${fact.content}`;
1472
+ const selected = selectRouteRule(routeText, routeRules, routeOptions);
1473
+ if (selected) {
1474
+ routedRuleId = selected.rule.id;
1475
+ if (selected.target.category) {
1476
+ writeCategory = selected.target.category;
1477
+ }
1478
+ if (selected.target.namespace) {
1479
+ routedNamespaceExplicit = true;
1480
+ targetStorage = await this.deps.getStorageRouter().storageFor(
1481
+ selected.target.namespace,
1482
+ );
1483
+ targetNamespaceName = selected.target.namespace;
1484
+ }
1485
+ }
1486
+ } catch (err) {
1487
+ log.warn(
1488
+ `routing evaluation failed; fail-open to extracted category/namespace: ${err}`,
1489
+ );
1490
+ }
1491
+ }
1492
+
1493
+ // Scope-based namespace routing: when scope classification is enabled
1494
+ // and the LLM tagged this fact as "global", route it to the shared
1495
+ // namespace so cross-project knowledge is visible everywhere. Only
1496
+ // applies when namespaces are enabled and the fact was not already
1497
+ // routed to a specific namespace by a routing rule (routing rules
1498
+ // that set an explicit namespace take precedence; category-only rules
1499
+ // do not block scope routing). Rule 30: gated by
1500
+ // extractionScopeClassificationEnabled.
1501
+ if (
1502
+ lifecycleCaps.extractionScopeClassification &&
1503
+ namespacesEnabled &&
1504
+ fact.scope === "global" &&
1505
+ !routedNamespaceExplicit
1506
+ ) {
1507
+ const currentNs = this.deps.storageDirNamespace(targetStorage.dir);
1508
+ if (currentNs !== this.deps.config.sharedNamespace && profileAllowsSharedWrites) {
1509
+ try {
1510
+ targetStorage = await this.deps.getStorageRouter().storageFor(
1511
+ this.deps.config.sharedNamespace,
1512
+ );
1513
+ targetNamespaceName = this.deps.config.sharedNamespace;
1514
+ log.debug(
1515
+ `scope-routing: fact "${fact.content.slice(0, 60)}…" routed to shared namespace (scope=global)`,
1516
+ );
1517
+ } catch (scopeRouteErr) {
1518
+ log.warn(
1519
+ `scope-routing: failed to resolve shared namespace storage; writing to session namespace (fail-open): ${scopeRouteErr}`,
1520
+ );
1521
+ }
1522
+ } else if (currentNs !== this.deps.config.sharedNamespace) {
1523
+ log.debug(
1524
+ `scope-routing: skipped shared namespace for global fact because active scope profile ${scopeProfileWritePlan?.profileId ?? "none"} does not authorize serverShared writes`,
1525
+ );
1526
+ }
1527
+ }
1528
+ // #1669 redaction-rule gate: consult BOTH source and target namespace
1529
+ // rules before any write. A never-store pattern registered under the
1530
+ // source namespace must survive scope-routing to a different target
1531
+ // (review thread #2). Fails open on read error.
1532
+ try {
1533
+ const redactionRules = await redactionRulesFor(sourceStorageDir, targetStorage.dir);
1534
+ const redactionCandidate = fact.content
1535
+ + (fact.structuredAttributes ? " " + JSON.stringify(fact.structuredAttributes) : "")
1536
+ + (fact.procedureSteps ? " " + fact.procedureSteps.map((s) => `${s.intent} ${s.expectedOutcome ?? ""} ${s.toolCall ? `${s.toolCall.kind} ${s.toolCall.signature}` : ""}`.trim()).join(" ") : "");
1537
+ if (redactionRules.length > 0 && contentMatchesRedactionRules(redactionCandidate, redactionRules)) {
1538
+ redactionGatedCount++;
1539
+ log.debug(`extraction: redaction-rule withheld fact #${redactionGatedCount} in ${targetStorage.dir}`);
1540
+ continue;
1541
+ }
1542
+ } catch (redactionErr) {
1543
+ log.warn(`extraction: redaction-rule gate failed open: ${redactionErr}`);
1544
+ }
1545
+
1546
+ // Procedures: fingerprint the full serialized body (title + steps), not
1547
+ // the title alone, so distinct step lists are not collapsed (issue #519).
1548
+ const canonicalContentForHash =
1549
+ citationEnabled &&
1550
+ hasCitationForTemplate(fact.content, citationTemplate)
1551
+ ? stripCitationForTemplate(fact.content, citationTemplate)
1552
+ : fact.content;
1553
+ const contentHashDedupKey =
1554
+ writeCategory === "procedure"
1555
+ ? buildProcedurePersistBody(fact.content, fact.procedureSteps)
1556
+ : canonicalContentForHash;
1557
+ // Importance is scored before the dedup short-circuit so #1671 backfill
1558
+ // can gate on it (cursor PRRT_OvKnS): a low-value duplicate that the
1559
+ // importance write-gate (#372) or the judge pre-filter would drop must
1560
+ // not expire an active fact.
1561
+ const importance = scoreImportance(
1562
+ fact.content,
1563
+ writeCategory,
1564
+ fact.tags,
1565
+ );
1566
+ let exactDuplicate = false;
1567
+ try {
1568
+ exactDuplicate = await this.deps.hasContentHashDedup(
1569
+ targetStorage,
1570
+ contentHashDedupKey,
1571
+ );
1572
+ } catch (err) {
1573
+ log.warn(
1574
+ `content-hash dedup lookup failed for storage ${targetStorage.dir}; writing fact fail-open: ${err}`,
1575
+ );
1576
+ }
1577
+ if (exactDuplicate) {
1578
+ // #1671 — before short-circuiting, backfill bi-temporal bounds
1579
+ // onto the existing source-namespace copy if it lacks bounds the
1580
+ // incoming fact now carries (re-extraction with a resolved invalidAt).
1581
+ // Skip when the fact would be rejected/deferred/pending by downstream
1582
+ // gates — a non-durable candidate must not expire an active fact
1583
+ // (chatgpt-codex P1: faithfulness, requireSpans, extraction judge).
1584
+ // #1671 + #1707: backfill bi-temporal bounds onto the existing
1585
+ // source-namespace copy if it lacks bounds the incoming fact now
1586
+ // carries (re-extraction with a resolved bound). #1707 thread 3:
1587
+ // gate on writeCategory === "fact" — the helper only matches facts,
1588
+ // so a non-fact duplicate must not reach the fact-only scan.
1589
+ // #1707 thread 2: also fire when only a corrected start bound
1590
+ // (validFrom) is present, not just an end bound (validUntil). The
1591
+ // downstream-gate skip still applies — a non-durable candidate must
1592
+ // not expire an active fact (chatgpt-codex P1).
1593
+ if (
1594
+ biTemporal &&
1595
+ writeCategory === "fact" &&
1596
+ (biTemporal.validUntil || biTemporal.validFrom)
1597
+ ) {
1598
+ const fr = faithfulnessResultsByFactIndex?.get(factLoopIndex);
1599
+ const faithfulnessWouldPending =
1600
+ faithfulnessMode === "enforce" &&
1601
+ fr?.ok === true &&
1602
+ (fr.verdict === "unsupported" || fr.verdict === "contradicted");
1603
+ const requireSpansWouldPending =
1604
+ this.deps.config.provenance?.requireSpans === true &&
1605
+ fact.requireSpansPending === true;
1606
+ const judgeVerdict = judgeVerdictsByFactIndex?.get(factLoopIndex);
1607
+ const judgeWouldGate =
1608
+ !this.deps.config.extractionJudgeShadow &&
1609
+ judgeVerdict !== undefined &&
1610
+ !judgeVerdict.durable;
1611
+ // Importance gate (cursor PRRT_OvKnS): below-threshold duplicates
1612
+ // would never persist (#372) and carry no judge verdict (the
1613
+ // pre-filter skips them), so they must not expire an active fact.
1614
+ if (
1615
+ isAboveImportanceThreshold(importance.level, this.deps.config.extractionMinImportanceLevel) &&
1616
+ !faithfulnessWouldPending &&
1617
+ !requireSpansWouldPending &&
1618
+ !judgeWouldGate
1619
+ ) {
1620
+ await this.deps.backfillTemporalBoundsOnDedupHit(
1621
+ targetStorage,
1622
+ contentHashDedupKey,
1623
+ {
1624
+ invalidAt: biTemporal.validUntil,
1625
+ // #1707 thread 2 — carry the corrected start bound too.
1626
+ validFrom: biTemporal.validFrom,
1627
+ observedAt: biTemporal.observedAt,
1628
+ eventTimeSource: biTemporal.eventTimeSource,
1629
+ },
1630
+ fact.entityRef,
1631
+ );
1632
+ // #1707 thread 1 — the source branch short-circuits (`continue`
1633
+ // below) before the promotion dedup paths run, so promoted
1634
+ // shared/profile copies written before the source fact carried
1635
+ // resolved bounds stay stale and cross-namespace recall surfaces
1636
+ // an expired fact. Backfill the promotion targets too (fail-open,
1637
+ // backfill-only — never writes a new promoted copy).
1638
+ await backfillTemporalBoundsOnPromotionCopies({
1639
+ sourceStorage: targetStorage,
1640
+ content: fact.content,
1641
+ category: writeCategory,
1642
+ confidence: fact.confidence,
1643
+ entityRef: fact.entityRef,
1644
+ structuredAttributes: fact.structuredAttributes,
1645
+ bounds: {
1646
+ invalidAt: biTemporal.validUntil,
1647
+ validFrom: biTemporal.validFrom,
1648
+ observedAt: biTemporal.observedAt,
1649
+ eventTimeSource: biTemporal.eventTimeSource,
1650
+ },
1651
+ });
1652
+ }
1653
+ }
1654
+ log.debug(
1655
+ `dedup: skipping duplicate fact "${fact.content.slice(0, 60)}…" in storage ${targetStorage.dir}`,
1656
+ );
1657
+ dedupedCount++;
1658
+ continue;
1659
+ }
1660
+
1661
+ if (writeCategory === "procedure" && this.deps.config.procedural?.enabled !== true) {
1662
+ log.debug("persistExtraction: skip procedure memory (procedural.enabled is false)");
1663
+ continue;
1664
+ }
1665
+
1666
+ // Importance write-gate (issue #372). Drop facts whose locally-scored
1667
+ // level falls below the configured minimum BEFORE the semantic dedup
1668
+ // lookup so that low-importance facts never incur an embedding search.
1669
+ // scoreImportance() already applies category boosts (e.g. corrections
1670
+ // +0.15) before deriving the level, so a correction at raw ~0.35
1671
+ // still lands at "normal" and passes the default gate. Without this
1672
+ // gate, trivial turn-level chatter ("hi", "k", heartbeat pings) gets
1673
+ // persisted as a fact memory and dilutes the store.
1674
+ if (
1675
+ !isAboveImportanceThreshold(
1676
+ importance.level,
1677
+ this.deps.config.extractionMinImportanceLevel,
1678
+ )
1679
+ ) {
1680
+ importanceGatedCount++;
1681
+ const snippet = fact.content.slice(0, 60).replace(/\s+/g, " ").trim();
1682
+ log.debug(`extraction: skip trivial "${snippet}"`);
1683
+ // Log-based counter (no dedicated metric bus in remnic-core yet).
1684
+ // Operators can grep for `metric:importance_gated` in gateway.log
1685
+ // to tune extractionMinImportanceLevel.
1686
+ log.debug(
1687
+ `metric:importance_gated level=${importance.level} threshold=${this.deps.config.extractionMinImportanceLevel} category=${writeCategory} count=${importanceGatedCount}`,
1688
+ );
1689
+ continue;
1690
+ }
1691
+
1692
+ // Extraction judge gate (issue #376 + #562 PR 2). After the local
1693
+ // importance gate passes, consult the judge verdict (computed before
1694
+ // the loop). In active mode, non-durable facts are dropped. In shadow
1695
+ // mode, verdicts are logged but all facts proceed to write.
1696
+ //
1697
+ // Defer verdicts (issue #562): do not persist now, but also do not
1698
+ // cache the outcome so the candidate is re-evaluated on a later
1699
+ // extraction pass. The judge module tracks how many times the same
1700
+ // content has been deferred and converts to reject at the configured
1701
+ // cap, so the orchestrator only needs to skip the write here.
1702
+ if (judgeVerdictsByFactIndex) {
1703
+ const verdict = judgeVerdictsByFactIndex.get(factLoopIndex);
1704
+ if (verdict && !verdict.durable) {
1705
+ const verdictKind = getVerdictKind(verdict);
1706
+ if (this.deps.config.extractionJudgeShadow) {
1707
+ log.info(
1708
+ `extraction-judge[shadow]: would ${verdictKind} "${fact.content.slice(0, 60)}…" reason="${verdict.reason}"`,
1709
+ );
1710
+ } else if (verdictKind === "defer") {
1711
+ judgeGatedCount++;
1712
+ log.debug(
1713
+ `extraction-judge: deferred "${fact.content.slice(0, 60)}…" reason="${verdict.reason}"`,
1714
+ );
1715
+ continue;
1716
+ } else {
1717
+ judgeGatedCount++;
1718
+ log.debug(
1719
+ `extraction-judge: rejected "${fact.content.slice(0, 60)}…" reason="${verdict.reason}"`,
1720
+ );
1721
+ continue;
1722
+ }
1723
+ }
1724
+ }
1725
+
1726
+ // Procedure extraction gate (issue #519): ≥2 steps + trigger phrasing.
1727
+ // Runs even when extractionJudgeEnabled is false (durability judge is unrelated).
1728
+ // Never tied to extractionJudgeShadow — that flag is only for the LLM durability judge.
1729
+ if (writeCategory === "procedure") {
1730
+ const procGate = validateProcedureExtraction({
1731
+ content: fact.content,
1732
+ procedureSteps: fact.procedureSteps,
1733
+ });
1734
+ if (!procGate.durable) {
1735
+ log.debug(
1736
+ `extraction-procedure-gate: rejected "${fact.content.slice(0, 60)}…" reason="${procGate.reason}"`,
1737
+ );
1738
+ continue;
1739
+ }
1740
+ }
1741
+
1742
+ // Faithfulness gate verdict application (issue #1576). Look up the
1743
+ // pre-computed verdict for this fact and translate it to frontmatter +
1744
+ // an optional enforce-mode pending_review status. Logic lives in the
1745
+ // pure module; this is thin read-through (ground rule 4).
1746
+ const { faithfulness: faithfulnessFm, enforceStatus: faithfulnessGateStatus } =
1747
+ applyFaithfulnessVerdict(
1748
+ faithfulnessResultsByFactIndex,
1749
+ factLoopIndex,
1750
+ faithfulnessMode,
1751
+ fact.content,
1752
+ this.deps.getFaithfulnessCounters(),
1753
+ );
1754
+
1755
+ // requireSpans enforcement (issue #1575 PR 2): when an operator opts
1756
+ // into provenance.requireSpans, a fact whose quote could not be located
1757
+ // in any source turn (carried as the transient requireSpansPending
1758
+ // signal from the extraction validator) routes to pending_review — the
1759
+ // same review queue an unsupported faithfulness verdict uses. This is
1760
+ // the persist-path wiring ProvenanceConfig.requireSpans documents.
1761
+ // Faithfulness takes precedence when it already routed the fact; both
1762
+ // gates agree on pending_review so the merge is a simple coalesce
1763
+ // (chatgpt-codex-connector thread 4xB).
1764
+ const requireSpansPendingStatus =
1765
+ this.deps.config.provenance?.requireSpans === true &&
1766
+ fact.requireSpansPending === true
1767
+ ? ("pending_review" as const)
1768
+ : undefined;
1769
+ const faithfulnessEnforceStatus = faithfulnessGateStatus ?? requireSpansPendingStatus;
1770
+
1771
+ // Issue #373 — write-time semantic similarity guard. Hook runs after
1772
+ // the exact content-hash miss and the importance gate so that:
1773
+ // (a) paraphrased near-duplicates never reach writeMemory(), and
1774
+ // (b) low-importance facts that will be dropped never trigger an
1775
+ // embedding lookup (avoids unnecessary API latency/cost).
1776
+ // Fails open when the embedding backend is unavailable.
1777
+ //
1778
+ // Defense in depth (PR #399 review): decideSemanticDedup already
1779
+ // catches lookup errors internally, and the embedding fetch is
1780
+ // bounded by a timeout in embedding-fallback.ts. We still wrap the
1781
+ // whole call in its own try/catch here so that any unexpected
1782
+ // rejection (future refactors, misbehaving custom backends, etc.)
1783
+ // can never block the persist loop — a failure in the dedup path
1784
+ // must always default to "not a duplicate".
1785
+ // Track a pending semantic-skip decision (populated inside the block
1786
+ // below). The actual drop happens AFTER contradiction detection so that
1787
+ // a high-similarity update/correction is linked as a superseding
1788
+ // contradiction rather than silently dropped.
1789
+ let pendingSemanticSkip: (SemanticDedupDecision & { action: "skip" }) | null = null;
1790
+ if (resolvePipelineProcessingCapabilities(this.deps.config).semanticDedup) {
1791
+ let semanticDecision: SemanticDedupDecision;
1792
+ // UUI2: skip embedding lookup for the rest of this batch once we know
1793
+ // the backend is unavailable. The flag is reset per-batch (set to false
1794
+ // at the top of persistExtraction), so a transient hiccup in one call
1795
+ // does not permanently disable dedup in subsequent calls.
1796
+ if (batchBackendUnavailable) {
1797
+ semanticDecision = { action: "keep", reason: "backend_unavailable" };
1798
+ } else {
1799
+ try {
1800
+ // Pass the resolved target storage so the lookup scopes the
1801
+ // embedding index to the target namespace (PR #399 P1 fix).
1802
+ // Without this, a high-similarity hit in a different namespace
1803
+ // would cause the fact to be dropped here — cross-namespace
1804
+ // write suppression / data loss.
1805
+ const lookupStorage = targetStorage;
1806
+ semanticDecision = await decideSemanticDedup(
1807
+ fact.content,
1808
+ (content, limit) =>
1809
+ this.deps.semanticDedupLookup(content, limit, lookupStorage),
1810
+ {
1811
+ enabled: true,
1812
+ threshold: this.deps.config.semanticDedupThreshold,
1813
+ candidates: this.deps.config.semanticDedupCandidates,
1814
+ },
1815
+ );
1816
+ } catch (err) {
1817
+ log.warn(
1818
+ `semantic dedup decision failed; failing open and writing fact: ${err}`,
1819
+ );
1820
+ semanticDecision = {
1821
+ action: "keep",
1822
+ reason: "backend_unavailable",
1823
+ };
1824
+ }
1825
+ // UUI2: cache the backend-unavailable signal for the rest of this batch.
1826
+ if (semanticDecision.reason === "backend_unavailable") {
1827
+ batchBackendUnavailable = true;
1828
+ }
1829
+ }
1830
+ if (semanticDecision.action === "skip") {
1831
+ pendingSemanticSkip = semanticDecision;
1832
+ }
1833
+ }
1834
+
1835
+ const inferredIntent = resolveConversationContextCapabilities(this.deps.config).intentRouting
1836
+ ? inferIntentFromText(
1837
+ `${writeCategory} ${fact.tags.join(" ")} ${fact.content}`,
1838
+ )
1839
+ : null;
1840
+ const extractionWriteSource =
1841
+ (fact as any).source === "proactive"
1842
+ ? "extraction-proactive"
1843
+ : "extraction";
1844
+
1845
+ // Check for contradictions before writing (Phase 2B).
1846
+ // NOTE: This block was moved above the chunking branch so that the
1847
+ // pendingSemanticSkip guard (below) can also protect the chunking path.
1848
+ // Previously, contradiction detection only ran on the non-chunked path,
1849
+ // meaning chunked facts could be persisted even when semanticDecision was
1850
+ // "skip" (the deferred guard was bypassed by the chunking `continue`).
1851
+ let supersedes: string | undefined;
1852
+ let links: MemoryLink[] = [];
1853
+ // True when contradiction detection ran and confirmed a contradiction,
1854
+ // regardless of whether auto-resolve is enabled. Used by the
1855
+ // semantic-skip guard so that contradictory updates are never silently
1856
+ // dropped — even when `contradictionAutoResolve=false` (in which case
1857
+ // `supersedes` is intentionally left unset to avoid retiring the old
1858
+ // memory without user confirmation).
1859
+ let contradictionDetected = false;
1860
+ // #1645: hoist the contradiction result so the deferred auto-resolve
1861
+ // (post-write, gated on tombstone status) can read its fields.
1862
+ let contradiction: {
1863
+ supersededId: string;
1864
+ confidence: number;
1865
+ reason: string;
1866
+ supersededPath: string;
1867
+ supersededCreated: string;
1868
+ supersededTags: string[];
1869
+ } | null | undefined;
1870
+
1871
+ // Faithfulness gate (#1576, chatgpt P2): skip contradiction detection
1872
+ // for a pending_review fact — an unfaithful extraction in the review queue
1873
+ // must not trigger auto-resolve and retire an existing active memory.
1874
+ if (
1875
+ resolveRecallEnhancementCapabilities(this.deps.config).contradictionDetection &&
1876
+ this.deps.getQmd().isAvailable() &&
1877
+ faithfulnessEnforceStatus !== "pending_review"
1878
+ ) {
1879
+ const targetNamespace = this.deps.storageDirNamespace(targetStorage.dir);
1880
+ contradiction = await this.deps.checkForContradiction(
1881
+ fact.content,
1882
+ writeCategory,
1883
+ targetNamespace,
1884
+ );
1885
+ if (contradiction) {
1886
+ contradictionDetected = true;
1887
+ // When auto-resolve is enabled the existing memory has already been
1888
+ // marked superseded; set `supersedes` so the new write carries the
1889
+ // relationship. When auto-resolve is disabled we still record the
1890
+ // contradiction link (so the memory is annotated for manual review)
1891
+ // but do NOT set `supersedes` on the new write — the old memory
1892
+ // remains active until a human resolves it.
1893
+ if (this.deps.config.contradictionAutoResolve) {
1894
+ supersedes = contradiction.supersededId;
1895
+ }
1896
+ links.push({
1897
+ targetId: contradiction.supersededId,
1898
+ linkType: "contradicts",
1899
+ strength: contradiction.confidence,
1900
+ reason: contradiction.reason,
1901
+ });
1902
+ // #1645: deindex + supersede are deferred to after writeMemory so the
1903
+ // caller can gate them on the new write's tombstone status. A
1904
+ // tombstone-blocked write (pending_review) must not deindex or retire
1905
+ // the existing active memory — see the post-write guard below.
1906
+ }
1907
+ }
1908
+
1909
+ // Apply the deferred semantic-skip now that contradiction detection has
1910
+ // run. If a contradiction was found (contradictionDetected is true), the
1911
+ // candidate is a contradictory update and must be written — do not skip
1912
+ // it. Only drop it when there is no detected contradiction (true
1913
+ // near-duplicate). This check intentionally runs BEFORE the chunking
1914
+ // branch so that a fact flagged as a semantic near-duplicate cannot be
1915
+ // persisted (with its hash registered) simply because it was long enough
1916
+ // to trigger chunking.
1917
+ //
1918
+ // NOTE: We use `contradictionDetected` rather than `!!supersedes` here
1919
+ // so that facts are preserved even when `contradictionAutoResolve=false`.
1920
+ // When auto-resolve is disabled `supersedes` is intentionally unset, but
1921
+ // the write must still proceed so the user can manually reconcile the
1922
+ // two memories later.
1923
+ //
1924
+ // UUI1: correction category writes are NEVER suppressed by the semantic
1925
+ // skip fallback, regardless of whether supersedes is set. When contradiction
1926
+ // detection is disabled or QMD is unavailable, supersedes is never set —
1927
+ // without this exemption a high-similarity correction would be silently
1928
+ // dropped, leaving a stale fact active. writeCategory (not fact.category)
1929
+ // is used because routing rules may have overridden the raw category.
1930
+ const isCorrection = writeCategory === "correction";
1931
+ // Faithfulness gate (#1576, cursor High): a pending_review fact must
1932
+ // bypass the semantic-dedup skip so it reaches the review queue — the
1933
+ // gate's contract is "persists with status: pending_review, never
1934
+ // silently dropped" (issue #1576).
1935
+ if (
1936
+ pendingSemanticSkip &&
1937
+ !contradictionDetected &&
1938
+ !isCorrection &&
1939
+ faithfulnessEnforceStatus !== "pending_review"
1940
+ ) {
1941
+ log.debug(
1942
+ `dedup: skipping semantic near-duplicate fact "${fact.content
1943
+ .slice(0, 60)
1944
+ .replace(/\s+/g, " ")}…" score=${pendingSemanticSkip.topScore.toFixed(
1945
+ 3,
1946
+ )} neighbor=${pendingSemanticSkip.topId}`,
1947
+ );
1948
+ dedupedCount++;
1949
+ // Do NOT add fact.content to contentHashIndex here. No memory was
1950
+ // persisted for this fact, so registering a synthetic hash would
1951
+ // permanently suppress exact-copy writes once the neighbor memory is
1952
+ // archived or deleted (the hash would linger with no backing record).
1953
+ continue;
1954
+ }
1955
+
1956
+ // Check if chunking is enabled and content should be chunked.
1957
+ // When semanticChunkingEnabled is true, prefer the embedding-based
1958
+ // semantic chunker which produces more coherent topic-aligned segments.
1959
+ // Falls back to the recursive sentence-boundary chunker on failure.
1960
+ if (resolvePipelineProcessingCapabilities(this.deps.config).chunking && writeCategory !== "procedure") {
1961
+ let chunkResult: { chunked: boolean; chunks: { content: string; index: number; tokenCount: number }[] };
1962
+
1963
+ if (resolvePipelineProcessingCapabilities(this.deps.config).semanticChunking) {
1964
+ try {
1965
+ const embedFn = this.deps.getEmbeddingFallback().embedTexts.bind(this.deps.getEmbeddingFallback());
1966
+ const semanticResult: SemanticChunkResult = await semanticChunkContent(
1967
+ fact.content,
1968
+ embedFn,
1969
+ this.deps.config.semanticChunkingConfig,
1970
+ );
1971
+ chunkResult = semanticResult;
1972
+ } catch (err) {
1973
+ // Honor the fallbackToRecursive contract: when the user explicitly
1974
+ // disables fallback, re-throw so extraction fails fast instead of
1975
+ // silently using the recursive chunker. semanticChunkContent already
1976
+ // throws when fallback is disabled, but this outer catch swallowed
1977
+ // that signal. (PR #439 post-merge Finding 1.)
1978
+ if (this.deps.config.semanticChunkingConfig?.fallbackToRecursive === false) {
1979
+ throw err;
1980
+ }
1981
+ log.debug(
1982
+ `semantic chunking failed, falling back to recursive chunker: ${err}`,
1983
+ );
1984
+ chunkResult = chunkContent(fact.content, chunkingConfig);
1985
+ }
1986
+ } else {
1987
+ chunkResult = chunkContent(fact.content, chunkingConfig);
1988
+ }
1989
+
1990
+ if (chunkResult.chunked && chunkResult.chunks.length > 1) {
1991
+ // Classify memory kind (v8.0 Phase 2B: HiMem episode/note dual store)
1992
+ const memoryKind = resolvePresentationCapabilities(this.deps.config).episodeNoteMode
1993
+ ? classifyMemoryKind(fact.content, fact.tags ?? [], writeCategory)
1994
+ : undefined;
1995
+
1996
+ // Write the parent memory first (with full content for reference).
1997
+ //
1998
+ // Compute the cited content once so that writeMemory and writeArtifact
1999
+ // (when verbatim artifacts are enabled) share the same citation timestamp.
2000
+ // See the normal write path comment for the full dedup rationale.
2001
+ //
2002
+ // Propagate supersedes/links from contradiction detection (round 6
2003
+ // fix): contradiction detection now runs BEFORE this branch so the
2004
+ // parent must carry the supersession relationship — without it the
2005
+ // old memory is deindexed but the new chunked parent has no link
2006
+ // back, leaving a dangling deindex with no replacement reference.
2007
+ // Child chunks intentionally do NOT carry supersedes; only the
2008
+ // parent represents the logical memory unit.
2009
+ // Canonicalize contentHashSource before writing (Thread 3 — Codex P2,
2010
+ // issue #369). If fact.content already carries an inline citation
2011
+ // (e.g. re-processed or relayed fact), strip it so contentHashSource
2012
+ // records the raw un-cited body — matching what the dedup check hashes
2013
+ // via stripCitationForTemplate before calling hasFactContentHash.
2014
+ const rawChunkedContent =
2015
+ citationEnabled &&
2016
+ hasCitationForTemplate(fact.content, citationTemplate)
2017
+ ? stripCitationForTemplate(fact.content, citationTemplate)
2018
+ : fact.content;
2019
+ const citedChunkedContent = applyInlineCitation(rawChunkedContent);
2020
+ const parentWrite = await targetStorage.writeMemory(
2021
+ writeCategory,
2022
+ citedChunkedContent,
2023
+ {
2024
+ confidence: fact.confidence,
2025
+ tags: [...fact.tags, "chunked"],
2026
+ entityRef: fact.entityRef,
2027
+ source: extractionWriteSource,
2028
+ importance,
2029
+ supersedes,
2030
+ links: links.length > 0 ? links : undefined,
2031
+ intentGoal: inferredIntent?.goal,
2032
+ intentActionType: inferredIntent?.actionType,
2033
+ intentEntityTypes: inferredIntent?.entityTypes,
2034
+ memoryKind,
2035
+ structuredAttributes: fact.structuredAttributes,
2036
+ validAt: biTemporal ? biTemporal.validFrom : sourceContext?.validAt,
2037
+ ...(biTemporal ? { observedAt: biTemporal.observedAt, eventTimeSource: biTemporal.eventTimeSource, ...(biTemporal.validUntil ? { invalidAt: biTemporal.validUntil } : {}) } : {}),
2038
+ contentHashSource: rawChunkedContent,
2039
+ // Faithfulness gate (issue #1576).
2040
+ ...(faithfulnessFm ? { faithfulness: faithfulnessFm } : {}),
2041
+ ...(faithfulnessEnforceStatus ? { status: faithfulnessEnforceStatus } : {}),
2042
+ // Claim-level provenance spans (issue #1575 PR 2).
2043
+ ...(fact.sources && fact.sources.length > 0 ? { sources: fact.sources } : {}),
2044
+ ...(fact.provenance ? { provenance: fact.provenance } : {}),
2045
+ },
2046
+ );
2047
+ const parentId = parentWrite.id;
2048
+ // #1645: surface the tombstone block and gate active post-write paths
2049
+ // (chunks, supersession, shared promotion, graph/artifact) like #1576.
2050
+ const tombstoneBlocked = parentWrite.tombstoneBlocked;
2051
+ const postWriteGuard =
2052
+ faithfulnessEnforceStatus === "pending_review" || tombstoneBlocked;
2053
+ // #1645: defer contradiction auto-resolve until tombstone status is
2054
+ // known (see applyDeferredContradictionResolve).
2055
+ await this.deps.applyDeferredContradictionResolve(
2056
+ contradiction,
2057
+ targetStorage,
2058
+ parentId,
2059
+ postWriteGuard,
2060
+ );
2061
+ try {
2062
+ // Write individual chunks with parent reference
2063
+ for (const chunk of chunkResult.chunks) {
2064
+ // Score each chunk's importance separately
2065
+ const chunkImportance = scoreImportance(
2066
+ chunk.content,
2067
+ writeCategory,
2068
+ fact.tags,
2069
+ );
2070
+ const chunkWriteSource =
2071
+ (fact as any).source === "proactive"
2072
+ ? "chunking-proactive"
2073
+ : "chunking";
2074
+
2075
+ await targetStorage.writeChunk(
2076
+ parentId,
2077
+ chunk.index,
2078
+ chunkResult.chunks.length,
2079
+ writeCategory,
2080
+ // Each chunk carries its own inline citation so provenance
2081
+ // survives when a single chunk is quoted in isolation.
2082
+ applyInlineCitation(chunk.content),
2083
+ {
2084
+ confidence: fact.confidence,
2085
+ tags: fact.tags,
2086
+ entityRef: fact.entityRef,
2087
+ source: chunkWriteSource,
2088
+ importance: chunkImportance,
2089
+ intentGoal: inferredIntent?.goal,
2090
+ intentActionType: inferredIntent?.actionType,
2091
+ intentEntityTypes: inferredIntent?.entityTypes,
2092
+ memoryKind,
2093
+ validAt: biTemporal ? biTemporal.validFrom : sourceContext?.validAt,
2094
+ // #1578: propagate end bound + provenance to chunks (cursor bugbot).
2095
+ ...(biTemporal
2096
+ ? {
2097
+ observedAt: biTemporal.observedAt,
2098
+ eventTimeSource: biTemporal.eventTimeSource,
2099
+ ...(biTemporal.validUntil
2100
+ ? { invalidAt: biTemporal.validUntil }
2101
+ : {}),
2102
+ }
2103
+ : {}),
2104
+ // Faithfulness gate (issue #1576): propagate the parent
2105
+ // fact's verdict + enforce status so a pending_review fact
2106
+ // is not indexed as active through its chunks (chatgpt P2).
2107
+ ...(faithfulnessFm ? { faithfulness: faithfulnessFm } : {}),
2108
+ // #1645 (OchiE): inherit pending_review + blockedBy so no chunk lands active.
2109
+ ...(postWriteGuard
2110
+ ? { status: "pending_review" as const }
2111
+ : (faithfulnessEnforceStatus ? { status: faithfulnessEnforceStatus } : {})),
2112
+ ...(tombstoneBlocked && parentWrite.blockedBy
2113
+ ? { blockedBy: parentWrite.blockedBy }
2114
+ : {}),
2115
+ // Claim-level provenance (issue #1575 PR 2): mirror the
2116
+ // parent's spans onto each chunk so a chunk surfaced
2117
+ // independently (memory_get/x-ray on a chunk ID) preserves
2118
+ // the verified span (chatgpt-codex-connector thread Ocvmo).
2119
+ ...(fact.sources && fact.sources.length > 0 ? { sources: fact.sources } : {}),
2120
+ ...(fact.provenance ? { provenance: fact.provenance } : {}),
2121
+ },
2122
+ );
2123
+ }
2124
+ } finally {
2125
+ // The parent memory is durable once writeMemory returns `parentId`.
2126
+ // Touch immediately around the chunk-write loop so a later chunk
2127
+ // failure still surfaces the partially durable parent/chunk files to
2128
+ // catalog-driven `writtenSince` maintenance. The final touch below
2129
+ // still refreshes `lastWriteAt` after later durable writes on success.
2130
+ }
2131
+
2132
+ if (routedRuleId) {
2133
+ log.debug(
2134
+ `routing applied for chunked memory ${parentId}: rule=${routedRuleId} category=${writeCategory} storage=${targetStorage.dir}`,
2135
+ );
2136
+ }
2137
+ log.debug(
2138
+ `chunked memory ${parentId} into ${chunkResult.chunks.length} chunks`,
2139
+ );
2140
+ trackPersistedId(targetStorage, parentId, {
2141
+ pendingReview: postWriteGuard,
2142
+ });
2143
+ // #1576 (cursor Medium): keep pending_review ids out of threadEpisodeIdsForGraph — else later active facts build thread-predecessor edges to an unfaithful memory.
2144
+ if (
2145
+ !postWriteGuard &&
2146
+ threadEpisodeIdsForGraph &&
2147
+ !threadEpisodeIdsForGraph.includes(parentId)
2148
+ ) {
2149
+ threadEpisodeIdsForGraph.push(parentId);
2150
+ }
2151
+ // #1645: same gate as the non-chunked path — a blocked chunked parent
2152
+ // must not enter the embedding-fallback index (resurrection).
2153
+ if (!postWriteGuard) {
2154
+ await this.deps.indexPersistedMemory(targetStorage, parentId);
2155
+ }
2156
+ // PR #402 Thread 1 fix: run source-namespace temporal supersession for
2157
+ // chunked writes, matching the non-chunked path. Without this the
2158
+ // source namespace retains stale facts that should have been superseded.
2159
+ // Faithfulness gate (#1576, cursor High): skip supersession for a
2160
+ // pending_review fact — an unfaithful extraction in the review queue
2161
+ // must NOT retire older active memories.
2162
+ if (!postWriteGuard) {
2163
+ try {
2164
+ const supersessionEntityRef =
2165
+ typeof (fact as any).entityRef === "string"
2166
+ ? ((fact as any).entityRef as string)
2167
+ : undefined;
2168
+ await applyTemporalSupersession({
2169
+ storage: targetStorage,
2170
+ newMemoryId: parentId,
2171
+ entityRef: supersessionEntityRef,
2172
+ structuredAttributes: fact.structuredAttributes,
2173
+ createdAt: supersessionOrderingAt(biTemporal?.validFrom ?? sourceContext?.validAt),
2174
+ // #1578 r3: an extracted end-only bound (validFrom absent) is
2175
+ // historical, not a new authoritative state — never let it
2176
+ // supersede a later active fact (codex P1 on :15534).
2177
+ enabled: lifecycleCaps.temporalSupersession &&
2178
+ !(biTemporal && !biTemporal.validFrom),
2179
+ });
2180
+ } catch (err) {
2181
+ log.warn(`temporal-supersession (chunked): unexpected error: ${err}`);
2182
+ }
2183
+ }
2184
+ // Faithfulness gate (#1576, chatgpt P2): do not promote a
2185
+ // pending_review fact to shared/profile — it must enter the review
2186
+ // queue without active copies that bypass the gate.
2187
+ if (!postWriteGuard) await promoteMemoryToShared({
2188
+ sourceStorage: targetStorage,
2189
+ category: writeCategory,
2190
+ content: fact.content,
2191
+ confidence: fact.confidence,
2192
+ tags: fact.tags,
2193
+ entityRef: fact.entityRef,
2194
+ structuredAttributes: fact.structuredAttributes,
2195
+ sourceMemoryId: parentId,
2196
+ importance,
2197
+ intentGoal: inferredIntent?.goal,
2198
+ intentActionType: inferredIntent?.actionType,
2199
+ intentEntityTypes: inferredIntent?.entityTypes,
2200
+ memoryKind,
2201
+ validAt: biTemporal ? biTemporal.validFrom : sourceContext?.validAt,
2202
+ ...(biTemporal
2203
+ ? {
2204
+ observedAt: biTemporal.observedAt,
2205
+ eventTimeSource: biTemporal.eventTimeSource,
2206
+ ...(biTemporal.validUntil ? { invalidAt: biTemporal.validUntil } : {}),
2207
+ }
2208
+ : {}),
2209
+ source: extractionWriteSource,
2210
+ ...(fact.sources && fact.sources.length > 0 ? { sources: fact.sources } : {}),
2211
+ ...(fact.provenance ? { provenance: fact.provenance } : {}),
2212
+ });
2213
+ // Register chunked content in the target storage hash index too.
2214
+ // Thread 3 fix: canonicalize by stripping any pre-existing citation
2215
+ // so the stored hash matches what the dedup check computes.
2216
+ try {
2217
+ const canonicalChunkedContent =
2218
+ citationEnabled &&
2219
+ hasCitationForTemplate(fact.content, citationTemplate)
2220
+ ? stripCitationForTemplate(fact.content, citationTemplate)
2221
+ : fact.content;
2222
+ // #1645: do NOT register a tombstone-blocked fact's content in the
2223
+ // dedup index — writeMemory already skipped it (rule 44). Re-adding
2224
+ // would let the next extraction dedup-skip the tombstone chokepoint
2225
+ // and silently ban the retired content (no pending_review row).
2226
+ if (!tombstoneBlocked) {
2227
+ await this.deps.addContentHashDedup(targetStorage, canonicalChunkedContent);
2228
+ }
2229
+ } catch (err) {
2230
+ log.warn(
2231
+ `content-hash dedup registration failed for chunked memory ${parentId}: ${err}`,
2232
+ );
2233
+ }
2234
+
2235
+ for (const chunk of chunkResult.chunks) {
2236
+ const chunkId = `${parentId}-chunk-${chunk.index}`;
2237
+ // Do NOT push chunkId into persistedIds — chunk IDs must not leak
2238
+ // into boxBuilder.onExtraction() or threading.processTurn(), which
2239
+ // only expect canonical parent memory IDs. Call indexPersistedMemory
2240
+ // directly for embedding-fallback sync of each chunk document.
2241
+ // #1645: chunks inherit pending_review under postWriteGuard — don't
2242
+ // index them into the embedding fallback (resurrection).
2243
+ if (!postWriteGuard) {
2244
+ await this.deps.indexPersistedMemory(targetStorage, chunkId);
2245
+ }
2246
+ }
2247
+ try {
2248
+ if (
2249
+ resolvePresentationCapabilities(this.deps.config).verbatimArtifacts &&
2250
+ this.deps.config.verbatimArtifactCategories.includes(writeCategory) &&
2251
+ fact.confidence >= this.deps.config.verbatimArtifactsMinConfidence &&
2252
+ !postWriteGuard
2253
+ ) {
2254
+ // Reuse citedChunkedContent so the artifact carries the same citation
2255
+ // timestamp as the parent memory write above (Fix #3 — duplicate-citation).
2256
+ await targetStorage.writeArtifact(citedChunkedContent, {
2257
+ confidence: fact.confidence,
2258
+ tags: [...fact.tags, "artifact", "chunked-parent"],
2259
+ artifactType: this.deps.artifactTypeForCategory(writeCategory),
2260
+ sourceMemoryId: parentId,
2261
+ intentGoal: inferredIntent?.goal,
2262
+ intentActionType: inferredIntent?.actionType,
2263
+ intentEntityTypes: inferredIntent?.entityTypes,
2264
+ });
2265
+ }
2266
+ // v8.2: graph edge building for chunked memories. #1576: skip pending_review.
2267
+ if (graphCaps.multiGraphMemory && !postWriteGuard) {
2268
+ try {
2269
+ const graphContext = await ensureGraphContext(targetStorage);
2270
+ const entityRef =
2271
+ typeof (fact as any).entityRef === "string"
2272
+ ? (fact as any).entityRef
2273
+ : undefined;
2274
+ const parentRelPath = resolvePersistedMemoryRelativePath({
2275
+ memoryId: parentId,
2276
+ pathById: graphContext.memoryPathById,
2277
+ category: writeCategory,
2278
+ });
2279
+ graphContext.memoryPathById.set(parentId, parentRelPath);
2280
+ appendMemoryToGraphContext({
2281
+ allMemsForGraph: graphContext.allMemsForGraph,
2282
+ storageDir: targetStorage.dir,
2283
+ memoryRelPath: parentRelPath,
2284
+ memoryId: parentId,
2285
+ category: writeCategory,
2286
+ content: fact.content ?? "",
2287
+ entityRef,
2288
+ });
2289
+ await this.deps.buildGraphEdge(
2290
+ targetStorage,
2291
+ parentRelPath,
2292
+ entityRef,
2293
+ parentId,
2294
+ fact.content ?? "",
2295
+ graphContext.allMemsForGraph,
2296
+ graphContext.memoryPathById,
2297
+ threadIdForExtraction ?? undefined,
2298
+ threadEpisodeIdsForGraph,
2299
+ graphContext.previousPersistedRelPath,
2300
+ graphCaps,
2301
+ );
2302
+ graphContext.previousPersistedRelPath = parentRelPath;
2303
+ } catch {
2304
+ /* fail-open */
2305
+ }
2306
+ }
2307
+ } finally {
2308
+ // Catalog touch (issue #1499): refresh AFTER later chunked
2309
+ // source-namespace durable mutations — temporal supersession, shared
2310
+ // promotion, optional artifact writes, and graph-edge writes — so
2311
+ // `lastWriteAt` cannot precede later file changes on successful
2312
+ // completion. Use the KNOWN routed name, not a dir-decoded guess.
2313
+ }
2314
+ trackBehaviorSignals(
2315
+ targetStorage,
2316
+ buildBehaviorSignalsForMemory({
2317
+ memoryId: parentId,
2318
+ category: writeCategory,
2319
+ content: fact.content,
2320
+ namespace: this.deps.storageDirNamespace(targetStorage.dir),
2321
+ confidence: fact.confidence,
2322
+ source: "extraction",
2323
+ }),
2324
+ );
2325
+ continue; // Skip the normal write below
2326
+ }
2327
+ }
2328
+
2329
+ // Suggest links for this memory (Phase 3A)
2330
+ if (resolveRecallEnhancementCapabilities(this.deps.config).memoryLinking && this.deps.getQmd().isAvailable()) {
2331
+ const targetNamespace = this.deps.storageDirNamespace(targetStorage.dir);
2332
+ const suggestedLinks = await this.deps.suggestLinksForMemory(
2333
+ fact.content,
2334
+ writeCategory,
2335
+ targetNamespace,
2336
+ );
2337
+ if (suggestedLinks.length > 0) {
2338
+ links.push(...suggestedLinks);
2339
+ }
2340
+ }
2341
+
2342
+ // Classify memory kind (v8.0 Phase 2B: HiMem episode/note dual store)
2343
+ const memoryKind =
2344
+ writeCategory === "procedure"
2345
+ ? undefined
2346
+ : resolvePresentationCapabilities(this.deps.config).episodeNoteMode
2347
+ ? classifyMemoryKind(fact.content, fact.tags ?? [], writeCategory)
2348
+ : undefined;
2349
+
2350
+ // Normal write (no chunking)
2351
+ // Compute the cited content once so that writeMemory and writeArtifact
2352
+ // (when verbatim artifacts are enabled) share the same citation timestamp.
2353
+ // Calling applyInlineCitation twice on the same raw content would produce
2354
+ // two different timestamps, creating duplicate citations with divergent
2355
+ // provenance metadata on the memory and artifact copies of the same fact.
2356
+ // Pass the RAW (pre-citation) fact as `contentHashSource` so the
2357
+ // fact-content hash index records the hash of the canonical fact text
2358
+ // rather than the citation-annotated variant. When inline attribution is
2359
+ // enabled, `applyInlineCitation` appends a timestamp-bearing marker, so
2360
+ // hashing the persisted body would produce a different hash on every
2361
+ // write and defeat cross-session dedup (see `findDuplicateExplicitCapture`
2362
+ // in explicit-capture.ts which calls `hasFactContentHash(candidate.content)`
2363
+ // on raw content).
2364
+ const rawPersistBody =
2365
+ writeCategory === "procedure"
2366
+ ? buildProcedurePersistBody(fact.content, fact.procedureSteps)
2367
+ : fact.content;
2368
+ const citedFactContent = applyInlineCitation(rawPersistBody);
2369
+ const factWrite = await targetStorage.writeMemory(
2370
+ writeCategory,
2371
+ citedFactContent,
2372
+ {
2373
+ confidence: fact.confidence,
2374
+ tags: fact.tags,
2375
+ entityRef:
2376
+ typeof (fact as any).entityRef === "string"
2377
+ ? (fact as any).entityRef
2378
+ : undefined,
2379
+ source: extractionWriteSource,
2380
+ importance,
2381
+ supersedes,
2382
+ links: links.length > 0 ? links : undefined,
2383
+ intentGoal: inferredIntent?.goal,
2384
+ intentActionType: inferredIntent?.actionType,
2385
+ intentEntityTypes: inferredIntent?.entityTypes,
2386
+ memoryKind,
2387
+ structuredAttributes: fact.structuredAttributes,
2388
+ validAt: biTemporal ? biTemporal.validFrom : sourceContext?.validAt,
2389
+ ...(biTemporal ? { observedAt: biTemporal.observedAt, eventTimeSource: biTemporal.eventTimeSource, ...(biTemporal.validUntil ? { invalidAt: biTemporal.validUntil } : {}) } : {}),
2390
+ contentHashSource: writeCategory === "fact" ? fact.content : undefined,
2391
+ // Faithfulness gate (issue #1576).
2392
+ ...(faithfulnessFm ? { faithfulness: faithfulnessFm } : {}),
2393
+ ...(faithfulnessEnforceStatus ? { status: faithfulnessEnforceStatus } : {}),
2394
+ // Claim-level provenance spans (issue #1575 PR 2). Carry verified
2395
+ // sources + the coarse strength tag from the extraction validator
2396
+ // through to frontmatter so they survive end-to-end.
2397
+ ...(fact.sources && fact.sources.length > 0 ? { sources: fact.sources } : {}),
2398
+ ...(fact.provenance ? { provenance: fact.provenance } : {}),
2399
+ },
2400
+ );
2401
+ const memoryId = factWrite.id;
2402
+ // #1645: surface the tombstone block; gate active post-write paths like #1576
2403
+ // so a blocked fact creates no active shared copy / supersession / graph entry.
2404
+ const tombstoneBlocked = factWrite.tombstoneBlocked;
2405
+ const postWriteGuard =
2406
+ faithfulnessEnforceStatus === "pending_review" || tombstoneBlocked;
2407
+ // #1645: defer contradiction auto-resolve until tombstone status is
2408
+ // known (see applyDeferredContradictionResolve).
2409
+ await this.deps.applyDeferredContradictionResolve(
2410
+ contradiction,
2411
+ targetStorage,
2412
+ memoryId,
2413
+ postWriteGuard,
2414
+ );
2415
+ if (routedRuleId) {
2416
+ log.debug(
2417
+ `routing applied for memory ${memoryId}: rule=${routedRuleId} category=${writeCategory} storage=${targetStorage.dir}`,
2418
+ );
2419
+ }
2420
+ // Temporal supersession (issue #375): when the new fact has structured
2421
+ // attributes, retire any older fact with the same entity + attribute
2422
+ // key that has a conflicting value. Faithfulness gate (#1576, cursor
2423
+ // High): skip for a pending_review fact — an unfaithful extraction in
2424
+ // the review queue must NOT retire older active memories.
2425
+ if (!postWriteGuard) {
2426
+ try {
2427
+ const supersessionEntityRef =
2428
+ typeof (fact as any).entityRef === "string"
2429
+ ? ((fact as any).entityRef as string)
2430
+ : undefined;
2431
+ await applyTemporalSupersession({
2432
+ storage: targetStorage,
2433
+ newMemoryId: memoryId,
2434
+ entityRef: supersessionEntityRef,
2435
+ structuredAttributes: fact.structuredAttributes,
2436
+ createdAt: supersessionOrderingAt(biTemporal?.validFrom ?? sourceContext?.validAt),
2437
+ enabled: lifecycleCaps.temporalSupersession &&
2438
+ !(biTemporal && !biTemporal.validFrom),
2439
+ });
2440
+ } catch (err) {
2441
+ log.warn(`temporal-supersession: unexpected error: ${err}`);
2442
+ }
2443
+ }
2444
+ try {
2445
+ trackBehaviorSignals(
2446
+ targetStorage,
2447
+ buildBehaviorSignalsForMemory({
2448
+ memoryId,
2449
+ category: writeCategory,
2450
+ content: fact.content,
2451
+ namespace: this.deps.storageDirNamespace(targetStorage.dir),
2452
+ confidence: fact.confidence,
2453
+ source: "extraction",
2454
+ }),
2455
+ );
2456
+ trackPersistedId(targetStorage, memoryId, {
2457
+ pendingReview: postWriteGuard,
2458
+ });
2459
+ if (
2460
+ !postWriteGuard &&
2461
+ threadEpisodeIdsForGraph &&
2462
+ !threadEpisodeIdsForGraph.includes(memoryId)
2463
+ ) {
2464
+ threadEpisodeIdsForGraph.push(memoryId);
2465
+ }
2466
+ // #1645: a tombstone-blocked / pending_review fact must NOT enter the
2467
+ // embedding-fallback index — otherwise embedding recall surfaces the
2468
+ // pending_review row (resurrection). Gate on postWriteGuard like the
2469
+ // surrounding supersession / promotion / graph paths.
2470
+ if (!postWriteGuard) {
2471
+ await this.deps.indexPersistedMemory(targetStorage, memoryId);
2472
+ }
2473
+ // Faithfulness gate (#1576, chatgpt P2): skip promotion for a
2474
+ // pending_review fact so no active shared/profile copy bypasses the gate.
2475
+ if (!postWriteGuard) await promoteMemoryToShared({
2476
+ sourceStorage: targetStorage,
2477
+ category: writeCategory,
2478
+ content: fact.content,
2479
+ confidence: fact.confidence,
2480
+ tags: fact.tags,
2481
+ entityRef:
2482
+ typeof (fact as any).entityRef === "string"
2483
+ ? (fact as any).entityRef
2484
+ : undefined,
2485
+ structuredAttributes: fact.structuredAttributes,
2486
+ sourceMemoryId: memoryId,
2487
+ importance,
2488
+ intentGoal: inferredIntent?.goal,
2489
+ intentActionType: inferredIntent?.actionType,
2490
+ intentEntityTypes: inferredIntent?.entityTypes,
2491
+ memoryKind,
2492
+ validAt: biTemporal ? biTemporal.validFrom : sourceContext?.validAt,
2493
+ ...(biTemporal
2494
+ ? {
2495
+ observedAt: biTemporal.observedAt,
2496
+ eventTimeSource: biTemporal.eventTimeSource,
2497
+ ...(biTemporal.validUntil ? { invalidAt: biTemporal.validUntil } : {}),
2498
+ }
2499
+ : {}),
2500
+ source: extractionWriteSource,
2501
+ ...(fact.sources && fact.sources.length > 0 ? { sources: fact.sources } : {}),
2502
+ ...(fact.provenance ? { provenance: fact.provenance } : {}),
2503
+ });
2504
+ // v8.2: graph edge building (fail-open). #1576: skip pending_review facts.
2505
+ if (graphCaps.multiGraphMemory && !postWriteGuard) {
2506
+ try {
2507
+ const graphContext = await ensureGraphContext(targetStorage);
2508
+ const entityRef =
2509
+ typeof (fact as any).entityRef === "string"
2510
+ ? (fact as any).entityRef
2511
+ : undefined;
2512
+ const memoryRelPath = resolvePersistedMemoryRelativePath({
2513
+ memoryId,
2514
+ pathById: graphContext.memoryPathById,
2515
+ category: writeCategory,
2516
+ });
2517
+ graphContext.memoryPathById.set(memoryId, memoryRelPath);
2518
+ appendMemoryToGraphContext({
2519
+ allMemsForGraph: graphContext.allMemsForGraph,
2520
+ storageDir: targetStorage.dir,
2521
+ memoryRelPath: memoryRelPath,
2522
+ memoryId,
2523
+ category: writeCategory,
2524
+ content: fact.content ?? "",
2525
+ entityRef,
2526
+ });
2527
+ await this.deps.buildGraphEdge(
2528
+ targetStorage,
2529
+ memoryRelPath,
2530
+ entityRef,
2531
+ memoryId,
2532
+ fact.content ?? "",
2533
+ graphContext.allMemsForGraph,
2534
+ graphContext.memoryPathById,
2535
+ threadIdForExtraction ?? undefined,
2536
+ threadEpisodeIdsForGraph,
2537
+ graphContext.previousPersistedRelPath,
2538
+ graphCaps,
2539
+ );
2540
+ graphContext.previousPersistedRelPath = memoryRelPath;
2541
+ } catch {
2542
+ /* fail-open */
2543
+ }
2544
+ }
2545
+ if (
2546
+ resolvePresentationCapabilities(this.deps.config).verbatimArtifacts &&
2547
+ this.deps.config.verbatimArtifactCategories.includes(writeCategory) &&
2548
+ fact.confidence >= this.deps.config.verbatimArtifactsMinConfidence &&
2549
+ !postWriteGuard
2550
+ ) {
2551
+ // Reuse citedFactContent so the artifact carries the same citation
2552
+ // timestamp as the memory write above (Fix #3 — duplicate-citation).
2553
+ await targetStorage.writeArtifact(citedFactContent, {
2554
+ confidence: fact.confidence,
2555
+ tags: [...fact.tags, "artifact"],
2556
+ artifactType: this.deps.artifactTypeForCategory(writeCategory),
2557
+ sourceMemoryId: memoryId,
2558
+ intentGoal: inferredIntent?.goal,
2559
+ intentActionType: inferredIntent?.actionType,
2560
+ intentEntityTypes: inferredIntent?.entityTypes,
2561
+ });
2562
+ }
2563
+ // Register in the target storage content-hash index after successful
2564
+ // write. Thread 3 fix: canonicalize by stripping any pre-existing
2565
+ // citation so the stored hash matches what the dedup check computes.
2566
+ try {
2567
+ const canonicalFactContent =
2568
+ citationEnabled &&
2569
+ hasCitationForTemplate(fact.content, citationTemplate)
2570
+ ? stripCitationForTemplate(fact.content, citationTemplate)
2571
+ : fact.content;
2572
+ const hashRegisterKey =
2573
+ writeCategory === "procedure"
2574
+ ? buildProcedurePersistBody(fact.content, fact.procedureSteps)
2575
+ : canonicalFactContent;
2576
+ // #1645: do NOT register a tombstone-blocked fact's content in the dedup
2577
+ // index (rule 44 defeat) — see chunked path comment.
2578
+ if (!tombstoneBlocked) {
2579
+ await this.deps.addContentHashDedup(targetStorage, hashRegisterKey);
2580
+ }
2581
+ } catch (err) {
2582
+ log.warn(
2583
+ `content-hash dedup registration failed for memory ${memoryId}: ${err}`,
2584
+ );
2585
+ }
2586
+ } finally {
2587
+ // Catalog touch (issue #1499): record AFTER every synchronous
2588
+ // source-namespace mutation in the non-chunked path: writeMemory,
2589
+ // temporal supersession, graph edges, and optional verbatim artifacts.
2590
+ // The `finally` preserves the write touch when post-write indexing or
2591
+ // promotion fails after the canonical memory is already durable. Use the
2592
+ // KNOWN routed name, not a dir-decoded guess (NCQI0).
2593
+ }
2594
+ }
2595
+
2596
+ // Tracks whether THIS extraction persisted any durable, non-fact output to the
2597
+ // BASE namespace's storage (entity / relationship / profile / question). The
2598
+ // per-fact catalog touch (storage chokepoint #1522) only fires inside the fact write loop, so a
2599
+ // fact-less extraction that still persists durable data must record exactly one
2600
+ // base-namespace catalog touch after all writes complete (NHZEZ, codex P2).
2601
+ let durableNonFactWritten = false;
2602
+ let durableNonFactTouchRecorded = false;
2603
+ const touchBaseNonFactNamespace = () => {
2604
+ const baseTouchNamespace =
2605
+ baseNamespace && baseNamespace.length > 0
2606
+ ? baseNamespace
2607
+ : this.deps.storageDirNamespace(storage.dir);
2608
+ };
2609
+ const recordDurableNonFactWrite = () => {
2610
+ durableNonFactWritten = true;
2611
+ if (durableNonFactTouchRecorded) return;
2612
+ durableNonFactTouchRecorded = true;
2613
+ touchBaseNonFactNamespace();
2614
+ };
2615
+ for (const entity of entities) {
2616
+ try {
2617
+ const name = (entity as any)?.name;
2618
+ const type = (entity as any)?.type;
2619
+ if (
2620
+ typeof name !== "string" ||
2621
+ !name.trim() ||
2622
+ typeof type !== "string" ||
2623
+ !type.trim()
2624
+ ) {
2625
+ continue;
2626
+ }
2627
+ const safeFacts = Array.isArray((entity as any)?.facts)
2628
+ ? (entity as any).facts.filter((f: any) => typeof f === "string")
2629
+ : [];
2630
+ const id = await storage.writeEntity(name, type, safeFacts, {
2631
+ source: typeof (entity as any)?.source === "string" ? (entity as any).source : "extraction",
2632
+ timestamp: sourceContext?.validAt,
2633
+ sessionKey: sourceContext?.sessionKey,
2634
+ principal: sourceContext?.principal,
2635
+ structuredSections: Array.isArray((entity as any)?.structuredSections)
2636
+ ? (entity as any).structuredSections
2637
+ : undefined,
2638
+ });
2639
+ if (id) {
2640
+ trackPersistedId(storage, id);
2641
+ recordDurableNonFactWrite();
2642
+ }
2643
+ } catch (err) {
2644
+ log.warn(`persistExtraction: entity write failed: ${err}`);
2645
+ }
2646
+ }
2647
+
2648
+ // Persist entity relationships (v7.0)
2649
+ if (
2650
+ resolveRecallEnhancementCapabilities(this.deps.config).entityRelationships &&
2651
+ Array.isArray(result.relationships)
2652
+ ) {
2653
+ for (const rel of result.relationships.slice(0, 5)) {
2654
+ if (!rel.source || !rel.target || !rel.label) continue;
2655
+ try {
2656
+ // Add bidirectional relationship
2657
+ await storage.addEntityRelationship(rel.source, {
2658
+ target: rel.target,
2659
+ label: rel.label,
2660
+ });
2661
+ recordDurableNonFactWrite();
2662
+ await storage.addEntityRelationship(rel.target, {
2663
+ target: rel.source,
2664
+ label: `${rel.label} (reverse)`,
2665
+ });
2666
+ recordDurableNonFactWrite();
2667
+ } catch (err) {
2668
+ log.debug(`relationship persist failed: ${err}`);
2669
+ }
2670
+ }
2671
+ }
2672
+
2673
+ // Persist entity activity (v7.0)
2674
+ if (resolveRecallEnhancementCapabilities(this.deps.config).entityActivityLog) {
2675
+ const today = new Date().toISOString().slice(0, 10);
2676
+ for (const entity of entities) {
2677
+ const name = (entity as any)?.name;
2678
+ const type = (entity as any)?.type;
2679
+ if (typeof name !== "string" || typeof type !== "string") continue;
2680
+ try {
2681
+ const normalized = storage.normalizeEntityName(name, type);
2682
+ await storage.addEntityActivity(
2683
+ normalized,
2684
+ { date: today, note: "Mentioned in conversation" },
2685
+ this.deps.config.entityActivityLogMaxEntries,
2686
+ );
2687
+ } catch (err) {
2688
+ log.debug(`activity persist failed: ${err}`);
2689
+ }
2690
+ }
2691
+ }
2692
+
2693
+ if (profileUpdates.length > 0) {
2694
+ await storage.appendToProfile(profileUpdates);
2695
+ recordDurableNonFactWrite();
2696
+ }
2697
+
2698
+ // Persist questions
2699
+ for (const q of questions) {
2700
+ const id = await storage.writeQuestion(q.question, q.context, q.priority);
2701
+ if (id) {
2702
+ trackPersistedId(storage, id);
2703
+ recordDurableNonFactWrite();
2704
+ }
2705
+ }
2706
+
2707
+ // Persist identity reflection. This writes durable namespace-local state, so
2708
+ // an identity-ONLY extraction (no facts/entities/profile/questions) still
2709
+ // counts as a durable non-fact write for the catalog touch below (NIIly).
2710
+ // Only count it when the write actually succeeds (best-effort write); the
2711
+ // touch is recorded AFTER this so a rolled-back/failed write never touches.
2712
+ if (resolveRecallEnhancementCapabilities(this.deps.config).identity && result.identityReflection) {
2713
+ try {
2714
+ await storage.appendIdentityReflection(result.identityReflection);
2715
+ recordDurableNonFactWrite();
2716
+ } catch (err) {
2717
+ log.debug(`identity reflection write failed: ${err}`);
2718
+ }
2719
+ }
2720
+
2721
+ // Catalog touch for durable NON-FACT outputs (NHZEZ / NIIly, codex P2). The
2722
+ // per-fact catalog touch (storage chokepoint #1522) above only fires inside the fact write loop, so
2723
+ // an extraction that persists ONLY entities, relationships, profile updates,
2724
+ // questions, or an identity reflection (no facts) would record durable data to
2725
+ // the BASE namespace's storage without ever touching the catalog — leaving that
2726
+ // namespace's `lastWriteAt` stale so `listNamespaces({writtenSince})` /
2727
+ // write-recency QMD maintenance miss the write. All of these are written to the
2728
+ // BASE `storage` (not the per-fact routed `targetStorage`), so we record ONE
2729
+ // base-namespace touch here, AFTER every non-fact write completes. Use the
2730
+ // KNOWN base namespace name, not a dir-decoded guess (NCQI0). One touch per
2731
+ // namespace per extraction — `markWrite` is idempotent, so if the fact path
2732
+ // already touched the base namespace this only refreshes `lastWriteAt`.
2733
+ // Best-effort and failure-tolerant (storage chokepoint #1522 swallows errors).
2734
+ if (durableNonFactWritten) {
2735
+ touchBaseNonFactNamespace();
2736
+ }
2737
+
2738
+ // Save any content-hash indexes touched during the batch.
2739
+ await this.deps.saveContentHashIndexes().catch((err) =>
2740
+ log.warn(`content-hash index save failed: ${err}`),
2741
+ );
2742
+
2743
+ for (const {
2744
+ storage: targetStorage,
2745
+ events,
2746
+ } of behaviorSignalsByStorage.values()) {
2747
+ const dedupedSignals = dedupeBehaviorSignalsByMemoryAndHash(events);
2748
+ if (dedupedSignals.length === 0) continue;
2749
+ await targetStorage
2750
+ .appendBehaviorSignals(dedupedSignals)
2751
+ .catch((err) =>
2752
+ log.warn(`appendBehaviorSignals failed (non-fatal): ${err}`),
2753
+ );
2754
+ }
2755
+
2756
+ const dedupSuffix = dedupedCount > 0 ? ` (${dedupedCount} deduped)` : "";
2757
+ const gatedSuffix =
2758
+ importanceGatedCount > 0 ? ` (${importanceGatedCount} gated)` : "";
2759
+ const judgeSuffix =
2760
+ judgeGatedCount > 0 ? ` (${judgeGatedCount} judge-rejected)` : "";
2761
+ const redactionSuffix =
2762
+ redactionGatedCount > 0 ? ` (${redactionGatedCount} redacted)` : "";
2763
+ log.info(
2764
+ `persisted: ${facts.length - dedupedCount - importanceGatedCount - judgeGatedCount - redactionGatedCount} facts${dedupSuffix}${gatedSuffix}${judgeSuffix}${redactionSuffix}, ${entities.length} entities, ${questions.length} questions, ${profileUpdates.length} profile updates`,
2765
+ );
2766
+
2767
+ // Update temporal + tag indexes (v8.1) — fire-and-forget, fail-open
2768
+ void (async () => {
2769
+ if (persistedIdsByStorage.size === 0) {
2770
+ await this.deps.updateTemporalTagIndexes(storage, []);
2771
+ return;
2772
+ }
2773
+ for (const entry of persistedIdsByStorage.values()) {
2774
+ await this.deps.updateTemporalTagIndexes(entry.storage, entry.ids);
2775
+ }
2776
+ })().catch((err) =>
2777
+ log.debug(`temporal-index update error (non-fatal): ${err}`),
2778
+ );
2779
+
2780
+ // #1635: surface pending_review ids so the thread episode set excludes them.
2781
+ this.deps.setLastPersistExtractionPendingReviewIds(pendingReviewPersistedIds);
2782
+ // Return the persisted fact IDs for threading
2783
+ return persistedIds;
2784
+ }
2785
+ }