@remnic/core 9.3.701 → 9.3.702

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (301) hide show
  1. package/dist/access-boundary.d.ts +7 -7
  2. package/dist/access-boundary.js +15 -12
  3. package/dist/access-cli.js +36 -33
  4. package/dist/access-cli.js.map +1 -1
  5. package/dist/access-http.d.ts +7 -7
  6. package/dist/access-http.js +18 -15
  7. package/dist/access-mcp.d.ts +7 -7
  8. package/dist/access-mcp.js +17 -14
  9. package/dist/access-operations.d.ts +7 -7
  10. package/dist/access-operations.js +16 -13
  11. package/dist/{access-service-CGVWK6lZ.d.ts → access-service-COCzhgEL.d.ts} +397 -19
  12. package/dist/access-service.d.ts +6 -6
  13. package/dist/access-service.js +14 -11
  14. package/dist/access-surface-catalog.d.ts +7 -7
  15. package/dist/action-confidence.d.ts +1 -1
  16. package/dist/active-memory-bridge.d.ts +1 -1
  17. package/dist/active-memory-bridge.js +2 -2
  18. package/dist/active-recall.d.ts +1 -1
  19. package/dist/active-recall.js +7 -6
  20. package/dist/active-recall.js.map +1 -1
  21. package/dist/behavior-learner.d.ts +1 -1
  22. package/dist/behavior-signals.d.ts +1 -1
  23. package/dist/bootstrap.d.ts +5 -5
  24. package/dist/briefing.d.ts +1 -1
  25. package/dist/briefing.js +6 -5
  26. package/dist/buffer-surprise-report.d.ts +1 -1
  27. package/dist/buffer.d.ts +1 -1
  28. package/dist/calibration.d.ts +1 -1
  29. package/dist/capabilities.d.ts +1 -1
  30. package/dist/{catalog-DBIghceA.d.ts → catalog-DxCzhjE6.d.ts} +2 -2
  31. package/dist/causal-behavior.d.ts +1 -1
  32. package/dist/causal-consolidation.d.ts +1 -1
  33. package/dist/causal-consolidation.js +7 -6
  34. package/dist/causal-consolidation.js.map +1 -1
  35. package/dist/{chunk-YXLT4EMM.js → chunk-4463LUIE.js} +12 -2
  36. package/dist/chunk-4463LUIE.js.map +1 -0
  37. package/dist/{chunk-R5DB26G6.js → chunk-4SBW47WK.js} +9 -23
  38. package/dist/chunk-4SBW47WK.js.map +1 -0
  39. package/dist/{chunk-KF4TXW7Z.js → chunk-5VXVXJH6.js} +149 -6
  40. package/dist/{chunk-KF4TXW7Z.js.map → chunk-5VXVXJH6.js.map} +1 -1
  41. package/dist/{chunk-ODTWHSY2.js → chunk-64XXOQRW.js} +2 -2
  42. package/dist/{chunk-ZT7B64BE.js → chunk-6TAETM63.js} +2 -2
  43. package/dist/{chunk-HF4N43Q7.js → chunk-6VVP6NK7.js} +2 -2
  44. package/dist/{chunk-FN2SM5SN.js → chunk-ABCGDW3A.js} +75 -12
  45. package/dist/chunk-ABCGDW3A.js.map +1 -0
  46. package/dist/{chunk-7TAQEPLE.js → chunk-AMNLZ6SF.js} +2 -2
  47. package/dist/{chunk-KS7WQ4BZ.js → chunk-CZJ6QSKG.js} +3 -3
  48. package/dist/{chunk-27LQPUMZ.js → chunk-EOOVJK2U.js} +3 -3
  49. package/dist/{chunk-QP37KL5H.js → chunk-EYJD6KIO.js} +2 -2
  50. package/dist/{chunk-WFEZUGU5.js → chunk-EZR35XHX.js} +2 -2
  51. package/dist/{chunk-UU6MVCJ6.js → chunk-FIOYURII.js} +32 -25
  52. package/dist/chunk-FIOYURII.js.map +1 -0
  53. package/dist/{chunk-JO3E5VGS.js → chunk-FVI5B7DE.js} +2 -2
  54. package/dist/{chunk-SFMRLXIV.js → chunk-GFIARMA7.js} +15 -7
  55. package/dist/chunk-GFIARMA7.js.map +1 -0
  56. package/dist/{chunk-PONNZ54D.js → chunk-GY3SKOS4.js} +4 -4
  57. package/dist/{chunk-U33LWTQQ.js → chunk-HV57RHMD.js} +4 -4
  58. package/dist/{chunk-IKNQAGBV.js → chunk-IX3UQT4H.js} +1 -1
  59. package/dist/{chunk-IKNQAGBV.js.map → chunk-IX3UQT4H.js.map} +1 -1
  60. package/dist/{chunk-IYOPIG3E.js → chunk-IX72AAMZ.js} +2 -2
  61. package/dist/{chunk-JKOKX3PS.js → chunk-JMA4RYRN.js} +2 -2
  62. package/dist/{chunk-DEDQXIDL.js → chunk-JTKFZMZ7.js} +2 -2
  63. package/dist/{chunk-OLOYQZFB.js → chunk-KC6TCAWV.js} +5 -5
  64. package/dist/{chunk-ROZJACKP.js → chunk-KKK7YTYN.js} +4 -1
  65. package/dist/chunk-KKK7YTYN.js.map +1 -0
  66. package/dist/{chunk-EC2AYKRX.js → chunk-L6W77GWW.js} +10 -24
  67. package/dist/chunk-L6W77GWW.js.map +1 -0
  68. package/dist/{chunk-ED35D32I.js → chunk-LQ4J7ELC.js} +2 -2
  69. package/dist/chunk-LUPVCGYK.js +63 -0
  70. package/dist/chunk-LUPVCGYK.js.map +1 -0
  71. package/dist/{chunk-TFVVONWD.js → chunk-MBUM2Y3L.js} +2 -2
  72. package/dist/{chunk-T5QAZIBO.js → chunk-MOXFPLD6.js} +3 -3
  73. package/dist/{chunk-XTIRCSIH.js → chunk-MXEWQKM7.js} +2 -2
  74. package/dist/{chunk-ZYNMX6IU.js → chunk-NUIJEGVD.js} +6 -6
  75. package/dist/{chunk-FUCJAZ25.js → chunk-OXEAMU42.js} +529 -9
  76. package/dist/chunk-OXEAMU42.js.map +1 -0
  77. package/dist/{chunk-YO4MBK3I.js → chunk-QGJAGC2J.js} +2 -2
  78. package/dist/{chunk-ISLJ5WIM.js → chunk-QIMFOCSH.js} +2 -2
  79. package/dist/{chunk-IJEZMWKA.js → chunk-SFOAQQDJ.js} +3 -3
  80. package/dist/{chunk-HRUULBBV.js → chunk-SHRRWOVY.js} +81 -4
  81. package/dist/chunk-SHRRWOVY.js.map +1 -0
  82. package/dist/{chunk-NINRTFSV.js → chunk-SINGJCUR.js} +5 -5
  83. package/dist/{chunk-ZPQVJEVQ.js → chunk-SK2CR6MW.js} +146 -2
  84. package/dist/chunk-SK2CR6MW.js.map +1 -0
  85. package/dist/{chunk-6W2D6FGG.js → chunk-TGAHHCB6.js} +2 -2
  86. package/dist/{chunk-D75JXBV4.js → chunk-WN4GHSDH.js} +2 -2
  87. package/dist/{chunk-K4DWSPMW.js → chunk-WRGPE6AW.js} +2 -2
  88. package/dist/{chunk-4HIAWLA2.js → chunk-XGMCUY5P.js} +60 -114
  89. package/dist/chunk-XGMCUY5P.js.map +1 -0
  90. package/dist/{chunk-SDPDU2PM.js → chunk-YTMDF6S7.js} +2 -2
  91. package/dist/{chunk-MNU5G4TK.js → chunk-Z7XEIAV4.js} +2 -2
  92. package/dist/{chunk-HXHKLVAS.js → chunk-ZCEI242W.js} +23 -23
  93. package/dist/{cli-D3XeenwN.d.ts → cli-BM4xQPp4.d.ts} +3 -3
  94. package/dist/cli.d.ts +7 -7
  95. package/dist/cli.js +33 -32
  96. package/dist/compounding/engine.d.ts +1 -1
  97. package/dist/compounding/engine.js +6 -5
  98. package/dist/compounding/preference-consolidator.d.ts +1 -1
  99. package/dist/compression-optimizer.d.ts +1 -1
  100. package/dist/config.d.ts +1 -1
  101. package/dist/config.js +4 -2
  102. package/dist/connectors/codex-materialize-runner.d.ts +1 -1
  103. package/dist/connectors/codex-materialize-runner.js +6 -5
  104. package/dist/connectors/codex-materialize.d.ts +1 -1
  105. package/dist/connectors/index.d.ts +1 -1
  106. package/dist/connectors/index.js +7 -6
  107. package/dist/consolidation-provenance-check.d.ts +1 -1
  108. package/dist/consolidation-undo.d.ts +1 -1
  109. package/dist/contradiction/index.d.ts +2 -2
  110. package/dist/conversation-index/backend.d.ts +1 -1
  111. package/dist/conversation-index/chunker.d.ts +1 -1
  112. package/dist/conversation-index/faiss-adapter.d.ts +1 -1
  113. package/dist/conversation-index/indexer.d.ts +1 -1
  114. package/dist/conversation-index/search.d.ts +1 -1
  115. package/dist/day-summary.d.ts +1 -1
  116. package/dist/delinearize.d.ts +1 -1
  117. package/dist/direct-answer-wiring.d.ts +1 -1
  118. package/dist/direct-answer.d.ts +1 -1
  119. package/dist/embedding-fallback.d.ts +1 -1
  120. package/dist/enrichment/index.d.ts +1 -1
  121. package/dist/entity-retrieval.d.ts +1 -1
  122. package/dist/entity-retrieval.js +6 -5
  123. package/dist/entity-schema.d.ts +1 -1
  124. package/dist/event-order-recall.js +2 -1
  125. package/dist/explicit-capture.d.ts +5 -5
  126. package/dist/explicit-capture.js +2 -2
  127. package/dist/explicit-cue-recall.js +2 -1
  128. package/dist/extraction-faithfulness.d.ts +1 -1
  129. package/dist/extraction-judge-telemetry.d.ts +1 -1
  130. package/dist/extraction-judge-training.d.ts +1 -1
  131. package/dist/extraction-judge.d.ts +1 -1
  132. package/dist/extraction.d.ts +14 -1
  133. package/dist/extraction.js +5 -3
  134. package/dist/fallback-llm.d.ts +1 -1
  135. package/dist/identity-continuity.d.ts +1 -1
  136. package/dist/importance.d.ts +1 -1
  137. package/dist/index.d.ts +123 -123
  138. package/dist/index.js +53 -52
  139. package/dist/index.js.map +1 -1
  140. package/dist/intent.d.ts +1 -1
  141. package/dist/lcm/engine.d.ts +1 -1
  142. package/dist/lcm/index.d.ts +1 -1
  143. package/dist/lcm/tools.d.ts +1 -1
  144. package/dist/lifecycle.d.ts +1 -1
  145. package/dist/live-connectors-runner.d.ts +1 -1
  146. package/dist/local-llm.d.ts +1 -1
  147. package/dist/maintenance/memory-governance.d.ts +1 -1
  148. package/dist/maintenance/memory-governance.js +6 -5
  149. package/dist/maintenance/rebuild-memory-lifecycle-ledger.js +6 -5
  150. package/dist/maintenance/rebuild-memory-projection.js +7 -6
  151. package/dist/mcp-memory-inspector-app.d.ts +7 -7
  152. package/dist/memory-action-policy.d.ts +1 -1
  153. package/dist/memory-cache.d.ts +1 -1
  154. package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
  155. package/dist/memory-projection-store.d.ts +1 -1
  156. package/dist/memory-provenance.d.ts +1 -1
  157. package/dist/memory-worth-outcomes.d.ts +1 -1
  158. package/dist/models-json.d.ts +1 -1
  159. package/dist/namespaces/migrate.d.ts +2 -2
  160. package/dist/namespaces/migrate.js +7 -6
  161. package/dist/namespaces/principal.d.ts +1 -1
  162. package/dist/namespaces/search.d.ts +1 -1
  163. package/dist/namespaces/storage.d.ts +2 -2
  164. package/dist/namespaces/storage.js +6 -5
  165. package/dist/native-knowledge.d.ts +1 -1
  166. package/dist/operator-toolkit.d.ts +1 -1
  167. package/dist/operator-toolkit.js +12 -11
  168. package/dist/orchestration/maintenance.d.ts +2 -2
  169. package/dist/orchestration/maintenance.js +8 -7
  170. package/dist/{orchestrator-BzMCZlKn.d.ts → orchestrator-C9CDWAm6.d.ts} +4 -4
  171. package/dist/orchestrator.d.ts +5 -5
  172. package/dist/orchestrator.js +29 -27
  173. package/dist/patterns-cli.d.ts +1 -1
  174. package/dist/policy-runtime.d.ts +1 -1
  175. package/dist/provenance.d.ts +36 -2
  176. package/dist/provenance.js +5 -1
  177. package/dist/qmd-recall-cache.d.ts +1 -1
  178. package/dist/qmd.d.ts +1 -1
  179. package/dist/recall-disclosure-escalation.d.ts +1 -1
  180. package/dist/recall-explain-renderer.d.ts +1 -1
  181. package/dist/recall-explain-renderer.js +3 -3
  182. package/dist/recall-pipeline-stages.d.ts +15 -0
  183. package/dist/recall-pipeline-stages.js +3 -56
  184. package/dist/recall-pipeline-stages.js.map +1 -1
  185. package/dist/recall-planner-llm.d.ts +1 -1
  186. package/dist/recall-state.d.ts +1 -1
  187. package/dist/recall-tag-filter.d.ts +1 -1
  188. package/dist/recall-xray-cli.d.ts +1 -1
  189. package/dist/recall-xray-cli.js +4 -4
  190. package/dist/recall-xray-renderer.d.ts +1 -1
  191. package/dist/recall-xray-renderer.js +3 -3
  192. package/dist/recall-xray.d.ts +1 -1
  193. package/dist/recall-xray.js +2 -2
  194. package/dist/resolve-auth-token.d.ts +1 -1
  195. package/dist/response-guidance-recall.js +2 -1
  196. package/dist/resume-bundles.js +7 -5
  197. package/dist/retrieval-agents.d.ts +1 -1
  198. package/dist/retrieval-tiers.d.ts +1 -1
  199. package/dist/routing/engine.d.ts +1 -1
  200. package/dist/routing/store.d.ts +1 -1
  201. package/dist/schemas.d.ts +19 -0
  202. package/dist/schemas.js +1 -1
  203. package/dist/search/embed-helper.d.ts +1 -1
  204. package/dist/search/factory.d.ts +1 -1
  205. package/dist/search/index.d.ts +1 -1
  206. package/dist/search/lancedb-backend.d.ts +1 -1
  207. package/dist/search/meilisearch-backend.d.ts +1 -1
  208. package/dist/search/noop-backend.d.ts +1 -1
  209. package/dist/search/orama-backend.d.ts +1 -1
  210. package/dist/search/port.d.ts +1 -1
  211. package/dist/search/remote-backend.d.ts +1 -1
  212. package/dist/{semantic-DJR8_DMQ.d.ts → semantic-SLAa_prH.d.ts} +1 -1
  213. package/dist/{semantic-consolidation-BtUfv-AL.d.ts → semantic-consolidation-DgFXyALl.d.ts} +1 -1
  214. package/dist/semantic-consolidation.d.ts +2 -2
  215. package/dist/semantic-consolidation.js +7 -6
  216. package/dist/semantic-rule-promotion.js +6 -5
  217. package/dist/semantic-rule-verifier.d.ts +1 -1
  218. package/dist/semantic-rule-verifier.js +6 -5
  219. package/dist/session-observer-bands.d.ts +1 -1
  220. package/dist/session-observer-state.d.ts +1 -1
  221. package/dist/shared-context/manager.d.ts +1 -1
  222. package/dist/signal.d.ts +1 -1
  223. package/dist/storage.d.ts +13 -1
  224. package/dist/storage.js +5 -4
  225. package/dist/summarizer.d.ts +1 -1
  226. package/dist/summary-snapshot.d.ts +1 -1
  227. package/dist/targeted-fact-recall.js +2 -1
  228. package/dist/temporal-supersession.d.ts +1 -1
  229. package/dist/temporal-validity.d.ts +1 -1
  230. package/dist/threading.d.ts +1 -1
  231. package/dist/tier-migration.d.ts +1 -1
  232. package/dist/tier-routing.d.ts +1 -1
  233. package/dist/topics.d.ts +1 -1
  234. package/dist/transcript.d.ts +1 -1
  235. package/dist/transcript.js +2 -2
  236. package/dist/{types-PuiPZ9iE.d.ts → types-MWKPZnM0.d.ts} +25 -1
  237. package/dist/types.d.ts +1 -1
  238. package/dist/types.js +1 -1
  239. package/dist/utility-runtime.d.ts +1 -1
  240. package/dist/verified-recall.js +6 -5
  241. package/package.json +2 -2
  242. package/src/access-http.ts +162 -0
  243. package/src/access-service.ts +215 -0
  244. package/src/admin/admin-surfaces.test.ts +458 -0
  245. package/src/admin/admin-surfaces.ts +819 -0
  246. package/src/event-order-recall.test.ts +53 -0
  247. package/src/event-order-recall.ts +61 -37
  248. package/src/explicit-cue-recall.ts +21 -1
  249. package/src/extraction.ts +91 -6
  250. package/src/orchestrator.ts +46 -5
  251. package/src/provenance-extraction.test.ts +838 -0
  252. package/src/provenance.ts +413 -0
  253. package/src/recall-pipeline-parity.test.ts +288 -0
  254. package/src/recall-pipeline-stages.test.ts +43 -0
  255. package/src/recall-pipeline-stages.ts +22 -0
  256. package/src/response-guidance-recall.ts +32 -36
  257. package/src/schemas.ts +7 -0
  258. package/src/storage.ts +23 -0
  259. package/src/targeted-fact-recall.ts +25 -32
  260. package/src/types.ts +24 -0
  261. package/dist/chunk-4HIAWLA2.js.map +0 -1
  262. package/dist/chunk-EC2AYKRX.js.map +0 -1
  263. package/dist/chunk-FN2SM5SN.js.map +0 -1
  264. package/dist/chunk-FUCJAZ25.js.map +0 -1
  265. package/dist/chunk-HRUULBBV.js.map +0 -1
  266. package/dist/chunk-R5DB26G6.js.map +0 -1
  267. package/dist/chunk-ROZJACKP.js.map +0 -1
  268. package/dist/chunk-SFMRLXIV.js.map +0 -1
  269. package/dist/chunk-UU6MVCJ6.js.map +0 -1
  270. package/dist/chunk-YXLT4EMM.js.map +0 -1
  271. package/dist/chunk-ZPQVJEVQ.js.map +0 -1
  272. /package/dist/{chunk-ODTWHSY2.js.map → chunk-64XXOQRW.js.map} +0 -0
  273. /package/dist/{chunk-ZT7B64BE.js.map → chunk-6TAETM63.js.map} +0 -0
  274. /package/dist/{chunk-HF4N43Q7.js.map → chunk-6VVP6NK7.js.map} +0 -0
  275. /package/dist/{chunk-7TAQEPLE.js.map → chunk-AMNLZ6SF.js.map} +0 -0
  276. /package/dist/{chunk-KS7WQ4BZ.js.map → chunk-CZJ6QSKG.js.map} +0 -0
  277. /package/dist/{chunk-27LQPUMZ.js.map → chunk-EOOVJK2U.js.map} +0 -0
  278. /package/dist/{chunk-QP37KL5H.js.map → chunk-EYJD6KIO.js.map} +0 -0
  279. /package/dist/{chunk-WFEZUGU5.js.map → chunk-EZR35XHX.js.map} +0 -0
  280. /package/dist/{chunk-JO3E5VGS.js.map → chunk-FVI5B7DE.js.map} +0 -0
  281. /package/dist/{chunk-PONNZ54D.js.map → chunk-GY3SKOS4.js.map} +0 -0
  282. /package/dist/{chunk-U33LWTQQ.js.map → chunk-HV57RHMD.js.map} +0 -0
  283. /package/dist/{chunk-IYOPIG3E.js.map → chunk-IX72AAMZ.js.map} +0 -0
  284. /package/dist/{chunk-JKOKX3PS.js.map → chunk-JMA4RYRN.js.map} +0 -0
  285. /package/dist/{chunk-DEDQXIDL.js.map → chunk-JTKFZMZ7.js.map} +0 -0
  286. /package/dist/{chunk-OLOYQZFB.js.map → chunk-KC6TCAWV.js.map} +0 -0
  287. /package/dist/{chunk-ED35D32I.js.map → chunk-LQ4J7ELC.js.map} +0 -0
  288. /package/dist/{chunk-TFVVONWD.js.map → chunk-MBUM2Y3L.js.map} +0 -0
  289. /package/dist/{chunk-T5QAZIBO.js.map → chunk-MOXFPLD6.js.map} +0 -0
  290. /package/dist/{chunk-XTIRCSIH.js.map → chunk-MXEWQKM7.js.map} +0 -0
  291. /package/dist/{chunk-ZYNMX6IU.js.map → chunk-NUIJEGVD.js.map} +0 -0
  292. /package/dist/{chunk-YO4MBK3I.js.map → chunk-QGJAGC2J.js.map} +0 -0
  293. /package/dist/{chunk-ISLJ5WIM.js.map → chunk-QIMFOCSH.js.map} +0 -0
  294. /package/dist/{chunk-IJEZMWKA.js.map → chunk-SFOAQQDJ.js.map} +0 -0
  295. /package/dist/{chunk-NINRTFSV.js.map → chunk-SINGJCUR.js.map} +0 -0
  296. /package/dist/{chunk-6W2D6FGG.js.map → chunk-TGAHHCB6.js.map} +0 -0
  297. /package/dist/{chunk-D75JXBV4.js.map → chunk-WN4GHSDH.js.map} +0 -0
  298. /package/dist/{chunk-K4DWSPMW.js.map → chunk-WRGPE6AW.js.map} +0 -0
  299. /package/dist/{chunk-SDPDU2PM.js.map → chunk-YTMDF6S7.js.map} +0 -0
  300. /package/dist/{chunk-MNU5G4TK.js.map → chunk-Z7XEIAV4.js.map} +0 -0
  301. /package/dist/{chunk-HXHKLVAS.js.map → chunk-ZCEI242W.js.map} +0 -0
package/src/provenance.ts CHANGED
@@ -27,7 +27,9 @@ import { z } from "zod";
27
27
 
28
28
  import { coerceBool, coerceNumber } from "./connectors/coerce.js";
29
29
  import { readEnvVar } from "./runtime/env.js";
30
+ import { collapseWhitespace } from "./whitespace.js";
30
31
  import type { MemoryFrontmatter, ProvenanceConfig, ProvenanceSource } from "./types.js";
32
+ import { isSafeMemoryContent } from "./sanitize.js";
31
33
 
32
34
  /**
33
35
  * Canonical key order for a serialized `ProvenanceSource` (issue #1575).
@@ -171,6 +173,71 @@ function isStrictIsoTimestamp(s: string): boolean {
171
173
  );
172
174
  }
173
175
 
176
+ /**
177
+ * Coerce a turn timestamp to a strict ISO-8601 string (cursor thread Ocveu —
178
+ * "Non-ISO turn timestamps drop sources"). The write-path `ProvenanceSourceSchema`
179
+ * rejects anything `isStrictIsoTimestamp` fails, so a turn whose `timestamp`
180
+ * parses via `new Date(...)` but isn't already strict ISO (e.g.
181
+ * `"2026/01/15 12:00:00"` or `"Jan 15 2026"`) would be silently dropped at
182
+ * serialization — clearing the whole `sources` array and downgrading the tag
183
+ * to `"none"`. Normalizing at extraction time preserves the source.
184
+ *
185
+ * Returns the strict ISO string when the timestamp already passes or
186
+ * `Date.parse` can round-trip it; `undefined` for empty / unparseable input
187
+ * so callers can decide whether to skip the source or fall back.
188
+ */
189
+ function toStrictIsoTimestamp(ts: string | undefined | null): string | undefined {
190
+ if (typeof ts !== "string" || ts.length === 0) return undefined;
191
+ if (isStrictIsoTimestamp(ts)) return ts;
192
+ const parsed = Date.parse(ts);
193
+ if (Number.isNaN(parsed)) return undefined;
194
+ // Reject bare-year / numeric-only strings that Date.parse accepts (e.g.
195
+ // "123") — isStrictIsoTimestamp already rejected them, and the round-trip
196
+ // below would otherwise resurrect them. Require at least one date/time
197
+ // separator so a plain number never round-trips into a fake epoch.
198
+ if (!/[-:T]/.test(ts)) return undefined;
199
+ // Reject calendar-overflow dates (chatgpt-codex-connector thread dANc):
200
+ // Date.parse silently rolls over invalid components — "2026-02-30" becomes
201
+ // 2026-03-02 — fabricating the observation date. For strings that carry an
202
+ // explicit numeric Y/M/D prefix (the common import/provider shape), validate
203
+ // the calendar components are in range before accepting the shifted result.
204
+ // isStrictIsoTimestamp already does this for full ISO strings; this closes
205
+ // the gap for the non-ISO normalization path.
206
+ const ymd = /^(\d{4})\D(\d{1,2})\D(\d{1,2})/.exec(ts);
207
+ if (ymd) {
208
+ const [_, ys, ms, ds] = ymd;
209
+ const y = Number(ys), mo = Number(ms), da = Number(ds);
210
+ if (!isValidCalendarDate(y, mo, da)) return undefined;
211
+ }
212
+ // Also validate M/D/Y or D/M/Y formats (provider/import common shapes:
213
+ // 02/30/2026, 30-02-2026, etc.). Date.parse silently shifts overflow in
214
+ // these too. Try both month-first and day-first interpretations; if
215
+ // neither is a valid calendar date, reject (chatgpt-codex-connector thread
216
+ // dH47 — non-YMD overflow timestamps).
217
+ const mdy = /^(\d{1,2})\D(\d{1,2})\D(\d{4})/.exec(ts);
218
+ if (mdy) {
219
+ const a = Number(mdy[1]), b = Number(mdy[2]), yr = Number(mdy[3]);
220
+ if (!isValidCalendarDate(yr, a, b) && !isValidCalendarDate(yr, b, a)) {
221
+ return undefined;
222
+ }
223
+ }
224
+ const iso = new Date(parsed).toISOString();
225
+ return isStrictIsoTimestamp(iso) ? iso : undefined;
226
+ }
227
+
228
+ /**
229
+ * Validate numeric calendar components without Date overflow (review thread
230
+ * dANc). Date.UTC silently normalizes Feb 30 -> Mar 2; a component round-trip
231
+ * catches what Date.parse accepts. Mirrors the same technique isStrictIsoTimestamp
232
+ * uses for full ISO strings, applied here to the non-ISO normalization path.
233
+ */
234
+ function isValidCalendarDate(y: number, mo: number, da: number): boolean {
235
+ if (mo < 1 || mo > 12 || da < 1) return false;
236
+ const leap = (y % 4 === 0 && (y % 100 !== 0 || y % 400 === 0)) ? 29 : 28;
237
+ const daysInMonth = [31, leap, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31];
238
+ return da <= daysInMonth[mo - 1]!;
239
+ }
240
+
174
241
  /**
175
242
  * Zod schema for a single `ProvenanceSource` entry (issue #1575). Parsed
176
243
  * JSON from frontmatter is external data, so each entry is validated here
@@ -303,3 +370,349 @@ export function parseProvenanceConfig(raw: unknown): ProvenanceConfig {
303
370
  })(),
304
371
  };
305
372
  }
373
+
374
+ // ---------------------------------------------------------------------------
375
+ // Issue #1575 PR 2 — extraction-side post-parse validator.
376
+ // ---------------------------------------------------------------------------
377
+ //
378
+ // This runs once at write time (inside `ExtractionEngine.extract`, after the
379
+ // LLM output is parsed and sanitized, before the result is returned for
380
+ // persistence). Its job: locate each fact's LLM-provided `quote` in the
381
+ // buffered turn texts and build a `ProvenanceSource[]` with verified
382
+ // offsets. Never throws, never drops a fact (rule 34 spirit — an
383
+ // unverifiable span is a tagged state, not a silent failure).
384
+ //
385
+ // Reuses `collapseWhitespace` from `whitespace.ts` for normalization so there
386
+ // is exactly one normalizer in the codebase (issue #1575 pitfall: "do not
387
+ // write a second normalizer").
388
+ // ---------------------------------------------------------------------------
389
+
390
+ /**
391
+ * A turn in the buffered conversation, reduced to the fields the provenance
392
+ * validator needs. `ExtractionEngine.extract` maps its `BufferTurn[]` to this
393
+ * shape so the validator is pure and testable without the full BufferTurn type.
394
+ */
395
+ export interface ProvenanceTurnInput {
396
+ content: string;
397
+ sessionKey?: string;
398
+ logicalSessionKey?: string;
399
+ timestamp: string;
400
+ turnId?: string;
401
+ }
402
+
403
+ /**
404
+ * Result of building provenance for a single extracted fact.
405
+ */
406
+ export interface ProvenanceBuildResult {
407
+ /** Verified sources (one per matching turn). Absent when no quote survives. */
408
+ sources?: ProvenanceSource[];
409
+ /** Coarse strength tag persisted to frontmatter. */
410
+ provenance: "verified" | "unverified" | "none";
411
+ /**
412
+ * Transient signal (never persisted): `true` when `config.requireSpans`
413
+ * is enabled and the LLM-provided quote could not be located in ANY source
414
+ * turn (the strict case `ProvenanceConfig.requireSpans` documents: "facts
415
+ * whose quote cannot be located are routed to `pending_review`"). The
416
+ * extraction consumer carries this onto the in-memory `ExtractedFact` so
417
+ * the persist path can route the fact to the review queue. A quote that
418
+ * WAS located but whose source was dropped (e.g. un-coercible timestamp)
419
+ * does NOT set this flag — the span was found, so requireSpans is
420
+ * satisfied (chatgpt-codex-connector thread 4xB).
421
+ */
422
+ requireSpansPending?: boolean;
423
+ }
424
+
425
+ /**
426
+ * Cap a quote string at `maxChars`, truncating at the last word boundary that
427
+ * fits and appending an ellipsis marker. Quotes at or under the cap pass
428
+ * through unchanged. Operates on Unicode code points (not UTF-16 units) so
429
+ * emoji and astral-plane characters are not split mid-glyph.
430
+ */
431
+ function capQuote(quote: string, maxChars: number): string {
432
+ const glyphs = Array.from(quote);
433
+ if (glyphs.length <= maxChars) return quote;
434
+ // Walk backward from the cap to find a word boundary (space). If the entire
435
+ // span is one long token (no spaces), cut at the cap — a hard cut is better
436
+ // than no quote at all.
437
+ let cut = maxChars;
438
+ while (cut > 0 && !/\s/.test(glyphs[cut - 1]!)) cut--;
439
+ if (cut === 0) cut = maxChars; // no word boundary found
440
+ return glyphs.slice(0, cut).join("").trimEnd() + "\u2026";
441
+ }
442
+
443
+ /**
444
+ * Casefold a string for normalized matching. Uses `toLowerCase` (not
445
+ * `toLocaleLowerCase`) for determinism across runtime locales — the existing
446
+ * `normalizeFactKey` helper in extraction.ts follows the same convention.
447
+ */
448
+ function casefold(s: string): string {
449
+ return s.toLowerCase();
450
+ }
451
+
452
+ /**
453
+ * Locate `quote` within `text`, returning whether a match was found and, when
454
+ * recoverable, the half-open `[start, end)` offsets. Tries exact substring
455
+ * first, then whitespace/case-normalized match.
456
+ *
457
+ * For the normalized path, the mapping from the normalized string back to
458
+ * original offsets is recovered by walking both strings in lockstep: we know
459
+ * `collapseWhitespace(text)` is a subsequence of `text` with whitespace runs
460
+ * collapsed, so we can track the original offset as we scan.
461
+ *
462
+ * Returns `{ matched: false }` when neither match succeeds. Returns
463
+ * `{ matched: true, offsets }` when offsets are recoverable. Returns
464
+ * `{ matched: true }` (offsets omitted) when the normalized substring was
465
+ * found but original offsets could not be recovered — callers still treat
466
+ * this as a verified match and record a source without offsets (cursor
467
+ * thread Ocver — "Normalized match drops verified provenance").
468
+ */
469
+ type LocateQuoteResult =
470
+ | { matched: false }
471
+ | { matched: true; offsets?: { charStart: number; charEnd: number } };
472
+
473
+ function locateQuoteOffsets(quote: string, text: string): LocateQuoteResult {
474
+ // 1. Exact substring match (handles unicode, curly quotes, emoji verbatim).
475
+ const exactIdx = text.indexOf(quote);
476
+ if (exactIdx >= 0) {
477
+ return { matched: true, offsets: { charStart: exactIdx, charEnd: exactIdx + quote.length } };
478
+ }
479
+
480
+ // 2. Whitespace/case-normalized match. Collapse runs of whitespace and
481
+ // casefold both sides, then find the normalized quote in the normalized
482
+ // text. Recover original offsets by scanning forward from the normalized
483
+ // match start through the original text, accumulating non-whitespace
484
+ // glyphs until we've consumed the normalized quote length.
485
+ const normQuote = collapseWhitespace(casefold(quote));
486
+ if (normQuote.length === 0) return { matched: false };
487
+ const normText = collapseWhitespace(casefold(text));
488
+ const normIdx = normText.indexOf(normQuote);
489
+ if (normIdx < 0) return { matched: false };
490
+
491
+ // Recover original offsets: walk the original text, skipping leading
492
+ // whitespace to align with the collapsed form, then track how many
493
+ // normalized chars we've consumed.
494
+ const normQuoteLen = normQuote.length;
495
+ let origIdx = 0;
496
+ // Skip leading whitespace in original text (collapseWhitespace trims both ends).
497
+ while (origIdx < text.length && /\s/.test(text[origIdx]!)) origIdx++;
498
+ // Walk through original text, counting normalized chars consumed.
499
+ // Each non-whitespace char in original = 1 normalized char.
500
+ // Whitespace runs in original = 1 space in normalized (but only between non-ws).
501
+ let normPos = 0;
502
+ let origStart = -1;
503
+ let origEnd = -1;
504
+ let prevWasWs = true; // suppress the leading space in collapsed form
505
+ while (origIdx < text.length && origEnd < 0) {
506
+ const ch = text[origIdx]!;
507
+ if (/\s/.test(ch)) {
508
+ if (!prevWasWs) {
509
+ // This whitespace run collapses to a single space in normalized text.
510
+ if (normPos === normIdx) origStart = origIdx; // normalized space aligns
511
+ normPos++;
512
+ prevWasWs = true;
513
+ }
514
+ origIdx++;
515
+ continue;
516
+ }
517
+ // Non-whitespace char.
518
+ if (normPos === normIdx && origStart < 0) {
519
+ origStart = origIdx;
520
+ }
521
+ normPos++;
522
+ prevWasWs = false;
523
+ origIdx++;
524
+ if (normPos >= normIdx + normQuoteLen) {
525
+ origEnd = origIdx;
526
+ }
527
+ }
528
+ if (origStart >= 0 && origEnd >= 0) {
529
+ return { matched: true, offsets: { charStart: origStart, charEnd: origEnd } };
530
+ }
531
+ // Normalized substring was found, but original offsets are not recoverable
532
+ // (edge case in the normalized mapping). The match still counts as verified
533
+ // — record a source without offsets rather than dropping the turn entirely
534
+ // (cursor thread Ocver). charStart/charEnd are best-effort debugging aids,
535
+ // not a precondition for a verified source.
536
+ return { matched: true };
537
+ }
538
+
539
+ /**
540
+ * Build provenance sources for a single extracted fact by locating its
541
+ * LLM-provided `quote` in the buffered turns (issue #1575 PR 2).
542
+ *
543
+ * Matching strategy (per issue design):
544
+ * 1. Exact substring match in a turn → `provenance: "verified"` with
545
+ * `charStart`/`charEnd` offsets.
546
+ * 2. Whitespace/case-normalized match → `"verified"`, offsets recovered
547
+ * when possible, else omitted.
548
+ * 3. No match in any turn → `"unverified"` — the quote survives as a
549
+ * source (the LLM vouched for it) but without located offsets.
550
+ * 4. No quote provided by the LLM → `"none"` — no evidence to record.
551
+ *
552
+ * A quote appearing in multiple turns produces multiple sources (one per
553
+ * matching turn) so a repeated utterance backs the fact from each occurrence.
554
+ *
555
+ * The quote is capped at `config.maxQuoteChars` (truncated at a word boundary
556
+ * with an ellipsis marker) BEFORE storage. Locating uses the original
557
+ * (untruncated) quote for maximum match fidelity; the capped excerpt is what
558
+ * gets persisted.
559
+ *
560
+ * When `config.enabled === false`, returns `{ provenance: "none" }`
561
+ * immediately — byte-identical to pre-feature extraction (rule 39).
562
+ *
563
+ * Never throws. An unexpected error degrades to `{ provenance: "none" }` so
564
+ * extraction never crashes on a provenance hiccup (rule 13/18).
565
+ */
566
+ /**
567
+ * Strip a leading extraction-prompt role label from a quote (cursor thread Oc3Z2
568
+ * — "Quote prompt mismatches validator"). The extraction prompt renders each
569
+ * buffered turn as `[role] content` (or `[context role] content`), so a
570
+ * faithful LLM may include that prefix in its verbatim quote. buildFactProvenance
571
+ * searches the raw `turn.content` (no prefix), so a quote carrying the label
572
+ * would never match. Stripping the leading label lets the actual utterance
573
+ * verify against the turn text. Only a SINGLE leading label is stripped — a
574
+ * multi-turn quote (with embedded labels) is left untouched and will simply
575
+ * fail to match a single turn, as before.
576
+ *
577
+ * The regex is constrained to the labels the prompt actually emits — `user`,
578
+ * `assistant`, optionally prefixed with `context ` (extraction.ts renders
579
+ * `[user]`, `[assistant]`, `[context user]`, `[context assistant]`). A
580
+ * real utterance that happens to start with other bracketed text (e.g.
581
+ * `[do not] deploy before approval`, `[P1] fix the cache`) is preserved
582
+ * verbatim so the quote can match the turn and the persisted span retains its
583
+ * original meaning (chatgpt-codex-connector thread 4xA — limiting role-prefix
584
+ * stripping to actual prompt labels).
585
+ */
586
+ function stripLeadingRolePrefix(quote: string): string {
587
+ return quote.replace(/^\s*\[(?:context\s+)?(?:user|assistant)\]\s+/i, "");
588
+ }
589
+
590
+ export function buildFactProvenance(
591
+ factQuote: string | null | undefined,
592
+ turns: ReadonlyArray<ProvenanceTurnInput>,
593
+ config: ProvenanceConfig,
594
+ ): ProvenanceBuildResult {
595
+ if (!config.enabled) return { provenance: "none" };
596
+ const rawQuote = typeof factQuote === "string" ? factQuote.trim() : "";
597
+ // requireSpans (chatgpt-codex-connector thread dEsu + cursor thread dGKJ):
598
+ // every early exit that drops a fact's span (no quote, empty after label
599
+ // strip, or unsafe quote) must flag requireSpansPending when the operator
600
+ // opted into requireSpans, so the persist path routes the fact to
601
+ // pending_review instead of active. Only the disabled-feature exit (above)
602
+ // and the unexpected-error catch (below) omit the flag — those are
603
+ // policy-neutral degradations, not a missing-span decision.
604
+ const noneResult = (): ProvenanceBuildResult =>
605
+ config.requireSpans === true
606
+ ? { provenance: "none", requireSpansPending: true }
607
+ : { provenance: "none" };
608
+ if (rawQuote.length === 0) return noneResult();
609
+ // Strip a leading prompt role label so a faithful quote verifies against
610
+ // the raw turn content (cursor thread Oc3Z2). Prefer the RAW quote when it
611
+ // matches at least one turn — an utterance that literally begins with
612
+ // [user]/[assistant] must verify as-is, not be truncated to the post-label
613
+ // text (cursor/codex thread dEsw). The strip handles the common case where
614
+ // the LLM includes the prompt label; this preserves the rare case where the
615
+ // utterance itself starts with that text.
616
+ const strippedQuote = stripLeadingRolePrefix(rawQuote);
617
+ const useRawQuote =
618
+ rawQuote === strippedQuote ||
619
+ turns.some(
620
+ (t) =>
621
+ typeof t?.content === "string" &&
622
+ t.content.length > 0 &&
623
+ locateQuoteOffsets(rawQuote, t.content).matched,
624
+ );
625
+ const quote = useRawQuote ? rawQuote : strippedQuote;
626
+ if (quote.length === 0) return noneResult();
627
+ // Sanitize the quote before persisting it as a provenance span
628
+ // (chatgpt-codex-connector thread dANZ): the fact body is sanitized via
629
+ // sanitizeMemoryContent, but sources[].quote was persisted verbatim,
630
+ // reintroducing unsafe memory text (e.g. "ignore previous instructions")
631
+ // through the provenance field exposed to memory_get/x-ray/faithfulness.
632
+ // An unsafe quote cannot serve as evidence — drop the source entirely
633
+ // (consistent with how the body is sanitized: unsafe text is redacted,
634
+ // and a redacted quote is useless as a verbatim span).
635
+ if (!isSafeMemoryContent(quote)) return noneResult();
636
+
637
+ try {
638
+ // Search every turn for the quote. Collect verified sources.
639
+ const sources: ProvenanceSource[] = [];
640
+ // Track the first turn where the quote was located even when its
641
+ // timestamp can't be coerced (cursor thread 4Pj — "Bad timestamp drops
642
+ // matched source"). Without this, a located-but-unverifiable quote falls
643
+ // through to the unverified branch and is attributed to the *last* turn's
644
+ // session, mislabeling the source's origin session. Also drives the
645
+ // requireSpans signal: only a quote that was NOT located in any turn
646
+ // qualifies for pending_review routing under requireSpans (thread 4xB).
647
+ let locatedTurn: ProvenanceTurnInput | undefined;
648
+ for (const turn of turns) {
649
+ if (!turn || typeof turn.content !== "string" || turn.content.length === 0) continue;
650
+ const located = locateQuoteOffsets(quote, turn.content);
651
+ // cursor thread Ocver: a normalized match counts as verified even when
652
+ // original offsets are unrecoverable — record the source without
653
+ // charStart/charEnd instead of skipping the turn.
654
+ if (!located.matched) continue;
655
+ if (!locatedTurn) locatedTurn = turn;
656
+ // cursor thread Ocveu: normalize the turn timestamp to strict ISO so
657
+ // the write-path ProvenanceSourceSchema keeps the source. Skip the turn
658
+ // when the timestamp can't be coerced — pushing it would guarantee a
659
+ // serialization drop (and the tag downgrade) for this source.
660
+ const observedAt = toStrictIsoTimestamp(turn.timestamp);
661
+ if (!observedAt) continue;
662
+ sources.push({
663
+ sessionKey: turn.sessionKey ?? turn.logicalSessionKey ?? "unknown",
664
+ ...(turn.turnId ? { turnId: turn.turnId } : {}),
665
+ observedAt,
666
+ quote: capQuote(quote, config.maxQuoteChars),
667
+ ...(located.offsets
668
+ ? { charStart: located.offsets.charStart, charEnd: located.offsets.charEnd }
669
+ : {}),
670
+ });
671
+ }
672
+
673
+ if (sources.length > 0) {
674
+ return { sources, provenance: "verified" };
675
+ }
676
+
677
+ // Quote provided but could not be turned into a verified source.
678
+ // Determine the session to attribute: prefer the turn where the quote was
679
+ // LOCATED even if its timestamp couldn't be coerced (cursor thread 4Pj — a
680
+ // located quote must not be tied to the *last* turn's session). If the
681
+ // quote was not located in any turn at all, fall back to the last turn's
682
+ // session (the documented unverified behavior: the LLM vouched for the
683
+ // excerpt but we couldn't pin it to a character offset). Normalize the
684
+ // timestamp (thread Ocveu); fall back to epoch when no turn supplies a
685
+ // coercible timestamp so the unverified source still survives the
686
+ // write-path schema.
687
+ const fallbackTurn = locatedTurn ?? turns[turns.length - 1];
688
+ const fallbackSessionKey = fallbackTurn
689
+ ? (fallbackTurn.sessionKey ?? fallbackTurn.logicalSessionKey ?? "unknown")
690
+ : "unknown";
691
+ const fallbackObservedAt = fallbackTurn
692
+ ? (toStrictIsoTimestamp(fallbackTurn.timestamp) ?? new Date(0).toISOString())
693
+ : new Date(0).toISOString();
694
+ // requireSpans signal (chatgpt-codex-connector thread 4xB): when an
695
+ // operator opts into provenance.requireSpans, a fact whose quote could
696
+ // not be located in ANY turn (locatedTurn undefined) is flagged so the
697
+ // persist path routes it to pending_review instead of active. A quote
698
+ // that WAS located (locatedTurn set) satisfies requireSpans even when
699
+ // its source was dropped for an un-coercible timestamp — the span was
700
+ // found, so the fact has the grounding requireSpans demands.
701
+ const requireSpansPending =
702
+ config.requireSpans === true && locatedTurn === undefined;
703
+ return {
704
+ sources: [
705
+ {
706
+ sessionKey: fallbackSessionKey,
707
+ observedAt: fallbackObservedAt,
708
+ quote: capQuote(quote, config.maxQuoteChars),
709
+ },
710
+ ],
711
+ provenance: "unverified",
712
+ ...(requireSpansPending ? { requireSpansPending: true } : {}),
713
+ };
714
+ } catch {
715
+ // Never crash extraction on a provenance error (rule 13/18).
716
+ return { provenance: "none" };
717
+ }
718
+ }