@monoes/monomindcli 2.14.0 → 2.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (291) hide show
  1. package/.claude/helpers/control-start.cjs +10 -13
  2. package/.claude/skills/monodesign/scripts/context.mjs +5 -5
  3. package/.claude/skills/monodesign/scripts/critique-storage.mjs +1 -1
  4. package/.claude/skills/monodesign/scripts/detect-csp.mjs +1 -6
  5. package/.claude/skills/monodesign/scripts/detector/browser/injected/index.mjs +16 -18
  6. package/.claude/skills/monodesign/scripts/detector/cli/main.mjs +3 -3
  7. package/.claude/skills/monodesign/scripts/detector/detect-antipatterns-browser.js +69 -49
  8. package/.claude/skills/monodesign/scripts/detector/engines/regex/detect-text.mjs +6 -6
  9. package/.claude/skills/monodesign/scripts/detector/engines/static-html/css-cascade.mjs +6 -5
  10. package/.claude/skills/monodesign/scripts/detector/engines/static-html/detect-html.mjs +6 -6
  11. package/.claude/skills/monodesign/scripts/detector/fix/index.mjs +3 -3
  12. package/.claude/skills/monodesign/scripts/detector/node/file-system.mjs +1 -1
  13. package/.claude/skills/monodesign/scripts/detector/registry/antipatterns.mjs +1 -1
  14. package/.claude/skills/monodesign/scripts/detector/rules/checks.mjs +26 -27
  15. package/.claude/skills/monodesign/scripts/detector/shared/color.mjs +2 -2
  16. package/.claude/skills/monodesign/scripts/detector/shared/constants.mjs +1 -1
  17. package/.claude/skills/monodesign/scripts/detector/shared/inline-ignores.mjs +2 -2
  18. package/.claude/skills/monodesign/scripts/hook-admin.mjs +6 -6
  19. package/.claude/skills/monodesign/scripts/hook-before-edit.mjs +4 -4
  20. package/.claude/skills/monodesign/scripts/hook-lib.mjs +20 -11
  21. package/.claude/skills/monodesign/scripts/hook.mjs +1 -1
  22. package/.claude/skills/monodesign/scripts/lib/design-parser.mjs +9 -42
  23. package/.claude/skills/monodesign/scripts/lib/is-generated.mjs +1 -1
  24. package/.claude/skills/monodesign/scripts/lib/monodesign-config.mjs +5 -5
  25. package/.claude/skills/monodesign/scripts/lib/monodesign-paths.mjs +1 -1
  26. package/.claude/skills/monodesign/scripts/live/event-validation.mjs +13 -13
  27. package/.claude/skills/monodesign/scripts/live/manual-apply.mjs +4 -4
  28. package/.claude/skills/monodesign/scripts/live/manual-edits-buffer.mjs +1 -1
  29. package/.claude/skills/monodesign/scripts/live/session-store.mjs +5 -5
  30. package/.claude/skills/monodesign/scripts/live/svelte-component.mjs +13 -13
  31. package/.claude/skills/monodesign/scripts/live/sveltekit-adapter.mjs +4 -4
  32. package/.claude/skills/monodesign/scripts/live/ui-core.mjs +1 -1
  33. package/.claude/skills/monodesign/scripts/live-accept.mjs +24 -24
  34. package/.claude/skills/monodesign/scripts/live-browser-dom.js +8 -7
  35. package/.claude/skills/monodesign/scripts/live-browser-session.js +8 -7
  36. package/.claude/skills/monodesign/scripts/live-browser.js +406 -433
  37. package/.claude/skills/monodesign/scripts/live-commit-manual-edits.mjs +6 -6
  38. package/.claude/skills/monodesign/scripts/live-copy-edit-agent.mjs +2 -2
  39. package/.claude/skills/monodesign/scripts/live-discard-manual-edits.mjs +1 -1
  40. package/.claude/skills/monodesign/scripts/live-inject.mjs +7 -7
  41. package/.claude/skills/monodesign/scripts/live-insert.mjs +11 -11
  42. package/.claude/skills/monodesign/scripts/live-manual-edit-evidence.mjs +4 -4
  43. package/.claude/skills/monodesign/scripts/live-poll.mjs +5 -5
  44. package/.claude/skills/monodesign/scripts/live-server.mjs +10 -10
  45. package/.claude/skills/monodesign/scripts/live-wrap.mjs +31 -41
  46. package/.claude/skills/monodesign/scripts/live.mjs +5 -5
  47. package/.claude/skills/monodesign/scripts/palette.mjs +3 -1
  48. package/.claude/skills/monodesign/scripts/pin.mjs +1 -1
  49. package/bin/cli.js +21 -1
  50. package/bin/mcp-server.js +21 -1
  51. package/dist/src/capabilities/cap-documents.d.ts.map +1 -1
  52. package/dist/src/capabilities/cap-documents.js +23 -0
  53. package/dist/src/capabilities/cap-documents.js.map +1 -1
  54. package/dist/src/commands/doc-filters.d.ts +19 -0
  55. package/dist/src/commands/doc-filters.d.ts.map +1 -0
  56. package/dist/src/commands/doc-filters.js +64 -0
  57. package/dist/src/commands/doc-filters.js.map +1 -0
  58. package/dist/src/commands/doc-library.d.ts +22 -0
  59. package/dist/src/commands/doc-library.d.ts.map +1 -0
  60. package/dist/src/commands/doc-library.js +454 -0
  61. package/dist/src/commands/doc-library.js.map +1 -0
  62. package/dist/src/commands/doc-list.d.ts +9 -0
  63. package/dist/src/commands/doc-list.d.ts.map +1 -0
  64. package/dist/src/commands/doc-list.js +99 -0
  65. package/dist/src/commands/doc-list.js.map +1 -0
  66. package/dist/src/commands/doc.d.ts.map +1 -1
  67. package/dist/src/commands/doc.js +51 -37
  68. package/dist/src/commands/doc.js.map +1 -1
  69. package/dist/src/commands/ui.d.ts.map +1 -1
  70. package/dist/src/commands/ui.js +60 -0
  71. package/dist/src/commands/ui.js.map +1 -1
  72. package/dist/src/knowledge/capture-envelope.d.ts +85 -0
  73. package/dist/src/knowledge/capture-envelope.d.ts.map +1 -0
  74. package/dist/src/knowledge/capture-envelope.js +167 -0
  75. package/dist/src/knowledge/capture-envelope.js.map +1 -0
  76. package/dist/src/knowledge/capture-text.d.ts +23 -0
  77. package/dist/src/knowledge/capture-text.d.ts.map +1 -0
  78. package/dist/src/knowledge/capture-text.js +49 -0
  79. package/dist/src/knowledge/capture-text.js.map +1 -0
  80. package/dist/src/knowledge/citation.d.ts +121 -0
  81. package/dist/src/knowledge/citation.d.ts.map +1 -0
  82. package/dist/src/knowledge/citation.js +252 -0
  83. package/dist/src/knowledge/citation.js.map +1 -0
  84. package/dist/src/knowledge/document-chunking.d.ts +42 -0
  85. package/dist/src/knowledge/document-chunking.d.ts.map +1 -0
  86. package/dist/src/knowledge/document-chunking.js +288 -0
  87. package/dist/src/knowledge/document-chunking.js.map +1 -0
  88. package/dist/src/knowledge/document-index.d.ts +124 -0
  89. package/dist/src/knowledge/document-index.d.ts.map +1 -0
  90. package/dist/src/knowledge/document-index.js +314 -0
  91. package/dist/src/knowledge/document-index.js.map +1 -0
  92. package/dist/src/knowledge/document-ingest.d.ts +23 -0
  93. package/dist/src/knowledge/document-ingest.d.ts.map +1 -0
  94. package/dist/src/knowledge/document-ingest.js +368 -0
  95. package/dist/src/knowledge/document-ingest.js.map +1 -0
  96. package/dist/src/knowledge/document-pipeline.d.ts +26 -148
  97. package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
  98. package/dist/src/knowledge/document-pipeline.js +25 -920
  99. package/dist/src/knowledge/document-pipeline.js.map +1 -1
  100. package/dist/src/knowledge/document-search.d.ts +27 -0
  101. package/dist/src/knowledge/document-search.d.ts.map +1 -0
  102. package/dist/src/knowledge/document-search.js +138 -0
  103. package/dist/src/knowledge/document-search.js.map +1 -0
  104. package/dist/src/knowledge/document-store.d.ts +28 -0
  105. package/dist/src/knowledge/document-store.d.ts.map +1 -0
  106. package/dist/src/knowledge/document-store.js +61 -0
  107. package/dist/src/knowledge/document-store.js.map +1 -0
  108. package/dist/src/knowledge/document-types.d.ts +103 -0
  109. package/dist/src/knowledge/document-types.d.ts.map +1 -0
  110. package/dist/src/knowledge/document-types.js +12 -0
  111. package/dist/src/knowledge/document-types.js.map +1 -0
  112. package/dist/src/knowledge/highlights.d.ts +168 -0
  113. package/dist/src/knowledge/highlights.d.ts.map +1 -0
  114. package/dist/src/knowledge/highlights.js +311 -0
  115. package/dist/src/knowledge/highlights.js.map +1 -0
  116. package/dist/src/knowledge/html-extract.d.ts +43 -0
  117. package/dist/src/knowledge/html-extract.d.ts.map +1 -0
  118. package/dist/src/knowledge/html-extract.js +286 -0
  119. package/dist/src/knowledge/html-extract.js.map +1 -0
  120. package/dist/src/knowledge/html-tags.d.ts +30 -0
  121. package/dist/src/knowledge/html-tags.d.ts.map +1 -0
  122. package/dist/src/knowledge/html-tags.js +229 -0
  123. package/dist/src/knowledge/html-tags.js.map +1 -0
  124. package/dist/src/knowledge/library.d.ts +106 -0
  125. package/dist/src/knowledge/library.d.ts.map +1 -0
  126. package/dist/src/knowledge/library.js +195 -0
  127. package/dist/src/knowledge/library.js.map +1 -0
  128. package/dist/src/knowledge/lookup.d.ts +80 -0
  129. package/dist/src/knowledge/lookup.d.ts.map +1 -0
  130. package/dist/src/knowledge/lookup.js +156 -0
  131. package/dist/src/knowledge/lookup.js.map +1 -0
  132. package/dist/src/knowledge/mhtml.d.ts +60 -0
  133. package/dist/src/knowledge/mhtml.d.ts.map +1 -0
  134. package/dist/src/knowledge/mhtml.js +154 -0
  135. package/dist/src/knowledge/mhtml.js.map +1 -0
  136. package/dist/src/knowledge/okf-bundle.d.ts +19 -0
  137. package/dist/src/knowledge/okf-bundle.d.ts.map +1 -0
  138. package/dist/src/knowledge/okf-bundle.js +106 -0
  139. package/dist/src/knowledge/okf-bundle.js.map +1 -0
  140. package/dist/src/knowledge/profile-store.d.ts +134 -0
  141. package/dist/src/knowledge/profile-store.d.ts.map +1 -0
  142. package/dist/src/knowledge/profile-store.js +237 -0
  143. package/dist/src/knowledge/profile-store.js.map +1 -0
  144. package/dist/src/knowledge/related.d.ts +58 -0
  145. package/dist/src/knowledge/related.d.ts.map +1 -0
  146. package/dist/src/knowledge/related.js +222 -0
  147. package/dist/src/knowledge/related.js.map +1 -0
  148. package/dist/src/knowledge/section-diff.d.ts +48 -0
  149. package/dist/src/knowledge/section-diff.d.ts.map +1 -0
  150. package/dist/src/knowledge/section-diff.js +151 -0
  151. package/dist/src/knowledge/section-diff.js.map +1 -0
  152. package/dist/src/knowledge/watch.d.ts +92 -0
  153. package/dist/src/knowledge/watch.d.ts.map +1 -0
  154. package/dist/src/knowledge/watch.js +221 -0
  155. package/dist/src/knowledge/watch.js.map +1 -0
  156. package/dist/src/mcp-server.d.ts.map +1 -1
  157. package/dist/src/mcp-server.js +44 -87
  158. package/dist/src/mcp-server.js.map +1 -1
  159. package/dist/src/mcp-tools/browser-instrument-tools.d.ts +35 -0
  160. package/dist/src/mcp-tools/browser-instrument-tools.d.ts.map +1 -0
  161. package/dist/src/mcp-tools/browser-instrument-tools.js +359 -0
  162. package/dist/src/mcp-tools/browser-instrument-tools.js.map +1 -0
  163. package/dist/src/mcp-tools/browser-metrics.d.ts +42 -0
  164. package/dist/src/mcp-tools/browser-metrics.d.ts.map +1 -0
  165. package/dist/src/mcp-tools/browser-metrics.js +91 -0
  166. package/dist/src/mcp-tools/browser-metrics.js.map +1 -0
  167. package/dist/src/mcp-tools/browser-profile-tools.d.ts +16 -0
  168. package/dist/src/mcp-tools/browser-profile-tools.d.ts.map +1 -0
  169. package/dist/src/mcp-tools/browser-profile-tools.js +324 -0
  170. package/dist/src/mcp-tools/browser-profile-tools.js.map +1 -0
  171. package/dist/src/mcp-tools/browser-session.d.ts +68 -0
  172. package/dist/src/mcp-tools/browser-session.d.ts.map +1 -0
  173. package/dist/src/mcp-tools/browser-session.js +224 -0
  174. package/dist/src/mcp-tools/browser-session.js.map +1 -0
  175. package/dist/src/mcp-tools/browser-tools.d.ts.map +1 -1
  176. package/dist/src/mcp-tools/browser-tools.js +10 -176
  177. package/dist/src/mcp-tools/browser-tools.js.map +1 -1
  178. package/dist/src/mcp-tools/capture-resource-read.d.ts +115 -0
  179. package/dist/src/mcp-tools/capture-resource-read.d.ts.map +1 -0
  180. package/dist/src/mcp-tools/capture-resource-read.js +296 -0
  181. package/dist/src/mcp-tools/capture-resource-read.js.map +1 -0
  182. package/dist/src/mcp-tools/capture-resource-tools.d.ts +22 -0
  183. package/dist/src/mcp-tools/capture-resource-tools.d.ts.map +1 -0
  184. package/dist/src/mcp-tools/capture-resource-tools.js +182 -0
  185. package/dist/src/mcp-tools/capture-resource-tools.js.map +1 -0
  186. package/dist/src/mcp-tools/capture-resources.d.ts +142 -0
  187. package/dist/src/mcp-tools/capture-resources.d.ts.map +1 -0
  188. package/dist/src/mcp-tools/capture-resources.js +289 -0
  189. package/dist/src/mcp-tools/capture-resources.js.map +1 -0
  190. package/dist/src/mcp-tools/index.d.ts +3 -0
  191. package/dist/src/mcp-tools/index.d.ts.map +1 -1
  192. package/dist/src/mcp-tools/index.js +6 -0
  193. package/dist/src/mcp-tools/index.js.map +1 -1
  194. package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
  195. package/dist/src/mcp-tools/knowledge-tools.js +9 -1
  196. package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
  197. package/dist/src/mcp-tools/resource-router.d.ts +86 -0
  198. package/dist/src/mcp-tools/resource-router.d.ts.map +1 -0
  199. package/dist/src/mcp-tools/resource-router.js +181 -0
  200. package/dist/src/mcp-tools/resource-router.js.map +1 -0
  201. package/dist/src/memory/memory-bridge.js +1 -1
  202. package/dist/src/memory/memory-bridge.js.map +1 -1
  203. package/dist/src/orgrt/agent-runner.d.ts +4 -0
  204. package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
  205. package/dist/src/orgrt/agent-runner.js +22 -0
  206. package/dist/src/orgrt/agent-runner.js.map +1 -1
  207. package/dist/src/orgrt/antigravity-runner.d.ts.map +1 -1
  208. package/dist/src/orgrt/antigravity-runner.js +2 -1
  209. package/dist/src/orgrt/antigravity-runner.js.map +1 -1
  210. package/dist/src/orgrt/authority-mask.d.ts +32 -0
  211. package/dist/src/orgrt/authority-mask.d.ts.map +1 -0
  212. package/dist/src/orgrt/authority-mask.js +135 -0
  213. package/dist/src/orgrt/authority-mask.js.map +1 -0
  214. package/dist/src/orgrt/codex-runner.d.ts.map +1 -1
  215. package/dist/src/orgrt/codex-runner.js +2 -1
  216. package/dist/src/orgrt/codex-runner.js.map +1 -1
  217. package/dist/src/orgrt/copilot-runner.d.ts.map +1 -1
  218. package/dist/src/orgrt/copilot-runner.js +2 -1
  219. package/dist/src/orgrt/copilot-runner.js.map +1 -1
  220. package/dist/src/orgrt/crush-runner.d.ts.map +1 -1
  221. package/dist/src/orgrt/crush-runner.js +6 -1
  222. package/dist/src/orgrt/crush-runner.js.map +1 -1
  223. package/dist/src/orgrt/daemon.d.ts +4 -0
  224. package/dist/src/orgrt/daemon.d.ts.map +1 -1
  225. package/dist/src/orgrt/daemon.js +13 -2
  226. package/dist/src/orgrt/daemon.js.map +1 -1
  227. package/dist/src/orgrt/decisions.d.ts +9 -0
  228. package/dist/src/orgrt/decisions.d.ts.map +1 -1
  229. package/dist/src/orgrt/decisions.js +32 -5
  230. package/dist/src/orgrt/decisions.js.map +1 -1
  231. package/dist/src/orgrt/file-roots.d.ts +2 -1
  232. package/dist/src/orgrt/file-roots.d.ts.map +1 -1
  233. package/dist/src/orgrt/file-roots.js +7 -3
  234. package/dist/src/orgrt/file-roots.js.map +1 -1
  235. package/dist/src/orgrt/grok-runner.d.ts.map +1 -1
  236. package/dist/src/orgrt/grok-runner.js +2 -1
  237. package/dist/src/orgrt/grok-runner.js.map +1 -1
  238. package/dist/src/orgrt/hermes-runner.d.ts.map +1 -1
  239. package/dist/src/orgrt/hermes-runner.js +2 -1
  240. package/dist/src/orgrt/hermes-runner.js.map +1 -1
  241. package/dist/src/orgrt/inbox.d.ts.map +1 -1
  242. package/dist/src/orgrt/inbox.js +70 -4
  243. package/dist/src/orgrt/inbox.js.map +1 -1
  244. package/dist/src/orgrt/kimicode-runner.d.ts.map +1 -1
  245. package/dist/src/orgrt/kimicode-runner.js +2 -1
  246. package/dist/src/orgrt/kimicode-runner.js.map +1 -1
  247. package/dist/src/orgrt/opencode-runner.d.ts.map +1 -1
  248. package/dist/src/orgrt/opencode-runner.js +3 -2
  249. package/dist/src/orgrt/opencode-runner.js.map +1 -1
  250. package/dist/src/orgrt/pi-rpc-runner.d.ts.map +1 -1
  251. package/dist/src/orgrt/pi-rpc-runner.js +3 -2
  252. package/dist/src/orgrt/pi-rpc-runner.js.map +1 -1
  253. package/dist/src/orgrt/pi-runner.d.ts.map +1 -1
  254. package/dist/src/orgrt/pi-runner.js +2 -1
  255. package/dist/src/orgrt/pi-runner.js.map +1 -1
  256. package/dist/src/orgrt/policy.d.ts.map +1 -1
  257. package/dist/src/orgrt/policy.js +3 -0
  258. package/dist/src/orgrt/policy.js.map +1 -1
  259. package/dist/src/orgrt/qwen-rpc-runner.d.ts.map +1 -1
  260. package/dist/src/orgrt/qwen-rpc-runner.js +3 -2
  261. package/dist/src/orgrt/qwen-rpc-runner.js.map +1 -1
  262. package/dist/src/orgrt/qwen-runner.d.ts.map +1 -1
  263. package/dist/src/orgrt/qwen-runner.js +2 -1
  264. package/dist/src/orgrt/qwen-runner.js.map +1 -1
  265. package/dist/src/orgrt/role-sandbox.d.ts +21 -0
  266. package/dist/src/orgrt/role-sandbox.d.ts.map +1 -1
  267. package/dist/src/orgrt/role-sandbox.js +31 -6
  268. package/dist/src/orgrt/role-sandbox.js.map +1 -1
  269. package/dist/src/orgrt/session-ledger.d.ts +7 -0
  270. package/dist/src/orgrt/session-ledger.d.ts.map +1 -1
  271. package/dist/src/orgrt/session-ledger.js +15 -0
  272. package/dist/src/orgrt/session-ledger.js.map +1 -1
  273. package/dist/src/orgrt/session.d.ts.map +1 -1
  274. package/dist/src/orgrt/session.js +42 -10
  275. package/dist/src/orgrt/session.js.map +1 -1
  276. package/dist/src/orgrt/task-dag.d.ts +3 -0
  277. package/dist/src/orgrt/task-dag.d.ts.map +1 -1
  278. package/dist/src/orgrt/task-dag.js +1 -0
  279. package/dist/src/orgrt/task-dag.js.map +1 -1
  280. package/dist/src/orgrt/types.d.ts +5 -0
  281. package/dist/src/orgrt/types.d.ts.map +1 -1
  282. package/dist/src/orgrt/types.js +23 -1
  283. package/dist/src/orgrt/types.js.map +1 -1
  284. package/dist/src/ui/dashboard.html +28 -848
  285. package/dist/src/ui/human-auth.mjs +103 -0
  286. package/dist/src/ui/org-hil.mjs +86 -37
  287. package/dist/src/ui/org-runtime.mjs +21 -13
  288. package/dist/src/ui/routes-org.mjs +49 -445
  289. package/dist/src/ui/server.mjs +90 -2
  290. package/dist/tsconfig.tsbuildinfo +1 -1
  291. package/package.json +9 -9
@@ -1,925 +1,30 @@
1
1
  /**
2
- * Document Pipeline — wires text extraction, chunking, embedding, and SQLite storage
3
- * into an end-to-end ingest/search/export pipeline for the Second Brain.
2
+ * Document Pipeline — the Second Brain's end-to-end ingest/search/export path,
3
+ * and the single import point for it.
4
+ *
5
+ * The implementation lives in focused modules; this file is the public surface
6
+ * every caller (CLI commands, MCP tools, the dashboard, the eval harness)
7
+ * imports from, so the map is here:
8
+ *
9
+ * - `document-types.ts` the shapes: IngestResult, KnowledgeExcerpt,
10
+ * DocumentMeta, ReconcileReport.
11
+ * - `document-store.ts` which store a scope writes to, and the lazy memory
12
+ * bridge that reaches it.
13
+ * - `document-chunking.ts` text → heading-anchored, context-enriched chunks
14
+ * (and the spans that make one citable).
15
+ * - `document-index.ts` the append-only metadata log: versions,
16
+ * tombstones, lookups, superseded filtering,
17
+ * filesystem reconciliation.
18
+ * - `document-ingest.ts` file → extracted → chunked → stored → committed.
19
+ * - `document-search.ts` query → live chunks, decorated for citation.
20
+ * - `okf-bundle.ts` export to / import from a portable OKF bundle.
4
21
  *
5
22
  * @module v1/cli/knowledge/document-pipeline
6
23
  */
7
- import * as crypto from 'node:crypto';
8
- import * as fs from 'node:fs';
9
- import * as os from 'node:os';
10
- import * as path from 'node:path';
11
- import { DOC_EXTENSIONS, extractText } from '../capabilities/cap-documents.js';
12
- // Static import is safe and deliberate: memory-bridge imports only node builtins
13
- // at module scope (everything heavy is lazy), and the project-root rule must not
14
- // be duplicated — two copies of "which directory is this project" is exactly the
15
- // bug this default exists to fix.
16
- import { getProjectRoot } from '../memory/memory-bridge.js';
17
- const DEFAULT_CHUNK_SIZE = 3200;
18
- const DEFAULT_OVERLAP = 400;
19
- // Head-of-chunk cap for text served by searchKnowledge — chunks are
20
- // heading-anchored, so the head carries the most relevant content.
21
- const SEARCH_EXCERPT_TEXT_CAP = 800;
22
- // Inline fallback identical to @monoes/memory's knowledge/document-chunker.ts —
23
- // used only if the dynamic import below fails (package not installed/built).
24
- // Keep in sync if the shared chunker's boundary-snapping logic changes.
25
- const HEADING_LINE_RE = /^#{1,6} /;
26
- const FENCE_LINE_RE = /^\s{0,3}(`{3,}|~{3,})/;
27
- function fenceTogglesInline(text) {
28
- const toggles = [];
29
- let lineStart = 0;
30
- while (lineStart <= text.length) {
31
- const eol = text.indexOf('\n', lineStart);
32
- const line = text.slice(lineStart, eol === -1 ? undefined : eol);
33
- if (FENCE_LINE_RE.test(line))
34
- toggles.push(lineStart);
35
- if (eol === -1)
36
- break;
37
- lineStart = eol + 1;
38
- }
39
- return toggles;
40
- }
41
- function inFenceInline(toggles, pos) {
42
- let lo = 0, hi = toggles.length;
43
- while (lo < hi) {
44
- const mid = (lo + hi) >> 1;
45
- if (toggles[mid] <= pos)
46
- lo = mid + 1;
47
- else
48
- hi = mid;
49
- }
50
- return (lo & 1) === 1;
51
- }
52
- function lastHeadingBefore(text, pos, toggles) {
53
- let i = text.lastIndexOf('\n#', pos - 1);
54
- while (i !== -1) {
55
- const eol = text.indexOf('\n', i + 1);
56
- const line = text.slice(i + 1, eol === -1 ? undefined : eol);
57
- if (HEADING_LINE_RE.test(line) && !inFenceInline(toggles, i + 1))
58
- return line.replace(/^#+ /, '').trim();
59
- i = i > 0 ? text.lastIndexOf('\n#', i - 1) : -1; // fromIndex -1 clamps to 0 — would loop on a match at 0
60
- }
61
- const firstEol = text.indexOf('\n');
62
- const firstLine = firstEol === -1 ? text : text.slice(0, firstEol);
63
- return HEADING_LINE_RE.test(firstLine) &&
64
- !inFenceInline(toggles, 0) &&
65
- firstEol !== -1 &&
66
- firstEol < pos
67
- ? firstLine.replace(/^#+ /, '').trim()
68
- : null;
69
- }
70
- function chunkDocumentInline(docId, text) {
71
- if (text.includes('\r\n'))
72
- text = text.replace(/\r\n/g, '\n');
73
- if (text.length === 0)
74
- return [];
75
- const toggles = fenceTogglesInline(text);
76
- const chunks = [];
77
- let startChar = 0;
78
- let chunkIndex = 0;
79
- while (startChar < text.length) {
80
- let endChar = Math.min(startChar + DEFAULT_CHUNK_SIZE, text.length);
81
- let brokeAtHeading = false;
82
- if (endChar < text.length) {
83
- const windowStart = Math.max(startChar, endChar - Math.floor(DEFAULT_CHUNK_SIZE * 0.2));
84
- const window = text.slice(windowStart, endChar);
85
- let h = window.lastIndexOf('\n#');
86
- while (h !== -1) {
87
- const eol = window.indexOf('\n', h + 1);
88
- const line = window.slice(h + 1, eol === -1 ? undefined : eol);
89
- if (HEADING_LINE_RE.test(line) &&
90
- windowStart + h > startChar &&
91
- !inFenceInline(toggles, windowStart + h + 1))
92
- break;
93
- h = h > 0 ? window.lastIndexOf('\n#', h - 1) : -1;
94
- }
95
- if (h !== -1 && windowStart + h > startChar) {
96
- endChar = windowStart + h + 1;
97
- brokeAtHeading = true;
98
- }
99
- else {
100
- let lastParagraph = window.lastIndexOf('\n\n');
101
- while (lastParagraph > 0 && inFenceInline(toggles, windowStart + lastParagraph + 1)) {
102
- lastParagraph = window.lastIndexOf('\n\n', lastParagraph - 1);
103
- }
104
- if (lastParagraph === 0 && inFenceInline(toggles, windowStart + 1))
105
- lastParagraph = -1;
106
- if (lastParagraph !== -1)
107
- endChar = windowStart + lastParagraph + 2;
108
- }
109
- }
110
- let chunkText = text.slice(startChar, endChar);
111
- const heading = lastHeadingBefore(text, startChar + 1, toggles);
112
- if (heading && !HEADING_LINE_RE.test(chunkText.trimStart()))
113
- chunkText = `§ ${heading}\n${chunkText}`;
114
- chunks.push({
115
- chunkId: `${docId}:${chunkIndex}`,
116
- docId,
117
- text: chunkText,
118
- startChar,
119
- endChar,
120
- chunkIndex,
121
- });
122
- chunkIndex++;
123
- if (endChar >= text.length)
124
- break;
125
- startChar += brokeAtHeading
126
- ? Math.max(1, endChar - startChar)
127
- : Math.max(1, endChar - startChar - DEFAULT_OVERLAP);
128
- }
129
- return chunks;
130
- }
131
- async function chunkDocument(docId, text) {
132
- try {
133
- const mod = await import('@monoes/memory');
134
- return mod.chunkDocument(docId, text, DEFAULT_CHUNK_SIZE, DEFAULT_OVERLAP);
135
- }
136
- catch {
137
- return chunkDocumentInline(docId, text);
138
- }
139
- }
140
- // ── Contextual chunk enrichment (item 6a) ─────────────────────────
141
- // Prepend a situating blurb per chunk before embedding: full heading
142
- // path + doc title + doc summary. No LLM, no network.
143
- //
144
- // The chunker's `§ heading` prefix (line 105) provides only the nearest
145
- // leaf heading. This replaces it with doc-level context so the embedding
146
- // model can distinguish "Memory Coordination" in a hooks doc from
147
- // "Memory Coordination" in a concepts doc.
148
- //
149
- // Applied at INGEST TIME (after chunking, before embedding), so:
150
- // - Works identically regardless of which chunker ran (inline or @monoes/memory)
151
- // - Works identically for both better-sqlite3 and sql.js (pure string ops)
152
- // - Zero dependencies, zero network
153
- const SECTION_PREFIX_RE = /^§ [^\n]+\n/;
154
- /** Cap on the situating summary prepended to each chunk, ellipsis included. */
155
- const SUMMARY_MAX_CHARS = 120;
156
- function extractDocTitle(text, filePath) {
157
- const eol = text.indexOf('\n');
158
- const first = eol === -1 ? text : text.slice(0, eol);
159
- return HEADING_LINE_RE.test(first)
160
- ? first.replace(/^#+ /, '').trim()
161
- : path.basename(filePath, path.extname(filePath)).replace(/[-_]/g, ' ');
162
- }
163
- function extractDocSummary(text) {
164
- const lines = text.split('\n');
165
- let inFence = false;
166
- const parts = [];
167
- for (const line of lines) {
168
- if (FENCE_LINE_RE.test(line)) {
169
- inFence = !inFence;
170
- continue;
171
- }
172
- if (inFence)
173
- continue;
174
- if (HEADING_LINE_RE.test(line)) {
175
- if (parts.length > 0)
176
- break;
177
- continue;
178
- }
179
- const t = line.trim();
180
- if (!t || /^[|=-]/.test(t)) {
181
- if (parts.length > 0)
182
- break;
183
- continue;
184
- }
185
- parts.push(t.startsWith('>') ? t.replace(/^>\s*/, '') : t);
186
- }
187
- const joined = parts.join(' ');
188
- // Truncate with an ellipsis so a clipped summary is visibly clipped. The bare
189
- // 150-char slice this replaces gave no signal that anything was cut, which
190
- // read as a complete sentence to both a human and the embedding model.
191
- return joined.length > SUMMARY_MAX_CHARS
192
- ? `${joined.slice(0, SUMMARY_MAX_CHARS - 3).trimEnd()}...`
193
- : joined;
194
- }
195
- function buildHeadingHierarchy(text, toggles) {
196
- const out = [];
197
- const eol0 = text.indexOf('\n');
198
- const line0 = eol0 === -1 ? text : text.slice(0, eol0);
199
- const firstLevel = line0.match(/^(#{1,6}) /)?.[1]?.length;
200
- if (firstLevel !== undefined && !inFenceInline(toggles, 0)) {
201
- out.push({
202
- level: firstLevel,
203
- text: line0.replace(/^#+ /, '').trim(),
204
- offset: 0,
205
- });
206
- }
207
- let i = text.indexOf('\n#', 0);
208
- while (i !== -1) {
209
- const ls = i + 1;
210
- const e = text.indexOf('\n', ls);
211
- const line = text.slice(ls, e === -1 ? undefined : e);
212
- const level = line.match(/^(#{1,6}) /)?.[1]?.length;
213
- if (level !== undefined && !inFenceInline(toggles, ls)) {
214
- out.push({
215
- level,
216
- text: line.replace(/^#+ /, '').trim(),
217
- offset: ls,
218
- });
219
- }
220
- i = text.indexOf('\n#', ls);
221
- }
222
- return out;
223
- }
224
- function headingPathAt(hierarchy, pos) {
225
- const stack = [];
226
- for (const h of hierarchy) {
227
- if (h.offset >= pos)
228
- break;
229
- while (stack.length > 0 && stack[stack.length - 1].level >= h.level)
230
- stack.pop();
231
- stack.push(h);
232
- }
233
- return stack.map((s) => s.text);
234
- }
235
- /**
236
- * Replace each chunk's `§ heading` prefix with a richer situating blurb:
237
- * § <doc title> · <full heading path>
238
- * <doc summary for non-first chunks>
239
- *
240
- * First chunks that start with their own heading are left untouched (the
241
- * heading IS the context). The summary line is omitted for the first
242
- * chunk since it is adjacent to the summary text anyway.
243
- */
244
- function enrichChunks(chunks, fullText, filePath) {
245
- if (chunks.length === 0)
246
- return chunks;
247
- const toggles = fenceTogglesInline(fullText);
248
- const hierarchy = buildHeadingHierarchy(fullText, toggles);
249
- const title = extractDocTitle(fullText, filePath);
250
- const summary = extractDocSummary(fullText);
251
- return chunks.map((c) => {
252
- let text = c.text;
253
- // First chunk starting with its own heading — the heading IS the context
254
- if (c.chunkIndex === 0 && HEADING_LINE_RE.test(text.trimStart()))
255
- return c;
256
- // Strip the old § leaf-heading prefix; we replace it with a richer one
257
- text = text.replace(SECTION_PREFIX_RE, '');
258
- const hpath = headingPathAt(hierarchy, c.startChar + 1);
259
- const parts = [];
260
- // Title + full heading path
261
- if (hpath.length > 0 && hpath[0] !== title) {
262
- parts.push(`§ ${title} · ${hpath.join(' > ')}`);
263
- }
264
- else if (hpath.length > 1) {
265
- parts.push(`§ ${hpath.join(' > ')}`);
266
- }
267
- else {
268
- parts.push(`§ ${title}`);
269
- }
270
- // Summary for non-first chunks — they are far from the doc intro
271
- if (c.chunkIndex > 0 && summary) {
272
- const snip = summary.length > 120 ? `${summary.slice(0, 117)}...` : summary;
273
- parts.push(snip);
274
- }
275
- return { ...c, text: `${parts.join('\n')}\n${text}` };
276
- });
277
- }
278
- // ── Constants ──────────────────────────────────────────────────────
279
- const KNOWLEDGE_NS_PREFIX = 'knowledge:';
280
- const METADATA_FILE = 'doc-metadata.jsonl';
281
- // Global brain constants — canonical definitions live in memory-bridge.ts
282
- // (GLOBAL_BRAIN / GLOBAL_BRAIN_DIR); duplicated here because the bridge is
283
- // imported lazily and these are needed synchronously.
284
- const GLOBAL_BRAIN_SENTINEL = '@global';
285
- const globalBrainRoot = () => process.env.MONOMIND_GLOBAL_BRAIN_DIR || path.join(os.homedir(), '.monomind', 'global-brain');
286
- /** scope 'global' routes to the personal cross-project store. */
287
- const isGlobalScope = (scope) => scope === 'global';
288
- const effectiveRoot = (scope, rootDir) => isGlobalScope(scope) ? globalBrainRoot() : rootDir;
289
- const storeDbPath = (scope) => isGlobalScope(scope) ? GLOBAL_BRAIN_SENTINEL : undefined;
290
- const IGNORE_DIRS = new Set([
291
- 'node_modules',
292
- '.git',
293
- 'dist',
294
- '.monomind',
295
- '.claude',
296
- '.next',
297
- '__pycache__',
298
- '.venv',
299
- 'vendor',
300
- ]);
301
- const MAX_FILE_SIZE = 50 * 1024 * 1024; // 50MB
302
- // ── Helpers ────────────────────────────────────────────────────────
303
- function namespace(scope) {
304
- return `${KNOWLEDGE_NS_PREFIX}${scope}`;
305
- }
306
- function contentHash(content) {
307
- return crypto.createHash('sha256').update(content).digest('hex');
308
- }
309
- function metadataPath(rootDir) {
310
- const dir = path.join(rootDir, '.monomind', 'knowledge');
311
- fs.mkdirSync(dir, { recursive: true });
312
- return path.join(dir, METADATA_FILE);
313
- }
314
- function readMetadata(rootDir) {
315
- const file = metadataPath(rootDir);
316
- if (!fs.existsSync(file))
317
- return [];
318
- // Last-wins per (filePath, scope): the file is append-only under concurrent
319
- // ingests (session-start detached reindex + a manual `doc ingest` can
320
- // overlap), so duplicates are expected and the newest record is truth.
321
- // Corrupt lines (torn concurrent writes) are skipped, not fatal.
322
- const latest = new Map();
323
- for (const l of fs.readFileSync(file, 'utf-8').split('\n')) {
324
- if (!l.trim())
325
- continue;
326
- try {
327
- const m = JSON.parse(l);
328
- latest.set(`${m.filePath} ${m.scope}`, m);
329
- }
330
- catch {
331
- /* torn line */
332
- }
333
- }
334
- // chunkCount -1 records are removal tombstones (see removeMetadataEntry)
335
- const live = [...latest.values()].filter((m) => m.chunkCount >= 0);
336
- // Occasional compaction: append-only + tombstones grow without bound; when
337
- // the log gets big, rewrite it deduped (atomic rename — a concurrent append
338
- // in the tiny window loses only its own record and self-heals on re-ingest).
339
- try {
340
- if (fs.statSync(file).size > 1024 * 1024) {
341
- const tmp = `${file}.${process.pid}.compact`;
342
- fs.writeFileSync(tmp, live.map((r) => JSON.stringify(r)).join('\n') + (live.length ? '\n' : ''), 'utf-8');
343
- fs.renameSync(tmp, file);
344
- }
345
- }
346
- catch {
347
- /* compaction is best-effort */
348
- }
349
- return live;
350
- }
351
- function appendMetadata(rootDir, meta) {
352
- fs.appendFileSync(metadataPath(rootDir), `${JSON.stringify(meta)}\n`, 'utf-8');
353
- }
354
- function removeMetadataEntry(rootDir, filePath, scope) {
355
- const file = metadataPath(rootDir);
356
- if (!fs.existsSync(file))
357
- return;
358
- // Tombstone by APPEND (chunkCount -1) instead of read-filter-rewrite — the
359
- // rewrite raced concurrent appends and silently dropped them.
360
- appendMetadata(rootDir, {
361
- filePath,
362
- scope,
363
- contentHash: '',
364
- chunkCount: -1,
365
- indexedAt: new Date().toISOString(),
366
- size: 0,
367
- });
368
- }
369
- function toFileEntry(filePath) {
370
- const stat = fs.statSync(filePath);
371
- return {
372
- path: filePath,
373
- absolutePath: path.resolve(filePath),
374
- extension: path.extname(filePath).toLowerCase(),
375
- size: stat.size,
376
- modified: stat.mtime,
377
- created: stat.birthtime,
378
- };
379
- }
380
- // ── Lazy bridge import ─────────────────────────────────────────────
381
- let _bridge;
382
- async function getBridge() {
383
- if (_bridge === null)
384
- return null;
385
- if (_bridge)
386
- return _bridge;
387
- try {
388
- _bridge = await import('../memory/memory-bridge.js');
389
- return _bridge;
390
- }
391
- catch {
392
- _bridge = null;
393
- return null;
394
- }
395
- }
396
- // ── Core Pipeline ──────────────────────────────────────────────────
397
- export async function ingestDocument(filePath, scope = 'shared', rootDir = getProjectRoot(), _metadataCache) {
398
- const resolved = path.resolve(filePath);
399
- const ext = path.extname(resolved).toLowerCase();
400
- // AppleDouble resource forks (`._name.md`) are binary macOS sidecars, not
401
- // documents. The directory walk has skipped dotfiles since 3e429194
402
- // (2026-07-19), but that walk is only ONE of six callers that reach this
403
- // function — the CLI `doc ingest`, the MCP `knowledge_ingest` tool, the
404
- // dashboard's live fs.watch and its polling sweep, the eval harness, and
405
- // `ingestDirectory` all land here, and four of them had no guard at all.
406
- //
407
- // Guarding at the boundary covers every caller at once, including callers
408
- // added later. Guarding at each call site covers only the ones we thought to
409
- // enumerate — which is how two `._` files reached the live index despite a
410
- // working guard in the walk.
411
- //
412
- // Measured on this repo 2026-07-28: 96 `._` entries in the live index, 91 of
413
- // them shadowing a real document of the same name and competing with it for
414
- // top-k slots. That is a direct Recall@5/MRR@10 loss, not wasted storage.
415
- if (isResourceFork(resolved)) {
416
- return {
417
- filePath: resolved,
418
- chunksIndexed: 0,
419
- scope,
420
- skipped: true,
421
- error: 'AppleDouble resource fork',
422
- };
423
- }
424
- if (!DOC_EXTENSIONS.has(ext)) {
425
- return {
426
- filePath: resolved,
427
- chunksIndexed: 0,
428
- scope,
429
- skipped: true,
430
- error: `unsupported extension: ${ext}`,
431
- };
432
- }
433
- if (!fs.existsSync(resolved)) {
434
- return { filePath: resolved, chunksIndexed: 0, scope, skipped: true, error: 'file not found' };
435
- }
436
- const stat = fs.statSync(resolved);
437
- if (stat.size > MAX_FILE_SIZE) {
438
- return {
439
- filePath: resolved,
440
- chunksIndexed: 0,
441
- scope,
442
- skipped: true,
443
- error: 'file too large (>50MB)',
444
- };
445
- }
446
- rootDir = effectiveRoot(scope, rootDir);
447
- const meta = _metadataCache ?? readMetadata(rootDir);
448
- const existing = meta.find((m) => m.filePath === resolved && m.scope === scope);
449
- let fullContent;
450
- try {
451
- const entry = toFileEntry(resolved);
452
- fullContent = await extractText(entry);
453
- }
454
- catch (err) {
455
- return { filePath: resolved, chunksIndexed: 0, scope, skipped: false, error: String(err) };
456
- }
457
- if (!fullContent || fullContent.trim().length === 0) {
458
- return {
459
- filePath: resolved,
460
- chunksIndexed: 0,
461
- scope,
462
- skipped: true,
463
- error: 'no text extracted',
464
- };
465
- }
466
- const hash = contentHash(fullContent);
467
- if (existing && existing.contentHash === hash) {
468
- return { filePath: resolved, chunksIndexed: existing.chunkCount, scope, skipped: true };
469
- }
470
- // NOTE: the previous version's metadata record is deliberately NOT tombstoned
471
- // here. `readMetadata` is last-wins per (filePath, scope), so appending the
472
- // new record below already supersedes the old one — the tombstone was a no-op
473
- // on the success path and destructive on the failure path: it retired a
474
- // perfectly good previous index before knowing whether the replacement would
475
- // land, so a failed re-ingest left the document with NO live version at all.
476
- const docId = `${scope}:${resolved}`;
477
- const rawChunks = await chunkDocument(docId, fullContent);
478
- // monolean: [re-enabled] item 2 shipped 768d gte-modernbert-base — capacity handles enrichment
479
- const chunks = enrichChunks(rawChunks, fullContent, resolved);
480
- const bridge = await getBridge();
481
- let indexed = 0;
482
- for (const chunk of chunks) {
483
- const key = `doc:${hash}:${chunk.chunkIndex}`;
484
- if (bridge) {
485
- try {
486
- const storeResult = await bridge.bridgeStoreEntry({
487
- key,
488
- value: chunk.text,
489
- namespace: namespace(scope),
490
- generateEmbeddingFlag: true,
491
- tags: ['document', ext, `src:${resolved}`],
492
- upsert: true,
493
- dbPath: storeDbPath(scope),
494
- });
495
- if (storeResult?.success)
496
- indexed++;
497
- }
498
- catch (e) {
499
- if (process.env.DEBUG || process.env.MONOMIND_DEBUG)
500
- console.error(`[ingestDocument] failed to store chunk ${chunk.chunkIndex} of ${resolved}:`, e);
501
- }
502
- }
503
- }
504
- // Commit the document version ONLY when EVERY chunk stored. Recording the
505
- // content hash after a partial store was the worse half of this bug: the
506
- // hash check above then skipped the file on every future ingest, so the
507
- // chunks that failed were never retried — a permanently, silently
508
- // half-indexed document feeding knowledge retrieval with no signal at all.
509
- // (Total failure was already handled; partial success was not.)
510
- //
511
- // Not committing is what makes a retry work: chunk keys are
512
- // `doc:<contentHash>:<index>` and stores are upserts, so re-ingesting the
513
- // same bytes rewrites the same keys and fills the gaps. Until it succeeds the
514
- // partially-written chunks sit under a hash that is not live, and superseded
515
- // filtering keeps them out of search (see `liveContentHashes`).
516
- const complete = indexed === chunks.length;
517
- if (complete) {
518
- appendMetadata(rootDir, {
519
- filePath: resolved,
520
- contentHash: hash,
521
- chunkCount: indexed,
522
- indexedAt: new Date().toISOString(),
523
- scope,
524
- size: stat.size,
525
- });
526
- }
527
- return {
528
- filePath: resolved,
529
- chunksIndexed: indexed,
530
- scope,
531
- skipped: false,
532
- ...(complete
533
- ? {}
534
- : indexed > 0
535
- ? {
536
- partial: true,
537
- error: `partial store: ${indexed}/${chunks.length} chunks — version not committed, re-ingest to repair`,
538
- }
539
- : {
540
- error: bridge
541
- ? 'all chunk stores failed'
542
- : 'memory bridge unavailable — nothing indexed',
543
- }),
544
- };
545
- }
546
- export async function ingestDirectory(dirPath, scope = 'shared', opts) {
547
- const scanDir = path.resolve(dirPath);
548
- const rootDir = path.resolve(opts?.rootDir ?? getProjectRoot());
549
- const files = [];
550
- function walk(dir, depth = 0) {
551
- if (depth > 10)
552
- return;
553
- let entries;
554
- try {
555
- entries = fs.readdirSync(dir, { withFileTypes: true });
556
- }
557
- catch {
558
- return;
559
- }
560
- for (const entry of entries) {
561
- // Skip dotfiles/dot-dirs (incl. exFAT `._*` junk) — except `.monodesign`,
562
- // whose critique snapshots are markdown worth surfacing in the Second Brain.
563
- if (entry.name.startsWith('.') && entry.name !== '.monodesign')
564
- continue;
565
- const full = path.join(dir, entry.name);
566
- if (entry.isDirectory()) {
567
- if (!IGNORE_DIRS.has(entry.name))
568
- walk(full, depth + 1);
569
- }
570
- else if (entry.isFile()) {
571
- const ext = path.extname(entry.name).toLowerCase();
572
- if (DOC_EXTENSIONS.has(ext))
573
- files.push(full);
574
- }
575
- }
576
- }
577
- walk(scanDir);
578
- const metadataCache = readMetadata(rootDir);
579
- const result = {
580
- filesProcessed: 0,
581
- filesSkipped: 0,
582
- totalChunks: 0,
583
- errors: [],
584
- results: [],
585
- };
586
- for (let i = 0; i < files.length; i++) {
587
- opts?.onProgress?.(files[i], i, files.length);
588
- const r = await ingestDocument(files[i], scope, rootDir, metadataCache);
589
- result.results.push(r);
590
- if (r.skipped) {
591
- result.filesSkipped++;
592
- }
593
- else {
594
- result.filesProcessed++;
595
- result.totalChunks += r.chunksIndexed;
596
- }
597
- if (r.error && !r.skipped) {
598
- result.errors.push(`${r.filePath}: ${r.error}`);
599
- }
600
- }
601
- return result;
602
- }
603
- // ── Search ─────────────────────────────────────────────────────────
604
- /** Small additive boost so project knowledge wins ties against the global
605
- * brain — local context is more likely to be what the user means. */
606
- const PROJECT_SCOPE_BOOST = 0.05;
607
- // ── Superseded-version filtering ───────────────────────────────────
608
- //
609
- // Chunk keys are `doc:<contentHash>:<chunkIndex>`. Re-ingesting a changed file
610
- // produces a NEW contentHash, so its chunks land under new keys — the previous
611
- // version's rows are never touched (`removeDocument` only tombstones metadata;
612
- // the bridge exposes no delete-by-prefix). The store therefore accumulates every
613
- // version a document has ever had, and all of them stay searchable.
614
- //
615
- // Measured on this repo's own store (2026-07-26): 9,067 `doc:`-keyed rows in
616
- // `knowledge:shared` spanning 798 distinct content hashes, of which only 139
617
- // are current — 8,542 rows (94.2%) are orphaned older versions.
618
- //
619
- // Nothing is deleted here. The current-hash set from doc-metadata.jsonl is used
620
- // to decide what search RETURNS; `includeSuperseded` puts the old versions back
621
- // (flagged `superseded: true`) for anyone who wants document history.
622
- /** Content hashes of the documents currently indexed under `rootDir`. */
623
- export function liveContentHashes(rootDir) {
624
- const live = new Set();
625
- for (const m of readMetadata(rootDir))
626
- if (m.contentHash)
627
- live.add(m.contentHash);
628
- return live;
629
- }
630
- /** True when a metadata log exists under `rootDir`.
631
- *
632
- * An empty live-hash set has two very different causes: the log is missing (we
633
- * cannot judge what is current) or the log exists and every document has been
634
- * removed (nothing is current). Collapsing them made `doc remove` of the LAST
635
- * document a no-op — the tombstoned chunks came straight back in search.
636
- *
637
- * Reads the path directly instead of via `metadataPath`, which mkdir's. */
638
- export function hasKnowledgeMetadata(rootDir) {
639
- return fs.existsSync(path.join(rootDir, '.monomind', 'knowledge', METADATA_FILE));
640
- }
641
- /**
642
- * True when `key` is a document chunk whose version is no longer current.
643
- * Non-`doc:` keys are never superseded. When no metadata is available nothing
644
- * is filtered, because "no metadata" must not read as "everything is stale".
645
- *
646
- * `metadataPresent` defaults to the old `live.size > 0` heuristic so existing
647
- * two-argument callers keep their exact behaviour; pass `hasKnowledgeMetadata`
648
- * to also filter correctly once the last document has been removed.
649
- */
650
- export function isSupersededKey(key, live, metadataPresent = live.size > 0) {
651
- if (!key?.startsWith('doc:'))
652
- return false;
653
- if (!metadataPresent)
654
- return false;
655
- return !live.has(key.split(':')[1] ?? '');
656
- }
657
- /** How many rows to ask the backend for per requested result when superseded
658
- * filtering is active — most rows in a long-lived store are old versions, so
659
- * a 1:1 fetch would return an almost-empty page. */
660
- const SUPERSEDED_OVERFETCH = 20;
661
- const SUPERSEDED_OVERFETCH_CAP = 300;
662
- export function supersededOverfetchLimit(limit, live) {
663
- if (live.size === 0)
664
- return limit;
665
- return Math.min(Math.max(limit * SUPERSEDED_OVERFETCH, limit), SUPERSEDED_OVERFETCH_CAP);
666
- }
667
- export async function searchKnowledge(query, opts) {
668
- const bridge = await getBridge();
669
- if (!bridge)
670
- return [];
671
- const scope = opts?.scope ?? 'shared';
672
- const limit = opts?.limit ?? 10;
673
- const minScore = opts?.minScore ?? 0.3;
674
- const store = opts?.store ?? 'all';
675
- const targets = [];
676
- if (store !== 'global') {
677
- targets.push({
678
- ns: namespace(scope),
679
- root: opts?.rootDir ?? getProjectRoot(),
680
- label: scope,
681
- boost: PROJECT_SCOPE_BOOST,
682
- });
683
- }
684
- if (store !== 'project') {
685
- targets.push({
686
- ns: namespace('global'),
687
- dbPath: GLOBAL_BRAIN_SENTINEL,
688
- root: globalBrainRoot(),
689
- label: 'global',
690
- boost: 0,
691
- });
692
- }
693
- const includeSuperseded = opts?.includeSuperseded === true;
694
- const perTarget = await Promise.all(targets.map(async (t) => {
695
- const meta = readMetadata(t.root);
696
- const hasMeta = hasKnowledgeMetadata(t.root);
697
- const live = new Set();
698
- for (const m of meta)
699
- if (m.contentHash)
700
- live.add(m.contentHash);
701
- // Old versions dominate a long-lived store, so a 1:1 fetch would come back
702
- // nearly empty once they are filtered out. Over-fetch, then trim.
703
- const fetchLimit = includeSuperseded ? limit : supersededOverfetchLimit(limit, live);
704
- const result = await bridge
705
- .bridgeSearchEntries({
706
- query,
707
- namespace: t.ns,
708
- limit: fetchLimit,
709
- threshold: minScore,
710
- dbPath: t.dbPath,
711
- skipRerank: opts?.skipRerank,
712
- includeSuperseded,
713
- rootDir: t.root,
714
- })
715
- .catch(() => null);
716
- if (!result?.success || !result.results.length)
717
- return [];
718
- const hashToFile = new Map();
719
- for (const m of meta)
720
- hashToFile.set(m.contentHash, m.filePath);
721
- const kept = includeSuperseded
722
- ? result.results
723
- : result.results.filter((r) => !isSupersededKey(String(r.key ?? ''), live, hasMeta));
724
- return kept.slice(0, limit).map((r) => {
725
- const parts = r.key.startsWith('doc:') ? r.key.split(':') : [];
726
- const hash = parts[1] ?? '';
727
- const idx = parseInt(parts[2] ?? '0', 10);
728
- // The src: tag stored at ingest is the chunk's OWN provenance — the
729
- // hash→file map can misattribute when two documents share identical
730
- // content, and goes empty when a re-ingested file's hash changed.
731
- const srcTag = (r.tags ?? []).find((tag) => tag.startsWith('src:'));
732
- const superseded = includeSuperseded && isSupersededKey(String(r.key ?? ''), live, hasMeta);
733
- return {
734
- id: r.id,
735
- filePath: srcTag ? srcTag.slice(4) : (hashToFile.get(hash) ?? ''),
736
- // Serve the head of the chunk only — chunks are heading-anchored, so the
737
- // head carries the most relevant text, and full chunks (up to ~3.2K
738
- // chars) bloat every search response.
739
- text: typeof r.content === 'string' && r.content.length > SEARCH_EXCERPT_TEXT_CAP
740
- ? r.content.slice(0, SEARCH_EXCERPT_TEXT_CAP)
741
- : r.content,
742
- similarity: r.score + t.boost,
743
- chunkIndex: Number.isNaN(idx) ? 0 : idx,
744
- scope: t.label,
745
- ...(superseded ? { superseded: true } : {}),
746
- };
747
- });
748
- }));
749
- return perTarget
750
- .flat()
751
- .sort((a, b) => b.similarity - a.similarity)
752
- .slice(0, limit);
753
- }
754
- // ── List / Remove ──────────────────────────────────────────────────
755
- export function listDocuments(rootDir = getProjectRoot(), scope) {
756
- const all = readMetadata(rootDir);
757
- return scope ? all.filter((m) => m.scope === scope) : all;
758
- }
759
- export async function removeDocument(filePath, scope = 'shared', rootDir = getProjectRoot()) {
760
- removeMetadataEntry(rootDir, path.resolve(filePath), scope);
761
- // SQLite cleanup: bridge doesn't expose delete-by-key, so metadata removal is sufficient.
762
- // Orphaned SQLite entries get swept on next full re-index or TTL expiry.
763
- }
764
- // ── Filesystem reconciliation (item 4b-i) ──────────────────────────
765
- /**
766
- * True for macOS AppleDouble sidecars (`._name`).
767
- *
768
- * Matches on the BASENAME PREFIX only. A legitimate document may contain `._`
769
- * elsewhere in its name (`v1._2-release.md`), or live under a dot-directory
770
- * that is deliberately indexed (`.monodesign/` critique snapshots), and
771
- * neither may be rejected.
772
- */
773
- export function isResourceFork(filePath) {
774
- return path.basename(filePath).startsWith('._');
775
- }
776
- /**
777
- * Reconcile the document index against the filesystem: find index entries whose
778
- * source file no longer exists and, only when explicitly asked, tombstone them.
779
- *
780
- * WHY — `removeDocument` only ever tombstoned metadata, and nothing has ever
781
- * compared the index against the disk, so a deleted file stayed searchable
782
- * forever. Measured 2026-07-28: 109 of 257 live entries (42.4%) had no file
783
- * behind them, including `docs/concepts/memory.md`. The Second Brain was
784
- * answering questions from documents the user had deleted.
785
- *
786
- * WHY IT IS THIS CAUTIOUS — "drop the index entry when the file is missing" is
787
- * a rule with a known catastrophic reading. A missing file is also an unmounted
788
- * volume, a checked-out branch, a partial clone, or a permissions failure. Two
789
- * guards were tried against real data and REJECTED; they are recorded here so
790
- * they are not re-proposed:
791
- *
792
- * - "abort if >50% of entries are missing" — the real, legitimate missing
793
- * fraction was 42.4%, so the threshold never fires in the one case we have.
794
- * Any threshold that would have blocked this reconcile is fitted to nothing.
795
- * - "only reconcile when the parent directory still exists" — 26 of the 109
796
- * missing files had no parent directory, because `docs/concepts`,
797
- * `docs/adrs` and `docs/commands` were legitimately deleted wholesale. A
798
- * deleted directory and an unmounted volume are indistinguishable there.
799
- *
800
- * What does discriminate is the ROOT. An intact, readable root carrying a
801
- * metadata log means the tree is genuinely present, so a missing file is
802
- * genuinely gone. A missing root means nothing beneath it is knowable and
803
- * nothing may be removed — hence throw rather than reconcile.
804
- *
805
- * Removal tombstones metadata; it does not delete store rows. Chunks stay on
806
- * disk and fall out of search through the existing superseded filter, which
807
- * keeps this consistent with the mark-don't-destroy rule and leaves the whole
808
- * operation reversible from the archive.
809
- */
810
- export async function reconcileIndex(rootDir = getProjectRoot(), opts) {
811
- const apply = opts?.apply === true;
812
- // Root guard — the unmounted-volume case. Every file below a missing root
813
- // looks deleted, so this must abort rather than reconcile.
814
- if (!rootDir || !fs.existsSync(rootDir)) {
815
- throw new Error(`reconcileIndex: project root does not exist: ${rootDir} — refusing to reconcile ` +
816
- `(an unmounted volume makes every indexed file look deleted)`);
817
- }
818
- if (!hasKnowledgeMetadata(rootDir)) {
819
- throw new Error(`reconcileIndex: no knowledge metadata log under ${rootDir} — refusing to reconcile ` +
820
- `("no metadata" must not read as "everything is stale")`);
821
- }
822
- const records = readMetadata(rootDir).filter((m) => !opts?.scope || m.scope === opts.scope);
823
- const missing = records.filter((m) => !fs.existsSync(m.filePath));
824
- if (!apply || missing.length === 0) {
825
- return { missing, scanned: records.length, applied: apply, removed: 0 };
826
- }
827
- // Archive BEFORE removing, inside the operation so no caller can bypass it
828
- // by forgetting — the same precondition rule the delete path uses.
829
- const dir = path.join(rootDir, '.monomind', 'knowledge', 'archive');
830
- fs.mkdirSync(dir, { recursive: true });
831
- const stamp = new Date().toISOString().replace(/[:.]/g, '-');
832
- const archivePath = path.join(dir, `reconcile-${stamp}.jsonl`);
833
- fs.writeFileSync(archivePath, `${missing.map((m) => JSON.stringify(m)).join('\n')}\n`, 'utf-8');
834
- let removed = 0;
835
- for (const m of missing) {
836
- removeMetadataEntry(rootDir, m.filePath, m.scope);
837
- removed++;
838
- }
839
- return { missing, scanned: records.length, applied: true, removed, archivePath };
840
- }
841
- // ── OKF Export ─────────────────────────────────────────────────────
842
- export async function exportToOKF(outputDir, rootDir = getProjectRoot(), scope = 'shared') {
843
- const docs = listDocuments(rootDir, scope);
844
- fs.mkdirSync(outputDir, { recursive: true });
845
- let exported = 0;
846
- const indexEntries = [];
847
- for (const doc of docs) {
848
- // Read original content
849
- let content = '';
850
- try {
851
- if (fs.existsSync(doc.filePath)) {
852
- const entry = toFileEntry(doc.filePath);
853
- content = await extractText(entry);
854
- }
855
- }
856
- catch {
857
- continue;
858
- }
859
- if (!content)
860
- continue;
861
- const title = path.basename(doc.filePath, path.extname(doc.filePath));
862
- const ext = path.extname(doc.filePath).toLowerCase();
863
- const relativePath = path.relative(rootDir, doc.filePath);
864
- const slug = title.replace(/[^a-zA-Z0-9._-]+/g, '-').toLowerCase();
865
- const outFile = path.join(outputDir, `${slug}.md`);
866
- const yamlEscape = (s) => /[:"'[\]{}#&*!|>%@`]/.test(s) ? `"${s.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"` : s;
867
- const frontmatter = [
868
- '---',
869
- `type: Document`,
870
- `title: ${yamlEscape(title)}`,
871
- `description: ${yamlEscape(`Extracted from ${path.basename(doc.filePath)}`)}`,
872
- `resource: ${yamlEscape(relativePath)}`,
873
- `tags: ["document", ${yamlEscape(ext.slice(1))}]`,
874
- `timestamp: ${yamlEscape(doc.indexedAt)}`,
875
- `contentHash: ${yamlEscape(doc.contentHash)}`,
876
- `chunkCount: ${doc.chunkCount}`,
877
- '---',
878
- '',
879
- ].join('\n');
880
- fs.writeFileSync(outFile, frontmatter + content, 'utf-8');
881
- indexEntries.push(`* [${title}](${slug}.md) - ${path.basename(doc.filePath)} (${doc.chunkCount} chunks)`);
882
- exported++;
883
- }
884
- // Write index.md
885
- const indexContent = [
886
- `# Knowledge Bundle`,
887
- '',
888
- `Exported from monomind on ${new Date().toISOString().slice(0, 10)}`,
889
- '',
890
- ...indexEntries,
891
- '',
892
- ].join('\n');
893
- fs.writeFileSync(path.join(outputDir, 'index.md'), indexContent, 'utf-8');
894
- return { exported, outputDir };
895
- }
896
- // ── OKF Import ─────────────────────────────────────────────────────
897
- export async function importFromOKF(bundleDir, scope = 'shared', rootDir = getProjectRoot()) {
898
- const resolved = path.resolve(bundleDir);
899
- const files = fs
900
- .readdirSync(resolved)
901
- .filter((f) => f.endsWith('.md') && f !== 'index.md' && f !== 'log.md')
902
- .map((f) => path.join(resolved, f));
903
- const result = {
904
- filesProcessed: 0,
905
- filesSkipped: 0,
906
- totalChunks: 0,
907
- errors: [],
908
- results: [],
909
- };
910
- for (const file of files) {
911
- const r = await ingestDocument(file, scope, rootDir);
912
- result.results.push(r);
913
- if (r.skipped) {
914
- result.filesSkipped++;
915
- }
916
- else {
917
- result.filesProcessed++;
918
- result.totalChunks += r.chunksIndexed;
919
- }
920
- if (r.error && !r.skipped)
921
- result.errors.push(`${r.filePath}: ${r.error}`);
922
- }
923
- return result;
924
- }
24
+ export { chunkSpans } from './document-chunking.js';
25
+ export { findDocumentRecord, hasKnowledgeMetadata, isResourceFork, isSupersededKey, listDocuments, listDocumentVersions, liveContentHashes, reconcileIndex, removeDocument, supersededOverfetchLimit, } from './document-index.js';
26
+ export { ingestDirectory, ingestDocument } from './document-ingest.js';
27
+ export { searchKnowledge } from './document-search.js';
28
+ export { getKnowledgeRoot } from './document-store.js';
29
+ export { exportToOKF, importFromOKF } from './okf-bundle.js';
925
30
  //# sourceMappingURL=document-pipeline.js.map