@monoes/monomindcli 2.14.0 → 2.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/helpers/control-start.cjs +10 -13
- package/.claude/skills/monodesign/scripts/context.mjs +5 -5
- package/.claude/skills/monodesign/scripts/critique-storage.mjs +1 -1
- package/.claude/skills/monodesign/scripts/detect-csp.mjs +1 -6
- package/.claude/skills/monodesign/scripts/detector/browser/injected/index.mjs +16 -18
- package/.claude/skills/monodesign/scripts/detector/cli/main.mjs +3 -3
- package/.claude/skills/monodesign/scripts/detector/detect-antipatterns-browser.js +69 -49
- package/.claude/skills/monodesign/scripts/detector/engines/regex/detect-text.mjs +6 -6
- package/.claude/skills/monodesign/scripts/detector/engines/static-html/css-cascade.mjs +6 -5
- package/.claude/skills/monodesign/scripts/detector/engines/static-html/detect-html.mjs +6 -6
- package/.claude/skills/monodesign/scripts/detector/fix/index.mjs +3 -3
- package/.claude/skills/monodesign/scripts/detector/node/file-system.mjs +1 -1
- package/.claude/skills/monodesign/scripts/detector/registry/antipatterns.mjs +1 -1
- package/.claude/skills/monodesign/scripts/detector/rules/checks.mjs +26 -27
- package/.claude/skills/monodesign/scripts/detector/shared/color.mjs +2 -2
- package/.claude/skills/monodesign/scripts/detector/shared/constants.mjs +1 -1
- package/.claude/skills/monodesign/scripts/detector/shared/inline-ignores.mjs +2 -2
- package/.claude/skills/monodesign/scripts/hook-admin.mjs +6 -6
- package/.claude/skills/monodesign/scripts/hook-before-edit.mjs +4 -4
- package/.claude/skills/monodesign/scripts/hook-lib.mjs +20 -11
- package/.claude/skills/monodesign/scripts/hook.mjs +1 -1
- package/.claude/skills/monodesign/scripts/lib/design-parser.mjs +9 -42
- package/.claude/skills/monodesign/scripts/lib/is-generated.mjs +1 -1
- package/.claude/skills/monodesign/scripts/lib/monodesign-config.mjs +5 -5
- package/.claude/skills/monodesign/scripts/lib/monodesign-paths.mjs +1 -1
- package/.claude/skills/monodesign/scripts/live/event-validation.mjs +13 -13
- package/.claude/skills/monodesign/scripts/live/manual-apply.mjs +4 -4
- package/.claude/skills/monodesign/scripts/live/manual-edits-buffer.mjs +1 -1
- package/.claude/skills/monodesign/scripts/live/session-store.mjs +5 -5
- package/.claude/skills/monodesign/scripts/live/svelte-component.mjs +13 -13
- package/.claude/skills/monodesign/scripts/live/sveltekit-adapter.mjs +4 -4
- package/.claude/skills/monodesign/scripts/live/ui-core.mjs +1 -1
- package/.claude/skills/monodesign/scripts/live-accept.mjs +24 -24
- package/.claude/skills/monodesign/scripts/live-browser-dom.js +8 -7
- package/.claude/skills/monodesign/scripts/live-browser-session.js +8 -7
- package/.claude/skills/monodesign/scripts/live-browser.js +406 -433
- package/.claude/skills/monodesign/scripts/live-commit-manual-edits.mjs +6 -6
- package/.claude/skills/monodesign/scripts/live-copy-edit-agent.mjs +2 -2
- package/.claude/skills/monodesign/scripts/live-discard-manual-edits.mjs +1 -1
- package/.claude/skills/monodesign/scripts/live-inject.mjs +7 -7
- package/.claude/skills/monodesign/scripts/live-insert.mjs +11 -11
- package/.claude/skills/monodesign/scripts/live-manual-edit-evidence.mjs +4 -4
- package/.claude/skills/monodesign/scripts/live-poll.mjs +5 -5
- package/.claude/skills/monodesign/scripts/live-server.mjs +10 -10
- package/.claude/skills/monodesign/scripts/live-wrap.mjs +31 -41
- package/.claude/skills/monodesign/scripts/live.mjs +5 -5
- package/.claude/skills/monodesign/scripts/palette.mjs +3 -1
- package/.claude/skills/monodesign/scripts/pin.mjs +1 -1
- package/bin/cli.js +21 -1
- package/bin/mcp-server.js +21 -1
- package/dist/src/capabilities/cap-documents.d.ts.map +1 -1
- package/dist/src/capabilities/cap-documents.js +23 -0
- package/dist/src/capabilities/cap-documents.js.map +1 -1
- package/dist/src/commands/doc-filters.d.ts +19 -0
- package/dist/src/commands/doc-filters.d.ts.map +1 -0
- package/dist/src/commands/doc-filters.js +64 -0
- package/dist/src/commands/doc-filters.js.map +1 -0
- package/dist/src/commands/doc-library.d.ts +22 -0
- package/dist/src/commands/doc-library.d.ts.map +1 -0
- package/dist/src/commands/doc-library.js +454 -0
- package/dist/src/commands/doc-library.js.map +1 -0
- package/dist/src/commands/doc-list.d.ts +9 -0
- package/dist/src/commands/doc-list.d.ts.map +1 -0
- package/dist/src/commands/doc-list.js +99 -0
- package/dist/src/commands/doc-list.js.map +1 -0
- package/dist/src/commands/doc.d.ts.map +1 -1
- package/dist/src/commands/doc.js +51 -37
- package/dist/src/commands/doc.js.map +1 -1
- package/dist/src/commands/ui.d.ts.map +1 -1
- package/dist/src/commands/ui.js +60 -0
- package/dist/src/commands/ui.js.map +1 -1
- package/dist/src/knowledge/capture-envelope.d.ts +85 -0
- package/dist/src/knowledge/capture-envelope.d.ts.map +1 -0
- package/dist/src/knowledge/capture-envelope.js +167 -0
- package/dist/src/knowledge/capture-envelope.js.map +1 -0
- package/dist/src/knowledge/capture-text.d.ts +23 -0
- package/dist/src/knowledge/capture-text.d.ts.map +1 -0
- package/dist/src/knowledge/capture-text.js +49 -0
- package/dist/src/knowledge/capture-text.js.map +1 -0
- package/dist/src/knowledge/citation.d.ts +121 -0
- package/dist/src/knowledge/citation.d.ts.map +1 -0
- package/dist/src/knowledge/citation.js +252 -0
- package/dist/src/knowledge/citation.js.map +1 -0
- package/dist/src/knowledge/document-chunking.d.ts +42 -0
- package/dist/src/knowledge/document-chunking.d.ts.map +1 -0
- package/dist/src/knowledge/document-chunking.js +288 -0
- package/dist/src/knowledge/document-chunking.js.map +1 -0
- package/dist/src/knowledge/document-index.d.ts +124 -0
- package/dist/src/knowledge/document-index.d.ts.map +1 -0
- package/dist/src/knowledge/document-index.js +314 -0
- package/dist/src/knowledge/document-index.js.map +1 -0
- package/dist/src/knowledge/document-ingest.d.ts +23 -0
- package/dist/src/knowledge/document-ingest.d.ts.map +1 -0
- package/dist/src/knowledge/document-ingest.js +368 -0
- package/dist/src/knowledge/document-ingest.js.map +1 -0
- package/dist/src/knowledge/document-pipeline.d.ts +26 -148
- package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
- package/dist/src/knowledge/document-pipeline.js +25 -920
- package/dist/src/knowledge/document-pipeline.js.map +1 -1
- package/dist/src/knowledge/document-search.d.ts +27 -0
- package/dist/src/knowledge/document-search.d.ts.map +1 -0
- package/dist/src/knowledge/document-search.js +138 -0
- package/dist/src/knowledge/document-search.js.map +1 -0
- package/dist/src/knowledge/document-store.d.ts +28 -0
- package/dist/src/knowledge/document-store.d.ts.map +1 -0
- package/dist/src/knowledge/document-store.js +61 -0
- package/dist/src/knowledge/document-store.js.map +1 -0
- package/dist/src/knowledge/document-types.d.ts +103 -0
- package/dist/src/knowledge/document-types.d.ts.map +1 -0
- package/dist/src/knowledge/document-types.js +12 -0
- package/dist/src/knowledge/document-types.js.map +1 -0
- package/dist/src/knowledge/highlights.d.ts +168 -0
- package/dist/src/knowledge/highlights.d.ts.map +1 -0
- package/dist/src/knowledge/highlights.js +311 -0
- package/dist/src/knowledge/highlights.js.map +1 -0
- package/dist/src/knowledge/html-extract.d.ts +43 -0
- package/dist/src/knowledge/html-extract.d.ts.map +1 -0
- package/dist/src/knowledge/html-extract.js +286 -0
- package/dist/src/knowledge/html-extract.js.map +1 -0
- package/dist/src/knowledge/html-tags.d.ts +30 -0
- package/dist/src/knowledge/html-tags.d.ts.map +1 -0
- package/dist/src/knowledge/html-tags.js +229 -0
- package/dist/src/knowledge/html-tags.js.map +1 -0
- package/dist/src/knowledge/library.d.ts +106 -0
- package/dist/src/knowledge/library.d.ts.map +1 -0
- package/dist/src/knowledge/library.js +195 -0
- package/dist/src/knowledge/library.js.map +1 -0
- package/dist/src/knowledge/lookup.d.ts +80 -0
- package/dist/src/knowledge/lookup.d.ts.map +1 -0
- package/dist/src/knowledge/lookup.js +156 -0
- package/dist/src/knowledge/lookup.js.map +1 -0
- package/dist/src/knowledge/mhtml.d.ts +60 -0
- package/dist/src/knowledge/mhtml.d.ts.map +1 -0
- package/dist/src/knowledge/mhtml.js +154 -0
- package/dist/src/knowledge/mhtml.js.map +1 -0
- package/dist/src/knowledge/okf-bundle.d.ts +19 -0
- package/dist/src/knowledge/okf-bundle.d.ts.map +1 -0
- package/dist/src/knowledge/okf-bundle.js +106 -0
- package/dist/src/knowledge/okf-bundle.js.map +1 -0
- package/dist/src/knowledge/profile-store.d.ts +134 -0
- package/dist/src/knowledge/profile-store.d.ts.map +1 -0
- package/dist/src/knowledge/profile-store.js +237 -0
- package/dist/src/knowledge/profile-store.js.map +1 -0
- package/dist/src/knowledge/related.d.ts +58 -0
- package/dist/src/knowledge/related.d.ts.map +1 -0
- package/dist/src/knowledge/related.js +222 -0
- package/dist/src/knowledge/related.js.map +1 -0
- package/dist/src/knowledge/section-diff.d.ts +48 -0
- package/dist/src/knowledge/section-diff.d.ts.map +1 -0
- package/dist/src/knowledge/section-diff.js +151 -0
- package/dist/src/knowledge/section-diff.js.map +1 -0
- package/dist/src/knowledge/watch.d.ts +92 -0
- package/dist/src/knowledge/watch.d.ts.map +1 -0
- package/dist/src/knowledge/watch.js +221 -0
- package/dist/src/knowledge/watch.js.map +1 -0
- package/dist/src/mcp-server.d.ts.map +1 -1
- package/dist/src/mcp-server.js +44 -87
- package/dist/src/mcp-server.js.map +1 -1
- package/dist/src/mcp-tools/browser-instrument-tools.d.ts +35 -0
- package/dist/src/mcp-tools/browser-instrument-tools.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-instrument-tools.js +359 -0
- package/dist/src/mcp-tools/browser-instrument-tools.js.map +1 -0
- package/dist/src/mcp-tools/browser-metrics.d.ts +42 -0
- package/dist/src/mcp-tools/browser-metrics.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-metrics.js +91 -0
- package/dist/src/mcp-tools/browser-metrics.js.map +1 -0
- package/dist/src/mcp-tools/browser-profile-tools.d.ts +16 -0
- package/dist/src/mcp-tools/browser-profile-tools.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-profile-tools.js +324 -0
- package/dist/src/mcp-tools/browser-profile-tools.js.map +1 -0
- package/dist/src/mcp-tools/browser-session.d.ts +68 -0
- package/dist/src/mcp-tools/browser-session.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-session.js +224 -0
- package/dist/src/mcp-tools/browser-session.js.map +1 -0
- package/dist/src/mcp-tools/browser-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/browser-tools.js +10 -176
- package/dist/src/mcp-tools/browser-tools.js.map +1 -1
- package/dist/src/mcp-tools/capture-resource-read.d.ts +115 -0
- package/dist/src/mcp-tools/capture-resource-read.d.ts.map +1 -0
- package/dist/src/mcp-tools/capture-resource-read.js +296 -0
- package/dist/src/mcp-tools/capture-resource-read.js.map +1 -0
- package/dist/src/mcp-tools/capture-resource-tools.d.ts +22 -0
- package/dist/src/mcp-tools/capture-resource-tools.d.ts.map +1 -0
- package/dist/src/mcp-tools/capture-resource-tools.js +182 -0
- package/dist/src/mcp-tools/capture-resource-tools.js.map +1 -0
- package/dist/src/mcp-tools/capture-resources.d.ts +142 -0
- package/dist/src/mcp-tools/capture-resources.d.ts.map +1 -0
- package/dist/src/mcp-tools/capture-resources.js +289 -0
- package/dist/src/mcp-tools/capture-resources.js.map +1 -0
- package/dist/src/mcp-tools/index.d.ts +3 -0
- package/dist/src/mcp-tools/index.d.ts.map +1 -1
- package/dist/src/mcp-tools/index.js +6 -0
- package/dist/src/mcp-tools/index.js.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.js +9 -1
- package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
- package/dist/src/mcp-tools/resource-router.d.ts +86 -0
- package/dist/src/mcp-tools/resource-router.d.ts.map +1 -0
- package/dist/src/mcp-tools/resource-router.js +181 -0
- package/dist/src/mcp-tools/resource-router.js.map +1 -0
- package/dist/src/memory/memory-bridge.js +1 -1
- package/dist/src/memory/memory-bridge.js.map +1 -1
- package/dist/src/orgrt/agent-runner.d.ts +4 -0
- package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
- package/dist/src/orgrt/agent-runner.js +22 -0
- package/dist/src/orgrt/agent-runner.js.map +1 -1
- package/dist/src/orgrt/antigravity-runner.d.ts.map +1 -1
- package/dist/src/orgrt/antigravity-runner.js +2 -1
- package/dist/src/orgrt/antigravity-runner.js.map +1 -1
- package/dist/src/orgrt/authority-mask.d.ts +32 -0
- package/dist/src/orgrt/authority-mask.d.ts.map +1 -0
- package/dist/src/orgrt/authority-mask.js +135 -0
- package/dist/src/orgrt/authority-mask.js.map +1 -0
- package/dist/src/orgrt/codex-runner.d.ts.map +1 -1
- package/dist/src/orgrt/codex-runner.js +2 -1
- package/dist/src/orgrt/codex-runner.js.map +1 -1
- package/dist/src/orgrt/copilot-runner.d.ts.map +1 -1
- package/dist/src/orgrt/copilot-runner.js +2 -1
- package/dist/src/orgrt/copilot-runner.js.map +1 -1
- package/dist/src/orgrt/crush-runner.d.ts.map +1 -1
- package/dist/src/orgrt/crush-runner.js +6 -1
- package/dist/src/orgrt/crush-runner.js.map +1 -1
- package/dist/src/orgrt/daemon.d.ts +4 -0
- package/dist/src/orgrt/daemon.d.ts.map +1 -1
- package/dist/src/orgrt/daemon.js +13 -2
- package/dist/src/orgrt/daemon.js.map +1 -1
- package/dist/src/orgrt/decisions.d.ts +9 -0
- package/dist/src/orgrt/decisions.d.ts.map +1 -1
- package/dist/src/orgrt/decisions.js +32 -5
- package/dist/src/orgrt/decisions.js.map +1 -1
- package/dist/src/orgrt/file-roots.d.ts +2 -1
- package/dist/src/orgrt/file-roots.d.ts.map +1 -1
- package/dist/src/orgrt/file-roots.js +7 -3
- package/dist/src/orgrt/file-roots.js.map +1 -1
- package/dist/src/orgrt/grok-runner.d.ts.map +1 -1
- package/dist/src/orgrt/grok-runner.js +2 -1
- package/dist/src/orgrt/grok-runner.js.map +1 -1
- package/dist/src/orgrt/hermes-runner.d.ts.map +1 -1
- package/dist/src/orgrt/hermes-runner.js +2 -1
- package/dist/src/orgrt/hermes-runner.js.map +1 -1
- package/dist/src/orgrt/inbox.d.ts.map +1 -1
- package/dist/src/orgrt/inbox.js +70 -4
- package/dist/src/orgrt/inbox.js.map +1 -1
- package/dist/src/orgrt/kimicode-runner.d.ts.map +1 -1
- package/dist/src/orgrt/kimicode-runner.js +2 -1
- package/dist/src/orgrt/kimicode-runner.js.map +1 -1
- package/dist/src/orgrt/opencode-runner.d.ts.map +1 -1
- package/dist/src/orgrt/opencode-runner.js +3 -2
- package/dist/src/orgrt/opencode-runner.js.map +1 -1
- package/dist/src/orgrt/pi-rpc-runner.d.ts.map +1 -1
- package/dist/src/orgrt/pi-rpc-runner.js +3 -2
- package/dist/src/orgrt/pi-rpc-runner.js.map +1 -1
- package/dist/src/orgrt/pi-runner.d.ts.map +1 -1
- package/dist/src/orgrt/pi-runner.js +2 -1
- package/dist/src/orgrt/pi-runner.js.map +1 -1
- package/dist/src/orgrt/policy.d.ts.map +1 -1
- package/dist/src/orgrt/policy.js +3 -0
- package/dist/src/orgrt/policy.js.map +1 -1
- package/dist/src/orgrt/qwen-rpc-runner.d.ts.map +1 -1
- package/dist/src/orgrt/qwen-rpc-runner.js +3 -2
- package/dist/src/orgrt/qwen-rpc-runner.js.map +1 -1
- package/dist/src/orgrt/qwen-runner.d.ts.map +1 -1
- package/dist/src/orgrt/qwen-runner.js +2 -1
- package/dist/src/orgrt/qwen-runner.js.map +1 -1
- package/dist/src/orgrt/role-sandbox.d.ts +21 -0
- package/dist/src/orgrt/role-sandbox.d.ts.map +1 -1
- package/dist/src/orgrt/role-sandbox.js +31 -6
- package/dist/src/orgrt/role-sandbox.js.map +1 -1
- package/dist/src/orgrt/session-ledger.d.ts +7 -0
- package/dist/src/orgrt/session-ledger.d.ts.map +1 -1
- package/dist/src/orgrt/session-ledger.js +15 -0
- package/dist/src/orgrt/session-ledger.js.map +1 -1
- package/dist/src/orgrt/session.d.ts.map +1 -1
- package/dist/src/orgrt/session.js +42 -10
- package/dist/src/orgrt/session.js.map +1 -1
- package/dist/src/orgrt/task-dag.d.ts +3 -0
- package/dist/src/orgrt/task-dag.d.ts.map +1 -1
- package/dist/src/orgrt/task-dag.js +1 -0
- package/dist/src/orgrt/task-dag.js.map +1 -1
- package/dist/src/orgrt/types.d.ts +5 -0
- package/dist/src/orgrt/types.d.ts.map +1 -1
- package/dist/src/orgrt/types.js +23 -1
- package/dist/src/orgrt/types.js.map +1 -1
- package/dist/src/ui/dashboard.html +28 -848
- package/dist/src/ui/human-auth.mjs +103 -0
- package/dist/src/ui/org-hil.mjs +86 -37
- package/dist/src/ui/org-runtime.mjs +21 -13
- package/dist/src/ui/routes-org.mjs +49 -445
- package/dist/src/ui/server.mjs +90 -2
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/package.json +9 -9
|
@@ -1,925 +1,30 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Document Pipeline —
|
|
3
|
-
*
|
|
2
|
+
* Document Pipeline — the Second Brain's end-to-end ingest/search/export path,
|
|
3
|
+
* and the single import point for it.
|
|
4
|
+
*
|
|
5
|
+
* The implementation lives in focused modules; this file is the public surface
|
|
6
|
+
* every caller (CLI commands, MCP tools, the dashboard, the eval harness)
|
|
7
|
+
* imports from, so the map is here:
|
|
8
|
+
*
|
|
9
|
+
* - `document-types.ts` the shapes: IngestResult, KnowledgeExcerpt,
|
|
10
|
+
* DocumentMeta, ReconcileReport.
|
|
11
|
+
* - `document-store.ts` which store a scope writes to, and the lazy memory
|
|
12
|
+
* bridge that reaches it.
|
|
13
|
+
* - `document-chunking.ts` text → heading-anchored, context-enriched chunks
|
|
14
|
+
* (and the spans that make one citable).
|
|
15
|
+
* - `document-index.ts` the append-only metadata log: versions,
|
|
16
|
+
* tombstones, lookups, superseded filtering,
|
|
17
|
+
* filesystem reconciliation.
|
|
18
|
+
* - `document-ingest.ts` file → extracted → chunked → stored → committed.
|
|
19
|
+
* - `document-search.ts` query → live chunks, decorated for citation.
|
|
20
|
+
* - `okf-bundle.ts` export to / import from a portable OKF bundle.
|
|
4
21
|
*
|
|
5
22
|
* @module v1/cli/knowledge/document-pipeline
|
|
6
23
|
*/
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
// at module scope (everything heavy is lazy), and the project-root rule must not
|
|
14
|
-
// be duplicated — two copies of "which directory is this project" is exactly the
|
|
15
|
-
// bug this default exists to fix.
|
|
16
|
-
import { getProjectRoot } from '../memory/memory-bridge.js';
|
|
17
|
-
const DEFAULT_CHUNK_SIZE = 3200;
|
|
18
|
-
const DEFAULT_OVERLAP = 400;
|
|
19
|
-
// Head-of-chunk cap for text served by searchKnowledge — chunks are
|
|
20
|
-
// heading-anchored, so the head carries the most relevant content.
|
|
21
|
-
const SEARCH_EXCERPT_TEXT_CAP = 800;
|
|
22
|
-
// Inline fallback identical to @monoes/memory's knowledge/document-chunker.ts —
|
|
23
|
-
// used only if the dynamic import below fails (package not installed/built).
|
|
24
|
-
// Keep in sync if the shared chunker's boundary-snapping logic changes.
|
|
25
|
-
const HEADING_LINE_RE = /^#{1,6} /;
|
|
26
|
-
const FENCE_LINE_RE = /^\s{0,3}(`{3,}|~{3,})/;
|
|
27
|
-
function fenceTogglesInline(text) {
|
|
28
|
-
const toggles = [];
|
|
29
|
-
let lineStart = 0;
|
|
30
|
-
while (lineStart <= text.length) {
|
|
31
|
-
const eol = text.indexOf('\n', lineStart);
|
|
32
|
-
const line = text.slice(lineStart, eol === -1 ? undefined : eol);
|
|
33
|
-
if (FENCE_LINE_RE.test(line))
|
|
34
|
-
toggles.push(lineStart);
|
|
35
|
-
if (eol === -1)
|
|
36
|
-
break;
|
|
37
|
-
lineStart = eol + 1;
|
|
38
|
-
}
|
|
39
|
-
return toggles;
|
|
40
|
-
}
|
|
41
|
-
function inFenceInline(toggles, pos) {
|
|
42
|
-
let lo = 0, hi = toggles.length;
|
|
43
|
-
while (lo < hi) {
|
|
44
|
-
const mid = (lo + hi) >> 1;
|
|
45
|
-
if (toggles[mid] <= pos)
|
|
46
|
-
lo = mid + 1;
|
|
47
|
-
else
|
|
48
|
-
hi = mid;
|
|
49
|
-
}
|
|
50
|
-
return (lo & 1) === 1;
|
|
51
|
-
}
|
|
52
|
-
function lastHeadingBefore(text, pos, toggles) {
|
|
53
|
-
let i = text.lastIndexOf('\n#', pos - 1);
|
|
54
|
-
while (i !== -1) {
|
|
55
|
-
const eol = text.indexOf('\n', i + 1);
|
|
56
|
-
const line = text.slice(i + 1, eol === -1 ? undefined : eol);
|
|
57
|
-
if (HEADING_LINE_RE.test(line) && !inFenceInline(toggles, i + 1))
|
|
58
|
-
return line.replace(/^#+ /, '').trim();
|
|
59
|
-
i = i > 0 ? text.lastIndexOf('\n#', i - 1) : -1; // fromIndex -1 clamps to 0 — would loop on a match at 0
|
|
60
|
-
}
|
|
61
|
-
const firstEol = text.indexOf('\n');
|
|
62
|
-
const firstLine = firstEol === -1 ? text : text.slice(0, firstEol);
|
|
63
|
-
return HEADING_LINE_RE.test(firstLine) &&
|
|
64
|
-
!inFenceInline(toggles, 0) &&
|
|
65
|
-
firstEol !== -1 &&
|
|
66
|
-
firstEol < pos
|
|
67
|
-
? firstLine.replace(/^#+ /, '').trim()
|
|
68
|
-
: null;
|
|
69
|
-
}
|
|
70
|
-
function chunkDocumentInline(docId, text) {
|
|
71
|
-
if (text.includes('\r\n'))
|
|
72
|
-
text = text.replace(/\r\n/g, '\n');
|
|
73
|
-
if (text.length === 0)
|
|
74
|
-
return [];
|
|
75
|
-
const toggles = fenceTogglesInline(text);
|
|
76
|
-
const chunks = [];
|
|
77
|
-
let startChar = 0;
|
|
78
|
-
let chunkIndex = 0;
|
|
79
|
-
while (startChar < text.length) {
|
|
80
|
-
let endChar = Math.min(startChar + DEFAULT_CHUNK_SIZE, text.length);
|
|
81
|
-
let brokeAtHeading = false;
|
|
82
|
-
if (endChar < text.length) {
|
|
83
|
-
const windowStart = Math.max(startChar, endChar - Math.floor(DEFAULT_CHUNK_SIZE * 0.2));
|
|
84
|
-
const window = text.slice(windowStart, endChar);
|
|
85
|
-
let h = window.lastIndexOf('\n#');
|
|
86
|
-
while (h !== -1) {
|
|
87
|
-
const eol = window.indexOf('\n', h + 1);
|
|
88
|
-
const line = window.slice(h + 1, eol === -1 ? undefined : eol);
|
|
89
|
-
if (HEADING_LINE_RE.test(line) &&
|
|
90
|
-
windowStart + h > startChar &&
|
|
91
|
-
!inFenceInline(toggles, windowStart + h + 1))
|
|
92
|
-
break;
|
|
93
|
-
h = h > 0 ? window.lastIndexOf('\n#', h - 1) : -1;
|
|
94
|
-
}
|
|
95
|
-
if (h !== -1 && windowStart + h > startChar) {
|
|
96
|
-
endChar = windowStart + h + 1;
|
|
97
|
-
brokeAtHeading = true;
|
|
98
|
-
}
|
|
99
|
-
else {
|
|
100
|
-
let lastParagraph = window.lastIndexOf('\n\n');
|
|
101
|
-
while (lastParagraph > 0 && inFenceInline(toggles, windowStart + lastParagraph + 1)) {
|
|
102
|
-
lastParagraph = window.lastIndexOf('\n\n', lastParagraph - 1);
|
|
103
|
-
}
|
|
104
|
-
if (lastParagraph === 0 && inFenceInline(toggles, windowStart + 1))
|
|
105
|
-
lastParagraph = -1;
|
|
106
|
-
if (lastParagraph !== -1)
|
|
107
|
-
endChar = windowStart + lastParagraph + 2;
|
|
108
|
-
}
|
|
109
|
-
}
|
|
110
|
-
let chunkText = text.slice(startChar, endChar);
|
|
111
|
-
const heading = lastHeadingBefore(text, startChar + 1, toggles);
|
|
112
|
-
if (heading && !HEADING_LINE_RE.test(chunkText.trimStart()))
|
|
113
|
-
chunkText = `§ ${heading}\n${chunkText}`;
|
|
114
|
-
chunks.push({
|
|
115
|
-
chunkId: `${docId}:${chunkIndex}`,
|
|
116
|
-
docId,
|
|
117
|
-
text: chunkText,
|
|
118
|
-
startChar,
|
|
119
|
-
endChar,
|
|
120
|
-
chunkIndex,
|
|
121
|
-
});
|
|
122
|
-
chunkIndex++;
|
|
123
|
-
if (endChar >= text.length)
|
|
124
|
-
break;
|
|
125
|
-
startChar += brokeAtHeading
|
|
126
|
-
? Math.max(1, endChar - startChar)
|
|
127
|
-
: Math.max(1, endChar - startChar - DEFAULT_OVERLAP);
|
|
128
|
-
}
|
|
129
|
-
return chunks;
|
|
130
|
-
}
|
|
131
|
-
async function chunkDocument(docId, text) {
|
|
132
|
-
try {
|
|
133
|
-
const mod = await import('@monoes/memory');
|
|
134
|
-
return mod.chunkDocument(docId, text, DEFAULT_CHUNK_SIZE, DEFAULT_OVERLAP);
|
|
135
|
-
}
|
|
136
|
-
catch {
|
|
137
|
-
return chunkDocumentInline(docId, text);
|
|
138
|
-
}
|
|
139
|
-
}
|
|
140
|
-
// ── Contextual chunk enrichment (item 6a) ─────────────────────────
|
|
141
|
-
// Prepend a situating blurb per chunk before embedding: full heading
|
|
142
|
-
// path + doc title + doc summary. No LLM, no network.
|
|
143
|
-
//
|
|
144
|
-
// The chunker's `§ heading` prefix (line 105) provides only the nearest
|
|
145
|
-
// leaf heading. This replaces it with doc-level context so the embedding
|
|
146
|
-
// model can distinguish "Memory Coordination" in a hooks doc from
|
|
147
|
-
// "Memory Coordination" in a concepts doc.
|
|
148
|
-
//
|
|
149
|
-
// Applied at INGEST TIME (after chunking, before embedding), so:
|
|
150
|
-
// - Works identically regardless of which chunker ran (inline or @monoes/memory)
|
|
151
|
-
// - Works identically for both better-sqlite3 and sql.js (pure string ops)
|
|
152
|
-
// - Zero dependencies, zero network
|
|
153
|
-
const SECTION_PREFIX_RE = /^§ [^\n]+\n/;
|
|
154
|
-
/** Cap on the situating summary prepended to each chunk, ellipsis included. */
|
|
155
|
-
const SUMMARY_MAX_CHARS = 120;
|
|
156
|
-
function extractDocTitle(text, filePath) {
|
|
157
|
-
const eol = text.indexOf('\n');
|
|
158
|
-
const first = eol === -1 ? text : text.slice(0, eol);
|
|
159
|
-
return HEADING_LINE_RE.test(first)
|
|
160
|
-
? first.replace(/^#+ /, '').trim()
|
|
161
|
-
: path.basename(filePath, path.extname(filePath)).replace(/[-_]/g, ' ');
|
|
162
|
-
}
|
|
163
|
-
function extractDocSummary(text) {
|
|
164
|
-
const lines = text.split('\n');
|
|
165
|
-
let inFence = false;
|
|
166
|
-
const parts = [];
|
|
167
|
-
for (const line of lines) {
|
|
168
|
-
if (FENCE_LINE_RE.test(line)) {
|
|
169
|
-
inFence = !inFence;
|
|
170
|
-
continue;
|
|
171
|
-
}
|
|
172
|
-
if (inFence)
|
|
173
|
-
continue;
|
|
174
|
-
if (HEADING_LINE_RE.test(line)) {
|
|
175
|
-
if (parts.length > 0)
|
|
176
|
-
break;
|
|
177
|
-
continue;
|
|
178
|
-
}
|
|
179
|
-
const t = line.trim();
|
|
180
|
-
if (!t || /^[|=-]/.test(t)) {
|
|
181
|
-
if (parts.length > 0)
|
|
182
|
-
break;
|
|
183
|
-
continue;
|
|
184
|
-
}
|
|
185
|
-
parts.push(t.startsWith('>') ? t.replace(/^>\s*/, '') : t);
|
|
186
|
-
}
|
|
187
|
-
const joined = parts.join(' ');
|
|
188
|
-
// Truncate with an ellipsis so a clipped summary is visibly clipped. The bare
|
|
189
|
-
// 150-char slice this replaces gave no signal that anything was cut, which
|
|
190
|
-
// read as a complete sentence to both a human and the embedding model.
|
|
191
|
-
return joined.length > SUMMARY_MAX_CHARS
|
|
192
|
-
? `${joined.slice(0, SUMMARY_MAX_CHARS - 3).trimEnd()}...`
|
|
193
|
-
: joined;
|
|
194
|
-
}
|
|
195
|
-
function buildHeadingHierarchy(text, toggles) {
|
|
196
|
-
const out = [];
|
|
197
|
-
const eol0 = text.indexOf('\n');
|
|
198
|
-
const line0 = eol0 === -1 ? text : text.slice(0, eol0);
|
|
199
|
-
const firstLevel = line0.match(/^(#{1,6}) /)?.[1]?.length;
|
|
200
|
-
if (firstLevel !== undefined && !inFenceInline(toggles, 0)) {
|
|
201
|
-
out.push({
|
|
202
|
-
level: firstLevel,
|
|
203
|
-
text: line0.replace(/^#+ /, '').trim(),
|
|
204
|
-
offset: 0,
|
|
205
|
-
});
|
|
206
|
-
}
|
|
207
|
-
let i = text.indexOf('\n#', 0);
|
|
208
|
-
while (i !== -1) {
|
|
209
|
-
const ls = i + 1;
|
|
210
|
-
const e = text.indexOf('\n', ls);
|
|
211
|
-
const line = text.slice(ls, e === -1 ? undefined : e);
|
|
212
|
-
const level = line.match(/^(#{1,6}) /)?.[1]?.length;
|
|
213
|
-
if (level !== undefined && !inFenceInline(toggles, ls)) {
|
|
214
|
-
out.push({
|
|
215
|
-
level,
|
|
216
|
-
text: line.replace(/^#+ /, '').trim(),
|
|
217
|
-
offset: ls,
|
|
218
|
-
});
|
|
219
|
-
}
|
|
220
|
-
i = text.indexOf('\n#', ls);
|
|
221
|
-
}
|
|
222
|
-
return out;
|
|
223
|
-
}
|
|
224
|
-
function headingPathAt(hierarchy, pos) {
|
|
225
|
-
const stack = [];
|
|
226
|
-
for (const h of hierarchy) {
|
|
227
|
-
if (h.offset >= pos)
|
|
228
|
-
break;
|
|
229
|
-
while (stack.length > 0 && stack[stack.length - 1].level >= h.level)
|
|
230
|
-
stack.pop();
|
|
231
|
-
stack.push(h);
|
|
232
|
-
}
|
|
233
|
-
return stack.map((s) => s.text);
|
|
234
|
-
}
|
|
235
|
-
/**
|
|
236
|
-
* Replace each chunk's `§ heading` prefix with a richer situating blurb:
|
|
237
|
-
* § <doc title> · <full heading path>
|
|
238
|
-
* <doc summary for non-first chunks>
|
|
239
|
-
*
|
|
240
|
-
* First chunks that start with their own heading are left untouched (the
|
|
241
|
-
* heading IS the context). The summary line is omitted for the first
|
|
242
|
-
* chunk since it is adjacent to the summary text anyway.
|
|
243
|
-
*/
|
|
244
|
-
function enrichChunks(chunks, fullText, filePath) {
|
|
245
|
-
if (chunks.length === 0)
|
|
246
|
-
return chunks;
|
|
247
|
-
const toggles = fenceTogglesInline(fullText);
|
|
248
|
-
const hierarchy = buildHeadingHierarchy(fullText, toggles);
|
|
249
|
-
const title = extractDocTitle(fullText, filePath);
|
|
250
|
-
const summary = extractDocSummary(fullText);
|
|
251
|
-
return chunks.map((c) => {
|
|
252
|
-
let text = c.text;
|
|
253
|
-
// First chunk starting with its own heading — the heading IS the context
|
|
254
|
-
if (c.chunkIndex === 0 && HEADING_LINE_RE.test(text.trimStart()))
|
|
255
|
-
return c;
|
|
256
|
-
// Strip the old § leaf-heading prefix; we replace it with a richer one
|
|
257
|
-
text = text.replace(SECTION_PREFIX_RE, '');
|
|
258
|
-
const hpath = headingPathAt(hierarchy, c.startChar + 1);
|
|
259
|
-
const parts = [];
|
|
260
|
-
// Title + full heading path
|
|
261
|
-
if (hpath.length > 0 && hpath[0] !== title) {
|
|
262
|
-
parts.push(`§ ${title} · ${hpath.join(' > ')}`);
|
|
263
|
-
}
|
|
264
|
-
else if (hpath.length > 1) {
|
|
265
|
-
parts.push(`§ ${hpath.join(' > ')}`);
|
|
266
|
-
}
|
|
267
|
-
else {
|
|
268
|
-
parts.push(`§ ${title}`);
|
|
269
|
-
}
|
|
270
|
-
// Summary for non-first chunks — they are far from the doc intro
|
|
271
|
-
if (c.chunkIndex > 0 && summary) {
|
|
272
|
-
const snip = summary.length > 120 ? `${summary.slice(0, 117)}...` : summary;
|
|
273
|
-
parts.push(snip);
|
|
274
|
-
}
|
|
275
|
-
return { ...c, text: `${parts.join('\n')}\n${text}` };
|
|
276
|
-
});
|
|
277
|
-
}
|
|
278
|
-
// ── Constants ──────────────────────────────────────────────────────
|
|
279
|
-
const KNOWLEDGE_NS_PREFIX = 'knowledge:';
|
|
280
|
-
const METADATA_FILE = 'doc-metadata.jsonl';
|
|
281
|
-
// Global brain constants — canonical definitions live in memory-bridge.ts
|
|
282
|
-
// (GLOBAL_BRAIN / GLOBAL_BRAIN_DIR); duplicated here because the bridge is
|
|
283
|
-
// imported lazily and these are needed synchronously.
|
|
284
|
-
const GLOBAL_BRAIN_SENTINEL = '@global';
|
|
285
|
-
const globalBrainRoot = () => process.env.MONOMIND_GLOBAL_BRAIN_DIR || path.join(os.homedir(), '.monomind', 'global-brain');
|
|
286
|
-
/** scope 'global' routes to the personal cross-project store. */
|
|
287
|
-
const isGlobalScope = (scope) => scope === 'global';
|
|
288
|
-
const effectiveRoot = (scope, rootDir) => isGlobalScope(scope) ? globalBrainRoot() : rootDir;
|
|
289
|
-
const storeDbPath = (scope) => isGlobalScope(scope) ? GLOBAL_BRAIN_SENTINEL : undefined;
|
|
290
|
-
const IGNORE_DIRS = new Set([
|
|
291
|
-
'node_modules',
|
|
292
|
-
'.git',
|
|
293
|
-
'dist',
|
|
294
|
-
'.monomind',
|
|
295
|
-
'.claude',
|
|
296
|
-
'.next',
|
|
297
|
-
'__pycache__',
|
|
298
|
-
'.venv',
|
|
299
|
-
'vendor',
|
|
300
|
-
]);
|
|
301
|
-
const MAX_FILE_SIZE = 50 * 1024 * 1024; // 50MB
|
|
302
|
-
// ── Helpers ────────────────────────────────────────────────────────
|
|
303
|
-
function namespace(scope) {
|
|
304
|
-
return `${KNOWLEDGE_NS_PREFIX}${scope}`;
|
|
305
|
-
}
|
|
306
|
-
function contentHash(content) {
|
|
307
|
-
return crypto.createHash('sha256').update(content).digest('hex');
|
|
308
|
-
}
|
|
309
|
-
function metadataPath(rootDir) {
|
|
310
|
-
const dir = path.join(rootDir, '.monomind', 'knowledge');
|
|
311
|
-
fs.mkdirSync(dir, { recursive: true });
|
|
312
|
-
return path.join(dir, METADATA_FILE);
|
|
313
|
-
}
|
|
314
|
-
function readMetadata(rootDir) {
|
|
315
|
-
const file = metadataPath(rootDir);
|
|
316
|
-
if (!fs.existsSync(file))
|
|
317
|
-
return [];
|
|
318
|
-
// Last-wins per (filePath, scope): the file is append-only under concurrent
|
|
319
|
-
// ingests (session-start detached reindex + a manual `doc ingest` can
|
|
320
|
-
// overlap), so duplicates are expected and the newest record is truth.
|
|
321
|
-
// Corrupt lines (torn concurrent writes) are skipped, not fatal.
|
|
322
|
-
const latest = new Map();
|
|
323
|
-
for (const l of fs.readFileSync(file, 'utf-8').split('\n')) {
|
|
324
|
-
if (!l.trim())
|
|
325
|
-
continue;
|
|
326
|
-
try {
|
|
327
|
-
const m = JSON.parse(l);
|
|
328
|
-
latest.set(`${m.filePath} ${m.scope}`, m);
|
|
329
|
-
}
|
|
330
|
-
catch {
|
|
331
|
-
/* torn line */
|
|
332
|
-
}
|
|
333
|
-
}
|
|
334
|
-
// chunkCount -1 records are removal tombstones (see removeMetadataEntry)
|
|
335
|
-
const live = [...latest.values()].filter((m) => m.chunkCount >= 0);
|
|
336
|
-
// Occasional compaction: append-only + tombstones grow without bound; when
|
|
337
|
-
// the log gets big, rewrite it deduped (atomic rename — a concurrent append
|
|
338
|
-
// in the tiny window loses only its own record and self-heals on re-ingest).
|
|
339
|
-
try {
|
|
340
|
-
if (fs.statSync(file).size > 1024 * 1024) {
|
|
341
|
-
const tmp = `${file}.${process.pid}.compact`;
|
|
342
|
-
fs.writeFileSync(tmp, live.map((r) => JSON.stringify(r)).join('\n') + (live.length ? '\n' : ''), 'utf-8');
|
|
343
|
-
fs.renameSync(tmp, file);
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
|
-
catch {
|
|
347
|
-
/* compaction is best-effort */
|
|
348
|
-
}
|
|
349
|
-
return live;
|
|
350
|
-
}
|
|
351
|
-
function appendMetadata(rootDir, meta) {
|
|
352
|
-
fs.appendFileSync(metadataPath(rootDir), `${JSON.stringify(meta)}\n`, 'utf-8');
|
|
353
|
-
}
|
|
354
|
-
function removeMetadataEntry(rootDir, filePath, scope) {
|
|
355
|
-
const file = metadataPath(rootDir);
|
|
356
|
-
if (!fs.existsSync(file))
|
|
357
|
-
return;
|
|
358
|
-
// Tombstone by APPEND (chunkCount -1) instead of read-filter-rewrite — the
|
|
359
|
-
// rewrite raced concurrent appends and silently dropped them.
|
|
360
|
-
appendMetadata(rootDir, {
|
|
361
|
-
filePath,
|
|
362
|
-
scope,
|
|
363
|
-
contentHash: '',
|
|
364
|
-
chunkCount: -1,
|
|
365
|
-
indexedAt: new Date().toISOString(),
|
|
366
|
-
size: 0,
|
|
367
|
-
});
|
|
368
|
-
}
|
|
369
|
-
function toFileEntry(filePath) {
|
|
370
|
-
const stat = fs.statSync(filePath);
|
|
371
|
-
return {
|
|
372
|
-
path: filePath,
|
|
373
|
-
absolutePath: path.resolve(filePath),
|
|
374
|
-
extension: path.extname(filePath).toLowerCase(),
|
|
375
|
-
size: stat.size,
|
|
376
|
-
modified: stat.mtime,
|
|
377
|
-
created: stat.birthtime,
|
|
378
|
-
};
|
|
379
|
-
}
|
|
380
|
-
// ── Lazy bridge import ─────────────────────────────────────────────
|
|
381
|
-
let _bridge;
|
|
382
|
-
async function getBridge() {
|
|
383
|
-
if (_bridge === null)
|
|
384
|
-
return null;
|
|
385
|
-
if (_bridge)
|
|
386
|
-
return _bridge;
|
|
387
|
-
try {
|
|
388
|
-
_bridge = await import('../memory/memory-bridge.js');
|
|
389
|
-
return _bridge;
|
|
390
|
-
}
|
|
391
|
-
catch {
|
|
392
|
-
_bridge = null;
|
|
393
|
-
return null;
|
|
394
|
-
}
|
|
395
|
-
}
|
|
396
|
-
// ── Core Pipeline ──────────────────────────────────────────────────
|
|
397
|
-
export async function ingestDocument(filePath, scope = 'shared', rootDir = getProjectRoot(), _metadataCache) {
|
|
398
|
-
const resolved = path.resolve(filePath);
|
|
399
|
-
const ext = path.extname(resolved).toLowerCase();
|
|
400
|
-
// AppleDouble resource forks (`._name.md`) are binary macOS sidecars, not
|
|
401
|
-
// documents. The directory walk has skipped dotfiles since 3e429194
|
|
402
|
-
// (2026-07-19), but that walk is only ONE of six callers that reach this
|
|
403
|
-
// function — the CLI `doc ingest`, the MCP `knowledge_ingest` tool, the
|
|
404
|
-
// dashboard's live fs.watch and its polling sweep, the eval harness, and
|
|
405
|
-
// `ingestDirectory` all land here, and four of them had no guard at all.
|
|
406
|
-
//
|
|
407
|
-
// Guarding at the boundary covers every caller at once, including callers
|
|
408
|
-
// added later. Guarding at each call site covers only the ones we thought to
|
|
409
|
-
// enumerate — which is how two `._` files reached the live index despite a
|
|
410
|
-
// working guard in the walk.
|
|
411
|
-
//
|
|
412
|
-
// Measured on this repo 2026-07-28: 96 `._` entries in the live index, 91 of
|
|
413
|
-
// them shadowing a real document of the same name and competing with it for
|
|
414
|
-
// top-k slots. That is a direct Recall@5/MRR@10 loss, not wasted storage.
|
|
415
|
-
if (isResourceFork(resolved)) {
|
|
416
|
-
return {
|
|
417
|
-
filePath: resolved,
|
|
418
|
-
chunksIndexed: 0,
|
|
419
|
-
scope,
|
|
420
|
-
skipped: true,
|
|
421
|
-
error: 'AppleDouble resource fork',
|
|
422
|
-
};
|
|
423
|
-
}
|
|
424
|
-
if (!DOC_EXTENSIONS.has(ext)) {
|
|
425
|
-
return {
|
|
426
|
-
filePath: resolved,
|
|
427
|
-
chunksIndexed: 0,
|
|
428
|
-
scope,
|
|
429
|
-
skipped: true,
|
|
430
|
-
error: `unsupported extension: ${ext}`,
|
|
431
|
-
};
|
|
432
|
-
}
|
|
433
|
-
if (!fs.existsSync(resolved)) {
|
|
434
|
-
return { filePath: resolved, chunksIndexed: 0, scope, skipped: true, error: 'file not found' };
|
|
435
|
-
}
|
|
436
|
-
const stat = fs.statSync(resolved);
|
|
437
|
-
if (stat.size > MAX_FILE_SIZE) {
|
|
438
|
-
return {
|
|
439
|
-
filePath: resolved,
|
|
440
|
-
chunksIndexed: 0,
|
|
441
|
-
scope,
|
|
442
|
-
skipped: true,
|
|
443
|
-
error: 'file too large (>50MB)',
|
|
444
|
-
};
|
|
445
|
-
}
|
|
446
|
-
rootDir = effectiveRoot(scope, rootDir);
|
|
447
|
-
const meta = _metadataCache ?? readMetadata(rootDir);
|
|
448
|
-
const existing = meta.find((m) => m.filePath === resolved && m.scope === scope);
|
|
449
|
-
let fullContent;
|
|
450
|
-
try {
|
|
451
|
-
const entry = toFileEntry(resolved);
|
|
452
|
-
fullContent = await extractText(entry);
|
|
453
|
-
}
|
|
454
|
-
catch (err) {
|
|
455
|
-
return { filePath: resolved, chunksIndexed: 0, scope, skipped: false, error: String(err) };
|
|
456
|
-
}
|
|
457
|
-
if (!fullContent || fullContent.trim().length === 0) {
|
|
458
|
-
return {
|
|
459
|
-
filePath: resolved,
|
|
460
|
-
chunksIndexed: 0,
|
|
461
|
-
scope,
|
|
462
|
-
skipped: true,
|
|
463
|
-
error: 'no text extracted',
|
|
464
|
-
};
|
|
465
|
-
}
|
|
466
|
-
const hash = contentHash(fullContent);
|
|
467
|
-
if (existing && existing.contentHash === hash) {
|
|
468
|
-
return { filePath: resolved, chunksIndexed: existing.chunkCount, scope, skipped: true };
|
|
469
|
-
}
|
|
470
|
-
// NOTE: the previous version's metadata record is deliberately NOT tombstoned
|
|
471
|
-
// here. `readMetadata` is last-wins per (filePath, scope), so appending the
|
|
472
|
-
// new record below already supersedes the old one — the tombstone was a no-op
|
|
473
|
-
// on the success path and destructive on the failure path: it retired a
|
|
474
|
-
// perfectly good previous index before knowing whether the replacement would
|
|
475
|
-
// land, so a failed re-ingest left the document with NO live version at all.
|
|
476
|
-
const docId = `${scope}:${resolved}`;
|
|
477
|
-
const rawChunks = await chunkDocument(docId, fullContent);
|
|
478
|
-
// monolean: [re-enabled] item 2 shipped 768d gte-modernbert-base — capacity handles enrichment
|
|
479
|
-
const chunks = enrichChunks(rawChunks, fullContent, resolved);
|
|
480
|
-
const bridge = await getBridge();
|
|
481
|
-
let indexed = 0;
|
|
482
|
-
for (const chunk of chunks) {
|
|
483
|
-
const key = `doc:${hash}:${chunk.chunkIndex}`;
|
|
484
|
-
if (bridge) {
|
|
485
|
-
try {
|
|
486
|
-
const storeResult = await bridge.bridgeStoreEntry({
|
|
487
|
-
key,
|
|
488
|
-
value: chunk.text,
|
|
489
|
-
namespace: namespace(scope),
|
|
490
|
-
generateEmbeddingFlag: true,
|
|
491
|
-
tags: ['document', ext, `src:${resolved}`],
|
|
492
|
-
upsert: true,
|
|
493
|
-
dbPath: storeDbPath(scope),
|
|
494
|
-
});
|
|
495
|
-
if (storeResult?.success)
|
|
496
|
-
indexed++;
|
|
497
|
-
}
|
|
498
|
-
catch (e) {
|
|
499
|
-
if (process.env.DEBUG || process.env.MONOMIND_DEBUG)
|
|
500
|
-
console.error(`[ingestDocument] failed to store chunk ${chunk.chunkIndex} of ${resolved}:`, e);
|
|
501
|
-
}
|
|
502
|
-
}
|
|
503
|
-
}
|
|
504
|
-
// Commit the document version ONLY when EVERY chunk stored. Recording the
|
|
505
|
-
// content hash after a partial store was the worse half of this bug: the
|
|
506
|
-
// hash check above then skipped the file on every future ingest, so the
|
|
507
|
-
// chunks that failed were never retried — a permanently, silently
|
|
508
|
-
// half-indexed document feeding knowledge retrieval with no signal at all.
|
|
509
|
-
// (Total failure was already handled; partial success was not.)
|
|
510
|
-
//
|
|
511
|
-
// Not committing is what makes a retry work: chunk keys are
|
|
512
|
-
// `doc:<contentHash>:<index>` and stores are upserts, so re-ingesting the
|
|
513
|
-
// same bytes rewrites the same keys and fills the gaps. Until it succeeds the
|
|
514
|
-
// partially-written chunks sit under a hash that is not live, and superseded
|
|
515
|
-
// filtering keeps them out of search (see `liveContentHashes`).
|
|
516
|
-
const complete = indexed === chunks.length;
|
|
517
|
-
if (complete) {
|
|
518
|
-
appendMetadata(rootDir, {
|
|
519
|
-
filePath: resolved,
|
|
520
|
-
contentHash: hash,
|
|
521
|
-
chunkCount: indexed,
|
|
522
|
-
indexedAt: new Date().toISOString(),
|
|
523
|
-
scope,
|
|
524
|
-
size: stat.size,
|
|
525
|
-
});
|
|
526
|
-
}
|
|
527
|
-
return {
|
|
528
|
-
filePath: resolved,
|
|
529
|
-
chunksIndexed: indexed,
|
|
530
|
-
scope,
|
|
531
|
-
skipped: false,
|
|
532
|
-
...(complete
|
|
533
|
-
? {}
|
|
534
|
-
: indexed > 0
|
|
535
|
-
? {
|
|
536
|
-
partial: true,
|
|
537
|
-
error: `partial store: ${indexed}/${chunks.length} chunks — version not committed, re-ingest to repair`,
|
|
538
|
-
}
|
|
539
|
-
: {
|
|
540
|
-
error: bridge
|
|
541
|
-
? 'all chunk stores failed'
|
|
542
|
-
: 'memory bridge unavailable — nothing indexed',
|
|
543
|
-
}),
|
|
544
|
-
};
|
|
545
|
-
}
|
|
546
|
-
export async function ingestDirectory(dirPath, scope = 'shared', opts) {
|
|
547
|
-
const scanDir = path.resolve(dirPath);
|
|
548
|
-
const rootDir = path.resolve(opts?.rootDir ?? getProjectRoot());
|
|
549
|
-
const files = [];
|
|
550
|
-
function walk(dir, depth = 0) {
|
|
551
|
-
if (depth > 10)
|
|
552
|
-
return;
|
|
553
|
-
let entries;
|
|
554
|
-
try {
|
|
555
|
-
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
556
|
-
}
|
|
557
|
-
catch {
|
|
558
|
-
return;
|
|
559
|
-
}
|
|
560
|
-
for (const entry of entries) {
|
|
561
|
-
// Skip dotfiles/dot-dirs (incl. exFAT `._*` junk) — except `.monodesign`,
|
|
562
|
-
// whose critique snapshots are markdown worth surfacing in the Second Brain.
|
|
563
|
-
if (entry.name.startsWith('.') && entry.name !== '.monodesign')
|
|
564
|
-
continue;
|
|
565
|
-
const full = path.join(dir, entry.name);
|
|
566
|
-
if (entry.isDirectory()) {
|
|
567
|
-
if (!IGNORE_DIRS.has(entry.name))
|
|
568
|
-
walk(full, depth + 1);
|
|
569
|
-
}
|
|
570
|
-
else if (entry.isFile()) {
|
|
571
|
-
const ext = path.extname(entry.name).toLowerCase();
|
|
572
|
-
if (DOC_EXTENSIONS.has(ext))
|
|
573
|
-
files.push(full);
|
|
574
|
-
}
|
|
575
|
-
}
|
|
576
|
-
}
|
|
577
|
-
walk(scanDir);
|
|
578
|
-
const metadataCache = readMetadata(rootDir);
|
|
579
|
-
const result = {
|
|
580
|
-
filesProcessed: 0,
|
|
581
|
-
filesSkipped: 0,
|
|
582
|
-
totalChunks: 0,
|
|
583
|
-
errors: [],
|
|
584
|
-
results: [],
|
|
585
|
-
};
|
|
586
|
-
for (let i = 0; i < files.length; i++) {
|
|
587
|
-
opts?.onProgress?.(files[i], i, files.length);
|
|
588
|
-
const r = await ingestDocument(files[i], scope, rootDir, metadataCache);
|
|
589
|
-
result.results.push(r);
|
|
590
|
-
if (r.skipped) {
|
|
591
|
-
result.filesSkipped++;
|
|
592
|
-
}
|
|
593
|
-
else {
|
|
594
|
-
result.filesProcessed++;
|
|
595
|
-
result.totalChunks += r.chunksIndexed;
|
|
596
|
-
}
|
|
597
|
-
if (r.error && !r.skipped) {
|
|
598
|
-
result.errors.push(`${r.filePath}: ${r.error}`);
|
|
599
|
-
}
|
|
600
|
-
}
|
|
601
|
-
return result;
|
|
602
|
-
}
|
|
603
|
-
// ── Search ─────────────────────────────────────────────────────────
|
|
604
|
-
/** Small additive boost so project knowledge wins ties against the global
|
|
605
|
-
* brain — local context is more likely to be what the user means. */
|
|
606
|
-
const PROJECT_SCOPE_BOOST = 0.05;
|
|
607
|
-
// ── Superseded-version filtering ───────────────────────────────────
|
|
608
|
-
//
|
|
609
|
-
// Chunk keys are `doc:<contentHash>:<chunkIndex>`. Re-ingesting a changed file
|
|
610
|
-
// produces a NEW contentHash, so its chunks land under new keys — the previous
|
|
611
|
-
// version's rows are never touched (`removeDocument` only tombstones metadata;
|
|
612
|
-
// the bridge exposes no delete-by-prefix). The store therefore accumulates every
|
|
613
|
-
// version a document has ever had, and all of them stay searchable.
|
|
614
|
-
//
|
|
615
|
-
// Measured on this repo's own store (2026-07-26): 9,067 `doc:`-keyed rows in
|
|
616
|
-
// `knowledge:shared` spanning 798 distinct content hashes, of which only 139
|
|
617
|
-
// are current — 8,542 rows (94.2%) are orphaned older versions.
|
|
618
|
-
//
|
|
619
|
-
// Nothing is deleted here. The current-hash set from doc-metadata.jsonl is used
|
|
620
|
-
// to decide what search RETURNS; `includeSuperseded` puts the old versions back
|
|
621
|
-
// (flagged `superseded: true`) for anyone who wants document history.
|
|
622
|
-
/** Content hashes of the documents currently indexed under `rootDir`. */
|
|
623
|
-
export function liveContentHashes(rootDir) {
|
|
624
|
-
const live = new Set();
|
|
625
|
-
for (const m of readMetadata(rootDir))
|
|
626
|
-
if (m.contentHash)
|
|
627
|
-
live.add(m.contentHash);
|
|
628
|
-
return live;
|
|
629
|
-
}
|
|
630
|
-
/** True when a metadata log exists under `rootDir`.
|
|
631
|
-
*
|
|
632
|
-
* An empty live-hash set has two very different causes: the log is missing (we
|
|
633
|
-
* cannot judge what is current) or the log exists and every document has been
|
|
634
|
-
* removed (nothing is current). Collapsing them made `doc remove` of the LAST
|
|
635
|
-
* document a no-op — the tombstoned chunks came straight back in search.
|
|
636
|
-
*
|
|
637
|
-
* Reads the path directly instead of via `metadataPath`, which mkdir's. */
|
|
638
|
-
export function hasKnowledgeMetadata(rootDir) {
|
|
639
|
-
return fs.existsSync(path.join(rootDir, '.monomind', 'knowledge', METADATA_FILE));
|
|
640
|
-
}
|
|
641
|
-
/**
|
|
642
|
-
* True when `key` is a document chunk whose version is no longer current.
|
|
643
|
-
* Non-`doc:` keys are never superseded. When no metadata is available nothing
|
|
644
|
-
* is filtered, because "no metadata" must not read as "everything is stale".
|
|
645
|
-
*
|
|
646
|
-
* `metadataPresent` defaults to the old `live.size > 0` heuristic so existing
|
|
647
|
-
* two-argument callers keep their exact behaviour; pass `hasKnowledgeMetadata`
|
|
648
|
-
* to also filter correctly once the last document has been removed.
|
|
649
|
-
*/
|
|
650
|
-
export function isSupersededKey(key, live, metadataPresent = live.size > 0) {
|
|
651
|
-
if (!key?.startsWith('doc:'))
|
|
652
|
-
return false;
|
|
653
|
-
if (!metadataPresent)
|
|
654
|
-
return false;
|
|
655
|
-
return !live.has(key.split(':')[1] ?? '');
|
|
656
|
-
}
|
|
657
|
-
/** How many rows to ask the backend for per requested result when superseded
|
|
658
|
-
* filtering is active — most rows in a long-lived store are old versions, so
|
|
659
|
-
* a 1:1 fetch would return an almost-empty page. */
|
|
660
|
-
const SUPERSEDED_OVERFETCH = 20;
|
|
661
|
-
const SUPERSEDED_OVERFETCH_CAP = 300;
|
|
662
|
-
export function supersededOverfetchLimit(limit, live) {
|
|
663
|
-
if (live.size === 0)
|
|
664
|
-
return limit;
|
|
665
|
-
return Math.min(Math.max(limit * SUPERSEDED_OVERFETCH, limit), SUPERSEDED_OVERFETCH_CAP);
|
|
666
|
-
}
|
|
667
|
-
export async function searchKnowledge(query, opts) {
|
|
668
|
-
const bridge = await getBridge();
|
|
669
|
-
if (!bridge)
|
|
670
|
-
return [];
|
|
671
|
-
const scope = opts?.scope ?? 'shared';
|
|
672
|
-
const limit = opts?.limit ?? 10;
|
|
673
|
-
const minScore = opts?.minScore ?? 0.3;
|
|
674
|
-
const store = opts?.store ?? 'all';
|
|
675
|
-
const targets = [];
|
|
676
|
-
if (store !== 'global') {
|
|
677
|
-
targets.push({
|
|
678
|
-
ns: namespace(scope),
|
|
679
|
-
root: opts?.rootDir ?? getProjectRoot(),
|
|
680
|
-
label: scope,
|
|
681
|
-
boost: PROJECT_SCOPE_BOOST,
|
|
682
|
-
});
|
|
683
|
-
}
|
|
684
|
-
if (store !== 'project') {
|
|
685
|
-
targets.push({
|
|
686
|
-
ns: namespace('global'),
|
|
687
|
-
dbPath: GLOBAL_BRAIN_SENTINEL,
|
|
688
|
-
root: globalBrainRoot(),
|
|
689
|
-
label: 'global',
|
|
690
|
-
boost: 0,
|
|
691
|
-
});
|
|
692
|
-
}
|
|
693
|
-
const includeSuperseded = opts?.includeSuperseded === true;
|
|
694
|
-
const perTarget = await Promise.all(targets.map(async (t) => {
|
|
695
|
-
const meta = readMetadata(t.root);
|
|
696
|
-
const hasMeta = hasKnowledgeMetadata(t.root);
|
|
697
|
-
const live = new Set();
|
|
698
|
-
for (const m of meta)
|
|
699
|
-
if (m.contentHash)
|
|
700
|
-
live.add(m.contentHash);
|
|
701
|
-
// Old versions dominate a long-lived store, so a 1:1 fetch would come back
|
|
702
|
-
// nearly empty once they are filtered out. Over-fetch, then trim.
|
|
703
|
-
const fetchLimit = includeSuperseded ? limit : supersededOverfetchLimit(limit, live);
|
|
704
|
-
const result = await bridge
|
|
705
|
-
.bridgeSearchEntries({
|
|
706
|
-
query,
|
|
707
|
-
namespace: t.ns,
|
|
708
|
-
limit: fetchLimit,
|
|
709
|
-
threshold: minScore,
|
|
710
|
-
dbPath: t.dbPath,
|
|
711
|
-
skipRerank: opts?.skipRerank,
|
|
712
|
-
includeSuperseded,
|
|
713
|
-
rootDir: t.root,
|
|
714
|
-
})
|
|
715
|
-
.catch(() => null);
|
|
716
|
-
if (!result?.success || !result.results.length)
|
|
717
|
-
return [];
|
|
718
|
-
const hashToFile = new Map();
|
|
719
|
-
for (const m of meta)
|
|
720
|
-
hashToFile.set(m.contentHash, m.filePath);
|
|
721
|
-
const kept = includeSuperseded
|
|
722
|
-
? result.results
|
|
723
|
-
: result.results.filter((r) => !isSupersededKey(String(r.key ?? ''), live, hasMeta));
|
|
724
|
-
return kept.slice(0, limit).map((r) => {
|
|
725
|
-
const parts = r.key.startsWith('doc:') ? r.key.split(':') : [];
|
|
726
|
-
const hash = parts[1] ?? '';
|
|
727
|
-
const idx = parseInt(parts[2] ?? '0', 10);
|
|
728
|
-
// The src: tag stored at ingest is the chunk's OWN provenance — the
|
|
729
|
-
// hash→file map can misattribute when two documents share identical
|
|
730
|
-
// content, and goes empty when a re-ingested file's hash changed.
|
|
731
|
-
const srcTag = (r.tags ?? []).find((tag) => tag.startsWith('src:'));
|
|
732
|
-
const superseded = includeSuperseded && isSupersededKey(String(r.key ?? ''), live, hasMeta);
|
|
733
|
-
return {
|
|
734
|
-
id: r.id,
|
|
735
|
-
filePath: srcTag ? srcTag.slice(4) : (hashToFile.get(hash) ?? ''),
|
|
736
|
-
// Serve the head of the chunk only — chunks are heading-anchored, so the
|
|
737
|
-
// head carries the most relevant text, and full chunks (up to ~3.2K
|
|
738
|
-
// chars) bloat every search response.
|
|
739
|
-
text: typeof r.content === 'string' && r.content.length > SEARCH_EXCERPT_TEXT_CAP
|
|
740
|
-
? r.content.slice(0, SEARCH_EXCERPT_TEXT_CAP)
|
|
741
|
-
: r.content,
|
|
742
|
-
similarity: r.score + t.boost,
|
|
743
|
-
chunkIndex: Number.isNaN(idx) ? 0 : idx,
|
|
744
|
-
scope: t.label,
|
|
745
|
-
...(superseded ? { superseded: true } : {}),
|
|
746
|
-
};
|
|
747
|
-
});
|
|
748
|
-
}));
|
|
749
|
-
return perTarget
|
|
750
|
-
.flat()
|
|
751
|
-
.sort((a, b) => b.similarity - a.similarity)
|
|
752
|
-
.slice(0, limit);
|
|
753
|
-
}
|
|
754
|
-
// ── List / Remove ──────────────────────────────────────────────────
|
|
755
|
-
export function listDocuments(rootDir = getProjectRoot(), scope) {
|
|
756
|
-
const all = readMetadata(rootDir);
|
|
757
|
-
return scope ? all.filter((m) => m.scope === scope) : all;
|
|
758
|
-
}
|
|
759
|
-
export async function removeDocument(filePath, scope = 'shared', rootDir = getProjectRoot()) {
|
|
760
|
-
removeMetadataEntry(rootDir, path.resolve(filePath), scope);
|
|
761
|
-
// SQLite cleanup: bridge doesn't expose delete-by-key, so metadata removal is sufficient.
|
|
762
|
-
// Orphaned SQLite entries get swept on next full re-index or TTL expiry.
|
|
763
|
-
}
|
|
764
|
-
// ── Filesystem reconciliation (item 4b-i) ──────────────────────────
|
|
765
|
-
/**
|
|
766
|
-
* True for macOS AppleDouble sidecars (`._name`).
|
|
767
|
-
*
|
|
768
|
-
* Matches on the BASENAME PREFIX only. A legitimate document may contain `._`
|
|
769
|
-
* elsewhere in its name (`v1._2-release.md`), or live under a dot-directory
|
|
770
|
-
* that is deliberately indexed (`.monodesign/` critique snapshots), and
|
|
771
|
-
* neither may be rejected.
|
|
772
|
-
*/
|
|
773
|
-
export function isResourceFork(filePath) {
|
|
774
|
-
return path.basename(filePath).startsWith('._');
|
|
775
|
-
}
|
|
776
|
-
/**
|
|
777
|
-
* Reconcile the document index against the filesystem: find index entries whose
|
|
778
|
-
* source file no longer exists and, only when explicitly asked, tombstone them.
|
|
779
|
-
*
|
|
780
|
-
* WHY — `removeDocument` only ever tombstoned metadata, and nothing has ever
|
|
781
|
-
* compared the index against the disk, so a deleted file stayed searchable
|
|
782
|
-
* forever. Measured 2026-07-28: 109 of 257 live entries (42.4%) had no file
|
|
783
|
-
* behind them, including `docs/concepts/memory.md`. The Second Brain was
|
|
784
|
-
* answering questions from documents the user had deleted.
|
|
785
|
-
*
|
|
786
|
-
* WHY IT IS THIS CAUTIOUS — "drop the index entry when the file is missing" is
|
|
787
|
-
* a rule with a known catastrophic reading. A missing file is also an unmounted
|
|
788
|
-
* volume, a checked-out branch, a partial clone, or a permissions failure. Two
|
|
789
|
-
* guards were tried against real data and REJECTED; they are recorded here so
|
|
790
|
-
* they are not re-proposed:
|
|
791
|
-
*
|
|
792
|
-
* - "abort if >50% of entries are missing" — the real, legitimate missing
|
|
793
|
-
* fraction was 42.4%, so the threshold never fires in the one case we have.
|
|
794
|
-
* Any threshold that would have blocked this reconcile is fitted to nothing.
|
|
795
|
-
* - "only reconcile when the parent directory still exists" — 26 of the 109
|
|
796
|
-
* missing files had no parent directory, because `docs/concepts`,
|
|
797
|
-
* `docs/adrs` and `docs/commands` were legitimately deleted wholesale. A
|
|
798
|
-
* deleted directory and an unmounted volume are indistinguishable there.
|
|
799
|
-
*
|
|
800
|
-
* What does discriminate is the ROOT. An intact, readable root carrying a
|
|
801
|
-
* metadata log means the tree is genuinely present, so a missing file is
|
|
802
|
-
* genuinely gone. A missing root means nothing beneath it is knowable and
|
|
803
|
-
* nothing may be removed — hence throw rather than reconcile.
|
|
804
|
-
*
|
|
805
|
-
* Removal tombstones metadata; it does not delete store rows. Chunks stay on
|
|
806
|
-
* disk and fall out of search through the existing superseded filter, which
|
|
807
|
-
* keeps this consistent with the mark-don't-destroy rule and leaves the whole
|
|
808
|
-
* operation reversible from the archive.
|
|
809
|
-
*/
|
|
810
|
-
export async function reconcileIndex(rootDir = getProjectRoot(), opts) {
|
|
811
|
-
const apply = opts?.apply === true;
|
|
812
|
-
// Root guard — the unmounted-volume case. Every file below a missing root
|
|
813
|
-
// looks deleted, so this must abort rather than reconcile.
|
|
814
|
-
if (!rootDir || !fs.existsSync(rootDir)) {
|
|
815
|
-
throw new Error(`reconcileIndex: project root does not exist: ${rootDir} — refusing to reconcile ` +
|
|
816
|
-
`(an unmounted volume makes every indexed file look deleted)`);
|
|
817
|
-
}
|
|
818
|
-
if (!hasKnowledgeMetadata(rootDir)) {
|
|
819
|
-
throw new Error(`reconcileIndex: no knowledge metadata log under ${rootDir} — refusing to reconcile ` +
|
|
820
|
-
`("no metadata" must not read as "everything is stale")`);
|
|
821
|
-
}
|
|
822
|
-
const records = readMetadata(rootDir).filter((m) => !opts?.scope || m.scope === opts.scope);
|
|
823
|
-
const missing = records.filter((m) => !fs.existsSync(m.filePath));
|
|
824
|
-
if (!apply || missing.length === 0) {
|
|
825
|
-
return { missing, scanned: records.length, applied: apply, removed: 0 };
|
|
826
|
-
}
|
|
827
|
-
// Archive BEFORE removing, inside the operation so no caller can bypass it
|
|
828
|
-
// by forgetting — the same precondition rule the delete path uses.
|
|
829
|
-
const dir = path.join(rootDir, '.monomind', 'knowledge', 'archive');
|
|
830
|
-
fs.mkdirSync(dir, { recursive: true });
|
|
831
|
-
const stamp = new Date().toISOString().replace(/[:.]/g, '-');
|
|
832
|
-
const archivePath = path.join(dir, `reconcile-${stamp}.jsonl`);
|
|
833
|
-
fs.writeFileSync(archivePath, `${missing.map((m) => JSON.stringify(m)).join('\n')}\n`, 'utf-8');
|
|
834
|
-
let removed = 0;
|
|
835
|
-
for (const m of missing) {
|
|
836
|
-
removeMetadataEntry(rootDir, m.filePath, m.scope);
|
|
837
|
-
removed++;
|
|
838
|
-
}
|
|
839
|
-
return { missing, scanned: records.length, applied: true, removed, archivePath };
|
|
840
|
-
}
|
|
841
|
-
// ── OKF Export ─────────────────────────────────────────────────────
|
|
842
|
-
export async function exportToOKF(outputDir, rootDir = getProjectRoot(), scope = 'shared') {
|
|
843
|
-
const docs = listDocuments(rootDir, scope);
|
|
844
|
-
fs.mkdirSync(outputDir, { recursive: true });
|
|
845
|
-
let exported = 0;
|
|
846
|
-
const indexEntries = [];
|
|
847
|
-
for (const doc of docs) {
|
|
848
|
-
// Read original content
|
|
849
|
-
let content = '';
|
|
850
|
-
try {
|
|
851
|
-
if (fs.existsSync(doc.filePath)) {
|
|
852
|
-
const entry = toFileEntry(doc.filePath);
|
|
853
|
-
content = await extractText(entry);
|
|
854
|
-
}
|
|
855
|
-
}
|
|
856
|
-
catch {
|
|
857
|
-
continue;
|
|
858
|
-
}
|
|
859
|
-
if (!content)
|
|
860
|
-
continue;
|
|
861
|
-
const title = path.basename(doc.filePath, path.extname(doc.filePath));
|
|
862
|
-
const ext = path.extname(doc.filePath).toLowerCase();
|
|
863
|
-
const relativePath = path.relative(rootDir, doc.filePath);
|
|
864
|
-
const slug = title.replace(/[^a-zA-Z0-9._-]+/g, '-').toLowerCase();
|
|
865
|
-
const outFile = path.join(outputDir, `${slug}.md`);
|
|
866
|
-
const yamlEscape = (s) => /[:"'[\]{}#&*!|>%@`]/.test(s) ? `"${s.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"` : s;
|
|
867
|
-
const frontmatter = [
|
|
868
|
-
'---',
|
|
869
|
-
`type: Document`,
|
|
870
|
-
`title: ${yamlEscape(title)}`,
|
|
871
|
-
`description: ${yamlEscape(`Extracted from ${path.basename(doc.filePath)}`)}`,
|
|
872
|
-
`resource: ${yamlEscape(relativePath)}`,
|
|
873
|
-
`tags: ["document", ${yamlEscape(ext.slice(1))}]`,
|
|
874
|
-
`timestamp: ${yamlEscape(doc.indexedAt)}`,
|
|
875
|
-
`contentHash: ${yamlEscape(doc.contentHash)}`,
|
|
876
|
-
`chunkCount: ${doc.chunkCount}`,
|
|
877
|
-
'---',
|
|
878
|
-
'',
|
|
879
|
-
].join('\n');
|
|
880
|
-
fs.writeFileSync(outFile, frontmatter + content, 'utf-8');
|
|
881
|
-
indexEntries.push(`* [${title}](${slug}.md) - ${path.basename(doc.filePath)} (${doc.chunkCount} chunks)`);
|
|
882
|
-
exported++;
|
|
883
|
-
}
|
|
884
|
-
// Write index.md
|
|
885
|
-
const indexContent = [
|
|
886
|
-
`# Knowledge Bundle`,
|
|
887
|
-
'',
|
|
888
|
-
`Exported from monomind on ${new Date().toISOString().slice(0, 10)}`,
|
|
889
|
-
'',
|
|
890
|
-
...indexEntries,
|
|
891
|
-
'',
|
|
892
|
-
].join('\n');
|
|
893
|
-
fs.writeFileSync(path.join(outputDir, 'index.md'), indexContent, 'utf-8');
|
|
894
|
-
return { exported, outputDir };
|
|
895
|
-
}
|
|
896
|
-
// ── OKF Import ─────────────────────────────────────────────────────
|
|
897
|
-
export async function importFromOKF(bundleDir, scope = 'shared', rootDir = getProjectRoot()) {
|
|
898
|
-
const resolved = path.resolve(bundleDir);
|
|
899
|
-
const files = fs
|
|
900
|
-
.readdirSync(resolved)
|
|
901
|
-
.filter((f) => f.endsWith('.md') && f !== 'index.md' && f !== 'log.md')
|
|
902
|
-
.map((f) => path.join(resolved, f));
|
|
903
|
-
const result = {
|
|
904
|
-
filesProcessed: 0,
|
|
905
|
-
filesSkipped: 0,
|
|
906
|
-
totalChunks: 0,
|
|
907
|
-
errors: [],
|
|
908
|
-
results: [],
|
|
909
|
-
};
|
|
910
|
-
for (const file of files) {
|
|
911
|
-
const r = await ingestDocument(file, scope, rootDir);
|
|
912
|
-
result.results.push(r);
|
|
913
|
-
if (r.skipped) {
|
|
914
|
-
result.filesSkipped++;
|
|
915
|
-
}
|
|
916
|
-
else {
|
|
917
|
-
result.filesProcessed++;
|
|
918
|
-
result.totalChunks += r.chunksIndexed;
|
|
919
|
-
}
|
|
920
|
-
if (r.error && !r.skipped)
|
|
921
|
-
result.errors.push(`${r.filePath}: ${r.error}`);
|
|
922
|
-
}
|
|
923
|
-
return result;
|
|
924
|
-
}
|
|
24
|
+
export { chunkSpans } from './document-chunking.js';
|
|
25
|
+
export { findDocumentRecord, hasKnowledgeMetadata, isResourceFork, isSupersededKey, listDocuments, listDocumentVersions, liveContentHashes, reconcileIndex, removeDocument, supersededOverfetchLimit, } from './document-index.js';
|
|
26
|
+
export { ingestDirectory, ingestDocument } from './document-ingest.js';
|
|
27
|
+
export { searchKnowledge } from './document-search.js';
|
|
28
|
+
export { getKnowledgeRoot } from './document-store.js';
|
|
29
|
+
export { exportToOKF, importFromOKF } from './okf-bundle.js';
|
|
925
30
|
//# sourceMappingURL=document-pipeline.js.map
|