@tekmidian/pai 0.66.1 → 0.66.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{aibroker-client-B8c42Lh8.mjs → aibroker-client-CHEEvJZW.mjs} +2 -2
- package/dist/{aibroker-client-C5Fw7DNz.mjs → aibroker-client-Dfv7j1Od.mjs} +3 -3
- package/dist/{aibroker-client-C5Fw7DNz.mjs.map → aibroker-client-Dfv7j1Od.mjs.map} +1 -1
- package/dist/{auto-route-o0BOsXn7.mjs → auto-route-BnizyALK.mjs} +65 -4
- package/dist/auto-route-BnizyALK.mjs.map +1 -0
- package/dist/auto-route-D6fW1Q8z.mjs +3 -0
- package/dist/{chain-CZfKIEq9.mjs → chain-D5876UQX.mjs} +3 -4
- package/dist/{chain-CZfKIEq9.mjs.map → chain-D5876UQX.mjs.map} +1 -1
- package/dist/cli/index.mjs +18 -23
- package/dist/cli/index.mjs.map +1 -1
- package/dist/cli/program.mjs +17 -22
- package/dist/{config-BbLFD7Uf.mjs → config-C-LiGlop.mjs} +2 -2
- package/dist/{config-BbLFD7Uf.mjs.map → config-C-LiGlop.mjs.map} +1 -1
- package/dist/{config-YinjgXEJ.mjs → config-DKIA7wgF.mjs} +1 -1
- package/dist/{context-handover-cache-pkHzmL3c.mjs → context-handover-cache-9PGvXRIw.mjs} +984 -10
- package/dist/context-handover-cache-9PGvXRIw.mjs.map +1 -0
- package/dist/daemon/index.mjs +17 -19
- package/dist/daemon/index.mjs.map +1 -1
- package/dist/daemon-BUgNvMUh.mjs +5029 -0
- package/dist/daemon-BUgNvMUh.mjs.map +1 -0
- package/dist/daemon-DL0eglMJ.mjs +18 -0
- package/dist/daemon-mcp/index.mjs +12 -9
- package/dist/daemon-mcp/index.mjs.map +1 -1
- package/dist/{embeddings-CcscYWwk.mjs → embeddings-CEBGrzwu.mjs} +1 -1
- package/dist/{embeddings-Bx3q0QOY.mjs → embeddings-DOLZnT1X.mjs} +1 -1
- package/dist/{embeddings-Bx3q0QOY.mjs.map → embeddings-DOLZnT1X.mjs.map} +1 -1
- package/dist/{env-JNEIrQWg.mjs → env-DiolswKQ.mjs} +1 -1
- package/dist/{env-JNEIrQWg.mjs.map → env-DiolswKQ.mjs.map} +1 -1
- package/dist/factory-DrBg24UT.mjs +8 -0
- package/dist/factory-Ypf8r0fK.mjs +2892 -0
- package/dist/factory-Ypf8r0fK.mjs.map +1 -0
- package/dist/{fallback-CIiq2Raw.mjs → fallback-D51R4shY.mjs} +23 -11
- package/dist/fallback-D51R4shY.mjs.map +1 -0
- package/dist/{chunker-BH4i-F2b.mjs → helpers-CZsi_49C.mjs} +221 -2
- package/dist/helpers-CZsi_49C.mjs.map +1 -0
- package/dist/hooks/worker-status-line.mjs.map +2 -2
- package/dist/index.mjs +6 -7
- package/dist/{ipc-client-D16Xw6Uo.mjs → ipc-client-BLYX51nG.mjs} +2 -2
- package/dist/{ipc-client-D16Xw6Uo.mjs.map → ipc-client-BLYX51nG.mjs.map} +1 -1
- package/dist/main-resolver-BYgP86C1.mjs +7 -0
- package/dist/{main-resolver-DPPwHNtn.mjs → main-resolver-kVkMJZUD.mjs} +6 -6
- package/dist/{main-resolver-DPPwHNtn.mjs.map → main-resolver-kVkMJZUD.mjs.map} +1 -1
- package/dist/{migrate-BD7D8EEh.mjs → migrate-Bq0esMOa.mjs} +2 -2
- package/dist/{migrate-BD7D8EEh.mjs.map → migrate-Bq0esMOa.mjs.map} +1 -1
- package/dist/{pai-marker-D1MMswkz.mjs → pai-marker-DXVpFsYz.mjs} +1 -1
- package/dist/{pai-marker-D1MMswkz.mjs.map → pai-marker-DXVpFsYz.mjs.map} +1 -1
- package/dist/{planner-Dm5Acyo1.mjs → planner-D5yt0EfA.mjs} +10 -7
- package/dist/{planner-Dm5Acyo1.mjs.map → planner-D5yt0EfA.mjs.map} +1 -1
- package/dist/postgres-BmLr0MUm.mjs +5 -0
- package/dist/{postgres-Ceqsa64C.mjs → postgres-D3xc2RB4.mjs} +364 -9
- package/dist/postgres-D3xc2RB4.mjs.map +1 -0
- package/dist/{program-JJ0TwSHI.mjs → program-DIXRtAFy.mjs} +77 -70
- package/dist/program-DIXRtAFy.mjs.map +1 -0
- package/dist/{query-feedback-BIaZTTFO.mjs → query-feedback-B7FYE4JR.mjs} +1 -1
- package/dist/{query-feedback-BIaZTTFO.mjs.map → query-feedback-B7FYE4JR.mjs.map} +1 -1
- package/dist/{reranker-DKv80KO5.mjs → reranker-3lnggwgq.mjs} +1 -1
- package/dist/{reranker-DKv80KO5.mjs.map → reranker-3lnggwgq.mjs.map} +1 -1
- package/dist/{reranker-CFiEzHQu.mjs → reranker-CZ2mP4cf.mjs} +1 -1
- package/dist/router-BK-hFeQ6.mjs +3 -0
- package/dist/{router-a5G7q8_3.mjs → router-BaTbc9VX.mjs} +2 -2
- package/dist/{router-a5G7q8_3.mjs.map → router-BaTbc9VX.mjs.map} +1 -1
- package/dist/{run-BCpbxqA5.mjs → run-CLYTOm0x.mjs} +599 -22
- package/dist/run-CLYTOm0x.mjs.map +1 -0
- package/dist/{run-env-DRcp7A8K.mjs → run-env-Bnmf6bRI.mjs} +2 -2
- package/dist/{run-env-DRcp7A8K.mjs.map → run-env-Bnmf6bRI.mjs.map} +1 -1
- package/dist/{runtime-paths-CHTg3ywb.mjs → runtime-paths-D3FKAm69.mjs} +1 -1
- package/dist/{runtime-paths-CHTg3ywb.mjs.map → runtime-paths-D3FKAm69.mjs.map} +1 -1
- package/dist/{search-BOSphCJ1.mjs → search-CNAGTiJP.mjs} +2 -2
- package/dist/{search-BOSphCJ1.mjs.map → search-CNAGTiJP.mjs.map} +1 -1
- package/dist/{sources-CijVso3n.mjs → sources-CM2g-CLT.mjs} +2 -2
- package/dist/{sources-CijVso3n.mjs.map → sources-CM2g-CLT.mjs.map} +1 -1
- package/dist/{stop-words-BdQuaE9K.mjs → stop-words-DtxaTWU_.mjs} +1 -1
- package/dist/{stop-words-BdQuaE9K.mjs.map → stop-words-DtxaTWU_.mjs.map} +1 -1
- package/dist/{utils-C9HsYDpO.mjs → utils-DddRwMsG.mjs} +2 -2
- package/dist/{utils-C9HsYDpO.mjs.map → utils-DddRwMsG.mjs.map} +1 -1
- package/dist/{zettelkasten-CtGHQWPU.mjs → zettelkasten-BhZMvmoK.mjs} +149 -7
- package/dist/zettelkasten-BhZMvmoK.mjs.map +1 -0
- package/dist/zettelkasten-BwV1pluL.mjs +5 -0
- package/docs/commands/worker.md +2 -0
- package/package.json +1 -1
- package/dist/async-C2Bm_Lal.mjs +0 -300
- package/dist/async-C2Bm_Lal.mjs.map +0 -1
- package/dist/auto-route-o0BOsXn7.mjs.map +0 -1
- package/dist/chunker-BH4i-F2b.mjs.map +0 -1
- package/dist/clusters-BCtD3fbe.mjs +0 -169
- package/dist/clusters-BCtD3fbe.mjs.map +0 -1
- package/dist/context-handover-cache-pkHzmL3c.mjs.map +0 -1
- package/dist/daemon-DXlaVCkw.mjs +0 -1741
- package/dist/daemon-DXlaVCkw.mjs.map +0 -1
- package/dist/daemon-DyORmxZu.mjs +0 -19
- package/dist/detector-BEPFsINR.mjs +0 -3
- package/dist/detector-BRTtAWrM.mjs +0 -65
- package/dist/detector-BRTtAWrM.mjs.map +0 -1
- package/dist/factory-CNpPQQ-5.mjs +0 -249
- package/dist/factory-CNpPQQ-5.mjs.map +0 -1
- package/dist/factory-DY7x73mE.mjs +0 -3
- package/dist/fallback-CIiq2Raw.mjs.map +0 -1
- package/dist/federation-db-BTyoufBh.mjs +0 -139
- package/dist/federation-db-BTyoufBh.mjs.map +0 -1
- package/dist/federation-db-HfIFc7FG.mjs +0 -3
- package/dist/helpers-BRCJg0G3.mjs +0 -223
- package/dist/helpers-BRCJg0G3.mjs.map +0 -1
- package/dist/indexer-backend-DFF2FrYx.mjs +0 -5
- package/dist/indexer-backend-isSLg6yE.mjs +0 -1
- package/dist/kg-entity-DCOcsVFD.mjs +0 -29
- package/dist/kg-entity-DCOcsVFD.mjs.map +0 -1
- package/dist/latent-ideas-wwAXquGe.mjs +0 -191
- package/dist/latent-ideas-wwAXquGe.mjs.map +0 -1
- package/dist/link-boost-CnI7UVMJ.mjs +0 -34
- package/dist/link-boost-CnI7UVMJ.mjs.map +0 -1
- package/dist/main-resolver-BKz_OuEh.mjs +0 -7
- package/dist/merge-2gqRPFu2.mjs +0 -3
- package/dist/merge-DgU9OgZy.mjs +0 -6
- package/dist/merge-DgU9OgZy.mjs.map +0 -1
- package/dist/module-paths-DdRzbkUI.mjs +0 -44
- package/dist/module-paths-DdRzbkUI.mjs.map +0 -1
- package/dist/neighborhood-2FsmzoxG.mjs +0 -114
- package/dist/neighborhood-2FsmzoxG.mjs.map +0 -1
- package/dist/note-context-De63k8na.mjs +0 -106
- package/dist/note-context-De63k8na.mjs.map +0 -1
- package/dist/postgres-Ceqsa64C.mjs.map +0 -1
- package/dist/program-JJ0TwSHI.mjs.map +0 -1
- package/dist/query-feedback-B_iigYj-.mjs +0 -3
- package/dist/registry-db-C7voqML9.mjs +0 -213
- package/dist/registry-db-C7voqML9.mjs.map +0 -1
- package/dist/registry-db-JHPhA8vF.mjs +0 -3
- package/dist/registry-postgres-nOfyBSIW.mjs +0 -795
- package/dist/registry-postgres-nOfyBSIW.mjs.map +0 -1
- package/dist/registry-sqlite-DrCW1aRK.mjs +0 -593
- package/dist/registry-sqlite-DrCW1aRK.mjs.map +0 -1
- package/dist/router-CXUGsv85.mjs +0 -3
- package/dist/run-BCpbxqA5.mjs.map +0 -1
- package/dist/search-Bf3Kub3F.mjs +0 -3
- package/dist/server-DgmAHyFK.mjs +0 -403
- package/dist/server-DgmAHyFK.mjs.map +0 -1
- package/dist/session-keepalive-DxjTHGcK.mjs +0 -582
- package/dist/session-keepalive-DxjTHGcK.mjs.map +0 -1
- package/dist/sqlite-CqsTy6xo.mjs +0 -936
- package/dist/sqlite-CqsTy6xo.mjs.map +0 -1
- package/dist/state-DW8zdweW.mjs +0 -3
- package/dist/state-HyjqTihC.mjs +0 -76
- package/dist/state-HyjqTihC.mjs.map +0 -1
- package/dist/themes-CTaOj3e1.mjs +0 -148
- package/dist/themes-CTaOj3e1.mjs.map +0 -1
- package/dist/tools-BcOKPBtg.mjs +0 -1489
- package/dist/tools-BcOKPBtg.mjs.map +0 -1
- package/dist/tools-C-_l0o8Z.mjs +0 -6
- package/dist/trace-CtV0RO6n.mjs +0 -137
- package/dist/trace-CtV0RO6n.mjs.map +0 -1
- package/dist/utils-BrX9FLeK.mjs +0 -3
- package/dist/vault-indexer-BG1sEVyM.mjs +0 -537
- package/dist/vault-indexer-BG1sEVyM.mjs.map +0 -1
- package/dist/wakeup-D8n33hYV.mjs +0 -335
- package/dist/wakeup-D8n33hYV.mjs.map +0 -1
- package/dist/work-queue-worker-4tLa5gpQ.mjs +0 -567
- package/dist/work-queue-worker-4tLa5gpQ.mjs.map +0 -1
- package/dist/work-queue-worker-rFZvTO3p.mjs +0 -11
- package/dist/zettelkasten-CtGHQWPU.mjs.map +0 -1
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"chunker-BH4i-F2b.mjs","names":[],"sources":["../src/utils/hash.ts","../src/memory/chunker.ts"],"sourcesContent":["/**\n * Shared hashing utilities. Centralises all SHA-256 usage so every module\n * obtains digests through the same function rather than inlining createHash.\n */\n\nimport { createHash } from \"node:crypto\";\n\n/**\n * Compute a SHA-256 hex digest of the given string.\n * Aliased as sha256File for compatibility with existing call-sites that use\n * that name to hash file contents.\n */\nexport function sha256(content: string): string {\n return createHash(\"sha256\").update(content).digest(\"hex\");\n}\n\n/** Alias kept for backwards compatibility with memory/indexer call-sites. */\nexport const sha256File = sha256;\n","/**\n * Markdown text chunker for the PAI memory engine.\n *\n * Splits markdown files into overlapping text segments suitable for BM25\n * full-text indexing. Respects heading boundaries where possible, falling\n * back to paragraph and sentence splitting when sections are large.\n */\n\nimport { sha256 } from \"../utils/hash.js\";\n\n/** Bump whenever chunk text or boundaries change; forces one re-chunk of every file. */\nexport const CHUNKER_VERSION = 2;\n\n/** File-level change-detection hash: content plus chunker version. */\nexport function fileContentHash(content: string, version: number = CHUNKER_VERSION): string {\n return sha256(`chunker v${version}\\n${content}`);\n}\n\nexport interface Chunk {\n text: string;\n startLine: number; // 1-indexed\n endLine: number; // 1-indexed, inclusive\n hash: string; // SHA-256 of text\n /** Ancestor heading titles, outermost first, including the chunk's own section heading. */\n headingPath: string[];\n}\n\nexport interface ChunkOptions {\n /** Approximate maximum tokens per chunk. Default 400. */\n maxTokens?: number;\n /** Overlap in tokens from the previous chunk. Default 80. */\n overlap?: number;\n}\n\nconst DEFAULT_MAX_TOKENS = 400;\nconst DEFAULT_OVERLAP = 80;\n\n/**\n * Approximate token count using a words * 1.3 heuristic.\n * Matches the OpenClaw estimate approach.\n */\nexport function estimateTokens(text: string): number {\n const wordCount = text.split(/\\s+/).filter(Boolean).length;\n return Math.ceil(wordCount * 1.3);\n}\n\n// sha256 imported from utils/hash.ts\n\n// ---------------------------------------------------------------------------\n// Heading parser (the only one — chunker and memory_outline both use it)\n// ---------------------------------------------------------------------------\n\nexport interface Heading {\n level: number; // 1-6\n title: string;\n line: number; // 1-indexed\n}\n\n/**\n * Parse ATX headings (levels 1-6) from lines. Lines inside ``` or ~~~ fences\n * are never headings.\n */\nexport function parseHeadings(lines: string[]): Heading[] {\n const headings: Heading[] = [];\n let fence: { ch: string; len: number } | null = null;\n\n for (let i = 0; i < lines.length; i++) {\n const text = lines[i] ?? \"\";\n const f = /^ {0,3}(`{3,}|~{3,})/.exec(text);\n if (f) {\n const marker = f[1]!;\n if (!fence) fence = { ch: marker[0]!, len: marker.length };\n else if (marker[0] === fence.ch && marker.length >= fence.len && !text.slice(f[0].length).trim()) fence = null;\n continue;\n }\n if (fence) continue;\n const h = /^ {0,3}(#{1,6})\\s+(.*?)(?:\\s+#+)?\\s*$/.exec(text);\n if (h && h[2]) headings.push({ level: h[1]!.length, title: h[2], line: i + 1 });\n }\n return headings;\n}\n\n/** Heading path (outermost first) in effect at each heading, via a level stack. */\nfunction headingPaths(headings: Heading[]): Map<number, string[]> {\n const paths = new Map<number, string[]>();\n const stack: Heading[] = [];\n for (const h of headings) {\n while (stack.length > 0 && stack[stack.length - 1]!.level >= h.level) stack.pop();\n stack.push(h);\n paths.set(h.line, stack.map((x) => x.title));\n }\n return paths;\n}\n\nexport interface OutlineNode {\n level: number;\n title: string;\n startLine: number;\n endLine: number;\n tokens: number;\n children: OutlineNode[];\n}\n\n/**\n * Heading tree of a markdown file. A section ends on the line before the next\n * heading of the same or higher level (or at EOF) and includes its children.\n */\nexport function buildOutline(content: string): OutlineNode[] {\n const lines = content.split(\"\\n\");\n const last = lines[lines.length - 1] === \"\" ? lines.length - 1 : lines.length;\n const headings = parseHeadings(lines);\n const roots: OutlineNode[] = [];\n const stack: OutlineNode[] = [];\n\n headings.forEach((h, i) => {\n const next = headings.slice(i + 1).find((n) => n.level <= h.level);\n const endLine = Math.max(h.line, next ? next.line - 1 : last);\n const node: OutlineNode = {\n level: h.level,\n title: h.title,\n startLine: h.line,\n endLine,\n tokens: estimateTokens(lines.slice(h.line - 1, endLine).join(\"\\n\")),\n children: [],\n };\n while (stack.length > 0 && stack[stack.length - 1]!.level >= h.level) stack.pop();\n (stack.length > 0 ? stack[stack.length - 1]!.children : roots).push(node);\n stack.push(node);\n });\n return roots;\n}\n\n// ---------------------------------------------------------------------------\n// Internal section / paragraph / sentence splitters\n// ---------------------------------------------------------------------------\n\n/**\n * A contiguous block of lines associated with an approximate token count.\n */\ninterface LineBlock {\n lines: Array<{ text: string; lineNo: number }>;\n tokens: number;\n headingPath: string[];\n}\n\n/**\n * Split content into sections delimited by ## or ### headings.\n * Each section starts at its heading line (or at line 1 for a preamble).\n */\nfunction splitBySections(\n lines: Array<{ text: string; lineNo: number }>,\n): LineBlock[] {\n const sections: LineBlock[] = [];\n let current: Array<{ text: string; lineNo: number }> = [];\n let currentPath: string[] = [];\n\n const headings = parseHeadings(lines.map((l) => l.text));\n const paths = headingPaths(headings);\n const splitAt = new Set(headings.filter((h) => h.level <= 3).map((h) => h.line));\n\n const flush = () => {\n const text = current.map((l) => l.text).join(\"\\n\");\n sections.push({ lines: current, tokens: estimateTokens(text), headingPath: currentPath });\n current = [];\n };\n\n for (const line of lines) {\n if (splitAt.has(line.lineNo)) {\n if (current.length > 0) flush();\n currentPath = paths.get(line.lineNo) ?? [];\n }\n current.push(line);\n }\n\n if (current.length > 0) flush();\n\n return sections;\n}\n\n/**\n * Split a LineBlock by double-newline paragraph boundaries.\n */\nfunction splitByParagraphs(block: LineBlock): LineBlock[] {\n const paragraphs: LineBlock[] = [];\n let current: Array<{ text: string; lineNo: number }> = [];\n\n for (const line of block.lines) {\n if (line.text.trim() === \"\" && current.length > 0) {\n // Empty line — potential paragraph boundary\n const text = current.map((l) => l.text).join(\"\\n\");\n paragraphs.push({ lines: [...current], tokens: estimateTokens(text), headingPath: block.headingPath });\n current = [];\n } else {\n current.push(line);\n }\n }\n\n if (current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n paragraphs.push({ lines: current, tokens: estimateTokens(text), headingPath: block.headingPath });\n }\n\n return paragraphs.length > 0 ? paragraphs : [block];\n}\n\n/**\n * Split a LineBlock by sentence boundaries (. ! ?) when even paragraphs are\n * too large. Works character-by-character within joined lines.\n */\nfunction splitBySentences(block: LineBlock, maxTokens: number): LineBlock[] {\n const fullText = block.lines.map((l) => l.text).join(\" \");\n // Very rough sentence split — split on '. ', '! ', '? ' followed by uppercase\n const sentenceRe = /(?<=[.!?])\\s+(?=[A-Z\"'])/g;\n const sentences = fullText.split(sentenceRe);\n\n const result: LineBlock[] = [];\n let accText = \"\";\n // We can't recover exact line numbers inside a single oversized paragraph,\n // so we approximate using the block's start/end lines distributed evenly.\n const startLine = block.lines[0]?.lineNo ?? 1;\n const endLine = block.lines[block.lines.length - 1]?.lineNo ?? startLine;\n const totalLines = endLine - startLine + 1;\n const linesPerSentence = Math.max(1, Math.floor(totalLines / Math.max(1, sentences.length)));\n\n let sentenceIdx = 0;\n let approxLine = startLine;\n\n const flush = () => {\n if (!accText.trim()) return;\n const endApprox = Math.min(approxLine + linesPerSentence - 1, endLine);\n result.push({\n lines: [{ text: accText.trim(), lineNo: approxLine }],\n tokens: estimateTokens(accText),\n headingPath: block.headingPath,\n });\n approxLine = endApprox + 1;\n accText = \"\";\n };\n\n for (const sentence of sentences) {\n sentenceIdx++;\n const candidateText = accText ? accText + \" \" + sentence : sentence;\n if (estimateTokens(candidateText) > maxTokens && accText) {\n flush();\n accText = sentence;\n } else {\n accText = candidateText;\n }\n }\n void sentenceIdx; // used only for iteration count\n flush();\n\n return result.length > 0 ? result : [block];\n}\n\n// ---------------------------------------------------------------------------\n// Overlap helper\n// ---------------------------------------------------------------------------\n\n/**\n * Extract the last `overlapTokens` worth of text from a list of previously\n * emitted chunks to prepend to the next chunk.\n */\nfunction buildOverlapPrefix(\n lastChunk: { text: string; startLine: number; endLine: number } | undefined,\n overlapTokens: number,\n): Array<{ text: string; lineNo: number }> {\n if (overlapTokens <= 0 || !lastChunk) return [];\n\n const lines = lastChunk.text.split(\"\\n\");\n const kept: string[] = [];\n let acc = 0;\n\n for (let i = lines.length - 1; i >= 0; i--) {\n const lineTokens = estimateTokens(lines[i] ?? \"\");\n acc += lineTokens;\n kept.unshift(lines[i] ?? \"\");\n if (acc >= overlapTokens) break;\n }\n\n // Distribute overlap lines across the lastChunk's line range\n const startLine = lastChunk.endLine - kept.length + 1;\n return kept.map((text, idx) => ({ text, lineNo: Math.max(lastChunk.startLine, startLine + idx) }));\n}\n\n// ---------------------------------------------------------------------------\n// Public API\n// ---------------------------------------------------------------------------\n\n/**\n * Chunk a markdown file into overlapping segments for BM25 indexing.\n *\n * Strategy:\n * 1. Split by headings (##, ###) as natural boundaries.\n * 2. If a section exceeds maxTokens, split by paragraphs.\n * 3. If a paragraph still exceeds maxTokens, split by sentences.\n * 4. Apply overlap: each chunk includes the last `overlap` tokens from the\n * previous chunk.\n */\n/**\n * Strip `<private>...</private>` blocks from content before indexing.\n * Content within these tags is excluded from memory — never stored or searched.\n */\nexport function stripPrivateTags(content: string): string {\n return content.replace(/<private>[\\s\\S]*?<\\/private>/gi, \"\");\n}\n\n/** One breadcrumb line (\"[A > B]\\n\") for a non-empty heading path, else \"\". */\nfunction breadcrumb(path: string[]): string {\n return path.length > 0 ? `[${path.join(\" > \")}]\\n` : \"\";\n}\n\nexport function chunkMarkdown(content: string, opts?: ChunkOptions): Chunk[] {\n const maxTokens = opts?.maxTokens ?? DEFAULT_MAX_TOKENS;\n const overlapTokens = opts?.overlap ?? DEFAULT_OVERLAP;\n\n // Strip private content before indexing\n content = stripPrivateTags(content);\n\n if (!content.trim()) return [];\n\n const rawLines = content.split(\"\\n\");\n const lines: Array<{ text: string; lineNo: number }> = rawLines.map((text, idx) => ({\n text,\n lineNo: idx + 1, // 1-indexed\n }));\n\n // Step 1: section split\n const sections = splitBySections(lines);\n\n // Step 2 & 3: further split oversized sections. The breadcrumb line is\n // prepended to every chunk, so it counts against the token budget.\n const finalBlocks: LineBlock[] = [];\n for (const section of sections) {\n const budget = Math.max(1, maxTokens - estimateTokens(breadcrumb(section.headingPath)));\n if (section.tokens <= budget) {\n finalBlocks.push(section);\n continue;\n }\n // Too big — split by paragraphs\n const paras = splitByParagraphs(section);\n for (const para of paras) {\n if (para.tokens <= budget) {\n finalBlocks.push(para);\n continue;\n }\n // Still too big — split by sentences\n const sentences = splitBySentences(para, budget);\n finalBlocks.push(...sentences);\n }\n }\n\n // Step 4: build final chunks with overlap\n const chunks: Chunk[] = [];\n let prev: { text: string; startLine: number; endLine: number } | undefined;\n\n for (const block of finalBlocks) {\n if (block.lines.length === 0) continue;\n\n // Build overlap prefix from the previous chunk's raw (unprefixed) text\n const overlapLines = buildOverlapPrefix(prev, overlapTokens);\n\n // Combine overlap + block lines\n const allLines = [...overlapLines, ...block.lines];\n const raw = allLines.map((l) => l.text).join(\"\\n\").trim();\n\n if (!raw) continue;\n\n const startLine = block.lines[0]?.lineNo ?? 1;\n const endLine = block.lines[block.lines.length - 1]?.lineNo ?? startLine;\n const text = breadcrumb(block.headingPath) + raw;\n\n prev = { text: raw, startLine, endLine };\n chunks.push({\n text,\n startLine,\n endLine,\n hash: sha256(text),\n headingPath: block.headingPath,\n });\n }\n\n return chunks;\n}\n"],"mappings":";;;;;;;;;;;;AAYA,SAAgB,OAAO,SAAyB;AAC9C,QAAO,WAAW,SAAS,CAAC,OAAO,QAAQ,CAAC,OAAO,MAAM;;;AAI3D,MAAa,aAAa;;;;;;;;;;;;ACN1B,MAAa,kBAAkB;;AAG/B,SAAgB,gBAAgB,SAAiB,UAAkB,iBAAyB;AAC1F,QAAO,OAAO,YAAY,QAAQ,IAAI,UAAU;;AAmBlD,MAAM,qBAAqB;AAC3B,MAAM,kBAAkB;;;;;AAMxB,SAAgB,eAAe,MAAsB;CACnD,MAAM,YAAY,KAAK,MAAM,MAAM,CAAC,OAAO,QAAQ,CAAC;AACpD,QAAO,KAAK,KAAK,YAAY,IAAI;;;;;;AAmBnC,SAAgB,cAAc,OAA4B;CACxD,MAAM,WAAsB,EAAE;CAC9B,IAAI,QAA4C;AAEhD,MAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;EACrC,MAAM,OAAO,MAAM,MAAM;EACzB,MAAM,IAAI,uBAAuB,KAAK,KAAK;AAC3C,MAAI,GAAG;GACL,MAAM,SAAS,EAAE;AACjB,OAAI,CAAC,MAAO,SAAQ;IAAE,IAAI,OAAO;IAAK,KAAK,OAAO;IAAQ;YACjD,OAAO,OAAO,MAAM,MAAM,OAAO,UAAU,MAAM,OAAO,CAAC,KAAK,MAAM,EAAE,GAAG,OAAO,CAAC,MAAM,CAAE,SAAQ;AAC1G;;AAEF,MAAI,MAAO;EACX,MAAM,IAAI,wCAAwC,KAAK,KAAK;AAC5D,MAAI,KAAK,EAAE,GAAI,UAAS,KAAK;GAAE,OAAO,EAAE,GAAI;GAAQ,OAAO,EAAE;GAAI,MAAM,IAAI;GAAG,CAAC;;AAEjF,QAAO;;;AAIT,SAAS,aAAa,UAA4C;CAChE,MAAM,wBAAQ,IAAI,KAAuB;CACzC,MAAM,QAAmB,EAAE;AAC3B,MAAK,MAAM,KAAK,UAAU;AACxB,SAAO,MAAM,SAAS,KAAK,MAAM,MAAM,SAAS,GAAI,SAAS,EAAE,MAAO,OAAM,KAAK;AACjF,QAAM,KAAK,EAAE;AACb,QAAM,IAAI,EAAE,MAAM,MAAM,KAAK,MAAM,EAAE,MAAM,CAAC;;AAE9C,QAAO;;;;;;AAgBT,SAAgB,aAAa,SAAgC;CAC3D,MAAM,QAAQ,QAAQ,MAAM,KAAK;CACjC,MAAM,OAAO,MAAM,MAAM,SAAS,OAAO,KAAK,MAAM,SAAS,IAAI,MAAM;CACvE,MAAM,WAAW,cAAc,MAAM;CACrC,MAAM,QAAuB,EAAE;CAC/B,MAAM,QAAuB,EAAE;AAE/B,UAAS,SAAS,GAAG,MAAM;EACzB,MAAM,OAAO,SAAS,MAAM,IAAI,EAAE,CAAC,MAAM,MAAM,EAAE,SAAS,EAAE,MAAM;EAClE,MAAM,UAAU,KAAK,IAAI,EAAE,MAAM,OAAO,KAAK,OAAO,IAAI,KAAK;EAC7D,MAAM,OAAoB;GACxB,OAAO,EAAE;GACT,OAAO,EAAE;GACT,WAAW,EAAE;GACb;GACA,QAAQ,eAAe,MAAM,MAAM,EAAE,OAAO,GAAG,QAAQ,CAAC,KAAK,KAAK,CAAC;GACnE,UAAU,EAAE;GACb;AACD,SAAO,MAAM,SAAS,KAAK,MAAM,MAAM,SAAS,GAAI,SAAS,EAAE,MAAO,OAAM,KAAK;AACjF,GAAC,MAAM,SAAS,IAAI,MAAM,MAAM,SAAS,GAAI,WAAW,OAAO,KAAK,KAAK;AACzE,QAAM,KAAK,KAAK;GAChB;AACF,QAAO;;;;;;AAoBT,SAAS,gBACP,OACa;CACb,MAAM,WAAwB,EAAE;CAChC,IAAI,UAAmD,EAAE;CACzD,IAAI,cAAwB,EAAE;CAE9B,MAAM,WAAW,cAAc,MAAM,KAAK,MAAM,EAAE,KAAK,CAAC;CACxD,MAAM,QAAQ,aAAa,SAAS;CACpC,MAAM,UAAU,IAAI,IAAI,SAAS,QAAQ,MAAM,EAAE,SAAS,EAAE,CAAC,KAAK,MAAM,EAAE,KAAK,CAAC;CAEhF,MAAM,cAAc;EAClB,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,WAAS,KAAK;GAAE,OAAO;GAAS,QAAQ,eAAe,KAAK;GAAE,aAAa;GAAa,CAAC;AACzF,YAAU,EAAE;;AAGd,MAAK,MAAM,QAAQ,OAAO;AACxB,MAAI,QAAQ,IAAI,KAAK,OAAO,EAAE;AAC5B,OAAI,QAAQ,SAAS,EAAG,QAAO;AAC/B,iBAAc,MAAM,IAAI,KAAK,OAAO,IAAI,EAAE;;AAE5C,UAAQ,KAAK,KAAK;;AAGpB,KAAI,QAAQ,SAAS,EAAG,QAAO;AAE/B,QAAO;;;;;AAMT,SAAS,kBAAkB,OAA+B;CACxD,MAAM,aAA0B,EAAE;CAClC,IAAI,UAAmD,EAAE;AAEzD,MAAK,MAAM,QAAQ,MAAM,MACvB,KAAI,KAAK,KAAK,MAAM,KAAK,MAAM,QAAQ,SAAS,GAAG;EAEjD,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,aAAW,KAAK;GAAE,OAAO,CAAC,GAAG,QAAQ;GAAE,QAAQ,eAAe,KAAK;GAAE,aAAa,MAAM;GAAa,CAAC;AACtG,YAAU,EAAE;OAEZ,SAAQ,KAAK,KAAK;AAItB,KAAI,QAAQ,SAAS,GAAG;EACtB,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,aAAW,KAAK;GAAE,OAAO;GAAS,QAAQ,eAAe,KAAK;GAAE,aAAa,MAAM;GAAa,CAAC;;AAGnG,QAAO,WAAW,SAAS,IAAI,aAAa,CAAC,MAAM;;;;;;AAOrD,SAAS,iBAAiB,OAAkB,WAAgC;CAI1E,MAAM,YAHW,MAAM,MAAM,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,IAAI,CAG9B,MADR,4BACyB;CAE5C,MAAM,SAAsB,EAAE;CAC9B,IAAI,UAAU;CAGd,MAAM,YAAY,MAAM,MAAM,IAAI,UAAU;CAC5C,MAAM,UAAU,MAAM,MAAM,MAAM,MAAM,SAAS,IAAI,UAAU;CAC/D,MAAM,aAAa,UAAU,YAAY;CACzC,MAAM,mBAAmB,KAAK,IAAI,GAAG,KAAK,MAAM,aAAa,KAAK,IAAI,GAAG,UAAU,OAAO,CAAC,CAAC;CAE5F,IAAI,cAAc;CAClB,IAAI,aAAa;CAEjB,MAAM,cAAc;AAClB,MAAI,CAAC,QAAQ,MAAM,CAAE;EACrB,MAAM,YAAY,KAAK,IAAI,aAAa,mBAAmB,GAAG,QAAQ;AACtE,SAAO,KAAK;GACV,OAAO,CAAC;IAAE,MAAM,QAAQ,MAAM;IAAE,QAAQ;IAAY,CAAC;GACrD,QAAQ,eAAe,QAAQ;GAC/B,aAAa,MAAM;GACpB,CAAC;AACF,eAAa,YAAY;AACzB,YAAU;;AAGZ,MAAK,MAAM,YAAY,WAAW;AAChC;EACA,MAAM,gBAAgB,UAAU,UAAU,MAAM,WAAW;AAC3D,MAAI,eAAe,cAAc,GAAG,aAAa,SAAS;AACxD,UAAO;AACP,aAAU;QAEV,WAAU;;AAId,QAAO;AAEP,QAAO,OAAO,SAAS,IAAI,SAAS,CAAC,MAAM;;;;;;AAW7C,SAAS,mBACP,WACA,eACyC;AACzC,KAAI,iBAAiB,KAAK,CAAC,UAAW,QAAO,EAAE;CAE/C,MAAM,QAAQ,UAAU,KAAK,MAAM,KAAK;CACxC,MAAM,OAAiB,EAAE;CACzB,IAAI,MAAM;AAEV,MAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAAK;EAC1C,MAAM,aAAa,eAAe,MAAM,MAAM,GAAG;AACjD,SAAO;AACP,OAAK,QAAQ,MAAM,MAAM,GAAG;AAC5B,MAAI,OAAO,cAAe;;CAI5B,MAAM,YAAY,UAAU,UAAU,KAAK,SAAS;AACpD,QAAO,KAAK,KAAK,MAAM,SAAS;EAAE;EAAM,QAAQ,KAAK,IAAI,UAAU,WAAW,YAAY,IAAI;EAAE,EAAE;;;;;;;;;;;;;;;;AAqBpG,SAAgB,iBAAiB,SAAyB;AACxD,QAAO,QAAQ,QAAQ,kCAAkC,GAAG;;;AAI9D,SAAS,WAAW,MAAwB;AAC1C,QAAO,KAAK,SAAS,IAAI,IAAI,KAAK,KAAK,MAAM,CAAC,OAAO;;AAGvD,SAAgB,cAAc,SAAiB,MAA8B;CAC3E,MAAM,YAAY,MAAM,aAAa;CACrC,MAAM,gBAAgB,MAAM,WAAW;AAGvC,WAAU,iBAAiB,QAAQ;AAEnC,KAAI,CAAC,QAAQ,MAAM,CAAE,QAAO,EAAE;CAS9B,MAAM,WAAW,gBAPA,QAAQ,MAAM,KAAK,CAC4B,KAAK,MAAM,SAAS;EAClF;EACA,QAAQ,MAAM;EACf,EAAE,CAGoC;CAIvC,MAAM,cAA2B,EAAE;AACnC,MAAK,MAAM,WAAW,UAAU;EAC9B,MAAM,SAAS,KAAK,IAAI,GAAG,YAAY,eAAe,WAAW,QAAQ,YAAY,CAAC,CAAC;AACvF,MAAI,QAAQ,UAAU,QAAQ;AAC5B,eAAY,KAAK,QAAQ;AACzB;;EAGF,MAAM,QAAQ,kBAAkB,QAAQ;AACxC,OAAK,MAAM,QAAQ,OAAO;AACxB,OAAI,KAAK,UAAU,QAAQ;AACzB,gBAAY,KAAK,KAAK;AACtB;;GAGF,MAAM,YAAY,iBAAiB,MAAM,OAAO;AAChD,eAAY,KAAK,GAAG,UAAU;;;CAKlC,MAAM,SAAkB,EAAE;CAC1B,IAAI;AAEJ,MAAK,MAAM,SAAS,aAAa;AAC/B,MAAI,MAAM,MAAM,WAAW,EAAG;EAO9B,MAAM,MADW,CAAC,GAHG,mBAAmB,MAAM,cAAc,EAGzB,GAAG,MAAM,MAAM,CAC7B,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK,CAAC,MAAM;AAEzD,MAAI,CAAC,IAAK;EAEV,MAAM,YAAY,MAAM,MAAM,IAAI,UAAU;EAC5C,MAAM,UAAU,MAAM,MAAM,MAAM,MAAM,SAAS,IAAI,UAAU;EAC/D,MAAM,OAAO,WAAW,MAAM,YAAY,GAAG;AAE7C,SAAO;GAAE,MAAM;GAAK;GAAW;GAAS;AACxC,SAAO,KAAK;GACV;GACA;GACA;GACA,MAAM,OAAO,KAAK;GAClB,aAAa,MAAM;GACpB,CAAC;;AAGJ,QAAO"}
|
|
@@ -1,169 +0,0 @@
|
|
|
1
|
-
import { t as STOP_WORDS } from "./stop-words-BdQuaE9K.mjs";
|
|
2
|
-
|
|
3
|
-
//#region src/graph/clusters.ts
|
|
4
|
-
/**
|
|
5
|
-
* Aggregate per-path observation type counts into cluster-level counts,
|
|
6
|
-
* then pick the dominant type.
|
|
7
|
-
*/
|
|
8
|
-
function aggregateObservationTypes(paths, byPath) {
|
|
9
|
-
const counts = {};
|
|
10
|
-
for (const path of paths) {
|
|
11
|
-
const pathCounts = byPath.get(path);
|
|
12
|
-
if (!pathCounts) continue;
|
|
13
|
-
for (const [type, n] of Object.entries(pathCounts)) counts[type] = (counts[type] ?? 0) + n;
|
|
14
|
-
}
|
|
15
|
-
let dominant = "unknown";
|
|
16
|
-
let maxCount = 0;
|
|
17
|
-
for (const [type, n] of Object.entries(counts)) if (n > maxCount) {
|
|
18
|
-
maxCount = n;
|
|
19
|
-
dominant = type;
|
|
20
|
-
}
|
|
21
|
-
return {
|
|
22
|
-
dominant,
|
|
23
|
-
counts
|
|
24
|
-
};
|
|
25
|
-
}
|
|
26
|
-
const SKIP_PREFIXES = [
|
|
27
|
-
"Attachments/",
|
|
28
|
-
"🗓️ Daily Notes/",
|
|
29
|
-
"Copilot/copilot-conversations/",
|
|
30
|
-
"Z - Zettelkasten/Tweets/"
|
|
31
|
-
];
|
|
32
|
-
/**
|
|
33
|
-
* Cluster vault notes by wikilink connectivity when embeddings aren't available.
|
|
34
|
-
* Uses BFS to find connected components in the link graph, then picks the
|
|
35
|
-
* largest components as clusters. Labels are derived from the most common
|
|
36
|
-
* title words in each component.
|
|
37
|
-
*/
|
|
38
|
-
async function clusterByLinks(backend, lookbackDays, minSize, maxClusters) {
|
|
39
|
-
const now = Date.now();
|
|
40
|
-
const from = now - lookbackDays * 864e5;
|
|
41
|
-
const recentNotes = (await backend.getRecentVaultFiles(from)).filter((f) => f.vaultPath.endsWith(".md"));
|
|
42
|
-
const noteMap = /* @__PURE__ */ new Map();
|
|
43
|
-
for (const n of recentNotes) noteMap.set(n.vaultPath, {
|
|
44
|
-
title: n.title,
|
|
45
|
-
indexed_at: n.indexedAt
|
|
46
|
-
});
|
|
47
|
-
const adj = /* @__PURE__ */ new Map();
|
|
48
|
-
for (const path of noteMap.keys()) if (!adj.has(path)) adj.set(path, /* @__PURE__ */ new Set());
|
|
49
|
-
const linkGraph = await backend.getVaultLinkGraph();
|
|
50
|
-
for (const { source_path, target_path } of linkGraph) if (noteMap.has(source_path) && noteMap.has(target_path)) {
|
|
51
|
-
adj.get(source_path).add(target_path);
|
|
52
|
-
adj.get(target_path).add(source_path);
|
|
53
|
-
}
|
|
54
|
-
const degrees = [...adj.entries()].map(([p, s]) => ({
|
|
55
|
-
path: p,
|
|
56
|
-
degree: s.size
|
|
57
|
-
}));
|
|
58
|
-
degrees.sort((a, b) => b.degree - a.degree);
|
|
59
|
-
const hubThreshold = Math.max(10, degrees[Math.floor(degrees.length * .05)]?.degree ?? 10);
|
|
60
|
-
const hubNodes = /* @__PURE__ */ new Set();
|
|
61
|
-
for (const { path, degree } of degrees) if (degree >= hubThreshold) hubNodes.add(path);
|
|
62
|
-
else break;
|
|
63
|
-
for (const hub of hubNodes) adj.delete(hub);
|
|
64
|
-
for (const [, neighbors] of adj) for (const hub of hubNodes) neighbors.delete(hub);
|
|
65
|
-
const visited = /* @__PURE__ */ new Set();
|
|
66
|
-
const components = [];
|
|
67
|
-
for (const path of noteMap.keys()) {
|
|
68
|
-
if (visited.has(path) || hubNodes.has(path)) continue;
|
|
69
|
-
if (SKIP_PREFIXES.some((p) => path.startsWith(p))) {
|
|
70
|
-
visited.add(path);
|
|
71
|
-
continue;
|
|
72
|
-
}
|
|
73
|
-
const component = [];
|
|
74
|
-
const queue = [path];
|
|
75
|
-
visited.add(path);
|
|
76
|
-
while (queue.length > 0) {
|
|
77
|
-
const current = queue.shift();
|
|
78
|
-
component.push(current);
|
|
79
|
-
const neighbors = adj.get(current);
|
|
80
|
-
if (!neighbors) continue;
|
|
81
|
-
for (const neighbor of neighbors) if (!visited.has(neighbor) && !SKIP_PREFIXES.some((p) => neighbor.startsWith(p))) {
|
|
82
|
-
visited.add(neighbor);
|
|
83
|
-
queue.push(neighbor);
|
|
84
|
-
}
|
|
85
|
-
}
|
|
86
|
-
if (component.length >= minSize) components.push(component);
|
|
87
|
-
}
|
|
88
|
-
components.sort((a, b) => b.length - a.length);
|
|
89
|
-
const topComponents = components.slice(0, maxClusters);
|
|
90
|
-
function generateLinkLabel(paths) {
|
|
91
|
-
const wordCounts = /* @__PURE__ */ new Map();
|
|
92
|
-
for (const p of paths) {
|
|
93
|
-
const title = noteMap.get(p)?.title;
|
|
94
|
-
if (!title) continue;
|
|
95
|
-
const words = title.toLowerCase().replace(/[^a-z0-9äöüàéèêëçñß\s]/g, " ").split(/\s+/).filter((w) => w.length > 2 && !STOP_WORDS.has(w));
|
|
96
|
-
for (const word of words) wordCounts.set(word, (wordCounts.get(word) ?? 0) + 1);
|
|
97
|
-
}
|
|
98
|
-
return [...wordCounts.entries()].sort((a, b) => b[1] - a[1]).slice(0, 3).map(([w]) => w).join(" / ") || "Linked Notes";
|
|
99
|
-
}
|
|
100
|
-
return {
|
|
101
|
-
themes: topComponents.map((component, idx) => {
|
|
102
|
-
const notes = component.map((p) => ({
|
|
103
|
-
path: p,
|
|
104
|
-
title: noteMap.get(p)?.title ?? null
|
|
105
|
-
}));
|
|
106
|
-
const avgRecency = component.reduce((sum, p) => sum + (noteMap.get(p)?.indexed_at ?? 0), 0) / component.length;
|
|
107
|
-
const uniqueFolders = new Set(component.map((p) => p.split("/")[0]));
|
|
108
|
-
return {
|
|
109
|
-
id: idx,
|
|
110
|
-
label: generateLinkLabel(component),
|
|
111
|
-
notes,
|
|
112
|
-
size: component.length,
|
|
113
|
-
folderDiversity: uniqueFolders.size / component.length,
|
|
114
|
-
avgRecency,
|
|
115
|
-
linkedRatio: 1,
|
|
116
|
-
suggestIndexNote: component.length >= 10
|
|
117
|
-
};
|
|
118
|
-
}),
|
|
119
|
-
totalNotesAnalyzed: recentNotes.length,
|
|
120
|
-
timeWindow: {
|
|
121
|
-
from,
|
|
122
|
-
to: now
|
|
123
|
-
}
|
|
124
|
-
};
|
|
125
|
-
}
|
|
126
|
-
async function handleGraphClusters(backend, params) {
|
|
127
|
-
const minSize = params.min_size ?? 3;
|
|
128
|
-
const maxClusters = params.max_clusters ?? 20;
|
|
129
|
-
const lookbackDays = params.lookback_days ?? 90;
|
|
130
|
-
if (!(params.project_id ?? 0)) throw new Error("graph_clusters: project_id is required (pass the vault project's numeric ID)");
|
|
131
|
-
const themeResult = await clusterByLinks(backend, lookbackDays, minSize, maxClusters);
|
|
132
|
-
const allPaths = themeResult.themes.flatMap((t) => t.notes.map((n) => n.path));
|
|
133
|
-
const observationsByPath = await backend.getObservationTypesForPaths(allPaths, params.project_id);
|
|
134
|
-
const fileRows = await backend.getVaultFilesByPaths(allPaths);
|
|
135
|
-
const indexedAtMap = new Map(fileRows.map((f) => [f.vaultPath, f.indexedAt]));
|
|
136
|
-
const clusters = themeResult.themes.map((theme) => {
|
|
137
|
-
const notePaths = theme.notes.map((n) => n.path);
|
|
138
|
-
const notesWithTimestamps = theme.notes.map((n) => ({
|
|
139
|
-
vault_path: n.path,
|
|
140
|
-
title: n.title ?? n.path.split("/").pop() ?? n.path,
|
|
141
|
-
indexed_at: indexedAtMap.get(n.path) ?? 0
|
|
142
|
-
}));
|
|
143
|
-
const avgRecency = theme.avgRecency;
|
|
144
|
-
const { dominant, counts } = aggregateObservationTypes(notePaths, observationsByPath);
|
|
145
|
-
return {
|
|
146
|
-
id: theme.id,
|
|
147
|
-
label: theme.label,
|
|
148
|
-
size: theme.size,
|
|
149
|
-
folder_diversity: theme.folderDiversity,
|
|
150
|
-
avg_recency: avgRecency,
|
|
151
|
-
linked_ratio: theme.linkedRatio,
|
|
152
|
-
dominant_observation_type: dominant,
|
|
153
|
-
observation_type_counts: counts,
|
|
154
|
-
suggest_index_note: theme.suggestIndexNote,
|
|
155
|
-
has_idea_note: false,
|
|
156
|
-
notes: notesWithTimestamps
|
|
157
|
-
};
|
|
158
|
-
});
|
|
159
|
-
clusters.sort((a, b) => b.size - a.size);
|
|
160
|
-
return {
|
|
161
|
-
clusters: clusters.slice(0, maxClusters),
|
|
162
|
-
total_notes_analyzed: themeResult.totalNotesAnalyzed,
|
|
163
|
-
time_window: themeResult.timeWindow
|
|
164
|
-
};
|
|
165
|
-
}
|
|
166
|
-
|
|
167
|
-
//#endregion
|
|
168
|
-
export { handleGraphClusters };
|
|
169
|
-
//# sourceMappingURL=clusters-BCtD3fbe.mjs.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"clusters-BCtD3fbe.mjs","names":[],"sources":["../src/graph/clusters.ts"],"sourcesContent":["/**\n * clusters.ts — graph_clusters endpoint handler\n *\n * Reuses the zettelThemes() agglomerative clustering algorithm and enriches\n * each cluster with observation-type statistics, avg_recency from member\n * timestamps, and helper flags for the Obsidian knowledge plugin.\n */\n\nimport type { StorageBackend } from \"../storage/interface.js\";\nimport { STOP_WORDS } from \"../utils/stop-words.js\";\n\n// ---------------------------------------------------------------------------\n// Public param / result types\n// ---------------------------------------------------------------------------\n\nexport interface GraphClustersParams {\n project_id?: number;\n min_size?: number;\n max_clusters?: number;\n lookback_days?: number;\n similarity_threshold?: number;\n}\n\nexport interface ClusterNode {\n id: number;\n label: string;\n size: number;\n folder_diversity: number;\n avg_recency: number;\n linked_ratio: number;\n dominant_observation_type: string;\n observation_type_counts: Record<string, number>;\n suggest_index_note: boolean;\n has_idea_note: boolean;\n notes: Array<{ vault_path: string; title: string; indexed_at: number }>;\n}\n\nexport interface GraphClustersResult {\n clusters: ClusterNode[];\n total_notes_analyzed: number;\n time_window: { from: number; to: number };\n}\n\n/**\n * Aggregate per-path observation type counts into cluster-level counts,\n * then pick the dominant type.\n */\nfunction aggregateObservationTypes(\n paths: string[],\n byPath: Map<string, Record<string, number>>\n): { dominant: string; counts: Record<string, number> } {\n const counts: Record<string, number> = {};\n for (const path of paths) {\n const pathCounts = byPath.get(path);\n if (!pathCounts) continue;\n for (const [type, n] of Object.entries(pathCounts)) {\n counts[type] = (counts[type] ?? 0) + n;\n }\n }\n\n let dominant = \"unknown\";\n let maxCount = 0;\n for (const [type, n] of Object.entries(counts)) {\n if (n > maxCount) {\n maxCount = n;\n dominant = type;\n }\n }\n\n return { dominant, counts };\n}\n\n// ---------------------------------------------------------------------------\n// Link-based fallback clustering (wikilink connected components)\n// ---------------------------------------------------------------------------\n\nconst SKIP_PREFIXES = [\n \"Attachments/\", \"🗓️ Daily Notes/\", \"Copilot/copilot-conversations/\",\n \"Z - Zettelkasten/Tweets/\",\n];\n\n/**\n * Cluster vault notes by wikilink connectivity when embeddings aren't available.\n * Uses BFS to find connected components in the link graph, then picks the\n * largest components as clusters. Labels are derived from the most common\n * title words in each component.\n */\nasync function clusterByLinks(\n backend: StorageBackend,\n lookbackDays: number,\n minSize: number,\n maxClusters: number,\n): Promise<{ themes: Array<{ id: number; label: string; notes: Array<{ path: string; title: string | null }>; size: number; folderDiversity: number; avgRecency: number; linkedRatio: number; suggestIndexNote: boolean }>; totalNotesAnalyzed: number; timeWindow: { from: number; to: number } }> {\n const now = Date.now();\n const from = now - lookbackDays * 86400000;\n\n // Get recent notes\n const recentFiles = await backend.getRecentVaultFiles(from);\n const recentNotes = recentFiles.filter(f => f.vaultPath.endsWith(\".md\"));\n\n const noteMap = new Map<string, { title: string | null; indexed_at: number }>();\n for (const n of recentNotes) {\n noteMap.set(n.vaultPath, { title: n.title, indexed_at: n.indexedAt });\n }\n\n // Build adjacency list from vault_links (only for recent notes)\n const adj = new Map<string, Set<string>>();\n for (const path of noteMap.keys()) {\n if (!adj.has(path)) adj.set(path, new Set());\n }\n\n const linkGraph = await backend.getVaultLinkGraph();\n\n for (const { source_path, target_path } of linkGraph) {\n if (noteMap.has(source_path) && noteMap.has(target_path)) {\n adj.get(source_path)!.add(target_path);\n adj.get(target_path)!.add(source_path);\n }\n }\n\n // Remove hub nodes before BFS\n const degrees = [...adj.entries()].map(([p, s]) => ({ path: p, degree: s.size }));\n degrees.sort((a, b) => b.degree - a.degree);\n const hubThreshold = Math.max(10, degrees[Math.floor(degrees.length * 0.05)]?.degree ?? 10);\n const hubNodes = new Set<string>();\n for (const { path, degree } of degrees) {\n if (degree >= hubThreshold) hubNodes.add(path);\n else break;\n }\n\n for (const hub of hubNodes) {\n adj.delete(hub);\n }\n for (const [, neighbors] of adj) {\n for (const hub of hubNodes) {\n neighbors.delete(hub);\n }\n }\n\n // BFS connected components\n const visited = new Set<string>();\n const components: string[][] = [];\n\n for (const path of noteMap.keys()) {\n if (visited.has(path) || hubNodes.has(path)) continue;\n if (SKIP_PREFIXES.some(p => path.startsWith(p))) { visited.add(path); continue; }\n const component: string[] = [];\n const queue = [path];\n visited.add(path);\n\n while (queue.length > 0) {\n const current = queue.shift()!;\n component.push(current);\n const neighbors = adj.get(current);\n if (!neighbors) continue;\n for (const neighbor of neighbors) {\n if (!visited.has(neighbor) && !SKIP_PREFIXES.some(p => neighbor.startsWith(p))) {\n visited.add(neighbor);\n queue.push(neighbor);\n }\n }\n }\n\n if (component.length >= minSize) {\n components.push(component);\n }\n }\n\n components.sort((a, b) => b.length - a.length);\n const topComponents = components.slice(0, maxClusters);\n\n // STOP_WORDS imported from utils/stop-words.ts (module-level import)\n\n function generateLinkLabel(paths: string[]): string {\n const wordCounts = new Map<string, number>();\n for (const p of paths) {\n const title = noteMap.get(p)?.title;\n if (!title) continue;\n const words = title.toLowerCase().replace(/[^a-z0-9äöüàéèêëçñß\\s]/g, \" \").split(/\\s+/)\n .filter(w => w.length > 2 && !STOP_WORDS.has(w));\n for (const word of words) {\n wordCounts.set(word, (wordCounts.get(word) ?? 0) + 1);\n }\n }\n const sorted = [...wordCounts.entries()].sort((a, b) => b[1] - a[1]);\n return sorted.slice(0, 3).map(([w]) => w).join(\" / \") || \"Linked Notes\";\n }\n\n const themes = topComponents.map((component, idx) => {\n const notes = component.map(p => ({\n path: p,\n title: noteMap.get(p)?.title ?? null,\n }));\n const avgRecency = component.reduce((sum, p) => sum + (noteMap.get(p)?.indexed_at ?? 0), 0) / component.length;\n const uniqueFolders = new Set(component.map(p => p.split(\"/\")[0]));\n\n return {\n id: idx,\n label: generateLinkLabel(component),\n notes,\n size: component.length,\n folderDiversity: uniqueFolders.size / component.length,\n avgRecency,\n linkedRatio: 1.0,\n suggestIndexNote: component.length >= 10,\n };\n });\n\n return {\n themes,\n totalNotesAnalyzed: recentNotes.length,\n timeWindow: { from, to: now },\n };\n}\n\n// ---------------------------------------------------------------------------\n// Main handler\n// ---------------------------------------------------------------------------\n\nexport async function handleGraphClusters(\n backend: StorageBackend,\n params: GraphClustersParams\n): Promise<GraphClustersResult> {\n const minSize = params.min_size ?? 3;\n const maxClusters = params.max_clusters ?? 20;\n const lookbackDays = params.lookback_days ?? 90;\n\n const vaultProjectId = params.project_id ?? 0;\n\n if (!vaultProjectId) {\n throw new Error(\n \"graph_clusters: project_id is required (pass the vault project's numeric ID)\"\n );\n }\n\n const themeResult = await clusterByLinks(backend, lookbackDays, minSize, maxClusters);\n\n const allPaths = themeResult.themes.flatMap((t) => t.notes.map((n) => n.path));\n\n const observationsByPath = await backend.getObservationTypesForPaths(allPaths, params.project_id);\n\n // Fetch indexed_at timestamps for all notes in bulk\n const fileRows = await backend.getVaultFilesByPaths(allPaths);\n const indexedAtMap = new Map<string, number>(fileRows.map(f => [f.vaultPath, f.indexedAt]));\n\n const clusters: ClusterNode[] = themeResult.themes.map((theme) => {\n const notePaths = theme.notes.map((n) => n.path);\n\n const notesWithTimestamps = theme.notes.map((n) => ({\n vault_path: n.path,\n title: n.title ?? n.path.split(\"/\").pop() ?? n.path,\n indexed_at: indexedAtMap.get(n.path) ?? 0,\n }));\n\n const avgRecency = theme.avgRecency;\n\n const { dominant, counts } = aggregateObservationTypes(\n notePaths,\n observationsByPath\n );\n\n return {\n id: theme.id,\n label: theme.label,\n size: theme.size,\n folder_diversity: theme.folderDiversity,\n avg_recency: avgRecency,\n linked_ratio: theme.linkedRatio,\n dominant_observation_type: dominant,\n observation_type_counts: counts,\n suggest_index_note: theme.suggestIndexNote,\n has_idea_note: false,\n notes: notesWithTimestamps,\n };\n });\n\n clusters.sort((a, b) => b.size - a.size);\n\n return {\n clusters: clusters.slice(0, maxClusters),\n total_notes_analyzed: themeResult.totalNotesAnalyzed,\n time_window: themeResult.timeWindow,\n };\n}\n"],"mappings":";;;;;;;AA+CA,SAAS,0BACP,OACA,QACsD;CACtD,MAAM,SAAiC,EAAE;AACzC,MAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,aAAa,OAAO,IAAI,KAAK;AACnC,MAAI,CAAC,WAAY;AACjB,OAAK,MAAM,CAAC,MAAM,MAAM,OAAO,QAAQ,WAAW,CAChD,QAAO,SAAS,OAAO,SAAS,KAAK;;CAIzC,IAAI,WAAW;CACf,IAAI,WAAW;AACf,MAAK,MAAM,CAAC,MAAM,MAAM,OAAO,QAAQ,OAAO,CAC5C,KAAI,IAAI,UAAU;AAChB,aAAW;AACX,aAAW;;AAIf,QAAO;EAAE;EAAU;EAAQ;;AAO7B,MAAM,gBAAgB;CACpB;CAAgB;CAAoB;CACpC;CACD;;;;;;;AAQD,eAAe,eACb,SACA,cACA,SACA,aACkS;CAClS,MAAM,MAAM,KAAK,KAAK;CACtB,MAAM,OAAO,MAAM,eAAe;CAIlC,MAAM,eADc,MAAM,QAAQ,oBAAoB,KAAK,EAC3B,QAAO,MAAK,EAAE,UAAU,SAAS,MAAM,CAAC;CAExE,MAAM,0BAAU,IAAI,KAA2D;AAC/E,MAAK,MAAM,KAAK,YACd,SAAQ,IAAI,EAAE,WAAW;EAAE,OAAO,EAAE;EAAO,YAAY,EAAE;EAAW,CAAC;CAIvE,MAAM,sBAAM,IAAI,KAA0B;AAC1C,MAAK,MAAM,QAAQ,QAAQ,MAAM,CAC/B,KAAI,CAAC,IAAI,IAAI,KAAK,CAAE,KAAI,IAAI,sBAAM,IAAI,KAAK,CAAC;CAG9C,MAAM,YAAY,MAAM,QAAQ,mBAAmB;AAEnD,MAAK,MAAM,EAAE,aAAa,iBAAiB,UACzC,KAAI,QAAQ,IAAI,YAAY,IAAI,QAAQ,IAAI,YAAY,EAAE;AACxD,MAAI,IAAI,YAAY,CAAE,IAAI,YAAY;AACtC,MAAI,IAAI,YAAY,CAAE,IAAI,YAAY;;CAK1C,MAAM,UAAU,CAAC,GAAG,IAAI,SAAS,CAAC,CAAC,KAAK,CAAC,GAAG,QAAQ;EAAE,MAAM;EAAG,QAAQ,EAAE;EAAM,EAAE;AACjF,SAAQ,MAAM,GAAG,MAAM,EAAE,SAAS,EAAE,OAAO;CAC3C,MAAM,eAAe,KAAK,IAAI,IAAI,QAAQ,KAAK,MAAM,QAAQ,SAAS,IAAK,GAAG,UAAU,GAAG;CAC3F,MAAM,2BAAW,IAAI,KAAa;AAClC,MAAK,MAAM,EAAE,MAAM,YAAY,QAC7B,KAAI,UAAU,aAAc,UAAS,IAAI,KAAK;KACzC;AAGP,MAAK,MAAM,OAAO,SAChB,KAAI,OAAO,IAAI;AAEjB,MAAK,MAAM,GAAG,cAAc,IAC1B,MAAK,MAAM,OAAO,SAChB,WAAU,OAAO,IAAI;CAKzB,MAAM,0BAAU,IAAI,KAAa;CACjC,MAAM,aAAyB,EAAE;AAEjC,MAAK,MAAM,QAAQ,QAAQ,MAAM,EAAE;AACjC,MAAI,QAAQ,IAAI,KAAK,IAAI,SAAS,IAAI,KAAK,CAAE;AAC7C,MAAI,cAAc,MAAK,MAAK,KAAK,WAAW,EAAE,CAAC,EAAE;AAAE,WAAQ,IAAI,KAAK;AAAE;;EACtE,MAAM,YAAsB,EAAE;EAC9B,MAAM,QAAQ,CAAC,KAAK;AACpB,UAAQ,IAAI,KAAK;AAEjB,SAAO,MAAM,SAAS,GAAG;GACvB,MAAM,UAAU,MAAM,OAAO;AAC7B,aAAU,KAAK,QAAQ;GACvB,MAAM,YAAY,IAAI,IAAI,QAAQ;AAClC,OAAI,CAAC,UAAW;AAChB,QAAK,MAAM,YAAY,UACrB,KAAI,CAAC,QAAQ,IAAI,SAAS,IAAI,CAAC,cAAc,MAAK,MAAK,SAAS,WAAW,EAAE,CAAC,EAAE;AAC9E,YAAQ,IAAI,SAAS;AACrB,UAAM,KAAK,SAAS;;;AAK1B,MAAI,UAAU,UAAU,QACtB,YAAW,KAAK,UAAU;;AAI9B,YAAW,MAAM,GAAG,MAAM,EAAE,SAAS,EAAE,OAAO;CAC9C,MAAM,gBAAgB,WAAW,MAAM,GAAG,YAAY;CAItD,SAAS,kBAAkB,OAAyB;EAClD,MAAM,6BAAa,IAAI,KAAqB;AAC5C,OAAK,MAAM,KAAK,OAAO;GACrB,MAAM,QAAQ,QAAQ,IAAI,EAAE,EAAE;AAC9B,OAAI,CAAC,MAAO;GACZ,MAAM,QAAQ,MAAM,aAAa,CAAC,QAAQ,2BAA2B,IAAI,CAAC,MAAM,MAAM,CACnF,QAAO,MAAK,EAAE,SAAS,KAAK,CAAC,WAAW,IAAI,EAAE,CAAC;AAClD,QAAK,MAAM,QAAQ,MACjB,YAAW,IAAI,OAAO,WAAW,IAAI,KAAK,IAAI,KAAK,EAAE;;AAIzD,SADe,CAAC,GAAG,WAAW,SAAS,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,GAAG,CACtD,MAAM,GAAG,EAAE,CAAC,KAAK,CAAC,OAAO,EAAE,CAAC,KAAK,MAAM,IAAI;;AAuB3D,QAAO;EACL,QArBa,cAAc,KAAK,WAAW,QAAQ;GACnD,MAAM,QAAQ,UAAU,KAAI,OAAM;IAChC,MAAM;IACN,OAAO,QAAQ,IAAI,EAAE,EAAE,SAAS;IACjC,EAAE;GACH,MAAM,aAAa,UAAU,QAAQ,KAAK,MAAM,OAAO,QAAQ,IAAI,EAAE,EAAE,cAAc,IAAI,EAAE,GAAG,UAAU;GACxG,MAAM,gBAAgB,IAAI,IAAI,UAAU,KAAI,MAAK,EAAE,MAAM,IAAI,CAAC,GAAG,CAAC;AAElE,UAAO;IACL,IAAI;IACJ,OAAO,kBAAkB,UAAU;IACnC;IACA,MAAM,UAAU;IAChB,iBAAiB,cAAc,OAAO,UAAU;IAChD;IACA,aAAa;IACb,kBAAkB,UAAU,UAAU;IACvC;IACD;EAIA,oBAAoB,YAAY;EAChC,YAAY;GAAE;GAAM,IAAI;GAAK;EAC9B;;AAOH,eAAsB,oBACpB,SACA,QAC8B;CAC9B,MAAM,UAAU,OAAO,YAAY;CACnC,MAAM,cAAc,OAAO,gBAAgB;CAC3C,MAAM,eAAe,OAAO,iBAAiB;AAI7C,KAAI,EAFmB,OAAO,cAAc,GAG1C,OAAM,IAAI,MACR,+EACD;CAGH,MAAM,cAAc,MAAM,eAAe,SAAS,cAAc,SAAS,YAAY;CAErF,MAAM,WAAW,YAAY,OAAO,SAAS,MAAM,EAAE,MAAM,KAAK,MAAM,EAAE,KAAK,CAAC;CAE9E,MAAM,qBAAqB,MAAM,QAAQ,4BAA4B,UAAU,OAAO,WAAW;CAGjG,MAAM,WAAW,MAAM,QAAQ,qBAAqB,SAAS;CAC7D,MAAM,eAAe,IAAI,IAAoB,SAAS,KAAI,MAAK,CAAC,EAAE,WAAW,EAAE,UAAU,CAAC,CAAC;CAE3F,MAAM,WAA0B,YAAY,OAAO,KAAK,UAAU;EAChE,MAAM,YAAY,MAAM,MAAM,KAAK,MAAM,EAAE,KAAK;EAEhD,MAAM,sBAAsB,MAAM,MAAM,KAAK,OAAO;GAClD,YAAY,EAAE;GACd,OAAO,EAAE,SAAS,EAAE,KAAK,MAAM,IAAI,CAAC,KAAK,IAAI,EAAE;GAC/C,YAAY,aAAa,IAAI,EAAE,KAAK,IAAI;GACzC,EAAE;EAEH,MAAM,aAAa,MAAM;EAEzB,MAAM,EAAE,UAAU,WAAW,0BAC3B,WACA,mBACD;AAED,SAAO;GACL,IAAI,MAAM;GACV,OAAO,MAAM;GACb,MAAM,MAAM;GACZ,kBAAkB,MAAM;GACxB,aAAa;GACb,cAAc,MAAM;GACpB,2BAA2B;GAC3B,yBAAyB;GACzB,oBAAoB,MAAM;GAC1B,eAAe;GACf,OAAO;GACR;GACD;AAEF,UAAS,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,KAAK;AAExC,QAAO;EACL,UAAU,SAAS,MAAM,GAAG,YAAY;EACxC,sBAAsB,YAAY;EAClC,aAAa,YAAY;EAC1B"}
|