@tekmidian/pai 0.66.1 → 0.66.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/dist/{aibroker-client-B8c42Lh8.mjs → aibroker-client-CHEEvJZW.mjs} +2 -2
  2. package/dist/{aibroker-client-C5Fw7DNz.mjs → aibroker-client-Dfv7j1Od.mjs} +3 -3
  3. package/dist/{aibroker-client-C5Fw7DNz.mjs.map → aibroker-client-Dfv7j1Od.mjs.map} +1 -1
  4. package/dist/{auto-route-o0BOsXn7.mjs → auto-route-BnizyALK.mjs} +65 -4
  5. package/dist/auto-route-BnizyALK.mjs.map +1 -0
  6. package/dist/auto-route-D6fW1Q8z.mjs +3 -0
  7. package/dist/{chain-CZfKIEq9.mjs → chain-D5876UQX.mjs} +3 -4
  8. package/dist/{chain-CZfKIEq9.mjs.map → chain-D5876UQX.mjs.map} +1 -1
  9. package/dist/cli/index.mjs +18 -23
  10. package/dist/cli/index.mjs.map +1 -1
  11. package/dist/cli/program.mjs +17 -22
  12. package/dist/{config-BbLFD7Uf.mjs → config-C-LiGlop.mjs} +2 -2
  13. package/dist/{config-BbLFD7Uf.mjs.map → config-C-LiGlop.mjs.map} +1 -1
  14. package/dist/{config-YinjgXEJ.mjs → config-DKIA7wgF.mjs} +1 -1
  15. package/dist/{context-handover-cache-pkHzmL3c.mjs → context-handover-cache-9PGvXRIw.mjs} +984 -10
  16. package/dist/context-handover-cache-9PGvXRIw.mjs.map +1 -0
  17. package/dist/daemon/index.mjs +17 -19
  18. package/dist/daemon/index.mjs.map +1 -1
  19. package/dist/daemon-BUgNvMUh.mjs +5029 -0
  20. package/dist/daemon-BUgNvMUh.mjs.map +1 -0
  21. package/dist/daemon-DL0eglMJ.mjs +18 -0
  22. package/dist/daemon-mcp/index.mjs +12 -9
  23. package/dist/daemon-mcp/index.mjs.map +1 -1
  24. package/dist/{embeddings-CcscYWwk.mjs → embeddings-CEBGrzwu.mjs} +1 -1
  25. package/dist/{embeddings-Bx3q0QOY.mjs → embeddings-DOLZnT1X.mjs} +1 -1
  26. package/dist/{embeddings-Bx3q0QOY.mjs.map → embeddings-DOLZnT1X.mjs.map} +1 -1
  27. package/dist/{env-JNEIrQWg.mjs → env-DiolswKQ.mjs} +1 -1
  28. package/dist/{env-JNEIrQWg.mjs.map → env-DiolswKQ.mjs.map} +1 -1
  29. package/dist/factory-DrBg24UT.mjs +8 -0
  30. package/dist/factory-Ypf8r0fK.mjs +2892 -0
  31. package/dist/factory-Ypf8r0fK.mjs.map +1 -0
  32. package/dist/{fallback-CIiq2Raw.mjs → fallback-D51R4shY.mjs} +23 -11
  33. package/dist/fallback-D51R4shY.mjs.map +1 -0
  34. package/dist/{chunker-BH4i-F2b.mjs → helpers-CZsi_49C.mjs} +221 -2
  35. package/dist/helpers-CZsi_49C.mjs.map +1 -0
  36. package/dist/hooks/worker-status-line.mjs.map +2 -2
  37. package/dist/index.mjs +6 -7
  38. package/dist/{ipc-client-D16Xw6Uo.mjs → ipc-client-BLYX51nG.mjs} +2 -2
  39. package/dist/{ipc-client-D16Xw6Uo.mjs.map → ipc-client-BLYX51nG.mjs.map} +1 -1
  40. package/dist/main-resolver-BYgP86C1.mjs +7 -0
  41. package/dist/{main-resolver-DPPwHNtn.mjs → main-resolver-kVkMJZUD.mjs} +6 -6
  42. package/dist/{main-resolver-DPPwHNtn.mjs.map → main-resolver-kVkMJZUD.mjs.map} +1 -1
  43. package/dist/{migrate-BD7D8EEh.mjs → migrate-Bq0esMOa.mjs} +2 -2
  44. package/dist/{migrate-BD7D8EEh.mjs.map → migrate-Bq0esMOa.mjs.map} +1 -1
  45. package/dist/{pai-marker-D1MMswkz.mjs → pai-marker-DXVpFsYz.mjs} +1 -1
  46. package/dist/{pai-marker-D1MMswkz.mjs.map → pai-marker-DXVpFsYz.mjs.map} +1 -1
  47. package/dist/{planner-Dm5Acyo1.mjs → planner-D5yt0EfA.mjs} +10 -7
  48. package/dist/{planner-Dm5Acyo1.mjs.map → planner-D5yt0EfA.mjs.map} +1 -1
  49. package/dist/postgres-BmLr0MUm.mjs +5 -0
  50. package/dist/{postgres-Ceqsa64C.mjs → postgres-D3xc2RB4.mjs} +364 -9
  51. package/dist/postgres-D3xc2RB4.mjs.map +1 -0
  52. package/dist/{program-JJ0TwSHI.mjs → program-DIXRtAFy.mjs} +77 -70
  53. package/dist/program-DIXRtAFy.mjs.map +1 -0
  54. package/dist/{query-feedback-BIaZTTFO.mjs → query-feedback-B7FYE4JR.mjs} +1 -1
  55. package/dist/{query-feedback-BIaZTTFO.mjs.map → query-feedback-B7FYE4JR.mjs.map} +1 -1
  56. package/dist/{reranker-DKv80KO5.mjs → reranker-3lnggwgq.mjs} +1 -1
  57. package/dist/{reranker-DKv80KO5.mjs.map → reranker-3lnggwgq.mjs.map} +1 -1
  58. package/dist/{reranker-CFiEzHQu.mjs → reranker-CZ2mP4cf.mjs} +1 -1
  59. package/dist/router-BK-hFeQ6.mjs +3 -0
  60. package/dist/{router-a5G7q8_3.mjs → router-BaTbc9VX.mjs} +2 -2
  61. package/dist/{router-a5G7q8_3.mjs.map → router-BaTbc9VX.mjs.map} +1 -1
  62. package/dist/{run-BCpbxqA5.mjs → run-CLYTOm0x.mjs} +599 -22
  63. package/dist/run-CLYTOm0x.mjs.map +1 -0
  64. package/dist/{run-env-DRcp7A8K.mjs → run-env-Bnmf6bRI.mjs} +2 -2
  65. package/dist/{run-env-DRcp7A8K.mjs.map → run-env-Bnmf6bRI.mjs.map} +1 -1
  66. package/dist/{runtime-paths-CHTg3ywb.mjs → runtime-paths-D3FKAm69.mjs} +1 -1
  67. package/dist/{runtime-paths-CHTg3ywb.mjs.map → runtime-paths-D3FKAm69.mjs.map} +1 -1
  68. package/dist/{search-BOSphCJ1.mjs → search-CNAGTiJP.mjs} +2 -2
  69. package/dist/{search-BOSphCJ1.mjs.map → search-CNAGTiJP.mjs.map} +1 -1
  70. package/dist/{sources-CijVso3n.mjs → sources-CM2g-CLT.mjs} +2 -2
  71. package/dist/{sources-CijVso3n.mjs.map → sources-CM2g-CLT.mjs.map} +1 -1
  72. package/dist/{stop-words-BdQuaE9K.mjs → stop-words-DtxaTWU_.mjs} +1 -1
  73. package/dist/{stop-words-BdQuaE9K.mjs.map → stop-words-DtxaTWU_.mjs.map} +1 -1
  74. package/dist/{utils-C9HsYDpO.mjs → utils-DddRwMsG.mjs} +2 -2
  75. package/dist/{utils-C9HsYDpO.mjs.map → utils-DddRwMsG.mjs.map} +1 -1
  76. package/dist/{zettelkasten-CtGHQWPU.mjs → zettelkasten-BhZMvmoK.mjs} +149 -7
  77. package/dist/zettelkasten-BhZMvmoK.mjs.map +1 -0
  78. package/dist/zettelkasten-BwV1pluL.mjs +5 -0
  79. package/docs/commands/worker.md +2 -0
  80. package/package.json +1 -1
  81. package/dist/async-C2Bm_Lal.mjs +0 -300
  82. package/dist/async-C2Bm_Lal.mjs.map +0 -1
  83. package/dist/auto-route-o0BOsXn7.mjs.map +0 -1
  84. package/dist/chunker-BH4i-F2b.mjs.map +0 -1
  85. package/dist/clusters-BCtD3fbe.mjs +0 -169
  86. package/dist/clusters-BCtD3fbe.mjs.map +0 -1
  87. package/dist/context-handover-cache-pkHzmL3c.mjs.map +0 -1
  88. package/dist/daemon-DXlaVCkw.mjs +0 -1741
  89. package/dist/daemon-DXlaVCkw.mjs.map +0 -1
  90. package/dist/daemon-DyORmxZu.mjs +0 -19
  91. package/dist/detector-BEPFsINR.mjs +0 -3
  92. package/dist/detector-BRTtAWrM.mjs +0 -65
  93. package/dist/detector-BRTtAWrM.mjs.map +0 -1
  94. package/dist/factory-CNpPQQ-5.mjs +0 -249
  95. package/dist/factory-CNpPQQ-5.mjs.map +0 -1
  96. package/dist/factory-DY7x73mE.mjs +0 -3
  97. package/dist/fallback-CIiq2Raw.mjs.map +0 -1
  98. package/dist/federation-db-BTyoufBh.mjs +0 -139
  99. package/dist/federation-db-BTyoufBh.mjs.map +0 -1
  100. package/dist/federation-db-HfIFc7FG.mjs +0 -3
  101. package/dist/helpers-BRCJg0G3.mjs +0 -223
  102. package/dist/helpers-BRCJg0G3.mjs.map +0 -1
  103. package/dist/indexer-backend-DFF2FrYx.mjs +0 -5
  104. package/dist/indexer-backend-isSLg6yE.mjs +0 -1
  105. package/dist/kg-entity-DCOcsVFD.mjs +0 -29
  106. package/dist/kg-entity-DCOcsVFD.mjs.map +0 -1
  107. package/dist/latent-ideas-wwAXquGe.mjs +0 -191
  108. package/dist/latent-ideas-wwAXquGe.mjs.map +0 -1
  109. package/dist/link-boost-CnI7UVMJ.mjs +0 -34
  110. package/dist/link-boost-CnI7UVMJ.mjs.map +0 -1
  111. package/dist/main-resolver-BKz_OuEh.mjs +0 -7
  112. package/dist/merge-2gqRPFu2.mjs +0 -3
  113. package/dist/merge-DgU9OgZy.mjs +0 -6
  114. package/dist/merge-DgU9OgZy.mjs.map +0 -1
  115. package/dist/module-paths-DdRzbkUI.mjs +0 -44
  116. package/dist/module-paths-DdRzbkUI.mjs.map +0 -1
  117. package/dist/neighborhood-2FsmzoxG.mjs +0 -114
  118. package/dist/neighborhood-2FsmzoxG.mjs.map +0 -1
  119. package/dist/note-context-De63k8na.mjs +0 -106
  120. package/dist/note-context-De63k8na.mjs.map +0 -1
  121. package/dist/postgres-Ceqsa64C.mjs.map +0 -1
  122. package/dist/program-JJ0TwSHI.mjs.map +0 -1
  123. package/dist/query-feedback-B_iigYj-.mjs +0 -3
  124. package/dist/registry-db-C7voqML9.mjs +0 -213
  125. package/dist/registry-db-C7voqML9.mjs.map +0 -1
  126. package/dist/registry-db-JHPhA8vF.mjs +0 -3
  127. package/dist/registry-postgres-nOfyBSIW.mjs +0 -795
  128. package/dist/registry-postgres-nOfyBSIW.mjs.map +0 -1
  129. package/dist/registry-sqlite-DrCW1aRK.mjs +0 -593
  130. package/dist/registry-sqlite-DrCW1aRK.mjs.map +0 -1
  131. package/dist/router-CXUGsv85.mjs +0 -3
  132. package/dist/run-BCpbxqA5.mjs.map +0 -1
  133. package/dist/search-Bf3Kub3F.mjs +0 -3
  134. package/dist/server-DgmAHyFK.mjs +0 -403
  135. package/dist/server-DgmAHyFK.mjs.map +0 -1
  136. package/dist/session-keepalive-DxjTHGcK.mjs +0 -582
  137. package/dist/session-keepalive-DxjTHGcK.mjs.map +0 -1
  138. package/dist/sqlite-CqsTy6xo.mjs +0 -936
  139. package/dist/sqlite-CqsTy6xo.mjs.map +0 -1
  140. package/dist/state-DW8zdweW.mjs +0 -3
  141. package/dist/state-HyjqTihC.mjs +0 -76
  142. package/dist/state-HyjqTihC.mjs.map +0 -1
  143. package/dist/themes-CTaOj3e1.mjs +0 -148
  144. package/dist/themes-CTaOj3e1.mjs.map +0 -1
  145. package/dist/tools-BcOKPBtg.mjs +0 -1489
  146. package/dist/tools-BcOKPBtg.mjs.map +0 -1
  147. package/dist/tools-C-_l0o8Z.mjs +0 -6
  148. package/dist/trace-CtV0RO6n.mjs +0 -137
  149. package/dist/trace-CtV0RO6n.mjs.map +0 -1
  150. package/dist/utils-BrX9FLeK.mjs +0 -3
  151. package/dist/vault-indexer-BG1sEVyM.mjs +0 -537
  152. package/dist/vault-indexer-BG1sEVyM.mjs.map +0 -1
  153. package/dist/wakeup-D8n33hYV.mjs +0 -335
  154. package/dist/wakeup-D8n33hYV.mjs.map +0 -1
  155. package/dist/work-queue-worker-4tLa5gpQ.mjs +0 -567
  156. package/dist/work-queue-worker-4tLa5gpQ.mjs.map +0 -1
  157. package/dist/work-queue-worker-rFZvTO3p.mjs +0 -11
  158. package/dist/zettelkasten-CtGHQWPU.mjs.map +0 -1
@@ -1 +0,0 @@
1
- {"version":3,"file":"federation-db-BTyoufBh.mjs","names":[],"sources":["../src/storage/sqlite/federation-schema.ts","../src/storage/sqlite/federation-db.ts"],"sourcesContent":["/**\n * SQLite DDL for the PAI federation database (federation.db).\n *\n * The federation DB is the cross-project search index — a single SQLite file\n * at ~/.pai/federation.db that holds chunked text from every registered\n * project's memory/ and Notes/ directories.\n *\n * Tables:\n * - memory_files — file-level metadata (hash, mtime, size) for change detection\n * - memory_chunks — chunked text with line numbers, tier classification, and optional embedding\n * - memory_fts — FTS5 virtual table backed by memory_chunks text\n *\n * Vault tables (vault_files, vault_aliases, vault_links, vault_name_index, vault_health)\n * have been migrated to Postgres (docker/init.sql) and are no longer created here.\n *\n * Schema version history:\n * v1 — initial schema (BM25 search only)\n * v2 — added embedding BLOB column to memory_chunks (Phase 2.5, vector search)\n * v3 — added vault tables (now removed — vault tables live in Postgres)\n */\n\nimport type { Database } from \"better-sqlite3\";\n\n/** Current schema version. Bump when adding new columns or tables. */\nexport const SCHEMA_VERSION = 5;\n\nexport const FEDERATION_SCHEMA_SQL = `\nPRAGMA journal_mode = WAL;\nPRAGMA foreign_keys = ON;\n\nCREATE TABLE IF NOT EXISTS memory_files (\n project_id INTEGER NOT NULL,\n path TEXT NOT NULL,\n source TEXT NOT NULL DEFAULT 'memory',\n tier TEXT NOT NULL DEFAULT 'topic',\n hash TEXT NOT NULL,\n mtime INTEGER NOT NULL,\n size INTEGER NOT NULL,\n PRIMARY KEY (project_id, path)\n);\n\nCREATE TABLE IF NOT EXISTS memory_chunks (\n id TEXT PRIMARY KEY,\n project_id INTEGER NOT NULL,\n source TEXT NOT NULL DEFAULT 'memory',\n tier TEXT NOT NULL DEFAULT 'topic',\n path TEXT NOT NULL,\n start_line INTEGER NOT NULL,\n end_line INTEGER NOT NULL,\n hash TEXT NOT NULL,\n text TEXT NOT NULL,\n updated_at INTEGER NOT NULL,\n last_accessed_at INTEGER,\n relevance_score REAL DEFAULT 0.5,\n embedding BLOB\n);\n\nCREATE VIRTUAL TABLE IF NOT EXISTS memory_fts USING fts5(\n text,\n id UNINDEXED,\n project_id UNINDEXED,\n path UNINDEXED,\n source UNINDEXED,\n tier UNINDEXED,\n start_line UNINDEXED,\n end_line UNINDEXED\n);\n\nCREATE INDEX IF NOT EXISTS idx_mc_project ON memory_chunks(project_id);\nCREATE INDEX IF NOT EXISTS idx_mc_source ON memory_chunks(project_id, source);\nCREATE INDEX IF NOT EXISTS idx_mc_tier ON memory_chunks(tier);\nCREATE INDEX IF NOT EXISTS idx_mf_project ON memory_files(project_id);\n\nCREATE TABLE IF NOT EXISTS kg_entities (\n entity_id TEXT PRIMARY KEY,\n tenant_id TEXT NOT NULL DEFAULT 'default',\n name TEXT NOT NULL,\n type TEXT NOT NULL DEFAULT 'unknown',\n description TEXT,\n first_seen INTEGER,\n last_seen INTEGER,\n mention_count INTEGER NOT NULL DEFAULT 1,\n feedback_weight REAL NOT NULL DEFAULT 0.5,\n UNIQUE(tenant_id, entity_id)\n);\n\nCREATE INDEX IF NOT EXISTS idx_kge_tenant ON kg_entities(tenant_id);\nCREATE INDEX IF NOT EXISTS idx_kge_name ON kg_entities(tenant_id, name);\nCREATE INDEX IF NOT EXISTS idx_kge_type ON kg_entities(tenant_id, type);\n`;\n\n/**\n * Apply the full federation schema to an open database.\n *\n * Idempotent — all statements use IF NOT EXISTS so calling this on an\n * already-initialised database is safe.\n *\n * Also runs any necessary migrations for existing databases (e.g. adding the\n * embedding column to an older schema that was created without it).\n */\nexport function initializeFederationSchema(db: Database): void {\n db.exec(FEDERATION_SCHEMA_SQL);\n runMigrations(db);\n}\n\n// ---------------------------------------------------------------------------\n// Migrations\n// ---------------------------------------------------------------------------\n\n/**\n * Apply incremental migrations to an existing database.\n *\n * Each migration is idempotent — safe to call on a database that has already\n * been migrated.\n */\nfunction runMigrations(db: Database): void {\n const columns = db.prepare(\"PRAGMA table_info(memory_chunks)\").all() as Array<{\n name: string;\n }>;\n\n // Migration v1→v2: add embedding BLOB column (schema v2, Phase 2.5)\n const hasEmbedding = columns.some((c) => c.name === \"embedding\");\n if (!hasEmbedding) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN embedding BLOB\");\n }\n\n // Create the partial index for embedded chunks (safe now that the column exists)\n db.exec(\n \"CREATE INDEX IF NOT EXISTS idx_mc_embedding ON memory_chunks(id) WHERE embedding IS NOT NULL\",\n );\n\n // Migration v4→v5: add last_accessed_at and relevance_score columns (QW2 + MR2)\n const hasLastAccessedAt = columns.some((c) => c.name === \"last_accessed_at\");\n if (!hasLastAccessedAt) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN last_accessed_at INTEGER\");\n }\n\n const hasRelevanceScore = columns.some((c) => c.name === \"relevance_score\");\n if (!hasRelevanceScore) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN relevance_score REAL DEFAULT 0.5\");\n }\n}\n","/**\n * Database connection helper for the PAI federation DB.\n *\n * Uses better-sqlite3 (synchronous API) to open or create federation.db.\n * On first open it runs the full DDL via initializeFederationSchema().\n */\n\nimport { mkdirSync } from \"node:fs\";\nimport { homedir } from \"node:os\";\nimport { dirname, join } from \"node:path\";\nimport BetterSqlite3 from \"better-sqlite3\";\nimport type { Database } from \"better-sqlite3\";\nimport { initializeFederationSchema } from \"./federation-schema.js\";\nimport { paiHomePath, resolvePaiFile } from \"../../config/pai-home.js\";\n\nexport type { Database };\n\n/** Old federation.db path, inside the ~/.pai/ directory (pre-2026-09-19). */\nexport function oldFederationPath(): string {\n return join(homedir(), \".pai\", \"federation.db\");\n}\n\n/** Federation DB path: PAI_HOME/federation.db if present, else the old\n * ~/.pai/federation.db (one-time stderr notice), else the new path. */\nexport function federationDbPath(): string {\n return resolvePaiFile(paiHomePath(\"federation.db\"), [oldFederationPath()], \"pai config migrate --federation\");\n}\n\n/**\n * Open (or create) the PAI federation database.\n *\n * @param path Absolute path to federation.db. Defaults to PAI_HOME/federation.db\n * (falling back to the pre-2026-09-19 ~/.pai/federation.db).\n * @returns An open better-sqlite3 Database instance.\n *\n * Side effects on first call:\n * - Creates the parent directory if it does not exist.\n * - Enables WAL journal mode.\n * - Runs initializeFederationSchema() to ensure tables exist.\n */\nexport function openFederation(path: string = federationDbPath()): Database {\n // Ensure the directory exists before SQLite tries to create the file\n mkdirSync(dirname(path), { recursive: true });\n\n const db = new BetterSqlite3(path);\n\n // WAL gives better concurrent read performance and crash safety\n db.pragma(\"journal_mode = WAL\");\n db.pragma(\"foreign_keys = ON\");\n\n // Apply schema (idempotent — all statements use IF NOT EXISTS)\n initializeFederationSchema(db);\n\n return db;\n}\n"],"mappings":";;;;;;;AA0BA,MAAa,wBAAwB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA0ErC,SAAgB,2BAA2B,IAAoB;AAC7D,IAAG,KAAK,sBAAsB;AAC9B,eAAc,GAAG;;;;;;;;AAanB,SAAS,cAAc,IAAoB;CACzC,MAAM,UAAU,GAAG,QAAQ,mCAAmC,CAAC,KAAK;AAMpE,KAAI,CADiB,QAAQ,MAAM,MAAM,EAAE,SAAS,YAAY,CAE9D,IAAG,KAAK,sDAAsD;AAIhE,IAAG,KACD,+FACD;AAID,KAAI,CADsB,QAAQ,MAAM,MAAM,EAAE,SAAS,mBAAmB,CAE1E,IAAG,KAAK,gEAAgE;AAI1E,KAAI,CADsB,QAAQ,MAAM,MAAM,EAAE,SAAS,kBAAkB,CAEzE,IAAG,KAAK,wEAAwE;;;;;;;;;;;;ACzHpF,SAAgB,oBAA4B;AAC1C,QAAO,KAAK,SAAS,EAAE,QAAQ,gBAAgB;;;;AAKjD,SAAgB,mBAA2B;AACzC,QAAO,eAAe,YAAY,gBAAgB,EAAE,CAAC,mBAAmB,CAAC,EAAE,kCAAkC;;;;;;;;;;;;;;AAe/G,SAAgB,eAAe,OAAe,kBAAkB,EAAY;AAE1E,WAAU,QAAQ,KAAK,EAAE,EAAE,WAAW,MAAM,CAAC;CAE7C,MAAM,KAAK,IAAI,cAAc,KAAK;AAGlC,IAAG,OAAO,qBAAqB;AAC/B,IAAG,OAAO,oBAAoB;AAG9B,4BAA2B,GAAG;AAE9B,QAAO"}
@@ -1,3 +0,0 @@
1
- import { n as oldFederationPath, r as openFederation, t as federationDbPath } from "./federation-db-BTyoufBh.mjs";
2
-
3
- export { openFederation };
@@ -1,223 +0,0 @@
1
- import { existsSync, readdirSync } from "node:fs";
2
- import { homedir } from "node:os";
3
- import { basename, join, normalize } from "node:path";
4
- import { createHash } from "node:crypto";
5
-
6
- //#region src/memory/indexer/helpers.ts
7
- /**
8
- * Shared helpers for the PAI memory indexers.
9
- *
10
- * Contains utilities used by both the sync (SQLite) and async (StorageBackend)
11
- * indexer paths: hashing, chunk ID generation, directory walking, and path guards.
12
- */
13
- /**
14
- * Classify a relative file path into one of the four memory tiers.
15
- *
16
- * Rules (in priority order):
17
- * - MEMORY.md anywhere in memory/ → 'evergreen'
18
- * - YYYY-MM-DD.md in memory/ → 'daily'
19
- * - anything else in memory/ → 'topic'
20
- * - anything in Notes/ → 'session'
21
- */
22
- function detectTier(relativePath) {
23
- const p = relativePath.replace(/\\/g, "/").replace(/^\.\//, "");
24
- if (p.startsWith("Notes/") || p === "Notes") return "session";
25
- const fileName = basename(p);
26
- if (fileName === "MEMORY.md") return "evergreen";
27
- if (/^\d{4}-\d{2}-\d{2}\.md$/.test(fileName)) return "daily";
28
- return "topic";
29
- }
30
- /**
31
- * Generate a deterministic chunk ID from its coordinates.
32
- * Format: sha256("projectId:path:chunkIndex:startLine:endLine")
33
- *
34
- * The chunkIndex (0-based position within the file) is included so that
35
- * chunks with approximated line numbers (e.g. from splitBySentences) never
36
- * produce colliding IDs even when multiple chunks share the same startLine/endLine.
37
- */
38
- function chunkId(projectId, path, chunkIndex, startLine, endLine) {
39
- return createHash("sha256").update(`${projectId}:${path}:${chunkIndex}:${startLine}:${endLine}`).digest("hex");
40
- }
41
- /**
42
- * Yield to the Node.js event loop so that IPC server can process requests
43
- * during long index runs.
44
- *
45
- * Uses setTimeout(10ms) rather than setImmediate — the 10ms pause gives the
46
- * event loop enough time to accept and process incoming IPC connections
47
- * (socket data, new connections, etc.). Without this, synchronous ONNX
48
- * inference blocks IPC for the full duration of each embedding (~50-100ms
49
- * per chunk).
50
- */
51
- function yieldToEventLoop() {
52
- return new Promise((resolve) => setTimeout(resolve, 10));
53
- }
54
- /**
55
- * Directories to ALWAYS skip, at any depth, during any directory walk.
56
- * These are build artifacts, dependency trees, and VCS internals that
57
- * should never be indexed regardless of where they appear in the tree.
58
- */
59
- const ALWAYS_SKIP_DIRS = new Set([
60
- ".git",
61
- "node_modules",
62
- "vendor",
63
- "Pods",
64
- "dist",
65
- "build",
66
- "out",
67
- "DerivedData",
68
- ".next",
69
- ".venv",
70
- "venv",
71
- "__pycache__",
72
- ".cache",
73
- ".bun",
74
- "snaps",
75
- ".Trashes",
76
- "worktrees"
77
- ]);
78
- /**
79
- * Directories to skip when doing a root-level content scan.
80
- * These are either already handled by dedicated scans or should never be indexed.
81
- */
82
- const ROOT_SCAN_SKIP_DIRS = new Set([
83
- "memory",
84
- "Notes",
85
- ".claude",
86
- ".DS_Store",
87
- ...ALWAYS_SKIP_DIRS
88
- ]);
89
- /**
90
- * Additional directories to skip at the content-scan level (first level below root).
91
- * These are common macOS/Linux home-directory or repo noise directories that are
92
- * never meaningful as project content.
93
- */
94
- const CONTENT_SCAN_SKIP_DIRS = new Set([
95
- "Library",
96
- "Applications",
97
- "Music",
98
- "Movies",
99
- "Pictures",
100
- "Desktop",
101
- "Downloads",
102
- "Public",
103
- "coverage",
104
- ...ALWAYS_SKIP_DIRS
105
- ]);
106
- /**
107
- * Safety cap: maximum number of .md files collected per project scan.
108
- * Prevents runaway scans on huge root paths (e.g. home directory).
109
- * Projects with more files than this are scanned up to the cap only.
110
- */
111
- const MAX_FILES_PER_PROJECT = 5e3;
112
- /**
113
- * Maximum recursion depth for directory walks.
114
- * Prevents deep traversal of large directory trees (e.g. development repos).
115
- * Depth 0 = the given directory itself (no recursion).
116
- * Value 6 allows: root → subdirs → sub-subdirs → ... up to 6 levels.
117
- * Sufficient for memory/, Notes/, and typical docs structures.
118
- */
119
- const MAX_WALK_DEPTH = 6;
120
- /**
121
- * Recursively collect all .md files under a directory.
122
- * Returns absolute paths. Stops early if the accumulated count hits the cap
123
- * or if the recursion depth exceeds MAX_WALK_DEPTH.
124
- *
125
- * @param dir Directory to scan.
126
- * @param acc Shared accumulator array (mutated in place for early exit).
127
- * @param cap Maximum number of files to collect (across all recursive calls).
128
- * @param depth Current recursion depth (0 = the initial call).
129
- */
130
- function walkMdFiles(dir, acc, cap = MAX_FILES_PER_PROJECT, depth = 0) {
131
- const results = acc ?? [];
132
- if (!existsSync(dir)) return results;
133
- if (results.length >= cap) return results;
134
- if (depth > MAX_WALK_DEPTH) return results;
135
- try {
136
- for (const entry of readdirSync(dir, { withFileTypes: true })) {
137
- if (results.length >= cap) break;
138
- if (entry.isSymbolicLink()) continue;
139
- if (ALWAYS_SKIP_DIRS.has(entry.name)) continue;
140
- const full = join(dir, entry.name);
141
- if (entry.isDirectory()) walkMdFiles(full, results, cap, depth + 1);
142
- else if (entry.isFile() && entry.name.endsWith(".md")) results.push(full);
143
- }
144
- } catch {}
145
- return results;
146
- }
147
- /**
148
- * Recursively collect all .md files under rootPath, excluding directories
149
- * that are already covered by dedicated scans (memory/, Notes/) and
150
- * common noise directories (.git, node_modules, etc.).
151
- *
152
- * Returns absolute paths for files NOT already handled by the specific scanners.
153
- * Stops collecting once MAX_FILES_PER_PROJECT is reached.
154
- */
155
- function walkContentFiles(rootPath) {
156
- if (!existsSync(rootPath)) return [];
157
- const results = [];
158
- try {
159
- for (const entry of readdirSync(rootPath, { withFileTypes: true })) {
160
- if (results.length >= MAX_FILES_PER_PROJECT) break;
161
- if (entry.isSymbolicLink()) continue;
162
- if (ROOT_SCAN_SKIP_DIRS.has(entry.name)) continue;
163
- if (CONTENT_SCAN_SKIP_DIRS.has(entry.name)) continue;
164
- const full = join(rootPath, entry.name);
165
- if (entry.isDirectory()) walkMdFiles(full, results, MAX_FILES_PER_PROJECT);
166
- else if (entry.isFile() && entry.name.endsWith(".md")) {
167
- if (entry.name !== "MEMORY.md") results.push(full);
168
- }
169
- }
170
- } catch {}
171
- return results;
172
- }
173
- /** Paths that must never be indexed — system/temp dirs that can contain backup snapshots. */
174
- const BLOCKED_ROOTS = new Set([
175
- "/tmp",
176
- "/private/tmp",
177
- "/var",
178
- "/private/var"
179
- ]);
180
- /**
181
- * Returns true if rootPath should skip the recursive content scan.
182
- *
183
- * Skips content scanning for:
184
- * - The home directory itself or any ancestor (too broad — millions of files)
185
- * - Git repositories (code repos — index memory/ and Notes/ only, not all .md files)
186
- *
187
- * The content scan is still useful for Obsidian vaults, Notes folders, and
188
- * other doc-centric project trees where ALL markdown files are meaningful.
189
- *
190
- * The memory/, Notes/, and claude_notes_dir scans always run regardless.
191
- */
192
- function isPathTooBroadForContentScan(rootPath) {
193
- const normalized = normalize(rootPath);
194
- if (BLOCKED_ROOTS.has(normalized)) return true;
195
- for (const blocked of BLOCKED_ROOTS) if (normalized.startsWith(blocked + "/")) return true;
196
- const home = homedir();
197
- if (home.startsWith(normalized) || normalized === "/") return true;
198
- if (normalized.startsWith(home)) {
199
- const rel = normalized.slice(home.length).replace(/^\//, "");
200
- if ((rel ? rel.split("/").length : 0) === 0) return true;
201
- }
202
- if (existsSync(join(normalized, ".git"))) return true;
203
- return false;
204
- }
205
- const SESSION_TITLE_RE = /^(\d{4})\s*-\s*(\d{4}-\d{2}-\d{2})\s*-\s*(.+)\.md$/;
206
- /**
207
- * Parse a session title from a Notes filename.
208
- * Format: "NNNN - YYYY-MM-DD - Descriptive Title.md"
209
- * Returns a synthetic chunk text like "Session #0086 2026-02-23: Pai Daemon Background Service"
210
- * or null if the filename doesn't match the expected pattern.
211
- */
212
- function parseSessionTitleChunk(fileName) {
213
- const m = SESSION_TITLE_RE.exec(fileName);
214
- if (!m) return null;
215
- const [, num, date, title] = m;
216
- return `Session #${num} ${date}: ${title}`;
217
- }
218
- /** Number of files to process before yielding to the event loop inside indexProject. */
219
- const INDEX_YIELD_EVERY = 10;
220
-
221
- //#endregion
222
- export { parseSessionTitleChunk as a, yieldToEventLoop as c, isPathTooBroadForContentScan as i, chunkId as n, walkContentFiles as o, detectTier as r, walkMdFiles as s, INDEX_YIELD_EVERY as t };
223
- //# sourceMappingURL=helpers-BRCJg0G3.mjs.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"helpers-BRCJg0G3.mjs","names":[],"sources":["../src/memory/indexer/helpers.ts"],"sourcesContent":["/**\n * Shared helpers for the PAI memory indexers.\n *\n * Contains utilities used by both the sync (SQLite) and async (StorageBackend)\n * indexer paths: hashing, chunk ID generation, directory walking, and path guards.\n */\n\nimport { readdirSync, existsSync } from \"node:fs\";\nimport { createHash } from \"node:crypto\";\nimport { sha256File } from \"../../utils/hash.js\";\nimport { join, normalize } from \"node:path\";\nimport { homedir } from \"node:os\";\nimport { basename } from \"node:path\";\n\n// ---------------------------------------------------------------------------\n// Tier detection\n// ---------------------------------------------------------------------------\n\n/**\n * Classify a relative file path into one of the four memory tiers.\n *\n * Rules (in priority order):\n * - MEMORY.md anywhere in memory/ → 'evergreen'\n * - YYYY-MM-DD.md in memory/ → 'daily'\n * - anything else in memory/ → 'topic'\n * - anything in Notes/ → 'session'\n */\nexport function detectTier(\n relativePath: string,\n): \"evergreen\" | \"daily\" | \"topic\" | \"session\" {\n // Normalise to forward slashes and strip leading ./\n const p = relativePath.replace(/\\\\/g, \"/\").replace(/^\\.\\//, \"\");\n\n // Notes directory → session tier\n if (p.startsWith(\"Notes/\") || p === \"Notes\") {\n return \"session\";\n }\n\n const fileName = basename(p);\n\n // MEMORY.md (case-sensitive match) → evergreen\n if (fileName === \"MEMORY.md\") {\n return \"evergreen\";\n }\n\n // YYYY-MM-DD.md → daily\n if (/^\\d{4}-\\d{2}-\\d{2}\\.md$/.test(fileName)) {\n return \"daily\";\n }\n\n // Default for memory/ files\n return \"topic\";\n}\n\n// ---------------------------------------------------------------------------\n// Hashing and chunk ID generation\n// ---------------------------------------------------------------------------\n\n// sha256File imported from ../../utils/hash.js\nexport { sha256File } from \"../../utils/hash.js\";\n\n/**\n * Generate a deterministic chunk ID from its coordinates.\n * Format: sha256(\"projectId:path:chunkIndex:startLine:endLine\")\n *\n * The chunkIndex (0-based position within the file) is included so that\n * chunks with approximated line numbers (e.g. from splitBySentences) never\n * produce colliding IDs even when multiple chunks share the same startLine/endLine.\n */\nexport function chunkId(\n projectId: number,\n path: string,\n chunkIndex: number,\n startLine: number,\n endLine: number,\n): string {\n return createHash(\"sha256\")\n .update(`${projectId}:${path}:${chunkIndex}:${startLine}:${endLine}`)\n .digest(\"hex\");\n}\n\n// ---------------------------------------------------------------------------\n// Event loop yield\n// ---------------------------------------------------------------------------\n\n/**\n * Yield to the Node.js event loop so that IPC server can process requests\n * during long index runs.\n *\n * Uses setTimeout(10ms) rather than setImmediate — the 10ms pause gives the\n * event loop enough time to accept and process incoming IPC connections\n * (socket data, new connections, etc.). Without this, synchronous ONNX\n * inference blocks IPC for the full duration of each embedding (~50-100ms\n * per chunk).\n */\nexport function yieldToEventLoop(): Promise<void> {\n return new Promise((resolve) => setTimeout(resolve, 10));\n}\n\n// ---------------------------------------------------------------------------\n// Directory skip sets\n// ---------------------------------------------------------------------------\n\n/**\n * Directories to ALWAYS skip, at any depth, during any directory walk.\n * These are build artifacts, dependency trees, and VCS internals that\n * should never be indexed regardless of where they appear in the tree.\n */\nexport const ALWAYS_SKIP_DIRS = new Set([\n // Version control\n \".git\",\n // Dependency directories (any language)\n \"node_modules\",\n \"vendor\",\n \"Pods\", // CocoaPods (iOS/macOS)\n // Build / compile output\n \"dist\",\n \"build\",\n \"out\",\n \"DerivedData\", // Xcode\n \".next\", // Next.js\n // Python virtual environments and caches\n \".venv\",\n \"venv\",\n \"__pycache__\",\n // General caches\n \".cache\",\n \".bun\",\n // Backup snapshots (Carbon Copy Cloner, Time Machine, etc.)\n \"snaps\",\n \".Trashes\",\n // Worker worktrees — full repo copies checked out per worker run\n \"worktrees\",\n]);\n\n/**\n * Directories to skip when doing a root-level content scan.\n * These are either already handled by dedicated scans or should never be indexed.\n */\nexport const ROOT_SCAN_SKIP_DIRS = new Set([\n \"memory\",\n \"Notes\",\n \".claude\",\n \".DS_Store\",\n // Everything in ALWAYS_SKIP_DIRS is also excluded at root level\n ...ALWAYS_SKIP_DIRS,\n]);\n\n/**\n * Additional directories to skip at the content-scan level (first level below root).\n * These are common macOS/Linux home-directory or repo noise directories that are\n * never meaningful as project content.\n */\nexport const CONTENT_SCAN_SKIP_DIRS = new Set([\n // macOS home directory standard folders\n \"Library\",\n \"Applications\",\n \"Music\",\n \"Movies\",\n \"Pictures\",\n \"Desktop\",\n \"Downloads\",\n \"Public\",\n // Common dev noise\n \"coverage\",\n // Everything in ALWAYS_SKIP_DIRS is also excluded at this level\n ...ALWAYS_SKIP_DIRS,\n]);\n\n// ---------------------------------------------------------------------------\n// Directory walkers\n// ---------------------------------------------------------------------------\n\n/**\n * Safety cap: maximum number of .md files collected per project scan.\n * Prevents runaway scans on huge root paths (e.g. home directory).\n * Projects with more files than this are scanned up to the cap only.\n */\nconst MAX_FILES_PER_PROJECT = 5_000;\n\n/**\n * Maximum recursion depth for directory walks.\n * Prevents deep traversal of large directory trees (e.g. development repos).\n * Depth 0 = the given directory itself (no recursion).\n * Value 6 allows: root → subdirs → sub-subdirs → ... up to 6 levels.\n * Sufficient for memory/, Notes/, and typical docs structures.\n */\nconst MAX_WALK_DEPTH = 6;\n\n/**\n * Recursively collect all .md files under a directory.\n * Returns absolute paths. Stops early if the accumulated count hits the cap\n * or if the recursion depth exceeds MAX_WALK_DEPTH.\n *\n * @param dir Directory to scan.\n * @param acc Shared accumulator array (mutated in place for early exit).\n * @param cap Maximum number of files to collect (across all recursive calls).\n * @param depth Current recursion depth (0 = the initial call).\n */\nexport function walkMdFiles(\n dir: string,\n acc?: string[],\n cap = MAX_FILES_PER_PROJECT,\n depth = 0,\n): string[] {\n const results = acc ?? [];\n if (!existsSync(dir)) return results;\n if (results.length >= cap) return results;\n if (depth > MAX_WALK_DEPTH) return results;\n\n try {\n for (const entry of readdirSync(dir, { withFileTypes: true })) {\n if (results.length >= cap) break;\n if (entry.isSymbolicLink()) continue;\n // Skip known junk directories at every recursion depth\n if (ALWAYS_SKIP_DIRS.has(entry.name)) continue;\n const full = join(dir, entry.name);\n if (entry.isDirectory()) {\n walkMdFiles(full, results, cap, depth + 1);\n } else if (entry.isFile() && entry.name.endsWith(\".md\")) {\n results.push(full);\n }\n }\n } catch {\n // Unreadable directory — skip\n }\n return results;\n}\n\n/**\n * Recursively collect all .md files under rootPath, excluding directories\n * that are already covered by dedicated scans (memory/, Notes/) and\n * common noise directories (.git, node_modules, etc.).\n *\n * Returns absolute paths for files NOT already handled by the specific scanners.\n * Stops collecting once MAX_FILES_PER_PROJECT is reached.\n */\nexport function walkContentFiles(rootPath: string): string[] {\n if (!existsSync(rootPath)) return [];\n\n const results: string[] = [];\n try {\n for (const entry of readdirSync(rootPath, { withFileTypes: true })) {\n if (results.length >= MAX_FILES_PER_PROJECT) break;\n if (entry.isSymbolicLink()) continue;\n if (ROOT_SCAN_SKIP_DIRS.has(entry.name)) continue;\n if (CONTENT_SCAN_SKIP_DIRS.has(entry.name)) continue;\n\n const full = join(rootPath, entry.name);\n if (entry.isDirectory()) {\n walkMdFiles(full, results, MAX_FILES_PER_PROJECT);\n } else if (entry.isFile() && entry.name.endsWith(\".md\")) {\n // Skip root-level MEMORY.md — handled by the dedicated evergreen scan\n if (entry.name !== \"MEMORY.md\") {\n results.push(full);\n }\n }\n }\n } catch {\n // Unreadable directory — skip\n }\n return results;\n}\n\n// ---------------------------------------------------------------------------\n// Path safety guard\n// ---------------------------------------------------------------------------\n\n/** Paths that must never be indexed — system/temp dirs that can contain backup snapshots. */\nconst BLOCKED_ROOTS = new Set([\"/tmp\", \"/private/tmp\", \"/var\", \"/private/var\"]);\n\n/**\n * Returns true if rootPath should skip the recursive content scan.\n *\n * Skips content scanning for:\n * - The home directory itself or any ancestor (too broad — millions of files)\n * - Git repositories (code repos — index memory/ and Notes/ only, not all .md files)\n *\n * The content scan is still useful for Obsidian vaults, Notes folders, and\n * other doc-centric project trees where ALL markdown files are meaningful.\n *\n * The memory/, Notes/, and claude_notes_dir scans always run regardless.\n */\nexport function isPathTooBroadForContentScan(rootPath: string): boolean {\n const normalized = normalize(rootPath);\n\n // Block system/temp directories outright (CCC snapshots live here)\n if (BLOCKED_ROOTS.has(normalized)) return true;\n for (const blocked of BLOCKED_ROOTS) {\n if (normalized.startsWith(blocked + \"/\")) return true;\n }\n\n const home = homedir();\n\n // Skip the home directory itself or any ancestor of home\n if (home.startsWith(normalized) || normalized === \"/\") {\n return true;\n }\n\n // Skip home directory itself (depth 0)\n if (normalized.startsWith(home)) {\n const rel = normalized.slice(home.length).replace(/^\\//, \"\");\n const depth = rel ? rel.split(\"/\").length : 0;\n if (depth === 0) return true;\n }\n\n // Skip git repositories — content scan is only for doc-centric projects\n // (Obsidian vaults, knowledge bases). Code repos use memory/ and Notes/ only.\n if (existsSync(join(normalized, \".git\"))) {\n return true;\n }\n\n return false;\n}\n\n// ---------------------------------------------------------------------------\n// Session title parser\n// ---------------------------------------------------------------------------\n\nconst SESSION_TITLE_RE = /^(\\d{4})\\s*-\\s*(\\d{4}-\\d{2}-\\d{2})\\s*-\\s*(.+)\\.md$/;\n\n/**\n * Parse a session title from a Notes filename.\n * Format: \"NNNN - YYYY-MM-DD - Descriptive Title.md\"\n * Returns a synthetic chunk text like \"Session #0086 2026-02-23: Pai Daemon Background Service\"\n * or null if the filename doesn't match the expected pattern.\n */\nexport function parseSessionTitleChunk(fileName: string): string | null {\n const m = SESSION_TITLE_RE.exec(fileName);\n if (!m) return null;\n const [, num, date, title] = m;\n return `Session #${num} ${date}: ${title}`;\n}\n\n/** Number of files to process before yielding to the event loop inside indexProject. */\nexport const INDEX_YIELD_EVERY = 10;\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;AA2BA,SAAgB,WACd,cAC6C;CAE7C,MAAM,IAAI,aAAa,QAAQ,OAAO,IAAI,CAAC,QAAQ,SAAS,GAAG;AAG/D,KAAI,EAAE,WAAW,SAAS,IAAI,MAAM,QAClC,QAAO;CAGT,MAAM,WAAW,SAAS,EAAE;AAG5B,KAAI,aAAa,YACf,QAAO;AAIT,KAAI,0BAA0B,KAAK,SAAS,CAC1C,QAAO;AAIT,QAAO;;;;;;;;;;AAkBT,SAAgB,QACd,WACA,MACA,YACA,WACA,SACQ;AACR,QAAO,WAAW,SAAS,CACxB,OAAO,GAAG,UAAU,GAAG,KAAK,GAAG,WAAW,GAAG,UAAU,GAAG,UAAU,CACpE,OAAO,MAAM;;;;;;;;;;;;AAiBlB,SAAgB,mBAAkC;AAChD,QAAO,IAAI,SAAS,YAAY,WAAW,SAAS,GAAG,CAAC;;;;;;;AAY1D,MAAa,mBAAmB,IAAI,IAAI;CAEtC;CAEA;CACA;CACA;CAEA;CACA;CACA;CACA;CACA;CAEA;CACA;CACA;CAEA;CACA;CAEA;CACA;CAEA;CACD,CAAC;;;;;AAMF,MAAa,sBAAsB,IAAI,IAAI;CACzC;CACA;CACA;CACA;CAEA,GAAG;CACJ,CAAC;;;;;;AAOF,MAAa,yBAAyB,IAAI,IAAI;CAE5C;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CAEA;CAEA,GAAG;CACJ,CAAC;;;;;;AAWF,MAAM,wBAAwB;;;;;;;;AAS9B,MAAM,iBAAiB;;;;;;;;;;;AAYvB,SAAgB,YACd,KACA,KACA,MAAM,uBACN,QAAQ,GACE;CACV,MAAM,UAAU,OAAO,EAAE;AACzB,KAAI,CAAC,WAAW,IAAI,CAAE,QAAO;AAC7B,KAAI,QAAQ,UAAU,IAAK,QAAO;AAClC,KAAI,QAAQ,eAAgB,QAAO;AAEnC,KAAI;AACF,OAAK,MAAM,SAAS,YAAY,KAAK,EAAE,eAAe,MAAM,CAAC,EAAE;AAC7D,OAAI,QAAQ,UAAU,IAAK;AAC3B,OAAI,MAAM,gBAAgB,CAAE;AAE5B,OAAI,iBAAiB,IAAI,MAAM,KAAK,CAAE;GACtC,MAAM,OAAO,KAAK,KAAK,MAAM,KAAK;AAClC,OAAI,MAAM,aAAa,CACrB,aAAY,MAAM,SAAS,KAAK,QAAQ,EAAE;YACjC,MAAM,QAAQ,IAAI,MAAM,KAAK,SAAS,MAAM,CACrD,SAAQ,KAAK,KAAK;;SAGhB;AAGR,QAAO;;;;;;;;;;AAWT,SAAgB,iBAAiB,UAA4B;AAC3D,KAAI,CAAC,WAAW,SAAS,CAAE,QAAO,EAAE;CAEpC,MAAM,UAAoB,EAAE;AAC5B,KAAI;AACF,OAAK,MAAM,SAAS,YAAY,UAAU,EAAE,eAAe,MAAM,CAAC,EAAE;AAClE,OAAI,QAAQ,UAAU,sBAAuB;AAC7C,OAAI,MAAM,gBAAgB,CAAE;AAC5B,OAAI,oBAAoB,IAAI,MAAM,KAAK,CAAE;AACzC,OAAI,uBAAuB,IAAI,MAAM,KAAK,CAAE;GAE5C,MAAM,OAAO,KAAK,UAAU,MAAM,KAAK;AACvC,OAAI,MAAM,aAAa,CACrB,aAAY,MAAM,SAAS,sBAAsB;YACxC,MAAM,QAAQ,IAAI,MAAM,KAAK,SAAS,MAAM,EAErD;QAAI,MAAM,SAAS,YACjB,SAAQ,KAAK,KAAK;;;SAIlB;AAGR,QAAO;;;AAQT,MAAM,gBAAgB,IAAI,IAAI;CAAC;CAAQ;CAAgB;CAAQ;CAAe,CAAC;;;;;;;;;;;;;AAc/E,SAAgB,6BAA6B,UAA2B;CACtE,MAAM,aAAa,UAAU,SAAS;AAGtC,KAAI,cAAc,IAAI,WAAW,CAAE,QAAO;AAC1C,MAAK,MAAM,WAAW,cACpB,KAAI,WAAW,WAAW,UAAU,IAAI,CAAE,QAAO;CAGnD,MAAM,OAAO,SAAS;AAGtB,KAAI,KAAK,WAAW,WAAW,IAAI,eAAe,IAChD,QAAO;AAIT,KAAI,WAAW,WAAW,KAAK,EAAE;EAC/B,MAAM,MAAM,WAAW,MAAM,KAAK,OAAO,CAAC,QAAQ,OAAO,GAAG;AAE5D,OADc,MAAM,IAAI,MAAM,IAAI,CAAC,SAAS,OAC9B,EAAG,QAAO;;AAK1B,KAAI,WAAW,KAAK,YAAY,OAAO,CAAC,CACtC,QAAO;AAGT,QAAO;;AAOT,MAAM,mBAAmB;;;;;;;AAQzB,SAAgB,uBAAuB,UAAiC;CACtE,MAAM,IAAI,iBAAiB,KAAK,SAAS;AACzC,KAAI,CAAC,EAAG,QAAO;CACf,MAAM,GAAG,KAAK,MAAM,SAAS;AAC7B,QAAO,YAAY,IAAI,GAAG,KAAK,IAAI;;;AAIrC,MAAa,oBAAoB"}
@@ -1,5 +0,0 @@
1
- import { a as parseSessionTitleChunk } from "./helpers-BRCJg0G3.mjs";
2
- import { i as indexProjectWithBackend, n as indexAllWithBackend, r as indexFileWithBackend, t as embedChunksWithBackend } from "./async-C2Bm_Lal.mjs";
3
- import "./indexer-backend-isSLg6yE.mjs";
4
-
5
- export { embedChunksWithBackend };
@@ -1 +0,0 @@
1
- export { };
@@ -1,29 +0,0 @@
1
- import { createHash } from "node:crypto";
2
-
3
- //#region src/memory/kg-entity.ts
4
- /**
5
- * kg-entity.ts — Entity content-addressing with multi-tenant support.
6
- *
7
- * Provides UUID5-style deterministic content hashes for KG entities and edges,
8
- * ensuring that the same entity name always maps to the same ID within a tenant.
9
- * This enables idempotent upserts and stable foreign keys for kg_triples.
10
- *
11
- * Multi-tenant support: each tenant namespace gets its own entity ID space.
12
- * The default tenant is "default" for single-user deployments.
13
- */
14
- /**
15
- * Generate a deterministic entity ID (UUID5-style) for a given name and tenant.
16
- *
17
- * The ID is a hex digest derived from "tenant_id:name" so the same entity
18
- * always receives the same ID within a tenant namespace.
19
- *
20
- * @param name Entity name (case-preserved)
21
- * @param tenantId Tenant namespace (default: "default")
22
- */
23
- function entityContentId(name, tenantId = "default") {
24
- return createHash("sha256").update(`entity:${tenantId}:${name}`).digest("hex").slice(0, 32);
25
- }
26
-
27
- //#endregion
28
- export { entityContentId as t };
29
- //# sourceMappingURL=kg-entity-DCOcsVFD.mjs.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"kg-entity-DCOcsVFD.mjs","names":[],"sources":["../src/memory/kg-entity.ts"],"sourcesContent":["/**\n * kg-entity.ts — Entity content-addressing with multi-tenant support.\n *\n * Provides UUID5-style deterministic content hashes for KG entities and edges,\n * ensuring that the same entity name always maps to the same ID within a tenant.\n * This enables idempotent upserts and stable foreign keys for kg_triples.\n *\n * Multi-tenant support: each tenant namespace gets its own entity ID space.\n * The default tenant is \"default\" for single-user deployments.\n */\n\nimport { createHash } from \"node:crypto\";\n\n// ---------------------------------------------------------------------------\n// Types\n// ---------------------------------------------------------------------------\n\nexport interface KgEntity {\n entity_id: string;\n tenant_id: string;\n name: string;\n type: string;\n description?: string;\n first_seen?: number;\n last_seen?: number;\n mention_count: number;\n feedback_weight: number;\n}\n\nexport interface KgEntityUpsertParams {\n name: string;\n type?: string;\n description?: string;\n tenantId?: string;\n}\n\n// ---------------------------------------------------------------------------\n// Content addressing\n// ---------------------------------------------------------------------------\n\n/**\n * Generate a deterministic entity ID (UUID5-style) for a given name and tenant.\n *\n * The ID is a hex digest derived from \"tenant_id:name\" so the same entity\n * always receives the same ID within a tenant namespace.\n *\n * @param name Entity name (case-preserved)\n * @param tenantId Tenant namespace (default: \"default\")\n */\nexport function entityContentId(name: string, tenantId = \"default\"): string {\n return createHash(\"sha256\")\n .update(`entity:${tenantId}:${name}`)\n .digest(\"hex\")\n .slice(0, 32); // 128-bit hex string — UUID5-compatible length\n}\n\n/**\n * Generate a deterministic edge ID for a (source, relation, target) triple\n * within a tenant namespace.\n *\n * @param source Source entity name\n * @param relation Relation/predicate verb phrase\n * @param target Target entity name\n * @param tenantId Tenant namespace (default: \"default\")\n */\nexport function edgeContentId(\n source: string,\n relation: string,\n target: string,\n tenantId = \"default\"\n): string {\n return createHash(\"sha256\")\n .update(`edge:${tenantId}:${source}:${relation}:${target}`)\n .digest(\"hex\")\n .slice(0, 32);\n}\n\n// SQLite's kg_entities CRUD (upsertKgEntity/findKgEntity/listKgEntities/\n// updateEntityFeedbackWeight) lives in storage/sqlite/kg-entity.ts —\n// PostgresBackend implements the same operations inline against its own\n// pool, so there is no cross-backend function to share here.\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAiDA,SAAgB,gBAAgB,MAAc,WAAW,WAAmB;AAC1E,QAAO,WAAW,SAAS,CACxB,OAAO,UAAU,SAAS,GAAG,OAAO,CACpC,OAAO,MAAM,CACb,MAAM,GAAG,GAAG"}
@@ -1,191 +0,0 @@
1
- import { n as TITLE_STOP_WORDS } from "./stop-words-BdQuaE9K.mjs";
2
- import "./embeddings-Bx3q0QOY.mjs";
3
- import { t as zettelThemes } from "./themes-CTaOj3e1.mjs";
4
- import { mkdirSync, writeFileSync } from "node:fs";
5
- import { dirname, join } from "node:path";
6
-
7
- //#region src/graph/latent-ideas.ts
8
- /**
9
- * latent-ideas.ts — graph_latent_ideas and idea_materialize endpoint handlers
10
- *
11
- * "Latent ideas" are recurring themes in the vault that exist as embedding
12
- * clusters but have NO dedicated note written about them yet. PAI surfaces
13
- * these by running the same agglomerative clustering used by graph_clusters /
14
- * zettelThemes and then filtering OUT any cluster whose label is well-matched
15
- * by an existing note title.
16
- *
17
- * The materialize endpoint writes a new Markdown note to the vault filesystem
18
- * and returns its content so the plugin can open it immediately.
19
- */
20
- /**
21
- * Returns true when any existing vault note title closely matches the cluster
22
- * label — meaning a dedicated note already exists for this topic.
23
- *
24
- * Matching strategy (simple, fast, no embeddings needed):
25
- * 1. Lowercase both sides and split into words.
26
- * 2. Remove stop words from the label words.
27
- * 3. If ≥ 60% of the significant label words appear in a note title → match.
28
- */
29
- function labelMatchesTitle(label, title) {
30
- const labelWords = label.toLowerCase().split(/[\s\-_/]+/).filter((w) => w.length > 2 && !TITLE_STOP_WORDS.has(w));
31
- if (labelWords.length === 0) return false;
32
- const titleLower = title.toLowerCase();
33
- return labelWords.filter((w) => titleLower.includes(w)).length / labelWords.length >= .6;
34
- }
35
- /**
36
- * Check whether any note indexed in the vault has a title matching the label.
37
- * Fetches all vault file rows via StorageBackend for efficiency.
38
- */
39
- async function clusterHasMatchingNote(backend, label, notePaths) {
40
- const pathSet = new Set(notePaths);
41
- const rows = await backend.getAllVaultFiles();
42
- for (const row of rows) {
43
- if (!row.title) continue;
44
- if (pathSet.has(row.vaultPath)) continue;
45
- if (labelMatchesTitle(label, row.title)) return true;
46
- }
47
- return false;
48
- }
49
- function toSuggestedTitle(label) {
50
- return label.trim().split(/\s+/).map((w, i) => {
51
- const lower = w.toLowerCase();
52
- if (i === 0 && TITLE_STOP_WORDS.has(lower) && label.trim().split(/\s+/).length > 1) return "";
53
- return w.charAt(0).toUpperCase() + w.slice(1);
54
- }).filter(Boolean).join(" ") || label;
55
- }
56
- function mostCommonFolder(vaultPaths) {
57
- const counts = /* @__PURE__ */ new Map();
58
- for (const p of vaultPaths) {
59
- const parts = p.split("/");
60
- const folder = parts.length > 1 ? parts.slice(0, -1).join("/") : "";
61
- counts.set(folder, (counts.get(folder) ?? 0) + 1);
62
- }
63
- let best = "";
64
- let bestCount = 0;
65
- for (const [folder, count] of counts) if (count > bestCount) {
66
- bestCount = count;
67
- best = folder;
68
- }
69
- return best;
70
- }
71
- /**
72
- * Heuristic: vault notes are often stored in date-based folders like
73
- * "2026/03/15" or "Daily/2026-03". We extract the first numeric path
74
- * segment that looks like a year (2020-2030) and group by year+month.
75
- *
76
- * Falls back to counting distinct top-level folders.
77
- */
78
- function countDistinctSessions(vaultPaths) {
79
- const sessions = /* @__PURE__ */ new Set();
80
- const yearMonthRe = /\b(202\d)\D?(0[1-9]|1[0-2])\b/;
81
- for (const p of vaultPaths) {
82
- const m = yearMonthRe.exec(p);
83
- if (m) sessions.add(`${m[1]}-${m[2]}`);
84
- else {
85
- const topFolder = p.split("/")[0];
86
- sessions.add(topFolder);
87
- }
88
- }
89
- return sessions.size;
90
- }
91
- /**
92
- * Confidence combines:
93
- * - Cluster size (normalized, capped at 20 for max contribution)
94
- * - Folder diversity (0-1 already)
95
- * - Sessions count (normalized, capped at 5)
96
- *
97
- * Formula: 0.4 * sizeScore + 0.35 * folderDiversity + 0.25 * sessionScore
98
- */
99
- function calcConfidence(size, folderDiversity, sessionsCount) {
100
- const sizeScore = Math.min(size / 20, 1);
101
- const sessionScore = Math.min(sessionsCount / 5, 1);
102
- const raw = .4 * sizeScore + .35 * folderDiversity + .25 * sessionScore;
103
- return Math.round(raw * 100) / 100;
104
- }
105
- async function handleGraphLatentIdeas(backend, params) {
106
- const minClusterSize = params.min_cluster_size ?? 3;
107
- const maxIdeas = params.max_ideas ?? 15;
108
- const lookbackDays = params.lookback_days ?? 180;
109
- const similarityThreshold = params.similarity_threshold ?? .65;
110
- const { project_id: vaultProjectId } = params;
111
- if (!vaultProjectId) throw new Error("graph_latent_ideas: project_id is required (pass the vault project's numeric ID)");
112
- const themeResult = await zettelThemes(backend, {
113
- vaultProjectId,
114
- lookbackDays,
115
- minClusterSize,
116
- maxThemes: maxIdeas * 3,
117
- similarityThreshold
118
- });
119
- const ideas = [];
120
- let materializedCount = 0;
121
- for (const theme of themeResult.themes) {
122
- const notePaths = theme.notes.map((n) => n.path);
123
- if (await clusterHasMatchingNote(backend, theme.label, notePaths)) {
124
- materializedCount++;
125
- continue;
126
- }
127
- const suggestedFolder = mostCommonFolder(notePaths);
128
- const sessionsCount = countDistinctSessions(notePaths);
129
- const confidence = calcConfidence(theme.size, theme.folderDiversity, sessionsCount);
130
- const sourceNotes = theme.notes.map((n, idx) => ({
131
- vault_path: n.path,
132
- title: n.title ?? n.path.split("/").pop()?.replace(/\.md$/i, "") ?? n.path,
133
- relevance: Math.round((1 - idx / Math.max(theme.notes.length - 1, 1) * .5) * 100) / 100
134
- }));
135
- ideas.push({
136
- id: theme.id,
137
- label: theme.label,
138
- size: theme.size,
139
- confidence,
140
- source_notes: sourceNotes,
141
- suggested_title: toSuggestedTitle(theme.label),
142
- suggested_folder: suggestedFolder,
143
- sessions_count: sessionsCount
144
- });
145
- if (ideas.length >= maxIdeas) break;
146
- }
147
- ideas.sort((a, b) => b.confidence - a.confidence);
148
- return {
149
- ideas,
150
- total_clusters_analyzed: themeResult.themes.length + materializedCount,
151
- materialized_count: materializedCount
152
- };
153
- }
154
- function handleIdeaMaterialize(params, vaultPath) {
155
- const { idea_label, title, folder, source_paths } = params;
156
- const fileName = `${title.replace(/[/\\:*?"<>|]/g, "-")}.md`;
157
- const relFolder = folder.replace(/^\/+|\/+$/g, "");
158
- const vault_path = relFolder ? `${relFolder}/${fileName}` : fileName;
159
- const absPath = join(vaultPath, vault_path);
160
- const absDir = dirname(absPath);
161
- const wikilinks = source_paths.map((p) => {
162
- return `- [[${p.split("/").pop()?.replace(/\.md$/i, "") ?? p}]]`;
163
- }).join("\n");
164
- const links_created = source_paths.length;
165
- const content = [
166
- `# ${title}`,
167
- "",
168
- `*Materialized from latent idea: "${idea_label}"*`,
169
- `*Sources: ${links_created} notes*`,
170
- "",
171
- "## Related Notes",
172
- "",
173
- wikilinks || "*(no source notes)*",
174
- "",
175
- "## Notes",
176
- "",
177
- "<!-- Add your thoughts about this idea here -->",
178
- ""
179
- ].join("\n");
180
- mkdirSync(absDir, { recursive: true });
181
- writeFileSync(absPath, content, "utf-8");
182
- return {
183
- vault_path,
184
- content,
185
- links_created
186
- };
187
- }
188
-
189
- //#endregion
190
- export { handleGraphLatentIdeas, handleIdeaMaterialize };
191
- //# sourceMappingURL=latent-ideas-wwAXquGe.mjs.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"latent-ideas-wwAXquGe.mjs","names":[],"sources":["../src/graph/latent-ideas.ts"],"sourcesContent":["/**\n * latent-ideas.ts — graph_latent_ideas and idea_materialize endpoint handlers\n *\n * \"Latent ideas\" are recurring themes in the vault that exist as embedding\n * clusters but have NO dedicated note written about them yet. PAI surfaces\n * these by running the same agglomerative clustering used by graph_clusters /\n * zettelThemes and then filtering OUT any cluster whose label is well-matched\n * by an existing note title.\n *\n * The materialize endpoint writes a new Markdown note to the vault filesystem\n * and returns its content so the plugin can open it immediately.\n */\n\nimport { mkdirSync, writeFileSync } from \"node:fs\";\nimport { dirname, join } from \"node:path\";\nimport { TITLE_STOP_WORDS } from \"../utils/stop-words.js\";\nimport type { StorageBackend } from \"../storage/interface.js\";\nimport { zettelThemes } from \"../zettelkasten/themes.js\";\n\n// ---------------------------------------------------------------------------\n// Public param / result types\n// ---------------------------------------------------------------------------\n\nexport interface GraphLatentIdeasParams {\n project_id: number;\n /** Minimum notes in a cluster (default: 3) */\n min_cluster_size?: number;\n /** Cap on returned ideas (default: 15) */\n max_ideas?: number;\n /** How far back to look in days (default: 180) */\n lookback_days?: number;\n /** Cosine similarity clustering threshold (default: 0.65) */\n similarity_threshold?: number;\n}\n\nexport interface LatentIdeaSourceNote {\n vault_path: string;\n title: string;\n /** How strongly this note relates to the theme (0-1) */\n relevance: number;\n}\n\nexport interface LatentIdea {\n id: number;\n /** Auto-generated cluster label from zettelThemes */\n label: string;\n /** Number of notes touching this theme */\n size: number;\n /** 0-1, how likely this is a real coherent idea */\n confidence: number;\n /** Notes that contribute to this theme */\n source_notes: LatentIdeaSourceNote[];\n /** Cleaned-up version of label for a potential note title */\n suggested_title: string;\n /** Most common folder among source notes */\n suggested_folder: string;\n /** Number of distinct session date-folders (e.g. \"2026/03\") touching this theme */\n sessions_count: number;\n}\n\nexport interface GraphLatentIdeasResult {\n ideas: LatentIdea[];\n total_clusters_analyzed: number;\n /** How many clusters already have a matching note (excluded from results) */\n materialized_count: number;\n}\n\n// ---------------------------------------------------------------------------\n// Materialize params / result\n// ---------------------------------------------------------------------------\n\nexport interface IdeaMaterializeParams {\n idea_label: string;\n /** User-chosen title for the new note */\n title: string;\n /** Vault-relative folder path where the note should be created */\n folder: string;\n /** Vault-relative paths of the source notes to link from the new note */\n source_paths: string[];\n project_id: number;\n}\n\nexport interface IdeaMaterializeResult {\n /** Vault-relative path of the created note */\n vault_path: string;\n /** Generated markdown content */\n content: string;\n /** Number of wikilinks inserted */\n links_created: number;\n}\n\n// ---------------------------------------------------------------------------\n// Helper: check if a cluster already has a matching note\n// ---------------------------------------------------------------------------\n\n/**\n * Returns true when any existing vault note title closely matches the cluster\n * label — meaning a dedicated note already exists for this topic.\n *\n * Matching strategy (simple, fast, no embeddings needed):\n * 1. Lowercase both sides and split into words.\n * 2. Remove stop words from the label words.\n * 3. If ≥ 60% of the significant label words appear in a note title → match.\n */\n// TITLE_STOP_WORDS imported from utils/stop-words.ts\n\nfunction labelMatchesTitle(label: string, title: string): boolean {\n const labelWords = label\n .toLowerCase()\n .split(/[\\s\\-_/]+/)\n .filter((w) => w.length > 2 && !TITLE_STOP_WORDS.has(w));\n\n if (labelWords.length === 0) return false;\n\n const titleLower = title.toLowerCase();\n const matchCount = labelWords.filter((w) => titleLower.includes(w)).length;\n return matchCount / labelWords.length >= 0.6;\n}\n\n/**\n * Check whether any note indexed in the vault has a title matching the label.\n * Fetches all vault file rows via StorageBackend for efficiency.\n */\nasync function clusterHasMatchingNote(\n backend: StorageBackend,\n label: string,\n notePaths: string[]\n): Promise<boolean> {\n // First check the notes already in the cluster themselves — if any cluster\n // member's title matches the label it IS the index note → materialized.\n const pathSet = new Set(notePaths);\n\n // Fetch all vault files (bounded — vault rarely > 50k notes)\n const rows = await backend.getAllVaultFiles();\n\n for (const row of rows) {\n if (!row.title) continue;\n // Skip notes already counted inside the cluster — they don't count as\n // \"dedicated notes\"; we only skip a cluster if a SEPARATE note exists.\n if (pathSet.has(row.vaultPath)) continue;\n if (labelMatchesTitle(label, row.title)) return true;\n }\n return false;\n}\n\n// ---------------------------------------------------------------------------\n// Helper: generate a clean suggested title\n// ---------------------------------------------------------------------------\n\nfunction toSuggestedTitle(label: string): string {\n // Remove leading/trailing whitespace, capitalize each word, remove stop words\n // that are all-lowercase at the start of the title.\n const words = label\n .trim()\n .split(/\\s+/)\n .map((w, i) => {\n const lower = w.toLowerCase();\n // Drop leading stop words (but keep if they're the only word)\n if (i === 0 && TITLE_STOP_WORDS.has(lower) && label.trim().split(/\\s+/).length > 1) {\n return \"\";\n }\n return w.charAt(0).toUpperCase() + w.slice(1);\n })\n .filter(Boolean);\n\n return words.join(\" \") || label;\n}\n\n// ---------------------------------------------------------------------------\n// Helper: find most common folder\n// ---------------------------------------------------------------------------\n\nfunction mostCommonFolder(vaultPaths: string[]): string {\n const counts = new Map<string, number>();\n for (const p of vaultPaths) {\n const parts = p.split(\"/\");\n const folder = parts.length > 1 ? parts.slice(0, -1).join(\"/\") : \"\";\n counts.set(folder, (counts.get(folder) ?? 0) + 1);\n }\n\n let best = \"\";\n let bestCount = 0;\n for (const [folder, count] of counts) {\n if (count > bestCount) {\n bestCount = count;\n best = folder;\n }\n }\n return best;\n}\n\n// ---------------------------------------------------------------------------\n// Helper: count distinct session date-folders\n// ---------------------------------------------------------------------------\n\n/**\n * Heuristic: vault notes are often stored in date-based folders like\n * \"2026/03/15\" or \"Daily/2026-03\". We extract the first numeric path\n * segment that looks like a year (2020-2030) and group by year+month.\n *\n * Falls back to counting distinct top-level folders.\n */\nfunction countDistinctSessions(vaultPaths: string[]): number {\n const sessions = new Set<string>();\n const yearMonthRe = /\\b(202\\d)\\D?(0[1-9]|1[0-2])\\b/;\n\n for (const p of vaultPaths) {\n const m = yearMonthRe.exec(p);\n if (m) {\n sessions.add(`${m[1]}-${m[2]}`);\n } else {\n // Fallback: use top-level folder as a proxy for \"session bucket\"\n const topFolder = p.split(\"/\")[0];\n sessions.add(topFolder);\n }\n }\n return sessions.size;\n}\n\n// ---------------------------------------------------------------------------\n// Helper: calculate confidence score\n// ---------------------------------------------------------------------------\n\n/**\n * Confidence combines:\n * - Cluster size (normalized, capped at 20 for max contribution)\n * - Folder diversity (0-1 already)\n * - Sessions count (normalized, capped at 5)\n *\n * Formula: 0.4 * sizeScore + 0.35 * folderDiversity + 0.25 * sessionScore\n */\nfunction calcConfidence(\n size: number,\n folderDiversity: number,\n sessionsCount: number\n): number {\n const sizeScore = Math.min(size / 20, 1.0);\n const sessionScore = Math.min(sessionsCount / 5, 1.0);\n const raw = 0.4 * sizeScore + 0.35 * folderDiversity + 0.25 * sessionScore;\n return Math.round(raw * 100) / 100;\n}\n\n// ---------------------------------------------------------------------------\n// Main handler: graph_latent_ideas\n// ---------------------------------------------------------------------------\n\nexport async function handleGraphLatentIdeas(\n backend: StorageBackend,\n params: GraphLatentIdeasParams\n): Promise<GraphLatentIdeasResult> {\n const minClusterSize = params.min_cluster_size ?? 3;\n const maxIdeas = params.max_ideas ?? 15;\n const lookbackDays = params.lookback_days ?? 180;\n const similarityThreshold = params.similarity_threshold ?? 0.65;\n\n const { project_id: vaultProjectId } = params;\n if (!vaultProjectId) {\n throw new Error(\n \"graph_latent_ideas: project_id is required (pass the vault project's numeric ID)\"\n );\n }\n\n // Run the same clustering algorithm used by graph_clusters\n const themeResult = await zettelThemes(backend, {\n vaultProjectId,\n lookbackDays,\n minClusterSize,\n maxThemes: maxIdeas * 3, // Over-fetch — many will be filtered as materialized\n similarityThreshold,\n });\n\n const ideas: LatentIdea[] = [];\n let materializedCount = 0;\n\n for (const theme of themeResult.themes) {\n const notePaths = theme.notes.map((n) => n.path);\n\n // Check if a dedicated note already exists for this theme\n if (await clusterHasMatchingNote(backend, theme.label, notePaths)) {\n materializedCount++;\n continue;\n }\n\n // This is a latent idea — no dedicated note exists yet\n const suggestedFolder = mostCommonFolder(notePaths);\n const sessionsCount = countDistinctSessions(notePaths);\n const confidence = calcConfidence(theme.size, theme.folderDiversity, sessionsCount);\n\n // Build source notes with relevance scores\n // Relevance is approximated by position in cluster (centroid-closest first)\n // zettelThemes returns notes in no guaranteed order; assign uniform relevance\n // decreasing from 1.0 to 0.5 across the list.\n const sourceNotes: LatentIdeaSourceNote[] = theme.notes.map((n, idx) => ({\n vault_path: n.path,\n title: n.title ?? n.path.split(\"/\").pop()?.replace(/\\.md$/i, \"\") ?? n.path,\n relevance: Math.round((1.0 - (idx / Math.max(theme.notes.length - 1, 1)) * 0.5) * 100) / 100,\n }));\n\n ideas.push({\n id: theme.id,\n label: theme.label,\n size: theme.size,\n confidence,\n source_notes: sourceNotes,\n suggested_title: toSuggestedTitle(theme.label),\n suggested_folder: suggestedFolder,\n sessions_count: sessionsCount,\n });\n\n if (ideas.length >= maxIdeas) break;\n }\n\n // Sort by confidence descending\n ideas.sort((a, b) => b.confidence - a.confidence);\n\n return {\n ideas,\n total_clusters_analyzed: themeResult.themes.length + materializedCount,\n materialized_count: materializedCount,\n };\n}\n\n// ---------------------------------------------------------------------------\n// Materialize handler: idea_materialize\n// ---------------------------------------------------------------------------\n\nexport function handleIdeaMaterialize(\n params: IdeaMaterializeParams,\n vaultPath: string\n): IdeaMaterializeResult {\n const { idea_label, title, folder, source_paths } = params;\n\n // Sanitize filename: replace characters illegal in filenames\n const safeTitle = title.replace(/[/\\\\:*?\"<>|]/g, \"-\");\n const fileName = `${safeTitle}.md`;\n\n // Vault-relative path (forward slashes, no leading slash)\n const relFolder = folder.replace(/^\\/+|\\/+$/g, \"\");\n const vault_path = relFolder ? `${relFolder}/${fileName}` : fileName;\n\n // Absolute filesystem path\n const absPath = join(vaultPath, vault_path);\n const absDir = dirname(absPath);\n\n // Build wikilinks from source_paths\n const wikilinks = source_paths\n .map((p) => {\n // Derive a display name: filename without extension\n const name = p.split(\"/\").pop()?.replace(/\\.md$/i, \"\") ?? p;\n // Relative wikilink — use just the filename (Obsidian resolves by title)\n return `- [[${name}]]`;\n })\n .join(\"\\n\");\n\n const links_created = source_paths.length;\n\n const content = [\n `# ${title}`,\n \"\",\n `*Materialized from latent idea: \"${idea_label}\"*`,\n `*Sources: ${links_created} notes*`,\n \"\",\n \"## Related Notes\",\n \"\",\n wikilinks || \"*(no source notes)*\",\n \"\",\n \"## Notes\",\n \"\",\n \"<!-- Add your thoughts about this idea here -->\",\n \"\",\n ].join(\"\\n\");\n\n // Write the file (create parent directories as needed)\n mkdirSync(absDir, { recursive: true });\n writeFileSync(absPath, content, \"utf-8\");\n\n return {\n vault_path,\n content,\n links_created,\n };\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;AA0GA,SAAS,kBAAkB,OAAe,OAAwB;CAChE,MAAM,aAAa,MAChB,aAAa,CACb,MAAM,YAAY,CAClB,QAAQ,MAAM,EAAE,SAAS,KAAK,CAAC,iBAAiB,IAAI,EAAE,CAAC;AAE1D,KAAI,WAAW,WAAW,EAAG,QAAO;CAEpC,MAAM,aAAa,MAAM,aAAa;AAEtC,QADmB,WAAW,QAAQ,MAAM,WAAW,SAAS,EAAE,CAAC,CAAC,SAChD,WAAW,UAAU;;;;;;AAO3C,eAAe,uBACb,SACA,OACA,WACkB;CAGlB,MAAM,UAAU,IAAI,IAAI,UAAU;CAGlC,MAAM,OAAO,MAAM,QAAQ,kBAAkB;AAE7C,MAAK,MAAM,OAAO,MAAM;AACtB,MAAI,CAAC,IAAI,MAAO;AAGhB,MAAI,QAAQ,IAAI,IAAI,UAAU,CAAE;AAChC,MAAI,kBAAkB,OAAO,IAAI,MAAM,CAAE,QAAO;;AAElD,QAAO;;AAOT,SAAS,iBAAiB,OAAuB;AAgB/C,QAbc,MACX,MAAM,CACN,MAAM,MAAM,CACZ,KAAK,GAAG,MAAM;EACb,MAAM,QAAQ,EAAE,aAAa;AAE7B,MAAI,MAAM,KAAK,iBAAiB,IAAI,MAAM,IAAI,MAAM,MAAM,CAAC,MAAM,MAAM,CAAC,SAAS,EAC/E,QAAO;AAET,SAAO,EAAE,OAAO,EAAE,CAAC,aAAa,GAAG,EAAE,MAAM,EAAE;GAC7C,CACD,OAAO,QAAQ,CAEL,KAAK,IAAI,IAAI;;AAO5B,SAAS,iBAAiB,YAA8B;CACtD,MAAM,yBAAS,IAAI,KAAqB;AACxC,MAAK,MAAM,KAAK,YAAY;EAC1B,MAAM,QAAQ,EAAE,MAAM,IAAI;EAC1B,MAAM,SAAS,MAAM,SAAS,IAAI,MAAM,MAAM,GAAG,GAAG,CAAC,KAAK,IAAI,GAAG;AACjE,SAAO,IAAI,SAAS,OAAO,IAAI,OAAO,IAAI,KAAK,EAAE;;CAGnD,IAAI,OAAO;CACX,IAAI,YAAY;AAChB,MAAK,MAAM,CAAC,QAAQ,UAAU,OAC5B,KAAI,QAAQ,WAAW;AACrB,cAAY;AACZ,SAAO;;AAGX,QAAO;;;;;;;;;AAcT,SAAS,sBAAsB,YAA8B;CAC3D,MAAM,2BAAW,IAAI,KAAa;CAClC,MAAM,cAAc;AAEpB,MAAK,MAAM,KAAK,YAAY;EAC1B,MAAM,IAAI,YAAY,KAAK,EAAE;AAC7B,MAAI,EACF,UAAS,IAAI,GAAG,EAAE,GAAG,GAAG,EAAE,KAAK;OAC1B;GAEL,MAAM,YAAY,EAAE,MAAM,IAAI,CAAC;AAC/B,YAAS,IAAI,UAAU;;;AAG3B,QAAO,SAAS;;;;;;;;;;AAelB,SAAS,eACP,MACA,iBACA,eACQ;CACR,MAAM,YAAY,KAAK,IAAI,OAAO,IAAI,EAAI;CAC1C,MAAM,eAAe,KAAK,IAAI,gBAAgB,GAAG,EAAI;CACrD,MAAM,MAAM,KAAM,YAAY,MAAO,kBAAkB,MAAO;AAC9D,QAAO,KAAK,MAAM,MAAM,IAAI,GAAG;;AAOjC,eAAsB,uBACpB,SACA,QACiC;CACjC,MAAM,iBAAiB,OAAO,oBAAoB;CAClD,MAAM,WAAW,OAAO,aAAa;CACrC,MAAM,eAAe,OAAO,iBAAiB;CAC7C,MAAM,sBAAsB,OAAO,wBAAwB;CAE3D,MAAM,EAAE,YAAY,mBAAmB;AACvC,KAAI,CAAC,eACH,OAAM,IAAI,MACR,mFACD;CAIH,MAAM,cAAc,MAAM,aAAa,SAAS;EAC9C;EACA;EACA;EACA,WAAW,WAAW;EACtB;EACD,CAAC;CAEF,MAAM,QAAsB,EAAE;CAC9B,IAAI,oBAAoB;AAExB,MAAK,MAAM,SAAS,YAAY,QAAQ;EACtC,MAAM,YAAY,MAAM,MAAM,KAAK,MAAM,EAAE,KAAK;AAGhD,MAAI,MAAM,uBAAuB,SAAS,MAAM,OAAO,UAAU,EAAE;AACjE;AACA;;EAIF,MAAM,kBAAkB,iBAAiB,UAAU;EACnD,MAAM,gBAAgB,sBAAsB,UAAU;EACtD,MAAM,aAAa,eAAe,MAAM,MAAM,MAAM,iBAAiB,cAAc;EAMnF,MAAM,cAAsC,MAAM,MAAM,KAAK,GAAG,SAAS;GACvE,YAAY,EAAE;GACd,OAAO,EAAE,SAAS,EAAE,KAAK,MAAM,IAAI,CAAC,KAAK,EAAE,QAAQ,UAAU,GAAG,IAAI,EAAE;GACtE,WAAW,KAAK,OAAO,IAAO,MAAM,KAAK,IAAI,MAAM,MAAM,SAAS,GAAG,EAAE,GAAI,MAAO,IAAI,GAAG;GAC1F,EAAE;AAEH,QAAM,KAAK;GACT,IAAI,MAAM;GACV,OAAO,MAAM;GACb,MAAM,MAAM;GACZ;GACA,cAAc;GACd,iBAAiB,iBAAiB,MAAM,MAAM;GAC9C,kBAAkB;GAClB,gBAAgB;GACjB,CAAC;AAEF,MAAI,MAAM,UAAU,SAAU;;AAIhC,OAAM,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,WAAW;AAEjD,QAAO;EACL;EACA,yBAAyB,YAAY,OAAO,SAAS;EACrD,oBAAoB;EACrB;;AAOH,SAAgB,sBACd,QACA,WACuB;CACvB,MAAM,EAAE,YAAY,OAAO,QAAQ,iBAAiB;CAIpD,MAAM,WAAW,GADC,MAAM,QAAQ,iBAAiB,IAAI,CACvB;CAG9B,MAAM,YAAY,OAAO,QAAQ,cAAc,GAAG;CAClD,MAAM,aAAa,YAAY,GAAG,UAAU,GAAG,aAAa;CAG5D,MAAM,UAAU,KAAK,WAAW,WAAW;CAC3C,MAAM,SAAS,QAAQ,QAAQ;CAG/B,MAAM,YAAY,aACf,KAAK,MAAM;AAIV,SAAO,OAFM,EAAE,MAAM,IAAI,CAAC,KAAK,EAAE,QAAQ,UAAU,GAAG,IAAI,EAEvC;GACnB,CACD,KAAK,KAAK;CAEb,MAAM,gBAAgB,aAAa;CAEnC,MAAM,UAAU;EACd,KAAK;EACL;EACA,oCAAoC,WAAW;EAC/C,aAAa,cAAc;EAC3B;EACA;EACA;EACA,aAAa;EACb;EACA;EACA;EACA;EACA;EACD,CAAC,KAAK,KAAK;AAGZ,WAAU,QAAQ,EAAE,WAAW,MAAM,CAAC;AACtC,eAAc,SAAS,SAAS,QAAQ;AAExC,QAAO;EACL;EACA;EACA;EACD"}
@@ -1,34 +0,0 @@
1
- //#region src/memory/link-boost.ts
2
- /**
3
- * Re-rank results by inbound links *from other results in the same set*.
4
- *
5
- * Returns a new array, sorted by the adjusted score. Input is not mutated.
6
- * Results whose paths carry no inbound links are unchanged, so a corpus with
7
- * no links at all is a no-op rather than a distortion.
8
- */
9
- function applyLinkBoost(results, edges, opts) {
10
- const weight = opts?.weight ?? .25;
11
- if (results.length === 0 || edges.length === 0 || weight === 0) return [...results];
12
- const paths = new Set(results.map((r) => r.path));
13
- const inbound = /* @__PURE__ */ new Map();
14
- for (const e of edges) {
15
- if (e.sourcePath === e.targetPath) continue;
16
- if (!paths.has(e.sourcePath) || !paths.has(e.targetPath)) continue;
17
- inbound.set(e.targetPath, (inbound.get(e.targetPath) ?? 0) + 1);
18
- }
19
- if (inbound.size === 0) return [...results];
20
- const maxInbound = Math.max(...inbound.values());
21
- return results.map((r) => {
22
- const links = inbound.get(r.path) ?? 0;
23
- if (links === 0) return { ...r };
24
- const factor = 1 + weight * (links / maxInbound);
25
- return {
26
- ...r,
27
- score: r.score * factor
28
- };
29
- }).sort((a, b) => b.score - a.score);
30
- }
31
-
32
- //#endregion
33
- export { applyLinkBoost };
34
- //# sourceMappingURL=link-boost-CnI7UVMJ.mjs.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"link-boost-CnI7UVMJ.mjs","names":[],"sources":["../src/memory/link-boost.ts"],"sourcesContent":["/**\n * Rank search results by how the corpus links to them, not only by similarity.\n *\n * Why this exists: the store holds 33,709 wikilinks that are *facts* — one note\n * pointing at another, written by a person — alongside 2.4M chunks whose only\n * ranking signal is embedding similarity. Similarity answers \"what reads like\n * the query\". It cannot answer \"which of these is the one the others refer\n * back to\", which is usually the note worth reading first.\n *\n * The boost is deliberately query-local: it counts links *between the results\n * themselves*, not global popularity. A note linked by many other notes that\n * also match the query is a hub for that question. A note linked by half the\n * vault is merely popular, which is not the same thing and would flatten every\n * ranking toward the same few index pages.\n *\n * Links cost nothing to maintain — no embedding pass, no model call — so this\n * signal stays correct while the embedding backlog drains, and works for chunks\n * that have no embedding at all.\n */\n\nimport type { SearchResult } from \"./search.js\";\n\n/** A directed link between two note paths, as stored in vault_links. */\nexport interface LinkEdge {\n sourcePath: string;\n targetPath: string;\n}\n\nexport interface LinkBoostOptions {\n /**\n * How much the boost may move a result, as a fraction of its current score.\n * 0.25 means the most-linked result gains 25%. Kept modest by default: the\n * link graph is a supporting signal, and a note nobody links to can still be\n * the right answer.\n */\n weight?: number;\n}\n\n/**\n * Re-rank results by inbound links *from other results in the same set*.\n *\n * Returns a new array, sorted by the adjusted score. Input is not mutated.\n * Results whose paths carry no inbound links are unchanged, so a corpus with\n * no links at all is a no-op rather than a distortion.\n */\nexport function applyLinkBoost(\n results: SearchResult[],\n edges: LinkEdge[],\n opts?: LinkBoostOptions,\n): SearchResult[] {\n const weight = opts?.weight ?? 0.25;\n if (results.length === 0 || edges.length === 0 || weight === 0) {\n return [...results];\n }\n\n // Only links whose BOTH ends are in the result set count. An edge pointing\n // out of the set says nothing about the relative rank of results inside it.\n const paths = new Set(results.map((r) => r.path));\n const inbound = new Map<string, number>();\n for (const e of edges) {\n if (e.sourcePath === e.targetPath) continue; // self-links are noise\n if (!paths.has(e.sourcePath) || !paths.has(e.targetPath)) continue;\n inbound.set(e.targetPath, (inbound.get(e.targetPath) ?? 0) + 1);\n }\n if (inbound.size === 0) return [...results];\n\n // Normalise against the most-linked result so the boost is bounded by\n // `weight` regardless of corpus size. Without this, a densely linked project\n // would swamp similarity entirely while a sparse one would see no effect.\n const maxInbound = Math.max(...inbound.values());\n\n return results\n .map((r) => {\n const links = inbound.get(r.path) ?? 0;\n if (links === 0) return { ...r };\n const factor = 1 + weight * (links / maxInbound);\n return { ...r, score: r.score * factor };\n })\n .sort((a, b) => b.score - a.score);\n}\n"],"mappings":";;;;;;;;AA6CA,SAAgB,eACd,SACA,OACA,MACgB;CAChB,MAAM,SAAS,MAAM,UAAU;AAC/B,KAAI,QAAQ,WAAW,KAAK,MAAM,WAAW,KAAK,WAAW,EAC3D,QAAO,CAAC,GAAG,QAAQ;CAKrB,MAAM,QAAQ,IAAI,IAAI,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC;CACjD,MAAM,0BAAU,IAAI,KAAqB;AACzC,MAAK,MAAM,KAAK,OAAO;AACrB,MAAI,EAAE,eAAe,EAAE,WAAY;AACnC,MAAI,CAAC,MAAM,IAAI,EAAE,WAAW,IAAI,CAAC,MAAM,IAAI,EAAE,WAAW,CAAE;AAC1D,UAAQ,IAAI,EAAE,aAAa,QAAQ,IAAI,EAAE,WAAW,IAAI,KAAK,EAAE;;AAEjE,KAAI,QAAQ,SAAS,EAAG,QAAO,CAAC,GAAG,QAAQ;CAK3C,MAAM,aAAa,KAAK,IAAI,GAAG,QAAQ,QAAQ,CAAC;AAEhD,QAAO,QACJ,KAAK,MAAM;EACV,MAAM,QAAQ,QAAQ,IAAI,EAAE,KAAK,IAAI;AACrC,MAAI,UAAU,EAAG,QAAO,EAAE,GAAG,GAAG;EAChC,MAAM,SAAS,IAAI,UAAU,QAAQ;AACrC,SAAO;GAAE,GAAG;GAAG,OAAO,EAAE,QAAQ;GAAQ;GACxC,CACD,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,MAAM"}
@@ -1,7 +0,0 @@
1
- import "./utils-C9HsYDpO.mjs";
2
- import "./config-BbLFD7Uf.mjs";
3
- import "./run-env-DRcp7A8K.mjs";
4
- import { i as resumeTargetFor, n as resolveSessionDir, r as resumeCandidateFor, t as cmdMain } from "./main-resolver-DPPwHNtn.mjs";
5
- import "./aibroker-client-C5Fw7DNz.mjs";
6
-
7
- export { cmdMain };
@@ -1,3 +0,0 @@
1
- import { t as MergeError } from "./merge-DgU9OgZy.mjs";
2
-
3
- export { MergeError };
@@ -1,6 +0,0 @@
1
- //#region src/registry/merge.ts
2
- var MergeError = class extends Error {};
3
-
4
- //#endregion
5
- export { MergeError as t };
6
- //# sourceMappingURL=merge-DgU9OgZy.mjs.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"merge-DgU9OgZy.mjs","names":[],"sources":["../src/registry/merge.ts"],"sourcesContent":["/**\n * merge.ts — pure types and error class shared by the merge implementations.\n *\n * The actual SQLite implementation (planMerge/applyMerge, taking a raw\n * better-sqlite3 Database) lives in src/storage/sqlite/registry-merge.ts —\n * moved there so nothing outside src/storage/ imports better-sqlite3. The\n * Postgres implementation lives in src/storage/registry-postgres.ts. Both are\n * reached through RegistryBackend.planProjectMerge/applyProjectMerge\n * (src/storage/registry-interface.ts), never directly.\n *\n * See src/storage/sqlite/registry-merge.ts for the \"why one function\" design\n * rationale (five tables reference a project, foreign_keys is off).\n */\n\nexport interface MergePlan {\n fromId: number;\n fromSlug: string;\n intoId: number;\n intoSlug: string;\n /** Sessions to move, with the numbers they will be given. */\n sessions: { id: number; from: number; to: number }[];\n tags: number;\n aliases: number;\n compactions: number;\n links: number;\n /** The losing slug is kept as an alias, so `pai <old-name>` still resolves. */\n aliasToAdd?: string;\n}\n\nexport class MergeError extends Error {}\n"],"mappings":";AA6BA,IAAa,aAAb,cAAgC,MAAM"}