@tekmidian/pai 0.35.1 → 0.36.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/dist/.metadata_never_index +0 -0
  2. package/dist/{auto-route-DVM3U2ZY.mjs → auto-route-Byf8ENXj.mjs} +4 -4
  3. package/dist/{auto-route-DVM3U2ZY.mjs.map → auto-route-Byf8ENXj.mjs.map} +1 -1
  4. package/dist/{checkpoint-block-D3rm4dAJ.mjs → checkpoint-block-DKYxCkBL.mjs} +7 -6
  5. package/dist/checkpoint-block-DKYxCkBL.mjs.map +1 -0
  6. package/dist/cli/index.mjs +25 -17
  7. package/dist/cli/index.mjs.map +1 -1
  8. package/dist/cli/probe2.mjs +2 -0
  9. package/dist/cli/program.d.mts.map +1 -1
  10. package/dist/cli/program.mjs +16 -267
  11. package/dist/{clusters-CzGxefB7.mjs → clusters-wZgTCYCB.mjs} +2 -2
  12. package/dist/{clusters-CzGxefB7.mjs.map → clusters-wZgTCYCB.mjs.map} +1 -1
  13. package/dist/{config-BSkVcvfq.mjs → config-B64vFg14.mjs} +3 -12
  14. package/dist/{config-BSkVcvfq.mjs.map → config-B64vFg14.mjs.map} +1 -1
  15. package/dist/config-C_ErGddD.mjs +3 -0
  16. package/dist/daemon/index.mjs +19 -19
  17. package/dist/{daemon-Hnu6-HDD.mjs → daemon--N2JFnUs.mjs} +39 -39
  18. package/dist/daemon--N2JFnUs.mjs.map +1 -0
  19. package/dist/daemon-DJEFqV84.mjs +20 -0
  20. package/dist/daemon-mcp/index.mjs +2 -2
  21. package/dist/{db-BtuN768f.mjs → db-Ca5qfsMC.mjs} +2 -4
  22. package/dist/{db-BtuN768f.mjs.map → db-Ca5qfsMC.mjs.map} +1 -1
  23. package/dist/db-O-cyAPfS.mjs +3 -0
  24. package/dist/db-XEwJbuGO.mjs +3 -0
  25. package/dist/{db-CYmBWcjh.mjs → db-a1ixZQjr.mjs} +2 -4
  26. package/dist/{db-CYmBWcjh.mjs.map → db-a1ixZQjr.mjs.map} +1 -1
  27. package/dist/{detect-Bf2z-oKB.mjs → detect-CdaA48EI.mjs} +1 -1
  28. package/dist/{detect-Bf2z-oKB.mjs.map → detect-CdaA48EI.mjs.map} +1 -1
  29. package/dist/{detector-BU-bsDXs.mjs → detector--Gg5JRN5.mjs} +3 -5
  30. package/dist/{detector-BU-bsDXs.mjs.map → detector--Gg5JRN5.mjs.map} +1 -1
  31. package/dist/detector-DGAk1iBR.mjs +5 -0
  32. package/dist/embeddings-CEBGrzwu.mjs +3 -0
  33. package/dist/{embeddings-Bn86ssxR.mjs → embeddings-DOLZnT1X.mjs} +2 -12
  34. package/dist/{embeddings-Bn86ssxR.mjs.map → embeddings-DOLZnT1X.mjs.map} +1 -1
  35. package/dist/{factory-BGH0COXb.mjs → factory-Bsp7xOpO.mjs} +9 -12
  36. package/dist/{factory-BGH0COXb.mjs.map → factory-Bsp7xOpO.mjs.map} +1 -1
  37. package/dist/factory-CrokPMk2.mjs +3 -0
  38. package/dist/{helpers-crDEr6S2.mjs → helpers-IjZkXBhj.mjs} +1 -1
  39. package/dist/{helpers-crDEr6S2.mjs.map → helpers-IjZkXBhj.mjs.map} +1 -1
  40. package/dist/hooks/context-compression-hook.mjs +209 -70
  41. package/dist/hooks/context-compression-hook.mjs.map +4 -4
  42. package/dist/hooks/whisper-reinject.mjs +59 -0
  43. package/dist/hooks/whisper-reinject.mjs.map +7 -0
  44. package/dist/hooks/whisper-rules.mjs +24 -9
  45. package/dist/hooks/whisper-rules.mjs.map +2 -2
  46. package/dist/index.mjs +10 -10
  47. package/dist/{indexer-backend-Bg7VDpGt.mjs → indexer-backend-nQZuEx6N.mjs} +3 -3
  48. package/dist/{indexer-backend-Bg7VDpGt.mjs.map → indexer-backend-nQZuEx6N.mjs.map} +1 -1
  49. package/dist/{ipc-client-aVKVERjJ.mjs → ipc-client-BmypMNYk.mjs} +13 -7
  50. package/dist/ipc-client-BmypMNYk.mjs.map +1 -0
  51. package/dist/{kg-entity-r8duqhi9.mjs → kg-entity-DbOMPdF9.mjs} +1 -1
  52. package/dist/{kg-entity-r8duqhi9.mjs.map → kg-entity-DbOMPdF9.mjs.map} +1 -1
  53. package/dist/{latent-ideas-BL9m2HF9.mjs → latent-ideas-Bn6A5-5P.mjs} +4 -4
  54. package/dist/{latent-ideas-BL9m2HF9.mjs.map → latent-ideas-Bn6A5-5P.mjs.map} +1 -1
  55. package/dist/{link-boost-QFLrJwD6.mjs → link-boost-fYjUnxCN.mjs} +1 -1
  56. package/dist/{link-boost-QFLrJwD6.mjs.map → link-boost-fYjUnxCN.mjs.map} +1 -1
  57. package/dist/{main-resolver-CNSqU8wo.mjs → main-resolver-BAbhKpeX.mjs} +12 -14
  58. package/dist/main-resolver-BAbhKpeX.mjs.map +1 -0
  59. package/dist/main-resolver-Dxh444GO.mjs +4 -0
  60. package/dist/{migrate-fLD6rAdO.mjs → migrate-Cjzeefn9.mjs} +2 -2
  61. package/dist/{migrate-fLD6rAdO.mjs.map → migrate-Cjzeefn9.mjs.map} +1 -1
  62. package/dist/{neighborhood-BX89_nty.mjs → neighborhood-DpaEM991.mjs} +2 -2
  63. package/dist/{neighborhood-BX89_nty.mjs.map → neighborhood-DpaEM991.mjs.map} +1 -1
  64. package/dist/{note-context-d1wT_-GA.mjs → note-context-DrcY4cWm.mjs} +1 -1
  65. package/dist/{note-context-d1wT_-GA.mjs.map → note-context-DrcY4cWm.mjs.map} +1 -1
  66. package/dist/{pai-marker-B20KqhA8.mjs → pai-marker-CHtbJMwJ.mjs} +1 -1
  67. package/dist/{pai-marker-B20KqhA8.mjs.map → pai-marker-CHtbJMwJ.mjs.map} +1 -1
  68. package/dist/{postgres-BALUE11K.mjs → postgres-BVme6qX0.mjs} +7 -4
  69. package/dist/postgres-BVme6qX0.mjs.map +1 -0
  70. package/dist/{pick-aWhenqjE.mjs → program-BnMNFb4O.mjs} +1171 -223
  71. package/dist/program-BnMNFb4O.mjs.map +1 -0
  72. package/dist/query-feedback-BBMBp96K.mjs +3 -0
  73. package/dist/{query-feedback-D4U56Hz6.mjs → query-feedback-C1T6kS18.mjs} +2 -4
  74. package/dist/{query-feedback-D4U56Hz6.mjs.map → query-feedback-C1T6kS18.mjs.map} +1 -1
  75. package/dist/reranker-CwTCNsgA.mjs +3 -0
  76. package/dist/{reranker-CMNZcfVx.mjs → reranker-xPm04PXx.mjs} +2 -8
  77. package/dist/{reranker-CMNZcfVx.mjs.map → reranker-xPm04PXx.mjs.map} +1 -1
  78. package/dist/router-BMkOb62X.mjs +3 -0
  79. package/dist/{router-i9S19Usg.mjs → router-CsDm7HvK.mjs} +2 -4
  80. package/dist/{router-i9S19Usg.mjs.map → router-CsDm7HvK.mjs.map} +1 -1
  81. package/dist/{runtime-paths-B0P1TvUr.mjs → runtime-paths-rni52zHX.mjs} +1 -1
  82. package/dist/{runtime-paths-B0P1TvUr.mjs.map → runtime-paths-rni52zHX.mjs.map} +1 -1
  83. package/dist/search-CfPpJAWQ.mjs +4 -0
  84. package/dist/{search-C32zQ0V0.mjs → search-Rpk1cSBC.mjs} +4 -15
  85. package/dist/{search-C32zQ0V0.mjs.map → search-Rpk1cSBC.mjs.map} +1 -1
  86. package/dist/{sources-BDwN0B8i.mjs → sources-D8ZdNfvK.mjs} +2 -2
  87. package/dist/{sources-BDwN0B8i.mjs.map → sources-D8ZdNfvK.mjs.map} +1 -1
  88. package/dist/{sqlite-C6FHnMkn.mjs → sqlite-D1IaR8Am.mjs} +3 -3
  89. package/dist/{sqlite-C6FHnMkn.mjs.map → sqlite-D1IaR8Am.mjs.map} +1 -1
  90. package/dist/state-WaXhLr6R.mjs +70 -0
  91. package/dist/{state-DTvy-jRB.mjs.map → state-WaXhLr6R.mjs.map} +1 -1
  92. package/dist/state-qtmrBWCm.mjs +3 -0
  93. package/dist/{stop-words-BaMEGVeY.mjs → stop-words-Hfu8u22w.mjs} +1 -1
  94. package/dist/{stop-words-BaMEGVeY.mjs.map → stop-words-Hfu8u22w.mjs.map} +1 -1
  95. package/dist/{sync--BoxBBok.mjs → sync-BWbe8JTg.mjs} +3 -3
  96. package/dist/{sync--BoxBBok.mjs.map → sync-BWbe8JTg.mjs.map} +1 -1
  97. package/dist/{themes-BObEGMWn.mjs → themes-XPkj_bfP.mjs} +3 -3
  98. package/dist/{themes-BObEGMWn.mjs.map → themes-XPkj_bfP.mjs.map} +1 -1
  99. package/dist/tools-DEt6YPfc.mjs +5 -0
  100. package/dist/{tools-C1lCHerL.mjs → tools-ceiy7ANX.mjs} +28 -65
  101. package/dist/tools-ceiy7ANX.mjs.map +1 -0
  102. package/dist/{trace-h23JCcFD.mjs → trace-DfyGmMG_.mjs} +1 -1
  103. package/dist/{trace-h23JCcFD.mjs.map → trace-DfyGmMG_.mjs.map} +1 -1
  104. package/dist/{utils-BAxjW3j8.mjs → utils-9Err2RBW.mjs} +2 -22
  105. package/dist/{utils-BAxjW3j8.mjs.map → utils-9Err2RBW.mjs.map} +1 -1
  106. package/dist/utils-DhMex3Ox.mjs +3 -0
  107. package/dist/{vault-indexer-CUF9edbW.mjs → vault-indexer-CFvlPUMB.mjs} +2 -2
  108. package/dist/{vault-indexer-CUF9edbW.mjs.map → vault-indexer-CFvlPUMB.mjs.map} +1 -1
  109. package/dist/{work-queue-worker-BcDGAcF3.mjs → work-queue-worker-228XjABm.mjs} +234 -14
  110. package/dist/work-queue-worker-228XjABm.mjs.map +1 -0
  111. package/dist/work-queue-worker-LRA9Fj9z.mjs +11 -0
  112. package/dist/{zettelkasten-W-h8G2is.mjs → zettelkasten-CvjmMghT.mjs} +4 -4
  113. package/dist/{zettelkasten-W-h8G2is.mjs.map → zettelkasten-CvjmMghT.mjs.map} +1 -1
  114. package/package.json +1 -1
  115. package/src/hooks/ts/lib/context-fill.test.ts +515 -0
  116. package/src/hooks/ts/lib/context-fill.ts +585 -0
  117. package/src/hooks/ts/lib/context-handover-cache.ts +46 -0
  118. package/src/hooks/ts/lib/transcript-text.test.ts +125 -0
  119. package/src/hooks/ts/lib/transcript-text.ts +71 -0
  120. package/src/hooks/ts/post-tool-use/whisper-reinject.ts +88 -0
  121. package/src/hooks/ts/pre-compact/context-compression-hook.ts +112 -30
  122. package/src/hooks/ts/user-prompt/whisper-rules.ts +62 -8
  123. package/statusline-command.sh +16 -0
  124. package/dist/checkpoint-block-D3rm4dAJ.mjs.map +0 -1
  125. package/dist/daemon-Hnu6-HDD.mjs.map +0 -1
  126. package/dist/ipc-client-aVKVERjJ.mjs.map +0 -1
  127. package/dist/main-resolver-CNSqU8wo.mjs.map +0 -1
  128. package/dist/pick-aWhenqjE.mjs.map +0 -1
  129. package/dist/postgres-BALUE11K.mjs.map +0 -1
  130. package/dist/rolldown-runtime-95iHPtFO.mjs +0 -18
  131. package/dist/state-DTvy-jRB.mjs +0 -102
  132. package/dist/tools-C1lCHerL.mjs.map +0 -1
  133. package/dist/work-queue-worker-BcDGAcF3.mjs.map +0 -1
  134. /package/dist/{indexer-AEcT8wHf.mjs → indexer-D7MvSQPY.mjs} +0 -0
@@ -1 +1 @@
1
- {"version":3,"file":"db-CYmBWcjh.mjs","names":[],"sources":["../src/memory/schema.ts","../src/memory/db.ts"],"sourcesContent":["/**\n * SQLite DDL for the PAI federation database (federation.db).\n *\n * The federation DB is the cross-project search index — a single SQLite file\n * at ~/.pai/federation.db that holds chunked text from every registered\n * project's memory/ and Notes/ directories.\n *\n * Tables:\n * - memory_files — file-level metadata (hash, mtime, size) for change detection\n * - memory_chunks — chunked text with line numbers, tier classification, and optional embedding\n * - memory_fts — FTS5 virtual table backed by memory_chunks text\n *\n * Vault tables (vault_files, vault_aliases, vault_links, vault_name_index, vault_health)\n * have been migrated to Postgres (docker/init.sql) and are no longer created here.\n *\n * Schema version history:\n * v1 — initial schema (BM25 search only)\n * v2 — added embedding BLOB column to memory_chunks (Phase 2.5, vector search)\n * v3 — added vault tables (now removed — vault tables live in Postgres)\n */\n\nimport type { Database } from \"better-sqlite3\";\n\n/** Current schema version. Bump when adding new columns or tables. */\nexport const SCHEMA_VERSION = 5;\n\nexport const FEDERATION_SCHEMA_SQL = `\nPRAGMA journal_mode = WAL;\nPRAGMA foreign_keys = ON;\n\nCREATE TABLE IF NOT EXISTS memory_files (\n project_id INTEGER NOT NULL,\n path TEXT NOT NULL,\n source TEXT NOT NULL DEFAULT 'memory',\n tier TEXT NOT NULL DEFAULT 'topic',\n hash TEXT NOT NULL,\n mtime INTEGER NOT NULL,\n size INTEGER NOT NULL,\n PRIMARY KEY (project_id, path)\n);\n\nCREATE TABLE IF NOT EXISTS memory_chunks (\n id TEXT PRIMARY KEY,\n project_id INTEGER NOT NULL,\n source TEXT NOT NULL DEFAULT 'memory',\n tier TEXT NOT NULL DEFAULT 'topic',\n path TEXT NOT NULL,\n start_line INTEGER NOT NULL,\n end_line INTEGER NOT NULL,\n hash TEXT NOT NULL,\n text TEXT NOT NULL,\n updated_at INTEGER NOT NULL,\n last_accessed_at INTEGER,\n relevance_score REAL DEFAULT 0.5,\n embedding BLOB\n);\n\nCREATE VIRTUAL TABLE IF NOT EXISTS memory_fts USING fts5(\n text,\n id UNINDEXED,\n project_id UNINDEXED,\n path UNINDEXED,\n source UNINDEXED,\n tier UNINDEXED,\n start_line UNINDEXED,\n end_line UNINDEXED\n);\n\nCREATE INDEX IF NOT EXISTS idx_mc_project ON memory_chunks(project_id);\nCREATE INDEX IF NOT EXISTS idx_mc_source ON memory_chunks(project_id, source);\nCREATE INDEX IF NOT EXISTS idx_mc_tier ON memory_chunks(tier);\nCREATE INDEX IF NOT EXISTS idx_mf_project ON memory_files(project_id);\n\nCREATE TABLE IF NOT EXISTS kg_entities (\n entity_id TEXT PRIMARY KEY,\n tenant_id TEXT NOT NULL DEFAULT 'default',\n name TEXT NOT NULL,\n type TEXT NOT NULL DEFAULT 'unknown',\n description TEXT,\n first_seen INTEGER,\n last_seen INTEGER,\n mention_count INTEGER NOT NULL DEFAULT 1,\n feedback_weight REAL NOT NULL DEFAULT 0.5,\n UNIQUE(tenant_id, entity_id)\n);\n\nCREATE INDEX IF NOT EXISTS idx_kge_tenant ON kg_entities(tenant_id);\nCREATE INDEX IF NOT EXISTS idx_kge_name ON kg_entities(tenant_id, name);\nCREATE INDEX IF NOT EXISTS idx_kge_type ON kg_entities(tenant_id, type);\n`;\n\n/**\n * Apply the full federation schema to an open database.\n *\n * Idempotent — all statements use IF NOT EXISTS so calling this on an\n * already-initialised database is safe.\n *\n * Also runs any necessary migrations for existing databases (e.g. adding the\n * embedding column to an older schema that was created without it).\n */\nexport function initializeFederationSchema(db: Database): void {\n db.exec(FEDERATION_SCHEMA_SQL);\n runMigrations(db);\n}\n\n// ---------------------------------------------------------------------------\n// Migrations\n// ---------------------------------------------------------------------------\n\n/**\n * Apply incremental migrations to an existing database.\n *\n * Each migration is idempotent — safe to call on a database that has already\n * been migrated.\n */\nfunction runMigrations(db: Database): void {\n const columns = db.prepare(\"PRAGMA table_info(memory_chunks)\").all() as Array<{\n name: string;\n }>;\n\n // Migration v1→v2: add embedding BLOB column (schema v2, Phase 2.5)\n const hasEmbedding = columns.some((c) => c.name === \"embedding\");\n if (!hasEmbedding) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN embedding BLOB\");\n }\n\n // Create the partial index for embedded chunks (safe now that the column exists)\n db.exec(\n \"CREATE INDEX IF NOT EXISTS idx_mc_embedding ON memory_chunks(id) WHERE embedding IS NOT NULL\",\n );\n\n // Migration v4→v5: add last_accessed_at and relevance_score columns (QW2 + MR2)\n const hasLastAccessedAt = columns.some((c) => c.name === \"last_accessed_at\");\n if (!hasLastAccessedAt) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN last_accessed_at INTEGER\");\n }\n\n const hasRelevanceScore = columns.some((c) => c.name === \"relevance_score\");\n if (!hasRelevanceScore) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN relevance_score REAL DEFAULT 0.5\");\n }\n}\n","/**\n * Database connection helper for the PAI federation DB.\n *\n * Uses better-sqlite3 (synchronous API) to open or create federation.db.\n * On first open it runs the full DDL via initializeFederationSchema().\n */\n\nimport { mkdirSync } from \"node:fs\";\nimport { homedir } from \"node:os\";\nimport { dirname, join } from \"node:path\";\nimport BetterSqlite3 from \"better-sqlite3\";\nimport type { Database } from \"better-sqlite3\";\nimport { initializeFederationSchema } from \"./schema.js\";\n\nexport type { Database };\n\n/** Default federation DB path inside the ~/.pai/ directory. */\nconst DEFAULT_FEDERATION_PATH = join(homedir(), \".pai\", \"federation.db\");\n\n/**\n * Open (or create) the PAI federation database.\n *\n * @param path Absolute path to federation.db. Defaults to ~/.pai/federation.db.\n * @returns An open better-sqlite3 Database instance.\n *\n * Side effects on first call:\n * - Creates the parent directory if it does not exist.\n * - Enables WAL journal mode.\n * - Runs initializeFederationSchema() to ensure tables exist.\n */\nexport function openFederation(path: string = DEFAULT_FEDERATION_PATH): Database {\n // Ensure the directory exists before SQLite tries to create the file\n mkdirSync(dirname(path), { recursive: true });\n\n const db = new BetterSqlite3(path);\n\n // WAL gives better concurrent read performance and crash safety\n db.pragma(\"journal_mode = WAL\");\n db.pragma(\"foreign_keys = ON\");\n\n // Apply schema (idempotent — all statements use IF NOT EXISTS)\n initializeFederationSchema(db);\n\n return db;\n}\n"],"mappings":";;;;;;;AA0BA,MAAa,wBAAwB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA0ErC,SAAgB,2BAA2B,IAAoB;AAC7D,IAAG,KAAK,sBAAsB;AAC9B,eAAc,GAAG;;;;;;;;AAanB,SAAS,cAAc,IAAoB;CACzC,MAAM,UAAU,GAAG,QAAQ,mCAAmC,CAAC,KAAK;AAMpE,KAAI,CADiB,QAAQ,MAAM,MAAM,EAAE,SAAS,YAAY,CAE9D,IAAG,KAAK,sDAAsD;AAIhE,IAAG,KACD,+FACD;AAID,KAAI,CADsB,QAAQ,MAAM,MAAM,EAAE,SAAS,mBAAmB,CAE1E,IAAG,KAAK,gEAAgE;AAI1E,KAAI,CADsB,QAAQ,MAAM,MAAM,EAAE,SAAS,kBAAkB,CAEzE,IAAG,KAAK,wEAAwE;;;;;;;;;;;;;AC1HpF,MAAM,0BAA0B,KAAK,SAAS,EAAE,QAAQ,gBAAgB;;;;;;;;;;;;AAaxE,SAAgB,eAAe,OAAe,yBAAmC;AAE/E,WAAU,QAAQ,KAAK,EAAE,EAAE,WAAW,MAAM,CAAC;CAE7C,MAAM,KAAK,IAAI,cAAc,KAAK;AAGlC,IAAG,OAAO,qBAAqB;AAC/B,IAAG,OAAO,oBAAoB;AAG9B,4BAA2B,GAAG;AAE9B,QAAO"}
1
+ {"version":3,"file":"db-a1ixZQjr.mjs","names":[],"sources":["../src/memory/schema.ts","../src/memory/db.ts"],"sourcesContent":["/**\n * SQLite DDL for the PAI federation database (federation.db).\n *\n * The federation DB is the cross-project search index — a single SQLite file\n * at ~/.pai/federation.db that holds chunked text from every registered\n * project's memory/ and Notes/ directories.\n *\n * Tables:\n * - memory_files — file-level metadata (hash, mtime, size) for change detection\n * - memory_chunks — chunked text with line numbers, tier classification, and optional embedding\n * - memory_fts — FTS5 virtual table backed by memory_chunks text\n *\n * Vault tables (vault_files, vault_aliases, vault_links, vault_name_index, vault_health)\n * have been migrated to Postgres (docker/init.sql) and are no longer created here.\n *\n * Schema version history:\n * v1 — initial schema (BM25 search only)\n * v2 — added embedding BLOB column to memory_chunks (Phase 2.5, vector search)\n * v3 — added vault tables (now removed — vault tables live in Postgres)\n */\n\nimport type { Database } from \"better-sqlite3\";\n\n/** Current schema version. Bump when adding new columns or tables. */\nexport const SCHEMA_VERSION = 5;\n\nexport const FEDERATION_SCHEMA_SQL = `\nPRAGMA journal_mode = WAL;\nPRAGMA foreign_keys = ON;\n\nCREATE TABLE IF NOT EXISTS memory_files (\n project_id INTEGER NOT NULL,\n path TEXT NOT NULL,\n source TEXT NOT NULL DEFAULT 'memory',\n tier TEXT NOT NULL DEFAULT 'topic',\n hash TEXT NOT NULL,\n mtime INTEGER NOT NULL,\n size INTEGER NOT NULL,\n PRIMARY KEY (project_id, path)\n);\n\nCREATE TABLE IF NOT EXISTS memory_chunks (\n id TEXT PRIMARY KEY,\n project_id INTEGER NOT NULL,\n source TEXT NOT NULL DEFAULT 'memory',\n tier TEXT NOT NULL DEFAULT 'topic',\n path TEXT NOT NULL,\n start_line INTEGER NOT NULL,\n end_line INTEGER NOT NULL,\n hash TEXT NOT NULL,\n text TEXT NOT NULL,\n updated_at INTEGER NOT NULL,\n last_accessed_at INTEGER,\n relevance_score REAL DEFAULT 0.5,\n embedding BLOB\n);\n\nCREATE VIRTUAL TABLE IF NOT EXISTS memory_fts USING fts5(\n text,\n id UNINDEXED,\n project_id UNINDEXED,\n path UNINDEXED,\n source UNINDEXED,\n tier UNINDEXED,\n start_line UNINDEXED,\n end_line UNINDEXED\n);\n\nCREATE INDEX IF NOT EXISTS idx_mc_project ON memory_chunks(project_id);\nCREATE INDEX IF NOT EXISTS idx_mc_source ON memory_chunks(project_id, source);\nCREATE INDEX IF NOT EXISTS idx_mc_tier ON memory_chunks(tier);\nCREATE INDEX IF NOT EXISTS idx_mf_project ON memory_files(project_id);\n\nCREATE TABLE IF NOT EXISTS kg_entities (\n entity_id TEXT PRIMARY KEY,\n tenant_id TEXT NOT NULL DEFAULT 'default',\n name TEXT NOT NULL,\n type TEXT NOT NULL DEFAULT 'unknown',\n description TEXT,\n first_seen INTEGER,\n last_seen INTEGER,\n mention_count INTEGER NOT NULL DEFAULT 1,\n feedback_weight REAL NOT NULL DEFAULT 0.5,\n UNIQUE(tenant_id, entity_id)\n);\n\nCREATE INDEX IF NOT EXISTS idx_kge_tenant ON kg_entities(tenant_id);\nCREATE INDEX IF NOT EXISTS idx_kge_name ON kg_entities(tenant_id, name);\nCREATE INDEX IF NOT EXISTS idx_kge_type ON kg_entities(tenant_id, type);\n`;\n\n/**\n * Apply the full federation schema to an open database.\n *\n * Idempotent — all statements use IF NOT EXISTS so calling this on an\n * already-initialised database is safe.\n *\n * Also runs any necessary migrations for existing databases (e.g. adding the\n * embedding column to an older schema that was created without it).\n */\nexport function initializeFederationSchema(db: Database): void {\n db.exec(FEDERATION_SCHEMA_SQL);\n runMigrations(db);\n}\n\n// ---------------------------------------------------------------------------\n// Migrations\n// ---------------------------------------------------------------------------\n\n/**\n * Apply incremental migrations to an existing database.\n *\n * Each migration is idempotent — safe to call on a database that has already\n * been migrated.\n */\nfunction runMigrations(db: Database): void {\n const columns = db.prepare(\"PRAGMA table_info(memory_chunks)\").all() as Array<{\n name: string;\n }>;\n\n // Migration v1→v2: add embedding BLOB column (schema v2, Phase 2.5)\n const hasEmbedding = columns.some((c) => c.name === \"embedding\");\n if (!hasEmbedding) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN embedding BLOB\");\n }\n\n // Create the partial index for embedded chunks (safe now that the column exists)\n db.exec(\n \"CREATE INDEX IF NOT EXISTS idx_mc_embedding ON memory_chunks(id) WHERE embedding IS NOT NULL\",\n );\n\n // Migration v4→v5: add last_accessed_at and relevance_score columns (QW2 + MR2)\n const hasLastAccessedAt = columns.some((c) => c.name === \"last_accessed_at\");\n if (!hasLastAccessedAt) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN last_accessed_at INTEGER\");\n }\n\n const hasRelevanceScore = columns.some((c) => c.name === \"relevance_score\");\n if (!hasRelevanceScore) {\n db.exec(\"ALTER TABLE memory_chunks ADD COLUMN relevance_score REAL DEFAULT 0.5\");\n }\n}\n","/**\n * Database connection helper for the PAI federation DB.\n *\n * Uses better-sqlite3 (synchronous API) to open or create federation.db.\n * On first open it runs the full DDL via initializeFederationSchema().\n */\n\nimport { mkdirSync } from \"node:fs\";\nimport { homedir } from \"node:os\";\nimport { dirname, join } from \"node:path\";\nimport BetterSqlite3 from \"better-sqlite3\";\nimport type { Database } from \"better-sqlite3\";\nimport { initializeFederationSchema } from \"./schema.js\";\n\nexport type { Database };\n\n/** Default federation DB path inside the ~/.pai/ directory. */\nconst DEFAULT_FEDERATION_PATH = join(homedir(), \".pai\", \"federation.db\");\n\n/**\n * Open (or create) the PAI federation database.\n *\n * @param path Absolute path to federation.db. Defaults to ~/.pai/federation.db.\n * @returns An open better-sqlite3 Database instance.\n *\n * Side effects on first call:\n * - Creates the parent directory if it does not exist.\n * - Enables WAL journal mode.\n * - Runs initializeFederationSchema() to ensure tables exist.\n */\nexport function openFederation(path: string = DEFAULT_FEDERATION_PATH): Database {\n // Ensure the directory exists before SQLite tries to create the file\n mkdirSync(dirname(path), { recursive: true });\n\n const db = new BetterSqlite3(path);\n\n // WAL gives better concurrent read performance and crash safety\n db.pragma(\"journal_mode = WAL\");\n db.pragma(\"foreign_keys = ON\");\n\n // Apply schema (idempotent — all statements use IF NOT EXISTS)\n initializeFederationSchema(db);\n\n return db;\n}\n"],"mappings":";;;;;;AA0BA,MAAa,wBAAwB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA0ErC,SAAgB,2BAA2B,IAAoB;AAC7D,IAAG,KAAK,sBAAsB;AAC9B,eAAc,GAAG;;;;;;;;AAanB,SAAS,cAAc,IAAoB;CACzC,MAAM,UAAU,GAAG,QAAQ,mCAAmC,CAAC,KAAK;AAMpE,KAAI,CADiB,QAAQ,MAAM,MAAM,EAAE,SAAS,YAAY,CAE9D,IAAG,KAAK,sDAAsD;AAIhE,IAAG,KACD,+FACD;AAID,KAAI,CADsB,QAAQ,MAAM,MAAM,EAAE,SAAS,mBAAmB,CAE1E,IAAG,KAAK,gEAAgE;AAI1E,KAAI,CADsB,QAAQ,MAAM,MAAM,EAAE,SAAS,kBAAkB,CAEzE,IAAG,KAAK,wEAAwE;;;;;;;;;;;;AC1HpF,MAAM,0BAA0B,KAAK,SAAS,EAAE,QAAQ,gBAAgB;;;;;;;;;;;;AAaxE,SAAgB,eAAe,OAAe,yBAAmC;AAE/E,WAAU,QAAQ,KAAK,EAAE,EAAE,WAAW,MAAM,CAAC;CAE7C,MAAM,KAAK,IAAI,cAAc,KAAK;AAGlC,IAAG,OAAO,qBAAqB;AAC/B,IAAG,OAAO,oBAAoB;AAG9B,4BAA2B,GAAG;AAE9B,QAAO"}
@@ -83,4 +83,4 @@ function formatDetectionJson(d) {
83
83
 
84
84
  //#endregion
85
85
  export { formatDetection as n, formatDetectionJson as r, detectProject as t };
86
- //# sourceMappingURL=detect-Bf2z-oKB.mjs.map
86
+ //# sourceMappingURL=detect-CdaA48EI.mjs.map
@@ -1 +1 @@
1
- {"version":3,"file":"detect-Bf2z-oKB.mjs","names":[],"sources":["../src/cli/commands/detect.ts"],"sourcesContent":["/**\n * Project detection logic for PAI.\n *\n * detectProject(cwd) — given a filesystem path, returns the best matching\n * project from the registry:\n * 1. Exact path match\n * 2. Longest parent match (project whose root_path is an ancestor of cwd)\n *\n * Exported for use by the CLI `pai project detect` command and the MCP\n * `project_detect` tool.\n */\n\nimport type { Database } from \"better-sqlite3\";\nimport { resolve } from \"node:path\";\n\n// ---------------------------------------------------------------------------\n// Types\n// ---------------------------------------------------------------------------\n\nexport interface DetectedProject {\n id: number;\n slug: string;\n display_name: string;\n root_path: string;\n encoded_dir: string;\n type: string;\n status: string;\n session_count: number;\n last_session_date: string | null;\n match_type: \"exact\" | \"parent\";\n /** Only set when match_type is 'parent' — the portion of cwd below root_path */\n relative_path: string | null;\n}\n\ninterface ProjectRow {\n id: number;\n slug: string;\n display_name: string;\n root_path: string;\n encoded_dir: string;\n type: string;\n status: string;\n}\n\n// ---------------------------------------------------------------------------\n// Core detection function\n// ---------------------------------------------------------------------------\n\n/**\n * Detect which registered project a filesystem path belongs to.\n *\n * @param db Open registry database\n * @param cwd Absolute path to detect (defaults to process.cwd())\n * @returns The best matching project, or null if no match\n */\nexport function detectProject(\n db: Database,\n cwd?: string\n): DetectedProject | null {\n const target = resolve(cwd ?? process.cwd());\n\n // Load all active projects ordered by root_path length descending\n // so the longest (most specific) match wins in a linear scan.\n const projects = db\n .prepare(\n `SELECT id, slug, display_name, root_path, encoded_dir, type, status\n FROM projects\n WHERE status != 'archived'\n ORDER BY LENGTH(root_path) DESC`\n )\n .all() as ProjectRow[];\n\n let matched: ProjectRow | null = null;\n let matchType: \"exact\" | \"parent\" = \"exact\";\n\n for (const p of projects) {\n const root = resolve(p.root_path);\n if (target === root) {\n matched = p;\n matchType = \"exact\";\n break;\n }\n if (!matched && target.startsWith(root + \"/\")) {\n matched = p;\n matchType = \"parent\";\n // Keep scanning — a longer root_path match might exist (but shouldn't\n // since we sorted by length desc). Safety break anyway once found.\n break;\n }\n }\n\n if (!matched) return null;\n\n // Enrich with session stats\n const sessionStats = db\n .prepare(\n `SELECT COUNT(*) AS cnt, MAX(date) AS last_date\n FROM sessions WHERE project_id = ?`\n )\n .get(matched.id) as { cnt: number; last_date: string | null };\n\n const relative =\n matchType === \"parent\"\n ? target.slice(resolve(matched.root_path).length + 1)\n : null;\n\n return {\n id: matched.id,\n slug: matched.slug,\n display_name: matched.display_name,\n root_path: matched.root_path,\n encoded_dir: matched.encoded_dir,\n type: matched.type,\n status: matched.status,\n session_count: sessionStats.cnt,\n last_session_date: sessionStats.last_date,\n match_type: matchType,\n relative_path: relative,\n };\n}\n\n// ---------------------------------------------------------------------------\n// Format helpers\n// ---------------------------------------------------------------------------\n\n/**\n * Format a DetectedProject for human-readable CLI output.\n */\nexport function formatDetection(d: DetectedProject): string {\n const lines: string[] = [\n `slug: ${d.slug}`,\n `display_name: ${d.display_name}`,\n `root_path: ${d.root_path}`,\n `type: ${d.type}`,\n `status: ${d.status}`,\n `match: ${d.match_type}${d.relative_path ? ` (+${d.relative_path})` : \"\"}`,\n `sessions: ${d.session_count}`,\n ];\n if (d.last_session_date) {\n lines.push(`last_session: ${d.last_session_date}`);\n }\n return lines.join(\"\\n\");\n}\n\n/**\n * Format a DetectedProject as JSON for machine consumption.\n */\nexport function formatDetectionJson(d: DetectedProject): string {\n return JSON.stringify(\n {\n slug: d.slug,\n display_name: d.display_name,\n root_path: d.root_path,\n encoded_dir: d.encoded_dir,\n type: d.type,\n status: d.status,\n match_type: d.match_type,\n relative_path: d.relative_path,\n session_count: d.session_count,\n last_session_date: d.last_session_date,\n },\n null,\n 2\n );\n}\n"],"mappings":";;;;;;;;;;AAuDA,SAAgB,cACd,IACA,KACwB;CACxB,MAAM,SAAS,QAAQ,OAAO,QAAQ,KAAK,CAAC;CAI5C,MAAM,WAAW,GACd,QACC;;;wCAID,CACA,KAAK;CAER,IAAI,UAA6B;CACjC,IAAI,YAAgC;AAEpC,MAAK,MAAM,KAAK,UAAU;EACxB,MAAM,OAAO,QAAQ,EAAE,UAAU;AACjC,MAAI,WAAW,MAAM;AACnB,aAAU;AACV,eAAY;AACZ;;AAEF,MAAI,CAAC,WAAW,OAAO,WAAW,OAAO,IAAI,EAAE;AAC7C,aAAU;AACV,eAAY;AAGZ;;;AAIJ,KAAI,CAAC,QAAS,QAAO;CAGrB,MAAM,eAAe,GAClB,QACC;2CAED,CACA,IAAI,QAAQ,GAAG;CAElB,MAAM,WACJ,cAAc,WACV,OAAO,MAAM,QAAQ,QAAQ,UAAU,CAAC,SAAS,EAAE,GACnD;AAEN,QAAO;EACL,IAAI,QAAQ;EACZ,MAAM,QAAQ;EACd,cAAc,QAAQ;EACtB,WAAW,QAAQ;EACnB,aAAa,QAAQ;EACrB,MAAM,QAAQ;EACd,QAAQ,QAAQ;EAChB,eAAe,aAAa;EAC5B,mBAAmB,aAAa;EAChC,YAAY;EACZ,eAAe;EAChB;;;;;AAUH,SAAgB,gBAAgB,GAA4B;CAC1D,MAAM,QAAkB;EACtB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE,aAAa,EAAE,gBAAgB,MAAM,EAAE,cAAc,KAAK;EAC7E,iBAAiB,EAAE;EACpB;AACD,KAAI,EAAE,kBACJ,OAAM,KAAK,iBAAiB,EAAE,oBAAoB;AAEpD,QAAO,MAAM,KAAK,KAAK;;;;;AAMzB,SAAgB,oBAAoB,GAA4B;AAC9D,QAAO,KAAK,UACV;EACE,MAAM,EAAE;EACR,cAAc,EAAE;EAChB,WAAW,EAAE;EACb,aAAa,EAAE;EACf,MAAM,EAAE;EACR,QAAQ,EAAE;EACV,YAAY,EAAE;EACd,eAAe,EAAE;EACjB,eAAe,EAAE;EACjB,mBAAmB,EAAE;EACtB,EACD,MACA,EACD"}
1
+ {"version":3,"file":"detect-CdaA48EI.mjs","names":[],"sources":["../src/cli/commands/detect.ts"],"sourcesContent":["/**\n * Project detection logic for PAI.\n *\n * detectProject(cwd) — given a filesystem path, returns the best matching\n * project from the registry:\n * 1. Exact path match\n * 2. Longest parent match (project whose root_path is an ancestor of cwd)\n *\n * Exported for use by the CLI `pai project detect` command and the MCP\n * `project_detect` tool.\n */\n\nimport type { Database } from \"better-sqlite3\";\nimport { resolve } from \"node:path\";\n\n// ---------------------------------------------------------------------------\n// Types\n// ---------------------------------------------------------------------------\n\nexport interface DetectedProject {\n id: number;\n slug: string;\n display_name: string;\n root_path: string;\n encoded_dir: string;\n type: string;\n status: string;\n session_count: number;\n last_session_date: string | null;\n match_type: \"exact\" | \"parent\";\n /** Only set when match_type is 'parent' — the portion of cwd below root_path */\n relative_path: string | null;\n}\n\ninterface ProjectRow {\n id: number;\n slug: string;\n display_name: string;\n root_path: string;\n encoded_dir: string;\n type: string;\n status: string;\n}\n\n// ---------------------------------------------------------------------------\n// Core detection function\n// ---------------------------------------------------------------------------\n\n/**\n * Detect which registered project a filesystem path belongs to.\n *\n * @param db Open registry database\n * @param cwd Absolute path to detect (defaults to process.cwd())\n * @returns The best matching project, or null if no match\n */\nexport function detectProject(\n db: Database,\n cwd?: string\n): DetectedProject | null {\n const target = resolve(cwd ?? process.cwd());\n\n // Load all active projects ordered by root_path length descending\n // so the longest (most specific) match wins in a linear scan.\n const projects = db\n .prepare(\n `SELECT id, slug, display_name, root_path, encoded_dir, type, status\n FROM projects\n WHERE status != 'archived'\n ORDER BY LENGTH(root_path) DESC`\n )\n .all() as ProjectRow[];\n\n let matched: ProjectRow | null = null;\n let matchType: \"exact\" | \"parent\" = \"exact\";\n\n for (const p of projects) {\n const root = resolve(p.root_path);\n if (target === root) {\n matched = p;\n matchType = \"exact\";\n break;\n }\n if (!matched && target.startsWith(root + \"/\")) {\n matched = p;\n matchType = \"parent\";\n // Keep scanning — a longer root_path match might exist (but shouldn't\n // since we sorted by length desc). Safety break anyway once found.\n break;\n }\n }\n\n if (!matched) return null;\n\n // Enrich with session stats\n const sessionStats = db\n .prepare(\n `SELECT COUNT(*) AS cnt, MAX(date) AS last_date\n FROM sessions WHERE project_id = ?`\n )\n .get(matched.id) as { cnt: number; last_date: string | null };\n\n const relative =\n matchType === \"parent\"\n ? target.slice(resolve(matched.root_path).length + 1)\n : null;\n\n return {\n id: matched.id,\n slug: matched.slug,\n display_name: matched.display_name,\n root_path: matched.root_path,\n encoded_dir: matched.encoded_dir,\n type: matched.type,\n status: matched.status,\n session_count: sessionStats.cnt,\n last_session_date: sessionStats.last_date,\n match_type: matchType,\n relative_path: relative,\n };\n}\n\n// ---------------------------------------------------------------------------\n// Format helpers\n// ---------------------------------------------------------------------------\n\n/**\n * Format a DetectedProject for human-readable CLI output.\n */\nexport function formatDetection(d: DetectedProject): string {\n const lines: string[] = [\n `slug: ${d.slug}`,\n `display_name: ${d.display_name}`,\n `root_path: ${d.root_path}`,\n `type: ${d.type}`,\n `status: ${d.status}`,\n `match: ${d.match_type}${d.relative_path ? ` (+${d.relative_path})` : \"\"}`,\n `sessions: ${d.session_count}`,\n ];\n if (d.last_session_date) {\n lines.push(`last_session: ${d.last_session_date}`);\n }\n return lines.join(\"\\n\");\n}\n\n/**\n * Format a DetectedProject as JSON for machine consumption.\n */\nexport function formatDetectionJson(d: DetectedProject): string {\n return JSON.stringify(\n {\n slug: d.slug,\n display_name: d.display_name,\n root_path: d.root_path,\n encoded_dir: d.encoded_dir,\n type: d.type,\n status: d.status,\n match_type: d.match_type,\n relative_path: d.relative_path,\n session_count: d.session_count,\n last_session_date: d.last_session_date,\n },\n null,\n 2\n );\n}\n"],"mappings":";;;;;;;;;;AAuDA,SAAgB,cACd,IACA,KACwB;CACxB,MAAM,SAAS,QAAQ,OAAO,QAAQ,KAAK,CAAC;CAI5C,MAAM,WAAW,GACd,QACC;;;wCAID,CACA,KAAK;CAER,IAAI,UAA6B;CACjC,IAAI,YAAgC;AAEpC,MAAK,MAAM,KAAK,UAAU;EACxB,MAAM,OAAO,QAAQ,EAAE,UAAU;AACjC,MAAI,WAAW,MAAM;AACnB,aAAU;AACV,eAAY;AACZ;;AAEF,MAAI,CAAC,WAAW,OAAO,WAAW,OAAO,IAAI,EAAE;AAC7C,aAAU;AACV,eAAY;AAGZ;;;AAIJ,KAAI,CAAC,QAAS,QAAO;CAGrB,MAAM,eAAe,GAClB,QACC;2CAED,CACA,IAAI,QAAQ,GAAG;CAElB,MAAM,WACJ,cAAc,WACV,OAAO,MAAM,QAAQ,QAAQ,UAAU,CAAC,SAAS,EAAE,GACnD;AAEN,QAAO;EACL,IAAI,QAAQ;EACZ,MAAM,QAAQ;EACd,cAAc,QAAQ;EACtB,WAAW,QAAQ;EACnB,aAAa,QAAQ;EACrB,MAAM,QAAQ;EACd,QAAQ,QAAQ;EAChB,eAAe,aAAa;EAC5B,mBAAmB,aAAa;EAChC,YAAY;EACZ,eAAe;EAChB;;;;;AAUH,SAAgB,gBAAgB,GAA4B;CAC1D,MAAM,QAAkB;EACtB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE;EACnB,iBAAiB,EAAE,aAAa,EAAE,gBAAgB,MAAM,EAAE,cAAc,KAAK;EAC7E,iBAAiB,EAAE;EACpB;AACD,KAAI,EAAE,kBACJ,OAAM,KAAK,iBAAiB,EAAE,oBAAoB;AAEpD,QAAO,MAAM,KAAK,KAAK;;;;;AAMzB,SAAgB,oBAAoB,GAA4B;AAC9D,QAAO,KAAK,UACV;EACE,MAAM,EAAE;EACR,cAAc,EAAE;EAChB,WAAW,EAAE;EACb,aAAa,EAAE;EACf,MAAM,EAAE;EACR,QAAQ,EAAE;EACV,YAAY,EAAE;EACd,eAAe,EAAE;EACjB,eAAe,EAAE;EACjB,mBAAmB,EAAE;EACtB,EACD,MACA,EACD"}
@@ -1,8 +1,6 @@
1
- import { t as __exportAll } from "./rolldown-runtime-95iHPtFO.mjs";
2
- import { n as populateSlugs, r as searchMemory } from "./search-C32zQ0V0.mjs";
1
+ import { a as searchMemory, i as populateSlugs } from "./search-Rpk1cSBC.mjs";
3
2
 
4
3
  //#region src/topics/detector.ts
5
- var detector_exports = /* @__PURE__ */ __exportAll({ detectTopicShift: () => detectTopicShift });
6
4
  /**
7
5
  * Detect whether the provided context text best matches a different project
8
6
  * than the session's current routing.
@@ -70,5 +68,5 @@ async function detectTopicShift(registryDb, federation, params) {
70
68
  }
71
69
 
72
70
  //#endregion
73
- export { detector_exports as n, detectTopicShift as t };
74
- //# sourceMappingURL=detector-BU-bsDXs.mjs.map
71
+ export { detectTopicShift as t };
72
+ //# sourceMappingURL=detector--Gg5JRN5.mjs.map
@@ -1 +1 @@
1
- {"version":3,"file":"detector-BU-bsDXs.mjs","names":[],"sources":["../src/topics/detector.ts"],"sourcesContent":["/**\n * Topic shift detection engine.\n *\n * Accepts a context summary (recent conversation text) and determines whether\n * the conversation has drifted away from the currently-routed project.\n *\n * Algorithm:\n * 1. Run keyword memory_search against the context text (no project filter)\n * 2. Score results by project — sum of BM25 scores per project\n * 3. Compare the top-scoring project against the current project\n * 4. If a different project dominates by more than the confidence threshold,\n * report a topic shift.\n *\n * Design decisions:\n * - Keyword search only (no semantic) — fast, no embedding requirement\n * - Works with or without an active daemon (direct DB access path)\n * - Stateless: callers supply currentProject; detector has no session memory\n * - Minimal: returns a plain result object, not MCP content arrays\n */\n\nimport type { Database } from \"better-sqlite3\";\nimport type { StorageBackend } from \"../storage/interface.js\";\nimport { searchMemory, populateSlugs } from \"../memory/search.js\";\nimport type { SearchResult } from \"../memory/search.js\";\n\n// ---------------------------------------------------------------------------\n// Types\n// ---------------------------------------------------------------------------\n\nexport interface TopicCheckParams {\n /** Recent conversation context (a few sentences or tool call summaries) */\n context: string;\n /** The project slug the session is currently routed to. May be null/empty. */\n currentProject?: string;\n /**\n * Minimum confidence [0,1] to declare a shift. Default: 0.6.\n * Higher = less sensitive, fewer false positives.\n */\n threshold?: number;\n /**\n * Maximum results to draw from memory search (candidates). Default: 20.\n * More candidates = more accurate scoring, slightly slower.\n */\n candidates?: number;\n}\n\nexport interface TopicCheckResult {\n /** Whether a significant topic shift was detected. */\n shifted: boolean;\n /** The project slug the session is currently routed to (echoed from input). */\n currentProject: string | null;\n /** The project slug that best matches the context, or null if no clear match. */\n suggestedProject: string | null;\n /**\n * Confidence score for the suggested project [0,1].\n * Represents the fraction of total score mass held by the top project.\n * 1.0 = all matching chunks belong to one project.\n * 0.5 = two projects are equally matched.\n */\n confidence: number;\n /** Number of memory chunks that contributed to scoring. */\n chunkCount: number;\n /** Top-3 scoring projects with their normalised scores (for debugging). */\n topProjects: Array<{ slug: string; score: number }>;\n}\n\n// ---------------------------------------------------------------------------\n// Core algorithm\n// ---------------------------------------------------------------------------\n\n/**\n * Detect whether the provided context text best matches a different project\n * than the session's current routing.\n *\n * Works with either a raw SQLite Database or a StorageBackend.\n * For the StorageBackend path, keyword search is used.\n * For the raw Database path (legacy/direct), searchMemory() is called.\n */\nexport async function detectTopicShift(\n registryDb: Database,\n federation: Database | StorageBackend,\n params: TopicCheckParams\n): Promise<TopicCheckResult> {\n const threshold = params.threshold ?? 0.6;\n const candidates = params.candidates ?? 20;\n const currentProject = params.currentProject?.trim() || null;\n\n if (!params.context || params.context.trim().length === 0) {\n return {\n shifted: false,\n currentProject,\n suggestedProject: null,\n confidence: 0,\n chunkCount: 0,\n topProjects: [],\n };\n }\n\n // -------------------------------------------------------------------------\n // Run memory search across ALL projects (no project filter)\n // -------------------------------------------------------------------------\n\n let results: SearchResult[];\n\n const isBackend = (x: Database | StorageBackend): x is StorageBackend =>\n \"backendType\" in x;\n\n if (isBackend(federation)) {\n results = await federation.searchKeyword(params.context, {\n maxResults: candidates,\n });\n } else {\n results = searchMemory(federation, params.context, {\n maxResults: candidates,\n });\n }\n\n if (results.length === 0) {\n return {\n shifted: false,\n currentProject,\n suggestedProject: null,\n confidence: 0,\n chunkCount: 0,\n topProjects: [],\n };\n }\n\n // Populate project slugs from the registry\n const withSlugs = populateSlugs(results, registryDb);\n\n // -------------------------------------------------------------------------\n // Score projects by summing BM25 scores of matching chunks\n // -------------------------------------------------------------------------\n\n const projectScores = new Map<string, number>();\n\n for (const r of withSlugs) {\n const slug = r.projectSlug;\n if (!slug) continue;\n projectScores.set(slug, (projectScores.get(slug) ?? 0) + r.score);\n }\n\n if (projectScores.size === 0) {\n return {\n shifted: false,\n currentProject,\n suggestedProject: null,\n confidence: 0,\n chunkCount: withSlugs.length,\n topProjects: [],\n };\n }\n\n // Sort by total score descending\n const ranked = Array.from(projectScores.entries())\n .sort((a, b) => b[1] - a[1]);\n\n const totalScore = ranked.reduce((sum, [, s]) => sum + s, 0);\n\n // Top-3 for reporting (normalised to [0,1] fraction of total mass)\n const topProjects = ranked.slice(0, 3).map(([slug, score]) => ({\n slug,\n score: totalScore > 0 ? score / totalScore : 0,\n }));\n\n const topSlug = ranked[0][0];\n const topRawScore = ranked[0][1];\n const confidence = totalScore > 0 ? topRawScore / totalScore : 0;\n\n // -------------------------------------------------------------------------\n // Determine if a shift occurred\n // -------------------------------------------------------------------------\n\n // A shift is detected when:\n // 1. confidence >= threshold (the top project dominates)\n // 2. The top project is different from currentProject\n // 3. There is a currentProject to compare against\n // (if no current project, we still return the best match but no \"shift\")\n\n const isDifferent =\n currentProject !== null &&\n topSlug !== currentProject;\n\n const shifted = isDifferent && confidence >= threshold;\n\n return {\n shifted,\n currentProject,\n suggestedProject: topSlug,\n confidence,\n chunkCount: withSlugs.length,\n topProjects,\n };\n}\n"],"mappings":";;;;;;;;;;;;;AA8EA,eAAsB,iBACpB,YACA,YACA,QAC2B;CAC3B,MAAM,YAAY,OAAO,aAAa;CACtC,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,iBAAiB,OAAO,gBAAgB,MAAM,IAAI;AAExD,KAAI,CAAC,OAAO,WAAW,OAAO,QAAQ,MAAM,CAAC,WAAW,EACtD,QAAO;EACL,SAAS;EACT;EACA,kBAAkB;EAClB,YAAY;EACZ,YAAY;EACZ,aAAa,EAAE;EAChB;CAOH,IAAI;CAEJ,MAAM,aAAa,MACjB,iBAAiB;AAEnB,KAAI,UAAU,WAAW,CACvB,WAAU,MAAM,WAAW,cAAc,OAAO,SAAS,EACvD,YAAY,YACb,CAAC;KAEF,WAAU,aAAa,YAAY,OAAO,SAAS,EACjD,YAAY,YACb,CAAC;AAGJ,KAAI,QAAQ,WAAW,EACrB,QAAO;EACL,SAAS;EACT;EACA,kBAAkB;EAClB,YAAY;EACZ,YAAY;EACZ,aAAa,EAAE;EAChB;CAIH,MAAM,YAAY,cAAc,SAAS,WAAW;CAMpD,MAAM,gCAAgB,IAAI,KAAqB;AAE/C,MAAK,MAAM,KAAK,WAAW;EACzB,MAAM,OAAO,EAAE;AACf,MAAI,CAAC,KAAM;AACX,gBAAc,IAAI,OAAO,cAAc,IAAI,KAAK,IAAI,KAAK,EAAE,MAAM;;AAGnE,KAAI,cAAc,SAAS,EACzB,QAAO;EACL,SAAS;EACT;EACA,kBAAkB;EAClB,YAAY;EACZ,YAAY,UAAU;EACtB,aAAa,EAAE;EAChB;CAIH,MAAM,SAAS,MAAM,KAAK,cAAc,SAAS,CAAC,CAC/C,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,GAAG;CAE9B,MAAM,aAAa,OAAO,QAAQ,KAAK,GAAG,OAAO,MAAM,GAAG,EAAE;CAG5D,MAAM,cAAc,OAAO,MAAM,GAAG,EAAE,CAAC,KAAK,CAAC,MAAM,YAAY;EAC7D;EACA,OAAO,aAAa,IAAI,QAAQ,aAAa;EAC9C,EAAE;CAEH,MAAM,UAAU,OAAO,GAAG;CAC1B,MAAM,cAAc,OAAO,GAAG;CAC9B,MAAM,aAAa,aAAa,IAAI,cAAc,aAAa;AAkB/D,QAAO;EACL,SANA,mBAAmB,QACnB,YAAY,kBAEiB,cAAc;EAI3C;EACA,kBAAkB;EAClB;EACA,YAAY,UAAU;EACtB;EACD"}
1
+ {"version":3,"file":"detector--Gg5JRN5.mjs","names":[],"sources":["../src/topics/detector.ts"],"sourcesContent":["/**\n * Topic shift detection engine.\n *\n * Accepts a context summary (recent conversation text) and determines whether\n * the conversation has drifted away from the currently-routed project.\n *\n * Algorithm:\n * 1. Run keyword memory_search against the context text (no project filter)\n * 2. Score results by project — sum of BM25 scores per project\n * 3. Compare the top-scoring project against the current project\n * 4. If a different project dominates by more than the confidence threshold,\n * report a topic shift.\n *\n * Design decisions:\n * - Keyword search only (no semantic) — fast, no embedding requirement\n * - Works with or without an active daemon (direct DB access path)\n * - Stateless: callers supply currentProject; detector has no session memory\n * - Minimal: returns a plain result object, not MCP content arrays\n */\n\nimport type { Database } from \"better-sqlite3\";\nimport type { StorageBackend } from \"../storage/interface.js\";\nimport { searchMemory, populateSlugs } from \"../memory/search.js\";\nimport type { SearchResult } from \"../memory/search.js\";\n\n// ---------------------------------------------------------------------------\n// Types\n// ---------------------------------------------------------------------------\n\nexport interface TopicCheckParams {\n /** Recent conversation context (a few sentences or tool call summaries) */\n context: string;\n /** The project slug the session is currently routed to. May be null/empty. */\n currentProject?: string;\n /**\n * Minimum confidence [0,1] to declare a shift. Default: 0.6.\n * Higher = less sensitive, fewer false positives.\n */\n threshold?: number;\n /**\n * Maximum results to draw from memory search (candidates). Default: 20.\n * More candidates = more accurate scoring, slightly slower.\n */\n candidates?: number;\n}\n\nexport interface TopicCheckResult {\n /** Whether a significant topic shift was detected. */\n shifted: boolean;\n /** The project slug the session is currently routed to (echoed from input). */\n currentProject: string | null;\n /** The project slug that best matches the context, or null if no clear match. */\n suggestedProject: string | null;\n /**\n * Confidence score for the suggested project [0,1].\n * Represents the fraction of total score mass held by the top project.\n * 1.0 = all matching chunks belong to one project.\n * 0.5 = two projects are equally matched.\n */\n confidence: number;\n /** Number of memory chunks that contributed to scoring. */\n chunkCount: number;\n /** Top-3 scoring projects with their normalised scores (for debugging). */\n topProjects: Array<{ slug: string; score: number }>;\n}\n\n// ---------------------------------------------------------------------------\n// Core algorithm\n// ---------------------------------------------------------------------------\n\n/**\n * Detect whether the provided context text best matches a different project\n * than the session's current routing.\n *\n * Works with either a raw SQLite Database or a StorageBackend.\n * For the StorageBackend path, keyword search is used.\n * For the raw Database path (legacy/direct), searchMemory() is called.\n */\nexport async function detectTopicShift(\n registryDb: Database,\n federation: Database | StorageBackend,\n params: TopicCheckParams\n): Promise<TopicCheckResult> {\n const threshold = params.threshold ?? 0.6;\n const candidates = params.candidates ?? 20;\n const currentProject = params.currentProject?.trim() || null;\n\n if (!params.context || params.context.trim().length === 0) {\n return {\n shifted: false,\n currentProject,\n suggestedProject: null,\n confidence: 0,\n chunkCount: 0,\n topProjects: [],\n };\n }\n\n // -------------------------------------------------------------------------\n // Run memory search across ALL projects (no project filter)\n // -------------------------------------------------------------------------\n\n let results: SearchResult[];\n\n const isBackend = (x: Database | StorageBackend): x is StorageBackend =>\n \"backendType\" in x;\n\n if (isBackend(federation)) {\n results = await federation.searchKeyword(params.context, {\n maxResults: candidates,\n });\n } else {\n results = searchMemory(federation, params.context, {\n maxResults: candidates,\n });\n }\n\n if (results.length === 0) {\n return {\n shifted: false,\n currentProject,\n suggestedProject: null,\n confidence: 0,\n chunkCount: 0,\n topProjects: [],\n };\n }\n\n // Populate project slugs from the registry\n const withSlugs = populateSlugs(results, registryDb);\n\n // -------------------------------------------------------------------------\n // Score projects by summing BM25 scores of matching chunks\n // -------------------------------------------------------------------------\n\n const projectScores = new Map<string, number>();\n\n for (const r of withSlugs) {\n const slug = r.projectSlug;\n if (!slug) continue;\n projectScores.set(slug, (projectScores.get(slug) ?? 0) + r.score);\n }\n\n if (projectScores.size === 0) {\n return {\n shifted: false,\n currentProject,\n suggestedProject: null,\n confidence: 0,\n chunkCount: withSlugs.length,\n topProjects: [],\n };\n }\n\n // Sort by total score descending\n const ranked = Array.from(projectScores.entries())\n .sort((a, b) => b[1] - a[1]);\n\n const totalScore = ranked.reduce((sum, [, s]) => sum + s, 0);\n\n // Top-3 for reporting (normalised to [0,1] fraction of total mass)\n const topProjects = ranked.slice(0, 3).map(([slug, score]) => ({\n slug,\n score: totalScore > 0 ? score / totalScore : 0,\n }));\n\n const topSlug = ranked[0][0];\n const topRawScore = ranked[0][1];\n const confidence = totalScore > 0 ? topRawScore / totalScore : 0;\n\n // -------------------------------------------------------------------------\n // Determine if a shift occurred\n // -------------------------------------------------------------------------\n\n // A shift is detected when:\n // 1. confidence >= threshold (the top project dominates)\n // 2. The top project is different from currentProject\n // 3. There is a currentProject to compare against\n // (if no current project, we still return the best match but no \"shift\")\n\n const isDifferent =\n currentProject !== null &&\n topSlug !== currentProject;\n\n const shifted = isDifferent && confidence >= threshold;\n\n return {\n shifted,\n currentProject,\n suggestedProject: topSlug,\n confidence,\n chunkCount: withSlugs.length,\n topProjects,\n };\n}\n"],"mappings":";;;;;;;;;;;AA8EA,eAAsB,iBACpB,YACA,YACA,QAC2B;CAC3B,MAAM,YAAY,OAAO,aAAa;CACtC,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,iBAAiB,OAAO,gBAAgB,MAAM,IAAI;AAExD,KAAI,CAAC,OAAO,WAAW,OAAO,QAAQ,MAAM,CAAC,WAAW,EACtD,QAAO;EACL,SAAS;EACT;EACA,kBAAkB;EAClB,YAAY;EACZ,YAAY;EACZ,aAAa,EAAE;EAChB;CAOH,IAAI;CAEJ,MAAM,aAAa,MACjB,iBAAiB;AAEnB,KAAI,UAAU,WAAW,CACvB,WAAU,MAAM,WAAW,cAAc,OAAO,SAAS,EACvD,YAAY,YACb,CAAC;KAEF,WAAU,aAAa,YAAY,OAAO,SAAS,EACjD,YAAY,YACb,CAAC;AAGJ,KAAI,QAAQ,WAAW,EACrB,QAAO;EACL,SAAS;EACT;EACA,kBAAkB;EAClB,YAAY;EACZ,YAAY;EACZ,aAAa,EAAE;EAChB;CAIH,MAAM,YAAY,cAAc,SAAS,WAAW;CAMpD,MAAM,gCAAgB,IAAI,KAAqB;AAE/C,MAAK,MAAM,KAAK,WAAW;EACzB,MAAM,OAAO,EAAE;AACf,MAAI,CAAC,KAAM;AACX,gBAAc,IAAI,OAAO,cAAc,IAAI,KAAK,IAAI,KAAK,EAAE,MAAM;;AAGnE,KAAI,cAAc,SAAS,EACzB,QAAO;EACL,SAAS;EACT;EACA,kBAAkB;EAClB,YAAY;EACZ,YAAY,UAAU;EACtB,aAAa,EAAE;EAChB;CAIH,MAAM,SAAS,MAAM,KAAK,cAAc,SAAS,CAAC,CAC/C,MAAM,GAAG,MAAM,EAAE,KAAK,EAAE,GAAG;CAE9B,MAAM,aAAa,OAAO,QAAQ,KAAK,GAAG,OAAO,MAAM,GAAG,EAAE;CAG5D,MAAM,cAAc,OAAO,MAAM,GAAG,EAAE,CAAC,KAAK,CAAC,MAAM,YAAY;EAC7D;EACA,OAAO,aAAa,IAAI,QAAQ,aAAa;EAC9C,EAAE;CAEH,MAAM,UAAU,OAAO,GAAG;CAC1B,MAAM,cAAc,OAAO,GAAG;CAC9B,MAAM,aAAa,aAAa,IAAI,cAAc,aAAa;AAkB/D,QAAO;EACL,SANA,mBAAmB,QACnB,YAAY,kBAEiB,cAAc;EAI3C;EACA,kBAAkB;EAClB;EACA,YAAY,UAAU;EACtB;EACD"}
@@ -0,0 +1,5 @@
1
+ import "./embeddings-DOLZnT1X.mjs";
2
+ import "./search-Rpk1cSBC.mjs";
3
+ import { t as detectTopicShift } from "./detector--Gg5JRN5.mjs";
4
+
5
+ export { detectTopicShift };
@@ -0,0 +1,3 @@
1
+ import { a as generateEmbeddings, i as generateEmbedding, n as cosineSimilarity, o as serializeEmbedding, r as deserializeEmbedding, t as configureEmbeddingModel } from "./embeddings-DOLZnT1X.mjs";
2
+
3
+ export { generateEmbedding, generateEmbeddings, serializeEmbedding };
@@ -1,14 +1,4 @@
1
- import { t as __exportAll } from "./rolldown-runtime-95iHPtFO.mjs";
2
-
3
1
  //#region src/memory/embeddings.ts
4
- var embeddings_exports = /* @__PURE__ */ __exportAll({
5
- configureEmbeddingModel: () => configureEmbeddingModel,
6
- cosineSimilarity: () => cosineSimilarity,
7
- deserializeEmbedding: () => deserializeEmbedding,
8
- generateEmbedding: () => generateEmbedding,
9
- generateEmbeddings: () => generateEmbeddings,
10
- serializeEmbedding: () => serializeEmbedding
11
- });
12
2
  const DEFAULT_EMBEDDING_MODEL = "Snowflake/snowflake-arctic-embed-m-v1.5";
13
3
  /** Query prefix required by Snowflake Arctic Embed for retrieval tasks. */
14
4
  const QUERY_PREFIX = "Represent this sentence for searching relevant passages: ";
@@ -115,5 +105,5 @@ function cosineSimilarity(a, b) {
115
105
  }
116
106
 
117
107
  //#endregion
118
- export { generateEmbedding as a, embeddings_exports as i, cosineSimilarity as n, deserializeEmbedding as r, configureEmbeddingModel as t };
119
- //# sourceMappingURL=embeddings-Bn86ssxR.mjs.map
108
+ export { generateEmbeddings as a, generateEmbedding as i, cosineSimilarity as n, serializeEmbedding as o, deserializeEmbedding as r, configureEmbeddingModel as t };
109
+ //# sourceMappingURL=embeddings-DOLZnT1X.mjs.map
@@ -1 +1 @@
1
- {"version":3,"file":"embeddings-Bn86ssxR.mjs","names":[],"sources":["../src/memory/embeddings.ts"],"sourcesContent":["/**\n * Embedding generation for the PAI federation memory engine (Phase 2.5).\n *\n * Uses @huggingface/transformers with the Snowflake/snowflake-arctic-embed-m-v1.5 model\n * (768 dims, q8 quantization, MTEB strong retrieval quality).\n *\n * The model uses CLS pooling (first token) — NOT mean pooling.\n * For retrieval, queries require a prefix: \"Represent this sentence for searching relevant passages: \"\n * Documents should be embedded WITHOUT a prefix.\n *\n * The pipeline is a lazy singleton — loaded on first call, reused thereafter.\n * This avoids loading the heavy ML model on every CLI invocation.\n */\n\n// ---------------------------------------------------------------------------\n// Constants\n// ---------------------------------------------------------------------------\n\nexport const EMBEDDING_DIM = 768;\nconst DEFAULT_EMBEDDING_MODEL = \"Snowflake/snowflake-arctic-embed-m-v1.5\";\n\n/** Query prefix required by Snowflake Arctic Embed for retrieval tasks. */\nconst QUERY_PREFIX = \"Represent this sentence for searching relevant passages: \";\n\n// ---------------------------------------------------------------------------\n// Lazy pipeline singleton\n// ---------------------------------------------------------------------------\n\n// eslint-disable-next-line @typescript-eslint/no-explicit-any\nlet _embeddingPipeline: any = null;\nlet _currentModel: string | null = null;\n\n/**\n * Configure the embedding model to use.\n * Must be called before the first generateEmbedding() call.\n * If the pipeline is already loaded with a different model, it will be reloaded.\n *\n * @param model HuggingFace model ID (e.g. \"Snowflake/snowflake-arctic-embed-m-v1.5\").\n * Pass undefined or empty string to use the default model.\n */\nexport function configureEmbeddingModel(model?: string): void {\n const resolved = model?.trim() || DEFAULT_EMBEDDING_MODEL;\n if (_currentModel !== null && _currentModel !== resolved) {\n // Model changed — force reload on next call\n _embeddingPipeline = null;\n }\n _currentModel = resolved;\n}\n\nasync function getEmbedder() {\n const model = _currentModel ?? DEFAULT_EMBEDDING_MODEL;\n if (!_embeddingPipeline) {\n // Dynamic import to avoid loading the ML runtime on startup\n const { pipeline } = await import(\"@huggingface/transformers\");\n _embeddingPipeline = await pipeline(\n \"feature-extraction\",\n model,\n { dtype: \"q8\" },\n );\n }\n return _embeddingPipeline;\n}\n\n// ---------------------------------------------------------------------------\n// Embedding generation\n// ---------------------------------------------------------------------------\n\n/**\n * Generate a normalized 768-dim embedding for the given text.\n *\n * Uses CLS pooling (first token) and L2 normalization (cosine similarity ready).\n *\n * @param text The text to embed.\n * @param isQuery If true, prepend the Snowflake query prefix. Use for search queries.\n * Documents should be embedded without the prefix (default: false).\n */\nexport async function generateEmbedding(text: string, isQuery: boolean = false): Promise<Float32Array> {\n const prefix = isQuery ? QUERY_PREFIX : \"\";\n const input = prefix + text;\n const extractor = await getEmbedder();\n // Snowflake Arctic Embed uses CLS pooling (first token), not mean pooling\n const output = await extractor(input, { pooling: \"cls\", normalize: true });\n return new Float32Array(output.data);\n}\n\n/**\n * Generate normalized embeddings for several documents in one forward pass.\n *\n * The model call dominates embedding cost, and calling it once per chunk leaves\n * most of the available throughput unused: measured on this machine, one-at-a-\n * time runs at ~5 chunks/s, which turns a six-figure backlog into days of\n * work. Batching amortises the per-call overhead across the batch.\n *\n * The pipeline returns one flat buffer for the whole batch, so it is sliced\n * back apart by row. Order is preserved: result[i] corresponds to texts[i].\n *\n * Documents only — queries take the prefixed single-text path, since a search\n * embeds exactly one string and gains nothing here.\n */\nexport async function generateEmbeddings(texts: string[]): Promise<Float32Array[]> {\n if (texts.length === 0) return [];\n if (texts.length === 1) return [await generateEmbedding(texts[0])];\n\n const extractor = await getEmbedder();\n const output = await extractor(texts, { pooling: \"cls\", normalize: true });\n const flat = output.data as Float32Array;\n\n const dim = flat.length / texts.length;\n if (!Number.isInteger(dim)) {\n throw new Error(\n `Batched embedding returned ${flat.length} values for ${texts.length} inputs — not divisible`\n );\n }\n\n const out: Float32Array[] = [];\n for (let i = 0; i < texts.length; i++) {\n // Copy rather than subarray: the caller stores these, and a view would\n // pin the whole batch buffer in memory for the lifetime of one vector.\n out.push(new Float32Array(flat.slice(i * dim, (i + 1) * dim)));\n }\n return out;\n}\n\n// ---------------------------------------------------------------------------\n// Serialization helpers\n// ---------------------------------------------------------------------------\n\n/**\n * Serialize a Float32Array to a Buffer for storage in a SQLite BLOB column.\n */\nexport function serializeEmbedding(vec: Float32Array): Buffer {\n return Buffer.from(vec.buffer, vec.byteOffset, vec.byteLength);\n}\n\n/**\n * Deserialize a Buffer (from a SQLite BLOB column) back into a Float32Array.\n */\nexport function deserializeEmbedding(blob: Buffer): Float32Array {\n return new Float32Array(blob.buffer, blob.byteOffset, blob.byteLength / 4);\n}\n\n// ---------------------------------------------------------------------------\n// Similarity computation\n// ---------------------------------------------------------------------------\n\n/**\n * Compute cosine similarity between two normalized embedding vectors.\n *\n * Since both vectors are already L2-normalized by the embedding model,\n * cosine similarity reduces to a dot product — but we compute the full\n * formula for correctness when embeddings may not be pre-normalized.\n *\n * Returns a value in [-1, 1] where 1 = identical.\n */\nexport function cosineSimilarity(a: Float32Array, b: Float32Array): number {\n let dot = 0;\n let normA = 0;\n let normB = 0;\n for (let i = 0; i < a.length; i++) {\n dot += a[i] * b[i];\n normA += a[i] * a[i];\n normB += b[i] * b[i];\n }\n const denom = Math.sqrt(normA) * Math.sqrt(normB);\n if (denom === 0) return 0;\n return dot / denom;\n}\n"],"mappings":";;;;;;;;;;;AAmBA,MAAM,0BAA0B;;AAGhC,MAAM,eAAe;AAOrB,IAAI,qBAA0B;AAC9B,IAAI,gBAA+B;;;;;;;;;AAUnC,SAAgB,wBAAwB,OAAsB;CAC5D,MAAM,WAAW,OAAO,MAAM,IAAI;AAClC,KAAI,kBAAkB,QAAQ,kBAAkB,SAE9C,sBAAqB;AAEvB,iBAAgB;;AAGlB,eAAe,cAAc;CAC3B,MAAM,QAAQ,iBAAiB;AAC/B,KAAI,CAAC,oBAAoB;EAEvB,MAAM,EAAE,aAAa,MAAM,OAAO;AAClC,uBAAqB,MAAM,SACzB,sBACA,OACA,EAAE,OAAO,MAAM,CAChB;;AAEH,QAAO;;;;;;;;;;;AAgBT,eAAsB,kBAAkB,MAAc,UAAmB,OAA8B;CAErG,MAAM,SADS,UAAU,eAAe,MACjB;CAGvB,MAAM,SAAS,OAFG,MAAM,aAAa,EAEN,OAAO;EAAE,SAAS;EAAO,WAAW;EAAM,CAAC;AAC1E,QAAO,IAAI,aAAa,OAAO,KAAK;;;;;;;;;;;;;;;;AAiBtC,eAAsB,mBAAmB,OAA0C;AACjF,KAAI,MAAM,WAAW,EAAG,QAAO,EAAE;AACjC,KAAI,MAAM,WAAW,EAAG,QAAO,CAAC,MAAM,kBAAkB,MAAM,GAAG,CAAC;CAIlE,MAAM,QADS,OADG,MAAM,aAAa,EACN,OAAO;EAAE,SAAS;EAAO,WAAW;EAAM,CAAC,EACtD;CAEpB,MAAM,MAAM,KAAK,SAAS,MAAM;AAChC,KAAI,CAAC,OAAO,UAAU,IAAI,CACxB,OAAM,IAAI,MACR,8BAA8B,KAAK,OAAO,cAAc,MAAM,OAAO,yBACtE;CAGH,MAAM,MAAsB,EAAE;AAC9B,MAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,IAGhC,KAAI,KAAK,IAAI,aAAa,KAAK,MAAM,IAAI,MAAM,IAAI,KAAK,IAAI,CAAC,CAAC;AAEhE,QAAO;;;;;AAUT,SAAgB,mBAAmB,KAA2B;AAC5D,QAAO,OAAO,KAAK,IAAI,QAAQ,IAAI,YAAY,IAAI,WAAW;;;;;AAMhE,SAAgB,qBAAqB,MAA4B;AAC/D,QAAO,IAAI,aAAa,KAAK,QAAQ,KAAK,YAAY,KAAK,aAAa,EAAE;;;;;;;;;;;AAgB5E,SAAgB,iBAAiB,GAAiB,GAAyB;CACzE,IAAI,MAAM;CACV,IAAI,QAAQ;CACZ,IAAI,QAAQ;AACZ,MAAK,IAAI,IAAI,GAAG,IAAI,EAAE,QAAQ,KAAK;AACjC,SAAO,EAAE,KAAK,EAAE;AAChB,WAAS,EAAE,KAAK,EAAE;AAClB,WAAS,EAAE,KAAK,EAAE;;CAEpB,MAAM,QAAQ,KAAK,KAAK,MAAM,GAAG,KAAK,KAAK,MAAM;AACjD,KAAI,UAAU,EAAG,QAAO;AACxB,QAAO,MAAM"}
1
+ {"version":3,"file":"embeddings-DOLZnT1X.mjs","names":[],"sources":["../src/memory/embeddings.ts"],"sourcesContent":["/**\n * Embedding generation for the PAI federation memory engine (Phase 2.5).\n *\n * Uses @huggingface/transformers with the Snowflake/snowflake-arctic-embed-m-v1.5 model\n * (768 dims, q8 quantization, MTEB strong retrieval quality).\n *\n * The model uses CLS pooling (first token) — NOT mean pooling.\n * For retrieval, queries require a prefix: \"Represent this sentence for searching relevant passages: \"\n * Documents should be embedded WITHOUT a prefix.\n *\n * The pipeline is a lazy singleton — loaded on first call, reused thereafter.\n * This avoids loading the heavy ML model on every CLI invocation.\n */\n\n// ---------------------------------------------------------------------------\n// Constants\n// ---------------------------------------------------------------------------\n\nexport const EMBEDDING_DIM = 768;\nconst DEFAULT_EMBEDDING_MODEL = \"Snowflake/snowflake-arctic-embed-m-v1.5\";\n\n/** Query prefix required by Snowflake Arctic Embed for retrieval tasks. */\nconst QUERY_PREFIX = \"Represent this sentence for searching relevant passages: \";\n\n// ---------------------------------------------------------------------------\n// Lazy pipeline singleton\n// ---------------------------------------------------------------------------\n\n// eslint-disable-next-line @typescript-eslint/no-explicit-any\nlet _embeddingPipeline: any = null;\nlet _currentModel: string | null = null;\n\n/**\n * Configure the embedding model to use.\n * Must be called before the first generateEmbedding() call.\n * If the pipeline is already loaded with a different model, it will be reloaded.\n *\n * @param model HuggingFace model ID (e.g. \"Snowflake/snowflake-arctic-embed-m-v1.5\").\n * Pass undefined or empty string to use the default model.\n */\nexport function configureEmbeddingModel(model?: string): void {\n const resolved = model?.trim() || DEFAULT_EMBEDDING_MODEL;\n if (_currentModel !== null && _currentModel !== resolved) {\n // Model changed — force reload on next call\n _embeddingPipeline = null;\n }\n _currentModel = resolved;\n}\n\nasync function getEmbedder() {\n const model = _currentModel ?? DEFAULT_EMBEDDING_MODEL;\n if (!_embeddingPipeline) {\n // Dynamic import to avoid loading the ML runtime on startup\n const { pipeline } = await import(\"@huggingface/transformers\");\n _embeddingPipeline = await pipeline(\n \"feature-extraction\",\n model,\n { dtype: \"q8\" },\n );\n }\n return _embeddingPipeline;\n}\n\n// ---------------------------------------------------------------------------\n// Embedding generation\n// ---------------------------------------------------------------------------\n\n/**\n * Generate a normalized 768-dim embedding for the given text.\n *\n * Uses CLS pooling (first token) and L2 normalization (cosine similarity ready).\n *\n * @param text The text to embed.\n * @param isQuery If true, prepend the Snowflake query prefix. Use for search queries.\n * Documents should be embedded without the prefix (default: false).\n */\nexport async function generateEmbedding(text: string, isQuery: boolean = false): Promise<Float32Array> {\n const prefix = isQuery ? QUERY_PREFIX : \"\";\n const input = prefix + text;\n const extractor = await getEmbedder();\n // Snowflake Arctic Embed uses CLS pooling (first token), not mean pooling\n const output = await extractor(input, { pooling: \"cls\", normalize: true });\n return new Float32Array(output.data);\n}\n\n/**\n * Generate normalized embeddings for several documents in one forward pass.\n *\n * The model call dominates embedding cost, and calling it once per chunk leaves\n * most of the available throughput unused: measured on this machine, one-at-a-\n * time runs at ~5 chunks/s, which turns a six-figure backlog into days of\n * work. Batching amortises the per-call overhead across the batch.\n *\n * The pipeline returns one flat buffer for the whole batch, so it is sliced\n * back apart by row. Order is preserved: result[i] corresponds to texts[i].\n *\n * Documents only — queries take the prefixed single-text path, since a search\n * embeds exactly one string and gains nothing here.\n */\nexport async function generateEmbeddings(texts: string[]): Promise<Float32Array[]> {\n if (texts.length === 0) return [];\n if (texts.length === 1) return [await generateEmbedding(texts[0])];\n\n const extractor = await getEmbedder();\n const output = await extractor(texts, { pooling: \"cls\", normalize: true });\n const flat = output.data as Float32Array;\n\n const dim = flat.length / texts.length;\n if (!Number.isInteger(dim)) {\n throw new Error(\n `Batched embedding returned ${flat.length} values for ${texts.length} inputs — not divisible`\n );\n }\n\n const out: Float32Array[] = [];\n for (let i = 0; i < texts.length; i++) {\n // Copy rather than subarray: the caller stores these, and a view would\n // pin the whole batch buffer in memory for the lifetime of one vector.\n out.push(new Float32Array(flat.slice(i * dim, (i + 1) * dim)));\n }\n return out;\n}\n\n// ---------------------------------------------------------------------------\n// Serialization helpers\n// ---------------------------------------------------------------------------\n\n/**\n * Serialize a Float32Array to a Buffer for storage in a SQLite BLOB column.\n */\nexport function serializeEmbedding(vec: Float32Array): Buffer {\n return Buffer.from(vec.buffer, vec.byteOffset, vec.byteLength);\n}\n\n/**\n * Deserialize a Buffer (from a SQLite BLOB column) back into a Float32Array.\n */\nexport function deserializeEmbedding(blob: Buffer): Float32Array {\n return new Float32Array(blob.buffer, blob.byteOffset, blob.byteLength / 4);\n}\n\n// ---------------------------------------------------------------------------\n// Similarity computation\n// ---------------------------------------------------------------------------\n\n/**\n * Compute cosine similarity between two normalized embedding vectors.\n *\n * Since both vectors are already L2-normalized by the embedding model,\n * cosine similarity reduces to a dot product — but we compute the full\n * formula for correctness when embeddings may not be pre-normalized.\n *\n * Returns a value in [-1, 1] where 1 = identical.\n */\nexport function cosineSimilarity(a: Float32Array, b: Float32Array): number {\n let dot = 0;\n let normA = 0;\n let normB = 0;\n for (let i = 0; i < a.length; i++) {\n dot += a[i] * b[i];\n normA += a[i] * a[i];\n normB += b[i] * b[i];\n }\n const denom = Math.sqrt(normA) * Math.sqrt(normB);\n if (denom === 0) return 0;\n return dot / denom;\n}\n"],"mappings":";AAmBA,MAAM,0BAA0B;;AAGhC,MAAM,eAAe;AAOrB,IAAI,qBAA0B;AAC9B,IAAI,gBAA+B;;;;;;;;;AAUnC,SAAgB,wBAAwB,OAAsB;CAC5D,MAAM,WAAW,OAAO,MAAM,IAAI;AAClC,KAAI,kBAAkB,QAAQ,kBAAkB,SAE9C,sBAAqB;AAEvB,iBAAgB;;AAGlB,eAAe,cAAc;CAC3B,MAAM,QAAQ,iBAAiB;AAC/B,KAAI,CAAC,oBAAoB;EAEvB,MAAM,EAAE,aAAa,MAAM,OAAO;AAClC,uBAAqB,MAAM,SACzB,sBACA,OACA,EAAE,OAAO,MAAM,CAChB;;AAEH,QAAO;;;;;;;;;;;AAgBT,eAAsB,kBAAkB,MAAc,UAAmB,OAA8B;CAErG,MAAM,SADS,UAAU,eAAe,MACjB;CAGvB,MAAM,SAAS,OAFG,MAAM,aAAa,EAEN,OAAO;EAAE,SAAS;EAAO,WAAW;EAAM,CAAC;AAC1E,QAAO,IAAI,aAAa,OAAO,KAAK;;;;;;;;;;;;;;;;AAiBtC,eAAsB,mBAAmB,OAA0C;AACjF,KAAI,MAAM,WAAW,EAAG,QAAO,EAAE;AACjC,KAAI,MAAM,WAAW,EAAG,QAAO,CAAC,MAAM,kBAAkB,MAAM,GAAG,CAAC;CAIlE,MAAM,QADS,OADG,MAAM,aAAa,EACN,OAAO;EAAE,SAAS;EAAO,WAAW;EAAM,CAAC,EACtD;CAEpB,MAAM,MAAM,KAAK,SAAS,MAAM;AAChC,KAAI,CAAC,OAAO,UAAU,IAAI,CACxB,OAAM,IAAI,MACR,8BAA8B,KAAK,OAAO,cAAc,MAAM,OAAO,yBACtE;CAGH,MAAM,MAAsB,EAAE;AAC9B,MAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,IAGhC,KAAI,KAAK,IAAI,aAAa,KAAK,MAAM,IAAI,MAAM,IAAI,KAAK,IAAI,CAAC,CAAC;AAEhE,QAAO;;;;;AAUT,SAAgB,mBAAmB,KAA2B;AAC5D,QAAO,OAAO,KAAK,IAAI,QAAQ,IAAI,YAAY,IAAI,WAAW;;;;;AAMhE,SAAgB,qBAAqB,MAA4B;AAC/D,QAAO,IAAI,aAAa,KAAK,QAAQ,KAAK,YAAY,KAAK,aAAa,EAAE;;;;;;;;;;;AAgB5E,SAAgB,iBAAiB,GAAiB,GAAyB;CACzE,IAAI,MAAM;CACV,IAAI,QAAQ;CACZ,IAAI,QAAQ;AACZ,MAAK,IAAI,IAAI,GAAG,IAAI,EAAE,QAAQ,KAAK;AACjC,SAAO,EAAE,KAAK,EAAE;AAChB,WAAS,EAAE,KAAK,EAAE;AAClB,WAAS,EAAE,KAAK,EAAE;;CAEpB,MAAM,QAAQ,KAAK,KAAK,MAAM,GAAG,KAAK,KAAK,MAAM;AACjD,KAAI,UAAU,EAAG,QAAO;AACxB,QAAO,MAAM"}
@@ -1,5 +1,3 @@
1
- import { t as __exportAll } from "./rolldown-runtime-95iHPtFO.mjs";
2
-
3
1
  //#region src/storage/outage.ts
4
2
  let current = null;
5
3
  /** Record that the backend is currently unreachable. */
@@ -34,7 +32,6 @@ function humanDuration(ms) {
34
32
 
35
33
  //#endregion
36
34
  //#region src/storage/factory.ts
37
- var factory_exports = /* @__PURE__ */ __exportAll({ createStorageBackend: () => createStorageBackend });
38
35
  /** Backoff schedule (ms) for the bounded CLI retry path. */
39
36
  const CLI_RETRY_DELAYS_MS = [
40
37
  500,
@@ -59,7 +56,7 @@ async function createStorageBackend(config, opts = {}) {
59
56
  * backend on success, or an error string describing why it failed.
60
57
  */
61
58
  async function attemptPostgres(config) {
62
- const { PostgresBackend } = await import("./postgres-BALUE11K.mjs");
59
+ const { PostgresBackend } = await import("./postgres-BVme6qX0.mjs");
63
60
  const pgConfig = config.postgres ?? {};
64
61
  let backend = null;
65
62
  try {
@@ -88,8 +85,8 @@ const ESCALATE_AFTER_ATTEMPTS = 5;
88
85
  /** Tell the user the backend is down, through whatever channels are configured. */
89
86
  async function notifyBackendDown(attempts, lastError) {
90
87
  try {
91
- const { routeNotification } = await import("./router-i9S19Usg.mjs").then((n) => n.n);
92
- const { loadConfig } = await import("./config-BSkVcvfq.mjs").then((n) => n.r);
88
+ const { routeNotification } = await import("./router-BMkOb62X.mjs");
89
+ const { loadConfig } = await import("./config-C_ErGddD.mjs");
93
90
  await routeNotification({
94
91
  event: "error",
95
92
  title: "PAI: storage backend unreachable",
@@ -100,8 +97,8 @@ async function notifyBackendDown(attempts, lastError) {
100
97
  /** And say when it comes back, so the alert is not left hanging. */
101
98
  async function notifyBackendRecovered(attempts, since) {
102
99
  try {
103
- const { routeNotification } = await import("./router-i9S19Usg.mjs").then((n) => n.n);
104
- const { loadConfig } = await import("./config-BSkVcvfq.mjs").then((n) => n.r);
100
+ const { routeNotification } = await import("./router-BMkOb62X.mjs");
101
+ const { loadConfig } = await import("./config-C_ErGddD.mjs");
105
102
  const mins = since ? Math.max(1, Math.round((Date.now() - since) / 6e4)) : null;
106
103
  await routeNotification({
107
104
  event: "completion",
@@ -143,11 +140,11 @@ async function connectPostgres(config, waitForever) {
143
140
  }
144
141
  }
145
142
  async function createSQLiteBackend() {
146
- const { openFederation } = await import("./db-CYmBWcjh.mjs").then((n) => n.t);
147
- const { SQLiteBackend } = await import("./sqlite-C6FHnMkn.mjs");
143
+ const { openFederation } = await import("./db-O-cyAPfS.mjs");
144
+ const { SQLiteBackend } = await import("./sqlite-D1IaR8Am.mjs");
148
145
  return new SQLiteBackend(openFederation());
149
146
  }
150
147
 
151
148
  //#endregion
152
- export { humanDuration as i, factory_exports as n, getBackendOutage as r, createStorageBackend as t };
153
- //# sourceMappingURL=factory-BGH0COXb.mjs.map
149
+ export { getBackendOutage as n, humanDuration as r, createStorageBackend as t };
150
+ //# sourceMappingURL=factory-Bsp7xOpO.mjs.map
@@ -1 +1 @@
1
- {"version":3,"file":"factory-BGH0COXb.mjs","names":[],"sources":["../src/storage/outage.ts","../src/storage/factory.ts"],"sourcesContent":["/**\n * outage.ts — is the storage backend actually reachable right now?\n *\n * The daemon retries a dead backend forever, which is correct: a Postgres\n * container that is down will usually come back, and giving up would lose the\n * work queue. What was wrong is that it did so in complete silence.\n *\n * Observed 2026-07-26: the container was down for roughly two days. The daemon\n * logged \"Postgres unavailable\" 144 times over 36 minutes, the work queue backed\n * up, session notes for the whole period were never written — and\n * `pai daemon status` reported \"Index: idle\" throughout. The one command anyone\n * would run to check said everything was fine.\n *\n * So the outage is recorded where the status command can see it, and escalated\n * once through the notification channels that were already configured and\n * already unused for this.\n */\n\nexport interface BackendOutage {\n backend: string;\n /** When the current run of failures began. */\n since: number;\n /** Consecutive failed attempts so far. */\n attempts: number;\n lastError: string;\n}\n\nlet current: BackendOutage | null = null;\n\n/** Record that the backend is currently unreachable. */\nexport function setBackendOutage(outage: BackendOutage): void {\n current = outage;\n}\n\n/**\n * Record that the backend answered.\n *\n * Called on every successful connection, including the first — so a daemon that\n * never had a problem reports none, and one that recovered stops reporting an\n * outage that has ended.\n */\nexport function clearBackendOutage(): void {\n current = null;\n}\n\n/** The current outage, or null when the backend is answering. */\nexport function getBackendOutage(): BackendOutage | null {\n return current;\n}\n\n/**\n * Human-readable elapsed time, for a status line rather than a log.\n *\n * Exported because the status command needs the duration on its own, to colour\n * it separately from the rest of the sentence. Without this it kept a second\n * copy of the same rounding, which is how two renderings of one outage drift.\n */\nexport function humanDuration(ms: number): string {\n const mins = Math.max(1, Math.round(ms / 60_000));\n return mins < 60 ? `${mins} min` : `${(mins / 60).toFixed(1)} h`;\n}\n\n/** Human-readable duration, for a status line rather than a log. */\nexport function describeOutage(o: BackendOutage, now = Date.now()): string {\n return `${o.backend} unreachable for ${humanDuration(now - o.since)}, ${o.attempts} attempts — ${o.lastError}`;\n}\n","/**\n * Storage backend factory.\n *\n * Reads the daemon config and returns the appropriate StorageBackend.\n *\n * When Postgres is the configured backend we NEVER silently fall back to\n * SQLite — doing so would split the corpus across two databases. Instead:\n * - Daemon (waitForPostgres: true) retries Postgres forever with capped\n * backoff until it comes up (handles the boot race where launchd starts\n * the daemon before Docker Desktop / Postgres is ready).\n * - CLI / one-shot callers (default) retry a few times, then throw a clear\n * error rather than returning a wrong/empty SQLite database.\n */\n\nimport type { PaiDaemonConfig } from \"../daemon/config.js\";\nimport type { StorageBackend } from \"./interface.js\";\nimport { setBackendOutage, clearBackendOutage } from \"./outage.js\";\n\nexport interface StorageBackendOptions {\n /**\n * When true, retry Postgres indefinitely instead of giving up. Used by the\n * long-lived daemon so a not-yet-ready Postgres at boot is tolerated.\n * Defaults to false (one-shot CLI behaviour: bounded retries, then throw).\n */\n waitForPostgres?: boolean;\n}\n\n/** Backoff schedule (ms) for the bounded CLI retry path. */\nconst CLI_RETRY_DELAYS_MS = [500, 1_000, 2_000];\n\n/** Backoff cap (ms) for the daemon's infinite retry path. */\nconst DAEMON_RETRY_CAP_MS = 15_000;\n\n/**\n * Create and return the configured StorageBackend.\n *\n * Auto-behaviour:\n * - storageBackend = \"sqlite\" → SQLiteBackend always\n * - storageBackend = \"postgres\" → PostgresBackend (retried; never falls back)\n */\nexport async function createStorageBackend(\n config: PaiDaemonConfig,\n opts: StorageBackendOptions = {}\n): Promise<StorageBackend> {\n if (config.storageBackend === \"postgres\") {\n return await connectPostgres(config, opts.waitForPostgres ?? false);\n }\n\n // Default: SQLite\n return createSQLiteBackend();\n}\n\n/**\n * Attempt a single Postgres connection (ensure DB + test). Returns the live\n * backend on success, or an error string describing why it failed.\n */\nasync function attemptPostgres(\n config: PaiDaemonConfig\n): Promise<{ backend: StorageBackend } | { error: string }> {\n const { PostgresBackend } = await import(\"./postgres.js\");\n const pgConfig = config.postgres ?? {};\n\n let backend: InstanceType<typeof PostgresBackend> | null = null;\n try {\n // Ensure the per-user database exists and has the schema applied.\n await PostgresBackend.ensureDatabase(pgConfig);\n\n backend = new PostgresBackend(pgConfig);\n const err = await backend.testConnection();\n if (err) {\n await backend.close().catch(() => {});\n return { error: err };\n }\n return { backend };\n } catch (e) {\n if (backend) await backend.close().catch(() => {});\n return { error: e instanceof Error ? e.message : String(e) };\n }\n}\n\n/**\n * Consecutive failures before the outage is escalated to the user.\n *\n * Not the first failure: a container restarting, or the daemon starting before\n * Docker is up, recovers within a few seconds and is not worth a notification.\n * By the fifth attempt the backoff has already spent tens of seconds, which is\n * long enough that something is actually wrong.\n */\nconst ESCALATE_AFTER_ATTEMPTS = 5;\n\n/** Tell the user the backend is down, through whatever channels are configured. */\nasync function notifyBackendDown(attempts: number, lastError: string): Promise<void> {\n try {\n const { routeNotification } = await import(\"../notifications/router.js\");\n const { loadConfig } = await import(\"../daemon/config.js\");\n await routeNotification(\n {\n event: \"error\",\n title: \"PAI: storage backend unreachable\",\n message:\n `Postgres has not answered in ${attempts} attempts (${lastError}). ` +\n `Indexing, session notes and the work queue are stalled until it returns. ` +\n `Check the container, then \\`pai daemon status\\`.`,\n },\n loadConfig().notifications\n );\n } catch {\n // A notification that cannot be sent must never take the daemon down with\n // it — the daemon retrying is still the useful behaviour here.\n }\n}\n\n/** And say when it comes back, so the alert is not left hanging. */\nasync function notifyBackendRecovered(\n attempts: number,\n since: number | null\n): Promise<void> {\n try {\n const { routeNotification } = await import(\"../notifications/router.js\");\n const { loadConfig } = await import(\"../daemon/config.js\");\n const mins = since ? Math.max(1, Math.round((Date.now() - since) / 60_000)) : null;\n await routeNotification(\n {\n event: \"completion\",\n title: \"PAI: storage backend back\",\n message:\n `Postgres answered after ${attempts} attempts` +\n (mins ? `, ${mins} min down` : \"\") +\n `. The queue will drain on its own.`,\n },\n loadConfig().notifications\n );\n } catch {\n /* same reasoning as above */\n }\n}\n\nasync function connectPostgres(\n config: PaiDaemonConfig,\n waitForever: boolean\n): Promise<StorageBackend> {\n let attempt = 0;\n let lastError = \"unknown error\";\n let outageSince: number | null = null;\n let escalated = false;\n\n // eslint-disable-next-line no-constant-condition\n while (true) {\n attempt++;\n const result = await attemptPostgres(config);\n if (\"backend\" in result) {\n if (attempt > 1) {\n process.stderr.write(\n `[pai-daemon] Connected to PostgreSQL backend (after ${attempt} attempts).\\n`\n );\n } else {\n process.stderr.write(\"[pai-daemon] Connected to PostgreSQL backend.\\n\");\n }\n // An outage that ended must stop being reported, or the status command\n // trades one wrong answer for another.\n clearBackendOutage();\n if (escalated) void notifyBackendRecovered(attempt, outageSince);\n return result.backend;\n }\n\n lastError = result.error;\n\n if (!waitForever && attempt > CLI_RETRY_DELAYS_MS.length) {\n // Bounded CLI path exhausted — fail loudly, never silently use SQLite.\n throw new Error(\n `Postgres backend unreachable after ${attempt} attempts: ${lastError}. ` +\n `Is Docker Desktop / Postgres running? Refusing to fall back to SQLite ` +\n `(would split the corpus). Start Postgres and retry.`\n );\n }\n\n const delayMs = waitForever\n ? Math.min(DAEMON_RETRY_CAP_MS, 1_000 * 2 ** Math.min(attempt - 1, 4))\n : CLI_RETRY_DELAYS_MS[attempt - 1];\n\n process.stderr.write(\n `[pai-daemon] Postgres unavailable (${lastError}). ` +\n `Retry ${attempt}${waitForever ? \"\" : `/${CLI_RETRY_DELAYS_MS.length + 1}`} ` +\n `in ${delayMs}ms...\\n`\n );\n\n // Publish the outage so `pai daemon status` can report it. Without this the\n // daemon retries silently forever and status still reads \"idle\" — which is\n // what happened for two days in July: 144 retries over 36 minutes, session\n // notes never written, and the one command anyone would run to check\n // reporting that everything was fine.\n setBackendOutage({\n backend: \"postgres\",\n since: outageSince ?? (outageSince = Date.now()),\n attempts: attempt,\n lastError: String(lastError),\n });\n\n // And escalate once, out loud, rather than only into a log nobody tails.\n // Once — not per retry — because a notification that repeats every few\n // seconds is filtered within a minute and stops being a signal at all.\n if (waitForever && attempt === ESCALATE_AFTER_ATTEMPTS && !escalated) {\n escalated = true;\n void notifyBackendDown(attempt, String(lastError));\n }\n\n await new Promise((r) => setTimeout(r, delayMs));\n }\n}\n\nasync function createSQLiteBackend(): Promise<StorageBackend> {\n const { openFederation } = await import(\"../memory/db.js\");\n const { SQLiteBackend } = await import(\"./sqlite.js\");\n const db = openFederation();\n return new SQLiteBackend(db);\n}\n"],"mappings":";;;AA2BA,IAAI,UAAgC;;AAGpC,SAAgB,iBAAiB,QAA6B;AAC5D,WAAU;;;;;;;;;AAUZ,SAAgB,qBAA2B;AACzC,WAAU;;;AAIZ,SAAgB,mBAAyC;AACvD,QAAO;;;;;;;;;AAUT,SAAgB,cAAc,IAAoB;CAChD,MAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,KAAK,IAAO,CAAC;AACjD,QAAO,OAAO,KAAK,GAAG,KAAK,QAAQ,IAAI,OAAO,IAAI,QAAQ,EAAE,CAAC;;;;;;;AC/B/D,MAAM,sBAAsB;CAAC;CAAK;CAAO;CAAM;;AAG/C,MAAM,sBAAsB;;;;;;;;AAS5B,eAAsB,qBACpB,QACA,OAA8B,EAAE,EACP;AACzB,KAAI,OAAO,mBAAmB,WAC5B,QAAO,MAAM,gBAAgB,QAAQ,KAAK,mBAAmB,MAAM;AAIrE,QAAO,qBAAqB;;;;;;AAO9B,eAAe,gBACb,QAC0D;CAC1D,MAAM,EAAE,oBAAoB,MAAM,OAAO;CACzC,MAAM,WAAW,OAAO,YAAY,EAAE;CAEtC,IAAI,UAAuD;AAC3D,KAAI;AAEF,QAAM,gBAAgB,eAAe,SAAS;AAE9C,YAAU,IAAI,gBAAgB,SAAS;EACvC,MAAM,MAAM,MAAM,QAAQ,gBAAgB;AAC1C,MAAI,KAAK;AACP,SAAM,QAAQ,OAAO,CAAC,YAAY,GAAG;AACrC,UAAO,EAAE,OAAO,KAAK;;AAEvB,SAAO,EAAE,SAAS;UACX,GAAG;AACV,MAAI,QAAS,OAAM,QAAQ,OAAO,CAAC,YAAY,GAAG;AAClD,SAAO,EAAE,OAAO,aAAa,QAAQ,EAAE,UAAU,OAAO,EAAE,EAAE;;;;;;;;;;;AAYhE,MAAM,0BAA0B;;AAGhC,eAAe,kBAAkB,UAAkB,WAAkC;AACnF,KAAI;EACF,MAAM,EAAE,sBAAsB,MAAM,OAAO;EAC3C,MAAM,EAAE,eAAe,MAAM,OAAO;AACpC,QAAM,kBACJ;GACE,OAAO;GACP,OAAO;GACP,SACE,gCAAgC,SAAS,aAAa,UAAU;GAGnE,EACD,YAAY,CAAC,cACd;SACK;;;AAOV,eAAe,uBACb,UACA,OACe;AACf,KAAI;EACF,MAAM,EAAE,sBAAsB,MAAM,OAAO;EAC3C,MAAM,EAAE,eAAe,MAAM,OAAO;EACpC,MAAM,OAAO,QAAQ,KAAK,IAAI,GAAG,KAAK,OAAO,KAAK,KAAK,GAAG,SAAS,IAAO,CAAC,GAAG;AAC9E,QAAM,kBACJ;GACE,OAAO;GACP,OAAO;GACP,SACE,2BAA2B,SAAS,cACnC,OAAO,KAAK,KAAK,aAAa,MAC/B;GACH,EACD,YAAY,CAAC,cACd;SACK;;AAKV,eAAe,gBACb,QACA,aACyB;CACzB,IAAI,UAAU;CACd,IAAI,YAAY;CAChB,IAAI,cAA6B;CACjC,IAAI,YAAY;AAGhB,QAAO,MAAM;AACX;EACA,MAAM,SAAS,MAAM,gBAAgB,OAAO;AAC5C,MAAI,aAAa,QAAQ;AACvB,OAAI,UAAU,EACZ,SAAQ,OAAO,MACb,uDAAuD,QAAQ,eAChE;OAED,SAAQ,OAAO,MAAM,kDAAkD;AAIzE,uBAAoB;AACpB,OAAI,UAAW,CAAK,uBAAuB,SAAS,YAAY;AAChE,UAAO,OAAO;;AAGhB,cAAY,OAAO;AAEnB,MAAI,CAAC,eAAe,UAAU,oBAAoB,OAEhD,OAAM,IAAI,MACR,sCAAsC,QAAQ,aAAa,UAAU,6HAGtE;EAGH,MAAM,UAAU,cACZ,KAAK,IAAI,qBAAqB,MAAQ,KAAK,KAAK,IAAI,UAAU,GAAG,EAAE,CAAC,GACpE,oBAAoB,UAAU;AAElC,UAAQ,OAAO,MACb,sCAAsC,UAAU,WACrC,UAAU,cAAc,KAAK,IAAI,oBAAoB,SAAS,IAAI,MACrE,QAAQ,SACjB;AAOD,mBAAiB;GACf,SAAS;GACT,OAAO,gBAAgB,cAAc,KAAK,KAAK;GAC/C,UAAU;GACV,WAAW,OAAO,UAAU;GAC7B,CAAC;AAKF,MAAI,eAAe,YAAY,2BAA2B,CAAC,WAAW;AACpE,eAAY;AACZ,GAAK,kBAAkB,SAAS,OAAO,UAAU,CAAC;;AAGpD,QAAM,IAAI,SAAS,MAAM,WAAW,GAAG,QAAQ,CAAC;;;AAIpD,eAAe,sBAA+C;CAC5D,MAAM,EAAE,mBAAmB,MAAM,OAAO;CACxC,MAAM,EAAE,kBAAkB,MAAM,OAAO;AAEvC,QAAO,IAAI,cADA,gBAAgB,CACC"}
1
+ {"version":3,"file":"factory-Bsp7xOpO.mjs","names":[],"sources":["../src/storage/outage.ts","../src/storage/factory.ts"],"sourcesContent":["/**\n * outage.ts — is the storage backend actually reachable right now?\n *\n * The daemon retries a dead backend forever, which is correct: a Postgres\n * container that is down will usually come back, and giving up would lose the\n * work queue. What was wrong is that it did so in complete silence.\n *\n * Observed 2026-07-26: the container was down for roughly two days. The daemon\n * logged \"Postgres unavailable\" 144 times over 36 minutes, the work queue backed\n * up, session notes for the whole period were never written — and\n * `pai daemon status` reported \"Index: idle\" throughout. The one command anyone\n * would run to check said everything was fine.\n *\n * So the outage is recorded where the status command can see it, and escalated\n * once through the notification channels that were already configured and\n * already unused for this.\n */\n\nexport interface BackendOutage {\n backend: string;\n /** When the current run of failures began. */\n since: number;\n /** Consecutive failed attempts so far. */\n attempts: number;\n lastError: string;\n}\n\nlet current: BackendOutage | null = null;\n\n/** Record that the backend is currently unreachable. */\nexport function setBackendOutage(outage: BackendOutage): void {\n current = outage;\n}\n\n/**\n * Record that the backend answered.\n *\n * Called on every successful connection, including the first — so a daemon that\n * never had a problem reports none, and one that recovered stops reporting an\n * outage that has ended.\n */\nexport function clearBackendOutage(): void {\n current = null;\n}\n\n/** The current outage, or null when the backend is answering. */\nexport function getBackendOutage(): BackendOutage | null {\n return current;\n}\n\n/**\n * Human-readable elapsed time, for a status line rather than a log.\n *\n * Exported because the status command needs the duration on its own, to colour\n * it separately from the rest of the sentence. Without this it kept a second\n * copy of the same rounding, which is how two renderings of one outage drift.\n */\nexport function humanDuration(ms: number): string {\n const mins = Math.max(1, Math.round(ms / 60_000));\n return mins < 60 ? `${mins} min` : `${(mins / 60).toFixed(1)} h`;\n}\n\n/** Human-readable duration, for a status line rather than a log. */\nexport function describeOutage(o: BackendOutage, now = Date.now()): string {\n return `${o.backend} unreachable for ${humanDuration(now - o.since)}, ${o.attempts} attempts — ${o.lastError}`;\n}\n","/**\n * Storage backend factory.\n *\n * Reads the daemon config and returns the appropriate StorageBackend.\n *\n * When Postgres is the configured backend we NEVER silently fall back to\n * SQLite — doing so would split the corpus across two databases. Instead:\n * - Daemon (waitForPostgres: true) retries Postgres forever with capped\n * backoff until it comes up (handles the boot race where launchd starts\n * the daemon before Docker Desktop / Postgres is ready).\n * - CLI / one-shot callers (default) retry a few times, then throw a clear\n * error rather than returning a wrong/empty SQLite database.\n */\n\nimport type { PaiDaemonConfig } from \"../daemon/config.js\";\nimport type { StorageBackend } from \"./interface.js\";\nimport { setBackendOutage, clearBackendOutage } from \"./outage.js\";\n\nexport interface StorageBackendOptions {\n /**\n * When true, retry Postgres indefinitely instead of giving up. Used by the\n * long-lived daemon so a not-yet-ready Postgres at boot is tolerated.\n * Defaults to false (one-shot CLI behaviour: bounded retries, then throw).\n */\n waitForPostgres?: boolean;\n}\n\n/** Backoff schedule (ms) for the bounded CLI retry path. */\nconst CLI_RETRY_DELAYS_MS = [500, 1_000, 2_000];\n\n/** Backoff cap (ms) for the daemon's infinite retry path. */\nconst DAEMON_RETRY_CAP_MS = 15_000;\n\n/**\n * Create and return the configured StorageBackend.\n *\n * Auto-behaviour:\n * - storageBackend = \"sqlite\" → SQLiteBackend always\n * - storageBackend = \"postgres\" → PostgresBackend (retried; never falls back)\n */\nexport async function createStorageBackend(\n config: PaiDaemonConfig,\n opts: StorageBackendOptions = {}\n): Promise<StorageBackend> {\n if (config.storageBackend === \"postgres\") {\n return await connectPostgres(config, opts.waitForPostgres ?? false);\n }\n\n // Default: SQLite\n return createSQLiteBackend();\n}\n\n/**\n * Attempt a single Postgres connection (ensure DB + test). Returns the live\n * backend on success, or an error string describing why it failed.\n */\nasync function attemptPostgres(\n config: PaiDaemonConfig\n): Promise<{ backend: StorageBackend } | { error: string }> {\n const { PostgresBackend } = await import(\"./postgres.js\");\n const pgConfig = config.postgres ?? {};\n\n let backend: InstanceType<typeof PostgresBackend> | null = null;\n try {\n // Ensure the per-user database exists and has the schema applied.\n await PostgresBackend.ensureDatabase(pgConfig);\n\n backend = new PostgresBackend(pgConfig);\n const err = await backend.testConnection();\n if (err) {\n await backend.close().catch(() => {});\n return { error: err };\n }\n return { backend };\n } catch (e) {\n if (backend) await backend.close().catch(() => {});\n return { error: e instanceof Error ? e.message : String(e) };\n }\n}\n\n/**\n * Consecutive failures before the outage is escalated to the user.\n *\n * Not the first failure: a container restarting, or the daemon starting before\n * Docker is up, recovers within a few seconds and is not worth a notification.\n * By the fifth attempt the backoff has already spent tens of seconds, which is\n * long enough that something is actually wrong.\n */\nconst ESCALATE_AFTER_ATTEMPTS = 5;\n\n/** Tell the user the backend is down, through whatever channels are configured. */\nasync function notifyBackendDown(attempts: number, lastError: string): Promise<void> {\n try {\n const { routeNotification } = await import(\"../notifications/router.js\");\n const { loadConfig } = await import(\"../daemon/config.js\");\n await routeNotification(\n {\n event: \"error\",\n title: \"PAI: storage backend unreachable\",\n message:\n `Postgres has not answered in ${attempts} attempts (${lastError}). ` +\n `Indexing, session notes and the work queue are stalled until it returns. ` +\n `Check the container, then \\`pai daemon status\\`.`,\n },\n loadConfig().notifications\n );\n } catch {\n // A notification that cannot be sent must never take the daemon down with\n // it — the daemon retrying is still the useful behaviour here.\n }\n}\n\n/** And say when it comes back, so the alert is not left hanging. */\nasync function notifyBackendRecovered(\n attempts: number,\n since: number | null\n): Promise<void> {\n try {\n const { routeNotification } = await import(\"../notifications/router.js\");\n const { loadConfig } = await import(\"../daemon/config.js\");\n const mins = since ? Math.max(1, Math.round((Date.now() - since) / 60_000)) : null;\n await routeNotification(\n {\n event: \"completion\",\n title: \"PAI: storage backend back\",\n message:\n `Postgres answered after ${attempts} attempts` +\n (mins ? `, ${mins} min down` : \"\") +\n `. The queue will drain on its own.`,\n },\n loadConfig().notifications\n );\n } catch {\n /* same reasoning as above */\n }\n}\n\nasync function connectPostgres(\n config: PaiDaemonConfig,\n waitForever: boolean\n): Promise<StorageBackend> {\n let attempt = 0;\n let lastError = \"unknown error\";\n let outageSince: number | null = null;\n let escalated = false;\n\n // eslint-disable-next-line no-constant-condition\n while (true) {\n attempt++;\n const result = await attemptPostgres(config);\n if (\"backend\" in result) {\n if (attempt > 1) {\n process.stderr.write(\n `[pai-daemon] Connected to PostgreSQL backend (after ${attempt} attempts).\\n`\n );\n } else {\n process.stderr.write(\"[pai-daemon] Connected to PostgreSQL backend.\\n\");\n }\n // An outage that ended must stop being reported, or the status command\n // trades one wrong answer for another.\n clearBackendOutage();\n if (escalated) void notifyBackendRecovered(attempt, outageSince);\n return result.backend;\n }\n\n lastError = result.error;\n\n if (!waitForever && attempt > CLI_RETRY_DELAYS_MS.length) {\n // Bounded CLI path exhausted — fail loudly, never silently use SQLite.\n throw new Error(\n `Postgres backend unreachable after ${attempt} attempts: ${lastError}. ` +\n `Is Docker Desktop / Postgres running? Refusing to fall back to SQLite ` +\n `(would split the corpus). Start Postgres and retry.`\n );\n }\n\n const delayMs = waitForever\n ? Math.min(DAEMON_RETRY_CAP_MS, 1_000 * 2 ** Math.min(attempt - 1, 4))\n : CLI_RETRY_DELAYS_MS[attempt - 1];\n\n process.stderr.write(\n `[pai-daemon] Postgres unavailable (${lastError}). ` +\n `Retry ${attempt}${waitForever ? \"\" : `/${CLI_RETRY_DELAYS_MS.length + 1}`} ` +\n `in ${delayMs}ms...\\n`\n );\n\n // Publish the outage so `pai daemon status` can report it. Without this the\n // daemon retries silently forever and status still reads \"idle\" — which is\n // what happened for two days in July: 144 retries over 36 minutes, session\n // notes never written, and the one command anyone would run to check\n // reporting that everything was fine.\n setBackendOutage({\n backend: \"postgres\",\n since: outageSince ?? (outageSince = Date.now()),\n attempts: attempt,\n lastError: String(lastError),\n });\n\n // And escalate once, out loud, rather than only into a log nobody tails.\n // Once — not per retry — because a notification that repeats every few\n // seconds is filtered within a minute and stops being a signal at all.\n if (waitForever && attempt === ESCALATE_AFTER_ATTEMPTS && !escalated) {\n escalated = true;\n void notifyBackendDown(attempt, String(lastError));\n }\n\n await new Promise((r) => setTimeout(r, delayMs));\n }\n}\n\nasync function createSQLiteBackend(): Promise<StorageBackend> {\n const { openFederation } = await import(\"../memory/db.js\");\n const { SQLiteBackend } = await import(\"./sqlite.js\");\n const db = openFederation();\n return new SQLiteBackend(db);\n}\n"],"mappings":";AA2BA,IAAI,UAAgC;;AAGpC,SAAgB,iBAAiB,QAA6B;AAC5D,WAAU;;;;;;;;;AAUZ,SAAgB,qBAA2B;AACzC,WAAU;;;AAIZ,SAAgB,mBAAyC;AACvD,QAAO;;;;;;;;;AAUT,SAAgB,cAAc,IAAoB;CAChD,MAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,KAAK,IAAO,CAAC;AACjD,QAAO,OAAO,KAAK,GAAG,KAAK,QAAQ,IAAI,OAAO,IAAI,QAAQ,EAAE,CAAC;;;;;;AC/B/D,MAAM,sBAAsB;CAAC;CAAK;CAAO;CAAM;;AAG/C,MAAM,sBAAsB;;;;;;;;AAS5B,eAAsB,qBACpB,QACA,OAA8B,EAAE,EACP;AACzB,KAAI,OAAO,mBAAmB,WAC5B,QAAO,MAAM,gBAAgB,QAAQ,KAAK,mBAAmB,MAAM;AAIrE,QAAO,qBAAqB;;;;;;AAO9B,eAAe,gBACb,QAC0D;CAC1D,MAAM,EAAE,oBAAoB,MAAM,OAAO;CACzC,MAAM,WAAW,OAAO,YAAY,EAAE;CAEtC,IAAI,UAAuD;AAC3D,KAAI;AAEF,QAAM,gBAAgB,eAAe,SAAS;AAE9C,YAAU,IAAI,gBAAgB,SAAS;EACvC,MAAM,MAAM,MAAM,QAAQ,gBAAgB;AAC1C,MAAI,KAAK;AACP,SAAM,QAAQ,OAAO,CAAC,YAAY,GAAG;AACrC,UAAO,EAAE,OAAO,KAAK;;AAEvB,SAAO,EAAE,SAAS;UACX,GAAG;AACV,MAAI,QAAS,OAAM,QAAQ,OAAO,CAAC,YAAY,GAAG;AAClD,SAAO,EAAE,OAAO,aAAa,QAAQ,EAAE,UAAU,OAAO,EAAE,EAAE;;;;;;;;;;;AAYhE,MAAM,0BAA0B;;AAGhC,eAAe,kBAAkB,UAAkB,WAAkC;AACnF,KAAI;EACF,MAAM,EAAE,sBAAsB,MAAM,OAAO;EAC3C,MAAM,EAAE,eAAe,MAAM,OAAO;AACpC,QAAM,kBACJ;GACE,OAAO;GACP,OAAO;GACP,SACE,gCAAgC,SAAS,aAAa,UAAU;GAGnE,EACD,YAAY,CAAC,cACd;SACK;;;AAOV,eAAe,uBACb,UACA,OACe;AACf,KAAI;EACF,MAAM,EAAE,sBAAsB,MAAM,OAAO;EAC3C,MAAM,EAAE,eAAe,MAAM,OAAO;EACpC,MAAM,OAAO,QAAQ,KAAK,IAAI,GAAG,KAAK,OAAO,KAAK,KAAK,GAAG,SAAS,IAAO,CAAC,GAAG;AAC9E,QAAM,kBACJ;GACE,OAAO;GACP,OAAO;GACP,SACE,2BAA2B,SAAS,cACnC,OAAO,KAAK,KAAK,aAAa,MAC/B;GACH,EACD,YAAY,CAAC,cACd;SACK;;AAKV,eAAe,gBACb,QACA,aACyB;CACzB,IAAI,UAAU;CACd,IAAI,YAAY;CAChB,IAAI,cAA6B;CACjC,IAAI,YAAY;AAGhB,QAAO,MAAM;AACX;EACA,MAAM,SAAS,MAAM,gBAAgB,OAAO;AAC5C,MAAI,aAAa,QAAQ;AACvB,OAAI,UAAU,EACZ,SAAQ,OAAO,MACb,uDAAuD,QAAQ,eAChE;OAED,SAAQ,OAAO,MAAM,kDAAkD;AAIzE,uBAAoB;AACpB,OAAI,UAAW,CAAK,uBAAuB,SAAS,YAAY;AAChE,UAAO,OAAO;;AAGhB,cAAY,OAAO;AAEnB,MAAI,CAAC,eAAe,UAAU,oBAAoB,OAEhD,OAAM,IAAI,MACR,sCAAsC,QAAQ,aAAa,UAAU,6HAGtE;EAGH,MAAM,UAAU,cACZ,KAAK,IAAI,qBAAqB,MAAQ,KAAK,KAAK,IAAI,UAAU,GAAG,EAAE,CAAC,GACpE,oBAAoB,UAAU;AAElC,UAAQ,OAAO,MACb,sCAAsC,UAAU,WACrC,UAAU,cAAc,KAAK,IAAI,oBAAoB,SAAS,IAAI,MACrE,QAAQ,SACjB;AAOD,mBAAiB;GACf,SAAS;GACT,OAAO,gBAAgB,cAAc,KAAK,KAAK;GAC/C,UAAU;GACV,WAAW,OAAO,UAAU;GAC7B,CAAC;AAKF,MAAI,eAAe,YAAY,2BAA2B,CAAC,WAAW;AACpE,eAAY;AACZ,GAAK,kBAAkB,SAAS,OAAO,UAAU,CAAC;;AAGpD,QAAM,IAAI,SAAS,MAAM,WAAW,GAAG,QAAQ,CAAC;;;AAIpD,eAAe,sBAA+C;CAC5D,MAAM,EAAE,mBAAmB,MAAM,OAAO;CACxC,MAAM,EAAE,kBAAkB,MAAM,OAAO;AAEvC,QAAO,IAAI,cADA,gBAAgB,CACC"}
@@ -0,0 +1,3 @@
1
+ import { t as createStorageBackend } from "./factory-Bsp7xOpO.mjs";
2
+
3
+ export { createStorageBackend };
@@ -425,4 +425,4 @@ const INDEX_YIELD_EVERY = 10;
425
425
 
426
426
  //#endregion
427
427
  export { parseSessionTitleChunk as a, yieldToEventLoop as c, sha256 as d, sha256File as f, isPathTooBroadForContentScan as i, chunkMarkdown as l, chunkId as n, walkContentFiles as o, detectTier as r, walkMdFiles as s, INDEX_YIELD_EVERY as t, estimateTokens as u };
428
- //# sourceMappingURL=helpers-crDEr6S2.mjs.map
428
+ //# sourceMappingURL=helpers-IjZkXBhj.mjs.map
@@ -1 +1 @@
1
- {"version":3,"file":"helpers-crDEr6S2.mjs","names":[],"sources":["../src/utils/hash.ts","../src/memory/chunker.ts","../src/memory/indexer/helpers.ts"],"sourcesContent":["/**\n * Shared hashing utilities. Centralises all SHA-256 usage so every module\n * obtains digests through the same function rather than inlining createHash.\n */\n\nimport { createHash } from \"node:crypto\";\n\n/**\n * Compute a SHA-256 hex digest of the given string.\n * Aliased as sha256File for compatibility with existing call-sites that use\n * that name to hash file contents.\n */\nexport function sha256(content: string): string {\n return createHash(\"sha256\").update(content).digest(\"hex\");\n}\n\n/** Alias kept for backwards compatibility with memory/indexer call-sites. */\nexport const sha256File = sha256;\n","/**\n * Markdown text chunker for the PAI memory engine.\n *\n * Splits markdown files into overlapping text segments suitable for BM25\n * full-text indexing. Respects heading boundaries where possible, falling\n * back to paragraph and sentence splitting when sections are large.\n */\n\nimport { sha256 } from \"../utils/hash.js\";\n\nexport interface Chunk {\n text: string;\n startLine: number; // 1-indexed\n endLine: number; // 1-indexed, inclusive\n hash: string; // SHA-256 of text\n}\n\nexport interface ChunkOptions {\n /** Approximate maximum tokens per chunk. Default 400. */\n maxTokens?: number;\n /** Overlap in tokens from the previous chunk. Default 80. */\n overlap?: number;\n}\n\nconst DEFAULT_MAX_TOKENS = 400;\nconst DEFAULT_OVERLAP = 80;\n\n/**\n * Approximate token count using a words * 1.3 heuristic.\n * Matches the OpenClaw estimate approach.\n */\nexport function estimateTokens(text: string): number {\n const wordCount = text.split(/\\s+/).filter(Boolean).length;\n return Math.ceil(wordCount * 1.3);\n}\n\n// sha256 imported from utils/hash.ts\n\n// ---------------------------------------------------------------------------\n// Internal section / paragraph / sentence splitters\n// ---------------------------------------------------------------------------\n\n/**\n * A contiguous block of lines associated with an approximate token count.\n */\ninterface LineBlock {\n lines: Array<{ text: string; lineNo: number }>;\n tokens: number;\n}\n\n/**\n * Split content into sections delimited by ## or ### headings.\n * Each section starts at its heading line (or at line 1 for a preamble).\n */\nfunction splitBySections(\n lines: Array<{ text: string; lineNo: number }>,\n): LineBlock[] {\n const sections: LineBlock[] = [];\n let current: Array<{ text: string; lineNo: number }> = [];\n\n for (const line of lines) {\n const isHeading = /^#{1,3}\\s/.test(line.text);\n if (isHeading && current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n sections.push({ lines: current, tokens: estimateTokens(text) });\n current = [];\n }\n current.push(line);\n }\n\n if (current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n sections.push({ lines: current, tokens: estimateTokens(text) });\n }\n\n return sections;\n}\n\n/**\n * Split a LineBlock by double-newline paragraph boundaries.\n */\nfunction splitByParagraphs(block: LineBlock): LineBlock[] {\n const paragraphs: LineBlock[] = [];\n let current: Array<{ text: string; lineNo: number }> = [];\n\n for (const line of block.lines) {\n if (line.text.trim() === \"\" && current.length > 0) {\n // Empty line — potential paragraph boundary\n const text = current.map((l) => l.text).join(\"\\n\");\n paragraphs.push({ lines: [...current], tokens: estimateTokens(text) });\n current = [];\n } else {\n current.push(line);\n }\n }\n\n if (current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n paragraphs.push({ lines: current, tokens: estimateTokens(text) });\n }\n\n return paragraphs.length > 0 ? paragraphs : [block];\n}\n\n/**\n * Split a LineBlock by sentence boundaries (. ! ?) when even paragraphs are\n * too large. Works character-by-character within joined lines.\n */\nfunction splitBySentences(block: LineBlock, maxTokens: number): LineBlock[] {\n const fullText = block.lines.map((l) => l.text).join(\" \");\n // Very rough sentence split — split on '. ', '! ', '? ' followed by uppercase\n const sentenceRe = /(?<=[.!?])\\s+(?=[A-Z\"'])/g;\n const sentences = fullText.split(sentenceRe);\n\n const result: LineBlock[] = [];\n let accText = \"\";\n // We can't recover exact line numbers inside a single oversized paragraph,\n // so we approximate using the block's start/end lines distributed evenly.\n const startLine = block.lines[0]?.lineNo ?? 1;\n const endLine = block.lines[block.lines.length - 1]?.lineNo ?? startLine;\n const totalLines = endLine - startLine + 1;\n const linesPerSentence = Math.max(1, Math.floor(totalLines / Math.max(1, sentences.length)));\n\n let sentenceIdx = 0;\n let approxLine = startLine;\n\n const flush = () => {\n if (!accText.trim()) return;\n const endApprox = Math.min(approxLine + linesPerSentence - 1, endLine);\n result.push({\n lines: [{ text: accText.trim(), lineNo: approxLine }],\n tokens: estimateTokens(accText),\n });\n approxLine = endApprox + 1;\n accText = \"\";\n };\n\n for (const sentence of sentences) {\n sentenceIdx++;\n const candidateText = accText ? accText + \" \" + sentence : sentence;\n if (estimateTokens(candidateText) > maxTokens && accText) {\n flush();\n accText = sentence;\n } else {\n accText = candidateText;\n }\n }\n void sentenceIdx; // used only for iteration count\n flush();\n\n return result.length > 0 ? result : [block];\n}\n\n// ---------------------------------------------------------------------------\n// Overlap helper\n// ---------------------------------------------------------------------------\n\n/**\n * Extract the last `overlapTokens` worth of text from a list of previously\n * emitted chunks to prepend to the next chunk.\n */\nfunction buildOverlapPrefix(\n chunks: Chunk[],\n overlapTokens: number,\n): Array<{ text: string; lineNo: number }> {\n if (overlapTokens <= 0 || chunks.length === 0) return [];\n\n const lastChunk = chunks[chunks.length - 1];\n if (!lastChunk) return [];\n\n const lines = lastChunk.text.split(\"\\n\");\n const kept: string[] = [];\n let acc = 0;\n\n for (let i = lines.length - 1; i >= 0; i--) {\n const lineTokens = estimateTokens(lines[i] ?? \"\");\n acc += lineTokens;\n kept.unshift(lines[i] ?? \"\");\n if (acc >= overlapTokens) break;\n }\n\n // Distribute overlap lines across the lastChunk's line range\n const startLine = lastChunk.endLine - kept.length + 1;\n return kept.map((text, idx) => ({ text, lineNo: Math.max(lastChunk.startLine, startLine + idx) }));\n}\n\n// ---------------------------------------------------------------------------\n// Public API\n// ---------------------------------------------------------------------------\n\n/**\n * Chunk a markdown file into overlapping segments for BM25 indexing.\n *\n * Strategy:\n * 1. Split by headings (##, ###) as natural boundaries.\n * 2. If a section exceeds maxTokens, split by paragraphs.\n * 3. If a paragraph still exceeds maxTokens, split by sentences.\n * 4. Apply overlap: each chunk includes the last `overlap` tokens from the\n * previous chunk.\n */\n/**\n * Strip `<private>...</private>` blocks from content before indexing.\n * Content within these tags is excluded from memory — never stored or searched.\n */\nexport function stripPrivateTags(content: string): string {\n return content.replace(/<private>[\\s\\S]*?<\\/private>/gi, \"\");\n}\n\nexport function chunkMarkdown(content: string, opts?: ChunkOptions): Chunk[] {\n const maxTokens = opts?.maxTokens ?? DEFAULT_MAX_TOKENS;\n const overlapTokens = opts?.overlap ?? DEFAULT_OVERLAP;\n\n // Strip private content before indexing\n content = stripPrivateTags(content);\n\n if (!content.trim()) return [];\n\n const rawLines = content.split(\"\\n\");\n const lines: Array<{ text: string; lineNo: number }> = rawLines.map((text, idx) => ({\n text,\n lineNo: idx + 1, // 1-indexed\n }));\n\n // Step 1: section split\n const sections = splitBySections(lines);\n\n // Step 2 & 3: further split oversized sections\n const finalBlocks: LineBlock[] = [];\n for (const section of sections) {\n if (section.tokens <= maxTokens) {\n finalBlocks.push(section);\n continue;\n }\n // Too big — split by paragraphs\n const paras = splitByParagraphs(section);\n for (const para of paras) {\n if (para.tokens <= maxTokens) {\n finalBlocks.push(para);\n continue;\n }\n // Still too big — split by sentences\n const sentences = splitBySentences(para, maxTokens);\n finalBlocks.push(...sentences);\n }\n }\n\n // Step 4: build final chunks with overlap\n const chunks: Chunk[] = [];\n\n for (const block of finalBlocks) {\n if (block.lines.length === 0) continue;\n\n // Build overlap prefix from previous chunks\n const overlapLines = buildOverlapPrefix(chunks, overlapTokens);\n\n // Combine overlap + block lines\n const allLines = [...overlapLines, ...block.lines];\n const text = allLines.map((l) => l.text).join(\"\\n\").trim();\n\n if (!text) continue;\n\n const startLine = block.lines[0]?.lineNo ?? 1;\n const endLine = block.lines[block.lines.length - 1]?.lineNo ?? startLine;\n\n chunks.push({\n text,\n startLine,\n endLine,\n hash: sha256(text),\n });\n }\n\n return chunks;\n}\n","/**\n * Shared helpers for the PAI memory indexers.\n *\n * Contains utilities used by both the sync (SQLite) and async (StorageBackend)\n * indexer paths: hashing, chunk ID generation, directory walking, and path guards.\n */\n\nimport { readdirSync, existsSync } from \"node:fs\";\nimport { createHash } from \"node:crypto\";\nimport { sha256File } from \"../../utils/hash.js\";\nimport { join, normalize } from \"node:path\";\nimport { homedir } from \"node:os\";\nimport { basename } from \"node:path\";\n\n// ---------------------------------------------------------------------------\n// Tier detection\n// ---------------------------------------------------------------------------\n\n/**\n * Classify a relative file path into one of the four memory tiers.\n *\n * Rules (in priority order):\n * - MEMORY.md anywhere in memory/ → 'evergreen'\n * - YYYY-MM-DD.md in memory/ → 'daily'\n * - anything else in memory/ → 'topic'\n * - anything in Notes/ → 'session'\n */\nexport function detectTier(\n relativePath: string,\n): \"evergreen\" | \"daily\" | \"topic\" | \"session\" {\n // Normalise to forward slashes and strip leading ./\n const p = relativePath.replace(/\\\\/g, \"/\").replace(/^\\.\\//, \"\");\n\n // Notes directory → session tier\n if (p.startsWith(\"Notes/\") || p === \"Notes\") {\n return \"session\";\n }\n\n const fileName = basename(p);\n\n // MEMORY.md (case-sensitive match) → evergreen\n if (fileName === \"MEMORY.md\") {\n return \"evergreen\";\n }\n\n // YYYY-MM-DD.md → daily\n if (/^\\d{4}-\\d{2}-\\d{2}\\.md$/.test(fileName)) {\n return \"daily\";\n }\n\n // Default for memory/ files\n return \"topic\";\n}\n\n// ---------------------------------------------------------------------------\n// Hashing and chunk ID generation\n// ---------------------------------------------------------------------------\n\n// sha256File imported from ../../utils/hash.js\nexport { sha256File } from \"../../utils/hash.js\";\n\n/**\n * Generate a deterministic chunk ID from its coordinates.\n * Format: sha256(\"projectId:path:chunkIndex:startLine:endLine\")\n *\n * The chunkIndex (0-based position within the file) is included so that\n * chunks with approximated line numbers (e.g. from splitBySentences) never\n * produce colliding IDs even when multiple chunks share the same startLine/endLine.\n */\nexport function chunkId(\n projectId: number,\n path: string,\n chunkIndex: number,\n startLine: number,\n endLine: number,\n): string {\n return createHash(\"sha256\")\n .update(`${projectId}:${path}:${chunkIndex}:${startLine}:${endLine}`)\n .digest(\"hex\");\n}\n\n// ---------------------------------------------------------------------------\n// Event loop yield\n// ---------------------------------------------------------------------------\n\n/**\n * Yield to the Node.js event loop so that IPC server can process requests\n * during long index runs.\n *\n * Uses setTimeout(10ms) rather than setImmediate — the 10ms pause gives the\n * event loop enough time to accept and process incoming IPC connections\n * (socket data, new connections, etc.). Without this, synchronous ONNX\n * inference blocks IPC for the full duration of each embedding (~50-100ms\n * per chunk).\n */\nexport function yieldToEventLoop(): Promise<void> {\n return new Promise((resolve) => setTimeout(resolve, 10));\n}\n\n// ---------------------------------------------------------------------------\n// Directory skip sets\n// ---------------------------------------------------------------------------\n\n/**\n * Directories to ALWAYS skip, at any depth, during any directory walk.\n * These are build artifacts, dependency trees, and VCS internals that\n * should never be indexed regardless of where they appear in the tree.\n */\nexport const ALWAYS_SKIP_DIRS = new Set([\n // Version control\n \".git\",\n // Dependency directories (any language)\n \"node_modules\",\n \"vendor\",\n \"Pods\", // CocoaPods (iOS/macOS)\n // Build / compile output\n \"dist\",\n \"build\",\n \"out\",\n \"DerivedData\", // Xcode\n \".next\", // Next.js\n // Python virtual environments and caches\n \".venv\",\n \"venv\",\n \"__pycache__\",\n // General caches\n \".cache\",\n \".bun\",\n // Backup snapshots (Carbon Copy Cloner, Time Machine, etc.)\n \"snaps\",\n \".Trashes\",\n]);\n\n/**\n * Directories to skip when doing a root-level content scan.\n * These are either already handled by dedicated scans or should never be indexed.\n */\nexport const ROOT_SCAN_SKIP_DIRS = new Set([\n \"memory\",\n \"Notes\",\n \".claude\",\n \".DS_Store\",\n // Everything in ALWAYS_SKIP_DIRS is also excluded at root level\n ...ALWAYS_SKIP_DIRS,\n]);\n\n/**\n * Additional directories to skip at the content-scan level (first level below root).\n * These are common macOS/Linux home-directory or repo noise directories that are\n * never meaningful as project content.\n */\nexport const CONTENT_SCAN_SKIP_DIRS = new Set([\n // macOS home directory standard folders\n \"Library\",\n \"Applications\",\n \"Music\",\n \"Movies\",\n \"Pictures\",\n \"Desktop\",\n \"Downloads\",\n \"Public\",\n // Common dev noise\n \"coverage\",\n // Everything in ALWAYS_SKIP_DIRS is also excluded at this level\n ...ALWAYS_SKIP_DIRS,\n]);\n\n// ---------------------------------------------------------------------------\n// Directory walkers\n// ---------------------------------------------------------------------------\n\n/**\n * Safety cap: maximum number of .md files collected per project scan.\n * Prevents runaway scans on huge root paths (e.g. home directory).\n * Projects with more files than this are scanned up to the cap only.\n */\nconst MAX_FILES_PER_PROJECT = 5_000;\n\n/**\n * Maximum recursion depth for directory walks.\n * Prevents deep traversal of large directory trees (e.g. development repos).\n * Depth 0 = the given directory itself (no recursion).\n * Value 6 allows: root → subdirs → sub-subdirs → ... up to 6 levels.\n * Sufficient for memory/, Notes/, and typical docs structures.\n */\nconst MAX_WALK_DEPTH = 6;\n\n/**\n * Recursively collect all .md files under a directory.\n * Returns absolute paths. Stops early if the accumulated count hits the cap\n * or if the recursion depth exceeds MAX_WALK_DEPTH.\n *\n * @param dir Directory to scan.\n * @param acc Shared accumulator array (mutated in place for early exit).\n * @param cap Maximum number of files to collect (across all recursive calls).\n * @param depth Current recursion depth (0 = the initial call).\n */\nexport function walkMdFiles(\n dir: string,\n acc?: string[],\n cap = MAX_FILES_PER_PROJECT,\n depth = 0,\n): string[] {\n const results = acc ?? [];\n if (!existsSync(dir)) return results;\n if (results.length >= cap) return results;\n if (depth > MAX_WALK_DEPTH) return results;\n\n try {\n for (const entry of readdirSync(dir, { withFileTypes: true })) {\n if (results.length >= cap) break;\n if (entry.isSymbolicLink()) continue;\n // Skip known junk directories at every recursion depth\n if (ALWAYS_SKIP_DIRS.has(entry.name)) continue;\n const full = join(dir, entry.name);\n if (entry.isDirectory()) {\n walkMdFiles(full, results, cap, depth + 1);\n } else if (entry.isFile() && entry.name.endsWith(\".md\")) {\n results.push(full);\n }\n }\n } catch {\n // Unreadable directory — skip\n }\n return results;\n}\n\n/**\n * Recursively collect all .md files under rootPath, excluding directories\n * that are already covered by dedicated scans (memory/, Notes/) and\n * common noise directories (.git, node_modules, etc.).\n *\n * Returns absolute paths for files NOT already handled by the specific scanners.\n * Stops collecting once MAX_FILES_PER_PROJECT is reached.\n */\nexport function walkContentFiles(rootPath: string): string[] {\n if (!existsSync(rootPath)) return [];\n\n const results: string[] = [];\n try {\n for (const entry of readdirSync(rootPath, { withFileTypes: true })) {\n if (results.length >= MAX_FILES_PER_PROJECT) break;\n if (entry.isSymbolicLink()) continue;\n if (ROOT_SCAN_SKIP_DIRS.has(entry.name)) continue;\n if (CONTENT_SCAN_SKIP_DIRS.has(entry.name)) continue;\n\n const full = join(rootPath, entry.name);\n if (entry.isDirectory()) {\n walkMdFiles(full, results, MAX_FILES_PER_PROJECT);\n } else if (entry.isFile() && entry.name.endsWith(\".md\")) {\n // Skip root-level MEMORY.md — handled by the dedicated evergreen scan\n if (entry.name !== \"MEMORY.md\") {\n results.push(full);\n }\n }\n }\n } catch {\n // Unreadable directory — skip\n }\n return results;\n}\n\n// ---------------------------------------------------------------------------\n// Path safety guard\n// ---------------------------------------------------------------------------\n\n/** Paths that must never be indexed — system/temp dirs that can contain backup snapshots. */\nconst BLOCKED_ROOTS = new Set([\"/tmp\", \"/private/tmp\", \"/var\", \"/private/var\"]);\n\n/**\n * Returns true if rootPath should skip the recursive content scan.\n *\n * Skips content scanning for:\n * - The home directory itself or any ancestor (too broad — millions of files)\n * - Git repositories (code repos — index memory/ and Notes/ only, not all .md files)\n *\n * The content scan is still useful for Obsidian vaults, Notes folders, and\n * other doc-centric project trees where ALL markdown files are meaningful.\n *\n * The memory/, Notes/, and claude_notes_dir scans always run regardless.\n */\nexport function isPathTooBroadForContentScan(rootPath: string): boolean {\n const normalized = normalize(rootPath);\n\n // Block system/temp directories outright (CCC snapshots live here)\n if (BLOCKED_ROOTS.has(normalized)) return true;\n for (const blocked of BLOCKED_ROOTS) {\n if (normalized.startsWith(blocked + \"/\")) return true;\n }\n\n const home = homedir();\n\n // Skip the home directory itself or any ancestor of home\n if (home.startsWith(normalized) || normalized === \"/\") {\n return true;\n }\n\n // Skip home directory itself (depth 0)\n if (normalized.startsWith(home)) {\n const rel = normalized.slice(home.length).replace(/^\\//, \"\");\n const depth = rel ? rel.split(\"/\").length : 0;\n if (depth === 0) return true;\n }\n\n // Skip git repositories — content scan is only for doc-centric projects\n // (Obsidian vaults, knowledge bases). Code repos use memory/ and Notes/ only.\n if (existsSync(join(normalized, \".git\"))) {\n return true;\n }\n\n return false;\n}\n\n// ---------------------------------------------------------------------------\n// Session title parser\n// ---------------------------------------------------------------------------\n\nconst SESSION_TITLE_RE = /^(\\d{4})\\s*-\\s*(\\d{4}-\\d{2}-\\d{2})\\s*-\\s*(.+)\\.md$/;\n\n/**\n * Parse a session title from a Notes filename.\n * Format: \"NNNN - YYYY-MM-DD - Descriptive Title.md\"\n * Returns a synthetic chunk text like \"Session #0086 2026-02-23: Pai Daemon Background Service\"\n * or null if the filename doesn't match the expected pattern.\n */\nexport function parseSessionTitleChunk(fileName: string): string | null {\n const m = SESSION_TITLE_RE.exec(fileName);\n if (!m) return null;\n const [, num, date, title] = m;\n return `Session #${num} ${date}: ${title}`;\n}\n\n/** Number of files to process before yielding to the event loop inside indexProject. */\nexport const INDEX_YIELD_EVERY = 10;\n"],"mappings":";;;;;;;;;;;;;;;AAYA,SAAgB,OAAO,SAAyB;AAC9C,QAAO,WAAW,SAAS,CAAC,OAAO,QAAQ,CAAC,OAAO,MAAM;;;AAI3D,MAAa,aAAa;;;;;;;;;;;ACO1B,MAAM,qBAAqB;AAC3B,MAAM,kBAAkB;;;;;AAMxB,SAAgB,eAAe,MAAsB;CACnD,MAAM,YAAY,KAAK,MAAM,MAAM,CAAC,OAAO,QAAQ,CAAC;AACpD,QAAO,KAAK,KAAK,YAAY,IAAI;;;;;;AAqBnC,SAAS,gBACP,OACa;CACb,MAAM,WAAwB,EAAE;CAChC,IAAI,UAAmD,EAAE;AAEzD,MAAK,MAAM,QAAQ,OAAO;AAExB,MADkB,YAAY,KAAK,KAAK,KAAK,IAC5B,QAAQ,SAAS,GAAG;GACnC,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,YAAS,KAAK;IAAE,OAAO;IAAS,QAAQ,eAAe,KAAK;IAAE,CAAC;AAC/D,aAAU,EAAE;;AAEd,UAAQ,KAAK,KAAK;;AAGpB,KAAI,QAAQ,SAAS,GAAG;EACtB,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,WAAS,KAAK;GAAE,OAAO;GAAS,QAAQ,eAAe,KAAK;GAAE,CAAC;;AAGjE,QAAO;;;;;AAMT,SAAS,kBAAkB,OAA+B;CACxD,MAAM,aAA0B,EAAE;CAClC,IAAI,UAAmD,EAAE;AAEzD,MAAK,MAAM,QAAQ,MAAM,MACvB,KAAI,KAAK,KAAK,MAAM,KAAK,MAAM,QAAQ,SAAS,GAAG;EAEjD,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,aAAW,KAAK;GAAE,OAAO,CAAC,GAAG,QAAQ;GAAE,QAAQ,eAAe,KAAK;GAAE,CAAC;AACtE,YAAU,EAAE;OAEZ,SAAQ,KAAK,KAAK;AAItB,KAAI,QAAQ,SAAS,GAAG;EACtB,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,aAAW,KAAK;GAAE,OAAO;GAAS,QAAQ,eAAe,KAAK;GAAE,CAAC;;AAGnE,QAAO,WAAW,SAAS,IAAI,aAAa,CAAC,MAAM;;;;;;AAOrD,SAAS,iBAAiB,OAAkB,WAAgC;CAI1E,MAAM,YAHW,MAAM,MAAM,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,IAAI,CAG9B,MADR,4BACyB;CAE5C,MAAM,SAAsB,EAAE;CAC9B,IAAI,UAAU;CAGd,MAAM,YAAY,MAAM,MAAM,IAAI,UAAU;CAC5C,MAAM,UAAU,MAAM,MAAM,MAAM,MAAM,SAAS,IAAI,UAAU;CAC/D,MAAM,aAAa,UAAU,YAAY;CACzC,MAAM,mBAAmB,KAAK,IAAI,GAAG,KAAK,MAAM,aAAa,KAAK,IAAI,GAAG,UAAU,OAAO,CAAC,CAAC;CAE5F,IAAI,cAAc;CAClB,IAAI,aAAa;CAEjB,MAAM,cAAc;AAClB,MAAI,CAAC,QAAQ,MAAM,CAAE;EACrB,MAAM,YAAY,KAAK,IAAI,aAAa,mBAAmB,GAAG,QAAQ;AACtE,SAAO,KAAK;GACV,OAAO,CAAC;IAAE,MAAM,QAAQ,MAAM;IAAE,QAAQ;IAAY,CAAC;GACrD,QAAQ,eAAe,QAAQ;GAChC,CAAC;AACF,eAAa,YAAY;AACzB,YAAU;;AAGZ,MAAK,MAAM,YAAY,WAAW;AAChC;EACA,MAAM,gBAAgB,UAAU,UAAU,MAAM,WAAW;AAC3D,MAAI,eAAe,cAAc,GAAG,aAAa,SAAS;AACxD,UAAO;AACP,aAAU;QAEV,WAAU;;AAId,QAAO;AAEP,QAAO,OAAO,SAAS,IAAI,SAAS,CAAC,MAAM;;;;;;AAW7C,SAAS,mBACP,QACA,eACyC;AACzC,KAAI,iBAAiB,KAAK,OAAO,WAAW,EAAG,QAAO,EAAE;CAExD,MAAM,YAAY,OAAO,OAAO,SAAS;AACzC,KAAI,CAAC,UAAW,QAAO,EAAE;CAEzB,MAAM,QAAQ,UAAU,KAAK,MAAM,KAAK;CACxC,MAAM,OAAiB,EAAE;CACzB,IAAI,MAAM;AAEV,MAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAAK;EAC1C,MAAM,aAAa,eAAe,MAAM,MAAM,GAAG;AACjD,SAAO;AACP,OAAK,QAAQ,MAAM,MAAM,GAAG;AAC5B,MAAI,OAAO,cAAe;;CAI5B,MAAM,YAAY,UAAU,UAAU,KAAK,SAAS;AACpD,QAAO,KAAK,KAAK,MAAM,SAAS;EAAE;EAAM,QAAQ,KAAK,IAAI,UAAU,WAAW,YAAY,IAAI;EAAE,EAAE;;;;;;;;;;;;;;;;AAqBpG,SAAgB,iBAAiB,SAAyB;AACxD,QAAO,QAAQ,QAAQ,kCAAkC,GAAG;;AAG9D,SAAgB,cAAc,SAAiB,MAA8B;CAC3E,MAAM,YAAY,MAAM,aAAa;CACrC,MAAM,gBAAgB,MAAM,WAAW;AAGvC,WAAU,iBAAiB,QAAQ;AAEnC,KAAI,CAAC,QAAQ,MAAM,CAAE,QAAO,EAAE;CAS9B,MAAM,WAAW,gBAPA,QAAQ,MAAM,KAAK,CAC4B,KAAK,MAAM,SAAS;EAClF;EACA,QAAQ,MAAM;EACf,EAAE,CAGoC;CAGvC,MAAM,cAA2B,EAAE;AACnC,MAAK,MAAM,WAAW,UAAU;AAC9B,MAAI,QAAQ,UAAU,WAAW;AAC/B,eAAY,KAAK,QAAQ;AACzB;;EAGF,MAAM,QAAQ,kBAAkB,QAAQ;AACxC,OAAK,MAAM,QAAQ,OAAO;AACxB,OAAI,KAAK,UAAU,WAAW;AAC5B,gBAAY,KAAK,KAAK;AACtB;;GAGF,MAAM,YAAY,iBAAiB,MAAM,UAAU;AACnD,eAAY,KAAK,GAAG,UAAU;;;CAKlC,MAAM,SAAkB,EAAE;AAE1B,MAAK,MAAM,SAAS,aAAa;AAC/B,MAAI,MAAM,MAAM,WAAW,EAAG;EAO9B,MAAM,OADW,CAAC,GAHG,mBAAmB,QAAQ,cAAc,EAG3B,GAAG,MAAM,MAAM,CAC5B,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK,CAAC,MAAM;AAE1D,MAAI,CAAC,KAAM;EAEX,MAAM,YAAY,MAAM,MAAM,IAAI,UAAU;EAC5C,MAAM,UAAU,MAAM,MAAM,MAAM,MAAM,SAAS,IAAI,UAAU;AAE/D,SAAO,KAAK;GACV;GACA;GACA;GACA,MAAM,OAAO,KAAK;GACnB,CAAC;;AAGJ,QAAO;;;;;;;;;;;;;;;;;;;;ACrPT,SAAgB,WACd,cAC6C;CAE7C,MAAM,IAAI,aAAa,QAAQ,OAAO,IAAI,CAAC,QAAQ,SAAS,GAAG;AAG/D,KAAI,EAAE,WAAW,SAAS,IAAI,MAAM,QAClC,QAAO;CAGT,MAAM,WAAW,SAAS,EAAE;AAG5B,KAAI,aAAa,YACf,QAAO;AAIT,KAAI,0BAA0B,KAAK,SAAS,CAC1C,QAAO;AAIT,QAAO;;;;;;;;;;AAkBT,SAAgB,QACd,WACA,MACA,YACA,WACA,SACQ;AACR,QAAO,WAAW,SAAS,CACxB,OAAO,GAAG,UAAU,GAAG,KAAK,GAAG,WAAW,GAAG,UAAU,GAAG,UAAU,CACpE,OAAO,MAAM;;;;;;;;;;;;AAiBlB,SAAgB,mBAAkC;AAChD,QAAO,IAAI,SAAS,YAAY,WAAW,SAAS,GAAG,CAAC;;;;;;;AAY1D,MAAa,mBAAmB,IAAI,IAAI;CAEtC;CAEA;CACA;CACA;CAEA;CACA;CACA;CACA;CACA;CAEA;CACA;CACA;CAEA;CACA;CAEA;CACA;CACD,CAAC;;;;;AAMF,MAAa,sBAAsB,IAAI,IAAI;CACzC;CACA;CACA;CACA;CAEA,GAAG;CACJ,CAAC;;;;;;AAOF,MAAa,yBAAyB,IAAI,IAAI;CAE5C;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CAEA;CAEA,GAAG;CACJ,CAAC;;;;;;AAWF,MAAM,wBAAwB;;;;;;;;AAS9B,MAAM,iBAAiB;;;;;;;;;;;AAYvB,SAAgB,YACd,KACA,KACA,MAAM,uBACN,QAAQ,GACE;CACV,MAAM,UAAU,OAAO,EAAE;AACzB,KAAI,CAAC,WAAW,IAAI,CAAE,QAAO;AAC7B,KAAI,QAAQ,UAAU,IAAK,QAAO;AAClC,KAAI,QAAQ,eAAgB,QAAO;AAEnC,KAAI;AACF,OAAK,MAAM,SAAS,YAAY,KAAK,EAAE,eAAe,MAAM,CAAC,EAAE;AAC7D,OAAI,QAAQ,UAAU,IAAK;AAC3B,OAAI,MAAM,gBAAgB,CAAE;AAE5B,OAAI,iBAAiB,IAAI,MAAM,KAAK,CAAE;GACtC,MAAM,OAAO,KAAK,KAAK,MAAM,KAAK;AAClC,OAAI,MAAM,aAAa,CACrB,aAAY,MAAM,SAAS,KAAK,QAAQ,EAAE;YACjC,MAAM,QAAQ,IAAI,MAAM,KAAK,SAAS,MAAM,CACrD,SAAQ,KAAK,KAAK;;SAGhB;AAGR,QAAO;;;;;;;;;;AAWT,SAAgB,iBAAiB,UAA4B;AAC3D,KAAI,CAAC,WAAW,SAAS,CAAE,QAAO,EAAE;CAEpC,MAAM,UAAoB,EAAE;AAC5B,KAAI;AACF,OAAK,MAAM,SAAS,YAAY,UAAU,EAAE,eAAe,MAAM,CAAC,EAAE;AAClE,OAAI,QAAQ,UAAU,sBAAuB;AAC7C,OAAI,MAAM,gBAAgB,CAAE;AAC5B,OAAI,oBAAoB,IAAI,MAAM,KAAK,CAAE;AACzC,OAAI,uBAAuB,IAAI,MAAM,KAAK,CAAE;GAE5C,MAAM,OAAO,KAAK,UAAU,MAAM,KAAK;AACvC,OAAI,MAAM,aAAa,CACrB,aAAY,MAAM,SAAS,sBAAsB;YACxC,MAAM,QAAQ,IAAI,MAAM,KAAK,SAAS,MAAM,EAErD;QAAI,MAAM,SAAS,YACjB,SAAQ,KAAK,KAAK;;;SAIlB;AAGR,QAAO;;;AAQT,MAAM,gBAAgB,IAAI,IAAI;CAAC;CAAQ;CAAgB;CAAQ;CAAe,CAAC;;;;;;;;;;;;;AAc/E,SAAgB,6BAA6B,UAA2B;CACtE,MAAM,aAAa,UAAU,SAAS;AAGtC,KAAI,cAAc,IAAI,WAAW,CAAE,QAAO;AAC1C,MAAK,MAAM,WAAW,cACpB,KAAI,WAAW,WAAW,UAAU,IAAI,CAAE,QAAO;CAGnD,MAAM,OAAO,SAAS;AAGtB,KAAI,KAAK,WAAW,WAAW,IAAI,eAAe,IAChD,QAAO;AAIT,KAAI,WAAW,WAAW,KAAK,EAAE;EAC/B,MAAM,MAAM,WAAW,MAAM,KAAK,OAAO,CAAC,QAAQ,OAAO,GAAG;AAE5D,OADc,MAAM,IAAI,MAAM,IAAI,CAAC,SAAS,OAC9B,EAAG,QAAO;;AAK1B,KAAI,WAAW,KAAK,YAAY,OAAO,CAAC,CACtC,QAAO;AAGT,QAAO;;AAOT,MAAM,mBAAmB;;;;;;;AAQzB,SAAgB,uBAAuB,UAAiC;CACtE,MAAM,IAAI,iBAAiB,KAAK,SAAS;AACzC,KAAI,CAAC,EAAG,QAAO;CACf,MAAM,GAAG,KAAK,MAAM,SAAS;AAC7B,QAAO,YAAY,IAAI,GAAG,KAAK,IAAI;;;AAIrC,MAAa,oBAAoB"}
1
+ {"version":3,"file":"helpers-IjZkXBhj.mjs","names":[],"sources":["../src/utils/hash.ts","../src/memory/chunker.ts","../src/memory/indexer/helpers.ts"],"sourcesContent":["/**\n * Shared hashing utilities. Centralises all SHA-256 usage so every module\n * obtains digests through the same function rather than inlining createHash.\n */\n\nimport { createHash } from \"node:crypto\";\n\n/**\n * Compute a SHA-256 hex digest of the given string.\n * Aliased as sha256File for compatibility with existing call-sites that use\n * that name to hash file contents.\n */\nexport function sha256(content: string): string {\n return createHash(\"sha256\").update(content).digest(\"hex\");\n}\n\n/** Alias kept for backwards compatibility with memory/indexer call-sites. */\nexport const sha256File = sha256;\n","/**\n * Markdown text chunker for the PAI memory engine.\n *\n * Splits markdown files into overlapping text segments suitable for BM25\n * full-text indexing. Respects heading boundaries where possible, falling\n * back to paragraph and sentence splitting when sections are large.\n */\n\nimport { sha256 } from \"../utils/hash.js\";\n\nexport interface Chunk {\n text: string;\n startLine: number; // 1-indexed\n endLine: number; // 1-indexed, inclusive\n hash: string; // SHA-256 of text\n}\n\nexport interface ChunkOptions {\n /** Approximate maximum tokens per chunk. Default 400. */\n maxTokens?: number;\n /** Overlap in tokens from the previous chunk. Default 80. */\n overlap?: number;\n}\n\nconst DEFAULT_MAX_TOKENS = 400;\nconst DEFAULT_OVERLAP = 80;\n\n/**\n * Approximate token count using a words * 1.3 heuristic.\n * Matches the OpenClaw estimate approach.\n */\nexport function estimateTokens(text: string): number {\n const wordCount = text.split(/\\s+/).filter(Boolean).length;\n return Math.ceil(wordCount * 1.3);\n}\n\n// sha256 imported from utils/hash.ts\n\n// ---------------------------------------------------------------------------\n// Internal section / paragraph / sentence splitters\n// ---------------------------------------------------------------------------\n\n/**\n * A contiguous block of lines associated with an approximate token count.\n */\ninterface LineBlock {\n lines: Array<{ text: string; lineNo: number }>;\n tokens: number;\n}\n\n/**\n * Split content into sections delimited by ## or ### headings.\n * Each section starts at its heading line (or at line 1 for a preamble).\n */\nfunction splitBySections(\n lines: Array<{ text: string; lineNo: number }>,\n): LineBlock[] {\n const sections: LineBlock[] = [];\n let current: Array<{ text: string; lineNo: number }> = [];\n\n for (const line of lines) {\n const isHeading = /^#{1,3}\\s/.test(line.text);\n if (isHeading && current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n sections.push({ lines: current, tokens: estimateTokens(text) });\n current = [];\n }\n current.push(line);\n }\n\n if (current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n sections.push({ lines: current, tokens: estimateTokens(text) });\n }\n\n return sections;\n}\n\n/**\n * Split a LineBlock by double-newline paragraph boundaries.\n */\nfunction splitByParagraphs(block: LineBlock): LineBlock[] {\n const paragraphs: LineBlock[] = [];\n let current: Array<{ text: string; lineNo: number }> = [];\n\n for (const line of block.lines) {\n if (line.text.trim() === \"\" && current.length > 0) {\n // Empty line — potential paragraph boundary\n const text = current.map((l) => l.text).join(\"\\n\");\n paragraphs.push({ lines: [...current], tokens: estimateTokens(text) });\n current = [];\n } else {\n current.push(line);\n }\n }\n\n if (current.length > 0) {\n const text = current.map((l) => l.text).join(\"\\n\");\n paragraphs.push({ lines: current, tokens: estimateTokens(text) });\n }\n\n return paragraphs.length > 0 ? paragraphs : [block];\n}\n\n/**\n * Split a LineBlock by sentence boundaries (. ! ?) when even paragraphs are\n * too large. Works character-by-character within joined lines.\n */\nfunction splitBySentences(block: LineBlock, maxTokens: number): LineBlock[] {\n const fullText = block.lines.map((l) => l.text).join(\" \");\n // Very rough sentence split — split on '. ', '! ', '? ' followed by uppercase\n const sentenceRe = /(?<=[.!?])\\s+(?=[A-Z\"'])/g;\n const sentences = fullText.split(sentenceRe);\n\n const result: LineBlock[] = [];\n let accText = \"\";\n // We can't recover exact line numbers inside a single oversized paragraph,\n // so we approximate using the block's start/end lines distributed evenly.\n const startLine = block.lines[0]?.lineNo ?? 1;\n const endLine = block.lines[block.lines.length - 1]?.lineNo ?? startLine;\n const totalLines = endLine - startLine + 1;\n const linesPerSentence = Math.max(1, Math.floor(totalLines / Math.max(1, sentences.length)));\n\n let sentenceIdx = 0;\n let approxLine = startLine;\n\n const flush = () => {\n if (!accText.trim()) return;\n const endApprox = Math.min(approxLine + linesPerSentence - 1, endLine);\n result.push({\n lines: [{ text: accText.trim(), lineNo: approxLine }],\n tokens: estimateTokens(accText),\n });\n approxLine = endApprox + 1;\n accText = \"\";\n };\n\n for (const sentence of sentences) {\n sentenceIdx++;\n const candidateText = accText ? accText + \" \" + sentence : sentence;\n if (estimateTokens(candidateText) > maxTokens && accText) {\n flush();\n accText = sentence;\n } else {\n accText = candidateText;\n }\n }\n void sentenceIdx; // used only for iteration count\n flush();\n\n return result.length > 0 ? result : [block];\n}\n\n// ---------------------------------------------------------------------------\n// Overlap helper\n// ---------------------------------------------------------------------------\n\n/**\n * Extract the last `overlapTokens` worth of text from a list of previously\n * emitted chunks to prepend to the next chunk.\n */\nfunction buildOverlapPrefix(\n chunks: Chunk[],\n overlapTokens: number,\n): Array<{ text: string; lineNo: number }> {\n if (overlapTokens <= 0 || chunks.length === 0) return [];\n\n const lastChunk = chunks[chunks.length - 1];\n if (!lastChunk) return [];\n\n const lines = lastChunk.text.split(\"\\n\");\n const kept: string[] = [];\n let acc = 0;\n\n for (let i = lines.length - 1; i >= 0; i--) {\n const lineTokens = estimateTokens(lines[i] ?? \"\");\n acc += lineTokens;\n kept.unshift(lines[i] ?? \"\");\n if (acc >= overlapTokens) break;\n }\n\n // Distribute overlap lines across the lastChunk's line range\n const startLine = lastChunk.endLine - kept.length + 1;\n return kept.map((text, idx) => ({ text, lineNo: Math.max(lastChunk.startLine, startLine + idx) }));\n}\n\n// ---------------------------------------------------------------------------\n// Public API\n// ---------------------------------------------------------------------------\n\n/**\n * Chunk a markdown file into overlapping segments for BM25 indexing.\n *\n * Strategy:\n * 1. Split by headings (##, ###) as natural boundaries.\n * 2. If a section exceeds maxTokens, split by paragraphs.\n * 3. If a paragraph still exceeds maxTokens, split by sentences.\n * 4. Apply overlap: each chunk includes the last `overlap` tokens from the\n * previous chunk.\n */\n/**\n * Strip `<private>...</private>` blocks from content before indexing.\n * Content within these tags is excluded from memory — never stored or searched.\n */\nexport function stripPrivateTags(content: string): string {\n return content.replace(/<private>[\\s\\S]*?<\\/private>/gi, \"\");\n}\n\nexport function chunkMarkdown(content: string, opts?: ChunkOptions): Chunk[] {\n const maxTokens = opts?.maxTokens ?? DEFAULT_MAX_TOKENS;\n const overlapTokens = opts?.overlap ?? DEFAULT_OVERLAP;\n\n // Strip private content before indexing\n content = stripPrivateTags(content);\n\n if (!content.trim()) return [];\n\n const rawLines = content.split(\"\\n\");\n const lines: Array<{ text: string; lineNo: number }> = rawLines.map((text, idx) => ({\n text,\n lineNo: idx + 1, // 1-indexed\n }));\n\n // Step 1: section split\n const sections = splitBySections(lines);\n\n // Step 2 & 3: further split oversized sections\n const finalBlocks: LineBlock[] = [];\n for (const section of sections) {\n if (section.tokens <= maxTokens) {\n finalBlocks.push(section);\n continue;\n }\n // Too big — split by paragraphs\n const paras = splitByParagraphs(section);\n for (const para of paras) {\n if (para.tokens <= maxTokens) {\n finalBlocks.push(para);\n continue;\n }\n // Still too big — split by sentences\n const sentences = splitBySentences(para, maxTokens);\n finalBlocks.push(...sentences);\n }\n }\n\n // Step 4: build final chunks with overlap\n const chunks: Chunk[] = [];\n\n for (const block of finalBlocks) {\n if (block.lines.length === 0) continue;\n\n // Build overlap prefix from previous chunks\n const overlapLines = buildOverlapPrefix(chunks, overlapTokens);\n\n // Combine overlap + block lines\n const allLines = [...overlapLines, ...block.lines];\n const text = allLines.map((l) => l.text).join(\"\\n\").trim();\n\n if (!text) continue;\n\n const startLine = block.lines[0]?.lineNo ?? 1;\n const endLine = block.lines[block.lines.length - 1]?.lineNo ?? startLine;\n\n chunks.push({\n text,\n startLine,\n endLine,\n hash: sha256(text),\n });\n }\n\n return chunks;\n}\n","/**\n * Shared helpers for the PAI memory indexers.\n *\n * Contains utilities used by both the sync (SQLite) and async (StorageBackend)\n * indexer paths: hashing, chunk ID generation, directory walking, and path guards.\n */\n\nimport { readdirSync, existsSync } from \"node:fs\";\nimport { createHash } from \"node:crypto\";\nimport { sha256File } from \"../../utils/hash.js\";\nimport { join, normalize } from \"node:path\";\nimport { homedir } from \"node:os\";\nimport { basename } from \"node:path\";\n\n// ---------------------------------------------------------------------------\n// Tier detection\n// ---------------------------------------------------------------------------\n\n/**\n * Classify a relative file path into one of the four memory tiers.\n *\n * Rules (in priority order):\n * - MEMORY.md anywhere in memory/ → 'evergreen'\n * - YYYY-MM-DD.md in memory/ → 'daily'\n * - anything else in memory/ → 'topic'\n * - anything in Notes/ → 'session'\n */\nexport function detectTier(\n relativePath: string,\n): \"evergreen\" | \"daily\" | \"topic\" | \"session\" {\n // Normalise to forward slashes and strip leading ./\n const p = relativePath.replace(/\\\\/g, \"/\").replace(/^\\.\\//, \"\");\n\n // Notes directory → session tier\n if (p.startsWith(\"Notes/\") || p === \"Notes\") {\n return \"session\";\n }\n\n const fileName = basename(p);\n\n // MEMORY.md (case-sensitive match) → evergreen\n if (fileName === \"MEMORY.md\") {\n return \"evergreen\";\n }\n\n // YYYY-MM-DD.md → daily\n if (/^\\d{4}-\\d{2}-\\d{2}\\.md$/.test(fileName)) {\n return \"daily\";\n }\n\n // Default for memory/ files\n return \"topic\";\n}\n\n// ---------------------------------------------------------------------------\n// Hashing and chunk ID generation\n// ---------------------------------------------------------------------------\n\n// sha256File imported from ../../utils/hash.js\nexport { sha256File } from \"../../utils/hash.js\";\n\n/**\n * Generate a deterministic chunk ID from its coordinates.\n * Format: sha256(\"projectId:path:chunkIndex:startLine:endLine\")\n *\n * The chunkIndex (0-based position within the file) is included so that\n * chunks with approximated line numbers (e.g. from splitBySentences) never\n * produce colliding IDs even when multiple chunks share the same startLine/endLine.\n */\nexport function chunkId(\n projectId: number,\n path: string,\n chunkIndex: number,\n startLine: number,\n endLine: number,\n): string {\n return createHash(\"sha256\")\n .update(`${projectId}:${path}:${chunkIndex}:${startLine}:${endLine}`)\n .digest(\"hex\");\n}\n\n// ---------------------------------------------------------------------------\n// Event loop yield\n// ---------------------------------------------------------------------------\n\n/**\n * Yield to the Node.js event loop so that IPC server can process requests\n * during long index runs.\n *\n * Uses setTimeout(10ms) rather than setImmediate — the 10ms pause gives the\n * event loop enough time to accept and process incoming IPC connections\n * (socket data, new connections, etc.). Without this, synchronous ONNX\n * inference blocks IPC for the full duration of each embedding (~50-100ms\n * per chunk).\n */\nexport function yieldToEventLoop(): Promise<void> {\n return new Promise((resolve) => setTimeout(resolve, 10));\n}\n\n// ---------------------------------------------------------------------------\n// Directory skip sets\n// ---------------------------------------------------------------------------\n\n/**\n * Directories to ALWAYS skip, at any depth, during any directory walk.\n * These are build artifacts, dependency trees, and VCS internals that\n * should never be indexed regardless of where they appear in the tree.\n */\nexport const ALWAYS_SKIP_DIRS = new Set([\n // Version control\n \".git\",\n // Dependency directories (any language)\n \"node_modules\",\n \"vendor\",\n \"Pods\", // CocoaPods (iOS/macOS)\n // Build / compile output\n \"dist\",\n \"build\",\n \"out\",\n \"DerivedData\", // Xcode\n \".next\", // Next.js\n // Python virtual environments and caches\n \".venv\",\n \"venv\",\n \"__pycache__\",\n // General caches\n \".cache\",\n \".bun\",\n // Backup snapshots (Carbon Copy Cloner, Time Machine, etc.)\n \"snaps\",\n \".Trashes\",\n]);\n\n/**\n * Directories to skip when doing a root-level content scan.\n * These are either already handled by dedicated scans or should never be indexed.\n */\nexport const ROOT_SCAN_SKIP_DIRS = new Set([\n \"memory\",\n \"Notes\",\n \".claude\",\n \".DS_Store\",\n // Everything in ALWAYS_SKIP_DIRS is also excluded at root level\n ...ALWAYS_SKIP_DIRS,\n]);\n\n/**\n * Additional directories to skip at the content-scan level (first level below root).\n * These are common macOS/Linux home-directory or repo noise directories that are\n * never meaningful as project content.\n */\nexport const CONTENT_SCAN_SKIP_DIRS = new Set([\n // macOS home directory standard folders\n \"Library\",\n \"Applications\",\n \"Music\",\n \"Movies\",\n \"Pictures\",\n \"Desktop\",\n \"Downloads\",\n \"Public\",\n // Common dev noise\n \"coverage\",\n // Everything in ALWAYS_SKIP_DIRS is also excluded at this level\n ...ALWAYS_SKIP_DIRS,\n]);\n\n// ---------------------------------------------------------------------------\n// Directory walkers\n// ---------------------------------------------------------------------------\n\n/**\n * Safety cap: maximum number of .md files collected per project scan.\n * Prevents runaway scans on huge root paths (e.g. home directory).\n * Projects with more files than this are scanned up to the cap only.\n */\nconst MAX_FILES_PER_PROJECT = 5_000;\n\n/**\n * Maximum recursion depth for directory walks.\n * Prevents deep traversal of large directory trees (e.g. development repos).\n * Depth 0 = the given directory itself (no recursion).\n * Value 6 allows: root → subdirs → sub-subdirs → ... up to 6 levels.\n * Sufficient for memory/, Notes/, and typical docs structures.\n */\nconst MAX_WALK_DEPTH = 6;\n\n/**\n * Recursively collect all .md files under a directory.\n * Returns absolute paths. Stops early if the accumulated count hits the cap\n * or if the recursion depth exceeds MAX_WALK_DEPTH.\n *\n * @param dir Directory to scan.\n * @param acc Shared accumulator array (mutated in place for early exit).\n * @param cap Maximum number of files to collect (across all recursive calls).\n * @param depth Current recursion depth (0 = the initial call).\n */\nexport function walkMdFiles(\n dir: string,\n acc?: string[],\n cap = MAX_FILES_PER_PROJECT,\n depth = 0,\n): string[] {\n const results = acc ?? [];\n if (!existsSync(dir)) return results;\n if (results.length >= cap) return results;\n if (depth > MAX_WALK_DEPTH) return results;\n\n try {\n for (const entry of readdirSync(dir, { withFileTypes: true })) {\n if (results.length >= cap) break;\n if (entry.isSymbolicLink()) continue;\n // Skip known junk directories at every recursion depth\n if (ALWAYS_SKIP_DIRS.has(entry.name)) continue;\n const full = join(dir, entry.name);\n if (entry.isDirectory()) {\n walkMdFiles(full, results, cap, depth + 1);\n } else if (entry.isFile() && entry.name.endsWith(\".md\")) {\n results.push(full);\n }\n }\n } catch {\n // Unreadable directory — skip\n }\n return results;\n}\n\n/**\n * Recursively collect all .md files under rootPath, excluding directories\n * that are already covered by dedicated scans (memory/, Notes/) and\n * common noise directories (.git, node_modules, etc.).\n *\n * Returns absolute paths for files NOT already handled by the specific scanners.\n * Stops collecting once MAX_FILES_PER_PROJECT is reached.\n */\nexport function walkContentFiles(rootPath: string): string[] {\n if (!existsSync(rootPath)) return [];\n\n const results: string[] = [];\n try {\n for (const entry of readdirSync(rootPath, { withFileTypes: true })) {\n if (results.length >= MAX_FILES_PER_PROJECT) break;\n if (entry.isSymbolicLink()) continue;\n if (ROOT_SCAN_SKIP_DIRS.has(entry.name)) continue;\n if (CONTENT_SCAN_SKIP_DIRS.has(entry.name)) continue;\n\n const full = join(rootPath, entry.name);\n if (entry.isDirectory()) {\n walkMdFiles(full, results, MAX_FILES_PER_PROJECT);\n } else if (entry.isFile() && entry.name.endsWith(\".md\")) {\n // Skip root-level MEMORY.md — handled by the dedicated evergreen scan\n if (entry.name !== \"MEMORY.md\") {\n results.push(full);\n }\n }\n }\n } catch {\n // Unreadable directory — skip\n }\n return results;\n}\n\n// ---------------------------------------------------------------------------\n// Path safety guard\n// ---------------------------------------------------------------------------\n\n/** Paths that must never be indexed — system/temp dirs that can contain backup snapshots. */\nconst BLOCKED_ROOTS = new Set([\"/tmp\", \"/private/tmp\", \"/var\", \"/private/var\"]);\n\n/**\n * Returns true if rootPath should skip the recursive content scan.\n *\n * Skips content scanning for:\n * - The home directory itself or any ancestor (too broad — millions of files)\n * - Git repositories (code repos — index memory/ and Notes/ only, not all .md files)\n *\n * The content scan is still useful for Obsidian vaults, Notes folders, and\n * other doc-centric project trees where ALL markdown files are meaningful.\n *\n * The memory/, Notes/, and claude_notes_dir scans always run regardless.\n */\nexport function isPathTooBroadForContentScan(rootPath: string): boolean {\n const normalized = normalize(rootPath);\n\n // Block system/temp directories outright (CCC snapshots live here)\n if (BLOCKED_ROOTS.has(normalized)) return true;\n for (const blocked of BLOCKED_ROOTS) {\n if (normalized.startsWith(blocked + \"/\")) return true;\n }\n\n const home = homedir();\n\n // Skip the home directory itself or any ancestor of home\n if (home.startsWith(normalized) || normalized === \"/\") {\n return true;\n }\n\n // Skip home directory itself (depth 0)\n if (normalized.startsWith(home)) {\n const rel = normalized.slice(home.length).replace(/^\\//, \"\");\n const depth = rel ? rel.split(\"/\").length : 0;\n if (depth === 0) return true;\n }\n\n // Skip git repositories — content scan is only for doc-centric projects\n // (Obsidian vaults, knowledge bases). Code repos use memory/ and Notes/ only.\n if (existsSync(join(normalized, \".git\"))) {\n return true;\n }\n\n return false;\n}\n\n// ---------------------------------------------------------------------------\n// Session title parser\n// ---------------------------------------------------------------------------\n\nconst SESSION_TITLE_RE = /^(\\d{4})\\s*-\\s*(\\d{4}-\\d{2}-\\d{2})\\s*-\\s*(.+)\\.md$/;\n\n/**\n * Parse a session title from a Notes filename.\n * Format: \"NNNN - YYYY-MM-DD - Descriptive Title.md\"\n * Returns a synthetic chunk text like \"Session #0086 2026-02-23: Pai Daemon Background Service\"\n * or null if the filename doesn't match the expected pattern.\n */\nexport function parseSessionTitleChunk(fileName: string): string | null {\n const m = SESSION_TITLE_RE.exec(fileName);\n if (!m) return null;\n const [, num, date, title] = m;\n return `Session #${num} ${date}: ${title}`;\n}\n\n/** Number of files to process before yielding to the event loop inside indexProject. */\nexport const INDEX_YIELD_EVERY = 10;\n"],"mappings":";;;;;;;;;;;;;;;AAYA,SAAgB,OAAO,SAAyB;AAC9C,QAAO,WAAW,SAAS,CAAC,OAAO,QAAQ,CAAC,OAAO,MAAM;;;AAI3D,MAAa,aAAa;;;;;;;;;;;ACO1B,MAAM,qBAAqB;AAC3B,MAAM,kBAAkB;;;;;AAMxB,SAAgB,eAAe,MAAsB;CACnD,MAAM,YAAY,KAAK,MAAM,MAAM,CAAC,OAAO,QAAQ,CAAC;AACpD,QAAO,KAAK,KAAK,YAAY,IAAI;;;;;;AAqBnC,SAAS,gBACP,OACa;CACb,MAAM,WAAwB,EAAE;CAChC,IAAI,UAAmD,EAAE;AAEzD,MAAK,MAAM,QAAQ,OAAO;AAExB,MADkB,YAAY,KAAK,KAAK,KAAK,IAC5B,QAAQ,SAAS,GAAG;GACnC,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,YAAS,KAAK;IAAE,OAAO;IAAS,QAAQ,eAAe,KAAK;IAAE,CAAC;AAC/D,aAAU,EAAE;;AAEd,UAAQ,KAAK,KAAK;;AAGpB,KAAI,QAAQ,SAAS,GAAG;EACtB,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,WAAS,KAAK;GAAE,OAAO;GAAS,QAAQ,eAAe,KAAK;GAAE,CAAC;;AAGjE,QAAO;;;;;AAMT,SAAS,kBAAkB,OAA+B;CACxD,MAAM,aAA0B,EAAE;CAClC,IAAI,UAAmD,EAAE;AAEzD,MAAK,MAAM,QAAQ,MAAM,MACvB,KAAI,KAAK,KAAK,MAAM,KAAK,MAAM,QAAQ,SAAS,GAAG;EAEjD,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,aAAW,KAAK;GAAE,OAAO,CAAC,GAAG,QAAQ;GAAE,QAAQ,eAAe,KAAK;GAAE,CAAC;AACtE,YAAU,EAAE;OAEZ,SAAQ,KAAK,KAAK;AAItB,KAAI,QAAQ,SAAS,GAAG;EACtB,MAAM,OAAO,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK;AAClD,aAAW,KAAK;GAAE,OAAO;GAAS,QAAQ,eAAe,KAAK;GAAE,CAAC;;AAGnE,QAAO,WAAW,SAAS,IAAI,aAAa,CAAC,MAAM;;;;;;AAOrD,SAAS,iBAAiB,OAAkB,WAAgC;CAI1E,MAAM,YAHW,MAAM,MAAM,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,IAAI,CAG9B,MADR,4BACyB;CAE5C,MAAM,SAAsB,EAAE;CAC9B,IAAI,UAAU;CAGd,MAAM,YAAY,MAAM,MAAM,IAAI,UAAU;CAC5C,MAAM,UAAU,MAAM,MAAM,MAAM,MAAM,SAAS,IAAI,UAAU;CAC/D,MAAM,aAAa,UAAU,YAAY;CACzC,MAAM,mBAAmB,KAAK,IAAI,GAAG,KAAK,MAAM,aAAa,KAAK,IAAI,GAAG,UAAU,OAAO,CAAC,CAAC;CAE5F,IAAI,cAAc;CAClB,IAAI,aAAa;CAEjB,MAAM,cAAc;AAClB,MAAI,CAAC,QAAQ,MAAM,CAAE;EACrB,MAAM,YAAY,KAAK,IAAI,aAAa,mBAAmB,GAAG,QAAQ;AACtE,SAAO,KAAK;GACV,OAAO,CAAC;IAAE,MAAM,QAAQ,MAAM;IAAE,QAAQ;IAAY,CAAC;GACrD,QAAQ,eAAe,QAAQ;GAChC,CAAC;AACF,eAAa,YAAY;AACzB,YAAU;;AAGZ,MAAK,MAAM,YAAY,WAAW;AAChC;EACA,MAAM,gBAAgB,UAAU,UAAU,MAAM,WAAW;AAC3D,MAAI,eAAe,cAAc,GAAG,aAAa,SAAS;AACxD,UAAO;AACP,aAAU;QAEV,WAAU;;AAId,QAAO;AAEP,QAAO,OAAO,SAAS,IAAI,SAAS,CAAC,MAAM;;;;;;AAW7C,SAAS,mBACP,QACA,eACyC;AACzC,KAAI,iBAAiB,KAAK,OAAO,WAAW,EAAG,QAAO,EAAE;CAExD,MAAM,YAAY,OAAO,OAAO,SAAS;AACzC,KAAI,CAAC,UAAW,QAAO,EAAE;CAEzB,MAAM,QAAQ,UAAU,KAAK,MAAM,KAAK;CACxC,MAAM,OAAiB,EAAE;CACzB,IAAI,MAAM;AAEV,MAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAAK;EAC1C,MAAM,aAAa,eAAe,MAAM,MAAM,GAAG;AACjD,SAAO;AACP,OAAK,QAAQ,MAAM,MAAM,GAAG;AAC5B,MAAI,OAAO,cAAe;;CAI5B,MAAM,YAAY,UAAU,UAAU,KAAK,SAAS;AACpD,QAAO,KAAK,KAAK,MAAM,SAAS;EAAE;EAAM,QAAQ,KAAK,IAAI,UAAU,WAAW,YAAY,IAAI;EAAE,EAAE;;;;;;;;;;;;;;;;AAqBpG,SAAgB,iBAAiB,SAAyB;AACxD,QAAO,QAAQ,QAAQ,kCAAkC,GAAG;;AAG9D,SAAgB,cAAc,SAAiB,MAA8B;CAC3E,MAAM,YAAY,MAAM,aAAa;CACrC,MAAM,gBAAgB,MAAM,WAAW;AAGvC,WAAU,iBAAiB,QAAQ;AAEnC,KAAI,CAAC,QAAQ,MAAM,CAAE,QAAO,EAAE;CAS9B,MAAM,WAAW,gBAPA,QAAQ,MAAM,KAAK,CAC4B,KAAK,MAAM,SAAS;EAClF;EACA,QAAQ,MAAM;EACf,EAAE,CAGoC;CAGvC,MAAM,cAA2B,EAAE;AACnC,MAAK,MAAM,WAAW,UAAU;AAC9B,MAAI,QAAQ,UAAU,WAAW;AAC/B,eAAY,KAAK,QAAQ;AACzB;;EAGF,MAAM,QAAQ,kBAAkB,QAAQ;AACxC,OAAK,MAAM,QAAQ,OAAO;AACxB,OAAI,KAAK,UAAU,WAAW;AAC5B,gBAAY,KAAK,KAAK;AACtB;;GAGF,MAAM,YAAY,iBAAiB,MAAM,UAAU;AACnD,eAAY,KAAK,GAAG,UAAU;;;CAKlC,MAAM,SAAkB,EAAE;AAE1B,MAAK,MAAM,SAAS,aAAa;AAC/B,MAAI,MAAM,MAAM,WAAW,EAAG;EAO9B,MAAM,OADW,CAAC,GAHG,mBAAmB,QAAQ,cAAc,EAG3B,GAAG,MAAM,MAAM,CAC5B,KAAK,MAAM,EAAE,KAAK,CAAC,KAAK,KAAK,CAAC,MAAM;AAE1D,MAAI,CAAC,KAAM;EAEX,MAAM,YAAY,MAAM,MAAM,IAAI,UAAU;EAC5C,MAAM,UAAU,MAAM,MAAM,MAAM,MAAM,SAAS,IAAI,UAAU;AAE/D,SAAO,KAAK;GACV;GACA;GACA;GACA,MAAM,OAAO,KAAK;GACnB,CAAC;;AAGJ,QAAO;;;;;;;;;;;;;;;;;;;;ACrPT,SAAgB,WACd,cAC6C;CAE7C,MAAM,IAAI,aAAa,QAAQ,OAAO,IAAI,CAAC,QAAQ,SAAS,GAAG;AAG/D,KAAI,EAAE,WAAW,SAAS,IAAI,MAAM,QAClC,QAAO;CAGT,MAAM,WAAW,SAAS,EAAE;AAG5B,KAAI,aAAa,YACf,QAAO;AAIT,KAAI,0BAA0B,KAAK,SAAS,CAC1C,QAAO;AAIT,QAAO;;;;;;;;;;AAkBT,SAAgB,QACd,WACA,MACA,YACA,WACA,SACQ;AACR,QAAO,WAAW,SAAS,CACxB,OAAO,GAAG,UAAU,GAAG,KAAK,GAAG,WAAW,GAAG,UAAU,GAAG,UAAU,CACpE,OAAO,MAAM;;;;;;;;;;;;AAiBlB,SAAgB,mBAAkC;AAChD,QAAO,IAAI,SAAS,YAAY,WAAW,SAAS,GAAG,CAAC;;;;;;;AAY1D,MAAa,mBAAmB,IAAI,IAAI;CAEtC;CAEA;CACA;CACA;CAEA;CACA;CACA;CACA;CACA;CAEA;CACA;CACA;CAEA;CACA;CAEA;CACA;CACD,CAAC;;;;;AAMF,MAAa,sBAAsB,IAAI,IAAI;CACzC;CACA;CACA;CACA;CAEA,GAAG;CACJ,CAAC;;;;;;AAOF,MAAa,yBAAyB,IAAI,IAAI;CAE5C;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CAEA;CAEA,GAAG;CACJ,CAAC;;;;;;AAWF,MAAM,wBAAwB;;;;;;;;AAS9B,MAAM,iBAAiB;;;;;;;;;;;AAYvB,SAAgB,YACd,KACA,KACA,MAAM,uBACN,QAAQ,GACE;CACV,MAAM,UAAU,OAAO,EAAE;AACzB,KAAI,CAAC,WAAW,IAAI,CAAE,QAAO;AAC7B,KAAI,QAAQ,UAAU,IAAK,QAAO;AAClC,KAAI,QAAQ,eAAgB,QAAO;AAEnC,KAAI;AACF,OAAK,MAAM,SAAS,YAAY,KAAK,EAAE,eAAe,MAAM,CAAC,EAAE;AAC7D,OAAI,QAAQ,UAAU,IAAK;AAC3B,OAAI,MAAM,gBAAgB,CAAE;AAE5B,OAAI,iBAAiB,IAAI,MAAM,KAAK,CAAE;GACtC,MAAM,OAAO,KAAK,KAAK,MAAM,KAAK;AAClC,OAAI,MAAM,aAAa,CACrB,aAAY,MAAM,SAAS,KAAK,QAAQ,EAAE;YACjC,MAAM,QAAQ,IAAI,MAAM,KAAK,SAAS,MAAM,CACrD,SAAQ,KAAK,KAAK;;SAGhB;AAGR,QAAO;;;;;;;;;;AAWT,SAAgB,iBAAiB,UAA4B;AAC3D,KAAI,CAAC,WAAW,SAAS,CAAE,QAAO,EAAE;CAEpC,MAAM,UAAoB,EAAE;AAC5B,KAAI;AACF,OAAK,MAAM,SAAS,YAAY,UAAU,EAAE,eAAe,MAAM,CAAC,EAAE;AAClE,OAAI,QAAQ,UAAU,sBAAuB;AAC7C,OAAI,MAAM,gBAAgB,CAAE;AAC5B,OAAI,oBAAoB,IAAI,MAAM,KAAK,CAAE;AACzC,OAAI,uBAAuB,IAAI,MAAM,KAAK,CAAE;GAE5C,MAAM,OAAO,KAAK,UAAU,MAAM,KAAK;AACvC,OAAI,MAAM,aAAa,CACrB,aAAY,MAAM,SAAS,sBAAsB;YACxC,MAAM,QAAQ,IAAI,MAAM,KAAK,SAAS,MAAM,EAErD;QAAI,MAAM,SAAS,YACjB,SAAQ,KAAK,KAAK;;;SAIlB;AAGR,QAAO;;;AAQT,MAAM,gBAAgB,IAAI,IAAI;CAAC;CAAQ;CAAgB;CAAQ;CAAe,CAAC;;;;;;;;;;;;;AAc/E,SAAgB,6BAA6B,UAA2B;CACtE,MAAM,aAAa,UAAU,SAAS;AAGtC,KAAI,cAAc,IAAI,WAAW,CAAE,QAAO;AAC1C,MAAK,MAAM,WAAW,cACpB,KAAI,WAAW,WAAW,UAAU,IAAI,CAAE,QAAO;CAGnD,MAAM,OAAO,SAAS;AAGtB,KAAI,KAAK,WAAW,WAAW,IAAI,eAAe,IAChD,QAAO;AAIT,KAAI,WAAW,WAAW,KAAK,EAAE;EAC/B,MAAM,MAAM,WAAW,MAAM,KAAK,OAAO,CAAC,QAAQ,OAAO,GAAG;AAE5D,OADc,MAAM,IAAI,MAAM,IAAI,CAAC,SAAS,OAC9B,EAAG,QAAO;;AAK1B,KAAI,WAAW,KAAK,YAAY,OAAO,CAAC,CACtC,QAAO;AAGT,QAAO;;AAOT,MAAM,mBAAmB;;;;;;;AAQzB,SAAgB,uBAAuB,UAAiC;CACtE,MAAM,IAAI,iBAAiB,KAAK,SAAS;AACzC,KAAI,CAAC,EAAG,QAAO;CACf,MAAM,GAAG,KAAK,MAAM,SAAS;AAC7B,QAAO,YAAY,IAAI,GAAG,KAAK,IAAI;;;AAIrC,MAAa,oBAAoB"}