@tekmidian/pai 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -3
- package/dist/{auto-route-CLYcToZ5.mjs → auto-route-CdvjEWs5.mjs} +4 -4
- package/dist/{auto-route-CLYcToZ5.mjs.map → auto-route-CdvjEWs5.mjs.map} +1 -1
- package/dist/auto-route-DkIRampF.mjs +86 -0
- package/dist/auto-route-DkIRampF.mjs.map +1 -0
- package/dist/{kg-extraction-C8DEUHTS.mjs → checkpoint-block-Bl0HD3U2.mjs} +442 -13
- package/dist/checkpoint-block-Bl0HD3U2.mjs.map +1 -0
- package/dist/checkpoint-block-C05sK_HK.mjs +1056 -0
- package/dist/checkpoint-block-C05sK_HK.mjs.map +1 -0
- package/dist/checkpoint-block-CUCG10Uh.mjs +1022 -0
- package/dist/checkpoint-block-CUCG10Uh.mjs.map +1 -0
- package/dist/checkpoint-block-D5IyLLTr.mjs +1062 -0
- package/dist/checkpoint-block-D5IyLLTr.mjs.map +1 -0
- package/dist/cli/index.mjs +14 -12
- package/dist/cli/index.mjs.map +1 -1
- package/dist/cli/program.d.mts.map +1 -1
- package/dist/cli/program.mjs +31 -16
- package/dist/cli/program.mjs.map +1 -1
- package/dist/{clusters-BYPw7vfW.mjs → clusters-CKWDwcMX.mjs} +2 -2
- package/dist/{clusters-BYPw7vfW.mjs.map → clusters-CKWDwcMX.mjs.map} +1 -1
- package/dist/{config-C8m-tPhP.mjs → config-DGOfoGfm.mjs} +1 -1
- package/dist/{config-C8m-tPhP.mjs.map → config-DGOfoGfm.mjs.map} +1 -1
- package/dist/daemon/index.mjs +18 -16
- package/dist/daemon/index.mjs.map +1 -1
- package/dist/{daemon-COSvHTNu.mjs → daemon-1rKbN-fd.mjs} +31 -30
- package/dist/{daemon-COSvHTNu.mjs.map → daemon-1rKbN-fd.mjs.map} +1 -1
- package/dist/daemon-9AL6Pwh-.mjs +1581 -0
- package/dist/daemon-9AL6Pwh-.mjs.map +1 -0
- package/dist/daemon-Bbk-_rV6.mjs +1581 -0
- package/dist/daemon-Bbk-_rV6.mjs.map +1 -0
- package/dist/daemon-BvcaMk4g.mjs +1581 -0
- package/dist/daemon-BvcaMk4g.mjs.map +1 -0
- package/dist/daemon-CTDEfwcq.mjs +1576 -0
- package/dist/daemon-CTDEfwcq.mjs.map +1 -0
- package/dist/daemon-ClbfaF60.mjs +1576 -0
- package/dist/daemon-ClbfaF60.mjs.map +1 -0
- package/dist/daemon-DDQNDgYF.mjs +1576 -0
- package/dist/daemon-DDQNDgYF.mjs.map +1 -0
- package/dist/daemon-DU8yDWlQ.mjs +1576 -0
- package/dist/daemon-DU8yDWlQ.mjs.map +1 -0
- package/dist/daemon-NJr4iSW3.mjs +1576 -0
- package/dist/daemon-NJr4iSW3.mjs.map +1 -0
- package/dist/daemon-mcp/index.mjs +2 -2
- package/dist/db-BtuN768f.mjs +206 -0
- package/dist/db-BtuN768f.mjs.map +1 -0
- package/dist/{db-CmYbAVCD.mjs → db-CYmBWcjh.mjs} +1 -1
- package/dist/{db-CmYbAVCD.mjs.map → db-CYmBWcjh.mjs.map} +1 -1
- package/dist/{detect-KjycLtXM.mjs → detect-Bf2z-oKB.mjs} +1 -1
- package/dist/{detect-KjycLtXM.mjs.map → detect-Bf2z-oKB.mjs.map} +1 -1
- package/dist/{detector-ZiOhHszd.mjs → detector-Bwzm3E_x.mjs} +2 -2
- package/dist/{detector-ZiOhHszd.mjs.map → detector-Bwzm3E_x.mjs.map} +1 -1
- package/dist/detector-rq9JGTmS.mjs +74 -0
- package/dist/detector-rq9JGTmS.mjs.map +1 -0
- package/dist/embeddings-Bn86ssxR.mjs +119 -0
- package/dist/embeddings-Bn86ssxR.mjs.map +1 -0
- package/dist/{embeddings-BJPOcbik.mjs → embeddings-DGRAPAYb.mjs} +1 -1
- package/dist/{embeddings-BJPOcbik.mjs.map → embeddings-DGRAPAYb.mjs.map} +1 -1
- package/dist/{factory-Q88X1bAN.mjs → factory-0-57Ac-7.mjs} +8 -5
- package/dist/{factory-Q88X1bAN.mjs.map → factory-0-57Ac-7.mjs.map} +1 -1
- package/dist/factory-HcQsQqh0.mjs +72 -0
- package/dist/factory-HcQsQqh0.mjs.map +1 -0
- package/dist/factory-PDXQdkLO.mjs +72 -0
- package/dist/factory-PDXQdkLO.mjs.map +1 -0
- package/dist/{helpers-IjZkXBhj.mjs → helpers-crDEr6S2.mjs} +1 -1
- package/dist/{helpers-IjZkXBhj.mjs.map → helpers-crDEr6S2.mjs.map} +1 -1
- package/dist/hooks/capture-all-events.mjs.map +1 -1
- package/dist/hooks/capture-session-summary.mjs.map +1 -1
- package/dist/hooks/capture-tool-output.mjs.map +1 -1
- package/dist/hooks/cleanup-session-files.mjs.map +1 -1
- package/dist/hooks/context-compression-hook.mjs +336 -48
- package/dist/hooks/context-compression-hook.mjs.map +4 -4
- package/dist/hooks/initialize-session.mjs.map +1 -1
- package/dist/hooks/inject-observations.mjs.map +1 -1
- package/dist/hooks/load-core-context.mjs.map +1 -1
- package/dist/hooks/load-project-context.mjs +188 -6
- package/dist/hooks/load-project-context.mjs.map +4 -4
- package/dist/hooks/observe.mjs.map +1 -1
- package/dist/hooks/post-compact-inject.mjs.map +1 -1
- package/dist/hooks/security-validator.mjs.map +1 -1
- package/dist/hooks/session-commands.mjs.map +1 -1
- package/dist/hooks/stop-hook.mjs +331 -43
- package/dist/hooks/stop-hook.mjs.map +4 -4
- package/dist/hooks/subagent-stop-hook.mjs.map +1 -1
- package/dist/hooks/sync-todo-to-md.mjs.map +1 -1
- package/dist/hooks/update-tab-on-action.mjs.map +1 -1
- package/dist/hooks/update-tab-titles.mjs.map +1 -1
- package/dist/hooks/whisper-rules.mjs.map +1 -1
- package/dist/index.d.mts +0 -9
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +10 -9
- package/dist/indexer-AEcT8wHf.mjs +1 -0
- package/dist/{indexer-backend-Cm5RS7y0.mjs → indexer-backend-BQZXoarl.mjs} +3 -3
- package/dist/{indexer-backend-Cm5RS7y0.mjs.map → indexer-backend-BQZXoarl.mjs.map} +1 -1
- package/dist/indexer-backend-BmaS2VcD.mjs +299 -0
- package/dist/indexer-backend-BmaS2VcD.mjs.map +1 -0
- package/dist/indexer-backend-COrJ5rjS.mjs +299 -0
- package/dist/indexer-backend-COrJ5rjS.mjs.map +1 -0
- package/dist/indexer-backend-NwD7TBGh.mjs +301 -0
- package/dist/indexer-backend-NwD7TBGh.mjs.map +1 -0
- package/dist/{ipc-client-CoyUHPod.mjs → ipc-client-vO2GV325.mjs} +1 -1
- package/dist/{ipc-client-CoyUHPod.mjs.map → ipc-client-vO2GV325.mjs.map} +1 -1
- package/dist/{kg-entity-LXblD7LZ.mjs → kg-entity-DbOMPdF9.mjs} +1 -1
- package/dist/{kg-entity-LXblD7LZ.mjs.map → kg-entity-DbOMPdF9.mjs.map} +1 -1
- package/dist/{latent-ideas-RyC6eyqI.mjs → latent-ideas-BwM3WdBp.mjs} +4 -4
- package/dist/{latent-ideas-RyC6eyqI.mjs.map → latent-ideas-BwM3WdBp.mjs.map} +1 -1
- package/dist/latent-ideas-C6IFFT0-.mjs +191 -0
- package/dist/latent-ideas-C6IFFT0-.mjs.map +1 -0
- package/dist/{main-resolver-uFxNiDy7.mjs → main-resolver-DimnS2qP.mjs} +22 -22
- package/dist/{main-resolver-uFxNiDy7.mjs.map → main-resolver-DimnS2qP.mjs.map} +1 -1
- package/dist/{migrate-B02wDDgS.mjs → migrate-fLD6rAdO.mjs} +2 -2
- package/dist/{migrate-B02wDDgS.mjs.map → migrate-fLD6rAdO.mjs.map} +1 -1
- package/dist/{neighborhood-CklGIB8r.mjs → neighborhood-CzRNG2Oy.mjs} +2 -2
- package/dist/{neighborhood-CklGIB8r.mjs.map → neighborhood-CzRNG2Oy.mjs.map} +1 -1
- package/dist/neighborhood-DZY0wkpE.mjs +135 -0
- package/dist/neighborhood-DZY0wkpE.mjs.map +1 -0
- package/dist/{note-context-qvZlVXVO.mjs → note-context-CrfMbr4R.mjs} +1 -1
- package/dist/{note-context-qvZlVXVO.mjs.map → note-context-CrfMbr4R.mjs.map} +1 -1
- package/dist/{pai-marker-HVBwwBW-.mjs → pai-marker-B20KqhA8.mjs} +44 -2
- package/dist/pai-marker-B20KqhA8.mjs.map +1 -0
- package/dist/pick-B0q5cy79.mjs +11785 -0
- package/dist/pick-B0q5cy79.mjs.map +1 -0
- package/dist/{pick-r95zvydr.mjs → pick-B3zEYjin.mjs} +2288 -497
- package/dist/pick-B3zEYjin.mjs.map +1 -0
- package/dist/pick-BHE9_iA0.mjs +11947 -0
- package/dist/pick-BHE9_iA0.mjs.map +1 -0
- package/dist/pick-BaTjoNqM.mjs +11544 -0
- package/dist/pick-BaTjoNqM.mjs.map +1 -0
- package/dist/pick-C6JbZfN4.mjs +11893 -0
- package/dist/pick-C6JbZfN4.mjs.map +1 -0
- package/dist/pick-CK-bJyjp.mjs +11883 -0
- package/dist/pick-CK-bJyjp.mjs.map +1 -0
- package/dist/pick-CLBkGHyI.mjs +11826 -0
- package/dist/pick-CLBkGHyI.mjs.map +1 -0
- package/dist/pick-CTOif21h.mjs +11544 -0
- package/dist/pick-CTOif21h.mjs.map +1 -0
- package/dist/pick-Ck2cdQgD.mjs +11893 -0
- package/dist/pick-Ck2cdQgD.mjs.map +1 -0
- package/dist/pick-Cr4nJzHA.mjs +11914 -0
- package/dist/pick-Cr4nJzHA.mjs.map +1 -0
- package/dist/pick-D4dFfKAu.mjs +11893 -0
- package/dist/pick-D4dFfKAu.mjs.map +1 -0
- package/dist/pick-D7-lYiGX.mjs +11543 -0
- package/dist/pick-D7-lYiGX.mjs.map +1 -0
- package/dist/pick-DMjCY0np.mjs +11875 -0
- package/dist/pick-DMjCY0np.mjs.map +1 -0
- package/dist/pick-DbX4OrCP.mjs +11543 -0
- package/dist/pick-DbX4OrCP.mjs.map +1 -0
- package/dist/pick-iCiPjlnC.mjs +11893 -0
- package/dist/pick-iCiPjlnC.mjs.map +1 -0
- package/dist/pick-sF9GnQUS.mjs +11893 -0
- package/dist/pick-sF9GnQUS.mjs.map +1 -0
- package/dist/pick-yHKUrnfG.mjs +11793 -0
- package/dist/pick-yHKUrnfG.mjs.map +1 -0
- package/dist/postgres-BJxiqtqg.mjs +891 -0
- package/dist/postgres-BJxiqtqg.mjs.map +1 -0
- package/dist/{postgres-B451Hnkm.mjs → postgres-zjKqPl4X.mjs} +2 -2
- package/dist/{postgres-B451Hnkm.mjs.map → postgres-zjKqPl4X.mjs.map} +1 -1
- package/dist/{query-feedback-fapbqRgh.mjs → query-feedback-BoY8_Dbb.mjs} +1 -1
- package/dist/{query-feedback-fapbqRgh.mjs.map → query-feedback-BoY8_Dbb.mjs.map} +1 -1
- package/dist/{reranker-C08R99zn.mjs → reranker-CMNZcfVx.mjs} +1 -1
- package/dist/{reranker-C08R99zn.mjs.map → reranker-CMNZcfVx.mjs.map} +1 -1
- package/dist/{search-HcdKtMla.mjs → search-CpTv1I24.mjs} +3 -3
- package/dist/{search-HcdKtMla.mjs.map → search-CpTv1I24.mjs.map} +1 -1
- package/dist/search-i2nlQ-JM.mjs +298 -0
- package/dist/search-i2nlQ-JM.mjs.map +1 -0
- package/dist/skills/End/SKILL.md +78 -0
- package/dist/skills/Pause/SKILL.md +84 -0
- package/dist/sqlite-CZ0LFili.mjs +271 -0
- package/dist/sqlite-CZ0LFili.mjs.map +1 -0
- package/dist/{sqlite-DQpY1Esi.mjs → sqlite-ChSCYwiC.mjs} +3 -3
- package/dist/{sqlite-DQpY1Esi.mjs.map → sqlite-ChSCYwiC.mjs.map} +1 -1
- package/dist/sqlite-Nw8ZW5Rj.mjs +271 -0
- package/dist/sqlite-Nw8ZW5Rj.mjs.map +1 -0
- package/dist/{state-BIlxNRUn.mjs → state-BXIdbxDs.mjs} +1 -1
- package/dist/{state-BIlxNRUn.mjs.map → state-BXIdbxDs.mjs.map} +1 -1
- package/dist/{stop-words-BwplsQ3z.mjs → stop-words-BaMEGVeY.mjs} +1 -1
- package/dist/{stop-words-BwplsQ3z.mjs.map → stop-words-BaMEGVeY.mjs.map} +1 -1
- package/dist/{sync-uPR4g438.mjs → sync--BoxBBok.mjs} +6 -205
- package/dist/sync--BoxBBok.mjs.map +1 -0
- package/dist/sync-CmBKOL3K.mjs +310 -0
- package/dist/sync-CmBKOL3K.mjs.map +1 -0
- package/dist/{themes-Cg8oG9Ra.mjs → themes-BBOlGXAg.mjs} +3 -3
- package/dist/{themes-Cg8oG9Ra.mjs.map → themes-BBOlGXAg.mjs.map} +1 -1
- package/dist/themes-nvRM8v84.mjs +148 -0
- package/dist/themes-nvRM8v84.mjs.map +1 -0
- package/dist/tools-B6BmaAvz.mjs +1939 -0
- package/dist/tools-B6BmaAvz.mjs.map +1 -0
- package/dist/tools-CAIY8TAb.mjs +1939 -0
- package/dist/tools-CAIY8TAb.mjs.map +1 -0
- package/dist/{tools-Op3C2Nm6.mjs → tools-CznjzfYz.mjs} +26 -26
- package/dist/{tools-Op3C2Nm6.mjs.map → tools-CznjzfYz.mjs.map} +1 -1
- package/dist/{trace-CLK-NPkb.mjs → trace-B8oz1ok0.mjs} +1 -1
- package/dist/{trace-CLK-NPkb.mjs.map → trace-B8oz1ok0.mjs.map} +1 -1
- package/dist/{utils-CqqgB0dH.mjs → utils-BAxjW3j8.mjs} +1 -1
- package/dist/{utils-CqqgB0dH.mjs.map → utils-BAxjW3j8.mjs.map} +1 -1
- package/dist/{vault-indexer-Dn_BccVK.mjs → vault-indexer-DrttVeEk.mjs} +2 -2
- package/dist/{vault-indexer-Dn_BccVK.mjs.map → vault-indexer-DrttVeEk.mjs.map} +1 -1
- package/dist/{work-queue-worker-BwouhVkb.mjs → work-queue-worker-BPXLkEik.mjs} +47 -30
- package/dist/work-queue-worker-BPXLkEik.mjs.map +1 -0
- package/dist/work-queue-worker-BoHIH70K.mjs +1856 -0
- package/dist/work-queue-worker-BoHIH70K.mjs.map +1 -0
- package/dist/work-queue-worker-CZGcQoLw.mjs +1856 -0
- package/dist/work-queue-worker-CZGcQoLw.mjs.map +1 -0
- package/dist/work-queue-worker-D05Ze9e9.mjs +1856 -0
- package/dist/work-queue-worker-D05Ze9e9.mjs.map +1 -0
- package/dist/work-queue-worker-DdXFINfH.mjs +1856 -0
- package/dist/work-queue-worker-DdXFINfH.mjs.map +1 -0
- package/dist/{zettelkasten-NwBD1NsT.mjs → zettelkasten-CTlt1dGv.mjs} +4 -4
- package/dist/{zettelkasten-NwBD1NsT.mjs.map → zettelkasten-CTlt1dGv.mjs.map} +1 -1
- package/dist/zettelkasten-D0N_A_OE.mjs +1063 -0
- package/dist/zettelkasten-D0N_A_OE.mjs.map +1 -0
- package/docs/commands/README.md +18 -0
- package/docs/commands/backup.md +1 -1
- package/docs/commands/clear-names.md +1 -1
- package/docs/commands/daemon.md +1 -1
- package/docs/commands/db.md +1 -1
- package/docs/commands/end.md +4 -1
- package/docs/commands/help.md +1 -1
- package/docs/commands/kg.md +1 -1
- package/docs/commands/mcp.md +1 -1
- package/docs/commands/memory.md +1 -1
- package/docs/commands/notify.md +1 -1
- package/docs/commands/observation.md +1 -1
- package/docs/commands/obsidian.md +1 -1
- package/docs/commands/pause.md +5 -2
- package/docs/commands/project.md +1 -1
- package/docs/commands/projects.md +1 -1
- package/docs/commands/registry.md +18 -1
- package/docs/commands/restore.md +1 -1
- package/docs/commands/session.md +295 -0
- package/docs/commands/sessions.md +1 -1
- package/docs/commands/setup.md +1 -1
- package/docs/commands/shell-init.md +1 -1
- package/docs/commands/skill.md +1 -1
- package/docs/commands/task.md +1 -1
- package/docs/commands/topic.md +1 -1
- package/docs/commands/update.md +1 -1
- package/docs/commands/zettel.md +1 -1
- package/package.json +1 -1
- package/plugins/context-preservation/hooks/hooks.json +12 -1
- package/scripts/build-hooks.mjs +33 -3
- package/src/hooks/session-autosave.sh +69 -0
- package/src/hooks/session-stop.sh +32 -1
- package/src/hooks/ts/lib/project-utils/todo.test.ts +132 -0
- package/src/hooks/ts/lib/project-utils/todo.ts +60 -34
- package/src/hooks/ts/session-start/load-project-context.ts +44 -0
- package/dist/kg-extraction-C8DEUHTS.mjs.map +0 -1
- package/dist/pai-marker-HVBwwBW-.mjs.map +0 -1
- package/dist/pick-r95zvydr.mjs.map +0 -1
- package/dist/sync-uPR4g438.mjs.map +0 -1
- package/dist/work-queue-worker-BwouhVkb.mjs.map +0 -1
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
import { a as parseSessionTitleChunk, c as yieldToEventLoop, f as sha256File, i as isPathTooBroadForContentScan, l as chunkMarkdown, n as chunkId, o as walkContentFiles, r as detectTier, s as walkMdFiles, t as INDEX_YIELD_EVERY } from "./helpers-crDEr6S2.mjs";
|
|
2
|
+
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
3
|
+
import { basename, join, relative } from "node:path";
|
|
4
|
+
|
|
5
|
+
//#region src/memory/indexer/async.ts
|
|
6
|
+
/**
|
|
7
|
+
* Backend-aware async indexer for PAI federation memory.
|
|
8
|
+
*
|
|
9
|
+
* Provides the same functionality as sync.ts but writes through the
|
|
10
|
+
* StorageBackend interface instead of directly to better-sqlite3.
|
|
11
|
+
* Used when the daemon is configured with the Postgres backend.
|
|
12
|
+
*
|
|
13
|
+
* The SQLite path still uses sync.ts directly (which is faster for SQLite
|
|
14
|
+
* due to synchronous transactions).
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* Index a single file through the StorageBackend interface.
|
|
18
|
+
* Returns true if the file was re-indexed (changed or new), false if skipped.
|
|
19
|
+
*/
|
|
20
|
+
async function indexFileWithBackend(backend, projectId, rootPath, relativePath, source, tier) {
|
|
21
|
+
const absPath = join(rootPath, relativePath);
|
|
22
|
+
let content;
|
|
23
|
+
let stat;
|
|
24
|
+
try {
|
|
25
|
+
content = readFileSync(absPath, "utf8");
|
|
26
|
+
stat = statSync(absPath);
|
|
27
|
+
} catch {
|
|
28
|
+
return false;
|
|
29
|
+
}
|
|
30
|
+
const hash = sha256File(content);
|
|
31
|
+
const mtime = Math.floor(stat.mtimeMs);
|
|
32
|
+
const size = stat.size;
|
|
33
|
+
if (await backend.getFileHash(projectId, relativePath) === hash) return false;
|
|
34
|
+
await backend.deleteChunksForFile(projectId, relativePath);
|
|
35
|
+
const rawChunks = chunkMarkdown(content);
|
|
36
|
+
const updatedAt = Date.now();
|
|
37
|
+
const chunks = rawChunks.map((c, i) => ({
|
|
38
|
+
id: chunkId(projectId, relativePath, i, c.startLine, c.endLine),
|
|
39
|
+
projectId,
|
|
40
|
+
source,
|
|
41
|
+
tier,
|
|
42
|
+
path: relativePath,
|
|
43
|
+
startLine: c.startLine,
|
|
44
|
+
endLine: c.endLine,
|
|
45
|
+
hash: c.hash,
|
|
46
|
+
text: c.text,
|
|
47
|
+
updatedAt,
|
|
48
|
+
embedding: null
|
|
49
|
+
}));
|
|
50
|
+
await backend.insertChunks(chunks);
|
|
51
|
+
await backend.upsertFile({
|
|
52
|
+
projectId,
|
|
53
|
+
path: relativePath,
|
|
54
|
+
source,
|
|
55
|
+
tier,
|
|
56
|
+
hash,
|
|
57
|
+
mtime,
|
|
58
|
+
size
|
|
59
|
+
});
|
|
60
|
+
return true;
|
|
61
|
+
}
|
|
62
|
+
async function indexProjectWithBackend(backend, projectId, rootPath, claudeNotesDir) {
|
|
63
|
+
const result = {
|
|
64
|
+
filesProcessed: 0,
|
|
65
|
+
chunksCreated: 0,
|
|
66
|
+
filesSkipped: 0
|
|
67
|
+
};
|
|
68
|
+
const filesToIndex = [];
|
|
69
|
+
const rootMemoryMd = join(rootPath, "MEMORY.md");
|
|
70
|
+
if (existsSync(rootMemoryMd)) filesToIndex.push({
|
|
71
|
+
absPath: rootMemoryMd,
|
|
72
|
+
rootBase: rootPath,
|
|
73
|
+
source: "memory",
|
|
74
|
+
tier: "evergreen"
|
|
75
|
+
});
|
|
76
|
+
const memoryDir = join(rootPath, "memory");
|
|
77
|
+
for (const absPath of walkMdFiles(memoryDir)) {
|
|
78
|
+
const tier = detectTier(relative(rootPath, absPath));
|
|
79
|
+
filesToIndex.push({
|
|
80
|
+
absPath,
|
|
81
|
+
rootBase: rootPath,
|
|
82
|
+
source: "memory",
|
|
83
|
+
tier
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
const notesDir = join(rootPath, "Notes");
|
|
87
|
+
for (const absPath of walkMdFiles(notesDir)) filesToIndex.push({
|
|
88
|
+
absPath,
|
|
89
|
+
rootBase: rootPath,
|
|
90
|
+
source: "notes",
|
|
91
|
+
tier: "session"
|
|
92
|
+
});
|
|
93
|
+
{
|
|
94
|
+
const updatedAt = Date.now();
|
|
95
|
+
for (const absPath of walkMdFiles(notesDir)) {
|
|
96
|
+
const text = parseSessionTitleChunk(basename(absPath));
|
|
97
|
+
if (!text) continue;
|
|
98
|
+
const syntheticPath = `${relative(rootPath, absPath)}::title`;
|
|
99
|
+
const titleChunk = {
|
|
100
|
+
id: chunkId(projectId, syntheticPath, 0, 0, 0),
|
|
101
|
+
projectId,
|
|
102
|
+
source: "notes",
|
|
103
|
+
tier: "session",
|
|
104
|
+
path: syntheticPath,
|
|
105
|
+
startLine: 0,
|
|
106
|
+
endLine: 0,
|
|
107
|
+
hash: sha256File(text),
|
|
108
|
+
text,
|
|
109
|
+
updatedAt,
|
|
110
|
+
embedding: null
|
|
111
|
+
};
|
|
112
|
+
try {
|
|
113
|
+
await backend.insertChunks([titleChunk]);
|
|
114
|
+
} catch {}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
if (!isPathTooBroadForContentScan(rootPath)) for (const absPath of walkContentFiles(rootPath)) filesToIndex.push({
|
|
118
|
+
absPath,
|
|
119
|
+
rootBase: rootPath,
|
|
120
|
+
source: "content",
|
|
121
|
+
tier: "topic"
|
|
122
|
+
});
|
|
123
|
+
if (claudeNotesDir && claudeNotesDir !== notesDir) {
|
|
124
|
+
for (const absPath of walkMdFiles(claudeNotesDir)) filesToIndex.push({
|
|
125
|
+
absPath,
|
|
126
|
+
rootBase: claudeNotesDir,
|
|
127
|
+
source: "notes",
|
|
128
|
+
tier: "session"
|
|
129
|
+
});
|
|
130
|
+
{
|
|
131
|
+
const updatedAt = Date.now();
|
|
132
|
+
for (const absPath of walkMdFiles(claudeNotesDir)) {
|
|
133
|
+
const text = parseSessionTitleChunk(basename(absPath));
|
|
134
|
+
if (!text) continue;
|
|
135
|
+
const syntheticPath = `${relative(claudeNotesDir, absPath)}::title`;
|
|
136
|
+
const titleChunk = {
|
|
137
|
+
id: chunkId(projectId, syntheticPath, 0, 0, 0),
|
|
138
|
+
projectId,
|
|
139
|
+
source: "notes",
|
|
140
|
+
tier: "session",
|
|
141
|
+
path: syntheticPath,
|
|
142
|
+
startLine: 0,
|
|
143
|
+
endLine: 0,
|
|
144
|
+
hash: sha256File(text),
|
|
145
|
+
text,
|
|
146
|
+
updatedAt,
|
|
147
|
+
embedding: null
|
|
148
|
+
};
|
|
149
|
+
try {
|
|
150
|
+
await backend.insertChunks([titleChunk]);
|
|
151
|
+
} catch {}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
if (claudeNotesDir.endsWith("/Notes")) {
|
|
155
|
+
const claudeProjectDir = claudeNotesDir.slice(0, -6);
|
|
156
|
+
const claudeMemoryMd = join(claudeProjectDir, "MEMORY.md");
|
|
157
|
+
if (existsSync(claudeMemoryMd)) filesToIndex.push({
|
|
158
|
+
absPath: claudeMemoryMd,
|
|
159
|
+
rootBase: claudeProjectDir,
|
|
160
|
+
source: "memory",
|
|
161
|
+
tier: "evergreen"
|
|
162
|
+
});
|
|
163
|
+
const claudeMemoryDir = join(claudeProjectDir, "memory");
|
|
164
|
+
for (const absPath of walkMdFiles(claudeMemoryDir)) {
|
|
165
|
+
const tier = detectTier(relative(claudeProjectDir, absPath));
|
|
166
|
+
filesToIndex.push({
|
|
167
|
+
absPath,
|
|
168
|
+
rootBase: claudeProjectDir,
|
|
169
|
+
source: "memory",
|
|
170
|
+
tier
|
|
171
|
+
});
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
await yieldToEventLoop();
|
|
176
|
+
let filesSinceYield = 0;
|
|
177
|
+
for (const { absPath, rootBase, source, tier } of filesToIndex) {
|
|
178
|
+
if (filesSinceYield >= INDEX_YIELD_EVERY) {
|
|
179
|
+
await yieldToEventLoop();
|
|
180
|
+
filesSinceYield = 0;
|
|
181
|
+
}
|
|
182
|
+
filesSinceYield++;
|
|
183
|
+
const relPath = relative(rootBase, absPath);
|
|
184
|
+
try {
|
|
185
|
+
if (await indexFileWithBackend(backend, projectId, rootBase, relPath, source, tier)) {
|
|
186
|
+
const ids = await backend.getChunkIds(projectId, relPath);
|
|
187
|
+
result.filesProcessed++;
|
|
188
|
+
result.chunksCreated += ids.length;
|
|
189
|
+
} else result.filesSkipped++;
|
|
190
|
+
} catch {
|
|
191
|
+
result.filesSkipped++;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
const livePaths = /* @__PURE__ */ new Set();
|
|
195
|
+
for (const { absPath, rootBase } of filesToIndex) livePaths.add(relative(rootBase, absPath));
|
|
196
|
+
const dbChunkPaths = await backend.getDistinctChunkPaths(projectId);
|
|
197
|
+
const stalePaths = [];
|
|
198
|
+
for (const p of dbChunkPaths) {
|
|
199
|
+
const basePath = p.endsWith("::title") ? p.slice(0, -7) : p;
|
|
200
|
+
if (!livePaths.has(basePath)) stalePaths.push(p);
|
|
201
|
+
}
|
|
202
|
+
if (stalePaths.length > 0) await backend.deletePaths(projectId, stalePaths);
|
|
203
|
+
return result;
|
|
204
|
+
}
|
|
205
|
+
const EMBED_BATCH_SIZE = 50;
|
|
206
|
+
/** Default ceiling on one pass. See EmbedPassOptions. */
|
|
207
|
+
const DEFAULT_MAX_CHUNKS_PER_PASS = 5e3;
|
|
208
|
+
/** Default wall-clock ceiling on one pass, in milliseconds. */
|
|
209
|
+
const DEFAULT_MAX_MILLIS_PER_PASS = 12e4;
|
|
210
|
+
/**
|
|
211
|
+
* Generate and store embeddings for unembedded chunks via the StorageBackend.
|
|
212
|
+
*
|
|
213
|
+
* The pass is deliberately bounded (see EmbedPassOptions) and resumable: it
|
|
214
|
+
* takes a slice of the backlog, embeds it, and returns so the scheduler can run
|
|
215
|
+
* an index pass before coming back. Draining the whole backlog in one call
|
|
216
|
+
* starves indexing for as long as the call runs.
|
|
217
|
+
*
|
|
218
|
+
* The optional `shouldStop` callback is checked between every batch. When it
|
|
219
|
+
* returns true the embed loop exits early so the caller (e.g. the daemon
|
|
220
|
+
* shutdown handler) can close the pool without racing against active queries.
|
|
221
|
+
*
|
|
222
|
+
* Returns the number of newly embedded chunks.
|
|
223
|
+
*/
|
|
224
|
+
async function embedChunksWithBackend(backend, shouldStop, projectNames, options) {
|
|
225
|
+
const { generateEmbeddings, serializeEmbedding } = await import("./embeddings-Bn86ssxR.mjs").then((n) => n.i);
|
|
226
|
+
const maxChunks = options?.maxChunks ?? DEFAULT_MAX_CHUNKS_PER_PASS;
|
|
227
|
+
const maxMillis = options?.maxMillis ?? DEFAULT_MAX_MILLIS_PER_PASS;
|
|
228
|
+
const deadline = Date.now() + maxMillis;
|
|
229
|
+
const rows = await backend.getUnembeddedChunkIds(void 0, maxChunks);
|
|
230
|
+
if (rows.length === 0) return 0;
|
|
231
|
+
const total = rows.length;
|
|
232
|
+
let embedded = 0;
|
|
233
|
+
const projectChunkCounts = /* @__PURE__ */ new Map();
|
|
234
|
+
for (const row of rows) {
|
|
235
|
+
const entry = projectChunkCounts.get(row.project_id);
|
|
236
|
+
if (entry) entry.count++;
|
|
237
|
+
else projectChunkCounts.set(row.project_id, {
|
|
238
|
+
count: 1,
|
|
239
|
+
samplePath: row.path
|
|
240
|
+
});
|
|
241
|
+
}
|
|
242
|
+
const pName = (pid) => projectNames?.get(pid) ?? `project ${pid}`;
|
|
243
|
+
const projectSummary = Array.from(projectChunkCounts.entries()).map(([pid, { count, samplePath }]) => ` ${pName(pid)}: ${count} chunks (e.g. ${samplePath})`).join("\n");
|
|
244
|
+
process.stderr.write(`[pai-daemon] Embed pass: ${total} unembedded chunks across ${projectChunkCounts.size} project(s)\n${projectSummary}\n`);
|
|
245
|
+
let currentProjectId = -1;
|
|
246
|
+
let projectEmbedded = 0;
|
|
247
|
+
for (let i = 0; i < rows.length; i += EMBED_BATCH_SIZE) {
|
|
248
|
+
if (shouldStop?.()) {
|
|
249
|
+
process.stderr.write(`[pai-daemon] Embed pass cancelled after ${embedded}/${total} chunks (shutdown requested)\n`);
|
|
250
|
+
break;
|
|
251
|
+
}
|
|
252
|
+
if (Date.now() >= deadline) {
|
|
253
|
+
process.stderr.write(`[pai-daemon] Embed pass yielding after ${embedded}/${total} chunks (${maxMillis}ms budget spent); resuming next pass\n`);
|
|
254
|
+
break;
|
|
255
|
+
}
|
|
256
|
+
const batch = rows.slice(i, i + EMBED_BATCH_SIZE);
|
|
257
|
+
await yieldToEventLoop();
|
|
258
|
+
const vecs = await generateEmbeddings(batch.map((r) => r.text));
|
|
259
|
+
for (let j = 0; j < batch.length; j++) await backend.updateEmbedding(batch[j].id, serializeEmbedding(vecs[j]));
|
|
260
|
+
for (const { project_id, path } of batch) {
|
|
261
|
+
if (project_id !== currentProjectId) {
|
|
262
|
+
if (currentProjectId !== -1) process.stderr.write(`[pai-daemon] Finished ${pName(currentProjectId)}: ${projectEmbedded} chunks embedded\n`);
|
|
263
|
+
const info = projectChunkCounts.get(project_id);
|
|
264
|
+
process.stderr.write(`[pai-daemon] Embedding ${pName(project_id)} (${info?.count ?? "?"} chunks, starting at ${path})\n`);
|
|
265
|
+
currentProjectId = project_id;
|
|
266
|
+
projectEmbedded = 0;
|
|
267
|
+
}
|
|
268
|
+
projectEmbedded++;
|
|
269
|
+
}
|
|
270
|
+
embedded += batch.length;
|
|
271
|
+
const lastChunk = batch[batch.length - 1];
|
|
272
|
+
process.stderr.write(`[pai-daemon] Embedded ${embedded}/${total} chunks (${pName(lastChunk.project_id)}: ${lastChunk.path})\n`);
|
|
273
|
+
}
|
|
274
|
+
if (currentProjectId !== -1) process.stderr.write(`[pai-daemon] Finished ${pName(currentProjectId)}: ${projectEmbedded} chunks embedded\n`);
|
|
275
|
+
return embedded;
|
|
276
|
+
}
|
|
277
|
+
async function indexAllWithBackend(backend, registryDb) {
|
|
278
|
+
const projects = registryDb.prepare("SELECT id, root_path, claude_notes_dir FROM projects WHERE status = 'active'").all();
|
|
279
|
+
const totals = {
|
|
280
|
+
filesProcessed: 0,
|
|
281
|
+
chunksCreated: 0,
|
|
282
|
+
filesSkipped: 0
|
|
283
|
+
};
|
|
284
|
+
for (const project of projects) {
|
|
285
|
+
await yieldToEventLoop();
|
|
286
|
+
const r = await indexProjectWithBackend(backend, project.id, project.root_path, project.claude_notes_dir);
|
|
287
|
+
totals.filesProcessed += r.filesProcessed;
|
|
288
|
+
totals.chunksCreated += r.chunksCreated;
|
|
289
|
+
totals.filesSkipped += r.filesSkipped;
|
|
290
|
+
}
|
|
291
|
+
return {
|
|
292
|
+
projects: projects.length,
|
|
293
|
+
result: totals
|
|
294
|
+
};
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
//#endregion
|
|
298
|
+
export { embedChunksWithBackend, indexAllWithBackend };
|
|
299
|
+
//# sourceMappingURL=indexer-backend-BmaS2VcD.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"indexer-backend-BmaS2VcD.mjs","names":[],"sources":["../src/memory/indexer/async.ts"],"sourcesContent":["/**\n * Backend-aware async indexer for PAI federation memory.\n *\n * Provides the same functionality as sync.ts but writes through the\n * StorageBackend interface instead of directly to better-sqlite3.\n * Used when the daemon is configured with the Postgres backend.\n *\n * The SQLite path still uses sync.ts directly (which is faster for SQLite\n * due to synchronous transactions).\n */\n\nimport { readFileSync, statSync, existsSync } from \"node:fs\";\nimport { join, relative, basename } from \"node:path\";\nimport type { Database } from \"better-sqlite3\";\nimport type { StorageBackend, ChunkRow } from \"../../storage/interface.js\";\nimport { chunkMarkdown } from \"../chunker.js\";\nimport {\n sha256File,\n chunkId,\n detectTier,\n walkMdFiles,\n walkContentFiles,\n isPathTooBroadForContentScan,\n parseSessionTitleChunk,\n yieldToEventLoop,\n INDEX_YIELD_EVERY,\n} from \"./helpers.js\";\nimport type { IndexResult } from \"./types.js\";\n\nexport type { IndexResult };\n\n// ---------------------------------------------------------------------------\n// Single-file indexing via StorageBackend\n// ---------------------------------------------------------------------------\n\n/**\n * Index a single file through the StorageBackend interface.\n * Returns true if the file was re-indexed (changed or new), false if skipped.\n */\nexport async function indexFileWithBackend(\n backend: StorageBackend,\n projectId: number,\n rootPath: string,\n relativePath: string,\n source: string,\n tier: string,\n): Promise<boolean> {\n const absPath = join(rootPath, relativePath);\n\n let content: string;\n let stat: ReturnType<typeof statSync>;\n try {\n content = readFileSync(absPath, \"utf8\");\n stat = statSync(absPath);\n } catch {\n return false;\n }\n\n const hash = sha256File(content);\n const mtime = Math.floor(stat.mtimeMs);\n const size = stat.size;\n\n // Change detection\n const existingHash = await backend.getFileHash(projectId, relativePath);\n if (existingHash === hash) return false;\n\n // Delete old chunks\n await backend.deleteChunksForFile(projectId, relativePath);\n\n // Chunk the content\n const rawChunks = chunkMarkdown(content);\n const updatedAt = Date.now();\n\n const chunks: ChunkRow[] = rawChunks.map((c, i) => ({\n id: chunkId(projectId, relativePath, i, c.startLine, c.endLine),\n projectId,\n source,\n tier,\n path: relativePath,\n startLine: c.startLine,\n endLine: c.endLine,\n hash: c.hash,\n text: c.text,\n updatedAt,\n embedding: null,\n }));\n\n // Insert chunks + update file record\n await backend.insertChunks(chunks);\n await backend.upsertFile({ projectId, path: relativePath, source, tier, hash, mtime, size });\n\n return true;\n}\n\n// ---------------------------------------------------------------------------\n// Project-level indexing via StorageBackend\n// ---------------------------------------------------------------------------\n\nexport async function indexProjectWithBackend(\n backend: StorageBackend,\n projectId: number,\n rootPath: string,\n claudeNotesDir?: string | null,\n): Promise<IndexResult> {\n const result: IndexResult = { filesProcessed: 0, chunksCreated: 0, filesSkipped: 0 };\n\n const filesToIndex: Array<{ absPath: string; rootBase: string; source: string; tier: string }> = [];\n\n const rootMemoryMd = join(rootPath, \"MEMORY.md\");\n if (existsSync(rootMemoryMd)) {\n filesToIndex.push({ absPath: rootMemoryMd, rootBase: rootPath, source: \"memory\", tier: \"evergreen\" });\n }\n\n const memoryDir = join(rootPath, \"memory\");\n for (const absPath of walkMdFiles(memoryDir)) {\n const relPath = relative(rootPath, absPath);\n const tier = detectTier(relPath);\n filesToIndex.push({ absPath, rootBase: rootPath, source: \"memory\", tier });\n }\n\n const notesDir = join(rootPath, \"Notes\");\n for (const absPath of walkMdFiles(notesDir)) {\n filesToIndex.push({ absPath, rootBase: rootPath, source: \"notes\", tier: \"session\" });\n }\n\n // Synthetic session-title chunks for Notes files\n {\n const updatedAt = Date.now();\n for (const absPath of walkMdFiles(notesDir)) {\n const fileName = basename(absPath);\n const text = parseSessionTitleChunk(fileName);\n if (!text) continue;\n const relPath = relative(rootPath, absPath);\n const syntheticPath = `${relPath}::title`;\n const id = chunkId(projectId, syntheticPath, 0, 0, 0);\n const hash = sha256File(text);\n const titleChunk: ChunkRow = {\n id, projectId, source: \"notes\", tier: \"session\",\n path: syntheticPath, startLine: 0, endLine: 0,\n hash, text, updatedAt, embedding: null,\n };\n try {\n await backend.insertChunks([titleChunk]);\n } catch {\n // Skip title chunks that cause backend errors\n }\n }\n }\n\n if (!isPathTooBroadForContentScan(rootPath)) {\n for (const absPath of walkContentFiles(rootPath)) {\n filesToIndex.push({ absPath, rootBase: rootPath, source: \"content\", tier: \"topic\" });\n }\n }\n\n if (claudeNotesDir && claudeNotesDir !== notesDir) {\n for (const absPath of walkMdFiles(claudeNotesDir)) {\n filesToIndex.push({ absPath, rootBase: claudeNotesDir, source: \"notes\", tier: \"session\" });\n }\n\n // Synthetic title chunks for claude notes dir\n {\n const updatedAt = Date.now();\n for (const absPath of walkMdFiles(claudeNotesDir)) {\n const fileName = basename(absPath);\n const text = parseSessionTitleChunk(fileName);\n if (!text) continue;\n const relPath = relative(claudeNotesDir, absPath);\n const syntheticPath = `${relPath}::title`;\n const id = chunkId(projectId, syntheticPath, 0, 0, 0);\n const hash = sha256File(text);\n const titleChunk: ChunkRow = {\n id, projectId, source: \"notes\", tier: \"session\",\n path: syntheticPath, startLine: 0, endLine: 0,\n hash, text, updatedAt, embedding: null,\n };\n try {\n await backend.insertChunks([titleChunk]);\n } catch {\n // Skip title chunks that cause backend errors\n }\n }\n }\n\n if (claudeNotesDir.endsWith(\"/Notes\")) {\n const claudeProjectDir = claudeNotesDir.slice(0, -\"/Notes\".length);\n const claudeMemoryMd = join(claudeProjectDir, \"MEMORY.md\");\n if (existsSync(claudeMemoryMd)) {\n filesToIndex.push({ absPath: claudeMemoryMd, rootBase: claudeProjectDir, source: \"memory\", tier: \"evergreen\" });\n }\n const claudeMemoryDir = join(claudeProjectDir, \"memory\");\n for (const absPath of walkMdFiles(claudeMemoryDir)) {\n const relPath = relative(claudeProjectDir, absPath);\n const tier = detectTier(relPath);\n filesToIndex.push({ absPath, rootBase: claudeProjectDir, source: \"memory\", tier });\n }\n }\n }\n\n await yieldToEventLoop();\n\n let filesSinceYield = 0;\n\n for (const { absPath, rootBase, source, tier } of filesToIndex) {\n if (filesSinceYield >= INDEX_YIELD_EVERY) {\n await yieldToEventLoop();\n filesSinceYield = 0;\n }\n filesSinceYield++;\n\n const relPath = relative(rootBase, absPath);\n try {\n const changed = await indexFileWithBackend(backend, projectId, rootBase, relPath, source, tier);\n\n if (changed) {\n const ids = await backend.getChunkIds(projectId, relPath);\n result.filesProcessed++;\n result.chunksCreated += ids.length;\n } else {\n result.filesSkipped++;\n }\n } catch {\n // Skip files that cause backend errors (e.g. null bytes in Postgres)\n result.filesSkipped++;\n }\n }\n\n // Prune stale paths\n const livePaths = new Set<string>();\n for (const { absPath, rootBase } of filesToIndex) {\n livePaths.add(relative(rootBase, absPath));\n }\n\n const dbChunkPaths = await backend.getDistinctChunkPaths(projectId);\n\n const stalePaths: string[] = [];\n for (const p of dbChunkPaths) {\n const basePath = p.endsWith(\"::title\") ? p.slice(0, -\"::title\".length) : p;\n if (!livePaths.has(basePath)) {\n stalePaths.push(p);\n }\n }\n\n if (stalePaths.length > 0) {\n await backend.deletePaths(projectId, stalePaths);\n }\n\n return result;\n}\n\n// ---------------------------------------------------------------------------\n// Embedding generation via StorageBackend\n// ---------------------------------------------------------------------------\n\nconst EMBED_BATCH_SIZE = 50;\nconst EMBED_YIELD_EVERY = 1;\n\n/** Default ceiling on one pass. See EmbedPassOptions. */\nconst DEFAULT_MAX_CHUNKS_PER_PASS = 5_000;\n/** Default wall-clock ceiling on one pass, in milliseconds. */\nconst DEFAULT_MAX_MILLIS_PER_PASS = 120_000;\n\nexport interface EmbedPassOptions {\n /**\n * Stop the pass after this many chunks. Unbounded passes are the reason a\n * six-figure backlog starves the indexer: the daemon serialises indexing and\n * embedding against each other, so a pass that runs for hours means nothing\n * new gets indexed for hours. Bounding the pass costs nothing — every\n * embedding is written as it is produced, so the next pass resumes exactly\n * where this one stopped.\n */\n maxChunks?: number;\n /** Stop the pass after this much wall-clock time, whichever bound hits first. */\n maxMillis?: number;\n}\n\n/**\n * Generate and store embeddings for unembedded chunks via the StorageBackend.\n *\n * The pass is deliberately bounded (see EmbedPassOptions) and resumable: it\n * takes a slice of the backlog, embeds it, and returns so the scheduler can run\n * an index pass before coming back. Draining the whole backlog in one call\n * starves indexing for as long as the call runs.\n *\n * The optional `shouldStop` callback is checked between every batch. When it\n * returns true the embed loop exits early so the caller (e.g. the daemon\n * shutdown handler) can close the pool without racing against active queries.\n *\n * Returns the number of newly embedded chunks.\n */\nexport async function embedChunksWithBackend(\n backend: StorageBackend,\n shouldStop?: () => boolean,\n projectNames?: Map<number, string>,\n options?: EmbedPassOptions,\n): Promise<number> {\n const { generateEmbeddings, serializeEmbedding } = await import(\"../embeddings.js\");\n\n const maxChunks = options?.maxChunks ?? DEFAULT_MAX_CHUNKS_PER_PASS;\n const maxMillis = options?.maxMillis ?? DEFAULT_MAX_MILLIS_PER_PASS;\n const deadline = Date.now() + maxMillis;\n\n const rows = await backend.getUnembeddedChunkIds(undefined, maxChunks);\n if (rows.length === 0) return 0;\n\n const total = rows.length;\n let embedded = 0;\n\n // Build a summary of what needs embedding: count chunks per project_id\n const projectChunkCounts = new Map<number, { count: number; samplePath: string }>();\n for (const row of rows) {\n const entry = projectChunkCounts.get(row.project_id);\n if (entry) {\n entry.count++;\n } else {\n projectChunkCounts.set(row.project_id, { count: 1, samplePath: row.path });\n }\n }\n const pName = (pid: number) => projectNames?.get(pid) ?? `project ${pid}`;\n const projectSummary = Array.from(projectChunkCounts.entries())\n .map(([pid, { count, samplePath }]) => ` ${pName(pid)}: ${count} chunks (e.g. ${samplePath})`)\n .join(\"\\n\");\n process.stderr.write(\n `[pai-daemon] Embed pass: ${total} unembedded chunks across ${projectChunkCounts.size} project(s)\\n${projectSummary}\\n`\n );\n\n // Track current project for transition logging\n let currentProjectId = -1;\n let projectEmbedded = 0;\n\n for (let i = 0; i < rows.length; i += EMBED_BATCH_SIZE) {\n // Check cancellation between every batch before touching the pool again\n if (shouldStop?.()) {\n process.stderr.write(\n `[pai-daemon] Embed pass cancelled after ${embedded}/${total} chunks (shutdown requested)\\n`\n );\n break;\n }\n\n // Yield the daemon back to the indexer rather than run past the deadline.\n // The remaining rows are simply picked up by the next scheduled pass.\n if (Date.now() >= deadline) {\n process.stderr.write(\n `[pai-daemon] Embed pass yielding after ${embedded}/${total} chunks (${maxMillis}ms budget spent); resuming next pass\\n`\n );\n break;\n }\n\n const batch = rows.slice(i, i + EMBED_BATCH_SIZE);\n\n // Keep IPC responsive: the forward pass below is synchronous inside the\n // model, so yield before entering it rather than between chunks.\n await yieldToEventLoop();\n\n const vecs = await generateEmbeddings(batch.map((r) => r.text));\n for (let j = 0; j < batch.length; j++) {\n await backend.updateEmbedding(batch[j].id, serializeEmbedding(vecs[j]));\n }\n\n // Attribute the batch to projects only after it is durably stored, and walk\n // it in order so a batch straddling a project boundary credits each side\n // correctly. Rows arrive ordered by project, so a boundary is a real\n // transition, not the per-chunk flapping this logging used to produce.\n for (const { project_id, path } of batch) {\n if (project_id !== currentProjectId) {\n if (currentProjectId !== -1) {\n process.stderr.write(\n `[pai-daemon] Finished ${pName(currentProjectId)}: ${projectEmbedded} chunks embedded\\n`\n );\n }\n const info = projectChunkCounts.get(project_id);\n process.stderr.write(\n `[pai-daemon] Embedding ${pName(project_id)} (${info?.count ?? \"?\"} chunks, starting at ${path})\\n`\n );\n currentProjectId = project_id;\n projectEmbedded = 0;\n }\n projectEmbedded++;\n }\n\n embedded += batch.length;\n\n // Log progress with current file path for context\n const lastChunk = batch[batch.length - 1];\n process.stderr.write(\n `[pai-daemon] Embedded ${embedded}/${total} chunks (${pName(lastChunk.project_id)}: ${lastChunk.path})\\n`\n );\n }\n\n // Log final project completion\n if (currentProjectId !== -1) {\n process.stderr.write(\n `[pai-daemon] Finished ${pName(currentProjectId)}: ${projectEmbedded} chunks embedded\\n`\n );\n }\n\n return embedded;\n}\n\n// ---------------------------------------------------------------------------\n// Global indexing via StorageBackend\n// ---------------------------------------------------------------------------\n\nexport async function indexAllWithBackend(\n backend: StorageBackend,\n registryDb: Database,\n): Promise<{ projects: number; result: IndexResult }> {\n const projects = registryDb\n .prepare(\"SELECT id, root_path, claude_notes_dir FROM projects WHERE status = 'active'\")\n .all() as Array<{ id: number; root_path: string; claude_notes_dir: string | null }>;\n\n const totals: IndexResult = { filesProcessed: 0, chunksCreated: 0, filesSkipped: 0 };\n\n for (const project of projects) {\n await yieldToEventLoop();\n const r = await indexProjectWithBackend(backend, project.id, project.root_path, project.claude_notes_dir);\n totals.filesProcessed += r.filesProcessed;\n totals.chunksCreated += r.chunksCreated;\n totals.filesSkipped += r.filesSkipped;\n }\n\n return { projects: projects.length, result: totals };\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAuCA,eAAsB,qBACpB,SACA,WACA,UACA,cACA,QACA,MACkB;CAClB,MAAM,UAAU,KAAK,UAAU,aAAa;CAE5C,IAAI;CACJ,IAAI;AACJ,KAAI;AACF,YAAU,aAAa,SAAS,OAAO;AACvC,SAAO,SAAS,QAAQ;SAClB;AACN,SAAO;;CAGT,MAAM,OAAO,WAAW,QAAQ;CAChC,MAAM,QAAQ,KAAK,MAAM,KAAK,QAAQ;CACtC,MAAM,OAAO,KAAK;AAIlB,KADqB,MAAM,QAAQ,YAAY,WAAW,aAAa,KAClD,KAAM,QAAO;AAGlC,OAAM,QAAQ,oBAAoB,WAAW,aAAa;CAG1D,MAAM,YAAY,cAAc,QAAQ;CACxC,MAAM,YAAY,KAAK,KAAK;CAE5B,MAAM,SAAqB,UAAU,KAAK,GAAG,OAAO;EAClD,IAAI,QAAQ,WAAW,cAAc,GAAG,EAAE,WAAW,EAAE,QAAQ;EAC/D;EACA;EACA;EACA,MAAM;EACN,WAAW,EAAE;EACb,SAAS,EAAE;EACX,MAAM,EAAE;EACR,MAAM,EAAE;EACR;EACA,WAAW;EACZ,EAAE;AAGH,OAAM,QAAQ,aAAa,OAAO;AAClC,OAAM,QAAQ,WAAW;EAAE;EAAW,MAAM;EAAc;EAAQ;EAAM;EAAM;EAAO;EAAM,CAAC;AAE5F,QAAO;;AAOT,eAAsB,wBACpB,SACA,WACA,UACA,gBACsB;CACtB,MAAM,SAAsB;EAAE,gBAAgB;EAAG,eAAe;EAAG,cAAc;EAAG;CAEpF,MAAM,eAA2F,EAAE;CAEnG,MAAM,eAAe,KAAK,UAAU,YAAY;AAChD,KAAI,WAAW,aAAa,CAC1B,cAAa,KAAK;EAAE,SAAS;EAAc,UAAU;EAAU,QAAQ;EAAU,MAAM;EAAa,CAAC;CAGvG,MAAM,YAAY,KAAK,UAAU,SAAS;AAC1C,MAAK,MAAM,WAAW,YAAY,UAAU,EAAE;EAE5C,MAAM,OAAO,WADG,SAAS,UAAU,QAAQ,CACX;AAChC,eAAa,KAAK;GAAE;GAAS,UAAU;GAAU,QAAQ;GAAU;GAAM,CAAC;;CAG5E,MAAM,WAAW,KAAK,UAAU,QAAQ;AACxC,MAAK,MAAM,WAAW,YAAY,SAAS,CACzC,cAAa,KAAK;EAAE;EAAS,UAAU;EAAU,QAAQ;EAAS,MAAM;EAAW,CAAC;CAItF;EACE,MAAM,YAAY,KAAK,KAAK;AAC5B,OAAK,MAAM,WAAW,YAAY,SAAS,EAAE;GAE3C,MAAM,OAAO,uBADI,SAAS,QAAQ,CACW;AAC7C,OAAI,CAAC,KAAM;GAEX,MAAM,gBAAgB,GADN,SAAS,UAAU,QAAQ,CACV;GAGjC,MAAM,aAAuB;IAC3B,IAHS,QAAQ,WAAW,eAAe,GAAG,GAAG,EAAE;IAG/C;IAAW,QAAQ;IAAS,MAAM;IACtC,MAAM;IAAe,WAAW;IAAG,SAAS;IAC5C,MAJW,WAAW,KAAK;IAIrB;IAAM;IAAW,WAAW;IACnC;AACD,OAAI;AACF,UAAM,QAAQ,aAAa,CAAC,WAAW,CAAC;WAClC;;;AAMZ,KAAI,CAAC,6BAA6B,SAAS,CACzC,MAAK,MAAM,WAAW,iBAAiB,SAAS,CAC9C,cAAa,KAAK;EAAE;EAAS,UAAU;EAAU,QAAQ;EAAW,MAAM;EAAS,CAAC;AAIxF,KAAI,kBAAkB,mBAAmB,UAAU;AACjD,OAAK,MAAM,WAAW,YAAY,eAAe,CAC/C,cAAa,KAAK;GAAE;GAAS,UAAU;GAAgB,QAAQ;GAAS,MAAM;GAAW,CAAC;EAI5F;GACE,MAAM,YAAY,KAAK,KAAK;AAC5B,QAAK,MAAM,WAAW,YAAY,eAAe,EAAE;IAEjD,MAAM,OAAO,uBADI,SAAS,QAAQ,CACW;AAC7C,QAAI,CAAC,KAAM;IAEX,MAAM,gBAAgB,GADN,SAAS,gBAAgB,QAAQ,CAChB;IAGjC,MAAM,aAAuB;KAC3B,IAHS,QAAQ,WAAW,eAAe,GAAG,GAAG,EAAE;KAG/C;KAAW,QAAQ;KAAS,MAAM;KACtC,MAAM;KAAe,WAAW;KAAG,SAAS;KAC5C,MAJW,WAAW,KAAK;KAIrB;KAAM;KAAW,WAAW;KACnC;AACD,QAAI;AACF,WAAM,QAAQ,aAAa,CAAC,WAAW,CAAC;YAClC;;;AAMZ,MAAI,eAAe,SAAS,SAAS,EAAE;GACrC,MAAM,mBAAmB,eAAe,MAAM,GAAG,GAAiB;GAClE,MAAM,iBAAiB,KAAK,kBAAkB,YAAY;AAC1D,OAAI,WAAW,eAAe,CAC5B,cAAa,KAAK;IAAE,SAAS;IAAgB,UAAU;IAAkB,QAAQ;IAAU,MAAM;IAAa,CAAC;GAEjH,MAAM,kBAAkB,KAAK,kBAAkB,SAAS;AACxD,QAAK,MAAM,WAAW,YAAY,gBAAgB,EAAE;IAElD,MAAM,OAAO,WADG,SAAS,kBAAkB,QAAQ,CACnB;AAChC,iBAAa,KAAK;KAAE;KAAS,UAAU;KAAkB,QAAQ;KAAU;KAAM,CAAC;;;;AAKxF,OAAM,kBAAkB;CAExB,IAAI,kBAAkB;AAEtB,MAAK,MAAM,EAAE,SAAS,UAAU,QAAQ,UAAU,cAAc;AAC9D,MAAI,mBAAmB,mBAAmB;AACxC,SAAM,kBAAkB;AACxB,qBAAkB;;AAEpB;EAEA,MAAM,UAAU,SAAS,UAAU,QAAQ;AAC3C,MAAI;AAGF,OAFgB,MAAM,qBAAqB,SAAS,WAAW,UAAU,SAAS,QAAQ,KAAK,EAElF;IACX,MAAM,MAAM,MAAM,QAAQ,YAAY,WAAW,QAAQ;AACzD,WAAO;AACP,WAAO,iBAAiB,IAAI;SAE5B,QAAO;UAEH;AAEN,UAAO;;;CAKX,MAAM,4BAAY,IAAI,KAAa;AACnC,MAAK,MAAM,EAAE,SAAS,cAAc,aAClC,WAAU,IAAI,SAAS,UAAU,QAAQ,CAAC;CAG5C,MAAM,eAAe,MAAM,QAAQ,sBAAsB,UAAU;CAEnE,MAAM,aAAuB,EAAE;AAC/B,MAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,WAAW,EAAE,SAAS,UAAU,GAAG,EAAE,MAAM,GAAG,GAAkB,GAAG;AACzE,MAAI,CAAC,UAAU,IAAI,SAAS,CAC1B,YAAW,KAAK,EAAE;;AAItB,KAAI,WAAW,SAAS,EACtB,OAAM,QAAQ,YAAY,WAAW,WAAW;AAGlD,QAAO;;AAOT,MAAM,mBAAmB;;AAIzB,MAAM,8BAA8B;;AAEpC,MAAM,8BAA8B;;;;;;;;;;;;;;;AA8BpC,eAAsB,uBACpB,SACA,YACA,cACA,SACiB;CACjB,MAAM,EAAE,oBAAoB,uBAAuB,MAAM,OAAO;CAEhE,MAAM,YAAY,SAAS,aAAa;CACxC,MAAM,YAAY,SAAS,aAAa;CACxC,MAAM,WAAW,KAAK,KAAK,GAAG;CAE9B,MAAM,OAAO,MAAM,QAAQ,sBAAsB,QAAW,UAAU;AACtE,KAAI,KAAK,WAAW,EAAG,QAAO;CAE9B,MAAM,QAAQ,KAAK;CACnB,IAAI,WAAW;CAGf,MAAM,qCAAqB,IAAI,KAAoD;AACnF,MAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,mBAAmB,IAAI,IAAI,WAAW;AACpD,MAAI,MACF,OAAM;MAEN,oBAAmB,IAAI,IAAI,YAAY;GAAE,OAAO;GAAG,YAAY,IAAI;GAAM,CAAC;;CAG9E,MAAM,SAAS,QAAgB,cAAc,IAAI,IAAI,IAAI,WAAW;CACpE,MAAM,iBAAiB,MAAM,KAAK,mBAAmB,SAAS,CAAC,CAC5D,KAAK,CAAC,KAAK,EAAE,OAAO,kBAAkB,KAAK,MAAM,IAAI,CAAC,IAAI,MAAM,gBAAgB,WAAW,GAAG,CAC9F,KAAK,KAAK;AACb,SAAQ,OAAO,MACb,4BAA4B,MAAM,4BAA4B,mBAAmB,KAAK,eAAe,eAAe,IACrH;CAGD,IAAI,mBAAmB;CACvB,IAAI,kBAAkB;AAEtB,MAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK,kBAAkB;AAEtD,MAAI,cAAc,EAAE;AAClB,WAAQ,OAAO,MACb,2CAA2C,SAAS,GAAG,MAAM,gCAC9D;AACD;;AAKF,MAAI,KAAK,KAAK,IAAI,UAAU;AAC1B,WAAQ,OAAO,MACb,0CAA0C,SAAS,GAAG,MAAM,WAAW,UAAU,wCAClF;AACD;;EAGF,MAAM,QAAQ,KAAK,MAAM,GAAG,IAAI,iBAAiB;AAIjD,QAAM,kBAAkB;EAExB,MAAM,OAAO,MAAM,mBAAmB,MAAM,KAAK,MAAM,EAAE,KAAK,CAAC;AAC/D,OAAK,IAAI,IAAI,GAAG,IAAI,MAAM,QAAQ,IAChC,OAAM,QAAQ,gBAAgB,MAAM,GAAG,IAAI,mBAAmB,KAAK,GAAG,CAAC;AAOzE,OAAK,MAAM,EAAE,YAAY,UAAU,OAAO;AACxC,OAAI,eAAe,kBAAkB;AACnC,QAAI,qBAAqB,GACvB,SAAQ,OAAO,MACb,yBAAyB,MAAM,iBAAiB,CAAC,IAAI,gBAAgB,oBACtE;IAEH,MAAM,OAAO,mBAAmB,IAAI,WAAW;AAC/C,YAAQ,OAAO,MACb,0BAA0B,MAAM,WAAW,CAAC,IAAI,MAAM,SAAS,IAAI,uBAAuB,KAAK,KAChG;AACD,uBAAmB;AACnB,sBAAkB;;AAEpB;;AAGF,cAAY,MAAM;EAGlB,MAAM,YAAY,MAAM,MAAM,SAAS;AACvC,UAAQ,OAAO,MACb,yBAAyB,SAAS,GAAG,MAAM,WAAW,MAAM,UAAU,WAAW,CAAC,IAAI,UAAU,KAAK,KACtG;;AAIH,KAAI,qBAAqB,GACvB,SAAQ,OAAO,MACb,yBAAyB,MAAM,iBAAiB,CAAC,IAAI,gBAAgB,oBACtE;AAGH,QAAO;;AAOT,eAAsB,oBACpB,SACA,YACoD;CACpD,MAAM,WAAW,WACd,QAAQ,+EAA+E,CACvF,KAAK;CAER,MAAM,SAAsB;EAAE,gBAAgB;EAAG,eAAe;EAAG,cAAc;EAAG;AAEpF,MAAK,MAAM,WAAW,UAAU;AAC9B,QAAM,kBAAkB;EACxB,MAAM,IAAI,MAAM,wBAAwB,SAAS,QAAQ,IAAI,QAAQ,WAAW,QAAQ,iBAAiB;AACzG,SAAO,kBAAkB,EAAE;AAC3B,SAAO,iBAAiB,EAAE;AAC1B,SAAO,gBAAgB,EAAE;;AAG3B,QAAO;EAAE,UAAU,SAAS;EAAQ,QAAQ;EAAQ"}
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
import { a as parseSessionTitleChunk, c as yieldToEventLoop, f as sha256File, i as isPathTooBroadForContentScan, l as chunkMarkdown, n as chunkId, o as walkContentFiles, r as detectTier, s as walkMdFiles, t as INDEX_YIELD_EVERY } from "./helpers-crDEr6S2.mjs";
|
|
2
|
+
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
3
|
+
import { basename, join, relative } from "node:path";
|
|
4
|
+
|
|
5
|
+
//#region src/memory/indexer/async.ts
|
|
6
|
+
/**
|
|
7
|
+
* Backend-aware async indexer for PAI federation memory.
|
|
8
|
+
*
|
|
9
|
+
* Provides the same functionality as sync.ts but writes through the
|
|
10
|
+
* StorageBackend interface instead of directly to better-sqlite3.
|
|
11
|
+
* Used when the daemon is configured with the Postgres backend.
|
|
12
|
+
*
|
|
13
|
+
* The SQLite path still uses sync.ts directly (which is faster for SQLite
|
|
14
|
+
* due to synchronous transactions).
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* Index a single file through the StorageBackend interface.
|
|
18
|
+
* Returns true if the file was re-indexed (changed or new), false if skipped.
|
|
19
|
+
*/
|
|
20
|
+
async function indexFileWithBackend(backend, projectId, rootPath, relativePath, source, tier) {
|
|
21
|
+
const absPath = join(rootPath, relativePath);
|
|
22
|
+
let content;
|
|
23
|
+
let stat;
|
|
24
|
+
try {
|
|
25
|
+
content = readFileSync(absPath, "utf8");
|
|
26
|
+
stat = statSync(absPath);
|
|
27
|
+
} catch {
|
|
28
|
+
return false;
|
|
29
|
+
}
|
|
30
|
+
const hash = sha256File(content);
|
|
31
|
+
const mtime = Math.floor(stat.mtimeMs);
|
|
32
|
+
const size = stat.size;
|
|
33
|
+
if (await backend.getFileHash(projectId, relativePath) === hash) return false;
|
|
34
|
+
await backend.deleteChunksForFile(projectId, relativePath);
|
|
35
|
+
const rawChunks = chunkMarkdown(content);
|
|
36
|
+
const updatedAt = Date.now();
|
|
37
|
+
const chunks = rawChunks.map((c, i) => ({
|
|
38
|
+
id: chunkId(projectId, relativePath, i, c.startLine, c.endLine),
|
|
39
|
+
projectId,
|
|
40
|
+
source,
|
|
41
|
+
tier,
|
|
42
|
+
path: relativePath,
|
|
43
|
+
startLine: c.startLine,
|
|
44
|
+
endLine: c.endLine,
|
|
45
|
+
hash: c.hash,
|
|
46
|
+
text: c.text,
|
|
47
|
+
updatedAt,
|
|
48
|
+
embedding: null
|
|
49
|
+
}));
|
|
50
|
+
await backend.insertChunks(chunks);
|
|
51
|
+
await backend.upsertFile({
|
|
52
|
+
projectId,
|
|
53
|
+
path: relativePath,
|
|
54
|
+
source,
|
|
55
|
+
tier,
|
|
56
|
+
hash,
|
|
57
|
+
mtime,
|
|
58
|
+
size
|
|
59
|
+
});
|
|
60
|
+
return true;
|
|
61
|
+
}
|
|
62
|
+
async function indexProjectWithBackend(backend, projectId, rootPath, claudeNotesDir) {
|
|
63
|
+
const result = {
|
|
64
|
+
filesProcessed: 0,
|
|
65
|
+
chunksCreated: 0,
|
|
66
|
+
filesSkipped: 0
|
|
67
|
+
};
|
|
68
|
+
const filesToIndex = [];
|
|
69
|
+
const rootMemoryMd = join(rootPath, "MEMORY.md");
|
|
70
|
+
if (existsSync(rootMemoryMd)) filesToIndex.push({
|
|
71
|
+
absPath: rootMemoryMd,
|
|
72
|
+
rootBase: rootPath,
|
|
73
|
+
source: "memory",
|
|
74
|
+
tier: "evergreen"
|
|
75
|
+
});
|
|
76
|
+
const memoryDir = join(rootPath, "memory");
|
|
77
|
+
for (const absPath of walkMdFiles(memoryDir)) {
|
|
78
|
+
const tier = detectTier(relative(rootPath, absPath));
|
|
79
|
+
filesToIndex.push({
|
|
80
|
+
absPath,
|
|
81
|
+
rootBase: rootPath,
|
|
82
|
+
source: "memory",
|
|
83
|
+
tier
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
const notesDir = join(rootPath, "Notes");
|
|
87
|
+
for (const absPath of walkMdFiles(notesDir)) filesToIndex.push({
|
|
88
|
+
absPath,
|
|
89
|
+
rootBase: rootPath,
|
|
90
|
+
source: "notes",
|
|
91
|
+
tier: "session"
|
|
92
|
+
});
|
|
93
|
+
{
|
|
94
|
+
const updatedAt = Date.now();
|
|
95
|
+
for (const absPath of walkMdFiles(notesDir)) {
|
|
96
|
+
const text = parseSessionTitleChunk(basename(absPath));
|
|
97
|
+
if (!text) continue;
|
|
98
|
+
const syntheticPath = `${relative(rootPath, absPath)}::title`;
|
|
99
|
+
const titleChunk = {
|
|
100
|
+
id: chunkId(projectId, syntheticPath, 0, 0, 0),
|
|
101
|
+
projectId,
|
|
102
|
+
source: "notes",
|
|
103
|
+
tier: "session",
|
|
104
|
+
path: syntheticPath,
|
|
105
|
+
startLine: 0,
|
|
106
|
+
endLine: 0,
|
|
107
|
+
hash: sha256File(text),
|
|
108
|
+
text,
|
|
109
|
+
updatedAt,
|
|
110
|
+
embedding: null
|
|
111
|
+
};
|
|
112
|
+
try {
|
|
113
|
+
await backend.insertChunks([titleChunk]);
|
|
114
|
+
} catch {}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
if (!isPathTooBroadForContentScan(rootPath)) for (const absPath of walkContentFiles(rootPath)) filesToIndex.push({
|
|
118
|
+
absPath,
|
|
119
|
+
rootBase: rootPath,
|
|
120
|
+
source: "content",
|
|
121
|
+
tier: "topic"
|
|
122
|
+
});
|
|
123
|
+
if (claudeNotesDir && claudeNotesDir !== notesDir) {
|
|
124
|
+
for (const absPath of walkMdFiles(claudeNotesDir)) filesToIndex.push({
|
|
125
|
+
absPath,
|
|
126
|
+
rootBase: claudeNotesDir,
|
|
127
|
+
source: "notes",
|
|
128
|
+
tier: "session"
|
|
129
|
+
});
|
|
130
|
+
{
|
|
131
|
+
const updatedAt = Date.now();
|
|
132
|
+
for (const absPath of walkMdFiles(claudeNotesDir)) {
|
|
133
|
+
const text = parseSessionTitleChunk(basename(absPath));
|
|
134
|
+
if (!text) continue;
|
|
135
|
+
const syntheticPath = `${relative(claudeNotesDir, absPath)}::title`;
|
|
136
|
+
const titleChunk = {
|
|
137
|
+
id: chunkId(projectId, syntheticPath, 0, 0, 0),
|
|
138
|
+
projectId,
|
|
139
|
+
source: "notes",
|
|
140
|
+
tier: "session",
|
|
141
|
+
path: syntheticPath,
|
|
142
|
+
startLine: 0,
|
|
143
|
+
endLine: 0,
|
|
144
|
+
hash: sha256File(text),
|
|
145
|
+
text,
|
|
146
|
+
updatedAt,
|
|
147
|
+
embedding: null
|
|
148
|
+
};
|
|
149
|
+
try {
|
|
150
|
+
await backend.insertChunks([titleChunk]);
|
|
151
|
+
} catch {}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
if (claudeNotesDir.endsWith("/Notes")) {
|
|
155
|
+
const claudeProjectDir = claudeNotesDir.slice(0, -6);
|
|
156
|
+
const claudeMemoryMd = join(claudeProjectDir, "MEMORY.md");
|
|
157
|
+
if (existsSync(claudeMemoryMd)) filesToIndex.push({
|
|
158
|
+
absPath: claudeMemoryMd,
|
|
159
|
+
rootBase: claudeProjectDir,
|
|
160
|
+
source: "memory",
|
|
161
|
+
tier: "evergreen"
|
|
162
|
+
});
|
|
163
|
+
const claudeMemoryDir = join(claudeProjectDir, "memory");
|
|
164
|
+
for (const absPath of walkMdFiles(claudeMemoryDir)) {
|
|
165
|
+
const tier = detectTier(relative(claudeProjectDir, absPath));
|
|
166
|
+
filesToIndex.push({
|
|
167
|
+
absPath,
|
|
168
|
+
rootBase: claudeProjectDir,
|
|
169
|
+
source: "memory",
|
|
170
|
+
tier
|
|
171
|
+
});
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
await yieldToEventLoop();
|
|
176
|
+
let filesSinceYield = 0;
|
|
177
|
+
for (const { absPath, rootBase, source, tier } of filesToIndex) {
|
|
178
|
+
if (filesSinceYield >= INDEX_YIELD_EVERY) {
|
|
179
|
+
await yieldToEventLoop();
|
|
180
|
+
filesSinceYield = 0;
|
|
181
|
+
}
|
|
182
|
+
filesSinceYield++;
|
|
183
|
+
const relPath = relative(rootBase, absPath);
|
|
184
|
+
try {
|
|
185
|
+
if (await indexFileWithBackend(backend, projectId, rootBase, relPath, source, tier)) {
|
|
186
|
+
const ids = await backend.getChunkIds(projectId, relPath);
|
|
187
|
+
result.filesProcessed++;
|
|
188
|
+
result.chunksCreated += ids.length;
|
|
189
|
+
} else result.filesSkipped++;
|
|
190
|
+
} catch {
|
|
191
|
+
result.filesSkipped++;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
const livePaths = /* @__PURE__ */ new Set();
|
|
195
|
+
for (const { absPath, rootBase } of filesToIndex) livePaths.add(relative(rootBase, absPath));
|
|
196
|
+
const dbChunkPaths = await backend.getDistinctChunkPaths(projectId);
|
|
197
|
+
const stalePaths = [];
|
|
198
|
+
for (const p of dbChunkPaths) {
|
|
199
|
+
const basePath = p.endsWith("::title") ? p.slice(0, -7) : p;
|
|
200
|
+
if (!livePaths.has(basePath)) stalePaths.push(p);
|
|
201
|
+
}
|
|
202
|
+
if (stalePaths.length > 0) await backend.deletePaths(projectId, stalePaths);
|
|
203
|
+
return result;
|
|
204
|
+
}
|
|
205
|
+
const EMBED_BATCH_SIZE = 50;
|
|
206
|
+
/** Default ceiling on one pass. See EmbedPassOptions. */
|
|
207
|
+
const DEFAULT_MAX_CHUNKS_PER_PASS = 5e3;
|
|
208
|
+
/** Default wall-clock ceiling on one pass, in milliseconds. */
|
|
209
|
+
const DEFAULT_MAX_MILLIS_PER_PASS = 12e4;
|
|
210
|
+
/**
|
|
211
|
+
* Generate and store embeddings for unembedded chunks via the StorageBackend.
|
|
212
|
+
*
|
|
213
|
+
* The pass is deliberately bounded (see EmbedPassOptions) and resumable: it
|
|
214
|
+
* takes a slice of the backlog, embeds it, and returns so the scheduler can run
|
|
215
|
+
* an index pass before coming back. Draining the whole backlog in one call
|
|
216
|
+
* starves indexing for as long as the call runs.
|
|
217
|
+
*
|
|
218
|
+
* The optional `shouldStop` callback is checked between every batch. When it
|
|
219
|
+
* returns true the embed loop exits early so the caller (e.g. the daemon
|
|
220
|
+
* shutdown handler) can close the pool without racing against active queries.
|
|
221
|
+
*
|
|
222
|
+
* Returns the number of newly embedded chunks.
|
|
223
|
+
*/
|
|
224
|
+
async function embedChunksWithBackend(backend, shouldStop, projectNames, options) {
|
|
225
|
+
const { generateEmbeddings, serializeEmbedding } = await import("./embeddings-Bn86ssxR.mjs").then((n) => n.i);
|
|
226
|
+
const maxChunks = options?.maxChunks ?? DEFAULT_MAX_CHUNKS_PER_PASS;
|
|
227
|
+
const maxMillis = options?.maxMillis ?? DEFAULT_MAX_MILLIS_PER_PASS;
|
|
228
|
+
const deadline = Date.now() + maxMillis;
|
|
229
|
+
const rows = await backend.getUnembeddedChunkIds(void 0, maxChunks);
|
|
230
|
+
if (rows.length === 0) return 0;
|
|
231
|
+
const total = rows.length;
|
|
232
|
+
let embedded = 0;
|
|
233
|
+
const projectChunkCounts = /* @__PURE__ */ new Map();
|
|
234
|
+
for (const row of rows) {
|
|
235
|
+
const entry = projectChunkCounts.get(row.project_id);
|
|
236
|
+
if (entry) entry.count++;
|
|
237
|
+
else projectChunkCounts.set(row.project_id, {
|
|
238
|
+
count: 1,
|
|
239
|
+
samplePath: row.path
|
|
240
|
+
});
|
|
241
|
+
}
|
|
242
|
+
const pName = (pid) => projectNames?.get(pid) ?? `project ${pid}`;
|
|
243
|
+
const projectSummary = Array.from(projectChunkCounts.entries()).map(([pid, { count, samplePath }]) => ` ${pName(pid)}: ${count} chunks (e.g. ${samplePath})`).join("\n");
|
|
244
|
+
process.stderr.write(`[pai-daemon] Embed pass: ${total} unembedded chunks across ${projectChunkCounts.size} project(s)\n${projectSummary}\n`);
|
|
245
|
+
let currentProjectId = -1;
|
|
246
|
+
let projectEmbedded = 0;
|
|
247
|
+
for (let i = 0; i < rows.length; i += EMBED_BATCH_SIZE) {
|
|
248
|
+
if (shouldStop?.()) {
|
|
249
|
+
process.stderr.write(`[pai-daemon] Embed pass cancelled after ${embedded}/${total} chunks (shutdown requested)\n`);
|
|
250
|
+
break;
|
|
251
|
+
}
|
|
252
|
+
if (Date.now() >= deadline) {
|
|
253
|
+
process.stderr.write(`[pai-daemon] Embed pass yielding after ${embedded}/${total} chunks (${maxMillis}ms budget spent); resuming next pass\n`);
|
|
254
|
+
break;
|
|
255
|
+
}
|
|
256
|
+
const batch = rows.slice(i, i + EMBED_BATCH_SIZE);
|
|
257
|
+
await yieldToEventLoop();
|
|
258
|
+
const vecs = await generateEmbeddings(batch.map((r) => r.text));
|
|
259
|
+
await Promise.all(batch.map((row, j) => backend.updateEmbedding(row.id, serializeEmbedding(vecs[j]))));
|
|
260
|
+
for (const { project_id, path } of batch) {
|
|
261
|
+
if (project_id !== currentProjectId) {
|
|
262
|
+
if (currentProjectId !== -1) process.stderr.write(`[pai-daemon] Finished ${pName(currentProjectId)}: ${projectEmbedded} chunks embedded\n`);
|
|
263
|
+
const info = projectChunkCounts.get(project_id);
|
|
264
|
+
process.stderr.write(`[pai-daemon] Embedding ${pName(project_id)} (${info?.count ?? "?"} chunks, starting at ${path})\n`);
|
|
265
|
+
currentProjectId = project_id;
|
|
266
|
+
projectEmbedded = 0;
|
|
267
|
+
}
|
|
268
|
+
projectEmbedded++;
|
|
269
|
+
}
|
|
270
|
+
embedded += batch.length;
|
|
271
|
+
const lastChunk = batch[batch.length - 1];
|
|
272
|
+
process.stderr.write(`[pai-daemon] Embedded ${embedded}/${total} chunks (${pName(lastChunk.project_id)}: ${lastChunk.path})\n`);
|
|
273
|
+
}
|
|
274
|
+
if (currentProjectId !== -1) process.stderr.write(`[pai-daemon] Finished ${pName(currentProjectId)}: ${projectEmbedded} chunks embedded\n`);
|
|
275
|
+
return embedded;
|
|
276
|
+
}
|
|
277
|
+
async function indexAllWithBackend(backend, registryDb) {
|
|
278
|
+
const projects = registryDb.prepare("SELECT id, root_path, claude_notes_dir FROM projects WHERE status = 'active'").all();
|
|
279
|
+
const totals = {
|
|
280
|
+
filesProcessed: 0,
|
|
281
|
+
chunksCreated: 0,
|
|
282
|
+
filesSkipped: 0
|
|
283
|
+
};
|
|
284
|
+
for (const project of projects) {
|
|
285
|
+
await yieldToEventLoop();
|
|
286
|
+
const r = await indexProjectWithBackend(backend, project.id, project.root_path, project.claude_notes_dir);
|
|
287
|
+
totals.filesProcessed += r.filesProcessed;
|
|
288
|
+
totals.chunksCreated += r.chunksCreated;
|
|
289
|
+
totals.filesSkipped += r.filesSkipped;
|
|
290
|
+
}
|
|
291
|
+
return {
|
|
292
|
+
projects: projects.length,
|
|
293
|
+
result: totals
|
|
294
|
+
};
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
//#endregion
|
|
298
|
+
export { embedChunksWithBackend, indexAllWithBackend };
|
|
299
|
+
//# sourceMappingURL=indexer-backend-COrJ5rjS.mjs.map
|