thincoder 0.12.61 → 0.12.63
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +34 -1
- package/README.md +13 -12
- package/bin/thincoder.mjs +63 -28
- package/package.json +6 -5
- package/src/acp/bridge.mjs +35 -15
- package/src/acp/client-caps.mjs +86 -0
- package/src/acp/ext.mjs +86 -0
- package/src/acp/handlers-session.mjs +240 -0
- package/src/acp/handlers-slots.mjs +196 -0
- package/src/acp/login.mjs +48 -0
- package/src/acp/session.mjs +6 -4
- package/src/acp.mjs +67 -371
- package/src/cli/distill-command.mjs +3 -3
- package/src/cli/make-agent.mjs +59 -17
- package/src/cli/memory-command.mjs +3 -3
- package/src/cli/permission.mjs +4 -48
- package/src/cli/setup-wizard.mjs +1 -1
- package/src/completions.mjs +3 -1
- package/src/crash-reports.mjs +32 -10
- package/src/distill.mjs +4 -4
- package/src/heap-watch.mjs +88 -0
- package/src/prompt-injections.mjs +20 -0
- package/src/tui/agent-turn.mjs +40 -9
- package/src/tui/cmd-advisor.mjs +5 -5
- package/src/tui/cmd-clear.mjs +2 -0
- package/src/tui/cmd-config.mjs +8 -8
- package/src/tui/cmd-eng.mjs +25 -9
- package/src/tui/cmd-mcp.mjs +9 -8
- package/src/tui/cmd-model.mjs +1 -1
- package/src/tui/cmd-new.mjs +10 -5
- package/src/tui/cmd-reindex.mjs +1 -1
- package/src/tui/cmd-restore.mjs +2 -2
- package/src/tui/cmd-session.mjs +31 -4
- package/src/tui/cmd-skills.mjs +1 -1
- package/src/tui/cmd-think.mjs +22 -9
- package/src/tui/config-helpers.mjs +1 -1
- package/src/tui/display-budget.mjs +206 -0
- package/src/tui/index.mjs +40 -12
- package/src/tui/interaction.mjs +16 -7
- package/src/tui/key-handler-search.mjs +9 -1
- package/src/tui/key-modes.mjs +9 -4
- package/src/tui/ledger-surface.mjs +85 -0
- package/src/tui/model-catalog.mjs +4 -4
- package/src/tui/model-picker.mjs +8 -7
- package/src/tui/mouse.mjs +11 -6
- package/src/tui/pickers.mjs +15 -2
- package/src/tui/render-conversation.mjs +1 -1
- package/src/tui/render-frame.mjs +17 -6
- package/src/tui/render-loop.mjs +1 -1
- package/src/tui/render-segments.mjs +3 -1
- package/src/tui/slash-commands.mjs +1 -1
- package/src/tui/startup.mjs +49 -17
- package/src/tui/subagent-blocks.mjs +21 -3
- package/src/tui/subagent-children.mjs +86 -14
- package/src/tui/subagent-freeze.mjs +80 -3
- package/src/tui/suspension-drive.mjs +48 -22
- package/src/tui/tool-args.mjs +5 -2
- package/src/tui/tool-display.mjs +16 -2
- package/src/tui/tool-events.mjs +64 -22
- package/src/tui/tui-lifecycle.mjs +9 -2
- package/src/tui/wizard.mjs +3 -3
- package/src/tui/wrapped-spawn.mjs +21 -5
- package/src/abort-provenance.mjs +0 -116
- package/src/advisor/citations.mjs +0 -139
- package/src/advisor/compaction.mjs +0 -174
- package/src/advisor/convergence.mjs +0 -80
- package/src/advisor/history.mjs +0 -77
- package/src/advisor/loop.mjs +0 -293
- package/src/advisor/messages.mjs +0 -299
- package/src/advisor/project-context.mjs +0 -194
- package/src/advisor/repos.mjs +0 -150
- package/src/advisor/run.mjs +0 -293
- package/src/advisor/truncate.mjs +0 -57
- package/src/advisor.mjs +0 -290
- package/src/agent/completion.mjs +0 -146
- package/src/agent/dispatch.mjs +0 -489
- package/src/agent/helpers.mjs +0 -384
- package/src/agent/post-turn.mjs +0 -70
- package/src/agent/record-results.mjs +0 -174
- package/src/agent/relay-prefix.mjs +0 -39
- package/src/agent/run-stages.mjs +0 -242
- package/src/agent/setup-reminders.mjs +0 -69
- package/src/agent/setup.mjs +0 -354
- package/src/agent/spawn-child.mjs +0 -228
- package/src/agent-tools/advisor-async.mjs +0 -346
- package/src/agent-tools/advisor-settle.mjs +0 -231
- package/src/agent-tools/advisor.mjs +0 -260
- package/src/agent-tools/async-settle.mjs +0 -191
- package/src/agent-tools/batch-segment.mjs +0 -195
- package/src/agent-tools/consult.mjs +0 -468
- package/src/agent-tools/design-token.mjs +0 -117
- package/src/agent-tools/digest-budget.mjs +0 -76
- package/src/agent-tools/eng.mjs +0 -67
- package/src/agent-tools/escalate-async.mjs +0 -289
- package/src/agent-tools/goal.mjs +0 -119
- package/src/agent-tools/plan.mjs +0 -81
- package/src/agent-tools/read-history.mjs +0 -294
- package/src/agent-tools/recent-changes.mjs +0 -24
- package/src/agent-tools/review-streak.mjs +0 -93
- package/src/agent-tools/settings.mjs +0 -265
- package/src/agent-tools/skill.mjs +0 -47
- package/src/agent-tools/subagent-actions.mjs +0 -479
- package/src/agent-tools/subagent-async.mjs +0 -434
- package/src/agent-tools/subagent-panel.mjs +0 -160
- package/src/agent-tools/subagent-run.mjs +0 -205
- package/src/agent-tools/subagent-scheduler.mjs +0 -392
- package/src/agent-tools/subagent-spawn.mjs +0 -453
- package/src/agent-tools/subagent.mjs +0 -404
- package/src/agent-tools/task.mjs +0 -87
- package/src/agent-tools/timer.mjs +0 -46
- package/src/agent-tools/verify.mjs +0 -271
- package/src/agent-tools.mjs +0 -17
- package/src/agent.mjs +0 -413
- package/src/auto-think.mjs +0 -115
- package/src/config-migrate.mjs +0 -70
- package/src/config.mjs +0 -496
- package/src/context.mjs +0 -381
- package/src/conventions.mjs +0 -223
- package/src/embedding.mjs +0 -120
- package/src/escape.mjs +0 -152
- package/src/expand-home.mjs +0 -16
- package/src/explore-distill.mjs +0 -155
- package/src/generate-title.mjs +0 -83
- package/src/git/checkpoint.mjs +0 -448
- package/src/git/gitmem.mjs +0 -100
- package/src/hooks.mjs +0 -97
- package/src/log.mjs +0 -195
- package/src/markdown.mjs +0 -106
- package/src/mcp/helpers.mjs +0 -51
- package/src/mcp/transport-http.mjs +0 -248
- package/src/mcp/transport-stdio.mjs +0 -140
- package/src/mcp/transport-ws.mjs +0 -122
- package/src/mcp.mjs +0 -295
- package/src/memory/code-index.mjs +0 -219
- package/src/memory/code-sync.mjs +0 -413
- package/src/memory/core.mjs +0 -300
- package/src/memory/delete.mjs +0 -236
- package/src/memory/docs.mjs +0 -417
- package/src/memory/file-walk.mjs +0 -109
- package/src/memory/schema.mjs +0 -452
- package/src/memory.mjs +0 -21
- package/src/model-ref.mjs +0 -66
- package/src/model-specs.mjs +0 -179
- package/src/peer-domains.mjs +0 -265
- package/src/peer-instances.mjs +0 -231
- package/src/prompt-overlays.mjs +0 -82
- package/src/prompts/advisor-design.md +0 -41
- package/src/prompts/advisor-round1.md +0 -41
- package/src/prompts/advisor-round2.md +0 -46
- package/src/prompts/advisor-round3.md +0 -42
- package/src/prompts/common.md +0 -115
- package/src/prompts/consult-base.md +0 -19
- package/src/prompts/discipline-engineering.md +0 -217
- package/src/prompts/discipline-normal.md +0 -179
- package/src/prompts/persona-coder.md +0 -21
- package/src/prompts/persona-eng-coder.md +0 -37
- package/src/prompts/persona-eng-designer.md +0 -55
- package/src/prompts/persona-engineering.md +0 -54
- package/src/prompts/persona-explore.md +0 -15
- package/src/prompts/persona-normal.md +0 -27
- package/src/prompts/persona-plan.md +0 -26
- package/src/provider/anthropic.mjs +0 -225
- package/src/provider/core.mjs +0 -476
- package/src/provider/errors.mjs +0 -101
- package/src/provider/google.mjs +0 -257
- package/src/provider/index.mjs +0 -7
- package/src/provider/list-models.mjs +0 -93
- package/src/provider/normalize.mjs +0 -81
- package/src/provider/rate.mjs +0 -108
- package/src/provider/responses.mjs +0 -495
- package/src/provider/retry.mjs +0 -88
- package/src/provider/sse.mjs +0 -264
- package/src/proxy.mjs +0 -261
- package/src/rules.mjs +0 -53
- package/src/session-gc.mjs +0 -214
- package/src/session-guard.mjs +0 -47
- package/src/session-migrate.mjs +0 -48
- package/src/session-rename.mjs +0 -38
- package/src/session-slots.mjs +0 -489
- package/src/session.mjs +0 -475
- package/src/skills.mjs +0 -153
- package/src/token-ttl.mjs +0 -274
- package/src/tools/apply_patch.md +0 -15
- package/src/tools/bash.md +0 -37
- package/src/tools/bash.mjs +0 -268
- package/src/tools/checklist-sync.mjs +0 -181
- package/src/tools/checklist.md +0 -13
- package/src/tools/checklist.mjs +0 -299
- package/src/tools/delete.md +0 -13
- package/src/tools/edit-batch.mjs +0 -191
- package/src/tools/edit-diff.mjs +0 -348
- package/src/tools/edit.md +0 -30
- package/src/tools/execute.md +0 -21
- package/src/tools/execute.mjs +0 -228
- package/src/tools/fetch.md +0 -12
- package/src/tools/file.mjs +0 -469
- package/src/tools/file_ops.md +0 -17
- package/src/tools/get_current_time.md +0 -8
- package/src/tools/git-checkpoint.mjs +0 -143
- package/src/tools/git-ext.mjs +0 -173
- package/src/tools/git.md +0 -54
- package/src/tools/git.mjs +0 -356
- package/src/tools/glob-dialect.mjs +0 -130
- package/src/tools/glob.md +0 -11
- package/src/tools/grep.md +0 -19
- package/src/tools/hashline_edit.md +0 -14
- package/src/tools/index.mjs +0 -36
- package/src/tools/insert_after.md +0 -15
- package/src/tools/lint.md +0 -10
- package/src/tools/linter.mjs +0 -128
- package/src/tools/ls.md +0 -12
- package/src/tools/lsp.md +0 -10
- package/src/tools/lsp.mjs +0 -316
- package/src/tools/ops.mjs +0 -299
- package/src/tools/patch.mjs +0 -282
- package/src/tools/process.md +0 -10
- package/src/tools/question.md +0 -16
- package/src/tools/question.mjs +0 -26
- package/src/tools/read.md +0 -20
- package/src/tools/read_image.md +0 -8
- package/src/tools/repomap.mjs +0 -314
- package/src/tools/search.mjs +0 -236
- package/src/tools/shared.mjs +0 -446
- package/src/tools/tree.md +0 -14
- package/src/tools/tree.mjs +0 -66
- package/src/tools/wait_for.md +0 -22
- package/src/tools/web.mjs +0 -224
- package/src/tools/websearch.md +0 -16
- package/src/tools/write.md +0 -11
- package/src/traces/trace-store.mjs +0 -224
|
@@ -1,219 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* memory/code-index.mjs — code and document chunking, language detection, symbol extraction
|
|
3
|
-
*/
|
|
4
|
-
|
|
5
|
-
import { segmentCJK, CODE_EXTS, DOC_EXTS, SKIP_DIRS, BIG_FILE_LINES } from "./schema.mjs"
|
|
6
|
-
|
|
7
|
-
/** Infer file language by extension */
|
|
8
|
-
export function detectLanguage(filename) {
|
|
9
|
-
const ext = filename.slice(filename.lastIndexOf(".")).toLowerCase()
|
|
10
|
-
const map = {
|
|
11
|
-
".mjs": "javascript", ".js": "javascript", ".cjs": "javascript", ".jsx": "jsx",
|
|
12
|
-
".ts": "typescript", ".tsx": "tsx", ".mts": "typescript", ".cts": "typescript",
|
|
13
|
-
".py": "python", ".rs": "rust", ".go": "go", ".java": "java",
|
|
14
|
-
".c": "c", ".h": "c", ".cpp": "cpp", ".hpp": "cpp",
|
|
15
|
-
".rb": "ruby", ".swift": "swift", ".kt": "kotlin", ".dart": "dart", ".lua": "lua",
|
|
16
|
-
".cs": "csharp", ".fs": "fsharp", ".fsx": "fsharp",
|
|
17
|
-
".clj": "clojure", ".cljs": "clojure", ".ex": "elixir", ".exs": "elixir",
|
|
18
|
-
".erl": "erlang", ".hrl": "erlang", ".scala": "scala", ".groovy": "groovy",
|
|
19
|
-
".pl": "perl", ".pm": "perl", ".r": "r", ".jl": "julia", ".zig": "zig",
|
|
20
|
-
".ps1": "powershell", ".proto": "protobuf", ".graphql": "graphql", ".tf": "terraform", ".hcl": "hcl",
|
|
21
|
-
".sh": "bash", ".bash": "bash", ".sql": "sql",
|
|
22
|
-
".yaml": "yaml", ".yml": "yaml", ".toml": "toml", ".json": "json",
|
|
23
|
-
".css": "css", ".html": "html", ".vue": "vue", ".svelte": "svelte",
|
|
24
|
-
".md": "markdown", ".mdc": "markdown", ".mdx": "markdown", ".org": "org", ".wiki": "wiki", ".tex": "tex",
|
|
25
|
-
}
|
|
26
|
-
return map[ext] ?? ext.slice(1)
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
/**
|
|
30
|
-
* Extract top-level symbol declarations (functions, classes, const exports, etc.)
|
|
31
|
-
* from JS/TS files using regex. Returns [{ name, line, kind }].
|
|
32
|
-
*/
|
|
33
|
-
export function extractSymbols(lines, ext) {
|
|
34
|
-
const jsish = new Set([".mjs", ".js", ".ts", ".jsx", ".tsx"])
|
|
35
|
-
if (!jsish.has(ext)) return []
|
|
36
|
-
|
|
37
|
-
const symbols = []
|
|
38
|
-
const text = lines.join("\n")
|
|
39
|
-
const re = /(?:export\s+)?(?:(?:async\s+)?function\s+(\w+)|class\s+(\w+)|(?:export\s+)?(?:const|let|var)\s+(\w+))/gm
|
|
40
|
-
let m
|
|
41
|
-
while ((m = re.exec(text))) {
|
|
42
|
-
const name = m[1] || m[2] || m[3]
|
|
43
|
-
if (!name || name[0] !== name[0].toLowerCase() && name.length < 2) continue
|
|
44
|
-
const line = text.slice(0, m.index).split("\n").length
|
|
45
|
-
symbols.push({ name, line, kind: m[1] ? "function" : m[2] ? "class" : "variable" })
|
|
46
|
-
}
|
|
47
|
-
return symbols
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
/** Extract top-level def/class from Python files. */
|
|
51
|
-
export function extractPySymbols(lines) {
|
|
52
|
-
const symbols = []
|
|
53
|
-
const re = /^(?:async\s+)?(?:def|class)\s+(\w+)/gm
|
|
54
|
-
const text = lines.join("\n")
|
|
55
|
-
let m
|
|
56
|
-
while ((m = re.exec(text))) {
|
|
57
|
-
symbols.push({ name: m[1], line: text.slice(0, m.index).split("\n").length, kind: text[m.index] === "c" ? "class" : "function" })
|
|
58
|
-
}
|
|
59
|
-
return symbols
|
|
60
|
-
}
|
|
61
|
-
|
|
62
|
-
/**
|
|
63
|
-
* Split a file into code chunks. Small files become a single chunk;
|
|
64
|
-
* large files are split by symbol, with inter-symbol content merged into the preceding symbol chunk.
|
|
65
|
-
* Each chunk includes the JSDoc/docstring before its symbol to improve search quality.
|
|
66
|
-
*/
|
|
67
|
-
export function chunkCode(lines, filepath) {
|
|
68
|
-
const ext = filepath.slice(filepath.lastIndexOf(".")).toLowerCase()
|
|
69
|
-
const chunks = []
|
|
70
|
-
|
|
71
|
-
if (lines.length <= BIG_FILE_LINES) {
|
|
72
|
-
const doc = extractLeadingDoc(lines, 1, ext)
|
|
73
|
-
const content = (doc ? doc + "\n" : "") + lines.join("\n").trimEnd()
|
|
74
|
-
chunks.push({ name: filepath, line_start: 1, line_end: lines.length, content })
|
|
75
|
-
return chunks
|
|
76
|
-
}
|
|
77
|
-
|
|
78
|
-
const symbols = ext === ".py" ? extractPySymbols(lines) : extractSymbols(lines, ext)
|
|
79
|
-
if (symbols.length <= 1) {
|
|
80
|
-
const doc = extractLeadingDoc(lines, 1, ext)
|
|
81
|
-
const content = (doc ? doc + "\n" : "") + lines.join("\n").trimEnd()
|
|
82
|
-
chunks.push({ name: filepath, line_start: 1, line_end: lines.length, content })
|
|
83
|
-
return chunks
|
|
84
|
-
}
|
|
85
|
-
|
|
86
|
-
for (let i = 0; i < symbols.length; i++) {
|
|
87
|
-
const sym = symbols[i]
|
|
88
|
-
const start = sym.line
|
|
89
|
-
const end = i + 1 < symbols.length ? symbols[i + 1].line - 1 : lines.length
|
|
90
|
-
if (start > end) continue
|
|
91
|
-
const doc = extractLeadingDoc(lines, start, ext)
|
|
92
|
-
const body = lines.slice(start - 1, end).join("\n").trimEnd()
|
|
93
|
-
const content = (doc ? doc + "\n" : "") + body
|
|
94
|
-
if (!content) continue
|
|
95
|
-
chunks.push({ name: `${filepath}:${sym.name}`, line_start: start, line_end: end, content })
|
|
96
|
-
}
|
|
97
|
-
return chunks
|
|
98
|
-
}
|
|
99
|
-
|
|
100
|
-
/**
|
|
101
|
-
* Extract the JSDoc/docstring comment preceding a given line.
|
|
102
|
-
* JS/TS: scan backwards for JSDoc block comments or consecutive // comment lines
|
|
103
|
-
* Python: look for a """...""" docstring on the line after the symbol definition
|
|
104
|
-
*/
|
|
105
|
-
export function extractLeadingDoc(lines, lineNum, ext) {
|
|
106
|
-
if (ext === ".py") {
|
|
107
|
-
if (lineNum >= lines.length) return ""
|
|
108
|
-
const next = lines[lineNum]
|
|
109
|
-
const m = next?.match(/^\s*"""(.+?)"""\s*$/)
|
|
110
|
-
if (m) return m[1].trim()
|
|
111
|
-
if (/^\s*"""\s*$/.test(next)) {
|
|
112
|
-
const parts = []
|
|
113
|
-
for (let i = lineNum + 1; i < lines.length && i < lineNum + 8; i++) {
|
|
114
|
-
if (/^\s*"""\s*$/.test(lines[i])) break
|
|
115
|
-
parts.push(lines[i].trim())
|
|
116
|
-
}
|
|
117
|
-
const text = parts.join(" ").trim()
|
|
118
|
-
return text.length > 0 && text.length < 300 ? text : ""
|
|
119
|
-
}
|
|
120
|
-
return ""
|
|
121
|
-
}
|
|
122
|
-
|
|
123
|
-
const jsish = new Set([".mjs", ".js", ".ts", ".jsx", ".tsx"])
|
|
124
|
-
if (!jsish.has(ext)) return ""
|
|
125
|
-
|
|
126
|
-
const parts = []
|
|
127
|
-
let i = lineNum - 2
|
|
128
|
-
if (i >= 0 && /^\s*\*\/\s*$/.test(lines[i])) {
|
|
129
|
-
while (i >= 0) {
|
|
130
|
-
const line = lines[i].trim()
|
|
131
|
-
if (/^\s*\/\*\*/.test(line)) {
|
|
132
|
-
parts.unshift(line.replace(/^\s*\/\*\*\s*/, "").replace(/\s*\*\/\s*$/, "").trim())
|
|
133
|
-
break
|
|
134
|
-
}
|
|
135
|
-
parts.unshift(line.replace(/^\s*\*\s?/, "").trim())
|
|
136
|
-
i--
|
|
137
|
-
}
|
|
138
|
-
} else {
|
|
139
|
-
while (i >= 0 && /^\s*\/\//.test(lines[i])) {
|
|
140
|
-
parts.unshift(lines[i].replace(/^\s*\/\/\s*/, "").trim())
|
|
141
|
-
i--
|
|
142
|
-
}
|
|
143
|
-
}
|
|
144
|
-
|
|
145
|
-
const text = parts.join(" ").trim()
|
|
146
|
-
return text.length > 0 && text.length < 300 ? text : ""
|
|
147
|
-
}
|
|
148
|
-
|
|
149
|
-
/** Yield control to the event loop for one tick (allows keyboard input to be processed). Uses setImmediate for lower latency than setTimeout(0). */
|
|
150
|
-
export function yieldTick() {
|
|
151
|
-
return new Promise((r) => setImmediate(r))
|
|
152
|
-
}
|
|
153
|
-
|
|
154
|
-
/** Index a single file: delete old chunks → chunk → insert new chunks */
|
|
155
|
-
export function _upsertCodeFile(memory, origin, rel, lines, lang, mtimeMs) {
|
|
156
|
-
const chunks = chunkCode(lines, rel)
|
|
157
|
-
memory.db.exec("BEGIN")
|
|
158
|
-
try {
|
|
159
|
-
memory.db.prepare(`DELETE FROM code_chunks WHERE origin = ? AND path = ?`).run(origin, rel)
|
|
160
|
-
const insert = memory.db.prepare(`
|
|
161
|
-
INSERT INTO code_chunks (origin, path, language, chunk_type, symbol_name, content, line_start, line_end, mtime_ms, seg_content)
|
|
162
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
163
|
-
`)
|
|
164
|
-
for (const c of chunks) {
|
|
165
|
-
const isFile = c.name === rel
|
|
166
|
-
insert.run(origin, rel, lang, isFile ? "file" : "symbol", isFile ? "" : c.name.slice(rel.length + 1), c.content, c.line_start, c.line_end, mtimeMs, segmentCJK(c.content))
|
|
167
|
-
}
|
|
168
|
-
memory.db.exec("COMMIT")
|
|
169
|
-
} catch (e) {
|
|
170
|
-
memory.db.exec("ROLLBACK")
|
|
171
|
-
throw e
|
|
172
|
-
}
|
|
173
|
-
}
|
|
174
|
-
|
|
175
|
-
/**
|
|
176
|
-
* Split a markdown file by ## headings. Each ## section is indexed independently,
|
|
177
|
-
* with the heading path as the heading label (e.g. "README.md > Deployment > Docker") for easy retrieval.
|
|
178
|
-
*/
|
|
179
|
-
export function chunkMarkdown(lines, filepath) {
|
|
180
|
-
const chunks = []
|
|
181
|
-
let start = 1
|
|
182
|
-
let heading = filepath
|
|
183
|
-
|
|
184
|
-
for (let i = 0; i < lines.length; i++) {
|
|
185
|
-
const m = lines[i].match(/^(#{1,4})\s+(.+)/)
|
|
186
|
-
if (m) {
|
|
187
|
-
if (i > start) {
|
|
188
|
-
chunks.push({ heading, line_start: start, line_end: i, content: lines.slice(start - 1, i).join("\n").trimEnd() })
|
|
189
|
-
}
|
|
190
|
-
heading = `${filepath} > ${m[2].trim()}`
|
|
191
|
-
start = i + 1
|
|
192
|
-
}
|
|
193
|
-
}
|
|
194
|
-
if (start <= lines.length) {
|
|
195
|
-
chunks.push({ heading, line_start: start, line_end: lines.length, content: lines.slice(start - 1).join("\n").trimEnd() })
|
|
196
|
-
}
|
|
197
|
-
return chunks.filter((c) => c.content)
|
|
198
|
-
}
|
|
199
|
-
|
|
200
|
-
/** Upsert a documentation file's chunks into the doc_chunks table within a transaction */
|
|
201
|
-
export function _upsertDocFile(memory, origin, rel, lines, mtimeMs) {
|
|
202
|
-
const chunks = chunkMarkdown(lines, rel)
|
|
203
|
-
const lang = rel.endsWith(".rst") ? "rst" : rel.endsWith(".adoc") ? "asciidoc" : rel.endsWith(".txt") ? "text" : "markdown"
|
|
204
|
-
memory.db.exec("BEGIN")
|
|
205
|
-
try {
|
|
206
|
-
memory.db.prepare(`DELETE FROM doc_chunks WHERE origin = ? AND path = ?`).run(origin, rel)
|
|
207
|
-
const insert = memory.db.prepare(`
|
|
208
|
-
INSERT INTO doc_chunks (origin, path, language, heading, content, line_start, line_end, mtime_ms, seg_content)
|
|
209
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
210
|
-
`)
|
|
211
|
-
for (const c of chunks) {
|
|
212
|
-
insert.run(origin, rel, lang, c.heading, c.content, c.line_start, c.line_end, mtimeMs, segmentCJK(c.content))
|
|
213
|
-
}
|
|
214
|
-
memory.db.exec("COMMIT")
|
|
215
|
-
} catch (e) {
|
|
216
|
-
memory.db.exec("ROLLBACK")
|
|
217
|
-
throw e
|
|
218
|
-
}
|
|
219
|
-
}
|
package/src/memory/code-sync.mjs
DELETED
|
@@ -1,413 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* memory/code-sync.mjs — code index sync, retrieval, incremental update
|
|
3
|
-
*/
|
|
4
|
-
import { readFile, stat } from "node:fs/promises"
|
|
5
|
-
import { join, relative } from "node:path"
|
|
6
|
-
import { embed, cosine, toBlob, fromBlob } from "../embedding.mjs"
|
|
7
|
-
import { CODE_EXTS, DOC_EXTS, MAX_CODE_FILE_BYTES, MAX_DOC_FILE_BYTES } from "./schema.mjs"
|
|
8
|
-
import { buildFtsQuery, ensureEmbeddings, EMBED_TEXT_MAX_LEN } from "./core.mjs"
|
|
9
|
-
import { detectLanguage, _upsertCodeFile, _upsertDocFile, yieldTick } from "./code-index.mjs"
|
|
10
|
-
import { walkProjectFiles, isSkippedRelPath, extensionOf, createUnlistedTally, MAX_WALK_FILES } from "./file-walk.mjs"
|
|
11
|
-
import { loadConventions } from "../conventions.mjs"
|
|
12
|
-
import { logEvent } from "../log.mjs"
|
|
13
|
-
|
|
14
|
-
const DIFF_FULL_SYNC_THRESHOLD = 200
|
|
15
|
-
const CODE_EMBED_BATCH = 64
|
|
16
|
-
|
|
17
|
-
/**
|
|
18
|
-
* Index extension sets = built-in tables ∪ project declaration
|
|
19
|
-
* (`.thincoder/conventions.json` → index.codeExtensions / index.docExtensions;
|
|
20
|
-
* PORTABILITY PO-9/§3.7 — a project may declare extensions this product does not
|
|
21
|
-
* ship a default for). Declaration only ADDS (union), never removes.
|
|
22
|
-
*/
|
|
23
|
-
export function indexExtensions(dir) {
|
|
24
|
-
const conv = loadConventions(dir)
|
|
25
|
-
return {
|
|
26
|
-
code: new Set([...CODE_EXTS, ...conv.index.codeExtensions]),
|
|
27
|
-
doc: new Set([...DOC_EXTS, ...conv.index.docExtensions]),
|
|
28
|
-
}
|
|
29
|
-
}
|
|
30
|
-
|
|
31
|
-
/**
|
|
32
|
-
* git-driven incremental indexing: use git diff to find files changed since
|
|
33
|
-
* the last index, and only rebuild FTS5 chunks for those files (vectors are untouched).
|
|
34
|
-
* An order of magnitude faster than full mtime scanning.
|
|
35
|
-
* Returns { updated, removed, skipped } or null (git unavailable).
|
|
36
|
-
*/
|
|
37
|
-
export async function gitSync(memory, dir, { onProgress } = {}) {
|
|
38
|
-
const { execFile: _execFile } = await import("node:child_process")
|
|
39
|
-
const { code: codeExts, doc: docExts } = indexExtensions(dir)
|
|
40
|
-
const gitRun = (args) => new Promise((resolve, reject) => {
|
|
41
|
-
_execFile("git", args, { cwd: dir, encoding: "utf8", timeout: 10000, windowsHide: true }, (err, stdout) => {
|
|
42
|
-
if (err) reject(err); else resolve(stdout)
|
|
43
|
-
})
|
|
44
|
-
})
|
|
45
|
-
const mergeDiff = (text) => text.trim().split("\n").filter(Boolean)
|
|
46
|
-
|
|
47
|
-
let head
|
|
48
|
-
try { head = (await gitRun(["rev-parse", "HEAD"])).trim() } catch { return null }
|
|
49
|
-
|
|
50
|
-
const stored = memory.db.prepare(`SELECT value FROM meta WHERE key = 'last_indexed_commit'`).get()?.value
|
|
51
|
-
if (!stored) return null
|
|
52
|
-
|
|
53
|
-
let diffOut
|
|
54
|
-
try {
|
|
55
|
-
const committed = mergeDiff(await gitRun(["diff", "--name-only", "--diff-filter=ACMRTD", stored, "HEAD"]))
|
|
56
|
-
const dirty = mergeDiff(await gitRun(["diff", "--name-only", "--diff-filter=ACMRTD"]))
|
|
57
|
-
const lines = [...new Set([...committed, ...dirty])]
|
|
58
|
-
diffOut = lines
|
|
59
|
-
} catch {
|
|
60
|
-
return null
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
if (diffOut.length > DIFF_FULL_SYNC_THRESHOLD) {
|
|
64
|
-
// diff too large, incremental is useless — fall back to full sync and update anchor
|
|
65
|
-
await codeSync(memory, dir, { onProgress })
|
|
66
|
-
const { docSync } = await import("./docs.mjs")
|
|
67
|
-
await docSync(memory, dir, { onProgress })
|
|
68
|
-
return { updated: -1, removed: 0, skipped: 0, failed: 0, errors: [], fallback: true }
|
|
69
|
-
}
|
|
70
|
-
|
|
71
|
-
let updated = 0, removed = 0, skipped = 0, failed = 0
|
|
72
|
-
const errors = []
|
|
73
|
-
for (let i = 0; i < diffOut.length; i++) {
|
|
74
|
-
const rel = diffOut[i].replaceAll("\\", "/")
|
|
75
|
-
const abs = join(dir, rel)
|
|
76
|
-
const ext = extensionOf(rel)
|
|
77
|
-
|
|
78
|
-
if (isSkippedRelPath(rel)) continue
|
|
79
|
-
|
|
80
|
-
if (!codeExts.has(ext) && !docExts.has(ext)) { skipped++; continue }
|
|
81
|
-
|
|
82
|
-
try {
|
|
83
|
-
const text = await readFile(abs, "utf8")
|
|
84
|
-
const lines = text.split("\n")
|
|
85
|
-
if (codeExts.has(ext)) {
|
|
86
|
-
const lang = detectLanguage(abs)
|
|
87
|
-
let mtimeMs = 0
|
|
88
|
-
try { mtimeMs = Math.floor((await stat(abs)).mtimeMs) } catch { /* new file */ }
|
|
89
|
-
_upsertCodeFile(memory, dir, rel, lines, lang, mtimeMs)
|
|
90
|
-
} else {
|
|
91
|
-
let mtimeMs = 0
|
|
92
|
-
try { mtimeMs = Math.floor((await stat(abs)).mtimeMs) } catch { /* new file */ }
|
|
93
|
-
_upsertDocFile(memory, dir, rel, lines, mtimeMs)
|
|
94
|
-
}
|
|
95
|
-
updated++
|
|
96
|
-
} catch (e) {
|
|
97
|
-
const isDeleted = e.code === "ENOENT"
|
|
98
|
-
if (isDeleted) {
|
|
99
|
-
if (codeExts.has(ext)) memory.db.prepare(`DELETE FROM code_chunks WHERE origin = ? AND path = ?`).run(dir, rel)
|
|
100
|
-
else memory.db.prepare(`DELETE FROM doc_chunks WHERE origin = ? AND path = ?`).run(dir, rel)
|
|
101
|
-
removed++
|
|
102
|
-
} else {
|
|
103
|
-
failed++
|
|
104
|
-
if (errors.length < 5) errors.push(`${rel}: ${e.message}`)
|
|
105
|
-
}
|
|
106
|
-
}
|
|
107
|
-
await yieldTick()
|
|
108
|
-
if (onProgress && i % 5 === 0) {
|
|
109
|
-
onProgress({ phase: "index", current: i + 1, total: diffOut.length, updated, removed, skipped })
|
|
110
|
-
}
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
if (failed === 0) {
|
|
114
|
-
memory.db.prepare(`INSERT INTO meta (key, value) VALUES ('last_indexed_commit', ?)
|
|
115
|
-
ON CONFLICT (key) DO UPDATE SET value = excluded.value`).run(head)
|
|
116
|
-
}
|
|
117
|
-
|
|
118
|
-
onProgress?.({ phase: "done", total: diffOut.length, updated, removed, skipped, failed })
|
|
119
|
-
return { updated, removed, skipped, failed, errors }
|
|
120
|
-
}
|
|
121
|
-
|
|
122
|
-
/**
|
|
123
|
-
* List project files matching the given extensions.
|
|
124
|
-
* Preferred source = git (`git ls-files --cached --others --exclude-standard`:
|
|
125
|
-
* tracked + untracked-not-ignored, .gitignore respected). Projects WITHOUT git
|
|
126
|
-
* fall back to a filesystem walk (PORTABILITY FR15/P8) — before, the index was
|
|
127
|
-
* silently empty there. The walk cannot honour .gitignore (no git → no such
|
|
128
|
-
* concept) and skips the shared SKIP_DIRS/dot-directory set instead.
|
|
129
|
-
* Returns { entries, unlisted }: `entries` = { abs, rel } pairs (rel relative to
|
|
130
|
-
* dir); `unlisted` = files whose extension is in NO index list (count + sample),
|
|
131
|
-
* the visible signal behind "my .xyz files are not searchable" (PORTABILITY PO-9).
|
|
132
|
-
* `truncated` = the non-git walk hit its file cap (capped trees are logged, never
|
|
133
|
-
* silently indexed-partial). opts.maxFiles = the walk cap (test seam).
|
|
134
|
-
*/
|
|
135
|
-
export async function listProjectFiles(dir, exts, { maxFiles } = {}) {
|
|
136
|
-
const { execFile: _execFile } = await import("node:child_process")
|
|
137
|
-
const { join: joinPath } = await import("node:path")
|
|
138
|
-
const { code, doc } = indexExtensions(dir)
|
|
139
|
-
const tally = createUnlistedTally(new Set([...code, ...doc]))
|
|
140
|
-
|
|
141
|
-
const files = []
|
|
142
|
-
const finish = (truncated = false) => ({ entries: files, unlisted: tally.result(), truncated })
|
|
143
|
-
|
|
144
|
-
// Git listing when available; a non-git project falls through to the walk.
|
|
145
|
-
let gitTop
|
|
146
|
-
try {
|
|
147
|
-
gitTop = (await new Promise((resolve, reject) => {
|
|
148
|
-
_execFile("git", ["rev-parse", "--show-toplevel"], { cwd: dir, encoding: "utf8", timeout: 5000, windowsHide: true },
|
|
149
|
-
(err, stdout) => { if (err) reject(err); else resolve(stdout.trim()) })
|
|
150
|
-
})).replace(/\\/g, "/")
|
|
151
|
-
} catch {
|
|
152
|
-
// Not a git repo → filesystem walk (FR15: the index must still be usable).
|
|
153
|
-
const walked = await walkProjectFiles(dir, exts, { knownExts: new Set([...code, ...doc]), ...(maxFiles !== undefined ? { maxFiles } : {}) })
|
|
154
|
-
files.push(...walked.files)
|
|
155
|
-
// The walk's cap must stay visible end-to-end (file-walk.mjs: "never silently
|
|
156
|
-
// dropped") — a capped listing says so in the log, not only in a discarded flag.
|
|
157
|
-
if (walked.truncated) {
|
|
158
|
-
logEvent("index:truncated", { dir, files: walked.files.length, maxFiles: maxFiles ?? MAX_WALK_FILES })
|
|
159
|
-
}
|
|
160
|
-
return { entries: files, unlisted: walked.unlisted, truncated: walked.truncated }
|
|
161
|
-
}
|
|
162
|
-
|
|
163
|
-
try {
|
|
164
|
-
const raw = await new Promise((resolve, reject) => {
|
|
165
|
-
_execFile("git", ["ls-files", "--cached", "--others", "--exclude-standard"],
|
|
166
|
-
{ cwd: dir, encoding: "utf8", timeout: 15000, windowsHide: true, maxBuffer: 10 * 1024 * 1024 },
|
|
167
|
-
(err, stdout) => { if (err) reject(err); else resolve(stdout) })
|
|
168
|
-
})
|
|
169
|
-
for (const line of raw.trim().split("\n")) {
|
|
170
|
-
const p = line.trim()
|
|
171
|
-
if (!p) continue
|
|
172
|
-
const rel = p.replace(/\\/g, "/")
|
|
173
|
-
if (isSkippedRelPath(rel)) continue
|
|
174
|
-
const ext = extensionOf(rel)
|
|
175
|
-
if (!ext || !exts.has(ext)) { tally.note(rel); continue }
|
|
176
|
-
files.push({ abs: joinPath(dir, rel), rel })
|
|
177
|
-
}
|
|
178
|
-
} catch { /* ls-files failed */ }
|
|
179
|
-
|
|
180
|
-
return finish()
|
|
181
|
-
}
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
/**
|
|
185
|
-
* Sync code index: scan all source files under dir → chunk → upsert into code_chunks.
|
|
186
|
-
* Incremental by mtime — only rebuilds chunks for files that have changed.
|
|
187
|
-
*/
|
|
188
|
-
export async function codeSync(memory, dir, { onProgress } = {}) {
|
|
189
|
-
const { code: exts } = indexExtensions(dir)
|
|
190
|
-
const { entries, unlisted } = await listProjectFiles(dir, exts)
|
|
191
|
-
const files = [] // { abs, rel, mtimeMs }
|
|
192
|
-
let overSizeSkipped = 0
|
|
193
|
-
for (const { abs, rel } of entries) {
|
|
194
|
-
let st
|
|
195
|
-
try { st = await stat(abs) } catch { continue }
|
|
196
|
-
if (st.size > MAX_CODE_FILE_BYTES) { overSizeSkipped++; continue }
|
|
197
|
-
files.push({ abs, rel, mtimeMs: Math.floor(st.mtimeMs) })
|
|
198
|
-
}
|
|
199
|
-
|
|
200
|
-
const indexed = new Map(
|
|
201
|
-
memory.db.prepare(`SELECT path, mtime_ms FROM code_chunks WHERE origin = ?`).all(dir).map((r) => [r.path, r.mtime_ms])
|
|
202
|
-
)
|
|
203
|
-
const seen = new Set()
|
|
204
|
-
|
|
205
|
-
onProgress?.({ phase: "scan", total: files.length, overSizeSkipped })
|
|
206
|
-
|
|
207
|
-
let updated = 0, removed = 0, skipped = 0, failed = 0
|
|
208
|
-
const errors = []
|
|
209
|
-
for (let i = 0; i < files.length; i++) {
|
|
210
|
-
const { abs, rel, mtimeMs } = files[i]
|
|
211
|
-
seen.add(rel)
|
|
212
|
-
|
|
213
|
-
if (indexed.get(rel) === mtimeMs) {
|
|
214
|
-
skipped++
|
|
215
|
-
continue
|
|
216
|
-
}
|
|
217
|
-
|
|
218
|
-
try {
|
|
219
|
-
const text = await readFile(abs, "utf8")
|
|
220
|
-
const lines = text.split("\n")
|
|
221
|
-
const lang = detectLanguage(abs)
|
|
222
|
-
_upsertCodeFile(memory, dir, rel, lines, lang, mtimeMs)
|
|
223
|
-
updated++
|
|
224
|
-
} catch (e) {
|
|
225
|
-
failed++
|
|
226
|
-
if (errors.length < 5) errors.push(`${rel}: ${e.message}`)
|
|
227
|
-
}
|
|
228
|
-
await yieldTick()
|
|
229
|
-
|
|
230
|
-
if (onProgress && i % 10 === 0) {
|
|
231
|
-
onProgress({ phase: "index", current: i + 1, total: files.length, updated, removed, skipped, failed })
|
|
232
|
-
}
|
|
233
|
-
}
|
|
234
|
-
|
|
235
|
-
for (const stale of indexed.keys()) {
|
|
236
|
-
if (!seen.has(stale)) {
|
|
237
|
-
memory.db.prepare(`DELETE FROM code_chunks WHERE origin = ? AND path = ?`).run(dir, stale)
|
|
238
|
-
removed++
|
|
239
|
-
}
|
|
240
|
-
}
|
|
241
|
-
|
|
242
|
-
onProgress?.({ phase: "done", total: files.length, updated, removed, skipped, failed, overSizeSkipped })
|
|
243
|
-
markIndexedCommit(memory, dir)
|
|
244
|
-
if (unlisted.count > 0) {
|
|
245
|
-
logEvent("index:unlisted", { dir, kind: "code", count: unlisted.count, exts: unlisted.exts.map((e) => e.ext) })
|
|
246
|
-
}
|
|
247
|
-
return { updated, removed, skipped, failed, errors, total: files.length, overSizeSkipped, unlistedExts: unlisted }
|
|
248
|
-
}
|
|
249
|
-
|
|
250
|
-
/** Record current HEAD as the index anchor (gitSync incremental diff baseline); silently skip non-git repos */
|
|
251
|
-
export async function markIndexedCommit(memory, dir) {
|
|
252
|
-
try {
|
|
253
|
-
const { execFile } = await import("node:child_process")
|
|
254
|
-
const head = await new Promise((resolve, reject) => {
|
|
255
|
-
execFile("git", ["rev-parse", "HEAD"], { cwd: dir, encoding: "utf8", timeout: 5000, windowsHide: true }, (err, stdout) => {
|
|
256
|
-
if (err) reject(err); else resolve(stdout)
|
|
257
|
-
})
|
|
258
|
-
})
|
|
259
|
-
memory.db.prepare(`INSERT INTO meta (key, value) VALUES ('last_indexed_commit', ?)
|
|
260
|
-
ON CONFLICT (key) DO UPDATE SET value = excluded.value`).run(head.trim())
|
|
261
|
-
} catch { /* not a git repo or git unavailable, skip */ }
|
|
262
|
-
}
|
|
263
|
-
|
|
264
|
-
/**
|
|
265
|
-
* Code search: FTS5(BM25) + optional vector cosine, RRF merged.
|
|
266
|
-
* Falls back to pure FTS when no embedder; falls back to pure vector when ftsQuery is empty and embedder is present.
|
|
267
|
-
*/
|
|
268
|
-
export async function codeSearch(memory, query, { limit = 5 } = {}) {
|
|
269
|
-
const ftsQuery = buildFtsQuery(query)
|
|
270
|
-
if (!ftsQuery && !memory.embedder) return []
|
|
271
|
-
|
|
272
|
-
const ftsOriginFilter = memory.codeOrigin ? `AND c.origin = ?` : ""
|
|
273
|
-
const vecOriginFilter = memory.codeOrigin ? `AND origin = ?` : ""
|
|
274
|
-
const originParams = memory.codeOrigin ? [memory.codeOrigin] : []
|
|
275
|
-
|
|
276
|
-
const ftsList = ftsQuery ? memory.db.prepare(`
|
|
277
|
-
SELECT c.rowid, c.path, c.language, c.symbol_name, c.content, c.line_start, c.line_end, bm25(code_chunks_fts) AS rank
|
|
278
|
-
FROM code_chunks_fts JOIN code_chunks c ON c.rowid = code_chunks_fts.rowid
|
|
279
|
-
WHERE code_chunks_fts MATCH ? ${ftsOriginFilter}
|
|
280
|
-
ORDER BY rank LIMIT ?
|
|
281
|
-
`).all(ftsQuery, ...originParams, Math.max(limit * 4, 20)) : []
|
|
282
|
-
|
|
283
|
-
if (!memory.embedder) return ftsList.slice(0, limit)
|
|
284
|
-
|
|
285
|
-
try { await ensureEmbeddings(memory) } catch (e) {
|
|
286
|
-
console.error(`[code] embedding ensure failed, falling back to FTS-only: ${e.message}`)
|
|
287
|
-
return ftsList.slice(0, limit)
|
|
288
|
-
}
|
|
289
|
-
let qvec
|
|
290
|
-
try { [qvec] = await embed(memory.embedder, [query]) } catch (e) {
|
|
291
|
-
console.error(`[code] query embedding failed, falling back to FTS-only: ${e.message}`)
|
|
292
|
-
return ftsList.slice(0, limit)
|
|
293
|
-
}
|
|
294
|
-
const rows = memory.db.prepare(`SELECT rowid, embedding FROM code_chunks WHERE embedding IS NOT NULL ${vecOriginFilter}`).all(...originParams)
|
|
295
|
-
const vecList = rows
|
|
296
|
-
.map((r) => ({ rowid: r.rowid, score: cosine(qvec, fromBlob(r.embedding)) }))
|
|
297
|
-
.sort((a, b) => b.score - a.score)
|
|
298
|
-
.slice(0, Math.max(limit * 4, 20))
|
|
299
|
-
|
|
300
|
-
const K = 60
|
|
301
|
-
const scores = new Map()
|
|
302
|
-
ftsList.forEach((r, i) => scores.set(r.rowid, (scores.get(r.rowid) ?? 0) + 1 / (K + i + 1)))
|
|
303
|
-
vecList.forEach((r, i) => scores.set(r.rowid, (scores.get(r.rowid) ?? 0) + 1 / (K + i + 1)))
|
|
304
|
-
|
|
305
|
-
const fetchChunk = memory.db.prepare(`
|
|
306
|
-
SELECT path, language, symbol_name, content, line_start, line_end FROM code_chunks WHERE rowid = ?
|
|
307
|
-
`)
|
|
308
|
-
const sorted = [...scores.entries()]
|
|
309
|
-
.sort((a, b) => b[1] - a[1])
|
|
310
|
-
.slice(0, limit)
|
|
311
|
-
return sorted
|
|
312
|
-
.map(([rowid, score]) => {
|
|
313
|
-
const chunk = fetchChunk.get(rowid)
|
|
314
|
-
if (!chunk) return null
|
|
315
|
-
chunk._score = Math.round(score * 100) / 100
|
|
316
|
-
return chunk
|
|
317
|
-
})
|
|
318
|
-
.filter(Boolean)
|
|
319
|
-
}
|
|
320
|
-
|
|
321
|
-
/** Lazily backfill missing vectors for code_chunks. Guarded against concurrent calls. */
|
|
322
|
-
let _codeEmbedLock = null
|
|
323
|
-
export function ensureCodeEmbeddings(memory) {
|
|
324
|
-
if (_codeEmbedLock) return _codeEmbedLock
|
|
325
|
-
_codeEmbedLock = _runEnsureCodeEmbeddings(memory).finally(() => { _codeEmbedLock = null })
|
|
326
|
-
return _codeEmbedLock
|
|
327
|
-
}
|
|
328
|
-
|
|
329
|
-
async function _runEnsureCodeEmbeddings(memory) {
|
|
330
|
-
if (!memory.embedder) return
|
|
331
|
-
const modelKey = memory.embedder.model
|
|
332
|
-
const stored = memory.db.prepare(`SELECT value FROM meta WHERE key = 'code_embedding_model'`).get()?.value
|
|
333
|
-
if (stored !== modelKey) {
|
|
334
|
-
memory.db.prepare(`UPDATE code_chunks SET embedding = NULL`).run()
|
|
335
|
-
memory.db.prepare(`INSERT INTO meta (key, value) VALUES ('code_embedding_model', ?)
|
|
336
|
-
ON CONFLICT (key) DO UPDATE SET value = excluded.value`).run(modelKey)
|
|
337
|
-
}
|
|
338
|
-
|
|
339
|
-
const pending = memory.db.prepare(`SELECT rowid, path, symbol_name, content FROM code_chunks WHERE embedding IS NULL LIMIT ${CODE_EMBED_BATCH}`).all()
|
|
340
|
-
if (pending.length === 0) return
|
|
341
|
-
|
|
342
|
-
const texts = pending.map((r) => `${r.path}${r.symbol_name ? " :: " + r.symbol_name : ""}\n${r.content.slice(0, EMBED_TEXT_MAX_LEN)}`)
|
|
343
|
-
const vecs = await embed(memory.embedder, texts)
|
|
344
|
-
|
|
345
|
-
const update = memory.db.prepare(`UPDATE code_chunks SET embedding = ? WHERE rowid = ?`)
|
|
346
|
-
pending.forEach((r, i) => update.run(toBlob(vecs[i]), r.rowid))
|
|
347
|
-
}
|
|
348
|
-
|
|
349
|
-
/** Generate the code_search tool (read-only). */
|
|
350
|
-
export function codeSearchTool(memory) {
|
|
351
|
-
return {
|
|
352
|
-
name: "code_search",
|
|
353
|
-
description:
|
|
354
|
-
"Search the project's source code for relevant code. Use this to find functions, classes, or code patterns across the codebase. Supports natural language queries and code snippets. Returns matching code chunks with file paths and line numbers. Prefer doc_search for the intended design (design docs, conventions); code_search for the implementation as written. " +
|
|
355
|
-
"For what was said in a session (conversation/chat history), use read_history.",
|
|
356
|
-
parameters: {
|
|
357
|
-
type: "object",
|
|
358
|
-
properties: {
|
|
359
|
-
query: { type: "string", description: "Natural language or code snippet to search for" },
|
|
360
|
-
limit: { type: "number", description: "Max results (default 5)" },
|
|
361
|
-
},
|
|
362
|
-
required: ["query"],
|
|
363
|
-
},
|
|
364
|
-
readonly: true,
|
|
365
|
-
async execute(args) {
|
|
366
|
-
const results = await codeSearch(memory, args.query, { limit: args.limit ?? 5 })
|
|
367
|
-
if (results.length === 0) return "(no matching code)"
|
|
368
|
-
return results.map((r) =>
|
|
369
|
-
`${r.path}${r.symbol_name ? ` :: ${r.symbol_name}` : ""} (L${r.line_start}-L${r.line_end}, relevance ${r._score?.toFixed(2) ?? "?"}):\n${r.content.slice(0, 2000)}`
|
|
370
|
-
).join("\n\n---\n\n")
|
|
371
|
-
},
|
|
372
|
-
}
|
|
373
|
-
}
|
|
374
|
-
|
|
375
|
-
/**
|
|
376
|
-
* Single-file incremental reindex: called after write/edit/delete, only rebuilds this one path.
|
|
377
|
-
*/
|
|
378
|
-
export async function reindexFile(memory, cwd, absPath) {
|
|
379
|
-
const ext = extensionOf(absPath)
|
|
380
|
-
const rel = relative(cwd, absPath).replaceAll("\\", "/")
|
|
381
|
-
if (rel === ".." || rel.startsWith("../")) return
|
|
382
|
-
if (isSkippedRelPath(rel)) return
|
|
383
|
-
// Declared extensions count for the single-file path too (union with built-ins).
|
|
384
|
-
const { code: codeExts, doc: docExts } = indexExtensions(cwd)
|
|
385
|
-
|
|
386
|
-
// skip oversized files (minified bundles, test fixtures, generated code)
|
|
387
|
-
const maxBytes = codeExts.has(ext) ? MAX_CODE_FILE_BYTES : docExts.has(ext) ? MAX_DOC_FILE_BYTES : 0
|
|
388
|
-
if (maxBytes > 0) {
|
|
389
|
-
try { const st = await stat(absPath); if (st.size > maxBytes) return } catch { /* can't stat, proceed */ }
|
|
390
|
-
}
|
|
391
|
-
|
|
392
|
-
let text
|
|
393
|
-
try { text = await readFile(absPath, "utf8") } catch {
|
|
394
|
-
if (codeExts.has(ext)) memory.db.prepare(`DELETE FROM code_chunks WHERE origin = ? AND path = ?`).run(cwd, rel)
|
|
395
|
-
else if (docExts.has(ext)) memory.db.prepare(`DELETE FROM doc_chunks WHERE origin = ? AND path = ?`).run(cwd, rel)
|
|
396
|
-
return
|
|
397
|
-
}
|
|
398
|
-
const lines = text.split("\n")
|
|
399
|
-
|
|
400
|
-
if (codeExts.has(ext)) {
|
|
401
|
-
const lang = detectLanguage(absPath)
|
|
402
|
-
let mtimeMs = 0
|
|
403
|
-
try { mtimeMs = Math.floor((await stat(absPath)).mtimeMs) } catch { /* new file */ }
|
|
404
|
-
_upsertCodeFile(memory, cwd, rel, lines, lang, mtimeMs)
|
|
405
|
-
} else if (docExts.has(ext)) {
|
|
406
|
-
let mtimeMs = 0
|
|
407
|
-
try { mtimeMs = Math.floor((await stat(absPath)).mtimeMs) } catch { /* new file */ }
|
|
408
|
-
_upsertDocFile(memory, cwd, rel, lines, mtimeMs)
|
|
409
|
-
}
|
|
410
|
-
if (memory.embedder) {
|
|
411
|
-
try { await ensureEmbeddings(memory) } catch { /* embedding failure is non-blocking */ }
|
|
412
|
-
}
|
|
413
|
-
}
|