thincoder 0.12.62 → 0.12.63
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/README.md +13 -12
- package/bin/thincoder.mjs +53 -27
- package/package.json +6 -5
- package/src/acp/bridge.mjs +35 -15
- package/src/acp/client-caps.mjs +86 -0
- package/src/acp/ext.mjs +86 -0
- package/src/acp/handlers-session.mjs +240 -0
- package/src/acp/handlers-slots.mjs +196 -0
- package/src/acp/login.mjs +48 -0
- package/src/acp/session.mjs +6 -4
- package/src/acp.mjs +67 -379
- package/src/cli/distill-command.mjs +3 -3
- package/src/cli/make-agent.mjs +59 -17
- package/src/cli/memory-command.mjs +3 -3
- package/src/cli/permission.mjs +4 -48
- package/src/cli/setup-wizard.mjs +1 -1
- package/src/completions.mjs +3 -1
- package/src/crash-reports.mjs +1 -1
- package/src/distill.mjs +4 -4
- package/src/heap-watch.mjs +1 -1
- package/src/prompt-injections.mjs +20 -0
- package/src/tui/agent-turn.mjs +40 -9
- package/src/tui/cmd-advisor.mjs +5 -5
- package/src/tui/cmd-config.mjs +8 -8
- package/src/tui/cmd-eng.mjs +25 -9
- package/src/tui/cmd-mcp.mjs +9 -8
- package/src/tui/cmd-model.mjs +1 -1
- package/src/tui/cmd-new.mjs +6 -5
- package/src/tui/cmd-reindex.mjs +1 -1
- package/src/tui/cmd-restore.mjs +2 -2
- package/src/tui/cmd-session.mjs +24 -1
- package/src/tui/cmd-skills.mjs +1 -1
- package/src/tui/cmd-think.mjs +22 -9
- package/src/tui/config-helpers.mjs +1 -1
- package/src/tui/display-budget.mjs +33 -11
- package/src/tui/index.mjs +9 -9
- package/src/tui/interaction.mjs +16 -7
- package/src/tui/key-modes.mjs +9 -4
- package/src/tui/ledger-surface.mjs +26 -10
- package/src/tui/model-catalog.mjs +4 -4
- package/src/tui/model-picker.mjs +8 -7
- package/src/tui/mouse.mjs +11 -6
- package/src/tui/pickers.mjs +15 -2
- package/src/tui/render-conversation.mjs +1 -1
- package/src/tui/render-frame.mjs +10 -5
- package/src/tui/render-loop.mjs +1 -1
- package/src/tui/render-segments.mjs +3 -1
- package/src/tui/slash-commands.mjs +1 -1
- package/src/tui/startup.mjs +14 -14
- package/src/tui/subagent-blocks.mjs +20 -3
- package/src/tui/subagent-freeze.mjs +73 -2
- package/src/tui/suspension-drive.mjs +46 -22
- package/src/tui/tool-events.mjs +11 -8
- package/src/tui/wizard.mjs +3 -3
- package/src/abort-provenance.mjs +0 -116
- package/src/advisor/citations.mjs +0 -139
- package/src/advisor/compaction.mjs +0 -174
- package/src/advisor/convergence.mjs +0 -80
- package/src/advisor/history.mjs +0 -77
- package/src/advisor/loop.mjs +0 -293
- package/src/advisor/messages.mjs +0 -299
- package/src/advisor/project-context.mjs +0 -194
- package/src/advisor/repos.mjs +0 -150
- package/src/advisor/run.mjs +0 -293
- package/src/advisor/truncate.mjs +0 -57
- package/src/advisor.mjs +0 -290
- package/src/agent/completion.mjs +0 -146
- package/src/agent/dispatch.mjs +0 -489
- package/src/agent/helpers.mjs +0 -384
- package/src/agent/post-turn.mjs +0 -70
- package/src/agent/record-results.mjs +0 -174
- package/src/agent/relay-prefix.mjs +0 -39
- package/src/agent/run-stages.mjs +0 -244
- package/src/agent/setup-reminders.mjs +0 -69
- package/src/agent/setup.mjs +0 -354
- package/src/agent/spawn-child.mjs +0 -243
- package/src/agent-tools/advisor-async.mjs +0 -346
- package/src/agent-tools/advisor-settle.mjs +0 -231
- package/src/agent-tools/advisor.mjs +0 -260
- package/src/agent-tools/async-settle.mjs +0 -204
- package/src/agent-tools/batch-segment.mjs +0 -195
- package/src/agent-tools/consult.mjs +0 -473
- package/src/agent-tools/design-token.mjs +0 -117
- package/src/agent-tools/digest-budget.mjs +0 -76
- package/src/agent-tools/eng.mjs +0 -67
- package/src/agent-tools/escalate-async.mjs +0 -295
- package/src/agent-tools/goal.mjs +0 -119
- package/src/agent-tools/plan.mjs +0 -81
- package/src/agent-tools/read-history.mjs +0 -309
- package/src/agent-tools/recent-changes.mjs +0 -24
- package/src/agent-tools/review-streak.mjs +0 -93
- package/src/agent-tools/settings.mjs +0 -265
- package/src/agent-tools/skill.mjs +0 -47
- package/src/agent-tools/subagent-actions.mjs +0 -482
- package/src/agent-tools/subagent-async.mjs +0 -434
- package/src/agent-tools/subagent-panel.mjs +0 -160
- package/src/agent-tools/subagent-run.mjs +0 -205
- package/src/agent-tools/subagent-scheduler.mjs +0 -392
- package/src/agent-tools/subagent-spawn.mjs +0 -459
- package/src/agent-tools/subagent.mjs +0 -404
- package/src/agent-tools/task.mjs +0 -87
- package/src/agent-tools/timer.mjs +0 -46
- package/src/agent-tools/verify.mjs +0 -271
- package/src/agent-tools.mjs +0 -17
- package/src/agent.mjs +0 -417
- package/src/auto-think.mjs +0 -115
- package/src/config-migrate.mjs +0 -70
- package/src/config.mjs +0 -496
- package/src/context.mjs +0 -392
- package/src/conventions.mjs +0 -223
- package/src/embedding.mjs +0 -120
- package/src/escape.mjs +0 -152
- package/src/expand-home.mjs +0 -16
- package/src/explore-distill.mjs +0 -155
- package/src/generate-title.mjs +0 -88
- package/src/git/checkpoint.mjs +0 -448
- package/src/git/gitmem.mjs +0 -100
- package/src/hooks.mjs +0 -97
- package/src/ledger.mjs +0 -227
- package/src/log.mjs +0 -195
- package/src/markdown.mjs +0 -106
- package/src/mcp/helpers.mjs +0 -51
- package/src/mcp/transport-http.mjs +0 -248
- package/src/mcp/transport-stdio.mjs +0 -140
- package/src/mcp/transport-ws.mjs +0 -122
- package/src/mcp.mjs +0 -295
- package/src/memory/code-index.mjs +0 -219
- package/src/memory/code-sync.mjs +0 -415
- package/src/memory/core.mjs +0 -299
- package/src/memory/delete.mjs +0 -236
- package/src/memory/docs.mjs +0 -419
- package/src/memory/file-walk.mjs +0 -109
- package/src/memory/scan.mjs +0 -95
- package/src/memory/schema.mjs +0 -452
- package/src/memory.mjs +0 -21
- package/src/model-ref.mjs +0 -66
- package/src/model-specs.mjs +0 -179
- package/src/peer-domains.mjs +0 -265
- package/src/peer-instances.mjs +0 -231
- package/src/prompt-overlays.mjs +0 -82
- package/src/prompts/advisor-design.md +0 -41
- package/src/prompts/advisor-round1.md +0 -41
- package/src/prompts/advisor-round2.md +0 -46
- package/src/prompts/advisor-round3.md +0 -42
- package/src/prompts/common.md +0 -115
- package/src/prompts/consult-base.md +0 -19
- package/src/prompts/discipline-engineering.md +0 -258
- package/src/prompts/discipline-normal.md +0 -185
- package/src/prompts/persona-coder.md +0 -21
- package/src/prompts/persona-eng-coder.md +0 -37
- package/src/prompts/persona-eng-designer.md +0 -60
- package/src/prompts/persona-engineering.md +0 -55
- package/src/prompts/persona-explore.md +0 -15
- package/src/prompts/persona-normal.md +0 -27
- package/src/prompts/persona-plan.md +0 -26
- package/src/provider/anthropic.mjs +0 -225
- package/src/provider/core.mjs +0 -476
- package/src/provider/errors.mjs +0 -101
- package/src/provider/google.mjs +0 -257
- package/src/provider/index.mjs +0 -7
- package/src/provider/list-models.mjs +0 -93
- package/src/provider/normalize.mjs +0 -81
- package/src/provider/rate.mjs +0 -108
- package/src/provider/responses.mjs +0 -495
- package/src/provider/retry.mjs +0 -88
- package/src/provider/sse.mjs +0 -264
- package/src/proxy.mjs +0 -261
- package/src/rules.mjs +0 -53
- package/src/session-gc.mjs +0 -221
- package/src/session-guard.mjs +0 -59
- package/src/session-migrate.mjs +0 -48
- package/src/session-rename.mjs +0 -38
- package/src/session-segments.mjs +0 -100
- package/src/session-slots.mjs +0 -492
- package/src/session-store.mjs +0 -441
- package/src/session.mjs +0 -492
- package/src/skills.mjs +0 -153
- package/src/text-budget.mjs +0 -46
- package/src/token-ttl.mjs +0 -274
- package/src/tools/apply_patch.md +0 -15
- package/src/tools/bash.md +0 -37
- package/src/tools/bash.mjs +0 -268
- package/src/tools/checklist-sync.mjs +0 -181
- package/src/tools/checklist.md +0 -13
- package/src/tools/checklist.mjs +0 -299
- package/src/tools/delete.md +0 -13
- package/src/tools/edit-batch.mjs +0 -191
- package/src/tools/edit-diff.mjs +0 -348
- package/src/tools/edit.md +0 -30
- package/src/tools/execute.md +0 -21
- package/src/tools/execute.mjs +0 -228
- package/src/tools/fetch.md +0 -12
- package/src/tools/file.mjs +0 -469
- package/src/tools/file_ops.md +0 -17
- package/src/tools/get_current_time.md +0 -8
- package/src/tools/git-checkpoint.mjs +0 -143
- package/src/tools/git-ext.mjs +0 -173
- package/src/tools/git.md +0 -54
- package/src/tools/git.mjs +0 -356
- package/src/tools/glob-dialect.mjs +0 -130
- package/src/tools/glob.md +0 -11
- package/src/tools/grep.md +0 -19
- package/src/tools/hashline_edit.md +0 -14
- package/src/tools/index.mjs +0 -36
- package/src/tools/insert_after.md +0 -15
- package/src/tools/lint.md +0 -10
- package/src/tools/linter.mjs +0 -128
- package/src/tools/ls.md +0 -12
- package/src/tools/lsp.md +0 -10
- package/src/tools/lsp.mjs +0 -316
- package/src/tools/ops.mjs +0 -299
- package/src/tools/patch.mjs +0 -282
- package/src/tools/process.md +0 -10
- package/src/tools/question.md +0 -16
- package/src/tools/question.mjs +0 -26
- package/src/tools/read.md +0 -20
- package/src/tools/read_image.md +0 -8
- package/src/tools/repomap.mjs +0 -314
- package/src/tools/search.mjs +0 -236
- package/src/tools/shared.mjs +0 -446
- package/src/tools/tree.md +0 -14
- package/src/tools/tree.mjs +0 -66
- package/src/tools/wait_for.md +0 -22
- package/src/tools/web.mjs +0 -224
- package/src/tools/websearch.md +0 -16
- package/src/tools/write.md +0 -11
- package/src/traces/trace-store.mjs +0 -355
package/src/memory/docs.mjs
DELETED
|
@@ -1,419 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* memory/docs.mjs — doc index sync, retrieval, agent tool generation
|
|
3
|
-
*/
|
|
4
|
-
|
|
5
|
-
import { readFile, stat } from "node:fs/promises"
|
|
6
|
-
import { isAbsolute, join } from "node:path"
|
|
7
|
-
import { embed, cosine, toBlob, fromBlob } from "../embedding.mjs"
|
|
8
|
-
import { scanVectors, createTopK } from "./scan.mjs"
|
|
9
|
-
import { commitAndPush } from "../git/gitmem.mjs"
|
|
10
|
-
import { MAX_DOC_FILE_BYTES } from "./schema.mjs"
|
|
11
|
-
import { buildFtsQuery, put, search, putMarkdown, clearPersonal, EMBED_TEXT_MAX_LEN } from "./core.mjs"
|
|
12
|
-
import { deleteByUid, matchMemoryRows, deleteWhere } from "./delete.mjs"
|
|
13
|
-
import { _upsertDocFile, yieldTick } from "./code-index.mjs"
|
|
14
|
-
import { markIndexedCommit, listProjectFiles, indexExtensions } from "./code-sync.mjs"
|
|
15
|
-
import { logEvent } from "../log.mjs"
|
|
16
|
-
|
|
17
|
-
const DOC_EMBED_BATCH = 64
|
|
18
|
-
|
|
19
|
-
/**
|
|
20
|
-
* Sync doc index: scan all .md/.mdc/.txt/.rst/.adoc under dir → chunk → upsert into doc_chunks.
|
|
21
|
-
* Incremental by mtime.
|
|
22
|
-
*/
|
|
23
|
-
export async function docSync(memory, dir, { onProgress } = {}) {
|
|
24
|
-
const { entries, unlisted } = await listProjectFiles(dir, indexExtensions(dir).doc)
|
|
25
|
-
const files = [] // { abs, rel, mtimeMs }
|
|
26
|
-
let overSizeSkipped = 0
|
|
27
|
-
for (const { abs, rel } of entries) {
|
|
28
|
-
let st
|
|
29
|
-
try { st = await stat(abs) } catch { continue }
|
|
30
|
-
if (st.size > MAX_DOC_FILE_BYTES) { overSizeSkipped++; continue }
|
|
31
|
-
files.push({ abs, rel, mtimeMs: Math.floor(st.mtimeMs) })
|
|
32
|
-
}
|
|
33
|
-
|
|
34
|
-
const indexed = new Map(
|
|
35
|
-
memory.db.prepare(`SELECT path, mtime_ms FROM doc_chunks WHERE origin = ?`).all(dir).map((r) => [r.path, r.mtime_ms])
|
|
36
|
-
)
|
|
37
|
-
const seen = new Set()
|
|
38
|
-
|
|
39
|
-
onProgress?.({ phase: "scan", total: files.length, overSizeSkipped })
|
|
40
|
-
|
|
41
|
-
let updated = 0, removed = 0, skipped = 0, failed = 0
|
|
42
|
-
const errors = []
|
|
43
|
-
for (let i = 0; i < files.length; i++) {
|
|
44
|
-
const { abs, rel, mtimeMs } = files[i]
|
|
45
|
-
seen.add(rel)
|
|
46
|
-
|
|
47
|
-
if (indexed.get(rel) === mtimeMs) {
|
|
48
|
-
skipped++
|
|
49
|
-
continue
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
try {
|
|
53
|
-
const text = await readFile(abs, "utf8")
|
|
54
|
-
const lines = text.split("\n")
|
|
55
|
-
_upsertDocFile(memory, dir, rel, lines, mtimeMs)
|
|
56
|
-
updated++
|
|
57
|
-
} catch (e) {
|
|
58
|
-
failed++
|
|
59
|
-
if (errors.length < 5) errors.push(`${rel}: ${e.message}`)
|
|
60
|
-
}
|
|
61
|
-
await yieldTick()
|
|
62
|
-
|
|
63
|
-
if (onProgress && i % 10 === 0) {
|
|
64
|
-
onProgress({ phase: "index", current: i + 1, total: files.length, updated, removed, skipped, failed })
|
|
65
|
-
}
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
for (const stale of indexed.keys()) {
|
|
69
|
-
if (!seen.has(stale)) {
|
|
70
|
-
memory.db.prepare(`DELETE FROM doc_chunks WHERE origin = ? AND path = ?`).run(dir, stale)
|
|
71
|
-
removed++
|
|
72
|
-
}
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
onProgress?.({ phase: "done", total: files.length, updated, removed, skipped, failed, overSizeSkipped })
|
|
76
|
-
markIndexedCommit(memory, dir)
|
|
77
|
-
if (unlisted.count > 0) {
|
|
78
|
-
logEvent("index:unlisted", { dir, kind: "doc", count: unlisted.count, exts: unlisted.exts.map((e) => e.ext) })
|
|
79
|
-
}
|
|
80
|
-
return { updated, removed, skipped, failed, errors, total: files.length, overSizeSkipped, unlistedExts: unlisted }
|
|
81
|
-
}
|
|
82
|
-
|
|
83
|
-
/**
|
|
84
|
-
* Doc search: FTS5(BM25) + optional vector cosine, RRF merged.
|
|
85
|
-
* Falls back to pure FTS when no embedder; falls back to pure vector when ftsQuery is empty and embedder is present.
|
|
86
|
-
*/
|
|
87
|
-
export async function docSearch(memory, query, { limit = 5 } = {}) {
|
|
88
|
-
const ftsQuery = buildFtsQuery(query)
|
|
89
|
-
if (!ftsQuery && !memory.embedder) return []
|
|
90
|
-
|
|
91
|
-
const ftsOriginFilter = memory.codeOrigin ? `AND d.origin = ?` : ""
|
|
92
|
-
const vecOriginFilter = memory.codeOrigin ? `AND origin = ?` : ""
|
|
93
|
-
const originParams = memory.codeOrigin ? [memory.codeOrigin] : []
|
|
94
|
-
|
|
95
|
-
const ftsList = ftsQuery ? memory.db.prepare(`
|
|
96
|
-
SELECT d.rowid, d.path, d.language, d.heading, d.content, d.line_start, d.line_end, bm25(doc_chunks_fts) AS rank
|
|
97
|
-
FROM doc_chunks_fts JOIN doc_chunks d ON d.rowid = doc_chunks_fts.rowid
|
|
98
|
-
WHERE doc_chunks_fts MATCH ? ${ftsOriginFilter}
|
|
99
|
-
ORDER BY rank LIMIT ?
|
|
100
|
-
`).all(ftsQuery, ...originParams, Math.max(limit * 4, 20)) : []
|
|
101
|
-
|
|
102
|
-
if (!memory.embedder) return ftsList.slice(0, limit)
|
|
103
|
-
|
|
104
|
-
try { await ensureDocEmbeddings(memory) } catch (e) {
|
|
105
|
-
console.error(`[docs] embedding ensure failed, falling back to FTS-only: ${e.message}`)
|
|
106
|
-
return ftsList.slice(0, limit)
|
|
107
|
-
}
|
|
108
|
-
let qvec
|
|
109
|
-
try { [qvec] = await embed(memory.embedder, [query]) } catch (e) {
|
|
110
|
-
console.error(`[docs] query embedding failed, falling back to FTS-only: ${e.message}`)
|
|
111
|
-
return ftsList.slice(0, limit)
|
|
112
|
-
}
|
|
113
|
-
// TUI-OOM-ROOTCAUSE(MEMORY.md §10.3):分块扫描 + 有界 top-K(原全表 .all()——峰值 = 块 + K)
|
|
114
|
-
const top = createTopK(Math.max(limit * 4, 20))
|
|
115
|
-
scanVectors(memory.db, `SELECT rowid, embedding FROM doc_chunks WHERE embedding IS NOT NULL ${vecOriginFilter}`, originParams, {
|
|
116
|
-
onRow: (r) => top.push({ id: r.rowid, rowid: r.rowid, score: cosine(qvec, fromBlob(r.embedding)) }),
|
|
117
|
-
})
|
|
118
|
-
const vecList = top.list().map((c) => ({ rowid: c.id, score: c.score }))
|
|
119
|
-
|
|
120
|
-
const K = 60
|
|
121
|
-
const scores = new Map()
|
|
122
|
-
ftsList.forEach((r, i) => scores.set(r.rowid, (scores.get(r.rowid) ?? 0) + 1 / (K + i + 1)))
|
|
123
|
-
vecList.forEach((r, i) => scores.set(r.rowid, (scores.get(r.rowid) ?? 0) + 1 / (K + i + 1)))
|
|
124
|
-
|
|
125
|
-
const fetchChunk = memory.db.prepare(`
|
|
126
|
-
SELECT path, language, heading, content, line_start, line_end FROM doc_chunks WHERE rowid = ?
|
|
127
|
-
`)
|
|
128
|
-
const sorted = [...scores.entries()]
|
|
129
|
-
.sort((a, b) => b[1] - a[1])
|
|
130
|
-
.slice(0, limit)
|
|
131
|
-
return sorted
|
|
132
|
-
.map(([rowid, score]) => {
|
|
133
|
-
const chunk = fetchChunk.get(rowid)
|
|
134
|
-
if (!chunk) return null
|
|
135
|
-
chunk._score = Math.round(score * 100) / 100
|
|
136
|
-
return chunk
|
|
137
|
-
})
|
|
138
|
-
.filter(Boolean)
|
|
139
|
-
}
|
|
140
|
-
|
|
141
|
-
/** Lazily backfill missing vectors for doc_chunks. Guarded against concurrent calls. */
|
|
142
|
-
let _docEmbedLock = null
|
|
143
|
-
export function ensureDocEmbeddings(memory) {
|
|
144
|
-
if (_docEmbedLock) return _docEmbedLock
|
|
145
|
-
_docEmbedLock = _runEnsureDocEmbeddings(memory).finally(() => { _docEmbedLock = null })
|
|
146
|
-
return _docEmbedLock
|
|
147
|
-
}
|
|
148
|
-
|
|
149
|
-
async function _runEnsureDocEmbeddings(memory) {
|
|
150
|
-
if (!memory.embedder) return
|
|
151
|
-
const modelKey = memory.embedder.model
|
|
152
|
-
const stored = memory.db.prepare(`SELECT value FROM meta WHERE key = 'doc_embedding_model'`).get()?.value
|
|
153
|
-
if (stored !== modelKey) {
|
|
154
|
-
memory.db.prepare(`UPDATE doc_chunks SET embedding = NULL`).run()
|
|
155
|
-
memory.db.prepare(`INSERT INTO meta (key, value) VALUES ('doc_embedding_model', ?)
|
|
156
|
-
ON CONFLICT (key) DO UPDATE SET value = excluded.value`).run(modelKey)
|
|
157
|
-
}
|
|
158
|
-
|
|
159
|
-
const pending = memory.db.prepare(`SELECT rowid, path, heading, content FROM doc_chunks WHERE embedding IS NULL LIMIT ${DOC_EMBED_BATCH}`).all()
|
|
160
|
-
if (pending.length === 0) return
|
|
161
|
-
|
|
162
|
-
const texts = pending.map((r) => `${r.heading || r.path}\n${r.content.slice(0, EMBED_TEXT_MAX_LEN)}`)
|
|
163
|
-
const vecs = await embed(memory.embedder, texts)
|
|
164
|
-
|
|
165
|
-
const update = memory.db.prepare(`UPDATE doc_chunks SET embedding = ? WHERE rowid = ?`)
|
|
166
|
-
pending.forEach((r, i) => update.run(toBlob(vecs[i]), r.rowid))
|
|
167
|
-
}
|
|
168
|
-
|
|
169
|
-
/** Generate the doc_search tool (read-only). */
|
|
170
|
-
export function docSearchTool(memory) {
|
|
171
|
-
return {
|
|
172
|
-
name: "doc_search",
|
|
173
|
-
description:
|
|
174
|
-
"Search the project's documentation (README, design docs, guides, markdown files) for relevant information. Use this to find design decisions, coding conventions, architecture docs, or project rules. Prefer this over code_search when you need to understand the project's intended design rather than existing implementation. " +
|
|
175
|
-
"Returns matching doc chunks: path, heading, line range, relevance score, content excerpt. " +
|
|
176
|
-
"For what was said in sessions (conversation/chat history — decisions, rulings), use read_history.",
|
|
177
|
-
parameters: {
|
|
178
|
-
type: "object",
|
|
179
|
-
properties: {
|
|
180
|
-
query: { type: "string", description: "Natural language search query" },
|
|
181
|
-
limit: { type: "number", description: "Max results (default 5)" },
|
|
182
|
-
},
|
|
183
|
-
required: ["query"],
|
|
184
|
-
},
|
|
185
|
-
readonly: true,
|
|
186
|
-
async execute(args) {
|
|
187
|
-
const results = await docSearch(memory, args.query, { limit: args.limit ?? 5 })
|
|
188
|
-
if (results.length === 0) return "(no matching documentation)"
|
|
189
|
-
return results.map((r) =>
|
|
190
|
-
`${r.path}${r.heading ? ` > ${r.heading}` : ""} (L${r.line_start}-L${r.line_end}, relevance ${r._score?.toFixed(2) ?? "?"}):\n${r.content.slice(0, 2000)}`
|
|
191
|
-
).join("\n\n---\n\n")
|
|
192
|
-
},
|
|
193
|
-
}
|
|
194
|
-
}
|
|
195
|
-
|
|
196
|
-
// ---------------------------------------------------------------- agent tools
|
|
197
|
-
|
|
198
|
-
/** §6 shared tool surface — action enum / parameter shapes / descriptions byte-identical
|
|
199
|
-
* with thincoder-vscode/src/memory.mjs (MEMORY.md §6 D-M1/F-M6); layer VALUES per end
|
|
200
|
-
* (VS Code has no team layer and rejects it with CLI guidance). */
|
|
201
|
-
const MEMORY_ACTIONS = ["search", "put", "list", "delete", "clear"]
|
|
202
|
-
const MEMORY_LAYERS = ["personal", "project", "team"]
|
|
203
|
-
const MEMORY_TOOL_DESCRIPTION =
|
|
204
|
-
"Manage long-term memory in ONE tool — the action parameter picks the operation:\n" +
|
|
205
|
-
"- search — find knowledge saved in previous sessions (query, optional layer/limit); result rows start with a [layer] tag and carry the entry id (id prefix = the layer)\n" +
|
|
206
|
-
"- put — save a piece of knowledge for future sessions (type: rule = coding standards, knowledge = project facts, decision = architecture decisions, pattern = debugging/workflow patterns; title/content/tags; layer defaults to personal)\n" +
|
|
207
|
-
"- list — inventory what memory holds (optional layer/type/keyword filters, limit default 50); one row per entry: [layer] id [type] title (date); a truncated list notes the full count\n" +
|
|
208
|
-
"- delete — SINGLE: {id, layer} removes the entry shown in put/search/list output — layer is optional: when passed it is validated against the id prefix (a mismatch is refused — guards against deleting the wrong entry); when omitted the id prefix routes the delete, so any id search/list returned is directly deletable. BATCH (no id): {layer, type and/or keyword} removes every matching entry in that layer — layer and at least one of type/keyword are required, plus confirm:true (without confirm it returns the count plus a preview); layer-wide wipes without filters are refused on every layer\n" +
|
|
209
|
-
"- clear — {layer: \"personal\", confirm: true} wipes ALL personal memory entries. clear is personal-only: a missing layer or a project/team layer is refused (use delete batch filters on shared layers)\n" +
|
|
210
|
-
"Layer is the memory tier: personal (private), project (shared via this repo's .thincoder/memory/), team (CLI only, git-synced). The [layer] tag on search/list result rows, the row's id prefix, and the layer parameter are the same concept — pass a result row's [layer] as layer, or omit it on a single delete to auto-route by the id prefix.\n" +
|
|
211
|
-
"Deleting project/team (CLI) entries removes the local markdown file and its index row — team deletion is local only and a later team sync may resurrect the file while the remote still has it.\n" +
|
|
212
|
-
"Save bugs, conventions, and preferences here — they persist across sessions.\n" +
|
|
213
|
-
"Session message history (what was said in this or past sessions) is NOT in memory — search session messages with read_history."
|
|
214
|
-
|
|
215
|
-
function validateTypeFilter(type) {
|
|
216
|
-
if (type === undefined || type === null || type === "") return null
|
|
217
|
-
const t = String(type)
|
|
218
|
-
if (!["rule", "knowledge", "decision", "pattern"].includes(t)) throw new Error(`Invalid memory type "${t}"; expected one of: rule, knowledge, decision, pattern`)
|
|
219
|
-
return t
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
function normalizeLimit(limit, dflt) {
|
|
223
|
-
const n = Number(limit)
|
|
224
|
-
return Number.isFinite(n) && n > 0 ? Math.floor(n) : dflt
|
|
225
|
-
}
|
|
226
|
-
|
|
227
|
-
function fmtDate(ts) {
|
|
228
|
-
return ts ? new Date(ts).toISOString().slice(0, 10) : "?"
|
|
229
|
-
}
|
|
230
|
-
|
|
231
|
-
const listRowLine = (r) => `[${r.layer}] ${r.id} [${r.type}] ${r.title}(${fmtDate(r.ts)})`
|
|
232
|
-
|
|
233
|
-
/**
|
|
234
|
-
* Generate the memory agent tool — ONE `memory` tool with five actions (MEMORY.md §6 D-M1).
|
|
235
|
-
* search/list are read-only actions (planMode pass / no permission ask — dispatch classifies
|
|
236
|
-
* them action-level, same as subagent check/status); put keeps its side-effect permission
|
|
237
|
-
* gate; batch delete/clear gate on confirm:true + layer inside the tool (direct-delete
|
|
238
|
-
* ruling — the confirm parameter IS the gate) and stay non-readonly like the retired tools.
|
|
239
|
-
* opts: { cwd, projectDir, author, team: { dir, name } | null }
|
|
240
|
-
*/
|
|
241
|
-
export function memoryTools(memory, opts = {}) {
|
|
242
|
-
const projectDir = opts.projectDir ? (isAbsolute(opts.projectDir) ? opts.projectDir : join(opts.cwd ?? process.cwd(), opts.projectDir)) : null
|
|
243
|
-
const dirs = { project: projectDir, team: opts.team?.dir ?? null }
|
|
244
|
-
return [
|
|
245
|
-
{
|
|
246
|
-
name: "memory",
|
|
247
|
-
description: MEMORY_TOOL_DESCRIPTION,
|
|
248
|
-
parameters: {
|
|
249
|
-
type: "object",
|
|
250
|
-
properties: {
|
|
251
|
-
action: { type: "string", enum: MEMORY_ACTIONS, description: "Operation to run (required)" },
|
|
252
|
-
layer: { type: "string", enum: MEMORY_LAYERS, description: "The memory layer: personal (private), project (shared via this repo's .thincoder/memory/), team (CLI only). Same concept as the [layer] tag and the id prefix on search/list result rows. put/search/list: optional (put defaults to personal; search/list omit = all layers). single delete: optional (omit = route by id prefix). batch delete/clear: required" },
|
|
253
|
-
type: { type: "string", enum: ["rule", "knowledge", "decision", "pattern"], description: "Entry type: put = what to save; list/delete batch = filter by type" },
|
|
254
|
-
title: { type: "string", description: "put: short title" },
|
|
255
|
-
content: { type: "string", description: "put: full content to remember" },
|
|
256
|
-
tags: { type: "string", description: "put: space-separated tags" },
|
|
257
|
-
query: { type: "string", description: "search: natural-language query" },
|
|
258
|
-
keyword: { type: "string", description: "list/delete batch: filter matching title/content" },
|
|
259
|
-
id: { type: "string", description: "delete single: the entry id from put/search/list output" },
|
|
260
|
-
limit: { type: "number", description: "Max rows: list 50 by default, search 5 by default" },
|
|
261
|
-
confirm: { type: "boolean", description: "delete batch/clear: must be true — without it the tool refuses" },
|
|
262
|
-
},
|
|
263
|
-
required: ["action"],
|
|
264
|
-
},
|
|
265
|
-
readonly: false,
|
|
266
|
-
async execute(args) {
|
|
267
|
-
const action = String(args?.action ?? "")
|
|
268
|
-
if (!MEMORY_ACTIONS.includes(action)) {
|
|
269
|
-
throw new Error(`memory: unknown action "${action}" — expected one of: ${MEMORY_ACTIONS.join("/")}`)
|
|
270
|
-
}
|
|
271
|
-
switch (action) {
|
|
272
|
-
case "search": return execSearch(memory, args)
|
|
273
|
-
case "put": return execPut(memory, args, opts, dirs)
|
|
274
|
-
case "list": return execList(memory, args, dirs)
|
|
275
|
-
case "delete": return execDelete(memory, args, dirs)
|
|
276
|
-
case "clear": return execClear(memory, args)
|
|
277
|
-
}
|
|
278
|
-
},
|
|
279
|
-
},
|
|
280
|
-
]
|
|
281
|
-
}
|
|
282
|
-
|
|
283
|
-
/** action search — the retired search tool surface (read-only, same output contract). */
|
|
284
|
-
async function execSearch(memory, args) {
|
|
285
|
-
const layer = args.layer
|
|
286
|
-
if (layer !== undefined && layer !== null && !MEMORY_LAYERS.includes(String(layer))) {
|
|
287
|
-
throw new Error(`memory search: invalid layer "${layer}"`)
|
|
288
|
-
}
|
|
289
|
-
const query = String(args.query ?? "").trim()
|
|
290
|
-
if (!query) return "(no matching memories)" // 空 query 短路——两端同语义(评审 code review #4)
|
|
291
|
-
const limit = normalizeLimit(args.limit, 5)
|
|
292
|
-
let results
|
|
293
|
-
if (!layer) {
|
|
294
|
-
results = await search(memory, query, { limit })
|
|
295
|
-
} else {
|
|
296
|
-
// layer filter: oversample then slice the requested layer (results keep global rank order).
|
|
297
|
-
// 窗口 = max(limit*4, 20) 是召回上限——大库 + 高 limit 时该层结果可能不足 limit(接受的取舍——评审 code review #3)
|
|
298
|
-
const wide = await search(memory, query, { limit: Math.max(limit * 4, 20) })
|
|
299
|
-
results = wide.filter((r) => r.layer === String(layer)).slice(0, limit)
|
|
300
|
-
}
|
|
301
|
-
if (results.length === 0) return "(no matching memories)"
|
|
302
|
-
return results.map((r) => `[${r.layer}][${r.type}] ${r.title} (id=${r.id})\n${r.content}`).join("\n\n")
|
|
303
|
-
}
|
|
304
|
-
|
|
305
|
-
/** action put — the retired put tool surface (side-effect gate, unchanged semantics). */
|
|
306
|
-
async function execPut(memory, args, opts, dirs) {
|
|
307
|
-
const layer = String(args.layer ?? "personal")
|
|
308
|
-
if (!MEMORY_LAYERS.includes(layer)) throw new Error(`memory put: invalid layer "${layer}"`)
|
|
309
|
-
if (layer === "personal") {
|
|
310
|
-
const id = await put(memory, { type: args.type, title: args.title, content: args.content, tags: args.tags ?? "" })
|
|
311
|
-
return `Saved to personal memory (id=personal:${id}): [${args.type}] ${args.title}`
|
|
312
|
-
}
|
|
313
|
-
if (layer === "project") {
|
|
314
|
-
if (!dirs.project) throw new Error("project layer unavailable: no project directory configured")
|
|
315
|
-
const filename = await putMarkdown(memory, {
|
|
316
|
-
layer: "project",
|
|
317
|
-
dir: dirs.project,
|
|
318
|
-
type: args.type,
|
|
319
|
-
title: args.title,
|
|
320
|
-
content: args.content,
|
|
321
|
-
tags: (args.tags ?? "").split(/\s+/).filter(Boolean),
|
|
322
|
-
author: opts.author ?? "unknown",
|
|
323
|
-
})
|
|
324
|
-
return `Saved to project memory (id=project:${dirs.project}:${filename}): [${args.type}] ${args.title}`
|
|
325
|
-
}
|
|
326
|
-
if (!dirs.team) {
|
|
327
|
-
throw new Error("team layer not configured: set memory.team in ~/.thincoder/config.json")
|
|
328
|
-
}
|
|
329
|
-
const filename = await putMarkdown(memory, {
|
|
330
|
-
layer: "team",
|
|
331
|
-
dir: dirs.team,
|
|
332
|
-
type: args.type,
|
|
333
|
-
title: args.title,
|
|
334
|
-
content: args.content,
|
|
335
|
-
tags: (args.tags ?? "").split(/\s+/).filter(Boolean),
|
|
336
|
-
author: opts.author ?? "unknown",
|
|
337
|
-
})
|
|
338
|
-
await commitAndPush(dirs.team, filename, `memory: [${args.type}] ${args.title}`)
|
|
339
|
-
return `Saved to team memory and pushed (id=team:${dirs.team}:${filename}): [${args.type}] ${args.title}`
|
|
340
|
-
}
|
|
341
|
-
|
|
342
|
-
/** action list — new inventory action (read-only): layer/type/keyword filters + limit truncation note. */
|
|
343
|
-
async function execList(memory, args, dirs) {
|
|
344
|
-
const layer = args.layer ?? null
|
|
345
|
-
if (layer && !MEMORY_LAYERS.includes(String(layer))) throw new Error(`memory list: invalid layer "${layer}"`)
|
|
346
|
-
const rows = await matchMemoryRows(memory, {
|
|
347
|
-
layer: layer ? String(layer) : null,
|
|
348
|
-
type: validateTypeFilter(args.type),
|
|
349
|
-
keyword: args.keyword ? String(args.keyword).trim() : null,
|
|
350
|
-
projectDir: dirs.project,
|
|
351
|
-
teamDir: dirs.team,
|
|
352
|
-
})
|
|
353
|
-
if (rows.length === 0) return "0 条匹配"
|
|
354
|
-
const limit = normalizeLimit(args.limit, 50)
|
|
355
|
-
const shown = rows.slice(0, limit)
|
|
356
|
-
const lines = shown.map(listRowLine)
|
|
357
|
-
if (rows.length > shown.length) lines.unshift(`${shown.length} 条——截断前 ${rows.length}`)
|
|
358
|
-
return lines.join("\n")
|
|
359
|
-
}
|
|
360
|
-
|
|
361
|
-
/** action delete — single ({ id, layer? } — MEMORY.md §6.2: layer OPTIONAL, validated when
|
|
362
|
-
* passed, else the id prefix routes the delete) + batch (layer + type/keyword + confirm). */
|
|
363
|
-
async function execDelete(memory, args, dirs) {
|
|
364
|
-
const hasId = args.id !== undefined && args.id !== null && String(args.id) !== ""
|
|
365
|
-
if (hasId) return execDeleteSingle(memory, args, dirs)
|
|
366
|
-
// batch form
|
|
367
|
-
const layer = args.layer
|
|
368
|
-
if (!layer) throw new Error("batch delete requires layer plus type and/or keyword filter")
|
|
369
|
-
if (!MEMORY_LAYERS.includes(String(layer))) throw new Error(`memory delete: invalid layer "${layer}"`)
|
|
370
|
-
const type = validateTypeFilter(args.type)
|
|
371
|
-
const keyword = args.keyword ? String(args.keyword).trim() : null
|
|
372
|
-
if (!type && !keyword) {
|
|
373
|
-
throw new Error("batch delete requires type and/or keyword filter — a layer-wide wipe without filters is refused (personal full wipe is the clear action)")
|
|
374
|
-
}
|
|
375
|
-
if (layer === "project" && !dirs.project) throw new Error("project layer unavailable: no project directory configured")
|
|
376
|
-
if (layer === "team" && !dirs.team) throw new Error("team layer not configured: set memory.team in ~/.thincoder/config.json")
|
|
377
|
-
const filters = { layer: String(layer), type, keyword }
|
|
378
|
-
const rows = await matchMemoryRows(memory, { ...filters, projectDir: dirs.project, teamDir: dirs.team })
|
|
379
|
-
if (rows.length === 0) return "0 条匹配"
|
|
380
|
-
if (args.confirm !== true) {
|
|
381
|
-
const lines = [rows.length > 5 ? `将删 ${rows.length} 条:前 5 条预览` : `将删 ${rows.length} 条`]
|
|
382
|
-
lines.push(...rows.slice(0, 5).map(listRowLine))
|
|
383
|
-
if (rows.length > 5) lines.push(`5 条——截断前 ${rows.length}`)
|
|
384
|
-
lines.push("confirm:true required — re-send with it to execute the deletion")
|
|
385
|
-
return lines.join("\n")
|
|
386
|
-
}
|
|
387
|
-
const n = await deleteWhere(memory, filters, { dirs })
|
|
388
|
-
return `Deleted ${n} entries in layer ${layer}`
|
|
389
|
-
}
|
|
390
|
-
|
|
391
|
-
/** Single-entry delete — MEMORY.md §6.2: layer is OPTIONAL. When passed it is validated
|
|
392
|
-
* against the id prefix (mismatch refused — guards against deleting the wrong entry); when
|
|
393
|
-
* omitted the delete routes by the id prefix alone, so any id search/list returned is
|
|
394
|
-
* directly deletable (deleteByUid already resolves the layer from the uid prefix). */
|
|
395
|
-
async function execDeleteSingle(memory, args, dirs) {
|
|
396
|
-
const uid = String(args.id)
|
|
397
|
-
const prefix = uid.split(":")[0]
|
|
398
|
-
const uidLayer = prefix === "personal" || prefix === "project" || prefix === "team" ? prefix : /^\d+$/.test(prefix) ? "personal" : null
|
|
399
|
-
if (!uidLayer) throw new Error(`invalid memory id: ${uid}`)
|
|
400
|
-
const layer = args.layer
|
|
401
|
-
if (layer !== undefined && layer !== null && String(layer) !== uidLayer) {
|
|
402
|
-
throw new Error(`id prefix ${prefix}: 与 layer ${layer} 不匹配`)
|
|
403
|
-
}
|
|
404
|
-
const entry = await deleteByUid(memory, uid, { dirs })
|
|
405
|
-
return `Deleted ${entry.id}: ${entry.title}\n${(entry.content ?? "").slice(0, 500)}`
|
|
406
|
-
}
|
|
407
|
-
|
|
408
|
-
/** action clear — personal-only full wipe (layer + confirm:true gates; project/team refused). */
|
|
409
|
-
function execClear(memory, args) {
|
|
410
|
-
const layer = args.layer
|
|
411
|
-
if (!layer) throw new Error('clear requires layer "personal" — pass layer: "personal" plus confirm: true')
|
|
412
|
-
if (String(layer) !== "personal") {
|
|
413
|
-
if (!MEMORY_LAYERS.includes(String(layer))) throw new Error(`memory clear: invalid layer "${layer}"`)
|
|
414
|
-
throw new Error("shared layers don't support clear — use delete with type/keyword batch filters instead")
|
|
415
|
-
}
|
|
416
|
-
if (args.confirm !== true) throw new Error("clear requires confirm:true — this wipes ALL personal memory")
|
|
417
|
-
const n = clearPersonal(memory)
|
|
418
|
-
return `Cleared personal memory (${n} entries deleted)`
|
|
419
|
-
}
|
package/src/memory/file-walk.mjs
DELETED
|
@@ -1,109 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* memory/file-walk.mjs — filesystem walk fallback + the shared file-listing predicates.
|
|
3
|
-
*
|
|
4
|
-
* Why this module (PORTABILITY FR15 / P8): the index was built from `git ls-files`
|
|
5
|
-
* only — on a project that is not a git repository the code and doc indexes came
|
|
6
|
-
* back EMPTY, silently (the reviewer/agent then "searched" an empty index). The
|
|
7
|
-
* fallback traverses the filesystem instead, and the skip predicate lives here so
|
|
8
|
-
* the git path, the walk and the single-file reindex cannot drift apart against
|
|
9
|
-
* each other (they were three copies before).
|
|
10
|
-
*/
|
|
11
|
-
import { readdir } from "node:fs/promises"
|
|
12
|
-
import { join } from "node:path"
|
|
13
|
-
import { SKIP_DIRS } from "./schema.mjs"
|
|
14
|
-
|
|
15
|
-
/** Upper guard for the walk: stop after this many matched files (runaway trees). */
|
|
16
|
-
export const MAX_WALK_FILES = 20000
|
|
17
|
-
|
|
18
|
-
/**
|
|
19
|
-
* Shared skip predicate for project-relative paths: any SKIP_DIRS basename, or any
|
|
20
|
-
* dot-prefixed segment (`.git`, `.cache`, `.thincoder`…). Accepts both separators.
|
|
21
|
-
*/
|
|
22
|
-
export function isSkippedRelPath(rel) {
|
|
23
|
-
return String(rel ?? "")
|
|
24
|
-
.replace(/\\/g, "/")
|
|
25
|
-
.split("/")
|
|
26
|
-
.some((seg) => seg !== "" && (seg.startsWith(".") || SKIP_DIRS.has(seg)))
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
/** Lower-cased ".ext" of a path, or "" when it has no extension. */
|
|
30
|
-
export function extensionOf(p) {
|
|
31
|
-
const s = String(p ?? "")
|
|
32
|
-
const base = s.slice(Math.max(s.lastIndexOf("/"), s.lastIndexOf("\\")) + 1)
|
|
33
|
-
const i = base.lastIndexOf(".")
|
|
34
|
-
return i > 0 ? base.slice(i).toLowerCase() : ""
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
/**
|
|
38
|
-
* Unlisted-extension tally (PORTABILITY PO-9 — visibility): counts files the index
|
|
39
|
-
* will NOT pick up because their extension is in no list, so "my .xyz files are not
|
|
40
|
-
* searchable" stops being invisible. `exts` is a capped sample (the count is the
|
|
41
|
-
* signal; the list is the hint).
|
|
42
|
-
* @param {Set<string>} knownExts — every extension that IS indexed (code ∪ doc ∪ declared)
|
|
43
|
-
*/
|
|
44
|
-
export function createUnlistedTally(knownExts) {
|
|
45
|
-
const counts = new Map()
|
|
46
|
-
let count = 0
|
|
47
|
-
return {
|
|
48
|
-
note(rel) {
|
|
49
|
-
const ext = extensionOf(rel)
|
|
50
|
-
if (!ext || knownExts.has(ext)) return
|
|
51
|
-
count++
|
|
52
|
-
counts.set(ext, (counts.get(ext) ?? 0) + 1)
|
|
53
|
-
},
|
|
54
|
-
result(limit = 12) {
|
|
55
|
-
const exts = [...counts.entries()]
|
|
56
|
-
.sort((a, b) => b[1] - a[1])
|
|
57
|
-
.slice(0, limit)
|
|
58
|
-
.map(([ext, n]) => ({ ext, count: n }))
|
|
59
|
-
return { count, exts }
|
|
60
|
-
},
|
|
61
|
-
}
|
|
62
|
-
}
|
|
63
|
-
|
|
64
|
-
/**
|
|
65
|
-
* Recursive project walk — the no-git listing source. Symlinks are never followed
|
|
66
|
-
* (loop/escape guard); skipped directories are pruned; only files whose extension
|
|
67
|
-
* is in `exts` are returned. Hitting maxFiles is reported as `truncated`, never
|
|
68
|
-
* silently dropped.
|
|
69
|
-
* @param {string} dir — project root
|
|
70
|
-
* @param {Set<string>} exts — extensions to keep (lower-case, leading dot)
|
|
71
|
-
* @param {{maxFiles?: number, knownExts?: Set<string>|null}} [opts]
|
|
72
|
-
* knownExts → also tally files whose extension is in NO index list
|
|
73
|
-
* @returns {Promise<{files: {abs: string, rel: string}[], truncated: boolean,
|
|
74
|
-
* unlisted: {count: number, exts: {ext: string, count: number}[]}}>}
|
|
75
|
-
*/
|
|
76
|
-
export async function walkProjectFiles(dir, exts, { maxFiles = MAX_WALK_FILES, knownExts = null } = {}) {
|
|
77
|
-
const files = []
|
|
78
|
-
const tally = knownExts ? createUnlistedTally(knownExts) : null
|
|
79
|
-
let truncated = false
|
|
80
|
-
const stack = [{ abs: dir, rel: "" }]
|
|
81
|
-
while (stack.length > 0) {
|
|
82
|
-
const cur = stack.pop()
|
|
83
|
-
let entries
|
|
84
|
-
try {
|
|
85
|
-
entries = await readdir(cur.abs, { withFileTypes: true })
|
|
86
|
-
} catch { continue /* unreadable dir — skip, the walk must not fail the sync */ }
|
|
87
|
-
for (const ent of entries) {
|
|
88
|
-
const rel = cur.rel ? `${cur.rel}/${ent.name}` : ent.name
|
|
89
|
-
if (ent.isSymbolicLink()) continue // never follow symlinks
|
|
90
|
-
if (ent.isDirectory()) {
|
|
91
|
-
if (!isSkippedRelPath(rel)) stack.push({ abs: join(cur.abs, ent.name), rel })
|
|
92
|
-
continue
|
|
93
|
-
}
|
|
94
|
-
if (!ent.isFile() || isSkippedRelPath(rel)) continue
|
|
95
|
-
const ext = extensionOf(ent.name)
|
|
96
|
-
if (!ext || !exts.has(ext)) {
|
|
97
|
-
tally?.note(rel)
|
|
98
|
-
continue
|
|
99
|
-
}
|
|
100
|
-
if (files.length >= maxFiles) {
|
|
101
|
-
truncated = true
|
|
102
|
-
break
|
|
103
|
-
}
|
|
104
|
-
files.push({ abs: join(cur.abs, ent.name), rel })
|
|
105
|
-
}
|
|
106
|
-
if (truncated) break
|
|
107
|
-
}
|
|
108
|
-
return { files, truncated, unlisted: tally ? tally.result() : { count: 0, exts: [] } }
|
|
109
|
-
}
|
package/src/memory/scan.mjs
DELETED
|
@@ -1,95 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* memory/scan.mjs — 检索向量通道的分块扫描 + 有界 top-K(TUI-OOM-ROOTCAUSE 批——MEMORY.md §10)。
|
|
3
|
-
*
|
|
4
|
-
* 病灶(§10.1):三张表的向量通道全表 `.all()` 物化后逐行 cosine + 全量排序——无 SQL LIMIT、
|
|
5
|
-
* 无分块(本机 memory.db ~736MB,embedding BLOB 为体量主源)。本模块把扫描改为
|
|
6
|
-
* **rowid 游标分块**(块 `SCAN_CHUNK_ROWS`)+ **有界 top-K**(升序小顶堆——候选集
|
|
7
|
-
* ≤ max(limit×4, 20) 既有口径):峰值 = 块 + K;召回语义不变(仍全表评分,D-M1/D-M4)。
|
|
8
|
-
*
|
|
9
|
-
* 对外结构不变:调用方拿到的仍是 `[{id, score}]` 降序候选(RRF 融合输入)——并列分数按
|
|
10
|
-
* 既有排序稳定性规则(先到先留——与「全量稳定 sort + slice(0,K)」等价)。
|
|
11
|
-
*
|
|
12
|
-
* 可测缝(N-M1):`scanVectors` 接受注入的 `runChunkedQuery`(默认真实 DB 实现——
|
|
13
|
-
* `AND rowid > ? ORDER BY rowid LIMIT ?`);测试以假数据源直测块大小/top-K/等价性
|
|
14
|
-
* (不建真实大表)。
|
|
15
|
-
*/
|
|
16
|
-
|
|
17
|
-
/** 单块行数(单源——三处共用;D-M2:块内存量级 KB~MB,随维度有界)。 */
|
|
18
|
-
export const SCAN_CHUNK_ROWS = 2_000
|
|
19
|
-
|
|
20
|
-
/**
|
|
21
|
-
* 有界 top-K(升序小顶堆):`push({id, score})` 摊销 O(log K);`list()` 返回降序
|
|
22
|
-
* `[{id, score}]`。并列分数先到先留(与全量稳定排序一致)。
|
|
23
|
-
*/
|
|
24
|
-
export function createTopK(k) {
|
|
25
|
-
const cap = Math.max(1, k)
|
|
26
|
-
const heap = [] // 升序小顶堆(堆顶 = 当前最差)
|
|
27
|
-
let seq = 0
|
|
28
|
-
|
|
29
|
-
/** a 比 b 更差?(分低者差;同分 → 迟到者差(seq 大)——堆顶恒为最差) */
|
|
30
|
-
const worse = (a, b) => a.score < b.score || (a.score === b.score && a.seq > b.seq)
|
|
31
|
-
/** a 严格优于 b?(同分不互优——先到先留) */
|
|
32
|
-
const strictlyBetter = (a, b) => a.score > b.score || (a.score === b.score && a.seq < b.seq)
|
|
33
|
-
const siftUp = (i) => {
|
|
34
|
-
while (i > 0) {
|
|
35
|
-
const p = (i - 1) >> 1
|
|
36
|
-
if (worse(heap[i], heap[p])) { const t = heap[i]; heap[i] = heap[p]; heap[p] = t; i = p; continue }
|
|
37
|
-
break
|
|
38
|
-
}
|
|
39
|
-
}
|
|
40
|
-
const siftDown = (i) => {
|
|
41
|
-
for (;;) {
|
|
42
|
-
const l = i * 2 + 1
|
|
43
|
-
const r = l + 1
|
|
44
|
-
let worst = i
|
|
45
|
-
if (l < heap.length && worse(heap[l], heap[worst])) worst = l
|
|
46
|
-
if (r < heap.length && worse(heap[r], heap[worst])) worst = r
|
|
47
|
-
if (worst === i) break
|
|
48
|
-
const t = heap[i]; heap[i] = heap[worst]; heap[worst] = t
|
|
49
|
-
i = worst
|
|
50
|
-
}
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
return {
|
|
54
|
-
get size() { return heap.length },
|
|
55
|
-
push(item) {
|
|
56
|
-
const node = { id: item.id, score: item.score, seq: seq++ }
|
|
57
|
-
if (heap.length < cap) { heap.push(node); siftUp(heap.length - 1); return true }
|
|
58
|
-
if (!strictlyBetter(node, heap[0])) return false // 不优于当前最差(含同分迟到)→ 丢弃(先到先留)
|
|
59
|
-
heap[0] = node
|
|
60
|
-
siftDown(0)
|
|
61
|
-
return true
|
|
62
|
-
},
|
|
63
|
-
/** 降序 [{id, score}](K 上限——与「全量 sort desc + slice(0,K)」逐条等价)。 */
|
|
64
|
-
list() {
|
|
65
|
-
return [...heap]
|
|
66
|
-
.sort((a, b) => (b.score - a.score) || (a.seq - b.seq))
|
|
67
|
-
.map((n) => ({ id: n.id, score: n.score }))
|
|
68
|
-
},
|
|
69
|
-
}
|
|
70
|
-
}
|
|
71
|
-
|
|
72
|
-
/**
|
|
73
|
-
* 分块扫描(rowid 游标):逐块物化 → 逐行 `onRow` → 块内存随迭代释放。
|
|
74
|
-
* @param {object} db sqlite 句柄(`.prepare(sql).all(...params)`)
|
|
75
|
-
* @param {string} sql 单表查询(须含 `rowid` 列;不含 LIMIT)
|
|
76
|
-
* @param {Array} params 绑定参数(`?` 占位——顺序与 SQL 一致)
|
|
77
|
-
* @param {object} opts `{ chunk = SCAN_CHUNK_ROWS, onRow, runChunkedQuery }`
|
|
78
|
-
* @returns {number} 扫描到的行数
|
|
79
|
-
*/
|
|
80
|
-
export function scanVectors(db, sql, params = [], { chunk = SCAN_CHUNK_ROWS, onRow = null, runChunkedQuery = null } = {}) {
|
|
81
|
-
const run = runChunkedQuery ?? ((after, take) =>
|
|
82
|
-
db.prepare(`${sql} AND rowid > ? ORDER BY rowid LIMIT ?`).all(...params, after, take))
|
|
83
|
-
let after = 0
|
|
84
|
-
let total = 0
|
|
85
|
-
for (;;) {
|
|
86
|
-
const rows = run(after, chunk) ?? []
|
|
87
|
-
if (rows.length === 0) break
|
|
88
|
-
for (const r of rows) { total++; onRow?.(r) }
|
|
89
|
-
const next = rows[rows.length - 1]?.rowid
|
|
90
|
-
if (next === undefined || next === null || !(next > after)) break // 防御:游标不前进即止(不空转)
|
|
91
|
-
after = next
|
|
92
|
-
if (rows.length < chunk) break // 尾块
|
|
93
|
-
}
|
|
94
|
-
return total
|
|
95
|
-
}
|