claude-mem-lite 6.8.1 → 6.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/adopt-cli.mjs +11 -0
- package/bash-utils.mjs +17 -1
- package/claudemd.mjs +95 -7
- package/hook-context.mjs +48 -4
- package/hook-llm.mjs +11 -4
- package/hook-memory.mjs +5 -6
- package/lib/activity.mjs +7 -2
- package/lib/db-unusable.mjs +43 -0
- package/lib/deferred-work.mjs +7 -2
- package/lib/file-edge-match.mjs +46 -2
- package/lib/import-jsonl.mjs +11 -4
- package/lib/injected-ids.mjs +67 -8
- package/lib/save-observation.mjs +7 -3
- package/lib/scrub-record.mjs +66 -2
- package/mem-cli.mjs +7 -0
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
- package/scripts/pre-tool-recall.js +10 -2
- package/scripts/prompt-search-utils.mjs +19 -2
- package/scripts/user-prompt-search.js +10 -0
- package/server.mjs +12 -1
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
"plugins": [
|
|
10
10
|
{
|
|
11
11
|
"name": "claude-mem-lite",
|
|
12
|
-
"version": "6.8.
|
|
12
|
+
"version": "6.8.3",
|
|
13
13
|
"source": "./",
|
|
14
14
|
"homepage": "https://github.com/sdsrss/claude-mem-lite",
|
|
15
15
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. Hybrid FTS5 + TF-IDF search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.8.
|
|
3
|
+
"version": "6.8.3",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. Hybrid FTS5 + TF-IDF search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "sdsrss"
|
package/adopt-cli.mjs
CHANGED
|
@@ -425,6 +425,13 @@ function unadoptAll(args) {
|
|
|
425
425
|
log(`[unadopt --all] ${dir} → cleaned partial residue (detail doc/state, no block)`);
|
|
426
426
|
partial++;
|
|
427
427
|
}
|
|
428
|
+
// OUTSIDE the branch, because residue is orthogonal to what happened to the block: a
|
|
429
|
+
// project can have its block removed AND still carry an unpaired sentinel. An orphan is
|
|
430
|
+
// the one kind of residue the sweep cannot finish — its block has no end marker, so its
|
|
431
|
+
// extent is unknowable — and the two lines above would otherwise imply the project is
|
|
432
|
+
// clean. Inside the else-branch it was also unreachable for the 'removed' case, which is
|
|
433
|
+
// how the print survived a mutation with the whole suite green (pre-ship review P2-2).
|
|
434
|
+
if (r.residue) log(` ⚠ ${r.residue}`);
|
|
428
435
|
}
|
|
429
436
|
|
|
430
437
|
// 2. Legacy memory-dir cleanup across every memdir (foreign-content guarded).
|
|
@@ -481,4 +488,8 @@ export function cmdUnadopt(args = []) {
|
|
|
481
488
|
const mig = migrateLegacyMemoryDir(cwd, PLUGIN_SLUG, { force });
|
|
482
489
|
const migNote = mig.action === 'removed' ? ' (+cleaned legacy memdir)' : '';
|
|
483
490
|
log(`[unadopt] ${cwd} → ${r.action}${migNote}`);
|
|
491
|
+
// 'partial' is the outcome that used to print as 'absent': the sidecar files are gone but
|
|
492
|
+
// an unpaired sentinel still holds steering text in the user's CLAUDE.md, and only they can
|
|
493
|
+
// decide where that text ends. Silence here is what let it survive every sweep.
|
|
494
|
+
if (r.residue) log(` ⚠ ${r.residue}`);
|
|
484
495
|
}
|
package/bash-utils.mjs
CHANGED
|
@@ -2,6 +2,13 @@
|
|
|
2
2
|
// Extracted from utils.mjs for focused responsibility
|
|
3
3
|
|
|
4
4
|
import { basename } from 'path';
|
|
5
|
+
// One import into a module that had exactly one, and it buys the single home for
|
|
6
|
+
// the file_path/notebook_path rule (lib/file-edge-match.mjs's own header: "a second
|
|
7
|
+
// copy is exactly what produced R12 B-1"). No cycle: file-edge-match reaches only
|
|
8
|
+
// project-utils + scrub-record -> secret-scrub -> private-strip, none of which
|
|
9
|
+
// import this file. Cold-start scripts are unaffected — scripts/pre-tool-recall.js
|
|
10
|
+
// deliberately imports nothing from the utils.mjs barrel that re-exports this.
|
|
11
|
+
import { toolEditPath } from './lib/file-edge-match.mjs';
|
|
5
12
|
|
|
6
13
|
// Read/search commands whose output legitimately contains "error"-like keywords without
|
|
7
14
|
// being a failure. Matched against the PRIMARY command (see isReadOnlyCommand).
|
|
@@ -441,7 +448,16 @@ export function extractFilePaths(input) {
|
|
|
441
448
|
// Direct fields (Edit/Write file_path) are kept unconditionally — an explicit edit to a
|
|
442
449
|
// /tmp path is real work the user chose to make, unlike a /tmp path that merely appears as
|
|
443
450
|
// a transient argument inside a Bash command (excluded as noise in the command branch below).
|
|
444
|
-
|
|
451
|
+
//
|
|
452
|
+
// `toolEditPath`, not a fourth hand-spelling of the same rule: this function knew
|
|
453
|
+
// file_path/path/filePath and not `notebook_path`, while hooks.json matches PostToolUse
|
|
454
|
+
// on `Edit|Write|NotebookEdit` and EDIT_TOOLS already counts NotebookEdit as significant.
|
|
455
|
+
// A notebook edit therefore produced a captured, significant episode entry carrying NO
|
|
456
|
+
// files, so the observation built from it got no observation_files edge and no file-keyed
|
|
457
|
+
// recall could reach it. Same root cause as R12 B-2; the fourth site, and the one its
|
|
458
|
+
// own follow-up note did not name.
|
|
459
|
+
const editedPath = toolEditPath(input);
|
|
460
|
+
if (editedPath) paths.push(editedPath);
|
|
445
461
|
if (input.path) paths.push(input.path);
|
|
446
462
|
if (input.filePath) paths.push(input.filePath);
|
|
447
463
|
if (input.command) {
|
package/claudemd.mjs
CHANGED
|
@@ -59,8 +59,50 @@ function escapeRe(s) {
|
|
|
59
59
|
// are `\r?\n` (not bare `\n`) so a CLAUDE.md re-saved with Windows CRLF endings
|
|
60
60
|
// still matches — otherwise the block read as "absent" and a fresh LF copy got
|
|
61
61
|
// appended every SessionStart, growing the file without bound (review C1/H2).
|
|
62
|
+
// The body may not contain ANOTHER sentinel of the same slug. `[\s\S]*?` could, and that is
|
|
63
|
+
// not a tidiness point — it is how a match stopped being one block. Drop the `:end` line by
|
|
64
|
+
// hand (a merge resolution, an editor, another tool) and the next adopt appends a second
|
|
65
|
+
// block below whatever the user has written since; the adopt after THAT matched from the
|
|
66
|
+
// orphaned begin, lazily, to the only `:end` in the file — which now sits past the user's
|
|
67
|
+
// text and past the second begin — so `raw.replace(m[0], section)` deleted all of it.
|
|
68
|
+
// Measured 2026-09-13: a "## Deployment runbook" section appended after adoption was gone
|
|
69
|
+
// after two further adopts, silently, with both runs reporting success.
|
|
70
|
+
//
|
|
71
|
+
// The tempered token below cannot span a sentinel, so the engine backtracks to the
|
|
72
|
+
// WELL-FORMED pair and the orphan is simply left alone — which is the right answer for a
|
|
73
|
+
// file we do not own: where the orphaned body ends is genuinely unknowable, so removeManaged
|
|
74
|
+
// reports it (action 'partial') rather than guessing a span to delete.
|
|
75
|
+
//
|
|
76
|
+
// Safe by construction for legitimate blocks: the shipped body carries the slug twice and
|
|
77
|
+
// never as a sentinel (measured: 1304 bytes — 1296 UTF-16 units, the body has em dashes —
|
|
78
|
+
// with zero `:begin` / `:end` occurrences), and the
|
|
79
|
+
// separators stay `\r?\n` for the CRLF reason below.
|
|
62
80
|
function blockBody(esc) {
|
|
63
|
-
|
|
81
|
+
const sentinel = `<!-- ${esc}:(?:begin|end)`;
|
|
82
|
+
return `<!-- ${esc}:begin (v\\d+) -->\\r?\\n((?:(?!${sentinel})[\\s\\S])*?)\\r?\\n<!-- ${esc}:end -->`;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// Any sentinel LINE of our slug, paired or not. The pair regex above is deliberately blind to
|
|
86
|
+
// an unpaired one; this is what lets residue reporting see what it cannot safely remove.
|
|
87
|
+
function sentinelLineRegexG(slug) {
|
|
88
|
+
return new RegExp(`<!-- ${escapeRe(slug)}:(?:begin|end)\\b[^>]*-->`, 'g');
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Sentinel lines of this slug left in `raw` that no well-formed block accounts for.
|
|
93
|
+
* Zero on a healthy file (every sentinel belongs to a matched pair) and on a clean one.
|
|
94
|
+
* @param {string} raw
|
|
95
|
+
* @param {string} slug
|
|
96
|
+
* @returns {number}
|
|
97
|
+
*/
|
|
98
|
+
function orphanSentinelCount(raw, slug) {
|
|
99
|
+
const total = (raw.match(sentinelLineRegexG(slug)) || []).length;
|
|
100
|
+
let paired = 0;
|
|
101
|
+
raw.replace(blockRegexG(slug), (whole) => {
|
|
102
|
+
paired += (whole.match(sentinelLineRegexG(slug)) || []).length;
|
|
103
|
+
return whole;
|
|
104
|
+
});
|
|
105
|
+
return total - paired;
|
|
64
106
|
}
|
|
65
107
|
function blockRegex(slug) {
|
|
66
108
|
return new RegExp(blockBody(escapeRe(slug)));
|
|
@@ -137,13 +179,24 @@ export function isAdopted(cwd, slug) {
|
|
|
137
179
|
* removed it. removeManaged cleans all three pieces, so sweep on any of them.
|
|
138
180
|
*/
|
|
139
181
|
export function hasResidue(cwd, slug) {
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
existsSync(detailDocPath(cwd, slug)) ||
|
|
143
|
-
existsSync(stateFilePath(cwd, slug))
|
|
144
|
-
);
|
|
182
|
+
const blk = readBlock(cwd, slug);
|
|
183
|
+
return blk.body !== null || existsSync(detailDocPath(cwd, slug)) || existsSync(stateFilePath(cwd, slug));
|
|
145
184
|
}
|
|
146
185
|
|
|
186
|
+
// An unpaired sentinel is deliberately NOT in the list above (pre-ship review P2-3). A first
|
|
187
|
+
// cut added it, reasoning that the sweep should see what the pair regex cannot. But
|
|
188
|
+
// orphanSentinelCount counts sentinel-shaped TEXT, and a project the plugin never touched can
|
|
189
|
+
// mention the marker in prose — documenting it, pasting half an example, a changelog line.
|
|
190
|
+
// That one mention let `unadopt --all` into a stranger's project, where removeManaged
|
|
191
|
+
// unconditionally deletes the detail doc and state sidecar and rmdir's an empty `.claude/`,
|
|
192
|
+
// then printed "remove those lines by hand" at the user's own paragraph — and never
|
|
193
|
+
// converged, because the mention is still there on the next sweep.
|
|
194
|
+
//
|
|
195
|
+
// The three entries above are all things the PLUGIN WROTE; a sentinel in prose is not. And
|
|
196
|
+
// the sweep gains nothing by entering: removeManaged cannot clean an orphan anyway, by
|
|
197
|
+
// design. The orphan is reported by removeManaged when unadopt genuinely runs — which is the
|
|
198
|
+
// real failure case, where the doc and sidecar are still present and do bring it in.
|
|
199
|
+
|
|
147
200
|
/**
|
|
148
201
|
* Whether the installed block/doc has drifted from the shipped content — i.e.
|
|
149
202
|
* a version bump or a template edit means we should refresh. Returns true when
|
|
@@ -232,11 +285,25 @@ export function writeManaged(cwd, { slug, version, block, doc }) {
|
|
|
232
285
|
* Remove our managed block from CLAUDE.md (preserving all other content) and
|
|
233
286
|
* delete the detail doc + state sidecar. Best-effort removes an emptied
|
|
234
287
|
* .claude/ directory.
|
|
235
|
-
*
|
|
288
|
+
*
|
|
289
|
+
* THREE outcomes, not two — the same rule lib/db-unusable.mjs states about backups: "there
|
|
290
|
+
* is nothing to do" and "I could not finish" must not print in the same voice, because a
|
|
291
|
+
* green-sounding line ends the reader's search. `absent` used to cover both: with one
|
|
292
|
+
* sentinel line missing, the pair regex matched nothing, so this returned 'absent' — while
|
|
293
|
+
* having already deleted the detail doc and the state sidecar and left ~1.3 KB of managed
|
|
294
|
+
* steering text in the user's CLAUDE.md, which it is then loaded from on every session.
|
|
295
|
+
* `partial` is that case, and `residue` names what is left so the caller can say so.
|
|
296
|
+
*
|
|
297
|
+
* Deliberately does NOT delete an orphaned sentinel's body: where it ends is unknowable
|
|
298
|
+
* (that is the defect, not a detail), and guessing a span in a file we do not own is how the
|
|
299
|
+
* adopt side came to delete a user's runbook. Report, do not repair.
|
|
300
|
+
*
|
|
301
|
+
* @returns {{action: 'removed'|'partial'|'absent', residue?: string}}
|
|
236
302
|
*/
|
|
237
303
|
export function removeManaged(cwd, slug) {
|
|
238
304
|
const p = claudeMdPath(cwd);
|
|
239
305
|
let action = 'absent';
|
|
306
|
+
let orphans = 0;
|
|
240
307
|
if (existsSync(p)) {
|
|
241
308
|
let raw = readFileSync(p, 'utf8');
|
|
242
309
|
// H2: loop so ALL same-slug blocks are removed, not just the first (a
|
|
@@ -283,8 +350,17 @@ export function removeManaged(cwd, slug) {
|
|
|
283
350
|
atomicWrite(p, raw);
|
|
284
351
|
}
|
|
285
352
|
}
|
|
353
|
+
// Counted on what is left AFTER the loop, so a healthy file (every sentinel consumed by
|
|
354
|
+
// a matched pair) reports zero and only a genuinely unpaired line survives the count.
|
|
355
|
+
orphans = orphanSentinelCount(raw, slug);
|
|
286
356
|
}
|
|
357
|
+
// Captured BEFORE the deletions below, because they are what it asks about: is there any
|
|
358
|
+
// evidence the plugin ever wrote in this project? An unpaired sentinel is NOT such
|
|
359
|
+
// evidence — it is text, and a project that merely documents the marker in prose has one
|
|
360
|
+
// (pre-ship review P2-3). Reporting residue there means telling a stranger to delete their
|
|
361
|
+
// own paragraph, on a project this tool has never touched.
|
|
287
362
|
const dp = detailDocPath(cwd, slug);
|
|
363
|
+
const wasOurs = action === 'removed' || existsSync(dp) || existsSync(stateFilePath(cwd, slug));
|
|
288
364
|
if (existsSync(dp))
|
|
289
365
|
try {
|
|
290
366
|
unlinkSync(dp);
|
|
@@ -300,6 +376,18 @@ export function removeManaged(cwd, slug) {
|
|
|
300
376
|
} catch {
|
|
301
377
|
/* best-effort */
|
|
302
378
|
}
|
|
379
|
+
// `action` answers ONE question — what happened to the block — and `residue` is an
|
|
380
|
+
// independent fact that rides alongside it. A first cut let an orphan override 'removed'
|
|
381
|
+
// too, on the reasoning that both are "unfinished". Pre-ship review P2-1: unadoptAll's
|
|
382
|
+
// else-branch prints "cleaned partial residue (detail doc/state, no block)" and counts
|
|
383
|
+
// `partial++`, so a sweep that DID remove a block reported "no block" and tallied zero
|
|
384
|
+
// removals. Two facts, two fields.
|
|
385
|
+
const residue =
|
|
386
|
+
orphans > 0 && wasOurs
|
|
387
|
+
? `${orphans} unpaired \`${slug}\` sentinel line(s) remain in ${claudeMdPath(cwd)} — the block they opened has no matching end marker, so its extent cannot be determined safely. Remove those lines and the text they wrap by hand.`
|
|
388
|
+
: null;
|
|
389
|
+
if (action === 'removed') return residue ? { action, residue } : { action };
|
|
390
|
+
if (residue) return { action: 'partial', residue };
|
|
303
391
|
return { action };
|
|
304
392
|
}
|
|
305
393
|
|
package/hook-context.mjs
CHANGED
|
@@ -496,6 +496,50 @@ export function cleanupClaudeMdLegacyBlock() {
|
|
|
496
496
|
}
|
|
497
497
|
}
|
|
498
498
|
|
|
499
|
+
/**
|
|
500
|
+
* How many locally-selected observations count as a thick enough pool to skip the
|
|
501
|
+
* cross-project fallback. Named because the number appeared three times: the guard
|
|
502
|
+
* that DECIDES to run the fallback query and the two sites that consume it.
|
|
503
|
+
*/
|
|
504
|
+
const MIN_LOCAL_OBS = 3;
|
|
505
|
+
|
|
506
|
+
/**
|
|
507
|
+
* The rows to render: everything `selectWithTokenBudget` chose, topped up with any
|
|
508
|
+
* fallback rows it does not already cover.
|
|
509
|
+
*
|
|
510
|
+
* R12 A2 — this used to be `observations.length >= 3 ? observations : fallbackObs`,
|
|
511
|
+
* written inline at BOTH consumer sites. The two row sets come from different windows
|
|
512
|
+
* (obsPool's low-speed tier1 is 48h/imp>=1; the fallback query is 24h/imp>=1 OR
|
|
513
|
+
* 7d/imp>=2) and neither contains the other, so a whole-set switch could hand back
|
|
514
|
+
* FEWER rows than it was given — measured on tests/hook-context.test.mjs's own `thin(n)`
|
|
515
|
+
* fixture: 2 rows in the DB, both selected, and a 0-byte context block, against 241 bytes
|
|
516
|
+
* and 3 table rows at n=3. Output was non-monotonic in corpus size, and the victims were
|
|
517
|
+
* exactly the thin/new projects the 60-day tier exists to serve. (An earlier revision of
|
|
518
|
+
* this line said 391 bytes, carried over from the audit's own differently-seeded fixture
|
|
519
|
+
* rather than measured here.)
|
|
520
|
+
*
|
|
521
|
+
* Top-up, not truncate-to-three: the union never renders fewer rows than the old
|
|
522
|
+
* expression did on any input, which a "fill to 3" fix would not hold — at 0 selected rows
|
|
523
|
+
* the fallback query's own LIMIT 5 already governs, and capping at 3 would have been a
|
|
524
|
+
* second, unrelated behaviour change. Growth is bounded: the union only runs below
|
|
525
|
+
* MIN_LOCAL_OBS, so the table gains at most two rows.
|
|
526
|
+
*
|
|
527
|
+
* The COUNT never shrinks; the composition can, and that is a ranking decision rather than
|
|
528
|
+
* an accident. At the uncapped table site this only adds. At the `.slice(0, MIN_LOCAL_OBS)`
|
|
529
|
+
* site the selected rows are PREPENDED, so with 1 local row and 3+ fallback rows the third
|
|
530
|
+
* fallback title is evicted — still three rows, one of them now local. Exhaustive
|
|
531
|
+
* enumeration over 42,436 ordered input pairs: 0 cases render fewer rows, 3,000 evict a
|
|
532
|
+
* fallback row that way, max gain 2. Preferring the project's own rows over cross-project
|
|
533
|
+
* ones inside a fixed window is the intended order, and saying so is the point: a new
|
|
534
|
+
* population entering a limit-bounded window without a stated ranking is this repo's
|
|
535
|
+
* recorded failure shape.
|
|
536
|
+
*/
|
|
537
|
+
function withFallbackTopUp(observations, fallbackObs) {
|
|
538
|
+
if (observations.length >= MIN_LOCAL_OBS) return observations;
|
|
539
|
+
const seen = new Set(observations.map((o) => o.id));
|
|
540
|
+
return [...observations, ...fallbackObs.filter((o) => !seen.has(o.id))];
|
|
541
|
+
}
|
|
542
|
+
|
|
499
543
|
/**
|
|
500
544
|
* Assemble the full markdown body that goes inside the <claude-mem-context>
|
|
501
545
|
* block emitted at session start. Same shape as the inline builder hook.mjs
|
|
@@ -536,7 +580,7 @@ export function buildSessionContextLines(
|
|
|
536
580
|
|
|
537
581
|
// 2. Fallback: recent across all projects with tiered windows (when local pool is thin)
|
|
538
582
|
let fallbackObs = [];
|
|
539
|
-
if (observations.length <
|
|
583
|
+
if (observations.length < MIN_LOCAL_OBS) {
|
|
540
584
|
const fbOneDayAgo = now.getTime() - STALE_SESSION_MS;
|
|
541
585
|
const fbSevenDaysAgo = now.getTime() - RELATED_OBS_WINDOW_MS;
|
|
542
586
|
fallbackObs = db
|
|
@@ -663,8 +707,8 @@ export function buildSessionContextLines(
|
|
|
663
707
|
// Slice FIRST, then sort: the slice is the selection (top 3 by value density) and must
|
|
664
708
|
// stay that way; only the order they are printed in is corrected, same as the Recent
|
|
665
709
|
// table below. Sorting before the slice would silently change WHICH three are injected.
|
|
666
|
-
const recentObs = (observations
|
|
667
|
-
.slice(0,
|
|
710
|
+
const recentObs = withFallbackTopUp(observations, fallbackObs)
|
|
711
|
+
.slice(0, MIN_LOCAL_OBS)
|
|
668
712
|
.sort((a, b) => Date.parse(b.created_at) - Date.parse(a.created_at) || b.id - a.id);
|
|
669
713
|
if (recentObs.length > 0) {
|
|
670
714
|
summaryLines.push('### Recent Activity');
|
|
@@ -784,7 +828,7 @@ export function buildSessionContextLines(
|
|
|
784
828
|
// tests/hook-context.test.mjs). Tiebroken on id for the same reason D#9 gives — an
|
|
785
829
|
// untiebroken tie flips direction, and two saves in one millisecond are common.
|
|
786
830
|
const obsLines = [];
|
|
787
|
-
const obsToShow = [...(observations
|
|
831
|
+
const obsToShow = [...withFallbackTopUp(observations, fallbackObs)].sort(
|
|
788
832
|
(a, b) => Date.parse(b.created_at) - Date.parse(a.created_at) || b.id - a.id,
|
|
789
833
|
);
|
|
790
834
|
if (obsToShow.length > 0) {
|
package/hook-llm.mjs
CHANGED
|
@@ -24,7 +24,7 @@ import {
|
|
|
24
24
|
} from './utils.mjs';
|
|
25
25
|
import { acquireLLMSlot, releaseLLMSlot } from './hook-semaphore.mjs';
|
|
26
26
|
import { BG_LLM_TIMEOUT_MS } from './haiku-client.mjs';
|
|
27
|
-
import { scrubRecord } from './lib/scrub-record.mjs';
|
|
27
|
+
import { scrubRecord, scrubFilePaths } from './lib/scrub-record.mjs';
|
|
28
28
|
import {
|
|
29
29
|
insertObservationRow,
|
|
30
30
|
insertObservationFiles,
|
|
@@ -356,6 +356,13 @@ export function saveObservation(obs, projectOverride, sessionIdOverride, externa
|
|
|
356
356
|
search_aliases: obs.searchAliases || null,
|
|
357
357
|
});
|
|
358
358
|
|
|
359
|
+
// D#44: derive the scrubbed path arrays ONCE. `obs.files` feeds two sinks —
|
|
360
|
+
// the files_modified JSON column and the observation_files junction below —
|
|
361
|
+
// and the junction value is also the recall key, so a per-sink scrub is how
|
|
362
|
+
// the stored key and the indexed column drift apart.
|
|
363
|
+
const safeFiles = scrubFilePaths(obs.files || []);
|
|
364
|
+
const safeFilesRead = scrubFilePaths(obs.filesRead || []);
|
|
365
|
+
|
|
359
366
|
// Atomic: observation INSERT + observation_files in one transaction.
|
|
360
367
|
// Column list single-sourced in lib/observation-write (shared with manual mem_save).
|
|
361
368
|
const savedId = db.transaction(() => {
|
|
@@ -369,8 +376,8 @@ export function saveObservation(obs, projectOverride, sessionIdOverride, externa
|
|
|
369
376
|
narrative: safe.narrative,
|
|
370
377
|
concepts: safe.concepts,
|
|
371
378
|
facts: safe.facts,
|
|
372
|
-
files_read: JSON.stringify(
|
|
373
|
-
files_modified: JSON.stringify(
|
|
379
|
+
files_read: JSON.stringify(safeFilesRead),
|
|
380
|
+
files_modified: JSON.stringify(safeFiles),
|
|
374
381
|
importance: obs.importance ?? 1,
|
|
375
382
|
minhash_sig: minhashSig,
|
|
376
383
|
lesson_learned: safe.lesson_learned,
|
|
@@ -384,7 +391,7 @@ export function saveObservation(obs, projectOverride, sessionIdOverride, externa
|
|
|
384
391
|
scope: normalizeScope(obs.scope),
|
|
385
392
|
});
|
|
386
393
|
|
|
387
|
-
insertObservationFiles(db, id,
|
|
394
|
+
insertObservationFiles(db, id, safeFiles);
|
|
388
395
|
|
|
389
396
|
return id;
|
|
390
397
|
})();
|
package/hook-memory.mjs
CHANGED
|
@@ -435,12 +435,11 @@ export function searchRelevantMemories(
|
|
|
435
435
|
debugCatch(e, 'crossProjectSearch');
|
|
436
436
|
}
|
|
437
437
|
|
|
438
|
-
// Merge and score: same-project full weight, cross-project
|
|
439
|
-
//
|
|
440
|
-
//
|
|
441
|
-
//
|
|
442
|
-
//
|
|
443
|
-
// same-project focus in noisy cross-project environments.
|
|
438
|
+
// Merge and score: same-project full weight, cross-project penalised by
|
|
439
|
+
// getCrossProjectBoost(). R12 A6 — this block used to restate that knob's default as
|
|
440
|
+
// 0.7 twice over, which v2.41 changed to 0.4 while updating only the comment at the
|
|
441
|
+
// function. The value and its rationale live at getCrossProjectBoost() and nowhere
|
|
442
|
+
// else; a second copy of a tuned number is a second answer to "what is the baseline".
|
|
444
443
|
//
|
|
445
444
|
// OR-fallback results get 0.4x penalty — they matched individual words, not the full intent
|
|
446
445
|
// v26 P0: noise_penalty (from SQL) shrinks high-inject/low-cite rows.
|
package/lib/activity.mjs
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
// so they don't pollute the L1 system-prompt memory section.
|
|
6
6
|
|
|
7
7
|
import { sanitizeFtsQuery } from '../utils.mjs';
|
|
8
|
-
import { scrubRecord } from './scrub-record.mjs';
|
|
8
|
+
import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
|
|
9
9
|
import { saveObservation } from './save-observation.mjs';
|
|
10
10
|
// Pure title-only builder: this query runs on the EVENTS table, which has no
|
|
11
11
|
// lesson_learned column — the lesson-escape variant would be a SQL error here.
|
|
@@ -67,6 +67,11 @@ export function saveEvent(
|
|
|
67
67
|
// paths are covered — title/body otherwise land verbatim and are FTS-indexed,
|
|
68
68
|
// searchable, and exportable (HIGH-2 at-rest leak).
|
|
69
69
|
const safe = scrubRecord('events', { title, body });
|
|
70
|
+
// D#44: `file_paths` is a JSON array, so scrubRecord deliberately skips it —
|
|
71
|
+
// element-level pre-scrub is the prescribed remedy. Measured on the
|
|
72
|
+
// maintainer's live DB before the fix: 3 of 2140 stored elements carried a
|
|
73
|
+
// credential-shaped segment, and all 3 were in THIS column.
|
|
74
|
+
const safeFilePaths = scrubFilePaths(file_paths);
|
|
70
75
|
const info = db
|
|
71
76
|
.prepare(
|
|
72
77
|
`
|
|
@@ -79,7 +84,7 @@ export function saveEvent(
|
|
|
79
84
|
event_type,
|
|
80
85
|
safe.title,
|
|
81
86
|
safe.body,
|
|
82
|
-
|
|
87
|
+
safeFilePaths ? JSON.stringify(safeFilePaths) : null,
|
|
83
88
|
git_sha,
|
|
84
89
|
importance,
|
|
85
90
|
created_at_epoch,
|
package/lib/db-unusable.mjs
CHANGED
|
@@ -67,6 +67,49 @@ export function isFtsCorruptionError(err) {
|
|
|
67
67
|
return /SQLITE_CORRUPT_VTAB/i.test(`${err?.code || ''}`);
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
/**
|
|
71
|
+
* What to do about a damaged FTS5 INDEX — the query-time half of isFtsCorruptionError.
|
|
72
|
+
*
|
|
73
|
+
* R10 P3-9 wired the classifier into `ensureDbWithWalRecovery`, which rebuilds and retries.
|
|
74
|
+
* That covers OPEN time. A structure record damaged inside the index opens fine — nothing
|
|
75
|
+
* reads it until the first MATCH — so the fault surfaces at QUERY time, and there the two
|
|
76
|
+
* faces that carry it to a reader (the CLI catch-all, the MCP safeHandler) passed SQLite's
|
|
77
|
+
* own sentence through with no next step. `recent` / `recall` / `browse` / `context` never
|
|
78
|
+
* touch FTS and keep answering, which makes that dead end easy to misread as "search found
|
|
79
|
+
* nothing" rather than "search is broken".
|
|
80
|
+
*
|
|
81
|
+
* ONE STRING FOR BOTH CHANNELS, unlike the file-level family below. The split there exists
|
|
82
|
+
* because that remedy OVERWRITES the database and must not be handed ready-to-run to an
|
|
83
|
+
* agent holding Bash. This one re-derives every index from its own content table: the rows
|
|
84
|
+
* are never read from the index, so a rebuild is lossless and idempotent, and `doctor`
|
|
85
|
+
* already runs it unprompted. Naming "intact" is load-bearing — the reading this line
|
|
86
|
+
* exists to prevent is "my memories are corrupt".
|
|
87
|
+
*
|
|
88
|
+
* SCOPE, measured rather than assumed: a damaged index does not always reach the caller as
|
|
89
|
+
* SQLITE_CORRUPT_VTAB. `observations_fts_data` holds three rows on a one-observation store —
|
|
90
|
+
* id=1 (averages), id=10 (the STRUCTURE record) and one leaf page — and which row the damage
|
|
91
|
+
* lands on decides the code. Over 100 trials each:
|
|
92
|
+
*
|
|
93
|
+
* UPDATE … SET block = randomblob(32) WHERE id > 1 (structure + leaf) 98 VTAB, 2 NOMEM
|
|
94
|
+
* UPDATE … SET block = randomblob(32) WHERE id > 10 (leaf only) 100 VTAB, 0 NOMEM
|
|
95
|
+
* DELETE … WHERE id > 10 (leaf only) 100 VTAB, 0 NOMEM
|
|
96
|
+
*
|
|
97
|
+
* So NOMEM comes from a mangled STRUCTURE record, where SQLite reads a corrupt varint and asks
|
|
98
|
+
* for an absurd allocation — not from leaf damage. (A first draft of this paragraph said
|
|
99
|
+
* "leaf pages", which would send anyone re-measuring `WHERE id > 10` to 0/N and make the
|
|
100
|
+
* boundary below look vacuous. Caught by the pre-ship claims audit.) That case gets no
|
|
101
|
+
* remedy, on purpose:
|
|
102
|
+
* isFtsCorruptionError is what isDbCorruptionError and isDbUnusableError consult, so
|
|
103
|
+
* admitting SQLITE_NOMEM would answer a real out-of-memory with a full FTS rebuild, and
|
|
104
|
+
* would tell a user their index is damaged when it may be their RAM. The code cannot
|
|
105
|
+
* discriminate the two, so this covers the fault it can name.
|
|
106
|
+
* `tests/fts-corruption-query-time-remedy.test.mjs` pins that boundary.
|
|
107
|
+
*/
|
|
108
|
+
export const FTS_CORRUPTION_REMEDY =
|
|
109
|
+
'The FTS5 search index is damaged; the stored observations are intact. ' +
|
|
110
|
+
'Rebuild it losslessly with `claude-mem-lite fts-check rebuild` ' +
|
|
111
|
+
'(`claude-mem-lite doctor` rebuilds it too, and re-checks everything else).';
|
|
112
|
+
|
|
70
113
|
/**
|
|
71
114
|
* True when `err` means "this file exists and SQLite cannot use it as a database".
|
|
72
115
|
*
|
package/lib/deferred-work.mjs
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { DAY_MS } from './time-constants.mjs';
|
|
2
|
-
import { scrubRecord } from './scrub-record.mjs';
|
|
2
|
+
import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
|
|
3
3
|
// claude-mem-lite — deferred_work data layer
|
|
4
4
|
// Pure-data CRUD + ordinal resolver + transactional closure helper.
|
|
5
5
|
// Decoupled from observations table: different lifecycle, different scoring.
|
|
@@ -49,7 +49,12 @@ export function insertDeferred(db, args) {
|
|
|
49
49
|
Date.now(),
|
|
50
50
|
source_session_id,
|
|
51
51
|
source_prompt_id,
|
|
52
|
-
|
|
52
|
+
// The SIXTH path column. `files` is agent-writable on two live faces (MCP
|
|
53
|
+
// `mem_defer(files=…)`, CLI `defer add --files`) and was stringified raw beside a
|
|
54
|
+
// title scrubRecord had already cleaned — the same asymmetry the D#44 round is named
|
|
55
|
+
// after, on the column scrub-record.mjs's header names explicitly. A non-array falls
|
|
56
|
+
// through untouched, so a null stays null rather than becoming '[]'.
|
|
57
|
+
files ? JSON.stringify(scrubFilePaths(files)) : null,
|
|
53
58
|
);
|
|
54
59
|
return { id: Number(r.lastInsertRowid) };
|
|
55
60
|
}
|
package/lib/file-edge-match.mjs
CHANGED
|
@@ -48,6 +48,11 @@
|
|
|
48
48
|
// The one import below does not cost that: project-utils.mjs is a leaf over
|
|
49
49
|
// `node:path`, and pre-tool-recall.js already imports it for inferProject.
|
|
50
50
|
import { likeLiteral } from '../project-utils.mjs';
|
|
51
|
+
// The second import, and it costs no more than the first: scrub-record.mjs ->
|
|
52
|
+
// secret-scrub.mjs -> lib/private-strip.mjs are three regex-only leaves — no
|
|
53
|
+
// node builtins, no DB handle, nothing child_process-shaped. Checked rather
|
|
54
|
+
// than asserted, because the dependency note in this file has been wrong twice.
|
|
55
|
+
import { scrubFilePath } from './scrub-record.mjs';
|
|
51
56
|
|
|
52
57
|
/**
|
|
53
58
|
* SQL boolean expression for the four-arm match. Placeholder order matches
|
|
@@ -233,8 +238,47 @@ export function toolEditPath(input) {
|
|
|
233
238
|
return input.file_path ?? input.notebook_path;
|
|
234
239
|
}
|
|
235
240
|
|
|
236
|
-
/**
|
|
237
|
-
|
|
241
|
+
/**
|
|
242
|
+
* Bind values for fileMatchClause, in placeholder order.
|
|
243
|
+
*
|
|
244
|
+
* D#44: the query path is scrubbed with the SAME helper the writers use, because
|
|
245
|
+
* `observation_files.filename` is a stored value AND the key matched against it.
|
|
246
|
+
* Once the write side pre-scrubs each element, a raw query path derives a
|
|
247
|
+
* different key from the row it is looking for — and only for the one shape that
|
|
248
|
+
* matters: a credential in the BASENAME rewrites the token arms 2-4 bind
|
|
249
|
+
* (`/repo/ghp_….mjs` -> `/repo/***.mjs`), so the lesson about that file becomes
|
|
250
|
+
* unreachable through the file itself. A credential confined to a directory
|
|
251
|
+
* segment is invisible to this, which is why the guard for it uses the basename
|
|
252
|
+
* shape; a case built on the directory shape passes with the reader unscrubbed.
|
|
253
|
+
*
|
|
254
|
+
* That second sentence is true BY CONSTRUCTION and was not, for one commit. Whole-path
|
|
255
|
+
* scrubbing let the eight KV-shaped patterns eat the separator, so
|
|
256
|
+
* `/repo/token=<secret>/notes.mjs` derived `/repo/token=***` — basename destroyed, and
|
|
257
|
+
* every file under that directory sharing one key. `scrubFilePath` splits on the separator
|
|
258
|
+
* first, which bounds every pattern to its own segment; the pre-ship review caught the
|
|
259
|
+
* window between.
|
|
260
|
+
*
|
|
261
|
+
* One consequence is deliberate and stays: a row written BEFORE 6.8.2 holds a raw path,
|
|
262
|
+
* and the query key is now derived. For a directory-segment credential the basename is
|
|
263
|
+
* untouched on both sides, so arms 2-4 still reach it (measured: 4 of 5 shapes). A row
|
|
264
|
+
* whose BASENAME is itself a credential is the exception — it is unreachable by its raw
|
|
265
|
+
* name until D#49 backfills the stored value, and it is by definition one of the leaking
|
|
266
|
+
* rows that backfill exists for.
|
|
267
|
+
*
|
|
268
|
+
* Doing it HERE rather than in each caller is this module's whole premise: the
|
|
269
|
+
* header above requires pre-tool-recall.js and edge-attribution.mjs to stay in
|
|
270
|
+
* byte-identical agreement, and recall-core.mjs + searchByFile bind the same
|
|
271
|
+
* clause. FIVE call sites across four modules — recall-core.mjs has two,
|
|
272
|
+
* recallByFile and countRecallableByFile — and one derivation. Counted by call
|
|
273
|
+
* site, because that is the unit an auditor asking "did every consumer move?"
|
|
274
|
+
* actually walks; the same change described it as four elsewhere.
|
|
275
|
+
*
|
|
276
|
+
* Not measurable on the maintainer's corpus — 0 of 2340 stored path values, and
|
|
277
|
+
* 0 of 100 junction rows, change under scrubSecrets (readonly census,
|
|
278
|
+
* 2026-09-13). That is a property of that corpus on that date, not of the code.
|
|
279
|
+
*/
|
|
280
|
+
export function fileMatchParams(rawFilePath) {
|
|
281
|
+
const filePath = scrubFilePath(rawFilePath);
|
|
238
282
|
const fname = basenameAnySep(filePath);
|
|
239
283
|
const escaped = likeLiteral(fname);
|
|
240
284
|
// `%\\` before the basename: under ESCAPE '\', a literal backslash is
|
package/lib/import-jsonl.mjs
CHANGED
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
import { readFileSync, statSync } from 'fs';
|
|
18
18
|
import { createHash } from 'crypto';
|
|
19
19
|
import { scrubSecrets } from '../secret-scrub.mjs';
|
|
20
|
-
import { scrubRecord } from './scrub-record.mjs';
|
|
20
|
+
import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
|
|
21
21
|
import { toolEditPath } from './file-edge-match.mjs';
|
|
22
22
|
import { insertObservationFiles } from './observation-write.mjs';
|
|
23
23
|
|
|
@@ -160,15 +160,22 @@ function importToolPair(db, toolUse, toolResult, project) {
|
|
|
160
160
|
// while being unable to fire for it, and every imported notebook edit built
|
|
161
161
|
// no (obs,file) edge. `toolEditPath` is the single home for that rule.
|
|
162
162
|
const editedPath = toolEditPath(toolUse.input);
|
|
163
|
-
|
|
163
|
+
// D#44: scrubbed at the derivation, so the files_modified JSON column and the
|
|
164
|
+
// observation_files junction (both fed from this one array below) cannot
|
|
165
|
+
// disagree. The title built from this same path was already scrubbed via
|
|
166
|
+
// scrubRecord — that asymmetry was the defect.
|
|
167
|
+
const filesModified = scrubFilePaths(
|
|
164
168
|
(toolName === 'Edit' || toolName === 'Write' || toolName === 'NotebookEdit') && editedPath
|
|
165
169
|
? [editedPath]
|
|
166
|
-
: []
|
|
170
|
+
: [],
|
|
171
|
+
);
|
|
167
172
|
// `file_path`, not the shared `editedPath`: `toolEditPath` answers "which path did this
|
|
168
173
|
// tool WRITE", and falls back to `notebook_path` for NotebookEdit's sake. Routing the read
|
|
169
174
|
// column through it too gave a `Read` carrying only `notebook_path` a files_read entry —
|
|
170
175
|
// a shape no tool emits, and a behaviour change D#35 made without declaring it (P3-5).
|
|
171
|
-
const filesRead =
|
|
176
|
+
const filesRead = scrubFilePaths(
|
|
177
|
+
toolName === 'Read' && toolUse.input?.file_path ? [toolUse.input.file_path] : [],
|
|
178
|
+
);
|
|
172
179
|
|
|
173
180
|
// `narrative` carries the body and `text` is the derived search blob
|
|
174
181
|
// (lib/observation-write.mjs rebuildObservationDerived). Writing the payload to `text`
|
package/lib/injected-ids.mjs
CHANGED
|
@@ -72,18 +72,27 @@ export function injectedIdsFileName(project, sessionId) {
|
|
|
72
72
|
* false and the next person to diff the two would have had to rediscover why.
|
|
73
73
|
*/
|
|
74
74
|
export function readInjectedMarker(file, { sessionId, maxAgeMs } = {}) {
|
|
75
|
-
const empty = { ids: [], count: 0, fresh: false };
|
|
75
|
+
const empty = { ids: [], count: 0, upsCount: 0, upsTs: 0, fresh: false };
|
|
76
76
|
try {
|
|
77
|
-
const { ids, ts, count, session } = JSON.parse(readFileSync(file, 'utf8'));
|
|
77
|
+
const { ids, ts, count, upsCount, upsTs, session } = JSON.parse(readFileSync(file, 'utf8'));
|
|
78
78
|
if (session && sessionId && session !== sessionId) return empty;
|
|
79
79
|
if (!ts || Date.now() - ts > maxAgeMs) return empty;
|
|
80
80
|
if (!Array.isArray(ids)) return empty;
|
|
81
|
-
return { ids, count: count || 0, fresh: true };
|
|
81
|
+
return { ids, count: count || 0, upsCount: upsCount || 0, upsTs: upsTs || 0, fresh: true };
|
|
82
82
|
} catch {
|
|
83
83
|
return empty;
|
|
84
84
|
}
|
|
85
85
|
}
|
|
86
86
|
|
|
87
|
+
/**
|
|
88
|
+
* Hard ceiling on the marker's `ids` array, applied to BOTH arms.
|
|
89
|
+
*
|
|
90
|
+
* 2x the largest seen-set this repo has measured in the wild (16, histogram at
|
|
91
|
+
* scripts/pre-tool-recall.js:92-93). See mergeInjectedMarker for why an unbounded set is
|
|
92
|
+
* not merely a size problem.
|
|
93
|
+
*/
|
|
94
|
+
export const MAX_MARKER_IDS = 32;
|
|
95
|
+
|
|
87
96
|
/**
|
|
88
97
|
* Write `newIds` into a marker, unioning with a fresh same-session payload or replacing it.
|
|
89
98
|
*
|
|
@@ -92,12 +101,21 @@ export function readInjectedMarker(file, { sessionId, maxAgeMs } = {}) {
|
|
|
92
101
|
* - `union` stringifies the whole result. Both union callers already did
|
|
93
102
|
* (`prev.ids.map(String)` plus new ids that are `D<id>` strings anyway), so
|
|
94
103
|
* this is their behaviour, not a new normalisation.
|
|
95
|
-
* - `replace` writes `newIds` verbatim
|
|
96
|
-
*
|
|
97
|
-
*
|
|
104
|
+
* - `replace` writes `newIds` verbatim and carries every OTHER id forward as a string.
|
|
105
|
+
* The UPS main leg passes `candidateIds`, a MIX of raw observation numbers
|
|
106
|
+
* and `P<id>` strings, and writing those unchanged is exactly the state
|
|
107
|
+
* D#213 measures. Stringifying them here would change it. It replaces the
|
|
108
|
+
* caller's own slice, not the file (R12 B-6 — see mergeInjectedMarker).
|
|
98
109
|
* Keeping both under one function is the point: the next writer picks a mode instead of
|
|
99
110
|
* copying a fifth predicate and inventing a fifth typing rule.
|
|
100
111
|
*
|
|
112
|
+
* TWO counters, and they answer different questions (R12 B-5). `count` is "how many times
|
|
113
|
+
* has anything written this marker" and stays as it was. `upsCount` is "how many times has
|
|
114
|
+
* the UPS face injected", and ONLY the caller that owns `MAX_SESSION_INJECTIONS` opts into
|
|
115
|
+
* bumping it. They were one field, so pre-tool-recall — which writes once per triggered
|
|
116
|
+
* Edit/Read and shares this file — spent the fyi face's entire per-session budget without
|
|
117
|
+
* the fyi face emitting a line. A budget has to be charged to the spender.
|
|
118
|
+
*
|
|
101
119
|
* The write is atomic for the reason M-6 recorded: a plain write torn by a concurrent hook
|
|
102
120
|
* left the shared marker as invalid JSON, silently disabling cross-hook dedup for the window.
|
|
103
121
|
*
|
|
@@ -108,15 +126,56 @@ export function readInjectedMarker(file, { sessionId, maxAgeMs } = {}) {
|
|
|
108
126
|
* @param {number} opts.maxAgeMs
|
|
109
127
|
* @param {'union'|'replace'} opts.mode
|
|
110
128
|
*/
|
|
111
|
-
export function mergeInjectedMarker(file, newIds, { sessionId, maxAgeMs, mode } = {}) {
|
|
129
|
+
export function mergeInjectedMarker(file, newIds, { sessionId, maxAgeMs, mode, bumpUpsCount = false } = {}) {
|
|
112
130
|
const prev = readInjectedMarker(file, { sessionId, maxAgeMs });
|
|
113
|
-
|
|
131
|
+
// R12 B-6. `replace` used to write `newIds` as the WHOLE array, so the UPS main leg's
|
|
132
|
+
// one write per prompt erased everything pre-tool-recall had accumulated in the window
|
|
133
|
+
// and that face re-injected lessons it had already shown. It now replaces the CALLER's
|
|
134
|
+
// slice and carries the rest.
|
|
135
|
+
//
|
|
136
|
+
// The carried ids are stringified and the caller's are not, and that asymmetry is the
|
|
137
|
+
// load-bearing part rather than an oversight: D#213's exclude is inert BECAUSE
|
|
138
|
+
// `new Set(excludeIds).has(<number from SQLite>)` misses a string key, and the UPS leg's
|
|
139
|
+
// raw numbers are the one population that is NOT inert. Writing `newIds` verbatim keeps
|
|
140
|
+
// that population byte-identical to what it was, and everything carried in arrives as a
|
|
141
|
+
// string, so this cannot widen the live exclude. Repairing D#213 is a separate decision
|
|
142
|
+
// with its own ruler (lib/patha-exclude-meter.mjs).
|
|
143
|
+
const newKeys = new Set(newIds.map(String));
|
|
144
|
+
const carried = [...new Set(prev.ids.map(String))].filter((id) => !newKeys.has(id));
|
|
145
|
+
// Both arms are built NEWEST-FIRST and then capped. Until B-6, `replace` overwriting the
|
|
146
|
+
// whole array was the only thing that ever shrank this file — `union` has always
|
|
147
|
+
// accumulated, and `ts` is refreshed on every write, so the staleness gate never fires in
|
|
148
|
+
// a session where any hook writes inside the window. Making replace carry the other
|
|
149
|
+
// hook's ids therefore removed a bound nobody had written down: measured 520 ids after 40
|
|
150
|
+
// rounds of 5 pre-tool-recall triggers plus one prompt, linear and unbounded.
|
|
151
|
+
//
|
|
152
|
+
// A large seen-set does not just cost bytes, it starves the face it exists to help.
|
|
153
|
+
// scripts/pre-tool-recall.js sizes its over-fetch as `min(seenSize, 5)` and then drops
|
|
154
|
+
// every fetched row that is IN the set, so past a handful of entries a Read can fetch six
|
|
155
|
+
// candidates and filter all six. That failure is described at pre-tool-recall.js:100-105,
|
|
156
|
+
// derived at a seen-set of 16 — which is the largest this repo has measured in the wild
|
|
157
|
+
// (`pre-tool-recall.js:92-93`, histogram 1x9 2x1 3x2 16x1 over n=13).
|
|
158
|
+
//
|
|
159
|
+
// So the cap is 2x the largest observed set, not a round number pulled from nowhere. It
|
|
160
|
+
// BOUNDS that failure; it does not remove it — six candidates can still all be seen — and
|
|
161
|
+
// removing it is ALGO-4's problem, not this one.
|
|
162
|
+
const ids = (
|
|
163
|
+
mode === 'union'
|
|
164
|
+
? [...new Set([...newIds.map(String), ...prev.ids.map(String)])]
|
|
165
|
+
: [...newIds, ...carried]
|
|
166
|
+
).slice(0, MAX_MARKER_IDS);
|
|
114
167
|
atomicWriteFileSync(
|
|
115
168
|
file,
|
|
116
169
|
JSON.stringify({
|
|
117
170
|
ids,
|
|
118
171
|
ts: Date.now(),
|
|
119
172
|
count: prev.count + 1,
|
|
173
|
+
upsCount: prev.upsCount + (bumpUpsCount ? 1 : 0),
|
|
174
|
+
// The UPS face's budget needs the UPS face's CLOCK, not the shared one. `ts` is
|
|
175
|
+
// refreshed by every writer, and readInjectedMarker zeroes upsCount off `ts`, so a
|
|
176
|
+
// tool-heavy session kept the fyi face's spent budget alive across gaps that should
|
|
177
|
+
// have released it — B-5 moved the counter to the spender and left the clock shared.
|
|
178
|
+
upsTs: bumpUpsCount ? Date.now() : prev.upsTs,
|
|
120
179
|
...(sessionId ? { session: sessionId } : {}),
|
|
121
180
|
}),
|
|
122
181
|
);
|
package/lib/save-observation.mjs
CHANGED
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
|
|
14
14
|
import { jaccardSimilarity, scrubSecrets, computeMinHash, cjkBigrams, getCurrentBranch } from '../utils.mjs';
|
|
15
15
|
import { DEDUP_JACCARD_THRESHOLD } from './dedup-constants.mjs';
|
|
16
|
+
import { scrubFilePaths } from './scrub-record.mjs';
|
|
16
17
|
import { insertObservationRow, insertObservationFiles } from './observation-write.mjs';
|
|
17
18
|
// The SAME predicate every read path uses. Imported rather than re-typed: a hand-written
|
|
18
19
|
// `superseded_at IS NULL` here would drift from the read side on the next column change,
|
|
@@ -174,9 +175,12 @@ export function saveObservation(db, params) {
|
|
|
174
175
|
throw new Error('mem_save: content is empty or whitespace-only');
|
|
175
176
|
}
|
|
176
177
|
const importance = params.importance ?? 2;
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
178
|
+
// D#44: scrub HERE, at the derivation, not at the two sinks below — this one
|
|
179
|
+
// array feeds both `files_modified` (JSON column) and the observation_files
|
|
180
|
+
// junction, and scrubbing at each sink separately is how they drift apart.
|
|
181
|
+
const files = scrubFilePaths(
|
|
182
|
+
Array.isArray(params.files) ? params.files.filter((f) => typeof f === 'string' && f.length > 0) : [],
|
|
183
|
+
);
|
|
180
184
|
const rawLesson =
|
|
181
185
|
typeof params.lesson_learned === 'string' && params.lesson_learned.length > 0
|
|
182
186
|
? params.lesson_learned
|
package/lib/scrub-record.mjs
CHANGED
|
@@ -10,7 +10,13 @@
|
|
|
10
10
|
// session_handoffs.match_keywords-when-array) are NOT listed here — running
|
|
11
11
|
// scrubSecrets over the JSON string can rewrite quoted values and break
|
|
12
12
|
// downstream JSON.parse. Pre-scrub each element upstream of the
|
|
13
|
-
// JSON.stringify call instead.
|
|
13
|
+
// JSON.stringify call instead, via `scrubFilePaths` below.
|
|
14
|
+
//
|
|
15
|
+
// D#44: that instruction sat here for four releases and exactly ONE call site
|
|
16
|
+
// followed it (hook-handoff.mjs, session_handoffs.key_files). observations
|
|
17
|
+
// .files_modified / .files_read, observation_files.filename and events
|
|
18
|
+
// .file_paths all stored raw paths while the title DERIVED FROM THE SAME PATH
|
|
19
|
+
// was scrubbed. A prescription in a comment is not a mechanism; the helper is.
|
|
14
20
|
|
|
15
21
|
import { scrubSecrets } from '../secret-scrub.mjs';
|
|
16
22
|
|
|
@@ -62,11 +68,69 @@ export const TEXT_FIELDS_BY_TABLE = {
|
|
|
62
68
|
// the unknown-table failsafe ever ran. title/detail are written verbatim by the agent
|
|
63
69
|
// ("rotate ghp_… before release", a connection string in detail) and replayed into
|
|
64
70
|
// model context by the SessionStart dashboard, mem_defer_list and mem_get D#N.
|
|
65
|
-
// files — JSON.stringify(array); pre-
|
|
71
|
+
// files — JSON.stringify(array); pre-scrubbed element-wise at insertDeferred via
|
|
72
|
+
// `scrubFilePaths`. It needed to be: `mem_defer` and `defer add --files` both
|
|
73
|
+
// take agent-supplied paths, and this line prescribed the remedy for four
|
|
74
|
+
// releases while the call site stored them raw.
|
|
66
75
|
// project / status — identifiers and an enum.
|
|
67
76
|
deferred_work: ['title', 'detail', 'drop_reason'],
|
|
68
77
|
};
|
|
69
78
|
|
|
79
|
+
/**
|
|
80
|
+
* Scrub one filesystem path — for PERSISTENCE and for KEY DERIVATION, which is
|
|
81
|
+
* why this is a named export and not an inline `scrubSecrets(f)` at each site.
|
|
82
|
+
*
|
|
83
|
+
* `observation_files.filename` is both a stored value and the recall key
|
|
84
|
+
* (lib/file-edge-match.mjs binds it four ways). If the write side scrubs and the
|
|
85
|
+
* read side does not, the two derive different keys from the same path and a
|
|
86
|
+
* lesson becomes unreachable through the very file it is about. Both sides call
|
|
87
|
+
* THIS, so they cannot drift apart — the same rule this repo already enforces
|
|
88
|
+
* for `fileMatchClause`'s two consumers.
|
|
89
|
+
*
|
|
90
|
+
* Total by construction: it never throws, because one caller is a hook on the
|
|
91
|
+
* PreToolUse path. Nullish becomes '' (`p ?? ''`); anything else becomes its
|
|
92
|
+
* String() form, so 42 yields '42', not ''.
|
|
93
|
+
*/
|
|
94
|
+
export function scrubFilePath(p) {
|
|
95
|
+
// SEGMENT-WISE, and that is the whole point rather than a micro-optimisation.
|
|
96
|
+
// Eight SECRET_PATTERNS carry a value class that does not exclude `/`
|
|
97
|
+
// (secret-scrub.mjs:33/74/78/83/98/109/259/260). On prose that is correct; run
|
|
98
|
+
// whole-path, the match eats the separator and everything after it, so
|
|
99
|
+
// `/repo/token=<secret>/notes.mjs` became `/repo/token=***` — the filename
|
|
100
|
+
// destroyed at WRITE time and unrecoverable, and every file under such a
|
|
101
|
+
// directory collapsing onto one recall key (measured: an untouched `gamma.mjs`
|
|
102
|
+
// recalled another file's observations). Splitting first bounds every pattern to
|
|
103
|
+
// the segment it matched in.
|
|
104
|
+
//
|
|
105
|
+
// The trade, stated rather than glossed: a credential whose own syntax spans a
|
|
106
|
+
// separator is no longer caught here — the Slack webhook path and the
|
|
107
|
+
// `scheme://user:pass@host` arms both need their `/` characters. Those are URL
|
|
108
|
+
// shapes, and these columns hold filesystem paths; `scrubSecrets` still runs
|
|
109
|
+
// whole-string on every prose field, which is where a URL actually lands.
|
|
110
|
+
// Preserving path structure wins because the path IS the recall key.
|
|
111
|
+
//
|
|
112
|
+
// The capture group keeps the separators in the split output, so join()
|
|
113
|
+
// reconstructs the original byte-for-byte when nothing matches.
|
|
114
|
+
return String(p ?? '')
|
|
115
|
+
.split(/([/\\])/)
|
|
116
|
+
.map((part) => (part === '/' || part === '\\' ? part : scrubSecrets(part)))
|
|
117
|
+
.join('');
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Element-wise scrub for a path ARRAY, to be called upstream of the
|
|
122
|
+
* `JSON.stringify` / junction INSERT this module's header points at.
|
|
123
|
+
*
|
|
124
|
+
* A non-array flows through UNTOUCHED, mirroring scrubRecord's own contract for
|
|
125
|
+
* non-string fields. That is load-bearing rather than defensive: call sites pass
|
|
126
|
+
* `undefined` on purpose (`JSON.stringify(undefined)` is `undefined`, which is
|
|
127
|
+
* how a column stays NULL), and coercing it to `[]` here would quietly rewrite
|
|
128
|
+
* NULL to '[]' in columns other code tests with `IS NULL` / `NOT IN (NULL,'[]')`.
|
|
129
|
+
*/
|
|
130
|
+
export function scrubFilePaths(paths) {
|
|
131
|
+
return Array.isArray(paths) ? paths.map(scrubFilePath) : paths;
|
|
132
|
+
}
|
|
133
|
+
|
|
70
134
|
/**
|
|
71
135
|
* Scrub the text fields of a record before INSERT.
|
|
72
136
|
* Returns a shallow copy with string text-fields scrubbed; the input object
|
package/mem-cli.mjs
CHANGED
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
import { homedir } from 'os';
|
|
6
6
|
import { ensureDbWithWalRecovery, DB_PATH, DB_DIR, CODE_DIR } from './schema.mjs';
|
|
7
7
|
import { resolveRuntimeDir } from './lib/resolve-data-dir.mjs';
|
|
8
|
+
import { isFtsCorruptionError, FTS_CORRUPTION_REMEDY } from './lib/db-unusable.mjs';
|
|
8
9
|
import { truncate, typeIcon, inferProject, scrubSecrets, COMPRESSED_PENDING_PURGE } from './utils.mjs';
|
|
9
10
|
import { resolveProject } from './project-utils.mjs';
|
|
10
11
|
// READ commands resolve the project DB-aware: a subdirectory whose own name holds no rows
|
|
@@ -3847,6 +3848,12 @@ export async function run(argv) {
|
|
|
3847
3848
|
// agent runs the CLI, the model's context — got a raw Node stack trace. Print the
|
|
3848
3849
|
// message, keep the stack behind CLAUDE_MEM_DEBUG for whoever is actually debugging.
|
|
3849
3850
|
process.stderr.write(`[mem] ${cmd || 'command'} failed: ${(e && e.message) || e}\n`);
|
|
3851
|
+
// A damaged FTS5 index reaches here as SQLITE_CORRUPT_VTAB from the first MATCH.
|
|
3852
|
+
// `fts-check` and `doctor` touch the index too — and both already explain themselves —
|
|
3853
|
+
// so `search` is the one command that DEAD-ENDS on SQLite's sentence, while `recent` /
|
|
3854
|
+
// `recall` / `browse` / `context` / `stats` never read the index and keep working.
|
|
3855
|
+
// The remedy is lossless (see FTS_CORRUPTION_REMEDY); the exit code stays 1.
|
|
3856
|
+
if (isFtsCorruptionError(e)) process.stderr.write(`[mem] ${FTS_CORRUPTION_REMEDY}\n`);
|
|
3850
3857
|
if (process.env.CLAUDE_MEM_DEBUG) process.stderr.write(`${(e && e.stack) || ''}\n`);
|
|
3851
3858
|
process.exitCode = 1;
|
|
3852
3859
|
} finally {
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.8.
|
|
3
|
+
"version": "6.8.3",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "claude-mem-lite",
|
|
9
|
-
"version": "6.8.
|
|
9
|
+
"version": "6.8.3",
|
|
10
10
|
"os": [
|
|
11
11
|
"darwin",
|
|
12
12
|
"linux",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.8.
|
|
3
|
+
"version": "6.8.3",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. Hybrid FTS5 + TF-IDF search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"packageManager": "npm@10.9.2",
|
|
@@ -27,6 +27,7 @@ import {
|
|
|
27
27
|
jsonArrayLikeNeedle,
|
|
28
28
|
toolEditPath,
|
|
29
29
|
} from '../lib/file-edge-match.mjs';
|
|
30
|
+
import { scrubFilePath } from '../lib/scrub-record.mjs';
|
|
30
31
|
import { fileIntelFor } from '../lib/file-intel.mjs';
|
|
31
32
|
import { shouldWarnReread, buildRereadWarning, readFileMeta } from '../lib/reread-guard.mjs';
|
|
32
33
|
import { recordMetric } from '../lib/metrics.mjs';
|
|
@@ -538,7 +539,14 @@ try {
|
|
|
538
539
|
// needs the same key, and host-native `basename` gave it the whole path for a
|
|
539
540
|
// Windows-shaped payload. Fixing the observations leg alone would have left this
|
|
540
541
|
// hook recalling lessons but no events.
|
|
541
|
-
|
|
542
|
+
// D#44: the KEY derivation is scrubbed; `filePath` itself is NOT reused for
|
|
543
|
+
// this, because the same variable is still handed to readFileMeta and friends
|
|
544
|
+
// to stat a real file on disk — a globally scrubbed path would point nowhere.
|
|
545
|
+
// events.file_paths is now written pre-scrubbed (lib/activity.mjs), and the
|
|
546
|
+
// observations leg below scrubs inside fileMatchParams, so both legs of this
|
|
547
|
+
// hook keep deriving the same key — which is what the note above requires.
|
|
548
|
+
const keyPath = scrubFilePath(filePath);
|
|
549
|
+
const fname = basenameAnySep(keyPath);
|
|
542
550
|
// Needle for the events leg's JSON-array column — see jsonArrayLikeNeedle for
|
|
543
551
|
// why the JSON escape has to run before the LIKE one. The observations leg
|
|
544
552
|
// below matches a plain column and gets its params from fileMatchParams.
|
|
@@ -652,7 +660,7 @@ try {
|
|
|
652
660
|
// patterns match both basename and full-path entries. JSON quoting
|
|
653
661
|
// (`"<name>"`) prevents partial-match false positives like "foo.mjs"
|
|
654
662
|
// matching "myfoo.mjs".
|
|
655
|
-
const fullPathNeedle = jsonArrayLikeNeedle(
|
|
663
|
+
const fullPathNeedle = jsonArrayLikeNeedle(keyPath);
|
|
656
664
|
// v2.34.6: Read also tightens the events query — only rows with a non-empty
|
|
657
665
|
// body (= lesson equivalent). Edit path keeps a wider net, but P0 (D#78)
|
|
658
666
|
// closes the parallel-path drift vs the observations query: a bodyless row
|
|
@@ -225,9 +225,26 @@ export function shouldSkipByDedup(newIds, injectedFile, sessionId) {
|
|
|
225
225
|
if (!newIds || newIds.length === 0) return true;
|
|
226
226
|
try {
|
|
227
227
|
const raw = readFileSync(injectedFile, 'utf8');
|
|
228
|
-
const { ids: prevIds, ts,
|
|
228
|
+
const { ids: prevIds, ts, upsCount = 0, upsTs = 0, session } = JSON.parse(raw);
|
|
229
229
|
if (session && sessionId && session !== sessionId) return false;
|
|
230
|
-
|
|
230
|
+
// R12 B-5, two defects on these two lines, and they had to be fixed together.
|
|
231
|
+
//
|
|
232
|
+
// (1) The cap reads `upsCount`, not the shared `count`. `count` is bumped by every
|
|
233
|
+
// hook that writes this marker — pre-tool-recall does so once per triggered Edit/Read
|
|
234
|
+
// — so a session that touched 15 lesson-bearing files had already spent this face's
|
|
235
|
+
// whole budget before it injected anything. Markers written before this field existed
|
|
236
|
+
// read 0, which releases a cap that should never have been charged.
|
|
237
|
+
//
|
|
238
|
+
// (2) Freshness is judged FIRST. The cap used to sit above it, so once `count` hit the
|
|
239
|
+
// ceiling the marker suppressed injection even after it went stale — and nothing
|
|
240
|
+
// lowers it again, since the reset only happens on the next WRITE and the write never
|
|
241
|
+
// comes. The only escape was a new session id.
|
|
242
|
+
// (3) The cap is judged on `upsTs`, the UPS face's OWN clock. `ts` moves on every
|
|
243
|
+
// writer's write, so judging the budget against it let another hook's activity keep a
|
|
244
|
+
// spent budget alive indefinitely — the same spender/charged mismatch as (1), arriving
|
|
245
|
+
// through the clock instead of the counter. A marker with no `upsTs` (anything written
|
|
246
|
+
// before this field existed) reads 0 and never caps, matching upsCount's default.
|
|
247
|
+
if (upsTs && Date.now() - upsTs <= DEDUP_STALE_MS && upsCount >= MAX_SESSION_INJECTIONS) return true;
|
|
231
248
|
if (!ts || Date.now() - ts > DEDUP_STALE_MS) return false;
|
|
232
249
|
if (!Array.isArray(prevIds) || prevIds.length === 0) return false;
|
|
233
250
|
// Normalize both sides to strings before comparing: UPS writes obs ids as numbers
|
|
@@ -798,6 +798,12 @@ async function main() {
|
|
|
798
798
|
sessionId: hookData.session_id,
|
|
799
799
|
maxAgeMs: DEDUP_STALE_MS,
|
|
800
800
|
mode: 'union',
|
|
801
|
+
// This leg is GATED by shouldSkipByDedup, so it has to charge the cap it reads.
|
|
802
|
+
// Before B-5 it did, because the cap was the shared `count` this write bumps;
|
|
803
|
+
// moving the cap to `upsCount` left the gated population {main leg, D#N leg}
|
|
804
|
+
// larger than the charged population {main leg}. Same spender/charged mismatch
|
|
805
|
+
// B-5 fixed, on the sibling call site.
|
|
806
|
+
bumpUpsCount: true,
|
|
801
807
|
});
|
|
802
808
|
} catch {}
|
|
803
809
|
}
|
|
@@ -1083,6 +1089,10 @@ async function main() {
|
|
|
1083
1089
|
sessionId: hookData.session_id,
|
|
1084
1090
|
maxAgeMs: DEDUP_STALE_MS,
|
|
1085
1091
|
mode: 'replace',
|
|
1092
|
+
// R12 B-5: this is the leg MAX_SESSION_INJECTIONS budgets, so it is the only
|
|
1093
|
+
// one that charges against it. The shared `count` is bumped by every hook that
|
|
1094
|
+
// touches this file and is no longer what the cap reads.
|
|
1095
|
+
bumpUpsCount: true,
|
|
1086
1096
|
});
|
|
1087
1097
|
} catch {}
|
|
1088
1098
|
// v26 P0: bump injection_count for obs-based emits only (prompt-corpus
|
package/server.mjs
CHANGED
|
@@ -118,6 +118,7 @@ import { saveWithClosures, formatSupersedeSkipped, formatSupersededNote } from '
|
|
|
118
118
|
import { applyObsUpdate } from './lib/observation-write.mjs';
|
|
119
119
|
import { EXPORT_COLUMNS_SQL, buildExportWhere } from './lib/export-columns.mjs';
|
|
120
120
|
import { recallByFile } from './lib/recall-core.mjs';
|
|
121
|
+
import { isFtsCorruptionError, FTS_CORRUPTION_REMEDY } from './lib/db-unusable.mjs';
|
|
121
122
|
import { fetchRecent } from './lib/recent-core.mjs';
|
|
122
123
|
import { AUTO_MERGE_THRESHOLD } from './lib/dedup-constants.mjs';
|
|
123
124
|
import {
|
|
@@ -374,7 +375,17 @@ function safeHandler(fn, { verbatim = false } = {}) {
|
|
|
374
375
|
const result = await fn(args, extra);
|
|
375
376
|
return verbatim ? result : defangResult(result, { skillBlocks: true });
|
|
376
377
|
} catch (err) {
|
|
377
|
-
|
|
378
|
+
// A damaged FTS5 index arrives here as SQLITE_CORRUPT_VTAB from the first MATCH.
|
|
379
|
+
// Without this the model got SQLite's own sentence and nothing else, on a fault it
|
|
380
|
+
// could have had fixed in one command — and mem_recent / mem_recall / mem_browse
|
|
381
|
+
// keep answering, so the dead end reads as "nothing matched". Both channels carry the
|
|
382
|
+
// same string here, deliberately: unlike the file-level remedy this one is lossless
|
|
383
|
+
// (see FTS_CORRUPTION_REMEDY in lib/db-unusable.mjs).
|
|
384
|
+
const hint = isFtsCorruptionError(err) ? `\n${FTS_CORRUPTION_REMEDY}` : '';
|
|
385
|
+
return defangResult({
|
|
386
|
+
content: [{ type: 'text', text: `Error: ${err.message}${hint}` }],
|
|
387
|
+
isError: true,
|
|
388
|
+
});
|
|
378
389
|
}
|
|
379
390
|
};
|
|
380
391
|
}
|