claude-mem-lite 6.8.2 → 6.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@
9
9
  "plugins": [
10
10
  {
11
11
  "name": "claude-mem-lite",
12
- "version": "6.8.2",
12
+ "version": "6.9.0",
13
13
  "source": "./",
14
14
  "homepage": "https://github.com/sdsrss/claude-mem-lite",
15
15
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. Hybrid FTS5 + TF-IDF search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.8.2",
3
+ "version": "6.9.0",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. Hybrid FTS5 + TF-IDF search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "author": {
6
6
  "name": "sdsrss"
package/README.md CHANGED
@@ -128,7 +128,7 @@ How claude-mem-lite differs from the major neighbors in the LLM-memory space (ve
128
128
  - **FTS integrity management** -- `mem_fts_check` tool verifies FTS5 index health or rebuilds indexes on demand, useful after database recovery or when search results seem wrong
129
129
  - **Atomic multi-table writes** -- `saveObservation` wraps the observations + observation_files INSERTs in a single `db.transaction()`, preventing orphaned rows on crash
130
130
  - **Modular NLP pipeline** -- Synonym maps, stop words, scoring constants, and query building extracted into focused modules (`synonyms.mjs`, `stop-words.mjs`, `scoring-sql.mjs`, `nlp.mjs`) for independent testing and maintenance
131
- - **Porter-aligned PRF** -- Pseudo-relevance feedback terms are now stemmed with the same Porter algorithm used by FTS5, ensuring PRF expansion terms match the search index
131
+ - **Surface-form PRF expansion** -- `observations_fts` is built on FTS5's default `unicode61` tokenizer, so the index is **not stemmed**: a query term matches the word forms actually stored, and `crash` does not match a row that only contains `crashes`. Pseudo-relevance feedback therefore uses the Porter stemmer only to *bucket* morphological variants when judging which candidate terms are discriminative, and emits the most frequent **surface** form of each — emitting a bare stem (`cach`) would match nothing and kill expansion recall
132
132
 
133
133
  ## Platform Support
134
134
 
@@ -735,7 +735,7 @@ claude-mem-lite/
735
735
  hook-semaphore.mjs # LLM concurrency control: file-based semaphore for background workers
736
736
  schema.mjs # Database schema: single source of truth for tables, migrations, FTS5
737
737
  tool-schemas.mjs # Shared Zod schemas for MCP tool validation
738
- tfidf.mjs # tokenization + Porter stemming (name is historical: the TF-IDF vector engine it held was removed)
738
+ tfidf.mjs # the Porter stemmer (name is historical: the TF-IDF vector engine it held was removed)
739
739
  tier.mjs # Temporal tier system: activity-based time window classification
740
740
  utils.mjs # Re-export hub: backward-compatible surface for all utility modules
741
741
  nlp.mjs # FTS5 query building: synonym expansion, CJK bigrams, sanitization
package/adopt-cli.mjs CHANGED
@@ -425,6 +425,13 @@ function unadoptAll(args) {
425
425
  log(`[unadopt --all] ${dir} → cleaned partial residue (detail doc/state, no block)`);
426
426
  partial++;
427
427
  }
428
+ // OUTSIDE the branch, because residue is orthogonal to what happened to the block: a
429
+ // project can have its block removed AND still carry an unpaired sentinel. An orphan is
430
+ // the one kind of residue the sweep cannot finish — its block has no end marker, so its
431
+ // extent is unknowable — and the two lines above would otherwise imply the project is
432
+ // clean. Inside the else-branch it was also unreachable for the 'removed' case, which is
433
+ // how the print survived a mutation with the whole suite green (pre-ship review P2-2).
434
+ if (r.residue) log(` ⚠ ${r.residue}`);
428
435
  }
429
436
 
430
437
  // 2. Legacy memory-dir cleanup across every memdir (foreign-content guarded).
@@ -481,4 +488,8 @@ export function cmdUnadopt(args = []) {
481
488
  const mig = migrateLegacyMemoryDir(cwd, PLUGIN_SLUG, { force });
482
489
  const migNote = mig.action === 'removed' ? ' (+cleaned legacy memdir)' : '';
483
490
  log(`[unadopt] ${cwd} → ${r.action}${migNote}`);
491
+ // 'partial' is the outcome that used to print as 'absent': the sidecar files are gone but
492
+ // an unpaired sentinel still holds steering text in the user's CLAUDE.md, and only they can
493
+ // decide where that text ends. Silence here is what let it survive every sweep.
494
+ if (r.residue) log(` ⚠ ${r.residue}`);
484
495
  }
package/claudemd.mjs CHANGED
@@ -59,8 +59,50 @@ function escapeRe(s) {
59
59
  // are `\r?\n` (not bare `\n`) so a CLAUDE.md re-saved with Windows CRLF endings
60
60
  // still matches — otherwise the block read as "absent" and a fresh LF copy got
61
61
  // appended every SessionStart, growing the file without bound (review C1/H2).
62
+ // The body may not contain ANOTHER sentinel of the same slug. `[\s\S]*?` could, and that is
63
+ // not a tidiness point — it is how a match stopped being one block. Drop the `:end` line by
64
+ // hand (a merge resolution, an editor, another tool) and the next adopt appends a second
65
+ // block below whatever the user has written since; the adopt after THAT matched from the
66
+ // orphaned begin, lazily, to the only `:end` in the file — which now sits past the user's
67
+ // text and past the second begin — so `raw.replace(m[0], section)` deleted all of it.
68
+ // Measured 2026-09-13: a "## Deployment runbook" section appended after adoption was gone
69
+ // after two further adopts, silently, with both runs reporting success.
70
+ //
71
+ // The tempered token below cannot span a sentinel, so the engine backtracks to the
72
+ // WELL-FORMED pair and the orphan is simply left alone — which is the right answer for a
73
+ // file we do not own: where the orphaned body ends is genuinely unknowable, so removeManaged
74
+ // reports it (action 'partial') rather than guessing a span to delete.
75
+ //
76
+ // Safe by construction for legitimate blocks: the shipped body carries the slug twice and
77
+ // never as a sentinel (measured: 1304 bytes — 1296 UTF-16 units, the body has em dashes —
78
+ // with zero `:begin` / `:end` occurrences), and the
79
+ // separators stay `\r?\n` for the CRLF reason below.
62
80
  function blockBody(esc) {
63
- return `<!-- ${esc}:begin (v\\d+) -->\\r?\\n([\\s\\S]*?)\\r?\\n<!-- ${esc}:end -->`;
81
+ const sentinel = `<!-- ${esc}:(?:begin|end)`;
82
+ return `<!-- ${esc}:begin (v\\d+) -->\\r?\\n((?:(?!${sentinel})[\\s\\S])*?)\\r?\\n<!-- ${esc}:end -->`;
83
+ }
84
+
85
+ // Any sentinel LINE of our slug, paired or not. The pair regex above is deliberately blind to
86
+ // an unpaired one; this is what lets residue reporting see what it cannot safely remove.
87
+ function sentinelLineRegexG(slug) {
88
+ return new RegExp(`<!-- ${escapeRe(slug)}:(?:begin|end)\\b[^>]*-->`, 'g');
89
+ }
90
+
91
+ /**
92
+ * Sentinel lines of this slug left in `raw` that no well-formed block accounts for.
93
+ * Zero on a healthy file (every sentinel belongs to a matched pair) and on a clean one.
94
+ * @param {string} raw
95
+ * @param {string} slug
96
+ * @returns {number}
97
+ */
98
+ function orphanSentinelCount(raw, slug) {
99
+ const total = (raw.match(sentinelLineRegexG(slug)) || []).length;
100
+ let paired = 0;
101
+ raw.replace(blockRegexG(slug), (whole) => {
102
+ paired += (whole.match(sentinelLineRegexG(slug)) || []).length;
103
+ return whole;
104
+ });
105
+ return total - paired;
64
106
  }
65
107
  function blockRegex(slug) {
66
108
  return new RegExp(blockBody(escapeRe(slug)));
@@ -137,13 +179,24 @@ export function isAdopted(cwd, slug) {
137
179
  * removed it. removeManaged cleans all three pieces, so sweep on any of them.
138
180
  */
139
181
  export function hasResidue(cwd, slug) {
140
- return (
141
- readBlock(cwd, slug).body !== null ||
142
- existsSync(detailDocPath(cwd, slug)) ||
143
- existsSync(stateFilePath(cwd, slug))
144
- );
182
+ const blk = readBlock(cwd, slug);
183
+ return blk.body !== null || existsSync(detailDocPath(cwd, slug)) || existsSync(stateFilePath(cwd, slug));
145
184
  }
146
185
 
186
+ // An unpaired sentinel is deliberately NOT in the list above (pre-ship review P2-3). A first
187
+ // cut added it, reasoning that the sweep should see what the pair regex cannot. But
188
+ // orphanSentinelCount counts sentinel-shaped TEXT, and a project the plugin never touched can
189
+ // mention the marker in prose — documenting it, pasting half an example, a changelog line.
190
+ // That one mention let `unadopt --all` into a stranger's project, where removeManaged
191
+ // unconditionally deletes the detail doc and state sidecar and rmdir's an empty `.claude/`,
192
+ // then printed "remove those lines by hand" at the user's own paragraph — and never
193
+ // converged, because the mention is still there on the next sweep.
194
+ //
195
+ // The three entries above are all things the PLUGIN WROTE; a sentinel in prose is not. And
196
+ // the sweep gains nothing by entering: removeManaged cannot clean an orphan anyway, by
197
+ // design. The orphan is reported by removeManaged when unadopt genuinely runs — which is the
198
+ // real failure case, where the doc and sidecar are still present and do bring it in.
199
+
147
200
  /**
148
201
  * Whether the installed block/doc has drifted from the shipped content — i.e.
149
202
  * a version bump or a template edit means we should refresh. Returns true when
@@ -232,11 +285,25 @@ export function writeManaged(cwd, { slug, version, block, doc }) {
232
285
  * Remove our managed block from CLAUDE.md (preserving all other content) and
233
286
  * delete the detail doc + state sidecar. Best-effort removes an emptied
234
287
  * .claude/ directory.
235
- * @returns {{action: 'removed'|'absent'}}
288
+ *
289
+ * THREE outcomes, not two — the same rule lib/db-unusable.mjs states about backups: "there
290
+ * is nothing to do" and "I could not finish" must not print in the same voice, because a
291
+ * green-sounding line ends the reader's search. `absent` used to cover both: with one
292
+ * sentinel line missing, the pair regex matched nothing, so this returned 'absent' — while
293
+ * having already deleted the detail doc and the state sidecar and left ~1.3 KB of managed
294
+ * steering text in the user's CLAUDE.md, which it is then loaded from on every session.
295
+ * `partial` is that case, and `residue` names what is left so the caller can say so.
296
+ *
297
+ * Deliberately does NOT delete an orphaned sentinel's body: where it ends is unknowable
298
+ * (that is the defect, not a detail), and guessing a span in a file we do not own is how the
299
+ * adopt side came to delete a user's runbook. Report, do not repair.
300
+ *
301
+ * @returns {{action: 'removed'|'partial'|'absent', residue?: string}}
236
302
  */
237
303
  export function removeManaged(cwd, slug) {
238
304
  const p = claudeMdPath(cwd);
239
305
  let action = 'absent';
306
+ let orphans = 0;
240
307
  if (existsSync(p)) {
241
308
  let raw = readFileSync(p, 'utf8');
242
309
  // H2: loop so ALL same-slug blocks are removed, not just the first (a
@@ -283,8 +350,17 @@ export function removeManaged(cwd, slug) {
283
350
  atomicWrite(p, raw);
284
351
  }
285
352
  }
353
+ // Counted on what is left AFTER the loop, so a healthy file (every sentinel consumed by
354
+ // a matched pair) reports zero and only a genuinely unpaired line survives the count.
355
+ orphans = orphanSentinelCount(raw, slug);
286
356
  }
357
+ // Captured BEFORE the deletions below, because they are what it asks about: is there any
358
+ // evidence the plugin ever wrote in this project? An unpaired sentinel is NOT such
359
+ // evidence — it is text, and a project that merely documents the marker in prose has one
360
+ // (pre-ship review P2-3). Reporting residue there means telling a stranger to delete their
361
+ // own paragraph, on a project this tool has never touched.
287
362
  const dp = detailDocPath(cwd, slug);
363
+ const wasOurs = action === 'removed' || existsSync(dp) || existsSync(stateFilePath(cwd, slug));
288
364
  if (existsSync(dp))
289
365
  try {
290
366
  unlinkSync(dp);
@@ -300,6 +376,18 @@ export function removeManaged(cwd, slug) {
300
376
  } catch {
301
377
  /* best-effort */
302
378
  }
379
+ // `action` answers ONE question — what happened to the block — and `residue` is an
380
+ // independent fact that rides alongside it. A first cut let an orphan override 'removed'
381
+ // too, on the reasoning that both are "unfinished". Pre-ship review P2-1: unadoptAll's
382
+ // else-branch prints "cleaned partial residue (detail doc/state, no block)" and counts
383
+ // `partial++`, so a sweep that DID remove a block reported "no block" and tallied zero
384
+ // removals. Two facts, two fields.
385
+ const residue =
386
+ orphans > 0 && wasOurs
387
+ ? `${orphans} unpaired \`${slug}\` sentinel line(s) remain in ${claudeMdPath(cwd)} — the block they opened has no matching end marker, so its extent cannot be determined safely. Remove those lines and the text they wrap by hand.`
388
+ : null;
389
+ if (action === 'removed') return residue ? { action, residue } : { action };
390
+ if (residue) return { action: 'partial', residue };
303
391
  return { action };
304
392
  }
305
393
 
package/cli/common.mjs CHANGED
@@ -83,7 +83,120 @@ export function parseArgs(argv) {
83
83
  i++;
84
84
  }
85
85
  }
86
- return { positional, flags };
86
+ return { positional, flags: trackFlagReads(flags) };
87
+ }
88
+
89
+ // ─── Inert selection flags ───────────────────────────────────────────────────
90
+ //
91
+ // The third cause of the harm parseArgs' docblock names twice. `--include_noise` (underscore
92
+ // spelling) and `--obs_type` (an MCP field name) both parsed, matched no reader, and let the
93
+ // command answer the unfiltered question; both are fixed above. `suggestUnknownFlags` fixes a
94
+ // third case, flags unknown to the whole CLI. What is left is the case where NOTHING is
95
+ // misspelled: `--type` is canonical and real — on `search`, `save` and `export` — so it clears
96
+ // the global known-flag set, and `browse` simply never reads it. Measured before the fix:
97
+ // `browse --type bugfix` printed rows of every type, exit 0, not a word.
98
+ //
99
+ // The check is read-tracking, not a per-command flag manifest, and that is the whole design.
100
+ // A manifest rots, and worse, it cannot see a flag that a command forwards wholesale into a
101
+ // core helper (`cmdSearch` hands the object to the pipeline, `cmdExport` to the writer) —
102
+ // deriving "which flags does this command read" from its own body would fire on every one of
103
+ // those. Watching what the object is actually ASKED for is exact in both directions.
104
+ //
105
+ // Scope is SELECTION-shaped flags only. Those are the ones whose silent drop hands back a
106
+ // wider set that reads as the answer, which is the harm. A `--confirm` that a short-circuited
107
+ // branch never reached is deliberately out of scope: nothing there is wrong, and a warning on
108
+ // correct usage is worse than the silence it replaces.
109
+ //
110
+ // `prompts-limit` is NOT here, and the reason generalises: `doctor --benchmark --prompts-limit`
111
+ // is read off raw `process.argv` in cli/doctor.mjs and never touches a flags object, so
112
+ // read-tracking would call a working flag inert. Anything read off argv must stay out.
113
+ export const FILTER_FLAGS = new Set([
114
+ 'type',
115
+ 'source',
116
+ 'project',
117
+ 'tier',
118
+ 'since',
119
+ 'from',
120
+ 'to',
121
+ 'branch',
122
+ 'scope',
123
+ 'importance',
124
+ 'limit',
125
+ 'offset',
126
+ 'sort',
127
+ 'days',
128
+ 'age-days',
129
+ // Inclusion toggles, added after the same correct-usage sweep the first batch passed. They
130
+ // widen or narrow the SET rather than filter within it, and the harm runs the other way:
131
+ // an ignored `--include-noise` hands back FEWER rows than asked for, and "I searched and it
132
+ // was not there" is the worst answer a memory tool can give. Held back at first only
133
+ // because booleans are read inside branches and were the likelier false-alarm shape; the
134
+ // exit-code gate below turned out to cover that class, and the sweep reads zero.
135
+ 'all',
136
+ 'include-noise',
137
+ 'include-compressed',
138
+ 'deep',
139
+ 'no-deep',
140
+ 'or',
141
+ 'rerank',
142
+ ]);
143
+
144
+ let suppliedFlags = new Set();
145
+ let readFlags = new Set();
146
+
147
+ /**
148
+ * Wrap a parsed flags object so every lookup is recorded.
149
+ *
150
+ * `get` and `has` are both trapped: a reader spelled `if ('tier' in flags)` must count as a
151
+ * read exactly like `flags.tier`. `ownKeys` deliberately is NOT — `Object.keys(flags)` is
152
+ * enumeration, not consumption, and `suggestUnknownFlags` does exactly that on its own parse.
153
+ */
154
+ function trackFlagReads(flags) {
155
+ for (const k of Object.keys(flags)) suppliedFlags.add(k);
156
+ return new Proxy(flags, {
157
+ get(target, prop, recv) {
158
+ if (typeof prop === 'string') readFlags.add(prop);
159
+ return Reflect.get(target, prop, recv);
160
+ },
161
+ has(target, prop) {
162
+ if (typeof prop === 'string') readFlags.add(prop);
163
+ return Reflect.has(target, prop);
164
+ },
165
+ });
166
+ }
167
+
168
+ /** Start a fresh recording. `run()` calls this per invocation so tests can drive it in a loop. */
169
+ export function resetFlagTracking() {
170
+ suppliedFlags = new Set();
171
+ readFlags = new Set();
172
+ }
173
+
174
+ /**
175
+ * Selection flags the user supplied that nothing looked at, sorted.
176
+ *
177
+ * Read AFTER the command has finished — a flag consumed late (inside a branch, or by a helper
178
+ * the command awaits) has still been read, and reporting it early would be a false alarm.
179
+ * @returns {string[]}
180
+ */
181
+ export function inertFilterFlags() {
182
+ return [...suppliedFlags].filter((f) => FILTER_FLAGS.has(f) && !readFlags.has(f)).sort();
183
+ }
184
+
185
+ /**
186
+ * The sentence the user gets. Says what happened to their results, not what the parser did:
187
+ * "ignored" alone leaves them to work out whether the output is still the answer they asked
188
+ * for. It is not.
189
+ * @param {string} cmd
190
+ * @param {string[]} inert
191
+ * @returns {string}
192
+ */
193
+ export function inertFilterFlagNotice(cmd, inert) {
194
+ const names = inert.map((f) => `--${f}`).join(' and ');
195
+ const verb = inert.length > 1 ? 'were' : 'was';
196
+ return (
197
+ `[mem] ${names} ${verb} ignored — \`${cmd}\` does not filter on ${inert.length > 1 ? 'them' : 'it'}, ` +
198
+ `so the results above are UNFILTERED. Run "claude-mem-lite help" for the flags this command reads.`
199
+ );
87
200
  }
88
201
 
89
202
  // ─── Output Helpers ──────────────────────────────────────────────────────────
package/cli.mjs CHANGED
@@ -81,6 +81,28 @@ function fileFromError(e) {
81
81
  return quoted ? quoted[1] : null;
82
82
  }
83
83
 
84
+ /**
85
+ * Is this `doctor` invocation one of the DB-layer modes?
86
+ *
87
+ * DYNAMIC on purpose. A static import here would put lib/doctor-modes.mjs in the LAUNCHER's
88
+ * load graph, and a missing file there kills cli.mjs before any of its own error handling
89
+ * exists — the user gets a raw ERR_MODULE_NOT_FOUND instead of "this install is incomplete,
90
+ * run repair". tests/doctor-startup-closure.test.mjs caught exactly that when the import was
91
+ * static. Same rule as "a recovery path must not import the thing it recovers", applied to
92
+ * the entry point: nothing the launcher needs before it can speak may be a hard edge.
93
+ * install.mjs may import it statically — that failure is caught by loadInstaller() below.
94
+ */
95
+ async function isDoctorDbMode() {
96
+ try {
97
+ const { DOCTOR_DB_MODES } = await import('./lib/doctor-modes.mjs');
98
+ return process.argv.slice(3).some((a) => DOCTOR_DB_MODES.some((m) => a === `--${m}`));
99
+ } catch {
100
+ // Unreadable module → treat as the plain install health check, which is the branch that
101
+ // can still explain a broken install.
102
+ return false;
103
+ }
104
+ }
105
+
84
106
  async function loadInstaller() {
85
107
  let mod;
86
108
  try {
@@ -129,10 +151,7 @@ if (cmd === '--version' || cmd === '-v' || cmd === '-V' || cmd === 'version') {
129
151
  } else if (cmd === '--help' || cmd === '-h') {
130
152
  const { run } = await import('./mem-cli.mjs');
131
153
  await run(['help']);
132
- } else if (
133
- cmd === 'doctor' &&
134
- process.argv.slice(3).some((a) => a === '--benchmark' || a === '--metrics' || a === '--session-audit')
135
- ) {
154
+ } else if (cmd === 'doctor' && (await isDoctorDbMode())) {
136
155
  // Per #8217: the DB-layer doctor modes (--benchmark / --metrics / --session-audit,
137
156
  // each implemented in cli/doctor.mjs) route to mem-cli. Everything else — plain
138
157
  // `doctor`, `doctor --` (POSIX end-of-options), and `doctor --json` — stays with
package/format-utils.mjs CHANGED
@@ -25,6 +25,36 @@ export function truncate(str, max = 80) {
25
25
  return str.slice(0, end) + '\u2026';
26
26
  }
27
27
 
28
+ /**
29
+ * Longest query echoed back verbatim in a result label. Long enough that no query a human
30
+ * or an agent actually types is touched — the shapes that exceed it are pasted stack traces,
31
+ * file dumps and multi-paragraph questions.
32
+ */
33
+ export const QUERY_LABEL_MAX = 200;
34
+
35
+ /**
36
+ * A query as it should appear in output handed back to whoever asked.
37
+ *
38
+ * Every search surface labels its answer with the query (`Found N result(s) for "<query>"`),
39
+ * which is how a caller confirms what was actually searched. Unbounded, that made the size
40
+ * of the answer track the size of the question: a 50,000-character query produced 50,024
41
+ * characters of CLI output, and the MCP face — with no argv ceiling — carried the whole
42
+ * thing back into the model's context. For a tool whose purpose is to spend context
43
+ * carefully, returning several KB of the caller's own input is the budget it was invoked to
44
+ * protect.
45
+ *
46
+ * Bounded, not silently truncated: the real length travels with the prefix, so the label
47
+ * still answers the question it exists for. `truncate` handles the surrogate-pair boundary.
48
+ *
49
+ * @param {string} query
50
+ * @returns {string}
51
+ */
52
+ export function queryLabel(query) {
53
+ if (typeof query !== 'string') return '';
54
+ if (query.length <= QUERY_LABEL_MAX) return query;
55
+ return `${truncate(query, QUERY_LABEL_MAX)} [query truncated; ${query.length} chars]`;
56
+ }
57
+
28
58
  // Two delimiter classes are defanged here:
29
59
  // 1. The blocks claude-mem-lite wraps injected context in (claude-mem-context /
30
60
  // memory-context / session-handoff). User-derived text containing one LITERALLY
package/hook-episode.mjs CHANGED
@@ -18,12 +18,21 @@ import { inferProject, EDIT_TOOLS } from './utils.mjs';
18
18
  import { RUNTIME_DIR } from './hook-shared.mjs';
19
19
 
20
20
  /**
21
- * Read episode file without locking (for signal handlers only).
21
+ * Read the episode buffer WITHOUT holding the lock: the dying-process salvage in hook.mjs's
22
+ * signal handler, and the snapshot Stop / SessionStart take before they flush.
23
+ *
24
+ * Same body as `readEpisode` on purpose — the two names record which locking contract the
25
+ * CALLER is under, and neither function takes a lock itself. What must not differ is the
26
+ * PATH, so both go through `episodeFile()`. This one used to re-spell it inline, which put
27
+ * the buffer's name in three places (here, `episodeFile`, and the signal handler's unlink)
28
+ * and left the salvage path — the one that runs while the process is dying, and the hardest
29
+ * to notice when it is wrong — as the only one not reading the accessor.
30
+ *
22
31
  * @returns {object|null} Parsed episode or null on failure
23
32
  */
24
33
  export function readEpisodeRaw() {
25
34
  try {
26
- return JSON.parse(readFileSync(join(RUNTIME_DIR, `ep-${inferProject()}.json`), 'utf8'));
35
+ return JSON.parse(readFileSync(episodeFile(), 'utf8'));
27
36
  } catch {
28
37
  return null;
29
38
  }
package/hook-llm.mjs CHANGED
@@ -924,7 +924,7 @@ export async function handleLLMEpisode() {
924
924
  type: pick by strongest signal. decision = explicit tradeoff / "chose X over Y because Z" / rejected an approach (e.g. "Rejected schema migration — single-source module + sync test instead"; "Heterogeneous hook events → heterogeneous context budgets"). bugfix = prior-failing path fixed with a named root cause. feature = new user-visible capability. refactor = behavior unchanged but structure improved. discovery = learned how a system works (read-heavy, no writes). change = routine edit with no new principle (default if unsure and nothing else fits).
925
925
  Facts: each MUST be (1) atomic—one claim, (2) self-contained—no pronouns, include file/function name, (3) specific—"refreshToken() in auth.ts:45 uses 1h TTL" not "handles tokens"
926
926
  importance: Be strict — default to 1. 0=pure browsing with zero learning value. 1=routine file edits, standard changes, normal workflow (MOST episodes). 2=notable ONLY if it reveals something non-obvious: error fix with discovered root cause, architectural decision with explicit tradeoff, config change with unexpected side effects. 3=critical: breaking change affecting users, security vulnerability fix, data migration. Ask yourself: "would a future session benefit from knowing this?" — if not, it's importance=1.
927
- lesson_learned: The non-obvious insight a future session would benefit from. Examples: "FTS5 porter stemmer doesn't tokenize CJK — need bigram workaround", "vitest --reporter=verbose hangs on large test suites, use default reporter". Look hard before giving up — most coding episodes contain at least one micro-lesson (an undocumented flag, a surprising default, a debugging shortcut, an unexpected interaction). If literally no insight worth teaching (e.g. version bump, whitespace fix, file rename), output JSON null. Do NOT invent a lesson, do NOT write the strings "none"/"n/a"/"todo"/"tbd"/"-" — those will be discarded as noise.
927
+ lesson_learned: The non-obvious insight a future session would benefit from. Examples: "FTS5's default tokenizer doesn't split CJK — need bigram workaround", "vitest --reporter=verbose hangs on large test suites, use default reporter". Look hard before giving up — most coding episodes contain at least one micro-lesson (an undocumented flag, a surprising default, a debugging shortcut, an unexpected interaction). If literally no insight worth teaching (e.g. version bump, whitespace fix, file rename), output JSON null. Do NOT invent a lesson, do NOT write the strings "none"/"n/a"/"todo"/"tbd"/"-" — those will be discarded as noise.
928
928
  scope: ${SCOPE_PROMPT_LEGEND}
929
929
  search_aliases: 2-6 alternative search terms someone might use to find this memory later (include CJK if project uses Chinese)`;
930
930
 
package/hook.mjs CHANGED
@@ -288,8 +288,13 @@ for (const sig of ['SIGTERM', 'SIGINT']) {
288
288
  if (db) {
289
289
  try {
290
290
  for (const sub of planEpisodeFlush(ep)) saveEpisodeImmediate(sub, db);
291
+ // episodeFile(), not a second spelling of `ep-<project>.json`: this is a
292
+ // DESTRUCTIVE step on the buffer the salvage just persisted, and a path that
293
+ // drifts from the accessor deletes nothing while reporting success — the next
294
+ // fire then salvages the same entries again. Every other flush path
295
+ // (PostToolUse, Stop, SessionStart) already unlinks through the accessor.
291
296
  try {
292
- unlinkSync(join(RUNTIME_DIR, `ep-${inferProject()}.json`));
297
+ unlinkSync(episodeFile());
293
298
  } catch {}
294
299
  } finally {
295
300
  try {
package/install.mjs CHANGED
@@ -55,6 +55,7 @@ const HOOK_PATH = join(INSTALL_DIR, 'hook.mjs');
55
55
  // P2-7: both constants and the predicate come from lib/plugin-key.mjs, which hook.mjs also
56
56
  // imports — this pair used to be typed out in each.
57
57
  import { MARKETPLACE_KEY, PLUGIN_KEY, PLUGIN_NAME, isPluginExplicitlyDisabled } from './lib/plugin-key.mjs';
58
+ import { doctorDbModeHint } from './lib/doctor-modes.mjs';
58
59
  const NPM_INSTALL_CMD = 'npm install --omit=dev --no-audit --no-fund';
59
60
 
60
61
  import {
@@ -70,7 +71,7 @@ import {
70
71
  nativeBindingRepairHint,
71
72
  isNativeBindingError,
72
73
  } from './lib/binding-probe.mjs';
73
- import { detectInstallShape, probeRuntimeRoots } from './lib/install-shape.mjs';
74
+ import { detectInstallShape, probeRuntimeRoots, hasAnyManagedCode } from './lib/install-shape.mjs';
74
75
  import { probeSchemaCompat, schemaSkewRemedy } from './lib/schema-skew.mjs';
75
76
  import { clearNativeBindingBreakage, readNativeBindingBreakage } from './lib/native-binding-hint.mjs';
76
77
  import { sweepStaleTestFixtures } from './lib/tmp-fixture-sweep.mjs';
@@ -2366,6 +2367,24 @@ async function doctor() {
2366
2367
  // was ever deployed there, so every entry reads as "missing" and this reported
2367
2368
  // `⚠ Managed files: 121 missing` + an issue on a correct install — prescribing
2368
2369
  // a repair against a path that does not exist.
2370
+ // ...and a THIRD state under the same `!shape.managed`: nothing was ever deployed here.
2371
+ // Both checks below otherwise prescribe `repair`, which re-syncs an install from the signed
2372
+ // release and runs from `<INSTALL_DIR>/cli.mjs` — one of the very entry points whose absence
2373
+ // produced the verdict, so on this shape it hands the reader a command that cannot start.
2374
+ // Damaged (some of the managed files survive) and never-deployed (none do) are different
2375
+ // populations with opposite commands, the same conflation the plugin-only branch above fixed
2376
+ // once already. Declared out here because the hook-script check needs it too and a `const`
2377
+ // inside the try below is not in scope there.
2378
+ //
2379
+ // The population is SOURCE_FILES, not the two entry points. Asking about the entry points
2380
+ // alone made this verdict a claim about two files while the message it gates says "none
2381
+ // present" about all of them: an install holding cli.mjs and every lib/ module, with only
2382
+ // server.mjs and hook.mjs gone, was reported as a data directory with no install behind it
2383
+ // — and sent to re-`install` instead of `repair`, which was runnable from the cli.mjs
2384
+ // already there.
2385
+ const noCodeInstall =
2386
+ !shape.managed && !shape.activePluginVersion && !hasAnyManagedCode(INSTALL_DIR, SOURCE_FILES);
2387
+ const installRemedy = `node ${join(PROJECT_DIR, 'install.mjs')} install`;
2369
2388
  try {
2370
2389
  const skipDrift = !shape.managed && !!shape.activePluginVersion;
2371
2390
  const { checkDevDrift } = await import('./lib/doctor-drift.mjs');
@@ -2436,9 +2455,13 @@ async function doctor() {
2436
2455
  // self-updater is `self-update`. Naming the wrong one sent the user to a
2437
2456
  // usage error at the exact moment their install was incomplete.
2438
2457
  issueWarn(
2439
- `Managed files: ${r.missingCount} missing (${parts.join('; ')}) — a copy install resolves ` +
2440
- `imports against the install dir, so these throw at hook time. Fix: claude-mem-lite self-update ` +
2441
- `(or: node ${join(INSTALL_DIR, 'cli.mjs')} repair)`,
2458
+ noCodeInstall
2459
+ ? `Managed files: no claude-mem-lite code is deployed in ${INSTALL_DIR} (${r.missingCount} ` +
2460
+ `file(s) absent, none present) — this is a data directory with no install behind it, not ` +
2461
+ `a damaged one. Fix: ${installRemedy}`
2462
+ : `Managed files: ${r.missingCount} missing (${parts.join('; ')}) — a copy install resolves ` +
2463
+ `imports against the install dir, so these throw at hook time. Fix: claude-mem-lite self-update ` +
2464
+ `(or: node ${join(INSTALL_DIR, 'cli.mjs')} repair)`,
2442
2465
  );
2443
2466
  }
2444
2467
  // Complete copy install: no message — drift is a dev-install concern.
@@ -2462,7 +2485,11 @@ async function doctor() {
2462
2485
  // missing files, and install.mjs is the one entry that cannot survive that —
2463
2486
  // its static imports resolve before its first statement. cli.mjs has no static
2464
2487
  // local imports and catches the failure (D#26). Same route, same command.
2465
- const scriptRemedy = `claude-mem-lite self-update (or: node ${join(INSTALL_DIR, 'cli.mjs')} repair)`;
2488
+ // Never-deployed gets the install command instead, for the reason spelled out at
2489
+ // `noCodeInstall` above: the `repair` route runs from an entry point that is itself absent.
2490
+ const scriptRemedy = noCodeInstall
2491
+ ? installRemedy
2492
+ : `claude-mem-lite self-update (or: node ${join(INSTALL_DIR, 'cli.mjs')} repair)`;
2466
2493
  if (skipScripts) {
2467
2494
  ok('Hook scripts: n/a (plugin-only install — hooks run from the plugin cache)');
2468
2495
  } else if (!h.present) {
@@ -2671,7 +2698,17 @@ async function doctor() {
2671
2698
  ),
2672
2699
  );
2673
2700
  } else {
2674
- console.log(`\n ${buildDoctorSummary(issues, warnings)}\n`);
2701
+ console.log(`\n ${buildDoctorSummary(issues, warnings)}`);
2702
+ // This run checked the INSTALL. The DB-layer modes are a different implementation reached
2703
+ // through the same command name, and nothing else told the user they exist -- a healthy
2704
+ // install with bad retrieval read "All checks passed!" and ended there. Derived from
2705
+ // DOCTOR_DB_MODES so it cannot become a second list to forget. Text only: the exit-code
2706
+ // contract `claude-mem-lite doctor || alert` depends on is untouched.
2707
+ // Phrased as prose, not as `doctor a | b | c`: a line that looks like a command gets
2708
+ // copy-pasted, and `|` is a shell pipe. See doctorDbModeHint()'s note.
2709
+ console.log(
2710
+ ` Deeper checks (database layer): run \`claude-mem-lite doctor\` with ${doctorDbModeHint()}\n`,
2711
+ );
2675
2712
  }
2676
2713
  // Diagnostic-tool exit-code contract: any ✗-level finding must propagate non-zero
2677
2714
  // so CI / wrapper scripts (`claude-mem-lite doctor || alert`) actually trip. Keeps
@@ -67,6 +67,49 @@ export function isFtsCorruptionError(err) {
67
67
  return /SQLITE_CORRUPT_VTAB/i.test(`${err?.code || ''}`);
68
68
  }
69
69
 
70
+ /**
71
+ * What to do about a damaged FTS5 INDEX — the query-time half of isFtsCorruptionError.
72
+ *
73
+ * R10 P3-9 wired the classifier into `ensureDbWithWalRecovery`, which rebuilds and retries.
74
+ * That covers OPEN time. A structure record damaged inside the index opens fine — nothing
75
+ * reads it until the first MATCH — so the fault surfaces at QUERY time, and there the two
76
+ * faces that carry it to a reader (the CLI catch-all, the MCP safeHandler) passed SQLite's
77
+ * own sentence through with no next step. `recent` / `recall` / `browse` / `context` never
78
+ * touch FTS and keep answering, which makes that dead end easy to misread as "search found
79
+ * nothing" rather than "search is broken".
80
+ *
81
+ * ONE STRING FOR BOTH CHANNELS, unlike the file-level family below. The split there exists
82
+ * because that remedy OVERWRITES the database and must not be handed ready-to-run to an
83
+ * agent holding Bash. This one re-derives every index from its own content table: the rows
84
+ * are never read from the index, so a rebuild is lossless and idempotent, and `doctor`
85
+ * already runs it unprompted. Naming "intact" is load-bearing — the reading this line
86
+ * exists to prevent is "my memories are corrupt".
87
+ *
88
+ * SCOPE, measured rather than assumed: a damaged index does not always reach the caller as
89
+ * SQLITE_CORRUPT_VTAB. `observations_fts_data` holds three rows on a one-observation store —
90
+ * id=1 (averages), id=10 (the STRUCTURE record) and one leaf page — and which row the damage
91
+ * lands on decides the code. Over 100 trials each:
92
+ *
93
+ * UPDATE … SET block = randomblob(32) WHERE id > 1 (structure + leaf) 98 VTAB, 2 NOMEM
94
+ * UPDATE … SET block = randomblob(32) WHERE id > 10 (leaf only) 100 VTAB, 0 NOMEM
95
+ * DELETE … WHERE id > 10 (leaf only) 100 VTAB, 0 NOMEM
96
+ *
97
+ * So NOMEM comes from a mangled STRUCTURE record, where SQLite reads a corrupt varint and asks
98
+ * for an absurd allocation — not from leaf damage. (A first draft of this paragraph said
99
+ * "leaf pages", which would send anyone re-measuring `WHERE id > 10` to 0/N and make the
100
+ * boundary below look vacuous. Caught by the pre-ship claims audit.) That case gets no
101
+ * remedy, on purpose:
102
+ * isFtsCorruptionError is what isDbCorruptionError and isDbUnusableError consult, so
103
+ * admitting SQLITE_NOMEM would answer a real out-of-memory with a full FTS rebuild, and
104
+ * would tell a user their index is damaged when it may be their RAM. The code cannot
105
+ * discriminate the two, so this covers the fault it can name.
106
+ * `tests/fts-corruption-query-time-remedy.test.mjs` pins that boundary.
107
+ */
108
+ export const FTS_CORRUPTION_REMEDY =
109
+ 'The FTS5 search index is damaged; the stored observations are intact. ' +
110
+ 'Rebuild it losslessly with `claude-mem-lite fts-check rebuild` ' +
111
+ '(`claude-mem-lite doctor` rebuilds it too, and re-checks everything else).';
112
+
70
113
  /**
71
114
  * True when `err` means "this file exists and SQLite cannot use it as a database".
72
115
  *
@@ -0,0 +1,31 @@
1
+ // lib/doctor-modes.mjs — the DB-layer `doctor` modes, in one place.
2
+ //
3
+ // `doctor` is one command name with two implementations: install.mjs's health check, which
4
+ // owns `--json`, and cli/doctor.mjs's DB-layer modes. cli.mjs decides between them by looking
5
+ // for one of these flags, and its comment used to say so with a warning attached — "Adding a
6
+ // NEW DB-layer mode requires extending this list — a deliberate trade for a working --json".
7
+ // A mode added to cli/doctor.mjs and not to that list is answered silently by the install
8
+ // check instead, which is a command answering as a different command.
9
+ //
10
+ // Three consumers now read this instead of spelling the list:
11
+ // cli.mjs — the router condition
12
+ // install.mjs — the pointer plain `doctor` prints, so the modes are discoverable at all
13
+ // tests/doctor-mode-router-sync.test.mjs — pins it against what cli/doctor.mjs implements
14
+ //
15
+ // A zero-dependency leaf on purpose. install.mjs is a recovery path and must not import
16
+ // anything that drags a load graph behind it (the lesson lib/data-paths.mjs exists for).
17
+ export const DOCTOR_DB_MODES = ['benchmark', 'metrics', 'session-audit'];
18
+
19
+ /**
20
+ * The modes as PROSE — `--benchmark, --metrics or --session-audit`.
21
+ *
22
+ * Not `a | b | c`. A line shaped like a command invites a copy-paste, and `|` is a shell
23
+ * pipe: pasting the first draft produced `bash: --metrics: command not found`. That is the
24
+ * same defect 32c8923 fixed one commit earlier in this branch (a remedy that named a binary
25
+ * which could not run), reintroduced two commits later in a different spelling. The wording
26
+ * around it has to make clear this is a list of options, not a command line.
27
+ */
28
+ export function doctorDbModeHint() {
29
+ const flags = DOCTOR_DB_MODES.map((m) => `--${m}`);
30
+ return `${flags.slice(0, -1).join(', ')} or ${flags[flags.length - 1]}`;
31
+ }