claude-mem-lite 3.74.0 → 3.75.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +122 -4
- package/README.zh-CN.md +18 -10
- package/bash-utils.mjs +49 -18
- package/hook-context.mjs +4 -7
- package/hook-llm.mjs +7 -2
- package/hook-memory.mjs +4 -6
- package/hook.mjs +72 -175
- package/hooks/hooks.json +1 -1
- package/install.mjs +35 -2
- package/lib/citation-tracker.mjs +15 -46
- package/lib/cite-back-hint.mjs +4 -18
- package/lib/doctor-drift.mjs +8 -1
- package/lib/fast-summary.mjs +87 -0
- package/lib/get-core.mjs +35 -0
- package/lib/maintain-core.mjs +95 -1
- package/lib/registry-core.mjs +94 -0
- package/lib/resolve-data-dir.mjs +53 -4
- package/lib/summary-extractor.mjs +2 -8
- package/lib/transcript-scan.mjs +66 -0
- package/mem-cli.mjs +43 -39
- package/nlp.mjs +20 -2
- package/npm-shrinkwrap.json +2 -2
- package/package.json +5 -1
- package/registry-recommend.mjs +30 -4
- package/scoring-sql.mjs +21 -7
- package/scripts/pre-agent-inject.sh +43 -0
- package/scripts/user-prompt-search.js +17 -3
- package/server.mjs +30 -38
- package/source-files.mjs +12 -0
package/lib/doctor-drift.mjs
CHANGED
|
@@ -82,9 +82,16 @@ export function checkDevDrift(installDir, sourceFiles) {
|
|
|
82
82
|
// stale in silence: tests/doctor-hook-script-manifest.test.mjs re-derives it from the
|
|
83
83
|
// `command` strings in hooks/hooks.json and asserts equality, so registering a new hook
|
|
84
84
|
// script without classifying it here goes red.
|
|
85
|
+
//
|
|
86
|
+
// `pre-agent-inject.js` moved OUT of this set on 2026-08-22 (audit P2-5): no command line
|
|
87
|
+
// names it any more — `pre-agent-inject.sh` does, and execs the .js only when the feature
|
|
88
|
+
// is switched on. It is still shipped and still executed, so `checkHookScriptDrift` keeps
|
|
89
|
+
// grading it; it now reports as the module class, whose message ("ERR_MODULE_NOT_FOUND at
|
|
90
|
+
// hook time") is the truer description of what its absence does. Same severity either way,
|
|
91
|
+
// which is why this set is documented as driving the MESSAGE, not the grade.
|
|
85
92
|
export const HOOK_SCRIPT_ENTRY_POINTS = new Set([
|
|
86
93
|
'post-tool-use.sh', 'user-prompt-search.js', 'pre-tool-recall.js',
|
|
87
|
-
'post-tool-recall.js', 'pre-skill-bridge.js', 'pre-agent-inject.
|
|
94
|
+
'post-tool-recall.js', 'pre-skill-bridge.js', 'pre-agent-inject.sh',
|
|
88
95
|
'hook-launcher.mjs',
|
|
89
96
|
]);
|
|
90
97
|
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
// The non-LLM session summary — one shape, three callers.
|
|
2
|
+
//
|
|
3
|
+
// Audit 2026-08-22 P2-9. hook.mjs carried three hand-copied versions of "read this
|
|
4
|
+
// session's first prompt and its last few observation titles, scrub them, insert a
|
|
5
|
+
// session_summaries row": the Stop-time fast path, the SessionStart previous-session
|
|
6
|
+
// path, and the SessionStart /exit-restart fallback. They had already drifted — the
|
|
7
|
+
// 13-column INSERT was retyped each time, and the truncation limits split 600/600/400
|
|
8
|
+
// against 300/200. Copy-and-miss on exactly this kind of triplicate is what broke
|
|
9
|
+
// v3.35.2, and the comment in one of these blocks announcing "parity with the other"
|
|
10
|
+
// is the tell that parity was being maintained by hand.
|
|
11
|
+
//
|
|
12
|
+
// NOT collapsed in here: hook-llm.mjs's summary insert. That row is produced by the
|
|
13
|
+
// model and carries two more columns (lessons, key_decisions); it is a different
|
|
14
|
+
// record that happens to share a table, and merging it would mean inventing a shape
|
|
15
|
+
// that fits neither.
|
|
16
|
+
import { scrubRecord } from './scrub-record.mjs';
|
|
17
|
+
import { truncate } from '../format-utils.mjs';
|
|
18
|
+
|
|
19
|
+
/** Column list + placeholder row, written once. */
|
|
20
|
+
const INSERT_SQL = `
|
|
21
|
+
INSERT INTO session_summaries
|
|
22
|
+
(memory_session_id, project, request, investigated, learned, completed, next_steps,
|
|
23
|
+
remaining_items, files_read, files_edited, notes, created_at, created_at_epoch)
|
|
24
|
+
VALUES (?, ?, ?, '', '', ?, '', ?, '[]', '[]', ?, ?, ?)
|
|
25
|
+
`;
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Per-caller truncation limits. These are NOT unified on purpose: the Stop path stores
|
|
29
|
+
* roughly twice what the two SessionStart paths do, and every one of these strings is
|
|
30
|
+
* re-injected into a later session's context. Collapsing them to one number changes how
|
|
31
|
+
* much text the product injects, which is a measurable behaviour change and not
|
|
32
|
+
* something a refactor gets to decide. Passed explicitly so the difference is visible
|
|
33
|
+
* at the call site instead of living in three retyped `truncate(...)` arguments.
|
|
34
|
+
*/
|
|
35
|
+
export const FAST_SUMMARY_LIMITS = {
|
|
36
|
+
stop: { request: 200, completed: 600, remaining: 600, notes: 400 },
|
|
37
|
+
sessionStart: { request: 200, completed: 300, remaining: 200, notes: 400 },
|
|
38
|
+
exitRestart: { request: 200, completed: 300, remaining: 200, notes: 400 },
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* The two reads every fast summary is built from: the session's opening prompt, and the
|
|
43
|
+
* titles of its most recent observations.
|
|
44
|
+
* @returns {{request: string, completed: string}} raw (unscrubbed, untruncated) values
|
|
45
|
+
*/
|
|
46
|
+
export function readFastSummarySource(db, sessionId) {
|
|
47
|
+
const firstPrompt = db.prepare(`
|
|
48
|
+
SELECT prompt_text FROM user_prompts
|
|
49
|
+
WHERE content_session_id = ?
|
|
50
|
+
ORDER BY prompt_number ASC LIMIT 1
|
|
51
|
+
`).get(sessionId);
|
|
52
|
+
const recentObs = db.prepare(`
|
|
53
|
+
SELECT title FROM observations
|
|
54
|
+
WHERE memory_session_id = ? AND COALESCE(compressed_into, 0) = 0
|
|
55
|
+
ORDER BY created_at_epoch DESC LIMIT 5
|
|
56
|
+
`).all(sessionId);
|
|
57
|
+
return {
|
|
58
|
+
request: firstPrompt?.prompt_text || '',
|
|
59
|
+
completed: recentObs.map((o) => o.title).filter(Boolean).join('; '),
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Scrub, truncate, insert. Raw values go into scrubRecord and truncation happens after,
|
|
65
|
+
* at the bind site — a secret straddling the truncation boundary would otherwise fall
|
|
66
|
+
* below scrubSecrets' length floors and survive into the row (privacy review, v3.x).
|
|
67
|
+
* That ordering is the reason this function exists in one copy.
|
|
68
|
+
*
|
|
69
|
+
* @param {{request?: string, completed?: string, remaining?: string, notes?: string}} values raw
|
|
70
|
+
* @param {{request: number, completed: number, remaining: number, notes: number}} limits
|
|
71
|
+
*/
|
|
72
|
+
export function insertFastSummary(db, { sessionId, project, values, limits, now }) {
|
|
73
|
+
const safe = scrubRecord('session_summaries', {
|
|
74
|
+
request: values.request || '',
|
|
75
|
+
completed: values.completed || '',
|
|
76
|
+
remaining_items: values.remaining || '',
|
|
77
|
+
notes: values.notes || 'fast',
|
|
78
|
+
});
|
|
79
|
+
db.prepare(INSERT_SQL).run(
|
|
80
|
+
sessionId, project,
|
|
81
|
+
truncate(safe.request, limits.request),
|
|
82
|
+
truncate(safe.completed, limits.completed),
|
|
83
|
+
truncate(safe.remaining_items, limits.remaining),
|
|
84
|
+
truncate(safe.notes, limits.notes),
|
|
85
|
+
now.toISOString(), now.getTime(),
|
|
86
|
+
);
|
|
87
|
+
}
|
package/lib/get-core.mjs
CHANGED
|
@@ -15,6 +15,41 @@ export const OBS_FIELDS = ['id', 'type', 'title', 'subtitle', 'narrative', 'text
|
|
|
15
15
|
* 6-field subset made notes/remaining_items/files_* searchable-but-unrenderable. */
|
|
16
16
|
export const SESSION_DETAIL_FIELDS = ['id', 'request', 'investigated', 'learned', 'completed', 'next_steps', 'remaining_items', 'files_read', 'files_edited', 'notes', 'project', 'created_at', 'memory_session_id', 'prompt_number'];
|
|
17
17
|
|
|
18
|
+
/** User-prompt detail render set. The CLI face rendered only prompt_text +
|
|
19
|
+
* content_session_id while MCP rendered prompt_number and created_at too — the same
|
|
20
|
+
* searchable-but-invisible shape SESSION_DETAIL_FIELDS was created to close, reopened on
|
|
21
|
+
* the prompt source (audit 2026-08-22, P2-6). */
|
|
22
|
+
export const PROMPT_DETAIL_FIELDS = ['id', 'prompt_text', 'content_session_id', 'prompt_number', 'created_at'];
|
|
23
|
+
|
|
24
|
+
/** Event detail render set. `body` carries the distilled lesson persistHaikuSummary
|
|
25
|
+
* writes (lesson_learned || narrative). */
|
|
26
|
+
export const EVENT_DETAIL_FIELDS = ['id', 'event_type', 'title', 'body', 'project', 'importance', 'file_paths', 'git_sha', 'created_at'];
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Fetch user-prompt detail rows, oldest-first.
|
|
30
|
+
* No access bump: prompts are not ranked, so reading one is not a usage signal.
|
|
31
|
+
*/
|
|
32
|
+
export function fetchPromptDetail(db, ids) {
|
|
33
|
+
const ph = ids.map(() => '?').join(',');
|
|
34
|
+
return db.prepare(`SELECT * FROM user_prompts WHERE id IN (${ph}) ORDER BY created_at_epoch ASC`).all(...ids);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Fetch event detail rows, oldest-first.
|
|
39
|
+
*
|
|
40
|
+
* The events table has no `created_at` ISO column — only `created_at_epoch`. Both faces
|
|
41
|
+
* used to derive the ISO string themselves; deriving it once here means EVENT_DETAIL_FIELDS
|
|
42
|
+
* can name `created_at` like every other field set and neither face special-cases it.
|
|
43
|
+
*/
|
|
44
|
+
export function fetchEventDetail(db, ids) {
|
|
45
|
+
const ph = ids.map(() => '?').join(',');
|
|
46
|
+
const rows = db.prepare(`SELECT * FROM events WHERE id IN (${ph}) ORDER BY created_at_epoch ASC`).all(...ids);
|
|
47
|
+
return rows.map(r => ({
|
|
48
|
+
...r,
|
|
49
|
+
created_at: r.created_at_epoch ? new Date(r.created_at_epoch).toISOString() : null,
|
|
50
|
+
}));
|
|
51
|
+
}
|
|
52
|
+
|
|
18
53
|
/**
|
|
19
54
|
* Fetch observation detail rows: bump access_count/last_accessed_at (reading a
|
|
20
55
|
* detail IS an access signal — feeds noisePenalty's ratio guard), run the
|
package/lib/maintain-core.mjs
CHANGED
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
// auto-maintain block). `ctx` carries the per-caller knobs:
|
|
13
13
|
// { projectFilter: 'AND project = ?' | '', baseParams: [project?] , staleAge, opCap }
|
|
14
14
|
|
|
15
|
-
import { COMPRESSED_PENDING_PURGE, computeMinHash, estimateJaccardFromMinHash, jaccardSimilarity } from '../utils.mjs';
|
|
15
|
+
import { COMPRESSED_AUTO, COMPRESSED_PENDING_PURGE, computeMinHash, estimateJaccardFromMinHash, jaccardSimilarity } from '../utils.mjs';
|
|
16
16
|
import { rebuildVocabulary, computeVector, _resetVocabCache, vectorsEnabled, vecTextForRow } from '../tfidf.mjs';
|
|
17
17
|
import { DEDUP_JACCARD_THRESHOLD, MINHASH_PRE_THRESHOLD as MINHASH_PRE_THRESHOLD_SRC, FUZZY_DEDUP_THRESHOLD, FUZZY_BODY_THRESHOLD, MINHASH_PREFILTER } from './dedup-constants.mjs';
|
|
18
18
|
import { liveObsFilterSql } from './inject-search-core.mjs';
|
|
@@ -81,6 +81,100 @@ export function selectFuzzyDedupeIds(rows, {
|
|
|
81
81
|
return removeIds;
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
/**
|
|
85
|
+
* Tombstone the ids an auto-dedup pass selected.
|
|
86
|
+
*
|
|
87
|
+
* `AND superseded_at IS NULL` is the load-bearing half. Both dedup channels select live
|
|
88
|
+
* rows and then stamp them, and between those two steps another writer can supersede a
|
|
89
|
+
* row — `save --supersedes=[#A]` writes A.superseded_by = B's NUMERIC id, and the chain
|
|
90
|
+
* citation-tracker's decay hand-off and timeline re-anchoring both follow. An unguarded
|
|
91
|
+
* UPDATE overwrites that numeric chain with the string marker and the chain silently dead-
|
|
92
|
+
* ends. The window is narrow and the failure is not reproducible on demand, which is
|
|
93
|
+
* exactly why it survived seven rounds of this invariant being re-broken.
|
|
94
|
+
*
|
|
95
|
+
* The exact channel grew this guard in v3.63; the fuzzy channel did not, and the audit of
|
|
96
|
+
* 2026-08-22 (P2-1) found the asymmetry still open. Both channels now stamp through this
|
|
97
|
+
* one function, so the guard cannot be present on one and missing on the other.
|
|
98
|
+
*
|
|
99
|
+
* @param {object} db open DB handle
|
|
100
|
+
* @param {number[]} ids observation ids selected for superseding
|
|
101
|
+
* @param {string} marker superseded_by value ('auto-dedup' | 'auto-dedup-fuzzy')
|
|
102
|
+
* @returns {number} rows actually stamped — less than ids.length means the guard held
|
|
103
|
+
*/
|
|
104
|
+
export function stampDedupSuperseded(db, ids, marker) {
|
|
105
|
+
if (!Array.isArray(ids) || ids.length === 0) return 0;
|
|
106
|
+
const ph = ids.map(() => '?').join(',');
|
|
107
|
+
const res = db.prepare(
|
|
108
|
+
`UPDATE observations SET superseded_at = ?, superseded_by = ?
|
|
109
|
+
WHERE id IN (${ph}) AND superseded_at IS NULL`
|
|
110
|
+
).run(Date.now(), marker, ...ids);
|
|
111
|
+
return res.changes;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Mark auto-compressible observations for one project — the two full-table conditional
|
|
116
|
+
* UPDATEs that used to run inside the SessionStart transaction on every single boot
|
|
117
|
+
* (audit 2026-08-22, P2-11). They are maintenance by definition: nothing about a new
|
|
118
|
+
* session makes a 30-day-old row newly compressible, and the same 24h auto-maintain gate
|
|
119
|
+
* that guards decay/purge/backup was already sitting right next to them.
|
|
120
|
+
*
|
|
121
|
+
* Both predicates are unchanged, including the project scope — this moves *when* the
|
|
122
|
+
* marking runs, not *what* it marks.
|
|
123
|
+
*
|
|
124
|
+
* @param {object} db open DB handle
|
|
125
|
+
* @param {string} project project scope (unchanged from the SessionStart behavior)
|
|
126
|
+
* @param {object} [opts]
|
|
127
|
+
* @param {number} [opts.agedAgeMs] age floor for the general 30d pass
|
|
128
|
+
* @param {number} [opts.noiseAgeMs] age floor for the accelerated 7d LOW_SIGNAL pass
|
|
129
|
+
* @returns {{aged: number, noise: number}} rows marked by each pass
|
|
130
|
+
*/
|
|
131
|
+
export function markAutoCompressible(db, project, {
|
|
132
|
+
agedAgeMs = 30 * DAY_MS,
|
|
133
|
+
noiseAgeMs = 7 * DAY_MS,
|
|
134
|
+
} = {}) {
|
|
135
|
+
if (!project) return { aged: 0, noise: 0 };
|
|
136
|
+
const now = Date.now();
|
|
137
|
+
|
|
138
|
+
// v2.56.0 #4: protect injection_count > 0 obs (proven contextually relevant via
|
|
139
|
+
// hook-memory injection, even if the user never explicitly fetched them).
|
|
140
|
+
// `<= 1` (was `= 1`): citation-decay floors importance at 0 and the LLM low-signal
|
|
141
|
+
// filter saves at imp=0 — those rows are STRICTLY lower value than imp=1 yet escaped
|
|
142
|
+
// GC, accumulating to ~40% of a mature DB (immortal: hidden from injection by the
|
|
143
|
+
// imp>=1 floor, but visible as explicit-search noise).
|
|
144
|
+
// v3.23: never auto-hide a row that carries a real lesson — compression folds sources
|
|
145
|
+
// into a title-only summary (lesson lost) and COMPRESSED_AUTO hides the row from search
|
|
146
|
+
// entirely (audit: 62 lessons buried this way).
|
|
147
|
+
const aged = db.prepare(`
|
|
148
|
+
UPDATE observations SET compressed_into = ${COMPRESSED_AUTO}
|
|
149
|
+
WHERE COALESCE(compressed_into, 0) = 0
|
|
150
|
+
AND COALESCE(importance, 1) <= 1
|
|
151
|
+
AND COALESCE(injection_count, 0) = 0
|
|
152
|
+
AND (lesson_learned IS NULL OR lesson_learned = '' OR lesson_learned = 'none')
|
|
153
|
+
AND created_at_epoch < ?
|
|
154
|
+
AND project = ?
|
|
155
|
+
`).run(now - agedAgeMs, project).changes;
|
|
156
|
+
|
|
157
|
+
// v2.47 P0-3: accelerated pass for LOW_SIGNAL + no-signal noise — 7 days instead of 30.
|
|
158
|
+
// The write-side capNoiseImportance already forces imp=1 on these; this only shrinks GC
|
|
159
|
+
// latency so the corpus reduction materializes within a week instead of bleeding into
|
|
160
|
+
// the 30-day tier.
|
|
161
|
+
const noise = db.prepare(`
|
|
162
|
+
UPDATE observations SET compressed_into = ${COMPRESSED_AUTO}
|
|
163
|
+
WHERE COALESCE(compressed_into, 0) = 0
|
|
164
|
+
AND COALESCE(importance, 1) <= 1
|
|
165
|
+
AND (lesson_learned IS NULL OR lesson_learned = '' OR lesson_learned = 'none')
|
|
166
|
+
AND (facts IS NULL OR facts = '' OR facts = '[]')
|
|
167
|
+
AND (
|
|
168
|
+
title LIKE 'Modified %' OR title LIKE 'Worked on %'
|
|
169
|
+
OR title LIKE 'Reviewed %' OR title LIKE 'Error%'
|
|
170
|
+
)
|
|
171
|
+
AND created_at_epoch < ?
|
|
172
|
+
AND project = ?
|
|
173
|
+
`).run(now - noiseAgeMs, project).changes;
|
|
174
|
+
|
|
175
|
+
return { aged, noise };
|
|
176
|
+
}
|
|
177
|
+
|
|
84
178
|
/** Delete broken observations (no title AND no narrative). Returns rows deleted. */
|
|
85
179
|
// Before hard-deleting observations, un-hide any rows merged INTO them. A child has
|
|
86
180
|
// compressed_into = <keeperId>; deleting that keeper (compressed_into has no FK) would
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
// Shared body for the registry write actions — import / remove / reindex.
|
|
2
|
+
//
|
|
3
|
+
// These were the last un-collapsed CLI/MCP twin (audit 2026-08-22 P1-3): mem-cli.mjs and
|
|
4
|
+
// server.mjs each wrote their own SQL for the same three actions, and had already drifted —
|
|
5
|
+
// the CLI granted a user-initiated import `quality_tier = 'installed'` and the MCP twin did
|
|
6
|
+
// not, so the same intent produced differently-ranked rows depending on which surface the
|
|
7
|
+
// user reached for. That is this project's first-listed病类 ("a guard wired into one face,
|
|
8
|
+
// missing on the other"), so the fix is a single body both faces call, not a second patch.
|
|
9
|
+
//
|
|
10
|
+
// Contract: these functions decide and write. They return structured results and never
|
|
11
|
+
// format output or touch process state — rendering (and CLI-only concerns like bare-flag
|
|
12
|
+
// rejection or the "add --capability-summary" tip) stays in the surface.
|
|
13
|
+
|
|
14
|
+
import { upsertResource } from '../registry.mjs';
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* String columns an import may set, in canonical snake_case. Single source: the CLI derives
|
|
18
|
+
* its kebab-case flag names from this list, the MCP tool reads its args by these keys, so a
|
|
19
|
+
* new column is added once and both surfaces pick it up.
|
|
20
|
+
*/
|
|
21
|
+
export const IMPORT_STRING_FIELDS = [
|
|
22
|
+
'repo_url',
|
|
23
|
+
'local_path',
|
|
24
|
+
'invocation_name',
|
|
25
|
+
'intent_tags',
|
|
26
|
+
'domain_tags',
|
|
27
|
+
'trigger_patterns',
|
|
28
|
+
'capability_summary',
|
|
29
|
+
'keywords',
|
|
30
|
+
'tech_stack',
|
|
31
|
+
'use_cases',
|
|
32
|
+
];
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Resolve the `source` column for an import.
|
|
36
|
+
*
|
|
37
|
+
* Preserve provenance on a metadata-only re-import: default to 'user' only for a genuinely
|
|
38
|
+
* NEW resource. Re-importing an existing github/preinstalled row without an explicit source
|
|
39
|
+
* must not flip it to 'user' (which also mis-grants the user-source rank boost).
|
|
40
|
+
*/
|
|
41
|
+
function resolveSource(db, { type, name, source }) {
|
|
42
|
+
if (source) return source;
|
|
43
|
+
const existing = db.prepare('SELECT source FROM resources WHERE type = ? AND name = ?').get(type, name);
|
|
44
|
+
return existing ? existing.source : 'user';
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Upsert one resource.
|
|
49
|
+
*
|
|
50
|
+
* @param {object} db open resource-registry.db handle
|
|
51
|
+
* @param {object} params
|
|
52
|
+
* @param {string} params.name
|
|
53
|
+
* @param {string} params.type 'skill' | 'agent'
|
|
54
|
+
* @param {string} [params.source] explicit provenance; absent = user-initiated
|
|
55
|
+
* @param {object} [params.fields] IMPORT_STRING_FIELDS values (snake_case keys)
|
|
56
|
+
* @returns {{id: number, source: string, installedTierGranted: boolean}}
|
|
57
|
+
*/
|
|
58
|
+
export function importResource(db, { name, type, source, fields = {} }) {
|
|
59
|
+
const resolvedSource = resolveSource(db, { type, name, source });
|
|
60
|
+
|
|
61
|
+
const row = { name, type, status: 'active', source: resolvedSource };
|
|
62
|
+
for (const f of IMPORT_STRING_FIELDS) row[f] = fields[f] || '';
|
|
63
|
+
|
|
64
|
+
const id = upsertResource(db, row);
|
|
65
|
+
|
|
66
|
+
// A user-initiated import (no explicit --source/source arg) means the user deliberately
|
|
67
|
+
// added this resource — it gets the 'installed' quality tier, which the retriever reads as
|
|
68
|
+
// a ranking bonus and the recommendation gate reads as a precision signal.
|
|
69
|
+
const installedTierGranted = Boolean(id) && !source;
|
|
70
|
+
if (installedTierGranted) {
|
|
71
|
+
db.prepare("UPDATE resources SET quality_tier = 'installed' WHERE id = ?").run(id);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
return { id, source: resolvedSource, installedTierGranted };
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Delete one resource.
|
|
79
|
+
* @returns {{removed: boolean}} removed=false means nothing matched (not an error).
|
|
80
|
+
*/
|
|
81
|
+
export function removeResource(db, { name, type }) {
|
|
82
|
+
const result = db.prepare('DELETE FROM resources WHERE type = ? AND name = ?').run(type, name);
|
|
83
|
+
return { removed: result.changes > 0 };
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Rebuild the FTS5 index over the resources table.
|
|
88
|
+
* @returns {{activeCount: number}}
|
|
89
|
+
*/
|
|
90
|
+
export function reindexResources(db) {
|
|
91
|
+
db.exec("INSERT INTO resources_fts(resources_fts) VALUES('rebuild')");
|
|
92
|
+
const row = db.prepare('SELECT COUNT(*) as c FROM resources WHERE status = ?').get('active');
|
|
93
|
+
return { activeCount: row.c };
|
|
94
|
+
}
|
package/lib/resolve-data-dir.mjs
CHANGED
|
@@ -13,8 +13,57 @@
|
|
|
13
13
|
// - A relative path silently resolves against each process's cwd, scattering
|
|
14
14
|
// state across directories.
|
|
15
15
|
// Falsy (unset/empty) is the only non-absolute value we treat as "use default".
|
|
16
|
-
import { homedir } from 'node:os';
|
|
17
|
-
import { join, isAbsolute } from 'node:path';
|
|
16
|
+
import { homedir, tmpdir } from 'node:os';
|
|
17
|
+
import { join, isAbsolute, resolve } from 'node:path';
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Test-run containment (audit 2026-08-22 P2-4).
|
|
21
|
+
*
|
|
22
|
+
* vitest.config.mjs clears CLAUDE_MEM_DIR for every worker, which stops a relocated
|
|
23
|
+
* dev DB from being read — but it cannot stop a test that never sets the var at all:
|
|
24
|
+
* that test simply resolves the DEFAULT, and the default is the maintainer's real
|
|
25
|
+
* `~/.claude-mem-lite`. It is not hypothetical. During the v3.73.0 release a test
|
|
26
|
+
* wrote a `rateLimited` marker into the live data dir, because the module computed
|
|
27
|
+
* its path at import time and the test's later env stub arrived too late.
|
|
28
|
+
*
|
|
29
|
+
* The guard rides the same channel as the leak: a subprocess that inherits
|
|
30
|
+
* `...process.env` (how e2e tests spawn hooks, and how the var goes missing in the
|
|
31
|
+
* first place) also inherits CLAUDE_MEM_TEST_GUARD, so the redirect follows it there.
|
|
32
|
+
* A subprocess given a curated env has to pass CLAUDE_MEM_DIR explicitly and is safe
|
|
33
|
+
* by construction.
|
|
34
|
+
*
|
|
35
|
+
* It REDIRECTS rather than throws, deliberately. Throwing was tried first and took 181
|
|
36
|
+
* of 289 test files down at collection: schema.mjs resolves the dir at IMPORT time, so
|
|
37
|
+
* every test that imports it — including the hundreds that then use `:memory:` and never
|
|
38
|
+
* touch a file — would have to opt out of a guard they never needed. "Imported the
|
|
39
|
+
* module" is not the failure; "wrote to the real directory" is, and a redirect makes
|
|
40
|
+
* that one impossible while leaving the import harmless.
|
|
41
|
+
*
|
|
42
|
+
* The redirect target is shared per run (CLAUDE_MEM_TEST_SANDBOX, set by
|
|
43
|
+
* tests/global-setup.mjs) so a parent and the subprocess it spawns still agree on one
|
|
44
|
+
* directory — the same reason the ambient env is inherited at all.
|
|
45
|
+
*
|
|
46
|
+
* Escape hatch: CLAUDE_MEM_TEST_GUARD=off, for a test that must exercise real default
|
|
47
|
+
* resolution. It should be rare and it should say why.
|
|
48
|
+
*/
|
|
49
|
+
function containInTests(dir) {
|
|
50
|
+
if (process.env.CLAUDE_MEM_TEST_GUARD !== '1') return dir;
|
|
51
|
+
// Block ONE directory: the real one. "Anywhere outside os.tmpdir()" was tried first
|
|
52
|
+
// and was wrong twice over — fixtures hardcode '/tmp' while os.tmpdir() follows
|
|
53
|
+
// $TMPDIR (relocated under $HOME by a sandboxed shell), and several suites keep their
|
|
54
|
+
// scratch DB inside the repo at tests/.tmp-*. Both are isolated; neither is the leak.
|
|
55
|
+
// The leak is always the same shape: nobody set CLAUDE_MEM_DIR, so the DEFAULT
|
|
56
|
+
// resolved, and the default is the developer's live database.
|
|
57
|
+
//
|
|
58
|
+
// Compared against the path captured by tests/global-setup.mjs BEFORE the suite
|
|
59
|
+
// relocated anything, not against homedir() — several suites deliberately run with
|
|
60
|
+
// HOME pointed at a fixture, and re-deriving the default here would redirect exactly
|
|
61
|
+
// those legitimate cases while missing the real dir once HOME moved.
|
|
62
|
+
const real = process.env.CLAUDE_MEM_TEST_REALDIR || join(homedir(), '.claude-mem-lite');
|
|
63
|
+
if (resolve(dir) !== resolve(real)) return dir;
|
|
64
|
+
const sandbox = process.env.CLAUDE_MEM_TEST_SANDBOX;
|
|
65
|
+
return sandbox && isAbsolute(sandbox) ? sandbox : join(resolve(tmpdir()), 'claude-mem-test-fallback');
|
|
66
|
+
}
|
|
18
67
|
|
|
19
68
|
/**
|
|
20
69
|
* @param {string|undefined|null} raw Typically process.env.CLAUDE_MEM_DIR.
|
|
@@ -23,7 +72,7 @@ import { join, isAbsolute } from 'node:path';
|
|
|
23
72
|
*/
|
|
24
73
|
export function resolveDataDir(raw) {
|
|
25
74
|
if (raw === undefined || raw === null || raw === '') {
|
|
26
|
-
return join(homedir(), '.claude-mem-lite');
|
|
75
|
+
return containInTests(join(homedir(), '.claude-mem-lite'));
|
|
27
76
|
}
|
|
28
77
|
if (typeof raw !== 'string' || raw === 'undefined' || raw === 'null' || !isAbsolute(raw)) {
|
|
29
78
|
throw new Error(
|
|
@@ -31,5 +80,5 @@ export function resolveDataDir(raw) {
|
|
|
31
80
|
`Leave it unset to use ~/.claude-mem-lite.`
|
|
32
81
|
);
|
|
33
82
|
}
|
|
34
|
-
return raw;
|
|
83
|
+
return containInTests(raw);
|
|
35
84
|
}
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
// with empty remaining_items. This extractor runs synchronously in
|
|
10
10
|
// handleStop and gives a deterministic floor.
|
|
11
11
|
|
|
12
|
-
import {
|
|
12
|
+
import { readTranscriptEntries } from './transcript-scan.mjs';
|
|
13
13
|
|
|
14
14
|
const EN_HEADER = /^[\s●*>-]*(Done|Not\s+done|Failed|Uncertain)\s*[::]\s*/im;
|
|
15
15
|
const ZH_HEADER = /^[\s●*>-]*(剩下的?|剩余|还剩|未完成|下次(?:要做|做|继续)?|待做|未做)\s*[::]?\s*/m;
|
|
@@ -27,14 +27,8 @@ const ZH_KEY_IS_NOTDONE = /剩下|剩余|还剩|未完成|下次|待做|未做/;
|
|
|
27
27
|
* @returns {string|null}
|
|
28
28
|
*/
|
|
29
29
|
export function extractTailAssistantText(transcriptPath) {
|
|
30
|
-
if (!transcriptPath || !existsSync(transcriptPath)) return null;
|
|
31
|
-
let raw;
|
|
32
|
-
try { raw = readFileSync(transcriptPath, 'utf8'); } catch { return null; }
|
|
33
30
|
let last = null;
|
|
34
|
-
for (const
|
|
35
|
-
if (!line.trim()) continue;
|
|
36
|
-
let entry;
|
|
37
|
-
try { entry = JSON.parse(line); } catch { continue; }
|
|
31
|
+
for (const entry of readTranscriptEntries(transcriptPath)) {
|
|
38
32
|
if (entry.type !== 'assistant' || !entry.message) continue;
|
|
39
33
|
const content = entry.message.content;
|
|
40
34
|
if (!Array.isArray(content)) continue;
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// One parse of a Claude Code transcript, shared by everything that scans it.
|
|
2
|
+
//
|
|
3
|
+
// Audit 2026-08-22 P2-8. handleStop asked the same .jsonl the same question eight
|
|
4
|
+
// different ways — tail assistant text, citations, citations again with mainOnly,
|
|
5
|
+
// injected-by-surface, cite-back signals, main-thread text, cite-recall, bugfix shape —
|
|
6
|
+
// and every one of them did its own readFileSync + split('\n') + JSON.parse per line.
|
|
7
|
+
// Measured on a real 5.7MB transcript: ~25ms per pass, so the repeated work is roughly
|
|
8
|
+
// 175ms of a 5s Stop budget, and it scales linearly with a session that runs long.
|
|
9
|
+
// Parsing is nearly all of it; iterating an already-parsed array of 2908 entries is
|
|
10
|
+
// 0.1ms.
|
|
11
|
+
//
|
|
12
|
+
// This is a memo, not a rewrite: each scanner keeps its own per-entry logic exactly as
|
|
13
|
+
// it was, and only stops re-reading and re-parsing the file to get at it.
|
|
14
|
+
import { readFileSync, existsSync, statSync } from 'fs';
|
|
15
|
+
|
|
16
|
+
// Retention cap. Parsed entries cost ~3.45× the file size in heap (measured, same
|
|
17
|
+
// transcript: 5.7MB → 19.5MB). Holding that across a whole Stop is fine for the sessions
|
|
18
|
+
// people actually have; holding it for a 50MB transcript is not, and a hook that gets
|
|
19
|
+
// OOM-killed loses the session's work outright. Above the cap each caller parses on its
|
|
20
|
+
// own exactly as before, so the worst case is today's behaviour rather than a new one.
|
|
21
|
+
export const TRANSCRIPT_CACHE_MAX_BYTES = 24 * 1024 * 1024;
|
|
22
|
+
|
|
23
|
+
let cacheKey = '';
|
|
24
|
+
let cacheEntries = null;
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Parsed transcript records, in file order, unparsable lines dropped.
|
|
28
|
+
*
|
|
29
|
+
* The cache key carries size and mtime: a transcript is append-only and Claude Code is
|
|
30
|
+
* still writing to it while the Stop hook runs, so a later caller in the same process
|
|
31
|
+
* must see the same fresh data it would have read for itself.
|
|
32
|
+
*
|
|
33
|
+
* @param {string|null|undefined} transcriptPath
|
|
34
|
+
* @returns {object[]} entries (empty array when the path is missing or unreadable)
|
|
35
|
+
*/
|
|
36
|
+
export function readTranscriptEntries(transcriptPath) {
|
|
37
|
+
if (!transcriptPath || !existsSync(transcriptPath)) return [];
|
|
38
|
+
let st;
|
|
39
|
+
try { st = statSync(transcriptPath); } catch { return []; }
|
|
40
|
+
const key = `${transcriptPath} ${st.size} ${st.mtimeMs}`;
|
|
41
|
+
if (key === cacheKey && cacheEntries) return cacheEntries;
|
|
42
|
+
|
|
43
|
+
let raw;
|
|
44
|
+
try { raw = readFileSync(transcriptPath, 'utf8'); } catch { return []; }
|
|
45
|
+
const entries = [];
|
|
46
|
+
for (const line of raw.split('\n')) {
|
|
47
|
+
if (!line.trim()) continue;
|
|
48
|
+
try { entries.push(JSON.parse(line)); } catch { /* a partially written tail line */ }
|
|
49
|
+
}
|
|
50
|
+
if (st.size <= TRANSCRIPT_CACHE_MAX_BYTES) {
|
|
51
|
+
cacheKey = key;
|
|
52
|
+
cacheEntries = entries;
|
|
53
|
+
} else {
|
|
54
|
+
// Drop whatever was held: an oversized transcript should not keep an older, smaller
|
|
55
|
+
// one alive in memory for the rest of the process either.
|
|
56
|
+
cacheKey = '';
|
|
57
|
+
cacheEntries = null;
|
|
58
|
+
}
|
|
59
|
+
return entries;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Test seam: forget the memo so a fixture rewritten within one mtime tick is re-read. */
|
|
63
|
+
export function _resetTranscriptCache() {
|
|
64
|
+
cacheKey = '';
|
|
65
|
+
cacheEntries = null;
|
|
66
|
+
}
|
package/mem-cli.mjs
CHANGED
|
@@ -17,10 +17,11 @@ import { resolveCliProject as cliProject } from './lib/cli-project.mjs';
|
|
|
17
17
|
import { _resetVocabCache, vecTextForRow, vectorsEnabled } from './tfidf.mjs';
|
|
18
18
|
import { reRankWithContext } from './search-scoring.mjs';
|
|
19
19
|
import { searchObservationsHybrid } from './search-engine.mjs';
|
|
20
|
-
import { fetchObsDetail, OBS_FIELDS, SESSION_DETAIL_FIELDS, supersededNotice } from './lib/get-core.mjs';
|
|
20
|
+
import { fetchObsDetail, fetchPromptDetail, fetchEventDetail, OBS_FIELDS, SESSION_DETAIL_FIELDS, PROMPT_DETAIL_FIELDS, EVENT_DETAIL_FIELDS, supersededNotice } from './lib/get-core.mjs';
|
|
21
21
|
import { collectBrowseTiers, getActiveMemorySessionId, BROWSE_TIERS, BROWSE_TIER_LABELS } from './lib/browse-core.mjs';
|
|
22
22
|
import { deepSearch, resolveDeepMode, shouldEscalateToDeep, autoDeepLlmReady } from './deep-search.mjs';
|
|
23
|
-
import { ensureRegistryDb,
|
|
23
|
+
import { ensureRegistryDb, collectRegistryStats, listResourcesRanked, formatRegistryListLine } from './registry.mjs';
|
|
24
|
+
import { IMPORT_STRING_FIELDS, importResource, removeResource, reindexResources } from './lib/registry-core.mjs';
|
|
24
25
|
import { searchResources } from './registry-retriever.mjs';
|
|
25
26
|
import { computeFunnel, formatFunnel, computeSweep, formatSweep, DEFAULT_SWEEP_FLOORS, DEFAULT_SWEEP_MARGINS } from './registry-recommend.mjs';
|
|
26
27
|
import { selectCompressionCandidates, groupByProjectWeek, compressGroup } from './lib/compress-core.mjs';
|
|
@@ -579,34 +580,50 @@ function renderSessionRows(db, ids) {
|
|
|
579
580
|
return { text: parts.join('\n\n'), count: rows.length };
|
|
580
581
|
}
|
|
581
582
|
|
|
583
|
+
// The CLI's established labels for the prompt/event detail faces. Sharing the FIELD SET
|
|
584
|
+
// with MCP (P2-6) must not rename what users already grep for, so the columns that had a
|
|
585
|
+
// label keep it; anything added later falls back to title-case.
|
|
586
|
+
const CLI_DETAIL_LABELS = {
|
|
587
|
+
prompt_text: 'Text',
|
|
588
|
+
content_session_id: 'Session',
|
|
589
|
+
file_paths: 'Files',
|
|
590
|
+
git_sha: 'Git',
|
|
591
|
+
};
|
|
592
|
+
|
|
593
|
+
/** Label a column for the CLI's `Label: value` render style. */
|
|
594
|
+
const cliFieldLabel = (f) =>
|
|
595
|
+
CLI_DETAIL_LABELS[f] || f[0].toUpperCase() + f.slice(1).replace(/_/g, ' ');
|
|
596
|
+
|
|
582
597
|
function renderPromptRows(db, ids) {
|
|
583
|
-
const
|
|
584
|
-
const rows = db.prepare(`SELECT * FROM user_prompts WHERE id IN (${placeholders}) ORDER BY created_at_epoch ASC`).all(...ids);
|
|
598
|
+
const rows = fetchPromptDetail(db, ids);
|
|
585
599
|
if (rows.length === 0) return null;
|
|
586
600
|
const parts = [];
|
|
587
601
|
for (const r of rows) {
|
|
588
602
|
const lines = [`P#${r.id} ${fmtDateShort(r.created_at)}`];
|
|
589
|
-
|
|
590
|
-
|
|
603
|
+
// id and created_at are already in the header (same convention as the session face).
|
|
604
|
+
for (const f of PROMPT_DETAIL_FIELDS) {
|
|
605
|
+
if (f === 'id' || f === 'created_at') continue;
|
|
606
|
+
const val = r[f];
|
|
607
|
+
if (val === null || val === undefined || val === '') continue;
|
|
608
|
+
lines.push(`${cliFieldLabel(f)}: ${val}`);
|
|
609
|
+
}
|
|
591
610
|
parts.push(lines.join('\n'));
|
|
592
611
|
}
|
|
593
612
|
return { text: parts.join('\n\n'), count: rows.length };
|
|
594
613
|
}
|
|
595
614
|
|
|
596
615
|
function renderEventRows(db, ids) {
|
|
597
|
-
const
|
|
598
|
-
const rows = db.prepare(`SELECT * FROM events WHERE id IN (${placeholders}) ORDER BY created_at_epoch ASC`).all(...ids);
|
|
616
|
+
const rows = fetchEventDetail(db, ids); // derives created_at from created_at_epoch
|
|
599
617
|
if (rows.length === 0) return null;
|
|
600
618
|
const parts = [];
|
|
601
619
|
for (const r of rows) {
|
|
602
|
-
|
|
603
|
-
const
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
if (r.git_sha) lines.push(`Git: ${r.git_sha}`);
|
|
620
|
+
const lines = [`E#${r.id} [${r.event_type}] ${r.created_at ? fmtDateShort(r.created_at) : ''}`];
|
|
621
|
+
for (const f of EVENT_DETAIL_FIELDS) {
|
|
622
|
+
if (f === 'id' || f === 'event_type' || f === 'created_at') continue; // in the header
|
|
623
|
+
const val = r[f];
|
|
624
|
+
if (val === null || val === undefined || val === '') continue;
|
|
625
|
+
lines.push(`${cliFieldLabel(f)}: ${val}`);
|
|
626
|
+
}
|
|
610
627
|
parts.push(lines.join('\n'));
|
|
611
628
|
}
|
|
612
629
|
return { text: parts.join('\n\n'), count: rows.length };
|
|
@@ -2339,24 +2356,12 @@ function cmdRegistry(_memDb, args) {
|
|
|
2339
2356
|
fail(`[mem] Invalid --source "${flags.source}". Valid: preinstalled, user, github`);
|
|
2340
2357
|
return;
|
|
2341
2358
|
}
|
|
2342
|
-
//
|
|
2343
|
-
//
|
|
2344
|
-
//
|
|
2345
|
-
|
|
2346
|
-
|
|
2347
|
-
|
|
2348
|
-
source = existing ? existing.source : 'user';
|
|
2349
|
-
}
|
|
2350
|
-
const fields = { name, type: resourceType, status: 'active', source };
|
|
2351
|
-
for (const f of ['repo-url', 'local-path', 'invocation-name', 'intent-tags', 'domain-tags', 'trigger-patterns', 'capability-summary', 'keywords', 'tech-stack', 'use-cases']) {
|
|
2352
|
-
const camel = f.replace(/-([a-z])/g, (_, c) => '_' + c);
|
|
2353
|
-
fields[camel] = flags[f] || '';
|
|
2354
|
-
}
|
|
2355
|
-
const id = upsertResource(rdb, fields);
|
|
2356
|
-
// User-imported resources get 'installed' quality tier (user explicitly chose to add them)
|
|
2357
|
-
if (id && !flags.source) {
|
|
2358
|
-
rdb.prepare("UPDATE resources SET quality_tier = 'installed' WHERE id = ?").run(id);
|
|
2359
|
-
}
|
|
2359
|
+
// Provenance preservation + the 'installed' tier grant live in lib/registry-core.mjs,
|
|
2360
|
+
// shared with the mem_registry MCP twin (audit 2026-08-22 P1-3). CLI flags are
|
|
2361
|
+
// kebab-case; the core's field list is the canonical snake_case source.
|
|
2362
|
+
const fields = {};
|
|
2363
|
+
for (const f of IMPORT_STRING_FIELDS) fields[f] = flags[f.replace(/_/g, '-')] || '';
|
|
2364
|
+
const { id } = importResource(rdb, { name, type: resourceType, source: flags.source, fields });
|
|
2360
2365
|
out(`[mem] Imported: ${resourceType}:${name} (id=${id})`);
|
|
2361
2366
|
if (!flags['capability-summary'] && !flags['use-cases']) {
|
|
2362
2367
|
out('[mem] Tip: Add --capability-summary or --use-cases so the resource appears in searches.');
|
|
@@ -2371,17 +2376,16 @@ function cmdRegistry(_memDb, args) {
|
|
|
2371
2376
|
const name = flags.name;
|
|
2372
2377
|
const resourceType = flags['resource-type'];
|
|
2373
2378
|
if (!name || !resourceType) { fail('[mem] Usage: claude-mem-lite registry remove --name N --resource-type skill|agent'); return; }
|
|
2374
|
-
const
|
|
2375
|
-
out(
|
|
2379
|
+
const { removed } = removeResource(rdb, { name, type: resourceType });
|
|
2380
|
+
out(removed
|
|
2376
2381
|
? `[mem] Removed: ${resourceType}:${name}`
|
|
2377
2382
|
: `[mem] Not found: ${resourceType}:${name}`);
|
|
2378
2383
|
return;
|
|
2379
2384
|
}
|
|
2380
2385
|
|
|
2381
2386
|
if (action === 'reindex') {
|
|
2382
|
-
|
|
2383
|
-
|
|
2384
|
-
out(`[mem] FTS5 reindexed. ${count.c} active resources.`);
|
|
2387
|
+
const { activeCount } = reindexResources(rdb);
|
|
2388
|
+
out(`[mem] FTS5 reindexed. ${activeCount} active resources.`);
|
|
2385
2389
|
return;
|
|
2386
2390
|
}
|
|
2387
2391
|
} finally {
|