claude-mem-lite 3.77.0 → 3.79.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +4 -1
- package/bash-utils.mjs +117 -2
- package/hook.mjs +137 -37
- package/hooks/hooks.json +12 -0
- package/install.mjs +14 -0
- package/lib/citation-tracker.mjs +27 -6
- package/lib/error-recall-core.mjs +402 -0
- package/lib/inject-search-core.mjs +10 -2
- package/lib/relevance-floor.mjs +136 -0
- package/lib/tool-refusal.mjs +117 -0
- package/npm-shrinkwrap.json +2 -2
- package/package.json +4 -1
- package/schema.mjs +16 -1
- package/scripts/user-prompt-search.js +9 -96
- package/source-files.mjs +12 -0
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
// Error-triggered recall — the SELECTION half of the surface.
|
|
2
|
+
//
|
|
3
|
+
// Why this is a shared core and not left inline in hook.mjs (project convention
|
|
4
|
+
// "shared by two or more faces → lib/"): the offline calibration suite
|
|
5
|
+
// (benchmark/error-recall-suite.mjs) must score THE QUERY THIS SURFACE ACTUALLY
|
|
6
|
+
// RUNS. A benchmark that re-types the SQL measures a second program that merely
|
|
7
|
+
// looks like the first — the failure mode recorded in
|
|
8
|
+
// reference_verify_cc_host_behavior_in_bundle ("copy a call, swap one argument,
|
|
9
|
+
// and you have measured something else"). One body, two consumers: the hook
|
|
10
|
+
// injects from it, the suite scores it.
|
|
11
|
+
//
|
|
12
|
+
// hook.mjs keeps what is genuinely its own: project inference, rendering
|
|
13
|
+
// (formatErrorRecallHints), metering, and the stdout envelope.
|
|
14
|
+
|
|
15
|
+
import { planErrorRecall } from '../bash-utils.mjs';
|
|
16
|
+
import { OBS_BM25, notLowSignalTitleClause } from '../scoring-sql.mjs';
|
|
17
|
+
import { liveObsFilterSql, recencyDecaySql } from './inject-search-core.mjs';
|
|
18
|
+
import { corpusFloorScale } from './relevance-floor.mjs';
|
|
19
|
+
|
|
20
|
+
/** Rows injected per fired error-recall. Historically a bare `LIMIT 3`. */
|
|
21
|
+
export const ERROR_RECALL_LIMIT = 3;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Floor default. **0 = OFF**, and that is a measured decision, not an oversight.
|
|
25
|
+
* The calibrated value if you want to enable it is 10.5 (`CLAUDE_MEM_ERROR_RECALL_BM25_MIN`).
|
|
26
|
+
* See errorRecallBm25Floor for the numbers behind both halves of that sentence.
|
|
27
|
+
*/
|
|
28
|
+
export const DEFAULT_ERROR_RECALL_BM25_FLOOR = 0;
|
|
29
|
+
|
|
30
|
+
/** The value calibrated for this face, used when the floor is switched on. */
|
|
31
|
+
export const CALIBRATED_ERROR_RECALL_BM25_FLOOR = 10.5;
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Set-level |bm25| floor for this surface. **Default 0 = off.**
|
|
35
|
+
*
|
|
36
|
+
* WHY IT IS BUILT, CALIBRATED, AND STILL OFF. Short version: measured, the lever is
|
|
37
|
+
* small where it is safe and large where it is unmeasured, so nothing justifies moving
|
|
38
|
+
* it into everyone's default path. Long version, because whoever reaches for this next
|
|
39
|
+
* needs the numbers rather than the conclusion:
|
|
40
|
+
*
|
|
41
|
+
* The DISTRIBUTION suggested a floor, and then stopped suggesting it once the fixture
|
|
42
|
+
* got more honest. Measured over 7 well-served cases (618 rows), |bm25_raw| by class:
|
|
43
|
+
*
|
|
44
|
+
* class n min p25 med p75 max
|
|
45
|
+
* relevant 9 10.93 11.17 20.75 30.27 42.97
|
|
46
|
+
* negative 9 11.28 22.35 24.39 27.28 31.89
|
|
47
|
+
* filler 19 8.33 8.51 9.55 10.13 29.57
|
|
48
|
+
*
|
|
49
|
+
* 10.5 sits in that gap — filler p75 10.13, relevant min 10.93 — which is how the UPS
|
|
50
|
+
* face's OR floor got its 30 in its own 22->41 gap. But adding two NO-GOOD-MATCH cases
|
|
51
|
+
* (real failures the corpus cannot explain, which is the common case in production and
|
|
52
|
+
* which the fixture originally lacked entirely) moves it (9 cases, 620 rows):
|
|
53
|
+
*
|
|
54
|
+
* class n min p25 med p75 max
|
|
55
|
+
* relevant 9 10.93 11.18 20.77 29.99 43.00
|
|
56
|
+
* negative 11 10.59 20.87 24.41 29.25 31.91
|
|
57
|
+
* filler 30 8.16 8.89 9.56 10.99 29.25
|
|
58
|
+
*
|
|
59
|
+
* filler's p75 is now 10.99, ABOVE relevant's min of 10.93: the gap is gone. It was an
|
|
60
|
+
* artifact of a fixture where every case had something good to find. Note too that
|
|
61
|
+
* `relevant` and `negative` overlap almost entirely in both tables — per #8858 a
|
|
62
|
+
* magnitude gate can remove genuinely off-topic rows and cannot tell an explaining row
|
|
63
|
+
* from a merely topical one.
|
|
64
|
+
*
|
|
65
|
+
* The SWEEP says the achievable gain is small, because a floor only affects rows that
|
|
66
|
+
* reach the injection cap and weak rows mostly do not. On the 7-case fixture a PER-ROW
|
|
67
|
+
* floor at 10.5 moved 20 -> 18 injected rows, both removed rows off-topic filler,
|
|
68
|
+
* relevant and hit-rate unchanged: +3.3pp precision. That was the best case for this
|
|
69
|
+
* lever and it is worth 2 rows.
|
|
70
|
+
*
|
|
71
|
+
* The LIVE DATABASE says the cost is neither small nor measured. Pre-release review ran
|
|
72
|
+
* the per-row form over 8 projects x 10 real hard-error shapes on the maintainer's own
|
|
73
|
+
* DB: injected rows 221 -> 112 (-49%), and 25 of 80 firing cases (31%) went to
|
|
74
|
+
* injecting nothing at all. The loss sits entirely in SMALL projects (under ~500
|
|
75
|
+
* observations: -47..-87%; above ~800: -0..-3%), and corpusFloorScale cannot correct it
|
|
76
|
+
* — it normalises over the WHOLE observations table, which is right for FTS5's IDF and
|
|
77
|
+
* deliberately project-blind, whereas what collapses on a small project is how good the
|
|
78
|
+
* best available memory is. That is v3.61.0's failure mode relocated from install scope
|
|
79
|
+
* to project scope.
|
|
80
|
+
*
|
|
81
|
+
* The obvious repair — make the gate SET-LEVEL, matching what UPS actually does (read
|
|
82
|
+
* the top row, drop the whole set on failure) so a case with one strong row keeps its
|
|
83
|
+
* supporting rows — is the form implemented in selectErrorRecall. AND THE FIXTURE SAYS
|
|
84
|
+
* IT IS FREE, WHICH IS ALSO WRONG. On the ruler it is a no-op at 10.5 (26 injected rows
|
|
85
|
+
* at every floor from 0 to 20; it only bites at 25, costing hit-rate 85.7% -> 42.9%).
|
|
86
|
+
* On the live DB, same threshold, 8 projects x 9 shapes, 69 firing cases:
|
|
87
|
+
*
|
|
88
|
+
* base (off) 201 rows
|
|
89
|
+
* set-level 10.5 126 rows (-37%) 27 of 69 cases silenced (39%)
|
|
90
|
+
* per-row 10.5 97 rows (-52%) 26 of 69 cases silenced (38%)
|
|
91
|
+
*
|
|
92
|
+
* The set-level form trims fewer rows in the cases it spares, and silences just as many
|
|
93
|
+
* cases. The fixture cannot see this because ITS hard negatives are constructed to score
|
|
94
|
+
* high — every fixture case, including the no-good-match ones, has a top row above 10.5,
|
|
95
|
+
* while real small projects frequently have nothing above it. That is the same failure
|
|
96
|
+
* as the per-row measurement one level up: a fixture number standing in for a live one.
|
|
97
|
+
*
|
|
98
|
+
* So the per-row form buys +3.3pp on a fixture at a live cost nothing has shown to be
|
|
99
|
+
* noise (citation_surface_log stores counts, not ids, so the |bm25| of the 15 cited rows
|
|
100
|
+
* is unrecoverable), and the set-level form costs a third of the face for a benefit
|
|
101
|
+
* measured nowhere. Neither earns a default. The defect this face actually has is that
|
|
102
|
+
* command words dominate BM25 — semantic, and out of reach of any magnitude gate (D#167).
|
|
103
|
+
*
|
|
104
|
+
* Enable with CLAUDE_MEM_ERROR_RECALL_BM25_MIN=10.5 (see CALIBRATED_… above). Read at
|
|
105
|
+
* call time so tests and a redirected environment both take effect.
|
|
106
|
+
*/
|
|
107
|
+
export function errorRecallBm25Floor() {
|
|
108
|
+
const raw = process.env.CLAUDE_MEM_ERROR_RECALL_BM25_MIN;
|
|
109
|
+
if (raw === undefined || raw === '') return DEFAULT_ERROR_RECALL_BM25_FLOOR;
|
|
110
|
+
const n = Number(raw);
|
|
111
|
+
return Number.isFinite(n) && n >= 0 ? n : DEFAULT_ERROR_RECALL_BM25_FLOOR;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Fixed 14d half-life: for a failure, recency matters more than observation
|
|
116
|
+
* type, so this surface deliberately does NOT use TYPE_DECAY_CASE.
|
|
117
|
+
*/
|
|
118
|
+
const ERROR_RECALL_HALF_LIFE_MS = '1209600000.0';
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Build the error-recall SELECT. Exported so the calibration suite can assert it
|
|
122
|
+
* is scoring the same statement the hook runs, and so a future floor lands in
|
|
123
|
+
* exactly one place.
|
|
124
|
+
*
|
|
125
|
+
* @param {number} limit Row cap; coerced to a safe integer before interpolation.
|
|
126
|
+
* @returns {string} SQL taking NAMED parameters @q (MATCH), @project, @now
|
|
127
|
+
* (decay reference) and @floor (|bm25| minimum). Named rather than positional
|
|
128
|
+
* so the binding cannot silently renumber when the statement is rearranged.
|
|
129
|
+
*/
|
|
130
|
+
/**
|
|
131
|
+
* The row cap, coerced. ONE body, because the SQL builder and the rerank's control flow
|
|
132
|
+
* both need it and a second copy let them disagree: at `limit: 0` the rerank compared
|
|
133
|
+
* against the raw value, short-circuited its fallback and became a filter, while the SQL
|
|
134
|
+
* had already fallen back to 3.
|
|
135
|
+
*
|
|
136
|
+
* Number.isFinite before trunc: `Number('Infinity')` is a number and truncates to
|
|
137
|
+
* Infinity, which interpolates as `LIMIT Infinity` and throws at prepare(). Not an
|
|
138
|
+
* injection (every string form coerces to the default) but a crash where a fallback
|
|
139
|
+
* belongs.
|
|
140
|
+
*/
|
|
141
|
+
function sanitizeErrorRecallLimit(limit) {
|
|
142
|
+
const asNum = Number(limit);
|
|
143
|
+
return Number.isFinite(asNum) && asNum >= 1
|
|
144
|
+
? Math.max(1, Math.trunc(asNum))
|
|
145
|
+
: ERROR_RECALL_LIMIT;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export function errorRecallSql(limit = ERROR_RECALL_LIMIT) {
|
|
149
|
+
const n = sanitizeErrorRecallLimit(limit);
|
|
150
|
+
// No floor in the SQL: the gate is SET-LEVEL and lives in selectErrorRecall. See
|
|
151
|
+
// its docblock for why. This statement is the pre-floor one plus a bm25_raw column.
|
|
152
|
+
return `
|
|
153
|
+
SELECT o.id, o.type, o.title, o.lesson_learned,
|
|
154
|
+
-- Raw match quality, exposed so the set-level floor can read the top row.
|
|
155
|
+
-- Deliberately the UNDECAYED bm25: an old-but-exact row should be able to clear
|
|
156
|
+
-- the floor, and gating on the decayed score would make the floor an age cutoff.
|
|
157
|
+
-- Note the gate reads the RANK-top row (ordered by bm25 x decay below), not the
|
|
158
|
+
-- highest |bm25| in the set — see selectErrorRecall.
|
|
159
|
+
${OBS_BM25} AS bm25_raw
|
|
160
|
+
FROM observations_fts
|
|
161
|
+
JOIN observations o ON observations_fts.rowid = o.id
|
|
162
|
+
WHERE observations_fts MATCH @q AND o.project = @project
|
|
163
|
+
-- Live-row invariant, same as every other model-facing retrieval path
|
|
164
|
+
-- (hook-context obsPool/fallbackObs/keyObs, hook-memory, search-engine,
|
|
165
|
+
-- recent/search/timeline/recall-core, pre-tool-recall, user-prompt-search).
|
|
166
|
+
-- This surface INLINES rows[0].lesson_learned into the model context, so an
|
|
167
|
+
-- unfiltered SELECT handed a retracted lesson to the agent verbatim while its
|
|
168
|
+
-- correction trailed as a bare pointer. compressed_into is filtered too, not
|
|
169
|
+
-- only for symmetry: the block's own footer is a mem_get(ids=...) pointer, and a
|
|
170
|
+
-- COMPRESSED_PENDING_PURGE row is queued for deletion by maintain purge_stale,
|
|
171
|
+
-- so that pointer would resolve to nothing.
|
|
172
|
+
AND ${liveObsFilterSql('o')}
|
|
173
|
+
AND ${notLowSignalTitleClause('o')}
|
|
174
|
+
-- Decay via the shared core (P2-11): the M-1 MAX(0,…) age clamp lives there.
|
|
175
|
+
ORDER BY ${OBS_BM25}
|
|
176
|
+
* ${recencyDecaySql({
|
|
177
|
+
tsExpr: 'o.created_at_epoch',
|
|
178
|
+
halfLifeSql: ERROR_RECALL_HALF_LIFE_MS,
|
|
179
|
+
// NAMED, not positional. The statement binds @q and @project ahead of this
|
|
180
|
+
// expression, and better-sqlite3 forbids mixing the two styles; an earlier
|
|
181
|
+
// revision moved this expression into the SELECT list with a positional `?`
|
|
182
|
+
// and MATCH silently received the project name (FTS5: `no such column`).
|
|
183
|
+
nowParam: '@now',
|
|
184
|
+
})}
|
|
185
|
+
LIMIT ${n}
|
|
186
|
+
`;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Turn planErrorRecall's terms into the OR-query this surface matches on.
|
|
191
|
+
* @returns {string} FTS5 MATCH expression, or '' when there is nothing to run.
|
|
192
|
+
*/
|
|
193
|
+
export function errorRecallFtsQuery(terms) {
|
|
194
|
+
return (terms || []).map((t) => `"${String(t).replace(/"/g, '""')}"`).join(' OR ');
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* The ERROR-FIRST form of the same expression: a row must carry at least one term that
|
|
199
|
+
* came from the FAILURE, while every term — command words included — stays in the
|
|
200
|
+
* expression so bm25 still sums their contribution.
|
|
201
|
+
*
|
|
202
|
+
* The redundant-looking second clause is the point. Dropping command words from the
|
|
203
|
+
* QUERY was measured against the live DB in D#136 and regressed two of five replays:
|
|
204
|
+
* `database` and `vitest` were carrying domain anchoring, not noise. This keeps them
|
|
205
|
+
* scoring while denying them the power to admit a row on their own.
|
|
206
|
+
*
|
|
207
|
+
* Module-private on purpose: selectErrorRecall is the only caller, and the tests assert
|
|
208
|
+
* on the expression it REPORTS (`errorFirstQuery`) rather than importing the builder —
|
|
209
|
+
* a guard through the real path beats one through a side door. Exporting it would add a
|
|
210
|
+
* name to the knip baseline for no consumer (the #9675 precedent).
|
|
211
|
+
*
|
|
212
|
+
* @returns {string|null} null when there is no error term to require.
|
|
213
|
+
*/
|
|
214
|
+
function errorRecallErrorFirstQuery(terms, errWords) {
|
|
215
|
+
if (!errWords || !errWords.length) return null;
|
|
216
|
+
const or = (list) => list.map((t) => `"${String(t).replace(/"/g, '""')}"`).join(' OR ');
|
|
217
|
+
return `(${or(errWords)}) AND (${or(terms)})`;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Rerank kill-switch. Default ON — see selectErrorRecall for the measurement.
|
|
222
|
+
* `CLAUDE_MEM_ERROR_RECALL_RERANK=off` restores the flat-OR ordering exactly.
|
|
223
|
+
*
|
|
224
|
+
* Module-private for the same reason as the builder above; the switch is exercised
|
|
225
|
+
* end-to-end through selectErrorRecall in the test suite, which is where it has to work.
|
|
226
|
+
*/
|
|
227
|
+
function errorRecallRerankEnabled() {
|
|
228
|
+
return String(process.env.CLAUDE_MEM_ERROR_RECALL_RERANK || '').toLowerCase() !== 'off';
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Decide whether error-recall fires, and select its rows.
|
|
233
|
+
*
|
|
234
|
+
* Returns null — meaning DO NOT INJECT — in the two cases the surface already
|
|
235
|
+
* treated as silence: planErrorRecall found no usable error term (see its own
|
|
236
|
+
* docblock for why silence beats querying the command's topic), or the terms
|
|
237
|
+
* produced an empty MATCH expression.
|
|
238
|
+
*
|
|
239
|
+
* @param {object} db Open better-sqlite3 handle.
|
|
240
|
+
* @param {{cmd: string, response: string, project: string, now?: number,
|
|
241
|
+
* limit?: number}} opts
|
|
242
|
+
* @returns {{rows: object[], terms: string[], ftsQuery: string}|null}
|
|
243
|
+
*/
|
|
244
|
+
export function selectErrorRecall(db, {
|
|
245
|
+
cmd, response, project, now = Date.now(), limit = ERROR_RECALL_LIMIT, floor,
|
|
246
|
+
}) {
|
|
247
|
+
const plan = planErrorRecall(cmd, response);
|
|
248
|
+
if (!plan) return null;
|
|
249
|
+
|
|
250
|
+
const ftsQuery = errorRecallFtsQuery(plan.terms);
|
|
251
|
+
if (!ftsQuery) return null;
|
|
252
|
+
|
|
253
|
+
// A null / NaN / '' floor means "not specified", same as omitting it — NOT "no
|
|
254
|
+
// floor". Disabling the gate must take an explicit 0. The env reader above is
|
|
255
|
+
// hardened the same way, and the asymmetry between the two was a review finding.
|
|
256
|
+
// Only a real, finite, non-negative NUMBER counts as specified. Coercing first would
|
|
257
|
+
// make `''` mean 0 — Number('') is 0 and is finite — i.e. an empty string would
|
|
258
|
+
// silently disable the gate. Same posture as the env reader above.
|
|
259
|
+
const base = (typeof floor === 'number' && Number.isFinite(floor) && floor >= 0)
|
|
260
|
+
? floor
|
|
261
|
+
: errorRecallBm25Floor();
|
|
262
|
+
// Scale by corpus size for the same reason UPS does: bm25 carries IDF, so a fixed
|
|
263
|
+
// magnitude means "weak match" on an established index and "small index" on a new
|
|
264
|
+
// one. v3.61.0 shipped an unscaled floor and injected 0/8 on fresh installs.
|
|
265
|
+
const effective = base > 0 ? base * corpusFloorScale(db) : 0;
|
|
266
|
+
|
|
267
|
+
const stmt = db.prepare(errorRecallSql(limit));
|
|
268
|
+
|
|
269
|
+
// ── ERROR-FIRST RERANK (D#167) ──────────────────────────────────────────────
|
|
270
|
+
// The flat OR admits a row on ANY term, so a memory that merely shares the command's
|
|
271
|
+
// vocabulary competes for the three slots on equal footing with one that names the
|
|
272
|
+
// failure. Measured on the live DB over 52 real failing commands (with their real
|
|
273
|
+
// stderr, extracted from 1110 transcripts) x 15 projects, ~715 firing cases. Rows that
|
|
274
|
+
// match NO error term at all, and cases whose TOP-1 row is one of those — the top row
|
|
275
|
+
// being the one whose lesson_learned is inlined verbatim into the model's context:
|
|
276
|
+
//
|
|
277
|
+
// cmd-only rows cmd-only at TOP-1
|
|
278
|
+
// v3.78.0 (flat OR, banner terms) 764/1947 39.2% 302/714 42.3%
|
|
279
|
+
// flat OR + ERROR_NAMER_RE 780/1941 40.2% 301/715 42.1%
|
|
280
|
+
// + error-first rerank (shipped) 434/1941 22.4% 154/715 21.5%
|
|
281
|
+
//
|
|
282
|
+
// Read the middle row before reaching for the term fix alone: naming the failure did
|
|
283
|
+
// NOT reduce command-vocabulary injection, it nudged it up. Better terms are more
|
|
284
|
+
// specific, so they match fewer rows, so the flat OR has MORE room to fill the three
|
|
285
|
+
// slots with whatever shares the command's words. Terms and ranking are two
|
|
286
|
+
// independent defects and only the pair moves this number.
|
|
287
|
+
//
|
|
288
|
+
// The obvious alternative — make the error term MANDATORY and stop there — was
|
|
289
|
+
// measured too: 1499 rows (−22.8%) and 157 of 715 cases (22.0%) injecting nothing,
|
|
290
|
+
// with the loss concentrated in small projects. That is the magnitude floor's failure
|
|
291
|
+
// mode wearing a different hat (see errorRecallBm25Floor above), so it is not what
|
|
292
|
+
// ships. This REORDERS and never removes: rows that only match the command fall to
|
|
293
|
+
// slots 2-3 instead of being deleted, and when NOTHING in the project matches an
|
|
294
|
+
// error term the result is byte-identical to the flat OR. The residual ~21.5% is
|
|
295
|
+
// essentially that set — verified, not assumed: of the cases the mandatory form
|
|
296
|
+
// silences, 0 had a base set containing an error-matching row.
|
|
297
|
+
//
|
|
298
|
+
// Cost is one extra query only when the primary comes up short of the cap. D#136
|
|
299
|
+
// rejected this shape on the grounds that "the primary always filled its LIMIT 3" —
|
|
300
|
+
// true of the five cases it replayed, false at 715, where the primary leaves hundreds
|
|
301
|
+
// of slots for the fallback to fill.
|
|
302
|
+
let rows;
|
|
303
|
+
const errorFirst = errorRecallRerankEnabled()
|
|
304
|
+
? errorRecallErrorFirstQuery(plan.terms, plan.errWords)
|
|
305
|
+
: null;
|
|
306
|
+
const primary = errorFirst ? stmt.all({ q: errorFirst, project, now }) : [];
|
|
307
|
+
// Compare against the SANITIZED cap, not the raw argument. errorRecallSql coerces
|
|
308
|
+
// (finite, >= 1, else the default) before interpolating, so the raw value and the one
|
|
309
|
+
// the SQL used can disagree — and at `limit: 0` the raw comparison short-circuits the
|
|
310
|
+
// fallback, turning the rerank into the filter this face deliberately rejected.
|
|
311
|
+
// Unreachable from the hook (it passes no limit), found by fuzzing in review.
|
|
312
|
+
const cap = sanitizeErrorRecallLimit(limit);
|
|
313
|
+
if (primary.length >= cap) {
|
|
314
|
+
rows = primary;
|
|
315
|
+
} else {
|
|
316
|
+
// Fallback fills the remaining slots from the unchanged flat-OR result, skipping
|
|
317
|
+
// what the primary already returned. `rows` therefore has the same LENGTH as the
|
|
318
|
+
// pre-rerank behaviour in every case, including the case where primary is empty.
|
|
319
|
+
const flat = stmt.all({ q: ftsQuery, project, now });
|
|
320
|
+
rows = [...primary];
|
|
321
|
+
for (const r of flat) {
|
|
322
|
+
if (rows.length >= cap) break;
|
|
323
|
+
// The id check is load-bearing, not defensive: the error-first match set is a
|
|
324
|
+
// SUBSET of the flat one, so every primary row appears again in `flat`. Without
|
|
325
|
+
// this, the top row is injected twice and one of three slots is wasted. Review
|
|
326
|
+
// measured the mutant: ids 1,2,3 -> 1,1,2.
|
|
327
|
+
if (!rows.some((x) => x.id === r.id)) rows.push(r);
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// SET-LEVEL gate, the shape UPS already uses (read the top row; on failure drop the
|
|
332
|
+
// WHOLE set) rather than filtering row by row.
|
|
333
|
+
//
|
|
334
|
+
// Why not per-row: measured in pre-release review against the maintainer's live DB,
|
|
335
|
+
// a per-row floor cut injected rows 221 → 112 (−49%) across 8 projects, and the loss
|
|
336
|
+
// was concentrated entirely in SMALL ones — projects under ~500 observations lost
|
|
337
|
+
// 47–87%, those above ~800 lost 0–3%. corpusFloorScale cannot see that by design: it
|
|
338
|
+
// normalises over the WHOLE observations table, which is correct for FTS5's IDF and
|
|
339
|
+
// is documented as deliberately project-blind. But what collapses on a small project
|
|
340
|
+
// is not IDF, it is how good the best available memory is. So a per-row floor also
|
|
341
|
+
// shortened sets that DID have a good top row — a different and unjustified
|
|
342
|
+
// behaviour from "this failure has nothing worth recalling".
|
|
343
|
+
//
|
|
344
|
+
// The set-level shape says exactly the intended thing and nothing more: if the best
|
|
345
|
+
// match is not about the failure, stay silent (D#136's stance); otherwise inject the
|
|
346
|
+
// set unchanged. Rows 2..n are never judged on their own, so a case with one strong
|
|
347
|
+
// row keeps its supporting rows.
|
|
348
|
+
//
|
|
349
|
+
// `rows[0]` is the RANK-top row (ordered by bm25 x decay), NOT the highest |bm25|.
|
|
350
|
+
// Those differ: with a 14-day half-life the multiplier reaches 2x, so a fresh weaker
|
|
351
|
+
// row can outrank an older stronger one and veto a set the stronger row would have
|
|
352
|
+
// admitted. Measured, this is why the set-level form silences marginally MORE cases
|
|
353
|
+
// than the per-row one (27 vs 26 of 69 on the live DB). Kept as-is rather than
|
|
354
|
+
// switched to max(): UPS gates on `ftsRows[0]` the same way, and one face quietly
|
|
355
|
+
// disagreeing with the other about what "the top hit" means is worse than the
|
|
356
|
+
// occasional veto. Stated here because the comment above used to claim otherwise.
|
|
357
|
+
//
|
|
358
|
+
// With the rerank above, `rows[0]` is the top row of the ERROR-FIRST result whenever
|
|
359
|
+
// one exists. The gate is therefore applied to the row it is actually about — the
|
|
360
|
+
// best row that mentions the failure — instead of to whatever the flat OR floated up.
|
|
361
|
+
//
|
|
362
|
+
// AND IT ALSO CHANGES THE SCALE, which the first version of this comment missed and
|
|
363
|
+
// pre-release review caught. `(errWords) AND (allTerms)` repeats every error term, and
|
|
364
|
+
// FTS5's bm25() sums over phrase instances, so a PRIMARY row's |bm25_raw| is
|
|
365
|
+
// systematically larger than the same row's flat-OR score — measured on one row in an
|
|
366
|
+
// in-memory index, 1.334 -> 1.779. Rows in the same result set can therefore carry two
|
|
367
|
+
// different scales (primary rows error-first, fallback rows flat).
|
|
368
|
+
//
|
|
369
|
+
// What that costs: CALIBRATED_ERROR_RECALL_BM25_FLOOR = 10.5 was derived from the
|
|
370
|
+
// PRE-RERANK distribution — the gap between filler p75 10.99 and relevant min 10.93 in
|
|
371
|
+
// the table above. Re-running `benchmark/error-recall-suite.mjs --scores` with the
|
|
372
|
+
// rerank ON moves filler p75 to 20.27 and relevant min to 21.87, so 10.5 no longer
|
|
373
|
+
// sits in any gap. The drift is toward a LOOSER gate on sets that have a primary and
|
|
374
|
+
// an unchanged one on sets that do not (empty primary keeps the flat scale), which may
|
|
375
|
+
// be an improvement; nothing has measured it.
|
|
376
|
+
//
|
|
377
|
+
// This is documented rather than fixed because the floor's default is 0 — nothing
|
|
378
|
+
// ships gated. Anyone switching it on must re-derive the constant with the rerank in
|
|
379
|
+
// whatever state they intend to run, NOT reuse the table above.
|
|
380
|
+
if (effective > 0 && rows.length && Math.abs(rows[0].bm25_raw) < effective) {
|
|
381
|
+
return {
|
|
382
|
+
rows: [],
|
|
383
|
+
terms: plan.terms,
|
|
384
|
+
ftsQuery,
|
|
385
|
+
// Reported here too: this is the path where knowing which expression ranked the
|
|
386
|
+
// row the gate just rejected matters MOST, and the first version omitted it.
|
|
387
|
+
errorFirstQuery: errorFirst,
|
|
388
|
+
floor: effective,
|
|
389
|
+
suppressed: rows.length,
|
|
390
|
+
};
|
|
391
|
+
}
|
|
392
|
+
return {
|
|
393
|
+
rows,
|
|
394
|
+
terms: plan.terms,
|
|
395
|
+
ftsQuery,
|
|
396
|
+
// The expression that actually ranked the set, so the calibration suite and any
|
|
397
|
+
// future debugging read the query that ran rather than the one that would have.
|
|
398
|
+
errorFirstQuery: errorFirst,
|
|
399
|
+
floor: effective,
|
|
400
|
+
suppressed: 0,
|
|
401
|
+
};
|
|
402
|
+
}
|
|
@@ -61,10 +61,18 @@ export function liveObsFilterSql(alias = 'o') {
|
|
|
61
61
|
* (e.g. 'o.created_at_epoch', or the created/last-accessed MAX search-engine uses)
|
|
62
62
|
* @param {string} [opts.halfLifeSql=TYPE_DECAY_CASE] - SQL expression for the
|
|
63
63
|
* half-life in ms (constant or per-type CASE; TYPE_DECAY_CASE assumes alias 'o')
|
|
64
|
+
* @param {string} [opts.nowParam='?'] - placeholder text for `now`. Defaults to a
|
|
65
|
+
* POSITIONAL `?`, which makes the binding order depend on where this expression
|
|
66
|
+
* lands in the statement — moving it from ORDER BY into a SELECT list silently
|
|
67
|
+
* renumbers every other placeholder (observed 2026-08-24: the MATCH argument
|
|
68
|
+
* received a project name and FTS5 reported `no such column`). A caller that
|
|
69
|
+
* would rather not carry that coupling passes a NAMED placeholder such as
|
|
70
|
+
* '@now' and binds by object instead. better-sqlite3 does not allow mixing the
|
|
71
|
+
* two styles in one statement, so this is per-statement, all or nothing.
|
|
64
72
|
* @returns {string} SQL numeric expression (parenthesized)
|
|
65
73
|
*/
|
|
66
|
-
export function recencyDecaySql({ tsExpr, halfLifeSql = TYPE_DECAY_CASE }) {
|
|
67
|
-
return `(1.0 + EXP(-0.693 * MAX(0,
|
|
74
|
+
export function recencyDecaySql({ tsExpr, halfLifeSql = TYPE_DECAY_CASE, nowParam = '?' }) {
|
|
75
|
+
return `(1.0 + EXP(-0.693 * MAX(0, ${nowParam} - ${tsExpr}) / ${halfLifeSql}))`;
|
|
68
76
|
}
|
|
69
77
|
|
|
70
78
|
/**
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
// Corpus-size normalization for ABSOLUTE relevance floors.
|
|
2
|
+
//
|
|
3
|
+
// Extracted from scripts/user-prompt-search.js (v3.61.0) when a second injection
|
|
4
|
+
// face — error-recall — needed the same ramp. Two faces, one body: the project's
|
|
5
|
+
// "shared by two or more faces → lib/" rule, and specifically the rule that exists
|
|
6
|
+
// because absolute-floor logic re-typed per face is how v3.61.0 shipped a constant
|
|
7
|
+
// gating an IDF-bearing quantity and injected 0/8 on fresh installs.
|
|
8
|
+
//
|
|
9
|
+
// scripts/user-prompt-search.js re-exports corpusFloorScale so its own callers and
|
|
10
|
+
// tests (tests/ups-corpus-floor-scale.test.mjs) keep their import path.
|
|
11
|
+
//
|
|
12
|
+
// ONE DELIBERATE CHANGE ON EXTRACTION: the reference corpus is read at CALL time,
|
|
13
|
+
// not at module load. In production this is indistinguishable — a hook process reads
|
|
14
|
+
// a fixed environment — but it removes a trap for callers: a re-import with a
|
|
15
|
+
// cache-busting query string reloads the FACE, not this module, so a load-time
|
|
16
|
+
// constant here would have frozen at whatever the first import saw.
|
|
17
|
+
|
|
18
|
+
// Default reference corpus. Overridable per the historical UPS env name.
|
|
19
|
+
// Module-private: exported by habit in the first cut, and knip correctly flagged it —
|
|
20
|
+
// this project treats a new unused export as a defect to fix, not a baseline to carry.
|
|
21
|
+
const DEFAULT_FLOOR_REF_CORPUS = 584;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* The reference corpus the absolute floors are calibrated against.
|
|
25
|
+
* @returns {number}
|
|
26
|
+
*/
|
|
27
|
+
function floorRefCorpus() {
|
|
28
|
+
return Number(process.env.CLAUDE_MEM_UPS_FLOOR_REF_CORPUS || DEFAULT_FLOOR_REF_CORPUS);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* FTS5's IDF term for the best case a query can hit: a term appearing in exactly
|
|
33
|
+
* one row (df=1) of an n-row index. SQLite computes
|
|
34
|
+
* `log((n - df + 0.5) / (df + 0.5))`, so this is the ceiling any single-term bm25
|
|
35
|
+
* contribution can reach at corpus size n — the quantity the absolute floors are
|
|
36
|
+
* implicitly denominated in. Clamped at 0: below n=2 the formula goes negative,
|
|
37
|
+
* which as a scale would flip the comparison rather than relax it.
|
|
38
|
+
* @param {number} n Row count.
|
|
39
|
+
* @returns {number} Max attainable IDF at this corpus size, ≥ 0.
|
|
40
|
+
*/
|
|
41
|
+
function maxIdf(n) {
|
|
42
|
+
// n <= 1 makes the numerator non-positive → Math.log returns NaN or -Infinity, and
|
|
43
|
+
// Math.max(0, NaN) is NaN, not 0. Short-circuit instead: a 0- or 1-row index has no
|
|
44
|
+
// term that can discriminate, so the max attainable IDF is 0.
|
|
45
|
+
if (!(n > 1)) return 0;
|
|
46
|
+
return Math.max(0, Math.log((n - 1 + 0.5) / 1.5));
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// ─── Why the ramp exists (v3.61.0 calibration history, moved with the code) ───
|
|
50
|
+
//
|
|
51
|
+
// The floors this scales are ABSOLUTE magnitudes (UPS: TOP_REL_FLOOR /
|
|
52
|
+
// OR_TOP_BM25_FLOOR; error-recall: ERROR_RECALL_BM25_FLOOR), but the quantity they
|
|
53
|
+
// gate is not scale-free: FTS5 bm25 carries an IDF term ≈ ln(N/df), so the SAME hit
|
|
54
|
+
// scores higher on a bigger index. Measured on one fixed query + one fixed target
|
|
55
|
+
// row, padding the corpus with distinct filler (2026-08-13 dogfood):
|
|
56
|
+
//
|
|
57
|
+
// totalObs 10 40 100 300
|
|
58
|
+
// top|bm25| 10.0 18.6 24.2 30.7 ← same row, same query
|
|
59
|
+
//
|
|
60
|
+
// The floors were calibrated at `projects--mem, 584 obs` (CHANGELOG v2.43.x /
|
|
61
|
+
// v2.34.3). Comparing a log-N quantity against that constant therefore does not
|
|
62
|
+
// mean "weak match" on a small index — it means "small index". A brand-new
|
|
63
|
+
// install measured 0/8 injections on a realistic first-day corpus (10 memories,
|
|
64
|
+
// 8 recall questions whose correct target ranked #1 in 4/5 scored cases): every
|
|
65
|
+
// one was dropped by the OR floor at |bm25| 3.8–15.2 < 30. The plugin is inert
|
|
66
|
+
// during exactly the window where a new user decides whether it earns its keep.
|
|
67
|
+
//
|
|
68
|
+
// Fix: scale the floors by the corpus's MAX ATTAINABLE IDF over the reference
|
|
69
|
+
// corpus's, capped at 1.0, so the SIGNAL↔NOISE separation the maintainer measured
|
|
70
|
+
// (signal ≥41, noise ≤22 at N_REF) is preserved proportionally at any N. At
|
|
71
|
+
// N ≥ N_REF the factor is exactly 1.0, so every established install keeps
|
|
72
|
+
// byte-identical behavior; only genuinely-new installs relax.
|
|
73
|
+
//
|
|
74
|
+
// 2026-08-17 e2e round — the ramp shape. The first cut used ln(N+1)/ln(N_REF+1), which has the
|
|
75
|
+
// right asymptotics but the wrong small-N behavior: FTS5's IDF term is
|
|
76
|
+
// `log((N - df + 0.5) / (df + 0.5))`, which is EXACTLY 0 at N=2/df=1 and stays far
|
|
77
|
+
// below ln(N+1) for the whole first-week window. Re-measured end-to-end through the
|
|
78
|
+
// production write path (lib/save-observation.mjs — a raw INSERT skips CJK bigram
|
|
79
|
+
// expansion and understates the corpus, which is how the first cut's ramp table was
|
|
80
|
+
// misread), 1 planted target + topically clustered filler, CJK prose prompt carrying
|
|
81
|
+
// no identifier for the bypass to rescue:
|
|
82
|
+
//
|
|
83
|
+
// N 2 3 4 5 6 10 25 80
|
|
84
|
+
// top|bm25| 0.0 5.1 7.0 9.5 11.4 15.5 22.2 30.3
|
|
85
|
+
// ln ramp floor 5.2 6.5 7.6 8.4 9.2 11.3 15.3 20.7 ← DROP at N≤4
|
|
86
|
+
// idf ramp floor 0.0 2.6 4.3 5.5 6.5 9.3 14.1 20.0 ← admits all
|
|
87
|
+
//
|
|
88
|
+
// The two ramps agree within 8% at N≥30 and within 2% at N≥200, so this re-shape is
|
|
89
|
+
// confined to the window it is meant to fix. Accepted tradeoff: on a ≤2-row corpus the
|
|
90
|
+
// scale is EXACTLY 0 (FTS5's max IDF is 0 there), which disables the set-level floors
|
|
91
|
+
// rather than lowering them, and just above that they are small; so the best lexical match
|
|
92
|
+
// is injected even when it is weak.
|
|
93
|
+
// That is the intended trade — the alternative measured behavior is total silence, and
|
|
94
|
+
// a 4-row corpus has no room to bury signal under noise. For UPS the upstream
|
|
95
|
+
// hasExplicitSignal gate, not these floors, is what suppresses noise prompts.
|
|
96
|
+
//
|
|
97
|
+
// N counts the WHOLE observations table, not the project: FTS5 computes IDF over
|
|
98
|
+
// the entire index and `o.project = ?` is a post-MATCH filter. Verified — a
|
|
99
|
+
// 2-row project on a 302-row install scores 31.5, matching the 300-row global
|
|
100
|
+
// baseline, not the 10-row one. So a new project on an established install is
|
|
101
|
+
// (correctly) unaffected by this ramp.
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Scale factor in [0, 1] for the absolute score floors, by total corpus size.
|
|
105
|
+
*
|
|
106
|
+
* Short-circuits with a bounded probe: if a row exists at offset N_REF-1 the
|
|
107
|
+
* corpus is at or above the reference and the factor is 1.0 — no COUNT scan on
|
|
108
|
+
* the large corpora where the answer is always 1.0 anyway.
|
|
109
|
+
*
|
|
110
|
+
* Note the ceiling: this only ever RELAXES a floor on a small corpus, never
|
|
111
|
+
* tightens one on a large corpus. A face calibrated at or above the reference
|
|
112
|
+
* therefore keeps its measured value everywhere, and a bigger index (higher IDF,
|
|
113
|
+
* higher scores) makes the same floor comparatively more permissive — the safe
|
|
114
|
+
* direction, since the failure this ramp exists to prevent is silence.
|
|
115
|
+
*
|
|
116
|
+
* @param {object} db Open better-sqlite3 handle.
|
|
117
|
+
* @returns {number} Multiplier for an absolute score floor.
|
|
118
|
+
*/
|
|
119
|
+
export function corpusFloorScale(db) {
|
|
120
|
+
const ref = floorRefCorpus();
|
|
121
|
+
if (ref <= 1) return 1;
|
|
122
|
+
try {
|
|
123
|
+
const atRef = db.prepare('SELECT 1 FROM observations LIMIT 1 OFFSET ?').get(ref - 1);
|
|
124
|
+
if (atRef) return 1;
|
|
125
|
+
const { c = 0 } = db.prepare('SELECT count(*) AS c FROM observations').get() || {};
|
|
126
|
+
const refIdf = maxIdf(ref);
|
|
127
|
+
// Degenerate reference (CLAUDE_MEM_UPS_FLOOR_REF_CORPUS set to 2 or 3, where
|
|
128
|
+
// maxIdf is 0 or near it): division would blow up or divide by zero. Treat the
|
|
129
|
+
// floors as fully calibrated, matching the ref <= 1 guard above.
|
|
130
|
+
if (!(refIdf > 0)) return 1;
|
|
131
|
+
return Math.min(1, maxIdf(c) / refIdf);
|
|
132
|
+
} catch {
|
|
133
|
+
// Any probe failure → behave exactly as before the ramp existed.
|
|
134
|
+
return 1;
|
|
135
|
+
}
|
|
136
|
+
}
|