claude-mem-lite 6.18.0 → 6.19.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +14 -0
- package/README.zh-CN.md +11 -0
- package/format-utils.mjs +4 -2
- package/hook-llm.mjs +1 -0
- package/hook-optimize.mjs +27 -2
- package/lib/activity.mjs +2 -0
- package/lib/deferred-work.mjs +24 -0
- package/lib/fast-summary.mjs +3 -2
- package/lib/lesson-bridge.mjs +2 -1
- package/lib/provenance.mjs +33 -0
- package/lib/save-observation.mjs +5 -1
- package/lib/search-core.mjs +19 -0
- package/mem-cli.mjs +9 -6
- package/npm-shrinkwrap.json +2 -2
- package/package.json +2 -1
- package/search-engine.mjs +23 -3
- package/secret-scrub.mjs +101 -19
- package/server.mjs +8 -7
- package/source-files.mjs +2 -0
- package/utils.mjs +93 -4
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
"plugins": [
|
|
10
10
|
{
|
|
11
11
|
"name": "claude-mem-lite",
|
|
12
|
-
"version": "6.
|
|
12
|
+
"version": "6.19.1",
|
|
13
13
|
"source": "./",
|
|
14
14
|
"homepage": "https://github.com/sdsrss/claude-mem-lite",
|
|
15
15
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.19.1",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "sdsrss"
|
package/README.md
CHANGED
|
@@ -237,6 +237,20 @@ rm -rf ~/claude-mem-lite/ # pre-v0.5 unhidden (if not auto-moved)
|
|
|
237
237
|
repos/ # Shallow-cloned source repos
|
|
238
238
|
```
|
|
239
239
|
|
|
240
|
+
## Upgrading to 6.19.0
|
|
241
|
+
|
|
242
|
+
**Search output changes; no switch.** No schema change and no migration, so reverting is
|
|
243
|
+
pinning `claude-mem-lite@6.18.0`.
|
|
244
|
+
|
|
245
|
+
- **`search` / `mem_search` mark machine-written observations with `🤖`**, and `get` /
|
|
246
|
+
`mem_get` name them in the header. Hook-captured, imported and compressed rows carry it;
|
|
247
|
+
explicit saves and events do not. The result line explains the mark whenever a shown row
|
|
248
|
+
has it, and `search --json` adds `auto` to observation rows.
|
|
249
|
+
- **`mem_search` shows a snippet line only when it adds to the title.**
|
|
250
|
+
- Fixes: a lone search match in a new per-project store no longer sorts last; a Bash
|
|
251
|
+
step's stored description keeps the end of its output; the secret scrubber catches values
|
|
252
|
+
behind markdown labels and stays linear-time on crafted input.
|
|
253
|
+
|
|
240
254
|
## Upgrading to 6.18.0
|
|
241
255
|
|
|
242
256
|
**One default changes, with a switch.** No schema change and no migration. Pinning
|
package/README.zh-CN.md
CHANGED
|
@@ -199,6 +199,17 @@ rm -rf ~/claude-mem-lite/ # v0.5 前的非隐藏目录(如未自动迁移)
|
|
|
199
199
|
repos/ # 浅克隆的源代码仓库
|
|
200
200
|
```
|
|
201
201
|
|
|
202
|
+
## 升级到 6.19.0
|
|
203
|
+
|
|
204
|
+
**搜索输出有变化,没有开关。** 没有 schema 变更、不需要迁移,回退只需固定 `claude-mem-lite@6.18.0`。
|
|
205
|
+
|
|
206
|
+
- **`search` / `mem_search` 用 `🤖` 标出机器写入的 observation**,`get` / `mem_get` 在标题行注明。
|
|
207
|
+
hook 采集、导入和压缩生成的行带这个标记;显式保存的和 events 不带。只要显示的行里有带标记的,
|
|
208
|
+
结果行就会解释它的含义;`search --json` 的 observation 行多一个 `auto` 字段。
|
|
209
|
+
- **`mem_search` 只在摘录比标题多出信息时才显示摘录行。**
|
|
210
|
+
- 修复:新的按项目存储里,单条搜索命中不再排在最后;Bash 步骤存下的描述会保留输出的结尾;
|
|
211
|
+
密钥擦除器能识别 markdown 标签后面的值,并且在构造输入下保持线性时间。
|
|
212
|
+
|
|
202
213
|
## 升级到 6.18.0
|
|
203
214
|
|
|
204
215
|
**一处默认行为改变,有开关。** 没有 schema 变更、不需要迁移。固定 `claude-mem-lite@6.17.1`
|
package/format-utils.mjs
CHANGED
|
@@ -92,7 +92,9 @@ export function queryLabel(query) {
|
|
|
92
92
|
|
|
93
93
|
// Two delimiter classes are defanged here:
|
|
94
94
|
// 1. The blocks claude-mem-lite wraps injected context in (claude-mem-context /
|
|
95
|
-
// memory-context / session-handoff
|
|
95
|
+
// memory-context / session-handoff / the handoff's inner session-summary — D#129: a
|
|
96
|
+
// summary carrying `</session-summary><session-summary source="report">` closed the
|
|
97
|
+
// real block and forged its provenance). User-derived text containing one LITERALLY
|
|
96
98
|
// would prematurely open/close the block it lands in, spilling the rest as
|
|
97
99
|
// undelimited context.
|
|
98
100
|
// 2. Harness-authority + tool-call tags the runtime injects (system-reminder /
|
|
@@ -113,7 +115,7 @@ export function queryLabel(query) {
|
|
|
113
115
|
// Reachable by editing files that contain these tokens \u2014 e.g. developing claude-mem-lite
|
|
114
116
|
// itself, where source/observations carry the delimiter names.
|
|
115
117
|
const CONTEXT_DELIMITER_RE =
|
|
116
|
-
/<\/?(?:claude-mem-context|memory-context|session-handoff|system-reminder|task-notification|(?:antml:)?function_calls|(?:antml:)?function_results|(?:antml:)?invoke|(?:antml:)?parameter)(?:\s[^>]*)?>/gi;
|
|
118
|
+
/<\/?(?:claude-mem-context|memory-context|session-handoff|session-summary|system-reminder|task-notification|(?:antml:)?function_calls|(?:antml:)?function_results|(?:antml:)?invoke|(?:antml:)?parameter)(?:\s[^>]*)?>/gi;
|
|
117
119
|
|
|
118
120
|
// Pass cap for the fixpoint loop below. 32 nested layers of a forged delimiter is far past
|
|
119
121
|
// anything prose produces; the cap exists only to bound the ADVERSARIAL cost (an unbounded
|
package/hook-llm.mjs
CHANGED
|
@@ -991,6 +991,7 @@ export async function handleLLMEpisode() {
|
|
|
991
991
|
// events; treating them as a separate role + boundary marker reduces the
|
|
992
992
|
// attack surface for memory poisoning via crafted file content.
|
|
993
993
|
const SHARED_OBS_SCHEMA_TAIL = `${MEMORY_INPUT_GUARD}
|
|
994
|
+
Grounding: state only outcomes the user message shows. An action description may be cut short ("…"); if its result is not visible, say what was run, not how it turned out. Never write that something passed, failed, was verified or confirmed unless that result appears in the user message.
|
|
994
995
|
type: pick by strongest signal. decision = explicit tradeoff / "chose X over Y because Z" / rejected an approach (e.g. "Rejected schema migration — single-source module + sync test instead"; "Heterogeneous hook events → heterogeneous context budgets"). bugfix = prior-failing path fixed with a named root cause. feature = new user-visible capability. refactor = behavior unchanged but structure improved. discovery = learned how a system works (read-heavy, no writes). change = routine edit with no new principle (default if unsure and nothing else fits).
|
|
995
996
|
Facts: each MUST be (1) atomic—one claim, (2) self-contained—no pronouns, include file/function name, (3) specific—"refreshToken() in auth.ts:45 uses 1h TTL" not "handles tokens"
|
|
996
997
|
importance: Be strict — default to 1. 0=pure browsing with zero learning value. 1=routine file edits, standard changes, normal workflow (MOST episodes). 2=notable ONLY if it reveals something non-obvious: error fix with discovered root cause, architectural decision with explicit tradeoff, config change with unexpected side effects. 3=critical: breaking change affecting users, security vulnerability fix, data migration. Ask yourself: "would a future session benefit from knowing this?" — if not, it's importance=1.
|
package/hook-optimize.mjs
CHANGED
|
@@ -32,6 +32,7 @@ import { normalizeScope, SCOPE_PROMPT_LEGEND, insertObservationRow } from './lib
|
|
|
32
32
|
import { liveObsFilterSql } from './lib/inject-search-core.mjs';
|
|
33
33
|
import { resolveRuntimeDir } from './lib/resolve-data-dir.mjs';
|
|
34
34
|
import { MEMORY_INPUT_GUARD } from './lib/memory-input-guard.mjs';
|
|
35
|
+
import { MANUAL_SESSION_ID_PREFIX } from './lib/provenance.mjs';
|
|
35
36
|
|
|
36
37
|
import { DAY_MS } from './lib/time-constants.mjs';
|
|
37
38
|
// P1-14: same resolver as hook-shared.mjs — this was the second module that had never
|
|
@@ -1108,7 +1109,7 @@ export function findMergeCandidates(db, maxClusters = 5, { project } = {}) {
|
|
|
1108
1109
|
-- keeper.search_aliases when it rebuilt the keeper's TF-IDF vector. Phase-2 removed that
|
|
1109
1110
|
-- rebuild, so the column had no reader left and went with it. Do NOT re-add it on the
|
|
1110
1111
|
-- strength of R10 P3-7 -- that finding is moot, not pending. executeMergeCluster reads
|
|
1111
|
-
-- keeper.{id,importance,narrative,concepts,facts} and o.{id,title,type,narrative,
|
|
1112
|
+
-- keeper.{id,project,importance,narrative,concepts,facts} and o.{id,title,type,narrative,
|
|
1112
1113
|
-- importance,access_count,lesson_learned}, and nothing else off these rows.
|
|
1113
1114
|
SELECT id, title, narrative, project, type, access_count, importance, created_at_epoch, minhash_sig, lesson_learned, concepts, facts
|
|
1114
1115
|
FROM observations
|
|
@@ -1311,10 +1312,33 @@ Return ONLY valid JSON:
|
|
|
1311
1312
|
SELECT ${snapColList}, ? FROM observations WHERE id = ?`,
|
|
1312
1313
|
).run(keeper.id, keeper.id);
|
|
1313
1314
|
|
|
1315
|
+
// The keeper now holds model text. Search reads authorship from memory_session_id, so an
|
|
1316
|
+
// explicit save's `manual-` id would mark it as one (D#138); it moves to the compression
|
|
1317
|
+
// writer's id. A machine-written keeper keeps its id, and so does the snapshot above, which
|
|
1318
|
+
// is the original save. The writer's session row is best-effort: sdk_sessions refuses an id
|
|
1319
|
+
// shaped like a uuid, which `compress-<project>` is for a 27-character project shaped
|
|
1320
|
+
// xxxx-xxxx-xxxx-xxxxxxxxxxxx, and that refusal must not fail the merge (v6.19.1 pre-tag F5a).
|
|
1321
|
+
let rewriteSessionId = null;
|
|
1322
|
+
const keeperSession = db
|
|
1323
|
+
.prepare('SELECT memory_session_id FROM observations WHERE id = ?')
|
|
1324
|
+
.get(keeper.id);
|
|
1325
|
+
if (keeperSession?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
|
|
1326
|
+
const compressSessionId = `compress-${keeper.project}`;
|
|
1327
|
+
try {
|
|
1328
|
+
db.prepare(
|
|
1329
|
+
`INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
|
|
1330
|
+
VALUES (?, ?, ?, ?, ?, 'active')`,
|
|
1331
|
+
).run(compressSessionId, compressSessionId, keeper.project, new Date().toISOString(), Date.now());
|
|
1332
|
+
rewriteSessionId = compressSessionId;
|
|
1333
|
+
} catch (e) {
|
|
1334
|
+
debugCatch(e, 'cluster-merge writer session');
|
|
1335
|
+
}
|
|
1336
|
+
}
|
|
1314
1337
|
db.prepare(
|
|
1315
1338
|
`
|
|
1316
1339
|
UPDATE observations SET title=?, narrative=?, concepts=?, facts=?, text=?,
|
|
1317
|
-
importance=?, lesson_learned=?, minhash_sig=?, optimized_at
|
|
1340
|
+
importance=?, lesson_learned=?, minhash_sig=?, optimized_at=?,
|
|
1341
|
+
memory_session_id = COALESCE(?, memory_session_id)
|
|
1318
1342
|
WHERE id = ?
|
|
1319
1343
|
`,
|
|
1320
1344
|
).run(
|
|
@@ -1327,6 +1351,7 @@ Return ONLY valid JSON:
|
|
|
1327
1351
|
safe.lesson_learned,
|
|
1328
1352
|
minhashSig,
|
|
1329
1353
|
Date.now(),
|
|
1354
|
+
rewriteSessionId,
|
|
1330
1355
|
keeper.id,
|
|
1331
1356
|
);
|
|
1332
1357
|
|
package/lib/activity.mjs
CHANGED
|
@@ -199,6 +199,8 @@ export function promoteInsightEvents(
|
|
|
199
199
|
// manual-save contract) so it lands in the high-weight FTS field.
|
|
200
200
|
lesson_learned: ev.body.slice(0, 500),
|
|
201
201
|
now: new Date(ev.created_at_epoch),
|
|
202
|
+
// Event bodies are machine-written (episode summaries), so the row is not an explicit save.
|
|
203
|
+
sessionId: `promote-${ev.project}`,
|
|
202
204
|
});
|
|
203
205
|
mark.run(Date.now(), ev.id);
|
|
204
206
|
return r;
|
package/lib/deferred-work.mjs
CHANGED
|
@@ -86,6 +86,30 @@ export function listOpenWithOrdinal(db, project, limit = 10) {
|
|
|
86
86
|
.all(project, limit);
|
|
87
87
|
}
|
|
88
88
|
|
|
89
|
+
/**
|
|
90
|
+
* One open row's ordinal, numbered as listOpenWithOrdinal numbers it; null when the row is not
|
|
91
|
+
* open. `defer add` / `mem_defer` read it from a 50-row page before, so an item added past 50
|
|
92
|
+
* open printed `(item ?)` (D#123).
|
|
93
|
+
* @param {Database} db
|
|
94
|
+
* @param {string} project
|
|
95
|
+
* @param {number} id
|
|
96
|
+
* @returns {number|null}
|
|
97
|
+
*/
|
|
98
|
+
export function openOrdinalOf(db, project, id) {
|
|
99
|
+
const row = db
|
|
100
|
+
.prepare(
|
|
101
|
+
`
|
|
102
|
+
SELECT ordinal FROM (
|
|
103
|
+
SELECT id, ROW_NUMBER() OVER (ORDER BY priority DESC, created_at_epoch ASC, id ASC) AS ordinal
|
|
104
|
+
FROM deferred_work
|
|
105
|
+
WHERE project = ? AND status = 'open'
|
|
106
|
+
) WHERE id = ?
|
|
107
|
+
`,
|
|
108
|
+
)
|
|
109
|
+
.get(project, id);
|
|
110
|
+
return row ? row.ordinal : null;
|
|
111
|
+
}
|
|
112
|
+
|
|
89
113
|
/**
|
|
90
114
|
* How many open rows `defer list` / `mem_defer_list` left off their page, as a line to
|
|
91
115
|
* print — '' when the page held them all. Without it a page one short of the open set
|
package/lib/fast-summary.mjs
CHANGED
|
@@ -110,8 +110,9 @@ function replyLines(text) {
|
|
|
110
110
|
|
|
111
111
|
/**
|
|
112
112
|
* Whether the scrubber finds a secret in `text` as written OR with inline markup (`, |, *)
|
|
113
|
-
* removed. Markup can hide a secret from the scrubber —
|
|
114
|
-
*
|
|
113
|
+
* removed. Markup can hide a secret from the scrubber — a key split by a backtick or `*`,
|
|
114
|
+
* `gh|p_…` are missed as written and caught stripped (a markup-wrapped LABEL such as
|
|
115
|
+
* `**password**=…` is caught as written since D#128) — and storing the
|
|
115
116
|
* stripped text instead would leave a split key's tail behind. The caller asks it about the
|
|
116
117
|
* exact text it stores, not the reply as written: a list or quote marker between a label and
|
|
117
118
|
* its value hid the pair from a check on the raw reply, and removing the marker for storage
|
package/lib/lesson-bridge.mjs
CHANGED
|
@@ -26,7 +26,8 @@ export function buildBridgePrompt(lesson, hunk) {
|
|
|
26
26
|
|
|
27
27
|
// { ok:true, check } when the bridge produced a usable, applicable check;
|
|
28
28
|
// { ok:false } on N/A / empty / error / timeout. NEVER throws — the caller
|
|
29
|
-
// falls back to the
|
|
29
|
+
// falls back to the arm's plain directive on { ok:false } (VERDICT_DIRECTIVE since 6.17.0,
|
|
30
|
+
// scripts/pre-tool-recall.js ACTIVE_DIRECTIVE).
|
|
30
31
|
export async function bridgeLesson({ lesson, hunk, timeoutMs = 2500, _callLLM = callLLM }) {
|
|
31
32
|
try {
|
|
32
33
|
const raw = await _callLLM(buildBridgePrompt(lesson, hunk), timeoutMs);
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
// Which writer produced an observation, read back from its memory_session_id. saveObservation
|
|
2
|
+
// writes every explicit save under MANUAL_SESSION_ID_PREFIX. The hook capture (`hook-`),
|
|
3
|
+
// transcript import (`import-`), compression summaries and cluster-merge keepers (`compress-`),
|
|
4
|
+
// promoted events (`promote-`) and rows imported from older stores (a bare session uuid) are
|
|
5
|
+
// machine-written.
|
|
6
|
+
//
|
|
7
|
+
// Search and get mark the machine-written side, because explicit saves are the large majority
|
|
8
|
+
// of a typical store and a mark on nearly every line would carry no information. An unknown id
|
|
9
|
+
// renders unmarked, as every row did before the mark existed.
|
|
10
|
+
|
|
11
|
+
export const MANUAL_SESSION_ID_PREFIX = 'manual-';
|
|
12
|
+
|
|
13
|
+
const AUTO_MARK = '🤖';
|
|
14
|
+
const AUTO_TEXT = 'auto-written, not an explicit save';
|
|
15
|
+
|
|
16
|
+
/** @param {string|null|undefined} memorySessionId */
|
|
17
|
+
export function isAutoWritten(memorySessionId) {
|
|
18
|
+
return (
|
|
19
|
+
typeof memorySessionId === 'string' &&
|
|
20
|
+
memorySessionId !== '' &&
|
|
21
|
+
!memorySessionId.startsWith(MANUAL_SESSION_ID_PREFIX)
|
|
22
|
+
);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** Search row tag; `auto` is set on obs rows by attachBodyTokens. */
|
|
26
|
+
export const autoTag = (r) => (r.auto ? ` ${AUTO_MARK}` : '');
|
|
27
|
+
|
|
28
|
+
/** Search result-line legend, shown only when a rendered row carries the tag. */
|
|
29
|
+
export const autoLegend = (rows) => (rows.some((r) => r.auto) ? ` · ${AUTO_MARK} = ${AUTO_TEXT}` : '');
|
|
30
|
+
|
|
31
|
+
/** Get header note for a full observation row. */
|
|
32
|
+
export const autoHeaderNote = (row) =>
|
|
33
|
+
isAutoWritten(row.memory_session_id) ? ` · ${AUTO_MARK} ${AUTO_TEXT}` : '';
|
package/lib/save-observation.mjs
CHANGED
|
@@ -22,6 +22,7 @@ import { liveObsFilterSql } from './inject-search-core.mjs';
|
|
|
22
22
|
// Imported, not injected: `allowStatuses` below is the POLICY this function exists to hold,
|
|
23
23
|
// and a caller free to pass its own resolver could reinstate the one-way gate D#195 closed.
|
|
24
24
|
import { resolveDeferredIds, closeDeferredItems } from './deferred-work.mjs';
|
|
25
|
+
import { MANUAL_SESSION_ID_PREFIX } from './provenance.mjs';
|
|
25
26
|
|
|
26
27
|
const DEDUP_WINDOW_MS = 5 * 60 * 1000;
|
|
27
28
|
const DEDUP_RECENT_LIMIT = 50;
|
|
@@ -143,6 +144,7 @@ export function formatSupersedeSkipped(skipped) {
|
|
|
143
144
|
* @param {string|null} [params.lesson_learned] Caller validates ≤500 chars.
|
|
144
145
|
* @param {boolean} [params.force=false] Skip the near-duplicate window (below).
|
|
145
146
|
* @param {Date} [params.now] Override for tests.
|
|
147
|
+
* @param {string} [params.sessionId] Writer id; default `manual-<project>` (an explicit save).
|
|
146
148
|
* Both result shapes carry `supersededIds` (observations actually tombstoned) and
|
|
147
149
|
* `supersedeSkipped` (requested but NOT tombstoned, each with a `reason`:
|
|
148
150
|
* `malformed-id` | `no-such-observation` | `no-such-event` | `other-project` |
|
|
@@ -197,7 +199,9 @@ export function saveObservation(db, params) {
|
|
|
197
199
|
const safeTitle = scrubSecrets(rawTitle);
|
|
198
200
|
const safeLesson = rawLesson ? scrubSecrets(rawLesson) : null;
|
|
199
201
|
|
|
200
|
-
|
|
202
|
+
// An explicit save by default; a caller storing machine-written text names its own writer
|
|
203
|
+
// (lib/provenance.mjs reads authorship from this id).
|
|
204
|
+
const sessionId = params.sessionId || `${MANUAL_SESSION_ID_PREFIX}${project}`;
|
|
201
205
|
|
|
202
206
|
// Ensure session exists (FK constraint). INSERT OR IGNORE makes this safe
|
|
203
207
|
// under concurrent calls.
|
package/lib/search-core.mjs
CHANGED
|
@@ -409,6 +409,20 @@ const SINGLE_MATCH_BANDS = [
|
|
|
409
409
|
[0, -0.25],
|
|
410
410
|
];
|
|
411
411
|
|
|
412
|
+
// #36: the ratio above assumes the lone hit's magnitude says how well it matched. It does not
|
|
413
|
+
// when FTS5 clamped the IDF: a term in at least half of a table's rows gets idf <= 0, which
|
|
414
|
+
// FTS5 replaces with 1e-6, so the raw bm25 is at most 1e-6 × (k1+1 = 2.2) per phrase whatever
|
|
415
|
+
// the match. A new per-project store reaches it on every query (2 observations, term in 1:
|
|
416
|
+
// idf = ln(1.5/1.5) = 0). Such a lone row is scored the way a multi-row source's best already
|
|
417
|
+
// is (-1): of two clamped rows the better was normalized to -1; a single one was sunk to -0.25.
|
|
418
|
+
// The test reads the RAW bm25 on the obs leg (rawScore): FULL_SCORE's multipliers go down to
|
|
419
|
+
// 0.5 × 0.5 × 0.2 × 0.4 = 0.02, so a demoted row with an informative IDF (0.18) scored 1.35e-4
|
|
420
|
+
// and was lifted from last to first when the final score was tested (v6.19.0 pre-tag review
|
|
421
|
+
// P3-3). Session and event multipliers are >= 1 and prompts carry raw bm25, so their final
|
|
422
|
+
// score is tested. An obs row without rawScore keeps the bands. A near-zero IDF that FTS5 did not
|
|
423
|
+
// clamp (a term in very nearly half the rows) can land below the scale too, and says as little.
|
|
424
|
+
const CLAMPED_IDF_SCALE = 1e-4;
|
|
425
|
+
|
|
412
426
|
/**
|
|
413
427
|
* Normalize each source's BM25 scores to [-1, 0] before cross-source merge.
|
|
414
428
|
* Prevents observations (BM25 can reach -40) from systematically outranking
|
|
@@ -437,6 +451,11 @@ export function normalizeCrossSourceScores(results, sourceKey) {
|
|
|
437
451
|
);
|
|
438
452
|
if (srcResults.length === 0) continue;
|
|
439
453
|
if (srcResults.length === 1) {
|
|
454
|
+
const magnitude = src === 'obs' ? srcResults[0].rawScore : srcResults[0].score;
|
|
455
|
+
if (typeof magnitude === 'number' && Math.abs(magnitude) < CLAMPED_IDF_SCALE) {
|
|
456
|
+
srcResults[0].score = -1;
|
|
457
|
+
continue;
|
|
458
|
+
}
|
|
440
459
|
const ratio = globalMaxAbs > 0 ? Math.abs(srcResults[0].score) / globalMaxAbs : 0;
|
|
441
460
|
srcResults[0].score = SINGLE_MATCH_BANDS.find(([floor]) => ratio >= floor)[1];
|
|
442
461
|
continue;
|
package/mem-cli.mjs
CHANGED
|
@@ -25,6 +25,7 @@ import { resolveProject } from './project-utils.mjs';
|
|
|
25
25
|
import { resolveCliProject as cliProject } from './lib/cli-project.mjs';
|
|
26
26
|
import { reRankWithContext } from './search-scoring.mjs';
|
|
27
27
|
import { searchObservationsHybrid } from './search-engine.mjs';
|
|
28
|
+
import { autoHeaderNote, autoLegend, autoTag } from './lib/provenance.mjs';
|
|
28
29
|
import {
|
|
29
30
|
fetchObsDetail,
|
|
30
31
|
fetchPromptDetail,
|
|
@@ -162,6 +163,7 @@ import { aggregateMetrics, readMetrics } from './lib/metrics.mjs';
|
|
|
162
163
|
import {
|
|
163
164
|
insertDeferred,
|
|
164
165
|
listOpenWithOrdinal,
|
|
166
|
+
openOrdinalOf,
|
|
165
167
|
dropDeferred,
|
|
166
168
|
formatDropReasonHint,
|
|
167
169
|
resolveDeferredIds,
|
|
@@ -565,6 +567,8 @@ async function cmdSearch(db, args, { llm } = {}) {
|
|
|
565
567
|
importance: r.importance ?? null,
|
|
566
568
|
files_modified: r.files_modified || null,
|
|
567
569
|
body_tokens: r.bodyTokens ?? null,
|
|
570
|
+
// Events carry no session id, so only observations can say who wrote them.
|
|
571
|
+
...(r.source === 'obs' ? { auto: r.auto === true } : {}),
|
|
568
572
|
};
|
|
569
573
|
});
|
|
570
574
|
out(
|
|
@@ -588,7 +592,7 @@ async function cmdSearch(db, args, { llm } = {}) {
|
|
|
588
592
|
// Pluralize on total — "Found 1 of 44 result" reads wrong; the population (44) drives
|
|
589
593
|
// grammatical number, not the page slice (1).
|
|
590
594
|
out(
|
|
591
|
-
`[mem] Found ${countLabel} result${total !== 1 ? 's' : ''} for "${queryLabel(query)}"${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}`,
|
|
595
|
+
`[mem] Found ${countLabel} result${total !== 1 ? 's' : ''} for "${queryLabel(query)}"${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}${autoLegend(paged)}`,
|
|
592
596
|
);
|
|
593
597
|
// `~Nt` = est. tokens to fetch this row's full body via mem_get (attachBodyTokens, paired with
|
|
594
598
|
// MCP). Conditional so a row that skipped enrichment renders cleanly, not "~undefinedt".
|
|
@@ -614,7 +618,7 @@ async function cmdSearch(db, args, { llm } = {}) {
|
|
|
614
618
|
} else {
|
|
615
619
|
const date = fmtDateShort(r.created_at);
|
|
616
620
|
const title = truncate(r.title || r.subtitle || '(untitled)', 80);
|
|
617
|
-
out(`#${r.id} ${typeIcon(r.type)} ${date}${timeStr} ${title}${tok(r)}`);
|
|
621
|
+
out(`#${r.id} ${typeIcon(r.type)}${autoTag(r)} ${date}${timeStr} ${title}${tok(r)}`);
|
|
618
622
|
if (r.lesson_learned) {
|
|
619
623
|
out(` -> ${truncate(r.lesson_learned, 80)}`);
|
|
620
624
|
}
|
|
@@ -793,7 +797,7 @@ function renderObsRows(db, ids, requestedFields) {
|
|
|
793
797
|
const fields = requestedFields || OBS_FIELDS;
|
|
794
798
|
const parts = [];
|
|
795
799
|
for (const r of rows) {
|
|
796
|
-
const lines = [`#${r.id} [${r.type}] ${fmtDateShort(r.created_at)}`];
|
|
800
|
+
const lines = [`#${r.id} [${r.type}] ${fmtDateShort(r.created_at)}${autoHeaderNote(r)}`];
|
|
797
801
|
// Retraction first (shared with mem_get via get-core) — see supersededNotice.
|
|
798
802
|
const retracted = supersededNotice(r);
|
|
799
803
|
if (retracted) lines.push(retracted);
|
|
@@ -1445,9 +1449,8 @@ function cmdDeferAdd(db, args) {
|
|
|
1445
1449
|
return;
|
|
1446
1450
|
}
|
|
1447
1451
|
// Compute the freshly-inserted row's ordinal for an immediately-actionable
|
|
1448
|
-
// response ("ok, deferred this as item N")
|
|
1449
|
-
const
|
|
1450
|
-
const ord = open.find((o) => o.id === r.id)?.ordinal ?? '?';
|
|
1452
|
+
// response ("ok, deferred this as item N"), as mem_defer does.
|
|
1453
|
+
const ord = openOrdinalOf(db, project, r.id) ?? '?';
|
|
1451
1454
|
out(`[mem] Deferred as D#${r.id} (item ${ord}) in project "${project}".`);
|
|
1452
1455
|
}
|
|
1453
1456
|
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.19.1",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "claude-mem-lite",
|
|
9
|
-
"version": "6.
|
|
9
|
+
"version": "6.19.1",
|
|
10
10
|
"os": [
|
|
11
11
|
"darwin",
|
|
12
12
|
"linux",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.19.1",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"packageManager": "npm@10.9.2",
|
|
@@ -82,6 +82,7 @@
|
|
|
82
82
|
"lib/stats-quality.mjs",
|
|
83
83
|
"lib/low-signal-patterns.mjs",
|
|
84
84
|
"lib/private-strip.mjs",
|
|
85
|
+
"lib/provenance.mjs",
|
|
85
86
|
"lib/citation-tracker.mjs",
|
|
86
87
|
"lib/edge-attribution.mjs",
|
|
87
88
|
"lib/file-edge-match.mjs",
|
package/search-engine.mjs
CHANGED
|
@@ -21,6 +21,7 @@ import {
|
|
|
21
21
|
import { citeFactorClause } from './scoring-sql.mjs';
|
|
22
22
|
import { extractPRFTerms, expandQueryByConcepts } from './search-scoring.mjs';
|
|
23
23
|
import { liveObsFilterSql, recencyDecaySql } from './lib/inject-search-core.mjs';
|
|
24
|
+
import { isAutoWritten } from './lib/provenance.mjs';
|
|
24
25
|
|
|
25
26
|
// Scoring expressions — full adds project boost + access bonus; simple is for
|
|
26
27
|
// expansion paths where boost would over-amplify already-loose matches.
|
|
@@ -69,7 +70,8 @@ export function buildObsFtsQuery(scoring, { multiplier, withSnippet, withOffset,
|
|
|
69
70
|
SELECT o.id, o.type, o.title, o.subtitle, o.project, o.created_at, o.created_at_epoch, o.importance,
|
|
70
71
|
o.files_modified, o.lesson_learned,
|
|
71
72
|
${withSnippet ? "snippet(observations_fts, 2, '»', '«', '…', 10) as match_snippet," : ''}
|
|
72
|
-
${scoreExpr}${mult} as score
|
|
73
|
+
${scoreExpr}${mult} as score,
|
|
74
|
+
${OBS_BM25} as raw_bm25
|
|
73
75
|
FROM observations_fts
|
|
74
76
|
JOIN observations o ON observations_fts.rowid = o.id
|
|
75
77
|
WHERE observations_fts MATCH ?
|
|
@@ -330,6 +332,19 @@ export function countSearchTotal(
|
|
|
330
332
|
return total;
|
|
331
333
|
}
|
|
332
334
|
|
|
335
|
+
/**
|
|
336
|
+
* True when an FTS excerpt says something the title does not. snippet() wraps matches in »« and
|
|
337
|
+
* cuts with …, and a save's title is the start of its narrative, so an excerpt that is only a
|
|
338
|
+
* piece of the title repeats it once markers and ellipses are stripped.
|
|
339
|
+
* @param {string|null|undefined} snippet
|
|
340
|
+
* @param {string|null|undefined} title
|
|
341
|
+
*/
|
|
342
|
+
export function snippetAddsInfo(snippet, title) {
|
|
343
|
+
if (typeof snippet !== 'string' || snippet.length <= 10) return false;
|
|
344
|
+
const bare = snippet.replace(/[»«]/g, '').replace(/^…|…$/g, '').trim();
|
|
345
|
+
return !(title || '').includes(bare);
|
|
346
|
+
}
|
|
347
|
+
|
|
333
348
|
export function ftsRowToResult(r, { scoreMultiplier, snippet } = {}) {
|
|
334
349
|
return {
|
|
335
350
|
source: 'obs',
|
|
@@ -346,6 +361,9 @@ export function ftsRowToResult(r, { scoreMultiplier, snippet } = {}) {
|
|
|
346
361
|
created_at: r.created_at,
|
|
347
362
|
created_at_epoch: r.created_at_epoch,
|
|
348
363
|
score: scoreMultiplier ? r.score * scoreMultiplier : r.score,
|
|
364
|
+
// bm25 before FULL_SCORE's multipliers, which can shrink it 50x: only this can say whether
|
|
365
|
+
// FTS5 clamped the IDF (normalizeCrossSourceScores, CLAMPED_IDF_SCALE).
|
|
366
|
+
rawScore: r.raw_bm25,
|
|
349
367
|
files_modified: r.files_modified,
|
|
350
368
|
importance: r.importance,
|
|
351
369
|
lesson_learned: r.lesson_learned,
|
|
@@ -362,7 +380,8 @@ export function ftsRowToResult(r, { scoreMultiplier, snippet } = {}) {
|
|
|
362
380
|
// heavy obs fields are batch-fetched by id HERE rather than carried on every result. The source
|
|
363
381
|
// key is read as `source || _source` because the two render paths disagree (#8654): MCP sets
|
|
364
382
|
// `source`+`text`, CLI sets `_source`+`prompt_text`. estimateTokens floors at 1, so a missing row
|
|
365
|
-
// or empty body yields 1 — never 0/NaN.
|
|
383
|
+
// or empty body yields 1 — never 0/NaN. The same by-id read sets `auto` (lib/provenance.mjs) on
|
|
384
|
+
// the rendered page only, so no result producer has to carry memory_session_id.
|
|
366
385
|
export function attachBodyTokens(db, results) {
|
|
367
386
|
if (!Array.isArray(results) || results.length === 0) return results;
|
|
368
387
|
const obsIds = results
|
|
@@ -373,7 +392,7 @@ export function attachBodyTokens(db, results) {
|
|
|
373
392
|
try {
|
|
374
393
|
const ph = obsIds.map(() => '?').join(',');
|
|
375
394
|
const rows = db
|
|
376
|
-
.prepare(`SELECT id, narrative, facts, text FROM observations WHERE id IN (${ph})`)
|
|
395
|
+
.prepare(`SELECT id, narrative, facts, text, memory_session_id FROM observations WHERE id IN (${ph})`)
|
|
377
396
|
.all(...obsIds);
|
|
378
397
|
for (const row of rows) bodyById.set(row.id, row);
|
|
379
398
|
} catch (e) {
|
|
@@ -385,6 +404,7 @@ export function attachBodyTokens(db, results) {
|
|
|
385
404
|
let parts;
|
|
386
405
|
if (src === 'obs') {
|
|
387
406
|
const row = bodyById.get(r.id) || {};
|
|
407
|
+
r.auto = isAutoWritten(row.memory_session_id);
|
|
388
408
|
parts = [r.title, r.subtitle, r.lesson_learned, row.narrative, row.facts, row.text];
|
|
389
409
|
} else if (src === 'session') {
|
|
390
410
|
parts = [r.request, r.completed, r.working_on];
|
package/secret-scrub.mjs
CHANGED
|
@@ -28,9 +28,41 @@ export const SECRET_PATTERNS = [
|
|
|
28
28
|
// keyword. Allowing a leading `_` catches those while the prose lookbehind still
|
|
29
29
|
// excludes "Marker token: …". `secret` added so a bare SECRET=… with a mixed-alnum
|
|
30
30
|
// value is covered (the hex-only assignment pattern below misses non-hex values).
|
|
31
|
+
//
|
|
32
|
+
// MARKDOWN AROUND A LABEL (D#128). `- **Password**: \`<v>\`` put `**` between the noun and
|
|
33
|
+
// its separator, so no label pattern matched and the value was stored. Models write labels
|
|
34
|
+
// that way all the time. So a label may carry up to three markup characters after the noun
|
|
35
|
+
// (`[*_~\`]`), and up to two of `*_~` between the separator and the value
|
|
36
|
+
// (`**Password:** <v>`). Three rules keep this from widening what counts as a value:
|
|
37
|
+
// - no backtick after the separator: `` `password:` <next word> `` names the label and
|
|
38
|
+
// then goes on in prose, and the next word (or a CJK run with no spaces) was scrubbed;
|
|
39
|
+
// - at most two characters there, so the scrubber's own `***` is never taken for markup
|
|
40
|
+
// (`password=*** token=abc` would scrub `token=abc` on the second pass);
|
|
41
|
+
// - the prose check looks through emphasis (`the **password**: …` is prose) but not a
|
|
42
|
+
// backtick, since `` word `token: <v>` `` is code, not prose. A label wrapped in code
|
|
43
|
+
// (`` `GH_TOKEN`: <v> ``) has its own branch, whose prose check looks past the opening
|
|
44
|
+
// backtick. That branch starts with a cheap lookahead: without it the 40-char lookbehind
|
|
45
|
+
// ran at every position, and a 500k-char input went from 407 ms to 3,387 ms.
|
|
46
|
+
// Measured 2026-09-27 over 831,328 unique lines (this repo's tracked text plus local
|
|
47
|
+
// transcripts: prompts, replies, tool output), old and new run back to back: 28 lines
|
|
48
|
+
// differ. All the catches are fixtures quoted from the D#128 audit; the rest are
|
|
49
|
+
// `- **token**:assistant …` and `` `PGPASSWORD`=PG+password ``, which the same text with
|
|
50
|
+
// no markup scrubs too, plus a JSON-escaped `\n` taken as a value. Idempotence failures: 0
|
|
51
|
+
// and 0. A 13,770-case ground-truth fuzz (labels × 8 wrappings × separators × values ×
|
|
52
|
+
// positions): leaks 12,580 → 421, none newly opened. All 421 are `<word> token|secret|bearer:`
|
|
53
|
+
// in prose, left open by design (below).
|
|
54
|
+
//
|
|
55
|
+
// ACCEPTED GAPS (D#131). This scrubber stops ACCIDENTAL persistence; an author who wants a
|
|
56
|
+
// secret stored can always encode it. So shapes only an adversary writes stay open:
|
|
57
|
+
// zero-width characters inside a label, fullwidth letters. Shapes that occur naturally but
|
|
58
|
+
// cannot be told from prose stay open too: `password is <v>` and `<word> token: <v>` (the
|
|
59
|
+
// guard below). A pattern that caught them would also rewrite ordinary sentences, and
|
|
60
|
+
// v3.61.0 already had to undo exactly that. `| password | <v> |` table rows stay open for a
|
|
61
|
+
// different reason: those 831k lines held 1 such row and none with a credential-shaped
|
|
62
|
+
// value, so a table pattern's false-positive rate cannot be measured here.
|
|
31
63
|
// 1a. `=` assignment → ALWAYS scrub (config syntax, never prose):
|
|
32
64
|
[
|
|
33
|
-
/((?:\b|_)(?:password|passwd|passphrase|token|bearer|secret)
|
|
65
|
+
/((?:\b|_)(?:password|passwd|passphrase|token|bearer|secret)(?:[*_~`]{1,3})?\s*=(?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
34
66
|
'$1***',
|
|
35
67
|
],
|
|
36
68
|
// 1b. `:` separator, PASSWORD nouns. Position decides how permissive the value
|
|
@@ -71,16 +103,16 @@ export const SECRET_PATTERNS = [
|
|
|
71
103
|
// Both arms emit `***` (3 chars, under the {6,} floor), so they cannot
|
|
72
104
|
// double-apply.
|
|
73
105
|
[
|
|
74
|
-
/((?<![A-Za-z][ \t])(?:\b|_)(?:password|passwd|passphrase)\s
|
|
106
|
+
/((?:(?<![A-Za-z][ \t][*_~]{0,3})(?:\b|_)(?:password|passwd|passphrase)(?:[*_~]{1,3})?|(?=_?(?:password|passwd|passphrase)`)(?<![A-Za-z][ \t]`[\w-]{0,40})(?<=`[\w-]{0,40})(?:\b|_)(?:password|passwd|passphrase)`)\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
75
107
|
'$1***',
|
|
76
108
|
],
|
|
77
109
|
[
|
|
78
|
-
/((?:\b|_)(?:password|passwd|passphrase)
|
|
110
|
+
/((?:\b|_)(?:password|passwd|passphrase)(?:[*_~`]{1,3})?\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)(?![A-Za-z]{1,15}(?=[`*~]*(?:[\s,;'"}\]]|$)))[^\s,;'"}\]]{6,}/gi,
|
|
79
111
|
'$1***',
|
|
80
112
|
],
|
|
81
113
|
// 1c. `:` separator, prose-ambiguous nouns → keep the lookbehind ("the token: alicebob"):
|
|
82
114
|
[
|
|
83
|
-
/((?<![A-Za-z][ \t])(?:\b|_)(?:token|bearer|secret)\s
|
|
115
|
+
/((?:(?<![A-Za-z][ \t][*_~]{0,3})(?:\b|_)(?:token|bearer|secret)(?:[*_~]{1,3})?|(?=_?(?:token|bearer|secret)`)(?<![A-Za-z][ \t]`[\w-]{0,40})(?<=`[\w-]{0,40})(?:\b|_)(?:token|bearer|secret)`)\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
84
116
|
'$1***',
|
|
85
117
|
],
|
|
86
118
|
// access_token / refresh_token are the canonical OAuth2 field names — they were
|
|
@@ -95,7 +127,7 @@ export const SECRET_PATTERNS = [
|
|
|
95
127
|
// low-FP decision that `topsecret=` / `access_token_count:` are non-credentials
|
|
96
128
|
// (#8283 + utils.test.mjs:1089-1100); bare `pwd` is omitted so `PWD=` (a path) survives.
|
|
97
129
|
[
|
|
98
|
-
/((?:\b|_)(?:api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token|pgpassword|pgpass|mysql_pwd)
|
|
130
|
+
/((?:\b|_)(?:api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token|pgpassword|pgpass|mysql_pwd)(?:[*_~`]{1,3})?\s*[=::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
99
131
|
'$1***',
|
|
100
132
|
],
|
|
101
133
|
// Space-separated credential CLI flag: `--password <value>` (long-form). The KV
|
|
@@ -118,19 +150,28 @@ export const SECRET_PATTERNS = [
|
|
|
118
150
|
// (a) bare credential nouns: `=` always scrubs; `:` keeps the prose lookbehind
|
|
119
151
|
// (mirrors the unquoted 1a/1b split — a quoted value doesn't turn `:` prose
|
|
120
152
|
// into config, but `<word> password="x"` is still a leak):
|
|
121
|
-
[
|
|
122
|
-
|
|
123
|
-
|
|
153
|
+
[
|
|
154
|
+
/((?:\b|_)(?:password|passwd|passphrase|token|bearer|secret)(?:[*_~`]{1,3})?\s*=(?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
155
|
+
'$1$2***$2',
|
|
156
|
+
],
|
|
157
|
+
[
|
|
158
|
+
/((?:\b|_)(?:password|passwd|passphrase)(?:[*_~`]{1,3})?\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
159
|
+
'$1$2***$2',
|
|
160
|
+
],
|
|
161
|
+
[
|
|
162
|
+
/((?:(?<![A-Za-z][ \t][*_~]{0,3})(?:\b|_)(?:token|bearer|secret)(?:[*_~]{1,3})?|(?=_?(?:token|bearer|secret)`)(?<![A-Za-z][ \t]`[\w-]{0,40})(?<=`[\w-]{0,40})(?:\b|_)(?:token|bearer|secret)`)\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
163
|
+
'$1$2***$2',
|
|
164
|
+
],
|
|
124
165
|
// (b) structured keys + named env vars are unambiguous config even after a word
|
|
125
166
|
// (`see api_key: "x"` DOES scrub, mirroring the unquoted structured-key path):
|
|
126
167
|
[
|
|
127
|
-
/((?:\b|_)(?:pgpassword|pgpass|mysql_pwd|api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token)
|
|
168
|
+
/((?:\b|_)(?:pgpassword|pgpass|mysql_pwd|api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token)(?:[*_~`]{1,3})?\s*[=::](?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
128
169
|
'$1$2***$2',
|
|
129
170
|
],
|
|
130
171
|
// AWS access keys: AKIA (long-term) + ASIA (STS temp) + AROA (role) + AIDA
|
|
131
172
|
// (user) + ANPA/ANVA/AGPA (other principal types). All share the 4-letter
|
|
132
173
|
// prefix + exactly 16 base32 chars shape — specific enough for near-zero FP.
|
|
133
|
-
[/\b(?:AKIA|ASIA|AROA|AIDA|ANPA|ANVA|AGPA)[A-Z0-9]{16}
|
|
174
|
+
[/\b(?:AKIA|ASIA|AROA|AIDA|ANPA|ANVA|AGPA)[A-Z0-9]{16}(?![A-Za-z0-9])/g, '***'],
|
|
134
175
|
// OpenAI / Anthropic keys (sk-...) — specific prefixes have lower length threshold
|
|
135
176
|
[/\bsk-(?:proj|ant|ant-api\d{2})-[a-zA-Z0-9_-]{8,}\b/g, '***'],
|
|
136
177
|
[/\bsk-[a-zA-Z0-9_-]{20,}\b/g, '***'],
|
|
@@ -143,15 +184,50 @@ export const SECRET_PATTERNS = [
|
|
|
143
184
|
[/\b(?:xox[bpasr]|xapp|xoxe)-[a-zA-Z0-9-]{10,}\b/g, '***'],
|
|
144
185
|
// Slack incoming-webhook URL — the path after /services/ is the shared secret.
|
|
145
186
|
[/(https:\/\/hooks\.slack\.com\/services\/)[A-Za-z0-9/]+/g, '$1***'],
|
|
146
|
-
// JWT tokens (eyJ...eyJ...)
|
|
147
|
-
|
|
187
|
+
// JWT tokens (eyJ...eyJ...), from any `eyJ`: one glued to a prefix (`my-sess-eyJ…`,
|
|
188
|
+
// `session-<uuid>-eyJ…`, `tok_eyJ…`) is still a JWT. Every `eyJ` of one dotless run used to be a
|
|
189
|
+
// fresh start that rescanned the run to its end — quadratic, 9.6 s on 200k chars of `eyJ-`
|
|
190
|
+
// (D#130) — and v6.19.0's bounded lookbehind that fixed it missed a prefix over 40 characters
|
|
191
|
+
// (round-3 review P3-2). Here a failed start consumes its run instead (`|eyJ[\w-]*`, returned
|
|
192
|
+
// unchanged), so the scan resumes after it. Nothing is lost: the first segment cannot contain a
|
|
193
|
+
// `.`, so every later `eyJ` of the same run reaches the same run end and fails the same way.
|
|
194
|
+
[
|
|
195
|
+
/eyJ[a-zA-Z0-9_-]{10,}\.eyJ[a-zA-Z0-9_-]{10,}\.[a-zA-Z0-9_-]+\b|eyJ[a-zA-Z0-9_-]*/g,
|
|
196
|
+
(m) => (m.includes('.') ? '***' : m),
|
|
197
|
+
],
|
|
148
198
|
// PEM private key blocks. `[A-Z0-9 ]*` covers every armor label — RSA/EC/DSA/
|
|
149
199
|
// OPENSSH plus ENCRYPTED and PGP (… PRIVATE KEY BLOCK) — that the fixed
|
|
150
200
|
// alternation missed; the block delimiters make FP impossible.
|
|
201
|
+
// The body stops at the next `-----BEGIN ` (D#130): with `[\s\S]*?` every header with no END
|
|
202
|
+
// scanned to the end of the text, on each of scrubSecrets' passes — quadratic, 8.3 s on 500k
|
|
203
|
+
// chars. A block whose END is missing is left to the next pattern.
|
|
204
|
+
[
|
|
205
|
+
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
|
|
206
|
+
'***PEM_KEY***',
|
|
207
|
+
],
|
|
208
|
+
// A cut-off key (`head id_rsa`, a tool output cut mid-key): the header and the WHOLE lines of
|
|
209
|
+
// base64 (or RFC 1421 headers) that follow it. Stored whole through v6.19.0 when no later key
|
|
210
|
+
// header followed. Whole lines only, so prose naming a header mid-sentence keeps its text
|
|
211
|
+
// (v6.19.0 round-3 P3-3: ending the block at the next key header erased the prose between two).
|
|
212
|
+
// A body line starts with 16+ base64 characters, and one shorter whole line may follow the last
|
|
213
|
+
// of them (a key's last line): a line of one word or number is prose (v6.19.1 pre-tag review F3),
|
|
214
|
+
// so the body needs a long line, and blank lines count only between long ones. A long line need
|
|
215
|
+
// not be whole, so a key cut mid-line loses the cut line too. A line break may be JSON-escaped
|
|
216
|
+
// (`\n` as two characters) and a quote may end the last line (v6.19.1 claims review F1). A header
|
|
217
|
+
// value may hold a backslash that does not start an escaped break (a Windows path), and it does
|
|
218
|
+
// not share its whitespace with a second quantifier, which was quadratic (delta review P1).
|
|
151
219
|
[
|
|
152
|
-
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----[\
|
|
220
|
+
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:[ \t]*(?:\r?\n|\\r\\n|\\n)(?=[ \t]*(?:\r?\n|\\r\\n|\\n)))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*(?:[A-Za-z0-9+/=]{16,}|(?:Proc-Type|DEK-Info|Version|Comment|Hash|Charset):(?:[^\r\n\\]|\\(?![rn]))*(?=\r?\n|\\[rn]|["']|$)))(?:(?:[ \t]*(?:\r?\n|\\r\\n|\\n)(?=[ \t]*(?:\r?\n|\\r\\n|\\n)))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*(?:[A-Za-z0-9+/=]{16,}|(?:Proc-Type|DEK-Info|Version|Comment|Hash|Charset):(?:[^\r\n\\]|\\(?![rn]))*(?=\r?\n|\\[rn]|["']|$))))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*[A-Za-z0-9+/=]{1,15}[ \t]*(?=\r?\n|\\[rn]|["']|$))?/g,
|
|
153
221
|
'***PEM_KEY***',
|
|
154
222
|
],
|
|
223
|
+
// The other end (`tail key.pem`): whole base64 lines ending in a private-key END, the same line
|
|
224
|
+
// rule as above but with two shorter lines allowed before the END (an armored PGP key's last data
|
|
225
|
+
// line and its `=XXXX` checksum; delta review P2). A line may start after a quote. A run of lines
|
|
226
|
+
// that does not end there is consumed and returned unchanged, so no line starts a second scan.
|
|
227
|
+
[
|
|
228
|
+
/(?:(?<![^\n])|(?<=\\n|["']))(?:[ \t]*[A-Za-z0-9+/=]{16,}[ \t]*(?:\r?\n|\\r\\n|\\n))+(?:[ \t]*[A-Za-z0-9+/=]{1,15}[ \t]*(?:\r?\n|\\r\\n|\\n)){0,2}(?:-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----)?/g,
|
|
229
|
+
(m) => (m.endsWith('-----') ? '***PEM_KEY***' : m),
|
|
230
|
+
],
|
|
155
231
|
// Long hex strings in credential assignments (e.g. SECRET_KEY=abc123def456...).
|
|
156
232
|
// `hash` deliberately excluded: `hash: <40hex>` / `hash=<md5>` are git SHAs and
|
|
157
233
|
// checksums (real, preserved data in this hash-heavy repo), not credentials.
|
|
@@ -160,7 +236,10 @@ export const SECRET_PATTERNS = [
|
|
|
160
236
|
[/\bAIza[A-Za-z0-9_-]{35}\b/g, '***'],
|
|
161
237
|
// Authorization header credentials — Bearer (opaque), Basic (base64 user:pass),
|
|
162
238
|
// and GitHub's `token` scheme all carry secrets after the scheme word.
|
|
163
|
-
[
|
|
239
|
+
[
|
|
240
|
+
/(Authorization(?:[*_~`]{1,3})?[::](?:[*_~]{1,2}(?=\s))?\s*(?:Bearer|Basic|token)\s+)[^\s,;'"}\]]+/gi,
|
|
241
|
+
'$1***',
|
|
242
|
+
],
|
|
164
243
|
// R10 P1-6: the same header as a QUOTED KEY — `{"Authorization":"Bearer …"}`. The
|
|
165
244
|
// pattern above needs `Authorization:` literally, and in JSON a quote sits between the
|
|
166
245
|
// name and the colon, so a `curl -v` / fetch header dump walked straight through. The
|
|
@@ -194,10 +273,10 @@ export const SECRET_PATTERNS = [
|
|
|
194
273
|
'$1$2://***',
|
|
195
274
|
],
|
|
196
275
|
// npm tokens (npm_...)
|
|
197
|
-
[/\bnpm_[a-zA-Z0-9]{36,}
|
|
276
|
+
[/\bnpm_[a-zA-Z0-9]{36,}(?![A-Za-z0-9])/g, '***'],
|
|
198
277
|
// Stripe keys (sk_live_, rk_live_, pk_live_, sk_test_, pk_test_) + webhook signing secret (whsec_)
|
|
199
|
-
[/\b[srp]k_(?:live|test)_[a-zA-Z0-9]{20,}
|
|
200
|
-
[/\bwhsec_[a-zA-Z0-9]{20,}
|
|
278
|
+
[/\b[srp]k_(?:live|test)_[a-zA-Z0-9]{20,}(?![A-Za-z0-9])/g, '***'],
|
|
279
|
+
[/\bwhsec_[a-zA-Z0-9]{20,}(?![A-Za-z0-9])/g, '***'],
|
|
201
280
|
// SendGrid API keys: SG.<22>.<43> — two dots at fixed offsets make this
|
|
202
281
|
// structurally unmistakable; near-zero false-positive risk.
|
|
203
282
|
[/\bSG\.[A-Za-z0-9_-]{22}\.[A-Za-z0-9_-]{43}\b/g, '***'],
|
|
@@ -224,8 +303,11 @@ export const SECRET_PATTERNS = [
|
|
|
224
303
|
// (password|secret|api_key|auth_token|access_token|private_key) so a benign
|
|
225
304
|
// `"token_count"` value (numeric, <6 non-quote chars after scrub) and prose
|
|
226
305
|
// keys stay low-FP; over-scrub is the safe direction for at-rest memory.
|
|
306
|
+
// `\w{0,64}`, not `\w*`, on both sides of the noun here and in the next pattern (D#130): from
|
|
307
|
+
// one quote, `\w*` ran to the end of a word run and backtracked through every keyword in it,
|
|
308
|
+
// each rescanning the run — `"` + `secret` x 33k took 4.6 s. 64 is far past any key name.
|
|
227
309
|
[
|
|
228
|
-
/("\w
|
|
310
|
+
/("\w{0,64}(?:password|passwd|secret|api[_-]?key|auth[_-]?token|access[_-]?token|private[_-]?key)\w{0,64}"\s*:\s*")[^"]{6,}(")/gi,
|
|
229
311
|
'$1***$2',
|
|
230
312
|
],
|
|
231
313
|
// Quoted-KEY credential values — Python dict reprs `{'api_key': '...'}`, single-quoted
|
|
@@ -241,7 +323,7 @@ export const SECRET_PATTERNS = [
|
|
|
241
323
|
// `'token_count': 123456`); `passphrase` added here too (double-quoted JSON passphrase is
|
|
242
324
|
// subsumed by this pattern since `['"]` matches `"`). Over-scrub is the safe direction.
|
|
243
325
|
[
|
|
244
|
-
/(['"]
|
|
326
|
+
/(['"](?:\w{0,64}(?:password|passwd|passphrase|secret|api[_-]?key|auth[_-]?token|access[_-]?token|private[_-]?key)\w{0,64}|\w{1,64}_token)['"]\s*:\s*)(['"])[^'"]{6,}\2/gi,
|
|
245
327
|
'$1$2***$2',
|
|
246
328
|
],
|
|
247
329
|
// Session cookies in headers / urlencoded bodies (sessionid=, session_id=, JSESSIONID=, PHPSESSID=).
|
package/server.mjs
CHANGED
|
@@ -17,7 +17,8 @@ import {
|
|
|
17
17
|
formatSchemaSkewNotice,
|
|
18
18
|
} from './lib/schema-skew.mjs';
|
|
19
19
|
import { reRankWithContext, runIdleCleanup, buildServerInstructions } from './search-scoring.mjs';
|
|
20
|
-
import { searchObservationsHybrid } from './search-engine.mjs';
|
|
20
|
+
import { searchObservationsHybrid, snippetAddsInfo } from './search-engine.mjs';
|
|
21
|
+
import { autoHeaderNote, autoLegend, autoTag } from './lib/provenance.mjs';
|
|
21
22
|
import {
|
|
22
23
|
deepSearch,
|
|
23
24
|
resolveDeepMode,
|
|
@@ -124,6 +125,7 @@ import { AUTO_MERGE_THRESHOLD } from './lib/dedup-constants.mjs';
|
|
|
124
125
|
import {
|
|
125
126
|
insertDeferred,
|
|
126
127
|
listOpenWithOrdinal,
|
|
128
|
+
openOrdinalOf,
|
|
127
129
|
dropDeferred,
|
|
128
130
|
formatDropReasonHint,
|
|
129
131
|
resolveDeferredIds,
|
|
@@ -461,7 +463,7 @@ function formatSearchOutput(
|
|
|
461
463
|
// explicitly requested OR semantics — there's no "fallback" in that path.
|
|
462
464
|
const fallbackHint = orFallbackFired && !args.or ? ' (relaxed AND→OR)' : '';
|
|
463
465
|
lines.push(
|
|
464
|
-
`Found ${countLabel} result(s)${qLabel}${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}\n`,
|
|
466
|
+
`Found ${countLabel} result(s)${qLabel}${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}${autoLegend(paginatedResults)}\n`,
|
|
465
467
|
);
|
|
466
468
|
|
|
467
469
|
// `~Nt` = estimated tokens to fetch this row's full body via mem_get (attachBodyTokens).
|
|
@@ -470,9 +472,9 @@ function formatSearchOutput(
|
|
|
470
472
|
for (const r of paginatedResults) {
|
|
471
473
|
if (r.source === 'obs') {
|
|
472
474
|
lines.push(
|
|
473
|
-
`#${r.id} ${typeIcon(r.type)} [${r.type}] ${truncate(r.title || r.subtitle || '(untitled)')} | ${r.project} | ${fmtDate(r.date)}${tok(r)}`,
|
|
475
|
+
`#${r.id} ${typeIcon(r.type)} [${r.type}]${autoTag(r)} ${truncate(r.title || r.subtitle || '(untitled)')} | ${r.project} | ${fmtDate(r.date)}${tok(r)}`,
|
|
474
476
|
);
|
|
475
|
-
if (r.snippet
|
|
477
|
+
if (snippetAddsInfo(r.snippet, r.title)) {
|
|
476
478
|
lines.push(` ${truncate(r.snippet, 100)}`);
|
|
477
479
|
}
|
|
478
480
|
} else if (r.source === 'session') {
|
|
@@ -900,7 +902,7 @@ server.registerTool(
|
|
|
900
902
|
const renderFields = obsFieldFilter || OBS_FIELDS;
|
|
901
903
|
for (const row of rows) {
|
|
902
904
|
foundBySource.obs.add(row.id);
|
|
903
|
-
const lines = [`── #${row.id} ──`];
|
|
905
|
+
const lines = [`── #${row.id}${autoHeaderNote(row)} ──`];
|
|
904
906
|
// Retraction first (shared with the CLI `get` via get-core) — see supersededNotice.
|
|
905
907
|
const retracted = supersededNotice(row);
|
|
906
908
|
if (retracted) lines.push(retracted);
|
|
@@ -1220,8 +1222,7 @@ server.registerTool(
|
|
|
1220
1222
|
});
|
|
1221
1223
|
// Compute the ordinal for the freshly-inserted row so the response is
|
|
1222
1224
|
// immediately actionable ("ok, I deferred this as item 1").
|
|
1223
|
-
const
|
|
1224
|
-
const ord = open.find((o) => o.id === r.id)?.ordinal ?? null;
|
|
1225
|
+
const ord = openOrdinalOf(db, project, r.id);
|
|
1225
1226
|
return {
|
|
1226
1227
|
content: [
|
|
1227
1228
|
{
|
package/source-files.mjs
CHANGED
|
@@ -81,6 +81,8 @@ export const SOURCE_FILES = [
|
|
|
81
81
|
'lib/stats-quality.mjs',
|
|
82
82
|
'lib/low-signal-patterns.mjs',
|
|
83
83
|
'lib/private-strip.mjs',
|
|
84
|
+
// Which writer produced an observation (explicit save vs machine-written); search + get marks.
|
|
85
|
+
'lib/provenance.mjs',
|
|
84
86
|
'lib/citation-tracker.mjs',
|
|
85
87
|
// v3.47 (D#78 P1): per-(obs,file) edge attribution. Imported by hook.mjs
|
|
86
88
|
// (handleStop edge resolution). Missing from manifest → tarball hook.mjs
|
package/utils.mjs
CHANGED
|
@@ -56,7 +56,8 @@ export {
|
|
|
56
56
|
} from './bash-utils.mjs';
|
|
57
57
|
|
|
58
58
|
// Internal imports for functions that remain in this module
|
|
59
|
-
import { truncate } from './format-utils.mjs';
|
|
59
|
+
import { normalizeInline, truncate } from './format-utils.mjs';
|
|
60
|
+
import { stripPrivate } from './lib/private-strip.mjs';
|
|
60
61
|
import { stripTestSuffix } from './bash-utils.mjs';
|
|
61
62
|
// Static, and deliberately the dependency-free resolver (node:os + node:path only) —
|
|
62
63
|
// debugCatch's sampler must not pull in the DB layer. See its comment below.
|
|
@@ -282,9 +283,97 @@ export function isRelatedToEpisode(episode, newFiles) {
|
|
|
282
283
|
// magnitude above the longest cut here, so a secret that begins before the cut is still
|
|
283
284
|
// seen whole by the patterns, at bounded cost.
|
|
284
285
|
const DESC_SCRUB_WINDOW = 4096;
|
|
285
|
-
|
|
286
|
+
|
|
287
|
+
// Every field is cut strictly around private spans: closed `<private>` spans are redacted across
|
|
288
|
+
// the WHOLE input first (one linear pass; hook input is capped at 256 KiB), then the window stops
|
|
289
|
+
// before the first remaining `<private>` or private-key BEGIN, so no field shows a span whose end
|
|
290
|
+
// is out of view. Before v6.19.0 an unclosed opener in the window (a Grep line
|
|
291
|
+
// `notes.md:3:<private>bank pin 4412`) was shown as written (v6.19.0 pre-tag reviews P3-1, r3 P2-1).
|
|
292
|
+
// A `</private>` or private-key END that comes first is a span whose START is out of view (a
|
|
293
|
+
// nested span, `tail key.pem`), so everything before it may be its inside: the field shows nothing
|
|
294
|
+
// (round-3 P3-1). The PEM half is case-sensitive, like the scrubber's PEM pattern.
|
|
295
|
+
const PRIVATE_MARK_RE = /<\/?[Pp][Rr][Ii][Vv][Aa][Tt][Ee]>|-----(?:BEGIN|END) [A-Z0-9 ]*PRIVATE KEY/;
|
|
296
|
+
// A window edge that cuts a token leaves a fragment shorter than its pattern needs (`ghp_` and 12
|
|
297
|
+
// of its 36 characters), and whitespace collapsing can bring it into view: the cut token is
|
|
298
|
+
// dropped (defect review P3-6, round-3 P3-4/P3-6). A head window with no whitespace keeps it: its
|
|
299
|
+
// start is intact and it is the only part shown. A tail window with no whitespace is all one cut token.
|
|
300
|
+
const isWs = (c) => c === ' ' || c === '\n' || c === '\t' || c === '\r' || /\s/.test(c);
|
|
301
|
+
function dropCutTokenAtEnd(win, next) {
|
|
302
|
+
if (next === undefined || isWs(next)) return win;
|
|
303
|
+
let i = win.length;
|
|
304
|
+
while (i > 0 && !isWs(win[i - 1])) i--;
|
|
305
|
+
return i > 0 ? win.slice(0, i) : win;
|
|
306
|
+
}
|
|
307
|
+
function dropCutTokenAtStart(win, prev) {
|
|
308
|
+
if (prev === undefined || isWs(prev)) return win;
|
|
309
|
+
let i = 0;
|
|
310
|
+
while (i < win.length && !isWs(win[i])) i++;
|
|
311
|
+
return win.slice(i);
|
|
312
|
+
}
|
|
313
|
+
function scrubTruncate(str, max, window = DESC_SCRUB_WINDOW) {
|
|
286
314
|
if (typeof str !== 'string' || str === '') return truncate(str, max);
|
|
287
|
-
|
|
315
|
+
const stripped = stripPrivate(str);
|
|
316
|
+
let win = stripped.slice(0, window);
|
|
317
|
+
const mark = PRIVATE_MARK_RE.exec(win);
|
|
318
|
+
if (mark) win = mark[0][1] === '/' || mark[0].startsWith('-----END') ? '' : win.slice(0, mark.index);
|
|
319
|
+
else win = dropCutTokenAtEnd(win, stripped[window]);
|
|
320
|
+
// The window edge can split a surrogate pair; truncate only guards a cut it makes itself.
|
|
321
|
+
if (/[\uD800-\uDBFF]$/.test(win)) win = win.slice(0, -1);
|
|
322
|
+
return truncate(_scrubSecrets(win), max);
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
// Head+tail cut for command output. A check usually prints its verdict last ("...parsed: 0",
|
|
326
|
+
// "3 passed"); a head-only cut hands the episode summarizer an unresolved-looking fragment,
|
|
327
|
+
// and it has written a bugfix narrative for a check that passed. The tail comes from its own
|
|
328
|
+
// scrub window at the END of the original string: taking it from the head window would drop
|
|
329
|
+
// the verdict of any output longer than DESC_SCRUB_WINDOW.
|
|
330
|
+
//
|
|
331
|
+
// Output with a private span anywhere in it (a `<private>` tag or a PEM private-key marker) shows
|
|
332
|
+
// no tail, only v6.18.0's 60-character head (cut as above). The tail is scrubbed in its own
|
|
333
|
+
// window, which cannot see a span that crosses its edge, and pairing the markers per window
|
|
334
|
+
// stored span text two ways in the v6.19.0 pre-tag review (a cut inside a closed `<private>`, a
|
|
335
|
+
// key with no END more than 4096 characters back).
|
|
336
|
+
const PRIVATE_TAG_HINT_RE = /<\/?private>/i;
|
|
337
|
+
const HEAD_ONLY_MAX = 60;
|
|
338
|
+
// The tail window is scrubbed with TAIL_CONTEXT characters before it, which are then dropped: a
|
|
339
|
+
// label cut by the window's edge (`pass|word: <value>`) is seen whole, so its value is not left
|
|
340
|
+
// unlabelled at the window's start, where whitespace collapsing can bring it into view (v6.19.1
|
|
341
|
+
// pre-tag review F1). A replacement inside the context moves the cut by its length change; the
|
|
342
|
+
// token the cut lands in is dropped either way.
|
|
343
|
+
const TAIL_CONTEXT = 256;
|
|
344
|
+
function scrubTailWindow(str, window) {
|
|
345
|
+
const scrubbed = _scrubSecrets(str.slice(-(window + TAIL_CONTEXT)));
|
|
346
|
+
return dropCutTokenAtStart(scrubbed.slice(TAIL_CONTEXT), scrubbed[TAIL_CONTEXT - 1]);
|
|
347
|
+
}
|
|
348
|
+
const oneSpace = (s) => s.replace(/\s+/g, ' ');
|
|
349
|
+
// The early return needs the WHOLE output inside the head window: a long output whose first
|
|
350
|
+
// 4096 characters collapse to a few (whitespace) still has a tail to show (v6.19.0 pre-tag
|
|
351
|
+
// claims review F2).
|
|
352
|
+
function scrubTruncateEnds(str, max) {
|
|
353
|
+
if (typeof str === 'string' && (str.includes('PRIVATE KEY') || PRIVATE_TAG_HINT_RE.test(str))) {
|
|
354
|
+
return scrubTruncate(str, HEAD_ONLY_MAX);
|
|
355
|
+
}
|
|
356
|
+
// Up to two windows long, one window covers the whole output: two overlapping windows showed
|
|
357
|
+
// the same text twice (delta review P3-4).
|
|
358
|
+
const window =
|
|
359
|
+
typeof str === 'string' && str.length <= 2 * DESC_SCRUB_WINDOW
|
|
360
|
+
? 2 * DESC_SCRUB_WINDOW
|
|
361
|
+
: DESC_SCRUB_WINDOW;
|
|
362
|
+
// Whitespace runs are one space: a blank-line run no longer spends the budget (or, before the
|
|
363
|
+
// one-window rule, hid behind the window edge).
|
|
364
|
+
const flat = oneSpace(scrubTruncate(str, window, window));
|
|
365
|
+
const whole = typeof str !== 'string' || str.length <= window;
|
|
366
|
+
if (whole && flat.length <= max) return flat;
|
|
367
|
+
const tailLen = Math.floor(max / 2) - 1;
|
|
368
|
+
const tailSrc = whole ? flat : oneSpace(normalizeInline(scrubTailWindow(str, window)));
|
|
369
|
+
if (tailSrc === '') return truncate(flat, max);
|
|
370
|
+
// Drop a lone low surrogate the tail's cut may start on.
|
|
371
|
+
const tail = tailSrc.slice(-tailLen).replace(/^[\uDC00-\uDFFF]/, '');
|
|
372
|
+
// A head short enough to escape truncate's own "…" still gets one before the tail.
|
|
373
|
+
const budget = max - tailLen;
|
|
374
|
+
let head = truncate(flat, budget);
|
|
375
|
+
if (!head.endsWith('…')) head = head.length < budget ? `${head}…` : truncate(flat, budget - 1);
|
|
376
|
+
return head + tail;
|
|
288
377
|
}
|
|
289
378
|
|
|
290
379
|
export function makeEntryDesc(toolName, input, resp, opts) {
|
|
@@ -302,7 +391,7 @@ export function makeEntryDesc(toolName, input, resp, opts) {
|
|
|
302
391
|
const isErr =
|
|
303
392
|
opts?.isError ??
|
|
304
393
|
(/\berror\b|\bfail(ed|ure)?\b|\bexception\b|\bpanic\b/i.test(resp) && resp.length > 30);
|
|
305
|
-
const snippet =
|
|
394
|
+
const snippet = scrubTruncateEnds(resp, 100);
|
|
306
395
|
return isErr ? `${cmd} → ERROR: ${snippet}` : `${cmd} → ${snippet}`;
|
|
307
396
|
}
|
|
308
397
|
case 'Grep':
|