claude-mem-lite 6.19.0 → 6.19.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@
9
9
  "plugins": [
10
10
  {
11
11
  "name": "claude-mem-lite",
12
- "version": "6.19.0",
12
+ "version": "6.19.2",
13
13
  "source": "./",
14
14
  "homepage": "https://github.com/sdsrss/claude-mem-lite",
15
15
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.0",
3
+ "version": "6.19.2",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "author": {
6
6
  "name": "sdsrss"
package/hook-optimize.mjs CHANGED
@@ -32,6 +32,7 @@ import { normalizeScope, SCOPE_PROMPT_LEGEND, insertObservationRow } from './lib
32
32
  import { liveObsFilterSql } from './lib/inject-search-core.mjs';
33
33
  import { resolveRuntimeDir } from './lib/resolve-data-dir.mjs';
34
34
  import { MEMORY_INPUT_GUARD } from './lib/memory-input-guard.mjs';
35
+ import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './lib/provenance.mjs';
35
36
 
36
37
  import { DAY_MS } from './lib/time-constants.mjs';
37
38
  // P1-14: same resolver as hook-shared.mjs — this was the second module that had never
@@ -582,12 +583,34 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
582
583
  // text over it, wide wrote back the pre-edit narrative it read before the call
583
584
  // (tests/verify-reenrich-race.test.mjs). It also stops two overlapping runs rewriting a
584
585
  // row twice.
586
+ //
587
+ // Narrow replaces title and narrative with model text, so an explicit save it rewrites (one
588
+ // whose save-time enrich failed) moves to the re-enrich writer's id, as a cluster-merge
589
+ // keeper does (D#146, D#138). Wide keeps the stored title and narrative (it fills the lesson
590
+ // and the side fields), so the row stays an explicit save. The writer's session row is
591
+ // best-effort.
592
+ let rewriteSessionId = null;
593
+ if (!isWide) {
594
+ const cur = db.prepare('SELECT memory_session_id FROM observations WHERE id = ?').get(cand.id);
595
+ if (cur?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
596
+ const enrichSessionId = writerSessionId('enrich-', cand.project);
597
+ try {
598
+ db.prepare(
599
+ `INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
600
+ VALUES (?, ?, ?, ?, ?, 'active')`,
601
+ ).run(enrichSessionId, enrichSessionId, cand.project, new Date().toISOString(), Date.now());
602
+ rewriteSessionId = enrichSessionId;
603
+ } catch (e) {
604
+ debugCatch(e, 'reenrich writer session');
605
+ }
606
+ }
607
+ }
585
608
  const res = db
586
609
  .prepare(
587
610
  `
588
611
  UPDATE observations SET type=?, title=?, narrative=?, concepts=?, facts=?,
589
612
  text=?, importance=?, lesson_learned=?, search_aliases=?, minhash_sig=?, optimized_at=?,
590
- scope=COALESCE(?, scope)
613
+ scope=COALESCE(?, scope), memory_session_id = COALESCE(?, memory_session_id)
591
614
  WHERE id = ? AND ${liveObsFilterSql('')} AND optimized_at IS NULL
592
615
  `,
593
616
  )
@@ -607,6 +630,7 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
607
630
  // scope, or emits an off-enum value, must never blank an existing label —
608
631
  // and THIS update stamps optimized_at, so the loss would be permanent.
609
632
  normalizeScope(parsed.scope),
633
+ rewriteSessionId,
610
634
  cand.id,
611
635
  );
612
636
  if (res.changes === 0) {
@@ -1108,7 +1132,7 @@ export function findMergeCandidates(db, maxClusters = 5, { project } = {}) {
1108
1132
  -- keeper.search_aliases when it rebuilt the keeper's TF-IDF vector. Phase-2 removed that
1109
1133
  -- rebuild, so the column had no reader left and went with it. Do NOT re-add it on the
1110
1134
  -- strength of R10 P3-7 -- that finding is moot, not pending. executeMergeCluster reads
1111
- -- keeper.{id,importance,narrative,concepts,facts} and o.{id,title,type,narrative,
1135
+ -- keeper.{id,project,importance,narrative,concepts,facts} and o.{id,title,type,narrative,
1112
1136
  -- importance,access_count,lesson_learned}, and nothing else off these rows.
1113
1137
  SELECT id, title, narrative, project, type, access_count, importance, created_at_epoch, minhash_sig, lesson_learned, concepts, facts
1114
1138
  FROM observations
@@ -1311,10 +1335,33 @@ Return ONLY valid JSON:
1311
1335
  SELECT ${snapColList}, ? FROM observations WHERE id = ?`,
1312
1336
  ).run(keeper.id, keeper.id);
1313
1337
 
1338
+ // The keeper now holds model text. Search reads authorship from memory_session_id, so an
1339
+ // explicit save's `manual-` id would mark it as one (D#138); it moves to the compression
1340
+ // writer's id. A machine-written keeper keeps its id, and so does the snapshot above, which
1341
+ // is the original save. writerSessionId keeps the id out of the uuid shape sdk_sessions
1342
+ // refuses (D#147); the row stays best-effort, so no refusal can fail the merge (v6.19.1
1343
+ // pre-tag F5a).
1344
+ let rewriteSessionId = null;
1345
+ const keeperSession = db
1346
+ .prepare('SELECT memory_session_id FROM observations WHERE id = ?')
1347
+ .get(keeper.id);
1348
+ if (keeperSession?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
1349
+ const compressSessionId = writerSessionId('compress-', keeper.project);
1350
+ try {
1351
+ db.prepare(
1352
+ `INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
1353
+ VALUES (?, ?, ?, ?, ?, 'active')`,
1354
+ ).run(compressSessionId, compressSessionId, keeper.project, new Date().toISOString(), Date.now());
1355
+ rewriteSessionId = compressSessionId;
1356
+ } catch (e) {
1357
+ debugCatch(e, 'cluster-merge writer session');
1358
+ }
1359
+ }
1314
1360
  db.prepare(
1315
1361
  `
1316
1362
  UPDATE observations SET title=?, narrative=?, concepts=?, facts=?, text=?,
1317
- importance=?, lesson_learned=?, minhash_sig=?, optimized_at=?
1363
+ importance=?, lesson_learned=?, minhash_sig=?, optimized_at=?,
1364
+ memory_session_id = COALESCE(?, memory_session_id)
1318
1365
  WHERE id = ?
1319
1366
  `,
1320
1367
  ).run(
@@ -1327,6 +1374,7 @@ Return ONLY valid JSON:
1327
1374
  safe.lesson_learned,
1328
1375
  minhashSig,
1329
1376
  Date.now(),
1377
+ rewriteSessionId,
1330
1378
  keeper.id,
1331
1379
  );
1332
1380
 
@@ -1555,7 +1603,7 @@ export async function executeSmartCompressCluster(db, observations, project) {
1555
1603
  ) {
1556
1604
  return null;
1557
1605
  }
1558
- const sessionId = `compress-${project}`;
1606
+ const sessionId = writerSessionId('compress-', project);
1559
1607
  const now = new Date();
1560
1608
  db.prepare(
1561
1609
  `INSERT OR IGNORE INTO sdk_sessions
package/lib/activity.mjs CHANGED
@@ -7,6 +7,7 @@
7
7
  import { sanitizeFtsQuery } from '../utils.mjs';
8
8
  import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
9
9
  import { saveObservation } from './save-observation.mjs';
10
+ import { writerSessionId } from './provenance.mjs';
10
11
  // Pure title-only builder: this query runs on the EVENTS table, which has no
11
12
  // lesson_learned column — the lesson-escape variant would be a SQL error here.
12
13
  import { buildNotLowSignalSql } from './low-signal-patterns.mjs';
@@ -199,6 +200,8 @@ export function promoteInsightEvents(
199
200
  // manual-save contract) so it lands in the high-weight FTS field.
200
201
  lesson_learned: ev.body.slice(0, 500),
201
202
  now: new Date(ev.created_at_epoch),
203
+ // Event bodies are machine-written (episode summaries), so the row is not an explicit save.
204
+ sessionId: writerSessionId('promote-', ev.project),
202
205
  });
203
206
  mark.run(Date.now(), ev.id);
204
207
  return r;
@@ -17,6 +17,7 @@
17
17
 
18
18
  import { isoWeekKey, COMPRESSED_AUTO } from '../utils.mjs';
19
19
  import { scrubRecord } from './scrub-record.mjs';
20
+ import { writerSessionId } from './provenance.mjs';
20
21
 
21
22
  /**
22
23
  * Low-value compression candidates: importance<=1, never accessed, older than
@@ -88,7 +89,7 @@ export function compressGroup(db, proj, obs) {
88
89
  const dominantType = Object.entries(types).sort((a, b) => b[1] - a[1])[0][0];
89
90
  const title = `Weekly summary: ${obs.length} ${dominantType} observations`;
90
91
  const narrative = obs.map((o) => `- ${o.title || '(untitled)'}`).join('\n');
91
- const sessionId = `compress-${proj}`;
92
+ const sessionId = writerSessionId('compress-', proj);
92
93
 
93
94
  const sortedEpochs = obs.map((o) => o.created_at_epoch).sort((a, b) => a - b);
94
95
  const medianEpoch = sortedEpochs[Math.floor(sortedEpochs.length / 2)];
@@ -86,6 +86,30 @@ export function listOpenWithOrdinal(db, project, limit = 10) {
86
86
  .all(project, limit);
87
87
  }
88
88
 
89
+ /**
90
+ * One open row's ordinal, numbered as listOpenWithOrdinal numbers it; null when the row is not
91
+ * open. `defer add` / `mem_defer` read it from a 50-row page before, so an item added past 50
92
+ * open printed `(item ?)` (D#123).
93
+ * @param {Database} db
94
+ * @param {string} project
95
+ * @param {number} id
96
+ * @returns {number|null}
97
+ */
98
+ export function openOrdinalOf(db, project, id) {
99
+ const row = db
100
+ .prepare(
101
+ `
102
+ SELECT ordinal FROM (
103
+ SELECT id, ROW_NUMBER() OVER (ORDER BY priority DESC, created_at_epoch ASC, id ASC) AS ordinal
104
+ FROM deferred_work
105
+ WHERE project = ? AND status = 'open'
106
+ ) WHERE id = ?
107
+ `,
108
+ )
109
+ .get(project, id);
110
+ return row ? row.ordinal : null;
111
+ }
112
+
89
113
  /**
90
114
  * How many open rows `defer list` / `mem_defer_list` left off their page, as a line to
91
115
  * print — '' when the page held them all. Without it a page one short of the open set
@@ -26,7 +26,8 @@ export function buildBridgePrompt(lesson, hunk) {
26
26
 
27
27
  // { ok:true, check } when the bridge produced a usable, applicable check;
28
28
  // { ok:false } on N/A / empty / error / timeout. NEVER throws — the caller
29
- // falls back to the baseline ACK_DIRECTIVE on { ok:false }.
29
+ // falls back to the arm's plain directive on { ok:false } (VERDICT_DIRECTIVE since 6.17.0,
30
+ // scripts/pre-tool-recall.js ACTIVE_DIRECTIVE).
30
31
  export async function bridgeLesson({ lesson, hunk, timeoutMs = 2500, _callLLM = callLLM }) {
31
32
  try {
32
33
  const raw = await _callLLM(buildBridgePrompt(lesson, hunk), timeoutMs);
@@ -1,6 +1,7 @@
1
1
  // Which writer produced an observation, read back from its memory_session_id. saveObservation
2
2
  // writes every explicit save under MANUAL_SESSION_ID_PREFIX. The hook capture (`hook-`),
3
- // transcript import (`import-`), compression summaries (`compress-`) and rows imported from
3
+ // transcript import (`import-`), compression summaries and cluster-merge keepers (`compress-`),
4
+ // rows narrow re-enrich rewrote (`enrich-`), promoted events (`promote-`) and rows imported from
4
5
  // older stores (a bare session uuid) are machine-written.
5
6
  //
6
7
  // Search and get mark the machine-written side, because explicit saves are the large majority
@@ -9,6 +10,20 @@
9
10
 
10
11
  export const MANUAL_SESSION_ID_PREFIX = 'manual-';
11
12
 
13
+ // sdk_sessions refuses a row whose two ids are equal and 36 characters long with dashes where a
14
+ // uuid has them (schema.mjs, sdk_sessions_id_mix_check_ai). A writer row uses `<prefix><project>`
15
+ // for both, which has that shape for one project length and dash layout (27 characters under
16
+ // `compress-`, 29 under `manual-`), and the refusal failed every write in such a project (D#147).
17
+ // That id gets one more character; every other id is unchanged. Counted in code points, as
18
+ // SQLite's length() and LIKE's `_` count them.
19
+ const UUID_SHAPED_RE = /^[\s\S]{8}-[\s\S]{4}-[\s\S]{4}-[\s\S]{4}-[\s\S]{12}$/u;
20
+
21
+ /** The session id a non-hook writer (`manual-`, `compress-`, `enrich-`, `promote-`) stores for a project. */
22
+ export function writerSessionId(prefix, project) {
23
+ const id = `${prefix}${project}`;
24
+ return UUID_SHAPED_RE.test(id) ? `${id}~` : id;
25
+ }
26
+
12
27
  const AUTO_MARK = '🤖';
13
28
  const AUTO_TEXT = 'auto-written, not an explicit save';
14
29
 
@@ -22,7 +22,7 @@ import { liveObsFilterSql } from './inject-search-core.mjs';
22
22
  // Imported, not injected: `allowStatuses` below is the POLICY this function exists to hold,
23
23
  // and a caller free to pass its own resolver could reinstate the one-way gate D#195 closed.
24
24
  import { resolveDeferredIds, closeDeferredItems } from './deferred-work.mjs';
25
- import { MANUAL_SESSION_ID_PREFIX } from './provenance.mjs';
25
+ import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './provenance.mjs';
26
26
 
27
27
  const DEDUP_WINDOW_MS = 5 * 60 * 1000;
28
28
  const DEDUP_RECENT_LIMIT = 50;
@@ -144,6 +144,7 @@ export function formatSupersedeSkipped(skipped) {
144
144
  * @param {string|null} [params.lesson_learned] Caller validates ≤500 chars.
145
145
  * @param {boolean} [params.force=false] Skip the near-duplicate window (below).
146
146
  * @param {Date} [params.now] Override for tests.
147
+ * @param {string} [params.sessionId] Writer id; default `manual-<project>` (an explicit save).
147
148
  * Both result shapes carry `supersededIds` (observations actually tombstoned) and
148
149
  * `supersedeSkipped` (requested but NOT tombstoned, each with a `reason`:
149
150
  * `malformed-id` | `no-such-observation` | `no-such-event` | `other-project` |
@@ -198,7 +199,9 @@ export function saveObservation(db, params) {
198
199
  const safeTitle = scrubSecrets(rawTitle);
199
200
  const safeLesson = rawLesson ? scrubSecrets(rawLesson) : null;
200
201
 
201
- const sessionId = `${MANUAL_SESSION_ID_PREFIX}${project}`;
202
+ // An explicit save by default; a caller storing machine-written text names its own writer
203
+ // (lib/provenance.mjs reads authorship from this id).
204
+ const sessionId = params.sessionId || writerSessionId(MANUAL_SESSION_ID_PREFIX, project);
202
205
 
203
206
  // Ensure session exists (FK constraint). INSERT OR IGNORE makes this safe
204
207
  // under concurrent calls.
package/mem-cli.mjs CHANGED
@@ -163,6 +163,7 @@ import { aggregateMetrics, readMetrics } from './lib/metrics.mjs';
163
163
  import {
164
164
  insertDeferred,
165
165
  listOpenWithOrdinal,
166
+ openOrdinalOf,
166
167
  dropDeferred,
167
168
  formatDropReasonHint,
168
169
  resolveDeferredIds,
@@ -1448,9 +1449,8 @@ function cmdDeferAdd(db, args) {
1448
1449
  return;
1449
1450
  }
1450
1451
  // Compute the freshly-inserted row's ordinal for an immediately-actionable
1451
- // response ("ok, deferred this as item N"). Mirrors server.mjs:980.
1452
- const open = listOpenWithOrdinal(db, project, 50);
1453
- const ord = open.find((o) => o.id === r.id)?.ordinal ?? '?';
1452
+ // response ("ok, deferred this as item N"), as mem_defer does.
1453
+ const ord = openOrdinalOf(db, project, r.id) ?? '?';
1454
1454
  out(`[mem] Deferred as D#${r.id} (item ${ord}) in project "${project}".`);
1455
1455
  }
1456
1456
 
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.0",
3
+ "version": "6.19.2",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "claude-mem-lite",
9
- "version": "6.19.0",
9
+ "version": "6.19.2",
10
10
  "os": [
11
11
  "darwin",
12
12
  "linux",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.0",
3
+ "version": "6.19.2",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "type": "module",
6
6
  "packageManager": "npm@10.9.2",
package/secret-scrub.mjs CHANGED
@@ -184,32 +184,35 @@ export const SECRET_PATTERNS = [
184
184
  [/\b(?:xox[bpasr]|xapp|xoxe)-[a-zA-Z0-9-]{10,}\b/g, '***'],
185
185
  // Slack incoming-webhook URL — the path after /services/ is the shared secret.
186
186
  [/(https:\/\/hooks\.slack\.com\/services\/)[A-Za-z0-9/]+/g, '$1***'],
187
- // JWT tokens (eyJ...eyJ...). A JWT begins its own token, so the start is `(?<![\w-])`, not
188
- // `\b`: `-` is a base64url character, and `\b` let every `-eyJ` inside one dotless run be a
189
- // fresh start that rescans the run to its end — quadratic, 9.6 s on 200k chars of `eyJ-`
190
- // (D#130). A `-eyJ` start is the middle of a run, never a JWT's first character.
191
- // A JWT glued by a hyphen to up to 40 characters of hyphenated words (`my-sess-eyJ…`,
192
- // `X-Auth-Token-eyJ…`) is a start too. `\b` allowed any length; past 40 characters it is missed. The lookbehind needs a token
193
- // boundary within those 41 characters, so a hyphen run has at most ~10 starts, not one per
194
- // `-eyJ` (v6.19.0 pre-tag reviews P3-2, delta P3-1).
187
+ // JWT tokens (eyJ...eyJ...), from any `eyJ`: one glued to a prefix (`my-sess-eyJ…`,
188
+ // `session-<uuid>-eyJ…`, `tok_eyJ…`) is still a JWT. Every `eyJ` of one dotless run used to be a
189
+ // fresh start that rescanned the run to its end — quadratic, 9.6 s on 200k chars of `eyJ-`
190
+ // (D#130) — and v6.19.0's bounded lookbehind that fixed it missed a prefix over 40 characters
191
+ // (round-3 review P3-2). Here a failed start consumes its run instead (`|eyJ[\w-]*`, returned
192
+ // unchanged), so the scan resumes after it. Nothing is lost: the first segment cannot contain a
193
+ // `.`, so every later `eyJ` of the same run reaches the same run end and fails the same way.
195
194
  [
196
- /(?:(?<![\w-])|(?<=(?:^|[^\w-])[\w-]{1,40}-))eyJ[a-zA-Z0-9_-]{10,}\.eyJ[a-zA-Z0-9_-]{10,}\.[a-zA-Z0-9_-]+\b/g,
197
- '***',
195
+ /eyJ[a-zA-Z0-9_-]{10,}\.eyJ[a-zA-Z0-9_-]{10,}\.[a-zA-Z0-9_-]+\b|eyJ[a-zA-Z0-9_-]*/g,
196
+ (m) => (m.includes('.') ? '***' : m),
198
197
  ],
199
198
  // PEM private key blocks. `[A-Z0-9 ]*` covers every armor label — RSA/EC/DSA/
200
199
  // OPENSSH plus ENCRYPTED and PGP (… PRIVATE KEY BLOCK) — that the fixed
201
200
  // alternation missed; the block delimiters make FP impossible.
202
201
  // The body stops at the next `-----BEGIN ` (D#130): with `[\s\S]*?` every header with no END
203
202
  // scanned to the end of the text, on each of scrubSecrets' passes — quadratic, 8.3 s on 500k
204
- // chars. A block whose END is missing ends where the next private-key BEGIN starts, so a cut-off
205
- // key's body is scrubbed too (v6.19.0 pre-tag review P2-1: requiring the END there stored that
206
- // body); any other BEGIN (a certificate) does not end it, so a bare header in prose does not
207
- // erase the text up to one (delta review P3-2). A header with no END and no later key header
208
- // is left as it was in v6.18.0.
203
+ // chars. A block whose END is missing is left to the next pattern. Nor does it cross a mark this
204
+ // scrubber wrote: on a later pass, a BEGIN that a scrubbed key used to block reached a far END
205
+ // and erased the prose between (v6.19.2 pre-tag defect review F5).
209
206
  [
210
- /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY)[\s\S])*?(?:-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----|(?=-----BEGIN [A-Z0-9 ]*PRIVATE KEY))/g,
207
+ /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY|\*\*\*PEM_KEY\*\*\*)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
211
208
  '***PEM_KEY***',
212
209
  ],
210
+ // A cut-off key (`head id_rsa`, a tool output cut mid-key) and a headerless key tail (`tail
211
+ // key.pem`). These are line scanners, not patterns (D#145): three rounds of review each found a
212
+ // line shape the regex versions misread, and each fix to one shape erased prose in another.
213
+ // See scrubCutOffKeys and scrubKeyTails below for the line rules.
214
+ [{ [Symbol.replace]: (text) => scrubCutOffKeys(text) }, null],
215
+ [{ [Symbol.replace]: (text) => scrubKeyTails(text) }, null],
213
216
  // Long hex strings in credential assignments (e.g. SECRET_KEY=abc123def456...).
214
217
  // `hash` deliberately excluded: `hash: <40hex>` / `hash=<md5>` are git SHAs and
215
218
  // checksums (real, preserved data in this hash-heavy repo), not credentials.
@@ -333,6 +336,463 @@ export const SECRET_PATTERNS = [
333
336
  // bare-token pattern here: don't — anchor it to a provider prefix instead.
334
337
  ];
335
338
 
339
+ // ─── Cut-off private keys (D#145) ──────────────────────────────────────────
340
+ // A key's text reaches the scrubber in many line shapes: plain LF/CRLF/CR lines, JSON-escaped
341
+ // breaks (`\n` as two characters, or `\\n` when serialised twice), lines carrying a prefix (the
342
+ // Read tool's ` 2\t` or `2→`, grep's `id_rsa:`, a `> ` quote, a diff `-`), and lines inside a
343
+ // quoted string. The scanners read the text as lines of ONE shape per key, decided at the key:
344
+ // - the break after the BEGIN line (or before the END line) says whether breaks are real or
345
+ // escaped, and at which depth; in real-break text a backslash is never a break, so a Windows
346
+ // path in a `Comment:` value no longer ends the header (v6.19.1 round-3 P3-E);
347
+ // - the text before the BEGIN (or END) on its line is the line prefix, and every other line
348
+ // loses a prefix of the same shape (digits may differ, `:` and `-` swap for grep context)
349
+ // before it is judged;
350
+ // - lines end at breaks only. A header value that runs on into the next JSON fields is still
351
+ // one header line, so it erases nothing without a base64 line under it (round-3 P3-C).
352
+ // A body line is WHOLE base64: 16+ characters, or 1-15 for the last line. So a word under a key
353
+ // body keeps its text (round-3 P3-A: `Don't` lost `Don`), as does a path or an identifier that
354
+ // starts the next line (P3-B), and a body needs one long line, so a header followed by words is
355
+ // not a key (F3). A string's closing quote, a backtick or a closing tag after the line is not part
356
+ // of it. A line whose base64 run is followed by something else is a cut or annotated key line
357
+ // when the run is 40+ characters or a truncation mark or delimiter follows it (`…`, `...`,
358
+ // `[truncated]`, `<`): its base64 goes and the rest stays (delta review P3-2; v6.19.2 pre-tag
359
+ // defect review F4). A shorter run followed by words starts a line of prose.
360
+
361
+ const KEY_BEGIN_RE = /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
362
+ const KEY_END_RE = /-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
363
+ const KEY_END_LINE_RE = /^-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/;
364
+ const B64_LONG_RE = /^[A-Za-z0-9+/=]{16,}$/;
365
+ const B64_SHORT_RE = /^[A-Za-z0-9+/=]{1,15}$/;
366
+ // Space-separated chunks, allowed only on the BEGIN line itself (a key pasted onto one line).
367
+ const B64_CHUNKS_RE = /^[A-Za-z0-9+/=]+(?:[ \t]+[A-Za-z0-9+/=]+)*$/;
368
+ const B64_RUN_RE = /^[A-Za-z0-9+/=]{16,}/;
369
+ // What may follow a key line that was cut or annotated: a truncation mark or a delimiter. A quote
370
+ // ends the string a key sits in (`…', 'rc': 0}`, `…","stderr":…`): the scanners do not track which
371
+ // quote opened a string, since no character before a quote tells an opening quote from prose
372
+ // (`Here's`, `Run 'head id_rsa'`, `b'…'`; v6.19.2 pre-tag reviews, delta F1 and round-3 F1/F2).
373
+ // Two escapes deep a string ends at `\"` (round-4 F2). Five dashes are the next key's marker glued
374
+ // to a cut line (`for i in …; do head -c 80 k; done`).
375
+ const CUT_MARK_RE = /^(?: ?…| ?\.\.\.| ?\[| ?<|[`"']|\\+["']|-----)/;
376
+ const PGP_CRC_RE = /^=[A-Za-z0-9+/]{4}$/;
377
+ // How a key's base64 starts: a DER SEQUENCE (PKCS#1, PKCS#8, SEC1) or OpenSSH's `openssh-key-v1`;
378
+ // under a PGP header, a secret-key packet in the old or new format (`lQ…`, `xc…`/`xV…`).
379
+ const KEY_MAGIC_RE = /^(?:MII|MIG|MC4C|MHcC|b3BlbnNzaC1rZXktdjE)/;
380
+ const PGP_MAGIC_RE = /^(?:lQ|x[cV])/;
381
+ // RFC 1421 / RFC 4880 armor headers, before the body. Named, not any `Word:`: a `Note:` line is
382
+ // prose, and taking it for a header made the lone base64 line under it a key.
383
+ // RFC 1421's full set and tool-written `X-` headers count too (v6.19.2 pre-tag delta review F4:
384
+ // `Content-Domain` or `X-Custom` before the body stored the whole key).
385
+ const ARMOR_HEADER_RE =
386
+ /^(?:Proc-Type|DEK-Info|Content-Domain|Originator-ID-(?:Asymmetric|Symmetric)|Originator-Certificate|Issuer-Certificate|MIC-Info|Key-Info|Recipient-ID-(?:Asymmetric|Symmetric)|CRL|Version|Comment|Hash|Charset|MessageID|X-[A-Za-z0-9-]+)[ \t]*:/i;
387
+ const MAX_ARMOR_HEADERS = 16;
388
+ // A JS/Python string split across source lines: `…\n" +` then `"…` on the next line. Not a comma:
389
+ // `'…\n',` then `'…'` is the next element of a list, and its first word is not the key's last line.
390
+ // One whitespace quantifier on each side of the `+`: `[ \t]*\+?[ \t]*` split a run between two
391
+ // and was quadratic when no break followed (v6.19.2 pre-tag defect review F1: 16-19 s at 200k).
392
+ const CONCAT_AFTER_RE = /["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']/y;
393
+ const CONCAT_BEFORE_RE = /(?:\\r)?\\n["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']$/;
394
+ const PEM_MARK = '***PEM_KEY***';
395
+ // After an END that ends its line: a closing quote or punctuation, then a break or the text end.
396
+ // Closing tags may be nested (`END</code></pre>`; round-4 F3), and a string's closing quote, escaped
397
+ // or not, ends the line whatever follows it: `…END-----","stderr":""` is `tail -n 2` in a tool
398
+ // result (round-5 F3).
399
+ const CLEAN_AFTER_END_RE = /(?:[ \t`,;)\]}]|<\/[A-Za-z][\w:.-]{0,40}>)*(?:\r|\n|\\+[nr]|\\*["']|$)/y;
400
+
401
+ const BLANK = 0;
402
+ const LONG = 1;
403
+ const SHORT = 2;
404
+ const END = 3;
405
+ const ARMOR = 4;
406
+ const CUT = 5;
407
+ const OTHER = 6;
408
+
409
+ function backslashesBefore(text, i, floor) {
410
+ let j = i;
411
+ while (j > floor && text[j - 1] === '\\') j--;
412
+ return i - j;
413
+ }
414
+
415
+ // Break depth `u`: 0 = real breaks only, 1 = `\n`, 2 = `\\n`, … A run of r backslashes then `n`
416
+ // is a break at depth u when r % 2u === u (the backslashes before it are escaped backslashes).
417
+ const isEscBreak = (r, u) => u > 0 && r % (2 * u) === u;
418
+ // An unescaped quote at depth u ends the string: r % 2u < u.
419
+ const isStringEnd = (r, u) => u > 0 && r % (2 * u) < u;
420
+
421
+ /** The line starting at s: its end, and where the next line starts (-1 at the end of the text). */
422
+ function nextLine(text, s, u) {
423
+ const n = text.length;
424
+ let i = s;
425
+ while (i < n) {
426
+ const ch = text[i];
427
+ if (ch === '\n') return { end: i, next: i + 1 };
428
+ if (ch === '\r') return { end: i, next: text[i + 1] === '\n' ? i + 2 : i + 1 };
429
+ if (u > 0 && ch === '\\') {
430
+ let j = i;
431
+ while (j < n && text[j] === '\\') j++;
432
+ const r = j - i;
433
+ const c = text[j];
434
+ if ((c === 'n' || c === 'r') && isEscBreak(r, u)) {
435
+ let next = j + 1;
436
+ if (c === 'r' && text.startsWith('\\'.repeat(u) + 'n', next)) next += u + 1;
437
+ if (u === 1) {
438
+ CONCAT_AFTER_RE.lastIndex = next;
439
+ const m = CONCAT_AFTER_RE.exec(text);
440
+ if (m) next += m[0].length;
441
+ }
442
+ return { end: j - u, next };
443
+ }
444
+ i = j + 1;
445
+ continue;
446
+ }
447
+ i++;
448
+ }
449
+ return { end: n, next: -1 };
450
+ }
451
+
452
+ /** The line that ends at the break before `ls`, or null when `ls` starts the text or string. */
453
+ function prevLine(text, ls, u, floor) {
454
+ if (ls - 1 < floor) return null;
455
+ const c = text[ls - 1];
456
+ let bs = -1;
457
+ if (c === '\n') bs = ls - 2 >= floor && text[ls - 2] === '\r' ? ls - 2 : ls - 1;
458
+ else if (c === '\r') bs = ls - 1;
459
+ else if ((c === 'n' || c === 'r') && isEscBreak(backslashesBefore(text, ls - 1, floor), u)) {
460
+ bs = ls - 1 - u;
461
+ const r2 = bs - 1 >= floor && text[bs - 1] === 'r' ? backslashesBefore(text, bs - 1, floor) : 0;
462
+ if (c === 'n' && isEscBreak(r2, u)) bs -= u + 1;
463
+ } else if (u === 1 && (c === '"' || c === "'")) {
464
+ const m = CONCAT_BEFORE_RE.exec(text.slice(Math.max(floor, ls - 64), ls));
465
+ if (m) bs = ls - m[0].length;
466
+ }
467
+ if (bs === -1) return null;
468
+ let i = bs - 1;
469
+ while (i >= floor) {
470
+ const ch = text[i];
471
+ if (ch === '\n' || ch === '\r') break;
472
+ if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, i, floor), u)) break;
473
+ if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, i, floor), u)) break;
474
+ i--;
475
+ }
476
+ return { start: i + 1, end: bs };
477
+ }
478
+
479
+ /** A regex for the line prefix `prefix` has, with digits free and grep's `:`/`-` interchangeable. */
480
+ function prefixShape(prefix) {
481
+ const body = prefix.replace(/^[ \t]+/, '');
482
+ if (!body || body.length > 256) return null;
483
+ if (body === '-' || body === '+') return /^[ \t]*[-+ ]/;
484
+ const src = body
485
+ .replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')
486
+ .replace(/\d+/g, '\\d+')
487
+ .replace(/[:-]/g, '[:-]');
488
+ return new RegExp(`^[ \\t]*${src}`);
489
+ }
490
+
491
+ /** The judged part of line [s, e): no prefix of the key's shape, no padding, no string quotes. */
492
+ function lineCore(text, s, e, shape) {
493
+ let line = text.slice(s, e);
494
+ let off = s;
495
+ if (shape) {
496
+ const m = shape.exec(line);
497
+ if (m) {
498
+ line = line.slice(m[0].length);
499
+ off += m[0].length;
500
+ }
501
+ }
502
+ let a = 0;
503
+ let b = line.length;
504
+ while (a < b && (line[a] === ' ' || line[a] === '\t')) a++;
505
+ while (b > a && (line[b - 1] === ' ' || line[b - 1] === '\t')) b--;
506
+ if (a < b && (line[a] === '"' || line[a] === "'" || line[a] === '`')) a++;
507
+ // A closing quote or backtick with what may follow it (`",`, `" +`, `')`), or a closing tag
508
+ // (`</key>`), and escaped breaks before it. Read from the end: an unanchored
509
+ // `(?:\\+[rn])*["']…$` retried every start in a backslash run.
510
+ let q = b;
511
+ while (q > a && /[ \t+,;)\]}]/.test(line[q - 1])) q--;
512
+ const tag = q > a && line[q - 1] === '>' ? /<\/[A-Za-z][\w:.-]{0,40}>$/.exec(line.slice(a, q)) : null;
513
+ if (tag) b = q - tag[0].length;
514
+ else if (q > a && (line[q - 1] === '"' || line[q - 1] === "'" || line[q - 1] === '`')) {
515
+ q--;
516
+ for (;;) {
517
+ if (q - 1 <= a || (line[q - 1] !== 'n' && line[q - 1] !== 'r')) break;
518
+ let k = q - 1;
519
+ while (k > a && line[k - 1] === '\\') k--;
520
+ if (k === q - 1) break;
521
+ q = k;
522
+ }
523
+ b = q;
524
+ }
525
+ return { core: line.slice(a, b), start: off + a, end: off + b };
526
+ }
527
+
528
+ function classify(core) {
529
+ if (core === '') return BLANK;
530
+ if (B64_LONG_RE.test(core)) return LONG;
531
+ if (B64_SHORT_RE.test(core)) return SHORT;
532
+ if (KEY_END_LINE_RE.test(core)) return END;
533
+ if (ARMOR_HEADER_RE.test(core)) return ARMOR;
534
+ if (cutRun(core)) return CUT;
535
+ return OTHER;
536
+ }
537
+
538
+ /**
539
+ * The base64 run a cut or annotated key line starts with, or null: 16+ characters followed by a
540
+ * truncation mark or a delimiter (`…`, `...`, `[truncated]`, `<`, a backtick), or 40+ followed by
541
+ * anything (`<64> see above`). A shorter run followed by text is an identifier or a path starting
542
+ * a line of prose (`exportedArmoredPrivateKey = …`, round-3 P3-B), which stays.
543
+ */
544
+ function cutRun(core) {
545
+ const m = B64_RUN_RE.exec(core);
546
+ if (!m) return null;
547
+ return m[0].length >= 40 || CUT_MARK_RE.test(core.slice(m[0].length)) ? m[0] : null;
548
+ }
549
+
550
+ /**
551
+ * Where the key that starts with the BEGIN at [b, be) ends, or -1 when no key material follows it.
552
+ * Header lines and blank lines may come first; then 16+-character base64 lines (blank lines only
553
+ * between them), one shorter last line (and a PGP `=XXXX` checksum after it), and the END if it
554
+ * is there. One base64 line alone is a key only when it is 40+ characters, starts the way a key
555
+ * encoding starts (DER `MII…`, OpenSSH `b3BlbnNzaC1rZXktdjE…`) or follows armor headers: a path
556
+ * or an identifier of 16-39 characters under a header is prose (delta review P3-5). A complete
557
+ * block never gets here; the block pattern above takes it first.
558
+ */
559
+ function cutOffKeyEnd(text, b, be) {
560
+ // The rest of the BEGIN line: nothing, or base64 chunks, then a break that sets the depth.
561
+ let i = be;
562
+ while (i < text.length && /[A-Za-z0-9+/= \t]/.test(text[i])) i++;
563
+ const rest = text.slice(be, i).trim();
564
+ let u;
565
+ let next;
566
+ if (i >= text.length) return -1;
567
+ if (text[i] === '\n' || text[i] === '\r') {
568
+ u = 0;
569
+ next = text[i] === '\r' && text[i + 1] === '\n' ? i + 2 : i + 1;
570
+ } else if (text[i] === '\\') {
571
+ let j = i;
572
+ while (j < text.length && text[j] === '\\') j++;
573
+ const r = j - i;
574
+ if (text[j] !== 'n' && text[j] !== 'r') return -1;
575
+ u = r & -r;
576
+ if (r !== u) return -1; // a literal backslash on the BEGIN line: not a key line
577
+ ({ next } = nextLine(text, i, u));
578
+ } else return -1;
579
+ let longs = 0;
580
+ let longest = 0;
581
+ let first = '';
582
+ let end = -1;
583
+ if (rest) {
584
+ // A key pasted onto its BEGIN line: every chunk but the last is a full 16+ line. Words there
585
+ // are prose, even one of 16+ letters (v6.19.2 pre-tag defect review F9). A loop, not a spread:
586
+ // Math.max(...chunks) overflowed the stack past ~125k chunks (F2).
587
+ if (!B64_CHUNKS_RE.test(rest)) return -1;
588
+ const chunks = rest.split(/[ \t]+/);
589
+ for (let k = 0; k < chunks.length; k++) {
590
+ if (k < chunks.length - 1 && chunks[k].length < 16) return -1;
591
+ if (chunks[k].length > longest) longest = chunks[k].length;
592
+ }
593
+ if (longest < 16) return -1;
594
+ longs = 1;
595
+ first = chunks[0];
596
+ end = be + text.slice(be, i).trimEnd().length;
597
+ }
598
+ // The BEGIN line's prefix.
599
+ let ls = b;
600
+ while (ls > 0) {
601
+ const ch = text[ls - 1];
602
+ if (ch === '\n' || ch === '\r') break;
603
+ if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, ls - 1, 0), u)) break;
604
+ if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, ls - 1, 0), u)) break;
605
+ ls--;
606
+ }
607
+ // A 16+ base64 run glued to the BEGIN is the previous key's cut line, not a line prefix: taken
608
+ // for one, it stripped this key's identical body line and one copy of a repeated cut key went
609
+ // per pass (v6.19.2 pre-tag round-5 review F2).
610
+ let run = 0;
611
+ while (run < 16 && b - run > ls && /[A-Za-z0-9+/=]/.test(text[b - run - 1])) run++;
612
+ const shape = run >= 16 ? null : prefixShape(text.slice(ls, b));
613
+
614
+ let armors = 0;
615
+ let shorts = 0;
616
+ let pendingBlank = false;
617
+ while (next !== -1) {
618
+ const line = nextLine(text, next, u);
619
+ const { core, start, end: coreEnd } = lineCore(text, next, line.end, shape);
620
+ const kind = classify(core);
621
+ next = line.next;
622
+ if (kind === END) {
623
+ if (longs > 0) end = start + KEY_END_LINE_RE.exec(core)[0].length;
624
+ break;
625
+ }
626
+ if (kind === BLANK) {
627
+ if (shorts > 0) break;
628
+ pendingBlank = true;
629
+ continue;
630
+ }
631
+ if (longs === 0) {
632
+ if (kind === ARMOR && ++armors <= MAX_ARMOR_HEADERS) continue;
633
+ if (kind !== LONG && kind !== CUT) break;
634
+ }
635
+ if (kind === LONG && shorts === 0) {
636
+ if (longs++ === 0) first = core;
637
+ longest = Math.max(longest, core.length);
638
+ end = coreEnd;
639
+ pendingBlank = false;
640
+ continue;
641
+ }
642
+ // A cut line ends the key whatever came before it, blank lines included: a PGP or encrypted
643
+ // key has one before its body, and `head` of it in a JSON string ends in a cut line (round-4
644
+ // F1: checked after the blank-line stop, it was never read and the whole key was stored).
645
+ if (kind === CUT && shorts === 0) {
646
+ const run = cutRun(core);
647
+ if (longs++ === 0) first = run;
648
+ longest = Math.max(longest, run.length);
649
+ end = start + run.length;
650
+ break;
651
+ }
652
+ if (pendingBlank) break;
653
+ if (kind === SHORT && (shorts === 0 || (shorts === 1 && PGP_CRC_RE.test(core)))) {
654
+ shorts++;
655
+ end = coreEnd;
656
+ continue;
657
+ }
658
+ break;
659
+ }
660
+ if (longs === 0) return -1;
661
+ const magic = KEY_MAGIC_RE.test(first) || (text.slice(b, be).includes('PGP') && PGP_MAGIC_RE.test(first));
662
+ return longs >= 2 || armors > 0 || longest >= 40 || magic ? end : -1;
663
+ }
664
+
665
+ function scrubCutOffKeys(text) {
666
+ if (!text.includes('PRIVATE KEY')) return text;
667
+ KEY_BEGIN_RE.lastIndex = 0;
668
+ let out = '';
669
+ let last = 0;
670
+ let m;
671
+ while ((m = KEY_BEGIN_RE.exec(text))) {
672
+ const end = cutOffKeyEnd(text, m.index, m.index + m[0].length);
673
+ if (end === -1) continue;
674
+ out += text.slice(last, m.index) + PEM_MARK;
675
+ last = end;
676
+ KEY_BEGIN_RE.lastIndex = end;
677
+ }
678
+ return last === 0 ? text : out + text.slice(last);
679
+ }
680
+
681
+ /**
682
+ * The span [start, end) of the key tail that ends with the END at [e, ee), or null: whole base64
683
+ * lines of 16+ characters directly above it, with one shorter line before the END, or two when the
684
+ * one before the END is a PGP `=XXXX` checksum (delta review P2; two short words above an END are
685
+ * prose, round-3 P3-D). The span takes the END too, unless words precede the END on its line (a
686
+ * sentence naming it): then the lines above go and the sentence stays (F7). `floor` is the end of
687
+ * the previous END, so no line is read twice.
688
+ */
689
+ function keyTailSpan(text, e, ee, floor) {
690
+ let ls = e;
691
+ let u = 0;
692
+ while (ls > floor) {
693
+ const ch = text[ls - 1];
694
+ if (ch === '\n' || ch === '\r') break;
695
+ if (ch === 'n' || ch === 'r') {
696
+ const r = backslashesBefore(text, ls - 1, floor);
697
+ if (r > 0) {
698
+ u = r & -r;
699
+ break;
700
+ }
701
+ }
702
+ ls--;
703
+ }
704
+ let longs = 0;
705
+ let shorts = 0;
706
+ let crc = false;
707
+ let top = -1;
708
+ let topCore = '';
709
+ let longest = 0;
710
+ let bottom = -1;
711
+ const take = (core, start, end) => {
712
+ if (!accept(core, start)) return false;
713
+ if (bottom === -1) bottom = end;
714
+ return true;
715
+ };
716
+ const accept = (core, start) => {
717
+ const kind = classify(core);
718
+ if (kind === LONG) {
719
+ longs++;
720
+ top = start;
721
+ topCore = core;
722
+ if (core.length > longest) longest = core.length;
723
+ return true;
724
+ }
725
+ if (longs > 0 || kind !== SHORT) return false;
726
+ if (shorts === 0) {
727
+ shorts = 1;
728
+ crc = PGP_CRC_RE.test(core);
729
+ return true;
730
+ }
731
+ if (shorts === 1 && crc) {
732
+ shorts = 2;
733
+ return true;
734
+ }
735
+ return false;
736
+ };
737
+ // The END line's own prefix: a line prefix, the key's last base64 run glued to the END, or words.
738
+ // Words with spaces are a line prefix when the line above starts the same way (`web-1 | `, a
739
+ // syslog stamp, `> > `; v6.19.2 pre-tag delta review F2), and a sentence naming the END if not.
740
+ const prefix = text.slice(ls, e);
741
+ const trimmed = prefix.trim();
742
+ let shape = null;
743
+ let sentence = false;
744
+ let prefixed = false;
745
+ if (/^[ \t]*[A-Za-z0-9+/=]+$/.test(prefix)) {
746
+ const at = ls + prefix.indexOf(trimmed);
747
+ if (!take(trimmed, at, at + trimmed.length)) return null;
748
+ } else {
749
+ shape = prefixShape(prefix);
750
+ const above = shape && prevLine(text, ls, u, floor);
751
+ prefixed = Boolean(above && shape.test(text.slice(above.start, above.end)));
752
+ if (/\S\s+\S/.test(trimmed) && !prefixed) {
753
+ sentence = true;
754
+ shape = null;
755
+ }
756
+ }
757
+ // An END alone on its line (after nothing but a line prefix, before nothing but a quote,
758
+ // punctuation or a closing tag) is evidence enough for one line over it that has a digit, a `+`
759
+ // or `=` padding: `tail -n 2` of a key whose last line is 16-39 characters (delta F3). A 16+
760
+ // run of random base64 almost always has one; a camelCase identifier or a path has none
761
+ // (round-3 F3). An END in a sentence or in inline code followed by words is not alone.
762
+ CLEAN_AFTER_END_RE.lastIndex = ee;
763
+ const clean = !sentence && (trimmed === '' || prefixed) && CLEAN_AFTER_END_RE.test(text);
764
+ let cur = ls;
765
+ for (let line; (line = prevLine(text, cur, u, floor)); cur = line.start) {
766
+ const { core, start, end } = lineCore(text, line.start, line.end, shape);
767
+ if (!take(core, start, end)) break;
768
+ }
769
+ // Otherwise the same evidence a cut-off key needs: one base64 line alone is a key tail only when
770
+ // it is 40+ characters, starts like a key encoding or sits over a PGP checksum; an identifier of
771
+ // 16-39 characters over an END named in prose is not (F7).
772
+ if (longs === 0) return null;
773
+ const b64ish = /[0-9+]|=$/.test(topCore);
774
+ if (!(longs >= 2 || longest >= 40 || crc || (clean && b64ish) || KEY_MAGIC_RE.test(topCore))) return null;
775
+ return [top, sentence ? bottom : ee];
776
+ }
777
+
778
+ function scrubKeyTails(text) {
779
+ if (!text.includes('PRIVATE KEY')) return text;
780
+ KEY_END_RE.lastIndex = 0;
781
+ let out = '';
782
+ let last = 0;
783
+ let floor = 0;
784
+ let m;
785
+ while ((m = KEY_END_RE.exec(text))) {
786
+ const span = keyTailSpan(text, m.index, m.index + m[0].length, Math.max(floor, last));
787
+ if (span) {
788
+ out += text.slice(last, span[0]) + PEM_MARK;
789
+ last = span[1];
790
+ }
791
+ floor = m.index + m[0].length;
792
+ }
793
+ return last === 0 ? text : out + text.slice(last);
794
+ }
795
+
336
796
  /**
337
797
  * Scrub known secret patterns (API keys, tokens, credentials) from text.
338
798
  * Also strips user-marked `<private>...</private>` blocks first, so every
package/server.mjs CHANGED
@@ -125,6 +125,7 @@ import { AUTO_MERGE_THRESHOLD } from './lib/dedup-constants.mjs';
125
125
  import {
126
126
  insertDeferred,
127
127
  listOpenWithOrdinal,
128
+ openOrdinalOf,
128
129
  dropDeferred,
129
130
  formatDropReasonHint,
130
131
  resolveDeferredIds,
@@ -1221,8 +1222,7 @@ server.registerTool(
1221
1222
  });
1222
1223
  // Compute the ordinal for the freshly-inserted row so the response is
1223
1224
  // immediately actionable ("ok, I deferred this as item 1").
1224
- const open = listOpenWithOrdinal(db, project, 50);
1225
- const ord = open.find((o) => o.id === r.id)?.ordinal ?? null;
1225
+ const ord = openOrdinalOf(db, project, r.id);
1226
1226
  return {
1227
1227
  content: [
1228
1228
  {
package/utils.mjs CHANGED
@@ -289,13 +289,34 @@ const DESC_SCRUB_WINDOW = 4096;
289
289
  // before the first remaining `<private>` or private-key BEGIN, so no field shows a span whose end
290
290
  // is out of view. Before v6.19.0 an unclosed opener in the window (a Grep line
291
291
  // `notes.md:3:<private>bank pin 4412`) was shown as written (v6.19.0 pre-tag reviews P3-1, r3 P2-1).
292
- // The PEM half is case-sensitive, like the scrubber's PEM pattern.
293
- const PRIVATE_OPENER_RE = /<[Pp][Rr][Ii][Vv][Aa][Tt][Ee]>|-----BEGIN [A-Z0-9 ]*PRIVATE KEY/;
294
- function scrubTruncate(str, max) {
292
+ // A `</private>` or private-key END that comes first is a span whose START is out of view (a
293
+ // nested span, `tail key.pem`), so everything before it may be its inside: the field shows nothing
294
+ // (round-3 P3-1). The PEM half is case-sensitive, like the scrubber's PEM pattern.
295
+ const PRIVATE_MARK_RE = /<\/?[Pp][Rr][Ii][Vv][Aa][Tt][Ee]>|-----(?:BEGIN|END) [A-Z0-9 ]*PRIVATE KEY/;
296
+ // A window edge that cuts a token leaves a fragment shorter than its pattern needs (`ghp_` and 12
297
+ // of its 36 characters), and whitespace collapsing can bring it into view: the cut token is
298
+ // dropped (defect review P3-6, round-3 P3-4/P3-6). A head window with no whitespace keeps it: its
299
+ // start is intact and it is the only part shown. A tail window with no whitespace is all one cut token.
300
+ const isWs = (c) => c === ' ' || c === '\n' || c === '\t' || c === '\r' || /\s/.test(c);
301
+ function dropCutTokenAtEnd(win, next) {
302
+ if (next === undefined || isWs(next)) return win;
303
+ let i = win.length;
304
+ while (i > 0 && !isWs(win[i - 1])) i--;
305
+ return i > 0 ? win.slice(0, i) : win;
306
+ }
307
+ function dropCutTokenAtStart(win, prev) {
308
+ if (prev === undefined || isWs(prev)) return win;
309
+ let i = 0;
310
+ while (i < win.length && !isWs(win[i])) i++;
311
+ return win.slice(i);
312
+ }
313
+ function scrubTruncate(str, max, window = DESC_SCRUB_WINDOW) {
295
314
  if (typeof str !== 'string' || str === '') return truncate(str, max);
296
- let win = stripPrivate(str).slice(0, DESC_SCRUB_WINDOW);
297
- const at = win.search(PRIVATE_OPENER_RE);
298
- if (at >= 0) win = win.slice(0, at);
315
+ const stripped = stripPrivate(str);
316
+ let win = stripped.slice(0, window);
317
+ const mark = PRIVATE_MARK_RE.exec(win);
318
+ if (mark) win = mark[0][1] === '/' || mark[0].startsWith('-----END') ? '' : win.slice(0, mark.index);
319
+ else win = dropCutTokenAtEnd(win, stripped[window]);
299
320
  // The window edge can split a surrogate pair; truncate only guards a cut it makes itself.
300
321
  if (/[\uD800-\uDBFF]$/.test(win)) win = win.slice(0, -1);
301
322
  return truncate(_scrubSecrets(win), max);
@@ -314,6 +335,17 @@ function scrubTruncate(str, max) {
314
335
  // key with no END more than 4096 characters back).
315
336
  const PRIVATE_TAG_HINT_RE = /<\/?private>/i;
316
337
  const HEAD_ONLY_MAX = 60;
338
+ // The tail window is scrubbed with TAIL_CONTEXT characters before it, which are then dropped: a
339
+ // label cut by the window's edge (`pass|word: <value>`) is seen whole, so its value is not left
340
+ // unlabelled at the window's start, where whitespace collapsing can bring it into view (v6.19.1
341
+ // pre-tag review F1). A replacement inside the context moves the cut by its length change; the
342
+ // token the cut lands in is dropped either way.
343
+ const TAIL_CONTEXT = 256;
344
+ function scrubTailWindow(str, window) {
345
+ const scrubbed = _scrubSecrets(str.slice(-(window + TAIL_CONTEXT)));
346
+ return dropCutTokenAtStart(scrubbed.slice(TAIL_CONTEXT), scrubbed[TAIL_CONTEXT - 1]);
347
+ }
348
+ const oneSpace = (s) => s.replace(/\s+/g, ' ');
317
349
  // The early return needs the WHOLE output inside the head window: a long output whose first
318
350
  // 4096 characters collapse to a few (whitespace) still has a tail to show (v6.19.0 pre-tag
319
351
  // claims review F2).
@@ -321,11 +353,20 @@ function scrubTruncateEnds(str, max) {
321
353
  if (typeof str === 'string' && (str.includes('PRIVATE KEY') || PRIVATE_TAG_HINT_RE.test(str))) {
322
354
  return scrubTruncate(str, HEAD_ONLY_MAX);
323
355
  }
324
- const flat = scrubTruncate(str, DESC_SCRUB_WINDOW);
325
- const whole = typeof str !== 'string' || str.length <= DESC_SCRUB_WINDOW;
356
+ // Up to two windows long, one window covers the whole output: two overlapping windows showed
357
+ // the same text twice (delta review P3-4).
358
+ const window =
359
+ typeof str === 'string' && str.length <= 2 * DESC_SCRUB_WINDOW
360
+ ? 2 * DESC_SCRUB_WINDOW
361
+ : DESC_SCRUB_WINDOW;
362
+ // Whitespace runs are one space: a blank-line run no longer spends the budget (or, before the
363
+ // one-window rule, hid behind the window edge).
364
+ const flat = oneSpace(scrubTruncate(str, window, window));
365
+ const whole = typeof str !== 'string' || str.length <= window;
326
366
  if (whole && flat.length <= max) return flat;
327
367
  const tailLen = Math.floor(max / 2) - 1;
328
- const tailSrc = whole ? flat : normalizeInline(_scrubSecrets(str.slice(-DESC_SCRUB_WINDOW)));
368
+ const tailSrc = whole ? flat : oneSpace(normalizeInline(scrubTailWindow(str, window)));
369
+ if (tailSrc === '') return truncate(flat, max);
329
370
  // Drop a lone low surrogate the tail's cut may start on.
330
371
  const tail = tailSrc.slice(-tailLen).replace(/^[\uDC00-\uDFFF]/, '');
331
372
  // A head short enough to escape truncate's own "…" still gets one before the tail.