claude-mem-lite 6.19.1 → 6.19.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@
9
9
  "plugins": [
10
10
  {
11
11
  "name": "claude-mem-lite",
12
- "version": "6.19.1",
12
+ "version": "6.19.2",
13
13
  "source": "./",
14
14
  "homepage": "https://github.com/sdsrss/claude-mem-lite",
15
15
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.1",
3
+ "version": "6.19.2",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "author": {
6
6
  "name": "sdsrss"
package/hook-optimize.mjs CHANGED
@@ -32,7 +32,7 @@ import { normalizeScope, SCOPE_PROMPT_LEGEND, insertObservationRow } from './lib
32
32
  import { liveObsFilterSql } from './lib/inject-search-core.mjs';
33
33
  import { resolveRuntimeDir } from './lib/resolve-data-dir.mjs';
34
34
  import { MEMORY_INPUT_GUARD } from './lib/memory-input-guard.mjs';
35
- import { MANUAL_SESSION_ID_PREFIX } from './lib/provenance.mjs';
35
+ import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './lib/provenance.mjs';
36
36
 
37
37
  import { DAY_MS } from './lib/time-constants.mjs';
38
38
  // P1-14: same resolver as hook-shared.mjs — this was the second module that had never
@@ -583,12 +583,34 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
583
583
  // text over it, wide wrote back the pre-edit narrative it read before the call
584
584
  // (tests/verify-reenrich-race.test.mjs). It also stops two overlapping runs rewriting a
585
585
  // row twice.
586
+ //
587
+ // Narrow replaces title and narrative with model text, so an explicit save it rewrites (one
588
+ // whose save-time enrich failed) moves to the re-enrich writer's id, as a cluster-merge
589
+ // keeper does (D#146, D#138). Wide keeps the stored title and narrative (it fills the lesson
590
+ // and the side fields), so the row stays an explicit save. The writer's session row is
591
+ // best-effort.
592
+ let rewriteSessionId = null;
593
+ if (!isWide) {
594
+ const cur = db.prepare('SELECT memory_session_id FROM observations WHERE id = ?').get(cand.id);
595
+ if (cur?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
596
+ const enrichSessionId = writerSessionId('enrich-', cand.project);
597
+ try {
598
+ db.prepare(
599
+ `INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
600
+ VALUES (?, ?, ?, ?, ?, 'active')`,
601
+ ).run(enrichSessionId, enrichSessionId, cand.project, new Date().toISOString(), Date.now());
602
+ rewriteSessionId = enrichSessionId;
603
+ } catch (e) {
604
+ debugCatch(e, 'reenrich writer session');
605
+ }
606
+ }
607
+ }
586
608
  const res = db
587
609
  .prepare(
588
610
  `
589
611
  UPDATE observations SET type=?, title=?, narrative=?, concepts=?, facts=?,
590
612
  text=?, importance=?, lesson_learned=?, search_aliases=?, minhash_sig=?, optimized_at=?,
591
- scope=COALESCE(?, scope)
613
+ scope=COALESCE(?, scope), memory_session_id = COALESCE(?, memory_session_id)
592
614
  WHERE id = ? AND ${liveObsFilterSql('')} AND optimized_at IS NULL
593
615
  `,
594
616
  )
@@ -608,6 +630,7 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
608
630
  // scope, or emits an off-enum value, must never blank an existing label —
609
631
  // and THIS update stamps optimized_at, so the loss would be permanent.
610
632
  normalizeScope(parsed.scope),
633
+ rewriteSessionId,
611
634
  cand.id,
612
635
  );
613
636
  if (res.changes === 0) {
@@ -1315,15 +1338,15 @@ Return ONLY valid JSON:
1315
1338
  // The keeper now holds model text. Search reads authorship from memory_session_id, so an
1316
1339
  // explicit save's `manual-` id would mark it as one (D#138); it moves to the compression
1317
1340
  // writer's id. A machine-written keeper keeps its id, and so does the snapshot above, which
1318
- // is the original save. The writer's session row is best-effort: sdk_sessions refuses an id
1319
- // shaped like a uuid, which `compress-<project>` is for a 27-character project shaped
1320
- // xxxx-xxxx-xxxx-xxxxxxxxxxxx, and that refusal must not fail the merge (v6.19.1 pre-tag F5a).
1341
+ // is the original save. writerSessionId keeps the id out of the uuid shape sdk_sessions
1342
+ // refuses (D#147); the row stays best-effort, so no refusal can fail the merge (v6.19.1
1343
+ // pre-tag F5a).
1321
1344
  let rewriteSessionId = null;
1322
1345
  const keeperSession = db
1323
1346
  .prepare('SELECT memory_session_id FROM observations WHERE id = ?')
1324
1347
  .get(keeper.id);
1325
1348
  if (keeperSession?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
1326
- const compressSessionId = `compress-${keeper.project}`;
1349
+ const compressSessionId = writerSessionId('compress-', keeper.project);
1327
1350
  try {
1328
1351
  db.prepare(
1329
1352
  `INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
@@ -1580,7 +1603,7 @@ export async function executeSmartCompressCluster(db, observations, project) {
1580
1603
  ) {
1581
1604
  return null;
1582
1605
  }
1583
- const sessionId = `compress-${project}`;
1606
+ const sessionId = writerSessionId('compress-', project);
1584
1607
  const now = new Date();
1585
1608
  db.prepare(
1586
1609
  `INSERT OR IGNORE INTO sdk_sessions
package/lib/activity.mjs CHANGED
@@ -7,6 +7,7 @@
7
7
  import { sanitizeFtsQuery } from '../utils.mjs';
8
8
  import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
9
9
  import { saveObservation } from './save-observation.mjs';
10
+ import { writerSessionId } from './provenance.mjs';
10
11
  // Pure title-only builder: this query runs on the EVENTS table, which has no
11
12
  // lesson_learned column — the lesson-escape variant would be a SQL error here.
12
13
  import { buildNotLowSignalSql } from './low-signal-patterns.mjs';
@@ -200,7 +201,7 @@ export function promoteInsightEvents(
200
201
  lesson_learned: ev.body.slice(0, 500),
201
202
  now: new Date(ev.created_at_epoch),
202
203
  // Event bodies are machine-written (episode summaries), so the row is not an explicit save.
203
- sessionId: `promote-${ev.project}`,
204
+ sessionId: writerSessionId('promote-', ev.project),
204
205
  });
205
206
  mark.run(Date.now(), ev.id);
206
207
  return r;
@@ -17,6 +17,7 @@
17
17
 
18
18
  import { isoWeekKey, COMPRESSED_AUTO } from '../utils.mjs';
19
19
  import { scrubRecord } from './scrub-record.mjs';
20
+ import { writerSessionId } from './provenance.mjs';
20
21
 
21
22
  /**
22
23
  * Low-value compression candidates: importance<=1, never accessed, older than
@@ -88,7 +89,7 @@ export function compressGroup(db, proj, obs) {
88
89
  const dominantType = Object.entries(types).sort((a, b) => b[1] - a[1])[0][0];
89
90
  const title = `Weekly summary: ${obs.length} ${dominantType} observations`;
90
91
  const narrative = obs.map((o) => `- ${o.title || '(untitled)'}`).join('\n');
91
- const sessionId = `compress-${proj}`;
92
+ const sessionId = writerSessionId('compress-', proj);
92
93
 
93
94
  const sortedEpochs = obs.map((o) => o.created_at_epoch).sort((a, b) => a - b);
94
95
  const medianEpoch = sortedEpochs[Math.floor(sortedEpochs.length / 2)];
@@ -1,8 +1,8 @@
1
1
  // Which writer produced an observation, read back from its memory_session_id. saveObservation
2
2
  // writes every explicit save under MANUAL_SESSION_ID_PREFIX. The hook capture (`hook-`),
3
3
  // transcript import (`import-`), compression summaries and cluster-merge keepers (`compress-`),
4
- // promoted events (`promote-`) and rows imported from older stores (a bare session uuid) are
5
- // machine-written.
4
+ // rows narrow re-enrich rewrote (`enrich-`), promoted events (`promote-`) and rows imported from
5
+ // older stores (a bare session uuid) are machine-written.
6
6
  //
7
7
  // Search and get mark the machine-written side, because explicit saves are the large majority
8
8
  // of a typical store and a mark on nearly every line would carry no information. An unknown id
@@ -10,6 +10,20 @@
10
10
 
11
11
  export const MANUAL_SESSION_ID_PREFIX = 'manual-';
12
12
 
13
+ // sdk_sessions refuses a row whose two ids are equal and 36 characters long with dashes where a
14
+ // uuid has them (schema.mjs, sdk_sessions_id_mix_check_ai). A writer row uses `<prefix><project>`
15
+ // for both, which has that shape for one project length and dash layout (27 characters under
16
+ // `compress-`, 29 under `manual-`), and the refusal failed every write in such a project (D#147).
17
+ // That id gets one more character; every other id is unchanged. Counted in code points, as
18
+ // SQLite's length() and LIKE's `_` count them.
19
+ const UUID_SHAPED_RE = /^[\s\S]{8}-[\s\S]{4}-[\s\S]{4}-[\s\S]{4}-[\s\S]{12}$/u;
20
+
21
+ /** The session id a non-hook writer (`manual-`, `compress-`, `enrich-`, `promote-`) stores for a project. */
22
+ export function writerSessionId(prefix, project) {
23
+ const id = `${prefix}${project}`;
24
+ return UUID_SHAPED_RE.test(id) ? `${id}~` : id;
25
+ }
26
+
13
27
  const AUTO_MARK = '🤖';
14
28
  const AUTO_TEXT = 'auto-written, not an explicit save';
15
29
 
@@ -22,7 +22,7 @@ import { liveObsFilterSql } from './inject-search-core.mjs';
22
22
  // Imported, not injected: `allowStatuses` below is the POLICY this function exists to hold,
23
23
  // and a caller free to pass its own resolver could reinstate the one-way gate D#195 closed.
24
24
  import { resolveDeferredIds, closeDeferredItems } from './deferred-work.mjs';
25
- import { MANUAL_SESSION_ID_PREFIX } from './provenance.mjs';
25
+ import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './provenance.mjs';
26
26
 
27
27
  const DEDUP_WINDOW_MS = 5 * 60 * 1000;
28
28
  const DEDUP_RECENT_LIMIT = 50;
@@ -201,7 +201,7 @@ export function saveObservation(db, params) {
201
201
 
202
202
  // An explicit save by default; a caller storing machine-written text names its own writer
203
203
  // (lib/provenance.mjs reads authorship from this id).
204
- const sessionId = params.sessionId || `${MANUAL_SESSION_ID_PREFIX}${project}`;
204
+ const sessionId = params.sessionId || writerSessionId(MANUAL_SESSION_ID_PREFIX, project);
205
205
 
206
206
  // Ensure session exists (FK constraint). INSERT OR IGNORE makes this safe
207
207
  // under concurrent calls.
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.1",
3
+ "version": "6.19.2",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "claude-mem-lite",
9
- "version": "6.19.1",
9
+ "version": "6.19.2",
10
10
  "os": [
11
11
  "darwin",
12
12
  "linux",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.1",
3
+ "version": "6.19.2",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "type": "module",
6
6
  "packageManager": "npm@10.9.2",
package/secret-scrub.mjs CHANGED
@@ -200,34 +200,19 @@ export const SECRET_PATTERNS = [
200
200
  // alternation missed; the block delimiters make FP impossible.
201
201
  // The body stops at the next `-----BEGIN ` (D#130): with `[\s\S]*?` every header with no END
202
202
  // scanned to the end of the text, on each of scrubSecrets' passes — quadratic, 8.3 s on 500k
203
- // chars. A block whose END is missing is left to the next pattern.
203
+ // chars. A block whose END is missing is left to the next pattern. Nor does it cross a mark this
204
+ // scrubber wrote: on a later pass, a BEGIN that a scrubbed key used to block reached a far END
205
+ // and erased the prose between (v6.19.2 pre-tag defect review F5).
204
206
  [
205
- /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
207
+ /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY|\*\*\*PEM_KEY\*\*\*)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
206
208
  '***PEM_KEY***',
207
209
  ],
208
- // A cut-off key (`head id_rsa`, a tool output cut mid-key): the header and the WHOLE lines of
209
- // base64 (or RFC 1421 headers) that follow it. Stored whole through v6.19.0 when no later key
210
- // header followed. Whole lines only, so prose naming a header mid-sentence keeps its text
211
- // (v6.19.0 round-3 P3-3: ending the block at the next key header erased the prose between two).
212
- // A body line starts with 16+ base64 characters, and one shorter whole line may follow the last
213
- // of them (a key's last line): a line of one word or number is prose (v6.19.1 pre-tag review F3),
214
- // so the body needs a long line, and blank lines count only between long ones. A long line need
215
- // not be whole, so a key cut mid-line loses the cut line too. A line break may be JSON-escaped
216
- // (`\n` as two characters) and a quote may end the last line (v6.19.1 claims review F1). A header
217
- // value may hold a backslash that does not start an escaped break (a Windows path), and it does
218
- // not share its whitespace with a second quantifier, which was quadratic (delta review P1).
219
- [
220
- /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:[ \t]*(?:\r?\n|\\r\\n|\\n)(?=[ \t]*(?:\r?\n|\\r\\n|\\n)))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*(?:[A-Za-z0-9+/=]{16,}|(?:Proc-Type|DEK-Info|Version|Comment|Hash|Charset):(?:[^\r\n\\]|\\(?![rn]))*(?=\r?\n|\\[rn]|["']|$)))(?:(?:[ \t]*(?:\r?\n|\\r\\n|\\n)(?=[ \t]*(?:\r?\n|\\r\\n|\\n)))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*(?:[A-Za-z0-9+/=]{16,}|(?:Proc-Type|DEK-Info|Version|Comment|Hash|Charset):(?:[^\r\n\\]|\\(?![rn]))*(?=\r?\n|\\[rn]|["']|$))))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*[A-Za-z0-9+/=]{1,15}[ \t]*(?=\r?\n|\\[rn]|["']|$))?/g,
221
- '***PEM_KEY***',
222
- ],
223
- // The other end (`tail key.pem`): whole base64 lines ending in a private-key END, the same line
224
- // rule as above but with two shorter lines allowed before the END (an armored PGP key's last data
225
- // line and its `=XXXX` checksum; delta review P2). A line may start after a quote. A run of lines
226
- // that does not end there is consumed and returned unchanged, so no line starts a second scan.
227
- [
228
- /(?:(?<![^\n])|(?<=\\n|["']))(?:[ \t]*[A-Za-z0-9+/=]{16,}[ \t]*(?:\r?\n|\\r\\n|\\n))+(?:[ \t]*[A-Za-z0-9+/=]{1,15}[ \t]*(?:\r?\n|\\r\\n|\\n)){0,2}(?:-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----)?/g,
229
- (m) => (m.endsWith('-----') ? '***PEM_KEY***' : m),
230
- ],
210
+ // A cut-off key (`head id_rsa`, a tool output cut mid-key) and a headerless key tail (`tail
211
+ // key.pem`). These are line scanners, not patterns (D#145): three rounds of review each found a
212
+ // line shape the regex versions misread, and each fix to one shape erased prose in another.
213
+ // See scrubCutOffKeys and scrubKeyTails below for the line rules.
214
+ [{ [Symbol.replace]: (text) => scrubCutOffKeys(text) }, null],
215
+ [{ [Symbol.replace]: (text) => scrubKeyTails(text) }, null],
231
216
  // Long hex strings in credential assignments (e.g. SECRET_KEY=abc123def456...).
232
217
  // `hash` deliberately excluded: `hash: <40hex>` / `hash=<md5>` are git SHAs and
233
218
  // checksums (real, preserved data in this hash-heavy repo), not credentials.
@@ -351,6 +336,463 @@ export const SECRET_PATTERNS = [
351
336
  // bare-token pattern here: don't — anchor it to a provider prefix instead.
352
337
  ];
353
338
 
339
+ // ─── Cut-off private keys (D#145) ──────────────────────────────────────────
340
+ // A key's text reaches the scrubber in many line shapes: plain LF/CRLF/CR lines, JSON-escaped
341
+ // breaks (`\n` as two characters, or `\\n` when serialised twice), lines carrying a prefix (the
342
+ // Read tool's ` 2\t` or `2→`, grep's `id_rsa:`, a `> ` quote, a diff `-`), and lines inside a
343
+ // quoted string. The scanners read the text as lines of ONE shape per key, decided at the key:
344
+ // - the break after the BEGIN line (or before the END line) says whether breaks are real or
345
+ // escaped, and at which depth; in real-break text a backslash is never a break, so a Windows
346
+ // path in a `Comment:` value no longer ends the header (v6.19.1 round-3 P3-E);
347
+ // - the text before the BEGIN (or END) on its line is the line prefix, and every other line
348
+ // loses a prefix of the same shape (digits may differ, `:` and `-` swap for grep context)
349
+ // before it is judged;
350
+ // - lines end at breaks only. A header value that runs on into the next JSON fields is still
351
+ // one header line, so it erases nothing without a base64 line under it (round-3 P3-C).
352
+ // A body line is WHOLE base64: 16+ characters, or 1-15 for the last line. So a word under a key
353
+ // body keeps its text (round-3 P3-A: `Don't` lost `Don`), as does a path or an identifier that
354
+ // starts the next line (P3-B), and a body needs one long line, so a header followed by words is
355
+ // not a key (F3). A string's closing quote, a backtick or a closing tag after the line is not part
356
+ // of it. A line whose base64 run is followed by something else is a cut or annotated key line
357
+ // when the run is 40+ characters or a truncation mark or delimiter follows it (`…`, `...`,
358
+ // `[truncated]`, `<`): its base64 goes and the rest stays (delta review P3-2; v6.19.2 pre-tag
359
+ // defect review F4). A shorter run followed by words starts a line of prose.
360
+
361
+ const KEY_BEGIN_RE = /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
362
+ const KEY_END_RE = /-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
363
+ const KEY_END_LINE_RE = /^-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/;
364
+ const B64_LONG_RE = /^[A-Za-z0-9+/=]{16,}$/;
365
+ const B64_SHORT_RE = /^[A-Za-z0-9+/=]{1,15}$/;
366
+ // Space-separated chunks, allowed only on the BEGIN line itself (a key pasted onto one line).
367
+ const B64_CHUNKS_RE = /^[A-Za-z0-9+/=]+(?:[ \t]+[A-Za-z0-9+/=]+)*$/;
368
+ const B64_RUN_RE = /^[A-Za-z0-9+/=]{16,}/;
369
+ // What may follow a key line that was cut or annotated: a truncation mark or a delimiter. A quote
370
+ // ends the string a key sits in (`…', 'rc': 0}`, `…","stderr":…`): the scanners do not track which
371
+ // quote opened a string, since no character before a quote tells an opening quote from prose
372
+ // (`Here's`, `Run 'head id_rsa'`, `b'…'`; v6.19.2 pre-tag reviews, delta F1 and round-3 F1/F2).
373
+ // Two escapes deep a string ends at `\"` (round-4 F2). Five dashes are the next key's marker glued
374
+ // to a cut line (`for i in …; do head -c 80 k; done`).
375
+ const CUT_MARK_RE = /^(?: ?…| ?\.\.\.| ?\[| ?<|[`"']|\\+["']|-----)/;
376
+ const PGP_CRC_RE = /^=[A-Za-z0-9+/]{4}$/;
377
+ // How a key's base64 starts: a DER SEQUENCE (PKCS#1, PKCS#8, SEC1) or OpenSSH's `openssh-key-v1`;
378
+ // under a PGP header, a secret-key packet in the old or new format (`lQ…`, `xc…`/`xV…`).
379
+ const KEY_MAGIC_RE = /^(?:MII|MIG|MC4C|MHcC|b3BlbnNzaC1rZXktdjE)/;
380
+ const PGP_MAGIC_RE = /^(?:lQ|x[cV])/;
381
+ // RFC 1421 / RFC 4880 armor headers, before the body. Named, not any `Word:`: a `Note:` line is
382
+ // prose, and taking it for a header made the lone base64 line under it a key.
383
+ // RFC 1421's full set and tool-written `X-` headers count too (v6.19.2 pre-tag delta review F4:
384
+ // `Content-Domain` or `X-Custom` before the body stored the whole key).
385
+ const ARMOR_HEADER_RE =
386
+ /^(?:Proc-Type|DEK-Info|Content-Domain|Originator-ID-(?:Asymmetric|Symmetric)|Originator-Certificate|Issuer-Certificate|MIC-Info|Key-Info|Recipient-ID-(?:Asymmetric|Symmetric)|CRL|Version|Comment|Hash|Charset|MessageID|X-[A-Za-z0-9-]+)[ \t]*:/i;
387
+ const MAX_ARMOR_HEADERS = 16;
388
+ // A JS/Python string split across source lines: `…\n" +` then `"…` on the next line. Not a comma:
389
+ // `'…\n',` then `'…'` is the next element of a list, and its first word is not the key's last line.
390
+ // One whitespace quantifier on each side of the `+`: `[ \t]*\+?[ \t]*` split a run between two
391
+ // and was quadratic when no break followed (v6.19.2 pre-tag defect review F1: 16-19 s at 200k).
392
+ const CONCAT_AFTER_RE = /["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']/y;
393
+ const CONCAT_BEFORE_RE = /(?:\\r)?\\n["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']$/;
394
+ const PEM_MARK = '***PEM_KEY***';
395
+ // After an END that ends its line: a closing quote or punctuation, then a break or the text end.
396
+ // Closing tags may be nested (`END</code></pre>`; round-4 F3), and a string's closing quote, escaped
397
+ // or not, ends the line whatever follows it: `…END-----","stderr":""` is `tail -n 2` in a tool
398
+ // result (round-5 F3).
399
+ const CLEAN_AFTER_END_RE = /(?:[ \t`,;)\]}]|<\/[A-Za-z][\w:.-]{0,40}>)*(?:\r|\n|\\+[nr]|\\*["']|$)/y;
400
+
401
+ const BLANK = 0;
402
+ const LONG = 1;
403
+ const SHORT = 2;
404
+ const END = 3;
405
+ const ARMOR = 4;
406
+ const CUT = 5;
407
+ const OTHER = 6;
408
+
409
+ function backslashesBefore(text, i, floor) {
410
+ let j = i;
411
+ while (j > floor && text[j - 1] === '\\') j--;
412
+ return i - j;
413
+ }
414
+
415
+ // Break depth `u`: 0 = real breaks only, 1 = `\n`, 2 = `\\n`, … A run of r backslashes then `n`
416
+ // is a break at depth u when r % 2u === u (the backslashes before it are escaped backslashes).
417
+ const isEscBreak = (r, u) => u > 0 && r % (2 * u) === u;
418
+ // An unescaped quote at depth u ends the string: r % 2u < u.
419
+ const isStringEnd = (r, u) => u > 0 && r % (2 * u) < u;
420
+
421
+ /** The line starting at s: its end, and where the next line starts (-1 at the end of the text). */
422
+ function nextLine(text, s, u) {
423
+ const n = text.length;
424
+ let i = s;
425
+ while (i < n) {
426
+ const ch = text[i];
427
+ if (ch === '\n') return { end: i, next: i + 1 };
428
+ if (ch === '\r') return { end: i, next: text[i + 1] === '\n' ? i + 2 : i + 1 };
429
+ if (u > 0 && ch === '\\') {
430
+ let j = i;
431
+ while (j < n && text[j] === '\\') j++;
432
+ const r = j - i;
433
+ const c = text[j];
434
+ if ((c === 'n' || c === 'r') && isEscBreak(r, u)) {
435
+ let next = j + 1;
436
+ if (c === 'r' && text.startsWith('\\'.repeat(u) + 'n', next)) next += u + 1;
437
+ if (u === 1) {
438
+ CONCAT_AFTER_RE.lastIndex = next;
439
+ const m = CONCAT_AFTER_RE.exec(text);
440
+ if (m) next += m[0].length;
441
+ }
442
+ return { end: j - u, next };
443
+ }
444
+ i = j + 1;
445
+ continue;
446
+ }
447
+ i++;
448
+ }
449
+ return { end: n, next: -1 };
450
+ }
451
+
452
+ /** The line that ends at the break before `ls`, or null when `ls` starts the text or string. */
453
+ function prevLine(text, ls, u, floor) {
454
+ if (ls - 1 < floor) return null;
455
+ const c = text[ls - 1];
456
+ let bs = -1;
457
+ if (c === '\n') bs = ls - 2 >= floor && text[ls - 2] === '\r' ? ls - 2 : ls - 1;
458
+ else if (c === '\r') bs = ls - 1;
459
+ else if ((c === 'n' || c === 'r') && isEscBreak(backslashesBefore(text, ls - 1, floor), u)) {
460
+ bs = ls - 1 - u;
461
+ const r2 = bs - 1 >= floor && text[bs - 1] === 'r' ? backslashesBefore(text, bs - 1, floor) : 0;
462
+ if (c === 'n' && isEscBreak(r2, u)) bs -= u + 1;
463
+ } else if (u === 1 && (c === '"' || c === "'")) {
464
+ const m = CONCAT_BEFORE_RE.exec(text.slice(Math.max(floor, ls - 64), ls));
465
+ if (m) bs = ls - m[0].length;
466
+ }
467
+ if (bs === -1) return null;
468
+ let i = bs - 1;
469
+ while (i >= floor) {
470
+ const ch = text[i];
471
+ if (ch === '\n' || ch === '\r') break;
472
+ if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, i, floor), u)) break;
473
+ if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, i, floor), u)) break;
474
+ i--;
475
+ }
476
+ return { start: i + 1, end: bs };
477
+ }
478
+
479
+ /** A regex for the line prefix `prefix` has, with digits free and grep's `:`/`-` interchangeable. */
480
+ function prefixShape(prefix) {
481
+ const body = prefix.replace(/^[ \t]+/, '');
482
+ if (!body || body.length > 256) return null;
483
+ if (body === '-' || body === '+') return /^[ \t]*[-+ ]/;
484
+ const src = body
485
+ .replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')
486
+ .replace(/\d+/g, '\\d+')
487
+ .replace(/[:-]/g, '[:-]');
488
+ return new RegExp(`^[ \\t]*${src}`);
489
+ }
490
+
491
+ /** The judged part of line [s, e): no prefix of the key's shape, no padding, no string quotes. */
492
+ function lineCore(text, s, e, shape) {
493
+ let line = text.slice(s, e);
494
+ let off = s;
495
+ if (shape) {
496
+ const m = shape.exec(line);
497
+ if (m) {
498
+ line = line.slice(m[0].length);
499
+ off += m[0].length;
500
+ }
501
+ }
502
+ let a = 0;
503
+ let b = line.length;
504
+ while (a < b && (line[a] === ' ' || line[a] === '\t')) a++;
505
+ while (b > a && (line[b - 1] === ' ' || line[b - 1] === '\t')) b--;
506
+ if (a < b && (line[a] === '"' || line[a] === "'" || line[a] === '`')) a++;
507
+ // A closing quote or backtick with what may follow it (`",`, `" +`, `')`), or a closing tag
508
+ // (`</key>`), and escaped breaks before it. Read from the end: an unanchored
509
+ // `(?:\\+[rn])*["']…$` retried every start in a backslash run.
510
+ let q = b;
511
+ while (q > a && /[ \t+,;)\]}]/.test(line[q - 1])) q--;
512
+ const tag = q > a && line[q - 1] === '>' ? /<\/[A-Za-z][\w:.-]{0,40}>$/.exec(line.slice(a, q)) : null;
513
+ if (tag) b = q - tag[0].length;
514
+ else if (q > a && (line[q - 1] === '"' || line[q - 1] === "'" || line[q - 1] === '`')) {
515
+ q--;
516
+ for (;;) {
517
+ if (q - 1 <= a || (line[q - 1] !== 'n' && line[q - 1] !== 'r')) break;
518
+ let k = q - 1;
519
+ while (k > a && line[k - 1] === '\\') k--;
520
+ if (k === q - 1) break;
521
+ q = k;
522
+ }
523
+ b = q;
524
+ }
525
+ return { core: line.slice(a, b), start: off + a, end: off + b };
526
+ }
527
+
528
+ function classify(core) {
529
+ if (core === '') return BLANK;
530
+ if (B64_LONG_RE.test(core)) return LONG;
531
+ if (B64_SHORT_RE.test(core)) return SHORT;
532
+ if (KEY_END_LINE_RE.test(core)) return END;
533
+ if (ARMOR_HEADER_RE.test(core)) return ARMOR;
534
+ if (cutRun(core)) return CUT;
535
+ return OTHER;
536
+ }
537
+
538
+ /**
539
+ * The base64 run a cut or annotated key line starts with, or null: 16+ characters followed by a
540
+ * truncation mark or a delimiter (`…`, `...`, `[truncated]`, `<`, a backtick), or 40+ followed by
541
+ * anything (`<64> see above`). A shorter run followed by text is an identifier or a path starting
542
+ * a line of prose (`exportedArmoredPrivateKey = …`, round-3 P3-B), which stays.
543
+ */
544
+ function cutRun(core) {
545
+ const m = B64_RUN_RE.exec(core);
546
+ if (!m) return null;
547
+ return m[0].length >= 40 || CUT_MARK_RE.test(core.slice(m[0].length)) ? m[0] : null;
548
+ }
549
+
550
+ /**
551
+ * Where the key that starts with the BEGIN at [b, be) ends, or -1 when no key material follows it.
552
+ * Header lines and blank lines may come first; then 16+-character base64 lines (blank lines only
553
+ * between them), one shorter last line (and a PGP `=XXXX` checksum after it), and the END if it
554
+ * is there. One base64 line alone is a key only when it is 40+ characters, starts the way a key
555
+ * encoding starts (DER `MII…`, OpenSSH `b3BlbnNzaC1rZXktdjE…`) or follows armor headers: a path
556
+ * or an identifier of 16-39 characters under a header is prose (delta review P3-5). A complete
557
+ * block never gets here; the block pattern above takes it first.
558
+ */
559
+ function cutOffKeyEnd(text, b, be) {
560
+ // The rest of the BEGIN line: nothing, or base64 chunks, then a break that sets the depth.
561
+ let i = be;
562
+ while (i < text.length && /[A-Za-z0-9+/= \t]/.test(text[i])) i++;
563
+ const rest = text.slice(be, i).trim();
564
+ let u;
565
+ let next;
566
+ if (i >= text.length) return -1;
567
+ if (text[i] === '\n' || text[i] === '\r') {
568
+ u = 0;
569
+ next = text[i] === '\r' && text[i + 1] === '\n' ? i + 2 : i + 1;
570
+ } else if (text[i] === '\\') {
571
+ let j = i;
572
+ while (j < text.length && text[j] === '\\') j++;
573
+ const r = j - i;
574
+ if (text[j] !== 'n' && text[j] !== 'r') return -1;
575
+ u = r & -r;
576
+ if (r !== u) return -1; // a literal backslash on the BEGIN line: not a key line
577
+ ({ next } = nextLine(text, i, u));
578
+ } else return -1;
579
+ let longs = 0;
580
+ let longest = 0;
581
+ let first = '';
582
+ let end = -1;
583
+ if (rest) {
584
+ // A key pasted onto its BEGIN line: every chunk but the last is a full 16+ line. Words there
585
+ // are prose, even one of 16+ letters (v6.19.2 pre-tag defect review F9). A loop, not a spread:
586
+ // Math.max(...chunks) overflowed the stack past ~125k chunks (F2).
587
+ if (!B64_CHUNKS_RE.test(rest)) return -1;
588
+ const chunks = rest.split(/[ \t]+/);
589
+ for (let k = 0; k < chunks.length; k++) {
590
+ if (k < chunks.length - 1 && chunks[k].length < 16) return -1;
591
+ if (chunks[k].length > longest) longest = chunks[k].length;
592
+ }
593
+ if (longest < 16) return -1;
594
+ longs = 1;
595
+ first = chunks[0];
596
+ end = be + text.slice(be, i).trimEnd().length;
597
+ }
598
+ // The BEGIN line's prefix.
599
+ let ls = b;
600
+ while (ls > 0) {
601
+ const ch = text[ls - 1];
602
+ if (ch === '\n' || ch === '\r') break;
603
+ if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, ls - 1, 0), u)) break;
604
+ if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, ls - 1, 0), u)) break;
605
+ ls--;
606
+ }
607
+ // A 16+ base64 run glued to the BEGIN is the previous key's cut line, not a line prefix: taken
608
+ // for one, it stripped this key's identical body line and one copy of a repeated cut key went
609
+ // per pass (v6.19.2 pre-tag round-5 review F2).
610
+ let run = 0;
611
+ while (run < 16 && b - run > ls && /[A-Za-z0-9+/=]/.test(text[b - run - 1])) run++;
612
+ const shape = run >= 16 ? null : prefixShape(text.slice(ls, b));
613
+
614
+ let armors = 0;
615
+ let shorts = 0;
616
+ let pendingBlank = false;
617
+ while (next !== -1) {
618
+ const line = nextLine(text, next, u);
619
+ const { core, start, end: coreEnd } = lineCore(text, next, line.end, shape);
620
+ const kind = classify(core);
621
+ next = line.next;
622
+ if (kind === END) {
623
+ if (longs > 0) end = start + KEY_END_LINE_RE.exec(core)[0].length;
624
+ break;
625
+ }
626
+ if (kind === BLANK) {
627
+ if (shorts > 0) break;
628
+ pendingBlank = true;
629
+ continue;
630
+ }
631
+ if (longs === 0) {
632
+ if (kind === ARMOR && ++armors <= MAX_ARMOR_HEADERS) continue;
633
+ if (kind !== LONG && kind !== CUT) break;
634
+ }
635
+ if (kind === LONG && shorts === 0) {
636
+ if (longs++ === 0) first = core;
637
+ longest = Math.max(longest, core.length);
638
+ end = coreEnd;
639
+ pendingBlank = false;
640
+ continue;
641
+ }
642
+ // A cut line ends the key whatever came before it, blank lines included: a PGP or encrypted
643
+ // key has one before its body, and `head` of it in a JSON string ends in a cut line (round-4
644
+ // F1: checked after the blank-line stop, it was never read and the whole key was stored).
645
+ if (kind === CUT && shorts === 0) {
646
+ const run = cutRun(core);
647
+ if (longs++ === 0) first = run;
648
+ longest = Math.max(longest, run.length);
649
+ end = start + run.length;
650
+ break;
651
+ }
652
+ if (pendingBlank) break;
653
+ if (kind === SHORT && (shorts === 0 || (shorts === 1 && PGP_CRC_RE.test(core)))) {
654
+ shorts++;
655
+ end = coreEnd;
656
+ continue;
657
+ }
658
+ break;
659
+ }
660
+ if (longs === 0) return -1;
661
+ const magic = KEY_MAGIC_RE.test(first) || (text.slice(b, be).includes('PGP') && PGP_MAGIC_RE.test(first));
662
+ return longs >= 2 || armors > 0 || longest >= 40 || magic ? end : -1;
663
+ }
664
+
665
+ function scrubCutOffKeys(text) {
666
+ if (!text.includes('PRIVATE KEY')) return text;
667
+ KEY_BEGIN_RE.lastIndex = 0;
668
+ let out = '';
669
+ let last = 0;
670
+ let m;
671
+ while ((m = KEY_BEGIN_RE.exec(text))) {
672
+ const end = cutOffKeyEnd(text, m.index, m.index + m[0].length);
673
+ if (end === -1) continue;
674
+ out += text.slice(last, m.index) + PEM_MARK;
675
+ last = end;
676
+ KEY_BEGIN_RE.lastIndex = end;
677
+ }
678
+ return last === 0 ? text : out + text.slice(last);
679
+ }
680
+
681
+ /**
682
+ * The span [start, end) of the key tail that ends with the END at [e, ee), or null: whole base64
683
+ * lines of 16+ characters directly above it, with one shorter line before the END, or two when the
684
+ * one before the END is a PGP `=XXXX` checksum (delta review P2; two short words above an END are
685
+ * prose, round-3 P3-D). The span takes the END too, unless words precede the END on its line (a
686
+ * sentence naming it): then the lines above go and the sentence stays (F7). `floor` is the end of
687
+ * the previous END, so no line is read twice.
688
+ */
689
+ function keyTailSpan(text, e, ee, floor) {
690
+ let ls = e;
691
+ let u = 0;
692
+ while (ls > floor) {
693
+ const ch = text[ls - 1];
694
+ if (ch === '\n' || ch === '\r') break;
695
+ if (ch === 'n' || ch === 'r') {
696
+ const r = backslashesBefore(text, ls - 1, floor);
697
+ if (r > 0) {
698
+ u = r & -r;
699
+ break;
700
+ }
701
+ }
702
+ ls--;
703
+ }
704
+ let longs = 0;
705
+ let shorts = 0;
706
+ let crc = false;
707
+ let top = -1;
708
+ let topCore = '';
709
+ let longest = 0;
710
+ let bottom = -1;
711
+ const take = (core, start, end) => {
712
+ if (!accept(core, start)) return false;
713
+ if (bottom === -1) bottom = end;
714
+ return true;
715
+ };
716
+ const accept = (core, start) => {
717
+ const kind = classify(core);
718
+ if (kind === LONG) {
719
+ longs++;
720
+ top = start;
721
+ topCore = core;
722
+ if (core.length > longest) longest = core.length;
723
+ return true;
724
+ }
725
+ if (longs > 0 || kind !== SHORT) return false;
726
+ if (shorts === 0) {
727
+ shorts = 1;
728
+ crc = PGP_CRC_RE.test(core);
729
+ return true;
730
+ }
731
+ if (shorts === 1 && crc) {
732
+ shorts = 2;
733
+ return true;
734
+ }
735
+ return false;
736
+ };
737
+ // The END line's own prefix: a line prefix, the key's last base64 run glued to the END, or words.
738
+ // Words with spaces are a line prefix when the line above starts the same way (`web-1 | `, a
739
+ // syslog stamp, `> > `; v6.19.2 pre-tag delta review F2), and a sentence naming the END if not.
740
+ const prefix = text.slice(ls, e);
741
+ const trimmed = prefix.trim();
742
+ let shape = null;
743
+ let sentence = false;
744
+ let prefixed = false;
745
+ if (/^[ \t]*[A-Za-z0-9+/=]+$/.test(prefix)) {
746
+ const at = ls + prefix.indexOf(trimmed);
747
+ if (!take(trimmed, at, at + trimmed.length)) return null;
748
+ } else {
749
+ shape = prefixShape(prefix);
750
+ const above = shape && prevLine(text, ls, u, floor);
751
+ prefixed = Boolean(above && shape.test(text.slice(above.start, above.end)));
752
+ if (/\S\s+\S/.test(trimmed) && !prefixed) {
753
+ sentence = true;
754
+ shape = null;
755
+ }
756
+ }
757
+ // An END alone on its line (after nothing but a line prefix, before nothing but a quote,
758
+ // punctuation or a closing tag) is evidence enough for one line over it that has a digit, a `+`
759
+ // or `=` padding: `tail -n 2` of a key whose last line is 16-39 characters (delta F3). A 16+
760
+ // run of random base64 almost always has one; a camelCase identifier or a path has none
761
+ // (round-3 F3). An END in a sentence or in inline code followed by words is not alone.
762
+ CLEAN_AFTER_END_RE.lastIndex = ee;
763
+ const clean = !sentence && (trimmed === '' || prefixed) && CLEAN_AFTER_END_RE.test(text);
764
+ let cur = ls;
765
+ for (let line; (line = prevLine(text, cur, u, floor)); cur = line.start) {
766
+ const { core, start, end } = lineCore(text, line.start, line.end, shape);
767
+ if (!take(core, start, end)) break;
768
+ }
769
+ // Otherwise the same evidence a cut-off key needs: one base64 line alone is a key tail only when
770
+ // it is 40+ characters, starts like a key encoding or sits over a PGP checksum; an identifier of
771
+ // 16-39 characters over an END named in prose is not (F7).
772
+ if (longs === 0) return null;
773
+ const b64ish = /[0-9+]|=$/.test(topCore);
774
+ if (!(longs >= 2 || longest >= 40 || crc || (clean && b64ish) || KEY_MAGIC_RE.test(topCore))) return null;
775
+ return [top, sentence ? bottom : ee];
776
+ }
777
+
778
+ function scrubKeyTails(text) {
779
+ if (!text.includes('PRIVATE KEY')) return text;
780
+ KEY_END_RE.lastIndex = 0;
781
+ let out = '';
782
+ let last = 0;
783
+ let floor = 0;
784
+ let m;
785
+ while ((m = KEY_END_RE.exec(text))) {
786
+ const span = keyTailSpan(text, m.index, m.index + m[0].length, Math.max(floor, last));
787
+ if (span) {
788
+ out += text.slice(last, span[0]) + PEM_MARK;
789
+ last = span[1];
790
+ }
791
+ floor = m.index + m[0].length;
792
+ }
793
+ return last === 0 ? text : out + text.slice(last);
794
+ }
795
+
354
796
  /**
355
797
  * Scrub known secret patterns (API keys, tokens, credentials) from text.
356
798
  * Also strips user-marked `<private>...</private>` blocks first, so every