claude-mem-lite 6.19.1 → 6.19.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@
9
9
  "plugins": [
10
10
  {
11
11
  "name": "claude-mem-lite",
12
- "version": "6.19.1",
12
+ "version": "6.19.3",
13
13
  "source": "./",
14
14
  "homepage": "https://github.com/sdsrss/claude-mem-lite",
15
15
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.1",
3
+ "version": "6.19.3",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "author": {
6
6
  "name": "sdsrss"
package/hook-optimize.mjs CHANGED
@@ -32,7 +32,7 @@ import { normalizeScope, SCOPE_PROMPT_LEGEND, insertObservationRow } from './lib
32
32
  import { liveObsFilterSql } from './lib/inject-search-core.mjs';
33
33
  import { resolveRuntimeDir } from './lib/resolve-data-dir.mjs';
34
34
  import { MEMORY_INPUT_GUARD } from './lib/memory-input-guard.mjs';
35
- import { MANUAL_SESSION_ID_PREFIX } from './lib/provenance.mjs';
35
+ import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './lib/provenance.mjs';
36
36
 
37
37
  import { DAY_MS } from './lib/time-constants.mjs';
38
38
  // P1-14: same resolver as hook-shared.mjs — this was the second module that had never
@@ -583,12 +583,34 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
583
583
  // text over it, wide wrote back the pre-edit narrative it read before the call
584
584
  // (tests/verify-reenrich-race.test.mjs). It also stops two overlapping runs rewriting a
585
585
  // row twice.
586
+ //
587
+ // Narrow replaces title and narrative with model text, so an explicit save it rewrites (one
588
+ // whose save-time enrich failed) moves to the re-enrich writer's id, as a cluster-merge
589
+ // keeper does (D#146, D#138). Wide keeps the stored title and narrative (it fills the lesson
590
+ // and the side fields), so the row stays an explicit save. The writer's session row is
591
+ // best-effort.
592
+ let rewriteSessionId = null;
593
+ if (!isWide) {
594
+ const cur = db.prepare('SELECT memory_session_id FROM observations WHERE id = ?').get(cand.id);
595
+ if (cur?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
596
+ const enrichSessionId = writerSessionId('enrich-', cand.project);
597
+ try {
598
+ db.prepare(
599
+ `INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
600
+ VALUES (?, ?, ?, ?, ?, 'active')`,
601
+ ).run(enrichSessionId, enrichSessionId, cand.project, new Date().toISOString(), Date.now());
602
+ rewriteSessionId = enrichSessionId;
603
+ } catch (e) {
604
+ debugCatch(e, 'reenrich writer session');
605
+ }
606
+ }
607
+ }
586
608
  const res = db
587
609
  .prepare(
588
610
  `
589
611
  UPDATE observations SET type=?, title=?, narrative=?, concepts=?, facts=?,
590
612
  text=?, importance=?, lesson_learned=?, search_aliases=?, minhash_sig=?, optimized_at=?,
591
- scope=COALESCE(?, scope)
613
+ scope=COALESCE(?, scope), memory_session_id = COALESCE(?, memory_session_id)
592
614
  WHERE id = ? AND ${liveObsFilterSql('')} AND optimized_at IS NULL
593
615
  `,
594
616
  )
@@ -608,6 +630,7 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
608
630
  // scope, or emits an off-enum value, must never blank an existing label —
609
631
  // and THIS update stamps optimized_at, so the loss would be permanent.
610
632
  normalizeScope(parsed.scope),
633
+ rewriteSessionId,
611
634
  cand.id,
612
635
  );
613
636
  if (res.changes === 0) {
@@ -1315,15 +1338,15 @@ Return ONLY valid JSON:
1315
1338
  // The keeper now holds model text. Search reads authorship from memory_session_id, so an
1316
1339
  // explicit save's `manual-` id would mark it as one (D#138); it moves to the compression
1317
1340
  // writer's id. A machine-written keeper keeps its id, and so does the snapshot above, which
1318
- // is the original save. The writer's session row is best-effort: sdk_sessions refuses an id
1319
- // shaped like a uuid, which `compress-<project>` is for a 27-character project shaped
1320
- // xxxx-xxxx-xxxx-xxxxxxxxxxxx, and that refusal must not fail the merge (v6.19.1 pre-tag F5a).
1341
+ // is the original save. writerSessionId keeps the id out of the uuid shape sdk_sessions
1342
+ // refuses (D#147); the row stays best-effort, so no refusal can fail the merge (v6.19.1
1343
+ // pre-tag F5a).
1321
1344
  let rewriteSessionId = null;
1322
1345
  const keeperSession = db
1323
1346
  .prepare('SELECT memory_session_id FROM observations WHERE id = ?')
1324
1347
  .get(keeper.id);
1325
1348
  if (keeperSession?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
1326
- const compressSessionId = `compress-${keeper.project}`;
1349
+ const compressSessionId = writerSessionId('compress-', keeper.project);
1327
1350
  try {
1328
1351
  db.prepare(
1329
1352
  `INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
@@ -1580,7 +1603,7 @@ export async function executeSmartCompressCluster(db, observations, project) {
1580
1603
  ) {
1581
1604
  return null;
1582
1605
  }
1583
- const sessionId = `compress-${project}`;
1606
+ const sessionId = writerSessionId('compress-', project);
1584
1607
  const now = new Date();
1585
1608
  db.prepare(
1586
1609
  `INSERT OR IGNORE INTO sdk_sessions
package/hook-shared.mjs CHANGED
@@ -30,6 +30,7 @@ import {
30
30
  } from './lib/schema-skew.mjs';
31
31
  import { isDbUnusableError, DB_UNUSABLE_MARKER_PREFIX } from './lib/db-unusable.mjs';
32
32
  import { shouldRecordOnce } from './lib/record-once.mjs';
33
+ import { hookSessionId } from './lib/provenance.mjs';
33
34
  // Audit 2026-09-05 P1-2 (carried from 2026-09-02 P2-9): `callLLM`, the quiet/adoption
34
35
  // predicates and the handoff constants moved into `lib/` because two lib modules
35
36
  // imported them from here and dragged this file's whole import graph — haiku-client,
@@ -385,7 +386,7 @@ export function getSessionId() {
385
386
 
386
387
  export function createSessionId() {
387
388
  const project = inferProject();
388
- const id = `hook-${project}-${randomUUID().slice(0, 8)}`;
389
+ const id = hookSessionId(project, randomUUID().slice(0, 8));
389
390
  const file = sessionFile();
390
391
  const tmp = file + `.tmp-${process.pid}`;
391
392
  writeFileSync(tmp, JSON.stringify({ id, startedAt: Date.now(), project }), { mode: 0o600 });
package/lib/activity.mjs CHANGED
@@ -7,6 +7,7 @@
7
7
  import { sanitizeFtsQuery } from '../utils.mjs';
8
8
  import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
9
9
  import { saveObservation } from './save-observation.mjs';
10
+ import { writerSessionId } from './provenance.mjs';
10
11
  // Pure title-only builder: this query runs on the EVENTS table, which has no
11
12
  // lesson_learned column — the lesson-escape variant would be a SQL error here.
12
13
  import { buildNotLowSignalSql } from './low-signal-patterns.mjs';
@@ -200,7 +201,7 @@ export function promoteInsightEvents(
200
201
  lesson_learned: ev.body.slice(0, 500),
201
202
  now: new Date(ev.created_at_epoch),
202
203
  // Event bodies are machine-written (episode summaries), so the row is not an explicit save.
203
- sessionId: `promote-${ev.project}`,
204
+ sessionId: writerSessionId('promote-', ev.project),
204
205
  });
205
206
  mark.run(Date.now(), ev.id);
206
207
  return r;
@@ -6,6 +6,7 @@
6
6
 
7
7
  import { TIER_CASE_SQL, tierSqlParams } from '../tier.mjs';
8
8
  import { liveObsFilterSql } from './inject-search-core.mjs';
9
+ import { HOOK_SESSION_ID_PREFIX } from './provenance.mjs';
9
10
 
10
11
  export const BROWSE_TIERS = ['working', 'active', 'archive'];
11
12
  export const BROWSE_TIER_LABELS = {
@@ -14,14 +15,17 @@ export const BROWSE_TIER_LABELS = {
14
15
  archive: '🔵 Archive',
15
16
  };
16
17
 
17
- /** Newest active memory_session_id for the project ('' when none) — the tier
18
- * classifier's "current session" input, needed identically by both faces. */
18
+ /** Newest active hook session id for the project ('' when none) — the tier classifier's
19
+ * "current session" input, needed identically by both faces. Only the hook's own sessions
20
+ * count: a writer's row (`manual-`, `compress-`, …) is inserted active on that writer's first
21
+ * write in a project and stays active until the 24h sweep, so taken as the current session it
22
+ * put all of that writer's rows, of any age, in the working tier (D#153). */
19
23
  export function getActiveMemorySessionId(db, project) {
20
24
  const row = db
21
25
  .prepare(
22
- "SELECT memory_session_id FROM sdk_sessions WHERE project = ? AND status = 'active' ORDER BY started_at_epoch DESC LIMIT 1",
26
+ "SELECT memory_session_id FROM sdk_sessions WHERE project = ? AND status = 'active' AND memory_session_id LIKE ? ORDER BY started_at_epoch DESC LIMIT 1",
23
27
  )
24
- .get(project);
28
+ .get(project, `${HOOK_SESSION_ID_PREFIX}%`);
25
29
  return row?.memory_session_id ?? '';
26
30
  }
27
31
 
@@ -17,6 +17,7 @@
17
17
 
18
18
  import { isoWeekKey, COMPRESSED_AUTO } from '../utils.mjs';
19
19
  import { scrubRecord } from './scrub-record.mjs';
20
+ import { writerSessionId } from './provenance.mjs';
20
21
 
21
22
  /**
22
23
  * Low-value compression candidates: importance<=1, never accessed, older than
@@ -88,7 +89,7 @@ export function compressGroup(db, proj, obs) {
88
89
  const dominantType = Object.entries(types).sort((a, b) => b[1] - a[1])[0][0];
89
90
  const title = `Weekly summary: ${obs.length} ${dominantType} observations`;
90
91
  const narrative = obs.map((o) => `- ${o.title || '(untitled)'}`).join('\n');
91
- const sessionId = `compress-${proj}`;
92
+ const sessionId = writerSessionId('compress-', proj);
92
93
 
93
94
  const sortedEpochs = obs.map((o) => o.created_at_epoch).sort((a, b) => a - b);
94
95
  const medianEpoch = sortedEpochs[Math.floor(sortedEpochs.length / 2)];
@@ -11,8 +11,9 @@
11
11
  // twin-drift class permanently.
12
12
  //
13
13
  // Full round-trippable set: content + value-signals (access/cited/uncited/injection/decay)
14
- // + branch + timing. `id` + `memory_session_id` are informational (restore remaps id and
15
- // buckets under a synthetic restore session). Session-idempotency keys (last_decided/
14
+ // + branch + timing. `id` + `memory_session_id` are informational: restore remaps id, and of the
15
+ // session id it keeps only whether the row was an explicit save (`manual-<project>`) or not
16
+ // (`restore-<project>`; see cmdRestore). Session-idempotency keys (last_decided/
16
17
  // last_cited/last_access_session_id, demoted_at, optimized_at) are intentionally NOT
17
18
  // exported — they are meaningless after a row is re-bucketed under a restore session.
18
19
  // last_access_session_id (v48) joins that set for the same reason: it answers "did THIS cc
@@ -1,8 +1,9 @@
1
1
  // Which writer produced an observation, read back from its memory_session_id. saveObservation
2
2
  // writes every explicit save under MANUAL_SESSION_ID_PREFIX. The hook capture (`hook-`),
3
3
  // transcript import (`import-`), compression summaries and cluster-merge keepers (`compress-`),
4
- // promoted events (`promote-`) and rows imported from older stores (a bare session uuid) are
5
- // machine-written.
4
+ // rows narrow re-enrich rewrote (`enrich-`), promoted events (`promote-`), machine-written rows
5
+ // `restore` brought back from a backup (`restore-`) and rows imported from older stores (a bare
6
+ // session uuid) are machine-written.
6
7
  //
7
8
  // Search and get mark the machine-written side, because explicit saves are the large majority
8
9
  // of a typical store and a mark on nearly every line would carry no information. An unknown id
@@ -10,6 +11,28 @@
10
11
 
11
12
  export const MANUAL_SESSION_ID_PREFIX = 'manual-';
12
13
 
14
+ // sdk_sessions refuses a row whose two ids are equal and 36 characters long with dashes where a
15
+ // uuid has them (schema.mjs, sdk_sessions_id_mix_check_ai). A writer row uses `<prefix><project>`
16
+ // for both, which has that shape for one project length and dash layout (27 characters under
17
+ // `compress-`, 29 under `manual-`), and the refusal failed every write in such a project (D#147).
18
+ // The hook's own id, `hook-<project>-<8 hex>`, has it for a 22-character project with dashes at
19
+ // 3, 8, 13 and 18 (D#156). That id gets one more character; every other id is unchanged. Counted
20
+ // in code points, as SQLite's length() and LIKE's `_` count them.
21
+ const UUID_SHAPED_RE = /^[\s\S]{8}-[\s\S]{4}-[\s\S]{4}-[\s\S]{4}-[\s\S]{12}$/u;
22
+ const outOfUuidShape = (id) => (UUID_SHAPED_RE.test(id) ? `${id}~` : id);
23
+
24
+ /** The session id a non-hook writer (`manual-`, `compress-`, `enrich-`, `promote-`, `restore-`) stores for a project. */
25
+ export function writerSessionId(prefix, project) {
26
+ return outOfUuidShape(`${prefix}${project}`);
27
+ }
28
+
29
+ export const HOOK_SESSION_ID_PREFIX = 'hook-';
30
+
31
+ /** The hook's id for one session in a project: `hook-<project>-<tail>`. */
32
+ export function hookSessionId(project, tail) {
33
+ return outOfUuidShape(`${HOOK_SESSION_ID_PREFIX}${project}-${tail}`);
34
+ }
35
+
13
36
  const AUTO_MARK = '🤖';
14
37
  const AUTO_TEXT = 'auto-written, not an explicit save';
15
38
 
@@ -22,7 +22,7 @@ import { liveObsFilterSql } from './inject-search-core.mjs';
22
22
  // Imported, not injected: `allowStatuses` below is the POLICY this function exists to hold,
23
23
  // and a caller free to pass its own resolver could reinstate the one-way gate D#195 closed.
24
24
  import { resolveDeferredIds, closeDeferredItems } from './deferred-work.mjs';
25
- import { MANUAL_SESSION_ID_PREFIX } from './provenance.mjs';
25
+ import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './provenance.mjs';
26
26
 
27
27
  const DEDUP_WINDOW_MS = 5 * 60 * 1000;
28
28
  const DEDUP_RECENT_LIMIT = 50;
@@ -201,7 +201,7 @@ export function saveObservation(db, params) {
201
201
 
202
202
  // An explicit save by default; a caller storing machine-written text names its own writer
203
203
  // (lib/provenance.mjs reads authorship from this id).
204
- const sessionId = params.sessionId || `${MANUAL_SESSION_ID_PREFIX}${project}`;
204
+ const sessionId = params.sessionId || writerSessionId(MANUAL_SESSION_ID_PREFIX, project);
205
205
 
206
206
  // Ensure session exists (FK constraint). INSERT OR IGNORE makes this safe
207
207
  // under concurrent calls.
package/mem-cli.mjs CHANGED
@@ -25,7 +25,7 @@ import { resolveProject } from './project-utils.mjs';
25
25
  import { resolveCliProject as cliProject } from './lib/cli-project.mjs';
26
26
  import { reRankWithContext } from './search-scoring.mjs';
27
27
  import { searchObservationsHybrid } from './search-engine.mjs';
28
- import { autoHeaderNote, autoLegend, autoTag } from './lib/provenance.mjs';
28
+ import { autoHeaderNote, autoLegend, autoTag, isAutoWritten, writerSessionId } from './lib/provenance.mjs';
29
29
  import {
30
30
  fetchObsDetail,
31
31
  fetchPromptDetail,
@@ -2268,8 +2268,18 @@ function cmdExport(db, args) {
2268
2268
  // (access/cited/uncited/injection/decay), branch, and concepts/facts/files_read that
2269
2269
  // saveObservation derives or zeros — so a restored backup keeps its citation-decay
2270
2270
  // history and original timing (created_at via the `now` param). Source ids are
2271
- // discarded (local AUTOINCREMENT; export omits related_ids); session provenance
2272
- // collapses to saveObservation's manual-<project> bucket (documented MVP tradeoff).
2271
+ // discarded (local AUTOINCREMENT; export omits related_ids). Session ids are not restored,
2272
+ // only whether a row was an explicit save: see RESTORE_SESSION_ID_PREFIX.
2273
+ // A machine-written row (any exported id but `manual-`) is restored under `restore-<project>`, so
2274
+ // lib/provenance.mjs still reads it as auto-written; before D#157 every row went under
2275
+ // `manual-<project>` and rendered as an explicit save. The exported id is not reused: saveObservation
2276
+ // stores it as both session ids of an active session row, and a bare session uuid
2277
+ // (rows imported from older stores) is the shape sdk_sessions refuses, a `hook-` id could become
2278
+ // browse's current session (the only active hook row once Stop has marked the real one completed)
2279
+ // until a SessionStart sweeps it, and under --project the id names another project. A row exported
2280
+ // without the column restores as an explicit save, as before.
2281
+ const RESTORE_SESSION_ID_PREFIX = 'restore-';
2282
+
2273
2283
  function cmdRestore(db, argv) {
2274
2284
  const { positional, flags } = parseArgs(argv);
2275
2285
  const file = positional[0];
@@ -2396,6 +2406,9 @@ function cmdRestore(db, argv) {
2396
2406
  files,
2397
2407
  lesson_learned: r.lesson_learned || null,
2398
2408
  now: new Date(createdEpoch),
2409
+ sessionId: isAutoWritten(r.memory_session_id)
2410
+ ? writerSessionId(RESTORE_SESSION_ID_PREFIX, project)
2411
+ : undefined,
2399
2412
  });
2400
2413
  if (res.kind !== 'saved') {
2401
2414
  skipped++;
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.1",
3
+ "version": "6.19.3",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "claude-mem-lite",
9
- "version": "6.19.1",
9
+ "version": "6.19.3",
10
10
  "os": [
11
11
  "darwin",
12
12
  "linux",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-mem-lite",
3
- "version": "6.19.1",
3
+ "version": "6.19.3",
4
4
  "description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
5
5
  "type": "module",
6
6
  "packageManager": "npm@10.9.2",
package/secret-scrub.mjs CHANGED
@@ -197,37 +197,26 @@ export const SECRET_PATTERNS = [
197
197
  ],
198
198
  // PEM private key blocks. `[A-Z0-9 ]*` covers every armor label — RSA/EC/DSA/
199
199
  // OPENSSH plus ENCRYPTED and PGP (… PRIVATE KEY BLOCK) — that the fixed
200
- // alternation missed; the block delimiters make FP impossible.
200
+ // alternation missed.
201
201
  // The body stops at the next `-----BEGIN ` (D#130): with `[\s\S]*?` every header with no END
202
202
  // scanned to the end of the text, on each of scrubSecrets' passes — quadratic, 8.3 s on 500k
203
- // chars. A block whose END is missing is left to the next pattern.
203
+ // chars. A block whose END is missing is left to the next pattern. Nor does it cross a mark this
204
+ // scrubber wrote: on a later pass, a BEGIN that a scrubbed key used to block reached a far END
205
+ // and erased the prose between (v6.19.2 pre-tag defect review F5).
206
+ // Text that names a BEGIN and, later, an END with no key between them loses the text between
207
+ // (D#155, open). Keeping such a block was tried for 6.19.3 and withdrawn before the tag: each of
208
+ // the three keep rules measured stored keys this pattern erases, either a key it took for text
209
+ // or a key tail the kept markers hid from scrubKeyTails (docs/audits/20260928-v6.19.3-pretag-*.md).
204
210
  [
205
- /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
211
+ /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY|\*\*\*PEM_KEY\*\*\*)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
206
212
  '***PEM_KEY***',
207
213
  ],
208
- // A cut-off key (`head id_rsa`, a tool output cut mid-key): the header and the WHOLE lines of
209
- // base64 (or RFC 1421 headers) that follow it. Stored whole through v6.19.0 when no later key
210
- // header followed. Whole lines only, so prose naming a header mid-sentence keeps its text
211
- // (v6.19.0 round-3 P3-3: ending the block at the next key header erased the prose between two).
212
- // A body line starts with 16+ base64 characters, and one shorter whole line may follow the last
213
- // of them (a key's last line): a line of one word or number is prose (v6.19.1 pre-tag review F3),
214
- // so the body needs a long line, and blank lines count only between long ones. A long line need
215
- // not be whole, so a key cut mid-line loses the cut line too. A line break may be JSON-escaped
216
- // (`\n` as two characters) and a quote may end the last line (v6.19.1 claims review F1). A header
217
- // value may hold a backslash that does not start an escaped break (a Windows path), and it does
218
- // not share its whitespace with a second quantifier, which was quadratic (delta review P1).
219
- [
220
- /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:[ \t]*(?:\r?\n|\\r\\n|\\n)(?=[ \t]*(?:\r?\n|\\r\\n|\\n)))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*(?:[A-Za-z0-9+/=]{16,}|(?:Proc-Type|DEK-Info|Version|Comment|Hash|Charset):(?:[^\r\n\\]|\\(?![rn]))*(?=\r?\n|\\[rn]|["']|$)))(?:(?:[ \t]*(?:\r?\n|\\r\\n|\\n)(?=[ \t]*(?:\r?\n|\\r\\n|\\n)))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*(?:[A-Za-z0-9+/=]{16,}|(?:Proc-Type|DEK-Info|Version|Comment|Hash|Charset):(?:[^\r\n\\]|\\(?![rn]))*(?=\r?\n|\\[rn]|["']|$))))*(?:[ \t]*(?:\r?\n|\\r\\n|\\n)[ \t]*[A-Za-z0-9+/=]{1,15}[ \t]*(?=\r?\n|\\[rn]|["']|$))?/g,
221
- '***PEM_KEY***',
222
- ],
223
- // The other end (`tail key.pem`): whole base64 lines ending in a private-key END, the same line
224
- // rule as above but with two shorter lines allowed before the END (an armored PGP key's last data
225
- // line and its `=XXXX` checksum; delta review P2). A line may start after a quote. A run of lines
226
- // that does not end there is consumed and returned unchanged, so no line starts a second scan.
227
- [
228
- /(?:(?<![^\n])|(?<=\\n|["']))(?:[ \t]*[A-Za-z0-9+/=]{16,}[ \t]*(?:\r?\n|\\r\\n|\\n))+(?:[ \t]*[A-Za-z0-9+/=]{1,15}[ \t]*(?:\r?\n|\\r\\n|\\n)){0,2}(?:-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----)?/g,
229
- (m) => (m.endsWith('-----') ? '***PEM_KEY***' : m),
230
- ],
214
+ // A cut-off key (`head id_rsa`, a tool output cut mid-key) and a headerless key tail (`tail
215
+ // key.pem`). These are line scanners, not patterns (D#145): three rounds of review each found a
216
+ // line shape the regex versions misread, and each fix to one shape erased prose in another.
217
+ // See scrubCutOffKeys and scrubKeyTails below for the line rules.
218
+ [{ [Symbol.replace]: (text) => scrubCutOffKeys(text) }, null],
219
+ [{ [Symbol.replace]: (text) => scrubKeyTails(text) }, null],
231
220
  // Long hex strings in credential assignments (e.g. SECRET_KEY=abc123def456...).
232
221
  // `hash` deliberately excluded: `hash: <40hex>` / `hash=<md5>` are git SHAs and
233
222
  // checksums (real, preserved data in this hash-heavy repo), not credentials.
@@ -351,6 +340,479 @@ export const SECRET_PATTERNS = [
351
340
  // bare-token pattern here: don't — anchor it to a provider prefix instead.
352
341
  ];
353
342
 
343
+ // ─── Cut-off private keys (D#145) ──────────────────────────────────────────
344
+ // A key's text reaches the scrubber in many line shapes: plain LF/CRLF/CR lines, JSON-escaped
345
+ // breaks (`\n` as two characters, or `\\n` when serialised twice), lines carrying a prefix (the
346
+ // Read tool's ` 2\t` or `2→`, grep's `id_rsa:`, a `> ` quote, a diff `-`), and lines inside a
347
+ // quoted string. The scanners read the text as lines of ONE shape per key, decided at the key:
348
+ // - the break after the BEGIN line (or before the END line) says whether breaks are real or
349
+ // escaped, and at which depth; in real-break text a backslash is never a break, so a Windows
350
+ // path in a `Comment:` value no longer ends the header (v6.19.1 round-3 P3-E);
351
+ // - the text before the BEGIN (or END) on its line is the line prefix, and every other line
352
+ // loses a prefix of the same shape (digits may differ, `:` and `-` swap for grep context)
353
+ // before it is judged;
354
+ // - lines end at breaks only. A header value that runs on into the next JSON fields is still
355
+ // one header line, so it erases nothing without a base64 line under it (round-3 P3-C).
356
+ // A body line is WHOLE base64: 16+ characters, or 1-15 for the last line. So a word under a key
357
+ // body keeps its text (round-3 P3-A: `Don't` lost `Don`), as does a path or an identifier that
358
+ // starts the next line (P3-B), and a body needs one long line, so a header followed by words is
359
+ // not a key (F3). A string's closing quote, a backtick or a closing tag after the line is not part
360
+ // of it. A line whose base64 run is followed by something else is a cut or annotated key line
361
+ // when the run is 40+ characters or a truncation mark or delimiter follows it (`…`, `...`,
362
+ // `[truncated]`, `<`): its base64 goes and the rest stays (delta review P3-2; v6.19.2 pre-tag
363
+ // defect review F4). A shorter run followed by words starts a line of prose.
364
+
365
+ const KEY_BEGIN_RE = /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
366
+ const KEY_END_RE = /-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
367
+ const KEY_END_LINE_RE = /^-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/;
368
+ const B64_LONG_RE = /^[A-Za-z0-9+/=]{16,}$/;
369
+ const B64_SHORT_RE = /^[A-Za-z0-9+/=]{1,15}$/;
370
+ // Space-separated chunks, allowed only on the BEGIN line itself (a key pasted onto one line).
371
+ const B64_CHUNKS_RE = /^[A-Za-z0-9+/=]+(?:[ \t]+[A-Za-z0-9+/=]+)*$/;
372
+ const B64_RUN_RE = /^[A-Za-z0-9+/=]{16,}/;
373
+ // What may follow a key line that was cut or annotated: a truncation mark or a delimiter. A quote
374
+ // ends the string a key sits in (`…', 'rc': 0}`, `…","stderr":…`): the scanners do not track which
375
+ // quote opened a string, since no character before a quote tells an opening quote from prose
376
+ // (`Here's`, `Run 'head id_rsa'`, `b'…'`; v6.19.2 pre-tag reviews, delta F1 and round-3 F1/F2).
377
+ // Two escapes deep a string ends at `\"` (round-4 F2). Five dashes are the next key's marker glued
378
+ // to a cut line (`for i in …; do head -c 80 k; done`).
379
+ const CUT_MARK_RE = /^(?: ?…| ?\.\.\.| ?\[| ?<|[`"']|\\+["']|-----)/;
380
+ const PGP_CRC_RE = /^=[A-Za-z0-9+/]{4}$/;
381
+ // How a key's base64 starts: a DER SEQUENCE (PKCS#1, PKCS#8, SEC1) or OpenSSH's `openssh-key-v1`;
382
+ // under a PGP header, a secret-key packet in the old or new format (`lQ…`, `xc…`/`xV…`).
383
+ const KEY_MAGIC_RE = /^(?:MII|MIG|MC4C|MHcC|b3BlbnNzaC1rZXktdjE)/;
384
+ const PGP_MAGIC_RE = /^(?:lQ|x[cV])/;
385
+ // RFC 1421 / RFC 4880 armor headers, before the body. Named, not any `Word:`: a `Note:` line is
386
+ // prose, and taking it for a header made the lone base64 line under it a key.
387
+ // RFC 1421's full set and tool-written `X-` headers count too (v6.19.2 pre-tag delta review F4:
388
+ // `Content-Domain` or `X-Custom` before the body stored the whole key).
389
+ const ARMOR_HEADER_RE =
390
+ /^(?:Proc-Type|DEK-Info|Content-Domain|Originator-ID-(?:Asymmetric|Symmetric)|Originator-Certificate|Issuer-Certificate|MIC-Info|Key-Info|Recipient-ID-(?:Asymmetric|Symmetric)|CRL|Version|Comment|Hash|Charset|MessageID|X-[A-Za-z0-9-]+)[ \t]*:/i;
391
+ const MAX_ARMOR_HEADERS = 16;
392
+ // A JS/Python string split across source lines: `…\n" +` then `"…` on the next line. Not a comma:
393
+ // `'…\n',` then `'…'` is the next element of a list, and its first word is not the key's last line.
394
+ // One whitespace quantifier on each side of the `+`: `[ \t]*\+?[ \t]*` split a run between two
395
+ // and was quadratic when no break followed (v6.19.2 pre-tag defect review F1: 16-19 s at 200k).
396
+ const CONCAT_AFTER_RE = /["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']/y;
397
+ const CONCAT_BEFORE_RE = /(?:\\r)?\\n["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']$/;
398
+ const PEM_MARK = '***PEM_KEY***';
399
+ // After an END that ends its line: a closing quote or punctuation, then a break or the text end.
400
+ // Closing tags may be nested (`END</code></pre>`; round-4 F3), and a string's closing quote, escaped
401
+ // or not, ends the line whatever follows it: `…END-----","stderr":""` is `tail -n 2` in a tool
402
+ // result (round-5 F3).
403
+ const CLEAN_AFTER_END_RE = /(?:[ \t`,;)\]}]|<\/[A-Za-z][\w:.-]{0,40}>)*(?:\r|\n|\\+[nr]|\\*["']|$)/y;
404
+
405
+ const BLANK = 0;
406
+ const LONG = 1;
407
+ const SHORT = 2;
408
+ const END = 3;
409
+ const ARMOR = 4;
410
+ const CUT = 5;
411
+ const OTHER = 6;
412
+
413
+ // Character-class checks by code, not a regex per character. (What brought the many-BEGINs linearity
414
+ // shape back under its budget after the v6.19.2 CI run was cutOffKeyEnd stopping before another
415
+ // BEGIN line, not these checks: under coverage the shape measured 5.6-6.2x benign with these checks
416
+ // alone, 1.9-2.3x with the stop alone.)
417
+ function isB64Code(c) {
418
+ return (
419
+ (c >= 65 && c <= 90) || (c >= 97 && c <= 122) || (c >= 48 && c <= 57) || c === 43 || c === 47 || c === 61
420
+ );
421
+ }
422
+
423
+ function backslashesBefore(text, i, floor) {
424
+ let j = i;
425
+ while (j > floor && text[j - 1] === '\\') j--;
426
+ return i - j;
427
+ }
428
+
429
+ // Break depth `u`: 0 = real breaks only, 1 = `\n`, 2 = `\\n`, … A run of r backslashes then `n`
430
+ // is a break at depth u when r % 2u === u (the backslashes before it are escaped backslashes).
431
+ const isEscBreak = (r, u) => u > 0 && r % (2 * u) === u;
432
+ // An unescaped quote at depth u ends the string: r % 2u < u.
433
+ const isStringEnd = (r, u) => u > 0 && r % (2 * u) < u;
434
+
435
+ /** The line starting at s: its end, and where the next line starts (-1 at the end of the text). */
436
+ function nextLine(text, s, u) {
437
+ const n = text.length;
438
+ let i = s;
439
+ while (i < n) {
440
+ const ch = text[i];
441
+ if (ch === '\n') return { end: i, next: i + 1 };
442
+ if (ch === '\r') return { end: i, next: text[i + 1] === '\n' ? i + 2 : i + 1 };
443
+ if (u > 0 && ch === '\\') {
444
+ let j = i;
445
+ while (j < n && text[j] === '\\') j++;
446
+ const r = j - i;
447
+ const c = text[j];
448
+ if ((c === 'n' || c === 'r') && isEscBreak(r, u)) {
449
+ let next = j + 1;
450
+ if (c === 'r' && text.startsWith('\\'.repeat(u) + 'n', next)) next += u + 1;
451
+ if (u === 1) {
452
+ CONCAT_AFTER_RE.lastIndex = next;
453
+ const m = CONCAT_AFTER_RE.exec(text);
454
+ if (m) next += m[0].length;
455
+ }
456
+ return { end: j - u, next };
457
+ }
458
+ i = j + 1;
459
+ continue;
460
+ }
461
+ i++;
462
+ }
463
+ return { end: n, next: -1 };
464
+ }
465
+
466
+ /** The line that ends at the break before `ls`, or null when `ls` starts the text or string. */
467
+ function prevLine(text, ls, u, floor) {
468
+ if (ls - 1 < floor) return null;
469
+ const c = text[ls - 1];
470
+ let bs = -1;
471
+ if (c === '\n') bs = ls - 2 >= floor && text[ls - 2] === '\r' ? ls - 2 : ls - 1;
472
+ else if (c === '\r') bs = ls - 1;
473
+ else if ((c === 'n' || c === 'r') && isEscBreak(backslashesBefore(text, ls - 1, floor), u)) {
474
+ bs = ls - 1 - u;
475
+ const r2 = bs - 1 >= floor && text[bs - 1] === 'r' ? backslashesBefore(text, bs - 1, floor) : 0;
476
+ if (c === 'n' && isEscBreak(r2, u)) bs -= u + 1;
477
+ } else if (u === 1 && (c === '"' || c === "'")) {
478
+ const m = CONCAT_BEFORE_RE.exec(text.slice(Math.max(floor, ls - 64), ls));
479
+ if (m) bs = ls - m[0].length;
480
+ }
481
+ if (bs === -1) return null;
482
+ let i = bs - 1;
483
+ while (i >= floor) {
484
+ const ch = text[i];
485
+ if (ch === '\n' || ch === '\r') break;
486
+ if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, i, floor), u)) break;
487
+ if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, i, floor), u)) break;
488
+ i--;
489
+ }
490
+ return { start: i + 1, end: bs };
491
+ }
492
+
493
+ /** A regex for the line prefix `prefix` has, with digits free and grep's `:`/`-` interchangeable. */
494
+ function prefixShape(prefix) {
495
+ const body = prefix.replace(/^[ \t]+/, '');
496
+ if (!body || body.length > 256) return null;
497
+ if (body === '-' || body === '+') return /^[ \t]*[-+ ]/;
498
+ const src = body
499
+ .replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')
500
+ .replace(/\d+/g, '\\d+')
501
+ .replace(/[:-]/g, '[:-]');
502
+ return new RegExp(`^[ \\t]*${src}`);
503
+ }
504
+
505
+ /** The judged part of line [s, e): no prefix of the key's shape, no padding, no string quotes. */
506
+ function lineCore(text, s, e, shape) {
507
+ let line = text.slice(s, e);
508
+ let off = s;
509
+ if (shape) {
510
+ const m = shape.exec(line);
511
+ if (m) {
512
+ line = line.slice(m[0].length);
513
+ off += m[0].length;
514
+ }
515
+ }
516
+ let a = 0;
517
+ let b = line.length;
518
+ while (a < b && (line[a] === ' ' || line[a] === '\t')) a++;
519
+ while (b > a && (line[b - 1] === ' ' || line[b - 1] === '\t')) b--;
520
+ if (a < b && (line[a] === '"' || line[a] === "'" || line[a] === '`')) a++;
521
+ // A closing quote or backtick with what may follow it (`",`, `" +`, `')`), or a closing tag
522
+ // (`</key>`), and escaped breaks before it. Read from the end: an unanchored
523
+ // `(?:\\+[rn])*["']…$` retried every start in a backslash run.
524
+ let q = b;
525
+ while (q > a && ' \t+,;)]}'.includes(line[q - 1])) q--;
526
+ const tag = q > a && line[q - 1] === '>' ? /<\/[A-Za-z][\w:.-]{0,40}>$/.exec(line.slice(a, q)) : null;
527
+ if (tag) b = q - tag[0].length;
528
+ else if (q > a && (line[q - 1] === '"' || line[q - 1] === "'" || line[q - 1] === '`')) {
529
+ q--;
530
+ for (;;) {
531
+ if (q - 1 <= a || (line[q - 1] !== 'n' && line[q - 1] !== 'r')) break;
532
+ let k = q - 1;
533
+ while (k > a && line[k - 1] === '\\') k--;
534
+ if (k === q - 1) break;
535
+ q = k;
536
+ }
537
+ b = q;
538
+ }
539
+ return { core: line.slice(a, b), start: off + a, end: off + b };
540
+ }
541
+
542
+ function classify(core) {
543
+ if (core === '') return BLANK;
544
+ if (B64_LONG_RE.test(core)) return LONG;
545
+ if (B64_SHORT_RE.test(core)) return SHORT;
546
+ if (KEY_END_LINE_RE.test(core)) return END;
547
+ if (ARMOR_HEADER_RE.test(core)) return ARMOR;
548
+ if (cutRun(core)) return CUT;
549
+ return OTHER;
550
+ }
551
+
552
+ /**
553
+ * The base64 run a cut or annotated key line starts with, or null: 16+ characters followed by a
554
+ * truncation mark or a delimiter (`…`, `...`, `[truncated]`, `<`, a backtick), or 40+ followed by
555
+ * anything (`<64> see above`). A shorter run followed by text is an identifier or a path starting
556
+ * a line of prose (`exportedArmoredPrivateKey = …`, round-3 P3-B), which stays.
557
+ */
558
+ function cutRun(core) {
559
+ const m = B64_RUN_RE.exec(core);
560
+ if (!m) return null;
561
+ return m[0].length >= 40 || CUT_MARK_RE.test(core.slice(m[0].length)) ? m[0] : null;
562
+ }
563
+
564
+ /**
565
+ * Where the key that starts with the BEGIN at [b, be) ends, or -1 when no key material follows it.
566
+ * Header lines and blank lines may come first; then 16+-character base64 lines (blank lines only
567
+ * between them), one shorter last line (and a PGP `=XXXX` checksum after it), and the END if it
568
+ * is there. One base64 line alone is a key only when it is 40+ characters, starts the way a key
569
+ * encoding starts (DER `MII…`, OpenSSH `b3BlbnNzaC1rZXktdjE…`) or follows armor headers: a path
570
+ * or an identifier of 16-39 characters under a header is prose (delta review P3-5). A complete
571
+ * block never gets here; the block pattern above takes it first.
572
+ */
573
+ function cutOffKeyEnd(text, b, be) {
574
+ // The rest of the BEGIN line: nothing, or base64 chunks, then a break that sets the depth.
575
+ let i = be;
576
+ for (
577
+ let c = text.charCodeAt(i);
578
+ i < text.length && (isB64Code(c) || c === 32 || c === 9);
579
+ c = text.charCodeAt(++i)
580
+ );
581
+ const rest = text.slice(be, i).trim();
582
+ let u;
583
+ let next;
584
+ if (i >= text.length) return -1;
585
+ if (text[i] === '\n' || text[i] === '\r') {
586
+ u = 0;
587
+ next = text[i] === '\r' && text[i + 1] === '\n' ? i + 2 : i + 1;
588
+ } else if (text[i] === '\\') {
589
+ let j = i;
590
+ while (j < text.length && text[j] === '\\') j++;
591
+ const r = j - i;
592
+ if (text[j] !== 'n' && text[j] !== 'r') return -1;
593
+ u = r & -r;
594
+ if (r !== u) return -1; // a literal backslash on the BEGIN line: not a key line
595
+ ({ next } = nextLine(text, i, u));
596
+ } else return -1;
597
+ let longs = 0;
598
+ let longest = 0;
599
+ let first = '';
600
+ let end = -1;
601
+ if (rest) {
602
+ // A key pasted onto its BEGIN line: every chunk but the last is a full 16+ line. Words there
603
+ // are prose, even one of 16+ letters (v6.19.2 pre-tag defect review F9). A loop, not a spread:
604
+ // Math.max(...chunks) overflowed the stack past ~125k chunks (F2).
605
+ if (!B64_CHUNKS_RE.test(rest)) return -1;
606
+ const chunks = rest.split(/[ \t]+/);
607
+ for (let k = 0; k < chunks.length; k++) {
608
+ if (k < chunks.length - 1 && chunks[k].length < 16) return -1;
609
+ if (chunks[k].length > longest) longest = chunks[k].length;
610
+ }
611
+ if (longest < 16) return -1;
612
+ longs = 1;
613
+ first = chunks[0];
614
+ end = be + text.slice(be, i).trimEnd().length;
615
+ }
616
+ // The BEGIN line's prefix.
617
+ let ls = b;
618
+ while (ls > 0) {
619
+ const ch = text[ls - 1];
620
+ if (ch === '\n' || ch === '\r') break;
621
+ if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, ls - 1, 0), u)) break;
622
+ if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, ls - 1, 0), u)) break;
623
+ ls--;
624
+ }
625
+ // A 16+ base64 run glued to the BEGIN is the previous key's cut line, not a line prefix: taken
626
+ // for one, it stripped this key's identical body line and one copy of a repeated cut key went
627
+ // per pass (v6.19.2 pre-tag round-5 review F2).
628
+ let run = 0;
629
+ while (run < 16 && b - run > ls && isB64Code(text.charCodeAt(b - run - 1))) run++;
630
+ const shape = run >= 16 ? null : prefixShape(text.slice(ls, b));
631
+
632
+ let armors = 0;
633
+ let shorts = 0;
634
+ let pendingBlank = false;
635
+ while (next !== -1) {
636
+ // Another BEGIN line is never part of this key; stop before reading it whole.
637
+ if (!shape && text.startsWith('-----BEGIN ', next)) break;
638
+ const line = nextLine(text, next, u);
639
+ const { core, start, end: coreEnd } = lineCore(text, next, line.end, shape);
640
+ const kind = classify(core);
641
+ next = line.next;
642
+ if (kind === END) {
643
+ if (longs > 0) end = start + KEY_END_LINE_RE.exec(core)[0].length;
644
+ break;
645
+ }
646
+ if (kind === BLANK) {
647
+ if (shorts > 0) break;
648
+ pendingBlank = true;
649
+ continue;
650
+ }
651
+ if (longs === 0) {
652
+ if (kind === ARMOR && ++armors <= MAX_ARMOR_HEADERS) continue;
653
+ if (kind !== LONG && kind !== CUT) break;
654
+ }
655
+ if (kind === LONG && shorts === 0) {
656
+ if (longs++ === 0) first = core;
657
+ longest = Math.max(longest, core.length);
658
+ end = coreEnd;
659
+ pendingBlank = false;
660
+ continue;
661
+ }
662
+ // A cut line ends the key whatever came before it, blank lines included: a PGP or encrypted
663
+ // key has one before its body, and `head` of it in a JSON string ends in a cut line (round-4
664
+ // F1: checked after the blank-line stop, it was never read and the whole key was stored).
665
+ if (kind === CUT && shorts === 0) {
666
+ const run = cutRun(core);
667
+ if (longs++ === 0) first = run;
668
+ longest = Math.max(longest, run.length);
669
+ end = start + run.length;
670
+ break;
671
+ }
672
+ if (pendingBlank) break;
673
+ if (kind === SHORT && (shorts === 0 || (shorts === 1 && PGP_CRC_RE.test(core)))) {
674
+ shorts++;
675
+ end = coreEnd;
676
+ continue;
677
+ }
678
+ break;
679
+ }
680
+ if (longs === 0) return -1;
681
+ const magic = KEY_MAGIC_RE.test(first) || (text.slice(b, be).includes('PGP') && PGP_MAGIC_RE.test(first));
682
+ return longs >= 2 || armors > 0 || longest >= 40 || magic ? end : -1;
683
+ }
684
+
685
+ function scrubCutOffKeys(text) {
686
+ if (!text.includes('PRIVATE KEY')) return text;
687
+ KEY_BEGIN_RE.lastIndex = 0;
688
+ let out = '';
689
+ let last = 0;
690
+ let m;
691
+ while ((m = KEY_BEGIN_RE.exec(text))) {
692
+ const end = cutOffKeyEnd(text, m.index, m.index + m[0].length);
693
+ if (end === -1) continue;
694
+ out += text.slice(last, m.index) + PEM_MARK;
695
+ last = end;
696
+ KEY_BEGIN_RE.lastIndex = end;
697
+ }
698
+ return last === 0 ? text : out + text.slice(last);
699
+ }
700
+
701
+ /**
702
+ * The span [start, end) of the key tail that ends with the END at [e, ee), or null: whole base64
703
+ * lines of 16+ characters directly above it, with one shorter line before the END, or two when the
704
+ * one before the END is a PGP `=XXXX` checksum (delta review P2; two short words above an END are
705
+ * prose, round-3 P3-D). The span takes the END too, unless words precede the END on its line (a
706
+ * sentence naming it): then the lines above go and the sentence stays (F7). `floor` is the end of
707
+ * the previous END, so no line is read twice.
708
+ */
709
+ function keyTailSpan(text, e, ee, floor) {
710
+ let ls = e;
711
+ let u = 0;
712
+ while (ls > floor) {
713
+ const ch = text[ls - 1];
714
+ if (ch === '\n' || ch === '\r') break;
715
+ if (ch === 'n' || ch === 'r') {
716
+ const r = backslashesBefore(text, ls - 1, floor);
717
+ if (r > 0) {
718
+ u = r & -r;
719
+ break;
720
+ }
721
+ }
722
+ ls--;
723
+ }
724
+ let longs = 0;
725
+ let shorts = 0;
726
+ let crc = false;
727
+ let top = -1;
728
+ let topCore = '';
729
+ let longest = 0;
730
+ let bottom = -1;
731
+ const take = (core, start, end) => {
732
+ if (!accept(core, start)) return false;
733
+ if (bottom === -1) bottom = end;
734
+ return true;
735
+ };
736
+ const accept = (core, start) => {
737
+ const kind = classify(core);
738
+ if (kind === LONG) {
739
+ longs++;
740
+ top = start;
741
+ topCore = core;
742
+ if (core.length > longest) longest = core.length;
743
+ return true;
744
+ }
745
+ if (longs > 0 || kind !== SHORT) return false;
746
+ if (shorts === 0) {
747
+ shorts = 1;
748
+ crc = PGP_CRC_RE.test(core);
749
+ return true;
750
+ }
751
+ if (shorts === 1 && crc) {
752
+ shorts = 2;
753
+ return true;
754
+ }
755
+ return false;
756
+ };
757
+ // The END line's own prefix: a line prefix, the key's last base64 run glued to the END, or words.
758
+ // Words with spaces are a line prefix when the line above starts the same way (`web-1 | `, a
759
+ // syslog stamp, `> > `; v6.19.2 pre-tag delta review F2), and a sentence naming the END if not.
760
+ const prefix = text.slice(ls, e);
761
+ const trimmed = prefix.trim();
762
+ let shape = null;
763
+ let sentence = false;
764
+ let prefixed = false;
765
+ if (/^[ \t]*[A-Za-z0-9+/=]+$/.test(prefix)) {
766
+ const at = ls + prefix.indexOf(trimmed);
767
+ if (!take(trimmed, at, at + trimmed.length)) return null;
768
+ } else {
769
+ shape = prefixShape(prefix);
770
+ const above = shape && prevLine(text, ls, u, floor);
771
+ prefixed = Boolean(above && shape.test(text.slice(above.start, above.end)));
772
+ if (/\S\s+\S/.test(trimmed) && !prefixed) {
773
+ sentence = true;
774
+ shape = null;
775
+ }
776
+ }
777
+ // An END alone on its line (after nothing but a line prefix, before nothing but a quote,
778
+ // punctuation or a closing tag) is evidence enough for one line over it that has a digit, a `+`
779
+ // or `=` padding: `tail -n 2` of a key whose last line is 16-39 characters (delta F3). A 16+
780
+ // run of random base64 almost always has one; a camelCase identifier or a path has none
781
+ // (round-3 F3). An END in a sentence or in inline code followed by words is not alone.
782
+ CLEAN_AFTER_END_RE.lastIndex = ee;
783
+ const clean = !sentence && (trimmed === '' || prefixed) && CLEAN_AFTER_END_RE.test(text);
784
+ let cur = ls;
785
+ for (let line; (line = prevLine(text, cur, u, floor)); cur = line.start) {
786
+ const { core, start, end } = lineCore(text, line.start, line.end, shape);
787
+ if (!take(core, start, end)) break;
788
+ }
789
+ // Otherwise the same evidence a cut-off key needs: one base64 line alone is a key tail only when
790
+ // it is 40+ characters, starts like a key encoding or sits over a PGP checksum; an identifier of
791
+ // 16-39 characters over an END named in prose is not (F7).
792
+ if (longs === 0) return null;
793
+ const b64ish = /[0-9+]|=$/.test(topCore);
794
+ if (!(longs >= 2 || longest >= 40 || crc || (clean && b64ish) || KEY_MAGIC_RE.test(topCore))) return null;
795
+ return [top, sentence ? bottom : ee];
796
+ }
797
+
798
+ function scrubKeyTails(text) {
799
+ if (!text.includes('PRIVATE KEY')) return text;
800
+ KEY_END_RE.lastIndex = 0;
801
+ let out = '';
802
+ let last = 0;
803
+ let floor = 0;
804
+ let m;
805
+ while ((m = KEY_END_RE.exec(text))) {
806
+ const span = keyTailSpan(text, m.index, m.index + m[0].length, Math.max(floor, last));
807
+ if (span) {
808
+ out += text.slice(last, span[0]) + PEM_MARK;
809
+ last = span[1];
810
+ }
811
+ floor = m.index + m[0].length;
812
+ }
813
+ return last === 0 ? text : out + text.slice(last);
814
+ }
815
+
354
816
  /**
355
817
  * Scrub known secret patterns (API keys, tokens, credentials) from text.
356
818
  * Also strips user-marked `<private>...</private>` blocks first, so every