claude-mem-lite 6.19.0 → 6.19.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/hook-optimize.mjs +52 -4
- package/lib/activity.mjs +3 -0
- package/lib/compress-core.mjs +2 -1
- package/lib/deferred-work.mjs +24 -0
- package/lib/lesson-bridge.mjs +2 -1
- package/lib/provenance.mjs +16 -1
- package/lib/save-observation.mjs +5 -2
- package/mem-cli.mjs +3 -3
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
- package/secret-scrub.mjs +476 -16
- package/server.mjs +2 -2
- package/utils.mjs +50 -9
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
"plugins": [
|
|
10
10
|
{
|
|
11
11
|
"name": "claude-mem-lite",
|
|
12
|
-
"version": "6.19.
|
|
12
|
+
"version": "6.19.2",
|
|
13
13
|
"source": "./",
|
|
14
14
|
"homepage": "https://github.com/sdsrss/claude-mem-lite",
|
|
15
15
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.19.
|
|
3
|
+
"version": "6.19.2",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "sdsrss"
|
package/hook-optimize.mjs
CHANGED
|
@@ -32,6 +32,7 @@ import { normalizeScope, SCOPE_PROMPT_LEGEND, insertObservationRow } from './lib
|
|
|
32
32
|
import { liveObsFilterSql } from './lib/inject-search-core.mjs';
|
|
33
33
|
import { resolveRuntimeDir } from './lib/resolve-data-dir.mjs';
|
|
34
34
|
import { MEMORY_INPUT_GUARD } from './lib/memory-input-guard.mjs';
|
|
35
|
+
import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './lib/provenance.mjs';
|
|
35
36
|
|
|
36
37
|
import { DAY_MS } from './lib/time-constants.mjs';
|
|
37
38
|
// P1-14: same resolver as hook-shared.mjs — this was the second module that had never
|
|
@@ -582,12 +583,34 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
|
|
|
582
583
|
// text over it, wide wrote back the pre-edit narrative it read before the call
|
|
583
584
|
// (tests/verify-reenrich-race.test.mjs). It also stops two overlapping runs rewriting a
|
|
584
585
|
// row twice.
|
|
586
|
+
//
|
|
587
|
+
// Narrow replaces title and narrative with model text, so an explicit save it rewrites (one
|
|
588
|
+
// whose save-time enrich failed) moves to the re-enrich writer's id, as a cluster-merge
|
|
589
|
+
// keeper does (D#146, D#138). Wide keeps the stored title and narrative (it fills the lesson
|
|
590
|
+
// and the side fields), so the row stays an explicit save. The writer's session row is
|
|
591
|
+
// best-effort.
|
|
592
|
+
let rewriteSessionId = null;
|
|
593
|
+
if (!isWide) {
|
|
594
|
+
const cur = db.prepare('SELECT memory_session_id FROM observations WHERE id = ?').get(cand.id);
|
|
595
|
+
if (cur?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
|
|
596
|
+
const enrichSessionId = writerSessionId('enrich-', cand.project);
|
|
597
|
+
try {
|
|
598
|
+
db.prepare(
|
|
599
|
+
`INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
|
|
600
|
+
VALUES (?, ?, ?, ?, ?, 'active')`,
|
|
601
|
+
).run(enrichSessionId, enrichSessionId, cand.project, new Date().toISOString(), Date.now());
|
|
602
|
+
rewriteSessionId = enrichSessionId;
|
|
603
|
+
} catch (e) {
|
|
604
|
+
debugCatch(e, 'reenrich writer session');
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
}
|
|
585
608
|
const res = db
|
|
586
609
|
.prepare(
|
|
587
610
|
`
|
|
588
611
|
UPDATE observations SET type=?, title=?, narrative=?, concepts=?, facts=?,
|
|
589
612
|
text=?, importance=?, lesson_learned=?, search_aliases=?, minhash_sig=?, optimized_at=?,
|
|
590
|
-
scope=COALESCE(?, scope)
|
|
613
|
+
scope=COALESCE(?, scope), memory_session_id = COALESCE(?, memory_session_id)
|
|
591
614
|
WHERE id = ? AND ${liveObsFilterSql('')} AND optimized_at IS NULL
|
|
592
615
|
`,
|
|
593
616
|
)
|
|
@@ -607,6 +630,7 @@ scope: ${SCOPE_PROMPT_LEGEND}`;
|
|
|
607
630
|
// scope, or emits an off-enum value, must never blank an existing label —
|
|
608
631
|
// and THIS update stamps optimized_at, so the loss would be permanent.
|
|
609
632
|
normalizeScope(parsed.scope),
|
|
633
|
+
rewriteSessionId,
|
|
610
634
|
cand.id,
|
|
611
635
|
);
|
|
612
636
|
if (res.changes === 0) {
|
|
@@ -1108,7 +1132,7 @@ export function findMergeCandidates(db, maxClusters = 5, { project } = {}) {
|
|
|
1108
1132
|
-- keeper.search_aliases when it rebuilt the keeper's TF-IDF vector. Phase-2 removed that
|
|
1109
1133
|
-- rebuild, so the column had no reader left and went with it. Do NOT re-add it on the
|
|
1110
1134
|
-- strength of R10 P3-7 -- that finding is moot, not pending. executeMergeCluster reads
|
|
1111
|
-
-- keeper.{id,importance,narrative,concepts,facts} and o.{id,title,type,narrative,
|
|
1135
|
+
-- keeper.{id,project,importance,narrative,concepts,facts} and o.{id,title,type,narrative,
|
|
1112
1136
|
-- importance,access_count,lesson_learned}, and nothing else off these rows.
|
|
1113
1137
|
SELECT id, title, narrative, project, type, access_count, importance, created_at_epoch, minhash_sig, lesson_learned, concepts, facts
|
|
1114
1138
|
FROM observations
|
|
@@ -1311,10 +1335,33 @@ Return ONLY valid JSON:
|
|
|
1311
1335
|
SELECT ${snapColList}, ? FROM observations WHERE id = ?`,
|
|
1312
1336
|
).run(keeper.id, keeper.id);
|
|
1313
1337
|
|
|
1338
|
+
// The keeper now holds model text. Search reads authorship from memory_session_id, so an
|
|
1339
|
+
// explicit save's `manual-` id would mark it as one (D#138); it moves to the compression
|
|
1340
|
+
// writer's id. A machine-written keeper keeps its id, and so does the snapshot above, which
|
|
1341
|
+
// is the original save. writerSessionId keeps the id out of the uuid shape sdk_sessions
|
|
1342
|
+
// refuses (D#147); the row stays best-effort, so no refusal can fail the merge (v6.19.1
|
|
1343
|
+
// pre-tag F5a).
|
|
1344
|
+
let rewriteSessionId = null;
|
|
1345
|
+
const keeperSession = db
|
|
1346
|
+
.prepare('SELECT memory_session_id FROM observations WHERE id = ?')
|
|
1347
|
+
.get(keeper.id);
|
|
1348
|
+
if (keeperSession?.memory_session_id?.startsWith(MANUAL_SESSION_ID_PREFIX)) {
|
|
1349
|
+
const compressSessionId = writerSessionId('compress-', keeper.project);
|
|
1350
|
+
try {
|
|
1351
|
+
db.prepare(
|
|
1352
|
+
`INSERT OR IGNORE INTO sdk_sessions (content_session_id, memory_session_id, project, started_at, started_at_epoch, status)
|
|
1353
|
+
VALUES (?, ?, ?, ?, ?, 'active')`,
|
|
1354
|
+
).run(compressSessionId, compressSessionId, keeper.project, new Date().toISOString(), Date.now());
|
|
1355
|
+
rewriteSessionId = compressSessionId;
|
|
1356
|
+
} catch (e) {
|
|
1357
|
+
debugCatch(e, 'cluster-merge writer session');
|
|
1358
|
+
}
|
|
1359
|
+
}
|
|
1314
1360
|
db.prepare(
|
|
1315
1361
|
`
|
|
1316
1362
|
UPDATE observations SET title=?, narrative=?, concepts=?, facts=?, text=?,
|
|
1317
|
-
importance=?, lesson_learned=?, minhash_sig=?, optimized_at
|
|
1363
|
+
importance=?, lesson_learned=?, minhash_sig=?, optimized_at=?,
|
|
1364
|
+
memory_session_id = COALESCE(?, memory_session_id)
|
|
1318
1365
|
WHERE id = ?
|
|
1319
1366
|
`,
|
|
1320
1367
|
).run(
|
|
@@ -1327,6 +1374,7 @@ Return ONLY valid JSON:
|
|
|
1327
1374
|
safe.lesson_learned,
|
|
1328
1375
|
minhashSig,
|
|
1329
1376
|
Date.now(),
|
|
1377
|
+
rewriteSessionId,
|
|
1330
1378
|
keeper.id,
|
|
1331
1379
|
);
|
|
1332
1380
|
|
|
@@ -1555,7 +1603,7 @@ export async function executeSmartCompressCluster(db, observations, project) {
|
|
|
1555
1603
|
) {
|
|
1556
1604
|
return null;
|
|
1557
1605
|
}
|
|
1558
|
-
const sessionId =
|
|
1606
|
+
const sessionId = writerSessionId('compress-', project);
|
|
1559
1607
|
const now = new Date();
|
|
1560
1608
|
db.prepare(
|
|
1561
1609
|
`INSERT OR IGNORE INTO sdk_sessions
|
package/lib/activity.mjs
CHANGED
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
import { sanitizeFtsQuery } from '../utils.mjs';
|
|
8
8
|
import { scrubRecord, scrubFilePaths } from './scrub-record.mjs';
|
|
9
9
|
import { saveObservation } from './save-observation.mjs';
|
|
10
|
+
import { writerSessionId } from './provenance.mjs';
|
|
10
11
|
// Pure title-only builder: this query runs on the EVENTS table, which has no
|
|
11
12
|
// lesson_learned column — the lesson-escape variant would be a SQL error here.
|
|
12
13
|
import { buildNotLowSignalSql } from './low-signal-patterns.mjs';
|
|
@@ -199,6 +200,8 @@ export function promoteInsightEvents(
|
|
|
199
200
|
// manual-save contract) so it lands in the high-weight FTS field.
|
|
200
201
|
lesson_learned: ev.body.slice(0, 500),
|
|
201
202
|
now: new Date(ev.created_at_epoch),
|
|
203
|
+
// Event bodies are machine-written (episode summaries), so the row is not an explicit save.
|
|
204
|
+
sessionId: writerSessionId('promote-', ev.project),
|
|
202
205
|
});
|
|
203
206
|
mark.run(Date.now(), ev.id);
|
|
204
207
|
return r;
|
package/lib/compress-core.mjs
CHANGED
|
@@ -17,6 +17,7 @@
|
|
|
17
17
|
|
|
18
18
|
import { isoWeekKey, COMPRESSED_AUTO } from '../utils.mjs';
|
|
19
19
|
import { scrubRecord } from './scrub-record.mjs';
|
|
20
|
+
import { writerSessionId } from './provenance.mjs';
|
|
20
21
|
|
|
21
22
|
/**
|
|
22
23
|
* Low-value compression candidates: importance<=1, never accessed, older than
|
|
@@ -88,7 +89,7 @@ export function compressGroup(db, proj, obs) {
|
|
|
88
89
|
const dominantType = Object.entries(types).sort((a, b) => b[1] - a[1])[0][0];
|
|
89
90
|
const title = `Weekly summary: ${obs.length} ${dominantType} observations`;
|
|
90
91
|
const narrative = obs.map((o) => `- ${o.title || '(untitled)'}`).join('\n');
|
|
91
|
-
const sessionId =
|
|
92
|
+
const sessionId = writerSessionId('compress-', proj);
|
|
92
93
|
|
|
93
94
|
const sortedEpochs = obs.map((o) => o.created_at_epoch).sort((a, b) => a - b);
|
|
94
95
|
const medianEpoch = sortedEpochs[Math.floor(sortedEpochs.length / 2)];
|
package/lib/deferred-work.mjs
CHANGED
|
@@ -86,6 +86,30 @@ export function listOpenWithOrdinal(db, project, limit = 10) {
|
|
|
86
86
|
.all(project, limit);
|
|
87
87
|
}
|
|
88
88
|
|
|
89
|
+
/**
|
|
90
|
+
* One open row's ordinal, numbered as listOpenWithOrdinal numbers it; null when the row is not
|
|
91
|
+
* open. `defer add` / `mem_defer` read it from a 50-row page before, so an item added past 50
|
|
92
|
+
* open printed `(item ?)` (D#123).
|
|
93
|
+
* @param {Database} db
|
|
94
|
+
* @param {string} project
|
|
95
|
+
* @param {number} id
|
|
96
|
+
* @returns {number|null}
|
|
97
|
+
*/
|
|
98
|
+
export function openOrdinalOf(db, project, id) {
|
|
99
|
+
const row = db
|
|
100
|
+
.prepare(
|
|
101
|
+
`
|
|
102
|
+
SELECT ordinal FROM (
|
|
103
|
+
SELECT id, ROW_NUMBER() OVER (ORDER BY priority DESC, created_at_epoch ASC, id ASC) AS ordinal
|
|
104
|
+
FROM deferred_work
|
|
105
|
+
WHERE project = ? AND status = 'open'
|
|
106
|
+
) WHERE id = ?
|
|
107
|
+
`,
|
|
108
|
+
)
|
|
109
|
+
.get(project, id);
|
|
110
|
+
return row ? row.ordinal : null;
|
|
111
|
+
}
|
|
112
|
+
|
|
89
113
|
/**
|
|
90
114
|
* How many open rows `defer list` / `mem_defer_list` left off their page, as a line to
|
|
91
115
|
* print — '' when the page held them all. Without it a page one short of the open set
|
package/lib/lesson-bridge.mjs
CHANGED
|
@@ -26,7 +26,8 @@ export function buildBridgePrompt(lesson, hunk) {
|
|
|
26
26
|
|
|
27
27
|
// { ok:true, check } when the bridge produced a usable, applicable check;
|
|
28
28
|
// { ok:false } on N/A / empty / error / timeout. NEVER throws — the caller
|
|
29
|
-
// falls back to the
|
|
29
|
+
// falls back to the arm's plain directive on { ok:false } (VERDICT_DIRECTIVE since 6.17.0,
|
|
30
|
+
// scripts/pre-tool-recall.js ACTIVE_DIRECTIVE).
|
|
30
31
|
export async function bridgeLesson({ lesson, hunk, timeoutMs = 2500, _callLLM = callLLM }) {
|
|
31
32
|
try {
|
|
32
33
|
const raw = await _callLLM(buildBridgePrompt(lesson, hunk), timeoutMs);
|
package/lib/provenance.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// Which writer produced an observation, read back from its memory_session_id. saveObservation
|
|
2
2
|
// writes every explicit save under MANUAL_SESSION_ID_PREFIX. The hook capture (`hook-`),
|
|
3
|
-
// transcript import (`import-`), compression summaries (`compress-`)
|
|
3
|
+
// transcript import (`import-`), compression summaries and cluster-merge keepers (`compress-`),
|
|
4
|
+
// rows narrow re-enrich rewrote (`enrich-`), promoted events (`promote-`) and rows imported from
|
|
4
5
|
// older stores (a bare session uuid) are machine-written.
|
|
5
6
|
//
|
|
6
7
|
// Search and get mark the machine-written side, because explicit saves are the large majority
|
|
@@ -9,6 +10,20 @@
|
|
|
9
10
|
|
|
10
11
|
export const MANUAL_SESSION_ID_PREFIX = 'manual-';
|
|
11
12
|
|
|
13
|
+
// sdk_sessions refuses a row whose two ids are equal and 36 characters long with dashes where a
|
|
14
|
+
// uuid has them (schema.mjs, sdk_sessions_id_mix_check_ai). A writer row uses `<prefix><project>`
|
|
15
|
+
// for both, which has that shape for one project length and dash layout (27 characters under
|
|
16
|
+
// `compress-`, 29 under `manual-`), and the refusal failed every write in such a project (D#147).
|
|
17
|
+
// That id gets one more character; every other id is unchanged. Counted in code points, as
|
|
18
|
+
// SQLite's length() and LIKE's `_` count them.
|
|
19
|
+
const UUID_SHAPED_RE = /^[\s\S]{8}-[\s\S]{4}-[\s\S]{4}-[\s\S]{4}-[\s\S]{12}$/u;
|
|
20
|
+
|
|
21
|
+
/** The session id a non-hook writer (`manual-`, `compress-`, `enrich-`, `promote-`) stores for a project. */
|
|
22
|
+
export function writerSessionId(prefix, project) {
|
|
23
|
+
const id = `${prefix}${project}`;
|
|
24
|
+
return UUID_SHAPED_RE.test(id) ? `${id}~` : id;
|
|
25
|
+
}
|
|
26
|
+
|
|
12
27
|
const AUTO_MARK = '🤖';
|
|
13
28
|
const AUTO_TEXT = 'auto-written, not an explicit save';
|
|
14
29
|
|
package/lib/save-observation.mjs
CHANGED
|
@@ -22,7 +22,7 @@ import { liveObsFilterSql } from './inject-search-core.mjs';
|
|
|
22
22
|
// Imported, not injected: `allowStatuses` below is the POLICY this function exists to hold,
|
|
23
23
|
// and a caller free to pass its own resolver could reinstate the one-way gate D#195 closed.
|
|
24
24
|
import { resolveDeferredIds, closeDeferredItems } from './deferred-work.mjs';
|
|
25
|
-
import { MANUAL_SESSION_ID_PREFIX } from './provenance.mjs';
|
|
25
|
+
import { MANUAL_SESSION_ID_PREFIX, writerSessionId } from './provenance.mjs';
|
|
26
26
|
|
|
27
27
|
const DEDUP_WINDOW_MS = 5 * 60 * 1000;
|
|
28
28
|
const DEDUP_RECENT_LIMIT = 50;
|
|
@@ -144,6 +144,7 @@ export function formatSupersedeSkipped(skipped) {
|
|
|
144
144
|
* @param {string|null} [params.lesson_learned] Caller validates ≤500 chars.
|
|
145
145
|
* @param {boolean} [params.force=false] Skip the near-duplicate window (below).
|
|
146
146
|
* @param {Date} [params.now] Override for tests.
|
|
147
|
+
* @param {string} [params.sessionId] Writer id; default `manual-<project>` (an explicit save).
|
|
147
148
|
* Both result shapes carry `supersededIds` (observations actually tombstoned) and
|
|
148
149
|
* `supersedeSkipped` (requested but NOT tombstoned, each with a `reason`:
|
|
149
150
|
* `malformed-id` | `no-such-observation` | `no-such-event` | `other-project` |
|
|
@@ -198,7 +199,9 @@ export function saveObservation(db, params) {
|
|
|
198
199
|
const safeTitle = scrubSecrets(rawTitle);
|
|
199
200
|
const safeLesson = rawLesson ? scrubSecrets(rawLesson) : null;
|
|
200
201
|
|
|
201
|
-
|
|
202
|
+
// An explicit save by default; a caller storing machine-written text names its own writer
|
|
203
|
+
// (lib/provenance.mjs reads authorship from this id).
|
|
204
|
+
const sessionId = params.sessionId || writerSessionId(MANUAL_SESSION_ID_PREFIX, project);
|
|
202
205
|
|
|
203
206
|
// Ensure session exists (FK constraint). INSERT OR IGNORE makes this safe
|
|
204
207
|
// under concurrent calls.
|
package/mem-cli.mjs
CHANGED
|
@@ -163,6 +163,7 @@ import { aggregateMetrics, readMetrics } from './lib/metrics.mjs';
|
|
|
163
163
|
import {
|
|
164
164
|
insertDeferred,
|
|
165
165
|
listOpenWithOrdinal,
|
|
166
|
+
openOrdinalOf,
|
|
166
167
|
dropDeferred,
|
|
167
168
|
formatDropReasonHint,
|
|
168
169
|
resolveDeferredIds,
|
|
@@ -1448,9 +1449,8 @@ function cmdDeferAdd(db, args) {
|
|
|
1448
1449
|
return;
|
|
1449
1450
|
}
|
|
1450
1451
|
// Compute the freshly-inserted row's ordinal for an immediately-actionable
|
|
1451
|
-
// response ("ok, deferred this as item N")
|
|
1452
|
-
const
|
|
1453
|
-
const ord = open.find((o) => o.id === r.id)?.ordinal ?? '?';
|
|
1452
|
+
// response ("ok, deferred this as item N"), as mem_defer does.
|
|
1453
|
+
const ord = openOrdinalOf(db, project, r.id) ?? '?';
|
|
1454
1454
|
out(`[mem] Deferred as D#${r.id} (item ${ord}) in project "${project}".`);
|
|
1455
1455
|
}
|
|
1456
1456
|
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.19.
|
|
3
|
+
"version": "6.19.2",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "claude-mem-lite",
|
|
9
|
-
"version": "6.19.
|
|
9
|
+
"version": "6.19.2",
|
|
10
10
|
"os": [
|
|
11
11
|
"darwin",
|
|
12
12
|
"linux",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.19.
|
|
3
|
+
"version": "6.19.2",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"packageManager": "npm@10.9.2",
|
package/secret-scrub.mjs
CHANGED
|
@@ -184,32 +184,35 @@ export const SECRET_PATTERNS = [
|
|
|
184
184
|
[/\b(?:xox[bpasr]|xapp|xoxe)-[a-zA-Z0-9-]{10,}\b/g, '***'],
|
|
185
185
|
// Slack incoming-webhook URL — the path after /services/ is the shared secret.
|
|
186
186
|
[/(https:\/\/hooks\.slack\.com\/services\/)[A-Za-z0-9/]+/g, '$1***'],
|
|
187
|
-
// JWT tokens (eyJ...eyJ...)
|
|
188
|
-
//
|
|
189
|
-
// fresh start that
|
|
190
|
-
// (D#130)
|
|
191
|
-
//
|
|
192
|
-
//
|
|
193
|
-
//
|
|
194
|
-
// `-eyJ` (v6.19.0 pre-tag reviews P3-2, delta P3-1).
|
|
187
|
+
// JWT tokens (eyJ...eyJ...), from any `eyJ`: one glued to a prefix (`my-sess-eyJ…`,
|
|
188
|
+
// `session-<uuid>-eyJ…`, `tok_eyJ…`) is still a JWT. Every `eyJ` of one dotless run used to be a
|
|
189
|
+
// fresh start that rescanned the run to its end — quadratic, 9.6 s on 200k chars of `eyJ-`
|
|
190
|
+
// (D#130) — and v6.19.0's bounded lookbehind that fixed it missed a prefix over 40 characters
|
|
191
|
+
// (round-3 review P3-2). Here a failed start consumes its run instead (`|eyJ[\w-]*`, returned
|
|
192
|
+
// unchanged), so the scan resumes after it. Nothing is lost: the first segment cannot contain a
|
|
193
|
+
// `.`, so every later `eyJ` of the same run reaches the same run end and fails the same way.
|
|
195
194
|
[
|
|
196
|
-
/
|
|
197
|
-
'***',
|
|
195
|
+
/eyJ[a-zA-Z0-9_-]{10,}\.eyJ[a-zA-Z0-9_-]{10,}\.[a-zA-Z0-9_-]+\b|eyJ[a-zA-Z0-9_-]*/g,
|
|
196
|
+
(m) => (m.includes('.') ? '***' : m),
|
|
198
197
|
],
|
|
199
198
|
// PEM private key blocks. `[A-Z0-9 ]*` covers every armor label — RSA/EC/DSA/
|
|
200
199
|
// OPENSSH plus ENCRYPTED and PGP (… PRIVATE KEY BLOCK) — that the fixed
|
|
201
200
|
// alternation missed; the block delimiters make FP impossible.
|
|
202
201
|
// The body stops at the next `-----BEGIN ` (D#130): with `[\s\S]*?` every header with no END
|
|
203
202
|
// scanned to the end of the text, on each of scrubSecrets' passes — quadratic, 8.3 s on 500k
|
|
204
|
-
// chars. A block whose END is missing
|
|
205
|
-
//
|
|
206
|
-
//
|
|
207
|
-
// erase the text up to one (delta review P3-2). A header with no END and no later key header
|
|
208
|
-
// is left as it was in v6.18.0.
|
|
203
|
+
// chars. A block whose END is missing is left to the next pattern. Nor does it cross a mark this
|
|
204
|
+
// scrubber wrote: on a later pass, a BEGIN that a scrubbed key used to block reached a far END
|
|
205
|
+
// and erased the prose between (v6.19.2 pre-tag defect review F5).
|
|
209
206
|
[
|
|
210
|
-
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY)[\s\S])
|
|
207
|
+
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY|\*\*\*PEM_KEY\*\*\*)[\s\S])*?-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g,
|
|
211
208
|
'***PEM_KEY***',
|
|
212
209
|
],
|
|
210
|
+
// A cut-off key (`head id_rsa`, a tool output cut mid-key) and a headerless key tail (`tail
|
|
211
|
+
// key.pem`). These are line scanners, not patterns (D#145): three rounds of review each found a
|
|
212
|
+
// line shape the regex versions misread, and each fix to one shape erased prose in another.
|
|
213
|
+
// See scrubCutOffKeys and scrubKeyTails below for the line rules.
|
|
214
|
+
[{ [Symbol.replace]: (text) => scrubCutOffKeys(text) }, null],
|
|
215
|
+
[{ [Symbol.replace]: (text) => scrubKeyTails(text) }, null],
|
|
213
216
|
// Long hex strings in credential assignments (e.g. SECRET_KEY=abc123def456...).
|
|
214
217
|
// `hash` deliberately excluded: `hash: <40hex>` / `hash=<md5>` are git SHAs and
|
|
215
218
|
// checksums (real, preserved data in this hash-heavy repo), not credentials.
|
|
@@ -333,6 +336,463 @@ export const SECRET_PATTERNS = [
|
|
|
333
336
|
// bare-token pattern here: don't — anchor it to a provider prefix instead.
|
|
334
337
|
];
|
|
335
338
|
|
|
339
|
+
// ─── Cut-off private keys (D#145) ──────────────────────────────────────────
|
|
340
|
+
// A key's text reaches the scrubber in many line shapes: plain LF/CRLF/CR lines, JSON-escaped
|
|
341
|
+
// breaks (`\n` as two characters, or `\\n` when serialised twice), lines carrying a prefix (the
|
|
342
|
+
// Read tool's ` 2\t` or `2→`, grep's `id_rsa:`, a `> ` quote, a diff `-`), and lines inside a
|
|
343
|
+
// quoted string. The scanners read the text as lines of ONE shape per key, decided at the key:
|
|
344
|
+
// - the break after the BEGIN line (or before the END line) says whether breaks are real or
|
|
345
|
+
// escaped, and at which depth; in real-break text a backslash is never a break, so a Windows
|
|
346
|
+
// path in a `Comment:` value no longer ends the header (v6.19.1 round-3 P3-E);
|
|
347
|
+
// - the text before the BEGIN (or END) on its line is the line prefix, and every other line
|
|
348
|
+
// loses a prefix of the same shape (digits may differ, `:` and `-` swap for grep context)
|
|
349
|
+
// before it is judged;
|
|
350
|
+
// - lines end at breaks only. A header value that runs on into the next JSON fields is still
|
|
351
|
+
// one header line, so it erases nothing without a base64 line under it (round-3 P3-C).
|
|
352
|
+
// A body line is WHOLE base64: 16+ characters, or 1-15 for the last line. So a word under a key
|
|
353
|
+
// body keeps its text (round-3 P3-A: `Don't` lost `Don`), as does a path or an identifier that
|
|
354
|
+
// starts the next line (P3-B), and a body needs one long line, so a header followed by words is
|
|
355
|
+
// not a key (F3). A string's closing quote, a backtick or a closing tag after the line is not part
|
|
356
|
+
// of it. A line whose base64 run is followed by something else is a cut or annotated key line
|
|
357
|
+
// when the run is 40+ characters or a truncation mark or delimiter follows it (`…`, `...`,
|
|
358
|
+
// `[truncated]`, `<`): its base64 goes and the rest stays (delta review P3-2; v6.19.2 pre-tag
|
|
359
|
+
// defect review F4). A shorter run followed by words starts a line of prose.
|
|
360
|
+
|
|
361
|
+
const KEY_BEGIN_RE = /-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
|
|
362
|
+
const KEY_END_RE = /-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/g;
|
|
363
|
+
const KEY_END_LINE_RE = /^-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----/;
|
|
364
|
+
const B64_LONG_RE = /^[A-Za-z0-9+/=]{16,}$/;
|
|
365
|
+
const B64_SHORT_RE = /^[A-Za-z0-9+/=]{1,15}$/;
|
|
366
|
+
// Space-separated chunks, allowed only on the BEGIN line itself (a key pasted onto one line).
|
|
367
|
+
const B64_CHUNKS_RE = /^[A-Za-z0-9+/=]+(?:[ \t]+[A-Za-z0-9+/=]+)*$/;
|
|
368
|
+
const B64_RUN_RE = /^[A-Za-z0-9+/=]{16,}/;
|
|
369
|
+
// What may follow a key line that was cut or annotated: a truncation mark or a delimiter. A quote
|
|
370
|
+
// ends the string a key sits in (`…', 'rc': 0}`, `…","stderr":…`): the scanners do not track which
|
|
371
|
+
// quote opened a string, since no character before a quote tells an opening quote from prose
|
|
372
|
+
// (`Here's`, `Run 'head id_rsa'`, `b'…'`; v6.19.2 pre-tag reviews, delta F1 and round-3 F1/F2).
|
|
373
|
+
// Two escapes deep a string ends at `\"` (round-4 F2). Five dashes are the next key's marker glued
|
|
374
|
+
// to a cut line (`for i in …; do head -c 80 k; done`).
|
|
375
|
+
const CUT_MARK_RE = /^(?: ?…| ?\.\.\.| ?\[| ?<|[`"']|\\+["']|-----)/;
|
|
376
|
+
const PGP_CRC_RE = /^=[A-Za-z0-9+/]{4}$/;
|
|
377
|
+
// How a key's base64 starts: a DER SEQUENCE (PKCS#1, PKCS#8, SEC1) or OpenSSH's `openssh-key-v1`;
|
|
378
|
+
// under a PGP header, a secret-key packet in the old or new format (`lQ…`, `xc…`/`xV…`).
|
|
379
|
+
const KEY_MAGIC_RE = /^(?:MII|MIG|MC4C|MHcC|b3BlbnNzaC1rZXktdjE)/;
|
|
380
|
+
const PGP_MAGIC_RE = /^(?:lQ|x[cV])/;
|
|
381
|
+
// RFC 1421 / RFC 4880 armor headers, before the body. Named, not any `Word:`: a `Note:` line is
|
|
382
|
+
// prose, and taking it for a header made the lone base64 line under it a key.
|
|
383
|
+
// RFC 1421's full set and tool-written `X-` headers count too (v6.19.2 pre-tag delta review F4:
|
|
384
|
+
// `Content-Domain` or `X-Custom` before the body stored the whole key).
|
|
385
|
+
const ARMOR_HEADER_RE =
|
|
386
|
+
/^(?:Proc-Type|DEK-Info|Content-Domain|Originator-ID-(?:Asymmetric|Symmetric)|Originator-Certificate|Issuer-Certificate|MIC-Info|Key-Info|Recipient-ID-(?:Asymmetric|Symmetric)|CRL|Version|Comment|Hash|Charset|MessageID|X-[A-Za-z0-9-]+)[ \t]*:/i;
|
|
387
|
+
const MAX_ARMOR_HEADERS = 16;
|
|
388
|
+
// A JS/Python string split across source lines: `…\n" +` then `"…` on the next line. Not a comma:
|
|
389
|
+
// `'…\n',` then `'…'` is the next element of a list, and its first word is not the key's last line.
|
|
390
|
+
// One whitespace quantifier on each side of the `+`: `[ \t]*\+?[ \t]*` split a run between two
|
|
391
|
+
// and was quadratic when no break followed (v6.19.2 pre-tag defect review F1: 16-19 s at 200k).
|
|
392
|
+
const CONCAT_AFTER_RE = /["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']/y;
|
|
393
|
+
const CONCAT_BEFORE_RE = /(?:\\r)?\\n["'](?:[ \t]*\+)?[ \t]*(?:\r\n|\n|\r)[ \t]*["']$/;
|
|
394
|
+
const PEM_MARK = '***PEM_KEY***';
|
|
395
|
+
// After an END that ends its line: a closing quote or punctuation, then a break or the text end.
|
|
396
|
+
// Closing tags may be nested (`END</code></pre>`; round-4 F3), and a string's closing quote, escaped
|
|
397
|
+
// or not, ends the line whatever follows it: `…END-----","stderr":""` is `tail -n 2` in a tool
|
|
398
|
+
// result (round-5 F3).
|
|
399
|
+
const CLEAN_AFTER_END_RE = /(?:[ \t`,;)\]}]|<\/[A-Za-z][\w:.-]{0,40}>)*(?:\r|\n|\\+[nr]|\\*["']|$)/y;
|
|
400
|
+
|
|
401
|
+
const BLANK = 0;
|
|
402
|
+
const LONG = 1;
|
|
403
|
+
const SHORT = 2;
|
|
404
|
+
const END = 3;
|
|
405
|
+
const ARMOR = 4;
|
|
406
|
+
const CUT = 5;
|
|
407
|
+
const OTHER = 6;
|
|
408
|
+
|
|
409
|
+
function backslashesBefore(text, i, floor) {
|
|
410
|
+
let j = i;
|
|
411
|
+
while (j > floor && text[j - 1] === '\\') j--;
|
|
412
|
+
return i - j;
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Break depth `u`: 0 = real breaks only, 1 = `\n`, 2 = `\\n`, … A run of r backslashes then `n`
|
|
416
|
+
// is a break at depth u when r % 2u === u (the backslashes before it are escaped backslashes).
|
|
417
|
+
const isEscBreak = (r, u) => u > 0 && r % (2 * u) === u;
|
|
418
|
+
// An unescaped quote at depth u ends the string: r % 2u < u.
|
|
419
|
+
const isStringEnd = (r, u) => u > 0 && r % (2 * u) < u;
|
|
420
|
+
|
|
421
|
+
/** The line starting at s: its end, and where the next line starts (-1 at the end of the text). */
|
|
422
|
+
function nextLine(text, s, u) {
|
|
423
|
+
const n = text.length;
|
|
424
|
+
let i = s;
|
|
425
|
+
while (i < n) {
|
|
426
|
+
const ch = text[i];
|
|
427
|
+
if (ch === '\n') return { end: i, next: i + 1 };
|
|
428
|
+
if (ch === '\r') return { end: i, next: text[i + 1] === '\n' ? i + 2 : i + 1 };
|
|
429
|
+
if (u > 0 && ch === '\\') {
|
|
430
|
+
let j = i;
|
|
431
|
+
while (j < n && text[j] === '\\') j++;
|
|
432
|
+
const r = j - i;
|
|
433
|
+
const c = text[j];
|
|
434
|
+
if ((c === 'n' || c === 'r') && isEscBreak(r, u)) {
|
|
435
|
+
let next = j + 1;
|
|
436
|
+
if (c === 'r' && text.startsWith('\\'.repeat(u) + 'n', next)) next += u + 1;
|
|
437
|
+
if (u === 1) {
|
|
438
|
+
CONCAT_AFTER_RE.lastIndex = next;
|
|
439
|
+
const m = CONCAT_AFTER_RE.exec(text);
|
|
440
|
+
if (m) next += m[0].length;
|
|
441
|
+
}
|
|
442
|
+
return { end: j - u, next };
|
|
443
|
+
}
|
|
444
|
+
i = j + 1;
|
|
445
|
+
continue;
|
|
446
|
+
}
|
|
447
|
+
i++;
|
|
448
|
+
}
|
|
449
|
+
return { end: n, next: -1 };
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
/** The line that ends at the break before `ls`, or null when `ls` starts the text or string. */
|
|
453
|
+
function prevLine(text, ls, u, floor) {
|
|
454
|
+
if (ls - 1 < floor) return null;
|
|
455
|
+
const c = text[ls - 1];
|
|
456
|
+
let bs = -1;
|
|
457
|
+
if (c === '\n') bs = ls - 2 >= floor && text[ls - 2] === '\r' ? ls - 2 : ls - 1;
|
|
458
|
+
else if (c === '\r') bs = ls - 1;
|
|
459
|
+
else if ((c === 'n' || c === 'r') && isEscBreak(backslashesBefore(text, ls - 1, floor), u)) {
|
|
460
|
+
bs = ls - 1 - u;
|
|
461
|
+
const r2 = bs - 1 >= floor && text[bs - 1] === 'r' ? backslashesBefore(text, bs - 1, floor) : 0;
|
|
462
|
+
if (c === 'n' && isEscBreak(r2, u)) bs -= u + 1;
|
|
463
|
+
} else if (u === 1 && (c === '"' || c === "'")) {
|
|
464
|
+
const m = CONCAT_BEFORE_RE.exec(text.slice(Math.max(floor, ls - 64), ls));
|
|
465
|
+
if (m) bs = ls - m[0].length;
|
|
466
|
+
}
|
|
467
|
+
if (bs === -1) return null;
|
|
468
|
+
let i = bs - 1;
|
|
469
|
+
while (i >= floor) {
|
|
470
|
+
const ch = text[i];
|
|
471
|
+
if (ch === '\n' || ch === '\r') break;
|
|
472
|
+
if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, i, floor), u)) break;
|
|
473
|
+
if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, i, floor), u)) break;
|
|
474
|
+
i--;
|
|
475
|
+
}
|
|
476
|
+
return { start: i + 1, end: bs };
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
/** A regex for the line prefix `prefix` has, with digits free and grep's `:`/`-` interchangeable. */
|
|
480
|
+
function prefixShape(prefix) {
|
|
481
|
+
const body = prefix.replace(/^[ \t]+/, '');
|
|
482
|
+
if (!body || body.length > 256) return null;
|
|
483
|
+
if (body === '-' || body === '+') return /^[ \t]*[-+ ]/;
|
|
484
|
+
const src = body
|
|
485
|
+
.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')
|
|
486
|
+
.replace(/\d+/g, '\\d+')
|
|
487
|
+
.replace(/[:-]/g, '[:-]');
|
|
488
|
+
return new RegExp(`^[ \\t]*${src}`);
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
/** The judged part of line [s, e): no prefix of the key's shape, no padding, no string quotes. */
|
|
492
|
+
function lineCore(text, s, e, shape) {
|
|
493
|
+
let line = text.slice(s, e);
|
|
494
|
+
let off = s;
|
|
495
|
+
if (shape) {
|
|
496
|
+
const m = shape.exec(line);
|
|
497
|
+
if (m) {
|
|
498
|
+
line = line.slice(m[0].length);
|
|
499
|
+
off += m[0].length;
|
|
500
|
+
}
|
|
501
|
+
}
|
|
502
|
+
let a = 0;
|
|
503
|
+
let b = line.length;
|
|
504
|
+
while (a < b && (line[a] === ' ' || line[a] === '\t')) a++;
|
|
505
|
+
while (b > a && (line[b - 1] === ' ' || line[b - 1] === '\t')) b--;
|
|
506
|
+
if (a < b && (line[a] === '"' || line[a] === "'" || line[a] === '`')) a++;
|
|
507
|
+
// A closing quote or backtick with what may follow it (`",`, `" +`, `')`), or a closing tag
|
|
508
|
+
// (`</key>`), and escaped breaks before it. Read from the end: an unanchored
|
|
509
|
+
// `(?:\\+[rn])*["']…$` retried every start in a backslash run.
|
|
510
|
+
let q = b;
|
|
511
|
+
while (q > a && /[ \t+,;)\]}]/.test(line[q - 1])) q--;
|
|
512
|
+
const tag = q > a && line[q - 1] === '>' ? /<\/[A-Za-z][\w:.-]{0,40}>$/.exec(line.slice(a, q)) : null;
|
|
513
|
+
if (tag) b = q - tag[0].length;
|
|
514
|
+
else if (q > a && (line[q - 1] === '"' || line[q - 1] === "'" || line[q - 1] === '`')) {
|
|
515
|
+
q--;
|
|
516
|
+
for (;;) {
|
|
517
|
+
if (q - 1 <= a || (line[q - 1] !== 'n' && line[q - 1] !== 'r')) break;
|
|
518
|
+
let k = q - 1;
|
|
519
|
+
while (k > a && line[k - 1] === '\\') k--;
|
|
520
|
+
if (k === q - 1) break;
|
|
521
|
+
q = k;
|
|
522
|
+
}
|
|
523
|
+
b = q;
|
|
524
|
+
}
|
|
525
|
+
return { core: line.slice(a, b), start: off + a, end: off + b };
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
function classify(core) {
|
|
529
|
+
if (core === '') return BLANK;
|
|
530
|
+
if (B64_LONG_RE.test(core)) return LONG;
|
|
531
|
+
if (B64_SHORT_RE.test(core)) return SHORT;
|
|
532
|
+
if (KEY_END_LINE_RE.test(core)) return END;
|
|
533
|
+
if (ARMOR_HEADER_RE.test(core)) return ARMOR;
|
|
534
|
+
if (cutRun(core)) return CUT;
|
|
535
|
+
return OTHER;
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
/**
|
|
539
|
+
* The base64 run a cut or annotated key line starts with, or null: 16+ characters followed by a
|
|
540
|
+
* truncation mark or a delimiter (`…`, `...`, `[truncated]`, `<`, a backtick), or 40+ followed by
|
|
541
|
+
* anything (`<64> see above`). A shorter run followed by text is an identifier or a path starting
|
|
542
|
+
* a line of prose (`exportedArmoredPrivateKey = …`, round-3 P3-B), which stays.
|
|
543
|
+
*/
|
|
544
|
+
function cutRun(core) {
|
|
545
|
+
const m = B64_RUN_RE.exec(core);
|
|
546
|
+
if (!m) return null;
|
|
547
|
+
return m[0].length >= 40 || CUT_MARK_RE.test(core.slice(m[0].length)) ? m[0] : null;
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
/**
|
|
551
|
+
* Where the key that starts with the BEGIN at [b, be) ends, or -1 when no key material follows it.
|
|
552
|
+
* Header lines and blank lines may come first; then 16+-character base64 lines (blank lines only
|
|
553
|
+
* between them), one shorter last line (and a PGP `=XXXX` checksum after it), and the END if it
|
|
554
|
+
* is there. One base64 line alone is a key only when it is 40+ characters, starts the way a key
|
|
555
|
+
* encoding starts (DER `MII…`, OpenSSH `b3BlbnNzaC1rZXktdjE…`) or follows armor headers: a path
|
|
556
|
+
* or an identifier of 16-39 characters under a header is prose (delta review P3-5). A complete
|
|
557
|
+
* block never gets here; the block pattern above takes it first.
|
|
558
|
+
*/
|
|
559
|
+
function cutOffKeyEnd(text, b, be) {
|
|
560
|
+
// The rest of the BEGIN line: nothing, or base64 chunks, then a break that sets the depth.
|
|
561
|
+
let i = be;
|
|
562
|
+
while (i < text.length && /[A-Za-z0-9+/= \t]/.test(text[i])) i++;
|
|
563
|
+
const rest = text.slice(be, i).trim();
|
|
564
|
+
let u;
|
|
565
|
+
let next;
|
|
566
|
+
if (i >= text.length) return -1;
|
|
567
|
+
if (text[i] === '\n' || text[i] === '\r') {
|
|
568
|
+
u = 0;
|
|
569
|
+
next = text[i] === '\r' && text[i + 1] === '\n' ? i + 2 : i + 1;
|
|
570
|
+
} else if (text[i] === '\\') {
|
|
571
|
+
let j = i;
|
|
572
|
+
while (j < text.length && text[j] === '\\') j++;
|
|
573
|
+
const r = j - i;
|
|
574
|
+
if (text[j] !== 'n' && text[j] !== 'r') return -1;
|
|
575
|
+
u = r & -r;
|
|
576
|
+
if (r !== u) return -1; // a literal backslash on the BEGIN line: not a key line
|
|
577
|
+
({ next } = nextLine(text, i, u));
|
|
578
|
+
} else return -1;
|
|
579
|
+
let longs = 0;
|
|
580
|
+
let longest = 0;
|
|
581
|
+
let first = '';
|
|
582
|
+
let end = -1;
|
|
583
|
+
if (rest) {
|
|
584
|
+
// A key pasted onto its BEGIN line: every chunk but the last is a full 16+ line. Words there
|
|
585
|
+
// are prose, even one of 16+ letters (v6.19.2 pre-tag defect review F9). A loop, not a spread:
|
|
586
|
+
// Math.max(...chunks) overflowed the stack past ~125k chunks (F2).
|
|
587
|
+
if (!B64_CHUNKS_RE.test(rest)) return -1;
|
|
588
|
+
const chunks = rest.split(/[ \t]+/);
|
|
589
|
+
for (let k = 0; k < chunks.length; k++) {
|
|
590
|
+
if (k < chunks.length - 1 && chunks[k].length < 16) return -1;
|
|
591
|
+
if (chunks[k].length > longest) longest = chunks[k].length;
|
|
592
|
+
}
|
|
593
|
+
if (longest < 16) return -1;
|
|
594
|
+
longs = 1;
|
|
595
|
+
first = chunks[0];
|
|
596
|
+
end = be + text.slice(be, i).trimEnd().length;
|
|
597
|
+
}
|
|
598
|
+
// The BEGIN line's prefix.
|
|
599
|
+
let ls = b;
|
|
600
|
+
while (ls > 0) {
|
|
601
|
+
const ch = text[ls - 1];
|
|
602
|
+
if (ch === '\n' || ch === '\r') break;
|
|
603
|
+
if (u > 0 && (ch === 'n' || ch === 'r') && isEscBreak(backslashesBefore(text, ls - 1, 0), u)) break;
|
|
604
|
+
if (u > 0 && (ch === '"' || ch === "'") && isStringEnd(backslashesBefore(text, ls - 1, 0), u)) break;
|
|
605
|
+
ls--;
|
|
606
|
+
}
|
|
607
|
+
// A 16+ base64 run glued to the BEGIN is the previous key's cut line, not a line prefix: taken
|
|
608
|
+
// for one, it stripped this key's identical body line and one copy of a repeated cut key went
|
|
609
|
+
// per pass (v6.19.2 pre-tag round-5 review F2).
|
|
610
|
+
let run = 0;
|
|
611
|
+
while (run < 16 && b - run > ls && /[A-Za-z0-9+/=]/.test(text[b - run - 1])) run++;
|
|
612
|
+
const shape = run >= 16 ? null : prefixShape(text.slice(ls, b));
|
|
613
|
+
|
|
614
|
+
let armors = 0;
|
|
615
|
+
let shorts = 0;
|
|
616
|
+
let pendingBlank = false;
|
|
617
|
+
while (next !== -1) {
|
|
618
|
+
const line = nextLine(text, next, u);
|
|
619
|
+
const { core, start, end: coreEnd } = lineCore(text, next, line.end, shape);
|
|
620
|
+
const kind = classify(core);
|
|
621
|
+
next = line.next;
|
|
622
|
+
if (kind === END) {
|
|
623
|
+
if (longs > 0) end = start + KEY_END_LINE_RE.exec(core)[0].length;
|
|
624
|
+
break;
|
|
625
|
+
}
|
|
626
|
+
if (kind === BLANK) {
|
|
627
|
+
if (shorts > 0) break;
|
|
628
|
+
pendingBlank = true;
|
|
629
|
+
continue;
|
|
630
|
+
}
|
|
631
|
+
if (longs === 0) {
|
|
632
|
+
if (kind === ARMOR && ++armors <= MAX_ARMOR_HEADERS) continue;
|
|
633
|
+
if (kind !== LONG && kind !== CUT) break;
|
|
634
|
+
}
|
|
635
|
+
if (kind === LONG && shorts === 0) {
|
|
636
|
+
if (longs++ === 0) first = core;
|
|
637
|
+
longest = Math.max(longest, core.length);
|
|
638
|
+
end = coreEnd;
|
|
639
|
+
pendingBlank = false;
|
|
640
|
+
continue;
|
|
641
|
+
}
|
|
642
|
+
// A cut line ends the key whatever came before it, blank lines included: a PGP or encrypted
|
|
643
|
+
// key has one before its body, and `head` of it in a JSON string ends in a cut line (round-4
|
|
644
|
+
// F1: checked after the blank-line stop, it was never read and the whole key was stored).
|
|
645
|
+
if (kind === CUT && shorts === 0) {
|
|
646
|
+
const run = cutRun(core);
|
|
647
|
+
if (longs++ === 0) first = run;
|
|
648
|
+
longest = Math.max(longest, run.length);
|
|
649
|
+
end = start + run.length;
|
|
650
|
+
break;
|
|
651
|
+
}
|
|
652
|
+
if (pendingBlank) break;
|
|
653
|
+
if (kind === SHORT && (shorts === 0 || (shorts === 1 && PGP_CRC_RE.test(core)))) {
|
|
654
|
+
shorts++;
|
|
655
|
+
end = coreEnd;
|
|
656
|
+
continue;
|
|
657
|
+
}
|
|
658
|
+
break;
|
|
659
|
+
}
|
|
660
|
+
if (longs === 0) return -1;
|
|
661
|
+
const magic = KEY_MAGIC_RE.test(first) || (text.slice(b, be).includes('PGP') && PGP_MAGIC_RE.test(first));
|
|
662
|
+
return longs >= 2 || armors > 0 || longest >= 40 || magic ? end : -1;
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
function scrubCutOffKeys(text) {
|
|
666
|
+
if (!text.includes('PRIVATE KEY')) return text;
|
|
667
|
+
KEY_BEGIN_RE.lastIndex = 0;
|
|
668
|
+
let out = '';
|
|
669
|
+
let last = 0;
|
|
670
|
+
let m;
|
|
671
|
+
while ((m = KEY_BEGIN_RE.exec(text))) {
|
|
672
|
+
const end = cutOffKeyEnd(text, m.index, m.index + m[0].length);
|
|
673
|
+
if (end === -1) continue;
|
|
674
|
+
out += text.slice(last, m.index) + PEM_MARK;
|
|
675
|
+
last = end;
|
|
676
|
+
KEY_BEGIN_RE.lastIndex = end;
|
|
677
|
+
}
|
|
678
|
+
return last === 0 ? text : out + text.slice(last);
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
/**
|
|
682
|
+
* The span [start, end) of the key tail that ends with the END at [e, ee), or null: whole base64
|
|
683
|
+
* lines of 16+ characters directly above it, with one shorter line before the END, or two when the
|
|
684
|
+
* one before the END is a PGP `=XXXX` checksum (delta review P2; two short words above an END are
|
|
685
|
+
* prose, round-3 P3-D). The span takes the END too, unless words precede the END on its line (a
|
|
686
|
+
* sentence naming it): then the lines above go and the sentence stays (F7). `floor` is the end of
|
|
687
|
+
* the previous END, so no line is read twice.
|
|
688
|
+
*/
|
|
689
|
+
function keyTailSpan(text, e, ee, floor) {
|
|
690
|
+
let ls = e;
|
|
691
|
+
let u = 0;
|
|
692
|
+
while (ls > floor) {
|
|
693
|
+
const ch = text[ls - 1];
|
|
694
|
+
if (ch === '\n' || ch === '\r') break;
|
|
695
|
+
if (ch === 'n' || ch === 'r') {
|
|
696
|
+
const r = backslashesBefore(text, ls - 1, floor);
|
|
697
|
+
if (r > 0) {
|
|
698
|
+
u = r & -r;
|
|
699
|
+
break;
|
|
700
|
+
}
|
|
701
|
+
}
|
|
702
|
+
ls--;
|
|
703
|
+
}
|
|
704
|
+
let longs = 0;
|
|
705
|
+
let shorts = 0;
|
|
706
|
+
let crc = false;
|
|
707
|
+
let top = -1;
|
|
708
|
+
let topCore = '';
|
|
709
|
+
let longest = 0;
|
|
710
|
+
let bottom = -1;
|
|
711
|
+
const take = (core, start, end) => {
|
|
712
|
+
if (!accept(core, start)) return false;
|
|
713
|
+
if (bottom === -1) bottom = end;
|
|
714
|
+
return true;
|
|
715
|
+
};
|
|
716
|
+
const accept = (core, start) => {
|
|
717
|
+
const kind = classify(core);
|
|
718
|
+
if (kind === LONG) {
|
|
719
|
+
longs++;
|
|
720
|
+
top = start;
|
|
721
|
+
topCore = core;
|
|
722
|
+
if (core.length > longest) longest = core.length;
|
|
723
|
+
return true;
|
|
724
|
+
}
|
|
725
|
+
if (longs > 0 || kind !== SHORT) return false;
|
|
726
|
+
if (shorts === 0) {
|
|
727
|
+
shorts = 1;
|
|
728
|
+
crc = PGP_CRC_RE.test(core);
|
|
729
|
+
return true;
|
|
730
|
+
}
|
|
731
|
+
if (shorts === 1 && crc) {
|
|
732
|
+
shorts = 2;
|
|
733
|
+
return true;
|
|
734
|
+
}
|
|
735
|
+
return false;
|
|
736
|
+
};
|
|
737
|
+
// The END line's own prefix: a line prefix, the key's last base64 run glued to the END, or words.
|
|
738
|
+
// Words with spaces are a line prefix when the line above starts the same way (`web-1 | `, a
|
|
739
|
+
// syslog stamp, `> > `; v6.19.2 pre-tag delta review F2), and a sentence naming the END if not.
|
|
740
|
+
const prefix = text.slice(ls, e);
|
|
741
|
+
const trimmed = prefix.trim();
|
|
742
|
+
let shape = null;
|
|
743
|
+
let sentence = false;
|
|
744
|
+
let prefixed = false;
|
|
745
|
+
if (/^[ \t]*[A-Za-z0-9+/=]+$/.test(prefix)) {
|
|
746
|
+
const at = ls + prefix.indexOf(trimmed);
|
|
747
|
+
if (!take(trimmed, at, at + trimmed.length)) return null;
|
|
748
|
+
} else {
|
|
749
|
+
shape = prefixShape(prefix);
|
|
750
|
+
const above = shape && prevLine(text, ls, u, floor);
|
|
751
|
+
prefixed = Boolean(above && shape.test(text.slice(above.start, above.end)));
|
|
752
|
+
if (/\S\s+\S/.test(trimmed) && !prefixed) {
|
|
753
|
+
sentence = true;
|
|
754
|
+
shape = null;
|
|
755
|
+
}
|
|
756
|
+
}
|
|
757
|
+
// An END alone on its line (after nothing but a line prefix, before nothing but a quote,
|
|
758
|
+
// punctuation or a closing tag) is evidence enough for one line over it that has a digit, a `+`
|
|
759
|
+
// or `=` padding: `tail -n 2` of a key whose last line is 16-39 characters (delta F3). A 16+
|
|
760
|
+
// run of random base64 almost always has one; a camelCase identifier or a path has none
|
|
761
|
+
// (round-3 F3). An END in a sentence or in inline code followed by words is not alone.
|
|
762
|
+
CLEAN_AFTER_END_RE.lastIndex = ee;
|
|
763
|
+
const clean = !sentence && (trimmed === '' || prefixed) && CLEAN_AFTER_END_RE.test(text);
|
|
764
|
+
let cur = ls;
|
|
765
|
+
for (let line; (line = prevLine(text, cur, u, floor)); cur = line.start) {
|
|
766
|
+
const { core, start, end } = lineCore(text, line.start, line.end, shape);
|
|
767
|
+
if (!take(core, start, end)) break;
|
|
768
|
+
}
|
|
769
|
+
// Otherwise the same evidence a cut-off key needs: one base64 line alone is a key tail only when
|
|
770
|
+
// it is 40+ characters, starts like a key encoding or sits over a PGP checksum; an identifier of
|
|
771
|
+
// 16-39 characters over an END named in prose is not (F7).
|
|
772
|
+
if (longs === 0) return null;
|
|
773
|
+
const b64ish = /[0-9+]|=$/.test(topCore);
|
|
774
|
+
if (!(longs >= 2 || longest >= 40 || crc || (clean && b64ish) || KEY_MAGIC_RE.test(topCore))) return null;
|
|
775
|
+
return [top, sentence ? bottom : ee];
|
|
776
|
+
}
|
|
777
|
+
|
|
778
|
+
function scrubKeyTails(text) {
|
|
779
|
+
if (!text.includes('PRIVATE KEY')) return text;
|
|
780
|
+
KEY_END_RE.lastIndex = 0;
|
|
781
|
+
let out = '';
|
|
782
|
+
let last = 0;
|
|
783
|
+
let floor = 0;
|
|
784
|
+
let m;
|
|
785
|
+
while ((m = KEY_END_RE.exec(text))) {
|
|
786
|
+
const span = keyTailSpan(text, m.index, m.index + m[0].length, Math.max(floor, last));
|
|
787
|
+
if (span) {
|
|
788
|
+
out += text.slice(last, span[0]) + PEM_MARK;
|
|
789
|
+
last = span[1];
|
|
790
|
+
}
|
|
791
|
+
floor = m.index + m[0].length;
|
|
792
|
+
}
|
|
793
|
+
return last === 0 ? text : out + text.slice(last);
|
|
794
|
+
}
|
|
795
|
+
|
|
336
796
|
/**
|
|
337
797
|
* Scrub known secret patterns (API keys, tokens, credentials) from text.
|
|
338
798
|
* Also strips user-marked `<private>...</private>` blocks first, so every
|
package/server.mjs
CHANGED
|
@@ -125,6 +125,7 @@ import { AUTO_MERGE_THRESHOLD } from './lib/dedup-constants.mjs';
|
|
|
125
125
|
import {
|
|
126
126
|
insertDeferred,
|
|
127
127
|
listOpenWithOrdinal,
|
|
128
|
+
openOrdinalOf,
|
|
128
129
|
dropDeferred,
|
|
129
130
|
formatDropReasonHint,
|
|
130
131
|
resolveDeferredIds,
|
|
@@ -1221,8 +1222,7 @@ server.registerTool(
|
|
|
1221
1222
|
});
|
|
1222
1223
|
// Compute the ordinal for the freshly-inserted row so the response is
|
|
1223
1224
|
// immediately actionable ("ok, I deferred this as item 1").
|
|
1224
|
-
const
|
|
1225
|
-
const ord = open.find((o) => o.id === r.id)?.ordinal ?? null;
|
|
1225
|
+
const ord = openOrdinalOf(db, project, r.id);
|
|
1226
1226
|
return {
|
|
1227
1227
|
content: [
|
|
1228
1228
|
{
|
package/utils.mjs
CHANGED
|
@@ -289,13 +289,34 @@ const DESC_SCRUB_WINDOW = 4096;
|
|
|
289
289
|
// before the first remaining `<private>` or private-key BEGIN, so no field shows a span whose end
|
|
290
290
|
// is out of view. Before v6.19.0 an unclosed opener in the window (a Grep line
|
|
291
291
|
// `notes.md:3:<private>bank pin 4412`) was shown as written (v6.19.0 pre-tag reviews P3-1, r3 P2-1).
|
|
292
|
-
//
|
|
293
|
-
|
|
294
|
-
|
|
292
|
+
// A `</private>` or private-key END that comes first is a span whose START is out of view (a
|
|
293
|
+
// nested span, `tail key.pem`), so everything before it may be its inside: the field shows nothing
|
|
294
|
+
// (round-3 P3-1). The PEM half is case-sensitive, like the scrubber's PEM pattern.
|
|
295
|
+
const PRIVATE_MARK_RE = /<\/?[Pp][Rr][Ii][Vv][Aa][Tt][Ee]>|-----(?:BEGIN|END) [A-Z0-9 ]*PRIVATE KEY/;
|
|
296
|
+
// A window edge that cuts a token leaves a fragment shorter than its pattern needs (`ghp_` and 12
|
|
297
|
+
// of its 36 characters), and whitespace collapsing can bring it into view: the cut token is
|
|
298
|
+
// dropped (defect review P3-6, round-3 P3-4/P3-6). A head window with no whitespace keeps it: its
|
|
299
|
+
// start is intact and it is the only part shown. A tail window with no whitespace is all one cut token.
|
|
300
|
+
const isWs = (c) => c === ' ' || c === '\n' || c === '\t' || c === '\r' || /\s/.test(c);
|
|
301
|
+
function dropCutTokenAtEnd(win, next) {
|
|
302
|
+
if (next === undefined || isWs(next)) return win;
|
|
303
|
+
let i = win.length;
|
|
304
|
+
while (i > 0 && !isWs(win[i - 1])) i--;
|
|
305
|
+
return i > 0 ? win.slice(0, i) : win;
|
|
306
|
+
}
|
|
307
|
+
function dropCutTokenAtStart(win, prev) {
|
|
308
|
+
if (prev === undefined || isWs(prev)) return win;
|
|
309
|
+
let i = 0;
|
|
310
|
+
while (i < win.length && !isWs(win[i])) i++;
|
|
311
|
+
return win.slice(i);
|
|
312
|
+
}
|
|
313
|
+
function scrubTruncate(str, max, window = DESC_SCRUB_WINDOW) {
|
|
295
314
|
if (typeof str !== 'string' || str === '') return truncate(str, max);
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
315
|
+
const stripped = stripPrivate(str);
|
|
316
|
+
let win = stripped.slice(0, window);
|
|
317
|
+
const mark = PRIVATE_MARK_RE.exec(win);
|
|
318
|
+
if (mark) win = mark[0][1] === '/' || mark[0].startsWith('-----END') ? '' : win.slice(0, mark.index);
|
|
319
|
+
else win = dropCutTokenAtEnd(win, stripped[window]);
|
|
299
320
|
// The window edge can split a surrogate pair; truncate only guards a cut it makes itself.
|
|
300
321
|
if (/[\uD800-\uDBFF]$/.test(win)) win = win.slice(0, -1);
|
|
301
322
|
return truncate(_scrubSecrets(win), max);
|
|
@@ -314,6 +335,17 @@ function scrubTruncate(str, max) {
|
|
|
314
335
|
// key with no END more than 4096 characters back).
|
|
315
336
|
const PRIVATE_TAG_HINT_RE = /<\/?private>/i;
|
|
316
337
|
const HEAD_ONLY_MAX = 60;
|
|
338
|
+
// The tail window is scrubbed with TAIL_CONTEXT characters before it, which are then dropped: a
|
|
339
|
+
// label cut by the window's edge (`pass|word: <value>`) is seen whole, so its value is not left
|
|
340
|
+
// unlabelled at the window's start, where whitespace collapsing can bring it into view (v6.19.1
|
|
341
|
+
// pre-tag review F1). A replacement inside the context moves the cut by its length change; the
|
|
342
|
+
// token the cut lands in is dropped either way.
|
|
343
|
+
const TAIL_CONTEXT = 256;
|
|
344
|
+
function scrubTailWindow(str, window) {
|
|
345
|
+
const scrubbed = _scrubSecrets(str.slice(-(window + TAIL_CONTEXT)));
|
|
346
|
+
return dropCutTokenAtStart(scrubbed.slice(TAIL_CONTEXT), scrubbed[TAIL_CONTEXT - 1]);
|
|
347
|
+
}
|
|
348
|
+
const oneSpace = (s) => s.replace(/\s+/g, ' ');
|
|
317
349
|
// The early return needs the WHOLE output inside the head window: a long output whose first
|
|
318
350
|
// 4096 characters collapse to a few (whitespace) still has a tail to show (v6.19.0 pre-tag
|
|
319
351
|
// claims review F2).
|
|
@@ -321,11 +353,20 @@ function scrubTruncateEnds(str, max) {
|
|
|
321
353
|
if (typeof str === 'string' && (str.includes('PRIVATE KEY') || PRIVATE_TAG_HINT_RE.test(str))) {
|
|
322
354
|
return scrubTruncate(str, HEAD_ONLY_MAX);
|
|
323
355
|
}
|
|
324
|
-
|
|
325
|
-
|
|
356
|
+
// Up to two windows long, one window covers the whole output: two overlapping windows showed
|
|
357
|
+
// the same text twice (delta review P3-4).
|
|
358
|
+
const window =
|
|
359
|
+
typeof str === 'string' && str.length <= 2 * DESC_SCRUB_WINDOW
|
|
360
|
+
? 2 * DESC_SCRUB_WINDOW
|
|
361
|
+
: DESC_SCRUB_WINDOW;
|
|
362
|
+
// Whitespace runs are one space: a blank-line run no longer spends the budget (or, before the
|
|
363
|
+
// one-window rule, hid behind the window edge).
|
|
364
|
+
const flat = oneSpace(scrubTruncate(str, window, window));
|
|
365
|
+
const whole = typeof str !== 'string' || str.length <= window;
|
|
326
366
|
if (whole && flat.length <= max) return flat;
|
|
327
367
|
const tailLen = Math.floor(max / 2) - 1;
|
|
328
|
-
const tailSrc = whole ? flat : normalizeInline(
|
|
368
|
+
const tailSrc = whole ? flat : oneSpace(normalizeInline(scrubTailWindow(str, window)));
|
|
369
|
+
if (tailSrc === '') return truncate(flat, max);
|
|
329
370
|
// Drop a lone low surrogate the tail's cut may start on.
|
|
330
371
|
const tail = tailSrc.slice(-tailLen).replace(/^[\uDC00-\uDFFF]/, '');
|
|
331
372
|
// A head short enough to escape truncate's own "…" still gets one before the tail.
|