claude-mem-lite 6.18.0 → 6.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +14 -0
- package/README.zh-CN.md +11 -0
- package/format-utils.mjs +4 -2
- package/hook-llm.mjs +1 -0
- package/lib/fast-summary.mjs +3 -2
- package/lib/provenance.mjs +32 -0
- package/lib/save-observation.mjs +2 -1
- package/lib/search-core.mjs +19 -0
- package/mem-cli.mjs +6 -3
- package/npm-shrinkwrap.json +2 -2
- package/package.json +2 -1
- package/search-engine.mjs +23 -3
- package/secret-scrub.mjs +83 -19
- package/server.mjs +6 -5
- package/source-files.mjs +2 -0
- package/utils.mjs +51 -3
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
"plugins": [
|
|
10
10
|
{
|
|
11
11
|
"name": "claude-mem-lite",
|
|
12
|
-
"version": "6.
|
|
12
|
+
"version": "6.19.0",
|
|
13
13
|
"source": "./",
|
|
14
14
|
"homepage": "https://github.com/sdsrss/claude-mem-lite",
|
|
15
15
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark)."
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.19.0",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "sdsrss"
|
package/README.md
CHANGED
|
@@ -237,6 +237,20 @@ rm -rf ~/claude-mem-lite/ # pre-v0.5 unhidden (if not auto-moved)
|
|
|
237
237
|
repos/ # Shallow-cloned source repos
|
|
238
238
|
```
|
|
239
239
|
|
|
240
|
+
## Upgrading to 6.19.0
|
|
241
|
+
|
|
242
|
+
**Search output changes; no switch.** No schema change and no migration, so reverting is
|
|
243
|
+
pinning `claude-mem-lite@6.18.0`.
|
|
244
|
+
|
|
245
|
+
- **`search` / `mem_search` mark machine-written observations with `🤖`**, and `get` /
|
|
246
|
+
`mem_get` name them in the header. Hook-captured, imported and compressed rows carry it;
|
|
247
|
+
explicit saves and events do not. The result line explains the mark whenever a shown row
|
|
248
|
+
has it, and `search --json` adds `auto` to observation rows.
|
|
249
|
+
- **`mem_search` shows a snippet line only when it adds to the title.**
|
|
250
|
+
- Fixes: a lone search match in a new per-project store no longer sorts last; a Bash
|
|
251
|
+
step's stored description keeps the end of its output; the secret scrubber catches values
|
|
252
|
+
behind markdown labels and stays linear-time on crafted input.
|
|
253
|
+
|
|
240
254
|
## Upgrading to 6.18.0
|
|
241
255
|
|
|
242
256
|
**One default changes, with a switch.** No schema change and no migration. Pinning
|
package/README.zh-CN.md
CHANGED
|
@@ -199,6 +199,17 @@ rm -rf ~/claude-mem-lite/ # v0.5 前的非隐藏目录(如未自动迁移)
|
|
|
199
199
|
repos/ # 浅克隆的源代码仓库
|
|
200
200
|
```
|
|
201
201
|
|
|
202
|
+
## 升级到 6.19.0
|
|
203
|
+
|
|
204
|
+
**搜索输出有变化,没有开关。** 没有 schema 变更、不需要迁移,回退只需固定 `claude-mem-lite@6.18.0`。
|
|
205
|
+
|
|
206
|
+
- **`search` / `mem_search` 用 `🤖` 标出机器写入的 observation**,`get` / `mem_get` 在标题行注明。
|
|
207
|
+
hook 采集、导入和压缩生成的行带这个标记;显式保存的和 events 不带。只要显示的行里有带标记的,
|
|
208
|
+
结果行就会解释它的含义;`search --json` 的 observation 行多一个 `auto` 字段。
|
|
209
|
+
- **`mem_search` 只在摘录比标题多出信息时才显示摘录行。**
|
|
210
|
+
- 修复:新的按项目存储里,单条搜索命中不再排在最后;Bash 步骤存下的描述会保留输出的结尾;
|
|
211
|
+
密钥擦除器能识别 markdown 标签后面的值,并且在构造输入下保持线性时间。
|
|
212
|
+
|
|
202
213
|
## 升级到 6.18.0
|
|
203
214
|
|
|
204
215
|
**一处默认行为改变,有开关。** 没有 schema 变更、不需要迁移。固定 `claude-mem-lite@6.17.1`
|
package/format-utils.mjs
CHANGED
|
@@ -92,7 +92,9 @@ export function queryLabel(query) {
|
|
|
92
92
|
|
|
93
93
|
// Two delimiter classes are defanged here:
|
|
94
94
|
// 1. The blocks claude-mem-lite wraps injected context in (claude-mem-context /
|
|
95
|
-
// memory-context / session-handoff
|
|
95
|
+
// memory-context / session-handoff / the handoff's inner session-summary — D#129: a
|
|
96
|
+
// summary carrying `</session-summary><session-summary source="report">` closed the
|
|
97
|
+
// real block and forged its provenance). User-derived text containing one LITERALLY
|
|
96
98
|
// would prematurely open/close the block it lands in, spilling the rest as
|
|
97
99
|
// undelimited context.
|
|
98
100
|
// 2. Harness-authority + tool-call tags the runtime injects (system-reminder /
|
|
@@ -113,7 +115,7 @@ export function queryLabel(query) {
|
|
|
113
115
|
// Reachable by editing files that contain these tokens \u2014 e.g. developing claude-mem-lite
|
|
114
116
|
// itself, where source/observations carry the delimiter names.
|
|
115
117
|
const CONTEXT_DELIMITER_RE =
|
|
116
|
-
/<\/?(?:claude-mem-context|memory-context|session-handoff|system-reminder|task-notification|(?:antml:)?function_calls|(?:antml:)?function_results|(?:antml:)?invoke|(?:antml:)?parameter)(?:\s[^>]*)?>/gi;
|
|
118
|
+
/<\/?(?:claude-mem-context|memory-context|session-handoff|session-summary|system-reminder|task-notification|(?:antml:)?function_calls|(?:antml:)?function_results|(?:antml:)?invoke|(?:antml:)?parameter)(?:\s[^>]*)?>/gi;
|
|
117
119
|
|
|
118
120
|
// Pass cap for the fixpoint loop below. 32 nested layers of a forged delimiter is far past
|
|
119
121
|
// anything prose produces; the cap exists only to bound the ADVERSARIAL cost (an unbounded
|
package/hook-llm.mjs
CHANGED
|
@@ -991,6 +991,7 @@ export async function handleLLMEpisode() {
|
|
|
991
991
|
// events; treating them as a separate role + boundary marker reduces the
|
|
992
992
|
// attack surface for memory poisoning via crafted file content.
|
|
993
993
|
const SHARED_OBS_SCHEMA_TAIL = `${MEMORY_INPUT_GUARD}
|
|
994
|
+
Grounding: state only outcomes the user message shows. An action description may be cut short ("…"); if its result is not visible, say what was run, not how it turned out. Never write that something passed, failed, was verified or confirmed unless that result appears in the user message.
|
|
994
995
|
type: pick by strongest signal. decision = explicit tradeoff / "chose X over Y because Z" / rejected an approach (e.g. "Rejected schema migration — single-source module + sync test instead"; "Heterogeneous hook events → heterogeneous context budgets"). bugfix = prior-failing path fixed with a named root cause. feature = new user-visible capability. refactor = behavior unchanged but structure improved. discovery = learned how a system works (read-heavy, no writes). change = routine edit with no new principle (default if unsure and nothing else fits).
|
|
995
996
|
Facts: each MUST be (1) atomic—one claim, (2) self-contained—no pronouns, include file/function name, (3) specific—"refreshToken() in auth.ts:45 uses 1h TTL" not "handles tokens"
|
|
996
997
|
importance: Be strict — default to 1. 0=pure browsing with zero learning value. 1=routine file edits, standard changes, normal workflow (MOST episodes). 2=notable ONLY if it reveals something non-obvious: error fix with discovered root cause, architectural decision with explicit tradeoff, config change with unexpected side effects. 3=critical: breaking change affecting users, security vulnerability fix, data migration. Ask yourself: "would a future session benefit from knowing this?" — if not, it's importance=1.
|
package/lib/fast-summary.mjs
CHANGED
|
@@ -110,8 +110,9 @@ function replyLines(text) {
|
|
|
110
110
|
|
|
111
111
|
/**
|
|
112
112
|
* Whether the scrubber finds a secret in `text` as written OR with inline markup (`, |, *)
|
|
113
|
-
* removed. Markup can hide a secret from the scrubber —
|
|
114
|
-
*
|
|
113
|
+
* removed. Markup can hide a secret from the scrubber — a key split by a backtick or `*`,
|
|
114
|
+
* `gh|p_…` are missed as written and caught stripped (a markup-wrapped LABEL such as
|
|
115
|
+
* `**password**=…` is caught as written since D#128) — and storing the
|
|
115
116
|
* stripped text instead would leave a split key's tail behind. The caller asks it about the
|
|
116
117
|
* exact text it stores, not the reply as written: a list or quote marker between a label and
|
|
117
118
|
* its value hid the pair from a check on the raw reply, and removing the marker for storage
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
// Which writer produced an observation, read back from its memory_session_id. saveObservation
|
|
2
|
+
// writes every explicit save under MANUAL_SESSION_ID_PREFIX. The hook capture (`hook-`),
|
|
3
|
+
// transcript import (`import-`), compression summaries (`compress-`) and rows imported from
|
|
4
|
+
// older stores (a bare session uuid) are machine-written.
|
|
5
|
+
//
|
|
6
|
+
// Search and get mark the machine-written side, because explicit saves are the large majority
|
|
7
|
+
// of a typical store and a mark on nearly every line would carry no information. An unknown id
|
|
8
|
+
// renders unmarked, as every row did before the mark existed.
|
|
9
|
+
|
|
10
|
+
export const MANUAL_SESSION_ID_PREFIX = 'manual-';
|
|
11
|
+
|
|
12
|
+
const AUTO_MARK = '🤖';
|
|
13
|
+
const AUTO_TEXT = 'auto-written, not an explicit save';
|
|
14
|
+
|
|
15
|
+
/** @param {string|null|undefined} memorySessionId */
|
|
16
|
+
export function isAutoWritten(memorySessionId) {
|
|
17
|
+
return (
|
|
18
|
+
typeof memorySessionId === 'string' &&
|
|
19
|
+
memorySessionId !== '' &&
|
|
20
|
+
!memorySessionId.startsWith(MANUAL_SESSION_ID_PREFIX)
|
|
21
|
+
);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/** Search row tag; `auto` is set on obs rows by attachBodyTokens. */
|
|
25
|
+
export const autoTag = (r) => (r.auto ? ` ${AUTO_MARK}` : '');
|
|
26
|
+
|
|
27
|
+
/** Search result-line legend, shown only when a rendered row carries the tag. */
|
|
28
|
+
export const autoLegend = (rows) => (rows.some((r) => r.auto) ? ` · ${AUTO_MARK} = ${AUTO_TEXT}` : '');
|
|
29
|
+
|
|
30
|
+
/** Get header note for a full observation row. */
|
|
31
|
+
export const autoHeaderNote = (row) =>
|
|
32
|
+
isAutoWritten(row.memory_session_id) ? ` · ${AUTO_MARK} ${AUTO_TEXT}` : '';
|
package/lib/save-observation.mjs
CHANGED
|
@@ -22,6 +22,7 @@ import { liveObsFilterSql } from './inject-search-core.mjs';
|
|
|
22
22
|
// Imported, not injected: `allowStatuses` below is the POLICY this function exists to hold,
|
|
23
23
|
// and a caller free to pass its own resolver could reinstate the one-way gate D#195 closed.
|
|
24
24
|
import { resolveDeferredIds, closeDeferredItems } from './deferred-work.mjs';
|
|
25
|
+
import { MANUAL_SESSION_ID_PREFIX } from './provenance.mjs';
|
|
25
26
|
|
|
26
27
|
const DEDUP_WINDOW_MS = 5 * 60 * 1000;
|
|
27
28
|
const DEDUP_RECENT_LIMIT = 50;
|
|
@@ -197,7 +198,7 @@ export function saveObservation(db, params) {
|
|
|
197
198
|
const safeTitle = scrubSecrets(rawTitle);
|
|
198
199
|
const safeLesson = rawLesson ? scrubSecrets(rawLesson) : null;
|
|
199
200
|
|
|
200
|
-
const sessionId =
|
|
201
|
+
const sessionId = `${MANUAL_SESSION_ID_PREFIX}${project}`;
|
|
201
202
|
|
|
202
203
|
// Ensure session exists (FK constraint). INSERT OR IGNORE makes this safe
|
|
203
204
|
// under concurrent calls.
|
package/lib/search-core.mjs
CHANGED
|
@@ -409,6 +409,20 @@ const SINGLE_MATCH_BANDS = [
|
|
|
409
409
|
[0, -0.25],
|
|
410
410
|
];
|
|
411
411
|
|
|
412
|
+
// #36: the ratio above assumes the lone hit's magnitude says how well it matched. It does not
|
|
413
|
+
// when FTS5 clamped the IDF: a term in at least half of a table's rows gets idf <= 0, which
|
|
414
|
+
// FTS5 replaces with 1e-6, so the raw bm25 is at most 1e-6 × (k1+1 = 2.2) per phrase whatever
|
|
415
|
+
// the match. A new per-project store reaches it on every query (2 observations, term in 1:
|
|
416
|
+
// idf = ln(1.5/1.5) = 0). Such a lone row is scored the way a multi-row source's best already
|
|
417
|
+
// is (-1): of two clamped rows the better was normalized to -1; a single one was sunk to -0.25.
|
|
418
|
+
// The test reads the RAW bm25 on the obs leg (rawScore): FULL_SCORE's multipliers go down to
|
|
419
|
+
// 0.5 × 0.5 × 0.2 × 0.4 = 0.02, so a demoted row with an informative IDF (0.18) scored 1.35e-4
|
|
420
|
+
// and was lifted from last to first when the final score was tested (v6.19.0 pre-tag review
|
|
421
|
+
// P3-3). Session and event multipliers are >= 1 and prompts carry raw bm25, so their final
|
|
422
|
+
// score is tested. An obs row without rawScore keeps the bands. A near-zero IDF that FTS5 did not
|
|
423
|
+
// clamp (a term in very nearly half the rows) can land below the scale too, and says as little.
|
|
424
|
+
const CLAMPED_IDF_SCALE = 1e-4;
|
|
425
|
+
|
|
412
426
|
/**
|
|
413
427
|
* Normalize each source's BM25 scores to [-1, 0] before cross-source merge.
|
|
414
428
|
* Prevents observations (BM25 can reach -40) from systematically outranking
|
|
@@ -437,6 +451,11 @@ export function normalizeCrossSourceScores(results, sourceKey) {
|
|
|
437
451
|
);
|
|
438
452
|
if (srcResults.length === 0) continue;
|
|
439
453
|
if (srcResults.length === 1) {
|
|
454
|
+
const magnitude = src === 'obs' ? srcResults[0].rawScore : srcResults[0].score;
|
|
455
|
+
if (typeof magnitude === 'number' && Math.abs(magnitude) < CLAMPED_IDF_SCALE) {
|
|
456
|
+
srcResults[0].score = -1;
|
|
457
|
+
continue;
|
|
458
|
+
}
|
|
440
459
|
const ratio = globalMaxAbs > 0 ? Math.abs(srcResults[0].score) / globalMaxAbs : 0;
|
|
441
460
|
srcResults[0].score = SINGLE_MATCH_BANDS.find(([floor]) => ratio >= floor)[1];
|
|
442
461
|
continue;
|
package/mem-cli.mjs
CHANGED
|
@@ -25,6 +25,7 @@ import { resolveProject } from './project-utils.mjs';
|
|
|
25
25
|
import { resolveCliProject as cliProject } from './lib/cli-project.mjs';
|
|
26
26
|
import { reRankWithContext } from './search-scoring.mjs';
|
|
27
27
|
import { searchObservationsHybrid } from './search-engine.mjs';
|
|
28
|
+
import { autoHeaderNote, autoLegend, autoTag } from './lib/provenance.mjs';
|
|
28
29
|
import {
|
|
29
30
|
fetchObsDetail,
|
|
30
31
|
fetchPromptDetail,
|
|
@@ -565,6 +566,8 @@ async function cmdSearch(db, args, { llm } = {}) {
|
|
|
565
566
|
importance: r.importance ?? null,
|
|
566
567
|
files_modified: r.files_modified || null,
|
|
567
568
|
body_tokens: r.bodyTokens ?? null,
|
|
569
|
+
// Events carry no session id, so only observations can say who wrote them.
|
|
570
|
+
...(r.source === 'obs' ? { auto: r.auto === true } : {}),
|
|
568
571
|
};
|
|
569
572
|
});
|
|
570
573
|
out(
|
|
@@ -588,7 +591,7 @@ async function cmdSearch(db, args, { llm } = {}) {
|
|
|
588
591
|
// Pluralize on total — "Found 1 of 44 result" reads wrong; the population (44) drives
|
|
589
592
|
// grammatical number, not the page slice (1).
|
|
590
593
|
out(
|
|
591
|
-
`[mem] Found ${countLabel} result${total !== 1 ? 's' : ''} for "${queryLabel(query)}"${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}`,
|
|
594
|
+
`[mem] Found ${countLabel} result${total !== 1 ? 's' : ''} for "${queryLabel(query)}"${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}${autoLegend(paged)}`,
|
|
592
595
|
);
|
|
593
596
|
// `~Nt` = est. tokens to fetch this row's full body via mem_get (attachBodyTokens, paired with
|
|
594
597
|
// MCP). Conditional so a row that skipped enrichment renders cleanly, not "~undefinedt".
|
|
@@ -614,7 +617,7 @@ async function cmdSearch(db, args, { llm } = {}) {
|
|
|
614
617
|
} else {
|
|
615
618
|
const date = fmtDateShort(r.created_at);
|
|
616
619
|
const title = truncate(r.title || r.subtitle || '(untitled)', 80);
|
|
617
|
-
out(`#${r.id} ${typeIcon(r.type)} ${date}${timeStr} ${title}${tok(r)}`);
|
|
620
|
+
out(`#${r.id} ${typeIcon(r.type)}${autoTag(r)} ${date}${timeStr} ${title}${tok(r)}`);
|
|
618
621
|
if (r.lesson_learned) {
|
|
619
622
|
out(` -> ${truncate(r.lesson_learned, 80)}`);
|
|
620
623
|
}
|
|
@@ -793,7 +796,7 @@ function renderObsRows(db, ids, requestedFields) {
|
|
|
793
796
|
const fields = requestedFields || OBS_FIELDS;
|
|
794
797
|
const parts = [];
|
|
795
798
|
for (const r of rows) {
|
|
796
|
-
const lines = [`#${r.id} [${r.type}] ${fmtDateShort(r.created_at)}`];
|
|
799
|
+
const lines = [`#${r.id} [${r.type}] ${fmtDateShort(r.created_at)}${autoHeaderNote(r)}`];
|
|
797
800
|
// Retraction first (shared with mem_get via get-core) — see supersededNotice.
|
|
798
801
|
const retracted = supersededNotice(r);
|
|
799
802
|
if (retracted) lines.push(retracted);
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.19.0",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "claude-mem-lite",
|
|
9
|
-
"version": "6.
|
|
9
|
+
"version": "6.19.0",
|
|
10
10
|
"os": [
|
|
11
11
|
"darwin",
|
|
12
12
|
"linux",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-mem-lite",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.19.0",
|
|
4
4
|
"description": "Persistent long-term memory for Claude Code via MCP — captures coding decisions, bugfixes, and context across sessions. FTS5 BM25 keyword search with episode batching. Single SQLite DB, no external services. A lighter, lower-cost alternative to claude-mem (episode batching + a smaller model; cost savings are an internal estimate, not a measured benchmark).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"packageManager": "npm@10.9.2",
|
|
@@ -82,6 +82,7 @@
|
|
|
82
82
|
"lib/stats-quality.mjs",
|
|
83
83
|
"lib/low-signal-patterns.mjs",
|
|
84
84
|
"lib/private-strip.mjs",
|
|
85
|
+
"lib/provenance.mjs",
|
|
85
86
|
"lib/citation-tracker.mjs",
|
|
86
87
|
"lib/edge-attribution.mjs",
|
|
87
88
|
"lib/file-edge-match.mjs",
|
package/search-engine.mjs
CHANGED
|
@@ -21,6 +21,7 @@ import {
|
|
|
21
21
|
import { citeFactorClause } from './scoring-sql.mjs';
|
|
22
22
|
import { extractPRFTerms, expandQueryByConcepts } from './search-scoring.mjs';
|
|
23
23
|
import { liveObsFilterSql, recencyDecaySql } from './lib/inject-search-core.mjs';
|
|
24
|
+
import { isAutoWritten } from './lib/provenance.mjs';
|
|
24
25
|
|
|
25
26
|
// Scoring expressions — full adds project boost + access bonus; simple is for
|
|
26
27
|
// expansion paths where boost would over-amplify already-loose matches.
|
|
@@ -69,7 +70,8 @@ export function buildObsFtsQuery(scoring, { multiplier, withSnippet, withOffset,
|
|
|
69
70
|
SELECT o.id, o.type, o.title, o.subtitle, o.project, o.created_at, o.created_at_epoch, o.importance,
|
|
70
71
|
o.files_modified, o.lesson_learned,
|
|
71
72
|
${withSnippet ? "snippet(observations_fts, 2, '»', '«', '…', 10) as match_snippet," : ''}
|
|
72
|
-
${scoreExpr}${mult} as score
|
|
73
|
+
${scoreExpr}${mult} as score,
|
|
74
|
+
${OBS_BM25} as raw_bm25
|
|
73
75
|
FROM observations_fts
|
|
74
76
|
JOIN observations o ON observations_fts.rowid = o.id
|
|
75
77
|
WHERE observations_fts MATCH ?
|
|
@@ -330,6 +332,19 @@ export function countSearchTotal(
|
|
|
330
332
|
return total;
|
|
331
333
|
}
|
|
332
334
|
|
|
335
|
+
/**
|
|
336
|
+
* True when an FTS excerpt says something the title does not. snippet() wraps matches in »« and
|
|
337
|
+
* cuts with …, and a save's title is the start of its narrative, so an excerpt that is only a
|
|
338
|
+
* piece of the title repeats it once markers and ellipses are stripped.
|
|
339
|
+
* @param {string|null|undefined} snippet
|
|
340
|
+
* @param {string|null|undefined} title
|
|
341
|
+
*/
|
|
342
|
+
export function snippetAddsInfo(snippet, title) {
|
|
343
|
+
if (typeof snippet !== 'string' || snippet.length <= 10) return false;
|
|
344
|
+
const bare = snippet.replace(/[»«]/g, '').replace(/^…|…$/g, '').trim();
|
|
345
|
+
return !(title || '').includes(bare);
|
|
346
|
+
}
|
|
347
|
+
|
|
333
348
|
export function ftsRowToResult(r, { scoreMultiplier, snippet } = {}) {
|
|
334
349
|
return {
|
|
335
350
|
source: 'obs',
|
|
@@ -346,6 +361,9 @@ export function ftsRowToResult(r, { scoreMultiplier, snippet } = {}) {
|
|
|
346
361
|
created_at: r.created_at,
|
|
347
362
|
created_at_epoch: r.created_at_epoch,
|
|
348
363
|
score: scoreMultiplier ? r.score * scoreMultiplier : r.score,
|
|
364
|
+
// bm25 before FULL_SCORE's multipliers, which can shrink it 50x: only this can say whether
|
|
365
|
+
// FTS5 clamped the IDF (normalizeCrossSourceScores, CLAMPED_IDF_SCALE).
|
|
366
|
+
rawScore: r.raw_bm25,
|
|
349
367
|
files_modified: r.files_modified,
|
|
350
368
|
importance: r.importance,
|
|
351
369
|
lesson_learned: r.lesson_learned,
|
|
@@ -362,7 +380,8 @@ export function ftsRowToResult(r, { scoreMultiplier, snippet } = {}) {
|
|
|
362
380
|
// heavy obs fields are batch-fetched by id HERE rather than carried on every result. The source
|
|
363
381
|
// key is read as `source || _source` because the two render paths disagree (#8654): MCP sets
|
|
364
382
|
// `source`+`text`, CLI sets `_source`+`prompt_text`. estimateTokens floors at 1, so a missing row
|
|
365
|
-
// or empty body yields 1 — never 0/NaN.
|
|
383
|
+
// or empty body yields 1 — never 0/NaN. The same by-id read sets `auto` (lib/provenance.mjs) on
|
|
384
|
+
// the rendered page only, so no result producer has to carry memory_session_id.
|
|
366
385
|
export function attachBodyTokens(db, results) {
|
|
367
386
|
if (!Array.isArray(results) || results.length === 0) return results;
|
|
368
387
|
const obsIds = results
|
|
@@ -373,7 +392,7 @@ export function attachBodyTokens(db, results) {
|
|
|
373
392
|
try {
|
|
374
393
|
const ph = obsIds.map(() => '?').join(',');
|
|
375
394
|
const rows = db
|
|
376
|
-
.prepare(`SELECT id, narrative, facts, text FROM observations WHERE id IN (${ph})`)
|
|
395
|
+
.prepare(`SELECT id, narrative, facts, text, memory_session_id FROM observations WHERE id IN (${ph})`)
|
|
377
396
|
.all(...obsIds);
|
|
378
397
|
for (const row of rows) bodyById.set(row.id, row);
|
|
379
398
|
} catch (e) {
|
|
@@ -385,6 +404,7 @@ export function attachBodyTokens(db, results) {
|
|
|
385
404
|
let parts;
|
|
386
405
|
if (src === 'obs') {
|
|
387
406
|
const row = bodyById.get(r.id) || {};
|
|
407
|
+
r.auto = isAutoWritten(row.memory_session_id);
|
|
388
408
|
parts = [r.title, r.subtitle, r.lesson_learned, row.narrative, row.facts, row.text];
|
|
389
409
|
} else if (src === 'session') {
|
|
390
410
|
parts = [r.request, r.completed, r.working_on];
|
package/secret-scrub.mjs
CHANGED
|
@@ -28,9 +28,41 @@ export const SECRET_PATTERNS = [
|
|
|
28
28
|
// keyword. Allowing a leading `_` catches those while the prose lookbehind still
|
|
29
29
|
// excludes "Marker token: …". `secret` added so a bare SECRET=… with a mixed-alnum
|
|
30
30
|
// value is covered (the hex-only assignment pattern below misses non-hex values).
|
|
31
|
+
//
|
|
32
|
+
// MARKDOWN AROUND A LABEL (D#128). `- **Password**: \`<v>\`` put `**` between the noun and
|
|
33
|
+
// its separator, so no label pattern matched and the value was stored. Models write labels
|
|
34
|
+
// that way all the time. So a label may carry up to three markup characters after the noun
|
|
35
|
+
// (`[*_~\`]`), and up to two of `*_~` between the separator and the value
|
|
36
|
+
// (`**Password:** <v>`). Three rules keep this from widening what counts as a value:
|
|
37
|
+
// - no backtick after the separator: `` `password:` <next word> `` names the label and
|
|
38
|
+
// then goes on in prose, and the next word (or a CJK run with no spaces) was scrubbed;
|
|
39
|
+
// - at most two characters there, so the scrubber's own `***` is never taken for markup
|
|
40
|
+
// (`password=*** token=abc` would scrub `token=abc` on the second pass);
|
|
41
|
+
// - the prose check looks through emphasis (`the **password**: …` is prose) but not a
|
|
42
|
+
// backtick, since `` word `token: <v>` `` is code, not prose. A label wrapped in code
|
|
43
|
+
// (`` `GH_TOKEN`: <v> ``) has its own branch, whose prose check looks past the opening
|
|
44
|
+
// backtick. That branch starts with a cheap lookahead: without it the 40-char lookbehind
|
|
45
|
+
// ran at every position, and a 500k-char input went from 407 ms to 3,387 ms.
|
|
46
|
+
// Measured 2026-09-27 over 831,328 unique lines (this repo's tracked text plus local
|
|
47
|
+
// transcripts: prompts, replies, tool output), old and new run back to back: 28 lines
|
|
48
|
+
// differ. All the catches are fixtures quoted from the D#128 audit; the rest are
|
|
49
|
+
// `- **token**:assistant …` and `` `PGPASSWORD`=PG+password ``, which the same text with
|
|
50
|
+
// no markup scrubs too, plus a JSON-escaped `\n` taken as a value. Idempotence failures: 0
|
|
51
|
+
// and 0. A 13,770-case ground-truth fuzz (labels × 8 wrappings × separators × values ×
|
|
52
|
+
// positions): leaks 12,580 → 421, none newly opened. All 421 are `<word> token|secret|bearer:`
|
|
53
|
+
// in prose, left open by design (below).
|
|
54
|
+
//
|
|
55
|
+
// ACCEPTED GAPS (D#131). This scrubber stops ACCIDENTAL persistence; an author who wants a
|
|
56
|
+
// secret stored can always encode it. So shapes only an adversary writes stay open:
|
|
57
|
+
// zero-width characters inside a label, fullwidth letters. Shapes that occur naturally but
|
|
58
|
+
// cannot be told from prose stay open too: `password is <v>` and `<word> token: <v>` (the
|
|
59
|
+
// guard below). A pattern that caught them would also rewrite ordinary sentences, and
|
|
60
|
+
// v3.61.0 already had to undo exactly that. `| password | <v> |` table rows stay open for a
|
|
61
|
+
// different reason: those 831k lines held 1 such row and none with a credential-shaped
|
|
62
|
+
// value, so a table pattern's false-positive rate cannot be measured here.
|
|
31
63
|
// 1a. `=` assignment → ALWAYS scrub (config syntax, never prose):
|
|
32
64
|
[
|
|
33
|
-
/((?:\b|_)(?:password|passwd|passphrase|token|bearer|secret)
|
|
65
|
+
/((?:\b|_)(?:password|passwd|passphrase|token|bearer|secret)(?:[*_~`]{1,3})?\s*=(?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
34
66
|
'$1***',
|
|
35
67
|
],
|
|
36
68
|
// 1b. `:` separator, PASSWORD nouns. Position decides how permissive the value
|
|
@@ -71,16 +103,16 @@ export const SECRET_PATTERNS = [
|
|
|
71
103
|
// Both arms emit `***` (3 chars, under the {6,} floor), so they cannot
|
|
72
104
|
// double-apply.
|
|
73
105
|
[
|
|
74
|
-
/((?<![A-Za-z][ \t])(?:\b|_)(?:password|passwd|passphrase)\s
|
|
106
|
+
/((?:(?<![A-Za-z][ \t][*_~]{0,3})(?:\b|_)(?:password|passwd|passphrase)(?:[*_~]{1,3})?|(?=_?(?:password|passwd|passphrase)`)(?<![A-Za-z][ \t]`[\w-]{0,40})(?<=`[\w-]{0,40})(?:\b|_)(?:password|passwd|passphrase)`)\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
75
107
|
'$1***',
|
|
76
108
|
],
|
|
77
109
|
[
|
|
78
|
-
/((?:\b|_)(?:password|passwd|passphrase)
|
|
110
|
+
/((?:\b|_)(?:password|passwd|passphrase)(?:[*_~`]{1,3})?\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)(?![A-Za-z]{1,15}(?=[`*~]*(?:[\s,;'"}\]]|$)))[^\s,;'"}\]]{6,}/gi,
|
|
79
111
|
'$1***',
|
|
80
112
|
],
|
|
81
113
|
// 1c. `:` separator, prose-ambiguous nouns → keep the lookbehind ("the token: alicebob"):
|
|
82
114
|
[
|
|
83
|
-
/((?<![A-Za-z][ \t])(?:\b|_)(?:token|bearer|secret)\s
|
|
115
|
+
/((?:(?<![A-Za-z][ \t][*_~]{0,3})(?:\b|_)(?:token|bearer|secret)(?:[*_~]{1,3})?|(?=_?(?:token|bearer|secret)`)(?<![A-Za-z][ \t]`[\w-]{0,40})(?<=`[\w-]{0,40})(?:\b|_)(?:token|bearer|secret)`)\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
84
116
|
'$1***',
|
|
85
117
|
],
|
|
86
118
|
// access_token / refresh_token are the canonical OAuth2 field names — they were
|
|
@@ -95,7 +127,7 @@ export const SECRET_PATTERNS = [
|
|
|
95
127
|
// low-FP decision that `topsecret=` / `access_token_count:` are non-credentials
|
|
96
128
|
// (#8283 + utils.test.mjs:1089-1100); bare `pwd` is omitted so `PWD=` (a path) survives.
|
|
97
129
|
[
|
|
98
|
-
/((?:\b|_)(?:api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token|pgpassword|pgpass|mysql_pwd)
|
|
130
|
+
/((?:\b|_)(?:api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token|pgpassword|pgpass|mysql_pwd)(?:[*_~`]{1,3})?\s*[=::](?:[*_~]{1,2}(?=\s))?\s*)(?!process\.env\.)(?!new\s)(?!\w+\()(?!(?:null|undefined|true|false|None|nil|empty|""|''|0)\b)[^\s,;'"}\]]{6,}/gi,
|
|
99
131
|
'$1***',
|
|
100
132
|
],
|
|
101
133
|
// Space-separated credential CLI flag: `--password <value>` (long-form). The KV
|
|
@@ -118,19 +150,28 @@ export const SECRET_PATTERNS = [
|
|
|
118
150
|
// (a) bare credential nouns: `=` always scrubs; `:` keeps the prose lookbehind
|
|
119
151
|
// (mirrors the unquoted 1a/1b split — a quoted value doesn't turn `:` prose
|
|
120
152
|
// into config, but `<word> password="x"` is still a leak):
|
|
121
|
-
[
|
|
122
|
-
|
|
123
|
-
|
|
153
|
+
[
|
|
154
|
+
/((?:\b|_)(?:password|passwd|passphrase|token|bearer|secret)(?:[*_~`]{1,3})?\s*=(?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
155
|
+
'$1$2***$2',
|
|
156
|
+
],
|
|
157
|
+
[
|
|
158
|
+
/((?:\b|_)(?:password|passwd|passphrase)(?:[*_~`]{1,3})?\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
159
|
+
'$1$2***$2',
|
|
160
|
+
],
|
|
161
|
+
[
|
|
162
|
+
/((?:(?<![A-Za-z][ \t][*_~]{0,3})(?:\b|_)(?:token|bearer|secret)(?:[*_~]{1,3})?|(?=_?(?:token|bearer|secret)`)(?<![A-Za-z][ \t]`[\w-]{0,40})(?<=`[\w-]{0,40})(?:\b|_)(?:token|bearer|secret)`)\s*[::](?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
163
|
+
'$1$2***$2',
|
|
164
|
+
],
|
|
124
165
|
// (b) structured keys + named env vars are unambiguous config even after a word
|
|
125
166
|
// (`see api_key: "x"` DOES scrub, mirroring the unquoted structured-key path):
|
|
126
167
|
[
|
|
127
|
-
/((?:\b|_)(?:pgpassword|pgpass|mysql_pwd|api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token)
|
|
168
|
+
/((?:\b|_)(?:pgpassword|pgpass|mysql_pwd|api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?key|private[_-]?key|client[_-]?secret|auth[_-]?token|access[_-]?token|refresh[_-]?token)(?:[*_~`]{1,3})?\s*[=::](?:[*_~]{1,2}(?=\s))?\s*)(['"])[^'"]{6,}\2/gi,
|
|
128
169
|
'$1$2***$2',
|
|
129
170
|
],
|
|
130
171
|
// AWS access keys: AKIA (long-term) + ASIA (STS temp) + AROA (role) + AIDA
|
|
131
172
|
// (user) + ANPA/ANVA/AGPA (other principal types). All share the 4-letter
|
|
132
173
|
// prefix + exactly 16 base32 chars shape — specific enough for near-zero FP.
|
|
133
|
-
[/\b(?:AKIA|ASIA|AROA|AIDA|ANPA|ANVA|AGPA)[A-Z0-9]{16}
|
|
174
|
+
[/\b(?:AKIA|ASIA|AROA|AIDA|ANPA|ANVA|AGPA)[A-Z0-9]{16}(?![A-Za-z0-9])/g, '***'],
|
|
134
175
|
// OpenAI / Anthropic keys (sk-...) — specific prefixes have lower length threshold
|
|
135
176
|
[/\bsk-(?:proj|ant|ant-api\d{2})-[a-zA-Z0-9_-]{8,}\b/g, '***'],
|
|
136
177
|
[/\bsk-[a-zA-Z0-9_-]{20,}\b/g, '***'],
|
|
@@ -143,13 +184,30 @@ export const SECRET_PATTERNS = [
|
|
|
143
184
|
[/\b(?:xox[bpasr]|xapp|xoxe)-[a-zA-Z0-9-]{10,}\b/g, '***'],
|
|
144
185
|
// Slack incoming-webhook URL — the path after /services/ is the shared secret.
|
|
145
186
|
[/(https:\/\/hooks\.slack\.com\/services\/)[A-Za-z0-9/]+/g, '$1***'],
|
|
146
|
-
// JWT tokens (eyJ...eyJ...)
|
|
147
|
-
|
|
187
|
+
// JWT tokens (eyJ...eyJ...). A JWT begins its own token, so the start is `(?<![\w-])`, not
|
|
188
|
+
// `\b`: `-` is a base64url character, and `\b` let every `-eyJ` inside one dotless run be a
|
|
189
|
+
// fresh start that rescans the run to its end — quadratic, 9.6 s on 200k chars of `eyJ-`
|
|
190
|
+
// (D#130). A `-eyJ` start is the middle of a run, never a JWT's first character.
|
|
191
|
+
// A JWT glued by a hyphen to up to 40 characters of hyphenated words (`my-sess-eyJ…`,
|
|
192
|
+
// `X-Auth-Token-eyJ…`) is a start too. `\b` allowed any length; past 40 characters it is missed. The lookbehind needs a token
|
|
193
|
+
// boundary within those 41 characters, so a hyphen run has at most ~10 starts, not one per
|
|
194
|
+
// `-eyJ` (v6.19.0 pre-tag reviews P3-2, delta P3-1).
|
|
195
|
+
[
|
|
196
|
+
/(?:(?<![\w-])|(?<=(?:^|[^\w-])[\w-]{1,40}-))eyJ[a-zA-Z0-9_-]{10,}\.eyJ[a-zA-Z0-9_-]{10,}\.[a-zA-Z0-9_-]+\b/g,
|
|
197
|
+
'***',
|
|
198
|
+
],
|
|
148
199
|
// PEM private key blocks. `[A-Z0-9 ]*` covers every armor label — RSA/EC/DSA/
|
|
149
200
|
// OPENSSH plus ENCRYPTED and PGP (… PRIVATE KEY BLOCK) — that the fixed
|
|
150
201
|
// alternation missed; the block delimiters make FP impossible.
|
|
202
|
+
// The body stops at the next `-----BEGIN ` (D#130): with `[\s\S]*?` every header with no END
|
|
203
|
+
// scanned to the end of the text, on each of scrubSecrets' passes — quadratic, 8.3 s on 500k
|
|
204
|
+
// chars. A block whose END is missing ends where the next private-key BEGIN starts, so a cut-off
|
|
205
|
+
// key's body is scrubbed too (v6.19.0 pre-tag review P2-1: requiring the END there stored that
|
|
206
|
+
// body); any other BEGIN (a certificate) does not end it, so a bare header in prose does not
|
|
207
|
+
// erase the text up to one (delta review P3-2). A header with no END and no later key header
|
|
208
|
+
// is left as it was in v6.18.0.
|
|
151
209
|
[
|
|
152
|
-
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----[\s\S]
|
|
210
|
+
/-----BEGIN [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----(?:(?!-----BEGIN [A-Z0-9 ]*PRIVATE KEY)[\s\S])*?(?:-----END [A-Z0-9 ]*PRIVATE KEY(?: BLOCK)?-----|(?=-----BEGIN [A-Z0-9 ]*PRIVATE KEY))/g,
|
|
153
211
|
'***PEM_KEY***',
|
|
154
212
|
],
|
|
155
213
|
// Long hex strings in credential assignments (e.g. SECRET_KEY=abc123def456...).
|
|
@@ -160,7 +218,10 @@ export const SECRET_PATTERNS = [
|
|
|
160
218
|
[/\bAIza[A-Za-z0-9_-]{35}\b/g, '***'],
|
|
161
219
|
// Authorization header credentials — Bearer (opaque), Basic (base64 user:pass),
|
|
162
220
|
// and GitHub's `token` scheme all carry secrets after the scheme word.
|
|
163
|
-
[
|
|
221
|
+
[
|
|
222
|
+
/(Authorization(?:[*_~`]{1,3})?[::](?:[*_~]{1,2}(?=\s))?\s*(?:Bearer|Basic|token)\s+)[^\s,;'"}\]]+/gi,
|
|
223
|
+
'$1***',
|
|
224
|
+
],
|
|
164
225
|
// R10 P1-6: the same header as a QUOTED KEY — `{"Authorization":"Bearer …"}`. The
|
|
165
226
|
// pattern above needs `Authorization:` literally, and in JSON a quote sits between the
|
|
166
227
|
// name and the colon, so a `curl -v` / fetch header dump walked straight through. The
|
|
@@ -194,10 +255,10 @@ export const SECRET_PATTERNS = [
|
|
|
194
255
|
'$1$2://***',
|
|
195
256
|
],
|
|
196
257
|
// npm tokens (npm_...)
|
|
197
|
-
[/\bnpm_[a-zA-Z0-9]{36,}
|
|
258
|
+
[/\bnpm_[a-zA-Z0-9]{36,}(?![A-Za-z0-9])/g, '***'],
|
|
198
259
|
// Stripe keys (sk_live_, rk_live_, pk_live_, sk_test_, pk_test_) + webhook signing secret (whsec_)
|
|
199
|
-
[/\b[srp]k_(?:live|test)_[a-zA-Z0-9]{20,}
|
|
200
|
-
[/\bwhsec_[a-zA-Z0-9]{20,}
|
|
260
|
+
[/\b[srp]k_(?:live|test)_[a-zA-Z0-9]{20,}(?![A-Za-z0-9])/g, '***'],
|
|
261
|
+
[/\bwhsec_[a-zA-Z0-9]{20,}(?![A-Za-z0-9])/g, '***'],
|
|
201
262
|
// SendGrid API keys: SG.<22>.<43> — two dots at fixed offsets make this
|
|
202
263
|
// structurally unmistakable; near-zero false-positive risk.
|
|
203
264
|
[/\bSG\.[A-Za-z0-9_-]{22}\.[A-Za-z0-9_-]{43}\b/g, '***'],
|
|
@@ -224,8 +285,11 @@ export const SECRET_PATTERNS = [
|
|
|
224
285
|
// (password|secret|api_key|auth_token|access_token|private_key) so a benign
|
|
225
286
|
// `"token_count"` value (numeric, <6 non-quote chars after scrub) and prose
|
|
226
287
|
// keys stay low-FP; over-scrub is the safe direction for at-rest memory.
|
|
288
|
+
// `\w{0,64}`, not `\w*`, on both sides of the noun here and in the next pattern (D#130): from
|
|
289
|
+
// one quote, `\w*` ran to the end of a word run and backtracked through every keyword in it,
|
|
290
|
+
// each rescanning the run — `"` + `secret` x 33k took 4.6 s. 64 is far past any key name.
|
|
227
291
|
[
|
|
228
|
-
/("\w
|
|
292
|
+
/("\w{0,64}(?:password|passwd|secret|api[_-]?key|auth[_-]?token|access[_-]?token|private[_-]?key)\w{0,64}"\s*:\s*")[^"]{6,}(")/gi,
|
|
229
293
|
'$1***$2',
|
|
230
294
|
],
|
|
231
295
|
// Quoted-KEY credential values — Python dict reprs `{'api_key': '...'}`, single-quoted
|
|
@@ -241,7 +305,7 @@ export const SECRET_PATTERNS = [
|
|
|
241
305
|
// `'token_count': 123456`); `passphrase` added here too (double-quoted JSON passphrase is
|
|
242
306
|
// subsumed by this pattern since `['"]` matches `"`). Over-scrub is the safe direction.
|
|
243
307
|
[
|
|
244
|
-
/(['"]
|
|
308
|
+
/(['"](?:\w{0,64}(?:password|passwd|passphrase|secret|api[_-]?key|auth[_-]?token|access[_-]?token|private[_-]?key)\w{0,64}|\w{1,64}_token)['"]\s*:\s*)(['"])[^'"]{6,}\2/gi,
|
|
245
309
|
'$1$2***$2',
|
|
246
310
|
],
|
|
247
311
|
// Session cookies in headers / urlencoded bodies (sessionid=, session_id=, JSESSIONID=, PHPSESSID=).
|
package/server.mjs
CHANGED
|
@@ -17,7 +17,8 @@ import {
|
|
|
17
17
|
formatSchemaSkewNotice,
|
|
18
18
|
} from './lib/schema-skew.mjs';
|
|
19
19
|
import { reRankWithContext, runIdleCleanup, buildServerInstructions } from './search-scoring.mjs';
|
|
20
|
-
import { searchObservationsHybrid } from './search-engine.mjs';
|
|
20
|
+
import { searchObservationsHybrid, snippetAddsInfo } from './search-engine.mjs';
|
|
21
|
+
import { autoHeaderNote, autoLegend, autoTag } from './lib/provenance.mjs';
|
|
21
22
|
import {
|
|
22
23
|
deepSearch,
|
|
23
24
|
resolveDeepMode,
|
|
@@ -461,7 +462,7 @@ function formatSearchOutput(
|
|
|
461
462
|
// explicitly requested OR semantics — there's no "fallback" in that path.
|
|
462
463
|
const fallbackHint = orFallbackFired && !args.or ? ' (relaxed AND→OR)' : '';
|
|
463
464
|
lines.push(
|
|
464
|
-
`Found ${countLabel} result(s)${qLabel}${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}\n`,
|
|
465
|
+
`Found ${countLabel} result(s)${qLabel}${fallbackHint}:${hasMixed ? ' (# observation, S# session, P# prompt, E# event)' : ''}${autoLegend(paginatedResults)}\n`,
|
|
465
466
|
);
|
|
466
467
|
|
|
467
468
|
// `~Nt` = estimated tokens to fetch this row's full body via mem_get (attachBodyTokens).
|
|
@@ -470,9 +471,9 @@ function formatSearchOutput(
|
|
|
470
471
|
for (const r of paginatedResults) {
|
|
471
472
|
if (r.source === 'obs') {
|
|
472
473
|
lines.push(
|
|
473
|
-
`#${r.id} ${typeIcon(r.type)} [${r.type}] ${truncate(r.title || r.subtitle || '(untitled)')} | ${r.project} | ${fmtDate(r.date)}${tok(r)}`,
|
|
474
|
+
`#${r.id} ${typeIcon(r.type)} [${r.type}]${autoTag(r)} ${truncate(r.title || r.subtitle || '(untitled)')} | ${r.project} | ${fmtDate(r.date)}${tok(r)}`,
|
|
474
475
|
);
|
|
475
|
-
if (r.snippet
|
|
476
|
+
if (snippetAddsInfo(r.snippet, r.title)) {
|
|
476
477
|
lines.push(` ${truncate(r.snippet, 100)}`);
|
|
477
478
|
}
|
|
478
479
|
} else if (r.source === 'session') {
|
|
@@ -900,7 +901,7 @@ server.registerTool(
|
|
|
900
901
|
const renderFields = obsFieldFilter || OBS_FIELDS;
|
|
901
902
|
for (const row of rows) {
|
|
902
903
|
foundBySource.obs.add(row.id);
|
|
903
|
-
const lines = [`── #${row.id} ──`];
|
|
904
|
+
const lines = [`── #${row.id}${autoHeaderNote(row)} ──`];
|
|
904
905
|
// Retraction first (shared with the CLI `get` via get-core) — see supersededNotice.
|
|
905
906
|
const retracted = supersededNotice(row);
|
|
906
907
|
if (retracted) lines.push(retracted);
|
package/source-files.mjs
CHANGED
|
@@ -81,6 +81,8 @@ export const SOURCE_FILES = [
|
|
|
81
81
|
'lib/stats-quality.mjs',
|
|
82
82
|
'lib/low-signal-patterns.mjs',
|
|
83
83
|
'lib/private-strip.mjs',
|
|
84
|
+
// Which writer produced an observation (explicit save vs machine-written); search + get marks.
|
|
85
|
+
'lib/provenance.mjs',
|
|
84
86
|
'lib/citation-tracker.mjs',
|
|
85
87
|
// v3.47 (D#78 P1): per-(obs,file) edge attribution. Imported by hook.mjs
|
|
86
88
|
// (handleStop edge resolution). Missing from manifest → tarball hook.mjs
|
package/utils.mjs
CHANGED
|
@@ -56,7 +56,8 @@ export {
|
|
|
56
56
|
} from './bash-utils.mjs';
|
|
57
57
|
|
|
58
58
|
// Internal imports for functions that remain in this module
|
|
59
|
-
import { truncate } from './format-utils.mjs';
|
|
59
|
+
import { normalizeInline, truncate } from './format-utils.mjs';
|
|
60
|
+
import { stripPrivate } from './lib/private-strip.mjs';
|
|
60
61
|
import { stripTestSuffix } from './bash-utils.mjs';
|
|
61
62
|
// Static, and deliberately the dependency-free resolver (node:os + node:path only) —
|
|
62
63
|
// debugCatch's sampler must not pull in the DB layer. See its comment below.
|
|
@@ -282,9 +283,56 @@ export function isRelatedToEpisode(episode, newFiles) {
|
|
|
282
283
|
// magnitude above the longest cut here, so a secret that begins before the cut is still
|
|
283
284
|
// seen whole by the patterns, at bounded cost.
|
|
284
285
|
const DESC_SCRUB_WINDOW = 4096;
|
|
286
|
+
|
|
287
|
+
// Every field is cut strictly around private spans: closed `<private>` spans are redacted across
|
|
288
|
+
// the WHOLE input first (one linear pass; hook input is capped at 256 KiB), then the window stops
|
|
289
|
+
// before the first remaining `<private>` or private-key BEGIN, so no field shows a span whose end
|
|
290
|
+
// is out of view. Before v6.19.0 an unclosed opener in the window (a Grep line
|
|
291
|
+
// `notes.md:3:<private>bank pin 4412`) was shown as written (v6.19.0 pre-tag reviews P3-1, r3 P2-1).
|
|
292
|
+
// The PEM half is case-sensitive, like the scrubber's PEM pattern.
|
|
293
|
+
const PRIVATE_OPENER_RE = /<[Pp][Rr][Ii][Vv][Aa][Tt][Ee]>|-----BEGIN [A-Z0-9 ]*PRIVATE KEY/;
|
|
285
294
|
function scrubTruncate(str, max) {
|
|
286
295
|
if (typeof str !== 'string' || str === '') return truncate(str, max);
|
|
287
|
-
|
|
296
|
+
let win = stripPrivate(str).slice(0, DESC_SCRUB_WINDOW);
|
|
297
|
+
const at = win.search(PRIVATE_OPENER_RE);
|
|
298
|
+
if (at >= 0) win = win.slice(0, at);
|
|
299
|
+
// The window edge can split a surrogate pair; truncate only guards a cut it makes itself.
|
|
300
|
+
if (/[\uD800-\uDBFF]$/.test(win)) win = win.slice(0, -1);
|
|
301
|
+
return truncate(_scrubSecrets(win), max);
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// Head+tail cut for command output. A check usually prints its verdict last ("...parsed: 0",
|
|
305
|
+
// "3 passed"); a head-only cut hands the episode summarizer an unresolved-looking fragment,
|
|
306
|
+
// and it has written a bugfix narrative for a check that passed. The tail comes from its own
|
|
307
|
+
// scrub window at the END of the original string: taking it from the head window would drop
|
|
308
|
+
// the verdict of any output longer than DESC_SCRUB_WINDOW.
|
|
309
|
+
//
|
|
310
|
+
// Output with a private span anywhere in it (a `<private>` tag or a PEM private-key marker) shows
|
|
311
|
+
// no tail, only v6.18.0's 60-character head (cut as above). The tail is scrubbed in its own
|
|
312
|
+
// window, which cannot see a span that crosses its edge, and pairing the markers per window
|
|
313
|
+
// stored span text two ways in the v6.19.0 pre-tag review (a cut inside a closed `<private>`, a
|
|
314
|
+
// key with no END more than 4096 characters back).
|
|
315
|
+
const PRIVATE_TAG_HINT_RE = /<\/?private>/i;
|
|
316
|
+
const HEAD_ONLY_MAX = 60;
|
|
317
|
+
// The early return needs the WHOLE output inside the head window: a long output whose first
|
|
318
|
+
// 4096 characters collapse to a few (whitespace) still has a tail to show (v6.19.0 pre-tag
|
|
319
|
+
// claims review F2).
|
|
320
|
+
function scrubTruncateEnds(str, max) {
|
|
321
|
+
if (typeof str === 'string' && (str.includes('PRIVATE KEY') || PRIVATE_TAG_HINT_RE.test(str))) {
|
|
322
|
+
return scrubTruncate(str, HEAD_ONLY_MAX);
|
|
323
|
+
}
|
|
324
|
+
const flat = scrubTruncate(str, DESC_SCRUB_WINDOW);
|
|
325
|
+
const whole = typeof str !== 'string' || str.length <= DESC_SCRUB_WINDOW;
|
|
326
|
+
if (whole && flat.length <= max) return flat;
|
|
327
|
+
const tailLen = Math.floor(max / 2) - 1;
|
|
328
|
+
const tailSrc = whole ? flat : normalizeInline(_scrubSecrets(str.slice(-DESC_SCRUB_WINDOW)));
|
|
329
|
+
// Drop a lone low surrogate the tail's cut may start on.
|
|
330
|
+
const tail = tailSrc.slice(-tailLen).replace(/^[\uDC00-\uDFFF]/, '');
|
|
331
|
+
// A head short enough to escape truncate's own "…" still gets one before the tail.
|
|
332
|
+
const budget = max - tailLen;
|
|
333
|
+
let head = truncate(flat, budget);
|
|
334
|
+
if (!head.endsWith('…')) head = head.length < budget ? `${head}…` : truncate(flat, budget - 1);
|
|
335
|
+
return head + tail;
|
|
288
336
|
}
|
|
289
337
|
|
|
290
338
|
export function makeEntryDesc(toolName, input, resp, opts) {
|
|
@@ -302,7 +350,7 @@ export function makeEntryDesc(toolName, input, resp, opts) {
|
|
|
302
350
|
const isErr =
|
|
303
351
|
opts?.isError ??
|
|
304
352
|
(/\berror\b|\bfail(ed|ure)?\b|\bexception\b|\bpanic\b/i.test(resp) && resp.length > 30);
|
|
305
|
-
const snippet =
|
|
353
|
+
const snippet = scrubTruncateEnds(resp, 100);
|
|
306
354
|
return isErr ? `${cmd} → ERROR: ${snippet}` : `${cmd} → ${snippet}`;
|
|
307
355
|
}
|
|
308
356
|
case 'Grep':
|