open-memex 0.1.0 → 0.2.0-alpha
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +31 -5
- package/CONTRIBUTING.md +31 -0
- package/README.md +17 -7
- package/docs/SCOPES.md +81 -0
- package/docs/V2-DESIGN.md +484 -0
- package/package.json +1 -1
- package/scripts/smoke-pure.ts +209 -9
- package/src/cli.ts +138 -22
- package/src/config.ts +3 -10
- package/src/index.ts +18 -6
- package/src/redact.ts +151 -4
- package/src/retrieve/cjk.ts +63 -0
- package/src/retrieve/inject.ts +2 -2
- package/src/retrieve/search.ts +115 -28
- package/src/scope.ts +7 -2
- package/src/store/db.ts +62 -14
- package/src/store/lifecycle.ts +280 -0
- package/src/store/markdown.ts +163 -11
- package/src/store/sync.ts +53 -9
- package/src/store/v2migrate.ts +190 -0
- package/src/tools/memory.ts +88 -27
package/src/redact.ts
CHANGED
|
@@ -1,5 +1,104 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
1
|
+
/**
|
|
2
|
+
* Redaction: secrets must never land in memory files or the index.
|
|
3
|
+
*
|
|
4
|
+
* Three layers, checked in order by redact():
|
|
5
|
+
* 1. <private>...</private> regions are stripped first (explicit opt-out —
|
|
6
|
+
* the author marked this span as sensitive, so it is replaced with
|
|
7
|
+
* [REDACTED] before any detection runs).
|
|
8
|
+
* 2. Built-in provider patterns (always on, reported by id).
|
|
9
|
+
* 3. User patterns from config redactPatterns (reported verbatim).
|
|
10
|
+
* 4. High-entropy assignment heuristic (catches secrets whose provider we
|
|
11
|
+
* don't have a pattern for, e.g. `deploy_key = "aB3d..."`).
|
|
12
|
+
*
|
|
13
|
+
* Any hit refuses the whole write — a memory with a hole in it is worse
|
|
14
|
+
* than no memory, because the hole invites reconstruction.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
export interface SecretPattern {
|
|
18
|
+
/** Stable id reported in matchedPattern. */
|
|
19
|
+
id: string;
|
|
20
|
+
/** Regex source (no slashes). */
|
|
21
|
+
source: string;
|
|
22
|
+
/** Optional RegExp flags, e.g. "i". */
|
|
23
|
+
flags?: string;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Prefixes like `sk-` also occur inside ordinary English words ("task-…",
|
|
28
|
+
* "risk-…", "disk-…"), which caused confirmed false positives. Require the
|
|
29
|
+
* token prefix NOT to be preceded by a word/hyphen char, so a real key after
|
|
30
|
+
* "=", ":", space, or a quote still matches.
|
|
31
|
+
*/
|
|
32
|
+
const TOKEN_BOUNDARY = "(?<![A-Za-z0-9_-])";
|
|
33
|
+
|
|
34
|
+
export const BUILTIN_SECRET_PATTERNS: SecretPattern[] = [
|
|
35
|
+
{ id: "openai-key", source: `${TOKEN_BOUNDARY}sk-[A-Za-z0-9_-]{20,}` },
|
|
36
|
+
{
|
|
37
|
+
id: "openai-admin-key",
|
|
38
|
+
source: `${TOKEN_BOUNDARY}sk-admin-[A-Za-z0-9_-]{20,}`,
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
id: "openai-session-key",
|
|
42
|
+
source: `${TOKEN_BOUNDARY}sm_[A-Za-z0-9_-]{20,}`,
|
|
43
|
+
},
|
|
44
|
+
{ id: "github-pat", source: `${TOKEN_BOUNDARY}ghp_[A-Za-z0-9]{30,}` },
|
|
45
|
+
{ id: "github-oauth-token", source: `${TOKEN_BOUNDARY}gho_[A-Za-z0-9]{30,}` },
|
|
46
|
+
{ id: "github-user-token", source: `${TOKEN_BOUNDARY}ghu_[A-Za-z0-9]{30,}` },
|
|
47
|
+
{
|
|
48
|
+
id: "github-refresh-token",
|
|
49
|
+
source: `${TOKEN_BOUNDARY}ghr_[A-Za-z0-9]{30,}`,
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
id: "github-fine-grained-pat",
|
|
53
|
+
source: `${TOKEN_BOUNDARY}github_pat_[A-Za-z0-9_]{40,}`,
|
|
54
|
+
},
|
|
55
|
+
{ id: "aws-access-key-id", source: `${TOKEN_BOUNDARY}AKIA[0-9A-Z]{16}` },
|
|
56
|
+
{
|
|
57
|
+
id: "aws-secret-access-key",
|
|
58
|
+
source:
|
|
59
|
+
"aws[_-]?secret[_-]?access[_-]?key[\"']?\\s*[:=]\\s*[\"']?[A-Za-z0-9/+=]{40}",
|
|
60
|
+
flags: "i",
|
|
61
|
+
},
|
|
62
|
+
{ id: "slack-token", source: `${TOKEN_BOUNDARY}xox[baprs]-[A-Za-z0-9-]{10,}` },
|
|
63
|
+
{ id: "google-api-key", source: `${TOKEN_BOUNDARY}AIza[0-9A-Za-z_-]{30,}` },
|
|
64
|
+
{ id: "npm-token", source: `${TOKEN_BOUNDARY}npm_[A-Za-z0-9]{30,}` },
|
|
65
|
+
{ id: "gitlab-pat", source: `${TOKEN_BOUNDARY}glpat-[A-Za-z0-9_-]{20,}` },
|
|
66
|
+
{
|
|
67
|
+
id: "stripe-restricted-key",
|
|
68
|
+
source: `${TOKEN_BOUNDARY}rk_(live|test)_[A-Za-z0-9]{20,}`,
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
id: "stripe-webhook-secret",
|
|
72
|
+
source: `${TOKEN_BOUNDARY}whsec_[A-Za-z0-9]{20,}`,
|
|
73
|
+
},
|
|
74
|
+
{ id: "private-key-block", source: "-----BEGIN [A-Z ]*PRIVATE KEY-----" },
|
|
75
|
+
{
|
|
76
|
+
id: "jwt",
|
|
77
|
+
source: `${TOKEN_BOUNDARY}eyJ[A-Za-z0-9_-]{10,}\\.[A-Za-z0-9_-]{10,}\\.[A-Za-z0-9_-]{10,}`,
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
id: "generic-secret-assignment",
|
|
81
|
+
source:
|
|
82
|
+
"(api[_-]?key|secret|passwd|password|auth[_-]?token|access[_-]?token)[\"']?\\s*[:=]\\s*[\"']?[A-Za-z0-9_\\-./+=]{16,}[\"']?",
|
|
83
|
+
flags: "i",
|
|
84
|
+
},
|
|
85
|
+
];
|
|
86
|
+
|
|
87
|
+
function compile(p: SecretPattern): RegExp | null {
|
|
88
|
+
try {
|
|
89
|
+
return new RegExp(p.source, p.flags ?? "");
|
|
90
|
+
} catch {
|
|
91
|
+
return null; // ignore malformed builtin (should never happen)
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Returns the matched builtin pattern id, or null. */
|
|
96
|
+
export function findBuiltinSecret(text: string): string | null {
|
|
97
|
+
for (const p of BUILTIN_SECRET_PATTERNS) {
|
|
98
|
+
const re = compile(p);
|
|
99
|
+
if (re && re.test(text)) return p.id;
|
|
100
|
+
}
|
|
101
|
+
return null;
|
|
3
102
|
}
|
|
4
103
|
|
|
5
104
|
export function findSecret(text: string, patterns: string[]): string | null {
|
|
@@ -14,11 +113,59 @@ export function findSecret(text: string, patterns: string[]): string | null {
|
|
|
14
113
|
return null;
|
|
15
114
|
}
|
|
16
115
|
|
|
116
|
+
function shannonEntropy(s: string): number {
|
|
117
|
+
const freq = new Map<string, number>();
|
|
118
|
+
for (const c of s) freq.set(c, (freq.get(c) ?? 0) + 1);
|
|
119
|
+
let h = 0;
|
|
120
|
+
for (const n of freq.values()) {
|
|
121
|
+
const p = n / s.length;
|
|
122
|
+
h -= p * Math.log2(p);
|
|
123
|
+
}
|
|
124
|
+
return h;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
// name = value assignments with a long token-ish value.
|
|
128
|
+
const ASSIGNMENT_RE =
|
|
129
|
+
/([A-Za-z_][A-Za-z0-9_]{1,63})\s*[:=]\s*["']?([A-Za-z0-9_\-./+=]{24,})["']?/g;
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Heuristic last line of defense: a long, high-entropy value assigned to a
|
|
133
|
+
* name is almost certainly a credential, even when no provider pattern
|
|
134
|
+
* matches. Tuned conservatively (length >= 24, entropy >= 4.5 bits/char):
|
|
135
|
+
* hex digests (<= 4.0) and prose (~4.0) pass through; base64-ish randomness
|
|
136
|
+
* does not. URLs are skipped.
|
|
137
|
+
*/
|
|
138
|
+
export function findHighEntropySecret(text: string): string | null {
|
|
139
|
+
for (const m of text.matchAll(ASSIGNMENT_RE)) {
|
|
140
|
+
const name = m[1];
|
|
141
|
+
const value = m[2];
|
|
142
|
+
if (value.includes("://")) continue; // URL, not a secret
|
|
143
|
+
if (shannonEntropy(value) >= 4.5) return `high-entropy-secret:${name}`;
|
|
144
|
+
}
|
|
145
|
+
return null;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export function stripPrivate(text: string): string {
|
|
149
|
+
// Closed pairs first...
|
|
150
|
+
let out = text.replace(/<private>[\s\S]*?<\/private>/gi, "[REDACTED]");
|
|
151
|
+
// ...then an unclosed <private> redacts everything after it. A dangling
|
|
152
|
+
// tag almost always means the author intended the rest to be private.
|
|
153
|
+
out = out.replace(/<private>[\s\S]*$/gi, "[REDACTED]");
|
|
154
|
+
return out;
|
|
155
|
+
}
|
|
156
|
+
|
|
17
157
|
export function redact(
|
|
18
158
|
text: string,
|
|
19
159
|
patterns: string[],
|
|
20
160
|
): { content: string; hadSecret: boolean; matchedPattern: string | null } {
|
|
21
161
|
const stripped = stripPrivate(text);
|
|
22
|
-
const
|
|
23
|
-
|
|
162
|
+
const builtin = findBuiltinSecret(stripped);
|
|
163
|
+
if (builtin)
|
|
164
|
+
return { content: stripped, hadSecret: true, matchedPattern: builtin };
|
|
165
|
+
const user = findSecret(stripped, patterns);
|
|
166
|
+
if (user) return { content: stripped, hadSecret: true, matchedPattern: user };
|
|
167
|
+
const entropic = findHighEntropySecret(stripped);
|
|
168
|
+
if (entropic)
|
|
169
|
+
return { content: stripped, hadSecret: true, matchedPattern: entropic };
|
|
170
|
+
return { content: stripped, hadSecret: false, matchedPattern: null };
|
|
24
171
|
}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
// CJK retrieval helpers (pure logic, no sqlite).
|
|
2
|
+
//
|
|
3
|
+
// FTS5's `porter unicode61` tokenizer treats a run of CJK ideographs as one
|
|
4
|
+
// token ("中文记忆" is a single token), so substring queries never match, and
|
|
5
|
+
// the old query builder dropped CJK characters entirely. Per D8 the default
|
|
6
|
+
// CJK strategy is bigram: at write time we pre-tokenize CJK runs into
|
|
7
|
+
// space-separated unigrams + overlapping bigrams stored in the `cjk` FTS
|
|
8
|
+
// column; at query time the CJK part of the query becomes an OR of bigrams
|
|
9
|
+
// against that column. Unigrams are included so single-character queries
|
|
10
|
+
// ("猫") still match. Works identically on bun:sqlite and better-sqlite3
|
|
11
|
+
// with no native tokenizer dependency.
|
|
12
|
+
|
|
13
|
+
const CJK_RE =
|
|
14
|
+
/[\u3400-\u4DBF\u4E00-\u9FFF\u3040-\u309F\u30A0-\u30FF\uAC00-\uD7AF\u1100-\u11FF\u{20000}-\u{2A6DF}]/u;
|
|
15
|
+
|
|
16
|
+
const CJK_RUN_RE =
|
|
17
|
+
/[\u3400-\u4DBF\u4E00-\u9FFF\u3040-\u309F\u30A0-\u30FF\uAC00-\uD7AF\u1100-\u11FF\u{20000}-\u{2A6DF}]+/gu;
|
|
18
|
+
|
|
19
|
+
/** True if the string contains any CJK (Han/Hiragana/Katakana/Hangul) character. */
|
|
20
|
+
export function hasCjk(s: string): boolean {
|
|
21
|
+
return CJK_RE.test(s);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Build the index text for the `cjk` FTS column: for every CJK run emit each
|
|
26
|
+
* character (unigram) then every overlapping bigram, space-separated.
|
|
27
|
+
* "中文记忆" -> "中 文 记 忆 中文 文记 记忆". Non-CJK text yields "".
|
|
28
|
+
*/
|
|
29
|
+
export function cjkIndexText(text: string): string {
|
|
30
|
+
const out: string[] = [];
|
|
31
|
+
for (const m of text.matchAll(CJK_RUN_RE)) {
|
|
32
|
+
const chars = [...m[0]];
|
|
33
|
+
for (const ch of chars) out.push(ch);
|
|
34
|
+
for (let i = 0; i + 1 < chars.length; i++) out.push(chars[i] + chars[i + 1]);
|
|
35
|
+
}
|
|
36
|
+
return out.join(" ");
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Escape a raw token for embedding in a double-quoted FTS5 phrase. */
|
|
40
|
+
function esc(t: string): string {
|
|
41
|
+
return t.replace(/"/g, '""');
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Convert the CJK runs of a query into an FTS5 expression for the `cjk`
|
|
46
|
+
* column. Multi-char runs become an OR of bigrams (recall-oriented; bm25
|
|
47
|
+
* ranks docs matching more bigrams higher). Single chars stay unigrams.
|
|
48
|
+
* Returns "" when the query has no CJK.
|
|
49
|
+
*/
|
|
50
|
+
export function cjkQueryExpr(query: string): string {
|
|
51
|
+
const parts: string[] = [];
|
|
52
|
+
for (const m of query.matchAll(CJK_RUN_RE)) {
|
|
53
|
+
const chars = [...m[0]];
|
|
54
|
+
if (chars.length === 1) {
|
|
55
|
+
parts.push(`"${esc(chars[0])}"`);
|
|
56
|
+
} else {
|
|
57
|
+
for (let i = 0; i + 1 < chars.length; i++) {
|
|
58
|
+
parts.push(`"${esc(chars[i] + chars[i + 1])}"`);
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
return parts.join(" OR ");
|
|
63
|
+
}
|
package/src/retrieve/inject.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { list } from "./search.ts";
|
|
2
2
|
import type { Scope } from "../scope.ts";
|
|
3
|
-
import {
|
|
3
|
+
import { PERSONAL_SCOPE } from "../scope.ts";
|
|
4
4
|
import type { MyOMemoryConfig } from "../config.ts";
|
|
5
5
|
|
|
6
6
|
function oneLine(s: string, max = 240): string {
|
|
@@ -10,7 +10,7 @@ function oneLine(s: string, max = 240): string {
|
|
|
10
10
|
|
|
11
11
|
export function buildContextBlock(scope: Scope, cfg: MyOMemoryConfig): string | null {
|
|
12
12
|
const project = list(scope.key, { limit: cfg.maxProjectMemories });
|
|
13
|
-
const user = list(
|
|
13
|
+
const user = list(PERSONAL_SCOPE.key, { limit: cfg.maxProfileItems });
|
|
14
14
|
|
|
15
15
|
if (project.length === 0 && user.length === 0) return null;
|
|
16
16
|
|
package/src/retrieve/search.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { db } from "../store/db.ts";
|
|
2
|
+
import { cjkQueryExpr, hasCjk } from "./cjk.ts";
|
|
2
3
|
|
|
3
4
|
export interface SearchHit {
|
|
4
5
|
id: string;
|
|
@@ -9,13 +10,105 @@ export interface SearchHit {
|
|
|
9
10
|
snippet: string;
|
|
10
11
|
score: number;
|
|
11
12
|
updated_at: number;
|
|
13
|
+
status: string;
|
|
12
14
|
}
|
|
13
15
|
|
|
14
|
-
|
|
16
|
+
interface RawRow {
|
|
17
|
+
id: string;
|
|
18
|
+
scope_key: string;
|
|
19
|
+
project_name: string;
|
|
20
|
+
type: string;
|
|
21
|
+
tags: string;
|
|
22
|
+
updated_at: number;
|
|
23
|
+
snippet: string;
|
|
24
|
+
score: number;
|
|
25
|
+
status: string;
|
|
26
|
+
superseded_by: string | null;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function toHit(r: RawRow): SearchHit {
|
|
30
|
+
return {
|
|
31
|
+
id: r.id,
|
|
32
|
+
scope_key: r.scope_key,
|
|
33
|
+
project_name: r.project_name,
|
|
34
|
+
type: r.type,
|
|
35
|
+
tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
|
|
36
|
+
snippet: r.snippet ?? "",
|
|
37
|
+
score: r.score,
|
|
38
|
+
updated_at: r.updated_at,
|
|
39
|
+
status: r.status,
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Lifecycle-aware post-processing (§3.3):
|
|
45
|
+
* - retracted / archived are excluded from retrieval (kept for audit);
|
|
46
|
+
* - a superseded memory resolves to the newest of its chain (cycle-safe);
|
|
47
|
+
* - deprecated stays visible as a warning but ranks after active.
|
|
48
|
+
*/
|
|
49
|
+
function resolveVisible(rows: RawRow[], limit: number): SearchHit[] {
|
|
50
|
+
const byId = new Map(rows.map((r) => [r.id, r]));
|
|
51
|
+
const fullRow = (id: string): RawRow | undefined => {
|
|
52
|
+
const cached = byId.get(id);
|
|
53
|
+
if (cached) return cached;
|
|
54
|
+
const r = db()
|
|
55
|
+
.prepare(
|
|
56
|
+
`SELECT id, scope_key, project_name, type, tags, updated_at, status,
|
|
57
|
+
superseded_by, substr(content, 1, 240) AS snippet
|
|
58
|
+
FROM memories WHERE id = ?`,
|
|
59
|
+
)
|
|
60
|
+
.get(id) as
|
|
61
|
+
| (Omit<RawRow, "score" | "snippet"> & { snippet: string })
|
|
62
|
+
| undefined;
|
|
63
|
+
if (!r) return undefined;
|
|
64
|
+
const full: RawRow = { ...r, score: 0 };
|
|
65
|
+
byId.set(id, full);
|
|
66
|
+
return full;
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
const seen = new Set<string>();
|
|
70
|
+
const active: SearchHit[] = [];
|
|
71
|
+
const deprecated: SearchHit[] = [];
|
|
72
|
+
|
|
73
|
+
for (const r of rows) {
|
|
74
|
+
if (r.status === "retracted" || r.status === "archived") continue;
|
|
75
|
+
let target = r;
|
|
76
|
+
if (r.status === "superseded") {
|
|
77
|
+
let cur = r;
|
|
78
|
+
const chain = new Set([r.id]);
|
|
79
|
+
while (cur.status === "superseded" && cur.superseded_by) {
|
|
80
|
+
if (chain.has(cur.superseded_by)) break; // cycle guard
|
|
81
|
+
chain.add(cur.superseded_by);
|
|
82
|
+
const nxt = fullRow(cur.superseded_by);
|
|
83
|
+
if (!nxt) break;
|
|
84
|
+
cur = nxt;
|
|
85
|
+
}
|
|
86
|
+
if (cur.status === "retracted" || cur.status === "archived") continue;
|
|
87
|
+
target = cur;
|
|
88
|
+
}
|
|
89
|
+
if (seen.has(target.id)) continue;
|
|
90
|
+
seen.add(target.id);
|
|
91
|
+
const hit = toHit(target);
|
|
92
|
+
if (target.status === "deprecated") deprecated.push(hit);
|
|
93
|
+
else active.push(hit);
|
|
94
|
+
}
|
|
95
|
+
return [...active, ...deprecated].slice(0, limit);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Convert free-text query into a safe FTS5 MATCH expression.
|
|
100
|
+
* Latin tokens keep the old behavior (prefix match on content/tags/type).
|
|
101
|
+
* CJK runs become an OR of bigrams against the `cjk` column (see cjk.ts).
|
|
102
|
+
* Mixed queries OR the two parts together.
|
|
103
|
+
*/
|
|
15
104
|
function toFtsQuery(q: string): string {
|
|
16
|
-
const
|
|
17
|
-
|
|
18
|
-
|
|
105
|
+
const latin = (q.toLowerCase().match(/[a-z0-9_.\-]+/g) ?? [])
|
|
106
|
+
.map((t) => `"${t.replace(/"/g, '""')}"*`)
|
|
107
|
+
.join(" OR ");
|
|
108
|
+
const cjk = hasCjk(q) ? cjkQueryExpr(q) : "";
|
|
109
|
+
if (latin && cjk) return `(${latin}) OR {cjk}:(${cjk})`;
|
|
110
|
+
if (cjk) return `{cjk}:(${cjk})`;
|
|
111
|
+
return latin;
|
|
19
112
|
}
|
|
20
113
|
|
|
21
114
|
export function search(
|
|
@@ -25,6 +118,9 @@ export function search(
|
|
|
25
118
|
const q = toFtsQuery(query);
|
|
26
119
|
if (!q) return [];
|
|
27
120
|
const limit = Math.max(1, Math.min(opts.limit ?? 8, 50));
|
|
121
|
+
// Over-fetch: lifecycle filtering (chain resolution, exclusions) happens
|
|
122
|
+
// after the FTS query, so candidates must survive it.
|
|
123
|
+
const fetchLimit = Math.min(limit * 3 + 10, 150);
|
|
28
124
|
|
|
29
125
|
const scopeFilter =
|
|
30
126
|
opts.scopeKeys && opts.scopeKeys.length > 0
|
|
@@ -34,6 +130,7 @@ export function search(
|
|
|
34
130
|
|
|
35
131
|
const sql = `
|
|
36
132
|
SELECT m.id, m.scope_key, m.project_name, m.type, m.tags, m.updated_at,
|
|
133
|
+
m.status, m.superseded_by,
|
|
37
134
|
snippet(memories_fts, 0, '[', ']', ' ... ', 12) AS snippet,
|
|
38
135
|
bm25(memories_fts) AS score
|
|
39
136
|
FROM memories_fts
|
|
@@ -45,7 +142,7 @@ export function search(
|
|
|
45
142
|
const params: unknown[] = [q];
|
|
46
143
|
if (opts.scopeKeys && opts.scopeKeys.length > 0) params.push(...opts.scopeKeys);
|
|
47
144
|
if (opts.type) params.push(opts.type);
|
|
48
|
-
params.push(
|
|
145
|
+
params.push(fetchLimit);
|
|
49
146
|
|
|
50
147
|
const rows = db()
|
|
51
148
|
.prepare(sql)
|
|
@@ -56,20 +153,15 @@ export function search(
|
|
|
56
153
|
type: string;
|
|
57
154
|
tags: string;
|
|
58
155
|
updated_at: number;
|
|
156
|
+
status: string;
|
|
157
|
+
superseded_by: string | null;
|
|
59
158
|
snippet: string;
|
|
60
159
|
score: number;
|
|
61
160
|
}>;
|
|
62
161
|
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
project_name: r.project_name,
|
|
67
|
-
type: r.type,
|
|
68
|
-
tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
|
|
69
|
-
snippet: r.snippet ?? "",
|
|
70
|
-
score: -r.score, // FTS5 bm25: lower = better; invert for intuition
|
|
71
|
-
updated_at: r.updated_at,
|
|
72
|
-
}));
|
|
162
|
+
// FTS5 bm25: lower = better; invert for intuition.
|
|
163
|
+
const raw: RawRow[] = rows.map((r) => ({ ...r, score: -r.score }));
|
|
164
|
+
return resolveVisible(raw, limit);
|
|
73
165
|
}
|
|
74
166
|
|
|
75
167
|
export function list(
|
|
@@ -77,10 +169,11 @@ export function list(
|
|
|
77
169
|
opts: { type?: string; limit?: number } = {},
|
|
78
170
|
): SearchHit[] {
|
|
79
171
|
const limit = Math.max(1, Math.min(opts.limit ?? 20, 100));
|
|
172
|
+
const fetchLimit = Math.min(limit * 3 + 10, 150);
|
|
80
173
|
const typeFilter = opts.type ? ` AND type = ?` : "";
|
|
81
174
|
const sql = `
|
|
82
|
-
SELECT id, scope_key, project_name, type, tags, updated_at,
|
|
83
|
-
substr(content, 1, 240) AS snippet
|
|
175
|
+
SELECT id, scope_key, project_name, type, tags, updated_at, status,
|
|
176
|
+
superseded_by, substr(content, 1, 240) AS snippet
|
|
84
177
|
FROM memories
|
|
85
178
|
WHERE scope_key = ?${typeFilter}
|
|
86
179
|
ORDER BY updated_at DESC
|
|
@@ -88,7 +181,7 @@ export function list(
|
|
|
88
181
|
`;
|
|
89
182
|
const params: unknown[] = [scopeKey];
|
|
90
183
|
if (opts.type) params.push(opts.type);
|
|
91
|
-
params.push(
|
|
184
|
+
params.push(fetchLimit);
|
|
92
185
|
|
|
93
186
|
const rows = db()
|
|
94
187
|
.prepare(sql)
|
|
@@ -99,17 +192,11 @@ export function list(
|
|
|
99
192
|
type: string;
|
|
100
193
|
tags: string;
|
|
101
194
|
updated_at: number;
|
|
195
|
+
status: string;
|
|
196
|
+
superseded_by: string | null;
|
|
102
197
|
snippet: string;
|
|
103
198
|
}>;
|
|
104
199
|
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
scope_key: r.scope_key,
|
|
108
|
-
project_name: r.project_name,
|
|
109
|
-
type: r.type,
|
|
110
|
-
tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
|
|
111
|
-
snippet: r.snippet ?? "",
|
|
112
|
-
score: 1,
|
|
113
|
-
updated_at: r.updated_at,
|
|
114
|
-
}));
|
|
200
|
+
const raw: RawRow[] = rows.map((r) => ({ ...r, score: 1 }));
|
|
201
|
+
return resolveVisible(raw, limit);
|
|
115
202
|
}
|
package/src/scope.ts
CHANGED
|
@@ -4,11 +4,16 @@ import path from "node:path";
|
|
|
4
4
|
|
|
5
5
|
export interface Scope {
|
|
6
6
|
key: string;
|
|
7
|
-
kind: "
|
|
7
|
+
kind: "personal" | "project" | "org";
|
|
8
8
|
projectName: string;
|
|
9
9
|
}
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
/** v2: v1 `user` scope is renamed to `personal` (§19). */
|
|
12
|
+
export const PERSONAL_SCOPE: Scope = {
|
|
13
|
+
key: "personal",
|
|
14
|
+
kind: "personal",
|
|
15
|
+
projectName: "personal",
|
|
16
|
+
};
|
|
12
17
|
|
|
13
18
|
function normalizeRemote(url: string): string {
|
|
14
19
|
return url
|
package/src/store/db.ts
CHANGED
|
@@ -23,18 +23,25 @@ function loadDatabase(): AnyDatabaseCtor {
|
|
|
23
23
|
|
|
24
24
|
let _db: AnyDatabase | null = null;
|
|
25
25
|
|
|
26
|
-
const
|
|
26
|
+
const TABLE_SCHEMA = `
|
|
27
27
|
CREATE TABLE IF NOT EXISTS memories (
|
|
28
28
|
id TEXT PRIMARY KEY,
|
|
29
29
|
scope_key TEXT NOT NULL,
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
30
|
+
scope TEXT NOT NULL,
|
|
31
|
+
visibility TEXT NOT NULL DEFAULT 'private',
|
|
32
|
+
project_name TEXT NOT NULL DEFAULT '',
|
|
33
|
+
type TEXT NOT NULL DEFAULT 'fact',
|
|
34
|
+
role TEXT NOT NULL DEFAULT 'knowledge',
|
|
35
|
+
importance TEXT NOT NULL DEFAULT 'normal',
|
|
36
|
+
status TEXT NOT NULL DEFAULT 'active',
|
|
33
37
|
tags TEXT NOT NULL DEFAULT '',
|
|
34
38
|
content TEXT NOT NULL,
|
|
39
|
+
cjk TEXT NOT NULL DEFAULT '',
|
|
40
|
+
content_hash TEXT NOT NULL DEFAULT '',
|
|
41
|
+
superseded_by TEXT,
|
|
35
42
|
source TEXT NOT NULL DEFAULT '',
|
|
36
43
|
file_path TEXT NOT NULL,
|
|
37
|
-
mtime_ms
|
|
44
|
+
mtime_ms REAL NOT NULL,
|
|
38
45
|
created_at INTEGER NOT NULL,
|
|
39
46
|
updated_at INTEGER NOT NULL
|
|
40
47
|
);
|
|
@@ -42,11 +49,20 @@ CREATE TABLE IF NOT EXISTS memories (
|
|
|
42
49
|
CREATE INDEX IF NOT EXISTS idx_memories_scope_updated
|
|
43
50
|
ON memories(scope_key, updated_at DESC);
|
|
44
51
|
CREATE INDEX IF NOT EXISTS idx_memories_type ON memories(type);
|
|
52
|
+
CREATE INDEX IF NOT EXISTS idx_memories_status ON memories(status);
|
|
53
|
+
`;
|
|
45
54
|
|
|
55
|
+
// The `cjk` column holds pre-tokenized CJK unigrams+bigrams (see
|
|
56
|
+
// src/retrieve/cjk.ts). FTS5's unicode61 treats a CJK run as one token, so
|
|
57
|
+
// without this column CJK substring search cannot work. Kept as a separate
|
|
58
|
+
// column (rather than a custom tokenizer) so both runtimes stay on stock
|
|
59
|
+
// SQLite with zero native dependencies.
|
|
60
|
+
const FTS_SCHEMA = `
|
|
46
61
|
CREATE VIRTUAL TABLE IF NOT EXISTS memories_fts USING fts5(
|
|
47
62
|
content,
|
|
48
63
|
tags,
|
|
49
64
|
type,
|
|
65
|
+
cjk,
|
|
50
66
|
scope_key UNINDEXED,
|
|
51
67
|
content='memories',
|
|
52
68
|
content_rowid='rowid',
|
|
@@ -54,23 +70,53 @@ CREATE VIRTUAL TABLE IF NOT EXISTS memories_fts USING fts5(
|
|
|
54
70
|
);
|
|
55
71
|
|
|
56
72
|
CREATE TRIGGER IF NOT EXISTS memories_ai AFTER INSERT ON memories BEGIN
|
|
57
|
-
INSERT INTO memories_fts(rowid, content, tags, type, scope_key)
|
|
58
|
-
VALUES (new.rowid, new.content, new.tags, new.type, new.scope_key);
|
|
73
|
+
INSERT INTO memories_fts(rowid, content, tags, type, cjk, scope_key)
|
|
74
|
+
VALUES (new.rowid, new.content, new.tags, new.type, new.cjk, new.scope_key);
|
|
59
75
|
END;
|
|
60
76
|
|
|
61
77
|
CREATE TRIGGER IF NOT EXISTS memories_ad AFTER DELETE ON memories BEGIN
|
|
62
|
-
INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, scope_key)
|
|
63
|
-
VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.scope_key);
|
|
78
|
+
INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, cjk, scope_key)
|
|
79
|
+
VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.cjk, old.scope_key);
|
|
64
80
|
END;
|
|
65
81
|
|
|
66
82
|
CREATE TRIGGER IF NOT EXISTS memories_au AFTER UPDATE ON memories BEGIN
|
|
67
|
-
INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, scope_key)
|
|
68
|
-
VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.scope_key);
|
|
69
|
-
INSERT INTO memories_fts(rowid, content, tags, type, scope_key)
|
|
70
|
-
VALUES (new.rowid, new.content, new.tags, new.type, new.scope_key);
|
|
83
|
+
INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, cjk, scope_key)
|
|
84
|
+
VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.cjk, old.scope_key);
|
|
85
|
+
INSERT INTO memories_fts(rowid, content, tags, type, cjk, scope_key)
|
|
86
|
+
VALUES (new.rowid, new.content, new.tags, new.type, new.cjk, new.scope_key);
|
|
71
87
|
END;
|
|
72
88
|
`;
|
|
73
89
|
|
|
90
|
+
/** Current index schema version. Bump when TABLE_SCHEMA/FTS_SCHEMA change. */
|
|
91
|
+
const SCHEMA_VERSION = 5;
|
|
92
|
+
|
|
93
|
+
function userVersion(d: AnyDatabase): number {
|
|
94
|
+
const row = d.prepare("PRAGMA user_version").get() as { user_version: number };
|
|
95
|
+
return row.user_version;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Any schema change: the query layer is fully derived from markdown (D1),
|
|
100
|
+
* so wipe it and let the next syncScope() repopulate. The markdown files
|
|
101
|
+
* themselves are untouched — `migrate --to-v2` handles the file format.
|
|
102
|
+
* NOTE ordering matters: triggers are dropped BEFORE the wipe, and the FTS
|
|
103
|
+
* table is rebuilt after. FTS5's 'delete' command corrupts
|
|
104
|
+
* (SQLITE_CORRUPT_VTAB) when it targets a rowid that was never indexed, so
|
|
105
|
+
* the wipe must not fire any FTS trigger while the index is out of sync
|
|
106
|
+
* with the table (found 2026-09-26).
|
|
107
|
+
*/
|
|
108
|
+
function rebuildIndexSchema(d: AnyDatabase): void {
|
|
109
|
+
d.exec(`DROP TRIGGER IF EXISTS memories_ai;
|
|
110
|
+
DROP TRIGGER IF EXISTS memories_ad;
|
|
111
|
+
DROP TRIGGER IF EXISTS memories_au;`);
|
|
112
|
+
d.exec("DELETE FROM memories");
|
|
113
|
+
d.exec(`DROP TABLE IF EXISTS memories_fts;`);
|
|
114
|
+
d.exec(`DROP TABLE IF EXISTS memories;`);
|
|
115
|
+
d.exec(TABLE_SCHEMA);
|
|
116
|
+
d.exec(FTS_SCHEMA);
|
|
117
|
+
d.exec(`PRAGMA user_version = ${SCHEMA_VERSION}`);
|
|
118
|
+
}
|
|
119
|
+
|
|
74
120
|
export function db(): AnyDatabase {
|
|
75
121
|
if (_db) return _db;
|
|
76
122
|
const { indexDb } = paths();
|
|
@@ -79,7 +125,9 @@ export function db(): AnyDatabase {
|
|
|
79
125
|
d.exec("PRAGMA journal_mode = WAL;");
|
|
80
126
|
d.exec("PRAGMA synchronous = NORMAL;");
|
|
81
127
|
d.exec("PRAGMA foreign_keys = ON;");
|
|
82
|
-
d.exec(
|
|
128
|
+
d.exec(TABLE_SCHEMA);
|
|
129
|
+
d.exec(FTS_SCHEMA);
|
|
130
|
+
if (userVersion(d) < SCHEMA_VERSION) rebuildIndexSchema(d);
|
|
83
131
|
_db = d;
|
|
84
132
|
return d;
|
|
85
133
|
}
|