open-memex 0.1.0 → 0.2.0-alpha

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/redact.ts CHANGED
@@ -1,5 +1,104 @@
1
- export function stripPrivate(text: string): string {
2
- return text.replace(/<private>[\s\S]*?<\/private>/gi, "[REDACTED]");
1
+ /**
2
+ * Redaction: secrets must never land in memory files or the index.
3
+ *
4
+ * Three layers, checked in order by redact():
5
+ * 1. <private>...</private> regions are stripped first (explicit opt-out —
6
+ * the author marked this span as sensitive, so it is replaced with
7
+ * [REDACTED] before any detection runs).
8
+ * 2. Built-in provider patterns (always on, reported by id).
9
+ * 3. User patterns from config redactPatterns (reported verbatim).
10
+ * 4. High-entropy assignment heuristic (catches secrets whose provider we
11
+ * don't have a pattern for, e.g. `deploy_key = "aB3d..."`).
12
+ *
13
+ * Any hit refuses the whole write — a memory with a hole in it is worse
14
+ * than no memory, because the hole invites reconstruction.
15
+ */
16
+
17
+ export interface SecretPattern {
18
+ /** Stable id reported in matchedPattern. */
19
+ id: string;
20
+ /** Regex source (no slashes). */
21
+ source: string;
22
+ /** Optional RegExp flags, e.g. "i". */
23
+ flags?: string;
24
+ }
25
+
26
+ /**
27
+ * Prefixes like `sk-` also occur inside ordinary English words ("task-…",
28
+ * "risk-…", "disk-…"), which caused confirmed false positives. Require the
29
+ * token prefix NOT to be preceded by a word/hyphen char, so a real key after
30
+ * "=", ":", space, or a quote still matches.
31
+ */
32
+ const TOKEN_BOUNDARY = "(?<![A-Za-z0-9_-])";
33
+
34
+ export const BUILTIN_SECRET_PATTERNS: SecretPattern[] = [
35
+ { id: "openai-key", source: `${TOKEN_BOUNDARY}sk-[A-Za-z0-9_-]{20,}` },
36
+ {
37
+ id: "openai-admin-key",
38
+ source: `${TOKEN_BOUNDARY}sk-admin-[A-Za-z0-9_-]{20,}`,
39
+ },
40
+ {
41
+ id: "openai-session-key",
42
+ source: `${TOKEN_BOUNDARY}sm_[A-Za-z0-9_-]{20,}`,
43
+ },
44
+ { id: "github-pat", source: `${TOKEN_BOUNDARY}ghp_[A-Za-z0-9]{30,}` },
45
+ { id: "github-oauth-token", source: `${TOKEN_BOUNDARY}gho_[A-Za-z0-9]{30,}` },
46
+ { id: "github-user-token", source: `${TOKEN_BOUNDARY}ghu_[A-Za-z0-9]{30,}` },
47
+ {
48
+ id: "github-refresh-token",
49
+ source: `${TOKEN_BOUNDARY}ghr_[A-Za-z0-9]{30,}`,
50
+ },
51
+ {
52
+ id: "github-fine-grained-pat",
53
+ source: `${TOKEN_BOUNDARY}github_pat_[A-Za-z0-9_]{40,}`,
54
+ },
55
+ { id: "aws-access-key-id", source: `${TOKEN_BOUNDARY}AKIA[0-9A-Z]{16}` },
56
+ {
57
+ id: "aws-secret-access-key",
58
+ source:
59
+ "aws[_-]?secret[_-]?access[_-]?key[\"']?\\s*[:=]\\s*[\"']?[A-Za-z0-9/+=]{40}",
60
+ flags: "i",
61
+ },
62
+ { id: "slack-token", source: `${TOKEN_BOUNDARY}xox[baprs]-[A-Za-z0-9-]{10,}` },
63
+ { id: "google-api-key", source: `${TOKEN_BOUNDARY}AIza[0-9A-Za-z_-]{30,}` },
64
+ { id: "npm-token", source: `${TOKEN_BOUNDARY}npm_[A-Za-z0-9]{30,}` },
65
+ { id: "gitlab-pat", source: `${TOKEN_BOUNDARY}glpat-[A-Za-z0-9_-]{20,}` },
66
+ {
67
+ id: "stripe-restricted-key",
68
+ source: `${TOKEN_BOUNDARY}rk_(live|test)_[A-Za-z0-9]{20,}`,
69
+ },
70
+ {
71
+ id: "stripe-webhook-secret",
72
+ source: `${TOKEN_BOUNDARY}whsec_[A-Za-z0-9]{20,}`,
73
+ },
74
+ { id: "private-key-block", source: "-----BEGIN [A-Z ]*PRIVATE KEY-----" },
75
+ {
76
+ id: "jwt",
77
+ source: `${TOKEN_BOUNDARY}eyJ[A-Za-z0-9_-]{10,}\\.[A-Za-z0-9_-]{10,}\\.[A-Za-z0-9_-]{10,}`,
78
+ },
79
+ {
80
+ id: "generic-secret-assignment",
81
+ source:
82
+ "(api[_-]?key|secret|passwd|password|auth[_-]?token|access[_-]?token)[\"']?\\s*[:=]\\s*[\"']?[A-Za-z0-9_\\-./+=]{16,}[\"']?",
83
+ flags: "i",
84
+ },
85
+ ];
86
+
87
+ function compile(p: SecretPattern): RegExp | null {
88
+ try {
89
+ return new RegExp(p.source, p.flags ?? "");
90
+ } catch {
91
+ return null; // ignore malformed builtin (should never happen)
92
+ }
93
+ }
94
+
95
+ /** Returns the matched builtin pattern id, or null. */
96
+ export function findBuiltinSecret(text: string): string | null {
97
+ for (const p of BUILTIN_SECRET_PATTERNS) {
98
+ const re = compile(p);
99
+ if (re && re.test(text)) return p.id;
100
+ }
101
+ return null;
3
102
  }
4
103
 
5
104
  export function findSecret(text: string, patterns: string[]): string | null {
@@ -14,11 +113,59 @@ export function findSecret(text: string, patterns: string[]): string | null {
14
113
  return null;
15
114
  }
16
115
 
116
+ function shannonEntropy(s: string): number {
117
+ const freq = new Map<string, number>();
118
+ for (const c of s) freq.set(c, (freq.get(c) ?? 0) + 1);
119
+ let h = 0;
120
+ for (const n of freq.values()) {
121
+ const p = n / s.length;
122
+ h -= p * Math.log2(p);
123
+ }
124
+ return h;
125
+ }
126
+
127
+ // name = value assignments with a long token-ish value.
128
+ const ASSIGNMENT_RE =
129
+ /([A-Za-z_][A-Za-z0-9_]{1,63})\s*[:=]\s*["']?([A-Za-z0-9_\-./+=]{24,})["']?/g;
130
+
131
+ /**
132
+ * Heuristic last line of defense: a long, high-entropy value assigned to a
133
+ * name is almost certainly a credential, even when no provider pattern
134
+ * matches. Tuned conservatively (length >= 24, entropy >= 4.5 bits/char):
135
+ * hex digests (<= 4.0) and prose (~4.0) pass through; base64-ish randomness
136
+ * does not. URLs are skipped.
137
+ */
138
+ export function findHighEntropySecret(text: string): string | null {
139
+ for (const m of text.matchAll(ASSIGNMENT_RE)) {
140
+ const name = m[1];
141
+ const value = m[2];
142
+ if (value.includes("://")) continue; // URL, not a secret
143
+ if (shannonEntropy(value) >= 4.5) return `high-entropy-secret:${name}`;
144
+ }
145
+ return null;
146
+ }
147
+
148
+ export function stripPrivate(text: string): string {
149
+ // Closed pairs first...
150
+ let out = text.replace(/<private>[\s\S]*?<\/private>/gi, "[REDACTED]");
151
+ // ...then an unclosed <private> redacts everything after it. A dangling
152
+ // tag almost always means the author intended the rest to be private.
153
+ out = out.replace(/<private>[\s\S]*$/gi, "[REDACTED]");
154
+ return out;
155
+ }
156
+
17
157
  export function redact(
18
158
  text: string,
19
159
  patterns: string[],
20
160
  ): { content: string; hadSecret: boolean; matchedPattern: string | null } {
21
161
  const stripped = stripPrivate(text);
22
- const matched = findSecret(stripped, patterns);
23
- return { content: stripped, hadSecret: matched !== null, matchedPattern: matched };
162
+ const builtin = findBuiltinSecret(stripped);
163
+ if (builtin)
164
+ return { content: stripped, hadSecret: true, matchedPattern: builtin };
165
+ const user = findSecret(stripped, patterns);
166
+ if (user) return { content: stripped, hadSecret: true, matchedPattern: user };
167
+ const entropic = findHighEntropySecret(stripped);
168
+ if (entropic)
169
+ return { content: stripped, hadSecret: true, matchedPattern: entropic };
170
+ return { content: stripped, hadSecret: false, matchedPattern: null };
24
171
  }
@@ -0,0 +1,63 @@
1
+ // CJK retrieval helpers (pure logic, no sqlite).
2
+ //
3
+ // FTS5's `porter unicode61` tokenizer treats a run of CJK ideographs as one
4
+ // token ("中文记忆" is a single token), so substring queries never match, and
5
+ // the old query builder dropped CJK characters entirely. Per D8 the default
6
+ // CJK strategy is bigram: at write time we pre-tokenize CJK runs into
7
+ // space-separated unigrams + overlapping bigrams stored in the `cjk` FTS
8
+ // column; at query time the CJK part of the query becomes an OR of bigrams
9
+ // against that column. Unigrams are included so single-character queries
10
+ // ("猫") still match. Works identically on bun:sqlite and better-sqlite3
11
+ // with no native tokenizer dependency.
12
+
13
+ const CJK_RE =
14
+ /[\u3400-\u4DBF\u4E00-\u9FFF\u3040-\u309F\u30A0-\u30FF\uAC00-\uD7AF\u1100-\u11FF\u{20000}-\u{2A6DF}]/u;
15
+
16
+ const CJK_RUN_RE =
17
+ /[\u3400-\u4DBF\u4E00-\u9FFF\u3040-\u309F\u30A0-\u30FF\uAC00-\uD7AF\u1100-\u11FF\u{20000}-\u{2A6DF}]+/gu;
18
+
19
+ /** True if the string contains any CJK (Han/Hiragana/Katakana/Hangul) character. */
20
+ export function hasCjk(s: string): boolean {
21
+ return CJK_RE.test(s);
22
+ }
23
+
24
+ /**
25
+ * Build the index text for the `cjk` FTS column: for every CJK run emit each
26
+ * character (unigram) then every overlapping bigram, space-separated.
27
+ * "中文记忆" -> "中 文 记 忆 中文 文记 记忆". Non-CJK text yields "".
28
+ */
29
+ export function cjkIndexText(text: string): string {
30
+ const out: string[] = [];
31
+ for (const m of text.matchAll(CJK_RUN_RE)) {
32
+ const chars = [...m[0]];
33
+ for (const ch of chars) out.push(ch);
34
+ for (let i = 0; i + 1 < chars.length; i++) out.push(chars[i] + chars[i + 1]);
35
+ }
36
+ return out.join(" ");
37
+ }
38
+
39
+ /** Escape a raw token for embedding in a double-quoted FTS5 phrase. */
40
+ function esc(t: string): string {
41
+ return t.replace(/"/g, '""');
42
+ }
43
+
44
+ /**
45
+ * Convert the CJK runs of a query into an FTS5 expression for the `cjk`
46
+ * column. Multi-char runs become an OR of bigrams (recall-oriented; bm25
47
+ * ranks docs matching more bigrams higher). Single chars stay unigrams.
48
+ * Returns "" when the query has no CJK.
49
+ */
50
+ export function cjkQueryExpr(query: string): string {
51
+ const parts: string[] = [];
52
+ for (const m of query.matchAll(CJK_RUN_RE)) {
53
+ const chars = [...m[0]];
54
+ if (chars.length === 1) {
55
+ parts.push(`"${esc(chars[0])}"`);
56
+ } else {
57
+ for (let i = 0; i + 1 < chars.length; i++) {
58
+ parts.push(`"${esc(chars[i] + chars[i + 1])}"`);
59
+ }
60
+ }
61
+ }
62
+ return parts.join(" OR ");
63
+ }
@@ -1,6 +1,6 @@
1
1
  import { list } from "./search.ts";
2
2
  import type { Scope } from "../scope.ts";
3
- import { USER_SCOPE } from "../scope.ts";
3
+ import { PERSONAL_SCOPE } from "../scope.ts";
4
4
  import type { MyOMemoryConfig } from "../config.ts";
5
5
 
6
6
  function oneLine(s: string, max = 240): string {
@@ -10,7 +10,7 @@ function oneLine(s: string, max = 240): string {
10
10
 
11
11
  export function buildContextBlock(scope: Scope, cfg: MyOMemoryConfig): string | null {
12
12
  const project = list(scope.key, { limit: cfg.maxProjectMemories });
13
- const user = list(USER_SCOPE.key, { limit: cfg.maxProfileItems });
13
+ const user = list(PERSONAL_SCOPE.key, { limit: cfg.maxProfileItems });
14
14
 
15
15
  if (project.length === 0 && user.length === 0) return null;
16
16
 
@@ -1,4 +1,5 @@
1
1
  import { db } from "../store/db.ts";
2
+ import { cjkQueryExpr, hasCjk } from "./cjk.ts";
2
3
 
3
4
  export interface SearchHit {
4
5
  id: string;
@@ -9,13 +10,105 @@ export interface SearchHit {
9
10
  snippet: string;
10
11
  score: number;
11
12
  updated_at: number;
13
+ status: string;
12
14
  }
13
15
 
14
- /** Convert free-text query into a safe FTS5 MATCH expression. */
16
+ interface RawRow {
17
+ id: string;
18
+ scope_key: string;
19
+ project_name: string;
20
+ type: string;
21
+ tags: string;
22
+ updated_at: number;
23
+ snippet: string;
24
+ score: number;
25
+ status: string;
26
+ superseded_by: string | null;
27
+ }
28
+
29
+ function toHit(r: RawRow): SearchHit {
30
+ return {
31
+ id: r.id,
32
+ scope_key: r.scope_key,
33
+ project_name: r.project_name,
34
+ type: r.type,
35
+ tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
36
+ snippet: r.snippet ?? "",
37
+ score: r.score,
38
+ updated_at: r.updated_at,
39
+ status: r.status,
40
+ };
41
+ }
42
+
43
+ /**
44
+ * Lifecycle-aware post-processing (§3.3):
45
+ * - retracted / archived are excluded from retrieval (kept for audit);
46
+ * - a superseded memory resolves to the newest of its chain (cycle-safe);
47
+ * - deprecated stays visible as a warning but ranks after active.
48
+ */
49
+ function resolveVisible(rows: RawRow[], limit: number): SearchHit[] {
50
+ const byId = new Map(rows.map((r) => [r.id, r]));
51
+ const fullRow = (id: string): RawRow | undefined => {
52
+ const cached = byId.get(id);
53
+ if (cached) return cached;
54
+ const r = db()
55
+ .prepare(
56
+ `SELECT id, scope_key, project_name, type, tags, updated_at, status,
57
+ superseded_by, substr(content, 1, 240) AS snippet
58
+ FROM memories WHERE id = ?`,
59
+ )
60
+ .get(id) as
61
+ | (Omit<RawRow, "score" | "snippet"> & { snippet: string })
62
+ | undefined;
63
+ if (!r) return undefined;
64
+ const full: RawRow = { ...r, score: 0 };
65
+ byId.set(id, full);
66
+ return full;
67
+ };
68
+
69
+ const seen = new Set<string>();
70
+ const active: SearchHit[] = [];
71
+ const deprecated: SearchHit[] = [];
72
+
73
+ for (const r of rows) {
74
+ if (r.status === "retracted" || r.status === "archived") continue;
75
+ let target = r;
76
+ if (r.status === "superseded") {
77
+ let cur = r;
78
+ const chain = new Set([r.id]);
79
+ while (cur.status === "superseded" && cur.superseded_by) {
80
+ if (chain.has(cur.superseded_by)) break; // cycle guard
81
+ chain.add(cur.superseded_by);
82
+ const nxt = fullRow(cur.superseded_by);
83
+ if (!nxt) break;
84
+ cur = nxt;
85
+ }
86
+ if (cur.status === "retracted" || cur.status === "archived") continue;
87
+ target = cur;
88
+ }
89
+ if (seen.has(target.id)) continue;
90
+ seen.add(target.id);
91
+ const hit = toHit(target);
92
+ if (target.status === "deprecated") deprecated.push(hit);
93
+ else active.push(hit);
94
+ }
95
+ return [...active, ...deprecated].slice(0, limit);
96
+ }
97
+
98
+ /**
99
+ * Convert free-text query into a safe FTS5 MATCH expression.
100
+ * Latin tokens keep the old behavior (prefix match on content/tags/type).
101
+ * CJK runs become an OR of bigrams against the `cjk` column (see cjk.ts).
102
+ * Mixed queries OR the two parts together.
103
+ */
15
104
  function toFtsQuery(q: string): string {
16
- const tokens = q.toLowerCase().match(/[a-z0-9_.\-]+/g) ?? [];
17
- if (tokens.length === 0) return "";
18
- return tokens.map((t) => `"${t.replace(/"/g, '""')}"*`).join(" OR ");
105
+ const latin = (q.toLowerCase().match(/[a-z0-9_.\-]+/g) ?? [])
106
+ .map((t) => `"${t.replace(/"/g, '""')}"*`)
107
+ .join(" OR ");
108
+ const cjk = hasCjk(q) ? cjkQueryExpr(q) : "";
109
+ if (latin && cjk) return `(${latin}) OR {cjk}:(${cjk})`;
110
+ if (cjk) return `{cjk}:(${cjk})`;
111
+ return latin;
19
112
  }
20
113
 
21
114
  export function search(
@@ -25,6 +118,9 @@ export function search(
25
118
  const q = toFtsQuery(query);
26
119
  if (!q) return [];
27
120
  const limit = Math.max(1, Math.min(opts.limit ?? 8, 50));
121
+ // Over-fetch: lifecycle filtering (chain resolution, exclusions) happens
122
+ // after the FTS query, so candidates must survive it.
123
+ const fetchLimit = Math.min(limit * 3 + 10, 150);
28
124
 
29
125
  const scopeFilter =
30
126
  opts.scopeKeys && opts.scopeKeys.length > 0
@@ -34,6 +130,7 @@ export function search(
34
130
 
35
131
  const sql = `
36
132
  SELECT m.id, m.scope_key, m.project_name, m.type, m.tags, m.updated_at,
133
+ m.status, m.superseded_by,
37
134
  snippet(memories_fts, 0, '[', ']', ' ... ', 12) AS snippet,
38
135
  bm25(memories_fts) AS score
39
136
  FROM memories_fts
@@ -45,7 +142,7 @@ export function search(
45
142
  const params: unknown[] = [q];
46
143
  if (opts.scopeKeys && opts.scopeKeys.length > 0) params.push(...opts.scopeKeys);
47
144
  if (opts.type) params.push(opts.type);
48
- params.push(limit);
145
+ params.push(fetchLimit);
49
146
 
50
147
  const rows = db()
51
148
  .prepare(sql)
@@ -56,20 +153,15 @@ export function search(
56
153
  type: string;
57
154
  tags: string;
58
155
  updated_at: number;
156
+ status: string;
157
+ superseded_by: string | null;
59
158
  snippet: string;
60
159
  score: number;
61
160
  }>;
62
161
 
63
- return rows.map((r) => ({
64
- id: r.id,
65
- scope_key: r.scope_key,
66
- project_name: r.project_name,
67
- type: r.type,
68
- tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
69
- snippet: r.snippet ?? "",
70
- score: -r.score, // FTS5 bm25: lower = better; invert for intuition
71
- updated_at: r.updated_at,
72
- }));
162
+ // FTS5 bm25: lower = better; invert for intuition.
163
+ const raw: RawRow[] = rows.map((r) => ({ ...r, score: -r.score }));
164
+ return resolveVisible(raw, limit);
73
165
  }
74
166
 
75
167
  export function list(
@@ -77,10 +169,11 @@ export function list(
77
169
  opts: { type?: string; limit?: number } = {},
78
170
  ): SearchHit[] {
79
171
  const limit = Math.max(1, Math.min(opts.limit ?? 20, 100));
172
+ const fetchLimit = Math.min(limit * 3 + 10, 150);
80
173
  const typeFilter = opts.type ? ` AND type = ?` : "";
81
174
  const sql = `
82
- SELECT id, scope_key, project_name, type, tags, updated_at,
83
- substr(content, 1, 240) AS snippet
175
+ SELECT id, scope_key, project_name, type, tags, updated_at, status,
176
+ superseded_by, substr(content, 1, 240) AS snippet
84
177
  FROM memories
85
178
  WHERE scope_key = ?${typeFilter}
86
179
  ORDER BY updated_at DESC
@@ -88,7 +181,7 @@ export function list(
88
181
  `;
89
182
  const params: unknown[] = [scopeKey];
90
183
  if (opts.type) params.push(opts.type);
91
- params.push(limit);
184
+ params.push(fetchLimit);
92
185
 
93
186
  const rows = db()
94
187
  .prepare(sql)
@@ -99,17 +192,11 @@ export function list(
99
192
  type: string;
100
193
  tags: string;
101
194
  updated_at: number;
195
+ status: string;
196
+ superseded_by: string | null;
102
197
  snippet: string;
103
198
  }>;
104
199
 
105
- return rows.map((r) => ({
106
- id: r.id,
107
- scope_key: r.scope_key,
108
- project_name: r.project_name,
109
- type: r.type,
110
- tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
111
- snippet: r.snippet ?? "",
112
- score: 1,
113
- updated_at: r.updated_at,
114
- }));
200
+ const raw: RawRow[] = rows.map((r) => ({ ...r, score: 1 }));
201
+ return resolveVisible(raw, limit);
115
202
  }
package/src/scope.ts CHANGED
@@ -4,11 +4,16 @@ import path from "node:path";
4
4
 
5
5
  export interface Scope {
6
6
  key: string;
7
- kind: "user" | "project";
7
+ kind: "personal" | "project" | "org";
8
8
  projectName: string;
9
9
  }
10
10
 
11
- export const USER_SCOPE: Scope = { key: "user", kind: "user", projectName: "user" };
11
+ /** v2: v1 `user` scope is renamed to `personal` (§19). */
12
+ export const PERSONAL_SCOPE: Scope = {
13
+ key: "personal",
14
+ kind: "personal",
15
+ projectName: "personal",
16
+ };
12
17
 
13
18
  function normalizeRemote(url: string): string {
14
19
  return url
package/src/store/db.ts CHANGED
@@ -23,18 +23,25 @@ function loadDatabase(): AnyDatabaseCtor {
23
23
 
24
24
  let _db: AnyDatabase | null = null;
25
25
 
26
- const SCHEMA = `
26
+ const TABLE_SCHEMA = `
27
27
  CREATE TABLE IF NOT EXISTS memories (
28
28
  id TEXT PRIMARY KEY,
29
29
  scope_key TEXT NOT NULL,
30
- scope_kind TEXT NOT NULL,
31
- project_name TEXT NOT NULL,
32
- type TEXT NOT NULL DEFAULT 'note',
30
+ scope TEXT NOT NULL,
31
+ visibility TEXT NOT NULL DEFAULT 'private',
32
+ project_name TEXT NOT NULL DEFAULT '',
33
+ type TEXT NOT NULL DEFAULT 'fact',
34
+ role TEXT NOT NULL DEFAULT 'knowledge',
35
+ importance TEXT NOT NULL DEFAULT 'normal',
36
+ status TEXT NOT NULL DEFAULT 'active',
33
37
  tags TEXT NOT NULL DEFAULT '',
34
38
  content TEXT NOT NULL,
39
+ cjk TEXT NOT NULL DEFAULT '',
40
+ content_hash TEXT NOT NULL DEFAULT '',
41
+ superseded_by TEXT,
35
42
  source TEXT NOT NULL DEFAULT '',
36
43
  file_path TEXT NOT NULL,
37
- mtime_ms INTEGER NOT NULL,
44
+ mtime_ms REAL NOT NULL,
38
45
  created_at INTEGER NOT NULL,
39
46
  updated_at INTEGER NOT NULL
40
47
  );
@@ -42,11 +49,20 @@ CREATE TABLE IF NOT EXISTS memories (
42
49
  CREATE INDEX IF NOT EXISTS idx_memories_scope_updated
43
50
  ON memories(scope_key, updated_at DESC);
44
51
  CREATE INDEX IF NOT EXISTS idx_memories_type ON memories(type);
52
+ CREATE INDEX IF NOT EXISTS idx_memories_status ON memories(status);
53
+ `;
45
54
 
55
+ // The `cjk` column holds pre-tokenized CJK unigrams+bigrams (see
56
+ // src/retrieve/cjk.ts). FTS5's unicode61 treats a CJK run as one token, so
57
+ // without this column CJK substring search cannot work. Kept as a separate
58
+ // column (rather than a custom tokenizer) so both runtimes stay on stock
59
+ // SQLite with zero native dependencies.
60
+ const FTS_SCHEMA = `
46
61
  CREATE VIRTUAL TABLE IF NOT EXISTS memories_fts USING fts5(
47
62
  content,
48
63
  tags,
49
64
  type,
65
+ cjk,
50
66
  scope_key UNINDEXED,
51
67
  content='memories',
52
68
  content_rowid='rowid',
@@ -54,23 +70,53 @@ CREATE VIRTUAL TABLE IF NOT EXISTS memories_fts USING fts5(
54
70
  );
55
71
 
56
72
  CREATE TRIGGER IF NOT EXISTS memories_ai AFTER INSERT ON memories BEGIN
57
- INSERT INTO memories_fts(rowid, content, tags, type, scope_key)
58
- VALUES (new.rowid, new.content, new.tags, new.type, new.scope_key);
73
+ INSERT INTO memories_fts(rowid, content, tags, type, cjk, scope_key)
74
+ VALUES (new.rowid, new.content, new.tags, new.type, new.cjk, new.scope_key);
59
75
  END;
60
76
 
61
77
  CREATE TRIGGER IF NOT EXISTS memories_ad AFTER DELETE ON memories BEGIN
62
- INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, scope_key)
63
- VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.scope_key);
78
+ INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, cjk, scope_key)
79
+ VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.cjk, old.scope_key);
64
80
  END;
65
81
 
66
82
  CREATE TRIGGER IF NOT EXISTS memories_au AFTER UPDATE ON memories BEGIN
67
- INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, scope_key)
68
- VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.scope_key);
69
- INSERT INTO memories_fts(rowid, content, tags, type, scope_key)
70
- VALUES (new.rowid, new.content, new.tags, new.type, new.scope_key);
83
+ INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, cjk, scope_key)
84
+ VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.cjk, old.scope_key);
85
+ INSERT INTO memories_fts(rowid, content, tags, type, cjk, scope_key)
86
+ VALUES (new.rowid, new.content, new.tags, new.type, new.cjk, new.scope_key);
71
87
  END;
72
88
  `;
73
89
 
90
+ /** Current index schema version. Bump when TABLE_SCHEMA/FTS_SCHEMA change. */
91
+ const SCHEMA_VERSION = 5;
92
+
93
+ function userVersion(d: AnyDatabase): number {
94
+ const row = d.prepare("PRAGMA user_version").get() as { user_version: number };
95
+ return row.user_version;
96
+ }
97
+
98
+ /**
99
+ * Any schema change: the query layer is fully derived from markdown (D1),
100
+ * so wipe it and let the next syncScope() repopulate. The markdown files
101
+ * themselves are untouched — `migrate --to-v2` handles the file format.
102
+ * NOTE ordering matters: triggers are dropped BEFORE the wipe, and the FTS
103
+ * table is rebuilt after. FTS5's 'delete' command corrupts
104
+ * (SQLITE_CORRUPT_VTAB) when it targets a rowid that was never indexed, so
105
+ * the wipe must not fire any FTS trigger while the index is out of sync
106
+ * with the table (found 2026-09-26).
107
+ */
108
+ function rebuildIndexSchema(d: AnyDatabase): void {
109
+ d.exec(`DROP TRIGGER IF EXISTS memories_ai;
110
+ DROP TRIGGER IF EXISTS memories_ad;
111
+ DROP TRIGGER IF EXISTS memories_au;`);
112
+ d.exec("DELETE FROM memories");
113
+ d.exec(`DROP TABLE IF EXISTS memories_fts;`);
114
+ d.exec(`DROP TABLE IF EXISTS memories;`);
115
+ d.exec(TABLE_SCHEMA);
116
+ d.exec(FTS_SCHEMA);
117
+ d.exec(`PRAGMA user_version = ${SCHEMA_VERSION}`);
118
+ }
119
+
74
120
  export function db(): AnyDatabase {
75
121
  if (_db) return _db;
76
122
  const { indexDb } = paths();
@@ -79,7 +125,9 @@ export function db(): AnyDatabase {
79
125
  d.exec("PRAGMA journal_mode = WAL;");
80
126
  d.exec("PRAGMA synchronous = NORMAL;");
81
127
  d.exec("PRAGMA foreign_keys = ON;");
82
- d.exec(SCHEMA);
128
+ d.exec(TABLE_SCHEMA);
129
+ d.exec(FTS_SCHEMA);
130
+ if (userVersion(d) < SCHEMA_VERSION) rebuildIndexSchema(d);
83
131
  _db = d;
84
132
  return d;
85
133
  }