open-memex 0.1.0 → 0.3.0-alpha

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/store/db.ts CHANGED
@@ -23,18 +23,25 @@ function loadDatabase(): AnyDatabaseCtor {
23
23
 
24
24
  let _db: AnyDatabase | null = null;
25
25
 
26
- const SCHEMA = `
26
+ const TABLE_SCHEMA = `
27
27
  CREATE TABLE IF NOT EXISTS memories (
28
28
  id TEXT PRIMARY KEY,
29
29
  scope_key TEXT NOT NULL,
30
- scope_kind TEXT NOT NULL,
31
- project_name TEXT NOT NULL,
32
- type TEXT NOT NULL DEFAULT 'note',
30
+ scope TEXT NOT NULL,
31
+ visibility TEXT NOT NULL DEFAULT 'private',
32
+ project_name TEXT NOT NULL DEFAULT '',
33
+ type TEXT NOT NULL DEFAULT 'fact',
34
+ role TEXT NOT NULL DEFAULT 'knowledge',
35
+ importance TEXT NOT NULL DEFAULT 'normal',
36
+ status TEXT NOT NULL DEFAULT 'active',
33
37
  tags TEXT NOT NULL DEFAULT '',
34
38
  content TEXT NOT NULL,
39
+ cjk TEXT NOT NULL DEFAULT '',
40
+ content_hash TEXT NOT NULL DEFAULT '',
41
+ superseded_by TEXT,
35
42
  source TEXT NOT NULL DEFAULT '',
36
43
  file_path TEXT NOT NULL,
37
- mtime_ms INTEGER NOT NULL,
44
+ mtime_ms REAL NOT NULL,
38
45
  created_at INTEGER NOT NULL,
39
46
  updated_at INTEGER NOT NULL
40
47
  );
@@ -42,11 +49,20 @@ CREATE TABLE IF NOT EXISTS memories (
42
49
  CREATE INDEX IF NOT EXISTS idx_memories_scope_updated
43
50
  ON memories(scope_key, updated_at DESC);
44
51
  CREATE INDEX IF NOT EXISTS idx_memories_type ON memories(type);
52
+ CREATE INDEX IF NOT EXISTS idx_memories_status ON memories(status);
53
+ `;
45
54
 
55
+ // The `cjk` column holds pre-tokenized CJK unigrams+bigrams (see
56
+ // src/retrieve/cjk.ts). FTS5's unicode61 treats a CJK run as one token, so
57
+ // without this column CJK substring search cannot work. Kept as a separate
58
+ // column (rather than a custom tokenizer) so both runtimes stay on stock
59
+ // SQLite with zero native dependencies.
60
+ const FTS_SCHEMA = `
46
61
  CREATE VIRTUAL TABLE IF NOT EXISTS memories_fts USING fts5(
47
62
  content,
48
63
  tags,
49
64
  type,
65
+ cjk,
50
66
  scope_key UNINDEXED,
51
67
  content='memories',
52
68
  content_rowid='rowid',
@@ -54,23 +70,53 @@ CREATE VIRTUAL TABLE IF NOT EXISTS memories_fts USING fts5(
54
70
  );
55
71
 
56
72
  CREATE TRIGGER IF NOT EXISTS memories_ai AFTER INSERT ON memories BEGIN
57
- INSERT INTO memories_fts(rowid, content, tags, type, scope_key)
58
- VALUES (new.rowid, new.content, new.tags, new.type, new.scope_key);
73
+ INSERT INTO memories_fts(rowid, content, tags, type, cjk, scope_key)
74
+ VALUES (new.rowid, new.content, new.tags, new.type, new.cjk, new.scope_key);
59
75
  END;
60
76
 
61
77
  CREATE TRIGGER IF NOT EXISTS memories_ad AFTER DELETE ON memories BEGIN
62
- INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, scope_key)
63
- VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.scope_key);
78
+ INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, cjk, scope_key)
79
+ VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.cjk, old.scope_key);
64
80
  END;
65
81
 
66
82
  CREATE TRIGGER IF NOT EXISTS memories_au AFTER UPDATE ON memories BEGIN
67
- INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, scope_key)
68
- VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.scope_key);
69
- INSERT INTO memories_fts(rowid, content, tags, type, scope_key)
70
- VALUES (new.rowid, new.content, new.tags, new.type, new.scope_key);
83
+ INSERT INTO memories_fts(memories_fts, rowid, content, tags, type, cjk, scope_key)
84
+ VALUES ('delete', old.rowid, old.content, old.tags, old.type, old.cjk, old.scope_key);
85
+ INSERT INTO memories_fts(rowid, content, tags, type, cjk, scope_key)
86
+ VALUES (new.rowid, new.content, new.tags, new.type, new.cjk, new.scope_key);
71
87
  END;
72
88
  `;
73
89
 
90
+ /** Current index schema version. Bump when TABLE_SCHEMA/FTS_SCHEMA change. */
91
+ const SCHEMA_VERSION = 5;
92
+
93
+ function userVersion(d: AnyDatabase): number {
94
+ const row = d.prepare("PRAGMA user_version").get() as { user_version: number };
95
+ return row.user_version;
96
+ }
97
+
98
+ /**
99
+ * Any schema change: the query layer is fully derived from markdown (D1),
100
+ * so wipe it and let the next syncScope() repopulate. The markdown files
101
+ * themselves are untouched — `migrate --to-v2` handles the file format.
102
+ * NOTE ordering matters: triggers are dropped BEFORE the wipe, and the FTS
103
+ * table is rebuilt after. FTS5's 'delete' command corrupts
104
+ * (SQLITE_CORRUPT_VTAB) when it targets a rowid that was never indexed, so
105
+ * the wipe must not fire any FTS trigger while the index is out of sync
106
+ * with the table (found 2026-09-26).
107
+ */
108
+ function rebuildIndexSchema(d: AnyDatabase): void {
109
+ d.exec(`DROP TRIGGER IF EXISTS memories_ai;
110
+ DROP TRIGGER IF EXISTS memories_ad;
111
+ DROP TRIGGER IF EXISTS memories_au;`);
112
+ d.exec("DELETE FROM memories");
113
+ d.exec(`DROP TABLE IF EXISTS memories_fts;`);
114
+ d.exec(`DROP TABLE IF EXISTS memories;`);
115
+ d.exec(TABLE_SCHEMA);
116
+ d.exec(FTS_SCHEMA);
117
+ d.exec(`PRAGMA user_version = ${SCHEMA_VERSION}`);
118
+ }
119
+
74
120
  export function db(): AnyDatabase {
75
121
  if (_db) return _db;
76
122
  const { indexDb } = paths();
@@ -79,7 +125,9 @@ export function db(): AnyDatabase {
79
125
  d.exec("PRAGMA journal_mode = WAL;");
80
126
  d.exec("PRAGMA synchronous = NORMAL;");
81
127
  d.exec("PRAGMA foreign_keys = ON;");
82
- d.exec(SCHEMA);
128
+ d.exec(TABLE_SCHEMA);
129
+ d.exec(FTS_SCHEMA);
130
+ if (userVersion(d) < SCHEMA_VERSION) rebuildIndexSchema(d);
83
131
  _db = d;
84
132
  return d;
85
133
  }
@@ -0,0 +1,280 @@
1
+ import { createHash } from "node:crypto";
2
+ import fs from "node:fs";
3
+ import path from "node:path";
4
+ import { db } from "./db.ts";
5
+ import { paths, memoriesDirPath } from "../paths.ts";
6
+ import { cjkIndexText } from "../retrieve/cjk.ts";
7
+ import {
8
+ parse,
9
+ serialize,
10
+ ulid,
11
+ msToRfc3339,
12
+ type Frontmatter,
13
+ type MemoryFile,
14
+ type MemoryStatus,
15
+ } from "./markdown.ts";
16
+
17
+ /**
18
+ * Dedup + lifecycle (V2-DESIGN §3.3, §3.4).
19
+ *
20
+ * - Dedup on write: exact content hash → idempotent add; fuzzy token
21
+ * overlap (Jaccard ≥ 0.8, CJK-bigram aware) → near-duplicate notice.
22
+ * - Lifecycle: active → superseded | deprecated | retracted | archived.
23
+ * Superseding never overwrites: the old memory keeps its history with a
24
+ * bidirectional supersedes/superseded_by chain. Retrieval (§3.3) returns
25
+ * only the newest of a chain and excludes retracted/archived.
26
+ */
27
+
28
+ /** sha256 of whitespace-normalized body. Stored in the index (derived). */
29
+ export function contentHash(body: string): string {
30
+ const norm = body.replace(/\s+/g, " ").trim();
31
+ return createHash("sha256").update(norm, "utf8").digest("hex");
32
+ }
33
+
34
+ function tokenSet(text: string): Set<string> {
35
+ const latin = text.toLowerCase().match(/[a-z0-9_\-]+/g) ?? [];
36
+ const cjk = cjkIndexText(text).split(" ").filter(Boolean);
37
+ return new Set([...latin, ...cjk]);
38
+ }
39
+
40
+ /** Jaccard similarity over latin tokens + CJK bigrams. 1 = identical. */
41
+ export function similarity(a: string, b: string): number {
42
+ const sa = tokenSet(a);
43
+ const sb = tokenSet(b);
44
+ if (sa.size === 0 || sb.size === 0) return 0;
45
+ let inter = 0;
46
+ for (const t of sa) if (sb.has(t)) inter++;
47
+ const union = sa.size + sb.size - inter;
48
+ return union === 0 ? 0 : inter / union;
49
+ }
50
+
51
+ export interface DuplicateCandidate {
52
+ id: string;
53
+ score: number;
54
+ snippet: string;
55
+ }
56
+
57
+ export interface DupResult {
58
+ exact?: DuplicateCandidate;
59
+ near: DuplicateCandidate[];
60
+ }
61
+
62
+ export const NEAR_DUP_THRESHOLD = 0.8;
63
+
64
+ interface IndexRow {
65
+ id: string;
66
+ content_hash: string;
67
+ content: string;
68
+ }
69
+
70
+ /** Find exact + near duplicates of `body` among ACTIVE memories in scope. */
71
+ export function findDuplicates(
72
+ scopeKey: string,
73
+ body: string,
74
+ excludeId?: string,
75
+ ): DupResult {
76
+ const hash = contentHash(body);
77
+ const rows = db()
78
+ .prepare(
79
+ `SELECT id, content_hash, content FROM memories
80
+ WHERE scope_key = ? AND status = 'active'`,
81
+ )
82
+ .all(scopeKey) as IndexRow[];
83
+
84
+ let exact: DuplicateCandidate | undefined;
85
+ const near: DuplicateCandidate[] = [];
86
+ for (const r of rows) {
87
+ if (r.id === excludeId) continue;
88
+ const snippet = r.content.replace(/\s+/g, " ").trim().slice(0, 120);
89
+ if (r.content_hash && r.content_hash === hash) {
90
+ exact = { id: r.id, score: 1, snippet };
91
+ continue;
92
+ }
93
+ const score = similarity(body, r.content);
94
+ if (score >= NEAR_DUP_THRESHOLD) near.push({ id: r.id, score, snippet });
95
+ }
96
+ near.sort((x, y) => y.score - x.score);
97
+ return { exact, near: near.slice(0, 3) };
98
+ }
99
+
100
+ /** Locate a memory file by id: index first, then a full dir scan fallback. */
101
+ export function findMemoryFile(id: string): MemoryFile | null {
102
+ const { memories } = paths();
103
+ let filePath: string | null = null;
104
+ try {
105
+ const row = db()
106
+ .prepare(`SELECT scope_key FROM memories WHERE id = ?`)
107
+ .get(id) as { scope_key: string } | undefined;
108
+ if (row) {
109
+ const p = path.join(memoriesDirPath(row.scope_key), `${id}.md`);
110
+ if (fs.existsSync(p)) filePath = p;
111
+ }
112
+ } catch {
113
+ // index unavailable — fall through to scan
114
+ }
115
+ if (!filePath && fs.existsSync(memories)) {
116
+ for (const entry of fs.readdirSync(memories, { withFileTypes: true })) {
117
+ if (!entry.isDirectory()) continue;
118
+ const p = path.join(memories, entry.name, `${id}.md`);
119
+ if (fs.existsSync(p)) {
120
+ filePath = p;
121
+ break;
122
+ }
123
+ }
124
+ }
125
+ if (!filePath) return null;
126
+ const raw = fs.readFileSync(filePath, "utf8");
127
+ const parsed = parse(raw);
128
+ if (!parsed) return null;
129
+ const st = fs.statSync(filePath);
130
+ return { fm: parsed.fm, body: parsed.body, filePath, mtimeMs: st.mtimeMs };
131
+ }
132
+
133
+ export function rewriteMemoryFile(mf: MemoryFile): void {
134
+ fs.writeFileSync(mf.filePath, serialize(mf.fm, mf.body), "utf8");
135
+ }
136
+
137
+ export interface ChainRepairResult {
138
+ warnings: string[];
139
+ /** Linked memories whose files were rewritten; caller should re-index them. */
140
+ repaired: MemoryFile[];
141
+ }
142
+
143
+ /**
144
+ * Chain integrity (§3.3): supersedes/superseded_by must be pairwise
145
+ * consistent. A missing side is auto-completed and reported — a broken chain
146
+ * must never silently degrade retrieval. Dangling pointers (target file
147
+ * gone) can only be reported.
148
+ */
149
+ export function repairChain(mf: MemoryFile): ChainRepairResult {
150
+ const warnings: string[] = [];
151
+ const repaired: MemoryFile[] = [];
152
+ const fm = mf.fm;
153
+
154
+ if (fm.supersedes) {
155
+ const prev = findMemoryFile(fm.supersedes);
156
+ if (!prev) {
157
+ warnings.push(
158
+ `${fm.id}: supersedes target ${fm.supersedes} not found (dangling)`,
159
+ );
160
+ } else if (prev.fm.superseded_by !== fm.id) {
161
+ prev.fm.superseded_by = fm.id;
162
+ if (prev.fm.status === "active") prev.fm.status = "superseded";
163
+ prev.fm.updated_at = msToRfc3339(Date.now());
164
+ rewriteMemoryFile(prev);
165
+ repaired.push(prev);
166
+ warnings.push(
167
+ `${fm.id}: auto-completed ${prev.fm.id}.superseded_by → ${fm.id}`,
168
+ );
169
+ }
170
+ }
171
+
172
+ if (fm.superseded_by) {
173
+ const next = findMemoryFile(fm.superseded_by);
174
+ if (!next) {
175
+ warnings.push(
176
+ `${fm.id}: superseded_by target ${fm.superseded_by} not found (dangling)`,
177
+ );
178
+ } else if (next.fm.supersedes !== fm.id) {
179
+ next.fm.supersedes = fm.id;
180
+ next.fm.updated_at = msToRfc3339(Date.now());
181
+ rewriteMemoryFile(next);
182
+ repaired.push(next);
183
+ warnings.push(
184
+ `${fm.id}: auto-completed ${next.fm.id}.supersedes → ${fm.id}`,
185
+ );
186
+ }
187
+ }
188
+
189
+ return { warnings, repaired };
190
+ }
191
+
192
+ export interface SupersedeInput {
193
+ body: string;
194
+ type?: string;
195
+ tags?: string[];
196
+ source?: string;
197
+ }
198
+
199
+ /**
200
+ * Replace an active memory with a new one. The old memory is NOT overwritten:
201
+ * it becomes `status: superseded` with a forward pointer; the new memory
202
+ * points back. History preserved; retrieval returns the newest (§3.4).
203
+ */
204
+ export function supersede(
205
+ oldId: string,
206
+ input: SupersedeInput,
207
+ ): { oldMf: MemoryFile; newMf: MemoryFile } {
208
+ const oldMf = findMemoryFile(oldId);
209
+ if (!oldMf) throw new Error(`no memory with id ${oldId}`);
210
+ if (oldMf.fm.status !== "active") {
211
+ throw new Error(
212
+ `cannot supersede memory with status '${oldMf.fm.status}' (id ${oldId}); only active memories can be superseded`,
213
+ );
214
+ }
215
+
216
+ const now = msToRfc3339(Date.now());
217
+ const newFm: Frontmatter = {
218
+ ...oldMf.fm,
219
+ id: ulid(),
220
+ type: input.type ?? oldMf.fm.type,
221
+ tags: input.tags ?? oldMf.fm.tags,
222
+ source: input.source ?? oldMf.fm.source,
223
+ status: "active",
224
+ created_at: now,
225
+ updated_at: now,
226
+ supersedes: oldId,
227
+ superseded_by: null,
228
+ };
229
+ const dir = path.dirname(oldMf.filePath);
230
+ const newPath = path.join(dir, `${newFm.id}.md`);
231
+ fs.writeFileSync(newPath, serialize(newFm, input.body), "utf8");
232
+ const st = fs.statSync(newPath);
233
+ const newMf: MemoryFile = {
234
+ fm: newFm,
235
+ body: input.body,
236
+ filePath: newPath,
237
+ mtimeMs: st.mtimeMs,
238
+ };
239
+
240
+ oldMf.fm.status = "superseded";
241
+ oldMf.fm.superseded_by = newFm.id;
242
+ oldMf.fm.updated_at = now;
243
+ rewriteMemoryFile(oldMf);
244
+
245
+ return { oldMf, newMf };
246
+ }
247
+
248
+ const SETTABLE_STATUSES: ReadonlyArray<MemoryStatus> = [
249
+ "active",
250
+ "deprecated",
251
+ "retracted",
252
+ "archived",
253
+ ];
254
+
255
+ export function isSettableStatus(s: string): s is MemoryStatus {
256
+ return (SETTABLE_STATUSES as ReadonlyArray<string>).includes(s);
257
+ }
258
+
259
+ /**
260
+ * Direct lifecycle transition. `superseded` is NOT settable here — it is
261
+ * managed exclusively by `supersede()` so the chain stays consistent.
262
+ */
263
+ export function setStatus(id: string, status: MemoryStatus): MemoryFile {
264
+ if (!isSettableStatus(status)) {
265
+ throw new Error(
266
+ `invalid status '${status}'; use one of: ${SETTABLE_STATUSES.join(", ")}`,
267
+ );
268
+ }
269
+ const mf = findMemoryFile(id);
270
+ if (!mf) throw new Error(`no memory with id ${id}`);
271
+ if (mf.fm.status === "superseded") {
272
+ throw new Error(
273
+ `memory ${id} is superseded (chain-managed); supersede it again instead of changing status directly`,
274
+ );
275
+ }
276
+ mf.fm.status = status;
277
+ mf.fm.updated_at = msToRfc3339(Date.now());
278
+ rewriteMemoryFile(mf);
279
+ return mf;
280
+ }
@@ -4,16 +4,54 @@ import { randomBytes } from "node:crypto";
4
4
  import yaml from "js-yaml";
5
5
  import { memoriesDirFor, memoriesDirPath } from "../paths.ts";
6
6
 
7
+ /** v2 content-kind taxonomy (V2-DESIGN §3.1). `type` = what the memory IS. */
8
+ export const MEMORY_TYPE_TAXONOMY = [
9
+ "preference",
10
+ "fact",
11
+ "decision",
12
+ "lesson",
13
+ "warning",
14
+ "workflow",
15
+ "architecture",
16
+ "constraint",
17
+ "todo",
18
+ "knowledge",
19
+ "observation",
20
+ ] as const;
21
+
22
+ const TAXONOMY = new Set<string>(MEMORY_TYPE_TAXONOMY);
23
+
24
+ export type ScopeKind = "personal" | "project" | "org";
25
+ export type Visibility = "private" | "internal" | "shared";
26
+ export type MemoryRole = "knowledge" | "instruction";
27
+ export type Importance = "low" | "normal" | "high";
28
+ export type MemoryStatus =
29
+ | "active"
30
+ | "superseded"
31
+ | "deprecated"
32
+ | "retracted"
33
+ | "archived";
34
+
35
+ /** v2 frontmatter (V2-DESIGN §3). Markdown is the source of truth; the
36
+ * SQLite index is derived and rebuildable (D1). `scope_key` is kept as the
37
+ * local storage address; `scope` is the semantic ownership. */
7
38
  export interface Frontmatter {
8
39
  id: string;
40
+ schema_version: 2;
9
41
  scope_key: string;
10
- scope_kind: "user" | "project";
42
+ scope: ScopeKind;
43
+ visibility: Visibility;
11
44
  project_name: string;
12
45
  type: string;
46
+ role: MemoryRole;
47
+ importance: Importance;
48
+ status: MemoryStatus;
13
49
  tags: string[];
14
50
  source: string;
15
- created_at: number;
16
- updated_at: number;
51
+ created_at: string; // RFC 3339, never bare epoch (§3)
52
+ updated_at: string; // RFC 3339
53
+ supersedes: string | null; // on the NEW memory → points BACK (§3.3)
54
+ superseded_by: string | null; // on the OLD memory → points FORWARD
17
55
  }
18
56
 
19
57
  export interface MemoryFile {
@@ -39,28 +77,142 @@ export function ulid(): string {
39
77
  return ts + rand;
40
78
  }
41
79
 
80
+ /** epoch ms → RFC 3339 (v2 times). */
81
+ export function msToRfc3339(ms: number): string {
82
+ return new Date(ms).toISOString().replace(/\.000Z$/, "Z");
83
+ }
84
+
85
+ /** RFC 3339 (or epoch ms) → epoch ms. Falls back to now on garbage. */
86
+ export function timeToMs(v: unknown): number {
87
+ if (typeof v === "number" && Number.isFinite(v)) return Math.round(v);
88
+ if (typeof v === "string") {
89
+ const t = Date.parse(v);
90
+ if (Number.isFinite(t)) return t;
91
+ }
92
+ return Date.now();
93
+ }
94
+
95
+ function asScopeKind(v: unknown, scopeKind: unknown): ScopeKind {
96
+ if (v === "personal" || v === "project" || v === "org") return v;
97
+ if (scopeKind === "user") return "personal"; // v1 → v2 (§19)
98
+ if (scopeKind === "project") return "project";
99
+ return "personal";
100
+ }
101
+
102
+ function asVisibility(v: unknown, scope: ScopeKind): Visibility {
103
+ if (v === "private" || v === "internal" || v === "shared") return v;
104
+ // v2 defaults (§4): personal stays local, project is internal.
105
+ return scope === "personal" ? "private" : "internal";
106
+ }
107
+
108
+ function asRole(v: unknown): MemoryRole {
109
+ if (v === "knowledge" || v === "instruction") return v;
110
+ return "knowledge";
111
+ }
112
+
113
+ function asImportance(v: unknown, priority: unknown): Importance {
114
+ if (v === "low" || v === "normal" || v === "high") return v;
115
+ // v1 `priority: N` → importance (§19): 1–3 low, 8–10 high, else normal.
116
+ if (typeof priority === "number" && Number.isFinite(priority)) {
117
+ if (priority <= 3) return "low";
118
+ if (priority >= 8) return "high";
119
+ }
120
+ return "normal";
121
+ }
122
+
123
+ function asStatus(v: unknown): MemoryStatus {
124
+ if (
125
+ v === "active" ||
126
+ v === "superseded" ||
127
+ v === "deprecated" ||
128
+ v === "retracted" ||
129
+ v === "archived"
130
+ )
131
+ return v;
132
+ return "active";
133
+ }
134
+
135
+ /**
136
+ * Normalize raw (possibly v1) frontmatter into v2 shape. Lenient on read:
137
+ * v1 files (epoch times, scope_kind, priority, type: instruction) are mapped
138
+ * per §19 so old files keep working even before `migrate --to-v2` rewrites
139
+ * them. Use `planConversion` in v2migrate.ts for the explicit, reporting
140
+ * migration path.
141
+ */
142
+ export function normalizeFrontmatter(
143
+ raw: Record<string, unknown>,
144
+ ): Frontmatter {
145
+ const scope = asScopeKind(raw.scope, raw.scope_kind);
146
+ let scopeKey =
147
+ typeof raw.scope_key === "string" && raw.scope_key ? raw.scope_key : scope;
148
+ if (scopeKey === "user") scopeKey = "personal"; // v1 storage dir → v2
149
+
150
+ let type = typeof raw.type === "string" && raw.type ? raw.type : "fact";
151
+ let role = asRole(raw.role);
152
+ if (type === "instruction") {
153
+ // §19: v1 `type: instruction` → content-kind + role split (D11).
154
+ type = "knowledge";
155
+ role = "instruction";
156
+ }
157
+
158
+ return {
159
+ id: String(raw.id),
160
+ schema_version: 2,
161
+ scope_key: scopeKey,
162
+ scope,
163
+ visibility: asVisibility(raw.visibility, scope),
164
+ project_name:
165
+ typeof raw.project_name === "string" ? raw.project_name : scopeKey,
166
+ type,
167
+ role,
168
+ importance: asImportance(raw.importance, raw.priority),
169
+ status: asStatus(raw.status),
170
+ tags: Array.isArray(raw.tags)
171
+ ? raw.tags.filter((t): t is string => typeof t === "string")
172
+ : [],
173
+ source: typeof raw.source === "string" ? raw.source : "",
174
+ created_at: msToRfc3339(timeToMs(raw.created_at)),
175
+ updated_at: msToRfc3339(timeToMs(raw.updated_at)),
176
+ supersedes:
177
+ typeof raw.supersedes === "string" && raw.supersedes ? raw.supersedes : null,
178
+ superseded_by:
179
+ typeof raw.superseded_by === "string" && raw.superseded_by
180
+ ? raw.superseded_by
181
+ : null,
182
+ };
183
+ }
184
+
185
+ export function isTaxonomyType(t: string): boolean {
186
+ return TAXONOMY.has(t);
187
+ }
188
+
42
189
  export function serialize(fm: Frontmatter, body: string): string {
43
190
  const yml = yaml.dump(fm, { lineWidth: -1, quotingType: '"' });
44
191
  return `---\n${yml}---\n\n${body.trimEnd()}\n`;
45
192
  }
46
193
 
47
- export function parse(raw: string): { fm: Frontmatter; body: string } | null {
194
+ /** Raw frontmatter parse (no normalization) — for migration tooling. */
195
+ export function parseRawFrontmatter(
196
+ raw: string,
197
+ ): { rawFm: Record<string, unknown>; body: string } | null {
48
198
  const m = raw.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/);
49
199
  if (!m) return null;
50
200
  try {
51
- const fm = yaml.load(m[1]!) as Frontmatter;
52
- if (!fm || typeof fm !== "object" || !fm.id) return null;
53
- // Normalize
54
- fm.tags = Array.isArray(fm.tags) ? fm.tags : [];
55
- fm.type = fm.type ?? "note";
56
- fm.source = fm.source ?? "";
201
+ const rawFm = yaml.load(m[1]!) as Record<string, unknown>;
202
+ if (!rawFm || typeof rawFm !== "object" || !rawFm.id) return null;
57
203
  const body = (m[2] ?? "").replace(/^\n+/, "");
58
- return { fm, body };
204
+ return { rawFm, body };
59
205
  } catch {
60
206
  return null;
61
207
  }
62
208
  }
63
209
 
210
+ export function parse(raw: string): { fm: Frontmatter; body: string } | null {
211
+ const parsed = parseRawFrontmatter(raw);
212
+ if (!parsed) return null;
213
+ return { fm: normalizeFrontmatter(parsed.rawFm), body: parsed.body };
214
+ }
215
+
64
216
  export function writeMemoryFile(
65
217
  fm: Frontmatter,
66
218
  body: string,