open-memex 0.1.0 → 0.2.0-alpha
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +31 -5
- package/CONTRIBUTING.md +31 -0
- package/README.md +17 -7
- package/docs/SCOPES.md +81 -0
- package/docs/V2-DESIGN.md +484 -0
- package/package.json +1 -1
- package/scripts/smoke-pure.ts +209 -9
- package/src/cli.ts +138 -22
- package/src/config.ts +3 -10
- package/src/index.ts +18 -6
- package/src/redact.ts +151 -4
- package/src/retrieve/cjk.ts +63 -0
- package/src/retrieve/inject.ts +2 -2
- package/src/retrieve/search.ts +115 -28
- package/src/scope.ts +7 -2
- package/src/store/db.ts +62 -14
- package/src/store/lifecycle.ts +280 -0
- package/src/store/markdown.ts +163 -11
- package/src/store/sync.ts +53 -9
- package/src/store/v2migrate.ts +190 -0
- package/src/tools/memory.ts +88 -27
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import fs from "node:fs";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import { db } from "./db.ts";
|
|
5
|
+
import { paths, memoriesDirPath } from "../paths.ts";
|
|
6
|
+
import { cjkIndexText } from "../retrieve/cjk.ts";
|
|
7
|
+
import {
|
|
8
|
+
parse,
|
|
9
|
+
serialize,
|
|
10
|
+
ulid,
|
|
11
|
+
msToRfc3339,
|
|
12
|
+
type Frontmatter,
|
|
13
|
+
type MemoryFile,
|
|
14
|
+
type MemoryStatus,
|
|
15
|
+
} from "./markdown.ts";
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Dedup + lifecycle (V2-DESIGN §3.3, §3.4).
|
|
19
|
+
*
|
|
20
|
+
* - Dedup on write: exact content hash → idempotent add; fuzzy token
|
|
21
|
+
* overlap (Jaccard ≥ 0.8, CJK-bigram aware) → near-duplicate notice.
|
|
22
|
+
* - Lifecycle: active → superseded | deprecated | retracted | archived.
|
|
23
|
+
* Superseding never overwrites: the old memory keeps its history with a
|
|
24
|
+
* bidirectional supersedes/superseded_by chain. Retrieval (§3.3) returns
|
|
25
|
+
* only the newest of a chain and excludes retracted/archived.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
/** sha256 of whitespace-normalized body. Stored in the index (derived). */
|
|
29
|
+
export function contentHash(body: string): string {
|
|
30
|
+
const norm = body.replace(/\s+/g, " ").trim();
|
|
31
|
+
return createHash("sha256").update(norm, "utf8").digest("hex");
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function tokenSet(text: string): Set<string> {
|
|
35
|
+
const latin = text.toLowerCase().match(/[a-z0-9_\-]+/g) ?? [];
|
|
36
|
+
const cjk = cjkIndexText(text).split(" ").filter(Boolean);
|
|
37
|
+
return new Set([...latin, ...cjk]);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Jaccard similarity over latin tokens + CJK bigrams. 1 = identical. */
|
|
41
|
+
export function similarity(a: string, b: string): number {
|
|
42
|
+
const sa = tokenSet(a);
|
|
43
|
+
const sb = tokenSet(b);
|
|
44
|
+
if (sa.size === 0 || sb.size === 0) return 0;
|
|
45
|
+
let inter = 0;
|
|
46
|
+
for (const t of sa) if (sb.has(t)) inter++;
|
|
47
|
+
const union = sa.size + sb.size - inter;
|
|
48
|
+
return union === 0 ? 0 : inter / union;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
export interface DuplicateCandidate {
|
|
52
|
+
id: string;
|
|
53
|
+
score: number;
|
|
54
|
+
snippet: string;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export interface DupResult {
|
|
58
|
+
exact?: DuplicateCandidate;
|
|
59
|
+
near: DuplicateCandidate[];
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export const NEAR_DUP_THRESHOLD = 0.8;
|
|
63
|
+
|
|
64
|
+
interface IndexRow {
|
|
65
|
+
id: string;
|
|
66
|
+
content_hash: string;
|
|
67
|
+
content: string;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Find exact + near duplicates of `body` among ACTIVE memories in scope. */
|
|
71
|
+
export function findDuplicates(
|
|
72
|
+
scopeKey: string,
|
|
73
|
+
body: string,
|
|
74
|
+
excludeId?: string,
|
|
75
|
+
): DupResult {
|
|
76
|
+
const hash = contentHash(body);
|
|
77
|
+
const rows = db()
|
|
78
|
+
.prepare(
|
|
79
|
+
`SELECT id, content_hash, content FROM memories
|
|
80
|
+
WHERE scope_key = ? AND status = 'active'`,
|
|
81
|
+
)
|
|
82
|
+
.all(scopeKey) as IndexRow[];
|
|
83
|
+
|
|
84
|
+
let exact: DuplicateCandidate | undefined;
|
|
85
|
+
const near: DuplicateCandidate[] = [];
|
|
86
|
+
for (const r of rows) {
|
|
87
|
+
if (r.id === excludeId) continue;
|
|
88
|
+
const snippet = r.content.replace(/\s+/g, " ").trim().slice(0, 120);
|
|
89
|
+
if (r.content_hash && r.content_hash === hash) {
|
|
90
|
+
exact = { id: r.id, score: 1, snippet };
|
|
91
|
+
continue;
|
|
92
|
+
}
|
|
93
|
+
const score = similarity(body, r.content);
|
|
94
|
+
if (score >= NEAR_DUP_THRESHOLD) near.push({ id: r.id, score, snippet });
|
|
95
|
+
}
|
|
96
|
+
near.sort((x, y) => y.score - x.score);
|
|
97
|
+
return { exact, near: near.slice(0, 3) };
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** Locate a memory file by id: index first, then a full dir scan fallback. */
|
|
101
|
+
export function findMemoryFile(id: string): MemoryFile | null {
|
|
102
|
+
const { memories } = paths();
|
|
103
|
+
let filePath: string | null = null;
|
|
104
|
+
try {
|
|
105
|
+
const row = db()
|
|
106
|
+
.prepare(`SELECT scope_key FROM memories WHERE id = ?`)
|
|
107
|
+
.get(id) as { scope_key: string } | undefined;
|
|
108
|
+
if (row) {
|
|
109
|
+
const p = path.join(memoriesDirPath(row.scope_key), `${id}.md`);
|
|
110
|
+
if (fs.existsSync(p)) filePath = p;
|
|
111
|
+
}
|
|
112
|
+
} catch {
|
|
113
|
+
// index unavailable — fall through to scan
|
|
114
|
+
}
|
|
115
|
+
if (!filePath && fs.existsSync(memories)) {
|
|
116
|
+
for (const entry of fs.readdirSync(memories, { withFileTypes: true })) {
|
|
117
|
+
if (!entry.isDirectory()) continue;
|
|
118
|
+
const p = path.join(memories, entry.name, `${id}.md`);
|
|
119
|
+
if (fs.existsSync(p)) {
|
|
120
|
+
filePath = p;
|
|
121
|
+
break;
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
if (!filePath) return null;
|
|
126
|
+
const raw = fs.readFileSync(filePath, "utf8");
|
|
127
|
+
const parsed = parse(raw);
|
|
128
|
+
if (!parsed) return null;
|
|
129
|
+
const st = fs.statSync(filePath);
|
|
130
|
+
return { fm: parsed.fm, body: parsed.body, filePath, mtimeMs: st.mtimeMs };
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
export function rewriteMemoryFile(mf: MemoryFile): void {
|
|
134
|
+
fs.writeFileSync(mf.filePath, serialize(mf.fm, mf.body), "utf8");
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export interface ChainRepairResult {
|
|
138
|
+
warnings: string[];
|
|
139
|
+
/** Linked memories whose files were rewritten; caller should re-index them. */
|
|
140
|
+
repaired: MemoryFile[];
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Chain integrity (§3.3): supersedes/superseded_by must be pairwise
|
|
145
|
+
* consistent. A missing side is auto-completed and reported — a broken chain
|
|
146
|
+
* must never silently degrade retrieval. Dangling pointers (target file
|
|
147
|
+
* gone) can only be reported.
|
|
148
|
+
*/
|
|
149
|
+
export function repairChain(mf: MemoryFile): ChainRepairResult {
|
|
150
|
+
const warnings: string[] = [];
|
|
151
|
+
const repaired: MemoryFile[] = [];
|
|
152
|
+
const fm = mf.fm;
|
|
153
|
+
|
|
154
|
+
if (fm.supersedes) {
|
|
155
|
+
const prev = findMemoryFile(fm.supersedes);
|
|
156
|
+
if (!prev) {
|
|
157
|
+
warnings.push(
|
|
158
|
+
`${fm.id}: supersedes target ${fm.supersedes} not found (dangling)`,
|
|
159
|
+
);
|
|
160
|
+
} else if (prev.fm.superseded_by !== fm.id) {
|
|
161
|
+
prev.fm.superseded_by = fm.id;
|
|
162
|
+
if (prev.fm.status === "active") prev.fm.status = "superseded";
|
|
163
|
+
prev.fm.updated_at = msToRfc3339(Date.now());
|
|
164
|
+
rewriteMemoryFile(prev);
|
|
165
|
+
repaired.push(prev);
|
|
166
|
+
warnings.push(
|
|
167
|
+
`${fm.id}: auto-completed ${prev.fm.id}.superseded_by → ${fm.id}`,
|
|
168
|
+
);
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
if (fm.superseded_by) {
|
|
173
|
+
const next = findMemoryFile(fm.superseded_by);
|
|
174
|
+
if (!next) {
|
|
175
|
+
warnings.push(
|
|
176
|
+
`${fm.id}: superseded_by target ${fm.superseded_by} not found (dangling)`,
|
|
177
|
+
);
|
|
178
|
+
} else if (next.fm.supersedes !== fm.id) {
|
|
179
|
+
next.fm.supersedes = fm.id;
|
|
180
|
+
next.fm.updated_at = msToRfc3339(Date.now());
|
|
181
|
+
rewriteMemoryFile(next);
|
|
182
|
+
repaired.push(next);
|
|
183
|
+
warnings.push(
|
|
184
|
+
`${fm.id}: auto-completed ${next.fm.id}.supersedes → ${fm.id}`,
|
|
185
|
+
);
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
return { warnings, repaired };
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
export interface SupersedeInput {
|
|
193
|
+
body: string;
|
|
194
|
+
type?: string;
|
|
195
|
+
tags?: string[];
|
|
196
|
+
source?: string;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Replace an active memory with a new one. The old memory is NOT overwritten:
|
|
201
|
+
* it becomes `status: superseded` with a forward pointer; the new memory
|
|
202
|
+
* points back. History preserved; retrieval returns the newest (§3.4).
|
|
203
|
+
*/
|
|
204
|
+
export function supersede(
|
|
205
|
+
oldId: string,
|
|
206
|
+
input: SupersedeInput,
|
|
207
|
+
): { oldMf: MemoryFile; newMf: MemoryFile } {
|
|
208
|
+
const oldMf = findMemoryFile(oldId);
|
|
209
|
+
if (!oldMf) throw new Error(`no memory with id ${oldId}`);
|
|
210
|
+
if (oldMf.fm.status !== "active") {
|
|
211
|
+
throw new Error(
|
|
212
|
+
`cannot supersede memory with status '${oldMf.fm.status}' (id ${oldId}); only active memories can be superseded`,
|
|
213
|
+
);
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
const now = msToRfc3339(Date.now());
|
|
217
|
+
const newFm: Frontmatter = {
|
|
218
|
+
...oldMf.fm,
|
|
219
|
+
id: ulid(),
|
|
220
|
+
type: input.type ?? oldMf.fm.type,
|
|
221
|
+
tags: input.tags ?? oldMf.fm.tags,
|
|
222
|
+
source: input.source ?? oldMf.fm.source,
|
|
223
|
+
status: "active",
|
|
224
|
+
created_at: now,
|
|
225
|
+
updated_at: now,
|
|
226
|
+
supersedes: oldId,
|
|
227
|
+
superseded_by: null,
|
|
228
|
+
};
|
|
229
|
+
const dir = path.dirname(oldMf.filePath);
|
|
230
|
+
const newPath = path.join(dir, `${newFm.id}.md`);
|
|
231
|
+
fs.writeFileSync(newPath, serialize(newFm, input.body), "utf8");
|
|
232
|
+
const st = fs.statSync(newPath);
|
|
233
|
+
const newMf: MemoryFile = {
|
|
234
|
+
fm: newFm,
|
|
235
|
+
body: input.body,
|
|
236
|
+
filePath: newPath,
|
|
237
|
+
mtimeMs: st.mtimeMs,
|
|
238
|
+
};
|
|
239
|
+
|
|
240
|
+
oldMf.fm.status = "superseded";
|
|
241
|
+
oldMf.fm.superseded_by = newFm.id;
|
|
242
|
+
oldMf.fm.updated_at = now;
|
|
243
|
+
rewriteMemoryFile(oldMf);
|
|
244
|
+
|
|
245
|
+
return { oldMf, newMf };
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
const SETTABLE_STATUSES: ReadonlyArray<MemoryStatus> = [
|
|
249
|
+
"active",
|
|
250
|
+
"deprecated",
|
|
251
|
+
"retracted",
|
|
252
|
+
"archived",
|
|
253
|
+
];
|
|
254
|
+
|
|
255
|
+
export function isSettableStatus(s: string): s is MemoryStatus {
|
|
256
|
+
return (SETTABLE_STATUSES as ReadonlyArray<string>).includes(s);
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Direct lifecycle transition. `superseded` is NOT settable here — it is
|
|
261
|
+
* managed exclusively by `supersede()` so the chain stays consistent.
|
|
262
|
+
*/
|
|
263
|
+
export function setStatus(id: string, status: MemoryStatus): MemoryFile {
|
|
264
|
+
if (!isSettableStatus(status)) {
|
|
265
|
+
throw new Error(
|
|
266
|
+
`invalid status '${status}'; use one of: ${SETTABLE_STATUSES.join(", ")}`,
|
|
267
|
+
);
|
|
268
|
+
}
|
|
269
|
+
const mf = findMemoryFile(id);
|
|
270
|
+
if (!mf) throw new Error(`no memory with id ${id}`);
|
|
271
|
+
if (mf.fm.status === "superseded") {
|
|
272
|
+
throw new Error(
|
|
273
|
+
`memory ${id} is superseded (chain-managed); supersede it again instead of changing status directly`,
|
|
274
|
+
);
|
|
275
|
+
}
|
|
276
|
+
mf.fm.status = status;
|
|
277
|
+
mf.fm.updated_at = msToRfc3339(Date.now());
|
|
278
|
+
rewriteMemoryFile(mf);
|
|
279
|
+
return mf;
|
|
280
|
+
}
|
package/src/store/markdown.ts
CHANGED
|
@@ -4,16 +4,54 @@ import { randomBytes } from "node:crypto";
|
|
|
4
4
|
import yaml from "js-yaml";
|
|
5
5
|
import { memoriesDirFor, memoriesDirPath } from "../paths.ts";
|
|
6
6
|
|
|
7
|
+
/** v2 content-kind taxonomy (V2-DESIGN §3.1). `type` = what the memory IS. */
|
|
8
|
+
export const MEMORY_TYPE_TAXONOMY = [
|
|
9
|
+
"preference",
|
|
10
|
+
"fact",
|
|
11
|
+
"decision",
|
|
12
|
+
"lesson",
|
|
13
|
+
"warning",
|
|
14
|
+
"workflow",
|
|
15
|
+
"architecture",
|
|
16
|
+
"constraint",
|
|
17
|
+
"todo",
|
|
18
|
+
"knowledge",
|
|
19
|
+
"observation",
|
|
20
|
+
] as const;
|
|
21
|
+
|
|
22
|
+
const TAXONOMY = new Set<string>(MEMORY_TYPE_TAXONOMY);
|
|
23
|
+
|
|
24
|
+
export type ScopeKind = "personal" | "project" | "org";
|
|
25
|
+
export type Visibility = "private" | "internal" | "shared";
|
|
26
|
+
export type MemoryRole = "knowledge" | "instruction";
|
|
27
|
+
export type Importance = "low" | "normal" | "high";
|
|
28
|
+
export type MemoryStatus =
|
|
29
|
+
| "active"
|
|
30
|
+
| "superseded"
|
|
31
|
+
| "deprecated"
|
|
32
|
+
| "retracted"
|
|
33
|
+
| "archived";
|
|
34
|
+
|
|
35
|
+
/** v2 frontmatter (V2-DESIGN §3). Markdown is the source of truth; the
|
|
36
|
+
* SQLite index is derived and rebuildable (D1). `scope_key` is kept as the
|
|
37
|
+
* local storage address; `scope` is the semantic ownership. */
|
|
7
38
|
export interface Frontmatter {
|
|
8
39
|
id: string;
|
|
40
|
+
schema_version: 2;
|
|
9
41
|
scope_key: string;
|
|
10
|
-
|
|
42
|
+
scope: ScopeKind;
|
|
43
|
+
visibility: Visibility;
|
|
11
44
|
project_name: string;
|
|
12
45
|
type: string;
|
|
46
|
+
role: MemoryRole;
|
|
47
|
+
importance: Importance;
|
|
48
|
+
status: MemoryStatus;
|
|
13
49
|
tags: string[];
|
|
14
50
|
source: string;
|
|
15
|
-
created_at:
|
|
16
|
-
updated_at:
|
|
51
|
+
created_at: string; // RFC 3339, never bare epoch (§3)
|
|
52
|
+
updated_at: string; // RFC 3339
|
|
53
|
+
supersedes: string | null; // on the NEW memory → points BACK (§3.3)
|
|
54
|
+
superseded_by: string | null; // on the OLD memory → points FORWARD
|
|
17
55
|
}
|
|
18
56
|
|
|
19
57
|
export interface MemoryFile {
|
|
@@ -39,28 +77,142 @@ export function ulid(): string {
|
|
|
39
77
|
return ts + rand;
|
|
40
78
|
}
|
|
41
79
|
|
|
80
|
+
/** epoch ms → RFC 3339 (v2 times). */
|
|
81
|
+
export function msToRfc3339(ms: number): string {
|
|
82
|
+
return new Date(ms).toISOString().replace(/\.000Z$/, "Z");
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** RFC 3339 (or epoch ms) → epoch ms. Falls back to now on garbage. */
|
|
86
|
+
export function timeToMs(v: unknown): number {
|
|
87
|
+
if (typeof v === "number" && Number.isFinite(v)) return Math.round(v);
|
|
88
|
+
if (typeof v === "string") {
|
|
89
|
+
const t = Date.parse(v);
|
|
90
|
+
if (Number.isFinite(t)) return t;
|
|
91
|
+
}
|
|
92
|
+
return Date.now();
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
function asScopeKind(v: unknown, scopeKind: unknown): ScopeKind {
|
|
96
|
+
if (v === "personal" || v === "project" || v === "org") return v;
|
|
97
|
+
if (scopeKind === "user") return "personal"; // v1 → v2 (§19)
|
|
98
|
+
if (scopeKind === "project") return "project";
|
|
99
|
+
return "personal";
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function asVisibility(v: unknown, scope: ScopeKind): Visibility {
|
|
103
|
+
if (v === "private" || v === "internal" || v === "shared") return v;
|
|
104
|
+
// v2 defaults (§4): personal stays local, project is internal.
|
|
105
|
+
return scope === "personal" ? "private" : "internal";
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function asRole(v: unknown): MemoryRole {
|
|
109
|
+
if (v === "knowledge" || v === "instruction") return v;
|
|
110
|
+
return "knowledge";
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function asImportance(v: unknown, priority: unknown): Importance {
|
|
114
|
+
if (v === "low" || v === "normal" || v === "high") return v;
|
|
115
|
+
// v1 `priority: N` → importance (§19): 1–3 low, 8–10 high, else normal.
|
|
116
|
+
if (typeof priority === "number" && Number.isFinite(priority)) {
|
|
117
|
+
if (priority <= 3) return "low";
|
|
118
|
+
if (priority >= 8) return "high";
|
|
119
|
+
}
|
|
120
|
+
return "normal";
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function asStatus(v: unknown): MemoryStatus {
|
|
124
|
+
if (
|
|
125
|
+
v === "active" ||
|
|
126
|
+
v === "superseded" ||
|
|
127
|
+
v === "deprecated" ||
|
|
128
|
+
v === "retracted" ||
|
|
129
|
+
v === "archived"
|
|
130
|
+
)
|
|
131
|
+
return v;
|
|
132
|
+
return "active";
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Normalize raw (possibly v1) frontmatter into v2 shape. Lenient on read:
|
|
137
|
+
* v1 files (epoch times, scope_kind, priority, type: instruction) are mapped
|
|
138
|
+
* per §19 so old files keep working even before `migrate --to-v2` rewrites
|
|
139
|
+
* them. Use `planConversion` in v2migrate.ts for the explicit, reporting
|
|
140
|
+
* migration path.
|
|
141
|
+
*/
|
|
142
|
+
export function normalizeFrontmatter(
|
|
143
|
+
raw: Record<string, unknown>,
|
|
144
|
+
): Frontmatter {
|
|
145
|
+
const scope = asScopeKind(raw.scope, raw.scope_kind);
|
|
146
|
+
let scopeKey =
|
|
147
|
+
typeof raw.scope_key === "string" && raw.scope_key ? raw.scope_key : scope;
|
|
148
|
+
if (scopeKey === "user") scopeKey = "personal"; // v1 storage dir → v2
|
|
149
|
+
|
|
150
|
+
let type = typeof raw.type === "string" && raw.type ? raw.type : "fact";
|
|
151
|
+
let role = asRole(raw.role);
|
|
152
|
+
if (type === "instruction") {
|
|
153
|
+
// §19: v1 `type: instruction` → content-kind + role split (D11).
|
|
154
|
+
type = "knowledge";
|
|
155
|
+
role = "instruction";
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
return {
|
|
159
|
+
id: String(raw.id),
|
|
160
|
+
schema_version: 2,
|
|
161
|
+
scope_key: scopeKey,
|
|
162
|
+
scope,
|
|
163
|
+
visibility: asVisibility(raw.visibility, scope),
|
|
164
|
+
project_name:
|
|
165
|
+
typeof raw.project_name === "string" ? raw.project_name : scopeKey,
|
|
166
|
+
type,
|
|
167
|
+
role,
|
|
168
|
+
importance: asImportance(raw.importance, raw.priority),
|
|
169
|
+
status: asStatus(raw.status),
|
|
170
|
+
tags: Array.isArray(raw.tags)
|
|
171
|
+
? raw.tags.filter((t): t is string => typeof t === "string")
|
|
172
|
+
: [],
|
|
173
|
+
source: typeof raw.source === "string" ? raw.source : "",
|
|
174
|
+
created_at: msToRfc3339(timeToMs(raw.created_at)),
|
|
175
|
+
updated_at: msToRfc3339(timeToMs(raw.updated_at)),
|
|
176
|
+
supersedes:
|
|
177
|
+
typeof raw.supersedes === "string" && raw.supersedes ? raw.supersedes : null,
|
|
178
|
+
superseded_by:
|
|
179
|
+
typeof raw.superseded_by === "string" && raw.superseded_by
|
|
180
|
+
? raw.superseded_by
|
|
181
|
+
: null,
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
export function isTaxonomyType(t: string): boolean {
|
|
186
|
+
return TAXONOMY.has(t);
|
|
187
|
+
}
|
|
188
|
+
|
|
42
189
|
export function serialize(fm: Frontmatter, body: string): string {
|
|
43
190
|
const yml = yaml.dump(fm, { lineWidth: -1, quotingType: '"' });
|
|
44
191
|
return `---\n${yml}---\n\n${body.trimEnd()}\n`;
|
|
45
192
|
}
|
|
46
193
|
|
|
47
|
-
|
|
194
|
+
/** Raw frontmatter parse (no normalization) — for migration tooling. */
|
|
195
|
+
export function parseRawFrontmatter(
|
|
196
|
+
raw: string,
|
|
197
|
+
): { rawFm: Record<string, unknown>; body: string } | null {
|
|
48
198
|
const m = raw.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/);
|
|
49
199
|
if (!m) return null;
|
|
50
200
|
try {
|
|
51
|
-
const
|
|
52
|
-
if (!
|
|
53
|
-
// Normalize
|
|
54
|
-
fm.tags = Array.isArray(fm.tags) ? fm.tags : [];
|
|
55
|
-
fm.type = fm.type ?? "note";
|
|
56
|
-
fm.source = fm.source ?? "";
|
|
201
|
+
const rawFm = yaml.load(m[1]!) as Record<string, unknown>;
|
|
202
|
+
if (!rawFm || typeof rawFm !== "object" || !rawFm.id) return null;
|
|
57
203
|
const body = (m[2] ?? "").replace(/^\n+/, "");
|
|
58
|
-
return {
|
|
204
|
+
return { rawFm, body };
|
|
59
205
|
} catch {
|
|
60
206
|
return null;
|
|
61
207
|
}
|
|
62
208
|
}
|
|
63
209
|
|
|
210
|
+
export function parse(raw: string): { fm: Frontmatter; body: string } | null {
|
|
211
|
+
const parsed = parseRawFrontmatter(raw);
|
|
212
|
+
if (!parsed) return null;
|
|
213
|
+
return { fm: normalizeFrontmatter(parsed.rawFm), body: parsed.body };
|
|
214
|
+
}
|
|
215
|
+
|
|
64
216
|
export function writeMemoryFile(
|
|
65
217
|
fm: Frontmatter,
|
|
66
218
|
body: string,
|
package/src/store/sync.ts
CHANGED
|
@@ -1,17 +1,31 @@
|
|
|
1
1
|
import fs from "node:fs";
|
|
2
2
|
import { db } from "./db.ts";
|
|
3
|
-
import {
|
|
3
|
+
import { cjkIndexText } from "../retrieve/cjk.ts";
|
|
4
|
+
import { contentHash, repairChain } from "./lifecycle.ts";
|
|
5
|
+
import {
|
|
6
|
+
iterMemoryFiles,
|
|
7
|
+
readMemoryFile,
|
|
8
|
+
timeToMs,
|
|
9
|
+
type MemoryFile,
|
|
10
|
+
} from "./markdown.ts";
|
|
4
11
|
|
|
5
12
|
const UPSERT_SQL = `
|
|
6
|
-
INSERT INTO memories (id, scope_key,
|
|
7
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
13
|
+
INSERT INTO memories (id, scope_key, scope, visibility, project_name, type, role, importance, status, tags, content, cjk, content_hash, superseded_by, source, file_path, mtime_ms, created_at, updated_at)
|
|
14
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
8
15
|
ON CONFLICT(id) DO UPDATE SET
|
|
9
16
|
scope_key = excluded.scope_key,
|
|
10
|
-
|
|
17
|
+
scope = excluded.scope,
|
|
18
|
+
visibility = excluded.visibility,
|
|
11
19
|
project_name = excluded.project_name,
|
|
12
20
|
type = excluded.type,
|
|
21
|
+
role = excluded.role,
|
|
22
|
+
importance = excluded.importance,
|
|
23
|
+
status = excluded.status,
|
|
13
24
|
tags = excluded.tags,
|
|
14
25
|
content = excluded.content,
|
|
26
|
+
cjk = excluded.cjk,
|
|
27
|
+
content_hash = excluded.content_hash,
|
|
28
|
+
superseded_by = excluded.superseded_by,
|
|
15
29
|
source = excluded.source,
|
|
16
30
|
file_path = excluded.file_path,
|
|
17
31
|
mtime_ms = excluded.mtime_ms,
|
|
@@ -19,20 +33,50 @@ const UPSERT_SQL = `
|
|
|
19
33
|
`;
|
|
20
34
|
|
|
21
35
|
export function upsertFromFile(mf: MemoryFile): void {
|
|
22
|
-
|
|
36
|
+
// Chain integrity (§3.3): every write path through Core validates the
|
|
37
|
+
// supersedes/superseded_by pair and auto-completes a missing side.
|
|
38
|
+
const { warnings, repaired } = repairChain(mf);
|
|
39
|
+
for (const w of warnings) console.warn(`[open-memex] chain: ${w}`);
|
|
40
|
+
const mtimeOf = (m: MemoryFile): number => {
|
|
41
|
+
try {
|
|
42
|
+
return fs.statSync(m.filePath).mtimeMs;
|
|
43
|
+
} catch {
|
|
44
|
+
return m.mtimeMs; // file vanished mid-repair; keep the old mtime
|
|
45
|
+
}
|
|
46
|
+
};
|
|
47
|
+
writeRow(
|
|
48
|
+
mf,
|
|
49
|
+
repaired.some((r) => r.filePath === mf.filePath) ? mtimeOf(mf) : mf.mtimeMs,
|
|
50
|
+
);
|
|
51
|
+
for (const r of repaired) {
|
|
52
|
+
if (r.filePath === mf.filePath) continue;
|
|
53
|
+
writeRow(r, mtimeOf(r));
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function writeRow(mf: MemoryFile, mtimeMs: number): void {
|
|
58
|
+
const { fm, body, filePath } = mf;
|
|
59
|
+
const tags = (fm.tags ?? []).join(",");
|
|
23
60
|
db().prepare(UPSERT_SQL).run(
|
|
24
61
|
fm.id,
|
|
25
62
|
fm.scope_key,
|
|
26
|
-
fm.
|
|
63
|
+
fm.scope,
|
|
64
|
+
fm.visibility,
|
|
27
65
|
fm.project_name,
|
|
28
66
|
fm.type,
|
|
29
|
-
|
|
67
|
+
fm.role,
|
|
68
|
+
fm.importance,
|
|
69
|
+
fm.status,
|
|
70
|
+
tags,
|
|
30
71
|
body,
|
|
72
|
+
cjkIndexText(body + "\n" + tags),
|
|
73
|
+
contentHash(body),
|
|
74
|
+
fm.superseded_by ?? null,
|
|
31
75
|
fm.source ?? "",
|
|
32
76
|
filePath,
|
|
33
77
|
mtimeMs,
|
|
34
|
-
fm.created_at,
|
|
35
|
-
fm.updated_at,
|
|
78
|
+
timeToMs(fm.created_at),
|
|
79
|
+
timeToMs(fm.updated_at),
|
|
36
80
|
);
|
|
37
81
|
}
|
|
38
82
|
|