open-memex 0.3.0-alpha → 0.3.0-alpha.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +12 -6
- package/README.md +16 -7
- package/README.zh-CN.md +15 -6
- package/dist/capture/keywords.js +43 -0
- package/dist/cli.js +531 -0
- package/dist/config.js +123 -0
- package/dist/doctor.js +151 -0
- package/dist/index.js +121 -0
- package/dist/init.js +323 -0
- package/dist/mcp.js +85 -0
- package/dist/paths.js +36 -0
- package/dist/redact.js +249 -0
- package/dist/retrieve/cjk.js +58 -0
- package/dist/retrieve/inject.js +25 -0
- package/dist/retrieve/search.js +147 -0
- package/dist/scope.js +70 -0
- package/dist/store/db.js +132 -0
- package/dist/store/lifecycle.js +214 -0
- package/dist/store/markdown.js +197 -0
- package/dist/store/migrate.js +109 -0
- package/dist/store/sync.js +88 -0
- package/dist/store/v2migrate.js +158 -0
- package/dist/tools/memory.js +40 -0
- package/dist/tools/ops.js +186 -0
- package/docs/V2-DESIGN.md +19 -0
- package/package.json +5 -3
- package/src/cli.ts +37 -21
- package/src/doctor.ts +16 -5
- package/src/init.ts +55 -6
- package/tsconfig.build.json +1 -0
- package/bin/open-memex.js +0 -28
package/dist/paths.js
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import path from "node:path";
|
|
2
|
+
import fs from "node:fs";
|
|
3
|
+
import os from "node:os";
|
|
4
|
+
function dataRoot() {
|
|
5
|
+
if (process.env.MY_O_MEMORY_HOME)
|
|
6
|
+
return path.resolve(process.env.MY_O_MEMORY_HOME);
|
|
7
|
+
if (process.platform === "win32") {
|
|
8
|
+
const appData = process.env.APPDATA ?? path.join(os.homedir(), "AppData", "Roaming");
|
|
9
|
+
return path.join(appData, "open-memex");
|
|
10
|
+
}
|
|
11
|
+
const xdg = process.env.XDG_DATA_HOME ?? path.join(os.homedir(), ".local", "share");
|
|
12
|
+
return path.join(xdg, "open-memex");
|
|
13
|
+
}
|
|
14
|
+
let _cached = null;
|
|
15
|
+
export function paths() {
|
|
16
|
+
if (_cached)
|
|
17
|
+
return _cached;
|
|
18
|
+
const root = dataRoot();
|
|
19
|
+
const memories = path.join(root, "memories");
|
|
20
|
+
const indexDb = path.join(root, "index.db");
|
|
21
|
+
fs.mkdirSync(memories, { recursive: true });
|
|
22
|
+
_cached = { root, memories, indexDb };
|
|
23
|
+
return _cached;
|
|
24
|
+
}
|
|
25
|
+
export function memoriesDirFor(scopeKey) {
|
|
26
|
+
const { memories } = paths();
|
|
27
|
+
const dir = path.join(memories, scopeKey);
|
|
28
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
29
|
+
return dir;
|
|
30
|
+
}
|
|
31
|
+
/** Same as `memoriesDirFor` but never creates the directory. For read-only
|
|
32
|
+
* callers (iteration, existence checks) that must not pollute storage. */
|
|
33
|
+
export function memoriesDirPath(scopeKey) {
|
|
34
|
+
const { memories } = paths();
|
|
35
|
+
return path.join(memories, scopeKey);
|
|
36
|
+
}
|
package/dist/redact.js
ADDED
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Redaction: secrets must never land in memory files or the index in
|
|
3
|
+
* readable form.
|
|
4
|
+
*
|
|
5
|
+
* Layers, in order:
|
|
6
|
+
* 1. <private>...</private> regions are stripped first (explicit opt-out —
|
|
7
|
+
* the author marked this span as sensitive, so it is replaced with
|
|
8
|
+
* [REDACTED] before any detection runs).
|
|
9
|
+
* 2. Built-in provider patterns (always on, reported by id).
|
|
10
|
+
* 3. User patterns from config redactPatterns (reported verbatim).
|
|
11
|
+
* 4. High-entropy assignment heuristic (catches secrets whose provider we
|
|
12
|
+
* don't have a pattern for, e.g. `deploy_key = "aB3d..."`).
|
|
13
|
+
*
|
|
14
|
+
* A detected secret does NOT refuse the write: the matched string is masked
|
|
15
|
+
* in place — first 4 characters kept, the rest replaced with 'x' — and the
|
|
16
|
+
* write proceeds with the masked content. hadSecret reports that masking
|
|
17
|
+
* happened so callers can surface a notice. A partially-masked memory still
|
|
18
|
+
* identifies which credential it referred to without storing the secret.
|
|
19
|
+
*/
|
|
20
|
+
/**
|
|
21
|
+
* Prefixes like `sk-` also occur inside ordinary English words ("task-…",
|
|
22
|
+
* "risk-…", "disk-…"), which caused confirmed false positives. Require the
|
|
23
|
+
* token prefix NOT to be preceded by a word/hyphen char, so a real key after
|
|
24
|
+
* "=", ":", space, or a quote still matches.
|
|
25
|
+
*/
|
|
26
|
+
const TOKEN_BOUNDARY = "(?<![A-Za-z0-9_-])";
|
|
27
|
+
export const BUILTIN_SECRET_PATTERNS = [
|
|
28
|
+
{ id: "openai-key", source: `${TOKEN_BOUNDARY}sk-[A-Za-z0-9_-]{20,}` },
|
|
29
|
+
{
|
|
30
|
+
id: "openai-admin-key",
|
|
31
|
+
source: `${TOKEN_BOUNDARY}sk-admin-[A-Za-z0-9_-]{20,}`,
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
id: "openai-session-key",
|
|
35
|
+
source: `${TOKEN_BOUNDARY}sm_[A-Za-z0-9_-]{20,}`,
|
|
36
|
+
},
|
|
37
|
+
{ id: "github-pat", source: `${TOKEN_BOUNDARY}ghp_[A-Za-z0-9]{30,}` },
|
|
38
|
+
{ id: "github-oauth-token", source: `${TOKEN_BOUNDARY}gho_[A-Za-z0-9]{30,}` },
|
|
39
|
+
{ id: "github-user-token", source: `${TOKEN_BOUNDARY}ghu_[A-Za-z0-9]{30,}` },
|
|
40
|
+
{
|
|
41
|
+
id: "github-refresh-token",
|
|
42
|
+
source: `${TOKEN_BOUNDARY}ghr_[A-Za-z0-9]{30,}`,
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
id: "github-fine-grained-pat",
|
|
46
|
+
source: `${TOKEN_BOUNDARY}github_pat_[A-Za-z0-9_]{40,}`,
|
|
47
|
+
},
|
|
48
|
+
{ id: "aws-access-key-id", source: `${TOKEN_BOUNDARY}AKIA[0-9A-Z]{16}` },
|
|
49
|
+
{
|
|
50
|
+
id: "aws-secret-access-key",
|
|
51
|
+
source: "(aws[_-]?secret[_-]?access[_-]?key)([\"']?\\s*[:=]\\s*[\"']?)([A-Za-z0-9/+=]{40})",
|
|
52
|
+
flags: "i",
|
|
53
|
+
valueGroup: 3,
|
|
54
|
+
},
|
|
55
|
+
{ id: "slack-token", source: `${TOKEN_BOUNDARY}xox[baprs]-[A-Za-z0-9-]{10,}` },
|
|
56
|
+
{ id: "google-api-key", source: `${TOKEN_BOUNDARY}AIza[0-9A-Za-z_-]{30,}` },
|
|
57
|
+
{ id: "npm-token", source: `${TOKEN_BOUNDARY}npm_[A-Za-z0-9]{30,}` },
|
|
58
|
+
{ id: "gitlab-pat", source: `${TOKEN_BOUNDARY}glpat-[A-Za-z0-9_-]{20,}` },
|
|
59
|
+
{
|
|
60
|
+
id: "stripe-restricted-key",
|
|
61
|
+
source: `${TOKEN_BOUNDARY}rk_(live|test)_[A-Za-z0-9]{20,}`,
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
id: "stripe-webhook-secret",
|
|
65
|
+
source: `${TOKEN_BOUNDARY}whsec_[A-Za-z0-9]{20,}`,
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
id: "private-key-block",
|
|
69
|
+
// Whole block: masking only the BEGIN header would leave the base64 body
|
|
70
|
+
// readable in the memory file. Non-greedy so two blocks mask separately.
|
|
71
|
+
source: "-----BEGIN [A-Z ]*PRIVATE KEY-----[\\s\\S]*?-----END [A-Z ]*PRIVATE KEY-----",
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
id: "private-key-truncated",
|
|
75
|
+
// No END marker: mask from the header to end of text. A header without a
|
|
76
|
+
// body is still key material; over-masking is the safe direction.
|
|
77
|
+
source: "-----BEGIN [A-Z ]*PRIVATE KEY-----[\\s\\S]*$",
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
id: "jwt",
|
|
81
|
+
source: `${TOKEN_BOUNDARY}eyJ[A-Za-z0-9_-]{10,}\\.[A-Za-z0-9_-]{10,}\\.[A-Za-z0-9_-]{10,}`,
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
id: "generic-secret-assignment",
|
|
85
|
+
source: "(api[_-]?key|secret|passwd|password|auth[_-]?token|access[_-]?token)([\"']?\\s*[:=]\\s*[\"']?)([A-Za-z0-9_\\-./+=]{16,})([\"']?)",
|
|
86
|
+
flags: "i",
|
|
87
|
+
valueGroup: 3,
|
|
88
|
+
},
|
|
89
|
+
];
|
|
90
|
+
function compile(p) {
|
|
91
|
+
try {
|
|
92
|
+
return new RegExp(p.source, p.flags ?? "");
|
|
93
|
+
}
|
|
94
|
+
catch {
|
|
95
|
+
return null; // ignore malformed builtin (should never happen)
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
/** Returns the matched builtin pattern id, or null. */
|
|
99
|
+
export function findBuiltinSecret(text) {
|
|
100
|
+
for (const p of BUILTIN_SECRET_PATTERNS) {
|
|
101
|
+
const re = compile(p);
|
|
102
|
+
if (re && re.test(text))
|
|
103
|
+
return p.id;
|
|
104
|
+
}
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
export function findSecret(text, patterns) {
|
|
108
|
+
for (const p of patterns) {
|
|
109
|
+
try {
|
|
110
|
+
const re = new RegExp(p);
|
|
111
|
+
if (re.test(text))
|
|
112
|
+
return p;
|
|
113
|
+
}
|
|
114
|
+
catch {
|
|
115
|
+
// ignore malformed regex
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
120
|
+
function shannonEntropy(s) {
|
|
121
|
+
const freq = new Map();
|
|
122
|
+
for (const c of s)
|
|
123
|
+
freq.set(c, (freq.get(c) ?? 0) + 1);
|
|
124
|
+
let h = 0;
|
|
125
|
+
for (const n of freq.values()) {
|
|
126
|
+
const p = n / s.length;
|
|
127
|
+
h -= p * Math.log2(p);
|
|
128
|
+
}
|
|
129
|
+
return h;
|
|
130
|
+
}
|
|
131
|
+
// name = value assignments with a long token-ish value.
|
|
132
|
+
const ASSIGNMENT_RE = /([A-Za-z_][A-Za-z0-9_]{1,63})\s*[:=]\s*["']?([A-Za-z0-9_\-./+=]{24,})["']?/g;
|
|
133
|
+
/**
|
|
134
|
+
* Heuristic last line of defense: a long, high-entropy value assigned to a
|
|
135
|
+
* name is almost certainly a credential, even when no provider pattern
|
|
136
|
+
* matches. Tuned conservatively (length >= 24, entropy >= 4.5 bits/char):
|
|
137
|
+
* hex digests (<= 4.0) and prose (~4.0) pass through; base64-ish randomness
|
|
138
|
+
* does not. URLs are skipped.
|
|
139
|
+
*/
|
|
140
|
+
export function findHighEntropySecret(text) {
|
|
141
|
+
for (const m of text.matchAll(ASSIGNMENT_RE)) {
|
|
142
|
+
const name = m[1];
|
|
143
|
+
const value = m[2];
|
|
144
|
+
if (!isSuspectAssignmentValue(value, m[0], m.index ?? 0, text))
|
|
145
|
+
continue;
|
|
146
|
+
return `high-entropy-secret:${name}`;
|
|
147
|
+
}
|
|
148
|
+
return null;
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* Shared benign-value rule for detection AND masking: a URL (e.g. the value
|
|
152
|
+
* after "https:") or a low-entropy value must never be touched. Note the
|
|
153
|
+
* "://" check: ASSIGNMENT_RE consumes the colon of "https:" as the separator,
|
|
154
|
+
* so the captured value starts with "//..." — check for "://" in the
|
|
155
|
+
* original text around the match, not just inside the value.
|
|
156
|
+
*/
|
|
157
|
+
function isSuspectAssignmentValue(value, fullMatch, offset, text) {
|
|
158
|
+
// Bare URL: ASSIGNMENT_RE consumed the scheme colon ("https:") as the
|
|
159
|
+
// separator, so the match itself starts with "scheme://".
|
|
160
|
+
if (/^[a-zA-Z][a-zA-Z0-9+.-]*:\/\//.test(fullMatch))
|
|
161
|
+
return false;
|
|
162
|
+
if (value.includes("://"))
|
|
163
|
+
return false;
|
|
164
|
+
// Reconstruct what preceded the value inside the match (name + separator);
|
|
165
|
+
// if the text right before the value looks like scheme://, it is a URL.
|
|
166
|
+
const before = text.slice(Math.max(0, offset - 12), offset);
|
|
167
|
+
if (/^[a-zA-Z][a-zA-Z0-9+.-]*:(\/\/)?$/.test(before.trim()) || before.includes("://"))
|
|
168
|
+
return false;
|
|
169
|
+
return shannonEntropy(value) >= 4.5;
|
|
170
|
+
}
|
|
171
|
+
export function stripPrivate(text) {
|
|
172
|
+
// Closed pairs first...
|
|
173
|
+
let out = text.replace(/<private>[\s\S]*?<\/private>/gi, "[REDACTED]");
|
|
174
|
+
// ...then an unclosed <private> redacts everything after it. A dangling
|
|
175
|
+
// tag almost always means the author intended the rest to be private.
|
|
176
|
+
out = out.replace(/<private>[\s\S]*$/gi, "[REDACTED]");
|
|
177
|
+
return out;
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Mask a matched secret: keep the first 4 characters, replace the rest
|
|
181
|
+
* with 'x' (length-preserving). "sk-1234567890abcdefghij" → "sk-1xxxxxxxxxxxxx".
|
|
182
|
+
*/
|
|
183
|
+
function maskMatch(m) {
|
|
184
|
+
return m.slice(0, 4) + "x".repeat(Math.max(0, m.length - 4));
|
|
185
|
+
}
|
|
186
|
+
/** Apply the builtin patterns as masking (global, all matches). */
|
|
187
|
+
function maskBuiltin(content) {
|
|
188
|
+
let out = content;
|
|
189
|
+
for (const p of BUILTIN_SECRET_PATTERNS) {
|
|
190
|
+
const re = compile(p);
|
|
191
|
+
if (!re)
|
|
192
|
+
continue;
|
|
193
|
+
const g = new RegExp(re.source, re.flags + "g");
|
|
194
|
+
if (p.valueGroup == null) {
|
|
195
|
+
out = out.replace(g, (m) => maskMatch(m));
|
|
196
|
+
}
|
|
197
|
+
else {
|
|
198
|
+
// Mask only the secret-value group; the credential name stays readable.
|
|
199
|
+
out = out.replace(g, (...args) => {
|
|
200
|
+
const m = args[0];
|
|
201
|
+
const val = args[p.valueGroup];
|
|
202
|
+
if (!val)
|
|
203
|
+
return m;
|
|
204
|
+
const idx = m.lastIndexOf(val);
|
|
205
|
+
if (idx < 0)
|
|
206
|
+
return m;
|
|
207
|
+
return m.slice(0, idx) + maskMatch(val) + m.slice(idx + val.length);
|
|
208
|
+
});
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
return out;
|
|
212
|
+
}
|
|
213
|
+
/** Apply user patterns as masking. Invalid regexes are ignored. */
|
|
214
|
+
function maskUser(content, patterns) {
|
|
215
|
+
let out = content;
|
|
216
|
+
for (const p of patterns) {
|
|
217
|
+
try {
|
|
218
|
+
out = out.replace(new RegExp(p, "g"), (m) => maskMatch(m));
|
|
219
|
+
}
|
|
220
|
+
catch {
|
|
221
|
+
// ignore malformed regex
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
return out;
|
|
225
|
+
}
|
|
226
|
+
/** Mask the value side of high-entropy assignments (name stays readable).
|
|
227
|
+
* Uses the exact same benign-value rule as detection, so URLs and
|
|
228
|
+
* low-entropy values are never touched even when another secret triggers. */
|
|
229
|
+
function maskEntropy(content) {
|
|
230
|
+
return content.replace(ASSIGNMENT_RE, (m, name, value, offset) => {
|
|
231
|
+
if (!isSuspectAssignmentValue(value, m, offset, content))
|
|
232
|
+
return m;
|
|
233
|
+
return m.split(value).join(maskMatch(value));
|
|
234
|
+
});
|
|
235
|
+
}
|
|
236
|
+
export function redact(text, patterns) {
|
|
237
|
+
const stripped = stripPrivate(text);
|
|
238
|
+
// Detection (for reporting) runs in priority order: builtin → user → entropy.
|
|
239
|
+
const builtin = findBuiltinSecret(stripped);
|
|
240
|
+
const user = builtin ? null : findSecret(stripped, patterns);
|
|
241
|
+
const entropic = builtin || user ? null : findHighEntropySecret(stripped);
|
|
242
|
+
const matched = builtin ?? user ?? entropic;
|
|
243
|
+
if (!matched)
|
|
244
|
+
return { content: stripped, hadSecret: false, matchedPattern: null };
|
|
245
|
+
// Masking runs every family over the text (not just the reported one) so
|
|
246
|
+
// multiple credentials in one memory are all masked.
|
|
247
|
+
const masked = maskEntropy(maskUser(maskBuiltin(stripped), patterns));
|
|
248
|
+
return { content: masked, hadSecret: true, matchedPattern: matched };
|
|
249
|
+
}
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
// CJK retrieval helpers (pure logic, no sqlite).
|
|
2
|
+
//
|
|
3
|
+
// FTS5's `porter unicode61` tokenizer treats a run of CJK ideographs as one
|
|
4
|
+
// token ("中文记忆" is a single token), so substring queries never match, and
|
|
5
|
+
// the old query builder dropped CJK characters entirely. Per D8 the default
|
|
6
|
+
// CJK strategy is bigram: at write time we pre-tokenize CJK runs into
|
|
7
|
+
// space-separated unigrams + overlapping bigrams stored in the `cjk` FTS
|
|
8
|
+
// column; at query time the CJK part of the query becomes an OR of bigrams
|
|
9
|
+
// against that column. Unigrams are included so single-character queries
|
|
10
|
+
// ("猫") still match. Works identically on bun:sqlite and better-sqlite3
|
|
11
|
+
// with no native tokenizer dependency.
|
|
12
|
+
const CJK_RE = /[\u3400-\u4DBF\u4E00-\u9FFF\u3040-\u309F\u30A0-\u30FF\uAC00-\uD7AF\u1100-\u11FF\u{20000}-\u{2A6DF}]/u;
|
|
13
|
+
const CJK_RUN_RE = /[\u3400-\u4DBF\u4E00-\u9FFF\u3040-\u309F\u30A0-\u30FF\uAC00-\uD7AF\u1100-\u11FF\u{20000}-\u{2A6DF}]+/gu;
|
|
14
|
+
/** True if the string contains any CJK (Han/Hiragana/Katakana/Hangul) character. */
|
|
15
|
+
export function hasCjk(s) {
|
|
16
|
+
return CJK_RE.test(s);
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Build the index text for the `cjk` FTS column: for every CJK run emit each
|
|
20
|
+
* character (unigram) then every overlapping bigram, space-separated.
|
|
21
|
+
* "中文记忆" -> "中 文 记 忆 中文 文记 记忆". Non-CJK text yields "".
|
|
22
|
+
*/
|
|
23
|
+
export function cjkIndexText(text) {
|
|
24
|
+
const out = [];
|
|
25
|
+
for (const m of text.matchAll(CJK_RUN_RE)) {
|
|
26
|
+
const chars = [...m[0]];
|
|
27
|
+
for (const ch of chars)
|
|
28
|
+
out.push(ch);
|
|
29
|
+
for (let i = 0; i + 1 < chars.length; i++)
|
|
30
|
+
out.push(chars[i] + chars[i + 1]);
|
|
31
|
+
}
|
|
32
|
+
return out.join(" ");
|
|
33
|
+
}
|
|
34
|
+
/** Escape a raw token for embedding in a double-quoted FTS5 phrase. */
|
|
35
|
+
function esc(t) {
|
|
36
|
+
return t.replace(/"/g, '""');
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Convert the CJK runs of a query into an FTS5 expression for the `cjk`
|
|
40
|
+
* column. Multi-char runs become an OR of bigrams (recall-oriented; bm25
|
|
41
|
+
* ranks docs matching more bigrams higher). Single chars stay unigrams.
|
|
42
|
+
* Returns "" when the query has no CJK.
|
|
43
|
+
*/
|
|
44
|
+
export function cjkQueryExpr(query) {
|
|
45
|
+
const parts = [];
|
|
46
|
+
for (const m of query.matchAll(CJK_RUN_RE)) {
|
|
47
|
+
const chars = [...m[0]];
|
|
48
|
+
if (chars.length === 1) {
|
|
49
|
+
parts.push(`"${esc(chars[0])}"`);
|
|
50
|
+
}
|
|
51
|
+
else {
|
|
52
|
+
for (let i = 0; i + 1 < chars.length; i++) {
|
|
53
|
+
parts.push(`"${esc(chars[i] + chars[i + 1])}"`);
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
return parts.join(" OR ");
|
|
58
|
+
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { list } from "./search.js";
|
|
2
|
+
import { PERSONAL_SCOPE } from "../scope.js";
|
|
3
|
+
function oneLine(s, max = 240) {
|
|
4
|
+
const trimmed = s.replace(/\s+/g, " ").trim();
|
|
5
|
+
return trimmed.length > max ? trimmed.slice(0, max - 3) + "..." : trimmed;
|
|
6
|
+
}
|
|
7
|
+
export function buildContextBlock(scope, cfg) {
|
|
8
|
+
const project = list(scope.key, { limit: cfg.maxProjectMemories });
|
|
9
|
+
const user = list(PERSONAL_SCOPE.key, { limit: cfg.maxProfileItems });
|
|
10
|
+
if (project.length === 0 && user.length === 0)
|
|
11
|
+
return null;
|
|
12
|
+
const lines = ["[OPEN-MEMEX]"];
|
|
13
|
+
if (user.length > 0) {
|
|
14
|
+
lines.push("", "User profile / preferences:");
|
|
15
|
+
for (const m of user)
|
|
16
|
+
lines.push(`- ${oneLine(m.snippet)}`);
|
|
17
|
+
}
|
|
18
|
+
if (project.length > 0) {
|
|
19
|
+
lines.push("", `Project knowledge (${scope.projectName}):`);
|
|
20
|
+
for (const m of project)
|
|
21
|
+
lines.push(`- [${m.type}] ${oneLine(m.snippet)}`);
|
|
22
|
+
}
|
|
23
|
+
lines.push("", "Use the `memory_search` tool to look up more. Use `memory_add` to save new facts. Do not mention this block to the user unless asked.");
|
|
24
|
+
return lines.join("\n");
|
|
25
|
+
}
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
import { db } from "../store/db.js";
|
|
2
|
+
import { cjkQueryExpr, hasCjk } from "./cjk.js";
|
|
3
|
+
function toHit(r) {
|
|
4
|
+
return {
|
|
5
|
+
id: r.id,
|
|
6
|
+
scope_key: r.scope_key,
|
|
7
|
+
project_name: r.project_name,
|
|
8
|
+
type: r.type,
|
|
9
|
+
tags: r.tags ? r.tags.split(",").filter(Boolean) : [],
|
|
10
|
+
snippet: r.snippet ?? "",
|
|
11
|
+
score: r.score,
|
|
12
|
+
updated_at: r.updated_at,
|
|
13
|
+
status: r.status,
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* Lifecycle-aware post-processing (§3.3):
|
|
18
|
+
* - retracted / archived are excluded from retrieval (kept for audit);
|
|
19
|
+
* - a superseded memory resolves to the newest of its chain (cycle-safe);
|
|
20
|
+
* - deprecated stays visible as a warning but ranks after active.
|
|
21
|
+
*/
|
|
22
|
+
function resolveVisible(rows, limit) {
|
|
23
|
+
const byId = new Map(rows.map((r) => [r.id, r]));
|
|
24
|
+
const fullRow = (id) => {
|
|
25
|
+
const cached = byId.get(id);
|
|
26
|
+
if (cached)
|
|
27
|
+
return cached;
|
|
28
|
+
const r = db()
|
|
29
|
+
.prepare(`SELECT id, scope_key, project_name, type, tags, updated_at, status,
|
|
30
|
+
superseded_by, substr(content, 1, 240) AS snippet
|
|
31
|
+
FROM memories WHERE id = ?`)
|
|
32
|
+
.get(id);
|
|
33
|
+
if (!r)
|
|
34
|
+
return undefined;
|
|
35
|
+
const full = { ...r, score: 0 };
|
|
36
|
+
byId.set(id, full);
|
|
37
|
+
return full;
|
|
38
|
+
};
|
|
39
|
+
const seen = new Set();
|
|
40
|
+
const active = [];
|
|
41
|
+
const deprecated = [];
|
|
42
|
+
for (const r of rows) {
|
|
43
|
+
if (r.status === "retracted" || r.status === "archived")
|
|
44
|
+
continue;
|
|
45
|
+
let target = r;
|
|
46
|
+
if (r.status === "superseded") {
|
|
47
|
+
let cur = r;
|
|
48
|
+
const chain = new Set([r.id]);
|
|
49
|
+
while (cur.status === "superseded" && cur.superseded_by) {
|
|
50
|
+
if (chain.has(cur.superseded_by))
|
|
51
|
+
break; // cycle guard
|
|
52
|
+
chain.add(cur.superseded_by);
|
|
53
|
+
const nxt = fullRow(cur.superseded_by);
|
|
54
|
+
if (!nxt)
|
|
55
|
+
break;
|
|
56
|
+
cur = nxt;
|
|
57
|
+
}
|
|
58
|
+
if (cur.status === "retracted" || cur.status === "archived")
|
|
59
|
+
continue;
|
|
60
|
+
target = cur;
|
|
61
|
+
}
|
|
62
|
+
if (seen.has(target.id))
|
|
63
|
+
continue;
|
|
64
|
+
seen.add(target.id);
|
|
65
|
+
const hit = toHit(target);
|
|
66
|
+
if (target.status === "deprecated")
|
|
67
|
+
deprecated.push(hit);
|
|
68
|
+
else
|
|
69
|
+
active.push(hit);
|
|
70
|
+
}
|
|
71
|
+
return [...active, ...deprecated].slice(0, limit);
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Convert free-text query into a safe FTS5 MATCH expression.
|
|
75
|
+
* Latin tokens keep the old behavior (prefix match on content/tags/type).
|
|
76
|
+
* CJK runs become an OR of bigrams against the `cjk` column (see cjk.ts).
|
|
77
|
+
* Mixed queries OR the two parts together.
|
|
78
|
+
*/
|
|
79
|
+
function toFtsQuery(q) {
|
|
80
|
+
const latin = (q.toLowerCase().match(/[a-z0-9_.\-]+/g) ?? [])
|
|
81
|
+
.map((t) => `"${t.replace(/"/g, '""')}"*`)
|
|
82
|
+
.join(" OR ");
|
|
83
|
+
const cjk = hasCjk(q) ? cjkQueryExpr(q) : "";
|
|
84
|
+
if (latin && cjk)
|
|
85
|
+
return `(${latin}) OR {cjk}:(${cjk})`;
|
|
86
|
+
if (cjk)
|
|
87
|
+
return `{cjk}:(${cjk})`;
|
|
88
|
+
return latin;
|
|
89
|
+
}
|
|
90
|
+
export function search(query, opts = {}) {
|
|
91
|
+
const q = toFtsQuery(query);
|
|
92
|
+
if (!q)
|
|
93
|
+
return [];
|
|
94
|
+
const limit = Math.max(1, Math.min(opts.limit ?? 8, 50));
|
|
95
|
+
// Over-fetch: lifecycle filtering (chain resolution, exclusions) happens
|
|
96
|
+
// after the FTS query, so candidates must survive it.
|
|
97
|
+
const fetchLimit = Math.min(limit * 3 + 10, 150);
|
|
98
|
+
const scopeFilter = opts.scopeKeys && opts.scopeKeys.length > 0
|
|
99
|
+
? ` AND m.scope_key IN (${opts.scopeKeys.map(() => "?").join(",")})`
|
|
100
|
+
: "";
|
|
101
|
+
const typeFilter = opts.type ? ` AND m.type = ?` : "";
|
|
102
|
+
const sql = `
|
|
103
|
+
SELECT m.id, m.scope_key, m.project_name, m.type, m.tags, m.updated_at,
|
|
104
|
+
m.status, m.superseded_by,
|
|
105
|
+
snippet(memories_fts, 0, '[', ']', ' ... ', 12) AS snippet,
|
|
106
|
+
bm25(memories_fts) AS score
|
|
107
|
+
FROM memories_fts
|
|
108
|
+
JOIN memories m ON m.rowid = memories_fts.rowid
|
|
109
|
+
WHERE memories_fts MATCH ?${scopeFilter}${typeFilter}
|
|
110
|
+
ORDER BY score ASC
|
|
111
|
+
LIMIT ?
|
|
112
|
+
`;
|
|
113
|
+
const params = [q];
|
|
114
|
+
if (opts.scopeKeys && opts.scopeKeys.length > 0)
|
|
115
|
+
params.push(...opts.scopeKeys);
|
|
116
|
+
if (opts.type)
|
|
117
|
+
params.push(opts.type);
|
|
118
|
+
params.push(fetchLimit);
|
|
119
|
+
const rows = db()
|
|
120
|
+
.prepare(sql)
|
|
121
|
+
.all(...params);
|
|
122
|
+
// FTS5 bm25: lower = better; invert for intuition.
|
|
123
|
+
const raw = rows.map((r) => ({ ...r, score: -r.score }));
|
|
124
|
+
return resolveVisible(raw, limit);
|
|
125
|
+
}
|
|
126
|
+
export function list(scopeKey, opts = {}) {
|
|
127
|
+
const limit = Math.max(1, Math.min(opts.limit ?? 20, 100));
|
|
128
|
+
const fetchLimit = Math.min(limit * 3 + 10, 150);
|
|
129
|
+
const typeFilter = opts.type ? ` AND type = ?` : "";
|
|
130
|
+
const sql = `
|
|
131
|
+
SELECT id, scope_key, project_name, type, tags, updated_at, status,
|
|
132
|
+
superseded_by, substr(content, 1, 240) AS snippet
|
|
133
|
+
FROM memories
|
|
134
|
+
WHERE scope_key = ?${typeFilter}
|
|
135
|
+
ORDER BY updated_at DESC
|
|
136
|
+
LIMIT ?
|
|
137
|
+
`;
|
|
138
|
+
const params = [scopeKey];
|
|
139
|
+
if (opts.type)
|
|
140
|
+
params.push(opts.type);
|
|
141
|
+
params.push(fetchLimit);
|
|
142
|
+
const rows = db()
|
|
143
|
+
.prepare(sql)
|
|
144
|
+
.all(...params);
|
|
145
|
+
const raw = rows.map((r) => ({ ...r, score: 1 }));
|
|
146
|
+
return resolveVisible(raw, limit);
|
|
147
|
+
}
|
package/dist/scope.js
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { execSync } from "node:child_process";
|
|
2
|
+
import crypto from "node:crypto";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
/** v2: v1 `user` scope is renamed to `personal` (§19). */
|
|
5
|
+
export const PERSONAL_SCOPE = {
|
|
6
|
+
key: "personal",
|
|
7
|
+
kind: "personal",
|
|
8
|
+
projectName: "personal",
|
|
9
|
+
};
|
|
10
|
+
function normalizeRemote(url) {
|
|
11
|
+
return url
|
|
12
|
+
.trim()
|
|
13
|
+
.replace(/\.git$/i, "")
|
|
14
|
+
.replace(/^git@([^:]+):/, "https://$1/")
|
|
15
|
+
.replace(/^ssh:\/\/git@/, "https://")
|
|
16
|
+
.toLowerCase();
|
|
17
|
+
}
|
|
18
|
+
function tryGitRemote(cwd) {
|
|
19
|
+
try {
|
|
20
|
+
const url = execSync("git config --get remote.origin.url", {
|
|
21
|
+
cwd,
|
|
22
|
+
stdio: ["ignore", "pipe", "ignore"],
|
|
23
|
+
encoding: "utf8",
|
|
24
|
+
}).trim();
|
|
25
|
+
return url || null;
|
|
26
|
+
}
|
|
27
|
+
catch {
|
|
28
|
+
return null;
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
function sanitize(name) {
|
|
32
|
+
const cleaned = name.replace(/[^a-zA-Z0-9._-]/g, "-").slice(0, 40);
|
|
33
|
+
return cleaned.length > 0 ? cleaned : "project";
|
|
34
|
+
}
|
|
35
|
+
function scopeFromSeed(seed, projectName) {
|
|
36
|
+
const hash = crypto.createHash("sha256").update(seed).digest("hex").slice(0, 12);
|
|
37
|
+
return {
|
|
38
|
+
key: `project__${sanitize(projectName)}__${hash}`,
|
|
39
|
+
kind: "project",
|
|
40
|
+
projectName,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
/** Scope keyed purely on the absolute cwd. Never consults git. Used both as
|
|
44
|
+
* the fallback for `resolveProjectScope` when no remote is set, and as the
|
|
45
|
+
* "legacy" scope detector for migration when a remote is added later. */
|
|
46
|
+
export function resolveCwdScope(worktree) {
|
|
47
|
+
const target = worktree && worktree.length > 0 ? worktree : ".";
|
|
48
|
+
const resolved = path.resolve(target);
|
|
49
|
+
const seed = resolved.toLowerCase();
|
|
50
|
+
// basename can return "" on Windows drive roots (e.g. "C:\\"); walk the path
|
|
51
|
+
// segments and pick the last non-empty, non-drive-letter component.
|
|
52
|
+
const parts = resolved
|
|
53
|
+
.split(/[\\/]+/)
|
|
54
|
+
.filter((p) => p && !/^[A-Za-z]:$/.test(p));
|
|
55
|
+
let projectName = parts[parts.length - 1] ?? path.basename(resolved) ?? "workspace";
|
|
56
|
+
if (!projectName)
|
|
57
|
+
projectName = "workspace";
|
|
58
|
+
return scopeFromSeed(seed, projectName);
|
|
59
|
+
}
|
|
60
|
+
export function resolveProjectScope(worktree) {
|
|
61
|
+
const target = worktree && worktree.length > 0 ? worktree : ".";
|
|
62
|
+
const remote = tryGitRemote(target);
|
|
63
|
+
if (remote) {
|
|
64
|
+
const norm = normalizeRemote(remote);
|
|
65
|
+
const m = norm.match(/\/([^/]+)$/);
|
|
66
|
+
const projectName = m?.[1] ?? "workspace";
|
|
67
|
+
return scopeFromSeed(norm, projectName);
|
|
68
|
+
}
|
|
69
|
+
return resolveCwdScope(target);
|
|
70
|
+
}
|