memorix 1.2.2 → 1.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/README.md +3 -3
- package/README.zh-CN.md +3 -3
- package/TEAM.md +86 -86
- package/dist/cli/index.js +5199 -4726
- package/dist/cli/index.js.map +1 -1
- package/dist/index.js +428 -49
- package/dist/index.js.map +1 -1
- package/dist/maintenance-runner.js +97 -18
- package/dist/maintenance-runner.js.map +1 -1
- package/dist/memcode-runtime/CHANGELOG.md +27 -0
- package/dist/sdk.js +428 -49
- package/dist/sdk.js.map +1 -1
- package/docs/1.2.4-PERSISTENT-MEMORY-DELIVERY.md +86 -0
- package/docs/AGENT_OPERATOR_PLAYBOOK.md +13 -1
- package/docs/API_REFERENCE.md +13 -3
- package/docs/DESIGN_DECISIONS.md +357 -357
- package/docs/dev-log/progress.txt +60 -9
- package/package.json +1 -1
- package/plugins/codex/memorix/.codex-plugin/plugin.json +1 -1
- package/src/audit/index.ts +156 -156
- package/src/cli/capability-map.ts +1 -1
- package/src/cli/command-guide.ts +4 -1
- package/src/cli/commands/agent-integrations.ts +5 -1
- package/src/cli/commands/audit-list.ts +89 -89
- package/src/cli/commands/background.ts +659 -659
- package/src/cli/commands/codegraph.ts +1 -1
- package/src/cli/commands/context.ts +9 -1
- package/src/cli/commands/formation.ts +48 -48
- package/src/cli/commands/git-hook-install.ts +111 -111
- package/src/cli/commands/handoff.ts +54 -54
- package/src/cli/commands/hooks-status.ts +63 -63
- package/src/cli/commands/ingest-commit.ts +153 -153
- package/src/cli/commands/ingest-image.ts +66 -66
- package/src/cli/commands/ingest-log.ts +180 -180
- package/src/cli/commands/ingest.ts +44 -44
- package/src/cli/commands/integrate-shared.ts +15 -15
- package/src/cli/commands/lock.ts +82 -82
- package/src/cli/commands/message.ts +104 -104
- package/src/cli/commands/poll.ts +58 -58
- package/src/cli/commands/purge-all-memory.ts +85 -85
- package/src/cli/commands/purge-project-memory.ts +83 -83
- package/src/cli/commands/reasoning.ts +118 -118
- package/src/cli/commands/resume.ts +31 -0
- package/src/cli/commands/serve-shared.ts +118 -118
- package/src/cli/commands/session.ts +15 -7
- package/src/cli/commands/skills.ts +114 -114
- package/src/cli/commands/task.ts +167 -167
- package/src/cli/commands/transfer.ts +47 -47
- package/src/cli/commands/uninstall-project-artifacts.ts +85 -85
- package/src/cli/index.ts +3 -1
- package/src/cli/tui/ChatView.tsx +234 -234
- package/src/cli/tui/CommandBar.tsx +312 -312
- package/src/cli/tui/ContextRail.tsx +118 -118
- package/src/cli/tui/HeaderBar.tsx +72 -72
- package/src/cli/tui/LogoBanner.tsx +51 -51
- package/src/cli/tui/Sidebar.tsx +179 -179
- package/src/cli/tui/index.ts +41 -41
- package/src/cli/tui/markdown-render.tsx +371 -371
- package/src/cli/tui/session-service.ts +3 -2
- package/src/cli/tui/use-mouse.ts +157 -157
- package/src/cli/tui/useNavigation.ts +56 -56
- package/src/cli/update-checker.ts +211 -211
- package/src/cli/version.ts +7 -7
- package/src/cli/workbench.ts +1 -1
- package/src/codegraph/auto-context.ts +54 -1
- package/src/codegraph/task-lens.ts +29 -0
- package/src/compact/token-budget.ts +89 -74
- package/src/config/toml-loader.ts +9 -5
- package/src/dashboard/project-classification.ts +64 -64
- package/src/embedding/fastembed-provider.ts +142 -142
- package/src/embedding/transformers-provider.ts +111 -111
- package/src/git/extractor.ts +209 -209
- package/src/git/hooks-path.ts +85 -85
- package/src/hooks/handler.ts +127 -66
- package/src/hooks/installers/index.ts +5 -4
- package/src/hooks/official-skills.ts +6 -4
- package/src/hooks/pattern-detector.ts +173 -173
- package/src/hooks/rules/memorix-agent-rules.md +9 -7
- package/src/hooks/significance-filter.ts +250 -250
- package/src/knowledge/context-assembly.ts +4 -1
- package/src/knowledge/workset.ts +89 -1
- package/src/llm/memory-manager.ts +328 -328
- package/src/llm/provider.ts +885 -885
- package/src/llm/quality.ts +248 -248
- package/src/memory/attribution-guard.ts +249 -249
- package/src/memory/disclosure-policy.ts +135 -135
- package/src/memory/entity-extractor.ts +197 -197
- package/src/memory/formation/evaluate.ts +217 -217
- package/src/memory/formation/extract.ts +361 -361
- package/src/memory/formation/index.ts +417 -417
- package/src/memory/formation/resolve.ts +344 -344
- package/src/memory/formation/types.ts +315 -315
- package/src/memory/freshness.ts +122 -122
- package/src/memory/graph.ts +197 -197
- package/src/memory/refs.ts +94 -94
- package/src/memory/secret-filter.ts +79 -79
- package/src/memory/session.ts +158 -9
- package/src/multimodal/image-loader.ts +143 -143
- package/src/orchestrate/adapters/claude-stream.ts +192 -192
- package/src/orchestrate/adapters/claude.ts +111 -111
- package/src/orchestrate/adapters/codex-stream.ts +134 -134
- package/src/orchestrate/adapters/codex.ts +41 -41
- package/src/orchestrate/adapters/gemini-stream.ts +166 -166
- package/src/orchestrate/adapters/gemini.ts +42 -42
- package/src/orchestrate/adapters/index.ts +73 -73
- package/src/orchestrate/adapters/opencode-stream.ts +143 -143
- package/src/orchestrate/adapters/opencode.ts +47 -47
- package/src/orchestrate/adapters/spawn-helper.ts +286 -286
- package/src/orchestrate/adapters/types.ts +77 -77
- package/src/orchestrate/capability-router.ts +284 -284
- package/src/orchestrate/context-compact.ts +188 -188
- package/src/orchestrate/cost-tracker.ts +219 -219
- package/src/orchestrate/error-recovery.ts +191 -191
- package/src/orchestrate/evidence.ts +140 -140
- package/src/orchestrate/ledger.ts +110 -110
- package/src/orchestrate/memorix-bridge.ts +343 -343
- package/src/orchestrate/output-budget.ts +80 -80
- package/src/orchestrate/permission.ts +152 -152
- package/src/orchestrate/pipeline-trace.ts +131 -131
- package/src/orchestrate/prompt-builder.ts +155 -155
- package/src/orchestrate/ring-buffer.ts +37 -37
- package/src/orchestrate/task-graph.ts +389 -389
- package/src/orchestrate/worktree.ts +232 -232
- package/src/project/aliases.ts +374 -374
- package/src/project/detector.ts +268 -268
- package/src/rules/adapters/claude-code.ts +99 -99
- package/src/rules/adapters/codex.ts +97 -97
- package/src/rules/adapters/copilot.ts +124 -124
- package/src/rules/adapters/cursor.ts +114 -114
- package/src/rules/adapters/kiro.ts +126 -126
- package/src/rules/adapters/trae.ts +56 -56
- package/src/rules/adapters/windsurf.ts +83 -83
- package/src/rules/syncer.ts +235 -235
- package/src/sdk.ts +299 -299
- package/src/search/intent-detector.ts +289 -289
- package/src/search/query-expansion.ts +52 -52
- package/src/server/formation-timeout.ts +27 -27
- package/src/server.ts +144 -10
- package/src/skills/mini-skills.ts +386 -386
- package/src/store/bun-sqlite-compat.ts +118 -15
- package/src/store/chat-store.ts +119 -119
- package/src/store/graph-store.ts +249 -249
- package/src/store/mini-skill-store.ts +349 -349
- package/src/store/persistence-json.ts +212 -212
- package/src/store/persistence.ts +291 -291
- package/src/store/project-affinity.ts +195 -195
- package/src/store/sqlite-db.ts +3 -3
- package/src/team/event-bus.ts +76 -76
- package/src/team/file-locks.ts +173 -173
- package/src/team/handoff.ts +161 -161
- package/src/team/messages.ts +203 -203
- package/src/team/poll.ts +132 -132
- package/src/team/tasks.ts +211 -211
- package/src/workspace/mcp-adapters/codex.ts +191 -191
- package/src/workspace/mcp-adapters/copilot.ts +105 -105
- package/src/workspace/mcp-adapters/cursor.ts +53 -53
- package/src/workspace/mcp-adapters/kiro.ts +64 -64
- package/src/workspace/mcp-adapters/opencode.ts +123 -123
- package/src/workspace/mcp-adapters/trae.ts +134 -134
- package/src/workspace/mcp-adapters/windsurf.ts +91 -91
- package/src/workspace/sanitizer.ts +60 -60
- package/src/workspace/workflow-sync.ts +131 -131
|
@@ -1,74 +1,89 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Token Budget Manager
|
|
3
|
-
*
|
|
4
|
-
* Provides token counting and budget management for Progressive Disclosure.
|
|
5
|
-
* Source: gpt-tokenizer (737 stars, JS port of OpenAI's tiktoken)
|
|
6
|
-
*
|
|
7
|
-
* Used by the Compact Engine to determine which layer of detail
|
|
8
|
-
* fits within the caller's token budget.
|
|
9
|
-
*/
|
|
10
|
-
|
|
11
|
-
import { countTokens, isWithinTokenLimit } from 'gpt-tokenizer';
|
|
12
|
-
|
|
13
|
-
/**
|
|
14
|
-
* Count tokens in a string.
|
|
15
|
-
*/
|
|
16
|
-
export function countTextTokens(text: string): number {
|
|
17
|
-
return countTokens(text);
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
/**
|
|
21
|
-
* Check if text fits within a token limit.
|
|
22
|
-
* Returns the token count if within limit, false otherwise.
|
|
23
|
-
*/
|
|
24
|
-
export function fitsInBudget(text: string, limit: number): number | false {
|
|
25
|
-
return isWithinTokenLimit(text, limit);
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
/**
|
|
29
|
-
* Truncate text to fit within a token budget.
|
|
30
|
-
* Truncates at sentence boundaries when possible.
|
|
31
|
-
*/
|
|
32
|
-
export function truncateToTokenBudget(text: string, budget: number): string {
|
|
33
|
-
if (fitsInBudget(text, budget) !== false) {
|
|
34
|
-
return text;
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
// Binary search for the right length
|
|
38
|
-
const sentences = text.split(/(?<=[.!?])\s+/);
|
|
39
|
-
let result = '';
|
|
40
|
-
|
|
41
|
-
for (const sentence of sentences) {
|
|
42
|
-
const candidate = result ? `${result} ${sentence}` : sentence;
|
|
43
|
-
if (fitsInBudget(candidate, budget) === false) {
|
|
44
|
-
break;
|
|
45
|
-
}
|
|
46
|
-
result = candidate;
|
|
47
|
-
}
|
|
48
|
-
|
|
49
|
-
// If no complete sentence fits, truncate by characters
|
|
50
|
-
if (!result) {
|
|
51
|
-
// Rough estimate: 1 token ≈ 4 chars for English, ≈ 1.5 chars for Chinese
|
|
52
|
-
const estimatedChars = budget * 2;
|
|
53
|
-
result = text.slice(0, estimatedChars);
|
|
54
|
-
// Refine
|
|
55
|
-
while (fitsInBudget(result, budget) === false && result.length > 0) {
|
|
56
|
-
result = result.slice(0, Math.floor(result.length * 0.9));
|
|
57
|
-
}
|
|
58
|
-
if (result.length < text.length) {
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
1
|
+
/**
|
|
2
|
+
* Token Budget Manager
|
|
3
|
+
*
|
|
4
|
+
* Provides token counting and budget management for Progressive Disclosure.
|
|
5
|
+
* Source: gpt-tokenizer (737 stars, JS port of OpenAI's tiktoken)
|
|
6
|
+
*
|
|
7
|
+
* Used by the Compact Engine to determine which layer of detail
|
|
8
|
+
* fits within the caller's token budget.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { countTokens, isWithinTokenLimit } from 'gpt-tokenizer';
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Count tokens in a string.
|
|
15
|
+
*/
|
|
16
|
+
export function countTextTokens(text: string): number {
|
|
17
|
+
return countTokens(text);
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Check if text fits within a token limit.
|
|
22
|
+
* Returns the token count if within limit, false otherwise.
|
|
23
|
+
*/
|
|
24
|
+
export function fitsInBudget(text: string, limit: number): number | false {
|
|
25
|
+
return isWithinTokenLimit(text, limit);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Truncate text to fit within a token budget.
|
|
30
|
+
* Truncates at sentence boundaries when possible.
|
|
31
|
+
*/
|
|
32
|
+
export function truncateToTokenBudget(text: string, budget: number): string {
|
|
33
|
+
if (fitsInBudget(text, budget) !== false) {
|
|
34
|
+
return text;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// Binary search for the right length
|
|
38
|
+
const sentences = text.split(/(?<=[.!?])\s+/);
|
|
39
|
+
let result = '';
|
|
40
|
+
|
|
41
|
+
for (const sentence of sentences) {
|
|
42
|
+
const candidate = result ? `${result} ${sentence}` : sentence;
|
|
43
|
+
if (fitsInBudget(candidate, budget) === false) {
|
|
44
|
+
break;
|
|
45
|
+
}
|
|
46
|
+
result = candidate;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// If no complete sentence fits, truncate by characters
|
|
50
|
+
if (!result) {
|
|
51
|
+
// Rough estimate: 1 token ≈ 4 chars for English, ≈ 1.5 chars for Chinese
|
|
52
|
+
const estimatedChars = budget * 2;
|
|
53
|
+
result = text.slice(0, estimatedChars);
|
|
54
|
+
// Refine
|
|
55
|
+
while (fitsInBudget(result, budget) === false && result.length > 0) {
|
|
56
|
+
result = result.slice(0, Math.floor(result.length * 0.9));
|
|
57
|
+
}
|
|
58
|
+
if (result.length < text.length) {
|
|
59
|
+
// A character estimate can end halfway through a flag, path, or symbol.
|
|
60
|
+
// Drop the incomplete whitespace-delimited token instead of returning
|
|
61
|
+
// misleading fragments such as `AUTH...`.
|
|
62
|
+
const nextCharacter = text.charAt(result.length);
|
|
63
|
+
if (nextCharacter && !/[\s.,;:!?)}\]]/.test(nextCharacter)) {
|
|
64
|
+
const boundary = result.search(/\s+\S*$/);
|
|
65
|
+
result = boundary > 0 ? result.slice(0, boundary).trimEnd() : '';
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
// Keep the suffix inside the stated budget when there is room. An empty
|
|
69
|
+
// prefix is more honest than a partial identifier that appears valid.
|
|
70
|
+
while (result && fitsInBudget(result + '...', budget) === false) {
|
|
71
|
+
const boundary = result.lastIndexOf(' ');
|
|
72
|
+
result = boundary > 0 ? result.slice(0, boundary).trimEnd() : '';
|
|
73
|
+
}
|
|
74
|
+
result = result ? result + '...' : '...';
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
return result;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Estimate the token cost of an IndexEntry line.
|
|
83
|
+
* Used to predict compact index size.
|
|
84
|
+
*/
|
|
85
|
+
export function estimateIndexEntryTokens(title: string): number {
|
|
86
|
+
// Format: "| #ID | Time | Icon | Title | ~Tokens |"
|
|
87
|
+
// Overhead is roughly 15 tokens for formatting
|
|
88
|
+
return countTextTokens(title) + 15;
|
|
89
|
+
}
|
|
@@ -144,6 +144,9 @@ function parseTomlValue(raw: string, filePath: string, line: number): unknown {
|
|
|
144
144
|
if (raw.startsWith('"') && raw.endsWith('"')) {
|
|
145
145
|
return raw.slice(1, -1).replace(/\\"/g, '"').replace(/\\\\/g, '\\');
|
|
146
146
|
}
|
|
147
|
+
if (raw.startsWith("'") && raw.endsWith("'")) {
|
|
148
|
+
return raw.slice(1, -1);
|
|
149
|
+
}
|
|
147
150
|
if (raw === 'true') return true;
|
|
148
151
|
if (raw === 'false') return false;
|
|
149
152
|
if (/^-?\d+$/.test(raw)) return Number.parseInt(raw, 10);
|
|
@@ -157,7 +160,7 @@ function parseTomlValue(raw: string, filePath: string, line: number): unknown {
|
|
|
157
160
|
}
|
|
158
161
|
|
|
159
162
|
function stripComment(line: string): string {
|
|
160
|
-
let
|
|
163
|
+
let quote: '"' | "'" | null = null;
|
|
161
164
|
let escaped = false;
|
|
162
165
|
for (let i = 0; i < line.length; i++) {
|
|
163
166
|
const char = line[i];
|
|
@@ -165,15 +168,16 @@ function stripComment(line: string): string {
|
|
|
165
168
|
escaped = false;
|
|
166
169
|
continue;
|
|
167
170
|
}
|
|
168
|
-
if (char === '\\' &&
|
|
171
|
+
if (char === '\\' && quote === '"') {
|
|
169
172
|
escaped = true;
|
|
170
173
|
continue;
|
|
171
174
|
}
|
|
172
|
-
if (char === '"') {
|
|
173
|
-
|
|
175
|
+
if (char === '"' || char === "'") {
|
|
176
|
+
if (quote === char) quote = null;
|
|
177
|
+
else if (quote === null) quote = char;
|
|
174
178
|
continue;
|
|
175
179
|
}
|
|
176
|
-
if (char === '#' &&
|
|
180
|
+
if (char === '#' && quote === null) {
|
|
177
181
|
return line.slice(0, i);
|
|
178
182
|
}
|
|
179
183
|
}
|
|
@@ -1,64 +1,64 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Project classification helpers — shared by /api/projects and /api/identity.
|
|
3
|
-
*
|
|
4
|
-
* Three kinds:
|
|
5
|
-
* - 'real': genuine user projects (e.g. AVIDS2/memorix, github.com/org/repo)
|
|
6
|
-
* - 'temporary': test/demo/smoke/e2e scratch projects (local/task-*, local/smoke-*, etc.)
|
|
7
|
-
* - 'placeholder': unresolved / obviously broken IDs (placeholder/*, __unresolved__, System32)
|
|
8
|
-
*
|
|
9
|
-
* A 'dirty' project is one with a clearly broken canonical ID (System32 etc.).
|
|
10
|
-
* 'dirty' and 'temporary' are orthogonal axes:
|
|
11
|
-
* - local/task-abc → temporary, NOT dirty
|
|
12
|
-
* - placeholder/xxx → placeholder, dirty
|
|
13
|
-
* - System32\something → real-looking location but dirty (broken ID)
|
|
14
|
-
*/
|
|
15
|
-
|
|
16
|
-
export type ProjectKind = 'real' | 'temporary' | 'placeholder';
|
|
17
|
-
|
|
18
|
-
/** Regex list — anything matching is temporary (scratch projects) */
|
|
19
|
-
const TEMPORARY_PATTERNS: RegExp[] = [
|
|
20
|
-
/^local\/task-/i,
|
|
21
|
-
/^local\/smoke-/i,
|
|
22
|
-
/^local\/release-smoke-/i,
|
|
23
|
-
/^local\/memorix-e2e-/i,
|
|
24
|
-
/^local\/orchestrate-/i,
|
|
25
|
-
/^local\/scratch-/i,
|
|
26
|
-
/^local\/tmp-/i,
|
|
27
|
-
];
|
|
28
|
-
|
|
29
|
-
/** Regex list — anything matching is placeholder/unresolved */
|
|
30
|
-
const PLACEHOLDER_PATTERNS: RegExp[] = [
|
|
31
|
-
/^__unresolved__$/,
|
|
32
|
-
/^placeholder\//i,
|
|
33
|
-
];
|
|
34
|
-
|
|
35
|
-
/** Regex list — IDs that indicate a broken canonical ID (dirty). */
|
|
36
|
-
const DIRTY_PATTERNS: RegExp[] = [
|
|
37
|
-
/^placeholder\//i,
|
|
38
|
-
/System32/i,
|
|
39
|
-
/Microsoft VS Code/i,
|
|
40
|
-
/node_modules/i,
|
|
41
|
-
/\.vscode/i,
|
|
42
|
-
/^local\/[A-Z]:\\/i,
|
|
43
|
-
];
|
|
44
|
-
|
|
45
|
-
export function classifyProjectId(id: string): ProjectKind {
|
|
46
|
-
if (!id) return 'placeholder';
|
|
47
|
-
if (PLACEHOLDER_PATTERNS.some(p => p.test(id))) return 'placeholder';
|
|
48
|
-
if (TEMPORARY_PATTERNS.some(p => p.test(id))) return 'temporary';
|
|
49
|
-
return 'real';
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
export function isDirtyProjectId(id: string): boolean {
|
|
53
|
-
if (!id) return false;
|
|
54
|
-
return DIRTY_PATTERNS.some(p => p.test(id));
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
/** Friendly label for UI badges */
|
|
58
|
-
export function projectKindLabel(kind: ProjectKind): string {
|
|
59
|
-
switch (kind) {
|
|
60
|
-
case 'real': return 'real';
|
|
61
|
-
case 'temporary': return 'temporary';
|
|
62
|
-
case 'placeholder': return 'placeholder';
|
|
63
|
-
}
|
|
64
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* Project classification helpers — shared by /api/projects and /api/identity.
|
|
3
|
+
*
|
|
4
|
+
* Three kinds:
|
|
5
|
+
* - 'real': genuine user projects (e.g. AVIDS2/memorix, github.com/org/repo)
|
|
6
|
+
* - 'temporary': test/demo/smoke/e2e scratch projects (local/task-*, local/smoke-*, etc.)
|
|
7
|
+
* - 'placeholder': unresolved / obviously broken IDs (placeholder/*, __unresolved__, System32)
|
|
8
|
+
*
|
|
9
|
+
* A 'dirty' project is one with a clearly broken canonical ID (System32 etc.).
|
|
10
|
+
* 'dirty' and 'temporary' are orthogonal axes:
|
|
11
|
+
* - local/task-abc → temporary, NOT dirty
|
|
12
|
+
* - placeholder/xxx → placeholder, dirty
|
|
13
|
+
* - System32\something → real-looking location but dirty (broken ID)
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
export type ProjectKind = 'real' | 'temporary' | 'placeholder';
|
|
17
|
+
|
|
18
|
+
/** Regex list — anything matching is temporary (scratch projects) */
|
|
19
|
+
const TEMPORARY_PATTERNS: RegExp[] = [
|
|
20
|
+
/^local\/task-/i,
|
|
21
|
+
/^local\/smoke-/i,
|
|
22
|
+
/^local\/release-smoke-/i,
|
|
23
|
+
/^local\/memorix-e2e-/i,
|
|
24
|
+
/^local\/orchestrate-/i,
|
|
25
|
+
/^local\/scratch-/i,
|
|
26
|
+
/^local\/tmp-/i,
|
|
27
|
+
];
|
|
28
|
+
|
|
29
|
+
/** Regex list — anything matching is placeholder/unresolved */
|
|
30
|
+
const PLACEHOLDER_PATTERNS: RegExp[] = [
|
|
31
|
+
/^__unresolved__$/,
|
|
32
|
+
/^placeholder\//i,
|
|
33
|
+
];
|
|
34
|
+
|
|
35
|
+
/** Regex list — IDs that indicate a broken canonical ID (dirty). */
|
|
36
|
+
const DIRTY_PATTERNS: RegExp[] = [
|
|
37
|
+
/^placeholder\//i,
|
|
38
|
+
/System32/i,
|
|
39
|
+
/Microsoft VS Code/i,
|
|
40
|
+
/node_modules/i,
|
|
41
|
+
/\.vscode/i,
|
|
42
|
+
/^local\/[A-Z]:\\/i,
|
|
43
|
+
];
|
|
44
|
+
|
|
45
|
+
export function classifyProjectId(id: string): ProjectKind {
|
|
46
|
+
if (!id) return 'placeholder';
|
|
47
|
+
if (PLACEHOLDER_PATTERNS.some(p => p.test(id))) return 'placeholder';
|
|
48
|
+
if (TEMPORARY_PATTERNS.some(p => p.test(id))) return 'temporary';
|
|
49
|
+
return 'real';
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export function isDirtyProjectId(id: string): boolean {
|
|
53
|
+
if (!id) return false;
|
|
54
|
+
return DIRTY_PATTERNS.some(p => p.test(id));
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** Friendly label for UI badges */
|
|
58
|
+
export function projectKindLabel(kind: ProjectKind): string {
|
|
59
|
+
switch (kind) {
|
|
60
|
+
case 'real': return 'real';
|
|
61
|
+
case 'temporary': return 'temporary';
|
|
62
|
+
case 'placeholder': return 'placeholder';
|
|
63
|
+
}
|
|
64
|
+
}
|
|
@@ -1,142 +1,142 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* FastEmbed Provider
|
|
3
|
-
*
|
|
4
|
-
* Local ONNX-based embedding using fastembed (Qdrant).
|
|
5
|
-
* Model: BAAI/bge-small-en-v1.5 (384 dimensions, ~30MB)
|
|
6
|
-
*
|
|
7
|
-
* This is an optional dependency — if fastembed is not installed,
|
|
8
|
-
* the provider module gracefully falls back to fulltext-only search.
|
|
9
|
-
*
|
|
10
|
-
* Persistent disk cache: embeddings are saved to ~/.memorix/data/.embedding-cache.json
|
|
11
|
-
* so server restarts don't need to regenerate them (saves minutes of CPU on 500+ obs).
|
|
12
|
-
*/
|
|
13
|
-
|
|
14
|
-
import { createHash } from 'node:crypto';
|
|
15
|
-
import { readFile, writeFile, mkdir } from 'node:fs/promises';
|
|
16
|
-
import { join } from 'node:path';
|
|
17
|
-
import { homedir } from 'node:os';
|
|
18
|
-
import type { EmbeddingProvider } from './provider.js';
|
|
19
|
-
|
|
20
|
-
const CACHE_DIR = process.env.MEMORIX_DATA_DIR || join(homedir(), '.memorix', 'data');
|
|
21
|
-
const CACHE_FILE = join(CACHE_DIR, '.embedding-cache.json');
|
|
22
|
-
|
|
23
|
-
// In-memory cache keyed by text hash → embedding
|
|
24
|
-
const cache = new Map<string, number[]>();
|
|
25
|
-
const MAX_CACHE_SIZE = 5000;
|
|
26
|
-
let diskCacheDirty = false;
|
|
27
|
-
|
|
28
|
-
function textHash(text: string): string {
|
|
29
|
-
return createHash('sha256').update(text).digest('hex').slice(0, 16);
|
|
30
|
-
}
|
|
31
|
-
|
|
32
|
-
async function loadDiskCache(): Promise<void> {
|
|
33
|
-
try {
|
|
34
|
-
const raw = await readFile(CACHE_FILE, 'utf-8');
|
|
35
|
-
const entries: [string, number[]][] = JSON.parse(raw);
|
|
36
|
-
for (const [k, v] of entries) cache.set(k, v);
|
|
37
|
-
console.error(`[memorix] Loaded ${entries.length} cached embeddings from disk`);
|
|
38
|
-
} catch {
|
|
39
|
-
// No cache file or corrupt — start fresh
|
|
40
|
-
}
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
async function saveDiskCache(): Promise<void> {
|
|
44
|
-
if (!diskCacheDirty) return;
|
|
45
|
-
try {
|
|
46
|
-
await mkdir(CACHE_DIR, { recursive: true });
|
|
47
|
-
const entries = Array.from(cache.entries());
|
|
48
|
-
await writeFile(CACHE_FILE, JSON.stringify(entries));
|
|
49
|
-
diskCacheDirty = false;
|
|
50
|
-
} catch {
|
|
51
|
-
// Ignore write errors — cache is an optimization, not critical
|
|
52
|
-
}
|
|
53
|
-
}
|
|
54
|
-
|
|
55
|
-
export class FastEmbedProvider implements EmbeddingProvider {
|
|
56
|
-
readonly name = 'fastembed-bge-small';
|
|
57
|
-
readonly dimensions = 384;
|
|
58
|
-
|
|
59
|
-
private model: { embed: (docs: string[], batchSize?: number) => AsyncGenerator<number[][]>; queryEmbed: (query: string) => Promise<number[]> };
|
|
60
|
-
|
|
61
|
-
private constructor(model: FastEmbedProvider['model']) {
|
|
62
|
-
this.model = model;
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
/**
|
|
66
|
-
* Initialize the FastEmbed provider.
|
|
67
|
-
* Downloads model on first use (~30MB), cached locally after.
|
|
68
|
-
* Loads persistent embedding cache from disk.
|
|
69
|
-
*/
|
|
70
|
-
static async create(): Promise<FastEmbedProvider> {
|
|
71
|
-
// Dynamic import — throws if fastembed is not installed
|
|
72
|
-
const { EmbeddingModel, FlagEmbedding } = await import('fastembed');
|
|
73
|
-
const model = await FlagEmbedding.init({
|
|
74
|
-
model: EmbeddingModel.BGESmallENV15,
|
|
75
|
-
});
|
|
76
|
-
// Load disk cache before returning — subsequent embedBatch calls will hit cache
|
|
77
|
-
await loadDiskCache();
|
|
78
|
-
return new FastEmbedProvider(model);
|
|
79
|
-
}
|
|
80
|
-
|
|
81
|
-
async embed(text: string): Promise<number[]> {
|
|
82
|
-
const hash = textHash(text);
|
|
83
|
-
const cached = cache.get(hash);
|
|
84
|
-
if (cached) return cached;
|
|
85
|
-
|
|
86
|
-
const raw = await this.model.queryEmbed(text);
|
|
87
|
-
// Ensure plain number[] (fastembed may return Float32Array)
|
|
88
|
-
const result = Array.from(raw) as number[];
|
|
89
|
-
if (result.length !== this.dimensions) {
|
|
90
|
-
throw new Error(`Expected ${this.dimensions}d embedding, got ${result.length}d`);
|
|
91
|
-
}
|
|
92
|
-
this.cacheSet(hash, result);
|
|
93
|
-
return result;
|
|
94
|
-
}
|
|
95
|
-
|
|
96
|
-
async embedBatch(texts: string[]): Promise<number[][]> {
|
|
97
|
-
const results: number[][] = new Array(texts.length);
|
|
98
|
-
const uncachedIndices: number[] = [];
|
|
99
|
-
const uncachedTexts: string[] = [];
|
|
100
|
-
|
|
101
|
-
// Check cache for each text (by hash)
|
|
102
|
-
for (let i = 0; i < texts.length; i++) {
|
|
103
|
-
const hash = textHash(texts[i]);
|
|
104
|
-
const cached = cache.get(hash);
|
|
105
|
-
if (cached) {
|
|
106
|
-
results[i] = cached;
|
|
107
|
-
} else {
|
|
108
|
-
uncachedIndices.push(i);
|
|
109
|
-
uncachedTexts.push(texts[i]);
|
|
110
|
-
}
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
// Batch embed uncached texts
|
|
114
|
-
if (uncachedTexts.length > 0) {
|
|
115
|
-
console.error(`[memorix] Embedding ${uncachedTexts.length}/${texts.length} uncached texts (${texts.length - uncachedTexts.length} from cache)`);
|
|
116
|
-
let batchIdx = 0;
|
|
117
|
-
for await (const batch of this.model.embed(uncachedTexts, 64)) {
|
|
118
|
-
for (const vec of batch) {
|
|
119
|
-
const originalIdx = uncachedIndices[batchIdx];
|
|
120
|
-
const plain = Array.from(vec) as number[];
|
|
121
|
-
results[originalIdx] = plain;
|
|
122
|
-
this.cacheSet(textHash(uncachedTexts[batchIdx]), plain);
|
|
123
|
-
batchIdx++;
|
|
124
|
-
}
|
|
125
|
-
}
|
|
126
|
-
// Persist cache to disk after batch operations
|
|
127
|
-
await saveDiskCache();
|
|
128
|
-
}
|
|
129
|
-
|
|
130
|
-
return results;
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
private cacheSet(hash: string, value: number[]): void {
|
|
134
|
-
// Evict oldest entries if cache is full
|
|
135
|
-
if (cache.size >= MAX_CACHE_SIZE) {
|
|
136
|
-
const firstKey = cache.keys().next().value;
|
|
137
|
-
if (firstKey !== undefined) cache.delete(firstKey);
|
|
138
|
-
}
|
|
139
|
-
cache.set(hash, value);
|
|
140
|
-
diskCacheDirty = true;
|
|
141
|
-
}
|
|
142
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* FastEmbed Provider
|
|
3
|
+
*
|
|
4
|
+
* Local ONNX-based embedding using fastembed (Qdrant).
|
|
5
|
+
* Model: BAAI/bge-small-en-v1.5 (384 dimensions, ~30MB)
|
|
6
|
+
*
|
|
7
|
+
* This is an optional dependency — if fastembed is not installed,
|
|
8
|
+
* the provider module gracefully falls back to fulltext-only search.
|
|
9
|
+
*
|
|
10
|
+
* Persistent disk cache: embeddings are saved to ~/.memorix/data/.embedding-cache.json
|
|
11
|
+
* so server restarts don't need to regenerate them (saves minutes of CPU on 500+ obs).
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { createHash } from 'node:crypto';
|
|
15
|
+
import { readFile, writeFile, mkdir } from 'node:fs/promises';
|
|
16
|
+
import { join } from 'node:path';
|
|
17
|
+
import { homedir } from 'node:os';
|
|
18
|
+
import type { EmbeddingProvider } from './provider.js';
|
|
19
|
+
|
|
20
|
+
const CACHE_DIR = process.env.MEMORIX_DATA_DIR || join(homedir(), '.memorix', 'data');
|
|
21
|
+
const CACHE_FILE = join(CACHE_DIR, '.embedding-cache.json');
|
|
22
|
+
|
|
23
|
+
// In-memory cache keyed by text hash → embedding
|
|
24
|
+
const cache = new Map<string, number[]>();
|
|
25
|
+
const MAX_CACHE_SIZE = 5000;
|
|
26
|
+
let diskCacheDirty = false;
|
|
27
|
+
|
|
28
|
+
function textHash(text: string): string {
|
|
29
|
+
return createHash('sha256').update(text).digest('hex').slice(0, 16);
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
async function loadDiskCache(): Promise<void> {
|
|
33
|
+
try {
|
|
34
|
+
const raw = await readFile(CACHE_FILE, 'utf-8');
|
|
35
|
+
const entries: [string, number[]][] = JSON.parse(raw);
|
|
36
|
+
for (const [k, v] of entries) cache.set(k, v);
|
|
37
|
+
console.error(`[memorix] Loaded ${entries.length} cached embeddings from disk`);
|
|
38
|
+
} catch {
|
|
39
|
+
// No cache file or corrupt — start fresh
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
async function saveDiskCache(): Promise<void> {
|
|
44
|
+
if (!diskCacheDirty) return;
|
|
45
|
+
try {
|
|
46
|
+
await mkdir(CACHE_DIR, { recursive: true });
|
|
47
|
+
const entries = Array.from(cache.entries());
|
|
48
|
+
await writeFile(CACHE_FILE, JSON.stringify(entries));
|
|
49
|
+
diskCacheDirty = false;
|
|
50
|
+
} catch {
|
|
51
|
+
// Ignore write errors — cache is an optimization, not critical
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export class FastEmbedProvider implements EmbeddingProvider {
|
|
56
|
+
readonly name = 'fastembed-bge-small';
|
|
57
|
+
readonly dimensions = 384;
|
|
58
|
+
|
|
59
|
+
private model: { embed: (docs: string[], batchSize?: number) => AsyncGenerator<number[][]>; queryEmbed: (query: string) => Promise<number[]> };
|
|
60
|
+
|
|
61
|
+
private constructor(model: FastEmbedProvider['model']) {
|
|
62
|
+
this.model = model;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Initialize the FastEmbed provider.
|
|
67
|
+
* Downloads model on first use (~30MB), cached locally after.
|
|
68
|
+
* Loads persistent embedding cache from disk.
|
|
69
|
+
*/
|
|
70
|
+
static async create(): Promise<FastEmbedProvider> {
|
|
71
|
+
// Dynamic import — throws if fastembed is not installed
|
|
72
|
+
const { EmbeddingModel, FlagEmbedding } = await import('fastembed');
|
|
73
|
+
const model = await FlagEmbedding.init({
|
|
74
|
+
model: EmbeddingModel.BGESmallENV15,
|
|
75
|
+
});
|
|
76
|
+
// Load disk cache before returning — subsequent embedBatch calls will hit cache
|
|
77
|
+
await loadDiskCache();
|
|
78
|
+
return new FastEmbedProvider(model);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
async embed(text: string): Promise<number[]> {
|
|
82
|
+
const hash = textHash(text);
|
|
83
|
+
const cached = cache.get(hash);
|
|
84
|
+
if (cached) return cached;
|
|
85
|
+
|
|
86
|
+
const raw = await this.model.queryEmbed(text);
|
|
87
|
+
// Ensure plain number[] (fastembed may return Float32Array)
|
|
88
|
+
const result = Array.from(raw) as number[];
|
|
89
|
+
if (result.length !== this.dimensions) {
|
|
90
|
+
throw new Error(`Expected ${this.dimensions}d embedding, got ${result.length}d`);
|
|
91
|
+
}
|
|
92
|
+
this.cacheSet(hash, result);
|
|
93
|
+
return result;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
async embedBatch(texts: string[]): Promise<number[][]> {
|
|
97
|
+
const results: number[][] = new Array(texts.length);
|
|
98
|
+
const uncachedIndices: number[] = [];
|
|
99
|
+
const uncachedTexts: string[] = [];
|
|
100
|
+
|
|
101
|
+
// Check cache for each text (by hash)
|
|
102
|
+
for (let i = 0; i < texts.length; i++) {
|
|
103
|
+
const hash = textHash(texts[i]);
|
|
104
|
+
const cached = cache.get(hash);
|
|
105
|
+
if (cached) {
|
|
106
|
+
results[i] = cached;
|
|
107
|
+
} else {
|
|
108
|
+
uncachedIndices.push(i);
|
|
109
|
+
uncachedTexts.push(texts[i]);
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// Batch embed uncached texts
|
|
114
|
+
if (uncachedTexts.length > 0) {
|
|
115
|
+
console.error(`[memorix] Embedding ${uncachedTexts.length}/${texts.length} uncached texts (${texts.length - uncachedTexts.length} from cache)`);
|
|
116
|
+
let batchIdx = 0;
|
|
117
|
+
for await (const batch of this.model.embed(uncachedTexts, 64)) {
|
|
118
|
+
for (const vec of batch) {
|
|
119
|
+
const originalIdx = uncachedIndices[batchIdx];
|
|
120
|
+
const plain = Array.from(vec) as number[];
|
|
121
|
+
results[originalIdx] = plain;
|
|
122
|
+
this.cacheSet(textHash(uncachedTexts[batchIdx]), plain);
|
|
123
|
+
batchIdx++;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
// Persist cache to disk after batch operations
|
|
127
|
+
await saveDiskCache();
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
return results;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
private cacheSet(hash: string, value: number[]): void {
|
|
134
|
+
// Evict oldest entries if cache is full
|
|
135
|
+
if (cache.size >= MAX_CACHE_SIZE) {
|
|
136
|
+
const firstKey = cache.keys().next().value;
|
|
137
|
+
if (firstKey !== undefined) cache.delete(firstKey);
|
|
138
|
+
}
|
|
139
|
+
cache.set(hash, value);
|
|
140
|
+
diskCacheDirty = true;
|
|
141
|
+
}
|
|
142
|
+
}
|