memorix 1.2.2 → 1.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/CHANGELOG.md +7 -0
  2. package/TEAM.md +86 -86
  3. package/dist/cli/index.js +34 -18
  4. package/dist/cli/index.js.map +1 -1
  5. package/dist/index.js +17 -8
  6. package/dist/index.js.map +1 -1
  7. package/dist/maintenance-runner.js.map +1 -1
  8. package/dist/memcode-runtime/CHANGELOG.md +7 -0
  9. package/dist/sdk.js +17 -8
  10. package/dist/sdk.js.map +1 -1
  11. package/docs/DESIGN_DECISIONS.md +357 -357
  12. package/docs/dev-log/progress.txt +18 -8
  13. package/package.json +1 -1
  14. package/plugins/codex/memorix/.codex-plugin/plugin.json +1 -1
  15. package/src/audit/index.ts +156 -156
  16. package/src/cli/commands/audit-list.ts +89 -89
  17. package/src/cli/commands/background.ts +659 -659
  18. package/src/cli/commands/formation.ts +48 -48
  19. package/src/cli/commands/git-hook-install.ts +111 -111
  20. package/src/cli/commands/handoff.ts +54 -54
  21. package/src/cli/commands/hooks-status.ts +63 -63
  22. package/src/cli/commands/ingest-commit.ts +153 -153
  23. package/src/cli/commands/ingest-image.ts +66 -66
  24. package/src/cli/commands/ingest-log.ts +180 -180
  25. package/src/cli/commands/ingest.ts +44 -44
  26. package/src/cli/commands/integrate-shared.ts +15 -15
  27. package/src/cli/commands/lock.ts +82 -82
  28. package/src/cli/commands/message.ts +104 -104
  29. package/src/cli/commands/poll.ts +58 -58
  30. package/src/cli/commands/purge-all-memory.ts +85 -85
  31. package/src/cli/commands/purge-project-memory.ts +83 -83
  32. package/src/cli/commands/reasoning.ts +118 -118
  33. package/src/cli/commands/serve-shared.ts +118 -118
  34. package/src/cli/commands/session.ts +15 -7
  35. package/src/cli/commands/skills.ts +114 -114
  36. package/src/cli/commands/task.ts +167 -167
  37. package/src/cli/commands/transfer.ts +47 -47
  38. package/src/cli/commands/uninstall-project-artifacts.ts +85 -85
  39. package/src/cli/tui/ChatView.tsx +234 -234
  40. package/src/cli/tui/CommandBar.tsx +312 -312
  41. package/src/cli/tui/ContextRail.tsx +118 -118
  42. package/src/cli/tui/HeaderBar.tsx +72 -72
  43. package/src/cli/tui/LogoBanner.tsx +51 -51
  44. package/src/cli/tui/Sidebar.tsx +179 -179
  45. package/src/cli/tui/index.ts +41 -41
  46. package/src/cli/tui/markdown-render.tsx +371 -371
  47. package/src/cli/tui/session-service.ts +3 -2
  48. package/src/cli/tui/use-mouse.ts +157 -157
  49. package/src/cli/tui/useNavigation.ts +56 -56
  50. package/src/cli/update-checker.ts +211 -211
  51. package/src/cli/version.ts +7 -7
  52. package/src/cli/workbench.ts +1 -1
  53. package/src/compact/token-budget.ts +74 -74
  54. package/src/dashboard/project-classification.ts +64 -64
  55. package/src/embedding/fastembed-provider.ts +142 -142
  56. package/src/embedding/transformers-provider.ts +111 -111
  57. package/src/git/extractor.ts +209 -209
  58. package/src/git/hooks-path.ts +85 -85
  59. package/src/hooks/pattern-detector.ts +173 -173
  60. package/src/hooks/significance-filter.ts +250 -250
  61. package/src/llm/memory-manager.ts +328 -328
  62. package/src/llm/provider.ts +885 -885
  63. package/src/llm/quality.ts +248 -248
  64. package/src/memory/attribution-guard.ts +249 -249
  65. package/src/memory/disclosure-policy.ts +135 -135
  66. package/src/memory/entity-extractor.ts +197 -197
  67. package/src/memory/formation/evaluate.ts +217 -217
  68. package/src/memory/formation/extract.ts +361 -361
  69. package/src/memory/formation/index.ts +417 -417
  70. package/src/memory/formation/resolve.ts +344 -344
  71. package/src/memory/formation/types.ts +315 -315
  72. package/src/memory/freshness.ts +122 -122
  73. package/src/memory/graph.ts +197 -197
  74. package/src/memory/refs.ts +94 -94
  75. package/src/memory/secret-filter.ts +79 -79
  76. package/src/memory/session.ts +24 -9
  77. package/src/multimodal/image-loader.ts +143 -143
  78. package/src/orchestrate/adapters/claude-stream.ts +192 -192
  79. package/src/orchestrate/adapters/claude.ts +111 -111
  80. package/src/orchestrate/adapters/codex-stream.ts +134 -134
  81. package/src/orchestrate/adapters/codex.ts +41 -41
  82. package/src/orchestrate/adapters/gemini-stream.ts +166 -166
  83. package/src/orchestrate/adapters/gemini.ts +42 -42
  84. package/src/orchestrate/adapters/index.ts +73 -73
  85. package/src/orchestrate/adapters/opencode-stream.ts +143 -143
  86. package/src/orchestrate/adapters/opencode.ts +47 -47
  87. package/src/orchestrate/adapters/spawn-helper.ts +286 -286
  88. package/src/orchestrate/adapters/types.ts +77 -77
  89. package/src/orchestrate/capability-router.ts +284 -284
  90. package/src/orchestrate/context-compact.ts +188 -188
  91. package/src/orchestrate/cost-tracker.ts +219 -219
  92. package/src/orchestrate/error-recovery.ts +191 -191
  93. package/src/orchestrate/evidence.ts +140 -140
  94. package/src/orchestrate/ledger.ts +110 -110
  95. package/src/orchestrate/memorix-bridge.ts +343 -343
  96. package/src/orchestrate/output-budget.ts +80 -80
  97. package/src/orchestrate/permission.ts +152 -152
  98. package/src/orchestrate/pipeline-trace.ts +131 -131
  99. package/src/orchestrate/prompt-builder.ts +155 -155
  100. package/src/orchestrate/ring-buffer.ts +37 -37
  101. package/src/orchestrate/task-graph.ts +389 -389
  102. package/src/orchestrate/worktree.ts +232 -232
  103. package/src/project/aliases.ts +374 -374
  104. package/src/project/detector.ts +268 -268
  105. package/src/rules/adapters/claude-code.ts +99 -99
  106. package/src/rules/adapters/codex.ts +97 -97
  107. package/src/rules/adapters/copilot.ts +124 -124
  108. package/src/rules/adapters/cursor.ts +114 -114
  109. package/src/rules/adapters/kiro.ts +126 -126
  110. package/src/rules/adapters/trae.ts +56 -56
  111. package/src/rules/adapters/windsurf.ts +83 -83
  112. package/src/rules/syncer.ts +235 -235
  113. package/src/sdk.ts +299 -299
  114. package/src/search/intent-detector.ts +289 -289
  115. package/src/search/query-expansion.ts +52 -52
  116. package/src/server/formation-timeout.ts +27 -27
  117. package/src/server.ts +7 -2
  118. package/src/skills/mini-skills.ts +386 -386
  119. package/src/store/chat-store.ts +119 -119
  120. package/src/store/graph-store.ts +249 -249
  121. package/src/store/mini-skill-store.ts +349 -349
  122. package/src/store/persistence-json.ts +212 -212
  123. package/src/store/persistence.ts +291 -291
  124. package/src/store/project-affinity.ts +195 -195
  125. package/src/team/event-bus.ts +76 -76
  126. package/src/team/file-locks.ts +173 -173
  127. package/src/team/handoff.ts +161 -161
  128. package/src/team/messages.ts +203 -203
  129. package/src/team/poll.ts +132 -132
  130. package/src/team/tasks.ts +211 -211
  131. package/src/workspace/mcp-adapters/codex.ts +191 -191
  132. package/src/workspace/mcp-adapters/copilot.ts +105 -105
  133. package/src/workspace/mcp-adapters/cursor.ts +53 -53
  134. package/src/workspace/mcp-adapters/kiro.ts +64 -64
  135. package/src/workspace/mcp-adapters/opencode.ts +123 -123
  136. package/src/workspace/mcp-adapters/trae.ts +134 -134
  137. package/src/workspace/mcp-adapters/windsurf.ts +91 -91
  138. package/src/workspace/sanitizer.ts +60 -60
  139. package/src/workspace/workflow-sync.ts +131 -131
@@ -1,64 +1,64 @@
1
- /**
2
- * Project classification helpers — shared by /api/projects and /api/identity.
3
- *
4
- * Three kinds:
5
- * - 'real': genuine user projects (e.g. AVIDS2/memorix, github.com/org/repo)
6
- * - 'temporary': test/demo/smoke/e2e scratch projects (local/task-*, local/smoke-*, etc.)
7
- * - 'placeholder': unresolved / obviously broken IDs (placeholder/*, __unresolved__, System32)
8
- *
9
- * A 'dirty' project is one with a clearly broken canonical ID (System32 etc.).
10
- * 'dirty' and 'temporary' are orthogonal axes:
11
- * - local/task-abc → temporary, NOT dirty
12
- * - placeholder/xxx → placeholder, dirty
13
- * - System32\something → real-looking location but dirty (broken ID)
14
- */
15
-
16
- export type ProjectKind = 'real' | 'temporary' | 'placeholder';
17
-
18
- /** Regex list — anything matching is temporary (scratch projects) */
19
- const TEMPORARY_PATTERNS: RegExp[] = [
20
- /^local\/task-/i,
21
- /^local\/smoke-/i,
22
- /^local\/release-smoke-/i,
23
- /^local\/memorix-e2e-/i,
24
- /^local\/orchestrate-/i,
25
- /^local\/scratch-/i,
26
- /^local\/tmp-/i,
27
- ];
28
-
29
- /** Regex list — anything matching is placeholder/unresolved */
30
- const PLACEHOLDER_PATTERNS: RegExp[] = [
31
- /^__unresolved__$/,
32
- /^placeholder\//i,
33
- ];
34
-
35
- /** Regex list — IDs that indicate a broken canonical ID (dirty). */
36
- const DIRTY_PATTERNS: RegExp[] = [
37
- /^placeholder\//i,
38
- /System32/i,
39
- /Microsoft VS Code/i,
40
- /node_modules/i,
41
- /\.vscode/i,
42
- /^local\/[A-Z]:\\/i,
43
- ];
44
-
45
- export function classifyProjectId(id: string): ProjectKind {
46
- if (!id) return 'placeholder';
47
- if (PLACEHOLDER_PATTERNS.some(p => p.test(id))) return 'placeholder';
48
- if (TEMPORARY_PATTERNS.some(p => p.test(id))) return 'temporary';
49
- return 'real';
50
- }
51
-
52
- export function isDirtyProjectId(id: string): boolean {
53
- if (!id) return false;
54
- return DIRTY_PATTERNS.some(p => p.test(id));
55
- }
56
-
57
- /** Friendly label for UI badges */
58
- export function projectKindLabel(kind: ProjectKind): string {
59
- switch (kind) {
60
- case 'real': return 'real';
61
- case 'temporary': return 'temporary';
62
- case 'placeholder': return 'placeholder';
63
- }
64
- }
1
+ /**
2
+ * Project classification helpers — shared by /api/projects and /api/identity.
3
+ *
4
+ * Three kinds:
5
+ * - 'real': genuine user projects (e.g. AVIDS2/memorix, github.com/org/repo)
6
+ * - 'temporary': test/demo/smoke/e2e scratch projects (local/task-*, local/smoke-*, etc.)
7
+ * - 'placeholder': unresolved / obviously broken IDs (placeholder/*, __unresolved__, System32)
8
+ *
9
+ * A 'dirty' project is one with a clearly broken canonical ID (System32 etc.).
10
+ * 'dirty' and 'temporary' are orthogonal axes:
11
+ * - local/task-abc → temporary, NOT dirty
12
+ * - placeholder/xxx → placeholder, dirty
13
+ * - System32\something → real-looking location but dirty (broken ID)
14
+ */
15
+
16
+ export type ProjectKind = 'real' | 'temporary' | 'placeholder';
17
+
18
+ /** Regex list — anything matching is temporary (scratch projects) */
19
+ const TEMPORARY_PATTERNS: RegExp[] = [
20
+ /^local\/task-/i,
21
+ /^local\/smoke-/i,
22
+ /^local\/release-smoke-/i,
23
+ /^local\/memorix-e2e-/i,
24
+ /^local\/orchestrate-/i,
25
+ /^local\/scratch-/i,
26
+ /^local\/tmp-/i,
27
+ ];
28
+
29
+ /** Regex list — anything matching is placeholder/unresolved */
30
+ const PLACEHOLDER_PATTERNS: RegExp[] = [
31
+ /^__unresolved__$/,
32
+ /^placeholder\//i,
33
+ ];
34
+
35
+ /** Regex list — IDs that indicate a broken canonical ID (dirty). */
36
+ const DIRTY_PATTERNS: RegExp[] = [
37
+ /^placeholder\//i,
38
+ /System32/i,
39
+ /Microsoft VS Code/i,
40
+ /node_modules/i,
41
+ /\.vscode/i,
42
+ /^local\/[A-Z]:\\/i,
43
+ ];
44
+
45
+ export function classifyProjectId(id: string): ProjectKind {
46
+ if (!id) return 'placeholder';
47
+ if (PLACEHOLDER_PATTERNS.some(p => p.test(id))) return 'placeholder';
48
+ if (TEMPORARY_PATTERNS.some(p => p.test(id))) return 'temporary';
49
+ return 'real';
50
+ }
51
+
52
+ export function isDirtyProjectId(id: string): boolean {
53
+ if (!id) return false;
54
+ return DIRTY_PATTERNS.some(p => p.test(id));
55
+ }
56
+
57
+ /** Friendly label for UI badges */
58
+ export function projectKindLabel(kind: ProjectKind): string {
59
+ switch (kind) {
60
+ case 'real': return 'real';
61
+ case 'temporary': return 'temporary';
62
+ case 'placeholder': return 'placeholder';
63
+ }
64
+ }
@@ -1,142 +1,142 @@
1
- /**
2
- * FastEmbed Provider
3
- *
4
- * Local ONNX-based embedding using fastembed (Qdrant).
5
- * Model: BAAI/bge-small-en-v1.5 (384 dimensions, ~30MB)
6
- *
7
- * This is an optional dependency — if fastembed is not installed,
8
- * the provider module gracefully falls back to fulltext-only search.
9
- *
10
- * Persistent disk cache: embeddings are saved to ~/.memorix/data/.embedding-cache.json
11
- * so server restarts don't need to regenerate them (saves minutes of CPU on 500+ obs).
12
- */
13
-
14
- import { createHash } from 'node:crypto';
15
- import { readFile, writeFile, mkdir } from 'node:fs/promises';
16
- import { join } from 'node:path';
17
- import { homedir } from 'node:os';
18
- import type { EmbeddingProvider } from './provider.js';
19
-
20
- const CACHE_DIR = process.env.MEMORIX_DATA_DIR || join(homedir(), '.memorix', 'data');
21
- const CACHE_FILE = join(CACHE_DIR, '.embedding-cache.json');
22
-
23
- // In-memory cache keyed by text hash → embedding
24
- const cache = new Map<string, number[]>();
25
- const MAX_CACHE_SIZE = 5000;
26
- let diskCacheDirty = false;
27
-
28
- function textHash(text: string): string {
29
- return createHash('sha256').update(text).digest('hex').slice(0, 16);
30
- }
31
-
32
- async function loadDiskCache(): Promise<void> {
33
- try {
34
- const raw = await readFile(CACHE_FILE, 'utf-8');
35
- const entries: [string, number[]][] = JSON.parse(raw);
36
- for (const [k, v] of entries) cache.set(k, v);
37
- console.error(`[memorix] Loaded ${entries.length} cached embeddings from disk`);
38
- } catch {
39
- // No cache file or corrupt — start fresh
40
- }
41
- }
42
-
43
- async function saveDiskCache(): Promise<void> {
44
- if (!diskCacheDirty) return;
45
- try {
46
- await mkdir(CACHE_DIR, { recursive: true });
47
- const entries = Array.from(cache.entries());
48
- await writeFile(CACHE_FILE, JSON.stringify(entries));
49
- diskCacheDirty = false;
50
- } catch {
51
- // Ignore write errors — cache is an optimization, not critical
52
- }
53
- }
54
-
55
- export class FastEmbedProvider implements EmbeddingProvider {
56
- readonly name = 'fastembed-bge-small';
57
- readonly dimensions = 384;
58
-
59
- private model: { embed: (docs: string[], batchSize?: number) => AsyncGenerator<number[][]>; queryEmbed: (query: string) => Promise<number[]> };
60
-
61
- private constructor(model: FastEmbedProvider['model']) {
62
- this.model = model;
63
- }
64
-
65
- /**
66
- * Initialize the FastEmbed provider.
67
- * Downloads model on first use (~30MB), cached locally after.
68
- * Loads persistent embedding cache from disk.
69
- */
70
- static async create(): Promise<FastEmbedProvider> {
71
- // Dynamic import — throws if fastembed is not installed
72
- const { EmbeddingModel, FlagEmbedding } = await import('fastembed');
73
- const model = await FlagEmbedding.init({
74
- model: EmbeddingModel.BGESmallENV15,
75
- });
76
- // Load disk cache before returning — subsequent embedBatch calls will hit cache
77
- await loadDiskCache();
78
- return new FastEmbedProvider(model);
79
- }
80
-
81
- async embed(text: string): Promise<number[]> {
82
- const hash = textHash(text);
83
- const cached = cache.get(hash);
84
- if (cached) return cached;
85
-
86
- const raw = await this.model.queryEmbed(text);
87
- // Ensure plain number[] (fastembed may return Float32Array)
88
- const result = Array.from(raw) as number[];
89
- if (result.length !== this.dimensions) {
90
- throw new Error(`Expected ${this.dimensions}d embedding, got ${result.length}d`);
91
- }
92
- this.cacheSet(hash, result);
93
- return result;
94
- }
95
-
96
- async embedBatch(texts: string[]): Promise<number[][]> {
97
- const results: number[][] = new Array(texts.length);
98
- const uncachedIndices: number[] = [];
99
- const uncachedTexts: string[] = [];
100
-
101
- // Check cache for each text (by hash)
102
- for (let i = 0; i < texts.length; i++) {
103
- const hash = textHash(texts[i]);
104
- const cached = cache.get(hash);
105
- if (cached) {
106
- results[i] = cached;
107
- } else {
108
- uncachedIndices.push(i);
109
- uncachedTexts.push(texts[i]);
110
- }
111
- }
112
-
113
- // Batch embed uncached texts
114
- if (uncachedTexts.length > 0) {
115
- console.error(`[memorix] Embedding ${uncachedTexts.length}/${texts.length} uncached texts (${texts.length - uncachedTexts.length} from cache)`);
116
- let batchIdx = 0;
117
- for await (const batch of this.model.embed(uncachedTexts, 64)) {
118
- for (const vec of batch) {
119
- const originalIdx = uncachedIndices[batchIdx];
120
- const plain = Array.from(vec) as number[];
121
- results[originalIdx] = plain;
122
- this.cacheSet(textHash(uncachedTexts[batchIdx]), plain);
123
- batchIdx++;
124
- }
125
- }
126
- // Persist cache to disk after batch operations
127
- await saveDiskCache();
128
- }
129
-
130
- return results;
131
- }
132
-
133
- private cacheSet(hash: string, value: number[]): void {
134
- // Evict oldest entries if cache is full
135
- if (cache.size >= MAX_CACHE_SIZE) {
136
- const firstKey = cache.keys().next().value;
137
- if (firstKey !== undefined) cache.delete(firstKey);
138
- }
139
- cache.set(hash, value);
140
- diskCacheDirty = true;
141
- }
142
- }
1
+ /**
2
+ * FastEmbed Provider
3
+ *
4
+ * Local ONNX-based embedding using fastembed (Qdrant).
5
+ * Model: BAAI/bge-small-en-v1.5 (384 dimensions, ~30MB)
6
+ *
7
+ * This is an optional dependency — if fastembed is not installed,
8
+ * the provider module gracefully falls back to fulltext-only search.
9
+ *
10
+ * Persistent disk cache: embeddings are saved to ~/.memorix/data/.embedding-cache.json
11
+ * so server restarts don't need to regenerate them (saves minutes of CPU on 500+ obs).
12
+ */
13
+
14
+ import { createHash } from 'node:crypto';
15
+ import { readFile, writeFile, mkdir } from 'node:fs/promises';
16
+ import { join } from 'node:path';
17
+ import { homedir } from 'node:os';
18
+ import type { EmbeddingProvider } from './provider.js';
19
+
20
+ const CACHE_DIR = process.env.MEMORIX_DATA_DIR || join(homedir(), '.memorix', 'data');
21
+ const CACHE_FILE = join(CACHE_DIR, '.embedding-cache.json');
22
+
23
+ // In-memory cache keyed by text hash → embedding
24
+ const cache = new Map<string, number[]>();
25
+ const MAX_CACHE_SIZE = 5000;
26
+ let diskCacheDirty = false;
27
+
28
+ function textHash(text: string): string {
29
+ return createHash('sha256').update(text).digest('hex').slice(0, 16);
30
+ }
31
+
32
+ async function loadDiskCache(): Promise<void> {
33
+ try {
34
+ const raw = await readFile(CACHE_FILE, 'utf-8');
35
+ const entries: [string, number[]][] = JSON.parse(raw);
36
+ for (const [k, v] of entries) cache.set(k, v);
37
+ console.error(`[memorix] Loaded ${entries.length} cached embeddings from disk`);
38
+ } catch {
39
+ // No cache file or corrupt — start fresh
40
+ }
41
+ }
42
+
43
+ async function saveDiskCache(): Promise<void> {
44
+ if (!diskCacheDirty) return;
45
+ try {
46
+ await mkdir(CACHE_DIR, { recursive: true });
47
+ const entries = Array.from(cache.entries());
48
+ await writeFile(CACHE_FILE, JSON.stringify(entries));
49
+ diskCacheDirty = false;
50
+ } catch {
51
+ // Ignore write errors — cache is an optimization, not critical
52
+ }
53
+ }
54
+
55
+ export class FastEmbedProvider implements EmbeddingProvider {
56
+ readonly name = 'fastembed-bge-small';
57
+ readonly dimensions = 384;
58
+
59
+ private model: { embed: (docs: string[], batchSize?: number) => AsyncGenerator<number[][]>; queryEmbed: (query: string) => Promise<number[]> };
60
+
61
+ private constructor(model: FastEmbedProvider['model']) {
62
+ this.model = model;
63
+ }
64
+
65
+ /**
66
+ * Initialize the FastEmbed provider.
67
+ * Downloads model on first use (~30MB), cached locally after.
68
+ * Loads persistent embedding cache from disk.
69
+ */
70
+ static async create(): Promise<FastEmbedProvider> {
71
+ // Dynamic import — throws if fastembed is not installed
72
+ const { EmbeddingModel, FlagEmbedding } = await import('fastembed');
73
+ const model = await FlagEmbedding.init({
74
+ model: EmbeddingModel.BGESmallENV15,
75
+ });
76
+ // Load disk cache before returning — subsequent embedBatch calls will hit cache
77
+ await loadDiskCache();
78
+ return new FastEmbedProvider(model);
79
+ }
80
+
81
+ async embed(text: string): Promise<number[]> {
82
+ const hash = textHash(text);
83
+ const cached = cache.get(hash);
84
+ if (cached) return cached;
85
+
86
+ const raw = await this.model.queryEmbed(text);
87
+ // Ensure plain number[] (fastembed may return Float32Array)
88
+ const result = Array.from(raw) as number[];
89
+ if (result.length !== this.dimensions) {
90
+ throw new Error(`Expected ${this.dimensions}d embedding, got ${result.length}d`);
91
+ }
92
+ this.cacheSet(hash, result);
93
+ return result;
94
+ }
95
+
96
+ async embedBatch(texts: string[]): Promise<number[][]> {
97
+ const results: number[][] = new Array(texts.length);
98
+ const uncachedIndices: number[] = [];
99
+ const uncachedTexts: string[] = [];
100
+
101
+ // Check cache for each text (by hash)
102
+ for (let i = 0; i < texts.length; i++) {
103
+ const hash = textHash(texts[i]);
104
+ const cached = cache.get(hash);
105
+ if (cached) {
106
+ results[i] = cached;
107
+ } else {
108
+ uncachedIndices.push(i);
109
+ uncachedTexts.push(texts[i]);
110
+ }
111
+ }
112
+
113
+ // Batch embed uncached texts
114
+ if (uncachedTexts.length > 0) {
115
+ console.error(`[memorix] Embedding ${uncachedTexts.length}/${texts.length} uncached texts (${texts.length - uncachedTexts.length} from cache)`);
116
+ let batchIdx = 0;
117
+ for await (const batch of this.model.embed(uncachedTexts, 64)) {
118
+ for (const vec of batch) {
119
+ const originalIdx = uncachedIndices[batchIdx];
120
+ const plain = Array.from(vec) as number[];
121
+ results[originalIdx] = plain;
122
+ this.cacheSet(textHash(uncachedTexts[batchIdx]), plain);
123
+ batchIdx++;
124
+ }
125
+ }
126
+ // Persist cache to disk after batch operations
127
+ await saveDiskCache();
128
+ }
129
+
130
+ return results;
131
+ }
132
+
133
+ private cacheSet(hash: string, value: number[]): void {
134
+ // Evict oldest entries if cache is full
135
+ if (cache.size >= MAX_CACHE_SIZE) {
136
+ const firstKey = cache.keys().next().value;
137
+ if (firstKey !== undefined) cache.delete(firstKey);
138
+ }
139
+ cache.set(hash, value);
140
+ diskCacheDirty = true;
141
+ }
142
+ }
@@ -1,111 +1,111 @@
1
- /**
2
- * Transformers.js Provider
3
- *
4
- * Pure JavaScript embedding using @huggingface/transformers (HuggingFace).
5
- * Model: Xenova/all-MiniLM-L6-v2 (384 dimensions, ~22MB quantized)
6
- *
7
- * Key advantages over fastembed:
8
- * - No native ONNX binding required (pure JS / WASM)
9
- * - Works out-of-the-box on Windows, macOS, Linux
10
- * - Supports quantized models (q8, q4) for smaller footprint
11
- *
12
- * This is an optional dependency — if @huggingface/transformers is not
13
- * installed, the provider module gracefully falls back to the next option.
14
- *
15
- * Inspired by Mem0's multi-provider embedding architecture.
16
- */
17
-
18
- import type { EmbeddingProvider } from './provider.js';
19
-
20
- // In-memory LRU cache
21
- const cache = new Map<string, number[]>();
22
- const MAX_CACHE_SIZE = 5000;
23
-
24
- export class TransformersProvider implements EmbeddingProvider {
25
- readonly name = 'transformers-minilm';
26
- readonly dimensions = 384;
27
-
28
- private extractor: any; // Pipeline instance
29
-
30
- private constructor(extractor: any) {
31
- this.extractor = extractor;
32
- }
33
-
34
- /**
35
- * Initialize the Transformers.js provider.
36
- * Downloads model on first use (~22MB quantized), cached locally after.
37
- */
38
- static async create(): Promise<TransformersProvider> {
39
- // Dynamic import — throws if @huggingface/transformers is not installed
40
- const { pipeline } = await import('@huggingface/transformers');
41
- const extractor = await pipeline(
42
- 'feature-extraction',
43
- 'Xenova/all-MiniLM-L6-v2',
44
- { dtype: 'q8' }, // Quantized for small footprint
45
- );
46
- return new TransformersProvider(extractor);
47
- }
48
-
49
- async embed(text: string): Promise<number[]> {
50
- // Check cache first
51
- const cached = cache.get(text);
52
- if (cached) return cached;
53
-
54
- const output = await this.extractor(text, {
55
- pooling: 'mean',
56
- normalize: true,
57
- });
58
-
59
- // output.tolist() returns [[...384 floats]]
60
- const result: number[] = Array.from(output.tolist()[0]);
61
- if (result.length !== this.dimensions) {
62
- throw new Error(`Expected ${this.dimensions}d embedding, got ${result.length}d`);
63
- }
64
-
65
- this.cacheSet(text, result);
66
- return result;
67
- }
68
-
69
- async embedBatch(texts: string[]): Promise<number[][]> {
70
- const results: number[][] = new Array(texts.length);
71
- const uncachedIndices: number[] = [];
72
- const uncachedTexts: string[] = [];
73
-
74
- // Check cache for each text
75
- for (let i = 0; i < texts.length; i++) {
76
- const cached = cache.get(texts[i]);
77
- if (cached) {
78
- results[i] = cached;
79
- } else {
80
- uncachedIndices.push(i);
81
- uncachedTexts.push(texts[i]);
82
- }
83
- }
84
-
85
- // Batch embed uncached texts
86
- if (uncachedTexts.length > 0) {
87
- const output = await this.extractor(uncachedTexts, {
88
- pooling: 'mean',
89
- normalize: true,
90
- });
91
- const allVecs: number[][] = output.tolist();
92
-
93
- for (let i = 0; i < allVecs.length; i++) {
94
- const vec = Array.from(allVecs[i]) as number[];
95
- const originalIdx = uncachedIndices[i];
96
- results[originalIdx] = vec;
97
- this.cacheSet(uncachedTexts[i], vec);
98
- }
99
- }
100
-
101
- return results;
102
- }
103
-
104
- private cacheSet(key: string, value: number[]): void {
105
- if (cache.size >= MAX_CACHE_SIZE) {
106
- const firstKey = cache.keys().next().value;
107
- if (firstKey !== undefined) cache.delete(firstKey);
108
- }
109
- cache.set(key, value);
110
- }
111
- }
1
+ /**
2
+ * Transformers.js Provider
3
+ *
4
+ * Pure JavaScript embedding using @huggingface/transformers (HuggingFace).
5
+ * Model: Xenova/all-MiniLM-L6-v2 (384 dimensions, ~22MB quantized)
6
+ *
7
+ * Key advantages over fastembed:
8
+ * - No native ONNX binding required (pure JS / WASM)
9
+ * - Works out-of-the-box on Windows, macOS, Linux
10
+ * - Supports quantized models (q8, q4) for smaller footprint
11
+ *
12
+ * This is an optional dependency — if @huggingface/transformers is not
13
+ * installed, the provider module gracefully falls back to the next option.
14
+ *
15
+ * Inspired by Mem0's multi-provider embedding architecture.
16
+ */
17
+
18
+ import type { EmbeddingProvider } from './provider.js';
19
+
20
+ // In-memory LRU cache
21
+ const cache = new Map<string, number[]>();
22
+ const MAX_CACHE_SIZE = 5000;
23
+
24
+ export class TransformersProvider implements EmbeddingProvider {
25
+ readonly name = 'transformers-minilm';
26
+ readonly dimensions = 384;
27
+
28
+ private extractor: any; // Pipeline instance
29
+
30
+ private constructor(extractor: any) {
31
+ this.extractor = extractor;
32
+ }
33
+
34
+ /**
35
+ * Initialize the Transformers.js provider.
36
+ * Downloads model on first use (~22MB quantized), cached locally after.
37
+ */
38
+ static async create(): Promise<TransformersProvider> {
39
+ // Dynamic import — throws if @huggingface/transformers is not installed
40
+ const { pipeline } = await import('@huggingface/transformers');
41
+ const extractor = await pipeline(
42
+ 'feature-extraction',
43
+ 'Xenova/all-MiniLM-L6-v2',
44
+ { dtype: 'q8' }, // Quantized for small footprint
45
+ );
46
+ return new TransformersProvider(extractor);
47
+ }
48
+
49
+ async embed(text: string): Promise<number[]> {
50
+ // Check cache first
51
+ const cached = cache.get(text);
52
+ if (cached) return cached;
53
+
54
+ const output = await this.extractor(text, {
55
+ pooling: 'mean',
56
+ normalize: true,
57
+ });
58
+
59
+ // output.tolist() returns [[...384 floats]]
60
+ const result: number[] = Array.from(output.tolist()[0]);
61
+ if (result.length !== this.dimensions) {
62
+ throw new Error(`Expected ${this.dimensions}d embedding, got ${result.length}d`);
63
+ }
64
+
65
+ this.cacheSet(text, result);
66
+ return result;
67
+ }
68
+
69
+ async embedBatch(texts: string[]): Promise<number[][]> {
70
+ const results: number[][] = new Array(texts.length);
71
+ const uncachedIndices: number[] = [];
72
+ const uncachedTexts: string[] = [];
73
+
74
+ // Check cache for each text
75
+ for (let i = 0; i < texts.length; i++) {
76
+ const cached = cache.get(texts[i]);
77
+ if (cached) {
78
+ results[i] = cached;
79
+ } else {
80
+ uncachedIndices.push(i);
81
+ uncachedTexts.push(texts[i]);
82
+ }
83
+ }
84
+
85
+ // Batch embed uncached texts
86
+ if (uncachedTexts.length > 0) {
87
+ const output = await this.extractor(uncachedTexts, {
88
+ pooling: 'mean',
89
+ normalize: true,
90
+ });
91
+ const allVecs: number[][] = output.tolist();
92
+
93
+ for (let i = 0; i < allVecs.length; i++) {
94
+ const vec = Array.from(allVecs[i]) as number[];
95
+ const originalIdx = uncachedIndices[i];
96
+ results[originalIdx] = vec;
97
+ this.cacheSet(uncachedTexts[i], vec);
98
+ }
99
+ }
100
+
101
+ return results;
102
+ }
103
+
104
+ private cacheSet(key: string, value: number[]): void {
105
+ if (cache.size >= MAX_CACHE_SIZE) {
106
+ const firstKey = cache.keys().next().value;
107
+ if (firstKey !== undefined) cache.delete(firstKey);
108
+ }
109
+ cache.set(key, value);
110
+ }
111
+ }