memorix 1.1.7 → 1.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/CLAUDE.md +6 -1
  3. package/README.md +21 -0
  4. package/README.zh-CN.md +21 -0
  5. package/TEAM.md +86 -86
  6. package/dist/cli/index.js +852 -214
  7. package/dist/cli/index.js.map +1 -1
  8. package/dist/dashboard/static/index.html +201 -201
  9. package/dist/dashboard/static/style.css +3584 -3584
  10. package/dist/index.js +129 -62
  11. package/dist/index.js.map +1 -1
  12. package/dist/memcode-runtime/CHANGELOG.md +22 -0
  13. package/dist/memcode-runtime/package.json +4 -4
  14. package/dist/sdk.js +129 -62
  15. package/dist/sdk.js.map +1 -1
  16. package/docs/AGENT_OPERATOR_PLAYBOOK.md +18 -0
  17. package/docs/API_REFERENCE.md +2 -0
  18. package/docs/CONFIGURATION.md +18 -0
  19. package/docs/DESIGN_DECISIONS.md +357 -357
  20. package/docs/SETUP.md +10 -0
  21. package/docs/dev-log/progress.txt +23 -30
  22. package/package.json +1 -1
  23. package/src/audit/index.ts +156 -156
  24. package/src/cli/commands/agent-integrations.ts +623 -0
  25. package/src/cli/commands/audit-list.ts +89 -89
  26. package/src/cli/commands/background.ts +659 -659
  27. package/src/cli/commands/cleanup.ts +255 -255
  28. package/src/cli/commands/codegraph.ts +4 -0
  29. package/src/cli/commands/config-get.ts +9 -2
  30. package/src/cli/commands/doctor.ts +26 -0
  31. package/src/cli/commands/formation.ts +48 -48
  32. package/src/cli/commands/git-hook-install.ts +111 -111
  33. package/src/cli/commands/handoff.ts +66 -66
  34. package/src/cli/commands/hooks-status.ts +63 -63
  35. package/src/cli/commands/ingest-commit.ts +153 -153
  36. package/src/cli/commands/ingest-image.ts +73 -73
  37. package/src/cli/commands/ingest-log.ts +180 -180
  38. package/src/cli/commands/ingest.ts +44 -44
  39. package/src/cli/commands/integrate-shared.ts +15 -15
  40. package/src/cli/commands/lock.ts +96 -96
  41. package/src/cli/commands/message.ts +121 -121
  42. package/src/cli/commands/poll.ts +70 -70
  43. package/src/cli/commands/purge-all-memory.ts +85 -85
  44. package/src/cli/commands/purge-project-memory.ts +83 -83
  45. package/src/cli/commands/reasoning.ts +132 -132
  46. package/src/cli/commands/repair.ts +60 -0
  47. package/src/cli/commands/retention.ts +108 -108
  48. package/src/cli/commands/serve-shared.ts +118 -118
  49. package/src/cli/commands/setup.ts +3 -3
  50. package/src/cli/commands/skills.ts +123 -123
  51. package/src/cli/commands/task.ts +192 -192
  52. package/src/cli/commands/transfer.ts +73 -73
  53. package/src/cli/commands/uninstall-project-artifacts.ts +85 -85
  54. package/src/cli/index.ts +3 -1
  55. package/src/cli/tui/ChatView.tsx +234 -234
  56. package/src/cli/tui/CommandBar.tsx +312 -312
  57. package/src/cli/tui/ContextRail.tsx +118 -118
  58. package/src/cli/tui/HeaderBar.tsx +72 -72
  59. package/src/cli/tui/LogoBanner.tsx +51 -51
  60. package/src/cli/tui/Panels.tsx +632 -632
  61. package/src/cli/tui/Sidebar.tsx +179 -179
  62. package/src/cli/tui/chat-service.ts +742 -742
  63. package/src/cli/tui/data.ts +547 -547
  64. package/src/cli/tui/index.ts +41 -41
  65. package/src/cli/tui/markdown-render.tsx +371 -371
  66. package/src/cli/tui/theme.ts +178 -178
  67. package/src/cli/tui/use-mouse.ts +157 -157
  68. package/src/cli/tui/useNavigation.ts +56 -56
  69. package/src/cli/update-checker.ts +211 -211
  70. package/src/cli/version.ts +7 -7
  71. package/src/cli/workbench.ts +1 -1
  72. package/src/codegraph/auto-context.ts +6 -0
  73. package/src/codegraph/context-pack.ts +7 -6
  74. package/src/codegraph/exclude.ts +47 -0
  75. package/src/codegraph/lite-provider.ts +5 -24
  76. package/src/codegraph/project-context.ts +13 -15
  77. package/src/compact/token-budget.ts +74 -74
  78. package/src/config/behavior.ts +59 -59
  79. package/src/config/resolved-config.ts +6 -0
  80. package/src/config/toml-loader.ts +4 -0
  81. package/src/config/yaml-loader.ts +7 -0
  82. package/src/dashboard/project-classification.ts +64 -64
  83. package/src/dashboard/static/index.html +201 -201
  84. package/src/dashboard/static/style.css +3584 -3584
  85. package/src/embedding/fastembed-provider.ts +142 -142
  86. package/src/embedding/transformers-provider.ts +111 -111
  87. package/src/git/extractor.ts +209 -209
  88. package/src/git/hooks-path.ts +85 -85
  89. package/src/git/noise-filter.ts +210 -210
  90. package/src/hooks/installers/index.ts +4 -4
  91. package/src/hooks/official-skills.ts +1 -1
  92. package/src/hooks/pattern-detector.ts +173 -173
  93. package/src/hooks/rules/memorix-agent-rules.md +2 -2
  94. package/src/hooks/significance-filter.ts +250 -250
  95. package/src/llm/memory-manager.ts +328 -328
  96. package/src/llm/provider.ts +885 -885
  97. package/src/llm/quality.ts +248 -248
  98. package/src/memory/attribution-guard.ts +249 -249
  99. package/src/memory/auto-relations.ts +107 -107
  100. package/src/memory/consolidation.ts +302 -302
  101. package/src/memory/disclosure-policy.ts +141 -141
  102. package/src/memory/entity-extractor.ts +197 -197
  103. package/src/memory/formation/evaluate.ts +217 -217
  104. package/src/memory/formation/extract.ts +361 -361
  105. package/src/memory/formation/index.ts +417 -417
  106. package/src/memory/formation/resolve.ts +344 -344
  107. package/src/memory/formation/types.ts +315 -315
  108. package/src/memory/freshness.ts +122 -122
  109. package/src/memory/graph.ts +197 -197
  110. package/src/memory/refs.ts +94 -94
  111. package/src/memory/retention.ts +433 -433
  112. package/src/memory/secret-filter.ts +79 -79
  113. package/src/memory/session.ts +523 -523
  114. package/src/multimodal/image-loader.ts +143 -143
  115. package/src/orchestrate/adapters/claude-stream.ts +192 -192
  116. package/src/orchestrate/adapters/claude.ts +111 -111
  117. package/src/orchestrate/adapters/codex-stream.ts +134 -134
  118. package/src/orchestrate/adapters/codex.ts +41 -41
  119. package/src/orchestrate/adapters/gemini-stream.ts +166 -166
  120. package/src/orchestrate/adapters/gemini.ts +42 -42
  121. package/src/orchestrate/adapters/index.ts +73 -73
  122. package/src/orchestrate/adapters/opencode-stream.ts +143 -143
  123. package/src/orchestrate/adapters/opencode.ts +47 -47
  124. package/src/orchestrate/adapters/spawn-helper.ts +286 -286
  125. package/src/orchestrate/adapters/types.ts +77 -77
  126. package/src/orchestrate/capability-router.ts +284 -284
  127. package/src/orchestrate/context-compact.ts +188 -188
  128. package/src/orchestrate/cost-tracker.ts +219 -219
  129. package/src/orchestrate/error-recovery.ts +191 -191
  130. package/src/orchestrate/evidence.ts +140 -140
  131. package/src/orchestrate/ledger.ts +110 -110
  132. package/src/orchestrate/memorix-bridge.ts +380 -380
  133. package/src/orchestrate/output-budget.ts +80 -80
  134. package/src/orchestrate/permission.ts +152 -152
  135. package/src/orchestrate/pipeline-trace.ts +131 -131
  136. package/src/orchestrate/prompt-builder.ts +155 -155
  137. package/src/orchestrate/ring-buffer.ts +37 -37
  138. package/src/orchestrate/task-graph.ts +389 -389
  139. package/src/orchestrate/verify-gate.ts +219 -219
  140. package/src/orchestrate/worktree.ts +232 -232
  141. package/src/project/aliases.ts +374 -374
  142. package/src/project/detector.ts +268 -268
  143. package/src/rules/adapters/claude-code.ts +99 -99
  144. package/src/rules/adapters/codex.ts +97 -97
  145. package/src/rules/adapters/copilot.ts +124 -124
  146. package/src/rules/adapters/cursor.ts +114 -114
  147. package/src/rules/adapters/kiro.ts +126 -126
  148. package/src/rules/adapters/trae.ts +56 -56
  149. package/src/rules/adapters/windsurf.ts +83 -83
  150. package/src/rules/syncer.ts +235 -235
  151. package/src/sdk.ts +327 -327
  152. package/src/search/intent-detector.ts +289 -289
  153. package/src/search/query-expansion.ts +52 -52
  154. package/src/server/formation-timeout.ts +27 -27
  155. package/src/server.ts +3 -0
  156. package/src/skills/mini-skills.ts +386 -386
  157. package/src/store/chat-store.ts +119 -119
  158. package/src/store/file-lock.ts +100 -100
  159. package/src/store/graph-store.ts +249 -249
  160. package/src/store/mini-skill-store.ts +349 -349
  161. package/src/store/obs-store.ts +255 -255
  162. package/src/store/orama-store.ts +15 -8
  163. package/src/store/persistence-json.ts +212 -212
  164. package/src/store/persistence.ts +291 -291
  165. package/src/store/project-affinity.ts +195 -195
  166. package/src/store/session-store.ts +259 -259
  167. package/src/store/sqlite-store.ts +339 -339
  168. package/src/team/event-bus.ts +76 -76
  169. package/src/team/file-locks.ts +173 -173
  170. package/src/team/handoff.ts +167 -167
  171. package/src/team/messages.ts +203 -203
  172. package/src/team/poll.ts +132 -132
  173. package/src/team/tasks.ts +211 -211
  174. package/src/wiki/generator.ts +237 -237
  175. package/src/wiki/knowledge-graph.ts +334 -334
  176. package/src/wiki/types.ts +85 -85
  177. package/src/workspace/mcp-adapters/codex.ts +191 -191
  178. package/src/workspace/mcp-adapters/copilot.ts +105 -105
  179. package/src/workspace/mcp-adapters/cursor.ts +53 -53
  180. package/src/workspace/mcp-adapters/kiro.ts +64 -64
  181. package/src/workspace/mcp-adapters/opencode.ts +123 -123
  182. package/src/workspace/mcp-adapters/trae.ts +134 -134
  183. package/src/workspace/mcp-adapters/windsurf.ts +91 -91
  184. package/src/workspace/sanitizer.ts +60 -60
  185. package/src/workspace/workflow-sync.ts +131 -131
@@ -1,248 +1,248 @@
1
- /**
2
- * LLM Quality Enhancements
3
- *
4
- * Premium memory quality features powered by LLM:
5
- * 1. Narrative Compression — compress verbose narratives into concise core knowledge
6
- * 2. Search Reranking — rerank search results by relevance to current task context
7
- *
8
- * Both features gracefully degrade: when LLM is not configured, they return
9
- * the original data unchanged.
10
- *
11
- * Performance targets:
12
- * - Compression: ~60% token reduction per stored memory
13
- * - Reranking: ~40% improvement in Top-5 precision
14
- */
15
-
16
- import { callLLM, isLLMEnabled } from './provider.js';
17
-
18
- // ── Narrative Compression ────────────────────────────────────────
19
-
20
- const COMPRESS_PROMPT = `You are a memory compression engine for a coding assistant.
21
-
22
- Compress the given narrative while preserving ALL technical facts and reasoning.
23
-
24
- Rules:
25
- - Remove: filler words, debugging journey, repeated info already in facts
26
- - Keep: specific values, file paths, error messages, version numbers, config keys, causal relationships, design reasoning
27
- - Merge related points into dense sentences
28
- - If facts are provided separately, do NOT repeat them in the compressed narrative
29
- - Output the compressed text ONLY, no explanation or wrapper
30
-
31
- Examples:
32
- Input: "我在调试过程中发现JWT token的refresh机制存在问题,具体来说是因为服务端没有实现自动续签,导致用户在24小时后会遇到静默的认证失败,之前我一直以为是网络问题但后来排查发现是token过期了"
33
- Output: "JWT refresh无自动续签→24h后静默认证失败(非网络问题)"
34
-
35
- Input: "Final deployment model for shadcn-blog is stable: GitHub Actions build locally, SCP artifacts to VPS, systemd manages the process. Docker was considered but rejected due to complexity overhead for a simple blog. The whole pipeline takes about 2 minutes from push to live."
36
- Output: "shadcn-blog部署: GH Actions构建→SCP到VPS→systemd管理, 弃Docker(复杂度过高), push到上线~2min"`;
37
-
38
- /** Gentler prompt for high-value types where reasoning context matters */
39
- const COMPRESS_PROMPT_GENTLE = `You are a memory compression engine for a coding assistant.
40
-
41
- Lightly compress the given narrative — preserve reasoning, trade-offs, and "why" context.
42
-
43
- Rules:
44
- - Only remove: obvious filler, debugging detours, info already in the separate facts list
45
- - PRESERVE: design reasoning, rejected alternatives, trade-off analysis, causal chains
46
- - Aim for ~70-80% of original length, NOT aggressive compression
47
- - If facts are provided separately, do NOT repeat them in the compressed narrative
48
- - Output the compressed text ONLY, no explanation or wrapper`;
49
-
50
- /**
51
- * Compress a narrative to its essential core using LLM.
52
- *
53
- * Returns the original narrative if:
54
- * - LLM is not enabled
55
- * - Narrative is already short (≤80 chars)
56
- * - Narrative is already concise (commands, file paths, git operations)
57
- * - LLM call fails
58
- */
59
- export async function compressNarrative(
60
- narrative: string,
61
- facts?: string[],
62
- type?: string,
63
- ): Promise<{ compressed: string; saved: number; usedLLM: boolean }> {
64
- const originalTokens = estimateTokens(narrative);
65
-
66
- // Skip compression for short narratives (≤150 chars is already concise)
67
- if (!isLLMEnabled() || narrative.length <= 150) {
68
- return { compressed: narrative, saved: 0, usedLLM: false };
69
- }
70
-
71
- // Skip compression for already-concise content that LLM can't meaningfully compress
72
- if (shouldSkipCompression(narrative, type)) {
73
- return { compressed: narrative, saved: 0, usedLLM: false };
74
- }
75
-
76
- try {
77
- const factsContext = facts && facts.length > 0
78
- ? `\n\nSeparate facts (already stored, don't repeat): ${facts.join('; ')}`
79
- : '';
80
-
81
- // Use gentler compression for high-value types where reasoning matters
82
- const HIGH_VALUE_TYPES = new Set(['decision', 'trade-off', 'why-it-exists', 'how-it-works']);
83
- const prompt = (type && HIGH_VALUE_TYPES.has(type)) ? COMPRESS_PROMPT_GENTLE : COMPRESS_PROMPT;
84
- const response = await callLLM(prompt, narrative + factsContext);
85
- const compressed = response.content.trim();
86
-
87
- // Sanity check: compressed should be shorter and non-empty
88
- if (!compressed || compressed.length >= narrative.length) {
89
- return { compressed: narrative, saved: 0, usedLLM: true };
90
- }
91
-
92
- const compressedTokens = estimateTokens(compressed);
93
- return {
94
- compressed,
95
- saved: originalTokens - compressedTokens,
96
- usedLLM: true,
97
- };
98
- } catch {
99
- return { compressed: narrative, saved: 0, usedLLM: false };
100
- }
101
- }
102
-
103
- // ── Search Reranking ─────────────────────────────────────────────
104
-
105
- /** Minimal search result for reranking */
106
- export interface RerankCandidate {
107
- id: string;
108
- title: string;
109
- type: string;
110
- score: number;
111
- narrative?: string;
112
- }
113
-
114
- const RERANK_PROMPT = `You are a memory relevance ranker for a coding assistant.
115
-
116
- Given a QUERY (what the user/agent is looking for) and a list of CANDIDATE memories,
117
- rerank them by relevance to the query.
118
-
119
- Rules:
120
- - Consider semantic relevance, not just keyword overlap
121
- - Gotchas and decisions related to the query topic should rank higher
122
- - Recent problem-solutions for the same component should rank higher
123
- - Command or audit-log memories (titles starting with "Ran:" or "Command:") should rank lower for natural-language questions unless the query is explicitly about commands, scripts, or audit history
124
- - Generic or loosely related memories should rank lower
125
- - Output ONLY a JSON array of IDs in order of relevance (most relevant first)
126
- - Include ALL candidate IDs, just reorder them
127
-
128
- Example output: ["r1", "r3", "r2"]`;
129
-
130
- /**
131
- * Rerank search results using LLM contextual understanding.
132
- *
133
- * Takes Orama's initial ranking and improves it by considering
134
- * semantic relevance to the current query/task context.
135
- *
136
- * Returns original order if LLM is not enabled or call fails.
137
- */
138
- export async function rerankResults(
139
- query: string,
140
- candidates: RerankCandidate[],
141
- ): Promise<{ reranked: RerankCandidate[]; usedLLM: boolean }> {
142
- // Skip if too few results or LLM not available
143
- if (!isLLMEnabled() || candidates.length <= 2) {
144
- return { reranked: candidates, usedLLM: false };
145
- }
146
-
147
- // Only rerank top-N to save LLM tokens (reranking 20+ is wasteful)
148
- const MAX_RERANK = 10;
149
- const toRerank = candidates.slice(0, MAX_RERANK);
150
- const rest = candidates.slice(MAX_RERANK);
151
-
152
- try {
153
- const candidateList = toRerank.map(c =>
154
- `[ID: ${c.id}] (${c.type}) ${c.title}${c.narrative ? ` — ${c.narrative.substring(0, 100)}` : ''}`,
155
- ).join('\n');
156
-
157
- const response = await callLLM(RERANK_PROMPT, `QUERY: ${query}\n\nCANDIDATES:\n${candidateList}`);
158
-
159
- // Parse response — handle markdown code blocks
160
- let content = response.content.trim();
161
- if (content.startsWith('```')) {
162
- content = content.replace(/^```(?:json)?\s*/, '').replace(/\s*```$/, '');
163
- }
164
-
165
- const rankedIds = JSON.parse(content) as string[];
166
-
167
- // Validate: must be an array of IDs matching our candidates
168
- if (!Array.isArray(rankedIds) || rankedIds.length === 0) {
169
- return { reranked: candidates, usedLLM: true };
170
- }
171
-
172
- // Build reranked list preserving original scores for display
173
- const idMap = new Map(toRerank.map(c => [c.id, c]));
174
- const reranked: RerankCandidate[] = [];
175
- const seen = new Set<string>();
176
-
177
- // Add IDs in LLM-reranked order
178
- for (const id of rankedIds) {
179
- const candidate = idMap.get(id);
180
- if (candidate && !seen.has(id)) {
181
- reranked.push(candidate);
182
- seen.add(id);
183
- }
184
- }
185
-
186
- // Add any candidates the LLM missed (safety: never lose results)
187
- for (const c of toRerank) {
188
- if (!seen.has(c.id)) {
189
- reranked.push(c);
190
- }
191
- }
192
-
193
- // Append non-reranked tail
194
- reranked.push(...rest);
195
-
196
- return { reranked, usedLLM: true };
197
- } catch {
198
- return { reranked: candidates, usedLLM: false };
199
- }
200
- }
201
-
202
- // ── Smart Compression Filtering ──────────────────────────────────
203
-
204
- /** Patterns that indicate already-concise content not worth compressing */
205
- const SKIP_PATTERNS = [
206
- /^(?:Command|Run|Execute):\s/i, // Shell commands
207
- /^(?:File|Edit|Changed):\s/i, // File change descriptions
208
- /^git\s+(?:add|commit|push|pull|log)/i, // Git operations
209
- /^(?:npm|npx|pnpm|yarn|bun)\s/i, // Package manager commands
210
- /^(?:Remove-Item|New-Item|Set-Content)/i, // PowerShell commands
211
- /^[A-Za-z]:\\[\w\\]/, // Windows file paths
212
- /^\/(?:usr|home|var|etc|opt)\//, // Unix file paths
213
- ];
214
-
215
- /** Low-value observation types that hooks auto-capture (usually already terse) */
216
- const LOW_COMPRESSION_TYPES = new Set(['what-changed', 'discovery', 'session-request']);
217
-
218
- /**
219
- * Determine if a narrative should skip LLM compression.
220
- *
221
- * Skip when:
222
- * - Content starts with command/path patterns (already structured, not prose)
223
- * - Type is hooks-auto-captured AND narrative is relatively short
224
- * - Narrative is mostly code/paths (high ratio of special chars)
225
- */
226
- function shouldSkipCompression(narrative: string, type?: string): boolean {
227
- // Skip command/path-like content
228
- const firstLine = narrative.split('\n')[0];
229
- if (SKIP_PATTERNS.some(p => p.test(firstLine))) return true;
230
-
231
- // Skip short auto-captured observations (hooks produce terse what-changed)
232
- if (type && LOW_COMPRESSION_TYPES.has(type) && narrative.length < 200) return true;
233
-
234
- // Skip if narrative is mostly code/structured data (high special char ratio)
235
- const specialChars = (narrative.match(/[{}()\[\]<>:;=|\\\/\-_\.@#$%^&*+~`"']/g) || []).length;
236
- if (specialChars / narrative.length > 0.35) return true;
237
-
238
- return false;
239
- }
240
-
241
- // ── Utility ──────────────────────────────────────────────────────
242
-
243
- /** Rough token estimate: ~4 chars per token for English, ~2 for CJK */
244
- function estimateTokens(text: string): number {
245
- const cjkChars = (text.match(/[\u4e00-\u9fff\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/g) || []).length;
246
- const otherChars = text.length - cjkChars;
247
- return Math.ceil(cjkChars / 1.5 + otherChars / 4);
248
- }
1
+ /**
2
+ * LLM Quality Enhancements
3
+ *
4
+ * Premium memory quality features powered by LLM:
5
+ * 1. Narrative Compression — compress verbose narratives into concise core knowledge
6
+ * 2. Search Reranking — rerank search results by relevance to current task context
7
+ *
8
+ * Both features gracefully degrade: when LLM is not configured, they return
9
+ * the original data unchanged.
10
+ *
11
+ * Performance targets:
12
+ * - Compression: ~60% token reduction per stored memory
13
+ * - Reranking: ~40% improvement in Top-5 precision
14
+ */
15
+
16
+ import { callLLM, isLLMEnabled } from './provider.js';
17
+
18
+ // ── Narrative Compression ────────────────────────────────────────
19
+
20
+ const COMPRESS_PROMPT = `You are a memory compression engine for a coding assistant.
21
+
22
+ Compress the given narrative while preserving ALL technical facts and reasoning.
23
+
24
+ Rules:
25
+ - Remove: filler words, debugging journey, repeated info already in facts
26
+ - Keep: specific values, file paths, error messages, version numbers, config keys, causal relationships, design reasoning
27
+ - Merge related points into dense sentences
28
+ - If facts are provided separately, do NOT repeat them in the compressed narrative
29
+ - Output the compressed text ONLY, no explanation or wrapper
30
+
31
+ Examples:
32
+ Input: "我在调试过程中发现JWT token的refresh机制存在问题,具体来说是因为服务端没有实现自动续签,导致用户在24小时后会遇到静默的认证失败,之前我一直以为是网络问题但后来排查发现是token过期了"
33
+ Output: "JWT refresh无自动续签→24h后静默认证失败(非网络问题)"
34
+
35
+ Input: "Final deployment model for shadcn-blog is stable: GitHub Actions build locally, SCP artifacts to VPS, systemd manages the process. Docker was considered but rejected due to complexity overhead for a simple blog. The whole pipeline takes about 2 minutes from push to live."
36
+ Output: "shadcn-blog部署: GH Actions构建→SCP到VPS→systemd管理, 弃Docker(复杂度过高), push到上线~2min"`;
37
+
38
+ /** Gentler prompt for high-value types where reasoning context matters */
39
+ const COMPRESS_PROMPT_GENTLE = `You are a memory compression engine for a coding assistant.
40
+
41
+ Lightly compress the given narrative — preserve reasoning, trade-offs, and "why" context.
42
+
43
+ Rules:
44
+ - Only remove: obvious filler, debugging detours, info already in the separate facts list
45
+ - PRESERVE: design reasoning, rejected alternatives, trade-off analysis, causal chains
46
+ - Aim for ~70-80% of original length, NOT aggressive compression
47
+ - If facts are provided separately, do NOT repeat them in the compressed narrative
48
+ - Output the compressed text ONLY, no explanation or wrapper`;
49
+
50
+ /**
51
+ * Compress a narrative to its essential core using LLM.
52
+ *
53
+ * Returns the original narrative if:
54
+ * - LLM is not enabled
55
+ * - Narrative is already short (≤80 chars)
56
+ * - Narrative is already concise (commands, file paths, git operations)
57
+ * - LLM call fails
58
+ */
59
+ export async function compressNarrative(
60
+ narrative: string,
61
+ facts?: string[],
62
+ type?: string,
63
+ ): Promise<{ compressed: string; saved: number; usedLLM: boolean }> {
64
+ const originalTokens = estimateTokens(narrative);
65
+
66
+ // Skip compression for short narratives (≤150 chars is already concise)
67
+ if (!isLLMEnabled() || narrative.length <= 150) {
68
+ return { compressed: narrative, saved: 0, usedLLM: false };
69
+ }
70
+
71
+ // Skip compression for already-concise content that LLM can't meaningfully compress
72
+ if (shouldSkipCompression(narrative, type)) {
73
+ return { compressed: narrative, saved: 0, usedLLM: false };
74
+ }
75
+
76
+ try {
77
+ const factsContext = facts && facts.length > 0
78
+ ? `\n\nSeparate facts (already stored, don't repeat): ${facts.join('; ')}`
79
+ : '';
80
+
81
+ // Use gentler compression for high-value types where reasoning matters
82
+ const HIGH_VALUE_TYPES = new Set(['decision', 'trade-off', 'why-it-exists', 'how-it-works']);
83
+ const prompt = (type && HIGH_VALUE_TYPES.has(type)) ? COMPRESS_PROMPT_GENTLE : COMPRESS_PROMPT;
84
+ const response = await callLLM(prompt, narrative + factsContext);
85
+ const compressed = response.content.trim();
86
+
87
+ // Sanity check: compressed should be shorter and non-empty
88
+ if (!compressed || compressed.length >= narrative.length) {
89
+ return { compressed: narrative, saved: 0, usedLLM: true };
90
+ }
91
+
92
+ const compressedTokens = estimateTokens(compressed);
93
+ return {
94
+ compressed,
95
+ saved: originalTokens - compressedTokens,
96
+ usedLLM: true,
97
+ };
98
+ } catch {
99
+ return { compressed: narrative, saved: 0, usedLLM: false };
100
+ }
101
+ }
102
+
103
+ // ── Search Reranking ─────────────────────────────────────────────
104
+
105
+ /** Minimal search result for reranking */
106
+ export interface RerankCandidate {
107
+ id: string;
108
+ title: string;
109
+ type: string;
110
+ score: number;
111
+ narrative?: string;
112
+ }
113
+
114
+ const RERANK_PROMPT = `You are a memory relevance ranker for a coding assistant.
115
+
116
+ Given a QUERY (what the user/agent is looking for) and a list of CANDIDATE memories,
117
+ rerank them by relevance to the query.
118
+
119
+ Rules:
120
+ - Consider semantic relevance, not just keyword overlap
121
+ - Gotchas and decisions related to the query topic should rank higher
122
+ - Recent problem-solutions for the same component should rank higher
123
+ - Command or audit-log memories (titles starting with "Ran:" or "Command:") should rank lower for natural-language questions unless the query is explicitly about commands, scripts, or audit history
124
+ - Generic or loosely related memories should rank lower
125
+ - Output ONLY a JSON array of IDs in order of relevance (most relevant first)
126
+ - Include ALL candidate IDs, just reorder them
127
+
128
+ Example output: ["r1", "r3", "r2"]`;
129
+
130
+ /**
131
+ * Rerank search results using LLM contextual understanding.
132
+ *
133
+ * Takes Orama's initial ranking and improves it by considering
134
+ * semantic relevance to the current query/task context.
135
+ *
136
+ * Returns original order if LLM is not enabled or call fails.
137
+ */
138
+ export async function rerankResults(
139
+ query: string,
140
+ candidates: RerankCandidate[],
141
+ ): Promise<{ reranked: RerankCandidate[]; usedLLM: boolean }> {
142
+ // Skip if too few results or LLM not available
143
+ if (!isLLMEnabled() || candidates.length <= 2) {
144
+ return { reranked: candidates, usedLLM: false };
145
+ }
146
+
147
+ // Only rerank top-N to save LLM tokens (reranking 20+ is wasteful)
148
+ const MAX_RERANK = 10;
149
+ const toRerank = candidates.slice(0, MAX_RERANK);
150
+ const rest = candidates.slice(MAX_RERANK);
151
+
152
+ try {
153
+ const candidateList = toRerank.map(c =>
154
+ `[ID: ${c.id}] (${c.type}) ${c.title}${c.narrative ? ` — ${c.narrative.substring(0, 100)}` : ''}`,
155
+ ).join('\n');
156
+
157
+ const response = await callLLM(RERANK_PROMPT, `QUERY: ${query}\n\nCANDIDATES:\n${candidateList}`);
158
+
159
+ // Parse response — handle markdown code blocks
160
+ let content = response.content.trim();
161
+ if (content.startsWith('```')) {
162
+ content = content.replace(/^```(?:json)?\s*/, '').replace(/\s*```$/, '');
163
+ }
164
+
165
+ const rankedIds = JSON.parse(content) as string[];
166
+
167
+ // Validate: must be an array of IDs matching our candidates
168
+ if (!Array.isArray(rankedIds) || rankedIds.length === 0) {
169
+ return { reranked: candidates, usedLLM: true };
170
+ }
171
+
172
+ // Build reranked list preserving original scores for display
173
+ const idMap = new Map(toRerank.map(c => [c.id, c]));
174
+ const reranked: RerankCandidate[] = [];
175
+ const seen = new Set<string>();
176
+
177
+ // Add IDs in LLM-reranked order
178
+ for (const id of rankedIds) {
179
+ const candidate = idMap.get(id);
180
+ if (candidate && !seen.has(id)) {
181
+ reranked.push(candidate);
182
+ seen.add(id);
183
+ }
184
+ }
185
+
186
+ // Add any candidates the LLM missed (safety: never lose results)
187
+ for (const c of toRerank) {
188
+ if (!seen.has(c.id)) {
189
+ reranked.push(c);
190
+ }
191
+ }
192
+
193
+ // Append non-reranked tail
194
+ reranked.push(...rest);
195
+
196
+ return { reranked, usedLLM: true };
197
+ } catch {
198
+ return { reranked: candidates, usedLLM: false };
199
+ }
200
+ }
201
+
202
+ // ── Smart Compression Filtering ──────────────────────────────────
203
+
204
+ /** Patterns that indicate already-concise content not worth compressing */
205
+ const SKIP_PATTERNS = [
206
+ /^(?:Command|Run|Execute):\s/i, // Shell commands
207
+ /^(?:File|Edit|Changed):\s/i, // File change descriptions
208
+ /^git\s+(?:add|commit|push|pull|log)/i, // Git operations
209
+ /^(?:npm|npx|pnpm|yarn|bun)\s/i, // Package manager commands
210
+ /^(?:Remove-Item|New-Item|Set-Content)/i, // PowerShell commands
211
+ /^[A-Za-z]:\\[\w\\]/, // Windows file paths
212
+ /^\/(?:usr|home|var|etc|opt)\//, // Unix file paths
213
+ ];
214
+
215
+ /** Low-value observation types that hooks auto-capture (usually already terse) */
216
+ const LOW_COMPRESSION_TYPES = new Set(['what-changed', 'discovery', 'session-request']);
217
+
218
+ /**
219
+ * Determine if a narrative should skip LLM compression.
220
+ *
221
+ * Skip when:
222
+ * - Content starts with command/path patterns (already structured, not prose)
223
+ * - Type is hooks-auto-captured AND narrative is relatively short
224
+ * - Narrative is mostly code/paths (high ratio of special chars)
225
+ */
226
+ function shouldSkipCompression(narrative: string, type?: string): boolean {
227
+ // Skip command/path-like content
228
+ const firstLine = narrative.split('\n')[0];
229
+ if (SKIP_PATTERNS.some(p => p.test(firstLine))) return true;
230
+
231
+ // Skip short auto-captured observations (hooks produce terse what-changed)
232
+ if (type && LOW_COMPRESSION_TYPES.has(type) && narrative.length < 200) return true;
233
+
234
+ // Skip if narrative is mostly code/structured data (high special char ratio)
235
+ const specialChars = (narrative.match(/[{}()\[\]<>:;=|\\\/\-_\.@#$%^&*+~`"']/g) || []).length;
236
+ if (specialChars / narrative.length > 0.35) return true;
237
+
238
+ return false;
239
+ }
240
+
241
+ // ── Utility ──────────────────────────────────────────────────────
242
+
243
+ /** Rough token estimate: ~4 chars per token for English, ~2 for CJK */
244
+ function estimateTokens(text: string): number {
245
+ const cjkChars = (text.match(/[\u4e00-\u9fff\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/g) || []).length;
246
+ const otherChars = text.length - cjkChars;
247
+ return Math.ceil(cjkChars / 1.5 + otherChars / 4);
248
+ }