memorix 1.1.7 → 1.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/CLAUDE.md +6 -1
  3. package/README.md +21 -0
  4. package/README.zh-CN.md +21 -0
  5. package/TEAM.md +86 -86
  6. package/dist/cli/index.js +852 -214
  7. package/dist/cli/index.js.map +1 -1
  8. package/dist/dashboard/static/index.html +201 -201
  9. package/dist/dashboard/static/style.css +3584 -3584
  10. package/dist/index.js +129 -62
  11. package/dist/index.js.map +1 -1
  12. package/dist/memcode-runtime/CHANGELOG.md +22 -0
  13. package/dist/memcode-runtime/package.json +4 -4
  14. package/dist/sdk.js +129 -62
  15. package/dist/sdk.js.map +1 -1
  16. package/docs/AGENT_OPERATOR_PLAYBOOK.md +18 -0
  17. package/docs/API_REFERENCE.md +2 -0
  18. package/docs/CONFIGURATION.md +18 -0
  19. package/docs/DESIGN_DECISIONS.md +357 -357
  20. package/docs/SETUP.md +10 -0
  21. package/docs/dev-log/progress.txt +23 -30
  22. package/package.json +1 -1
  23. package/src/audit/index.ts +156 -156
  24. package/src/cli/commands/agent-integrations.ts +623 -0
  25. package/src/cli/commands/audit-list.ts +89 -89
  26. package/src/cli/commands/background.ts +659 -659
  27. package/src/cli/commands/cleanup.ts +255 -255
  28. package/src/cli/commands/codegraph.ts +4 -0
  29. package/src/cli/commands/config-get.ts +9 -2
  30. package/src/cli/commands/doctor.ts +26 -0
  31. package/src/cli/commands/formation.ts +48 -48
  32. package/src/cli/commands/git-hook-install.ts +111 -111
  33. package/src/cli/commands/handoff.ts +66 -66
  34. package/src/cli/commands/hooks-status.ts +63 -63
  35. package/src/cli/commands/ingest-commit.ts +153 -153
  36. package/src/cli/commands/ingest-image.ts +73 -73
  37. package/src/cli/commands/ingest-log.ts +180 -180
  38. package/src/cli/commands/ingest.ts +44 -44
  39. package/src/cli/commands/integrate-shared.ts +15 -15
  40. package/src/cli/commands/lock.ts +96 -96
  41. package/src/cli/commands/message.ts +121 -121
  42. package/src/cli/commands/poll.ts +70 -70
  43. package/src/cli/commands/purge-all-memory.ts +85 -85
  44. package/src/cli/commands/purge-project-memory.ts +83 -83
  45. package/src/cli/commands/reasoning.ts +132 -132
  46. package/src/cli/commands/repair.ts +60 -0
  47. package/src/cli/commands/retention.ts +108 -108
  48. package/src/cli/commands/serve-shared.ts +118 -118
  49. package/src/cli/commands/setup.ts +3 -3
  50. package/src/cli/commands/skills.ts +123 -123
  51. package/src/cli/commands/task.ts +192 -192
  52. package/src/cli/commands/transfer.ts +73 -73
  53. package/src/cli/commands/uninstall-project-artifacts.ts +85 -85
  54. package/src/cli/index.ts +3 -1
  55. package/src/cli/tui/ChatView.tsx +234 -234
  56. package/src/cli/tui/CommandBar.tsx +312 -312
  57. package/src/cli/tui/ContextRail.tsx +118 -118
  58. package/src/cli/tui/HeaderBar.tsx +72 -72
  59. package/src/cli/tui/LogoBanner.tsx +51 -51
  60. package/src/cli/tui/Panels.tsx +632 -632
  61. package/src/cli/tui/Sidebar.tsx +179 -179
  62. package/src/cli/tui/chat-service.ts +742 -742
  63. package/src/cli/tui/data.ts +547 -547
  64. package/src/cli/tui/index.ts +41 -41
  65. package/src/cli/tui/markdown-render.tsx +371 -371
  66. package/src/cli/tui/theme.ts +178 -178
  67. package/src/cli/tui/use-mouse.ts +157 -157
  68. package/src/cli/tui/useNavigation.ts +56 -56
  69. package/src/cli/update-checker.ts +211 -211
  70. package/src/cli/version.ts +7 -7
  71. package/src/cli/workbench.ts +1 -1
  72. package/src/codegraph/auto-context.ts +6 -0
  73. package/src/codegraph/context-pack.ts +7 -6
  74. package/src/codegraph/exclude.ts +47 -0
  75. package/src/codegraph/lite-provider.ts +5 -24
  76. package/src/codegraph/project-context.ts +13 -15
  77. package/src/compact/token-budget.ts +74 -74
  78. package/src/config/behavior.ts +59 -59
  79. package/src/config/resolved-config.ts +6 -0
  80. package/src/config/toml-loader.ts +4 -0
  81. package/src/config/yaml-loader.ts +7 -0
  82. package/src/dashboard/project-classification.ts +64 -64
  83. package/src/dashboard/static/index.html +201 -201
  84. package/src/dashboard/static/style.css +3584 -3584
  85. package/src/embedding/fastembed-provider.ts +142 -142
  86. package/src/embedding/transformers-provider.ts +111 -111
  87. package/src/git/extractor.ts +209 -209
  88. package/src/git/hooks-path.ts +85 -85
  89. package/src/git/noise-filter.ts +210 -210
  90. package/src/hooks/installers/index.ts +4 -4
  91. package/src/hooks/official-skills.ts +1 -1
  92. package/src/hooks/pattern-detector.ts +173 -173
  93. package/src/hooks/rules/memorix-agent-rules.md +2 -2
  94. package/src/hooks/significance-filter.ts +250 -250
  95. package/src/llm/memory-manager.ts +328 -328
  96. package/src/llm/provider.ts +885 -885
  97. package/src/llm/quality.ts +248 -248
  98. package/src/memory/attribution-guard.ts +249 -249
  99. package/src/memory/auto-relations.ts +107 -107
  100. package/src/memory/consolidation.ts +302 -302
  101. package/src/memory/disclosure-policy.ts +141 -141
  102. package/src/memory/entity-extractor.ts +197 -197
  103. package/src/memory/formation/evaluate.ts +217 -217
  104. package/src/memory/formation/extract.ts +361 -361
  105. package/src/memory/formation/index.ts +417 -417
  106. package/src/memory/formation/resolve.ts +344 -344
  107. package/src/memory/formation/types.ts +315 -315
  108. package/src/memory/freshness.ts +122 -122
  109. package/src/memory/graph.ts +197 -197
  110. package/src/memory/refs.ts +94 -94
  111. package/src/memory/retention.ts +433 -433
  112. package/src/memory/secret-filter.ts +79 -79
  113. package/src/memory/session.ts +523 -523
  114. package/src/multimodal/image-loader.ts +143 -143
  115. package/src/orchestrate/adapters/claude-stream.ts +192 -192
  116. package/src/orchestrate/adapters/claude.ts +111 -111
  117. package/src/orchestrate/adapters/codex-stream.ts +134 -134
  118. package/src/orchestrate/adapters/codex.ts +41 -41
  119. package/src/orchestrate/adapters/gemini-stream.ts +166 -166
  120. package/src/orchestrate/adapters/gemini.ts +42 -42
  121. package/src/orchestrate/adapters/index.ts +73 -73
  122. package/src/orchestrate/adapters/opencode-stream.ts +143 -143
  123. package/src/orchestrate/adapters/opencode.ts +47 -47
  124. package/src/orchestrate/adapters/spawn-helper.ts +286 -286
  125. package/src/orchestrate/adapters/types.ts +77 -77
  126. package/src/orchestrate/capability-router.ts +284 -284
  127. package/src/orchestrate/context-compact.ts +188 -188
  128. package/src/orchestrate/cost-tracker.ts +219 -219
  129. package/src/orchestrate/error-recovery.ts +191 -191
  130. package/src/orchestrate/evidence.ts +140 -140
  131. package/src/orchestrate/ledger.ts +110 -110
  132. package/src/orchestrate/memorix-bridge.ts +380 -380
  133. package/src/orchestrate/output-budget.ts +80 -80
  134. package/src/orchestrate/permission.ts +152 -152
  135. package/src/orchestrate/pipeline-trace.ts +131 -131
  136. package/src/orchestrate/prompt-builder.ts +155 -155
  137. package/src/orchestrate/ring-buffer.ts +37 -37
  138. package/src/orchestrate/task-graph.ts +389 -389
  139. package/src/orchestrate/verify-gate.ts +219 -219
  140. package/src/orchestrate/worktree.ts +232 -232
  141. package/src/project/aliases.ts +374 -374
  142. package/src/project/detector.ts +268 -268
  143. package/src/rules/adapters/claude-code.ts +99 -99
  144. package/src/rules/adapters/codex.ts +97 -97
  145. package/src/rules/adapters/copilot.ts +124 -124
  146. package/src/rules/adapters/cursor.ts +114 -114
  147. package/src/rules/adapters/kiro.ts +126 -126
  148. package/src/rules/adapters/trae.ts +56 -56
  149. package/src/rules/adapters/windsurf.ts +83 -83
  150. package/src/rules/syncer.ts +235 -235
  151. package/src/sdk.ts +327 -327
  152. package/src/search/intent-detector.ts +289 -289
  153. package/src/search/query-expansion.ts +52 -52
  154. package/src/server/formation-timeout.ts +27 -27
  155. package/src/server.ts +3 -0
  156. package/src/skills/mini-skills.ts +386 -386
  157. package/src/store/chat-store.ts +119 -119
  158. package/src/store/file-lock.ts +100 -100
  159. package/src/store/graph-store.ts +249 -249
  160. package/src/store/mini-skill-store.ts +349 -349
  161. package/src/store/obs-store.ts +255 -255
  162. package/src/store/orama-store.ts +15 -8
  163. package/src/store/persistence-json.ts +212 -212
  164. package/src/store/persistence.ts +291 -291
  165. package/src/store/project-affinity.ts +195 -195
  166. package/src/store/session-store.ts +259 -259
  167. package/src/store/sqlite-store.ts +339 -339
  168. package/src/team/event-bus.ts +76 -76
  169. package/src/team/file-locks.ts +173 -173
  170. package/src/team/handoff.ts +167 -167
  171. package/src/team/messages.ts +203 -203
  172. package/src/team/poll.ts +132 -132
  173. package/src/team/tasks.ts +211 -211
  174. package/src/wiki/generator.ts +237 -237
  175. package/src/wiki/knowledge-graph.ts +334 -334
  176. package/src/wiki/types.ts +85 -85
  177. package/src/workspace/mcp-adapters/codex.ts +191 -191
  178. package/src/workspace/mcp-adapters/copilot.ts +105 -105
  179. package/src/workspace/mcp-adapters/cursor.ts +53 -53
  180. package/src/workspace/mcp-adapters/kiro.ts +64 -64
  181. package/src/workspace/mcp-adapters/opencode.ts +123 -123
  182. package/src/workspace/mcp-adapters/trae.ts +134 -134
  183. package/src/workspace/mcp-adapters/windsurf.ts +91 -91
  184. package/src/workspace/sanitizer.ts +60 -60
  185. package/src/workspace/workflow-sync.ts +131 -131
@@ -1,366 +1,366 @@
1
- /**
2
- * Memory Formation — Stage 1: Extract
3
- *
4
- * Enriches raw memory input with system-extracted facts, normalized titles,
5
- * resolved entity names, and verified observation types.
6
- *
7
- * Rules-based mode (no LLM):
8
- * - Fact extraction: key-value patterns, error messages, version numbers, paths
9
- * - Title normalization: replace generic titles with first meaningful sentence
10
- * - Entity resolution: match against existing Knowledge Graph entities
11
- * - Type inference: verify type matches content signals
12
- */
13
-
14
- import type { ObservationType } from '../../types.js';
15
- import type { FormationInput, ExtractResult } from './types.js';
16
-
17
- // ── Fact Extraction Patterns ──────────────────────────────────────
18
-
19
- /** Patterns that extract structured facts from narrative text */
20
- const FACT_PATTERNS: Array<{ pattern: RegExp; format: (m: RegExpMatchArray) => string }> = [
21
- // Key: Value pairs (e.g., "Port: 3000", "Timeout = 60s")
22
- {
23
- pattern: /\b([A-Z][a-zA-Z_-]{2,30})\s*[:=]\s*([^\n,;]{2,60})/g,
24
- format: (m) => `${m[1]}: ${m[2].trim()}`,
25
- },
1
+ /**
2
+ * Memory Formation — Stage 1: Extract
3
+ *
4
+ * Enriches raw memory input with system-extracted facts, normalized titles,
5
+ * resolved entity names, and verified observation types.
6
+ *
7
+ * Rules-based mode (no LLM):
8
+ * - Fact extraction: key-value patterns, error messages, version numbers, paths
9
+ * - Title normalization: replace generic titles with first meaningful sentence
10
+ * - Entity resolution: match against existing Knowledge Graph entities
11
+ * - Type inference: verify type matches content signals
12
+ */
13
+
14
+ import type { ObservationType } from '../../types.js';
15
+ import type { FormationInput, ExtractResult } from './types.js';
16
+
17
+ // ── Fact Extraction Patterns ──────────────────────────────────────
18
+
19
+ /** Patterns that extract structured facts from narrative text */
20
+ const FACT_PATTERNS: Array<{ pattern: RegExp; format: (m: RegExpMatchArray) => string }> = [
21
+ // Key: Value pairs (e.g., "Port: 3000", "Timeout = 60s")
22
+ {
23
+ pattern: /\b([A-Z][a-zA-Z_-]{2,30})\s*[:=]\s*([^\n,;]{2,60})/g,
24
+ format: (m) => `${m[1]}: ${m[2].trim()}`,
25
+ },
26
26
  // Arrow notation (e.g., "MySQL -> PostgreSQL", "v1.0 -> v2.0")
27
27
  {
28
28
  pattern: /\b(\S{2,30})\s*(?:->|=>|>)\s*(\S{2,30})/g,
29
29
  format: (m) => `${m[1]} -> ${m[2]}`,
30
30
  },
31
- // Version numbers (e.g., "v1.2.3", "version 2.0")
32
- {
33
- pattern: /\b(?:v(?:ersion)?\s*)(\d+\.\d+(?:\.\d+)?(?:-[\w.]+)?)\b/gi,
34
- format: (m) => `Version: ${m[1]}`,
35
- },
36
- // Error messages (e.g., "Error: ...", "ERR_...")
37
- {
38
- pattern: /\b(?:Error|ERR|ENOENT|ECONNREFUSED|TypeError|RangeError|SyntaxError|ReferenceError)[:\s]+([^\n]{5,80})/gi,
39
- format: (m) => `Error: ${m[1].trim()}`,
40
- },
41
- // Port numbers in context
42
- {
43
- pattern: /\b(?:port|PORT)\s*[:=]?\s*(\d{2,5})\b/gi,
44
- format: (m) => `Port: ${m[1]}`,
45
- },
46
- // Environment variables
47
- {
48
- pattern: /\b([A-Z][A-Z0-9_]{3,30})\s*=\s*(\S{1,60})/g,
49
- format: (m) => `${m[1]}=${m[2]}`,
50
- },
51
- // npm/package versions (e.g., "react@18.2.0")
52
- {
53
- pattern: /\b([@a-z][\w./-]+)@(\d+\.\d+\.\d+(?:-[\w.]+)?)\b/g,
54
- format: (m) => `${m[1]}@${m[2]}`,
55
- },
56
- ];
57
-
58
- /** Patterns indicating generic/low-quality titles that should be improved */
59
- const GENERIC_TITLE_PATTERNS = [
60
- /^Updated \S+\.\w+$/i,
61
- /^Created \S+\.\w+$/i,
62
- /^Deleted \S+\.\w+$/i,
63
- /^Modified \S+\.\w+$/i,
64
- /^Changed \S+\.\w+$/i,
65
- /^Session activity/i,
66
- /^Activity \(/i,
67
- /^Used \w+$/i,
68
- /^Ran: /i,
69
- ];
70
-
71
- /** Content signals mapped to observation types */
72
- const TYPE_SIGNALS: Array<{ type: ObservationType; patterns: RegExp[] }> = [
73
- {
74
- type: 'problem-solution',
75
- patterns: [
76
- /\b(fix|fixed|bug|error|issue|crash|broken|resolved|workaround|patch)\b/i,
77
- /\b(修复|修正|解决|报错|崩溃|异常)\b/,
78
- ],
79
- },
80
- {
81
- type: 'gotcha',
82
- patterns: [
83
- /\b(gotcha|pitfall|trap|careful|warning|caveat|footgun|unexpected|beware)\b/i,
84
- /\b(坑|陷阱|注意|小心|踩坑)\b/,
85
- ],
86
- },
87
- {
88
- type: 'decision',
89
- patterns: [
90
- /\b(decided|chose|chosen|selected|adopted|rejected|evaluated|compared)\b/i,
91
- /\b(决定|选择|采用|弃用|对比|评估)\b/,
92
- ],
93
- },
94
- {
95
- type: 'what-changed',
96
- patterns: [
97
- /\b(changed|migrated|upgraded|refactored|replaced|renamed|moved|removed|added)\b/i,
98
- /\b(改|迁移|升级|重构|替换|重命名|删除|新增)\b/,
99
- ],
100
- },
101
- {
102
- type: 'how-it-works',
103
- patterns: [
104
- /\b(works by|architecture|mechanism|pipeline|flow|under the hood|internally)\b/i,
105
- /\b(原理|机制|流程|架构|内部)\b/,
106
- ],
107
- },
108
- {
109
- type: 'trade-off',
110
- patterns: [
111
- /\b(trade.?off|compromise|downside|cost|benefit|pro|con|versus|vs)\b/i,
112
- /\b(权衡|折中|代价|收益|优缺点)\b/,
113
- ],
114
- },
115
- ];
116
-
117
- // ── Extract Implementation ───────────────────────────────────────
118
-
119
- /**
120
- * Extract structured facts from narrative text using regex patterns.
121
- * Returns only facts not already present in the caller-provided list.
122
- */
123
- function extractFacts(narrative: string, existingFacts: string[]): string[] {
124
- const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
125
- const extracted: string[] = [];
126
- const seen = new Set<string>();
127
-
128
- for (const { pattern, format } of FACT_PATTERNS) {
129
- pattern.lastIndex = 0;
130
- let match: RegExpExecArray | null;
131
- while ((match = pattern.exec(narrative)) !== null) {
132
- const fact = format(match);
133
- const normalized = fact.toLowerCase().trim();
134
-
135
- // Skip if already provided by caller or already extracted
136
- if (existingLower.has(normalized) || seen.has(normalized)) continue;
137
-
138
- // Skip very short or very long facts
139
- if (fact.length < 5 || fact.length > 120) continue;
140
-
141
- seen.add(normalized);
142
- extracted.push(fact);
143
- }
144
- }
145
-
146
- return extracted.slice(0, 10); // Cap at 10 system-extracted facts
147
- }
148
-
149
- /**
150
- * Improve a generic title by extracting the first meaningful sentence
151
- * from the narrative.
152
- */
153
- function improveTitle(title: string, narrative: string): { title: string; improved: boolean } {
154
- const isGeneric = GENERIC_TITLE_PATTERNS.some(p => p.test(title));
155
- if (!isGeneric) return { title, improved: false };
156
-
157
- // Try to extract first meaningful sentence from narrative
158
- const sentences = narrative
159
- .replace(/```[\s\S]*?```/g, '') // Remove code blocks
160
- .split(/[.。!!?\n]/)
161
- .map(s => s.trim())
162
- .filter(s => s.length >= 15);
163
-
164
- if (sentences.length > 0) {
165
- return { title: sentences[0].slice(0, 60), improved: true };
166
- }
167
-
168
- return { title, improved: false };
169
- }
170
-
171
- /**
172
- * Resolve entity name against existing Knowledge Graph entities.
173
- * If a close match is found, use the canonical entity name.
174
- */
175
- function resolveEntity(
176
- entityName: string,
177
- existingEntities: string[],
178
- ): { entityName: string; resolved: boolean } {
179
- if (existingEntities.length === 0) return { entityName, resolved: false };
180
-
181
- const lower = entityName.toLowerCase().replace(/[-_]/g, '');
182
-
183
- for (const existing of existingEntities) {
184
- const existingLower = existing.toLowerCase().replace(/[-_]/g, '');
185
-
186
- // Exact match (case-insensitive, ignoring hyphens/underscores)
187
- if (lower === existingLower) {
188
- return { entityName: existing, resolved: existing !== entityName };
189
- }
190
-
191
- // Substring match: one contains the other (e.g., "auth" matches "auth-module")
192
- if (lower.length >= 3 && existingLower.length >= 3) {
193
- if (existingLower.includes(lower) || lower.includes(existingLower)) {
194
- // Prefer the longer (more specific) name
195
- const canonical = existing.length >= entityName.length ? existing : entityName;
196
- return { entityName: canonical, resolved: canonical !== entityName };
197
- }
198
- }
199
- }
200
-
201
- return { entityName, resolved: false };
202
- }
203
-
204
- /**
205
- * Verify observation type against content signals.
206
- * If content strongly suggests a different type, correct it.
207
- */
208
- function verifyType(
209
- declaredType: ObservationType,
210
- narrative: string,
211
- title: string,
212
- ): { type: ObservationType; corrected: boolean } {
213
- const content = `${title} ${narrative}`;
214
-
215
- // Score each type by counting individual keyword hits across all patterns
216
- const scores: Array<{ type: ObservationType; score: number }> = [];
217
- for (const { type, patterns } of TYPE_SIGNALS) {
218
- let score = 0;
219
- for (const p of patterns) {
220
- // Use matchAll to count individual keyword matches
221
- const regex = new RegExp(p.source, p.flags.includes('g') ? p.flags : p.flags + 'g');
222
- const matches = [...content.matchAll(regex)];
223
- score += matches.length;
224
- }
225
- if (score > 0) scores.push({ type, score });
226
- }
227
-
228
- if (scores.length === 0) return { type: declaredType, corrected: false };
229
-
230
- // Sort by score descending
231
- scores.sort((a, b) => b.score - a.score);
232
- const best = scores[0];
233
-
234
- // Only correct if the best match is significantly stronger than declared type
235
- // Requires: best type has >= 2 keyword hits AND declared type has 0 signals
236
- if (best.type !== declaredType && best.score >= 2) {
237
- const declaredScore = scores.find(s => s.type === declaredType)?.score ?? 0;
238
- if (declaredScore === 0) {
239
- return { type: best.type, corrected: true };
240
- }
241
- }
242
-
243
- return { type: declaredType, corrected: false };
244
- }
245
-
246
- // ── LLM Fact Extraction ─────────────────────────────────────────
247
-
248
- /** Prompt for LLM-based fact extraction (inspired by Mem0's approach) */
249
- const LLM_EXTRACT_PROMPT = `You are a Software Engineering Knowledge Extractor.
250
- Extract structured facts from the given development context.
251
-
252
- Focus on:
253
- 1. Technical decisions and their reasoning
254
- 2. Bug root causes and fixes
255
- 3. Configuration values (ports, versions, env vars)
256
- 4. Architecture patterns and constraints
257
- 5. Gotchas, pitfalls, and workarounds
258
- 6. File paths and their roles
259
-
260
- Rules:
261
- - Return ONLY a JSON object with a "facts" key containing an array of strings
262
- - Each fact should be a concise, self-contained statement
263
- - Include specific values (versions, ports, paths) when present
264
- - Detect the language of the input and record facts in the same language
265
- - If no meaningful facts exist, return {"facts": []}
266
- - Do NOT include trivial information (file read, directory listing)
267
- - Maximum 10 facts
268
-
269
- Example:
270
- Input: "Fixed Redis connection leak. The pool wasn't being closed on shutdown. Added defer pool.Close() in main.go. Port 6379."
271
- Output: {"facts": ["Redis connection leak caused by pool not closed on shutdown", "Fix: added defer pool.Close() in main.go", "Redis port: 6379"]}`;
272
-
273
- /**
274
- * Extract facts using LLM (Mem0-style structured extraction).
275
- * Returns extracted facts or empty array on failure.
276
- */
277
- async function extractFactsWithLLM(
278
- narrative: string,
279
- title: string,
280
- existingFacts: string[],
281
- ): Promise<string[]> {
282
- try {
283
- const { callLLM } = await import('../../llm/provider.js');
284
- const input = `Title: ${title}\nContent: ${narrative}${existingFacts.length > 0 ? `\nAlready known facts (don't repeat): ${existingFacts.join('; ')}` : ''}`;
285
- const response = await callLLM(LLM_EXTRACT_PROMPT, input);
286
- const text = response.content.trim();
287
-
288
- // Parse JSON response
289
- const jsonMatch = text.match(/\{[\s\S]*\}/);
290
- if (!jsonMatch) return [];
291
- const parsed = JSON.parse(jsonMatch[0]);
292
- const facts = parsed.facts;
293
- if (!Array.isArray(facts)) return [];
294
-
295
- // Filter: remove duplicates with existing facts
296
- const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
297
- return facts
298
- .filter((f: unknown): f is string => typeof f === 'string' && f.length >= 5)
299
- .filter((f: string) => !existingLower.has(f.toLowerCase().trim()))
300
- .slice(0, 10);
301
- } catch {
302
- return []; // LLM failure → fall back to rules
303
- }
304
- }
305
-
306
- // ── Public API ───────────────────────────────────────────────────
307
-
308
- /**
309
- * Run Stage 1: Extract.
310
- *
311
- * Enriches raw input with system-extracted facts, normalized titles,
312
- * resolved entities, and verified types.
313
- *
314
- * When useLLM=true, uses LLM for fact extraction (Mem0-style).
315
- * Falls back to rules-based extraction on LLM failure.
316
- */
317
- export async function runExtract(
318
- input: FormationInput,
319
- existingEntities: string[],
320
- useLLM = false,
321
- ): Promise<ExtractResult> {
322
- const callerFacts = input.facts ?? [];
323
-
324
- // 1. Extract facts from narrative
325
- let extractedFacts: string[];
326
- if (useLLM) {
327
- // LLM extraction (quality-first, Mem0-style)
328
- extractedFacts = await extractFactsWithLLM(input.narrative, input.title, callerFacts);
329
- // If LLM returned nothing, fall back to rules
330
- if (extractedFacts.length === 0) {
331
- extractedFacts = extractFacts(input.narrative, callerFacts);
332
- }
333
- } else {
334
- // Rules-based extraction (free mode)
335
- extractedFacts = extractFacts(input.narrative, callerFacts);
336
- }
337
- const allFacts = [...callerFacts, ...extractedFacts];
338
-
339
- // 2. Improve title if generic
340
- const { title, improved: titleImproved } = improveTitle(input.title, input.narrative);
341
-
342
- // 3. Resolve entity name
343
- const { entityName, resolved: entityResolved } = resolveEntity(
344
- input.entityName,
345
- existingEntities,
346
- );
347
-
348
- // 4. Verify observation type
349
- const { type, corrected: typeCorrected } = verifyType(
350
- input.type,
351
- input.narrative,
352
- input.title,
353
- );
354
-
355
- return {
356
- title,
357
- titleImproved,
358
- narrative: input.narrative,
359
- facts: allFacts,
360
- extractedFacts,
361
- entityName,
362
- entityResolved,
363
- type,
364
- typeCorrected,
365
- };
366
- }
31
+ // Version numbers (e.g., "v1.2.3", "version 2.0")
32
+ {
33
+ pattern: /\b(?:v(?:ersion)?\s*)(\d+\.\d+(?:\.\d+)?(?:-[\w.]+)?)\b/gi,
34
+ format: (m) => `Version: ${m[1]}`,
35
+ },
36
+ // Error messages (e.g., "Error: ...", "ERR_...")
37
+ {
38
+ pattern: /\b(?:Error|ERR|ENOENT|ECONNREFUSED|TypeError|RangeError|SyntaxError|ReferenceError)[:\s]+([^\n]{5,80})/gi,
39
+ format: (m) => `Error: ${m[1].trim()}`,
40
+ },
41
+ // Port numbers in context
42
+ {
43
+ pattern: /\b(?:port|PORT)\s*[:=]?\s*(\d{2,5})\b/gi,
44
+ format: (m) => `Port: ${m[1]}`,
45
+ },
46
+ // Environment variables
47
+ {
48
+ pattern: /\b([A-Z][A-Z0-9_]{3,30})\s*=\s*(\S{1,60})/g,
49
+ format: (m) => `${m[1]}=${m[2]}`,
50
+ },
51
+ // npm/package versions (e.g., "react@18.2.0")
52
+ {
53
+ pattern: /\b([@a-z][\w./-]+)@(\d+\.\d+\.\d+(?:-[\w.]+)?)\b/g,
54
+ format: (m) => `${m[1]}@${m[2]}`,
55
+ },
56
+ ];
57
+
58
+ /** Patterns indicating generic/low-quality titles that should be improved */
59
+ const GENERIC_TITLE_PATTERNS = [
60
+ /^Updated \S+\.\w+$/i,
61
+ /^Created \S+\.\w+$/i,
62
+ /^Deleted \S+\.\w+$/i,
63
+ /^Modified \S+\.\w+$/i,
64
+ /^Changed \S+\.\w+$/i,
65
+ /^Session activity/i,
66
+ /^Activity \(/i,
67
+ /^Used \w+$/i,
68
+ /^Ran: /i,
69
+ ];
70
+
71
+ /** Content signals mapped to observation types */
72
+ const TYPE_SIGNALS: Array<{ type: ObservationType; patterns: RegExp[] }> = [
73
+ {
74
+ type: 'problem-solution',
75
+ patterns: [
76
+ /\b(fix|fixed|bug|error|issue|crash|broken|resolved|workaround|patch)\b/i,
77
+ /\b(修复|修正|解决|报错|崩溃|异常)\b/,
78
+ ],
79
+ },
80
+ {
81
+ type: 'gotcha',
82
+ patterns: [
83
+ /\b(gotcha|pitfall|trap|careful|warning|caveat|footgun|unexpected|beware)\b/i,
84
+ /\b(坑|陷阱|注意|小心|踩坑)\b/,
85
+ ],
86
+ },
87
+ {
88
+ type: 'decision',
89
+ patterns: [
90
+ /\b(decided|chose|chosen|selected|adopted|rejected|evaluated|compared)\b/i,
91
+ /\b(决定|选择|采用|弃用|对比|评估)\b/,
92
+ ],
93
+ },
94
+ {
95
+ type: 'what-changed',
96
+ patterns: [
97
+ /\b(changed|migrated|upgraded|refactored|replaced|renamed|moved|removed|added)\b/i,
98
+ /\b(改|迁移|升级|重构|替换|重命名|删除|新增)\b/,
99
+ ],
100
+ },
101
+ {
102
+ type: 'how-it-works',
103
+ patterns: [
104
+ /\b(works by|architecture|mechanism|pipeline|flow|under the hood|internally)\b/i,
105
+ /\b(原理|机制|流程|架构|内部)\b/,
106
+ ],
107
+ },
108
+ {
109
+ type: 'trade-off',
110
+ patterns: [
111
+ /\b(trade.?off|compromise|downside|cost|benefit|pro|con|versus|vs)\b/i,
112
+ /\b(权衡|折中|代价|收益|优缺点)\b/,
113
+ ],
114
+ },
115
+ ];
116
+
117
+ // ── Extract Implementation ───────────────────────────────────────
118
+
119
+ /**
120
+ * Extract structured facts from narrative text using regex patterns.
121
+ * Returns only facts not already present in the caller-provided list.
122
+ */
123
+ function extractFacts(narrative: string, existingFacts: string[]): string[] {
124
+ const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
125
+ const extracted: string[] = [];
126
+ const seen = new Set<string>();
127
+
128
+ for (const { pattern, format } of FACT_PATTERNS) {
129
+ pattern.lastIndex = 0;
130
+ let match: RegExpExecArray | null;
131
+ while ((match = pattern.exec(narrative)) !== null) {
132
+ const fact = format(match);
133
+ const normalized = fact.toLowerCase().trim();
134
+
135
+ // Skip if already provided by caller or already extracted
136
+ if (existingLower.has(normalized) || seen.has(normalized)) continue;
137
+
138
+ // Skip very short or very long facts
139
+ if (fact.length < 5 || fact.length > 120) continue;
140
+
141
+ seen.add(normalized);
142
+ extracted.push(fact);
143
+ }
144
+ }
145
+
146
+ return extracted.slice(0, 10); // Cap at 10 system-extracted facts
147
+ }
148
+
149
+ /**
150
+ * Improve a generic title by extracting the first meaningful sentence
151
+ * from the narrative.
152
+ */
153
+ function improveTitle(title: string, narrative: string): { title: string; improved: boolean } {
154
+ const isGeneric = GENERIC_TITLE_PATTERNS.some(p => p.test(title));
155
+ if (!isGeneric) return { title, improved: false };
156
+
157
+ // Try to extract first meaningful sentence from narrative
158
+ const sentences = narrative
159
+ .replace(/```[\s\S]*?```/g, '') // Remove code blocks
160
+ .split(/[.。!!?\n]/)
161
+ .map(s => s.trim())
162
+ .filter(s => s.length >= 15);
163
+
164
+ if (sentences.length > 0) {
165
+ return { title: sentences[0].slice(0, 60), improved: true };
166
+ }
167
+
168
+ return { title, improved: false };
169
+ }
170
+
171
+ /**
172
+ * Resolve entity name against existing Knowledge Graph entities.
173
+ * If a close match is found, use the canonical entity name.
174
+ */
175
+ function resolveEntity(
176
+ entityName: string,
177
+ existingEntities: string[],
178
+ ): { entityName: string; resolved: boolean } {
179
+ if (existingEntities.length === 0) return { entityName, resolved: false };
180
+
181
+ const lower = entityName.toLowerCase().replace(/[-_]/g, '');
182
+
183
+ for (const existing of existingEntities) {
184
+ const existingLower = existing.toLowerCase().replace(/[-_]/g, '');
185
+
186
+ // Exact match (case-insensitive, ignoring hyphens/underscores)
187
+ if (lower === existingLower) {
188
+ return { entityName: existing, resolved: existing !== entityName };
189
+ }
190
+
191
+ // Substring match: one contains the other (e.g., "auth" matches "auth-module")
192
+ if (lower.length >= 3 && existingLower.length >= 3) {
193
+ if (existingLower.includes(lower) || lower.includes(existingLower)) {
194
+ // Prefer the longer (more specific) name
195
+ const canonical = existing.length >= entityName.length ? existing : entityName;
196
+ return { entityName: canonical, resolved: canonical !== entityName };
197
+ }
198
+ }
199
+ }
200
+
201
+ return { entityName, resolved: false };
202
+ }
203
+
204
+ /**
205
+ * Verify observation type against content signals.
206
+ * If content strongly suggests a different type, correct it.
207
+ */
208
+ function verifyType(
209
+ declaredType: ObservationType,
210
+ narrative: string,
211
+ title: string,
212
+ ): { type: ObservationType; corrected: boolean } {
213
+ const content = `${title} ${narrative}`;
214
+
215
+ // Score each type by counting individual keyword hits across all patterns
216
+ const scores: Array<{ type: ObservationType; score: number }> = [];
217
+ for (const { type, patterns } of TYPE_SIGNALS) {
218
+ let score = 0;
219
+ for (const p of patterns) {
220
+ // Use matchAll to count individual keyword matches
221
+ const regex = new RegExp(p.source, p.flags.includes('g') ? p.flags : p.flags + 'g');
222
+ const matches = [...content.matchAll(regex)];
223
+ score += matches.length;
224
+ }
225
+ if (score > 0) scores.push({ type, score });
226
+ }
227
+
228
+ if (scores.length === 0) return { type: declaredType, corrected: false };
229
+
230
+ // Sort by score descending
231
+ scores.sort((a, b) => b.score - a.score);
232
+ const best = scores[0];
233
+
234
+ // Only correct if the best match is significantly stronger than declared type
235
+ // Requires: best type has >= 2 keyword hits AND declared type has 0 signals
236
+ if (best.type !== declaredType && best.score >= 2) {
237
+ const declaredScore = scores.find(s => s.type === declaredType)?.score ?? 0;
238
+ if (declaredScore === 0) {
239
+ return { type: best.type, corrected: true };
240
+ }
241
+ }
242
+
243
+ return { type: declaredType, corrected: false };
244
+ }
245
+
246
+ // ── LLM Fact Extraction ─────────────────────────────────────────
247
+
248
+ /** Prompt for LLM-based fact extraction (inspired by Mem0's approach) */
249
+ const LLM_EXTRACT_PROMPT = `You are a Software Engineering Knowledge Extractor.
250
+ Extract structured facts from the given development context.
251
+
252
+ Focus on:
253
+ 1. Technical decisions and their reasoning
254
+ 2. Bug root causes and fixes
255
+ 3. Configuration values (ports, versions, env vars)
256
+ 4. Architecture patterns and constraints
257
+ 5. Gotchas, pitfalls, and workarounds
258
+ 6. File paths and their roles
259
+
260
+ Rules:
261
+ - Return ONLY a JSON object with a "facts" key containing an array of strings
262
+ - Each fact should be a concise, self-contained statement
263
+ - Include specific values (versions, ports, paths) when present
264
+ - Detect the language of the input and record facts in the same language
265
+ - If no meaningful facts exist, return {"facts": []}
266
+ - Do NOT include trivial information (file read, directory listing)
267
+ - Maximum 10 facts
268
+
269
+ Example:
270
+ Input: "Fixed Redis connection leak. The pool wasn't being closed on shutdown. Added defer pool.Close() in main.go. Port 6379."
271
+ Output: {"facts": ["Redis connection leak caused by pool not closed on shutdown", "Fix: added defer pool.Close() in main.go", "Redis port: 6379"]}`;
272
+
273
+ /**
274
+ * Extract facts using LLM (Mem0-style structured extraction).
275
+ * Returns extracted facts or empty array on failure.
276
+ */
277
+ async function extractFactsWithLLM(
278
+ narrative: string,
279
+ title: string,
280
+ existingFacts: string[],
281
+ ): Promise<string[]> {
282
+ try {
283
+ const { callLLM } = await import('../../llm/provider.js');
284
+ const input = `Title: ${title}\nContent: ${narrative}${existingFacts.length > 0 ? `\nAlready known facts (don't repeat): ${existingFacts.join('; ')}` : ''}`;
285
+ const response = await callLLM(LLM_EXTRACT_PROMPT, input);
286
+ const text = response.content.trim();
287
+
288
+ // Parse JSON response
289
+ const jsonMatch = text.match(/\{[\s\S]*\}/);
290
+ if (!jsonMatch) return [];
291
+ const parsed = JSON.parse(jsonMatch[0]);
292
+ const facts = parsed.facts;
293
+ if (!Array.isArray(facts)) return [];
294
+
295
+ // Filter: remove duplicates with existing facts
296
+ const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
297
+ return facts
298
+ .filter((f: unknown): f is string => typeof f === 'string' && f.length >= 5)
299
+ .filter((f: string) => !existingLower.has(f.toLowerCase().trim()))
300
+ .slice(0, 10);
301
+ } catch {
302
+ return []; // LLM failure → fall back to rules
303
+ }
304
+ }
305
+
306
+ // ── Public API ───────────────────────────────────────────────────
307
+
308
+ /**
309
+ * Run Stage 1: Extract.
310
+ *
311
+ * Enriches raw input with system-extracted facts, normalized titles,
312
+ * resolved entities, and verified types.
313
+ *
314
+ * When useLLM=true, uses LLM for fact extraction (Mem0-style).
315
+ * Falls back to rules-based extraction on LLM failure.
316
+ */
317
+ export async function runExtract(
318
+ input: FormationInput,
319
+ existingEntities: string[],
320
+ useLLM = false,
321
+ ): Promise<ExtractResult> {
322
+ const callerFacts = input.facts ?? [];
323
+
324
+ // 1. Extract facts from narrative
325
+ let extractedFacts: string[];
326
+ if (useLLM) {
327
+ // LLM extraction (quality-first, Mem0-style)
328
+ extractedFacts = await extractFactsWithLLM(input.narrative, input.title, callerFacts);
329
+ // If LLM returned nothing, fall back to rules
330
+ if (extractedFacts.length === 0) {
331
+ extractedFacts = extractFacts(input.narrative, callerFacts);
332
+ }
333
+ } else {
334
+ // Rules-based extraction (free mode)
335
+ extractedFacts = extractFacts(input.narrative, callerFacts);
336
+ }
337
+ const allFacts = [...callerFacts, ...extractedFacts];
338
+
339
+ // 2. Improve title if generic
340
+ const { title, improved: titleImproved } = improveTitle(input.title, input.narrative);
341
+
342
+ // 3. Resolve entity name
343
+ const { entityName, resolved: entityResolved } = resolveEntity(
344
+ input.entityName,
345
+ existingEntities,
346
+ );
347
+
348
+ // 4. Verify observation type
349
+ const { type, corrected: typeCorrected } = verifyType(
350
+ input.type,
351
+ input.narrative,
352
+ input.title,
353
+ );
354
+
355
+ return {
356
+ title,
357
+ titleImproved,
358
+ narrative: input.narrative,
359
+ facts: allFacts,
360
+ extractedFacts,
361
+ entityName,
362
+ entityResolved,
363
+ type,
364
+ typeCorrected,
365
+ };
366
+ }