memorix 1.2.0 → 1.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (212) hide show
  1. package/CHANGELOG.md +30 -1
  2. package/README.md +18 -4
  3. package/README.zh-CN.md +18 -4
  4. package/TEAM.md +86 -86
  5. package/dist/cli/index.js +15919 -14055
  6. package/dist/cli/index.js.map +1 -1
  7. package/dist/index.js +1997 -1021
  8. package/dist/index.js.map +1 -1
  9. package/dist/maintenance-runner.d.ts +1 -1
  10. package/dist/maintenance-runner.js +8481 -8005
  11. package/dist/maintenance-runner.js.map +1 -1
  12. package/dist/memcode-runtime/CHANGELOG.md +30 -1
  13. package/dist/sdk.d.ts +7 -2
  14. package/dist/sdk.js +2022 -1024
  15. package/dist/sdk.js.map +1 -1
  16. package/dist/types.d.ts +49 -1
  17. package/dist/types.js.map +1 -1
  18. package/docs/1.2.2-MEMORY-CONTROL-PLANE.md +434 -0
  19. package/docs/AGENT_OPERATOR_PLAYBOOK.md +4 -0
  20. package/docs/API_REFERENCE.md +27 -5
  21. package/docs/DESIGN_DECISIONS.md +357 -357
  22. package/docs/DEVELOPMENT.md +4 -0
  23. package/docs/README.md +1 -1
  24. package/docs/SETUP.md +7 -1
  25. package/docs/dev-log/progress.txt +91 -11
  26. package/docs/knowledge/workflows/memorix-release.md +57 -0
  27. package/package.json +1 -1
  28. package/plugins/codex/memorix/.codex-plugin/plugin.json +1 -1
  29. package/src/audit/index.ts +156 -156
  30. package/src/cli/command-guide.ts +192 -0
  31. package/src/cli/commands/audit-list.ts +89 -89
  32. package/src/cli/commands/audit.ts +9 -4
  33. package/src/cli/commands/background.ts +659 -659
  34. package/src/cli/commands/cleanup.ts +5 -1
  35. package/src/cli/commands/codegraph.ts +17 -8
  36. package/src/cli/commands/context.ts +3 -2
  37. package/src/cli/commands/doctor.ts +4 -2
  38. package/src/cli/commands/explain.ts +9 -3
  39. package/src/cli/commands/formation.ts +48 -48
  40. package/src/cli/commands/git-hook-install.ts +111 -111
  41. package/src/cli/commands/handoff.ts +75 -61
  42. package/src/cli/commands/hooks-status.ts +63 -63
  43. package/src/cli/commands/identity.ts +116 -0
  44. package/src/cli/commands/ingest-commit.ts +153 -153
  45. package/src/cli/commands/ingest-image.ts +71 -69
  46. package/src/cli/commands/ingest-log.ts +180 -180
  47. package/src/cli/commands/ingest.ts +44 -44
  48. package/src/cli/commands/integrate-shared.ts +15 -15
  49. package/src/cli/commands/knowledge.ts +40 -0
  50. package/src/cli/commands/lock.ts +93 -92
  51. package/src/cli/commands/memory.ts +58 -21
  52. package/src/cli/commands/message.ts +123 -118
  53. package/src/cli/commands/operator-shared.ts +98 -3
  54. package/src/cli/commands/poll.ts +74 -64
  55. package/src/cli/commands/purge-all-memory.ts +85 -85
  56. package/src/cli/commands/purge-project-memory.ts +83 -83
  57. package/src/cli/commands/reasoning.ts +135 -121
  58. package/src/cli/commands/retention.ts +9 -4
  59. package/src/cli/commands/serve-http.ts +22 -43
  60. package/src/cli/commands/serve-shared.ts +118 -118
  61. package/src/cli/commands/session.ts +29 -3
  62. package/src/cli/commands/setup.ts +9 -3
  63. package/src/cli/commands/skills.ts +124 -119
  64. package/src/cli/commands/status.ts +4 -3
  65. package/src/cli/commands/task.ts +193 -184
  66. package/src/cli/commands/team.ts +14 -10
  67. package/src/cli/commands/transfer.ts +108 -55
  68. package/src/cli/commands/uninstall-project-artifacts.ts +85 -85
  69. package/src/cli/identity.ts +89 -0
  70. package/src/cli/index.ts +96 -19
  71. package/src/cli/invocation.ts +115 -0
  72. package/src/cli/tui/ChatView.tsx +234 -234
  73. package/src/cli/tui/CommandBar.tsx +312 -312
  74. package/src/cli/tui/ContextRail.tsx +118 -118
  75. package/src/cli/tui/HeaderBar.tsx +72 -72
  76. package/src/cli/tui/LogoBanner.tsx +51 -51
  77. package/src/cli/tui/Sidebar.tsx +179 -179
  78. package/src/cli/tui/chat-service.ts +41 -18
  79. package/src/cli/tui/data.ts +23 -44
  80. package/src/cli/tui/index.ts +41 -41
  81. package/src/cli/tui/markdown-render.tsx +371 -371
  82. package/src/cli/tui/operator-context.ts +60 -0
  83. package/src/cli/tui/use-mouse.ts +157 -157
  84. package/src/cli/tui/useNavigation.ts +56 -56
  85. package/src/cli/tui/views/MemoryView.tsx +10 -8
  86. package/src/cli/update-checker.ts +211 -211
  87. package/src/cli/version.ts +7 -7
  88. package/src/cli/workbench.ts +1 -1
  89. package/src/codegraph/auto-context.ts +34 -17
  90. package/src/codegraph/context-pack.ts +1 -0
  91. package/src/codegraph/current-facts.ts +19 -1
  92. package/src/codegraph/project-context.ts +2 -0
  93. package/src/codegraph/task-lens.ts +49 -5
  94. package/src/compact/engine.ts +26 -10
  95. package/src/compact/index-format.ts +25 -2
  96. package/src/compact/token-budget.ts +74 -74
  97. package/src/dashboard/project-classification.ts +64 -64
  98. package/src/dashboard/server.ts +58 -52
  99. package/src/embedding/fastembed-provider.ts +142 -142
  100. package/src/embedding/transformers-provider.ts +111 -111
  101. package/src/git/extractor.ts +209 -209
  102. package/src/git/hooks-path.ts +85 -85
  103. package/src/hooks/admission.ts +117 -0
  104. package/src/hooks/handler.ts +98 -91
  105. package/src/hooks/pattern-detector.ts +173 -173
  106. package/src/hooks/significance-filter.ts +250 -250
  107. package/src/knowledge/claims.ts +51 -1
  108. package/src/knowledge/context-assembly.ts +97 -0
  109. package/src/knowledge/types.ts +1 -0
  110. package/src/knowledge/workflows.ts +34 -3
  111. package/src/knowledge/workset.ts +179 -10
  112. package/src/llm/memory-manager.ts +328 -328
  113. package/src/llm/provider.ts +885 -885
  114. package/src/llm/quality.ts +248 -248
  115. package/src/memory/admission.ts +57 -0
  116. package/src/memory/attribution-guard.ts +249 -249
  117. package/src/memory/auto-relations.ts +21 -0
  118. package/src/memory/consolidation.ts +13 -2
  119. package/src/memory/disclosure-policy.ts +140 -135
  120. package/src/memory/entity-extractor.ts +197 -197
  121. package/src/memory/export-import.ts +11 -3
  122. package/src/memory/formation/evaluate.ts +217 -217
  123. package/src/memory/formation/extract.ts +361 -361
  124. package/src/memory/formation/index.ts +417 -417
  125. package/src/memory/formation/resolve.ts +344 -344
  126. package/src/memory/formation/types.ts +315 -315
  127. package/src/memory/freshness.ts +122 -122
  128. package/src/memory/graph-context.ts +8 -2
  129. package/src/memory/graph-scope.ts +46 -0
  130. package/src/memory/graph.ts +197 -197
  131. package/src/memory/observations.ts +162 -4
  132. package/src/memory/quality-audit.ts +2 -0
  133. package/src/memory/refs.ts +94 -94
  134. package/src/memory/retention.ts +22 -2
  135. package/src/memory/secret-filter.ts +79 -79
  136. package/src/memory/session.ts +5 -2
  137. package/src/memory/visibility.ts +80 -0
  138. package/src/multimodal/image-loader.ts +143 -143
  139. package/src/orchestrate/adapters/claude-stream.ts +192 -192
  140. package/src/orchestrate/adapters/claude.ts +111 -111
  141. package/src/orchestrate/adapters/codex-stream.ts +134 -134
  142. package/src/orchestrate/adapters/codex.ts +41 -41
  143. package/src/orchestrate/adapters/gemini-stream.ts +166 -166
  144. package/src/orchestrate/adapters/gemini.ts +42 -42
  145. package/src/orchestrate/adapters/index.ts +73 -73
  146. package/src/orchestrate/adapters/opencode-stream.ts +143 -143
  147. package/src/orchestrate/adapters/opencode.ts +47 -47
  148. package/src/orchestrate/adapters/spawn-helper.ts +286 -286
  149. package/src/orchestrate/adapters/types.ts +77 -77
  150. package/src/orchestrate/capability-router.ts +284 -284
  151. package/src/orchestrate/context-compact.ts +188 -188
  152. package/src/orchestrate/cost-tracker.ts +219 -219
  153. package/src/orchestrate/error-recovery.ts +191 -191
  154. package/src/orchestrate/evidence.ts +140 -140
  155. package/src/orchestrate/ledger.ts +110 -110
  156. package/src/orchestrate/memorix-bridge.ts +378 -340
  157. package/src/orchestrate/output-budget.ts +80 -80
  158. package/src/orchestrate/permission.ts +152 -152
  159. package/src/orchestrate/pipeline-trace.ts +131 -131
  160. package/src/orchestrate/prompt-builder.ts +155 -155
  161. package/src/orchestrate/ring-buffer.ts +37 -37
  162. package/src/orchestrate/task-graph.ts +389 -389
  163. package/src/orchestrate/verify-gate.ts +33 -10
  164. package/src/orchestrate/worktree.ts +232 -232
  165. package/src/project/aliases.ts +374 -374
  166. package/src/project/detector.ts +268 -268
  167. package/src/rules/adapters/claude-code.ts +99 -99
  168. package/src/rules/adapters/codex.ts +97 -97
  169. package/src/rules/adapters/copilot.ts +124 -124
  170. package/src/rules/adapters/cursor.ts +114 -114
  171. package/src/rules/adapters/kiro.ts +126 -126
  172. package/src/rules/adapters/trae.ts +56 -56
  173. package/src/rules/adapters/windsurf.ts +83 -83
  174. package/src/rules/syncer.ts +235 -235
  175. package/src/runtime/control-plane-maintenance.ts +1 -0
  176. package/src/runtime/isolated-maintenance.ts +1 -0
  177. package/src/runtime/lifecycle.ts +18 -0
  178. package/src/runtime/maintenance-jobs.ts +1 -0
  179. package/src/runtime/maintenance-runner.ts +2 -0
  180. package/src/runtime/project-maintenance.ts +89 -0
  181. package/src/sdk.ts +334 -304
  182. package/src/search/intent-detector.ts +289 -289
  183. package/src/search/query-expansion.ts +52 -52
  184. package/src/server/formation-timeout.ts +27 -27
  185. package/src/server.ts +334 -93
  186. package/src/skills/mini-skills.ts +386 -386
  187. package/src/store/chat-store.ts +119 -119
  188. package/src/store/graph-store.ts +249 -249
  189. package/src/store/mini-skill-store.ts +349 -349
  190. package/src/store/orama-store.ts +61 -6
  191. package/src/store/persistence-json.ts +212 -212
  192. package/src/store/persistence.ts +291 -291
  193. package/src/store/project-affinity.ts +195 -195
  194. package/src/store/sqlite-db.ts +23 -1
  195. package/src/store/sqlite-store.ts +12 -2
  196. package/src/team/event-bus.ts +76 -76
  197. package/src/team/file-locks.ts +173 -173
  198. package/src/team/handoff.ts +168 -161
  199. package/src/team/messages.ts +203 -203
  200. package/src/team/poll.ts +132 -132
  201. package/src/team/tasks.ts +211 -211
  202. package/src/types.ts +51 -0
  203. package/src/wiki/generator.ts +2 -0
  204. package/src/workspace/mcp-adapters/codex.ts +191 -191
  205. package/src/workspace/mcp-adapters/copilot.ts +105 -105
  206. package/src/workspace/mcp-adapters/cursor.ts +53 -53
  207. package/src/workspace/mcp-adapters/kiro.ts +64 -64
  208. package/src/workspace/mcp-adapters/opencode.ts +123 -123
  209. package/src/workspace/mcp-adapters/trae.ts +134 -134
  210. package/src/workspace/mcp-adapters/windsurf.ts +91 -91
  211. package/src/workspace/sanitizer.ts +60 -60
  212. package/src/workspace/workflow-sync.ts +131 -131
@@ -1,366 +1,366 @@
1
- /**
2
- * Memory Formation — Stage 1: Extract
3
- *
4
- * Enriches raw memory input with system-extracted facts, normalized titles,
5
- * resolved entity names, and verified observation types.
6
- *
7
- * Rules-based mode (no LLM):
8
- * - Fact extraction: key-value patterns, error messages, version numbers, paths
9
- * - Title normalization: replace generic titles with first meaningful sentence
10
- * - Entity resolution: match against existing Knowledge Graph entities
11
- * - Type inference: verify type matches content signals
12
- */
13
-
14
- import type { ObservationType } from '../../types.js';
15
- import type { FormationInput, ExtractResult } from './types.js';
16
-
17
- // ── Fact Extraction Patterns ──────────────────────────────────────
18
-
19
- /** Patterns that extract structured facts from narrative text */
20
- const FACT_PATTERNS: Array<{ pattern: RegExp; format: (m: RegExpMatchArray) => string }> = [
21
- // Key: Value pairs (e.g., "Port: 3000", "Timeout = 60s")
22
- {
23
- pattern: /\b([A-Z][a-zA-Z_-]{2,30})\s*[:=]\s*([^\n,;]{2,60})/g,
24
- format: (m) => `${m[1]}: ${m[2].trim()}`,
25
- },
1
+ /**
2
+ * Memory Formation — Stage 1: Extract
3
+ *
4
+ * Enriches raw memory input with system-extracted facts, normalized titles,
5
+ * resolved entity names, and verified observation types.
6
+ *
7
+ * Rules-based mode (no LLM):
8
+ * - Fact extraction: key-value patterns, error messages, version numbers, paths
9
+ * - Title normalization: replace generic titles with first meaningful sentence
10
+ * - Entity resolution: match against existing Knowledge Graph entities
11
+ * - Type inference: verify type matches content signals
12
+ */
13
+
14
+ import type { ObservationType } from '../../types.js';
15
+ import type { FormationInput, ExtractResult } from './types.js';
16
+
17
+ // ── Fact Extraction Patterns ──────────────────────────────────────
18
+
19
+ /** Patterns that extract structured facts from narrative text */
20
+ const FACT_PATTERNS: Array<{ pattern: RegExp; format: (m: RegExpMatchArray) => string }> = [
21
+ // Key: Value pairs (e.g., "Port: 3000", "Timeout = 60s")
22
+ {
23
+ pattern: /\b([A-Z][a-zA-Z_-]{2,30})\s*[:=]\s*([^\n,;]{2,60})/g,
24
+ format: (m) => `${m[1]}: ${m[2].trim()}`,
25
+ },
26
26
  // Arrow notation (e.g., "MySQL -> PostgreSQL", "v1.0 -> v2.0")
27
27
  {
28
28
  pattern: /\b(\S{2,30})\s*(?:->|=>|>)\s*(\S{2,30})/g,
29
29
  format: (m) => `${m[1]} -> ${m[2]}`,
30
30
  },
31
- // Version numbers (e.g., "v1.2.3", "version 2.0")
32
- {
33
- pattern: /\b(?:v(?:ersion)?\s*)(\d+\.\d+(?:\.\d+)?(?:-[\w.]+)?)\b/gi,
34
- format: (m) => `Version: ${m[1]}`,
35
- },
36
- // Error messages (e.g., "Error: ...", "ERR_...")
37
- {
38
- pattern: /\b(?:Error|ERR|ENOENT|ECONNREFUSED|TypeError|RangeError|SyntaxError|ReferenceError)[:\s]+([^\n]{5,80})/gi,
39
- format: (m) => `Error: ${m[1].trim()}`,
40
- },
41
- // Port numbers in context
42
- {
43
- pattern: /\b(?:port|PORT)\s*[:=]?\s*(\d{2,5})\b/gi,
44
- format: (m) => `Port: ${m[1]}`,
45
- },
46
- // Environment variables
47
- {
48
- pattern: /\b([A-Z][A-Z0-9_]{3,30})\s*=\s*(\S{1,60})/g,
49
- format: (m) => `${m[1]}=${m[2]}`,
50
- },
51
- // npm/package versions (e.g., "react@18.2.0")
52
- {
53
- pattern: /\b([@a-z][\w./-]+)@(\d+\.\d+\.\d+(?:-[\w.]+)?)\b/g,
54
- format: (m) => `${m[1]}@${m[2]}`,
55
- },
56
- ];
57
-
58
- /** Patterns indicating generic/low-quality titles that should be improved */
59
- const GENERIC_TITLE_PATTERNS = [
60
- /^Updated \S+\.\w+$/i,
61
- /^Created \S+\.\w+$/i,
62
- /^Deleted \S+\.\w+$/i,
63
- /^Modified \S+\.\w+$/i,
64
- /^Changed \S+\.\w+$/i,
65
- /^Session activity/i,
66
- /^Activity \(/i,
67
- /^Used \w+$/i,
68
- /^Ran: /i,
69
- ];
70
-
71
- /** Content signals mapped to observation types */
72
- const TYPE_SIGNALS: Array<{ type: ObservationType; patterns: RegExp[] }> = [
73
- {
74
- type: 'problem-solution',
75
- patterns: [
76
- /\b(fix|fixed|bug|error|issue|crash|broken|resolved|workaround|patch)\b/i,
77
- /\b(修复|修正|解决|报错|崩溃|异常)\b/,
78
- ],
79
- },
80
- {
81
- type: 'gotcha',
82
- patterns: [
83
- /\b(gotcha|pitfall|trap|careful|warning|caveat|footgun|unexpected|beware)\b/i,
84
- /\b(坑|陷阱|注意|小心|踩坑)\b/,
85
- ],
86
- },
87
- {
88
- type: 'decision',
89
- patterns: [
90
- /\b(decided|chose|chosen|selected|adopted|rejected|evaluated|compared)\b/i,
91
- /\b(决定|选择|采用|弃用|对比|评估)\b/,
92
- ],
93
- },
94
- {
95
- type: 'what-changed',
96
- patterns: [
97
- /\b(changed|migrated|upgraded|refactored|replaced|renamed|moved|removed|added)\b/i,
98
- /\b(改|迁移|升级|重构|替换|重命名|删除|新增)\b/,
99
- ],
100
- },
101
- {
102
- type: 'how-it-works',
103
- patterns: [
104
- /\b(works by|architecture|mechanism|pipeline|flow|under the hood|internally)\b/i,
105
- /\b(原理|机制|流程|架构|内部)\b/,
106
- ],
107
- },
108
- {
109
- type: 'trade-off',
110
- patterns: [
111
- /\b(trade.?off|compromise|downside|cost|benefit|pro|con|versus|vs)\b/i,
112
- /\b(权衡|折中|代价|收益|优缺点)\b/,
113
- ],
114
- },
115
- ];
116
-
117
- // ── Extract Implementation ───────────────────────────────────────
118
-
119
- /**
120
- * Extract structured facts from narrative text using regex patterns.
121
- * Returns only facts not already present in the caller-provided list.
122
- */
123
- function extractFacts(narrative: string, existingFacts: string[]): string[] {
124
- const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
125
- const extracted: string[] = [];
126
- const seen = new Set<string>();
127
-
128
- for (const { pattern, format } of FACT_PATTERNS) {
129
- pattern.lastIndex = 0;
130
- let match: RegExpExecArray | null;
131
- while ((match = pattern.exec(narrative)) !== null) {
132
- const fact = format(match);
133
- const normalized = fact.toLowerCase().trim();
134
-
135
- // Skip if already provided by caller or already extracted
136
- if (existingLower.has(normalized) || seen.has(normalized)) continue;
137
-
138
- // Skip very short or very long facts
139
- if (fact.length < 5 || fact.length > 120) continue;
140
-
141
- seen.add(normalized);
142
- extracted.push(fact);
143
- }
144
- }
145
-
146
- return extracted.slice(0, 10); // Cap at 10 system-extracted facts
147
- }
148
-
149
- /**
150
- * Improve a generic title by extracting the first meaningful sentence
151
- * from the narrative.
152
- */
153
- function improveTitle(title: string, narrative: string): { title: string; improved: boolean } {
154
- const isGeneric = GENERIC_TITLE_PATTERNS.some(p => p.test(title));
155
- if (!isGeneric) return { title, improved: false };
156
-
157
- // Try to extract first meaningful sentence from narrative
158
- const sentences = narrative
159
- .replace(/```[\s\S]*?```/g, '') // Remove code blocks
160
- .split(/[.。!!?\n]/)
161
- .map(s => s.trim())
162
- .filter(s => s.length >= 15);
163
-
164
- if (sentences.length > 0) {
165
- return { title: sentences[0].slice(0, 60), improved: true };
166
- }
167
-
168
- return { title, improved: false };
169
- }
170
-
171
- /**
172
- * Resolve entity name against existing Knowledge Graph entities.
173
- * If a close match is found, use the canonical entity name.
174
- */
175
- function resolveEntity(
176
- entityName: string,
177
- existingEntities: string[],
178
- ): { entityName: string; resolved: boolean } {
179
- if (existingEntities.length === 0) return { entityName, resolved: false };
180
-
181
- const lower = entityName.toLowerCase().replace(/[-_]/g, '');
182
-
183
- for (const existing of existingEntities) {
184
- const existingLower = existing.toLowerCase().replace(/[-_]/g, '');
185
-
186
- // Exact match (case-insensitive, ignoring hyphens/underscores)
187
- if (lower === existingLower) {
188
- return { entityName: existing, resolved: existing !== entityName };
189
- }
190
-
191
- // Substring match: one contains the other (e.g., "auth" matches "auth-module")
192
- if (lower.length >= 3 && existingLower.length >= 3) {
193
- if (existingLower.includes(lower) || lower.includes(existingLower)) {
194
- // Prefer the longer (more specific) name
195
- const canonical = existing.length >= entityName.length ? existing : entityName;
196
- return { entityName: canonical, resolved: canonical !== entityName };
197
- }
198
- }
199
- }
200
-
201
- return { entityName, resolved: false };
202
- }
203
-
204
- /**
205
- * Verify observation type against content signals.
206
- * If content strongly suggests a different type, correct it.
207
- */
208
- function verifyType(
209
- declaredType: ObservationType,
210
- narrative: string,
211
- title: string,
212
- ): { type: ObservationType; corrected: boolean } {
213
- const content = `${title} ${narrative}`;
214
-
215
- // Score each type by counting individual keyword hits across all patterns
216
- const scores: Array<{ type: ObservationType; score: number }> = [];
217
- for (const { type, patterns } of TYPE_SIGNALS) {
218
- let score = 0;
219
- for (const p of patterns) {
220
- // Use matchAll to count individual keyword matches
221
- const regex = new RegExp(p.source, p.flags.includes('g') ? p.flags : p.flags + 'g');
222
- const matches = [...content.matchAll(regex)];
223
- score += matches.length;
224
- }
225
- if (score > 0) scores.push({ type, score });
226
- }
227
-
228
- if (scores.length === 0) return { type: declaredType, corrected: false };
229
-
230
- // Sort by score descending
231
- scores.sort((a, b) => b.score - a.score);
232
- const best = scores[0];
233
-
234
- // Only correct if the best match is significantly stronger than declared type
235
- // Requires: best type has >= 2 keyword hits AND declared type has 0 signals
236
- if (best.type !== declaredType && best.score >= 2) {
237
- const declaredScore = scores.find(s => s.type === declaredType)?.score ?? 0;
238
- if (declaredScore === 0) {
239
- return { type: best.type, corrected: true };
240
- }
241
- }
242
-
243
- return { type: declaredType, corrected: false };
244
- }
245
-
246
- // ── LLM Fact Extraction ─────────────────────────────────────────
247
-
248
- /** Prompt for LLM-based fact extraction (inspired by Mem0's approach) */
249
- const LLM_EXTRACT_PROMPT = `You are a Software Engineering Knowledge Extractor.
250
- Extract structured facts from the given development context.
251
-
252
- Focus on:
253
- 1. Technical decisions and their reasoning
254
- 2. Bug root causes and fixes
255
- 3. Configuration values (ports, versions, env vars)
256
- 4. Architecture patterns and constraints
257
- 5. Gotchas, pitfalls, and workarounds
258
- 6. File paths and their roles
259
-
260
- Rules:
261
- - Return ONLY a JSON object with a "facts" key containing an array of strings
262
- - Each fact should be a concise, self-contained statement
263
- - Include specific values (versions, ports, paths) when present
264
- - Detect the language of the input and record facts in the same language
265
- - If no meaningful facts exist, return {"facts": []}
266
- - Do NOT include trivial information (file read, directory listing)
267
- - Maximum 10 facts
268
-
269
- Example:
270
- Input: "Fixed Redis connection leak. The pool wasn't being closed on shutdown. Added defer pool.Close() in main.go. Port 6379."
271
- Output: {"facts": ["Redis connection leak caused by pool not closed on shutdown", "Fix: added defer pool.Close() in main.go", "Redis port: 6379"]}`;
272
-
273
- /**
274
- * Extract facts using LLM (Mem0-style structured extraction).
275
- * Returns extracted facts or empty array on failure.
276
- */
277
- async function extractFactsWithLLM(
278
- narrative: string,
279
- title: string,
280
- existingFacts: string[],
281
- ): Promise<string[]> {
282
- try {
283
- const { callLLM } = await import('../../llm/provider.js');
284
- const input = `Title: ${title}\nContent: ${narrative}${existingFacts.length > 0 ? `\nAlready known facts (don't repeat): ${existingFacts.join('; ')}` : ''}`;
285
- const response = await callLLM(LLM_EXTRACT_PROMPT, input);
286
- const text = response.content.trim();
287
-
288
- // Parse JSON response
289
- const jsonMatch = text.match(/\{[\s\S]*\}/);
290
- if (!jsonMatch) return [];
291
- const parsed = JSON.parse(jsonMatch[0]);
292
- const facts = parsed.facts;
293
- if (!Array.isArray(facts)) return [];
294
-
295
- // Filter: remove duplicates with existing facts
296
- const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
297
- return facts
298
- .filter((f: unknown): f is string => typeof f === 'string' && f.length >= 5)
299
- .filter((f: string) => !existingLower.has(f.toLowerCase().trim()))
300
- .slice(0, 10);
301
- } catch {
302
- return []; // LLM failure → fall back to rules
303
- }
304
- }
305
-
306
- // ── Public API ───────────────────────────────────────────────────
307
-
308
- /**
309
- * Run Stage 1: Extract.
310
- *
311
- * Enriches raw input with system-extracted facts, normalized titles,
312
- * resolved entities, and verified types.
313
- *
314
- * When useLLM=true, uses LLM for fact extraction (Mem0-style).
315
- * Falls back to rules-based extraction on LLM failure.
316
- */
317
- export async function runExtract(
318
- input: FormationInput,
319
- existingEntities: string[],
320
- useLLM = false,
321
- ): Promise<ExtractResult> {
322
- const callerFacts = input.facts ?? [];
323
-
324
- // 1. Extract facts from narrative
325
- let extractedFacts: string[];
326
- if (useLLM) {
327
- // LLM extraction (quality-first, Mem0-style)
328
- extractedFacts = await extractFactsWithLLM(input.narrative, input.title, callerFacts);
329
- // If LLM returned nothing, fall back to rules
330
- if (extractedFacts.length === 0) {
331
- extractedFacts = extractFacts(input.narrative, callerFacts);
332
- }
333
- } else {
334
- // Rules-based extraction (free mode)
335
- extractedFacts = extractFacts(input.narrative, callerFacts);
336
- }
337
- const allFacts = [...callerFacts, ...extractedFacts];
338
-
339
- // 2. Improve title if generic
340
- const { title, improved: titleImproved } = improveTitle(input.title, input.narrative);
341
-
342
- // 3. Resolve entity name
343
- const { entityName, resolved: entityResolved } = resolveEntity(
344
- input.entityName,
345
- existingEntities,
346
- );
347
-
348
- // 4. Verify observation type
349
- const { type, corrected: typeCorrected } = verifyType(
350
- input.type,
351
- input.narrative,
352
- input.title,
353
- );
354
-
355
- return {
356
- title,
357
- titleImproved,
358
- narrative: input.narrative,
359
- facts: allFacts,
360
- extractedFacts,
361
- entityName,
362
- entityResolved,
363
- type,
364
- typeCorrected,
365
- };
366
- }
31
+ // Version numbers (e.g., "v1.2.3", "version 2.0")
32
+ {
33
+ pattern: /\b(?:v(?:ersion)?\s*)(\d+\.\d+(?:\.\d+)?(?:-[\w.]+)?)\b/gi,
34
+ format: (m) => `Version: ${m[1]}`,
35
+ },
36
+ // Error messages (e.g., "Error: ...", "ERR_...")
37
+ {
38
+ pattern: /\b(?:Error|ERR|ENOENT|ECONNREFUSED|TypeError|RangeError|SyntaxError|ReferenceError)[:\s]+([^\n]{5,80})/gi,
39
+ format: (m) => `Error: ${m[1].trim()}`,
40
+ },
41
+ // Port numbers in context
42
+ {
43
+ pattern: /\b(?:port|PORT)\s*[:=]?\s*(\d{2,5})\b/gi,
44
+ format: (m) => `Port: ${m[1]}`,
45
+ },
46
+ // Environment variables
47
+ {
48
+ pattern: /\b([A-Z][A-Z0-9_]{3,30})\s*=\s*(\S{1,60})/g,
49
+ format: (m) => `${m[1]}=${m[2]}`,
50
+ },
51
+ // npm/package versions (e.g., "react@18.2.0")
52
+ {
53
+ pattern: /\b([@a-z][\w./-]+)@(\d+\.\d+\.\d+(?:-[\w.]+)?)\b/g,
54
+ format: (m) => `${m[1]}@${m[2]}`,
55
+ },
56
+ ];
57
+
58
+ /** Patterns indicating generic/low-quality titles that should be improved */
59
+ const GENERIC_TITLE_PATTERNS = [
60
+ /^Updated \S+\.\w+$/i,
61
+ /^Created \S+\.\w+$/i,
62
+ /^Deleted \S+\.\w+$/i,
63
+ /^Modified \S+\.\w+$/i,
64
+ /^Changed \S+\.\w+$/i,
65
+ /^Session activity/i,
66
+ /^Activity \(/i,
67
+ /^Used \w+$/i,
68
+ /^Ran: /i,
69
+ ];
70
+
71
+ /** Content signals mapped to observation types */
72
+ const TYPE_SIGNALS: Array<{ type: ObservationType; patterns: RegExp[] }> = [
73
+ {
74
+ type: 'problem-solution',
75
+ patterns: [
76
+ /\b(fix|fixed|bug|error|issue|crash|broken|resolved|workaround|patch)\b/i,
77
+ /\b(修复|修正|解决|报错|崩溃|异常)\b/,
78
+ ],
79
+ },
80
+ {
81
+ type: 'gotcha',
82
+ patterns: [
83
+ /\b(gotcha|pitfall|trap|careful|warning|caveat|footgun|unexpected|beware)\b/i,
84
+ /\b(坑|陷阱|注意|小心|踩坑)\b/,
85
+ ],
86
+ },
87
+ {
88
+ type: 'decision',
89
+ patterns: [
90
+ /\b(decided|chose|chosen|selected|adopted|rejected|evaluated|compared)\b/i,
91
+ /\b(决定|选择|采用|弃用|对比|评估)\b/,
92
+ ],
93
+ },
94
+ {
95
+ type: 'what-changed',
96
+ patterns: [
97
+ /\b(changed|migrated|upgraded|refactored|replaced|renamed|moved|removed|added)\b/i,
98
+ /\b(改|迁移|升级|重构|替换|重命名|删除|新增)\b/,
99
+ ],
100
+ },
101
+ {
102
+ type: 'how-it-works',
103
+ patterns: [
104
+ /\b(works by|architecture|mechanism|pipeline|flow|under the hood|internally)\b/i,
105
+ /\b(原理|机制|流程|架构|内部)\b/,
106
+ ],
107
+ },
108
+ {
109
+ type: 'trade-off',
110
+ patterns: [
111
+ /\b(trade.?off|compromise|downside|cost|benefit|pro|con|versus|vs)\b/i,
112
+ /\b(权衡|折中|代价|收益|优缺点)\b/,
113
+ ],
114
+ },
115
+ ];
116
+
117
+ // ── Extract Implementation ───────────────────────────────────────
118
+
119
+ /**
120
+ * Extract structured facts from narrative text using regex patterns.
121
+ * Returns only facts not already present in the caller-provided list.
122
+ */
123
+ function extractFacts(narrative: string, existingFacts: string[]): string[] {
124
+ const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
125
+ const extracted: string[] = [];
126
+ const seen = new Set<string>();
127
+
128
+ for (const { pattern, format } of FACT_PATTERNS) {
129
+ pattern.lastIndex = 0;
130
+ let match: RegExpExecArray | null;
131
+ while ((match = pattern.exec(narrative)) !== null) {
132
+ const fact = format(match);
133
+ const normalized = fact.toLowerCase().trim();
134
+
135
+ // Skip if already provided by caller or already extracted
136
+ if (existingLower.has(normalized) || seen.has(normalized)) continue;
137
+
138
+ // Skip very short or very long facts
139
+ if (fact.length < 5 || fact.length > 120) continue;
140
+
141
+ seen.add(normalized);
142
+ extracted.push(fact);
143
+ }
144
+ }
145
+
146
+ return extracted.slice(0, 10); // Cap at 10 system-extracted facts
147
+ }
148
+
149
+ /**
150
+ * Improve a generic title by extracting the first meaningful sentence
151
+ * from the narrative.
152
+ */
153
+ function improveTitle(title: string, narrative: string): { title: string; improved: boolean } {
154
+ const isGeneric = GENERIC_TITLE_PATTERNS.some(p => p.test(title));
155
+ if (!isGeneric) return { title, improved: false };
156
+
157
+ // Try to extract first meaningful sentence from narrative
158
+ const sentences = narrative
159
+ .replace(/```[\s\S]*?```/g, '') // Remove code blocks
160
+ .split(/[.。!!?\n]/)
161
+ .map(s => s.trim())
162
+ .filter(s => s.length >= 15);
163
+
164
+ if (sentences.length > 0) {
165
+ return { title: sentences[0].slice(0, 60), improved: true };
166
+ }
167
+
168
+ return { title, improved: false };
169
+ }
170
+
171
+ /**
172
+ * Resolve entity name against existing Knowledge Graph entities.
173
+ * If a close match is found, use the canonical entity name.
174
+ */
175
+ function resolveEntity(
176
+ entityName: string,
177
+ existingEntities: string[],
178
+ ): { entityName: string; resolved: boolean } {
179
+ if (existingEntities.length === 0) return { entityName, resolved: false };
180
+
181
+ const lower = entityName.toLowerCase().replace(/[-_]/g, '');
182
+
183
+ for (const existing of existingEntities) {
184
+ const existingLower = existing.toLowerCase().replace(/[-_]/g, '');
185
+
186
+ // Exact match (case-insensitive, ignoring hyphens/underscores)
187
+ if (lower === existingLower) {
188
+ return { entityName: existing, resolved: existing !== entityName };
189
+ }
190
+
191
+ // Substring match: one contains the other (e.g., "auth" matches "auth-module")
192
+ if (lower.length >= 3 && existingLower.length >= 3) {
193
+ if (existingLower.includes(lower) || lower.includes(existingLower)) {
194
+ // Prefer the longer (more specific) name
195
+ const canonical = existing.length >= entityName.length ? existing : entityName;
196
+ return { entityName: canonical, resolved: canonical !== entityName };
197
+ }
198
+ }
199
+ }
200
+
201
+ return { entityName, resolved: false };
202
+ }
203
+
204
+ /**
205
+ * Verify observation type against content signals.
206
+ * If content strongly suggests a different type, correct it.
207
+ */
208
+ function verifyType(
209
+ declaredType: ObservationType,
210
+ narrative: string,
211
+ title: string,
212
+ ): { type: ObservationType; corrected: boolean } {
213
+ const content = `${title} ${narrative}`;
214
+
215
+ // Score each type by counting individual keyword hits across all patterns
216
+ const scores: Array<{ type: ObservationType; score: number }> = [];
217
+ for (const { type, patterns } of TYPE_SIGNALS) {
218
+ let score = 0;
219
+ for (const p of patterns) {
220
+ // Use matchAll to count individual keyword matches
221
+ const regex = new RegExp(p.source, p.flags.includes('g') ? p.flags : p.flags + 'g');
222
+ const matches = [...content.matchAll(regex)];
223
+ score += matches.length;
224
+ }
225
+ if (score > 0) scores.push({ type, score });
226
+ }
227
+
228
+ if (scores.length === 0) return { type: declaredType, corrected: false };
229
+
230
+ // Sort by score descending
231
+ scores.sort((a, b) => b.score - a.score);
232
+ const best = scores[0];
233
+
234
+ // Only correct if the best match is significantly stronger than declared type
235
+ // Requires: best type has >= 2 keyword hits AND declared type has 0 signals
236
+ if (best.type !== declaredType && best.score >= 2) {
237
+ const declaredScore = scores.find(s => s.type === declaredType)?.score ?? 0;
238
+ if (declaredScore === 0) {
239
+ return { type: best.type, corrected: true };
240
+ }
241
+ }
242
+
243
+ return { type: declaredType, corrected: false };
244
+ }
245
+
246
+ // ── LLM Fact Extraction ─────────────────────────────────────────
247
+
248
+ /** Prompt for LLM-based fact extraction (inspired by Mem0's approach) */
249
+ const LLM_EXTRACT_PROMPT = `You are a Software Engineering Knowledge Extractor.
250
+ Extract structured facts from the given development context.
251
+
252
+ Focus on:
253
+ 1. Technical decisions and their reasoning
254
+ 2. Bug root causes and fixes
255
+ 3. Configuration values (ports, versions, env vars)
256
+ 4. Architecture patterns and constraints
257
+ 5. Gotchas, pitfalls, and workarounds
258
+ 6. File paths and their roles
259
+
260
+ Rules:
261
+ - Return ONLY a JSON object with a "facts" key containing an array of strings
262
+ - Each fact should be a concise, self-contained statement
263
+ - Include specific values (versions, ports, paths) when present
264
+ - Detect the language of the input and record facts in the same language
265
+ - If no meaningful facts exist, return {"facts": []}
266
+ - Do NOT include trivial information (file read, directory listing)
267
+ - Maximum 10 facts
268
+
269
+ Example:
270
+ Input: "Fixed Redis connection leak. The pool wasn't being closed on shutdown. Added defer pool.Close() in main.go. Port 6379."
271
+ Output: {"facts": ["Redis connection leak caused by pool not closed on shutdown", "Fix: added defer pool.Close() in main.go", "Redis port: 6379"]}`;
272
+
273
+ /**
274
+ * Extract facts using LLM (Mem0-style structured extraction).
275
+ * Returns extracted facts or empty array on failure.
276
+ */
277
+ async function extractFactsWithLLM(
278
+ narrative: string,
279
+ title: string,
280
+ existingFacts: string[],
281
+ ): Promise<string[]> {
282
+ try {
283
+ const { callLLM } = await import('../../llm/provider.js');
284
+ const input = `Title: ${title}\nContent: ${narrative}${existingFacts.length > 0 ? `\nAlready known facts (don't repeat): ${existingFacts.join('; ')}` : ''}`;
285
+ const response = await callLLM(LLM_EXTRACT_PROMPT, input);
286
+ const text = response.content.trim();
287
+
288
+ // Parse JSON response
289
+ const jsonMatch = text.match(/\{[\s\S]*\}/);
290
+ if (!jsonMatch) return [];
291
+ const parsed = JSON.parse(jsonMatch[0]);
292
+ const facts = parsed.facts;
293
+ if (!Array.isArray(facts)) return [];
294
+
295
+ // Filter: remove duplicates with existing facts
296
+ const existingLower = new Set(existingFacts.map(f => f.toLowerCase().trim()));
297
+ return facts
298
+ .filter((f: unknown): f is string => typeof f === 'string' && f.length >= 5)
299
+ .filter((f: string) => !existingLower.has(f.toLowerCase().trim()))
300
+ .slice(0, 10);
301
+ } catch {
302
+ return []; // LLM failure → fall back to rules
303
+ }
304
+ }
305
+
306
+ // ── Public API ───────────────────────────────────────────────────
307
+
308
+ /**
309
+ * Run Stage 1: Extract.
310
+ *
311
+ * Enriches raw input with system-extracted facts, normalized titles,
312
+ * resolved entities, and verified types.
313
+ *
314
+ * When useLLM=true, uses LLM for fact extraction (Mem0-style).
315
+ * Falls back to rules-based extraction on LLM failure.
316
+ */
317
+ export async function runExtract(
318
+ input: FormationInput,
319
+ existingEntities: string[],
320
+ useLLM = false,
321
+ ): Promise<ExtractResult> {
322
+ const callerFacts = input.facts ?? [];
323
+
324
+ // 1. Extract facts from narrative
325
+ let extractedFacts: string[];
326
+ if (useLLM) {
327
+ // LLM extraction (quality-first, Mem0-style)
328
+ extractedFacts = await extractFactsWithLLM(input.narrative, input.title, callerFacts);
329
+ // If LLM returned nothing, fall back to rules
330
+ if (extractedFacts.length === 0) {
331
+ extractedFacts = extractFacts(input.narrative, callerFacts);
332
+ }
333
+ } else {
334
+ // Rules-based extraction (free mode)
335
+ extractedFacts = extractFacts(input.narrative, callerFacts);
336
+ }
337
+ const allFacts = [...callerFacts, ...extractedFacts];
338
+
339
+ // 2. Improve title if generic
340
+ const { title, improved: titleImproved } = improveTitle(input.title, input.narrative);
341
+
342
+ // 3. Resolve entity name
343
+ const { entityName, resolved: entityResolved } = resolveEntity(
344
+ input.entityName,
345
+ existingEntities,
346
+ );
347
+
348
+ // 4. Verify observation type
349
+ const { type, corrected: typeCorrected } = verifyType(
350
+ input.type,
351
+ input.narrative,
352
+ input.title,
353
+ );
354
+
355
+ return {
356
+ title,
357
+ titleImproved,
358
+ narrative: input.narrative,
359
+ facts: allFacts,
360
+ extractedFacts,
361
+ entityName,
362
+ entityResolved,
363
+ type,
364
+ typeCorrected,
365
+ };
366
+ }