@wooojin/forgen 0.4.9 → 0.4.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/CHANGELOG.md +46 -0
  3. package/README.md +33 -1
  4. package/assets/claude/agents/forgen-verify.md +65 -0
  5. package/assets/claude/workflows/compound-extract.js +136 -0
  6. package/assets/claude/workflows/evidence-gate-audit.js +107 -0
  7. package/assets/shared/hook-registry.json +1 -0
  8. package/dist/checks/_shared/meta-guard-dispatch.d.ts +38 -0
  9. package/dist/checks/_shared/meta-guard-dispatch.js +80 -0
  10. package/dist/checks/_shared/text-sanitizer.js +15 -0
  11. package/dist/cli.js +65 -2
  12. package/dist/core/changelog-cli.d.ts +7 -0
  13. package/dist/core/changelog-cli.js +100 -0
  14. package/dist/core/doctor.d.ts +3 -0
  15. package/dist/core/doctor.js +75 -1
  16. package/dist/core/effort-advisory.d.ts +23 -0
  17. package/dist/core/effort-advisory.js +29 -0
  18. package/dist/core/explain-cli.d.ts +6 -0
  19. package/dist/core/explain-cli.js +99 -0
  20. package/dist/core/git-stats.d.ts +23 -0
  21. package/dist/core/git-stats.js +50 -0
  22. package/dist/core/health-cli.d.ts +23 -0
  23. package/dist/core/health-cli.js +86 -0
  24. package/dist/core/inspect-cli.js +60 -1
  25. package/dist/core/probe-workflow-cli.d.ts +72 -0
  26. package/dist/core/probe-workflow-cli.js +282 -0
  27. package/dist/core/regress-map-cli.d.ts +7 -0
  28. package/dist/core/regress-map-cli.js +59 -0
  29. package/dist/core/stats-cli.d.ts +22 -9
  30. package/dist/core/stats-cli.js +149 -0
  31. package/dist/core/watch-cli.d.ts +7 -0
  32. package/dist/core/watch-cli.js +185 -0
  33. package/dist/core/workflows-cli.d.ts +26 -0
  34. package/dist/core/workflows-cli.js +120 -0
  35. package/dist/engine/compound-export.d.ts +12 -0
  36. package/dist/engine/compound-export.js +136 -14
  37. package/dist/engine/compound-extractor.d.ts +12 -43
  38. package/dist/engine/compound-extractor.js +27 -756
  39. package/dist/engine/extraction-diff.d.ts +11 -0
  40. package/dist/engine/extraction-diff.js +105 -0
  41. package/dist/engine/extraction-gates.d.ts +37 -0
  42. package/dist/engine/extraction-gates.js +100 -0
  43. package/dist/engine/extraction-git.d.ts +20 -0
  44. package/dist/engine/extraction-git.js +75 -0
  45. package/dist/engine/extraction-persistence.d.ts +27 -0
  46. package/dist/engine/extraction-persistence.js +140 -0
  47. package/dist/engine/extraction-session.d.ts +26 -0
  48. package/dist/engine/extraction-session.js +230 -0
  49. package/dist/engine/lifecycle/types.d.ts +1 -1
  50. package/dist/engine/meta-learning/matcher-weight-loader.d.ts +16 -0
  51. package/dist/engine/meta-learning/matcher-weight-loader.js +45 -0
  52. package/dist/engine/precision-guards.d.ts +14 -0
  53. package/dist/engine/precision-guards.js +39 -0
  54. package/dist/engine/ranking-pipeline.d.ts +45 -0
  55. package/dist/engine/ranking-pipeline.js +66 -0
  56. package/dist/engine/relevance-scorer.d.ts +43 -0
  57. package/dist/engine/relevance-scorer.js +81 -0
  58. package/dist/engine/scoring-algorithms.d.ts +31 -0
  59. package/dist/engine/scoring-algorithms.js +109 -0
  60. package/dist/engine/solution-matcher-eval.d.ts +97 -0
  61. package/dist/engine/solution-matcher-eval.js +122 -0
  62. package/dist/engine/solution-matcher.d.ts +21 -380
  63. package/dist/engine/solution-matcher.js +27 -828
  64. package/dist/fgx.d.ts +4 -1
  65. package/dist/fgx.js +46 -10
  66. package/dist/hooks/notepad-injector.js +7 -0
  67. package/dist/hooks/post-tool-use.js +8 -1
  68. package/dist/hooks/shared/preflight-check.d.ts +15 -0
  69. package/dist/hooks/shared/preflight-check.js +51 -0
  70. package/dist/hooks/stop-guard.js +19 -60
  71. package/dist/hooks/subagent-stop-guard.d.ts +23 -0
  72. package/dist/hooks/subagent-stop-guard.js +158 -0
  73. package/dist/hooks/subagent-tracker.d.ts +36 -3
  74. package/dist/hooks/subagent-tracker.js +86 -39
  75. package/dist/store/evidence-store.js +9 -2
  76. package/hooks/hooks.json +6 -1
  77. package/package.json +7 -7
  78. package/plugin.json +1 -1
  79. package/scripts/postinstall.js +10 -7
@@ -1,649 +1,32 @@
1
1
  /**
2
- * Forgen — Compound Knowledge Extractor
2
+ * Forgen — Compound Knowledge Extractor (facade)
3
3
  *
4
- * Extracts reusable patterns and decisions from git history and session context.
5
- * Runs quality gates (structure, toxicity, trivial, dedup) before persisting solutions.
4
+ * Orchestrates extraction pipeline: git analysis → quality gates →
5
+ * pattern extraction → persistence. Re-exports from decomposed modules.
6
6
  *
7
- * Module Structure:
8
- * - Lines 1-50: Imports, constants, SHA validation, LastExtraction/ExtractedSolution interfaces
9
- * - Lines 50-115: Git helpers — getNewCommits, getCommitMessages, getGitDiff, getDiffStats
10
- * - Lines 115-190: Quality Gates — gate0 (worth extracting), gate1 (structure), gate2 (toxicity),
11
- * gateTrivial (trivial rejection), gate3 (dedup)
12
- * - Lines 190-275: extractFromDiff — pattern extraction from git diff (modules, errors, imports, commits)
13
- * - Lines 275-395: extractFromSessionContext — prompt/write history analysis (actions, hotspots, tech)
14
- * - Lines 396-475: saveExtractedSolution, updateReExtractedCounter — solution persistence
15
- * - Lines 477-555: runExtraction — main entry point orchestrating gates + extraction + state
16
- * - Lines 557-634: processExtractionResults, isExtractionPaused, pauseExtraction, resumeExtraction
7
+ * Module layout (post-decomposition):
8
+ * extraction-git.ts — getNewCommits, getCommitMessages, getGitDiff, getDiffStats
9
+ * extraction-gates.ts — gate0-gate3, gateTrivial, evaluateExtractedSolution, ExtractedSolution
10
+ * extraction-diff.ts — extractFromDiff, findCommonPrefix
11
+ * extraction-session.ts — extractFromSessionContext, loadClaudeProjectSessionContext
12
+ * extraction-persistence.ts — saveExtractedSolution, updateReExtractedCounter, LastExtraction
17
13
  */
18
14
  import * as fs from 'node:fs';
19
15
  import * as path from 'node:path';
20
16
  import { execFileSync } from 'node:child_process';
21
17
  import { execHost } from '../host/exec-host.js';
22
- import { serializeSolutionV3, DEFAULT_EVIDENCE, extractTags } from './solution-format.js';
23
18
  import { createLogger } from '../core/logger.js';
24
- import { emitSolutionEvent } from '../core/observability-store.js';
19
+ import { STATE_DIR } from '../core/paths.js';
20
+ import { getNewCommits, getCommitMessages, getGitDiff, getDiffStats } from './extraction-git.js';
21
+ import { gate0, evaluateExtractedSolution } from './extraction-gates.js';
22
+ import { extractFromDiff } from './extraction-diff.js';
23
+ import { extractFromSessionContext } from './extraction-session.js';
24
+ import { loadLastExtraction, saveLastExtraction, saveExtractedSolution, updateReExtractedCounter, emitCompoundExtractActedOn, } from './extraction-persistence.js';
25
25
  const log = createLogger('compound-extractor');
26
- import { CLAUDE_DIR, ME_SOLUTIONS, STATE_DIR } from '../core/paths.js';
27
- import { atomicWriteJSON, atomicWriteText } from '../hooks/shared/atomic-write.js';
28
- import { mutateSolutionFile } from './solution-writer.js';
29
- const LAST_EXTRACTION_PATH = path.join(STATE_DIR, 'last-extraction.json');
30
26
  const MAX_EXTRACTIONS_PER_DAY = 5;
31
- const MAX_DIFF_LENGTH = 3000;
32
- /** Validate that a string is a valid git SHA (7-64 hex chars) */
33
- function isValidSha(sha) {
34
- return /^[a-f0-9]{7,64}$/.test(sha);
35
- }
36
- /** Load last extraction state */
37
- function loadLastExtraction() {
38
- try {
39
- if (fs.existsSync(LAST_EXTRACTION_PATH)) {
40
- return JSON.parse(fs.readFileSync(LAST_EXTRACTION_PATH, 'utf-8'));
41
- }
42
- }
43
- catch (e) {
44
- log.debug('last extraction state read failed — may cause duplicate extractions', e);
45
- }
46
- return { lastCommitSha: '', lastExtractedAt: '', extractionsToday: 0, todayDate: '' };
47
- }
48
- /** Save last extraction state */
49
- function saveLastExtraction(state) {
50
- fs.mkdirSync(STATE_DIR, { recursive: true });
51
- atomicWriteJSON(LAST_EXTRACTION_PATH, state);
52
- }
53
- /** Get new commits since last extraction — uses execFileSync to prevent injection */
54
- function getNewCommits(cwd, lastSha) {
55
- try {
56
- if (!lastSha || !isValidSha(lastSha)) {
57
- return execFileSync('git', ['log', '--oneline', '-5'], { cwd, encoding: 'utf-8', timeout: 5000, stdio: ['pipe', 'pipe', 'pipe'] });
58
- }
59
- return execFileSync('git', ['log', '--oneline', `${lastSha}..HEAD`], { cwd, encoding: 'utf-8', timeout: 5000, stdio: ['pipe', 'pipe', 'pipe'] });
60
- }
61
- catch {
62
- return '';
63
- }
64
- }
65
- /** Get commit messages for "why" context enrichment */
66
- function getCommitMessages(cwd, lastSha) {
67
- try {
68
- const args = lastSha && isValidSha(lastSha)
69
- ? ['log', '--format=%B', `${lastSha}..HEAD`]
70
- : ['log', '--format=%B', '-5'];
71
- const msgs = execFileSync('git', args, { cwd, encoding: 'utf-8', timeout: 5000 });
72
- return msgs.slice(0, 1000).trim();
73
- }
74
- catch {
75
- return '';
76
- }
77
- }
78
- /** Get git diff for extraction */
79
- function getGitDiff(cwd, lastSha) {
80
- try {
81
- const args = lastSha && isValidSha(lastSha)
82
- ? ['diff', `${lastSha}..HEAD`]
83
- : ['diff', 'HEAD~1'];
84
- const diff = execFileSync('git', args, { cwd, encoding: 'utf-8', timeout: 10000 });
85
- return diff.slice(0, MAX_DIFF_LENGTH);
86
- }
87
- catch {
88
- return '';
89
- }
90
- }
91
- /** Get diff stats for Gate 0 */
92
- function getDiffStats(cwd, lastSha) {
93
- try {
94
- const args = lastSha && isValidSha(lastSha)
95
- ? ['diff', '--stat', `${lastSha}..HEAD`]
96
- : ['diff', '--stat', 'HEAD~1'];
97
- const stat = execFileSync('git', args, { cwd, encoding: 'utf-8', timeout: 5000 });
98
- const lines = stat.split('\n').filter(l => l.trim());
99
- const codeExts = /\.(ts|tsx|js|jsx|py|rs|go|java|rb|c|cpp|h|swift|kt)$/;
100
- const hasCodeFiles = lines.some(line => {
101
- const filePath = line.split('|')[0]?.trim() ?? '';
102
- return codeExts.test(filePath);
103
- });
104
- const lastLine = lines[lines.length - 1] ?? '';
105
- const changedMatch = lastLine.match(/(\d+)\s+files?\s+changed/);
106
- const insertMatch = lastLine.match(/(\d+)\s+insertion/);
107
- const deleteMatch = lastLine.match(/(\d+)\s+deletion/);
108
- const fileCount = parseInt(changedMatch?.[1] ?? '0', 10);
109
- const lineCount = parseInt(insertMatch?.[1] ?? '0', 10) + parseInt(deleteMatch?.[1] ?? '0', 10);
110
- return { files: fileCount, lines: lineCount, hasCodeFiles };
111
- }
112
- catch {
113
- return { files: 0, lines: 0, hasCodeFiles: false };
114
- }
115
- }
116
- // --- Blocklist for Gate 2 (Toxicity Filter) ---
117
- const TOXICITY_PATTERNS = [
118
- /@ts-ignore/i, /@ts-nocheck/i, /as\s+any\b/i,
119
- /--force\b/i, /--no-verify\b/i, /--skip-ci\b/i,
120
- /eslint-disable/i, /prettier-ignore/i, /noqa/i,
121
- /\bTODO:/i, /\bFIXME:/i, /\bHACK:/i, /\bXXX:/i,
122
- /\/Users\//i, /\/home\//i, /C:\\\\Users/i,
123
- ];
124
- // --- Quality Gates ---
125
- /** Gate 0: Is this extraction worth doing? */
126
- function gate0(stats) {
127
- if (stats.files < 1)
128
- return false;
129
- if (stats.lines < 30)
130
- return false;
131
- if (!stats.hasCodeFiles)
132
- return false;
133
- return true;
134
- }
135
- /** Gate 1: Structural validation (pure — does not mutate input) */
136
- function gate1(sol) {
137
- if (!sol.name || sol.name.length < 3)
138
- return false;
139
- if (!sol.tags || sol.tags.length === 0)
140
- return false;
141
- if (!sol.content || sol.content.length < 50)
142
- return false;
143
- if (!sol.context)
144
- return false;
145
- return true;
146
- }
147
- /** Gate 2: Toxicity filter */
148
- function gate2(sol) {
149
- const text = `${sol.context} ${sol.content}`;
150
- return !TOXICITY_PATTERNS.some(p => p.test(text));
151
- }
152
- /**
153
- * Gate 2.5: Trivial pattern rejection — 자명한 패턴은 축적할 가치 없음.
154
- * "주로 TypeScript를 작성합니다" 수준의 솔루션은 Claude가 코드를 보면 알 수 있으므로
155
- * compound에 저장하면 컨텍스트만 낭비됨.
156
- */
157
- function gateTrivial(sol) {
158
- const content = sol.content.trim();
159
- // 내용이 너무 짧으면 자명함 (한 줄짜리)
160
- if (content.length < 80)
161
- return false;
162
- // "주로 X를 Y합니다" 패턴
163
- if (/^주로\s/.test(content) && content.split('\n').length < 3)
164
- return false;
165
- // 식별자가 하나도 없으면 구체적인 기술 패턴이 아님
166
- if (sol.identifiers.length === 0 && sol.tags.length < 3)
167
- return false;
168
- return true;
169
- }
170
- /** Gate 3: Dedup check against existing solutions */
171
- function gate3(sol) {
172
- if (!fs.existsSync(ME_SOLUTIONS))
173
- return 'new';
174
- try {
175
- const files = fs.readdirSync(ME_SOLUTIONS).filter(f => f.endsWith('.md'));
176
- for (const file of files) {
177
- const content = fs.readFileSync(path.join(ME_SOLUTIONS, file), 'utf-8');
178
- const tagMatch = content.match(/tags:\s*\[([^\]]*)\]/);
179
- if (!tagMatch)
180
- continue;
181
- const existingTags = tagMatch[1].split(',').map(t => t.trim().replace(/"/g, ''));
182
- const overlap = sol.tags.filter(t => existingTags.includes(t));
183
- const overlapRatio = overlap.length / Math.max(sol.tags.length, existingTags.length, 1);
184
- if (overlapRatio >= 0.7) {
185
- if (content.includes('status: "experiment"') || content.includes("status: 'experiment'") || content.includes('status: experiment')) {
186
- return 're-extract';
187
- }
188
- return 'duplicate';
189
- }
190
- }
191
- }
192
- catch (e) {
193
- log.debug('gate3 기존 솔루션 파일 읽기 실패 — new로 간주', e);
194
- }
195
- return 'new';
196
- }
197
- /** Simple local extraction from git diff (no LLM needed) */
198
- function extractFromDiff(gitLog, gitDiff) {
199
- const solutions = [];
200
- // 1. Detect new files/modules created
201
- const newFiles = gitDiff.match(/^\+\+\+ b\/(.+)$/gm);
202
- if (newFiles && newFiles.length >= 2) {
203
- const fileNames = newFiles.map(f => f.replace('+++ b/', ''));
204
- const ext = path.extname(fileNames[0]);
205
- const dir = path.dirname(fileNames[0]).split('/').pop() ?? '';
206
- if (ext && dir) {
207
- const basenames = fileNames.map(f => path.basename(f, ext));
208
- const commonPrefix = findCommonPrefix(basenames);
209
- if (commonPrefix.length >= 3) {
210
- solutions.push({
211
- name: `module-${commonPrefix}-pattern`,
212
- type: 'pattern',
213
- tags: extractTags(`${fileNames.join(' ')} ${dir}`),
214
- identifiers: basenames.filter(b => b.length >= 4).slice(0, 5),
215
- context: `File organization pattern in ${dir}/`,
216
- content: `Files follow the naming pattern: ${commonPrefix}*${ext} in ${dir}/`,
217
- });
218
- }
219
- }
220
- }
221
- // 2. Detect error handling patterns from diff
222
- const errorPatterns = gitDiff.match(/^\+.*(?:try\s*\{|catch\s*[({]|\.catch\(|throw new|Error\()/gm);
223
- if (errorPatterns && errorPatterns.length >= 3) {
224
- const sample = errorPatterns.slice(0, 3).map(l => l.replace(/^\+\s*/, '').trim());
225
- solutions.push({
226
- name: 'error-handling-pattern',
227
- type: 'pattern',
228
- tags: ['error', 'handling', 'try-catch', 'pattern'],
229
- identifiers: sample.filter(s => s.length >= 4).slice(0, 3),
230
- context: 'Error handling approach used in this codebase',
231
- content: `Consistent error handling: ${sample.join('; ')}`.slice(0, 500),
232
- });
233
- }
234
- // 3. Detect import/dependency patterns
235
- const imports = gitDiff.match(/^\+\s*import\s+.+from\s+['"]([^'"]+)['"]/gm);
236
- if (imports && imports.length >= 3) {
237
- const packages = imports
238
- .map(i => i.match(/from\s+['"]([^'"]+)['"]/)?.[1])
239
- .filter((p) => !!p && !p.startsWith('.'))
240
- .filter((v, i, a) => a.indexOf(v) === i);
241
- if (packages.length >= 2) {
242
- solutions.push({
243
- name: 'dependency-stack',
244
- type: 'decision',
245
- tags: ['dependency', 'stack', ...packages.slice(0, 3)],
246
- identifiers: packages.filter(p => p.length >= 4).slice(0, 5),
247
- context: 'Technology stack and dependency choices',
248
- content: `Project uses: ${packages.join(', ')}`,
249
- });
250
- }
251
- }
252
- // 4. Detect from commit messages
253
- const commitKeywords = {
254
- 'fix': { type: 'troubleshoot', tags: ['bugfix', 'troubleshoot'] },
255
- 'refactor': { type: 'pattern', tags: ['refactor', 'cleanup'] },
256
- 'test': { type: 'pattern', tags: ['testing', 'tdd'] },
257
- 'security': { type: 'pattern', tags: ['security', 'hardening'] },
258
- };
259
- for (const [keyword, meta] of Object.entries(commitKeywords)) {
260
- const re = new RegExp(`^[a-f0-9]+\\s+${keyword}[:\\s](.+)$`, 'gim');
261
- const matches = [...gitLog.matchAll(re)];
262
- if (matches.length >= 2) {
263
- const descriptions = matches.map(m => m[1].trim()).slice(0, 3);
264
- // commit 메시지에서 identifier 후보 추출 (camelCase, PascalCase, snake_case, 6자 이상)
265
- const commitIdentifiers = descriptions
266
- .join(' ')
267
- .match(/\b[a-zA-Z][a-zA-Z0-9]*(?:[A-Z][a-z]+)+\b|\b[a-z]+(?:_[a-z]+)+\b/g)
268
- ?.filter(id => id.length >= 6)
269
- ?.filter((v, i, a) => a.indexOf(v) === i)
270
- ?.slice(0, 5) ?? [];
271
- solutions.push({
272
- name: `${keyword}-pattern`,
273
- type: meta.type,
274
- tags: [...meta.tags, keyword],
275
- identifiers: commitIdentifiers,
276
- context: `Recurring ${keyword} pattern from commit history`,
277
- content: descriptions.join('. ').slice(0, 500),
278
- });
279
- }
280
- }
281
- return solutions.slice(0, 3); // max 3
282
- }
283
- /** Extract patterns from accumulated session context (prompts + writes + diff) */
284
- function extractFromSessionContext(gitDiff, cwd, lastExtractedAt) {
285
- const solutions = [];
286
- const claudeContext = loadClaudeProjectSessionContext(cwd, lastExtractedAt);
287
- // Load recent prompts (still consumed by tech-stack-decision below).
288
- let prompts = claudeContext.prompts;
289
- if (prompts.length === 0) {
290
- prompts = loadPromptHistoryFallback();
291
- }
292
- // C4 removal (2026-04-09): `recurring-task-pattern` and `modification-
293
- // hotspot` extractors were deleted here. They produced word-frequency
294
- // histograms and directory counts masquerading as "patterns", with
295
- // generic "consider automating/refactoring" advice that applied to
296
- // any project. Observed in production: one `recurring-task-pattern`
297
- // solution whose entire content was `User frequently requests:
298
- // test(39회), 테스트(36회), 추가(32회). Consider automating...` was
299
- // injected into 105 sessions before the quality problem was spotted.
300
- // The `extractFromSessionContext` function now only emits the
301
- // tech-stack-decision pattern below, which at least cross-validates
302
- // against both prompts and diff before writing. If session-level
303
- // extraction needs to come back, the replacement MUST pass a
304
- // content-level sniff test: does the extracted solution teach
305
- // something a new developer wouldn't already infer from `git log`?
306
- //
307
- // M-1 (review follow-up): the `writes` loader was removed from this
308
- // function as dead code. Pre-M-1 `loadWriteHistoryFallback()` ran on
309
- // every extraction, touching the filesystem to load data no
310
- // downstream extractor consumed. If a future extractor needs writes,
311
- // restore `claudeContext.writes` (already loaded above) or reintroduce
312
- // the fallback loader at that point.
313
- // 3. Detect decision patterns from prompt + diff correlation
314
- // When user asks about X and diff shows Y, the decision is "for X, use Y"
315
- const techDecisions = [];
316
- const techTerms = ['react', 'vue', 'next', 'express', 'fastify', 'prisma', 'drizzle', 'zustand', 'redux', 'tailwind', 'styled', 'vitest', 'jest', 'playwright', 'cypress'];
317
- for (const term of techTerms) {
318
- const inPrompts = prompts.some(p => p.toLowerCase().includes(term));
319
- const inDiff = gitDiff.toLowerCase().includes(term);
320
- if (inPrompts && inDiff) {
321
- techDecisions.push(term);
322
- }
323
- }
324
- if (techDecisions.length >= 2) {
325
- solutions.push({
326
- name: 'tech-stack-decision',
327
- type: 'decision',
328
- tags: ['stack', 'technology', ...techDecisions.slice(0, 5)],
329
- identifiers: techDecisions.filter(t => t.length >= 4).slice(0, 5),
330
- context: 'Technology choices confirmed by both discussion and implementation',
331
- content: `Active technology stack: ${techDecisions.join(', ')}. Both discussed in prompts and present in code changes.`,
332
- });
333
- }
334
- return solutions;
335
- }
336
- function normalizeProjectPath(cwd) {
337
- const resolved = path.resolve(cwd);
338
- try {
339
- return typeof fs.realpathSync.native === 'function'
340
- ? fs.realpathSync.native(resolved)
341
- : fs.realpathSync(resolved);
342
- }
343
- catch {
344
- return resolved;
345
- }
346
- }
347
- function getProjectPathCandidates(cwd) {
348
- const resolved = path.resolve(cwd);
349
- const candidates = new Set([resolved, normalizeProjectPath(cwd)]);
350
- try {
351
- if (fs.lstatSync(resolved).isSymbolicLink()) {
352
- candidates.add(path.resolve(path.dirname(resolved), fs.readlinkSync(resolved)));
353
- }
354
- }
355
- catch {
356
- // Ignore lstat/readlink failures; raw + realpath candidates are enough.
357
- }
358
- for (const candidate of [...candidates]) {
359
- candidates.add(normalizeProjectPath(candidate));
360
- }
361
- return [...candidates];
362
- }
363
- function getClaudeProjectDirs(cwd) {
364
- return getProjectPathCandidates(cwd)
365
- .map(candidate => path.join(CLAUDE_DIR, 'projects', candidate.replace(/[:\\/]/g, '-')));
366
- }
367
- function listClaudeSessionFiles(projectDirs, maxFiles) {
368
- // Symlink hardening (INFO from security review, 2026-04-09):
369
- // `~/.claude/projects/` is inside the user's HOME so in the normal
370
- // threat model it's trusted, but we mirror the `solution-index.ts:135`
371
- // defensive posture and refuse to follow symlinks. A local attacker
372
- // with HOME write access could otherwise plant a symlink pointing at
373
- // arbitrary JSONL files on disk and cause `collectClaudeProjectSessionContext`
374
- // to ingest their contents as "Claude session prompts".
375
- return projectDirs
376
- .flatMap(projectDir => {
377
- let entries;
378
- try {
379
- entries = fs.readdirSync(projectDir);
380
- }
381
- catch {
382
- return [];
383
- }
384
- const out = [];
385
- for (const file of entries) {
386
- if (!file.endsWith('.jsonl'))
387
- continue;
388
- const filePath = path.join(projectDir, file);
389
- try {
390
- if (fs.lstatSync(filePath).isSymbolicLink())
391
- continue;
392
- out.push({ filePath, mtimeMs: fs.statSync(filePath).mtimeMs });
393
- }
394
- catch {
395
- // unreadable / vanished between readdir and stat — skip
396
- }
397
- }
398
- return out;
399
- })
400
- .sort((a, b) => b.mtimeMs - a.mtimeMs)
401
- .slice(0, maxFiles);
402
- }
403
- function getAllClaudeProjectDirs() {
404
- const projectsRoot = path.join(CLAUDE_DIR, 'projects');
405
- if (!fs.existsSync(projectsRoot))
406
- return [];
407
- return fs.readdirSync(projectsRoot)
408
- .map(name => path.join(projectsRoot, name))
409
- .filter(dir => {
410
- try {
411
- return fs.statSync(dir).isDirectory();
412
- }
413
- catch {
414
- return false;
415
- }
416
- });
417
- }
418
- function collectClaudeProjectSessionContext(files, cwdCandidates, cutoffMs) {
419
- const prompts = [];
420
- const writes = [];
421
- for (const file of files) {
422
- // Defense in depth: even though listClaudeSessionFiles already
423
- // rejects symlinks, re-check here in case a caller bypasses the
424
- // lister. A TOCTOU race between lister's lstat and this read is
425
- // theoretically possible but requires local HOME write access,
426
- // at which point the attacker already has easier vectors.
427
- try {
428
- if (fs.lstatSync(file.filePath).isSymbolicLink())
429
- continue;
430
- }
431
- catch {
432
- continue;
433
- }
434
- let lines;
435
- try {
436
- lines = fs.readFileSync(file.filePath, 'utf-8').split('\n').filter(Boolean);
437
- }
438
- catch {
439
- continue;
440
- }
441
- for (const line of lines) {
442
- let entry;
443
- try {
444
- entry = JSON.parse(line);
445
- }
446
- catch {
447
- continue;
448
- }
449
- const entryCandidates = typeof entry.cwd === 'string' ? getProjectPathCandidates(entry.cwd) : [];
450
- if (!entryCandidates.some(candidate => cwdCandidates.has(candidate)))
451
- continue;
452
- const timestamp = typeof entry.timestamp === 'string' ? new Date(entry.timestamp).getTime() : Number.NaN;
453
- if (cutoffMs && Number.isFinite(timestamp) && timestamp <= cutoffMs)
454
- continue;
455
- if (entry.type === 'user') {
456
- const message = entry.message;
457
- if (message?.role === 'user' && typeof message.content === 'string') {
458
- prompts.push(message.content);
459
- }
460
- continue;
461
- }
462
- if (entry.type !== 'assistant')
463
- continue;
464
- const message = entry.message;
465
- if (message?.role !== 'assistant' || !Array.isArray(message.content))
466
- continue;
467
- for (const item of message.content) {
468
- if (typeof item !== 'object' || item === null)
469
- continue;
470
- const toolUse = item;
471
- if (toolUse.type !== 'tool_use')
472
- continue;
473
- if (toolUse.name !== 'Write' && toolUse.name !== 'Edit')
474
- continue;
475
- const filePath = String(toolUse.input?.file_path ?? toolUse.input?.filePath ?? '');
476
- const content = String(toolUse.input?.content ?? toolUse.input?.new_string ?? '');
477
- if (!filePath || !content)
478
- continue;
479
- writes.push({
480
- filePath: filePath.slice(-100),
481
- contentSnippet: content.slice(0, 200),
482
- fileExtension: path.extname(filePath).toLowerCase(),
483
- });
484
- }
485
- }
486
- }
487
- return {
488
- prompts: prompts.slice(-50),
489
- writes: writes.slice(-30),
490
- };
491
- }
492
- function loadPromptHistoryFallback() {
493
- const promptHistoryPath = path.join(STATE_DIR, 'prompt-history.jsonl');
494
- try {
495
- if (!fs.existsSync(promptHistoryPath))
496
- return [];
497
- const lines = fs.readFileSync(promptHistoryPath, 'utf-8').split('\n').filter(Boolean);
498
- return lines.slice(-50).map(l => {
499
- try {
500
- return JSON.parse(l).prompt;
501
- }
502
- catch {
503
- return '';
504
- }
505
- }).filter(Boolean);
506
- }
507
- catch (e) {
508
- log.debug('prompt-history.jsonl 읽기 실패 — session context fallback 건너뜀', e);
509
- return [];
510
- }
511
- }
512
- // M-1 follow-up (2026-04-09): `loadWriteHistoryFallback` was removed
513
- // alongside the C4 extractor cleanup. Its only caller was the now-deleted
514
- // session-context loader for writes, so keeping it would be dead code on
515
- // the extraction hot path (per-session I/O against a file that nobody
516
- // reads). If a future write-based extractor is reintroduced, either
517
- // restore this loader or call it via `claudeContext.writes` once session
518
- // correlation picks writes up.
519
- /**
520
- * Load Claude session prompts + writes correlated to `cwd`.
521
- *
522
- * Exported primarily for test assertions (the `claude-session-context`
523
- * tests need to verify that correlation picks the right project's
524
- * sessions and ignores unrelated ones). Before C4 the tests could
525
- * observe this indirectly via the now-removed `recurring-task-pattern`
526
- * extractor; now they check this loader directly. Not intended for
527
- * production callers outside the extractor pipeline.
528
- */
529
- export function loadClaudeProjectSessionContext(cwd, lastExtractedAt) {
530
- const cwdCandidates = new Set(getProjectPathCandidates(cwd));
531
- const projectDirs = getClaudeProjectDirs(cwd).filter(dir => fs.existsSync(dir));
532
- const cutoffMs = lastExtractedAt ? new Date(lastExtractedAt).getTime() : 0;
533
- try {
534
- if (projectDirs.length > 0) {
535
- const primary = collectClaudeProjectSessionContext(listClaudeSessionFiles(projectDirs, 5), cwdCandidates, cutoffMs);
536
- if (primary.prompts.length > 0 || primary.writes.length > 0)
537
- return primary;
538
- }
539
- const fallbackDirs = getAllClaudeProjectDirs().filter(dir => !projectDirs.includes(dir));
540
- if (fallbackDirs.length === 0)
541
- return { prompts: [], writes: [] };
542
- return collectClaudeProjectSessionContext(listClaudeSessionFiles(fallbackDirs, 20), cwdCandidates, cutoffMs);
543
- }
544
- catch (e) {
545
- log.debug('Claude project session context 로드 실패 — fallback 사용', e);
546
- return { prompts: [], writes: [] };
547
- }
548
- }
549
- function findCommonPrefix(strings) {
550
- if (strings.length === 0)
551
- return '';
552
- let prefix = strings[0];
553
- for (const s of strings.slice(1)) {
554
- while (!s.startsWith(prefix) && prefix.length > 0) {
555
- prefix = prefix.slice(0, -1);
556
- }
557
- }
558
- return prefix.replace(/-$/, '');
559
- }
560
- /** Save an extracted solution as experiment */
561
- function saveExtractedSolution(sol, _sessionId) {
562
- const today = new Date().toISOString().split('T')[0];
563
- const slugName = sol.name.toLowerCase()
564
- .replace(/[^a-z0-9가-힣\s-]/g, '')
565
- .replace(/\s+/g, '-')
566
- .replace(/-+/g, '-')
567
- .replace(/^-|-$/g, '')
568
- .slice(0, 60) || `untitled-${Date.now()}`;
569
- const solution = {
570
- frontmatter: {
571
- name: slugName,
572
- version: 1,
573
- status: 'experiment',
574
- confidence: 0.3,
575
- type: sol.type,
576
- scope: 'me',
577
- tags: sol.tags.slice(0, 5),
578
- identifiers: sol.identifiers.filter(id => id.length >= 4),
579
- evidence: { ...DEFAULT_EVIDENCE },
580
- created: today,
581
- updated: today,
582
- supersedes: null,
583
- extractedBy: 'auto',
584
- },
585
- context: sol.context,
586
- content: sol.content,
587
- };
588
- const filePath = path.join(ME_SOLUTIONS, `${slugName}.md`);
589
- if (fs.existsSync(filePath))
590
- return null;
591
- fs.mkdirSync(ME_SOLUTIONS, { recursive: true });
592
- // PR2b: 새 파일 create는 atomicWriteText로. O_EXCL이 race를 차단한다.
593
- atomicWriteText(filePath, serializeSolutionV3(solution));
594
- return slugName;
595
- }
596
- /**
597
- * Increment reExtracted counter on existing solution that matches given tags.
598
- * PR2b 라운드 2 (M-2 fix): mutateSolutionFile로 통합. parse → 카운터 증가 →
599
- * serialize. 이전 regex in-place mutation은 frontmatter 외 body의 우연 매칭
600
- * 위험이 있었고 다른 mutator와 일관성이 깨졌다.
601
- */
602
- function updateReExtractedCounter(tags) {
603
- if (!fs.existsSync(ME_SOLUTIONS))
604
- return;
605
- const files = fs.readdirSync(ME_SOLUTIONS).filter(f => f.endsWith('.md'));
606
- for (const file of files) {
607
- const filePath = path.join(ME_SOLUTIONS, file);
608
- // PR2c-4 (security L-1): symlink을 통한 임의 파일 read 차단.
609
- try {
610
- if (fs.lstatSync(filePath).isSymbolicLink())
611
- continue;
612
- }
613
- catch {
614
- continue;
615
- }
616
- // 사전 필터 (lock 없이 read) — frontmatter parse가 더 정확하지만,
617
- // 70% overlap 조건은 frontmatter 안의 tags만 보는 게 의도라
618
- // tagMatch regex가 frontmatter에 우선 매칭됨 (frontmatter가 항상 앞).
619
- let preview;
620
- try {
621
- preview = fs.readFileSync(filePath, 'utf-8');
622
- }
623
- catch {
624
- continue;
625
- }
626
- const tagMatch = preview.match(/tags:\s*\[([^\]]*)\]/);
627
- if (!tagMatch)
628
- continue;
629
- const existingTags = tagMatch[1].split(',').map(t => t.trim().replace(/"/g, ''));
630
- const overlap = tags.filter(t => existingTags.includes(t));
631
- if (overlap.length / Math.max(tags.length, existingTags.length, 1) < 0.7)
632
- continue;
633
- // lock + fresh re-read + parse-modify-serialize
634
- mutateSolutionFile(filePath, sol => {
635
- sol.frontmatter.evidence.reExtracted = (sol.frontmatter.evidence.reExtracted ?? 0) + 1;
636
- return true;
637
- });
638
- return;
639
- }
640
- }
641
- /**
642
- * Optional LLM enrichment for thin solution content.
643
- * Uses execFileSync (synchronous) to keep callers synchronous.
644
- * Completely fail-open: any error returns null and the regex-extracted content is kept.
645
- * Budget: max 2 calls per extraction run, 15s timeout each.
646
- */
27
+ // ── Re-exports (backward compatibility) ──
28
+ export { loadClaudeProjectSessionContext } from './extraction-session.js';
29
+ // ── LLM enrichment (kept inline — ~30 lines, not worth a separate file) ──
647
30
  function enrichSolutionContent(solution, diffSnippet) {
648
31
  try {
649
32
  const prompt = [
@@ -657,100 +40,53 @@ function enrichSolutionContent(solution, diffSnippet) {
657
40
  '코드 변경 (일부):',
658
41
  diffSnippet.slice(0, 2000),
659
42
  ].join('\n');
660
- // feat/codex-support P2-2 — host-aware exec via profile.default_host.
661
- // Codex 메인 사용자도 자동 추출 enrichment 가능 (해당 host CLI 호출).
662
- // fail-open 정책 유지 — LLM enrichment 실패는 추출 자체를 막지 않음.
663
43
  const { message } = execHost({ prompt, model: 'haiku', timeout: 15000 });
664
44
  if (message.length > 30 && message.length < 1000)
665
45
  return message;
666
46
  return null;
667
47
  }
668
48
  catch {
669
- // fail-open: LLM enrichment failure should never block extraction
670
49
  return null;
671
50
  }
672
51
  }
673
- /** Main extraction function — called from SessionStart or CLI */
52
+ // ── Orchestration ──
674
53
  function analyzeExtraction(cwd, options) {
675
54
  const state = loadLastExtraction();
676
55
  const today = new Date().toISOString().split('T')[0];
677
- // Reset daily counter if new day
678
56
  if (state.todayDate !== today) {
679
57
  state.extractionsToday = 0;
680
58
  state.todayDate = today;
681
59
  }
682
- // Daily limit check
683
60
  if (options?.enforceDailyLimit !== false && state.extractionsToday >= MAX_EXTRACTIONS_PER_DAY) {
684
61
  return {
685
- state,
686
- today,
687
- headSha: '',
688
- extracted: [],
62
+ state, today, headSha: '', extracted: [],
689
63
  reason: `일일 추출 한도 도달 (${MAX_EXTRACTIONS_PER_DAY}/일)`,
690
64
  persistStateWithoutSaving: false,
691
65
  };
692
66
  }
693
- // Check for new commits
694
67
  const gitLog = getNewCommits(cwd, state.lastCommitSha);
695
68
  if (!gitLog.trim()) {
696
- return {
697
- state,
698
- today,
699
- headSha: '',
700
- extracted: [],
701
- reason: '새 커밋 없음',
702
- persistStateWithoutSaving: false,
703
- };
69
+ return { state, today, headSha: '', extracted: [], reason: '새 커밋 없음', persistStateWithoutSaving: false };
704
70
  }
705
- // Get current HEAD sha
706
71
  let headSha = '';
707
72
  try {
708
73
  headSha = execFileSync('git', ['rev-parse', 'HEAD'], { cwd, encoding: 'utf-8', timeout: 3000, stdio: ['pipe', 'pipe', 'pipe'] }).trim();
709
74
  }
710
75
  catch {
711
- return {
712
- state,
713
- today,
714
- headSha: '',
715
- extracted: [],
716
- reason: 'git HEAD 조회 실패',
717
- persistStateWithoutSaving: false,
718
- };
76
+ return { state, today, headSha: '', extracted: [], reason: 'git HEAD 조회 실패', persistStateWithoutSaving: false };
719
77
  }
720
- // Gate 0: Worth extracting?
721
78
  const stats = getDiffStats(cwd, state.lastCommitSha);
722
79
  if (!gate0(stats)) {
723
80
  return {
724
- state,
725
- today,
726
- headSha,
727
- extracted: [],
81
+ state, today, headSha, extracted: [],
728
82
  reason: `Gate 0: 추출 가치 부족 (${stats.files} files, ${stats.lines} lines)`,
729
- stats,
730
- persistStateWithoutSaving: true,
83
+ stats, persistStateWithoutSaving: true,
731
84
  };
732
85
  }
733
- // Get diff for extraction prompt
734
86
  const gitDiff = getGitDiff(cwd, state.lastCommitSha);
735
- // Get commit messages for "why" context (addresses feedback: auto-extraction loses reasoning)
736
87
  const commitMessages = getCommitMessages(cwd, state.lastCommitSha);
737
- // Combine git diff analysis + session context analysis.
738
- // C3 fix: track provenance so commit context is only attached to
739
- // solutions that were actually derived from the diff. Pre-C3 the commit
740
- // message was blindly copy-pasted onto every extracted solution —
741
- // including session-context-derived patterns (word frequency
742
- // histograms, recurring task stats) that had nothing to do with the
743
- // commit. Observed failure mode: a "recurring-task-pattern" solution
744
- // about `test/테스트/추가` word counts was annotated with a completely
745
- // unrelated `Phase 1.5 + 2.5 — surprise detection + contextual bandit`
746
- // commit message, producing a misleading audit trail + noise in the
747
- // MCP context returned to Claude.
748
88
  const diffPatterns = extractFromDiff(gitLog, gitDiff);
749
89
  const contextPatterns = extractFromSessionContext(gitDiff, cwd, state.lastExtractedAt);
750
- // Attach commit context ONLY to diff-derived patterns (they're
751
- // genuinely about the commit). Session-context patterns keep their
752
- // own context unchanged — if they don't have a context, they don't
753
- // get a fake one.
754
90
  if (commitMessages) {
755
91
  for (const sol of diffPatterns) {
756
92
  sol.context = sol.context
@@ -758,31 +94,10 @@ function analyzeExtraction(cwd, options) {
758
94
  : `Commit context:\n${commitMessages.slice(0, 300)}`;
759
95
  }
760
96
  }
761
- const extracted = [...diffPatterns, ...contextPatterns].slice(0, 3); // max 3 total
762
- return {
763
- state,
764
- today,
765
- headSha,
766
- extracted,
767
- stats,
768
- persistStateWithoutSaving: false,
769
- gitDiff,
770
- };
771
- }
772
- function evaluateExtractedSolution(sol) {
773
- if (!gate1(sol))
774
- return { action: 'skip', message: `${sol.name ?? 'unnamed'}: Gate 1 실패 (구조 검증)` };
775
- if (!gate2(sol))
776
- return { action: 'skip', message: `${sol.name}: Gate 2 실패 (독성 필터)` };
777
- if (!gateTrivial(sol))
778
- return { action: 'skip', message: `${sol.name}: Gate 2.5 실패 (자명한 패턴)` };
779
- const dupResult = gate3(sol);
780
- if (dupResult === 'duplicate')
781
- return { action: 'duplicate', message: `${sol.name}: Gate 3 중복` };
782
- if (dupResult === 're-extract')
783
- return { action: 're-extract', message: `${sol.name}: 재추출 (기존 솔루션 강화)` };
784
- return { action: 'accept' };
97
+ const extracted = [...diffPatterns, ...contextPatterns].slice(0, 3);
98
+ return { state, today, headSha, extracted, stats, persistStateWithoutSaving: false, gitDiff };
785
99
  }
100
+ // ── Public API ──
786
101
  export async function previewExtraction(cwd) {
787
102
  const analysis = analyzeExtraction(cwd, { enforceDailyLimit: false });
788
103
  if (analysis.reason) {
@@ -804,7 +119,6 @@ export async function previewExtraction(cwd) {
804
119
  }
805
120
  return { preview, skipped };
806
121
  }
807
- /** Main extraction function — called from SessionStart or CLI */
808
122
  export async function runExtraction(cwd, sessionId) {
809
123
  const result = { extracted: [], skipped: [] };
810
124
  const analysis = analyzeExtraction(cwd);
@@ -819,7 +133,6 @@ export async function runExtraction(cwd, sessionId) {
819
133
  return { ...result, reason: analysis.reason };
820
134
  }
821
135
  if (analysis.extracted.length > 0) {
822
- // Enrich thin solutions with LLM context — max 2 per run, fail-open
823
136
  let enrichCount = 0;
824
137
  for (const sol of analysis.extracted) {
825
138
  if (enrichCount >= 2)
@@ -836,7 +149,6 @@ export async function runExtraction(cwd, sessionId) {
836
149
  result.extracted = saved;
837
150
  result.skipped = skipped;
838
151
  }
839
- // Update extraction state
840
152
  analysis.state.lastCommitSha = analysis.headSha;
841
153
  analysis.state.lastExtractedAt = new Date().toISOString();
842
154
  analysis.state.extractionsToday++;
@@ -846,42 +158,6 @@ export async function runExtraction(cwd, sessionId) {
846
158
  }
847
159
  return result;
848
160
  }
849
- /**
850
- * Observability P2: 새 솔루션 본문/supersedes 에서 기존 솔루션 참조 감지 → acted_on emit.
851
- * fail-open.
852
- */
853
- function emitCompoundExtractActedOn(sessionId, newSolutionName, newContent, newSupersedes) {
854
- try {
855
- if (!fs.existsSync(ME_SOLUTIONS))
856
- return;
857
- const bodyLower = newContent.toLowerCase();
858
- const supersedes = newSupersedes ?? '';
859
- const files = fs.readdirSync(ME_SOLUTIONS).filter(f => f.endsWith('.md'));
860
- for (const file of files) {
861
- const existingName = path.basename(file, '.md');
862
- if (existingName === newSolutionName)
863
- continue;
864
- const referenced = (supersedes && existingName === supersedes)
865
- || bodyLower.includes(existingName.toLowerCase());
866
- if (!referenced)
867
- continue;
868
- emitSolutionEvent({
869
- sessionId,
870
- solutionId: existingName,
871
- eventType: 'acted_on',
872
- signalSource: 'compound-extract',
873
- signalScore: 0.20,
874
- meta: {
875
- new_solution: newSolutionName,
876
- via: (supersedes && existingName === supersedes) ? 'supersedes' : 'body-mention',
877
- },
878
- });
879
- }
880
- }
881
- catch (e) {
882
- log.debug('emitCompoundExtractActedOn 실패', e);
883
- }
884
- }
885
161
  /** Process LLM extraction results (called after LLM returns) */
886
162
  export function processExtractionResults(rawJson, sessionId) {
887
163
  const saved = [];
@@ -895,7 +171,6 @@ export function processExtractionResults(rawJson, sessionId) {
895
171
  catch {
896
172
  return { saved, skipped };
897
173
  }
898
- // Max 3 per extraction
899
174
  for (const sol of solutions.slice(0, 3)) {
900
175
  const evaluation = evaluateExtractedSolution(sol);
901
176
  if (evaluation.action === 'skip' || evaluation.action === 'duplicate') {
@@ -903,7 +178,6 @@ export function processExtractionResults(rawJson, sessionId) {
903
178
  continue;
904
179
  }
905
180
  if (evaluation.action === 're-extract') {
906
- // Increment reExtracted counter on existing solution
907
181
  try {
908
182
  updateReExtractedCounter(sol.tags);
909
183
  }
@@ -913,13 +187,10 @@ export function processExtractionResults(rawJson, sessionId) {
913
187
  skipped.push(evaluation.message ?? `${sol.name}: 재추출`);
914
188
  continue;
915
189
  }
916
- // Clean identifiers before saving (short identifiers are noise)
917
190
  sol.identifiers = sol.identifiers.filter(id => id.length >= 4);
918
- // Save as experiment
919
191
  const savedName = saveExtractedSolution(sol, sessionId);
920
192
  if (savedName) {
921
193
  saved.push(savedName);
922
- // Observability P2: compound-extract acted_on signal
923
194
  emitCompoundExtractActedOn(sessionId, savedName, sol.content, null);
924
195
  }
925
196
  else {