@wooojin/forgen 0.4.10 → 0.4.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/CHANGELOG.md +46 -0
  3. package/README.md +33 -1
  4. package/assets/claude/agents/forgen-verify.md +65 -0
  5. package/assets/claude/workflows/compound-extract.js +136 -0
  6. package/assets/claude/workflows/evidence-gate-audit.js +107 -0
  7. package/assets/shared/hook-registry.json +1 -0
  8. package/dist/checks/_shared/meta-guard-dispatch.d.ts +38 -0
  9. package/dist/checks/_shared/meta-guard-dispatch.js +80 -0
  10. package/dist/checks/_shared/text-sanitizer.js +15 -0
  11. package/dist/cli.js +57 -2
  12. package/dist/core/changelog-cli.d.ts +7 -0
  13. package/dist/core/changelog-cli.js +100 -0
  14. package/dist/core/doctor.d.ts +3 -0
  15. package/dist/core/doctor.js +38 -0
  16. package/dist/core/effort-advisory.d.ts +23 -0
  17. package/dist/core/effort-advisory.js +29 -0
  18. package/dist/core/explain-cli.d.ts +6 -0
  19. package/dist/core/explain-cli.js +99 -0
  20. package/dist/core/health-cli.d.ts +23 -0
  21. package/dist/core/health-cli.js +86 -0
  22. package/dist/core/probe-workflow-cli.d.ts +72 -0
  23. package/dist/core/probe-workflow-cli.js +282 -0
  24. package/dist/core/stats-cli.d.ts +22 -9
  25. package/dist/core/stats-cli.js +149 -0
  26. package/dist/core/watch-cli.d.ts +7 -0
  27. package/dist/core/watch-cli.js +185 -0
  28. package/dist/core/workflows-cli.d.ts +26 -0
  29. package/dist/core/workflows-cli.js +120 -0
  30. package/dist/engine/compound-export.d.ts +12 -0
  31. package/dist/engine/compound-export.js +136 -14
  32. package/dist/engine/compound-extractor.d.ts +12 -43
  33. package/dist/engine/compound-extractor.js +27 -756
  34. package/dist/engine/extraction-diff.d.ts +11 -0
  35. package/dist/engine/extraction-diff.js +105 -0
  36. package/dist/engine/extraction-gates.d.ts +37 -0
  37. package/dist/engine/extraction-gates.js +100 -0
  38. package/dist/engine/extraction-git.d.ts +20 -0
  39. package/dist/engine/extraction-git.js +75 -0
  40. package/dist/engine/extraction-persistence.d.ts +27 -0
  41. package/dist/engine/extraction-persistence.js +140 -0
  42. package/dist/engine/extraction-session.d.ts +26 -0
  43. package/dist/engine/extraction-session.js +230 -0
  44. package/dist/engine/lifecycle/types.d.ts +1 -1
  45. package/dist/engine/meta-learning/matcher-weight-loader.d.ts +16 -0
  46. package/dist/engine/meta-learning/matcher-weight-loader.js +45 -0
  47. package/dist/engine/precision-guards.d.ts +14 -0
  48. package/dist/engine/precision-guards.js +39 -0
  49. package/dist/engine/ranking-pipeline.d.ts +45 -0
  50. package/dist/engine/ranking-pipeline.js +66 -0
  51. package/dist/engine/relevance-scorer.d.ts +43 -0
  52. package/dist/engine/relevance-scorer.js +81 -0
  53. package/dist/engine/scoring-algorithms.d.ts +31 -0
  54. package/dist/engine/scoring-algorithms.js +109 -0
  55. package/dist/engine/solution-matcher-eval.d.ts +97 -0
  56. package/dist/engine/solution-matcher-eval.js +122 -0
  57. package/dist/engine/solution-matcher.d.ts +21 -380
  58. package/dist/engine/solution-matcher.js +27 -828
  59. package/dist/fgx.js +1 -1
  60. package/dist/hooks/notepad-injector.js +7 -0
  61. package/dist/hooks/post-tool-use.js +8 -1
  62. package/dist/hooks/shared/preflight-check.d.ts +15 -0
  63. package/dist/hooks/shared/preflight-check.js +51 -0
  64. package/dist/hooks/stop-guard.js +19 -60
  65. package/dist/hooks/subagent-stop-guard.d.ts +23 -0
  66. package/dist/hooks/subagent-stop-guard.js +158 -0
  67. package/dist/hooks/subagent-tracker.d.ts +36 -3
  68. package/dist/hooks/subagent-tracker.js +86 -39
  69. package/hooks/hooks.json +6 -1
  70. package/package.json +7 -7
  71. package/plugin.json +1 -1
  72. package/scripts/postinstall.js +10 -7
@@ -0,0 +1,230 @@
1
+ /**
2
+ * Session context extraction for compound knowledge.
3
+ *
4
+ * Extracted from compound-extractor.ts — reads Claude session JSONL files
5
+ * to extract prompts and write history correlated to a specific project.
6
+ */
7
+ import * as fs from 'node:fs';
8
+ import * as path from 'node:path';
9
+ import { CLAUDE_DIR, STATE_DIR } from '../core/paths.js';
10
+ import { createLogger } from '../core/logger.js';
11
+ const log = createLogger('extraction-session');
12
+ function normalizeProjectPath(cwd) {
13
+ const resolved = path.resolve(cwd);
14
+ try {
15
+ return typeof fs.realpathSync.native === 'function'
16
+ ? fs.realpathSync.native(resolved)
17
+ : fs.realpathSync(resolved);
18
+ }
19
+ catch {
20
+ return resolved;
21
+ }
22
+ }
23
+ function getProjectPathCandidates(cwd) {
24
+ const resolved = path.resolve(cwd);
25
+ const candidates = new Set([resolved, normalizeProjectPath(cwd)]);
26
+ try {
27
+ if (fs.lstatSync(resolved).isSymbolicLink()) {
28
+ candidates.add(path.resolve(path.dirname(resolved), fs.readlinkSync(resolved)));
29
+ }
30
+ }
31
+ catch {
32
+ // Ignore lstat/readlink failures
33
+ }
34
+ for (const candidate of [...candidates]) {
35
+ candidates.add(normalizeProjectPath(candidate));
36
+ }
37
+ return [...candidates];
38
+ }
39
+ function getClaudeProjectDirs(cwd) {
40
+ return getProjectPathCandidates(cwd)
41
+ .map(candidate => path.join(CLAUDE_DIR, 'projects', candidate.replace(/[:\\/]/g, '-')));
42
+ }
43
+ function listClaudeSessionFiles(projectDirs, maxFiles) {
44
+ return projectDirs
45
+ .flatMap(projectDir => {
46
+ let entries;
47
+ try {
48
+ entries = fs.readdirSync(projectDir);
49
+ }
50
+ catch {
51
+ return [];
52
+ }
53
+ const out = [];
54
+ for (const file of entries) {
55
+ if (!file.endsWith('.jsonl'))
56
+ continue;
57
+ const filePath = path.join(projectDir, file);
58
+ try {
59
+ if (fs.lstatSync(filePath).isSymbolicLink())
60
+ continue;
61
+ out.push({ filePath, mtimeMs: fs.statSync(filePath).mtimeMs });
62
+ }
63
+ catch {
64
+ // unreadable / vanished between readdir and stat — skip
65
+ }
66
+ }
67
+ return out;
68
+ })
69
+ .sort((a, b) => b.mtimeMs - a.mtimeMs)
70
+ .slice(0, maxFiles);
71
+ }
72
+ function getAllClaudeProjectDirs() {
73
+ const projectsRoot = path.join(CLAUDE_DIR, 'projects');
74
+ if (!fs.existsSync(projectsRoot))
75
+ return [];
76
+ return fs.readdirSync(projectsRoot)
77
+ .map(name => path.join(projectsRoot, name))
78
+ .filter(dir => {
79
+ try {
80
+ return fs.statSync(dir).isDirectory();
81
+ }
82
+ catch {
83
+ return false;
84
+ }
85
+ });
86
+ }
87
+ function collectClaudeProjectSessionContext(files, cwdCandidates, cutoffMs) {
88
+ const prompts = [];
89
+ const writes = [];
90
+ for (const file of files) {
91
+ try {
92
+ if (fs.lstatSync(file.filePath).isSymbolicLink())
93
+ continue;
94
+ }
95
+ catch {
96
+ continue;
97
+ }
98
+ let lines;
99
+ try {
100
+ lines = fs.readFileSync(file.filePath, 'utf-8').split('\n').filter(Boolean);
101
+ }
102
+ catch {
103
+ continue;
104
+ }
105
+ for (const line of lines) {
106
+ let entry;
107
+ try {
108
+ entry = JSON.parse(line);
109
+ }
110
+ catch {
111
+ continue;
112
+ }
113
+ const entryCandidates = typeof entry.cwd === 'string' ? getProjectPathCandidates(entry.cwd) : [];
114
+ if (!entryCandidates.some(candidate => cwdCandidates.has(candidate)))
115
+ continue;
116
+ const timestamp = typeof entry.timestamp === 'string' ? new Date(entry.timestamp).getTime() : Number.NaN;
117
+ if (cutoffMs && Number.isFinite(timestamp) && timestamp <= cutoffMs)
118
+ continue;
119
+ if (entry.type === 'user') {
120
+ const message = entry.message;
121
+ if (message?.role === 'user' && typeof message.content === 'string') {
122
+ prompts.push(message.content);
123
+ }
124
+ continue;
125
+ }
126
+ if (entry.type !== 'assistant')
127
+ continue;
128
+ const message = entry.message;
129
+ if (message?.role !== 'assistant' || !Array.isArray(message.content))
130
+ continue;
131
+ for (const item of message.content) {
132
+ if (typeof item !== 'object' || item === null)
133
+ continue;
134
+ const toolUse = item;
135
+ if (toolUse.type !== 'tool_use')
136
+ continue;
137
+ if (toolUse.name !== 'Write' && toolUse.name !== 'Edit')
138
+ continue;
139
+ const filePath = String(toolUse.input?.file_path ?? toolUse.input?.filePath ?? '');
140
+ const content = String(toolUse.input?.content ?? toolUse.input?.new_string ?? '');
141
+ if (!filePath || !content)
142
+ continue;
143
+ writes.push({
144
+ filePath: filePath.slice(-100),
145
+ contentSnippet: content.slice(0, 200),
146
+ fileExtension: path.extname(filePath).toLowerCase(),
147
+ });
148
+ }
149
+ }
150
+ }
151
+ return {
152
+ prompts: prompts.slice(-50),
153
+ writes: writes.slice(-30),
154
+ };
155
+ }
156
+ export function loadPromptHistoryFallback() {
157
+ const promptHistoryPath = path.join(STATE_DIR, 'prompt-history.jsonl');
158
+ try {
159
+ if (!fs.existsSync(promptHistoryPath))
160
+ return [];
161
+ const lines = fs.readFileSync(promptHistoryPath, 'utf-8').split('\n').filter(Boolean);
162
+ return lines.slice(-50).map(l => {
163
+ try {
164
+ return JSON.parse(l).prompt;
165
+ }
166
+ catch {
167
+ return '';
168
+ }
169
+ }).filter(Boolean);
170
+ }
171
+ catch (e) {
172
+ log.debug('prompt-history.jsonl 읽기 실패 — session context fallback 건너뜀', e);
173
+ return [];
174
+ }
175
+ }
176
+ /**
177
+ * Load Claude session prompts + writes correlated to `cwd`.
178
+ *
179
+ * Exported primarily for test assertions (the `claude-session-context`
180
+ * tests verify correlation logic directly).
181
+ */
182
+ export function loadClaudeProjectSessionContext(cwd, lastExtractedAt) {
183
+ const cwdCandidates = new Set(getProjectPathCandidates(cwd));
184
+ const projectDirs = getClaudeProjectDirs(cwd).filter(dir => fs.existsSync(dir));
185
+ const cutoffMs = lastExtractedAt ? new Date(lastExtractedAt).getTime() : 0;
186
+ try {
187
+ if (projectDirs.length > 0) {
188
+ const primary = collectClaudeProjectSessionContext(listClaudeSessionFiles(projectDirs, 5), cwdCandidates, cutoffMs);
189
+ if (primary.prompts.length > 0 || primary.writes.length > 0)
190
+ return primary;
191
+ }
192
+ const fallbackDirs = getAllClaudeProjectDirs().filter(dir => !projectDirs.includes(dir));
193
+ if (fallbackDirs.length === 0)
194
+ return { prompts: [], writes: [] };
195
+ return collectClaudeProjectSessionContext(listClaudeSessionFiles(fallbackDirs, 20), cwdCandidates, cutoffMs);
196
+ }
197
+ catch (e) {
198
+ log.debug('Claude project session context 로드 실패 — fallback 사용', e);
199
+ return { prompts: [], writes: [] };
200
+ }
201
+ }
202
+ /** Extract patterns from accumulated session context (prompts + writes + diff) */
203
+ export function extractFromSessionContext(gitDiff, cwd, lastExtractedAt) {
204
+ const solutions = [];
205
+ const claudeContext = loadClaudeProjectSessionContext(cwd, lastExtractedAt);
206
+ let prompts = claudeContext.prompts;
207
+ if (prompts.length === 0) {
208
+ prompts = loadPromptHistoryFallback();
209
+ }
210
+ const techDecisions = [];
211
+ const techTerms = ['react', 'vue', 'next', 'express', 'fastify', 'prisma', 'drizzle', 'zustand', 'redux', 'tailwind', 'styled', 'vitest', 'jest', 'playwright', 'cypress'];
212
+ for (const term of techTerms) {
213
+ const inPrompts = prompts.some(p => p.toLowerCase().includes(term));
214
+ const inDiff = gitDiff.toLowerCase().includes(term);
215
+ if (inPrompts && inDiff) {
216
+ techDecisions.push(term);
217
+ }
218
+ }
219
+ if (techDecisions.length >= 2) {
220
+ solutions.push({
221
+ name: 'tech-stack-decision',
222
+ type: 'decision',
223
+ tags: ['stack', 'technology', ...techDecisions.slice(0, 5)],
224
+ identifiers: techDecisions.filter(t => t.length >= 4).slice(0, 5),
225
+ context: 'Technology choices confirmed by both discussion and implementation',
226
+ content: `Active technology stack: ${techDecisions.join(', ')}. Both discussed in prompts and present in code changes.`,
227
+ });
228
+ }
229
+ return solutions;
230
+ }
@@ -39,7 +39,7 @@ export interface ViolationEntry {
39
39
  at: string;
40
40
  rule_id: string;
41
41
  session_id: string;
42
- source: 'stop-guard' | 'pre-tool-guard' | 'post-tool-guard' | 'evidence-store' | 'manual';
42
+ source: 'stop-guard' | 'subagent-stop-guard' | 'pre-tool-guard' | 'post-tool-guard' | 'evidence-store' | 'manual';
43
43
  kind: 'block' | 'deny' | 'correction';
44
44
  message_preview?: string;
45
45
  }
@@ -0,0 +1,16 @@
1
+ /**
2
+ * Dynamic ensemble weight loader for the solution matcher.
3
+ *
4
+ * Extracted from solution-matcher.ts — loads tuned weights from meta-learning
5
+ * state with a 1-minute in-process TTL cache.
6
+ */
7
+ /**
8
+ * Load tuned matcher weights from meta-learning state.
9
+ * Returns undefined (use defaults) if no tuned weights exist.
10
+ * Cached for 1 minute to avoid re-reading per matchSolutions call.
11
+ */
12
+ export declare function loadTunedMatcherWeights(): {
13
+ tfidf: number;
14
+ bm25: number;
15
+ bigram: number;
16
+ } | undefined;
@@ -0,0 +1,45 @@
1
+ /**
2
+ * Dynamic ensemble weight loader for the solution matcher.
3
+ *
4
+ * Extracted from solution-matcher.ts — loads tuned weights from meta-learning
5
+ * state with a 1-minute in-process TTL cache.
6
+ */
7
+ import * as fs from 'node:fs';
8
+ import * as path from 'node:path';
9
+ import { META_LEARNING_DIR } from '../../core/paths.js';
10
+ let _cachedWeights;
11
+ let _weightsCacheTime = 0;
12
+ const WEIGHTS_CACHE_TTL = 60_000; // 1 minute cache
13
+ /**
14
+ * Load tuned matcher weights from meta-learning state.
15
+ * Returns undefined (use defaults) if no tuned weights exist.
16
+ * Cached for 1 minute to avoid re-reading per matchSolutions call.
17
+ */
18
+ export function loadTunedMatcherWeights() {
19
+ const now = Date.now();
20
+ if (_cachedWeights !== undefined && now - _weightsCacheTime < WEIGHTS_CACHE_TTL) {
21
+ return _cachedWeights ?? undefined;
22
+ }
23
+ try {
24
+ const weightsPath = path.join(META_LEARNING_DIR, 'matcher-weights.json');
25
+ if (!fs.existsSync(weightsPath)) {
26
+ _cachedWeights = null;
27
+ _weightsCacheTime = now;
28
+ return undefined;
29
+ }
30
+ const data = JSON.parse(fs.readFileSync(weightsPath, 'utf-8'));
31
+ if (typeof data.tfidf === 'number' &&
32
+ typeof data.bm25 === 'number' &&
33
+ typeof data.bigram === 'number') {
34
+ _cachedWeights = { tfidf: data.tfidf, bm25: data.bm25, bigram: data.bigram };
35
+ _weightsCacheTime = now;
36
+ return _cachedWeights;
37
+ }
38
+ }
39
+ catch {
40
+ /* fail-open: use defaults */
41
+ }
42
+ _cachedWeights = null;
43
+ _weightsCacheTime = now;
44
+ return undefined;
45
+ }
@@ -0,0 +1,14 @@
1
+ /**
2
+ * Query-side specificity guards for the solution matcher (R4-T3).
3
+ *
4
+ * Extracted from solution-matcher.ts — orchestration-layer precision rules
5
+ * applied AFTER calculateRelevance returns. These fix false positives from
6
+ * ambiguous single-tag matches without regressing legitimate results.
7
+ *
8
+ * Rule A: single-token query AND single-tag match → reject.
9
+ * Rule B: all matched tags came via synonym expansion (none literal in prompt)
10
+ * AND match is single-tag → reject.
11
+ *
12
+ * Returns true = reject the candidate, false = keep it.
13
+ */
14
+ export declare function shouldRejectByR4T3Rules(promptTags: readonly string[], matchedTags: readonly string[]): boolean;
@@ -0,0 +1,39 @@
1
+ /**
2
+ * Query-side specificity guards for the solution matcher (R4-T3).
3
+ *
4
+ * Extracted from solution-matcher.ts — orchestration-layer precision rules
5
+ * applied AFTER calculateRelevance returns. These fix false positives from
6
+ * ambiguous single-tag matches without regressing legitimate results.
7
+ *
8
+ * Rule A: single-token query AND single-tag match → reject.
9
+ * Rule B: all matched tags came via synonym expansion (none literal in prompt)
10
+ * AND match is single-tag → reject.
11
+ *
12
+ * Returns true = reject the candidate, false = keep it.
13
+ */
14
+ export function shouldRejectByR4T3Rules(promptTags, matchedTags) {
15
+ // Rule A
16
+ if (promptTags.length === 1 && matchedTags.length === 1) {
17
+ return true;
18
+ }
19
+ // Rule B
20
+ if (matchedTags.length === 1) {
21
+ const tag = matchedTags[0];
22
+ const literalHit = promptTags.includes(tag) ||
23
+ promptTags.some((pt) => {
24
+ if (pt.length <= 3 || tag.length <= 3)
25
+ return false;
26
+ if (pt.includes(tag) || tag.includes(pt))
27
+ return true;
28
+ // Morphological stem: shared prefix of length ≥ 4
29
+ let i = 0;
30
+ const limit = Math.min(pt.length, tag.length);
31
+ while (i < limit && pt[i] === tag[i])
32
+ i++;
33
+ return i >= 4;
34
+ });
35
+ if (!literalHit)
36
+ return true;
37
+ }
38
+ return false;
39
+ }
@@ -0,0 +1,45 @@
1
+ /**
2
+ * Shared ranking core for solution matching.
3
+ *
4
+ * Extracted from solution-matcher.ts — the ranking pipeline used by both
5
+ * production (matchSolutions) and the bootstrap evaluator
6
+ * (evaluateSolutionMatcher). Single source of truth for ranking behaviour.
7
+ */
8
+ /**
9
+ * Narrow input shape for the shared ranking pipeline. `matchSolutions` and the
10
+ * bootstrap evaluator both reduce to this contract — `LoadedSolution` is
11
+ * structurally compatible (it has more fields), and `EvalSolution` mirrors it
12
+ * exactly. Keeping the input narrow prevents the evaluator from leaking onto
13
+ * prod types and vice versa.
14
+ */
15
+ export interface RankableSolution {
16
+ name: string;
17
+ tags: string[];
18
+ identifiers?: string[];
19
+ confidence: number;
20
+ }
21
+ /**
22
+ * Intermediate ranked candidate. Generic over the source solution type so the
23
+ * caller can get back the exact object they passed in.
24
+ */
25
+ export interface RankedCandidate<T extends RankableSolution = RankableSolution> {
26
+ solution: T;
27
+ relevance: number;
28
+ matchedTags: string[];
29
+ matchedIdentifiers: string[];
30
+ }
31
+ /**
32
+ * Shared ranking core: tag-based relevance + identifier boost + top-5 sort.
33
+ *
34
+ * Contract:
35
+ * - identifier boost requires `id.length >= 4` and substring presence in
36
+ * the prompt (case-insensitive).
37
+ * - candidates with zero matched tags AND zero matched identifiers are dropped.
38
+ * - top-5 by `relevance` descending.
39
+ * - duplicate names are NOT deduplicated.
40
+ */
41
+ export declare function rankCandidates<T extends RankableSolution>(promptTags: string[], promptLower: string, solutions: readonly T[], ensembleWeights?: {
42
+ tfidf: number;
43
+ bm25: number;
44
+ bigram: number;
45
+ }): RankedCandidate<T>[];
@@ -0,0 +1,66 @@
1
+ /**
2
+ * Shared ranking core for solution matching.
3
+ *
4
+ * Extracted from solution-matcher.ts — the ranking pipeline used by both
5
+ * production (matchSolutions) and the bootstrap evaluator
6
+ * (evaluateSolutionMatcher). Single source of truth for ranking behaviour.
7
+ */
8
+ import { maskBlockedTokens } from './phrase-blocklist.js';
9
+ import { calculateRelevance } from './relevance-scorer.js';
10
+ import { expandCompoundTags, expandQueryBigrams } from './solution-format.js';
11
+ import { shouldRejectByR4T3Rules } from './precision-guards.js';
12
+ import { defaultNormalizer } from './term-normalizer.js';
13
+ /**
14
+ * Shared ranking core: tag-based relevance + identifier boost + top-5 sort.
15
+ *
16
+ * Contract:
17
+ * - identifier boost requires `id.length >= 4` and substring presence in
18
+ * the prompt (case-insensitive).
19
+ * - candidates with zero matched tags AND zero matched identifiers are dropped.
20
+ * - top-5 by `relevance` descending.
21
+ * - duplicate names are NOT deduplicated.
22
+ */
23
+ export function rankCandidates(promptTags, promptLower, solutions, ensembleWeights) {
24
+ // R4-T2: mask blocked tokens before expansion/normalization
25
+ const maskedPromptTags = maskBlockedTokens(promptLower, promptTags);
26
+ if (maskedPromptTags.length === 0)
27
+ return [];
28
+ // R4-T1: expand prompt tags with adjacent-token bigrams
29
+ const promptTagsWithBigrams = expandQueryBigrams(maskedPromptTags);
30
+ const normalizedPromptTags = defaultNormalizer.normalizeTerms(promptTagsWithBigrams);
31
+ return solutions
32
+ .map((sol) => {
33
+ const solTagsExpanded = expandCompoundTags(sol.tags);
34
+ const result = calculateRelevance(maskedPromptTags, sol.tags, sol.confidence, {
35
+ normalizedPromptTags,
36
+ solutionTagsExpanded: solTagsExpanded,
37
+ ensembleWeights,
38
+ });
39
+ let identifierBoost = 0;
40
+ const matchedIdentifiers = [];
41
+ for (const id of sol.identifiers ?? []) {
42
+ if (id.length >= 4 && promptLower.includes(id.toLowerCase())) {
43
+ identifierBoost += 0.15;
44
+ matchedIdentifiers.push(id);
45
+ }
46
+ }
47
+ // R4-T3: orchestration-layer specificity guards
48
+ let tagRelevance = result.relevance;
49
+ let tagMatches = result.matchedTags;
50
+ if (matchedIdentifiers.length === 0 &&
51
+ tagMatches.length > 0 &&
52
+ shouldRejectByR4T3Rules(maskedPromptTags, tagMatches)) {
53
+ tagRelevance = 0;
54
+ tagMatches = [];
55
+ }
56
+ return {
57
+ solution: sol,
58
+ relevance: tagRelevance + identifierBoost,
59
+ matchedTags: tagMatches,
60
+ matchedIdentifiers,
61
+ };
62
+ })
63
+ .filter((c) => c.matchedTags.length + c.matchedIdentifiers.length >= 1)
64
+ .sort((a, b) => b.relevance - a.relevance)
65
+ .slice(0, 5);
66
+ }
@@ -0,0 +1,43 @@
1
+ /**
2
+ * Tag-based relevance scoring for solution matching.
3
+ *
4
+ * Extracted from solution-matcher.ts — the core scoring logic that computes
5
+ * relevance between a prompt's tags and a solution's tags using a TF-IDF +
6
+ * BM25 + bigram ensemble.
7
+ */
8
+ /**
9
+ * Optional hints for the v3 `calculateRelevance` path. Used by hot-path
10
+ * callers (matchSolutions, searchSolutions) to avoid re-normalizing the
11
+ * same query tags on every solution.
12
+ */
13
+ export interface CalculateRelevanceOptions {
14
+ /**
15
+ * Pre-normalized prompt tags (produced by `defaultNormalizer.normalizeTerms`).
16
+ * If provided, skips the per-call expansion. Callers loop-running against
17
+ * many solutions should compute this once outside the loop and pass it in.
18
+ */
19
+ normalizedPromptTags?: string[];
20
+ /**
21
+ * R4-T1: solution tags expanded with compound-split alternatives
22
+ * (`expandCompoundTags`). When supplied, the intersection/partial-match
23
+ * step uses this set INSTEAD of `solutionTags`, but the Jaccard union
24
+ * denominator still uses `solutionTags` (raw) so the score normalization
25
+ * stays semantically stable. Caller responsibility to pass the matching
26
+ * pair — `solutionTagsExpanded` MUST be a superset of `solutionTags`.
27
+ */
28
+ solutionTagsExpanded?: string[];
29
+ /** Average document (solution) tag count for BM25 normalization. Defaults to 6. */
30
+ avgDocLength?: number;
31
+ /** Meta-learning: dynamic ensemble weights (sum must equal 1.0). Defaults to {tfidf:0.5, bm25:0.3, bigram:0.2}. */
32
+ ensembleWeights?: {
33
+ tfidf: number;
34
+ bm25: number;
35
+ bigram: number;
36
+ };
37
+ }
38
+ export declare function calculateRelevance(promptTags: string[], solutionTags: string[], confidence: number, options?: CalculateRelevanceOptions): {
39
+ relevance: number;
40
+ matchedTags: string[];
41
+ };
42
+ /** @deprecated */
43
+ export declare function calculateRelevance(prompt: string, keywords: string[]): number;
@@ -0,0 +1,81 @@
1
+ /**
2
+ * Tag-based relevance scoring for solution matching.
3
+ *
4
+ * Extracted from solution-matcher.ts — the core scoring logic that computes
5
+ * relevance between a prompt's tags and a solution's tags using a TF-IDF +
6
+ * BM25 + bigram ensemble.
7
+ */
8
+ import { bigramSimilarity, bm25Score, tagWeight } from './scoring-algorithms.js';
9
+ import { extractTags } from './solution-format.js';
10
+ import { defaultNormalizer } from './term-normalizer.js';
11
+ export function calculateRelevance(promptOrTags, keywordsOrTags, confidence, options) {
12
+ if (typeof promptOrTags === 'string') {
13
+ // Legacy mode: substring matching for backwards compatibility.
14
+ // Not a hot path — only hit by the (old) solution-matcher.test.ts cases.
15
+ const promptTags = extractTags(promptOrTags);
16
+ const intersection = keywordsOrTags.filter((kw) => promptTags.some((pt) => pt === kw || (pt.length > 3 && kw.length > 3 && (pt.startsWith(kw) || kw.startsWith(pt)))));
17
+ return Math.min(1, intersection.length / Math.max(promptTags.length * 0.5, 1));
18
+ }
19
+ // v3 mode: tag matching with synonym expansion + TF-IDF weighting.
20
+ const expandedPromptTags = options?.normalizedPromptTags ?? defaultNormalizer.normalizeTerms(promptOrTags);
21
+ // R4-T1: when the caller supplies a compound-expanded solution tag set,
22
+ // intersection and partial matching run against the expanded set.
23
+ const matchTags = options?.solutionTagsExpanded ?? keywordsOrTags;
24
+ const intersection = matchTags.filter((t) => expandedPromptTags.includes(t));
25
+ // partial/substring matches for longer tags (>3 chars)
26
+ const partialMatches = matchTags.filter((t) => t.length > 3 &&
27
+ !intersection.includes(t) &&
28
+ expandedPromptTags.some((pt) => pt.length > 3 && (pt.includes(t) || t.includes(pt))));
29
+ // Apply TF-IDF weighting: common tags count less
30
+ const weightedMatched = intersection.reduce((sum, t) => sum + tagWeight(t), 0) +
31
+ partialMatches.reduce((sum, t) => sum + tagWeight(t) * 0.5, 0);
32
+ // Bigram similarity boost for borderline cases
33
+ if (weightedMatched < 0.5) {
34
+ let bestBigramScore = 0;
35
+ const bigramMatchedTags = [];
36
+ for (const st of matchTags) {
37
+ for (const pt of expandedPromptTags) {
38
+ const sim = bigramSimilarity(pt, st);
39
+ if (sim > bestBigramScore) {
40
+ bestBigramScore = sim;
41
+ }
42
+ if (sim > 0.4 && !bigramMatchedTags.includes(st)) {
43
+ bigramMatchedTags.push(st);
44
+ }
45
+ }
46
+ }
47
+ if (bestBigramScore > 0.4) {
48
+ const union = new Set([...promptOrTags, ...keywordsOrTags]).size;
49
+ const tfidfScore = weightedMatched / Math.max(union, 1);
50
+ const blendedScore = tfidfScore * 0.8 + bestBigramScore * 0.2;
51
+ return {
52
+ relevance: blendedScore * (confidence ?? 1),
53
+ matchedTags: [
54
+ ...intersection,
55
+ ...partialMatches,
56
+ ...bigramMatchedTags.filter((t) => !intersection.includes(t) && !partialMatches.includes(t)),
57
+ ],
58
+ };
59
+ }
60
+ return { relevance: 0, matchedTags: [] };
61
+ }
62
+ // Ensemble: TF-IDF (Jaccard) 0.5 + BM25 0.3 + bigram 0.2
63
+ const union = new Set([...promptOrTags, ...keywordsOrTags]).size;
64
+ const tfidfScore = weightedMatched / Math.max(union, 1);
65
+ const avgDocLen = options?.avgDocLength ?? 6;
66
+ const bm25 = bm25Score(promptOrTags, keywordsOrTags, avgDocLen);
67
+ let bigramBoost = 0;
68
+ for (const st of matchTags) {
69
+ for (const pt of expandedPromptTags) {
70
+ const sim = bigramSimilarity(pt, st);
71
+ if (sim > bigramBoost)
72
+ bigramBoost = sim;
73
+ }
74
+ }
75
+ const w = options?.ensembleWeights ?? { tfidf: 0.5, bm25: 0.3, bigram: 0.2 };
76
+ const ensembleScore = tfidfScore * w.tfidf + bm25 * w.bm25 + bigramBoost * w.bigram;
77
+ return {
78
+ relevance: ensembleScore * (confidence ?? 1),
79
+ matchedTags: [...intersection, ...partialMatches],
80
+ };
81
+ }
@@ -0,0 +1,31 @@
1
+ /**
2
+ * Stateless scoring primitives for the solution matcher.
3
+ *
4
+ * Extracted from solution-matcher.ts — these are pure functions with no
5
+ * filesystem or module-level state dependencies.
6
+ */
7
+ /** High-frequency tags that should be weighted lower */
8
+ export declare const COMMON_TAGS: Set<string>;
9
+ /** Apply IDF-like weight: common tags get reduced weight */
10
+ export declare function tagWeight(tag: string): number;
11
+ /**
12
+ * Compute the Dice coefficient between two strings using character bigrams.
13
+ *
14
+ * Dice = 2 * |intersection| / (|A| + |B|)
15
+ *
16
+ * Both strings are lowercased and whitespace-stripped before bigram generation.
17
+ * Returns 0 for empty strings or single-character strings (no bigrams possible).
18
+ * Returns 1.0 for identical non-trivial strings.
19
+ *
20
+ * This is used as a lightweight fuzzy matching signal for borderline cases
21
+ * where the TF-IDF tag intersection produces a low score but the query and
22
+ * solution tags are character-similar (e.g., "database" vs "데이터베이스"
23
+ * won't match, but "database" vs "databse" will get a high score).
24
+ */
25
+ export declare function bigramSimilarity(a: string, b: string): number;
26
+ /**
27
+ * Simplified BM25 score for a single query-document pair.
28
+ * Uses tag overlap with term frequency normalization.
29
+ * k1=1.2, b=0.75 (standard BM25 parameters).
30
+ */
31
+ export declare function bm25Score(queryTags: string[], docTags: string[], avgDocLength: number): number;