@wooojin/forgen 0.4.10 → 0.4.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/CHANGELOG.md +62 -0
  3. package/README.md +33 -1
  4. package/assets/claude/agents/forgen-verify.md +65 -0
  5. package/assets/claude/workflows/compound-extract.js +136 -0
  6. package/assets/claude/workflows/evidence-gate-audit.js +107 -0
  7. package/assets/shared/hook-registry.json +1 -0
  8. package/dist/checks/_shared/meta-guard-dispatch.d.ts +38 -0
  9. package/dist/checks/_shared/meta-guard-dispatch.js +80 -0
  10. package/dist/checks/_shared/text-sanitizer.js +15 -0
  11. package/dist/cli.js +57 -2
  12. package/dist/core/changelog-cli.d.ts +7 -0
  13. package/dist/core/changelog-cli.js +100 -0
  14. package/dist/core/doctor.d.ts +3 -0
  15. package/dist/core/doctor.js +38 -0
  16. package/dist/core/effort-advisory.d.ts +23 -0
  17. package/dist/core/effort-advisory.js +29 -0
  18. package/dist/core/explain-cli.d.ts +6 -0
  19. package/dist/core/explain-cli.js +99 -0
  20. package/dist/core/health-cli.d.ts +23 -0
  21. package/dist/core/health-cli.js +86 -0
  22. package/dist/core/probe-workflow-cli.d.ts +72 -0
  23. package/dist/core/probe-workflow-cli.js +282 -0
  24. package/dist/core/spawn.d.ts +13 -0
  25. package/dist/core/spawn.js +36 -8
  26. package/dist/core/stats-cli.d.ts +22 -9
  27. package/dist/core/stats-cli.js +149 -0
  28. package/dist/core/watch-cli.d.ts +7 -0
  29. package/dist/core/watch-cli.js +185 -0
  30. package/dist/core/workflows-cli.d.ts +26 -0
  31. package/dist/core/workflows-cli.js +120 -0
  32. package/dist/engine/compound-export.d.ts +12 -0
  33. package/dist/engine/compound-export.js +136 -14
  34. package/dist/engine/compound-extractor.d.ts +12 -43
  35. package/dist/engine/compound-extractor.js +27 -756
  36. package/dist/engine/extraction-diff.d.ts +11 -0
  37. package/dist/engine/extraction-diff.js +105 -0
  38. package/dist/engine/extraction-gates.d.ts +37 -0
  39. package/dist/engine/extraction-gates.js +100 -0
  40. package/dist/engine/extraction-git.d.ts +20 -0
  41. package/dist/engine/extraction-git.js +75 -0
  42. package/dist/engine/extraction-persistence.d.ts +27 -0
  43. package/dist/engine/extraction-persistence.js +140 -0
  44. package/dist/engine/extraction-session.d.ts +26 -0
  45. package/dist/engine/extraction-session.js +230 -0
  46. package/dist/engine/lifecycle/types.d.ts +1 -1
  47. package/dist/engine/meta-learning/matcher-weight-loader.d.ts +16 -0
  48. package/dist/engine/meta-learning/matcher-weight-loader.js +45 -0
  49. package/dist/engine/precision-guards.d.ts +14 -0
  50. package/dist/engine/precision-guards.js +39 -0
  51. package/dist/engine/ranking-pipeline.d.ts +45 -0
  52. package/dist/engine/ranking-pipeline.js +66 -0
  53. package/dist/engine/relevance-scorer.d.ts +43 -0
  54. package/dist/engine/relevance-scorer.js +81 -0
  55. package/dist/engine/scoring-algorithms.d.ts +31 -0
  56. package/dist/engine/scoring-algorithms.js +109 -0
  57. package/dist/engine/solution-matcher-eval.d.ts +97 -0
  58. package/dist/engine/solution-matcher-eval.js +122 -0
  59. package/dist/engine/solution-matcher.d.ts +21 -380
  60. package/dist/engine/solution-matcher.js +27 -828
  61. package/dist/fgx.js +1 -1
  62. package/dist/hooks/notepad-injector.js +7 -0
  63. package/dist/hooks/post-tool-use.js +8 -1
  64. package/dist/hooks/secret-filter.d.ts +1 -0
  65. package/dist/hooks/secret-filter.js +17 -7
  66. package/dist/hooks/shared/preflight-check.d.ts +15 -0
  67. package/dist/hooks/shared/preflight-check.js +51 -0
  68. package/dist/hooks/stop-guard.js +19 -60
  69. package/dist/hooks/subagent-stop-guard.d.ts +23 -0
  70. package/dist/hooks/subagent-stop-guard.js +158 -0
  71. package/dist/hooks/subagent-tracker.d.ts +36 -3
  72. package/dist/hooks/subagent-tracker.js +86 -39
  73. package/hooks/hooks.json +6 -1
  74. package/package.json +7 -7
  75. package/plugin.json +1 -1
  76. package/scripts/postinstall.js +10 -7
@@ -1,832 +1,38 @@
1
- import * as fs from 'node:fs';
1
+ /**
2
+ * Solution matcher — thin facade re-exporting from decomposed modules.
3
+ *
4
+ * All public exports are preserved for backward compatibility. Internal
5
+ * callers that import from './solution-matcher.js' continue to work.
6
+ *
7
+ * Module layout (post-decomposition):
8
+ * scoring-algorithms.ts — bigramSimilarity, bm25Score, tagWeight, COMMON_TAGS
9
+ * relevance-scorer.ts — calculateRelevance, CalculateRelevanceOptions
10
+ * precision-guards.ts — shouldRejectByR4T3Rules
11
+ * ranking-pipeline.ts — rankCandidates, RankableSolution, RankedCandidate
12
+ * solution-matcher-eval.ts — EvalSolution/Query/Fixture/Result, ROUND3_BASELINE, evaluateQuery, evaluateSolutionMatcher
13
+ * meta-learning/matcher-weight-loader.ts — loadTunedMatcherWeights
14
+ */
2
15
  import * as path from 'node:path';
3
- import { ME_SOLUTIONS, META_LEARNING_DIR, PACKS_DIR } from '../core/paths.js';
4
- import { maskBlockedTokens } from './phrase-blocklist.js';
5
- import { expandCompoundTags, expandQueryBigrams, extractTags } from './solution-format.js';
16
+ import { ME_SOLUTIONS, PACKS_DIR } from '../core/paths.js';
17
+ import { extractTags } from './solution-format.js';
6
18
  import { getOrBuildIndex } from './solution-index.js';
7
19
  import { defaultNormalizer } from './term-normalizer.js';
8
- // ── Synonym expansion (delegates to term-normalizer) ──
9
- //
10
- // The old `SYNONYM_MAP` + `expandTagsWithSynonyms` pair had two problems:
11
- // 1. The reverse-lookup `Object.entries(SYNONYM_MAP).filter(v => v.includes(tag))`
12
- // was O(N) per term and ran once per (query, solution) pair — quadratic
13
- // on the solution count.
14
- // 2. Korean↔English cross-mapping was maintained as two separate map entries
15
- // that drifted (fixed in 5.1.2 but fragile).
16
- //
17
- // Both are now handled by `src/engine/term-normalizer.ts`. See that file for
18
- // the canonical registry (`DEFAULT_MATCH_TERMS`) and the `buildTermNormalizer`
19
- // implementation. Reverse lookup is an O(1) `Map<term, canonicals>` fetch.
20
- //
21
- // The export below is kept as a thin backwards-compatible wrapper so
22
- // downstream callers (and the existing `synonym-tfidf.test.ts` spot-checks)
23
- // continue to work — but the hot path in this module now passes
24
- // pre-normalized query tags via the new `calculateRelevance` options arg
25
- // and skips the wrapper entirely.
20
+ import { rankCandidates } from './ranking-pipeline.js';
21
+ import { loadTunedMatcherWeights } from './meta-learning/matcher-weight-loader.js';
22
+ // ── Re-exports (backward compatibility) ──
23
+ export { bigramSimilarity, bm25Score, COMMON_TAGS, tagWeight } from './scoring-algorithms.js';
24
+ export { calculateRelevance } from './relevance-scorer.js';
25
+ export { shouldRejectByR4T3Rules } from './precision-guards.js';
26
+ export { ROUND3_BASELINE, BASELINE_TOLERANCE, evaluateQuery, evaluateSolutionMatcher, } from './solution-matcher-eval.js';
27
+ // ── Deprecated wrapper (kept for synonym-tfidf.test.ts) ──
26
28
  /**
27
29
  * @deprecated Use `defaultNormalizer.normalizeTerms` from
28
- * `./term-normalizer.js` directly. Kept as a thin wrapper for the existing
29
- * `synonym-tfidf.test.ts` and any external consumers.
30
+ * `./term-normalizer.js` directly.
30
31
  */
31
32
  export function expandTagsWithSynonyms(tags) {
32
33
  return defaultNormalizer.normalizeTerms(tags);
33
34
  }
34
- // ── TF-IDF weighting for common tags ──
35
- // ── Character bigram similarity (Dice coefficient) ──
36
- /**
37
- * Compute the Dice coefficient between two strings using character bigrams.
38
- *
39
- * Dice = 2 * |intersection| / (|A| + |B|)
40
- *
41
- * Both strings are lowercased and whitespace-stripped before bigram generation.
42
- * Returns 0 for empty strings or single-character strings (no bigrams possible).
43
- * Returns 1.0 for identical non-trivial strings.
44
- *
45
- * This is used as a lightweight fuzzy matching signal for borderline cases
46
- * where the TF-IDF tag intersection produces a low score but the query and
47
- * solution tags are character-similar (e.g., "database" vs "데이터베이스"
48
- * won't match, but "database" vs "databse" will get a high score).
49
- */
50
- export function bigramSimilarity(a, b) {
51
- const na = a.toLowerCase().replace(/\s+/g, '');
52
- const nb = b.toLowerCase().replace(/\s+/g, '');
53
- if (na.length < 2 || nb.length < 2)
54
- return 0;
55
- if (na === nb)
56
- return 1.0;
57
- const bigramsA = new Map();
58
- for (let i = 0; i < na.length - 1; i++) {
59
- const bg = na.slice(i, i + 2);
60
- bigramsA.set(bg, (bigramsA.get(bg) ?? 0) + 1);
61
- }
62
- const bigramsB = new Map();
63
- for (let i = 0; i < nb.length - 1; i++) {
64
- const bg = nb.slice(i, i + 2);
65
- bigramsB.set(bg, (bigramsB.get(bg) ?? 0) + 1);
66
- }
67
- let intersectionSize = 0;
68
- for (const [bg, countA] of bigramsA) {
69
- const countB = bigramsB.get(bg) ?? 0;
70
- intersectionSize += Math.min(countA, countB);
71
- }
72
- const totalA = na.length - 1;
73
- const totalB = nb.length - 1;
74
- return (2 * intersectionSize) / (totalA + totalB);
75
- }
76
- // ── BM25-like scoring ──
77
- /**
78
- * Simplified BM25 score for a single query-document pair.
79
- * Uses tag overlap with term frequency normalization.
80
- * k1=1.2, b=0.75 (standard BM25 parameters).
81
- */
82
- export function bm25Score(queryTags, docTags, avgDocLength) {
83
- const k1 = 1.2;
84
- const b = 0.75;
85
- const docLen = docTags.length;
86
- if (docLen === 0 || queryTags.length === 0 || avgDocLength === 0)
87
- return 0;
88
- let score = 0;
89
- for (const qt of queryTags) {
90
- // Term frequency in document
91
- const tf = docTags.filter((dt) => dt === qt || (dt.length > 3 && qt.length > 3 && (dt.includes(qt) || qt.includes(dt)))).length;
92
- if (tf === 0)
93
- continue;
94
- // BM25 TF saturation
95
- const numerator = tf * (k1 + 1);
96
- const denominator = tf + k1 * (1 - b + b * (docLen / avgDocLength));
97
- score += numerator / denominator;
98
- }
99
- // Normalize by query length
100
- return score / queryTags.length;
101
- }
102
- /** High-frequency tags that should be weighted lower */
103
- const COMMON_TAGS = new Set([
104
- 'typescript',
105
- 'ts',
106
- 'javascript',
107
- 'js',
108
- 'fix',
109
- 'update',
110
- 'add',
111
- 'change',
112
- 'file',
113
- 'code',
114
- 'function',
115
- 'import',
116
- 'export',
117
- 'error',
118
- 'type',
119
- 'string',
120
- 'number',
121
- 'object',
122
- 'array',
123
- 'return',
124
- 'const',
125
- 'class',
126
- 'module',
127
- '코드',
128
- '파일',
129
- '함수',
130
- '수정',
131
- '추가',
132
- '변경',
133
- '에러',
134
- '타입',
135
- ]);
136
- /** Apply IDF-like weight: common tags get reduced weight */
137
- export function tagWeight(tag) {
138
- return COMMON_TAGS.has(tag) ? 0.5 : 1.0;
139
- }
140
- export function calculateRelevance(promptOrTags, keywordsOrTags, confidence, options) {
141
- if (typeof promptOrTags === 'string') {
142
- // Legacy mode: substring matching for backwards compatibility.
143
- // Not a hot path — only hit by the (old) solution-matcher.test.ts cases.
144
- const promptTags = extractTags(promptOrTags);
145
- const intersection = keywordsOrTags.filter((kw) => promptTags.some((pt) => pt === kw || (pt.length > 3 && kw.length > 3 && (pt.startsWith(kw) || kw.startsWith(pt)))));
146
- return Math.min(1, intersection.length / Math.max(promptTags.length * 0.5, 1));
147
- }
148
- // v3 mode: tag matching with synonym expansion + TF-IDF weighting.
149
- //
150
- // T2: the synonym expansion is now a hash-indexed lookup via
151
- // `defaultNormalizer.normalizeTerms` (see term-normalizer.ts). Callers in
152
- // the hot path pre-compute the expansion once per query and pass it via
153
- // `options.normalizedPromptTags`, so this function no longer repeats the
154
- // work per solution.
155
- const expandedPromptTags = options?.normalizedPromptTags ?? defaultNormalizer.normalizeTerms(promptOrTags);
156
- // R4-T1: when the caller supplies a compound-expanded solution tag set,
157
- // intersection and partial matching run against the expanded set (so
158
- // `api-key` matches `api`/`key` queries via the split parts), but the
159
- // Jaccard union denominator below still uses the RAW `keywordsOrTags`
160
- // for normalization stability.
161
- const matchTags = options?.solutionTagsExpanded ?? keywordsOrTags;
162
- const intersection = matchTags.filter((t) => expandedPromptTags.includes(t));
163
- // partial/substring matches for longer tags (>3 chars)
164
- const partialMatches = matchTags.filter((t) => t.length > 3 &&
165
- !intersection.includes(t) &&
166
- expandedPromptTags.some((pt) => pt.length > 3 && (pt.includes(t) || t.includes(pt))));
167
- // Apply TF-IDF weighting: common tags count less
168
- const weightedMatched = intersection.reduce((sum, t) => sum + tagWeight(t), 0) +
169
- partialMatches.reduce((sum, t) => sum + tagWeight(t) * 0.5, 0);
170
- // ── Bigram similarity boost for borderline cases ──
171
- //
172
- // When the TF-IDF intersection score is below the match threshold (0.5),
173
- // compute a character-bigram Dice coefficient between the query tags and
174
- // the solution tags. If the best bigram similarity is high enough, blend
175
- // it in at 20% weight (TF-IDF 80%, bigram 20%) to rescue fuzzy matches
176
- // that the exact/substring intersection missed (e.g., typos, slight
177
- // morphological variants).
178
- //
179
- // When TF-IDF score is already above threshold, the bigram boost is NOT
180
- // applied — this preserves existing match quality and avoids disturbing
181
- // already-good rankings. The bigram path is purely a rescue mechanism
182
- // for borderline cases.
183
- if (weightedMatched < 0.5) {
184
- // Compute best bigram similarity across all (promptTag, solutionTag) pairs
185
- let bestBigramScore = 0;
186
- const bigramMatchedTags = [];
187
- for (const st of matchTags) {
188
- for (const pt of expandedPromptTags) {
189
- const sim = bigramSimilarity(pt, st);
190
- if (sim > bestBigramScore) {
191
- bestBigramScore = sim;
192
- }
193
- // Track solution tags with meaningful bigram similarity (> 0.4)
194
- if (sim > 0.4 && !bigramMatchedTags.includes(st)) {
195
- bigramMatchedTags.push(st);
196
- }
197
- }
198
- }
199
- // Only rescue if the bigram signal is strong enough (> 0.4 threshold)
200
- // to avoid noise from weakly similar strings
201
- if (bestBigramScore > 0.4) {
202
- const union = new Set([...promptOrTags, ...keywordsOrTags]).size;
203
- const tfidfScore = weightedMatched / Math.max(union, 1);
204
- const blendedScore = tfidfScore * 0.8 + bestBigramScore * 0.2;
205
- return {
206
- relevance: blendedScore * (confidence ?? 1),
207
- matchedTags: [
208
- ...intersection,
209
- ...partialMatches,
210
- ...bigramMatchedTags.filter((t) => !intersection.includes(t) && !partialMatches.includes(t)),
211
- ],
212
- };
213
- }
214
- return { relevance: 0, matchedTags: [] };
215
- }
216
- // Ensemble: TF-IDF (Jaccard) 0.5 + BM25 0.3 + bigram 0.2
217
- const union = new Set([...promptOrTags, ...keywordsOrTags]).size;
218
- const tfidfScore = weightedMatched / Math.max(union, 1);
219
- // BM25 component: average doc length defaults to 6 tags (typical solution)
220
- const avgDocLen = options?.avgDocLength ?? 6;
221
- const bm25 = bm25Score(promptOrTags, keywordsOrTags, avgDocLen);
222
- // Bigram component (mild boost for partial string matches)
223
- let bigramBoost = 0;
224
- for (const st of matchTags) {
225
- for (const pt of expandedPromptTags) {
226
- const sim = bigramSimilarity(pt, st);
227
- if (sim > bigramBoost)
228
- bigramBoost = sim;
229
- }
230
- }
231
- const w = options?.ensembleWeights ?? { tfidf: 0.5, bm25: 0.3, bigram: 0.2 };
232
- const ensembleScore = tfidfScore * w.tfidf + bm25 * w.bm25 + bigramBoost * w.bigram;
233
- return {
234
- relevance: ensembleScore * (confidence ?? 1),
235
- matchedTags: [...intersection, ...partialMatches],
236
- };
237
- }
238
- // ── R4-T3: query-side specificity guards (orchestration layer) ──
239
- //
240
- // Two narrow precision rules applied AFTER `calculateRelevance` returns,
241
- // at the orchestration layer (`rankCandidates`, `searchSolutions`).
242
- // These rules fix the 2 surviving false positives from R4-T2 — the
243
- // "validation of insurance claims" and "database backup recovery
244
- // procedure" residuals — WITHOUT regressing any legitimate fixture
245
- // positive or paraphrase.
246
- //
247
- // Why orchestration-level (not inside calculateRelevance):
248
- // `calculateRelevance` is a pure scoring function with a stable
249
- // contract: given (promptTags, solutionTags, confidence), return the
250
- // relevance and the matched tag set. Several internal tests
251
- // (synonym-tfidf.test.ts) call it directly with single-token inputs
252
- // to verify synonym expansion in isolation. Embedding precision
253
- // filters in the scoring path would break those tests AND break the
254
- // semantic of "scoring is a pure function". The two rules below are
255
- // policy-layer decisions about which scored candidates to surface,
256
- // so they belong at the caller — not at the scorer.
257
- //
258
- // Rule A — single-token query AND single-tag match → reject.
259
- // Rationale: a query that's been reduced to a single dev token (after
260
- // R4-T2 phrase masking) is unlikely to be a real dev question. Combined
261
- // with a single-tag match, this is the "validation of insurance
262
- // claims" shape: masked to `[validation]`, matched a single ambiguous
263
- // tag `validation` on error-handling-patterns. No legitimate fixture
264
- // positive or paraphrase has both promptTags.length === 1 AND
265
- // matchedTags.length === 1.
266
- //
267
- // Rule B — all matched tags came via SYNONYM EXPANSION (none appear
268
- // literally in the prompt tokens) AND match is single-tag → reject.
269
- // Rationale: the "database backup recovery procedure" shape. After
270
- // R4-T2 masks `database`/`backup`, the residual tokens are `[recovery,
271
- // procedure]`. The matched tag is `handling` — which appears nowhere
272
- // in the query. It only matches because the term-normalizer's
273
- // `handling` canonical includes `recovery` as a matchTerm (legitimate
274
- // for "error recovery handler" queries). The rule rejects this
275
- // expansion-only single-tag match because the query carries no
276
- // LITERAL signal that the matched solution is relevant. Multi-tag
277
- // expansion matches are NOT rejected — those indicate the canonical
278
- // family is being hit from multiple angles ("버그 재현 시스템적으로"
279
- // hits debugging-systematic via both `debug` and `debugging` — two
280
- // distinct matches survive).
281
- //
282
- // Literal hit: a matched tag is "literal" with respect to the query if
283
- // any of the following holds for some prompt token `pt`:
284
- // 1. `pt === tag` (exact verbatim match in the query)
285
- // 2. `pt` is a substring of `tag` or vice versa, with both length > 3
286
- // (mirrors the partialMatches discovery rule in calculateRelevance —
287
- // e.g., `code` (query) ↔ `code-review` (matched tag))
288
- // 3. `pt` and `tag` share a common prefix of length ≥ 4 (catches
289
- // morphological variants like `caching` ↔ `cache`, `cached` ↔
290
- // `cache`, `documents` ↔ `document` where neither is a substring
291
- // of the other but both clearly come from the same stem)
292
- //
293
- // Rule (3) is the defensive precision fix: without it, a query like
294
- // "caching strategy" (which the term-normalizer expands `caching → cache`
295
- // via the cache canonical) would have its single-tag `cache` match
296
- // rejected by Rule B, even though `caching` is morphologically the same
297
- // concept. The 4-char threshold is the same as the partialMatches rule
298
- // to keep the literal-hit semantics consistent across the matcher.
299
- //
300
- // Returns true if the candidate should be rejected (caller filters
301
- // it out), false if the candidate passes both rules.
302
- export function shouldRejectByR4T3Rules(promptTags, matchedTags) {
303
- // Rule A
304
- if (promptTags.length === 1 && matchedTags.length === 1) {
305
- return true;
306
- }
307
- // Rule B
308
- if (matchedTags.length === 1) {
309
- const tag = matchedTags[0];
310
- const literalHit = promptTags.includes(tag) ||
311
- promptTags.some((pt) => {
312
- if (pt.length <= 3 || tag.length <= 3)
313
- return false;
314
- if (pt.includes(tag) || tag.includes(pt))
315
- return true;
316
- // Morphological stem: shared prefix of length ≥ 4
317
- let i = 0;
318
- const limit = Math.min(pt.length, tag.length);
319
- while (i < limit && pt[i] === tag[i])
320
- i++;
321
- return i >= 4;
322
- });
323
- if (!literalHit)
324
- return true;
325
- }
326
- return false;
327
- }
328
- /**
329
- * Shared ranking core: tag-based relevance + identifier boost + top-5 sort.
330
- *
331
- * Single source of truth for the matcher's ranking behaviour. Both
332
- * `matchSolutions` (production, reads from the index) and
333
- * `evaluateSolutionMatcher` (bootstrap eval, reads from an in-memory fixture)
334
- * call through here so the eval metrics track reality — any future
335
- * ranking-logic change only needs to happen in one place.
336
- *
337
- * Contract:
338
- * - identifier boost requires `id.length >= 4` (STRONG_ID_MIN_LENGTH mirror)
339
- * and substring presence in the prompt (case-insensitive).
340
- * - candidates with zero matched tags AND zero matched identifiers are dropped.
341
- * - top-5 by `relevance` descending.
342
- * - duplicate names are NOT deduplicated — that matches the pre-refactor
343
- * `matchSolutions` behaviour (both scopes could rank). Callers that want
344
- * first-wins scope precedence must dedupe on their side.
345
- */
346
- function rankCandidates(promptTags, promptLower, solutions, ensembleWeights) {
347
- // T2: normalize prompt tags ONCE per query (not once per solution).
348
- // Pre-T2 this expansion happened inside calculateRelevance and was
349
- // repeated N times for N solutions — the plan's primary hot-path win.
350
- //
351
- // R4-T2: BEFORE any expansion or normalization, mask out tokens that
352
- // belong to blocked English phrases ("performance review", "system
353
- // architecture", etc.). This is a precision filter for non-dev-context
354
- // false positives. The mask runs first so neither bigram expansion nor
355
- // canonical normalization can re-introduce a masked token via synonyms
356
- // or compound recovery — the masked tokens are simply removed from the
357
- // matching pipeline. See `phrase-blocklist.ts` for the full rationale
358
- // and the `maskBlockedTokens` contract.
359
- const maskedPromptTags = maskBlockedTokens(promptLower, promptTags);
360
- if (maskedPromptTags.length === 0)
361
- return [];
362
- //
363
- // R4-T1: also expand the prompt tags with adjacent-token bigrams BEFORE
364
- // running the canonical normalizer. `expandQueryBigrams` produces compound
365
- // forms like `api-key`, `apikey`, `api-keys`, `apikeys` from the raw
366
- // ['api', 'keys'] token pair, so a query "api keys" can hit a solution
367
- // tag `api-key` via direct intersection — without depending on the
368
- // partialMatches half-weight fallback. The bigram expansion is layered
369
- // BEFORE normalization so that `apikey → api` (via the api canonical
370
- // family) still works.
371
- //
372
- // Note: we intentionally do NOT use `sol.normalizedTags` (if present) for
373
- // the intersection. Using normalized on BOTH sides is bidirectional
374
- // expansion that inflates Jaccard intersection 5-10× and silently shifts
375
- // every baseline metric. `entry.normalizedTags` is populated by the
376
- // index but reserved for log explainability. If a future change uses it
377
- // in scoring, it must update ROUND3_BASELINE in the same PR.
378
- const promptTagsWithBigrams = expandQueryBigrams(maskedPromptTags);
379
- const normalizedPromptTags = defaultNormalizer.normalizeTerms(promptTagsWithBigrams);
380
- return solutions
381
- .map((sol) => {
382
- // R4-T1: solution-side compound-tag expansion. `api-key` becomes
383
- // {api-key, api, key} so a query token `api` (from "api keys") hits
384
- // it directly. Computed per solution because each sol.tags is
385
- // independent — caching across the rank loop is not worth the
386
- // bookkeeping for the corpus sizes Forgen targets (N ≤ 200).
387
- const solTagsExpanded = expandCompoundTags(sol.tags);
388
- // R4-T2: pass `maskedPromptTags` (not the original `promptTags`) as
389
- // the first arg so the Jaccard union denominator inside
390
- // calculateRelevance reflects the post-mask tag set. The matching
391
- // step (intersection/partialMatches) already uses the masked set
392
- // via `normalizedPromptTags` — the union must match for score
393
- // semantics to stay consistent.
394
- const result = calculateRelevance(maskedPromptTags, sol.tags, sol.confidence, {
395
- normalizedPromptTags,
396
- solutionTagsExpanded: solTagsExpanded,
397
- ensembleWeights,
398
- });
399
- // Compute identifier boost FIRST — independent of tag scoring so
400
- // R4-T3's tag-evidence precision rules below cannot silently drop
401
- // a candidate that has strong identifier-level evidence.
402
- let identifierBoost = 0;
403
- const matchedIdentifiers = [];
404
- for (const id of sol.identifiers ?? []) {
405
- if (id.length >= 4 && promptLower.includes(id.toLowerCase())) {
406
- identifierBoost += 0.15;
407
- matchedIdentifiers.push(id);
408
- }
409
- }
410
- // R4-T3: orchestration-layer specificity guards. Reject single-tag
411
- // matches that lack a corroborating signal (single-token query OR
412
- // all-via-expansion match). See `shouldRejectByR4T3Rules` for the
413
- // full rule rationale.
414
- //
415
- // Identifier evidence is the escape hatch: if the query literally
416
- // mentioned one of the solution's identifiers (e.g. a function or
417
- // file name), the R4-T3 tag-precision rules are bypassed because
418
- // the identifier hit is itself a strong-specificity signal. Only
419
- // the tag evidence is zeroed out when R4-T3 fires; the identifier
420
- // boost and matched identifiers are preserved, so a candidate with
421
- // a single weak tag match but a valid identifier still survives
422
- // the `matchedTags.length + matchedIdentifiers.length >= 1` filter.
423
- let tagRelevance = result.relevance;
424
- let tagMatches = result.matchedTags;
425
- if (matchedIdentifiers.length === 0 &&
426
- tagMatches.length > 0 &&
427
- shouldRejectByR4T3Rules(maskedPromptTags, tagMatches)) {
428
- tagRelevance = 0;
429
- tagMatches = [];
430
- }
431
- return {
432
- solution: sol,
433
- relevance: tagRelevance + identifierBoost,
434
- matchedTags: tagMatches,
435
- matchedIdentifiers,
436
- };
437
- })
438
- .filter((c) => c.matchedTags.length + c.matchedIdentifiers.length >= 1)
439
- .sort((a, b) => b.relevance - a.relevance)
440
- .slice(0, 5);
441
- }
442
- /**
443
- * Round 3 baseline metrics, recorded against the current `term-normalizer`
444
- * + `calculateRelevance` + fixture `solution-match-bootstrap.json`. Used as
445
- * a relative regression guard in `tests/solution-matcher-eval.test.ts` —
446
- * downstream PRs must not regress any field by more than `BASELINE_TOLERANCE`.
447
- *
448
- * History (chronological ascending — v1 at top, latest at bottom):
449
- * - v1 (2026-04-08, fixture v1, 41+10+10 queries): 1.0 / 1.0 / 0.0 / 0.1
450
- * Recorded against the original 61-query fixture, all positive queries
451
- * PASS@1. Indicated a measurement plateau but masked the matcher's true
452
- * ranking and false-positive weaknesses because the fixture queries were
453
- * too tag-aligned.
454
- *
455
- * - v2 (2026-04-08, fixture v2, 53+16+14 queries): 1.0 / 0.969 / 0.0 / 0.357
456
- * Expanded with 12 hard positive (multi-canonical / compound-tag tug-of-
457
- * war), 6 Korean subtle paraphrase, and 4 tricky negative queries. The
458
- * drops are intentional and represent genuine matcher behaviour:
459
- * * positive mrrAt5 1.0 → 0.959: 4 of 12 added positives rank #2-3:
460
- * (1) "managing api keys and credentials safely" → secret @3 vs
461
- * api-error-responses @1 — the `api` canonical in
462
- * DEFAULT_MATCH_TERMS expands to {api, rest, graphql, endpoint,
463
- * route}, so query `api` hits BOTH `api` AND `rest` on
464
- * starter-api-error-responses (matched=['api','rest']) — a
465
- * double-count numerator. starter-secret-management only scores
466
- * a single weak partial match on `credential`. The compound
467
- * `api-key` tag on secret-management is never reached because
468
- * extractTags strips the query-side hyphen and yields
469
- * ['api','keys'] (the solution-side tag remains hyphenated in
470
- * the index but has no query token to intersect with). T4 IDF
471
- * would down-weight both `api` and `rest`, neutralising the
472
- * double-count and letting `credential` outscore the noise.
473
- * (2) "avoiding hardcoded credentials in source code" → secret @2
474
- * vs code-review @1 — `code` partial-matches `code-review`
475
- * (len>3, code-review.includes('code')=true) at half weight.
476
- * secret-management's `credential` matches by partial too but
477
- * the union size differs.
478
- * (3) "red green refactor cycle for new features" → tdd @2 vs
479
- * refactor-safely @1 — `refactor` is a full-weight intersection
480
- * with both refactor-safely's `refactor` and `리팩토링` (via
481
- * the refactor canonical), giving 2 hits at 1.0 each. tdd-red-
482
- * green-refactor only matches the literal compound tag
483
- * `red-green-refactor` (one weighted hit) — the full-weight
484
- * generic `refactor` term overpowers the compound-tag specifity.
485
- * (4) "writing unit tests for a function with side effects" → tdd
486
- * @2 vs separation-of-concerns @1 — both solutions have a
487
- * SINGLE matching tag with weighted score 0.5: separation gets
488
- * `function` (COMMON_TAG, exact intersection, weight 0.5);
489
- * tdd-red-green-refactor gets `tests` partial-matching `test`
490
- * (len>3, partial weight 1.0 × 0.5 = 0.5). Both numerators are
491
- * identical. Separation wins because the `function` co-occurs
492
- * in both promptTags and solution.tags, shrinking its Jaccard
493
- * union by one element vs tdd's — a 1-element union-size
494
- * advantage drives the entire ranking. starter-dependency-
495
- * injection is *not* in top-5 despite having `testing`/`mock`/
496
- * `dependency` tags (`tests` does not partial-match `testing`
497
- * — neither is a substring of the other), so listing `di` in
498
- * expectAnyOf is purely defensive recall, not a live candidate.
499
- * T4 BM25 with proper length normalization would attack the
500
- * union-size tie-breaker more rigorously than current Jaccard.
501
- * * paraphrase mrrAt5 stays at 1.0: all 6 added Korean paraphrases
502
- * rank @1 (the originally hard "테스트 먼저 작성하고 리팩토링" is
503
- * documented in the fixture as legitimately matching either tdd
504
- * OR refactor-safely, since starter-refactor-safely's README also
505
- * covers test-first workflows — both are defensible answers).
506
- * * negativeAnyResultRate 0.1 → 0.357: 4 added tricky negatives all
507
- * trigger false positives via single common dev-adjacent words —
508
- * "performance review meeting notes" → caching (matches
509
- * `performance`), "system architecture overview document" →
510
- * separation-of-concerns (matches `architecture`), "database backup
511
- * recovery procedure" → n-plus-one-queries (matches `database`,
512
- * `query`, `데이터베이스`), "validation of insurance claims" →
513
- * error-handling (matches `validation`).
514
- * The original Round 3 plan staged these for T4 (BM25 + IDF). T4 was
515
- * EMPIRICALLY SKIPPED on 2026-04-08 — see
516
- * `docs/plans/2026-04-08-t4-bm25-skip-adr.md` for the full decision
517
- * record. Summary: BM25 prototypes (naive, hybrid Jaccard×IDF,
518
- * precision filter, soft penalty) all matched or underperformed the
519
- * current scorer on every metric. The starter corpus (N=15) is too
520
- * small for IDF to be informative, and the false positives are
521
- * semantic ("performance" is both a dev tag and an English noun) — not
522
- * statistical, so no frequency-based weighting can fix them. The real
523
- * follow-up candidates are tokenizer fix for compound tags, an n-gram
524
- * phrase matcher, and corpus growth — all deferred to Round 4 per the
525
- * ADR.
526
- *
527
- * - v3 (2026-04-08, fixture v2 + R4-T1 compound-tag fix): 1.0 / 0.986 / 0.0 / 0.357
528
- * R4-T1 added `expandCompoundTags` (solution-side) and
529
- * `expandQueryBigrams` (query-side) so hyphenated solution tags like
530
- * `api-key`, `code-review`, `red-green-refactor` participate in direct
531
- * intersection rather than relying on the half-weight partialMatches
532
- * fallback. positive `mrrAt5` improved 0.959 → 0.981 (+0.022). 2 of
533
- * the 4 v2 hard positive cases were resolved (`managing api keys and
534
- * credentials safely` and `red green refactor cycle for new features`
535
- * now rank @1). The remaining 2 (`avoiding hardcoded credentials …`
536
- * and `writing unit tests for a function with side effects`) require
537
- * R4-T2 (phrase matcher) or R4-T3 (specificity classifier) — they're
538
- * about query-side English semantics, not compound-tag tokenization.
539
- * `negativeAnyResultRate` is unchanged at 0.357 because R4-T1 is a
540
- * ranking-quality fix, not a false-positive filter.
541
- *
542
- * - v4 (2026-04-08, fixture v2 + R4-T1 + R4-T2 phrase blocklist):
543
- * 1.0 / 0.986 / 0.0 / 0.143
544
- * R4-T2 added `phrase-blocklist.ts` with 17 curated 2-word English
545
- * non-dev compounds ("performance review", "system architecture",
546
- * "database backup", etc.) and a `maskBlockedTokens` step at the
547
- * top of `rankCandidates` and `searchSolutions`. When a query
548
- * contains a blocked phrase, the constituent tokens are removed
549
- * from the prompt tag list before bigram expansion / canonical
550
- * normalization runs — so the false-positive evidence is removed
551
- * at the source rather than demoted in scoring.
552
- *
553
- * `negativeAnyResultRate` dropped 0.357 → 0.143 (3 of 5 v2 trigger
554
- * negatives fully blocked):
555
- * * "performance review meeting notes" — blocked via
556
- * `performance review` + `meeting notes`
557
- * * "system architecture overview document" — blocked via
558
- * `system architecture` + `overview document`
559
- * * "solar system planets astronomy" — blocked via `solar system`
560
- *
561
- * 2 false positives remain (both deferred to R4-T3 query-side
562
- * specificity classifier — the residuals share a common shape:
563
- * a single dev-tag homograph survives whatever masking is applied,
564
- * and the term-normalizer expansion still surfaces a false match):
565
- *
566
- * * "database backup recovery procedure" → error-handling-patterns:
567
- * `database backup` is blocked, but the residual tokens
568
- * {`recovery`, `procedure`} survive. `recovery` is in the
569
- * `handling` canonical's matchTerms (intentional, for legitimate
570
- * "error recovery handler" queries), so the masked query still
571
- * hits `starter-error-handling-patterns` via the handling
572
- * family. A 3-word `recovery procedure` blocklist entry was
573
- * considered and rejected — it would silently mask legitimate
574
- * dev SRE queries like "disaster recovery procedure" or
575
- * "rollback recovery procedure" without a fixture-driven
576
- * signal. The right fix is at the query-specificity layer
577
- * (R4-T3): require ≥ 2 distinct dev-context signals before any
578
- * match is returned, not at the phrase-blocklist layer.
579
- *
580
- * * "validation of insurance claims" → error-handling-patterns:
581
- * `insurance claim` is blocked, but the residual `validation`
582
- * token IS a legitimate dev tag (input-validation,
583
- * error-handling-patterns both have it). Same R4-T3 target.
584
- *
585
- * positive/paraphrase mrrAt5 are unchanged from v3 because no
586
- * legitimate dev query in the fixture contains a blocked phrase.
587
- *
588
- * - v5 (2026-04-08, fixture v2 + R4-T1 + R4-T2 + R4-T3 specificity guards):
589
- * 1.0 / 0.986 / 0.0 / 0.000
590
- * R4-T3 added two narrow precision rules at the ORCHESTRATION LAYER —
591
- * NOT inside `calculateRelevance` (which remains a pure scoring
592
- * function for test symmetry). The rules are implemented as the
593
- * exported helper `shouldRejectByR4T3Rules(promptTags, matchedTags)`
594
- * and called from both `rankCandidates` (hook path) and
595
- * `searchSolutions` (MCP path) right after the per-solution
596
- * `calculateRelevance` call:
597
- * (Rule A) single-token query AND single-tag match → reject;
598
- * (Rule B) single-tag match with no literal hit in the prompt
599
- * (verbatim match, or substring partial length > 3, or
600
- * shared prefix ≥ 4 for morphological stems) → reject.
601
- * Both rules are scoped narrowly enough to fix exactly the 2 R4-T2
602
- * residuals without recall regression — every fixture positive and
603
- * paraphrase still ranks identically:
604
- * * "validation of insurance claims" → masked to `[validation]`
605
- * (length 1) with single-tag match `validation` → Rule A reject.
606
- * * "database backup recovery procedure" → masked to
607
- * `[recovery, procedure]` with single-tag match `handling`
608
- * (zero literal hit; `handling` is reached via the `recovery`
609
- * canonical-family expansion in term-normalizer) → Rule B reject.
610
- * `negativeAnyResultRate` is now 0.000 — every fixture v2 negative
611
- * produces zero candidates. positive/paraphrase metrics unchanged
612
- * from v4 because no fixture positive matches the (single-token AND
613
- * single-tag) or (all-expansion AND single-tag) shape.
614
- *
615
- * Escape hatch: identifier-boost evidence (hook path) or name-match
616
- * evidence (MCP path) BYPASSES the R4-T3 rules. A candidate with
617
- * even a single weak tag match plus an identifier hit still
618
- * surfaces — the precision rules only fire when the candidate's
619
- * entire evidence pool is a single ambiguous tag.
620
- *
621
- * Defensive precision note: Rule B's "shared prefix ≥ 4"
622
- * morphological check is currently NOT fixture-driven (no fixture
623
- * query masks down to the `caching/cache`-style morphological gap).
624
- * It exists as a pre-emptive fix against silently rejecting
625
- * legitimate future queries where the term-normalizer synonym
626
- * expansion is the only bridge between the query token and the
627
- * solution tag. If a production query surfaces a case the prefix
628
- * check misses, extend it (e.g. by lowering the threshold or
629
- * adding a Levenshtein-1 check) rather than removing it.
630
- *
631
- * Known matcher quirks (separate from the T4 BM25 investigation):
632
- * - `term-normalizer.ts` `error` canonical contains `debug` as a matchTerm
633
- * (intentional for `bug → error` recall), which causes any prompt
634
- * containing `error` to expand to `debug` and over-rank
635
- * `starter-debugging-systematic` on otherwise unrelated queries. This
636
- * is why `async await error propagation` could not be added as a hard
637
- * case — the matcher returns debugging-systematic at #1, which is
638
- * defensible-but-noisy. The fix is at the normalizer level (split
639
- * `debug` out of the `error` family or remove the `error → debug`
640
- * edge entirely) and is queued as a Round 4 follow-up. T4 BM25 was
641
- * considered as a partial mitigation but the T4 skip ADR (referenced
642
- * in the Round 3 outcome paragraph above) shows it does not help.
643
- *
644
- * Long-tail caveat:
645
- * - `"trying to handle authentication errors gracefully when our backend
646
- * api returns inconsistent response formats from different
647
- * microservices"` is a 17-word query intentionally added to exercise
648
- * long-tail behaviour. Currently PASS@1. Originally flagged as BM25
649
- * length-normalization sensitive, but since T4 BM25 was skipped this
650
- * caveat is now informational only — no length-norm code path is
651
- * planned in Round 3.
652
- *
653
- * If a PR legitimately improves a metric, update this constant in the same
654
- * commit so future PRs guard against the new floor.
655
- */
656
- export const ROUND3_BASELINE = {
657
- recallAt5: 1.0,
658
- mrrAt5: 0.986,
659
- noResultRate: 0.0,
660
- negativeAnyResultRate: 0.0,
661
- byBucket: {
662
- positive: { recallAt5: 1.0, mrrAt5: 0.981, noResultRate: 0.0, total: 53 },
663
- paraphrase: { recallAt5: 1.0, mrrAt5: 1.0, noResultRate: 0.0, total: 16 },
664
- },
665
- total: { positive: 53, paraphrase: 16, negative: 14 },
666
- };
667
- /** Maximum allowed absolute regression per metric. 5% is tight enough to catch
668
- * ~3-4 query regressions in a 69-query combined bucket (positive+paraphrase)
669
- * but lenient enough that a single fixture edit won't spuriously fail the
670
- * guard. */
671
- export const BASELINE_TOLERANCE = 0.05;
672
- /** Run a single bucket through the ranking pipeline and aggregate IR metrics. */
673
- function computeBucketMetrics(queries, solutions) {
674
- let recallHits = 0;
675
- let reciprocalSum = 0;
676
- let noResultCount = 0;
677
- for (const q of queries) {
678
- const promptTags = extractTags(q.query);
679
- const ranked = rankCandidates(promptTags, q.query.toLowerCase(), solutions);
680
- if (ranked.length === 0) {
681
- noResultCount++;
682
- continue;
683
- }
684
- for (let i = 0; i < ranked.length; i++) {
685
- if (q.expectAnyOf.includes(ranked[i].solution.name)) {
686
- recallHits++;
687
- reciprocalSum += 1 / (i + 1);
688
- break;
689
- }
690
- }
691
- }
692
- const total = queries.length;
693
- return {
694
- recallAt5: total > 0 ? recallHits / total : 0,
695
- mrrAt5: total > 0 ? reciprocalSum / total : 0,
696
- noResultRate: total > 0 ? noResultCount / total : 0,
697
- total,
698
- };
699
- }
700
- /**
701
- * Test/diagnostic helper: evaluate one query against a fixture solution set
702
- * and return the top-5 ranked candidates with their relevance + matched tags.
703
- *
704
- * Exists so per-query regression tests (e.g. the R4-T1 hard-positive guards
705
- * in `tests/solution-matcher-eval.test.ts`) can assert specific ranking
706
- * outcomes without scraping aggregate metrics. Wraps `rankCandidates` so
707
- * the test path stays in sync with the production ranker.
708
- *
709
- * Returns the same shape as `rankCandidates` minus the generic carrier:
710
- * `{name, relevance, matchedTags}`. Use the names to assert "expected
711
- * solution at rank 1".
712
- */
713
- export function evaluateQuery(query, solutions) {
714
- const promptTags = extractTags(query);
715
- return rankCandidates(promptTags, query.toLowerCase(), solutions).map((c) => ({
716
- name: c.solution.name,
717
- relevance: c.relevance,
718
- matchedTags: c.matchedTags,
719
- }));
720
- }
721
- /**
722
- * Evaluate the current matcher against a labeled fixture and return IR
723
- * metrics. This is the Round 3 baseline — each downstream PR (T2/T3/T4) must
724
- * not regress any of the thresholds asserted in `solution-matcher-eval.test.ts`.
725
- *
726
- * Uses `rankCandidates` (shared with `matchSolutions`) so the evaluator can't
727
- * silently drift from production ranking behaviour.
728
- *
729
- * Metrics are reported both aggregated (positive ∪ paraphrase) and per-bucket,
730
- * so paraphrase-only regressions surface in `byBucket.paraphrase` even if the
731
- * aggregate looks fine.
732
- */
733
- export function evaluateSolutionMatcher(fixture) {
734
- const positiveM = computeBucketMetrics(fixture.positive, fixture.solutions);
735
- const paraphraseM = computeBucketMetrics(fixture.paraphrase, fixture.solutions);
736
- const combinedTotal = positiveM.total + paraphraseM.total;
737
- // Weighted aggregation: counts, not means — so a large positive bucket
738
- // doesn't drown a small paraphrase bucket but also a single-query bucket
739
- // doesn't dominate.
740
- const recallAt5 = combinedTotal > 0
741
- ? (positiveM.recallAt5 * positiveM.total + paraphraseM.recallAt5 * paraphraseM.total) /
742
- combinedTotal
743
- : 0;
744
- const mrrAt5 = combinedTotal > 0
745
- ? (positiveM.mrrAt5 * positiveM.total + paraphraseM.mrrAt5 * paraphraseM.total) /
746
- combinedTotal
747
- : 0;
748
- const noResultRate = combinedTotal > 0
749
- ? (positiveM.noResultRate * positiveM.total + paraphraseM.noResultRate * paraphraseM.total) /
750
- combinedTotal
751
- : 0;
752
- let negAnyResult = 0;
753
- for (const q of fixture.negative) {
754
- const promptTags = extractTags(q.query);
755
- const ranked = rankCandidates(promptTags, q.query.toLowerCase(), fixture.solutions);
756
- if (ranked.length >= 1)
757
- negAnyResult++;
758
- }
759
- const negTotal = fixture.negative.length;
760
- return {
761
- recallAt5,
762
- mrrAt5,
763
- noResultRate,
764
- negativeAnyResultRate: negTotal > 0 ? negAnyResult / negTotal : 0,
765
- byBucket: {
766
- positive: positiveM,
767
- paraphrase: paraphraseM,
768
- },
769
- total: {
770
- positive: fixture.positive.length,
771
- paraphrase: fixture.paraphrase.length,
772
- negative: fixture.negative.length,
773
- },
774
- };
775
- }
776
- // ── Meta-learning: dynamic ensemble weights ──
777
- let _cachedWeights;
778
- let _weightsCacheTime = 0;
779
- const WEIGHTS_CACHE_TTL = 60_000; // 1 minute cache
780
- /**
781
- * Load tuned matcher weights from meta-learning state.
782
- * Returns undefined (use defaults) if no tuned weights exist.
783
- * Cached for 1 minute to avoid re-reading per matchSolutions call.
784
- */
785
- function loadTunedMatcherWeights() {
786
- const now = Date.now();
787
- if (_cachedWeights !== undefined && now - _weightsCacheTime < WEIGHTS_CACHE_TTL) {
788
- return _cachedWeights ?? undefined;
789
- }
790
- try {
791
- const weightsPath = path.join(META_LEARNING_DIR, 'matcher-weights.json');
792
- if (!fs.existsSync(weightsPath)) {
793
- _cachedWeights = null;
794
- _weightsCacheTime = now;
795
- return undefined;
796
- }
797
- const data = JSON.parse(fs.readFileSync(weightsPath, 'utf-8'));
798
- if (typeof data.tfidf === 'number' &&
799
- typeof data.bm25 === 'number' &&
800
- typeof data.bigram === 'number') {
801
- _cachedWeights = { tfidf: data.tfidf, bm25: data.bm25, bigram: data.bigram };
802
- _weightsCacheTime = now;
803
- return _cachedWeights;
804
- }
805
- }
806
- catch {
807
- /* fail-open: use defaults */
808
- }
809
- _cachedWeights = null;
810
- _weightsCacheTime = now;
811
- return undefined;
812
- }
813
- /**
814
- * Cold-start exploration bonus for candidate solutions.
815
- *
816
- * Phase 4 evolution: newly proposed solutions enter at `status: candidate`.
817
- * Without a nudge they compete head-to-head with mature verified/champion
818
- * entries and almost always lose the first few rounds — not because
819
- * they're worse, but because matchers favor solutions with richer tag
820
- * histories. A small confidence multiplier lets candidates surface often
821
- * enough to accumulate reflected/sessions evidence, after which the
822
- * lifecycle loop decides their fate.
823
- *
824
- * The 1.3× factor is a starting point (Q1 in docs/design-solution-evolution.md).
825
- * Bonus deactivation happens implicitly when compound-lifecycle.ts::
826
- * runLifecycleCheck promotes the candidate to `verified` based on accumulated
827
- * reflected/sessions evidence. There is no inject-count-based auto promotion
828
- * (removed 2026-04-20 — see feedback_core_loop_invariant).
829
- */
35
+ // ── Candidate exploration bonus ──
830
36
  const CANDIDATE_EXPLORATION_MULTIPLIER = 1.3;
831
37
  function applyCandidateExplorationBonus(entries) {
832
38
  return entries.map((e) => {
@@ -835,25 +41,18 @@ function applyCandidateExplorationBonus(entries) {
835
41
  return { ...e, confidence: Math.min(1, e.confidence * CANDIDATE_EXPLORATION_MULTIPLIER) };
836
42
  });
837
43
  }
44
+ // ── Public API ──
838
45
  export function matchSolutions(prompt, scope, cwd) {
839
- // Build solution dirs for index cache
840
46
  const dirs = [{ dir: ME_SOLUTIONS, scope: 'me' }];
841
47
  if (scope.team) {
842
48
  dirs.push({ dir: path.join(PACKS_DIR, scope.team.name, 'solutions'), scope: 'team' });
843
49
  }
844
50
  dirs.push({ dir: path.join(cwd, '.compound', 'solutions'), scope: 'project' });
845
- // Use cached index (rebuilt only when dirs change)
846
51
  const index = getOrBuildIndex(dirs);
847
52
  const allSolutions = applyCandidateExplorationBonus(index.entries.map((e) => ({ ...e })));
848
53
  const promptTags = extractTags(prompt);
849
54
  const promptLower = prompt.toLowerCase();
850
- // Meta-learning: load tuned weights if available
851
55
  const tunedWeights = loadTunedMatcherWeights();
852
- // Delegate to shared ranking core. `rankCandidates` is generic so each
853
- // ranked candidate carries the original `LoadedSolution` reference — no
854
- // name-based re-lookup, so two scopes sharing a name (e.g. me/foo and
855
- // project/foo) can both appear in the result without a Map last-wins
856
- // scope-precedence bug.
857
56
  const ranked = rankCandidates(promptTags, promptLower, allSolutions, tunedWeights);
858
57
  return ranked.map((c) => ({
859
58
  name: c.solution.name,