@wooojin/forgen 0.4.10 → 0.4.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/CHANGELOG.md +46 -0
  3. package/README.md +33 -1
  4. package/assets/claude/agents/forgen-verify.md +65 -0
  5. package/assets/claude/workflows/compound-extract.js +136 -0
  6. package/assets/claude/workflows/evidence-gate-audit.js +107 -0
  7. package/assets/shared/hook-registry.json +1 -0
  8. package/dist/checks/_shared/meta-guard-dispatch.d.ts +38 -0
  9. package/dist/checks/_shared/meta-guard-dispatch.js +80 -0
  10. package/dist/checks/_shared/text-sanitizer.js +15 -0
  11. package/dist/cli.js +57 -2
  12. package/dist/core/changelog-cli.d.ts +7 -0
  13. package/dist/core/changelog-cli.js +100 -0
  14. package/dist/core/doctor.d.ts +3 -0
  15. package/dist/core/doctor.js +38 -0
  16. package/dist/core/effort-advisory.d.ts +23 -0
  17. package/dist/core/effort-advisory.js +29 -0
  18. package/dist/core/explain-cli.d.ts +6 -0
  19. package/dist/core/explain-cli.js +99 -0
  20. package/dist/core/health-cli.d.ts +23 -0
  21. package/dist/core/health-cli.js +86 -0
  22. package/dist/core/probe-workflow-cli.d.ts +72 -0
  23. package/dist/core/probe-workflow-cli.js +282 -0
  24. package/dist/core/stats-cli.d.ts +22 -9
  25. package/dist/core/stats-cli.js +149 -0
  26. package/dist/core/watch-cli.d.ts +7 -0
  27. package/dist/core/watch-cli.js +185 -0
  28. package/dist/core/workflows-cli.d.ts +26 -0
  29. package/dist/core/workflows-cli.js +120 -0
  30. package/dist/engine/compound-export.d.ts +12 -0
  31. package/dist/engine/compound-export.js +136 -14
  32. package/dist/engine/compound-extractor.d.ts +12 -43
  33. package/dist/engine/compound-extractor.js +27 -756
  34. package/dist/engine/extraction-diff.d.ts +11 -0
  35. package/dist/engine/extraction-diff.js +105 -0
  36. package/dist/engine/extraction-gates.d.ts +37 -0
  37. package/dist/engine/extraction-gates.js +100 -0
  38. package/dist/engine/extraction-git.d.ts +20 -0
  39. package/dist/engine/extraction-git.js +75 -0
  40. package/dist/engine/extraction-persistence.d.ts +27 -0
  41. package/dist/engine/extraction-persistence.js +140 -0
  42. package/dist/engine/extraction-session.d.ts +26 -0
  43. package/dist/engine/extraction-session.js +230 -0
  44. package/dist/engine/lifecycle/types.d.ts +1 -1
  45. package/dist/engine/meta-learning/matcher-weight-loader.d.ts +16 -0
  46. package/dist/engine/meta-learning/matcher-weight-loader.js +45 -0
  47. package/dist/engine/precision-guards.d.ts +14 -0
  48. package/dist/engine/precision-guards.js +39 -0
  49. package/dist/engine/ranking-pipeline.d.ts +45 -0
  50. package/dist/engine/ranking-pipeline.js +66 -0
  51. package/dist/engine/relevance-scorer.d.ts +43 -0
  52. package/dist/engine/relevance-scorer.js +81 -0
  53. package/dist/engine/scoring-algorithms.d.ts +31 -0
  54. package/dist/engine/scoring-algorithms.js +109 -0
  55. package/dist/engine/solution-matcher-eval.d.ts +97 -0
  56. package/dist/engine/solution-matcher-eval.js +122 -0
  57. package/dist/engine/solution-matcher.d.ts +21 -380
  58. package/dist/engine/solution-matcher.js +27 -828
  59. package/dist/fgx.js +1 -1
  60. package/dist/hooks/notepad-injector.js +7 -0
  61. package/dist/hooks/post-tool-use.js +8 -1
  62. package/dist/hooks/shared/preflight-check.d.ts +15 -0
  63. package/dist/hooks/shared/preflight-check.js +51 -0
  64. package/dist/hooks/stop-guard.js +19 -60
  65. package/dist/hooks/subagent-stop-guard.d.ts +23 -0
  66. package/dist/hooks/subagent-stop-guard.js +158 -0
  67. package/dist/hooks/subagent-tracker.d.ts +36 -3
  68. package/dist/hooks/subagent-tracker.js +86 -39
  69. package/hooks/hooks.json +6 -1
  70. package/package.json +7 -7
  71. package/plugin.json +1 -1
  72. package/scripts/postinstall.js +10 -7
@@ -1,34 +1,30 @@
1
- import type { ScopeInfo } from '../core/types.js';
2
- import type { SolutionStatus, SolutionType } from './solution-format.js';
3
1
  /**
4
- * @deprecated Use `defaultNormalizer.normalizeTerms` from
5
- * `./term-normalizer.js` directly. Kept as a thin wrapper for the existing
6
- * `synonym-tfidf.test.ts` and any external consumers.
7
- */
8
- export declare function expandTagsWithSynonyms(tags: string[]): string[];
9
- /**
10
- * Compute the Dice coefficient between two strings using character bigrams.
11
- *
12
- * Dice = 2 * |intersection| / (|A| + |B|)
2
+ * Solution matcher — thin facade re-exporting from decomposed modules.
13
3
  *
14
- * Both strings are lowercased and whitespace-stripped before bigram generation.
15
- * Returns 0 for empty strings or single-character strings (no bigrams possible).
16
- * Returns 1.0 for identical non-trivial strings.
4
+ * All public exports are preserved for backward compatibility. Internal
5
+ * callers that import from './solution-matcher.js' continue to work.
17
6
  *
18
- * This is used as a lightweight fuzzy matching signal for borderline cases
19
- * where the TF-IDF tag intersection produces a low score but the query and
20
- * solution tags are character-similar (e.g., "database" vs "데이터베이스"
21
- * won't match, but "database" vs "databse" will get a high score).
7
+ * Module layout (post-decomposition):
8
+ * scoring-algorithms.ts — bigramSimilarity, bm25Score, tagWeight, COMMON_TAGS
9
+ * relevance-scorer.ts — calculateRelevance, CalculateRelevanceOptions
10
+ * precision-guards.ts — shouldRejectByR4T3Rules
11
+ * ranking-pipeline.ts — rankCandidates, RankableSolution, RankedCandidate
12
+ * solution-matcher-eval.ts — EvalSolution/Query/Fixture/Result, ROUND3_BASELINE, evaluateQuery, evaluateSolutionMatcher
13
+ * meta-learning/matcher-weight-loader.ts — loadTunedMatcherWeights
22
14
  */
23
- export declare function bigramSimilarity(a: string, b: string): number;
15
+ import type { ScopeInfo } from '../core/types.js';
16
+ import type { SolutionStatus, SolutionType } from './solution-format.js';
17
+ export { bigramSimilarity, bm25Score, COMMON_TAGS, tagWeight } from './scoring-algorithms.js';
18
+ export { calculateRelevance } from './relevance-scorer.js';
19
+ export type { CalculateRelevanceOptions } from './relevance-scorer.js';
20
+ export { shouldRejectByR4T3Rules } from './precision-guards.js';
21
+ export { ROUND3_BASELINE, BASELINE_TOLERANCE, evaluateQuery, evaluateSolutionMatcher, } from './solution-matcher-eval.js';
22
+ export type { EvalSolution, EvalQuery, EvalFixture, BucketMetrics, EvalResult, } from './solution-matcher-eval.js';
24
23
  /**
25
- * Simplified BM25 score for a single query-document pair.
26
- * Uses tag overlap with term frequency normalization.
27
- * k1=1.2, b=0.75 (standard BM25 parameters).
24
+ * @deprecated Use `defaultNormalizer.normalizeTerms` from
25
+ * `./term-normalizer.js` directly.
28
26
  */
29
- export declare function bm25Score(queryTags: string[], docTags: string[], avgDocLength: number): number;
30
- /** Apply IDF-like weight: common tags get reduced weight */
31
- export declare function tagWeight(tag: string): number;
27
+ export declare function expandTagsWithSynonyms(tags: string[]): string[];
32
28
  export interface SolutionMatch {
33
29
  name: string;
34
30
  path: string;
@@ -41,361 +37,6 @@ export interface SolutionMatch {
41
37
  tags: string[];
42
38
  identifiers: string[];
43
39
  matchedTags: string[];
44
- /**
45
- * Identifier substrings (function/file names) that appeared literally in the
46
- * prompt. Added 2026-04-21 so solution-injector can enforce a precision gate
47
- * distinguishing "user typed a specific identifier" (strong signal, survives
48
- * 1-tag overlap) from "only 1 tag happens to overlap" (often noise — common
49
- * nouns like 'type', 'file', 'forgen' trigger rare-tag BM25 boost).
50
- */
51
40
  matchedIdentifiers: string[];
52
41
  }
53
- /**
54
- * Optional hints for the v3 `calculateRelevance` path. Used by hot-path
55
- * callers (matchSolutions, searchSolutions) to avoid re-normalizing the
56
- * same query tags on every solution.
57
- */
58
- export interface CalculateRelevanceOptions {
59
- /**
60
- * Pre-normalized prompt tags (produced by `defaultNormalizer.normalizeTerms`).
61
- * If provided, skips the per-call expansion. Callers loop-running against
62
- * many solutions should compute this once outside the loop and pass it in.
63
- */
64
- normalizedPromptTags?: string[];
65
- /**
66
- * R4-T1: solution tags expanded with compound-split alternatives
67
- * (`expandCompoundTags`). When supplied, the intersection/partial-match
68
- * step uses this set INSTEAD of `solutionTags`, but the Jaccard union
69
- * denominator still uses `solutionTags` (raw) so the score normalization
70
- * stays semantically stable. Caller responsibility to pass the matching
71
- * pair — `solutionTagsExpanded` MUST be a superset of `solutionTags`.
72
- */
73
- solutionTagsExpanded?: string[];
74
- /** Average document (solution) tag count for BM25 normalization. Defaults to 6. */
75
- avgDocLength?: number;
76
- /** Meta-learning: dynamic ensemble weights (sum must equal 1.0). Defaults to {tfidf:0.5, bm25:0.3, bigram:0.2}. */
77
- ensembleWeights?: {
78
- tfidf: number;
79
- bm25: number;
80
- bigram: number;
81
- };
82
- }
83
- export declare function calculateRelevance(promptTags: string[], solutionTags: string[], confidence: number, options?: CalculateRelevanceOptions): {
84
- relevance: number;
85
- matchedTags: string[];
86
- };
87
- /** @deprecated */
88
- export declare function calculateRelevance(prompt: string, keywords: string[]): number;
89
- export declare function shouldRejectByR4T3Rules(promptTags: readonly string[], matchedTags: readonly string[]): boolean;
90
- /**
91
- * In-memory solution shape for the bootstrap evaluator. Mirrors the index
92
- * entry fields that `matchSolutions` consumes (tags, identifiers, confidence)
93
- * but without any filesystem dependency — the evaluator is pure so CI can run
94
- * it without mounting a starter pack.
95
- */
96
- export interface EvalSolution {
97
- name: string;
98
- tags: string[];
99
- identifiers?: string[];
100
- confidence: number;
101
- }
102
- export interface EvalQuery {
103
- query: string;
104
- /** Names that should appear in the top-5. Empty array = expect no match (negative case). */
105
- expectAnyOf: string[];
106
- }
107
- export interface EvalFixture {
108
- solutions: EvalSolution[];
109
- positive: EvalQuery[];
110
- /** Bilingual or compound-word variants that exercise synonym expansion. */
111
- paraphrase: EvalQuery[];
112
- /** Unrelated queries that should not return a top-1 hit. */
113
- negative: EvalQuery[];
114
- }
115
- /** Per-bucket metrics. Paraphrase and positive are reported separately so a
116
- * bilingual regression (T2 synonym change) can't hide inside the aggregate. */
117
- export interface BucketMetrics {
118
- /** |{q : ∃i≤5, ranked[i] ∈ q.expectAnyOf}| / |q| */
119
- recallAt5: number;
120
- /** Σ (1 / firstMatchRank) / |q|; rank > 5 contributes 0. */
121
- mrrAt5: number;
122
- /** |{q : ranked is empty}| / |q| */
123
- noResultRate: number;
124
- /** Number of queries in this bucket. */
125
- total: number;
126
- }
127
- export interface EvalResult {
128
- /** Combined (positive ∪ paraphrase) metrics — backwards-compatible headline numbers. */
129
- recallAt5: number;
130
- mrrAt5: number;
131
- noResultRate: number;
132
- /**
133
- * Fraction of negative queries where the matcher returned ≥ 1 candidate
134
- * (regardless of rank). Name is honest: this is the "any result" rate on
135
- * the negative bucket, not a rank-1 precision metric. It's the correct
136
- * baseline for "did synonym/stemming leak into unrelated queries?".
137
- */
138
- negativeAnyResultRate: number;
139
- /** Per-bucket breakdown — use these to catch paraphrase-only regressions. */
140
- byBucket: {
141
- positive: BucketMetrics;
142
- paraphrase: BucketMetrics;
143
- };
144
- total: {
145
- positive: number;
146
- paraphrase: number;
147
- negative: number;
148
- };
149
- }
150
- /**
151
- * Round 3 baseline metrics, recorded against the current `term-normalizer`
152
- * + `calculateRelevance` + fixture `solution-match-bootstrap.json`. Used as
153
- * a relative regression guard in `tests/solution-matcher-eval.test.ts` —
154
- * downstream PRs must not regress any field by more than `BASELINE_TOLERANCE`.
155
- *
156
- * History (chronological ascending — v1 at top, latest at bottom):
157
- * - v1 (2026-04-08, fixture v1, 41+10+10 queries): 1.0 / 1.0 / 0.0 / 0.1
158
- * Recorded against the original 61-query fixture, all positive queries
159
- * PASS@1. Indicated a measurement plateau but masked the matcher's true
160
- * ranking and false-positive weaknesses because the fixture queries were
161
- * too tag-aligned.
162
- *
163
- * - v2 (2026-04-08, fixture v2, 53+16+14 queries): 1.0 / 0.969 / 0.0 / 0.357
164
- * Expanded with 12 hard positive (multi-canonical / compound-tag tug-of-
165
- * war), 6 Korean subtle paraphrase, and 4 tricky negative queries. The
166
- * drops are intentional and represent genuine matcher behaviour:
167
- * * positive mrrAt5 1.0 → 0.959: 4 of 12 added positives rank #2-3:
168
- * (1) "managing api keys and credentials safely" → secret @3 vs
169
- * api-error-responses @1 — the `api` canonical in
170
- * DEFAULT_MATCH_TERMS expands to {api, rest, graphql, endpoint,
171
- * route}, so query `api` hits BOTH `api` AND `rest` on
172
- * starter-api-error-responses (matched=['api','rest']) — a
173
- * double-count numerator. starter-secret-management only scores
174
- * a single weak partial match on `credential`. The compound
175
- * `api-key` tag on secret-management is never reached because
176
- * extractTags strips the query-side hyphen and yields
177
- * ['api','keys'] (the solution-side tag remains hyphenated in
178
- * the index but has no query token to intersect with). T4 IDF
179
- * would down-weight both `api` and `rest`, neutralising the
180
- * double-count and letting `credential` outscore the noise.
181
- * (2) "avoiding hardcoded credentials in source code" → secret @2
182
- * vs code-review @1 — `code` partial-matches `code-review`
183
- * (len>3, code-review.includes('code')=true) at half weight.
184
- * secret-management's `credential` matches by partial too but
185
- * the union size differs.
186
- * (3) "red green refactor cycle for new features" → tdd @2 vs
187
- * refactor-safely @1 — `refactor` is a full-weight intersection
188
- * with both refactor-safely's `refactor` and `리팩토링` (via
189
- * the refactor canonical), giving 2 hits at 1.0 each. tdd-red-
190
- * green-refactor only matches the literal compound tag
191
- * `red-green-refactor` (one weighted hit) — the full-weight
192
- * generic `refactor` term overpowers the compound-tag specifity.
193
- * (4) "writing unit tests for a function with side effects" → tdd
194
- * @2 vs separation-of-concerns @1 — both solutions have a
195
- * SINGLE matching tag with weighted score 0.5: separation gets
196
- * `function` (COMMON_TAG, exact intersection, weight 0.5);
197
- * tdd-red-green-refactor gets `tests` partial-matching `test`
198
- * (len>3, partial weight 1.0 × 0.5 = 0.5). Both numerators are
199
- * identical. Separation wins because the `function` co-occurs
200
- * in both promptTags and solution.tags, shrinking its Jaccard
201
- * union by one element vs tdd's — a 1-element union-size
202
- * advantage drives the entire ranking. starter-dependency-
203
- * injection is *not* in top-5 despite having `testing`/`mock`/
204
- * `dependency` tags (`tests` does not partial-match `testing`
205
- * — neither is a substring of the other), so listing `di` in
206
- * expectAnyOf is purely defensive recall, not a live candidate.
207
- * T4 BM25 with proper length normalization would attack the
208
- * union-size tie-breaker more rigorously than current Jaccard.
209
- * * paraphrase mrrAt5 stays at 1.0: all 6 added Korean paraphrases
210
- * rank @1 (the originally hard "테스트 먼저 작성하고 리팩토링" is
211
- * documented in the fixture as legitimately matching either tdd
212
- * OR refactor-safely, since starter-refactor-safely's README also
213
- * covers test-first workflows — both are defensible answers).
214
- * * negativeAnyResultRate 0.1 → 0.357: 4 added tricky negatives all
215
- * trigger false positives via single common dev-adjacent words —
216
- * "performance review meeting notes" → caching (matches
217
- * `performance`), "system architecture overview document" →
218
- * separation-of-concerns (matches `architecture`), "database backup
219
- * recovery procedure" → n-plus-one-queries (matches `database`,
220
- * `query`, `데이터베이스`), "validation of insurance claims" →
221
- * error-handling (matches `validation`).
222
- * The original Round 3 plan staged these for T4 (BM25 + IDF). T4 was
223
- * EMPIRICALLY SKIPPED on 2026-04-08 — see
224
- * `docs/plans/2026-04-08-t4-bm25-skip-adr.md` for the full decision
225
- * record. Summary: BM25 prototypes (naive, hybrid Jaccard×IDF,
226
- * precision filter, soft penalty) all matched or underperformed the
227
- * current scorer on every metric. The starter corpus (N=15) is too
228
- * small for IDF to be informative, and the false positives are
229
- * semantic ("performance" is both a dev tag and an English noun) — not
230
- * statistical, so no frequency-based weighting can fix them. The real
231
- * follow-up candidates are tokenizer fix for compound tags, an n-gram
232
- * phrase matcher, and corpus growth — all deferred to Round 4 per the
233
- * ADR.
234
- *
235
- * - v3 (2026-04-08, fixture v2 + R4-T1 compound-tag fix): 1.0 / 0.986 / 0.0 / 0.357
236
- * R4-T1 added `expandCompoundTags` (solution-side) and
237
- * `expandQueryBigrams` (query-side) so hyphenated solution tags like
238
- * `api-key`, `code-review`, `red-green-refactor` participate in direct
239
- * intersection rather than relying on the half-weight partialMatches
240
- * fallback. positive `mrrAt5` improved 0.959 → 0.981 (+0.022). 2 of
241
- * the 4 v2 hard positive cases were resolved (`managing api keys and
242
- * credentials safely` and `red green refactor cycle for new features`
243
- * now rank @1). The remaining 2 (`avoiding hardcoded credentials …`
244
- * and `writing unit tests for a function with side effects`) require
245
- * R4-T2 (phrase matcher) or R4-T3 (specificity classifier) — they're
246
- * about query-side English semantics, not compound-tag tokenization.
247
- * `negativeAnyResultRate` is unchanged at 0.357 because R4-T1 is a
248
- * ranking-quality fix, not a false-positive filter.
249
- *
250
- * - v4 (2026-04-08, fixture v2 + R4-T1 + R4-T2 phrase blocklist):
251
- * 1.0 / 0.986 / 0.0 / 0.143
252
- * R4-T2 added `phrase-blocklist.ts` with 17 curated 2-word English
253
- * non-dev compounds ("performance review", "system architecture",
254
- * "database backup", etc.) and a `maskBlockedTokens` step at the
255
- * top of `rankCandidates` and `searchSolutions`. When a query
256
- * contains a blocked phrase, the constituent tokens are removed
257
- * from the prompt tag list before bigram expansion / canonical
258
- * normalization runs — so the false-positive evidence is removed
259
- * at the source rather than demoted in scoring.
260
- *
261
- * `negativeAnyResultRate` dropped 0.357 → 0.143 (3 of 5 v2 trigger
262
- * negatives fully blocked):
263
- * * "performance review meeting notes" — blocked via
264
- * `performance review` + `meeting notes`
265
- * * "system architecture overview document" — blocked via
266
- * `system architecture` + `overview document`
267
- * * "solar system planets astronomy" — blocked via `solar system`
268
- *
269
- * 2 false positives remain (both deferred to R4-T3 query-side
270
- * specificity classifier — the residuals share a common shape:
271
- * a single dev-tag homograph survives whatever masking is applied,
272
- * and the term-normalizer expansion still surfaces a false match):
273
- *
274
- * * "database backup recovery procedure" → error-handling-patterns:
275
- * `database backup` is blocked, but the residual tokens
276
- * {`recovery`, `procedure`} survive. `recovery` is in the
277
- * `handling` canonical's matchTerms (intentional, for legitimate
278
- * "error recovery handler" queries), so the masked query still
279
- * hits `starter-error-handling-patterns` via the handling
280
- * family. A 3-word `recovery procedure` blocklist entry was
281
- * considered and rejected — it would silently mask legitimate
282
- * dev SRE queries like "disaster recovery procedure" or
283
- * "rollback recovery procedure" without a fixture-driven
284
- * signal. The right fix is at the query-specificity layer
285
- * (R4-T3): require ≥ 2 distinct dev-context signals before any
286
- * match is returned, not at the phrase-blocklist layer.
287
- *
288
- * * "validation of insurance claims" → error-handling-patterns:
289
- * `insurance claim` is blocked, but the residual `validation`
290
- * token IS a legitimate dev tag (input-validation,
291
- * error-handling-patterns both have it). Same R4-T3 target.
292
- *
293
- * positive/paraphrase mrrAt5 are unchanged from v3 because no
294
- * legitimate dev query in the fixture contains a blocked phrase.
295
- *
296
- * - v5 (2026-04-08, fixture v2 + R4-T1 + R4-T2 + R4-T3 specificity guards):
297
- * 1.0 / 0.986 / 0.0 / 0.000
298
- * R4-T3 added two narrow precision rules at the ORCHESTRATION LAYER —
299
- * NOT inside `calculateRelevance` (which remains a pure scoring
300
- * function for test symmetry). The rules are implemented as the
301
- * exported helper `shouldRejectByR4T3Rules(promptTags, matchedTags)`
302
- * and called from both `rankCandidates` (hook path) and
303
- * `searchSolutions` (MCP path) right after the per-solution
304
- * `calculateRelevance` call:
305
- * (Rule A) single-token query AND single-tag match → reject;
306
- * (Rule B) single-tag match with no literal hit in the prompt
307
- * (verbatim match, or substring partial length > 3, or
308
- * shared prefix ≥ 4 for morphological stems) → reject.
309
- * Both rules are scoped narrowly enough to fix exactly the 2 R4-T2
310
- * residuals without recall regression — every fixture positive and
311
- * paraphrase still ranks identically:
312
- * * "validation of insurance claims" → masked to `[validation]`
313
- * (length 1) with single-tag match `validation` → Rule A reject.
314
- * * "database backup recovery procedure" → masked to
315
- * `[recovery, procedure]` with single-tag match `handling`
316
- * (zero literal hit; `handling` is reached via the `recovery`
317
- * canonical-family expansion in term-normalizer) → Rule B reject.
318
- * `negativeAnyResultRate` is now 0.000 — every fixture v2 negative
319
- * produces zero candidates. positive/paraphrase metrics unchanged
320
- * from v4 because no fixture positive matches the (single-token AND
321
- * single-tag) or (all-expansion AND single-tag) shape.
322
- *
323
- * Escape hatch: identifier-boost evidence (hook path) or name-match
324
- * evidence (MCP path) BYPASSES the R4-T3 rules. A candidate with
325
- * even a single weak tag match plus an identifier hit still
326
- * surfaces — the precision rules only fire when the candidate's
327
- * entire evidence pool is a single ambiguous tag.
328
- *
329
- * Defensive precision note: Rule B's "shared prefix ≥ 4"
330
- * morphological check is currently NOT fixture-driven (no fixture
331
- * query masks down to the `caching/cache`-style morphological gap).
332
- * It exists as a pre-emptive fix against silently rejecting
333
- * legitimate future queries where the term-normalizer synonym
334
- * expansion is the only bridge between the query token and the
335
- * solution tag. If a production query surfaces a case the prefix
336
- * check misses, extend it (e.g. by lowering the threshold or
337
- * adding a Levenshtein-1 check) rather than removing it.
338
- *
339
- * Known matcher quirks (separate from the T4 BM25 investigation):
340
- * - `term-normalizer.ts` `error` canonical contains `debug` as a matchTerm
341
- * (intentional for `bug → error` recall), which causes any prompt
342
- * containing `error` to expand to `debug` and over-rank
343
- * `starter-debugging-systematic` on otherwise unrelated queries. This
344
- * is why `async await error propagation` could not be added as a hard
345
- * case — the matcher returns debugging-systematic at #1, which is
346
- * defensible-but-noisy. The fix is at the normalizer level (split
347
- * `debug` out of the `error` family or remove the `error → debug`
348
- * edge entirely) and is queued as a Round 4 follow-up. T4 BM25 was
349
- * considered as a partial mitigation but the T4 skip ADR (referenced
350
- * in the Round 3 outcome paragraph above) shows it does not help.
351
- *
352
- * Long-tail caveat:
353
- * - `"trying to handle authentication errors gracefully when our backend
354
- * api returns inconsistent response formats from different
355
- * microservices"` is a 17-word query intentionally added to exercise
356
- * long-tail behaviour. Currently PASS@1. Originally flagged as BM25
357
- * length-normalization sensitive, but since T4 BM25 was skipped this
358
- * caveat is now informational only — no length-norm code path is
359
- * planned in Round 3.
360
- *
361
- * If a PR legitimately improves a metric, update this constant in the same
362
- * commit so future PRs guard against the new floor.
363
- */
364
- export declare const ROUND3_BASELINE: EvalResult;
365
- /** Maximum allowed absolute regression per metric. 5% is tight enough to catch
366
- * ~3-4 query regressions in a 69-query combined bucket (positive+paraphrase)
367
- * but lenient enough that a single fixture edit won't spuriously fail the
368
- * guard. */
369
- export declare const BASELINE_TOLERANCE = 0.05;
370
- /**
371
- * Test/diagnostic helper: evaluate one query against a fixture solution set
372
- * and return the top-5 ranked candidates with their relevance + matched tags.
373
- *
374
- * Exists so per-query regression tests (e.g. the R4-T1 hard-positive guards
375
- * in `tests/solution-matcher-eval.test.ts`) can assert specific ranking
376
- * outcomes without scraping aggregate metrics. Wraps `rankCandidates` so
377
- * the test path stays in sync with the production ranker.
378
- *
379
- * Returns the same shape as `rankCandidates` minus the generic carrier:
380
- * `{name, relevance, matchedTags}`. Use the names to assert "expected
381
- * solution at rank 1".
382
- */
383
- export declare function evaluateQuery(query: string, solutions: readonly EvalSolution[]): Array<{
384
- name: string;
385
- relevance: number;
386
- matchedTags: string[];
387
- }>;
388
- /**
389
- * Evaluate the current matcher against a labeled fixture and return IR
390
- * metrics. This is the Round 3 baseline — each downstream PR (T2/T3/T4) must
391
- * not regress any of the thresholds asserted in `solution-matcher-eval.test.ts`.
392
- *
393
- * Uses `rankCandidates` (shared with `matchSolutions`) so the evaluator can't
394
- * silently drift from production ranking behaviour.
395
- *
396
- * Metrics are reported both aggregated (positive ∪ paraphrase) and per-bucket,
397
- * so paraphrase-only regressions surface in `byBucket.paraphrase` even if the
398
- * aggregate looks fine.
399
- */
400
- export declare function evaluateSolutionMatcher(fixture: EvalFixture): EvalResult;
401
42
  export declare function matchSolutions(prompt: string, scope: ScopeInfo, cwd: string): SolutionMatch[];