@wooojin/forgen 0.4.10 → 0.4.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/CHANGELOG.md +62 -0
- package/README.md +33 -1
- package/assets/claude/agents/forgen-verify.md +65 -0
- package/assets/claude/workflows/compound-extract.js +136 -0
- package/assets/claude/workflows/evidence-gate-audit.js +107 -0
- package/assets/shared/hook-registry.json +1 -0
- package/dist/checks/_shared/meta-guard-dispatch.d.ts +38 -0
- package/dist/checks/_shared/meta-guard-dispatch.js +80 -0
- package/dist/checks/_shared/text-sanitizer.js +15 -0
- package/dist/cli.js +57 -2
- package/dist/core/changelog-cli.d.ts +7 -0
- package/dist/core/changelog-cli.js +100 -0
- package/dist/core/doctor.d.ts +3 -0
- package/dist/core/doctor.js +38 -0
- package/dist/core/effort-advisory.d.ts +23 -0
- package/dist/core/effort-advisory.js +29 -0
- package/dist/core/explain-cli.d.ts +6 -0
- package/dist/core/explain-cli.js +99 -0
- package/dist/core/health-cli.d.ts +23 -0
- package/dist/core/health-cli.js +86 -0
- package/dist/core/probe-workflow-cli.d.ts +72 -0
- package/dist/core/probe-workflow-cli.js +282 -0
- package/dist/core/spawn.d.ts +13 -0
- package/dist/core/spawn.js +36 -8
- package/dist/core/stats-cli.d.ts +22 -9
- package/dist/core/stats-cli.js +149 -0
- package/dist/core/watch-cli.d.ts +7 -0
- package/dist/core/watch-cli.js +185 -0
- package/dist/core/workflows-cli.d.ts +26 -0
- package/dist/core/workflows-cli.js +120 -0
- package/dist/engine/compound-export.d.ts +12 -0
- package/dist/engine/compound-export.js +136 -14
- package/dist/engine/compound-extractor.d.ts +12 -43
- package/dist/engine/compound-extractor.js +27 -756
- package/dist/engine/extraction-diff.d.ts +11 -0
- package/dist/engine/extraction-diff.js +105 -0
- package/dist/engine/extraction-gates.d.ts +37 -0
- package/dist/engine/extraction-gates.js +100 -0
- package/dist/engine/extraction-git.d.ts +20 -0
- package/dist/engine/extraction-git.js +75 -0
- package/dist/engine/extraction-persistence.d.ts +27 -0
- package/dist/engine/extraction-persistence.js +140 -0
- package/dist/engine/extraction-session.d.ts +26 -0
- package/dist/engine/extraction-session.js +230 -0
- package/dist/engine/lifecycle/types.d.ts +1 -1
- package/dist/engine/meta-learning/matcher-weight-loader.d.ts +16 -0
- package/dist/engine/meta-learning/matcher-weight-loader.js +45 -0
- package/dist/engine/precision-guards.d.ts +14 -0
- package/dist/engine/precision-guards.js +39 -0
- package/dist/engine/ranking-pipeline.d.ts +45 -0
- package/dist/engine/ranking-pipeline.js +66 -0
- package/dist/engine/relevance-scorer.d.ts +43 -0
- package/dist/engine/relevance-scorer.js +81 -0
- package/dist/engine/scoring-algorithms.d.ts +31 -0
- package/dist/engine/scoring-algorithms.js +109 -0
- package/dist/engine/solution-matcher-eval.d.ts +97 -0
- package/dist/engine/solution-matcher-eval.js +122 -0
- package/dist/engine/solution-matcher.d.ts +21 -380
- package/dist/engine/solution-matcher.js +27 -828
- package/dist/fgx.js +1 -1
- package/dist/hooks/notepad-injector.js +7 -0
- package/dist/hooks/post-tool-use.js +8 -1
- package/dist/hooks/secret-filter.d.ts +1 -0
- package/dist/hooks/secret-filter.js +17 -7
- package/dist/hooks/shared/preflight-check.d.ts +15 -0
- package/dist/hooks/shared/preflight-check.js +51 -0
- package/dist/hooks/stop-guard.js +19 -60
- package/dist/hooks/subagent-stop-guard.d.ts +23 -0
- package/dist/hooks/subagent-stop-guard.js +158 -0
- package/dist/hooks/subagent-tracker.d.ts +36 -3
- package/dist/hooks/subagent-tracker.js +86 -39
- package/hooks/hooks.json +6 -1
- package/package.json +7 -7
- package/plugin.json +1 -1
- package/scripts/postinstall.js +10 -7
|
@@ -1,34 +1,30 @@
|
|
|
1
|
-
import type { ScopeInfo } from '../core/types.js';
|
|
2
|
-
import type { SolutionStatus, SolutionType } from './solution-format.js';
|
|
3
1
|
/**
|
|
4
|
-
*
|
|
5
|
-
* `./term-normalizer.js` directly. Kept as a thin wrapper for the existing
|
|
6
|
-
* `synonym-tfidf.test.ts` and any external consumers.
|
|
7
|
-
*/
|
|
8
|
-
export declare function expandTagsWithSynonyms(tags: string[]): string[];
|
|
9
|
-
/**
|
|
10
|
-
* Compute the Dice coefficient between two strings using character bigrams.
|
|
11
|
-
*
|
|
12
|
-
* Dice = 2 * |intersection| / (|A| + |B|)
|
|
2
|
+
* Solution matcher — thin facade re-exporting from decomposed modules.
|
|
13
3
|
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* Returns 1.0 for identical non-trivial strings.
|
|
4
|
+
* All public exports are preserved for backward compatibility. Internal
|
|
5
|
+
* callers that import from './solution-matcher.js' continue to work.
|
|
17
6
|
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
7
|
+
* Module layout (post-decomposition):
|
|
8
|
+
* scoring-algorithms.ts — bigramSimilarity, bm25Score, tagWeight, COMMON_TAGS
|
|
9
|
+
* relevance-scorer.ts — calculateRelevance, CalculateRelevanceOptions
|
|
10
|
+
* precision-guards.ts — shouldRejectByR4T3Rules
|
|
11
|
+
* ranking-pipeline.ts — rankCandidates, RankableSolution, RankedCandidate
|
|
12
|
+
* solution-matcher-eval.ts — EvalSolution/Query/Fixture/Result, ROUND3_BASELINE, evaluateQuery, evaluateSolutionMatcher
|
|
13
|
+
* meta-learning/matcher-weight-loader.ts — loadTunedMatcherWeights
|
|
22
14
|
*/
|
|
23
|
-
|
|
15
|
+
import type { ScopeInfo } from '../core/types.js';
|
|
16
|
+
import type { SolutionStatus, SolutionType } from './solution-format.js';
|
|
17
|
+
export { bigramSimilarity, bm25Score, COMMON_TAGS, tagWeight } from './scoring-algorithms.js';
|
|
18
|
+
export { calculateRelevance } from './relevance-scorer.js';
|
|
19
|
+
export type { CalculateRelevanceOptions } from './relevance-scorer.js';
|
|
20
|
+
export { shouldRejectByR4T3Rules } from './precision-guards.js';
|
|
21
|
+
export { ROUND3_BASELINE, BASELINE_TOLERANCE, evaluateQuery, evaluateSolutionMatcher, } from './solution-matcher-eval.js';
|
|
22
|
+
export type { EvalSolution, EvalQuery, EvalFixture, BucketMetrics, EvalResult, } from './solution-matcher-eval.js';
|
|
24
23
|
/**
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
* k1=1.2, b=0.75 (standard BM25 parameters).
|
|
24
|
+
* @deprecated Use `defaultNormalizer.normalizeTerms` from
|
|
25
|
+
* `./term-normalizer.js` directly.
|
|
28
26
|
*/
|
|
29
|
-
export declare function
|
|
30
|
-
/** Apply IDF-like weight: common tags get reduced weight */
|
|
31
|
-
export declare function tagWeight(tag: string): number;
|
|
27
|
+
export declare function expandTagsWithSynonyms(tags: string[]): string[];
|
|
32
28
|
export interface SolutionMatch {
|
|
33
29
|
name: string;
|
|
34
30
|
path: string;
|
|
@@ -41,361 +37,6 @@ export interface SolutionMatch {
|
|
|
41
37
|
tags: string[];
|
|
42
38
|
identifiers: string[];
|
|
43
39
|
matchedTags: string[];
|
|
44
|
-
/**
|
|
45
|
-
* Identifier substrings (function/file names) that appeared literally in the
|
|
46
|
-
* prompt. Added 2026-04-21 so solution-injector can enforce a precision gate
|
|
47
|
-
* distinguishing "user typed a specific identifier" (strong signal, survives
|
|
48
|
-
* 1-tag overlap) from "only 1 tag happens to overlap" (often noise — common
|
|
49
|
-
* nouns like 'type', 'file', 'forgen' trigger rare-tag BM25 boost).
|
|
50
|
-
*/
|
|
51
40
|
matchedIdentifiers: string[];
|
|
52
41
|
}
|
|
53
|
-
/**
|
|
54
|
-
* Optional hints for the v3 `calculateRelevance` path. Used by hot-path
|
|
55
|
-
* callers (matchSolutions, searchSolutions) to avoid re-normalizing the
|
|
56
|
-
* same query tags on every solution.
|
|
57
|
-
*/
|
|
58
|
-
export interface CalculateRelevanceOptions {
|
|
59
|
-
/**
|
|
60
|
-
* Pre-normalized prompt tags (produced by `defaultNormalizer.normalizeTerms`).
|
|
61
|
-
* If provided, skips the per-call expansion. Callers loop-running against
|
|
62
|
-
* many solutions should compute this once outside the loop and pass it in.
|
|
63
|
-
*/
|
|
64
|
-
normalizedPromptTags?: string[];
|
|
65
|
-
/**
|
|
66
|
-
* R4-T1: solution tags expanded with compound-split alternatives
|
|
67
|
-
* (`expandCompoundTags`). When supplied, the intersection/partial-match
|
|
68
|
-
* step uses this set INSTEAD of `solutionTags`, but the Jaccard union
|
|
69
|
-
* denominator still uses `solutionTags` (raw) so the score normalization
|
|
70
|
-
* stays semantically stable. Caller responsibility to pass the matching
|
|
71
|
-
* pair — `solutionTagsExpanded` MUST be a superset of `solutionTags`.
|
|
72
|
-
*/
|
|
73
|
-
solutionTagsExpanded?: string[];
|
|
74
|
-
/** Average document (solution) tag count for BM25 normalization. Defaults to 6. */
|
|
75
|
-
avgDocLength?: number;
|
|
76
|
-
/** Meta-learning: dynamic ensemble weights (sum must equal 1.0). Defaults to {tfidf:0.5, bm25:0.3, bigram:0.2}. */
|
|
77
|
-
ensembleWeights?: {
|
|
78
|
-
tfidf: number;
|
|
79
|
-
bm25: number;
|
|
80
|
-
bigram: number;
|
|
81
|
-
};
|
|
82
|
-
}
|
|
83
|
-
export declare function calculateRelevance(promptTags: string[], solutionTags: string[], confidence: number, options?: CalculateRelevanceOptions): {
|
|
84
|
-
relevance: number;
|
|
85
|
-
matchedTags: string[];
|
|
86
|
-
};
|
|
87
|
-
/** @deprecated */
|
|
88
|
-
export declare function calculateRelevance(prompt: string, keywords: string[]): number;
|
|
89
|
-
export declare function shouldRejectByR4T3Rules(promptTags: readonly string[], matchedTags: readonly string[]): boolean;
|
|
90
|
-
/**
|
|
91
|
-
* In-memory solution shape for the bootstrap evaluator. Mirrors the index
|
|
92
|
-
* entry fields that `matchSolutions` consumes (tags, identifiers, confidence)
|
|
93
|
-
* but without any filesystem dependency — the evaluator is pure so CI can run
|
|
94
|
-
* it without mounting a starter pack.
|
|
95
|
-
*/
|
|
96
|
-
export interface EvalSolution {
|
|
97
|
-
name: string;
|
|
98
|
-
tags: string[];
|
|
99
|
-
identifiers?: string[];
|
|
100
|
-
confidence: number;
|
|
101
|
-
}
|
|
102
|
-
export interface EvalQuery {
|
|
103
|
-
query: string;
|
|
104
|
-
/** Names that should appear in the top-5. Empty array = expect no match (negative case). */
|
|
105
|
-
expectAnyOf: string[];
|
|
106
|
-
}
|
|
107
|
-
export interface EvalFixture {
|
|
108
|
-
solutions: EvalSolution[];
|
|
109
|
-
positive: EvalQuery[];
|
|
110
|
-
/** Bilingual or compound-word variants that exercise synonym expansion. */
|
|
111
|
-
paraphrase: EvalQuery[];
|
|
112
|
-
/** Unrelated queries that should not return a top-1 hit. */
|
|
113
|
-
negative: EvalQuery[];
|
|
114
|
-
}
|
|
115
|
-
/** Per-bucket metrics. Paraphrase and positive are reported separately so a
|
|
116
|
-
* bilingual regression (T2 synonym change) can't hide inside the aggregate. */
|
|
117
|
-
export interface BucketMetrics {
|
|
118
|
-
/** |{q : ∃i≤5, ranked[i] ∈ q.expectAnyOf}| / |q| */
|
|
119
|
-
recallAt5: number;
|
|
120
|
-
/** Σ (1 / firstMatchRank) / |q|; rank > 5 contributes 0. */
|
|
121
|
-
mrrAt5: number;
|
|
122
|
-
/** |{q : ranked is empty}| / |q| */
|
|
123
|
-
noResultRate: number;
|
|
124
|
-
/** Number of queries in this bucket. */
|
|
125
|
-
total: number;
|
|
126
|
-
}
|
|
127
|
-
export interface EvalResult {
|
|
128
|
-
/** Combined (positive ∪ paraphrase) metrics — backwards-compatible headline numbers. */
|
|
129
|
-
recallAt5: number;
|
|
130
|
-
mrrAt5: number;
|
|
131
|
-
noResultRate: number;
|
|
132
|
-
/**
|
|
133
|
-
* Fraction of negative queries where the matcher returned ≥ 1 candidate
|
|
134
|
-
* (regardless of rank). Name is honest: this is the "any result" rate on
|
|
135
|
-
* the negative bucket, not a rank-1 precision metric. It's the correct
|
|
136
|
-
* baseline for "did synonym/stemming leak into unrelated queries?".
|
|
137
|
-
*/
|
|
138
|
-
negativeAnyResultRate: number;
|
|
139
|
-
/** Per-bucket breakdown — use these to catch paraphrase-only regressions. */
|
|
140
|
-
byBucket: {
|
|
141
|
-
positive: BucketMetrics;
|
|
142
|
-
paraphrase: BucketMetrics;
|
|
143
|
-
};
|
|
144
|
-
total: {
|
|
145
|
-
positive: number;
|
|
146
|
-
paraphrase: number;
|
|
147
|
-
negative: number;
|
|
148
|
-
};
|
|
149
|
-
}
|
|
150
|
-
/**
|
|
151
|
-
* Round 3 baseline metrics, recorded against the current `term-normalizer`
|
|
152
|
-
* + `calculateRelevance` + fixture `solution-match-bootstrap.json`. Used as
|
|
153
|
-
* a relative regression guard in `tests/solution-matcher-eval.test.ts` —
|
|
154
|
-
* downstream PRs must not regress any field by more than `BASELINE_TOLERANCE`.
|
|
155
|
-
*
|
|
156
|
-
* History (chronological ascending — v1 at top, latest at bottom):
|
|
157
|
-
* - v1 (2026-04-08, fixture v1, 41+10+10 queries): 1.0 / 1.0 / 0.0 / 0.1
|
|
158
|
-
* Recorded against the original 61-query fixture, all positive queries
|
|
159
|
-
* PASS@1. Indicated a measurement plateau but masked the matcher's true
|
|
160
|
-
* ranking and false-positive weaknesses because the fixture queries were
|
|
161
|
-
* too tag-aligned.
|
|
162
|
-
*
|
|
163
|
-
* - v2 (2026-04-08, fixture v2, 53+16+14 queries): 1.0 / 0.969 / 0.0 / 0.357
|
|
164
|
-
* Expanded with 12 hard positive (multi-canonical / compound-tag tug-of-
|
|
165
|
-
* war), 6 Korean subtle paraphrase, and 4 tricky negative queries. The
|
|
166
|
-
* drops are intentional and represent genuine matcher behaviour:
|
|
167
|
-
* * positive mrrAt5 1.0 → 0.959: 4 of 12 added positives rank #2-3:
|
|
168
|
-
* (1) "managing api keys and credentials safely" → secret @3 vs
|
|
169
|
-
* api-error-responses @1 — the `api` canonical in
|
|
170
|
-
* DEFAULT_MATCH_TERMS expands to {api, rest, graphql, endpoint,
|
|
171
|
-
* route}, so query `api` hits BOTH `api` AND `rest` on
|
|
172
|
-
* starter-api-error-responses (matched=['api','rest']) — a
|
|
173
|
-
* double-count numerator. starter-secret-management only scores
|
|
174
|
-
* a single weak partial match on `credential`. The compound
|
|
175
|
-
* `api-key` tag on secret-management is never reached because
|
|
176
|
-
* extractTags strips the query-side hyphen and yields
|
|
177
|
-
* ['api','keys'] (the solution-side tag remains hyphenated in
|
|
178
|
-
* the index but has no query token to intersect with). T4 IDF
|
|
179
|
-
* would down-weight both `api` and `rest`, neutralising the
|
|
180
|
-
* double-count and letting `credential` outscore the noise.
|
|
181
|
-
* (2) "avoiding hardcoded credentials in source code" → secret @2
|
|
182
|
-
* vs code-review @1 — `code` partial-matches `code-review`
|
|
183
|
-
* (len>3, code-review.includes('code')=true) at half weight.
|
|
184
|
-
* secret-management's `credential` matches by partial too but
|
|
185
|
-
* the union size differs.
|
|
186
|
-
* (3) "red green refactor cycle for new features" → tdd @2 vs
|
|
187
|
-
* refactor-safely @1 — `refactor` is a full-weight intersection
|
|
188
|
-
* with both refactor-safely's `refactor` and `리팩토링` (via
|
|
189
|
-
* the refactor canonical), giving 2 hits at 1.0 each. tdd-red-
|
|
190
|
-
* green-refactor only matches the literal compound tag
|
|
191
|
-
* `red-green-refactor` (one weighted hit) — the full-weight
|
|
192
|
-
* generic `refactor` term overpowers the compound-tag specifity.
|
|
193
|
-
* (4) "writing unit tests for a function with side effects" → tdd
|
|
194
|
-
* @2 vs separation-of-concerns @1 — both solutions have a
|
|
195
|
-
* SINGLE matching tag with weighted score 0.5: separation gets
|
|
196
|
-
* `function` (COMMON_TAG, exact intersection, weight 0.5);
|
|
197
|
-
* tdd-red-green-refactor gets `tests` partial-matching `test`
|
|
198
|
-
* (len>3, partial weight 1.0 × 0.5 = 0.5). Both numerators are
|
|
199
|
-
* identical. Separation wins because the `function` co-occurs
|
|
200
|
-
* in both promptTags and solution.tags, shrinking its Jaccard
|
|
201
|
-
* union by one element vs tdd's — a 1-element union-size
|
|
202
|
-
* advantage drives the entire ranking. starter-dependency-
|
|
203
|
-
* injection is *not* in top-5 despite having `testing`/`mock`/
|
|
204
|
-
* `dependency` tags (`tests` does not partial-match `testing`
|
|
205
|
-
* — neither is a substring of the other), so listing `di` in
|
|
206
|
-
* expectAnyOf is purely defensive recall, not a live candidate.
|
|
207
|
-
* T4 BM25 with proper length normalization would attack the
|
|
208
|
-
* union-size tie-breaker more rigorously than current Jaccard.
|
|
209
|
-
* * paraphrase mrrAt5 stays at 1.0: all 6 added Korean paraphrases
|
|
210
|
-
* rank @1 (the originally hard "테스트 먼저 작성하고 리팩토링" is
|
|
211
|
-
* documented in the fixture as legitimately matching either tdd
|
|
212
|
-
* OR refactor-safely, since starter-refactor-safely's README also
|
|
213
|
-
* covers test-first workflows — both are defensible answers).
|
|
214
|
-
* * negativeAnyResultRate 0.1 → 0.357: 4 added tricky negatives all
|
|
215
|
-
* trigger false positives via single common dev-adjacent words —
|
|
216
|
-
* "performance review meeting notes" → caching (matches
|
|
217
|
-
* `performance`), "system architecture overview document" →
|
|
218
|
-
* separation-of-concerns (matches `architecture`), "database backup
|
|
219
|
-
* recovery procedure" → n-plus-one-queries (matches `database`,
|
|
220
|
-
* `query`, `데이터베이스`), "validation of insurance claims" →
|
|
221
|
-
* error-handling (matches `validation`).
|
|
222
|
-
* The original Round 3 plan staged these for T4 (BM25 + IDF). T4 was
|
|
223
|
-
* EMPIRICALLY SKIPPED on 2026-04-08 — see
|
|
224
|
-
* `docs/plans/2026-04-08-t4-bm25-skip-adr.md` for the full decision
|
|
225
|
-
* record. Summary: BM25 prototypes (naive, hybrid Jaccard×IDF,
|
|
226
|
-
* precision filter, soft penalty) all matched or underperformed the
|
|
227
|
-
* current scorer on every metric. The starter corpus (N=15) is too
|
|
228
|
-
* small for IDF to be informative, and the false positives are
|
|
229
|
-
* semantic ("performance" is both a dev tag and an English noun) — not
|
|
230
|
-
* statistical, so no frequency-based weighting can fix them. The real
|
|
231
|
-
* follow-up candidates are tokenizer fix for compound tags, an n-gram
|
|
232
|
-
* phrase matcher, and corpus growth — all deferred to Round 4 per the
|
|
233
|
-
* ADR.
|
|
234
|
-
*
|
|
235
|
-
* - v3 (2026-04-08, fixture v2 + R4-T1 compound-tag fix): 1.0 / 0.986 / 0.0 / 0.357
|
|
236
|
-
* R4-T1 added `expandCompoundTags` (solution-side) and
|
|
237
|
-
* `expandQueryBigrams` (query-side) so hyphenated solution tags like
|
|
238
|
-
* `api-key`, `code-review`, `red-green-refactor` participate in direct
|
|
239
|
-
* intersection rather than relying on the half-weight partialMatches
|
|
240
|
-
* fallback. positive `mrrAt5` improved 0.959 → 0.981 (+0.022). 2 of
|
|
241
|
-
* the 4 v2 hard positive cases were resolved (`managing api keys and
|
|
242
|
-
* credentials safely` and `red green refactor cycle for new features`
|
|
243
|
-
* now rank @1). The remaining 2 (`avoiding hardcoded credentials …`
|
|
244
|
-
* and `writing unit tests for a function with side effects`) require
|
|
245
|
-
* R4-T2 (phrase matcher) or R4-T3 (specificity classifier) — they're
|
|
246
|
-
* about query-side English semantics, not compound-tag tokenization.
|
|
247
|
-
* `negativeAnyResultRate` is unchanged at 0.357 because R4-T1 is a
|
|
248
|
-
* ranking-quality fix, not a false-positive filter.
|
|
249
|
-
*
|
|
250
|
-
* - v4 (2026-04-08, fixture v2 + R4-T1 + R4-T2 phrase blocklist):
|
|
251
|
-
* 1.0 / 0.986 / 0.0 / 0.143
|
|
252
|
-
* R4-T2 added `phrase-blocklist.ts` with 17 curated 2-word English
|
|
253
|
-
* non-dev compounds ("performance review", "system architecture",
|
|
254
|
-
* "database backup", etc.) and a `maskBlockedTokens` step at the
|
|
255
|
-
* top of `rankCandidates` and `searchSolutions`. When a query
|
|
256
|
-
* contains a blocked phrase, the constituent tokens are removed
|
|
257
|
-
* from the prompt tag list before bigram expansion / canonical
|
|
258
|
-
* normalization runs — so the false-positive evidence is removed
|
|
259
|
-
* at the source rather than demoted in scoring.
|
|
260
|
-
*
|
|
261
|
-
* `negativeAnyResultRate` dropped 0.357 → 0.143 (3 of 5 v2 trigger
|
|
262
|
-
* negatives fully blocked):
|
|
263
|
-
* * "performance review meeting notes" — blocked via
|
|
264
|
-
* `performance review` + `meeting notes`
|
|
265
|
-
* * "system architecture overview document" — blocked via
|
|
266
|
-
* `system architecture` + `overview document`
|
|
267
|
-
* * "solar system planets astronomy" — blocked via `solar system`
|
|
268
|
-
*
|
|
269
|
-
* 2 false positives remain (both deferred to R4-T3 query-side
|
|
270
|
-
* specificity classifier — the residuals share a common shape:
|
|
271
|
-
* a single dev-tag homograph survives whatever masking is applied,
|
|
272
|
-
* and the term-normalizer expansion still surfaces a false match):
|
|
273
|
-
*
|
|
274
|
-
* * "database backup recovery procedure" → error-handling-patterns:
|
|
275
|
-
* `database backup` is blocked, but the residual tokens
|
|
276
|
-
* {`recovery`, `procedure`} survive. `recovery` is in the
|
|
277
|
-
* `handling` canonical's matchTerms (intentional, for legitimate
|
|
278
|
-
* "error recovery handler" queries), so the masked query still
|
|
279
|
-
* hits `starter-error-handling-patterns` via the handling
|
|
280
|
-
* family. A 3-word `recovery procedure` blocklist entry was
|
|
281
|
-
* considered and rejected — it would silently mask legitimate
|
|
282
|
-
* dev SRE queries like "disaster recovery procedure" or
|
|
283
|
-
* "rollback recovery procedure" without a fixture-driven
|
|
284
|
-
* signal. The right fix is at the query-specificity layer
|
|
285
|
-
* (R4-T3): require ≥ 2 distinct dev-context signals before any
|
|
286
|
-
* match is returned, not at the phrase-blocklist layer.
|
|
287
|
-
*
|
|
288
|
-
* * "validation of insurance claims" → error-handling-patterns:
|
|
289
|
-
* `insurance claim` is blocked, but the residual `validation`
|
|
290
|
-
* token IS a legitimate dev tag (input-validation,
|
|
291
|
-
* error-handling-patterns both have it). Same R4-T3 target.
|
|
292
|
-
*
|
|
293
|
-
* positive/paraphrase mrrAt5 are unchanged from v3 because no
|
|
294
|
-
* legitimate dev query in the fixture contains a blocked phrase.
|
|
295
|
-
*
|
|
296
|
-
* - v5 (2026-04-08, fixture v2 + R4-T1 + R4-T2 + R4-T3 specificity guards):
|
|
297
|
-
* 1.0 / 0.986 / 0.0 / 0.000
|
|
298
|
-
* R4-T3 added two narrow precision rules at the ORCHESTRATION LAYER —
|
|
299
|
-
* NOT inside `calculateRelevance` (which remains a pure scoring
|
|
300
|
-
* function for test symmetry). The rules are implemented as the
|
|
301
|
-
* exported helper `shouldRejectByR4T3Rules(promptTags, matchedTags)`
|
|
302
|
-
* and called from both `rankCandidates` (hook path) and
|
|
303
|
-
* `searchSolutions` (MCP path) right after the per-solution
|
|
304
|
-
* `calculateRelevance` call:
|
|
305
|
-
* (Rule A) single-token query AND single-tag match → reject;
|
|
306
|
-
* (Rule B) single-tag match with no literal hit in the prompt
|
|
307
|
-
* (verbatim match, or substring partial length > 3, or
|
|
308
|
-
* shared prefix ≥ 4 for morphological stems) → reject.
|
|
309
|
-
* Both rules are scoped narrowly enough to fix exactly the 2 R4-T2
|
|
310
|
-
* residuals without recall regression — every fixture positive and
|
|
311
|
-
* paraphrase still ranks identically:
|
|
312
|
-
* * "validation of insurance claims" → masked to `[validation]`
|
|
313
|
-
* (length 1) with single-tag match `validation` → Rule A reject.
|
|
314
|
-
* * "database backup recovery procedure" → masked to
|
|
315
|
-
* `[recovery, procedure]` with single-tag match `handling`
|
|
316
|
-
* (zero literal hit; `handling` is reached via the `recovery`
|
|
317
|
-
* canonical-family expansion in term-normalizer) → Rule B reject.
|
|
318
|
-
* `negativeAnyResultRate` is now 0.000 — every fixture v2 negative
|
|
319
|
-
* produces zero candidates. positive/paraphrase metrics unchanged
|
|
320
|
-
* from v4 because no fixture positive matches the (single-token AND
|
|
321
|
-
* single-tag) or (all-expansion AND single-tag) shape.
|
|
322
|
-
*
|
|
323
|
-
* Escape hatch: identifier-boost evidence (hook path) or name-match
|
|
324
|
-
* evidence (MCP path) BYPASSES the R4-T3 rules. A candidate with
|
|
325
|
-
* even a single weak tag match plus an identifier hit still
|
|
326
|
-
* surfaces — the precision rules only fire when the candidate's
|
|
327
|
-
* entire evidence pool is a single ambiguous tag.
|
|
328
|
-
*
|
|
329
|
-
* Defensive precision note: Rule B's "shared prefix ≥ 4"
|
|
330
|
-
* morphological check is currently NOT fixture-driven (no fixture
|
|
331
|
-
* query masks down to the `caching/cache`-style morphological gap).
|
|
332
|
-
* It exists as a pre-emptive fix against silently rejecting
|
|
333
|
-
* legitimate future queries where the term-normalizer synonym
|
|
334
|
-
* expansion is the only bridge between the query token and the
|
|
335
|
-
* solution tag. If a production query surfaces a case the prefix
|
|
336
|
-
* check misses, extend it (e.g. by lowering the threshold or
|
|
337
|
-
* adding a Levenshtein-1 check) rather than removing it.
|
|
338
|
-
*
|
|
339
|
-
* Known matcher quirks (separate from the T4 BM25 investigation):
|
|
340
|
-
* - `term-normalizer.ts` `error` canonical contains `debug` as a matchTerm
|
|
341
|
-
* (intentional for `bug → error` recall), which causes any prompt
|
|
342
|
-
* containing `error` to expand to `debug` and over-rank
|
|
343
|
-
* `starter-debugging-systematic` on otherwise unrelated queries. This
|
|
344
|
-
* is why `async await error propagation` could not be added as a hard
|
|
345
|
-
* case — the matcher returns debugging-systematic at #1, which is
|
|
346
|
-
* defensible-but-noisy. The fix is at the normalizer level (split
|
|
347
|
-
* `debug` out of the `error` family or remove the `error → debug`
|
|
348
|
-
* edge entirely) and is queued as a Round 4 follow-up. T4 BM25 was
|
|
349
|
-
* considered as a partial mitigation but the T4 skip ADR (referenced
|
|
350
|
-
* in the Round 3 outcome paragraph above) shows it does not help.
|
|
351
|
-
*
|
|
352
|
-
* Long-tail caveat:
|
|
353
|
-
* - `"trying to handle authentication errors gracefully when our backend
|
|
354
|
-
* api returns inconsistent response formats from different
|
|
355
|
-
* microservices"` is a 17-word query intentionally added to exercise
|
|
356
|
-
* long-tail behaviour. Currently PASS@1. Originally flagged as BM25
|
|
357
|
-
* length-normalization sensitive, but since T4 BM25 was skipped this
|
|
358
|
-
* caveat is now informational only — no length-norm code path is
|
|
359
|
-
* planned in Round 3.
|
|
360
|
-
*
|
|
361
|
-
* If a PR legitimately improves a metric, update this constant in the same
|
|
362
|
-
* commit so future PRs guard against the new floor.
|
|
363
|
-
*/
|
|
364
|
-
export declare const ROUND3_BASELINE: EvalResult;
|
|
365
|
-
/** Maximum allowed absolute regression per metric. 5% is tight enough to catch
|
|
366
|
-
* ~3-4 query regressions in a 69-query combined bucket (positive+paraphrase)
|
|
367
|
-
* but lenient enough that a single fixture edit won't spuriously fail the
|
|
368
|
-
* guard. */
|
|
369
|
-
export declare const BASELINE_TOLERANCE = 0.05;
|
|
370
|
-
/**
|
|
371
|
-
* Test/diagnostic helper: evaluate one query against a fixture solution set
|
|
372
|
-
* and return the top-5 ranked candidates with their relevance + matched tags.
|
|
373
|
-
*
|
|
374
|
-
* Exists so per-query regression tests (e.g. the R4-T1 hard-positive guards
|
|
375
|
-
* in `tests/solution-matcher-eval.test.ts`) can assert specific ranking
|
|
376
|
-
* outcomes without scraping aggregate metrics. Wraps `rankCandidates` so
|
|
377
|
-
* the test path stays in sync with the production ranker.
|
|
378
|
-
*
|
|
379
|
-
* Returns the same shape as `rankCandidates` minus the generic carrier:
|
|
380
|
-
* `{name, relevance, matchedTags}`. Use the names to assert "expected
|
|
381
|
-
* solution at rank 1".
|
|
382
|
-
*/
|
|
383
|
-
export declare function evaluateQuery(query: string, solutions: readonly EvalSolution[]): Array<{
|
|
384
|
-
name: string;
|
|
385
|
-
relevance: number;
|
|
386
|
-
matchedTags: string[];
|
|
387
|
-
}>;
|
|
388
|
-
/**
|
|
389
|
-
* Evaluate the current matcher against a labeled fixture and return IR
|
|
390
|
-
* metrics. This is the Round 3 baseline — each downstream PR (T2/T3/T4) must
|
|
391
|
-
* not regress any of the thresholds asserted in `solution-matcher-eval.test.ts`.
|
|
392
|
-
*
|
|
393
|
-
* Uses `rankCandidates` (shared with `matchSolutions`) so the evaluator can't
|
|
394
|
-
* silently drift from production ranking behaviour.
|
|
395
|
-
*
|
|
396
|
-
* Metrics are reported both aggregated (positive ∪ paraphrase) and per-bucket,
|
|
397
|
-
* so paraphrase-only regressions surface in `byBucket.paraphrase` even if the
|
|
398
|
-
* aggregate looks fine.
|
|
399
|
-
*/
|
|
400
|
-
export declare function evaluateSolutionMatcher(fixture: EvalFixture): EvalResult;
|
|
401
42
|
export declare function matchSolutions(prompt: string, scope: ScopeInfo, cwd: string): SolutionMatch[];
|