cans-spec 0.2.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,575 @@
1
+ /** Pure issue-aggregation core for `cans check` (issue #41).
2
+ *
3
+ * Turns the raw issue list into pattern groups: one group per root cause,
4
+ * compact per-file locations (`action:30,111,143`), count prefixes for the
5
+ * human printer (`61× <min children (2/3)`), ranked sub-items (missing
6
+ * targets, keywords, overlap pairs, referrers), a fix hint once per group,
7
+ * and the lossless `--json` wire shape.
8
+ *
9
+ * FULLY STANDALONE: zero imports — not even from src/types.ts. IssueLike is
10
+ * structurally compatible with src/types.ts Issue (which gains an optional
11
+ * machine-readable `rule` at integration). Every known engine message format
12
+ * has a normalizer, so the module behaves identically with and without
13
+ * `rule` annotations — the rule path and the fallback path MUST agree
14
+ * (test/report.test.ts pins this on a full mixed scenario).
15
+ *
16
+ * Design decisions pinned here (integration agents read this):
17
+ * - Group key = `rule + '|' + (key ?? pattern)`. Threshold-family rules
18
+ * (siblings/depth/node-length/node_chars/tbd.max/collapse/parse.indent)
19
+ * encode the observed ratio IN the pattern, so `<min children (2/3)` and
20
+ * `<min children (1/3)` are distinct patterns; identity-family rules
21
+ * (missing file, stale target, keyword, overlap class, duplicate-home
22
+ * concept, prefix word, malformed dir) use a stable pattern plus a `key`
23
+ * carrying the root-cause identity.
24
+ * - Group ordering: section (structure → style → refs → redundancy →
25
+ * overflow → parse → content → io → other), then count desc, then first
26
+ * occurrence in the input. Nothing is dropped: every input issue raises
27
+ * its group's count exactly once.
28
+ * - `category` on a group is the SECTION (rule prefix wins over the engine
29
+ * category — e.g. content.tbd.* lands under `content`, refs.chaining
30
+ * under `refs` even though the overflow engine emits it).
31
+ * - `level` is the first member's level; sections count errors/warnings
32
+ * from the raw issues, so a hypothetically mixed group never skews them.
33
+ * - topGroups: `folded` is the number of issue OCCURRENCES hidden in the
34
+ * tail groups (the "…and K more" line), not the group count.
35
+ * - Engine message formats with no id in the rule vocabulary (transient
36
+ * refs, _collab/ refs, possible typos) fall through to `<category>.other`
37
+ * (typos get the local extension `redundancy.typo` so they still group).
38
+ */
39
+
40
+ // issue #41: contract types — the human printer and the JSON emitter build on these.
41
+ export interface IssueLike {
42
+ file: string;
43
+ line: number;
44
+ level: 'error' | 'warning';
45
+ category: string;
46
+ message: string;
47
+ suggestion?: string;
48
+ rule?: string;
49
+ }
50
+
51
+ export interface GroupItem {
52
+ label: string;
53
+ count: number;
54
+ locations?: string[];
55
+ /** issue #41 integration: semantic size when the message carries one —
56
+ * keyword sprawl's node count (`"artifacts" × 105 nodes` → 105). Ranking
57
+ * and display use the metric; `count` stays the raw occurrence count. */
58
+ metric?: number;
59
+ }
60
+
61
+ export interface IssueGroup {
62
+ /** Section: structure|style|refs|redundancy|overflow|parse|content|io|other */
63
+ category: string;
64
+ /** Normalized rule id (issue.rule when present, else derived from the message). */
65
+ rule: string;
66
+ /** One-line pattern label, e.g. `<min children (2/3)` or `missing file`. */
67
+ pattern: string;
68
+ /** Occurrences. */
69
+ count: number;
70
+ level: 'error' | 'warning';
71
+ /** Compact per-file strings `action:30,111,143` (dedup, stable order). */
72
+ locations: string[];
73
+ /** Metric-ranked sub-items (keywords, missing targets, overlap pairs, …). */
74
+ items?: GroupItem[];
75
+ /** Representative detail, e.g. `budget.md:148 → interface.md#Refusals`. */
76
+ detail?: string;
77
+ /** Fix hint — once per group (first non-empty among members). */
78
+ suggestion?: string;
79
+ /** Extracted root-cause key (missing target path, keyword, pair, …). */
80
+ key?: string;
81
+ }
82
+
83
+ export interface SectionReport {
84
+ name: string;
85
+ errorCount: number;
86
+ warningCount: number;
87
+ groups: IssueGroup[];
88
+ }
89
+
90
+ export interface CheckReport {
91
+ /** Every group, ordered by section → count desc → first occurrence. */
92
+ groups: IssueGroup[];
93
+ sections: Record<string, SectionReport>;
94
+ }
95
+
96
+ /** Structural subset of CheckResult needed for the --json wire shape. */
97
+ export interface CheckResultLike {
98
+ ok: boolean;
99
+ exitCode: number;
100
+ files: number;
101
+ nodes: number;
102
+ maxDepth: number;
103
+ elapsedMs?: number;
104
+ refs: { total: number; broken: number; deepHops: number };
105
+ backPointers: { total: number; current: number; stale: number };
106
+ errorCount: number;
107
+ warningCount: number;
108
+ backPointersUpdated: number;
109
+ /** Issue #11: sorted spec-relative paths of the files --fix actually
110
+ * rewrote (empty without --fix or when nothing needed a write). */
111
+ backPointersUpdatedFiles: string[];
112
+ rulesSummary?: string;
113
+ issues: IssueLike[];
114
+ /** checkFail diagnosis (usage / no-workspace / invalid rules) — carried
115
+ * through the wire shape so agents get the real cause, not just exit 2. */
116
+ error?: string;
117
+ }
118
+
119
+ // ── Section/rule vocabulary ──
120
+
121
+ const SECTION_ORDER = ['structure', 'style', 'refs', 'redundancy', 'overflow', 'parse', 'content', 'io', 'other'];
122
+ const KNOWN_RULE_PREFIXES = new Set(['structure', 'style', 'refs', 'redundancy', 'overflow', 'content', 'parse', 'io']);
123
+ const ENGINE_CATEGORIES = new Set(['structure', 'style', 'refs', 'redundancy', 'overflow']);
124
+
125
+ function sectionFor(rule: string, category: string): string {
126
+ const dot = rule.indexOf('.');
127
+ const prefix = dot === -1 ? rule : rule.slice(0, dot);
128
+ if (KNOWN_RULE_PREFIXES.has(prefix)) return prefix;
129
+ if (ENGINE_CATEGORIES.has(category)) return category;
130
+ return 'other';
131
+ }
132
+
133
+ // ── Message normalizers (one per engine emission format) ──
134
+
135
+ const RE_BROKEN_FILE = /^broken ref: see (\S+) — file not found$/;
136
+ const RE_BROKEN_ANCHOR = /^broken anchor: (\S+)#(\S+) — no node matches$/;
137
+ const RE_STALE_BP = /^stale back-pointer: (\S+) no longer refs (\S+)$/;
138
+ const RE_SELF = /^self-reference: (\S+) → (\S+)$/;
139
+ const RE_ORPHAN = /^orphan: (\S+) has no incoming or outgoing refs$/;
140
+ const RE_DEEP_HOP = /^DEEP HOP: (.+)$/;
141
+ const RE_CHAINING = /^no chaining: overflow target (\S+) must not contain its own see: refs \(found see (\S+)\)$/;
142
+ const RE_PROSE = /^see-like prose: "see (\S+)" did not resolve to a spec file/;
143
+ const RE_KEYWORD = /^"([^"]+)" × (\d+) nodes \(threshold: (\d+)\)$/;
144
+ const RE_OVERLAP = /^(\d+)% overlap: (\S+) ↔ (\S+)$/;
145
+ const RE_TYPO = /^possible typo: "([^"]+)" \(([^)]*)\) ↔ "([^"]+)" \(([^)]*)\) — Levenshtein (\d+)$/;
146
+ const RE_DUP_HOME_RED = /^"([^"]+)" at depth 0-1 in (\d+)\+ files without see: \(([^)]*)\)$/;
147
+ const RE_SIBLINGS = /^".*" has (\d+) children \((min|max) (\d+)\)\.?$/;
148
+ const RE_DEPTH_MIN = /^Max depth (\d+) is below min (\d+)\b/;
149
+ const RE_DEPTH_MAX = /^Depth (\d+) exceeds max (\d+)\b/;
150
+ const RE_NODE_LONG = /^Node too long \((\d+) > (\d+)\)/;
151
+ const RE_NODE_SHORT = /^Node too short \((\d+) < (\d+)\)/;
152
+ const RE_EMPTY_NODE = /^Empty node\.?$/;
153
+ const RE_SINGLE_CHILD = /^".*" has exactly 1 child\. Collapse\.$/;
154
+ const RE_MALFORMED = /^malformed workspace entry: directory "([^"]+)" looks like a spec file/;
155
+ const RE_DUP_HOME = /^duplicate home: both (\S+) and (\S+) exist/;
156
+ const RE_SHARED_PREFIX = /^(\d+) siblings share prefix "([^"]+)"/;
157
+ const RE_COLLAPSE = /^".*" has (\d+) child(?:ren)?\. Collapse to sibling style\.$/;
158
+ const RE_TBD_MAX = /^(\d+) TBD nodes exceed content\.max_tbd_per_file \((\d+)\)$/;
159
+ const RE_TBD_DISALLOWED = /^TBD used but content\.tbd_allowed is false$/;
160
+ const RE_CONTENT_TYPE = /^(code fence|table|[a-z_ ]+?) detected — extract to file/;
161
+ const RE_NODE_CHARS = /^node exceeds max chars \((\d+) > (\d+)\)$/;
162
+ const RE_UNREADABLE = /^unreadable spec file: /;
163
+ const RE_PARSE_ERROR = /^parse error: /;
164
+ const RE_PARSE_INDENT = /^odd indentation \((\d+) spaces?\) —/;
165
+
166
+ interface Normalized {
167
+ rule: string;
168
+ pattern: string;
169
+ key?: string;
170
+ detail?: string;
171
+ /** Per-member item (merged by label across the group). */
172
+ item?: { label: string; sort: number; metric?: number };
173
+ /** How the merged items are ranked: by item count, by metric, or input order. */
174
+ itemSort?: 'count' | 'pct' | 'metric' | 'appearance';
175
+ /** refs.broken.file: single item {label: key, count: group count} at finalize. */
176
+ targetItem?: boolean;
177
+ }
178
+
179
+ function stripMd(file: string): string {
180
+ return file.endsWith('.md') ? file.slice(0, -3) : file;
181
+ }
182
+
183
+ /** issue #41: unknown message shapes → first 40 chars with quoted strings
184
+ * removed and digit runs replaced by '#' (stable shape, no leaking values). */
185
+ function genericPattern(message: string): string {
186
+ const stripped = message
187
+ .replace(/"[^"]*"/g, '')
188
+ .replace(/'[^']*'/g, '')
189
+ .replace(/\d+/g, '#')
190
+ .replace(/\s+/g, ' ')
191
+ .trim();
192
+ return stripped.length > 0 ? stripped.slice(0, 40) : '(unrecognized)';
193
+ }
194
+
195
+ function normalizeMessage(issue: IssueLike): Normalized {
196
+ const m = issue.message;
197
+ let mt: RegExpMatchArray | null;
198
+
199
+ // ── refs ──
200
+ if ((mt = m.match(RE_BROKEN_FILE)) !== null) {
201
+ // issue #41: group per missing target — pattern generic, key = target path.
202
+ return { rule: 'refs.broken.file', pattern: 'missing file', key: mt[1]!, targetItem: true };
203
+ }
204
+ if ((mt = m.match(RE_BROKEN_ANCHOR)) !== null) {
205
+ const detail = `${issue.file}:${issue.line} → ${mt[1]}#${mt[2]}`;
206
+ return { rule: 'refs.broken.anchor', pattern: 'broken anchor', detail, item: { label: detail, sort: 0 } };
207
+ }
208
+ if ((mt = m.match(RE_STALE_BP)) !== null) {
209
+ // One group per stale target; items = referrer basenames, ranked by count.
210
+ return {
211
+ rule: 'refs.backpointer.stale',
212
+ pattern: 'stale back-pointer',
213
+ key: mt[2]!,
214
+ item: { label: stripMd(mt[1]!), sort: 0 },
215
+ itemSort: 'count',
216
+ };
217
+ }
218
+ if ((mt = m.match(RE_SELF)) !== null) {
219
+ return { rule: 'refs.self', pattern: 'self reference' };
220
+ }
221
+ if ((mt = m.match(RE_ORPHAN)) !== null) {
222
+ return { rule: 'refs.orphan', pattern: 'orphan file' };
223
+ }
224
+ if ((mt = m.match(RE_DEEP_HOP)) !== null) {
225
+ return { rule: 'refs.deep_hop', pattern: 'deep hop chain', detail: mt[1]! };
226
+ }
227
+ if ((mt = m.match(RE_CHAINING)) !== null) {
228
+ return { rule: 'refs.chaining', pattern: 'chaining in overflow target', key: mt[1]! };
229
+ }
230
+ if ((mt = m.match(RE_PROSE)) !== null) {
231
+ // Main's see-like-prose finding (advisory rephrase hint) — per-target group.
232
+ return { rule: 'refs.prose', pattern: 'see-like prose', key: mt[1]!, detail: m };
233
+ }
234
+
235
+ // ── redundancy ──
236
+ if ((mt = m.match(RE_KEYWORD)) !== null) {
237
+ // issue #41: ONE group — items = keywords ranked by NODE count desc (the
238
+ // message's ×N), so the engine's one-issue-per-keyword shape renders
239
+ // `artifacts:105 db:74 …` exactly as the issue specifies.
240
+ return {
241
+ rule: 'redundancy.keyword',
242
+ pattern: 'keyword sprawl',
243
+ item: { label: mt[1]!, sort: 0, metric: Number(mt[2]) },
244
+ itemSort: 'metric',
245
+ };
246
+ }
247
+ if ((mt = m.match(RE_OVERLAP)) !== null) {
248
+ const pct = Number(mt[1]);
249
+ if (pct >= 100) {
250
+ return { rule: 'redundancy.overlap.exact', pattern: 'exact overlap (100%)', item: { label: `${mt[2]} ↔ ${mt[3]}`, sort: 0 } };
251
+ }
252
+ return {
253
+ rule: 'redundancy.overlap.fuzzy',
254
+ pattern: 'fuzzy overlap (<100%)',
255
+ item: { label: `${mt[2]} ↔ ${mt[3]} (${pct}%)`, sort: pct },
256
+ itemSort: 'pct',
257
+ };
258
+ }
259
+ if ((mt = m.match(RE_TYPO)) !== null) {
260
+ // Not in the rule vocabulary — local extension so typos still group.
261
+ return { rule: 'redundancy.typo', pattern: 'possible typo', item: { label: `"${mt[1]}" ↔ "${mt[3]}"`, sort: 0 } };
262
+ }
263
+ if ((mt = m.match(RE_DUP_HOME_RED)) !== null) {
264
+ return { rule: 'redundancy.duplicate_home', pattern: 'duplicate home', key: mt[1]!, detail: m };
265
+ }
266
+
267
+ // ── structure ──
268
+ if ((mt = m.match(RE_SIBLINGS)) !== null) {
269
+ const isMin = mt[2] === 'min';
270
+ return {
271
+ rule: isMin ? 'structure.siblings.min' : 'structure.siblings.max',
272
+ // issue #41: quoted node title dropped, ratio kept — `<min children (2/3)`.
273
+ pattern: isMin ? `<min children (${mt[1]}/${mt[3]})` : `>max children (${mt[1]}/${mt[3]})`,
274
+ };
275
+ }
276
+ if ((mt = m.match(RE_DEPTH_MIN)) !== null) {
277
+ return { rule: 'structure.depth.min', pattern: `depth <min (${mt[1]}/${mt[2]})` };
278
+ }
279
+ if ((mt = m.match(RE_DEPTH_MAX)) !== null) {
280
+ return { rule: 'structure.depth.max', pattern: `depth >max (${mt[1]}/${mt[2]})` };
281
+ }
282
+ if ((mt = m.match(RE_NODE_LONG)) !== null) {
283
+ return { rule: 'structure.node_length.max', pattern: `node chars >max (${mt[1]}/${mt[2]})` };
284
+ }
285
+ if ((mt = m.match(RE_NODE_SHORT)) !== null) {
286
+ return { rule: 'structure.node_length.min', pattern: `node chars <min (${mt[1]}/${mt[2]})` };
287
+ }
288
+ if (RE_EMPTY_NODE.test(m)) {
289
+ return { rule: 'structure.empty_node', pattern: 'empty node' };
290
+ }
291
+ if (RE_SINGLE_CHILD.test(m)) {
292
+ return { rule: 'structure.single_child', pattern: 'single child (collapse)' };
293
+ }
294
+ if ((mt = m.match(RE_MALFORMED)) !== null) {
295
+ return { rule: 'structure.malformed_dir', pattern: 'malformed dir', key: mt[1]! };
296
+ }
297
+ if ((mt = m.match(RE_DUP_HOME)) !== null) {
298
+ return { rule: 'structure.duplicate_home', pattern: 'duplicate home', key: `${mt[1]}|${mt[2]}` };
299
+ }
300
+
301
+ // ── style ──
302
+ if ((mt = m.match(RE_SHARED_PREFIX)) !== null) {
303
+ // Per-word groups: the fix (group under a nested style) differs per prefix.
304
+ return { rule: 'style.prefix.shared', pattern: `shared prefix "${mt[2]}"`, key: `prefix:${mt[2]}` };
305
+ }
306
+ if ((mt = m.match(RE_COLLAPSE)) !== null) {
307
+ return { rule: 'style.nesting.prefer', pattern: `collapse to sibling (${mt[1]})` };
308
+ }
309
+
310
+ // ── content ──
311
+ if ((mt = m.match(RE_TBD_MAX)) !== null) {
312
+ return { rule: 'content.tbd.max', pattern: `tbd nodes >max (${mt[1]}/${mt[2]})` };
313
+ }
314
+ if (RE_TBD_DISALLOWED.test(m)) {
315
+ return { rule: 'content.tbd.disallowed', pattern: 'tbd disallowed' };
316
+ }
317
+
318
+ // ── overflow ──
319
+ if ((mt = m.match(RE_CONTENT_TYPE)) !== null) {
320
+ const kind = mt[1]!.trim();
321
+ if (kind === 'code fence') return { rule: 'overflow.code_fence', pattern: 'code fence (extract to file)' };
322
+ if (kind === 'table') return { rule: 'overflow.table', pattern: 'table (extract to file)' };
323
+ return { rule: 'overflow.force_file', pattern: `${kind} forced to file` };
324
+ }
325
+ if ((mt = m.match(RE_NODE_CHARS)) !== null) {
326
+ return { rule: 'overflow.node_chars', pattern: `node chars >max (${mt[1]}/${mt[2]})` };
327
+ }
328
+
329
+ // ── io / parse ──
330
+ if (RE_UNREADABLE.test(m)) {
331
+ return { rule: 'io.unreadable', pattern: 'unreadable file', detail: m };
332
+ }
333
+ if (RE_PARSE_ERROR.test(m)) {
334
+ return { rule: 'parse.error', pattern: 'parse error', detail: m };
335
+ }
336
+ if ((mt = m.match(RE_PARSE_INDENT)) !== null) {
337
+ const n = Number(mt[1]);
338
+ return { rule: 'parse.indent', pattern: `odd indentation (${n} ${n === 1 ? 'space' : 'spaces'})` };
339
+ }
340
+
341
+ // issue #41: fallback — unknown shapes must still group, never be dropped.
342
+ return { rule: `${issue.category}.other`, pattern: genericPattern(m) };
343
+ }
344
+
345
+ // ── Public API ──
346
+
347
+ /** issue #41: compact per-file locations — `action:30,111,143`. Lines are
348
+ * deduped and sorted numerically; the `.md` extension is stripped only when
349
+ * lines are listed (a line-0/absent entry renders as the bare `file.md`).
350
+ * Empty file names (check-level failures) produce no location string. */
351
+ export function formatLocations(entries: Array<{ file: string; line: number }>): string[] {
352
+ const perFile = new Map<string, number[]>();
353
+ const seen = new Set<string>();
354
+ for (const e of entries) {
355
+ if (e.file === '') continue;
356
+ const k = `${e.file}\u0000${e.line}`;
357
+ if (seen.has(k)) continue;
358
+ seen.add(k);
359
+ const lines = perFile.get(e.file);
360
+ if (lines === undefined) perFile.set(e.file, [e.line]);
361
+ else lines.push(e.line);
362
+ }
363
+ const out: string[] = [];
364
+ for (const [file, lines] of perFile) {
365
+ const nonZero = lines.filter((l) => l > 0).sort((a, b) => a - b);
366
+ if (nonZero.length === 0) {
367
+ out.push(file); // line 0 / absent → bare path, extension kept
368
+ } else {
369
+ out.push(`${stripMd(file)}:${nonZero.join(',')}`);
370
+ }
371
+ }
372
+ return out;
373
+ }
374
+
375
+ interface Acc {
376
+ group: IssueGroup;
377
+ firstIndex: number;
378
+ entries: Array<{ file: string; line: number }>;
379
+ seenLoc: Set<string>;
380
+ itemMap: Map<string, { label: string; count: number; sort: number; order: number; locations: string[]; metric?: number }> | null;
381
+ itemSort: 'count' | 'pct' | 'metric' | 'appearance';
382
+ wantsTargetItem: boolean;
383
+ }
384
+
385
+ function finalizeGroup(acc: Acc): IssueGroup {
386
+ const group = acc.group;
387
+ group.locations = formatLocations(acc.entries);
388
+ if (acc.itemMap !== null && acc.itemMap.size > 0) {
389
+ const items = [...acc.itemMap.values()];
390
+ if (acc.itemSort === 'count') items.sort((a, b) => b.count - a.count || a.order - b.order);
391
+ else if (acc.itemSort === 'pct') items.sort((a, b) => b.sort - a.sort || a.order - b.order);
392
+ else if (acc.itemSort === 'metric') items.sort((a, b) => (b.metric ?? 0) - (a.metric ?? 0) || b.count - a.count || a.order - b.order);
393
+ else items.sort((a, b) => a.order - b.order);
394
+ group.items = items.map((it) => ({
395
+ label: it.label,
396
+ count: it.count,
397
+ ...(it.metric !== undefined ? { metric: it.metric } : {}),
398
+ ...(it.locations.length > 0 ? { locations: it.locations } : {}),
399
+ }));
400
+ } else if (acc.wantsTargetItem && group.key !== undefined) {
401
+ // issue #41: missing-target groups show the target as their single item.
402
+ group.items = [{ label: group.key, count: group.count }];
403
+ }
404
+ return group;
405
+ }
406
+
407
+ /** issue #41: the pure aggregation core. Groups every issue by
408
+ * `rule + '|' + (key ?? pattern)`, orders by section → count desc → first
409
+ * occurrence, and never drops an issue. With `opts.topN`, each section's
410
+ * groups are folded to the top N (the flat `groups` array always stays
411
+ * complete — nothing is dropped silently). */
412
+ export function buildReport(issues: IssueLike[], opts?: { topN?: number }): CheckReport {
413
+ const byKey = new Map<string, Acc>();
414
+ const sectionCounts = new Map<string, { errorCount: number; warningCount: number }>();
415
+
416
+ for (let i = 0; i < issues.length; i++) {
417
+ const issue = issues[i]!;
418
+ const norm = normalizeMessage(issue);
419
+ const rule = issue.rule ?? norm.rule; // issue #41: explicit rule wins, fallback agrees
420
+ const section = sectionFor(rule, issue.category);
421
+
422
+ const counts = sectionCounts.get(section) ?? { errorCount: 0, warningCount: 0 };
423
+ if (issue.level === 'error') counts.errorCount++;
424
+ else counts.warningCount++;
425
+ sectionCounts.set(section, counts);
426
+
427
+ const gkey = `${rule}|${norm.key ?? norm.pattern}`;
428
+ let acc = byKey.get(gkey);
429
+ if (acc === undefined) {
430
+ const group: IssueGroup = {
431
+ category: section,
432
+ rule,
433
+ pattern: norm.pattern,
434
+ count: 0,
435
+ level: issue.level, // first member's level
436
+ locations: [],
437
+ };
438
+ if (norm.key !== undefined) group.key = norm.key;
439
+ acc = {
440
+ group,
441
+ firstIndex: i,
442
+ entries: [],
443
+ seenLoc: new Set(),
444
+ itemMap: norm.item !== undefined || norm.targetItem === true ? new Map() : null,
445
+ itemSort: norm.itemSort ?? 'appearance',
446
+ wantsTargetItem: norm.targetItem === true,
447
+ };
448
+ byKey.set(gkey, acc);
449
+ }
450
+ acc.group.count++;
451
+ if (acc.group.detail === undefined && norm.detail !== undefined) acc.group.detail = norm.detail;
452
+ if (acc.group.suggestion === undefined && issue.suggestion !== undefined && issue.suggestion !== '') {
453
+ acc.group.suggestion = issue.suggestion; // fix hint once per group
454
+ }
455
+ if (issue.file !== '') {
456
+ const locKey = `${issue.file}\u0000${issue.line}`;
457
+ if (!acc.seenLoc.has(locKey)) {
458
+ acc.seenLoc.add(locKey);
459
+ acc.entries.push({ file: issue.file, line: issue.line });
460
+ }
461
+ }
462
+ if (norm.item !== undefined && acc.itemMap !== null) {
463
+ const entry = acc.itemMap.get(norm.item.label) ?? {
464
+ label: norm.item.label,
465
+ count: 0,
466
+ sort: norm.item.sort,
467
+ order: acc.itemMap.size,
468
+ locations: [],
469
+ ...(norm.item.metric !== undefined ? { metric: norm.item.metric } : {}),
470
+ };
471
+ entry.count++;
472
+ if (issue.file !== '') {
473
+ entry.locations.push(formatLocations([{ file: issue.file, line: issue.line }])[0]!);
474
+ }
475
+ acc.itemMap.set(norm.item.label, entry);
476
+ }
477
+ }
478
+
479
+ const groups = [...byKey.values()]
480
+ .sort((a, b) => {
481
+ const sa = SECTION_ORDER.indexOf(a.group.category);
482
+ const sb = SECTION_ORDER.indexOf(b.group.category);
483
+ if (sa !== sb) return sa - sb;
484
+ if (b.group.count !== a.group.count) return b.group.count - a.group.count;
485
+ return a.firstIndex - b.firstIndex;
486
+ })
487
+ .map(finalizeGroup);
488
+
489
+ const sections: Record<string, SectionReport> = {};
490
+ for (const group of groups) {
491
+ let s = sections[group.category];
492
+ if (s === undefined) {
493
+ s = { name: group.category, errorCount: 0, warningCount: 0, groups: [] };
494
+ sections[group.category] = s;
495
+ }
496
+ s.groups.push(group);
497
+ }
498
+ for (const section of Object.values(sections)) {
499
+ const counts = sectionCounts.get(section.name)!;
500
+ section.errorCount = counts.errorCount;
501
+ section.warningCount = counts.warningCount;
502
+ }
503
+ if (opts?.topN !== undefined) {
504
+ for (const section of Object.values(sections)) {
505
+ section.groups = topGroups(section, opts.topN).shown;
506
+ }
507
+ }
508
+ return { groups, sections };
509
+ }
510
+
511
+ /** issue #41: top-N + fold. `shown` is the leading group slice; `folded` is
512
+ * the number of issue OCCURRENCES hidden in the tail (the "…and K more"
513
+ * count), not the number of hidden groups. */
514
+ export function topGroups(section: SectionReport, n: number): { shown: IssueGroup[]; folded: number } {
515
+ const limit = Math.max(n, 0);
516
+ const shown = section.groups.slice(0, limit);
517
+ let folded = 0;
518
+ for (let i = limit; i < section.groups.length; i++) folded += section.groups[i]!.count;
519
+ return { shown, folded };
520
+ }
521
+
522
+ /** ~4 chars per token heuristic (issue #41: ≤500-token default output budget). */
523
+ export function estimateTokens(text: string): number {
524
+ return Math.ceil(text.length / 4);
525
+ }
526
+
527
+ /** issue #41: lossless `--json` wire shape. One entry per raw issue in source
528
+ * order. Each entry is `{file, line, rule, detail}` (the acceptance shape)
529
+ * plus `level` and `suggestion` — agents must be able to filter errors from
530
+ * warnings and read fix hints without re-deriving them, so the wire format
531
+ * is a strict superset of the acceptance shape. Unknown engine categories
532
+ * land in `other`; no folding. */
533
+ export function checkReportJson(result: CheckResultLike): unknown {
534
+ const sections: Record<string, Array<{ file: string; line: number; level: string; rule: string; detail: string; suggestion?: string }>> = {};
535
+ for (const name of ['structure', 'style', 'refs', 'redundancy', 'overflow', 'other']) {
536
+ sections[name] = [];
537
+ }
538
+ for (const issue of result.issues) {
539
+ const bucket = ENGINE_CATEGORIES.has(issue.category) ? issue.category : 'other';
540
+ sections[bucket]!.push({
541
+ file: issue.file,
542
+ line: issue.line,
543
+ level: issue.level,
544
+ rule: issue.rule ?? normalizeMessage(issue).rule,
545
+ detail: issue.message,
546
+ ...(issue.suggestion !== undefined ? { suggestion: issue.suggestion } : {}),
547
+ });
548
+ }
549
+ const summary: { files: number; nodes: number; maxDepth: number; elapsedMs?: number } = {
550
+ files: result.files,
551
+ nodes: result.nodes,
552
+ maxDepth: result.maxDepth,
553
+ };
554
+ if (result.elapsedMs !== undefined) summary.elapsedMs = result.elapsedMs;
555
+ const out: Record<string, unknown> = {
556
+ ok: result.ok,
557
+ command: 'check',
558
+ exitCode: result.exitCode,
559
+ summary,
560
+ refs: { total: result.refs.total, broken: result.refs.broken, deepHops: result.refs.deepHops },
561
+ backPointers: {
562
+ total: result.backPointers.total,
563
+ current: result.backPointers.current,
564
+ stale: result.backPointers.stale,
565
+ },
566
+ counts: { errors: result.errorCount, warnings: result.warningCount },
567
+ sections,
568
+ backPointersUpdated: result.backPointersUpdated,
569
+ // issue #11: name the files --fix actually rewrote (sorted, spec-relative).
570
+ backPointersUpdatedFiles: result.backPointersUpdatedFiles,
571
+ };
572
+ if (result.rulesSummary !== undefined) out.rulesSummary = result.rulesSummary;
573
+ if (result.error !== undefined) out.error = result.error;
574
+ return out;
575
+ }
package/src/core/rules.ts CHANGED
@@ -221,6 +221,7 @@ export function defaultRules(): Rules {
221
221
  word_frequency_threshold: 4,
222
222
  phrase_overlap_threshold: 0.7,
223
223
  cross_file_threshold: 2,
224
+ fuzzy: true,
224
225
  stopwords: ['the', 'a', 'an', 'of', 'to', 'in', 'for', 'and', 'or', 'with', 'must', 'shall', 'requires'],
225
226
  synonyms: [
226
227
  ['postgres', 'postgresql', 'pg'],
@@ -298,6 +299,21 @@ function validateRulesShape(merged: Record<string, unknown>, source: string): vo
298
299
  ) {
299
300
  throw new Error(`${at} — "${section}.${key}" must be a mapping like { min: 3, max: 120 }`);
300
301
  }
302
+ // Issue #2: style.prefer / references.mode are validated reserved values.
303
+ // The merged object always carries the (valid) defaults, so an invalid
304
+ // value can only come from the user's _rules.yaml. `null` (an empty
305
+ // `prefer:` / `mode:`) is accepted and means the documented default; a
306
+ // non-string (e.g. `prefer: 42`) fails the same got-value convention.
307
+ if (section === 'style' && key === 'prefer') {
308
+ if (v !== null && (typeof v !== 'string' || (v !== 'sibling' && v !== 'nested'))) {
309
+ throw new Error(`${at} — "style.prefer" must be "sibling" or "nested", got "${String(v)}"`);
310
+ }
311
+ }
312
+ if (section === 'references' && key === 'mode') {
313
+ if (v !== null && (typeof v !== 'string' || v !== 'pointer')) {
314
+ throw new Error(`${at} — "references.mode" must be "pointer", got "${String(v)}"`);
315
+ }
316
+ }
301
317
  }
302
318
  }
303
319
  }
@@ -307,7 +323,10 @@ function validateRulesShape(merged: Record<string, unknown>, source: string): vo
307
323
  * token_budget.enabled/default_limit/estimate_chars_per_token) are listed too:
308
324
  * they count towards a section's coverage but are never flipped OFF — omitted
309
325
  * parameters keep their documented defaults (§18 overrides only what the file
310
- * lists for them). */
326
+ * lists for them). references.mode is additionally a VALIDATED reserved
327
+ * parameter: validateRulesShape rejects any value other than 'pointer' (or a
328
+ * deleted/empty null) with a line-numbered error, and style.prefer is
329
+ * validated the same way against 'sibling' | 'nested'. */
311
330
  const SECTION_KEYS: Record<string, string[]> = {
312
331
  structure: ['node_length', 'siblings', 'depth', 'single_child_collapse', 'empty_nodes'],
313
332
  style: ['prefer', 'force_nested_above', 'force_sibling_below', 'shared_prefix_detection'],
@@ -318,6 +337,7 @@ const SECTION_KEYS: Record<string, string[]> = {
318
337
  'word_frequency_threshold',
319
338
  'phrase_overlap_threshold',
320
339
  'cross_file_threshold',
340
+ 'fuzzy',
321
341
  'stopwords',
322
342
  'synonyms',
323
343
  ],
@@ -387,7 +407,8 @@ export function loadRules(root: string): Rules {
387
407
  // from its deep-merged default to its OFF state:
388
408
  // boolean switch → false (single_child_collapse, empty_nodes, tbd_allowed,
389
409
  // shared_prefix_detection, back_pointers,
390
- // orphan_check, duplicate_home_check, redundancy.enabled;
410
+ // orphan_check, duplicate_home_check,
411
+ // redundancy.enabled, redundancy.fuzzy;
391
412
  // an explicit `false` stays false — same OFF result)
392
413
  // mapping/numeric → null (node_length, siblings, depth, max_tbd_per_file,
393
414
  // force_nested_above, force_sibling_below, max_hops,
@@ -498,9 +519,11 @@ export function loadRules(root: string): Rules {
498
519
  }
499
520
 
500
521
  // redundancy: enabled / word_frequency_threshold / phrase_overlap_threshold /
501
- // cross_file_threshold. `stopwords`/`synonyms` are parameters (§13 inputs) —
502
- // they keep their defaults when omitted so the remaining layers still
503
- // normalize text exactly as documented.
522
+ // cross_file_threshold / fuzzy. `stopwords`/`synonyms` are parameters (§13
523
+ // inputs) — they keep their defaults when omitted so the remaining layers
524
+ // still normalize text exactly as documented. `fuzzy` is layer 3's own check
525
+ // switch (issue #3): deleted → the near-miss layer alone turns OFF while
526
+ // layers 1/2/4 keep running.
504
527
  if (!listed('redundancy')) {
505
528
  if (deleteMode) {
506
529
  rules.redundancy = {
@@ -509,6 +532,7 @@ export function loadRules(root: string): Rules {
509
532
  word_frequency_threshold: null,
510
533
  phrase_overlap_threshold: null,
511
534
  cross_file_threshold: null,
535
+ fuzzy: false,
512
536
  };
513
537
  }
514
538
  } else {
@@ -524,6 +548,9 @@ export function loadRules(root: string): Rules {
524
548
  if (deleted('redundancy', 'cross_file_threshold')) {
525
549
  rules.redundancy = { ...rules.redundancy, cross_file_threshold: null };
526
550
  }
551
+ if (deleted('redundancy', 'fuzzy')) {
552
+ rules.redundancy = { ...rules.redundancy, fuzzy: false };
553
+ }
527
554
  }
528
555
 
529
556
  // token_budget: warn_threshold omitted (deleted) → null → no usage warning.