cans-spec 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,575 @@
1
+ /** Pure issue-aggregation core for `cans check` (issue #41).
2
+ *
3
+ * Turns the raw issue list into pattern groups: one group per root cause,
4
+ * compact per-file locations (`action:30,111,143`), count prefixes for the
5
+ * human printer (`61× <min children (2/3)`), ranked sub-items (missing
6
+ * targets, keywords, overlap pairs, referrers), a fix hint once per group,
7
+ * and the lossless `--json` wire shape.
8
+ *
9
+ * FULLY STANDALONE: zero imports — not even from src/types.ts. IssueLike is
10
+ * structurally compatible with src/types.ts Issue (which gains an optional
11
+ * machine-readable `rule` at integration). Every known engine message format
12
+ * has a normalizer, so the module behaves identically with and without
13
+ * `rule` annotations — the rule path and the fallback path MUST agree
14
+ * (test/report.test.ts pins this on a full mixed scenario).
15
+ *
16
+ * Design decisions pinned here (integration agents read this):
17
+ * - Group key = `rule + '|' + (key ?? pattern)`. Threshold-family rules
18
+ * (siblings/depth/node-length/node_chars/tbd.max/collapse/parse.indent)
19
+ * encode the observed ratio IN the pattern, so `<min children (2/3)` and
20
+ * `<min children (1/3)` are distinct patterns; identity-family rules
21
+ * (missing file, stale target, keyword, overlap class, duplicate-home
22
+ * concept, prefix word, malformed dir) use a stable pattern plus a `key`
23
+ * carrying the root-cause identity.
24
+ * - Group ordering: section (structure → style → refs → redundancy →
25
+ * overflow → parse → content → io → other), then count desc, then first
26
+ * occurrence in the input. Nothing is dropped: every input issue raises
27
+ * its group's count exactly once.
28
+ * - `category` on a group is the SECTION (rule prefix wins over the engine
29
+ * category — e.g. content.tbd.* lands under `content`, refs.chaining
30
+ * under `refs` even though the overflow engine emits it).
31
+ * - `level` is the first member's level; sections count errors/warnings
32
+ * from the raw issues, so a hypothetically mixed group never skews them.
33
+ * - topGroups: `folded` is the number of issue OCCURRENCES hidden in the
34
+ * tail groups (the "…and K more" line), not the group count.
35
+ * - Engine message formats with no id in the rule vocabulary (transient
36
+ * refs, _collab/ refs, possible typos) fall through to `<category>.other`
37
+ * (typos get the local extension `redundancy.typo` so they still group).
38
+ */
39
+
40
+ // issue #41: contract types — the human printer and the JSON emitter build on these.
41
+ export interface IssueLike {
42
+ file: string;
43
+ line: number;
44
+ level: 'error' | 'warning';
45
+ category: string;
46
+ message: string;
47
+ suggestion?: string;
48
+ rule?: string;
49
+ }
50
+
51
+ export interface GroupItem {
52
+ label: string;
53
+ count: number;
54
+ locations?: string[];
55
+ /** issue #41 integration: semantic size when the message carries one —
56
+ * keyword sprawl's node count (`"artifacts" × 105 nodes` → 105). Ranking
57
+ * and display use the metric; `count` stays the raw occurrence count. */
58
+ metric?: number;
59
+ }
60
+
61
+ export interface IssueGroup {
62
+ /** Section: structure|style|refs|redundancy|overflow|parse|content|io|other */
63
+ category: string;
64
+ /** Normalized rule id (issue.rule when present, else derived from the message). */
65
+ rule: string;
66
+ /** One-line pattern label, e.g. `<min children (2/3)` or `missing file`. */
67
+ pattern: string;
68
+ /** Occurrences. */
69
+ count: number;
70
+ level: 'error' | 'warning';
71
+ /** Compact per-file strings `action:30,111,143` (dedup, stable order). */
72
+ locations: string[];
73
+ /** Metric-ranked sub-items (keywords, missing targets, overlap pairs, …). */
74
+ items?: GroupItem[];
75
+ /** Representative detail, e.g. `budget.md:148 → interface.md#Refusals`. */
76
+ detail?: string;
77
+ /** Fix hint — once per group (first non-empty among members). */
78
+ suggestion?: string;
79
+ /** Extracted root-cause key (missing target path, keyword, pair, …). */
80
+ key?: string;
81
+ }
82
+
83
+ export interface SectionReport {
84
+ name: string;
85
+ errorCount: number;
86
+ warningCount: number;
87
+ groups: IssueGroup[];
88
+ }
89
+
90
+ export interface CheckReport {
91
+ /** Every group, ordered by section → count desc → first occurrence. */
92
+ groups: IssueGroup[];
93
+ sections: Record<string, SectionReport>;
94
+ }
95
+
96
+ /** Structural subset of CheckResult needed for the --json wire shape. */
97
+ export interface CheckResultLike {
98
+ ok: boolean;
99
+ exitCode: number;
100
+ files: number;
101
+ nodes: number;
102
+ maxDepth: number;
103
+ elapsedMs?: number;
104
+ refs: { total: number; broken: number; deepHops: number };
105
+ backPointers: { total: number; current: number; stale: number };
106
+ errorCount: number;
107
+ warningCount: number;
108
+ backPointersUpdated: number;
109
+ /** Issue #11: sorted spec-relative paths of the files --fix actually
110
+ * rewrote (empty without --fix or when nothing needed a write). */
111
+ backPointersUpdatedFiles: string[];
112
+ rulesSummary?: string;
113
+ issues: IssueLike[];
114
+ /** checkFail diagnosis (usage / no-workspace / invalid rules) — carried
115
+ * through the wire shape so agents get the real cause, not just exit 2. */
116
+ error?: string;
117
+ }
118
+
119
+ // ── Section/rule vocabulary ──
120
+
121
+ const SECTION_ORDER = ['structure', 'style', 'refs', 'redundancy', 'overflow', 'parse', 'content', 'io', 'other'];
122
+ const KNOWN_RULE_PREFIXES = new Set(['structure', 'style', 'refs', 'redundancy', 'overflow', 'content', 'parse', 'io']);
123
+ const ENGINE_CATEGORIES = new Set(['structure', 'style', 'refs', 'redundancy', 'overflow']);
124
+
125
+ function sectionFor(rule: string, category: string): string {
126
+ const dot = rule.indexOf('.');
127
+ const prefix = dot === -1 ? rule : rule.slice(0, dot);
128
+ if (KNOWN_RULE_PREFIXES.has(prefix)) return prefix;
129
+ if (ENGINE_CATEGORIES.has(category)) return category;
130
+ return 'other';
131
+ }
132
+
133
+ // ── Message normalizers (one per engine emission format) ──
134
+
135
+ const RE_BROKEN_FILE = /^broken ref: see (\S+) — file not found$/;
136
+ const RE_BROKEN_ANCHOR = /^broken anchor: (\S+)#(\S+) — no node matches$/;
137
+ const RE_STALE_BP = /^stale back-pointer: (\S+) no longer refs (\S+)$/;
138
+ const RE_SELF = /^self-reference: (\S+) → (\S+)$/;
139
+ const RE_ORPHAN = /^orphan: (\S+) has no incoming or outgoing refs$/;
140
+ const RE_DEEP_HOP = /^DEEP HOP: (.+)$/;
141
+ const RE_CHAINING = /^no chaining: overflow target (\S+) must not contain its own see: refs \(found see (\S+)\)$/;
142
+ const RE_PROSE = /^see-like prose: "see (\S+)" did not resolve to a spec file/;
143
+ const RE_KEYWORD = /^"([^"]+)" × (\d+) nodes \(threshold: (\d+)\)$/;
144
+ const RE_OVERLAP = /^(\d+)% overlap: (\S+) ↔ (\S+)$/;
145
+ const RE_TYPO = /^possible typo: "([^"]+)" \(([^)]*)\) ↔ "([^"]+)" \(([^)]*)\) — Levenshtein (\d+)$/;
146
+ const RE_DUP_HOME_RED = /^"([^"]+)" at depth 0-1 in (\d+)\+ files without see: \(([^)]*)\)$/;
147
+ const RE_SIBLINGS = /^".*" has (\d+) children \((min|max) (\d+)\)\.?$/;
148
+ const RE_DEPTH_MIN = /^Max depth (\d+) is below min (\d+)\b/;
149
+ const RE_DEPTH_MAX = /^Depth (\d+) exceeds max (\d+)\b/;
150
+ const RE_NODE_LONG = /^Node too long \((\d+) > (\d+)\)/;
151
+ const RE_NODE_SHORT = /^Node too short \((\d+) < (\d+)\)/;
152
+ const RE_EMPTY_NODE = /^Empty node\.?$/;
153
+ const RE_SINGLE_CHILD = /^".*" has exactly 1 child\. Collapse\.$/;
154
+ const RE_MALFORMED = /^malformed workspace entry: directory "([^"]+)" looks like a spec file/;
155
+ const RE_DUP_HOME = /^duplicate home: both (\S+) and (\S+) exist/;
156
+ const RE_SHARED_PREFIX = /^(\d+) siblings share prefix "([^"]+)"/;
157
+ const RE_COLLAPSE = /^".*" has (\d+) child(?:ren)?\. Collapse to sibling style\.$/;
158
+ const RE_TBD_MAX = /^(\d+) TBD nodes exceed content\.max_tbd_per_file \((\d+)\)$/;
159
+ const RE_TBD_DISALLOWED = /^TBD used but content\.tbd_allowed is false$/;
160
+ const RE_CONTENT_TYPE = /^(code fence|table|[a-z_ ]+?) detected — extract to file/;
161
+ const RE_NODE_CHARS = /^node exceeds max chars \((\d+) > (\d+)\)$/;
162
+ const RE_UNREADABLE = /^unreadable spec file: /;
163
+ const RE_PARSE_ERROR = /^parse error: /;
164
+ const RE_PARSE_INDENT = /^odd indentation \((\d+) spaces?\) —/;
165
+
166
+ interface Normalized {
167
+ rule: string;
168
+ pattern: string;
169
+ key?: string;
170
+ detail?: string;
171
+ /** Per-member item (merged by label across the group). */
172
+ item?: { label: string; sort: number; metric?: number };
173
+ /** How the merged items are ranked: by item count, by metric, or input order. */
174
+ itemSort?: 'count' | 'pct' | 'metric' | 'appearance';
175
+ /** refs.broken.file: single item {label: key, count: group count} at finalize. */
176
+ targetItem?: boolean;
177
+ }
178
+
179
+ function stripMd(file: string): string {
180
+ return file.endsWith('.md') ? file.slice(0, -3) : file;
181
+ }
182
+
183
+ /** issue #41: unknown message shapes → first 40 chars with quoted strings
184
+ * removed and digit runs replaced by '#' (stable shape, no leaking values). */
185
+ function genericPattern(message: string): string {
186
+ const stripped = message
187
+ .replace(/"[^"]*"/g, '')
188
+ .replace(/'[^']*'/g, '')
189
+ .replace(/\d+/g, '#')
190
+ .replace(/\s+/g, ' ')
191
+ .trim();
192
+ return stripped.length > 0 ? stripped.slice(0, 40) : '(unrecognized)';
193
+ }
194
+
195
+ function normalizeMessage(issue: IssueLike): Normalized {
196
+ const m = issue.message;
197
+ let mt: RegExpMatchArray | null;
198
+
199
+ // ── refs ──
200
+ if ((mt = m.match(RE_BROKEN_FILE)) !== null) {
201
+ // issue #41: group per missing target — pattern generic, key = target path.
202
+ return { rule: 'refs.broken.file', pattern: 'missing file', key: mt[1]!, targetItem: true };
203
+ }
204
+ if ((mt = m.match(RE_BROKEN_ANCHOR)) !== null) {
205
+ const detail = `${issue.file}:${issue.line} → ${mt[1]}#${mt[2]}`;
206
+ return { rule: 'refs.broken.anchor', pattern: 'broken anchor', detail, item: { label: detail, sort: 0 } };
207
+ }
208
+ if ((mt = m.match(RE_STALE_BP)) !== null) {
209
+ // One group per stale target; items = referrer basenames, ranked by count.
210
+ return {
211
+ rule: 'refs.backpointer.stale',
212
+ pattern: 'stale back-pointer',
213
+ key: mt[2]!,
214
+ item: { label: stripMd(mt[1]!), sort: 0 },
215
+ itemSort: 'count',
216
+ };
217
+ }
218
+ if ((mt = m.match(RE_SELF)) !== null) {
219
+ return { rule: 'refs.self', pattern: 'self reference' };
220
+ }
221
+ if ((mt = m.match(RE_ORPHAN)) !== null) {
222
+ return { rule: 'refs.orphan', pattern: 'orphan file' };
223
+ }
224
+ if ((mt = m.match(RE_DEEP_HOP)) !== null) {
225
+ return { rule: 'refs.deep_hop', pattern: 'deep hop chain', detail: mt[1]! };
226
+ }
227
+ if ((mt = m.match(RE_CHAINING)) !== null) {
228
+ return { rule: 'refs.chaining', pattern: 'chaining in overflow target', key: mt[1]! };
229
+ }
230
+ if ((mt = m.match(RE_PROSE)) !== null) {
231
+ // Main's see-like-prose finding (advisory rephrase hint) — per-target group.
232
+ return { rule: 'refs.prose', pattern: 'see-like prose', key: mt[1]!, detail: m };
233
+ }
234
+
235
+ // ── redundancy ──
236
+ if ((mt = m.match(RE_KEYWORD)) !== null) {
237
+ // issue #41: ONE group — items = keywords ranked by NODE count desc (the
238
+ // message's ×N), so the engine's one-issue-per-keyword shape renders
239
+ // `artifacts:105 db:74 …` exactly as the issue specifies.
240
+ return {
241
+ rule: 'redundancy.keyword',
242
+ pattern: 'keyword sprawl',
243
+ item: { label: mt[1]!, sort: 0, metric: Number(mt[2]) },
244
+ itemSort: 'metric',
245
+ };
246
+ }
247
+ if ((mt = m.match(RE_OVERLAP)) !== null) {
248
+ const pct = Number(mt[1]);
249
+ if (pct >= 100) {
250
+ return { rule: 'redundancy.overlap.exact', pattern: 'exact overlap (100%)', item: { label: `${mt[2]} ↔ ${mt[3]}`, sort: 0 } };
251
+ }
252
+ return {
253
+ rule: 'redundancy.overlap.fuzzy',
254
+ pattern: 'fuzzy overlap (<100%)',
255
+ item: { label: `${mt[2]} ↔ ${mt[3]} (${pct}%)`, sort: pct },
256
+ itemSort: 'pct',
257
+ };
258
+ }
259
+ if ((mt = m.match(RE_TYPO)) !== null) {
260
+ // Not in the rule vocabulary — local extension so typos still group.
261
+ return { rule: 'redundancy.typo', pattern: 'possible typo', item: { label: `"${mt[1]}" ↔ "${mt[3]}"`, sort: 0 } };
262
+ }
263
+ if ((mt = m.match(RE_DUP_HOME_RED)) !== null) {
264
+ return { rule: 'redundancy.duplicate_home', pattern: 'duplicate home', key: mt[1]!, detail: m };
265
+ }
266
+
267
+ // ── structure ──
268
+ if ((mt = m.match(RE_SIBLINGS)) !== null) {
269
+ const isMin = mt[2] === 'min';
270
+ return {
271
+ rule: isMin ? 'structure.siblings.min' : 'structure.siblings.max',
272
+ // issue #41: quoted node title dropped, ratio kept — `<min children (2/3)`.
273
+ pattern: isMin ? `<min children (${mt[1]}/${mt[3]})` : `>max children (${mt[1]}/${mt[3]})`,
274
+ };
275
+ }
276
+ if ((mt = m.match(RE_DEPTH_MIN)) !== null) {
277
+ return { rule: 'structure.depth.min', pattern: `depth <min (${mt[1]}/${mt[2]})` };
278
+ }
279
+ if ((mt = m.match(RE_DEPTH_MAX)) !== null) {
280
+ return { rule: 'structure.depth.max', pattern: `depth >max (${mt[1]}/${mt[2]})` };
281
+ }
282
+ if ((mt = m.match(RE_NODE_LONG)) !== null) {
283
+ return { rule: 'structure.node_length.max', pattern: `node chars >max (${mt[1]}/${mt[2]})` };
284
+ }
285
+ if ((mt = m.match(RE_NODE_SHORT)) !== null) {
286
+ return { rule: 'structure.node_length.min', pattern: `node chars <min (${mt[1]}/${mt[2]})` };
287
+ }
288
+ if (RE_EMPTY_NODE.test(m)) {
289
+ return { rule: 'structure.empty_node', pattern: 'empty node' };
290
+ }
291
+ if (RE_SINGLE_CHILD.test(m)) {
292
+ return { rule: 'structure.single_child', pattern: 'single child (collapse)' };
293
+ }
294
+ if ((mt = m.match(RE_MALFORMED)) !== null) {
295
+ return { rule: 'structure.malformed_dir', pattern: 'malformed dir', key: mt[1]! };
296
+ }
297
+ if ((mt = m.match(RE_DUP_HOME)) !== null) {
298
+ return { rule: 'structure.duplicate_home', pattern: 'duplicate home', key: `${mt[1]}|${mt[2]}` };
299
+ }
300
+
301
+ // ── style ──
302
+ if ((mt = m.match(RE_SHARED_PREFIX)) !== null) {
303
+ // Per-word groups: the fix (group under a nested style) differs per prefix.
304
+ return { rule: 'style.prefix.shared', pattern: `shared prefix "${mt[2]}"`, key: `prefix:${mt[2]}` };
305
+ }
306
+ if ((mt = m.match(RE_COLLAPSE)) !== null) {
307
+ return { rule: 'style.nesting.prefer', pattern: `collapse to sibling (${mt[1]})` };
308
+ }
309
+
310
+ // ── content ──
311
+ if ((mt = m.match(RE_TBD_MAX)) !== null) {
312
+ return { rule: 'content.tbd.max', pattern: `tbd nodes >max (${mt[1]}/${mt[2]})` };
313
+ }
314
+ if (RE_TBD_DISALLOWED.test(m)) {
315
+ return { rule: 'content.tbd.disallowed', pattern: 'tbd disallowed' };
316
+ }
317
+
318
+ // ── overflow ──
319
+ if ((mt = m.match(RE_CONTENT_TYPE)) !== null) {
320
+ const kind = mt[1]!.trim();
321
+ if (kind === 'code fence') return { rule: 'overflow.code_fence', pattern: 'code fence (extract to file)' };
322
+ if (kind === 'table') return { rule: 'overflow.table', pattern: 'table (extract to file)' };
323
+ return { rule: 'overflow.force_file', pattern: `${kind} forced to file` };
324
+ }
325
+ if ((mt = m.match(RE_NODE_CHARS)) !== null) {
326
+ return { rule: 'overflow.node_chars', pattern: `node chars >max (${mt[1]}/${mt[2]})` };
327
+ }
328
+
329
+ // ── io / parse ──
330
+ if (RE_UNREADABLE.test(m)) {
331
+ return { rule: 'io.unreadable', pattern: 'unreadable file', detail: m };
332
+ }
333
+ if (RE_PARSE_ERROR.test(m)) {
334
+ return { rule: 'parse.error', pattern: 'parse error', detail: m };
335
+ }
336
+ if ((mt = m.match(RE_PARSE_INDENT)) !== null) {
337
+ const n = Number(mt[1]);
338
+ return { rule: 'parse.indent', pattern: `odd indentation (${n} ${n === 1 ? 'space' : 'spaces'})` };
339
+ }
340
+
341
+ // issue #41: fallback — unknown shapes must still group, never be dropped.
342
+ return { rule: `${issue.category}.other`, pattern: genericPattern(m) };
343
+ }
344
+
345
+ // ── Public API ──
346
+
347
+ /** issue #41: compact per-file locations — `action:30,111,143`. Lines are
348
+ * deduped and sorted numerically; the `.md` extension is stripped only when
349
+ * lines are listed (a line-0/absent entry renders as the bare `file.md`).
350
+ * Empty file names (check-level failures) produce no location string. */
351
+ export function formatLocations(entries: Array<{ file: string; line: number }>): string[] {
352
+ const perFile = new Map<string, number[]>();
353
+ const seen = new Set<string>();
354
+ for (const e of entries) {
355
+ if (e.file === '') continue;
356
+ const k = `${e.file}\u0000${e.line}`;
357
+ if (seen.has(k)) continue;
358
+ seen.add(k);
359
+ const lines = perFile.get(e.file);
360
+ if (lines === undefined) perFile.set(e.file, [e.line]);
361
+ else lines.push(e.line);
362
+ }
363
+ const out: string[] = [];
364
+ for (const [file, lines] of perFile) {
365
+ const nonZero = lines.filter((l) => l > 0).sort((a, b) => a - b);
366
+ if (nonZero.length === 0) {
367
+ out.push(file); // line 0 / absent → bare path, extension kept
368
+ } else {
369
+ out.push(`${stripMd(file)}:${nonZero.join(',')}`);
370
+ }
371
+ }
372
+ return out;
373
+ }
374
+
375
+ interface Acc {
376
+ group: IssueGroup;
377
+ firstIndex: number;
378
+ entries: Array<{ file: string; line: number }>;
379
+ seenLoc: Set<string>;
380
+ itemMap: Map<string, { label: string; count: number; sort: number; order: number; locations: string[]; metric?: number }> | null;
381
+ itemSort: 'count' | 'pct' | 'metric' | 'appearance';
382
+ wantsTargetItem: boolean;
383
+ }
384
+
385
+ function finalizeGroup(acc: Acc): IssueGroup {
386
+ const group = acc.group;
387
+ group.locations = formatLocations(acc.entries);
388
+ if (acc.itemMap !== null && acc.itemMap.size > 0) {
389
+ const items = [...acc.itemMap.values()];
390
+ if (acc.itemSort === 'count') items.sort((a, b) => b.count - a.count || a.order - b.order);
391
+ else if (acc.itemSort === 'pct') items.sort((a, b) => b.sort - a.sort || a.order - b.order);
392
+ else if (acc.itemSort === 'metric') items.sort((a, b) => (b.metric ?? 0) - (a.metric ?? 0) || b.count - a.count || a.order - b.order);
393
+ else items.sort((a, b) => a.order - b.order);
394
+ group.items = items.map((it) => ({
395
+ label: it.label,
396
+ count: it.count,
397
+ ...(it.metric !== undefined ? { metric: it.metric } : {}),
398
+ ...(it.locations.length > 0 ? { locations: it.locations } : {}),
399
+ }));
400
+ } else if (acc.wantsTargetItem && group.key !== undefined) {
401
+ // issue #41: missing-target groups show the target as their single item.
402
+ group.items = [{ label: group.key, count: group.count }];
403
+ }
404
+ return group;
405
+ }
406
+
407
+ /** issue #41: the pure aggregation core. Groups every issue by
408
+ * `rule + '|' + (key ?? pattern)`, orders by section → count desc → first
409
+ * occurrence, and never drops an issue. With `opts.topN`, each section's
410
+ * groups are folded to the top N (the flat `groups` array always stays
411
+ * complete — nothing is dropped silently). */
412
+ export function buildReport(issues: IssueLike[], opts?: { topN?: number }): CheckReport {
413
+ const byKey = new Map<string, Acc>();
414
+ const sectionCounts = new Map<string, { errorCount: number; warningCount: number }>();
415
+
416
+ for (let i = 0; i < issues.length; i++) {
417
+ const issue = issues[i]!;
418
+ const norm = normalizeMessage(issue);
419
+ const rule = issue.rule ?? norm.rule; // issue #41: explicit rule wins, fallback agrees
420
+ const section = sectionFor(rule, issue.category);
421
+
422
+ const counts = sectionCounts.get(section) ?? { errorCount: 0, warningCount: 0 };
423
+ if (issue.level === 'error') counts.errorCount++;
424
+ else counts.warningCount++;
425
+ sectionCounts.set(section, counts);
426
+
427
+ const gkey = `${rule}|${norm.key ?? norm.pattern}`;
428
+ let acc = byKey.get(gkey);
429
+ if (acc === undefined) {
430
+ const group: IssueGroup = {
431
+ category: section,
432
+ rule,
433
+ pattern: norm.pattern,
434
+ count: 0,
435
+ level: issue.level, // first member's level
436
+ locations: [],
437
+ };
438
+ if (norm.key !== undefined) group.key = norm.key;
439
+ acc = {
440
+ group,
441
+ firstIndex: i,
442
+ entries: [],
443
+ seenLoc: new Set(),
444
+ itemMap: norm.item !== undefined || norm.targetItem === true ? new Map() : null,
445
+ itemSort: norm.itemSort ?? 'appearance',
446
+ wantsTargetItem: norm.targetItem === true,
447
+ };
448
+ byKey.set(gkey, acc);
449
+ }
450
+ acc.group.count++;
451
+ if (acc.group.detail === undefined && norm.detail !== undefined) acc.group.detail = norm.detail;
452
+ if (acc.group.suggestion === undefined && issue.suggestion !== undefined && issue.suggestion !== '') {
453
+ acc.group.suggestion = issue.suggestion; // fix hint once per group
454
+ }
455
+ if (issue.file !== '') {
456
+ const locKey = `${issue.file}\u0000${issue.line}`;
457
+ if (!acc.seenLoc.has(locKey)) {
458
+ acc.seenLoc.add(locKey);
459
+ acc.entries.push({ file: issue.file, line: issue.line });
460
+ }
461
+ }
462
+ if (norm.item !== undefined && acc.itemMap !== null) {
463
+ const entry = acc.itemMap.get(norm.item.label) ?? {
464
+ label: norm.item.label,
465
+ count: 0,
466
+ sort: norm.item.sort,
467
+ order: acc.itemMap.size,
468
+ locations: [],
469
+ ...(norm.item.metric !== undefined ? { metric: norm.item.metric } : {}),
470
+ };
471
+ entry.count++;
472
+ if (issue.file !== '') {
473
+ entry.locations.push(formatLocations([{ file: issue.file, line: issue.line }])[0]!);
474
+ }
475
+ acc.itemMap.set(norm.item.label, entry);
476
+ }
477
+ }
478
+
479
+ const groups = [...byKey.values()]
480
+ .sort((a, b) => {
481
+ const sa = SECTION_ORDER.indexOf(a.group.category);
482
+ const sb = SECTION_ORDER.indexOf(b.group.category);
483
+ if (sa !== sb) return sa - sb;
484
+ if (b.group.count !== a.group.count) return b.group.count - a.group.count;
485
+ return a.firstIndex - b.firstIndex;
486
+ })
487
+ .map(finalizeGroup);
488
+
489
+ const sections: Record<string, SectionReport> = {};
490
+ for (const group of groups) {
491
+ let s = sections[group.category];
492
+ if (s === undefined) {
493
+ s = { name: group.category, errorCount: 0, warningCount: 0, groups: [] };
494
+ sections[group.category] = s;
495
+ }
496
+ s.groups.push(group);
497
+ }
498
+ for (const section of Object.values(sections)) {
499
+ const counts = sectionCounts.get(section.name)!;
500
+ section.errorCount = counts.errorCount;
501
+ section.warningCount = counts.warningCount;
502
+ }
503
+ if (opts?.topN !== undefined) {
504
+ for (const section of Object.values(sections)) {
505
+ section.groups = topGroups(section, opts.topN).shown;
506
+ }
507
+ }
508
+ return { groups, sections };
509
+ }
510
+
511
+ /** issue #41: top-N + fold. `shown` is the leading group slice; `folded` is
512
+ * the number of issue OCCURRENCES hidden in the tail (the "…and K more"
513
+ * count), not the number of hidden groups. */
514
+ export function topGroups(section: SectionReport, n: number): { shown: IssueGroup[]; folded: number } {
515
+ const limit = Math.max(n, 0);
516
+ const shown = section.groups.slice(0, limit);
517
+ let folded = 0;
518
+ for (let i = limit; i < section.groups.length; i++) folded += section.groups[i]!.count;
519
+ return { shown, folded };
520
+ }
521
+
522
+ /** ~4 chars per token heuristic (issue #41: ≤500-token default output budget). */
523
+ export function estimateTokens(text: string): number {
524
+ return Math.ceil(text.length / 4);
525
+ }
526
+
527
+ /** issue #41: lossless `--json` wire shape. One entry per raw issue in source
528
+ * order. Each entry is `{file, line, rule, detail}` (the acceptance shape)
529
+ * plus `level` and `suggestion` — agents must be able to filter errors from
530
+ * warnings and read fix hints without re-deriving them, so the wire format
531
+ * is a strict superset of the acceptance shape. Unknown engine categories
532
+ * land in `other`; no folding. */
533
+ export function checkReportJson(result: CheckResultLike): unknown {
534
+ const sections: Record<string, Array<{ file: string; line: number; level: string; rule: string; detail: string; suggestion?: string }>> = {};
535
+ for (const name of ['structure', 'style', 'refs', 'redundancy', 'overflow', 'other']) {
536
+ sections[name] = [];
537
+ }
538
+ for (const issue of result.issues) {
539
+ const bucket = ENGINE_CATEGORIES.has(issue.category) ? issue.category : 'other';
540
+ sections[bucket]!.push({
541
+ file: issue.file,
542
+ line: issue.line,
543
+ level: issue.level,
544
+ rule: issue.rule ?? normalizeMessage(issue).rule,
545
+ detail: issue.message,
546
+ ...(issue.suggestion !== undefined ? { suggestion: issue.suggestion } : {}),
547
+ });
548
+ }
549
+ const summary: { files: number; nodes: number; maxDepth: number; elapsedMs?: number } = {
550
+ files: result.files,
551
+ nodes: result.nodes,
552
+ maxDepth: result.maxDepth,
553
+ };
554
+ if (result.elapsedMs !== undefined) summary.elapsedMs = result.elapsedMs;
555
+ const out: Record<string, unknown> = {
556
+ ok: result.ok,
557
+ command: 'check',
558
+ exitCode: result.exitCode,
559
+ summary,
560
+ refs: { total: result.refs.total, broken: result.refs.broken, deepHops: result.refs.deepHops },
561
+ backPointers: {
562
+ total: result.backPointers.total,
563
+ current: result.backPointers.current,
564
+ stale: result.backPointers.stale,
565
+ },
566
+ counts: { errors: result.errorCount, warnings: result.warningCount },
567
+ sections,
568
+ backPointersUpdated: result.backPointersUpdated,
569
+ // issue #11: name the files --fix actually rewrote (sorted, spec-relative).
570
+ backPointersUpdatedFiles: result.backPointersUpdatedFiles,
571
+ };
572
+ if (result.rulesSummary !== undefined) out.rulesSummary = result.rulesSummary;
573
+ if (result.error !== undefined) out.error = result.error;
574
+ return out;
575
+ }
@@ -30,6 +30,7 @@ export function checkStructure(
30
30
  level: 'error',
31
31
  category: 'structure',
32
32
  message: `Node too long (${len} > ${nl.max}). Split or move to file.`,
33
+ rule: 'structure.node_length.max', // issue #41: machine-readable rule key
33
34
  });
34
35
  } else if (nl !== null && nl.min !== null && len < nl.min) {
35
36
  issues.push({
@@ -38,6 +39,7 @@ export function checkStructure(
38
39
  level: 'warning',
39
40
  category: 'structure',
40
41
  message: `Node too short (${len} < ${nl.min}).`,
42
+ rule: 'structure.node_length.min', // issue #41: machine-readable rule key
41
43
  });
42
44
  }
43
45
 
@@ -50,6 +52,7 @@ export function checkStructure(
50
52
  level: 'error',
51
53
  category: 'structure',
52
54
  message: `Depth ${depth} exceeds max ${depthMax}. Flatten.`,
55
+ rule: 'structure.depth.max', // issue #41: machine-readable rule key
53
56
  });
54
57
  }
55
58
 
@@ -62,6 +65,7 @@ export function checkStructure(
62
65
  level: 'warning',
63
66
  category: 'structure',
64
67
  message: `"${node.text}" has ${count} children (max ${siblingsMax}).`,
68
+ rule: 'structure.siblings.max', // issue #41: machine-readable rule key
65
69
  });
66
70
  }
67
71
  // Issue #1: enforce siblings.min — a parent with 0 < count < min children
@@ -76,6 +80,7 @@ export function checkStructure(
76
80
  level: 'warning',
77
81
  category: 'structure',
78
82
  message: `"${node.text}" has ${count} children (min ${siblingsMin}).`,
83
+ rule: 'structure.siblings.min', // issue #41: machine-readable rule key
79
84
  });
80
85
  }
81
86
 
@@ -86,6 +91,7 @@ export function checkStructure(
86
91
  level: 'warning',
87
92
  category: 'structure',
88
93
  message: `"${node.text}" has exactly 1 child. Collapse.`,
94
+ rule: 'structure.single_child', // issue #41: machine-readable rule key
89
95
  });
90
96
  }
91
97
 
@@ -96,6 +102,7 @@ export function checkStructure(
96
102
  level: 'warning',
97
103
  category: 'structure',
98
104
  message: 'Empty node.',
105
+ rule: 'structure.empty_node', // issue #41: machine-readable rule key
99
106
  });
100
107
  }
101
108
  }
@@ -133,6 +140,7 @@ export function checkStructure(
133
140
  category: 'structure',
134
141
  message: `Max depth ${maxNodeDepth} is below min ${depthMin}. Deepen the outline.`,
135
142
  suggestion: `add nested sub-levels until the outline reaches depth ${depthMin}, or lower structure.depth.min in _rules.yaml`,
143
+ rule: 'structure.depth.min', // issue #41: machine-readable rule key
136
144
  });
137
145
  }
138
146
  }
@@ -161,6 +169,7 @@ export function checkTbdPolicy(
161
169
  level: 'warning',
162
170
  category: 'structure',
163
171
  message: 'TBD used but content.tbd_allowed is false',
172
+ rule: 'content.tbd.disallowed', // issue #41: machine-readable rule key
164
173
  suggestion: 'resolve the TBD nodes or set content.tbd_allowed: true',
165
174
  },
166
175
  ];
@@ -173,6 +182,7 @@ export function checkTbdPolicy(
173
182
  level: 'warning',
174
183
  category: 'structure',
175
184
  message: `${tbdNodes.length} TBD nodes exceed content.max_tbd_per_file (${rules.max_tbd_per_file})`,
185
+ rule: 'content.tbd.max', // issue #41: machine-readable rule key
176
186
  suggestion: 'resolve the TBD nodes or raise content.max_tbd_per_file',
177
187
  },
178
188
  ];
package/src/core/style.ts CHANGED
@@ -66,6 +66,7 @@ export function checkStyle(
66
66
  level: 'warning',
67
67
  category: 'style',
68
68
  message: `${size} siblings share prefix "${word}". Group under nested style.`,
69
+ rule: 'style.prefix.shared', // issue #41: machine-readable rule key
69
70
  });
70
71
  }
71
72
  }
@@ -95,6 +96,7 @@ export function checkStyle(
95
96
  // QA-03 F17 pluralization contract (unreachable for 1 — the ≥2 guard
96
97
  // above plus the structure engine owns the 1-child case).
97
98
  message: `"${node.text}" has ${children.length} ${children.length === 1 ? 'child' : 'children'}. Collapse to sibling style.`,
99
+ rule: 'style.nesting.prefer', // issue #41: machine-readable rule key
98
100
  });
99
101
  }
100
102
  }