vigiles 14.1.0 → 14.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -13
- package/dist/adapters/claude-code/dialect.d.ts +1 -1
- package/dist/audit-report.template.html +14 -14
- package/dist/cli.js +6 -4
- package/dist/core/rule-catalog.d.ts +1 -1
- package/dist/core/rule-catalog.js +1 -1
- package/dist/instruction-sources.d.ts +1 -1
- package/dist/instruction-sources.js +1 -1
- package/dist/rule-inventory.js +151 -2
- package/dist/rule-routing.d.ts +16 -1
- package/dist/rule-routing.js +172 -138
- package/dist/rule-signals.d.ts +46 -0
- package/dist/rule-signals.js +49 -0
- package/dist/segment.d.ts +16 -5
- package/dist/segment.js +219 -179
- package/package.json +1 -1
package/dist/segment.js
CHANGED
|
@@ -10,17 +10,10 @@
|
|
|
10
10
|
*/
|
|
11
11
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
12
12
|
exports.segmentInstructions = segmentInstructions;
|
|
13
|
+
const rule_signals_js_1 = require("./rule-signals.js");
|
|
13
14
|
// --- Heuristic vocabulary --------------------------------------------------
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
* "Require `curly` braces", "Disallow `var`" — so they belong here. NB "no" is
|
|
17
|
-
* `no\s` (the prohibition word + whitespace) NOT the old `no\s+\S`, which — via
|
|
18
|
-
* the shared trailing `\b` — only matched when the word after "No " began at a
|
|
19
|
-
* boundary, so "No bare except" / "No default exports" silently failed the form
|
|
20
|
-
* cue. Bare `no` + the shared `\b` (checked right after "no", a boundary before
|
|
21
|
-
* a space OR a backtick) matches "No bare" AND "No `any`", while "Note"/"Nowhere"
|
|
22
|
-
* (no boundary after "no") are still rejected. */
|
|
23
|
-
const FORM_HEAD = /^(?:use|avoid|prefer|never|always|don'?t|do not|no|must|should|keep|run|write|add|remove|only|require|requires?|disallow|forbid|ban|enforce)\b/i;
|
|
15
|
+
// The deontic/imperative lexicon (FORM_HEAD, RULE_PREDICATE) lives in
|
|
16
|
+
// ./rule-signals.ts so the routing stage's NORM_SIGNAL can't drift from it.
|
|
24
17
|
/**
|
|
25
18
|
* Rule-ish heading gate for prose-under-heading candidacy. Word-bounded so the
|
|
26
19
|
* `do` alternate can't match inside `Documentation`/`Adoption`/`Download` (the
|
|
@@ -83,13 +76,37 @@ function looksLikeIndexEntry(t) {
|
|
|
83
76
|
* follows the leading code span.
|
|
84
77
|
*/
|
|
85
78
|
const DESCRIPTION_LED = /^`[^`]+`\s+(?:is|are|was|were|lives?|live|contains?|holds?|handles?|executes?|provides?|represents?|maps?|points?|implements?|exports?|defines?|wraps?|stores?|returns?|the|a|an|class|function|module|component|file|package|hook|utility|helper|type|interface|enum|constant|method|directory|folder|dir)\b/i;
|
|
86
|
-
//
|
|
87
|
-
// ("`const` is preferred over `let`"
|
|
88
|
-
//
|
|
89
|
-
|
|
79
|
+
// RULE_PREDICATE (a deontic modal anywhere) makes a code-span-led sentence a
|
|
80
|
+
// RULE, not a description ("`const` is preferred over `let`") — so the
|
|
81
|
+
// description reject must NOT fire. It lives in ./rule-signals.ts alongside
|
|
82
|
+
// NORM_SIGNAL (routing's twin) to keep the two from drifting.
|
|
83
|
+
/**
|
|
84
|
+
* DETERMINER-LED description: a sentence that opens with a determiner + noun
|
|
85
|
+
* subject and a descriptive copula ("The v1 README lives on the `v1.x` branch",
|
|
86
|
+
* "Each test lives in its own folder", "Many fixtures now provide a config") —
|
|
87
|
+
* an architecture/layout FACT, not a norm. Requiring a DETERMINER lead
|
|
88
|
+
* (the/each/all/every/many/…) is what keeps this precise: it excludes
|
|
89
|
+
* verb-first imperatives that merely contain a later copula ("Check you are not
|
|
90
|
+
* on main", "otherwise use `rg` … fall back to `grep`") — the exact
|
|
91
|
+
* false-positives a bare subject-copula pattern hit on the corpus. The
|
|
92
|
+
* RULE_PREDICATE guard below still lets "Each PR **must** …" through as a rule.
|
|
93
|
+
*/
|
|
94
|
+
const DESCRIPTION_DET = /^(?:the|a|an|each|all|every|most|many|our|its|their|this|these|those)\s+[`"']?[a-z][\w./-]*[`"']?(?:\s+[a-z][\w./-]*){0,3}\s+(?:is|are|was|were|lives?|resides?|exists?|contains?|holds?|serves?|provides?|has|have|uses?|maps?|points?|relies|rely|defaults?|becomes?|gets?|auto-\w+)\b/i;
|
|
90
95
|
function looksLikeDescription(t) {
|
|
91
96
|
const s = t.trim();
|
|
92
|
-
return DESCRIPTION_LED.test(s)
|
|
97
|
+
return ((DESCRIPTION_LED.test(s) || DESCRIPTION_DET.test(s)) &&
|
|
98
|
+
!rule_signals_js_1.RULE_PREDICATE.test(s));
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* A rule SIGNAL present: an imperative/prohibitive HEAD (`FORM_HEAD` on the
|
|
102
|
+
* decoration-stripped text) OR a deontic modal ANYWHERE (`RULE_PREDICATE`). Used
|
|
103
|
+
* to spare a colon-terminated line from the `leadin` reject when it actually
|
|
104
|
+
* carries a norm ("`expect` must come from test context — never import …:",
|
|
105
|
+
* "Avoid large modules:") — those introduce examples but ARE rules; only
|
|
106
|
+
* signal-less headers ("Run the full test suite:", "Python check:") drop.
|
|
107
|
+
*/
|
|
108
|
+
function hasRuleSignal(head, full) {
|
|
109
|
+
return rule_signals_js_1.FORM_HEAD.test(head) || rule_signals_js_1.RULE_PREDICATE.test(full);
|
|
93
110
|
}
|
|
94
111
|
/**
|
|
95
112
|
* RULE-NAME cue: a backticked token that is SHAPED like an off-the-shelf lint
|
|
@@ -342,7 +359,14 @@ function gate(text, isBullet, underRuleHeading) {
|
|
|
342
359
|
// The form/declaration cues see the text with leading decoration stripped, so
|
|
343
360
|
// `- **Never** …` reads as imperative and `**We** …` still reads declarative.
|
|
344
361
|
const head = stripLeadDecoration(t);
|
|
345
|
-
|
|
362
|
+
// Reject a colon-terminated LEAD-IN header ("To add a setting:", "Run the full
|
|
363
|
+
// test suite:", "Python check:") — a procedure/enumeration heading whose real
|
|
364
|
+
// content sits in the sub-list/code-block it introduces (segmented on its
|
|
365
|
+
// own). Fires only when the header carries NO rule signal, so a norm-bearing
|
|
366
|
+
// header ("`expect` must come from test context — never …:") is kept.
|
|
367
|
+
if (/:\s*$/.test(t) && !hasRuleSignal(head, t))
|
|
368
|
+
return { reject: "leadin" };
|
|
369
|
+
const form = rule_signals_js_1.FORM_HEAD.test(head);
|
|
346
370
|
const shape = t.length >= 15 &&
|
|
347
371
|
t.length <= 300 &&
|
|
348
372
|
hasVerbish(t) &&
|
|
@@ -366,17 +390,9 @@ function trimSpan(src, span) {
|
|
|
366
390
|
end--;
|
|
367
391
|
return { start, end };
|
|
368
392
|
}
|
|
369
|
-
/**
|
|
370
|
-
*
|
|
371
|
-
|
|
372
|
-
* piece independently passes the gate; otherwise returns [whole].
|
|
373
|
-
*/
|
|
374
|
-
function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
375
|
-
const whole = trimSpan(src, contentSpan);
|
|
376
|
-
const wholeText = src.slice(whole.start, whole.end);
|
|
377
|
-
if (HAS_EXCEPT.test(wholeText))
|
|
378
|
-
return [whole];
|
|
379
|
-
// Candidate cut points: ';' and sentence terminators followed by a capital.
|
|
393
|
+
/** Candidate cut offsets inside a span: after every ';' and after a sentence
|
|
394
|
+
* terminator that is followed by whitespace + a capital (a real boundary). */
|
|
395
|
+
function findCutPoints(src, whole) {
|
|
380
396
|
const cuts = [];
|
|
381
397
|
for (let i = whole.start; i < whole.end; i++) {
|
|
382
398
|
const c = src[i];
|
|
@@ -384,44 +400,54 @@ function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
|
384
400
|
cuts.push(i + 1);
|
|
385
401
|
}
|
|
386
402
|
else if (c === "." || c === "!" || c === "?") {
|
|
387
|
-
|
|
388
|
-
const rest = src.slice(i + 1, whole.end);
|
|
389
|
-
const m = /^\s+[A-Z]/.exec(rest);
|
|
390
|
-
if (m)
|
|
403
|
+
if (/^\s+[A-Z]/.test(src.slice(i + 1, whole.end)))
|
|
391
404
|
cuts.push(i + 1);
|
|
392
405
|
}
|
|
393
406
|
}
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
407
|
+
return cuts;
|
|
408
|
+
}
|
|
409
|
+
/** Turn cut offsets into trimmed pieces (leading `;`/space stripped). Returns
|
|
410
|
+
* null if any piece is empty — the caller then keeps the span whole. */
|
|
411
|
+
function buildPieces(src, bounds) {
|
|
397
412
|
const pieces = [];
|
|
398
413
|
for (let i = 0; i < bounds.length - 1; i++) {
|
|
399
414
|
const piece = trimSpan(src, { start: bounds[i], end: bounds[i + 1] });
|
|
400
|
-
// strip a leading semicolon left by the cut
|
|
401
415
|
while (piece.start < piece.end &&
|
|
402
416
|
(src[piece.start] === ";" || /\s/.test(src[piece.start]))) {
|
|
403
417
|
piece.start++;
|
|
404
418
|
}
|
|
405
419
|
if (piece.start >= piece.end)
|
|
406
|
-
return
|
|
420
|
+
return null;
|
|
407
421
|
pieces.push(piece);
|
|
408
422
|
}
|
|
423
|
+
return pieces;
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* Try to split a single-line bullet's content span on ';' or sentence
|
|
427
|
+
* boundaries. Returns the resulting spans ONLY IF there is >1 and every
|
|
428
|
+
* piece independently passes the gate; otherwise returns [whole].
|
|
429
|
+
*/
|
|
430
|
+
function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
431
|
+
const whole = trimSpan(src, contentSpan);
|
|
432
|
+
if (HAS_EXCEPT.test(src.slice(whole.start, whole.end)))
|
|
433
|
+
return [whole];
|
|
434
|
+
const cuts = findCutPoints(src, whole);
|
|
435
|
+
if (cuts.length === 0)
|
|
436
|
+
return [whole];
|
|
437
|
+
const pieces = buildPieces(src, [whole.start, ...cuts, whole.end]);
|
|
438
|
+
if (pieces === null)
|
|
439
|
+
return [whole];
|
|
409
440
|
// Both/all halves must independently pass the gate, else keep whole.
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
if (confidenceOf(gate(text, isBullet, underRuleHeading)) === null)
|
|
413
|
-
return [whole];
|
|
414
|
-
}
|
|
415
|
-
return pieces.length > 1 ? pieces : [whole];
|
|
441
|
+
const allPass = pieces.every((p) => confidenceOf(gate(normalize(src.slice(p.start, p.end)), isBullet, underRuleHeading)) !== null);
|
|
442
|
+
return allPass && pieces.length > 1 ? pieces : [whole];
|
|
416
443
|
}
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
const exactQuote = src.slice(span.start, span.end);
|
|
444
|
+
function emitFromSpan(ctx, span, confidence) {
|
|
445
|
+
const exactQuote = ctx.src.slice(span.start, span.end);
|
|
420
446
|
return {
|
|
421
447
|
text: normalize(exactQuote),
|
|
422
|
-
file,
|
|
423
|
-
lineStart: offsetToLine(lineOffsets, span.start),
|
|
424
|
-
lineEnd: offsetToLine(lineOffsets, span.end - 1),
|
|
448
|
+
file: ctx.file,
|
|
449
|
+
lineStart: offsetToLine(ctx.lineOffsets, span.start),
|
|
450
|
+
lineEnd: offsetToLine(ctx.lineOffsets, span.end - 1),
|
|
425
451
|
exactQuote,
|
|
426
452
|
confidence,
|
|
427
453
|
};
|
|
@@ -433,26 +459,149 @@ const LIST_ITEM = /^(\s*)([-*+]|\d+[.)]|[✅❌☑✔✖✗])(\s+)(.*)$/u;
|
|
|
433
459
|
const HEADING = /^(#{1,6})\s+(.*)$/;
|
|
434
460
|
const FENCE = /^\s*(```|~~~)/;
|
|
435
461
|
const TABLE_LINE = /^\s*\|/;
|
|
462
|
+
/** Extend a list item over its continuation lines — deeper-indented, non-blank,
|
|
463
|
+
* not a new marker / heading / fence. Returns the item's last line index. */
|
|
464
|
+
function gatherListBody(lines, start, markerIndent) {
|
|
465
|
+
let end = start;
|
|
466
|
+
for (let j = start + 1; j < lines.length; j++) {
|
|
467
|
+
const cand = lines[j];
|
|
468
|
+
if (cand.trim() === "" || FENCE.test(cand) || HEADING.test(cand))
|
|
469
|
+
break;
|
|
470
|
+
if (cand.length - cand.trimStart().length <= markerIndent)
|
|
471
|
+
break;
|
|
472
|
+
if (LIST_ITEM.test(cand))
|
|
473
|
+
break; // nested/sibling bullet => separate candidate
|
|
474
|
+
end = j;
|
|
475
|
+
}
|
|
476
|
+
return end;
|
|
477
|
+
}
|
|
478
|
+
/** Extend a paragraph block until a blank / heading / list / fence / table. */
|
|
479
|
+
function gatherParagraph(lines, start) {
|
|
480
|
+
let end = start;
|
|
481
|
+
for (let j = start + 1; j < lines.length; j++) {
|
|
482
|
+
const cand = lines[j];
|
|
483
|
+
if (cand.trim() === "" || FENCE.test(cand) || HEADING.test(cand))
|
|
484
|
+
break;
|
|
485
|
+
if (LIST_ITEM.test(cand) || TABLE_LINE.test(cand))
|
|
486
|
+
break;
|
|
487
|
+
end = j;
|
|
488
|
+
}
|
|
489
|
+
return end;
|
|
490
|
+
}
|
|
491
|
+
/** Emit each atomized PIECE of a split bullet, re-gated independently (the
|
|
492
|
+
* whole-item case is handled by the caller, which emits the marker-inclusive
|
|
493
|
+
* span at its own gate result — so this only runs when `atomize` split). */
|
|
494
|
+
function emitSplitSpans(ctx, spans, ruleish) {
|
|
495
|
+
const out = [];
|
|
496
|
+
for (const s of spans) {
|
|
497
|
+
const text = normalize(ctx.src.slice(s.start, s.end));
|
|
498
|
+
const c = confidenceOf(gate(text, true, ruleish));
|
|
499
|
+
if (c !== null)
|
|
500
|
+
out.push(emitFromSpan(ctx, s, c));
|
|
501
|
+
}
|
|
502
|
+
return out;
|
|
503
|
+
}
|
|
504
|
+
/** Handle a list item at line `i`: gather its body, gate it, and either emit
|
|
505
|
+
* (whole or atomized) or record why it was skipped. */
|
|
506
|
+
function handleListItem(ctx, i, li, heading) {
|
|
507
|
+
const markerIndent = li[1].length;
|
|
508
|
+
const contentCol = li[1].length + li[2].length + li[3].length;
|
|
509
|
+
const endLine = gatherListBody(ctx.lines, i, markerIndent);
|
|
510
|
+
const contentStart = ctx.lineOffsets[i] + contentCol;
|
|
511
|
+
const contentEnd = ctx.lineOffsets[endLine] + ctx.lines[endLine].length;
|
|
512
|
+
const contentSpan = { start: contentStart, end: contentEnd };
|
|
513
|
+
const wholeText = normalize(ctx.src.slice(contentStart, contentEnd));
|
|
514
|
+
const g = gate(wholeText, true, heading.ruleish);
|
|
515
|
+
const conf = confidenceOf(g);
|
|
516
|
+
// Reject bullets under an anti-context heading (Commands/Setup/Key Files/…).
|
|
517
|
+
if (conf !== null && !heading.antiContext) {
|
|
518
|
+
// Only single-line items are split (keeps offsets exact). A single span
|
|
519
|
+
// emits the MARKER-INCLUSIVE full line at the whole-item confidence; a real
|
|
520
|
+
// split emits each piece re-gated at its own confidence.
|
|
521
|
+
const spans = endLine > i
|
|
522
|
+
? [trimSpan(ctx.src, contentSpan)]
|
|
523
|
+
: atomize(ctx.src, contentSpan, true, heading.ruleish);
|
|
524
|
+
const fullSpan = { start: ctx.lineOffsets[i], end: contentEnd };
|
|
525
|
+
return {
|
|
526
|
+
emitted: spans.length === 1
|
|
527
|
+
? [emitFromSpan(ctx, fullSpan, conf)]
|
|
528
|
+
: emitSplitSpans(ctx, spans, heading.ruleish),
|
|
529
|
+
skipped: [],
|
|
530
|
+
next: endLine + 1,
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
// NOT a rule — record it + why so the report is honest (§3). An anti-context
|
|
534
|
+
// rejection is a "section" skip; otherwise it's the gate's own reason.
|
|
535
|
+
return {
|
|
536
|
+
emitted: [],
|
|
537
|
+
skipped: [
|
|
538
|
+
{
|
|
539
|
+
text: wholeText,
|
|
540
|
+
file: ctx.file,
|
|
541
|
+
lineStart: offsetToLine(ctx.lineOffsets, contentStart),
|
|
542
|
+
lineEnd: offsetToLine(ctx.lineOffsets, contentEnd - 1),
|
|
543
|
+
reason: heading.antiContext || "confidence" in g ? "section" : g.reject,
|
|
544
|
+
},
|
|
545
|
+
],
|
|
546
|
+
next: endLine + 1,
|
|
547
|
+
};
|
|
548
|
+
}
|
|
549
|
+
/** Handle a paragraph block at line `i`: under a rule-ish heading, split into
|
|
550
|
+
* sentences and emit each that gates; otherwise emit nothing. Paragraph prose is
|
|
551
|
+
* never RECORDED as a skip (too noisy — see `SkippedBullet`), so `skipped` is
|
|
552
|
+
* always empty; it returns a `BlockResult` only so the dispatcher is uniform. */
|
|
553
|
+
function handleParagraph(ctx, i, heading) {
|
|
554
|
+
const endLine = gatherParagraph(ctx.lines, i);
|
|
555
|
+
const emitted = [];
|
|
556
|
+
if (heading.ruleish) {
|
|
557
|
+
const paraStart = ctx.lineOffsets[i];
|
|
558
|
+
const paraText = ctx.src.slice(paraStart, ctx.lineOffsets[endLine] + ctx.lines[endLine].length);
|
|
559
|
+
const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
|
|
560
|
+
let m;
|
|
561
|
+
while ((m = re.exec(paraText)) !== null) {
|
|
562
|
+
const s = trimSpan(ctx.src, {
|
|
563
|
+
start: paraStart + m.index,
|
|
564
|
+
end: paraStart + m.index + m[0].length,
|
|
565
|
+
});
|
|
566
|
+
if (s.start >= s.end)
|
|
567
|
+
continue;
|
|
568
|
+
const c = confidenceOf(gate(normalize(ctx.src.slice(s.start, s.end)), false, true));
|
|
569
|
+
if (c !== null)
|
|
570
|
+
emitted.push(emitFromSpan(ctx, s, c));
|
|
571
|
+
}
|
|
572
|
+
}
|
|
573
|
+
return { emitted, skipped: [], next: endLine + 1 };
|
|
574
|
+
}
|
|
575
|
+
/** Read a heading line into the rule-ish / anti-context state the gate keys on.
|
|
576
|
+
* Anti-context wins only when NOT also rule-ish, so an accept word wins a tie
|
|
577
|
+
* (`## Testing conventions` keeps its bullets; `## Testing` drops them). */
|
|
578
|
+
function headingStateFrom(headingText) {
|
|
579
|
+
const ruleish = RULE_HEADING.test(headingText);
|
|
580
|
+
return { ruleish, antiContext: ANTI_HEADING.test(headingText) && !ruleish };
|
|
581
|
+
}
|
|
436
582
|
/**
|
|
437
583
|
* Split a CLAUDE.md / AGENTS.md into atomic candidate rules with provenance.
|
|
438
584
|
*
|
|
439
585
|
* Deterministic Tier-A heuristic. Code fences and tables are excluded from
|
|
440
586
|
* candidacy. Candidate units are (a) list items with attached continuation
|
|
441
|
-
* lines and (b) sentences of paragraphs under a rule-ish heading.
|
|
587
|
+
* lines and (b) sentences of paragraphs under a rule-ish heading. This function
|
|
588
|
+
* is a thin DISPATCHER — each block type is handled by its own pure helper
|
|
589
|
+
* (`handleListItem` / `handleParagraph`); the state it threads is the fence
|
|
590
|
+
* toggle and the current `HeadingState`.
|
|
442
591
|
*/
|
|
443
592
|
function segmentInstructions(markdown, file, skipLines) {
|
|
444
593
|
const lines = markdown.split("\n");
|
|
445
|
-
const
|
|
594
|
+
const ctx = {
|
|
595
|
+
src: markdown,
|
|
596
|
+
lines,
|
|
597
|
+
lineOffsets: computeLineOffsets(lines),
|
|
598
|
+
file,
|
|
599
|
+
};
|
|
446
600
|
const out = [];
|
|
447
601
|
const skipped = [];
|
|
448
602
|
let inFence = false;
|
|
449
|
-
let
|
|
450
|
-
let currentHeadingIsAntiContext = false;
|
|
603
|
+
let heading = { ruleish: false, antiContext: false };
|
|
451
604
|
let i = 0;
|
|
452
|
-
const lineSpan = (a, b) => ({
|
|
453
|
-
start: lineOffsets[a],
|
|
454
|
-
end: lineOffsets[b] + lines[b].length,
|
|
455
|
-
});
|
|
456
605
|
while (i < lines.length) {
|
|
457
606
|
const line = lines[i];
|
|
458
607
|
// Code fences: toggle and skip everything inside (incl. the fence lines).
|
|
@@ -465,142 +614,33 @@ function segmentInstructions(markdown, file, skipLines) {
|
|
|
465
614
|
i++;
|
|
466
615
|
continue;
|
|
467
616
|
}
|
|
468
|
-
// Headings
|
|
617
|
+
// Headings update rule-ish context; not a candidate themselves.
|
|
469
618
|
const h = HEADING.exec(line);
|
|
470
619
|
if (h) {
|
|
471
|
-
|
|
472
|
-
// Anti-context only when it is NOT also rule-ish, so an accept word wins a
|
|
473
|
-
// tie (`## Testing conventions` keeps its bullets; `## Testing` drops them).
|
|
474
|
-
currentHeadingIsAntiContext =
|
|
475
|
-
ANTI_HEADING.test(h[2]) && !currentHeadingIsRuleish;
|
|
476
|
-
i++;
|
|
477
|
-
continue;
|
|
478
|
-
}
|
|
479
|
-
// Tables: excluded from candidacy.
|
|
480
|
-
if (TABLE_LINE.test(line)) {
|
|
620
|
+
heading = headingStateFrom(h[2]);
|
|
481
621
|
i++;
|
|
482
622
|
continue;
|
|
483
623
|
}
|
|
484
|
-
//
|
|
485
|
-
// section's body)
|
|
486
|
-
//
|
|
487
|
-
if (skipLines?.has(i + 1)) {
|
|
624
|
+
// Tables are excluded; so is a line already CONSUMED by the marker pre-pass
|
|
625
|
+
// (a marked section's body) — the span-consumption that stops a marked rule
|
|
626
|
+
// being double-counted by the heuristic (1-based).
|
|
627
|
+
if (TABLE_LINE.test(line) || skipLines?.has(i + 1)) {
|
|
488
628
|
i++;
|
|
489
629
|
continue;
|
|
490
630
|
}
|
|
491
|
-
// List items (with attached continuation lines).
|
|
492
631
|
const li = LIST_ITEM.exec(line);
|
|
493
632
|
if (li) {
|
|
494
|
-
const
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
// list marker, not a heading, not a fence.
|
|
499
|
-
let endLine = i;
|
|
500
|
-
let j = i + 1;
|
|
501
|
-
while (j < lines.length) {
|
|
502
|
-
const cand = lines[j];
|
|
503
|
-
if (cand.trim() === "")
|
|
504
|
-
break;
|
|
505
|
-
if (FENCE.test(cand))
|
|
506
|
-
break;
|
|
507
|
-
if (HEADING.test(cand))
|
|
508
|
-
break;
|
|
509
|
-
const indent = cand.length - cand.trimStart().length;
|
|
510
|
-
if (indent <= markerIndent)
|
|
511
|
-
break;
|
|
512
|
-
if (LIST_ITEM.test(cand))
|
|
513
|
-
break; // nested/sibling bullet => separate candidate
|
|
514
|
-
endLine = j;
|
|
515
|
-
j++;
|
|
516
|
-
}
|
|
517
|
-
const multiLine = endLine > startLine;
|
|
518
|
-
const contentStart = lineOffsets[startLine] + contentCol;
|
|
519
|
-
const contentEnd = lineOffsets[endLine] + lines[endLine].length;
|
|
520
|
-
const contentSpan = { start: contentStart, end: contentEnd };
|
|
521
|
-
const wholeText = normalize(markdown.slice(contentStart, contentEnd));
|
|
522
|
-
const g = gate(wholeText, true, currentHeadingIsRuleish);
|
|
523
|
-
const conf = confidenceOf(g);
|
|
524
|
-
// Reject bullets under an anti-context heading (Commands/Setup/Key Files/
|
|
525
|
-
// Architecture/…) — the corpus's dominant false-positive locus.
|
|
526
|
-
if (conf !== null && !currentHeadingIsAntiContext) {
|
|
527
|
-
// Only attempt splitting for single-line items (keeps offsets exact).
|
|
528
|
-
const spans = multiLine
|
|
529
|
-
? [trimSpan(markdown, contentSpan)]
|
|
530
|
-
: atomize(markdown, contentSpan, true, currentHeadingIsRuleish);
|
|
531
|
-
if (spans.length === 1) {
|
|
532
|
-
// Emit whole item; exactQuote is the full source span incl. marker.
|
|
533
|
-
out.push(emitFromSpan(markdown, lineOffsets, file, lineSpan(startLine, endLine), conf));
|
|
534
|
-
}
|
|
535
|
-
else {
|
|
536
|
-
for (const s of spans) {
|
|
537
|
-
const text = normalize(markdown.slice(s.start, s.end));
|
|
538
|
-
const c = confidenceOf(gate(text, true, currentHeadingIsRuleish));
|
|
539
|
-
if (c !== null)
|
|
540
|
-
out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
|
|
541
|
-
}
|
|
542
|
-
}
|
|
543
|
-
}
|
|
544
|
-
else {
|
|
545
|
-
// This bullet was NOT treated as a rule — record it + why, so the audit
|
|
546
|
-
// report can be honest about what it set aside (transparency, §3). A
|
|
547
|
-
// rejection under an anti-context heading is a "section" skip; otherwise
|
|
548
|
-
// it's the gate's own reason.
|
|
549
|
-
skipped.push({
|
|
550
|
-
text: wholeText,
|
|
551
|
-
file,
|
|
552
|
-
lineStart: offsetToLine(lineOffsets, contentStart),
|
|
553
|
-
lineEnd: offsetToLine(lineOffsets, contentEnd - 1),
|
|
554
|
-
reason: currentHeadingIsAntiContext || "confidence" in g
|
|
555
|
-
? "section"
|
|
556
|
-
: g.reject,
|
|
557
|
-
});
|
|
558
|
-
}
|
|
559
|
-
i = endLine + 1;
|
|
633
|
+
const r = handleListItem(ctx, i, li, heading);
|
|
634
|
+
out.push(...r.emitted);
|
|
635
|
+
skipped.push(...r.skipped);
|
|
636
|
+
i = r.next;
|
|
560
637
|
continue;
|
|
561
638
|
}
|
|
562
|
-
// Paragraph block: accumulate until blank / heading / list / fence / table.
|
|
563
639
|
if (line.trim() !== "") {
|
|
564
|
-
const
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
const cand = lines[j];
|
|
569
|
-
if (cand.trim() === "")
|
|
570
|
-
break;
|
|
571
|
-
if (FENCE.test(cand))
|
|
572
|
-
break;
|
|
573
|
-
if (HEADING.test(cand))
|
|
574
|
-
break;
|
|
575
|
-
if (LIST_ITEM.test(cand))
|
|
576
|
-
break;
|
|
577
|
-
if (TABLE_LINE.test(cand))
|
|
578
|
-
break;
|
|
579
|
-
endLine = j;
|
|
580
|
-
j++;
|
|
581
|
-
}
|
|
582
|
-
// Prose is only a candidate under a rule-ish heading.
|
|
583
|
-
if (currentHeadingIsRuleish) {
|
|
584
|
-
const paraStart = lineOffsets[startLine];
|
|
585
|
-
const paraEnd = lineOffsets[endLine] + lines[endLine].length;
|
|
586
|
-
const paraText = markdown.slice(paraStart, paraEnd);
|
|
587
|
-
// Sentence spans preserving absolute offsets.
|
|
588
|
-
const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
|
|
589
|
-
let m;
|
|
590
|
-
while ((m = re.exec(paraText)) !== null) {
|
|
591
|
-
const s = trimSpan(markdown, {
|
|
592
|
-
start: paraStart + m.index,
|
|
593
|
-
end: paraStart + m.index + m[0].length,
|
|
594
|
-
});
|
|
595
|
-
if (s.start >= s.end)
|
|
596
|
-
continue;
|
|
597
|
-
const text = normalize(markdown.slice(s.start, s.end));
|
|
598
|
-
const c = confidenceOf(gate(text, false, true));
|
|
599
|
-
if (c !== null)
|
|
600
|
-
out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
|
|
601
|
-
}
|
|
602
|
-
}
|
|
603
|
-
i = endLine + 1;
|
|
640
|
+
const r = handleParagraph(ctx, i, heading);
|
|
641
|
+
out.push(...r.emitted);
|
|
642
|
+
skipped.push(...r.skipped);
|
|
643
|
+
i = r.next;
|
|
604
644
|
continue;
|
|
605
645
|
}
|
|
606
646
|
i++;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "14.
|
|
3
|
+
"version": "14.2.0",
|
|
4
4
|
"description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|