vigiles 14.0.0 → 14.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -13
- package/dist/adapters/claude-code/dialect.d.ts +1 -1
- package/dist/audit-report.js +4 -1
- package/dist/audit-report.template.html +28 -28
- package/dist/cli.js +194 -35
- package/dist/core/rule-catalog.d.ts +49 -4
- package/dist/core/rule-catalog.js +151 -3
- package/dist/instruction-sources.d.ts +1 -1
- package/dist/instruction-sources.js +1 -1
- package/dist/rule-inventory.js +151 -2
- package/dist/rule-routing.d.ts +63 -1
- package/dist/rule-routing.js +231 -98
- package/dist/rule-signals.d.ts +46 -0
- package/dist/rule-signals.js +49 -0
- package/dist/segment.d.ts +35 -2
- package/dist/segment.js +233 -171
- package/package.json +1 -1
package/dist/segment.js
CHANGED
|
@@ -10,17 +10,10 @@
|
|
|
10
10
|
*/
|
|
11
11
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
12
12
|
exports.segmentInstructions = segmentInstructions;
|
|
13
|
+
const rule_signals_js_1 = require("./rule-signals.js");
|
|
13
14
|
// --- Heuristic vocabulary --------------------------------------------------
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
* "Require `curly` braces", "Disallow `var`" — so they belong here. NB "no" is
|
|
17
|
-
* `no\s` (the prohibition word + whitespace) NOT the old `no\s+\S`, which — via
|
|
18
|
-
* the shared trailing `\b` — only matched when the word after "No " began at a
|
|
19
|
-
* boundary, so "No bare except" / "No default exports" silently failed the form
|
|
20
|
-
* cue. Bare `no` + the shared `\b` (checked right after "no", a boundary before
|
|
21
|
-
* a space OR a backtick) matches "No bare" AND "No `any`", while "Note"/"Nowhere"
|
|
22
|
-
* (no boundary after "no") are still rejected. */
|
|
23
|
-
const FORM_HEAD = /^(?:use|avoid|prefer|never|always|don'?t|do not|no|must|should|keep|run|write|add|remove|only|require|requires?|disallow|forbid|ban|enforce)\b/i;
|
|
15
|
+
// The deontic/imperative lexicon (FORM_HEAD, RULE_PREDICATE) lives in
|
|
16
|
+
// ./rule-signals.ts so the routing stage's NORM_SIGNAL can't drift from it.
|
|
24
17
|
/**
|
|
25
18
|
* Rule-ish heading gate for prose-under-heading candidacy. Word-bounded so the
|
|
26
19
|
* `do` alternate can't match inside `Documentation`/`Adoption`/`Download` (the
|
|
@@ -83,13 +76,37 @@ function looksLikeIndexEntry(t) {
|
|
|
83
76
|
* follows the leading code span.
|
|
84
77
|
*/
|
|
85
78
|
const DESCRIPTION_LED = /^`[^`]+`\s+(?:is|are|was|were|lives?|live|contains?|holds?|handles?|executes?|provides?|represents?|maps?|points?|implements?|exports?|defines?|wraps?|stores?|returns?|the|a|an|class|function|module|component|file|package|hook|utility|helper|type|interface|enum|constant|method|directory|folder|dir)\b/i;
|
|
86
|
-
//
|
|
87
|
-
// ("`const` is preferred over `let`"
|
|
88
|
-
//
|
|
89
|
-
|
|
79
|
+
// RULE_PREDICATE (a deontic modal anywhere) makes a code-span-led sentence a
|
|
80
|
+
// RULE, not a description ("`const` is preferred over `let`") — so the
|
|
81
|
+
// description reject must NOT fire. It lives in ./rule-signals.ts alongside
|
|
82
|
+
// NORM_SIGNAL (routing's twin) to keep the two from drifting.
|
|
83
|
+
/**
|
|
84
|
+
* DETERMINER-LED description: a sentence that opens with a determiner + noun
|
|
85
|
+
* subject and a descriptive copula ("The v1 README lives on the `v1.x` branch",
|
|
86
|
+
* "Each test lives in its own folder", "Many fixtures now provide a config") —
|
|
87
|
+
* an architecture/layout FACT, not a norm. Requiring a DETERMINER lead
|
|
88
|
+
* (the/each/all/every/many/…) is what keeps this precise: it excludes
|
|
89
|
+
* verb-first imperatives that merely contain a later copula ("Check you are not
|
|
90
|
+
* on main", "otherwise use `rg` … fall back to `grep`") — the exact
|
|
91
|
+
* false-positives a bare subject-copula pattern hit on the corpus. The
|
|
92
|
+
* RULE_PREDICATE guard below still lets "Each PR **must** …" through as a rule.
|
|
93
|
+
*/
|
|
94
|
+
const DESCRIPTION_DET = /^(?:the|a|an|each|all|every|most|many|our|its|their|this|these|those)\s+[`"']?[a-z][\w./-]*[`"']?(?:\s+[a-z][\w./-]*){0,3}\s+(?:is|are|was|were|lives?|resides?|exists?|contains?|holds?|serves?|provides?|has|have|uses?|maps?|points?|relies|rely|defaults?|becomes?|gets?|auto-\w+)\b/i;
|
|
90
95
|
function looksLikeDescription(t) {
|
|
91
96
|
const s = t.trim();
|
|
92
|
-
return DESCRIPTION_LED.test(s)
|
|
97
|
+
return ((DESCRIPTION_LED.test(s) || DESCRIPTION_DET.test(s)) &&
|
|
98
|
+
!rule_signals_js_1.RULE_PREDICATE.test(s));
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* A rule SIGNAL present: an imperative/prohibitive HEAD (`FORM_HEAD` on the
|
|
102
|
+
* decoration-stripped text) OR a deontic modal ANYWHERE (`RULE_PREDICATE`). Used
|
|
103
|
+
* to spare a colon-terminated line from the `leadin` reject when it actually
|
|
104
|
+
* carries a norm ("`expect` must come from test context — never import …:",
|
|
105
|
+
* "Avoid large modules:") — those introduce examples but ARE rules; only
|
|
106
|
+
* signal-less headers ("Run the full test suite:", "Python check:") drop.
|
|
107
|
+
*/
|
|
108
|
+
function hasRuleSignal(head, full) {
|
|
109
|
+
return rule_signals_js_1.FORM_HEAD.test(head) || rule_signals_js_1.RULE_PREDICATE.test(full);
|
|
93
110
|
}
|
|
94
111
|
/**
|
|
95
112
|
* RULE-NAME cue: a backticked token that is SHAPED like an off-the-shelf lint
|
|
@@ -310,8 +327,13 @@ function isLinkOnly(text) {
|
|
|
310
327
|
const t = text.trim();
|
|
311
328
|
return URL_ONLY.test(t) || LINK_ONLY.test(t);
|
|
312
329
|
}
|
|
330
|
+
/** The accept/reject view of a gate result, for sites that only need the split
|
|
331
|
+
* (atomize, the sub-span loop) and don't care about the reason. */
|
|
332
|
+
function confidenceOf(g) {
|
|
333
|
+
return "confidence" in g ? g.confidence : null;
|
|
334
|
+
}
|
|
313
335
|
/**
|
|
314
|
-
* Score the 3 cues. Returns confidence
|
|
336
|
+
* Score the 3 cues. Returns a confidence OR a reject reason.
|
|
315
337
|
* - form: starts with an imperative/prohibitive head (or "No X").
|
|
316
338
|
* - context: is a bullet OR sits under a rule-ish heading.
|
|
317
339
|
* - shape: 15–300 chars, has a verb-ish token, not link-only, not a declaration.
|
|
@@ -322,22 +344,29 @@ function gate(text, isBullet, underRuleHeading) {
|
|
|
322
344
|
// `a.ts → b.ts`, `dir/x — …`, `Label: path`) — the corpus's dominant false
|
|
323
345
|
// positive. No cue count can rescue it.
|
|
324
346
|
if (looksLikeIndexEntry(t))
|
|
325
|
-
return
|
|
347
|
+
return { reject: "index" };
|
|
326
348
|
// Reject a DESCRIPTION-led sentence (`` `Foo` class in `x` executes … ``) — an
|
|
327
349
|
// architecture/index sentence, not a rule (the dogfood's #1 false positive).
|
|
328
350
|
if (looksLikeDescription(t))
|
|
329
|
-
return
|
|
351
|
+
return { reject: "description" };
|
|
330
352
|
const context = isBullet || underRuleHeading;
|
|
331
353
|
// RULE-NAME cue: a bullet/section line that NAMES an off-the-shelf rule is a
|
|
332
354
|
// strong signal it's enforceable, even without an imperative verb — promote it
|
|
333
355
|
// to high so the high-only default doesn't drop it (recovers rule-naming
|
|
334
356
|
// bullets like "No floating promises (`@ts.../no-floating-promises`)").
|
|
335
357
|
if (context && RULE_NAME_IN_CODE.test(t))
|
|
336
|
-
return "high";
|
|
358
|
+
return { confidence: "high" };
|
|
337
359
|
// The form/declaration cues see the text with leading decoration stripped, so
|
|
338
360
|
// `- **Never** …` reads as imperative and `**We** …` still reads declarative.
|
|
339
361
|
const head = stripLeadDecoration(t);
|
|
340
|
-
|
|
362
|
+
// Reject a colon-terminated LEAD-IN header ("To add a setting:", "Run the full
|
|
363
|
+
// test suite:", "Python check:") — a procedure/enumeration heading whose real
|
|
364
|
+
// content sits in the sub-list/code-block it introduces (segmented on its
|
|
365
|
+
// own). Fires only when the header carries NO rule signal, so a norm-bearing
|
|
366
|
+
// header ("`expect` must come from test context — never …:") is kept.
|
|
367
|
+
if (/:\s*$/.test(t) && !hasRuleSignal(head, t))
|
|
368
|
+
return { reject: "leadin" };
|
|
369
|
+
const form = rule_signals_js_1.FORM_HEAD.test(head);
|
|
341
370
|
const shape = t.length >= 15 &&
|
|
342
371
|
t.length <= 300 &&
|
|
343
372
|
hasVerbish(t) &&
|
|
@@ -345,10 +374,10 @@ function gate(text, isBullet, underRuleHeading) {
|
|
|
345
374
|
!DECLARATION.test(head);
|
|
346
375
|
const cues = (form ? 1 : 0) + (context ? 1 : 0) + (shape ? 1 : 0);
|
|
347
376
|
if (cues >= 3)
|
|
348
|
-
return "high";
|
|
377
|
+
return { confidence: "high" };
|
|
349
378
|
if (cues === 2)
|
|
350
|
-
return "medium";
|
|
351
|
-
return
|
|
379
|
+
return { confidence: "medium" };
|
|
380
|
+
return { reject: "no-signal" };
|
|
352
381
|
}
|
|
353
382
|
// --- Atomicity split -------------------------------------------------------
|
|
354
383
|
/** Never split when an exception clause carries polarity/meaning. */
|
|
@@ -361,17 +390,9 @@ function trimSpan(src, span) {
|
|
|
361
390
|
end--;
|
|
362
391
|
return { start, end };
|
|
363
392
|
}
|
|
364
|
-
/**
|
|
365
|
-
*
|
|
366
|
-
|
|
367
|
-
* piece independently passes the gate; otherwise returns [whole].
|
|
368
|
-
*/
|
|
369
|
-
function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
370
|
-
const whole = trimSpan(src, contentSpan);
|
|
371
|
-
const wholeText = src.slice(whole.start, whole.end);
|
|
372
|
-
if (HAS_EXCEPT.test(wholeText))
|
|
373
|
-
return [whole];
|
|
374
|
-
// Candidate cut points: ';' and sentence terminators followed by a capital.
|
|
393
|
+
/** Candidate cut offsets inside a span: after every ';' and after a sentence
|
|
394
|
+
* terminator that is followed by whitespace + a capital (a real boundary). */
|
|
395
|
+
function findCutPoints(src, whole) {
|
|
375
396
|
const cuts = [];
|
|
376
397
|
for (let i = whole.start; i < whole.end; i++) {
|
|
377
398
|
const c = src[i];
|
|
@@ -379,44 +400,54 @@ function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
|
379
400
|
cuts.push(i + 1);
|
|
380
401
|
}
|
|
381
402
|
else if (c === "." || c === "!" || c === "?") {
|
|
382
|
-
|
|
383
|
-
const rest = src.slice(i + 1, whole.end);
|
|
384
|
-
const m = /^\s+[A-Z]/.exec(rest);
|
|
385
|
-
if (m)
|
|
403
|
+
if (/^\s+[A-Z]/.test(src.slice(i + 1, whole.end)))
|
|
386
404
|
cuts.push(i + 1);
|
|
387
405
|
}
|
|
388
406
|
}
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
407
|
+
return cuts;
|
|
408
|
+
}
|
|
409
|
+
/** Turn cut offsets into trimmed pieces (leading `;`/space stripped). Returns
|
|
410
|
+
* null if any piece is empty — the caller then keeps the span whole. */
|
|
411
|
+
function buildPieces(src, bounds) {
|
|
392
412
|
const pieces = [];
|
|
393
413
|
for (let i = 0; i < bounds.length - 1; i++) {
|
|
394
414
|
const piece = trimSpan(src, { start: bounds[i], end: bounds[i + 1] });
|
|
395
|
-
// strip a leading semicolon left by the cut
|
|
396
415
|
while (piece.start < piece.end &&
|
|
397
416
|
(src[piece.start] === ";" || /\s/.test(src[piece.start]))) {
|
|
398
417
|
piece.start++;
|
|
399
418
|
}
|
|
400
419
|
if (piece.start >= piece.end)
|
|
401
|
-
return
|
|
420
|
+
return null;
|
|
402
421
|
pieces.push(piece);
|
|
403
422
|
}
|
|
423
|
+
return pieces;
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* Try to split a single-line bullet's content span on ';' or sentence
|
|
427
|
+
* boundaries. Returns the resulting spans ONLY IF there is >1 and every
|
|
428
|
+
* piece independently passes the gate; otherwise returns [whole].
|
|
429
|
+
*/
|
|
430
|
+
function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
431
|
+
const whole = trimSpan(src, contentSpan);
|
|
432
|
+
if (HAS_EXCEPT.test(src.slice(whole.start, whole.end)))
|
|
433
|
+
return [whole];
|
|
434
|
+
const cuts = findCutPoints(src, whole);
|
|
435
|
+
if (cuts.length === 0)
|
|
436
|
+
return [whole];
|
|
437
|
+
const pieces = buildPieces(src, [whole.start, ...cuts, whole.end]);
|
|
438
|
+
if (pieces === null)
|
|
439
|
+
return [whole];
|
|
404
440
|
// Both/all halves must independently pass the gate, else keep whole.
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
if (gate(text, isBullet, underRuleHeading) === null)
|
|
408
|
-
return [whole];
|
|
409
|
-
}
|
|
410
|
-
return pieces.length > 1 ? pieces : [whole];
|
|
441
|
+
const allPass = pieces.every((p) => confidenceOf(gate(normalize(src.slice(p.start, p.end)), isBullet, underRuleHeading)) !== null);
|
|
442
|
+
return allPass && pieces.length > 1 ? pieces : [whole];
|
|
411
443
|
}
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
const exactQuote = src.slice(span.start, span.end);
|
|
444
|
+
function emitFromSpan(ctx, span, confidence) {
|
|
445
|
+
const exactQuote = ctx.src.slice(span.start, span.end);
|
|
415
446
|
return {
|
|
416
447
|
text: normalize(exactQuote),
|
|
417
|
-
file,
|
|
418
|
-
lineStart: offsetToLine(lineOffsets, span.start),
|
|
419
|
-
lineEnd: offsetToLine(lineOffsets, span.end - 1),
|
|
448
|
+
file: ctx.file,
|
|
449
|
+
lineStart: offsetToLine(ctx.lineOffsets, span.start),
|
|
450
|
+
lineEnd: offsetToLine(ctx.lineOffsets, span.end - 1),
|
|
420
451
|
exactQuote,
|
|
421
452
|
confidence,
|
|
422
453
|
};
|
|
@@ -428,25 +459,149 @@ const LIST_ITEM = /^(\s*)([-*+]|\d+[.)]|[✅❌☑✔✖✗])(\s+)(.*)$/u;
|
|
|
428
459
|
const HEADING = /^(#{1,6})\s+(.*)$/;
|
|
429
460
|
const FENCE = /^\s*(```|~~~)/;
|
|
430
461
|
const TABLE_LINE = /^\s*\|/;
|
|
462
|
+
/** Extend a list item over its continuation lines — deeper-indented, non-blank,
|
|
463
|
+
* not a new marker / heading / fence. Returns the item's last line index. */
|
|
464
|
+
function gatherListBody(lines, start, markerIndent) {
|
|
465
|
+
let end = start;
|
|
466
|
+
for (let j = start + 1; j < lines.length; j++) {
|
|
467
|
+
const cand = lines[j];
|
|
468
|
+
if (cand.trim() === "" || FENCE.test(cand) || HEADING.test(cand))
|
|
469
|
+
break;
|
|
470
|
+
if (cand.length - cand.trimStart().length <= markerIndent)
|
|
471
|
+
break;
|
|
472
|
+
if (LIST_ITEM.test(cand))
|
|
473
|
+
break; // nested/sibling bullet => separate candidate
|
|
474
|
+
end = j;
|
|
475
|
+
}
|
|
476
|
+
return end;
|
|
477
|
+
}
|
|
478
|
+
/** Extend a paragraph block until a blank / heading / list / fence / table. */
|
|
479
|
+
function gatherParagraph(lines, start) {
|
|
480
|
+
let end = start;
|
|
481
|
+
for (let j = start + 1; j < lines.length; j++) {
|
|
482
|
+
const cand = lines[j];
|
|
483
|
+
if (cand.trim() === "" || FENCE.test(cand) || HEADING.test(cand))
|
|
484
|
+
break;
|
|
485
|
+
if (LIST_ITEM.test(cand) || TABLE_LINE.test(cand))
|
|
486
|
+
break;
|
|
487
|
+
end = j;
|
|
488
|
+
}
|
|
489
|
+
return end;
|
|
490
|
+
}
|
|
491
|
+
/** Emit each atomized PIECE of a split bullet, re-gated independently (the
|
|
492
|
+
* whole-item case is handled by the caller, which emits the marker-inclusive
|
|
493
|
+
* span at its own gate result — so this only runs when `atomize` split). */
|
|
494
|
+
function emitSplitSpans(ctx, spans, ruleish) {
|
|
495
|
+
const out = [];
|
|
496
|
+
for (const s of spans) {
|
|
497
|
+
const text = normalize(ctx.src.slice(s.start, s.end));
|
|
498
|
+
const c = confidenceOf(gate(text, true, ruleish));
|
|
499
|
+
if (c !== null)
|
|
500
|
+
out.push(emitFromSpan(ctx, s, c));
|
|
501
|
+
}
|
|
502
|
+
return out;
|
|
503
|
+
}
|
|
504
|
+
/** Handle a list item at line `i`: gather its body, gate it, and either emit
|
|
505
|
+
* (whole or atomized) or record why it was skipped. */
|
|
506
|
+
function handleListItem(ctx, i, li, heading) {
|
|
507
|
+
const markerIndent = li[1].length;
|
|
508
|
+
const contentCol = li[1].length + li[2].length + li[3].length;
|
|
509
|
+
const endLine = gatherListBody(ctx.lines, i, markerIndent);
|
|
510
|
+
const contentStart = ctx.lineOffsets[i] + contentCol;
|
|
511
|
+
const contentEnd = ctx.lineOffsets[endLine] + ctx.lines[endLine].length;
|
|
512
|
+
const contentSpan = { start: contentStart, end: contentEnd };
|
|
513
|
+
const wholeText = normalize(ctx.src.slice(contentStart, contentEnd));
|
|
514
|
+
const g = gate(wholeText, true, heading.ruleish);
|
|
515
|
+
const conf = confidenceOf(g);
|
|
516
|
+
// Reject bullets under an anti-context heading (Commands/Setup/Key Files/…).
|
|
517
|
+
if (conf !== null && !heading.antiContext) {
|
|
518
|
+
// Only single-line items are split (keeps offsets exact). A single span
|
|
519
|
+
// emits the MARKER-INCLUSIVE full line at the whole-item confidence; a real
|
|
520
|
+
// split emits each piece re-gated at its own confidence.
|
|
521
|
+
const spans = endLine > i
|
|
522
|
+
? [trimSpan(ctx.src, contentSpan)]
|
|
523
|
+
: atomize(ctx.src, contentSpan, true, heading.ruleish);
|
|
524
|
+
const fullSpan = { start: ctx.lineOffsets[i], end: contentEnd };
|
|
525
|
+
return {
|
|
526
|
+
emitted: spans.length === 1
|
|
527
|
+
? [emitFromSpan(ctx, fullSpan, conf)]
|
|
528
|
+
: emitSplitSpans(ctx, spans, heading.ruleish),
|
|
529
|
+
skipped: [],
|
|
530
|
+
next: endLine + 1,
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
// NOT a rule — record it + why so the report is honest (§3). An anti-context
|
|
534
|
+
// rejection is a "section" skip; otherwise it's the gate's own reason.
|
|
535
|
+
return {
|
|
536
|
+
emitted: [],
|
|
537
|
+
skipped: [
|
|
538
|
+
{
|
|
539
|
+
text: wholeText,
|
|
540
|
+
file: ctx.file,
|
|
541
|
+
lineStart: offsetToLine(ctx.lineOffsets, contentStart),
|
|
542
|
+
lineEnd: offsetToLine(ctx.lineOffsets, contentEnd - 1),
|
|
543
|
+
reason: heading.antiContext || "confidence" in g ? "section" : g.reject,
|
|
544
|
+
},
|
|
545
|
+
],
|
|
546
|
+
next: endLine + 1,
|
|
547
|
+
};
|
|
548
|
+
}
|
|
549
|
+
/** Handle a paragraph block at line `i`: under a rule-ish heading, split into
|
|
550
|
+
* sentences and emit each that gates; otherwise emit nothing. Paragraph prose is
|
|
551
|
+
* never RECORDED as a skip (too noisy — see `SkippedBullet`), so `skipped` is
|
|
552
|
+
* always empty; it returns a `BlockResult` only so the dispatcher is uniform. */
|
|
553
|
+
function handleParagraph(ctx, i, heading) {
|
|
554
|
+
const endLine = gatherParagraph(ctx.lines, i);
|
|
555
|
+
const emitted = [];
|
|
556
|
+
if (heading.ruleish) {
|
|
557
|
+
const paraStart = ctx.lineOffsets[i];
|
|
558
|
+
const paraText = ctx.src.slice(paraStart, ctx.lineOffsets[endLine] + ctx.lines[endLine].length);
|
|
559
|
+
const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
|
|
560
|
+
let m;
|
|
561
|
+
while ((m = re.exec(paraText)) !== null) {
|
|
562
|
+
const s = trimSpan(ctx.src, {
|
|
563
|
+
start: paraStart + m.index,
|
|
564
|
+
end: paraStart + m.index + m[0].length,
|
|
565
|
+
});
|
|
566
|
+
if (s.start >= s.end)
|
|
567
|
+
continue;
|
|
568
|
+
const c = confidenceOf(gate(normalize(ctx.src.slice(s.start, s.end)), false, true));
|
|
569
|
+
if (c !== null)
|
|
570
|
+
emitted.push(emitFromSpan(ctx, s, c));
|
|
571
|
+
}
|
|
572
|
+
}
|
|
573
|
+
return { emitted, skipped: [], next: endLine + 1 };
|
|
574
|
+
}
|
|
575
|
+
/** Read a heading line into the rule-ish / anti-context state the gate keys on.
|
|
576
|
+
* Anti-context wins only when NOT also rule-ish, so an accept word wins a tie
|
|
577
|
+
* (`## Testing conventions` keeps its bullets; `## Testing` drops them). */
|
|
578
|
+
function headingStateFrom(headingText) {
|
|
579
|
+
const ruleish = RULE_HEADING.test(headingText);
|
|
580
|
+
return { ruleish, antiContext: ANTI_HEADING.test(headingText) && !ruleish };
|
|
581
|
+
}
|
|
431
582
|
/**
|
|
432
583
|
* Split a CLAUDE.md / AGENTS.md into atomic candidate rules with provenance.
|
|
433
584
|
*
|
|
434
585
|
* Deterministic Tier-A heuristic. Code fences and tables are excluded from
|
|
435
586
|
* candidacy. Candidate units are (a) list items with attached continuation
|
|
436
|
-
* lines and (b) sentences of paragraphs under a rule-ish heading.
|
|
587
|
+
* lines and (b) sentences of paragraphs under a rule-ish heading. This function
|
|
588
|
+
* is a thin DISPATCHER — each block type is handled by its own pure helper
|
|
589
|
+
* (`handleListItem` / `handleParagraph`); the state it threads is the fence
|
|
590
|
+
* toggle and the current `HeadingState`.
|
|
437
591
|
*/
|
|
438
592
|
function segmentInstructions(markdown, file, skipLines) {
|
|
439
593
|
const lines = markdown.split("\n");
|
|
440
|
-
const
|
|
594
|
+
const ctx = {
|
|
595
|
+
src: markdown,
|
|
596
|
+
lines,
|
|
597
|
+
lineOffsets: computeLineOffsets(lines),
|
|
598
|
+
file,
|
|
599
|
+
};
|
|
441
600
|
const out = [];
|
|
601
|
+
const skipped = [];
|
|
442
602
|
let inFence = false;
|
|
443
|
-
let
|
|
444
|
-
let currentHeadingIsAntiContext = false;
|
|
603
|
+
let heading = { ruleish: false, antiContext: false };
|
|
445
604
|
let i = 0;
|
|
446
|
-
const lineSpan = (a, b) => ({
|
|
447
|
-
start: lineOffsets[a],
|
|
448
|
-
end: lineOffsets[b] + lines[b].length,
|
|
449
|
-
});
|
|
450
605
|
while (i < lines.length) {
|
|
451
606
|
const line = lines[i];
|
|
452
607
|
// Code fences: toggle and skip everything inside (incl. the fence lines).
|
|
@@ -459,130 +614,37 @@ function segmentInstructions(markdown, file, skipLines) {
|
|
|
459
614
|
i++;
|
|
460
615
|
continue;
|
|
461
616
|
}
|
|
462
|
-
// Headings
|
|
617
|
+
// Headings update rule-ish context; not a candidate themselves.
|
|
463
618
|
const h = HEADING.exec(line);
|
|
464
619
|
if (h) {
|
|
465
|
-
|
|
466
|
-
// Anti-context only when it is NOT also rule-ish, so an accept word wins a
|
|
467
|
-
// tie (`## Testing conventions` keeps its bullets; `## Testing` drops them).
|
|
468
|
-
currentHeadingIsAntiContext =
|
|
469
|
-
ANTI_HEADING.test(h[2]) && !currentHeadingIsRuleish;
|
|
470
|
-
i++;
|
|
471
|
-
continue;
|
|
472
|
-
}
|
|
473
|
-
// Tables: excluded from candidacy.
|
|
474
|
-
if (TABLE_LINE.test(line)) {
|
|
620
|
+
heading = headingStateFrom(h[2]);
|
|
475
621
|
i++;
|
|
476
622
|
continue;
|
|
477
623
|
}
|
|
478
|
-
//
|
|
479
|
-
// section's body)
|
|
480
|
-
//
|
|
481
|
-
if (skipLines?.has(i + 1)) {
|
|
624
|
+
// Tables are excluded; so is a line already CONSUMED by the marker pre-pass
|
|
625
|
+
// (a marked section's body) — the span-consumption that stops a marked rule
|
|
626
|
+
// being double-counted by the heuristic (1-based).
|
|
627
|
+
if (TABLE_LINE.test(line) || skipLines?.has(i + 1)) {
|
|
482
628
|
i++;
|
|
483
629
|
continue;
|
|
484
630
|
}
|
|
485
|
-
// List items (with attached continuation lines).
|
|
486
631
|
const li = LIST_ITEM.exec(line);
|
|
487
632
|
if (li) {
|
|
488
|
-
const
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
// list marker, not a heading, not a fence.
|
|
493
|
-
let endLine = i;
|
|
494
|
-
let j = i + 1;
|
|
495
|
-
while (j < lines.length) {
|
|
496
|
-
const cand = lines[j];
|
|
497
|
-
if (cand.trim() === "")
|
|
498
|
-
break;
|
|
499
|
-
if (FENCE.test(cand))
|
|
500
|
-
break;
|
|
501
|
-
if (HEADING.test(cand))
|
|
502
|
-
break;
|
|
503
|
-
const indent = cand.length - cand.trimStart().length;
|
|
504
|
-
if (indent <= markerIndent)
|
|
505
|
-
break;
|
|
506
|
-
if (LIST_ITEM.test(cand))
|
|
507
|
-
break; // nested/sibling bullet => separate candidate
|
|
508
|
-
endLine = j;
|
|
509
|
-
j++;
|
|
510
|
-
}
|
|
511
|
-
const multiLine = endLine > startLine;
|
|
512
|
-
const contentStart = lineOffsets[startLine] + contentCol;
|
|
513
|
-
const contentEnd = lineOffsets[endLine] + lines[endLine].length;
|
|
514
|
-
const contentSpan = { start: contentStart, end: contentEnd };
|
|
515
|
-
const wholeText = normalize(markdown.slice(contentStart, contentEnd));
|
|
516
|
-
const conf = gate(wholeText, true, currentHeadingIsRuleish);
|
|
517
|
-
// Reject bullets under an anti-context heading (Commands/Setup/Key Files/
|
|
518
|
-
// Architecture/…) — the corpus's dominant false-positive locus.
|
|
519
|
-
if (conf !== null && !currentHeadingIsAntiContext) {
|
|
520
|
-
// Only attempt splitting for single-line items (keeps offsets exact).
|
|
521
|
-
const spans = multiLine
|
|
522
|
-
? [trimSpan(markdown, contentSpan)]
|
|
523
|
-
: atomize(markdown, contentSpan, true, currentHeadingIsRuleish);
|
|
524
|
-
if (spans.length === 1) {
|
|
525
|
-
// Emit whole item; exactQuote is the full source span incl. marker.
|
|
526
|
-
out.push(emitFromSpan(markdown, lineOffsets, file, lineSpan(startLine, endLine), conf));
|
|
527
|
-
}
|
|
528
|
-
else {
|
|
529
|
-
for (const s of spans) {
|
|
530
|
-
const text = normalize(markdown.slice(s.start, s.end));
|
|
531
|
-
const c = gate(text, true, currentHeadingIsRuleish);
|
|
532
|
-
if (c !== null)
|
|
533
|
-
out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
|
|
534
|
-
}
|
|
535
|
-
}
|
|
536
|
-
}
|
|
537
|
-
i = endLine + 1;
|
|
633
|
+
const r = handleListItem(ctx, i, li, heading);
|
|
634
|
+
out.push(...r.emitted);
|
|
635
|
+
skipped.push(...r.skipped);
|
|
636
|
+
i = r.next;
|
|
538
637
|
continue;
|
|
539
638
|
}
|
|
540
|
-
// Paragraph block: accumulate until blank / heading / list / fence / table.
|
|
541
639
|
if (line.trim() !== "") {
|
|
542
|
-
const
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
const cand = lines[j];
|
|
547
|
-
if (cand.trim() === "")
|
|
548
|
-
break;
|
|
549
|
-
if (FENCE.test(cand))
|
|
550
|
-
break;
|
|
551
|
-
if (HEADING.test(cand))
|
|
552
|
-
break;
|
|
553
|
-
if (LIST_ITEM.test(cand))
|
|
554
|
-
break;
|
|
555
|
-
if (TABLE_LINE.test(cand))
|
|
556
|
-
break;
|
|
557
|
-
endLine = j;
|
|
558
|
-
j++;
|
|
559
|
-
}
|
|
560
|
-
// Prose is only a candidate under a rule-ish heading.
|
|
561
|
-
if (currentHeadingIsRuleish) {
|
|
562
|
-
const paraStart = lineOffsets[startLine];
|
|
563
|
-
const paraEnd = lineOffsets[endLine] + lines[endLine].length;
|
|
564
|
-
const paraText = markdown.slice(paraStart, paraEnd);
|
|
565
|
-
// Sentence spans preserving absolute offsets.
|
|
566
|
-
const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
|
|
567
|
-
let m;
|
|
568
|
-
while ((m = re.exec(paraText)) !== null) {
|
|
569
|
-
const s = trimSpan(markdown, {
|
|
570
|
-
start: paraStart + m.index,
|
|
571
|
-
end: paraStart + m.index + m[0].length,
|
|
572
|
-
});
|
|
573
|
-
if (s.start >= s.end)
|
|
574
|
-
continue;
|
|
575
|
-
const text = normalize(markdown.slice(s.start, s.end));
|
|
576
|
-
const c = gate(text, false, true);
|
|
577
|
-
if (c !== null)
|
|
578
|
-
out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
|
|
579
|
-
}
|
|
580
|
-
}
|
|
581
|
-
i = endLine + 1;
|
|
640
|
+
const r = handleParagraph(ctx, i, heading);
|
|
641
|
+
out.push(...r.emitted);
|
|
642
|
+
skipped.push(...r.skipped);
|
|
643
|
+
i = r.next;
|
|
582
644
|
continue;
|
|
583
645
|
}
|
|
584
646
|
i++;
|
|
585
647
|
}
|
|
586
|
-
return out;
|
|
648
|
+
return { segments: out, skipped };
|
|
587
649
|
}
|
|
588
650
|
//# sourceMappingURL=segment.js.map
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "14.
|
|
3
|
+
"version": "14.2.0",
|
|
4
4
|
"description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|