vigiles 14.0.0 → 14.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/segment.js CHANGED
@@ -10,17 +10,10 @@
10
10
  */
11
11
  Object.defineProperty(exports, "__esModule", { value: true });
12
12
  exports.segmentInstructions = segmentInstructions;
13
+ const rule_signals_js_1 = require("./rule-signals.js");
13
14
  // --- Heuristic vocabulary --------------------------------------------------
14
- /** Imperative/prohibitive head the candidate must START with (form cue). The
15
- * deontic verbs (require/disallow/forbid/ban/enforce) are common rule leads —
16
- * "Require `curly` braces", "Disallow `var`" — so they belong here. NB "no" is
17
- * `no\s` (the prohibition word + whitespace) NOT the old `no\s+\S`, which — via
18
- * the shared trailing `\b` — only matched when the word after "No " began at a
19
- * boundary, so "No bare except" / "No default exports" silently failed the form
20
- * cue. Bare `no` + the shared `\b` (checked right after "no", a boundary before
21
- * a space OR a backtick) matches "No bare" AND "No `any`", while "Note"/"Nowhere"
22
- * (no boundary after "no") are still rejected. */
23
- const FORM_HEAD = /^(?:use|avoid|prefer|never|always|don'?t|do not|no|must|should|keep|run|write|add|remove|only|require|requires?|disallow|forbid|ban|enforce)\b/i;
15
+ // The deontic/imperative lexicon (FORM_HEAD, RULE_PREDICATE) lives in
16
+ // ./rule-signals.ts so the routing stage's NORM_SIGNAL can't drift from it.
24
17
  /**
25
18
  * Rule-ish heading gate for prose-under-heading candidacy. Word-bounded so the
26
19
  * `do` alternate can't match inside `Documentation`/`Adoption`/`Download` (the
@@ -83,13 +76,37 @@ function looksLikeIndexEntry(t) {
83
76
  * follows the leading code span.
84
77
  */
85
78
  const DESCRIPTION_LED = /^`[^`]+`\s+(?:is|are|was|were|lives?|live|contains?|holds?|handles?|executes?|provides?|represents?|maps?|points?|implements?|exports?|defines?|wraps?|stores?|returns?|the|a|an|class|function|module|component|file|package|hook|utility|helper|type|interface|enum|constant|method|directory|folder|dir)\b/i;
86
- // A deontic predicate makes a code-span-led sentence a RULE, not a description
87
- // ("`const` is preferred over `let`", "`AbstractBase` class must be extended") —
88
- // so the description reject must NOT fire. Guards the copula/kind-noun ambiguity.
89
- const RULE_PREDICATE = /\b(?:must|should|shall|never|always|require|avoid|prefer|banned|forbidden|prohibited|allowed|disallowed|deprecated|discouraged|mandatory|do not|don'?t|only|instead)\b/i;
79
+ // RULE_PREDICATE (a deontic modal anywhere) makes a code-span-led sentence a
80
+ // RULE, not a description ("`const` is preferred over `let`") — so the
81
+ // description reject must NOT fire. It lives in ./rule-signals.ts alongside
82
+ // NORM_SIGNAL (routing's twin) to keep the two from drifting.
83
+ /**
84
+ * DETERMINER-LED description: a sentence that opens with a determiner + noun
85
+ * subject and a descriptive copula ("The v1 README lives on the `v1.x` branch",
86
+ * "Each test lives in its own folder", "Many fixtures now provide a config") —
87
+ * an architecture/layout FACT, not a norm. Requiring a DETERMINER lead
88
+ * (the/each/all/every/many/…) is what keeps this precise: it excludes
89
+ * verb-first imperatives that merely contain a later copula ("Check you are not
90
+ * on main", "otherwise use `rg` … fall back to `grep`") — the exact
91
+ * false-positives a bare subject-copula pattern hit on the corpus. The
92
+ * RULE_PREDICATE guard below still lets "Each PR **must** …" through as a rule.
93
+ */
94
+ const DESCRIPTION_DET = /^(?:the|a|an|each|all|every|most|many|our|its|their|this|these|those)\s+[`"']?[a-z][\w./-]*[`"']?(?:\s+[a-z][\w./-]*){0,3}\s+(?:is|are|was|were|lives?|resides?|exists?|contains?|holds?|serves?|provides?|has|have|uses?|maps?|points?|relies|rely|defaults?|becomes?|gets?|auto-\w+)\b/i;
90
95
  function looksLikeDescription(t) {
91
96
  const s = t.trim();
92
- return DESCRIPTION_LED.test(s) && !RULE_PREDICATE.test(s);
97
+ return ((DESCRIPTION_LED.test(s) || DESCRIPTION_DET.test(s)) &&
98
+ !rule_signals_js_1.RULE_PREDICATE.test(s));
99
+ }
100
+ /**
101
+ * A rule SIGNAL present: an imperative/prohibitive HEAD (`FORM_HEAD` on the
102
+ * decoration-stripped text) OR a deontic modal ANYWHERE (`RULE_PREDICATE`). Used
103
+ * to spare a colon-terminated line from the `leadin` reject when it actually
104
+ * carries a norm ("`expect` must come from test context — never import …:",
105
+ * "Avoid large modules:") — those introduce examples but ARE rules; only
106
+ * signal-less headers ("Run the full test suite:", "Python check:") drop.
107
+ */
108
+ function hasRuleSignal(head, full) {
109
+ return rule_signals_js_1.FORM_HEAD.test(head) || rule_signals_js_1.RULE_PREDICATE.test(full);
93
110
  }
94
111
  /**
95
112
  * RULE-NAME cue: a backticked token that is SHAPED like an off-the-shelf lint
@@ -310,8 +327,13 @@ function isLinkOnly(text) {
310
327
  const t = text.trim();
311
328
  return URL_ONLY.test(t) || LINK_ONLY.test(t);
312
329
  }
330
+ /** The accept/reject view of a gate result, for sites that only need the split
331
+ * (atomize, the sub-span loop) and don't care about the reason. */
332
+ function confidenceOf(g) {
333
+ return "confidence" in g ? g.confidence : null;
334
+ }
313
335
  /**
314
- * Score the 3 cues. Returns confidence or null (reject).
336
+ * Score the 3 cues. Returns a confidence OR a reject reason.
315
337
  * - form: starts with an imperative/prohibitive head (or "No X").
316
338
  * - context: is a bullet OR sits under a rule-ish heading.
317
339
  * - shape: 15–300 chars, has a verb-ish token, not link-only, not a declaration.
@@ -322,22 +344,29 @@ function gate(text, isBullet, underRuleHeading) {
322
344
  // `a.ts → b.ts`, `dir/x — …`, `Label: path`) — the corpus's dominant false
323
345
  // positive. No cue count can rescue it.
324
346
  if (looksLikeIndexEntry(t))
325
- return null;
347
+ return { reject: "index" };
326
348
  // Reject a DESCRIPTION-led sentence (`` `Foo` class in `x` executes … ``) — an
327
349
  // architecture/index sentence, not a rule (the dogfood's #1 false positive).
328
350
  if (looksLikeDescription(t))
329
- return null;
351
+ return { reject: "description" };
330
352
  const context = isBullet || underRuleHeading;
331
353
  // RULE-NAME cue: a bullet/section line that NAMES an off-the-shelf rule is a
332
354
  // strong signal it's enforceable, even without an imperative verb — promote it
333
355
  // to high so the high-only default doesn't drop it (recovers rule-naming
334
356
  // bullets like "No floating promises (`@ts.../no-floating-promises`)").
335
357
  if (context && RULE_NAME_IN_CODE.test(t))
336
- return "high";
358
+ return { confidence: "high" };
337
359
  // The form/declaration cues see the text with leading decoration stripped, so
338
360
  // `- **Never** …` reads as imperative and `**We** …` still reads declarative.
339
361
  const head = stripLeadDecoration(t);
340
- const form = FORM_HEAD.test(head);
362
+ // Reject a colon-terminated LEAD-IN header ("To add a setting:", "Run the full
363
+ // test suite:", "Python check:") — a procedure/enumeration heading whose real
364
+ // content sits in the sub-list/code-block it introduces (segmented on its
365
+ // own). Fires only when the header carries NO rule signal, so a norm-bearing
366
+ // header ("`expect` must come from test context — never …:") is kept.
367
+ if (/:\s*$/.test(t) && !hasRuleSignal(head, t))
368
+ return { reject: "leadin" };
369
+ const form = rule_signals_js_1.FORM_HEAD.test(head);
341
370
  const shape = t.length >= 15 &&
342
371
  t.length <= 300 &&
343
372
  hasVerbish(t) &&
@@ -345,10 +374,10 @@ function gate(text, isBullet, underRuleHeading) {
345
374
  !DECLARATION.test(head);
346
375
  const cues = (form ? 1 : 0) + (context ? 1 : 0) + (shape ? 1 : 0);
347
376
  if (cues >= 3)
348
- return "high";
377
+ return { confidence: "high" };
349
378
  if (cues === 2)
350
- return "medium";
351
- return null;
379
+ return { confidence: "medium" };
380
+ return { reject: "no-signal" };
352
381
  }
353
382
  // --- Atomicity split -------------------------------------------------------
354
383
  /** Never split when an exception clause carries polarity/meaning. */
@@ -361,17 +390,9 @@ function trimSpan(src, span) {
361
390
  end--;
362
391
  return { start, end };
363
392
  }
364
- /**
365
- * Try to split a single-line bullet's content span on ';' or sentence
366
- * boundaries. Returns the resulting spans ONLY IF there is >1 and every
367
- * piece independently passes the gate; otherwise returns [whole].
368
- */
369
- function atomize(src, contentSpan, isBullet, underRuleHeading) {
370
- const whole = trimSpan(src, contentSpan);
371
- const wholeText = src.slice(whole.start, whole.end);
372
- if (HAS_EXCEPT.test(wholeText))
373
- return [whole];
374
- // Candidate cut points: ';' and sentence terminators followed by a capital.
393
+ /** Candidate cut offsets inside a span: after every ';' and after a sentence
394
+ * terminator that is followed by whitespace + a capital (a real boundary). */
395
+ function findCutPoints(src, whole) {
375
396
  const cuts = [];
376
397
  for (let i = whole.start; i < whole.end; i++) {
377
398
  const c = src[i];
@@ -379,44 +400,54 @@ function atomize(src, contentSpan, isBullet, underRuleHeading) {
379
400
  cuts.push(i + 1);
380
401
  }
381
402
  else if (c === "." || c === "!" || c === "?") {
382
- // sentence boundary: terminator + whitespace + capital letter
383
- const rest = src.slice(i + 1, whole.end);
384
- const m = /^\s+[A-Z]/.exec(rest);
385
- if (m)
403
+ if (/^\s+[A-Z]/.test(src.slice(i + 1, whole.end)))
386
404
  cuts.push(i + 1);
387
405
  }
388
406
  }
389
- if (cuts.length === 0)
390
- return [whole];
391
- const bounds = [whole.start, ...cuts, whole.end];
407
+ return cuts;
408
+ }
409
+ /** Turn cut offsets into trimmed pieces (leading `;`/space stripped). Returns
410
+ * null if any piece is empty — the caller then keeps the span whole. */
411
+ function buildPieces(src, bounds) {
392
412
  const pieces = [];
393
413
  for (let i = 0; i < bounds.length - 1; i++) {
394
414
  const piece = trimSpan(src, { start: bounds[i], end: bounds[i + 1] });
395
- // strip a leading semicolon left by the cut
396
415
  while (piece.start < piece.end &&
397
416
  (src[piece.start] === ";" || /\s/.test(src[piece.start]))) {
398
417
  piece.start++;
399
418
  }
400
419
  if (piece.start >= piece.end)
401
- return [whole];
420
+ return null;
402
421
  pieces.push(piece);
403
422
  }
423
+ return pieces;
424
+ }
425
+ /**
426
+ * Try to split a single-line bullet's content span on ';' or sentence
427
+ * boundaries. Returns the resulting spans ONLY IF there is >1 and every
428
+ * piece independently passes the gate; otherwise returns [whole].
429
+ */
430
+ function atomize(src, contentSpan, isBullet, underRuleHeading) {
431
+ const whole = trimSpan(src, contentSpan);
432
+ if (HAS_EXCEPT.test(src.slice(whole.start, whole.end)))
433
+ return [whole];
434
+ const cuts = findCutPoints(src, whole);
435
+ if (cuts.length === 0)
436
+ return [whole];
437
+ const pieces = buildPieces(src, [whole.start, ...cuts, whole.end]);
438
+ if (pieces === null)
439
+ return [whole];
404
440
  // Both/all halves must independently pass the gate, else keep whole.
405
- for (const p of pieces) {
406
- const text = normalize(src.slice(p.start, p.end));
407
- if (gate(text, isBullet, underRuleHeading) === null)
408
- return [whole];
409
- }
410
- return pieces.length > 1 ? pieces : [whole];
441
+ const allPass = pieces.every((p) => confidenceOf(gate(normalize(src.slice(p.start, p.end)), isBullet, underRuleHeading)) !== null);
442
+ return allPass && pieces.length > 1 ? pieces : [whole];
411
443
  }
412
- // --- Emission --------------------------------------------------------------
413
- function emitFromSpan(src, lineOffsets, file, span, confidence) {
414
- const exactQuote = src.slice(span.start, span.end);
444
+ function emitFromSpan(ctx, span, confidence) {
445
+ const exactQuote = ctx.src.slice(span.start, span.end);
415
446
  return {
416
447
  text: normalize(exactQuote),
417
- file,
418
- lineStart: offsetToLine(lineOffsets, span.start),
419
- lineEnd: offsetToLine(lineOffsets, span.end - 1),
448
+ file: ctx.file,
449
+ lineStart: offsetToLine(ctx.lineOffsets, span.start),
450
+ lineEnd: offsetToLine(ctx.lineOffsets, span.end - 1),
420
451
  exactQuote,
421
452
  confidence,
422
453
  };
@@ -428,25 +459,149 @@ const LIST_ITEM = /^(\s*)([-*+]|\d+[.)]|[✅❌☑✔✖✗])(\s+)(.*)$/u;
428
459
  const HEADING = /^(#{1,6})\s+(.*)$/;
429
460
  const FENCE = /^\s*(```|~~~)/;
430
461
  const TABLE_LINE = /^\s*\|/;
462
+ /** Extend a list item over its continuation lines — deeper-indented, non-blank,
463
+ * not a new marker / heading / fence. Returns the item's last line index. */
464
+ function gatherListBody(lines, start, markerIndent) {
465
+ let end = start;
466
+ for (let j = start + 1; j < lines.length; j++) {
467
+ const cand = lines[j];
468
+ if (cand.trim() === "" || FENCE.test(cand) || HEADING.test(cand))
469
+ break;
470
+ if (cand.length - cand.trimStart().length <= markerIndent)
471
+ break;
472
+ if (LIST_ITEM.test(cand))
473
+ break; // nested/sibling bullet => separate candidate
474
+ end = j;
475
+ }
476
+ return end;
477
+ }
478
+ /** Extend a paragraph block until a blank / heading / list / fence / table. */
479
+ function gatherParagraph(lines, start) {
480
+ let end = start;
481
+ for (let j = start + 1; j < lines.length; j++) {
482
+ const cand = lines[j];
483
+ if (cand.trim() === "" || FENCE.test(cand) || HEADING.test(cand))
484
+ break;
485
+ if (LIST_ITEM.test(cand) || TABLE_LINE.test(cand))
486
+ break;
487
+ end = j;
488
+ }
489
+ return end;
490
+ }
491
+ /** Emit each atomized PIECE of a split bullet, re-gated independently (the
492
+ * whole-item case is handled by the caller, which emits the marker-inclusive
493
+ * span at its own gate result — so this only runs when `atomize` split). */
494
+ function emitSplitSpans(ctx, spans, ruleish) {
495
+ const out = [];
496
+ for (const s of spans) {
497
+ const text = normalize(ctx.src.slice(s.start, s.end));
498
+ const c = confidenceOf(gate(text, true, ruleish));
499
+ if (c !== null)
500
+ out.push(emitFromSpan(ctx, s, c));
501
+ }
502
+ return out;
503
+ }
504
+ /** Handle a list item at line `i`: gather its body, gate it, and either emit
505
+ * (whole or atomized) or record why it was skipped. */
506
+ function handleListItem(ctx, i, li, heading) {
507
+ const markerIndent = li[1].length;
508
+ const contentCol = li[1].length + li[2].length + li[3].length;
509
+ const endLine = gatherListBody(ctx.lines, i, markerIndent);
510
+ const contentStart = ctx.lineOffsets[i] + contentCol;
511
+ const contentEnd = ctx.lineOffsets[endLine] + ctx.lines[endLine].length;
512
+ const contentSpan = { start: contentStart, end: contentEnd };
513
+ const wholeText = normalize(ctx.src.slice(contentStart, contentEnd));
514
+ const g = gate(wholeText, true, heading.ruleish);
515
+ const conf = confidenceOf(g);
516
+ // Reject bullets under an anti-context heading (Commands/Setup/Key Files/…).
517
+ if (conf !== null && !heading.antiContext) {
518
+ // Only single-line items are split (keeps offsets exact). A single span
519
+ // emits the MARKER-INCLUSIVE full line at the whole-item confidence; a real
520
+ // split emits each piece re-gated at its own confidence.
521
+ const spans = endLine > i
522
+ ? [trimSpan(ctx.src, contentSpan)]
523
+ : atomize(ctx.src, contentSpan, true, heading.ruleish);
524
+ const fullSpan = { start: ctx.lineOffsets[i], end: contentEnd };
525
+ return {
526
+ emitted: spans.length === 1
527
+ ? [emitFromSpan(ctx, fullSpan, conf)]
528
+ : emitSplitSpans(ctx, spans, heading.ruleish),
529
+ skipped: [],
530
+ next: endLine + 1,
531
+ };
532
+ }
533
+ // NOT a rule — record it + why so the report is honest (§3). An anti-context
534
+ // rejection is a "section" skip; otherwise it's the gate's own reason.
535
+ return {
536
+ emitted: [],
537
+ skipped: [
538
+ {
539
+ text: wholeText,
540
+ file: ctx.file,
541
+ lineStart: offsetToLine(ctx.lineOffsets, contentStart),
542
+ lineEnd: offsetToLine(ctx.lineOffsets, contentEnd - 1),
543
+ reason: heading.antiContext || "confidence" in g ? "section" : g.reject,
544
+ },
545
+ ],
546
+ next: endLine + 1,
547
+ };
548
+ }
549
+ /** Handle a paragraph block at line `i`: under a rule-ish heading, split into
550
+ * sentences and emit each that gates; otherwise emit nothing. Paragraph prose is
551
+ * never RECORDED as a skip (too noisy — see `SkippedBullet`), so `skipped` is
552
+ * always empty; it returns a `BlockResult` only so the dispatcher is uniform. */
553
+ function handleParagraph(ctx, i, heading) {
554
+ const endLine = gatherParagraph(ctx.lines, i);
555
+ const emitted = [];
556
+ if (heading.ruleish) {
557
+ const paraStart = ctx.lineOffsets[i];
558
+ const paraText = ctx.src.slice(paraStart, ctx.lineOffsets[endLine] + ctx.lines[endLine].length);
559
+ const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
560
+ let m;
561
+ while ((m = re.exec(paraText)) !== null) {
562
+ const s = trimSpan(ctx.src, {
563
+ start: paraStart + m.index,
564
+ end: paraStart + m.index + m[0].length,
565
+ });
566
+ if (s.start >= s.end)
567
+ continue;
568
+ const c = confidenceOf(gate(normalize(ctx.src.slice(s.start, s.end)), false, true));
569
+ if (c !== null)
570
+ emitted.push(emitFromSpan(ctx, s, c));
571
+ }
572
+ }
573
+ return { emitted, skipped: [], next: endLine + 1 };
574
+ }
575
+ /** Read a heading line into the rule-ish / anti-context state the gate keys on.
576
+ * Anti-context wins only when NOT also rule-ish, so an accept word wins a tie
577
+ * (`## Testing conventions` keeps its bullets; `## Testing` drops them). */
578
+ function headingStateFrom(headingText) {
579
+ const ruleish = RULE_HEADING.test(headingText);
580
+ return { ruleish, antiContext: ANTI_HEADING.test(headingText) && !ruleish };
581
+ }
431
582
  /**
432
583
  * Split a CLAUDE.md / AGENTS.md into atomic candidate rules with provenance.
433
584
  *
434
585
  * Deterministic Tier-A heuristic. Code fences and tables are excluded from
435
586
  * candidacy. Candidate units are (a) list items with attached continuation
436
- * lines and (b) sentences of paragraphs under a rule-ish heading.
587
+ * lines and (b) sentences of paragraphs under a rule-ish heading. This function
588
+ * is a thin DISPATCHER — each block type is handled by its own pure helper
589
+ * (`handleListItem` / `handleParagraph`); the state it threads is the fence
590
+ * toggle and the current `HeadingState`.
437
591
  */
438
592
  function segmentInstructions(markdown, file, skipLines) {
439
593
  const lines = markdown.split("\n");
440
- const lineOffsets = computeLineOffsets(lines);
594
+ const ctx = {
595
+ src: markdown,
596
+ lines,
597
+ lineOffsets: computeLineOffsets(lines),
598
+ file,
599
+ };
441
600
  const out = [];
601
+ const skipped = [];
442
602
  let inFence = false;
443
- let currentHeadingIsRuleish = false;
444
- let currentHeadingIsAntiContext = false;
603
+ let heading = { ruleish: false, antiContext: false };
445
604
  let i = 0;
446
- const lineSpan = (a, b) => ({
447
- start: lineOffsets[a],
448
- end: lineOffsets[b] + lines[b].length,
449
- });
450
605
  while (i < lines.length) {
451
606
  const line = lines[i];
452
607
  // Code fences: toggle and skip everything inside (incl. the fence lines).
@@ -459,130 +614,37 @@ function segmentInstructions(markdown, file, skipLines) {
459
614
  i++;
460
615
  continue;
461
616
  }
462
- // Headings: update rule-ish context, not a candidate.
617
+ // Headings update rule-ish context; not a candidate themselves.
463
618
  const h = HEADING.exec(line);
464
619
  if (h) {
465
- currentHeadingIsRuleish = RULE_HEADING.test(h[2]);
466
- // Anti-context only when it is NOT also rule-ish, so an accept word wins a
467
- // tie (`## Testing conventions` keeps its bullets; `## Testing` drops them).
468
- currentHeadingIsAntiContext =
469
- ANTI_HEADING.test(h[2]) && !currentHeadingIsRuleish;
470
- i++;
471
- continue;
472
- }
473
- // Tables: excluded from candidacy.
474
- if (TABLE_LINE.test(line)) {
620
+ heading = headingStateFrom(h[2]);
475
621
  i++;
476
622
  continue;
477
623
  }
478
- // A line already CONSUMED by the structured-marker pre-pass (a marked
479
- // section's body) is not re-segmented — this is the span-consumption that
480
- // stops a marked rule being double-counted by the heuristic. (1-based.)
481
- if (skipLines?.has(i + 1)) {
624
+ // Tables are excluded; so is a line already CONSUMED by the marker pre-pass
625
+ // (a marked section's body) — the span-consumption that stops a marked rule
626
+ // being double-counted by the heuristic (1-based).
627
+ if (TABLE_LINE.test(line) || skipLines?.has(i + 1)) {
482
628
  i++;
483
629
  continue;
484
630
  }
485
- // List items (with attached continuation lines).
486
631
  const li = LIST_ITEM.exec(line);
487
632
  if (li) {
488
- const markerIndent = li[1].length;
489
- const contentCol = li[1].length + li[2].length + li[3].length;
490
- const startLine = i;
491
- // Gather continuation lines: deeper-indented, non-blank, not a new
492
- // list marker, not a heading, not a fence.
493
- let endLine = i;
494
- let j = i + 1;
495
- while (j < lines.length) {
496
- const cand = lines[j];
497
- if (cand.trim() === "")
498
- break;
499
- if (FENCE.test(cand))
500
- break;
501
- if (HEADING.test(cand))
502
- break;
503
- const indent = cand.length - cand.trimStart().length;
504
- if (indent <= markerIndent)
505
- break;
506
- if (LIST_ITEM.test(cand))
507
- break; // nested/sibling bullet => separate candidate
508
- endLine = j;
509
- j++;
510
- }
511
- const multiLine = endLine > startLine;
512
- const contentStart = lineOffsets[startLine] + contentCol;
513
- const contentEnd = lineOffsets[endLine] + lines[endLine].length;
514
- const contentSpan = { start: contentStart, end: contentEnd };
515
- const wholeText = normalize(markdown.slice(contentStart, contentEnd));
516
- const conf = gate(wholeText, true, currentHeadingIsRuleish);
517
- // Reject bullets under an anti-context heading (Commands/Setup/Key Files/
518
- // Architecture/…) — the corpus's dominant false-positive locus.
519
- if (conf !== null && !currentHeadingIsAntiContext) {
520
- // Only attempt splitting for single-line items (keeps offsets exact).
521
- const spans = multiLine
522
- ? [trimSpan(markdown, contentSpan)]
523
- : atomize(markdown, contentSpan, true, currentHeadingIsRuleish);
524
- if (spans.length === 1) {
525
- // Emit whole item; exactQuote is the full source span incl. marker.
526
- out.push(emitFromSpan(markdown, lineOffsets, file, lineSpan(startLine, endLine), conf));
527
- }
528
- else {
529
- for (const s of spans) {
530
- const text = normalize(markdown.slice(s.start, s.end));
531
- const c = gate(text, true, currentHeadingIsRuleish);
532
- if (c !== null)
533
- out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
534
- }
535
- }
536
- }
537
- i = endLine + 1;
633
+ const r = handleListItem(ctx, i, li, heading);
634
+ out.push(...r.emitted);
635
+ skipped.push(...r.skipped);
636
+ i = r.next;
538
637
  continue;
539
638
  }
540
- // Paragraph block: accumulate until blank / heading / list / fence / table.
541
639
  if (line.trim() !== "") {
542
- const startLine = i;
543
- let endLine = i;
544
- let j = i + 1;
545
- while (j < lines.length) {
546
- const cand = lines[j];
547
- if (cand.trim() === "")
548
- break;
549
- if (FENCE.test(cand))
550
- break;
551
- if (HEADING.test(cand))
552
- break;
553
- if (LIST_ITEM.test(cand))
554
- break;
555
- if (TABLE_LINE.test(cand))
556
- break;
557
- endLine = j;
558
- j++;
559
- }
560
- // Prose is only a candidate under a rule-ish heading.
561
- if (currentHeadingIsRuleish) {
562
- const paraStart = lineOffsets[startLine];
563
- const paraEnd = lineOffsets[endLine] + lines[endLine].length;
564
- const paraText = markdown.slice(paraStart, paraEnd);
565
- // Sentence spans preserving absolute offsets.
566
- const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
567
- let m;
568
- while ((m = re.exec(paraText)) !== null) {
569
- const s = trimSpan(markdown, {
570
- start: paraStart + m.index,
571
- end: paraStart + m.index + m[0].length,
572
- });
573
- if (s.start >= s.end)
574
- continue;
575
- const text = normalize(markdown.slice(s.start, s.end));
576
- const c = gate(text, false, true);
577
- if (c !== null)
578
- out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
579
- }
580
- }
581
- i = endLine + 1;
640
+ const r = handleParagraph(ctx, i, heading);
641
+ out.push(...r.emitted);
642
+ skipped.push(...r.skipped);
643
+ i = r.next;
582
644
  continue;
583
645
  }
584
646
  i++;
585
647
  }
586
- return out;
648
+ return { segments: out, skipped };
587
649
  }
588
650
  //# sourceMappingURL=segment.js.map
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "vigiles",
3
- "version": "14.0.0",
3
+ "version": "14.2.0",
4
4
  "description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
5
5
  "keywords": [
6
6
  "claude-code",