okf-kit 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -101,8 +101,8 @@ const RULE_ID = "citations-resolve";
101
101
  * citation, prose habitually repeats just the line (or range) for a later
102
102
  * reference in the same sentence rather than retyping the path, in three
103
103
  * forms:
104
- * - `` `:N` `` or `` `:N-M` `` -- a bare colon-prefixed line/range
105
- * - `` -`M` `` / `` –`M` `` -- a hyphen- or en-dash-led bare line, the
104
+ * - `` `:N` ``/`` `:N-M` `` -- a bare colon-prefixed line/range
105
+ * - `` -`M` ``/`` –`M` `` -- a hyphen- or en-dash-led bare line, the
106
106
  * tail half of a `` `path:N`-`M` `` split range
107
107
  * - `` (`N`) `` -- a parenthesized bare line
108
108
  * Each of these resolves against `governing`: the nearest *preceding*
@@ -198,6 +198,11 @@ const RULE_ID = "citations-resolve";
198
198
  * nowhere natural to hang an anchor on one without inventing a second,
199
199
  * detached syntax; the migration this rule was built for (CHANGELOG.md
200
200
  * citations) is written as full citations throughout the corpus it targets.
201
+ * Under `--require-anchors`, this remains true (no second anchor syntax is
202
+ * added), but a continuation or short-form that chains off an in-repo full
203
+ * citation is instead flagged, `anchor-required-continuation`, so it does
204
+ * not drift silently -- see the "Anchor strictness (opt-in)" doc block
205
+ * below for that check.
201
206
  *
202
207
  * Rejected alternatives (see the PR/CHANGELOG for the fuller writeup):
203
208
  * - Embedding the literal heading markup itself, e.g.
@@ -216,6 +221,123 @@ const RULE_ID = "citations-resolve";
216
221
  * plus prose) to stay in sync, the exact class of drift this rule
217
222
  * exists to catch.
218
223
  */
224
+ /**
225
+ * Anchor strictness (opt-in, `--require-anchors`, see `src/cli.ts`).
226
+ * Backward compatible by construction: every check in this section is
227
+ * gated on `ctx.requireAnchors` being present (`RequireAnchorsOptions`, see
228
+ * `src/types.ts`), so a consumer that never passes `--require-anchors`
229
+ * sees byte-identical findings to before this section existed. Three new
230
+ * rule ids, all `warning`-severity, mirroring this rule's own "a wrong
231
+ * start/target is strong drift evidence" posture rather than the `notice`
232
+ * posture reserved for the mechanical prose checks with a real
233
+ * false-positive risk (`markdown-range-boundary-bracket-or-fence` and
234
+ * friends):
235
+ *
236
+ * - `anchor-required`: an in-repo full citation (`path:N`/`path:N-M`,
237
+ * resolved to a real target -- an unresolved citation already gets its
238
+ * own `missing-file`/`unresolved-ambiguous` finding, not this one)
239
+ * carrying no `#anchor` at all. Exempted the same way this rule already
240
+ * exempts short-form matching for a reserved citing doc
241
+ * (`doc.isReserved`, e.g. `log.md`'s append-only line-number narration),
242
+ * and additionally by `requireAnchors.allow`: a citedPath matching one
243
+ * of its glob/exact patterns (matched against the citation's raw,
244
+ * as-written `citedPath` text -- see `matchesAllowPattern`) is exempt,
245
+ * for a doc category this check is not (yet) meant to cover, e.g. a
246
+ * README or install guide whose prose is not line-anchored the way a
247
+ * CHANGELOG or source citation is.
248
+ * - `anchor-not-on-last-line`: layered on `checkAnchor`'s existing string
249
+ * branch. An anchor sitting on the FIRST line of a wide range survives a
250
+ * k-line insertion above the range whenever k is smaller than the range
251
+ * itself, because the shifted window (the citation's own line numbers,
252
+ * re-read after the insertion) still contains the original first
253
+ * line's content, just at a different offset inside the window.
254
+ * Anchoring on the LAST line instead closes that: the original last
255
+ * line falls out of the shifted window on any insertion size >= 1, not
256
+ * only large ones.
257
+ * - `anchor-not-unique-in-range`: also layered on the string branch. An
258
+ * anchor text occurring more than once inside its own cited range is
259
+ * ambiguous evidence -- which occurrence is the one actually pinning
260
+ * the citation? A count of zero is unaffected (already
261
+ * `anchor-not-found-in-range`, unconditionally, not double-reported
262
+ * here).
263
+ *
264
+ * Both of the last two are checked only once a match exists at all (count
265
+ * >= 1); uniqueness is checked before last-line placement, since a
266
+ * non-unique anchor is the more fundamental problem (which of the several
267
+ * matching lines "is" the anchor is undefined before asking whether the
268
+ * one occurrence is on the right line).
269
+ *
270
+ * A fourth check, `test-range-straddles-block` (warning), applies to a FULL
271
+ * citation's own range into a `.test.`/`.spec.` target (`.ts`, `.js`,
272
+ * `.mjs`) under this same opt-in (see `checkTestRangeStraddle`, called
273
+ * from `checkFullTarget`): a
274
+ * full citation's range is expected to stay inside a single `describe`/
275
+ * `it`/`test` block once it starts, so any block-head line
276
+ * (`describe(`/`describe.only(`/`describe.skip(`/`describe.each(`/`it(`/
277
+ * `it.only(`/`it.skip(`/`it.each(`/`test(`/`test.only(`/`test.skip(`/
278
+ * `test.each(`, see `TEST_BLOCK_HEAD_RE`) found on any line of the range
279
+ * OTHER than its own first line means the citation ran into a sibling or
280
+ * outer block (or never really started on one). A range that leaves its
281
+ * block without a later head line inside it (ending on an outer block's
282
+ * closing line) is not detected by this line-based check. The range's start
283
+ * line is never itself checked here: it is either a legitimate block head
284
+ * (the range correctly starts a block) or legitimately inside a block's
285
+ * body (a partial citation into the middle of a block), and both are fine
286
+ * -- only a head line reappearing *after* the start is evidence of drift.
287
+ * Deliberately its own rule id rather than reusing `checkRangeBoundary`'s
288
+ * `test-range-start-not-head`/`test-range-end-not-closing` (which stay
289
+ * short-form-only, see `checkShortFormTarget`): those two require the range
290
+ * to start exactly on a head line and end exactly on the matching closing
291
+ * `});`, which is far stricter than a full citation's legitimate use
292
+ * (citing an arbitrary sub-range once a paragraph has already named the
293
+ * file) needs -- the straddle check only cares whether the range crossed a
294
+ * block boundary, not whether it captured a whole block precisely.
295
+ * Deliberately scoped to test-file targets only, same reasoning
296
+ * `checkShortFormTarget`'s own doc comment already gives for the Markdown
297
+ * half of the block-boundary check: a full citation into a Markdown target
298
+ * legitimately cites a couple of arbitrary lines all the time.
299
+ *
300
+ * A fifth check, `anchor-required-continuation` (warning), closes a gap the
301
+ * four checks above leave open: every one of them is gated on reaching a
302
+ * "full" atom, so a continuation (` :N`/`:N-M` `, `` -`M` ``/`` –`M` ``,
303
+ * `` (`N`) ``) or a paragraph-bound short-form `:N-M` chained off an
304
+ * ALREADY-ANCHORED full citation was previously invisible to
305
+ * `--require-anchors` entirely -- it carries no path of its own
306
+ * (see the "Continuation citations" / "Short-form citations" doc blocks),
307
+ * so it was never itself an in-repo full citation `anchor-required` could
308
+ * reach, and the anchor on the governing citation it chains off does not
309
+ * (and structurally cannot) cover it: a line-shift that only affects the
310
+ * continuation's own start/end line is exactly the drift this rule's base
311
+ * checks (blank-start-line and friends) already catch when it lands on
312
+ * unusable content, but a shift that lands the continuation on still
313
+ * non-blank, in-bounds, unrelated content is invisible to any of them --
314
+ * see `pushAnchorRequiredContinuation` for where this is pushed, from all
315
+ * three places a continuation or short-form atom is processed in `scanDoc`.
316
+ * Unlike `anchor-required`, there is no "already carries one" escape here:
317
+ * a continuation is structurally anchor-less regardless of whether its
318
+ * governing full citation has an anchor, so this fires unconditionally
319
+ * once the same exemptions (reserved citing doc, `requireAnchors.allow`
320
+ * matched against the GOVERNING citation's citedPath) clear. The remedy is
321
+ * always the same: lift the continuation into its own full,
322
+ * `path:N-M#anchor` citation.
323
+ */
324
+ /**
325
+ * True when `citedPath` (the citation's raw, as-written path text) matches
326
+ * one of `patterns`: an exact string match, or a glob with `*` (any run of
327
+ * characters) and `?` (any single character) as the only wildcards -- the
328
+ * minimal shape needed for `requireAnchors.allow`'s own documented example
329
+ * (`["README.md", "INSTALL-AGENT.md"]`), not a general-purpose glob engine.
330
+ */
331
+ function matchesAllowPattern(citedPath, patterns) {
332
+ return patterns.some((pattern) => {
333
+ if (!pattern.includes("*") && !pattern.includes("?")) {
334
+ return pattern === citedPath;
335
+ }
336
+ const escaped = pattern.replace(/[.+^${}()|[\]\\]/g, "\\$&");
337
+ const reSource = "^" + escaped.replace(/\*/g, ".*").replace(/\?/g, ".") + "$";
338
+ return new RegExp(reSource).test(citedPath);
339
+ });
340
+ }
219
341
  const ANCHOR_HEADING_MAX_LEVEL = 2;
220
342
  const MD_HEADING_RE = /^(#{1,6})\s+(.*)$/;
221
343
  /**
@@ -335,22 +457,73 @@ function findEnclosingHeading(lines, startLine, fencedLines) {
335
457
  }
336
458
  return null;
337
459
  }
460
+ // A trimmed line composed of nothing but closing brackets/braces/parens
461
+ // plus an optional trailing `,`/`;` -- bare closing boilerplate like `}`,
462
+ // `});`, `]);`, `}),`. Used only by `lastContentLineInRange` (opt-in
463
+ // `anchor-not-on-last-line`, see there): a range that ends on such a line
464
+ // (the common shape of a `});` that closes a whole cited `describe`/`it`
465
+ // block) should not force the anchor onto that boilerplate line itself --
466
+ // the last CONTENT line before it is the meaningful place for an anchor
467
+ // to live.
468
+ const CLOSING_BOILERPLATE_RE = /^[\]\)\};,]*$/;
469
+ function isContentLine(text) {
470
+ const trimmed = text.trim();
471
+ return trimmed !== "" && !CLOSING_BOILERPLATE_RE.test(trimmed);
472
+ }
473
+ /**
474
+ * The last line number (1-based, within `[startLine, endLine]`) whose text
475
+ * is a "content" line -- see `isContentLine`. Walks backward from `endLine`
476
+ * so a range ending on one or more bare closing-boilerplate lines (`});`,
477
+ * `]);`, `}),`, a lone `}`) resolves to the real content line before them
478
+ * instead of the boilerplate itself. Falls back to `endLine` unchanged when
479
+ * every line in the range is boilerplate or blank (nothing better to
480
+ * anchor against).
481
+ */
482
+ function lastContentLineInRange(lines, startLine, endLine) {
483
+ for (let ln = endLine; ln >= startLine; ln--) {
484
+ if (isContentLine(lines[ln - 1] ?? ""))
485
+ return ln;
486
+ }
487
+ return endLine;
488
+ }
338
489
  /**
339
490
  * Checks an anchored full citation's anchor against its already-resolved,
340
491
  * already-range-checked target -- see the "Anchored citations" doc block
341
492
  * above for the two forms' semantics. `endLine` is the citation's end line,
342
493
  * or its start line for a single-line citation (a size-1 range).
343
494
  */
344
- function checkAnchor(anchor, startLine, endLine, lines) {
495
+ function checkAnchor(anchor, startLine, endLine, lines, requireAnchors) {
345
496
  if (anchor.kind === "string") {
497
+ let count = 0;
498
+ let lastMatchLine = -1;
346
499
  for (let i = startLine - 1; i <= endLine - 1 && i < lines.length; i++) {
347
- if ((lines[i] ?? "").includes(anchor.text))
348
- return null;
500
+ if ((lines[i] ?? "").includes(anchor.text)) {
501
+ count++;
502
+ lastMatchLine = i + 1;
503
+ }
349
504
  }
350
- return {
351
- rule: "anchor-not-found-in-range",
352
- message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
353
- };
505
+ if (count === 0) {
506
+ return {
507
+ rule: "anchor-not-found-in-range",
508
+ message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
509
+ };
510
+ }
511
+ if (requireAnchors) {
512
+ if (count > 1) {
513
+ return {
514
+ rule: "anchor-not-unique-in-range",
515
+ message: `anchor "${anchor.text}" occurs on ${count} lines of the cited range (${startLine}-${endLine}); expected exactly one`,
516
+ };
517
+ }
518
+ const lastContentLine = lastContentLineInRange(lines, startLine, endLine);
519
+ if (lastMatchLine !== lastContentLine) {
520
+ return {
521
+ rule: "anchor-not-on-last-line",
522
+ message: `anchor "${anchor.text}" occurs on line ${lastMatchLine}, not the cited range's last content line (${lastContentLine})`,
523
+ };
524
+ }
525
+ }
526
+ return null;
354
527
  }
355
528
  const fencedLines = computeFencedLineIndices(lines);
356
529
  const heading = findEnclosingHeading(lines, startLine, fencedLines);
@@ -379,7 +552,90 @@ function checkAnchor(anchor, startLine, endLine, lines) {
379
552
  }
380
553
  return null;
381
554
  }
382
- const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
555
+ // Exported so `prose-line-references` (see src/rules/prose-line-references.ts)
556
+ // can exclude a real citations-resolve citation span from its own
557
+ // extraction, rather than re-deriving this grammar itself.
558
+ export const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
559
+ /**
560
+ * Heading-section citations. A CHANGELOG.md that grows by insertion at the
561
+ * top forces every later `path:N-M#anchor` citation to be re-pointed on
562
+ * every release, even with the heading anchor above closing the "wrong
563
+ * section" gap -- the line RANGE itself still drifts on every insertion,
564
+ * so the citation still needs editing, just not silently mis-validated.
565
+ * `path:#heading` sidesteps line numbers entirely: it names a heading
566
+ * (matched the same way the line-range anchor above already matches one,
567
+ * `heading.text.includes(...)`, reused rather than re-invented -- see
568
+ * `parseAnchor`/`findHeadingSection`) and resolves to that heading's own
569
+ * section (from the line after the heading up to, but not including, the
570
+ * next heading of the same or shallower level, or EOF) -- see
571
+ * `findHeadingSection`. The section must exist (`heading-section-not-found`
572
+ * when no such heading is found), must be unambiguous (more than one
573
+ * matching heading is `heading-section-ambiguous`, never silently the
574
+ * first hit -- the same posture `unresolved-ambiguous` already takes for
575
+ * path resolution), and must not be empty (`heading-section-empty`).
576
+ *
577
+ * Deliberately reuses `ANCHOR_HEADING_MAX_LEVEL` (2), not a second cap: the
578
+ * same Keep-a-Changelog nesting that motivates the line-range anchor's cap
579
+ * (`## [x.y.z]` release headings around identically-named, per-release
580
+ * `### Added`/`### Changed`/`### Fixed` subsections) means a level-3+
581
+ * heading name is *never* unique across a real CHANGELOG anyway --
582
+ * matching it would just turn every such citation into a guaranteed
583
+ * `heading-section-ambiguous`. Capping the search at level 2 keeps this
584
+ * form usable for exactly the case it exists for (a release section) and
585
+ * behaves identically to the line-range anchor for the same reason.
586
+ *
587
+ * Optional content anchor: `path:#heading#"text"`, always the quoted form
588
+ * (a section's body is prose/content, not a second heading to look up) --
589
+ * the text must occur on EXACTLY one line inside the resolved section
590
+ * (`heading-section-content-anchor-not-found` / `-ambiguous`), stricter
591
+ * than the line-range string anchor's "at least one line" (`checkAnchor`'s
592
+ * string branch): a whole section is a much larger haystack than a
593
+ * caller-chosen line range, so "present somewhere" is far weaker evidence
594
+ * the anchor still points at the intended spot, and "found on exactly one
595
+ * line" is what the task this form was built for (pinning a citation to
596
+ * one prose sentence inside a growing section) actually needs. The heading
597
+ * position itself keeps only the bare/bracketed form (`#heading`/
598
+ * `#[heading]`); the quoted alternative is reserved for the content anchor
599
+ * position so the two can never be confused by shape alone.
600
+ *
601
+ * Backtick-delimited, unlike the line-range form above (which does not
602
+ * require backticks -- see the "Anchor syntax note" in the README). This
603
+ * is a deliberate, narrower grammar for this form specifically: a bare
604
+ * `path#heading` (the grammar this form used in an earlier round) is
605
+ * indistinguishable from an ordinary Markdown link's target, which
606
+ * commonly has exactly that shape (`[install](docs/README.md#install)`)
607
+ * -- unlike the line-range form (gated by a `:N` no ordinary link ever
608
+ * contains), there was no character available to tell a real citation
609
+ * apart from a relative link's href written in prose, and measuring
610
+ * against real bundles (see the CHANGELOG entry that introduced the
611
+ * `:#` form) showed the collision is not hypothetical: existing docs write
612
+ * `` `path#heading` `` as inline prose pointing at an unrelated anchor,
613
+ * each producing a spurious `heading-section-not-found`. The colon
614
+ * (`path:#heading`) reuses this rule's existing citation signature (a
615
+ * literal `:` no Markdown link fragment ever contains) instead of relying
616
+ * on backticks alone to do that job, and is additionally restricted to a
617
+ * `.md` target: a link fragment's target is always the Markdown doc it
618
+ * points into, never a source or config file, so a non-`.md` path with
619
+ * this shape is never a live citation. Requiring the whole citation
620
+ * inside one pair of backticks stays in place as well, matching this
621
+ * rule's own general recommendation for the line-range form.
622
+ */
623
+ const HEADING_SECTION_CITATION_RE = /`([\w./-]+\.md):#(\[?\w(?:[\w.-]*\w)?\]?)(?:#("[^"\n`]+"))?`/g;
624
+ /**
625
+ * Companion to `HEADING_SECTION_CITATION_RE`: matches the same
626
+ * backtick + path + `:#` opener but accepts anything up to the closing
627
+ * backtick, so a heading-section citation that fails to parse (an
628
+ * unterminated or empty content-anchor quote, an unquoted third segment,
629
+ * or a non-`.md` target, which includes a non-lowercase `.MD` extension) is
630
+ * still recognised as an *attempt* rather than
631
+ * silently vanishing -- mirrors `anchor-malformed`'s
632
+ * "still-visible-as-a-notice" posture for the line-range anchor form (see
633
+ * `extractMalformedAnchorRaw`). Only used for a match whose span does not
634
+ * already overlap a successful `HEADING_SECTION_CITATION_RE` match (see
635
+ * `collectHeadingSectionMatches`); a well-formed citation never also
636
+ * produces a malformed notice.
637
+ */
638
+ const HEADING_SECTION_MALFORMED_RE = /`([\w./-]+):#([^`\n]*)`/g;
383
639
  // Continuation citation forms (see the "Continuation citations" doc block
384
640
  // above). Each requires the backtick delimiter as part of the match so it
385
641
  // can never overlap a CITATION_RE match: a full citation's regex match
@@ -404,6 +660,14 @@ const SHORT_FORM_COLON_RE = /:(\d+)-(\d+)/g;
404
660
  const TEST_FILE_RE = /\.(test|spec)\.(ts|js|mjs)$/i;
405
661
  const TEST_HEAD_LINE_RE = /^\s*(?:describe|it)\s*\(/;
406
662
  const TEST_CLOSING_LINE_RE = /^\s*\}\)\s*;\s*$/;
663
+ // Block-head detection for test-range-straddles-block (see the doc block
664
+ // above and checkTestRangeStraddle below): a wider set of head shapes than
665
+ // TEST_HEAD_LINE_RE (which stays short-form-only, unchanged, via
666
+ // checkRangeBoundary/checkShortFormTarget) -- describe/it/test, each with
667
+ // an optional .only/.skip/.each modifier. A leading `export `/`async ` is
668
+ // deliberately not handled: not needed for any bundle this rule has been
669
+ // measured against.
670
+ const TEST_BLOCK_HEAD_RE = /^\s*(?:describe|it|test)(?:\.(?:only|skip|each))?\s*\(/;
407
671
  // Markdown block-boundary check (see checkRangeBoundary): a range boundary
408
672
  // line that is nothing but a bracket (open or close), optionally with a
409
673
  // trailing `,`/`;`, is always a drift signal. A bare code-fence delimiter is
@@ -434,8 +698,13 @@ function isFile(p) {
434
698
  return false;
435
699
  }
436
700
  }
437
- /** True when citedPath has a literal `..` path segment. */
438
- function hasParentSegment(citedPath) {
701
+ /**
702
+ * True when citedPath has a literal `..` path segment. Exported so
703
+ * `prose-line-references` rejects the same shape before ever calling
704
+ * `resolveCitation` on a file-mention token, matching this rule's own
705
+ * `path-traversal-rejected` posture.
706
+ */
707
+ export function hasParentSegment(citedPath) {
439
708
  return citedPath.split("/").includes("..");
440
709
  }
441
710
  /**
@@ -538,8 +807,14 @@ function resolveViaAncestorClimb(root, docAbsPath, citedPath) {
538
807
  * plausible target exists, or `null` when nothing matches. Callers must
539
808
  * reject a citedPath with a `..` segment (see hasParentSegment) before
540
809
  * calling this; it is not re-checked here.
810
+ *
811
+ * Exported so `prose-line-references` (see
812
+ * src/rules/prose-line-references.ts) reuses this exact resolution order
813
+ * for binding a bare prose line reference to the file mention nearest it,
814
+ * rather than re-implementing (and risking drifting from) this rule's own
815
+ * path-resolution rules.
541
816
  */
542
- function resolveCitation(cache, root, docAbsPath, docContent, docSources, citedPath, matchIndex) {
817
+ export function resolveCitation(cache, root, docAbsPath, docContent, docSources, citedPath, matchIndex) {
543
818
  if (citedPath.startsWith("/")) {
544
819
  return { skip: true };
545
820
  }
@@ -679,17 +954,215 @@ function checkTarget(citedPath, startLine, endLine, resolvedPath) {
679
954
  * are full-citation-only (see that doc block for why), so `checkTarget`
680
955
  * itself is untouched and still used as-is for a cont-fresh atom.
681
956
  */
682
- function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor) {
957
+ function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor, requireAnchors) {
683
958
  const read = readTarget(resolvedPath);
684
959
  if ("rule" in read)
685
- return read;
960
+ return [read];
686
961
  const lines = splitLines(read.content);
687
962
  const base = checkTargetLines(citedPath, startLine, endLine, lines);
688
963
  if (base)
689
- return base;
690
- if (!anchor)
691
- return null;
692
- return checkAnchor(anchor, startLine, endLine ?? startLine, lines);
964
+ return [base];
965
+ const problems = [];
966
+ // test-range-straddles-block, opt-in only (see "Anchor strictness
967
+ // (opt-in)" above and checkTestRangeStraddle): a full citation's own
968
+ // range into a .test./.spec. target (.ts, .js, .mjs) must not cross
969
+ // into another block's head after its own first line. Deliberately
970
+ // scoped to test-file targets only (isTestFile), not also a markdown
971
+ // branch: a
972
+ // full citation into a markdown target legitimately cites a couple of
973
+ // arbitrary lines all the time (e.g. two lines of a code fence example),
974
+ // the exact reasoning checkShortFormTarget's own doc comment already
975
+ // gives for why the sibling block-boundary check was scoped to
976
+ // short-form in the first place. Only meaningful for an actual range
977
+ // (endLine !== null); a single-line citation trivially starts and ends
978
+ // on the same line and has nothing to straddle.
979
+ if (requireAnchors && endLine !== null && isTestFile(citedPath)) {
980
+ const straddle = checkTestRangeStraddle(startLine, endLine, lines);
981
+ if (straddle)
982
+ problems.push(straddle);
983
+ }
984
+ // The anchor check is run independently of the straddle check above
985
+ // (not gated on it having come back clean): a straddling range and a
986
+ // missing/misplaced anchor are two independent problems with the same
987
+ // citation, and reporting only the first one found would silently drop
988
+ // the other from the output every time both happen to co-occur.
989
+ if (anchor) {
990
+ const anchorProblem = checkAnchor(anchor, startLine, endLine ?? startLine, lines, requireAnchors);
991
+ if (anchorProblem)
992
+ problems.push(anchorProblem);
993
+ }
994
+ return problems;
995
+ }
996
+ /**
997
+ * Every Markdown heading in `lines` up to `ANCHOR_HEADING_MAX_LEVEL`, in
998
+ * document order -- the whole-document counterpart of `findEnclosingHeading`
999
+ * (which stops at the nearest heading at or before a given line): a
1000
+ * heading-section citation has no start line to search backward from, it
1001
+ * needs every candidate to test for a unique match. `fencedLines` (see
1002
+ * `computeFencedLineIndices`) excludes a `#`-led line inside a fenced
1003
+ * example, same as `findEnclosingHeading`.
1004
+ */
1005
+ function collectHeadingsUpToLevel(lines, fencedLines) {
1006
+ const headings = [];
1007
+ for (let i = 0; i < lines.length; i++) {
1008
+ if (fencedLines.has(i))
1009
+ continue;
1010
+ const m = (lines[i] ?? "").match(MD_HEADING_RE);
1011
+ if (m && m[1].length <= ANCHOR_HEADING_MAX_LEVEL) {
1012
+ headings.push({ level: m[1].length, text: m[2].trim(), lineNo: i + 1 });
1013
+ }
1014
+ }
1015
+ return headings;
1016
+ }
1017
+ /**
1018
+ * Resolves a heading-section citation's heading text to a single section:
1019
+ * every level <= `ANCHOR_HEADING_MAX_LEVEL` heading whose text contains
1020
+ * `headingText` (same containment check `checkAnchor`'s heading branch
1021
+ * already uses -- reused, not reinvented) is a candidate; zero is
1022
+ * `not-found`, more than one is `ambiguous` (never silently the first
1023
+ * match), exactly one resolves to the section running from the line right
1024
+ * after the heading up to (not including) the next heading at or above the
1025
+ * same level, or EOF -- the same enclosure boundary `checkAnchor`'s heading
1026
+ * branch walks forward to check, just producing a body range here instead
1027
+ * of a pass/fail against an already-known end line.
1028
+ */
1029
+ function findHeadingSection(lines, fencedLines, headingText) {
1030
+ const headings = collectHeadingsUpToLevel(lines, fencedLines);
1031
+ const matches = headings.filter((h) => h.text.includes(headingText));
1032
+ if (matches.length === 0)
1033
+ return { kind: "not-found" };
1034
+ if (matches.length > 1)
1035
+ return { kind: "ambiguous", matches };
1036
+ const heading = matches[0];
1037
+ let bodyEnd = lines.length;
1038
+ for (let i = heading.lineNo; i < lines.length; i++) {
1039
+ if (fencedLines.has(i))
1040
+ continue;
1041
+ const m = (lines[i] ?? "").match(MD_HEADING_RE);
1042
+ if (m && m[1].length <= heading.level) {
1043
+ bodyEnd = i;
1044
+ break;
1045
+ }
1046
+ }
1047
+ return { kind: "found", heading, bodyStart: heading.lineNo, bodyEnd };
1048
+ }
1049
+ /** True when every line in `[bodyStart, bodyEnd)` is empty or whitespace-only. */
1050
+ function isSectionEmpty(lines, bodyStart, bodyEnd) {
1051
+ for (let i = bodyStart; i < bodyEnd; i++) {
1052
+ if ((lines[i] ?? "").trim() !== "")
1053
+ return false;
1054
+ }
1055
+ return true;
1056
+ }
1057
+ /**
1058
+ * Count of lines in `[bodyStart, bodyEnd)` containing `text` -- a heading
1059
+ * section's content anchor must occur on exactly one such line (see the
1060
+ * "Heading-section citations" doc block above for why this is stricter
1061
+ * than the line-range string anchor's "at least one line").
1062
+ */
1063
+ function countAnchorOccurrences(lines, bodyStart, bodyEnd, text) {
1064
+ let count = 0;
1065
+ for (let i = bodyStart; i < bodyEnd; i++) {
1066
+ if ((lines[i] ?? "").includes(text))
1067
+ count++;
1068
+ }
1069
+ return count;
1070
+ }
1071
+ /**
1072
+ * A heading-section citation's complete target check (see the
1073
+ * "Heading-section citations" doc block above): the named heading must
1074
+ * exist exactly once, its section must be non-empty, and, when a content
1075
+ * anchor was given, it must occur on exactly one line inside that section.
1076
+ * One problem per citation, checked in that order, matching every other
1077
+ * check in this file's "base checks first" pattern.
1078
+ */
1079
+ function checkHeadingSectionTarget(headingAnchor, contentAnchor, resolvedPath) {
1080
+ const read = readTarget(resolvedPath);
1081
+ if ("rule" in read)
1082
+ return read;
1083
+ const lines = splitLines(read.content);
1084
+ const fencedLines = computeFencedLineIndices(lines);
1085
+ const section = findHeadingSection(lines, fencedLines, headingAnchor.text);
1086
+ if (section.kind === "not-found") {
1087
+ return {
1088
+ rule: "heading-section-not-found",
1089
+ message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) contains "${headingAnchor.text}"`,
1090
+ };
1091
+ }
1092
+ if (section.kind === "ambiguous") {
1093
+ return {
1094
+ rule: "heading-section-ambiguous",
1095
+ message: `${section.matches.length} headings contain "${headingAnchor.text}" (lines ${section.matches
1096
+ .map((m) => m.lineNo)
1097
+ .join(", ")}); not evaluated`,
1098
+ };
1099
+ }
1100
+ if (isSectionEmpty(lines, section.bodyStart, section.bodyEnd)) {
1101
+ return {
1102
+ rule: "heading-section-empty",
1103
+ message: `section under heading "${section.heading.text}" (line ${section.heading.lineNo}) has no non-blank content before the next heading`,
1104
+ };
1105
+ }
1106
+ if (contentAnchor) {
1107
+ const count = countAnchorOccurrences(lines, section.bodyStart, section.bodyEnd, contentAnchor.text);
1108
+ if (count === 0) {
1109
+ return {
1110
+ rule: "heading-section-content-anchor-not-found",
1111
+ message: `content anchor "${contentAnchor.text}" does not occur in the section under heading "${section.heading.text}"`,
1112
+ };
1113
+ }
1114
+ if (count > 1) {
1115
+ return {
1116
+ rule: "heading-section-content-anchor-ambiguous",
1117
+ message: `content anchor "${contentAnchor.text}" occurs on ${count} lines in the section under heading "${section.heading.text}"; expected exactly one`,
1118
+ };
1119
+ }
1120
+ }
1121
+ return null;
1122
+ }
1123
+ /**
1124
+ * Every well-formed heading-section citation in `content`, in document
1125
+ * order. Collected once, up front (before `CITATION_RE`'s own scan in
1126
+ * `scanDoc`), so its match spans can gate both `CITATION_RE` (a full
1127
+ * citation never fires inside a heading-section citation's own quoted
1128
+ * content anchor, see the "Heading-section citations" doc block above) and
1129
+ * the malformed companion scan (see `collectHeadingSectionMalformedMatches`)
1130
+ * without re-deriving the same spans twice.
1131
+ */
1132
+ function collectHeadingSectionMatches(content) {
1133
+ const out = [];
1134
+ const re = new RegExp(HEADING_SECTION_CITATION_RE.source, HEADING_SECTION_CITATION_RE.flags);
1135
+ let m;
1136
+ while ((m = re.exec(content)) !== null) {
1137
+ out.push({
1138
+ index: m.index,
1139
+ end: m.index + m[0].length,
1140
+ citedPath: m[1],
1141
+ headingAnchor: parseAnchor(m[2]), // group 2 is mandatory
1142
+ contentAnchor: parseAnchor(m[3]),
1143
+ });
1144
+ }
1145
+ return out;
1146
+ }
1147
+ /**
1148
+ * Every backtick + path + `:#` opener in `content` that did NOT parse as a
1149
+ * well-formed heading-section citation (see `HEADING_SECTION_MALFORMED_RE`'s
1150
+ * doc comment) -- an unterminated content-anchor quote, an unquoted third
1151
+ * segment, or a non-`.md` target. `wellFormedSpans` (from
1152
+ * `collectHeadingSectionMatches`) gates out any match that is really just
1153
+ * the successful citation seen from the outside; a well-formed citation
1154
+ * never also produces a malformed notice.
1155
+ */
1156
+ function collectHeadingSectionMalformedMatches(content, wellFormedSpans) {
1157
+ const out = [];
1158
+ const re = new RegExp(HEADING_SECTION_MALFORMED_RE.source, "g");
1159
+ let m;
1160
+ while ((m = re.exec(content)) !== null) {
1161
+ if (isWithinAnySpan(m.index, wellFormedSpans))
1162
+ continue;
1163
+ out.push({ index: m.index, end: m.index + m[0].length, raw: m[0] });
1164
+ }
1165
+ return out;
693
1166
  }
694
1167
  // A cont-ext atom only ever extends the *end* of a range whose start line
695
1168
  // was already fully checked (blank / closing-brace) when it was cited as
@@ -839,6 +1312,53 @@ function checkRangeBoundary(citedPath, startLine, endLine, lines) {
839
1312
  }
840
1313
  return null;
841
1314
  }
1315
+ /**
1316
+ * Width (in characters) of `text`'s leading run of spaces/tabs -- used by
1317
+ * `checkTestRangeStraddle` to tell a NESTED block head (indented deeper
1318
+ * than the range's own start line) apart from a SIBLING or OUTER one
1319
+ * (indented the same or shallower): citing a whole `describe` block
1320
+ * necessarily contains every `it(`/nested-`describe(` head inside its own
1321
+ * body, and those are not straddling anywhere, they are exactly what the
1322
+ * citation is about.
1323
+ */
1324
+ function leadingWhitespaceWidth(text) {
1325
+ return (text.match(/^[ \t]*/) ?? [""])[0].length;
1326
+ }
1327
+ /**
1328
+ * `test-range-straddles-block` (opt-in, warning): a FULL citation's own
1329
+ * range into a `.test.`/`.spec.` target (`.ts`, `.js`, `.mjs`), see the
1330
+ * "Anchor strictness (opt-in)" doc block above for the full rule text.
1331
+ * Checks every line of `[startLine, endLine]` EXCEPT the range's own
1332
+ * first line (`startLine`
1333
+ * itself is never checked -- see that doc block for why) against
1334
+ * `TEST_BLOCK_HEAD_RE`; the first hit (in document order) is reported,
1335
+ * matching this file's "one problem per citation" pattern. A block-head
1336
+ * line indented STRICTLY DEEPER than the range's own start line is a
1337
+ * NESTED block (e.g. the `it(`s inside a `describe(` the range cites in
1338
+ * full) and is skipped, not reported: a citation covering a whole block is
1339
+ * expected to contain every head line nested inside it. A block-head line
1340
+ * at the same or a shallower indent is a SIBLING or OUTER block and is
1341
+ * still reported -- this is an approximation of "which block scope is
1342
+ * this line lexically in" using indentation instead of a real AST, chosen
1343
+ * because it reproduces the AST-based reference verdict on every straddle
1344
+ * finding this rule has been measured against; a file that mixes tabs and
1345
+ * spaces, or that does not indent nested blocks at all, can defeat it.
1346
+ */
1347
+ function checkTestRangeStraddle(startLine, endLine, lines) {
1348
+ const startIndent = leadingWhitespaceWidth(lines[startLine - 1] ?? "");
1349
+ for (let i = startLine; i <= endLine - 1 && i < lines.length; i++) {
1350
+ const text = lines[i] ?? "";
1351
+ if (!TEST_BLOCK_HEAD_RE.test(text))
1352
+ continue;
1353
+ if (leadingWhitespaceWidth(text) > startIndent)
1354
+ continue; // nested block
1355
+ return {
1356
+ rule: "test-range-straddles-block",
1357
+ message: `range straddles into another block's head at line ${i + 1} ("${text.trim()}")`,
1358
+ };
1359
+ }
1360
+ return null;
1361
+ }
842
1362
  /**
843
1363
  * A short-form citation's full check: checkTarget's existing checks
844
1364
  * (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
@@ -891,6 +1411,7 @@ function collectContinuationAtoms(content) {
891
1411
  kind: "cont-ext",
892
1412
  index: m.index,
893
1413
  value: m[2] ? Number(m[2]) : Number(m[1]),
1414
+ raw: m[0],
894
1415
  });
895
1416
  }
896
1417
  else {
@@ -899,12 +1420,18 @@ function collectContinuationAtoms(content) {
899
1420
  index: m.index,
900
1421
  startLine: Number(m[1]),
901
1422
  endLine: m[2] ? Number(m[2]) : null,
1423
+ raw: m[0],
902
1424
  });
903
1425
  }
904
1426
  }
905
1427
  const dashRe = new RegExp(CONT_DASH_RE.source, "g");
906
1428
  while ((m = dashRe.exec(content)) !== null) {
907
- atoms.push({ kind: "cont-ext", index: m.index, value: Number(m[1]) });
1429
+ atoms.push({
1430
+ kind: "cont-ext",
1431
+ index: m.index,
1432
+ value: Number(m[1]),
1433
+ raw: m[0],
1434
+ });
908
1435
  }
909
1436
  const parenRe = new RegExp(CONT_PAREN_RE.source, "g");
910
1437
  while ((m = parenRe.exec(content)) !== null) {
@@ -913,6 +1440,7 @@ function collectContinuationAtoms(content) {
913
1440
  index: m.index,
914
1441
  startLine: Number(m[1]),
915
1442
  endLine: null,
1443
+ raw: m[0],
916
1444
  });
917
1445
  }
918
1446
  return atoms;
@@ -966,8 +1494,16 @@ function collectShortFormMatches(content, excludedSpans) {
966
1494
  * `scanFenceLines` (see there): per-line fenced/opens/closes state is
967
1495
  * converted to char-offset spans by tracking each line's `[start, end)`
968
1496
  * offset in `content` alongside it.
1497
+ *
1498
+ * Exported (with computeIndentedCodeSpans and computeTableRowSpans below)
1499
+ * so `prose-line-references` can build its OWN excluded-span set for file
1500
+ * mentions that leaves out computeInlineCodeSpans -- a bare backtick-
1501
+ * wrapped filename (`` `src/cli.ts` ``) is the normal, encouraged way to
1502
+ * write a file mention in prose, unlike a short-form bare number, so it
1503
+ * must NOT be excluded from mention detection the way it is excluded from
1504
+ * short-form citation matching here.
969
1505
  */
970
- function computeFencedSpans(content) {
1506
+ export function computeFencedSpans(content) {
971
1507
  const spans = [];
972
1508
  const lines = content.split("\n");
973
1509
  const states = scanFenceLines(lines);
@@ -998,7 +1534,7 @@ function computeFencedSpans(content) {
998
1534
  * spec (no list-item-context awareness); adequate for this mechanical,
999
1535
  * warn-only rule.
1000
1536
  */
1001
- function computeIndentedCodeSpans(content) {
1537
+ export function computeIndentedCodeSpans(content) {
1002
1538
  const spans = [];
1003
1539
  const lines = content.split("\n");
1004
1540
  let offset = 0;
@@ -1069,7 +1605,7 @@ function computeInlineCodeSpans(content) {
1069
1605
  * and can otherwise carry a range shape the gate would not reject (e.g.
1070
1606
  * `| col (5-9) |`).
1071
1607
  */
1072
- function computeTableRowSpans(content) {
1608
+ export function computeTableRowSpans(content) {
1073
1609
  const spans = [];
1074
1610
  const lines = content.split("\n");
1075
1611
  let offset = 0;
@@ -1090,8 +1626,13 @@ function computeTableRowSpans(content) {
1090
1626
  * per doc and combined with fullSpans (see scanDoc) via the existing
1091
1627
  * isWithinAnySpan helper -- the same mechanism a full citation's own span
1092
1628
  * already uses, not a second one.
1629
+ *
1630
+ * Exported so `prose-line-references` excludes the identical set of spans
1631
+ * from its own extraction (a code-fenced or inline-code "line N" is a code
1632
+ * example, not a live prose reference), rather than maintaining a second,
1633
+ * possibly-drifting copy of "what counts as excluded prose".
1093
1634
  */
1094
- function computeExcludedSpans(content) {
1635
+ export function computeExcludedSpans(content) {
1095
1636
  return [
1096
1637
  ...computeFencedSpans(content),
1097
1638
  ...computeIndentedCodeSpans(content),
@@ -1102,8 +1643,13 @@ function computeExcludedSpans(content) {
1102
1643
  /**
1103
1644
  * Paragraph-start offsets in `content`, ascending, always including 0. A
1104
1645
  * paragraph boundary is a blank (empty or whitespace-only) line.
1646
+ *
1647
+ * Exported (with paragraphStartFor below) so `prose-line-references` binds
1648
+ * against the same notion of "paragraph" this rule already uses for
1649
+ * short-form citation binding, instead of a second, possibly-inconsistent
1650
+ * definition.
1105
1651
  */
1106
- function computeParagraphStarts(content) {
1652
+ export function computeParagraphStarts(content) {
1107
1653
  const starts = [0];
1108
1654
  const re = /\n[ \t]*\n+/g;
1109
1655
  let m;
@@ -1113,7 +1659,7 @@ function computeParagraphStarts(content) {
1113
1659
  return starts;
1114
1660
  }
1115
1661
  /** The start offset of the paragraph containing `index` (see computeParagraphStarts). */
1116
- function paragraphStartFor(starts, index) {
1662
+ export function paragraphStartFor(starts, index) {
1117
1663
  let result = starts[0];
1118
1664
  for (const s of starts) {
1119
1665
  if (s > index)
@@ -1164,6 +1710,40 @@ function pushUnreadable(findings, file, citation, resolvedTo, code) {
1164
1710
  detail: `resolvedTo: ${resolvedTo}, errorCode: ${code}`,
1165
1711
  });
1166
1712
  }
1713
+ /**
1714
+ * `anchor-required-continuation` (opt-in, `--require-anchors`, warning):
1715
+ * see the "Anchor strictness (opt-in)" doc block above. A continuation
1716
+ * atom (cont-fresh or cont-ext, any of the three backtick forms, or a
1717
+ * paragraph-bound short-form `:N-M`) is structurally anchor-less by
1718
+ * construction -- see the "Continuation citations" doc block -- so
1719
+ * unlike `anchor-required` there is no "already carries one" escape: this
1720
+ * fires once its governing citation resolved in-repo, REGARDLESS of
1721
+ * whether that governing full citation itself carries an anchor (an
1722
+ * anchor on the full citation does not extend to a later continuation of
1723
+ * it). Exemptions mirror `anchor-required`'s exactly: the caller already
1724
+ * folds the reserved-citing-doc carve-out into `requireAnchorsForDoc`
1725
+ * (undefined for a reserved doc), and `requireAnchors.allow` is matched
1726
+ * here against `governing.citedPath` -- the path the continuation is
1727
+ * chained to, since a continuation carries no path of its own to match
1728
+ * against.
1729
+ */
1730
+ function pushAnchorRequiredContinuation(findings, file, citation, raw, governing, requireAnchorsForDoc, root) {
1731
+ if (!requireAnchorsForDoc ||
1732
+ matchesAllowPattern(governing.citedPath, requireAnchorsForDoc.allow)) {
1733
+ return;
1734
+ }
1735
+ const governingRange = `${governing.citedPath}:${governing.startLine}${governing.endLine ? "-" + governing.endLine : ""}`;
1736
+ // "chained to" rather than "extends": for a cont-ext that itself chains
1737
+ // off an earlier cont-fresh (not the original full citation), the range
1738
+ // it literally extends is the cont-fresh's own range, not `governing`'s
1739
+ // -- `governing` (see the loop above) always carries the ORIGINAL full
1740
+ // citation's own startLine/endLine, unchanged across intervening
1741
+ // cont-fresh atoms. Naming it "the governing citation" is honest for
1742
+ // every caller (cont-ext, cont-fresh, and the short-form call site
1743
+ // below) without having to thread the true immediate antecedent range
1744
+ // through here.
1745
+ pushDrift(findings, file, citation, "anchor-required-continuation", `continuation ${raw} is chained to the governing citation \`${governingRange}\`; a continuation cannot carry its own #anchor, lift it into a full \`path:N-M#anchor\` citation (--require-anchors is on)`, path.relative(root, governing.resolvedPath));
1746
+ }
1167
1747
  /**
1168
1748
  * True when a `CITATION_RE` match at `matchIndex` is the phantom tail of a
1169
1749
  * filename hard-wrapped across a line break (see the "Hard-wrapped prose"
@@ -1223,11 +1803,36 @@ function extractMalformedAnchorRaw(content, hashIndex) {
1223
1803
  }
1224
1804
  return content.slice(from, end);
1225
1805
  }
1226
- function scanDoc(cache, root, bundleDir, doc) {
1806
+ function scanDoc(cache, root, bundleDir, doc, requireAnchors) {
1227
1807
  const findings = [];
1228
1808
  const content = doc.raw;
1229
1809
  const sources = getValidSources(doc.frontmatter.parsed) ?? [];
1230
1810
  const docAbsPath = path.join(bundleDir, doc.relPath);
1811
+ // The four `--require-anchors` opt-in checks (anchor-required,
1812
+ // anchor-not-on-last-line, anchor-not-unique-in-range,
1813
+ // test-range-straddles-block) are all exempt for a reserved citing doc
1814
+ // (index.md/log.md, see doc.isReserved), the same carve-out this rule
1815
+ // already gives reserved docs for short-form matching: an append-only
1816
+ // narrative journal routinely narrates historical line-number deltas as
1817
+ // prose about the past, not live citations against current content.
1818
+ // Threaded as `undefined` here (rather than a second boolean everywhere
1819
+ // `requireAnchors` is read) so every opt-in check downstream stays gated
1820
+ // on the exact same "is this option object present" test it already
1821
+ // uses for the non-reserved case.
1822
+ const requireAnchorsForDoc = doc.isReserved ? undefined : requireAnchors;
1823
+ // Heading-section citations (well-formed and malformed) are collected up
1824
+ // front so their char spans can gate the `CITATION_RE` scan below: a
1825
+ // `CITATION_RE` match landing inside a heading-section citation's own
1826
+ // quoted content anchor (e.g. `` `x.md:#2.0.0#"see other.md:12 now"` ``)
1827
+ // is not a second, independent citation -- it is text the heading-section
1828
+ // citation already owns. Reused again lower down for the heading-section
1829
+ // findings themselves, rather than re-scanning by regex a second time.
1830
+ const headingSectionMatches = collectHeadingSectionMatches(content);
1831
+ const headingSectionSpans = headingSectionMatches.map((hs) => [hs.index, hs.end]);
1832
+ const headingSectionMalformedMatches = collectHeadingSectionMalformedMatches(content, headingSectionSpans);
1833
+ for (const hm of headingSectionMalformedMatches) {
1834
+ headingSectionSpans.push([hm.index, hm.end]);
1835
+ }
1231
1836
  const fullAtoms = [];
1232
1837
  // Char spans of every matched full citation, used to keep short-form
1233
1838
  // matching (see collectShortFormMatches) from re-matching the tail of a
@@ -1238,6 +1843,8 @@ function scanDoc(cache, root, bundleDir, doc) {
1238
1843
  while ((m = re.exec(content)) !== null) {
1239
1844
  if (isWrappedPathContinuation(content, m.index))
1240
1845
  continue;
1846
+ if (isWithinAnySpan(m.index, headingSectionSpans))
1847
+ continue;
1241
1848
  const matchEnd = m.index + m[0].length;
1242
1849
  // anchor-malformed detection (see the atom-processing loop below for
1243
1850
  // where the finding is actually pushed): a `#` immediately follows the
@@ -1267,9 +1874,13 @@ function scanDoc(cache, root, bundleDir, doc) {
1267
1874
  const atoms = [...fullAtoms, ...collectContinuationAtoms(content)].sort((a, b) => a.index - b.index);
1268
1875
  // `governing`: nearest preceding citation (full or continuation) that
1269
1876
  // resolved to a real file; see the "Continuation citations" doc block
1270
- // above for the reset rules. `lastStartLine`: the start line a "cont-ext"
1271
- // atom extends into a range; tracks the most recent full or cont-fresh
1272
- // atom's own startLine, scoped together with `governing`.
1877
+ // above for the reset rules. Carries the ORIGINAL full citation's own
1878
+ // startLine/endLine (not the extended range a later cont-ext might grow
1879
+ // into) so `anchor-required-continuation` (see `pushAnchorRequiredContinuation`
1880
+ // above) can name "the governing citation" the same way it was actually
1881
+ // written. `lastStartLine`: the start line a "cont-ext" atom extends into
1882
+ // a range; tracks the most recent full or cont-fresh atom's own
1883
+ // startLine, scoped together with `governing`.
1273
1884
  let governing = null;
1274
1885
  let lastStartLine = null;
1275
1886
  for (const atom of atoms) {
@@ -1277,6 +1888,7 @@ function scanDoc(cache, root, bundleDir, doc) {
1277
1888
  if (!governing || lastStartLine === null)
1278
1889
  continue; // nothing to extend
1279
1890
  const citation = `${governing.citedPath}:${lastStartLine}-${atom.value} (continuation)`;
1891
+ pushAnchorRequiredContinuation(findings, doc.relPath, citation, atom.raw, governing, requireAnchorsForDoc, root);
1280
1892
  const problem = checkRangeBoundOnly(lastStartLine, atom.value, governing.resolvedPath);
1281
1893
  if (problem?.rule === "unreadable-target") {
1282
1894
  pushUnreadable(findings, doc.relPath, citation, path.relative(root, governing.resolvedPath), problem.code ?? "UNKNOWN");
@@ -1291,6 +1903,7 @@ function scanDoc(cache, root, bundleDir, doc) {
1291
1903
  continue; // nothing to validate a bare continuation against
1292
1904
  const { startLine, endLine } = atom;
1293
1905
  const citation = `${governing.citedPath}:${startLine}${endLine ? "-" + endLine : ""} (continuation)`;
1906
+ pushAnchorRequiredContinuation(findings, doc.relPath, citation, atom.raw, governing, requireAnchorsForDoc, root);
1294
1907
  const problem = checkTarget(governing.citedPath, startLine, endLine, governing.resolvedPath);
1295
1908
  if (problem?.rule === "unreadable-target") {
1296
1909
  pushUnreadable(findings, doc.relPath, citation, path.relative(root, governing.resolvedPath), problem.code ?? "UNKNOWN");
@@ -1345,15 +1958,80 @@ function scanDoc(cache, root, bundleDir, doc) {
1345
1958
  if (atom.malformedAnchorRaw !== null) {
1346
1959
  pushDrift(findings, doc.relPath, citation, "anchor-malformed", `a "#" follows the citation's range but does not parse as a heading or string anchor (raw: "${atom.malformedAnchorRaw}")`, undefined, "notice");
1347
1960
  }
1348
- const problem = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor);
1961
+ // anchor-required (opt-in, warning): see the "Anchor strictness
1962
+ // (opt-in)" doc block above. Gated on the same posture every other
1963
+ // per-atom check already uses (only reached once the citation resolved
1964
+ // to a real, unambiguous, non-skipped target), plus a reserved citing
1965
+ // doc (e.g. log.md, folded into requireAnchorsForDoc above) and an
1966
+ // allowlisted citedPath being exempt.
1967
+ if (requireAnchorsForDoc &&
1968
+ !anchor &&
1969
+ !matchesAllowPattern(citedPath, requireAnchorsForDoc.allow)) {
1970
+ pushDrift(findings, doc.relPath, citation, "anchor-required", `full citation into an in-repo file carries no #anchor (--require-anchors is on)`, path.relative(root, resolution.path));
1971
+ }
1972
+ const problems = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor, requireAnchorsForDoc);
1973
+ for (const problem of problems) {
1974
+ if (problem.rule === "unreadable-target") {
1975
+ pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
1976
+ }
1977
+ else {
1978
+ pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
1979
+ }
1980
+ }
1981
+ governing = {
1982
+ citedPath,
1983
+ resolvedPath: resolution.path,
1984
+ startLine,
1985
+ endLine,
1986
+ };
1987
+ lastStartLine = startLine;
1988
+ }
1989
+ // Heading-section citations -- see the "Heading-section citations" doc
1990
+ // block above. Independent of `governing`/`lastStartLine`/`fullAtoms`:
1991
+ // this form carries its own path on every citation (never a continuation
1992
+ // or short form), so there is nothing to chain off. Iterates over
1993
+ // `headingSectionMatches`, collected up front (see above), rather than
1994
+ // re-scanning `content` with the regex a second time.
1995
+ for (const hs of headingSectionMatches) {
1996
+ const citedPath = hs.citedPath;
1997
+ const headingAnchor = hs.headingAnchor;
1998
+ const contentAnchor = hs.contentAnchor;
1999
+ const citation = `${citedPath}:${formatAnchorForLabel(headingAnchor)}${contentAnchor ? formatAnchorForLabel(contentAnchor) : ""}`;
2000
+ if (hasParentSegment(citedPath)) {
2001
+ pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
2002
+ continue;
2003
+ }
2004
+ const resolution = resolveCitation(cache, root, docAbsPath, content, sources, citedPath, hs.index);
2005
+ if (!resolution) {
2006
+ pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${citedPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
2007
+ continue;
2008
+ }
2009
+ if ("skip" in resolution)
2010
+ continue;
2011
+ if ("ambiguous" in resolution) {
2012
+ pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
2013
+ continue;
2014
+ }
2015
+ const problem = checkHeadingSectionTarget(headingAnchor, contentAnchor, resolution.path);
1349
2016
  if (problem?.rule === "unreadable-target") {
1350
2017
  pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
1351
2018
  }
1352
2019
  else if (problem) {
1353
2020
  pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
1354
2021
  }
1355
- governing = { citedPath, resolvedPath: resolution.path };
1356
- lastStartLine = startLine;
2022
+ }
2023
+ // heading-section-malformed (notice): a backtick + path + `:#` opener
2024
+ // that did not parse as a well-formed heading-section citation above --
2025
+ // mirrors `anchor-malformed`'s posture for the line-range anchor form
2026
+ // (see `extractMalformedAnchorRaw`'s doc comment): a typo should not
2027
+ // silently vanish from the very check it was written to exercise.
2028
+ for (const hm of headingSectionMalformedMatches) {
2029
+ // `hm.raw` includes the delimiting backticks (the regex match itself);
2030
+ // the citation label, like every other citation label in this file,
2031
+ // does not repeat them -- pushDrift's message template already wraps
2032
+ // the label in its own pair.
2033
+ const inner = hm.raw.slice(1, -1);
2034
+ pushDrift(findings, doc.relPath, inner, "heading-section-malformed", `a backtick-delimited "path:#" heading-section citation opener does not parse as a well-formed citation (raw: "${inner}")`, undefined, "notice");
1357
2035
  }
1358
2036
  // Short-form (paragraph-bound) citations -- see that doc block above.
1359
2037
  // Deliberately independent of `governing`/`lastStartLine`: short-form
@@ -1381,6 +2059,8 @@ function scanDoc(cache, root, bundleDir, doc) {
1381
2059
  namedFullAtoms.push({
1382
2060
  index: a.index,
1383
2061
  citedPath: a.citedPath,
2062
+ startLine: a.startLine,
2063
+ endLine: a.endLine,
1384
2064
  });
1385
2065
  }
1386
2066
  }
@@ -1410,6 +2090,27 @@ function scanDoc(cache, root, bundleDir, doc) {
1410
2090
  pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
1411
2091
  continue;
1412
2092
  }
2093
+ // anchor-required-continuation (opt-in, warning) also applies to a
2094
+ // bound short-form citation: it is just as anchor-less as a
2095
+ // backtick continuation (see the "Short-form citations" doc block
2096
+ // above), for the identical reason -- a bound short form cannot
2097
+ // carry an anchor either, so it is flagged as well, rather than
2098
+ // left as a silent gap in `--require-anchors`.
2099
+ //
2100
+ // `raw` here is synthesized from the parsed startLine/endLine
2101
+ // (`:${rangeLabel}`) rather than carried from the match itself --
2102
+ // `ShortFormMatch` (see there) does not retain the matched text.
2103
+ // This is exact for `SHORT_FORM_COLON_RE` (`/:(\d+)-(\d+)/`) as
2104
+ // written today, since re-stringifying the two captured numbers
2105
+ // reproduces the source bytes; it would stop being exact if that
2106
+ // regex ever tolerated something re-stringifying can't round-trip
2107
+ // (e.g. a leading zero).
2108
+ pushAnchorRequiredContinuation(findings, doc.relPath, citation, `:${rangeLabel}`, {
2109
+ citedPath: targetPath,
2110
+ resolvedPath: resolution.path,
2111
+ startLine: target.startLine,
2112
+ endLine: target.endLine,
2113
+ }, requireAnchorsForDoc, root);
1413
2114
  const problem = checkShortFormTarget(targetPath, sf.startLine, sf.endLine, resolution.path);
1414
2115
  if (problem?.rule === "unreadable-target") {
1415
2116
  pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
@@ -1423,7 +2124,7 @@ function scanDoc(cache, root, bundleDir, doc) {
1423
2124
  }
1424
2125
  export const citationsResolveRule = {
1425
2126
  id: RULE_ID,
1426
- description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations) must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
2127
+ description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations), and backtick-delimited `` `path:#heading` `` heading-section citations (`.md` targets only), must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
1427
2128
  run(ctx) {
1428
2129
  if (!ctx.repoRoot) {
1429
2130
  // Never silently skip: mirrors sources-fresh's posture so a "clean"
@@ -1442,7 +2143,7 @@ export const citationsResolveRule = {
1442
2143
  // Fresh per invocation: see findByBasename's doc comment for why this
1443
2144
  // is not held at module scope.
1444
2145
  const cache = new Map();
1445
- return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc));
2146
+ return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc, ctx.requireAnchors));
1446
2147
  },
1447
2148
  };
1448
2149
  //# sourceMappingURL=citations-resolve.js.map