okf-kit 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -173,7 +173,21 @@ const RULE_ID = "citations-resolve";
173
173
  * checks (blank-start-line, closing-brace-start-line, inverted-range,
174
174
  * range-exceeds-file) already came back clean -- same "one problem per
175
175
  * citation, base checks first" pattern `checkShortFormTarget` already uses
176
- * for the block-boundary check.
176
+ * for the block-boundary check. Every one of these four messages names the
177
+ * anchor text itself, and an anchored full citation's own finding label
178
+ * (`path:N-M#anchor`, see `formatAnchorForLabel`) carries the anchor too --
179
+ * without both, two citations to the identical range with different
180
+ * anchors would be indistinguishable in the output.
181
+ *
182
+ * A `#` that immediately follows a citation's range but does not parse as
183
+ * either anchor form at all (unbalanced quotes, a backtick inside a quoted
184
+ * anchor, or nothing after the `#`) is its own separate, `notice`-severity
185
+ * finding, `anchor-malformed` (see `scanDoc`'s main matching loop): the
186
+ * citation is still checked exactly as an anchorless one would be
187
+ * (backward compatible, see below), but silently checking it anchorless
188
+ * when the `#` right there looks like a typo'd anchor attempt would defeat
189
+ * the entire point of writing one -- a single misplaced character would
190
+ * quietly turn the very check the anchor was written for back off.
177
191
  *
178
192
  * Backward compatible by construction: the `#anchor` suffix is optional in
179
193
  * `CITATION_RE`, so an existing anchorless citation matches exactly as
@@ -202,8 +216,112 @@ const RULE_ID = "citations-resolve";
202
216
  * plus prose) to stay in sync, the exact class of drift this rule
203
217
  * exists to catch.
204
218
  */
219
+ /**
220
+ * Anchor strictness (opt-in, `--require-anchors`, see `src/cli.ts`).
221
+ * Backward compatible by construction: every check in this section is
222
+ * gated on `ctx.requireAnchors` being present (`RequireAnchorsOptions`, see
223
+ * `src/types.ts`), so a consumer that never passes `--require-anchors`
224
+ * sees byte-identical findings to before this section existed. Three new
225
+ * rule ids, all `warning`-severity, mirroring this rule's own "a wrong
226
+ * start/target is strong drift evidence" posture rather than the `notice`
227
+ * posture reserved for the mechanical prose checks with a real
228
+ * false-positive risk (`markdown-range-boundary-bracket-or-fence` and
229
+ * friends):
230
+ *
231
+ * - `anchor-required`: an in-repo full citation (`path:N`/`path:N-M`,
232
+ * resolved to a real target -- an unresolved citation already gets its
233
+ * own `missing-file`/`unresolved-ambiguous` finding, not this one)
234
+ * carrying no `#anchor` at all. Exempted the same way this rule already
235
+ * exempts short-form matching for a reserved citing doc
236
+ * (`doc.isReserved`, e.g. `log.md`'s append-only line-number narration),
237
+ * and additionally by `requireAnchors.allow`: a citedPath matching one
238
+ * of its glob/exact patterns (matched against the citation's raw,
239
+ * as-written `citedPath` text -- see `matchesAllowPattern`) is exempt,
240
+ * for a doc category this check is not (yet) meant to cover, e.g. a
241
+ * README or install guide whose prose is not line-anchored the way a
242
+ * CHANGELOG or source citation is.
243
+ * - `anchor-not-on-last-line`: layered on `checkAnchor`'s existing string
244
+ * branch. An anchor sitting on the FIRST line of a wide range survives a
245
+ * k-line insertion above the range whenever k is smaller than the range
246
+ * itself, because the shifted window (the citation's own line numbers,
247
+ * re-read after the insertion) still contains the original first
248
+ * line's content, just at a different offset inside the window.
249
+ * Anchoring on the LAST line instead closes that: the original last
250
+ * line falls out of the shifted window on any insertion size >= 1, not
251
+ * only large ones.
252
+ * - `anchor-not-unique-in-range`: also layered on the string branch. An
253
+ * anchor text occurring more than once inside its own cited range is
254
+ * ambiguous evidence -- which occurrence is the one actually pinning
255
+ * the citation? A count of zero is unaffected (already
256
+ * `anchor-not-found-in-range`, unconditionally, not double-reported
257
+ * here).
258
+ *
259
+ * Both of the last two are checked only once a match exists at all (count
260
+ * >= 1); uniqueness is checked before last-line placement, since a
261
+ * non-unique anchor is the more fundamental problem (which of the several
262
+ * matching lines "is" the anchor is undefined before asking whether the
263
+ * one occurrence is on the right line).
264
+ *
265
+ * A fourth check, `test-range-straddles-block` (warning), applies to a FULL
266
+ * citation's own range into a `.test.`/`.spec.` target (`.ts`, `.js`,
267
+ * `.mjs`) under this same opt-in (see `checkTestRangeStraddle`, called
268
+ * from `checkFullTarget`): a
269
+ * full citation's range is expected to stay inside a single `describe`/
270
+ * `it`/`test` block once it starts, so any block-head line
271
+ * (`describe(`/`describe.only(`/`describe.skip(`/`describe.each(`/`it(`/
272
+ * `it.only(`/`it.skip(`/`it.each(`/`test(`/`test.only(`/`test.skip(`/
273
+ * `test.each(`, see `TEST_BLOCK_HEAD_RE`) found on any line of the range
274
+ * OTHER than its own first line means the citation ran into a sibling or
275
+ * outer block (or never really started on one). A range that leaves its
276
+ * block without a later head line inside it (ending on an outer block's
277
+ * closing line) is not detected by this line-based check. The range's start
278
+ * line is never itself checked here: it is either a legitimate block head
279
+ * (the range correctly starts a block) or legitimately inside a block's
280
+ * body (a partial citation into the middle of a block), and both are fine
281
+ * -- only a head line reappearing *after* the start is evidence of drift.
282
+ * Deliberately its own rule id rather than reusing `checkRangeBoundary`'s
283
+ * `test-range-start-not-head`/`test-range-end-not-closing` (which stay
284
+ * short-form-only, see `checkShortFormTarget`): those two require the range
285
+ * to start exactly on a head line and end exactly on the matching closing
286
+ * `});`, which is far stricter than a full citation's legitimate use
287
+ * (citing an arbitrary sub-range once a paragraph has already named the
288
+ * file) needs -- the straddle check only cares whether the range crossed a
289
+ * block boundary, not whether it captured a whole block precisely.
290
+ * Deliberately scoped to test-file targets only, same reasoning
291
+ * `checkShortFormTarget`'s own doc comment already gives for the Markdown
292
+ * half of the block-boundary check: a full citation into a Markdown target
293
+ * legitimately cites a couple of arbitrary lines all the time.
294
+ */
295
+ /**
296
+ * True when `citedPath` (the citation's raw, as-written path text) matches
297
+ * one of `patterns`: an exact string match, or a glob with `*` (any run of
298
+ * characters) and `?` (any single character) as the only wildcards -- the
299
+ * minimal shape needed for `requireAnchors.allow`'s own documented example
300
+ * (`["README.md", "INSTALL-AGENT.md"]`), not a general-purpose glob engine.
301
+ */
302
+ function matchesAllowPattern(citedPath, patterns) {
303
+ return patterns.some((pattern) => {
304
+ if (!pattern.includes("*") && !pattern.includes("?")) {
305
+ return pattern === citedPath;
306
+ }
307
+ const escaped = pattern.replace(/[.+^${}()|[\]\\]/g, "\\$&");
308
+ const reSource = "^" + escaped.replace(/\*/g, ".*").replace(/\?/g, ".") + "$";
309
+ return new RegExp(reSource).test(citedPath);
310
+ });
311
+ }
205
312
  const ANCHOR_HEADING_MAX_LEVEL = 2;
206
313
  const MD_HEADING_RE = /^(#{1,6})\s+(.*)$/;
314
+ /**
315
+ * Renders a parsed `Anchor` back to the short form used to disambiguate a
316
+ * finding's citation label (see the `citation` variable in `scanDoc`'s main
317
+ * loop): `#text` for a heading anchor (already bracket-stripped by
318
+ * `parseAnchor`), `#"text"` for a string anchor. Two citations to the same
319
+ * range with different anchors would otherwise be indistinguishable in the
320
+ * output -- see the "Anchored citations" doc block above.
321
+ */
322
+ function formatAnchorForLabel(anchor) {
323
+ return anchor.kind === "string" ? `#"${anchor.text}"` : `#${anchor.text}`;
324
+ }
207
325
  /**
208
326
  * Parses `CITATION_RE`'s optional 4th capture group (the raw anchor text,
209
327
  * including its surrounding quotes or brackets if any) into an `Anchor`, or
@@ -225,13 +343,53 @@ function parseAnchor(raw) {
225
343
  const text = raw.startsWith("[") && raw.endsWith("]") ? raw.slice(1, -1) : raw;
226
344
  return { kind: "heading", text };
227
345
  }
346
+ /**
347
+ * The single fence state machine every fence-aware consumer in this file
348
+ * derives from: one forward pass over `lines`, replaying the same
349
+ * open/close logic (a line matching `MD_FENCE_DELIM_RE` opens a fence when
350
+ * none is open, a line starting with the *same* marker closes it) that
351
+ * `computeFencedSpans`, `computeFencedLineIndices`, and `isFenceOpeningLine`
352
+ * each used to implement as their own, independently-maintained copy --
353
+ * nothing enforced the three staying in agreement with each other. An
354
+ * unterminated fence (no matching close before the end of `lines`) is
355
+ * reflected by every remaining line coming back `fenced: true` -- each line
356
+ * is marked while the scan is still inside the open fence, so no separate
357
+ * post-loop fixup is needed the way the three original copies each had.
358
+ * State at line `i` never depends on any line after `i`, so a caller that
359
+ * only needs one line's state (see `isFenceOpeningLine`) can safely ignore
360
+ * the rest of the array without re-deriving the logic itself.
361
+ */
362
+ function scanFenceLines(lines) {
363
+ const states = [];
364
+ let fenceMarker;
365
+ for (let i = 0; i < lines.length; i++) {
366
+ const trimmed = (lines[i] ?? "").trim();
367
+ if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
368
+ fenceMarker = trimmed.slice(0, 3);
369
+ states.push({ fenced: true, opensFence: true, closesFence: false });
370
+ }
371
+ else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
372
+ states.push({ fenced: true, opensFence: false, closesFence: true });
373
+ fenceMarker = undefined;
374
+ }
375
+ else {
376
+ states.push({
377
+ fenced: fenceMarker !== undefined,
378
+ opensFence: false,
379
+ closesFence: false,
380
+ });
381
+ }
382
+ }
383
+ return states;
384
+ }
228
385
  /**
229
386
  * 0-based line indices that fall inside a fenced code block (```` ``` ````
230
387
  * or `~~~`, optionally with a trailing language tag), delimiters included --
231
388
  * the target-side twin of `computeFencedSpans` above, which does the same
232
- * job for the *citing* doc's short-form matching. Anchor heading-search
233
- * needs its own copy because it works from an already-split `lines` array
234
- * (see `checkFullTarget`), not the raw `content` string `computeFencedSpans`
389
+ * job for the *citing* doc's short-form matching. Derived from
390
+ * `scanFenceLines` (see there). Anchor heading-search needs its own copy
391
+ * because it works from an already-split `lines` array (see
392
+ * `checkFullTarget`), not the raw `content` string `computeFencedSpans`
235
393
  * takes, and because it must ignore a target's `# not a heading` sitting
236
394
  * inside a fenced example exactly the same way a citing doc's own fences
237
395
  * are already ignored for short-form matching -- without this, a `#`-led
@@ -243,25 +401,10 @@ function parseAnchor(raw) {
243
401
  */
244
402
  function computeFencedLineIndices(lines) {
245
403
  const fenced = new Set();
246
- let fenceMarker;
247
- let fenceStart = -1;
248
- for (let i = 0; i < lines.length; i++) {
249
- const trimmed = (lines[i] ?? "").trim();
250
- if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
251
- fenceMarker = trimmed.slice(0, 3);
252
- fenceStart = i;
253
- }
254
- else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
255
- for (let j = fenceStart; j <= i; j++)
256
- fenced.add(j);
257
- fenceMarker = undefined;
258
- fenceStart = -1;
259
- }
260
- }
261
- if (fenceMarker && fenceStart >= 0) {
262
- for (let j = fenceStart; j < lines.length; j++)
263
- fenced.add(j);
264
- }
404
+ scanFenceLines(lines).forEach((state, i) => {
405
+ if (state.fenced)
406
+ fenced.add(i);
407
+ });
265
408
  return fenced;
266
409
  }
267
410
  /**
@@ -285,29 +428,80 @@ function findEnclosingHeading(lines, startLine, fencedLines) {
285
428
  }
286
429
  return null;
287
430
  }
431
+ // A trimmed line composed of nothing but closing brackets/braces/parens
432
+ // plus an optional trailing `,`/`;` -- bare closing boilerplate like `}`,
433
+ // `});`, `]);`, `}),`. Used only by `lastContentLineInRange` (opt-in
434
+ // `anchor-not-on-last-line`, see there): a range that ends on such a line
435
+ // (the common shape of a `});` that closes a whole cited `describe`/`it`
436
+ // block) should not force the anchor onto that boilerplate line itself --
437
+ // the last CONTENT line before it is the meaningful place for an anchor
438
+ // to live.
439
+ const CLOSING_BOILERPLATE_RE = /^[\]\)\};,]*$/;
440
+ function isContentLine(text) {
441
+ const trimmed = text.trim();
442
+ return trimmed !== "" && !CLOSING_BOILERPLATE_RE.test(trimmed);
443
+ }
444
+ /**
445
+ * The last line number (1-based, within `[startLine, endLine]`) whose text
446
+ * is a "content" line -- see `isContentLine`. Walks backward from `endLine`
447
+ * so a range ending on one or more bare closing-boilerplate lines (`});`,
448
+ * `]);`, `}),`, a lone `}`) resolves to the real content line before them
449
+ * instead of the boilerplate itself. Falls back to `endLine` unchanged when
450
+ * every line in the range is boilerplate or blank (nothing better to
451
+ * anchor against).
452
+ */
453
+ function lastContentLineInRange(lines, startLine, endLine) {
454
+ for (let ln = endLine; ln >= startLine; ln--) {
455
+ if (isContentLine(lines[ln - 1] ?? ""))
456
+ return ln;
457
+ }
458
+ return endLine;
459
+ }
288
460
  /**
289
461
  * Checks an anchored full citation's anchor against its already-resolved,
290
462
  * already-range-checked target -- see the "Anchored citations" doc block
291
463
  * above for the two forms' semantics. `endLine` is the citation's end line,
292
464
  * or its start line for a single-line citation (a size-1 range).
293
465
  */
294
- function checkAnchor(anchor, startLine, endLine, lines) {
466
+ function checkAnchor(anchor, startLine, endLine, lines, requireAnchors) {
295
467
  if (anchor.kind === "string") {
468
+ let count = 0;
469
+ let lastMatchLine = -1;
296
470
  for (let i = startLine - 1; i <= endLine - 1 && i < lines.length; i++) {
297
- if ((lines[i] ?? "").includes(anchor.text))
298
- return null;
471
+ if ((lines[i] ?? "").includes(anchor.text)) {
472
+ count++;
473
+ lastMatchLine = i + 1;
474
+ }
299
475
  }
300
- return {
301
- rule: "anchor-not-found-in-range",
302
- message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
303
- };
476
+ if (count === 0) {
477
+ return {
478
+ rule: "anchor-not-found-in-range",
479
+ message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
480
+ };
481
+ }
482
+ if (requireAnchors) {
483
+ if (count > 1) {
484
+ return {
485
+ rule: "anchor-not-unique-in-range",
486
+ message: `anchor "${anchor.text}" occurs on ${count} lines of the cited range (${startLine}-${endLine}); expected exactly one`,
487
+ };
488
+ }
489
+ const lastContentLine = lastContentLineInRange(lines, startLine, endLine);
490
+ if (lastMatchLine !== lastContentLine) {
491
+ return {
492
+ rule: "anchor-not-on-last-line",
493
+ message: `anchor "${anchor.text}" occurs on line ${lastMatchLine}, not the cited range's last content line (${lastContentLine})`,
494
+ };
495
+ }
496
+ }
497
+ return null;
304
498
  }
305
499
  const fencedLines = computeFencedLineIndices(lines);
306
500
  const heading = findEnclosingHeading(lines, startLine, fencedLines);
307
501
  if (!heading) {
308
502
  return {
309
503
  rule: "anchor-heading-not-found",
310
- message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor against; use a string anchor (#"...") instead against a target with no heading structure`,
504
+ message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor "${anchor.text}" against; use a string anchor (#"...") instead against a target with no heading structure`,
311
505
  };
312
506
  }
313
507
  if (!heading.text.includes(anchor.text)) {
@@ -323,13 +517,93 @@ function checkAnchor(anchor, startLine, endLine, lines) {
323
517
  if (m && m[1].length <= heading.level) {
324
518
  return {
325
519
  rule: "anchor-heading-does-not-enclose",
326
- message: `range extends past its enclosing heading's section (next heading "${m[2].trim()}" at line ${i + 1})`,
520
+ message: `range extends past the section enclosing anchor "${anchor.text}" (next heading "${m[2].trim()}" at line ${i + 1})`,
327
521
  };
328
522
  }
329
523
  }
330
524
  return null;
331
525
  }
332
526
  const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
527
+ /**
528
+ * Heading-section citations. A CHANGELOG.md that grows by insertion at the
529
+ * top forces every later `path:N-M#anchor` citation to be re-pointed on
530
+ * every release, even with the heading anchor above closing the "wrong
531
+ * section" gap -- the line RANGE itself still drifts on every insertion,
532
+ * so the citation still needs editing, just not silently mis-validated.
533
+ * `path:#heading` sidesteps line numbers entirely: it names a heading
534
+ * (matched the same way the line-range anchor above already matches one,
535
+ * `heading.text.includes(...)`, reused rather than re-invented -- see
536
+ * `parseAnchor`/`findHeadingSection`) and resolves to that heading's own
537
+ * section (from the line after the heading up to, but not including, the
538
+ * next heading of the same or shallower level, or EOF) -- see
539
+ * `findHeadingSection`. The section must exist (`heading-section-not-found`
540
+ * when no such heading is found), must be unambiguous (more than one
541
+ * matching heading is `heading-section-ambiguous`, never silently the
542
+ * first hit -- the same posture `unresolved-ambiguous` already takes for
543
+ * path resolution), and must not be empty (`heading-section-empty`).
544
+ *
545
+ * Deliberately reuses `ANCHOR_HEADING_MAX_LEVEL` (2), not a second cap: the
546
+ * same Keep-a-Changelog nesting that motivates the line-range anchor's cap
547
+ * (`## [x.y.z]` release headings around identically-named, per-release
548
+ * `### Added`/`### Changed`/`### Fixed` subsections) means a level-3+
549
+ * heading name is *never* unique across a real CHANGELOG anyway --
550
+ * matching it would just turn every such citation into a guaranteed
551
+ * `heading-section-ambiguous`. Capping the search at level 2 keeps this
552
+ * form usable for exactly the case it exists for (a release section) and
553
+ * behaves identically to the line-range anchor for the same reason.
554
+ *
555
+ * Optional content anchor: `path:#heading#"text"`, always the quoted form
556
+ * (a section's body is prose/content, not a second heading to look up) --
557
+ * the text must occur on EXACTLY one line inside the resolved section
558
+ * (`heading-section-content-anchor-not-found` / `-ambiguous`), stricter
559
+ * than the line-range string anchor's "at least one line" (`checkAnchor`'s
560
+ * string branch): a whole section is a much larger haystack than a
561
+ * caller-chosen line range, so "present somewhere" is far weaker evidence
562
+ * the anchor still points at the intended spot, and "found on exactly one
563
+ * line" is what the task this form was built for (pinning a citation to
564
+ * one prose sentence inside a growing section) actually needs. The heading
565
+ * position itself keeps only the bare/bracketed form (`#heading`/
566
+ * `#[heading]`); the quoted alternative is reserved for the content anchor
567
+ * position so the two can never be confused by shape alone.
568
+ *
569
+ * Backtick-delimited, unlike the line-range form above (which does not
570
+ * require backticks -- see the "Anchor syntax note" in the README). This
571
+ * is a deliberate, narrower grammar for this form specifically: a bare
572
+ * `path#heading` (the grammar this form used in an earlier round) is
573
+ * indistinguishable from an ordinary Markdown link's target, which
574
+ * commonly has exactly that shape (`[install](docs/README.md#install)`)
575
+ * -- unlike the line-range form (gated by a `:N` no ordinary link ever
576
+ * contains), there was no character available to tell a real citation
577
+ * apart from a relative link's href written in prose, and measuring
578
+ * against real bundles (see the CHANGELOG entry that introduced the
579
+ * `:#` form) showed the collision is not hypothetical: existing docs write
580
+ * `` `path#heading` `` as inline prose pointing at an unrelated anchor,
581
+ * each producing a spurious `heading-section-not-found`. The colon
582
+ * (`path:#heading`) reuses this rule's existing citation signature (a
583
+ * literal `:` no Markdown link fragment ever contains) instead of relying
584
+ * on backticks alone to do that job, and is additionally restricted to a
585
+ * `.md` target: a link fragment's target is always the Markdown doc it
586
+ * points into, never a source or config file, so a non-`.md` path with
587
+ * this shape is never a live citation. Requiring the whole citation
588
+ * inside one pair of backticks stays in place as well, matching this
589
+ * rule's own general recommendation for the line-range form.
590
+ */
591
+ const HEADING_SECTION_CITATION_RE = /`([\w./-]+\.md):#(\[?\w(?:[\w.-]*\w)?\]?)(?:#("[^"\n`]+"))?`/g;
592
+ /**
593
+ * Companion to `HEADING_SECTION_CITATION_RE`: matches the same
594
+ * backtick + path + `:#` opener but accepts anything up to the closing
595
+ * backtick, so a heading-section citation that fails to parse (an
596
+ * unterminated or empty content-anchor quote, an unquoted third segment,
597
+ * or a non-`.md` target, which includes a non-lowercase `.MD` extension) is
598
+ * still recognised as an *attempt* rather than
599
+ * silently vanishing -- mirrors `anchor-malformed`'s
600
+ * "still-visible-as-a-notice" posture for the line-range anchor form (see
601
+ * `extractMalformedAnchorRaw`). Only used for a match whose span does not
602
+ * already overlap a successful `HEADING_SECTION_CITATION_RE` match (see
603
+ * `collectHeadingSectionMatches`); a well-formed citation never also
604
+ * produces a malformed notice.
605
+ */
606
+ const HEADING_SECTION_MALFORMED_RE = /`([\w./-]+):#([^`\n]*)`/g;
333
607
  // Continuation citation forms (see the "Continuation citations" doc block
334
608
  // above). Each requires the backtick delimiter as part of the match so it
335
609
  // can never overlap a CITATION_RE match: a full citation's regex match
@@ -354,6 +628,14 @@ const SHORT_FORM_COLON_RE = /:(\d+)-(\d+)/g;
354
628
  const TEST_FILE_RE = /\.(test|spec)\.(ts|js|mjs)$/i;
355
629
  const TEST_HEAD_LINE_RE = /^\s*(?:describe|it)\s*\(/;
356
630
  const TEST_CLOSING_LINE_RE = /^\s*\}\)\s*;\s*$/;
631
+ // Block-head detection for test-range-straddles-block (see the doc block
632
+ // above and checkTestRangeStraddle below): a wider set of head shapes than
633
+ // TEST_HEAD_LINE_RE (which stays short-form-only, unchanged, via
634
+ // checkRangeBoundary/checkShortFormTarget) -- describe/it/test, each with
635
+ // an optional .only/.skip/.each modifier. A leading `export `/`async ` is
636
+ // deliberately not handled: not needed for any bundle this rule has been
637
+ // measured against.
638
+ const TEST_BLOCK_HEAD_RE = /^\s*(?:describe|it|test)(?:\.(?:only|skip|each))?\s*\(/;
357
639
  // Markdown block-boundary check (see checkRangeBoundary): a range boundary
358
640
  // line that is nothing but a bracket (open or close), optionally with a
359
641
  // trailing `,`/`;`, is always a drift signal. A bare code-fence delimiter is
@@ -629,17 +911,215 @@ function checkTarget(citedPath, startLine, endLine, resolvedPath) {
629
911
  * are full-citation-only (see that doc block for why), so `checkTarget`
630
912
  * itself is untouched and still used as-is for a cont-fresh atom.
631
913
  */
632
- function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor) {
914
+ function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor, requireAnchors) {
633
915
  const read = readTarget(resolvedPath);
634
916
  if ("rule" in read)
635
- return read;
917
+ return [read];
636
918
  const lines = splitLines(read.content);
637
919
  const base = checkTargetLines(citedPath, startLine, endLine, lines);
638
920
  if (base)
639
- return base;
640
- if (!anchor)
641
- return null;
642
- return checkAnchor(anchor, startLine, endLine ?? startLine, lines);
921
+ return [base];
922
+ const problems = [];
923
+ // test-range-straddles-block, opt-in only (see "Anchor strictness
924
+ // (opt-in)" above and checkTestRangeStraddle): a full citation's own
925
+ // range into a .test./.spec. target (.ts, .js, .mjs) must not cross
926
+ // into another block's head after its own first line. Deliberately
927
+ // scoped to test-file targets only (isTestFile), not also a markdown
928
+ // branch: a
929
+ // full citation into a markdown target legitimately cites a couple of
930
+ // arbitrary lines all the time (e.g. two lines of a code fence example),
931
+ // the exact reasoning checkShortFormTarget's own doc comment already
932
+ // gives for why the sibling block-boundary check was scoped to
933
+ // short-form in the first place. Only meaningful for an actual range
934
+ // (endLine !== null); a single-line citation trivially starts and ends
935
+ // on the same line and has nothing to straddle.
936
+ if (requireAnchors && endLine !== null && isTestFile(citedPath)) {
937
+ const straddle = checkTestRangeStraddle(startLine, endLine, lines);
938
+ if (straddle)
939
+ problems.push(straddle);
940
+ }
941
+ // The anchor check is run independently of the straddle check above
942
+ // (not gated on it having come back clean): a straddling range and a
943
+ // missing/misplaced anchor are two independent problems with the same
944
+ // citation, and reporting only the first one found would silently drop
945
+ // the other from the output every time both happen to co-occur.
946
+ if (anchor) {
947
+ const anchorProblem = checkAnchor(anchor, startLine, endLine ?? startLine, lines, requireAnchors);
948
+ if (anchorProblem)
949
+ problems.push(anchorProblem);
950
+ }
951
+ return problems;
952
+ }
953
+ /**
954
+ * Every Markdown heading in `lines` up to `ANCHOR_HEADING_MAX_LEVEL`, in
955
+ * document order -- the whole-document counterpart of `findEnclosingHeading`
956
+ * (which stops at the nearest heading at or before a given line): a
957
+ * heading-section citation has no start line to search backward from, it
958
+ * needs every candidate to test for a unique match. `fencedLines` (see
959
+ * `computeFencedLineIndices`) excludes a `#`-led line inside a fenced
960
+ * example, same as `findEnclosingHeading`.
961
+ */
962
+ function collectHeadingsUpToLevel(lines, fencedLines) {
963
+ const headings = [];
964
+ for (let i = 0; i < lines.length; i++) {
965
+ if (fencedLines.has(i))
966
+ continue;
967
+ const m = (lines[i] ?? "").match(MD_HEADING_RE);
968
+ if (m && m[1].length <= ANCHOR_HEADING_MAX_LEVEL) {
969
+ headings.push({ level: m[1].length, text: m[2].trim(), lineNo: i + 1 });
970
+ }
971
+ }
972
+ return headings;
973
+ }
974
+ /**
975
+ * Resolves a heading-section citation's heading text to a single section:
976
+ * every level <= `ANCHOR_HEADING_MAX_LEVEL` heading whose text contains
977
+ * `headingText` (same containment check `checkAnchor`'s heading branch
978
+ * already uses -- reused, not reinvented) is a candidate; zero is
979
+ * `not-found`, more than one is `ambiguous` (never silently the first
980
+ * match), exactly one resolves to the section running from the line right
981
+ * after the heading up to (not including) the next heading at or above the
982
+ * same level, or EOF -- the same enclosure boundary `checkAnchor`'s heading
983
+ * branch walks forward to check, just producing a body range here instead
984
+ * of a pass/fail against an already-known end line.
985
+ */
986
+ function findHeadingSection(lines, fencedLines, headingText) {
987
+ const headings = collectHeadingsUpToLevel(lines, fencedLines);
988
+ const matches = headings.filter((h) => h.text.includes(headingText));
989
+ if (matches.length === 0)
990
+ return { kind: "not-found" };
991
+ if (matches.length > 1)
992
+ return { kind: "ambiguous", matches };
993
+ const heading = matches[0];
994
+ let bodyEnd = lines.length;
995
+ for (let i = heading.lineNo; i < lines.length; i++) {
996
+ if (fencedLines.has(i))
997
+ continue;
998
+ const m = (lines[i] ?? "").match(MD_HEADING_RE);
999
+ if (m && m[1].length <= heading.level) {
1000
+ bodyEnd = i;
1001
+ break;
1002
+ }
1003
+ }
1004
+ return { kind: "found", heading, bodyStart: heading.lineNo, bodyEnd };
1005
+ }
1006
+ /** True when every line in `[bodyStart, bodyEnd)` is empty or whitespace-only. */
1007
+ function isSectionEmpty(lines, bodyStart, bodyEnd) {
1008
+ for (let i = bodyStart; i < bodyEnd; i++) {
1009
+ if ((lines[i] ?? "").trim() !== "")
1010
+ return false;
1011
+ }
1012
+ return true;
1013
+ }
1014
+ /**
1015
+ * Count of lines in `[bodyStart, bodyEnd)` containing `text` -- a heading
1016
+ * section's content anchor must occur on exactly one such line (see the
1017
+ * "Heading-section citations" doc block above for why this is stricter
1018
+ * than the line-range string anchor's "at least one line").
1019
+ */
1020
+ function countAnchorOccurrences(lines, bodyStart, bodyEnd, text) {
1021
+ let count = 0;
1022
+ for (let i = bodyStart; i < bodyEnd; i++) {
1023
+ if ((lines[i] ?? "").includes(text))
1024
+ count++;
1025
+ }
1026
+ return count;
1027
+ }
1028
+ /**
1029
+ * A heading-section citation's complete target check (see the
1030
+ * "Heading-section citations" doc block above): the named heading must
1031
+ * exist exactly once, its section must be non-empty, and, when a content
1032
+ * anchor was given, it must occur on exactly one line inside that section.
1033
+ * One problem per citation, checked in that order, matching every other
1034
+ * check in this file's "base checks first" pattern.
1035
+ */
1036
+ function checkHeadingSectionTarget(headingAnchor, contentAnchor, resolvedPath) {
1037
+ const read = readTarget(resolvedPath);
1038
+ if ("rule" in read)
1039
+ return read;
1040
+ const lines = splitLines(read.content);
1041
+ const fencedLines = computeFencedLineIndices(lines);
1042
+ const section = findHeadingSection(lines, fencedLines, headingAnchor.text);
1043
+ if (section.kind === "not-found") {
1044
+ return {
1045
+ rule: "heading-section-not-found",
1046
+ message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) contains "${headingAnchor.text}"`,
1047
+ };
1048
+ }
1049
+ if (section.kind === "ambiguous") {
1050
+ return {
1051
+ rule: "heading-section-ambiguous",
1052
+ message: `${section.matches.length} headings contain "${headingAnchor.text}" (lines ${section.matches
1053
+ .map((m) => m.lineNo)
1054
+ .join(", ")}); not evaluated`,
1055
+ };
1056
+ }
1057
+ if (isSectionEmpty(lines, section.bodyStart, section.bodyEnd)) {
1058
+ return {
1059
+ rule: "heading-section-empty",
1060
+ message: `section under heading "${section.heading.text}" (line ${section.heading.lineNo}) has no non-blank content before the next heading`,
1061
+ };
1062
+ }
1063
+ if (contentAnchor) {
1064
+ const count = countAnchorOccurrences(lines, section.bodyStart, section.bodyEnd, contentAnchor.text);
1065
+ if (count === 0) {
1066
+ return {
1067
+ rule: "heading-section-content-anchor-not-found",
1068
+ message: `content anchor "${contentAnchor.text}" does not occur in the section under heading "${section.heading.text}"`,
1069
+ };
1070
+ }
1071
+ if (count > 1) {
1072
+ return {
1073
+ rule: "heading-section-content-anchor-ambiguous",
1074
+ message: `content anchor "${contentAnchor.text}" occurs on ${count} lines in the section under heading "${section.heading.text}"; expected exactly one`,
1075
+ };
1076
+ }
1077
+ }
1078
+ return null;
1079
+ }
1080
+ /**
1081
+ * Every well-formed heading-section citation in `content`, in document
1082
+ * order. Collected once, up front (before `CITATION_RE`'s own scan in
1083
+ * `scanDoc`), so its match spans can gate both `CITATION_RE` (a full
1084
+ * citation never fires inside a heading-section citation's own quoted
1085
+ * content anchor, see the "Heading-section citations" doc block above) and
1086
+ * the malformed companion scan (see `collectHeadingSectionMalformedMatches`)
1087
+ * without re-deriving the same spans twice.
1088
+ */
1089
+ function collectHeadingSectionMatches(content) {
1090
+ const out = [];
1091
+ const re = new RegExp(HEADING_SECTION_CITATION_RE.source, HEADING_SECTION_CITATION_RE.flags);
1092
+ let m;
1093
+ while ((m = re.exec(content)) !== null) {
1094
+ out.push({
1095
+ index: m.index,
1096
+ end: m.index + m[0].length,
1097
+ citedPath: m[1],
1098
+ headingAnchor: parseAnchor(m[2]), // group 2 is mandatory
1099
+ contentAnchor: parseAnchor(m[3]),
1100
+ });
1101
+ }
1102
+ return out;
1103
+ }
1104
+ /**
1105
+ * Every backtick + path + `:#` opener in `content` that did NOT parse as a
1106
+ * well-formed heading-section citation (see `HEADING_SECTION_MALFORMED_RE`'s
1107
+ * doc comment) -- an unterminated content-anchor quote, an unquoted third
1108
+ * segment, or a non-`.md` target. `wellFormedSpans` (from
1109
+ * `collectHeadingSectionMatches`) gates out any match that is really just
1110
+ * the successful citation seen from the outside; a well-formed citation
1111
+ * never also produces a malformed notice.
1112
+ */
1113
+ function collectHeadingSectionMalformedMatches(content, wellFormedSpans) {
1114
+ const out = [];
1115
+ const re = new RegExp(HEADING_SECTION_MALFORMED_RE.source, "g");
1116
+ let m;
1117
+ while ((m = re.exec(content)) !== null) {
1118
+ if (isWithinAnySpan(m.index, wellFormedSpans))
1119
+ continue;
1120
+ out.push({ index: m.index, end: m.index + m[0].length, raw: m[0] });
1121
+ }
1122
+ return out;
643
1123
  }
644
1124
  // A cont-ext atom only ever extends the *end* of a range whose start line
645
1125
  // was already fully checked (blank / closing-brace) when it was cited as
@@ -668,27 +1148,13 @@ function isTestFile(citedPath) {
668
1148
  * identical). Used by checkRangeBoundary's markdown branch: citing a
669
1149
  * fenced block starting at its own opening fence line is the natural,
670
1150
  * correct way to cite it, so that specific case is exempted from the
671
- * fence-as-drift-signal check (see there).
1151
+ * fence-as-drift-signal check (see there). Derived from `scanFenceLines`
1152
+ * (see there); state at `lineIndex` never depends on any line after it, so
1153
+ * scanning the whole array and reading one index back is equivalent to (and
1154
+ * replaces) the original's own up-to-`lineIndex`-only replay.
672
1155
  */
673
1156
  function isFenceOpeningLine(lines, lineIndex) {
674
- let inFence = false;
675
- let fenceMarker;
676
- for (let i = 0; i <= lineIndex; i++) {
677
- const trimmed = (lines[i] ?? "").trim();
678
- if (!inFence && MD_FENCE_DELIM_RE.test(trimmed)) {
679
- if (i === lineIndex)
680
- return true;
681
- inFence = true;
682
- fenceMarker = trimmed.slice(0, 3);
683
- }
684
- else if (inFence && fenceMarker && trimmed.startsWith(fenceMarker)) {
685
- if (i === lineIndex)
686
- return false;
687
- inFence = false;
688
- fenceMarker = undefined;
689
- }
690
- }
691
- return false;
1157
+ return scanFenceLines(lines)[lineIndex]?.opensFence ?? false;
692
1158
  }
693
1159
  /**
694
1160
  * True when `lines[endLineIndex]` is the closing delimiter that matches
@@ -803,6 +1269,53 @@ function checkRangeBoundary(citedPath, startLine, endLine, lines) {
803
1269
  }
804
1270
  return null;
805
1271
  }
1272
+ /**
1273
+ * Width (in characters) of `text`'s leading run of spaces/tabs -- used by
1274
+ * `checkTestRangeStraddle` to tell a NESTED block head (indented deeper
1275
+ * than the range's own start line) apart from a SIBLING or OUTER one
1276
+ * (indented the same or shallower): citing a whole `describe` block
1277
+ * necessarily contains every `it(`/nested-`describe(` head inside its own
1278
+ * body, and those are not straddling anywhere, they are exactly what the
1279
+ * citation is about.
1280
+ */
1281
+ function leadingWhitespaceWidth(text) {
1282
+ return (text.match(/^[ \t]*/) ?? [""])[0].length;
1283
+ }
1284
+ /**
1285
+ * `test-range-straddles-block` (opt-in, warning): a FULL citation's own
1286
+ * range into a `.test.`/`.spec.` target (`.ts`, `.js`, `.mjs`), see the
1287
+ * "Anchor strictness (opt-in)" doc block above for the full rule text.
1288
+ * Checks every line of `[startLine, endLine]` EXCEPT the range's own
1289
+ * first line (`startLine`
1290
+ * itself is never checked -- see that doc block for why) against
1291
+ * `TEST_BLOCK_HEAD_RE`; the first hit (in document order) is reported,
1292
+ * matching this file's "one problem per citation" pattern. A block-head
1293
+ * line indented STRICTLY DEEPER than the range's own start line is a
1294
+ * NESTED block (e.g. the `it(`s inside a `describe(` the range cites in
1295
+ * full) and is skipped, not reported: a citation covering a whole block is
1296
+ * expected to contain every head line nested inside it. A block-head line
1297
+ * at the same or a shallower indent is a SIBLING or OUTER block and is
1298
+ * still reported -- this is an approximation of "which block scope is
1299
+ * this line lexically in" using indentation instead of a real AST, chosen
1300
+ * because it reproduces the AST-based reference verdict on every straddle
1301
+ * finding this rule has been measured against; a file that mixes tabs and
1302
+ * spaces, or that does not indent nested blocks at all, can defeat it.
1303
+ */
1304
+ function checkTestRangeStraddle(startLine, endLine, lines) {
1305
+ const startIndent = leadingWhitespaceWidth(lines[startLine - 1] ?? "");
1306
+ for (let i = startLine; i <= endLine - 1 && i < lines.length; i++) {
1307
+ const text = lines[i] ?? "";
1308
+ if (!TEST_BLOCK_HEAD_RE.test(text))
1309
+ continue;
1310
+ if (leadingWhitespaceWidth(text) > startIndent)
1311
+ continue; // nested block
1312
+ return {
1313
+ rule: "test-range-straddles-block",
1314
+ message: `range straddles into another block's head at line ${i + 1} ("${text.trim()}")`,
1315
+ };
1316
+ }
1317
+ return null;
1318
+ }
806
1319
  /**
807
1320
  * A short-form citation's full check: checkTarget's existing checks
808
1321
  * (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
@@ -926,30 +1439,29 @@ function collectShortFormMatches(content, excludedSpans) {
926
1439
  * inclusive. An unterminated fence (no matching close before end of doc) is
927
1440
  * treated as running to the end of the content -- conservative, since an
928
1441
  * unterminated fence is itself a doc problem outside this rule's scope, not
929
- * a reason to scan its contents for short-form citations.
1442
+ * a reason to scan its contents for short-form citations. Derived from
1443
+ * `scanFenceLines` (see there): per-line fenced/opens/closes state is
1444
+ * converted to char-offset spans by tracking each line's `[start, end)`
1445
+ * offset in `content` alongside it.
930
1446
  */
931
1447
  function computeFencedSpans(content) {
932
1448
  const spans = [];
933
1449
  const lines = content.split("\n");
1450
+ const states = scanFenceLines(lines);
934
1451
  let offset = 0;
935
- let fenceMarker;
936
- let fenceStart = -1;
937
- for (const line of lines) {
938
- const trimmed = line.trim();
939
- const lineEnd = offset + line.length;
940
- if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
941
- fenceMarker = trimmed.slice(0, 3);
942
- fenceStart = offset;
943
- }
944
- else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
945
- spans.push([fenceStart, lineEnd]);
946
- fenceMarker = undefined;
947
- fenceStart = -1;
1452
+ let spanStart = -1;
1453
+ for (let i = 0; i < lines.length; i++) {
1454
+ const lineEnd = offset + lines[i].length;
1455
+ if (states[i].opensFence)
1456
+ spanStart = offset;
1457
+ if (states[i].closesFence && spanStart >= 0) {
1458
+ spans.push([spanStart, lineEnd]);
1459
+ spanStart = -1;
948
1460
  }
949
1461
  offset = lineEnd + 1; // +1 for the newline joining this line to the next
950
1462
  }
951
- if (fenceMarker && fenceStart >= 0) {
952
- spans.push([fenceStart, content.length]);
1463
+ if (spanStart >= 0) {
1464
+ spans.push([spanStart, content.length]);
953
1465
  }
954
1466
  return spans;
955
1467
  }
@@ -1148,11 +1660,76 @@ function isWrappedPathContinuation(content, matchIndex) {
1148
1660
  const prevLine = content.slice(prevLineStart, prevLineEnd);
1149
1661
  return /[-–]$/.test(prevLine);
1150
1662
  }
1151
- function scanDoc(cache, root, bundleDir, doc) {
1663
+ /**
1664
+ * Raw text following a malformed anchor's `#` (see `anchor-malformed` in
1665
+ * the atom-processing loop below), used only to make that finding's message
1666
+ * concrete -- bounded so it reports roughly what was actually typed as the
1667
+ * failed anchor attempt, not an unrelated run of later prose:
1668
+ * - a quoted-anchor attempt (the character right after `#` is `"`) stops
1669
+ * at the next `"` on the same line (the closing quote the author
1670
+ * presumably meant, embedded backticks and all -- that quote is exactly
1671
+ * what makes this a *quoted* anchor attempt rather than a heading one),
1672
+ * or at the end of the line when no such quote exists on it (matches
1673
+ * `CITATION_RE`'s own string alternative, which cannot cross a
1674
+ * newline either);
1675
+ * - any other character after `#` is a heading-anchor attempt, which
1676
+ * stops at the first whitespace (a heading anchor token, like
1677
+ * `CITATION_RE`'s own heading alternative, never contains whitespace)
1678
+ * or the end of the line, whichever comes first;
1679
+ * - either way, capped at `MAX_RAW_LEN` characters so a heading-form
1680
+ * attempt with no whitespace at all before the line ends (or a
1681
+ * pathological single long line) cannot make the message unbounded.
1682
+ */
1683
+ const MAX_MALFORMED_ANCHOR_RAW_LEN = 60;
1684
+ function extractMalformedAnchorRaw(content, hashIndex) {
1685
+ const from = hashIndex + 1;
1686
+ const nl = content.indexOf("\n", from);
1687
+ const lineEnd = nl === -1 ? content.length : nl;
1688
+ let end;
1689
+ if (content[from] === '"') {
1690
+ const q = content.indexOf('"', from + 1);
1691
+ end = q !== -1 && q < lineEnd ? q + 1 : lineEnd;
1692
+ }
1693
+ else {
1694
+ const rest = content.slice(from, lineEnd);
1695
+ const ws = rest.search(/\s/);
1696
+ end = ws === -1 ? lineEnd : from + ws;
1697
+ }
1698
+ if (end - from > MAX_MALFORMED_ANCHOR_RAW_LEN) {
1699
+ end = from + MAX_MALFORMED_ANCHOR_RAW_LEN;
1700
+ }
1701
+ return content.slice(from, end);
1702
+ }
1703
+ function scanDoc(cache, root, bundleDir, doc, requireAnchors) {
1152
1704
  const findings = [];
1153
1705
  const content = doc.raw;
1154
1706
  const sources = getValidSources(doc.frontmatter.parsed) ?? [];
1155
1707
  const docAbsPath = path.join(bundleDir, doc.relPath);
1708
+ // The four `--require-anchors` opt-in checks (anchor-required,
1709
+ // anchor-not-on-last-line, anchor-not-unique-in-range,
1710
+ // test-range-straddles-block) are all exempt for a reserved citing doc
1711
+ // (index.md/log.md, see doc.isReserved), the same carve-out this rule
1712
+ // already gives reserved docs for short-form matching: an append-only
1713
+ // narrative journal routinely narrates historical line-number deltas as
1714
+ // prose about the past, not live citations against current content.
1715
+ // Threaded as `undefined` here (rather than a second boolean everywhere
1716
+ // `requireAnchors` is read) so every opt-in check downstream stays gated
1717
+ // on the exact same "is this option object present" test it already
1718
+ // uses for the non-reserved case.
1719
+ const requireAnchorsForDoc = doc.isReserved ? undefined : requireAnchors;
1720
+ // Heading-section citations (well-formed and malformed) are collected up
1721
+ // front so their char spans can gate the `CITATION_RE` scan below: a
1722
+ // `CITATION_RE` match landing inside a heading-section citation's own
1723
+ // quoted content anchor (e.g. `` `x.md:#2.0.0#"see other.md:12 now"` ``)
1724
+ // is not a second, independent citation -- it is text the heading-section
1725
+ // citation already owns. Reused again lower down for the heading-section
1726
+ // findings themselves, rather than re-scanning by regex a second time.
1727
+ const headingSectionMatches = collectHeadingSectionMatches(content);
1728
+ const headingSectionSpans = headingSectionMatches.map((hs) => [hs.index, hs.end]);
1729
+ const headingSectionMalformedMatches = collectHeadingSectionMalformedMatches(content, headingSectionSpans);
1730
+ for (const hm of headingSectionMalformedMatches) {
1731
+ headingSectionSpans.push([hm.index, hm.end]);
1732
+ }
1156
1733
  const fullAtoms = [];
1157
1734
  // Char spans of every matched full citation, used to keep short-form
1158
1735
  // matching (see collectShortFormMatches) from re-matching the tail of a
@@ -1163,6 +1740,23 @@ function scanDoc(cache, root, bundleDir, doc) {
1163
1740
  while ((m = re.exec(content)) !== null) {
1164
1741
  if (isWrappedPathContinuation(content, m.index))
1165
1742
  continue;
1743
+ if (isWithinAnySpan(m.index, headingSectionSpans))
1744
+ continue;
1745
+ const matchEnd = m.index + m[0].length;
1746
+ // anchor-malformed detection (see the atom-processing loop below for
1747
+ // where the finding is actually pushed): a `#` immediately follows the
1748
+ // range but group 4 (the anchor) did not match -- unbalanced quotes, a
1749
+ // backtick inside a quoted anchor, or nothing at all after the `#`
1750
+ // (e.g. `path:N-M#` at end of line). Computed here, at match time,
1751
+ // because it needs `content`/`matchEnd`; carried on the atom rather
1752
+ // than pushed immediately so the citation's out-of-scope/resolution
1753
+ // posture (path-traversal-rejected, skip, missing-file, ambiguous) can
1754
+ // gate it exactly the same way every other check on this atom already
1755
+ // is -- an out-of-scope or unresolved citation gets none of those
1756
+ // checks either.
1757
+ const malformedAnchorRaw = m[4] === undefined && content[matchEnd] === "#"
1758
+ ? extractMalformedAnchorRaw(content, matchEnd)
1759
+ : null;
1166
1760
  fullAtoms.push({
1167
1761
  kind: "full",
1168
1762
  index: m.index,
@@ -1170,8 +1764,9 @@ function scanDoc(cache, root, bundleDir, doc) {
1170
1764
  startLine: Number(m[2]),
1171
1765
  endLine: m[3] ? Number(m[3]) : null,
1172
1766
  anchor: parseAnchor(m[4]),
1767
+ malformedAnchorRaw,
1173
1768
  });
1174
- fullSpans.push([m.index, m.index + m[0].length]);
1769
+ fullSpans.push([m.index, matchEnd]);
1175
1770
  }
1176
1771
  const atoms = [...fullAtoms, ...collectContinuationAtoms(content)].sort((a, b) => a.index - b.index);
1177
1772
  // `governing`: nearest preceding citation (full or continuation) that
@@ -1211,7 +1806,13 @@ function scanDoc(cache, root, bundleDir, doc) {
1211
1806
  continue; // governing (same file) carries over unchanged
1212
1807
  }
1213
1808
  const { citedPath, startLine, endLine, anchor } = atom;
1214
- const citation = `${citedPath}:${startLine}${endLine ? "-" + endLine : ""}`;
1809
+ // The anchor, when present, is carried in the citation label itself
1810
+ // (not just the anchor-check finding's own message) so two citations to
1811
+ // the same range with different anchors are distinguishable in the
1812
+ // output -- see formatAnchorForLabel and the "Anchored citations" doc
1813
+ // block above. Continuations and short-form citations never carry an
1814
+ // anchor (see there), so their own citation labels are unaffected.
1815
+ const citation = `${citedPath}:${startLine}${endLine ? "-" + endLine : ""}${anchor ? formatAnchorForLabel(anchor) : ""}`;
1215
1816
  if (hasParentSegment(citedPath)) {
1216
1817
  pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
1217
1818
  governing = null;
@@ -1236,15 +1837,87 @@ function scanDoc(cache, root, bundleDir, doc) {
1236
1837
  lastStartLine = null;
1237
1838
  continue;
1238
1839
  }
1239
- const problem = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor);
1840
+ // anchor-malformed (notice): only reached for a citation that resolved
1841
+ // to a real file -- see `malformedAnchorRaw`'s doc comment above for
1842
+ // why this is gated the same way missing-file/skip/path-traversal are
1843
+ // already gated for every other check on this atom. The citation is
1844
+ // still checked below via checkFullTarget exactly as an ordinary
1845
+ // anchorless citation would be (anchor is null here by construction --
1846
+ // see parseAnchor); this only ADDS a heads-up that the `#` sitting
1847
+ // right there was silently not read as the anchor it looks like it was
1848
+ // meant to be.
1849
+ if (atom.malformedAnchorRaw !== null) {
1850
+ pushDrift(findings, doc.relPath, citation, "anchor-malformed", `a "#" follows the citation's range but does not parse as a heading or string anchor (raw: "${atom.malformedAnchorRaw}")`, undefined, "notice");
1851
+ }
1852
+ // anchor-required (opt-in, warning): see the "Anchor strictness
1853
+ // (opt-in)" doc block above. Gated on the same posture every other
1854
+ // per-atom check already uses (only reached once the citation resolved
1855
+ // to a real, unambiguous, non-skipped target), plus a reserved citing
1856
+ // doc (e.g. log.md, folded into requireAnchorsForDoc above) and an
1857
+ // allowlisted citedPath being exempt.
1858
+ if (requireAnchorsForDoc &&
1859
+ !anchor &&
1860
+ !matchesAllowPattern(citedPath, requireAnchorsForDoc.allow)) {
1861
+ pushDrift(findings, doc.relPath, citation, "anchor-required", `full citation into an in-repo file carries no #anchor (--require-anchors is on)`, path.relative(root, resolution.path));
1862
+ }
1863
+ const problems = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor, requireAnchorsForDoc);
1864
+ for (const problem of problems) {
1865
+ if (problem.rule === "unreadable-target") {
1866
+ pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
1867
+ }
1868
+ else {
1869
+ pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
1870
+ }
1871
+ }
1872
+ governing = { citedPath, resolvedPath: resolution.path };
1873
+ lastStartLine = startLine;
1874
+ }
1875
+ // Heading-section citations -- see the "Heading-section citations" doc
1876
+ // block above. Independent of `governing`/`lastStartLine`/`fullAtoms`:
1877
+ // this form carries its own path on every citation (never a continuation
1878
+ // or short form), so there is nothing to chain off. Iterates over
1879
+ // `headingSectionMatches`, collected up front (see above), rather than
1880
+ // re-scanning `content` with the regex a second time.
1881
+ for (const hs of headingSectionMatches) {
1882
+ const citedPath = hs.citedPath;
1883
+ const headingAnchor = hs.headingAnchor;
1884
+ const contentAnchor = hs.contentAnchor;
1885
+ const citation = `${citedPath}:${formatAnchorForLabel(headingAnchor)}${contentAnchor ? formatAnchorForLabel(contentAnchor) : ""}`;
1886
+ if (hasParentSegment(citedPath)) {
1887
+ pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
1888
+ continue;
1889
+ }
1890
+ const resolution = resolveCitation(cache, root, docAbsPath, content, sources, citedPath, hs.index);
1891
+ if (!resolution) {
1892
+ pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${citedPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
1893
+ continue;
1894
+ }
1895
+ if ("skip" in resolution)
1896
+ continue;
1897
+ if ("ambiguous" in resolution) {
1898
+ pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
1899
+ continue;
1900
+ }
1901
+ const problem = checkHeadingSectionTarget(headingAnchor, contentAnchor, resolution.path);
1240
1902
  if (problem?.rule === "unreadable-target") {
1241
1903
  pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
1242
1904
  }
1243
1905
  else if (problem) {
1244
1906
  pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
1245
1907
  }
1246
- governing = { citedPath, resolvedPath: resolution.path };
1247
- lastStartLine = startLine;
1908
+ }
1909
+ // heading-section-malformed (notice): a backtick + path + `:#` opener
1910
+ // that did not parse as a well-formed heading-section citation above --
1911
+ // mirrors `anchor-malformed`'s posture for the line-range anchor form
1912
+ // (see `extractMalformedAnchorRaw`'s doc comment): a typo should not
1913
+ // silently vanish from the very check it was written to exercise.
1914
+ for (const hm of headingSectionMalformedMatches) {
1915
+ // `hm.raw` includes the delimiting backticks (the regex match itself);
1916
+ // the citation label, like every other citation label in this file,
1917
+ // does not repeat them -- pushDrift's message template already wraps
1918
+ // the label in its own pair.
1919
+ const inner = hm.raw.slice(1, -1);
1920
+ pushDrift(findings, doc.relPath, inner, "heading-section-malformed", `a backtick-delimited "path:#" heading-section citation opener does not parse as a well-formed citation (raw: "${inner}")`, undefined, "notice");
1248
1921
  }
1249
1922
  // Short-form (paragraph-bound) citations -- see that doc block above.
1250
1923
  // Deliberately independent of `governing`/`lastStartLine`: short-form
@@ -1314,7 +1987,7 @@ function scanDoc(cache, root, bundleDir, doc) {
1314
1987
  }
1315
1988
  export const citationsResolveRule = {
1316
1989
  id: RULE_ID,
1317
- description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations) must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
1990
+ description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations), and backtick-delimited `` `path:#heading` `` heading-section citations (`.md` targets only), must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
1318
1991
  run(ctx) {
1319
1992
  if (!ctx.repoRoot) {
1320
1993
  // Never silently skip: mirrors sources-fresh's posture so a "clean"
@@ -1333,7 +2006,7 @@ export const citationsResolveRule = {
1333
2006
  // Fresh per invocation: see findByBasename's doc comment for why this
1334
2007
  // is not held at module scope.
1335
2008
  const cache = new Map();
1336
- return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc));
2009
+ return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc, ctx.requireAnchors));
1337
2010
  },
1338
2011
  };
1339
2012
  //# sourceMappingURL=citations-resolve.js.map