okf-kit 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1339 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ import { getValidSources } from "../util.js";
4
+ const RULE_ID = "citations-resolve";
5
+ /**
6
+ * Warn-only checker for `path:N[-M]` citations inside a bundle's docs (and
7
+ * their continuation forms, see below).
8
+ *
9
+ * Ported from agent-grounding's `scripts/okf-citations-resolve.mjs`
10
+ * (PR #185), which was a repo-local, no-dependency spike at closing a gap
11
+ * `sources-fresh` cannot: `sources-fresh` only compares a doc's `sources`
12
+ * list against file mtimes, so it is structurally blind to an edit that
13
+ * shifts line numbers inside a still-fresh source file. This rule finds
14
+ * every `path:N` / `path:N-M` citation in a doc, resolves `path` to a real
15
+ * file, and warns when the citation clearly cannot be pointing at real
16
+ * content any more.
17
+ *
18
+ * What it checks (mechanical only, no symbol/AST resolution):
19
+ * - the target file does not exist (after resolution, see below)
20
+ * - a range's end line is before its start line (an inverted range)
21
+ * - the cited line (or the end of a range) is past the end of the file
22
+ * - the start line is blank
23
+ * - for non-markdown targets only, the start line is only a closing
24
+ * brace/bracket/paren (`}`, `)`, `]`, optionally with a trailing `,`/`;`)
25
+ * -- a common signature of "the cited block moved and this now points at
26
+ * the line right after it used to end"
27
+ *
28
+ * What it does NOT check: whether the cited line is semantically the right
29
+ * one (that needs the actual symbol; out of scope for a warn-only mechanical
30
+ * rule), nor citations without a file extension this rule recognises.
31
+ *
32
+ * Hard-wrapped prose. Some docs hard-wrap prose at a fixed column, which can
33
+ * split a hyphenated filename like `run-state-lifecycle-and-markers.md`
34
+ * across a line break right after the trailing `-`. `CITATION_RE` cannot
35
+ * (and should not) span the newline to reassemble the real citedPath, but
36
+ * without a guard it instead matches the phantom tail on its own
37
+ * (`markers.md:172`), which usually does not exist as a file and produces a
38
+ * false `missing-file`. A `CITATION_RE` match is skipped entirely (neither
39
+ * checked nor counted as a real citation) when it starts at column 0 of its
40
+ * line -- optionally after only whitespace or a list/quote marker
41
+ * (`-`, `*`, `>`, digits, `.`) -- and the previous line ends with `-` or
42
+ * `–`: the signature of a wrapped continuation, not a new citation.
43
+ *
44
+ * Unreadable targets. A resolved target file that exists but cannot be read
45
+ * (permission denied, and similar OS-level failures) is reported as a
46
+ * bundle-level `notice` tagged `unreadable-target` with the OS error code in
47
+ * `detail`, rather than throwing and aborting the whole check: a citation
48
+ * pointing at a file the checker itself cannot read is not evidence the
49
+ * citation is wrong.
50
+ *
51
+ * Porting decision (scope cut from the original): the original script also
52
+ * scanned any `docs/testing/*.md` file that an okf doc's frontmatter
53
+ * `sources` cited, an agent-grounding-specific convention (docs/testing is
54
+ * not part of the OKF spec) that is not exercised by any of the ported
55
+ * tests. This rule instead only scans the docs `loadBundle` already loaded
56
+ * for `ctx.bundleDir` -- a citing doc outside the bundle is out of scope.
57
+ *
58
+ * Path resolution. A bundle's prose citations are frequently written
59
+ * *relative to context* rather than as a full repo-root path, e.g. a doc's
60
+ * frontmatter `sources` list carries `packages/foo/src/cli.ts`, but the
61
+ * prose later just says `cli.ts:42` or `src/cli.ts:42` once the reader
62
+ * "knows" which package the section is about -- and source basenames can be
63
+ * reused across sibling packages, so a resolver that only tried
64
+ * repo-root-relative and doc-relative paths would either flag real,
65
+ * non-drifted citations as "missing file" or silently check the wrong
66
+ * same-named file. Resolution here instead tries, in order:
67
+ * 1. the citing doc's own frontmatter `sources` list, matched by exact
68
+ * suffix (`source === citedPath || source.endsWith('/' + citedPath)`),
69
+ * only when exactly one source matches -- the doc's author already
70
+ * disambiguated which physical file this citation means
71
+ * 2. for a citedPath with no `/` only: doc-relative, then each ancestor
72
+ * directory of the doc up to (and including) repoRoot, nearest first
73
+ * -- a bare filename like `README.md` almost always means "the README
74
+ * *for the package this doc lives in*", and a repo-root-relative
75
+ * lookup done first would instead silently resolve to an unrelated,
76
+ * differently-sized file of the same name that happens to sit at the
77
+ * repo root ("shadowing"). A citedPath containing a `/` skips straight
78
+ * to step 3: the caller already qualified enough of the path that
79
+ * "closest ancestor" guessing isn't needed.
80
+ * 3. repo-root-relative (`<repoRoot>/<citedPath>`)
81
+ * 4. doc-relative (`<dirname(doc)>/<citedPath>`) -- redundant with step 2
82
+ * for a no-`/` citedPath (already tried there), the effective first
83
+ * resolution attempt for a `/`-containing one
84
+ * 5. the nearest earlier citation *in the same doc* whose path contains a
85
+ * `/` and ends with the same suffix as citedPath (the "last full path
86
+ * mentioned" convention)
87
+ * 6. a repo-wide search for a file whose path ends with citedPath (or,
88
+ * for a bare filename, whose basename equals it)
89
+ * A citedPath starting with `/` is treated as out of scope (an absolute or
90
+ * placeholder path, e.g. inside a fabricated example stack trace) and
91
+ * skipped without a finding. When step 6 finds more than one candidate the
92
+ * citation is reported as a `notice`-severity "ambiguous target" finding
93
+ * rather than guessed at or false-flagged as missing (never counted toward
94
+ * `--strict`). A citation resolved by none of the above, with zero
95
+ * candidates at step 6, is reported as `missing-file`. A citedPath
96
+ * containing a `..` segment is rejected outright (`path-traversal-rejected`)
97
+ * without ever being resolved, so a malformed or hostile citation cannot
98
+ * walk resolution outside the repo.
99
+ *
100
+ * Continuation citations. Once a sentence has stated a full `path:N`
101
+ * citation, prose habitually repeats just the line (or range) for a later
102
+ * reference in the same sentence rather than retyping the path, in three
103
+ * forms:
104
+ * - `` `:N` `` or `` `:N-M` `` -- a bare colon-prefixed line/range
105
+ * - `` -`M` `` / `` –`M` `` -- a hyphen- or en-dash-led bare line, the
106
+ * tail half of a `` `path:N`-`M` `` split range
107
+ * - `` (`N`) `` -- a parenthesized bare line
108
+ * Each of these resolves against `governing`: the nearest *preceding*
109
+ * citation (full or itself a continuation) that resolved to a real file,
110
+ * scanned in document order. `governing` resets to none whenever the
111
+ * citation immediately before it failed to resolve, was ambiguous, or was
112
+ * out of scope (leading `/`) -- a continuation never silently inherits a
113
+ * stale or unrelated path from further up the doc. A continuation with no
114
+ * governing citation at all (e.g. very start of a doc) is skipped, not
115
+ * flagged: there is nothing to validate it against.
116
+ *
117
+ * A continuation is further split into two roles (see
118
+ * collectContinuationAtoms): "fresh" (a genuinely new start line, checked
119
+ * the same five ways as a full citation's start) versus "extension" (only
120
+ * ever the tail `M` of a split range whose start was already checked) --
121
+ * an extension gets *only* the range-bound checks (inverted-range,
122
+ * range-exceeds-file), never blank/closing-brace: that start line was
123
+ * already checked when it was first cited, so re-running it here would
124
+ * double-report the same drift, and a range legitimately ending on a
125
+ * closing brace is normal, not drift.
126
+ *
127
+ * Requires `ctx.repoRoot` (explicit `--repo-root` or auto-detected): target
128
+ * files usually live outside the bundle itself, so without a repo root
129
+ * there is no filesystem tree to resolve a citation against. Without one,
130
+ * this rule emits a single bundle-level notice, matching `sources-fresh`'s
131
+ * "not inside a git work tree" posture.
132
+ */
133
+ /**
134
+ * Anchored citations. A full citation (never a continuation or short-form
135
+ * atom -- see below) may carry an anchor directly after its range,
136
+ * `path:N-M#anchor`, e.g. `` `CHANGELOG.md:50-144#0.24.0` ``. Motivation: a
137
+ * CHANGELOG.md grows by insertion at the TOP (newest entry first), so every
138
+ * later entry's absolute line numbers shift on every release; the checks
139
+ * above (missing-file, inverted-range, range-exceeds-file, blank-start-line,
140
+ * closing-brace-start-line) are structurally blind to a citation that
141
+ * shifted a whole release section over and now lands, still non-blank and
142
+ * in-bounds, inside the WRONG section -- a citation `CHANGELOG.md:738-748`
143
+ * meant for the "0.7.4" entry that quietly now points at "0.7.3" prose is
144
+ * exactly as green as before the shift. An anchor closes that gap by
145
+ * pinning the citation to a piece of the target's own structure/content that
146
+ * the line-shift does not preserve automatically.
147
+ *
148
+ * Two anchor kinds, told apart by the raw anchor text (group 4 of
149
+ * `CITATION_RE`, see `parseAnchor`):
150
+ * - **Heading form** (bare, unquoted, e.g. `#0.24.0` or `#[0.24.0]`):
151
+ * the target's nearest *enclosing* Markdown heading -- see
152
+ * `findEnclosingHeading` -- must contain the anchor text, and no
153
+ * heading of the same or shallower level may start before the range's
154
+ * end line (i.e. the heading must actually enclose the whole range, not
155
+ * merely precede its start -- see `checkAnchor`). Deliberately capped at
156
+ * `ANCHOR_HEADING_MAX_LEVEL` (2): a Keep-a-Changelog CHANGELOG.md nests
157
+ * `## [x.y.z]` release headings around identically-named `### Added` /
158
+ * `### Changed` / `### Fixed` subsections repeated in every release;
159
+ * treating "nearest heading of any level" as the anchor target would
160
+ * make a heading anchor nearly useless here (matching the wrong
161
+ * release's own "Changed" subsection just as readily as the right
162
+ * one's), so subsection headings are transparent to this check and only
163
+ * level-1/level-2 headings are ever considered.
164
+ * - **String form** (double-quoted, e.g. `#"reproduction requirement"`):
165
+ * the anchor text must occur, verbatim, on at least one line of the
166
+ * cited range itself (`checkAnchor`'s string branch) -- "occurs inside
167
+ * it" rather than "encloses it". No heading structure required, so this
168
+ * form also works against a non-Markdown target (`.ts`/`.js`/...) where
169
+ * "nearest enclosing heading" has no meaning.
170
+ * An anchor mismatch is its own drift finding (`anchor-heading-not-found`,
171
+ * `anchor-heading-mismatch`, `anchor-heading-does-not-enclose`,
172
+ * `anchor-not-found-in-range`), checked only once the base start/range
173
+ * checks (blank-start-line, closing-brace-start-line, inverted-range,
174
+ * range-exceeds-file) already came back clean -- same "one problem per
175
+ * citation, base checks first" pattern `checkShortFormTarget` already uses
176
+ * for the block-boundary check.
177
+ *
178
+ * Backward compatible by construction: the `#anchor` suffix is optional in
179
+ * `CITATION_RE`, so an existing anchorless citation matches exactly as
180
+ * before and is checked exactly as before (this section adds a new check
181
+ * gated on the anchor being present, it does not change any existing one).
182
+ * Deliberately scoped to full citations only: a continuation (`` `:M` ``
183
+ * etc.) or a short-form `:N-M` never carries its own path, so there is
184
+ * nowhere natural to hang an anchor on one without inventing a second,
185
+ * detached syntax; the migration this rule was built for (CHANGELOG.md
186
+ * citations) is written as full citations throughout the corpus it targets.
187
+ *
188
+ * Rejected alternatives (see the PR/CHANGELOG for the fuller writeup):
189
+ * - Embedding the literal heading markup itself, e.g.
190
+ * `` `CHANGELOG.md:50-144#"## [0.24.0]"` `` -- verbatim-correct but
191
+ * `#`, `[`, `]`, and the space all need quoting/escaping right next to
192
+ * the citation, which reads worse in prose than a bare version token
193
+ * and gains nothing the structural heading-enclosure check does not
194
+ * already provide from the shorter form.
195
+ * - A detached anchor, e.g. a trailing parenthetical `(see "0.24.0")`
196
+ * elsewhere in the sentence -- unparseable without a second, separate
197
+ * grammar next to the existing continuation/short-form machinery, and
198
+ * easy to leave behind (or attach to the wrong citation) when a
199
+ * sentence is edited later.
200
+ * - A named-capture-group slug matching `sources-fresh`'s YAML shape --
201
+ * rejected because it would require a second citation site (frontmatter
202
+ * plus prose) to stay in sync, the exact class of drift this rule
203
+ * exists to catch.
204
+ */
205
+ const ANCHOR_HEADING_MAX_LEVEL = 2;
206
+ const MD_HEADING_RE = /^(#{1,6})\s+(.*)$/;
207
+ /**
208
+ * Parses `CITATION_RE`'s optional 4th capture group (the raw anchor text,
209
+ * including its surrounding quotes or brackets if any) into an `Anchor`, or
210
+ * `null` when the citation carried no `#anchor` suffix at all. A
211
+ * double-quoted raw value (`"..."`) is the string form, text taken verbatim
212
+ * between the quotes; anything else is the heading form, with a single
213
+ * wrapping `[...]` stripped (so `#[0.24.0]` and `#0.24.0` compare
214
+ * identically) -- compared as a plain substring against the heading's own
215
+ * raw text (`findEnclosingHeading`'s `text`, itself never stripped of any
216
+ * brackets it happens to carry), not a stripped copy of it -- see the
217
+ * "Anchored citations" doc block above.
218
+ */
219
+ function parseAnchor(raw) {
220
+ if (!raw)
221
+ return null;
222
+ if (raw.length >= 2 && raw.startsWith('"') && raw.endsWith('"')) {
223
+ return { kind: "string", text: raw.slice(1, -1) };
224
+ }
225
+ const text = raw.startsWith("[") && raw.endsWith("]") ? raw.slice(1, -1) : raw;
226
+ return { kind: "heading", text };
227
+ }
228
+ /**
229
+ * 0-based line indices that fall inside a fenced code block (```` ``` ````
230
+ * or `~~~`, optionally with a trailing language tag), delimiters included --
231
+ * the target-side twin of `computeFencedSpans` above, which does the same
232
+ * job for the *citing* doc's short-form matching. Anchor heading-search
233
+ * needs its own copy because it works from an already-split `lines` array
234
+ * (see `checkFullTarget`), not the raw `content` string `computeFencedSpans`
235
+ * takes, and because it must ignore a target's `# not a heading` sitting
236
+ * inside a fenced example exactly the same way a citing doc's own fences
237
+ * are already ignored for short-form matching -- without this, a `#`-led
238
+ * comment line inside e.g. a fenced shell example in the target is
239
+ * indistinguishable from a real Markdown heading to `MD_HEADING_RE`, and
240
+ * both `findEnclosingHeading` (picks the wrong "nearest" heading) and the
241
+ * enclosure walk in `checkAnchor` (treats it as a section boundary) would
242
+ * misfire on it.
243
+ */
244
+ function computeFencedLineIndices(lines) {
245
+ const fenced = new Set();
246
+ let fenceMarker;
247
+ let fenceStart = -1;
248
+ for (let i = 0; i < lines.length; i++) {
249
+ const trimmed = (lines[i] ?? "").trim();
250
+ if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
251
+ fenceMarker = trimmed.slice(0, 3);
252
+ fenceStart = i;
253
+ }
254
+ else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
255
+ for (let j = fenceStart; j <= i; j++)
256
+ fenced.add(j);
257
+ fenceMarker = undefined;
258
+ fenceStart = -1;
259
+ }
260
+ }
261
+ if (fenceMarker && fenceStart >= 0) {
262
+ for (let j = fenceStart; j < lines.length; j++)
263
+ fenced.add(j);
264
+ }
265
+ return fenced;
266
+ }
267
+ /**
268
+ * Nearest Markdown heading at or before `startLine` (1-based), considering
269
+ * only heading levels up to `ANCHOR_HEADING_MAX_LEVEL` -- see the "Anchored
270
+ * citations" doc block above for why subsection headings are transparent to
271
+ * this search. `fencedLines` (see `computeFencedLineIndices`) excludes any
272
+ * line inside a fenced code block from matching, so a `#`-led comment
273
+ * inside a fenced example is never mistaken for a heading. Returns `null`
274
+ * when no such heading precedes `startLine` at all (e.g. the citation
275
+ * lands above the target's first release heading).
276
+ */
277
+ function findEnclosingHeading(lines, startLine, fencedLines) {
278
+ for (let i = startLine - 1; i >= 0; i--) {
279
+ if (fencedLines.has(i))
280
+ continue;
281
+ const m = (lines[i] ?? "").match(MD_HEADING_RE);
282
+ if (m && m[1].length <= ANCHOR_HEADING_MAX_LEVEL) {
283
+ return { level: m[1].length, text: m[2].trim(), lineNo: i + 1 };
284
+ }
285
+ }
286
+ return null;
287
+ }
288
+ /**
289
+ * Checks an anchored full citation's anchor against its already-resolved,
290
+ * already-range-checked target -- see the "Anchored citations" doc block
291
+ * above for the two forms' semantics. `endLine` is the citation's end line,
292
+ * or its start line for a single-line citation (a size-1 range).
293
+ */
294
+ function checkAnchor(anchor, startLine, endLine, lines) {
295
+ if (anchor.kind === "string") {
296
+ for (let i = startLine - 1; i <= endLine - 1 && i < lines.length; i++) {
297
+ if ((lines[i] ?? "").includes(anchor.text))
298
+ return null;
299
+ }
300
+ return {
301
+ rule: "anchor-not-found-in-range",
302
+ message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
303
+ };
304
+ }
305
+ const fencedLines = computeFencedLineIndices(lines);
306
+ const heading = findEnclosingHeading(lines, startLine, fencedLines);
307
+ if (!heading) {
308
+ return {
309
+ rule: "anchor-heading-not-found",
310
+ message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor against; use a string anchor (#"...") instead against a target with no heading structure`,
311
+ };
312
+ }
313
+ if (!heading.text.includes(anchor.text)) {
314
+ return {
315
+ rule: "anchor-heading-mismatch",
316
+ message: `nearest enclosing heading ("${heading.text}") does not contain anchor "${anchor.text}"`,
317
+ };
318
+ }
319
+ for (let i = heading.lineNo; i <= endLine - 1; i++) {
320
+ if (fencedLines.has(i))
321
+ continue;
322
+ const m = (lines[i] ?? "").match(MD_HEADING_RE);
323
+ if (m && m[1].length <= heading.level) {
324
+ return {
325
+ rule: "anchor-heading-does-not-enclose",
326
+ message: `range extends past its enclosing heading's section (next heading "${m[2].trim()}" at line ${i + 1})`,
327
+ };
328
+ }
329
+ }
330
+ return null;
331
+ }
332
+ const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
333
+ // Continuation citation forms (see the "Continuation citations" doc block
334
+ // above). Each requires the backtick delimiter as part of the match so it
335
+ // can never overlap a CITATION_RE match: a full citation's regex match
336
+ // never includes the surrounding backticks, and none of these three
337
+ // require a `path.ext` prefix before the digits.
338
+ const CONT_COLON_RE = /`:(\d+)(?:-(\d+))?`/g;
339
+ const CONT_DASH_RE = /[-–]`(\d+)`/g;
340
+ const CONT_PAREN_RE = /\(`(\d+)`\)/g;
341
+ const CLOSING_ONLY_EXTS = new Set(["ts", "js", "mjs", "yml", "yaml", "json"]);
342
+ const CLOSING_BRACE_RE = /^[)\]}][;,]?$/;
343
+ // Short-form (paragraph-bound) citation form, see the "Short-form
344
+ // citations" doc block below. Requires an explicit N-M range: a bare single
345
+ // number (`:5`) is not matched -- see that doc block for why. Only the
346
+ // colon form is collected; a bare `(N-M)` is never a short-form citation
347
+ // candidate at all -- see the same doc block for why the paren form was
348
+ // dropped rather than gated.
349
+ const SHORT_FORM_COLON_RE = /:(\d+)-(\d+)/g;
350
+ // Test-file block-boundary check (see checkRangeBoundary): a citation's
351
+ // range into a `.test.ts`/`.spec.ts` (or `.js`/`.mjs` equivalent) target
352
+ // must start on a describe(/it( head line and end on its matching closing
353
+ // `});` line.
354
+ const TEST_FILE_RE = /\.(test|spec)\.(ts|js|mjs)$/i;
355
+ const TEST_HEAD_LINE_RE = /^\s*(?:describe|it)\s*\(/;
356
+ const TEST_CLOSING_LINE_RE = /^\s*\}\)\s*;\s*$/;
357
+ // Markdown block-boundary check (see checkRangeBoundary): a range boundary
358
+ // line that is nothing but a bracket (open or close), optionally with a
359
+ // trailing `,`/`;`, is always a drift signal. A bare code-fence delimiter is
360
+ // its own, separate signal (see MD_FENCE_DELIM_RE): unlike a bracket, a
361
+ // fence line is sometimes the deliberately-correct start of a citation (see
362
+ // checkRangeBoundary's markdown branch for the opening-fence exception).
363
+ const MD_BARE_BRACKET_RE = /^(?:[)\]}][;,]?|[[({])$/;
364
+ const MD_FENCE_DELIM_RE = /^(?:`{3,}\S*|~{3,}\S*)$/;
365
+ const EXCLUDED_DIRS = new Set([
366
+ "node_modules",
367
+ ".git",
368
+ "dist",
369
+ "build",
370
+ "coverage",
371
+ ".next",
372
+ ".turbo",
373
+ // Ported 1:1 from the original script's own disposable test-fixtures
374
+ // exclusion, kept so the ported EXCLUDED_DIRS regression test carries
375
+ // over verbatim; harmless for any consuming repo that has no directory
376
+ // with this name.
377
+ "okf-citations-resolve-fixtures",
378
+ ]);
379
+ function isFile(p) {
380
+ try {
381
+ return fs.statSync(p).isFile();
382
+ }
383
+ catch {
384
+ return false;
385
+ }
386
+ }
387
+ /** True when citedPath has a literal `..` path segment. */
388
+ function hasParentSegment(citedPath) {
389
+ return citedPath.split("/").includes("..");
390
+ }
391
+ /**
392
+ * Repo-wide search for files with an exact basename, memoized per root
393
+ * within `cache`. `cache` is built lazily and scoped to a single
394
+ * `citationsResolveRule.run(ctx)` invocation (see the rule's `run`), not
395
+ * held at module scope, so an in-process caller running the check
396
+ * repeatedly (e.g. in a long-lived process, or a test suite that edits
397
+ * fixtures between runs) never sees a stale index from an earlier run.
398
+ */
399
+ function findByBasename(cache, root, basename) {
400
+ let index = cache.get(root);
401
+ if (!index) {
402
+ index = new Map();
403
+ const found = index;
404
+ const walk = (dir) => {
405
+ let entries;
406
+ try {
407
+ entries = fs.readdirSync(dir, { withFileTypes: true });
408
+ }
409
+ catch {
410
+ return;
411
+ }
412
+ for (const entry of entries) {
413
+ if (entry.name.startsWith(".") && entry.name !== ".github")
414
+ continue;
415
+ if (EXCLUDED_DIRS.has(entry.name))
416
+ continue;
417
+ const full = path.join(dir, entry.name);
418
+ if (entry.isDirectory()) {
419
+ walk(full);
420
+ }
421
+ else if (entry.isFile()) {
422
+ const list = found.get(entry.name) ?? [];
423
+ list.push(full);
424
+ found.set(entry.name, list);
425
+ }
426
+ }
427
+ };
428
+ walk(root);
429
+ cache.set(root, index);
430
+ }
431
+ return index.get(basename) ?? [];
432
+ }
433
+ /**
434
+ * Finds the nearest citation earlier in the same doc whose cited path
435
+ * contains a `/` and ends with the same suffix as `citedPath`, the "full
436
+ * path was mentioned earlier in this section/doc" convention.
437
+ */
438
+ function findPriorQualifiedCitation(content, beforeIndex, citedPath) {
439
+ const suffix = "/" + citedPath;
440
+ let best = null;
441
+ let bestIndex = -1;
442
+ const re = new RegExp(CITATION_RE.source, "g");
443
+ let m;
444
+ while ((m = re.exec(content)) !== null) {
445
+ if (m.index >= beforeIndex)
446
+ break;
447
+ const candidate = m[1];
448
+ if (candidate === citedPath)
449
+ continue; // not more qualified than itself
450
+ if (candidate.includes("/") &&
451
+ (candidate === citedPath || candidate.endsWith(suffix))) {
452
+ if (m.index > bestIndex) {
453
+ bestIndex = m.index;
454
+ best = candidate;
455
+ }
456
+ }
457
+ }
458
+ return best;
459
+ }
460
+ /**
461
+ * For a bare (no `/`) citedPath, tries doc-relative first, then each
462
+ * ancestor directory of the doc, nearest first, up to and including `root`.
463
+ * Returns the first real file found, or `null`. See the "Path resolution"
464
+ * doc block above (step 2) for why this runs before the plain
465
+ * repo-root-relative lookup: a same-named file closer to the citing doc
466
+ * (e.g. a package's own `README.md`) should win over one that merely
467
+ * happens to also exist at the repo root.
468
+ */
469
+ function resolveViaAncestorClimb(root, docAbsPath, citedPath) {
470
+ const resolvedRoot = path.resolve(root);
471
+ let dir = path.dirname(docAbsPath);
472
+ for (;;) {
473
+ const candidate = path.resolve(dir, citedPath);
474
+ if (isFile(candidate))
475
+ return candidate;
476
+ if (path.resolve(dir) === resolvedRoot)
477
+ return null;
478
+ const parent = path.dirname(dir);
479
+ if (parent === dir)
480
+ return null; // reached the filesystem root
481
+ dir = parent;
482
+ }
483
+ }
484
+ /**
485
+ * Resolves a citation's path to a single real file. Returns `{ skip: true }`
486
+ * for a citedPath out of scope (leading `/`), `{ path }` on a definitive
487
+ * single resolution, `{ ambiguous: true, candidates }` when more than one
488
+ * plausible target exists, or `null` when nothing matches. Callers must
489
+ * reject a citedPath with a `..` segment (see hasParentSegment) before
490
+ * calling this; it is not re-checked here.
491
+ */
492
+ function resolveCitation(cache, root, docAbsPath, docContent, docSources, citedPath, matchIndex) {
493
+ if (citedPath.startsWith("/")) {
494
+ return { skip: true };
495
+ }
496
+ const sourceMatches = docSources.filter((s) => s === citedPath || s.endsWith("/" + citedPath));
497
+ if (sourceMatches.length === 1) {
498
+ const candidate = path.resolve(root, sourceMatches[0]);
499
+ if (isFile(candidate))
500
+ return { path: candidate };
501
+ }
502
+ if (!citedPath.includes("/")) {
503
+ const viaAncestor = resolveViaAncestorClimb(root, docAbsPath, citedPath);
504
+ if (viaAncestor)
505
+ return { path: viaAncestor };
506
+ }
507
+ const rootRelative = path.resolve(root, citedPath);
508
+ if (isFile(rootRelative))
509
+ return { path: rootRelative };
510
+ const docRelative = path.resolve(path.dirname(docAbsPath), citedPath);
511
+ if (isFile(docRelative))
512
+ return { path: docRelative };
513
+ const prior = findPriorQualifiedCitation(docContent, matchIndex, citedPath);
514
+ if (prior) {
515
+ const candidate = path.resolve(root, prior);
516
+ if (isFile(candidate))
517
+ return { path: candidate };
518
+ }
519
+ const base = citedPath.includes("/")
520
+ ? citedPath.split("/").pop()
521
+ : citedPath;
522
+ const bySuffix = findByBasename(cache, root, base).filter((m) => {
523
+ const normalized = m.split(path.sep).join("/");
524
+ return (normalized === citedPath ||
525
+ normalized.endsWith("/" + citedPath) ||
526
+ !citedPath.includes("/"));
527
+ });
528
+ if (bySuffix.length === 1)
529
+ return { path: bySuffix[0] };
530
+ if (bySuffix.length > 1) {
531
+ return {
532
+ ambiguous: true,
533
+ candidates: bySuffix.map((m) => path.relative(root, m)),
534
+ };
535
+ }
536
+ return null;
537
+ }
538
+ function splitLines(content) {
539
+ const lines = content.split("\n");
540
+ if (lines.length > 0 &&
541
+ lines[lines.length - 1] === "" &&
542
+ content.endsWith("\n")) {
543
+ lines.pop();
544
+ }
545
+ return lines;
546
+ }
547
+ // Range-bound checks shared by a full citation's own range (via
548
+ // checkTarget below) and a cont-ext atom's extension (via
549
+ // checkRangeBoundOnly): a citation's end before its start, or either bound
550
+ // past the end of the file. Both are pure "does this range make sense"
551
+ // checks, independent of what the start line's content actually is.
552
+ function checkRangeBound(startLine, endLine, lineCount) {
553
+ if (endLine !== null && endLine < startLine) {
554
+ return {
555
+ rule: "inverted-range",
556
+ message: `range end (${endLine}) is before its start (${startLine})`,
557
+ };
558
+ }
559
+ const last = endLine ?? startLine;
560
+ if (startLine > lineCount || last > lineCount) {
561
+ return {
562
+ rule: "range-exceeds-file",
563
+ message: `citation exceeds file length (${lineCount} line(s))`,
564
+ };
565
+ }
566
+ return null;
567
+ }
568
+ /**
569
+ * Reads a resolved target's content, or reports why it couldn't be read
570
+ * (permission denied, and similar OS-level failures short of the file not
571
+ * existing at all -- `resolveCitation` already confirmed the target exists
572
+ * via `isFile`/`fs.statSync`, which does not require read permission).
573
+ * Returned as a `Problem` with `rule: "unreadable-target"` and `code` set to
574
+ * the OS error code, so callers can route it to a `notice`-severity finding
575
+ * (see pushUnreadable) instead of the usual `warning`-severity drift finding
576
+ * -- an unreadable file is not evidence the citation itself is wrong.
577
+ */
578
+ function readTarget(resolvedPath) {
579
+ try {
580
+ return { content: fs.readFileSync(resolvedPath, "utf8") };
581
+ }
582
+ catch (err) {
583
+ const code = err.code ?? "UNKNOWN";
584
+ return {
585
+ rule: "unreadable-target",
586
+ message: "target file exists but could not be read",
587
+ code,
588
+ };
589
+ }
590
+ }
591
+ /**
592
+ * The non-I/O half of checkTarget: every check that only needs the
593
+ * target's already-read `lines`, not the filesystem. Split out so a caller
594
+ * that needs a second, different check against the same target (see
595
+ * checkShortFormTarget) can read the file once and reuse `lines` for both,
596
+ * instead of checkTarget re-reading it internally.
597
+ */
598
+ function checkTargetLines(citedPath, startLine, endLine, lines) {
599
+ const bound = checkRangeBound(startLine, endLine, lines.length);
600
+ if (bound)
601
+ return bound;
602
+ const startText = lines[startLine - 1] ?? "";
603
+ const trimmed = startText.trim();
604
+ if (trimmed === "") {
605
+ return { rule: "blank-start-line", message: "start line is blank" };
606
+ }
607
+ const ext = (citedPath.split(".").pop() ?? "").toLowerCase();
608
+ if (CLOSING_ONLY_EXTS.has(ext) && CLOSING_BRACE_RE.test(trimmed)) {
609
+ return {
610
+ rule: "closing-brace-start-line",
611
+ message: `start line is only a closing brace/bracket ("${trimmed}")`,
612
+ };
613
+ }
614
+ return null;
615
+ }
616
+ function checkTarget(citedPath, startLine, endLine, resolvedPath) {
617
+ const read = readTarget(resolvedPath);
618
+ if ("rule" in read)
619
+ return read;
620
+ return checkTargetLines(citedPath, startLine, endLine, splitLines(read.content));
621
+ }
622
+ /**
623
+ * A full citation's complete check: `checkTargetLines`'s existing checks
624
+ * (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
625
+ * closing-brace-start-line), plus, only when those all pass and the
626
+ * citation carried an anchor, the anchor check above (see the "Anchored
627
+ * citations" doc block). Reads the resolved target once, mirroring
628
+ * `checkShortFormTarget`'s reasoning for the block-boundary check. Anchors
629
+ * are full-citation-only (see that doc block for why), so `checkTarget`
630
+ * itself is untouched and still used as-is for a cont-fresh atom.
631
+ */
632
+ function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor) {
633
+ const read = readTarget(resolvedPath);
634
+ if ("rule" in read)
635
+ return read;
636
+ const lines = splitLines(read.content);
637
+ const base = checkTargetLines(citedPath, startLine, endLine, lines);
638
+ if (base)
639
+ return base;
640
+ if (!anchor)
641
+ return null;
642
+ return checkAnchor(anchor, startLine, endLine ?? startLine, lines);
643
+ }
644
+ // A cont-ext atom only ever extends the *end* of a range whose start line
645
+ // was already fully checked (blank / closing-brace) when it was cited as
646
+ // its own full citation or cont-fresh atom -- re-running checkTarget here
647
+ // would re-derive that same start-line check against the identical line
648
+ // and, on a real drift, double-report it as a second finding. This checks
649
+ // only whether the (possibly inverted, possibly out-of-file) range itself
650
+ // is sound.
651
+ function checkRangeBoundOnly(startLine, endLine, resolvedPath) {
652
+ const read = readTarget(resolvedPath);
653
+ if ("rule" in read)
654
+ return read;
655
+ const lineCount = splitLines(read.content).length;
656
+ return checkRangeBound(startLine, endLine, lineCount);
657
+ }
658
+ function isTestFile(citedPath) {
659
+ return TEST_FILE_RE.test(citedPath);
660
+ }
661
+ /**
662
+ * True when the line at `lines[lineIndex]` is a fence delimiter
663
+ * (```` ``` ```` or `~~~`, optionally with a trailing language tag) that
664
+ * *opens* a fenced code block, as opposed to closing one -- determined by
665
+ * replaying the same open/close state machine `stripFencedCode`-style
666
+ * scanners use from the top of the document, not by the line's own text
667
+ * (a bare closing fence and an untagged opening fence are lexically
668
+ * identical). Used by checkRangeBoundary's markdown branch: citing a
669
+ * fenced block starting at its own opening fence line is the natural,
670
+ * correct way to cite it, so that specific case is exempted from the
671
+ * fence-as-drift-signal check (see there).
672
+ */
673
+ function isFenceOpeningLine(lines, lineIndex) {
674
+ let inFence = false;
675
+ let fenceMarker;
676
+ for (let i = 0; i <= lineIndex; i++) {
677
+ const trimmed = (lines[i] ?? "").trim();
678
+ if (!inFence && MD_FENCE_DELIM_RE.test(trimmed)) {
679
+ if (i === lineIndex)
680
+ return true;
681
+ inFence = true;
682
+ fenceMarker = trimmed.slice(0, 3);
683
+ }
684
+ else if (inFence && fenceMarker && trimmed.startsWith(fenceMarker)) {
685
+ if (i === lineIndex)
686
+ return false;
687
+ inFence = false;
688
+ fenceMarker = undefined;
689
+ }
690
+ }
691
+ return false;
692
+ }
693
+ /**
694
+ * True when `lines[endLineIndex]` is the closing delimiter that matches
695
+ * the *opening* fence at `lines[startLineIndex]` (the caller must already
696
+ * have confirmed the start line is a genuine opening fence, via
697
+ * isFenceOpeningLine, before calling this) -- the first line after the
698
+ * opener whose trimmed text starts with the same fence marker. Used by
699
+ * checkRangeBoundary's markdown branch to also exempt a range's END from
700
+ * the fence-as-drift-signal check when the range legitimately cites a
701
+ * whole fenced code block from its own opening delimiter to its own
702
+ * closing delimiter: the natural, correct way to cite such a block, not
703
+ * drift, the same reasoning the start-side exception already applies.
704
+ */
705
+ function isMatchingFenceClosingLine(lines, startLineIndex, endLineIndex) {
706
+ const fenceMarker = (lines[startLineIndex] ?? "").trim().slice(0, 3);
707
+ for (let i = startLineIndex + 1; i < lines.length; i++) {
708
+ const trimmed = (lines[i] ?? "").trim();
709
+ if (trimmed.startsWith(fenceMarker)) {
710
+ return i === endLineIndex;
711
+ }
712
+ }
713
+ return false;
714
+ }
715
+ /**
716
+ * Additional block-boundary check for a short-form citation's range (see
717
+ * the "Short-form citations" doc block below), layered on top of
718
+ * checkTarget's existing per-start-line checks via checkShortFormTarget.
719
+ * Deliberately scoped to short-form citations only, not wired into
720
+ * checkTarget/checkRangeBoundOnly (the full/continuation citation paths):
721
+ * applying it there too was tried first and rejected -- the real bundle
722
+ * this rule was built against has legitimate full citations into test
723
+ * files that cite a couple of arbitrary lines (e.g. two lines of a shared
724
+ * regex definition), not a describe/it block, and flagging those as newly
725
+ * broken would have regressed the existing 0-warning baseline. Short-form
726
+ * citations are the demonstrated motivating case (a paragraph naming a
727
+ * test file once, then citing several of its describe/it blocks by bare
728
+ * range alone), so the check is scoped to exactly that mechanism.
729
+ *
730
+ * Test-file targets (`.test.ts`/`.spec.ts`, and their `.js`/`.mjs`
731
+ * equivalents): the range must start on a `describe(`/`it(` head line and
732
+ * end on a matching closing `});` line. Severity split, by evidence
733
+ * strength: a wrong START line is a
734
+ * *warning* (`test-range-start-not-head`) -- the range beginning somewhere
735
+ * other than a block head is strong drift evidence. A range whose start IS
736
+ * correct but whose END is not the matching `});` is only a *notice*
737
+ * (`test-range-end-not-closing`): this also matches a deliberate partial
738
+ * citation (citing from a block's head to partway through it), which is not
739
+ * drift. The start is checked first and returned alone, matching
740
+ * checkTarget's existing single-problem-per-citation pattern (never both at
741
+ * once).
742
+ *
743
+ * Markdown targets: mechanical verification of "is this still the same
744
+ * block" is far less reliable for prose than for TS/JS brace structure, so
745
+ * this is a notice, not a warning (see this rule's own risk note on
746
+ * markdown false positives). The range's start or end line landing on a
747
+ * bare bracket (MD_BARE_BRACKET_RE) is always a heads-up that the boundary
748
+ * likely drifted onto structural punctuation rather than real prose. A bare
749
+ * code-fence delimiter (MD_FENCE_DELIM_RE) is also always a drift signal at
750
+ * either boundary, with two exceptions carved out for the one legitimate
751
+ * way to cite a whole fenced code block by its own delimiters: the START is
752
+ * exempted when that line is itself a genuine *opening* fence (see
753
+ * isFenceOpeningLine), and, only when the start already qualified for that
754
+ * exemption, the END is exempted when it is that same fence's matching
755
+ * *closing* delimiter (see isMatchingFenceClosingLine) -- citing a fenced
756
+ * block from its own opening delimiter through its own closing delimiter is
757
+ * the natural, correct way to cite it, not drift.
758
+ */
759
+ function checkRangeBoundary(citedPath, startLine, endLine, lines) {
760
+ if (isTestFile(citedPath)) {
761
+ const startText = lines[startLine - 1] ?? "";
762
+ if (!TEST_HEAD_LINE_RE.test(startText)) {
763
+ return {
764
+ rule: "test-range-start-not-head",
765
+ message: `range start is not a "describe(" or "it(" head line ("${startText.trim()}")`,
766
+ };
767
+ }
768
+ const endText = lines[endLine - 1] ?? "";
769
+ if (!TEST_CLOSING_LINE_RE.test(endText)) {
770
+ return {
771
+ rule: "test-range-end-not-closing",
772
+ message: `range end is not a matching closing "});" line ("${endText.trim()}")`,
773
+ severity: "notice",
774
+ };
775
+ }
776
+ return null;
777
+ }
778
+ if (citedPath.toLowerCase().endsWith(".md")) {
779
+ const startTrim = (lines[startLine - 1] ?? "").trim();
780
+ const startIsFence = MD_FENCE_DELIM_RE.test(startTrim);
781
+ if (MD_BARE_BRACKET_RE.test(startTrim) ||
782
+ (startIsFence && !isFenceOpeningLine(lines, startLine - 1))) {
783
+ return {
784
+ rule: "markdown-range-boundary-bracket-or-fence",
785
+ message: `range start is a bare bracket/fence line ("${startTrim}")`,
786
+ severity: "notice",
787
+ };
788
+ }
789
+ const endTrim = (lines[endLine - 1] ?? "").trim();
790
+ const endIsFence = MD_FENCE_DELIM_RE.test(endTrim);
791
+ const endIsMatchingClose = startIsFence &&
792
+ endIsFence &&
793
+ isMatchingFenceClosingLine(lines, startLine - 1, endLine - 1);
794
+ if (MD_BARE_BRACKET_RE.test(endTrim) ||
795
+ (endIsFence && !endIsMatchingClose)) {
796
+ return {
797
+ rule: "markdown-range-boundary-bracket-or-fence",
798
+ message: `range end is a bare bracket/fence line ("${endTrim}")`,
799
+ severity: "notice",
800
+ };
801
+ }
802
+ return null;
803
+ }
804
+ return null;
805
+ }
806
+ /**
807
+ * A short-form citation's full check: checkTarget's existing checks
808
+ * (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
809
+ * closing-brace-start-line), plus, only when those all pass, the
810
+ * test-file/markdown block-boundary check above (see checkRangeBoundary).
811
+ * Reads the resolved target from disk exactly once (a dogfood bundle can
812
+ * run this against the same target file a dozen-plus times for one
813
+ * compound short-form list) and reuses the same `lines` array for both
814
+ * checks, instead of checkTarget and checkRangeBoundary each reading and
815
+ * re-splitting it independently.
816
+ */
817
+ function checkShortFormTarget(citedPath, startLine, endLine, resolvedPath) {
818
+ const read = readTarget(resolvedPath);
819
+ if ("rule" in read)
820
+ return read;
821
+ const lines = splitLines(read.content);
822
+ const base = checkTargetLines(citedPath, startLine, endLine, lines);
823
+ if (base)
824
+ return base;
825
+ return checkRangeBoundary(citedPath, startLine, endLine, lines);
826
+ }
827
+ /**
828
+ * Collects every continuation-citation atom (see the "Continuation
829
+ * citations" doc block above) in `content`, sorted by document position.
830
+ *
831
+ * Each atom is tagged with a role:
832
+ * - "cont-fresh": establishes a new start line (optionally with its own
833
+ * embedded end, e.g. `` `:75-78` ``), checked the same way as a full
834
+ * citation's start (blank / closing-brace / range-exceeds).
835
+ * - "cont-ext": purely extends the *end* of whatever start line came
836
+ * immediately before it (a `` -`M` `` / `` –`M` `` tail, or a
837
+ * `` `:M` `` directly preceded by a `-`/`–`). Ending a range on a
838
+ * closing brace is completely normal, so an extension is checked ONLY
839
+ * for the range-bound checks (inverted-range, range-exceeds-file),
840
+ * never blank/closing-brace.
841
+ * A colon-form match (`` `:N` ``) is "cont-ext" exactly when the nearest
842
+ * non-whitespace character before its opening backtick is `-` or `–`;
843
+ * otherwise it is "cont-fresh". Dash-form (`` -`M` ``/`` –`M` ``) is
844
+ * always "cont-ext" by construction. Paren-form (`` (`N`) ``) is always
845
+ * "cont-fresh".
846
+ */
847
+ function collectContinuationAtoms(content) {
848
+ const atoms = [];
849
+ const colonRe = new RegExp(CONT_COLON_RE.source, "g");
850
+ let m;
851
+ while ((m = colonRe.exec(content)) !== null) {
852
+ const before = content.slice(0, m.index).trimEnd();
853
+ if (/[-–]$/.test(before)) {
854
+ atoms.push({
855
+ kind: "cont-ext",
856
+ index: m.index,
857
+ value: m[2] ? Number(m[2]) : Number(m[1]),
858
+ });
859
+ }
860
+ else {
861
+ atoms.push({
862
+ kind: "cont-fresh",
863
+ index: m.index,
864
+ startLine: Number(m[1]),
865
+ endLine: m[2] ? Number(m[2]) : null,
866
+ });
867
+ }
868
+ }
869
+ const dashRe = new RegExp(CONT_DASH_RE.source, "g");
870
+ while ((m = dashRe.exec(content)) !== null) {
871
+ atoms.push({ kind: "cont-ext", index: m.index, value: Number(m[1]) });
872
+ }
873
+ const parenRe = new RegExp(CONT_PAREN_RE.source, "g");
874
+ while ((m = parenRe.exec(content)) !== null) {
875
+ atoms.push({
876
+ kind: "cont-fresh",
877
+ index: m.index,
878
+ startLine: Number(m[1]),
879
+ endLine: null,
880
+ });
881
+ }
882
+ return atoms;
883
+ }
884
+ /**
885
+ * True when the nearest non-whitespace text before `matchIndex` in
886
+ * `content` is a serial connective: the single character `,`, `;`, or `(`,
887
+ * or the word `and`/`or` (case-insensitive, word-boundary-matched so it
888
+ * does not fire on the tail of a longer word like "brand"). See the
889
+ * "Short-form citations" doc block above for why this is the gate.
890
+ */
891
+ function isSerialConnectivePreceded(content, matchIndex) {
892
+ const before = content.slice(0, matchIndex).trimEnd();
893
+ if (before === "")
894
+ return false;
895
+ const lastChar = before[before.length - 1];
896
+ if (lastChar === "," || lastChar === ";" || lastChar === "(")
897
+ return true;
898
+ return /\b(?:and|or)$/i.test(before);
899
+ }
900
+ function isWithinAnySpan(index, spans) {
901
+ return spans.some(([start, end]) => index >= start && index < end);
902
+ }
903
+ function collectShortFormMatches(content, excludedSpans) {
904
+ const out = [];
905
+ const colonRe = new RegExp(SHORT_FORM_COLON_RE.source, "g");
906
+ let m;
907
+ while ((m = colonRe.exec(content)) !== null) {
908
+ if (isWithinAnySpan(m.index, excludedSpans))
909
+ continue;
910
+ if (content[m.index - 1] === "`")
911
+ continue;
912
+ if (content[m.index + m[0].length] === "`")
913
+ continue;
914
+ if (!isSerialConnectivePreceded(content, m.index))
915
+ continue;
916
+ const startLine = Number(m[1]);
917
+ const endLine = Number(m[2]);
918
+ out.push({ index: m.index, startLine, endLine });
919
+ }
920
+ return out.sort((a, b) => a.index - b.index);
921
+ }
922
+ /**
923
+ * Char spans of every fenced code block in `content` (```` ``` ```` or
924
+ * `~~~`, optionally with a trailing language tag), each span running from
925
+ * the start of the opening fence line to the end of the closing fence line
926
+ * inclusive. An unterminated fence (no matching close before end of doc) is
927
+ * treated as running to the end of the content -- conservative, since an
928
+ * unterminated fence is itself a doc problem outside this rule's scope, not
929
+ * a reason to scan its contents for short-form citations.
930
+ */
931
+ function computeFencedSpans(content) {
932
+ const spans = [];
933
+ const lines = content.split("\n");
934
+ let offset = 0;
935
+ let fenceMarker;
936
+ let fenceStart = -1;
937
+ for (const line of lines) {
938
+ const trimmed = line.trim();
939
+ const lineEnd = offset + line.length;
940
+ if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
941
+ fenceMarker = trimmed.slice(0, 3);
942
+ fenceStart = offset;
943
+ }
944
+ else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
945
+ spans.push([fenceStart, lineEnd]);
946
+ fenceMarker = undefined;
947
+ fenceStart = -1;
948
+ }
949
+ offset = lineEnd + 1; // +1 for the newline joining this line to the next
950
+ }
951
+ if (fenceMarker && fenceStart >= 0) {
952
+ spans.push([fenceStart, content.length]);
953
+ }
954
+ return spans;
955
+ }
956
+ /**
957
+ * Char spans of every CommonMark-style indented code block in `content`: a
958
+ * maximal run of consecutive non-blank lines, each indented by at least
959
+ * four spaces or a leading tab, whose first line is preceded by a blank
960
+ * line or the start of the document (an indented code block cannot
961
+ * interrupt a paragraph). A blank line inside the run does not itself end
962
+ * it, matching CommonMark. Simplified relative to the full CommonMark
963
+ * spec (no list-item-context awareness); adequate for this mechanical,
964
+ * warn-only rule.
965
+ */
966
+ function computeIndentedCodeSpans(content) {
967
+ const spans = [];
968
+ const lines = content.split("\n");
969
+ let offset = 0;
970
+ let blockStart = -1;
971
+ let blockEnd = -1;
972
+ let prevBlank = true;
973
+ for (const line of lines) {
974
+ const lineStart = offset;
975
+ const lineEnd = offset + line.length;
976
+ const isBlank = line.trim() === "";
977
+ const isIndented = /^( {4,}|\t)/.test(line);
978
+ if (!isBlank && isIndented && (blockStart !== -1 || prevBlank)) {
979
+ if (blockStart === -1)
980
+ blockStart = lineStart;
981
+ blockEnd = lineEnd;
982
+ }
983
+ else if (isBlank && blockStart !== -1) {
984
+ // blank line inside an open block: keep it open, don't extend blockEnd
985
+ }
986
+ else if (!isBlank) {
987
+ if (blockStart !== -1)
988
+ spans.push([blockStart, blockEnd]);
989
+ blockStart = -1;
990
+ blockEnd = -1;
991
+ }
992
+ prevBlank = isBlank;
993
+ offset = lineEnd + 1;
994
+ }
995
+ if (blockStart !== -1)
996
+ spans.push([blockStart, blockEnd]);
997
+ return spans;
998
+ }
999
+ /**
1000
+ * Char spans of every inline code span in `content` (`` `...` ``, or a
1001
+ * longer run of backticks as the delimiter). This is a superset of the
1002
+ * narrower "immediately adjacent to a backtick" guard already applied in
1003
+ * collectShortFormMatches: that guard only rejects a match directly
1004
+ * touching a backtick, so a short form embedded further inside a longer
1005
+ * inline code span (e.g. `` `ports (1-3)` ``) was previously matched and
1006
+ * bound; this closes that gap.
1007
+ *
1008
+ * Confined to a single line (the pairing regex excludes `\n`): CommonMark
1009
+ * inline code spans cannot themselves span multiple lines in the sense
1010
+ * this rule cares about, but more importantly, an unmatched backtick run
1011
+ * (a typo, or a literal backtick in prose) previously paired greedily with
1012
+ * the *next* backtick run anywhere later in the whole document, silently
1013
+ * treating everything in between -- potentially several unrelated
1014
+ * sentences and any short-form citation among them -- as one giant inline
1015
+ * code span. Confining the match to a single line means a backtick run
1016
+ * with no partner on the same line produces no span at all, instead of
1017
+ * reaching across a newline for one.
1018
+ */
1019
+ function computeInlineCodeSpans(content) {
1020
+ const spans = [];
1021
+ const re = /(`+)[^`\n]*?\1/g;
1022
+ let m;
1023
+ while ((m = re.exec(content)) !== null) {
1024
+ spans.push([m.index, m.index + m[0].length]);
1025
+ }
1026
+ return spans;
1027
+ }
1028
+ /**
1029
+ * Char spans of every Markdown table row in `content`: a line whose
1030
+ * trimmed form starts and ends with `|`. Decision (documented in the
1031
+ * README): a short-form citation inside a table cell is never recognised,
1032
+ * the same way one inside a code span is not -- excluded here rather than
1033
+ * left to the plausibility gate, since a table cell's content is prose-like
1034
+ * and can otherwise carry a range shape the gate would not reject (e.g.
1035
+ * `| col (5-9) |`).
1036
+ */
1037
+ function computeTableRowSpans(content) {
1038
+ const spans = [];
1039
+ const lines = content.split("\n");
1040
+ let offset = 0;
1041
+ for (const line of lines) {
1042
+ const trimmed = line.trim();
1043
+ if (trimmed.startsWith("|") &&
1044
+ trimmed.endsWith("|") &&
1045
+ trimmed.length > 1) {
1046
+ spans.push([offset, offset + line.length]);
1047
+ }
1048
+ offset += line.length + 1;
1049
+ }
1050
+ return spans;
1051
+ }
1052
+ /**
1053
+ * All char spans short-form matching must never fire inside: fenced code,
1054
+ * indented code, inline code spans, and Markdown table rows. Computed once
1055
+ * per doc and combined with fullSpans (see scanDoc) via the existing
1056
+ * isWithinAnySpan helper -- the same mechanism a full citation's own span
1057
+ * already uses, not a second one.
1058
+ */
1059
+ function computeExcludedSpans(content) {
1060
+ return [
1061
+ ...computeFencedSpans(content),
1062
+ ...computeIndentedCodeSpans(content),
1063
+ ...computeInlineCodeSpans(content),
1064
+ ...computeTableRowSpans(content),
1065
+ ];
1066
+ }
1067
+ /**
1068
+ * Paragraph-start offsets in `content`, ascending, always including 0. A
1069
+ * paragraph boundary is a blank (empty or whitespace-only) line.
1070
+ */
1071
+ function computeParagraphStarts(content) {
1072
+ const starts = [0];
1073
+ const re = /\n[ \t]*\n+/g;
1074
+ let m;
1075
+ while ((m = re.exec(content)) !== null) {
1076
+ starts.push(m.index + m[0].length);
1077
+ }
1078
+ return starts;
1079
+ }
1080
+ /** The start offset of the paragraph containing `index` (see computeParagraphStarts). */
1081
+ function paragraphStartFor(starts, index) {
1082
+ let result = starts[0];
1083
+ for (const s of starts) {
1084
+ if (s > index)
1085
+ break;
1086
+ result = s;
1087
+ }
1088
+ return result;
1089
+ }
1090
+ /**
1091
+ * The nearest full citation strictly before `beforeIndex` and at or after
1092
+ * `paragraphStart` -- "the last target document named earlier in the same
1093
+ * paragraph" -- or null when none exists.
1094
+ */
1095
+ function findLastNamedTargetInParagraph(fullAtoms, paragraphStart, beforeIndex) {
1096
+ let best = null;
1097
+ for (const a of fullAtoms) {
1098
+ if (a.index >= paragraphStart && a.index < beforeIndex) {
1099
+ if (!best || a.index > best.index)
1100
+ best = a;
1101
+ }
1102
+ }
1103
+ return best;
1104
+ }
1105
+ function pushDrift(findings, file, citation, rule, message, resolvedTo, severity = "warning") {
1106
+ findings.push({
1107
+ ruleId: RULE_ID,
1108
+ severity,
1109
+ file,
1110
+ message: `\`${citation}\`: ${message} [${rule}]`,
1111
+ ...(resolvedTo ? { detail: `resolvedTo: ${resolvedTo}` } : {}),
1112
+ });
1113
+ }
1114
+ function pushAmbiguous(findings, file, citation, candidates) {
1115
+ findings.push({
1116
+ ruleId: RULE_ID,
1117
+ severity: "notice",
1118
+ file,
1119
+ message: `\`${citation}\`: ambiguous target, not evaluated [unresolved-ambiguous]`,
1120
+ detail: `candidates: ${candidates.join(", ")}`,
1121
+ });
1122
+ }
1123
+ function pushUnreadable(findings, file, citation, resolvedTo, code) {
1124
+ findings.push({
1125
+ ruleId: RULE_ID,
1126
+ severity: "notice",
1127
+ file,
1128
+ message: `\`${citation}\`: target file exists but could not be read [unreadable-target]`,
1129
+ detail: `resolvedTo: ${resolvedTo}, errorCode: ${code}`,
1130
+ });
1131
+ }
1132
+ /**
1133
+ * True when a `CITATION_RE` match at `matchIndex` is the phantom tail of a
1134
+ * filename hard-wrapped across a line break (see the "Hard-wrapped prose"
1135
+ * doc block above): the match starts at column 0 of its line, optionally
1136
+ * after only whitespace or a list/quote marker, and the previous line ends
1137
+ * with `-` or `–`.
1138
+ */
1139
+ function isWrappedPathContinuation(content, matchIndex) {
1140
+ const lineStart = content.lastIndexOf("\n", matchIndex - 1) + 1;
1141
+ if (lineStart === 0)
1142
+ return false; // no previous line to have wrapped from
1143
+ const prefix = content.slice(lineStart, matchIndex);
1144
+ if (!/^[\s>*.\d-]*$/.test(prefix))
1145
+ return false;
1146
+ const prevLineEnd = lineStart - 1; // index of the newline just before lineStart
1147
+ const prevLineStart = content.lastIndexOf("\n", prevLineEnd - 1) + 1;
1148
+ const prevLine = content.slice(prevLineStart, prevLineEnd);
1149
+ return /[-–]$/.test(prevLine);
1150
+ }
1151
+ function scanDoc(cache, root, bundleDir, doc) {
1152
+ const findings = [];
1153
+ const content = doc.raw;
1154
+ const sources = getValidSources(doc.frontmatter.parsed) ?? [];
1155
+ const docAbsPath = path.join(bundleDir, doc.relPath);
1156
+ const fullAtoms = [];
1157
+ // Char spans of every matched full citation, used to keep short-form
1158
+ // matching (see collectShortFormMatches) from re-matching the tail of a
1159
+ // real `path:N-M` citation as a bare short form.
1160
+ const fullSpans = [];
1161
+ const re = new RegExp(CITATION_RE.source, "g");
1162
+ let m;
1163
+ while ((m = re.exec(content)) !== null) {
1164
+ if (isWrappedPathContinuation(content, m.index))
1165
+ continue;
1166
+ fullAtoms.push({
1167
+ kind: "full",
1168
+ index: m.index,
1169
+ citedPath: m[1],
1170
+ startLine: Number(m[2]),
1171
+ endLine: m[3] ? Number(m[3]) : null,
1172
+ anchor: parseAnchor(m[4]),
1173
+ });
1174
+ fullSpans.push([m.index, m.index + m[0].length]);
1175
+ }
1176
+ const atoms = [...fullAtoms, ...collectContinuationAtoms(content)].sort((a, b) => a.index - b.index);
1177
+ // `governing`: nearest preceding citation (full or continuation) that
1178
+ // resolved to a real file; see the "Continuation citations" doc block
1179
+ // above for the reset rules. `lastStartLine`: the start line a "cont-ext"
1180
+ // atom extends into a range; tracks the most recent full or cont-fresh
1181
+ // atom's own startLine, scoped together with `governing`.
1182
+ let governing = null;
1183
+ let lastStartLine = null;
1184
+ for (const atom of atoms) {
1185
+ if (atom.kind === "cont-ext") {
1186
+ if (!governing || lastStartLine === null)
1187
+ continue; // nothing to extend
1188
+ const citation = `${governing.citedPath}:${lastStartLine}-${atom.value} (continuation)`;
1189
+ const problem = checkRangeBoundOnly(lastStartLine, atom.value, governing.resolvedPath);
1190
+ if (problem?.rule === "unreadable-target") {
1191
+ pushUnreadable(findings, doc.relPath, citation, path.relative(root, governing.resolvedPath), problem.code ?? "UNKNOWN");
1192
+ }
1193
+ else if (problem) {
1194
+ pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, governing.resolvedPath));
1195
+ }
1196
+ continue; // governing and lastStartLine both carry over unchanged
1197
+ }
1198
+ if (atom.kind === "cont-fresh") {
1199
+ if (!governing)
1200
+ continue; // nothing to validate a bare continuation against
1201
+ const { startLine, endLine } = atom;
1202
+ const citation = `${governing.citedPath}:${startLine}${endLine ? "-" + endLine : ""} (continuation)`;
1203
+ const problem = checkTarget(governing.citedPath, startLine, endLine, governing.resolvedPath);
1204
+ if (problem?.rule === "unreadable-target") {
1205
+ pushUnreadable(findings, doc.relPath, citation, path.relative(root, governing.resolvedPath), problem.code ?? "UNKNOWN");
1206
+ }
1207
+ else if (problem) {
1208
+ pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, governing.resolvedPath));
1209
+ }
1210
+ lastStartLine = startLine;
1211
+ continue; // governing (same file) carries over unchanged
1212
+ }
1213
+ const { citedPath, startLine, endLine, anchor } = atom;
1214
+ const citation = `${citedPath}:${startLine}${endLine ? "-" + endLine : ""}`;
1215
+ if (hasParentSegment(citedPath)) {
1216
+ pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
1217
+ governing = null;
1218
+ lastStartLine = null;
1219
+ continue;
1220
+ }
1221
+ const resolution = resolveCitation(cache, root, docAbsPath, content, sources, citedPath, atom.index);
1222
+ if (!resolution) {
1223
+ pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${citedPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
1224
+ governing = null;
1225
+ lastStartLine = null;
1226
+ continue;
1227
+ }
1228
+ if ("skip" in resolution) {
1229
+ governing = null;
1230
+ lastStartLine = null;
1231
+ continue;
1232
+ }
1233
+ if ("ambiguous" in resolution) {
1234
+ pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
1235
+ governing = null;
1236
+ lastStartLine = null;
1237
+ continue;
1238
+ }
1239
+ const problem = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor);
1240
+ if (problem?.rule === "unreadable-target") {
1241
+ pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
1242
+ }
1243
+ else if (problem) {
1244
+ pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
1245
+ }
1246
+ governing = { citedPath, resolvedPath: resolution.path };
1247
+ lastStartLine = startLine;
1248
+ }
1249
+ // Short-form (paragraph-bound) citations -- see that doc block above.
1250
+ // Deliberately independent of `governing`/`lastStartLine`: short-form
1251
+ // binding is paragraph-scoped by design, not chained through the
1252
+ // document-wide continuation state machine above. Reserved files
1253
+ // (index.md, log.md, see doc.isReserved) are skipped entirely: they are
1254
+ // append-only narrative journals, not `sources:`-driven reference docs,
1255
+ // and routinely narrate historical "old N-M -> new X-Y" line-number
1256
+ // deltas as prose data about past changes -- not live citations against
1257
+ // current content -- which this rule's bare-range matching cannot tell
1258
+ // apart from a real short-form citation. Full/continuation citations in
1259
+ // reserved files are still scanned as before; this carve-out is scoped
1260
+ // to short-form matching only.
1261
+ const shortFormMatches = doc.isReserved
1262
+ ? []
1263
+ : collectShortFormMatches(content, [
1264
+ ...fullSpans,
1265
+ ...computeExcludedSpans(content),
1266
+ ]);
1267
+ if (shortFormMatches.length > 0) {
1268
+ const paragraphStarts = computeParagraphStarts(content);
1269
+ const namedFullAtoms = [];
1270
+ for (const a of fullAtoms) {
1271
+ if (a.kind === "full") {
1272
+ namedFullAtoms.push({
1273
+ index: a.index,
1274
+ citedPath: a.citedPath,
1275
+ });
1276
+ }
1277
+ }
1278
+ for (const sf of shortFormMatches) {
1279
+ const paragraphStart = paragraphStartFor(paragraphStarts, sf.index);
1280
+ const target = findLastNamedTargetInParagraph(namedFullAtoms, paragraphStart, sf.index);
1281
+ const rangeLabel = `${sf.startLine}-${sf.endLine}`;
1282
+ if (!target) {
1283
+ pushDrift(findings, doc.relPath, `${rangeLabel} (short-form)`, "short-form-unbound", "no full `path:N` citation earlier in this paragraph to bind to", undefined, "notice");
1284
+ continue;
1285
+ }
1286
+ const targetPath = target.citedPath;
1287
+ const citation = `${targetPath}:${rangeLabel} (short-form)`;
1288
+ if (hasParentSegment(targetPath)) {
1289
+ // The full citation that named this target already reported
1290
+ // path-traversal-rejected for itself; not re-flagged a second time.
1291
+ continue;
1292
+ }
1293
+ const resolution = resolveCitation(cache, root, docAbsPath, content, sources, targetPath, sf.index);
1294
+ if (!resolution) {
1295
+ pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${targetPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
1296
+ continue;
1297
+ }
1298
+ if ("skip" in resolution)
1299
+ continue;
1300
+ if ("ambiguous" in resolution) {
1301
+ pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
1302
+ continue;
1303
+ }
1304
+ const problem = checkShortFormTarget(targetPath, sf.startLine, sf.endLine, resolution.path);
1305
+ if (problem?.rule === "unreadable-target") {
1306
+ pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
1307
+ }
1308
+ else if (problem) {
1309
+ pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path), problem.severity);
1310
+ }
1311
+ }
1312
+ }
1313
+ return findings;
1314
+ }
1315
+ export const citationsResolveRule = {
1316
+ id: RULE_ID,
1317
+ description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations) must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
1318
+ run(ctx) {
1319
+ if (!ctx.repoRoot) {
1320
+ // Never silently skip: mirrors sources-fresh's posture so a "clean"
1321
+ // run is never a fake pass when there is no filesystem tree to
1322
+ // resolve citation targets against.
1323
+ return [
1324
+ {
1325
+ ruleId: RULE_ID,
1326
+ severity: "notice",
1327
+ file: "",
1328
+ message: "citation resolution skipped: not inside a git work tree",
1329
+ },
1330
+ ];
1331
+ }
1332
+ const root = ctx.repoRoot;
1333
+ // Fresh per invocation: see findByBasename's doc comment for why this
1334
+ // is not held at module scope.
1335
+ const cache = new Map();
1336
+ return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc));
1337
+ },
1338
+ };
1339
+ //# sourceMappingURL=citations-resolve.js.map