okf-kit 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +286 -0
- package/README.md +58 -2
- package/dist/rules/citations-resolve.d.ts +2 -0
- package/dist/rules/citations-resolve.js +1339 -0
- package/dist/rules/citations-resolve.js.map +1 -0
- package/dist/rules/index.d.ts +2 -1
- package/dist/rules/index.js +3 -1
- package/dist/rules/index.js.map +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,1339 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { getValidSources } from "../util.js";
|
|
4
|
+
const RULE_ID = "citations-resolve";
|
|
5
|
+
/**
|
|
6
|
+
* Warn-only checker for `path:N[-M]` citations inside a bundle's docs (and
|
|
7
|
+
* their continuation forms, see below).
|
|
8
|
+
*
|
|
9
|
+
* Ported from agent-grounding's `scripts/okf-citations-resolve.mjs`
|
|
10
|
+
* (PR #185), which was a repo-local, no-dependency spike at closing a gap
|
|
11
|
+
* `sources-fresh` cannot: `sources-fresh` only compares a doc's `sources`
|
|
12
|
+
* list against file mtimes, so it is structurally blind to an edit that
|
|
13
|
+
* shifts line numbers inside a still-fresh source file. This rule finds
|
|
14
|
+
* every `path:N` / `path:N-M` citation in a doc, resolves `path` to a real
|
|
15
|
+
* file, and warns when the citation clearly cannot be pointing at real
|
|
16
|
+
* content any more.
|
|
17
|
+
*
|
|
18
|
+
* What it checks (mechanical only, no symbol/AST resolution):
|
|
19
|
+
* - the target file does not exist (after resolution, see below)
|
|
20
|
+
* - a range's end line is before its start line (an inverted range)
|
|
21
|
+
* - the cited line (or the end of a range) is past the end of the file
|
|
22
|
+
* - the start line is blank
|
|
23
|
+
* - for non-markdown targets only, the start line is only a closing
|
|
24
|
+
* brace/bracket/paren (`}`, `)`, `]`, optionally with a trailing `,`/`;`)
|
|
25
|
+
* -- a common signature of "the cited block moved and this now points at
|
|
26
|
+
* the line right after it used to end"
|
|
27
|
+
*
|
|
28
|
+
* What it does NOT check: whether the cited line is semantically the right
|
|
29
|
+
* one (that needs the actual symbol; out of scope for a warn-only mechanical
|
|
30
|
+
* rule), nor citations without a file extension this rule recognises.
|
|
31
|
+
*
|
|
32
|
+
* Hard-wrapped prose. Some docs hard-wrap prose at a fixed column, which can
|
|
33
|
+
* split a hyphenated filename like `run-state-lifecycle-and-markers.md`
|
|
34
|
+
* across a line break right after the trailing `-`. `CITATION_RE` cannot
|
|
35
|
+
* (and should not) span the newline to reassemble the real citedPath, but
|
|
36
|
+
* without a guard it instead matches the phantom tail on its own
|
|
37
|
+
* (`markers.md:172`), which usually does not exist as a file and produces a
|
|
38
|
+
* false `missing-file`. A `CITATION_RE` match is skipped entirely (neither
|
|
39
|
+
* checked nor counted as a real citation) when it starts at column 0 of its
|
|
40
|
+
* line -- optionally after only whitespace or a list/quote marker
|
|
41
|
+
* (`-`, `*`, `>`, digits, `.`) -- and the previous line ends with `-` or
|
|
42
|
+
* `–`: the signature of a wrapped continuation, not a new citation.
|
|
43
|
+
*
|
|
44
|
+
* Unreadable targets. A resolved target file that exists but cannot be read
|
|
45
|
+
* (permission denied, and similar OS-level failures) is reported as a
|
|
46
|
+
* bundle-level `notice` tagged `unreadable-target` with the OS error code in
|
|
47
|
+
* `detail`, rather than throwing and aborting the whole check: a citation
|
|
48
|
+
* pointing at a file the checker itself cannot read is not evidence the
|
|
49
|
+
* citation is wrong.
|
|
50
|
+
*
|
|
51
|
+
* Porting decision (scope cut from the original): the original script also
|
|
52
|
+
* scanned any `docs/testing/*.md` file that an okf doc's frontmatter
|
|
53
|
+
* `sources` cited, an agent-grounding-specific convention (docs/testing is
|
|
54
|
+
* not part of the OKF spec) that is not exercised by any of the ported
|
|
55
|
+
* tests. This rule instead only scans the docs `loadBundle` already loaded
|
|
56
|
+
* for `ctx.bundleDir` -- a citing doc outside the bundle is out of scope.
|
|
57
|
+
*
|
|
58
|
+
* Path resolution. A bundle's prose citations are frequently written
|
|
59
|
+
* *relative to context* rather than as a full repo-root path, e.g. a doc's
|
|
60
|
+
* frontmatter `sources` list carries `packages/foo/src/cli.ts`, but the
|
|
61
|
+
* prose later just says `cli.ts:42` or `src/cli.ts:42` once the reader
|
|
62
|
+
* "knows" which package the section is about -- and source basenames can be
|
|
63
|
+
* reused across sibling packages, so a resolver that only tried
|
|
64
|
+
* repo-root-relative and doc-relative paths would either flag real,
|
|
65
|
+
* non-drifted citations as "missing file" or silently check the wrong
|
|
66
|
+
* same-named file. Resolution here instead tries, in order:
|
|
67
|
+
* 1. the citing doc's own frontmatter `sources` list, matched by exact
|
|
68
|
+
* suffix (`source === citedPath || source.endsWith('/' + citedPath)`),
|
|
69
|
+
* only when exactly one source matches -- the doc's author already
|
|
70
|
+
* disambiguated which physical file this citation means
|
|
71
|
+
* 2. for a citedPath with no `/` only: doc-relative, then each ancestor
|
|
72
|
+
* directory of the doc up to (and including) repoRoot, nearest first
|
|
73
|
+
* -- a bare filename like `README.md` almost always means "the README
|
|
74
|
+
* *for the package this doc lives in*", and a repo-root-relative
|
|
75
|
+
* lookup done first would instead silently resolve to an unrelated,
|
|
76
|
+
* differently-sized file of the same name that happens to sit at the
|
|
77
|
+
* repo root ("shadowing"). A citedPath containing a `/` skips straight
|
|
78
|
+
* to step 3: the caller already qualified enough of the path that
|
|
79
|
+
* "closest ancestor" guessing isn't needed.
|
|
80
|
+
* 3. repo-root-relative (`<repoRoot>/<citedPath>`)
|
|
81
|
+
* 4. doc-relative (`<dirname(doc)>/<citedPath>`) -- redundant with step 2
|
|
82
|
+
* for a no-`/` citedPath (already tried there), the effective first
|
|
83
|
+
* resolution attempt for a `/`-containing one
|
|
84
|
+
* 5. the nearest earlier citation *in the same doc* whose path contains a
|
|
85
|
+
* `/` and ends with the same suffix as citedPath (the "last full path
|
|
86
|
+
* mentioned" convention)
|
|
87
|
+
* 6. a repo-wide search for a file whose path ends with citedPath (or,
|
|
88
|
+
* for a bare filename, whose basename equals it)
|
|
89
|
+
* A citedPath starting with `/` is treated as out of scope (an absolute or
|
|
90
|
+
* placeholder path, e.g. inside a fabricated example stack trace) and
|
|
91
|
+
* skipped without a finding. When step 6 finds more than one candidate the
|
|
92
|
+
* citation is reported as a `notice`-severity "ambiguous target" finding
|
|
93
|
+
* rather than guessed at or false-flagged as missing (never counted toward
|
|
94
|
+
* `--strict`). A citation resolved by none of the above, with zero
|
|
95
|
+
* candidates at step 6, is reported as `missing-file`. A citedPath
|
|
96
|
+
* containing a `..` segment is rejected outright (`path-traversal-rejected`)
|
|
97
|
+
* without ever being resolved, so a malformed or hostile citation cannot
|
|
98
|
+
* walk resolution outside the repo.
|
|
99
|
+
*
|
|
100
|
+
* Continuation citations. Once a sentence has stated a full `path:N`
|
|
101
|
+
* citation, prose habitually repeats just the line (or range) for a later
|
|
102
|
+
* reference in the same sentence rather than retyping the path, in three
|
|
103
|
+
* forms:
|
|
104
|
+
* - `` `:N` `` or `` `:N-M` `` -- a bare colon-prefixed line/range
|
|
105
|
+
* - `` -`M` `` / `` –`M` `` -- a hyphen- or en-dash-led bare line, the
|
|
106
|
+
* tail half of a `` `path:N`-`M` `` split range
|
|
107
|
+
* - `` (`N`) `` -- a parenthesized bare line
|
|
108
|
+
* Each of these resolves against `governing`: the nearest *preceding*
|
|
109
|
+
* citation (full or itself a continuation) that resolved to a real file,
|
|
110
|
+
* scanned in document order. `governing` resets to none whenever the
|
|
111
|
+
* citation immediately before it failed to resolve, was ambiguous, or was
|
|
112
|
+
* out of scope (leading `/`) -- a continuation never silently inherits a
|
|
113
|
+
* stale or unrelated path from further up the doc. A continuation with no
|
|
114
|
+
* governing citation at all (e.g. very start of a doc) is skipped, not
|
|
115
|
+
* flagged: there is nothing to validate it against.
|
|
116
|
+
*
|
|
117
|
+
* A continuation is further split into two roles (see
|
|
118
|
+
* collectContinuationAtoms): "fresh" (a genuinely new start line, checked
|
|
119
|
+
* the same five ways as a full citation's start) versus "extension" (only
|
|
120
|
+
* ever the tail `M` of a split range whose start was already checked) --
|
|
121
|
+
* an extension gets *only* the range-bound checks (inverted-range,
|
|
122
|
+
* range-exceeds-file), never blank/closing-brace: that start line was
|
|
123
|
+
* already checked when it was first cited, so re-running it here would
|
|
124
|
+
* double-report the same drift, and a range legitimately ending on a
|
|
125
|
+
* closing brace is normal, not drift.
|
|
126
|
+
*
|
|
127
|
+
* Requires `ctx.repoRoot` (explicit `--repo-root` or auto-detected): target
|
|
128
|
+
* files usually live outside the bundle itself, so without a repo root
|
|
129
|
+
* there is no filesystem tree to resolve a citation against. Without one,
|
|
130
|
+
* this rule emits a single bundle-level notice, matching `sources-fresh`'s
|
|
131
|
+
* "not inside a git work tree" posture.
|
|
132
|
+
*/
|
|
133
|
+
/**
|
|
134
|
+
* Anchored citations. A full citation (never a continuation or short-form
|
|
135
|
+
* atom -- see below) may carry an anchor directly after its range,
|
|
136
|
+
* `path:N-M#anchor`, e.g. `` `CHANGELOG.md:50-144#0.24.0` ``. Motivation: a
|
|
137
|
+
* CHANGELOG.md grows by insertion at the TOP (newest entry first), so every
|
|
138
|
+
* later entry's absolute line numbers shift on every release; the checks
|
|
139
|
+
* above (missing-file, inverted-range, range-exceeds-file, blank-start-line,
|
|
140
|
+
* closing-brace-start-line) are structurally blind to a citation that
|
|
141
|
+
* shifted a whole release section over and now lands, still non-blank and
|
|
142
|
+
* in-bounds, inside the WRONG section -- a citation `CHANGELOG.md:738-748`
|
|
143
|
+
* meant for the "0.7.4" entry that quietly now points at "0.7.3" prose is
|
|
144
|
+
* exactly as green as before the shift. An anchor closes that gap by
|
|
145
|
+
* pinning the citation to a piece of the target's own structure/content that
|
|
146
|
+
* the line-shift does not preserve automatically.
|
|
147
|
+
*
|
|
148
|
+
* Two anchor kinds, told apart by the raw anchor text (group 4 of
|
|
149
|
+
* `CITATION_RE`, see `parseAnchor`):
|
|
150
|
+
* - **Heading form** (bare, unquoted, e.g. `#0.24.0` or `#[0.24.0]`):
|
|
151
|
+
* the target's nearest *enclosing* Markdown heading -- see
|
|
152
|
+
* `findEnclosingHeading` -- must contain the anchor text, and no
|
|
153
|
+
* heading of the same or shallower level may start before the range's
|
|
154
|
+
* end line (i.e. the heading must actually enclose the whole range, not
|
|
155
|
+
* merely precede its start -- see `checkAnchor`). Deliberately capped at
|
|
156
|
+
* `ANCHOR_HEADING_MAX_LEVEL` (2): a Keep-a-Changelog CHANGELOG.md nests
|
|
157
|
+
* `## [x.y.z]` release headings around identically-named `### Added` /
|
|
158
|
+
* `### Changed` / `### Fixed` subsections repeated in every release;
|
|
159
|
+
* treating "nearest heading of any level" as the anchor target would
|
|
160
|
+
* make a heading anchor nearly useless here (matching the wrong
|
|
161
|
+
* release's own "Changed" subsection just as readily as the right
|
|
162
|
+
* one's), so subsection headings are transparent to this check and only
|
|
163
|
+
* level-1/level-2 headings are ever considered.
|
|
164
|
+
* - **String form** (double-quoted, e.g. `#"reproduction requirement"`):
|
|
165
|
+
* the anchor text must occur, verbatim, on at least one line of the
|
|
166
|
+
* cited range itself (`checkAnchor`'s string branch) -- "occurs inside
|
|
167
|
+
* it" rather than "encloses it". No heading structure required, so this
|
|
168
|
+
* form also works against a non-Markdown target (`.ts`/`.js`/...) where
|
|
169
|
+
* "nearest enclosing heading" has no meaning.
|
|
170
|
+
* An anchor mismatch is its own drift finding (`anchor-heading-not-found`,
|
|
171
|
+
* `anchor-heading-mismatch`, `anchor-heading-does-not-enclose`,
|
|
172
|
+
* `anchor-not-found-in-range`), checked only once the base start/range
|
|
173
|
+
* checks (blank-start-line, closing-brace-start-line, inverted-range,
|
|
174
|
+
* range-exceeds-file) already came back clean -- same "one problem per
|
|
175
|
+
* citation, base checks first" pattern `checkShortFormTarget` already uses
|
|
176
|
+
* for the block-boundary check.
|
|
177
|
+
*
|
|
178
|
+
* Backward compatible by construction: the `#anchor` suffix is optional in
|
|
179
|
+
* `CITATION_RE`, so an existing anchorless citation matches exactly as
|
|
180
|
+
* before and is checked exactly as before (this section adds a new check
|
|
181
|
+
* gated on the anchor being present, it does not change any existing one).
|
|
182
|
+
* Deliberately scoped to full citations only: a continuation (`` `:M` ``
|
|
183
|
+
* etc.) or a short-form `:N-M` never carries its own path, so there is
|
|
184
|
+
* nowhere natural to hang an anchor on one without inventing a second,
|
|
185
|
+
* detached syntax; the migration this rule was built for (CHANGELOG.md
|
|
186
|
+
* citations) is written as full citations throughout the corpus it targets.
|
|
187
|
+
*
|
|
188
|
+
* Rejected alternatives (see the PR/CHANGELOG for the fuller writeup):
|
|
189
|
+
* - Embedding the literal heading markup itself, e.g.
|
|
190
|
+
* `` `CHANGELOG.md:50-144#"## [0.24.0]"` `` -- verbatim-correct but
|
|
191
|
+
* `#`, `[`, `]`, and the space all need quoting/escaping right next to
|
|
192
|
+
* the citation, which reads worse in prose than a bare version token
|
|
193
|
+
* and gains nothing the structural heading-enclosure check does not
|
|
194
|
+
* already provide from the shorter form.
|
|
195
|
+
* - A detached anchor, e.g. a trailing parenthetical `(see "0.24.0")`
|
|
196
|
+
* elsewhere in the sentence -- unparseable without a second, separate
|
|
197
|
+
* grammar next to the existing continuation/short-form machinery, and
|
|
198
|
+
* easy to leave behind (or attach to the wrong citation) when a
|
|
199
|
+
* sentence is edited later.
|
|
200
|
+
* - A named-capture-group slug matching `sources-fresh`'s YAML shape --
|
|
201
|
+
* rejected because it would require a second citation site (frontmatter
|
|
202
|
+
* plus prose) to stay in sync, the exact class of drift this rule
|
|
203
|
+
* exists to catch.
|
|
204
|
+
*/
|
|
205
|
+
const ANCHOR_HEADING_MAX_LEVEL = 2;
|
|
206
|
+
const MD_HEADING_RE = /^(#{1,6})\s+(.*)$/;
|
|
207
|
+
/**
|
|
208
|
+
* Parses `CITATION_RE`'s optional 4th capture group (the raw anchor text,
|
|
209
|
+
* including its surrounding quotes or brackets if any) into an `Anchor`, or
|
|
210
|
+
* `null` when the citation carried no `#anchor` suffix at all. A
|
|
211
|
+
* double-quoted raw value (`"..."`) is the string form, text taken verbatim
|
|
212
|
+
* between the quotes; anything else is the heading form, with a single
|
|
213
|
+
* wrapping `[...]` stripped (so `#[0.24.0]` and `#0.24.0` compare
|
|
214
|
+
* identically) -- compared as a plain substring against the heading's own
|
|
215
|
+
* raw text (`findEnclosingHeading`'s `text`, itself never stripped of any
|
|
216
|
+
* brackets it happens to carry), not a stripped copy of it -- see the
|
|
217
|
+
* "Anchored citations" doc block above.
|
|
218
|
+
*/
|
|
219
|
+
function parseAnchor(raw) {
|
|
220
|
+
if (!raw)
|
|
221
|
+
return null;
|
|
222
|
+
if (raw.length >= 2 && raw.startsWith('"') && raw.endsWith('"')) {
|
|
223
|
+
return { kind: "string", text: raw.slice(1, -1) };
|
|
224
|
+
}
|
|
225
|
+
const text = raw.startsWith("[") && raw.endsWith("]") ? raw.slice(1, -1) : raw;
|
|
226
|
+
return { kind: "heading", text };
|
|
227
|
+
}
|
|
228
|
+
/**
|
|
229
|
+
* 0-based line indices that fall inside a fenced code block (```` ``` ````
|
|
230
|
+
* or `~~~`, optionally with a trailing language tag), delimiters included --
|
|
231
|
+
* the target-side twin of `computeFencedSpans` above, which does the same
|
|
232
|
+
* job for the *citing* doc's short-form matching. Anchor heading-search
|
|
233
|
+
* needs its own copy because it works from an already-split `lines` array
|
|
234
|
+
* (see `checkFullTarget`), not the raw `content` string `computeFencedSpans`
|
|
235
|
+
* takes, and because it must ignore a target's `# not a heading` sitting
|
|
236
|
+
* inside a fenced example exactly the same way a citing doc's own fences
|
|
237
|
+
* are already ignored for short-form matching -- without this, a `#`-led
|
|
238
|
+
* comment line inside e.g. a fenced shell example in the target is
|
|
239
|
+
* indistinguishable from a real Markdown heading to `MD_HEADING_RE`, and
|
|
240
|
+
* both `findEnclosingHeading` (picks the wrong "nearest" heading) and the
|
|
241
|
+
* enclosure walk in `checkAnchor` (treats it as a section boundary) would
|
|
242
|
+
* misfire on it.
|
|
243
|
+
*/
|
|
244
|
+
function computeFencedLineIndices(lines) {
|
|
245
|
+
const fenced = new Set();
|
|
246
|
+
let fenceMarker;
|
|
247
|
+
let fenceStart = -1;
|
|
248
|
+
for (let i = 0; i < lines.length; i++) {
|
|
249
|
+
const trimmed = (lines[i] ?? "").trim();
|
|
250
|
+
if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
251
|
+
fenceMarker = trimmed.slice(0, 3);
|
|
252
|
+
fenceStart = i;
|
|
253
|
+
}
|
|
254
|
+
else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
255
|
+
for (let j = fenceStart; j <= i; j++)
|
|
256
|
+
fenced.add(j);
|
|
257
|
+
fenceMarker = undefined;
|
|
258
|
+
fenceStart = -1;
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
if (fenceMarker && fenceStart >= 0) {
|
|
262
|
+
for (let j = fenceStart; j < lines.length; j++)
|
|
263
|
+
fenced.add(j);
|
|
264
|
+
}
|
|
265
|
+
return fenced;
|
|
266
|
+
}
|
|
267
|
+
/**
|
|
268
|
+
* Nearest Markdown heading at or before `startLine` (1-based), considering
|
|
269
|
+
* only heading levels up to `ANCHOR_HEADING_MAX_LEVEL` -- see the "Anchored
|
|
270
|
+
* citations" doc block above for why subsection headings are transparent to
|
|
271
|
+
* this search. `fencedLines` (see `computeFencedLineIndices`) excludes any
|
|
272
|
+
* line inside a fenced code block from matching, so a `#`-led comment
|
|
273
|
+
* inside a fenced example is never mistaken for a heading. Returns `null`
|
|
274
|
+
* when no such heading precedes `startLine` at all (e.g. the citation
|
|
275
|
+
* lands above the target's first release heading).
|
|
276
|
+
*/
|
|
277
|
+
function findEnclosingHeading(lines, startLine, fencedLines) {
|
|
278
|
+
for (let i = startLine - 1; i >= 0; i--) {
|
|
279
|
+
if (fencedLines.has(i))
|
|
280
|
+
continue;
|
|
281
|
+
const m = (lines[i] ?? "").match(MD_HEADING_RE);
|
|
282
|
+
if (m && m[1].length <= ANCHOR_HEADING_MAX_LEVEL) {
|
|
283
|
+
return { level: m[1].length, text: m[2].trim(), lineNo: i + 1 };
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
return null;
|
|
287
|
+
}
|
|
288
|
+
/**
|
|
289
|
+
* Checks an anchored full citation's anchor against its already-resolved,
|
|
290
|
+
* already-range-checked target -- see the "Anchored citations" doc block
|
|
291
|
+
* above for the two forms' semantics. `endLine` is the citation's end line,
|
|
292
|
+
* or its start line for a single-line citation (a size-1 range).
|
|
293
|
+
*/
|
|
294
|
+
function checkAnchor(anchor, startLine, endLine, lines) {
|
|
295
|
+
if (anchor.kind === "string") {
|
|
296
|
+
for (let i = startLine - 1; i <= endLine - 1 && i < lines.length; i++) {
|
|
297
|
+
if ((lines[i] ?? "").includes(anchor.text))
|
|
298
|
+
return null;
|
|
299
|
+
}
|
|
300
|
+
return {
|
|
301
|
+
rule: "anchor-not-found-in-range",
|
|
302
|
+
message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
|
|
303
|
+
};
|
|
304
|
+
}
|
|
305
|
+
const fencedLines = computeFencedLineIndices(lines);
|
|
306
|
+
const heading = findEnclosingHeading(lines, startLine, fencedLines);
|
|
307
|
+
if (!heading) {
|
|
308
|
+
return {
|
|
309
|
+
rule: "anchor-heading-not-found",
|
|
310
|
+
message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor against; use a string anchor (#"...") instead against a target with no heading structure`,
|
|
311
|
+
};
|
|
312
|
+
}
|
|
313
|
+
if (!heading.text.includes(anchor.text)) {
|
|
314
|
+
return {
|
|
315
|
+
rule: "anchor-heading-mismatch",
|
|
316
|
+
message: `nearest enclosing heading ("${heading.text}") does not contain anchor "${anchor.text}"`,
|
|
317
|
+
};
|
|
318
|
+
}
|
|
319
|
+
for (let i = heading.lineNo; i <= endLine - 1; i++) {
|
|
320
|
+
if (fencedLines.has(i))
|
|
321
|
+
continue;
|
|
322
|
+
const m = (lines[i] ?? "").match(MD_HEADING_RE);
|
|
323
|
+
if (m && m[1].length <= heading.level) {
|
|
324
|
+
return {
|
|
325
|
+
rule: "anchor-heading-does-not-enclose",
|
|
326
|
+
message: `range extends past its enclosing heading's section (next heading "${m[2].trim()}" at line ${i + 1})`,
|
|
327
|
+
};
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
return null;
|
|
331
|
+
}
|
|
332
|
+
const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
|
|
333
|
+
// Continuation citation forms (see the "Continuation citations" doc block
|
|
334
|
+
// above). Each requires the backtick delimiter as part of the match so it
|
|
335
|
+
// can never overlap a CITATION_RE match: a full citation's regex match
|
|
336
|
+
// never includes the surrounding backticks, and none of these three
|
|
337
|
+
// require a `path.ext` prefix before the digits.
|
|
338
|
+
const CONT_COLON_RE = /`:(\d+)(?:-(\d+))?`/g;
|
|
339
|
+
const CONT_DASH_RE = /[-–]`(\d+)`/g;
|
|
340
|
+
const CONT_PAREN_RE = /\(`(\d+)`\)/g;
|
|
341
|
+
const CLOSING_ONLY_EXTS = new Set(["ts", "js", "mjs", "yml", "yaml", "json"]);
|
|
342
|
+
const CLOSING_BRACE_RE = /^[)\]}][;,]?$/;
|
|
343
|
+
// Short-form (paragraph-bound) citation form, see the "Short-form
|
|
344
|
+
// citations" doc block below. Requires an explicit N-M range: a bare single
|
|
345
|
+
// number (`:5`) is not matched -- see that doc block for why. Only the
|
|
346
|
+
// colon form is collected; a bare `(N-M)` is never a short-form citation
|
|
347
|
+
// candidate at all -- see the same doc block for why the paren form was
|
|
348
|
+
// dropped rather than gated.
|
|
349
|
+
const SHORT_FORM_COLON_RE = /:(\d+)-(\d+)/g;
|
|
350
|
+
// Test-file block-boundary check (see checkRangeBoundary): a citation's
|
|
351
|
+
// range into a `.test.ts`/`.spec.ts` (or `.js`/`.mjs` equivalent) target
|
|
352
|
+
// must start on a describe(/it( head line and end on its matching closing
|
|
353
|
+
// `});` line.
|
|
354
|
+
const TEST_FILE_RE = /\.(test|spec)\.(ts|js|mjs)$/i;
|
|
355
|
+
const TEST_HEAD_LINE_RE = /^\s*(?:describe|it)\s*\(/;
|
|
356
|
+
const TEST_CLOSING_LINE_RE = /^\s*\}\)\s*;\s*$/;
|
|
357
|
+
// Markdown block-boundary check (see checkRangeBoundary): a range boundary
|
|
358
|
+
// line that is nothing but a bracket (open or close), optionally with a
|
|
359
|
+
// trailing `,`/`;`, is always a drift signal. A bare code-fence delimiter is
|
|
360
|
+
// its own, separate signal (see MD_FENCE_DELIM_RE): unlike a bracket, a
|
|
361
|
+
// fence line is sometimes the deliberately-correct start of a citation (see
|
|
362
|
+
// checkRangeBoundary's markdown branch for the opening-fence exception).
|
|
363
|
+
const MD_BARE_BRACKET_RE = /^(?:[)\]}][;,]?|[[({])$/;
|
|
364
|
+
const MD_FENCE_DELIM_RE = /^(?:`{3,}\S*|~{3,}\S*)$/;
|
|
365
|
+
const EXCLUDED_DIRS = new Set([
|
|
366
|
+
"node_modules",
|
|
367
|
+
".git",
|
|
368
|
+
"dist",
|
|
369
|
+
"build",
|
|
370
|
+
"coverage",
|
|
371
|
+
".next",
|
|
372
|
+
".turbo",
|
|
373
|
+
// Ported 1:1 from the original script's own disposable test-fixtures
|
|
374
|
+
// exclusion, kept so the ported EXCLUDED_DIRS regression test carries
|
|
375
|
+
// over verbatim; harmless for any consuming repo that has no directory
|
|
376
|
+
// with this name.
|
|
377
|
+
"okf-citations-resolve-fixtures",
|
|
378
|
+
]);
|
|
379
|
+
function isFile(p) {
|
|
380
|
+
try {
|
|
381
|
+
return fs.statSync(p).isFile();
|
|
382
|
+
}
|
|
383
|
+
catch {
|
|
384
|
+
return false;
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
/** True when citedPath has a literal `..` path segment. */
|
|
388
|
+
function hasParentSegment(citedPath) {
|
|
389
|
+
return citedPath.split("/").includes("..");
|
|
390
|
+
}
|
|
391
|
+
/**
|
|
392
|
+
* Repo-wide search for files with an exact basename, memoized per root
|
|
393
|
+
* within `cache`. `cache` is built lazily and scoped to a single
|
|
394
|
+
* `citationsResolveRule.run(ctx)` invocation (see the rule's `run`), not
|
|
395
|
+
* held at module scope, so an in-process caller running the check
|
|
396
|
+
* repeatedly (e.g. in a long-lived process, or a test suite that edits
|
|
397
|
+
* fixtures between runs) never sees a stale index from an earlier run.
|
|
398
|
+
*/
|
|
399
|
+
function findByBasename(cache, root, basename) {
|
|
400
|
+
let index = cache.get(root);
|
|
401
|
+
if (!index) {
|
|
402
|
+
index = new Map();
|
|
403
|
+
const found = index;
|
|
404
|
+
const walk = (dir) => {
|
|
405
|
+
let entries;
|
|
406
|
+
try {
|
|
407
|
+
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
408
|
+
}
|
|
409
|
+
catch {
|
|
410
|
+
return;
|
|
411
|
+
}
|
|
412
|
+
for (const entry of entries) {
|
|
413
|
+
if (entry.name.startsWith(".") && entry.name !== ".github")
|
|
414
|
+
continue;
|
|
415
|
+
if (EXCLUDED_DIRS.has(entry.name))
|
|
416
|
+
continue;
|
|
417
|
+
const full = path.join(dir, entry.name);
|
|
418
|
+
if (entry.isDirectory()) {
|
|
419
|
+
walk(full);
|
|
420
|
+
}
|
|
421
|
+
else if (entry.isFile()) {
|
|
422
|
+
const list = found.get(entry.name) ?? [];
|
|
423
|
+
list.push(full);
|
|
424
|
+
found.set(entry.name, list);
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
};
|
|
428
|
+
walk(root);
|
|
429
|
+
cache.set(root, index);
|
|
430
|
+
}
|
|
431
|
+
return index.get(basename) ?? [];
|
|
432
|
+
}
|
|
433
|
+
/**
|
|
434
|
+
* Finds the nearest citation earlier in the same doc whose cited path
|
|
435
|
+
* contains a `/` and ends with the same suffix as `citedPath`, the "full
|
|
436
|
+
* path was mentioned earlier in this section/doc" convention.
|
|
437
|
+
*/
|
|
438
|
+
function findPriorQualifiedCitation(content, beforeIndex, citedPath) {
|
|
439
|
+
const suffix = "/" + citedPath;
|
|
440
|
+
let best = null;
|
|
441
|
+
let bestIndex = -1;
|
|
442
|
+
const re = new RegExp(CITATION_RE.source, "g");
|
|
443
|
+
let m;
|
|
444
|
+
while ((m = re.exec(content)) !== null) {
|
|
445
|
+
if (m.index >= beforeIndex)
|
|
446
|
+
break;
|
|
447
|
+
const candidate = m[1];
|
|
448
|
+
if (candidate === citedPath)
|
|
449
|
+
continue; // not more qualified than itself
|
|
450
|
+
if (candidate.includes("/") &&
|
|
451
|
+
(candidate === citedPath || candidate.endsWith(suffix))) {
|
|
452
|
+
if (m.index > bestIndex) {
|
|
453
|
+
bestIndex = m.index;
|
|
454
|
+
best = candidate;
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
return best;
|
|
459
|
+
}
|
|
460
|
+
/**
|
|
461
|
+
* For a bare (no `/`) citedPath, tries doc-relative first, then each
|
|
462
|
+
* ancestor directory of the doc, nearest first, up to and including `root`.
|
|
463
|
+
* Returns the first real file found, or `null`. See the "Path resolution"
|
|
464
|
+
* doc block above (step 2) for why this runs before the plain
|
|
465
|
+
* repo-root-relative lookup: a same-named file closer to the citing doc
|
|
466
|
+
* (e.g. a package's own `README.md`) should win over one that merely
|
|
467
|
+
* happens to also exist at the repo root.
|
|
468
|
+
*/
|
|
469
|
+
function resolveViaAncestorClimb(root, docAbsPath, citedPath) {
|
|
470
|
+
const resolvedRoot = path.resolve(root);
|
|
471
|
+
let dir = path.dirname(docAbsPath);
|
|
472
|
+
for (;;) {
|
|
473
|
+
const candidate = path.resolve(dir, citedPath);
|
|
474
|
+
if (isFile(candidate))
|
|
475
|
+
return candidate;
|
|
476
|
+
if (path.resolve(dir) === resolvedRoot)
|
|
477
|
+
return null;
|
|
478
|
+
const parent = path.dirname(dir);
|
|
479
|
+
if (parent === dir)
|
|
480
|
+
return null; // reached the filesystem root
|
|
481
|
+
dir = parent;
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
/**
|
|
485
|
+
* Resolves a citation's path to a single real file. Returns `{ skip: true }`
|
|
486
|
+
* for a citedPath out of scope (leading `/`), `{ path }` on a definitive
|
|
487
|
+
* single resolution, `{ ambiguous: true, candidates }` when more than one
|
|
488
|
+
* plausible target exists, or `null` when nothing matches. Callers must
|
|
489
|
+
* reject a citedPath with a `..` segment (see hasParentSegment) before
|
|
490
|
+
* calling this; it is not re-checked here.
|
|
491
|
+
*/
|
|
492
|
+
function resolveCitation(cache, root, docAbsPath, docContent, docSources, citedPath, matchIndex) {
|
|
493
|
+
if (citedPath.startsWith("/")) {
|
|
494
|
+
return { skip: true };
|
|
495
|
+
}
|
|
496
|
+
const sourceMatches = docSources.filter((s) => s === citedPath || s.endsWith("/" + citedPath));
|
|
497
|
+
if (sourceMatches.length === 1) {
|
|
498
|
+
const candidate = path.resolve(root, sourceMatches[0]);
|
|
499
|
+
if (isFile(candidate))
|
|
500
|
+
return { path: candidate };
|
|
501
|
+
}
|
|
502
|
+
if (!citedPath.includes("/")) {
|
|
503
|
+
const viaAncestor = resolveViaAncestorClimb(root, docAbsPath, citedPath);
|
|
504
|
+
if (viaAncestor)
|
|
505
|
+
return { path: viaAncestor };
|
|
506
|
+
}
|
|
507
|
+
const rootRelative = path.resolve(root, citedPath);
|
|
508
|
+
if (isFile(rootRelative))
|
|
509
|
+
return { path: rootRelative };
|
|
510
|
+
const docRelative = path.resolve(path.dirname(docAbsPath), citedPath);
|
|
511
|
+
if (isFile(docRelative))
|
|
512
|
+
return { path: docRelative };
|
|
513
|
+
const prior = findPriorQualifiedCitation(docContent, matchIndex, citedPath);
|
|
514
|
+
if (prior) {
|
|
515
|
+
const candidate = path.resolve(root, prior);
|
|
516
|
+
if (isFile(candidate))
|
|
517
|
+
return { path: candidate };
|
|
518
|
+
}
|
|
519
|
+
const base = citedPath.includes("/")
|
|
520
|
+
? citedPath.split("/").pop()
|
|
521
|
+
: citedPath;
|
|
522
|
+
const bySuffix = findByBasename(cache, root, base).filter((m) => {
|
|
523
|
+
const normalized = m.split(path.sep).join("/");
|
|
524
|
+
return (normalized === citedPath ||
|
|
525
|
+
normalized.endsWith("/" + citedPath) ||
|
|
526
|
+
!citedPath.includes("/"));
|
|
527
|
+
});
|
|
528
|
+
if (bySuffix.length === 1)
|
|
529
|
+
return { path: bySuffix[0] };
|
|
530
|
+
if (bySuffix.length > 1) {
|
|
531
|
+
return {
|
|
532
|
+
ambiguous: true,
|
|
533
|
+
candidates: bySuffix.map((m) => path.relative(root, m)),
|
|
534
|
+
};
|
|
535
|
+
}
|
|
536
|
+
return null;
|
|
537
|
+
}
|
|
538
|
+
function splitLines(content) {
|
|
539
|
+
const lines = content.split("\n");
|
|
540
|
+
if (lines.length > 0 &&
|
|
541
|
+
lines[lines.length - 1] === "" &&
|
|
542
|
+
content.endsWith("\n")) {
|
|
543
|
+
lines.pop();
|
|
544
|
+
}
|
|
545
|
+
return lines;
|
|
546
|
+
}
|
|
547
|
+
// Range-bound checks shared by a full citation's own range (via
|
|
548
|
+
// checkTarget below) and a cont-ext atom's extension (via
|
|
549
|
+
// checkRangeBoundOnly): a citation's end before its start, or either bound
|
|
550
|
+
// past the end of the file. Both are pure "does this range make sense"
|
|
551
|
+
// checks, independent of what the start line's content actually is.
|
|
552
|
+
function checkRangeBound(startLine, endLine, lineCount) {
|
|
553
|
+
if (endLine !== null && endLine < startLine) {
|
|
554
|
+
return {
|
|
555
|
+
rule: "inverted-range",
|
|
556
|
+
message: `range end (${endLine}) is before its start (${startLine})`,
|
|
557
|
+
};
|
|
558
|
+
}
|
|
559
|
+
const last = endLine ?? startLine;
|
|
560
|
+
if (startLine > lineCount || last > lineCount) {
|
|
561
|
+
return {
|
|
562
|
+
rule: "range-exceeds-file",
|
|
563
|
+
message: `citation exceeds file length (${lineCount} line(s))`,
|
|
564
|
+
};
|
|
565
|
+
}
|
|
566
|
+
return null;
|
|
567
|
+
}
|
|
568
|
+
/**
|
|
569
|
+
* Reads a resolved target's content, or reports why it couldn't be read
|
|
570
|
+
* (permission denied, and similar OS-level failures short of the file not
|
|
571
|
+
* existing at all -- `resolveCitation` already confirmed the target exists
|
|
572
|
+
* via `isFile`/`fs.statSync`, which does not require read permission).
|
|
573
|
+
* Returned as a `Problem` with `rule: "unreadable-target"` and `code` set to
|
|
574
|
+
* the OS error code, so callers can route it to a `notice`-severity finding
|
|
575
|
+
* (see pushUnreadable) instead of the usual `warning`-severity drift finding
|
|
576
|
+
* -- an unreadable file is not evidence the citation itself is wrong.
|
|
577
|
+
*/
|
|
578
|
+
function readTarget(resolvedPath) {
|
|
579
|
+
try {
|
|
580
|
+
return { content: fs.readFileSync(resolvedPath, "utf8") };
|
|
581
|
+
}
|
|
582
|
+
catch (err) {
|
|
583
|
+
const code = err.code ?? "UNKNOWN";
|
|
584
|
+
return {
|
|
585
|
+
rule: "unreadable-target",
|
|
586
|
+
message: "target file exists but could not be read",
|
|
587
|
+
code,
|
|
588
|
+
};
|
|
589
|
+
}
|
|
590
|
+
}
|
|
591
|
+
/**
|
|
592
|
+
* The non-I/O half of checkTarget: every check that only needs the
|
|
593
|
+
* target's already-read `lines`, not the filesystem. Split out so a caller
|
|
594
|
+
* that needs a second, different check against the same target (see
|
|
595
|
+
* checkShortFormTarget) can read the file once and reuse `lines` for both,
|
|
596
|
+
* instead of checkTarget re-reading it internally.
|
|
597
|
+
*/
|
|
598
|
+
function checkTargetLines(citedPath, startLine, endLine, lines) {
|
|
599
|
+
const bound = checkRangeBound(startLine, endLine, lines.length);
|
|
600
|
+
if (bound)
|
|
601
|
+
return bound;
|
|
602
|
+
const startText = lines[startLine - 1] ?? "";
|
|
603
|
+
const trimmed = startText.trim();
|
|
604
|
+
if (trimmed === "") {
|
|
605
|
+
return { rule: "blank-start-line", message: "start line is blank" };
|
|
606
|
+
}
|
|
607
|
+
const ext = (citedPath.split(".").pop() ?? "").toLowerCase();
|
|
608
|
+
if (CLOSING_ONLY_EXTS.has(ext) && CLOSING_BRACE_RE.test(trimmed)) {
|
|
609
|
+
return {
|
|
610
|
+
rule: "closing-brace-start-line",
|
|
611
|
+
message: `start line is only a closing brace/bracket ("${trimmed}")`,
|
|
612
|
+
};
|
|
613
|
+
}
|
|
614
|
+
return null;
|
|
615
|
+
}
|
|
616
|
+
function checkTarget(citedPath, startLine, endLine, resolvedPath) {
|
|
617
|
+
const read = readTarget(resolvedPath);
|
|
618
|
+
if ("rule" in read)
|
|
619
|
+
return read;
|
|
620
|
+
return checkTargetLines(citedPath, startLine, endLine, splitLines(read.content));
|
|
621
|
+
}
|
|
622
|
+
/**
|
|
623
|
+
* A full citation's complete check: `checkTargetLines`'s existing checks
|
|
624
|
+
* (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
|
|
625
|
+
* closing-brace-start-line), plus, only when those all pass and the
|
|
626
|
+
* citation carried an anchor, the anchor check above (see the "Anchored
|
|
627
|
+
* citations" doc block). Reads the resolved target once, mirroring
|
|
628
|
+
* `checkShortFormTarget`'s reasoning for the block-boundary check. Anchors
|
|
629
|
+
* are full-citation-only (see that doc block for why), so `checkTarget`
|
|
630
|
+
* itself is untouched and still used as-is for a cont-fresh atom.
|
|
631
|
+
*/
|
|
632
|
+
function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor) {
|
|
633
|
+
const read = readTarget(resolvedPath);
|
|
634
|
+
if ("rule" in read)
|
|
635
|
+
return read;
|
|
636
|
+
const lines = splitLines(read.content);
|
|
637
|
+
const base = checkTargetLines(citedPath, startLine, endLine, lines);
|
|
638
|
+
if (base)
|
|
639
|
+
return base;
|
|
640
|
+
if (!anchor)
|
|
641
|
+
return null;
|
|
642
|
+
return checkAnchor(anchor, startLine, endLine ?? startLine, lines);
|
|
643
|
+
}
|
|
644
|
+
// A cont-ext atom only ever extends the *end* of a range whose start line
|
|
645
|
+
// was already fully checked (blank / closing-brace) when it was cited as
|
|
646
|
+
// its own full citation or cont-fresh atom -- re-running checkTarget here
|
|
647
|
+
// would re-derive that same start-line check against the identical line
|
|
648
|
+
// and, on a real drift, double-report it as a second finding. This checks
|
|
649
|
+
// only whether the (possibly inverted, possibly out-of-file) range itself
|
|
650
|
+
// is sound.
|
|
651
|
+
function checkRangeBoundOnly(startLine, endLine, resolvedPath) {
|
|
652
|
+
const read = readTarget(resolvedPath);
|
|
653
|
+
if ("rule" in read)
|
|
654
|
+
return read;
|
|
655
|
+
const lineCount = splitLines(read.content).length;
|
|
656
|
+
return checkRangeBound(startLine, endLine, lineCount);
|
|
657
|
+
}
|
|
658
|
+
function isTestFile(citedPath) {
|
|
659
|
+
return TEST_FILE_RE.test(citedPath);
|
|
660
|
+
}
|
|
661
|
+
/**
|
|
662
|
+
* True when the line at `lines[lineIndex]` is a fence delimiter
|
|
663
|
+
* (```` ``` ```` or `~~~`, optionally with a trailing language tag) that
|
|
664
|
+
* *opens* a fenced code block, as opposed to closing one -- determined by
|
|
665
|
+
* replaying the same open/close state machine `stripFencedCode`-style
|
|
666
|
+
* scanners use from the top of the document, not by the line's own text
|
|
667
|
+
* (a bare closing fence and an untagged opening fence are lexically
|
|
668
|
+
* identical). Used by checkRangeBoundary's markdown branch: citing a
|
|
669
|
+
* fenced block starting at its own opening fence line is the natural,
|
|
670
|
+
* correct way to cite it, so that specific case is exempted from the
|
|
671
|
+
* fence-as-drift-signal check (see there).
|
|
672
|
+
*/
|
|
673
|
+
function isFenceOpeningLine(lines, lineIndex) {
|
|
674
|
+
let inFence = false;
|
|
675
|
+
let fenceMarker;
|
|
676
|
+
for (let i = 0; i <= lineIndex; i++) {
|
|
677
|
+
const trimmed = (lines[i] ?? "").trim();
|
|
678
|
+
if (!inFence && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
679
|
+
if (i === lineIndex)
|
|
680
|
+
return true;
|
|
681
|
+
inFence = true;
|
|
682
|
+
fenceMarker = trimmed.slice(0, 3);
|
|
683
|
+
}
|
|
684
|
+
else if (inFence && fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
685
|
+
if (i === lineIndex)
|
|
686
|
+
return false;
|
|
687
|
+
inFence = false;
|
|
688
|
+
fenceMarker = undefined;
|
|
689
|
+
}
|
|
690
|
+
}
|
|
691
|
+
return false;
|
|
692
|
+
}
|
|
693
|
+
/**
|
|
694
|
+
* True when `lines[endLineIndex]` is the closing delimiter that matches
|
|
695
|
+
* the *opening* fence at `lines[startLineIndex]` (the caller must already
|
|
696
|
+
* have confirmed the start line is a genuine opening fence, via
|
|
697
|
+
* isFenceOpeningLine, before calling this) -- the first line after the
|
|
698
|
+
* opener whose trimmed text starts with the same fence marker. Used by
|
|
699
|
+
* checkRangeBoundary's markdown branch to also exempt a range's END from
|
|
700
|
+
* the fence-as-drift-signal check when the range legitimately cites a
|
|
701
|
+
* whole fenced code block from its own opening delimiter to its own
|
|
702
|
+
* closing delimiter: the natural, correct way to cite such a block, not
|
|
703
|
+
* drift, the same reasoning the start-side exception already applies.
|
|
704
|
+
*/
|
|
705
|
+
function isMatchingFenceClosingLine(lines, startLineIndex, endLineIndex) {
|
|
706
|
+
const fenceMarker = (lines[startLineIndex] ?? "").trim().slice(0, 3);
|
|
707
|
+
for (let i = startLineIndex + 1; i < lines.length; i++) {
|
|
708
|
+
const trimmed = (lines[i] ?? "").trim();
|
|
709
|
+
if (trimmed.startsWith(fenceMarker)) {
|
|
710
|
+
return i === endLineIndex;
|
|
711
|
+
}
|
|
712
|
+
}
|
|
713
|
+
return false;
|
|
714
|
+
}
|
|
715
|
+
/**
|
|
716
|
+
* Additional block-boundary check for a short-form citation's range (see
|
|
717
|
+
* the "Short-form citations" doc block below), layered on top of
|
|
718
|
+
* checkTarget's existing per-start-line checks via checkShortFormTarget.
|
|
719
|
+
* Deliberately scoped to short-form citations only, not wired into
|
|
720
|
+
* checkTarget/checkRangeBoundOnly (the full/continuation citation paths):
|
|
721
|
+
* applying it there too was tried first and rejected -- the real bundle
|
|
722
|
+
* this rule was built against has legitimate full citations into test
|
|
723
|
+
* files that cite a couple of arbitrary lines (e.g. two lines of a shared
|
|
724
|
+
* regex definition), not a describe/it block, and flagging those as newly
|
|
725
|
+
* broken would have regressed the existing 0-warning baseline. Short-form
|
|
726
|
+
* citations are the demonstrated motivating case (a paragraph naming a
|
|
727
|
+
* test file once, then citing several of its describe/it blocks by bare
|
|
728
|
+
* range alone), so the check is scoped to exactly that mechanism.
|
|
729
|
+
*
|
|
730
|
+
* Test-file targets (`.test.ts`/`.spec.ts`, and their `.js`/`.mjs`
|
|
731
|
+
* equivalents): the range must start on a `describe(`/`it(` head line and
|
|
732
|
+
* end on a matching closing `});` line. Severity split, by evidence
|
|
733
|
+
* strength: a wrong START line is a
|
|
734
|
+
* *warning* (`test-range-start-not-head`) -- the range beginning somewhere
|
|
735
|
+
* other than a block head is strong drift evidence. A range whose start IS
|
|
736
|
+
* correct but whose END is not the matching `});` is only a *notice*
|
|
737
|
+
* (`test-range-end-not-closing`): this also matches a deliberate partial
|
|
738
|
+
* citation (citing from a block's head to partway through it), which is not
|
|
739
|
+
* drift. The start is checked first and returned alone, matching
|
|
740
|
+
* checkTarget's existing single-problem-per-citation pattern (never both at
|
|
741
|
+
* once).
|
|
742
|
+
*
|
|
743
|
+
* Markdown targets: mechanical verification of "is this still the same
|
|
744
|
+
* block" is far less reliable for prose than for TS/JS brace structure, so
|
|
745
|
+
* this is a notice, not a warning (see this rule's own risk note on
|
|
746
|
+
* markdown false positives). The range's start or end line landing on a
|
|
747
|
+
* bare bracket (MD_BARE_BRACKET_RE) is always a heads-up that the boundary
|
|
748
|
+
* likely drifted onto structural punctuation rather than real prose. A bare
|
|
749
|
+
* code-fence delimiter (MD_FENCE_DELIM_RE) is also always a drift signal at
|
|
750
|
+
* either boundary, with two exceptions carved out for the one legitimate
|
|
751
|
+
* way to cite a whole fenced code block by its own delimiters: the START is
|
|
752
|
+
* exempted when that line is itself a genuine *opening* fence (see
|
|
753
|
+
* isFenceOpeningLine), and, only when the start already qualified for that
|
|
754
|
+
* exemption, the END is exempted when it is that same fence's matching
|
|
755
|
+
* *closing* delimiter (see isMatchingFenceClosingLine) -- citing a fenced
|
|
756
|
+
* block from its own opening delimiter through its own closing delimiter is
|
|
757
|
+
* the natural, correct way to cite it, not drift.
|
|
758
|
+
*/
|
|
759
|
+
function checkRangeBoundary(citedPath, startLine, endLine, lines) {
|
|
760
|
+
if (isTestFile(citedPath)) {
|
|
761
|
+
const startText = lines[startLine - 1] ?? "";
|
|
762
|
+
if (!TEST_HEAD_LINE_RE.test(startText)) {
|
|
763
|
+
return {
|
|
764
|
+
rule: "test-range-start-not-head",
|
|
765
|
+
message: `range start is not a "describe(" or "it(" head line ("${startText.trim()}")`,
|
|
766
|
+
};
|
|
767
|
+
}
|
|
768
|
+
const endText = lines[endLine - 1] ?? "";
|
|
769
|
+
if (!TEST_CLOSING_LINE_RE.test(endText)) {
|
|
770
|
+
return {
|
|
771
|
+
rule: "test-range-end-not-closing",
|
|
772
|
+
message: `range end is not a matching closing "});" line ("${endText.trim()}")`,
|
|
773
|
+
severity: "notice",
|
|
774
|
+
};
|
|
775
|
+
}
|
|
776
|
+
return null;
|
|
777
|
+
}
|
|
778
|
+
if (citedPath.toLowerCase().endsWith(".md")) {
|
|
779
|
+
const startTrim = (lines[startLine - 1] ?? "").trim();
|
|
780
|
+
const startIsFence = MD_FENCE_DELIM_RE.test(startTrim);
|
|
781
|
+
if (MD_BARE_BRACKET_RE.test(startTrim) ||
|
|
782
|
+
(startIsFence && !isFenceOpeningLine(lines, startLine - 1))) {
|
|
783
|
+
return {
|
|
784
|
+
rule: "markdown-range-boundary-bracket-or-fence",
|
|
785
|
+
message: `range start is a bare bracket/fence line ("${startTrim}")`,
|
|
786
|
+
severity: "notice",
|
|
787
|
+
};
|
|
788
|
+
}
|
|
789
|
+
const endTrim = (lines[endLine - 1] ?? "").trim();
|
|
790
|
+
const endIsFence = MD_FENCE_DELIM_RE.test(endTrim);
|
|
791
|
+
const endIsMatchingClose = startIsFence &&
|
|
792
|
+
endIsFence &&
|
|
793
|
+
isMatchingFenceClosingLine(lines, startLine - 1, endLine - 1);
|
|
794
|
+
if (MD_BARE_BRACKET_RE.test(endTrim) ||
|
|
795
|
+
(endIsFence && !endIsMatchingClose)) {
|
|
796
|
+
return {
|
|
797
|
+
rule: "markdown-range-boundary-bracket-or-fence",
|
|
798
|
+
message: `range end is a bare bracket/fence line ("${endTrim}")`,
|
|
799
|
+
severity: "notice",
|
|
800
|
+
};
|
|
801
|
+
}
|
|
802
|
+
return null;
|
|
803
|
+
}
|
|
804
|
+
return null;
|
|
805
|
+
}
|
|
806
|
+
/**
|
|
807
|
+
* A short-form citation's full check: checkTarget's existing checks
|
|
808
|
+
* (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
|
|
809
|
+
* closing-brace-start-line), plus, only when those all pass, the
|
|
810
|
+
* test-file/markdown block-boundary check above (see checkRangeBoundary).
|
|
811
|
+
* Reads the resolved target from disk exactly once (a dogfood bundle can
|
|
812
|
+
* run this against the same target file a dozen-plus times for one
|
|
813
|
+
* compound short-form list) and reuses the same `lines` array for both
|
|
814
|
+
* checks, instead of checkTarget and checkRangeBoundary each reading and
|
|
815
|
+
* re-splitting it independently.
|
|
816
|
+
*/
|
|
817
|
+
function checkShortFormTarget(citedPath, startLine, endLine, resolvedPath) {
|
|
818
|
+
const read = readTarget(resolvedPath);
|
|
819
|
+
if ("rule" in read)
|
|
820
|
+
return read;
|
|
821
|
+
const lines = splitLines(read.content);
|
|
822
|
+
const base = checkTargetLines(citedPath, startLine, endLine, lines);
|
|
823
|
+
if (base)
|
|
824
|
+
return base;
|
|
825
|
+
return checkRangeBoundary(citedPath, startLine, endLine, lines);
|
|
826
|
+
}
|
|
827
|
+
/**
|
|
828
|
+
* Collects every continuation-citation atom (see the "Continuation
|
|
829
|
+
* citations" doc block above) in `content`, sorted by document position.
|
|
830
|
+
*
|
|
831
|
+
* Each atom is tagged with a role:
|
|
832
|
+
* - "cont-fresh": establishes a new start line (optionally with its own
|
|
833
|
+
* embedded end, e.g. `` `:75-78` ``), checked the same way as a full
|
|
834
|
+
* citation's start (blank / closing-brace / range-exceeds).
|
|
835
|
+
* - "cont-ext": purely extends the *end* of whatever start line came
|
|
836
|
+
* immediately before it (a `` -`M` `` / `` –`M` `` tail, or a
|
|
837
|
+
* `` `:M` `` directly preceded by a `-`/`–`). Ending a range on a
|
|
838
|
+
* closing brace is completely normal, so an extension is checked ONLY
|
|
839
|
+
* for the range-bound checks (inverted-range, range-exceeds-file),
|
|
840
|
+
* never blank/closing-brace.
|
|
841
|
+
* A colon-form match (`` `:N` ``) is "cont-ext" exactly when the nearest
|
|
842
|
+
* non-whitespace character before its opening backtick is `-` or `–`;
|
|
843
|
+
* otherwise it is "cont-fresh". Dash-form (`` -`M` ``/`` –`M` ``) is
|
|
844
|
+
* always "cont-ext" by construction. Paren-form (`` (`N`) ``) is always
|
|
845
|
+
* "cont-fresh".
|
|
846
|
+
*/
|
|
847
|
+
function collectContinuationAtoms(content) {
|
|
848
|
+
const atoms = [];
|
|
849
|
+
const colonRe = new RegExp(CONT_COLON_RE.source, "g");
|
|
850
|
+
let m;
|
|
851
|
+
while ((m = colonRe.exec(content)) !== null) {
|
|
852
|
+
const before = content.slice(0, m.index).trimEnd();
|
|
853
|
+
if (/[-–]$/.test(before)) {
|
|
854
|
+
atoms.push({
|
|
855
|
+
kind: "cont-ext",
|
|
856
|
+
index: m.index,
|
|
857
|
+
value: m[2] ? Number(m[2]) : Number(m[1]),
|
|
858
|
+
});
|
|
859
|
+
}
|
|
860
|
+
else {
|
|
861
|
+
atoms.push({
|
|
862
|
+
kind: "cont-fresh",
|
|
863
|
+
index: m.index,
|
|
864
|
+
startLine: Number(m[1]),
|
|
865
|
+
endLine: m[2] ? Number(m[2]) : null,
|
|
866
|
+
});
|
|
867
|
+
}
|
|
868
|
+
}
|
|
869
|
+
const dashRe = new RegExp(CONT_DASH_RE.source, "g");
|
|
870
|
+
while ((m = dashRe.exec(content)) !== null) {
|
|
871
|
+
atoms.push({ kind: "cont-ext", index: m.index, value: Number(m[1]) });
|
|
872
|
+
}
|
|
873
|
+
const parenRe = new RegExp(CONT_PAREN_RE.source, "g");
|
|
874
|
+
while ((m = parenRe.exec(content)) !== null) {
|
|
875
|
+
atoms.push({
|
|
876
|
+
kind: "cont-fresh",
|
|
877
|
+
index: m.index,
|
|
878
|
+
startLine: Number(m[1]),
|
|
879
|
+
endLine: null,
|
|
880
|
+
});
|
|
881
|
+
}
|
|
882
|
+
return atoms;
|
|
883
|
+
}
|
|
884
|
+
/**
|
|
885
|
+
* True when the nearest non-whitespace text before `matchIndex` in
|
|
886
|
+
* `content` is a serial connective: the single character `,`, `;`, or `(`,
|
|
887
|
+
* or the word `and`/`or` (case-insensitive, word-boundary-matched so it
|
|
888
|
+
* does not fire on the tail of a longer word like "brand"). See the
|
|
889
|
+
* "Short-form citations" doc block above for why this is the gate.
|
|
890
|
+
*/
|
|
891
|
+
function isSerialConnectivePreceded(content, matchIndex) {
|
|
892
|
+
const before = content.slice(0, matchIndex).trimEnd();
|
|
893
|
+
if (before === "")
|
|
894
|
+
return false;
|
|
895
|
+
const lastChar = before[before.length - 1];
|
|
896
|
+
if (lastChar === "," || lastChar === ";" || lastChar === "(")
|
|
897
|
+
return true;
|
|
898
|
+
return /\b(?:and|or)$/i.test(before);
|
|
899
|
+
}
|
|
900
|
+
function isWithinAnySpan(index, spans) {
|
|
901
|
+
return spans.some(([start, end]) => index >= start && index < end);
|
|
902
|
+
}
|
|
903
|
+
function collectShortFormMatches(content, excludedSpans) {
|
|
904
|
+
const out = [];
|
|
905
|
+
const colonRe = new RegExp(SHORT_FORM_COLON_RE.source, "g");
|
|
906
|
+
let m;
|
|
907
|
+
while ((m = colonRe.exec(content)) !== null) {
|
|
908
|
+
if (isWithinAnySpan(m.index, excludedSpans))
|
|
909
|
+
continue;
|
|
910
|
+
if (content[m.index - 1] === "`")
|
|
911
|
+
continue;
|
|
912
|
+
if (content[m.index + m[0].length] === "`")
|
|
913
|
+
continue;
|
|
914
|
+
if (!isSerialConnectivePreceded(content, m.index))
|
|
915
|
+
continue;
|
|
916
|
+
const startLine = Number(m[1]);
|
|
917
|
+
const endLine = Number(m[2]);
|
|
918
|
+
out.push({ index: m.index, startLine, endLine });
|
|
919
|
+
}
|
|
920
|
+
return out.sort((a, b) => a.index - b.index);
|
|
921
|
+
}
|
|
922
|
+
/**
|
|
923
|
+
* Char spans of every fenced code block in `content` (```` ``` ```` or
|
|
924
|
+
* `~~~`, optionally with a trailing language tag), each span running from
|
|
925
|
+
* the start of the opening fence line to the end of the closing fence line
|
|
926
|
+
* inclusive. An unterminated fence (no matching close before end of doc) is
|
|
927
|
+
* treated as running to the end of the content -- conservative, since an
|
|
928
|
+
* unterminated fence is itself a doc problem outside this rule's scope, not
|
|
929
|
+
* a reason to scan its contents for short-form citations.
|
|
930
|
+
*/
|
|
931
|
+
function computeFencedSpans(content) {
|
|
932
|
+
const spans = [];
|
|
933
|
+
const lines = content.split("\n");
|
|
934
|
+
let offset = 0;
|
|
935
|
+
let fenceMarker;
|
|
936
|
+
let fenceStart = -1;
|
|
937
|
+
for (const line of lines) {
|
|
938
|
+
const trimmed = line.trim();
|
|
939
|
+
const lineEnd = offset + line.length;
|
|
940
|
+
if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
941
|
+
fenceMarker = trimmed.slice(0, 3);
|
|
942
|
+
fenceStart = offset;
|
|
943
|
+
}
|
|
944
|
+
else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
945
|
+
spans.push([fenceStart, lineEnd]);
|
|
946
|
+
fenceMarker = undefined;
|
|
947
|
+
fenceStart = -1;
|
|
948
|
+
}
|
|
949
|
+
offset = lineEnd + 1; // +1 for the newline joining this line to the next
|
|
950
|
+
}
|
|
951
|
+
if (fenceMarker && fenceStart >= 0) {
|
|
952
|
+
spans.push([fenceStart, content.length]);
|
|
953
|
+
}
|
|
954
|
+
return spans;
|
|
955
|
+
}
|
|
956
|
+
/**
|
|
957
|
+
* Char spans of every CommonMark-style indented code block in `content`: a
|
|
958
|
+
* maximal run of consecutive non-blank lines, each indented by at least
|
|
959
|
+
* four spaces or a leading tab, whose first line is preceded by a blank
|
|
960
|
+
* line or the start of the document (an indented code block cannot
|
|
961
|
+
* interrupt a paragraph). A blank line inside the run does not itself end
|
|
962
|
+
* it, matching CommonMark. Simplified relative to the full CommonMark
|
|
963
|
+
* spec (no list-item-context awareness); adequate for this mechanical,
|
|
964
|
+
* warn-only rule.
|
|
965
|
+
*/
|
|
966
|
+
function computeIndentedCodeSpans(content) {
|
|
967
|
+
const spans = [];
|
|
968
|
+
const lines = content.split("\n");
|
|
969
|
+
let offset = 0;
|
|
970
|
+
let blockStart = -1;
|
|
971
|
+
let blockEnd = -1;
|
|
972
|
+
let prevBlank = true;
|
|
973
|
+
for (const line of lines) {
|
|
974
|
+
const lineStart = offset;
|
|
975
|
+
const lineEnd = offset + line.length;
|
|
976
|
+
const isBlank = line.trim() === "";
|
|
977
|
+
const isIndented = /^( {4,}|\t)/.test(line);
|
|
978
|
+
if (!isBlank && isIndented && (blockStart !== -1 || prevBlank)) {
|
|
979
|
+
if (blockStart === -1)
|
|
980
|
+
blockStart = lineStart;
|
|
981
|
+
blockEnd = lineEnd;
|
|
982
|
+
}
|
|
983
|
+
else if (isBlank && blockStart !== -1) {
|
|
984
|
+
// blank line inside an open block: keep it open, don't extend blockEnd
|
|
985
|
+
}
|
|
986
|
+
else if (!isBlank) {
|
|
987
|
+
if (blockStart !== -1)
|
|
988
|
+
spans.push([blockStart, blockEnd]);
|
|
989
|
+
blockStart = -1;
|
|
990
|
+
blockEnd = -1;
|
|
991
|
+
}
|
|
992
|
+
prevBlank = isBlank;
|
|
993
|
+
offset = lineEnd + 1;
|
|
994
|
+
}
|
|
995
|
+
if (blockStart !== -1)
|
|
996
|
+
spans.push([blockStart, blockEnd]);
|
|
997
|
+
return spans;
|
|
998
|
+
}
|
|
999
|
+
/**
|
|
1000
|
+
* Char spans of every inline code span in `content` (`` `...` ``, or a
|
|
1001
|
+
* longer run of backticks as the delimiter). This is a superset of the
|
|
1002
|
+
* narrower "immediately adjacent to a backtick" guard already applied in
|
|
1003
|
+
* collectShortFormMatches: that guard only rejects a match directly
|
|
1004
|
+
* touching a backtick, so a short form embedded further inside a longer
|
|
1005
|
+
* inline code span (e.g. `` `ports (1-3)` ``) was previously matched and
|
|
1006
|
+
* bound; this closes that gap.
|
|
1007
|
+
*
|
|
1008
|
+
* Confined to a single line (the pairing regex excludes `\n`): CommonMark
|
|
1009
|
+
* inline code spans cannot themselves span multiple lines in the sense
|
|
1010
|
+
* this rule cares about, but more importantly, an unmatched backtick run
|
|
1011
|
+
* (a typo, or a literal backtick in prose) previously paired greedily with
|
|
1012
|
+
* the *next* backtick run anywhere later in the whole document, silently
|
|
1013
|
+
* treating everything in between -- potentially several unrelated
|
|
1014
|
+
* sentences and any short-form citation among them -- as one giant inline
|
|
1015
|
+
* code span. Confining the match to a single line means a backtick run
|
|
1016
|
+
* with no partner on the same line produces no span at all, instead of
|
|
1017
|
+
* reaching across a newline for one.
|
|
1018
|
+
*/
|
|
1019
|
+
function computeInlineCodeSpans(content) {
|
|
1020
|
+
const spans = [];
|
|
1021
|
+
const re = /(`+)[^`\n]*?\1/g;
|
|
1022
|
+
let m;
|
|
1023
|
+
while ((m = re.exec(content)) !== null) {
|
|
1024
|
+
spans.push([m.index, m.index + m[0].length]);
|
|
1025
|
+
}
|
|
1026
|
+
return spans;
|
|
1027
|
+
}
|
|
1028
|
+
/**
|
|
1029
|
+
* Char spans of every Markdown table row in `content`: a line whose
|
|
1030
|
+
* trimmed form starts and ends with `|`. Decision (documented in the
|
|
1031
|
+
* README): a short-form citation inside a table cell is never recognised,
|
|
1032
|
+
* the same way one inside a code span is not -- excluded here rather than
|
|
1033
|
+
* left to the plausibility gate, since a table cell's content is prose-like
|
|
1034
|
+
* and can otherwise carry a range shape the gate would not reject (e.g.
|
|
1035
|
+
* `| col (5-9) |`).
|
|
1036
|
+
*/
|
|
1037
|
+
function computeTableRowSpans(content) {
|
|
1038
|
+
const spans = [];
|
|
1039
|
+
const lines = content.split("\n");
|
|
1040
|
+
let offset = 0;
|
|
1041
|
+
for (const line of lines) {
|
|
1042
|
+
const trimmed = line.trim();
|
|
1043
|
+
if (trimmed.startsWith("|") &&
|
|
1044
|
+
trimmed.endsWith("|") &&
|
|
1045
|
+
trimmed.length > 1) {
|
|
1046
|
+
spans.push([offset, offset + line.length]);
|
|
1047
|
+
}
|
|
1048
|
+
offset += line.length + 1;
|
|
1049
|
+
}
|
|
1050
|
+
return spans;
|
|
1051
|
+
}
|
|
1052
|
+
/**
|
|
1053
|
+
* All char spans short-form matching must never fire inside: fenced code,
|
|
1054
|
+
* indented code, inline code spans, and Markdown table rows. Computed once
|
|
1055
|
+
* per doc and combined with fullSpans (see scanDoc) via the existing
|
|
1056
|
+
* isWithinAnySpan helper -- the same mechanism a full citation's own span
|
|
1057
|
+
* already uses, not a second one.
|
|
1058
|
+
*/
|
|
1059
|
+
function computeExcludedSpans(content) {
|
|
1060
|
+
return [
|
|
1061
|
+
...computeFencedSpans(content),
|
|
1062
|
+
...computeIndentedCodeSpans(content),
|
|
1063
|
+
...computeInlineCodeSpans(content),
|
|
1064
|
+
...computeTableRowSpans(content),
|
|
1065
|
+
];
|
|
1066
|
+
}
|
|
1067
|
+
/**
|
|
1068
|
+
* Paragraph-start offsets in `content`, ascending, always including 0. A
|
|
1069
|
+
* paragraph boundary is a blank (empty or whitespace-only) line.
|
|
1070
|
+
*/
|
|
1071
|
+
function computeParagraphStarts(content) {
|
|
1072
|
+
const starts = [0];
|
|
1073
|
+
const re = /\n[ \t]*\n+/g;
|
|
1074
|
+
let m;
|
|
1075
|
+
while ((m = re.exec(content)) !== null) {
|
|
1076
|
+
starts.push(m.index + m[0].length);
|
|
1077
|
+
}
|
|
1078
|
+
return starts;
|
|
1079
|
+
}
|
|
1080
|
+
/** The start offset of the paragraph containing `index` (see computeParagraphStarts). */
|
|
1081
|
+
function paragraphStartFor(starts, index) {
|
|
1082
|
+
let result = starts[0];
|
|
1083
|
+
for (const s of starts) {
|
|
1084
|
+
if (s > index)
|
|
1085
|
+
break;
|
|
1086
|
+
result = s;
|
|
1087
|
+
}
|
|
1088
|
+
return result;
|
|
1089
|
+
}
|
|
1090
|
+
/**
|
|
1091
|
+
* The nearest full citation strictly before `beforeIndex` and at or after
|
|
1092
|
+
* `paragraphStart` -- "the last target document named earlier in the same
|
|
1093
|
+
* paragraph" -- or null when none exists.
|
|
1094
|
+
*/
|
|
1095
|
+
function findLastNamedTargetInParagraph(fullAtoms, paragraphStart, beforeIndex) {
|
|
1096
|
+
let best = null;
|
|
1097
|
+
for (const a of fullAtoms) {
|
|
1098
|
+
if (a.index >= paragraphStart && a.index < beforeIndex) {
|
|
1099
|
+
if (!best || a.index > best.index)
|
|
1100
|
+
best = a;
|
|
1101
|
+
}
|
|
1102
|
+
}
|
|
1103
|
+
return best;
|
|
1104
|
+
}
|
|
1105
|
+
function pushDrift(findings, file, citation, rule, message, resolvedTo, severity = "warning") {
|
|
1106
|
+
findings.push({
|
|
1107
|
+
ruleId: RULE_ID,
|
|
1108
|
+
severity,
|
|
1109
|
+
file,
|
|
1110
|
+
message: `\`${citation}\`: ${message} [${rule}]`,
|
|
1111
|
+
...(resolvedTo ? { detail: `resolvedTo: ${resolvedTo}` } : {}),
|
|
1112
|
+
});
|
|
1113
|
+
}
|
|
1114
|
+
function pushAmbiguous(findings, file, citation, candidates) {
|
|
1115
|
+
findings.push({
|
|
1116
|
+
ruleId: RULE_ID,
|
|
1117
|
+
severity: "notice",
|
|
1118
|
+
file,
|
|
1119
|
+
message: `\`${citation}\`: ambiguous target, not evaluated [unresolved-ambiguous]`,
|
|
1120
|
+
detail: `candidates: ${candidates.join(", ")}`,
|
|
1121
|
+
});
|
|
1122
|
+
}
|
|
1123
|
+
function pushUnreadable(findings, file, citation, resolvedTo, code) {
|
|
1124
|
+
findings.push({
|
|
1125
|
+
ruleId: RULE_ID,
|
|
1126
|
+
severity: "notice",
|
|
1127
|
+
file,
|
|
1128
|
+
message: `\`${citation}\`: target file exists but could not be read [unreadable-target]`,
|
|
1129
|
+
detail: `resolvedTo: ${resolvedTo}, errorCode: ${code}`,
|
|
1130
|
+
});
|
|
1131
|
+
}
|
|
1132
|
+
/**
|
|
1133
|
+
* True when a `CITATION_RE` match at `matchIndex` is the phantom tail of a
|
|
1134
|
+
* filename hard-wrapped across a line break (see the "Hard-wrapped prose"
|
|
1135
|
+
* doc block above): the match starts at column 0 of its line, optionally
|
|
1136
|
+
* after only whitespace or a list/quote marker, and the previous line ends
|
|
1137
|
+
* with `-` or `–`.
|
|
1138
|
+
*/
|
|
1139
|
+
function isWrappedPathContinuation(content, matchIndex) {
|
|
1140
|
+
const lineStart = content.lastIndexOf("\n", matchIndex - 1) + 1;
|
|
1141
|
+
if (lineStart === 0)
|
|
1142
|
+
return false; // no previous line to have wrapped from
|
|
1143
|
+
const prefix = content.slice(lineStart, matchIndex);
|
|
1144
|
+
if (!/^[\s>*.\d-]*$/.test(prefix))
|
|
1145
|
+
return false;
|
|
1146
|
+
const prevLineEnd = lineStart - 1; // index of the newline just before lineStart
|
|
1147
|
+
const prevLineStart = content.lastIndexOf("\n", prevLineEnd - 1) + 1;
|
|
1148
|
+
const prevLine = content.slice(prevLineStart, prevLineEnd);
|
|
1149
|
+
return /[-–]$/.test(prevLine);
|
|
1150
|
+
}
|
|
1151
|
+
function scanDoc(cache, root, bundleDir, doc) {
|
|
1152
|
+
const findings = [];
|
|
1153
|
+
const content = doc.raw;
|
|
1154
|
+
const sources = getValidSources(doc.frontmatter.parsed) ?? [];
|
|
1155
|
+
const docAbsPath = path.join(bundleDir, doc.relPath);
|
|
1156
|
+
const fullAtoms = [];
|
|
1157
|
+
// Char spans of every matched full citation, used to keep short-form
|
|
1158
|
+
// matching (see collectShortFormMatches) from re-matching the tail of a
|
|
1159
|
+
// real `path:N-M` citation as a bare short form.
|
|
1160
|
+
const fullSpans = [];
|
|
1161
|
+
const re = new RegExp(CITATION_RE.source, "g");
|
|
1162
|
+
let m;
|
|
1163
|
+
while ((m = re.exec(content)) !== null) {
|
|
1164
|
+
if (isWrappedPathContinuation(content, m.index))
|
|
1165
|
+
continue;
|
|
1166
|
+
fullAtoms.push({
|
|
1167
|
+
kind: "full",
|
|
1168
|
+
index: m.index,
|
|
1169
|
+
citedPath: m[1],
|
|
1170
|
+
startLine: Number(m[2]),
|
|
1171
|
+
endLine: m[3] ? Number(m[3]) : null,
|
|
1172
|
+
anchor: parseAnchor(m[4]),
|
|
1173
|
+
});
|
|
1174
|
+
fullSpans.push([m.index, m.index + m[0].length]);
|
|
1175
|
+
}
|
|
1176
|
+
const atoms = [...fullAtoms, ...collectContinuationAtoms(content)].sort((a, b) => a.index - b.index);
|
|
1177
|
+
// `governing`: nearest preceding citation (full or continuation) that
|
|
1178
|
+
// resolved to a real file; see the "Continuation citations" doc block
|
|
1179
|
+
// above for the reset rules. `lastStartLine`: the start line a "cont-ext"
|
|
1180
|
+
// atom extends into a range; tracks the most recent full or cont-fresh
|
|
1181
|
+
// atom's own startLine, scoped together with `governing`.
|
|
1182
|
+
let governing = null;
|
|
1183
|
+
let lastStartLine = null;
|
|
1184
|
+
for (const atom of atoms) {
|
|
1185
|
+
if (atom.kind === "cont-ext") {
|
|
1186
|
+
if (!governing || lastStartLine === null)
|
|
1187
|
+
continue; // nothing to extend
|
|
1188
|
+
const citation = `${governing.citedPath}:${lastStartLine}-${atom.value} (continuation)`;
|
|
1189
|
+
const problem = checkRangeBoundOnly(lastStartLine, atom.value, governing.resolvedPath);
|
|
1190
|
+
if (problem?.rule === "unreadable-target") {
|
|
1191
|
+
pushUnreadable(findings, doc.relPath, citation, path.relative(root, governing.resolvedPath), problem.code ?? "UNKNOWN");
|
|
1192
|
+
}
|
|
1193
|
+
else if (problem) {
|
|
1194
|
+
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, governing.resolvedPath));
|
|
1195
|
+
}
|
|
1196
|
+
continue; // governing and lastStartLine both carry over unchanged
|
|
1197
|
+
}
|
|
1198
|
+
if (atom.kind === "cont-fresh") {
|
|
1199
|
+
if (!governing)
|
|
1200
|
+
continue; // nothing to validate a bare continuation against
|
|
1201
|
+
const { startLine, endLine } = atom;
|
|
1202
|
+
const citation = `${governing.citedPath}:${startLine}${endLine ? "-" + endLine : ""} (continuation)`;
|
|
1203
|
+
const problem = checkTarget(governing.citedPath, startLine, endLine, governing.resolvedPath);
|
|
1204
|
+
if (problem?.rule === "unreadable-target") {
|
|
1205
|
+
pushUnreadable(findings, doc.relPath, citation, path.relative(root, governing.resolvedPath), problem.code ?? "UNKNOWN");
|
|
1206
|
+
}
|
|
1207
|
+
else if (problem) {
|
|
1208
|
+
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, governing.resolvedPath));
|
|
1209
|
+
}
|
|
1210
|
+
lastStartLine = startLine;
|
|
1211
|
+
continue; // governing (same file) carries over unchanged
|
|
1212
|
+
}
|
|
1213
|
+
const { citedPath, startLine, endLine, anchor } = atom;
|
|
1214
|
+
const citation = `${citedPath}:${startLine}${endLine ? "-" + endLine : ""}`;
|
|
1215
|
+
if (hasParentSegment(citedPath)) {
|
|
1216
|
+
pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
|
|
1217
|
+
governing = null;
|
|
1218
|
+
lastStartLine = null;
|
|
1219
|
+
continue;
|
|
1220
|
+
}
|
|
1221
|
+
const resolution = resolveCitation(cache, root, docAbsPath, content, sources, citedPath, atom.index);
|
|
1222
|
+
if (!resolution) {
|
|
1223
|
+
pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${citedPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
|
|
1224
|
+
governing = null;
|
|
1225
|
+
lastStartLine = null;
|
|
1226
|
+
continue;
|
|
1227
|
+
}
|
|
1228
|
+
if ("skip" in resolution) {
|
|
1229
|
+
governing = null;
|
|
1230
|
+
lastStartLine = null;
|
|
1231
|
+
continue;
|
|
1232
|
+
}
|
|
1233
|
+
if ("ambiguous" in resolution) {
|
|
1234
|
+
pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
|
|
1235
|
+
governing = null;
|
|
1236
|
+
lastStartLine = null;
|
|
1237
|
+
continue;
|
|
1238
|
+
}
|
|
1239
|
+
const problem = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor);
|
|
1240
|
+
if (problem?.rule === "unreadable-target") {
|
|
1241
|
+
pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
|
|
1242
|
+
}
|
|
1243
|
+
else if (problem) {
|
|
1244
|
+
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
|
|
1245
|
+
}
|
|
1246
|
+
governing = { citedPath, resolvedPath: resolution.path };
|
|
1247
|
+
lastStartLine = startLine;
|
|
1248
|
+
}
|
|
1249
|
+
// Short-form (paragraph-bound) citations -- see that doc block above.
|
|
1250
|
+
// Deliberately independent of `governing`/`lastStartLine`: short-form
|
|
1251
|
+
// binding is paragraph-scoped by design, not chained through the
|
|
1252
|
+
// document-wide continuation state machine above. Reserved files
|
|
1253
|
+
// (index.md, log.md, see doc.isReserved) are skipped entirely: they are
|
|
1254
|
+
// append-only narrative journals, not `sources:`-driven reference docs,
|
|
1255
|
+
// and routinely narrate historical "old N-M -> new X-Y" line-number
|
|
1256
|
+
// deltas as prose data about past changes -- not live citations against
|
|
1257
|
+
// current content -- which this rule's bare-range matching cannot tell
|
|
1258
|
+
// apart from a real short-form citation. Full/continuation citations in
|
|
1259
|
+
// reserved files are still scanned as before; this carve-out is scoped
|
|
1260
|
+
// to short-form matching only.
|
|
1261
|
+
const shortFormMatches = doc.isReserved
|
|
1262
|
+
? []
|
|
1263
|
+
: collectShortFormMatches(content, [
|
|
1264
|
+
...fullSpans,
|
|
1265
|
+
...computeExcludedSpans(content),
|
|
1266
|
+
]);
|
|
1267
|
+
if (shortFormMatches.length > 0) {
|
|
1268
|
+
const paragraphStarts = computeParagraphStarts(content);
|
|
1269
|
+
const namedFullAtoms = [];
|
|
1270
|
+
for (const a of fullAtoms) {
|
|
1271
|
+
if (a.kind === "full") {
|
|
1272
|
+
namedFullAtoms.push({
|
|
1273
|
+
index: a.index,
|
|
1274
|
+
citedPath: a.citedPath,
|
|
1275
|
+
});
|
|
1276
|
+
}
|
|
1277
|
+
}
|
|
1278
|
+
for (const sf of shortFormMatches) {
|
|
1279
|
+
const paragraphStart = paragraphStartFor(paragraphStarts, sf.index);
|
|
1280
|
+
const target = findLastNamedTargetInParagraph(namedFullAtoms, paragraphStart, sf.index);
|
|
1281
|
+
const rangeLabel = `${sf.startLine}-${sf.endLine}`;
|
|
1282
|
+
if (!target) {
|
|
1283
|
+
pushDrift(findings, doc.relPath, `${rangeLabel} (short-form)`, "short-form-unbound", "no full `path:N` citation earlier in this paragraph to bind to", undefined, "notice");
|
|
1284
|
+
continue;
|
|
1285
|
+
}
|
|
1286
|
+
const targetPath = target.citedPath;
|
|
1287
|
+
const citation = `${targetPath}:${rangeLabel} (short-form)`;
|
|
1288
|
+
if (hasParentSegment(targetPath)) {
|
|
1289
|
+
// The full citation that named this target already reported
|
|
1290
|
+
// path-traversal-rejected for itself; not re-flagged a second time.
|
|
1291
|
+
continue;
|
|
1292
|
+
}
|
|
1293
|
+
const resolution = resolveCitation(cache, root, docAbsPath, content, sources, targetPath, sf.index);
|
|
1294
|
+
if (!resolution) {
|
|
1295
|
+
pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${targetPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
|
|
1296
|
+
continue;
|
|
1297
|
+
}
|
|
1298
|
+
if ("skip" in resolution)
|
|
1299
|
+
continue;
|
|
1300
|
+
if ("ambiguous" in resolution) {
|
|
1301
|
+
pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
|
|
1302
|
+
continue;
|
|
1303
|
+
}
|
|
1304
|
+
const problem = checkShortFormTarget(targetPath, sf.startLine, sf.endLine, resolution.path);
|
|
1305
|
+
if (problem?.rule === "unreadable-target") {
|
|
1306
|
+
pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
|
|
1307
|
+
}
|
|
1308
|
+
else if (problem) {
|
|
1309
|
+
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path), problem.severity);
|
|
1310
|
+
}
|
|
1311
|
+
}
|
|
1312
|
+
}
|
|
1313
|
+
return findings;
|
|
1314
|
+
}
|
|
1315
|
+
export const citationsResolveRule = {
|
|
1316
|
+
id: RULE_ID,
|
|
1317
|
+
description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations) must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
|
|
1318
|
+
run(ctx) {
|
|
1319
|
+
if (!ctx.repoRoot) {
|
|
1320
|
+
// Never silently skip: mirrors sources-fresh's posture so a "clean"
|
|
1321
|
+
// run is never a fake pass when there is no filesystem tree to
|
|
1322
|
+
// resolve citation targets against.
|
|
1323
|
+
return [
|
|
1324
|
+
{
|
|
1325
|
+
ruleId: RULE_ID,
|
|
1326
|
+
severity: "notice",
|
|
1327
|
+
file: "",
|
|
1328
|
+
message: "citation resolution skipped: not inside a git work tree",
|
|
1329
|
+
},
|
|
1330
|
+
];
|
|
1331
|
+
}
|
|
1332
|
+
const root = ctx.repoRoot;
|
|
1333
|
+
// Fresh per invocation: see findByBasename's doc comment for why this
|
|
1334
|
+
// is not held at module scope.
|
|
1335
|
+
const cache = new Map();
|
|
1336
|
+
return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc));
|
|
1337
|
+
},
|
|
1338
|
+
};
|
|
1339
|
+
//# sourceMappingURL=citations-resolve.js.map
|