okf-kit 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +297 -0
- package/README.md +36 -3
- package/dist/rules/citations-resolve.js +833 -14
- package/dist/rules/citations-resolve.js.map +1 -1
- package/package.json +1 -1
|
@@ -130,7 +130,256 @@ const RULE_ID = "citations-resolve";
|
|
|
130
130
|
* this rule emits a single bundle-level notice, matching `sources-fresh`'s
|
|
131
131
|
* "not inside a git work tree" posture.
|
|
132
132
|
*/
|
|
133
|
-
|
|
133
|
+
/**
|
|
134
|
+
* Anchored citations. A full citation (never a continuation or short-form
|
|
135
|
+
* atom -- see below) may carry an anchor directly after its range,
|
|
136
|
+
* `path:N-M#anchor`, e.g. `` `CHANGELOG.md:50-144#0.24.0` ``. Motivation: a
|
|
137
|
+
* CHANGELOG.md grows by insertion at the TOP (newest entry first), so every
|
|
138
|
+
* later entry's absolute line numbers shift on every release; the checks
|
|
139
|
+
* above (missing-file, inverted-range, range-exceeds-file, blank-start-line,
|
|
140
|
+
* closing-brace-start-line) are structurally blind to a citation that
|
|
141
|
+
* shifted a whole release section over and now lands, still non-blank and
|
|
142
|
+
* in-bounds, inside the WRONG section -- a citation `CHANGELOG.md:738-748`
|
|
143
|
+
* meant for the "0.7.4" entry that quietly now points at "0.7.3" prose is
|
|
144
|
+
* exactly as green as before the shift. An anchor closes that gap by
|
|
145
|
+
* pinning the citation to a piece of the target's own structure/content that
|
|
146
|
+
* the line-shift does not preserve automatically.
|
|
147
|
+
*
|
|
148
|
+
* Two anchor kinds, told apart by the raw anchor text (group 4 of
|
|
149
|
+
* `CITATION_RE`, see `parseAnchor`):
|
|
150
|
+
* - **Heading form** (bare, unquoted, e.g. `#0.24.0` or `#[0.24.0]`):
|
|
151
|
+
* the target's nearest *enclosing* Markdown heading -- see
|
|
152
|
+
* `findEnclosingHeading` -- must contain the anchor text, and no
|
|
153
|
+
* heading of the same or shallower level may start before the range's
|
|
154
|
+
* end line (i.e. the heading must actually enclose the whole range, not
|
|
155
|
+
* merely precede its start -- see `checkAnchor`). Deliberately capped at
|
|
156
|
+
* `ANCHOR_HEADING_MAX_LEVEL` (2): a Keep-a-Changelog CHANGELOG.md nests
|
|
157
|
+
* `## [x.y.z]` release headings around identically-named `### Added` /
|
|
158
|
+
* `### Changed` / `### Fixed` subsections repeated in every release;
|
|
159
|
+
* treating "nearest heading of any level" as the anchor target would
|
|
160
|
+
* make a heading anchor nearly useless here (matching the wrong
|
|
161
|
+
* release's own "Changed" subsection just as readily as the right
|
|
162
|
+
* one's), so subsection headings are transparent to this check and only
|
|
163
|
+
* level-1/level-2 headings are ever considered.
|
|
164
|
+
* - **String form** (double-quoted, e.g. `#"reproduction requirement"`):
|
|
165
|
+
* the anchor text must occur, verbatim, on at least one line of the
|
|
166
|
+
* cited range itself (`checkAnchor`'s string branch) -- "occurs inside
|
|
167
|
+
* it" rather than "encloses it". No heading structure required, so this
|
|
168
|
+
* form also works against a non-Markdown target (`.ts`/`.js`/...) where
|
|
169
|
+
* "nearest enclosing heading" has no meaning.
|
|
170
|
+
* An anchor mismatch is its own drift finding (`anchor-heading-not-found`,
|
|
171
|
+
* `anchor-heading-mismatch`, `anchor-heading-does-not-enclose`,
|
|
172
|
+
* `anchor-not-found-in-range`), checked only once the base start/range
|
|
173
|
+
* checks (blank-start-line, closing-brace-start-line, inverted-range,
|
|
174
|
+
* range-exceeds-file) already came back clean -- same "one problem per
|
|
175
|
+
* citation, base checks first" pattern `checkShortFormTarget` already uses
|
|
176
|
+
* for the block-boundary check. Every one of these four messages names the
|
|
177
|
+
* anchor text itself, and an anchored full citation's own finding label
|
|
178
|
+
* (`path:N-M#anchor`, see `formatAnchorForLabel`) carries the anchor too --
|
|
179
|
+
* without both, two citations to the identical range with different
|
|
180
|
+
* anchors would be indistinguishable in the output.
|
|
181
|
+
*
|
|
182
|
+
* A `#` that immediately follows a citation's range but does not parse as
|
|
183
|
+
* either anchor form at all (unbalanced quotes, a backtick inside a quoted
|
|
184
|
+
* anchor, or nothing after the `#`) is its own separate, `notice`-severity
|
|
185
|
+
* finding, `anchor-malformed` (see `scanDoc`'s main matching loop): the
|
|
186
|
+
* citation is still checked exactly as an anchorless one would be
|
|
187
|
+
* (backward compatible, see below), but silently checking it anchorless
|
|
188
|
+
* when the `#` right there looks like a typo'd anchor attempt would defeat
|
|
189
|
+
* the entire point of writing one -- a single misplaced character would
|
|
190
|
+
* quietly turn the very check the anchor was written for back off.
|
|
191
|
+
*
|
|
192
|
+
* Backward compatible by construction: the `#anchor` suffix is optional in
|
|
193
|
+
* `CITATION_RE`, so an existing anchorless citation matches exactly as
|
|
194
|
+
* before and is checked exactly as before (this section adds a new check
|
|
195
|
+
* gated on the anchor being present, it does not change any existing one).
|
|
196
|
+
* Deliberately scoped to full citations only: a continuation (`` `:M` ``
|
|
197
|
+
* etc.) or a short-form `:N-M` never carries its own path, so there is
|
|
198
|
+
* nowhere natural to hang an anchor on one without inventing a second,
|
|
199
|
+
* detached syntax; the migration this rule was built for (CHANGELOG.md
|
|
200
|
+
* citations) is written as full citations throughout the corpus it targets.
|
|
201
|
+
*
|
|
202
|
+
* Rejected alternatives (see the PR/CHANGELOG for the fuller writeup):
|
|
203
|
+
* - Embedding the literal heading markup itself, e.g.
|
|
204
|
+
* `` `CHANGELOG.md:50-144#"## [0.24.0]"` `` -- verbatim-correct but
|
|
205
|
+
* `#`, `[`, `]`, and the space all need quoting/escaping right next to
|
|
206
|
+
* the citation, which reads worse in prose than a bare version token
|
|
207
|
+
* and gains nothing the structural heading-enclosure check does not
|
|
208
|
+
* already provide from the shorter form.
|
|
209
|
+
* - A detached anchor, e.g. a trailing parenthetical `(see "0.24.0")`
|
|
210
|
+
* elsewhere in the sentence -- unparseable without a second, separate
|
|
211
|
+
* grammar next to the existing continuation/short-form machinery, and
|
|
212
|
+
* easy to leave behind (or attach to the wrong citation) when a
|
|
213
|
+
* sentence is edited later.
|
|
214
|
+
* - A named-capture-group slug matching `sources-fresh`'s YAML shape --
|
|
215
|
+
* rejected because it would require a second citation site (frontmatter
|
|
216
|
+
* plus prose) to stay in sync, the exact class of drift this rule
|
|
217
|
+
* exists to catch.
|
|
218
|
+
*/
|
|
219
|
+
const ANCHOR_HEADING_MAX_LEVEL = 2;
|
|
220
|
+
const MD_HEADING_RE = /^(#{1,6})\s+(.*)$/;
|
|
221
|
+
/**
|
|
222
|
+
* Renders a parsed `Anchor` back to the short form used to disambiguate a
|
|
223
|
+
* finding's citation label (see the `citation` variable in `scanDoc`'s main
|
|
224
|
+
* loop): `#text` for a heading anchor (already bracket-stripped by
|
|
225
|
+
* `parseAnchor`), `#"text"` for a string anchor. Two citations to the same
|
|
226
|
+
* range with different anchors would otherwise be indistinguishable in the
|
|
227
|
+
* output -- see the "Anchored citations" doc block above.
|
|
228
|
+
*/
|
|
229
|
+
function formatAnchorForLabel(anchor) {
|
|
230
|
+
return anchor.kind === "string" ? `#"${anchor.text}"` : `#${anchor.text}`;
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* Parses `CITATION_RE`'s optional 4th capture group (the raw anchor text,
|
|
234
|
+
* including its surrounding quotes or brackets if any) into an `Anchor`, or
|
|
235
|
+
* `null` when the citation carried no `#anchor` suffix at all. A
|
|
236
|
+
* double-quoted raw value (`"..."`) is the string form, text taken verbatim
|
|
237
|
+
* between the quotes; anything else is the heading form, with a single
|
|
238
|
+
* wrapping `[...]` stripped (so `#[0.24.0]` and `#0.24.0` compare
|
|
239
|
+
* identically) -- compared as a plain substring against the heading's own
|
|
240
|
+
* raw text (`findEnclosingHeading`'s `text`, itself never stripped of any
|
|
241
|
+
* brackets it happens to carry), not a stripped copy of it -- see the
|
|
242
|
+
* "Anchored citations" doc block above.
|
|
243
|
+
*/
|
|
244
|
+
function parseAnchor(raw) {
|
|
245
|
+
if (!raw)
|
|
246
|
+
return null;
|
|
247
|
+
if (raw.length >= 2 && raw.startsWith('"') && raw.endsWith('"')) {
|
|
248
|
+
return { kind: "string", text: raw.slice(1, -1) };
|
|
249
|
+
}
|
|
250
|
+
const text = raw.startsWith("[") && raw.endsWith("]") ? raw.slice(1, -1) : raw;
|
|
251
|
+
return { kind: "heading", text };
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* The single fence state machine every fence-aware consumer in this file
|
|
255
|
+
* derives from: one forward pass over `lines`, replaying the same
|
|
256
|
+
* open/close logic (a line matching `MD_FENCE_DELIM_RE` opens a fence when
|
|
257
|
+
* none is open, a line starting with the *same* marker closes it) that
|
|
258
|
+
* `computeFencedSpans`, `computeFencedLineIndices`, and `isFenceOpeningLine`
|
|
259
|
+
* each used to implement as their own, independently-maintained copy --
|
|
260
|
+
* nothing enforced the three staying in agreement with each other. An
|
|
261
|
+
* unterminated fence (no matching close before the end of `lines`) is
|
|
262
|
+
* reflected by every remaining line coming back `fenced: true` -- each line
|
|
263
|
+
* is marked while the scan is still inside the open fence, so no separate
|
|
264
|
+
* post-loop fixup is needed the way the three original copies each had.
|
|
265
|
+
* State at line `i` never depends on any line after `i`, so a caller that
|
|
266
|
+
* only needs one line's state (see `isFenceOpeningLine`) can safely ignore
|
|
267
|
+
* the rest of the array without re-deriving the logic itself.
|
|
268
|
+
*/
|
|
269
|
+
function scanFenceLines(lines) {
|
|
270
|
+
const states = [];
|
|
271
|
+
let fenceMarker;
|
|
272
|
+
for (let i = 0; i < lines.length; i++) {
|
|
273
|
+
const trimmed = (lines[i] ?? "").trim();
|
|
274
|
+
if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
275
|
+
fenceMarker = trimmed.slice(0, 3);
|
|
276
|
+
states.push({ fenced: true, opensFence: true, closesFence: false });
|
|
277
|
+
}
|
|
278
|
+
else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
279
|
+
states.push({ fenced: true, opensFence: false, closesFence: true });
|
|
280
|
+
fenceMarker = undefined;
|
|
281
|
+
}
|
|
282
|
+
else {
|
|
283
|
+
states.push({
|
|
284
|
+
fenced: fenceMarker !== undefined,
|
|
285
|
+
opensFence: false,
|
|
286
|
+
closesFence: false,
|
|
287
|
+
});
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
return states;
|
|
291
|
+
}
|
|
292
|
+
/**
|
|
293
|
+
* 0-based line indices that fall inside a fenced code block (```` ``` ````
|
|
294
|
+
* or `~~~`, optionally with a trailing language tag), delimiters included --
|
|
295
|
+
* the target-side twin of `computeFencedSpans` above, which does the same
|
|
296
|
+
* job for the *citing* doc's short-form matching. Derived from
|
|
297
|
+
* `scanFenceLines` (see there). Anchor heading-search needs its own copy
|
|
298
|
+
* because it works from an already-split `lines` array (see
|
|
299
|
+
* `checkFullTarget`), not the raw `content` string `computeFencedSpans`
|
|
300
|
+
* takes, and because it must ignore a target's `# not a heading` sitting
|
|
301
|
+
* inside a fenced example exactly the same way a citing doc's own fences
|
|
302
|
+
* are already ignored for short-form matching -- without this, a `#`-led
|
|
303
|
+
* comment line inside e.g. a fenced shell example in the target is
|
|
304
|
+
* indistinguishable from a real Markdown heading to `MD_HEADING_RE`, and
|
|
305
|
+
* both `findEnclosingHeading` (picks the wrong "nearest" heading) and the
|
|
306
|
+
* enclosure walk in `checkAnchor` (treats it as a section boundary) would
|
|
307
|
+
* misfire on it.
|
|
308
|
+
*/
|
|
309
|
+
function computeFencedLineIndices(lines) {
|
|
310
|
+
const fenced = new Set();
|
|
311
|
+
scanFenceLines(lines).forEach((state, i) => {
|
|
312
|
+
if (state.fenced)
|
|
313
|
+
fenced.add(i);
|
|
314
|
+
});
|
|
315
|
+
return fenced;
|
|
316
|
+
}
|
|
317
|
+
/**
|
|
318
|
+
* Nearest Markdown heading at or before `startLine` (1-based), considering
|
|
319
|
+
* only heading levels up to `ANCHOR_HEADING_MAX_LEVEL` -- see the "Anchored
|
|
320
|
+
* citations" doc block above for why subsection headings are transparent to
|
|
321
|
+
* this search. `fencedLines` (see `computeFencedLineIndices`) excludes any
|
|
322
|
+
* line inside a fenced code block from matching, so a `#`-led comment
|
|
323
|
+
* inside a fenced example is never mistaken for a heading. Returns `null`
|
|
324
|
+
* when no such heading precedes `startLine` at all (e.g. the citation
|
|
325
|
+
* lands above the target's first release heading).
|
|
326
|
+
*/
|
|
327
|
+
function findEnclosingHeading(lines, startLine, fencedLines) {
|
|
328
|
+
for (let i = startLine - 1; i >= 0; i--) {
|
|
329
|
+
if (fencedLines.has(i))
|
|
330
|
+
continue;
|
|
331
|
+
const m = (lines[i] ?? "").match(MD_HEADING_RE);
|
|
332
|
+
if (m && m[1].length <= ANCHOR_HEADING_MAX_LEVEL) {
|
|
333
|
+
return { level: m[1].length, text: m[2].trim(), lineNo: i + 1 };
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
return null;
|
|
337
|
+
}
|
|
338
|
+
/**
|
|
339
|
+
* Checks an anchored full citation's anchor against its already-resolved,
|
|
340
|
+
* already-range-checked target -- see the "Anchored citations" doc block
|
|
341
|
+
* above for the two forms' semantics. `endLine` is the citation's end line,
|
|
342
|
+
* or its start line for a single-line citation (a size-1 range).
|
|
343
|
+
*/
|
|
344
|
+
function checkAnchor(anchor, startLine, endLine, lines) {
|
|
345
|
+
if (anchor.kind === "string") {
|
|
346
|
+
for (let i = startLine - 1; i <= endLine - 1 && i < lines.length; i++) {
|
|
347
|
+
if ((lines[i] ?? "").includes(anchor.text))
|
|
348
|
+
return null;
|
|
349
|
+
}
|
|
350
|
+
return {
|
|
351
|
+
rule: "anchor-not-found-in-range",
|
|
352
|
+
message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
const fencedLines = computeFencedLineIndices(lines);
|
|
356
|
+
const heading = findEnclosingHeading(lines, startLine, fencedLines);
|
|
357
|
+
if (!heading) {
|
|
358
|
+
return {
|
|
359
|
+
rule: "anchor-heading-not-found",
|
|
360
|
+
message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor "${anchor.text}" against; use a string anchor (#"...") instead against a target with no heading structure`,
|
|
361
|
+
};
|
|
362
|
+
}
|
|
363
|
+
if (!heading.text.includes(anchor.text)) {
|
|
364
|
+
return {
|
|
365
|
+
rule: "anchor-heading-mismatch",
|
|
366
|
+
message: `nearest enclosing heading ("${heading.text}") does not contain anchor "${anchor.text}"`,
|
|
367
|
+
};
|
|
368
|
+
}
|
|
369
|
+
for (let i = heading.lineNo; i <= endLine - 1; i++) {
|
|
370
|
+
if (fencedLines.has(i))
|
|
371
|
+
continue;
|
|
372
|
+
const m = (lines[i] ?? "").match(MD_HEADING_RE);
|
|
373
|
+
if (m && m[1].length <= heading.level) {
|
|
374
|
+
return {
|
|
375
|
+
rule: "anchor-heading-does-not-enclose",
|
|
376
|
+
message: `range extends past the section enclosing anchor "${anchor.text}" (next heading "${m[2].trim()}" at line ${i + 1})`,
|
|
377
|
+
};
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
return null;
|
|
381
|
+
}
|
|
382
|
+
const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
|
|
134
383
|
// Continuation citation forms (see the "Continuation citations" doc block
|
|
135
384
|
// above). Each requires the backtick delimiter as part of the match so it
|
|
136
385
|
// can never overlap a CITATION_RE match: a full citation's regex match
|
|
@@ -141,6 +390,28 @@ const CONT_DASH_RE = /[-–]`(\d+)`/g;
|
|
|
141
390
|
const CONT_PAREN_RE = /\(`(\d+)`\)/g;
|
|
142
391
|
const CLOSING_ONLY_EXTS = new Set(["ts", "js", "mjs", "yml", "yaml", "json"]);
|
|
143
392
|
const CLOSING_BRACE_RE = /^[)\]}][;,]?$/;
|
|
393
|
+
// Short-form (paragraph-bound) citation form, see the "Short-form
|
|
394
|
+
// citations" doc block below. Requires an explicit N-M range: a bare single
|
|
395
|
+
// number (`:5`) is not matched -- see that doc block for why. Only the
|
|
396
|
+
// colon form is collected; a bare `(N-M)` is never a short-form citation
|
|
397
|
+
// candidate at all -- see the same doc block for why the paren form was
|
|
398
|
+
// dropped rather than gated.
|
|
399
|
+
const SHORT_FORM_COLON_RE = /:(\d+)-(\d+)/g;
|
|
400
|
+
// Test-file block-boundary check (see checkRangeBoundary): a citation's
|
|
401
|
+
// range into a `.test.ts`/`.spec.ts` (or `.js`/`.mjs` equivalent) target
|
|
402
|
+
// must start on a describe(/it( head line and end on its matching closing
|
|
403
|
+
// `});` line.
|
|
404
|
+
const TEST_FILE_RE = /\.(test|spec)\.(ts|js|mjs)$/i;
|
|
405
|
+
const TEST_HEAD_LINE_RE = /^\s*(?:describe|it)\s*\(/;
|
|
406
|
+
const TEST_CLOSING_LINE_RE = /^\s*\}\)\s*;\s*$/;
|
|
407
|
+
// Markdown block-boundary check (see checkRangeBoundary): a range boundary
|
|
408
|
+
// line that is nothing but a bracket (open or close), optionally with a
|
|
409
|
+
// trailing `,`/`;`, is always a drift signal. A bare code-fence delimiter is
|
|
410
|
+
// its own, separate signal (see MD_FENCE_DELIM_RE): unlike a bracket, a
|
|
411
|
+
// fence line is sometimes the deliberately-correct start of a citation (see
|
|
412
|
+
// checkRangeBoundary's markdown branch for the opening-fence exception).
|
|
413
|
+
const MD_BARE_BRACKET_RE = /^(?:[)\]}][;,]?|[[({])$/;
|
|
414
|
+
const MD_FENCE_DELIM_RE = /^(?:`{3,}\S*|~{3,}\S*)$/;
|
|
144
415
|
const EXCLUDED_DIRS = new Set([
|
|
145
416
|
"node_modules",
|
|
146
417
|
".git",
|
|
@@ -367,14 +638,15 @@ function readTarget(resolvedPath) {
|
|
|
367
638
|
};
|
|
368
639
|
}
|
|
369
640
|
}
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
641
|
+
/**
|
|
642
|
+
* The non-I/O half of checkTarget: every check that only needs the
|
|
643
|
+
* target's already-read `lines`, not the filesystem. Split out so a caller
|
|
644
|
+
* that needs a second, different check against the same target (see
|
|
645
|
+
* checkShortFormTarget) can read the file once and reuse `lines` for both,
|
|
646
|
+
* instead of checkTarget re-reading it internally.
|
|
647
|
+
*/
|
|
648
|
+
function checkTargetLines(citedPath, startLine, endLine, lines) {
|
|
649
|
+
const bound = checkRangeBound(startLine, endLine, lines.length);
|
|
378
650
|
if (bound)
|
|
379
651
|
return bound;
|
|
380
652
|
const startText = lines[startLine - 1] ?? "";
|
|
@@ -391,6 +663,34 @@ function checkTarget(citedPath, startLine, endLine, resolvedPath) {
|
|
|
391
663
|
}
|
|
392
664
|
return null;
|
|
393
665
|
}
|
|
666
|
+
function checkTarget(citedPath, startLine, endLine, resolvedPath) {
|
|
667
|
+
const read = readTarget(resolvedPath);
|
|
668
|
+
if ("rule" in read)
|
|
669
|
+
return read;
|
|
670
|
+
return checkTargetLines(citedPath, startLine, endLine, splitLines(read.content));
|
|
671
|
+
}
|
|
672
|
+
/**
|
|
673
|
+
* A full citation's complete check: `checkTargetLines`'s existing checks
|
|
674
|
+
* (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
|
|
675
|
+
* closing-brace-start-line), plus, only when those all pass and the
|
|
676
|
+
* citation carried an anchor, the anchor check above (see the "Anchored
|
|
677
|
+
* citations" doc block). Reads the resolved target once, mirroring
|
|
678
|
+
* `checkShortFormTarget`'s reasoning for the block-boundary check. Anchors
|
|
679
|
+
* are full-citation-only (see that doc block for why), so `checkTarget`
|
|
680
|
+
* itself is untouched and still used as-is for a cont-fresh atom.
|
|
681
|
+
*/
|
|
682
|
+
function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor) {
|
|
683
|
+
const read = readTarget(resolvedPath);
|
|
684
|
+
if ("rule" in read)
|
|
685
|
+
return read;
|
|
686
|
+
const lines = splitLines(read.content);
|
|
687
|
+
const base = checkTargetLines(citedPath, startLine, endLine, lines);
|
|
688
|
+
if (base)
|
|
689
|
+
return base;
|
|
690
|
+
if (!anchor)
|
|
691
|
+
return null;
|
|
692
|
+
return checkAnchor(anchor, startLine, endLine ?? startLine, lines);
|
|
693
|
+
}
|
|
394
694
|
// A cont-ext atom only ever extends the *end* of a range whose start line
|
|
395
695
|
// was already fully checked (blank / closing-brace) when it was cited as
|
|
396
696
|
// its own full citation or cont-fresh atom -- re-running checkTarget here
|
|
@@ -405,6 +705,161 @@ function checkRangeBoundOnly(startLine, endLine, resolvedPath) {
|
|
|
405
705
|
const lineCount = splitLines(read.content).length;
|
|
406
706
|
return checkRangeBound(startLine, endLine, lineCount);
|
|
407
707
|
}
|
|
708
|
+
function isTestFile(citedPath) {
|
|
709
|
+
return TEST_FILE_RE.test(citedPath);
|
|
710
|
+
}
|
|
711
|
+
/**
|
|
712
|
+
* True when the line at `lines[lineIndex]` is a fence delimiter
|
|
713
|
+
* (```` ``` ```` or `~~~`, optionally with a trailing language tag) that
|
|
714
|
+
* *opens* a fenced code block, as opposed to closing one -- determined by
|
|
715
|
+
* replaying the same open/close state machine `stripFencedCode`-style
|
|
716
|
+
* scanners use from the top of the document, not by the line's own text
|
|
717
|
+
* (a bare closing fence and an untagged opening fence are lexically
|
|
718
|
+
* identical). Used by checkRangeBoundary's markdown branch: citing a
|
|
719
|
+
* fenced block starting at its own opening fence line is the natural,
|
|
720
|
+
* correct way to cite it, so that specific case is exempted from the
|
|
721
|
+
* fence-as-drift-signal check (see there). Derived from `scanFenceLines`
|
|
722
|
+
* (see there); state at `lineIndex` never depends on any line after it, so
|
|
723
|
+
* scanning the whole array and reading one index back is equivalent to (and
|
|
724
|
+
* replaces) the original's own up-to-`lineIndex`-only replay.
|
|
725
|
+
*/
|
|
726
|
+
function isFenceOpeningLine(lines, lineIndex) {
|
|
727
|
+
return scanFenceLines(lines)[lineIndex]?.opensFence ?? false;
|
|
728
|
+
}
|
|
729
|
+
/**
|
|
730
|
+
* True when `lines[endLineIndex]` is the closing delimiter that matches
|
|
731
|
+
* the *opening* fence at `lines[startLineIndex]` (the caller must already
|
|
732
|
+
* have confirmed the start line is a genuine opening fence, via
|
|
733
|
+
* isFenceOpeningLine, before calling this) -- the first line after the
|
|
734
|
+
* opener whose trimmed text starts with the same fence marker. Used by
|
|
735
|
+
* checkRangeBoundary's markdown branch to also exempt a range's END from
|
|
736
|
+
* the fence-as-drift-signal check when the range legitimately cites a
|
|
737
|
+
* whole fenced code block from its own opening delimiter to its own
|
|
738
|
+
* closing delimiter: the natural, correct way to cite such a block, not
|
|
739
|
+
* drift, the same reasoning the start-side exception already applies.
|
|
740
|
+
*/
|
|
741
|
+
function isMatchingFenceClosingLine(lines, startLineIndex, endLineIndex) {
|
|
742
|
+
const fenceMarker = (lines[startLineIndex] ?? "").trim().slice(0, 3);
|
|
743
|
+
for (let i = startLineIndex + 1; i < lines.length; i++) {
|
|
744
|
+
const trimmed = (lines[i] ?? "").trim();
|
|
745
|
+
if (trimmed.startsWith(fenceMarker)) {
|
|
746
|
+
return i === endLineIndex;
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
return false;
|
|
750
|
+
}
|
|
751
|
+
/**
|
|
752
|
+
* Additional block-boundary check for a short-form citation's range (see
|
|
753
|
+
* the "Short-form citations" doc block below), layered on top of
|
|
754
|
+
* checkTarget's existing per-start-line checks via checkShortFormTarget.
|
|
755
|
+
* Deliberately scoped to short-form citations only, not wired into
|
|
756
|
+
* checkTarget/checkRangeBoundOnly (the full/continuation citation paths):
|
|
757
|
+
* applying it there too was tried first and rejected -- the real bundle
|
|
758
|
+
* this rule was built against has legitimate full citations into test
|
|
759
|
+
* files that cite a couple of arbitrary lines (e.g. two lines of a shared
|
|
760
|
+
* regex definition), not a describe/it block, and flagging those as newly
|
|
761
|
+
* broken would have regressed the existing 0-warning baseline. Short-form
|
|
762
|
+
* citations are the demonstrated motivating case (a paragraph naming a
|
|
763
|
+
* test file once, then citing several of its describe/it blocks by bare
|
|
764
|
+
* range alone), so the check is scoped to exactly that mechanism.
|
|
765
|
+
*
|
|
766
|
+
* Test-file targets (`.test.ts`/`.spec.ts`, and their `.js`/`.mjs`
|
|
767
|
+
* equivalents): the range must start on a `describe(`/`it(` head line and
|
|
768
|
+
* end on a matching closing `});` line. Severity split, by evidence
|
|
769
|
+
* strength: a wrong START line is a
|
|
770
|
+
* *warning* (`test-range-start-not-head`) -- the range beginning somewhere
|
|
771
|
+
* other than a block head is strong drift evidence. A range whose start IS
|
|
772
|
+
* correct but whose END is not the matching `});` is only a *notice*
|
|
773
|
+
* (`test-range-end-not-closing`): this also matches a deliberate partial
|
|
774
|
+
* citation (citing from a block's head to partway through it), which is not
|
|
775
|
+
* drift. The start is checked first and returned alone, matching
|
|
776
|
+
* checkTarget's existing single-problem-per-citation pattern (never both at
|
|
777
|
+
* once).
|
|
778
|
+
*
|
|
779
|
+
* Markdown targets: mechanical verification of "is this still the same
|
|
780
|
+
* block" is far less reliable for prose than for TS/JS brace structure, so
|
|
781
|
+
* this is a notice, not a warning (see this rule's own risk note on
|
|
782
|
+
* markdown false positives). The range's start or end line landing on a
|
|
783
|
+
* bare bracket (MD_BARE_BRACKET_RE) is always a heads-up that the boundary
|
|
784
|
+
* likely drifted onto structural punctuation rather than real prose. A bare
|
|
785
|
+
* code-fence delimiter (MD_FENCE_DELIM_RE) is also always a drift signal at
|
|
786
|
+
* either boundary, with two exceptions carved out for the one legitimate
|
|
787
|
+
* way to cite a whole fenced code block by its own delimiters: the START is
|
|
788
|
+
* exempted when that line is itself a genuine *opening* fence (see
|
|
789
|
+
* isFenceOpeningLine), and, only when the start already qualified for that
|
|
790
|
+
* exemption, the END is exempted when it is that same fence's matching
|
|
791
|
+
* *closing* delimiter (see isMatchingFenceClosingLine) -- citing a fenced
|
|
792
|
+
* block from its own opening delimiter through its own closing delimiter is
|
|
793
|
+
* the natural, correct way to cite it, not drift.
|
|
794
|
+
*/
|
|
795
|
+
function checkRangeBoundary(citedPath, startLine, endLine, lines) {
|
|
796
|
+
if (isTestFile(citedPath)) {
|
|
797
|
+
const startText = lines[startLine - 1] ?? "";
|
|
798
|
+
if (!TEST_HEAD_LINE_RE.test(startText)) {
|
|
799
|
+
return {
|
|
800
|
+
rule: "test-range-start-not-head",
|
|
801
|
+
message: `range start is not a "describe(" or "it(" head line ("${startText.trim()}")`,
|
|
802
|
+
};
|
|
803
|
+
}
|
|
804
|
+
const endText = lines[endLine - 1] ?? "";
|
|
805
|
+
if (!TEST_CLOSING_LINE_RE.test(endText)) {
|
|
806
|
+
return {
|
|
807
|
+
rule: "test-range-end-not-closing",
|
|
808
|
+
message: `range end is not a matching closing "});" line ("${endText.trim()}")`,
|
|
809
|
+
severity: "notice",
|
|
810
|
+
};
|
|
811
|
+
}
|
|
812
|
+
return null;
|
|
813
|
+
}
|
|
814
|
+
if (citedPath.toLowerCase().endsWith(".md")) {
|
|
815
|
+
const startTrim = (lines[startLine - 1] ?? "").trim();
|
|
816
|
+
const startIsFence = MD_FENCE_DELIM_RE.test(startTrim);
|
|
817
|
+
if (MD_BARE_BRACKET_RE.test(startTrim) ||
|
|
818
|
+
(startIsFence && !isFenceOpeningLine(lines, startLine - 1))) {
|
|
819
|
+
return {
|
|
820
|
+
rule: "markdown-range-boundary-bracket-or-fence",
|
|
821
|
+
message: `range start is a bare bracket/fence line ("${startTrim}")`,
|
|
822
|
+
severity: "notice",
|
|
823
|
+
};
|
|
824
|
+
}
|
|
825
|
+
const endTrim = (lines[endLine - 1] ?? "").trim();
|
|
826
|
+
const endIsFence = MD_FENCE_DELIM_RE.test(endTrim);
|
|
827
|
+
const endIsMatchingClose = startIsFence &&
|
|
828
|
+
endIsFence &&
|
|
829
|
+
isMatchingFenceClosingLine(lines, startLine - 1, endLine - 1);
|
|
830
|
+
if (MD_BARE_BRACKET_RE.test(endTrim) ||
|
|
831
|
+
(endIsFence && !endIsMatchingClose)) {
|
|
832
|
+
return {
|
|
833
|
+
rule: "markdown-range-boundary-bracket-or-fence",
|
|
834
|
+
message: `range end is a bare bracket/fence line ("${endTrim}")`,
|
|
835
|
+
severity: "notice",
|
|
836
|
+
};
|
|
837
|
+
}
|
|
838
|
+
return null;
|
|
839
|
+
}
|
|
840
|
+
return null;
|
|
841
|
+
}
|
|
842
|
+
/**
|
|
843
|
+
* A short-form citation's full check: checkTarget's existing checks
|
|
844
|
+
* (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
|
|
845
|
+
* closing-brace-start-line), plus, only when those all pass, the
|
|
846
|
+
* test-file/markdown block-boundary check above (see checkRangeBoundary).
|
|
847
|
+
* Reads the resolved target from disk exactly once (a dogfood bundle can
|
|
848
|
+
* run this against the same target file a dozen-plus times for one
|
|
849
|
+
* compound short-form list) and reuses the same `lines` array for both
|
|
850
|
+
* checks, instead of checkTarget and checkRangeBoundary each reading and
|
|
851
|
+
* re-splitting it independently.
|
|
852
|
+
*/
|
|
853
|
+
function checkShortFormTarget(citedPath, startLine, endLine, resolvedPath) {
|
|
854
|
+
const read = readTarget(resolvedPath);
|
|
855
|
+
if ("rule" in read)
|
|
856
|
+
return read;
|
|
857
|
+
const lines = splitLines(read.content);
|
|
858
|
+
const base = checkTargetLines(citedPath, startLine, endLine, lines);
|
|
859
|
+
if (base)
|
|
860
|
+
return base;
|
|
861
|
+
return checkRangeBoundary(citedPath, startLine, endLine, lines);
|
|
862
|
+
}
|
|
408
863
|
/**
|
|
409
864
|
* Collects every continuation-citation atom (see the "Continuation
|
|
410
865
|
* citations" doc block above) in `content`, sorted by document position.
|
|
@@ -462,10 +917,230 @@ function collectContinuationAtoms(content) {
|
|
|
462
917
|
}
|
|
463
918
|
return atoms;
|
|
464
919
|
}
|
|
465
|
-
|
|
920
|
+
/**
|
|
921
|
+
* True when the nearest non-whitespace text before `matchIndex` in
|
|
922
|
+
* `content` is a serial connective: the single character `,`, `;`, or `(`,
|
|
923
|
+
* or the word `and`/`or` (case-insensitive, word-boundary-matched so it
|
|
924
|
+
* does not fire on the tail of a longer word like "brand"). See the
|
|
925
|
+
* "Short-form citations" doc block above for why this is the gate.
|
|
926
|
+
*/
|
|
927
|
+
function isSerialConnectivePreceded(content, matchIndex) {
|
|
928
|
+
const before = content.slice(0, matchIndex).trimEnd();
|
|
929
|
+
if (before === "")
|
|
930
|
+
return false;
|
|
931
|
+
const lastChar = before[before.length - 1];
|
|
932
|
+
if (lastChar === "," || lastChar === ";" || lastChar === "(")
|
|
933
|
+
return true;
|
|
934
|
+
return /\b(?:and|or)$/i.test(before);
|
|
935
|
+
}
|
|
936
|
+
function isWithinAnySpan(index, spans) {
|
|
937
|
+
return spans.some(([start, end]) => index >= start && index < end);
|
|
938
|
+
}
|
|
939
|
+
function collectShortFormMatches(content, excludedSpans) {
|
|
940
|
+
const out = [];
|
|
941
|
+
const colonRe = new RegExp(SHORT_FORM_COLON_RE.source, "g");
|
|
942
|
+
let m;
|
|
943
|
+
while ((m = colonRe.exec(content)) !== null) {
|
|
944
|
+
if (isWithinAnySpan(m.index, excludedSpans))
|
|
945
|
+
continue;
|
|
946
|
+
if (content[m.index - 1] === "`")
|
|
947
|
+
continue;
|
|
948
|
+
if (content[m.index + m[0].length] === "`")
|
|
949
|
+
continue;
|
|
950
|
+
if (!isSerialConnectivePreceded(content, m.index))
|
|
951
|
+
continue;
|
|
952
|
+
const startLine = Number(m[1]);
|
|
953
|
+
const endLine = Number(m[2]);
|
|
954
|
+
out.push({ index: m.index, startLine, endLine });
|
|
955
|
+
}
|
|
956
|
+
return out.sort((a, b) => a.index - b.index);
|
|
957
|
+
}
|
|
958
|
+
/**
|
|
959
|
+
* Char spans of every fenced code block in `content` (```` ``` ```` or
|
|
960
|
+
* `~~~`, optionally with a trailing language tag), each span running from
|
|
961
|
+
* the start of the opening fence line to the end of the closing fence line
|
|
962
|
+
* inclusive. An unterminated fence (no matching close before end of doc) is
|
|
963
|
+
* treated as running to the end of the content -- conservative, since an
|
|
964
|
+
* unterminated fence is itself a doc problem outside this rule's scope, not
|
|
965
|
+
* a reason to scan its contents for short-form citations. Derived from
|
|
966
|
+
* `scanFenceLines` (see there): per-line fenced/opens/closes state is
|
|
967
|
+
* converted to char-offset spans by tracking each line's `[start, end)`
|
|
968
|
+
* offset in `content` alongside it.
|
|
969
|
+
*/
|
|
970
|
+
function computeFencedSpans(content) {
|
|
971
|
+
const spans = [];
|
|
972
|
+
const lines = content.split("\n");
|
|
973
|
+
const states = scanFenceLines(lines);
|
|
974
|
+
let offset = 0;
|
|
975
|
+
let spanStart = -1;
|
|
976
|
+
for (let i = 0; i < lines.length; i++) {
|
|
977
|
+
const lineEnd = offset + lines[i].length;
|
|
978
|
+
if (states[i].opensFence)
|
|
979
|
+
spanStart = offset;
|
|
980
|
+
if (states[i].closesFence && spanStart >= 0) {
|
|
981
|
+
spans.push([spanStart, lineEnd]);
|
|
982
|
+
spanStart = -1;
|
|
983
|
+
}
|
|
984
|
+
offset = lineEnd + 1; // +1 for the newline joining this line to the next
|
|
985
|
+
}
|
|
986
|
+
if (spanStart >= 0) {
|
|
987
|
+
spans.push([spanStart, content.length]);
|
|
988
|
+
}
|
|
989
|
+
return spans;
|
|
990
|
+
}
|
|
991
|
+
/**
|
|
992
|
+
* Char spans of every CommonMark-style indented code block in `content`: a
|
|
993
|
+
* maximal run of consecutive non-blank lines, each indented by at least
|
|
994
|
+
* four spaces or a leading tab, whose first line is preceded by a blank
|
|
995
|
+
* line or the start of the document (an indented code block cannot
|
|
996
|
+
* interrupt a paragraph). A blank line inside the run does not itself end
|
|
997
|
+
* it, matching CommonMark. Simplified relative to the full CommonMark
|
|
998
|
+
* spec (no list-item-context awareness); adequate for this mechanical,
|
|
999
|
+
* warn-only rule.
|
|
1000
|
+
*/
|
|
1001
|
+
function computeIndentedCodeSpans(content) {
|
|
1002
|
+
const spans = [];
|
|
1003
|
+
const lines = content.split("\n");
|
|
1004
|
+
let offset = 0;
|
|
1005
|
+
let blockStart = -1;
|
|
1006
|
+
let blockEnd = -1;
|
|
1007
|
+
let prevBlank = true;
|
|
1008
|
+
for (const line of lines) {
|
|
1009
|
+
const lineStart = offset;
|
|
1010
|
+
const lineEnd = offset + line.length;
|
|
1011
|
+
const isBlank = line.trim() === "";
|
|
1012
|
+
const isIndented = /^( {4,}|\t)/.test(line);
|
|
1013
|
+
if (!isBlank && isIndented && (blockStart !== -1 || prevBlank)) {
|
|
1014
|
+
if (blockStart === -1)
|
|
1015
|
+
blockStart = lineStart;
|
|
1016
|
+
blockEnd = lineEnd;
|
|
1017
|
+
}
|
|
1018
|
+
else if (isBlank && blockStart !== -1) {
|
|
1019
|
+
// blank line inside an open block: keep it open, don't extend blockEnd
|
|
1020
|
+
}
|
|
1021
|
+
else if (!isBlank) {
|
|
1022
|
+
if (blockStart !== -1)
|
|
1023
|
+
spans.push([blockStart, blockEnd]);
|
|
1024
|
+
blockStart = -1;
|
|
1025
|
+
blockEnd = -1;
|
|
1026
|
+
}
|
|
1027
|
+
prevBlank = isBlank;
|
|
1028
|
+
offset = lineEnd + 1;
|
|
1029
|
+
}
|
|
1030
|
+
if (blockStart !== -1)
|
|
1031
|
+
spans.push([blockStart, blockEnd]);
|
|
1032
|
+
return spans;
|
|
1033
|
+
}
|
|
1034
|
+
/**
|
|
1035
|
+
* Char spans of every inline code span in `content` (`` `...` ``, or a
|
|
1036
|
+
* longer run of backticks as the delimiter). This is a superset of the
|
|
1037
|
+
* narrower "immediately adjacent to a backtick" guard already applied in
|
|
1038
|
+
* collectShortFormMatches: that guard only rejects a match directly
|
|
1039
|
+
* touching a backtick, so a short form embedded further inside a longer
|
|
1040
|
+
* inline code span (e.g. `` `ports (1-3)` ``) was previously matched and
|
|
1041
|
+
* bound; this closes that gap.
|
|
1042
|
+
*
|
|
1043
|
+
* Confined to a single line (the pairing regex excludes `\n`): CommonMark
|
|
1044
|
+
* inline code spans cannot themselves span multiple lines in the sense
|
|
1045
|
+
* this rule cares about, but more importantly, an unmatched backtick run
|
|
1046
|
+
* (a typo, or a literal backtick in prose) previously paired greedily with
|
|
1047
|
+
* the *next* backtick run anywhere later in the whole document, silently
|
|
1048
|
+
* treating everything in between -- potentially several unrelated
|
|
1049
|
+
* sentences and any short-form citation among them -- as one giant inline
|
|
1050
|
+
* code span. Confining the match to a single line means a backtick run
|
|
1051
|
+
* with no partner on the same line produces no span at all, instead of
|
|
1052
|
+
* reaching across a newline for one.
|
|
1053
|
+
*/
|
|
1054
|
+
function computeInlineCodeSpans(content) {
|
|
1055
|
+
const spans = [];
|
|
1056
|
+
const re = /(`+)[^`\n]*?\1/g;
|
|
1057
|
+
let m;
|
|
1058
|
+
while ((m = re.exec(content)) !== null) {
|
|
1059
|
+
spans.push([m.index, m.index + m[0].length]);
|
|
1060
|
+
}
|
|
1061
|
+
return spans;
|
|
1062
|
+
}
|
|
1063
|
+
/**
|
|
1064
|
+
* Char spans of every Markdown table row in `content`: a line whose
|
|
1065
|
+
* trimmed form starts and ends with `|`. Decision (documented in the
|
|
1066
|
+
* README): a short-form citation inside a table cell is never recognised,
|
|
1067
|
+
* the same way one inside a code span is not -- excluded here rather than
|
|
1068
|
+
* left to the plausibility gate, since a table cell's content is prose-like
|
|
1069
|
+
* and can otherwise carry a range shape the gate would not reject (e.g.
|
|
1070
|
+
* `| col (5-9) |`).
|
|
1071
|
+
*/
|
|
1072
|
+
function computeTableRowSpans(content) {
|
|
1073
|
+
const spans = [];
|
|
1074
|
+
const lines = content.split("\n");
|
|
1075
|
+
let offset = 0;
|
|
1076
|
+
for (const line of lines) {
|
|
1077
|
+
const trimmed = line.trim();
|
|
1078
|
+
if (trimmed.startsWith("|") &&
|
|
1079
|
+
trimmed.endsWith("|") &&
|
|
1080
|
+
trimmed.length > 1) {
|
|
1081
|
+
spans.push([offset, offset + line.length]);
|
|
1082
|
+
}
|
|
1083
|
+
offset += line.length + 1;
|
|
1084
|
+
}
|
|
1085
|
+
return spans;
|
|
1086
|
+
}
|
|
1087
|
+
/**
|
|
1088
|
+
* All char spans short-form matching must never fire inside: fenced code,
|
|
1089
|
+
* indented code, inline code spans, and Markdown table rows. Computed once
|
|
1090
|
+
* per doc and combined with fullSpans (see scanDoc) via the existing
|
|
1091
|
+
* isWithinAnySpan helper -- the same mechanism a full citation's own span
|
|
1092
|
+
* already uses, not a second one.
|
|
1093
|
+
*/
|
|
1094
|
+
function computeExcludedSpans(content) {
|
|
1095
|
+
return [
|
|
1096
|
+
...computeFencedSpans(content),
|
|
1097
|
+
...computeIndentedCodeSpans(content),
|
|
1098
|
+
...computeInlineCodeSpans(content),
|
|
1099
|
+
...computeTableRowSpans(content),
|
|
1100
|
+
];
|
|
1101
|
+
}
|
|
1102
|
+
/**
|
|
1103
|
+
* Paragraph-start offsets in `content`, ascending, always including 0. A
|
|
1104
|
+
* paragraph boundary is a blank (empty or whitespace-only) line.
|
|
1105
|
+
*/
|
|
1106
|
+
function computeParagraphStarts(content) {
|
|
1107
|
+
const starts = [0];
|
|
1108
|
+
const re = /\n[ \t]*\n+/g;
|
|
1109
|
+
let m;
|
|
1110
|
+
while ((m = re.exec(content)) !== null) {
|
|
1111
|
+
starts.push(m.index + m[0].length);
|
|
1112
|
+
}
|
|
1113
|
+
return starts;
|
|
1114
|
+
}
|
|
1115
|
+
/** The start offset of the paragraph containing `index` (see computeParagraphStarts). */
|
|
1116
|
+
function paragraphStartFor(starts, index) {
|
|
1117
|
+
let result = starts[0];
|
|
1118
|
+
for (const s of starts) {
|
|
1119
|
+
if (s > index)
|
|
1120
|
+
break;
|
|
1121
|
+
result = s;
|
|
1122
|
+
}
|
|
1123
|
+
return result;
|
|
1124
|
+
}
|
|
1125
|
+
/**
|
|
1126
|
+
* The nearest full citation strictly before `beforeIndex` and at or after
|
|
1127
|
+
* `paragraphStart` -- "the last target document named earlier in the same
|
|
1128
|
+
* paragraph" -- or null when none exists.
|
|
1129
|
+
*/
|
|
1130
|
+
function findLastNamedTargetInParagraph(fullAtoms, paragraphStart, beforeIndex) {
|
|
1131
|
+
let best = null;
|
|
1132
|
+
for (const a of fullAtoms) {
|
|
1133
|
+
if (a.index >= paragraphStart && a.index < beforeIndex) {
|
|
1134
|
+
if (!best || a.index > best.index)
|
|
1135
|
+
best = a;
|
|
1136
|
+
}
|
|
1137
|
+
}
|
|
1138
|
+
return best;
|
|
1139
|
+
}
|
|
1140
|
+
function pushDrift(findings, file, citation, rule, message, resolvedTo, severity = "warning") {
|
|
466
1141
|
findings.push({
|
|
467
1142
|
ruleId: RULE_ID,
|
|
468
|
-
severity
|
|
1143
|
+
severity,
|
|
469
1144
|
file,
|
|
470
1145
|
message: `\`${citation}\`: ${message} [${rule}]`,
|
|
471
1146
|
...(resolvedTo ? { detail: `resolvedTo: ${resolvedTo}` } : {}),
|
|
@@ -508,24 +1183,86 @@ function isWrappedPathContinuation(content, matchIndex) {
|
|
|
508
1183
|
const prevLine = content.slice(prevLineStart, prevLineEnd);
|
|
509
1184
|
return /[-–]$/.test(prevLine);
|
|
510
1185
|
}
|
|
1186
|
+
/**
|
|
1187
|
+
* Raw text following a malformed anchor's `#` (see `anchor-malformed` in
|
|
1188
|
+
* the atom-processing loop below), used only to make that finding's message
|
|
1189
|
+
* concrete -- bounded so it reports roughly what was actually typed as the
|
|
1190
|
+
* failed anchor attempt, not an unrelated run of later prose:
|
|
1191
|
+
* - a quoted-anchor attempt (the character right after `#` is `"`) stops
|
|
1192
|
+
* at the next `"` on the same line (the closing quote the author
|
|
1193
|
+
* presumably meant, embedded backticks and all -- that quote is exactly
|
|
1194
|
+
* what makes this a *quoted* anchor attempt rather than a heading one),
|
|
1195
|
+
* or at the end of the line when no such quote exists on it (matches
|
|
1196
|
+
* `CITATION_RE`'s own string alternative, which cannot cross a
|
|
1197
|
+
* newline either);
|
|
1198
|
+
* - any other character after `#` is a heading-anchor attempt, which
|
|
1199
|
+
* stops at the first whitespace (a heading anchor token, like
|
|
1200
|
+
* `CITATION_RE`'s own heading alternative, never contains whitespace)
|
|
1201
|
+
* or the end of the line, whichever comes first;
|
|
1202
|
+
* - either way, capped at `MAX_RAW_LEN` characters so a heading-form
|
|
1203
|
+
* attempt with no whitespace at all before the line ends (or a
|
|
1204
|
+
* pathological single long line) cannot make the message unbounded.
|
|
1205
|
+
*/
|
|
1206
|
+
const MAX_MALFORMED_ANCHOR_RAW_LEN = 60;
|
|
1207
|
+
function extractMalformedAnchorRaw(content, hashIndex) {
|
|
1208
|
+
const from = hashIndex + 1;
|
|
1209
|
+
const nl = content.indexOf("\n", from);
|
|
1210
|
+
const lineEnd = nl === -1 ? content.length : nl;
|
|
1211
|
+
let end;
|
|
1212
|
+
if (content[from] === '"') {
|
|
1213
|
+
const q = content.indexOf('"', from + 1);
|
|
1214
|
+
end = q !== -1 && q < lineEnd ? q + 1 : lineEnd;
|
|
1215
|
+
}
|
|
1216
|
+
else {
|
|
1217
|
+
const rest = content.slice(from, lineEnd);
|
|
1218
|
+
const ws = rest.search(/\s/);
|
|
1219
|
+
end = ws === -1 ? lineEnd : from + ws;
|
|
1220
|
+
}
|
|
1221
|
+
if (end - from > MAX_MALFORMED_ANCHOR_RAW_LEN) {
|
|
1222
|
+
end = from + MAX_MALFORMED_ANCHOR_RAW_LEN;
|
|
1223
|
+
}
|
|
1224
|
+
return content.slice(from, end);
|
|
1225
|
+
}
|
|
511
1226
|
function scanDoc(cache, root, bundleDir, doc) {
|
|
512
1227
|
const findings = [];
|
|
513
1228
|
const content = doc.raw;
|
|
514
1229
|
const sources = getValidSources(doc.frontmatter.parsed) ?? [];
|
|
515
1230
|
const docAbsPath = path.join(bundleDir, doc.relPath);
|
|
516
1231
|
const fullAtoms = [];
|
|
1232
|
+
// Char spans of every matched full citation, used to keep short-form
|
|
1233
|
+
// matching (see collectShortFormMatches) from re-matching the tail of a
|
|
1234
|
+
// real `path:N-M` citation as a bare short form.
|
|
1235
|
+
const fullSpans = [];
|
|
517
1236
|
const re = new RegExp(CITATION_RE.source, "g");
|
|
518
1237
|
let m;
|
|
519
1238
|
while ((m = re.exec(content)) !== null) {
|
|
520
1239
|
if (isWrappedPathContinuation(content, m.index))
|
|
521
1240
|
continue;
|
|
1241
|
+
const matchEnd = m.index + m[0].length;
|
|
1242
|
+
// anchor-malformed detection (see the atom-processing loop below for
|
|
1243
|
+
// where the finding is actually pushed): a `#` immediately follows the
|
|
1244
|
+
// range but group 4 (the anchor) did not match -- unbalanced quotes, a
|
|
1245
|
+
// backtick inside a quoted anchor, or nothing at all after the `#`
|
|
1246
|
+
// (e.g. `path:N-M#` at end of line). Computed here, at match time,
|
|
1247
|
+
// because it needs `content`/`matchEnd`; carried on the atom rather
|
|
1248
|
+
// than pushed immediately so the citation's out-of-scope/resolution
|
|
1249
|
+
// posture (path-traversal-rejected, skip, missing-file, ambiguous) can
|
|
1250
|
+
// gate it exactly the same way every other check on this atom already
|
|
1251
|
+
// is -- an out-of-scope or unresolved citation gets none of those
|
|
1252
|
+
// checks either.
|
|
1253
|
+
const malformedAnchorRaw = m[4] === undefined && content[matchEnd] === "#"
|
|
1254
|
+
? extractMalformedAnchorRaw(content, matchEnd)
|
|
1255
|
+
: null;
|
|
522
1256
|
fullAtoms.push({
|
|
523
1257
|
kind: "full",
|
|
524
1258
|
index: m.index,
|
|
525
1259
|
citedPath: m[1],
|
|
526
1260
|
startLine: Number(m[2]),
|
|
527
1261
|
endLine: m[3] ? Number(m[3]) : null,
|
|
1262
|
+
anchor: parseAnchor(m[4]),
|
|
1263
|
+
malformedAnchorRaw,
|
|
528
1264
|
});
|
|
1265
|
+
fullSpans.push([m.index, matchEnd]);
|
|
529
1266
|
}
|
|
530
1267
|
const atoms = [...fullAtoms, ...collectContinuationAtoms(content)].sort((a, b) => a.index - b.index);
|
|
531
1268
|
// `governing`: nearest preceding citation (full or continuation) that
|
|
@@ -564,8 +1301,14 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
564
1301
|
lastStartLine = startLine;
|
|
565
1302
|
continue; // governing (same file) carries over unchanged
|
|
566
1303
|
}
|
|
567
|
-
const { citedPath, startLine, endLine } = atom;
|
|
568
|
-
|
|
1304
|
+
const { citedPath, startLine, endLine, anchor } = atom;
|
|
1305
|
+
// The anchor, when present, is carried in the citation label itself
|
|
1306
|
+
// (not just the anchor-check finding's own message) so two citations to
|
|
1307
|
+
// the same range with different anchors are distinguishable in the
|
|
1308
|
+
// output -- see formatAnchorForLabel and the "Anchored citations" doc
|
|
1309
|
+
// block above. Continuations and short-form citations never carry an
|
|
1310
|
+
// anchor (see there), so their own citation labels are unaffected.
|
|
1311
|
+
const citation = `${citedPath}:${startLine}${endLine ? "-" + endLine : ""}${anchor ? formatAnchorForLabel(anchor) : ""}`;
|
|
569
1312
|
if (hasParentSegment(citedPath)) {
|
|
570
1313
|
pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
|
|
571
1314
|
governing = null;
|
|
@@ -590,7 +1333,19 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
590
1333
|
lastStartLine = null;
|
|
591
1334
|
continue;
|
|
592
1335
|
}
|
|
593
|
-
|
|
1336
|
+
// anchor-malformed (notice): only reached for a citation that resolved
|
|
1337
|
+
// to a real file -- see `malformedAnchorRaw`'s doc comment above for
|
|
1338
|
+
// why this is gated the same way missing-file/skip/path-traversal are
|
|
1339
|
+
// already gated for every other check on this atom. The citation is
|
|
1340
|
+
// still checked below via checkFullTarget exactly as an ordinary
|
|
1341
|
+
// anchorless citation would be (anchor is null here by construction --
|
|
1342
|
+
// see parseAnchor); this only ADDS a heads-up that the `#` sitting
|
|
1343
|
+
// right there was silently not read as the anchor it looks like it was
|
|
1344
|
+
// meant to be.
|
|
1345
|
+
if (atom.malformedAnchorRaw !== null) {
|
|
1346
|
+
pushDrift(findings, doc.relPath, citation, "anchor-malformed", `a "#" follows the citation's range but does not parse as a heading or string anchor (raw: "${atom.malformedAnchorRaw}")`, undefined, "notice");
|
|
1347
|
+
}
|
|
1348
|
+
const problem = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor);
|
|
594
1349
|
if (problem?.rule === "unreadable-target") {
|
|
595
1350
|
pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
|
|
596
1351
|
}
|
|
@@ -600,6 +1355,70 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
600
1355
|
governing = { citedPath, resolvedPath: resolution.path };
|
|
601
1356
|
lastStartLine = startLine;
|
|
602
1357
|
}
|
|
1358
|
+
// Short-form (paragraph-bound) citations -- see that doc block above.
|
|
1359
|
+
// Deliberately independent of `governing`/`lastStartLine`: short-form
|
|
1360
|
+
// binding is paragraph-scoped by design, not chained through the
|
|
1361
|
+
// document-wide continuation state machine above. Reserved files
|
|
1362
|
+
// (index.md, log.md, see doc.isReserved) are skipped entirely: they are
|
|
1363
|
+
// append-only narrative journals, not `sources:`-driven reference docs,
|
|
1364
|
+
// and routinely narrate historical "old N-M -> new X-Y" line-number
|
|
1365
|
+
// deltas as prose data about past changes -- not live citations against
|
|
1366
|
+
// current content -- which this rule's bare-range matching cannot tell
|
|
1367
|
+
// apart from a real short-form citation. Full/continuation citations in
|
|
1368
|
+
// reserved files are still scanned as before; this carve-out is scoped
|
|
1369
|
+
// to short-form matching only.
|
|
1370
|
+
const shortFormMatches = doc.isReserved
|
|
1371
|
+
? []
|
|
1372
|
+
: collectShortFormMatches(content, [
|
|
1373
|
+
...fullSpans,
|
|
1374
|
+
...computeExcludedSpans(content),
|
|
1375
|
+
]);
|
|
1376
|
+
if (shortFormMatches.length > 0) {
|
|
1377
|
+
const paragraphStarts = computeParagraphStarts(content);
|
|
1378
|
+
const namedFullAtoms = [];
|
|
1379
|
+
for (const a of fullAtoms) {
|
|
1380
|
+
if (a.kind === "full") {
|
|
1381
|
+
namedFullAtoms.push({
|
|
1382
|
+
index: a.index,
|
|
1383
|
+
citedPath: a.citedPath,
|
|
1384
|
+
});
|
|
1385
|
+
}
|
|
1386
|
+
}
|
|
1387
|
+
for (const sf of shortFormMatches) {
|
|
1388
|
+
const paragraphStart = paragraphStartFor(paragraphStarts, sf.index);
|
|
1389
|
+
const target = findLastNamedTargetInParagraph(namedFullAtoms, paragraphStart, sf.index);
|
|
1390
|
+
const rangeLabel = `${sf.startLine}-${sf.endLine}`;
|
|
1391
|
+
if (!target) {
|
|
1392
|
+
pushDrift(findings, doc.relPath, `${rangeLabel} (short-form)`, "short-form-unbound", "no full `path:N` citation earlier in this paragraph to bind to", undefined, "notice");
|
|
1393
|
+
continue;
|
|
1394
|
+
}
|
|
1395
|
+
const targetPath = target.citedPath;
|
|
1396
|
+
const citation = `${targetPath}:${rangeLabel} (short-form)`;
|
|
1397
|
+
if (hasParentSegment(targetPath)) {
|
|
1398
|
+
// The full citation that named this target already reported
|
|
1399
|
+
// path-traversal-rejected for itself; not re-flagged a second time.
|
|
1400
|
+
continue;
|
|
1401
|
+
}
|
|
1402
|
+
const resolution = resolveCitation(cache, root, docAbsPath, content, sources, targetPath, sf.index);
|
|
1403
|
+
if (!resolution) {
|
|
1404
|
+
pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${targetPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
|
|
1405
|
+
continue;
|
|
1406
|
+
}
|
|
1407
|
+
if ("skip" in resolution)
|
|
1408
|
+
continue;
|
|
1409
|
+
if ("ambiguous" in resolution) {
|
|
1410
|
+
pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
|
|
1411
|
+
continue;
|
|
1412
|
+
}
|
|
1413
|
+
const problem = checkShortFormTarget(targetPath, sf.startLine, sf.endLine, resolution.path);
|
|
1414
|
+
if (problem?.rule === "unreadable-target") {
|
|
1415
|
+
pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
|
|
1416
|
+
}
|
|
1417
|
+
else if (problem) {
|
|
1418
|
+
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path), problem.severity);
|
|
1419
|
+
}
|
|
1420
|
+
}
|
|
1421
|
+
}
|
|
603
1422
|
return findings;
|
|
604
1423
|
}
|
|
605
1424
|
export const citationsResolveRule = {
|