okf-kit 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +112 -0
- package/README.md +28 -4
- package/dist/cli.d.ts +14 -0
- package/dist/cli.js +9 -0
- package/dist/cli.js.map +1 -1
- package/dist/rules/citations-resolve.js +754 -81
- package/dist/rules/citations-resolve.js.map +1 -1
- package/dist/types.d.ts +17 -0
- package/package.json +1 -1
|
@@ -173,7 +173,21 @@ const RULE_ID = "citations-resolve";
|
|
|
173
173
|
* checks (blank-start-line, closing-brace-start-line, inverted-range,
|
|
174
174
|
* range-exceeds-file) already came back clean -- same "one problem per
|
|
175
175
|
* citation, base checks first" pattern `checkShortFormTarget` already uses
|
|
176
|
-
* for the block-boundary check.
|
|
176
|
+
* for the block-boundary check. Every one of these four messages names the
|
|
177
|
+
* anchor text itself, and an anchored full citation's own finding label
|
|
178
|
+
* (`path:N-M#anchor`, see `formatAnchorForLabel`) carries the anchor too --
|
|
179
|
+
* without both, two citations to the identical range with different
|
|
180
|
+
* anchors would be indistinguishable in the output.
|
|
181
|
+
*
|
|
182
|
+
* A `#` that immediately follows a citation's range but does not parse as
|
|
183
|
+
* either anchor form at all (unbalanced quotes, a backtick inside a quoted
|
|
184
|
+
* anchor, or nothing after the `#`) is its own separate, `notice`-severity
|
|
185
|
+
* finding, `anchor-malformed` (see `scanDoc`'s main matching loop): the
|
|
186
|
+
* citation is still checked exactly as an anchorless one would be
|
|
187
|
+
* (backward compatible, see below), but silently checking it anchorless
|
|
188
|
+
* when the `#` right there looks like a typo'd anchor attempt would defeat
|
|
189
|
+
* the entire point of writing one -- a single misplaced character would
|
|
190
|
+
* quietly turn the very check the anchor was written for back off.
|
|
177
191
|
*
|
|
178
192
|
* Backward compatible by construction: the `#anchor` suffix is optional in
|
|
179
193
|
* `CITATION_RE`, so an existing anchorless citation matches exactly as
|
|
@@ -202,8 +216,112 @@ const RULE_ID = "citations-resolve";
|
|
|
202
216
|
* plus prose) to stay in sync, the exact class of drift this rule
|
|
203
217
|
* exists to catch.
|
|
204
218
|
*/
|
|
219
|
+
/**
|
|
220
|
+
* Anchor strictness (opt-in, `--require-anchors`, see `src/cli.ts`).
|
|
221
|
+
* Backward compatible by construction: every check in this section is
|
|
222
|
+
* gated on `ctx.requireAnchors` being present (`RequireAnchorsOptions`, see
|
|
223
|
+
* `src/types.ts`), so a consumer that never passes `--require-anchors`
|
|
224
|
+
* sees byte-identical findings to before this section existed. Three new
|
|
225
|
+
* rule ids, all `warning`-severity, mirroring this rule's own "a wrong
|
|
226
|
+
* start/target is strong drift evidence" posture rather than the `notice`
|
|
227
|
+
* posture reserved for the mechanical prose checks with a real
|
|
228
|
+
* false-positive risk (`markdown-range-boundary-bracket-or-fence` and
|
|
229
|
+
* friends):
|
|
230
|
+
*
|
|
231
|
+
* - `anchor-required`: an in-repo full citation (`path:N`/`path:N-M`,
|
|
232
|
+
* resolved to a real target -- an unresolved citation already gets its
|
|
233
|
+
* own `missing-file`/`unresolved-ambiguous` finding, not this one)
|
|
234
|
+
* carrying no `#anchor` at all. Exempted the same way this rule already
|
|
235
|
+
* exempts short-form matching for a reserved citing doc
|
|
236
|
+
* (`doc.isReserved`, e.g. `log.md`'s append-only line-number narration),
|
|
237
|
+
* and additionally by `requireAnchors.allow`: a citedPath matching one
|
|
238
|
+
* of its glob/exact patterns (matched against the citation's raw,
|
|
239
|
+
* as-written `citedPath` text -- see `matchesAllowPattern`) is exempt,
|
|
240
|
+
* for a doc category this check is not (yet) meant to cover, e.g. a
|
|
241
|
+
* README or install guide whose prose is not line-anchored the way a
|
|
242
|
+
* CHANGELOG or source citation is.
|
|
243
|
+
* - `anchor-not-on-last-line`: layered on `checkAnchor`'s existing string
|
|
244
|
+
* branch. An anchor sitting on the FIRST line of a wide range survives a
|
|
245
|
+
* k-line insertion above the range whenever k is smaller than the range
|
|
246
|
+
* itself, because the shifted window (the citation's own line numbers,
|
|
247
|
+
* re-read after the insertion) still contains the original first
|
|
248
|
+
* line's content, just at a different offset inside the window.
|
|
249
|
+
* Anchoring on the LAST line instead closes that: the original last
|
|
250
|
+
* line falls out of the shifted window on any insertion size >= 1, not
|
|
251
|
+
* only large ones.
|
|
252
|
+
* - `anchor-not-unique-in-range`: also layered on the string branch. An
|
|
253
|
+
* anchor text occurring more than once inside its own cited range is
|
|
254
|
+
* ambiguous evidence -- which occurrence is the one actually pinning
|
|
255
|
+
* the citation? A count of zero is unaffected (already
|
|
256
|
+
* `anchor-not-found-in-range`, unconditionally, not double-reported
|
|
257
|
+
* here).
|
|
258
|
+
*
|
|
259
|
+
* Both of the last two are checked only once a match exists at all (count
|
|
260
|
+
* >= 1); uniqueness is checked before last-line placement, since a
|
|
261
|
+
* non-unique anchor is the more fundamental problem (which of the several
|
|
262
|
+
* matching lines "is" the anchor is undefined before asking whether the
|
|
263
|
+
* one occurrence is on the right line).
|
|
264
|
+
*
|
|
265
|
+
* A fourth check, `test-range-straddles-block` (warning), applies to a FULL
|
|
266
|
+
* citation's own range into a `.test.`/`.spec.` target (`.ts`, `.js`,
|
|
267
|
+
* `.mjs`) under this same opt-in (see `checkTestRangeStraddle`, called
|
|
268
|
+
* from `checkFullTarget`): a
|
|
269
|
+
* full citation's range is expected to stay inside a single `describe`/
|
|
270
|
+
* `it`/`test` block once it starts, so any block-head line
|
|
271
|
+
* (`describe(`/`describe.only(`/`describe.skip(`/`describe.each(`/`it(`/
|
|
272
|
+
* `it.only(`/`it.skip(`/`it.each(`/`test(`/`test.only(`/`test.skip(`/
|
|
273
|
+
* `test.each(`, see `TEST_BLOCK_HEAD_RE`) found on any line of the range
|
|
274
|
+
* OTHER than its own first line means the citation ran into a sibling or
|
|
275
|
+
* outer block (or never really started on one). A range that leaves its
|
|
276
|
+
* block without a later head line inside it (ending on an outer block's
|
|
277
|
+
* closing line) is not detected by this line-based check. The range's start
|
|
278
|
+
* line is never itself checked here: it is either a legitimate block head
|
|
279
|
+
* (the range correctly starts a block) or legitimately inside a block's
|
|
280
|
+
* body (a partial citation into the middle of a block), and both are fine
|
|
281
|
+
* -- only a head line reappearing *after* the start is evidence of drift.
|
|
282
|
+
* Deliberately its own rule id rather than reusing `checkRangeBoundary`'s
|
|
283
|
+
* `test-range-start-not-head`/`test-range-end-not-closing` (which stay
|
|
284
|
+
* short-form-only, see `checkShortFormTarget`): those two require the range
|
|
285
|
+
* to start exactly on a head line and end exactly on the matching closing
|
|
286
|
+
* `});`, which is far stricter than a full citation's legitimate use
|
|
287
|
+
* (citing an arbitrary sub-range once a paragraph has already named the
|
|
288
|
+
* file) needs -- the straddle check only cares whether the range crossed a
|
|
289
|
+
* block boundary, not whether it captured a whole block precisely.
|
|
290
|
+
* Deliberately scoped to test-file targets only, same reasoning
|
|
291
|
+
* `checkShortFormTarget`'s own doc comment already gives for the Markdown
|
|
292
|
+
* half of the block-boundary check: a full citation into a Markdown target
|
|
293
|
+
* legitimately cites a couple of arbitrary lines all the time.
|
|
294
|
+
*/
|
|
295
|
+
/**
|
|
296
|
+
* True when `citedPath` (the citation's raw, as-written path text) matches
|
|
297
|
+
* one of `patterns`: an exact string match, or a glob with `*` (any run of
|
|
298
|
+
* characters) and `?` (any single character) as the only wildcards -- the
|
|
299
|
+
* minimal shape needed for `requireAnchors.allow`'s own documented example
|
|
300
|
+
* (`["README.md", "INSTALL-AGENT.md"]`), not a general-purpose glob engine.
|
|
301
|
+
*/
|
|
302
|
+
function matchesAllowPattern(citedPath, patterns) {
|
|
303
|
+
return patterns.some((pattern) => {
|
|
304
|
+
if (!pattern.includes("*") && !pattern.includes("?")) {
|
|
305
|
+
return pattern === citedPath;
|
|
306
|
+
}
|
|
307
|
+
const escaped = pattern.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
308
|
+
const reSource = "^" + escaped.replace(/\*/g, ".*").replace(/\?/g, ".") + "$";
|
|
309
|
+
return new RegExp(reSource).test(citedPath);
|
|
310
|
+
});
|
|
311
|
+
}
|
|
205
312
|
const ANCHOR_HEADING_MAX_LEVEL = 2;
|
|
206
313
|
const MD_HEADING_RE = /^(#{1,6})\s+(.*)$/;
|
|
314
|
+
/**
|
|
315
|
+
* Renders a parsed `Anchor` back to the short form used to disambiguate a
|
|
316
|
+
* finding's citation label (see the `citation` variable in `scanDoc`'s main
|
|
317
|
+
* loop): `#text` for a heading anchor (already bracket-stripped by
|
|
318
|
+
* `parseAnchor`), `#"text"` for a string anchor. Two citations to the same
|
|
319
|
+
* range with different anchors would otherwise be indistinguishable in the
|
|
320
|
+
* output -- see the "Anchored citations" doc block above.
|
|
321
|
+
*/
|
|
322
|
+
function formatAnchorForLabel(anchor) {
|
|
323
|
+
return anchor.kind === "string" ? `#"${anchor.text}"` : `#${anchor.text}`;
|
|
324
|
+
}
|
|
207
325
|
/**
|
|
208
326
|
* Parses `CITATION_RE`'s optional 4th capture group (the raw anchor text,
|
|
209
327
|
* including its surrounding quotes or brackets if any) into an `Anchor`, or
|
|
@@ -225,13 +343,53 @@ function parseAnchor(raw) {
|
|
|
225
343
|
const text = raw.startsWith("[") && raw.endsWith("]") ? raw.slice(1, -1) : raw;
|
|
226
344
|
return { kind: "heading", text };
|
|
227
345
|
}
|
|
346
|
+
/**
|
|
347
|
+
* The single fence state machine every fence-aware consumer in this file
|
|
348
|
+
* derives from: one forward pass over `lines`, replaying the same
|
|
349
|
+
* open/close logic (a line matching `MD_FENCE_DELIM_RE` opens a fence when
|
|
350
|
+
* none is open, a line starting with the *same* marker closes it) that
|
|
351
|
+
* `computeFencedSpans`, `computeFencedLineIndices`, and `isFenceOpeningLine`
|
|
352
|
+
* each used to implement as their own, independently-maintained copy --
|
|
353
|
+
* nothing enforced the three staying in agreement with each other. An
|
|
354
|
+
* unterminated fence (no matching close before the end of `lines`) is
|
|
355
|
+
* reflected by every remaining line coming back `fenced: true` -- each line
|
|
356
|
+
* is marked while the scan is still inside the open fence, so no separate
|
|
357
|
+
* post-loop fixup is needed the way the three original copies each had.
|
|
358
|
+
* State at line `i` never depends on any line after `i`, so a caller that
|
|
359
|
+
* only needs one line's state (see `isFenceOpeningLine`) can safely ignore
|
|
360
|
+
* the rest of the array without re-deriving the logic itself.
|
|
361
|
+
*/
|
|
362
|
+
function scanFenceLines(lines) {
|
|
363
|
+
const states = [];
|
|
364
|
+
let fenceMarker;
|
|
365
|
+
for (let i = 0; i < lines.length; i++) {
|
|
366
|
+
const trimmed = (lines[i] ?? "").trim();
|
|
367
|
+
if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
368
|
+
fenceMarker = trimmed.slice(0, 3);
|
|
369
|
+
states.push({ fenced: true, opensFence: true, closesFence: false });
|
|
370
|
+
}
|
|
371
|
+
else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
372
|
+
states.push({ fenced: true, opensFence: false, closesFence: true });
|
|
373
|
+
fenceMarker = undefined;
|
|
374
|
+
}
|
|
375
|
+
else {
|
|
376
|
+
states.push({
|
|
377
|
+
fenced: fenceMarker !== undefined,
|
|
378
|
+
opensFence: false,
|
|
379
|
+
closesFence: false,
|
|
380
|
+
});
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
return states;
|
|
384
|
+
}
|
|
228
385
|
/**
|
|
229
386
|
* 0-based line indices that fall inside a fenced code block (```` ``` ````
|
|
230
387
|
* or `~~~`, optionally with a trailing language tag), delimiters included --
|
|
231
388
|
* the target-side twin of `computeFencedSpans` above, which does the same
|
|
232
|
-
* job for the *citing* doc's short-form matching.
|
|
233
|
-
*
|
|
234
|
-
*
|
|
389
|
+
* job for the *citing* doc's short-form matching. Derived from
|
|
390
|
+
* `scanFenceLines` (see there). Anchor heading-search needs its own copy
|
|
391
|
+
* because it works from an already-split `lines` array (see
|
|
392
|
+
* `checkFullTarget`), not the raw `content` string `computeFencedSpans`
|
|
235
393
|
* takes, and because it must ignore a target's `# not a heading` sitting
|
|
236
394
|
* inside a fenced example exactly the same way a citing doc's own fences
|
|
237
395
|
* are already ignored for short-form matching -- without this, a `#`-led
|
|
@@ -243,25 +401,10 @@ function parseAnchor(raw) {
|
|
|
243
401
|
*/
|
|
244
402
|
function computeFencedLineIndices(lines) {
|
|
245
403
|
const fenced = new Set();
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
if (!fenceMarker && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
251
|
-
fenceMarker = trimmed.slice(0, 3);
|
|
252
|
-
fenceStart = i;
|
|
253
|
-
}
|
|
254
|
-
else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
255
|
-
for (let j = fenceStart; j <= i; j++)
|
|
256
|
-
fenced.add(j);
|
|
257
|
-
fenceMarker = undefined;
|
|
258
|
-
fenceStart = -1;
|
|
259
|
-
}
|
|
260
|
-
}
|
|
261
|
-
if (fenceMarker && fenceStart >= 0) {
|
|
262
|
-
for (let j = fenceStart; j < lines.length; j++)
|
|
263
|
-
fenced.add(j);
|
|
264
|
-
}
|
|
404
|
+
scanFenceLines(lines).forEach((state, i) => {
|
|
405
|
+
if (state.fenced)
|
|
406
|
+
fenced.add(i);
|
|
407
|
+
});
|
|
265
408
|
return fenced;
|
|
266
409
|
}
|
|
267
410
|
/**
|
|
@@ -285,29 +428,80 @@ function findEnclosingHeading(lines, startLine, fencedLines) {
|
|
|
285
428
|
}
|
|
286
429
|
return null;
|
|
287
430
|
}
|
|
431
|
+
// A trimmed line composed of nothing but closing brackets/braces/parens
|
|
432
|
+
// plus an optional trailing `,`/`;` -- bare closing boilerplate like `}`,
|
|
433
|
+
// `});`, `]);`, `}),`. Used only by `lastContentLineInRange` (opt-in
|
|
434
|
+
// `anchor-not-on-last-line`, see there): a range that ends on such a line
|
|
435
|
+
// (the common shape of a `});` that closes a whole cited `describe`/`it`
|
|
436
|
+
// block) should not force the anchor onto that boilerplate line itself --
|
|
437
|
+
// the last CONTENT line before it is the meaningful place for an anchor
|
|
438
|
+
// to live.
|
|
439
|
+
const CLOSING_BOILERPLATE_RE = /^[\]\)\};,]*$/;
|
|
440
|
+
function isContentLine(text) {
|
|
441
|
+
const trimmed = text.trim();
|
|
442
|
+
return trimmed !== "" && !CLOSING_BOILERPLATE_RE.test(trimmed);
|
|
443
|
+
}
|
|
444
|
+
/**
|
|
445
|
+
* The last line number (1-based, within `[startLine, endLine]`) whose text
|
|
446
|
+
* is a "content" line -- see `isContentLine`. Walks backward from `endLine`
|
|
447
|
+
* so a range ending on one or more bare closing-boilerplate lines (`});`,
|
|
448
|
+
* `]);`, `}),`, a lone `}`) resolves to the real content line before them
|
|
449
|
+
* instead of the boilerplate itself. Falls back to `endLine` unchanged when
|
|
450
|
+
* every line in the range is boilerplate or blank (nothing better to
|
|
451
|
+
* anchor against).
|
|
452
|
+
*/
|
|
453
|
+
function lastContentLineInRange(lines, startLine, endLine) {
|
|
454
|
+
for (let ln = endLine; ln >= startLine; ln--) {
|
|
455
|
+
if (isContentLine(lines[ln - 1] ?? ""))
|
|
456
|
+
return ln;
|
|
457
|
+
}
|
|
458
|
+
return endLine;
|
|
459
|
+
}
|
|
288
460
|
/**
|
|
289
461
|
* Checks an anchored full citation's anchor against its already-resolved,
|
|
290
462
|
* already-range-checked target -- see the "Anchored citations" doc block
|
|
291
463
|
* above for the two forms' semantics. `endLine` is the citation's end line,
|
|
292
464
|
* or its start line for a single-line citation (a size-1 range).
|
|
293
465
|
*/
|
|
294
|
-
function checkAnchor(anchor, startLine, endLine, lines) {
|
|
466
|
+
function checkAnchor(anchor, startLine, endLine, lines, requireAnchors) {
|
|
295
467
|
if (anchor.kind === "string") {
|
|
468
|
+
let count = 0;
|
|
469
|
+
let lastMatchLine = -1;
|
|
296
470
|
for (let i = startLine - 1; i <= endLine - 1 && i < lines.length; i++) {
|
|
297
|
-
if ((lines[i] ?? "").includes(anchor.text))
|
|
298
|
-
|
|
471
|
+
if ((lines[i] ?? "").includes(anchor.text)) {
|
|
472
|
+
count++;
|
|
473
|
+
lastMatchLine = i + 1;
|
|
474
|
+
}
|
|
299
475
|
}
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
476
|
+
if (count === 0) {
|
|
477
|
+
return {
|
|
478
|
+
rule: "anchor-not-found-in-range",
|
|
479
|
+
message: `anchor "${anchor.text}" does not occur in the cited range (${startLine}-${endLine})`,
|
|
480
|
+
};
|
|
481
|
+
}
|
|
482
|
+
if (requireAnchors) {
|
|
483
|
+
if (count > 1) {
|
|
484
|
+
return {
|
|
485
|
+
rule: "anchor-not-unique-in-range",
|
|
486
|
+
message: `anchor "${anchor.text}" occurs on ${count} lines of the cited range (${startLine}-${endLine}); expected exactly one`,
|
|
487
|
+
};
|
|
488
|
+
}
|
|
489
|
+
const lastContentLine = lastContentLineInRange(lines, startLine, endLine);
|
|
490
|
+
if (lastMatchLine !== lastContentLine) {
|
|
491
|
+
return {
|
|
492
|
+
rule: "anchor-not-on-last-line",
|
|
493
|
+
message: `anchor "${anchor.text}" occurs on line ${lastMatchLine}, not the cited range's last content line (${lastContentLine})`,
|
|
494
|
+
};
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
return null;
|
|
304
498
|
}
|
|
305
499
|
const fencedLines = computeFencedLineIndices(lines);
|
|
306
500
|
const heading = findEnclosingHeading(lines, startLine, fencedLines);
|
|
307
501
|
if (!heading) {
|
|
308
502
|
return {
|
|
309
503
|
rule: "anchor-heading-not-found",
|
|
310
|
-
message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor against; use a string anchor (#"...") instead against a target with no heading structure`,
|
|
504
|
+
message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) precedes line ${startLine} to anchor "${anchor.text}" against; use a string anchor (#"...") instead against a target with no heading structure`,
|
|
311
505
|
};
|
|
312
506
|
}
|
|
313
507
|
if (!heading.text.includes(anchor.text)) {
|
|
@@ -323,13 +517,93 @@ function checkAnchor(anchor, startLine, endLine, lines) {
|
|
|
323
517
|
if (m && m[1].length <= heading.level) {
|
|
324
518
|
return {
|
|
325
519
|
rule: "anchor-heading-does-not-enclose",
|
|
326
|
-
message: `range extends past
|
|
520
|
+
message: `range extends past the section enclosing anchor "${anchor.text}" (next heading "${m[2].trim()}" at line ${i + 1})`,
|
|
327
521
|
};
|
|
328
522
|
}
|
|
329
523
|
}
|
|
330
524
|
return null;
|
|
331
525
|
}
|
|
332
526
|
const CITATION_RE = /([\w./-]+\.(?:ts|js|mjs|md|yml|yaml|json)):(\d+)(?:-(\d+))?(?:#(\[?\w(?:[\w.-]*\w)?\]?|"[^"\n`]*"))?/g;
|
|
527
|
+
/**
|
|
528
|
+
* Heading-section citations. A CHANGELOG.md that grows by insertion at the
|
|
529
|
+
* top forces every later `path:N-M#anchor` citation to be re-pointed on
|
|
530
|
+
* every release, even with the heading anchor above closing the "wrong
|
|
531
|
+
* section" gap -- the line RANGE itself still drifts on every insertion,
|
|
532
|
+
* so the citation still needs editing, just not silently mis-validated.
|
|
533
|
+
* `path:#heading` sidesteps line numbers entirely: it names a heading
|
|
534
|
+
* (matched the same way the line-range anchor above already matches one,
|
|
535
|
+
* `heading.text.includes(...)`, reused rather than re-invented -- see
|
|
536
|
+
* `parseAnchor`/`findHeadingSection`) and resolves to that heading's own
|
|
537
|
+
* section (from the line after the heading up to, but not including, the
|
|
538
|
+
* next heading of the same or shallower level, or EOF) -- see
|
|
539
|
+
* `findHeadingSection`. The section must exist (`heading-section-not-found`
|
|
540
|
+
* when no such heading is found), must be unambiguous (more than one
|
|
541
|
+
* matching heading is `heading-section-ambiguous`, never silently the
|
|
542
|
+
* first hit -- the same posture `unresolved-ambiguous` already takes for
|
|
543
|
+
* path resolution), and must not be empty (`heading-section-empty`).
|
|
544
|
+
*
|
|
545
|
+
* Deliberately reuses `ANCHOR_HEADING_MAX_LEVEL` (2), not a second cap: the
|
|
546
|
+
* same Keep-a-Changelog nesting that motivates the line-range anchor's cap
|
|
547
|
+
* (`## [x.y.z]` release headings around identically-named, per-release
|
|
548
|
+
* `### Added`/`### Changed`/`### Fixed` subsections) means a level-3+
|
|
549
|
+
* heading name is *never* unique across a real CHANGELOG anyway --
|
|
550
|
+
* matching it would just turn every such citation into a guaranteed
|
|
551
|
+
* `heading-section-ambiguous`. Capping the search at level 2 keeps this
|
|
552
|
+
* form usable for exactly the case it exists for (a release section) and
|
|
553
|
+
* behaves identically to the line-range anchor for the same reason.
|
|
554
|
+
*
|
|
555
|
+
* Optional content anchor: `path:#heading#"text"`, always the quoted form
|
|
556
|
+
* (a section's body is prose/content, not a second heading to look up) --
|
|
557
|
+
* the text must occur on EXACTLY one line inside the resolved section
|
|
558
|
+
* (`heading-section-content-anchor-not-found` / `-ambiguous`), stricter
|
|
559
|
+
* than the line-range string anchor's "at least one line" (`checkAnchor`'s
|
|
560
|
+
* string branch): a whole section is a much larger haystack than a
|
|
561
|
+
* caller-chosen line range, so "present somewhere" is far weaker evidence
|
|
562
|
+
* the anchor still points at the intended spot, and "found on exactly one
|
|
563
|
+
* line" is what the task this form was built for (pinning a citation to
|
|
564
|
+
* one prose sentence inside a growing section) actually needs. The heading
|
|
565
|
+
* position itself keeps only the bare/bracketed form (`#heading`/
|
|
566
|
+
* `#[heading]`); the quoted alternative is reserved for the content anchor
|
|
567
|
+
* position so the two can never be confused by shape alone.
|
|
568
|
+
*
|
|
569
|
+
* Backtick-delimited, unlike the line-range form above (which does not
|
|
570
|
+
* require backticks -- see the "Anchor syntax note" in the README). This
|
|
571
|
+
* is a deliberate, narrower grammar for this form specifically: a bare
|
|
572
|
+
* `path#heading` (the grammar this form used in an earlier round) is
|
|
573
|
+
* indistinguishable from an ordinary Markdown link's target, which
|
|
574
|
+
* commonly has exactly that shape (`[install](docs/README.md#install)`)
|
|
575
|
+
* -- unlike the line-range form (gated by a `:N` no ordinary link ever
|
|
576
|
+
* contains), there was no character available to tell a real citation
|
|
577
|
+
* apart from a relative link's href written in prose, and measuring
|
|
578
|
+
* against real bundles (see the CHANGELOG entry that introduced the
|
|
579
|
+
* `:#` form) showed the collision is not hypothetical: existing docs write
|
|
580
|
+
* `` `path#heading` `` as inline prose pointing at an unrelated anchor,
|
|
581
|
+
* each producing a spurious `heading-section-not-found`. The colon
|
|
582
|
+
* (`path:#heading`) reuses this rule's existing citation signature (a
|
|
583
|
+
* literal `:` no Markdown link fragment ever contains) instead of relying
|
|
584
|
+
* on backticks alone to do that job, and is additionally restricted to a
|
|
585
|
+
* `.md` target: a link fragment's target is always the Markdown doc it
|
|
586
|
+
* points into, never a source or config file, so a non-`.md` path with
|
|
587
|
+
* this shape is never a live citation. Requiring the whole citation
|
|
588
|
+
* inside one pair of backticks stays in place as well, matching this
|
|
589
|
+
* rule's own general recommendation for the line-range form.
|
|
590
|
+
*/
|
|
591
|
+
const HEADING_SECTION_CITATION_RE = /`([\w./-]+\.md):#(\[?\w(?:[\w.-]*\w)?\]?)(?:#("[^"\n`]+"))?`/g;
|
|
592
|
+
/**
|
|
593
|
+
* Companion to `HEADING_SECTION_CITATION_RE`: matches the same
|
|
594
|
+
* backtick + path + `:#` opener but accepts anything up to the closing
|
|
595
|
+
* backtick, so a heading-section citation that fails to parse (an
|
|
596
|
+
* unterminated or empty content-anchor quote, an unquoted third segment,
|
|
597
|
+
* or a non-`.md` target, which includes a non-lowercase `.MD` extension) is
|
|
598
|
+
* still recognised as an *attempt* rather than
|
|
599
|
+
* silently vanishing -- mirrors `anchor-malformed`'s
|
|
600
|
+
* "still-visible-as-a-notice" posture for the line-range anchor form (see
|
|
601
|
+
* `extractMalformedAnchorRaw`). Only used for a match whose span does not
|
|
602
|
+
* already overlap a successful `HEADING_SECTION_CITATION_RE` match (see
|
|
603
|
+
* `collectHeadingSectionMatches`); a well-formed citation never also
|
|
604
|
+
* produces a malformed notice.
|
|
605
|
+
*/
|
|
606
|
+
const HEADING_SECTION_MALFORMED_RE = /`([\w./-]+):#([^`\n]*)`/g;
|
|
333
607
|
// Continuation citation forms (see the "Continuation citations" doc block
|
|
334
608
|
// above). Each requires the backtick delimiter as part of the match so it
|
|
335
609
|
// can never overlap a CITATION_RE match: a full citation's regex match
|
|
@@ -354,6 +628,14 @@ const SHORT_FORM_COLON_RE = /:(\d+)-(\d+)/g;
|
|
|
354
628
|
const TEST_FILE_RE = /\.(test|spec)\.(ts|js|mjs)$/i;
|
|
355
629
|
const TEST_HEAD_LINE_RE = /^\s*(?:describe|it)\s*\(/;
|
|
356
630
|
const TEST_CLOSING_LINE_RE = /^\s*\}\)\s*;\s*$/;
|
|
631
|
+
// Block-head detection for test-range-straddles-block (see the doc block
|
|
632
|
+
// above and checkTestRangeStraddle below): a wider set of head shapes than
|
|
633
|
+
// TEST_HEAD_LINE_RE (which stays short-form-only, unchanged, via
|
|
634
|
+
// checkRangeBoundary/checkShortFormTarget) -- describe/it/test, each with
|
|
635
|
+
// an optional .only/.skip/.each modifier. A leading `export `/`async ` is
|
|
636
|
+
// deliberately not handled: not needed for any bundle this rule has been
|
|
637
|
+
// measured against.
|
|
638
|
+
const TEST_BLOCK_HEAD_RE = /^\s*(?:describe|it|test)(?:\.(?:only|skip|each))?\s*\(/;
|
|
357
639
|
// Markdown block-boundary check (see checkRangeBoundary): a range boundary
|
|
358
640
|
// line that is nothing but a bracket (open or close), optionally with a
|
|
359
641
|
// trailing `,`/`;`, is always a drift signal. A bare code-fence delimiter is
|
|
@@ -629,17 +911,215 @@ function checkTarget(citedPath, startLine, endLine, resolvedPath) {
|
|
|
629
911
|
* are full-citation-only (see that doc block for why), so `checkTarget`
|
|
630
912
|
* itself is untouched and still used as-is for a cont-fresh atom.
|
|
631
913
|
*/
|
|
632
|
-
function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor) {
|
|
914
|
+
function checkFullTarget(citedPath, startLine, endLine, resolvedPath, anchor, requireAnchors) {
|
|
633
915
|
const read = readTarget(resolvedPath);
|
|
634
916
|
if ("rule" in read)
|
|
635
|
-
return read;
|
|
917
|
+
return [read];
|
|
636
918
|
const lines = splitLines(read.content);
|
|
637
919
|
const base = checkTargetLines(citedPath, startLine, endLine, lines);
|
|
638
920
|
if (base)
|
|
639
|
-
return base;
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
921
|
+
return [base];
|
|
922
|
+
const problems = [];
|
|
923
|
+
// test-range-straddles-block, opt-in only (see "Anchor strictness
|
|
924
|
+
// (opt-in)" above and checkTestRangeStraddle): a full citation's own
|
|
925
|
+
// range into a .test./.spec. target (.ts, .js, .mjs) must not cross
|
|
926
|
+
// into another block's head after its own first line. Deliberately
|
|
927
|
+
// scoped to test-file targets only (isTestFile), not also a markdown
|
|
928
|
+
// branch: a
|
|
929
|
+
// full citation into a markdown target legitimately cites a couple of
|
|
930
|
+
// arbitrary lines all the time (e.g. two lines of a code fence example),
|
|
931
|
+
// the exact reasoning checkShortFormTarget's own doc comment already
|
|
932
|
+
// gives for why the sibling block-boundary check was scoped to
|
|
933
|
+
// short-form in the first place. Only meaningful for an actual range
|
|
934
|
+
// (endLine !== null); a single-line citation trivially starts and ends
|
|
935
|
+
// on the same line and has nothing to straddle.
|
|
936
|
+
if (requireAnchors && endLine !== null && isTestFile(citedPath)) {
|
|
937
|
+
const straddle = checkTestRangeStraddle(startLine, endLine, lines);
|
|
938
|
+
if (straddle)
|
|
939
|
+
problems.push(straddle);
|
|
940
|
+
}
|
|
941
|
+
// The anchor check is run independently of the straddle check above
|
|
942
|
+
// (not gated on it having come back clean): a straddling range and a
|
|
943
|
+
// missing/misplaced anchor are two independent problems with the same
|
|
944
|
+
// citation, and reporting only the first one found would silently drop
|
|
945
|
+
// the other from the output every time both happen to co-occur.
|
|
946
|
+
if (anchor) {
|
|
947
|
+
const anchorProblem = checkAnchor(anchor, startLine, endLine ?? startLine, lines, requireAnchors);
|
|
948
|
+
if (anchorProblem)
|
|
949
|
+
problems.push(anchorProblem);
|
|
950
|
+
}
|
|
951
|
+
return problems;
|
|
952
|
+
}
|
|
953
|
+
/**
|
|
954
|
+
* Every Markdown heading in `lines` up to `ANCHOR_HEADING_MAX_LEVEL`, in
|
|
955
|
+
* document order -- the whole-document counterpart of `findEnclosingHeading`
|
|
956
|
+
* (which stops at the nearest heading at or before a given line): a
|
|
957
|
+
* heading-section citation has no start line to search backward from, it
|
|
958
|
+
* needs every candidate to test for a unique match. `fencedLines` (see
|
|
959
|
+
* `computeFencedLineIndices`) excludes a `#`-led line inside a fenced
|
|
960
|
+
* example, same as `findEnclosingHeading`.
|
|
961
|
+
*/
|
|
962
|
+
function collectHeadingsUpToLevel(lines, fencedLines) {
|
|
963
|
+
const headings = [];
|
|
964
|
+
for (let i = 0; i < lines.length; i++) {
|
|
965
|
+
if (fencedLines.has(i))
|
|
966
|
+
continue;
|
|
967
|
+
const m = (lines[i] ?? "").match(MD_HEADING_RE);
|
|
968
|
+
if (m && m[1].length <= ANCHOR_HEADING_MAX_LEVEL) {
|
|
969
|
+
headings.push({ level: m[1].length, text: m[2].trim(), lineNo: i + 1 });
|
|
970
|
+
}
|
|
971
|
+
}
|
|
972
|
+
return headings;
|
|
973
|
+
}
|
|
974
|
+
/**
|
|
975
|
+
* Resolves a heading-section citation's heading text to a single section:
|
|
976
|
+
* every level <= `ANCHOR_HEADING_MAX_LEVEL` heading whose text contains
|
|
977
|
+
* `headingText` (same containment check `checkAnchor`'s heading branch
|
|
978
|
+
* already uses -- reused, not reinvented) is a candidate; zero is
|
|
979
|
+
* `not-found`, more than one is `ambiguous` (never silently the first
|
|
980
|
+
* match), exactly one resolves to the section running from the line right
|
|
981
|
+
* after the heading up to (not including) the next heading at or above the
|
|
982
|
+
* same level, or EOF -- the same enclosure boundary `checkAnchor`'s heading
|
|
983
|
+
* branch walks forward to check, just producing a body range here instead
|
|
984
|
+
* of a pass/fail against an already-known end line.
|
|
985
|
+
*/
|
|
986
|
+
function findHeadingSection(lines, fencedLines, headingText) {
|
|
987
|
+
const headings = collectHeadingsUpToLevel(lines, fencedLines);
|
|
988
|
+
const matches = headings.filter((h) => h.text.includes(headingText));
|
|
989
|
+
if (matches.length === 0)
|
|
990
|
+
return { kind: "not-found" };
|
|
991
|
+
if (matches.length > 1)
|
|
992
|
+
return { kind: "ambiguous", matches };
|
|
993
|
+
const heading = matches[0];
|
|
994
|
+
let bodyEnd = lines.length;
|
|
995
|
+
for (let i = heading.lineNo; i < lines.length; i++) {
|
|
996
|
+
if (fencedLines.has(i))
|
|
997
|
+
continue;
|
|
998
|
+
const m = (lines[i] ?? "").match(MD_HEADING_RE);
|
|
999
|
+
if (m && m[1].length <= heading.level) {
|
|
1000
|
+
bodyEnd = i;
|
|
1001
|
+
break;
|
|
1002
|
+
}
|
|
1003
|
+
}
|
|
1004
|
+
return { kind: "found", heading, bodyStart: heading.lineNo, bodyEnd };
|
|
1005
|
+
}
|
|
1006
|
+
/** True when every line in `[bodyStart, bodyEnd)` is empty or whitespace-only. */
|
|
1007
|
+
function isSectionEmpty(lines, bodyStart, bodyEnd) {
|
|
1008
|
+
for (let i = bodyStart; i < bodyEnd; i++) {
|
|
1009
|
+
if ((lines[i] ?? "").trim() !== "")
|
|
1010
|
+
return false;
|
|
1011
|
+
}
|
|
1012
|
+
return true;
|
|
1013
|
+
}
|
|
1014
|
+
/**
|
|
1015
|
+
* Count of lines in `[bodyStart, bodyEnd)` containing `text` -- a heading
|
|
1016
|
+
* section's content anchor must occur on exactly one such line (see the
|
|
1017
|
+
* "Heading-section citations" doc block above for why this is stricter
|
|
1018
|
+
* than the line-range string anchor's "at least one line").
|
|
1019
|
+
*/
|
|
1020
|
+
function countAnchorOccurrences(lines, bodyStart, bodyEnd, text) {
|
|
1021
|
+
let count = 0;
|
|
1022
|
+
for (let i = bodyStart; i < bodyEnd; i++) {
|
|
1023
|
+
if ((lines[i] ?? "").includes(text))
|
|
1024
|
+
count++;
|
|
1025
|
+
}
|
|
1026
|
+
return count;
|
|
1027
|
+
}
|
|
1028
|
+
/**
|
|
1029
|
+
* A heading-section citation's complete target check (see the
|
|
1030
|
+
* "Heading-section citations" doc block above): the named heading must
|
|
1031
|
+
* exist exactly once, its section must be non-empty, and, when a content
|
|
1032
|
+
* anchor was given, it must occur on exactly one line inside that section.
|
|
1033
|
+
* One problem per citation, checked in that order, matching every other
|
|
1034
|
+
* check in this file's "base checks first" pattern.
|
|
1035
|
+
*/
|
|
1036
|
+
function checkHeadingSectionTarget(headingAnchor, contentAnchor, resolvedPath) {
|
|
1037
|
+
const read = readTarget(resolvedPath);
|
|
1038
|
+
if ("rule" in read)
|
|
1039
|
+
return read;
|
|
1040
|
+
const lines = splitLines(read.content);
|
|
1041
|
+
const fencedLines = computeFencedLineIndices(lines);
|
|
1042
|
+
const section = findHeadingSection(lines, fencedLines, headingAnchor.text);
|
|
1043
|
+
if (section.kind === "not-found") {
|
|
1044
|
+
return {
|
|
1045
|
+
rule: "heading-section-not-found",
|
|
1046
|
+
message: `no heading (level <= ${ANCHOR_HEADING_MAX_LEVEL}) contains "${headingAnchor.text}"`,
|
|
1047
|
+
};
|
|
1048
|
+
}
|
|
1049
|
+
if (section.kind === "ambiguous") {
|
|
1050
|
+
return {
|
|
1051
|
+
rule: "heading-section-ambiguous",
|
|
1052
|
+
message: `${section.matches.length} headings contain "${headingAnchor.text}" (lines ${section.matches
|
|
1053
|
+
.map((m) => m.lineNo)
|
|
1054
|
+
.join(", ")}); not evaluated`,
|
|
1055
|
+
};
|
|
1056
|
+
}
|
|
1057
|
+
if (isSectionEmpty(lines, section.bodyStart, section.bodyEnd)) {
|
|
1058
|
+
return {
|
|
1059
|
+
rule: "heading-section-empty",
|
|
1060
|
+
message: `section under heading "${section.heading.text}" (line ${section.heading.lineNo}) has no non-blank content before the next heading`,
|
|
1061
|
+
};
|
|
1062
|
+
}
|
|
1063
|
+
if (contentAnchor) {
|
|
1064
|
+
const count = countAnchorOccurrences(lines, section.bodyStart, section.bodyEnd, contentAnchor.text);
|
|
1065
|
+
if (count === 0) {
|
|
1066
|
+
return {
|
|
1067
|
+
rule: "heading-section-content-anchor-not-found",
|
|
1068
|
+
message: `content anchor "${contentAnchor.text}" does not occur in the section under heading "${section.heading.text}"`,
|
|
1069
|
+
};
|
|
1070
|
+
}
|
|
1071
|
+
if (count > 1) {
|
|
1072
|
+
return {
|
|
1073
|
+
rule: "heading-section-content-anchor-ambiguous",
|
|
1074
|
+
message: `content anchor "${contentAnchor.text}" occurs on ${count} lines in the section under heading "${section.heading.text}"; expected exactly one`,
|
|
1075
|
+
};
|
|
1076
|
+
}
|
|
1077
|
+
}
|
|
1078
|
+
return null;
|
|
1079
|
+
}
|
|
1080
|
+
/**
|
|
1081
|
+
* Every well-formed heading-section citation in `content`, in document
|
|
1082
|
+
* order. Collected once, up front (before `CITATION_RE`'s own scan in
|
|
1083
|
+
* `scanDoc`), so its match spans can gate both `CITATION_RE` (a full
|
|
1084
|
+
* citation never fires inside a heading-section citation's own quoted
|
|
1085
|
+
* content anchor, see the "Heading-section citations" doc block above) and
|
|
1086
|
+
* the malformed companion scan (see `collectHeadingSectionMalformedMatches`)
|
|
1087
|
+
* without re-deriving the same spans twice.
|
|
1088
|
+
*/
|
|
1089
|
+
function collectHeadingSectionMatches(content) {
|
|
1090
|
+
const out = [];
|
|
1091
|
+
const re = new RegExp(HEADING_SECTION_CITATION_RE.source, HEADING_SECTION_CITATION_RE.flags);
|
|
1092
|
+
let m;
|
|
1093
|
+
while ((m = re.exec(content)) !== null) {
|
|
1094
|
+
out.push({
|
|
1095
|
+
index: m.index,
|
|
1096
|
+
end: m.index + m[0].length,
|
|
1097
|
+
citedPath: m[1],
|
|
1098
|
+
headingAnchor: parseAnchor(m[2]), // group 2 is mandatory
|
|
1099
|
+
contentAnchor: parseAnchor(m[3]),
|
|
1100
|
+
});
|
|
1101
|
+
}
|
|
1102
|
+
return out;
|
|
1103
|
+
}
|
|
1104
|
+
/**
|
|
1105
|
+
* Every backtick + path + `:#` opener in `content` that did NOT parse as a
|
|
1106
|
+
* well-formed heading-section citation (see `HEADING_SECTION_MALFORMED_RE`'s
|
|
1107
|
+
* doc comment) -- an unterminated content-anchor quote, an unquoted third
|
|
1108
|
+
* segment, or a non-`.md` target. `wellFormedSpans` (from
|
|
1109
|
+
* `collectHeadingSectionMatches`) gates out any match that is really just
|
|
1110
|
+
* the successful citation seen from the outside; a well-formed citation
|
|
1111
|
+
* never also produces a malformed notice.
|
|
1112
|
+
*/
|
|
1113
|
+
function collectHeadingSectionMalformedMatches(content, wellFormedSpans) {
|
|
1114
|
+
const out = [];
|
|
1115
|
+
const re = new RegExp(HEADING_SECTION_MALFORMED_RE.source, "g");
|
|
1116
|
+
let m;
|
|
1117
|
+
while ((m = re.exec(content)) !== null) {
|
|
1118
|
+
if (isWithinAnySpan(m.index, wellFormedSpans))
|
|
1119
|
+
continue;
|
|
1120
|
+
out.push({ index: m.index, end: m.index + m[0].length, raw: m[0] });
|
|
1121
|
+
}
|
|
1122
|
+
return out;
|
|
643
1123
|
}
|
|
644
1124
|
// A cont-ext atom only ever extends the *end* of a range whose start line
|
|
645
1125
|
// was already fully checked (blank / closing-brace) when it was cited as
|
|
@@ -668,27 +1148,13 @@ function isTestFile(citedPath) {
|
|
|
668
1148
|
* identical). Used by checkRangeBoundary's markdown branch: citing a
|
|
669
1149
|
* fenced block starting at its own opening fence line is the natural,
|
|
670
1150
|
* correct way to cite it, so that specific case is exempted from the
|
|
671
|
-
* fence-as-drift-signal check (see there).
|
|
1151
|
+
* fence-as-drift-signal check (see there). Derived from `scanFenceLines`
|
|
1152
|
+
* (see there); state at `lineIndex` never depends on any line after it, so
|
|
1153
|
+
* scanning the whole array and reading one index back is equivalent to (and
|
|
1154
|
+
* replaces) the original's own up-to-`lineIndex`-only replay.
|
|
672
1155
|
*/
|
|
673
1156
|
function isFenceOpeningLine(lines, lineIndex) {
|
|
674
|
-
|
|
675
|
-
let fenceMarker;
|
|
676
|
-
for (let i = 0; i <= lineIndex; i++) {
|
|
677
|
-
const trimmed = (lines[i] ?? "").trim();
|
|
678
|
-
if (!inFence && MD_FENCE_DELIM_RE.test(trimmed)) {
|
|
679
|
-
if (i === lineIndex)
|
|
680
|
-
return true;
|
|
681
|
-
inFence = true;
|
|
682
|
-
fenceMarker = trimmed.slice(0, 3);
|
|
683
|
-
}
|
|
684
|
-
else if (inFence && fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
685
|
-
if (i === lineIndex)
|
|
686
|
-
return false;
|
|
687
|
-
inFence = false;
|
|
688
|
-
fenceMarker = undefined;
|
|
689
|
-
}
|
|
690
|
-
}
|
|
691
|
-
return false;
|
|
1157
|
+
return scanFenceLines(lines)[lineIndex]?.opensFence ?? false;
|
|
692
1158
|
}
|
|
693
1159
|
/**
|
|
694
1160
|
* True when `lines[endLineIndex]` is the closing delimiter that matches
|
|
@@ -803,6 +1269,53 @@ function checkRangeBoundary(citedPath, startLine, endLine, lines) {
|
|
|
803
1269
|
}
|
|
804
1270
|
return null;
|
|
805
1271
|
}
|
|
1272
|
+
/**
|
|
1273
|
+
* Width (in characters) of `text`'s leading run of spaces/tabs -- used by
|
|
1274
|
+
* `checkTestRangeStraddle` to tell a NESTED block head (indented deeper
|
|
1275
|
+
* than the range's own start line) apart from a SIBLING or OUTER one
|
|
1276
|
+
* (indented the same or shallower): citing a whole `describe` block
|
|
1277
|
+
* necessarily contains every `it(`/nested-`describe(` head inside its own
|
|
1278
|
+
* body, and those are not straddling anywhere, they are exactly what the
|
|
1279
|
+
* citation is about.
|
|
1280
|
+
*/
|
|
1281
|
+
function leadingWhitespaceWidth(text) {
|
|
1282
|
+
return (text.match(/^[ \t]*/) ?? [""])[0].length;
|
|
1283
|
+
}
|
|
1284
|
+
/**
|
|
1285
|
+
* `test-range-straddles-block` (opt-in, warning): a FULL citation's own
|
|
1286
|
+
* range into a `.test.`/`.spec.` target (`.ts`, `.js`, `.mjs`), see the
|
|
1287
|
+
* "Anchor strictness (opt-in)" doc block above for the full rule text.
|
|
1288
|
+
* Checks every line of `[startLine, endLine]` EXCEPT the range's own
|
|
1289
|
+
* first line (`startLine`
|
|
1290
|
+
* itself is never checked -- see that doc block for why) against
|
|
1291
|
+
* `TEST_BLOCK_HEAD_RE`; the first hit (in document order) is reported,
|
|
1292
|
+
* matching this file's "one problem per citation" pattern. A block-head
|
|
1293
|
+
* line indented STRICTLY DEEPER than the range's own start line is a
|
|
1294
|
+
* NESTED block (e.g. the `it(`s inside a `describe(` the range cites in
|
|
1295
|
+
* full) and is skipped, not reported: a citation covering a whole block is
|
|
1296
|
+
* expected to contain every head line nested inside it. A block-head line
|
|
1297
|
+
* at the same or a shallower indent is a SIBLING or OUTER block and is
|
|
1298
|
+
* still reported -- this is an approximation of "which block scope is
|
|
1299
|
+
* this line lexically in" using indentation instead of a real AST, chosen
|
|
1300
|
+
* because it reproduces the AST-based reference verdict on every straddle
|
|
1301
|
+
* finding this rule has been measured against; a file that mixes tabs and
|
|
1302
|
+
* spaces, or that does not indent nested blocks at all, can defeat it.
|
|
1303
|
+
*/
|
|
1304
|
+
function checkTestRangeStraddle(startLine, endLine, lines) {
|
|
1305
|
+
const startIndent = leadingWhitespaceWidth(lines[startLine - 1] ?? "");
|
|
1306
|
+
for (let i = startLine; i <= endLine - 1 && i < lines.length; i++) {
|
|
1307
|
+
const text = lines[i] ?? "";
|
|
1308
|
+
if (!TEST_BLOCK_HEAD_RE.test(text))
|
|
1309
|
+
continue;
|
|
1310
|
+
if (leadingWhitespaceWidth(text) > startIndent)
|
|
1311
|
+
continue; // nested block
|
|
1312
|
+
return {
|
|
1313
|
+
rule: "test-range-straddles-block",
|
|
1314
|
+
message: `range straddles into another block's head at line ${i + 1} ("${text.trim()}")`,
|
|
1315
|
+
};
|
|
1316
|
+
}
|
|
1317
|
+
return null;
|
|
1318
|
+
}
|
|
806
1319
|
/**
|
|
807
1320
|
* A short-form citation's full check: checkTarget's existing checks
|
|
808
1321
|
* (unreadable-target, inverted-range, range-exceeds-file, blank-start-line,
|
|
@@ -926,30 +1439,29 @@ function collectShortFormMatches(content, excludedSpans) {
|
|
|
926
1439
|
* inclusive. An unterminated fence (no matching close before end of doc) is
|
|
927
1440
|
* treated as running to the end of the content -- conservative, since an
|
|
928
1441
|
* unterminated fence is itself a doc problem outside this rule's scope, not
|
|
929
|
-
* a reason to scan its contents for short-form citations.
|
|
1442
|
+
* a reason to scan its contents for short-form citations. Derived from
|
|
1443
|
+
* `scanFenceLines` (see there): per-line fenced/opens/closes state is
|
|
1444
|
+
* converted to char-offset spans by tracking each line's `[start, end)`
|
|
1445
|
+
* offset in `content` alongside it.
|
|
930
1446
|
*/
|
|
931
1447
|
function computeFencedSpans(content) {
|
|
932
1448
|
const spans = [];
|
|
933
1449
|
const lines = content.split("\n");
|
|
1450
|
+
const states = scanFenceLines(lines);
|
|
934
1451
|
let offset = 0;
|
|
935
|
-
let
|
|
936
|
-
let
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
if (
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
}
|
|
944
|
-
else if (fenceMarker && trimmed.startsWith(fenceMarker)) {
|
|
945
|
-
spans.push([fenceStart, lineEnd]);
|
|
946
|
-
fenceMarker = undefined;
|
|
947
|
-
fenceStart = -1;
|
|
1452
|
+
let spanStart = -1;
|
|
1453
|
+
for (let i = 0; i < lines.length; i++) {
|
|
1454
|
+
const lineEnd = offset + lines[i].length;
|
|
1455
|
+
if (states[i].opensFence)
|
|
1456
|
+
spanStart = offset;
|
|
1457
|
+
if (states[i].closesFence && spanStart >= 0) {
|
|
1458
|
+
spans.push([spanStart, lineEnd]);
|
|
1459
|
+
spanStart = -1;
|
|
948
1460
|
}
|
|
949
1461
|
offset = lineEnd + 1; // +1 for the newline joining this line to the next
|
|
950
1462
|
}
|
|
951
|
-
if (
|
|
952
|
-
spans.push([
|
|
1463
|
+
if (spanStart >= 0) {
|
|
1464
|
+
spans.push([spanStart, content.length]);
|
|
953
1465
|
}
|
|
954
1466
|
return spans;
|
|
955
1467
|
}
|
|
@@ -1148,11 +1660,76 @@ function isWrappedPathContinuation(content, matchIndex) {
|
|
|
1148
1660
|
const prevLine = content.slice(prevLineStart, prevLineEnd);
|
|
1149
1661
|
return /[-–]$/.test(prevLine);
|
|
1150
1662
|
}
|
|
1151
|
-
|
|
1663
|
+
/**
|
|
1664
|
+
* Raw text following a malformed anchor's `#` (see `anchor-malformed` in
|
|
1665
|
+
* the atom-processing loop below), used only to make that finding's message
|
|
1666
|
+
* concrete -- bounded so it reports roughly what was actually typed as the
|
|
1667
|
+
* failed anchor attempt, not an unrelated run of later prose:
|
|
1668
|
+
* - a quoted-anchor attempt (the character right after `#` is `"`) stops
|
|
1669
|
+
* at the next `"` on the same line (the closing quote the author
|
|
1670
|
+
* presumably meant, embedded backticks and all -- that quote is exactly
|
|
1671
|
+
* what makes this a *quoted* anchor attempt rather than a heading one),
|
|
1672
|
+
* or at the end of the line when no such quote exists on it (matches
|
|
1673
|
+
* `CITATION_RE`'s own string alternative, which cannot cross a
|
|
1674
|
+
* newline either);
|
|
1675
|
+
* - any other character after `#` is a heading-anchor attempt, which
|
|
1676
|
+
* stops at the first whitespace (a heading anchor token, like
|
|
1677
|
+
* `CITATION_RE`'s own heading alternative, never contains whitespace)
|
|
1678
|
+
* or the end of the line, whichever comes first;
|
|
1679
|
+
* - either way, capped at `MAX_RAW_LEN` characters so a heading-form
|
|
1680
|
+
* attempt with no whitespace at all before the line ends (or a
|
|
1681
|
+
* pathological single long line) cannot make the message unbounded.
|
|
1682
|
+
*/
|
|
1683
|
+
const MAX_MALFORMED_ANCHOR_RAW_LEN = 60;
|
|
1684
|
+
function extractMalformedAnchorRaw(content, hashIndex) {
|
|
1685
|
+
const from = hashIndex + 1;
|
|
1686
|
+
const nl = content.indexOf("\n", from);
|
|
1687
|
+
const lineEnd = nl === -1 ? content.length : nl;
|
|
1688
|
+
let end;
|
|
1689
|
+
if (content[from] === '"') {
|
|
1690
|
+
const q = content.indexOf('"', from + 1);
|
|
1691
|
+
end = q !== -1 && q < lineEnd ? q + 1 : lineEnd;
|
|
1692
|
+
}
|
|
1693
|
+
else {
|
|
1694
|
+
const rest = content.slice(from, lineEnd);
|
|
1695
|
+
const ws = rest.search(/\s/);
|
|
1696
|
+
end = ws === -1 ? lineEnd : from + ws;
|
|
1697
|
+
}
|
|
1698
|
+
if (end - from > MAX_MALFORMED_ANCHOR_RAW_LEN) {
|
|
1699
|
+
end = from + MAX_MALFORMED_ANCHOR_RAW_LEN;
|
|
1700
|
+
}
|
|
1701
|
+
return content.slice(from, end);
|
|
1702
|
+
}
|
|
1703
|
+
function scanDoc(cache, root, bundleDir, doc, requireAnchors) {
|
|
1152
1704
|
const findings = [];
|
|
1153
1705
|
const content = doc.raw;
|
|
1154
1706
|
const sources = getValidSources(doc.frontmatter.parsed) ?? [];
|
|
1155
1707
|
const docAbsPath = path.join(bundleDir, doc.relPath);
|
|
1708
|
+
// The four `--require-anchors` opt-in checks (anchor-required,
|
|
1709
|
+
// anchor-not-on-last-line, anchor-not-unique-in-range,
|
|
1710
|
+
// test-range-straddles-block) are all exempt for a reserved citing doc
|
|
1711
|
+
// (index.md/log.md, see doc.isReserved), the same carve-out this rule
|
|
1712
|
+
// already gives reserved docs for short-form matching: an append-only
|
|
1713
|
+
// narrative journal routinely narrates historical line-number deltas as
|
|
1714
|
+
// prose about the past, not live citations against current content.
|
|
1715
|
+
// Threaded as `undefined` here (rather than a second boolean everywhere
|
|
1716
|
+
// `requireAnchors` is read) so every opt-in check downstream stays gated
|
|
1717
|
+
// on the exact same "is this option object present" test it already
|
|
1718
|
+
// uses for the non-reserved case.
|
|
1719
|
+
const requireAnchorsForDoc = doc.isReserved ? undefined : requireAnchors;
|
|
1720
|
+
// Heading-section citations (well-formed and malformed) are collected up
|
|
1721
|
+
// front so their char spans can gate the `CITATION_RE` scan below: a
|
|
1722
|
+
// `CITATION_RE` match landing inside a heading-section citation's own
|
|
1723
|
+
// quoted content anchor (e.g. `` `x.md:#2.0.0#"see other.md:12 now"` ``)
|
|
1724
|
+
// is not a second, independent citation -- it is text the heading-section
|
|
1725
|
+
// citation already owns. Reused again lower down for the heading-section
|
|
1726
|
+
// findings themselves, rather than re-scanning by regex a second time.
|
|
1727
|
+
const headingSectionMatches = collectHeadingSectionMatches(content);
|
|
1728
|
+
const headingSectionSpans = headingSectionMatches.map((hs) => [hs.index, hs.end]);
|
|
1729
|
+
const headingSectionMalformedMatches = collectHeadingSectionMalformedMatches(content, headingSectionSpans);
|
|
1730
|
+
for (const hm of headingSectionMalformedMatches) {
|
|
1731
|
+
headingSectionSpans.push([hm.index, hm.end]);
|
|
1732
|
+
}
|
|
1156
1733
|
const fullAtoms = [];
|
|
1157
1734
|
// Char spans of every matched full citation, used to keep short-form
|
|
1158
1735
|
// matching (see collectShortFormMatches) from re-matching the tail of a
|
|
@@ -1163,6 +1740,23 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
1163
1740
|
while ((m = re.exec(content)) !== null) {
|
|
1164
1741
|
if (isWrappedPathContinuation(content, m.index))
|
|
1165
1742
|
continue;
|
|
1743
|
+
if (isWithinAnySpan(m.index, headingSectionSpans))
|
|
1744
|
+
continue;
|
|
1745
|
+
const matchEnd = m.index + m[0].length;
|
|
1746
|
+
// anchor-malformed detection (see the atom-processing loop below for
|
|
1747
|
+
// where the finding is actually pushed): a `#` immediately follows the
|
|
1748
|
+
// range but group 4 (the anchor) did not match -- unbalanced quotes, a
|
|
1749
|
+
// backtick inside a quoted anchor, or nothing at all after the `#`
|
|
1750
|
+
// (e.g. `path:N-M#` at end of line). Computed here, at match time,
|
|
1751
|
+
// because it needs `content`/`matchEnd`; carried on the atom rather
|
|
1752
|
+
// than pushed immediately so the citation's out-of-scope/resolution
|
|
1753
|
+
// posture (path-traversal-rejected, skip, missing-file, ambiguous) can
|
|
1754
|
+
// gate it exactly the same way every other check on this atom already
|
|
1755
|
+
// is -- an out-of-scope or unresolved citation gets none of those
|
|
1756
|
+
// checks either.
|
|
1757
|
+
const malformedAnchorRaw = m[4] === undefined && content[matchEnd] === "#"
|
|
1758
|
+
? extractMalformedAnchorRaw(content, matchEnd)
|
|
1759
|
+
: null;
|
|
1166
1760
|
fullAtoms.push({
|
|
1167
1761
|
kind: "full",
|
|
1168
1762
|
index: m.index,
|
|
@@ -1170,8 +1764,9 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
1170
1764
|
startLine: Number(m[2]),
|
|
1171
1765
|
endLine: m[3] ? Number(m[3]) : null,
|
|
1172
1766
|
anchor: parseAnchor(m[4]),
|
|
1767
|
+
malformedAnchorRaw,
|
|
1173
1768
|
});
|
|
1174
|
-
fullSpans.push([m.index,
|
|
1769
|
+
fullSpans.push([m.index, matchEnd]);
|
|
1175
1770
|
}
|
|
1176
1771
|
const atoms = [...fullAtoms, ...collectContinuationAtoms(content)].sort((a, b) => a.index - b.index);
|
|
1177
1772
|
// `governing`: nearest preceding citation (full or continuation) that
|
|
@@ -1211,7 +1806,13 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
1211
1806
|
continue; // governing (same file) carries over unchanged
|
|
1212
1807
|
}
|
|
1213
1808
|
const { citedPath, startLine, endLine, anchor } = atom;
|
|
1214
|
-
|
|
1809
|
+
// The anchor, when present, is carried in the citation label itself
|
|
1810
|
+
// (not just the anchor-check finding's own message) so two citations to
|
|
1811
|
+
// the same range with different anchors are distinguishable in the
|
|
1812
|
+
// output -- see formatAnchorForLabel and the "Anchored citations" doc
|
|
1813
|
+
// block above. Continuations and short-form citations never carry an
|
|
1814
|
+
// anchor (see there), so their own citation labels are unaffected.
|
|
1815
|
+
const citation = `${citedPath}:${startLine}${endLine ? "-" + endLine : ""}${anchor ? formatAnchorForLabel(anchor) : ""}`;
|
|
1215
1816
|
if (hasParentSegment(citedPath)) {
|
|
1216
1817
|
pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
|
|
1217
1818
|
governing = null;
|
|
@@ -1236,15 +1837,87 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
1236
1837
|
lastStartLine = null;
|
|
1237
1838
|
continue;
|
|
1238
1839
|
}
|
|
1239
|
-
|
|
1840
|
+
// anchor-malformed (notice): only reached for a citation that resolved
|
|
1841
|
+
// to a real file -- see `malformedAnchorRaw`'s doc comment above for
|
|
1842
|
+
// why this is gated the same way missing-file/skip/path-traversal are
|
|
1843
|
+
// already gated for every other check on this atom. The citation is
|
|
1844
|
+
// still checked below via checkFullTarget exactly as an ordinary
|
|
1845
|
+
// anchorless citation would be (anchor is null here by construction --
|
|
1846
|
+
// see parseAnchor); this only ADDS a heads-up that the `#` sitting
|
|
1847
|
+
// right there was silently not read as the anchor it looks like it was
|
|
1848
|
+
// meant to be.
|
|
1849
|
+
if (atom.malformedAnchorRaw !== null) {
|
|
1850
|
+
pushDrift(findings, doc.relPath, citation, "anchor-malformed", `a "#" follows the citation's range but does not parse as a heading or string anchor (raw: "${atom.malformedAnchorRaw}")`, undefined, "notice");
|
|
1851
|
+
}
|
|
1852
|
+
// anchor-required (opt-in, warning): see the "Anchor strictness
|
|
1853
|
+
// (opt-in)" doc block above. Gated on the same posture every other
|
|
1854
|
+
// per-atom check already uses (only reached once the citation resolved
|
|
1855
|
+
// to a real, unambiguous, non-skipped target), plus a reserved citing
|
|
1856
|
+
// doc (e.g. log.md, folded into requireAnchorsForDoc above) and an
|
|
1857
|
+
// allowlisted citedPath being exempt.
|
|
1858
|
+
if (requireAnchorsForDoc &&
|
|
1859
|
+
!anchor &&
|
|
1860
|
+
!matchesAllowPattern(citedPath, requireAnchorsForDoc.allow)) {
|
|
1861
|
+
pushDrift(findings, doc.relPath, citation, "anchor-required", `full citation into an in-repo file carries no #anchor (--require-anchors is on)`, path.relative(root, resolution.path));
|
|
1862
|
+
}
|
|
1863
|
+
const problems = checkFullTarget(citedPath, startLine, endLine, resolution.path, anchor, requireAnchorsForDoc);
|
|
1864
|
+
for (const problem of problems) {
|
|
1865
|
+
if (problem.rule === "unreadable-target") {
|
|
1866
|
+
pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
|
|
1867
|
+
}
|
|
1868
|
+
else {
|
|
1869
|
+
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
|
|
1870
|
+
}
|
|
1871
|
+
}
|
|
1872
|
+
governing = { citedPath, resolvedPath: resolution.path };
|
|
1873
|
+
lastStartLine = startLine;
|
|
1874
|
+
}
|
|
1875
|
+
// Heading-section citations -- see the "Heading-section citations" doc
|
|
1876
|
+
// block above. Independent of `governing`/`lastStartLine`/`fullAtoms`:
|
|
1877
|
+
// this form carries its own path on every citation (never a continuation
|
|
1878
|
+
// or short form), so there is nothing to chain off. Iterates over
|
|
1879
|
+
// `headingSectionMatches`, collected up front (see above), rather than
|
|
1880
|
+
// re-scanning `content` with the regex a second time.
|
|
1881
|
+
for (const hs of headingSectionMatches) {
|
|
1882
|
+
const citedPath = hs.citedPath;
|
|
1883
|
+
const headingAnchor = hs.headingAnchor;
|
|
1884
|
+
const contentAnchor = hs.contentAnchor;
|
|
1885
|
+
const citation = `${citedPath}:${formatAnchorForLabel(headingAnchor)}${contentAnchor ? formatAnchorForLabel(contentAnchor) : ""}`;
|
|
1886
|
+
if (hasParentSegment(citedPath)) {
|
|
1887
|
+
pushDrift(findings, doc.relPath, citation, "path-traversal-rejected", `citedPath contains a ".." segment and was rejected without resolving: ${citedPath}`);
|
|
1888
|
+
continue;
|
|
1889
|
+
}
|
|
1890
|
+
const resolution = resolveCitation(cache, root, docAbsPath, content, sources, citedPath, hs.index);
|
|
1891
|
+
if (!resolution) {
|
|
1892
|
+
pushDrift(findings, doc.relPath, citation, "missing-file", `could not resolve ${citedPath}: tried doc sources, ancestor climb (bare filenames only), repo-root, doc-relative, nearest prior qualified mention, repo-wide search; no candidate file exists`);
|
|
1893
|
+
continue;
|
|
1894
|
+
}
|
|
1895
|
+
if ("skip" in resolution)
|
|
1896
|
+
continue;
|
|
1897
|
+
if ("ambiguous" in resolution) {
|
|
1898
|
+
pushAmbiguous(findings, doc.relPath, citation, resolution.candidates);
|
|
1899
|
+
continue;
|
|
1900
|
+
}
|
|
1901
|
+
const problem = checkHeadingSectionTarget(headingAnchor, contentAnchor, resolution.path);
|
|
1240
1902
|
if (problem?.rule === "unreadable-target") {
|
|
1241
1903
|
pushUnreadable(findings, doc.relPath, citation, path.relative(root, resolution.path), problem.code ?? "UNKNOWN");
|
|
1242
1904
|
}
|
|
1243
1905
|
else if (problem) {
|
|
1244
1906
|
pushDrift(findings, doc.relPath, citation, problem.rule, problem.message, path.relative(root, resolution.path));
|
|
1245
1907
|
}
|
|
1246
|
-
|
|
1247
|
-
|
|
1908
|
+
}
|
|
1909
|
+
// heading-section-malformed (notice): a backtick + path + `:#` opener
|
|
1910
|
+
// that did not parse as a well-formed heading-section citation above --
|
|
1911
|
+
// mirrors `anchor-malformed`'s posture for the line-range anchor form
|
|
1912
|
+
// (see `extractMalformedAnchorRaw`'s doc comment): a typo should not
|
|
1913
|
+
// silently vanish from the very check it was written to exercise.
|
|
1914
|
+
for (const hm of headingSectionMalformedMatches) {
|
|
1915
|
+
// `hm.raw` includes the delimiting backticks (the regex match itself);
|
|
1916
|
+
// the citation label, like every other citation label in this file,
|
|
1917
|
+
// does not repeat them -- pushDrift's message template already wraps
|
|
1918
|
+
// the label in its own pair.
|
|
1919
|
+
const inner = hm.raw.slice(1, -1);
|
|
1920
|
+
pushDrift(findings, doc.relPath, inner, "heading-section-malformed", `a backtick-delimited "path:#" heading-section citation opener does not parse as a well-formed citation (raw: "${inner}")`, undefined, "notice");
|
|
1248
1921
|
}
|
|
1249
1922
|
// Short-form (paragraph-bound) citations -- see that doc block above.
|
|
1250
1923
|
// Deliberately independent of `governing`/`lastStartLine`: short-form
|
|
@@ -1314,7 +1987,7 @@ function scanDoc(cache, root, bundleDir, doc) {
|
|
|
1314
1987
|
}
|
|
1315
1988
|
export const citationsResolveRule = {
|
|
1316
1989
|
id: RULE_ID,
|
|
1317
|
-
description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations) must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
|
|
1990
|
+
description: "`path:N`/`path:N-M` citations (and their `:N`, -`M`/–`M`, (`N`) continuations), and backtick-delimited `` `path:#heading` `` heading-section citations (`.md` targets only), must resolve to a real target file and land on real, non-blank content. Mechanical only: does not verify the cited line is semantically correct.",
|
|
1318
1991
|
run(ctx) {
|
|
1319
1992
|
if (!ctx.repoRoot) {
|
|
1320
1993
|
// Never silently skip: mirrors sources-fresh's posture so a "clean"
|
|
@@ -1333,7 +2006,7 @@ export const citationsResolveRule = {
|
|
|
1333
2006
|
// Fresh per invocation: see findByBasename's doc comment for why this
|
|
1334
2007
|
// is not held at module scope.
|
|
1335
2008
|
const cache = new Map();
|
|
1336
|
-
return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc));
|
|
2009
|
+
return ctx.docs.flatMap((doc) => scanDoc(cache, root, ctx.bundleDir, doc, ctx.requireAnchors));
|
|
1337
2010
|
},
|
|
1338
2011
|
};
|
|
1339
2012
|
//# sourceMappingURL=citations-resolve.js.map
|