@supersuit/hyperspec 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +143 -0
- package/README.md +50 -2
- package/SPEC.md +5 -5
- package/WRITING.md +510 -17
- package/bin/hyperspec.mjs +177 -2
- package/examples/writing/dna/essay-new-managers-teach/features.json +56 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/README.md +14 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/close.md +9 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/opening.md +9 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/status.md +10 -0
- package/examples/writing/dna/essay-new-managers-teach/scope.md +11 -0
- package/examples/writing/essay/claims.jsonl +9 -0
- package/examples/writing/essay/draft.md +82 -0
- package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +2 -2
- package/examples/writing/essay.hyperspec.md +32 -13
- package/examples/writing/story/claims.jsonl +9 -0
- package/examples/writing/story/draft.md +267 -0
- package/examples/writing/story.hyperspec.md +20 -5
- package/package.json +1 -1
- package/src/check.mjs +245 -0
- package/src/dna.mjs +474 -0
- package/src/stations/claims.mjs +150 -0
- package/src/stations/dna.mjs +126 -0
- package/src/stations/form.mjs +115 -0
- package/src/stations/index.mjs +27 -0
- package/src/stations/links.mjs +275 -0
- package/src/stations/private.mjs +117 -0
- package/src/stations/quotes.mjs +168 -0
- package/src/stations/terms.mjs +131 -0
- package/src/stations/util.mjs +99 -0
- package/src/writing-exports.mjs +8 -3
- package/src/writing-fields.mjs +197 -1
- package/src/writing-template.mjs +8 -0
- package/examples/writing/essay/goldens/close.md +0 -2
- package/examples/writing/essay/goldens/opening.md +0 -2
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
// Station "dna" (hyperspec 0.6). Measures the draft the way `hyperspec dna measure`
|
|
2
|
+
// measures a scope's goldens (dna.mjs's measureFeatures, the same function, never a second copy)
|
|
3
|
+
// and compares it with the scope's recorded features.json. Pure and deterministic: arithmetic on
|
|
4
|
+
// two features objects, no model call and no judgment about whether a difference matters beyond the
|
|
5
|
+
// band rule below.
|
|
6
|
+
//
|
|
7
|
+
// Runs only when writing.dna.scope_dir is set (no scope_dir: skip) and its features.json is
|
|
8
|
+
// current. "Current" is lint's own test-6 notion, read through the one function lint uses
|
|
9
|
+
// (writing-fields.mjs's featuresStaleness): a missing or stale features.json, or a scope whose
|
|
10
|
+
// goldens folder cannot be read, makes this station skip with the reason. Lint already fails the
|
|
11
|
+
// spec for each of those, and check runs no station on a spec lint fails, so the skip is defense in
|
|
12
|
+
// depth for a direct caller; comparing a draft against numbers that no longer describe the goldens
|
|
13
|
+
// would report drift from something that is not the writer's voice.
|
|
14
|
+
//
|
|
15
|
+
// Compared features, and only when the scope's features.json records them: sentence_length.mean,
|
|
16
|
+
// paragraph_length.mean_sentences and paragraph_length.mean_words (the two paragraph-length means),
|
|
17
|
+
// and every per-1000-words rate (each rates_per_1000_words entry, plus contraction_rate and the
|
|
18
|
+
// three person rates, which measureFeatures computes per 1000 words too). Counts, medians, p90,
|
|
19
|
+
// mean word length and signature words are not compared. For a scope value v the band is
|
|
20
|
+
// v / 1.5 (floor 0) to max(v * 1.5, v + 5), edges inside; a draft value outside it is one
|
|
21
|
+
// station-dna-drift finding carrying both values and the band.
|
|
22
|
+
//
|
|
23
|
+
// The em dash has one rule of its own: when the scope's em dash rate is 0 and the draft's is above
|
|
24
|
+
// 0, that is station-dna-em-dash, and the em dash's band finding is not also reported (it is the
|
|
25
|
+
// same defect, and the stricter rule already names it). Fenced and inline code are masked out of
|
|
26
|
+
// the draft before it is measured, like every station that reads prose.
|
|
27
|
+
//
|
|
28
|
+
// Severity: both findings are WARNINGS, never failures, so the station's status is pass
|
|
29
|
+
// whenever it runs. This station measures; judging whether the draft is in the writer's voice
|
|
30
|
+
// belongs to the lineup judge of a later release, and a band over a handful of goldens is evidence
|
|
31
|
+
// for that judge, not a verdict.
|
|
32
|
+
|
|
33
|
+
import { readFileSync } from "node:fs";
|
|
34
|
+
import { join, resolve } from "node:path";
|
|
35
|
+
import { str } from "../placeholder.mjs";
|
|
36
|
+
import { measureFeatures, readScope } from "../dna.mjs";
|
|
37
|
+
import { featuresStaleness } from "../writing-fields.mjs";
|
|
38
|
+
import { lineAt, maskCode } from "./util.mjs";
|
|
39
|
+
|
|
40
|
+
export const name = "dna";
|
|
41
|
+
|
|
42
|
+
const MEANS = [["sentence_length", "mean"], ["paragraph_length", "mean_sentences"], ["paragraph_length", "mean_words"]];
|
|
43
|
+
const FLAT_RATES = ["contraction_rate", "first_person_singular_rate", "first_person_plural_rate", "second_person_rate"];
|
|
44
|
+
const EM_DASH = "rates_per_1000_words.em_dash";
|
|
45
|
+
|
|
46
|
+
const isObj = (v) => v != null && typeof v === "object" && !Array.isArray(v);
|
|
47
|
+
const isNum = (v) => typeof v === "number" && Number.isFinite(v);
|
|
48
|
+
const fmt = (x) => String(Math.round(x * 1000) / 1000);
|
|
49
|
+
|
|
50
|
+
// Every compared feature as [dotted name, scope value, draft value], in a fixed order, for the
|
|
51
|
+
// features the scope records as numbers. A draft value that is not a number reads as 0.
|
|
52
|
+
function comparable(scope, draft) {
|
|
53
|
+
const out = [];
|
|
54
|
+
const add = (key, s, d) => { if (isNum(s)) out.push([key, s, isNum(d) ? d : 0]); };
|
|
55
|
+
for (const [group, field] of MEANS) add(`${group}.${field}`, scope?.[group]?.[field], draft?.[group]?.[field]);
|
|
56
|
+
const rates = isObj(scope?.rates_per_1000_words) ? scope.rates_per_1000_words : {};
|
|
57
|
+
for (const k of Object.keys(rates)) add(`rates_per_1000_words.${k}`, rates[k], draft?.rates_per_1000_words?.[k]);
|
|
58
|
+
for (const k of FLAT_RATES) add(k, scope?.[k], draft?.[k]);
|
|
59
|
+
return out;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// compareFeatures(scopeFeatures, draftFeatures): the band rule alone, over two measureFeatures-shaped
|
|
63
|
+
// objects. Returns [{ feature, scope, draft, low, high }] for every compared feature outside its band.
|
|
64
|
+
export function compareFeatures(scopeFeatures, draftFeatures) {
|
|
65
|
+
const drift = [];
|
|
66
|
+
for (const [feature, s, d] of comparable(scopeFeatures, draftFeatures)) {
|
|
67
|
+
const low = Math.max(0, s / 1.5);
|
|
68
|
+
const high = Math.max(s * 1.5, s + 5);
|
|
69
|
+
if (d < low || d > high) drift.push({ feature, scope: s, draft: d, low, high });
|
|
70
|
+
}
|
|
71
|
+
return drift;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const skip = (reason) => ({ station: name, status: "skip", findings: [], reason });
|
|
75
|
+
|
|
76
|
+
export function run(spec, draft) {
|
|
77
|
+
const scopeDir = str(spec?.data?.writing?.dna?.scope_dir);
|
|
78
|
+
if (!scopeDir) return skip("writing.dna.scope_dir is not set");
|
|
79
|
+
|
|
80
|
+
const scopeAbs = resolve(spec?.dir || ".", scopeDir);
|
|
81
|
+
const measure = `run \`hyperspec dna measure ${scopeDir}\``;
|
|
82
|
+
const disk = readScope(scopeAbs, { displayDir: scopeDir });
|
|
83
|
+
if (disk.findings.some((x) => x.id === "writing-dna-goldens-missing" || x.id === "writing-dna-goldens-outside")) {
|
|
84
|
+
return skip(`writing.dna.scope_dir "${scopeDir}": its goldens cannot be read (run \`hyperspec lint\` for details)`);
|
|
85
|
+
}
|
|
86
|
+
const featuresPath = join(scopeAbs, "features.json");
|
|
87
|
+
const stale = featuresStaleness(featuresPath, disk.scope, disk.goldens);
|
|
88
|
+
if (stale === "missing") return skip(`writing.dna.scope_dir "${scopeDir}" has no features.json (or it is not valid JSON); ${measure}`);
|
|
89
|
+
if (stale) return skip(`writing.dna.scope_dir "${scopeDir}"'s features.json is stale: ${stale}; ${measure}`);
|
|
90
|
+
|
|
91
|
+
let scopeFeatures;
|
|
92
|
+
try { scopeFeatures = JSON.parse(readFileSync(featuresPath, "utf8")).features; } catch { scopeFeatures = null; }
|
|
93
|
+
if (!isObj(scopeFeatures)) return skip(`writing.dna.scope_dir "${scopeDir}" has no features.json (or it is not valid JSON); ${measure}`);
|
|
94
|
+
|
|
95
|
+
const masked = maskCode(draft.text);
|
|
96
|
+
const draftFeatures = measureFeatures([masked]);
|
|
97
|
+
const findings = [];
|
|
98
|
+
|
|
99
|
+
const scopeEm = scopeFeatures.rates_per_1000_words?.em_dash;
|
|
100
|
+
const draftEm = draftFeatures.rates_per_1000_words.em_dash;
|
|
101
|
+
const emDashRule = scopeEm === 0 && draftEm > 0;
|
|
102
|
+
if (emDashRule) {
|
|
103
|
+
findings.push({
|
|
104
|
+
station: name,
|
|
105
|
+
id: "station-dna-em-dash",
|
|
106
|
+
severity: "warn",
|
|
107
|
+
line: lineAt(draft.text, masked.indexOf("\u2014")),
|
|
108
|
+
message: `the draft uses em dashes (${fmt(draftEm)} per 1000 words); the scope's goldens use none (0)`,
|
|
109
|
+
fix: "Rewrite each em dash as the punctuation the goldens use instead: a comma, a colon, parentheses or a new sentence.",
|
|
110
|
+
});
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
for (const d of compareFeatures(scopeFeatures, draftFeatures)) {
|
|
114
|
+
if (emDashRule && d.feature === EM_DASH) continue;
|
|
115
|
+
findings.push({
|
|
116
|
+
station: name,
|
|
117
|
+
id: "station-dna-drift",
|
|
118
|
+
severity: "warn",
|
|
119
|
+
message: `${d.feature} is ${fmt(d.draft)} in the draft; the scope's goldens measure ${fmt(d.scope)}, band ${fmt(d.low)} to ${fmt(d.high)}`,
|
|
120
|
+
fix: `Bring ${d.feature} back inside the band, or, if the scope no longer describes this writer, re-measure it with better goldens.`,
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// Every finding here is a warning, so the station passes whenever it runs.
|
|
125
|
+
return { station: name, status: findings.some((x) => x.severity === "fail") ? "fail" : "pass", findings };
|
|
126
|
+
}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
// Station "form" (hyperspec 0.6). The first of hyperspec check's deterministic
|
|
2
|
+
// stations: it checks a draft's word count against writing.form.length, and checks that every
|
|
3
|
+
// writing.form.required_parts entry actually shows up in the draft. Pure and deterministic, like
|
|
4
|
+
// every station: (spec, draft) in, a result out, no filesystem access beyond what the caller
|
|
5
|
+
// already read, no model call.
|
|
6
|
+
//
|
|
7
|
+
// Length: writing.form.length.unit is only measured when it is exactly "words" (lint already
|
|
8
|
+
// requires the key to be present; a spec naming an unmeasured unit, e.g. "characters" or
|
|
9
|
+
// "minutes", is not wrong, this build just cannot grade it yet). When the unit is not "words"
|
|
10
|
+
// the whole station skips, findings included, rather than silently passing or half-checking: a
|
|
11
|
+
// length this build cannot read is not evidence the length is fine, and running required_parts
|
|
12
|
+
// alone while staying silent about length would read as a check that covered more than it did.
|
|
13
|
+
//
|
|
14
|
+
// Required parts: a required part is judged present two ways, either one is enough, because
|
|
15
|
+
// required_parts sometimes names a heading-shaped thing ("claim", "evidence", "close") and
|
|
16
|
+
// sometimes names a field a form fills in inline rather than under its own heading (a memo's
|
|
17
|
+
// "To:", an email's "Subject:"). Neither the schema nor the draft says which kind a given part
|
|
18
|
+
// is, so both checks always run for every part, regardless of the form's name:
|
|
19
|
+
// - an ATX heading (up to three spaces of indent, 1 to 6 "#" characters, a space, the text, an
|
|
20
|
+
// optional closing run of "#"s) whose text, trimmed and case-folded, equals the part name; or
|
|
21
|
+
// - a line whose text, trimmed and case-folded, starts with the part name immediately
|
|
22
|
+
// followed by ":".
|
|
23
|
+
// Fenced and inline code are masked first. This is a literal, narrow reading on purpose: it will
|
|
24
|
+
// miss a heading spelled "## The Claim" against a required part "claim", a Setext heading
|
|
25
|
+
// (underlined with === or ---), or one styled "**Claim**". Widening the match is a later
|
|
26
|
+
// station's decision once real drafts show what this narrow reading actually misses.
|
|
27
|
+
|
|
28
|
+
import { wordsOf } from "../dna.mjs";
|
|
29
|
+
import { maskCode } from "./util.mjs";
|
|
30
|
+
|
|
31
|
+
export const name = "form";
|
|
32
|
+
|
|
33
|
+
// An ATX heading as CommonMark reads it: up to three spaces of indent, 1 to 6 "#"s, then a space
|
|
34
|
+
// or tab and the text (or nothing); an optional closing run of "#"s is not part of the text.
|
|
35
|
+
const ATX_HEADING = /^ {0,3}(#{1,6})(?:[ \t]+(.*))?$/;
|
|
36
|
+
const headingText = (m) => (m[2] ?? "").replace(/(^|[ \t]+)#+[ \t]*$/, "").trim();
|
|
37
|
+
|
|
38
|
+
// A finding id's slug half: lowercase, non [a-z0-9] runs collapsed to one "-", no leading or
|
|
39
|
+
// trailing "-". Falls back to `fallback` when nothing alphanumeric survives (e.g. a required
|
|
40
|
+
// part that is pure punctuation), so an id is never left with a trailing "station-form-required-part-".
|
|
41
|
+
function slug(text, fallback) {
|
|
42
|
+
const s = String(text).toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "");
|
|
43
|
+
return s || fallback;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Whether `part` shows up in the draft, either as a matching ATX heading or as a line starting
|
|
47
|
+
// "<part>:" (both compared trimmed and case-folded).
|
|
48
|
+
function partPresent(lines, part) {
|
|
49
|
+
const needle = part.trim().toLowerCase();
|
|
50
|
+
for (const line of lines) {
|
|
51
|
+
const heading = ATX_HEADING.exec(line);
|
|
52
|
+
if (heading && headingText(heading).toLowerCase() === needle) return true;
|
|
53
|
+
if (line.trim().toLowerCase().startsWith(`${needle}:`)) return true;
|
|
54
|
+
}
|
|
55
|
+
return false;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// run(spec, draft): spec is a loadSpec()-shaped object (spec.data.writing.form is what this
|
|
59
|
+
// station reads); draft is { path, text, lines, sha256 } as src/check.mjs builds it. ctx (a
|
|
60
|
+
// third argument every station receives) is unused here: form shares nothing with the other
|
|
61
|
+
// stations.
|
|
62
|
+
export function run(spec, draft) {
|
|
63
|
+
const form = spec?.data?.writing?.form ?? {};
|
|
64
|
+
const length = form.length && typeof form.length === "object" ? form.length : {};
|
|
65
|
+
// Trimmed and case-folded for the comparison: lint only requires length.unit to be non-
|
|
66
|
+
// placeholder text (writing-fields.mjs's formFields), not literally the lowercase word
|
|
67
|
+
// "words", so a spec author who writes "Words" or "WORDS" still gets the word count checked
|
|
68
|
+
// rather than a silent skip. The reason message on a genuine skip still shows the unit as
|
|
69
|
+
// written, never lowercased, since that is what the spec actually says.
|
|
70
|
+
const rawUnit = typeof length.unit === "string" ? length.unit.trim() : "";
|
|
71
|
+
const unit = rawUnit.toLowerCase();
|
|
72
|
+
|
|
73
|
+
if (unit !== "words") {
|
|
74
|
+
return { station: name, status: "skip", findings: [], reason: `length unit ${rawUnit || "(none)"} is not measured yet` };
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const findings = [];
|
|
78
|
+
const min = Number(length.min);
|
|
79
|
+
const max = Number(length.max);
|
|
80
|
+
const count = wordsOf(draft.text).length;
|
|
81
|
+
if (!(count >= min && count <= max)) {
|
|
82
|
+
findings.push({
|
|
83
|
+
station: name,
|
|
84
|
+
id: "station-form-length",
|
|
85
|
+
severity: "fail",
|
|
86
|
+
message: `word count ${count} is outside writing.form.length (${min} to ${max} words)`,
|
|
87
|
+
fix: `Trim or expand the draft to ${min}-${max} words; it is currently ${count}.`,
|
|
88
|
+
});
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// Code is masked first, like every station that reads prose: a heading shown inside a fenced
|
|
92
|
+
// block is an example, not the draft's own part.
|
|
93
|
+
const lines = maskCode(draft.text).split("\n").map((l) => (l.endsWith("\r") ? l.slice(0, -1) : l));
|
|
94
|
+
const parts = Array.isArray(form.required_parts) ? form.required_parts.filter((p) => typeof p === "string" && p.trim()) : [];
|
|
95
|
+
const usedIds = new Set();
|
|
96
|
+
for (const part of parts) {
|
|
97
|
+
if (partPresent(lines, part)) continue;
|
|
98
|
+
let id = `station-form-required-part-${slug(part, "part")}`;
|
|
99
|
+
// Two required parts that slug to the same string (e.g. "Close" and "close!") would
|
|
100
|
+
// otherwise collide on one finding id; the second and later ones get a numeric suffix so
|
|
101
|
+
// every missing part still gets its own finding.
|
|
102
|
+
let n = 2;
|
|
103
|
+
while (usedIds.has(id)) { id = `station-form-required-part-${slug(part, "part")}-${n}`; n += 1; }
|
|
104
|
+
usedIds.add(id);
|
|
105
|
+
findings.push({
|
|
106
|
+
station: name,
|
|
107
|
+
id,
|
|
108
|
+
severity: "fail",
|
|
109
|
+
message: `required part "${part}" does not appear as a heading or a "${part}:" line`,
|
|
110
|
+
fix: `Add a heading ("# ${part}") or a line starting "${part}:" for required part "${part}".`,
|
|
111
|
+
});
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
return { station: name, status: findings.length ? "fail" : "pass", findings };
|
|
115
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
// The station registry: every station `hyperspec check` knows about, in run order. Each entry's
|
|
2
|
+
// run is the pure function (spec, draft, ctx) -> { station, status, findings, reason? } that
|
|
3
|
+
// src/check.mjs calls; adding a station is adding one file plus one line here, which is the whole
|
|
4
|
+
// point of the registry existing rather than check.mjs importing each station by name itself.
|
|
5
|
+
//
|
|
6
|
+
// The order: form, terms, claims, quotes, private, dna, links. quotes and private share ctx (util.mjs's
|
|
7
|
+
// markedSegments caches the spec's marked materials there), so a check run reads them once.
|
|
8
|
+
|
|
9
|
+
import * as form from "./form.mjs";
|
|
10
|
+
import * as terms from "./terms.mjs";
|
|
11
|
+
import * as claims from "./claims.mjs";
|
|
12
|
+
import * as quotes from "./quotes.mjs";
|
|
13
|
+
import * as privateStation from "./private.mjs";
|
|
14
|
+
import * as dna from "./dna.mjs";
|
|
15
|
+
import * as links from "./links.mjs";
|
|
16
|
+
|
|
17
|
+
export const STATIONS = Object.freeze([
|
|
18
|
+
{ name: form.name, run: form.run },
|
|
19
|
+
{ name: terms.name, run: terms.run },
|
|
20
|
+
{ name: claims.name, run: claims.run },
|
|
21
|
+
{ name: quotes.name, run: quotes.run },
|
|
22
|
+
{ name: privateStation.name, run: privateStation.run },
|
|
23
|
+
{ name: dna.name, run: dna.run },
|
|
24
|
+
{ name: links.name, run: links.run },
|
|
25
|
+
]);
|
|
26
|
+
|
|
27
|
+
export const STATION_NAMES = Object.freeze(STATIONS.map((s) => s.name));
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
// Station "links" (hyperspec 0.6). Checks every Markdown link and
|
|
2
|
+
// every bare http(s):// URL in the draft: an absolute http/https URL must be well-formed (a real
|
|
3
|
+
// host), a mailto: link must carry an address, any other scheme fails outright, a link rooted at
|
|
4
|
+
// "/" is site-root-relative and warns rather than resolving locally (there is
|
|
5
|
+
// no site root here to resolve it against), and any other relative link must resolve to a file
|
|
6
|
+
// that actually exists, relative to the DRAFT's own directory (never the spec's). No network
|
|
7
|
+
// access, ever: well-formed means "the URL parses and names a host", not "the host answers".
|
|
8
|
+
//
|
|
9
|
+
// Three Markdown link forms are checked, all through the same checkUrl:
|
|
10
|
+
// - inline: [text](url)
|
|
11
|
+
// - reference, full or collapsed: [text][ref] / [ref][], resolved against a [ref]: url
|
|
12
|
+
// definition elsewhere in the draft (label matching is case-insensitive and whitespace-
|
|
13
|
+
// collapsed); a full or collapsed reference with no matching definition is its own
|
|
14
|
+
// finding, station-links-undefined-reference, naming the label, since the second bracket
|
|
15
|
+
// pair says a link was meant.
|
|
16
|
+
// - shortcut reference: [ref] alone (no second bracket pair), checked only when a definition
|
|
17
|
+
// for that label exists, once every inline link, reference definition and full/collapsed
|
|
18
|
+
// reference has already been matched and masked out of the text. With no definition it is
|
|
19
|
+
// ordinary text, as CommonMark renders it: [sic], a task-list [x], a footnote-style [1].
|
|
20
|
+
//
|
|
21
|
+
// Anchors: the "#fragment" part of any link is stripped before checking anything else, so
|
|
22
|
+
// "notes.md#section-two" is checked as "notes.md" (its target heading is never verified), and a
|
|
23
|
+
// link that is nothing but "#fragment" (empty path once the anchor is stripped) always resolves,
|
|
24
|
+
// since it points at the draft itself.
|
|
25
|
+
//
|
|
26
|
+
// Fenced code blocks and inline code spans are masked out (src/stations/util.mjs's maskCode)
|
|
27
|
+
// before ANY of the above runs, so a Markdown link or URL shown as illustrative syntax inside a
|
|
28
|
+
// fence or a `` `span` `` is never checked as a real, followable link.
|
|
29
|
+
|
|
30
|
+
import { existsSync } from "node:fs";
|
|
31
|
+
import { dirname, resolve } from "node:path";
|
|
32
|
+
import { lineAt, maskCode, maskRanges, truncate } from "./util.mjs";
|
|
33
|
+
|
|
34
|
+
export const name = "links";
|
|
35
|
+
|
|
36
|
+
// Every "[text](url)" span in `text`, in document order: { start, end, url }. The url is the raw
|
|
37
|
+
// content between the parens, up to the first run of whitespace (a Markdown link may carry a
|
|
38
|
+
// `"title"` after the url, separated by whitespace; that title is not itself a link and is
|
|
39
|
+
// ignored here). `end` is the index just past the closing ")", used both to report a line number
|
|
40
|
+
// and to mask the span out before later scans, so the same URL is never counted twice.
|
|
41
|
+
const MD_LINK_RE = /\[([^\]]*)\]\(([^)]+)\)/g;
|
|
42
|
+
|
|
43
|
+
function markdownLinks(text) {
|
|
44
|
+
const out = [];
|
|
45
|
+
let m;
|
|
46
|
+
MD_LINK_RE.lastIndex = 0;
|
|
47
|
+
while ((m = MD_LINK_RE.exec(text))) {
|
|
48
|
+
const inner = m[2].trim();
|
|
49
|
+
const url = inner.split(/\s+/)[0];
|
|
50
|
+
if (url) out.push({ start: m.index, end: m.index + m[0].length, url });
|
|
51
|
+
}
|
|
52
|
+
return out;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// A reference definition line: up to 3 spaces of indent, "[label]:", the URL, and an optional
|
|
56
|
+
// title in "quotes", 'quotes' or (parens). One definition per line, the common case; a multi-line
|
|
57
|
+
// definition (title on the following line) is not attempted.
|
|
58
|
+
const DEF_RE = /^[ \t]{0,3}\[([^\]]+)\]:[ \t]*(\S+)[ \t]*(?:"[^"]*"|'[^']*'|\([^)]*\))?[ \t]*$/gm;
|
|
59
|
+
|
|
60
|
+
const normalizeLabel = (label) => label.trim().toLowerCase().replace(/\s+/g, " ");
|
|
61
|
+
|
|
62
|
+
// Every reference definition in `text`: a Map from normalized label to { url }, plus the [start,
|
|
63
|
+
// end) span of each whole definition line, ready to mask out (so a definition's own "[label]:" is
|
|
64
|
+
// never later mistaken for a reference USE, and its URL is never also picked up by the bare-URL
|
|
65
|
+
// scan below). The FIRST definition for a given label wins on a duplicate, matching CommonMark.
|
|
66
|
+
function definitions(text) {
|
|
67
|
+
const defs = new Map();
|
|
68
|
+
const spans = [];
|
|
69
|
+
let m;
|
|
70
|
+
DEF_RE.lastIndex = 0;
|
|
71
|
+
while ((m = DEF_RE.exec(text))) {
|
|
72
|
+
const label = normalizeLabel(m[1]);
|
|
73
|
+
if (!defs.has(label)) defs.set(label, { url: m[2] });
|
|
74
|
+
spans.push({ start: m.index, end: m.index + m[0].length });
|
|
75
|
+
}
|
|
76
|
+
return { defs, spans };
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Every "[text][ref]" (full) or "[text][]" (collapsed, label = text) span, in document order:
|
|
80
|
+
// { start, end, label }. Run AFTER inline links and definitions are already masked out of `text`,
|
|
81
|
+
// so this can only match genuine two-bracket-pair reference syntax.
|
|
82
|
+
const FULL_REF_RE = /\[([^\]]*)\]\[([^\]]*)\]/g;
|
|
83
|
+
|
|
84
|
+
function fullReferences(text) {
|
|
85
|
+
const out = [];
|
|
86
|
+
let m;
|
|
87
|
+
FULL_REF_RE.lastIndex = 0;
|
|
88
|
+
while ((m = FULL_REF_RE.exec(text))) {
|
|
89
|
+
const label = (m[2].trim() || m[1].trim());
|
|
90
|
+
if (label) out.push({ start: m.index, end: m.index + m[0].length, label });
|
|
91
|
+
}
|
|
92
|
+
return out;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// Every remaining "[label]" span, in document order: { start, end, label }. Run AFTER inline
|
|
96
|
+
// links, definitions and full/collapsed references are already masked out, so whatever "[...]"
|
|
97
|
+
// is left is either a shortcut reference or ordinary bracketed prose. The caller keeps only the
|
|
98
|
+
// ones whose label has a definition: CommonMark renders an undefined "[label]" as plain text, so
|
|
99
|
+
// an editorial [sic], a task-list [x] or a footnote-style [1] is never a link.
|
|
100
|
+
const SHORTCUT_REF_RE = /\[([^\]]+)\]/g;
|
|
101
|
+
|
|
102
|
+
function shortcutReferences(text) {
|
|
103
|
+
const out = [];
|
|
104
|
+
let m;
|
|
105
|
+
SHORTCUT_REF_RE.lastIndex = 0;
|
|
106
|
+
while ((m = SHORTCUT_REF_RE.exec(text))) {
|
|
107
|
+
const label = m[1].trim();
|
|
108
|
+
if (label) out.push({ start: m.index, end: m.index + m[0].length, label });
|
|
109
|
+
}
|
|
110
|
+
return out;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// Every bare "http://" or "https://" URL left in `text` (after every other link form has been
|
|
114
|
+
// masked out of it), stopping at whitespace or a closing bracket/paren/angle-bracket that is more
|
|
115
|
+
// likely to be surrounding punctuation than part of the URL itself. The host part may be empty,
|
|
116
|
+
// so a bare "https://" is matched and fails as malformed rather than passing unseen.
|
|
117
|
+
const BARE_URL_RE = /https?:\/\/[^\s)>\]]*/g;
|
|
118
|
+
|
|
119
|
+
function bareUrls(text) {
|
|
120
|
+
const out = [];
|
|
121
|
+
let m;
|
|
122
|
+
BARE_URL_RE.lastIndex = 0;
|
|
123
|
+
while ((m = BARE_URL_RE.exec(text))) out.push({ start: m.index, end: m.index + m[0].length, url: m[0] });
|
|
124
|
+
return out;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const SCHEME_RE = /^([a-zA-Z][a-zA-Z0-9+.-]*):/;
|
|
128
|
+
|
|
129
|
+
// Classifies one URL, with its "#..." anchor already known to the caller as stripped: "http" (the
|
|
130
|
+
// scheme itself, lowercased, is on `scheme`), "mailto", "other-scheme" (anything else with a
|
|
131
|
+
// scheme prefix), or "relative" (no scheme prefix at all: a bare path, root-relative or relative).
|
|
132
|
+
function classify(withoutAnchor) {
|
|
133
|
+
const m = SCHEME_RE.exec(withoutAnchor);
|
|
134
|
+
if (!m) return { kind: "relative", path: withoutAnchor };
|
|
135
|
+
const scheme = m[1].toLowerCase();
|
|
136
|
+
if (scheme === "http" || scheme === "https") return { kind: "http", scheme };
|
|
137
|
+
if (scheme === "mailto") return { kind: "mailto" };
|
|
138
|
+
return { kind: "other-scheme", scheme };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// Why `url` is broken (or merely unresolvable locally), as one of "malformed-http",
|
|
142
|
+
// "malformed-mailto", "bad-scheme", "root-relative" or "broken-relative", or null when it is
|
|
143
|
+
// fine. draftDirAbs is the draft's own directory (resolved once by the caller), which every
|
|
144
|
+
// relative link resolves against, never the spec's directory.
|
|
145
|
+
function checkUrl(url, draftDirAbs) {
|
|
146
|
+
const hashIdx = url.indexOf("#");
|
|
147
|
+
const withoutAnchor = hashIdx === -1 ? url : url.slice(0, hashIdx);
|
|
148
|
+
const c = classify(withoutAnchor);
|
|
149
|
+
|
|
150
|
+
if (c.kind === "http") {
|
|
151
|
+
try {
|
|
152
|
+
const u = new URL(url);
|
|
153
|
+
if (!u.hostname) return "malformed-http";
|
|
154
|
+
} catch {
|
|
155
|
+
return "malformed-http";
|
|
156
|
+
}
|
|
157
|
+
return null;
|
|
158
|
+
}
|
|
159
|
+
if (c.kind === "mailto") {
|
|
160
|
+
const address = withoutAnchor.slice("mailto:".length).split("?")[0].trim();
|
|
161
|
+
return address ? null : "malformed-mailto";
|
|
162
|
+
}
|
|
163
|
+
if (c.kind === "other-scheme") return "bad-scheme";
|
|
164
|
+
|
|
165
|
+
// relative: an empty path (the link was nothing but "#fragment", or literally empty) always
|
|
166
|
+
// resolves, since it points at the draft's own file, which exists by construction (check.mjs
|
|
167
|
+
// only ever builds a draft object after successfully reading it).
|
|
168
|
+
if (!c.path) return null;
|
|
169
|
+
// A link rooted at "/" names a path from some site's root, which this station has no way to
|
|
170
|
+
// resolve (there is no "site" here, only the draft's own folder), so it is neither a pass nor a
|
|
171
|
+
// fail -- a warning, naming the fact that it cannot be checked locally.
|
|
172
|
+
if (c.path.startsWith("/")) return "root-relative";
|
|
173
|
+
const target = resolve(draftDirAbs, c.path);
|
|
174
|
+
return existsSync(target) ? null : "broken-relative";
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
const REASON = {
|
|
178
|
+
"malformed-http": {
|
|
179
|
+
id: "station-links-malformed",
|
|
180
|
+
severity: "fail",
|
|
181
|
+
text: (url) => `link "${url}" is not a well-formed http/https URL (no host)`,
|
|
182
|
+
fix: "Fix the URL to include a scheme (http:// or https://) and a real host.",
|
|
183
|
+
},
|
|
184
|
+
"malformed-mailto": {
|
|
185
|
+
id: "station-links-malformed",
|
|
186
|
+
severity: "fail",
|
|
187
|
+
text: (url) => `link "${url}" is a mailto: link with no address`,
|
|
188
|
+
fix: "Add a real address after mailto:, or remove the link.",
|
|
189
|
+
},
|
|
190
|
+
"bad-scheme": {
|
|
191
|
+
id: "station-links-bad-scheme",
|
|
192
|
+
severity: "fail",
|
|
193
|
+
text: (url) => `link "${url}" uses a scheme that is not http, https or mailto`,
|
|
194
|
+
fix: "Use an http(s):// URL, a mailto: link, or a path relative to the draft.",
|
|
195
|
+
},
|
|
196
|
+
"root-relative": {
|
|
197
|
+
id: "station-links-root-relative",
|
|
198
|
+
severity: "warn",
|
|
199
|
+
text: (url) => `link "${url}" is site-root-relative and cannot be resolved against a file on disk`,
|
|
200
|
+
fix: "Point the link at a path relative to the draft, or use a full URL, if it must be checked.",
|
|
201
|
+
},
|
|
202
|
+
"broken-relative": {
|
|
203
|
+
id: "station-links-broken-relative",
|
|
204
|
+
severity: "fail",
|
|
205
|
+
text: (url) => `relative link "${url}" does not resolve to a file next to the draft`,
|
|
206
|
+
fix: "Fix the path, or add the file the link points at.",
|
|
207
|
+
},
|
|
208
|
+
};
|
|
209
|
+
|
|
210
|
+
function reasonFinding(reason, url, line) {
|
|
211
|
+
const r = REASON[reason];
|
|
212
|
+
return { station: name, id: r.id, severity: r.severity, line, message: r.text(truncate(url, 80)), fix: r.fix };
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
function undefinedReferenceFinding(label, line) {
|
|
216
|
+
const shown = truncate(label, 80);
|
|
217
|
+
return {
|
|
218
|
+
station: name,
|
|
219
|
+
id: "station-links-undefined-reference",
|
|
220
|
+
severity: "fail",
|
|
221
|
+
line,
|
|
222
|
+
message: `reference "${shown}" has no matching "[${shown}]: url" definition`,
|
|
223
|
+
fix: `Add a "[${shown}]: <url>" definition, or fix the reference to match a label that already has one.`,
|
|
224
|
+
};
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
export function run(spec, draft) {
|
|
228
|
+
const draftDirAbs = dirname(resolve(draft.path));
|
|
229
|
+
|
|
230
|
+
// Masking pipeline: code first, then each link form in turn, each pass working on the text the
|
|
231
|
+
// previous pass left behind, so nothing is ever matched twice by a later, looser pattern (a
|
|
232
|
+
// reference definition's own brackets are not a shortcut reference; the leftover "[text]" half
|
|
233
|
+
// of a masked-out "[text][ref]" pair is not itself a shortcut reference; a URL already read as
|
|
234
|
+
// part of a Markdown link is not also a bare URL). Every span carries its ORIGINAL start offset
|
|
235
|
+
// throughout, since maskRanges never shifts anything, only blanks it, so draft.text and lineAt
|
|
236
|
+
// stay valid for every one of them regardless of how many passes it survived.
|
|
237
|
+
let working = maskCode(draft.text);
|
|
238
|
+
|
|
239
|
+
const mdLinks = markdownLinks(working);
|
|
240
|
+
working = maskRanges(working, mdLinks);
|
|
241
|
+
|
|
242
|
+
const { defs, spans: defSpans } = definitions(working);
|
|
243
|
+
working = maskRanges(working, defSpans);
|
|
244
|
+
|
|
245
|
+
const fullRefs = fullReferences(working);
|
|
246
|
+
working = maskRanges(working, fullRefs);
|
|
247
|
+
|
|
248
|
+
const shortcutRefs = shortcutReferences(working).filter((r) => defs.has(normalizeLabel(r.label)));
|
|
249
|
+
working = maskRanges(working, shortcutRefs);
|
|
250
|
+
|
|
251
|
+
const bare = bareUrls(working);
|
|
252
|
+
|
|
253
|
+
const entries = [];
|
|
254
|
+
|
|
255
|
+
for (const { start, url } of [...mdLinks, ...bare]) {
|
|
256
|
+
const reason = checkUrl(url, draftDirAbs);
|
|
257
|
+
if (reason) entries.push({ start, finding: reasonFinding(reason, url, lineAt(draft.text, start)) });
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
for (const { start, label } of [...fullRefs, ...shortcutRefs]) {
|
|
261
|
+
const line = lineAt(draft.text, start);
|
|
262
|
+
const def = defs.get(normalizeLabel(label));
|
|
263
|
+
if (!def) {
|
|
264
|
+
entries.push({ start, finding: undefinedReferenceFinding(label, line) });
|
|
265
|
+
continue;
|
|
266
|
+
}
|
|
267
|
+
const reason = checkUrl(def.url, draftDirAbs);
|
|
268
|
+
if (reason) entries.push({ start, finding: reasonFinding(reason, def.url, line) });
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
entries.sort((a, b) => a.start - b.start);
|
|
272
|
+
const findings = entries.map((e) => e.finding);
|
|
273
|
+
const status = findings.some((f) => f.severity === "fail") ? "fail" : "pass";
|
|
274
|
+
return { station: name, status, findings };
|
|
275
|
+
}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
// Station "private" (hyperspec 0.6). No run of 8 or more consecutive words from
|
|
2
|
+
// any `private` segment of a marked material may appear in the draft. Pure and deterministic: a
|
|
3
|
+
// word-level match over normalized text, no model call and no judgment about what would count as
|
|
4
|
+
// a paraphrase (a paraphrase is not caught, by design; a verbatim run is).
|
|
5
|
+
//
|
|
6
|
+
// Normalization, applied the same way to the draft and to every private segment: split into words
|
|
7
|
+
// with dna.mjs's wordsOf (already lowercased, the one word definition every station shares), then
|
|
8
|
+
// drop apostrophes from each word, so case, punctuation and whitespace never decide a match and
|
|
9
|
+
// "don't", "don’t" and "dont" are one word. Fenced code and inline code are masked out of the
|
|
10
|
+
// draft first (util.mjs's maskCode), like every other station that reads prose.
|
|
11
|
+
//
|
|
12
|
+
// A private segment of 8 or more words fails when any 8-word window of it appears in the draft. One
|
|
13
|
+
// of 4 to 7 words is checked as a whole: all of its words, in order, anywhere in the draft. One
|
|
14
|
+
// under 4 words is not checked at all: two or three words ("Yes, Tuesday.") match
|
|
15
|
+
// ordinary prose, so checking them would fail drafts that leak nothing. Skipping silently would
|
|
16
|
+
// hide that a private passage went unchecked, so the station reports how many it skipped as ONE
|
|
17
|
+
// warning, station-private-short-skipped, carrying the count and never the text (the text is
|
|
18
|
+
// the private part). Each leak is reported once, as the longest run the segment and the draft share from where
|
|
19
|
+
// the match starts, so a whole pasted paragraph is one finding rather than one per window; a
|
|
20
|
+
// segment that leaks in two separate places is two findings. Every finding names the material, the
|
|
21
|
+
// segment and the leaked run (normalized words, at most 80 characters), with the draft line where
|
|
22
|
+
// the run starts.
|
|
23
|
+
|
|
24
|
+
import { wordsOf } from "../dna.mjs";
|
|
25
|
+
import { lineAt, maskCode, markedSegments, truncate } from "./util.mjs";
|
|
26
|
+
|
|
27
|
+
export const name = "private";
|
|
28
|
+
|
|
29
|
+
const WINDOW = 8;
|
|
30
|
+
const MIN_CHECKED = 4;
|
|
31
|
+
// dna.mjs's WORD_RE, repeated here only because wordsOf returns words without their offsets and a
|
|
32
|
+
// finding needs the line a leak starts on; the segment side goes through wordsOf itself.
|
|
33
|
+
const WORD_RE = /[\p{L}\p{N}'’]+/gu;
|
|
34
|
+
const stripApostrophes = (w) => w.replace(/['’]/g, "");
|
|
35
|
+
|
|
36
|
+
// The draft's words, normalized, each with the offset it starts at (for the finding's line).
|
|
37
|
+
function draftWords(text) {
|
|
38
|
+
const out = [];
|
|
39
|
+
WORD_RE.lastIndex = 0;
|
|
40
|
+
let m;
|
|
41
|
+
while ((m = WORD_RE.exec(text))) {
|
|
42
|
+
const w = stripApostrophes(m[0].toLowerCase());
|
|
43
|
+
if (w) out.push({ w, at: m.index });
|
|
44
|
+
}
|
|
45
|
+
return out;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const segmentWords = (text) => wordsOf(text).map(stripApostrophes).filter(Boolean);
|
|
49
|
+
|
|
50
|
+
// The first index in `words` (draft word objects) where `seq` (strings) appears contiguously, or -1.
|
|
51
|
+
function findSequence(words, seq) {
|
|
52
|
+
outer: for (let i = 0; i + seq.length <= words.length; i++) {
|
|
53
|
+
for (let k = 0; k < seq.length; k++) if (words[i + k].w !== seq[k]) continue outer;
|
|
54
|
+
return i;
|
|
55
|
+
}
|
|
56
|
+
return -1;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function run(spec, draft, ctx) {
|
|
60
|
+
const words = draftWords(maskCode(draft.text));
|
|
61
|
+
// First draft position of every 8-word window, so each private window is one map lookup.
|
|
62
|
+
const windows = new Map();
|
|
63
|
+
for (let i = 0; i + WINDOW <= words.length; i++) {
|
|
64
|
+
const key = words.slice(i, i + WINDOW).map((x) => x.w).join(" ");
|
|
65
|
+
if (!windows.has(key)) windows.set(key, i);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const findings = [];
|
|
69
|
+
let skippedShort = 0;
|
|
70
|
+
const leak = (material, segId, run, at) => findings.push({
|
|
71
|
+
station: name,
|
|
72
|
+
id: "station-private-leak",
|
|
73
|
+
severity: "fail",
|
|
74
|
+
line: lineAt(draft.text, at),
|
|
75
|
+
message: `material ${material}, segment ${segId} is private, and the draft repeats it: "${truncate(run.join(" "), 80)}"`,
|
|
76
|
+
fix: "Rewrite the passage in words that do not repeat the private material, or relabel the segment if it is not private.",
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
for (const { material, segments } of markedSegments(spec, ctx)) {
|
|
80
|
+
for (const seg of segments) {
|
|
81
|
+
if (seg.label !== "private" || typeof seg.text !== "string") continue;
|
|
82
|
+
const segId = typeof seg.id === "string" && seg.id.trim() ? seg.id.trim() : "(no id)";
|
|
83
|
+
const tokens = segmentWords(seg.text);
|
|
84
|
+
if (!tokens.length) continue;
|
|
85
|
+
if (tokens.length < MIN_CHECKED) { skippedShort++; continue; }
|
|
86
|
+
|
|
87
|
+
if (tokens.length < WINDOW) {
|
|
88
|
+
const at = findSequence(words, tokens);
|
|
89
|
+
if (at !== -1) leak(material, segId, tokens, words[at].at);
|
|
90
|
+
continue;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
let i = 0;
|
|
94
|
+
while (i + WINDOW <= tokens.length) {
|
|
95
|
+
const start = windows.get(tokens.slice(i, i + WINDOW).join(" "));
|
|
96
|
+
if (start === undefined) { i++; continue; }
|
|
97
|
+
// Extend the shared run as far as the segment and the draft keep agreeing.
|
|
98
|
+
let len = WINDOW;
|
|
99
|
+
while (i + len < tokens.length && start + len < words.length && tokens[i + len] === words[start + len].w) len++;
|
|
100
|
+
leak(material, segId, tokens.slice(i, i + len), words[start].at);
|
|
101
|
+
i += len;
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
if (skippedShort) {
|
|
107
|
+
findings.push({
|
|
108
|
+
station: name,
|
|
109
|
+
id: "station-private-short-skipped",
|
|
110
|
+
severity: "warn",
|
|
111
|
+
message: `${skippedShort} private segment${skippedShort === 1 ? " is" : "s are"} under ${MIN_CHECKED} words and ${skippedShort === 1 ? "was" : "were"} not checked`,
|
|
112
|
+
fix: `Check the draft against ${skippedShort === 1 ? "it" : "them"} by hand, or widen the private segment to ${MIN_CHECKED} or more words.`,
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return { station: name, status: findings.some((x) => x.severity === "fail") ? "fail" : "pass", findings };
|
|
117
|
+
}
|