@supersuit/hyperspec 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,275 @@
1
+ // Station "sequence" (hyperspec 0.8). Every other station checks one piece against its spec. A
2
+ // work read in order (a course, a primer, a textbook, a book of lessons) makes promises ACROSS its
3
+ // pieces: lesson 5 is written for someone who has read lessons 1 to 4 and nothing else. This
4
+ // station holds the draft to those promises, and runs only when the spec declares
5
+ // writing.form.sequence.
6
+ //
7
+ // A unit is one lesson (or chapter, or whatever sequence.unit names): an ATX heading "<unit> <n>",
8
+ // optionally followed by ":" or "." and a title, running until the next unit heading, the next
9
+ // heading of its own level or above (a part's heading, whose introduction belongs to no lesson), or
10
+ // the end of the file it sits in. Headings inside fenced code are not headings.
11
+ //
12
+ // The guards, every one deterministic:
13
+ // sections every unit carries each of sequence.sections, as a "**Label:**" line or a heading
14
+ // defined every term is defined in exactly one unit's terms section
15
+ // order no unit uses a term before the unit that defines it; code is not prose, the terms
16
+ // section is not a use, and neither is the unit's closing teaser (a "**<teaser>" line
17
+ // and everything after it). Words in audience.knows and sequence.knows are exempt:
18
+ // listing a word there is a decision that the reader already has it
19
+ // outline each unit defines every term the outline promises for it
20
+ // numbering unit numbers increase through the work
21
+ // pointers (warning) a "<unit> N" pointer to a later unit is reported, so it stays a pointer
22
+ // and never becomes a dependency
23
+ //
24
+ // A use is a whole-word, case-insensitive match, where a hyphen is part of a word: "context-aware"
25
+ // does not use "context", and "skills" does not use "skill". The station reads the outline from
26
+ // disk, resolved against the spec's folder, the way claims reads its ledger. It never reads a
27
+ // part's text for meaning: a term used in a sense other than its definition still counts as a use.
28
+
29
+ import { readFileSync } from "node:fs";
30
+ import { resolve } from "node:path";
31
+ import { str } from "../placeholder.mjs";
32
+ import { maskCode } from "./util.mjs";
33
+
34
+ export const name = "sequence";
35
+
36
+ export const DEFAULTS = Object.freeze({
37
+ unit: "Lesson",
38
+ sections: Object.freeze(["After this lesson you can", "New terms", "Try this"]),
39
+ terms_section: "New terms",
40
+ teaser: "Next,",
41
+ });
42
+
43
+ const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
44
+ const list = (v) => (Array.isArray(v) ? v : []);
45
+ const ATX = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
46
+
47
+ // A term as the station compares it: lower case, no code or emphasis marks, no parenthetical, one
48
+ // space between words.
49
+ export const normTerm = (s) => String(s).toLowerCase().replace(/[`*]/g, "").replace(/\s*\(.*?\)\s*/g, " ").replace(/\s+/g, " ").trim();
50
+
51
+ // The sequence block with every default filled in. null when the spec declares none.
52
+ export function sequenceOf(spec) {
53
+ const raw = spec?.data?.writing?.form?.sequence;
54
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
55
+ const sections = list(raw.sections).map(str).filter(Boolean);
56
+ const knows = [...list(raw.knows), ...list(spec?.data?.writing?.audience?.knows)].map(str).filter(Boolean).map(normTerm);
57
+ return {
58
+ unit: str(raw.unit) || DEFAULTS.unit,
59
+ sections: sections.length ? sections : [...DEFAULTS.sections],
60
+ termsSection: str(raw.terms_section) || DEFAULTS.terms_section,
61
+ teaser: str(raw.teaser) || DEFAULTS.teaser,
62
+ outline: str(raw.outline) || null,
63
+ knows: new Set(knows),
64
+ };
65
+ }
66
+
67
+ // Every unit in the draft, in document order: { n, title, line, startLine, text, lines }, where
68
+ // line is the heading's 1-based line, and text is everything after the heading up to where the
69
+ // unit ends (see the header). startLine is the line text begins on. A draft assembled from several
70
+ // files (draft.sources, see src/sequence-draft.mjs) ends a unit at its file's end too.
71
+ export function parseUnits(draft, { unit }) {
72
+ const lines = draft.text.split("\n");
73
+ const masked = maskCode(draft.text).split("\n");
74
+ const unitRe = new RegExp(`^${escapeRe(unit)}[ \\t]+(\\d+)\\b[ \\t]*[:.]?[ \\t]*(.*)$`, "i");
75
+ const fileEnds = new Set(list(draft.sources).map((s) => s.startLine + s.lineCount - 1));
76
+ const units = [];
77
+ let open = null;
78
+ const close = (endLine) => {
79
+ if (!open) return;
80
+ const body = lines.slice(open.line, endLine);
81
+ units.push({ n: open.n, title: open.title, line: open.line, startLine: open.line + 1, lines: body, text: body.join("\n") });
82
+ open = null;
83
+ };
84
+ for (let i = 0; i < lines.length; i++) {
85
+ const h = ATX.exec(masked[i].replace(/\r$/, ""));
86
+ if (h) {
87
+ const level = h[1].length;
88
+ const m = unitRe.exec((h[2] ?? "").replace(/(^|[ \t]+)#+[ \t]*$/, "").trim());
89
+ if (m) {
90
+ close(i);
91
+ open = { n: Number(m[1]), title: m[2].trim(), line: i + 1, level };
92
+ } else if (open && level <= open.level) close(i);
93
+ }
94
+ if (open && fileEnds.has(i + 1)) close(i + 1);
95
+ }
96
+ close(lines.length);
97
+ return units;
98
+ }
99
+
100
+ // A section label line: "**Label:**", "**Label**:" or "**Label**" at the start of a line (text may
101
+ // follow), or a heading whose text is the label, with or without a closing colon. Case-insensitive.
102
+ function labelRe(label) {
103
+ const l = escapeRe(label);
104
+ return new RegExp(`^(?:[ \\t]*\\*\\*${l}(?::\\*\\*|\\*\\*:?)|[ \\t]{0,3}#{1,6}[ \\t]+${l}:?[ \\t]*#*[ \\t]*$)`, "i");
105
+ }
106
+
107
+ // The index (into unit.lines) of the first line carrying `label`, or -1. Code is masked first, so a
108
+ // label shown in an example is not the unit's own.
109
+ function labelLine(unit, label) {
110
+ const masked = maskCode(unit.text).split("\n");
111
+ const re = labelRe(label);
112
+ return masked.findIndex((l) => re.test(l.replace(/\r$/, "")));
113
+ }
114
+
115
+ // The [start, end) range of unit.lines holding the terms section: its label line, then the list
116
+ // under it, up to the first blank line after the list begins. null when the unit has no such section.
117
+ function termsRange(unit, termsSection) {
118
+ const at = labelLine(unit, termsSection);
119
+ if (at < 0) return null;
120
+ let end = at + 1;
121
+ while (end < unit.lines.length && !unit.lines[end].trim()) end++; // blank lines before the list
122
+ while (end < unit.lines.length && unit.lines[end].trim()) end++;
123
+ return { start: at, end };
124
+ }
125
+
126
+ // The terms a unit's terms section defines, normalized, each once, in order. A list item that
127
+ // reminds the reader of an earlier term, "(from Lesson 3)", defines nothing. Several bold terms
128
+ // before the item's first ":**" are all defined, so "- **A**, **B** and **C:** ..." defines three.
129
+ export function newTerms(unit, { unit: unitWord, termsSection }) {
130
+ const range = termsRange(unit, termsSection);
131
+ if (!range) return [];
132
+ const reminder = new RegExp(`\\(\\s*from\\s+${escapeRe(unitWord)}\\s+\\d+\\s*\\)`, "i");
133
+ const out = [];
134
+ for (const raw of unit.lines.slice(range.start + 1, range.end)) {
135
+ const line = raw.replace(/^[ \t]*[-*+][ \t]+/, "");
136
+ if (reminder.test(line)) continue;
137
+ let head;
138
+ if (line.includes(":**")) head = `${line.split(":**")[0]}**`;
139
+ else if (line.includes("**:")) head = line.split("**:")[0] + "**";
140
+ else head = (line.match(/\*\*(.+?)\*\*/) || [""])[0];
141
+ for (const m of head.matchAll(/\*\*(.+?)\*\*/g)) {
142
+ const t = normTerm(m[1]);
143
+ if (t && !out.includes(t)) out.push(t);
144
+ }
145
+ }
146
+ return out;
147
+ }
148
+
149
+ // The terms an outline promises per unit: { n: [term, ...] }. An outline item is a numbered list
150
+ // line, "N. ...", running until the next numbered item at its indent or less, or a heading; its
151
+ // promise is an emphasized "*Terms: a, b, c.*" anywhere in the item.
152
+ export function outlineTerms(text) {
153
+ const out = {};
154
+ const lines = String(text).replace(/\r/g, "").split("\n");
155
+ let cur = null;
156
+ const flush = () => {
157
+ if (!cur) return;
158
+ const m = cur.text.match(/\*Terms:\s*([^*]*?)\.?\s*\*/);
159
+ if (m) out[cur.n] = m[1].split(/,\s*/).map(normTerm).filter(Boolean);
160
+ cur = null;
161
+ };
162
+ for (const line of lines) {
163
+ const item = /^([ \t]*)(\d+)\.[ \t]/.exec(line);
164
+ if (item && (!cur || item[1].length <= cur.indent)) {
165
+ flush();
166
+ cur = { n: Number(item[2]), indent: item[1].length, text: line };
167
+ } else if (/^ {0,3}#{1,6}[ \t]/.test(line)) flush();
168
+ else if (cur) cur.text += `\n${line}`;
169
+ }
170
+ flush();
171
+ return out;
172
+ }
173
+
174
+ // The unit's prose as the order and pointer guards read it: code masked and the closing teaser
175
+ // (from its "**<teaser>" line to the unit's end) blanked, lines kept so a position still maps to a
176
+ // line. withTerms false blanks the terms section too, for the order guard: a unit's terms section
177
+ // is where it defines words, and a reminder there names an earlier unit's word on purpose. The
178
+ // pointer guard keeps it, since "(Lesson 18 covers this)" in a definition is still a pointer.
179
+ function proseOf(unit, seq, { withTerms }) {
180
+ const lines = maskCode(unit.text).split("\n");
181
+ const range = withTerms ? null : termsRange(unit, seq.termsSection);
182
+ if (range) for (let i = range.start; i < range.end; i++) lines[i] = "";
183
+ const teaserRe = new RegExp(`^[ \\t]*\\*\\*${escapeRe(seq.teaser)}`, "i");
184
+ const t = lines.findIndex((l) => teaserRe.test(l));
185
+ if (t >= 0) for (let i = t; i < lines.length; i++) lines[i] = "";
186
+ return lines;
187
+ }
188
+
189
+ // The 1-based draft line of the first whole-word use of `term` in the unit's prose, or 0.
190
+ function firstUse(unit, prose, term) {
191
+ const re = new RegExp(`(^|[^a-z0-9-])${escapeRe(term)}([^a-z0-9-]|$)`, "i");
192
+ const i = prose.findIndex((l) => re.test(l));
193
+ return i < 0 ? 0 : unit.startLine + i;
194
+ }
195
+
196
+ // One finding, in the shape every station returns; line only when there is one to point at.
197
+ const finding = ({ id, severity, message, fix, line }) => ({ station: name, id, severity, ...(line ? { line } : {}), message, fix });
198
+
199
+ export function run(spec, draft) {
200
+ const seq = sequenceOf(spec);
201
+ if (!seq) return { station: name, status: "skip", findings: [], reason: "the spec declares no writing.form.sequence" };
202
+ const U = seq.unit;
203
+ const findings = [];
204
+ const units = parseUnits(draft, seq);
205
+ if (!units.length) {
206
+ findings.push(finding({ id: "station-sequence-no-units", severity: "fail", message: `no "${U} <n>" heading in the draft`, fix: `Head each ${U.toLowerCase()} "## ${U} 1: <title>", or set writing.form.sequence.unit to the word the headings use.` }));
207
+ return { station: name, status: "fail", findings };
208
+ }
209
+
210
+ // numbering: each unit's number is greater than the one before it.
211
+ for (let i = 1; i < units.length; i++) {
212
+ if (units[i].n <= units[i - 1].n) {
213
+ findings.push(finding({ id: "station-sequence-numbering", severity: "fail", message: `${U} ${units[i].n} comes after ${U} ${units[i - 1].n}`, fix: `Number the ${U.toLowerCase()}s in reading order, or reorder writing.form.sequence.files.`, line: units[i].line }));
214
+ }
215
+ }
216
+
217
+ // sections, and where each term is defined.
218
+ const definedAt = new Map();
219
+ for (const u of units) {
220
+ for (const s of seq.sections) {
221
+ if (labelLine(u, s) < 0) findings.push(finding({ id: "station-sequence-missing-section", severity: "fail", message: `${U} ${u.n} has no "${s}" section`, fix: `Add a "**${s}:**" line (or a "${s}" heading) to ${U} ${u.n}.`, line: u.line }));
222
+ }
223
+ for (const t of newTerms(u, seq)) {
224
+ const first = definedAt.get(t);
225
+ if (first && first !== u) findings.push(finding({ id: "station-sequence-defined-twice", severity: "fail", message: `"${t}" is defined in ${U} ${first.n} and again in ${U} ${u.n}`, fix: `Define "${t}" once; later ${U.toLowerCase()}s remind the reader with "- **${t}** (from ${U} ${first.n}): ...".`, line: u.line }));
226
+ else if (!first) definedAt.set(t, u);
227
+ }
228
+ }
229
+
230
+ // order: no unit before the defining one uses the term.
231
+ const prose = new Map(units.map((u) => [u, proseOf(u, seq, { withTerms: false })]));
232
+ for (const [t, at] of definedAt) {
233
+ if (seq.knows.has(t)) continue;
234
+ for (const u of units) {
235
+ if (u === at) break;
236
+ const line = firstUse(u, prose.get(u), t);
237
+ if (line) findings.push(finding({ id: "station-sequence-used-before-defined", severity: "fail", message: `${U} ${u.n} uses "${t}" before ${U} ${at.n} defines it`, fix: `Rewrite ${U} ${u.n} without "${t}", define it earlier, or add it to writing.form.sequence.knows if the reader already has the word.`, line: line }));
238
+ }
239
+ }
240
+
241
+ // outline: each unit defines what the outline promises for it. A unit the outline promises that
242
+ // the draft does not hold yet is not a finding: a work is checked while it is being written.
243
+ if (seq.outline) {
244
+ let text = null;
245
+ try { text = readFileSync(resolve(spec?.dir || ".", seq.outline), "utf8"); } catch { /* reported below */ }
246
+ if (text === null) {
247
+ findings.push(finding({ id: "station-sequence-outline-unreadable", severity: "fail", message: `the outline "${seq.outline}" cannot be read`, fix: "Point writing.form.sequence.outline at the outline file, relative to the spec." }));
248
+ } else {
249
+ const promised = outlineTerms(text);
250
+ for (const u of units) {
251
+ const have = newTerms(u, seq);
252
+ for (const t of promised[u.n] ?? []) {
253
+ if (!have.includes(t)) findings.push(finding({ id: "station-sequence-outline-unkept", severity: "fail", message: `the outline promises "${t}" in ${U} ${u.n}, and ${U} ${u.n} does not define it`, fix: `Define "${t}" in ${U} ${u.n}'s ${seq.termsSection}, or change the outline.`, line: u.line }));
254
+ }
255
+ }
256
+ }
257
+ }
258
+
259
+ // pointers: a mention of a later unit, once per pair.
260
+ const pointerRe = new RegExp(`\\b${escapeRe(U)}\\s+(\\d+)`, "gi");
261
+ for (const u of units) {
262
+ const seen = new Set();
263
+ proseOf(u, seq, { withTerms: true }).forEach((l, i) => {
264
+ for (const m of l.matchAll(pointerRe)) {
265
+ const n = Number(m[1]);
266
+ if (n <= u.n || seen.has(n)) continue;
267
+ seen.add(n);
268
+ findings.push(finding({ id: "station-sequence-forward-pointer", severity: "warn", message: `${U} ${u.n} points forward to ${U} ${n}`, fix: `Keep it a pointer ("more in ${U} ${n}"): ${U} ${u.n} must make sense to a reader who has not read ${U} ${n}.`, line: u.startLine + i }));
269
+ }
270
+ });
271
+ }
272
+
273
+ const status = findings.some((f) => f.severity === "fail") ? "fail" : "pass";
274
+ return { station: name, status, findings };
275
+ }
@@ -23,6 +23,7 @@ import { basename, dirname, join, sep } from "node:path";
23
23
  import { str } from "./placeholder.mjs";
24
24
  import { readSegments } from "./segments.mjs";
25
25
  import { readScope, isGoldenFileName, measureFeatures, featuresText, DNA_FORMAT } from "./dna.mjs";
26
+ import { expandEntry } from "./sequence-draft.mjs";
26
27
 
27
28
  const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
28
29
  const list = (v) => (Array.isArray(v) ? v : []);
@@ -446,6 +447,39 @@ function formFields(raw, d, here, idPrefix) {
446
447
  if (minOk && maxOk && min > max) out.push(f(1, `${idPrefix}-length-range`, "fail", `writing.form.length.min (${min}) is greater than length.max (${max})`, "Set min to no more than max."));
447
448
  if (!str(length.unit)) out.push(f(1, `${idPrefix}-length-unit`, "fail", "writing.form.length has no unit", "Add length.unit:, e.g. words."));
448
449
  if (!list(raw.required_parts).some((x) => str(x))) out.push(f(1, `${idPrefix}-required-parts`, "fail", "writing.form has no required_parts", "List at least one required part."));
450
+ if ("sequence" in raw) out.push(...sequenceFields(raw.sequence, here, `${idPrefix}-sequence`));
451
+ return out;
452
+ }
453
+
454
+ // writing.form.sequence, optional: a work read in order (see WRITING.md, Sequential works). Every
455
+ // key is optional, and each one present has to be usable, because the sequence station reads it
456
+ // as given and a hollow value would switch a guard off without saying so.
457
+ function sequenceFields(seq, here, idPrefix) {
458
+ if (!isObj(seq)) return [f(1, idPrefix, "fail", "writing.form.sequence is not a map", "Write sequence: as a map of the keys WRITING.md lists, one per line.")];
459
+ const out = [];
460
+ for (const key of ["unit", "terms_section", "teaser"]) {
461
+ if (key in seq && !str(seq[key])) out.push(f(1, `${idPrefix}-${key.replace("_", "-")}`, "fail", `writing.form.sequence.${key} is empty or a placeholder`, `Give ${key} a real value, or delete it for the default.`));
462
+ }
463
+ const strings = (key, fix) => {
464
+ if (!(key in seq)) return null;
465
+ const items = list(seq[key]);
466
+ if (!Array.isArray(seq[key]) || !items.length || items.some((x) => typeof x !== "string" || !str(x))) {
467
+ out.push(f(1, `${idPrefix}-${key}`, "fail", `writing.form.sequence.${key} is not a list of real entries`, fix));
468
+ return null;
469
+ }
470
+ return items;
471
+ };
472
+ strings("sections", "List every section a unit carries, or delete sections: for the default three.");
473
+ strings("knows", "List the words a reader already has, one per entry, or delete knows:.");
474
+ const files = strings("files", "List the work's files in reading order, relative to the spec; a * in a file name matches.");
475
+ for (const entry of files ?? []) {
476
+ if (!expandEntry(here("."), entry).length) out.push(f(6, `${idPrefix}-files-missing`, "fail", `writing.form.sequence.files entry "${entry}" matches no file`, "Fix the path (relative to the spec), or remove the entry."));
477
+ }
478
+ if ("outline" in seq) {
479
+ const p = str(seq.outline);
480
+ if (!p) out.push(f(1, `${idPrefix}-outline`, "fail", "writing.form.sequence.outline is empty or a placeholder", "Name the outline file, or delete outline:."));
481
+ else out.push(...pathFindings(here, p, `${idPrefix}-outline`, "writing.form.sequence.outline", "Point outline: at the outline file, relative to the spec."));
482
+ }
449
483
  return out;
450
484
  }
451
485