@supersuit/hyperspec 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/README.md +71 -8
- package/SPEC.md +2 -2
- package/WRITING.md +680 -21
- package/bin/hyperspec.mjs +188 -4
- package/examples/writing/course/claims.jsonl +0 -0
- package/examples/writing/course/goldens/lesson.md +1 -0
- package/examples/writing/course/materials/brief.md +9 -0
- package/examples/writing/course/materials/brief.md.segments.jsonl +6 -0
- package/examples/writing/course/outline.md +11 -0
- package/examples/writing/course/part-1.md +47 -0
- package/examples/writing/course/part-2.md +40 -0
- package/examples/writing/course/runs.jsonl +0 -0
- package/examples/writing/course.hyperspec.md +205 -0
- package/examples/writing/essay/judge/doctor.packet.json +108 -0
- package/examples/writing/essay/judge/lineup.packet.json +64 -0
- package/examples/writing/essay/judge/persona.packet.json +73 -0
- package/examples/writing/essay/judge/reader.packet.json +93 -0
- package/examples/writing/essay/learn/first-draft.md +84 -0
- package/examples/writing/essay/learn/learn.packet.json +106 -0
- package/examples/writing/essay/sample-verdicts/doctor.verdict.json +43 -0
- package/examples/writing/essay/sample-verdicts/learn.verdict.json +30 -0
- package/examples/writing/essay/sample-verdicts/lineup.verdict.json +6 -0
- package/examples/writing/essay/sample-verdicts/persona.verdict.json +4 -0
- package/examples/writing/essay/sample-verdicts/reader.verdict.json +7 -0
- package/examples/writing/essay.hyperspec.md +6 -1
- package/examples/writing/story/judge/attribution.packet.json +194 -0
- package/examples/writing/story/judge/doctor.packet.json +108 -0
- package/examples/writing/story/judge/knowledge.packet.json +77 -0
- package/examples/writing/story/judge/persona.packet.json +73 -0
- package/examples/writing/story/judge/reader.packet.json +94 -0
- package/examples/writing/story/sample-verdicts/attribution.verdict.json +81 -0
- package/examples/writing/story/sample-verdicts/doctor.verdict.json +43 -0
- package/examples/writing/story/sample-verdicts/knowledge.verdict.json +4 -0
- package/examples/writing/story/sample-verdicts/persona.verdict.json +20 -0
- package/examples/writing/story/sample-verdicts/reader.verdict.json +16 -0
- package/examples/writing/story.hyperspec.md +7 -3
- package/package.json +1 -1
- package/src/check.mjs +96 -132
- package/src/draft.mjs +26 -0
- package/src/judge.mjs +386 -0
- package/src/judges/attribution.mjs +360 -0
- package/src/judges/doctor.mjs +126 -0
- package/src/judges/index.mjs +31 -0
- package/src/judges/knowledge.mjs +111 -0
- package/src/judges/lineup.mjs +272 -0
- package/src/judges/persona.mjs +137 -0
- package/src/judges/reader.mjs +111 -0
- package/src/learn.mjs +422 -0
- package/src/ledger.mjs +108 -0
- package/src/sentences.mjs +81 -0
- package/src/sequence-draft.mjs +75 -0
- package/src/stations/claims.mjs +44 -39
- package/src/stations/index.mjs +3 -1
- package/src/stations/links.mjs +11 -3
- package/src/stations/quotes.mjs +6 -4
- package/src/stations/sequence.mjs +275 -0
- package/src/writing-fields.mjs +34 -0
- package/src/writing.mjs +1 -1
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
// A work read in order often lives in several files (one per part, one per lesson). When a writing
|
|
2
|
+
// spec lists them as writing.form.sequence.files, `hyperspec check <spec>` needs no --draft: the
|
|
3
|
+
// draft is those files, joined in order. This module owns which files that is and how they join, so
|
|
4
|
+
// lint (every entry must match a file) and check (the draft) agree on one reading.
|
|
5
|
+
|
|
6
|
+
import { readdirSync, readFileSync, statSync } from "node:fs";
|
|
7
|
+
import { dirname, join, posix, resolve } from "node:path";
|
|
8
|
+
import { sha256 } from "./hash.mjs";
|
|
9
|
+
import { str } from "./placeholder.mjs";
|
|
10
|
+
import { splitLines } from "./draft.mjs";
|
|
11
|
+
|
|
12
|
+
const list = (v) => (Array.isArray(v) ? v : []);
|
|
13
|
+
const byNumber = (a, b) => a.localeCompare(b, "en", { numeric: true });
|
|
14
|
+
|
|
15
|
+
// The declared entries, as written: writing.form.sequence.files, strings only.
|
|
16
|
+
export function sequenceFilesDecl(spec) {
|
|
17
|
+
return list(spec?.data?.writing?.form?.sequence?.files).map(str).filter(Boolean);
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
// The files one entry names, relative to the spec's folder and written with "/": the entry itself
|
|
21
|
+
// when it has no "*", or every file in its folder whose name matches, sorted so part-2 comes before
|
|
22
|
+
// part-10. A "*" matches within a file name only; the folder part is taken literally. [] when
|
|
23
|
+
// nothing matches.
|
|
24
|
+
export function expandEntry(specDir, entry) {
|
|
25
|
+
const clean = entry.replace(/\\/g, "/");
|
|
26
|
+
const dir = posix.dirname(clean);
|
|
27
|
+
const base = posix.basename(clean);
|
|
28
|
+
const isFile = (rel) => { try { return statSync(resolve(specDir, rel)).isFile(); } catch { return false; } };
|
|
29
|
+
if (!base.includes("*")) return isFile(clean) ? [clean] : [];
|
|
30
|
+
const re = new RegExp(`^${base.split("*").map((s) => s.replace(/[.+?^${}()|[\]\\]/g, "\\$&")).join("[^/]*")}$`);
|
|
31
|
+
let names = [];
|
|
32
|
+
try { names = readdirSync(resolve(specDir, dir)); } catch { return []; }
|
|
33
|
+
return names.filter((n) => re.test(n)).sort(byNumber).map((n) => (dir === "." ? n : `${dir}/${n}`)).filter(isFile);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// Every file of the sequence in reading order, each once (at its first position).
|
|
37
|
+
export function sequenceFiles(specDir, entries) {
|
|
38
|
+
const out = [];
|
|
39
|
+
for (const e of entries) for (const f of expandEntry(specDir, e)) if (!out.includes(f)) out.push(f);
|
|
40
|
+
return out;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// The draft `check` grades when the spec lists sequence files and no --draft is given:
|
|
44
|
+
// { path, text, lines, sha256, sources }, the same shape src/draft.mjs's readDraft returns plus
|
|
45
|
+
// sources, one per file: { file, at, startLine, lineCount }. file is the path relative to the spec
|
|
46
|
+
// (what a finding names); at resolves from the working directory (what a station opens, such as
|
|
47
|
+
// links resolving a relative link beside the file that holds it). Each file's YAML frontmatter is
|
|
48
|
+
// blanked line for line, so its metadata is not prose and its line numbers stay its own. sha256
|
|
49
|
+
// covers every file's name and bytes. path is the first file's. null when no file matches.
|
|
50
|
+
export function readSequenceDraft(spec, specPathArg) {
|
|
51
|
+
const files = sequenceFiles(spec.dir, sequenceFilesDecl(spec));
|
|
52
|
+
if (!files.length) return null;
|
|
53
|
+
const parts = [];
|
|
54
|
+
const hashed = [];
|
|
55
|
+
const sources = [];
|
|
56
|
+
let startLine = 1;
|
|
57
|
+
for (const file of files) {
|
|
58
|
+
const buf = readFileSync(resolve(spec.dir, file));
|
|
59
|
+
hashed.push(Buffer.from(`${file}\n`), buf);
|
|
60
|
+
let text = buf.toString("utf8").replace(/^\uFEFF/, "");
|
|
61
|
+
text = text.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n/, (fm) => fm.replace(/[^\n]/g, ""));
|
|
62
|
+
if (!text.endsWith("\n")) text += "\n";
|
|
63
|
+
const lineCount = text.split("\n").length - 1;
|
|
64
|
+
sources.push({ file, at: join(dirname(specPathArg), file), startLine, lineCount });
|
|
65
|
+
parts.push(text);
|
|
66
|
+
startLine += lineCount;
|
|
67
|
+
}
|
|
68
|
+
const text = parts.join("");
|
|
69
|
+
return { path: sources[0].at, text, lines: splitLines(text), sha256: sha256(Buffer.concat(hashed)), sources };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// The source holding 1-based draft line `line`, or null.
|
|
73
|
+
export function sourceAt(draft, line) {
|
|
74
|
+
return list(draft?.sources).find((s) => line >= s.startLine && line < s.startLine + s.lineCount) ?? null;
|
|
75
|
+
}
|
package/src/stations/claims.mjs
CHANGED
|
@@ -44,6 +44,44 @@ function firstOccurrence(text, claimText) {
|
|
|
44
44
|
return m ? m.index : -1;
|
|
45
45
|
}
|
|
46
46
|
|
|
47
|
+
// The claims ledger as the claims station reads it, shared with the persona judge
|
|
48
|
+
// (src/judges/persona.mjs), so both see the same claims. { path, missing, lines }: path is
|
|
49
|
+
// writing.sources.ledger as written (null when unset); missing is true when it is set but cannot be
|
|
50
|
+
// read; lines holds every non-blank line, 1-based as n, each either { n, text } (a claim with
|
|
51
|
+
// non-empty text; claim is the parsed object) or { n, problem } naming why it is not one.
|
|
52
|
+
export function readClaimsLedger(spec) {
|
|
53
|
+
const ledgerPath = str(spec?.data?.writing?.sources?.ledger);
|
|
54
|
+
if (!ledgerPath) return { path: null, missing: false, lines: [] };
|
|
55
|
+
let raw;
|
|
56
|
+
try {
|
|
57
|
+
// A leading UTF-8 BOM (written by default by several Windows/Excel-adjacent editors) is not
|
|
58
|
+
// valid JSON leading whitespace, so it must come off before line 1 is parsed, or a genuinely
|
|
59
|
+
// well-formed first line reports as broken JSON for a reason that has nothing to do with its
|
|
60
|
+
// content.
|
|
61
|
+
raw = readFileSync(resolve(spec?.dir || ".", ledgerPath), "utf8").replace(/^\uFEFF/, "");
|
|
62
|
+
} catch {
|
|
63
|
+
return { path: ledgerPath, missing: true, lines: [] };
|
|
64
|
+
}
|
|
65
|
+
const lines = [];
|
|
66
|
+
raw.split("\n").forEach((text, i) => {
|
|
67
|
+
if (text.trim() === "") return;
|
|
68
|
+
const n = i + 1;
|
|
69
|
+
let obj;
|
|
70
|
+
try { obj = JSON.parse(text); } catch { lines.push({ n, problem: "is not valid JSON" }); return; }
|
|
71
|
+
if (!obj || typeof obj !== "object" || Array.isArray(obj)) { lines.push({ n, problem: "is not a JSON object" }); return; }
|
|
72
|
+
const claimText = typeof obj.text === "string" ? obj.text : "";
|
|
73
|
+
if (!claimText.trim()) { lines.push({ n, problem: "has no text" }); return; }
|
|
74
|
+
lines.push({ n, text: claimText, claim: obj });
|
|
75
|
+
});
|
|
76
|
+
return { path: ledgerPath, missing: false, lines };
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const PROBLEM_FIX = {
|
|
80
|
+
"is not valid JSON": "Fix the JSON on that line.",
|
|
81
|
+
"is not a JSON object": 'Each ledger line must be a JSON object: {"text": "...", "source": "..."}.',
|
|
82
|
+
"has no text": "Add text: the claim exactly as it appears in the draft.",
|
|
83
|
+
};
|
|
84
|
+
|
|
47
85
|
export function run(spec, draft) {
|
|
48
86
|
const sources = spec?.data?.writing?.sources ?? {};
|
|
49
87
|
const ledgerPath = str(sources.ledger);
|
|
@@ -57,15 +95,8 @@ export function run(spec, draft) {
|
|
|
57
95
|
return { station: name, status: "skip", findings: [], reason: "writing.sources.ledger is not set" };
|
|
58
96
|
}
|
|
59
97
|
|
|
60
|
-
const
|
|
61
|
-
|
|
62
|
-
try {
|
|
63
|
-
// A leading UTF-8 BOM (written by default by several Windows/Excel-adjacent editors) is not
|
|
64
|
-
// valid JSON leading whitespace, so it must come off before line 1 is parsed, or a genuinely
|
|
65
|
-
// well-formed first line reports as broken JSON for a reason that has nothing to do with its
|
|
66
|
-
// content.
|
|
67
|
-
raw = readFileSync(ledgerAbs, "utf8").replace(/^/, "");
|
|
68
|
-
} catch {
|
|
98
|
+
const ledger = readClaimsLedger(spec);
|
|
99
|
+
if (ledger.missing) {
|
|
69
100
|
return {
|
|
70
101
|
station: name,
|
|
71
102
|
status: "fail",
|
|
@@ -81,41 +112,15 @@ export function run(spec, draft) {
|
|
|
81
112
|
|
|
82
113
|
const findings = [];
|
|
83
114
|
const draftNorm = normalize(draft.text);
|
|
84
|
-
const ledgerLines = raw.split("\n").map((text, i) => ({ n: i + 1, text })).filter((l) => l.text.trim() !== "");
|
|
85
|
-
|
|
86
|
-
for (const { n, text } of ledgerLines) {
|
|
87
|
-
let obj;
|
|
88
|
-
try {
|
|
89
|
-
obj = JSON.parse(text);
|
|
90
|
-
} catch {
|
|
91
|
-
findings.push({
|
|
92
|
-
station: name,
|
|
93
|
-
id: `station-claims-json-line-${n}`,
|
|
94
|
-
severity: "fail",
|
|
95
|
-
message: `writing.sources.ledger "${ledgerPath}" line ${n} is not valid JSON`,
|
|
96
|
-
fix: "Fix the JSON on that line.",
|
|
97
|
-
});
|
|
98
|
-
continue;
|
|
99
|
-
}
|
|
100
|
-
if (!obj || typeof obj !== "object" || Array.isArray(obj)) {
|
|
101
|
-
findings.push({
|
|
102
|
-
station: name,
|
|
103
|
-
id: `station-claims-json-line-${n}`,
|
|
104
|
-
severity: "fail",
|
|
105
|
-
message: `writing.sources.ledger "${ledgerPath}" line ${n} is not a JSON object`,
|
|
106
|
-
fix: 'Each ledger line must be a JSON object: {"text": "...", "source": "..."}.',
|
|
107
|
-
});
|
|
108
|
-
continue;
|
|
109
|
-
}
|
|
110
115
|
|
|
111
|
-
|
|
112
|
-
if (
|
|
116
|
+
for (const { n, problem, text: claimText, claim: obj } of ledger.lines) {
|
|
117
|
+
if (problem) {
|
|
113
118
|
findings.push({
|
|
114
119
|
station: name,
|
|
115
120
|
id: `station-claims-json-line-${n}`,
|
|
116
121
|
severity: "fail",
|
|
117
|
-
message: `writing.sources.ledger "${ledgerPath}" line ${n}
|
|
118
|
-
fix:
|
|
122
|
+
message: `writing.sources.ledger "${ledgerPath}" line ${n} ${problem}`,
|
|
123
|
+
fix: PROBLEM_FIX[problem],
|
|
119
124
|
});
|
|
120
125
|
continue;
|
|
121
126
|
}
|
package/src/stations/index.mjs
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
// src/check.mjs calls; adding a station is adding one file plus one line here, which is the whole
|
|
4
4
|
// point of the registry existing rather than check.mjs importing each station by name itself.
|
|
5
5
|
//
|
|
6
|
-
// The order: form, terms, claims, quotes, private, dna, links. quotes and private share ctx (util.mjs's
|
|
6
|
+
// The order: form, terms, claims, quotes, private, dna, links, sequence. quotes and private share ctx (util.mjs's
|
|
7
7
|
// markedSegments caches the spec's marked materials there), so a check run reads them once.
|
|
8
8
|
|
|
9
9
|
import * as form from "./form.mjs";
|
|
@@ -13,6 +13,7 @@ import * as quotes from "./quotes.mjs";
|
|
|
13
13
|
import * as privateStation from "./private.mjs";
|
|
14
14
|
import * as dna from "./dna.mjs";
|
|
15
15
|
import * as links from "./links.mjs";
|
|
16
|
+
import * as sequence from "./sequence.mjs";
|
|
16
17
|
|
|
17
18
|
export const STATIONS = Object.freeze([
|
|
18
19
|
{ name: form.name, run: form.run },
|
|
@@ -22,6 +23,7 @@ export const STATIONS = Object.freeze([
|
|
|
22
23
|
{ name: privateStation.name, run: privateStation.run },
|
|
23
24
|
{ name: dna.name, run: dna.run },
|
|
24
25
|
{ name: links.name, run: links.run },
|
|
26
|
+
{ name: sequence.name, run: sequence.run },
|
|
25
27
|
]);
|
|
26
28
|
|
|
27
29
|
export const STATION_NAMES = Object.freeze(STATIONS.map((s) => s.name));
|
package/src/stations/links.mjs
CHANGED
|
@@ -30,6 +30,7 @@
|
|
|
30
30
|
import { existsSync } from "node:fs";
|
|
31
31
|
import { dirname, resolve } from "node:path";
|
|
32
32
|
import { lineAt, maskCode, maskRanges, truncate } from "./util.mjs";
|
|
33
|
+
import { sourceAt } from "../sequence-draft.mjs";
|
|
33
34
|
|
|
34
35
|
export const name = "links";
|
|
35
36
|
|
|
@@ -226,6 +227,12 @@ function undefinedReferenceFinding(label, line) {
|
|
|
226
227
|
|
|
227
228
|
export function run(spec, draft) {
|
|
228
229
|
const draftDirAbs = dirname(resolve(draft.path));
|
|
230
|
+
// A draft assembled from a sequence's files (src/sequence-draft.mjs) resolves each relative link
|
|
231
|
+
// beside the file that holds it.
|
|
232
|
+
const dirAt = (line) => {
|
|
233
|
+
const src = sourceAt(draft, line);
|
|
234
|
+
return src ? dirname(resolve(src.at)) : draftDirAbs;
|
|
235
|
+
};
|
|
229
236
|
|
|
230
237
|
// Masking pipeline: code first, then each link form in turn, each pass working on the text the
|
|
231
238
|
// previous pass left behind, so nothing is ever matched twice by a later, looser pattern (a
|
|
@@ -253,8 +260,9 @@ export function run(spec, draft) {
|
|
|
253
260
|
const entries = [];
|
|
254
261
|
|
|
255
262
|
for (const { start, url } of [...mdLinks, ...bare]) {
|
|
256
|
-
const
|
|
257
|
-
|
|
263
|
+
const line = lineAt(draft.text, start);
|
|
264
|
+
const reason = checkUrl(url, dirAt(line));
|
|
265
|
+
if (reason) entries.push({ start, finding: reasonFinding(reason, url, line) });
|
|
258
266
|
}
|
|
259
267
|
|
|
260
268
|
for (const { start, label } of [...fullRefs, ...shortcutRefs]) {
|
|
@@ -264,7 +272,7 @@ export function run(spec, draft) {
|
|
|
264
272
|
entries.push({ start, finding: undefinedReferenceFinding(label, line) });
|
|
265
273
|
continue;
|
|
266
274
|
}
|
|
267
|
-
const reason = checkUrl(def.url,
|
|
275
|
+
const reason = checkUrl(def.url, dirAt(line));
|
|
268
276
|
if (reason) entries.push({ start, finding: reasonFinding(reason, def.url, line) });
|
|
269
277
|
}
|
|
270
278
|
|
package/src/stations/quotes.mjs
CHANGED
|
@@ -59,15 +59,17 @@ function normalize(text) {
|
|
|
59
59
|
// what was said.
|
|
60
60
|
const matchKey = (inner) => normalize(inner).replace(/[.,]+$/, "").trim();
|
|
61
61
|
|
|
62
|
-
// Every quoted span in `text`, as { start, end, inner }: start/end bound the whole span
|
|
63
|
-
// its quote marks, inner is the text between them
|
|
64
|
-
|
|
62
|
+
// Every quoted span in `text`, as { start, end, inner, para }: start/end bound the whole span
|
|
63
|
+
// including its quote marks, inner is the text between them, para is { start, end } of the
|
|
64
|
+
// paragraph holding it. Paired per paragraph (see the header). Also how the attribution judge
|
|
65
|
+
// (src/judges/attribution.mjs) finds a draft's dialogue lines.
|
|
66
|
+
export function quotedSpans(text) {
|
|
65
67
|
const spans = [];
|
|
66
68
|
for (const para of splitSegments(text, { by: "paragraph" })) {
|
|
67
69
|
QUOTE_RE.lastIndex = 0;
|
|
68
70
|
let m;
|
|
69
71
|
while ((m = QUOTE_RE.exec(para.text))) {
|
|
70
|
-
spans.push({ start: para.start + m.index, end: para.start + m.index + m[0].length, inner: m[1] ?? m[2] ?? "" });
|
|
72
|
+
spans.push({ start: para.start + m.index, end: para.start + m.index + m[0].length, inner: m[1] ?? m[2] ?? "", para: { start: para.start, end: para.end } });
|
|
71
73
|
}
|
|
72
74
|
}
|
|
73
75
|
return spans;
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
// Station "sequence" (hyperspec 0.8). Every other station checks one piece against its spec. A
|
|
2
|
+
// work read in order (a course, a primer, a textbook, a book of lessons) makes promises ACROSS its
|
|
3
|
+
// pieces: lesson 5 is written for someone who has read lessons 1 to 4 and nothing else. This
|
|
4
|
+
// station holds the draft to those promises, and runs only when the spec declares
|
|
5
|
+
// writing.form.sequence.
|
|
6
|
+
//
|
|
7
|
+
// A unit is one lesson (or chapter, or whatever sequence.unit names): an ATX heading "<unit> <n>",
|
|
8
|
+
// optionally followed by ":" or "." and a title, running until the next unit heading, the next
|
|
9
|
+
// heading of its own level or above (a part's heading, whose introduction belongs to no lesson), or
|
|
10
|
+
// the end of the file it sits in. Headings inside fenced code are not headings.
|
|
11
|
+
//
|
|
12
|
+
// The guards, every one deterministic:
|
|
13
|
+
// sections every unit carries each of sequence.sections, as a "**Label:**" line or a heading
|
|
14
|
+
// defined every term is defined in exactly one unit's terms section
|
|
15
|
+
// order no unit uses a term before the unit that defines it; code is not prose, the terms
|
|
16
|
+
// section is not a use, and neither is the unit's closing teaser (a "**<teaser>" line
|
|
17
|
+
// and everything after it). Words in audience.knows and sequence.knows are exempt:
|
|
18
|
+
// listing a word there is a decision that the reader already has it
|
|
19
|
+
// outline each unit defines every term the outline promises for it
|
|
20
|
+
// numbering unit numbers increase through the work
|
|
21
|
+
// pointers (warning) a "<unit> N" pointer to a later unit is reported, so it stays a pointer
|
|
22
|
+
// and never becomes a dependency
|
|
23
|
+
//
|
|
24
|
+
// A use is a whole-word, case-insensitive match, where a hyphen is part of a word: "context-aware"
|
|
25
|
+
// does not use "context", and "skills" does not use "skill". The station reads the outline from
|
|
26
|
+
// disk, resolved against the spec's folder, the way claims reads its ledger. It never reads a
|
|
27
|
+
// part's text for meaning: a term used in a sense other than its definition still counts as a use.
|
|
28
|
+
|
|
29
|
+
import { readFileSync } from "node:fs";
|
|
30
|
+
import { resolve } from "node:path";
|
|
31
|
+
import { str } from "../placeholder.mjs";
|
|
32
|
+
import { maskCode } from "./util.mjs";
|
|
33
|
+
|
|
34
|
+
export const name = "sequence";
|
|
35
|
+
|
|
36
|
+
export const DEFAULTS = Object.freeze({
|
|
37
|
+
unit: "Lesson",
|
|
38
|
+
sections: Object.freeze(["After this lesson you can", "New terms", "Try this"]),
|
|
39
|
+
terms_section: "New terms",
|
|
40
|
+
teaser: "Next,",
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
44
|
+
const list = (v) => (Array.isArray(v) ? v : []);
|
|
45
|
+
const ATX = /^ {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*$/;
|
|
46
|
+
|
|
47
|
+
// A term as the station compares it: lower case, no code or emphasis marks, no parenthetical, one
|
|
48
|
+
// space between words.
|
|
49
|
+
export const normTerm = (s) => String(s).toLowerCase().replace(/[`*]/g, "").replace(/\s*\(.*?\)\s*/g, " ").replace(/\s+/g, " ").trim();
|
|
50
|
+
|
|
51
|
+
// The sequence block with every default filled in. null when the spec declares none.
|
|
52
|
+
export function sequenceOf(spec) {
|
|
53
|
+
const raw = spec?.data?.writing?.form?.sequence;
|
|
54
|
+
if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
|
|
55
|
+
const sections = list(raw.sections).map(str).filter(Boolean);
|
|
56
|
+
const knows = [...list(raw.knows), ...list(spec?.data?.writing?.audience?.knows)].map(str).filter(Boolean).map(normTerm);
|
|
57
|
+
return {
|
|
58
|
+
unit: str(raw.unit) || DEFAULTS.unit,
|
|
59
|
+
sections: sections.length ? sections : [...DEFAULTS.sections],
|
|
60
|
+
termsSection: str(raw.terms_section) || DEFAULTS.terms_section,
|
|
61
|
+
teaser: str(raw.teaser) || DEFAULTS.teaser,
|
|
62
|
+
outline: str(raw.outline) || null,
|
|
63
|
+
knows: new Set(knows),
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Every unit in the draft, in document order: { n, title, line, startLine, text, lines }, where
|
|
68
|
+
// line is the heading's 1-based line, and text is everything after the heading up to where the
|
|
69
|
+
// unit ends (see the header). startLine is the line text begins on. A draft assembled from several
|
|
70
|
+
// files (draft.sources, see src/sequence-draft.mjs) ends a unit at its file's end too.
|
|
71
|
+
export function parseUnits(draft, { unit }) {
|
|
72
|
+
const lines = draft.text.split("\n");
|
|
73
|
+
const masked = maskCode(draft.text).split("\n");
|
|
74
|
+
const unitRe = new RegExp(`^${escapeRe(unit)}[ \\t]+(\\d+)\\b[ \\t]*[:.]?[ \\t]*(.*)$`, "i");
|
|
75
|
+
const fileEnds = new Set(list(draft.sources).map((s) => s.startLine + s.lineCount - 1));
|
|
76
|
+
const units = [];
|
|
77
|
+
let open = null;
|
|
78
|
+
const close = (endLine) => {
|
|
79
|
+
if (!open) return;
|
|
80
|
+
const body = lines.slice(open.line, endLine);
|
|
81
|
+
units.push({ n: open.n, title: open.title, line: open.line, startLine: open.line + 1, lines: body, text: body.join("\n") });
|
|
82
|
+
open = null;
|
|
83
|
+
};
|
|
84
|
+
for (let i = 0; i < lines.length; i++) {
|
|
85
|
+
const h = ATX.exec(masked[i].replace(/\r$/, ""));
|
|
86
|
+
if (h) {
|
|
87
|
+
const level = h[1].length;
|
|
88
|
+
const m = unitRe.exec((h[2] ?? "").replace(/(^|[ \t]+)#+[ \t]*$/, "").trim());
|
|
89
|
+
if (m) {
|
|
90
|
+
close(i);
|
|
91
|
+
open = { n: Number(m[1]), title: m[2].trim(), line: i + 1, level };
|
|
92
|
+
} else if (open && level <= open.level) close(i);
|
|
93
|
+
}
|
|
94
|
+
if (open && fileEnds.has(i + 1)) close(i + 1);
|
|
95
|
+
}
|
|
96
|
+
close(lines.length);
|
|
97
|
+
return units;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// A section label line: "**Label:**", "**Label**:" or "**Label**" at the start of a line (text may
|
|
101
|
+
// follow), or a heading whose text is the label, with or without a closing colon. Case-insensitive.
|
|
102
|
+
function labelRe(label) {
|
|
103
|
+
const l = escapeRe(label);
|
|
104
|
+
return new RegExp(`^(?:[ \\t]*\\*\\*${l}(?::\\*\\*|\\*\\*:?)|[ \\t]{0,3}#{1,6}[ \\t]+${l}:?[ \\t]*#*[ \\t]*$)`, "i");
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// The index (into unit.lines) of the first line carrying `label`, or -1. Code is masked first, so a
|
|
108
|
+
// label shown in an example is not the unit's own.
|
|
109
|
+
function labelLine(unit, label) {
|
|
110
|
+
const masked = maskCode(unit.text).split("\n");
|
|
111
|
+
const re = labelRe(label);
|
|
112
|
+
return masked.findIndex((l) => re.test(l.replace(/\r$/, "")));
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// The [start, end) range of unit.lines holding the terms section: its label line, then the list
|
|
116
|
+
// under it, up to the first blank line after the list begins. null when the unit has no such section.
|
|
117
|
+
function termsRange(unit, termsSection) {
|
|
118
|
+
const at = labelLine(unit, termsSection);
|
|
119
|
+
if (at < 0) return null;
|
|
120
|
+
let end = at + 1;
|
|
121
|
+
while (end < unit.lines.length && !unit.lines[end].trim()) end++; // blank lines before the list
|
|
122
|
+
while (end < unit.lines.length && unit.lines[end].trim()) end++;
|
|
123
|
+
return { start: at, end };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
// The terms a unit's terms section defines, normalized, each once, in order. A list item that
|
|
127
|
+
// reminds the reader of an earlier term, "(from Lesson 3)", defines nothing. Several bold terms
|
|
128
|
+
// before the item's first ":**" are all defined, so "- **A**, **B** and **C:** ..." defines three.
|
|
129
|
+
export function newTerms(unit, { unit: unitWord, termsSection }) {
|
|
130
|
+
const range = termsRange(unit, termsSection);
|
|
131
|
+
if (!range) return [];
|
|
132
|
+
const reminder = new RegExp(`\\(\\s*from\\s+${escapeRe(unitWord)}\\s+\\d+\\s*\\)`, "i");
|
|
133
|
+
const out = [];
|
|
134
|
+
for (const raw of unit.lines.slice(range.start + 1, range.end)) {
|
|
135
|
+
const line = raw.replace(/^[ \t]*[-*+][ \t]+/, "");
|
|
136
|
+
if (reminder.test(line)) continue;
|
|
137
|
+
let head;
|
|
138
|
+
if (line.includes(":**")) head = `${line.split(":**")[0]}**`;
|
|
139
|
+
else if (line.includes("**:")) head = line.split("**:")[0] + "**";
|
|
140
|
+
else head = (line.match(/\*\*(.+?)\*\*/) || [""])[0];
|
|
141
|
+
for (const m of head.matchAll(/\*\*(.+?)\*\*/g)) {
|
|
142
|
+
const t = normTerm(m[1]);
|
|
143
|
+
if (t && !out.includes(t)) out.push(t);
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
return out;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// The terms an outline promises per unit: { n: [term, ...] }. An outline item is a numbered list
|
|
150
|
+
// line, "N. ...", running until the next numbered item at its indent or less, or a heading; its
|
|
151
|
+
// promise is an emphasized "*Terms: a, b, c.*" anywhere in the item.
|
|
152
|
+
export function outlineTerms(text) {
|
|
153
|
+
const out = {};
|
|
154
|
+
const lines = String(text).replace(/\r/g, "").split("\n");
|
|
155
|
+
let cur = null;
|
|
156
|
+
const flush = () => {
|
|
157
|
+
if (!cur) return;
|
|
158
|
+
const m = cur.text.match(/\*Terms:\s*([^*]*?)\.?\s*\*/);
|
|
159
|
+
if (m) out[cur.n] = m[1].split(/,\s*/).map(normTerm).filter(Boolean);
|
|
160
|
+
cur = null;
|
|
161
|
+
};
|
|
162
|
+
for (const line of lines) {
|
|
163
|
+
const item = /^([ \t]*)(\d+)\.[ \t]/.exec(line);
|
|
164
|
+
if (item && (!cur || item[1].length <= cur.indent)) {
|
|
165
|
+
flush();
|
|
166
|
+
cur = { n: Number(item[2]), indent: item[1].length, text: line };
|
|
167
|
+
} else if (/^ {0,3}#{1,6}[ \t]/.test(line)) flush();
|
|
168
|
+
else if (cur) cur.text += `\n${line}`;
|
|
169
|
+
}
|
|
170
|
+
flush();
|
|
171
|
+
return out;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// The unit's prose as the order and pointer guards read it: code masked and the closing teaser
|
|
175
|
+
// (from its "**<teaser>" line to the unit's end) blanked, lines kept so a position still maps to a
|
|
176
|
+
// line. withTerms false blanks the terms section too, for the order guard: a unit's terms section
|
|
177
|
+
// is where it defines words, and a reminder there names an earlier unit's word on purpose. The
|
|
178
|
+
// pointer guard keeps it, since "(Lesson 18 covers this)" in a definition is still a pointer.
|
|
179
|
+
function proseOf(unit, seq, { withTerms }) {
|
|
180
|
+
const lines = maskCode(unit.text).split("\n");
|
|
181
|
+
const range = withTerms ? null : termsRange(unit, seq.termsSection);
|
|
182
|
+
if (range) for (let i = range.start; i < range.end; i++) lines[i] = "";
|
|
183
|
+
const teaserRe = new RegExp(`^[ \\t]*\\*\\*${escapeRe(seq.teaser)}`, "i");
|
|
184
|
+
const t = lines.findIndex((l) => teaserRe.test(l));
|
|
185
|
+
if (t >= 0) for (let i = t; i < lines.length; i++) lines[i] = "";
|
|
186
|
+
return lines;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
// The 1-based draft line of the first whole-word use of `term` in the unit's prose, or 0.
|
|
190
|
+
function firstUse(unit, prose, term) {
|
|
191
|
+
const re = new RegExp(`(^|[^a-z0-9-])${escapeRe(term)}([^a-z0-9-]|$)`, "i");
|
|
192
|
+
const i = prose.findIndex((l) => re.test(l));
|
|
193
|
+
return i < 0 ? 0 : unit.startLine + i;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
// One finding, in the shape every station returns; line only when there is one to point at.
|
|
197
|
+
const finding = ({ id, severity, message, fix, line }) => ({ station: name, id, severity, ...(line ? { line } : {}), message, fix });
|
|
198
|
+
|
|
199
|
+
export function run(spec, draft) {
|
|
200
|
+
const seq = sequenceOf(spec);
|
|
201
|
+
if (!seq) return { station: name, status: "skip", findings: [], reason: "the spec declares no writing.form.sequence" };
|
|
202
|
+
const U = seq.unit;
|
|
203
|
+
const findings = [];
|
|
204
|
+
const units = parseUnits(draft, seq);
|
|
205
|
+
if (!units.length) {
|
|
206
|
+
findings.push(finding({ id: "station-sequence-no-units", severity: "fail", message: `no "${U} <n>" heading in the draft`, fix: `Head each ${U.toLowerCase()} "## ${U} 1: <title>", or set writing.form.sequence.unit to the word the headings use.` }));
|
|
207
|
+
return { station: name, status: "fail", findings };
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
// numbering: each unit's number is greater than the one before it.
|
|
211
|
+
for (let i = 1; i < units.length; i++) {
|
|
212
|
+
if (units[i].n <= units[i - 1].n) {
|
|
213
|
+
findings.push(finding({ id: "station-sequence-numbering", severity: "fail", message: `${U} ${units[i].n} comes after ${U} ${units[i - 1].n}`, fix: `Number the ${U.toLowerCase()}s in reading order, or reorder writing.form.sequence.files.`, line: units[i].line }));
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// sections, and where each term is defined.
|
|
218
|
+
const definedAt = new Map();
|
|
219
|
+
for (const u of units) {
|
|
220
|
+
for (const s of seq.sections) {
|
|
221
|
+
if (labelLine(u, s) < 0) findings.push(finding({ id: "station-sequence-missing-section", severity: "fail", message: `${U} ${u.n} has no "${s}" section`, fix: `Add a "**${s}:**" line (or a "${s}" heading) to ${U} ${u.n}.`, line: u.line }));
|
|
222
|
+
}
|
|
223
|
+
for (const t of newTerms(u, seq)) {
|
|
224
|
+
const first = definedAt.get(t);
|
|
225
|
+
if (first && first !== u) findings.push(finding({ id: "station-sequence-defined-twice", severity: "fail", message: `"${t}" is defined in ${U} ${first.n} and again in ${U} ${u.n}`, fix: `Define "${t}" once; later ${U.toLowerCase()}s remind the reader with "- **${t}** (from ${U} ${first.n}): ...".`, line: u.line }));
|
|
226
|
+
else if (!first) definedAt.set(t, u);
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
// order: no unit before the defining one uses the term.
|
|
231
|
+
const prose = new Map(units.map((u) => [u, proseOf(u, seq, { withTerms: false })]));
|
|
232
|
+
for (const [t, at] of definedAt) {
|
|
233
|
+
if (seq.knows.has(t)) continue;
|
|
234
|
+
for (const u of units) {
|
|
235
|
+
if (u === at) break;
|
|
236
|
+
const line = firstUse(u, prose.get(u), t);
|
|
237
|
+
if (line) findings.push(finding({ id: "station-sequence-used-before-defined", severity: "fail", message: `${U} ${u.n} uses "${t}" before ${U} ${at.n} defines it`, fix: `Rewrite ${U} ${u.n} without "${t}", define it earlier, or add it to writing.form.sequence.knows if the reader already has the word.`, line: line }));
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// outline: each unit defines what the outline promises for it. A unit the outline promises that
|
|
242
|
+
// the draft does not hold yet is not a finding: a work is checked while it is being written.
|
|
243
|
+
if (seq.outline) {
|
|
244
|
+
let text = null;
|
|
245
|
+
try { text = readFileSync(resolve(spec?.dir || ".", seq.outline), "utf8"); } catch { /* reported below */ }
|
|
246
|
+
if (text === null) {
|
|
247
|
+
findings.push(finding({ id: "station-sequence-outline-unreadable", severity: "fail", message: `the outline "${seq.outline}" cannot be read`, fix: "Point writing.form.sequence.outline at the outline file, relative to the spec." }));
|
|
248
|
+
} else {
|
|
249
|
+
const promised = outlineTerms(text);
|
|
250
|
+
for (const u of units) {
|
|
251
|
+
const have = newTerms(u, seq);
|
|
252
|
+
for (const t of promised[u.n] ?? []) {
|
|
253
|
+
if (!have.includes(t)) findings.push(finding({ id: "station-sequence-outline-unkept", severity: "fail", message: `the outline promises "${t}" in ${U} ${u.n}, and ${U} ${u.n} does not define it`, fix: `Define "${t}" in ${U} ${u.n}'s ${seq.termsSection}, or change the outline.`, line: u.line }));
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
// pointers: a mention of a later unit, once per pair.
|
|
260
|
+
const pointerRe = new RegExp(`\\b${escapeRe(U)}\\s+(\\d+)`, "gi");
|
|
261
|
+
for (const u of units) {
|
|
262
|
+
const seen = new Set();
|
|
263
|
+
proseOf(u, seq, { withTerms: true }).forEach((l, i) => {
|
|
264
|
+
for (const m of l.matchAll(pointerRe)) {
|
|
265
|
+
const n = Number(m[1]);
|
|
266
|
+
if (n <= u.n || seen.has(n)) continue;
|
|
267
|
+
seen.add(n);
|
|
268
|
+
findings.push(finding({ id: "station-sequence-forward-pointer", severity: "warn", message: `${U} ${u.n} points forward to ${U} ${n}`, fix: `Keep it a pointer ("more in ${U} ${n}"): ${U} ${u.n} must make sense to a reader who has not read ${U} ${n}.`, line: u.startLine + i }));
|
|
269
|
+
}
|
|
270
|
+
});
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
const status = findings.some((f) => f.severity === "fail") ? "fail" : "pass";
|
|
274
|
+
return { station: name, status, findings };
|
|
275
|
+
}
|
package/src/writing-fields.mjs
CHANGED
|
@@ -23,6 +23,7 @@ import { basename, dirname, join, sep } from "node:path";
|
|
|
23
23
|
import { str } from "./placeholder.mjs";
|
|
24
24
|
import { readSegments } from "./segments.mjs";
|
|
25
25
|
import { readScope, isGoldenFileName, measureFeatures, featuresText, DNA_FORMAT } from "./dna.mjs";
|
|
26
|
+
import { expandEntry } from "./sequence-draft.mjs";
|
|
26
27
|
|
|
27
28
|
const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
|
|
28
29
|
const list = (v) => (Array.isArray(v) ? v : []);
|
|
@@ -446,6 +447,39 @@ function formFields(raw, d, here, idPrefix) {
|
|
|
446
447
|
if (minOk && maxOk && min > max) out.push(f(1, `${idPrefix}-length-range`, "fail", `writing.form.length.min (${min}) is greater than length.max (${max})`, "Set min to no more than max."));
|
|
447
448
|
if (!str(length.unit)) out.push(f(1, `${idPrefix}-length-unit`, "fail", "writing.form.length has no unit", "Add length.unit:, e.g. words."));
|
|
448
449
|
if (!list(raw.required_parts).some((x) => str(x))) out.push(f(1, `${idPrefix}-required-parts`, "fail", "writing.form has no required_parts", "List at least one required part."));
|
|
450
|
+
if ("sequence" in raw) out.push(...sequenceFields(raw.sequence, here, `${idPrefix}-sequence`));
|
|
451
|
+
return out;
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
// writing.form.sequence, optional: a work read in order (see WRITING.md, Sequential works). Every
|
|
455
|
+
// key is optional, and each one present has to be usable, because the sequence station reads it
|
|
456
|
+
// as given and a hollow value would switch a guard off without saying so.
|
|
457
|
+
function sequenceFields(seq, here, idPrefix) {
|
|
458
|
+
if (!isObj(seq)) return [f(1, idPrefix, "fail", "writing.form.sequence is not a map", "Write sequence: as a map of the keys WRITING.md lists, one per line.")];
|
|
459
|
+
const out = [];
|
|
460
|
+
for (const key of ["unit", "terms_section", "teaser"]) {
|
|
461
|
+
if (key in seq && !str(seq[key])) out.push(f(1, `${idPrefix}-${key.replace("_", "-")}`, "fail", `writing.form.sequence.${key} is empty or a placeholder`, `Give ${key} a real value, or delete it for the default.`));
|
|
462
|
+
}
|
|
463
|
+
const strings = (key, fix) => {
|
|
464
|
+
if (!(key in seq)) return null;
|
|
465
|
+
const items = list(seq[key]);
|
|
466
|
+
if (!Array.isArray(seq[key]) || !items.length || items.some((x) => typeof x !== "string" || !str(x))) {
|
|
467
|
+
out.push(f(1, `${idPrefix}-${key}`, "fail", `writing.form.sequence.${key} is not a list of real entries`, fix));
|
|
468
|
+
return null;
|
|
469
|
+
}
|
|
470
|
+
return items;
|
|
471
|
+
};
|
|
472
|
+
strings("sections", "List every section a unit carries, or delete sections: for the default three.");
|
|
473
|
+
strings("knows", "List the words a reader already has, one per entry, or delete knows:.");
|
|
474
|
+
const files = strings("files", "List the work's files in reading order, relative to the spec; a * in a file name matches.");
|
|
475
|
+
for (const entry of files ?? []) {
|
|
476
|
+
if (!expandEntry(here("."), entry).length) out.push(f(6, `${idPrefix}-files-missing`, "fail", `writing.form.sequence.files entry "${entry}" matches no file`, "Fix the path (relative to the spec), or remove the entry."));
|
|
477
|
+
}
|
|
478
|
+
if ("outline" in seq) {
|
|
479
|
+
const p = str(seq.outline);
|
|
480
|
+
if (!p) out.push(f(1, `${idPrefix}-outline`, "fail", "writing.form.sequence.outline is empty or a placeholder", "Name the outline file, or delete outline:."));
|
|
481
|
+
else out.push(...pathFindings(here, p, `${idPrefix}-outline`, "writing.form.sequence.outline", "Point outline: at the outline file, relative to the spec."));
|
|
482
|
+
}
|
|
449
483
|
return out;
|
|
450
484
|
}
|
|
451
485
|
|
package/src/writing.mjs
CHANGED
|
@@ -36,7 +36,7 @@ const required = (block, fiction) => block !== "characters" || fiction;
|
|
|
36
36
|
// least one field. This is deliberately shallow: it is the bar for "something was written here",
|
|
37
37
|
// not the bar for "this block is correct", which is what checkOwner and the field rules in
|
|
38
38
|
// writing-fields.mjs are for.
|
|
39
|
-
function blockPresent(block, raw) {
|
|
39
|
+
export function blockPresent(block, raw) {
|
|
40
40
|
if (block === "characters") return Array.isArray(raw) && raw.length > 0;
|
|
41
41
|
if (block === "materials") return isObj(raw) && Array.isArray(raw.items) && raw.items.length > 0;
|
|
42
42
|
return isObj(raw) && Object.keys(raw).length > 0;
|