@supersuit/hyperspec 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +143 -0
  2. package/README.md +50 -2
  3. package/SPEC.md +5 -5
  4. package/WRITING.md +510 -17
  5. package/bin/hyperspec.mjs +177 -2
  6. package/examples/writing/dna/essay-new-managers-teach/features.json +56 -0
  7. package/examples/writing/dna/essay-new-managers-teach/goldens/README.md +14 -0
  8. package/examples/writing/dna/essay-new-managers-teach/goldens/close.md +9 -0
  9. package/examples/writing/dna/essay-new-managers-teach/goldens/opening.md +9 -0
  10. package/examples/writing/dna/essay-new-managers-teach/goldens/status.md +10 -0
  11. package/examples/writing/dna/essay-new-managers-teach/scope.md +11 -0
  12. package/examples/writing/essay/claims.jsonl +9 -0
  13. package/examples/writing/essay/draft.md +82 -0
  14. package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +2 -2
  15. package/examples/writing/essay.hyperspec.md +32 -13
  16. package/examples/writing/story/claims.jsonl +9 -0
  17. package/examples/writing/story/draft.md +267 -0
  18. package/examples/writing/story.hyperspec.md +20 -5
  19. package/package.json +1 -1
  20. package/src/check.mjs +245 -0
  21. package/src/dna.mjs +474 -0
  22. package/src/stations/claims.mjs +150 -0
  23. package/src/stations/dna.mjs +126 -0
  24. package/src/stations/form.mjs +115 -0
  25. package/src/stations/index.mjs +27 -0
  26. package/src/stations/links.mjs +275 -0
  27. package/src/stations/private.mjs +117 -0
  28. package/src/stations/quotes.mjs +168 -0
  29. package/src/stations/terms.mjs +131 -0
  30. package/src/stations/util.mjs +99 -0
  31. package/src/writing-exports.mjs +8 -3
  32. package/src/writing-fields.mjs +197 -1
  33. package/src/writing-template.mjs +8 -0
  34. package/examples/writing/essay/goldens/close.md +0 -2
  35. package/examples/writing/essay/goldens/opening.md +0 -2
@@ -0,0 +1,168 @@
1
+ // Station "quotes" (hyperspec 0.6). Every double-quoted span in the draft of 4 or
2
+ // more words must appear in a `quote` or `story` segment of a marked material; a quote the draft
3
+ // attributes to someone by name must appear in a `quote` segment whose speaker is that someone.
4
+ // Pure and deterministic: no model call, no judgment about what counts as quoting, just the closed
5
+ // rules below.
6
+ //
7
+ // What counts as a quoted span: text between a pair of straight double quotes ("...") or a pair of
8
+ // curly ones (U+201C ... U+201D), paired left to right WITHIN one paragraph, so a stray quote mark
9
+ // can never pair with one in a later paragraph and swallow everything between. Fenced code and
10
+ // inline code are masked first (util.mjs's maskCode): a quote inside example syntax is not the
11
+ // writer quoting anyone. A span is checked only when it holds 4 or more words (dna.mjs's wordsOf,
12
+ // the one word definition every station shares); shorter spans are scare quotes and titles as
13
+ // often as they are quotations.
14
+ //
15
+ // Matching: both sides are normalized the same way (curly quotes and apostrophes to straight,
16
+ // whitespace runs to one space, ends trimmed; case is kept, so a changed capital is a changed
17
+ // quote). The span also drops trailing commas and periods before matching, because typographic
18
+ // convention puts a sentence's own comma or period inside the closing quote ("...at a time," he
19
+ // said) whether or not the speaker's sentence ended there. The normalized span must then appear as
20
+ // a substring of some quote or story segment's text, in any marked material of the spec.
21
+ //
22
+ // Attribution: a speaker is any `speaker` value on a quote segment. It is named in the draft
23
+ // when the sentence that holds the quote contains, case-insensitively and as whole
24
+ // words, EITHER the full value (split on anything that is not a letter, digit or apostrophe, so the
25
+ // slug "maria-lopez" reads as "maria lopez", its words joined in the draft by whitespace, hyphens or
26
+ // underscores) OR the value's first word alone ("Maria said" names maria-lopez; "Mariana" does not).
27
+ // The first word alone counts only when it has 2 or more letters and is not one of dna.mjs's
28
+ // STOPWORDS: a speaker recorded as "the manager interviewed" would otherwise be named by every
29
+ // sentence holding "the". Such a speaker is still named by its full value. The sentence is read
30
+ // with the quote itself blanked out (a name inside the quoted words is what was said, not who said it). A named speaker
31
+ // means the span must be in a quote segment with that speaker; matching only some other speaker's
32
+ // quote, or only a story, is `station-quotes-misattributed`. With no speaker named, any quote or
33
+ // story segment is enough. Sentences come from splitSegments(text, { by: "sentence" }), and every
34
+ // sentence the span overlaps is read, so a curly quote the splitter cuts in two still keeps its
35
+ // attribution.
36
+
37
+ import { splitSegments } from "../segments.mjs";
38
+ import { STOPWORDS, wordsOf } from "../dna.mjs";
39
+ import { str } from "../placeholder.mjs";
40
+ import { lineAt, maskCode, markedSegments, truncate } from "./util.mjs";
41
+
42
+ export const name = "quotes";
43
+
44
+ const QUOTE_RE = /"([^"]*)"|“([^“”]*)”/g;
45
+ const WORD_CLASS = "\\p{L}\\p{N}'’";
46
+
47
+ function normalize(text) {
48
+ return String(text)
49
+ .replace(/[‘’‚‛]/g, "'")
50
+ .replace(/[“”„‟]/g, '"')
51
+ .replace(/\s+/g, " ")
52
+ .trim();
53
+ }
54
+
55
+ // What a quoted span is matched on: normalized, then trailing commas and periods dropped.
56
+ // Typographic convention puts the writer's own comma or period inside the closing quote
57
+ // ("...at a time," he said) whether or not the speaker's sentence ended there, so keeping them
58
+ // would fail nearly every quotation used mid-sentence. "?" and "!" are kept: adding either changes
59
+ // what was said.
60
+ const matchKey = (inner) => normalize(inner).replace(/[.,]+$/, "").trim();
61
+
62
+ // Every quoted span in `text`, as { start, end, inner }: start/end bound the whole span including
63
+ // its quote marks, inner is the text between them. Paired per paragraph (see the header).
64
+ function quotedSpans(text) {
65
+ const spans = [];
66
+ for (const para of splitSegments(text, { by: "paragraph" })) {
67
+ QUOTE_RE.lastIndex = 0;
68
+ let m;
69
+ while ((m = QUOTE_RE.exec(para.text))) {
70
+ spans.push({ start: para.start + m.index, end: para.start + m.index + m[0].length, inner: m[1] ?? m[2] ?? "" });
71
+ }
72
+ }
73
+ return spans;
74
+ }
75
+
76
+ const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
77
+
78
+ const STOPWORD_SET = new Set(STOPWORDS);
79
+
80
+ // Whether a speaker's first word may name the speaker on its own: 2 or more letters, and not a
81
+ // stopword (compared lowercased, apostrophes as written).
82
+ function firstWordNames(word) {
83
+ const letters = word.match(/\p{L}/gu)?.length ?? 0;
84
+ return letters >= 2 && !STOPWORD_SET.has(word.toLowerCase());
85
+ }
86
+
87
+ // A speaker value as a whole-word, case-insensitive pattern matching its full words, or its first
88
+ // word alone when firstWordNames allows it; null when it has no words at all.
89
+ export function speakerPattern(value) {
90
+ const words = value.split(new RegExp(`[^${WORD_CLASS}]+`, "u")).filter(Boolean);
91
+ if (!words.length) return null;
92
+ const full = words.map(escapeRe).join("[\\s_-]+");
93
+ const alts = firstWordNames(words[0]) && words.length > 1 ? `${full}|${escapeRe(words[0])}` : full;
94
+ return new RegExp(`(?<![${WORD_CLASS}])(?:${alts})(?![${WORD_CLASS}])`, "iu");
95
+ }
96
+
97
+ // The text of every sentence the span overlaps, with the span itself blanked out.
98
+ function attributionContext(text, sentences, span) {
99
+ const over = sentences.filter((s) => s.start < span.end && s.end > span.start);
100
+ const from = over.length ? Math.min(span.start, over[0].start) : span.start;
101
+ const to = over.length ? Math.max(span.end, over[over.length - 1].end) : span.end;
102
+ return `${text.slice(from, span.start)} ${text.slice(span.end, to)}`;
103
+ }
104
+
105
+ export function run(spec, draft, ctx) {
106
+ // Fiction: a character's line is invented, not quoted from a material, so holding dialogue to
107
+ // the marked quotes would fail every story that uses quotation marks. The character stations of a
108
+ // later release check dialogue against each character's golden and rejected lines instead.
109
+ if (str(spec?.data?.fiction) === "true") {
110
+ return { station: name, status: "skip", findings: [], reason: "fiction dialogue is checked by the character stations in a later release" };
111
+ }
112
+ const text = maskCode(draft.text);
113
+ const spans = quotedSpans(text).filter((s) => wordsOf(s.inner).length >= 4);
114
+ if (!spans.length) return { station: name, status: "pass", findings: [] };
115
+
116
+ // Every quote and story segment across every marked material, normalized once. A speaker's key
117
+ // is its value lowercased; shownSpeaker keeps the first spelling the material wrote, for findings.
118
+ const sources = [];
119
+ const shownSpeaker = new Map();
120
+ for (const { segments } of markedSegments(spec, ctx)) {
121
+ for (const seg of segments) {
122
+ if ((seg.label !== "quote" && seg.label !== "story") || typeof seg.text !== "string") continue;
123
+ const speaker = seg.label === "quote" && typeof seg.speaker === "string" ? seg.speaker.trim() : "";
124
+ const key = speaker.toLowerCase();
125
+ if (key && !shownSpeaker.has(key)) shownSpeaker.set(key, speaker);
126
+ sources.push({ label: seg.label, speaker: key, norm: normalize(seg.text) });
127
+ }
128
+ }
129
+ const speakers = [...shownSpeaker.keys()]
130
+ .map((key) => ({ key, re: speakerPattern(key) }))
131
+ .filter((sp) => sp.re);
132
+
133
+ const sentences = splitSegments(text, { by: "sentence" });
134
+ const findings = [];
135
+ for (const span of spans) {
136
+ const key = matchKey(span.inner);
137
+ if (!key) continue;
138
+ const shown = truncate(span.inner.replace(/\s+/g, " "), 80);
139
+ const line = lineAt(draft.text, span.start);
140
+ const hits = sources.filter((s) => s.norm.includes(key));
141
+ if (!hits.length) {
142
+ findings.push({
143
+ station: name,
144
+ id: "station-quotes-unmatched",
145
+ severity: "fail",
146
+ line,
147
+ message: `quoted span "${shown}" does not appear in any quote or story segment of a marked material`,
148
+ fix: "Quote the material verbatim, or drop the quotation marks and paraphrase.",
149
+ });
150
+ continue;
151
+ }
152
+ const around = attributionContext(text, sentences, span);
153
+ const named = speakers.filter((sp) => sp.re.test(around)).map((sp) => sp.key);
154
+ if (named.length && !hits.some((h) => h.label === "quote" && named.includes(h.speaker))) {
155
+ const who = named.map((k) => shownSpeaker.get(k) ?? k).join(", ");
156
+ findings.push({
157
+ station: name,
158
+ id: "station-quotes-misattributed",
159
+ severity: "fail",
160
+ line,
161
+ message: `quoted span "${shown}" is attributed to ${who} in its sentence, but no quote segment by ${who} contains it`,
162
+ fix: `Attribute the quote to whoever the material records saying it, or quote what ${who} actually said.`,
163
+ });
164
+ }
165
+ }
166
+
167
+ return { station: name, status: findings.length ? "fail" : "pass", findings };
168
+ }
@@ -0,0 +1,131 @@
1
+ // Station "terms" (hyperspec 0.6). Checks that every term in
2
+ // writing.audience.terms (the new optional list: terms the piece uses that the reader may not
3
+ // know) is defined the first time it appears in the draft. Pure and deterministic like every
4
+ // station: no model call, no judgment about which words are jargon, just the closed rule below.
5
+ //
6
+ // Matching a term against the draft is case-insensitive and whole-word (Unicode-aware: a match
7
+ // cannot start or end mid-word, using the same letter/number/apostrophe class dna.mjs's own word
8
+ // definition uses, so "AI" does not match inside "said" and a multi-word term like "context
9
+ // window" matches only as that exact phrase).
10
+ //
11
+ // "Defined at its first appearance" is read as: the sentence containing
12
+ // the term's first appearance, OR the sentence right after it, contains an occurrence of the term
13
+ // followed (within the next 6 words of that same sentence, or by a colon among them) by one of
14
+ // "is", "means", "refers to"; OR the term is immediately followed by a parenthetical ("term
15
+ // (short gloss)"). Sentence boundaries come from splitSegments(text, { by: "sentence" }) in
16
+ // src/segments.mjs, the same splitter every other station that reads sentences uses, so "which
17
+ // sentence is this in" never disagrees between stations.
18
+ //
19
+ // A term also listed in writing.audience.knows is never flagged, whatever the draft does with it:
20
+ // the reader is assumed to already have it. A term that never appears in the draft at all is not
21
+ // flagged either (nothing to define); writing.audience.terms with no real entries, or missing
22
+ // entirely, skips the whole station.
23
+ //
24
+ // Fenced code blocks and inline code spans are masked out (src/stations/util.mjs's maskCode)
25
+ // before any of the above runs: a term's only appearance inside example syntax like
26
+ // `` `ledger: check` `` is not a prose use, and the mechanical colon/`is` rule below would
27
+ // otherwise read that code span's own punctuation as a definition it never gave. Masking
28
+ // preserves every character offset, so every 1-based line number reported below is still a line
29
+ // of the ORIGINAL draft, never the masked copy.
30
+
31
+ import { str } from "../placeholder.mjs";
32
+ import { splitSegments } from "../segments.mjs";
33
+ import { lineAt, maskCode } from "./util.mjs";
34
+
35
+ export const name = "terms";
36
+
37
+ // The same word/number class dna.mjs's WORD_RE tokenizes with (letters, digits, straight or
38
+ // curly apostrophe for contractions), Unicode-aware so an accented or non-Latin term's edges are
39
+ // read correctly too.
40
+ const WORD_CHARS = "\\p{L}\\p{N}'’";
41
+
42
+ function escapeRegExp(s) {
43
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
44
+ }
45
+
46
+ // A case-insensitive, whole-phrase, Unicode-aware matcher for `term`: it cannot start or end
47
+ // mid-word (a lookbehind/lookahead against the word-char class above), but the term itself may
48
+ // contain internal spaces (a multi-word term matches only as that exact phrase, spaces and all).
49
+ function termRegex(term, flags) {
50
+ const body = escapeRegExp(term.trim()).replace(/\s+/g, "\\s+");
51
+ return new RegExp(`(?<![${WORD_CHARS}])${body}(?![${WORD_CHARS}])`, flags);
52
+ }
53
+
54
+ // Every { word, start, end } token in `text`, in order, using the same word-char class as above.
55
+ function tokenize(text) {
56
+ const re = new RegExp(`[${WORD_CHARS}]+`, "gu");
57
+ const out = [];
58
+ let m;
59
+ while ((m = re.exec(text))) out.push({ word: m[0].toLowerCase(), start: m.index, end: m.index + m[0].length });
60
+ return out;
61
+ }
62
+
63
+ // Whether `sentenceText` contains a definition of `term`: some occurrence of the term followed,
64
+ // within the next 6 words of the SAME sentence (or by a colon somewhere among them), by "is",
65
+ // "means" or "refers"+"to"; or immediately (skipping only whitespace) followed by "(".
66
+ function definesTerm(sentenceText, term) {
67
+ const re = termRegex(term, "giu");
68
+ let m;
69
+ while ((m = re.exec(sentenceText))) {
70
+ const tailStart = m.index + m[0].length;
71
+ const tail = sentenceText.slice(tailStart);
72
+ if (/^\s*\(/.test(tail)) return true;
73
+
74
+ const tokens = tokenize(tail).slice(0, 6);
75
+ const windowEnd = tokens.length ? tokens[tokens.length - 1].end : tail.length;
76
+ const window = tail.slice(0, windowEnd);
77
+ if (window.includes(":")) return true;
78
+ const words = tokens.map((t) => t.word);
79
+ if (words.includes("is") || words.includes("means")) return true;
80
+ if (words.some((w, i) => w === "refers" && words[i + 1] === "to")) return true;
81
+ // re is not global-sticky across iterations by construction (lastIndex advances past this
82
+ // match automatically since the "g" flag is set), so a term repeated in one sentence is
83
+ // still checked occurrence by occurrence rather than looping forever.
84
+ }
85
+ return false;
86
+ }
87
+
88
+ export function run(spec, draft) {
89
+ const audience = spec?.data?.writing?.audience ?? {};
90
+ const terms = Array.isArray(audience.terms) ? audience.terms.map(str).filter(Boolean) : [];
91
+ if (!terms.length) {
92
+ return { station: name, status: "skip", findings: [], reason: "writing.audience.terms is empty or not set" };
93
+ }
94
+
95
+ const knows = new Set((Array.isArray(audience.knows) ? audience.knows : []).map((x) => str(x).toLowerCase()).filter(Boolean));
96
+
97
+ const scanText = maskCode(draft.text);
98
+ const sentences = splitSegments(scanText, { by: "sentence" });
99
+ const findings = [];
100
+ const usedIds = new Set();
101
+
102
+ for (const term of terms) {
103
+ if (knows.has(term.toLowerCase())) continue;
104
+
105
+ const first = termRegex(term, "iu").exec(scanText);
106
+ if (!first) continue; // the term never appears outside code; nothing to define
107
+
108
+ const pos = first.index;
109
+ const idx = sentences.findIndex((s) => pos >= s.start && pos < s.end);
110
+ const candidates = idx === -1 ? [] : [sentences[idx], sentences[idx + 1]].filter(Boolean);
111
+ const defined = candidates.some((s) => definesTerm(s.text, term));
112
+ if (defined) continue;
113
+
114
+ const slug = term.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "") || "term";
115
+ let id = `station-terms-undefined-${slug}`;
116
+ let n = 2;
117
+ while (usedIds.has(id)) { id = `station-terms-undefined-${slug}-${n}`; n += 1; }
118
+ usedIds.add(id);
119
+
120
+ findings.push({
121
+ station: name,
122
+ id,
123
+ severity: "fail",
124
+ line: lineAt(draft.text, pos),
125
+ message: `term "${term}" is not defined at its first appearance (need "is", "means", "refers to", a colon within a few words, or an immediate parenthetical, in that sentence or the next)`,
126
+ fix: `Define "${term}" where it first appears, e.g. "${term} is ..." or "${term} (a short gloss)".`,
127
+ });
128
+ }
129
+
130
+ return { station: name, status: findings.length ? "fail" : "pass", findings };
131
+ }
@@ -0,0 +1,99 @@
1
+ import { resolve } from "node:path";
2
+ import { str } from "../placeholder.mjs";
3
+ import { readSegments } from "../segments.mjs";
4
+
5
+ // Shared helpers for the deterministic stations (hyperspec 0.6). "One place this pattern is
6
+ // defined" (src/placeholder.mjs's own header comment), so a helper more than one station needs
7
+ // lives here rather than being copied station to station: lineAt, truncate, maskCode, maskRanges
8
+ // and markedSegments.
9
+
10
+ // The 1-based line containing character offset `pos` of `text`. Counts "\n" characters directly
11
+ // rather than reading draft.lines: draft.lines' own CRLF handling ("\r" left attached to the
12
+ // previous line's entry, or stripped, depending on how src/check.mjs built it) is none of any
13
+ // station's business, and counting "\n" occurrences gives the right line number either way, since
14
+ // every line, CRLF or not, still carries exactly one "\n".
15
+ export function lineAt(text, pos) {
16
+ let line = 1;
17
+ for (let i = 0; i < pos && i < text.length; i++) if (text[i] === "\n") line++;
18
+ return line;
19
+ }
20
+
21
+ // Trims `text`, then truncates to at most `max` characters (an ellipsis replacing the last
22
+ // character when it is longer), matching the general finding-message rule that a quoted span of
23
+ // the draft is at most 80 characters. Trimming FIRST means a span with leading/trailing
24
+ // whitespace never wastes truncation budget on it.
25
+ export function truncate(text, max) {
26
+ const t = String(text).trim();
27
+ return t.length > max ? `${t.slice(0, max - 1)}…` : t;
28
+ }
29
+
30
+ // `text` with every [start, end) range in `ranges` blanked out: every character in the range
31
+ // becomes a space, EXCEPT "\n", which is left alone. Preserving newlines (rather than blanking
32
+ // them too) is what keeps every character offset, and therefore every 1-based line number
33
+ // computed afterward, identical to `text` before masking, even when a masked range spans more
34
+ // than one line (a fenced code block, in particular): blanking a "\n" would merge two lines into
35
+ // one for every offset that comes after it, which is exactly the bug a single-line-only masker
36
+ // would not have caught. Used by every "mask X out before scanning for Y" step across the
37
+ // stations, so a masked-and-rescanned span can never be matched a second time.
38
+ export function maskRanges(text, ranges) {
39
+ let out = text;
40
+ for (const { start, end } of ranges) {
41
+ const blanked = out.slice(start, end).replace(/[^\n]/g, " ");
42
+ out = out.slice(0, start) + blanked + out.slice(end);
43
+ }
44
+ return out;
45
+ }
46
+
47
+ // A fenced code block: a line starting (after up to 3 spaces of indent) with 3 or more backticks
48
+ // or tildes, some content, then a line starting with at least as many of the SAME fence
49
+ // character. Non-greedy so back-to-back fences in one draft are matched as separate blocks
50
+ // rather than one block swallowing everything between the first open and the last close.
51
+ const FENCE_RE = /^[ \t]{0,3}(`{3,}|~{3,})[^\n]*\n[\s\S]*?^[ \t]{0,3}\1[ \t]*$/gm;
52
+
53
+ // An inline code span: a single backtick, a run of non-backtick, non-newline characters, a
54
+ // closing backtick. Does not attempt CommonMark's full rule for spans opened with a run of two or
55
+ // more backticks (rare in practice, and unneeded for the case this exists to fix: an inline
56
+ // `like this` mention of syntax that must not be read as prose).
57
+ const INLINE_CODE_RE = /`[^`\n]+`/g;
58
+
59
+ // `text` with every fenced code block and inline code span blanked out (offsets and line numbers
60
+ // preserved, see maskRanges above). Both `links` (a URL shown as example syntax inside a fence or
61
+ // a span is not a real, followable link) and `terms` (a term's only appearance inside `` `code`
62
+ // `` is not a prose use, and must not count toward "does this term appear/get defined") mask code
63
+ // out before doing anything else, so later masking passes (Markdown links, reference definitions,
64
+ // bare URLs, sentence splitting) never see inside a code span or fence.
65
+ export function maskCode(text) {
66
+ const fenceRanges = [];
67
+ FENCE_RE.lastIndex = 0;
68
+ let m;
69
+ while ((m = FENCE_RE.exec(text))) fenceRanges.push({ start: m.index, end: m.index + m[0].length });
70
+ const withoutFences = maskRanges(text, fenceRanges);
71
+
72
+ const inlineRanges = [];
73
+ INLINE_CODE_RE.lastIndex = 0;
74
+ let im;
75
+ while ((im = INLINE_CODE_RE.exec(withoutFences))) inlineRanges.push({ start: im.index, end: im.index + im[0].length });
76
+ return maskRanges(withoutFences, inlineRanges);
77
+ }
78
+
79
+ // Every marked material of the spec, as [{ material, segments }], in writing.materials.items
80
+ // order: an item with a segments: field whose file readSegments can read. The segments file
81
+ // resolves relative to spec.dir, the way every writing path does. readSegments' own findings are
82
+ // dropped here on purpose: lint reports them (test 1 and 4) and check only runs once lint passes,
83
+ // so a station reading materials only needs the segments themselves. quotes and private both
84
+ // read this, so it is cached on ctx (one read per check run, whichever station asks first); a
85
+ // caller that passes no ctx simply reads the files again.
86
+ export function markedSegments(spec, ctx) {
87
+ if (ctx && ctx.markedSegments) return ctx.markedSegments;
88
+ const items = spec?.data?.writing?.materials?.items;
89
+ const out = [];
90
+ for (const item of Array.isArray(items) ? items : []) {
91
+ const segPath = str(item?.segments);
92
+ if (!segPath) continue;
93
+ const material = str(item?.id) || segPath;
94
+ const { segments } = readSegments(resolve(spec?.dir || ".", segPath), { materialId: str(item?.id) || undefined });
95
+ out.push({ material, segments: segments.filter((s) => s && typeof s === "object") });
96
+ }
97
+ if (ctx) ctx.markedSegments = out;
98
+ return out;
99
+ }
@@ -1,6 +1,11 @@
1
- // @supersuit/hyperspec/writing: the reading side of materials marking, for a tool outside this
2
- // package that labels materials (an agent, an editor, the writing engine's capture stage). It gets
1
+ // @supersuit/hyperspec/writing: the reading side of the writing profile, for a tool outside this
2
+ // package. For materials marking (an agent, an editor, the writing engine's capture stage) it gets
3
3
  // the closed label set and the same parse-and-validate the linter runs; labeling itself stays with
4
- // whoever calls this.
4
+ // whoever calls this. For scoped writer DNA it gets readScope, the same read of a scope folder
5
+ // (scope.md plus every golden, with their field findings) that `hyperspec lint` and
6
+ // `hyperspec dna measure` run, and measureFeatures, the pure function that turns golden passages
7
+ // into the numbers features.json records. Judging a draft against those numbers stays with the
8
+ // caller.
5
9
  export { MATERIAL_LABELS } from "./labels.mjs";
6
10
  export { readSegments } from "./segments.mjs";
11
+ export { readScope, measureFeatures } from "./dna.mjs";
@@ -18,9 +18,11 @@
18
18
  // (test 1), because dialogue cannot be specified without them. relationships stays optional: a
19
19
  // character may genuinely relate to no one yet, and nothing gives it a closed set or a count.
20
20
 
21
- import { statSync } from "node:fs";
21
+ import { readFileSync, realpathSync, statSync } from "node:fs";
22
+ import { basename, dirname, join, sep } from "node:path";
22
23
  import { str } from "./placeholder.mjs";
23
24
  import { readSegments } from "./segments.mjs";
25
+ import { readScope, isGoldenFileName, measureFeatures, featuresText, DNA_FORMAT } from "./dna.mjs";
24
26
 
25
27
  const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
26
28
  const list = (v) => (Array.isArray(v) ? v : []);
@@ -131,16 +133,151 @@ const UNRESOLVABLE_BECAUSE = {
131
133
 
132
134
  // ---------------------------------------------------------------- 2. dna ----------------------
133
135
 
136
+ // writing.dna.scope_dir (0.5, optional): the path (relative to the spec, like every other path in
137
+ // this file) to a scoped-DNA folder built by `hyperspec dna init`/`dna measure` (src/dna.mjs).
138
+ // The KEY being absent means none of the checks below run: 0.4 behavior, unchanged. A key that IS
139
+ // present but placeholder-ish (TODO, tbd, an empty string) fails on its own (test 1,
140
+ // writing-dna-scope-dir, naming the value) before any of this runs, since a real value is what
141
+ // every check below needs. Present with a real value, four things must all hold:
142
+ // - scope.md's writer/form/audience/purpose agree with dna.writer/dna.scope (test 1);
143
+ // - every dna.goldens[].path is one of the goldens readScope actually reads: resolved through
144
+ // any symlink, a golden-named file directly in the scope's REAL goldens/ folder (test 5,
145
+ // writing-dna-golden-leak, naming the golden and the scope). Anything else feeds the spec a
146
+ // passage that is never checked or measured under this scope. A goldens/ folder that itself
147
+ // resolves outside the scope is readScope's own finding (writing-dna-goldens-outside), which
148
+ // stands in for the per-golden findings it would otherwise cause;
149
+ // - every golden IN the scope passes its own field checks, exactly readScope's findings, reused
150
+ // rather than re-derived, with displayDir set to the scope_dir string the spec wrote (never a
151
+ // resolved filesystem path, so a finding here never names this machine's folders);
152
+ // - <scope_dir>/features.json is current: byte for byte what `dna measure` would write now
153
+ // (test 6), with the stale finding naming what differs.
154
+ function isInsideDir(parentAbs, childAbs) {
155
+ return childAbs === parentAbs || childAbs.startsWith(parentAbs + sep);
156
+ }
157
+
158
+ const realOrNull = (p) => { try { return realpathSync(p); } catch { return null; } };
159
+
160
+ // Why a listed golden is not one of the scope's goldens, as the words of the leak finding, or
161
+ // null when it is one. lexicalGoldens is <scope_dir>/goldens as written; realGoldens is where
162
+ // that folder really is.
163
+ function leakReason(lexicalGoldens, realGoldens, goldenAbs) {
164
+ const real = realOrNull(goldenAbs);
165
+ if (real && realGoldens && dirname(real) === realGoldens && isGoldenFileName(basename(real))) return null;
166
+ if (!isInsideDir(lexicalGoldens, goldenAbs)) return "outside";
167
+ if (!real || !realGoldens || !isInsideDir(realGoldens, real)) return "symlink";
168
+ return "not-a-golden";
169
+ }
170
+
171
+ function leakFinding(idPrefix, reason, p, i, scopeDirRaw) {
172
+ const where = `${scopeDirRaw}/goldens/`;
173
+ if (reason === "outside") {
174
+ return f(5, `${idPrefix}-golden-leak`, "fail",
175
+ `golden "${p}" feeds only work that shares its scope; it does not live under ${where}`,
176
+ `Move ${p} into ${where}, or point dna.goldens[${i + 1}].path at a golden already there.`);
177
+ }
178
+ if (reason === "symlink") {
179
+ return f(5, `${idPrefix}-golden-leak`, "fail",
180
+ `golden "${p}" is a symlink that resolves outside ${where} (or sits in a folder that does), so it feeds this scope a passage from somewhere else`,
181
+ `Replace the link at ${p} with the passage itself, as a file in ${where} with why, approved_by and source.`);
182
+ }
183
+ return f(5, `${idPrefix}-golden-leak`, "fail",
184
+ `golden "${p}" is not one of the scope's goldens: only .md files directly in ${where}, other than README.md, are read, checked and measured`,
185
+ `Put the passage in its own .md file directly in ${where}, with why, approved_by and source, and point dna.goldens[${i + 1}].path at it.`);
186
+ }
187
+
188
+ // One field of the spec's own dna claim against the same field read off scope.md. Silent when
189
+ // either side is empty: an empty spec-side value already fails its own presence check above (e.g.
190
+ // writing-dna-writer), and an empty disk-side value already fails as one of readScope's own
191
+ // findings (e.g. writing-dna-scope-file-writer); comparing two things when one is already known-broken
192
+ // would just be a second name for the same defect, not a second defect.
193
+ function scopeMismatch(out, idPrefix, scopeDirRaw, label, idSuffix, specVal, diskVal) {
194
+ const a = str(specVal);
195
+ const b = str(diskVal);
196
+ if (!a || !b || a.trim().toLowerCase() === b.trim().toLowerCase()) return;
197
+ out.push(f(1, `${idPrefix}-scope-mismatch-${idSuffix}`, "fail",
198
+ `writing.dna.scope_dir "${scopeDirRaw}": ${label} "${a}" does not match ${scopeDirRaw}/scope.md's ${label} "${b}"`,
199
+ `Make writing.dna's ${label} and ${scopeDirRaw}/scope.md's ${label} agree; one of them is wrong.`));
200
+ }
201
+
202
+ // Whether <scope>/features.json is what `dna measure` would write now. Returns "missing" (absent
203
+ // or not JSON), null (current), or the words naming what differs: goldens added, removed or
204
+ // changed by hash; scope fields that differ from scope.md; a dna format this linter does not
205
+ // know; features that differ from a fresh measurement (only named when the goldens themselves
206
+ // are unchanged, since changed goldens explain every number); and, when nothing more specific
207
+ // differs, bytes dna measure would not have written. Exported so the check command's dna station
208
+ // (src/stations/dna.mjs) skips on exactly the staleness lint fails a spec for, never a second
209
+ // definition of it.
210
+ export function featuresStaleness(featuresPath, diskScope, diskGoldens) {
211
+ let text;
212
+ let recorded;
213
+ try {
214
+ text = readFileSync(featuresPath, "utf8");
215
+ recorded = JSON.parse(text);
216
+ } catch {
217
+ return "missing";
218
+ }
219
+ if (!recorded || typeof recorded !== "object" || Array.isArray(recorded)) return "missing";
220
+ const parts = [];
221
+ const recordedGoldens = new Map(list(recorded.goldens).map((g) => [str(g?.path), str(g?.sha256)]));
222
+ const current = new Map(diskGoldens.map((g) => [g.path, g.sha256]));
223
+ const added = [...current.keys()].filter((p) => !recordedGoldens.has(p)).sort();
224
+ const removed = [...recordedGoldens.keys()].filter((p) => !current.has(p)).sort();
225
+ const changed = [...current.keys()].filter((p) => recordedGoldens.has(p) && recordedGoldens.get(p) !== current.get(p)).sort();
226
+ if (added.length) parts.push(`added ${added.join(", ")}`);
227
+ if (removed.length) parts.push(`removed ${removed.join(", ")}`);
228
+ if (changed.length) parts.push(`changed ${changed.join(", ")}`);
229
+ if (recorded.dna !== DNA_FORMAT) parts.push(`dna version ${JSON.stringify(recorded.dna ?? null)} is not the "${DNA_FORMAT}" this linter knows`);
230
+ if (!diskScope) return parts.length ? parts.join("; ") : null;
231
+
232
+ const recordedScope = isObj(recorded.scope) ? recorded.scope : {};
233
+ const scopeFields = ["writer", "form", "audience", "purpose"].filter((k) => recordedScope[k] !== diskScope[k]);
234
+ if (scopeFields.length) parts.push(`scope changed: ${scopeFields.join(", ")}`);
235
+ const features = measureFeatures(diskGoldens.map((g) => g.text));
236
+ if (!added.length && !removed.length && !changed.length) {
237
+ const recordedFeatures = isObj(recorded.features) ? recorded.features : {};
238
+ const keys = [...new Set([...Object.keys(features), ...Object.keys(recordedFeatures)])];
239
+ const differ = keys.filter((k) => JSON.stringify(features[k]) !== JSON.stringify(recordedFeatures[k]));
240
+ if (differ.length) parts.push(`features differ from a fresh measurement: ${differ.join(", ")}`);
241
+ }
242
+ if (!parts.length && text !== featuresText({ scope: diskScope, goldens: diskGoldens, features }).text) {
243
+ parts.push("the file is not byte for byte what `hyperspec dna measure` writes");
244
+ }
245
+ return parts.length ? parts.join("; ") : null;
246
+ }
247
+
134
248
  function dnaFields(raw, d, here, idPrefix) {
135
249
  const out = [];
136
250
  if (!str(raw.writer)) out.push(f(1, `${idPrefix}-writer`, "fail", "writing.dna has no writer", "Add writer:."));
137
251
  const scope = isObj(raw.scope) ? raw.scope : {};
252
+ // scope-<field> is the spec's own writing.dna.scope, the id 0.4.0 shipped; scope-file-<field>
253
+ // (src/dna.mjs) is a scope folder's scope.md.
138
254
  if (!str(scope.form)) out.push(f(1, `${idPrefix}-scope-form`, "fail", "writing.dna.scope has no form", "Add scope.form:."));
139
255
  if (!str(scope.audience)) out.push(f(1, `${idPrefix}-scope-audience`, "fail", "writing.dna.scope has no audience", "Add scope.audience:."));
140
256
  if (!str(scope.purpose)) out.push(f(1, `${idPrefix}-scope-purpose`, "fail", "writing.dna.scope has no purpose", "Add scope.purpose:."));
141
257
  const rulesPath = str(raw.rules);
142
258
  if (!rulesPath) out.push(f(1, `${idPrefix}-rules`, "fail", "writing.dna has no rules", "Add rules: the path to the always-on writing style."));
143
259
  else out.push(...pathFindings(here, rulesPath, `${idPrefix}-rules`, "dna.rules", "Fix the path, or add the file."));
260
+
261
+ // scope_dir is optional, so its KEY being absent from writing.dna is never a finding (the
262
+ // whole scope_dir section below simply does not run). But a key that IS present with a
263
+ // placeholder-ish value (TODO, tbd, an empty string, ...) is a different situation: the operator
264
+ // wrote something and str() silently reads it as "not there", which would otherwise make a
265
+ // half-filled skeleton lint clean by accident. That gets its own finding, naming the value, and
266
+ // is why this check reads raw.scope_dir directly rather than through scopeDirRaw.
267
+ if (raw.scope_dir !== undefined && !str(raw.scope_dir)) {
268
+ out.push(f(1, `${idPrefix}-scope-dir`, "fail",
269
+ `writing.dna.scope_dir "${raw.scope_dir}" looks like a placeholder`,
270
+ "Point scope_dir: at a real scope folder (built with hyperspec dna init), or remove the field entirely; it is optional."));
271
+ }
272
+ const scopeDirRaw = str(raw.scope_dir);
273
+ const scopeDirAbs = scopeDirRaw ? here(scopeDirRaw) : null;
274
+ const disk = scopeDirRaw ? readScope(scopeDirAbs, { displayDir: scopeDirRaw }) : null;
275
+ const diskIds = new Set((disk?.findings ?? []).map((x) => x.id));
276
+ const goldensOutside = diskIds.has("writing-dna-goldens-outside");
277
+ const lexicalGoldens = scopeDirRaw ? here(join(scopeDirRaw, "goldens")) : null;
278
+ const realScope = scopeDirRaw ? realOrNull(scopeDirAbs) : null;
279
+ const realGoldens = realScope ? join(realScope, "goldens") : null;
280
+
144
281
  const goldens = list(raw.goldens);
145
282
  if (!goldens.length) out.push(f(1, `${idPrefix}-goldens`, "fail", "writing.dna has no goldens", "Add at least one golden under dna.goldens."));
146
283
  goldens.forEach((g, i) => {
@@ -148,7 +285,44 @@ function dnaFields(raw, d, here, idPrefix) {
148
285
  if (!p) out.push(f(1, `${idPrefix}-golden-${i}-path`, "fail", `dna.goldens[${i + 1}] has no path`, "Add path: to the golden."));
149
286
  else out.push(...pathFindings(here, p, `${idPrefix}-golden-${i}`, "golden", "Fix the path, or add the golden file."));
150
287
  if (!str(g?.why)) out.push(f(6, `${idPrefix}-golden-${i}-why`, "fail", `golden "${p || `#${i + 1}`}" has no why`, "Add why: what it shows that an adjective could not."));
288
+ // Checked only once the path resolves to a real file: a missing or non-file path is already
289
+ // reported above under test 6, and is not also a leak.
290
+ if (scopeDirRaw && p && pathKind(here, p) === "file") {
291
+ const reason = leakReason(lexicalGoldens, realGoldens, here(p));
292
+ // A goldens/ folder that resolves outside the scope already has its own finding, which
293
+ // names the cause for every golden listed through it.
294
+ const coveredByFolder = goldensOutside && isInsideDir(lexicalGoldens, here(p));
295
+ if (reason && !coveredByFolder) out.push(leakFinding(idPrefix, reason, p, i, scopeDirRaw));
296
+ }
151
297
  });
298
+
299
+ if (scopeDirRaw) {
300
+ const { scope: diskScope, goldens: diskGoldens, findings: diskFindings } = disk;
301
+ out.push(...diskFindings);
302
+
303
+ if (diskScope) {
304
+ scopeMismatch(out, idPrefix, scopeDirRaw, "writer", "writer", raw.writer, diskScope.writer);
305
+ scopeMismatch(out, idPrefix, scopeDirRaw, "form", "form", scope.form, diskScope.form);
306
+ scopeMismatch(out, idPrefix, scopeDirRaw, "audience", "audience", scope.audience, diskScope.audience);
307
+ scopeMismatch(out, idPrefix, scopeDirRaw, "purpose", "purpose", scope.purpose, diskScope.purpose);
308
+ }
309
+
310
+ // With the goldens folder unreadable or somewhere else, there is nothing to compare
311
+ // features.json against; that folder's own finding says what to fix first.
312
+ if (!goldensOutside && !diskIds.has("writing-dna-goldens-missing")) {
313
+ const stale = featuresStaleness(join(scopeDirAbs, "features.json"), diskScope, diskGoldens);
314
+ if (stale === "missing") {
315
+ out.push(f(6, `${idPrefix}-features-missing`, "fail",
316
+ `writing.dna.scope_dir "${scopeDirRaw}" has no features.json (or it is not valid JSON)`,
317
+ `Run \`hyperspec dna measure ${scopeDirRaw}\`.`));
318
+ } else if (stale) {
319
+ out.push(f(6, `${idPrefix}-features-stale`, "fail",
320
+ `writing.dna.scope_dir "${scopeDirRaw}"'s features.json is stale: ${stale}`,
321
+ `Run \`hyperspec dna measure ${scopeDirRaw}\` again.`));
322
+ }
323
+ }
324
+ }
325
+
152
326
  return out;
153
327
  }
154
328
 
@@ -193,6 +367,28 @@ function audienceFields(raw, d, here, idPrefix) {
193
367
  if (!list(raw.knows).some((x) => str(x))) out.push(f(1, `${idPrefix}-knows`, "fail", "writing.audience has no knows", "List at least one term the reader already has."));
194
368
  const reader = str(raw.reader);
195
369
  if (!READER_VALUES.includes(reader)) out.push(f(1, `${idPrefix}-reader`, "fail", `writing.audience.reader is "${reader || "(none)"}"`, "Set reader to person or agent."));
370
+
371
+ // terms (0.6, optional): the reader-may-not-know terms the check command's `terms` station
372
+ // checks are defined at first use. The KEY being absent is fine and does not change the
373
+ // audience block's completeness (the same "absence is 0.4/0.5 behavior" shape scope_dir and
374
+ // characters use elsewhere in this file): nothing below runs, and the block can still be
375
+ // complete with no terms: field at all. Present, though, it must be a list of real text: a
376
+ // scalar is named as not a list, an empty list fails as a whole (the "declared but empty" shape
377
+ // audience.knows already uses one line up), and each non-string, placeholder or empty entry is
378
+ // named on its own (mirroring the top-level `rejects` list's own item check in src/rules.mjs).
379
+ if (raw.terms !== undefined) {
380
+ if (!Array.isArray(raw.terms)) {
381
+ out.push(f(1, `${idPrefix}-terms`, "fail", "writing.audience.terms is not a list", "Write terms as a list, one term per line starting \"- \"."));
382
+ } else if (!raw.terms.length) {
383
+ out.push(f(1, `${idPrefix}-terms`, "fail", "writing.audience.terms is present but has no real entries", "List at least one term, or remove terms: entirely; it is optional."));
384
+ } else {
385
+ raw.terms.forEach((x, i) => {
386
+ if (x != null && typeof x !== "string") out.push(f(1, `${idPrefix}-terms`, "fail", `writing.audience.terms item ${i + 1} is not a plain string`, "Write each term as plain text, e.g. a word or short phrase."));
387
+ else if (!str(x)) out.push(f(1, `${idPrefix}-terms`, "fail", `writing.audience.terms item ${i + 1} is ${typeof x === "string" && x.trim() ? "a placeholder" : "empty"}`, "Replace it with the real term, or remove the item."));
388
+ });
389
+ }
390
+ }
391
+
196
392
  return out;
197
393
  }
198
394