@supersuit/hyperspec 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +166 -0
  2. package/README.md +62 -1
  3. package/SPEC.md +4 -4
  4. package/WRITING.md +770 -9
  5. package/bin/hyperspec.mjs +252 -0
  6. package/examples/writing/essay/claims.jsonl +9 -0
  7. package/examples/writing/essay/draft.md +82 -0
  8. package/examples/writing/essay/judge/doctor.packet.json +108 -0
  9. package/examples/writing/essay/judge/lineup.packet.json +64 -0
  10. package/examples/writing/essay/judge/persona.packet.json +73 -0
  11. package/examples/writing/essay/judge/reader.packet.json +93 -0
  12. package/examples/writing/essay/learn/first-draft.md +84 -0
  13. package/examples/writing/essay/learn/learn.packet.json +106 -0
  14. package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +2 -2
  15. package/examples/writing/essay/sample-verdicts/doctor.verdict.json +43 -0
  16. package/examples/writing/essay/sample-verdicts/learn.verdict.json +30 -0
  17. package/examples/writing/essay/sample-verdicts/lineup.verdict.json +6 -0
  18. package/examples/writing/essay/sample-verdicts/persona.verdict.json +4 -0
  19. package/examples/writing/essay/sample-verdicts/reader.verdict.json +7 -0
  20. package/examples/writing/essay.hyperspec.md +23 -7
  21. package/examples/writing/story/claims.jsonl +9 -0
  22. package/examples/writing/story/draft.md +267 -0
  23. package/examples/writing/story/judge/attribution.packet.json +194 -0
  24. package/examples/writing/story/judge/doctor.packet.json +108 -0
  25. package/examples/writing/story/judge/knowledge.packet.json +77 -0
  26. package/examples/writing/story/judge/persona.packet.json +73 -0
  27. package/examples/writing/story/judge/reader.packet.json +94 -0
  28. package/examples/writing/story/sample-verdicts/attribution.verdict.json +81 -0
  29. package/examples/writing/story/sample-verdicts/doctor.verdict.json +43 -0
  30. package/examples/writing/story/sample-verdicts/knowledge.verdict.json +4 -0
  31. package/examples/writing/story/sample-verdicts/persona.verdict.json +20 -0
  32. package/examples/writing/story/sample-verdicts/reader.verdict.json +16 -0
  33. package/examples/writing/story.hyperspec.md +24 -5
  34. package/package.json +1 -1
  35. package/src/check.mjs +191 -0
  36. package/src/dna.mjs +4 -1
  37. package/src/draft.mjs +26 -0
  38. package/src/judge.mjs +386 -0
  39. package/src/judges/attribution.mjs +360 -0
  40. package/src/judges/doctor.mjs +126 -0
  41. package/src/judges/index.mjs +31 -0
  42. package/src/judges/knowledge.mjs +111 -0
  43. package/src/judges/lineup.mjs +272 -0
  44. package/src/judges/persona.mjs +137 -0
  45. package/src/judges/reader.mjs +111 -0
  46. package/src/learn.mjs +422 -0
  47. package/src/ledger.mjs +108 -0
  48. package/src/sentences.mjs +81 -0
  49. package/src/stations/claims.mjs +155 -0
  50. package/src/stations/dna.mjs +126 -0
  51. package/src/stations/form.mjs +115 -0
  52. package/src/stations/index.mjs +27 -0
  53. package/src/stations/links.mjs +275 -0
  54. package/src/stations/private.mjs +117 -0
  55. package/src/stations/quotes.mjs +170 -0
  56. package/src/stations/terms.mjs +131 -0
  57. package/src/stations/util.mjs +99 -0
  58. package/src/writing-fields.mjs +26 -2
  59. package/src/writing.mjs +1 -1
@@ -0,0 +1,131 @@
1
+ // Station "terms" (hyperspec 0.6). Checks that every term in
2
+ // writing.audience.terms (the new optional list: terms the piece uses that the reader may not
3
+ // know) is defined the first time it appears in the draft. Pure and deterministic like every
4
+ // station: no model call, no judgment about which words are jargon, just the closed rule below.
5
+ //
6
+ // Matching a term against the draft is case-insensitive and whole-word (Unicode-aware: a match
7
+ // cannot start or end mid-word, using the same letter/number/apostrophe class dna.mjs's own word
8
+ // definition uses, so "AI" does not match inside "said" and a multi-word term like "context
9
+ // window" matches only as that exact phrase).
10
+ //
11
+ // "Defined at its first appearance" is read as: the sentence containing
12
+ // the term's first appearance, OR the sentence right after it, contains an occurrence of the term
13
+ // followed (within the next 6 words of that same sentence, or by a colon among them) by one of
14
+ // "is", "means", "refers to"; OR the term is immediately followed by a parenthetical ("term
15
+ // (short gloss)"). Sentence boundaries come from splitSegments(text, { by: "sentence" }) in
16
+ // src/segments.mjs, the same splitter every other station that reads sentences uses, so "which
17
+ // sentence is this in" never disagrees between stations.
18
+ //
19
+ // A term also listed in writing.audience.knows is never flagged, whatever the draft does with it:
20
+ // the reader is assumed to already have it. A term that never appears in the draft at all is not
21
+ // flagged either (nothing to define); writing.audience.terms with no real entries, or missing
22
+ // entirely, skips the whole station.
23
+ //
24
+ // Fenced code blocks and inline code spans are masked out (src/stations/util.mjs's maskCode)
25
+ // before any of the above runs: a term's only appearance inside example syntax like
26
+ // `` `ledger: check` `` is not a prose use, and the mechanical colon/`is` rule below would
27
+ // otherwise read that code span's own punctuation as a definition it never gave. Masking
28
+ // preserves every character offset, so every 1-based line number reported below is still a line
29
+ // of the ORIGINAL draft, never the masked copy.
30
+
31
+ import { str } from "../placeholder.mjs";
32
+ import { splitSegments } from "../segments.mjs";
33
+ import { lineAt, maskCode } from "./util.mjs";
34
+
35
+ export const name = "terms";
36
+
37
+ // The same word/number class dna.mjs's WORD_RE tokenizes with (letters, digits, straight or
38
+ // curly apostrophe for contractions), Unicode-aware so an accented or non-Latin term's edges are
39
+ // read correctly too.
40
+ const WORD_CHARS = "\\p{L}\\p{N}'’";
41
+
42
+ function escapeRegExp(s) {
43
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
44
+ }
45
+
46
+ // A case-insensitive, whole-phrase, Unicode-aware matcher for `term`: it cannot start or end
47
+ // mid-word (a lookbehind/lookahead against the word-char class above), but the term itself may
48
+ // contain internal spaces (a multi-word term matches only as that exact phrase, spaces and all).
49
+ function termRegex(term, flags) {
50
+ const body = escapeRegExp(term.trim()).replace(/\s+/g, "\\s+");
51
+ return new RegExp(`(?<![${WORD_CHARS}])${body}(?![${WORD_CHARS}])`, flags);
52
+ }
53
+
54
+ // Every { word, start, end } token in `text`, in order, using the same word-char class as above.
55
+ function tokenize(text) {
56
+ const re = new RegExp(`[${WORD_CHARS}]+`, "gu");
57
+ const out = [];
58
+ let m;
59
+ while ((m = re.exec(text))) out.push({ word: m[0].toLowerCase(), start: m.index, end: m.index + m[0].length });
60
+ return out;
61
+ }
62
+
63
+ // Whether `sentenceText` contains a definition of `term`: some occurrence of the term followed,
64
+ // within the next 6 words of the SAME sentence (or by a colon somewhere among them), by "is",
65
+ // "means" or "refers"+"to"; or immediately (skipping only whitespace) followed by "(".
66
+ function definesTerm(sentenceText, term) {
67
+ const re = termRegex(term, "giu");
68
+ let m;
69
+ while ((m = re.exec(sentenceText))) {
70
+ const tailStart = m.index + m[0].length;
71
+ const tail = sentenceText.slice(tailStart);
72
+ if (/^\s*\(/.test(tail)) return true;
73
+
74
+ const tokens = tokenize(tail).slice(0, 6);
75
+ const windowEnd = tokens.length ? tokens[tokens.length - 1].end : tail.length;
76
+ const window = tail.slice(0, windowEnd);
77
+ if (window.includes(":")) return true;
78
+ const words = tokens.map((t) => t.word);
79
+ if (words.includes("is") || words.includes("means")) return true;
80
+ if (words.some((w, i) => w === "refers" && words[i + 1] === "to")) return true;
81
+ // re is not global-sticky across iterations by construction (lastIndex advances past this
82
+ // match automatically since the "g" flag is set), so a term repeated in one sentence is
83
+ // still checked occurrence by occurrence rather than looping forever.
84
+ }
85
+ return false;
86
+ }
87
+
88
+ export function run(spec, draft) {
89
+ const audience = spec?.data?.writing?.audience ?? {};
90
+ const terms = Array.isArray(audience.terms) ? audience.terms.map(str).filter(Boolean) : [];
91
+ if (!terms.length) {
92
+ return { station: name, status: "skip", findings: [], reason: "writing.audience.terms is empty or not set" };
93
+ }
94
+
95
+ const knows = new Set((Array.isArray(audience.knows) ? audience.knows : []).map((x) => str(x).toLowerCase()).filter(Boolean));
96
+
97
+ const scanText = maskCode(draft.text);
98
+ const sentences = splitSegments(scanText, { by: "sentence" });
99
+ const findings = [];
100
+ const usedIds = new Set();
101
+
102
+ for (const term of terms) {
103
+ if (knows.has(term.toLowerCase())) continue;
104
+
105
+ const first = termRegex(term, "iu").exec(scanText);
106
+ if (!first) continue; // the term never appears outside code; nothing to define
107
+
108
+ const pos = first.index;
109
+ const idx = sentences.findIndex((s) => pos >= s.start && pos < s.end);
110
+ const candidates = idx === -1 ? [] : [sentences[idx], sentences[idx + 1]].filter(Boolean);
111
+ const defined = candidates.some((s) => definesTerm(s.text, term));
112
+ if (defined) continue;
113
+
114
+ const slug = term.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "") || "term";
115
+ let id = `station-terms-undefined-${slug}`;
116
+ let n = 2;
117
+ while (usedIds.has(id)) { id = `station-terms-undefined-${slug}-${n}`; n += 1; }
118
+ usedIds.add(id);
119
+
120
+ findings.push({
121
+ station: name,
122
+ id,
123
+ severity: "fail",
124
+ line: lineAt(draft.text, pos),
125
+ message: `term "${term}" is not defined at its first appearance (need "is", "means", "refers to", a colon within a few words, or an immediate parenthetical, in that sentence or the next)`,
126
+ fix: `Define "${term}" where it first appears, e.g. "${term} is ..." or "${term} (a short gloss)".`,
127
+ });
128
+ }
129
+
130
+ return { station: name, status: findings.length ? "fail" : "pass", findings };
131
+ }
@@ -0,0 +1,99 @@
1
+ import { resolve } from "node:path";
2
+ import { str } from "../placeholder.mjs";
3
+ import { readSegments } from "../segments.mjs";
4
+
5
+ // Shared helpers for the deterministic stations (hyperspec 0.6). "One place this pattern is
6
+ // defined" (src/placeholder.mjs's own header comment), so a helper more than one station needs
7
+ // lives here rather than being copied station to station: lineAt, truncate, maskCode, maskRanges
8
+ // and markedSegments.
9
+
10
+ // The 1-based line containing character offset `pos` of `text`. Counts "\n" characters directly
11
+ // rather than reading draft.lines: draft.lines' own CRLF handling ("\r" left attached to the
12
+ // previous line's entry, or stripped, depending on how src/check.mjs built it) is none of any
13
+ // station's business, and counting "\n" occurrences gives the right line number either way, since
14
+ // every line, CRLF or not, still carries exactly one "\n".
15
+ export function lineAt(text, pos) {
16
+ let line = 1;
17
+ for (let i = 0; i < pos && i < text.length; i++) if (text[i] === "\n") line++;
18
+ return line;
19
+ }
20
+
21
+ // Trims `text`, then truncates to at most `max` characters (an ellipsis replacing the last
22
+ // character when it is longer), matching the general finding-message rule that a quoted span of
23
+ // the draft is at most 80 characters. Trimming FIRST means a span with leading/trailing
24
+ // whitespace never wastes truncation budget on it.
25
+ export function truncate(text, max) {
26
+ const t = String(text).trim();
27
+ return t.length > max ? `${t.slice(0, max - 1)}…` : t;
28
+ }
29
+
30
+ // `text` with every [start, end) range in `ranges` blanked out: every character in the range
31
+ // becomes a space, EXCEPT "\n", which is left alone. Preserving newlines (rather than blanking
32
+ // them too) is what keeps every character offset, and therefore every 1-based line number
33
+ // computed afterward, identical to `text` before masking, even when a masked range spans more
34
+ // than one line (a fenced code block, in particular): blanking a "\n" would merge two lines into
35
+ // one for every offset that comes after it, which is exactly the bug a single-line-only masker
36
+ // would not have caught. Used by every "mask X out before scanning for Y" step across the
37
+ // stations, so a masked-and-rescanned span can never be matched a second time.
38
+ export function maskRanges(text, ranges) {
39
+ let out = text;
40
+ for (const { start, end } of ranges) {
41
+ const blanked = out.slice(start, end).replace(/[^\n]/g, " ");
42
+ out = out.slice(0, start) + blanked + out.slice(end);
43
+ }
44
+ return out;
45
+ }
46
+
47
+ // A fenced code block: a line starting (after up to 3 spaces of indent) with 3 or more backticks
48
+ // or tildes, some content, then a line starting with at least as many of the SAME fence
49
+ // character. Non-greedy so back-to-back fences in one draft are matched as separate blocks
50
+ // rather than one block swallowing everything between the first open and the last close.
51
+ const FENCE_RE = /^[ \t]{0,3}(`{3,}|~{3,})[^\n]*\n[\s\S]*?^[ \t]{0,3}\1[ \t]*$/gm;
52
+
53
+ // An inline code span: a single backtick, a run of non-backtick, non-newline characters, a
54
+ // closing backtick. Does not attempt CommonMark's full rule for spans opened with a run of two or
55
+ // more backticks (rare in practice, and unneeded for the case this exists to fix: an inline
56
+ // `like this` mention of syntax that must not be read as prose).
57
+ const INLINE_CODE_RE = /`[^`\n]+`/g;
58
+
59
+ // `text` with every fenced code block and inline code span blanked out (offsets and line numbers
60
+ // preserved, see maskRanges above). Both `links` (a URL shown as example syntax inside a fence or
61
+ // a span is not a real, followable link) and `terms` (a term's only appearance inside `` `code`
62
+ // `` is not a prose use, and must not count toward "does this term appear/get defined") mask code
63
+ // out before doing anything else, so later masking passes (Markdown links, reference definitions,
64
+ // bare URLs, sentence splitting) never see inside a code span or fence.
65
+ export function maskCode(text) {
66
+ const fenceRanges = [];
67
+ FENCE_RE.lastIndex = 0;
68
+ let m;
69
+ while ((m = FENCE_RE.exec(text))) fenceRanges.push({ start: m.index, end: m.index + m[0].length });
70
+ const withoutFences = maskRanges(text, fenceRanges);
71
+
72
+ const inlineRanges = [];
73
+ INLINE_CODE_RE.lastIndex = 0;
74
+ let im;
75
+ while ((im = INLINE_CODE_RE.exec(withoutFences))) inlineRanges.push({ start: im.index, end: im.index + im[0].length });
76
+ return maskRanges(withoutFences, inlineRanges);
77
+ }
78
+
79
+ // Every marked material of the spec, as [{ material, segments }], in writing.materials.items
80
+ // order: an item with a segments: field whose file readSegments can read. The segments file
81
+ // resolves relative to spec.dir, the way every writing path does. readSegments' own findings are
82
+ // dropped here on purpose: lint reports them (test 1 and 4) and check only runs once lint passes,
83
+ // so a station reading materials only needs the segments themselves. quotes and private both
84
+ // read this, so it is cached on ctx (one read per check run, whichever station asks first); a
85
+ // caller that passes no ctx simply reads the files again.
86
+ export function markedSegments(spec, ctx) {
87
+ if (ctx && ctx.markedSegments) return ctx.markedSegments;
88
+ const items = spec?.data?.writing?.materials?.items;
89
+ const out = [];
90
+ for (const item of Array.isArray(items) ? items : []) {
91
+ const segPath = str(item?.segments);
92
+ if (!segPath) continue;
93
+ const material = str(item?.id) || segPath;
94
+ const { segments } = readSegments(resolve(spec?.dir || ".", segPath), { materialId: str(item?.id) || undefined });
95
+ out.push({ material, segments: segments.filter((s) => s && typeof s === "object") });
96
+ }
97
+ if (ctx) ctx.markedSegments = out;
98
+ return out;
99
+ }
@@ -204,8 +204,10 @@ function scopeMismatch(out, idPrefix, scopeDirRaw, label, idSuffix, specVal, dis
204
204
  // changed by hash; scope fields that differ from scope.md; a dna format this linter does not
205
205
  // know; features that differ from a fresh measurement (only named when the goldens themselves
206
206
  // are unchanged, since changed goldens explain every number); and, when nothing more specific
207
- // differs, bytes dna measure would not have written.
208
- function featuresStaleness(featuresPath, diskScope, diskGoldens) {
207
+ // differs, bytes dna measure would not have written. Exported so the check command's dna station
208
+ // (src/stations/dna.mjs) skips on exactly the staleness lint fails a spec for, never a second
209
+ // definition of it.
210
+ export function featuresStaleness(featuresPath, diskScope, diskGoldens) {
209
211
  let text;
210
212
  let recorded;
211
213
  try {
@@ -365,6 +367,28 @@ function audienceFields(raw, d, here, idPrefix) {
365
367
  if (!list(raw.knows).some((x) => str(x))) out.push(f(1, `${idPrefix}-knows`, "fail", "writing.audience has no knows", "List at least one term the reader already has."));
366
368
  const reader = str(raw.reader);
367
369
  if (!READER_VALUES.includes(reader)) out.push(f(1, `${idPrefix}-reader`, "fail", `writing.audience.reader is "${reader || "(none)"}"`, "Set reader to person or agent."));
370
+
371
+ // terms (0.6, optional): the reader-may-not-know terms the check command's `terms` station
372
+ // checks are defined at first use. The KEY being absent is fine and does not change the
373
+ // audience block's completeness (the same "absence is 0.4/0.5 behavior" shape scope_dir and
374
+ // characters use elsewhere in this file): nothing below runs, and the block can still be
375
+ // complete with no terms: field at all. Present, though, it must be a list of real text: a
376
+ // scalar is named as not a list, an empty list fails as a whole (the "declared but empty" shape
377
+ // audience.knows already uses one line up), and each non-string, placeholder or empty entry is
378
+ // named on its own (mirroring the top-level `rejects` list's own item check in src/rules.mjs).
379
+ if (raw.terms !== undefined) {
380
+ if (!Array.isArray(raw.terms)) {
381
+ out.push(f(1, `${idPrefix}-terms`, "fail", "writing.audience.terms is not a list", "Write terms as a list, one term per line starting \"- \"."));
382
+ } else if (!raw.terms.length) {
383
+ out.push(f(1, `${idPrefix}-terms`, "fail", "writing.audience.terms is present but has no real entries", "List at least one term, or remove terms: entirely; it is optional."));
384
+ } else {
385
+ raw.terms.forEach((x, i) => {
386
+ if (x != null && typeof x !== "string") out.push(f(1, `${idPrefix}-terms`, "fail", `writing.audience.terms item ${i + 1} is not a plain string`, "Write each term as plain text, e.g. a word or short phrase."));
387
+ else if (!str(x)) out.push(f(1, `${idPrefix}-terms`, "fail", `writing.audience.terms item ${i + 1} is ${typeof x === "string" && x.trim() ? "a placeholder" : "empty"}`, "Replace it with the real term, or remove the item."));
388
+ });
389
+ }
390
+ }
391
+
368
392
  return out;
369
393
  }
370
394
 
package/src/writing.mjs CHANGED
@@ -36,7 +36,7 @@ const required = (block, fiction) => block !== "characters" || fiction;
36
36
  // least one field. This is deliberately shallow: it is the bar for "something was written here",
37
37
  // not the bar for "this block is correct", which is what checkOwner and the field rules in
38
38
  // writing-fields.mjs are for.
39
- function blockPresent(block, raw) {
39
+ export function blockPresent(block, raw) {
40
40
  if (block === "characters") return Array.isArray(raw) && raw.length > 0;
41
41
  if (block === "materials") return isObj(raw) && Array.isArray(raw.items) && raw.items.length > 0;
42
42
  return isObj(raw) && Object.keys(raw).length > 0;