@supersuit/hyperspec 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +127 -0
  2. package/README.md +46 -1
  3. package/SPEC.md +5 -3
  4. package/WRITING.md +475 -40
  5. package/bin/hyperspec.mjs +157 -3
  6. package/examples/writing/dna/essay-new-managers-teach/features.json +56 -0
  7. package/examples/writing/dna/essay-new-managers-teach/goldens/README.md +14 -0
  8. package/examples/writing/dna/essay-new-managers-teach/goldens/close.md +9 -0
  9. package/examples/writing/dna/essay-new-managers-teach/goldens/opening.md +9 -0
  10. package/examples/writing/dna/essay-new-managers-teach/goldens/status.md +10 -0
  11. package/examples/writing/dna/essay-new-managers-teach/scope.md +11 -0
  12. package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +10 -0
  13. package/examples/writing/essay/materials/team-survey.md.segments.jsonl +6 -0
  14. package/examples/writing/essay/materials/voice-memo.md.segments.jsonl +7 -0
  15. package/examples/writing/essay.hyperspec.md +24 -13
  16. package/examples/writing/story/materials/bakery-visit.md +1 -0
  17. package/examples/writing/story/materials/bakery-visit.md.segments.jsonl +13 -0
  18. package/examples/writing/story/materials/notes.md +2 -0
  19. package/examples/writing/story/materials/notes.md.segments.jsonl +7 -0
  20. package/examples/writing/story/materials/scene-list.md.segments.jsonl +14 -0
  21. package/examples/writing/story.hyperspec.md +8 -5
  22. package/package.json +2 -1
  23. package/src/blobs.mjs +1 -1
  24. package/src/compare.mjs +6 -6
  25. package/src/dna.mjs +471 -0
  26. package/src/fsutil.mjs +1 -1
  27. package/src/labels.mjs +6 -0
  28. package/src/reproduce.mjs +5 -5
  29. package/src/segments.mjs +407 -0
  30. package/src/writing-exports.mjs +11 -0
  31. package/src/writing-fields.mjs +279 -6
  32. package/src/writing-template.mjs +16 -1
  33. package/src/writing.mjs +4 -4
  34. package/examples/writing/essay/goldens/close.md +0 -2
  35. package/examples/writing/essay/goldens/opening.md +0 -2
package/src/dna.mjs ADDED
@@ -0,0 +1,471 @@
1
+ // Scoped writer DNA (hyperspec 0.5). A writer does not have one voice: the same person writes
2
+ // differently for a theology journal and a landing page, so their DNA is kept per SCOPE (form,
3
+ // audience, purpose) as a folder of goldens, real passages a human approved, each carrying a note
4
+ // on why it is golden. This module never judges a passage and never calls a model: it reads a
5
+ // scope, checks the closed set of required fields on each golden (findings, in the lint shape),
6
+ // and measures a scope's style features deterministically from its goldens' text. Whether a
7
+ // generated passage matches those features is a later build's job (a station); this module only
8
+ // reads and counts.
9
+ //
10
+ // A scope is a folder: <dir>/scope.md (frontmatter writer, form, audience, purpose, optional
11
+ // notes) and <dir>/goldens/*.md, one golden per file (frontmatter why, approved_by, source,
12
+ // optional approved_on; the body is the passage, verbatim). readScope reads both and returns
13
+ // everything downstream code needs: the scope's own fields, every golden's data, and every
14
+ // finding, in one pass. readGoldens is the goldens-only half, exported on its own because a
15
+ // caller that already has the scope's fields (or does not need them) can read just the goldens.
16
+ //
17
+ // Sentence and paragraph splitting reuse splitSegments from segments.mjs (paragraph mode for
18
+ // paragraphs, sentence mode within each paragraph for sentences), so a golden's paragraph count
19
+ // and its sentence count are never two different notions of where a boundary falls.
20
+
21
+ import { readFileSync, readdirSync, realpathSync } from "node:fs";
22
+ import { join, relative } from "node:path";
23
+ import { parseSkillFile } from "@supersuit/superskill/yaml";
24
+ import { sha256 } from "./hash.mjs";
25
+ import { splitSegments } from "./segments.mjs";
26
+ import { str } from "./placeholder.mjs";
27
+ import { scalar } from "./template.mjs";
28
+ import { writeFileAtomic } from "./fsutil.mjs";
29
+
30
+ const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
31
+
32
+ // -------------------------------------------------------------------------------------------
33
+ // reading a scope
34
+
35
+ // The golden files in <dir>/goldens/: every *.md file except README.md (case-insensitive),
36
+ // which is the guidance file `dna init` writes into an otherwise-empty goldens/ folder. Sorted
37
+ // by filename, so golden order (and therefore features.json's goldens list and word pooling
38
+ // order) never depends on the filesystem's own directory-listing order.
39
+ // isGoldenFileName is the one definition of which names in goldens/ are goldens, shared with the
40
+ // spec linter, which refuses a listed golden the reader would never read.
41
+ export function isGoldenFileName(name) {
42
+ const lower = String(name).toLowerCase();
43
+ return lower.endsWith(".md") && lower !== "readme.md";
44
+ }
45
+
46
+ function listGoldenFiles(goldensDir) {
47
+ return readdirSync(goldensDir, { withFileTypes: true })
48
+ .filter((e) => e.isFile() && isGoldenFileName(e.name))
49
+ .map((e) => e.name)
50
+ .sort();
51
+ }
52
+
53
+ // The folder a scope's goldens really live in, when <dir>/goldens resolves somewhere else: a
54
+ // goldens/ folder that is a symlink (to another scope's goldens, say) would otherwise carry that
55
+ // scope's passages into this one under this scope's name. Returned relative to the scope's own
56
+ // real folder, so a message built from it never names a folder on this machine. null when the
57
+ // folder is where it should be, or cannot be resolved at all (a missing folder is reported as
58
+ // goldens-missing by the listing below).
59
+ function goldensElsewhere(dir) {
60
+ try {
61
+ const dirReal = realpathSync(dir);
62
+ const goldensReal = realpathSync(join(dir, "goldens"));
63
+ return goldensReal === join(dirReal, "goldens") ? null : relative(dirReal, goldensReal);
64
+ } catch {
65
+ return null;
66
+ }
67
+ }
68
+
69
+ // readGoldens(dir): reads <dir>/goldens/*.md. Returns { goldens, findings }. Every finding is in
70
+ // the lint shape (test, id, severity, message, fix), ids prefixed writing-dna-, messages naming
71
+ // the scope dir (displayDir, defaulting to dir itself) and the golden file (its path relative to
72
+ // dir, e.g. "goldens/opening.md"). A golden with a missing required field is still returned in
73
+ // goldens (so a caller can see what IS there); only a golden this function cannot even read
74
+ // (an fs error) is left out, since there is nothing to return for it.
75
+ export function readGoldens(dir, { displayDir } = {}) {
76
+ const shown = displayDir ?? dir;
77
+ const goldensDir = join(dir, "goldens");
78
+ const findings = [];
79
+
80
+ const elsewhere = goldensElsewhere(dir);
81
+ if (elsewhere !== null) {
82
+ findings.push(f(5, "writing-dna-goldens-outside", "fail",
83
+ `scope "${shown}": its goldens/ folder resolves to ${elsewhere}, outside the scope; a golden feeds only work that shares its scope`,
84
+ `Replace ${shown}/goldens with a real folder holding this scope's own goldens, copying in any passage that belongs to this scope too.`));
85
+ return { goldens: [], findings };
86
+ }
87
+
88
+ let names;
89
+ try {
90
+ names = listGoldenFiles(goldensDir);
91
+ } catch {
92
+ findings.push(f(1, "writing-dna-goldens-missing", "fail",
93
+ `scope "${shown}": goldens folder "goldens/" does not exist or cannot be read`,
94
+ `Run \`hyperspec dna init ${shown} --writer W --form F --audience A --purpose P\`, or create goldens/ yourself.`));
95
+ return { goldens: [], findings };
96
+ }
97
+
98
+ if (names.length === 0) {
99
+ findings.push(f(1, "writing-dna-goldens-empty", "fail",
100
+ `scope "${shown}" has no goldens in goldens/`,
101
+ "Add at least one golden file under goldens/."));
102
+ return { goldens: [], findings };
103
+ }
104
+
105
+ const goldens = [];
106
+ for (const name of names) {
107
+ const relPath = `goldens/${name}`;
108
+ let buf;
109
+ try {
110
+ buf = readFileSync(join(goldensDir, name));
111
+ } catch {
112
+ findings.push(f(1, "writing-dna-golden-unreadable", "fail",
113
+ `scope "${shown}", golden "${relPath}" cannot be read`,
114
+ `Fix or remove ${relPath}.`));
115
+ continue;
116
+ }
117
+
118
+ const sha = sha256(buf);
119
+ // parseSkillFile never throws: malformed or missing frontmatter comes back as data: {} (and,
120
+ // for missing frontmatter, the whole file as body), so every required field below is simply
121
+ // reported missing rather than needing a separate "malformed" finding. The one exception is
122
+ // frontmatter that OPENS (a first line of ---) and never closes: parseSkillFile has nowhere to
123
+ // end the block, so body comes back empty and the real passage text is invisible, meaning the
124
+ // three field checks below and golden-empty would all fire for what is one defect. So report
125
+ // that once, plainly, instead of a cascade that reads like four unrelated problems.
126
+ const { data, body, error } = parseSkillFile(buf.toString("utf8"));
127
+ if (error === "unterminated frontmatter") {
128
+ findings.push(f(1, "writing-dna-golden-frontmatter", "fail",
129
+ `scope "${shown}", golden "${relPath}"'s frontmatter opens with --- but never closes`,
130
+ `Close ${relPath}'s frontmatter with a second --- line.`));
131
+ continue;
132
+ }
133
+ const why = str(data.why);
134
+ const approvedBy = str(data.approved_by);
135
+ const source = str(data.source);
136
+ const approvedOn = str(data.approved_on);
137
+ // "The body is the passage, verbatim": trimmed only to drop the one blank line every golden
138
+ // carries between its closing "---" and the first line of the passage, never touched inside.
139
+ const text = body.trim();
140
+
141
+ if (!text) {
142
+ findings.push(f(1, "writing-dna-golden-empty", "fail",
143
+ `scope "${shown}", golden "${relPath}" has no passage text`,
144
+ "Add the passage as the file body."));
145
+ }
146
+ if (!why) {
147
+ findings.push(f(6, "writing-dna-golden-why", "fail",
148
+ `scope "${shown}", golden "${relPath}" has no why`,
149
+ "Add why: the move this golden teaches."));
150
+ }
151
+ if (!approvedBy) {
152
+ findings.push(f(4, "writing-dna-golden-approved-by", "fail",
153
+ `scope "${shown}", golden "${relPath}" has no approved_by`,
154
+ "Add approved_by: the person slug who approved this passage."));
155
+ } else if (approvedBy.toLowerCase().startsWith("agent:")) {
156
+ findings.push(f(4, "writing-dna-golden-approved-by-agent", "fail",
157
+ `scope "${shown}", golden "${relPath}": approved_by "${approvedBy}" is an agent; golden means a human approved it`,
158
+ "Set approved_by: to the person slug who actually approved this passage, not an agent."));
159
+ }
160
+ if (!source) {
161
+ findings.push(f(4, "writing-dna-golden-source", "fail",
162
+ `scope "${shown}", golden "${relPath}" has no source`,
163
+ "Add source: where this passage came from."));
164
+ }
165
+
166
+ goldens.push({ path: relPath, why, approved_by: approvedBy, source, approved_on: approvedOn, text, sha256: sha });
167
+ }
168
+
169
+ return { goldens, findings };
170
+ }
171
+
172
+ // readScope(dir): reads <dir>/scope.md and <dir>/goldens/*.md. Returns { scope, goldens,
173
+ // findings }. scope is { writer, form, audience, purpose, notes } (each a string, "" if missing
174
+ // or a placeholder), or null when scope.md itself could not be read at all. goldens and findings
175
+ // are as readGoldens above; findings from scope.md's own fields come first, then every golden's.
176
+ // displayDir overrides what messages call the scope (for a caller, e.g. the spec linter, that
177
+ // wants the scope_dir string as it was written, not a resolved filesystem path); it defaults to
178
+ // dir.
179
+ export function readScope(dir, { displayDir } = {}) {
180
+ const shown = displayDir ?? dir;
181
+ const findings = [];
182
+ let scope = null;
183
+
184
+ let raw;
185
+ try {
186
+ raw = readFileSync(join(dir, "scope.md"), "utf8");
187
+ } catch {
188
+ findings.push(f(1, "writing-dna-scope-missing", "fail",
189
+ `scope "${shown}": scope.md does not exist or cannot be read`,
190
+ `Run \`hyperspec dna init ${shown} --writer W --form F --audience A --purpose P\` to create it.`));
191
+ }
192
+
193
+ if (raw !== undefined) {
194
+ const { data, error } = parseSkillFile(raw);
195
+ if (error) {
196
+ findings.push(f(1, "writing-dna-scope-missing", "fail",
197
+ `scope "${shown}": scope.md is not valid (${error})`,
198
+ "Fix scope.md's frontmatter: writer, form, audience, purpose."));
199
+ } else {
200
+ const writer = str(data.writer);
201
+ const form = str(data.form);
202
+ const audience = str(data.audience);
203
+ const purpose = str(data.purpose);
204
+ scope = { writer, form, audience, purpose, notes: str(data.notes) };
205
+ for (const [field, value] of [["writer", writer], ["form", form], ["audience", audience], ["purpose", purpose]]) {
206
+ if (!value) {
207
+ findings.push(f(1, `writing-dna-scope-file-${field}`, "fail",
208
+ `scope "${shown}": scope.md has no ${field}`,
209
+ `Add ${field}: to scope.md.`));
210
+ }
211
+ }
212
+ }
213
+ }
214
+
215
+ const { goldens, findings: goldenFindings } = readGoldens(dir, { displayDir: shown });
216
+ findings.push(...goldenFindings);
217
+
218
+ return { scope, goldens, findings };
219
+ }
220
+
221
+ // -------------------------------------------------------------------------------------------
222
+ // measuring features, from text alone: no filesystem, no judgment, fully deterministic.
223
+
224
+ // Every number this module writes is rounded the same way: half away from zero, at the 3rd
225
+ // decimal. -0 never survives (JSON.stringify(-0) already prints "0", but this also normalizes
226
+ // the in-memory number, so a strict equality check on the returned object sees +0 too).
227
+ function round3(x) {
228
+ const r = Math.round(x * 1000) / 1000;
229
+ return r === 0 ? 0 : r;
230
+ }
231
+
232
+ const mean = (arr) => (arr.length === 0 ? 0 : arr.reduce((a, b) => a + b, 0) / arr.length);
233
+ function median(sortedArr) {
234
+ const n = sortedArr.length;
235
+ if (n === 0) return 0;
236
+ const mid = Math.floor(n / 2);
237
+ return n % 2 === 0 ? (sortedArr[mid - 1] + sortedArr[mid]) / 2 : sortedArr[mid];
238
+ }
239
+ // Nearest-rank percentile: rank = ceil(p/100 * n), 1-based, clamped to [1, n]. p90 on 10 values
240
+ // [1..10] is rank ceil(9) = 9, the 9th smallest, i.e. 9 itself; this is the "nearest rank"
241
+ // definition (not linear interpolation).
242
+ function percentileNearestRank(sortedArr, p) {
243
+ const n = sortedArr.length;
244
+ if (n === 0) return 0;
245
+ const rank = Math.min(Math.max(Math.ceil((p / 100) * n), 1), n);
246
+ return sortedArr[rank - 1];
247
+ }
248
+
249
+ // A word is a maximal run of Unicode letters, digits, and apostrophes (straight ' or curly U+2019),
250
+ // lowercased. A leading or trailing apostrophe (a quote mark hugging a word) is swept up into
251
+ // the token by this same rule; it is simply never "between letters", so it never makes the word
252
+ // count as a contraction below.
253
+ const WORD_RE = /[\p{L}\p{N}'\u2019]+/gu;
254
+ function wordsOf(text) {
255
+ const m = text.match(WORD_RE);
256
+ return m ? m.map((w) => w.toLowerCase()) : [];
257
+ }
258
+
259
+ // A contraction: the word contains an apostrophe (straight or curly) with a letter immediately
260
+ // before AND after it. "don't" and "y'all's" qualify; "'tis" (apostrophe at position 0, nothing
261
+ // before it) and a stray quote-wrapped word do not.
262
+ function isContraction(word) {
263
+ for (let i = 1; i < word.length - 1; i++) {
264
+ const ch = word[i];
265
+ if ((ch === "'" || ch === "\u2019") && /\p{L}/u.test(word[i - 1]) && /\p{L}/u.test(word[i + 1])) return true;
266
+ }
267
+ return false;
268
+ }
269
+
270
+ const FIRST_PERSON_SINGULAR = new Set(["i", "me", "my", "mine", "myself"]);
271
+ const FIRST_PERSON_PLURAL = new Set(["we", "us", "our", "ours", "ourselves"]);
272
+ const SECOND_PERSON = new Set(["you", "your", "yours", "yourself", "yourselves"]);
273
+
274
+ // A signature-word candidate: 4+ characters, letters and apostrophes only (no bare digit runs,
275
+ // so "2024" is never a signature word), and not a stopword. Frequency and the >=2 floor are
276
+ // applied afterward, over the whole scope's pooled words.
277
+ const CANDIDATE_RE = /^[\p{L}'\u2019]+$/u;
278
+ function isSignatureCandidate(word) {
279
+ return word.length >= 4 && CANDIDATE_RE.test(word) && !STOPWORD_SET.has(word.replace(/\u2019/g, "'"));
280
+ }
281
+
282
+ function signatureWords(allWords) {
283
+ const freq = new Map();
284
+ for (const w of allWords) {
285
+ if (!isSignatureCandidate(w)) continue;
286
+ freq.set(w, (freq.get(w) ?? 0) + 1);
287
+ }
288
+ const candidates = [...freq.entries()].filter(([, count]) => count >= 2);
289
+ // Frequency descending; ties broken alphabetically. Plain code-point comparison, not
290
+ // localeCompare, so the order never depends on the running Node build's ICU data.
291
+ candidates.sort((a, b) => b[1] - a[1] || (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : 0));
292
+ return candidates.slice(0, 15).map(([w]) => w);
293
+ }
294
+
295
+ const countChar = (text, re) => (text.match(re) || []).length;
296
+
297
+ // measureFeatures(texts): texts is an array of golden passage strings (readScope's goldens[].text),
298
+ // in any order; the result does not depend on it. Pure and deterministic: the same texts
299
+ // always produce the same features object, byte for byte once JSON.stringify'd. Every rate is
300
+ // per 1000 words; every number is rounded via round3.
301
+ export function measureFeatures(texts) {
302
+ const list = Array.isArray(texts) ? texts : [];
303
+ const allWords = [];
304
+ const sentenceWordCounts = [];
305
+ const paragraphSentenceCounts = [];
306
+ const paragraphWordCounts = [];
307
+ let commas = 0, semicolons = 0, colons = 0, emDashes = 0, enDashes = 0;
308
+ let exclaims = 0, questions = 0, parens = 0, quotes = 0;
309
+
310
+ for (const raw of list) {
311
+ const text = typeof raw === "string" ? raw : "";
312
+ commas += countChar(text, /,/g);
313
+ semicolons += countChar(text, /;/g);
314
+ colons += countChar(text, /:/g);
315
+ emDashes += countChar(text, /\u2014/g);
316
+ enDashes += countChar(text, /\u2013/g);
317
+ exclaims += countChar(text, /!/g);
318
+ questions += countChar(text, /\?/g);
319
+ parens += countChar(text, /[()]/g);
320
+ quotes += countChar(text, /["\u201C\u201D]/g);
321
+
322
+ const paragraphs = splitSegments(text, { by: "paragraph" }).map((s) => s.text);
323
+ for (const paragraph of paragraphs) {
324
+ const sentences = splitSegments(paragraph, { by: "sentence" }).map((s) => s.text);
325
+ let paragraphWords = 0;
326
+ for (const sentence of sentences) {
327
+ const sWords = wordsOf(sentence);
328
+ sentenceWordCounts.push(sWords.length);
329
+ paragraphWords += sWords.length;
330
+ allWords.push(...sWords);
331
+ }
332
+ paragraphSentenceCounts.push(sentences.length);
333
+ paragraphWordCounts.push(paragraphWords);
334
+ }
335
+ }
336
+
337
+ const totalWords = allWords.length;
338
+ const perThousand = (count) => (totalWords === 0 ? 0 : round3((count / totalWords) * 1000));
339
+ const sortedSentenceWords = [...sentenceWordCounts].sort((a, b) => a - b);
340
+ const contractionCount = allWords.filter(isContraction).length;
341
+ const fpSingular = allWords.filter((w) => FIRST_PERSON_SINGULAR.has(w)).length;
342
+ const fpPlural = allWords.filter((w) => FIRST_PERSON_PLURAL.has(w)).length;
343
+ const sp = allWords.filter((w) => SECOND_PERSON.has(w)).length;
344
+ const meanWordLength = totalWords === 0 ? 0 : round3(allWords.reduce((sum, w) => sum + w.length, 0) / totalWords);
345
+
346
+ return {
347
+ word_count: totalWords,
348
+ sentence_length: {
349
+ mean: round3(mean(sentenceWordCounts)),
350
+ median: round3(median(sortedSentenceWords)),
351
+ p90: round3(percentileNearestRank(sortedSentenceWords, 90)),
352
+ },
353
+ paragraph_length: {
354
+ mean_sentences: round3(mean(paragraphSentenceCounts)),
355
+ mean_words: round3(mean(paragraphWordCounts)),
356
+ },
357
+ rates_per_1000_words: {
358
+ comma: perThousand(commas),
359
+ semicolon: perThousand(semicolons),
360
+ colon: perThousand(colons),
361
+ em_dash: perThousand(emDashes),
362
+ en_dash: perThousand(enDashes),
363
+ exclamation: perThousand(exclaims),
364
+ question_mark: perThousand(questions),
365
+ parentheses: perThousand(parens),
366
+ quotation_marks: perThousand(quotes),
367
+ },
368
+ contraction_rate: perThousand(contractionCount),
369
+ first_person_singular_rate: perThousand(fpSingular),
370
+ first_person_plural_rate: perThousand(fpPlural),
371
+ second_person_rate: perThousand(sp),
372
+ mean_word_length: meanWordLength,
373
+ signature_words: signatureWords(allWords),
374
+ };
375
+ }
376
+
377
+ // features.json: 2-space JSON, a trailing newline, keys in the fixed order below, goldens sorted
378
+ // by path, so the same goldens always produce byte-identical bytes. DNA_FORMAT is the version of
379
+ // this shape, recorded in the file as "dna".
380
+ export const DNA_FORMAT = "0.1";
381
+
382
+ // featuresText({ scope, goldens, features }): the exact bytes writeFeatures writes, and the data
383
+ // they encode. The spec linter builds these from the scope as it reads now and compares them to
384
+ // the file, so a features.json is current only when it is what dna measure would write today.
385
+ export function featuresText({ scope, goldens, features }) {
386
+ const sortedGoldens = [...(goldens ?? [])].sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
387
+ const data = {
388
+ dna: DNA_FORMAT,
389
+ scope: {
390
+ writer: scope?.writer ?? "",
391
+ form: scope?.form ?? "",
392
+ audience: scope?.audience ?? "",
393
+ purpose: scope?.purpose ?? "",
394
+ },
395
+ goldens: sortedGoldens.map((g) => ({ path: g.path, sha256: g.sha256 })),
396
+ features,
397
+ };
398
+ return { data, text: `${JSON.stringify(data, null, 2)}\n` };
399
+ }
400
+
401
+ // writeFeatures(scopeDir, { scope, goldens, features }): writes <scopeDir>/features.json, the
402
+ // bytes featuresText returns. Returns { path, data }.
403
+ export function writeFeatures(scopeDir, input) {
404
+ const { data, text } = featuresText(input);
405
+ const path = join(scopeDir, "features.json");
406
+ writeFileAtomic(path, text);
407
+ return { path, data };
408
+ }
409
+
410
+ // -------------------------------------------------------------------------------------------
411
+ // dna init's skeleton, kept beside the module it belongs to (the same split as
412
+ // template.mjs/writingTemplate: string-building lives with the reader/measurer it seeds, file
413
+ // I/O and refusal checks live in the CLI).
414
+
415
+ export function scopeTemplate({ writer, form, audience, purpose } = {}) {
416
+ return `---
417
+ writer: ${scalar(String(writer))}
418
+ form: ${scalar(String(form))}
419
+ audience: ${scalar(String(audience))}
420
+ purpose: ${scalar(String(purpose))}
421
+ ---
422
+
423
+ # Writer DNA scope
424
+
425
+ Goldens live in goldens/. Run \`hyperspec dna measure <scope-dir>\` once every golden there has
426
+ why, approved_by and source.
427
+ `;
428
+ }
429
+
430
+ export const GOLDENS_README = `# Goldens
431
+
432
+ Each file in this folder except this one is a golden: a passage the writer marked as right,
433
+ filed under this scope.
434
+
435
+ Frontmatter:
436
+ - why (required): what makes it golden, the move it teaches.
437
+ - approved_by (required): a person slug. Golden means a human approved it; agent:* is refused.
438
+ - source (required): where the passage came from.
439
+ - approved_on (optional): a date.
440
+
441
+ The body is the passage, verbatim.
442
+
443
+ Run \`hyperspec dna measure <scope-dir>\` once every golden here has why, approved_by and source.
444
+ `;
445
+
446
+ // -------------------------------------------------------------------------------------------
447
+ // stopwords: excluded from signature_words so the list names what this writer says, not what
448
+ // every writer says. About 150 common English function words. Deliberately no contractions
449
+ // (don't, isn't, ...): a stopword is matched against a lowercased word as tokenized above, and
450
+ // keeping this list plain avoids it silently missing half its entries because a golden happened
451
+ // to use a curly apostrophe where this file used a straight one, or the other way around.
452
+ export const STOPWORDS = Object.freeze([
453
+ "a", "about", "above", "across", "after", "again", "against", "all", "almost", "along",
454
+ "already", "also", "although", "always", "am", "among", "an", "and", "another", "any",
455
+ "anyone", "anything", "are", "around", "as", "at", "away", "be", "because", "been", "before",
456
+ "being", "below", "between", "both", "but", "by", "can", "cannot", "could", "did", "do",
457
+ "does", "doing", "done", "down", "during", "each", "either", "else", "ever", "every",
458
+ "everyone", "everything", "few", "for", "from", "further", "had", "has", "have", "having",
459
+ "he", "her", "here", "hers", "herself", "him", "himself", "his", "how", "however", "i", "if",
460
+ "in", "into", "is", "it", "its", "itself", "just", "may", "me", "might", "mine", "more",
461
+ "most", "much", "must", "my", "myself", "neither", "never", "next", "no", "nobody", "none",
462
+ "nor", "not", "nothing", "now", "of", "off", "often", "on", "once", "one", "only", "onto",
463
+ "or", "other", "others", "our", "ours", "ourselves", "out", "over", "own", "same", "shall",
464
+ "she", "should", "since", "so", "some", "someone", "something", "sometimes", "still", "such",
465
+ "than", "that", "the", "their", "theirs", "them", "themselves", "then", "there", "these",
466
+ "they", "this", "those", "though", "through", "to", "too", "toward", "towards", "under",
467
+ "until", "up", "upon", "us", "very", "was", "we", "were", "what", "when", "where", "whether",
468
+ "which", "while", "who", "whom", "whose", "why", "will", "with", "within", "without", "would",
469
+ "yet", "you", "your", "yours", "yourself", "yourselves",
470
+ ]);
471
+ const STOPWORD_SET = new Set(STOPWORDS);
package/src/fsutil.mjs CHANGED
@@ -2,7 +2,7 @@ import { renameSync, unlinkSync, writeFileSync } from "node:fs";
2
2
  import { randomBytes } from "node:crypto";
3
3
  import { relative, resolve, sep } from "node:path";
4
4
 
5
- // True when `target`, resolved against `dir`, stays inside `dir` — refuses a `..`-escape and an
5
+ // True when `target`, resolved against `dir`, stays inside `dir`. It refuses a `..`-escape and an
6
6
  // absolute path pointing elsewhere. Both a relative `target` (including one that climbs out via
7
7
  // `../`) and an already-absolute `target` are handled the same way, since `path.resolve(dir,
8
8
  // target)` already treats an absolute second argument as overriding the first: either way, the
package/src/labels.mjs ADDED
@@ -0,0 +1,6 @@
1
+ // The closed vocabulary every segment of a marked material is labeled from (hyperspec 0.4). It lives in
2
+ // this leaf module, which imports nothing, so src/segments.mjs can read it without importing
3
+ // src/writing.mjs: writing.mjs imports writing-fields.mjs, which imports segments.mjs, and reading
4
+ // the labels through writing.mjs made that a cycle. writing.mjs re-exports it for callers that
5
+ // already import it from there.
6
+ export const MATERIAL_LABELS = Object.freeze(["claim", "story", "quote", "stance", "question", "aside", "private"]);
package/src/reproduce.mjs CHANGED
@@ -11,7 +11,7 @@ const MALFORMED = "recorded hash is not a SHA-256 hash (64 lowercase hex charact
11
11
  const blobWhy = (hex, why) => (present(hex) && !isSha256(hex) ? MALFORMED : why);
12
12
 
13
13
  // Spec requirement `reproduce`: replay the record and hash-check it. Never invokes a model,
14
- // never runs a command, never regenerates a byte of content — every blob this looks at already
14
+ // never runs a command, never regenerates a byte of content. Every blob this looks at already
15
15
  // exists in the store, and this only confirms the recipe's own claims about it still hold.
16
16
  //
17
17
  // Checks, in order (and every one is reported, nothing stops the walk early): every input blob
@@ -98,8 +98,8 @@ export function reproduce(recipePath, { store, restore = false } = {}) {
98
98
  // 5. The output file on disk, if present, hashes to output.sha256. Absent is vacuously fine:
99
99
  // reproduce doesn't require the file to already exist, only that it agrees when it does.
100
100
  //
101
- // The recipe is a plain JSON file on disk — exactly the kind of claim reproduce exists to
102
- // distrust — so output.path is never trusted blind. A hand-edited or corrupted recipe could
101
+ // The recipe is a plain JSON file on disk, exactly the kind of claim reproduce exists to
102
+ // distrust, so output.path is never trusted blind. A hand-edited or corrupted recipe could
103
103
  // name a path outside the recipe's own directory (a `../` climb, or an absolute path); refuse
104
104
  // before touching disk at all, rather than reading from or (worse, under --restore) writing to
105
105
  // wherever it points.
@@ -120,12 +120,12 @@ export function reproduce(recipePath, { store, restore = false } = {}) {
120
120
  }
121
121
 
122
122
  // restore: true writes the output file from its blob, but only when the output blob
123
- // itself verified (step 4) — restoring from an unverified blob would just write
123
+ // itself verified (step 4); restoring from an unverified blob would just write
124
124
  // different wrong bytes. Covers both a lost file (never existed / deleted) and an
125
125
  // edited one. The write is atomic (temp file + rename, same pattern as putBlob), so a
126
126
  // crash mid-write never leaves the file in a state that is neither the old nor the new
127
127
  // content, and a write failure is caught and reported as a failing step rather than
128
- // thrown out of reproduce() — this function always returns a structured result.
128
+ // thrown out of reproduce(): this function always returns a structured result.
129
129
  const needsRestore = !fileExists || !matches;
130
130
  if (restore && needsRestore && outputBlobOk) {
131
131
  const blob = getBlob(root, outputSha);