@supersuit/hyperspec 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +143 -0
- package/README.md +50 -2
- package/SPEC.md +5 -5
- package/WRITING.md +510 -17
- package/bin/hyperspec.mjs +177 -2
- package/examples/writing/dna/essay-new-managers-teach/features.json +56 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/README.md +14 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/close.md +9 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/opening.md +9 -0
- package/examples/writing/dna/essay-new-managers-teach/goldens/status.md +10 -0
- package/examples/writing/dna/essay-new-managers-teach/scope.md +11 -0
- package/examples/writing/essay/claims.jsonl +9 -0
- package/examples/writing/essay/draft.md +82 -0
- package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +2 -2
- package/examples/writing/essay.hyperspec.md +32 -13
- package/examples/writing/story/claims.jsonl +9 -0
- package/examples/writing/story/draft.md +267 -0
- package/examples/writing/story.hyperspec.md +20 -5
- package/package.json +1 -1
- package/src/check.mjs +245 -0
- package/src/dna.mjs +474 -0
- package/src/stations/claims.mjs +150 -0
- package/src/stations/dna.mjs +126 -0
- package/src/stations/form.mjs +115 -0
- package/src/stations/index.mjs +27 -0
- package/src/stations/links.mjs +275 -0
- package/src/stations/private.mjs +117 -0
- package/src/stations/quotes.mjs +168 -0
- package/src/stations/terms.mjs +131 -0
- package/src/stations/util.mjs +99 -0
- package/src/writing-exports.mjs +8 -3
- package/src/writing-fields.mjs +197 -1
- package/src/writing-template.mjs +8 -0
- package/examples/writing/essay/goldens/close.md +0 -2
- package/examples/writing/essay/goldens/opening.md +0 -2
package/src/dna.mjs
ADDED
|
@@ -0,0 +1,474 @@
|
|
|
1
|
+
// Scoped writer DNA (hyperspec 0.5). A writer does not have one voice: the same person writes
|
|
2
|
+
// differently for a theology journal and a landing page, so their DNA is kept per SCOPE (form,
|
|
3
|
+
// audience, purpose) as a folder of goldens, real passages a human approved, each carrying a note
|
|
4
|
+
// on why it is golden. This module never judges a passage and never calls a model: it reads a
|
|
5
|
+
// scope, checks the closed set of required fields on each golden (findings, in the lint shape),
|
|
6
|
+
// and measures a scope's style features deterministically from its goldens' text. Whether a
|
|
7
|
+
// generated passage matches those features is a later build's job (a station); this module only
|
|
8
|
+
// reads and counts.
|
|
9
|
+
//
|
|
10
|
+
// A scope is a folder: <dir>/scope.md (frontmatter writer, form, audience, purpose, optional
|
|
11
|
+
// notes) and <dir>/goldens/*.md, one golden per file (frontmatter why, approved_by, source,
|
|
12
|
+
// optional approved_on; the body is the passage, verbatim). readScope reads both and returns
|
|
13
|
+
// everything downstream code needs: the scope's own fields, every golden's data, and every
|
|
14
|
+
// finding, in one pass. readGoldens is the goldens-only half, exported on its own because a
|
|
15
|
+
// caller that already has the scope's fields (or does not need them) can read just the goldens.
|
|
16
|
+
//
|
|
17
|
+
// Sentence and paragraph splitting reuse splitSegments from segments.mjs (paragraph mode for
|
|
18
|
+
// paragraphs, sentence mode within each paragraph for sentences), so a golden's paragraph count
|
|
19
|
+
// and its sentence count are never two different notions of where a boundary falls.
|
|
20
|
+
|
|
21
|
+
import { readFileSync, readdirSync, realpathSync } from "node:fs";
|
|
22
|
+
import { join, relative } from "node:path";
|
|
23
|
+
import { parseSkillFile } from "@supersuit/superskill/yaml";
|
|
24
|
+
import { sha256 } from "./hash.mjs";
|
|
25
|
+
import { splitSegments } from "./segments.mjs";
|
|
26
|
+
import { str } from "./placeholder.mjs";
|
|
27
|
+
import { scalar } from "./template.mjs";
|
|
28
|
+
import { writeFileAtomic } from "./fsutil.mjs";
|
|
29
|
+
|
|
30
|
+
const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
|
|
31
|
+
|
|
32
|
+
// -------------------------------------------------------------------------------------------
|
|
33
|
+
// reading a scope
|
|
34
|
+
|
|
35
|
+
// The golden files in <dir>/goldens/: every *.md file except README.md (case-insensitive),
|
|
36
|
+
// which is the guidance file `dna init` writes into an otherwise-empty goldens/ folder. Sorted
|
|
37
|
+
// by filename, so golden order (and therefore features.json's goldens list and word pooling
|
|
38
|
+
// order) never depends on the filesystem's own directory-listing order.
|
|
39
|
+
// isGoldenFileName is the one definition of which names in goldens/ are goldens, shared with the
|
|
40
|
+
// spec linter, which refuses a listed golden the reader would never read.
|
|
41
|
+
export function isGoldenFileName(name) {
|
|
42
|
+
const lower = String(name).toLowerCase();
|
|
43
|
+
return lower.endsWith(".md") && lower !== "readme.md";
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function listGoldenFiles(goldensDir) {
|
|
47
|
+
return readdirSync(goldensDir, { withFileTypes: true })
|
|
48
|
+
.filter((e) => e.isFile() && isGoldenFileName(e.name))
|
|
49
|
+
.map((e) => e.name)
|
|
50
|
+
.sort();
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// The folder a scope's goldens really live in, when <dir>/goldens resolves somewhere else: a
|
|
54
|
+
// goldens/ folder that is a symlink (to another scope's goldens, say) would otherwise carry that
|
|
55
|
+
// scope's passages into this one under this scope's name. Returned relative to the scope's own
|
|
56
|
+
// real folder, so a message built from it never names a folder on this machine. null when the
|
|
57
|
+
// folder is where it should be, or cannot be resolved at all (a missing folder is reported as
|
|
58
|
+
// goldens-missing by the listing below).
|
|
59
|
+
function goldensElsewhere(dir) {
|
|
60
|
+
try {
|
|
61
|
+
const dirReal = realpathSync(dir);
|
|
62
|
+
const goldensReal = realpathSync(join(dir, "goldens"));
|
|
63
|
+
return goldensReal === join(dirReal, "goldens") ? null : relative(dirReal, goldensReal);
|
|
64
|
+
} catch {
|
|
65
|
+
return null;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// readGoldens(dir): reads <dir>/goldens/*.md. Returns { goldens, findings }. Every finding is in
|
|
70
|
+
// the lint shape (test, id, severity, message, fix), ids prefixed writing-dna-, messages naming
|
|
71
|
+
// the scope dir (displayDir, defaulting to dir itself) and the golden file (its path relative to
|
|
72
|
+
// dir, e.g. "goldens/opening.md"). A golden with a missing required field is still returned in
|
|
73
|
+
// goldens (so a caller can see what IS there); only a golden this function cannot even read
|
|
74
|
+
// (an fs error) is left out, since there is nothing to return for it.
|
|
75
|
+
export function readGoldens(dir, { displayDir } = {}) {
|
|
76
|
+
const shown = displayDir ?? dir;
|
|
77
|
+
const goldensDir = join(dir, "goldens");
|
|
78
|
+
const findings = [];
|
|
79
|
+
|
|
80
|
+
const elsewhere = goldensElsewhere(dir);
|
|
81
|
+
if (elsewhere !== null) {
|
|
82
|
+
findings.push(f(5, "writing-dna-goldens-outside", "fail",
|
|
83
|
+
`scope "${shown}": its goldens/ folder resolves to ${elsewhere}, outside the scope; a golden feeds only work that shares its scope`,
|
|
84
|
+
`Replace ${shown}/goldens with a real folder holding this scope's own goldens, copying in any passage that belongs to this scope too.`));
|
|
85
|
+
return { goldens: [], findings };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
let names;
|
|
89
|
+
try {
|
|
90
|
+
names = listGoldenFiles(goldensDir);
|
|
91
|
+
} catch {
|
|
92
|
+
findings.push(f(1, "writing-dna-goldens-missing", "fail",
|
|
93
|
+
`scope "${shown}": goldens folder "goldens/" does not exist or cannot be read`,
|
|
94
|
+
`Run \`hyperspec dna init ${shown} --writer W --form F --audience A --purpose P\`, or create goldens/ yourself.`));
|
|
95
|
+
return { goldens: [], findings };
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
if (names.length === 0) {
|
|
99
|
+
findings.push(f(1, "writing-dna-goldens-empty", "fail",
|
|
100
|
+
`scope "${shown}" has no goldens in goldens/`,
|
|
101
|
+
"Add at least one golden file under goldens/."));
|
|
102
|
+
return { goldens: [], findings };
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const goldens = [];
|
|
106
|
+
for (const name of names) {
|
|
107
|
+
const relPath = `goldens/${name}`;
|
|
108
|
+
let buf;
|
|
109
|
+
try {
|
|
110
|
+
buf = readFileSync(join(goldensDir, name));
|
|
111
|
+
} catch {
|
|
112
|
+
findings.push(f(1, "writing-dna-golden-unreadable", "fail",
|
|
113
|
+
`scope "${shown}", golden "${relPath}" cannot be read`,
|
|
114
|
+
`Fix or remove ${relPath}.`));
|
|
115
|
+
continue;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const sha = sha256(buf);
|
|
119
|
+
// parseSkillFile never throws: malformed or missing frontmatter comes back as data: {} (and,
|
|
120
|
+
// for missing frontmatter, the whole file as body), so every required field below is simply
|
|
121
|
+
// reported missing rather than needing a separate "malformed" finding. The one exception is
|
|
122
|
+
// frontmatter that OPENS (a first line of ---) and never closes: parseSkillFile has nowhere to
|
|
123
|
+
// end the block, so body comes back empty and the real passage text is invisible, meaning the
|
|
124
|
+
// three field checks below and golden-empty would all fire for what is one defect. So report
|
|
125
|
+
// that once, plainly, instead of a cascade that reads like four unrelated problems.
|
|
126
|
+
const { data, body, error } = parseSkillFile(buf.toString("utf8"));
|
|
127
|
+
if (error === "unterminated frontmatter") {
|
|
128
|
+
findings.push(f(1, "writing-dna-golden-frontmatter", "fail",
|
|
129
|
+
`scope "${shown}", golden "${relPath}"'s frontmatter opens with --- but never closes`,
|
|
130
|
+
`Close ${relPath}'s frontmatter with a second --- line.`));
|
|
131
|
+
continue;
|
|
132
|
+
}
|
|
133
|
+
const why = str(data.why);
|
|
134
|
+
const approvedBy = str(data.approved_by);
|
|
135
|
+
const source = str(data.source);
|
|
136
|
+
const approvedOn = str(data.approved_on);
|
|
137
|
+
// "The body is the passage, verbatim": trimmed only to drop the one blank line every golden
|
|
138
|
+
// carries between its closing "---" and the first line of the passage, never touched inside.
|
|
139
|
+
const text = body.trim();
|
|
140
|
+
|
|
141
|
+
if (!text) {
|
|
142
|
+
findings.push(f(1, "writing-dna-golden-empty", "fail",
|
|
143
|
+
`scope "${shown}", golden "${relPath}" has no passage text`,
|
|
144
|
+
"Add the passage as the file body."));
|
|
145
|
+
}
|
|
146
|
+
if (!why) {
|
|
147
|
+
findings.push(f(6, "writing-dna-golden-why", "fail",
|
|
148
|
+
`scope "${shown}", golden "${relPath}" has no why`,
|
|
149
|
+
"Add why: the move this golden teaches."));
|
|
150
|
+
}
|
|
151
|
+
if (!approvedBy) {
|
|
152
|
+
findings.push(f(4, "writing-dna-golden-approved-by", "fail",
|
|
153
|
+
`scope "${shown}", golden "${relPath}" has no approved_by`,
|
|
154
|
+
"Add approved_by: the person slug who approved this passage."));
|
|
155
|
+
} else if (approvedBy.toLowerCase().startsWith("agent:")) {
|
|
156
|
+
findings.push(f(4, "writing-dna-golden-approved-by-agent", "fail",
|
|
157
|
+
`scope "${shown}", golden "${relPath}": approved_by "${approvedBy}" is an agent; golden means a human approved it`,
|
|
158
|
+
"Set approved_by: to the person slug who actually approved this passage, not an agent."));
|
|
159
|
+
}
|
|
160
|
+
if (!source) {
|
|
161
|
+
findings.push(f(4, "writing-dna-golden-source", "fail",
|
|
162
|
+
`scope "${shown}", golden "${relPath}" has no source`,
|
|
163
|
+
"Add source: where this passage came from."));
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
goldens.push({ path: relPath, why, approved_by: approvedBy, source, approved_on: approvedOn, text, sha256: sha });
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return { goldens, findings };
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
// readScope(dir): reads <dir>/scope.md and <dir>/goldens/*.md. Returns { scope, goldens,
|
|
173
|
+
// findings }. scope is { writer, form, audience, purpose, notes } (each a string, "" if missing
|
|
174
|
+
// or a placeholder), or null when scope.md itself could not be read at all. goldens and findings
|
|
175
|
+
// are as readGoldens above; findings from scope.md's own fields come first, then every golden's.
|
|
176
|
+
// displayDir overrides what messages call the scope (for a caller, e.g. the spec linter, that
|
|
177
|
+
// wants the scope_dir string as it was written, not a resolved filesystem path); it defaults to
|
|
178
|
+
// dir.
|
|
179
|
+
export function readScope(dir, { displayDir } = {}) {
|
|
180
|
+
const shown = displayDir ?? dir;
|
|
181
|
+
const findings = [];
|
|
182
|
+
let scope = null;
|
|
183
|
+
|
|
184
|
+
let raw;
|
|
185
|
+
try {
|
|
186
|
+
raw = readFileSync(join(dir, "scope.md"), "utf8");
|
|
187
|
+
} catch {
|
|
188
|
+
findings.push(f(1, "writing-dna-scope-missing", "fail",
|
|
189
|
+
`scope "${shown}": scope.md does not exist or cannot be read`,
|
|
190
|
+
`Run \`hyperspec dna init ${shown} --writer W --form F --audience A --purpose P\` to create it.`));
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
if (raw !== undefined) {
|
|
194
|
+
const { data, error } = parseSkillFile(raw);
|
|
195
|
+
if (error) {
|
|
196
|
+
findings.push(f(1, "writing-dna-scope-missing", "fail",
|
|
197
|
+
`scope "${shown}": scope.md is not valid (${error})`,
|
|
198
|
+
"Fix scope.md's frontmatter: writer, form, audience, purpose."));
|
|
199
|
+
} else {
|
|
200
|
+
const writer = str(data.writer);
|
|
201
|
+
const form = str(data.form);
|
|
202
|
+
const audience = str(data.audience);
|
|
203
|
+
const purpose = str(data.purpose);
|
|
204
|
+
scope = { writer, form, audience, purpose, notes: str(data.notes) };
|
|
205
|
+
for (const [field, value] of [["writer", writer], ["form", form], ["audience", audience], ["purpose", purpose]]) {
|
|
206
|
+
if (!value) {
|
|
207
|
+
findings.push(f(1, `writing-dna-scope-file-${field}`, "fail",
|
|
208
|
+
`scope "${shown}": scope.md has no ${field}`,
|
|
209
|
+
`Add ${field}: to scope.md.`));
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const { goldens, findings: goldenFindings } = readGoldens(dir, { displayDir: shown });
|
|
216
|
+
findings.push(...goldenFindings);
|
|
217
|
+
|
|
218
|
+
return { scope, goldens, findings };
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// -------------------------------------------------------------------------------------------
|
|
222
|
+
// measuring features, from text alone: no filesystem, no judgment, fully deterministic.
|
|
223
|
+
|
|
224
|
+
// Every number this module writes is rounded the same way: half away from zero, at the 3rd
|
|
225
|
+
// decimal. -0 never survives (JSON.stringify(-0) already prints "0", but this also normalizes
|
|
226
|
+
// the in-memory number, so a strict equality check on the returned object sees +0 too).
|
|
227
|
+
function round3(x) {
|
|
228
|
+
const r = Math.round(x * 1000) / 1000;
|
|
229
|
+
return r === 0 ? 0 : r;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
const mean = (arr) => (arr.length === 0 ? 0 : arr.reduce((a, b) => a + b, 0) / arr.length);
|
|
233
|
+
function median(sortedArr) {
|
|
234
|
+
const n = sortedArr.length;
|
|
235
|
+
if (n === 0) return 0;
|
|
236
|
+
const mid = Math.floor(n / 2);
|
|
237
|
+
return n % 2 === 0 ? (sortedArr[mid - 1] + sortedArr[mid]) / 2 : sortedArr[mid];
|
|
238
|
+
}
|
|
239
|
+
// Nearest-rank percentile: rank = ceil(p/100 * n), 1-based, clamped to [1, n]. p90 on 10 values
|
|
240
|
+
// [1..10] is rank ceil(9) = 9, the 9th smallest, i.e. 9 itself; this is the "nearest rank"
|
|
241
|
+
// definition (not linear interpolation).
|
|
242
|
+
function percentileNearestRank(sortedArr, p) {
|
|
243
|
+
const n = sortedArr.length;
|
|
244
|
+
if (n === 0) return 0;
|
|
245
|
+
const rank = Math.min(Math.max(Math.ceil((p / 100) * n), 1), n);
|
|
246
|
+
return sortedArr[rank - 1];
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
// A word is a maximal run of Unicode letters, digits, and apostrophes (straight ' or curly U+2019),
|
|
250
|
+
// lowercased. A leading or trailing apostrophe (a quote mark hugging a word) is swept up into
|
|
251
|
+
// the token by this same rule; it is simply never "between letters", so it never makes the word
|
|
252
|
+
// count as a contraction below.
|
|
253
|
+
const WORD_RE = /[\p{L}\p{N}'\u2019]+/gu;
|
|
254
|
+
// Exported so a later reader of a whole document (the check command's "form" station counting a
|
|
255
|
+
// draft's words against writing.form.length) reuses this exact word definition rather than
|
|
256
|
+
// keeping a second one that could quietly disagree with it.
|
|
257
|
+
export function wordsOf(text) {
|
|
258
|
+
const m = text.match(WORD_RE);
|
|
259
|
+
return m ? m.map((w) => w.toLowerCase()) : [];
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// A contraction: the word contains an apostrophe (straight or curly) with a letter immediately
|
|
263
|
+
// before AND after it. "don't" and "y'all's" qualify; "'tis" (apostrophe at position 0, nothing
|
|
264
|
+
// before it) and a stray quote-wrapped word do not.
|
|
265
|
+
function isContraction(word) {
|
|
266
|
+
for (let i = 1; i < word.length - 1; i++) {
|
|
267
|
+
const ch = word[i];
|
|
268
|
+
if ((ch === "'" || ch === "\u2019") && /\p{L}/u.test(word[i - 1]) && /\p{L}/u.test(word[i + 1])) return true;
|
|
269
|
+
}
|
|
270
|
+
return false;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
const FIRST_PERSON_SINGULAR = new Set(["i", "me", "my", "mine", "myself"]);
|
|
274
|
+
const FIRST_PERSON_PLURAL = new Set(["we", "us", "our", "ours", "ourselves"]);
|
|
275
|
+
const SECOND_PERSON = new Set(["you", "your", "yours", "yourself", "yourselves"]);
|
|
276
|
+
|
|
277
|
+
// A signature-word candidate: 4+ characters, letters and apostrophes only (no bare digit runs,
|
|
278
|
+
// so "2024" is never a signature word), and not a stopword. Frequency and the >=2 floor are
|
|
279
|
+
// applied afterward, over the whole scope's pooled words.
|
|
280
|
+
const CANDIDATE_RE = /^[\p{L}'\u2019]+$/u;
|
|
281
|
+
function isSignatureCandidate(word) {
|
|
282
|
+
return word.length >= 4 && CANDIDATE_RE.test(word) && !STOPWORD_SET.has(word.replace(/\u2019/g, "'"));
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
function signatureWords(allWords) {
|
|
286
|
+
const freq = new Map();
|
|
287
|
+
for (const w of allWords) {
|
|
288
|
+
if (!isSignatureCandidate(w)) continue;
|
|
289
|
+
freq.set(w, (freq.get(w) ?? 0) + 1);
|
|
290
|
+
}
|
|
291
|
+
const candidates = [...freq.entries()].filter(([, count]) => count >= 2);
|
|
292
|
+
// Frequency descending; ties broken alphabetically. Plain code-point comparison, not
|
|
293
|
+
// localeCompare, so the order never depends on the running Node build's ICU data.
|
|
294
|
+
candidates.sort((a, b) => b[1] - a[1] || (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : 0));
|
|
295
|
+
return candidates.slice(0, 15).map(([w]) => w);
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
const countChar = (text, re) => (text.match(re) || []).length;
|
|
299
|
+
|
|
300
|
+
// measureFeatures(texts): texts is an array of golden passage strings (readScope's goldens[].text),
|
|
301
|
+
// in any order; the result does not depend on it. Pure and deterministic: the same texts
|
|
302
|
+
// always produce the same features object, byte for byte once JSON.stringify'd. Every rate is
|
|
303
|
+
// per 1000 words; every number is rounded via round3.
|
|
304
|
+
export function measureFeatures(texts) {
|
|
305
|
+
const list = Array.isArray(texts) ? texts : [];
|
|
306
|
+
const allWords = [];
|
|
307
|
+
const sentenceWordCounts = [];
|
|
308
|
+
const paragraphSentenceCounts = [];
|
|
309
|
+
const paragraphWordCounts = [];
|
|
310
|
+
let commas = 0, semicolons = 0, colons = 0, emDashes = 0, enDashes = 0;
|
|
311
|
+
let exclaims = 0, questions = 0, parens = 0, quotes = 0;
|
|
312
|
+
|
|
313
|
+
for (const raw of list) {
|
|
314
|
+
const text = typeof raw === "string" ? raw : "";
|
|
315
|
+
commas += countChar(text, /,/g);
|
|
316
|
+
semicolons += countChar(text, /;/g);
|
|
317
|
+
colons += countChar(text, /:/g);
|
|
318
|
+
emDashes += countChar(text, /\u2014/g);
|
|
319
|
+
enDashes += countChar(text, /\u2013/g);
|
|
320
|
+
exclaims += countChar(text, /!/g);
|
|
321
|
+
questions += countChar(text, /\?/g);
|
|
322
|
+
parens += countChar(text, /[()]/g);
|
|
323
|
+
quotes += countChar(text, /["\u201C\u201D]/g);
|
|
324
|
+
|
|
325
|
+
const paragraphs = splitSegments(text, { by: "paragraph" }).map((s) => s.text);
|
|
326
|
+
for (const paragraph of paragraphs) {
|
|
327
|
+
const sentences = splitSegments(paragraph, { by: "sentence" }).map((s) => s.text);
|
|
328
|
+
let paragraphWords = 0;
|
|
329
|
+
for (const sentence of sentences) {
|
|
330
|
+
const sWords = wordsOf(sentence);
|
|
331
|
+
sentenceWordCounts.push(sWords.length);
|
|
332
|
+
paragraphWords += sWords.length;
|
|
333
|
+
allWords.push(...sWords);
|
|
334
|
+
}
|
|
335
|
+
paragraphSentenceCounts.push(sentences.length);
|
|
336
|
+
paragraphWordCounts.push(paragraphWords);
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
const totalWords = allWords.length;
|
|
341
|
+
const perThousand = (count) => (totalWords === 0 ? 0 : round3((count / totalWords) * 1000));
|
|
342
|
+
const sortedSentenceWords = [...sentenceWordCounts].sort((a, b) => a - b);
|
|
343
|
+
const contractionCount = allWords.filter(isContraction).length;
|
|
344
|
+
const fpSingular = allWords.filter((w) => FIRST_PERSON_SINGULAR.has(w)).length;
|
|
345
|
+
const fpPlural = allWords.filter((w) => FIRST_PERSON_PLURAL.has(w)).length;
|
|
346
|
+
const sp = allWords.filter((w) => SECOND_PERSON.has(w)).length;
|
|
347
|
+
const meanWordLength = totalWords === 0 ? 0 : round3(allWords.reduce((sum, w) => sum + w.length, 0) / totalWords);
|
|
348
|
+
|
|
349
|
+
return {
|
|
350
|
+
word_count: totalWords,
|
|
351
|
+
sentence_length: {
|
|
352
|
+
mean: round3(mean(sentenceWordCounts)),
|
|
353
|
+
median: round3(median(sortedSentenceWords)),
|
|
354
|
+
p90: round3(percentileNearestRank(sortedSentenceWords, 90)),
|
|
355
|
+
},
|
|
356
|
+
paragraph_length: {
|
|
357
|
+
mean_sentences: round3(mean(paragraphSentenceCounts)),
|
|
358
|
+
mean_words: round3(mean(paragraphWordCounts)),
|
|
359
|
+
},
|
|
360
|
+
rates_per_1000_words: {
|
|
361
|
+
comma: perThousand(commas),
|
|
362
|
+
semicolon: perThousand(semicolons),
|
|
363
|
+
colon: perThousand(colons),
|
|
364
|
+
em_dash: perThousand(emDashes),
|
|
365
|
+
en_dash: perThousand(enDashes),
|
|
366
|
+
exclamation: perThousand(exclaims),
|
|
367
|
+
question_mark: perThousand(questions),
|
|
368
|
+
parentheses: perThousand(parens),
|
|
369
|
+
quotation_marks: perThousand(quotes),
|
|
370
|
+
},
|
|
371
|
+
contraction_rate: perThousand(contractionCount),
|
|
372
|
+
first_person_singular_rate: perThousand(fpSingular),
|
|
373
|
+
first_person_plural_rate: perThousand(fpPlural),
|
|
374
|
+
second_person_rate: perThousand(sp),
|
|
375
|
+
mean_word_length: meanWordLength,
|
|
376
|
+
signature_words: signatureWords(allWords),
|
|
377
|
+
};
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
// features.json: 2-space JSON, a trailing newline, keys in the fixed order below, goldens sorted
|
|
381
|
+
// by path, so the same goldens always produce byte-identical bytes. DNA_FORMAT is the version of
|
|
382
|
+
// this shape, recorded in the file as "dna".
|
|
383
|
+
export const DNA_FORMAT = "0.1";
|
|
384
|
+
|
|
385
|
+
// featuresText({ scope, goldens, features }): the exact bytes writeFeatures writes, and the data
|
|
386
|
+
// they encode. The spec linter builds these from the scope as it reads now and compares them to
|
|
387
|
+
// the file, so a features.json is current only when it is what dna measure would write today.
|
|
388
|
+
export function featuresText({ scope, goldens, features }) {
|
|
389
|
+
const sortedGoldens = [...(goldens ?? [])].sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
|
|
390
|
+
const data = {
|
|
391
|
+
dna: DNA_FORMAT,
|
|
392
|
+
scope: {
|
|
393
|
+
writer: scope?.writer ?? "",
|
|
394
|
+
form: scope?.form ?? "",
|
|
395
|
+
audience: scope?.audience ?? "",
|
|
396
|
+
purpose: scope?.purpose ?? "",
|
|
397
|
+
},
|
|
398
|
+
goldens: sortedGoldens.map((g) => ({ path: g.path, sha256: g.sha256 })),
|
|
399
|
+
features,
|
|
400
|
+
};
|
|
401
|
+
return { data, text: `${JSON.stringify(data, null, 2)}\n` };
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
// writeFeatures(scopeDir, { scope, goldens, features }): writes <scopeDir>/features.json, the
|
|
405
|
+
// bytes featuresText returns. Returns { path, data }.
|
|
406
|
+
export function writeFeatures(scopeDir, input) {
|
|
407
|
+
const { data, text } = featuresText(input);
|
|
408
|
+
const path = join(scopeDir, "features.json");
|
|
409
|
+
writeFileAtomic(path, text);
|
|
410
|
+
return { path, data };
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
// -------------------------------------------------------------------------------------------
|
|
414
|
+
// dna init's skeleton, kept beside the module it belongs to (the same split as
|
|
415
|
+
// template.mjs/writingTemplate: string-building lives with the reader/measurer it seeds, file
|
|
416
|
+
// I/O and refusal checks live in the CLI).
|
|
417
|
+
|
|
418
|
+
export function scopeTemplate({ writer, form, audience, purpose } = {}) {
|
|
419
|
+
return `---
|
|
420
|
+
writer: ${scalar(String(writer))}
|
|
421
|
+
form: ${scalar(String(form))}
|
|
422
|
+
audience: ${scalar(String(audience))}
|
|
423
|
+
purpose: ${scalar(String(purpose))}
|
|
424
|
+
---
|
|
425
|
+
|
|
426
|
+
# Writer DNA scope
|
|
427
|
+
|
|
428
|
+
Goldens live in goldens/. Run \`hyperspec dna measure <scope-dir>\` once every golden there has
|
|
429
|
+
why, approved_by and source.
|
|
430
|
+
`;
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
export const GOLDENS_README = `# Goldens
|
|
434
|
+
|
|
435
|
+
Each file in this folder except this one is a golden: a passage the writer marked as right,
|
|
436
|
+
filed under this scope.
|
|
437
|
+
|
|
438
|
+
Frontmatter:
|
|
439
|
+
- why (required): what makes it golden, the move it teaches.
|
|
440
|
+
- approved_by (required): a person slug. Golden means a human approved it; agent:* is refused.
|
|
441
|
+
- source (required): where the passage came from.
|
|
442
|
+
- approved_on (optional): a date.
|
|
443
|
+
|
|
444
|
+
The body is the passage, verbatim.
|
|
445
|
+
|
|
446
|
+
Run \`hyperspec dna measure <scope-dir>\` once every golden here has why, approved_by and source.
|
|
447
|
+
`;
|
|
448
|
+
|
|
449
|
+
// -------------------------------------------------------------------------------------------
|
|
450
|
+
// stopwords: excluded from signature_words so the list names what this writer says, not what
|
|
451
|
+
// every writer says. About 150 common English function words. Deliberately no contractions
|
|
452
|
+
// (don't, isn't, ...): a stopword is matched against a lowercased word as tokenized above, and
|
|
453
|
+
// keeping this list plain avoids it silently missing half its entries because a golden happened
|
|
454
|
+
// to use a curly apostrophe where this file used a straight one, or the other way around.
|
|
455
|
+
export const STOPWORDS = Object.freeze([
|
|
456
|
+
"a", "about", "above", "across", "after", "again", "against", "all", "almost", "along",
|
|
457
|
+
"already", "also", "although", "always", "am", "among", "an", "and", "another", "any",
|
|
458
|
+
"anyone", "anything", "are", "around", "as", "at", "away", "be", "because", "been", "before",
|
|
459
|
+
"being", "below", "between", "both", "but", "by", "can", "cannot", "could", "did", "do",
|
|
460
|
+
"does", "doing", "done", "down", "during", "each", "either", "else", "ever", "every",
|
|
461
|
+
"everyone", "everything", "few", "for", "from", "further", "had", "has", "have", "having",
|
|
462
|
+
"he", "her", "here", "hers", "herself", "him", "himself", "his", "how", "however", "i", "if",
|
|
463
|
+
"in", "into", "is", "it", "its", "itself", "just", "may", "me", "might", "mine", "more",
|
|
464
|
+
"most", "much", "must", "my", "myself", "neither", "never", "next", "no", "nobody", "none",
|
|
465
|
+
"nor", "not", "nothing", "now", "of", "off", "often", "on", "once", "one", "only", "onto",
|
|
466
|
+
"or", "other", "others", "our", "ours", "ourselves", "out", "over", "own", "same", "shall",
|
|
467
|
+
"she", "should", "since", "so", "some", "someone", "something", "sometimes", "still", "such",
|
|
468
|
+
"than", "that", "the", "their", "theirs", "them", "themselves", "then", "there", "these",
|
|
469
|
+
"they", "this", "those", "though", "through", "to", "too", "toward", "towards", "under",
|
|
470
|
+
"until", "up", "upon", "us", "very", "was", "we", "were", "what", "when", "where", "whether",
|
|
471
|
+
"which", "while", "who", "whom", "whose", "why", "will", "with", "within", "without", "would",
|
|
472
|
+
"yet", "you", "your", "yours", "yourself", "yourselves",
|
|
473
|
+
]);
|
|
474
|
+
const STOPWORD_SET = new Set(STOPWORDS);
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
// Station "claims" (hyperspec 0.6). Reads writing.sources.ledger, a JSONL file
|
|
2
|
+
// where each line is one claim: { "text": <claim as it appears in the draft>, "source":
|
|
3
|
+
// <non-empty>, "span"?: <quote or locator> }. Two things are checked per line, and whether a
|
|
4
|
+
// sentence in the draft even IS a factual claim is never attempted here: that is judgment, and
|
|
5
|
+
// the ledger is the closed list of what counts as a claim. This station only checks that the
|
|
6
|
+
// ledger and the draft agree with each other:
|
|
7
|
+
//
|
|
8
|
+
// - the claim's text still appears verbatim in the draft (normalized for whitespace and quote
|
|
9
|
+
// characters, but not case) -- otherwise the ledger is stale (station-claims-stale);
|
|
10
|
+
// - the claim carries a real, non-placeholder source -- otherwise it is unsourced
|
|
11
|
+
// (station-claims-unsourced, downgraded to a warning when writing.sources.unsourced_claim is
|
|
12
|
+
// "warn", the same closed-set field src/writing-fields.mjs already validates on the spec).
|
|
13
|
+
//
|
|
14
|
+
// The ledger path resolves relative to the spec (spec.dir), the same way every other path-bearing
|
|
15
|
+
// writing field does. A missing or unreadable ledger fails the whole station on its own
|
|
16
|
+
// (station-claims-ledger-missing); a malformed JSONL line (not JSON, not an object, or missing
|
|
17
|
+
// text) is its own finding naming the line number, and does not stop the rest of the file from
|
|
18
|
+
// being read.
|
|
19
|
+
|
|
20
|
+
import { readFileSync } from "node:fs";
|
|
21
|
+
import { resolve } from "node:path";
|
|
22
|
+
import { str } from "../placeholder.mjs";
|
|
23
|
+
import { lineAt, truncate } from "./util.mjs";
|
|
24
|
+
|
|
25
|
+
export const name = "claims";
|
|
26
|
+
|
|
27
|
+
// Quote characters normalized to their straight ASCII form, then whitespace runs collapsed to a
|
|
28
|
+
// single space and the ends trimmed. "text matching is exact after normalizing whitespace and
|
|
29
|
+
// quote characters": case is NOT normalized, so a claim's text must
|
|
30
|
+
// still match the draft's actual capitalization.
|
|
31
|
+
function normalize(text) {
|
|
32
|
+
return String(text)
|
|
33
|
+
.replace(/[‘’‚‛]/g, "'")
|
|
34
|
+
.replace(/[“”„‟]/g, '"')
|
|
35
|
+
.replace(/\s+/g, " ")
|
|
36
|
+
.trim();
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Where a claim's text first appears in the original draft, under the same normalization the match
|
|
40
|
+
// uses (any whitespace run for a space, any quote character for a quote), or -1.
|
|
41
|
+
function firstOccurrence(text, claimText) {
|
|
42
|
+
const body = [...normalize(claimText)].map((c) => (c === " " ? "\\s+" : c === "'" ? "['‘’‚‛]" : c === '"' ? '["“”„‟]' : c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"))).join("");
|
|
43
|
+
const m = new RegExp(body).exec(text);
|
|
44
|
+
return m ? m.index : -1;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export function run(spec, draft) {
|
|
48
|
+
const sources = spec?.data?.writing?.sources ?? {};
|
|
49
|
+
const ledgerPath = str(sources.ledger);
|
|
50
|
+
const unsourcedSeverity = str(sources.unsourced_claim) === "warn" ? "warn" : "fail";
|
|
51
|
+
|
|
52
|
+
if (!ledgerPath) {
|
|
53
|
+
// writing.sources.ledger is a required field (lint test 1), so by the time check runs (lint
|
|
54
|
+
// already passed) a real spec always has one; this is defense in depth for a caller that
|
|
55
|
+
// builds a spec object by hand and skips lint, mirroring how form.mjs treats its own inputs
|
|
56
|
+
// as never fully trusted either.
|
|
57
|
+
return { station: name, status: "skip", findings: [], reason: "writing.sources.ledger is not set" };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const ledgerAbs = resolve(spec?.dir || ".", ledgerPath);
|
|
61
|
+
let raw;
|
|
62
|
+
try {
|
|
63
|
+
// A leading UTF-8 BOM (written by default by several Windows/Excel-adjacent editors) is not
|
|
64
|
+
// valid JSON leading whitespace, so it must come off before line 1 is parsed, or a genuinely
|
|
65
|
+
// well-formed first line reports as broken JSON for a reason that has nothing to do with its
|
|
66
|
+
// content.
|
|
67
|
+
raw = readFileSync(ledgerAbs, "utf8").replace(/^/, "");
|
|
68
|
+
} catch {
|
|
69
|
+
return {
|
|
70
|
+
station: name,
|
|
71
|
+
status: "fail",
|
|
72
|
+
findings: [{
|
|
73
|
+
station: name,
|
|
74
|
+
id: "station-claims-ledger-missing",
|
|
75
|
+
severity: "fail",
|
|
76
|
+
message: `writing.sources.ledger "${ledgerPath}" does not exist or cannot be read`,
|
|
77
|
+
fix: `Create ${ledgerPath} as JSONL, one claim per line: {"text": "...", "source": "..."}.`,
|
|
78
|
+
}],
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const findings = [];
|
|
83
|
+
const draftNorm = normalize(draft.text);
|
|
84
|
+
const ledgerLines = raw.split("\n").map((text, i) => ({ n: i + 1, text })).filter((l) => l.text.trim() !== "");
|
|
85
|
+
|
|
86
|
+
for (const { n, text } of ledgerLines) {
|
|
87
|
+
let obj;
|
|
88
|
+
try {
|
|
89
|
+
obj = JSON.parse(text);
|
|
90
|
+
} catch {
|
|
91
|
+
findings.push({
|
|
92
|
+
station: name,
|
|
93
|
+
id: `station-claims-json-line-${n}`,
|
|
94
|
+
severity: "fail",
|
|
95
|
+
message: `writing.sources.ledger "${ledgerPath}" line ${n} is not valid JSON`,
|
|
96
|
+
fix: "Fix the JSON on that line.",
|
|
97
|
+
});
|
|
98
|
+
continue;
|
|
99
|
+
}
|
|
100
|
+
if (!obj || typeof obj !== "object" || Array.isArray(obj)) {
|
|
101
|
+
findings.push({
|
|
102
|
+
station: name,
|
|
103
|
+
id: `station-claims-json-line-${n}`,
|
|
104
|
+
severity: "fail",
|
|
105
|
+
message: `writing.sources.ledger "${ledgerPath}" line ${n} is not a JSON object`,
|
|
106
|
+
fix: 'Each ledger line must be a JSON object: {"text": "...", "source": "..."}.',
|
|
107
|
+
});
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
const claimText = typeof obj.text === "string" ? obj.text : "";
|
|
112
|
+
if (!claimText.trim()) {
|
|
113
|
+
findings.push({
|
|
114
|
+
station: name,
|
|
115
|
+
id: `station-claims-json-line-${n}`,
|
|
116
|
+
severity: "fail",
|
|
117
|
+
message: `writing.sources.ledger "${ledgerPath}" line ${n} has no text`,
|
|
118
|
+
fix: "Add text: the claim exactly as it appears in the draft.",
|
|
119
|
+
});
|
|
120
|
+
continue;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const tag = truncate(claimText, 80);
|
|
124
|
+
|
|
125
|
+
if (!draftNorm.includes(normalize(claimText))) {
|
|
126
|
+
findings.push({
|
|
127
|
+
station: name,
|
|
128
|
+
id: "station-claims-stale",
|
|
129
|
+
severity: "fail",
|
|
130
|
+
message: `ledger line ${n}, "${tag}", does not appear verbatim in the draft`,
|
|
131
|
+
fix: "Update the ledger's text to match the draft exactly, or remove the stale claim.",
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
if (!str(obj.source)) {
|
|
136
|
+
const at = firstOccurrence(draft.text, claimText);
|
|
137
|
+
findings.push({
|
|
138
|
+
station: name,
|
|
139
|
+
id: "station-claims-unsourced",
|
|
140
|
+
severity: unsourcedSeverity,
|
|
141
|
+
...(at >= 0 ? { line: lineAt(draft.text, at) } : {}),
|
|
142
|
+
message: `ledger line ${n}, "${tag}", has no source (or it is a placeholder)`,
|
|
143
|
+
fix: "Add a real source: to the claim, or remove it from the ledger.",
|
|
144
|
+
});
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const status = findings.some((x) => x.severity === "fail") ? "fail" : "pass";
|
|
149
|
+
return { station: name, status, findings };
|
|
150
|
+
}
|