@supersuit/hyperspec 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +59 -0
- package/README.md +22 -1
- package/SPEC.md +5 -3
- package/WRITING.md +203 -32
- package/bin/hyperspec.mjs +50 -2
- package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +10 -0
- package/examples/writing/essay/materials/team-survey.md.segments.jsonl +6 -0
- package/examples/writing/essay/materials/voice-memo.md.segments.jsonl +7 -0
- package/examples/writing/essay.hyperspec.md +9 -6
- package/examples/writing/story/materials/bakery-visit.md +1 -0
- package/examples/writing/story/materials/bakery-visit.md.segments.jsonl +13 -0
- package/examples/writing/story/materials/notes.md +2 -0
- package/examples/writing/story/materials/notes.md.segments.jsonl +7 -0
- package/examples/writing/story/materials/scene-list.md.segments.jsonl +14 -0
- package/examples/writing/story.hyperspec.md +8 -5
- package/package.json +2 -1
- package/src/blobs.mjs +1 -1
- package/src/compare.mjs +6 -6
- package/src/fsutil.mjs +1 -1
- package/src/labels.mjs +6 -0
- package/src/reproduce.mjs +5 -5
- package/src/segments.mjs +407 -0
- package/src/writing-exports.mjs +6 -0
- package/src/writing-fields.mjs +106 -5
- package/src/writing-template.mjs +8 -1
- package/src/writing.mjs +4 -4
package/src/segments.mjs
ADDED
|
@@ -0,0 +1,407 @@
|
|
|
1
|
+
// Materials marking (hyperspec 0.4). hyperspec never calls a model: a material is split into candidate
|
|
2
|
+
// segments deterministically (splitSegments), an agent or a person labels each one by hand-editing
|
|
3
|
+
// the JSONL file `segments init` wrote, and this module reads that file back and checks it
|
|
4
|
+
// (readSegments). Labeling itself is judgment and stays outside this file entirely; everything
|
|
5
|
+
// here is either splitting (no judgment) or checking (closed rules, no model).
|
|
6
|
+
//
|
|
7
|
+
// Segments file format (JSONL, one JSON object per line): line 1 is a header
|
|
8
|
+
// {"material":"<id>","path":"<material path>","sha256":"<hex>"}; every later line is one segment
|
|
9
|
+
// {"id":"s1","start":0,"end":212,"label":"claim"|"story"|"quote"|"stance"|"question"|"aside"|
|
|
10
|
+
// "private"|"unlabeled", ...label fields, "text":"..."}. start/end are JS string indices (UTF-16
|
|
11
|
+
// code units, the same indices readFileSync(path, "utf8") hands back) into the material's text,
|
|
12
|
+
// end exclusive; text must equal source.slice(start, end) exactly.
|
|
13
|
+
|
|
14
|
+
import { readFileSync } from "node:fs";
|
|
15
|
+
import { basename } from "node:path";
|
|
16
|
+
import { sha256 } from "./hash.mjs";
|
|
17
|
+
import { MATERIAL_LABELS } from "./labels.mjs";
|
|
18
|
+
import { str } from "./placeholder.mjs";
|
|
19
|
+
|
|
20
|
+
const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
|
|
21
|
+
// own counts as set when it is the JSON boolean true or the string "true"; anything else
|
|
22
|
+
// (false, "false", missing, any other value) is unset.
|
|
23
|
+
const ownIsSet = (obj) => obj?.own === true || obj?.own === "true";
|
|
24
|
+
// The SAME whitespace test the coverage check below uses (/\S/, JS's Unicode-aware class, which
|
|
25
|
+
// includes U+00A0 NBSP and the other Unicode space separators, not just the ASCII set). One
|
|
26
|
+
// definition shared by both: a gap the sentence splitter treats as pure separator and a gap the
|
|
27
|
+
// coverage check treats as "nothing to require a segment for" must never disagree, or a stretch
|
|
28
|
+
// of NBSP-only text could read as covered to one check and as real content to the other.
|
|
29
|
+
const isWhitespace = (ch) => /\s/.test(ch);
|
|
30
|
+
|
|
31
|
+
// -------------------------------------------------------------------------------------------
|
|
32
|
+
// splitSegments: deterministic, no judgment. Candidate segments only ("unlabeled"); a person or
|
|
33
|
+
// agent assigns real labels afterward by editing the JSONL file this feeds.
|
|
34
|
+
|
|
35
|
+
// A line's [start, contentEnd) span, contentEnd excluding that line's own terminator ("\n" or, for
|
|
36
|
+
// CRLF, "\r\n"; the \r is excluded from content the same way the \n is, so a blank CRLF line
|
|
37
|
+
// reads as blank, but the \r is never dropped from the original text itself: it simply falls
|
|
38
|
+
// outside every line's content span, same as \n does, and stays untouched wherever it sits inside
|
|
39
|
+
// a multi-line segment). The final line (no trailing "\n") has contentEnd === text.length.
|
|
40
|
+
function lineSpans(text) {
|
|
41
|
+
const spans = [];
|
|
42
|
+
let start = 0;
|
|
43
|
+
while (start <= text.length) {
|
|
44
|
+
const nl = text.indexOf("\n", start);
|
|
45
|
+
if (nl === -1) { spans.push({ start, contentEnd: text.length }); break; }
|
|
46
|
+
let contentEnd = nl;
|
|
47
|
+
if (contentEnd > start && text[contentEnd - 1] === "\r") contentEnd -= 1;
|
|
48
|
+
spans.push({ start, contentEnd });
|
|
49
|
+
start = nl + 1;
|
|
50
|
+
}
|
|
51
|
+
return spans;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const isBlankLine = (content) => /^[ \t]*$/.test(content);
|
|
55
|
+
|
|
56
|
+
// Paragraphs: a maximal run of consecutive non-blank lines. The blank line(s) between two
|
|
57
|
+
// paragraphs, and their terminators, are never part of either paragraph's span: "whitespace
|
|
58
|
+
// between segments is left out of every segment." A multi-line paragraph keeps the single "\n" (or
|
|
59
|
+
// "\r\n") between its own lines, since that is internal to the paragraph, not a separator.
|
|
60
|
+
function splitParagraphs(text) {
|
|
61
|
+
const lines = lineSpans(text);
|
|
62
|
+
const out = [];
|
|
63
|
+
let curStart = null;
|
|
64
|
+
let curEnd = null;
|
|
65
|
+
for (const line of lines) {
|
|
66
|
+
const content = text.slice(line.start, line.contentEnd);
|
|
67
|
+
if (isBlankLine(content)) {
|
|
68
|
+
if (curStart !== null) { out.push({ start: curStart, end: curEnd }); curStart = null; }
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
if (curStart === null) curStart = line.start;
|
|
72
|
+
curEnd = line.contentEnd;
|
|
73
|
+
}
|
|
74
|
+
if (curStart !== null) out.push({ start: curStart, end: curEnd });
|
|
75
|
+
return out;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// Sentences: split on ".", "?" or "!" followed by whitespace (or end of text), never while inside
|
|
79
|
+
// a quoted span (a run opened by '"' and not yet closed) on the SAME line. inQuote resets at
|
|
80
|
+
// every "\n" regardless of whether a quote is actually still open, so a stray unterminated quote
|
|
81
|
+
// can never suppress boundaries on a later line. Punctuation strictly INSIDE an open quote on the
|
|
82
|
+
// same line (the case this rule exists for, e.g. a title like "Wait." spoken mid-sentence) is
|
|
83
|
+
// never itself a boundary; the next real, unquoted terminator is.
|
|
84
|
+
//
|
|
85
|
+
// A terminal punctuation mark immediately followed by a closing '"' DOES still count as a
|
|
86
|
+
// boundary, and that "punctuation + closing quote" branch is reachable more often than "same
|
|
87
|
+
// line" alone suggests: because inQuote resets on every "\n", a quote opened on one line and
|
|
88
|
+
// closed with punctuation on a LATER line is, by the time that later line's punctuation is
|
|
89
|
+
// reached, read as not-currently-in-a-quote, so the branch fires there too, closing the segment
|
|
90
|
+
// right after the quote mark (`He said "long\nquote here." Next.` -> one segment ending after the
|
|
91
|
+
// closing quote, then "Next."). The same branch also fires on an orphan closing quote with no
|
|
92
|
+
// matching open one on its own line (`He was done." Next.` splits the same way); this is a
|
|
93
|
+
// side effect of the same rule rather than special-cased, and is deterministic either way.
|
|
94
|
+
//
|
|
95
|
+
// A line break followed by a list marker is also a boundary: optional spaces or tabs, then "-", "*",
|
|
96
|
+
// "+", or digits followed by "." or ")", then a space. Without this, a bullet that is entirely a
|
|
97
|
+
// quotation ending in `."` never meets an unquoted terminator (its punctuation sits inside the
|
|
98
|
+
// quote) and runs into the next bullet. A segment that opens on a list marker steps over the
|
|
99
|
+
// marker first, so a numbered item's own "1." or "2)" is never read as a sentence ending. A line
|
|
100
|
+
// that merely starts with a hyphenated word ("self-evident"), a decimal ("3.5") or a hyphen with
|
|
101
|
+
// no space after it ("-ish") is not a marker and does not split.
|
|
102
|
+
//
|
|
103
|
+
// Leading and trailing whitespace around the whole text, and the whitespace run between sentences,
|
|
104
|
+
// is left out of every segment, the same as splitParagraphs.
|
|
105
|
+
const LIST_MARKER = /[ \t]*(?:[-*+]|\d+[.)]) /y;
|
|
106
|
+
|
|
107
|
+
// If a list marker begins at `pos` (after optional spaces or tabs), the index just past the marker
|
|
108
|
+
// symbol (before its trailing space), else -1. Callers only ask at the start of a line.
|
|
109
|
+
function listMarkerEnd(text, pos) {
|
|
110
|
+
LIST_MARKER.lastIndex = pos;
|
|
111
|
+
return LIST_MARKER.test(text) ? LIST_MARKER.lastIndex - 1 : -1;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// True when `pos` sits at the start of a line once any leading spaces or tabs are skipped back over.
|
|
115
|
+
function atLineStart(text, pos) {
|
|
116
|
+
let k = pos - 1;
|
|
117
|
+
while (k >= 0 && (text[k] === " " || text[k] === "\t")) k--;
|
|
118
|
+
return k < 0 || text[k] === "\n";
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function splitSentences(text) {
|
|
122
|
+
const n = text.length;
|
|
123
|
+
const out = [];
|
|
124
|
+
let inQuote = false;
|
|
125
|
+
let i = 0;
|
|
126
|
+
while (i < n && isWhitespace(text[i])) i++;
|
|
127
|
+
let start = i;
|
|
128
|
+
// A segment that opens on a list marker begins scanning past it, so "1." is not a terminator.
|
|
129
|
+
const skipMarker = (pos) => (atLineStart(text, pos) ? Math.max(pos, listMarkerEnd(text, pos)) : pos);
|
|
130
|
+
i = skipMarker(start);
|
|
131
|
+
while (i < n) {
|
|
132
|
+
const ch = text[i];
|
|
133
|
+
if (ch === "\n") {
|
|
134
|
+
inQuote = false;
|
|
135
|
+
const markerEnd = listMarkerEnd(text, i + 1);
|
|
136
|
+
if (markerEnd !== -1) {
|
|
137
|
+
let end = i;
|
|
138
|
+
while (end > start && isWhitespace(text[end - 1])) end--;
|
|
139
|
+
if (end > start) out.push({ start, end });
|
|
140
|
+
let k = i + 1;
|
|
141
|
+
while (text[k] === " " || text[k] === "\t") k++;
|
|
142
|
+
start = k;
|
|
143
|
+
i = markerEnd;
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
i++;
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
if (ch === '"') { inQuote = !inQuote; i++; continue; }
|
|
150
|
+
if (!inQuote && (ch === "." || ch === "?" || ch === "!")) {
|
|
151
|
+
let j = i + 1;
|
|
152
|
+
while (j < n && (text[j] === "." || text[j] === "?" || text[j] === "!")) j++;
|
|
153
|
+
let end = j;
|
|
154
|
+
if (j < n && text[j] === '"') { end = j + 1; inQuote = false; }
|
|
155
|
+
if (end >= n || isWhitespace(text[end])) {
|
|
156
|
+
out.push({ start, end });
|
|
157
|
+
let k = end;
|
|
158
|
+
while (k < n && isWhitespace(text[k])) k++;
|
|
159
|
+
start = k;
|
|
160
|
+
i = skipMarker(k);
|
|
161
|
+
continue;
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
i++;
|
|
165
|
+
}
|
|
166
|
+
if (start < n) {
|
|
167
|
+
let end = n;
|
|
168
|
+
while (end > start && isWhitespace(text[end - 1])) end--;
|
|
169
|
+
if (end > start) out.push({ start, end });
|
|
170
|
+
}
|
|
171
|
+
return out;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// splitSegments(text, { by }): "paragraph" (default) or "sentence". Returns candidate segments,
|
|
175
|
+
// each { id: "s<n>", start, end, label: "unlabeled", text }, in document order, ids 1-based and
|
|
176
|
+
// dense. "unlabeled" is what this writes and is never a valid label in readSegments below.
|
|
177
|
+
export function splitSegments(text, { by = "paragraph" } = {}) {
|
|
178
|
+
if (by !== "paragraph" && by !== "sentence") {
|
|
179
|
+
throw new RangeError(`splitSegments: by must be "paragraph" or "sentence", got ${JSON.stringify(by)}`);
|
|
180
|
+
}
|
|
181
|
+
const spans = by === "sentence" ? splitSentences(text) : splitParagraphs(text);
|
|
182
|
+
return spans.map((s, i) => ({ id: `s${i + 1}`, start: s.start, end: s.end, label: "unlabeled", text: text.slice(s.start, s.end) }));
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// -------------------------------------------------------------------------------------------
|
|
186
|
+
// readSegments: parse + validate a marked-up segments file. Never throws on a bad or missing
|
|
187
|
+
// file; every failure mode becomes a finding in the lint shape (test, id, severity, message, fix),
|
|
188
|
+
// the same shape src/rules.mjs and src/writing-fields.mjs already use. readSegments never reads
|
|
189
|
+
// MATERIAL_LABELS' meaning into anything beyond the closed-set and per-label-field checks below;
|
|
190
|
+
// spine-ref resolution (m1#s3, and the private/question refusal) happens in
|
|
191
|
+
// src/writing-fields.mjs, which calls this and then checks refs against the returned segments.
|
|
192
|
+
//
|
|
193
|
+
// materialPath is optional: without it, only the structural checks that need no material text run
|
|
194
|
+
// (header shape, labels, per-label fields, duplicate ids). With it, the material-text-dependent
|
|
195
|
+
// checks run too: verbatim text, coverage, overlap and staleness. materialId, when given, is
|
|
196
|
+
// compared against the header's own "material" field.
|
|
197
|
+
//
|
|
198
|
+
// EVERY finding message names which material it is about (and the segment id, where one
|
|
199
|
+
// applies), so two materials that each have a broken "s1" never produce identical-looking
|
|
200
|
+
// findings. The finding `id` fields stay rule ids and may still repeat across materials (the same
|
|
201
|
+
// way core findings do); it is the message text this rule is about. The material tag preferred,
|
|
202
|
+
// in order: the caller's own materialId, else the header's own "material" field (once parsed),
|
|
203
|
+
// else the segments file's own basename. The last resort covers the missing-file and
|
|
204
|
+
// unparsable-header cases, where neither of the first two is available yet.
|
|
205
|
+
export function readSegments(segmentsPath, { materialPath, materialId, displayPath, materialDisplayPath } = {}) {
|
|
206
|
+
const findings = [];
|
|
207
|
+
const fallbackTag = basename(segmentsPath);
|
|
208
|
+
// What messages print for the two files: the caller's display paths when given (lint passes the
|
|
209
|
+
// paths as the spec wrote them), else the paths exactly as given. Never a path this function
|
|
210
|
+
// resolved itself, so a finding reads the same on every machine.
|
|
211
|
+
const segShown = displayPath ?? segmentsPath;
|
|
212
|
+
const matShown = materialDisplayPath ?? materialPath;
|
|
213
|
+
|
|
214
|
+
let raw;
|
|
215
|
+
try {
|
|
216
|
+
raw = readFileSync(segmentsPath, "utf8");
|
|
217
|
+
} catch {
|
|
218
|
+
const matTag = materialId ?? fallbackTag;
|
|
219
|
+
findings.push(f(1, "writing-materials-segments-missing", "fail",
|
|
220
|
+
`material ${matTag}: is not marked (segments file "${segShown}" does not exist or cannot be read)`,
|
|
221
|
+
`Run \`hyperspec segments init <material> --id <id>\` to write it, then label every segment.`));
|
|
222
|
+
return { header: null, segments: [], findings };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const rawLines = raw.split("\n");
|
|
226
|
+
const headerRaw = rawLines[0] ?? "";
|
|
227
|
+
// Any line whose whole trimmed content is empty is skipped (a trailing blank line left by a
|
|
228
|
+
// final "\n" is the common case, but a genuinely blank line mid-file is tolerated the same way
|
|
229
|
+
// ordinary JSONL readers tolerate one); line 1 is the only line this reader treats as load-
|
|
230
|
+
// bearing on its own, per the format.
|
|
231
|
+
const segLines = rawLines.slice(1).map((text, i) => ({ n: i + 2, text })).filter((l) => l.text.trim() !== "");
|
|
232
|
+
|
|
233
|
+
let header = null;
|
|
234
|
+
try {
|
|
235
|
+
const parsed = JSON.parse(headerRaw);
|
|
236
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) header = parsed;
|
|
237
|
+
else throw new Error("header is not a JSON object");
|
|
238
|
+
} catch {
|
|
239
|
+
// header stays null; matTag below falls through materialId -> fallbackTag, since there is no
|
|
240
|
+
// header.material to read yet.
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// Computed once header parsing has settled, so every finding from here on (including the
|
|
244
|
+
// header's own shape findings) can use it. header?.material is only trusted when it is a real,
|
|
245
|
+
// non-empty string; a header that fails its own presence check contributes nothing here.
|
|
246
|
+
const matTag = materialId ?? (header && str(header.material) ? str(header.material) : fallbackTag);
|
|
247
|
+
const matPrefix = (segId) => (segId ? `material ${matTag}, segment ${segId}: ` : `material ${matTag}: `);
|
|
248
|
+
|
|
249
|
+
if (!header) {
|
|
250
|
+
findings.push(f(1, "writing-materials-header", "fail",
|
|
251
|
+
`${matPrefix()}segments file "${segShown}" line 1 is not a valid JSON header object`,
|
|
252
|
+
`Fix line 1 to a JSON object: {"material":"<id>","path":"<material path>","sha256":"<hex>"}.`));
|
|
253
|
+
} else {
|
|
254
|
+
for (const key of ["material", "path", "sha256"]) {
|
|
255
|
+
if (!str(header[key])) {
|
|
256
|
+
findings.push(f(1, "writing-materials-header", "fail",
|
|
257
|
+
`${matPrefix()}segments file "${segShown}" header has no ${key}`,
|
|
258
|
+
`Add ${key}: to the header line (line 1).`));
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
if (materialId !== undefined && str(header.material) && str(header.material) !== materialId) {
|
|
262
|
+
findings.push(f(1, "writing-materials-header-material", "fail",
|
|
263
|
+
`${matPrefix()}segments file "${segShown}" header names material "${header.material}", not "${materialId}"`,
|
|
264
|
+
`Set the header's material to "${materialId}", or point the item at the right segments file.`));
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
const segments = [];
|
|
269
|
+
const idsSeen = new Set();
|
|
270
|
+
const idsReportedDup = new Set();
|
|
271
|
+
|
|
272
|
+
for (const { n, text } of segLines) {
|
|
273
|
+
let obj;
|
|
274
|
+
try {
|
|
275
|
+
obj = JSON.parse(text);
|
|
276
|
+
} catch {
|
|
277
|
+
findings.push(f(1, `writing-materials-json-line-${n}`, "fail",
|
|
278
|
+
`${matPrefix()}segments file "${segShown}" line ${n} is not valid JSON`,
|
|
279
|
+
"Fix the JSON on that line."));
|
|
280
|
+
continue;
|
|
281
|
+
}
|
|
282
|
+
if (!obj || typeof obj !== "object" || Array.isArray(obj)) {
|
|
283
|
+
findings.push(f(1, `writing-materials-json-line-${n}`, "fail",
|
|
284
|
+
`${matPrefix()}segments file "${segShown}" line ${n} is not a JSON object`,
|
|
285
|
+
"Each segment line must be a JSON object."));
|
|
286
|
+
continue;
|
|
287
|
+
}
|
|
288
|
+
segments.push(obj);
|
|
289
|
+
|
|
290
|
+
const id = str(obj.id);
|
|
291
|
+
const segTag = id || `line ${n}`;
|
|
292
|
+
if (!id) {
|
|
293
|
+
findings.push(f(1, `writing-materials-segment-id-${n}`, "fail", `${matPrefix()}segment on line ${n} has no id`, "Give it a short id, e.g. s1."));
|
|
294
|
+
} else if (idsSeen.has(id)) {
|
|
295
|
+
if (!idsReportedDup.has(id)) {
|
|
296
|
+
findings.push(f(1, "writing-materials-segment-id", "fail", `${matPrefix(id)}segment id is used twice`, "Ids must be unique within the segments file; rename one."));
|
|
297
|
+
}
|
|
298
|
+
idsReportedDup.add(id);
|
|
299
|
+
} else {
|
|
300
|
+
idsSeen.add(id);
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
const label = typeof obj.label === "string" ? obj.label : "";
|
|
304
|
+
if (!MATERIAL_LABELS.includes(label)) {
|
|
305
|
+
findings.push(f(1, `writing-materials-label-${segTag}`, "fail",
|
|
306
|
+
label === "unlabeled" || !label
|
|
307
|
+
? `${matPrefix(segTag)}is still unlabeled`
|
|
308
|
+
: `${matPrefix(segTag)}label "${label}" is outside ${MATERIAL_LABELS.join(", ")}`,
|
|
309
|
+
`Set label to one of: ${MATERIAL_LABELS.join(", ")}.`));
|
|
310
|
+
} else if (label === "claim" && !(str(obj.source) || ownIsSet(obj))) {
|
|
311
|
+
findings.push(f(4, `writing-materials-claim-source-${segTag}`, "fail",
|
|
312
|
+
`${matPrefix(segTag)}claim has no source and is not marked own`,
|
|
313
|
+
`Add source: to segment "${segTag}", or set own: true if it is the author's own claim, said as such.`));
|
|
314
|
+
} else if (label === "story" && !str(obj.teller)) {
|
|
315
|
+
findings.push(f(4, `writing-materials-story-teller-${segTag}`, "fail", `${matPrefix(segTag)}story has no teller`, `Add teller: to segment "${segTag}".`));
|
|
316
|
+
} else if (label === "quote" && !str(obj.speaker)) {
|
|
317
|
+
findings.push(f(4, `writing-materials-quote-speaker-${segTag}`, "fail", `${matPrefix(segTag)}quote has no speaker`, `Add speaker: to segment "${segTag}".`));
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Material-text-dependent checks: verbatim text, shape, overlap, coverage, staleness. Skipped
|
|
322
|
+
// entirely when no materialPath is given (structural-only mode).
|
|
323
|
+
if (materialPath) {
|
|
324
|
+
let materialBuffer = null;
|
|
325
|
+
try {
|
|
326
|
+
materialBuffer = readFileSync(materialPath);
|
|
327
|
+
} catch {
|
|
328
|
+
findings.push(f(1, "writing-materials-material-missing", "fail",
|
|
329
|
+
`${matPrefix()}the material file "${matShown}" does not exist or cannot be read`,
|
|
330
|
+
"Fix the material's path, or add the file."));
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
if (materialBuffer) {
|
|
334
|
+
const materialText = materialBuffer.toString("utf8");
|
|
335
|
+
|
|
336
|
+
if (header && str(header.sha256)) {
|
|
337
|
+
const current = sha256(materialBuffer);
|
|
338
|
+
if (current !== str(header.sha256)) {
|
|
339
|
+
findings.push(f(4, "writing-materials-stale", "fail",
|
|
340
|
+
`${matPrefix()}segments file "${segShown}" was marked against a different version of "${matShown}" (sha256 no longer matches)`,
|
|
341
|
+
"Re-run `hyperspec segments init` (or otherwise re-mark) against the current material, and re-label every segment."));
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
const valid = [];
|
|
346
|
+
segments.forEach((obj, i) => {
|
|
347
|
+
const id = str(obj.id) || `#${i + 1}`;
|
|
348
|
+
const { start, end } = obj;
|
|
349
|
+
const shapeOk = Number.isInteger(start) && Number.isInteger(end) && start >= 0 && end > start && end <= materialText.length;
|
|
350
|
+
if (!shapeOk) {
|
|
351
|
+
findings.push(f(1, `writing-materials-segment-shape-${id}`, "fail",
|
|
352
|
+
`${matPrefix(id)}start/end are not valid offsets into the material`,
|
|
353
|
+
"Set start and end to whole-number character offsets into the material, end greater than start and no greater than the material's length."));
|
|
354
|
+
return;
|
|
355
|
+
}
|
|
356
|
+
const text = typeof obj.text === "string" ? obj.text : "";
|
|
357
|
+
const expected = materialText.slice(start, end);
|
|
358
|
+
if (text !== expected) {
|
|
359
|
+
findings.push(f(4, `writing-materials-text-${id}`, "fail",
|
|
360
|
+
`${matPrefix(id)}text does not match the material verbatim at [${start}, ${end})`,
|
|
361
|
+
"Re-derive start/end/text from the material, or re-run hyperspec segments init."));
|
|
362
|
+
}
|
|
363
|
+
valid.push({ id, start, end });
|
|
364
|
+
});
|
|
365
|
+
|
|
366
|
+
valid.sort((a, b) => a.start - b.start);
|
|
367
|
+
for (let i = 1; i < valid.length; i++) {
|
|
368
|
+
if (valid[i].start < valid[i - 1].end) {
|
|
369
|
+
findings.push(f(1, "writing-materials-overlap", "fail",
|
|
370
|
+
`${matPrefix(valid[i].id)}overlaps segment "${valid[i - 1].id}"`,
|
|
371
|
+
"Adjust start/end so segments never overlap."));
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
// A material with no text that is not whitespace has nothing to mark, so it is not a material.
|
|
376
|
+
if (!/\S/.test(materialText)) {
|
|
377
|
+
findings.push(f(1, "writing-materials-empty", "fail",
|
|
378
|
+
`${matPrefix()}the material "${matShown}" has no text to mark`,
|
|
379
|
+
"Put the material's text in the file, or drop the item from materials.items."));
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
let cursor = 0;
|
|
383
|
+
const gaps = [];
|
|
384
|
+
for (const seg of valid) {
|
|
385
|
+
if (seg.start > cursor) gaps.push([cursor, seg.start]);
|
|
386
|
+
cursor = Math.max(cursor, seg.end);
|
|
387
|
+
}
|
|
388
|
+
if (cursor < materialText.length) gaps.push([cursor, materialText.length]);
|
|
389
|
+
const uncovered = gaps.filter(([s, e]) => /\S/.test(materialText.slice(s, e)));
|
|
390
|
+
if (uncovered.length) {
|
|
391
|
+
// Say where: the first uncovered stretch's offset (its first non-whitespace character) and
|
|
392
|
+
// up to 60 characters of it, trimmed, plus how many stretches there are in all.
|
|
393
|
+
const [gs, ge] = uncovered[0];
|
|
394
|
+
const gap = materialText.slice(gs, ge);
|
|
395
|
+
const offset = gs + gap.search(/\S/);
|
|
396
|
+
const trimmed = gap.trim();
|
|
397
|
+
const excerpt = trimmed.length > 60 ? `${trimmed.slice(0, 60)}...` : trimmed;
|
|
398
|
+
const more = uncovered.length > 1 ? ` (${uncovered.length} uncovered stretches in all)` : "";
|
|
399
|
+
findings.push(f(1, "writing-materials-coverage", "fail",
|
|
400
|
+
`${matPrefix()}text at offset ${offset} of "${matShown}" is not covered by any segment: ${JSON.stringify(excerpt)}${more}`,
|
|
401
|
+
"Add a segment for every non-whitespace span, or extend an existing segment's start/end."));
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
return { header, segments, findings };
|
|
407
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
// @supersuit/hyperspec/writing: the reading side of materials marking, for a tool outside this
|
|
2
|
+
// package that labels materials (an agent, an editor, the writing engine's capture stage). It gets
|
|
3
|
+
// the closed label set and the same parse-and-validate the linter runs; labeling itself stays with
|
|
4
|
+
// whoever calls this.
|
|
5
|
+
export { MATERIAL_LABELS } from "./labels.mjs";
|
|
6
|
+
export { readSegments } from "./segments.mjs";
|
package/src/writing-fields.mjs
CHANGED
|
@@ -20,11 +20,12 @@
|
|
|
20
20
|
|
|
21
21
|
import { statSync } from "node:fs";
|
|
22
22
|
import { str } from "./placeholder.mjs";
|
|
23
|
+
import { readSegments } from "./segments.mjs";
|
|
23
24
|
|
|
24
25
|
const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
|
|
25
26
|
const list = (v) => (Array.isArray(v) ? v : []);
|
|
26
27
|
const isObj = (v) => v != null && typeof v === "object" && !Array.isArray(v);
|
|
27
|
-
// "file", "other" (a directory or a device), or null when nothing is there at all
|
|
28
|
+
// "file", "other" (a directory or a device), or null when nothing is there at all. This is the same
|
|
28
29
|
// three-way classification the core examples rule uses (src/rules.mjs's kind()), so a real
|
|
29
30
|
// directory is reported as "is not a file" rather than the misleading "does not exist".
|
|
30
31
|
const pathKind = (here, p) => { try { return statSync(here(p)).isFile() ? "file" : "other"; } catch { return null; } };
|
|
@@ -72,10 +73,62 @@ function materialsFields(raw, d, here, idPrefix) {
|
|
|
72
73
|
const p = str(it?.path);
|
|
73
74
|
if (!p) out.push(f(1, `${idPrefix}-item-${i}-path`, "fail", `material ${tag} has no path`, "Add path: to the material."));
|
|
74
75
|
else out.push(...pathFindings(here, p, `${idPrefix}-item-${i}-path`, "material", "Fix the path, or add the material file."));
|
|
76
|
+
|
|
77
|
+
// Marking is required from 0.4 on. A material item with no segments: field is not
|
|
78
|
+
// marked at all (the design puts marking before specifying), so it fails on its own, distinct
|
|
79
|
+
// from the segments file existing but being broken (readSegments' own findings below). The
|
|
80
|
+
// material's text-dependent checks (verbatim, coverage, overlap, staleness) only run when the
|
|
81
|
+
// path itself already resolved to a real file, so a broken path is never reported twice: once
|
|
82
|
+
// here for the path field and again for the material readSegments could not read.
|
|
83
|
+
const segPath = str(it?.segments);
|
|
84
|
+
if (!segPath) {
|
|
85
|
+
out.push(f(1, "writing-materials-unmarked", "fail",
|
|
86
|
+
`material ${tag} is not marked (no segments field)`,
|
|
87
|
+
"Run `hyperspec segments init <material> --id <id>`, then add segments: to the material item."));
|
|
88
|
+
} else {
|
|
89
|
+
const { findings: segFindings } = readSegments(here(segPath), { materialPath: materialFilePath(it, here), materialId: id || undefined, ...shownPaths(it, segPath) });
|
|
90
|
+
out.push(...segFindings);
|
|
91
|
+
}
|
|
75
92
|
});
|
|
76
93
|
return out;
|
|
77
94
|
}
|
|
78
95
|
|
|
96
|
+
// The material file readSegments checks segment text against, or undefined when the item's own
|
|
97
|
+
// path is missing or is not a file. That broken path is already reported by pathFindings (test 6);
|
|
98
|
+
// passing it on would make readSegments report the same root cause a second time, under test 1.
|
|
99
|
+
// materialsFields and resolveMaterialSegments both resolve through here, so they cannot disagree.
|
|
100
|
+
function materialFilePath(item, here) {
|
|
101
|
+
const p = str(item?.path);
|
|
102
|
+
return p && pathKind(here, p) === "file" ? here(p) : undefined;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// The paths readSegments' messages print: exactly as the spec wrote them, the way every other path
|
|
106
|
+
// finding in the linter reads, never resolved against the spec's folder. A finding pasted into a
|
|
107
|
+
// public issue then names no one's home folder, and --json is the same on every machine.
|
|
108
|
+
function shownPaths(item, segPath) {
|
|
109
|
+
return { displayPath: segPath, materialDisplayPath: str(item?.path) || undefined };
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// Resolves ONE material item's segments (for spine ref resolution below). Never pushes
|
|
113
|
+
// readSegments' own findings: those are already reported once, by materialsFields, under the
|
|
114
|
+
// materials block; this is read-only lookup. When nothing resolves, `why` says which of the three
|
|
115
|
+
// causes it was, because each needs a different fix: "unmarked" (no segments: field), "unreadable"
|
|
116
|
+
// (the segments file does not exist or cannot be read) or "empty" (read, but no segment lines).
|
|
117
|
+
function resolveMaterialSegments(item, here) {
|
|
118
|
+
const segPath = str(item?.segments);
|
|
119
|
+
if (!segPath) return { segments: [], loaded: false, why: "unmarked" };
|
|
120
|
+
const { segments, findings } = readSegments(here(segPath), { materialPath: materialFilePath(item, here), materialId: str(item?.id) || undefined, ...shownPaths(item, segPath) });
|
|
121
|
+
if (segments.length > 0) return { segments, loaded: true };
|
|
122
|
+
const unreadable = findings.some((x) => x.id === "writing-materials-segments-missing");
|
|
123
|
+
return { segments, loaded: false, why: unreadable ? "unreadable" : "empty" };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const UNRESOLVABLE_BECAUSE = {
|
|
127
|
+
unmarked: "it is not marked (no segments field)",
|
|
128
|
+
unreadable: "its segments file could not be read",
|
|
129
|
+
empty: "its segments file has no segments",
|
|
130
|
+
};
|
|
131
|
+
|
|
79
132
|
// ---------------------------------------------------------------- 2. dna ----------------------
|
|
80
133
|
|
|
81
134
|
function dnaFields(raw, d, here, idPrefix) {
|
|
@@ -222,7 +275,19 @@ function spineFields(raw, d, here, idPrefix) {
|
|
|
222
275
|
distinct += 1;
|
|
223
276
|
});
|
|
224
277
|
if (distinct < 3 || distinct > 7) out.push(f(1, `${idPrefix}-claims-count`, "fail", `writing.spine has ${distinct} distinct claims, outside 3 to 7`, "List 3 to 7 claims, each with its own id, under spine.claims."));
|
|
225
|
-
const
|
|
278
|
+
const items = list(d.writing?.materials?.items);
|
|
279
|
+
const itemsById = new Map(items.map((m) => [str(m?.id), m]).filter(([id]) => id));
|
|
280
|
+
const materialIds = new Set(itemsById.keys());
|
|
281
|
+
// A material whose segments cannot be resolved at all (no segments: field, a segments file that
|
|
282
|
+
// cannot be read, or one with no segment lines) makes every #segment ref against it equally
|
|
283
|
+
// unresolvable. Report that once per material, not once per ref: two claims both pointing at
|
|
284
|
+
// "m1#s1" and "m1#s2" when m1 is unmarked are the same underlying problem, not two.
|
|
285
|
+
const segmentsCache = new Map();
|
|
286
|
+
const segmentsFor = (mid) => {
|
|
287
|
+
if (!segmentsCache.has(mid)) segmentsCache.set(mid, resolveMaterialSegments(itemsById.get(mid), here));
|
|
288
|
+
return segmentsCache.get(mid);
|
|
289
|
+
};
|
|
290
|
+
const reportedUnresolvable = new Set();
|
|
226
291
|
claims.forEach((c, i) => {
|
|
227
292
|
const cid = str(c?.id) || `#${i + 1}`;
|
|
228
293
|
if (!str(c?.id)) out.push(f(1, `${idPrefix}-claim-${i}-id`, "fail", `spine claim ${cid} has no id`, "Give it a short id, e.g. c1."));
|
|
@@ -232,8 +297,44 @@ function spineFields(raw, d, here, idPrefix) {
|
|
|
232
297
|
out.push(f(4, `${idPrefix}-claim-${i}-materials`, "fail", `spine claim "${cid}" has no materials`, "Point materials: at one or more material ids."));
|
|
233
298
|
} else {
|
|
234
299
|
refs.forEach((ref) => {
|
|
235
|
-
const
|
|
236
|
-
|
|
300
|
+
const hashIdx = ref.indexOf("#");
|
|
301
|
+
const mid = hashIdx === -1 ? ref : ref.slice(0, hashIdx);
|
|
302
|
+
const segId = hashIdx === -1 ? "" : ref.slice(hashIdx + 1);
|
|
303
|
+
if (!materialIds.has(mid)) {
|
|
304
|
+
out.push(f(4, `${idPrefix}-claim-${i}-materials-unknown`, "fail", `spine claim "${cid}" points at material "${ref}", which is not in writing.materials.items`, "Point materials: at an id that exists in writing.materials.items."));
|
|
305
|
+
return;
|
|
306
|
+
}
|
|
307
|
+
// A bare material id (no "#") stays valid on its own; only a ref naming a specific segment
|
|
308
|
+
// needs resolving against that material's segments file. "m1#" names an empty segment id,
|
|
309
|
+
// which is not a bare ref, so it goes on to fail as an unknown segment.
|
|
310
|
+
if (hashIdx === -1) return;
|
|
311
|
+
const { segments, loaded, why } = segmentsFor(mid);
|
|
312
|
+
if (!loaded) {
|
|
313
|
+
if (!reportedUnresolvable.has(mid)) {
|
|
314
|
+
out.push(f(4, `${idPrefix}-materials-segments-unresolvable-${mid}`, "fail",
|
|
315
|
+
`spine claims point at material "${mid}"'s segments, but ${UNRESOLVABLE_BECAUSE[why]}`,
|
|
316
|
+
"Run `hyperspec segments init` on the material, label every segment, then re-check the spine refs."));
|
|
317
|
+
reportedUnresolvable.add(mid);
|
|
318
|
+
}
|
|
319
|
+
return;
|
|
320
|
+
}
|
|
321
|
+
const seg = segId ? segments.find((s) => str(s?.id) === segId) : undefined;
|
|
322
|
+
if (!seg) {
|
|
323
|
+
out.push(f(4, `${idPrefix}-claim-${i}-materials-segment-unknown`, "fail",
|
|
324
|
+
`spine claim "${cid}" points at material "${ref}", which is not a segment in "${mid}"'s segments file`,
|
|
325
|
+
"Point materials: at a segment id that exists in the material's segments file, or drop the #segment suffix to reference the whole material."));
|
|
326
|
+
return;
|
|
327
|
+
}
|
|
328
|
+
const label = typeof seg.label === "string" ? seg.label : "";
|
|
329
|
+
if (label === "private") {
|
|
330
|
+
out.push(f(5, `${idPrefix}-claim-${i}-materials-segment-private`, "fail",
|
|
331
|
+
`spine claim "${cid}" points at material "${ref}", which is labeled private (private is never used)`,
|
|
332
|
+
"Point materials: at a different segment, or drop this ref."));
|
|
333
|
+
} else if (label === "question") {
|
|
334
|
+
out.push(f(5, `${idPrefix}-claim-${i}-materials-segment-question`, "fail",
|
|
335
|
+
`spine claim "${cid}" points at material "${ref}", which is labeled question (a question is never an assertion)`,
|
|
336
|
+
"Point materials: at a different segment, or drop this ref."));
|
|
337
|
+
}
|
|
237
338
|
});
|
|
238
339
|
}
|
|
239
340
|
});
|
|
@@ -266,7 +367,7 @@ function characterFields(c, here, idPrefix) {
|
|
|
266
367
|
if (!str(c?.id)) out.push(f(1, `${idPrefix}-id`, "fail", "character has no id", "Give it a short id."));
|
|
267
368
|
|
|
268
369
|
// Raw array length, matching dnaFields' goldens.length check: a present-but-malformed entry
|
|
269
|
-
// (missing by or knows) must not ALSO trigger "has no knowledge"
|
|
370
|
+
// (missing by or knows) must not ALSO trigger "has no knowledge"; that is only true when the
|
|
270
371
|
// list is literally empty.
|
|
271
372
|
const knowledge = list(c?.knowledge);
|
|
272
373
|
if (!knowledge.length) out.push(f(1, `${idPrefix}-knowledge`, "fail", `character "${tag}" has no knowledge`, "Add at least one { by, knows } entry under knowledge."));
|
package/src/writing-template.mjs
CHANGED
|
@@ -14,6 +14,12 @@
|
|
|
14
14
|
// block. Without that rule the character block, whose checks are all presence checks, would lint
|
|
15
15
|
// clean the moment init wrote it.
|
|
16
16
|
//
|
|
17
|
+
// Two values are not placeholders. The material item's segments: names the file `hyperspec
|
|
18
|
+
// segments init materials/TODO.md` would write, so it follows the path placeholder beside it and
|
|
19
|
+
// fails as a segments file that does not exist yet (every material must be marked). And the
|
|
20
|
+
// materials check.station is the marking station itself, since that check is the same for every
|
|
21
|
+
// writing spec: the linter enforces it, and there is nothing for the operator to decide there.
|
|
22
|
+
//
|
|
17
23
|
// init with no --profile never imports or calls this file: template.mjs's own template() is
|
|
18
24
|
// untouched, so a bare init is still byte-for-byte what it always was.
|
|
19
25
|
import { scalar } from "./template.mjs";
|
|
@@ -93,12 +99,13 @@ writing:
|
|
|
93
99
|
items:
|
|
94
100
|
- id: m1
|
|
95
101
|
path: materials/TODO.md
|
|
102
|
+
segments: materials/TODO.md.segments.jsonl
|
|
96
103
|
produced_by: TODO
|
|
97
104
|
captured: TODO
|
|
98
105
|
how: TODO
|
|
99
106
|
trust: TODO
|
|
100
107
|
check:
|
|
101
|
-
station:
|
|
108
|
+
station: every segment of every material carries a label from the closed set, matches its source verbatim, and the markings are current
|
|
102
109
|
source: TODO
|
|
103
110
|
author: TODO
|
|
104
111
|
dna:
|
package/src/writing.mjs
CHANGED
|
@@ -18,10 +18,10 @@ const f = (test, id, severity, message, fix) => ({ test, id, severity, message,
|
|
|
18
18
|
const list = (v) => (Array.isArray(v) ? v : []);
|
|
19
19
|
const isObj = (v) => v != null && typeof v === "object" && !Array.isArray(v);
|
|
20
20
|
|
|
21
|
-
// The closed vocabulary
|
|
22
|
-
//
|
|
23
|
-
//
|
|
24
|
-
export
|
|
21
|
+
// The closed vocabulary every segment of a material is labeled from. Defined once, in the leaf
|
|
22
|
+
// module src/labels.mjs (see there for why it is not defined here), and re-exported so existing
|
|
23
|
+
// imports from this file keep working.
|
|
24
|
+
export { MATERIAL_LABELS } from "./labels.mjs";
|
|
25
25
|
|
|
26
26
|
// The nine writing blocks, in schema order. "characters" is the one block that is not always
|
|
27
27
|
// required: it is required only when fiction: true, everywhere else in this file and in
|