@supersuit/hyperspec 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -0
- package/README.md +53 -3
- package/SPEC.md +24 -10
- package/WRITING.md +597 -0
- package/bin/hyperspec.mjs +94 -3
- package/examples/minimal.hyperspec.md +2 -2
- package/examples/writing/essay/goldens/close.md +2 -0
- package/examples/writing/essay/goldens/opening.md +2 -0
- package/examples/writing/essay/materials/interview-notes.md +12 -0
- package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +10 -0
- package/examples/writing/essay/materials/team-survey.md +7 -0
- package/examples/writing/essay/materials/team-survey.md.segments.jsonl +6 -0
- package/examples/writing/essay/materials/voice-memo.md +18 -0
- package/examples/writing/essay/materials/voice-memo.md.segments.jsonl +7 -0
- package/examples/writing/essay/runs.jsonl +0 -0
- package/examples/writing/essay.hyperspec.md +220 -0
- package/examples/writing/story/goldens/dialogue.md +3 -0
- package/examples/writing/story/goldens/opening.md +3 -0
- package/examples/writing/story/materials/bakery-visit.md +9 -0
- package/examples/writing/story/materials/bakery-visit.md.segments.jsonl +13 -0
- package/examples/writing/story/materials/notes.md +16 -0
- package/examples/writing/story/materials/notes.md.segments.jsonl +7 -0
- package/examples/writing/story/materials/scene-list.md +7 -0
- package/examples/writing/story/materials/scene-list.md.segments.jsonl +14 -0
- package/examples/writing/story/runs.jsonl +0 -0
- package/examples/writing/story.hyperspec.md +288 -0
- package/examples/writing/style-rules.md +19 -0
- package/package.json +4 -2
- package/src/blobs.mjs +1 -1
- package/src/compare.mjs +6 -6
- package/src/fsutil.mjs +1 -1
- package/src/labels.mjs +6 -0
- package/src/placeholder.mjs +20 -0
- package/src/profiles.mjs +50 -0
- package/src/reproduce.mjs +5 -5
- package/src/rules.mjs +25 -13
- package/src/score.mjs +7 -1
- package/src/segments.mjs +407 -0
- package/src/template.mjs +4 -1
- package/src/writing-exports.mjs +6 -0
- package/src/writing-fields.mjs +418 -0
- package/src/writing-template.mjs +199 -0
- package/src/writing.mjs +181 -0
package/src/segments.mjs
ADDED
|
@@ -0,0 +1,407 @@
|
|
|
1
|
+
// Materials marking (hyperspec 0.4). hyperspec never calls a model: a material is split into candidate
|
|
2
|
+
// segments deterministically (splitSegments), an agent or a person labels each one by hand-editing
|
|
3
|
+
// the JSONL file `segments init` wrote, and this module reads that file back and checks it
|
|
4
|
+
// (readSegments). Labeling itself is judgment and stays outside this file entirely; everything
|
|
5
|
+
// here is either splitting (no judgment) or checking (closed rules, no model).
|
|
6
|
+
//
|
|
7
|
+
// Segments file format (JSONL, one JSON object per line): line 1 is a header
|
|
8
|
+
// {"material":"<id>","path":"<material path>","sha256":"<hex>"}; every later line is one segment
|
|
9
|
+
// {"id":"s1","start":0,"end":212,"label":"claim"|"story"|"quote"|"stance"|"question"|"aside"|
|
|
10
|
+
// "private"|"unlabeled", ...label fields, "text":"..."}. start/end are JS string indices (UTF-16
|
|
11
|
+
// code units, the same indices readFileSync(path, "utf8") hands back) into the material's text,
|
|
12
|
+
// end exclusive; text must equal source.slice(start, end) exactly.
|
|
13
|
+
|
|
14
|
+
import { readFileSync } from "node:fs";
|
|
15
|
+
import { basename } from "node:path";
|
|
16
|
+
import { sha256 } from "./hash.mjs";
|
|
17
|
+
import { MATERIAL_LABELS } from "./labels.mjs";
|
|
18
|
+
import { str } from "./placeholder.mjs";
|
|
19
|
+
|
|
20
|
+
const f = (test, id, severity, message, fix) => ({ test, id, severity, message, fix });
|
|
21
|
+
// own counts as set when it is the JSON boolean true or the string "true"; anything else
|
|
22
|
+
// (false, "false", missing, any other value) is unset.
|
|
23
|
+
const ownIsSet = (obj) => obj?.own === true || obj?.own === "true";
|
|
24
|
+
// The SAME whitespace test the coverage check below uses (/\S/, JS's Unicode-aware class, which
|
|
25
|
+
// includes U+00A0 NBSP and the other Unicode space separators, not just the ASCII set). One
|
|
26
|
+
// definition shared by both: a gap the sentence splitter treats as pure separator and a gap the
|
|
27
|
+
// coverage check treats as "nothing to require a segment for" must never disagree, or a stretch
|
|
28
|
+
// of NBSP-only text could read as covered to one check and as real content to the other.
|
|
29
|
+
const isWhitespace = (ch) => /\s/.test(ch);
|
|
30
|
+
|
|
31
|
+
// -------------------------------------------------------------------------------------------
|
|
32
|
+
// splitSegments: deterministic, no judgment. Candidate segments only ("unlabeled"); a person or
|
|
33
|
+
// agent assigns real labels afterward by editing the JSONL file this feeds.
|
|
34
|
+
|
|
35
|
+
// A line's [start, contentEnd) span, contentEnd excluding that line's own terminator ("\n" or, for
|
|
36
|
+
// CRLF, "\r\n"; the \r is excluded from content the same way the \n is, so a blank CRLF line
|
|
37
|
+
// reads as blank, but the \r is never dropped from the original text itself: it simply falls
|
|
38
|
+
// outside every line's content span, same as \n does, and stays untouched wherever it sits inside
|
|
39
|
+
// a multi-line segment). The final line (no trailing "\n") has contentEnd === text.length.
|
|
40
|
+
function lineSpans(text) {
|
|
41
|
+
const spans = [];
|
|
42
|
+
let start = 0;
|
|
43
|
+
while (start <= text.length) {
|
|
44
|
+
const nl = text.indexOf("\n", start);
|
|
45
|
+
if (nl === -1) { spans.push({ start, contentEnd: text.length }); break; }
|
|
46
|
+
let contentEnd = nl;
|
|
47
|
+
if (contentEnd > start && text[contentEnd - 1] === "\r") contentEnd -= 1;
|
|
48
|
+
spans.push({ start, contentEnd });
|
|
49
|
+
start = nl + 1;
|
|
50
|
+
}
|
|
51
|
+
return spans;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const isBlankLine = (content) => /^[ \t]*$/.test(content);
|
|
55
|
+
|
|
56
|
+
// Paragraphs: a maximal run of consecutive non-blank lines. The blank line(s) between two
|
|
57
|
+
// paragraphs, and their terminators, are never part of either paragraph's span: "whitespace
|
|
58
|
+
// between segments is left out of every segment." A multi-line paragraph keeps the single "\n" (or
|
|
59
|
+
// "\r\n") between its own lines, since that is internal to the paragraph, not a separator.
|
|
60
|
+
function splitParagraphs(text) {
|
|
61
|
+
const lines = lineSpans(text);
|
|
62
|
+
const out = [];
|
|
63
|
+
let curStart = null;
|
|
64
|
+
let curEnd = null;
|
|
65
|
+
for (const line of lines) {
|
|
66
|
+
const content = text.slice(line.start, line.contentEnd);
|
|
67
|
+
if (isBlankLine(content)) {
|
|
68
|
+
if (curStart !== null) { out.push({ start: curStart, end: curEnd }); curStart = null; }
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
if (curStart === null) curStart = line.start;
|
|
72
|
+
curEnd = line.contentEnd;
|
|
73
|
+
}
|
|
74
|
+
if (curStart !== null) out.push({ start: curStart, end: curEnd });
|
|
75
|
+
return out;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// Sentences: split on ".", "?" or "!" followed by whitespace (or end of text), never while inside
|
|
79
|
+
// a quoted span (a run opened by '"' and not yet closed) on the SAME line. inQuote resets at
|
|
80
|
+
// every "\n" regardless of whether a quote is actually still open, so a stray unterminated quote
|
|
81
|
+
// can never suppress boundaries on a later line. Punctuation strictly INSIDE an open quote on the
|
|
82
|
+
// same line (the case this rule exists for, e.g. a title like "Wait." spoken mid-sentence) is
|
|
83
|
+
// never itself a boundary; the next real, unquoted terminator is.
|
|
84
|
+
//
|
|
85
|
+
// A terminal punctuation mark immediately followed by a closing '"' DOES still count as a
|
|
86
|
+
// boundary, and that "punctuation + closing quote" branch is reachable more often than "same
|
|
87
|
+
// line" alone suggests: because inQuote resets on every "\n", a quote opened on one line and
|
|
88
|
+
// closed with punctuation on a LATER line is, by the time that later line's punctuation is
|
|
89
|
+
// reached, read as not-currently-in-a-quote, so the branch fires there too, closing the segment
|
|
90
|
+
// right after the quote mark (`He said "long\nquote here." Next.` -> one segment ending after the
|
|
91
|
+
// closing quote, then "Next."). The same branch also fires on an orphan closing quote with no
|
|
92
|
+
// matching open one on its own line (`He was done." Next.` splits the same way); this is a
|
|
93
|
+
// side effect of the same rule rather than special-cased, and is deterministic either way.
|
|
94
|
+
//
|
|
95
|
+
// A line break followed by a list marker is also a boundary: optional spaces or tabs, then "-", "*",
|
|
96
|
+
// "+", or digits followed by "." or ")", then a space. Without this, a bullet that is entirely a
|
|
97
|
+
// quotation ending in `."` never meets an unquoted terminator (its punctuation sits inside the
|
|
98
|
+
// quote) and runs into the next bullet. A segment that opens on a list marker steps over the
|
|
99
|
+
// marker first, so a numbered item's own "1." or "2)" is never read as a sentence ending. A line
|
|
100
|
+
// that merely starts with a hyphenated word ("self-evident"), a decimal ("3.5") or a hyphen with
|
|
101
|
+
// no space after it ("-ish") is not a marker and does not split.
|
|
102
|
+
//
|
|
103
|
+
// Leading and trailing whitespace around the whole text, and the whitespace run between sentences,
|
|
104
|
+
// is left out of every segment, the same as splitParagraphs.
|
|
105
|
+
const LIST_MARKER = /[ \t]*(?:[-*+]|\d+[.)]) /y;
|
|
106
|
+
|
|
107
|
+
// If a list marker begins at `pos` (after optional spaces or tabs), the index just past the marker
|
|
108
|
+
// symbol (before its trailing space), else -1. Callers only ask at the start of a line.
|
|
109
|
+
function listMarkerEnd(text, pos) {
|
|
110
|
+
LIST_MARKER.lastIndex = pos;
|
|
111
|
+
return LIST_MARKER.test(text) ? LIST_MARKER.lastIndex - 1 : -1;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// True when `pos` sits at the start of a line once any leading spaces or tabs are skipped back over.
|
|
115
|
+
function atLineStart(text, pos) {
|
|
116
|
+
let k = pos - 1;
|
|
117
|
+
while (k >= 0 && (text[k] === " " || text[k] === "\t")) k--;
|
|
118
|
+
return k < 0 || text[k] === "\n";
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function splitSentences(text) {
|
|
122
|
+
const n = text.length;
|
|
123
|
+
const out = [];
|
|
124
|
+
let inQuote = false;
|
|
125
|
+
let i = 0;
|
|
126
|
+
while (i < n && isWhitespace(text[i])) i++;
|
|
127
|
+
let start = i;
|
|
128
|
+
// A segment that opens on a list marker begins scanning past it, so "1." is not a terminator.
|
|
129
|
+
const skipMarker = (pos) => (atLineStart(text, pos) ? Math.max(pos, listMarkerEnd(text, pos)) : pos);
|
|
130
|
+
i = skipMarker(start);
|
|
131
|
+
while (i < n) {
|
|
132
|
+
const ch = text[i];
|
|
133
|
+
if (ch === "\n") {
|
|
134
|
+
inQuote = false;
|
|
135
|
+
const markerEnd = listMarkerEnd(text, i + 1);
|
|
136
|
+
if (markerEnd !== -1) {
|
|
137
|
+
let end = i;
|
|
138
|
+
while (end > start && isWhitespace(text[end - 1])) end--;
|
|
139
|
+
if (end > start) out.push({ start, end });
|
|
140
|
+
let k = i + 1;
|
|
141
|
+
while (text[k] === " " || text[k] === "\t") k++;
|
|
142
|
+
start = k;
|
|
143
|
+
i = markerEnd;
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
i++;
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
if (ch === '"') { inQuote = !inQuote; i++; continue; }
|
|
150
|
+
if (!inQuote && (ch === "." || ch === "?" || ch === "!")) {
|
|
151
|
+
let j = i + 1;
|
|
152
|
+
while (j < n && (text[j] === "." || text[j] === "?" || text[j] === "!")) j++;
|
|
153
|
+
let end = j;
|
|
154
|
+
if (j < n && text[j] === '"') { end = j + 1; inQuote = false; }
|
|
155
|
+
if (end >= n || isWhitespace(text[end])) {
|
|
156
|
+
out.push({ start, end });
|
|
157
|
+
let k = end;
|
|
158
|
+
while (k < n && isWhitespace(text[k])) k++;
|
|
159
|
+
start = k;
|
|
160
|
+
i = skipMarker(k);
|
|
161
|
+
continue;
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
i++;
|
|
165
|
+
}
|
|
166
|
+
if (start < n) {
|
|
167
|
+
let end = n;
|
|
168
|
+
while (end > start && isWhitespace(text[end - 1])) end--;
|
|
169
|
+
if (end > start) out.push({ start, end });
|
|
170
|
+
}
|
|
171
|
+
return out;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// splitSegments(text, { by }): "paragraph" (default) or "sentence". Returns candidate segments,
|
|
175
|
+
// each { id: "s<n>", start, end, label: "unlabeled", text }, in document order, ids 1-based and
|
|
176
|
+
// dense. "unlabeled" is what this writes and is never a valid label in readSegments below.
|
|
177
|
+
export function splitSegments(text, { by = "paragraph" } = {}) {
|
|
178
|
+
if (by !== "paragraph" && by !== "sentence") {
|
|
179
|
+
throw new RangeError(`splitSegments: by must be "paragraph" or "sentence", got ${JSON.stringify(by)}`);
|
|
180
|
+
}
|
|
181
|
+
const spans = by === "sentence" ? splitSentences(text) : splitParagraphs(text);
|
|
182
|
+
return spans.map((s, i) => ({ id: `s${i + 1}`, start: s.start, end: s.end, label: "unlabeled", text: text.slice(s.start, s.end) }));
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// -------------------------------------------------------------------------------------------
|
|
186
|
+
// readSegments: parse + validate a marked-up segments file. Never throws on a bad or missing
|
|
187
|
+
// file; every failure mode becomes a finding in the lint shape (test, id, severity, message, fix),
|
|
188
|
+
// the same shape src/rules.mjs and src/writing-fields.mjs already use. readSegments never reads
|
|
189
|
+
// MATERIAL_LABELS' meaning into anything beyond the closed-set and per-label-field checks below;
|
|
190
|
+
// spine-ref resolution (m1#s3, and the private/question refusal) happens in
|
|
191
|
+
// src/writing-fields.mjs, which calls this and then checks refs against the returned segments.
|
|
192
|
+
//
|
|
193
|
+
// materialPath is optional: without it, only the structural checks that need no material text run
|
|
194
|
+
// (header shape, labels, per-label fields, duplicate ids). With it, the material-text-dependent
|
|
195
|
+
// checks run too: verbatim text, coverage, overlap and staleness. materialId, when given, is
|
|
196
|
+
// compared against the header's own "material" field.
|
|
197
|
+
//
|
|
198
|
+
// EVERY finding message names which material it is about (and the segment id, where one
|
|
199
|
+
// applies), so two materials that each have a broken "s1" never produce identical-looking
|
|
200
|
+
// findings. The finding `id` fields stay rule ids and may still repeat across materials (the same
|
|
201
|
+
// way core findings do); it is the message text this rule is about. The material tag preferred,
|
|
202
|
+
// in order: the caller's own materialId, else the header's own "material" field (once parsed),
|
|
203
|
+
// else the segments file's own basename. The last resort covers the missing-file and
|
|
204
|
+
// unparsable-header cases, where neither of the first two is available yet.
|
|
205
|
+
export function readSegments(segmentsPath, { materialPath, materialId, displayPath, materialDisplayPath } = {}) {
|
|
206
|
+
const findings = [];
|
|
207
|
+
const fallbackTag = basename(segmentsPath);
|
|
208
|
+
// What messages print for the two files: the caller's display paths when given (lint passes the
|
|
209
|
+
// paths as the spec wrote them), else the paths exactly as given. Never a path this function
|
|
210
|
+
// resolved itself, so a finding reads the same on every machine.
|
|
211
|
+
const segShown = displayPath ?? segmentsPath;
|
|
212
|
+
const matShown = materialDisplayPath ?? materialPath;
|
|
213
|
+
|
|
214
|
+
let raw;
|
|
215
|
+
try {
|
|
216
|
+
raw = readFileSync(segmentsPath, "utf8");
|
|
217
|
+
} catch {
|
|
218
|
+
const matTag = materialId ?? fallbackTag;
|
|
219
|
+
findings.push(f(1, "writing-materials-segments-missing", "fail",
|
|
220
|
+
`material ${matTag}: is not marked (segments file "${segShown}" does not exist or cannot be read)`,
|
|
221
|
+
`Run \`hyperspec segments init <material> --id <id>\` to write it, then label every segment.`));
|
|
222
|
+
return { header: null, segments: [], findings };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const rawLines = raw.split("\n");
|
|
226
|
+
const headerRaw = rawLines[0] ?? "";
|
|
227
|
+
// Any line whose whole trimmed content is empty is skipped (a trailing blank line left by a
|
|
228
|
+
// final "\n" is the common case, but a genuinely blank line mid-file is tolerated the same way
|
|
229
|
+
// ordinary JSONL readers tolerate one); line 1 is the only line this reader treats as load-
|
|
230
|
+
// bearing on its own, per the format.
|
|
231
|
+
const segLines = rawLines.slice(1).map((text, i) => ({ n: i + 2, text })).filter((l) => l.text.trim() !== "");
|
|
232
|
+
|
|
233
|
+
let header = null;
|
|
234
|
+
try {
|
|
235
|
+
const parsed = JSON.parse(headerRaw);
|
|
236
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) header = parsed;
|
|
237
|
+
else throw new Error("header is not a JSON object");
|
|
238
|
+
} catch {
|
|
239
|
+
// header stays null; matTag below falls through materialId -> fallbackTag, since there is no
|
|
240
|
+
// header.material to read yet.
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// Computed once header parsing has settled, so every finding from here on (including the
|
|
244
|
+
// header's own shape findings) can use it. header?.material is only trusted when it is a real,
|
|
245
|
+
// non-empty string; a header that fails its own presence check contributes nothing here.
|
|
246
|
+
const matTag = materialId ?? (header && str(header.material) ? str(header.material) : fallbackTag);
|
|
247
|
+
const matPrefix = (segId) => (segId ? `material ${matTag}, segment ${segId}: ` : `material ${matTag}: `);
|
|
248
|
+
|
|
249
|
+
if (!header) {
|
|
250
|
+
findings.push(f(1, "writing-materials-header", "fail",
|
|
251
|
+
`${matPrefix()}segments file "${segShown}" line 1 is not a valid JSON header object`,
|
|
252
|
+
`Fix line 1 to a JSON object: {"material":"<id>","path":"<material path>","sha256":"<hex>"}.`));
|
|
253
|
+
} else {
|
|
254
|
+
for (const key of ["material", "path", "sha256"]) {
|
|
255
|
+
if (!str(header[key])) {
|
|
256
|
+
findings.push(f(1, "writing-materials-header", "fail",
|
|
257
|
+
`${matPrefix()}segments file "${segShown}" header has no ${key}`,
|
|
258
|
+
`Add ${key}: to the header line (line 1).`));
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
if (materialId !== undefined && str(header.material) && str(header.material) !== materialId) {
|
|
262
|
+
findings.push(f(1, "writing-materials-header-material", "fail",
|
|
263
|
+
`${matPrefix()}segments file "${segShown}" header names material "${header.material}", not "${materialId}"`,
|
|
264
|
+
`Set the header's material to "${materialId}", or point the item at the right segments file.`));
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
const segments = [];
|
|
269
|
+
const idsSeen = new Set();
|
|
270
|
+
const idsReportedDup = new Set();
|
|
271
|
+
|
|
272
|
+
for (const { n, text } of segLines) {
|
|
273
|
+
let obj;
|
|
274
|
+
try {
|
|
275
|
+
obj = JSON.parse(text);
|
|
276
|
+
} catch {
|
|
277
|
+
findings.push(f(1, `writing-materials-json-line-${n}`, "fail",
|
|
278
|
+
`${matPrefix()}segments file "${segShown}" line ${n} is not valid JSON`,
|
|
279
|
+
"Fix the JSON on that line."));
|
|
280
|
+
continue;
|
|
281
|
+
}
|
|
282
|
+
if (!obj || typeof obj !== "object" || Array.isArray(obj)) {
|
|
283
|
+
findings.push(f(1, `writing-materials-json-line-${n}`, "fail",
|
|
284
|
+
`${matPrefix()}segments file "${segShown}" line ${n} is not a JSON object`,
|
|
285
|
+
"Each segment line must be a JSON object."));
|
|
286
|
+
continue;
|
|
287
|
+
}
|
|
288
|
+
segments.push(obj);
|
|
289
|
+
|
|
290
|
+
const id = str(obj.id);
|
|
291
|
+
const segTag = id || `line ${n}`;
|
|
292
|
+
if (!id) {
|
|
293
|
+
findings.push(f(1, `writing-materials-segment-id-${n}`, "fail", `${matPrefix()}segment on line ${n} has no id`, "Give it a short id, e.g. s1."));
|
|
294
|
+
} else if (idsSeen.has(id)) {
|
|
295
|
+
if (!idsReportedDup.has(id)) {
|
|
296
|
+
findings.push(f(1, "writing-materials-segment-id", "fail", `${matPrefix(id)}segment id is used twice`, "Ids must be unique within the segments file; rename one."));
|
|
297
|
+
}
|
|
298
|
+
idsReportedDup.add(id);
|
|
299
|
+
} else {
|
|
300
|
+
idsSeen.add(id);
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
const label = typeof obj.label === "string" ? obj.label : "";
|
|
304
|
+
if (!MATERIAL_LABELS.includes(label)) {
|
|
305
|
+
findings.push(f(1, `writing-materials-label-${segTag}`, "fail",
|
|
306
|
+
label === "unlabeled" || !label
|
|
307
|
+
? `${matPrefix(segTag)}is still unlabeled`
|
|
308
|
+
: `${matPrefix(segTag)}label "${label}" is outside ${MATERIAL_LABELS.join(", ")}`,
|
|
309
|
+
`Set label to one of: ${MATERIAL_LABELS.join(", ")}.`));
|
|
310
|
+
} else if (label === "claim" && !(str(obj.source) || ownIsSet(obj))) {
|
|
311
|
+
findings.push(f(4, `writing-materials-claim-source-${segTag}`, "fail",
|
|
312
|
+
`${matPrefix(segTag)}claim has no source and is not marked own`,
|
|
313
|
+
`Add source: to segment "${segTag}", or set own: true if it is the author's own claim, said as such.`));
|
|
314
|
+
} else if (label === "story" && !str(obj.teller)) {
|
|
315
|
+
findings.push(f(4, `writing-materials-story-teller-${segTag}`, "fail", `${matPrefix(segTag)}story has no teller`, `Add teller: to segment "${segTag}".`));
|
|
316
|
+
} else if (label === "quote" && !str(obj.speaker)) {
|
|
317
|
+
findings.push(f(4, `writing-materials-quote-speaker-${segTag}`, "fail", `${matPrefix(segTag)}quote has no speaker`, `Add speaker: to segment "${segTag}".`));
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Material-text-dependent checks: verbatim text, shape, overlap, coverage, staleness. Skipped
|
|
322
|
+
// entirely when no materialPath is given (structural-only mode).
|
|
323
|
+
if (materialPath) {
|
|
324
|
+
let materialBuffer = null;
|
|
325
|
+
try {
|
|
326
|
+
materialBuffer = readFileSync(materialPath);
|
|
327
|
+
} catch {
|
|
328
|
+
findings.push(f(1, "writing-materials-material-missing", "fail",
|
|
329
|
+
`${matPrefix()}the material file "${matShown}" does not exist or cannot be read`,
|
|
330
|
+
"Fix the material's path, or add the file."));
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
if (materialBuffer) {
|
|
334
|
+
const materialText = materialBuffer.toString("utf8");
|
|
335
|
+
|
|
336
|
+
if (header && str(header.sha256)) {
|
|
337
|
+
const current = sha256(materialBuffer);
|
|
338
|
+
if (current !== str(header.sha256)) {
|
|
339
|
+
findings.push(f(4, "writing-materials-stale", "fail",
|
|
340
|
+
`${matPrefix()}segments file "${segShown}" was marked against a different version of "${matShown}" (sha256 no longer matches)`,
|
|
341
|
+
"Re-run `hyperspec segments init` (or otherwise re-mark) against the current material, and re-label every segment."));
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
const valid = [];
|
|
346
|
+
segments.forEach((obj, i) => {
|
|
347
|
+
const id = str(obj.id) || `#${i + 1}`;
|
|
348
|
+
const { start, end } = obj;
|
|
349
|
+
const shapeOk = Number.isInteger(start) && Number.isInteger(end) && start >= 0 && end > start && end <= materialText.length;
|
|
350
|
+
if (!shapeOk) {
|
|
351
|
+
findings.push(f(1, `writing-materials-segment-shape-${id}`, "fail",
|
|
352
|
+
`${matPrefix(id)}start/end are not valid offsets into the material`,
|
|
353
|
+
"Set start and end to whole-number character offsets into the material, end greater than start and no greater than the material's length."));
|
|
354
|
+
return;
|
|
355
|
+
}
|
|
356
|
+
const text = typeof obj.text === "string" ? obj.text : "";
|
|
357
|
+
const expected = materialText.slice(start, end);
|
|
358
|
+
if (text !== expected) {
|
|
359
|
+
findings.push(f(4, `writing-materials-text-${id}`, "fail",
|
|
360
|
+
`${matPrefix(id)}text does not match the material verbatim at [${start}, ${end})`,
|
|
361
|
+
"Re-derive start/end/text from the material, or re-run hyperspec segments init."));
|
|
362
|
+
}
|
|
363
|
+
valid.push({ id, start, end });
|
|
364
|
+
});
|
|
365
|
+
|
|
366
|
+
valid.sort((a, b) => a.start - b.start);
|
|
367
|
+
for (let i = 1; i < valid.length; i++) {
|
|
368
|
+
if (valid[i].start < valid[i - 1].end) {
|
|
369
|
+
findings.push(f(1, "writing-materials-overlap", "fail",
|
|
370
|
+
`${matPrefix(valid[i].id)}overlaps segment "${valid[i - 1].id}"`,
|
|
371
|
+
"Adjust start/end so segments never overlap."));
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
// A material with no text that is not whitespace has nothing to mark, so it is not a material.
|
|
376
|
+
if (!/\S/.test(materialText)) {
|
|
377
|
+
findings.push(f(1, "writing-materials-empty", "fail",
|
|
378
|
+
`${matPrefix()}the material "${matShown}" has no text to mark`,
|
|
379
|
+
"Put the material's text in the file, or drop the item from materials.items."));
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
let cursor = 0;
|
|
383
|
+
const gaps = [];
|
|
384
|
+
for (const seg of valid) {
|
|
385
|
+
if (seg.start > cursor) gaps.push([cursor, seg.start]);
|
|
386
|
+
cursor = Math.max(cursor, seg.end);
|
|
387
|
+
}
|
|
388
|
+
if (cursor < materialText.length) gaps.push([cursor, materialText.length]);
|
|
389
|
+
const uncovered = gaps.filter(([s, e]) => /\S/.test(materialText.slice(s, e)));
|
|
390
|
+
if (uncovered.length) {
|
|
391
|
+
// Say where: the first uncovered stretch's offset (its first non-whitespace character) and
|
|
392
|
+
// up to 60 characters of it, trimmed, plus how many stretches there are in all.
|
|
393
|
+
const [gs, ge] = uncovered[0];
|
|
394
|
+
const gap = materialText.slice(gs, ge);
|
|
395
|
+
const offset = gs + gap.search(/\S/);
|
|
396
|
+
const trimmed = gap.trim();
|
|
397
|
+
const excerpt = trimmed.length > 60 ? `${trimmed.slice(0, 60)}...` : trimmed;
|
|
398
|
+
const more = uncovered.length > 1 ? ` (${uncovered.length} uncovered stretches in all)` : "";
|
|
399
|
+
findings.push(f(1, "writing-materials-coverage", "fail",
|
|
400
|
+
`${matPrefix()}text at offset ${offset} of "${matShown}" is not covered by any segment: ${JSON.stringify(excerpt)}${more}`,
|
|
401
|
+
"Add a segment for every non-whitespace span, or extend an existing segment's start/end."));
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
return { header, segments, findings };
|
|
407
|
+
}
|
package/src/template.mjs
CHANGED
|
@@ -4,7 +4,10 @@
|
|
|
4
4
|
// which is also a valid YAML double-quoted scalar.
|
|
5
5
|
const PLAIN = /^[A-Za-z0-9][A-Za-z0-9 _.,'()/-]*$/;
|
|
6
6
|
const RESOLVES = /^(null|~|true|false|yes|no|on|off|y|n|[-+]?(\d[\d_]*)?\.?\d+([eE][-+]?\d+)?|0x[0-9a-f]+|0o[0-7]+|\.inf|\.nan)$/i;
|
|
7
|
-
|
|
7
|
+
// Exported so any other init-time template (writing-template.mjs's writingTemplate, and whatever
|
|
8
|
+
// profile templates come after it) quotes titles, kinds and other free-text scalars the same way,
|
|
9
|
+
// rather than re-deriving this regex pair.
|
|
10
|
+
export const scalar = (v) => (PLAIN.test(v) && !/\s$/.test(v) && !RESOLVES.test(v) ? v : JSON.stringify(v));
|
|
8
11
|
|
|
9
12
|
export function template({ title = "Untitled", kind = "document" } = {}) {
|
|
10
13
|
const heading = String(title).replace(/\s+/g, " ").trim();
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
// @supersuit/hyperspec/writing: the reading side of materials marking, for a tool outside this
|
|
2
|
+
// package that labels materials (an agent, an editor, the writing engine's capture stage). It gets
|
|
3
|
+
// the closed label set and the same parse-and-validate the linter runs; labeling itself stays with
|
|
4
|
+
// whoever calls this.
|
|
5
|
+
export { MATERIAL_LABELS } from "./labels.mjs";
|
|
6
|
+
export { readSegments } from "./segments.mjs";
|