@supersuit/hyperspec 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +91 -0
- package/README.md +37 -0
- package/SPEC.md +2 -2
- package/WRITING.md +542 -10
- package/bin/hyperspec.mjs +183 -0
- package/examples/writing/essay/judge/doctor.packet.json +108 -0
- package/examples/writing/essay/judge/lineup.packet.json +64 -0
- package/examples/writing/essay/judge/persona.packet.json +73 -0
- package/examples/writing/essay/judge/reader.packet.json +93 -0
- package/examples/writing/essay/learn/first-draft.md +84 -0
- package/examples/writing/essay/learn/learn.packet.json +106 -0
- package/examples/writing/essay/sample-verdicts/doctor.verdict.json +43 -0
- package/examples/writing/essay/sample-verdicts/learn.verdict.json +30 -0
- package/examples/writing/essay/sample-verdicts/lineup.verdict.json +6 -0
- package/examples/writing/essay/sample-verdicts/persona.verdict.json +4 -0
- package/examples/writing/essay/sample-verdicts/reader.verdict.json +7 -0
- package/examples/writing/essay.hyperspec.md +6 -1
- package/examples/writing/story/judge/attribution.packet.json +194 -0
- package/examples/writing/story/judge/doctor.packet.json +108 -0
- package/examples/writing/story/judge/knowledge.packet.json +77 -0
- package/examples/writing/story/judge/persona.packet.json +73 -0
- package/examples/writing/story/judge/reader.packet.json +94 -0
- package/examples/writing/story/sample-verdicts/attribution.verdict.json +81 -0
- package/examples/writing/story/sample-verdicts/doctor.verdict.json +43 -0
- package/examples/writing/story/sample-verdicts/knowledge.verdict.json +4 -0
- package/examples/writing/story/sample-verdicts/persona.verdict.json +20 -0
- package/examples/writing/story/sample-verdicts/reader.verdict.json +16 -0
- package/examples/writing/story.hyperspec.md +7 -3
- package/package.json +1 -1
- package/src/check.mjs +75 -129
- package/src/draft.mjs +26 -0
- package/src/judge.mjs +386 -0
- package/src/judges/attribution.mjs +360 -0
- package/src/judges/doctor.mjs +126 -0
- package/src/judges/index.mjs +31 -0
- package/src/judges/knowledge.mjs +111 -0
- package/src/judges/lineup.mjs +272 -0
- package/src/judges/persona.mjs +137 -0
- package/src/judges/reader.mjs +111 -0
- package/src/learn.mjs +422 -0
- package/src/ledger.mjs +108 -0
- package/src/sentences.mjs +81 -0
- package/src/stations/claims.mjs +44 -39
- package/src/stations/quotes.mjs +6 -4
- package/src/writing.mjs +1 -1
package/src/learn.mjs
ADDED
|
@@ -0,0 +1,422 @@
|
|
|
1
|
+
// `hyperspec learn prepare` and `hyperspec learn record`: the Learn stage. A factory wrote a first
|
|
2
|
+
// draft; a person edited it into the draft they approved. Every edit is something the spec did not
|
|
3
|
+
// say well enough, or at all. `prepare` diffs the two drafts sentence by sentence and writes a
|
|
4
|
+
// PACKET: each edit as a hunk (E1..En), the spec's block names, fixed instructions and the verdict
|
|
5
|
+
// shape. An outside judge, a person or a model, names for each hunk the one block that should have
|
|
6
|
+
// prevented it. `record` validates that verdict, tallies it by block, names ONE next move for the
|
|
7
|
+
// block with the most edits, and appends a ledger line. hyperspec never calls a model, and learn
|
|
8
|
+
// never edits the spec: the move is a suggestion for whoever maintains it.
|
|
9
|
+
//
|
|
10
|
+
// The judge packets (src/judge.mjs) set the rules followed here: paths kept as given so the packet
|
|
11
|
+
// is byte-identical on rerun, and record never trusts the packet file: it rebuilds the packet from
|
|
12
|
+
// the files on disk and requires the file to be exactly those bytes.
|
|
13
|
+
|
|
14
|
+
import { appendFileSync, existsSync, readFileSync, statSync } from "node:fs";
|
|
15
|
+
import { join, resolve } from "node:path";
|
|
16
|
+
import { sha256 } from "./hash.mjs";
|
|
17
|
+
import { writeFileAtomic } from "./fsutil.mjs";
|
|
18
|
+
import { readDraft } from "./draft.mjs";
|
|
19
|
+
import { loadWritingSpec, lintBlock } from "./check.mjs";
|
|
20
|
+
import { openLedger, ledgerDraftKey } from "./ledger.mjs";
|
|
21
|
+
import { BLOCKS, blockPresent } from "./writing.mjs";
|
|
22
|
+
import { sentenceUnits, normalizeSentence } from "./sentences.mjs";
|
|
23
|
+
import { truncate } from "./stations/util.mjs";
|
|
24
|
+
import { hashMismatchMessage } from "./judge.mjs";
|
|
25
|
+
|
|
26
|
+
export const LEARN_VERSION = "0.1";
|
|
27
|
+
export const PACKET_NAME = "learn.packet.json";
|
|
28
|
+
|
|
29
|
+
// The closed list a verdict names blocks from, in the order every tally breaks ties by: the nine
|
|
30
|
+
// writing blocks in schema order, then "none" (no block of the spec could have prevented the edit).
|
|
31
|
+
export const LEARN_BLOCKS = Object.freeze([...BLOCKS, "none"]);
|
|
32
|
+
|
|
33
|
+
// The one next move per block, suggested for the block with the most edits. "none" has no move.
|
|
34
|
+
export const LEARN_MOVES = Object.freeze({
|
|
35
|
+
materials: "mark or add the material the edit drew on",
|
|
36
|
+
dna: "add a golden or a style rule",
|
|
37
|
+
persona: "tighten the persona's stance, may_assert or will_not_say",
|
|
38
|
+
audience: "extend the audience's knows or terms",
|
|
39
|
+
goal: "tighten a goal condition's fails_when",
|
|
40
|
+
form: "adjust the form block's length or shape",
|
|
41
|
+
spine: "restate the spine's claim or its order",
|
|
42
|
+
sources: "add or cite a source in the claims ledger",
|
|
43
|
+
characters: "extend a character's speech or knowledge",
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
export const LEARN_INSTRUCTIONS = [
|
|
47
|
+
"Each hunk is one edit a person made between the first draft a factory produced (first) and the draft they approved (approved): deleted (only in first), inserted (only in approved) or replaced (first became approved).",
|
|
48
|
+
"For every hunk, name the one block of the spec (read it at the spec path) that, had it said more or said it better, would have had the factory write the approved text in the first place:",
|
|
49
|
+
"materials (the source material the text drew on), dna (the writer's voice: goldens and style rules), persona (who is speaking: stance, what they may assert, what they will not say), audience (the reader: what they know, the terms they use), goal (the change the draft must produce and its conditions), form (length and shape), spine (the claims and their order), sources (the claims ledger), characters (fiction: a character's speech or knowledge),",
|
|
50
|
+
"or none when no block of the spec could have prevented it.",
|
|
51
|
+
"A hunk never crosses a paragraph or a heading; its sentences field counts the sentences it touched.",
|
|
52
|
+
"Name only a block listed in blocks. Answer only in the verdict shape given in verdict_schema: one entry per hunk id, each exactly once, with a why saying what the block should have said, or why no block could have.",
|
|
53
|
+
].join(" ");
|
|
54
|
+
|
|
55
|
+
const present = (v) => typeof v === "string" && v.trim().length > 0;
|
|
56
|
+
const isObject = (v) => Boolean(v) && typeof v === "object" && !Array.isArray(v);
|
|
57
|
+
const plural = (n, word) => `${n} ${word}${n === 1 ? "" : "s"}`;
|
|
58
|
+
|
|
59
|
+
// ---- the diff ----------------------------------------------------------------------------------
|
|
60
|
+
|
|
61
|
+
// The most cells the LCS table may have: past it, prepare refuses with a usage error
|
|
62
|
+
// rather than allocate gigabytes. The common start and end of the two drafts are trimmed first, so
|
|
63
|
+
// only the span that actually differs counts.
|
|
64
|
+
export const MAX_DIFF_CELLS = 10_000_000;
|
|
65
|
+
|
|
66
|
+
// Thrown by diffSentences when the differing span is too large to diff; prepare and record turn it
|
|
67
|
+
// into a usage error (exit 2) naming both sentence counts.
|
|
68
|
+
export class DiffTooLarge extends Error {
|
|
69
|
+
constructor(firstCount, approvedCount) {
|
|
70
|
+
super(`the drafts differ over ${firstCount} sentences of the first draft against ${approvedCount} of the approved draft (after their common start and end), more than learn diffs at once (${MAX_DIFF_CELLS.toLocaleString("en-US")} comparisons); learn from a smaller pair, a chapter or a scene at a time`);
|
|
71
|
+
this.firstCount = firstCount;
|
|
72
|
+
this.approvedCount = approvedCount;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// A boundary between two units: a blank line between paragraphs, or the end of a heading that
|
|
77
|
+
// opens its paragraph. Boundaries are tokens in the diff, equal to one another, so the LCS anchors
|
|
78
|
+
// on them and no hunk crosses one. Only a paragraph's first line is ever a heading
|
|
79
|
+
// (src/sentences.mjs), so a hard wrap that starts a line with "# " is text, not a
|
|
80
|
+
// boundary, and a reflow cannot make one.
|
|
81
|
+
const BOUNDARY = Symbol("boundary");
|
|
82
|
+
|
|
83
|
+
function tokens(units) {
|
|
84
|
+
const out = [];
|
|
85
|
+
let leading = true;
|
|
86
|
+
units.forEach((u, k) => {
|
|
87
|
+
const prev = units[k - 1];
|
|
88
|
+
if (prev) {
|
|
89
|
+
if (u.para !== prev.para) { out.push({ key: BOUNDARY }); leading = true; }
|
|
90
|
+
else if (prev.heading && leading) out.push({ key: BOUNDARY });
|
|
91
|
+
}
|
|
92
|
+
if (!u.heading) leading = false;
|
|
93
|
+
out.push({ key: normalizeSentence(u.text), unit: k });
|
|
94
|
+
});
|
|
95
|
+
return out;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// diffSentences(firstText, approvedText): the edits between two drafts, as hunks in document order,
|
|
99
|
+
// { id: "E1", kind: "deleted" | "inserted" | "replaced", first, approved, sentences }. Units
|
|
100
|
+
// (src/sentences.mjs) are compared with their whitespace normalized and matched by a longest common
|
|
101
|
+
// subsequence, boundaries included; every maximal run of unmatched units between two matches or
|
|
102
|
+
// boundaries is one hunk. A hunk's texts are the drafts' own text from its first unit to its last,
|
|
103
|
+
// as written (null on the side that has none), and `sentences` is the larger of its two unit
|
|
104
|
+
// counts. A run whose two sides read the same once whitespace is normalized is a reflow, not an
|
|
105
|
+
// edit: it gets no hunk and no id. Throws DiffTooLarge past MAX_DIFF_CELLS.
|
|
106
|
+
export function diffSentences(firstText, approvedText) {
|
|
107
|
+
const a = sentenceUnits(firstText);
|
|
108
|
+
const b = sentenceUnits(approvedText);
|
|
109
|
+
const ta = tokens(a);
|
|
110
|
+
const tb = tokens(b);
|
|
111
|
+
// Trim the common start and end: they are matches whatever the LCS says.
|
|
112
|
+
let lo = 0;
|
|
113
|
+
while (lo < ta.length && lo < tb.length && ta[lo].key === tb[lo].key) lo++;
|
|
114
|
+
let hiA = ta.length;
|
|
115
|
+
let hiB = tb.length;
|
|
116
|
+
while (hiA > lo && hiB > lo && ta[hiA - 1].key === tb[hiB - 1].key) { hiA--; hiB--; }
|
|
117
|
+
const n = hiA - lo;
|
|
118
|
+
const m = hiB - lo;
|
|
119
|
+
if ((n + 1) * (m + 1) > MAX_DIFF_CELLS) {
|
|
120
|
+
const count = (t, from, to) => t.slice(from, to).filter((x) => x.key !== BOUNDARY).length;
|
|
121
|
+
throw new DiffTooLarge(count(ta, lo, hiA), count(tb, lo, hiB));
|
|
122
|
+
}
|
|
123
|
+
// lcs[i * (m + 1) + j]: the LCS length of ta[lo + i..hiA) and tb[lo + j..hiB).
|
|
124
|
+
const lcs = new Uint32Array((n + 1) * (m + 1));
|
|
125
|
+
const at = (i, j) => lcs[i * (m + 1) + j];
|
|
126
|
+
for (let i = n - 1; i >= 0; i--) {
|
|
127
|
+
for (let j = m - 1; j >= 0; j--) {
|
|
128
|
+
lcs[i * (m + 1) + j] = ta[lo + i].key === tb[lo + j].key ? at(i + 1, j + 1) + 1 : Math.max(at(i + 1, j), at(i, j + 1));
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
const hunks = [];
|
|
133
|
+
let run = { a: [], b: [] };
|
|
134
|
+
const slice = (units, text, ks) => (ks.length ? text.slice(units[ks[0]].start, units[ks.at(-1)].end) : null);
|
|
135
|
+
const close = () => {
|
|
136
|
+
const first = slice(a, firstText, run.a);
|
|
137
|
+
const approved = slice(b, approvedText, run.b);
|
|
138
|
+
const sentences = Math.max(run.a.length, run.b.length);
|
|
139
|
+
run = { a: [], b: [] };
|
|
140
|
+
if (first === null && approved === null) return;
|
|
141
|
+
if (first !== null && approved !== null && normalizeSentence(first) === normalizeSentence(approved)) return;
|
|
142
|
+
hunks.push({
|
|
143
|
+
id: `E${hunks.length + 1}`,
|
|
144
|
+
kind: first === null ? "inserted" : approved === null ? "deleted" : "replaced",
|
|
145
|
+
first,
|
|
146
|
+
approved,
|
|
147
|
+
sentences,
|
|
148
|
+
});
|
|
149
|
+
};
|
|
150
|
+
let i = 0;
|
|
151
|
+
let j = 0;
|
|
152
|
+
while (i < n || j < m) {
|
|
153
|
+
if (i < n && j < m && ta[lo + i].key === tb[lo + j].key) { close(); i++; j++; continue; }
|
|
154
|
+
if (j >= m || (i < n && at(i + 1, j) >= at(i, j + 1))) {
|
|
155
|
+
const t = ta[lo + i++];
|
|
156
|
+
if (t.key === BOUNDARY) close(); else run.a.push(t.unit);
|
|
157
|
+
} else {
|
|
158
|
+
const t = tb[lo + j++];
|
|
159
|
+
if (t.key === BOUNDARY) close(); else run.b.push(t.unit);
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
close();
|
|
163
|
+
return hunks;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
// ---- the packet --------------------------------------------------------------------------------
|
|
167
|
+
|
|
168
|
+
// The blocks a verdict may name for this spec: each writing block the spec has written, in schema
|
|
169
|
+
// order, then "none". A deferred or absent block (characters outside fiction) is not listed.
|
|
170
|
+
export function specBlocks(spec) {
|
|
171
|
+
const writing = isObject(spec.data?.writing) ? spec.data.writing : {};
|
|
172
|
+
return [...BLOCKS.filter((b) => blockPresent(b, writing[b])), "none"];
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
const packetJson = (packet) => `${JSON.stringify(packet, null, 2)}\n`;
|
|
176
|
+
|
|
177
|
+
// The packet for this spec and these two drafts, built the one way both commands build it.
|
|
178
|
+
function buildPacket(specPathArg, spec, specSha, first, approved) {
|
|
179
|
+
const hunks = diffSentences(first.text, approved.text);
|
|
180
|
+
const blocks = specBlocks(spec);
|
|
181
|
+
const packet = {
|
|
182
|
+
hyperspec_learn: LEARN_VERSION,
|
|
183
|
+
spec: specPathArg,
|
|
184
|
+
spec_sha256: specSha,
|
|
185
|
+
first: first.path,
|
|
186
|
+
first_sha256: first.sha256,
|
|
187
|
+
approved: approved.path,
|
|
188
|
+
approved_sha256: approved.sha256,
|
|
189
|
+
blocks,
|
|
190
|
+
instructions: LEARN_INSTRUCTIONS,
|
|
191
|
+
hunks,
|
|
192
|
+
verdict_schema: {
|
|
193
|
+
type: "object",
|
|
194
|
+
required: ["edits"],
|
|
195
|
+
properties: {
|
|
196
|
+
edits: {
|
|
197
|
+
type: "array",
|
|
198
|
+
description: "one entry per hunk id, each exactly once",
|
|
199
|
+
items: {
|
|
200
|
+
type: "object",
|
|
201
|
+
required: ["id", "block", "why"],
|
|
202
|
+
properties: {
|
|
203
|
+
id: { enum: hunks.map((h) => h.id) },
|
|
204
|
+
block: { enum: blocks },
|
|
205
|
+
why: { type: "string", description: "what the block should have said to prevent this edit, or why no block could have" },
|
|
206
|
+
},
|
|
207
|
+
},
|
|
208
|
+
},
|
|
209
|
+
},
|
|
210
|
+
},
|
|
211
|
+
};
|
|
212
|
+
return { packet, packetBytes: packetJson(packet) };
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// "2 replaced, 1 deleted, 1 inserted": the kinds present, most first, ties in that fixed order.
|
|
216
|
+
function kindCounts(hunks) {
|
|
217
|
+
const order = ["replaced", "deleted", "inserted"];
|
|
218
|
+
return order
|
|
219
|
+
.map((k) => [k, hunks.filter((h) => h.kind === k).length])
|
|
220
|
+
.filter(([, c]) => c > 0)
|
|
221
|
+
.sort((x, y) => y[1] - x[1] || order.indexOf(x[0]) - order.indexOf(y[0]))
|
|
222
|
+
.map(([k, c]) => `${c} ${k}`)
|
|
223
|
+
.join(", ");
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
const NOTHING = "the first draft and the approved draft match";
|
|
227
|
+
|
|
228
|
+
// prepareLearn(specPathArg, { first, approved, out, force }): writes <out>/learn.packet.json.
|
|
229
|
+
export function prepareLearn(specPathArg, { first: firstArg, approved: approvedArg, out: outDirArg, force = false } = {}) {
|
|
230
|
+
if (!present(specPathArg)) return { usage: true, error: "learn prepare needs a spec path" };
|
|
231
|
+
if (!present(firstArg)) return { usage: true, error: "learn prepare needs --first <draft>" };
|
|
232
|
+
if (!present(approvedArg)) return { usage: true, error: "learn prepare needs --approved <draft>" };
|
|
233
|
+
if (!present(outDirArg)) return { usage: true, error: "learn prepare needs --out <dir>" };
|
|
234
|
+
|
|
235
|
+
const loaded = loadWritingSpec(specPathArg, "learn prepare");
|
|
236
|
+
if (loaded.usage) return loaded;
|
|
237
|
+
const { spec } = loaded;
|
|
238
|
+
|
|
239
|
+
let outStat = null;
|
|
240
|
+
try { outStat = statSync(resolve(outDirArg)); } catch { /* reported below */ }
|
|
241
|
+
if (!outStat) return { usage: true, error: `--out folder does not exist: ${outDirArg}; create it first` };
|
|
242
|
+
if (!outStat.isDirectory()) return { usage: true, error: `--out is not a folder: ${outDirArg}` };
|
|
243
|
+
|
|
244
|
+
const blocked = lintBlock(spec, specPathArg);
|
|
245
|
+
if (blocked) return blocked;
|
|
246
|
+
|
|
247
|
+
const first = readDraft(firstArg);
|
|
248
|
+
if (!first) return { usage: true, error: `cannot read the first draft: ${firstArg}` };
|
|
249
|
+
const approved = readDraft(approvedArg);
|
|
250
|
+
if (!approved) return { usage: true, error: `cannot read the approved draft: ${approvedArg}` };
|
|
251
|
+
const specSha = sha256(readFileSync(resolve(specPathArg)));
|
|
252
|
+
|
|
253
|
+
let built;
|
|
254
|
+
try { built = buildPacket(specPathArg, spec, specSha, first, approved); }
|
|
255
|
+
catch (e) { if (e instanceof DiffTooLarge) return { usage: true, error: e.message }; throw e; }
|
|
256
|
+
const { packet, packetBytes } = built;
|
|
257
|
+
const path = join(outDirArg, PACKET_NAME);
|
|
258
|
+
if (existsSync(resolve(path)) && !force) return { usage: true, error: `refusing to overwrite ${path}; pass --force to replace it` };
|
|
259
|
+
writeFileAtomic(resolve(path), packetBytes);
|
|
260
|
+
|
|
261
|
+
const n = packet.hunks.length;
|
|
262
|
+
return {
|
|
263
|
+
ok: true,
|
|
264
|
+
specPath: specPathArg,
|
|
265
|
+
path,
|
|
266
|
+
edits: n,
|
|
267
|
+
kinds: Object.fromEntries(["replaced", "deleted", "inserted"].map((k) => [k, packet.hunks.filter((h) => h.kind === k).length])),
|
|
268
|
+
summary: n ? `${plural(n, "edit")} over ${plural(packet.hunks.reduce((sum, h) => sum + h.sentences, 0), "sentence")}: ${kindCounts(packet.hunks)}` : `no edits: ${NOTHING}; nothing to learn`,
|
|
269
|
+
code: 0,
|
|
270
|
+
};
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
// ---- record ------------------------------------------------------------------------------------
|
|
274
|
+
|
|
275
|
+
const finding = (id, message, fix) => ({ station: "learn", id, severity: "fail", message, fix });
|
|
276
|
+
const shape = (message) => finding("learn-verdict-shape", message, "Write one object: { edits: [{ id, block, why }] }, one entry per hunk id.");
|
|
277
|
+
|
|
278
|
+
// Every problem with the verdict against the rebuilt packet; [] when it is valid.
|
|
279
|
+
function validate(verdict, packet) {
|
|
280
|
+
if (!isObject(verdict)) return [shape("the verdict is not a JSON object")];
|
|
281
|
+
if (!Array.isArray(verdict.edits)) return [shape("edits is missing or not a list")];
|
|
282
|
+
const out = [];
|
|
283
|
+
const ids = packet.hunks.map((h) => h.id);
|
|
284
|
+
const seen = new Set();
|
|
285
|
+
const dup = new Set();
|
|
286
|
+
verdict.edits.forEach((e, i) => {
|
|
287
|
+
if (!isObject(e)) { out.push(shape(`edits[${i}] is not an object`)); return; }
|
|
288
|
+
if (!present(e.id)) { out.push(shape(`edits[${i}].id is missing or not a string`)); return; }
|
|
289
|
+
const id = e.id;
|
|
290
|
+
if (!ids.includes(id)) out.push(finding("learn-edit-unknown", `${id} is not a hunk in the packet (hunks: ${ids.join(", ") || "none"})`, "Classify only the packet's hunks, by their ids."));
|
|
291
|
+
else if (seen.has(id)) { if (!dup.has(id)) out.push(finding("learn-edit-duplicate", `${id} is classified more than once`, "Classify each hunk exactly once.")); dup.add(id); }
|
|
292
|
+
seen.add(id);
|
|
293
|
+
if (typeof e.block !== "string" || !LEARN_BLOCKS.includes(e.block)) {
|
|
294
|
+
out.push(finding("learn-block-unknown", `${id}: block ${JSON.stringify(e.block ?? null)} is not one of ${LEARN_BLOCKS.join(", ")}`, "Name one block from the closed list, or none."));
|
|
295
|
+
} else if (!packet.blocks.includes(e.block)) {
|
|
296
|
+
out.push(finding("learn-block-absent", `${id}: the spec has no ${e.block} block`, `Name a block the spec has (${packet.blocks.join(", ")}).`));
|
|
297
|
+
}
|
|
298
|
+
if (!present(e.why)) out.push(finding("learn-why-missing", `${id}: why is empty or not a string`, "Say what the block should have said to prevent the edit, or why no block could have."));
|
|
299
|
+
});
|
|
300
|
+
for (const id of ids) {
|
|
301
|
+
if (!seen.has(id)) out.push(finding("learn-edit-missing", `${id} is not classified`, "Give every hunk in the packet exactly one entry."));
|
|
302
|
+
}
|
|
303
|
+
return out;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// { block: { edits, sentences } } for the blocks named (sentences: the sum of each hunk's sentence
|
|
307
|
+
// count), most sentences first, then most edits, then LEARN_BLOCKS order (a rewritten
|
|
308
|
+
// paragraph is one edit of many sentences, and weighs as the sentences it rewrote).
|
|
309
|
+
function tallyOf(edits, hunks) {
|
|
310
|
+
const size = new Map(hunks.map((h) => [h.id, h.sentences]));
|
|
311
|
+
const counts = new Map();
|
|
312
|
+
for (const e of edits) {
|
|
313
|
+
const c = counts.get(e.block) ?? { edits: 0, sentences: 0 };
|
|
314
|
+
c.edits += 1;
|
|
315
|
+
c.sentences += size.get(e.id);
|
|
316
|
+
counts.set(e.block, c);
|
|
317
|
+
}
|
|
318
|
+
return Object.fromEntries([...counts].sort((x, y) => y[1].sentences - x[1].sentences || y[1].edits - x[1].edits || LEARN_BLOCKS.indexOf(x[0]) - LEARN_BLOCKS.indexOf(y[0])));
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// "1 edit, 4 sentences": one block's line of the tally.
|
|
322
|
+
export const tallyLine = (c) => `${plural(c.edits, "edit")}, ${plural(c.sentences, "sentence")}`;
|
|
323
|
+
|
|
324
|
+
// recordLearn(packetPathArg, verdictPathArg): usage (exit 2); { invalid } (exit 1, nothing
|
|
325
|
+
// appended; also { stale } when a file changed since prepare); or the recorded result (exit 0).
|
|
326
|
+
export function recordLearn(packetPathArg, verdictPathArg) {
|
|
327
|
+
if (!present(packetPathArg)) return { usage: true, error: "learn record needs a packet path" };
|
|
328
|
+
if (!present(verdictPathArg)) return { usage: true, error: "learn record needs --verdict <file>" };
|
|
329
|
+
|
|
330
|
+
let packetText;
|
|
331
|
+
try { packetText = readFileSync(resolve(packetPathArg), "utf8"); }
|
|
332
|
+
catch { return { usage: true, error: `cannot read packet: ${packetPathArg}` }; }
|
|
333
|
+
let packet;
|
|
334
|
+
try { packet = JSON.parse(packetText); } catch { packet = null; }
|
|
335
|
+
if (!isObject(packet) || packet.hyperspec_learn !== LEARN_VERSION || !present(packet.spec) || !present(packet.first) || !present(packet.approved)) {
|
|
336
|
+
return { usage: true, error: `not a hyperspec learn packet: ${packetPathArg}` };
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
let verdictBuf;
|
|
340
|
+
try { verdictBuf = readFileSync(resolve(verdictPathArg)); }
|
|
341
|
+
catch { return { usage: true, error: `cannot read verdict: ${verdictPathArg}` }; }
|
|
342
|
+
|
|
343
|
+
const loaded = loadWritingSpec(packet.spec, "learn record");
|
|
344
|
+
if (loaded.usage) return { usage: true, error: `the packet's spec: ${loaded.error}` };
|
|
345
|
+
const { spec } = loaded;
|
|
346
|
+
const first = readDraft(packet.first);
|
|
347
|
+
if (!first) return { usage: true, error: `cannot read the packet's first draft: ${packet.first}` };
|
|
348
|
+
const approved = readDraft(packet.approved);
|
|
349
|
+
if (!approved) return { usage: true, error: `cannot read the packet's approved draft: ${packet.approved}` };
|
|
350
|
+
|
|
351
|
+
const base = { packetPath: packetPathArg, verdictPath: verdictPathArg };
|
|
352
|
+
const refuse = (f, extra = {}) => ({ ...base, ok: false, invalid: true, ...extra, findings: [f], code: 1 });
|
|
353
|
+
|
|
354
|
+
// A verdict on bytes other than the ones the packet was built from classifies edits that no
|
|
355
|
+
// longer exist: say which file no longer matches, claiming neither cause.
|
|
356
|
+
const specSha = sha256(readFileSync(resolve(packet.spec)));
|
|
357
|
+
const changed = [
|
|
358
|
+
specSha !== packet.spec_sha256 && "the spec",
|
|
359
|
+
first.sha256 !== packet.first_sha256 && "the first draft",
|
|
360
|
+
approved.sha256 !== packet.approved_sha256 && "the approved draft",
|
|
361
|
+
].filter(Boolean);
|
|
362
|
+
if (changed.length) {
|
|
363
|
+
return refuse(finding("learn-stale", hashMismatchMessage(changed), "Run learn prepare again (with --force) and classify the new packet."), { stale: true });
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
// Never trust the packet file: rebuild it and require these exact bytes, then validate against
|
|
367
|
+
// the rebuilt copy only.
|
|
368
|
+
let built;
|
|
369
|
+
try { built = buildPacket(packet.spec, spec, specSha, first, approved); }
|
|
370
|
+
catch (e) { if (e instanceof DiffTooLarge) return { usage: true, error: e.message }; throw e; }
|
|
371
|
+
const { packet: rebuilt, packetBytes } = built;
|
|
372
|
+
if (packetBytes !== packetText) {
|
|
373
|
+
return refuse(finding("learn-packet-altered", `${packetPathArg} is not the packet learn prepare builds from the spec and drafts on disk`, "Run learn prepare again (with --force) and classify the new packet; never edit a packet."));
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
let verdictJson;
|
|
377
|
+
try { verdictJson = JSON.parse(verdictBuf.toString("utf8").replace(/^/, "")); }
|
|
378
|
+
catch (e) {
|
|
379
|
+
return refuse(finding("learn-verdict-not-json", `the verdict is not valid JSON (${truncate(e instanceof Error ? e.message : String(e), 120)})`, "Write the verdict as one JSON object in the packet's verdict_schema shape."));
|
|
380
|
+
}
|
|
381
|
+
const problems = validate(verdictJson, rebuilt);
|
|
382
|
+
if (problems.length) return { ...base, ok: false, invalid: true, findings: problems, code: 1 };
|
|
383
|
+
|
|
384
|
+
const edits = verdictJson.edits.length;
|
|
385
|
+
const sentences = rebuilt.hunks.reduce((sum, h) => sum + h.sentences, 0);
|
|
386
|
+
const tally = tallyOf(verdictJson.edits, rebuilt.hunks);
|
|
387
|
+
const counts = Object.entries(tally).map(([b, c]) => `${b} ${plural(c.edits, "edit")} (${plural(c.sentences, "sentence")})`).join(", ");
|
|
388
|
+
const top = Object.entries(tally).find(([b]) => b !== "none");
|
|
389
|
+
const suggestion = top ? { block: top[0], edits: top[1].edits, sentences: top[1].sentences, move: LEARN_MOVES[top[0]] } : null;
|
|
390
|
+
// "none" means no block of the spec could have prevented the edit (the packet's instructions),
|
|
391
|
+
// so a verdict of nothing but none leaves the spec nothing to learn.
|
|
392
|
+
const NOTHING_TO_LEARN = "no block could have prevented any edit, so the spec has nothing to learn from this pair";
|
|
393
|
+
const next = suggestion
|
|
394
|
+
? `${suggestion.block}, ${suggestion.sentences} of ${plural(sentences, "sentence")} (${suggestion.edits} of ${plural(edits, "edit")}): ${suggestion.move}`
|
|
395
|
+
: edits ? `none; ${NOTHING_TO_LEARN}` : `none; ${NOTHING}, so there is nothing to learn`;
|
|
396
|
+
// The spec is the one the edits were classified against (a changed spec is stale, above), and
|
|
397
|
+
// learn never edits it: every count naming a block is an edit the spec does not yet prevent.
|
|
398
|
+
const reason = !edits ? `no edits: ${NOTHING}` : `edits by block: ${counts}; ${suggestion ? "not yet applied to the spec" : NOTHING_TO_LEARN}`;
|
|
399
|
+
|
|
400
|
+
let ledgerPath = null;
|
|
401
|
+
let ledgerWarning = null;
|
|
402
|
+
const ledger = openLedger(spec);
|
|
403
|
+
if (ledger?.warning) ledgerWarning = ledger.warning;
|
|
404
|
+
else if (ledger) {
|
|
405
|
+
const line = {
|
|
406
|
+
at: new Date().toISOString(),
|
|
407
|
+
kind: "learn",
|
|
408
|
+
first: ledgerDraftKey(spec.dir, packet.first),
|
|
409
|
+
first_sha256: first.sha256,
|
|
410
|
+
approved: ledgerDraftKey(spec.dir, packet.approved),
|
|
411
|
+
approved_sha256: approved.sha256,
|
|
412
|
+
spec_sha256: specSha,
|
|
413
|
+
verdict: "not-improved",
|
|
414
|
+
reason,
|
|
415
|
+
tally,
|
|
416
|
+
};
|
|
417
|
+
appendFileSync(ledger.abs, `${JSON.stringify(line)}\n`);
|
|
418
|
+
ledgerPath = ledger.decl;
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
return { ...base, ok: true, edits, sentences, tally, suggestion, next, verdict: ledger && !ledger.warning ? "not-improved" : null, reason, ledgerPath, ledgerWarning, code: 0 };
|
|
422
|
+
}
|
package/src/ledger.mjs
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
// The runs ledger, shared by every command that appends a verdict to a spec's improvement.ledger
|
|
2
|
+
// after grading a draft: `hyperspec check` (one line per run of the deterministic stations) and
|
|
3
|
+
// `hyperspec judge record` (one line per recorded judgment). Both follow the same truth rules, so
|
|
4
|
+
// the rules live here once: which file, how the draft is keyed, and which verdict a line earns.
|
|
5
|
+
//
|
|
6
|
+
// compare.mjs, the other ledger writer, grades recipes rather than drafts and keeps its own shape.
|
|
7
|
+
|
|
8
|
+
import { readFileSync } from "node:fs";
|
|
9
|
+
import { relative, resolve, sep } from "node:path";
|
|
10
|
+
import { insideDir } from "./fsutil.mjs";
|
|
11
|
+
|
|
12
|
+
const present = (v) => typeof v === "string" && v.trim().length > 0;
|
|
13
|
+
|
|
14
|
+
// The spec's ledger, ready to append to: { decl, abs, priorText }, where decl is improvement.ledger
|
|
15
|
+
// exactly as the spec wrote it (what a command reports) and priorText is the file's current text
|
|
16
|
+
// ("" when not written yet; the first run creates it). { warning } when the declared path leads
|
|
17
|
+
// outside the spec's folder, and null when the spec declares no ledger at all. Lint's test 9
|
|
18
|
+
// already requires a ledger before any draft is graded; the checks here are defense in depth.
|
|
19
|
+
export function openLedger(spec) {
|
|
20
|
+
const decl = spec.data?.improvement?.ledger;
|
|
21
|
+
if (!present(decl)) return null;
|
|
22
|
+
if (!insideDir(spec.dir, decl)) return { warning: "improvement.ledger escapes the spec's directory; not appended" };
|
|
23
|
+
const abs = resolve(spec.dir, decl);
|
|
24
|
+
let priorText = "";
|
|
25
|
+
try { priorText = readFileSync(abs, "utf8"); } catch { /* not written yet */ }
|
|
26
|
+
return { decl, abs, priorText };
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// Every well-formed line of the given kind already in the ledger, oldest first. A line that is not
|
|
30
|
+
// valid JSON, or not an object of that kind, is skipped: this is a read for verdict history, not a
|
|
31
|
+
// lint pass, and a malformed line is lint test 9's finding to report.
|
|
32
|
+
export function priorLines(text, kind) {
|
|
33
|
+
return (text ?? "")
|
|
34
|
+
.split("\n")
|
|
35
|
+
.filter((l) => l.trim())
|
|
36
|
+
.map((l) => { try { return JSON.parse(l); } catch { return null; } })
|
|
37
|
+
.filter((v) => v && typeof v === "object" && !Array.isArray(v) && v.kind === kind);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// The draft as the ledger records it: relative to the spec's folder, with forward slashes, so
|
|
41
|
+
// "./draft.md", "draft.md" and an absolute path are one history, and no absolute path lands in a
|
|
42
|
+
// ledger that is usually committed.
|
|
43
|
+
export function ledgerDraftKey(specDir, draftPathArg) {
|
|
44
|
+
return relative(resolve(specDir), resolve(draftPathArg)).split(sep).join("/");
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// The verdict a new line earns, compared with `last`, the most recent earlier line this one
|
|
48
|
+
// continues (the caller picks it: check uses the last full check of the same draft, judge the last
|
|
49
|
+
// judgment of the same station and draft). `last` is normalized to { stations, draft_sha256,
|
|
50
|
+
// spec_sha256 }, stations mapping each station name to its status then; statusNow maps each
|
|
51
|
+
// station run now to its status. "Changed" means the draft's bytes or the spec's bytes, unless the
|
|
52
|
+
// caller says more (what, below).
|
|
53
|
+
//
|
|
54
|
+
// none, and every station passes -> one-shot
|
|
55
|
+
// none, and a station fails -> not-improved "failing stations: X"
|
|
56
|
+
// it failed, every station passes now -> improved "stations now pass: X", exactly the
|
|
57
|
+
// stations that failed then and pass now
|
|
58
|
+
// it passed, nothing changed, passing -> not-improved "no change since the last passing <noun>"
|
|
59
|
+
// it passed, something changed, passing -> not-improved "<what> changed; every station still passes"
|
|
60
|
+
// failing now -> not-improved "[<what> changed; |no change since
|
|
61
|
+
// the last <noun>; ]still failing: X" when every
|
|
62
|
+
// failing station also failed then, else
|
|
63
|
+
// "[<what> changed; ]failing stations: X"
|
|
64
|
+
//
|
|
65
|
+
// Returns { verdict, detail }, detail holding either change (improved) or reason (not-improved),
|
|
66
|
+
// or nothing (one-shot). Every reason is literally true of the two lines compared.
|
|
67
|
+
//
|
|
68
|
+
// noun: what a line records, in the reasons that say nothing changed ("no change since the last
|
|
69
|
+
// passing check"); a judge line passes "judgment". Default "check".
|
|
70
|
+
//
|
|
71
|
+
// what: the caller's own account of what changed since `last`, when the draft and spec hashes are
|
|
72
|
+
// not the whole of it (a judge's packet also reads the DNA goldens or the claims ledger); a string
|
|
73
|
+
// such as "draft" or "the claims ledger (claims.jsonl)", or null for nothing. Omitted, it is worked
|
|
74
|
+
// out from the two hashes.
|
|
75
|
+
//
|
|
76
|
+
// unchangedImprovedReason: a caller whose station can flip from fail to pass with nothing changed
|
|
77
|
+
// (a judge answering differently; check cannot, being deterministic over the same files) passes the
|
|
78
|
+
// reason to record instead: "improved" then requires something changed since the failing line, an
|
|
79
|
+
// unchanged flip is not-improved with this reason, and an improved line's change names what
|
|
80
|
+
// changed ("draft changed; stations now pass: doctor").
|
|
81
|
+
export function ledgerVerdict({ last, statusNow, draftSha, specSha, unchangedImprovedReason, noun = "check", what: whatGiven }) {
|
|
82
|
+
const list = (names) => names.join(", ");
|
|
83
|
+
const failing = Object.entries(statusNow).filter(([, st]) => st === "fail").map(([n]) => n);
|
|
84
|
+
const passedNow = failing.length === 0;
|
|
85
|
+
if (!last) {
|
|
86
|
+
return passedNow
|
|
87
|
+
? { verdict: "one-shot", detail: {} }
|
|
88
|
+
: { verdict: "not-improved", detail: { reason: `failing stations: ${list(failing)}` } };
|
|
89
|
+
}
|
|
90
|
+
const failedThen = Object.entries(last.stations ?? {}).filter(([, st]) => st === "fail").map(([n]) => n);
|
|
91
|
+
const draftChanged = last.draft_sha256 !== draftSha;
|
|
92
|
+
const specChanged = last.spec_sha256 !== specSha;
|
|
93
|
+
const what = whatGiven !== undefined ? whatGiven
|
|
94
|
+
: draftChanged && specChanged ? "spec and draft" : specChanged ? "spec" : draftChanged ? "draft" : null;
|
|
95
|
+
|
|
96
|
+
if (passedNow && failedThen.length) {
|
|
97
|
+
const nowPass = failedThen.filter((n) => statusNow[n] === "pass");
|
|
98
|
+
if (nowPass.length && !what && unchangedImprovedReason) return { verdict: "not-improved", detail: { reason: unchangedImprovedReason } };
|
|
99
|
+
if (nowPass.length) return { verdict: "improved", detail: { change: `${unchangedImprovedReason ? `${what} changed; ` : ""}stations now pass: ${list(nowPass)}` } };
|
|
100
|
+
return { verdict: "not-improved", detail: { reason: `${what ? `${what} changed; ` : ""}stations that failed last time now skip: ${list(failedThen)}` } };
|
|
101
|
+
}
|
|
102
|
+
if (passedNow) {
|
|
103
|
+
return { verdict: "not-improved", detail: { reason: what ? `${what} changed; every station still passes` : `no change since the last passing ${noun}` } };
|
|
104
|
+
}
|
|
105
|
+
const still = failing.every((n) => failedThen.includes(n));
|
|
106
|
+
const prefix = what ? `${what} changed; ` : still ? `no change since the last ${noun}; ` : "";
|
|
107
|
+
return { verdict: "not-improved", detail: { reason: `${prefix}${still ? "still failing" : "failing stations"}: ${list(failing)}` } };
|
|
108
|
+
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
// Sentence units: the grain `hyperspec learn` diffs two drafts at. Deterministic, no judgment.
|
|
2
|
+
//
|
|
3
|
+
// A unit is one sentence, one markdown heading line, or one list item. Built from the splitters
|
|
4
|
+
// src/segments.mjs already ships, so "what is a sentence" never disagrees between commands:
|
|
5
|
+
// paragraphs first (a blank line always ends a unit, punctuation or not), then, inside each
|
|
6
|
+
// paragraph, its first line on its own when it is a heading, and the other lines through the sentence
|
|
7
|
+
// splitter (". ! ?" followed by whitespace, never inside a quotation on the same line, each list
|
|
8
|
+
// item on its own). Three rules of its own on top:
|
|
9
|
+
// - curly double quotes count as quotes, the same as straight ones (the shared splitter only
|
|
10
|
+
// knows '"'; each curly mark is one UTF-16 unit, so mapping them keeps every offset);
|
|
11
|
+
// - a unit that ends in a common abbreviation ("Dr.", "e.g.", an initial like "J.") is joined
|
|
12
|
+
// to the next unit of the same run, since that period ends a word, not a sentence;
|
|
13
|
+
// - so is any unit followed by one that starts with a lowercase letter ("the U.S. economy",
|
|
14
|
+
// "9 a.m. in"), since a sentence does not start lowercase.
|
|
15
|
+
|
|
16
|
+
import { splitSegments } from "./segments.mjs";
|
|
17
|
+
|
|
18
|
+
const HEADING = /^ {0,3}#{1,6}(?:[ \t]|$)/;
|
|
19
|
+
|
|
20
|
+
// Lowercased, leading punctuation stripped ("(e.g." reads as "e.g.").
|
|
21
|
+
const ABBREVIATIONS = new Set(["mr.", "mrs.", "ms.", "dr.", "prof.", "sr.", "jr.", "st.", "vs.", "cf.", "e.g.", "i.e."]);
|
|
22
|
+
|
|
23
|
+
function endsInAbbreviation(text) {
|
|
24
|
+
const last = text.trimEnd().split(/\s+/).at(-1) ?? "";
|
|
25
|
+
const word = last.replace(/^[^\p{L}]+/u, "");
|
|
26
|
+
return ABBREVIATIONS.has(word.toLowerCase()) || /^\p{Lu}\.$/u.test(word);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const startsLowercase = (text) => /^\p{Ll}/u.test(text);
|
|
30
|
+
|
|
31
|
+
// The lines of text[start, end) as { start, end } spans, a CRLF line's "\r" outside its span.
|
|
32
|
+
function lines(text, start, end) {
|
|
33
|
+
const out = [];
|
|
34
|
+
let at = start;
|
|
35
|
+
while (at <= end) {
|
|
36
|
+
let nl = text.indexOf("\n", at);
|
|
37
|
+
if (nl < 0 || nl > end) nl = end;
|
|
38
|
+
let e = nl;
|
|
39
|
+
if (e > at && text[e - 1] === "\r") e -= 1;
|
|
40
|
+
out.push({ start: at, end: e });
|
|
41
|
+
at = nl + 1;
|
|
42
|
+
}
|
|
43
|
+
return out;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Sentences of text[start, end), split on `scan` (the text with curly double quotes straightened,
|
|
47
|
+
// same offsets), with abbreviation and lowercase splits joined back.
|
|
48
|
+
function sentencesOf(text, scan, start, end) {
|
|
49
|
+
const units = splitSegments(scan.slice(start, end), { by: "sentence" }).map((s) => ({ start: start + s.start, end: start + s.end }));
|
|
50
|
+
const out = [];
|
|
51
|
+
for (const u of units) {
|
|
52
|
+
const prev = out.at(-1);
|
|
53
|
+
if (prev && (endsInAbbreviation(text.slice(prev.start, prev.end)) || startsLowercase(text.slice(u.start, u.end)))) prev.end = u.end;
|
|
54
|
+
else out.push({ ...u });
|
|
55
|
+
}
|
|
56
|
+
return out;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// sentenceUnits(text): [{ text, start, end, para, heading }] in document order, text exactly
|
|
60
|
+
// text.slice(start, end). para is the 0-based index of the paragraph the unit is in; heading is
|
|
61
|
+
// true for a markdown heading line.
|
|
62
|
+
export function sentenceUnits(text) {
|
|
63
|
+
const scan = text.replace(/[“”]/g, '"');
|
|
64
|
+
const out = [];
|
|
65
|
+
splitSegments(text).forEach((para, p) => {
|
|
66
|
+
let run = null;
|
|
67
|
+
const flush = () => { if (run) out.push(...sentencesOf(text, scan, run.start, run.end).map((u) => ({ ...u, para: p, heading: false }))); run = null; };
|
|
68
|
+
// Only a paragraph's first line can be a heading: a later line that starts with
|
|
69
|
+
// "#" is a hard wrap inside the text and is split into sentences like the rest of it.
|
|
70
|
+
lines(text, para.start, para.end).forEach((line, k) => {
|
|
71
|
+
if (k === 0 && HEADING.test(text.slice(line.start, line.end))) { out.push({ start: line.start, end: line.end, para: p, heading: true }); return; }
|
|
72
|
+
if (run) run.end = line.end; else run = { ...line };
|
|
73
|
+
});
|
|
74
|
+
flush();
|
|
75
|
+
});
|
|
76
|
+
return out.map((u) => ({ text: text.slice(u.start, u.end), start: u.start, end: u.end, para: u.para, heading: u.heading }));
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// The form two units are compared in: every whitespace run one space, trimmed. For comparison
|
|
80
|
+
// only; a hunk carries its units as written.
|
|
81
|
+
export const normalizeSentence = (s) => s.replace(/\s+/g, " ").trim();
|