@supersuit/hyperspec 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +166 -0
- package/README.md +62 -1
- package/SPEC.md +4 -4
- package/WRITING.md +770 -9
- package/bin/hyperspec.mjs +252 -0
- package/examples/writing/essay/claims.jsonl +9 -0
- package/examples/writing/essay/draft.md +82 -0
- package/examples/writing/essay/judge/doctor.packet.json +108 -0
- package/examples/writing/essay/judge/lineup.packet.json +64 -0
- package/examples/writing/essay/judge/persona.packet.json +73 -0
- package/examples/writing/essay/judge/reader.packet.json +93 -0
- package/examples/writing/essay/learn/first-draft.md +84 -0
- package/examples/writing/essay/learn/learn.packet.json +106 -0
- package/examples/writing/essay/materials/interview-notes.md.segments.jsonl +2 -2
- package/examples/writing/essay/sample-verdicts/doctor.verdict.json +43 -0
- package/examples/writing/essay/sample-verdicts/learn.verdict.json +30 -0
- package/examples/writing/essay/sample-verdicts/lineup.verdict.json +6 -0
- package/examples/writing/essay/sample-verdicts/persona.verdict.json +4 -0
- package/examples/writing/essay/sample-verdicts/reader.verdict.json +7 -0
- package/examples/writing/essay.hyperspec.md +23 -7
- package/examples/writing/story/claims.jsonl +9 -0
- package/examples/writing/story/draft.md +267 -0
- package/examples/writing/story/judge/attribution.packet.json +194 -0
- package/examples/writing/story/judge/doctor.packet.json +108 -0
- package/examples/writing/story/judge/knowledge.packet.json +77 -0
- package/examples/writing/story/judge/persona.packet.json +73 -0
- package/examples/writing/story/judge/reader.packet.json +94 -0
- package/examples/writing/story/sample-verdicts/attribution.verdict.json +81 -0
- package/examples/writing/story/sample-verdicts/doctor.verdict.json +43 -0
- package/examples/writing/story/sample-verdicts/knowledge.verdict.json +4 -0
- package/examples/writing/story/sample-verdicts/persona.verdict.json +20 -0
- package/examples/writing/story/sample-verdicts/reader.verdict.json +16 -0
- package/examples/writing/story.hyperspec.md +24 -5
- package/package.json +1 -1
- package/src/check.mjs +191 -0
- package/src/dna.mjs +4 -1
- package/src/draft.mjs +26 -0
- package/src/judge.mjs +386 -0
- package/src/judges/attribution.mjs +360 -0
- package/src/judges/doctor.mjs +126 -0
- package/src/judges/index.mjs +31 -0
- package/src/judges/knowledge.mjs +111 -0
- package/src/judges/lineup.mjs +272 -0
- package/src/judges/persona.mjs +137 -0
- package/src/judges/reader.mjs +111 -0
- package/src/learn.mjs +422 -0
- package/src/ledger.mjs +108 -0
- package/src/sentences.mjs +81 -0
- package/src/stations/claims.mjs +155 -0
- package/src/stations/dna.mjs +126 -0
- package/src/stations/form.mjs +115 -0
- package/src/stations/index.mjs +27 -0
- package/src/stations/links.mjs +275 -0
- package/src/stations/private.mjs +117 -0
- package/src/stations/quotes.mjs +170 -0
- package/src/stations/terms.mjs +131 -0
- package/src/stations/util.mjs +99 -0
- package/src/writing-fields.mjs +26 -2
- package/src/writing.mjs +1 -1
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
// Station "links" (hyperspec 0.6). Checks every Markdown link and
|
|
2
|
+
// every bare http(s):// URL in the draft: an absolute http/https URL must be well-formed (a real
|
|
3
|
+
// host), a mailto: link must carry an address, any other scheme fails outright, a link rooted at
|
|
4
|
+
// "/" is site-root-relative and warns rather than resolving locally (there is
|
|
5
|
+
// no site root here to resolve it against), and any other relative link must resolve to a file
|
|
6
|
+
// that actually exists, relative to the DRAFT's own directory (never the spec's). No network
|
|
7
|
+
// access, ever: well-formed means "the URL parses and names a host", not "the host answers".
|
|
8
|
+
//
|
|
9
|
+
// Three Markdown link forms are checked, all through the same checkUrl:
|
|
10
|
+
// - inline: [text](url)
|
|
11
|
+
// - reference, full or collapsed: [text][ref] / [ref][], resolved against a [ref]: url
|
|
12
|
+
// definition elsewhere in the draft (label matching is case-insensitive and whitespace-
|
|
13
|
+
// collapsed); a full or collapsed reference with no matching definition is its own
|
|
14
|
+
// finding, station-links-undefined-reference, naming the label, since the second bracket
|
|
15
|
+
// pair says a link was meant.
|
|
16
|
+
// - shortcut reference: [ref] alone (no second bracket pair), checked only when a definition
|
|
17
|
+
// for that label exists, once every inline link, reference definition and full/collapsed
|
|
18
|
+
// reference has already been matched and masked out of the text. With no definition it is
|
|
19
|
+
// ordinary text, as CommonMark renders it: [sic], a task-list [x], a footnote-style [1].
|
|
20
|
+
//
|
|
21
|
+
// Anchors: the "#fragment" part of any link is stripped before checking anything else, so
|
|
22
|
+
// "notes.md#section-two" is checked as "notes.md" (its target heading is never verified), and a
|
|
23
|
+
// link that is nothing but "#fragment" (empty path once the anchor is stripped) always resolves,
|
|
24
|
+
// since it points at the draft itself.
|
|
25
|
+
//
|
|
26
|
+
// Fenced code blocks and inline code spans are masked out (src/stations/util.mjs's maskCode)
|
|
27
|
+
// before ANY of the above runs, so a Markdown link or URL shown as illustrative syntax inside a
|
|
28
|
+
// fence or a `` `span` `` is never checked as a real, followable link.
|
|
29
|
+
|
|
30
|
+
import { existsSync } from "node:fs";
|
|
31
|
+
import { dirname, resolve } from "node:path";
|
|
32
|
+
import { lineAt, maskCode, maskRanges, truncate } from "./util.mjs";
|
|
33
|
+
|
|
34
|
+
export const name = "links";
|
|
35
|
+
|
|
36
|
+
// Every "[text](url)" span in `text`, in document order: { start, end, url }. The url is the raw
|
|
37
|
+
// content between the parens, up to the first run of whitespace (a Markdown link may carry a
|
|
38
|
+
// `"title"` after the url, separated by whitespace; that title is not itself a link and is
|
|
39
|
+
// ignored here). `end` is the index just past the closing ")", used both to report a line number
|
|
40
|
+
// and to mask the span out before later scans, so the same URL is never counted twice.
|
|
41
|
+
const MD_LINK_RE = /\[([^\]]*)\]\(([^)]+)\)/g;
|
|
42
|
+
|
|
43
|
+
function markdownLinks(text) {
|
|
44
|
+
const out = [];
|
|
45
|
+
let m;
|
|
46
|
+
MD_LINK_RE.lastIndex = 0;
|
|
47
|
+
while ((m = MD_LINK_RE.exec(text))) {
|
|
48
|
+
const inner = m[2].trim();
|
|
49
|
+
const url = inner.split(/\s+/)[0];
|
|
50
|
+
if (url) out.push({ start: m.index, end: m.index + m[0].length, url });
|
|
51
|
+
}
|
|
52
|
+
return out;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// A reference definition line: up to 3 spaces of indent, "[label]:", the URL, and an optional
|
|
56
|
+
// title in "quotes", 'quotes' or (parens). One definition per line, the common case; a multi-line
|
|
57
|
+
// definition (title on the following line) is not attempted.
|
|
58
|
+
const DEF_RE = /^[ \t]{0,3}\[([^\]]+)\]:[ \t]*(\S+)[ \t]*(?:"[^"]*"|'[^']*'|\([^)]*\))?[ \t]*$/gm;
|
|
59
|
+
|
|
60
|
+
const normalizeLabel = (label) => label.trim().toLowerCase().replace(/\s+/g, " ");
|
|
61
|
+
|
|
62
|
+
// Every reference definition in `text`: a Map from normalized label to { url }, plus the [start,
|
|
63
|
+
// end) span of each whole definition line, ready to mask out (so a definition's own "[label]:" is
|
|
64
|
+
// never later mistaken for a reference USE, and its URL is never also picked up by the bare-URL
|
|
65
|
+
// scan below). The FIRST definition for a given label wins on a duplicate, matching CommonMark.
|
|
66
|
+
function definitions(text) {
|
|
67
|
+
const defs = new Map();
|
|
68
|
+
const spans = [];
|
|
69
|
+
let m;
|
|
70
|
+
DEF_RE.lastIndex = 0;
|
|
71
|
+
while ((m = DEF_RE.exec(text))) {
|
|
72
|
+
const label = normalizeLabel(m[1]);
|
|
73
|
+
if (!defs.has(label)) defs.set(label, { url: m[2] });
|
|
74
|
+
spans.push({ start: m.index, end: m.index + m[0].length });
|
|
75
|
+
}
|
|
76
|
+
return { defs, spans };
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Every "[text][ref]" (full) or "[text][]" (collapsed, label = text) span, in document order:
|
|
80
|
+
// { start, end, label }. Run AFTER inline links and definitions are already masked out of `text`,
|
|
81
|
+
// so this can only match genuine two-bracket-pair reference syntax.
|
|
82
|
+
const FULL_REF_RE = /\[([^\]]*)\]\[([^\]]*)\]/g;
|
|
83
|
+
|
|
84
|
+
function fullReferences(text) {
|
|
85
|
+
const out = [];
|
|
86
|
+
let m;
|
|
87
|
+
FULL_REF_RE.lastIndex = 0;
|
|
88
|
+
while ((m = FULL_REF_RE.exec(text))) {
|
|
89
|
+
const label = (m[2].trim() || m[1].trim());
|
|
90
|
+
if (label) out.push({ start: m.index, end: m.index + m[0].length, label });
|
|
91
|
+
}
|
|
92
|
+
return out;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// Every remaining "[label]" span, in document order: { start, end, label }. Run AFTER inline
|
|
96
|
+
// links, definitions and full/collapsed references are already masked out, so whatever "[...]"
|
|
97
|
+
// is left is either a shortcut reference or ordinary bracketed prose. The caller keeps only the
|
|
98
|
+
// ones whose label has a definition: CommonMark renders an undefined "[label]" as plain text, so
|
|
99
|
+
// an editorial [sic], a task-list [x] or a footnote-style [1] is never a link.
|
|
100
|
+
const SHORTCUT_REF_RE = /\[([^\]]+)\]/g;
|
|
101
|
+
|
|
102
|
+
function shortcutReferences(text) {
|
|
103
|
+
const out = [];
|
|
104
|
+
let m;
|
|
105
|
+
SHORTCUT_REF_RE.lastIndex = 0;
|
|
106
|
+
while ((m = SHORTCUT_REF_RE.exec(text))) {
|
|
107
|
+
const label = m[1].trim();
|
|
108
|
+
if (label) out.push({ start: m.index, end: m.index + m[0].length, label });
|
|
109
|
+
}
|
|
110
|
+
return out;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// Every bare "http://" or "https://" URL left in `text` (after every other link form has been
|
|
114
|
+
// masked out of it), stopping at whitespace or a closing bracket/paren/angle-bracket that is more
|
|
115
|
+
// likely to be surrounding punctuation than part of the URL itself. The host part may be empty,
|
|
116
|
+
// so a bare "https://" is matched and fails as malformed rather than passing unseen.
|
|
117
|
+
const BARE_URL_RE = /https?:\/\/[^\s)>\]]*/g;
|
|
118
|
+
|
|
119
|
+
function bareUrls(text) {
|
|
120
|
+
const out = [];
|
|
121
|
+
let m;
|
|
122
|
+
BARE_URL_RE.lastIndex = 0;
|
|
123
|
+
while ((m = BARE_URL_RE.exec(text))) out.push({ start: m.index, end: m.index + m[0].length, url: m[0] });
|
|
124
|
+
return out;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const SCHEME_RE = /^([a-zA-Z][a-zA-Z0-9+.-]*):/;
|
|
128
|
+
|
|
129
|
+
// Classifies one URL, with its "#..." anchor already known to the caller as stripped: "http" (the
|
|
130
|
+
// scheme itself, lowercased, is on `scheme`), "mailto", "other-scheme" (anything else with a
|
|
131
|
+
// scheme prefix), or "relative" (no scheme prefix at all: a bare path, root-relative or relative).
|
|
132
|
+
function classify(withoutAnchor) {
|
|
133
|
+
const m = SCHEME_RE.exec(withoutAnchor);
|
|
134
|
+
if (!m) return { kind: "relative", path: withoutAnchor };
|
|
135
|
+
const scheme = m[1].toLowerCase();
|
|
136
|
+
if (scheme === "http" || scheme === "https") return { kind: "http", scheme };
|
|
137
|
+
if (scheme === "mailto") return { kind: "mailto" };
|
|
138
|
+
return { kind: "other-scheme", scheme };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// Why `url` is broken (or merely unresolvable locally), as one of "malformed-http",
|
|
142
|
+
// "malformed-mailto", "bad-scheme", "root-relative" or "broken-relative", or null when it is
|
|
143
|
+
// fine. draftDirAbs is the draft's own directory (resolved once by the caller), which every
|
|
144
|
+
// relative link resolves against, never the spec's directory.
|
|
145
|
+
function checkUrl(url, draftDirAbs) {
|
|
146
|
+
const hashIdx = url.indexOf("#");
|
|
147
|
+
const withoutAnchor = hashIdx === -1 ? url : url.slice(0, hashIdx);
|
|
148
|
+
const c = classify(withoutAnchor);
|
|
149
|
+
|
|
150
|
+
if (c.kind === "http") {
|
|
151
|
+
try {
|
|
152
|
+
const u = new URL(url);
|
|
153
|
+
if (!u.hostname) return "malformed-http";
|
|
154
|
+
} catch {
|
|
155
|
+
return "malformed-http";
|
|
156
|
+
}
|
|
157
|
+
return null;
|
|
158
|
+
}
|
|
159
|
+
if (c.kind === "mailto") {
|
|
160
|
+
const address = withoutAnchor.slice("mailto:".length).split("?")[0].trim();
|
|
161
|
+
return address ? null : "malformed-mailto";
|
|
162
|
+
}
|
|
163
|
+
if (c.kind === "other-scheme") return "bad-scheme";
|
|
164
|
+
|
|
165
|
+
// relative: an empty path (the link was nothing but "#fragment", or literally empty) always
|
|
166
|
+
// resolves, since it points at the draft's own file, which exists by construction (check.mjs
|
|
167
|
+
// only ever builds a draft object after successfully reading it).
|
|
168
|
+
if (!c.path) return null;
|
|
169
|
+
// A link rooted at "/" names a path from some site's root, which this station has no way to
|
|
170
|
+
// resolve (there is no "site" here, only the draft's own folder), so it is neither a pass nor a
|
|
171
|
+
// fail -- a warning, naming the fact that it cannot be checked locally.
|
|
172
|
+
if (c.path.startsWith("/")) return "root-relative";
|
|
173
|
+
const target = resolve(draftDirAbs, c.path);
|
|
174
|
+
return existsSync(target) ? null : "broken-relative";
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
const REASON = {
|
|
178
|
+
"malformed-http": {
|
|
179
|
+
id: "station-links-malformed",
|
|
180
|
+
severity: "fail",
|
|
181
|
+
text: (url) => `link "${url}" is not a well-formed http/https URL (no host)`,
|
|
182
|
+
fix: "Fix the URL to include a scheme (http:// or https://) and a real host.",
|
|
183
|
+
},
|
|
184
|
+
"malformed-mailto": {
|
|
185
|
+
id: "station-links-malformed",
|
|
186
|
+
severity: "fail",
|
|
187
|
+
text: (url) => `link "${url}" is a mailto: link with no address`,
|
|
188
|
+
fix: "Add a real address after mailto:, or remove the link.",
|
|
189
|
+
},
|
|
190
|
+
"bad-scheme": {
|
|
191
|
+
id: "station-links-bad-scheme",
|
|
192
|
+
severity: "fail",
|
|
193
|
+
text: (url) => `link "${url}" uses a scheme that is not http, https or mailto`,
|
|
194
|
+
fix: "Use an http(s):// URL, a mailto: link, or a path relative to the draft.",
|
|
195
|
+
},
|
|
196
|
+
"root-relative": {
|
|
197
|
+
id: "station-links-root-relative",
|
|
198
|
+
severity: "warn",
|
|
199
|
+
text: (url) => `link "${url}" is site-root-relative and cannot be resolved against a file on disk`,
|
|
200
|
+
fix: "Point the link at a path relative to the draft, or use a full URL, if it must be checked.",
|
|
201
|
+
},
|
|
202
|
+
"broken-relative": {
|
|
203
|
+
id: "station-links-broken-relative",
|
|
204
|
+
severity: "fail",
|
|
205
|
+
text: (url) => `relative link "${url}" does not resolve to a file next to the draft`,
|
|
206
|
+
fix: "Fix the path, or add the file the link points at.",
|
|
207
|
+
},
|
|
208
|
+
};
|
|
209
|
+
|
|
210
|
+
function reasonFinding(reason, url, line) {
|
|
211
|
+
const r = REASON[reason];
|
|
212
|
+
return { station: name, id: r.id, severity: r.severity, line, message: r.text(truncate(url, 80)), fix: r.fix };
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
function undefinedReferenceFinding(label, line) {
|
|
216
|
+
const shown = truncate(label, 80);
|
|
217
|
+
return {
|
|
218
|
+
station: name,
|
|
219
|
+
id: "station-links-undefined-reference",
|
|
220
|
+
severity: "fail",
|
|
221
|
+
line,
|
|
222
|
+
message: `reference "${shown}" has no matching "[${shown}]: url" definition`,
|
|
223
|
+
fix: `Add a "[${shown}]: <url>" definition, or fix the reference to match a label that already has one.`,
|
|
224
|
+
};
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
export function run(spec, draft) {
|
|
228
|
+
const draftDirAbs = dirname(resolve(draft.path));
|
|
229
|
+
|
|
230
|
+
// Masking pipeline: code first, then each link form in turn, each pass working on the text the
|
|
231
|
+
// previous pass left behind, so nothing is ever matched twice by a later, looser pattern (a
|
|
232
|
+
// reference definition's own brackets are not a shortcut reference; the leftover "[text]" half
|
|
233
|
+
// of a masked-out "[text][ref]" pair is not itself a shortcut reference; a URL already read as
|
|
234
|
+
// part of a Markdown link is not also a bare URL). Every span carries its ORIGINAL start offset
|
|
235
|
+
// throughout, since maskRanges never shifts anything, only blanks it, so draft.text and lineAt
|
|
236
|
+
// stay valid for every one of them regardless of how many passes it survived.
|
|
237
|
+
let working = maskCode(draft.text);
|
|
238
|
+
|
|
239
|
+
const mdLinks = markdownLinks(working);
|
|
240
|
+
working = maskRanges(working, mdLinks);
|
|
241
|
+
|
|
242
|
+
const { defs, spans: defSpans } = definitions(working);
|
|
243
|
+
working = maskRanges(working, defSpans);
|
|
244
|
+
|
|
245
|
+
const fullRefs = fullReferences(working);
|
|
246
|
+
working = maskRanges(working, fullRefs);
|
|
247
|
+
|
|
248
|
+
const shortcutRefs = shortcutReferences(working).filter((r) => defs.has(normalizeLabel(r.label)));
|
|
249
|
+
working = maskRanges(working, shortcutRefs);
|
|
250
|
+
|
|
251
|
+
const bare = bareUrls(working);
|
|
252
|
+
|
|
253
|
+
const entries = [];
|
|
254
|
+
|
|
255
|
+
for (const { start, url } of [...mdLinks, ...bare]) {
|
|
256
|
+
const reason = checkUrl(url, draftDirAbs);
|
|
257
|
+
if (reason) entries.push({ start, finding: reasonFinding(reason, url, lineAt(draft.text, start)) });
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
for (const { start, label } of [...fullRefs, ...shortcutRefs]) {
|
|
261
|
+
const line = lineAt(draft.text, start);
|
|
262
|
+
const def = defs.get(normalizeLabel(label));
|
|
263
|
+
if (!def) {
|
|
264
|
+
entries.push({ start, finding: undefinedReferenceFinding(label, line) });
|
|
265
|
+
continue;
|
|
266
|
+
}
|
|
267
|
+
const reason = checkUrl(def.url, draftDirAbs);
|
|
268
|
+
if (reason) entries.push({ start, finding: reasonFinding(reason, def.url, line) });
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
entries.sort((a, b) => a.start - b.start);
|
|
272
|
+
const findings = entries.map((e) => e.finding);
|
|
273
|
+
const status = findings.some((f) => f.severity === "fail") ? "fail" : "pass";
|
|
274
|
+
return { station: name, status, findings };
|
|
275
|
+
}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
// Station "private" (hyperspec 0.6). No run of 8 or more consecutive words from
|
|
2
|
+
// any `private` segment of a marked material may appear in the draft. Pure and deterministic: a
|
|
3
|
+
// word-level match over normalized text, no model call and no judgment about what would count as
|
|
4
|
+
// a paraphrase (a paraphrase is not caught, by design; a verbatim run is).
|
|
5
|
+
//
|
|
6
|
+
// Normalization, applied the same way to the draft and to every private segment: split into words
|
|
7
|
+
// with dna.mjs's wordsOf (already lowercased, the one word definition every station shares), then
|
|
8
|
+
// drop apostrophes from each word, so case, punctuation and whitespace never decide a match and
|
|
9
|
+
// "don't", "don’t" and "dont" are one word. Fenced code and inline code are masked out of the
|
|
10
|
+
// draft first (util.mjs's maskCode), like every other station that reads prose.
|
|
11
|
+
//
|
|
12
|
+
// A private segment of 8 or more words fails when any 8-word window of it appears in the draft. One
|
|
13
|
+
// of 4 to 7 words is checked as a whole: all of its words, in order, anywhere in the draft. One
|
|
14
|
+
// under 4 words is not checked at all: two or three words ("Yes, Tuesday.") match
|
|
15
|
+
// ordinary prose, so checking them would fail drafts that leak nothing. Skipping silently would
|
|
16
|
+
// hide that a private passage went unchecked, so the station reports how many it skipped as ONE
|
|
17
|
+
// warning, station-private-short-skipped, carrying the count and never the text (the text is
|
|
18
|
+
// the private part). Each leak is reported once, as the longest run the segment and the draft share from where
|
|
19
|
+
// the match starts, so a whole pasted paragraph is one finding rather than one per window; a
|
|
20
|
+
// segment that leaks in two separate places is two findings. Every finding names the material, the
|
|
21
|
+
// segment and the leaked run (normalized words, at most 80 characters), with the draft line where
|
|
22
|
+
// the run starts.
|
|
23
|
+
|
|
24
|
+
import { wordsOf } from "../dna.mjs";
|
|
25
|
+
import { lineAt, maskCode, markedSegments, truncate } from "./util.mjs";
|
|
26
|
+
|
|
27
|
+
export const name = "private";
|
|
28
|
+
|
|
29
|
+
const WINDOW = 8;
|
|
30
|
+
const MIN_CHECKED = 4;
|
|
31
|
+
// dna.mjs's WORD_RE, repeated here only because wordsOf returns words without their offsets and a
|
|
32
|
+
// finding needs the line a leak starts on; the segment side goes through wordsOf itself.
|
|
33
|
+
const WORD_RE = /[\p{L}\p{N}'’]+/gu;
|
|
34
|
+
const stripApostrophes = (w) => w.replace(/['’]/g, "");
|
|
35
|
+
|
|
36
|
+
// The draft's words, normalized, each with the offset it starts at (for the finding's line).
|
|
37
|
+
function draftWords(text) {
|
|
38
|
+
const out = [];
|
|
39
|
+
WORD_RE.lastIndex = 0;
|
|
40
|
+
let m;
|
|
41
|
+
while ((m = WORD_RE.exec(text))) {
|
|
42
|
+
const w = stripApostrophes(m[0].toLowerCase());
|
|
43
|
+
if (w) out.push({ w, at: m.index });
|
|
44
|
+
}
|
|
45
|
+
return out;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const segmentWords = (text) => wordsOf(text).map(stripApostrophes).filter(Boolean);
|
|
49
|
+
|
|
50
|
+
// The first index in `words` (draft word objects) where `seq` (strings) appears contiguously, or -1.
|
|
51
|
+
function findSequence(words, seq) {
|
|
52
|
+
outer: for (let i = 0; i + seq.length <= words.length; i++) {
|
|
53
|
+
for (let k = 0; k < seq.length; k++) if (words[i + k].w !== seq[k]) continue outer;
|
|
54
|
+
return i;
|
|
55
|
+
}
|
|
56
|
+
return -1;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function run(spec, draft, ctx) {
|
|
60
|
+
const words = draftWords(maskCode(draft.text));
|
|
61
|
+
// First draft position of every 8-word window, so each private window is one map lookup.
|
|
62
|
+
const windows = new Map();
|
|
63
|
+
for (let i = 0; i + WINDOW <= words.length; i++) {
|
|
64
|
+
const key = words.slice(i, i + WINDOW).map((x) => x.w).join(" ");
|
|
65
|
+
if (!windows.has(key)) windows.set(key, i);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const findings = [];
|
|
69
|
+
let skippedShort = 0;
|
|
70
|
+
const leak = (material, segId, run, at) => findings.push({
|
|
71
|
+
station: name,
|
|
72
|
+
id: "station-private-leak",
|
|
73
|
+
severity: "fail",
|
|
74
|
+
line: lineAt(draft.text, at),
|
|
75
|
+
message: `material ${material}, segment ${segId} is private, and the draft repeats it: "${truncate(run.join(" "), 80)}"`,
|
|
76
|
+
fix: "Rewrite the passage in words that do not repeat the private material, or relabel the segment if it is not private.",
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
for (const { material, segments } of markedSegments(spec, ctx)) {
|
|
80
|
+
for (const seg of segments) {
|
|
81
|
+
if (seg.label !== "private" || typeof seg.text !== "string") continue;
|
|
82
|
+
const segId = typeof seg.id === "string" && seg.id.trim() ? seg.id.trim() : "(no id)";
|
|
83
|
+
const tokens = segmentWords(seg.text);
|
|
84
|
+
if (!tokens.length) continue;
|
|
85
|
+
if (tokens.length < MIN_CHECKED) { skippedShort++; continue; }
|
|
86
|
+
|
|
87
|
+
if (tokens.length < WINDOW) {
|
|
88
|
+
const at = findSequence(words, tokens);
|
|
89
|
+
if (at !== -1) leak(material, segId, tokens, words[at].at);
|
|
90
|
+
continue;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
let i = 0;
|
|
94
|
+
while (i + WINDOW <= tokens.length) {
|
|
95
|
+
const start = windows.get(tokens.slice(i, i + WINDOW).join(" "));
|
|
96
|
+
if (start === undefined) { i++; continue; }
|
|
97
|
+
// Extend the shared run as far as the segment and the draft keep agreeing.
|
|
98
|
+
let len = WINDOW;
|
|
99
|
+
while (i + len < tokens.length && start + len < words.length && tokens[i + len] === words[start + len].w) len++;
|
|
100
|
+
leak(material, segId, tokens.slice(i, i + len), words[start].at);
|
|
101
|
+
i += len;
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
if (skippedShort) {
|
|
107
|
+
findings.push({
|
|
108
|
+
station: name,
|
|
109
|
+
id: "station-private-short-skipped",
|
|
110
|
+
severity: "warn",
|
|
111
|
+
message: `${skippedShort} private segment${skippedShort === 1 ? " is" : "s are"} under ${MIN_CHECKED} words and ${skippedShort === 1 ? "was" : "were"} not checked`,
|
|
112
|
+
fix: `Check the draft against ${skippedShort === 1 ? "it" : "them"} by hand, or widen the private segment to ${MIN_CHECKED} or more words.`,
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return { station: name, status: findings.some((x) => x.severity === "fail") ? "fail" : "pass", findings };
|
|
117
|
+
}
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
// Station "quotes" (hyperspec 0.6). Every double-quoted span in the draft of 4 or
|
|
2
|
+
// more words must appear in a `quote` or `story` segment of a marked material; a quote the draft
|
|
3
|
+
// attributes to someone by name must appear in a `quote` segment whose speaker is that someone.
|
|
4
|
+
// Pure and deterministic: no model call, no judgment about what counts as quoting, just the closed
|
|
5
|
+
// rules below.
|
|
6
|
+
//
|
|
7
|
+
// What counts as a quoted span: text between a pair of straight double quotes ("...") or a pair of
|
|
8
|
+
// curly ones (U+201C ... U+201D), paired left to right WITHIN one paragraph, so a stray quote mark
|
|
9
|
+
// can never pair with one in a later paragraph and swallow everything between. Fenced code and
|
|
10
|
+
// inline code are masked first (util.mjs's maskCode): a quote inside example syntax is not the
|
|
11
|
+
// writer quoting anyone. A span is checked only when it holds 4 or more words (dna.mjs's wordsOf,
|
|
12
|
+
// the one word definition every station shares); shorter spans are scare quotes and titles as
|
|
13
|
+
// often as they are quotations.
|
|
14
|
+
//
|
|
15
|
+
// Matching: both sides are normalized the same way (curly quotes and apostrophes to straight,
|
|
16
|
+
// whitespace runs to one space, ends trimmed; case is kept, so a changed capital is a changed
|
|
17
|
+
// quote). The span also drops trailing commas and periods before matching, because typographic
|
|
18
|
+
// convention puts a sentence's own comma or period inside the closing quote ("...at a time," he
|
|
19
|
+
// said) whether or not the speaker's sentence ended there. The normalized span must then appear as
|
|
20
|
+
// a substring of some quote or story segment's text, in any marked material of the spec.
|
|
21
|
+
//
|
|
22
|
+
// Attribution: a speaker is any `speaker` value on a quote segment. It is named in the draft
|
|
23
|
+
// when the sentence that holds the quote contains, case-insensitively and as whole
|
|
24
|
+
// words, EITHER the full value (split on anything that is not a letter, digit or apostrophe, so the
|
|
25
|
+
// slug "maria-lopez" reads as "maria lopez", its words joined in the draft by whitespace, hyphens or
|
|
26
|
+
// underscores) OR the value's first word alone ("Maria said" names maria-lopez; "Mariana" does not).
|
|
27
|
+
// The first word alone counts only when it has 2 or more letters and is not one of dna.mjs's
|
|
28
|
+
// STOPWORDS: a speaker recorded as "the manager interviewed" would otherwise be named by every
|
|
29
|
+
// sentence holding "the". Such a speaker is still named by its full value. The sentence is read
|
|
30
|
+
// with the quote itself blanked out (a name inside the quoted words is what was said, not who said it). A named speaker
|
|
31
|
+
// means the span must be in a quote segment with that speaker; matching only some other speaker's
|
|
32
|
+
// quote, or only a story, is `station-quotes-misattributed`. With no speaker named, any quote or
|
|
33
|
+
// story segment is enough. Sentences come from splitSegments(text, { by: "sentence" }), and every
|
|
34
|
+
// sentence the span overlaps is read, so a curly quote the splitter cuts in two still keeps its
|
|
35
|
+
// attribution.
|
|
36
|
+
|
|
37
|
+
import { splitSegments } from "../segments.mjs";
|
|
38
|
+
import { STOPWORDS, wordsOf } from "../dna.mjs";
|
|
39
|
+
import { str } from "../placeholder.mjs";
|
|
40
|
+
import { lineAt, maskCode, markedSegments, truncate } from "./util.mjs";
|
|
41
|
+
|
|
42
|
+
export const name = "quotes";
|
|
43
|
+
|
|
44
|
+
const QUOTE_RE = /"([^"]*)"|“([^“”]*)”/g;
|
|
45
|
+
const WORD_CLASS = "\\p{L}\\p{N}'’";
|
|
46
|
+
|
|
47
|
+
function normalize(text) {
|
|
48
|
+
return String(text)
|
|
49
|
+
.replace(/[‘’‚‛]/g, "'")
|
|
50
|
+
.replace(/[“”„‟]/g, '"')
|
|
51
|
+
.replace(/\s+/g, " ")
|
|
52
|
+
.trim();
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// What a quoted span is matched on: normalized, then trailing commas and periods dropped.
|
|
56
|
+
// Typographic convention puts the writer's own comma or period inside the closing quote
|
|
57
|
+
// ("...at a time," he said) whether or not the speaker's sentence ended there, so keeping them
|
|
58
|
+
// would fail nearly every quotation used mid-sentence. "?" and "!" are kept: adding either changes
|
|
59
|
+
// what was said.
|
|
60
|
+
const matchKey = (inner) => normalize(inner).replace(/[.,]+$/, "").trim();
|
|
61
|
+
|
|
62
|
+
// Every quoted span in `text`, as { start, end, inner, para }: start/end bound the whole span
|
|
63
|
+
// including its quote marks, inner is the text between them, para is { start, end } of the
|
|
64
|
+
// paragraph holding it. Paired per paragraph (see the header). Also how the attribution judge
|
|
65
|
+
// (src/judges/attribution.mjs) finds a draft's dialogue lines.
|
|
66
|
+
export function quotedSpans(text) {
|
|
67
|
+
const spans = [];
|
|
68
|
+
for (const para of splitSegments(text, { by: "paragraph" })) {
|
|
69
|
+
QUOTE_RE.lastIndex = 0;
|
|
70
|
+
let m;
|
|
71
|
+
while ((m = QUOTE_RE.exec(para.text))) {
|
|
72
|
+
spans.push({ start: para.start + m.index, end: para.start + m.index + m[0].length, inner: m[1] ?? m[2] ?? "", para: { start: para.start, end: para.end } });
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return spans;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
const escapeRe = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
79
|
+
|
|
80
|
+
const STOPWORD_SET = new Set(STOPWORDS);
|
|
81
|
+
|
|
82
|
+
// Whether a speaker's first word may name the speaker on its own: 2 or more letters, and not a
|
|
83
|
+
// stopword (compared lowercased, apostrophes as written).
|
|
84
|
+
function firstWordNames(word) {
|
|
85
|
+
const letters = word.match(/\p{L}/gu)?.length ?? 0;
|
|
86
|
+
return letters >= 2 && !STOPWORD_SET.has(word.toLowerCase());
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// A speaker value as a whole-word, case-insensitive pattern matching its full words, or its first
|
|
90
|
+
// word alone when firstWordNames allows it; null when it has no words at all.
|
|
91
|
+
export function speakerPattern(value) {
|
|
92
|
+
const words = value.split(new RegExp(`[^${WORD_CLASS}]+`, "u")).filter(Boolean);
|
|
93
|
+
if (!words.length) return null;
|
|
94
|
+
const full = words.map(escapeRe).join("[\\s_-]+");
|
|
95
|
+
const alts = firstWordNames(words[0]) && words.length > 1 ? `${full}|${escapeRe(words[0])}` : full;
|
|
96
|
+
return new RegExp(`(?<![${WORD_CLASS}])(?:${alts})(?![${WORD_CLASS}])`, "iu");
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// The text of every sentence the span overlaps, with the span itself blanked out.
|
|
100
|
+
function attributionContext(text, sentences, span) {
|
|
101
|
+
const over = sentences.filter((s) => s.start < span.end && s.end > span.start);
|
|
102
|
+
const from = over.length ? Math.min(span.start, over[0].start) : span.start;
|
|
103
|
+
const to = over.length ? Math.max(span.end, over[over.length - 1].end) : span.end;
|
|
104
|
+
return `${text.slice(from, span.start)} ${text.slice(span.end, to)}`;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export function run(spec, draft, ctx) {
|
|
108
|
+
// Fiction: a character's line is invented, not quoted from a material, so holding dialogue to
|
|
109
|
+
// the marked quotes would fail every story that uses quotation marks. The character stations of a
|
|
110
|
+
// later release check dialogue against each character's golden and rejected lines instead.
|
|
111
|
+
if (str(spec?.data?.fiction) === "true") {
|
|
112
|
+
return { station: name, status: "skip", findings: [], reason: "fiction dialogue is checked by the character stations in a later release" };
|
|
113
|
+
}
|
|
114
|
+
const text = maskCode(draft.text);
|
|
115
|
+
const spans = quotedSpans(text).filter((s) => wordsOf(s.inner).length >= 4);
|
|
116
|
+
if (!spans.length) return { station: name, status: "pass", findings: [] };
|
|
117
|
+
|
|
118
|
+
// Every quote and story segment across every marked material, normalized once. A speaker's key
|
|
119
|
+
// is its value lowercased; shownSpeaker keeps the first spelling the material wrote, for findings.
|
|
120
|
+
const sources = [];
|
|
121
|
+
const shownSpeaker = new Map();
|
|
122
|
+
for (const { segments } of markedSegments(spec, ctx)) {
|
|
123
|
+
for (const seg of segments) {
|
|
124
|
+
if ((seg.label !== "quote" && seg.label !== "story") || typeof seg.text !== "string") continue;
|
|
125
|
+
const speaker = seg.label === "quote" && typeof seg.speaker === "string" ? seg.speaker.trim() : "";
|
|
126
|
+
const key = speaker.toLowerCase();
|
|
127
|
+
if (key && !shownSpeaker.has(key)) shownSpeaker.set(key, speaker);
|
|
128
|
+
sources.push({ label: seg.label, speaker: key, norm: normalize(seg.text) });
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
const speakers = [...shownSpeaker.keys()]
|
|
132
|
+
.map((key) => ({ key, re: speakerPattern(key) }))
|
|
133
|
+
.filter((sp) => sp.re);
|
|
134
|
+
|
|
135
|
+
const sentences = splitSegments(text, { by: "sentence" });
|
|
136
|
+
const findings = [];
|
|
137
|
+
for (const span of spans) {
|
|
138
|
+
const key = matchKey(span.inner);
|
|
139
|
+
if (!key) continue;
|
|
140
|
+
const shown = truncate(span.inner.replace(/\s+/g, " "), 80);
|
|
141
|
+
const line = lineAt(draft.text, span.start);
|
|
142
|
+
const hits = sources.filter((s) => s.norm.includes(key));
|
|
143
|
+
if (!hits.length) {
|
|
144
|
+
findings.push({
|
|
145
|
+
station: name,
|
|
146
|
+
id: "station-quotes-unmatched",
|
|
147
|
+
severity: "fail",
|
|
148
|
+
line,
|
|
149
|
+
message: `quoted span "${shown}" does not appear in any quote or story segment of a marked material`,
|
|
150
|
+
fix: "Quote the material verbatim, or drop the quotation marks and paraphrase.",
|
|
151
|
+
});
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
const around = attributionContext(text, sentences, span);
|
|
155
|
+
const named = speakers.filter((sp) => sp.re.test(around)).map((sp) => sp.key);
|
|
156
|
+
if (named.length && !hits.some((h) => h.label === "quote" && named.includes(h.speaker))) {
|
|
157
|
+
const who = named.map((k) => shownSpeaker.get(k) ?? k).join(", ");
|
|
158
|
+
findings.push({
|
|
159
|
+
station: name,
|
|
160
|
+
id: "station-quotes-misattributed",
|
|
161
|
+
severity: "fail",
|
|
162
|
+
line,
|
|
163
|
+
message: `quoted span "${shown}" is attributed to ${who} in its sentence, but no quote segment by ${who} contains it`,
|
|
164
|
+
fix: `Attribute the quote to whoever the material records saying it, or quote what ${who} actually said.`,
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return { station: name, status: findings.length ? "fail" : "pass", findings };
|
|
170
|
+
}
|