@forwardimpact/libinvariant 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +136 -0
- package/package.json +60 -0
- package/src/enum-drift-grammar.js +565 -0
- package/src/enum-drift.js +315 -0
- package/src/index.js +9 -0
- package/src/instructions.js +394 -0
- package/src/invariant-kit.js +487 -0
- package/src/invariants.js +164 -0
- package/src/jtbd.js +517 -0
|
@@ -0,0 +1,565 @@
|
|
|
1
|
+
// The enumeration-drift grammar: source probes (fs-glob, md-table), the
|
|
2
|
+
// list/count value extractors, and the fenced-block consumer parser. These are
|
|
3
|
+
// the reusable mechanics the invariant kit injects as `kit.enumDrift`; they
|
|
4
|
+
// carry no policy. Filesystem access is passed in (`fsSync`) rather than
|
|
5
|
+
// imported, so this module stays clean under the repo's ambient-deps invariant;
|
|
6
|
+
// the orchestration that binds them lives in enum-drift.js.
|
|
7
|
+
|
|
8
|
+
import { isAbsolute, join } from "node:path";
|
|
9
|
+
|
|
10
|
+
export const VALID_PROPERTIES = new Set(["count", "list"]);
|
|
11
|
+
|
|
12
|
+
/** Reject a pattern/file that escapes `root` (absolute or `..`); else null.
|
|
13
|
+
*
|
|
14
|
+
* @param {string} pattern - A repo-relative source pattern or file path.
|
|
15
|
+
* @returns {string|null} An error message, or null when contained.
|
|
16
|
+
*/
|
|
17
|
+
export function checkContainment(pattern) {
|
|
18
|
+
if (typeof pattern !== "string" || pattern === "") {
|
|
19
|
+
return "missing or empty pattern";
|
|
20
|
+
}
|
|
21
|
+
if (isAbsolute(pattern)) return `absolute path not allowed: ${pattern}`;
|
|
22
|
+
if (pattern.split("/").includes("..")) {
|
|
23
|
+
return `path escapes the repo root: ${pattern}`;
|
|
24
|
+
}
|
|
25
|
+
return null;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// --- fs-glob probe ---------------------------------------------------------
|
|
29
|
+
|
|
30
|
+
/** Compile one path segment (e.g. `kata-*`) to a line-anchored, safe regex.
|
|
31
|
+
*
|
|
32
|
+
* @param {string} segment - A single glob path segment.
|
|
33
|
+
* @returns {RegExp} An anchored matcher for one path component.
|
|
34
|
+
*/
|
|
35
|
+
export function segmentToRegExp(segment) {
|
|
36
|
+
const escaped = segment.replace(/[.+^${}()|[\]\\?]/g, "\\$&");
|
|
37
|
+
return new RegExp(`^${escaped.replace(/\*/g, "[^/]+")}$`);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Derive an identifier from a matched path per the registry `id` rule.
|
|
41
|
+
*
|
|
42
|
+
* @param {string} relPath - The matched path, repo-relative.
|
|
43
|
+
* @param {"dirname"|"basename"|"basename-noext"} id - Derivation rule.
|
|
44
|
+
* @returns {string} The derived identifier.
|
|
45
|
+
*/
|
|
46
|
+
export function deriveId(relPath, id) {
|
|
47
|
+
const parts = relPath.split("/");
|
|
48
|
+
const base = parts[parts.length - 1];
|
|
49
|
+
if (id === "dirname") return parts[parts.length - 2] ?? base;
|
|
50
|
+
if (id === "basename-noext") return base.replace(/\.[^.]+$/, "");
|
|
51
|
+
return base;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Split a glob into its fixed (non-glob) prefix segments and its glob tail. */
|
|
55
|
+
function splitGlob(pattern) {
|
|
56
|
+
const segments = pattern.split("/");
|
|
57
|
+
let i = 0;
|
|
58
|
+
while (i < segments.length && !segments[i].includes("*")) i += 1;
|
|
59
|
+
return { fixed: segments.slice(0, i), tail: segments.slice(i) };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Walk `tail`-deep below `dir`, matching each level, collecting ids. */
|
|
63
|
+
function walkGlob(dir, tail, fixed, relParts, id, excludeSet, ids, fsSync) {
|
|
64
|
+
if (relParts.length === tail.length) {
|
|
65
|
+
const base = relParts[relParts.length - 1];
|
|
66
|
+
if (!excludeSet.has(base)) {
|
|
67
|
+
ids.add(deriveId([...fixed, ...relParts].join("/"), id));
|
|
68
|
+
}
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
const matcher = segmentToRegExp(tail[relParts.length]);
|
|
72
|
+
const isLast = relParts.length === tail.length - 1;
|
|
73
|
+
let entries;
|
|
74
|
+
try {
|
|
75
|
+
entries = fsSync.readdirSync(dir);
|
|
76
|
+
} catch {
|
|
77
|
+
return;
|
|
78
|
+
}
|
|
79
|
+
for (const entry of entries) {
|
|
80
|
+
if (!matcher.test(entry)) continue;
|
|
81
|
+
const full = join(dir, entry);
|
|
82
|
+
let isDir = false;
|
|
83
|
+
try {
|
|
84
|
+
isDir = fsSync.statSync(full).isDirectory();
|
|
85
|
+
} catch {
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
if (!isLast && !isDir) continue;
|
|
89
|
+
walkGlob(
|
|
90
|
+
full,
|
|
91
|
+
tail,
|
|
92
|
+
fixed,
|
|
93
|
+
[...relParts, entry],
|
|
94
|
+
id,
|
|
95
|
+
excludeSet,
|
|
96
|
+
ids,
|
|
97
|
+
fsSync,
|
|
98
|
+
);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** Probe a fs-glob source into a Set of identifiers.
|
|
103
|
+
*
|
|
104
|
+
* @param {{ pattern: string, id?: string, exclude?: string[] }} source
|
|
105
|
+
* @param {string} root - Repository root.
|
|
106
|
+
* @param {{ existsSync: Function, readdirSync: Function, statSync: Function }} fsSync
|
|
107
|
+
* @returns {Set<string>}
|
|
108
|
+
*/
|
|
109
|
+
export function probeFsGlob(source, root, fsSync) {
|
|
110
|
+
const { pattern, id = "basename", exclude = [] } = source;
|
|
111
|
+
const { fixed, tail } = splitGlob(pattern);
|
|
112
|
+
const baseDir = join(root, ...fixed);
|
|
113
|
+
const ids = new Set();
|
|
114
|
+
if (!fsSync.existsSync(baseDir)) return ids;
|
|
115
|
+
walkGlob(baseDir, tail, fixed, [], id, new Set(exclude), ids, fsSync);
|
|
116
|
+
return ids;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
// --- md-table probe --------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Reduce a composite-action cell to its bare slug. Unwraps a leading
|
|
123
|
+
* `[text](url)` markdown link to its link text, then drops backticks, the
|
|
124
|
+
* `forwardimpact/` scope, and a trailing `@version`.
|
|
125
|
+
*
|
|
126
|
+
* @param {string} cell - A raw table cell.
|
|
127
|
+
* @returns {string} The bare slug.
|
|
128
|
+
*/
|
|
129
|
+
export function bareSlug(cell) {
|
|
130
|
+
const trimmed = cell.trim();
|
|
131
|
+
const link = trimmed.match(/^\[([^\]]+)\]\([^)]*\)/);
|
|
132
|
+
return (link ? link[1] : trimmed)
|
|
133
|
+
.trim()
|
|
134
|
+
.replace(/^`+|`+$/g, "")
|
|
135
|
+
.trim()
|
|
136
|
+
.replace(/^forwardimpact\//, "")
|
|
137
|
+
.replace(/@[^\s`]+$/, "")
|
|
138
|
+
.trim();
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function escapeRegExp(s) {
|
|
142
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** Split a GFM table row `| a | b |` into cells, or null when not a row.
|
|
146
|
+
*
|
|
147
|
+
* @param {string} line - One source line.
|
|
148
|
+
* @returns {string[]|null} Trimmed cells, or null when the line is not a row.
|
|
149
|
+
*/
|
|
150
|
+
export function parseTableRow(line) {
|
|
151
|
+
const t = line.trim();
|
|
152
|
+
if (!t.startsWith("|")) return null;
|
|
153
|
+
return t
|
|
154
|
+
.replace(/^\|/, "")
|
|
155
|
+
.replace(/\|\s*$/, "")
|
|
156
|
+
.split("|")
|
|
157
|
+
.map((c) => c.trim());
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// The table rows under a `## <section>` heading, up to the next heading,
|
|
161
|
+
// skipping fenced-code regions. Returns parsed cell-arrays (header included).
|
|
162
|
+
function sectionTableRows(lines, section) {
|
|
163
|
+
const headingRe = new RegExp(`^#{1,6}\\s+${escapeRegExp(section)}\\s*$`);
|
|
164
|
+
let i = lines.findIndex((l) => headingRe.test(l));
|
|
165
|
+
if (i < 0) return [];
|
|
166
|
+
const rows = [];
|
|
167
|
+
let inFence = false;
|
|
168
|
+
for (i += 1; i < lines.length; i += 1) {
|
|
169
|
+
const line = lines[i];
|
|
170
|
+
if (/^#{1,6}\s/.test(line)) break;
|
|
171
|
+
if (/^(```|~~~)/.test(line.trim())) {
|
|
172
|
+
inFence = !inFence;
|
|
173
|
+
} else if (!inFence) {
|
|
174
|
+
const cells = parseTableRow(line);
|
|
175
|
+
if (cells) rows.push(cells);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
return rows;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** Probe a md-table source: filtered column cells under a section heading.
|
|
182
|
+
*
|
|
183
|
+
* @param {{ file: string, section: string, column: string, filter: string }} source
|
|
184
|
+
* @param {string} root - Repository root.
|
|
185
|
+
* @param {{ readFileSync: Function }} fsSync
|
|
186
|
+
* @returns {Set<string>}
|
|
187
|
+
*/
|
|
188
|
+
export function probeMdTable(source, root, fsSync) {
|
|
189
|
+
const { file, section, column, filter } = source;
|
|
190
|
+
const lines = fsSync.readFileSync(join(root, file), "utf8").split("\n");
|
|
191
|
+
const rows = sectionTableRows(lines, section);
|
|
192
|
+
if (rows.length === 0) return new Set();
|
|
193
|
+
const filterRe = new RegExp(filter);
|
|
194
|
+
const columnIndex = rows[0].findIndex((h) => h === column);
|
|
195
|
+
const out = new Set();
|
|
196
|
+
if (columnIndex < 0) return out;
|
|
197
|
+
for (const cells of rows.slice(1)) {
|
|
198
|
+
const raw = (cells[columnIndex] ?? "").trim();
|
|
199
|
+
if (raw === "" || /^[-:\s]+$/.test(raw)) continue;
|
|
200
|
+
if (filterRe.test(raw.replace(/^`+|`+$/g, ""))) out.add(bareSlug(raw));
|
|
201
|
+
}
|
|
202
|
+
return out;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** Resolve a topic's source to `{set}` or `{error}`.
|
|
206
|
+
*
|
|
207
|
+
* @param {{ type?: string }} source - A topic `source` descriptor.
|
|
208
|
+
* @param {string} root - Repository root.
|
|
209
|
+
* @param {object} [fsSync] - Sync filesystem surface (unused on error paths).
|
|
210
|
+
* @returns {{ set: Set<string> }|{ error: string }}
|
|
211
|
+
*/
|
|
212
|
+
export function probeSource(source, root, fsSync) {
|
|
213
|
+
if (!source || typeof source.type !== "string") {
|
|
214
|
+
return { error: "source missing `type`" };
|
|
215
|
+
}
|
|
216
|
+
if (source.type !== "fs-glob" && source.type !== "md-table") {
|
|
217
|
+
return { error: `unknown source type \`${source.type}\`` };
|
|
218
|
+
}
|
|
219
|
+
const target = source.type === "md-table" ? source.file : source.pattern;
|
|
220
|
+
const bad = checkContainment(target);
|
|
221
|
+
if (bad) return { error: bad };
|
|
222
|
+
try {
|
|
223
|
+
return {
|
|
224
|
+
set:
|
|
225
|
+
source.type === "fs-glob"
|
|
226
|
+
? probeFsGlob(source, root, fsSync)
|
|
227
|
+
: probeMdTable(source, root, fsSync),
|
|
228
|
+
};
|
|
229
|
+
} catch (err) {
|
|
230
|
+
return { error: `${source.type} probe failed: ${err.message}` };
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// --- value extraction ------------------------------------------------------
|
|
235
|
+
|
|
236
|
+
const WORD_NUMBERS = buildWordNumbers();
|
|
237
|
+
|
|
238
|
+
function buildWordNumbers() {
|
|
239
|
+
const ones = [
|
|
240
|
+
"zero",
|
|
241
|
+
"one",
|
|
242
|
+
"two",
|
|
243
|
+
"three",
|
|
244
|
+
"four",
|
|
245
|
+
"five",
|
|
246
|
+
"six",
|
|
247
|
+
"seven",
|
|
248
|
+
"eight",
|
|
249
|
+
"nine",
|
|
250
|
+
"ten",
|
|
251
|
+
"eleven",
|
|
252
|
+
"twelve",
|
|
253
|
+
"thirteen",
|
|
254
|
+
"fourteen",
|
|
255
|
+
"fifteen",
|
|
256
|
+
"sixteen",
|
|
257
|
+
"seventeen",
|
|
258
|
+
"eighteen",
|
|
259
|
+
"nineteen",
|
|
260
|
+
"twenty",
|
|
261
|
+
];
|
|
262
|
+
const out = {};
|
|
263
|
+
ones.forEach((w, n) => {
|
|
264
|
+
out[w] = n;
|
|
265
|
+
});
|
|
266
|
+
out.thirty = 30;
|
|
267
|
+
out.forty = 40;
|
|
268
|
+
out.fifty = 50;
|
|
269
|
+
return out;
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/** Every count in a span, in source order (digits + English word-numbers).
|
|
273
|
+
*
|
|
274
|
+
* @param {string} span - The text to scan.
|
|
275
|
+
* @returns {number[]} Counts in source order.
|
|
276
|
+
*/
|
|
277
|
+
export function extractCounts(span) {
|
|
278
|
+
const matches = [];
|
|
279
|
+
let m;
|
|
280
|
+
const intRe = /\b\d+\b/g;
|
|
281
|
+
while ((m = intRe.exec(span)) !== null) {
|
|
282
|
+
matches.push({ pos: m.index, value: Number(m[0]) });
|
|
283
|
+
}
|
|
284
|
+
const wordRe = new RegExp(
|
|
285
|
+
`\\b(${Object.keys(WORD_NUMBERS).join("|")})\\b`,
|
|
286
|
+
"gi",
|
|
287
|
+
);
|
|
288
|
+
while ((m = wordRe.exec(span)) !== null) {
|
|
289
|
+
matches.push({ pos: m.index, value: WORD_NUMBERS[m[1].toLowerCase()] });
|
|
290
|
+
}
|
|
291
|
+
return matches.sort((a, b) => a.pos - b.pos).map((x) => x.value);
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/** The first count in a span, or null.
|
|
295
|
+
*
|
|
296
|
+
* @param {string} span - The text to scan.
|
|
297
|
+
* @returns {number|null}
|
|
298
|
+
*/
|
|
299
|
+
export function extractCount(span) {
|
|
300
|
+
const all = extractCounts(span);
|
|
301
|
+
return all.length === 0 ? null : all[0];
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
function firstToken(s) {
|
|
305
|
+
return s.trim().split(/\s+/)[0] ?? "";
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/** Normalize a list token to its comparison form (slug, no slash, lowercase).
|
|
309
|
+
*
|
|
310
|
+
* @param {string} raw - A raw list token.
|
|
311
|
+
* @returns {string} The normalized identifier.
|
|
312
|
+
*/
|
|
313
|
+
export function normalizeToken(raw) {
|
|
314
|
+
let s = (raw ?? "").trim();
|
|
315
|
+
if (s === "") return "";
|
|
316
|
+
const link = s.match(/^\[([^\]]+)\]\([^)]*\)/);
|
|
317
|
+
if (link) s = link[1];
|
|
318
|
+
return s
|
|
319
|
+
.replace(/^\*\*|\*\*$/g, "")
|
|
320
|
+
.replace(/`/g, "")
|
|
321
|
+
.replace(/^forwardimpact\//, "")
|
|
322
|
+
.replace(/@[^\s/]+$/, "")
|
|
323
|
+
.replace(/\/+$/, "")
|
|
324
|
+
.replace(/[.,;:]+$/, "")
|
|
325
|
+
.trim()
|
|
326
|
+
.toLowerCase();
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
function tokensToSet(tokens) {
|
|
330
|
+
const ids = new Set();
|
|
331
|
+
for (const tok of tokens) {
|
|
332
|
+
const id = normalizeToken(tok);
|
|
333
|
+
if (id) ids.add(id);
|
|
334
|
+
}
|
|
335
|
+
return ids;
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
function matchAll(span, re) {
|
|
339
|
+
const out = [];
|
|
340
|
+
let m;
|
|
341
|
+
while ((m = re.exec(span)) !== null) out.push(...m[1].split(","));
|
|
342
|
+
return out;
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
function bulletTokens(lines) {
|
|
346
|
+
const out = [];
|
|
347
|
+
for (const line of lines) {
|
|
348
|
+
const b = line.trim().match(/^[-*+]\s+(.*)$/);
|
|
349
|
+
if (b) out.push(firstToken(b[1]));
|
|
350
|
+
}
|
|
351
|
+
return out;
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
// First non-empty cell of each GFM data row, dropping the header (the row right
|
|
355
|
+
// before the `|---|` alignment row).
|
|
356
|
+
function tableIds(lines) {
|
|
357
|
+
const rows = [];
|
|
358
|
+
let alignmentAt = -1;
|
|
359
|
+
let sawTable = false;
|
|
360
|
+
for (const line of lines) {
|
|
361
|
+
const cells = parseTableRow(line);
|
|
362
|
+
if (!cells) continue;
|
|
363
|
+
sawTable = true;
|
|
364
|
+
if (/^[\s|:-]+$/.test(line.trim()) && line.includes("-")) {
|
|
365
|
+
alignmentAt = rows.length;
|
|
366
|
+
} else {
|
|
367
|
+
rows.push(cells);
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
if (!sawTable) return null;
|
|
371
|
+
const ids = new Set();
|
|
372
|
+
rows.forEach((cells, idx) => {
|
|
373
|
+
if (idx === alignmentAt - 1) return;
|
|
374
|
+
const first = cells.find((c) => c !== "");
|
|
375
|
+
if (first && /[`a-z]/i.test(first)) {
|
|
376
|
+
const id = normalizeToken(first);
|
|
377
|
+
if (id) ids.add(id);
|
|
378
|
+
}
|
|
379
|
+
});
|
|
380
|
+
return ids;
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
// ASCII-tree leaf `name/ desc` — the trailing slash distinguishes a directory
|
|
384
|
+
// leaf from a prose sentence whose first word is capitalized.
|
|
385
|
+
function treeIds(lines) {
|
|
386
|
+
const ids = new Set();
|
|
387
|
+
for (const line of lines) {
|
|
388
|
+
const leaf = line.trim().match(/^([A-Za-z0-9._-]+)\/(\s|$|#|—|-)/);
|
|
389
|
+
if (leaf && /[A-Za-z]/.test(leaf[1])) {
|
|
390
|
+
const id = normalizeToken(leaf[1]);
|
|
391
|
+
if (id) ids.add(id);
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
return ids;
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
// A bare comma/space-separated run of inline code spans and nothing else, e.g.
|
|
398
|
+
// `a`, `b`, `c`. Returns the code-span tokens, or null when the span carries
|
|
399
|
+
// any other text (prose, a bullet marker, an item description). That ambiguity
|
|
400
|
+
// — commas inside a description versus commas between items — is exactly what
|
|
401
|
+
// the bracketed shapes resolve, so this fires only when the whole span is code
|
|
402
|
+
// spans plus separators, and needs at least two so a lone token is not a list.
|
|
403
|
+
function inlineCodeSpanList(span) {
|
|
404
|
+
const t = span.trim();
|
|
405
|
+
const spans = t.match(/`[^`]+`/g);
|
|
406
|
+
if (!spans || spans.length < 2) return null;
|
|
407
|
+
const residue = t.replace(/`[^`]+`/g, "").replace(/[\s,]+/g, "");
|
|
408
|
+
return residue === "" ? spans : null;
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
/**
|
|
412
|
+
* Identifier set from a list-shaped span. Precedence: brace expansion, bullets,
|
|
413
|
+
* a bare comma/space-separated run of code spans, GFM table, ASCII tree, then a
|
|
414
|
+
* parenthetical comma-list (last resort, since a bullet/tree leaf often carries
|
|
415
|
+
* a parenthetical aside whose commas are prose).
|
|
416
|
+
*
|
|
417
|
+
* @param {string} span - The fenced body to read.
|
|
418
|
+
* @returns {Set<string>} The normalized identifier set.
|
|
419
|
+
*/
|
|
420
|
+
export function extractList(span) {
|
|
421
|
+
const lines = span.split("\n");
|
|
422
|
+
const brace = matchAll(span, /\{([^{}]+)\}/g);
|
|
423
|
+
if (brace.length > 0) return tokensToSet(brace);
|
|
424
|
+
const bullets = bulletTokens(lines);
|
|
425
|
+
if (bullets.length > 0) return tokensToSet(bullets);
|
|
426
|
+
const codeSpans = inlineCodeSpanList(span);
|
|
427
|
+
if (codeSpans) return tokensToSet(codeSpans);
|
|
428
|
+
const table = tableIds(lines);
|
|
429
|
+
if (table) return table;
|
|
430
|
+
const tree = treeIds(lines);
|
|
431
|
+
if (tree.size > 0) return tree;
|
|
432
|
+
const paren = matchAll(span, /\(([^()]*,[^()]*)\)/g);
|
|
433
|
+
return tokensToSet(paren);
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
// --- consumer parser -------------------------------------------------------
|
|
437
|
+
|
|
438
|
+
const OPEN_RE = /^\s*<!--\s*(enum:[^>]*?)\s*-->\s*$/;
|
|
439
|
+
const CLOSE_RE = /^\s*<!--\s*\/enum\s*-->\s*$/;
|
|
440
|
+
const CLAIM_RE = /^enum:([a-z0-9-]+):([a-z]+)$/;
|
|
441
|
+
|
|
442
|
+
// Inline single-line fence: open + body + close on one line (counts embedded in
|
|
443
|
+
// prose). A fresh regex per call keeps no global `lastIndex` between lines.
|
|
444
|
+
function inlineFenceRe() {
|
|
445
|
+
return /<!--\s*(enum:[^>]*?)\s*-->(.*?)<!--\s*\/enum\s*-->/g;
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
function splitClaims(raw) {
|
|
449
|
+
return raw
|
|
450
|
+
.trim()
|
|
451
|
+
.split(/\s+/)
|
|
452
|
+
.map((tok) => {
|
|
453
|
+
const m = tok.match(CLAIM_RE);
|
|
454
|
+
return m ? { topic: m[1], property: m[2] } : { bad: tok };
|
|
455
|
+
});
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
function countRecord(topic, value, lineNo) {
|
|
459
|
+
return {
|
|
460
|
+
topic,
|
|
461
|
+
property: "count",
|
|
462
|
+
observed: value,
|
|
463
|
+
lineNo,
|
|
464
|
+
malformed: value === null ? "count span has no number" : undefined,
|
|
465
|
+
};
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
// Expand one fence (raw claim string + body span) into per-claim records.
|
|
469
|
+
function spanRecords(raw, span, lineNo) {
|
|
470
|
+
const counts = extractCounts(span);
|
|
471
|
+
let countIndex = 0;
|
|
472
|
+
const records = [];
|
|
473
|
+
for (const c of splitClaims(raw)) {
|
|
474
|
+
if (c.bad !== undefined) {
|
|
475
|
+
records.push({ lineNo, malformed: `bad claim token \`${c.bad}\`` });
|
|
476
|
+
} else if (!VALID_PROPERTIES.has(c.property)) {
|
|
477
|
+
records.push({
|
|
478
|
+
topic: c.topic,
|
|
479
|
+
property: c.property,
|
|
480
|
+
lineNo,
|
|
481
|
+
malformed: `unknown property \`${c.property}\``,
|
|
482
|
+
});
|
|
483
|
+
} else if (c.property === "count") {
|
|
484
|
+
const value = countIndex < counts.length ? counts[countIndex] : null;
|
|
485
|
+
countIndex += 1;
|
|
486
|
+
records.push(countRecord(c.topic, value, lineNo));
|
|
487
|
+
} else {
|
|
488
|
+
records.push({
|
|
489
|
+
topic: c.topic,
|
|
490
|
+
property: "list",
|
|
491
|
+
observed: extractList(span),
|
|
492
|
+
lineNo,
|
|
493
|
+
});
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
return records;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
const isCodeFence = (line) => /^\s*(```|~~~)/.test(line);
|
|
500
|
+
|
|
501
|
+
// Outside an enum span: toggle top-level code-fence state, emit any inline
|
|
502
|
+
// fences, and open a multi-line span. Returns the new `open` state (or null).
|
|
503
|
+
function scanOutside(line, lineNo, st, records) {
|
|
504
|
+
if (isCodeFence(line)) {
|
|
505
|
+
st.inFence = !st.inFence;
|
|
506
|
+
return null;
|
|
507
|
+
}
|
|
508
|
+
if (st.inFence) return null;
|
|
509
|
+
let sawInline = false;
|
|
510
|
+
for (const im of line.matchAll(inlineFenceRe())) {
|
|
511
|
+
sawInline = true;
|
|
512
|
+
records.push(...spanRecords(im[1], im[2], lineNo));
|
|
513
|
+
}
|
|
514
|
+
const om = sawInline ? null : line.match(OPEN_RE);
|
|
515
|
+
return om ? { raw: om[1], body: [], lineNo } : null;
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
// Inside an enum span: a span may enclose a code block, so a close marker only
|
|
519
|
+
// finalizes when we are not within an enclosed fence. Returns the `open` state
|
|
520
|
+
// (cleared to null when the span closes).
|
|
521
|
+
function scanInside(line, st, open, records) {
|
|
522
|
+
if (isCodeFence(line)) {
|
|
523
|
+
st.enclosed = !st.enclosed;
|
|
524
|
+
open.body.push(line);
|
|
525
|
+
} else if (!st.enclosed && CLOSE_RE.test(line)) {
|
|
526
|
+
records.push(...spanRecords(open.raw, open.body.join("\n"), open.lineNo));
|
|
527
|
+
return null;
|
|
528
|
+
} else {
|
|
529
|
+
open.body.push(line);
|
|
530
|
+
}
|
|
531
|
+
return open;
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
function unclosedRecords(open, records) {
|
|
535
|
+
for (const c of splitClaims(open.raw)) {
|
|
536
|
+
records.push({
|
|
537
|
+
topic: c.topic ?? null,
|
|
538
|
+
property: c.property ?? null,
|
|
539
|
+
lineNo: open.lineNo,
|
|
540
|
+
malformed: "unclosed fence (no <!-- /enum -->)",
|
|
541
|
+
});
|
|
542
|
+
}
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
/**
|
|
546
|
+
* Scan a consumer's text for enum fences outside top-level code blocks (a span
|
|
547
|
+
* may itself enclose a code block). Returns one record per enum:TOPIC:PROPERTY.
|
|
548
|
+
*
|
|
549
|
+
* @param {string} text - The consumer file's text.
|
|
550
|
+
* @returns {Array<object>} Per-claim records.
|
|
551
|
+
*/
|
|
552
|
+
export function parseConsumer(text) {
|
|
553
|
+
const lines = text.split("\n");
|
|
554
|
+
const records = [];
|
|
555
|
+
const st = { inFence: false, enclosed: false };
|
|
556
|
+
let open = null;
|
|
557
|
+
for (let n = 0; n < lines.length; n += 1) {
|
|
558
|
+
open =
|
|
559
|
+
open === null
|
|
560
|
+
? scanOutside(lines[n], n + 1, st, records)
|
|
561
|
+
: scanInside(lines[n], st, open, records);
|
|
562
|
+
}
|
|
563
|
+
if (open !== null) unclosedRecords(open, records);
|
|
564
|
+
return records;
|
|
565
|
+
}
|