rastack 0.0.46 → 0.0.48

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +4 -0
  2. package/dist/compile/entities.d.ts +0 -49
  3. package/dist/compile/entities.js +110 -22
  4. package/dist/compile/index.d.ts +1 -1
  5. package/dist/compile/index.js +1 -1
  6. package/dist/compile/program.js +1 -1
  7. package/dist/define/db.d.ts +1 -1
  8. package/dist/define/db.js +1 -1
  9. package/dist/import/index.d.ts +18 -0
  10. package/dist/import/index.js +57 -0
  11. package/dist/import/pdf.d.ts +25 -0
  12. package/dist/import/pdf.js +326 -0
  13. package/dist/import/tabular.d.ts +76 -0
  14. package/dist/import/tabular.js +98 -0
  15. package/dist/import/xlsx.d.ts +18 -0
  16. package/dist/import/xlsx.js +160 -0
  17. package/dist/rastack-import.d.ts +22 -0
  18. package/dist/rastack-import.js +165 -0
  19. package/dist/rastack.d.ts +1 -0
  20. package/dist/rastack.js +7 -0
  21. package/dist/validate/adapters.d.ts +42 -0
  22. package/dist/validate/adapters.js +109 -0
  23. package/dist/validate/index.d.ts +2 -0
  24. package/dist/validate/index.js +18 -0
  25. package/dist/validate/machine.d.ts +175 -0
  26. package/dist/validate/machine.js +347 -0
  27. package/dist/wasm/rastack_wasm_bg.wasm +0 -0
  28. package/hooks/form/form.ts +16 -5
  29. package/hooks/form/interfaces.ts +15 -0
  30. package/hooks/form/structure.ts +28 -0
  31. package/package.json +1 -1
  32. package/src/compile/entities.ts +161 -59
  33. package/src/compile/index.ts +1 -1
  34. package/src/compile/program.ts +1 -1
  35. package/src/define/db.ts +1 -1
  36. package/src/import/index.ts +41 -0
  37. package/src/import/pdf.ts +304 -0
  38. package/src/import/tabular.ts +159 -0
  39. package/src/import/xlsx.ts +186 -0
  40. package/src/rastack-import.ts +203 -0
  41. package/src/rastack.ts +7 -0
  42. package/src/validate/adapters.ts +118 -0
  43. package/src/validate/index.ts +2 -0
  44. package/src/validate/machine.ts +525 -0
  45. package/test/entities.spec.ts +99 -0
  46. package/test/import.spec.ts +241 -0
  47. package/test/validate.spec.ts +319 -0
  48. package/validate.ts +10 -0
  49. package/wasm/rastack_wasm_bg.wasm +0 -0
@@ -0,0 +1,326 @@
1
+ "use strict";
2
+ /**
3
+ * A tiny, dependency-free PDF *table* extractor for `rastack import`.
4
+ *
5
+ * PDFs carry no table structure — only positioned text — so this is
6
+ * deliberately a best-effort extractor for the common case of a machine-
7
+ * generated tabular report: it decodes every content stream (FlateDecode via
8
+ * node's zlib, plus raw streams), replays the text-positioning operators
9
+ * (`Tm`/`Td`/`TD`/`T*`) to give each shown string (`Tj`/`TJ`/`'`) an (x, y),
10
+ * clusters strings that share a baseline into rows, and orders each row's
11
+ * cells left-to-right. The first row becomes the header.
12
+ *
13
+ * Anything a text-line model can't represent — merged cells, wrapped cell
14
+ * text, scanned images — won't survive; for those, export the source data to
15
+ * CSV/XLSX instead. Every extracted value still goes through the constraint
16
+ * state machine, so a misread cell is *rejected with a located error*, never
17
+ * silently imported.
18
+ */
19
+ Object.defineProperty(exports, "__esModule", { value: true });
20
+ exports.parsePdfTable = parsePdfTable;
21
+ const zlib_1 = require("zlib");
22
+ /** How close two baselines must be (in text-space units) to share a row. */
23
+ const ROW_TOLERANCE = 2;
24
+ // ---------------------------------------------------------------------------
25
+ // Stream extraction
26
+ // ---------------------------------------------------------------------------
27
+ /** Decode every content stream in the document to latin1 text. */
28
+ function contentStreams(buf) {
29
+ const raw = buf.toString("latin1");
30
+ const streams = [];
31
+ const re = /stream\r?\n/g;
32
+ let m;
33
+ while ((m = re.exec(raw)) !== null) {
34
+ const start = m.index + m[0].length;
35
+ const end = raw.indexOf("endstream", start);
36
+ if (end === -1)
37
+ break;
38
+ // The stream dictionary sits just before the `stream` keyword.
39
+ const dictStart = raw.lastIndexOf("<<", m.index);
40
+ const dict = dictStart === -1 ? "" : raw.slice(dictStart, m.index);
41
+ let data = buf.subarray(start, end);
42
+ // Trim the EOL PDF writers place before `endstream`.
43
+ while (data.length &&
44
+ (data[data.length - 1] === 0x0a || data[data.length - 1] === 0x0d)) {
45
+ data = data.subarray(0, data.length - 1);
46
+ }
47
+ if (/\/FlateDecode/.test(dict)) {
48
+ try {
49
+ streams.push((0, zlib_1.inflateSync)(data).toString("latin1"));
50
+ }
51
+ catch {
52
+ // Not a decodable text stream (e.g. an image) — skip it.
53
+ }
54
+ }
55
+ else if (!/\/Filter/.test(dict)) {
56
+ streams.push(data.toString("latin1"));
57
+ }
58
+ re.lastIndex = end;
59
+ }
60
+ return streams;
61
+ }
62
+ // ---------------------------------------------------------------------------
63
+ // Content-stream text replay
64
+ // ---------------------------------------------------------------------------
65
+ /** Unescape a PDF literal string body: `\(`, `\)`, `\\`, `\n`, `\t`, `\ddd`. */
66
+ function unescapePdfString(body) {
67
+ let out = "";
68
+ for (let i = 0; i < body.length; i++) {
69
+ const ch = body[i];
70
+ if (ch !== "\\") {
71
+ out += ch;
72
+ continue;
73
+ }
74
+ const next = body[++i];
75
+ switch (next) {
76
+ case "n":
77
+ out += "\n";
78
+ break;
79
+ case "r":
80
+ out += "\r";
81
+ break;
82
+ case "t":
83
+ out += "\t";
84
+ break;
85
+ case "b":
86
+ out += "\b";
87
+ break;
88
+ case "f":
89
+ out += "\f";
90
+ break;
91
+ case "(":
92
+ out += "(";
93
+ break;
94
+ case ")":
95
+ out += ")";
96
+ break;
97
+ case "\\":
98
+ out += "\\";
99
+ break;
100
+ default: {
101
+ // Octal escape \ddd (1–3 digits)
102
+ if (next >= "0" && next <= "7") {
103
+ let digits = next;
104
+ while (digits.length < 3 &&
105
+ body[i + 1] >= "0" &&
106
+ body[i + 1] <= "7") {
107
+ digits += body[++i];
108
+ }
109
+ out += String.fromCharCode(parseInt(digits, 8));
110
+ }
111
+ else if (next !== undefined) {
112
+ out += next;
113
+ }
114
+ }
115
+ }
116
+ }
117
+ return out;
118
+ }
119
+ /**
120
+ * Tokenise a content stream into PDF operands/operators — enough of the
121
+ * grammar for text extraction: literal strings, numbers, names, arrays and
122
+ * bare operators.
123
+ */
124
+ function tokenize(stream) {
125
+ const tokens = [];
126
+ const stack = [];
127
+ let current = tokens;
128
+ let i = 0;
129
+ while (i < stream.length) {
130
+ const ch = stream[i];
131
+ if (/\s/.test(ch)) {
132
+ i++;
133
+ continue;
134
+ }
135
+ if (ch === "(") {
136
+ // Literal string — respects nesting and escapes.
137
+ let depth = 1;
138
+ let j = i + 1;
139
+ let body = "";
140
+ while (j < stream.length && depth > 0) {
141
+ const c = stream[j];
142
+ if (c === "\\") {
143
+ body += c + (stream[j + 1] ?? "");
144
+ j += 2;
145
+ continue;
146
+ }
147
+ if (c === "(")
148
+ depth++;
149
+ else if (c === ")") {
150
+ depth--;
151
+ if (depth === 0)
152
+ break;
153
+ }
154
+ body += c;
155
+ j++;
156
+ }
157
+ current.push({ str: unescapePdfString(body) });
158
+ i = j + 1;
159
+ }
160
+ else if (ch === "<" && stream[i + 1] !== "<") {
161
+ // Hex string
162
+ const end = stream.indexOf(">", i);
163
+ const hex = stream.slice(i + 1, end === -1 ? stream.length : end).replace(/\s/g, "");
164
+ let text = "";
165
+ for (let k = 0; k + 1 < hex.length; k += 2) {
166
+ text += String.fromCharCode(parseInt(hex.slice(k, k + 2), 16));
167
+ }
168
+ current.push({ str: text });
169
+ i = (end === -1 ? stream.length : end) + 1;
170
+ }
171
+ else if (ch === "[") {
172
+ const arr = [];
173
+ current.push(arr);
174
+ stack.push(current);
175
+ current = arr;
176
+ i++;
177
+ }
178
+ else if (ch === "]") {
179
+ current = stack.pop() ?? tokens;
180
+ i++;
181
+ }
182
+ else if (ch === "<" && stream[i + 1] === "<") {
183
+ // Inline dictionary — skip to the matching >>
184
+ let depth = 1;
185
+ let j = i + 2;
186
+ while (j < stream.length && depth > 0) {
187
+ if (stream.startsWith("<<", j)) {
188
+ depth++;
189
+ j += 2;
190
+ }
191
+ else if (stream.startsWith(">>", j)) {
192
+ depth--;
193
+ j += 2;
194
+ }
195
+ else
196
+ j++;
197
+ }
198
+ i = j;
199
+ }
200
+ else if (/[-+.\d]/.test(ch)) {
201
+ const m = /^[-+]?[\d.]+/.exec(stream.slice(i));
202
+ current.push(parseFloat(m[0]));
203
+ i += m[0].length;
204
+ }
205
+ else if (ch === "/") {
206
+ const m = /^\/[^\s()<>[\]{}/%]*/.exec(stream.slice(i));
207
+ current.push(m[0]);
208
+ i += m[0].length;
209
+ }
210
+ else {
211
+ const m = /^[^\s()<>[\]{}/%]+/.exec(stream.slice(i));
212
+ if (!m) {
213
+ i++;
214
+ continue;
215
+ }
216
+ current.push(m[0]);
217
+ i += m[0].length;
218
+ }
219
+ }
220
+ return tokens;
221
+ }
222
+ /** Replay text operators, collecting each shown string with its position. */
223
+ function extractChunks(stream) {
224
+ const chunks = [];
225
+ const tokens = tokenize(stream);
226
+ let x = 0;
227
+ let y = 0;
228
+ let leading = 0;
229
+ const operands = [];
230
+ const shown = (text) => {
231
+ if (text !== "")
232
+ chunks.push({ x, y, text });
233
+ };
234
+ const num = (v) => (typeof v === "number" ? v : 0);
235
+ for (const token of tokens) {
236
+ if (typeof token !== "string" || token.startsWith("/")) {
237
+ operands.push(token);
238
+ continue;
239
+ }
240
+ switch (token) {
241
+ case "Tm":
242
+ x = num(operands[operands.length - 2]);
243
+ y = num(operands[operands.length - 1]);
244
+ break;
245
+ case "Td":
246
+ x += num(operands[operands.length - 2]);
247
+ y += num(operands[operands.length - 1]);
248
+ break;
249
+ case "TD":
250
+ leading = -num(operands[operands.length - 1]);
251
+ x += num(operands[operands.length - 2]);
252
+ y += num(operands[operands.length - 1]);
253
+ break;
254
+ case "TL":
255
+ leading = num(operands[operands.length - 1]);
256
+ break;
257
+ case "T*":
258
+ y -= leading;
259
+ break;
260
+ case "Tj":
261
+ case "'": {
262
+ if (token === "'")
263
+ y -= leading;
264
+ const s = operands[operands.length - 1];
265
+ if (s && typeof s === "object" && "str" in s)
266
+ shown(s.str);
267
+ break;
268
+ }
269
+ case "TJ": {
270
+ const arr = operands[operands.length - 1];
271
+ if (Array.isArray(arr)) {
272
+ const text = arr
273
+ .filter((el) => Boolean(el && typeof el === "object" && "str" in el))
274
+ .map((el) => el.str)
275
+ .join("");
276
+ shown(text);
277
+ }
278
+ break;
279
+ }
280
+ case "BT":
281
+ x = 0;
282
+ y = 0;
283
+ break;
284
+ }
285
+ operands.length = 0;
286
+ }
287
+ return chunks;
288
+ }
289
+ // ---------------------------------------------------------------------------
290
+ // Row clustering
291
+ // ---------------------------------------------------------------------------
292
+ /**
293
+ * Parse a PDF buffer into headers + rows: text chunks sharing a baseline form
294
+ * a row (top of the page first), ordered left-to-right; the first row is the
295
+ * header. Rows with a different cell count than the header are padded or
296
+ * truncated — the state machine will flag what doesn't validate.
297
+ */
298
+ function parsePdfTable(buf) {
299
+ const chunks = contentStreams(buf).flatMap(extractChunks);
300
+ if (chunks.length === 0)
301
+ return { headers: [], rows: [] };
302
+ // Cluster by baseline (y), newest tolerance-merge wins.
303
+ const lines = [];
304
+ for (const chunk of chunks) {
305
+ const line = lines.find((l) => Math.abs(l.y - chunk.y) <= ROW_TOLERANCE);
306
+ if (line)
307
+ line.chunks.push(chunk);
308
+ else
309
+ lines.push({ y: chunk.y, chunks: [chunk] });
310
+ }
311
+ // Page order: highest y first (PDF's origin is bottom-left).
312
+ lines.sort((a, b) => b.y - a.y);
313
+ const grid = lines.map((line) => line.chunks
314
+ .sort((a, b) => a.x - b.x)
315
+ .map((c) => c.text.trim()));
316
+ const [headers = [], ...rows] = grid;
317
+ return {
318
+ headers,
319
+ rows: rows.map((row) => {
320
+ const cells = row.slice(0, headers.length);
321
+ while (cells.length < headers.length)
322
+ cells.push("");
323
+ return cells;
324
+ }),
325
+ };
326
+ }
@@ -0,0 +1,76 @@
1
+ /**
2
+ * The pure core of `rastack import`: map a tabular source (CSV/XLSX/PDF —
3
+ * already parsed to headers + string rows) onto a resource's fields, then run
4
+ * every row through the resource's **constraint state machine**
5
+ * (`rastack/validate`) with a dataset-level context, so `unique` sees
6
+ * duplicates across the file and `resolved` checks foreign keys against
7
+ * supplied related ids.
8
+ *
9
+ * Everything here is data-in/data-out (no filesystem), tested in
10
+ * `tools/test/import.spec.ts`; the parsers and the CLI are the IO shell.
11
+ */
12
+ import { type FieldSpec, type FieldState, type RecordRun, type SpecRelation } from "../validate";
13
+ /** A parsed tabular source: one header row + string cell rows. */
14
+ export interface TabularData {
15
+ headers: string[];
16
+ rows: string[][];
17
+ }
18
+ /** How one source column mapped onto the resource (or didn't). */
19
+ export interface ColumnMapping {
20
+ header: string;
21
+ field: string | null;
22
+ }
23
+ /** One constraint failure, located by row (0-based, excluding the header). */
24
+ export interface ImportIssue {
25
+ row: number;
26
+ field: string;
27
+ gate: FieldState;
28
+ constraint: string;
29
+ message: string;
30
+ }
31
+ /** The full validation report for one imported file. */
32
+ export interface ImportReport {
33
+ total: number;
34
+ valid: number;
35
+ invalid: number;
36
+ /** Source-column → field mapping (null = column ignored). */
37
+ columns: ColumnMapping[];
38
+ /** Resource fields no source column covered (they'll fail `present` if required). */
39
+ missingFields: string[];
40
+ /** Every constraint failure, in row order. */
41
+ issues: ImportIssue[];
42
+ /** The coerced records of the *valid* rows — ready to POST or seed. */
43
+ records: Record<string, unknown>[];
44
+ /** Per-row machine runs, index-aligned with the source rows. */
45
+ rows: RecordRun[];
46
+ }
47
+ /**
48
+ * Map source headers onto field specs by normalised name (case, `_`, `-` and
49
+ * spaces are insignificant). Unmatched headers are carried as ignored columns;
50
+ * unmatched fields are reported so a missing required column reads as a
51
+ * mapping problem, not just N× `present` failures.
52
+ */
53
+ export declare function mapHeaders(headers: string[], specs: FieldSpec[]): {
54
+ columns: ColumnMapping[];
55
+ missingFields: string[];
56
+ };
57
+ /** Optional dataset-level inputs for FK resolution. */
58
+ export interface ValidateDatasetOptions {
59
+ /** Known ids per `"app.model"` — engages the `resolved` gate. */
60
+ relatedIds?: Record<string, Set<string>>;
61
+ /** Custom FK resolver (wins over `relatedIds`). */
62
+ resolveRelation?: (relation: SpecRelation, value: unknown) => boolean;
63
+ }
64
+ /**
65
+ * Run a whole tabular source through a resource's record machine. Each row is
66
+ * projected onto the mapped fields and validated; the dataset context makes
67
+ * the `unique` gate reject in-file duplicates and (when related ids are
68
+ * given) the `resolved` gate reject dangling foreign keys.
69
+ */
70
+ export declare function validateDataset(specs: FieldSpec[], data: TabularData, options?: ValidateDatasetOptions): ImportReport;
71
+ /**
72
+ * Pull the id set out of an already-parsed related table (its `id` column,
73
+ * else its first column) — the ids `validateDataset` resolves foreign keys
74
+ * against when importing a dependent table alongside its target.
75
+ */
76
+ export declare function idsFromTabular(data: TabularData): Set<string>;
@@ -0,0 +1,98 @@
1
+ "use strict";
2
+ /**
3
+ * The pure core of `rastack import`: map a tabular source (CSV/XLSX/PDF —
4
+ * already parsed to headers + string rows) onto a resource's fields, then run
5
+ * every row through the resource's **constraint state machine**
6
+ * (`rastack/validate`) with a dataset-level context, so `unique` sees
7
+ * duplicates across the file and `resolved` checks foreign keys against
8
+ * supplied related ids.
9
+ *
10
+ * Everything here is data-in/data-out (no filesystem), tested in
11
+ * `tools/test/import.spec.ts`; the parsers and the CLI are the IO shell.
12
+ */
13
+ Object.defineProperty(exports, "__esModule", { value: true });
14
+ exports.mapHeaders = mapHeaders;
15
+ exports.validateDataset = validateDataset;
16
+ exports.idsFromTabular = idsFromTabular;
17
+ const validate_1 = require("../validate");
18
+ /** `"Airport Code"`, `airport_code`, `airportCode` → `airportcode`. */
19
+ function normalise(name) {
20
+ return name.toLowerCase().replace(/[^a-z0-9]/g, "");
21
+ }
22
+ /**
23
+ * Map source headers onto field specs by normalised name (case, `_`, `-` and
24
+ * spaces are insignificant). Unmatched headers are carried as ignored columns;
25
+ * unmatched fields are reported so a missing required column reads as a
26
+ * mapping problem, not just N× `present` failures.
27
+ */
28
+ function mapHeaders(headers, specs) {
29
+ const byNormalised = new Map(specs.map((s) => [normalise(s.name), s.name]));
30
+ const used = new Set();
31
+ const columns = headers.map((header) => {
32
+ const field = byNormalised.get(normalise(header)) ?? null;
33
+ if (field)
34
+ used.add(field);
35
+ return { header, field };
36
+ });
37
+ return {
38
+ columns,
39
+ missingFields: specs.map((s) => s.name).filter((n) => !used.has(n)),
40
+ };
41
+ }
42
+ /**
43
+ * Run a whole tabular source through a resource's record machine. Each row is
44
+ * projected onto the mapped fields and validated; the dataset context makes
45
+ * the `unique` gate reject in-file duplicates and (when related ids are
46
+ * given) the `resolved` gate reject dangling foreign keys.
47
+ */
48
+ function validateDataset(specs, data, options = {}) {
49
+ const { columns, missingFields } = mapHeaders(data.headers, specs);
50
+ const machine = (0, validate_1.compileRecordMachine)(specs);
51
+ const ctx = (0, validate_1.createDatasetContext)(machine, options);
52
+ const rows = [];
53
+ const issues = [];
54
+ const records = [];
55
+ data.rows.forEach((cells, index) => {
56
+ const record = {};
57
+ columns.forEach((col, i) => {
58
+ if (col.field !== null)
59
+ record[col.field] = cells[i] ?? "";
60
+ });
61
+ const run = (0, validate_1.runRecordMachine)(machine, record, ctx);
62
+ rows.push(run);
63
+ if (run.state === "valid") {
64
+ records.push(run.values);
65
+ }
66
+ else {
67
+ for (const error of run.errors)
68
+ issues.push({ row: index, ...error });
69
+ }
70
+ });
71
+ return {
72
+ total: data.rows.length,
73
+ valid: records.length,
74
+ invalid: data.rows.length - records.length,
75
+ columns,
76
+ missingFields,
77
+ issues,
78
+ records,
79
+ rows,
80
+ };
81
+ }
82
+ /**
83
+ * Pull the id set out of an already-parsed related table (its `id` column,
84
+ * else its first column) — the ids `validateDataset` resolves foreign keys
85
+ * against when importing a dependent table alongside its target.
86
+ */
87
+ function idsFromTabular(data) {
88
+ let idColumn = data.headers.findIndex((h) => normalise(h) === "id");
89
+ if (idColumn === -1)
90
+ idColumn = 0;
91
+ const ids = new Set();
92
+ for (const row of data.rows) {
93
+ const value = (row[idColumn] ?? "").trim();
94
+ if (value !== "")
95
+ ids.add(value);
96
+ }
97
+ return ids;
98
+ }
@@ -0,0 +1,18 @@
1
+ /**
2
+ * A tiny, dependency-free XLSX reader for `rastack import`.
3
+ *
4
+ * Why hand-roll it: the CLI stays dependency-light (the same reasoning as the
5
+ * hand-rolled ZIP *writer* in `deploy/zip.ts`), and an `.xlsx` workbook is
6
+ * just a ZIP of small XML parts — the reader below understands exactly the
7
+ * subset a tabular import needs: the ZIP central directory (STORE and DEFLATE
8
+ * entries, inflated via node's zlib), `xl/sharedStrings.xml`, and the cells of
9
+ * the first worksheet. Formulas are read through their cached `<v>` results;
10
+ * merged cells, styles and charts are ignored — an import wants values.
11
+ */
12
+ import type { TabularData } from "./tabular";
13
+ /**
14
+ * Parse an `.xlsx` workbook buffer into headers + rows (first worksheet, first
15
+ * row as the header). Cell values come back as strings — the constraint state
16
+ * machine's `typed` gate owns coercion, same as for CSV.
17
+ */
18
+ export declare function parseXlsx(buf: Buffer): TabularData;
@@ -0,0 +1,160 @@
1
+ "use strict";
2
+ /**
3
+ * A tiny, dependency-free XLSX reader for `rastack import`.
4
+ *
5
+ * Why hand-roll it: the CLI stays dependency-light (the same reasoning as the
6
+ * hand-rolled ZIP *writer* in `deploy/zip.ts`), and an `.xlsx` workbook is
7
+ * just a ZIP of small XML parts — the reader below understands exactly the
8
+ * subset a tabular import needs: the ZIP central directory (STORE and DEFLATE
9
+ * entries, inflated via node's zlib), `xl/sharedStrings.xml`, and the cells of
10
+ * the first worksheet. Formulas are read through their cached `<v>` results;
11
+ * merged cells, styles and charts are ignored — an import wants values.
12
+ */
13
+ Object.defineProperty(exports, "__esModule", { value: true });
14
+ exports.parseXlsx = parseXlsx;
15
+ const zlib_1 = require("zlib");
16
+ /** Locate the end-of-central-directory record (scan back over the comment). */
17
+ function findEocd(buf) {
18
+ for (let i = buf.length - 22; i >= 0; i--) {
19
+ if (buf.readUInt32LE(i) === 0x06054b50)
20
+ return i;
21
+ }
22
+ throw new Error("not a ZIP archive (no end-of-central-directory record)");
23
+ }
24
+ function readCentralDirectory(buf) {
25
+ const eocd = findEocd(buf);
26
+ const count = buf.readUInt16LE(eocd + 10);
27
+ let offset = buf.readUInt32LE(eocd + 16);
28
+ const entries = [];
29
+ for (let i = 0; i < count; i++) {
30
+ if (buf.readUInt32LE(offset) !== 0x02014b50) {
31
+ throw new Error("corrupt ZIP central directory");
32
+ }
33
+ const method = buf.readUInt16LE(offset + 10);
34
+ const compressedSize = buf.readUInt32LE(offset + 20);
35
+ const nameLength = buf.readUInt16LE(offset + 28);
36
+ const extraLength = buf.readUInt16LE(offset + 30);
37
+ const commentLength = buf.readUInt16LE(offset + 32);
38
+ const localHeaderOffset = buf.readUInt32LE(offset + 42);
39
+ const name = buf
40
+ .subarray(offset + 46, offset + 46 + nameLength)
41
+ .toString("utf-8");
42
+ entries.push({ name, method, compressedSize, localHeaderOffset });
43
+ offset += 46 + nameLength + extraLength + commentLength;
44
+ }
45
+ return entries;
46
+ }
47
+ function readEntry(buf, entry) {
48
+ const at = entry.localHeaderOffset;
49
+ if (buf.readUInt32LE(at) !== 0x04034b50) {
50
+ throw new Error(`corrupt ZIP local header for ${entry.name}`);
51
+ }
52
+ const nameLength = buf.readUInt16LE(at + 26);
53
+ const extraLength = buf.readUInt16LE(at + 28);
54
+ const start = at + 30 + nameLength + extraLength;
55
+ const data = buf.subarray(start, start + entry.compressedSize);
56
+ if (entry.method === 0)
57
+ return Buffer.from(data); // STORE
58
+ if (entry.method === 8)
59
+ return (0, zlib_1.inflateRawSync)(data); // DEFLATE
60
+ throw new Error(`unsupported ZIP compression method ${entry.method}`);
61
+ }
62
+ // ---------------------------------------------------------------------------
63
+ // Minimal XML helpers (the sheet parts are machine-written, flat XML)
64
+ // ---------------------------------------------------------------------------
65
+ function unescapeXml(text) {
66
+ return text
67
+ .replace(/&lt;/g, "<")
68
+ .replace(/&gt;/g, ">")
69
+ .replace(/&quot;/g, '"')
70
+ .replace(/&apos;/g, "'")
71
+ .replace(/&#(\d+);/g, (_, code) => String.fromCharCode(Number(code)))
72
+ .replace(/&amp;/g, "&");
73
+ }
74
+ /** All matches of `<tag ...>inner</tag>` (non-nested — true of sheet XML). */
75
+ function elements(xml, tag) {
76
+ const re = new RegExp(`<${tag}(?:\\s[^>]*)?>([\\s\\S]*?)</${tag}>`, "g");
77
+ const out = [];
78
+ let m;
79
+ while ((m = re.exec(xml)) !== null)
80
+ out.push(m[1]);
81
+ return out;
82
+ }
83
+ /** `"BC7"` → 0-based column index (54). */
84
+ function columnIndex(cellRef) {
85
+ let index = 0;
86
+ for (const ch of cellRef) {
87
+ if (ch < "A" || ch > "Z")
88
+ break;
89
+ index = index * 26 + (ch.charCodeAt(0) - 64);
90
+ }
91
+ return index - 1;
92
+ }
93
+ // ---------------------------------------------------------------------------
94
+ // Workbook parsing
95
+ // ---------------------------------------------------------------------------
96
+ function parseSharedStrings(xml) {
97
+ if (!xml)
98
+ return [];
99
+ // Each <si> may hold one <t> or several rich-text runs — concatenate them.
100
+ return elements(xml, "si").map((si) => elements(si, "t").map(unescapeXml).join(""));
101
+ }
102
+ /**
103
+ * Parse an `.xlsx` workbook buffer into headers + rows (first worksheet, first
104
+ * row as the header). Cell values come back as strings — the constraint state
105
+ * machine's `typed` gate owns coercion, same as for CSV.
106
+ */
107
+ function parseXlsx(buf) {
108
+ const entries = readCentralDirectory(buf);
109
+ const byName = new Map(entries.map((e) => [e.name, e]));
110
+ const sheetEntry = byName.get("xl/worksheets/sheet1.xml") ??
111
+ entries
112
+ .filter((e) => /^xl\/worksheets\/[^/]+\.xml$/.test(e.name))
113
+ .sort((a, b) => a.name.localeCompare(b.name))[0];
114
+ if (!sheetEntry)
115
+ throw new Error("no worksheet found in workbook");
116
+ const sharedEntry = byName.get("xl/sharedStrings.xml");
117
+ const shared = parseSharedStrings(sharedEntry ? readEntry(buf, sharedEntry).toString("utf-8") : undefined);
118
+ const sheetXml = readEntry(buf, sheetEntry).toString("utf-8");
119
+ const cellRe = /<c(?:\s([^>]*?))?(?:\/>|>([\s\S]*?)<\/c>)/g;
120
+ const attr = (attrs, name) => attrs?.match(new RegExp(`${name}="([^"]*)"`))?.[1];
121
+ const grid = [];
122
+ for (const rowXml of elements(sheetXml, "row")) {
123
+ const cells = [];
124
+ let m;
125
+ let fallbackColumn = 0;
126
+ cellRe.lastIndex = 0;
127
+ while ((m = cellRe.exec(rowXml)) !== null) {
128
+ const [, attrs, inner = ""] = m;
129
+ const ref = attr(attrs, "r");
130
+ const column = ref ? columnIndex(ref) : fallbackColumn;
131
+ fallbackColumn = column + 1;
132
+ const type = attr(attrs, "t");
133
+ let value = "";
134
+ if (type === "inlineStr") {
135
+ value = elements(inner, "t").map(unescapeXml).join("");
136
+ }
137
+ else {
138
+ const v = elements(inner, "v")[0];
139
+ value = v === undefined ? "" : unescapeXml(v);
140
+ if (type === "s")
141
+ value = shared[Number(value)] ?? "";
142
+ }
143
+ while (cells.length < column)
144
+ cells.push("");
145
+ cells[column] = value;
146
+ }
147
+ grid.push(cells);
148
+ }
149
+ const [headers = [], ...rows] = grid;
150
+ const width = headers.length;
151
+ return {
152
+ headers,
153
+ rows: rows.map((r) => {
154
+ const row = r.slice(0, Math.max(width, r.length));
155
+ while (row.length < width)
156
+ row.push("");
157
+ return row;
158
+ }),
159
+ };
160
+ }