rastack 0.0.46 → 0.0.48
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +4 -0
- package/dist/compile/entities.d.ts +0 -49
- package/dist/compile/entities.js +110 -22
- package/dist/compile/index.d.ts +1 -1
- package/dist/compile/index.js +1 -1
- package/dist/compile/program.js +1 -1
- package/dist/define/db.d.ts +1 -1
- package/dist/define/db.js +1 -1
- package/dist/import/index.d.ts +18 -0
- package/dist/import/index.js +57 -0
- package/dist/import/pdf.d.ts +25 -0
- package/dist/import/pdf.js +326 -0
- package/dist/import/tabular.d.ts +76 -0
- package/dist/import/tabular.js +98 -0
- package/dist/import/xlsx.d.ts +18 -0
- package/dist/import/xlsx.js +160 -0
- package/dist/rastack-import.d.ts +22 -0
- package/dist/rastack-import.js +165 -0
- package/dist/rastack.d.ts +1 -0
- package/dist/rastack.js +7 -0
- package/dist/validate/adapters.d.ts +42 -0
- package/dist/validate/adapters.js +109 -0
- package/dist/validate/index.d.ts +2 -0
- package/dist/validate/index.js +18 -0
- package/dist/validate/machine.d.ts +175 -0
- package/dist/validate/machine.js +347 -0
- package/dist/wasm/rastack_wasm_bg.wasm +0 -0
- package/hooks/form/form.ts +16 -5
- package/hooks/form/interfaces.ts +15 -0
- package/hooks/form/structure.ts +28 -0
- package/package.json +1 -1
- package/src/compile/entities.ts +161 -59
- package/src/compile/index.ts +1 -1
- package/src/compile/program.ts +1 -1
- package/src/define/db.ts +1 -1
- package/src/import/index.ts +41 -0
- package/src/import/pdf.ts +304 -0
- package/src/import/tabular.ts +159 -0
- package/src/import/xlsx.ts +186 -0
- package/src/rastack-import.ts +203 -0
- package/src/rastack.ts +7 -0
- package/src/validate/adapters.ts +118 -0
- package/src/validate/index.ts +2 -0
- package/src/validate/machine.ts +525 -0
- package/test/entities.spec.ts +99 -0
- package/test/import.spec.ts +241 -0
- package/test/validate.spec.ts +319 -0
- package/validate.ts +10 -0
- package/wasm/rastack_wasm_bg.wasm +0 -0
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A tiny, dependency-free PDF *table* extractor for `rastack import`.
|
|
3
|
+
*
|
|
4
|
+
* PDFs carry no table structure — only positioned text — so this is
|
|
5
|
+
* deliberately a best-effort extractor for the common case of a machine-
|
|
6
|
+
* generated tabular report: it decodes every content stream (FlateDecode via
|
|
7
|
+
* node's zlib, plus raw streams), replays the text-positioning operators
|
|
8
|
+
* (`Tm`/`Td`/`TD`/`T*`) to give each shown string (`Tj`/`TJ`/`'`) an (x, y),
|
|
9
|
+
* clusters strings that share a baseline into rows, and orders each row's
|
|
10
|
+
* cells left-to-right. The first row becomes the header.
|
|
11
|
+
*
|
|
12
|
+
* Anything a text-line model can't represent — merged cells, wrapped cell
|
|
13
|
+
* text, scanned images — won't survive; for those, export the source data to
|
|
14
|
+
* CSV/XLSX instead. Every extracted value still goes through the constraint
|
|
15
|
+
* state machine, so a misread cell is *rejected with a located error*, never
|
|
16
|
+
* silently imported.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { inflateSync } from "zlib";
|
|
20
|
+
import type { TabularData } from "./tabular";
|
|
21
|
+
|
|
22
|
+
interface TextChunk {
|
|
23
|
+
x: number;
|
|
24
|
+
y: number;
|
|
25
|
+
text: string;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** How close two baselines must be (in text-space units) to share a row. */
|
|
29
|
+
const ROW_TOLERANCE = 2;
|
|
30
|
+
|
|
31
|
+
// ---------------------------------------------------------------------------
|
|
32
|
+
// Stream extraction
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
/** Decode every content stream in the document to latin1 text. */
|
|
36
|
+
function contentStreams(buf: Buffer): string[] {
|
|
37
|
+
const raw = buf.toString("latin1");
|
|
38
|
+
const streams: string[] = [];
|
|
39
|
+
const re = /stream\r?\n/g;
|
|
40
|
+
let m: RegExpExecArray | null;
|
|
41
|
+
|
|
42
|
+
while ((m = re.exec(raw)) !== null) {
|
|
43
|
+
const start = m.index + m[0].length;
|
|
44
|
+
const end = raw.indexOf("endstream", start);
|
|
45
|
+
if (end === -1) break;
|
|
46
|
+
|
|
47
|
+
// The stream dictionary sits just before the `stream` keyword.
|
|
48
|
+
const dictStart = raw.lastIndexOf("<<", m.index);
|
|
49
|
+
const dict = dictStart === -1 ? "" : raw.slice(dictStart, m.index);
|
|
50
|
+
|
|
51
|
+
let data = buf.subarray(start, end);
|
|
52
|
+
// Trim the EOL PDF writers place before `endstream`.
|
|
53
|
+
while (
|
|
54
|
+
data.length &&
|
|
55
|
+
(data[data.length - 1] === 0x0a || data[data.length - 1] === 0x0d)
|
|
56
|
+
) {
|
|
57
|
+
data = data.subarray(0, data.length - 1);
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
if (/\/FlateDecode/.test(dict)) {
|
|
61
|
+
try {
|
|
62
|
+
streams.push(inflateSync(data).toString("latin1"));
|
|
63
|
+
} catch {
|
|
64
|
+
// Not a decodable text stream (e.g. an image) — skip it.
|
|
65
|
+
}
|
|
66
|
+
} else if (!/\/Filter/.test(dict)) {
|
|
67
|
+
streams.push(data.toString("latin1"));
|
|
68
|
+
}
|
|
69
|
+
re.lastIndex = end;
|
|
70
|
+
}
|
|
71
|
+
return streams;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// ---------------------------------------------------------------------------
|
|
75
|
+
// Content-stream text replay
|
|
76
|
+
// ---------------------------------------------------------------------------
|
|
77
|
+
|
|
78
|
+
/** Unescape a PDF literal string body: `\(`, `\)`, `\\`, `\n`, `\t`, `\ddd`. */
|
|
79
|
+
function unescapePdfString(body: string): string {
|
|
80
|
+
let out = "";
|
|
81
|
+
for (let i = 0; i < body.length; i++) {
|
|
82
|
+
const ch = body[i];
|
|
83
|
+
if (ch !== "\\") {
|
|
84
|
+
out += ch;
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
const next = body[++i];
|
|
88
|
+
switch (next) {
|
|
89
|
+
case "n": out += "\n"; break;
|
|
90
|
+
case "r": out += "\r"; break;
|
|
91
|
+
case "t": out += "\t"; break;
|
|
92
|
+
case "b": out += "\b"; break;
|
|
93
|
+
case "f": out += "\f"; break;
|
|
94
|
+
case "(": out += "("; break;
|
|
95
|
+
case ")": out += ")"; break;
|
|
96
|
+
case "\\": out += "\\"; break;
|
|
97
|
+
default: {
|
|
98
|
+
// Octal escape \ddd (1–3 digits)
|
|
99
|
+
if (next >= "0" && next <= "7") {
|
|
100
|
+
let digits = next;
|
|
101
|
+
while (
|
|
102
|
+
digits.length < 3 &&
|
|
103
|
+
body[i + 1] >= "0" &&
|
|
104
|
+
body[i + 1] <= "7"
|
|
105
|
+
) {
|
|
106
|
+
digits += body[++i];
|
|
107
|
+
}
|
|
108
|
+
out += String.fromCharCode(parseInt(digits, 8));
|
|
109
|
+
} else if (next !== undefined) {
|
|
110
|
+
out += next;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
return out;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Tokenise a content stream into PDF operands/operators — enough of the
|
|
120
|
+
* grammar for text extraction: literal strings, numbers, names, arrays and
|
|
121
|
+
* bare operators.
|
|
122
|
+
*/
|
|
123
|
+
function tokenize(stream: string): Array<string | number | { str: string } | unknown[]> {
|
|
124
|
+
const tokens: Array<string | number | { str: string } | unknown[]> = [];
|
|
125
|
+
const stack: Array<Array<string | number | { str: string } | unknown[]>> = [];
|
|
126
|
+
let current = tokens;
|
|
127
|
+
let i = 0;
|
|
128
|
+
|
|
129
|
+
while (i < stream.length) {
|
|
130
|
+
const ch = stream[i];
|
|
131
|
+
if (/\s/.test(ch)) { i++; continue; }
|
|
132
|
+
|
|
133
|
+
if (ch === "(") {
|
|
134
|
+
// Literal string — respects nesting and escapes.
|
|
135
|
+
let depth = 1;
|
|
136
|
+
let j = i + 1;
|
|
137
|
+
let body = "";
|
|
138
|
+
while (j < stream.length && depth > 0) {
|
|
139
|
+
const c = stream[j];
|
|
140
|
+
if (c === "\\") { body += c + (stream[j + 1] ?? ""); j += 2; continue; }
|
|
141
|
+
if (c === "(") depth++;
|
|
142
|
+
else if (c === ")") { depth--; if (depth === 0) break; }
|
|
143
|
+
body += c;
|
|
144
|
+
j++;
|
|
145
|
+
}
|
|
146
|
+
current.push({ str: unescapePdfString(body) });
|
|
147
|
+
i = j + 1;
|
|
148
|
+
} else if (ch === "<" && stream[i + 1] !== "<") {
|
|
149
|
+
// Hex string
|
|
150
|
+
const end = stream.indexOf(">", i);
|
|
151
|
+
const hex = stream.slice(i + 1, end === -1 ? stream.length : end).replace(/\s/g, "");
|
|
152
|
+
let text = "";
|
|
153
|
+
for (let k = 0; k + 1 < hex.length; k += 2) {
|
|
154
|
+
text += String.fromCharCode(parseInt(hex.slice(k, k + 2), 16));
|
|
155
|
+
}
|
|
156
|
+
current.push({ str: text });
|
|
157
|
+
i = (end === -1 ? stream.length : end) + 1;
|
|
158
|
+
} else if (ch === "[") {
|
|
159
|
+
const arr: unknown[] = [];
|
|
160
|
+
current.push(arr);
|
|
161
|
+
stack.push(current);
|
|
162
|
+
current = arr as typeof current;
|
|
163
|
+
i++;
|
|
164
|
+
} else if (ch === "]") {
|
|
165
|
+
current = stack.pop() ?? tokens;
|
|
166
|
+
i++;
|
|
167
|
+
} else if (ch === "<" && stream[i + 1] === "<") {
|
|
168
|
+
// Inline dictionary — skip to the matching >>
|
|
169
|
+
let depth = 1;
|
|
170
|
+
let j = i + 2;
|
|
171
|
+
while (j < stream.length && depth > 0) {
|
|
172
|
+
if (stream.startsWith("<<", j)) { depth++; j += 2; }
|
|
173
|
+
else if (stream.startsWith(">>", j)) { depth--; j += 2; }
|
|
174
|
+
else j++;
|
|
175
|
+
}
|
|
176
|
+
i = j;
|
|
177
|
+
} else if (/[-+.\d]/.test(ch)) {
|
|
178
|
+
const m = /^[-+]?[\d.]+/.exec(stream.slice(i))!;
|
|
179
|
+
current.push(parseFloat(m[0]));
|
|
180
|
+
i += m[0].length;
|
|
181
|
+
} else if (ch === "/") {
|
|
182
|
+
const m = /^\/[^\s()<>[\]{}/%]*/.exec(stream.slice(i))!;
|
|
183
|
+
current.push(m[0]);
|
|
184
|
+
i += m[0].length;
|
|
185
|
+
} else {
|
|
186
|
+
const m = /^[^\s()<>[\]{}/%]+/.exec(stream.slice(i));
|
|
187
|
+
if (!m) { i++; continue; }
|
|
188
|
+
current.push(m[0]);
|
|
189
|
+
i += m[0].length;
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return tokens;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** Replay text operators, collecting each shown string with its position. */
|
|
196
|
+
function extractChunks(stream: string): TextChunk[] {
|
|
197
|
+
const chunks: TextChunk[] = [];
|
|
198
|
+
const tokens = tokenize(stream);
|
|
199
|
+
|
|
200
|
+
let x = 0;
|
|
201
|
+
let y = 0;
|
|
202
|
+
let leading = 0;
|
|
203
|
+
const operands: Array<string | number | { str: string } | unknown[]> = [];
|
|
204
|
+
|
|
205
|
+
const shown = (text: string) => {
|
|
206
|
+
if (text !== "") chunks.push({ x, y, text });
|
|
207
|
+
};
|
|
208
|
+
const num = (v: unknown): number => (typeof v === "number" ? v : 0);
|
|
209
|
+
|
|
210
|
+
for (const token of tokens) {
|
|
211
|
+
if (typeof token !== "string" || token.startsWith("/")) {
|
|
212
|
+
operands.push(token);
|
|
213
|
+
continue;
|
|
214
|
+
}
|
|
215
|
+
switch (token) {
|
|
216
|
+
case "Tm":
|
|
217
|
+
x = num(operands[operands.length - 2]);
|
|
218
|
+
y = num(operands[operands.length - 1]);
|
|
219
|
+
break;
|
|
220
|
+
case "Td":
|
|
221
|
+
x += num(operands[operands.length - 2]);
|
|
222
|
+
y += num(operands[operands.length - 1]);
|
|
223
|
+
break;
|
|
224
|
+
case "TD":
|
|
225
|
+
leading = -num(operands[operands.length - 1]);
|
|
226
|
+
x += num(operands[operands.length - 2]);
|
|
227
|
+
y += num(operands[operands.length - 1]);
|
|
228
|
+
break;
|
|
229
|
+
case "TL":
|
|
230
|
+
leading = num(operands[operands.length - 1]);
|
|
231
|
+
break;
|
|
232
|
+
case "T*":
|
|
233
|
+
y -= leading;
|
|
234
|
+
break;
|
|
235
|
+
case "Tj":
|
|
236
|
+
case "'": {
|
|
237
|
+
if (token === "'") y -= leading;
|
|
238
|
+
const s = operands[operands.length - 1];
|
|
239
|
+
if (s && typeof s === "object" && "str" in s) shown((s as { str: string }).str);
|
|
240
|
+
break;
|
|
241
|
+
}
|
|
242
|
+
case "TJ": {
|
|
243
|
+
const arr = operands[operands.length - 1];
|
|
244
|
+
if (Array.isArray(arr)) {
|
|
245
|
+
const text = arr
|
|
246
|
+
.filter((el): el is { str: string } =>
|
|
247
|
+
Boolean(el && typeof el === "object" && "str" in el),
|
|
248
|
+
)
|
|
249
|
+
.map((el) => el.str)
|
|
250
|
+
.join("");
|
|
251
|
+
shown(text);
|
|
252
|
+
}
|
|
253
|
+
break;
|
|
254
|
+
}
|
|
255
|
+
case "BT":
|
|
256
|
+
x = 0;
|
|
257
|
+
y = 0;
|
|
258
|
+
break;
|
|
259
|
+
}
|
|
260
|
+
operands.length = 0;
|
|
261
|
+
}
|
|
262
|
+
return chunks;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
// ---------------------------------------------------------------------------
|
|
266
|
+
// Row clustering
|
|
267
|
+
// ---------------------------------------------------------------------------
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* Parse a PDF buffer into headers + rows: text chunks sharing a baseline form
|
|
271
|
+
* a row (top of the page first), ordered left-to-right; the first row is the
|
|
272
|
+
* header. Rows with a different cell count than the header are padded or
|
|
273
|
+
* truncated — the state machine will flag what doesn't validate.
|
|
274
|
+
*/
|
|
275
|
+
export function parsePdfTable(buf: Buffer): TabularData {
|
|
276
|
+
const chunks = contentStreams(buf).flatMap(extractChunks);
|
|
277
|
+
if (chunks.length === 0) return { headers: [], rows: [] };
|
|
278
|
+
|
|
279
|
+
// Cluster by baseline (y), newest tolerance-merge wins.
|
|
280
|
+
const lines: Array<{ y: number; chunks: TextChunk[] }> = [];
|
|
281
|
+
for (const chunk of chunks) {
|
|
282
|
+
const line = lines.find((l) => Math.abs(l.y - chunk.y) <= ROW_TOLERANCE);
|
|
283
|
+
if (line) line.chunks.push(chunk);
|
|
284
|
+
else lines.push({ y: chunk.y, chunks: [chunk] });
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
// Page order: highest y first (PDF's origin is bottom-left).
|
|
288
|
+
lines.sort((a, b) => b.y - a.y);
|
|
289
|
+
const grid = lines.map((line) =>
|
|
290
|
+
line.chunks
|
|
291
|
+
.sort((a, b) => a.x - b.x)
|
|
292
|
+
.map((c) => c.text.trim()),
|
|
293
|
+
);
|
|
294
|
+
|
|
295
|
+
const [headers = [], ...rows] = grid;
|
|
296
|
+
return {
|
|
297
|
+
headers,
|
|
298
|
+
rows: rows.map((row) => {
|
|
299
|
+
const cells = row.slice(0, headers.length);
|
|
300
|
+
while (cells.length < headers.length) cells.push("");
|
|
301
|
+
return cells;
|
|
302
|
+
}),
|
|
303
|
+
};
|
|
304
|
+
}
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The pure core of `rastack import`: map a tabular source (CSV/XLSX/PDF —
|
|
3
|
+
* already parsed to headers + string rows) onto a resource's fields, then run
|
|
4
|
+
* every row through the resource's **constraint state machine**
|
|
5
|
+
* (`rastack/validate`) with a dataset-level context, so `unique` sees
|
|
6
|
+
* duplicates across the file and `resolved` checks foreign keys against
|
|
7
|
+
* supplied related ids.
|
|
8
|
+
*
|
|
9
|
+
* Everything here is data-in/data-out (no filesystem), tested in
|
|
10
|
+
* `tools/test/import.spec.ts`; the parsers and the CLI are the IO shell.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import {
|
|
14
|
+
compileRecordMachine,
|
|
15
|
+
createDatasetContext,
|
|
16
|
+
runRecordMachine,
|
|
17
|
+
type FieldSpec,
|
|
18
|
+
type FieldState,
|
|
19
|
+
type RecordRun,
|
|
20
|
+
type SpecRelation,
|
|
21
|
+
} from "../validate";
|
|
22
|
+
|
|
23
|
+
/** A parsed tabular source: one header row + string cell rows. */
|
|
24
|
+
export interface TabularData {
|
|
25
|
+
headers: string[];
|
|
26
|
+
rows: string[][];
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** How one source column mapped onto the resource (or didn't). */
|
|
30
|
+
export interface ColumnMapping {
|
|
31
|
+
header: string;
|
|
32
|
+
field: string | null;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** One constraint failure, located by row (0-based, excluding the header). */
|
|
36
|
+
export interface ImportIssue {
|
|
37
|
+
row: number;
|
|
38
|
+
field: string;
|
|
39
|
+
gate: FieldState;
|
|
40
|
+
constraint: string;
|
|
41
|
+
message: string;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** The full validation report for one imported file. */
|
|
45
|
+
export interface ImportReport {
|
|
46
|
+
total: number;
|
|
47
|
+
valid: number;
|
|
48
|
+
invalid: number;
|
|
49
|
+
/** Source-column → field mapping (null = column ignored). */
|
|
50
|
+
columns: ColumnMapping[];
|
|
51
|
+
/** Resource fields no source column covered (they'll fail `present` if required). */
|
|
52
|
+
missingFields: string[];
|
|
53
|
+
/** Every constraint failure, in row order. */
|
|
54
|
+
issues: ImportIssue[];
|
|
55
|
+
/** The coerced records of the *valid* rows — ready to POST or seed. */
|
|
56
|
+
records: Record<string, unknown>[];
|
|
57
|
+
/** Per-row machine runs, index-aligned with the source rows. */
|
|
58
|
+
rows: RecordRun[];
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** `"Airport Code"`, `airport_code`, `airportCode` → `airportcode`. */
|
|
62
|
+
function normalise(name: string): string {
|
|
63
|
+
return name.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Map source headers onto field specs by normalised name (case, `_`, `-` and
|
|
68
|
+
* spaces are insignificant). Unmatched headers are carried as ignored columns;
|
|
69
|
+
* unmatched fields are reported so a missing required column reads as a
|
|
70
|
+
* mapping problem, not just N× `present` failures.
|
|
71
|
+
*/
|
|
72
|
+
export function mapHeaders(
|
|
73
|
+
headers: string[],
|
|
74
|
+
specs: FieldSpec[],
|
|
75
|
+
): { columns: ColumnMapping[]; missingFields: string[] } {
|
|
76
|
+
const byNormalised = new Map(specs.map((s) => [normalise(s.name), s.name]));
|
|
77
|
+
const used = new Set<string>();
|
|
78
|
+
|
|
79
|
+
const columns = headers.map((header) => {
|
|
80
|
+
const field = byNormalised.get(normalise(header)) ?? null;
|
|
81
|
+
if (field) used.add(field);
|
|
82
|
+
return { header, field };
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
return {
|
|
86
|
+
columns,
|
|
87
|
+
missingFields: specs.map((s) => s.name).filter((n) => !used.has(n)),
|
|
88
|
+
};
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** Optional dataset-level inputs for FK resolution. */
|
|
92
|
+
export interface ValidateDatasetOptions {
|
|
93
|
+
/** Known ids per `"app.model"` — engages the `resolved` gate. */
|
|
94
|
+
relatedIds?: Record<string, Set<string>>;
|
|
95
|
+
/** Custom FK resolver (wins over `relatedIds`). */
|
|
96
|
+
resolveRelation?: (relation: SpecRelation, value: unknown) => boolean;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Run a whole tabular source through a resource's record machine. Each row is
|
|
101
|
+
* projected onto the mapped fields and validated; the dataset context makes
|
|
102
|
+
* the `unique` gate reject in-file duplicates and (when related ids are
|
|
103
|
+
* given) the `resolved` gate reject dangling foreign keys.
|
|
104
|
+
*/
|
|
105
|
+
export function validateDataset(
|
|
106
|
+
specs: FieldSpec[],
|
|
107
|
+
data: TabularData,
|
|
108
|
+
options: ValidateDatasetOptions = {},
|
|
109
|
+
): ImportReport {
|
|
110
|
+
const { columns, missingFields } = mapHeaders(data.headers, specs);
|
|
111
|
+
const machine = compileRecordMachine(specs);
|
|
112
|
+
const ctx = createDatasetContext(machine, options);
|
|
113
|
+
|
|
114
|
+
const rows: RecordRun[] = [];
|
|
115
|
+
const issues: ImportIssue[] = [];
|
|
116
|
+
const records: Record<string, unknown>[] = [];
|
|
117
|
+
|
|
118
|
+
data.rows.forEach((cells, index) => {
|
|
119
|
+
const record: Record<string, unknown> = {};
|
|
120
|
+
columns.forEach((col, i) => {
|
|
121
|
+
if (col.field !== null) record[col.field] = cells[i] ?? "";
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
const run = runRecordMachine(machine, record, ctx);
|
|
125
|
+
rows.push(run);
|
|
126
|
+
if (run.state === "valid") {
|
|
127
|
+
records.push(run.values);
|
|
128
|
+
} else {
|
|
129
|
+
for (const error of run.errors) issues.push({ row: index, ...error });
|
|
130
|
+
}
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
return {
|
|
134
|
+
total: data.rows.length,
|
|
135
|
+
valid: records.length,
|
|
136
|
+
invalid: data.rows.length - records.length,
|
|
137
|
+
columns,
|
|
138
|
+
missingFields,
|
|
139
|
+
issues,
|
|
140
|
+
records,
|
|
141
|
+
rows,
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Pull the id set out of an already-parsed related table (its `id` column,
|
|
147
|
+
* else its first column) — the ids `validateDataset` resolves foreign keys
|
|
148
|
+
* against when importing a dependent table alongside its target.
|
|
149
|
+
*/
|
|
150
|
+
export function idsFromTabular(data: TabularData): Set<string> {
|
|
151
|
+
let idColumn = data.headers.findIndex((h) => normalise(h) === "id");
|
|
152
|
+
if (idColumn === -1) idColumn = 0;
|
|
153
|
+
const ids = new Set<string>();
|
|
154
|
+
for (const row of data.rows) {
|
|
155
|
+
const value = (row[idColumn] ?? "").trim();
|
|
156
|
+
if (value !== "") ids.add(value);
|
|
157
|
+
}
|
|
158
|
+
return ids;
|
|
159
|
+
}
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A tiny, dependency-free XLSX reader for `rastack import`.
|
|
3
|
+
*
|
|
4
|
+
* Why hand-roll it: the CLI stays dependency-light (the same reasoning as the
|
|
5
|
+
* hand-rolled ZIP *writer* in `deploy/zip.ts`), and an `.xlsx` workbook is
|
|
6
|
+
* just a ZIP of small XML parts — the reader below understands exactly the
|
|
7
|
+
* subset a tabular import needs: the ZIP central directory (STORE and DEFLATE
|
|
8
|
+
* entries, inflated via node's zlib), `xl/sharedStrings.xml`, and the cells of
|
|
9
|
+
* the first worksheet. Formulas are read through their cached `<v>` results;
|
|
10
|
+
* merged cells, styles and charts are ignored — an import wants values.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { inflateRawSync } from "zlib";
|
|
14
|
+
import type { TabularData } from "./tabular";
|
|
15
|
+
|
|
16
|
+
// ---------------------------------------------------------------------------
|
|
17
|
+
// ZIP reading
|
|
18
|
+
// ---------------------------------------------------------------------------
|
|
19
|
+
|
|
20
|
+
interface ZipEntryRecord {
|
|
21
|
+
name: string;
|
|
22
|
+
method: number;
|
|
23
|
+
compressedSize: number;
|
|
24
|
+
localHeaderOffset: number;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** Locate the end-of-central-directory record (scan back over the comment). */
|
|
28
|
+
function findEocd(buf: Buffer): number {
|
|
29
|
+
for (let i = buf.length - 22; i >= 0; i--) {
|
|
30
|
+
if (buf.readUInt32LE(i) === 0x06054b50) return i;
|
|
31
|
+
}
|
|
32
|
+
throw new Error("not a ZIP archive (no end-of-central-directory record)");
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function readCentralDirectory(buf: Buffer): ZipEntryRecord[] {
|
|
36
|
+
const eocd = findEocd(buf);
|
|
37
|
+
const count = buf.readUInt16LE(eocd + 10);
|
|
38
|
+
let offset = buf.readUInt32LE(eocd + 16);
|
|
39
|
+
|
|
40
|
+
const entries: ZipEntryRecord[] = [];
|
|
41
|
+
for (let i = 0; i < count; i++) {
|
|
42
|
+
if (buf.readUInt32LE(offset) !== 0x02014b50) {
|
|
43
|
+
throw new Error("corrupt ZIP central directory");
|
|
44
|
+
}
|
|
45
|
+
const method = buf.readUInt16LE(offset + 10);
|
|
46
|
+
const compressedSize = buf.readUInt32LE(offset + 20);
|
|
47
|
+
const nameLength = buf.readUInt16LE(offset + 28);
|
|
48
|
+
const extraLength = buf.readUInt16LE(offset + 30);
|
|
49
|
+
const commentLength = buf.readUInt16LE(offset + 32);
|
|
50
|
+
const localHeaderOffset = buf.readUInt32LE(offset + 42);
|
|
51
|
+
const name = buf
|
|
52
|
+
.subarray(offset + 46, offset + 46 + nameLength)
|
|
53
|
+
.toString("utf-8");
|
|
54
|
+
entries.push({ name, method, compressedSize, localHeaderOffset });
|
|
55
|
+
offset += 46 + nameLength + extraLength + commentLength;
|
|
56
|
+
}
|
|
57
|
+
return entries;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
function readEntry(buf: Buffer, entry: ZipEntryRecord): Buffer {
|
|
61
|
+
const at = entry.localHeaderOffset;
|
|
62
|
+
if (buf.readUInt32LE(at) !== 0x04034b50) {
|
|
63
|
+
throw new Error(`corrupt ZIP local header for ${entry.name}`);
|
|
64
|
+
}
|
|
65
|
+
const nameLength = buf.readUInt16LE(at + 26);
|
|
66
|
+
const extraLength = buf.readUInt16LE(at + 28);
|
|
67
|
+
const start = at + 30 + nameLength + extraLength;
|
|
68
|
+
const data = buf.subarray(start, start + entry.compressedSize);
|
|
69
|
+
|
|
70
|
+
if (entry.method === 0) return Buffer.from(data); // STORE
|
|
71
|
+
if (entry.method === 8) return inflateRawSync(data); // DEFLATE
|
|
72
|
+
throw new Error(`unsupported ZIP compression method ${entry.method}`);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// ---------------------------------------------------------------------------
|
|
76
|
+
// Minimal XML helpers (the sheet parts are machine-written, flat XML)
|
|
77
|
+
// ---------------------------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
function unescapeXml(text: string): string {
|
|
80
|
+
return text
|
|
81
|
+
.replace(/</g, "<")
|
|
82
|
+
.replace(/>/g, ">")
|
|
83
|
+
.replace(/"/g, '"')
|
|
84
|
+
.replace(/'/g, "'")
|
|
85
|
+
.replace(/&#(\d+);/g, (_, code) => String.fromCharCode(Number(code)))
|
|
86
|
+
.replace(/&/g, "&");
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** All matches of `<tag ...>inner</tag>` (non-nested — true of sheet XML). */
|
|
90
|
+
function elements(xml: string, tag: string): string[] {
|
|
91
|
+
const re = new RegExp(`<${tag}(?:\\s[^>]*)?>([\\s\\S]*?)</${tag}>`, "g");
|
|
92
|
+
const out: string[] = [];
|
|
93
|
+
let m: RegExpExecArray | null;
|
|
94
|
+
while ((m = re.exec(xml)) !== null) out.push(m[1]);
|
|
95
|
+
return out;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/** `"BC7"` → 0-based column index (54). */
|
|
99
|
+
function columnIndex(cellRef: string): number {
|
|
100
|
+
let index = 0;
|
|
101
|
+
for (const ch of cellRef) {
|
|
102
|
+
if (ch < "A" || ch > "Z") break;
|
|
103
|
+
index = index * 26 + (ch.charCodeAt(0) - 64);
|
|
104
|
+
}
|
|
105
|
+
return index - 1;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// ---------------------------------------------------------------------------
|
|
109
|
+
// Workbook parsing
|
|
110
|
+
// ---------------------------------------------------------------------------
|
|
111
|
+
|
|
112
|
+
function parseSharedStrings(xml: string | undefined): string[] {
|
|
113
|
+
if (!xml) return [];
|
|
114
|
+
// Each <si> may hold one <t> or several rich-text runs — concatenate them.
|
|
115
|
+
return elements(xml, "si").map((si) =>
|
|
116
|
+
elements(si, "t").map(unescapeXml).join(""),
|
|
117
|
+
);
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Parse an `.xlsx` workbook buffer into headers + rows (first worksheet, first
|
|
122
|
+
* row as the header). Cell values come back as strings — the constraint state
|
|
123
|
+
* machine's `typed` gate owns coercion, same as for CSV.
|
|
124
|
+
*/
|
|
125
|
+
export function parseXlsx(buf: Buffer): TabularData {
|
|
126
|
+
const entries = readCentralDirectory(buf);
|
|
127
|
+
const byName = new Map(entries.map((e) => [e.name, e]));
|
|
128
|
+
|
|
129
|
+
const sheetEntry =
|
|
130
|
+
byName.get("xl/worksheets/sheet1.xml") ??
|
|
131
|
+
entries
|
|
132
|
+
.filter((e) => /^xl\/worksheets\/[^/]+\.xml$/.test(e.name))
|
|
133
|
+
.sort((a, b) => a.name.localeCompare(b.name))[0];
|
|
134
|
+
if (!sheetEntry) throw new Error("no worksheet found in workbook");
|
|
135
|
+
|
|
136
|
+
const sharedEntry = byName.get("xl/sharedStrings.xml");
|
|
137
|
+
const shared = parseSharedStrings(
|
|
138
|
+
sharedEntry ? readEntry(buf, sharedEntry).toString("utf-8") : undefined,
|
|
139
|
+
);
|
|
140
|
+
|
|
141
|
+
const sheetXml = readEntry(buf, sheetEntry).toString("utf-8");
|
|
142
|
+
|
|
143
|
+
const cellRe =
|
|
144
|
+
/<c(?:\s([^>]*?))?(?:\/>|>([\s\S]*?)<\/c>)/g;
|
|
145
|
+
const attr = (attrs: string | undefined, name: string): string | undefined =>
|
|
146
|
+
attrs?.match(new RegExp(`${name}="([^"]*)"`))?.[1];
|
|
147
|
+
|
|
148
|
+
const grid: string[][] = [];
|
|
149
|
+
for (const rowXml of elements(sheetXml, "row")) {
|
|
150
|
+
const cells: string[] = [];
|
|
151
|
+
let m: RegExpExecArray | null;
|
|
152
|
+
let fallbackColumn = 0;
|
|
153
|
+
cellRe.lastIndex = 0;
|
|
154
|
+
while ((m = cellRe.exec(rowXml)) !== null) {
|
|
155
|
+
const [, attrs, inner = ""] = m;
|
|
156
|
+
const ref = attr(attrs, "r");
|
|
157
|
+
const column = ref ? columnIndex(ref) : fallbackColumn;
|
|
158
|
+
fallbackColumn = column + 1;
|
|
159
|
+
|
|
160
|
+
const type = attr(attrs, "t");
|
|
161
|
+
let value = "";
|
|
162
|
+
if (type === "inlineStr") {
|
|
163
|
+
value = elements(inner, "t").map(unescapeXml).join("");
|
|
164
|
+
} else {
|
|
165
|
+
const v = elements(inner, "v")[0];
|
|
166
|
+
value = v === undefined ? "" : unescapeXml(v);
|
|
167
|
+
if (type === "s") value = shared[Number(value)] ?? "";
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
while (cells.length < column) cells.push("");
|
|
171
|
+
cells[column] = value;
|
|
172
|
+
}
|
|
173
|
+
grid.push(cells);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const [headers = [], ...rows] = grid;
|
|
177
|
+
const width = headers.length;
|
|
178
|
+
return {
|
|
179
|
+
headers,
|
|
180
|
+
rows: rows.map((r) => {
|
|
181
|
+
const row = r.slice(0, Math.max(width, r.length));
|
|
182
|
+
while (row.length < width) row.push("");
|
|
183
|
+
return row;
|
|
184
|
+
}),
|
|
185
|
+
};
|
|
186
|
+
}
|