rastack 0.0.45 → 0.0.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +4 -0
- package/dist/import/index.d.ts +18 -0
- package/dist/import/index.js +57 -0
- package/dist/import/pdf.d.ts +25 -0
- package/dist/import/pdf.js +326 -0
- package/dist/import/tabular.d.ts +76 -0
- package/dist/import/tabular.js +98 -0
- package/dist/import/xlsx.d.ts +18 -0
- package/dist/import/xlsx.js +160 -0
- package/dist/rastack-import.d.ts +22 -0
- package/dist/rastack-import.js +165 -0
- package/dist/rastack.d.ts +1 -0
- package/dist/rastack.js +7 -0
- package/dist/validate/adapters.d.ts +42 -0
- package/dist/validate/adapters.js +109 -0
- package/dist/validate/index.d.ts +2 -0
- package/dist/validate/index.js +18 -0
- package/dist/validate/machine.d.ts +175 -0
- package/dist/validate/machine.js +347 -0
- package/dist/wasm/rastack_wasm_bg.wasm +0 -0
- package/hooks/form/form.ts +16 -5
- package/hooks/form/interfaces.ts +15 -0
- package/hooks/form/structure.ts +28 -0
- package/package.json +1 -1
- package/src/import/index.ts +41 -0
- package/src/import/pdf.ts +304 -0
- package/src/import/tabular.ts +159 -0
- package/src/import/xlsx.ts +186 -0
- package/src/rastack-import.ts +203 -0
- package/src/rastack.ts +7 -0
- package/src/validate/adapters.ts +118 -0
- package/src/validate/index.ts +2 -0
- package/src/validate/machine.ts +525 -0
- package/test/import.spec.ts +241 -0
- package/test/validate.spec.ts +319 -0
- package/validate.ts +10 -0
- package/wasm/rastack_wasm_bg.wasm +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,10 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
|
|
4
4
|
|
|
5
|
+
### [0.0.47](https://github.com/theserverkid/reactapistack/compare/v0.0.46...v0.0.47) (2026-07-13)
|
|
6
|
+
|
|
7
|
+
### [0.0.46](https://github.com/theserverkid/reactapistack/compare/v0.0.45...v0.0.46) (2026-07-13)
|
|
8
|
+
|
|
5
9
|
### [0.0.45](https://github.com/theserverkid/reactapistack/compare/v0.0.44...v0.0.45) (2026-07-13)
|
|
6
10
|
|
|
7
11
|
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `rastack import` — load external data (CSV, Excel, PDF) and validate it
|
|
3
|
+
* against a resource's constraint state machine.
|
|
4
|
+
*
|
|
5
|
+
* `parseTabular` dispatches on the file extension to the format parsers, all
|
|
6
|
+
* of which produce the same {@link TabularData}; `validateDataset` (in
|
|
7
|
+
* `tabular.ts`) is the shared, pure validation core.
|
|
8
|
+
*/
|
|
9
|
+
import type { TabularData } from "./tabular";
|
|
10
|
+
export * from "./tabular";
|
|
11
|
+
export { parseXlsx } from "./xlsx";
|
|
12
|
+
export { parsePdfTable } from "./pdf";
|
|
13
|
+
/** The source formats `rastack import` understands. */
|
|
14
|
+
export type TabularFormat = "csv" | "xlsx" | "pdf";
|
|
15
|
+
/** Infer the source format from a filename, or `null` if unsupported. */
|
|
16
|
+
export declare function formatOf(fileName: string): TabularFormat | null;
|
|
17
|
+
/** Parse a source file's bytes into headers + rows, by format. */
|
|
18
|
+
export declare function parseTabular(format: TabularFormat, data: Buffer): TabularData;
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* `rastack import` — load external data (CSV, Excel, PDF) and validate it
|
|
4
|
+
* against a resource's constraint state machine.
|
|
5
|
+
*
|
|
6
|
+
* `parseTabular` dispatches on the file extension to the format parsers, all
|
|
7
|
+
* of which produce the same {@link TabularData}; `validateDataset` (in
|
|
8
|
+
* `tabular.ts`) is the shared, pure validation core.
|
|
9
|
+
*/
|
|
10
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
11
|
+
if (k2 === undefined) k2 = k;
|
|
12
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
13
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
14
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
15
|
+
}
|
|
16
|
+
Object.defineProperty(o, k2, desc);
|
|
17
|
+
}) : (function(o, m, k, k2) {
|
|
18
|
+
if (k2 === undefined) k2 = k;
|
|
19
|
+
o[k2] = m[k];
|
|
20
|
+
}));
|
|
21
|
+
var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
22
|
+
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
23
|
+
};
|
|
24
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
25
|
+
exports.parsePdfTable = exports.parseXlsx = void 0;
|
|
26
|
+
exports.formatOf = formatOf;
|
|
27
|
+
exports.parseTabular = parseTabular;
|
|
28
|
+
const csv_schema_1 = require("../csv-schema");
|
|
29
|
+
const xlsx_1 = require("./xlsx");
|
|
30
|
+
const pdf_1 = require("./pdf");
|
|
31
|
+
__exportStar(require("./tabular"), exports);
|
|
32
|
+
var xlsx_2 = require("./xlsx");
|
|
33
|
+
Object.defineProperty(exports, "parseXlsx", { enumerable: true, get: function () { return xlsx_2.parseXlsx; } });
|
|
34
|
+
var pdf_2 = require("./pdf");
|
|
35
|
+
Object.defineProperty(exports, "parsePdfTable", { enumerable: true, get: function () { return pdf_2.parsePdfTable; } });
|
|
36
|
+
/** Infer the source format from a filename, or `null` if unsupported. */
|
|
37
|
+
function formatOf(fileName) {
|
|
38
|
+
const ext = fileName.toLowerCase().split(".").pop();
|
|
39
|
+
if (ext === "csv")
|
|
40
|
+
return "csv";
|
|
41
|
+
if (ext === "xlsx" || ext === "xlsm")
|
|
42
|
+
return "xlsx";
|
|
43
|
+
if (ext === "pdf")
|
|
44
|
+
return "pdf";
|
|
45
|
+
return null;
|
|
46
|
+
}
|
|
47
|
+
/** Parse a source file's bytes into headers + rows, by format. */
|
|
48
|
+
function parseTabular(format, data) {
|
|
49
|
+
switch (format) {
|
|
50
|
+
case "csv":
|
|
51
|
+
return (0, csv_schema_1.parseCsv)(data.toString("utf-8"));
|
|
52
|
+
case "xlsx":
|
|
53
|
+
return (0, xlsx_1.parseXlsx)(data);
|
|
54
|
+
case "pdf":
|
|
55
|
+
return (0, pdf_1.parsePdfTable)(data);
|
|
56
|
+
}
|
|
57
|
+
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A tiny, dependency-free PDF *table* extractor for `rastack import`.
|
|
3
|
+
*
|
|
4
|
+
* PDFs carry no table structure — only positioned text — so this is
|
|
5
|
+
* deliberately a best-effort extractor for the common case of a machine-
|
|
6
|
+
* generated tabular report: it decodes every content stream (FlateDecode via
|
|
7
|
+
* node's zlib, plus raw streams), replays the text-positioning operators
|
|
8
|
+
* (`Tm`/`Td`/`TD`/`T*`) to give each shown string (`Tj`/`TJ`/`'`) an (x, y),
|
|
9
|
+
* clusters strings that share a baseline into rows, and orders each row's
|
|
10
|
+
* cells left-to-right. The first row becomes the header.
|
|
11
|
+
*
|
|
12
|
+
* Anything a text-line model can't represent — merged cells, wrapped cell
|
|
13
|
+
* text, scanned images — won't survive; for those, export the source data to
|
|
14
|
+
* CSV/XLSX instead. Every extracted value still goes through the constraint
|
|
15
|
+
* state machine, so a misread cell is *rejected with a located error*, never
|
|
16
|
+
* silently imported.
|
|
17
|
+
*/
|
|
18
|
+
import type { TabularData } from "./tabular";
|
|
19
|
+
/**
|
|
20
|
+
* Parse a PDF buffer into headers + rows: text chunks sharing a baseline form
|
|
21
|
+
* a row (top of the page first), ordered left-to-right; the first row is the
|
|
22
|
+
* header. Rows with a different cell count than the header are padded or
|
|
23
|
+
* truncated — the state machine will flag what doesn't validate.
|
|
24
|
+
*/
|
|
25
|
+
export declare function parsePdfTable(buf: Buffer): TabularData;
|
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* A tiny, dependency-free PDF *table* extractor for `rastack import`.
|
|
4
|
+
*
|
|
5
|
+
* PDFs carry no table structure — only positioned text — so this is
|
|
6
|
+
* deliberately a best-effort extractor for the common case of a machine-
|
|
7
|
+
* generated tabular report: it decodes every content stream (FlateDecode via
|
|
8
|
+
* node's zlib, plus raw streams), replays the text-positioning operators
|
|
9
|
+
* (`Tm`/`Td`/`TD`/`T*`) to give each shown string (`Tj`/`TJ`/`'`) an (x, y),
|
|
10
|
+
* clusters strings that share a baseline into rows, and orders each row's
|
|
11
|
+
* cells left-to-right. The first row becomes the header.
|
|
12
|
+
*
|
|
13
|
+
* Anything a text-line model can't represent — merged cells, wrapped cell
|
|
14
|
+
* text, scanned images — won't survive; for those, export the source data to
|
|
15
|
+
* CSV/XLSX instead. Every extracted value still goes through the constraint
|
|
16
|
+
* state machine, so a misread cell is *rejected with a located error*, never
|
|
17
|
+
* silently imported.
|
|
18
|
+
*/
|
|
19
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
20
|
+
exports.parsePdfTable = parsePdfTable;
|
|
21
|
+
const zlib_1 = require("zlib");
|
|
22
|
+
/** How close two baselines must be (in text-space units) to share a row. */
|
|
23
|
+
const ROW_TOLERANCE = 2;
|
|
24
|
+
// ---------------------------------------------------------------------------
|
|
25
|
+
// Stream extraction
|
|
26
|
+
// ---------------------------------------------------------------------------
|
|
27
|
+
/** Decode every content stream in the document to latin1 text. */
|
|
28
|
+
function contentStreams(buf) {
|
|
29
|
+
const raw = buf.toString("latin1");
|
|
30
|
+
const streams = [];
|
|
31
|
+
const re = /stream\r?\n/g;
|
|
32
|
+
let m;
|
|
33
|
+
while ((m = re.exec(raw)) !== null) {
|
|
34
|
+
const start = m.index + m[0].length;
|
|
35
|
+
const end = raw.indexOf("endstream", start);
|
|
36
|
+
if (end === -1)
|
|
37
|
+
break;
|
|
38
|
+
// The stream dictionary sits just before the `stream` keyword.
|
|
39
|
+
const dictStart = raw.lastIndexOf("<<", m.index);
|
|
40
|
+
const dict = dictStart === -1 ? "" : raw.slice(dictStart, m.index);
|
|
41
|
+
let data = buf.subarray(start, end);
|
|
42
|
+
// Trim the EOL PDF writers place before `endstream`.
|
|
43
|
+
while (data.length &&
|
|
44
|
+
(data[data.length - 1] === 0x0a || data[data.length - 1] === 0x0d)) {
|
|
45
|
+
data = data.subarray(0, data.length - 1);
|
|
46
|
+
}
|
|
47
|
+
if (/\/FlateDecode/.test(dict)) {
|
|
48
|
+
try {
|
|
49
|
+
streams.push((0, zlib_1.inflateSync)(data).toString("latin1"));
|
|
50
|
+
}
|
|
51
|
+
catch {
|
|
52
|
+
// Not a decodable text stream (e.g. an image) — skip it.
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
else if (!/\/Filter/.test(dict)) {
|
|
56
|
+
streams.push(data.toString("latin1"));
|
|
57
|
+
}
|
|
58
|
+
re.lastIndex = end;
|
|
59
|
+
}
|
|
60
|
+
return streams;
|
|
61
|
+
}
|
|
62
|
+
// ---------------------------------------------------------------------------
|
|
63
|
+
// Content-stream text replay
|
|
64
|
+
// ---------------------------------------------------------------------------
|
|
65
|
+
/** Unescape a PDF literal string body: `\(`, `\)`, `\\`, `\n`, `\t`, `\ddd`. */
|
|
66
|
+
function unescapePdfString(body) {
|
|
67
|
+
let out = "";
|
|
68
|
+
for (let i = 0; i < body.length; i++) {
|
|
69
|
+
const ch = body[i];
|
|
70
|
+
if (ch !== "\\") {
|
|
71
|
+
out += ch;
|
|
72
|
+
continue;
|
|
73
|
+
}
|
|
74
|
+
const next = body[++i];
|
|
75
|
+
switch (next) {
|
|
76
|
+
case "n":
|
|
77
|
+
out += "\n";
|
|
78
|
+
break;
|
|
79
|
+
case "r":
|
|
80
|
+
out += "\r";
|
|
81
|
+
break;
|
|
82
|
+
case "t":
|
|
83
|
+
out += "\t";
|
|
84
|
+
break;
|
|
85
|
+
case "b":
|
|
86
|
+
out += "\b";
|
|
87
|
+
break;
|
|
88
|
+
case "f":
|
|
89
|
+
out += "\f";
|
|
90
|
+
break;
|
|
91
|
+
case "(":
|
|
92
|
+
out += "(";
|
|
93
|
+
break;
|
|
94
|
+
case ")":
|
|
95
|
+
out += ")";
|
|
96
|
+
break;
|
|
97
|
+
case "\\":
|
|
98
|
+
out += "\\";
|
|
99
|
+
break;
|
|
100
|
+
default: {
|
|
101
|
+
// Octal escape \ddd (1–3 digits)
|
|
102
|
+
if (next >= "0" && next <= "7") {
|
|
103
|
+
let digits = next;
|
|
104
|
+
while (digits.length < 3 &&
|
|
105
|
+
body[i + 1] >= "0" &&
|
|
106
|
+
body[i + 1] <= "7") {
|
|
107
|
+
digits += body[++i];
|
|
108
|
+
}
|
|
109
|
+
out += String.fromCharCode(parseInt(digits, 8));
|
|
110
|
+
}
|
|
111
|
+
else if (next !== undefined) {
|
|
112
|
+
out += next;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
return out;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Tokenise a content stream into PDF operands/operators — enough of the
|
|
121
|
+
* grammar for text extraction: literal strings, numbers, names, arrays and
|
|
122
|
+
* bare operators.
|
|
123
|
+
*/
|
|
124
|
+
function tokenize(stream) {
|
|
125
|
+
const tokens = [];
|
|
126
|
+
const stack = [];
|
|
127
|
+
let current = tokens;
|
|
128
|
+
let i = 0;
|
|
129
|
+
while (i < stream.length) {
|
|
130
|
+
const ch = stream[i];
|
|
131
|
+
if (/\s/.test(ch)) {
|
|
132
|
+
i++;
|
|
133
|
+
continue;
|
|
134
|
+
}
|
|
135
|
+
if (ch === "(") {
|
|
136
|
+
// Literal string — respects nesting and escapes.
|
|
137
|
+
let depth = 1;
|
|
138
|
+
let j = i + 1;
|
|
139
|
+
let body = "";
|
|
140
|
+
while (j < stream.length && depth > 0) {
|
|
141
|
+
const c = stream[j];
|
|
142
|
+
if (c === "\\") {
|
|
143
|
+
body += c + (stream[j + 1] ?? "");
|
|
144
|
+
j += 2;
|
|
145
|
+
continue;
|
|
146
|
+
}
|
|
147
|
+
if (c === "(")
|
|
148
|
+
depth++;
|
|
149
|
+
else if (c === ")") {
|
|
150
|
+
depth--;
|
|
151
|
+
if (depth === 0)
|
|
152
|
+
break;
|
|
153
|
+
}
|
|
154
|
+
body += c;
|
|
155
|
+
j++;
|
|
156
|
+
}
|
|
157
|
+
current.push({ str: unescapePdfString(body) });
|
|
158
|
+
i = j + 1;
|
|
159
|
+
}
|
|
160
|
+
else if (ch === "<" && stream[i + 1] !== "<") {
|
|
161
|
+
// Hex string
|
|
162
|
+
const end = stream.indexOf(">", i);
|
|
163
|
+
const hex = stream.slice(i + 1, end === -1 ? stream.length : end).replace(/\s/g, "");
|
|
164
|
+
let text = "";
|
|
165
|
+
for (let k = 0; k + 1 < hex.length; k += 2) {
|
|
166
|
+
text += String.fromCharCode(parseInt(hex.slice(k, k + 2), 16));
|
|
167
|
+
}
|
|
168
|
+
current.push({ str: text });
|
|
169
|
+
i = (end === -1 ? stream.length : end) + 1;
|
|
170
|
+
}
|
|
171
|
+
else if (ch === "[") {
|
|
172
|
+
const arr = [];
|
|
173
|
+
current.push(arr);
|
|
174
|
+
stack.push(current);
|
|
175
|
+
current = arr;
|
|
176
|
+
i++;
|
|
177
|
+
}
|
|
178
|
+
else if (ch === "]") {
|
|
179
|
+
current = stack.pop() ?? tokens;
|
|
180
|
+
i++;
|
|
181
|
+
}
|
|
182
|
+
else if (ch === "<" && stream[i + 1] === "<") {
|
|
183
|
+
// Inline dictionary — skip to the matching >>
|
|
184
|
+
let depth = 1;
|
|
185
|
+
let j = i + 2;
|
|
186
|
+
while (j < stream.length && depth > 0) {
|
|
187
|
+
if (stream.startsWith("<<", j)) {
|
|
188
|
+
depth++;
|
|
189
|
+
j += 2;
|
|
190
|
+
}
|
|
191
|
+
else if (stream.startsWith(">>", j)) {
|
|
192
|
+
depth--;
|
|
193
|
+
j += 2;
|
|
194
|
+
}
|
|
195
|
+
else
|
|
196
|
+
j++;
|
|
197
|
+
}
|
|
198
|
+
i = j;
|
|
199
|
+
}
|
|
200
|
+
else if (/[-+.\d]/.test(ch)) {
|
|
201
|
+
const m = /^[-+]?[\d.]+/.exec(stream.slice(i));
|
|
202
|
+
current.push(parseFloat(m[0]));
|
|
203
|
+
i += m[0].length;
|
|
204
|
+
}
|
|
205
|
+
else if (ch === "/") {
|
|
206
|
+
const m = /^\/[^\s()<>[\]{}/%]*/.exec(stream.slice(i));
|
|
207
|
+
current.push(m[0]);
|
|
208
|
+
i += m[0].length;
|
|
209
|
+
}
|
|
210
|
+
else {
|
|
211
|
+
const m = /^[^\s()<>[\]{}/%]+/.exec(stream.slice(i));
|
|
212
|
+
if (!m) {
|
|
213
|
+
i++;
|
|
214
|
+
continue;
|
|
215
|
+
}
|
|
216
|
+
current.push(m[0]);
|
|
217
|
+
i += m[0].length;
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
return tokens;
|
|
221
|
+
}
|
|
222
|
+
/** Replay text operators, collecting each shown string with its position. */
|
|
223
|
+
function extractChunks(stream) {
|
|
224
|
+
const chunks = [];
|
|
225
|
+
const tokens = tokenize(stream);
|
|
226
|
+
let x = 0;
|
|
227
|
+
let y = 0;
|
|
228
|
+
let leading = 0;
|
|
229
|
+
const operands = [];
|
|
230
|
+
const shown = (text) => {
|
|
231
|
+
if (text !== "")
|
|
232
|
+
chunks.push({ x, y, text });
|
|
233
|
+
};
|
|
234
|
+
const num = (v) => (typeof v === "number" ? v : 0);
|
|
235
|
+
for (const token of tokens) {
|
|
236
|
+
if (typeof token !== "string" || token.startsWith("/")) {
|
|
237
|
+
operands.push(token);
|
|
238
|
+
continue;
|
|
239
|
+
}
|
|
240
|
+
switch (token) {
|
|
241
|
+
case "Tm":
|
|
242
|
+
x = num(operands[operands.length - 2]);
|
|
243
|
+
y = num(operands[operands.length - 1]);
|
|
244
|
+
break;
|
|
245
|
+
case "Td":
|
|
246
|
+
x += num(operands[operands.length - 2]);
|
|
247
|
+
y += num(operands[operands.length - 1]);
|
|
248
|
+
break;
|
|
249
|
+
case "TD":
|
|
250
|
+
leading = -num(operands[operands.length - 1]);
|
|
251
|
+
x += num(operands[operands.length - 2]);
|
|
252
|
+
y += num(operands[operands.length - 1]);
|
|
253
|
+
break;
|
|
254
|
+
case "TL":
|
|
255
|
+
leading = num(operands[operands.length - 1]);
|
|
256
|
+
break;
|
|
257
|
+
case "T*":
|
|
258
|
+
y -= leading;
|
|
259
|
+
break;
|
|
260
|
+
case "Tj":
|
|
261
|
+
case "'": {
|
|
262
|
+
if (token === "'")
|
|
263
|
+
y -= leading;
|
|
264
|
+
const s = operands[operands.length - 1];
|
|
265
|
+
if (s && typeof s === "object" && "str" in s)
|
|
266
|
+
shown(s.str);
|
|
267
|
+
break;
|
|
268
|
+
}
|
|
269
|
+
case "TJ": {
|
|
270
|
+
const arr = operands[operands.length - 1];
|
|
271
|
+
if (Array.isArray(arr)) {
|
|
272
|
+
const text = arr
|
|
273
|
+
.filter((el) => Boolean(el && typeof el === "object" && "str" in el))
|
|
274
|
+
.map((el) => el.str)
|
|
275
|
+
.join("");
|
|
276
|
+
shown(text);
|
|
277
|
+
}
|
|
278
|
+
break;
|
|
279
|
+
}
|
|
280
|
+
case "BT":
|
|
281
|
+
x = 0;
|
|
282
|
+
y = 0;
|
|
283
|
+
break;
|
|
284
|
+
}
|
|
285
|
+
operands.length = 0;
|
|
286
|
+
}
|
|
287
|
+
return chunks;
|
|
288
|
+
}
|
|
289
|
+
// ---------------------------------------------------------------------------
|
|
290
|
+
// Row clustering
|
|
291
|
+
// ---------------------------------------------------------------------------
|
|
292
|
+
/**
|
|
293
|
+
* Parse a PDF buffer into headers + rows: text chunks sharing a baseline form
|
|
294
|
+
* a row (top of the page first), ordered left-to-right; the first row is the
|
|
295
|
+
* header. Rows with a different cell count than the header are padded or
|
|
296
|
+
* truncated — the state machine will flag what doesn't validate.
|
|
297
|
+
*/
|
|
298
|
+
function parsePdfTable(buf) {
|
|
299
|
+
const chunks = contentStreams(buf).flatMap(extractChunks);
|
|
300
|
+
if (chunks.length === 0)
|
|
301
|
+
return { headers: [], rows: [] };
|
|
302
|
+
// Cluster by baseline (y), newest tolerance-merge wins.
|
|
303
|
+
const lines = [];
|
|
304
|
+
for (const chunk of chunks) {
|
|
305
|
+
const line = lines.find((l) => Math.abs(l.y - chunk.y) <= ROW_TOLERANCE);
|
|
306
|
+
if (line)
|
|
307
|
+
line.chunks.push(chunk);
|
|
308
|
+
else
|
|
309
|
+
lines.push({ y: chunk.y, chunks: [chunk] });
|
|
310
|
+
}
|
|
311
|
+
// Page order: highest y first (PDF's origin is bottom-left).
|
|
312
|
+
lines.sort((a, b) => b.y - a.y);
|
|
313
|
+
const grid = lines.map((line) => line.chunks
|
|
314
|
+
.sort((a, b) => a.x - b.x)
|
|
315
|
+
.map((c) => c.text.trim()));
|
|
316
|
+
const [headers = [], ...rows] = grid;
|
|
317
|
+
return {
|
|
318
|
+
headers,
|
|
319
|
+
rows: rows.map((row) => {
|
|
320
|
+
const cells = row.slice(0, headers.length);
|
|
321
|
+
while (cells.length < headers.length)
|
|
322
|
+
cells.push("");
|
|
323
|
+
return cells;
|
|
324
|
+
}),
|
|
325
|
+
};
|
|
326
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The pure core of `rastack import`: map a tabular source (CSV/XLSX/PDF —
|
|
3
|
+
* already parsed to headers + string rows) onto a resource's fields, then run
|
|
4
|
+
* every row through the resource's **constraint state machine**
|
|
5
|
+
* (`rastack/validate`) with a dataset-level context, so `unique` sees
|
|
6
|
+
* duplicates across the file and `resolved` checks foreign keys against
|
|
7
|
+
* supplied related ids.
|
|
8
|
+
*
|
|
9
|
+
* Everything here is data-in/data-out (no filesystem), tested in
|
|
10
|
+
* `tools/test/import.spec.ts`; the parsers and the CLI are the IO shell.
|
|
11
|
+
*/
|
|
12
|
+
import { type FieldSpec, type FieldState, type RecordRun, type SpecRelation } from "../validate";
|
|
13
|
+
/** A parsed tabular source: one header row + string cell rows. */
|
|
14
|
+
export interface TabularData {
|
|
15
|
+
headers: string[];
|
|
16
|
+
rows: string[][];
|
|
17
|
+
}
|
|
18
|
+
/** How one source column mapped onto the resource (or didn't). */
|
|
19
|
+
export interface ColumnMapping {
|
|
20
|
+
header: string;
|
|
21
|
+
field: string | null;
|
|
22
|
+
}
|
|
23
|
+
/** One constraint failure, located by row (0-based, excluding the header). */
|
|
24
|
+
export interface ImportIssue {
|
|
25
|
+
row: number;
|
|
26
|
+
field: string;
|
|
27
|
+
gate: FieldState;
|
|
28
|
+
constraint: string;
|
|
29
|
+
message: string;
|
|
30
|
+
}
|
|
31
|
+
/** The full validation report for one imported file. */
|
|
32
|
+
export interface ImportReport {
|
|
33
|
+
total: number;
|
|
34
|
+
valid: number;
|
|
35
|
+
invalid: number;
|
|
36
|
+
/** Source-column → field mapping (null = column ignored). */
|
|
37
|
+
columns: ColumnMapping[];
|
|
38
|
+
/** Resource fields no source column covered (they'll fail `present` if required). */
|
|
39
|
+
missingFields: string[];
|
|
40
|
+
/** Every constraint failure, in row order. */
|
|
41
|
+
issues: ImportIssue[];
|
|
42
|
+
/** The coerced records of the *valid* rows — ready to POST or seed. */
|
|
43
|
+
records: Record<string, unknown>[];
|
|
44
|
+
/** Per-row machine runs, index-aligned with the source rows. */
|
|
45
|
+
rows: RecordRun[];
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* Map source headers onto field specs by normalised name (case, `_`, `-` and
|
|
49
|
+
* spaces are insignificant). Unmatched headers are carried as ignored columns;
|
|
50
|
+
* unmatched fields are reported so a missing required column reads as a
|
|
51
|
+
* mapping problem, not just N× `present` failures.
|
|
52
|
+
*/
|
|
53
|
+
export declare function mapHeaders(headers: string[], specs: FieldSpec[]): {
|
|
54
|
+
columns: ColumnMapping[];
|
|
55
|
+
missingFields: string[];
|
|
56
|
+
};
|
|
57
|
+
/** Optional dataset-level inputs for FK resolution. */
|
|
58
|
+
export interface ValidateDatasetOptions {
|
|
59
|
+
/** Known ids per `"app.model"` — engages the `resolved` gate. */
|
|
60
|
+
relatedIds?: Record<string, Set<string>>;
|
|
61
|
+
/** Custom FK resolver (wins over `relatedIds`). */
|
|
62
|
+
resolveRelation?: (relation: SpecRelation, value: unknown) => boolean;
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Run a whole tabular source through a resource's record machine. Each row is
|
|
66
|
+
* projected onto the mapped fields and validated; the dataset context makes
|
|
67
|
+
* the `unique` gate reject in-file duplicates and (when related ids are
|
|
68
|
+
* given) the `resolved` gate reject dangling foreign keys.
|
|
69
|
+
*/
|
|
70
|
+
export declare function validateDataset(specs: FieldSpec[], data: TabularData, options?: ValidateDatasetOptions): ImportReport;
|
|
71
|
+
/**
|
|
72
|
+
* Pull the id set out of an already-parsed related table (its `id` column,
|
|
73
|
+
* else its first column) — the ids `validateDataset` resolves foreign keys
|
|
74
|
+
* against when importing a dependent table alongside its target.
|
|
75
|
+
*/
|
|
76
|
+
export declare function idsFromTabular(data: TabularData): Set<string>;
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* The pure core of `rastack import`: map a tabular source (CSV/XLSX/PDF —
|
|
4
|
+
* already parsed to headers + string rows) onto a resource's fields, then run
|
|
5
|
+
* every row through the resource's **constraint state machine**
|
|
6
|
+
* (`rastack/validate`) with a dataset-level context, so `unique` sees
|
|
7
|
+
* duplicates across the file and `resolved` checks foreign keys against
|
|
8
|
+
* supplied related ids.
|
|
9
|
+
*
|
|
10
|
+
* Everything here is data-in/data-out (no filesystem), tested in
|
|
11
|
+
* `tools/test/import.spec.ts`; the parsers and the CLI are the IO shell.
|
|
12
|
+
*/
|
|
13
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
|
+
exports.mapHeaders = mapHeaders;
|
|
15
|
+
exports.validateDataset = validateDataset;
|
|
16
|
+
exports.idsFromTabular = idsFromTabular;
|
|
17
|
+
const validate_1 = require("../validate");
|
|
18
|
+
/** `"Airport Code"`, `airport_code`, `airportCode` → `airportcode`. */
|
|
19
|
+
function normalise(name) {
|
|
20
|
+
return name.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* Map source headers onto field specs by normalised name (case, `_`, `-` and
|
|
24
|
+
* spaces are insignificant). Unmatched headers are carried as ignored columns;
|
|
25
|
+
* unmatched fields are reported so a missing required column reads as a
|
|
26
|
+
* mapping problem, not just N× `present` failures.
|
|
27
|
+
*/
|
|
28
|
+
function mapHeaders(headers, specs) {
|
|
29
|
+
const byNormalised = new Map(specs.map((s) => [normalise(s.name), s.name]));
|
|
30
|
+
const used = new Set();
|
|
31
|
+
const columns = headers.map((header) => {
|
|
32
|
+
const field = byNormalised.get(normalise(header)) ?? null;
|
|
33
|
+
if (field)
|
|
34
|
+
used.add(field);
|
|
35
|
+
return { header, field };
|
|
36
|
+
});
|
|
37
|
+
return {
|
|
38
|
+
columns,
|
|
39
|
+
missingFields: specs.map((s) => s.name).filter((n) => !used.has(n)),
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Run a whole tabular source through a resource's record machine. Each row is
|
|
44
|
+
* projected onto the mapped fields and validated; the dataset context makes
|
|
45
|
+
* the `unique` gate reject in-file duplicates and (when related ids are
|
|
46
|
+
* given) the `resolved` gate reject dangling foreign keys.
|
|
47
|
+
*/
|
|
48
|
+
function validateDataset(specs, data, options = {}) {
|
|
49
|
+
const { columns, missingFields } = mapHeaders(data.headers, specs);
|
|
50
|
+
const machine = (0, validate_1.compileRecordMachine)(specs);
|
|
51
|
+
const ctx = (0, validate_1.createDatasetContext)(machine, options);
|
|
52
|
+
const rows = [];
|
|
53
|
+
const issues = [];
|
|
54
|
+
const records = [];
|
|
55
|
+
data.rows.forEach((cells, index) => {
|
|
56
|
+
const record = {};
|
|
57
|
+
columns.forEach((col, i) => {
|
|
58
|
+
if (col.field !== null)
|
|
59
|
+
record[col.field] = cells[i] ?? "";
|
|
60
|
+
});
|
|
61
|
+
const run = (0, validate_1.runRecordMachine)(machine, record, ctx);
|
|
62
|
+
rows.push(run);
|
|
63
|
+
if (run.state === "valid") {
|
|
64
|
+
records.push(run.values);
|
|
65
|
+
}
|
|
66
|
+
else {
|
|
67
|
+
for (const error of run.errors)
|
|
68
|
+
issues.push({ row: index, ...error });
|
|
69
|
+
}
|
|
70
|
+
});
|
|
71
|
+
return {
|
|
72
|
+
total: data.rows.length,
|
|
73
|
+
valid: records.length,
|
|
74
|
+
invalid: data.rows.length - records.length,
|
|
75
|
+
columns,
|
|
76
|
+
missingFields,
|
|
77
|
+
issues,
|
|
78
|
+
records,
|
|
79
|
+
rows,
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* Pull the id set out of an already-parsed related table (its `id` column,
|
|
84
|
+
* else its first column) — the ids `validateDataset` resolves foreign keys
|
|
85
|
+
* against when importing a dependent table alongside its target.
|
|
86
|
+
*/
|
|
87
|
+
function idsFromTabular(data) {
|
|
88
|
+
let idColumn = data.headers.findIndex((h) => normalise(h) === "id");
|
|
89
|
+
if (idColumn === -1)
|
|
90
|
+
idColumn = 0;
|
|
91
|
+
const ids = new Set();
|
|
92
|
+
for (const row of data.rows) {
|
|
93
|
+
const value = (row[idColumn] ?? "").trim();
|
|
94
|
+
if (value !== "")
|
|
95
|
+
ids.add(value);
|
|
96
|
+
}
|
|
97
|
+
return ids;
|
|
98
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A tiny, dependency-free XLSX reader for `rastack import`.
|
|
3
|
+
*
|
|
4
|
+
* Why hand-roll it: the CLI stays dependency-light (the same reasoning as the
|
|
5
|
+
* hand-rolled ZIP *writer* in `deploy/zip.ts`), and an `.xlsx` workbook is
|
|
6
|
+
* just a ZIP of small XML parts — the reader below understands exactly the
|
|
7
|
+
* subset a tabular import needs: the ZIP central directory (STORE and DEFLATE
|
|
8
|
+
* entries, inflated via node's zlib), `xl/sharedStrings.xml`, and the cells of
|
|
9
|
+
* the first worksheet. Formulas are read through their cached `<v>` results;
|
|
10
|
+
* merged cells, styles and charts are ignored — an import wants values.
|
|
11
|
+
*/
|
|
12
|
+
import type { TabularData } from "./tabular";
|
|
13
|
+
/**
|
|
14
|
+
* Parse an `.xlsx` workbook buffer into headers + rows (first worksheet, first
|
|
15
|
+
* row as the header). Cell values come back as strings — the constraint state
|
|
16
|
+
* machine's `typed` gate owns coercion, same as for CSV.
|
|
17
|
+
*/
|
|
18
|
+
export declare function parseXlsx(buf: Buffer): TabularData;
|