tranfi 0.2.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +215 -55
- package/csrc/pipeline.c +1 -1
- package/csrc/size_utils.c +4 -0
- package/package.json +2 -2
- package/src/cli.js +75 -32
- package/wasm/tranfi_core.js +0 -0
package/README.md
CHANGED
|
@@ -1,14 +1,11 @@
|
|
|
1
1
|
# tranfi (Node.js / WASM)
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
> **Unreleased main:** The prepared-transform API documented below targets
|
|
10
|
-
> Tranfi 0.2. Current npm 0.1.x installs do not include it; build this branch
|
|
11
|
-
> from source until 0.2 is published.
|
|
3
|
+
Tranfi transforms CSV, JSON Lines and plain text as streams. In Node.js it runs
|
|
4
|
+
on a native C core; in the browser it runs on WebAssembly. Data flows through in
|
|
5
|
+
chunks, so most pipelines use a small, fixed amount of memory however large the
|
|
6
|
+
input is. Operations that need the whole input, such as `sort`, must be allowed
|
|
7
|
+
explicitly. Tranfi works on byte streams; it is not an in-memory DataFrame
|
|
8
|
+
library.
|
|
12
9
|
|
|
13
10
|
Save this example as `quickstart.mjs`:
|
|
14
11
|
|
|
@@ -35,6 +32,7 @@ console.log(result.outputText)
|
|
|
35
32
|
|
|
36
33
|
Or use the pipe DSL for one-liners:
|
|
37
34
|
|
|
35
|
+
<!-- readme-test: js-api -->
|
|
38
36
|
```js
|
|
39
37
|
const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
|
|
40
38
|
.run({ inputFile: 'data.csv' })
|
|
@@ -46,17 +44,19 @@ const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | cs
|
|
|
46
44
|
npm install tranfi
|
|
47
45
|
```
|
|
48
46
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
47
|
+
Installation compiles Tranfi's C core as a native Node.js addon, so you need a C
|
|
48
|
+
compiler. If that compile fails, the install fails.
|
|
49
|
+
|
|
50
|
+
In the browser, import `tranfi/wasm`: it provides pipelines and prepared
|
|
51
|
+
transforms without the native addon. To install only the WebAssembly build, run
|
|
52
|
+
`TRANFI_SKIP_NATIVE_BUILD=1 npm install tranfi`. The root `tranfi` entry then
|
|
53
|
+
runs pipelines through WebAssembly, but its prepared-transform classes need the
|
|
54
|
+
native addon, so use `tranfi/wasm` for those.
|
|
55
55
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
56
|
+
| | Linux | Windows | macOS |
|
|
57
|
+
|---|---|---|---|
|
|
58
|
+
| Native addon | Tested | Not tested | Not tested |
|
|
59
|
+
| WebAssembly (`tranfi/wasm`) | Tested in Node.js and Chromium | Tested in Node.js | Not tested |
|
|
60
60
|
|
|
61
61
|
Use `pipeline(...)` for byte-stream ETL. The separate
|
|
62
62
|
`TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
|
|
@@ -70,7 +70,7 @@ Installing the package also provides the `tranfi` command:
|
|
|
70
70
|
|
|
71
71
|
```bash
|
|
72
72
|
# Via npx (no install)
|
|
73
|
-
|
|
73
|
+
printf 'name,age\nAlice,30\nBob,25\n' | npx tranfi -q 'csv | filter "age > 25" | csv'
|
|
74
74
|
|
|
75
75
|
# Or install globally
|
|
76
76
|
npm i -g tranfi
|
|
@@ -79,7 +79,11 @@ tranfi profile < data.csv
|
|
|
79
79
|
tranfi -R # list recipes
|
|
80
80
|
```
|
|
81
81
|
|
|
82
|
-
Run `tranfi -h` for
|
|
82
|
+
Run `tranfi -h` for the npm CLI options. Use `--allow-blocking` for known-small
|
|
83
|
+
full-input operations, `--memory max:64MB` for a memory policy, and
|
|
84
|
+
`--spill-dir DIR` for an existing private spill directory. `--stats-json FILE`
|
|
85
|
+
writes the stats channel; `--target json|sql` compiles without executing.
|
|
86
|
+
The detailed `--explain` report requires the standalone C CLI built from source.
|
|
83
87
|
|
|
84
88
|
## Quick start
|
|
85
89
|
|
|
@@ -87,6 +91,7 @@ Run `tranfi -h` for all options.
|
|
|
87
91
|
|
|
88
92
|
**Builder API** -- composable, structured, and IDE-friendly:
|
|
89
93
|
|
|
94
|
+
<!-- readme-test: js-api -->
|
|
90
95
|
```js
|
|
91
96
|
const p = pipeline([
|
|
92
97
|
codec.csv(),
|
|
@@ -100,6 +105,7 @@ const result = await p.run({ inputFile: 'students.csv' })
|
|
|
100
105
|
|
|
101
106
|
**DSL strings** -- compact, suitable for CLI-like use:
|
|
102
107
|
|
|
108
|
+
<!-- readme-test: js-api -->
|
|
103
109
|
```js
|
|
104
110
|
const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
|
|
105
111
|
const result = await p.run({ inputFile: 'students.csv' })
|
|
@@ -111,12 +117,13 @@ Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate
|
|
|
111
117
|
|
|
112
118
|
### Running pipelines
|
|
113
119
|
|
|
120
|
+
<!-- readme-test: js-pipeline -->
|
|
114
121
|
```js
|
|
115
122
|
// From string or Buffer
|
|
116
123
|
const result = await p.run({ input: 'name,age\nAlice,30\n' })
|
|
117
124
|
|
|
118
125
|
// From file (streamed in 64 KB chunks)
|
|
119
|
-
const
|
|
126
|
+
const fileResult = await p.run({ inputFile: 'data.csv' })
|
|
120
127
|
|
|
121
128
|
// Access results
|
|
122
129
|
result.output // Buffer
|
|
@@ -129,8 +136,9 @@ result.samples // Buffer (sample channel)
|
|
|
129
136
|
|
|
130
137
|
For large outputs, drain chunks instead of collecting `result.output`:
|
|
131
138
|
|
|
139
|
+
<!-- readme-test: js-pipeline -->
|
|
132
140
|
```js
|
|
133
|
-
|
|
141
|
+
import { createReadStream, createWriteStream } from 'node:fs'
|
|
134
142
|
|
|
135
143
|
// Write to a Node Writable and wait for `drain` when the sink applies backpressure.
|
|
136
144
|
const out = createWriteStream("out.csv")
|
|
@@ -159,12 +167,13 @@ for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
|
|
|
159
167
|
}
|
|
160
168
|
```
|
|
161
169
|
|
|
162
|
-
`sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file.
|
|
170
|
+
`sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. Repeated headers are kept by default. Use `codec.csv({ skipRepeatedHeader: true })` to skip a later file's first non-comment record when it exactly matches the original header; matching rows inside a file remain data.
|
|
163
171
|
|
|
164
172
|
Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
|
|
165
173
|
|
|
166
174
|
For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
|
|
167
175
|
|
|
176
|
+
<!-- readme-test: browser -->
|
|
168
177
|
```js
|
|
169
178
|
// main thread
|
|
170
179
|
import { createWorkerClient } from 'tranfi/wasm/worker'
|
|
@@ -189,21 +198,16 @@ The bundled Tranfi app runner is also preview-bounded by default: file chunks ar
|
|
|
189
198
|
|
|
190
199
|
### Prepared reusable transforms
|
|
191
200
|
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
The current slice accepts declared `float32`/`float64` columns. It supports numeric none/zero/constant/mean/exact-median imputation and none/standard/min-max normalization. Declared categorical columns support mode imputation with `allMissing: 'error' | 'zero'` or `impute.op: 'none'`; encoders discover finite typed categories, while the no-imputation/no-encoding combination needs no learned dictionary. A column may instead provide both branches with `kind.op: 'infer'`, `kind.rule: 'finite-integer-cardinality-v1'`, and `kind.maxCategories >= 2`: missing/NaN values are ignored, `2..maxCategories` distinct finite integers resolve categorical, while zero/one distinct value, any noninteger, or the next distinct value resolves numeric. `impute.op: 'none'` with `encode.op: 'none'` passes every finite value through and emits canonical qNaN for missing input. Mode with `encode.op: 'none'` retains its learned dictionary and rejects unseen finite values with code `108`. `encode.op: 'label'` freezes zero-based sorted ordinals and supports `unknown: 'error' | 'sentinel' | 'other'`; `encode.op: 'onehot'` emits source-ordered category blocks and supports `unknown: 'error' | 'all_zero' | 'other'`. Without categorical imputation, missing label/one-hot input follows that encoder's unknown policy. The label sentinel must be a safe integer outside the learned ordinal range; label/one-hot `other` appends the reserved ordinal/field after known categories. Generated output IDs/names and one-hot category metadata are deterministic, and collisions fail with code `102`. Exact median obeys the configured allocation and resident-state limits; inference, categorical discovery, and output expansion obey category, output-width, allocation, and resident-state limits. Limit failures use resource code `104`. Prepared-transform host-policy and spill fields are reserved but not implemented; nonempty use fails with unsupported-runtime code `113` instead of being silently ignored.
|
|
197
|
-
|
|
198
|
-
Declared categorical columns also accept a fixed, nonempty `encode.categories` array of finite, sorted, unique tags (`{ "t": "f64", "v": "4000000000000000" }` represents 2). Tags must match the input dtype; negative zero is represented as positive zero. Fixed dictionaries retain unobserved categories. Unknown training values follow the encoder policy and never vote for mode; an all-missing zero fallback must exist in the dictionary. Fixed encoding without imputation can finalize without training rows. Fixed dictionaries with kind inference, string categories, and categorical constant imputation remain unsupported.
|
|
201
|
+
Learn imputation, scaling or category mappings from reference data, then apply
|
|
202
|
+
the same plan to new batches. Analyze the reference batches, finalize the plan,
|
|
203
|
+
and apply it. Replaying the reference data through that plan gives a second-pass
|
|
204
|
+
`fit_transform` workflow.
|
|
199
205
|
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
`limits`, `cancelToken`, `hostPolicy`, and `spillDir`. The host/spill fields are
|
|
203
|
-
reserved as described above, and cancellation spellings are not interchangeable.
|
|
206
|
+
Inputs are typed `float32`/`float64` columns. This example learns mean imputation
|
|
207
|
+
and standard scaling, applies them to the reference batch, and exports the plan:
|
|
204
208
|
|
|
205
209
|
```js
|
|
206
|
-
|
|
210
|
+
import tf from 'tranfi'
|
|
207
211
|
|
|
208
212
|
const recipeSpec = {
|
|
209
213
|
format: 'tranfi.transform-recipe',
|
|
@@ -245,8 +249,85 @@ analyzer.close()
|
|
|
245
249
|
recipe.close()
|
|
246
250
|
```
|
|
247
251
|
|
|
252
|
+
<details markdown="1">
|
|
253
|
+
<summary>Available methods, missing values and kind inference</summary>
|
|
254
|
+
|
|
255
|
+
Recipe fields use the same names in all bindings.
|
|
256
|
+
|
|
257
|
+
| Recipe field | Choices / behavior |
|
|
258
|
+
|--------------|--------------------|
|
|
259
|
+
| Numeric `impute.op` | `none`, `zero`, `constant`, `mean`, exact `median` |
|
|
260
|
+
| Numeric `normalize.op` | `none`, `standard`, `minmax` |
|
|
261
|
+
| Categorical `impute.op` | `none` or mode; mode supports `allMissing` of `error` or `zero` |
|
|
262
|
+
| Categorical `encode.op` | `none`, `label`, `onehot` |
|
|
263
|
+
| Label unknown policy | `error`, `sentinel`, `other` |
|
|
264
|
+
| One-hot unknown policy | `error`, `all_zero`, `other` |
|
|
265
|
+
|
|
266
|
+
**No imputation or encoding:** finite values pass through; missing input becomes
|
|
267
|
+
canonical qNaN. This combination needs no learned category dictionary.
|
|
268
|
+
Mode imputation with no encoding retains a dictionary and rejects unseen finite
|
|
269
|
+
values with error `108`.
|
|
270
|
+
|
|
271
|
+
**Encoding:** label ordinals are zero-based and sorted; one-hot blocks follow
|
|
272
|
+
source order. A label sentinel must be a safe integer outside the known ordinal
|
|
273
|
+
range. `other` appends an ordinal or field after known categories. Without
|
|
274
|
+
imputation, missing input follows the encoder's unknown policy.
|
|
275
|
+
|
|
276
|
+
Generated IDs, names and category metadata are deterministic. Collisions fail
|
|
277
|
+
with error `102`.
|
|
278
|
+
|
|
279
|
+
**Kind inference:** configure both numeric and categorical branches, set
|
|
280
|
+
`kind.op` to `infer`, `kind.rule` to `finite-integer-cardinality-v1`, and
|
|
281
|
+
`kind.maxCategories` to at least `2`. Missing/NaN values are ignored.
|
|
282
|
+
|
|
283
|
+
| Observed reference values | Selected branch |
|
|
284
|
+
|---------------------------|-----------------|
|
|
285
|
+
| `2..maxCategories` distinct finite integers | Categorical |
|
|
286
|
+
| Zero or one distinct value, any noninteger, or more than `maxCategories` | Numeric |
|
|
287
|
+
|
|
288
|
+
</details>
|
|
289
|
+
|
|
290
|
+
<details markdown="1">
|
|
291
|
+
<summary>Fixed category dictionaries</summary>
|
|
292
|
+
|
|
293
|
+
Set `encode.categories` to a nonempty, sorted, unique array of finite typed tags.
|
|
294
|
+
For example, `{ "t": "f64", "v": "4000000000000000" }` represents `2`.
|
|
295
|
+
Tags must match the input dtype; represent negative zero as positive zero.
|
|
296
|
+
|
|
297
|
+
- Unobserved categories stay in the dictionary.
|
|
298
|
+
- Unknown training values follow the encoder policy and never vote for mode.
|
|
299
|
+
- An all-missing zero fallback must exist in the dictionary.
|
|
300
|
+
- Fixed encoding without imputation can finalize without training rows.
|
|
301
|
+
|
|
302
|
+
String categories, categorical constant imputation and fixed dictionaries
|
|
303
|
+
combined with kind inference are not supported.
|
|
304
|
+
|
|
305
|
+
</details>
|
|
306
|
+
|
|
307
|
+
<details markdown="1">
|
|
308
|
+
<summary>Limits, errors and cancellation</summary>
|
|
309
|
+
|
|
310
|
+
Exact median obeys allocation and resident-state limits. Inference, category
|
|
311
|
+
discovery and output expansion also enforce category/output-width limits.
|
|
312
|
+
Resource-limit failures use error `104`.
|
|
313
|
+
|
|
314
|
+
Host-policy and spill settings are reserved. A non-null host policy or nonempty
|
|
315
|
+
spill path fails with unsupported-runtime error `113`.
|
|
316
|
+
|
|
317
|
+
Runtime options reject unknown fields:
|
|
318
|
+
|
|
319
|
+
| Runtime | Options |
|
|
320
|
+
|---------|---------|
|
|
321
|
+
| Native Node | `limits`, `cancelFlag`, `hostPolicy`, `spillDir` |
|
|
322
|
+
| Standalone WASM | `limits`, `cancelToken`, `hostPolicy`, `spillDir` |
|
|
323
|
+
|
|
324
|
+
Cancellation spellings are not interchangeable. WASM tokens created with
|
|
325
|
+
`createTransformCancelToken()` expose a read-only `requested` boolean for host
|
|
326
|
+
conversion loops; reading a closed token fails.
|
|
327
|
+
|
|
248
328
|
`tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
|
|
249
329
|
|
|
330
|
+
<!-- readme-test: browser -->
|
|
250
331
|
```js
|
|
251
332
|
const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
|
|
252
333
|
const makeWorker = () => new Worker(workerUrl, { type: 'module' })
|
|
@@ -258,29 +339,30 @@ const client = createWorkerClient(makeWorker(), {
|
|
|
258
339
|
|
|
259
340
|
All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
|
|
260
341
|
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
|
|
342
|
+
</details>
|
|
264
343
|
|
|
265
|
-
|
|
266
|
-
|--------|-------------|
|
|
267
|
-
| `codec.csv({ delimiter, header, batchSize, repair, mode, strict, maxErrorBytes, maxRecordBytes, maxColumns, nulls, quotedNulls, skip, nMax, maxRows, comment, trimWs, skipEmptyRows, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | CSV decoder. `mode: 'strict'` fails on field-count mismatches; `repair: true` / `mode: 'repair'` emits repair diagnostics; `maxRecordBytes` bounds buffered records; `nulls: ['NA']` adds null sentinels; `skip: 2`, `nMax: 100`, `comment: '#'`, `trimWs: false`, and `skipEmptyRows: true` control row-local parsing; repair audit/raw diagnostics support privacy controls with pseudo-column `raw` |
|
|
268
|
-
| `codec.csvEncode({ delimiter })` | CSV encoder |
|
|
269
|
-
| `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | JSON Lines decoder. `onError` is `skip`, `fail`, `warn`, or `quarantine` |
|
|
270
|
-
| `codec.jsonlEncode()` | JSON Lines encoder |
|
|
271
|
-
| `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | Line-oriented text decoder (single `_line` column) |
|
|
272
|
-
| `codec.textEncode()` | Text encoder |
|
|
273
|
-
| `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
|
|
344
|
+
## Codecs
|
|
274
345
|
|
|
275
|
-
|
|
346
|
+
Start with a decoder and finish with an encoder. Input and output formats can differ.
|
|
276
347
|
|
|
277
|
-
|
|
348
|
+
| Format | Decoder | Encoder |
|
|
349
|
+
|--------|---------|---------|
|
|
350
|
+
| CSV | `codec.csv(options)` | `codec.csvEncode({ delimiter })` |
|
|
351
|
+
| JSON Lines | `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | `codec.jsonlEncode()` |
|
|
352
|
+
| Text (one `_line` column) | `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | `codec.textEncode()` |
|
|
353
|
+
| Markdown table | — | `codec.tableEncode({ maxWidth, maxRows })` |
|
|
278
354
|
|
|
279
|
-
|
|
355
|
+
**CSV defaults to permissive field-count handling.** Choose `mode: 'strict'`
|
|
356
|
+
(or `strict: true`) to reject width mismatches. Choose `repair: true`
|
|
357
|
+
(or `mode: 'repair'`) to pad/truncate rows and collect diagnostics in `result.errors`.
|
|
280
358
|
|
|
359
|
+
**Malformed JSONL is skipped by default.** Set `onError` to `fail` to stop,
|
|
360
|
+
`warn` to keep valid rows with diagnostics, or `quarantine` to route bad records
|
|
361
|
+
to error diagnostics. Read those diagnostics from `result.errors`.
|
|
281
362
|
|
|
282
363
|
Cross-codec pipelines work naturally:
|
|
283
364
|
|
|
365
|
+
<!-- readme-test: js-api -->
|
|
284
366
|
```js
|
|
285
367
|
// CSV in, JSONL out
|
|
286
368
|
pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
|
|
@@ -289,6 +371,57 @@ pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
|
|
|
289
371
|
pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
|
|
290
372
|
```
|
|
291
373
|
|
|
374
|
+
<details markdown="1">
|
|
375
|
+
<summary>CSV options for nulls, row limits and whitespace</summary>
|
|
376
|
+
|
|
377
|
+
| Need | Option | Behavior |
|
|
378
|
+
|------|--------|----------|
|
|
379
|
+
| Field separator / header | `delimiter`, `header` | Configure CSV decoding; encoder accepts `delimiter` |
|
|
380
|
+
| Add null markers | `nulls: ['NA', 'NULL']` | Adds to default unquoted-empty-field null handling |
|
|
381
|
+
| Preserve quoted markers | `quotedNulls: false` | Keeps quoted `"NA"` and `""` as strings |
|
|
382
|
+
| Skip a preamble | `skip: 2` | Before schema/header discovery; comments are handled afterward |
|
|
383
|
+
| Limit rows | `nMax: 100` or `maxRows: 100` | One shared counter after skip/comment/header handling |
|
|
384
|
+
| Keep only the schema | `nMax: 0` | Preserves a header-only batch |
|
|
385
|
+
| Comments | `comment: '#'` | Removes text after unquoted markers and skips comment-only rows; quoted markers remain |
|
|
386
|
+
| Preserve spaces/tabs | `trimWs: false` | Disables default trimming of unquoted values |
|
|
387
|
+
| Drop blank rows | `skipEmptyRows: true` | Otherwise blank physical rows after the header become all-null rows |
|
|
388
|
+
| Read CSV shards | `skipRepeatedHeader: true` | Skips matching headers only at explicit file boundaries |
|
|
389
|
+
|
|
390
|
+
</details>
|
|
391
|
+
|
|
392
|
+
<details markdown="1">
|
|
393
|
+
<summary>Decoder limits and repair diagnostics</summary>
|
|
394
|
+
|
|
395
|
+
All decoder size options are checked integers:
|
|
396
|
+
|
|
397
|
+
| Option | Range | Behavior |
|
|
398
|
+
|--------|-------|----------|
|
|
399
|
+
| `batchSize` | `1..65536` | Rows per batch |
|
|
400
|
+
| `maxErrorBytes` | `0..67108864` | Bounds raw diagnostic previews |
|
|
401
|
+
| `maxRecordBytes` | `0..1073741824` | Default `67108864` bytes (64 MiB); `0` disables the record guard |
|
|
402
|
+
| `maxColumns` (CSV) | `1..65536` | Default `8192`; overflow fails with `csv_too_many_columns` |
|
|
403
|
+
|
|
404
|
+
Record limits apply before a newline is seen, including CSV, JSONL and text.
|
|
405
|
+
CSV record overflow fails with `csv_record_too_large`.
|
|
406
|
+
|
|
407
|
+
Repair audit/raw payloads accept `auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes`. The raw preview is exposed to
|
|
408
|
+
those controls as pseudo-column `raw`. CSV repair also accepts
|
|
409
|
+
`audit, auditLimit`.
|
|
410
|
+
|
|
411
|
+
</details>
|
|
412
|
+
|
|
413
|
+
<details markdown="1">
|
|
414
|
+
<summary>JSONL record and schema rules</summary>
|
|
415
|
+
|
|
416
|
+
Each record must be one complete UTF-8 JSON object. Invalid number syntax,
|
|
417
|
+
trailing content, embedded NULs and numbers outside finite float64 range follow
|
|
418
|
+
the configured malformed-record policy. The encoder rejects nonfinite values.
|
|
419
|
+
|
|
420
|
+
The first valid record determines schema order. Repeated keys use their first
|
|
421
|
+
value. `maxErrorBytes` bounds diagnostic previews, and `maxRecordBytes` caps the current line.
|
|
422
|
+
|
|
423
|
+
</details>
|
|
424
|
+
|
|
292
425
|
## Operators
|
|
293
426
|
|
|
294
427
|
### Row filtering
|
|
@@ -351,6 +484,7 @@ Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selecte
|
|
|
351
484
|
| `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
|
|
352
485
|
| `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
|
|
353
486
|
|
|
487
|
+
<!-- readme-test: js-api -->
|
|
354
488
|
```js
|
|
355
489
|
// Group aggregation
|
|
356
490
|
ops.groupAgg(['city'], [
|
|
@@ -409,6 +543,7 @@ ops.groupAgg(['city'], [
|
|
|
409
543
|
|
|
410
544
|
Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
|
|
411
545
|
|
|
546
|
+
<!-- readme-test: js-api -->
|
|
412
547
|
```js
|
|
413
548
|
ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
|
|
414
549
|
ops.derive({
|
|
@@ -432,6 +567,7 @@ ops.derive({
|
|
|
432
567
|
|
|
433
568
|
Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
|
|
434
569
|
|
|
570
|
+
<!-- readme-test: js-api -->
|
|
435
571
|
```js
|
|
436
572
|
ops.derive({
|
|
437
573
|
year: expr("year(col('date'))"),
|
|
@@ -443,9 +579,10 @@ ops.derive({
|
|
|
443
579
|
|
|
444
580
|
Built-in named pipelines for common tasks. Use by name:
|
|
445
581
|
|
|
582
|
+
<!-- readme-test: js-api -->
|
|
446
583
|
```js
|
|
447
584
|
const result = await pipeline('preview').run({ inputFile: 'data.csv' })
|
|
448
|
-
const
|
|
585
|
+
const frequencies = await pipeline('freq').run({ inputFile: 'data.csv' })
|
|
449
586
|
```
|
|
450
587
|
|
|
451
588
|
| Recipe | Pipeline | Description |
|
|
@@ -488,12 +625,14 @@ for (const r of await recipes()) {
|
|
|
488
625
|
|
|
489
626
|
Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
|
|
490
627
|
|
|
628
|
+
<!-- readme-test: js-api -->
|
|
491
629
|
```js
|
|
492
630
|
await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
|
|
493
631
|
```
|
|
494
632
|
|
|
495
633
|
For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
|
|
496
634
|
|
|
635
|
+
<!-- readme-test: js-api -->
|
|
497
636
|
```js
|
|
498
637
|
await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
|
|
499
638
|
```
|
|
@@ -510,6 +649,7 @@ Run pipelines on DuckDB instead of the native C streaming core. The DSL is trans
|
|
|
510
649
|
npm install duckdb
|
|
511
650
|
```
|
|
512
651
|
|
|
652
|
+
<!-- readme-test: js-data -->
|
|
513
653
|
```js
|
|
514
654
|
import { pipeline, compileToSql } from 'tranfi'
|
|
515
655
|
|
|
@@ -526,6 +666,7 @@ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
|
|
|
526
666
|
|
|
527
667
|
Generate SQL directly from DSL strings:
|
|
528
668
|
|
|
669
|
+
<!-- readme-test: js-sql -->
|
|
529
670
|
```js
|
|
530
671
|
const sql = await compileToSql(
|
|
531
672
|
'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
|
|
@@ -546,6 +687,7 @@ their own compatibility tests and SQL-generation rules.
|
|
|
546
687
|
|
|
547
688
|
In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
|
|
548
689
|
|
|
690
|
+
<!-- readme-test: browser -->
|
|
549
691
|
```js
|
|
550
692
|
import createTranfi from 'tranfi/wasm'
|
|
551
693
|
import * as duckdb from '@duckdb/duckdb-wasm'
|
|
@@ -571,7 +713,7 @@ console.log(result.rows) // Array of row objects
|
|
|
571
713
|
### DSL compilation
|
|
572
714
|
|
|
573
715
|
```js
|
|
574
|
-
import { compileDsl, saveRecipe, loadRecipe } from 'tranfi'
|
|
716
|
+
import { compileDsl, saveRecipe, loadRecipe, codec, ops } from 'tranfi'
|
|
575
717
|
|
|
576
718
|
// Compile DSL to JSON plan
|
|
577
719
|
const json = await compileDsl('csv | filter "col(age) > 25" | sort -age | csv')
|
|
@@ -591,6 +733,7 @@ Every pipeline produces four output channels:
|
|
|
591
733
|
- **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
|
|
592
734
|
- **samples** -- reserved for sampling operators
|
|
593
735
|
|
|
736
|
+
<!-- readme-test: js-pipeline -->
|
|
594
737
|
```js
|
|
595
738
|
const result = await p.run({ inputFile: 'data.csv' })
|
|
596
739
|
console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
|
|
@@ -599,6 +742,8 @@ console.log(result.statsText) // newline-delimited JSON: summary plus step_sta
|
|
|
599
742
|
### Pipeline from JSON
|
|
600
743
|
|
|
601
744
|
```js
|
|
745
|
+
import { loadRecipe } from 'tranfi'
|
|
746
|
+
|
|
602
747
|
const p = await loadRecipe({
|
|
603
748
|
steps: [
|
|
604
749
|
{ op: 'codec.csv.decode', args: {} },
|
|
@@ -613,7 +758,7 @@ const p = await loadRecipe({
|
|
|
613
758
|
The package automatically selects the best backend:
|
|
614
759
|
|
|
615
760
|
1. **N-API** (Node.js) -- native C addon, fastest, used when available
|
|
616
|
-
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly,
|
|
761
|
+
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, single-file module
|
|
617
762
|
3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
|
|
618
763
|
|
|
619
764
|
```js
|
|
@@ -624,4 +769,19 @@ const tf = await createTranfi()
|
|
|
624
769
|
|
|
625
770
|
## Architecture
|
|
626
771
|
|
|
627
|
-
|
|
772
|
+
Node, Python, CLI and WASM use the same C11 engine. It processes columnar batches
|
|
773
|
+
with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`)
|
|
774
|
+
and per-cell null bitmaps.
|
|
775
|
+
|
|
776
|
+
| Need | Node API behavior |
|
|
777
|
+
|------|-------------------|
|
|
778
|
+
| Stream input | `run({ inputFile })` uses `createReadStream()` and drains output after each push and incremental finish boundary |
|
|
779
|
+
| Avoid collecting all output | Use `inputStream`, `toReadable()`, `writeTo()`, `iterChunks()`, or `onOutput` with `collectOutput: false` |
|
|
780
|
+
| Read gzip | `.gz` files use `zlib.createGunzip()`; override with `compression: "none"` or `"gzip"` for `inputFile`/`inputStream` |
|
|
781
|
+
| Permit a blocking operation | Set `allowBlocking: true` or use a supported `spillDir` path |
|
|
782
|
+
| Cap retained key state | Supply caps and check the plan with `memory: "64MB"` |
|
|
783
|
+
| Allow plan file reads | Supply explicit host-policy options |
|
|
784
|
+
|
|
785
|
+
Input streaming still collects the final output by default. Choose a streaming
|
|
786
|
+
sink for large outputs; row-local and bounded-state operators stream under the
|
|
787
|
+
default strict native memory policy.
|
package/csrc/pipeline.c
CHANGED
package/csrc/size_utils.c
CHANGED
|
@@ -104,6 +104,9 @@ char *tf_string_append_suffix_checked(const char *prefix, const char *suffix) {
|
|
|
104
104
|
return copy;
|
|
105
105
|
}
|
|
106
106
|
|
|
107
|
+
/* Platforms with qsort_r do not compile the portable introsort fallback. */
|
|
108
|
+
#if !defined(__GLIBC__) && !defined(__APPLE__) && !defined(__FreeBSD__) && \
|
|
109
|
+
!defined(__OpenBSD__) && !defined(__NetBSD__)
|
|
107
110
|
static void tf_index_swap(size_t *a, size_t *b) {
|
|
108
111
|
size_t tmp = *a;
|
|
109
112
|
*a = *b;
|
|
@@ -225,6 +228,7 @@ static void tf_sort_indices_intro_loop(size_t *indices, size_t lo, size_t hi, si
|
|
|
225
228
|
}
|
|
226
229
|
tf_sort_indices_insertion_range(indices, lo, hi, compare, ctx);
|
|
227
230
|
}
|
|
231
|
+
#endif
|
|
228
232
|
|
|
229
233
|
typedef struct {
|
|
230
234
|
tf_index_compare_fn compare;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "tranfi",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.1",
|
|
4
4
|
"description": "Streaming ETL language + runtime",
|
|
5
5
|
"type": "commonjs",
|
|
6
6
|
"main": "src/index.js",
|
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
"prepack": "node scripts/prepack.js",
|
|
40
40
|
"install": "node scripts/install-native.js",
|
|
41
41
|
"build:native": "node scripts/sync-csrc.js && node-gyp rebuild",
|
|
42
|
-
"test": "node --test ../test/test_install_native.js && node ../test/test_node.js && node ../test/test_transform_node.js && node ../test/test_transform_wasm_node.js && node ../test/test_transform_worker_node.js"
|
|
42
|
+
"test": "node --test ../test/test_install_native.js ../test/test_cli_node.js && node ../test/test_node.js && node ../test/test_transform_node.js && node ../test/test_transform_wasm_node.js && node ../test/test_transform_worker_node.js"
|
|
43
43
|
},
|
|
44
44
|
"peerDependencies": {
|
|
45
45
|
"duckdb": ">=1.0.0",
|
package/src/cli.js
CHANGED
|
@@ -9,8 +9,8 @@
|
|
|
9
9
|
* tranfi profile < data.csv
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
|
-
const { pipeline, compileDsl, recipes } = require('./index.js')
|
|
13
|
-
const { readFileSync, writeFileSync, existsSync } = require('fs')
|
|
12
|
+
const { pipeline, compileDsl, compileToSql, recipes } = require('./index.js')
|
|
13
|
+
const { readFileSync, writeFileSync, existsSync, openSync, closeSync } = require('fs')
|
|
14
14
|
const { resolve, dirname } = require('path')
|
|
15
15
|
const nativeBinding = require('./native.js')
|
|
16
16
|
|
|
@@ -34,6 +34,13 @@ Options:
|
|
|
34
34
|
-i FILE Read input from file instead of stdin
|
|
35
35
|
-o FILE Write output to file instead of stdout
|
|
36
36
|
-j Compile only, output JSON plan
|
|
37
|
+
--target native|json|sql Execute or compile a pipeline
|
|
38
|
+
--dialect NAME SQL dialect (duckdb)
|
|
39
|
+
--allow-blocking Permit full-input operations on known-small data
|
|
40
|
+
--fail-on-blocking Reject blocking operations (default)
|
|
41
|
+
--memory SIZE Native memory policy, e.g. max:64MB
|
|
42
|
+
--spill-dir DIR Existing private spill directory
|
|
43
|
+
--stats-json FILE Write stats NDJSON; - means stderr
|
|
37
44
|
-q Quiet mode (suppress stats)
|
|
38
45
|
-v Show version
|
|
39
46
|
-R List built-in recipes
|
|
@@ -102,10 +109,21 @@ async function main() {
|
|
|
102
109
|
let outputFile = null
|
|
103
110
|
let jsonMode = false
|
|
104
111
|
let quiet = false
|
|
112
|
+
let allowBlocking = false
|
|
113
|
+
let failOnBlocking = false
|
|
114
|
+
let memory, spillDir, statsFile
|
|
115
|
+
let target = 'native'
|
|
116
|
+
let dialect = 'duckdb'
|
|
117
|
+
function valueAt(i, inline, flag) {
|
|
118
|
+
const value = inline === undefined ? argv[i + 1] : inline
|
|
119
|
+
if (!value || value.startsWith('--')) throw new Error(`${flag} requires a value`)
|
|
120
|
+
return value
|
|
121
|
+
}
|
|
105
122
|
|
|
106
123
|
// Parse args
|
|
107
124
|
for (let i = 0; i < argv.length; i++) {
|
|
108
|
-
const arg = argv[i]
|
|
125
|
+
const [arg, ...assigned] = argv[i].split('=')
|
|
126
|
+
const inline = assigned.length ? assigned.join('=') : undefined
|
|
109
127
|
if (arg === '-h' || arg === '--help') { usage(); process.exit(0) }
|
|
110
128
|
else if (arg === '-v' || arg === '--version') {
|
|
111
129
|
const v = nativeBinding ? nativeBinding.version() : 'unknown'
|
|
@@ -121,18 +139,35 @@ async function main() {
|
|
|
121
139
|
}
|
|
122
140
|
process.exit(0)
|
|
123
141
|
}
|
|
124
|
-
else if (arg === '-j') jsonMode = true
|
|
125
|
-
else if (arg === '-q') quiet = true
|
|
126
|
-
else if (arg === '-
|
|
127
|
-
else if (arg === '-
|
|
128
|
-
else if (
|
|
129
|
-
|
|
142
|
+
else if (arg === '-j' || arg === '--json') jsonMode = true
|
|
143
|
+
else if (arg === '-q' || arg === '--quiet') quiet = true
|
|
144
|
+
else if (arg === '--allow-blocking') allowBlocking = true
|
|
145
|
+
else if (arg === '--fail-on-blocking') failOnBlocking = true
|
|
146
|
+
else if (['-f', '--file', '-i', '--input', '-o', '--output', '--memory', '--spill-dir', '--stats-json', '--target', '--dialect'].includes(arg)) {
|
|
147
|
+
const value = valueAt(i, inline, arg)
|
|
148
|
+
if (inline === undefined) i++
|
|
149
|
+
if (arg === '-f' || arg === '--file') pipelineFile = value
|
|
150
|
+
else if (arg === '-i' || arg === '--input') inputFile = value
|
|
151
|
+
else if (arg === '-o' || arg === '--output') outputFile = value
|
|
152
|
+
else if (arg === '--memory') memory = value
|
|
153
|
+
else if (arg === '--spill-dir') spillDir = value
|
|
154
|
+
else if (arg === '--stats-json') statsFile = value
|
|
155
|
+
else if (arg === '--target') target = value
|
|
156
|
+
else if (arg === '--dialect') dialect = value
|
|
157
|
+
}
|
|
158
|
+
else if (!argv[i].startsWith('-')) {
|
|
159
|
+
if (pipelineText !== null) throw new Error('only one pipeline argument is allowed')
|
|
160
|
+
pipelineText = argv[i]
|
|
161
|
+
}
|
|
130
162
|
else {
|
|
131
163
|
process.stderr.write(`error: unknown option '${arg}'\n`)
|
|
132
164
|
process.exit(1)
|
|
133
165
|
}
|
|
134
166
|
}
|
|
135
167
|
|
|
168
|
+
if (allowBlocking && failOnBlocking) throw new Error('--allow-blocking conflicts with --fail-on-blocking')
|
|
169
|
+
if (!['native', 'json', 'sql'].includes(target)) throw new Error('unknown --target; expected native, json or sql')
|
|
170
|
+
|
|
136
171
|
// Get pipeline text
|
|
137
172
|
if (pipelineFile) {
|
|
138
173
|
pipelineText = readFileSync(pipelineFile, 'utf-8')
|
|
@@ -143,26 +178,44 @@ async function main() {
|
|
|
143
178
|
process.exit(1)
|
|
144
179
|
}
|
|
145
180
|
|
|
146
|
-
|
|
147
|
-
if (
|
|
181
|
+
const named = (await recipes()).find(recipe => recipe.name === pipelineText.trim())
|
|
182
|
+
if (named) pipelineText = named.dsl
|
|
183
|
+
if (target === 'sql') {
|
|
184
|
+
process.stdout.write(await compileToSql(pipelineText, { dialect }) + '\n')
|
|
185
|
+
return
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
if (jsonMode || target === 'json') {
|
|
148
189
|
const json = await compileDsl(pipelineText)
|
|
149
190
|
process.stdout.write(json + '\n')
|
|
150
191
|
process.exit(0)
|
|
151
192
|
}
|
|
152
193
|
|
|
153
|
-
// Create and run pipeline
|
|
154
194
|
const p = pipeline(pipelineText)
|
|
155
|
-
const
|
|
156
|
-
const result = await p.run({
|
|
195
|
+
const options = {
|
|
157
196
|
inputFile: inputFile || undefined,
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
197
|
+
inputStream: inputFile ? undefined : process.stdin,
|
|
198
|
+
allowBlocking, memory, spillDir, allowFs: true, allowRulesFile: true
|
|
199
|
+
}
|
|
200
|
+
let result
|
|
162
201
|
if (outputFile) {
|
|
163
|
-
|
|
202
|
+
let fd
|
|
203
|
+
try {
|
|
204
|
+
result = await p.run({
|
|
205
|
+
...options,
|
|
206
|
+
collectOutput: false,
|
|
207
|
+
onOutput(chunk) {
|
|
208
|
+
// Open only after plan validation; a rejected plan must not truncate output.
|
|
209
|
+
if (fd === undefined) fd = openSync(outputFile, 'w')
|
|
210
|
+
writeFileSync(fd, chunk)
|
|
211
|
+
}
|
|
212
|
+
})
|
|
213
|
+
if (fd === undefined) fd = openSync(outputFile, 'w')
|
|
214
|
+
} finally {
|
|
215
|
+
if (fd !== undefined) closeSync(fd)
|
|
216
|
+
}
|
|
164
217
|
} else {
|
|
165
|
-
process.stdout
|
|
218
|
+
result = await p.writeTo(process.stdout, { ...options, end: false })
|
|
166
219
|
}
|
|
167
220
|
|
|
168
221
|
// Errors to stderr
|
|
@@ -170,18 +223,8 @@ async function main() {
|
|
|
170
223
|
process.stderr.write(result.errors)
|
|
171
224
|
}
|
|
172
225
|
|
|
173
|
-
|
|
174
|
-
if (!quiet && result.stats.length > 0)
|
|
175
|
-
process.stderr.write(result.stats)
|
|
176
|
-
}
|
|
177
|
-
}
|
|
178
|
-
|
|
179
|
-
function readStdin() {
|
|
180
|
-
return new Promise((resolve) => {
|
|
181
|
-
const chunks = []
|
|
182
|
-
process.stdin.on('data', chunk => chunks.push(chunk))
|
|
183
|
-
process.stdin.on('end', () => resolve(Buffer.concat(chunks)))
|
|
184
|
-
})
|
|
226
|
+
if (statsFile && statsFile !== '-') writeFileSync(statsFile, result.stats)
|
|
227
|
+
else if ((statsFile || !quiet) && result.stats.length > 0) process.stderr.write(result.stats)
|
|
185
228
|
}
|
|
186
229
|
|
|
187
230
|
main().catch(err => {
|
package/wasm/tranfi_core.js
CHANGED
|
Binary file
|