tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/README.md
CHANGED
|
@@ -1,14 +1,26 @@
|
|
|
1
1
|
# tranfi (Node.js / WASM)
|
|
2
2
|
|
|
3
|
-
Streaming ETL in JavaScript, powered by a native C11 core via N-API
|
|
3
|
+
Streaming-first ETL in JavaScript, powered by a native C11 core via N-API
|
|
4
|
+
(Node.js) or WASM (browsers). Tranfi processes CSV, JSONL, and text byte streams;
|
|
5
|
+
it is not an in-memory DataFrame API. Row-local operations stream, bounded
|
|
6
|
+
operators declare their limits, and full-input operators require an explicit
|
|
7
|
+
blocking or spill policy.
|
|
8
|
+
|
|
9
|
+
> **Unreleased main:** The prepared-transform API documented below targets
|
|
10
|
+
> Tranfi 0.2. Current npm 0.1.x installs do not include it; build this branch
|
|
11
|
+
> from source until 0.2 is published.
|
|
12
|
+
|
|
13
|
+
Save this example as `quickstart.mjs`:
|
|
4
14
|
|
|
5
15
|
```js
|
|
6
|
-
import
|
|
16
|
+
import tranfi from 'tranfi'
|
|
17
|
+
|
|
18
|
+
const { pipeline, codec, ops, expr } = tranfi
|
|
7
19
|
|
|
8
20
|
const result = await pipeline([
|
|
9
21
|
codec.csv(),
|
|
10
22
|
ops.filter(expr("col('age') > 25")),
|
|
11
|
-
ops.
|
|
23
|
+
ops.top(100, 'age'),
|
|
12
24
|
ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
|
|
13
25
|
ops.select(['name', 'age', 'label']),
|
|
14
26
|
codec.csvEncode(),
|
|
@@ -24,7 +36,7 @@ console.log(result.outputText)
|
|
|
24
36
|
Or use the pipe DSL for one-liners:
|
|
25
37
|
|
|
26
38
|
```js
|
|
27
|
-
const result = await pipeline('csv | filter "col(age) > 25" |
|
|
39
|
+
const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
|
|
28
40
|
.run({ inputFile: 'data.csv' })
|
|
29
41
|
```
|
|
30
42
|
|
|
@@ -34,7 +46,23 @@ const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
|
|
|
34
46
|
npm install tranfi
|
|
35
47
|
```
|
|
36
48
|
|
|
37
|
-
|
|
49
|
+
The default install compiles the N-API addon and reports a nonzero failure if
|
|
50
|
+
the native toolchain or synchronized C sources are unavailable. For an
|
|
51
|
+
intentional WASM-only/browser installation, set
|
|
52
|
+
`TRANFI_SKIP_NATIVE_BUILD=1` and import `tranfi/wasm` explicitly. The ordinary
|
|
53
|
+
pipeline can execute with WASM, but root-entry prepared transforms require the
|
|
54
|
+
native addon.
|
|
55
|
+
|
|
56
|
+
The native addon is currently built and tested on Linux. Windows is supported
|
|
57
|
+
through the packed package's `tranfi/wasm` entry with
|
|
58
|
+
`TRANFI_SKIP_NATIVE_BUILD=1`; the release CI runs streaming and prepared-transform
|
|
59
|
+
smokes in that configuration. This is not a Windows native-addon support claim.
|
|
60
|
+
|
|
61
|
+
Use `pipeline(...)` for byte-stream ETL. The separate
|
|
62
|
+
`TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
|
|
63
|
+
lifecycle is for typed batches whose learned state must be frozen and reused.
|
|
64
|
+
Calling `.run()` collects output for convenience; use `writeTo()` or
|
|
65
|
+
`toReadable()` for large outputs.
|
|
38
66
|
|
|
39
67
|
## CLI
|
|
40
68
|
|
|
@@ -42,11 +70,11 @@ Installing the package also provides the `tranfi` command:
|
|
|
42
70
|
|
|
43
71
|
```bash
|
|
44
72
|
# Via npx (no install)
|
|
45
|
-
echo 'name,age\nAlice,30\nBob,25' | npx tranfi 'csv | filter "age > 25" | csv'
|
|
73
|
+
echo 'name,age\nAlice,30\nBob,25' | npx tranfi -q 'csv | filter "age > 25" | csv'
|
|
46
74
|
|
|
47
75
|
# Or install globally
|
|
48
76
|
npm i -g tranfi
|
|
49
|
-
tranfi 'csv | filter "age > 25" |
|
|
77
|
+
tranfi -q 'csv | filter "age > 25" | top-k 100 age | csv' < data.csv
|
|
50
78
|
tranfi profile < data.csv
|
|
51
79
|
tranfi -R # list recipes
|
|
52
80
|
```
|
|
@@ -57,15 +85,14 @@ Run `tranfi -h` for all options.
|
|
|
57
85
|
|
|
58
86
|
### Two APIs
|
|
59
87
|
|
|
60
|
-
**Builder API** -- composable,
|
|
88
|
+
**Builder API** -- composable, structured, and IDE-friendly:
|
|
61
89
|
|
|
62
90
|
```js
|
|
63
91
|
const p = pipeline([
|
|
64
92
|
codec.csv(),
|
|
65
93
|
ops.filter(expr("col('score') >= 80")),
|
|
66
94
|
ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
|
|
67
|
-
ops.
|
|
68
|
-
ops.head(10),
|
|
95
|
+
ops.top(10, 'score'),
|
|
69
96
|
codec.csvEncode(),
|
|
70
97
|
])
|
|
71
98
|
const result = await p.run({ inputFile: 'students.csv' })
|
|
@@ -74,12 +101,14 @@ const result = await p.run({ inputFile: 'students.csv' })
|
|
|
74
101
|
**DSL strings** -- compact, suitable for CLI-like use:
|
|
75
102
|
|
|
76
103
|
```js
|
|
77
|
-
const p = pipeline('csv | filter "col(score) >= 80" |
|
|
104
|
+
const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
|
|
78
105
|
const result = await p.run({ inputFile: 'students.csv' })
|
|
79
106
|
```
|
|
80
107
|
|
|
81
108
|
Both produce identical pipelines under the hood.
|
|
82
109
|
|
|
110
|
+
Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate` -> `derive`, `summarise`/`summarize` -> `group-agg`, `distinct` -> `unique`, and `arrange` -> `sort`.
|
|
111
|
+
|
|
83
112
|
### Running pipelines
|
|
84
113
|
|
|
85
114
|
```js
|
|
@@ -98,20 +127,158 @@ result.statsText // string
|
|
|
98
127
|
result.samples // Buffer (sample channel)
|
|
99
128
|
```
|
|
100
129
|
|
|
130
|
+
For large outputs, drain chunks instead of collecting `result.output`:
|
|
131
|
+
|
|
132
|
+
```js
|
|
133
|
+
const { createReadStream, createWriteStream } = require("fs")
|
|
134
|
+
|
|
135
|
+
// Write to a Node Writable and wait for `drain` when the sink applies backpressure.
|
|
136
|
+
const out = createWriteStream("out.csv")
|
|
137
|
+
const result = await p.writeTo(out, { inputFile: "data.csv" })
|
|
138
|
+
|
|
139
|
+
// Or consume Tranfi output as a Node Readable.
|
|
140
|
+
for await (const chunk of p.toReadable({ inputFile: "data.csv" })) {
|
|
141
|
+
// process each output chunk
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// Existing input streams can feed the native pipeline without readFile() materialization.
|
|
145
|
+
const result2 = await p.run({ inputStream: createReadStream("data.csv") })
|
|
146
|
+
|
|
147
|
+
// .gz input files are decompressed as streaming sources by default.
|
|
148
|
+
const gz = await p.run({ inputFile: "events.jsonl.gz" })
|
|
149
|
+
const raw = await p.run({ inputFile: "events.jsonl.gz", compression: "none" })
|
|
150
|
+
const streamGz = await p.run({ inputStream: createReadStream("events.jsonl.gz"), compression: "gzip" })
|
|
151
|
+
|
|
152
|
+
// Multiple files stream sequentially through one pipeline.
|
|
153
|
+
const inputFiles = ["part-a.csv", "part-b.csv"]
|
|
154
|
+
const combined = await p.run({ inputFiles, sourceColumn: "src" })
|
|
155
|
+
|
|
156
|
+
// Low-level async iteration is still available.
|
|
157
|
+
for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
|
|
158
|
+
// process each output chunk
|
|
159
|
+
}
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
`sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. It does not remove repeated CSV headers from later files; use shards without repeated headers, `header: false`, or a pre-cleaning step when every file has its own header.
|
|
163
|
+
|
|
164
|
+
Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
|
|
165
|
+
|
|
166
|
+
For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
|
|
167
|
+
|
|
168
|
+
```js
|
|
169
|
+
// main thread
|
|
170
|
+
import { createWorkerClient } from 'tranfi/wasm/worker'
|
|
171
|
+
|
|
172
|
+
const client = createWorkerClient(new Worker(new URL('./tranfi-worker.js', import.meta.url), { type: 'module' }))
|
|
173
|
+
const result = await client.runFile('csv | filter "col(age) >= 18" | csv', file, {
|
|
174
|
+
chunkSize: 64 * 1024,
|
|
175
|
+
collectOutput: false,
|
|
176
|
+
onOutput: chunk => downloadSink.write(chunk),
|
|
177
|
+
onProgress: p => updateProgress(p),
|
|
178
|
+
signal: abortController.signal
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
// tranfi-worker.js, bundled by the app
|
|
182
|
+
import { runWorkerServer } from 'tranfi/wasm/worker'
|
|
183
|
+
runWorkerServer()
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
The Worker protocol streams chunks with transferable buffers and sends progress, stats, errors, and cancellation messages around the same WASM `create/push/pull/finish/free` API; it is not a separate IR target.
|
|
187
|
+
|
|
188
|
+
The bundled Tranfi app runner is also preview-bounded by default: file chunks are streamed into WASM, main output is drained incrementally into a table preview capped by hidden `preview_rows` (default 200), and full output text is only materialized when `collect_output` is explicitly true in the schema.
|
|
189
|
+
|
|
190
|
+
### Prepared reusable transforms
|
|
191
|
+
|
|
192
|
+
WASM cancellation tokens created with `createTransformCancelToken()` expose a read-only `requested` boolean for host-side conversion loops. Reading a closed token fails.
|
|
193
|
+
|
|
194
|
+
Prepared transforms are a separate typed-table API for operations whose parameters must be learned from reference data. `analyze` accumulates bounded statistics over one or more batches, `finalize` freezes an immutable plan and output schema, and `apply` runs that plan either over a second pass of the original data (`fit_transform`-style) or over later compatible batches. It does not replace the byte-stream pipeline API.
|
|
195
|
+
|
|
196
|
+
The current slice accepts declared `float32`/`float64` columns. It supports numeric none/zero/constant/mean/exact-median imputation and none/standard/min-max normalization. Declared categorical columns support mode imputation with `allMissing: 'error' | 'zero'` or `impute.op: 'none'`; encoders discover finite typed categories, while the no-imputation/no-encoding combination needs no learned dictionary. A column may instead provide both branches with `kind.op: 'infer'`, `kind.rule: 'finite-integer-cardinality-v1'`, and `kind.maxCategories >= 2`: missing/NaN values are ignored, `2..maxCategories` distinct finite integers resolve categorical, while zero/one distinct value, any noninteger, or the next distinct value resolves numeric. `impute.op: 'none'` with `encode.op: 'none'` passes every finite value through and emits canonical qNaN for missing input. Mode with `encode.op: 'none'` retains its learned dictionary and rejects unseen finite values with code `108`. `encode.op: 'label'` freezes zero-based sorted ordinals and supports `unknown: 'error' | 'sentinel' | 'other'`; `encode.op: 'onehot'` emits source-ordered category blocks and supports `unknown: 'error' | 'all_zero' | 'other'`. Without categorical imputation, missing label/one-hot input follows that encoder's unknown policy. The label sentinel must be a safe integer outside the learned ordinal range; label/one-hot `other` appends the reserved ordinal/field after known categories. Generated output IDs/names and one-hot category metadata are deterministic, and collisions fail with code `102`. Exact median obeys the configured allocation and resident-state limits; inference, categorical discovery, and output expansion obey category, output-width, allocation, and resident-state limits. Limit failures use resource code `104`. Prepared-transform host-policy and spill fields are reserved but not implemented; nonempty use fails with unsupported-runtime code `113` instead of being silently ignored.
|
|
197
|
+
|
|
198
|
+
Declared categorical columns also accept a fixed, nonempty `encode.categories` array of finite, sorted, unique tags (`{ "t": "f64", "v": "4000000000000000" }` represents 2). Tags must match the input dtype; negative zero is represented as positive zero. Fixed dictionaries retain unobserved categories. Unknown training values follow the encoder policy and never vote for mode; an all-missing zero fallback must exist in the dictionary. Fixed encoding without imputation can finalize without training rows. Fixed dictionaries with kind inference, string categories, and categorical constant imputation remain unsupported.
|
|
199
|
+
|
|
200
|
+
Prepared runtime option objects reject unknown fields. Native Node accepts
|
|
201
|
+
`limits`, `cancelFlag`, `hostPolicy`, and `spillDir`; standalone WASM accepts
|
|
202
|
+
`limits`, `cancelToken`, `hostPolicy`, and `spillDir`. The host/spill fields are
|
|
203
|
+
reserved as described above, and cancellation spellings are not interchangeable.
|
|
204
|
+
|
|
205
|
+
```js
|
|
206
|
+
const tf = require('tranfi')
|
|
207
|
+
|
|
208
|
+
const recipeSpec = {
|
|
209
|
+
format: 'tranfi.transform-recipe',
|
|
210
|
+
version: 1,
|
|
211
|
+
policyVersion: 1,
|
|
212
|
+
outputDtype: 'float64',
|
|
213
|
+
semanticLimits: {
|
|
214
|
+
maxOutputColumns: 65536,
|
|
215
|
+
maxOutputElementsPerApply: 134217728
|
|
216
|
+
},
|
|
217
|
+
columns: [{
|
|
218
|
+
sourceId: 'x0',
|
|
219
|
+
kind: { op: 'declared', value: 'numeric', rule: null, maxCategories: null },
|
|
220
|
+
numeric: {
|
|
221
|
+
impute: { op: 'mean', constant: null, allMissing: 'zero' },
|
|
222
|
+
normalize: { op: 'standard', ddof: 0 }
|
|
223
|
+
},
|
|
224
|
+
categorical: null
|
|
225
|
+
}]
|
|
226
|
+
}
|
|
227
|
+
const schema = [{ id: 'x0', dtype: 'float64' }]
|
|
228
|
+
|
|
229
|
+
const recipe = tf.TransformRecipe.fromJSON(recipeSpec)
|
|
230
|
+
const analyzer = recipe.analyzer(schema)
|
|
231
|
+
analyzer.push({ rows: 3, columns: [new Float64Array([1, NaN, 3])] })
|
|
232
|
+
const plan = analyzer.finalize()
|
|
233
|
+
|
|
234
|
+
const apply = plan.apply(schema)
|
|
235
|
+
const fittedReference = apply.run({
|
|
236
|
+
rows: 3,
|
|
237
|
+
columns: [new Float64Array([1, NaN, 3])]
|
|
238
|
+
})
|
|
239
|
+
const planBytes = plan.toBytes() // canonical TFTR artifact
|
|
240
|
+
const recipeSha256 = plan.recipeSha256() // canonical recipe + input-schema identity
|
|
241
|
+
|
|
242
|
+
apply.close()
|
|
243
|
+
plan.close()
|
|
244
|
+
analyzer.close()
|
|
245
|
+
recipe.close()
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
`tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
|
|
249
|
+
|
|
250
|
+
```js
|
|
251
|
+
const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
|
|
252
|
+
const makeWorker = () => new Worker(workerUrl, { type: 'module' })
|
|
253
|
+
const client = createWorkerClient(makeWorker(), {
|
|
254
|
+
workerFactory: makeWorker,
|
|
255
|
+
terminateOnDispose: true
|
|
256
|
+
})
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
|
|
260
|
+
|
|
101
261
|
## Codecs
|
|
102
262
|
|
|
103
263
|
Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
|
|
104
264
|
|
|
105
265
|
| Method | Description |
|
|
106
266
|
|--------|-------------|
|
|
107
|
-
| `codec.csv({ delimiter, header, batchSize, repair })` | CSV decoder. `repair: true`
|
|
267
|
+
| `codec.csv({ delimiter, header, batchSize, repair, mode, strict, maxErrorBytes, maxRecordBytes, maxColumns, nulls, quotedNulls, skip, nMax, maxRows, comment, trimWs, skipEmptyRows, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | CSV decoder. `mode: 'strict'` fails on field-count mismatches; `repair: true` / `mode: 'repair'` emits repair diagnostics; `maxRecordBytes` bounds buffered records; `nulls: ['NA']` adds null sentinels; `skip: 2`, `nMax: 100`, `comment: '#'`, `trimWs: false`, and `skipEmptyRows: true` control row-local parsing; repair audit/raw diagnostics support privacy controls with pseudo-column `raw` |
|
|
108
268
|
| `codec.csvEncode({ delimiter })` | CSV encoder |
|
|
109
|
-
| `codec.jsonl({ batchSize })` | JSON Lines decoder |
|
|
269
|
+
| `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | JSON Lines decoder. `onError` is `skip`, `fail`, `warn`, or `quarantine` |
|
|
110
270
|
| `codec.jsonlEncode()` | JSON Lines encoder |
|
|
111
|
-
| `codec.text({ batchSize })` | Line-oriented text decoder (single `_line` column) |
|
|
271
|
+
| `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | Line-oriented text decoder (single `_line` column) |
|
|
112
272
|
| `codec.textEncode()` | Text encoder |
|
|
113
273
|
| `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
|
|
114
274
|
|
|
275
|
+
CSV treats unquoted empty fields as null by default. Pass `nulls: ['NA', 'NULL']` to add sentinel strings; pass `quotedNulls: false` when quoted sentinels like `"NA"` or `""` should remain strings. Default field-count handling is permissive for compatibility. Use `mode: 'strict'` or `strict: true` to fail on row/header width mismatches. Use `repair: true` or `mode: 'repair'` to pad/truncate and collect JSONL diagnostics in `result.errors`; `maxErrorBytes` bounds raw previews, and `auditIncludeRow`, `auditColumns`, `auditRedact`, `auditHashColumns`, `auditMaxBytes`, and `auditMaxCellBytes` govern repair audit/raw payloads with the raw preview exposed as pseudo-column `raw`. `maxRecordBytes` defaults to `67108864` bytes, caps the current record buffer before a newline is seen, and rejects with a `csv_record_too_large` diagnostic when exceeded; pass `0` to disable the guard. `maxColumns` defaults to `8192`; records above the cap reject with a bounded `csv_too_many_columns` diagnostic instead of silently dropping columns. Decoder size options are checked before execution: `batchSize` must be `1..65536`, `maxErrorBytes` must be `0..67108864`, `maxRecordBytes` must be `0..1073741824`, and `maxColumns` must be `1..65536`. `skip: 2` discards preamble records before header/schema discovery; comments are applied after skipped rows. `nMax: 100` / `maxRows: 100` keeps at most that many decoded data rows after skip/comment/header handling and uses only one counter; `nMax: 0` preserves a header-only schema batch. `comment: '#'` removes text after an unquoted marker and skips comment-only rows; quoted markers are preserved. Unquoted spaces/tabs are trimmed by default; pass `trimWs: false` to preserve them. Blank physical rows after the header are preserved as all-null rows by default; pass `skipEmptyRows: true` to drop them.
|
|
276
|
+
|
|
277
|
+
Malformed JSONL records are skipped by default. Set `onError: 'warn'` or `onError: 'quarantine'` to keep valid rows and collect JSONL diagnostics in `result.errors`; set `onError: 'fail'` to reject on the first malformed line. `maxErrorBytes` bounds the raw preview stored in diagnostics, and `maxRecordBytes` applies the same current-line guard as CSV/text.
|
|
278
|
+
|
|
279
|
+
Each JSONL record must be one complete UTF-8 JSON object. Invalid number syntax, trailing content, embedded NULs, and numbers outside the finite float64 range follow the same malformed-record policy. The encoder rejects nonfinite values. Schema order comes from the first valid record; repeated keys use their first value.
|
|
280
|
+
|
|
281
|
+
|
|
115
282
|
Cross-codec pipelines work naturally:
|
|
116
283
|
|
|
117
284
|
```js
|
|
@@ -128,30 +295,46 @@ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
|
|
|
128
295
|
|
|
129
296
|
| Method | Description |
|
|
130
297
|
|--------|-------------|
|
|
131
|
-
| `ops.filter(expr)` | Keep rows matching expression |
|
|
298
|
+
| `ops.filter(expr, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Keep rows matching expression; optional dropped-row audit records support row omission, column allowlists, redaction, hashes, and payload caps |
|
|
132
299
|
| `ops.head(n)` | First N rows |
|
|
133
300
|
| `ops.tail(n)` | Last N rows |
|
|
134
301
|
| `ops.skip(n)` | Skip first N rows |
|
|
135
302
|
| `ops.top(n, column, desc?)` | Top N by column value |
|
|
136
|
-
| `ops.sample(n)` |
|
|
303
|
+
| `ops.sample(n, { seed })` | Deterministic bounded reservoir sampling; use `seed: 'random'` for nondeterministic mode |
|
|
137
304
|
| `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
|
|
138
|
-
| `ops.validate(expr)` | Add `_valid` boolean column, keep all rows |
|
|
305
|
+
| `ops.validate(expr, { rules, rulesFile, audit, auditLimit, maxFailures, warnFailureRate, maxFailureRate, name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Add `_valid` boolean column, keep all rows; supports inline rules or local JSON rule-suite files; bounded failure audit records support count/rate thresholds plus privacy controls |
|
|
306
|
+
| `ops.assert(expr, { action, name, message, result, aggregate, op, value, column, tolerance, rel, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Row-local data-quality rule or finish-time O(1) aggregate assertion; row-local failure side-channel records support privacy controls |
|
|
307
|
+
| `ops.quarantine(expr, { name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Route rows matching expression to `errors` and drop them from main output; row payloads support privacy controls |
|
|
308
|
+
| `ops.schema({ columns, required, nonNull, nullable, values, min, max, regex, mode='fail', result='_schema', maxRegexPatternBytes, maxRegexCellBytes, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Row-local table schema contract; fail, warn, filter, quarantine, or annotate; regex budgets default to 4096-byte patterns and 65536-byte cells; schema audit/error records support row omission, column allowlists, redaction, stable non-cryptographic hashes, and row/cell payload caps |
|
|
309
|
+
| `ops.schemaInfer({ rows })` | Bounded decoded-type/nullability schema report; defaults to 10000 sampled rows |
|
|
310
|
+
| `ops.tee({ expr, channel, columns, limit, every, name, includeRow, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Preserve main rows and write bounded JSONL row snapshots to a side channel; row payloads support privacy controls |
|
|
139
311
|
|
|
140
312
|
### Column operations
|
|
141
313
|
|
|
142
314
|
| Method | Description |
|
|
143
315
|
|--------|-------------|
|
|
144
316
|
| `ops.select(columns)` | Keep and reorder columns |
|
|
317
|
+
| `ops.relocate(columns, { before, after })` | Move columns while preserving all columns |
|
|
145
318
|
| `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
|
|
146
319
|
| `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
|
|
147
|
-
| `ops.
|
|
320
|
+
| `ops.sourceName({ result, defaultValue })` | Append the current host source path/name as a row-local string column |
|
|
321
|
+
| `ops.across(columns, { fn, functions, names, replace })` | Apply row-local functions over selected columns |
|
|
322
|
+
| `ops.cast(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Type conversion; optional bounded value/coercion audit records support privacy controls |
|
|
148
323
|
| `ops.trim(columns?)` | Strip whitespace |
|
|
149
|
-
| `ops.fillNull(mapping)` | Replace nulls
|
|
324
|
+
| `ops.fillNull(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Replace nulls; optional bounded audit records support privacy controls |
|
|
150
325
|
| `ops.fillDown(columns?)` | Forward-fill nulls |
|
|
151
326
|
| `ops.clip(column, { min, max })` | Clamp numeric values |
|
|
152
|
-
| `ops.replace(column, pattern, replacement, { regex })` | String find/replace |
|
|
327
|
+
| `ops.replace(column, pattern, replacement, { regex, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | String find/replace; optional bounded audit records support privacy controls |
|
|
153
328
|
| `ops.hash(columns?)` | Add `_hash` column (DJB2) |
|
|
154
|
-
| `ops.bin(column, boundaries)` | Discretize into bins |
|
|
329
|
+
| `ops.bin(column, boundaries, { missing, onTypeError })` | Discretize into bins with strict numeric-source defaults |
|
|
330
|
+
| `ops.ewma(column, alpha, { result, missing, onTypeError })` | Exponentially weighted moving average |
|
|
331
|
+
| `ops.anomaly(column, { threshold, result, missing, onTypeError })` | Streaming z-score anomaly flag |
|
|
332
|
+
| `ops.normalize(columns, { method, audit, auditLimit, missing, onTypeError, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Blocking minmax/zscore normalization; optional audit records support privacy controls |
|
|
333
|
+
| `ops.acf(column, { lags, missing, onTypeError })` | Blocking autocorrelation table |
|
|
334
|
+
|
|
335
|
+
`columns` may contain exact names or selector strings: `id:score`, `starts_with(score_)`, `ends_with(_id)`, `contains(temp)`, `matches(^score_)`, `where(numeric)`, strict `all_of(score,name)`, lenient `any_of(optional,score)`, exclusions with `!name` or `-name`, and boolean selector algebra such as `starts_with(score_)&where(numeric)`, `starts_with(score_)&!ends_with(raw)`, or `!(id:score)`. Ranges use input schema order and can be reversed. Helper matching is case-insensitive; exact names are case-sensitive. These selectors are resolved by the native schema-aware `select`, `relocate`, and `across` ops in `O(columns)` without retaining rows; SQL lowering rejects selector helpers without a known schema.
|
|
336
|
+
|
|
337
|
+
Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selected numeric columns per row; `ops.across(['name'], { functions: ['lower'], replace: false, names: '{col}_{fn}' })` appends templated columns. This is native/WASM row-local execution, not arbitrary lambdas or grouped dplyr evaluation.
|
|
155
338
|
|
|
156
339
|
### Sorting and deduplication
|
|
157
340
|
|
|
@@ -165,14 +348,15 @@ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
|
|
|
165
348
|
| Method | Description |
|
|
166
349
|
|--------|-------------|
|
|
167
350
|
| `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
|
|
168
|
-
| `ops.frequency(columns?)` | Value counts
|
|
169
|
-
| `ops.groupAgg(groupBy, aggs)` | Group by + aggregate |
|
|
351
|
+
| `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
|
|
352
|
+
| `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
|
|
170
353
|
|
|
171
354
|
```js
|
|
172
355
|
// Group aggregation
|
|
173
356
|
ops.groupAgg(['city'], [
|
|
174
|
-
{ column: 'price', func: 'sum',
|
|
175
|
-
{ column: 'price', func: 'avg',
|
|
357
|
+
{ column: 'price', func: 'sum', name: 'total' },
|
|
358
|
+
{ column: 'price', func: 'avg', name: 'avg_price' },
|
|
359
|
+
{ column: '*', func: 'count', name: 'rows' },
|
|
176
360
|
])
|
|
177
361
|
```
|
|
178
362
|
|
|
@@ -181,36 +365,49 @@ ops.groupAgg(['city'], [
|
|
|
181
365
|
| Method | Description |
|
|
182
366
|
|--------|-------------|
|
|
183
367
|
| `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
|
|
184
|
-
| `ops.window(column, size, func,
|
|
368
|
+
| `ops.window(column, size, func, resultOrOptions?)` | Sliding window: `avg`, `sum`, `min`, `max`; options include `result`, `missing`, `onTypeError` |
|
|
369
|
+
| `ops.rollingSum/rollingMean/rollingMin/rollingMax(column, size, { result, missing, onTypeError })` | Named trailing fixed-row numeric windows; default missing/non-numeric source fails |
|
|
370
|
+
| `ops.rollingAny/rollingAll(column, size, { result, nulls })` | Boolean trailing windows; `nulls`: `ignore`, `false`, `true`, `propagate` |
|
|
185
371
|
| `ops.lead(column, { offset, result })` | Lookahead N rows |
|
|
372
|
+
| `ops.lag(column, { offset, result })` | Previous-row shift |
|
|
373
|
+
| `ops.shift(column, { offset, result, type })` | Shift alias; `type='lag'` default or `type='lead'` |
|
|
374
|
+
| `ops.rowid(columns, { result, sorted, maxKeys })` | Global or per-key 1-based row ids |
|
|
375
|
+
| `ops.rleid(columns, { result })` | Consecutive run id by selected column(s) |
|
|
186
376
|
|
|
187
377
|
### Reshape
|
|
188
378
|
|
|
189
379
|
| Method | Description |
|
|
190
380
|
|--------|-------------|
|
|
191
|
-
| `ops.explode(column, delimiter
|
|
381
|
+
| `ops.explode(column, delimiter?, { maxTokensPerRow, maxOutputRowsPerInputRow, maxOutputRowsPerBatch, maxTokenBytes })` | Split delimited string into rows with optional expansion caps |
|
|
192
382
|
| `ops.split(column, names, delimiter?)` | Split column into multiple columns |
|
|
193
|
-
| `ops.unpivot(columns)` | Wide to long (melt) |
|
|
383
|
+
| `ops.unpivot(columns, { maxOutputRowsPerInputRow, maxOutputRowsPerBatch })` | Wide to long (melt) with optional expansion caps |
|
|
194
384
|
| `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
|
|
195
385
|
|
|
386
|
+
`explode` caps fail fast with `maxTokensPerRow`, `maxOutputRowsPerInputRow`, `maxOutputRowsPerBatch`, or `maxTokenBytes` when a single row or batch would expand beyond the configured limit. `unpivot` supports `maxOutputRowsPerInputRow` and `maxOutputRowsPerBatch`.
|
|
387
|
+
|
|
196
388
|
### Date/time
|
|
197
389
|
|
|
198
390
|
| Method | Description |
|
|
199
391
|
|--------|-------------|
|
|
200
|
-
| `ops.datetime(column,
|
|
201
|
-
| `ops.dateTrunc(column, trunc, { result })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
|
|
392
|
+
| `ops.datetime(column, extractOrOptions?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday`; accepts `{ extract, missing, onTypeError }` |
|
|
393
|
+
| `ops.dateTrunc(column, trunc, { result, missing, onTypeError })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
|
|
202
394
|
|
|
203
395
|
### Other
|
|
204
396
|
|
|
205
397
|
| Method | Description |
|
|
206
398
|
|--------|-------------|
|
|
207
399
|
| `ops.flatten()` | Flatten nested columns |
|
|
400
|
+
| `ops.interpolate(column, { method, missing, onTypeError })` | Fill nulls in a numeric column; default missing/non-numeric source fails |
|
|
401
|
+
| `ops.jsonExtract(path, result, { column, type })` | Extract JSON Pointer/simple JSONPath value into a new column |
|
|
402
|
+
| `ops.jsonFilter(path, { op, value, column, type })` | Filter rows by JSON Pointer/simple JSONPath predicate |
|
|
403
|
+
| `ops.jsonSchema(schema, { column, mode, result, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Validate JSON text with a supported JSON Schema subset; filter-mode audit records support privacy controls |
|
|
404
|
+
| `ops.jsonFlatten(fields, { column })` | Append declared JSON Pointer/simple JSONPath fields as bounded output columns |
|
|
208
405
|
| `ops.reorder(columns)` | Alias for `select` |
|
|
209
406
|
| `ops.dedup(columns?)` | Alias for `unique` |
|
|
210
407
|
|
|
211
408
|
## Expressions
|
|
212
409
|
|
|
213
|
-
Used in `filter`, `derive`, and `
|
|
410
|
+
Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
|
|
214
411
|
|
|
215
412
|
```js
|
|
216
413
|
ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
|
|
@@ -228,11 +425,19 @@ ops.derive({
|
|
|
228
425
|
| Comparison | `>` `>=` `<` `<=` `==` `!=` |
|
|
229
426
|
| Logic | `and` `or` `not` |
|
|
230
427
|
| String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
|
|
231
|
-
| Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` |
|
|
232
|
-
|
|
|
428
|
+
| Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` `between(x,left,right)` `inrange(x,left,right)` |
|
|
429
|
+
| Date/time | `year(x)` `month(x)` `day(x)` `hour(x)` `minute(x)` `second(x)` `weekday(x)` `epoch(x)` `date_trunc(x,unit)` |
|
|
430
|
+
| Conditional | `if(cond,then,else)` `case_when(cond,value,...,default)` `case_match(value,key,result,...,default)` `if_any(pred,...)` `if_all(pred,...)` `coalesce(a,b,...)` `nullif(a,b)` |
|
|
233
431
|
| Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
|
|
234
432
|
|
|
235
|
-
Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`.
|
|
433
|
+
Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
|
|
434
|
+
|
|
435
|
+
```js
|
|
436
|
+
ops.derive({
|
|
437
|
+
year: expr("year(col('date'))"),
|
|
438
|
+
monthStart: expr("date_trunc(col('date'), 'month')"),
|
|
439
|
+
})
|
|
440
|
+
```
|
|
236
441
|
|
|
237
442
|
## Recipes
|
|
238
443
|
|
|
@@ -248,6 +453,7 @@ const result = await pipeline('freq').run({ inputFile: 'data.csv' })
|
|
|
248
453
|
| `profile` | `csv \| stats \| csv` | Full data profiling |
|
|
249
454
|
| `preview` | `csv \| head 10 \| csv` | First 10 rows |
|
|
250
455
|
| `schema` | `csv \| head 0 \| csv` | Column names only |
|
|
456
|
+
| `sniff` | `csv \| schema infer rows=1000 \| csv` | Bounded-memory schema/type/nullability sniff |
|
|
251
457
|
| `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
|
|
252
458
|
| `count` | `csv \| stats count \| csv` | Row count |
|
|
253
459
|
| `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
|
|
@@ -277,6 +483,25 @@ for (const r of await recipes()) {
|
|
|
277
483
|
}
|
|
278
484
|
```
|
|
279
485
|
|
|
486
|
+
|
|
487
|
+
## Memory policy
|
|
488
|
+
|
|
489
|
+
Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
|
|
490
|
+
|
|
491
|
+
```js
|
|
492
|
+
await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
|
|
493
|
+
```
|
|
494
|
+
|
|
495
|
+
For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
|
|
496
|
+
|
|
497
|
+
```js
|
|
498
|
+
await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
|
|
499
|
+
```
|
|
500
|
+
|
|
501
|
+
Native Node execution supports spill-backed `sort`, capped unsorted `pivot`, unsorted `unique`/`dedup`, unsorted `group-agg`, capped unsorted `join` inner/left, unsorted `semi-join`/`anti-join`, unsorted set operations, and duplicate-eliminating `union` when `spillDir` is provided. `run()`, `iterChunks()`, `toReadable()`, and `writeTo()` drain finish-time merge output through N-API `finishStep()` instead of waiting for one whole `finish()`. Standalone WASM supports `allowBlocking` and `memory`, but rejects `spillDir` because browser/WASM spill storage would occupy WASM memory. Use the Node native addon, CLI/direct C, or `{ engine: 'duckdb' }` for external spill; with DuckDB, `memory` maps to `memory_limit` and `spillDir` maps to `temp_directory`.
|
|
502
|
+
|
|
503
|
+
Plan-internal file reads are denied by default in Node and WASM. Pass `allowFs: true` for trusted local lookup files used by `join`, set operations, `union`, or `stack`; pass both `allowFs: true` and `allowRulesFile: true` for `validate rules_file=...`; pass `workspaceRoot` to pin resolved core plan paths inside a trusted directory. Supplying `spillDir` opts into local filesystem spill for Node native execution; standalone WASM still rejects spill. `inputFile` and `inputFiles` are host source adapters and are not controlled by `allowFs`.
|
|
504
|
+
|
|
280
505
|
## DuckDB engine
|
|
281
506
|
|
|
282
507
|
Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
|
|
@@ -302,7 +527,10 @@ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
|
|
|
302
527
|
Generate SQL directly from DSL strings:
|
|
303
528
|
|
|
304
529
|
```js
|
|
305
|
-
const sql = await compileToSql(
|
|
530
|
+
const sql = await compileToSql(
|
|
531
|
+
'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
|
|
532
|
+
{ dialect: 'duckdb' }
|
|
533
|
+
)
|
|
306
534
|
console.log(sql)
|
|
307
535
|
// WITH
|
|
308
536
|
// step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
|
|
@@ -310,6 +538,10 @@ console.log(sql)
|
|
|
310
538
|
// SELECT * FROM step_2
|
|
311
539
|
```
|
|
312
540
|
|
|
541
|
+
DuckDB is the only implemented SQL dialect today. `dialect: 'sqlite'` and
|
|
542
|
+
`dialect: 'postgres'` are recognized but rejected until those dialects have
|
|
543
|
+
their own compatibility tests and SQL-generation rules.
|
|
544
|
+
|
|
313
545
|
### Browser (WASM + DuckDB-WASM)
|
|
314
546
|
|
|
315
547
|
In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
|
|
@@ -321,7 +553,7 @@ import * as duckdb from '@duckdb/duckdb-wasm'
|
|
|
321
553
|
const tf = await createTranfi()
|
|
322
554
|
|
|
323
555
|
// SQL generation (synchronous, no DuckDB needed)
|
|
324
|
-
const sql = tf.compileToSql('csv | filter "age > 25" | csv')
|
|
556
|
+
const sql = tf.compileToSql('csv | filter "age > 25" | csv', { dialect: 'duckdb' })
|
|
325
557
|
|
|
326
558
|
// Full execution with DuckDB-WASM
|
|
327
559
|
const db = new duckdb.AsyncDuckDB(...)
|
|
@@ -356,12 +588,12 @@ Every pipeline produces four output channels:
|
|
|
356
588
|
|
|
357
589
|
- **output** -- main pipeline result
|
|
358
590
|
- **errors** -- rows that failed processing
|
|
359
|
-
- **stats** --
|
|
591
|
+
- **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
|
|
360
592
|
- **samples** -- reserved for sampling operators
|
|
361
593
|
|
|
362
594
|
```js
|
|
363
595
|
const result = await p.run({ inputFile: 'data.csv' })
|
|
364
|
-
console.log(result.statsText) //
|
|
596
|
+
console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
|
|
365
597
|
```
|
|
366
598
|
|
|
367
599
|
### Pipeline from JSON
|
|
@@ -381,7 +613,7 @@ const p = await loadRecipe({
|
|
|
381
613
|
The package automatically selects the best backend:
|
|
382
614
|
|
|
383
615
|
1. **N-API** (Node.js) -- native C addon, fastest, used when available
|
|
384
|
-
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~
|
|
616
|
+
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~500 KB single-file
|
|
385
617
|
3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
|
|
386
618
|
|
|
387
619
|
```js
|
|
@@ -392,4 +624,4 @@ const tf = await createTranfi()
|
|
|
392
624
|
|
|
393
625
|
## Architecture
|
|
394
626
|
|
|
395
|
-
The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps.
|
|
627
|
+
The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. `run({ inputFile })` streams files with `createReadStream()` and drains native/WASM main output after each push and each incremental finish boundary; by default it still collects the final output for convenience. `.gz` input files are decompressed through a `zlib.createGunzip()` source transform; use `compression: "none"` to force raw bytes or `compression: "gzip"` to force gzip for `inputFile`/`inputStream`. Use `inputStream`, `toReadable()`, `writeTo()`, `onOutput` with `collectOutput: false`, or `iterChunks()` for large inputs/outputs and backpressure-aware sinks. Native execution is strict by default: row-local and bounded-state operators stream, blocking operators require `allowBlocking: true` or supported `spillDir`, capped key-state plans can be checked with `memory: "64MB"`, and core plan file reads require explicit host-policy options.
|