tranfi 0.0.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -21
- package/NOTICE +8 -0
- package/README.md +627 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-BIAIKnrp.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +121 -0
- package/csrc/arena.c +93 -0
- package/csrc/batch.c +976 -0
- package/csrc/buffer.c +154 -0
- package/csrc/cJSON.c +3386 -0
- package/csrc/cJSON.h +316 -0
- package/csrc/codec_csv.c +1951 -0
- package/csrc/codec_jsonl.c +1086 -0
- package/csrc/codec_table.c +248 -0
- package/csrc/codec_text.c +447 -0
- package/csrc/compiler.c +130 -0
- package/csrc/config.h +21 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +5417 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1553 -0
- package/csrc/expr.h +58 -0
- package/csrc/internal.h +539 -0
- package/csrc/ir.c +166 -0
- package/csrc/ir.h +208 -0
- package/csrc/ir_schema.c +75 -0
- package/csrc/ir_serialize.c +166 -0
- package/csrc/ir_sql.c +1822 -0
- package/csrc/ir_validate.c +576 -0
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +1241 -0
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +283 -0
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +255 -0
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +248 -0
- package/csrc/op_cast.c +523 -0
- package/csrc/op_clip.c +99 -0
- package/csrc/op_date_trunc.c +355 -0
- package/csrc/op_datetime.c +394 -0
- package/csrc/op_derive.c +216 -0
- package/csrc/op_diff.c +250 -0
- package/csrc/op_ewma.c +222 -0
- package/csrc/op_explode.c +206 -0
- package/csrc/op_fill_down.c +235 -0
- package/csrc/op_fill_null.c +268 -0
- package/csrc/op_filter.c +181 -0
- package/csrc/op_frequency.c +721 -0
- package/csrc/op_grep.c +181 -0
- package/csrc/op_group_agg.c +1956 -0
- package/csrc/op_hash.c +159 -0
- package/csrc/op_head.c +84 -0
- package/csrc/op_interpolate.c +445 -0
- package/csrc/op_join.c +2902 -0
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +419 -0
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +242 -0
- package/csrc/op_normalize.c +510 -0
- package/csrc/op_onehot.c +457 -0
- package/csrc/op_pivot.c +1754 -0
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +3044 -0
- package/csrc/op_rename.c +129 -0
- package/csrc/op_replace.c +354 -0
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +158 -0
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +340 -0
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +95 -0
- package/csrc/op_sort.c +819 -0
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +151 -0
- package/csrc/op_split_data.c +119 -0
- package/csrc/op_stack.c +271 -0
- package/csrc/op_stats.c +875 -0
- package/csrc/op_step.c +333 -0
- package/csrc/op_tail.c +105 -0
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +357 -0
- package/csrc/op_trim.c +138 -0
- package/csrc/op_unique.c +1343 -0
- package/csrc/op_unpivot.c +193 -0
- package/csrc/op_validate.c +648 -0
- package/csrc/op_window.c +591 -0
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +1088 -0
- package/csrc/recipes.c +104 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +506 -0
- package/csrc/report.h +22 -0
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +291 -0
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +218 -0
- package/napi_api.c +534 -0
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +64 -59
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +190 -0
- package/src/engines/duckdb.js +142 -0
- package/src/index.js +925 -0
- package/src/memory_policy.js +411 -0
- package/src/native.js +18 -0
- package/src/pipeline.js +709 -0
- package/src/recipe_json.js +80 -0
- package/src/server.js +279 -0
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +21 -0
- package/wasm/index.js +732 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/README.md
ADDED
|
@@ -0,0 +1,627 @@
|
|
|
1
|
+
# tranfi (Node.js / WASM)
|
|
2
|
+
|
|
3
|
+
Streaming-first ETL in JavaScript, powered by a native C11 core via N-API
|
|
4
|
+
(Node.js) or WASM (browsers). Tranfi processes CSV, JSONL, and text byte streams;
|
|
5
|
+
it is not an in-memory DataFrame API. Row-local operations stream, bounded
|
|
6
|
+
operators declare their limits, and full-input operators require an explicit
|
|
7
|
+
blocking or spill policy.
|
|
8
|
+
|
|
9
|
+
> **Unreleased main:** The prepared-transform API documented below targets
|
|
10
|
+
> Tranfi 0.2. Current npm 0.1.x installs do not include it; build this branch
|
|
11
|
+
> from source until 0.2 is published.
|
|
12
|
+
|
|
13
|
+
Save this example as `quickstart.mjs`:
|
|
14
|
+
|
|
15
|
+
```js
|
|
16
|
+
import tranfi from 'tranfi'
|
|
17
|
+
|
|
18
|
+
const { pipeline, codec, ops, expr } = tranfi
|
|
19
|
+
|
|
20
|
+
const result = await pipeline([
|
|
21
|
+
codec.csv(),
|
|
22
|
+
ops.filter(expr("col('age') > 25")),
|
|
23
|
+
ops.top(100, 'age'),
|
|
24
|
+
ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
|
|
25
|
+
ops.select(['name', 'age', 'label']),
|
|
26
|
+
codec.csvEncode(),
|
|
27
|
+
]).run({ input: 'name,age\nAlice,30\nBob,25\nCharlie,35\nDiana,28\n' })
|
|
28
|
+
|
|
29
|
+
console.log(result.outputText)
|
|
30
|
+
// name,age,label
|
|
31
|
+
// Charlie,35,senior
|
|
32
|
+
// Alice,30,junior
|
|
33
|
+
// Diana,28,junior
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Or use the pipe DSL for one-liners:
|
|
37
|
+
|
|
38
|
+
```js
|
|
39
|
+
const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
|
|
40
|
+
.run({ inputFile: 'data.csv' })
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Install
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
npm install tranfi
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The default install compiles the N-API addon and reports a nonzero failure if
|
|
50
|
+
the native toolchain or synchronized C sources are unavailable. For an
|
|
51
|
+
intentional WASM-only/browser installation, set
|
|
52
|
+
`TRANFI_SKIP_NATIVE_BUILD=1` and import `tranfi/wasm` explicitly. The ordinary
|
|
53
|
+
pipeline can execute with WASM, but root-entry prepared transforms require the
|
|
54
|
+
native addon.
|
|
55
|
+
|
|
56
|
+
The native addon is currently built and tested on Linux. Windows is supported
|
|
57
|
+
through the packed package's `tranfi/wasm` entry with
|
|
58
|
+
`TRANFI_SKIP_NATIVE_BUILD=1`; the release CI runs streaming and prepared-transform
|
|
59
|
+
smokes in that configuration. This is not a Windows native-addon support claim.
|
|
60
|
+
|
|
61
|
+
Use `pipeline(...)` for byte-stream ETL. The separate
|
|
62
|
+
`TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
|
|
63
|
+
lifecycle is for typed batches whose learned state must be frozen and reused.
|
|
64
|
+
Calling `.run()` collects output for convenience; use `writeTo()` or
|
|
65
|
+
`toReadable()` for large outputs.
|
|
66
|
+
|
|
67
|
+
## CLI
|
|
68
|
+
|
|
69
|
+
Installing the package also provides the `tranfi` command:
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
# Via npx (no install)
|
|
73
|
+
echo 'name,age\nAlice,30\nBob,25' | npx tranfi -q 'csv | filter "age > 25" | csv'
|
|
74
|
+
|
|
75
|
+
# Or install globally
|
|
76
|
+
npm i -g tranfi
|
|
77
|
+
tranfi -q 'csv | filter "age > 25" | top-k 100 age | csv' < data.csv
|
|
78
|
+
tranfi profile < data.csv
|
|
79
|
+
tranfi -R # list recipes
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Run `tranfi -h` for all options.
|
|
83
|
+
|
|
84
|
+
## Quick start
|
|
85
|
+
|
|
86
|
+
### Two APIs
|
|
87
|
+
|
|
88
|
+
**Builder API** -- composable, structured, and IDE-friendly:
|
|
89
|
+
|
|
90
|
+
```js
|
|
91
|
+
const p = pipeline([
|
|
92
|
+
codec.csv(),
|
|
93
|
+
ops.filter(expr("col('score') >= 80")),
|
|
94
|
+
ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
|
|
95
|
+
ops.top(10, 'score'),
|
|
96
|
+
codec.csvEncode(),
|
|
97
|
+
])
|
|
98
|
+
const result = await p.run({ inputFile: 'students.csv' })
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
**DSL strings** -- compact, suitable for CLI-like use:
|
|
102
|
+
|
|
103
|
+
```js
|
|
104
|
+
const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
|
|
105
|
+
const result = await p.run({ inputFile: 'students.csv' })
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Both produce identical pipelines under the hood.
|
|
109
|
+
|
|
110
|
+
Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate` -> `derive`, `summarise`/`summarize` -> `group-agg`, `distinct` -> `unique`, and `arrange` -> `sort`.
|
|
111
|
+
|
|
112
|
+
### Running pipelines
|
|
113
|
+
|
|
114
|
+
```js
|
|
115
|
+
// From string or Buffer
|
|
116
|
+
const result = await p.run({ input: 'name,age\nAlice,30\n' })
|
|
117
|
+
|
|
118
|
+
// From file (streamed in 64 KB chunks)
|
|
119
|
+
const result = await p.run({ inputFile: 'data.csv' })
|
|
120
|
+
|
|
121
|
+
// Access results
|
|
122
|
+
result.output // Buffer
|
|
123
|
+
result.outputText // string (UTF-8 decoded)
|
|
124
|
+
result.errors // Buffer (error channel)
|
|
125
|
+
result.stats // Buffer (pipeline stats)
|
|
126
|
+
result.statsText // string
|
|
127
|
+
result.samples // Buffer (sample channel)
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
For large outputs, drain chunks instead of collecting `result.output`:
|
|
131
|
+
|
|
132
|
+
```js
|
|
133
|
+
const { createReadStream, createWriteStream } = require("fs")
|
|
134
|
+
|
|
135
|
+
// Write to a Node Writable and wait for `drain` when the sink applies backpressure.
|
|
136
|
+
const out = createWriteStream("out.csv")
|
|
137
|
+
const result = await p.writeTo(out, { inputFile: "data.csv" })
|
|
138
|
+
|
|
139
|
+
// Or consume Tranfi output as a Node Readable.
|
|
140
|
+
for await (const chunk of p.toReadable({ inputFile: "data.csv" })) {
|
|
141
|
+
// process each output chunk
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// Existing input streams can feed the native pipeline without readFile() materialization.
|
|
145
|
+
const result2 = await p.run({ inputStream: createReadStream("data.csv") })
|
|
146
|
+
|
|
147
|
+
// .gz input files are decompressed as streaming sources by default.
|
|
148
|
+
const gz = await p.run({ inputFile: "events.jsonl.gz" })
|
|
149
|
+
const raw = await p.run({ inputFile: "events.jsonl.gz", compression: "none" })
|
|
150
|
+
const streamGz = await p.run({ inputStream: createReadStream("events.jsonl.gz"), compression: "gzip" })
|
|
151
|
+
|
|
152
|
+
// Multiple files stream sequentially through one pipeline.
|
|
153
|
+
const inputFiles = ["part-a.csv", "part-b.csv"]
|
|
154
|
+
const combined = await p.run({ inputFiles, sourceColumn: "src" })
|
|
155
|
+
|
|
156
|
+
// Low-level async iteration is still available.
|
|
157
|
+
for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
|
|
158
|
+
// process each output chunk
|
|
159
|
+
}
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
`sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. It does not remove repeated CSV headers from later files; use shards without repeated headers, `header: false`, or a pre-cleaning step when every file has its own header.
|
|
163
|
+
|
|
164
|
+
Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
|
|
165
|
+
|
|
166
|
+
For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
|
|
167
|
+
|
|
168
|
+
```js
|
|
169
|
+
// main thread
|
|
170
|
+
import { createWorkerClient } from 'tranfi/wasm/worker'
|
|
171
|
+
|
|
172
|
+
const client = createWorkerClient(new Worker(new URL('./tranfi-worker.js', import.meta.url), { type: 'module' }))
|
|
173
|
+
const result = await client.runFile('csv | filter "col(age) >= 18" | csv', file, {
|
|
174
|
+
chunkSize: 64 * 1024,
|
|
175
|
+
collectOutput: false,
|
|
176
|
+
onOutput: chunk => downloadSink.write(chunk),
|
|
177
|
+
onProgress: p => updateProgress(p),
|
|
178
|
+
signal: abortController.signal
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
// tranfi-worker.js, bundled by the app
|
|
182
|
+
import { runWorkerServer } from 'tranfi/wasm/worker'
|
|
183
|
+
runWorkerServer()
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
The Worker protocol streams chunks with transferable buffers and sends progress, stats, errors, and cancellation messages around the same WASM `create/push/pull/finish/free` API; it is not a separate IR target.
|
|
187
|
+
|
|
188
|
+
The bundled Tranfi app runner is also preview-bounded by default: file chunks are streamed into WASM, main output is drained incrementally into a table preview capped by hidden `preview_rows` (default 200), and full output text is only materialized when `collect_output` is explicitly true in the schema.
|
|
189
|
+
|
|
190
|
+
### Prepared reusable transforms
|
|
191
|
+
|
|
192
|
+
WASM cancellation tokens created with `createTransformCancelToken()` expose a read-only `requested` boolean for host-side conversion loops. Reading a closed token fails.
|
|
193
|
+
|
|
194
|
+
Prepared transforms are a separate typed-table API for operations whose parameters must be learned from reference data. `analyze` accumulates bounded statistics over one or more batches, `finalize` freezes an immutable plan and output schema, and `apply` runs that plan either over a second pass of the original data (`fit_transform`-style) or over later compatible batches. It does not replace the byte-stream pipeline API.
|
|
195
|
+
|
|
196
|
+
The current slice accepts declared `float32`/`float64` columns. It supports numeric none/zero/constant/mean/exact-median imputation and none/standard/min-max normalization. Declared categorical columns support mode imputation with `allMissing: 'error' | 'zero'` or `impute.op: 'none'`; encoders discover finite typed categories, while the no-imputation/no-encoding combination needs no learned dictionary. A column may instead provide both branches with `kind.op: 'infer'`, `kind.rule: 'finite-integer-cardinality-v1'`, and `kind.maxCategories >= 2`: missing/NaN values are ignored, `2..maxCategories` distinct finite integers resolve categorical, while zero/one distinct value, any noninteger, or the next distinct value resolves numeric. `impute.op: 'none'` with `encode.op: 'none'` passes every finite value through and emits canonical qNaN for missing input. Mode with `encode.op: 'none'` retains its learned dictionary and rejects unseen finite values with code `108`. `encode.op: 'label'` freezes zero-based sorted ordinals and supports `unknown: 'error' | 'sentinel' | 'other'`; `encode.op: 'onehot'` emits source-ordered category blocks and supports `unknown: 'error' | 'all_zero' | 'other'`. Without categorical imputation, missing label/one-hot input follows that encoder's unknown policy. The label sentinel must be a safe integer outside the learned ordinal range; label/one-hot `other` appends the reserved ordinal/field after known categories. Generated output IDs/names and one-hot category metadata are deterministic, and collisions fail with code `102`. Exact median obeys the configured allocation and resident-state limits; inference, categorical discovery, and output expansion obey category, output-width, allocation, and resident-state limits. Limit failures use resource code `104`. Prepared-transform host-policy and spill fields are reserved but not implemented; nonempty use fails with unsupported-runtime code `113` instead of being silently ignored.
|
|
197
|
+
|
|
198
|
+
Declared categorical columns also accept a fixed, nonempty `encode.categories` array of finite, sorted, unique tags (`{ "t": "f64", "v": "4000000000000000" }` represents 2). Tags must match the input dtype; negative zero is represented as positive zero. Fixed dictionaries retain unobserved categories. Unknown training values follow the encoder policy and never vote for mode; an all-missing zero fallback must exist in the dictionary. Fixed encoding without imputation can finalize without training rows. Fixed dictionaries with kind inference, string categories, and categorical constant imputation remain unsupported.
|
|
199
|
+
|
|
200
|
+
Prepared runtime option objects reject unknown fields. Native Node accepts
|
|
201
|
+
`limits`, `cancelFlag`, `hostPolicy`, and `spillDir`; standalone WASM accepts
|
|
202
|
+
`limits`, `cancelToken`, `hostPolicy`, and `spillDir`. The host/spill fields are
|
|
203
|
+
reserved as described above, and cancellation spellings are not interchangeable.
|
|
204
|
+
|
|
205
|
+
```js
|
|
206
|
+
const tf = require('tranfi')
|
|
207
|
+
|
|
208
|
+
const recipeSpec = {
|
|
209
|
+
format: 'tranfi.transform-recipe',
|
|
210
|
+
version: 1,
|
|
211
|
+
policyVersion: 1,
|
|
212
|
+
outputDtype: 'float64',
|
|
213
|
+
semanticLimits: {
|
|
214
|
+
maxOutputColumns: 65536,
|
|
215
|
+
maxOutputElementsPerApply: 134217728
|
|
216
|
+
},
|
|
217
|
+
columns: [{
|
|
218
|
+
sourceId: 'x0',
|
|
219
|
+
kind: { op: 'declared', value: 'numeric', rule: null, maxCategories: null },
|
|
220
|
+
numeric: {
|
|
221
|
+
impute: { op: 'mean', constant: null, allMissing: 'zero' },
|
|
222
|
+
normalize: { op: 'standard', ddof: 0 }
|
|
223
|
+
},
|
|
224
|
+
categorical: null
|
|
225
|
+
}]
|
|
226
|
+
}
|
|
227
|
+
const schema = [{ id: 'x0', dtype: 'float64' }]
|
|
228
|
+
|
|
229
|
+
const recipe = tf.TransformRecipe.fromJSON(recipeSpec)
|
|
230
|
+
const analyzer = recipe.analyzer(schema)
|
|
231
|
+
analyzer.push({ rows: 3, columns: [new Float64Array([1, NaN, 3])] })
|
|
232
|
+
const plan = analyzer.finalize()
|
|
233
|
+
|
|
234
|
+
const apply = plan.apply(schema)
|
|
235
|
+
const fittedReference = apply.run({
|
|
236
|
+
rows: 3,
|
|
237
|
+
columns: [new Float64Array([1, NaN, 3])]
|
|
238
|
+
})
|
|
239
|
+
const planBytes = plan.toBytes() // canonical TFTR artifact
|
|
240
|
+
const recipeSha256 = plan.recipeSha256() // canonical recipe + input-schema identity
|
|
241
|
+
|
|
242
|
+
apply.close()
|
|
243
|
+
plan.close()
|
|
244
|
+
analyzer.close()
|
|
245
|
+
recipe.close()
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
`tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
|
|
249
|
+
|
|
250
|
+
```js
|
|
251
|
+
const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
|
|
252
|
+
const makeWorker = () => new Worker(workerUrl, { type: 'module' })
|
|
253
|
+
const client = createWorkerClient(makeWorker(), {
|
|
254
|
+
workerFactory: makeWorker,
|
|
255
|
+
terminateOnDispose: true
|
|
256
|
+
})
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
|
|
260
|
+
|
|
261
|
+
## Codecs
|
|
262
|
+
|
|
263
|
+
Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
|
|
264
|
+
|
|
265
|
+
| Method | Description |
|
|
266
|
+
|--------|-------------|
|
|
267
|
+
| `codec.csv({ delimiter, header, batchSize, repair, mode, strict, maxErrorBytes, maxRecordBytes, maxColumns, nulls, quotedNulls, skip, nMax, maxRows, comment, trimWs, skipEmptyRows, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | CSV decoder. `mode: 'strict'` fails on field-count mismatches; `repair: true` / `mode: 'repair'` emits repair diagnostics; `maxRecordBytes` bounds buffered records; `nulls: ['NA']` adds null sentinels; `skip: 2`, `nMax: 100`, `comment: '#'`, `trimWs: false`, and `skipEmptyRows: true` control row-local parsing; repair audit/raw diagnostics support privacy controls with pseudo-column `raw` |
|
|
268
|
+
| `codec.csvEncode({ delimiter })` | CSV encoder |
|
|
269
|
+
| `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | JSON Lines decoder. `onError` is `skip`, `fail`, `warn`, or `quarantine` |
|
|
270
|
+
| `codec.jsonlEncode()` | JSON Lines encoder |
|
|
271
|
+
| `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | Line-oriented text decoder (single `_line` column) |
|
|
272
|
+
| `codec.textEncode()` | Text encoder |
|
|
273
|
+
| `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
|
|
274
|
+
|
|
275
|
+
CSV treats unquoted empty fields as null by default. Pass `nulls: ['NA', 'NULL']` to add sentinel strings; pass `quotedNulls: false` when quoted sentinels like `"NA"` or `""` should remain strings. Default field-count handling is permissive for compatibility. Use `mode: 'strict'` or `strict: true` to fail on row/header width mismatches. Use `repair: true` or `mode: 'repair'` to pad/truncate and collect JSONL diagnostics in `result.errors`; `maxErrorBytes` bounds raw previews, and `auditIncludeRow`, `auditColumns`, `auditRedact`, `auditHashColumns`, `auditMaxBytes`, and `auditMaxCellBytes` govern repair audit/raw payloads with the raw preview exposed as pseudo-column `raw`. `maxRecordBytes` defaults to `67108864` bytes, caps the current record buffer before a newline is seen, and rejects with a `csv_record_too_large` diagnostic when exceeded; pass `0` to disable the guard. `maxColumns` defaults to `8192`; records above the cap reject with a bounded `csv_too_many_columns` diagnostic instead of silently dropping columns. Decoder size options are checked before execution: `batchSize` must be `1..65536`, `maxErrorBytes` must be `0..67108864`, `maxRecordBytes` must be `0..1073741824`, and `maxColumns` must be `1..65536`. `skip: 2` discards preamble records before header/schema discovery; comments are applied after skipped rows. `nMax: 100` / `maxRows: 100` keeps at most that many decoded data rows after skip/comment/header handling and uses only one counter; `nMax: 0` preserves a header-only schema batch. `comment: '#'` removes text after an unquoted marker and skips comment-only rows; quoted markers are preserved. Unquoted spaces/tabs are trimmed by default; pass `trimWs: false` to preserve them. Blank physical rows after the header are preserved as all-null rows by default; pass `skipEmptyRows: true` to drop them.
|
|
276
|
+
|
|
277
|
+
Malformed JSONL records are skipped by default. Set `onError: 'warn'` or `onError: 'quarantine'` to keep valid rows and collect JSONL diagnostics in `result.errors`; set `onError: 'fail'` to reject on the first malformed line. `maxErrorBytes` bounds the raw preview stored in diagnostics, and `maxRecordBytes` applies the same current-line guard as CSV/text.
|
|
278
|
+
|
|
279
|
+
Each JSONL record must be one complete UTF-8 JSON object. Invalid number syntax, trailing content, embedded NULs, and numbers outside the finite float64 range follow the same malformed-record policy. The encoder rejects nonfinite values. Schema order comes from the first valid record; repeated keys use their first value.
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
Cross-codec pipelines work naturally:
|
|
283
|
+
|
|
284
|
+
```js
|
|
285
|
+
// CSV in, JSONL out
|
|
286
|
+
pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
|
|
287
|
+
|
|
288
|
+
// JSONL in, CSV out
|
|
289
|
+
pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
## Operators
|
|
293
|
+
|
|
294
|
+
### Row filtering
|
|
295
|
+
|
|
296
|
+
| Method | Description |
|
|
297
|
+
|--------|-------------|
|
|
298
|
+
| `ops.filter(expr, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Keep rows matching expression; optional dropped-row audit records support row omission, column allowlists, redaction, hashes, and payload caps |
|
|
299
|
+
| `ops.head(n)` | First N rows |
|
|
300
|
+
| `ops.tail(n)` | Last N rows |
|
|
301
|
+
| `ops.skip(n)` | Skip first N rows |
|
|
302
|
+
| `ops.top(n, column, desc?)` | Top N by column value |
|
|
303
|
+
| `ops.sample(n, { seed })` | Deterministic bounded reservoir sampling; use `seed: 'random'` for nondeterministic mode |
|
|
304
|
+
| `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
|
|
305
|
+
| `ops.validate(expr, { rules, rulesFile, audit, auditLimit, maxFailures, warnFailureRate, maxFailureRate, name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Add `_valid` boolean column, keep all rows; supports inline rules or local JSON rule-suite files; bounded failure audit records support count/rate thresholds plus privacy controls |
|
|
306
|
+
| `ops.assert(expr, { action, name, message, result, aggregate, op, value, column, tolerance, rel, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Row-local data-quality rule or finish-time O(1) aggregate assertion; row-local failure side-channel records support privacy controls |
|
|
307
|
+
| `ops.quarantine(expr, { name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Route rows matching expression to `errors` and drop them from main output; row payloads support privacy controls |
|
|
308
|
+
| `ops.schema({ columns, required, nonNull, nullable, values, min, max, regex, mode='fail', result='_schema', maxRegexPatternBytes, maxRegexCellBytes, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Row-local table schema contract; fail, warn, filter, quarantine, or annotate; regex budgets default to 4096-byte patterns and 65536-byte cells; schema audit/error records support row omission, column allowlists, redaction, stable non-cryptographic hashes, and row/cell payload caps |
|
|
309
|
+
| `ops.schemaInfer({ rows })` | Bounded decoded-type/nullability schema report; defaults to 10000 sampled rows |
|
|
310
|
+
| `ops.tee({ expr, channel, columns, limit, every, name, includeRow, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Preserve main rows and write bounded JSONL row snapshots to a side channel; row payloads support privacy controls |
|
|
311
|
+
|
|
312
|
+
### Column operations
|
|
313
|
+
|
|
314
|
+
| Method | Description |
|
|
315
|
+
|--------|-------------|
|
|
316
|
+
| `ops.select(columns)` | Keep and reorder columns |
|
|
317
|
+
| `ops.relocate(columns, { before, after })` | Move columns while preserving all columns |
|
|
318
|
+
| `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
|
|
319
|
+
| `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
|
|
320
|
+
| `ops.sourceName({ result, defaultValue })` | Append the current host source path/name as a row-local string column |
|
|
321
|
+
| `ops.across(columns, { fn, functions, names, replace })` | Apply row-local functions over selected columns |
|
|
322
|
+
| `ops.cast(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Type conversion; optional bounded value/coercion audit records support privacy controls |
|
|
323
|
+
| `ops.trim(columns?)` | Strip whitespace |
|
|
324
|
+
| `ops.fillNull(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Replace nulls; optional bounded audit records support privacy controls |
|
|
325
|
+
| `ops.fillDown(columns?)` | Forward-fill nulls |
|
|
326
|
+
| `ops.clip(column, { min, max })` | Clamp numeric values |
|
|
327
|
+
| `ops.replace(column, pattern, replacement, { regex, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | String find/replace; optional bounded audit records support privacy controls |
|
|
328
|
+
| `ops.hash(columns?)` | Add `_hash` column (DJB2) |
|
|
329
|
+
| `ops.bin(column, boundaries, { missing, onTypeError })` | Discretize into bins with strict numeric-source defaults |
|
|
330
|
+
| `ops.ewma(column, alpha, { result, missing, onTypeError })` | Exponentially weighted moving average |
|
|
331
|
+
| `ops.anomaly(column, { threshold, result, missing, onTypeError })` | Streaming z-score anomaly flag |
|
|
332
|
+
| `ops.normalize(columns, { method, audit, auditLimit, missing, onTypeError, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Blocking minmax/zscore normalization; optional audit records support privacy controls |
|
|
333
|
+
| `ops.acf(column, { lags, missing, onTypeError })` | Blocking autocorrelation table |
|
|
334
|
+
|
|
335
|
+
`columns` may contain exact names or selector strings: `id:score`, `starts_with(score_)`, `ends_with(_id)`, `contains(temp)`, `matches(^score_)`, `where(numeric)`, strict `all_of(score,name)`, lenient `any_of(optional,score)`, exclusions with `!name` or `-name`, and boolean selector algebra such as `starts_with(score_)&where(numeric)`, `starts_with(score_)&!ends_with(raw)`, or `!(id:score)`. Ranges use input schema order and can be reversed. Helper matching is case-insensitive; exact names are case-sensitive. These selectors are resolved by the native schema-aware `select`, `relocate`, and `across` ops in `O(columns)` without retaining rows; SQL lowering rejects selector helpers without a known schema.
|
|
336
|
+
|
|
337
|
+
Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selected numeric columns per row; `ops.across(['name'], { functions: ['lower'], replace: false, names: '{col}_{fn}' })` appends templated columns. This is native/WASM row-local execution, not arbitrary lambdas or grouped dplyr evaluation.
|
|
338
|
+
|
|
339
|
+
### Sorting and deduplication
|
|
340
|
+
|
|
341
|
+
| Method | Description |
|
|
342
|
+
|--------|-------------|
|
|
343
|
+
| `ops.sort(columns)` | Sort rows. Prefix `-` for descending: `sort(['-age', 'name'])` |
|
|
344
|
+
| `ops.unique(columns?)` | Deduplicate on specified columns |
|
|
345
|
+
|
|
346
|
+
### Aggregation
|
|
347
|
+
|
|
348
|
+
| Method | Description |
|
|
349
|
+
|--------|-------------|
|
|
350
|
+
| `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
|
|
351
|
+
| `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
|
|
352
|
+
| `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
|
|
353
|
+
|
|
354
|
+
```js
|
|
355
|
+
// Group aggregation
|
|
356
|
+
ops.groupAgg(['city'], [
|
|
357
|
+
{ column: 'price', func: 'sum', name: 'total' },
|
|
358
|
+
{ column: 'price', func: 'avg', name: 'avg_price' },
|
|
359
|
+
{ column: '*', func: 'count', name: 'rows' },
|
|
360
|
+
])
|
|
361
|
+
```
|
|
362
|
+
|
|
363
|
+
### Sequential / window
|
|
364
|
+
|
|
365
|
+
| Method | Description |
|
|
366
|
+
|--------|-------------|
|
|
367
|
+
| `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
|
|
368
|
+
| `ops.window(column, size, func, resultOrOptions?)` | Sliding window: `avg`, `sum`, `min`, `max`; options include `result`, `missing`, `onTypeError` |
|
|
369
|
+
| `ops.rollingSum/rollingMean/rollingMin/rollingMax(column, size, { result, missing, onTypeError })` | Named trailing fixed-row numeric windows; default missing/non-numeric source fails |
|
|
370
|
+
| `ops.rollingAny/rollingAll(column, size, { result, nulls })` | Boolean trailing windows; `nulls`: `ignore`, `false`, `true`, `propagate` |
|
|
371
|
+
| `ops.lead(column, { offset, result })` | Lookahead N rows |
|
|
372
|
+
| `ops.lag(column, { offset, result })` | Previous-row shift |
|
|
373
|
+
| `ops.shift(column, { offset, result, type })` | Shift alias; `type='lag'` default or `type='lead'` |
|
|
374
|
+
| `ops.rowid(columns, { result, sorted, maxKeys })` | Global or per-key 1-based row ids |
|
|
375
|
+
| `ops.rleid(columns, { result })` | Consecutive run id by selected column(s) |
|
|
376
|
+
|
|
377
|
+
### Reshape
|
|
378
|
+
|
|
379
|
+
| Method | Description |
|
|
380
|
+
|--------|-------------|
|
|
381
|
+
| `ops.explode(column, delimiter?, { maxTokensPerRow, maxOutputRowsPerInputRow, maxOutputRowsPerBatch, maxTokenBytes })` | Split delimited string into rows with optional expansion caps |
|
|
382
|
+
| `ops.split(column, names, delimiter?)` | Split column into multiple columns |
|
|
383
|
+
| `ops.unpivot(columns, { maxOutputRowsPerInputRow, maxOutputRowsPerBatch })` | Wide to long (melt) with optional expansion caps |
|
|
384
|
+
| `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
|
|
385
|
+
|
|
386
|
+
`explode` caps fail fast with `maxTokensPerRow`, `maxOutputRowsPerInputRow`, `maxOutputRowsPerBatch`, or `maxTokenBytes` when a single row or batch would expand beyond the configured limit. `unpivot` supports `maxOutputRowsPerInputRow` and `maxOutputRowsPerBatch`.
|
|
387
|
+
|
|
388
|
+
### Date/time
|
|
389
|
+
|
|
390
|
+
| Method | Description |
|
|
391
|
+
|--------|-------------|
|
|
392
|
+
| `ops.datetime(column, extractOrOptions?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday`; accepts `{ extract, missing, onTypeError }` |
|
|
393
|
+
| `ops.dateTrunc(column, trunc, { result, missing, onTypeError })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
|
|
394
|
+
|
|
395
|
+
### Other
|
|
396
|
+
|
|
397
|
+
| Method | Description |
|
|
398
|
+
|--------|-------------|
|
|
399
|
+
| `ops.flatten()` | Flatten nested columns |
|
|
400
|
+
| `ops.interpolate(column, { method, missing, onTypeError })` | Fill nulls in a numeric column; default missing/non-numeric source fails |
|
|
401
|
+
| `ops.jsonExtract(path, result, { column, type })` | Extract JSON Pointer/simple JSONPath value into a new column |
|
|
402
|
+
| `ops.jsonFilter(path, { op, value, column, type })` | Filter rows by JSON Pointer/simple JSONPath predicate |
|
|
403
|
+
| `ops.jsonSchema(schema, { column, mode, result, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Validate JSON text with a supported JSON Schema subset; filter-mode audit records support privacy controls |
|
|
404
|
+
| `ops.jsonFlatten(fields, { column })` | Append declared JSON Pointer/simple JSONPath fields as bounded output columns |
|
|
405
|
+
| `ops.reorder(columns)` | Alias for `select` |
|
|
406
|
+
| `ops.dedup(columns?)` | Alias for `unique` |
|
|
407
|
+
|
|
408
|
+
## Expressions
|
|
409
|
+
|
|
410
|
+
Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
|
|
411
|
+
|
|
412
|
+
```js
|
|
413
|
+
ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
|
|
414
|
+
ops.derive({
|
|
415
|
+
full: expr("concat(col('first'), ' ', col('last'))"),
|
|
416
|
+
grade: expr("if(col('score')>=90, 'A', if(col('score')>=80, 'B', 'C'))"),
|
|
417
|
+
})
|
|
418
|
+
```
|
|
419
|
+
|
|
420
|
+
### Available functions
|
|
421
|
+
|
|
422
|
+
| Category | Functions |
|
|
423
|
+
|----------|-----------|
|
|
424
|
+
| Arithmetic | `+` `-` `*` `/` |
|
|
425
|
+
| Comparison | `>` `>=` `<` `<=` `==` `!=` |
|
|
426
|
+
| Logic | `and` `or` `not` |
|
|
427
|
+
| String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
|
|
428
|
+
| Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` `between(x,left,right)` `inrange(x,left,right)` |
|
|
429
|
+
| Date/time | `year(x)` `month(x)` `day(x)` `hour(x)` `minute(x)` `second(x)` `weekday(x)` `epoch(x)` `date_trunc(x,unit)` |
|
|
430
|
+
| Conditional | `if(cond,then,else)` `case_when(cond,value,...,default)` `case_match(value,key,result,...,default)` `if_any(pred,...)` `if_all(pred,...)` `coalesce(a,b,...)` `nullif(a,b)` |
|
|
431
|
+
| Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
|
|
432
|
+
|
|
433
|
+
Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
|
|
434
|
+
|
|
435
|
+
```js
|
|
436
|
+
ops.derive({
|
|
437
|
+
year: expr("year(col('date'))"),
|
|
438
|
+
monthStart: expr("date_trunc(col('date'), 'month')"),
|
|
439
|
+
})
|
|
440
|
+
```
|
|
441
|
+
|
|
442
|
+
## Recipes
|
|
443
|
+
|
|
444
|
+
Built-in named pipelines for common tasks. Use by name:
|
|
445
|
+
|
|
446
|
+
```js
|
|
447
|
+
const result = await pipeline('preview').run({ inputFile: 'data.csv' })
|
|
448
|
+
const result = await pipeline('freq').run({ inputFile: 'data.csv' })
|
|
449
|
+
```
|
|
450
|
+
|
|
451
|
+
| Recipe | Pipeline | Description |
|
|
452
|
+
|--------|----------|-------------|
|
|
453
|
+
| `profile` | `csv \| stats \| csv` | Full data profiling |
|
|
454
|
+
| `preview` | `csv \| head 10 \| csv` | First 10 rows |
|
|
455
|
+
| `schema` | `csv \| head 0 \| csv` | Column names only |
|
|
456
|
+
| `sniff` | `csv \| schema infer rows=1000 \| csv` | Bounded-memory schema/type/nullability sniff |
|
|
457
|
+
| `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
|
|
458
|
+
| `count` | `csv \| stats count \| csv` | Row count |
|
|
459
|
+
| `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
|
|
460
|
+
| `distro` | `csv \| stats min,p25,median,p75,max \| csv` | Five-number summary |
|
|
461
|
+
| `freq` | `csv \| frequency \| csv` | Value frequency |
|
|
462
|
+
| `dedup` | `csv \| dedup \| csv` | Remove duplicates |
|
|
463
|
+
| `clean` | `csv \| trim \| csv` | Trim whitespace |
|
|
464
|
+
| `sample` | `csv \| sample 100 \| csv` | Random 100 rows |
|
|
465
|
+
| `head` | `csv \| head 20 \| csv` | First 20 rows |
|
|
466
|
+
| `tail` | `csv \| tail 20 \| csv` | Last 20 rows |
|
|
467
|
+
| `csv2json` | `csv \| jsonl` | CSV to JSONL |
|
|
468
|
+
| `json2csv` | `jsonl \| csv` | JSONL to CSV |
|
|
469
|
+
| `tsv2csv` | `csv delimiter="\t" \| csv` | TSV to CSV |
|
|
470
|
+
| `csv2tsv` | `csv \| csv delimiter="\t"` | CSV to TSV |
|
|
471
|
+
| `look` | `csv \| table` | Pretty-print table |
|
|
472
|
+
| `histogram` | `csv \| stats hist \| csv` | Distribution histograms |
|
|
473
|
+
| `hash` | `csv \| hash \| csv` | Row hash for change detection |
|
|
474
|
+
| `samples` | `csv \| stats sample \| csv` | Sample values per column |
|
|
475
|
+
|
|
476
|
+
List all recipes programmatically:
|
|
477
|
+
|
|
478
|
+
```js
|
|
479
|
+
import { recipes } from 'tranfi'
|
|
480
|
+
|
|
481
|
+
for (const r of await recipes()) {
|
|
482
|
+
console.log(`${r.name.padEnd(15)} ${r.description}`)
|
|
483
|
+
}
|
|
484
|
+
```
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
## Memory policy
|
|
488
|
+
|
|
489
|
+
Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
|
|
490
|
+
|
|
491
|
+
```js
|
|
492
|
+
await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
|
|
493
|
+
```
|
|
494
|
+
|
|
495
|
+
For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
|
|
496
|
+
|
|
497
|
+
```js
|
|
498
|
+
await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
|
|
499
|
+
```
|
|
500
|
+
|
|
501
|
+
Native Node execution supports spill-backed `sort`, capped unsorted `pivot`, unsorted `unique`/`dedup`, unsorted `group-agg`, capped unsorted `join` inner/left, unsorted `semi-join`/`anti-join`, unsorted set operations, and duplicate-eliminating `union` when `spillDir` is provided. `run()`, `iterChunks()`, `toReadable()`, and `writeTo()` drain finish-time merge output through N-API `finishStep()` instead of waiting for one whole `finish()`. Standalone WASM supports `allowBlocking` and `memory`, but rejects `spillDir` because browser/WASM spill storage would occupy WASM memory. Use the Node native addon, CLI/direct C, or `{ engine: 'duckdb' }` for external spill; with DuckDB, `memory` maps to `memory_limit` and `spillDir` maps to `temp_directory`.
|
|
502
|
+
|
|
503
|
+
Plan-internal file reads are denied by default in Node and WASM. Pass `allowFs: true` for trusted local lookup files used by `join`, set operations, `union`, or `stack`; pass both `allowFs: true` and `allowRulesFile: true` for `validate rules_file=...`; pass `workspaceRoot` to pin resolved core plan paths inside a trusted directory. Supplying `spillDir` opts into local filesystem spill for Node native execution; standalone WASM still rejects spill. `inputFile` and `inputFiles` are host source adapters and are not controlled by `allowFs`.
|
|
504
|
+
|
|
505
|
+
## DuckDB engine
|
|
506
|
+
|
|
507
|
+
Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
|
|
508
|
+
|
|
509
|
+
```bash
|
|
510
|
+
npm install duckdb
|
|
511
|
+
```
|
|
512
|
+
|
|
513
|
+
```js
|
|
514
|
+
import { pipeline, compileToSql } from 'tranfi'
|
|
515
|
+
|
|
516
|
+
// Run a pipeline via DuckDB
|
|
517
|
+
const result = await pipeline('csv | filter "age > 25" | sort -age | csv', { engine: 'duckdb' })
|
|
518
|
+
.run({ inputFile: 'data.csv' })
|
|
519
|
+
|
|
520
|
+
// Or with string/Buffer input
|
|
521
|
+
const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
|
|
522
|
+
.run({ input: csvString })
|
|
523
|
+
```
|
|
524
|
+
|
|
525
|
+
### SQL transpilation
|
|
526
|
+
|
|
527
|
+
Generate SQL directly from DSL strings:
|
|
528
|
+
|
|
529
|
+
```js
|
|
530
|
+
const sql = await compileToSql(
|
|
531
|
+
'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
|
|
532
|
+
{ dialect: 'duckdb' }
|
|
533
|
+
)
|
|
534
|
+
console.log(sql)
|
|
535
|
+
// WITH
|
|
536
|
+
// step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
|
|
537
|
+
// step_2 AS (SELECT * FROM step_1 ORDER BY "age" DESC LIMIT 10)
|
|
538
|
+
// SELECT * FROM step_2
|
|
539
|
+
```
|
|
540
|
+
|
|
541
|
+
DuckDB is the only implemented SQL dialect today. `dialect: 'sqlite'` and
|
|
542
|
+
`dialect: 'postgres'` are recognized but rejected until those dialects have
|
|
543
|
+
their own compatibility tests and SQL-generation rules.
|
|
544
|
+
|
|
545
|
+
### Browser (WASM + DuckDB-WASM)
|
|
546
|
+
|
|
547
|
+
In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
|
|
548
|
+
|
|
549
|
+
```js
|
|
550
|
+
import createTranfi from 'tranfi/wasm'
|
|
551
|
+
import * as duckdb from '@duckdb/duckdb-wasm'
|
|
552
|
+
|
|
553
|
+
const tf = await createTranfi()
|
|
554
|
+
|
|
555
|
+
// SQL generation (synchronous, no DuckDB needed)
|
|
556
|
+
const sql = tf.compileToSql('csv | filter "age > 25" | csv', { dialect: 'duckdb' })
|
|
557
|
+
|
|
558
|
+
// Full execution with DuckDB-WASM
|
|
559
|
+
const db = new duckdb.AsyncDuckDB(...)
|
|
560
|
+
await db.instantiate(...)
|
|
561
|
+
|
|
562
|
+
const result = await tf.runDuckDB(db, 'csv | filter "age > 25" | csv', csvData)
|
|
563
|
+
console.log(result.outputText) // CSV output
|
|
564
|
+
console.log(result.rows) // Array of row objects
|
|
565
|
+
```
|
|
566
|
+
|
|
567
|
+
`runDuckDB` accepts string, `Uint8Array`, or `File` objects as data input.
|
|
568
|
+
|
|
569
|
+
## Advanced
|
|
570
|
+
|
|
571
|
+
### DSL compilation
|
|
572
|
+
|
|
573
|
+
```js
|
|
574
|
+
import { compileDsl, saveRecipe, loadRecipe } from 'tranfi'
|
|
575
|
+
|
|
576
|
+
// Compile DSL to JSON plan
|
|
577
|
+
const json = await compileDsl('csv | filter "col(age) > 25" | sort -age | csv')
|
|
578
|
+
|
|
579
|
+
// Save / load recipes
|
|
580
|
+
await saveRecipe([codec.csv(), ops.head(10), codec.csvEncode()], 'preview.tranfi')
|
|
581
|
+
const p = await loadRecipe('preview.tranfi')
|
|
582
|
+
const result = await p.run({ inputFile: 'data.csv' })
|
|
583
|
+
```
|
|
584
|
+
|
|
585
|
+
### Side channels
|
|
586
|
+
|
|
587
|
+
Every pipeline produces four output channels:
|
|
588
|
+
|
|
589
|
+
- **output** -- main pipeline result
|
|
590
|
+
- **errors** -- rows that failed processing
|
|
591
|
+
- **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
|
|
592
|
+
- **samples** -- reserved for sampling operators
|
|
593
|
+
|
|
594
|
+
```js
|
|
595
|
+
const result = await p.run({ inputFile: 'data.csv' })
|
|
596
|
+
console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
|
|
597
|
+
```
|
|
598
|
+
|
|
599
|
+
### Pipeline from JSON
|
|
600
|
+
|
|
601
|
+
```js
|
|
602
|
+
const p = await loadRecipe({
|
|
603
|
+
steps: [
|
|
604
|
+
{ op: 'codec.csv.decode', args: {} },
|
|
605
|
+
{ op: 'head', args: { n: 5 } },
|
|
606
|
+
{ op: 'codec.csv.encode', args: {} },
|
|
607
|
+
]
|
|
608
|
+
})
|
|
609
|
+
```
|
|
610
|
+
|
|
611
|
+
### Backend selection
|
|
612
|
+
|
|
613
|
+
The package automatically selects the best backend:
|
|
614
|
+
|
|
615
|
+
1. **N-API** (Node.js) -- native C addon, fastest, used when available
|
|
616
|
+
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~500 KB single-file
|
|
617
|
+
3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
|
|
618
|
+
|
|
619
|
+
```js
|
|
620
|
+
// Force WASM backend (e.g., for testing)
|
|
621
|
+
import createTranfi from 'tranfi/wasm'
|
|
622
|
+
const tf = await createTranfi()
|
|
623
|
+
```
|
|
624
|
+
|
|
625
|
+
## Architecture
|
|
626
|
+
|
|
627
|
+
The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. `run({ inputFile })` streams files with `createReadStream()` and drains native/WASM main output after each push and each incremental finish boundary; by default it still collects the final output for convenience. `.gz` input files are decompressed through a `zlib.createGunzip()` source transform; use `compression: "none"` to force raw bytes or `compression: "gzip"` to force gzip for `inputFile`/`inputStream`. Use `inputStream`, `toReadable()`, `writeTo()`, `onOutput` with `collectOutput: false`, or `iterChunks()` for large inputs/outputs and backpressure-aware sinks. Native execution is strict by default: row-local and bounded-state operators stream, blocking operators require `allowBlocking: true` or supported `spillDir`, capped key-state plans can be checked with `memory: "64MB"`, and core plan file reads require explicit host-policy options.
|