tranfi 0.1.2 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +443 -51
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +352 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +81 -41
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/README.md
CHANGED
|
@@ -1,14 +1,23 @@
|
|
|
1
1
|
# tranfi (Node.js / WASM)
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Tranfi transforms CSV, JSON Lines and plain text as streams. In Node.js it runs
|
|
4
|
+
on a native C core; in the browser it runs on WebAssembly. Data flows through in
|
|
5
|
+
chunks, so most pipelines use a small, fixed amount of memory however large the
|
|
6
|
+
input is. Operations that need the whole input, such as `sort`, must be allowed
|
|
7
|
+
explicitly. Tranfi works on byte streams; it is not an in-memory DataFrame
|
|
8
|
+
library.
|
|
9
|
+
|
|
10
|
+
Save this example as `quickstart.mjs`:
|
|
4
11
|
|
|
5
12
|
```js
|
|
6
|
-
import
|
|
13
|
+
import tranfi from 'tranfi'
|
|
14
|
+
|
|
15
|
+
const { pipeline, codec, ops, expr } = tranfi
|
|
7
16
|
|
|
8
17
|
const result = await pipeline([
|
|
9
18
|
codec.csv(),
|
|
10
19
|
ops.filter(expr("col('age') > 25")),
|
|
11
|
-
ops.
|
|
20
|
+
ops.top(100, 'age'),
|
|
12
21
|
ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
|
|
13
22
|
ops.select(['name', 'age', 'label']),
|
|
14
23
|
codec.csvEncode(),
|
|
@@ -23,8 +32,9 @@ console.log(result.outputText)
|
|
|
23
32
|
|
|
24
33
|
Or use the pipe DSL for one-liners:
|
|
25
34
|
|
|
35
|
+
<!-- readme-test: js-api -->
|
|
26
36
|
```js
|
|
27
|
-
const result = await pipeline('csv | filter "col(age) > 25" |
|
|
37
|
+
const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
|
|
28
38
|
.run({ inputFile: 'data.csv' })
|
|
29
39
|
```
|
|
30
40
|
|
|
@@ -34,7 +44,25 @@ const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
|
|
|
34
44
|
npm install tranfi
|
|
35
45
|
```
|
|
36
46
|
|
|
37
|
-
|
|
47
|
+
Installation compiles Tranfi's C core as a native Node.js addon, so you need a C
|
|
48
|
+
compiler. If that compile fails, the install fails.
|
|
49
|
+
|
|
50
|
+
In the browser, import `tranfi/wasm`: it provides pipelines and prepared
|
|
51
|
+
transforms without the native addon. To install only the WebAssembly build, run
|
|
52
|
+
`TRANFI_SKIP_NATIVE_BUILD=1 npm install tranfi`. The root `tranfi` entry then
|
|
53
|
+
runs pipelines through WebAssembly, but its prepared-transform classes need the
|
|
54
|
+
native addon, so use `tranfi/wasm` for those.
|
|
55
|
+
|
|
56
|
+
| | Linux | Windows | macOS |
|
|
57
|
+
|---|---|---|---|
|
|
58
|
+
| Native addon | Tested | Not tested | Not tested |
|
|
59
|
+
| WebAssembly (`tranfi/wasm`) | Tested in Node.js and Chromium | Tested in Node.js | Not tested |
|
|
60
|
+
|
|
61
|
+
Use `pipeline(...)` for byte-stream ETL. The separate
|
|
62
|
+
`TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
|
|
63
|
+
lifecycle is for typed batches whose learned state must be frozen and reused.
|
|
64
|
+
Calling `.run()` collects output for convenience; use `writeTo()` or
|
|
65
|
+
`toReadable()` for large outputs.
|
|
38
66
|
|
|
39
67
|
## CLI
|
|
40
68
|
|
|
@@ -42,30 +70,34 @@ Installing the package also provides the `tranfi` command:
|
|
|
42
70
|
|
|
43
71
|
```bash
|
|
44
72
|
# Via npx (no install)
|
|
45
|
-
|
|
73
|
+
printf 'name,age\nAlice,30\nBob,25\n' | npx tranfi -q 'csv | filter "age > 25" | csv'
|
|
46
74
|
|
|
47
75
|
# Or install globally
|
|
48
76
|
npm i -g tranfi
|
|
49
|
-
tranfi 'csv | filter "age > 25" |
|
|
77
|
+
tranfi -q 'csv | filter "age > 25" | top-k 100 age | csv' < data.csv
|
|
50
78
|
tranfi profile < data.csv
|
|
51
79
|
tranfi -R # list recipes
|
|
52
80
|
```
|
|
53
81
|
|
|
54
|
-
Run `tranfi -h` for
|
|
82
|
+
Run `tranfi -h` for the npm CLI options. Use `--allow-blocking` for known-small
|
|
83
|
+
full-input operations, `--memory max:64MB` for a memory policy, and
|
|
84
|
+
`--spill-dir DIR` for an existing private spill directory. `--stats-json FILE`
|
|
85
|
+
writes the stats channel; `--target json|sql` compiles without executing.
|
|
86
|
+
The detailed `--explain` report requires the standalone C CLI built from source.
|
|
55
87
|
|
|
56
88
|
## Quick start
|
|
57
89
|
|
|
58
90
|
### Two APIs
|
|
59
91
|
|
|
60
|
-
**Builder API** -- composable,
|
|
92
|
+
**Builder API** -- composable, structured, and IDE-friendly:
|
|
61
93
|
|
|
94
|
+
<!-- readme-test: js-api -->
|
|
62
95
|
```js
|
|
63
96
|
const p = pipeline([
|
|
64
97
|
codec.csv(),
|
|
65
98
|
ops.filter(expr("col('score') >= 80")),
|
|
66
99
|
ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
|
|
67
|
-
ops.
|
|
68
|
-
ops.head(10),
|
|
100
|
+
ops.top(10, 'score'),
|
|
69
101
|
codec.csvEncode(),
|
|
70
102
|
])
|
|
71
103
|
const result = await p.run({ inputFile: 'students.csv' })
|
|
@@ -73,21 +105,25 @@ const result = await p.run({ inputFile: 'students.csv' })
|
|
|
73
105
|
|
|
74
106
|
**DSL strings** -- compact, suitable for CLI-like use:
|
|
75
107
|
|
|
108
|
+
<!-- readme-test: js-api -->
|
|
76
109
|
```js
|
|
77
|
-
const p = pipeline('csv | filter "col(score) >= 80" |
|
|
110
|
+
const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
|
|
78
111
|
const result = await p.run({ inputFile: 'students.csv' })
|
|
79
112
|
```
|
|
80
113
|
|
|
81
114
|
Both produce identical pipelines under the hood.
|
|
82
115
|
|
|
116
|
+
Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate` -> `derive`, `summarise`/`summarize` -> `group-agg`, `distinct` -> `unique`, and `arrange` -> `sort`.
|
|
117
|
+
|
|
83
118
|
### Running pipelines
|
|
84
119
|
|
|
120
|
+
<!-- readme-test: js-pipeline -->
|
|
85
121
|
```js
|
|
86
122
|
// From string or Buffer
|
|
87
123
|
const result = await p.run({ input: 'name,age\nAlice,30\n' })
|
|
88
124
|
|
|
89
125
|
// From file (streamed in 64 KB chunks)
|
|
90
|
-
const
|
|
126
|
+
const fileResult = await p.run({ inputFile: 'data.csv' })
|
|
91
127
|
|
|
92
128
|
// Access results
|
|
93
129
|
result.output // Buffer
|
|
@@ -98,22 +134,235 @@ result.statsText // string
|
|
|
98
134
|
result.samples // Buffer (sample channel)
|
|
99
135
|
```
|
|
100
136
|
|
|
137
|
+
For large outputs, drain chunks instead of collecting `result.output`:
|
|
138
|
+
|
|
139
|
+
<!-- readme-test: js-pipeline -->
|
|
140
|
+
```js
|
|
141
|
+
import { createReadStream, createWriteStream } from 'node:fs'
|
|
142
|
+
|
|
143
|
+
// Write to a Node Writable and wait for `drain` when the sink applies backpressure.
|
|
144
|
+
const out = createWriteStream("out.csv")
|
|
145
|
+
const result = await p.writeTo(out, { inputFile: "data.csv" })
|
|
146
|
+
|
|
147
|
+
// Or consume Tranfi output as a Node Readable.
|
|
148
|
+
for await (const chunk of p.toReadable({ inputFile: "data.csv" })) {
|
|
149
|
+
// process each output chunk
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// Existing input streams can feed the native pipeline without readFile() materialization.
|
|
153
|
+
const result2 = await p.run({ inputStream: createReadStream("data.csv") })
|
|
154
|
+
|
|
155
|
+
// .gz input files are decompressed as streaming sources by default.
|
|
156
|
+
const gz = await p.run({ inputFile: "events.jsonl.gz" })
|
|
157
|
+
const raw = await p.run({ inputFile: "events.jsonl.gz", compression: "none" })
|
|
158
|
+
const streamGz = await p.run({ inputStream: createReadStream("events.jsonl.gz"), compression: "gzip" })
|
|
159
|
+
|
|
160
|
+
// Multiple files stream sequentially through one pipeline.
|
|
161
|
+
const inputFiles = ["part-a.csv", "part-b.csv"]
|
|
162
|
+
const combined = await p.run({ inputFiles, sourceColumn: "src" })
|
|
163
|
+
|
|
164
|
+
// Low-level async iteration is still available.
|
|
165
|
+
for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
|
|
166
|
+
// process each output chunk
|
|
167
|
+
}
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
`sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. Repeated headers are kept by default. Use `codec.csv({ skipRepeatedHeader: true })` to skip a later file's first non-comment record when it exactly matches the original header; matching rows inside a file remain data.
|
|
171
|
+
|
|
172
|
+
Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
|
|
173
|
+
|
|
174
|
+
For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
|
|
175
|
+
|
|
176
|
+
<!-- readme-test: browser -->
|
|
177
|
+
```js
|
|
178
|
+
// main thread
|
|
179
|
+
import { createWorkerClient } from 'tranfi/wasm/worker'
|
|
180
|
+
|
|
181
|
+
const client = createWorkerClient(new Worker(new URL('./tranfi-worker.js', import.meta.url), { type: 'module' }))
|
|
182
|
+
const result = await client.runFile('csv | filter "col(age) >= 18" | csv', file, {
|
|
183
|
+
chunkSize: 64 * 1024,
|
|
184
|
+
collectOutput: false,
|
|
185
|
+
onOutput: chunk => downloadSink.write(chunk),
|
|
186
|
+
onProgress: p => updateProgress(p),
|
|
187
|
+
signal: abortController.signal
|
|
188
|
+
})
|
|
189
|
+
|
|
190
|
+
// tranfi-worker.js, bundled by the app
|
|
191
|
+
import { runWorkerServer } from 'tranfi/wasm/worker'
|
|
192
|
+
runWorkerServer()
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
The Worker protocol streams chunks with transferable buffers and sends progress, stats, errors, and cancellation messages around the same WASM `create/push/pull/finish/free` API; it is not a separate IR target.
|
|
196
|
+
|
|
197
|
+
The bundled Tranfi app runner is also preview-bounded by default: file chunks are streamed into WASM, main output is drained incrementally into a table preview capped by hidden `preview_rows` (default 200), and full output text is only materialized when `collect_output` is explicitly true in the schema.
|
|
198
|
+
|
|
199
|
+
### Prepared reusable transforms
|
|
200
|
+
|
|
201
|
+
Learn imputation, scaling or category mappings from reference data, then apply
|
|
202
|
+
the same plan to new batches. Analyze the reference batches, finalize the plan,
|
|
203
|
+
and apply it. Replaying the reference data through that plan gives a second-pass
|
|
204
|
+
`fit_transform` workflow.
|
|
205
|
+
|
|
206
|
+
Inputs are typed `float32`/`float64` columns. This example learns mean imputation
|
|
207
|
+
and standard scaling, applies them to the reference batch, and exports the plan:
|
|
208
|
+
|
|
209
|
+
```js
|
|
210
|
+
import tf from 'tranfi'
|
|
211
|
+
|
|
212
|
+
const recipeSpec = {
|
|
213
|
+
format: 'tranfi.transform-recipe',
|
|
214
|
+
version: 1,
|
|
215
|
+
policyVersion: 1,
|
|
216
|
+
outputDtype: 'float64',
|
|
217
|
+
semanticLimits: {
|
|
218
|
+
maxOutputColumns: 65536,
|
|
219
|
+
maxOutputElementsPerApply: 134217728
|
|
220
|
+
},
|
|
221
|
+
columns: [{
|
|
222
|
+
sourceId: 'x0',
|
|
223
|
+
kind: { op: 'declared', value: 'numeric', rule: null, maxCategories: null },
|
|
224
|
+
numeric: {
|
|
225
|
+
impute: { op: 'mean', constant: null, allMissing: 'zero' },
|
|
226
|
+
normalize: { op: 'standard', ddof: 0 }
|
|
227
|
+
},
|
|
228
|
+
categorical: null
|
|
229
|
+
}]
|
|
230
|
+
}
|
|
231
|
+
const schema = [{ id: 'x0', dtype: 'float64' }]
|
|
232
|
+
|
|
233
|
+
const recipe = tf.TransformRecipe.fromJSON(recipeSpec)
|
|
234
|
+
const analyzer = recipe.analyzer(schema)
|
|
235
|
+
analyzer.push({ rows: 3, columns: [new Float64Array([1, NaN, 3])] })
|
|
236
|
+
const plan = analyzer.finalize()
|
|
237
|
+
|
|
238
|
+
const apply = plan.apply(schema)
|
|
239
|
+
const fittedReference = apply.run({
|
|
240
|
+
rows: 3,
|
|
241
|
+
columns: [new Float64Array([1, NaN, 3])]
|
|
242
|
+
})
|
|
243
|
+
const planBytes = plan.toBytes() // canonical TFTR artifact
|
|
244
|
+
const recipeSha256 = plan.recipeSha256() // canonical recipe + input-schema identity
|
|
245
|
+
|
|
246
|
+
apply.close()
|
|
247
|
+
plan.close()
|
|
248
|
+
analyzer.close()
|
|
249
|
+
recipe.close()
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
<details markdown="1">
|
|
253
|
+
<summary>Available methods, missing values and kind inference</summary>
|
|
254
|
+
|
|
255
|
+
Recipe fields use the same names in all bindings.
|
|
256
|
+
|
|
257
|
+
| Recipe field | Choices / behavior |
|
|
258
|
+
|--------------|--------------------|
|
|
259
|
+
| Numeric `impute.op` | `none`, `zero`, `constant`, `mean`, exact `median` |
|
|
260
|
+
| Numeric `normalize.op` | `none`, `standard`, `minmax` |
|
|
261
|
+
| Categorical `impute.op` | `none` or mode; mode supports `allMissing` of `error` or `zero` |
|
|
262
|
+
| Categorical `encode.op` | `none`, `label`, `onehot` |
|
|
263
|
+
| Label unknown policy | `error`, `sentinel`, `other` |
|
|
264
|
+
| One-hot unknown policy | `error`, `all_zero`, `other` |
|
|
265
|
+
|
|
266
|
+
**No imputation or encoding:** finite values pass through; missing input becomes
|
|
267
|
+
canonical qNaN. This combination needs no learned category dictionary.
|
|
268
|
+
Mode imputation with no encoding retains a dictionary and rejects unseen finite
|
|
269
|
+
values with error `108`.
|
|
270
|
+
|
|
271
|
+
**Encoding:** label ordinals are zero-based and sorted; one-hot blocks follow
|
|
272
|
+
source order. A label sentinel must be a safe integer outside the known ordinal
|
|
273
|
+
range. `other` appends an ordinal or field after known categories. Without
|
|
274
|
+
imputation, missing input follows the encoder's unknown policy.
|
|
275
|
+
|
|
276
|
+
Generated IDs, names and category metadata are deterministic. Collisions fail
|
|
277
|
+
with error `102`.
|
|
278
|
+
|
|
279
|
+
**Kind inference:** configure both numeric and categorical branches, set
|
|
280
|
+
`kind.op` to `infer`, `kind.rule` to `finite-integer-cardinality-v1`, and
|
|
281
|
+
`kind.maxCategories` to at least `2`. Missing/NaN values are ignored.
|
|
282
|
+
|
|
283
|
+
| Observed reference values | Selected branch |
|
|
284
|
+
|---------------------------|-----------------|
|
|
285
|
+
| `2..maxCategories` distinct finite integers | Categorical |
|
|
286
|
+
| Zero or one distinct value, any noninteger, or more than `maxCategories` | Numeric |
|
|
287
|
+
|
|
288
|
+
</details>
|
|
289
|
+
|
|
290
|
+
<details markdown="1">
|
|
291
|
+
<summary>Fixed category dictionaries</summary>
|
|
292
|
+
|
|
293
|
+
Set `encode.categories` to a nonempty, sorted, unique array of finite typed tags.
|
|
294
|
+
For example, `{ "t": "f64", "v": "4000000000000000" }` represents `2`.
|
|
295
|
+
Tags must match the input dtype; represent negative zero as positive zero.
|
|
296
|
+
|
|
297
|
+
- Unobserved categories stay in the dictionary.
|
|
298
|
+
- Unknown training values follow the encoder policy and never vote for mode.
|
|
299
|
+
- An all-missing zero fallback must exist in the dictionary.
|
|
300
|
+
- Fixed encoding without imputation can finalize without training rows.
|
|
301
|
+
|
|
302
|
+
String categories, categorical constant imputation and fixed dictionaries
|
|
303
|
+
combined with kind inference are not supported.
|
|
304
|
+
|
|
305
|
+
</details>
|
|
306
|
+
|
|
307
|
+
<details markdown="1">
|
|
308
|
+
<summary>Limits, errors and cancellation</summary>
|
|
309
|
+
|
|
310
|
+
Exact median obeys allocation and resident-state limits. Inference, category
|
|
311
|
+
discovery and output expansion also enforce category/output-width limits.
|
|
312
|
+
Resource-limit failures use error `104`.
|
|
313
|
+
|
|
314
|
+
Host-policy and spill settings are reserved. A non-null host policy or nonempty
|
|
315
|
+
spill path fails with unsupported-runtime error `113`.
|
|
316
|
+
|
|
317
|
+
Runtime options reject unknown fields:
|
|
318
|
+
|
|
319
|
+
| Runtime | Options |
|
|
320
|
+
|---------|---------|
|
|
321
|
+
| Native Node | `limits`, `cancelFlag`, `hostPolicy`, `spillDir` |
|
|
322
|
+
| Standalone WASM | `limits`, `cancelToken`, `hostPolicy`, `spillDir` |
|
|
323
|
+
|
|
324
|
+
Cancellation spellings are not interchangeable. WASM tokens created with
|
|
325
|
+
`createTransformCancelToken()` expose a read-only `requested` boolean for host
|
|
326
|
+
conversion loops; reading a closed token fails.
|
|
327
|
+
|
|
328
|
+
`tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
|
|
329
|
+
|
|
330
|
+
<!-- readme-test: browser -->
|
|
331
|
+
```js
|
|
332
|
+
const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
|
|
333
|
+
const makeWorker = () => new Worker(workerUrl, { type: 'module' })
|
|
334
|
+
const client = createWorkerClient(makeWorker(), {
|
|
335
|
+
workerFactory: makeWorker,
|
|
336
|
+
terminateOnDispose: true
|
|
337
|
+
})
|
|
338
|
+
```
|
|
339
|
+
|
|
340
|
+
All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
|
|
341
|
+
|
|
342
|
+
</details>
|
|
343
|
+
|
|
101
344
|
## Codecs
|
|
102
345
|
|
|
103
|
-
|
|
346
|
+
Start with a decoder and finish with an encoder. Input and output formats can differ.
|
|
104
347
|
|
|
105
|
-
|
|
|
106
|
-
|
|
107
|
-
| `codec.csv(
|
|
108
|
-
| `codec.
|
|
109
|
-
| `codec.
|
|
110
|
-
| `codec.
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
348
|
+
| Format | Decoder | Encoder |
|
|
349
|
+
|--------|---------|---------|
|
|
350
|
+
| CSV | `codec.csv(options)` | `codec.csvEncode({ delimiter })` |
|
|
351
|
+
| JSON Lines | `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | `codec.jsonlEncode()` |
|
|
352
|
+
| Text (one `_line` column) | `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | `codec.textEncode()` |
|
|
353
|
+
| Markdown table | — | `codec.tableEncode({ maxWidth, maxRows })` |
|
|
354
|
+
|
|
355
|
+
**CSV defaults to permissive field-count handling.** Choose `mode: 'strict'`
|
|
356
|
+
(or `strict: true`) to reject width mismatches. Choose `repair: true`
|
|
357
|
+
(or `mode: 'repair'`) to pad/truncate rows and collect diagnostics in `result.errors`.
|
|
358
|
+
|
|
359
|
+
**Malformed JSONL is skipped by default.** Set `onError` to `fail` to stop,
|
|
360
|
+
`warn` to keep valid rows with diagnostics, or `quarantine` to route bad records
|
|
361
|
+
to error diagnostics. Read those diagnostics from `result.errors`.
|
|
114
362
|
|
|
115
363
|
Cross-codec pipelines work naturally:
|
|
116
364
|
|
|
365
|
+
<!-- readme-test: js-api -->
|
|
117
366
|
```js
|
|
118
367
|
// CSV in, JSONL out
|
|
119
368
|
pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
|
|
@@ -122,36 +371,103 @@ pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
|
|
|
122
371
|
pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
|
|
123
372
|
```
|
|
124
373
|
|
|
374
|
+
<details markdown="1">
|
|
375
|
+
<summary>CSV options for nulls, row limits and whitespace</summary>
|
|
376
|
+
|
|
377
|
+
| Need | Option | Behavior |
|
|
378
|
+
|------|--------|----------|
|
|
379
|
+
| Field separator / header | `delimiter`, `header` | Configure CSV decoding; encoder accepts `delimiter` |
|
|
380
|
+
| Add null markers | `nulls: ['NA', 'NULL']` | Adds to default unquoted-empty-field null handling |
|
|
381
|
+
| Preserve quoted markers | `quotedNulls: false` | Keeps quoted `"NA"` and `""` as strings |
|
|
382
|
+
| Skip a preamble | `skip: 2` | Before schema/header discovery; comments are handled afterward |
|
|
383
|
+
| Limit rows | `nMax: 100` or `maxRows: 100` | One shared counter after skip/comment/header handling |
|
|
384
|
+
| Keep only the schema | `nMax: 0` | Preserves a header-only batch |
|
|
385
|
+
| Comments | `comment: '#'` | Removes text after unquoted markers and skips comment-only rows; quoted markers remain |
|
|
386
|
+
| Preserve spaces/tabs | `trimWs: false` | Disables default trimming of unquoted values |
|
|
387
|
+
| Drop blank rows | `skipEmptyRows: true` | Otherwise blank physical rows after the header become all-null rows |
|
|
388
|
+
| Read CSV shards | `skipRepeatedHeader: true` | Skips matching headers only at explicit file boundaries |
|
|
389
|
+
|
|
390
|
+
</details>
|
|
391
|
+
|
|
392
|
+
<details markdown="1">
|
|
393
|
+
<summary>Decoder limits and repair diagnostics</summary>
|
|
394
|
+
|
|
395
|
+
All decoder size options are checked integers:
|
|
396
|
+
|
|
397
|
+
| Option | Range | Behavior |
|
|
398
|
+
|--------|-------|----------|
|
|
399
|
+
| `batchSize` | `1..65536` | Rows per batch |
|
|
400
|
+
| `maxErrorBytes` | `0..67108864` | Bounds raw diagnostic previews |
|
|
401
|
+
| `maxRecordBytes` | `0..1073741824` | Default `67108864` bytes (64 MiB); `0` disables the record guard |
|
|
402
|
+
| `maxColumns` (CSV) | `1..65536` | Default `8192`; overflow fails with `csv_too_many_columns` |
|
|
403
|
+
|
|
404
|
+
Record limits apply before a newline is seen, including CSV, JSONL and text.
|
|
405
|
+
CSV record overflow fails with `csv_record_too_large`.
|
|
406
|
+
|
|
407
|
+
Repair audit/raw payloads accept `auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes`. The raw preview is exposed to
|
|
408
|
+
those controls as pseudo-column `raw`. CSV repair also accepts
|
|
409
|
+
`audit, auditLimit`.
|
|
410
|
+
|
|
411
|
+
</details>
|
|
412
|
+
|
|
413
|
+
<details markdown="1">
|
|
414
|
+
<summary>JSONL record and schema rules</summary>
|
|
415
|
+
|
|
416
|
+
Each record must be one complete UTF-8 JSON object. Invalid number syntax,
|
|
417
|
+
trailing content, embedded NULs and numbers outside finite float64 range follow
|
|
418
|
+
the configured malformed-record policy. The encoder rejects nonfinite values.
|
|
419
|
+
|
|
420
|
+
The first valid record determines schema order. Repeated keys use their first
|
|
421
|
+
value. `maxErrorBytes` bounds diagnostic previews, and `maxRecordBytes` caps the current line.
|
|
422
|
+
|
|
423
|
+
</details>
|
|
424
|
+
|
|
125
425
|
## Operators
|
|
126
426
|
|
|
127
427
|
### Row filtering
|
|
128
428
|
|
|
129
429
|
| Method | Description |
|
|
130
430
|
|--------|-------------|
|
|
131
|
-
| `ops.filter(expr)` | Keep rows matching expression |
|
|
431
|
+
| `ops.filter(expr, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Keep rows matching expression; optional dropped-row audit records support row omission, column allowlists, redaction, hashes, and payload caps |
|
|
132
432
|
| `ops.head(n)` | First N rows |
|
|
133
433
|
| `ops.tail(n)` | Last N rows |
|
|
134
434
|
| `ops.skip(n)` | Skip first N rows |
|
|
135
435
|
| `ops.top(n, column, desc?)` | Top N by column value |
|
|
136
|
-
| `ops.sample(n)` |
|
|
436
|
+
| `ops.sample(n, { seed })` | Deterministic bounded reservoir sampling; use `seed: 'random'` for nondeterministic mode |
|
|
137
437
|
| `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
|
|
138
|
-
| `ops.validate(expr)` | Add `_valid` boolean column, keep all rows |
|
|
438
|
+
| `ops.validate(expr, { rules, rulesFile, audit, auditLimit, maxFailures, warnFailureRate, maxFailureRate, name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Add `_valid` boolean column, keep all rows; supports inline rules or local JSON rule-suite files; bounded failure audit records support count/rate thresholds plus privacy controls |
|
|
439
|
+
| `ops.assert(expr, { action, name, message, result, aggregate, op, value, column, tolerance, rel, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Row-local data-quality rule or finish-time O(1) aggregate assertion; row-local failure side-channel records support privacy controls |
|
|
440
|
+
| `ops.quarantine(expr, { name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Route rows matching expression to `errors` and drop them from main output; row payloads support privacy controls |
|
|
441
|
+
| `ops.schema({ columns, required, nonNull, nullable, values, min, max, regex, mode='fail', result='_schema', maxRegexPatternBytes, maxRegexCellBytes, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Row-local table schema contract; fail, warn, filter, quarantine, or annotate; regex budgets default to 4096-byte patterns and 65536-byte cells; schema audit/error records support row omission, column allowlists, redaction, stable non-cryptographic hashes, and row/cell payload caps |
|
|
442
|
+
| `ops.schemaInfer({ rows })` | Bounded decoded-type/nullability schema report; defaults to 10000 sampled rows |
|
|
443
|
+
| `ops.tee({ expr, channel, columns, limit, every, name, includeRow, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Preserve main rows and write bounded JSONL row snapshots to a side channel; row payloads support privacy controls |
|
|
139
444
|
|
|
140
445
|
### Column operations
|
|
141
446
|
|
|
142
447
|
| Method | Description |
|
|
143
448
|
|--------|-------------|
|
|
144
449
|
| `ops.select(columns)` | Keep and reorder columns |
|
|
450
|
+
| `ops.relocate(columns, { before, after })` | Move columns while preserving all columns |
|
|
145
451
|
| `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
|
|
146
452
|
| `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
|
|
147
|
-
| `ops.
|
|
453
|
+
| `ops.sourceName({ result, defaultValue })` | Append the current host source path/name as a row-local string column |
|
|
454
|
+
| `ops.across(columns, { fn, functions, names, replace })` | Apply row-local functions over selected columns |
|
|
455
|
+
| `ops.cast(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Type conversion; optional bounded value/coercion audit records support privacy controls |
|
|
148
456
|
| `ops.trim(columns?)` | Strip whitespace |
|
|
149
|
-
| `ops.fillNull(mapping)` | Replace nulls
|
|
457
|
+
| `ops.fillNull(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Replace nulls; optional bounded audit records support privacy controls |
|
|
150
458
|
| `ops.fillDown(columns?)` | Forward-fill nulls |
|
|
151
459
|
| `ops.clip(column, { min, max })` | Clamp numeric values |
|
|
152
|
-
| `ops.replace(column, pattern, replacement, { regex })` | String find/replace |
|
|
460
|
+
| `ops.replace(column, pattern, replacement, { regex, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | String find/replace; optional bounded audit records support privacy controls |
|
|
153
461
|
| `ops.hash(columns?)` | Add `_hash` column (DJB2) |
|
|
154
|
-
| `ops.bin(column, boundaries)` | Discretize into bins |
|
|
462
|
+
| `ops.bin(column, boundaries, { missing, onTypeError })` | Discretize into bins with strict numeric-source defaults |
|
|
463
|
+
| `ops.ewma(column, alpha, { result, missing, onTypeError })` | Exponentially weighted moving average |
|
|
464
|
+
| `ops.anomaly(column, { threshold, result, missing, onTypeError })` | Streaming z-score anomaly flag |
|
|
465
|
+
| `ops.normalize(columns, { method, audit, auditLimit, missing, onTypeError, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Blocking minmax/zscore normalization; optional audit records support privacy controls |
|
|
466
|
+
| `ops.acf(column, { lags, missing, onTypeError })` | Blocking autocorrelation table |
|
|
467
|
+
|
|
468
|
+
`columns` may contain exact names or selector strings: `id:score`, `starts_with(score_)`, `ends_with(_id)`, `contains(temp)`, `matches(^score_)`, `where(numeric)`, strict `all_of(score,name)`, lenient `any_of(optional,score)`, exclusions with `!name` or `-name`, and boolean selector algebra such as `starts_with(score_)&where(numeric)`, `starts_with(score_)&!ends_with(raw)`, or `!(id:score)`. Ranges use input schema order and can be reversed. Helper matching is case-insensitive; exact names are case-sensitive. These selectors are resolved by the native schema-aware `select`, `relocate`, and `across` ops in `O(columns)` without retaining rows; SQL lowering rejects selector helpers without a known schema.
|
|
469
|
+
|
|
470
|
+
Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selected numeric columns per row; `ops.across(['name'], { functions: ['lower'], replace: false, names: '{col}_{fn}' })` appends templated columns. This is native/WASM row-local execution, not arbitrary lambdas or grouped dplyr evaluation.
|
|
155
471
|
|
|
156
472
|
### Sorting and deduplication
|
|
157
473
|
|
|
@@ -165,14 +481,16 @@ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
|
|
|
165
481
|
| Method | Description |
|
|
166
482
|
|--------|-------------|
|
|
167
483
|
| `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
|
|
168
|
-
| `ops.frequency(columns?)` | Value counts
|
|
169
|
-
| `ops.groupAgg(groupBy, aggs)` | Group by + aggregate |
|
|
484
|
+
| `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
|
|
485
|
+
| `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
|
|
170
486
|
|
|
487
|
+
<!-- readme-test: js-api -->
|
|
171
488
|
```js
|
|
172
489
|
// Group aggregation
|
|
173
490
|
ops.groupAgg(['city'], [
|
|
174
|
-
{ column: 'price', func: 'sum',
|
|
175
|
-
{ column: 'price', func: 'avg',
|
|
491
|
+
{ column: 'price', func: 'sum', name: 'total' },
|
|
492
|
+
{ column: 'price', func: 'avg', name: 'avg_price' },
|
|
493
|
+
{ column: '*', func: 'count', name: 'rows' },
|
|
176
494
|
])
|
|
177
495
|
```
|
|
178
496
|
|
|
@@ -181,37 +499,51 @@ ops.groupAgg(['city'], [
|
|
|
181
499
|
| Method | Description |
|
|
182
500
|
|--------|-------------|
|
|
183
501
|
| `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
|
|
184
|
-
| `ops.window(column, size, func,
|
|
502
|
+
| `ops.window(column, size, func, resultOrOptions?)` | Sliding window: `avg`, `sum`, `min`, `max`; options include `result`, `missing`, `onTypeError` |
|
|
503
|
+
| `ops.rollingSum/rollingMean/rollingMin/rollingMax(column, size, { result, missing, onTypeError })` | Named trailing fixed-row numeric windows; default missing/non-numeric source fails |
|
|
504
|
+
| `ops.rollingAny/rollingAll(column, size, { result, nulls })` | Boolean trailing windows; `nulls`: `ignore`, `false`, `true`, `propagate` |
|
|
185
505
|
| `ops.lead(column, { offset, result })` | Lookahead N rows |
|
|
506
|
+
| `ops.lag(column, { offset, result })` | Previous-row shift |
|
|
507
|
+
| `ops.shift(column, { offset, result, type })` | Shift alias; `type='lag'` default or `type='lead'` |
|
|
508
|
+
| `ops.rowid(columns, { result, sorted, maxKeys })` | Global or per-key 1-based row ids |
|
|
509
|
+
| `ops.rleid(columns, { result })` | Consecutive run id by selected column(s) |
|
|
186
510
|
|
|
187
511
|
### Reshape
|
|
188
512
|
|
|
189
513
|
| Method | Description |
|
|
190
514
|
|--------|-------------|
|
|
191
|
-
| `ops.explode(column, delimiter
|
|
515
|
+
| `ops.explode(column, delimiter?, { maxTokensPerRow, maxOutputRowsPerInputRow, maxOutputRowsPerBatch, maxTokenBytes })` | Split delimited string into rows with optional expansion caps |
|
|
192
516
|
| `ops.split(column, names, delimiter?)` | Split column into multiple columns |
|
|
193
|
-
| `ops.unpivot(columns)` | Wide to long (melt) |
|
|
517
|
+
| `ops.unpivot(columns, { maxOutputRowsPerInputRow, maxOutputRowsPerBatch })` | Wide to long (melt) with optional expansion caps |
|
|
194
518
|
| `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
|
|
195
519
|
|
|
520
|
+
`explode` caps fail fast with `maxTokensPerRow`, `maxOutputRowsPerInputRow`, `maxOutputRowsPerBatch`, or `maxTokenBytes` when a single row or batch would expand beyond the configured limit. `unpivot` supports `maxOutputRowsPerInputRow` and `maxOutputRowsPerBatch`.
|
|
521
|
+
|
|
196
522
|
### Date/time
|
|
197
523
|
|
|
198
524
|
| Method | Description |
|
|
199
525
|
|--------|-------------|
|
|
200
|
-
| `ops.datetime(column,
|
|
201
|
-
| `ops.dateTrunc(column, trunc, { result })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
|
|
526
|
+
| `ops.datetime(column, extractOrOptions?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday`; accepts `{ extract, missing, onTypeError }` |
|
|
527
|
+
| `ops.dateTrunc(column, trunc, { result, missing, onTypeError })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
|
|
202
528
|
|
|
203
529
|
### Other
|
|
204
530
|
|
|
205
531
|
| Method | Description |
|
|
206
532
|
|--------|-------------|
|
|
207
533
|
| `ops.flatten()` | Flatten nested columns |
|
|
534
|
+
| `ops.interpolate(column, { method, missing, onTypeError })` | Fill nulls in a numeric column; default missing/non-numeric source fails |
|
|
535
|
+
| `ops.jsonExtract(path, result, { column, type })` | Extract JSON Pointer/simple JSONPath value into a new column |
|
|
536
|
+
| `ops.jsonFilter(path, { op, value, column, type })` | Filter rows by JSON Pointer/simple JSONPath predicate |
|
|
537
|
+
| `ops.jsonSchema(schema, { column, mode, result, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Validate JSON text with a supported JSON Schema subset; filter-mode audit records support privacy controls |
|
|
538
|
+
| `ops.jsonFlatten(fields, { column })` | Append declared JSON Pointer/simple JSONPath fields as bounded output columns |
|
|
208
539
|
| `ops.reorder(columns)` | Alias for `select` |
|
|
209
540
|
| `ops.dedup(columns?)` | Alias for `unique` |
|
|
210
541
|
|
|
211
542
|
## Expressions
|
|
212
543
|
|
|
213
|
-
Used in `filter`, `derive`, and `
|
|
544
|
+
Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
|
|
214
545
|
|
|
546
|
+
<!-- readme-test: js-api -->
|
|
215
547
|
```js
|
|
216
548
|
ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
|
|
217
549
|
ops.derive({
|
|
@@ -228,19 +560,29 @@ ops.derive({
|
|
|
228
560
|
| Comparison | `>` `>=` `<` `<=` `==` `!=` |
|
|
229
561
|
| Logic | `and` `or` `not` |
|
|
230
562
|
| String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
|
|
231
|
-
| Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` |
|
|
232
|
-
|
|
|
563
|
+
| Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` `between(x,left,right)` `inrange(x,left,right)` |
|
|
564
|
+
| Date/time | `year(x)` `month(x)` `day(x)` `hour(x)` `minute(x)` `second(x)` `weekday(x)` `epoch(x)` `date_trunc(x,unit)` |
|
|
565
|
+
| Conditional | `if(cond,then,else)` `case_when(cond,value,...,default)` `case_match(value,key,result,...,default)` `if_any(pred,...)` `if_all(pred,...)` `coalesce(a,b,...)` `nullif(a,b)` |
|
|
233
566
|
| Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
|
|
234
567
|
|
|
235
|
-
Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`.
|
|
568
|
+
Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
|
|
569
|
+
|
|
570
|
+
<!-- readme-test: js-api -->
|
|
571
|
+
```js
|
|
572
|
+
ops.derive({
|
|
573
|
+
year: expr("year(col('date'))"),
|
|
574
|
+
monthStart: expr("date_trunc(col('date'), 'month')"),
|
|
575
|
+
})
|
|
576
|
+
```
|
|
236
577
|
|
|
237
578
|
## Recipes
|
|
238
579
|
|
|
239
580
|
Built-in named pipelines for common tasks. Use by name:
|
|
240
581
|
|
|
582
|
+
<!-- readme-test: js-api -->
|
|
241
583
|
```js
|
|
242
584
|
const result = await pipeline('preview').run({ inputFile: 'data.csv' })
|
|
243
|
-
const
|
|
585
|
+
const frequencies = await pipeline('freq').run({ inputFile: 'data.csv' })
|
|
244
586
|
```
|
|
245
587
|
|
|
246
588
|
| Recipe | Pipeline | Description |
|
|
@@ -248,6 +590,7 @@ const result = await pipeline('freq').run({ inputFile: 'data.csv' })
|
|
|
248
590
|
| `profile` | `csv \| stats \| csv` | Full data profiling |
|
|
249
591
|
| `preview` | `csv \| head 10 \| csv` | First 10 rows |
|
|
250
592
|
| `schema` | `csv \| head 0 \| csv` | Column names only |
|
|
593
|
+
| `sniff` | `csv \| schema infer rows=1000 \| csv` | Bounded-memory schema/type/nullability sniff |
|
|
251
594
|
| `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
|
|
252
595
|
| `count` | `csv \| stats count \| csv` | Row count |
|
|
253
596
|
| `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
|
|
@@ -277,6 +620,27 @@ for (const r of await recipes()) {
|
|
|
277
620
|
}
|
|
278
621
|
```
|
|
279
622
|
|
|
623
|
+
|
|
624
|
+
## Memory policy
|
|
625
|
+
|
|
626
|
+
Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
|
|
627
|
+
|
|
628
|
+
<!-- readme-test: js-api -->
|
|
629
|
+
```js
|
|
630
|
+
await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
|
|
631
|
+
```
|
|
632
|
+
|
|
633
|
+
For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
|
|
634
|
+
|
|
635
|
+
<!-- readme-test: js-api -->
|
|
636
|
+
```js
|
|
637
|
+
await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
|
|
638
|
+
```
|
|
639
|
+
|
|
640
|
+
Native Node execution supports spill-backed `sort`, capped unsorted `pivot`, unsorted `unique`/`dedup`, unsorted `group-agg`, capped unsorted `join` inner/left, unsorted `semi-join`/`anti-join`, unsorted set operations, and duplicate-eliminating `union` when `spillDir` is provided. `run()`, `iterChunks()`, `toReadable()`, and `writeTo()` drain finish-time merge output through N-API `finishStep()` instead of waiting for one whole `finish()`. Standalone WASM supports `allowBlocking` and `memory`, but rejects `spillDir` because browser/WASM spill storage would occupy WASM memory. Use the Node native addon, CLI/direct C, or `{ engine: 'duckdb' }` for external spill; with DuckDB, `memory` maps to `memory_limit` and `spillDir` maps to `temp_directory`.
|
|
641
|
+
|
|
642
|
+
Plan-internal file reads are denied by default in Node and WASM. Pass `allowFs: true` for trusted local lookup files used by `join`, set operations, `union`, or `stack`; pass both `allowFs: true` and `allowRulesFile: true` for `validate rules_file=...`; pass `workspaceRoot` to pin resolved core plan paths inside a trusted directory. Supplying `spillDir` opts into local filesystem spill for Node native execution; standalone WASM still rejects spill. `inputFile` and `inputFiles` are host source adapters and are not controlled by `allowFs`.
|
|
643
|
+
|
|
280
644
|
## DuckDB engine
|
|
281
645
|
|
|
282
646
|
Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
|
|
@@ -285,6 +649,7 @@ Run pipelines on DuckDB instead of the native C streaming core. The DSL is trans
|
|
|
285
649
|
npm install duckdb
|
|
286
650
|
```
|
|
287
651
|
|
|
652
|
+
<!-- readme-test: js-data -->
|
|
288
653
|
```js
|
|
289
654
|
import { pipeline, compileToSql } from 'tranfi'
|
|
290
655
|
|
|
@@ -301,8 +666,12 @@ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
|
|
|
301
666
|
|
|
302
667
|
Generate SQL directly from DSL strings:
|
|
303
668
|
|
|
669
|
+
<!-- readme-test: js-sql -->
|
|
304
670
|
```js
|
|
305
|
-
const sql = await compileToSql(
|
|
671
|
+
const sql = await compileToSql(
|
|
672
|
+
'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
|
|
673
|
+
{ dialect: 'duckdb' }
|
|
674
|
+
)
|
|
306
675
|
console.log(sql)
|
|
307
676
|
// WITH
|
|
308
677
|
// step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
|
|
@@ -310,10 +679,15 @@ console.log(sql)
|
|
|
310
679
|
// SELECT * FROM step_2
|
|
311
680
|
```
|
|
312
681
|
|
|
682
|
+
DuckDB is the only implemented SQL dialect today. `dialect: 'sqlite'` and
|
|
683
|
+
`dialect: 'postgres'` are recognized but rejected until those dialects have
|
|
684
|
+
their own compatibility tests and SQL-generation rules.
|
|
685
|
+
|
|
313
686
|
### Browser (WASM + DuckDB-WASM)
|
|
314
687
|
|
|
315
688
|
In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
|
|
316
689
|
|
|
690
|
+
<!-- readme-test: browser -->
|
|
317
691
|
```js
|
|
318
692
|
import createTranfi from 'tranfi/wasm'
|
|
319
693
|
import * as duckdb from '@duckdb/duckdb-wasm'
|
|
@@ -321,7 +695,7 @@ import * as duckdb from '@duckdb/duckdb-wasm'
|
|
|
321
695
|
const tf = await createTranfi()
|
|
322
696
|
|
|
323
697
|
// SQL generation (synchronous, no DuckDB needed)
|
|
324
|
-
const sql = tf.compileToSql('csv | filter "age > 25" | csv')
|
|
698
|
+
const sql = tf.compileToSql('csv | filter "age > 25" | csv', { dialect: 'duckdb' })
|
|
325
699
|
|
|
326
700
|
// Full execution with DuckDB-WASM
|
|
327
701
|
const db = new duckdb.AsyncDuckDB(...)
|
|
@@ -339,7 +713,7 @@ console.log(result.rows) // Array of row objects
|
|
|
339
713
|
### DSL compilation
|
|
340
714
|
|
|
341
715
|
```js
|
|
342
|
-
import { compileDsl, saveRecipe, loadRecipe } from 'tranfi'
|
|
716
|
+
import { compileDsl, saveRecipe, loadRecipe, codec, ops } from 'tranfi'
|
|
343
717
|
|
|
344
718
|
// Compile DSL to JSON plan
|
|
345
719
|
const json = await compileDsl('csv | filter "col(age) > 25" | sort -age | csv')
|
|
@@ -356,17 +730,20 @@ Every pipeline produces four output channels:
|
|
|
356
730
|
|
|
357
731
|
- **output** -- main pipeline result
|
|
358
732
|
- **errors** -- rows that failed processing
|
|
359
|
-
- **stats** --
|
|
733
|
+
- **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
|
|
360
734
|
- **samples** -- reserved for sampling operators
|
|
361
735
|
|
|
736
|
+
<!-- readme-test: js-pipeline -->
|
|
362
737
|
```js
|
|
363
738
|
const result = await p.run({ inputFile: 'data.csv' })
|
|
364
|
-
console.log(result.statsText) //
|
|
739
|
+
console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
|
|
365
740
|
```
|
|
366
741
|
|
|
367
742
|
### Pipeline from JSON
|
|
368
743
|
|
|
369
744
|
```js
|
|
745
|
+
import { loadRecipe } from 'tranfi'
|
|
746
|
+
|
|
370
747
|
const p = await loadRecipe({
|
|
371
748
|
steps: [
|
|
372
749
|
{ op: 'codec.csv.decode', args: {} },
|
|
@@ -381,7 +758,7 @@ const p = await loadRecipe({
|
|
|
381
758
|
The package automatically selects the best backend:
|
|
382
759
|
|
|
383
760
|
1. **N-API** (Node.js) -- native C addon, fastest, used when available
|
|
384
|
-
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly,
|
|
761
|
+
2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, single-file module
|
|
385
762
|
3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
|
|
386
763
|
|
|
387
764
|
```js
|
|
@@ -392,4 +769,19 @@ const tf = await createTranfi()
|
|
|
392
769
|
|
|
393
770
|
## Architecture
|
|
394
771
|
|
|
395
|
-
|
|
772
|
+
Node, Python, CLI and WASM use the same C11 engine. It processes columnar batches
|
|
773
|
+
with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`)
|
|
774
|
+
and per-cell null bitmaps.
|
|
775
|
+
|
|
776
|
+
| Need | Node API behavior |
|
|
777
|
+
|------|-------------------|
|
|
778
|
+
| Stream input | `run({ inputFile })` uses `createReadStream()` and drains output after each push and incremental finish boundary |
|
|
779
|
+
| Avoid collecting all output | Use `inputStream`, `toReadable()`, `writeTo()`, `iterChunks()`, or `onOutput` with `collectOutput: false` |
|
|
780
|
+
| Read gzip | `.gz` files use `zlib.createGunzip()`; override with `compression: "none"` or `"gzip"` for `inputFile`/`inputStream` |
|
|
781
|
+
| Permit a blocking operation | Set `allowBlocking: true` or use a supported `spillDir` path |
|
|
782
|
+
| Cap retained key state | Supply caps and check the plan with `memory: "64MB"` |
|
|
783
|
+
| Allow plan file reads | Supply explicit host-policy options |
|
|
784
|
+
|
|
785
|
+
Input streaming still collects the final output by default. Choose a streaming
|
|
786
|
+
sink for large outputs; row-local and bounded-state operators stream under the
|
|
787
|
+
default strict native memory policy.
|