tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/README.md CHANGED
@@ -1,14 +1,23 @@
1
1
  # tranfi (Node.js / WASM)
2
2
 
3
- Streaming ETL in JavaScript, powered by a native C11 core via N-API (Node.js) or WASM (browsers). Process CSV, JSONL, and text data with composable pipelines that run in constant memory, no matter how large the input.
3
+ Tranfi transforms CSV, JSON Lines and plain text as streams. In Node.js it runs
4
+ on a native C core; in the browser it runs on WebAssembly. Data flows through in
5
+ chunks, so most pipelines use a small, fixed amount of memory however large the
6
+ input is. Operations that need the whole input, such as `sort`, must be allowed
7
+ explicitly. Tranfi works on byte streams; it is not an in-memory DataFrame
8
+ library.
9
+
10
+ Save this example as `quickstart.mjs`:
4
11
 
5
12
  ```js
6
- import { pipeline, codec, ops, expr } from 'tranfi'
13
+ import tranfi from 'tranfi'
14
+
15
+ const { pipeline, codec, ops, expr } = tranfi
7
16
 
8
17
  const result = await pipeline([
9
18
  codec.csv(),
10
19
  ops.filter(expr("col('age') > 25")),
11
- ops.sort(['-age']),
20
+ ops.top(100, 'age'),
12
21
  ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
13
22
  ops.select(['name', 'age', 'label']),
14
23
  codec.csvEncode(),
@@ -23,8 +32,9 @@ console.log(result.outputText)
23
32
 
24
33
  Or use the pipe DSL for one-liners:
25
34
 
35
+ <!-- readme-test: js-api -->
26
36
  ```js
27
- const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
37
+ const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
28
38
  .run({ inputFile: 'data.csv' })
29
39
  ```
30
40
 
@@ -34,7 +44,25 @@ const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
34
44
  npm install tranfi
35
45
  ```
36
46
 
37
- Uses N-API natively in Node.js, falls back to WASM in browsers automatically.
47
+ Installation compiles Tranfi's C core as a native Node.js addon, so you need a C
48
+ compiler. If that compile fails, the install fails.
49
+
50
+ In the browser, import `tranfi/wasm`: it provides pipelines and prepared
51
+ transforms without the native addon. To install only the WebAssembly build, run
52
+ `TRANFI_SKIP_NATIVE_BUILD=1 npm install tranfi`. The root `tranfi` entry then
53
+ runs pipelines through WebAssembly, but its prepared-transform classes need the
54
+ native addon, so use `tranfi/wasm` for those.
55
+
56
+ | | Linux | Windows | macOS |
57
+ |---|---|---|---|
58
+ | Native addon | Tested | Not tested | Not tested |
59
+ | WebAssembly (`tranfi/wasm`) | Tested in Node.js and Chromium | Tested in Node.js | Not tested |
60
+
61
+ Use `pipeline(...)` for byte-stream ETL. The separate
62
+ `TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
63
+ lifecycle is for typed batches whose learned state must be frozen and reused.
64
+ Calling `.run()` collects output for convenience; use `writeTo()` or
65
+ `toReadable()` for large outputs.
38
66
 
39
67
  ## CLI
40
68
 
@@ -42,30 +70,34 @@ Installing the package also provides the `tranfi` command:
42
70
 
43
71
  ```bash
44
72
  # Via npx (no install)
45
- echo 'name,age\nAlice,30\nBob,25' | npx tranfi 'csv | filter "age > 25" | csv'
73
+ printf 'name,age\nAlice,30\nBob,25\n' | npx tranfi -q 'csv | filter "age > 25" | csv'
46
74
 
47
75
  # Or install globally
48
76
  npm i -g tranfi
49
- tranfi 'csv | filter "age > 25" | sort -age | csv' < data.csv
77
+ tranfi -q 'csv | filter "age > 25" | top-k 100 age | csv' < data.csv
50
78
  tranfi profile < data.csv
51
79
  tranfi -R # list recipes
52
80
  ```
53
81
 
54
- Run `tranfi -h` for all options.
82
+ Run `tranfi -h` for the npm CLI options. Use `--allow-blocking` for known-small
83
+ full-input operations, `--memory max:64MB` for a memory policy, and
84
+ `--spill-dir DIR` for an existing private spill directory. `--stats-json FILE`
85
+ writes the stats channel; `--target json|sql` compiles without executing.
86
+ The detailed `--explain` report requires the standalone C CLI built from source.
55
87
 
56
88
  ## Quick start
57
89
 
58
90
  ### Two APIs
59
91
 
60
- **Builder API** -- composable, type-safe, IDE-friendly:
92
+ **Builder API** -- composable, structured, and IDE-friendly:
61
93
 
94
+ <!-- readme-test: js-api -->
62
95
  ```js
63
96
  const p = pipeline([
64
97
  codec.csv(),
65
98
  ops.filter(expr("col('score') >= 80")),
66
99
  ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
67
- ops.sort(['-score']),
68
- ops.head(10),
100
+ ops.top(10, 'score'),
69
101
  codec.csvEncode(),
70
102
  ])
71
103
  const result = await p.run({ inputFile: 'students.csv' })
@@ -73,21 +105,25 @@ const result = await p.run({ inputFile: 'students.csv' })
73
105
 
74
106
  **DSL strings** -- compact, suitable for CLI-like use:
75
107
 
108
+ <!-- readme-test: js-api -->
76
109
  ```js
77
- const p = pipeline('csv | filter "col(score) >= 80" | sort -score | head 10 | csv')
110
+ const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
78
111
  const result = await p.run({ inputFile: 'students.csv' })
79
112
  ```
80
113
 
81
114
  Both produce identical pipelines under the hood.
82
115
 
116
+ Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate` -> `derive`, `summarise`/`summarize` -> `group-agg`, `distinct` -> `unique`, and `arrange` -> `sort`.
117
+
83
118
  ### Running pipelines
84
119
 
120
+ <!-- readme-test: js-pipeline -->
85
121
  ```js
86
122
  // From string or Buffer
87
123
  const result = await p.run({ input: 'name,age\nAlice,30\n' })
88
124
 
89
125
  // From file (streamed in 64 KB chunks)
90
- const result = await p.run({ inputFile: 'data.csv' })
126
+ const fileResult = await p.run({ inputFile: 'data.csv' })
91
127
 
92
128
  // Access results
93
129
  result.output // Buffer
@@ -98,22 +134,235 @@ result.statsText // string
98
134
  result.samples // Buffer (sample channel)
99
135
  ```
100
136
 
137
+ For large outputs, drain chunks instead of collecting `result.output`:
138
+
139
+ <!-- readme-test: js-pipeline -->
140
+ ```js
141
+ import { createReadStream, createWriteStream } from 'node:fs'
142
+
143
+ // Write to a Node Writable and wait for `drain` when the sink applies backpressure.
144
+ const out = createWriteStream("out.csv")
145
+ const result = await p.writeTo(out, { inputFile: "data.csv" })
146
+
147
+ // Or consume Tranfi output as a Node Readable.
148
+ for await (const chunk of p.toReadable({ inputFile: "data.csv" })) {
149
+ // process each output chunk
150
+ }
151
+
152
+ // Existing input streams can feed the native pipeline without readFile() materialization.
153
+ const result2 = await p.run({ inputStream: createReadStream("data.csv") })
154
+
155
+ // .gz input files are decompressed as streaming sources by default.
156
+ const gz = await p.run({ inputFile: "events.jsonl.gz" })
157
+ const raw = await p.run({ inputFile: "events.jsonl.gz", compression: "none" })
158
+ const streamGz = await p.run({ inputStream: createReadStream("events.jsonl.gz"), compression: "gzip" })
159
+
160
+ // Multiple files stream sequentially through one pipeline.
161
+ const inputFiles = ["part-a.csv", "part-b.csv"]
162
+ const combined = await p.run({ inputFiles, sourceColumn: "src" })
163
+
164
+ // Low-level async iteration is still available.
165
+ for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
166
+ // process each output chunk
167
+ }
168
+ ```
169
+
170
+ `sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. Repeated headers are kept by default. Use `codec.csv({ skipRepeatedHeader: true })` to skip a later file's first non-comment record when it exactly matches the original header; matching rows inside a file remain data.
171
+
172
+ Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
173
+
174
+ For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
175
+
176
+ <!-- readme-test: browser -->
177
+ ```js
178
+ // main thread
179
+ import { createWorkerClient } from 'tranfi/wasm/worker'
180
+
181
+ const client = createWorkerClient(new Worker(new URL('./tranfi-worker.js', import.meta.url), { type: 'module' }))
182
+ const result = await client.runFile('csv | filter "col(age) >= 18" | csv', file, {
183
+ chunkSize: 64 * 1024,
184
+ collectOutput: false,
185
+ onOutput: chunk => downloadSink.write(chunk),
186
+ onProgress: p => updateProgress(p),
187
+ signal: abortController.signal
188
+ })
189
+
190
+ // tranfi-worker.js, bundled by the app
191
+ import { runWorkerServer } from 'tranfi/wasm/worker'
192
+ runWorkerServer()
193
+ ```
194
+
195
+ The Worker protocol streams chunks with transferable buffers and sends progress, stats, errors, and cancellation messages around the same WASM `create/push/pull/finish/free` API; it is not a separate IR target.
196
+
197
+ The bundled Tranfi app runner is also preview-bounded by default: file chunks are streamed into WASM, main output is drained incrementally into a table preview capped by hidden `preview_rows` (default 200), and full output text is only materialized when `collect_output` is explicitly true in the schema.
198
+
199
+ ### Prepared reusable transforms
200
+
201
+ Learn imputation, scaling or category mappings from reference data, then apply
202
+ the same plan to new batches. Analyze the reference batches, finalize the plan,
203
+ and apply it. Replaying the reference data through that plan gives a second-pass
204
+ `fit_transform` workflow.
205
+
206
+ Inputs are typed `float32`/`float64` columns. This example learns mean imputation
207
+ and standard scaling, applies them to the reference batch, and exports the plan:
208
+
209
+ ```js
210
+ import tf from 'tranfi'
211
+
212
+ const recipeSpec = {
213
+ format: 'tranfi.transform-recipe',
214
+ version: 1,
215
+ policyVersion: 1,
216
+ outputDtype: 'float64',
217
+ semanticLimits: {
218
+ maxOutputColumns: 65536,
219
+ maxOutputElementsPerApply: 134217728
220
+ },
221
+ columns: [{
222
+ sourceId: 'x0',
223
+ kind: { op: 'declared', value: 'numeric', rule: null, maxCategories: null },
224
+ numeric: {
225
+ impute: { op: 'mean', constant: null, allMissing: 'zero' },
226
+ normalize: { op: 'standard', ddof: 0 }
227
+ },
228
+ categorical: null
229
+ }]
230
+ }
231
+ const schema = [{ id: 'x0', dtype: 'float64' }]
232
+
233
+ const recipe = tf.TransformRecipe.fromJSON(recipeSpec)
234
+ const analyzer = recipe.analyzer(schema)
235
+ analyzer.push({ rows: 3, columns: [new Float64Array([1, NaN, 3])] })
236
+ const plan = analyzer.finalize()
237
+
238
+ const apply = plan.apply(schema)
239
+ const fittedReference = apply.run({
240
+ rows: 3,
241
+ columns: [new Float64Array([1, NaN, 3])]
242
+ })
243
+ const planBytes = plan.toBytes() // canonical TFTR artifact
244
+ const recipeSha256 = plan.recipeSha256() // canonical recipe + input-schema identity
245
+
246
+ apply.close()
247
+ plan.close()
248
+ analyzer.close()
249
+ recipe.close()
250
+ ```
251
+
252
+ <details markdown="1">
253
+ <summary>Available methods, missing values and kind inference</summary>
254
+
255
+ Recipe fields use the same names in all bindings.
256
+
257
+ | Recipe field | Choices / behavior |
258
+ |--------------|--------------------|
259
+ | Numeric `impute.op` | `none`, `zero`, `constant`, `mean`, exact `median` |
260
+ | Numeric `normalize.op` | `none`, `standard`, `minmax` |
261
+ | Categorical `impute.op` | `none` or mode; mode supports `allMissing` of `error` or `zero` |
262
+ | Categorical `encode.op` | `none`, `label`, `onehot` |
263
+ | Label unknown policy | `error`, `sentinel`, `other` |
264
+ | One-hot unknown policy | `error`, `all_zero`, `other` |
265
+
266
+ **No imputation or encoding:** finite values pass through; missing input becomes
267
+ canonical qNaN. This combination needs no learned category dictionary.
268
+ Mode imputation with no encoding retains a dictionary and rejects unseen finite
269
+ values with error `108`.
270
+
271
+ **Encoding:** label ordinals are zero-based and sorted; one-hot blocks follow
272
+ source order. A label sentinel must be a safe integer outside the known ordinal
273
+ range. `other` appends an ordinal or field after known categories. Without
274
+ imputation, missing input follows the encoder's unknown policy.
275
+
276
+ Generated IDs, names and category metadata are deterministic. Collisions fail
277
+ with error `102`.
278
+
279
+ **Kind inference:** configure both numeric and categorical branches, set
280
+ `kind.op` to `infer`, `kind.rule` to `finite-integer-cardinality-v1`, and
281
+ `kind.maxCategories` to at least `2`. Missing/NaN values are ignored.
282
+
283
+ | Observed reference values | Selected branch |
284
+ |---------------------------|-----------------|
285
+ | `2..maxCategories` distinct finite integers | Categorical |
286
+ | Zero or one distinct value, any noninteger, or more than `maxCategories` | Numeric |
287
+
288
+ </details>
289
+
290
+ <details markdown="1">
291
+ <summary>Fixed category dictionaries</summary>
292
+
293
+ Set `encode.categories` to a nonempty, sorted, unique array of finite typed tags.
294
+ For example, `{ "t": "f64", "v": "4000000000000000" }` represents `2`.
295
+ Tags must match the input dtype; represent negative zero as positive zero.
296
+
297
+ - Unobserved categories stay in the dictionary.
298
+ - Unknown training values follow the encoder policy and never vote for mode.
299
+ - An all-missing zero fallback must exist in the dictionary.
300
+ - Fixed encoding without imputation can finalize without training rows.
301
+
302
+ String categories, categorical constant imputation and fixed dictionaries
303
+ combined with kind inference are not supported.
304
+
305
+ </details>
306
+
307
+ <details markdown="1">
308
+ <summary>Limits, errors and cancellation</summary>
309
+
310
+ Exact median obeys allocation and resident-state limits. Inference, category
311
+ discovery and output expansion also enforce category/output-width limits.
312
+ Resource-limit failures use error `104`.
313
+
314
+ Host-policy and spill settings are reserved. A non-null host policy or nonempty
315
+ spill path fails with unsupported-runtime error `113`.
316
+
317
+ Runtime options reject unknown fields:
318
+
319
+ | Runtime | Options |
320
+ |---------|---------|
321
+ | Native Node | `limits`, `cancelFlag`, `hostPolicy`, `spillDir` |
322
+ | Standalone WASM | `limits`, `cancelToken`, `hostPolicy`, `spillDir` |
323
+
324
+ Cancellation spellings are not interchangeable. WASM tokens created with
325
+ `createTransformCancelToken()` expose a read-only `requested` boolean for host
326
+ conversion loops; reading a closed token fails.
327
+
328
+ `tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
329
+
330
+ <!-- readme-test: browser -->
331
+ ```js
332
+ const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
333
+ const makeWorker = () => new Worker(workerUrl, { type: 'module' })
334
+ const client = createWorkerClient(makeWorker(), {
335
+ workerFactory: makeWorker,
336
+ terminateOnDispose: true
337
+ })
338
+ ```
339
+
340
+ All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
341
+
342
+ </details>
343
+
101
344
  ## Codecs
102
345
 
103
- Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
346
+ Start with a decoder and finish with an encoder. Input and output formats can differ.
104
347
 
105
- | Method | Description |
106
- |--------|-------------|
107
- | `codec.csv({ delimiter, header, batchSize, repair })` | CSV decoder. `repair: true` pads short / truncates long rows |
108
- | `codec.csvEncode({ delimiter })` | CSV encoder |
109
- | `codec.jsonl({ batchSize })` | JSON Lines decoder |
110
- | `codec.jsonlEncode()` | JSON Lines encoder |
111
- | `codec.text({ batchSize })` | Line-oriented text decoder (single `_line` column) |
112
- | `codec.textEncode()` | Text encoder |
113
- | `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
348
+ | Format | Decoder | Encoder |
349
+ |--------|---------|---------|
350
+ | CSV | `codec.csv(options)` | `codec.csvEncode({ delimiter })` |
351
+ | JSON Lines | `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | `codec.jsonlEncode()` |
352
+ | Text (one `_line` column) | `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | `codec.textEncode()` |
353
+ | Markdown table | — | `codec.tableEncode({ maxWidth, maxRows })` |
354
+
355
+ **CSV defaults to permissive field-count handling.** Choose `mode: 'strict'`
356
+ (or `strict: true`) to reject width mismatches. Choose `repair: true`
357
+ (or `mode: 'repair'`) to pad/truncate rows and collect diagnostics in `result.errors`.
358
+
359
+ **Malformed JSONL is skipped by default.** Set `onError` to `fail` to stop,
360
+ `warn` to keep valid rows with diagnostics, or `quarantine` to route bad records
361
+ to error diagnostics. Read those diagnostics from `result.errors`.
114
362
 
115
363
  Cross-codec pipelines work naturally:
116
364
 
365
+ <!-- readme-test: js-api -->
117
366
  ```js
118
367
  // CSV in, JSONL out
119
368
  pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
@@ -122,36 +371,103 @@ pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
122
371
  pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
123
372
  ```
124
373
 
374
+ <details markdown="1">
375
+ <summary>CSV options for nulls, row limits and whitespace</summary>
376
+
377
+ | Need | Option | Behavior |
378
+ |------|--------|----------|
379
+ | Field separator / header | `delimiter`, `header` | Configure CSV decoding; encoder accepts `delimiter` |
380
+ | Add null markers | `nulls: ['NA', 'NULL']` | Adds to default unquoted-empty-field null handling |
381
+ | Preserve quoted markers | `quotedNulls: false` | Keeps quoted `"NA"` and `""` as strings |
382
+ | Skip a preamble | `skip: 2` | Before schema/header discovery; comments are handled afterward |
383
+ | Limit rows | `nMax: 100` or `maxRows: 100` | One shared counter after skip/comment/header handling |
384
+ | Keep only the schema | `nMax: 0` | Preserves a header-only batch |
385
+ | Comments | `comment: '#'` | Removes text after unquoted markers and skips comment-only rows; quoted markers remain |
386
+ | Preserve spaces/tabs | `trimWs: false` | Disables default trimming of unquoted values |
387
+ | Drop blank rows | `skipEmptyRows: true` | Otherwise blank physical rows after the header become all-null rows |
388
+ | Read CSV shards | `skipRepeatedHeader: true` | Skips matching headers only at explicit file boundaries |
389
+
390
+ </details>
391
+
392
+ <details markdown="1">
393
+ <summary>Decoder limits and repair diagnostics</summary>
394
+
395
+ All decoder size options are checked integers:
396
+
397
+ | Option | Range | Behavior |
398
+ |--------|-------|----------|
399
+ | `batchSize` | `1..65536` | Rows per batch |
400
+ | `maxErrorBytes` | `0..67108864` | Bounds raw diagnostic previews |
401
+ | `maxRecordBytes` | `0..1073741824` | Default `67108864` bytes (64 MiB); `0` disables the record guard |
402
+ | `maxColumns` (CSV) | `1..65536` | Default `8192`; overflow fails with `csv_too_many_columns` |
403
+
404
+ Record limits apply before a newline is seen, including CSV, JSONL and text.
405
+ CSV record overflow fails with `csv_record_too_large`.
406
+
407
+ Repair audit/raw payloads accept `auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes`. The raw preview is exposed to
408
+ those controls as pseudo-column `raw`. CSV repair also accepts
409
+ `audit, auditLimit`.
410
+
411
+ </details>
412
+
413
+ <details markdown="1">
414
+ <summary>JSONL record and schema rules</summary>
415
+
416
+ Each record must be one complete UTF-8 JSON object. Invalid number syntax,
417
+ trailing content, embedded NULs and numbers outside finite float64 range follow
418
+ the configured malformed-record policy. The encoder rejects nonfinite values.
419
+
420
+ The first valid record determines schema order. Repeated keys use their first
421
+ value. `maxErrorBytes` bounds diagnostic previews, and `maxRecordBytes` caps the current line.
422
+
423
+ </details>
424
+
125
425
  ## Operators
126
426
 
127
427
  ### Row filtering
128
428
 
129
429
  | Method | Description |
130
430
  |--------|-------------|
131
- | `ops.filter(expr)` | Keep rows matching expression |
431
+ | `ops.filter(expr, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Keep rows matching expression; optional dropped-row audit records support row omission, column allowlists, redaction, hashes, and payload caps |
132
432
  | `ops.head(n)` | First N rows |
133
433
  | `ops.tail(n)` | Last N rows |
134
434
  | `ops.skip(n)` | Skip first N rows |
135
435
  | `ops.top(n, column, desc?)` | Top N by column value |
136
- | `ops.sample(n)` | Reservoir sampling (uniform random) |
436
+ | `ops.sample(n, { seed })` | Deterministic bounded reservoir sampling; use `seed: 'random'` for nondeterministic mode |
137
437
  | `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
138
- | `ops.validate(expr)` | Add `_valid` boolean column, keep all rows |
438
+ | `ops.validate(expr, { rules, rulesFile, audit, auditLimit, maxFailures, warnFailureRate, maxFailureRate, name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Add `_valid` boolean column, keep all rows; supports inline rules or local JSON rule-suite files; bounded failure audit records support count/rate thresholds plus privacy controls |
439
+ | `ops.assert(expr, { action, name, message, result, aggregate, op, value, column, tolerance, rel, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Row-local data-quality rule or finish-time O(1) aggregate assertion; row-local failure side-channel records support privacy controls |
440
+ | `ops.quarantine(expr, { name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Route rows matching expression to `errors` and drop them from main output; row payloads support privacy controls |
441
+ | `ops.schema({ columns, required, nonNull, nullable, values, min, max, regex, mode='fail', result='_schema', maxRegexPatternBytes, maxRegexCellBytes, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Row-local table schema contract; fail, warn, filter, quarantine, or annotate; regex budgets default to 4096-byte patterns and 65536-byte cells; schema audit/error records support row omission, column allowlists, redaction, stable non-cryptographic hashes, and row/cell payload caps |
442
+ | `ops.schemaInfer({ rows })` | Bounded decoded-type/nullability schema report; defaults to 10000 sampled rows |
443
+ | `ops.tee({ expr, channel, columns, limit, every, name, includeRow, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Preserve main rows and write bounded JSONL row snapshots to a side channel; row payloads support privacy controls |
139
444
 
140
445
  ### Column operations
141
446
 
142
447
  | Method | Description |
143
448
  |--------|-------------|
144
449
  | `ops.select(columns)` | Keep and reorder columns |
450
+ | `ops.relocate(columns, { before, after })` | Move columns while preserving all columns |
145
451
  | `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
146
452
  | `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
147
- | `ops.cast(mapping)` | Type conversion: `cast({ age: 'int', score: 'float' })` |
453
+ | `ops.sourceName({ result, defaultValue })` | Append the current host source path/name as a row-local string column |
454
+ | `ops.across(columns, { fn, functions, names, replace })` | Apply row-local functions over selected columns |
455
+ | `ops.cast(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Type conversion; optional bounded value/coercion audit records support privacy controls |
148
456
  | `ops.trim(columns?)` | Strip whitespace |
149
- | `ops.fillNull(mapping)` | Replace nulls: `fillNull({ age: '0' })` |
457
+ | `ops.fillNull(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Replace nulls; optional bounded audit records support privacy controls |
150
458
  | `ops.fillDown(columns?)` | Forward-fill nulls |
151
459
  | `ops.clip(column, { min, max })` | Clamp numeric values |
152
- | `ops.replace(column, pattern, replacement, { regex })` | String find/replace |
460
+ | `ops.replace(column, pattern, replacement, { regex, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | String find/replace; optional bounded audit records support privacy controls |
153
461
  | `ops.hash(columns?)` | Add `_hash` column (DJB2) |
154
- | `ops.bin(column, boundaries)` | Discretize into bins |
462
+ | `ops.bin(column, boundaries, { missing, onTypeError })` | Discretize into bins with strict numeric-source defaults |
463
+ | `ops.ewma(column, alpha, { result, missing, onTypeError })` | Exponentially weighted moving average |
464
+ | `ops.anomaly(column, { threshold, result, missing, onTypeError })` | Streaming z-score anomaly flag |
465
+ | `ops.normalize(columns, { method, audit, auditLimit, missing, onTypeError, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Blocking minmax/zscore normalization; optional audit records support privacy controls |
466
+ | `ops.acf(column, { lags, missing, onTypeError })` | Blocking autocorrelation table |
467
+
468
+ `columns` may contain exact names or selector strings: `id:score`, `starts_with(score_)`, `ends_with(_id)`, `contains(temp)`, `matches(^score_)`, `where(numeric)`, strict `all_of(score,name)`, lenient `any_of(optional,score)`, exclusions with `!name` or `-name`, and boolean selector algebra such as `starts_with(score_)&where(numeric)`, `starts_with(score_)&!ends_with(raw)`, or `!(id:score)`. Ranges use input schema order and can be reversed. Helper matching is case-insensitive; exact names are case-sensitive. These selectors are resolved by the native schema-aware `select`, `relocate`, and `across` ops in `O(columns)` without retaining rows; SQL lowering rejects selector helpers without a known schema.
469
+
470
+ Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selected numeric columns per row; `ops.across(['name'], { functions: ['lower'], replace: false, names: '{col}_{fn}' })` appends templated columns. This is native/WASM row-local execution, not arbitrary lambdas or grouped dplyr evaluation.
155
471
 
156
472
  ### Sorting and deduplication
157
473
 
@@ -165,14 +481,16 @@ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
165
481
  | Method | Description |
166
482
  |--------|-------------|
167
483
  | `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
168
- | `ops.frequency(columns?)` | Value counts (descending) |
169
- | `ops.groupAgg(groupBy, aggs)` | Group by + aggregate |
484
+ | `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
485
+ | `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
170
486
 
487
+ <!-- readme-test: js-api -->
171
488
  ```js
172
489
  // Group aggregation
173
490
  ops.groupAgg(['city'], [
174
- { column: 'price', func: 'sum', result: 'total' },
175
- { column: 'price', func: 'avg', result: 'avg_price' },
491
+ { column: 'price', func: 'sum', name: 'total' },
492
+ { column: 'price', func: 'avg', name: 'avg_price' },
493
+ { column: '*', func: 'count', name: 'rows' },
176
494
  ])
177
495
  ```
178
496
 
@@ -181,37 +499,51 @@ ops.groupAgg(['city'], [
181
499
  | Method | Description |
182
500
  |--------|-------------|
183
501
  | `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
184
- | `ops.window(column, size, func, result?)` | Sliding window: `avg`, `sum`, `min`, `max` |
502
+ | `ops.window(column, size, func, resultOrOptions?)` | Sliding window: `avg`, `sum`, `min`, `max`; options include `result`, `missing`, `onTypeError` |
503
+ | `ops.rollingSum/rollingMean/rollingMin/rollingMax(column, size, { result, missing, onTypeError })` | Named trailing fixed-row numeric windows; default missing/non-numeric source fails |
504
+ | `ops.rollingAny/rollingAll(column, size, { result, nulls })` | Boolean trailing windows; `nulls`: `ignore`, `false`, `true`, `propagate` |
185
505
  | `ops.lead(column, { offset, result })` | Lookahead N rows |
506
+ | `ops.lag(column, { offset, result })` | Previous-row shift |
507
+ | `ops.shift(column, { offset, result, type })` | Shift alias; `type='lag'` default or `type='lead'` |
508
+ | `ops.rowid(columns, { result, sorted, maxKeys })` | Global or per-key 1-based row ids |
509
+ | `ops.rleid(columns, { result })` | Consecutive run id by selected column(s) |
186
510
 
187
511
  ### Reshape
188
512
 
189
513
  | Method | Description |
190
514
  |--------|-------------|
191
- | `ops.explode(column, delimiter?)` | Split delimited string into rows |
515
+ | `ops.explode(column, delimiter?, { maxTokensPerRow, maxOutputRowsPerInputRow, maxOutputRowsPerBatch, maxTokenBytes })` | Split delimited string into rows with optional expansion caps |
192
516
  | `ops.split(column, names, delimiter?)` | Split column into multiple columns |
193
- | `ops.unpivot(columns)` | Wide to long (melt) |
517
+ | `ops.unpivot(columns, { maxOutputRowsPerInputRow, maxOutputRowsPerBatch })` | Wide to long (melt) with optional expansion caps |
194
518
  | `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
195
519
 
520
+ `explode` caps fail fast with `maxTokensPerRow`, `maxOutputRowsPerInputRow`, `maxOutputRowsPerBatch`, or `maxTokenBytes` when a single row or batch would expand beyond the configured limit. `unpivot` supports `maxOutputRowsPerInputRow` and `maxOutputRowsPerBatch`.
521
+
196
522
  ### Date/time
197
523
 
198
524
  | Method | Description |
199
525
  |--------|-------------|
200
- | `ops.datetime(column, extract?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday` |
201
- | `ops.dateTrunc(column, trunc, { result })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
526
+ | `ops.datetime(column, extractOrOptions?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday`; accepts `{ extract, missing, onTypeError }` |
527
+ | `ops.dateTrunc(column, trunc, { result, missing, onTypeError })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
202
528
 
203
529
  ### Other
204
530
 
205
531
  | Method | Description |
206
532
  |--------|-------------|
207
533
  | `ops.flatten()` | Flatten nested columns |
534
+ | `ops.interpolate(column, { method, missing, onTypeError })` | Fill nulls in a numeric column; default missing/non-numeric source fails |
535
+ | `ops.jsonExtract(path, result, { column, type })` | Extract JSON Pointer/simple JSONPath value into a new column |
536
+ | `ops.jsonFilter(path, { op, value, column, type })` | Filter rows by JSON Pointer/simple JSONPath predicate |
537
+ | `ops.jsonSchema(schema, { column, mode, result, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Validate JSON text with a supported JSON Schema subset; filter-mode audit records support privacy controls |
538
+ | `ops.jsonFlatten(fields, { column })` | Append declared JSON Pointer/simple JSONPath fields as bounded output columns |
208
539
  | `ops.reorder(columns)` | Alias for `select` |
209
540
  | `ops.dedup(columns?)` | Alias for `unique` |
210
541
 
211
542
  ## Expressions
212
543
 
213
- Used in `filter`, `derive`, and `validate`. Reference columns with `col('name')`.
544
+ Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
214
545
 
546
+ <!-- readme-test: js-api -->
215
547
  ```js
216
548
  ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
217
549
  ops.derive({
@@ -228,19 +560,29 @@ ops.derive({
228
560
  | Comparison | `>` `>=` `<` `<=` `==` `!=` |
229
561
  | Logic | `and` `or` `not` |
230
562
  | String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
231
- | Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` |
232
- | Conditional | `if(cond,then,else)` `coalesce(a,b,...)` `nullif(a,b)` |
563
+ | Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` `between(x,left,right)` `inrange(x,left,right)` |
564
+ | Date/time | `year(x)` `month(x)` `day(x)` `hour(x)` `minute(x)` `second(x)` `weekday(x)` `epoch(x)` `date_trunc(x,unit)` |
565
+ | Conditional | `if(cond,then,else)` `case_when(cond,value,...,default)` `case_match(value,key,result,...,default)` `if_any(pred,...)` `if_all(pred,...)` `coalesce(a,b,...)` `nullif(a,b)` |
233
566
  | Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
234
567
 
235
- Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`.
568
+ Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
569
+
570
+ <!-- readme-test: js-api -->
571
+ ```js
572
+ ops.derive({
573
+ year: expr("year(col('date'))"),
574
+ monthStart: expr("date_trunc(col('date'), 'month')"),
575
+ })
576
+ ```
236
577
 
237
578
  ## Recipes
238
579
 
239
580
  Built-in named pipelines for common tasks. Use by name:
240
581
 
582
+ <!-- readme-test: js-api -->
241
583
  ```js
242
584
  const result = await pipeline('preview').run({ inputFile: 'data.csv' })
243
- const result = await pipeline('freq').run({ inputFile: 'data.csv' })
585
+ const frequencies = await pipeline('freq').run({ inputFile: 'data.csv' })
244
586
  ```
245
587
 
246
588
  | Recipe | Pipeline | Description |
@@ -248,6 +590,7 @@ const result = await pipeline('freq').run({ inputFile: 'data.csv' })
248
590
  | `profile` | `csv \| stats \| csv` | Full data profiling |
249
591
  | `preview` | `csv \| head 10 \| csv` | First 10 rows |
250
592
  | `schema` | `csv \| head 0 \| csv` | Column names only |
593
+ | `sniff` | `csv \| schema infer rows=1000 \| csv` | Bounded-memory schema/type/nullability sniff |
251
594
  | `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
252
595
  | `count` | `csv \| stats count \| csv` | Row count |
253
596
  | `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
@@ -277,6 +620,27 @@ for (const r of await recipes()) {
277
620
  }
278
621
  ```
279
622
 
623
+
624
+ ## Memory policy
625
+
626
+ Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
627
+
628
+ <!-- readme-test: js-api -->
629
+ ```js
630
+ await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
631
+ ```
632
+
633
+ For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
634
+
635
+ <!-- readme-test: js-api -->
636
+ ```js
637
+ await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
638
+ ```
639
+
640
+ Native Node execution supports spill-backed `sort`, capped unsorted `pivot`, unsorted `unique`/`dedup`, unsorted `group-agg`, capped unsorted `join` inner/left, unsorted `semi-join`/`anti-join`, unsorted set operations, and duplicate-eliminating `union` when `spillDir` is provided. `run()`, `iterChunks()`, `toReadable()`, and `writeTo()` drain finish-time merge output through N-API `finishStep()` instead of waiting for one whole `finish()`. Standalone WASM supports `allowBlocking` and `memory`, but rejects `spillDir` because browser/WASM spill storage would occupy WASM memory. Use the Node native addon, CLI/direct C, or `{ engine: 'duckdb' }` for external spill; with DuckDB, `memory` maps to `memory_limit` and `spillDir` maps to `temp_directory`.
641
+
642
+ Plan-internal file reads are denied by default in Node and WASM. Pass `allowFs: true` for trusted local lookup files used by `join`, set operations, `union`, or `stack`; pass both `allowFs: true` and `allowRulesFile: true` for `validate rules_file=...`; pass `workspaceRoot` to pin resolved core plan paths inside a trusted directory. Supplying `spillDir` opts into local filesystem spill for Node native execution; standalone WASM still rejects spill. `inputFile` and `inputFiles` are host source adapters and are not controlled by `allowFs`.
643
+
280
644
  ## DuckDB engine
281
645
 
282
646
  Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
@@ -285,6 +649,7 @@ Run pipelines on DuckDB instead of the native C streaming core. The DSL is trans
285
649
  npm install duckdb
286
650
  ```
287
651
 
652
+ <!-- readme-test: js-data -->
288
653
  ```js
289
654
  import { pipeline, compileToSql } from 'tranfi'
290
655
 
@@ -301,8 +666,12 @@ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
301
666
 
302
667
  Generate SQL directly from DSL strings:
303
668
 
669
+ <!-- readme-test: js-sql -->
304
670
  ```js
305
- const sql = await compileToSql('csv | filter "col(age) > 25" | sort -age | head 10 | csv')
671
+ const sql = await compileToSql(
672
+ 'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
673
+ { dialect: 'duckdb' }
674
+ )
306
675
  console.log(sql)
307
676
  // WITH
308
677
  // step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
@@ -310,10 +679,15 @@ console.log(sql)
310
679
  // SELECT * FROM step_2
311
680
  ```
312
681
 
682
+ DuckDB is the only implemented SQL dialect today. `dialect: 'sqlite'` and
683
+ `dialect: 'postgres'` are recognized but rejected until those dialects have
684
+ their own compatibility tests and SQL-generation rules.
685
+
313
686
  ### Browser (WASM + DuckDB-WASM)
314
687
 
315
688
  In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
316
689
 
690
+ <!-- readme-test: browser -->
317
691
  ```js
318
692
  import createTranfi from 'tranfi/wasm'
319
693
  import * as duckdb from '@duckdb/duckdb-wasm'
@@ -321,7 +695,7 @@ import * as duckdb from '@duckdb/duckdb-wasm'
321
695
  const tf = await createTranfi()
322
696
 
323
697
  // SQL generation (synchronous, no DuckDB needed)
324
- const sql = tf.compileToSql('csv | filter "age > 25" | csv')
698
+ const sql = tf.compileToSql('csv | filter "age > 25" | csv', { dialect: 'duckdb' })
325
699
 
326
700
  // Full execution with DuckDB-WASM
327
701
  const db = new duckdb.AsyncDuckDB(...)
@@ -339,7 +713,7 @@ console.log(result.rows) // Array of row objects
339
713
  ### DSL compilation
340
714
 
341
715
  ```js
342
- import { compileDsl, saveRecipe, loadRecipe } from 'tranfi'
716
+ import { compileDsl, saveRecipe, loadRecipe, codec, ops } from 'tranfi'
343
717
 
344
718
  // Compile DSL to JSON plan
345
719
  const json = await compileDsl('csv | filter "col(age) > 25" | sort -age | csv')
@@ -356,17 +730,20 @@ Every pipeline produces four output channels:
356
730
 
357
731
  - **output** -- main pipeline result
358
732
  - **errors** -- rows that failed processing
359
- - **stats** -- pipeline execution statistics (rows in/out, timing)
733
+ - **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
360
734
  - **samples** -- reserved for sampling operators
361
735
 
736
+ <!-- readme-test: js-pipeline -->
362
737
  ```js
363
738
  const result = await p.run({ inputFile: 'data.csv' })
364
- console.log(result.statsText) // {"rows_in": 1000, "rows_out": 42, ...}
739
+ console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
365
740
  ```
366
741
 
367
742
  ### Pipeline from JSON
368
743
 
369
744
  ```js
745
+ import { loadRecipe } from 'tranfi'
746
+
370
747
  const p = await loadRecipe({
371
748
  steps: [
372
749
  { op: 'codec.csv.decode', args: {} },
@@ -381,7 +758,7 @@ const p = await loadRecipe({
381
758
  The package automatically selects the best backend:
382
759
 
383
760
  1. **N-API** (Node.js) -- native C addon, fastest, used when available
384
- 2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~313 KB single-file
761
+ 2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, single-file module
385
762
  3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
386
763
 
387
764
  ```js
@@ -392,4 +769,19 @@ const tf = await createTranfi()
392
769
 
393
770
  ## Architecture
394
771
 
395
- The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. All operators are streaming with bounded memory, except those that require full input (sort, unique, stats, tail, top, group-agg, frequency, pivot).
772
+ Node, Python, CLI and WASM use the same C11 engine. It processes columnar batches
773
+ with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`)
774
+ and per-cell null bitmaps.
775
+
776
+ | Need | Node API behavior |
777
+ |------|-------------------|
778
+ | Stream input | `run({ inputFile })` uses `createReadStream()` and drains output after each push and incremental finish boundary |
779
+ | Avoid collecting all output | Use `inputStream`, `toReadable()`, `writeTo()`, `iterChunks()`, or `onOutput` with `collectOutput: false` |
780
+ | Read gzip | `.gz` files use `zlib.createGunzip()`; override with `compression: "none"` or `"gzip"` for `inputFile`/`inputStream` |
781
+ | Permit a blocking operation | Set `allowBlocking: true` or use a supported `spillDir` path |
782
+ | Cap retained key state | Supply caps and check the plan with `memory: "64MB"` |
783
+ | Allow plan file reads | Supply explicit host-policy options |
784
+
785
+ Input streaming still collects the final output by default. Choose a streaming
786
+ sink for large outputs; row-local and bounded-state operators stream under the
787
+ default strict native memory policy.