tranfi 0.0.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/LICENSE +177 -21
  2. package/NOTICE +8 -0
  3. package/README.md +627 -0
  4. package/app/assets/index-6quYZ5Ap.css +5 -0
  5. package/app/assets/index-BIAIKnrp.js +160 -0
  6. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  7. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  8. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  9. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  10. package/app/index.html +13 -0
  11. package/binding.gyp +121 -0
  12. package/csrc/arena.c +93 -0
  13. package/csrc/batch.c +976 -0
  14. package/csrc/buffer.c +154 -0
  15. package/csrc/cJSON.c +3386 -0
  16. package/csrc/cJSON.h +316 -0
  17. package/csrc/codec_csv.c +1951 -0
  18. package/csrc/codec_jsonl.c +1086 -0
  19. package/csrc/codec_table.c +248 -0
  20. package/csrc/codec_text.c +447 -0
  21. package/csrc/compiler.c +130 -0
  22. package/csrc/config.h +21 -0
  23. package/csrc/date_utils.h +94 -0
  24. package/csrc/dsl.c +5417 -0
  25. package/csrc/dsl.h +22 -0
  26. package/csrc/expr.c +1553 -0
  27. package/csrc/expr.h +58 -0
  28. package/csrc/internal.h +539 -0
  29. package/csrc/ir.c +166 -0
  30. package/csrc/ir.h +208 -0
  31. package/csrc/ir_schema.c +75 -0
  32. package/csrc/ir_serialize.c +166 -0
  33. package/csrc/ir_sql.c +1822 -0
  34. package/csrc/ir_validate.c +576 -0
  35. package/csrc/json_path.c +210 -0
  36. package/csrc/main.c +1241 -0
  37. package/csrc/memory_estimate.c +477 -0
  38. package/csrc/op_acf.c +283 -0
  39. package/csrc/op_across.c +477 -0
  40. package/csrc/op_anomaly.c +255 -0
  41. package/csrc/op_assert.c +761 -0
  42. package/csrc/op_bin.c +248 -0
  43. package/csrc/op_cast.c +523 -0
  44. package/csrc/op_clip.c +99 -0
  45. package/csrc/op_date_trunc.c +355 -0
  46. package/csrc/op_datetime.c +394 -0
  47. package/csrc/op_derive.c +216 -0
  48. package/csrc/op_diff.c +250 -0
  49. package/csrc/op_ewma.c +222 -0
  50. package/csrc/op_explode.c +206 -0
  51. package/csrc/op_fill_down.c +235 -0
  52. package/csrc/op_fill_null.c +268 -0
  53. package/csrc/op_filter.c +181 -0
  54. package/csrc/op_frequency.c +721 -0
  55. package/csrc/op_grep.c +181 -0
  56. package/csrc/op_group_agg.c +1956 -0
  57. package/csrc/op_hash.c +159 -0
  58. package/csrc/op_head.c +84 -0
  59. package/csrc/op_interpolate.c +445 -0
  60. package/csrc/op_join.c +2902 -0
  61. package/csrc/op_json_extract.c +227 -0
  62. package/csrc/op_json_filter.c +384 -0
  63. package/csrc/op_json_flatten.c +293 -0
  64. package/csrc/op_json_schema.c +503 -0
  65. package/csrc/op_label_encode.c +419 -0
  66. package/csrc/op_lag.c +181 -0
  67. package/csrc/op_lead.c +242 -0
  68. package/csrc/op_normalize.c +510 -0
  69. package/csrc/op_onehot.c +457 -0
  70. package/csrc/op_pivot.c +1754 -0
  71. package/csrc/op_quarantine.c +189 -0
  72. package/csrc/op_registry.c +3044 -0
  73. package/csrc/op_rename.c +129 -0
  74. package/csrc/op_replace.c +354 -0
  75. package/csrc/op_rleid.c +297 -0
  76. package/csrc/op_rowid.c +559 -0
  77. package/csrc/op_sample.c +158 -0
  78. package/csrc/op_schema.c +1341 -0
  79. package/csrc/op_schema_infer.c +252 -0
  80. package/csrc/op_select.c +340 -0
  81. package/csrc/op_set.c +3449 -0
  82. package/csrc/op_skip.c +95 -0
  83. package/csrc/op_sort.c +819 -0
  84. package/csrc/op_source_name.c +120 -0
  85. package/csrc/op_split.c +151 -0
  86. package/csrc/op_split_data.c +119 -0
  87. package/csrc/op_stack.c +271 -0
  88. package/csrc/op_stats.c +875 -0
  89. package/csrc/op_step.c +333 -0
  90. package/csrc/op_tail.c +105 -0
  91. package/csrc/op_tee.c +338 -0
  92. package/csrc/op_top.c +357 -0
  93. package/csrc/op_trim.c +138 -0
  94. package/csrc/op_unique.c +1343 -0
  95. package/csrc/op_unpivot.c +193 -0
  96. package/csrc/op_validate.c +648 -0
  97. package/csrc/op_window.c +591 -0
  98. package/csrc/path_policy.c +85 -0
  99. package/csrc/pipeline.c +1088 -0
  100. package/csrc/recipes.c +104 -0
  101. package/csrc/recipes.h +27 -0
  102. package/csrc/report.c +506 -0
  103. package/csrc/report.h +22 -0
  104. package/csrc/selector.c +1097 -0
  105. package/csrc/size_utils.c +348 -0
  106. package/csrc/spill.c +317 -0
  107. package/csrc/spill.h +21 -0
  108. package/csrc/tranfi.h +291 -0
  109. package/csrc/transform.h +209 -0
  110. package/csrc/transform_api.c +2237 -0
  111. package/csrc/transform_categorical.c +923 -0
  112. package/csrc/transform_internal.h +472 -0
  113. package/csrc/transform_json.c +3812 -0
  114. package/csrc/transform_numeric.c +1966 -0
  115. package/csrc/transform_sha256.c +154 -0
  116. package/csrc/transform_wasm.h +162 -0
  117. package/csrc/transform_wasm_api.c +1373 -0
  118. package/csrc/wasm_api.c +218 -0
  119. package/napi_api.c +534 -0
  120. package/napi_transform.c +1648 -0
  121. package/napi_transform.h +8 -0
  122. package/package.json +64 -59
  123. package/scripts/install-native.js +76 -0
  124. package/scripts/prepack.js +64 -0
  125. package/scripts/sync-csrc.js +23 -0
  126. package/src/cli.js +190 -0
  127. package/src/engines/duckdb.js +142 -0
  128. package/src/index.js +925 -0
  129. package/src/memory_policy.js +411 -0
  130. package/src/native.js +18 -0
  131. package/src/pipeline.js +709 -0
  132. package/src/recipe_json.js +80 -0
  133. package/src/server.js +279 -0
  134. package/src/transform.js +403 -0
  135. package/src/transform_error.js +10 -0
  136. package/src/wasm.js +21 -0
  137. package/wasm/index.js +732 -0
  138. package/wasm/package.json +1 -0
  139. package/wasm/tranfi_core.js +0 -0
  140. package/wasm/transform.js +1156 -0
  141. package/wasm/worker.js +786 -0
  142. package/dist/bundle.js +0 -1
  143. package/index.html +0 -18
  144. package/src/app.css +0 -169
  145. package/src/app.js +0 -203
  146. package/src/app.vue +0 -250
  147. package/src/bulma-input.vue +0 -110
  148. package/src/common-inputs.js +0 -28
  149. package/src/main.js +0 -20
  150. package/src/transforms.js +0 -166
  151. package/webpack.config.js +0 -108
package/README.md ADDED
@@ -0,0 +1,627 @@
1
+ # tranfi (Node.js / WASM)
2
+
3
+ Streaming-first ETL in JavaScript, powered by a native C11 core via N-API
4
+ (Node.js) or WASM (browsers). Tranfi processes CSV, JSONL, and text byte streams;
5
+ it is not an in-memory DataFrame API. Row-local operations stream, bounded
6
+ operators declare their limits, and full-input operators require an explicit
7
+ blocking or spill policy.
8
+
9
+ > **Unreleased main:** The prepared-transform API documented below targets
10
+ > Tranfi 0.2. Current npm 0.1.x installs do not include it; build this branch
11
+ > from source until 0.2 is published.
12
+
13
+ Save this example as `quickstart.mjs`:
14
+
15
+ ```js
16
+ import tranfi from 'tranfi'
17
+
18
+ const { pipeline, codec, ops, expr } = tranfi
19
+
20
+ const result = await pipeline([
21
+ codec.csv(),
22
+ ops.filter(expr("col('age') > 25")),
23
+ ops.top(100, 'age'),
24
+ ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
25
+ ops.select(['name', 'age', 'label']),
26
+ codec.csvEncode(),
27
+ ]).run({ input: 'name,age\nAlice,30\nBob,25\nCharlie,35\nDiana,28\n' })
28
+
29
+ console.log(result.outputText)
30
+ // name,age,label
31
+ // Charlie,35,senior
32
+ // Alice,30,junior
33
+ // Diana,28,junior
34
+ ```
35
+
36
+ Or use the pipe DSL for one-liners:
37
+
38
+ ```js
39
+ const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
40
+ .run({ inputFile: 'data.csv' })
41
+ ```
42
+
43
+ ## Install
44
+
45
+ ```bash
46
+ npm install tranfi
47
+ ```
48
+
49
+ The default install compiles the N-API addon and reports a nonzero failure if
50
+ the native toolchain or synchronized C sources are unavailable. For an
51
+ intentional WASM-only/browser installation, set
52
+ `TRANFI_SKIP_NATIVE_BUILD=1` and import `tranfi/wasm` explicitly. The ordinary
53
+ pipeline can execute with WASM, but root-entry prepared transforms require the
54
+ native addon.
55
+
56
+ The native addon is currently built and tested on Linux. Windows is supported
57
+ through the packed package's `tranfi/wasm` entry with
58
+ `TRANFI_SKIP_NATIVE_BUILD=1`; the release CI runs streaming and prepared-transform
59
+ smokes in that configuration. This is not a Windows native-addon support claim.
60
+
61
+ Use `pipeline(...)` for byte-stream ETL. The separate
62
+ `TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
63
+ lifecycle is for typed batches whose learned state must be frozen and reused.
64
+ Calling `.run()` collects output for convenience; use `writeTo()` or
65
+ `toReadable()` for large outputs.
66
+
67
+ ## CLI
68
+
69
+ Installing the package also provides the `tranfi` command:
70
+
71
+ ```bash
72
+ # Via npx (no install)
73
+ echo 'name,age\nAlice,30\nBob,25' | npx tranfi -q 'csv | filter "age > 25" | csv'
74
+
75
+ # Or install globally
76
+ npm i -g tranfi
77
+ tranfi -q 'csv | filter "age > 25" | top-k 100 age | csv' < data.csv
78
+ tranfi profile < data.csv
79
+ tranfi -R # list recipes
80
+ ```
81
+
82
+ Run `tranfi -h` for all options.
83
+
84
+ ## Quick start
85
+
86
+ ### Two APIs
87
+
88
+ **Builder API** -- composable, structured, and IDE-friendly:
89
+
90
+ ```js
91
+ const p = pipeline([
92
+ codec.csv(),
93
+ ops.filter(expr("col('score') >= 80")),
94
+ ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
95
+ ops.top(10, 'score'),
96
+ codec.csvEncode(),
97
+ ])
98
+ const result = await p.run({ inputFile: 'students.csv' })
99
+ ```
100
+
101
+ **DSL strings** -- compact, suitable for CLI-like use:
102
+
103
+ ```js
104
+ const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
105
+ const result = await p.run({ inputFile: 'students.csv' })
106
+ ```
107
+
108
+ Both produce identical pipelines under the hood.
109
+
110
+ Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate` -> `derive`, `summarise`/`summarize` -> `group-agg`, `distinct` -> `unique`, and `arrange` -> `sort`.
111
+
112
+ ### Running pipelines
113
+
114
+ ```js
115
+ // From string or Buffer
116
+ const result = await p.run({ input: 'name,age\nAlice,30\n' })
117
+
118
+ // From file (streamed in 64 KB chunks)
119
+ const result = await p.run({ inputFile: 'data.csv' })
120
+
121
+ // Access results
122
+ result.output // Buffer
123
+ result.outputText // string (UTF-8 decoded)
124
+ result.errors // Buffer (error channel)
125
+ result.stats // Buffer (pipeline stats)
126
+ result.statsText // string
127
+ result.samples // Buffer (sample channel)
128
+ ```
129
+
130
+ For large outputs, drain chunks instead of collecting `result.output`:
131
+
132
+ ```js
133
+ const { createReadStream, createWriteStream } = require("fs")
134
+
135
+ // Write to a Node Writable and wait for `drain` when the sink applies backpressure.
136
+ const out = createWriteStream("out.csv")
137
+ const result = await p.writeTo(out, { inputFile: "data.csv" })
138
+
139
+ // Or consume Tranfi output as a Node Readable.
140
+ for await (const chunk of p.toReadable({ inputFile: "data.csv" })) {
141
+ // process each output chunk
142
+ }
143
+
144
+ // Existing input streams can feed the native pipeline without readFile() materialization.
145
+ const result2 = await p.run({ inputStream: createReadStream("data.csv") })
146
+
147
+ // .gz input files are decompressed as streaming sources by default.
148
+ const gz = await p.run({ inputFile: "events.jsonl.gz" })
149
+ const raw = await p.run({ inputFile: "events.jsonl.gz", compression: "none" })
150
+ const streamGz = await p.run({ inputStream: createReadStream("events.jsonl.gz"), compression: "gzip" })
151
+
152
+ // Multiple files stream sequentially through one pipeline.
153
+ const inputFiles = ["part-a.csv", "part-b.csv"]
154
+ const combined = await p.run({ inputFiles, sourceColumn: "src" })
155
+
156
+ // Low-level async iteration is still available.
157
+ for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
158
+ // process each output chunk
159
+ }
160
+ ```
161
+
162
+ `sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. It does not remove repeated CSV headers from later files; use shards without repeated headers, `header: false`, or a pre-cleaning step when every file has its own header.
163
+
164
+ Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
165
+
166
+ For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
167
+
168
+ ```js
169
+ // main thread
170
+ import { createWorkerClient } from 'tranfi/wasm/worker'
171
+
172
+ const client = createWorkerClient(new Worker(new URL('./tranfi-worker.js', import.meta.url), { type: 'module' }))
173
+ const result = await client.runFile('csv | filter "col(age) >= 18" | csv', file, {
174
+ chunkSize: 64 * 1024,
175
+ collectOutput: false,
176
+ onOutput: chunk => downloadSink.write(chunk),
177
+ onProgress: p => updateProgress(p),
178
+ signal: abortController.signal
179
+ })
180
+
181
+ // tranfi-worker.js, bundled by the app
182
+ import { runWorkerServer } from 'tranfi/wasm/worker'
183
+ runWorkerServer()
184
+ ```
185
+
186
+ The Worker protocol streams chunks with transferable buffers and sends progress, stats, errors, and cancellation messages around the same WASM `create/push/pull/finish/free` API; it is not a separate IR target.
187
+
188
+ The bundled Tranfi app runner is also preview-bounded by default: file chunks are streamed into WASM, main output is drained incrementally into a table preview capped by hidden `preview_rows` (default 200), and full output text is only materialized when `collect_output` is explicitly true in the schema.
189
+
190
+ ### Prepared reusable transforms
191
+
192
+ WASM cancellation tokens created with `createTransformCancelToken()` expose a read-only `requested` boolean for host-side conversion loops. Reading a closed token fails.
193
+
194
+ Prepared transforms are a separate typed-table API for operations whose parameters must be learned from reference data. `analyze` accumulates bounded statistics over one or more batches, `finalize` freezes an immutable plan and output schema, and `apply` runs that plan either over a second pass of the original data (`fit_transform`-style) or over later compatible batches. It does not replace the byte-stream pipeline API.
195
+
196
+ The current slice accepts declared `float32`/`float64` columns. It supports numeric none/zero/constant/mean/exact-median imputation and none/standard/min-max normalization. Declared categorical columns support mode imputation with `allMissing: 'error' | 'zero'` or `impute.op: 'none'`; encoders discover finite typed categories, while the no-imputation/no-encoding combination needs no learned dictionary. A column may instead provide both branches with `kind.op: 'infer'`, `kind.rule: 'finite-integer-cardinality-v1'`, and `kind.maxCategories >= 2`: missing/NaN values are ignored, `2..maxCategories` distinct finite integers resolve categorical, while zero/one distinct value, any noninteger, or the next distinct value resolves numeric. `impute.op: 'none'` with `encode.op: 'none'` passes every finite value through and emits canonical qNaN for missing input. Mode with `encode.op: 'none'` retains its learned dictionary and rejects unseen finite values with code `108`. `encode.op: 'label'` freezes zero-based sorted ordinals and supports `unknown: 'error' | 'sentinel' | 'other'`; `encode.op: 'onehot'` emits source-ordered category blocks and supports `unknown: 'error' | 'all_zero' | 'other'`. Without categorical imputation, missing label/one-hot input follows that encoder's unknown policy. The label sentinel must be a safe integer outside the learned ordinal range; label/one-hot `other` appends the reserved ordinal/field after known categories. Generated output IDs/names and one-hot category metadata are deterministic, and collisions fail with code `102`. Exact median obeys the configured allocation and resident-state limits; inference, categorical discovery, and output expansion obey category, output-width, allocation, and resident-state limits. Limit failures use resource code `104`. Prepared-transform host-policy and spill fields are reserved but not implemented; nonempty use fails with unsupported-runtime code `113` instead of being silently ignored.
197
+
198
+ Declared categorical columns also accept a fixed, nonempty `encode.categories` array of finite, sorted, unique tags (`{ "t": "f64", "v": "4000000000000000" }` represents 2). Tags must match the input dtype; negative zero is represented as positive zero. Fixed dictionaries retain unobserved categories. Unknown training values follow the encoder policy and never vote for mode; an all-missing zero fallback must exist in the dictionary. Fixed encoding without imputation can finalize without training rows. Fixed dictionaries with kind inference, string categories, and categorical constant imputation remain unsupported.
199
+
200
+ Prepared runtime option objects reject unknown fields. Native Node accepts
201
+ `limits`, `cancelFlag`, `hostPolicy`, and `spillDir`; standalone WASM accepts
202
+ `limits`, `cancelToken`, `hostPolicy`, and `spillDir`. The host/spill fields are
203
+ reserved as described above, and cancellation spellings are not interchangeable.
204
+
205
+ ```js
206
+ const tf = require('tranfi')
207
+
208
+ const recipeSpec = {
209
+ format: 'tranfi.transform-recipe',
210
+ version: 1,
211
+ policyVersion: 1,
212
+ outputDtype: 'float64',
213
+ semanticLimits: {
214
+ maxOutputColumns: 65536,
215
+ maxOutputElementsPerApply: 134217728
216
+ },
217
+ columns: [{
218
+ sourceId: 'x0',
219
+ kind: { op: 'declared', value: 'numeric', rule: null, maxCategories: null },
220
+ numeric: {
221
+ impute: { op: 'mean', constant: null, allMissing: 'zero' },
222
+ normalize: { op: 'standard', ddof: 0 }
223
+ },
224
+ categorical: null
225
+ }]
226
+ }
227
+ const schema = [{ id: 'x0', dtype: 'float64' }]
228
+
229
+ const recipe = tf.TransformRecipe.fromJSON(recipeSpec)
230
+ const analyzer = recipe.analyzer(schema)
231
+ analyzer.push({ rows: 3, columns: [new Float64Array([1, NaN, 3])] })
232
+ const plan = analyzer.finalize()
233
+
234
+ const apply = plan.apply(schema)
235
+ const fittedReference = apply.run({
236
+ rows: 3,
237
+ columns: [new Float64Array([1, NaN, 3])]
238
+ })
239
+ const planBytes = plan.toBytes() // canonical TFTR artifact
240
+ const recipeSha256 = plan.recipeSha256() // canonical recipe + input-schema identity
241
+
242
+ apply.close()
243
+ plan.close()
244
+ analyzer.close()
245
+ recipe.close()
246
+ ```
247
+
248
+ `tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
249
+
250
+ ```js
251
+ const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
252
+ const makeWorker = () => new Worker(workerUrl, { type: 'module' })
253
+ const client = createWorkerClient(makeWorker(), {
254
+ workerFactory: makeWorker,
255
+ terminateOnDispose: true
256
+ })
257
+ ```
258
+
259
+ All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
260
+
261
+ ## Codecs
262
+
263
+ Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
264
+
265
+ | Method | Description |
266
+ |--------|-------------|
267
+ | `codec.csv({ delimiter, header, batchSize, repair, mode, strict, maxErrorBytes, maxRecordBytes, maxColumns, nulls, quotedNulls, skip, nMax, maxRows, comment, trimWs, skipEmptyRows, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | CSV decoder. `mode: 'strict'` fails on field-count mismatches; `repair: true` / `mode: 'repair'` emits repair diagnostics; `maxRecordBytes` bounds buffered records; `nulls: ['NA']` adds null sentinels; `skip: 2`, `nMax: 100`, `comment: '#'`, `trimWs: false`, and `skipEmptyRows: true` control row-local parsing; repair audit/raw diagnostics support privacy controls with pseudo-column `raw` |
268
+ | `codec.csvEncode({ delimiter })` | CSV encoder |
269
+ | `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | JSON Lines decoder. `onError` is `skip`, `fail`, `warn`, or `quarantine` |
270
+ | `codec.jsonlEncode()` | JSON Lines encoder |
271
+ | `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | Line-oriented text decoder (single `_line` column) |
272
+ | `codec.textEncode()` | Text encoder |
273
+ | `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
274
+
275
+ CSV treats unquoted empty fields as null by default. Pass `nulls: ['NA', 'NULL']` to add sentinel strings; pass `quotedNulls: false` when quoted sentinels like `"NA"` or `""` should remain strings. Default field-count handling is permissive for compatibility. Use `mode: 'strict'` or `strict: true` to fail on row/header width mismatches. Use `repair: true` or `mode: 'repair'` to pad/truncate and collect JSONL diagnostics in `result.errors`; `maxErrorBytes` bounds raw previews, and `auditIncludeRow`, `auditColumns`, `auditRedact`, `auditHashColumns`, `auditMaxBytes`, and `auditMaxCellBytes` govern repair audit/raw payloads with the raw preview exposed as pseudo-column `raw`. `maxRecordBytes` defaults to `67108864` bytes, caps the current record buffer before a newline is seen, and rejects with a `csv_record_too_large` diagnostic when exceeded; pass `0` to disable the guard. `maxColumns` defaults to `8192`; records above the cap reject with a bounded `csv_too_many_columns` diagnostic instead of silently dropping columns. Decoder size options are checked before execution: `batchSize` must be `1..65536`, `maxErrorBytes` must be `0..67108864`, `maxRecordBytes` must be `0..1073741824`, and `maxColumns` must be `1..65536`. `skip: 2` discards preamble records before header/schema discovery; comments are applied after skipped rows. `nMax: 100` / `maxRows: 100` keeps at most that many decoded data rows after skip/comment/header handling and uses only one counter; `nMax: 0` preserves a header-only schema batch. `comment: '#'` removes text after an unquoted marker and skips comment-only rows; quoted markers are preserved. Unquoted spaces/tabs are trimmed by default; pass `trimWs: false` to preserve them. Blank physical rows after the header are preserved as all-null rows by default; pass `skipEmptyRows: true` to drop them.
276
+
277
+ Malformed JSONL records are skipped by default. Set `onError: 'warn'` or `onError: 'quarantine'` to keep valid rows and collect JSONL diagnostics in `result.errors`; set `onError: 'fail'` to reject on the first malformed line. `maxErrorBytes` bounds the raw preview stored in diagnostics, and `maxRecordBytes` applies the same current-line guard as CSV/text.
278
+
279
+ Each JSONL record must be one complete UTF-8 JSON object. Invalid number syntax, trailing content, embedded NULs, and numbers outside the finite float64 range follow the same malformed-record policy. The encoder rejects nonfinite values. Schema order comes from the first valid record; repeated keys use their first value.
280
+
281
+
282
+ Cross-codec pipelines work naturally:
283
+
284
+ ```js
285
+ // CSV in, JSONL out
286
+ pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
287
+
288
+ // JSONL in, CSV out
289
+ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
290
+ ```
291
+
292
+ ## Operators
293
+
294
+ ### Row filtering
295
+
296
+ | Method | Description |
297
+ |--------|-------------|
298
+ | `ops.filter(expr, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Keep rows matching expression; optional dropped-row audit records support row omission, column allowlists, redaction, hashes, and payload caps |
299
+ | `ops.head(n)` | First N rows |
300
+ | `ops.tail(n)` | Last N rows |
301
+ | `ops.skip(n)` | Skip first N rows |
302
+ | `ops.top(n, column, desc?)` | Top N by column value |
303
+ | `ops.sample(n, { seed })` | Deterministic bounded reservoir sampling; use `seed: 'random'` for nondeterministic mode |
304
+ | `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
305
+ | `ops.validate(expr, { rules, rulesFile, audit, auditLimit, maxFailures, warnFailureRate, maxFailureRate, name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Add `_valid` boolean column, keep all rows; supports inline rules or local JSON rule-suite files; bounded failure audit records support count/rate thresholds plus privacy controls |
306
+ | `ops.assert(expr, { action, name, message, result, aggregate, op, value, column, tolerance, rel, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Row-local data-quality rule or finish-time O(1) aggregate assertion; row-local failure side-channel records support privacy controls |
307
+ | `ops.quarantine(expr, { name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Route rows matching expression to `errors` and drop them from main output; row payloads support privacy controls |
308
+ | `ops.schema({ columns, required, nonNull, nullable, values, min, max, regex, mode='fail', result='_schema', maxRegexPatternBytes, maxRegexCellBytes, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Row-local table schema contract; fail, warn, filter, quarantine, or annotate; regex budgets default to 4096-byte patterns and 65536-byte cells; schema audit/error records support row omission, column allowlists, redaction, stable non-cryptographic hashes, and row/cell payload caps |
309
+ | `ops.schemaInfer({ rows })` | Bounded decoded-type/nullability schema report; defaults to 10000 sampled rows |
310
+ | `ops.tee({ expr, channel, columns, limit, every, name, includeRow, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Preserve main rows and write bounded JSONL row snapshots to a side channel; row payloads support privacy controls |
311
+
312
+ ### Column operations
313
+
314
+ | Method | Description |
315
+ |--------|-------------|
316
+ | `ops.select(columns)` | Keep and reorder columns |
317
+ | `ops.relocate(columns, { before, after })` | Move columns while preserving all columns |
318
+ | `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
319
+ | `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
320
+ | `ops.sourceName({ result, defaultValue })` | Append the current host source path/name as a row-local string column |
321
+ | `ops.across(columns, { fn, functions, names, replace })` | Apply row-local functions over selected columns |
322
+ | `ops.cast(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Type conversion; optional bounded value/coercion audit records support privacy controls |
323
+ | `ops.trim(columns?)` | Strip whitespace |
324
+ | `ops.fillNull(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Replace nulls; optional bounded audit records support privacy controls |
325
+ | `ops.fillDown(columns?)` | Forward-fill nulls |
326
+ | `ops.clip(column, { min, max })` | Clamp numeric values |
327
+ | `ops.replace(column, pattern, replacement, { regex, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | String find/replace; optional bounded audit records support privacy controls |
328
+ | `ops.hash(columns?)` | Add `_hash` column (DJB2) |
329
+ | `ops.bin(column, boundaries, { missing, onTypeError })` | Discretize into bins with strict numeric-source defaults |
330
+ | `ops.ewma(column, alpha, { result, missing, onTypeError })` | Exponentially weighted moving average |
331
+ | `ops.anomaly(column, { threshold, result, missing, onTypeError })` | Streaming z-score anomaly flag |
332
+ | `ops.normalize(columns, { method, audit, auditLimit, missing, onTypeError, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Blocking minmax/zscore normalization; optional audit records support privacy controls |
333
+ | `ops.acf(column, { lags, missing, onTypeError })` | Blocking autocorrelation table |
334
+
335
+ `columns` may contain exact names or selector strings: `id:score`, `starts_with(score_)`, `ends_with(_id)`, `contains(temp)`, `matches(^score_)`, `where(numeric)`, strict `all_of(score,name)`, lenient `any_of(optional,score)`, exclusions with `!name` or `-name`, and boolean selector algebra such as `starts_with(score_)&where(numeric)`, `starts_with(score_)&!ends_with(raw)`, or `!(id:score)`. Ranges use input schema order and can be reversed. Helper matching is case-insensitive; exact names are case-sensitive. These selectors are resolved by the native schema-aware `select`, `relocate`, and `across` ops in `O(columns)` without retaining rows; SQL lowering rejects selector helpers without a known schema.
336
+
337
+ Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selected numeric columns per row; `ops.across(['name'], { functions: ['lower'], replace: false, names: '{col}_{fn}' })` appends templated columns. This is native/WASM row-local execution, not arbitrary lambdas or grouped dplyr evaluation.
338
+
339
+ ### Sorting and deduplication
340
+
341
+ | Method | Description |
342
+ |--------|-------------|
343
+ | `ops.sort(columns)` | Sort rows. Prefix `-` for descending: `sort(['-age', 'name'])` |
344
+ | `ops.unique(columns?)` | Deduplicate on specified columns |
345
+
346
+ ### Aggregation
347
+
348
+ | Method | Description |
349
+ |--------|-------------|
350
+ | `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
351
+ | `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
352
+ | `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
353
+
354
+ ```js
355
+ // Group aggregation
356
+ ops.groupAgg(['city'], [
357
+ { column: 'price', func: 'sum', name: 'total' },
358
+ { column: 'price', func: 'avg', name: 'avg_price' },
359
+ { column: '*', func: 'count', name: 'rows' },
360
+ ])
361
+ ```
362
+
363
+ ### Sequential / window
364
+
365
+ | Method | Description |
366
+ |--------|-------------|
367
+ | `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
368
+ | `ops.window(column, size, func, resultOrOptions?)` | Sliding window: `avg`, `sum`, `min`, `max`; options include `result`, `missing`, `onTypeError` |
369
+ | `ops.rollingSum/rollingMean/rollingMin/rollingMax(column, size, { result, missing, onTypeError })` | Named trailing fixed-row numeric windows; default missing/non-numeric source fails |
370
+ | `ops.rollingAny/rollingAll(column, size, { result, nulls })` | Boolean trailing windows; `nulls`: `ignore`, `false`, `true`, `propagate` |
371
+ | `ops.lead(column, { offset, result })` | Lookahead N rows |
372
+ | `ops.lag(column, { offset, result })` | Previous-row shift |
373
+ | `ops.shift(column, { offset, result, type })` | Shift alias; `type='lag'` default or `type='lead'` |
374
+ | `ops.rowid(columns, { result, sorted, maxKeys })` | Global or per-key 1-based row ids |
375
+ | `ops.rleid(columns, { result })` | Consecutive run id by selected column(s) |
376
+
377
+ ### Reshape
378
+
379
+ | Method | Description |
380
+ |--------|-------------|
381
+ | `ops.explode(column, delimiter?, { maxTokensPerRow, maxOutputRowsPerInputRow, maxOutputRowsPerBatch, maxTokenBytes })` | Split delimited string into rows with optional expansion caps |
382
+ | `ops.split(column, names, delimiter?)` | Split column into multiple columns |
383
+ | `ops.unpivot(columns, { maxOutputRowsPerInputRow, maxOutputRowsPerBatch })` | Wide to long (melt) with optional expansion caps |
384
+ | `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
385
+
386
+ `explode` caps fail fast with `maxTokensPerRow`, `maxOutputRowsPerInputRow`, `maxOutputRowsPerBatch`, or `maxTokenBytes` when a single row or batch would expand beyond the configured limit. `unpivot` supports `maxOutputRowsPerInputRow` and `maxOutputRowsPerBatch`.
387
+
388
+ ### Date/time
389
+
390
+ | Method | Description |
391
+ |--------|-------------|
392
+ | `ops.datetime(column, extractOrOptions?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday`; accepts `{ extract, missing, onTypeError }` |
393
+ | `ops.dateTrunc(column, trunc, { result, missing, onTypeError })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
394
+
395
+ ### Other
396
+
397
+ | Method | Description |
398
+ |--------|-------------|
399
+ | `ops.flatten()` | Flatten nested columns |
400
+ | `ops.interpolate(column, { method, missing, onTypeError })` | Fill nulls in a numeric column; default missing/non-numeric source fails |
401
+ | `ops.jsonExtract(path, result, { column, type })` | Extract JSON Pointer/simple JSONPath value into a new column |
402
+ | `ops.jsonFilter(path, { op, value, column, type })` | Filter rows by JSON Pointer/simple JSONPath predicate |
403
+ | `ops.jsonSchema(schema, { column, mode, result, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Validate JSON text with a supported JSON Schema subset; filter-mode audit records support privacy controls |
404
+ | `ops.jsonFlatten(fields, { column })` | Append declared JSON Pointer/simple JSONPath fields as bounded output columns |
405
+ | `ops.reorder(columns)` | Alias for `select` |
406
+ | `ops.dedup(columns?)` | Alias for `unique` |
407
+
408
+ ## Expressions
409
+
410
+ Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
411
+
412
+ ```js
413
+ ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
414
+ ops.derive({
415
+ full: expr("concat(col('first'), ' ', col('last'))"),
416
+ grade: expr("if(col('score')>=90, 'A', if(col('score')>=80, 'B', 'C'))"),
417
+ })
418
+ ```
419
+
420
+ ### Available functions
421
+
422
+ | Category | Functions |
423
+ |----------|-----------|
424
+ | Arithmetic | `+` `-` `*` `/` |
425
+ | Comparison | `>` `>=` `<` `<=` `==` `!=` |
426
+ | Logic | `and` `or` `not` |
427
+ | String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
428
+ | Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` `between(x,left,right)` `inrange(x,left,right)` |
429
+ | Date/time | `year(x)` `month(x)` `day(x)` `hour(x)` `minute(x)` `second(x)` `weekday(x)` `epoch(x)` `date_trunc(x,unit)` |
430
+ | Conditional | `if(cond,then,else)` `case_when(cond,value,...,default)` `case_match(value,key,result,...,default)` `if_any(pred,...)` `if_all(pred,...)` `coalesce(a,b,...)` `nullif(a,b)` |
431
+ | Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
432
+
433
+ Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
434
+
435
+ ```js
436
+ ops.derive({
437
+ year: expr("year(col('date'))"),
438
+ monthStart: expr("date_trunc(col('date'), 'month')"),
439
+ })
440
+ ```
441
+
442
+ ## Recipes
443
+
444
+ Built-in named pipelines for common tasks. Use by name:
445
+
446
+ ```js
447
+ const result = await pipeline('preview').run({ inputFile: 'data.csv' })
448
+ const result = await pipeline('freq').run({ inputFile: 'data.csv' })
449
+ ```
450
+
451
+ | Recipe | Pipeline | Description |
452
+ |--------|----------|-------------|
453
+ | `profile` | `csv \| stats \| csv` | Full data profiling |
454
+ | `preview` | `csv \| head 10 \| csv` | First 10 rows |
455
+ | `schema` | `csv \| head 0 \| csv` | Column names only |
456
+ | `sniff` | `csv \| schema infer rows=1000 \| csv` | Bounded-memory schema/type/nullability sniff |
457
+ | `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
458
+ | `count` | `csv \| stats count \| csv` | Row count |
459
+ | `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
460
+ | `distro` | `csv \| stats min,p25,median,p75,max \| csv` | Five-number summary |
461
+ | `freq` | `csv \| frequency \| csv` | Value frequency |
462
+ | `dedup` | `csv \| dedup \| csv` | Remove duplicates |
463
+ | `clean` | `csv \| trim \| csv` | Trim whitespace |
464
+ | `sample` | `csv \| sample 100 \| csv` | Random 100 rows |
465
+ | `head` | `csv \| head 20 \| csv` | First 20 rows |
466
+ | `tail` | `csv \| tail 20 \| csv` | Last 20 rows |
467
+ | `csv2json` | `csv \| jsonl` | CSV to JSONL |
468
+ | `json2csv` | `jsonl \| csv` | JSONL to CSV |
469
+ | `tsv2csv` | `csv delimiter="\t" \| csv` | TSV to CSV |
470
+ | `csv2tsv` | `csv \| csv delimiter="\t"` | CSV to TSV |
471
+ | `look` | `csv \| table` | Pretty-print table |
472
+ | `histogram` | `csv \| stats hist \| csv` | Distribution histograms |
473
+ | `hash` | `csv \| hash \| csv` | Row hash for change detection |
474
+ | `samples` | `csv \| stats sample \| csv` | Sample values per column |
475
+
476
+ List all recipes programmatically:
477
+
478
+ ```js
479
+ import { recipes } from 'tranfi'
480
+
481
+ for (const r of await recipes()) {
482
+ console.log(`${r.name.padEnd(15)} ${r.description}`)
483
+ }
484
+ ```
485
+
486
+
487
+ ## Memory policy
488
+
489
+ Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
490
+
491
+ ```js
492
+ await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
493
+ ```
494
+
495
+ For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
496
+
497
+ ```js
498
+ await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
499
+ ```
500
+
501
+ Native Node execution supports spill-backed `sort`, capped unsorted `pivot`, unsorted `unique`/`dedup`, unsorted `group-agg`, capped unsorted `join` inner/left, unsorted `semi-join`/`anti-join`, unsorted set operations, and duplicate-eliminating `union` when `spillDir` is provided. `run()`, `iterChunks()`, `toReadable()`, and `writeTo()` drain finish-time merge output through N-API `finishStep()` instead of waiting for one whole `finish()`. Standalone WASM supports `allowBlocking` and `memory`, but rejects `spillDir` because browser/WASM spill storage would occupy WASM memory. Use the Node native addon, CLI/direct C, or `{ engine: 'duckdb' }` for external spill; with DuckDB, `memory` maps to `memory_limit` and `spillDir` maps to `temp_directory`.
502
+
503
+ Plan-internal file reads are denied by default in Node and WASM. Pass `allowFs: true` for trusted local lookup files used by `join`, set operations, `union`, or `stack`; pass both `allowFs: true` and `allowRulesFile: true` for `validate rules_file=...`; pass `workspaceRoot` to pin resolved core plan paths inside a trusted directory. Supplying `spillDir` opts into local filesystem spill for Node native execution; standalone WASM still rejects spill. `inputFile` and `inputFiles` are host source adapters and are not controlled by `allowFs`.
504
+
505
+ ## DuckDB engine
506
+
507
+ Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
508
+
509
+ ```bash
510
+ npm install duckdb
511
+ ```
512
+
513
+ ```js
514
+ import { pipeline, compileToSql } from 'tranfi'
515
+
516
+ // Run a pipeline via DuckDB
517
+ const result = await pipeline('csv | filter "age > 25" | sort -age | csv', { engine: 'duckdb' })
518
+ .run({ inputFile: 'data.csv' })
519
+
520
+ // Or with string/Buffer input
521
+ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
522
+ .run({ input: csvString })
523
+ ```
524
+
525
+ ### SQL transpilation
526
+
527
+ Generate SQL directly from DSL strings:
528
+
529
+ ```js
530
+ const sql = await compileToSql(
531
+ 'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
532
+ { dialect: 'duckdb' }
533
+ )
534
+ console.log(sql)
535
+ // WITH
536
+ // step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
537
+ // step_2 AS (SELECT * FROM step_1 ORDER BY "age" DESC LIMIT 10)
538
+ // SELECT * FROM step_2
539
+ ```
540
+
541
+ DuckDB is the only implemented SQL dialect today. `dialect: 'sqlite'` and
542
+ `dialect: 'postgres'` are recognized but rejected until those dialects have
543
+ their own compatibility tests and SQL-generation rules.
544
+
545
+ ### Browser (WASM + DuckDB-WASM)
546
+
547
+ In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
548
+
549
+ ```js
550
+ import createTranfi from 'tranfi/wasm'
551
+ import * as duckdb from '@duckdb/duckdb-wasm'
552
+
553
+ const tf = await createTranfi()
554
+
555
+ // SQL generation (synchronous, no DuckDB needed)
556
+ const sql = tf.compileToSql('csv | filter "age > 25" | csv', { dialect: 'duckdb' })
557
+
558
+ // Full execution with DuckDB-WASM
559
+ const db = new duckdb.AsyncDuckDB(...)
560
+ await db.instantiate(...)
561
+
562
+ const result = await tf.runDuckDB(db, 'csv | filter "age > 25" | csv', csvData)
563
+ console.log(result.outputText) // CSV output
564
+ console.log(result.rows) // Array of row objects
565
+ ```
566
+
567
+ `runDuckDB` accepts string, `Uint8Array`, or `File` objects as data input.
568
+
569
+ ## Advanced
570
+
571
+ ### DSL compilation
572
+
573
+ ```js
574
+ import { compileDsl, saveRecipe, loadRecipe } from 'tranfi'
575
+
576
+ // Compile DSL to JSON plan
577
+ const json = await compileDsl('csv | filter "col(age) > 25" | sort -age | csv')
578
+
579
+ // Save / load recipes
580
+ await saveRecipe([codec.csv(), ops.head(10), codec.csvEncode()], 'preview.tranfi')
581
+ const p = await loadRecipe('preview.tranfi')
582
+ const result = await p.run({ inputFile: 'data.csv' })
583
+ ```
584
+
585
+ ### Side channels
586
+
587
+ Every pipeline produces four output channels:
588
+
589
+ - **output** -- main pipeline result
590
+ - **errors** -- rows that failed processing
591
+ - **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
592
+ - **samples** -- reserved for sampling operators
593
+
594
+ ```js
595
+ const result = await p.run({ inputFile: 'data.csv' })
596
+ console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
597
+ ```
598
+
599
+ ### Pipeline from JSON
600
+
601
+ ```js
602
+ const p = await loadRecipe({
603
+ steps: [
604
+ { op: 'codec.csv.decode', args: {} },
605
+ { op: 'head', args: { n: 5 } },
606
+ { op: 'codec.csv.encode', args: {} },
607
+ ]
608
+ })
609
+ ```
610
+
611
+ ### Backend selection
612
+
613
+ The package automatically selects the best backend:
614
+
615
+ 1. **N-API** (Node.js) -- native C addon, fastest, used when available
616
+ 2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~500 KB single-file
617
+ 3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
618
+
619
+ ```js
620
+ // Force WASM backend (e.g., for testing)
621
+ import createTranfi from 'tranfi/wasm'
622
+ const tf = await createTranfi()
623
+ ```
624
+
625
+ ## Architecture
626
+
627
+ The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. `run({ inputFile })` streams files with `createReadStream()` and drains native/WASM main output after each push and each incremental finish boundary; by default it still collects the final output for convenience. `.gz` input files are decompressed through a `zlib.createGunzip()` source transform; use `compression: "none"` to force raw bytes or `compression: "gzip"` to force gzip for `inputFile`/`inputStream`. Use `inputStream`, `toReadable()`, `writeTo()`, `onOutput` with `collectOutput: false`, or `iterChunks()` for large inputs/outputs and backpressure-aware sinks. Native execution is strict by default: row-local and bounded-state operators stream, blocking operators require `allowBlocking: true` or supported `spillDir`, capped key-state plans can be checked with `memory: "64MB"`, and core plan file reads require explicit host-policy options.