tranfi 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +272 -40
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +348 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +8 -11
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/README.md CHANGED
@@ -1,14 +1,26 @@
1
1
  # tranfi (Node.js / WASM)
2
2
 
3
- Streaming ETL in JavaScript, powered by a native C11 core via N-API (Node.js) or WASM (browsers). Process CSV, JSONL, and text data with composable pipelines that run in constant memory, no matter how large the input.
3
+ Streaming-first ETL in JavaScript, powered by a native C11 core via N-API
4
+ (Node.js) or WASM (browsers). Tranfi processes CSV, JSONL, and text byte streams;
5
+ it is not an in-memory DataFrame API. Row-local operations stream, bounded
6
+ operators declare their limits, and full-input operators require an explicit
7
+ blocking or spill policy.
8
+
9
+ > **Unreleased main:** The prepared-transform API documented below targets
10
+ > Tranfi 0.2. Current npm 0.1.x installs do not include it; build this branch
11
+ > from source until 0.2 is published.
12
+
13
+ Save this example as `quickstart.mjs`:
4
14
 
5
15
  ```js
6
- import { pipeline, codec, ops, expr } from 'tranfi'
16
+ import tranfi from 'tranfi'
17
+
18
+ const { pipeline, codec, ops, expr } = tranfi
7
19
 
8
20
  const result = await pipeline([
9
21
  codec.csv(),
10
22
  ops.filter(expr("col('age') > 25")),
11
- ops.sort(['-age']),
23
+ ops.top(100, 'age'),
12
24
  ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
13
25
  ops.select(['name', 'age', 'label']),
14
26
  codec.csvEncode(),
@@ -24,7 +36,7 @@ console.log(result.outputText)
24
36
  Or use the pipe DSL for one-liners:
25
37
 
26
38
  ```js
27
- const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
39
+ const result = await pipeline('csv | filter "col(age) > 25" | top-k 100 age | csv')
28
40
  .run({ inputFile: 'data.csv' })
29
41
  ```
30
42
 
@@ -34,7 +46,23 @@ const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
34
46
  npm install tranfi
35
47
  ```
36
48
 
37
- Uses N-API natively in Node.js, falls back to WASM in browsers automatically.
49
+ The default install compiles the N-API addon and reports a nonzero failure if
50
+ the native toolchain or synchronized C sources are unavailable. For an
51
+ intentional WASM-only/browser installation, set
52
+ `TRANFI_SKIP_NATIVE_BUILD=1` and import `tranfi/wasm` explicitly. The ordinary
53
+ pipeline can execute with WASM, but root-entry prepared transforms require the
54
+ native addon.
55
+
56
+ The native addon is currently built and tested on Linux. Windows is supported
57
+ through the packed package's `tranfi/wasm` entry with
58
+ `TRANFI_SKIP_NATIVE_BUILD=1`; the release CI runs streaming and prepared-transform
59
+ smokes in that configuration. This is not a Windows native-addon support claim.
60
+
61
+ Use `pipeline(...)` for byte-stream ETL. The separate
62
+ `TransformRecipe -> TransformAnalyzer -> TransformPlan -> TransformApply`
63
+ lifecycle is for typed batches whose learned state must be frozen and reused.
64
+ Calling `.run()` collects output for convenience; use `writeTo()` or
65
+ `toReadable()` for large outputs.
38
66
 
39
67
  ## CLI
40
68
 
@@ -42,11 +70,11 @@ Installing the package also provides the `tranfi` command:
42
70
 
43
71
  ```bash
44
72
  # Via npx (no install)
45
- echo 'name,age\nAlice,30\nBob,25' | npx tranfi 'csv | filter "age > 25" | csv'
73
+ echo 'name,age\nAlice,30\nBob,25' | npx tranfi -q 'csv | filter "age > 25" | csv'
46
74
 
47
75
  # Or install globally
48
76
  npm i -g tranfi
49
- tranfi 'csv | filter "age > 25" | sort -age | csv' < data.csv
77
+ tranfi -q 'csv | filter "age > 25" | top-k 100 age | csv' < data.csv
50
78
  tranfi profile < data.csv
51
79
  tranfi -R # list recipes
52
80
  ```
@@ -57,15 +85,14 @@ Run `tranfi -h` for all options.
57
85
 
58
86
  ### Two APIs
59
87
 
60
- **Builder API** -- composable, type-safe, IDE-friendly:
88
+ **Builder API** -- composable, structured, and IDE-friendly:
61
89
 
62
90
  ```js
63
91
  const p = pipeline([
64
92
  codec.csv(),
65
93
  ops.filter(expr("col('score') >= 80")),
66
94
  ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
67
- ops.sort(['-score']),
68
- ops.head(10),
95
+ ops.top(10, 'score'),
69
96
  codec.csvEncode(),
70
97
  ])
71
98
  const result = await p.run({ inputFile: 'students.csv' })
@@ -74,12 +101,14 @@ const result = await p.run({ inputFile: 'students.csv' })
74
101
  **DSL strings** -- compact, suitable for CLI-like use:
75
102
 
76
103
  ```js
77
- const p = pipeline('csv | filter "col(score) >= 80" | sort -score | head 10 | csv')
104
+ const p = pipeline('csv | filter "col(score) >= 80" | top-k 10 score | csv')
78
105
  const result = await p.run({ inputFile: 'students.csv' })
79
106
  ```
80
107
 
81
108
  Both produce identical pipelines under the hood.
82
109
 
110
+ Dataframe-style DSL aliases are accepted and normalize to canonical ops: `mutate` -> `derive`, `summarise`/`summarize` -> `group-agg`, `distinct` -> `unique`, and `arrange` -> `sort`.
111
+
83
112
  ### Running pipelines
84
113
 
85
114
  ```js
@@ -98,20 +127,158 @@ result.statsText // string
98
127
  result.samples // Buffer (sample channel)
99
128
  ```
100
129
 
130
+ For large outputs, drain chunks instead of collecting `result.output`:
131
+
132
+ ```js
133
+ const { createReadStream, createWriteStream } = require("fs")
134
+
135
+ // Write to a Node Writable and wait for `drain` when the sink applies backpressure.
136
+ const out = createWriteStream("out.csv")
137
+ const result = await p.writeTo(out, { inputFile: "data.csv" })
138
+
139
+ // Or consume Tranfi output as a Node Readable.
140
+ for await (const chunk of p.toReadable({ inputFile: "data.csv" })) {
141
+ // process each output chunk
142
+ }
143
+
144
+ // Existing input streams can feed the native pipeline without readFile() materialization.
145
+ const result2 = await p.run({ inputStream: createReadStream("data.csv") })
146
+
147
+ // .gz input files are decompressed as streaming sources by default.
148
+ const gz = await p.run({ inputFile: "events.jsonl.gz" })
149
+ const raw = await p.run({ inputFile: "events.jsonl.gz", compression: "none" })
150
+ const streamGz = await p.run({ inputStream: createReadStream("events.jsonl.gz"), compression: "gzip" })
151
+
152
+ // Multiple files stream sequentially through one pipeline.
153
+ const inputFiles = ["part-a.csv", "part-b.csv"]
154
+ const combined = await p.run({ inputFiles, sourceColumn: "src" })
155
+
156
+ // Low-level async iteration is still available.
157
+ for await (const chunk of p.iterChunks({ inputFiles, sourceColumn: "src" })) {
158
+ // process each output chunk
159
+ }
160
+ ```
161
+
162
+ `sourceColumn` appends the path for each input row without preloading file contents. Tranfi flushes decoder input at each file boundary so an unterminated final record belongs to the correct source file. It does not remove repeated CSV headers from later files; use shards without repeated headers, `header: false`, or a pre-cleaning step when every file has its own header.
163
+
164
+ Standalone WASM exposes the same non-collecting shape for in-memory data: `tf.run(dsl, data, { onOutput, collectOutput: false })` and `tf.iterChunks(dsl, data)`.
165
+
166
+ For browser UI work, keep Worker placement outside the core pipeline and use the optional WASM Worker adapter:
167
+
168
+ ```js
169
+ // main thread
170
+ import { createWorkerClient } from 'tranfi/wasm/worker'
171
+
172
+ const client = createWorkerClient(new Worker(new URL('./tranfi-worker.js', import.meta.url), { type: 'module' }))
173
+ const result = await client.runFile('csv | filter "col(age) >= 18" | csv', file, {
174
+ chunkSize: 64 * 1024,
175
+ collectOutput: false,
176
+ onOutput: chunk => downloadSink.write(chunk),
177
+ onProgress: p => updateProgress(p),
178
+ signal: abortController.signal
179
+ })
180
+
181
+ // tranfi-worker.js, bundled by the app
182
+ import { runWorkerServer } from 'tranfi/wasm/worker'
183
+ runWorkerServer()
184
+ ```
185
+
186
+ The Worker protocol streams chunks with transferable buffers and sends progress, stats, errors, and cancellation messages around the same WASM `create/push/pull/finish/free` API; it is not a separate IR target.
187
+
188
+ The bundled Tranfi app runner is also preview-bounded by default: file chunks are streamed into WASM, main output is drained incrementally into a table preview capped by hidden `preview_rows` (default 200), and full output text is only materialized when `collect_output` is explicitly true in the schema.
189
+
190
+ ### Prepared reusable transforms
191
+
192
+ WASM cancellation tokens created with `createTransformCancelToken()` expose a read-only `requested` boolean for host-side conversion loops. Reading a closed token fails.
193
+
194
+ Prepared transforms are a separate typed-table API for operations whose parameters must be learned from reference data. `analyze` accumulates bounded statistics over one or more batches, `finalize` freezes an immutable plan and output schema, and `apply` runs that plan either over a second pass of the original data (`fit_transform`-style) or over later compatible batches. It does not replace the byte-stream pipeline API.
195
+
196
+ The current slice accepts declared `float32`/`float64` columns. It supports numeric none/zero/constant/mean/exact-median imputation and none/standard/min-max normalization. Declared categorical columns support mode imputation with `allMissing: 'error' | 'zero'` or `impute.op: 'none'`; encoders discover finite typed categories, while the no-imputation/no-encoding combination needs no learned dictionary. A column may instead provide both branches with `kind.op: 'infer'`, `kind.rule: 'finite-integer-cardinality-v1'`, and `kind.maxCategories >= 2`: missing/NaN values are ignored, `2..maxCategories` distinct finite integers resolve categorical, while zero/one distinct value, any noninteger, or the next distinct value resolves numeric. `impute.op: 'none'` with `encode.op: 'none'` passes every finite value through and emits canonical qNaN for missing input. Mode with `encode.op: 'none'` retains its learned dictionary and rejects unseen finite values with code `108`. `encode.op: 'label'` freezes zero-based sorted ordinals and supports `unknown: 'error' | 'sentinel' | 'other'`; `encode.op: 'onehot'` emits source-ordered category blocks and supports `unknown: 'error' | 'all_zero' | 'other'`. Without categorical imputation, missing label/one-hot input follows that encoder's unknown policy. The label sentinel must be a safe integer outside the learned ordinal range; label/one-hot `other` appends the reserved ordinal/field after known categories. Generated output IDs/names and one-hot category metadata are deterministic, and collisions fail with code `102`. Exact median obeys the configured allocation and resident-state limits; inference, categorical discovery, and output expansion obey category, output-width, allocation, and resident-state limits. Limit failures use resource code `104`. Prepared-transform host-policy and spill fields are reserved but not implemented; nonempty use fails with unsupported-runtime code `113` instead of being silently ignored.
197
+
198
+ Declared categorical columns also accept a fixed, nonempty `encode.categories` array of finite, sorted, unique tags (`{ "t": "f64", "v": "4000000000000000" }` represents 2). Tags must match the input dtype; negative zero is represented as positive zero. Fixed dictionaries retain unobserved categories. Unknown training values follow the encoder policy and never vote for mode; an all-missing zero fallback must exist in the dictionary. Fixed encoding without imputation can finalize without training rows. Fixed dictionaries with kind inference, string categories, and categorical constant imputation remain unsupported.
199
+
200
+ Prepared runtime option objects reject unknown fields. Native Node accepts
201
+ `limits`, `cancelFlag`, `hostPolicy`, and `spillDir`; standalone WASM accepts
202
+ `limits`, `cancelToken`, `hostPolicy`, and `spillDir`. The host/spill fields are
203
+ reserved as described above, and cancellation spellings are not interchangeable.
204
+
205
+ ```js
206
+ const tf = require('tranfi')
207
+
208
+ const recipeSpec = {
209
+ format: 'tranfi.transform-recipe',
210
+ version: 1,
211
+ policyVersion: 1,
212
+ outputDtype: 'float64',
213
+ semanticLimits: {
214
+ maxOutputColumns: 65536,
215
+ maxOutputElementsPerApply: 134217728
216
+ },
217
+ columns: [{
218
+ sourceId: 'x0',
219
+ kind: { op: 'declared', value: 'numeric', rule: null, maxCategories: null },
220
+ numeric: {
221
+ impute: { op: 'mean', constant: null, allMissing: 'zero' },
222
+ normalize: { op: 'standard', ddof: 0 }
223
+ },
224
+ categorical: null
225
+ }]
226
+ }
227
+ const schema = [{ id: 'x0', dtype: 'float64' }]
228
+
229
+ const recipe = tf.TransformRecipe.fromJSON(recipeSpec)
230
+ const analyzer = recipe.analyzer(schema)
231
+ analyzer.push({ rows: 3, columns: [new Float64Array([1, NaN, 3])] })
232
+ const plan = analyzer.finalize()
233
+
234
+ const apply = plan.apply(schema)
235
+ const fittedReference = apply.run({
236
+ rows: 3,
237
+ columns: [new Float64Array([1, NaN, 3])]
238
+ })
239
+ const planBytes = plan.toBytes() // canonical TFTR artifact
240
+ const recipeSha256 = plan.recipeSha256() // canonical recipe + input-schema identity
241
+
242
+ apply.close()
243
+ plan.close()
244
+ analyzer.close()
245
+ recipe.close()
246
+ ```
247
+
248
+ `tranfi/wasm` exposes the same classes on the initialized module. The Worker adapter adds `analyzeTransform()` and `applyTransform()`. With `SharedArrayBuffer`, an `AbortSignal` interrupts a synchronous C call through an atomic poll cell. Without it, cancellation terminates the whole worker and reclaims its WASM heap. Pass a worker URL directly so the client can recreate it, or supply an owned worker plus `workerFactory`:
249
+
250
+ ```js
251
+ const workerUrl = new URL('./tranfi-worker.js', import.meta.url)
252
+ const makeWorker = () => new Worker(workerUrl, { type: 'module' })
253
+ const client = createWorkerClient(makeWorker(), {
254
+ workerFactory: makeWorker,
255
+ terminateOnDispose: true
256
+ })
257
+ ```
258
+
259
+ All prepared-transform failures use `TranfiTransformError`; its numeric `code` is stable across native Node and WASM. Native Node accepts a SharedArrayBuffer-backed `Int32Array` as `cancelFlag`: cell 0 is the cancellation request, and an optional cell 1 is incremented modulo 2^32 at every native poll so another realm can observe operation progress without a timer.
260
+
101
261
  ## Codecs
102
262
 
103
263
  Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
104
264
 
105
265
  | Method | Description |
106
266
  |--------|-------------|
107
- | `codec.csv({ delimiter, header, batchSize, repair })` | CSV decoder. `repair: true` pads short / truncates long rows |
267
+ | `codec.csv({ delimiter, header, batchSize, repair, mode, strict, maxErrorBytes, maxRecordBytes, maxColumns, nulls, quotedNulls, skip, nMax, maxRows, comment, trimWs, skipEmptyRows, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | CSV decoder. `mode: 'strict'` fails on field-count mismatches; `repair: true` / `mode: 'repair'` emits repair diagnostics; `maxRecordBytes` bounds buffered records; `nulls: ['NA']` adds null sentinels; `skip: 2`, `nMax: 100`, `comment: '#'`, `trimWs: false`, and `skipEmptyRows: true` control row-local parsing; repair audit/raw diagnostics support privacy controls with pseudo-column `raw` |
108
268
  | `codec.csvEncode({ delimiter })` | CSV encoder |
109
- | `codec.jsonl({ batchSize })` | JSON Lines decoder |
269
+ | `codec.jsonl({ batchSize, onError, maxErrorBytes, maxRecordBytes })` | JSON Lines decoder. `onError` is `skip`, `fail`, `warn`, or `quarantine` |
110
270
  | `codec.jsonlEncode()` | JSON Lines encoder |
111
- | `codec.text({ batchSize })` | Line-oriented text decoder (single `_line` column) |
271
+ | `codec.text({ batchSize, maxErrorBytes, maxRecordBytes })` | Line-oriented text decoder (single `_line` column) |
112
272
  | `codec.textEncode()` | Text encoder |
113
273
  | `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
114
274
 
275
+ CSV treats unquoted empty fields as null by default. Pass `nulls: ['NA', 'NULL']` to add sentinel strings; pass `quotedNulls: false` when quoted sentinels like `"NA"` or `""` should remain strings. Default field-count handling is permissive for compatibility. Use `mode: 'strict'` or `strict: true` to fail on row/header width mismatches. Use `repair: true` or `mode: 'repair'` to pad/truncate and collect JSONL diagnostics in `result.errors`; `maxErrorBytes` bounds raw previews, and `auditIncludeRow`, `auditColumns`, `auditRedact`, `auditHashColumns`, `auditMaxBytes`, and `auditMaxCellBytes` govern repair audit/raw payloads with the raw preview exposed as pseudo-column `raw`. `maxRecordBytes` defaults to `67108864` bytes, caps the current record buffer before a newline is seen, and rejects with a `csv_record_too_large` diagnostic when exceeded; pass `0` to disable the guard. `maxColumns` defaults to `8192`; records above the cap reject with a bounded `csv_too_many_columns` diagnostic instead of silently dropping columns. Decoder size options are checked before execution: `batchSize` must be `1..65536`, `maxErrorBytes` must be `0..67108864`, `maxRecordBytes` must be `0..1073741824`, and `maxColumns` must be `1..65536`. `skip: 2` discards preamble records before header/schema discovery; comments are applied after skipped rows. `nMax: 100` / `maxRows: 100` keeps at most that many decoded data rows after skip/comment/header handling and uses only one counter; `nMax: 0` preserves a header-only schema batch. `comment: '#'` removes text after an unquoted marker and skips comment-only rows; quoted markers are preserved. Unquoted spaces/tabs are trimmed by default; pass `trimWs: false` to preserve them. Blank physical rows after the header are preserved as all-null rows by default; pass `skipEmptyRows: true` to drop them.
276
+
277
+ Malformed JSONL records are skipped by default. Set `onError: 'warn'` or `onError: 'quarantine'` to keep valid rows and collect JSONL diagnostics in `result.errors`; set `onError: 'fail'` to reject on the first malformed line. `maxErrorBytes` bounds the raw preview stored in diagnostics, and `maxRecordBytes` applies the same current-line guard as CSV/text.
278
+
279
+ Each JSONL record must be one complete UTF-8 JSON object. Invalid number syntax, trailing content, embedded NULs, and numbers outside the finite float64 range follow the same malformed-record policy. The encoder rejects nonfinite values. Schema order comes from the first valid record; repeated keys use their first value.
280
+
281
+
115
282
  Cross-codec pipelines work naturally:
116
283
 
117
284
  ```js
@@ -128,30 +295,46 @@ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
128
295
 
129
296
  | Method | Description |
130
297
  |--------|-------------|
131
- | `ops.filter(expr)` | Keep rows matching expression |
298
+ | `ops.filter(expr, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Keep rows matching expression; optional dropped-row audit records support row omission, column allowlists, redaction, hashes, and payload caps |
132
299
  | `ops.head(n)` | First N rows |
133
300
  | `ops.tail(n)` | Last N rows |
134
301
  | `ops.skip(n)` | Skip first N rows |
135
302
  | `ops.top(n, column, desc?)` | Top N by column value |
136
- | `ops.sample(n)` | Reservoir sampling (uniform random) |
303
+ | `ops.sample(n, { seed })` | Deterministic bounded reservoir sampling; use `seed: 'random'` for nondeterministic mode |
137
304
  | `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
138
- | `ops.validate(expr)` | Add `_valid` boolean column, keep all rows |
305
+ | `ops.validate(expr, { rules, rulesFile, audit, auditLimit, maxFailures, warnFailureRate, maxFailureRate, name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Add `_valid` boolean column, keep all rows; supports inline rules or local JSON rule-suite files; bounded failure audit records support count/rate thresholds plus privacy controls |
306
+ | `ops.assert(expr, { action, name, message, result, aggregate, op, value, column, tolerance, rel, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Row-local data-quality rule or finish-time O(1) aggregate assertion; row-local failure side-channel records support privacy controls |
307
+ | `ops.quarantine(expr, { name, message, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Route rows matching expression to `errors` and drop them from main output; row payloads support privacy controls |
308
+ | `ops.schema({ columns, required, nonNull, nullable, values, min, max, regex, mode='fail', result='_schema', maxRegexPatternBytes, maxRegexCellBytes, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Row-local table schema contract; fail, warn, filter, quarantine, or annotate; regex budgets default to 4096-byte patterns and 65536-byte cells; schema audit/error records support row omission, column allowlists, redaction, stable non-cryptographic hashes, and row/cell payload caps |
309
+ | `ops.schemaInfer({ rows })` | Bounded decoded-type/nullability schema report; defaults to 10000 sampled rows |
310
+ | `ops.tee({ expr, channel, columns, limit, every, name, includeRow, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Preserve main rows and write bounded JSONL row snapshots to a side channel; row payloads support privacy controls |
139
311
 
140
312
  ### Column operations
141
313
 
142
314
  | Method | Description |
143
315
  |--------|-------------|
144
316
  | `ops.select(columns)` | Keep and reorder columns |
317
+ | `ops.relocate(columns, { before, after })` | Move columns while preserving all columns |
145
318
  | `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
146
319
  | `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
147
- | `ops.cast(mapping)` | Type conversion: `cast({ age: 'int', score: 'float' })` |
320
+ | `ops.sourceName({ result, defaultValue })` | Append the current host source path/name as a row-local string column |
321
+ | `ops.across(columns, { fn, functions, names, replace })` | Apply row-local functions over selected columns |
322
+ | `ops.cast(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Type conversion; optional bounded value/coercion audit records support privacy controls |
148
323
  | `ops.trim(columns?)` | Strip whitespace |
149
- | `ops.fillNull(mapping)` | Replace nulls: `fillNull({ age: '0' })` |
324
+ | `ops.fillNull(mapping, { audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Replace nulls; optional bounded audit records support privacy controls |
150
325
  | `ops.fillDown(columns?)` | Forward-fill nulls |
151
326
  | `ops.clip(column, { min, max })` | Clamp numeric values |
152
- | `ops.replace(column, pattern, replacement, { regex })` | String find/replace |
327
+ | `ops.replace(column, pattern, replacement, { regex, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | String find/replace; optional bounded audit records support privacy controls |
153
328
  | `ops.hash(columns?)` | Add `_hash` column (DJB2) |
154
- | `ops.bin(column, boundaries)` | Discretize into bins |
329
+ | `ops.bin(column, boundaries, { missing, onTypeError })` | Discretize into bins with strict numeric-source defaults |
330
+ | `ops.ewma(column, alpha, { result, missing, onTypeError })` | Exponentially weighted moving average |
331
+ | `ops.anomaly(column, { threshold, result, missing, onTypeError })` | Streaming z-score anomaly flag |
332
+ | `ops.normalize(columns, { method, audit, auditLimit, missing, onTypeError, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Blocking minmax/zscore normalization; optional audit records support privacy controls |
333
+ | `ops.acf(column, { lags, missing, onTypeError })` | Blocking autocorrelation table |
334
+
335
+ `columns` may contain exact names or selector strings: `id:score`, `starts_with(score_)`, `ends_with(_id)`, `contains(temp)`, `matches(^score_)`, `where(numeric)`, strict `all_of(score,name)`, lenient `any_of(optional,score)`, exclusions with `!name` or `-name`, and boolean selector algebra such as `starts_with(score_)&where(numeric)`, `starts_with(score_)&!ends_with(raw)`, or `!(id:score)`. Ranges use input schema order and can be reversed. Helper matching is case-insensitive; exact names are case-sensitive. These selectors are resolved by the native schema-aware `select`, `relocate`, and `across` ops in `O(columns)` without retaining rows; SQL lowering rejects selector helpers without a known schema.
336
+
337
+ Example: `ops.across(['starts_with(score_)'], { fn: 'round' })` replaces selected numeric columns per row; `ops.across(['name'], { functions: ['lower'], replace: false, names: '{col}_{fn}' })` appends templated columns. This is native/WASM row-local execution, not arbitrary lambdas or grouped dplyr evaluation.
155
338
 
156
339
  ### Sorting and deduplication
157
340
 
@@ -165,14 +348,15 @@ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
165
348
  | Method | Description |
166
349
  |--------|-------------|
167
350
  | `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
168
- | `ops.frequency(columns?)` | Value counts (descending) |
169
- | `ops.groupAgg(groupBy, aggs)` | Group by + aggregate |
351
+ | `ops.frequency(columns?, { maxValues, maxStateBytes, overflow, other, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes }?)` | Value counts; `overflow: "other"` can emit bounded category-overflow audit records with privacy controls |
352
+ | `ops.groupAgg(groupBy, aggs)` | Group by + aggregate. `count` on a column counts non-null values; `column: '*'` counts rows. |
170
353
 
171
354
  ```js
172
355
  // Group aggregation
173
356
  ops.groupAgg(['city'], [
174
- { column: 'price', func: 'sum', result: 'total' },
175
- { column: 'price', func: 'avg', result: 'avg_price' },
357
+ { column: 'price', func: 'sum', name: 'total' },
358
+ { column: 'price', func: 'avg', name: 'avg_price' },
359
+ { column: '*', func: 'count', name: 'rows' },
176
360
  ])
177
361
  ```
178
362
 
@@ -181,36 +365,49 @@ ops.groupAgg(['city'], [
181
365
  | Method | Description |
182
366
  |--------|-------------|
183
367
  | `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
184
- | `ops.window(column, size, func, result?)` | Sliding window: `avg`, `sum`, `min`, `max` |
368
+ | `ops.window(column, size, func, resultOrOptions?)` | Sliding window: `avg`, `sum`, `min`, `max`; options include `result`, `missing`, `onTypeError` |
369
+ | `ops.rollingSum/rollingMean/rollingMin/rollingMax(column, size, { result, missing, onTypeError })` | Named trailing fixed-row numeric windows; default missing/non-numeric source fails |
370
+ | `ops.rollingAny/rollingAll(column, size, { result, nulls })` | Boolean trailing windows; `nulls`: `ignore`, `false`, `true`, `propagate` |
185
371
  | `ops.lead(column, { offset, result })` | Lookahead N rows |
372
+ | `ops.lag(column, { offset, result })` | Previous-row shift |
373
+ | `ops.shift(column, { offset, result, type })` | Shift alias; `type='lag'` default or `type='lead'` |
374
+ | `ops.rowid(columns, { result, sorted, maxKeys })` | Global or per-key 1-based row ids |
375
+ | `ops.rleid(columns, { result })` | Consecutive run id by selected column(s) |
186
376
 
187
377
  ### Reshape
188
378
 
189
379
  | Method | Description |
190
380
  |--------|-------------|
191
- | `ops.explode(column, delimiter?)` | Split delimited string into rows |
381
+ | `ops.explode(column, delimiter?, { maxTokensPerRow, maxOutputRowsPerInputRow, maxOutputRowsPerBatch, maxTokenBytes })` | Split delimited string into rows with optional expansion caps |
192
382
  | `ops.split(column, names, delimiter?)` | Split column into multiple columns |
193
- | `ops.unpivot(columns)` | Wide to long (melt) |
383
+ | `ops.unpivot(columns, { maxOutputRowsPerInputRow, maxOutputRowsPerBatch })` | Wide to long (melt) with optional expansion caps |
194
384
  | `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
195
385
 
386
+ `explode` caps fail fast with `maxTokensPerRow`, `maxOutputRowsPerInputRow`, `maxOutputRowsPerBatch`, or `maxTokenBytes` when a single row or batch would expand beyond the configured limit. `unpivot` supports `maxOutputRowsPerInputRow` and `maxOutputRowsPerBatch`.
387
+
196
388
  ### Date/time
197
389
 
198
390
  | Method | Description |
199
391
  |--------|-------------|
200
- | `ops.datetime(column, extract?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday` |
201
- | `ops.dateTrunc(column, trunc, { result })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
392
+ | `ops.datetime(column, extractOrOptions?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday`; accepts `{ extract, missing, onTypeError }` |
393
+ | `ops.dateTrunc(column, trunc, { result, missing, onTypeError })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
202
394
 
203
395
  ### Other
204
396
 
205
397
  | Method | Description |
206
398
  |--------|-------------|
207
399
  | `ops.flatten()` | Flatten nested columns |
400
+ | `ops.interpolate(column, { method, missing, onTypeError })` | Fill nulls in a numeric column; default missing/non-numeric source fails |
401
+ | `ops.jsonExtract(path, result, { column, type })` | Extract JSON Pointer/simple JSONPath value into a new column |
402
+ | `ops.jsonFilter(path, { op, value, column, type })` | Filter rows by JSON Pointer/simple JSONPath predicate |
403
+ | `ops.jsonSchema(schema, { column, mode, result, audit, auditLimit, auditIncludeRow, auditColumns, auditRedact, auditHashColumns, auditMaxBytes, auditMaxCellBytes })` | Validate JSON text with a supported JSON Schema subset; filter-mode audit records support privacy controls |
404
+ | `ops.jsonFlatten(fields, { column })` | Append declared JSON Pointer/simple JSONPath fields as bounded output columns |
208
405
  | `ops.reorder(columns)` | Alias for `select` |
209
406
  | `ops.dedup(columns?)` | Alias for `unique` |
210
407
 
211
408
  ## Expressions
212
409
 
213
- Used in `filter`, `derive`, and `validate`. Reference columns with `col('name')`.
410
+ Used in `filter`, `derive`, `validate`, and `assert`. Reference columns with `col('name')`.
214
411
 
215
412
  ```js
216
413
  ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
@@ -228,11 +425,19 @@ ops.derive({
228
425
  | Comparison | `>` `>=` `<` `<=` `==` `!=` |
229
426
  | Logic | `and` `or` `not` |
230
427
  | String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
231
- | Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` |
232
- | Conditional | `if(cond,then,else)` `coalesce(a,b,...)` `nullif(a,b)` |
428
+ | Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` `between(x,left,right)` `inrange(x,left,right)` |
429
+ | Date/time | `year(x)` `month(x)` `day(x)` `hour(x)` `minute(x)` `second(x)` `weekday(x)` `epoch(x)` `date_trunc(x,unit)` |
430
+ | Conditional | `if(cond,then,else)` `case_when(cond,value,...,default)` `case_match(value,key,result,...,default)` `if_any(pred,...)` `if_all(pred,...)` `coalesce(a,b,...)` `nullif(a,b)` |
233
431
  | Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
234
432
 
235
- Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`.
433
+ Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`. Date/time functions are row-local and accept date/timestamp values plus parseable date/timestamp strings; `weekday()` returns `0=Sunday` through `6=Saturday`.
434
+
435
+ ```js
436
+ ops.derive({
437
+ year: expr("year(col('date'))"),
438
+ monthStart: expr("date_trunc(col('date'), 'month')"),
439
+ })
440
+ ```
236
441
 
237
442
  ## Recipes
238
443
 
@@ -248,6 +453,7 @@ const result = await pipeline('freq').run({ inputFile: 'data.csv' })
248
453
  | `profile` | `csv \| stats \| csv` | Full data profiling |
249
454
  | `preview` | `csv \| head 10 \| csv` | First 10 rows |
250
455
  | `schema` | `csv \| head 0 \| csv` | Column names only |
456
+ | `sniff` | `csv \| schema infer rows=1000 \| csv` | Bounded-memory schema/type/nullability sniff |
251
457
  | `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
252
458
  | `count` | `csv \| stats count \| csv` | Row count |
253
459
  | `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
@@ -277,6 +483,25 @@ for (const r of await recipes()) {
277
483
  }
278
484
  ```
279
485
 
486
+
487
+ ## Memory policy
488
+
489
+ Native Node/WASM execution rejects full-input blocking steps such as `sort`, `pivot`, `normalize`, `acf`, and table encoding unless you opt in for known-small data:
490
+
491
+ ```js
492
+ await pipeline('csv | sort age | csv').run({ inputFile: 'small.csv', allowBlocking: true })
493
+ ```
494
+
495
+ For capped key-state operators, pass `memory` to validate the conservative native state estimate before execution:
496
+
497
+ ```js
498
+ await pipeline('csv | unique city max_keys=10000 | csv').run({ inputFile: 'data.csv', memory: '64MB' })
499
+ ```
500
+
501
+ Native Node execution supports spill-backed `sort`, capped unsorted `pivot`, unsorted `unique`/`dedup`, unsorted `group-agg`, capped unsorted `join` inner/left, unsorted `semi-join`/`anti-join`, unsorted set operations, and duplicate-eliminating `union` when `spillDir` is provided. `run()`, `iterChunks()`, `toReadable()`, and `writeTo()` drain finish-time merge output through N-API `finishStep()` instead of waiting for one whole `finish()`. Standalone WASM supports `allowBlocking` and `memory`, but rejects `spillDir` because browser/WASM spill storage would occupy WASM memory. Use the Node native addon, CLI/direct C, or `{ engine: 'duckdb' }` for external spill; with DuckDB, `memory` maps to `memory_limit` and `spillDir` maps to `temp_directory`.
502
+
503
+ Plan-internal file reads are denied by default in Node and WASM. Pass `allowFs: true` for trusted local lookup files used by `join`, set operations, `union`, or `stack`; pass both `allowFs: true` and `allowRulesFile: true` for `validate rules_file=...`; pass `workspaceRoot` to pin resolved core plan paths inside a trusted directory. Supplying `spillDir` opts into local filesystem spill for Node native execution; standalone WASM still rejects spill. `inputFile` and `inputFiles` are host source adapters and are not controlled by `allowFs`.
504
+
280
505
  ## DuckDB engine
281
506
 
282
507
  Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
@@ -302,7 +527,10 @@ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
302
527
  Generate SQL directly from DSL strings:
303
528
 
304
529
  ```js
305
- const sql = await compileToSql('csv | filter "col(age) > 25" | sort -age | head 10 | csv')
530
+ const sql = await compileToSql(
531
+ 'csv | filter "col(age) > 25" | sort -age | head 10 | csv',
532
+ { dialect: 'duckdb' }
533
+ )
306
534
  console.log(sql)
307
535
  // WITH
308
536
  // step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
@@ -310,6 +538,10 @@ console.log(sql)
310
538
  // SELECT * FROM step_2
311
539
  ```
312
540
 
541
+ DuckDB is the only implemented SQL dialect today. `dialect: 'sqlite'` and
542
+ `dialect: 'postgres'` are recognized but rejected until those dialects have
543
+ their own compatibility tests and SQL-generation rules.
544
+
313
545
  ### Browser (WASM + DuckDB-WASM)
314
546
 
315
547
  In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
@@ -321,7 +553,7 @@ import * as duckdb from '@duckdb/duckdb-wasm'
321
553
  const tf = await createTranfi()
322
554
 
323
555
  // SQL generation (synchronous, no DuckDB needed)
324
- const sql = tf.compileToSql('csv | filter "age > 25" | csv')
556
+ const sql = tf.compileToSql('csv | filter "age > 25" | csv', { dialect: 'duckdb' })
325
557
 
326
558
  // Full execution with DuckDB-WASM
327
559
  const db = new duckdb.AsyncDuckDB(...)
@@ -356,12 +588,12 @@ Every pipeline produces four output channels:
356
588
 
357
589
  - **output** -- main pipeline result
358
590
  - **errors** -- rows that failed processing
359
- - **stats** -- pipeline execution statistics (rows in/out, timing)
591
+ - **stats** -- newline-delimited execution statistics: run summary plus per-step counters, state estimates, and warnings
360
592
  - **samples** -- reserved for sampling operators
361
593
 
362
594
  ```js
363
595
  const result = await p.run({ inputFile: 'data.csv' })
364
- console.log(result.statsText) // {"rows_in": 1000, "rows_out": 42, ...}
596
+ console.log(result.statsText) // newline-delimited JSON: summary plus step_stats/state_bytes_estimate/warnings
365
597
  ```
366
598
 
367
599
  ### Pipeline from JSON
@@ -381,7 +613,7 @@ const p = await loadRecipe({
381
613
  The package automatically selects the best backend:
382
614
 
383
615
  1. **N-API** (Node.js) -- native C addon, fastest, used when available
384
- 2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~313 KB single-file
616
+ 2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~500 KB single-file
385
617
  3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
386
618
 
387
619
  ```js
@@ -392,4 +624,4 @@ const tf = await createTranfi()
392
624
 
393
625
  ## Architecture
394
626
 
395
- The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. All operators are streaming with bounded memory, except those that require full input (sort, unique, stats, tail, top, group-agg, frequency, pivot).
627
+ The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. `run({ inputFile })` streams files with `createReadStream()` and drains native/WASM main output after each push and each incremental finish boundary; by default it still collects the final output for convenience. `.gz` input files are decompressed through a `zlib.createGunzip()` source transform; use `compression: "none"` to force raw bytes or `compression: "gzip"` to force gzip for `inputFile`/`inputStream`. Use `inputStream`, `toReadable()`, `writeTo()`, `onOutput` with `collectOutput: false`, or `iterChunks()` for large inputs/outputs and backpressure-aware sinks. Native execution is strict by default: row-local and bounded-state operators stream, blocking operators require `allowBlocking: true` or supported `spillDir`, capped key-state plans can be checked with `memory: "64MB"`, and core plan file reads require explicit host-policy options.