tranfi 0.0.1 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/README.md +395 -0
  2. package/app/assets/index-6quYZ5Ap.css +5 -0
  3. package/app/assets/index-pDFMluyz.js +160 -0
  4. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  5. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  6. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  7. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  8. package/app/index.html +13 -0
  9. package/binding.gyp +69 -0
  10. package/csrc/arena.c +91 -0
  11. package/csrc/batch.c +229 -0
  12. package/csrc/buffer.c +78 -0
  13. package/csrc/cJSON.c +3143 -0
  14. package/csrc/cJSON.h +300 -0
  15. package/csrc/codec_csv.c +1058 -0
  16. package/csrc/codec_jsonl.c +374 -0
  17. package/csrc/codec_table.c +218 -0
  18. package/csrc/codec_text.c +229 -0
  19. package/csrc/compiler.c +102 -0
  20. package/csrc/date_utils.h +94 -0
  21. package/csrc/dsl.c +1180 -0
  22. package/csrc/dsl.h +22 -0
  23. package/csrc/expr.c +1245 -0
  24. package/csrc/expr.h +56 -0
  25. package/csrc/internal.h +250 -0
  26. package/csrc/ir.c +119 -0
  27. package/csrc/ir.h +167 -0
  28. package/csrc/ir_schema.c +60 -0
  29. package/csrc/ir_serialize.c +104 -0
  30. package/csrc/ir_sql.c +1211 -0
  31. package/csrc/ir_validate.c +120 -0
  32. package/csrc/main.c +392 -0
  33. package/csrc/op_acf.c +133 -0
  34. package/csrc/op_anomaly.c +120 -0
  35. package/csrc/op_bin.c +109 -0
  36. package/csrc/op_cast.c +195 -0
  37. package/csrc/op_clip.c +88 -0
  38. package/csrc/op_date_trunc.c +181 -0
  39. package/csrc/op_datetime.c +212 -0
  40. package/csrc/op_derive.c +248 -0
  41. package/csrc/op_diff.c +134 -0
  42. package/csrc/op_ewma.c +103 -0
  43. package/csrc/op_explode.c +108 -0
  44. package/csrc/op_fill_down.c +163 -0
  45. package/csrc/op_fill_null.c +123 -0
  46. package/csrc/op_filter.c +132 -0
  47. package/csrc/op_frequency.c +193 -0
  48. package/csrc/op_grep.c +163 -0
  49. package/csrc/op_group_agg.c +285 -0
  50. package/csrc/op_hash.c +126 -0
  51. package/csrc/op_head.c +149 -0
  52. package/csrc/op_interpolate.c +239 -0
  53. package/csrc/op_join.c +384 -0
  54. package/csrc/op_label_encode.c +144 -0
  55. package/csrc/op_lead.c +190 -0
  56. package/csrc/op_normalize.c +226 -0
  57. package/csrc/op_onehot.c +185 -0
  58. package/csrc/op_pivot.c +370 -0
  59. package/csrc/op_registry.c +1148 -0
  60. package/csrc/op_rename.c +138 -0
  61. package/csrc/op_replace.c +202 -0
  62. package/csrc/op_sample.c +101 -0
  63. package/csrc/op_select.c +140 -0
  64. package/csrc/op_skip.c +152 -0
  65. package/csrc/op_sort.c +273 -0
  66. package/csrc/op_split.c +114 -0
  67. package/csrc/op_split_data.c +87 -0
  68. package/csrc/op_stack.c +315 -0
  69. package/csrc/op_stats.c +779 -0
  70. package/csrc/op_step.c +171 -0
  71. package/csrc/op_tail.c +96 -0
  72. package/csrc/op_top.c +150 -0
  73. package/csrc/op_trim.c +109 -0
  74. package/csrc/op_unique.c +300 -0
  75. package/csrc/op_unpivot.c +159 -0
  76. package/csrc/op_validate.c +71 -0
  77. package/csrc/op_window.c +150 -0
  78. package/csrc/pipeline.c +315 -0
  79. package/csrc/plan.c +206 -0
  80. package/csrc/recipes.c +102 -0
  81. package/csrc/recipes.h +27 -0
  82. package/csrc/report.c +463 -0
  83. package/csrc/report.h +22 -0
  84. package/csrc/tranfi.h +123 -0
  85. package/csrc/wasm_api.c +157 -0
  86. package/napi_api.c +326 -0
  87. package/package.json +46 -57
  88. package/src/cli.js +193 -0
  89. package/src/engines/duckdb.js +109 -0
  90. package/src/index.js +306 -0
  91. package/src/native.js +22 -0
  92. package/src/pipeline.js +286 -0
  93. package/src/server.js +277 -0
  94. package/src/wasm.js +19 -0
  95. package/wasm/index.js +244 -0
  96. package/wasm/package.json +1 -0
  97. package/wasm/tranfi_core.js +0 -0
  98. package/LICENSE +0 -21
  99. package/dist/bundle.js +0 -1
  100. package/index.html +0 -18
  101. package/logo.png +0 -0
  102. package/src/app.css +0 -160
  103. package/src/app.js +0 -203
  104. package/src/app.vue +0 -253
  105. package/src/bulma-input.vue +0 -110
  106. package/src/common-inputs.js +0 -28
  107. package/src/main.js +0 -18
  108. package/src/transforms.js +0 -166
  109. package/webpack.config.js +0 -108
package/README.md ADDED
@@ -0,0 +1,395 @@
1
+ # tranfi (Node.js / WASM)
2
+
3
+ Streaming ETL in JavaScript, powered by a native C11 core via N-API (Node.js) or WASM (browsers). Process CSV, JSONL, and text data with composable pipelines that run in constant memory, no matter how large the input.
4
+
5
+ ```js
6
+ import { pipeline, codec, ops, expr } from 'tranfi'
7
+
8
+ const result = await pipeline([
9
+ codec.csv(),
10
+ ops.filter(expr("col('age') > 25")),
11
+ ops.sort(['-age']),
12
+ ops.derive({ label: expr("if(col('age')>30, 'senior', 'junior')") }),
13
+ ops.select(['name', 'age', 'label']),
14
+ codec.csvEncode(),
15
+ ]).run({ input: 'name,age\nAlice,30\nBob,25\nCharlie,35\nDiana,28\n' })
16
+
17
+ console.log(result.outputText)
18
+ // name,age,label
19
+ // Charlie,35,senior
20
+ // Alice,30,junior
21
+ // Diana,28,junior
22
+ ```
23
+
24
+ Or use the pipe DSL for one-liners:
25
+
26
+ ```js
27
+ const result = await pipeline('csv | filter "col(age) > 25" | sort -age | csv')
28
+ .run({ inputFile: 'data.csv' })
29
+ ```
30
+
31
+ ## Install
32
+
33
+ ```bash
34
+ npm install tranfi
35
+ ```
36
+
37
+ Uses N-API natively in Node.js, falls back to WASM in browsers automatically.
38
+
39
+ ## CLI
40
+
41
+ Installing the package also provides the `tranfi` command:
42
+
43
+ ```bash
44
+ # Via npx (no install)
45
+ echo 'name,age\nAlice,30\nBob,25' | npx tranfi 'csv | filter "age > 25" | csv'
46
+
47
+ # Or install globally
48
+ npm i -g tranfi
49
+ tranfi 'csv | filter "age > 25" | sort -age | csv' < data.csv
50
+ tranfi profile < data.csv
51
+ tranfi -R # list recipes
52
+ ```
53
+
54
+ Run `tranfi -h` for all options.
55
+
56
+ ## Quick start
57
+
58
+ ### Two APIs
59
+
60
+ **Builder API** -- composable, type-safe, IDE-friendly:
61
+
62
+ ```js
63
+ const p = pipeline([
64
+ codec.csv(),
65
+ ops.filter(expr("col('score') >= 80")),
66
+ ops.derive({ grade: expr("if(col('score')>=90, 'A', 'B')") }),
67
+ ops.sort(['-score']),
68
+ ops.head(10),
69
+ codec.csvEncode(),
70
+ ])
71
+ const result = await p.run({ inputFile: 'students.csv' })
72
+ ```
73
+
74
+ **DSL strings** -- compact, suitable for CLI-like use:
75
+
76
+ ```js
77
+ const p = pipeline('csv | filter "col(score) >= 80" | sort -score | head 10 | csv')
78
+ const result = await p.run({ inputFile: 'students.csv' })
79
+ ```
80
+
81
+ Both produce identical pipelines under the hood.
82
+
83
+ ### Running pipelines
84
+
85
+ ```js
86
+ // From string or Buffer
87
+ const result = await p.run({ input: 'name,age\nAlice,30\n' })
88
+
89
+ // From file (streamed in 64 KB chunks)
90
+ const result = await p.run({ inputFile: 'data.csv' })
91
+
92
+ // Access results
93
+ result.output // Buffer
94
+ result.outputText // string (UTF-8 decoded)
95
+ result.errors // Buffer (error channel)
96
+ result.stats // Buffer (pipeline stats)
97
+ result.statsText // string
98
+ result.samples // Buffer (sample channel)
99
+ ```
100
+
101
+ ## Codecs
102
+
103
+ Codecs convert between raw bytes and columnar batches. Every pipeline starts with a decoder and ends with an encoder.
104
+
105
+ | Method | Description |
106
+ |--------|-------------|
107
+ | `codec.csv({ delimiter, header, batchSize, repair })` | CSV decoder. `repair: true` pads short / truncates long rows |
108
+ | `codec.csvEncode({ delimiter })` | CSV encoder |
109
+ | `codec.jsonl({ batchSize })` | JSON Lines decoder |
110
+ | `codec.jsonlEncode()` | JSON Lines encoder |
111
+ | `codec.text({ batchSize })` | Line-oriented text decoder (single `_line` column) |
112
+ | `codec.textEncode()` | Text encoder |
113
+ | `codec.tableEncode({ maxWidth, maxRows })` | Pretty-print Markdown table |
114
+
115
+ Cross-codec pipelines work naturally:
116
+
117
+ ```js
118
+ // CSV in, JSONL out
119
+ pipeline([codec.csv(), ops.head(5), codec.jsonlEncode()])
120
+
121
+ // JSONL in, CSV out
122
+ pipeline([codec.jsonl(), ops.sort(['name']), codec.csvEncode()])
123
+ ```
124
+
125
+ ## Operators
126
+
127
+ ### Row filtering
128
+
129
+ | Method | Description |
130
+ |--------|-------------|
131
+ | `ops.filter(expr)` | Keep rows matching expression |
132
+ | `ops.head(n)` | First N rows |
133
+ | `ops.tail(n)` | Last N rows |
134
+ | `ops.skip(n)` | Skip first N rows |
135
+ | `ops.top(n, column, desc?)` | Top N by column value |
136
+ | `ops.sample(n)` | Reservoir sampling (uniform random) |
137
+ | `ops.grep(pattern, { invert, column, regex })` | Substring/regex filter |
138
+ | `ops.validate(expr)` | Add `_valid` boolean column, keep all rows |
139
+
140
+ ### Column operations
141
+
142
+ | Method | Description |
143
+ |--------|-------------|
144
+ | `ops.select(columns)` | Keep and reorder columns |
145
+ | `ops.rename(mapping)` | Rename columns: `rename({ name: 'full_name' })` |
146
+ | `ops.derive(columns)` | Computed columns: `derive({ total: expr("col('a')*col('b')") })` |
147
+ | `ops.cast(mapping)` | Type conversion: `cast({ age: 'int', score: 'float' })` |
148
+ | `ops.trim(columns?)` | Strip whitespace |
149
+ | `ops.fillNull(mapping)` | Replace nulls: `fillNull({ age: '0' })` |
150
+ | `ops.fillDown(columns?)` | Forward-fill nulls |
151
+ | `ops.clip(column, { min, max })` | Clamp numeric values |
152
+ | `ops.replace(column, pattern, replacement, { regex })` | String find/replace |
153
+ | `ops.hash(columns?)` | Add `_hash` column (DJB2) |
154
+ | `ops.bin(column, boundaries)` | Discretize into bins |
155
+
156
+ ### Sorting and deduplication
157
+
158
+ | Method | Description |
159
+ |--------|-------------|
160
+ | `ops.sort(columns)` | Sort rows. Prefix `-` for descending: `sort(['-age', 'name'])` |
161
+ | `ops.unique(columns?)` | Deduplicate on specified columns |
162
+
163
+ ### Aggregation
164
+
165
+ | Method | Description |
166
+ |--------|-------------|
167
+ | `ops.stats(statsList?)` | Column statistics. Stats: `count`, `min`, `max`, `sum`, `avg`, `stddev`, `variance`, `median`, `p25`, `p75`, `p90`, `p99`, `distinct`, `hist`, `sample` |
168
+ | `ops.frequency(columns?)` | Value counts (descending) |
169
+ | `ops.groupAgg(groupBy, aggs)` | Group by + aggregate |
170
+
171
+ ```js
172
+ // Group aggregation
173
+ ops.groupAgg(['city'], [
174
+ { column: 'price', func: 'sum', result: 'total' },
175
+ { column: 'price', func: 'avg', result: 'avg_price' },
176
+ ])
177
+ ```
178
+
179
+ ### Sequential / window
180
+
181
+ | Method | Description |
182
+ |--------|-------------|
183
+ | `ops.step(column, func, result?)` | Running aggregation: `running-sum`, `running-avg`, `running-min`, `running-max`, `lag` |
184
+ | `ops.window(column, size, func, result?)` | Sliding window: `avg`, `sum`, `min`, `max` |
185
+ | `ops.lead(column, { offset, result })` | Lookahead N rows |
186
+
187
+ ### Reshape
188
+
189
+ | Method | Description |
190
+ |--------|-------------|
191
+ | `ops.explode(column, delimiter?)` | Split delimited string into rows |
192
+ | `ops.split(column, names, delimiter?)` | Split column into multiple columns |
193
+ | `ops.unpivot(columns)` | Wide to long (melt) |
194
+ | `ops.stack(file, { tag, tagValue })` | Vertically concatenate another CSV file |
195
+
196
+ ### Date/time
197
+
198
+ | Method | Description |
199
+ |--------|-------------|
200
+ | `ops.datetime(column, extract?)` | Extract parts: `year`, `month`, `day`, `hour`, `minute`, `second`, `weekday` |
201
+ | `ops.dateTrunc(column, trunc, { result })` | Truncate to: `year`, `month`, `day`, `hour`, `minute`, `second` |
202
+
203
+ ### Other
204
+
205
+ | Method | Description |
206
+ |--------|-------------|
207
+ | `ops.flatten()` | Flatten nested columns |
208
+ | `ops.reorder(columns)` | Alias for `select` |
209
+ | `ops.dedup(columns?)` | Alias for `unique` |
210
+
211
+ ## Expressions
212
+
213
+ Used in `filter`, `derive`, and `validate`. Reference columns with `col('name')`.
214
+
215
+ ```js
216
+ ops.filter(expr("col('age') > 25 and contains(col('name'), 'A')"))
217
+ ops.derive({
218
+ full: expr("concat(col('first'), ' ', col('last'))"),
219
+ grade: expr("if(col('score')>=90, 'A', if(col('score')>=80, 'B', 'C'))"),
220
+ })
221
+ ```
222
+
223
+ ### Available functions
224
+
225
+ | Category | Functions |
226
+ |----------|-----------|
227
+ | Arithmetic | `+` `-` `*` `/` |
228
+ | Comparison | `>` `>=` `<` `<=` `==` `!=` |
229
+ | Logic | `and` `or` `not` |
230
+ | String | `upper(s)` `lower(s)` `initcap(s)` `len(s)` `trim(s)` `left(s,n)` `right(s,n)` `concat(a,b,...)` `replace(s,old,new)` `slice(s,start,len)` `pad_left(s,w)` `pad_right(s,w)` |
231
+ | Predicates | `starts_with(s,prefix)` `ends_with(s,suffix)` `contains(s,sub)` |
232
+ | Conditional | `if(cond,then,else)` `coalesce(a,b,...)` `nullif(a,b)` |
233
+ | Math | `abs(x)` `round(x)` `floor(x)` `ceil(x)` `sign(x)` `pow(x,y)` `sqrt(x)` `log(x)` `exp(x)` `mod(a,b)` `greatest(a,b,...)` `least(a,b,...)` |
234
+
235
+ Aliases: `substr`=`slice`, `length`=`len`, `lpad`=`pad_left`, `rpad`=`pad_right`, `min`=`least`, `max`=`greatest`.
236
+
237
+ ## Recipes
238
+
239
+ Built-in named pipelines for common tasks. Use by name:
240
+
241
+ ```js
242
+ const result = await pipeline('preview').run({ inputFile: 'data.csv' })
243
+ const result = await pipeline('freq').run({ inputFile: 'data.csv' })
244
+ ```
245
+
246
+ | Recipe | Pipeline | Description |
247
+ |--------|----------|-------------|
248
+ | `profile` | `csv \| stats \| csv` | Full data profiling |
249
+ | `preview` | `csv \| head 10 \| csv` | First 10 rows |
250
+ | `schema` | `csv \| head 0 \| csv` | Column names only |
251
+ | `summary` | `csv \| stats count,min,max,avg,stddev \| csv` | Summary statistics |
252
+ | `count` | `csv \| stats count \| csv` | Row count |
253
+ | `cardinality` | `csv \| stats count,distinct \| csv` | Unique value counts |
254
+ | `distro` | `csv \| stats min,p25,median,p75,max \| csv` | Five-number summary |
255
+ | `freq` | `csv \| frequency \| csv` | Value frequency |
256
+ | `dedup` | `csv \| dedup \| csv` | Remove duplicates |
257
+ | `clean` | `csv \| trim \| csv` | Trim whitespace |
258
+ | `sample` | `csv \| sample 100 \| csv` | Random 100 rows |
259
+ | `head` | `csv \| head 20 \| csv` | First 20 rows |
260
+ | `tail` | `csv \| tail 20 \| csv` | Last 20 rows |
261
+ | `csv2json` | `csv \| jsonl` | CSV to JSONL |
262
+ | `json2csv` | `jsonl \| csv` | JSONL to CSV |
263
+ | `tsv2csv` | `csv delimiter="\t" \| csv` | TSV to CSV |
264
+ | `csv2tsv` | `csv \| csv delimiter="\t"` | CSV to TSV |
265
+ | `look` | `csv \| table` | Pretty-print table |
266
+ | `histogram` | `csv \| stats hist \| csv` | Distribution histograms |
267
+ | `hash` | `csv \| hash \| csv` | Row hash for change detection |
268
+ | `samples` | `csv \| stats sample \| csv` | Sample values per column |
269
+
270
+ List all recipes programmatically:
271
+
272
+ ```js
273
+ import { recipes } from 'tranfi'
274
+
275
+ for (const r of await recipes()) {
276
+ console.log(`${r.name.padEnd(15)} ${r.description}`)
277
+ }
278
+ ```
279
+
280
+ ## DuckDB engine
281
+
282
+ Run pipelines on DuckDB instead of the native C streaming core. The DSL is transpiled to SQL in C, then executed by DuckDB.
283
+
284
+ ```bash
285
+ npm install duckdb
286
+ ```
287
+
288
+ ```js
289
+ import { pipeline, compileToSql } from 'tranfi'
290
+
291
+ // Run a pipeline via DuckDB
292
+ const result = await pipeline('csv | filter "age > 25" | sort -age | csv', { engine: 'duckdb' })
293
+ .run({ inputFile: 'data.csv' })
294
+
295
+ // Or with string/Buffer input
296
+ const result2 = await pipeline('csv | head 10 | csv', { engine: 'duckdb' })
297
+ .run({ input: csvString })
298
+ ```
299
+
300
+ ### SQL transpilation
301
+
302
+ Generate SQL directly from DSL strings:
303
+
304
+ ```js
305
+ const sql = await compileToSql('csv | filter "col(age) > 25" | sort -age | head 10 | csv')
306
+ console.log(sql)
307
+ // WITH
308
+ // step_1 AS (SELECT * FROM input_data WHERE ("age" > 25)),
309
+ // step_2 AS (SELECT * FROM step_1 ORDER BY "age" DESC LIMIT 10)
310
+ // SELECT * FROM step_2
311
+ ```
312
+
313
+ ### Browser (WASM + DuckDB-WASM)
314
+
315
+ In the browser, use `@duckdb/duckdb-wasm` with the tranfi WASM module:
316
+
317
+ ```js
318
+ import createTranfi from 'tranfi/wasm'
319
+ import * as duckdb from '@duckdb/duckdb-wasm'
320
+
321
+ const tf = await createTranfi()
322
+
323
+ // SQL generation (synchronous, no DuckDB needed)
324
+ const sql = tf.compileToSql('csv | filter "age > 25" | csv')
325
+
326
+ // Full execution with DuckDB-WASM
327
+ const db = new duckdb.AsyncDuckDB(...)
328
+ await db.instantiate(...)
329
+
330
+ const result = await tf.runDuckDB(db, 'csv | filter "age > 25" | csv', csvData)
331
+ console.log(result.outputText) // CSV output
332
+ console.log(result.rows) // Array of row objects
333
+ ```
334
+
335
+ `runDuckDB` accepts string, `Uint8Array`, or `File` objects as data input.
336
+
337
+ ## Advanced
338
+
339
+ ### DSL compilation
340
+
341
+ ```js
342
+ import { compileDsl, saveRecipe, loadRecipe } from 'tranfi'
343
+
344
+ // Compile DSL to JSON plan
345
+ const json = await compileDsl('csv | filter "col(age) > 25" | sort -age | csv')
346
+
347
+ // Save / load recipes
348
+ await saveRecipe([codec.csv(), ops.head(10), codec.csvEncode()], 'preview.tranfi')
349
+ const p = await loadRecipe('preview.tranfi')
350
+ const result = await p.run({ inputFile: 'data.csv' })
351
+ ```
352
+
353
+ ### Side channels
354
+
355
+ Every pipeline produces four output channels:
356
+
357
+ - **output** -- main pipeline result
358
+ - **errors** -- rows that failed processing
359
+ - **stats** -- pipeline execution statistics (rows in/out, timing)
360
+ - **samples** -- reserved for sampling operators
361
+
362
+ ```js
363
+ const result = await p.run({ inputFile: 'data.csv' })
364
+ console.log(result.statsText) // {"rows_in": 1000, "rows_out": 42, ...}
365
+ ```
366
+
367
+ ### Pipeline from JSON
368
+
369
+ ```js
370
+ const p = await loadRecipe({
371
+ steps: [
372
+ { op: 'codec.csv.decode', args: {} },
373
+ { op: 'head', args: { n: 5 } },
374
+ { op: 'codec.csv.encode', args: {} },
375
+ ]
376
+ })
377
+ ```
378
+
379
+ ### Backend selection
380
+
381
+ The package automatically selects the best backend:
382
+
383
+ 1. **N-API** (Node.js) -- native C addon, fastest, used when available
384
+ 2. **WASM** (browsers/fallback) -- same C core compiled to WebAssembly, ~313 KB single-file
385
+ 3. **DuckDB** (opt-in) -- SQL execution via `{ engine: 'duckdb' }`, requires `npm install duckdb`
386
+
387
+ ```js
388
+ // Force WASM backend (e.g., for testing)
389
+ import createTranfi from 'tranfi/wasm'
390
+ const tf = await createTranfi()
391
+ ```
392
+
393
+ ## Architecture
394
+
395
+ The Node.js package wraps the same C11 core used by the CLI, Python, and WASM targets. Data flows through columnar batches with typed columns (`bool`, `int64`, `float64`, `string`, `date`, `timestamp`) and per-cell null bitmaps. All operators are streaming with bounded memory, except those that require full input (sort, unique, stats, tail, top, group-agg, frequency, pivot).