tranfi 0.0.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/LICENSE +177 -21
  2. package/NOTICE +8 -0
  3. package/README.md +627 -0
  4. package/app/assets/index-6quYZ5Ap.css +5 -0
  5. package/app/assets/index-BIAIKnrp.js +160 -0
  6. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  7. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  8. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  9. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  10. package/app/index.html +13 -0
  11. package/binding.gyp +121 -0
  12. package/csrc/arena.c +93 -0
  13. package/csrc/batch.c +976 -0
  14. package/csrc/buffer.c +154 -0
  15. package/csrc/cJSON.c +3386 -0
  16. package/csrc/cJSON.h +316 -0
  17. package/csrc/codec_csv.c +1951 -0
  18. package/csrc/codec_jsonl.c +1086 -0
  19. package/csrc/codec_table.c +248 -0
  20. package/csrc/codec_text.c +447 -0
  21. package/csrc/compiler.c +130 -0
  22. package/csrc/config.h +21 -0
  23. package/csrc/date_utils.h +94 -0
  24. package/csrc/dsl.c +5417 -0
  25. package/csrc/dsl.h +22 -0
  26. package/csrc/expr.c +1553 -0
  27. package/csrc/expr.h +58 -0
  28. package/csrc/internal.h +539 -0
  29. package/csrc/ir.c +166 -0
  30. package/csrc/ir.h +208 -0
  31. package/csrc/ir_schema.c +75 -0
  32. package/csrc/ir_serialize.c +166 -0
  33. package/csrc/ir_sql.c +1822 -0
  34. package/csrc/ir_validate.c +576 -0
  35. package/csrc/json_path.c +210 -0
  36. package/csrc/main.c +1241 -0
  37. package/csrc/memory_estimate.c +477 -0
  38. package/csrc/op_acf.c +283 -0
  39. package/csrc/op_across.c +477 -0
  40. package/csrc/op_anomaly.c +255 -0
  41. package/csrc/op_assert.c +761 -0
  42. package/csrc/op_bin.c +248 -0
  43. package/csrc/op_cast.c +523 -0
  44. package/csrc/op_clip.c +99 -0
  45. package/csrc/op_date_trunc.c +355 -0
  46. package/csrc/op_datetime.c +394 -0
  47. package/csrc/op_derive.c +216 -0
  48. package/csrc/op_diff.c +250 -0
  49. package/csrc/op_ewma.c +222 -0
  50. package/csrc/op_explode.c +206 -0
  51. package/csrc/op_fill_down.c +235 -0
  52. package/csrc/op_fill_null.c +268 -0
  53. package/csrc/op_filter.c +181 -0
  54. package/csrc/op_frequency.c +721 -0
  55. package/csrc/op_grep.c +181 -0
  56. package/csrc/op_group_agg.c +1956 -0
  57. package/csrc/op_hash.c +159 -0
  58. package/csrc/op_head.c +84 -0
  59. package/csrc/op_interpolate.c +445 -0
  60. package/csrc/op_join.c +2902 -0
  61. package/csrc/op_json_extract.c +227 -0
  62. package/csrc/op_json_filter.c +384 -0
  63. package/csrc/op_json_flatten.c +293 -0
  64. package/csrc/op_json_schema.c +503 -0
  65. package/csrc/op_label_encode.c +419 -0
  66. package/csrc/op_lag.c +181 -0
  67. package/csrc/op_lead.c +242 -0
  68. package/csrc/op_normalize.c +510 -0
  69. package/csrc/op_onehot.c +457 -0
  70. package/csrc/op_pivot.c +1754 -0
  71. package/csrc/op_quarantine.c +189 -0
  72. package/csrc/op_registry.c +3044 -0
  73. package/csrc/op_rename.c +129 -0
  74. package/csrc/op_replace.c +354 -0
  75. package/csrc/op_rleid.c +297 -0
  76. package/csrc/op_rowid.c +559 -0
  77. package/csrc/op_sample.c +158 -0
  78. package/csrc/op_schema.c +1341 -0
  79. package/csrc/op_schema_infer.c +252 -0
  80. package/csrc/op_select.c +340 -0
  81. package/csrc/op_set.c +3449 -0
  82. package/csrc/op_skip.c +95 -0
  83. package/csrc/op_sort.c +819 -0
  84. package/csrc/op_source_name.c +120 -0
  85. package/csrc/op_split.c +151 -0
  86. package/csrc/op_split_data.c +119 -0
  87. package/csrc/op_stack.c +271 -0
  88. package/csrc/op_stats.c +875 -0
  89. package/csrc/op_step.c +333 -0
  90. package/csrc/op_tail.c +105 -0
  91. package/csrc/op_tee.c +338 -0
  92. package/csrc/op_top.c +357 -0
  93. package/csrc/op_trim.c +138 -0
  94. package/csrc/op_unique.c +1343 -0
  95. package/csrc/op_unpivot.c +193 -0
  96. package/csrc/op_validate.c +648 -0
  97. package/csrc/op_window.c +591 -0
  98. package/csrc/path_policy.c +85 -0
  99. package/csrc/pipeline.c +1088 -0
  100. package/csrc/recipes.c +104 -0
  101. package/csrc/recipes.h +27 -0
  102. package/csrc/report.c +506 -0
  103. package/csrc/report.h +22 -0
  104. package/csrc/selector.c +1097 -0
  105. package/csrc/size_utils.c +348 -0
  106. package/csrc/spill.c +317 -0
  107. package/csrc/spill.h +21 -0
  108. package/csrc/tranfi.h +291 -0
  109. package/csrc/transform.h +209 -0
  110. package/csrc/transform_api.c +2237 -0
  111. package/csrc/transform_categorical.c +923 -0
  112. package/csrc/transform_internal.h +472 -0
  113. package/csrc/transform_json.c +3812 -0
  114. package/csrc/transform_numeric.c +1966 -0
  115. package/csrc/transform_sha256.c +154 -0
  116. package/csrc/transform_wasm.h +162 -0
  117. package/csrc/transform_wasm_api.c +1373 -0
  118. package/csrc/wasm_api.c +218 -0
  119. package/napi_api.c +534 -0
  120. package/napi_transform.c +1648 -0
  121. package/napi_transform.h +8 -0
  122. package/package.json +64 -59
  123. package/scripts/install-native.js +76 -0
  124. package/scripts/prepack.js +64 -0
  125. package/scripts/sync-csrc.js +23 -0
  126. package/src/cli.js +190 -0
  127. package/src/engines/duckdb.js +142 -0
  128. package/src/index.js +925 -0
  129. package/src/memory_policy.js +411 -0
  130. package/src/native.js +18 -0
  131. package/src/pipeline.js +709 -0
  132. package/src/recipe_json.js +80 -0
  133. package/src/server.js +279 -0
  134. package/src/transform.js +403 -0
  135. package/src/transform_error.js +10 -0
  136. package/src/wasm.js +21 -0
  137. package/wasm/index.js +732 -0
  138. package/wasm/package.json +1 -0
  139. package/wasm/tranfi_core.js +0 -0
  140. package/wasm/transform.js +1156 -0
  141. package/wasm/worker.js +786 -0
  142. package/dist/bundle.js +0 -1
  143. package/index.html +0 -18
  144. package/src/app.css +0 -169
  145. package/src/app.js +0 -203
  146. package/src/app.vue +0 -250
  147. package/src/bulma-input.vue +0 -110
  148. package/src/common-inputs.js +0 -28
  149. package/src/main.js +0 -20
  150. package/src/transforms.js +0 -166
  151. package/webpack.config.js +0 -108
package/csrc/main.c ADDED
@@ -0,0 +1,1241 @@
1
+ /*
2
+ * main.c — Tranfi CLI.
3
+ *
4
+ * Usage:
5
+ * tranfi 'csv | filter "col(age) > 25" | select name,age | csv' < in.csv
6
+ * tranfi -f pipeline.tf < in.csv > out.csv
7
+ * tranfi -j 'csv | head 5 | csv' # compile only, output JSON
8
+ * tranfi --target sql --dialect duckdb 'csv | head 5 | csv'
9
+ * tranfi -i input.csv -o output.csv 'csv | filter "col(age) > 25" | csv'
10
+ *
11
+ * Channels:
12
+ * stdout — main output (encoded data)
13
+ * stderr — stats and errors
14
+ */
15
+
16
+ #include "tranfi.h"
17
+ #include "internal.h"
18
+ #include "cJSON.h"
19
+ #include "ir.h"
20
+ #include "dsl.h"
21
+ #include "recipes.h"
22
+ #include "report.h"
23
+ #include <stdio.h>
24
+ #include <stdlib.h>
25
+ #include <string.h>
26
+ #include <unistd.h>
27
+ #include <ctype.h>
28
+ #include <errno.h>
29
+ #include <stdint.h>
30
+ #include <strings.h>
31
+
32
+ #define READ_BUF_SIZE (64 * 1024)
33
+ #define PULL_BUF_SIZE (64 * 1024)
34
+
35
+ static void usage(const char *prog) {
36
+ fprintf(stderr,
37
+ "Usage: %s [OPTIONS] PIPELINE\n"
38
+ " %s [OPTIONS] -f FILE\n"
39
+ "\n"
40
+ "Streaming ETL with a pipe-style DSL.\n"
41
+ "\n"
42
+ "Examples:\n"
43
+ " %s 'csv | csv' # passthrough\n"
44
+ " %s 'csv | filter \"col(age) > 25\" | csv' # filter rows\n"
45
+ " %s 'csv | select name,age | csv' # select columns\n"
46
+ " %s 'csv | rename name=full_name | csv' # rename columns\n"
47
+ " %s 'csv | head 10 | csv' # first N rows\n"
48
+ " %s 'csv | skip 5 | csv' # skip first 5 rows\n"
49
+ " %s 'csv | derive total=col(price)*col(qty) | csv' # computed columns\n"
50
+ " %s --allow-blocking 'csv | sort age | csv' # sort known-small input\n"
51
+ " %s 'csv | unique name | csv' # deduplicate\n"
52
+ " %s 'csv | stats | csv' # aggregate stats\n"
53
+ " %s 'jsonl | filter \"col(x) > 0\" | jsonl' # JSONL variant\n"
54
+ " %s --target sql --dialect duckdb 'csv | head 5 | csv' # print SQL\n"
55
+ "\n"
56
+ "Options:\n"
57
+ " -f FILE Read pipeline from file instead of argument\n"
58
+ " -i FILE Read input from file instead of stdin\n"
59
+ " -o FILE Write output to file instead of stdout\n"
60
+ " -j Output plan as JSON (compile only, don't execute)\n"
61
+ " --target NAME Compile target: native, json, or sql\n"
62
+ " --dialect NAME SQL dialect for --target sql (duckdb; sqlite/postgres planned)\n"
63
+ " --explain Output target/memory/emit/schema/state execution plan and exit\n"
64
+ " --memory max:SIZE Set memory cap policy, e.g. max:64MB\n"
65
+ " --spill-dir DIR Request disk spill directory for spillable plans\n"
66
+ " --engine NAME Execution engine: native or duckdb\n"
67
+ " --stats-json FILE Write stats side-channel NDJSON to FILE; use - for stderr\n"
68
+ " --allow-blocking Explicitly permit full-input native blocking steps\n"
69
+ " --fail-on-blocking Refuse native execution plans with blocking steps (default)\n"
70
+ " -p, --progress Show progress on stderr\n"
71
+ " -q Quiet mode (suppress stats on stderr)\n"
72
+ " --raw Force raw CSV stats output (disable report formatting)\n"
73
+ " -v Show version\n"
74
+ " -R, --recipes List built-in recipes\n"
75
+ " -h Show this help\n"
76
+ "\n"
77
+ "Recipes (use by name, e.g. %s profile):\n"
78
+ " profile, preview, schema, summary, count, cardinality,\n"
79
+ " distro, freq, dedup, clean, sample, head, tail, csv2json,\n"
80
+ " json2csv, tsv2csv, csv2tsv, histogram, hash, samples\n",
81
+ prog, prog, prog, prog, prog, prog, prog, prog,
82
+ prog, prog, prog, prog, prog, prog, prog);
83
+ }
84
+
85
+ static int cli_file_output_sink(int channel, const uint8_t *data, size_t len, void *user) {
86
+ (void)channel;
87
+ FILE *out = (FILE *)user;
88
+ if (!out || len == 0) return out ? TF_OK : TF_ERROR;
89
+ return fwrite(data, 1, len, out) == len ? TF_OK : TF_ERROR;
90
+ }
91
+
92
+ static char *read_file(const char *path) {
93
+ FILE *f = fopen(path, "r");
94
+ if (!f) return NULL;
95
+
96
+ if (fseek(f, 0, SEEK_END) != 0) { fclose(f); return NULL; }
97
+ long size = ftell(f);
98
+ if (fseek(f, 0, SEEK_SET) != 0) { fclose(f); return NULL; }
99
+
100
+ if (size <= 0) { fclose(f); return NULL; }
101
+
102
+ size_t alloc_len;
103
+ if (tf_size_add((size_t)size, 1, &alloc_len) != TF_OK) { fclose(f); return NULL; }
104
+ char *buf = tf_mallocarray_checked(alloc_len, sizeof(char));
105
+ if (!buf) { fclose(f); return NULL; }
106
+
107
+ size_t nread = fread(buf, 1, (size_t)size, f);
108
+ if (ferror(f)) {
109
+ free(buf);
110
+ fclose(f);
111
+ return NULL;
112
+ }
113
+ fclose(f);
114
+ buf[nread] = '\0';
115
+ return buf;
116
+ }
117
+
118
+ typedef enum {
119
+ CLI_ENGINE_NATIVE,
120
+ CLI_ENGINE_DUCKDB,
121
+ } cli_engine;
122
+
123
+ typedef struct {
124
+ cli_engine engine;
125
+ int allow_blocking;
126
+ int fail_on_blocking_seen;
127
+ int has_memory_limit;
128
+ size_t memory_limit_bytes;
129
+ const char *spill_dir;
130
+ } cli_memory_policy;
131
+
132
+ static const char *engine_name(cli_engine engine) {
133
+ switch (engine) {
134
+ case CLI_ENGINE_NATIVE: return "native";
135
+ case CLI_ENGINE_DUCKDB: return "duckdb";
136
+ default: return "unknown";
137
+ }
138
+ }
139
+
140
+ static const char *format_bytes(size_t bytes, char *buf, size_t buf_size) {
141
+ if (bytes < 1024) {
142
+ snprintf(buf, buf_size, "%zuB", bytes);
143
+ } else if (bytes < 1024 * 1024) {
144
+ snprintf(buf, buf_size, "%.1fKB", (double)bytes / 1024);
145
+ } else if (bytes < 1024 * 1024 * 1024) {
146
+ snprintf(buf, buf_size, "%.1fMB", (double)bytes / (1024 * 1024));
147
+ } else {
148
+ snprintf(buf, buf_size, "%.1fGB", (double)bytes / (1024 * 1024 * 1024));
149
+ }
150
+ return buf;
151
+ }
152
+
153
+ static int parse_size_bytes(const char *text, size_t *out, char *err, size_t err_size) {
154
+ const char *p = text;
155
+ if (!p || !*p) {
156
+ snprintf(err, err_size, "empty size");
157
+ return -1;
158
+ }
159
+ if (strncmp(p, "max:", 4) == 0 || strncmp(p, "max=", 4) == 0) p += 4;
160
+ while (isspace((unsigned char)*p)) p++;
161
+ if (!isdigit((unsigned char)*p)) {
162
+ snprintf(err, err_size, "expected size such as 64MB or max:64MB");
163
+ return -1;
164
+ }
165
+
166
+ errno = 0;
167
+ char *end = NULL;
168
+ unsigned long long value = strtoull(p, &end, 10);
169
+ if (errno != 0 || end == p || value == 0) {
170
+ snprintf(err, err_size, "invalid positive size '%s'", text);
171
+ return -1;
172
+ }
173
+ while (isspace((unsigned char)*end)) end++;
174
+
175
+ char suffix[8] = {0};
176
+ size_t si = 0;
177
+ while (*end && si + 1 < sizeof(suffix)) {
178
+ suffix[si++] = (char)tolower((unsigned char)*end++);
179
+ }
180
+ while (isspace((unsigned char)*end)) end++;
181
+ if (*end) {
182
+ snprintf(err, err_size, "invalid size suffix in '%s'", text);
183
+ return -1;
184
+ }
185
+
186
+ unsigned long long mul = 1;
187
+ if (suffix[0] == '\0' || strcmp(suffix, "b") == 0) mul = 1;
188
+ else if (strcmp(suffix, "k") == 0 || strcmp(suffix, "kb") == 0 || strcmp(suffix, "kib") == 0) mul = 1024ULL;
189
+ else if (strcmp(suffix, "m") == 0 || strcmp(suffix, "mb") == 0 || strcmp(suffix, "mib") == 0) mul = 1024ULL * 1024ULL;
190
+ else if (strcmp(suffix, "g") == 0 || strcmp(suffix, "gb") == 0 || strcmp(suffix, "gib") == 0) mul = 1024ULL * 1024ULL * 1024ULL;
191
+ else if (strcmp(suffix, "t") == 0 || strcmp(suffix, "tb") == 0 || strcmp(suffix, "tib") == 0) mul = 1024ULL * 1024ULL * 1024ULL * 1024ULL;
192
+ else {
193
+ snprintf(err, err_size, "unsupported size suffix '%s'", suffix);
194
+ return -1;
195
+ }
196
+
197
+ if (value > (unsigned long long)SIZE_MAX / mul) {
198
+ snprintf(err, err_size, "size '%s' is too large", text);
199
+ return -1;
200
+ }
201
+ *out = (size_t)(value * mul);
202
+ return 0;
203
+ }
204
+
205
+ static int set_engine(cli_memory_policy *policy, const char *name, char *err, size_t err_size) {
206
+ if (strcmp(name, "native") == 0) {
207
+ policy->engine = CLI_ENGINE_NATIVE;
208
+ return 0;
209
+ }
210
+ if (strcmp(name, "duckdb") == 0 || strcmp(name, "sql") == 0) {
211
+ policy->engine = CLI_ENGINE_DUCKDB;
212
+ return 0;
213
+ }
214
+ snprintf(err, err_size, "unknown engine '%s' (expected native or duckdb)", name);
215
+ return -1;
216
+ }
217
+
218
+
219
+ static int is_known_sql_dialect(const char *name) {
220
+ return name && (strcmp(name, "duckdb") == 0 ||
221
+ strcmp(name, "sqlite") == 0 ||
222
+ strcmp(name, "postgres") == 0);
223
+ }
224
+
225
+ static int sql_dialect_supported(const char *name) {
226
+ return name && strcmp(name, "duckdb") == 0;
227
+ }
228
+
229
+ static int memory_rank(tf_memory_class cls) {
230
+ switch (cls) {
231
+ case TF_MEM_ROW_LOCAL: return 0;
232
+ case TF_MEM_BOUNDED_STATE: return 1;
233
+ case TF_MEM_KEY_STATE: return 2;
234
+ case TF_MEM_EXTERNAL: return 3;
235
+ case TF_MEM_BLOCKING: return 4;
236
+ default: return 5;
237
+ }
238
+ }
239
+
240
+ static void print_cap_item(FILE *f, int *first, const char *name) {
241
+ fprintf(f, "%s%s", *first ? "" : ",", name);
242
+ *first = 0;
243
+ }
244
+
245
+
246
+ typedef struct {
247
+ char *data;
248
+ size_t len;
249
+ size_t cap;
250
+ } cli_string_builder;
251
+
252
+ static int sb_reserve(cli_string_builder *sb, size_t extra) {
253
+ if (!sb) return -1;
254
+ size_t need;
255
+ if (tf_size_add(sb->len, extra, &need) != TF_OK ||
256
+ tf_size_add(need, 1, &need) != TF_OK) {
257
+ return -1;
258
+ }
259
+ if (need <= sb->cap) return 0;
260
+ size_t new_cap;
261
+ if (tf_size_grow_pow2(sb->cap, need, 256, &new_cap) != TF_OK) return -1;
262
+ char *tmp = tf_reallocarray_checked(sb->data, new_cap, sizeof(char));
263
+ if (!tmp) return -1;
264
+ sb->data = tmp;
265
+ sb->cap = new_cap;
266
+ return 0;
267
+ }
268
+
269
+ static int sb_append_n(cli_string_builder *sb, const char *s, size_t n) {
270
+ if (!s) return 0;
271
+ if (sb_reserve(sb, n) != 0) return -1;
272
+ memcpy(sb->data + sb->len, s, n);
273
+ sb->len += n;
274
+ sb->data[sb->len] = '\0';
275
+ return 0;
276
+ }
277
+
278
+ static int sb_append(cli_string_builder *sb, const char *s) {
279
+ return sb_append_n(sb, s, s ? strlen(s) : 0);
280
+ }
281
+
282
+ static int sb_append_char(cli_string_builder *sb, char ch) {
283
+ if (sb_reserve(sb, 1) != 0) return -1;
284
+ sb->data[sb->len++] = ch;
285
+ sb->data[sb->len] = '\0';
286
+ return 0;
287
+ }
288
+
289
+ static int cli_report_buffer_append(char **buf, size_t *len, size_t *cap,
290
+ const uint8_t *data, size_t n) {
291
+ if (!buf || !len || !cap || !*buf || (!data && n > 0)) return -1;
292
+ size_t need;
293
+ if (tf_size_add(*len, n, &need) != TF_OK) return -1;
294
+ if (need > *cap) {
295
+ size_t new_cap;
296
+ if (tf_size_grow_pow2(*cap, need, PULL_BUF_SIZE, &new_cap) != TF_OK) return -1;
297
+ char *tmp = tf_reallocarray_checked(*buf, new_cap, sizeof(char));
298
+ if (!tmp) return -1;
299
+ *buf = tmp;
300
+ *cap = new_cap;
301
+ }
302
+ if (n > 0) memcpy(*buf + *len, data, n);
303
+ *len = need;
304
+ return 0;
305
+ }
306
+
307
+ static void cli_disable_report_buffer(FILE *fout, char **buf, size_t *len, size_t *cap) {
308
+ if (!buf || !len || !cap) return;
309
+ if (fout && *buf && *len > 0) fwrite(*buf, 1, *len, fout);
310
+ free(*buf);
311
+ *buf = NULL;
312
+ *len = 0;
313
+ *cap = 0;
314
+ }
315
+
316
+ static int cli_is_bare_dsl_token(const char *s) {
317
+ if (!s || !*s) return 0;
318
+ for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
319
+ if (isspace(*p) || *p == '|' || *p == '=' || *p == '"' || *p == '\'' || *p == '\\') return 0;
320
+ }
321
+ return 1;
322
+ }
323
+
324
+ static const char *cli_normalized_step_name(const char *op) {
325
+ if (!op) return "unknown";
326
+ if (strcmp(op, "codec.csv.decode") == 0 || strcmp(op, "codec.csv.encode") == 0) return "csv";
327
+ if (strcmp(op, "codec.jsonl.decode") == 0 || strcmp(op, "codec.jsonl.encode") == 0) return "jsonl";
328
+ if (strcmp(op, "codec.text.decode") == 0 || strcmp(op, "codec.text.encode") == 0) return "text";
329
+ if (strcmp(op, "codec.table.encode") == 0) return "table";
330
+ return op;
331
+ }
332
+
333
+ static int cli_append_json_value_token(cli_string_builder *sb, const cJSON *value) {
334
+ if (!value) return sb_append(sb, "null");
335
+ if (cJSON_IsString(value) && value->valuestring && cli_is_bare_dsl_token(value->valuestring)) {
336
+ return sb_append(sb, value->valuestring);
337
+ }
338
+ char *json = cJSON_PrintUnformatted((cJSON *)value);
339
+ if (!json) return -1;
340
+ int rc = sb_append(sb, json);
341
+ free(json);
342
+ return rc;
343
+ }
344
+
345
+ static int cli_append_normalized_arg(cli_string_builder *sb, const cJSON *arg) {
346
+ if (!arg || !arg->string) return 0;
347
+ if (sb_append_char(sb, ' ') != 0) return -1;
348
+ if (sb_append(sb, arg->string) != 0) return -1;
349
+ if (sb_append_char(sb, '=') != 0) return -1;
350
+ return cli_append_json_value_token(sb, arg);
351
+ }
352
+
353
+ static char *cli_ir_to_normalized_dsl(const tf_ir_plan *ir) {
354
+ if (!ir) return NULL;
355
+ cli_string_builder sb = {0};
356
+ for (size_t i = 0; i < ir->n_nodes; i++) {
357
+ const tf_ir_node *node = &ir->nodes[i];
358
+ if (i > 0 && sb_append(&sb, " | ") != 0) goto fail;
359
+ if (sb_append(&sb, cli_normalized_step_name(node->op)) != 0) goto fail;
360
+ if (node->args && cJSON_IsObject(node->args)) {
361
+ for (const cJSON *arg = node->args->child; arg; arg = arg->next) {
362
+ if (arg->string &&
363
+ (strcmp(arg->string, TF_POLICY_VALIDATED_FILE_PATH_ARG) == 0 ||
364
+ strcmp(arg->string, TF_POLICY_VALIDATED_RULES_FILE_PATH_ARG) == 0)) {
365
+ continue;
366
+ }
367
+ if (cli_append_normalized_arg(&sb, arg) != 0) goto fail;
368
+ }
369
+ }
370
+ }
371
+ if (!sb.data) {
372
+ sb.data = strdup("");
373
+ if (!sb.data) return NULL;
374
+ }
375
+ return sb.data;
376
+
377
+ fail:
378
+ free(sb.data);
379
+ return NULL;
380
+ }
381
+
382
+ static void print_caps(FILE *f, uint32_t caps) {
383
+ int first = 1;
384
+ if (caps & TF_CAP_STREAMING) print_cap_item(f, &first, "streaming");
385
+ if (caps & TF_CAP_BOUNDED_MEMORY) print_cap_item(f, &first, "bounded_memory");
386
+ if (caps & TF_CAP_BROWSER_SAFE) print_cap_item(f, &first, "browser_safe");
387
+ if (caps & TF_CAP_DETERMINISTIC) print_cap_item(f, &first, "deterministic");
388
+ if (caps & TF_CAP_FS) print_cap_item(f, &first, "fs");
389
+ if (caps & TF_CAP_NET) print_cap_item(f, &first, "net");
390
+ if (first) fprintf(f, "none");
391
+ }
392
+
393
+ static const tf_ir_node *first_node_with_memory_class(const tf_ir_plan *ir, tf_memory_class cls) {
394
+ for (size_t i = 0; i < ir->n_nodes; i++) {
395
+ if (ir->nodes[i].memory_class == cls) return &ir->nodes[i];
396
+ }
397
+ return NULL;
398
+ }
399
+
400
+ static const tf_ir_node *first_blocking_node(const tf_ir_plan *ir) {
401
+ return first_node_with_memory_class(ir, TF_MEM_BLOCKING);
402
+ }
403
+
404
+ static const tf_ir_node *first_key_state_node(const tf_ir_plan *ir) {
405
+ return first_node_with_memory_class(ir, TF_MEM_KEY_STATE);
406
+ }
407
+
408
+ static int has_sort_head_pattern(const tf_ir_plan *ir) {
409
+ for (size_t i = 0; i + 1 < ir->n_nodes; i++) {
410
+ if (strcmp(ir->nodes[i].op, "sort") == 0 && strcmp(ir->nodes[i + 1].op, "head") == 0)
411
+ return 1;
412
+ }
413
+ return 0;
414
+ }
415
+
416
+ static int node_arg_true(const tf_ir_node *node, const char *name) {
417
+ cJSON *item = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, name) : NULL;
418
+ return cJSON_IsTrue(item);
419
+ }
420
+
421
+ static int node_op_is_unique(const tf_ir_node *node) {
422
+ return node && node->op &&
423
+ (strcmp(node->op, "unique") == 0 || strcmp(node->op, "dedup") == 0);
424
+ }
425
+
426
+ static int node_array_arg_nonempty_main(const tf_ir_node *node, const char *name) {
427
+ cJSON *item = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, name) : NULL;
428
+ return cJSON_IsArray(item) && cJSON_GetArraySize(item) > 0;
429
+ }
430
+
431
+ static int node_positive_arg_main(const tf_ir_node *node, const char *name) {
432
+ cJSON *item = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, name) : NULL;
433
+ return cJSON_IsNumber(item) && item->valuedouble > 0.0;
434
+ }
435
+
436
+ static int node_op_is_pivot_spillable(const tf_ir_node *node) {
437
+ return node && node->op && strcmp(node->op, "pivot") == 0 &&
438
+ !node_arg_true(node, "sorted") &&
439
+ (node_array_arg_nonempty_main(node, "categories") || node_positive_arg_main(node, "max_categories"));
440
+ }
441
+
442
+
443
+ static int node_op_is_join_spillable(const tf_ir_node *node) {
444
+ if (!node || !node->op) return 0;
445
+ if (strcmp(node->op, "semi-join") == 0 || strcmp(node->op, "anti-join") == 0) return 1;
446
+ if (strcmp(node->op, "join") != 0) return 0;
447
+ if (!node->args) return 1;
448
+ cJSON *how = cJSON_GetObjectItemCaseSensitive(node->args, "how");
449
+ if (!cJSON_IsString(how) || !how->valuestring) return 1;
450
+ return strcmp(how->valuestring, "inner") == 0 || strcmp(how->valuestring, "left") == 0 ||
451
+ strcmp(how->valuestring, "semi") == 0 || strcmp(how->valuestring, "anti") == 0;
452
+ }
453
+
454
+ static int node_op_is_row_set_spillable(const tf_ir_node *node) {
455
+ return node && node->op &&
456
+ (strcmp(node->op, "intersect") == 0 || strcmp(node->op, "setdiff") == 0 ||
457
+ strcmp(node->op, "intersect-all") == 0 || strcmp(node->op, "setdiff-all") == 0 ||
458
+ strcmp(node->op, "union") == 0);
459
+ }
460
+
461
+ static int blocking_node_has_native_spill(const tf_ir_node *node) {
462
+ return node && node->op && (strcmp(node->op, "sort") == 0 || node_op_is_pivot_spillable(node));
463
+ }
464
+
465
+ static int key_state_node_has_native_spill(const tf_ir_node *node) {
466
+ return ((node_op_is_unique(node) ||
467
+ (node && node->op && strcmp(node->op, "group-agg") == 0) ||
468
+ node_op_is_join_spillable(node) ||
469
+ node_op_is_row_set_spillable(node)) &&
470
+ !node_arg_true(node, "sorted"));
471
+ }
472
+
473
+ static const tf_ir_node *first_key_state_node_without_native_spill(const tf_ir_plan *ir) {
474
+ for (size_t i = 0; i < ir->n_nodes; i++) {
475
+ const tf_ir_node *node = &ir->nodes[i];
476
+ if (node->memory_class == TF_MEM_KEY_STATE && !key_state_node_has_native_spill(node))
477
+ return node;
478
+ }
479
+ return NULL;
480
+ }
481
+
482
+ static int node_uses_native_spill(const tf_ir_node *node) {
483
+ return blocking_node_has_native_spill(node) || key_state_node_has_native_spill(node);
484
+ }
485
+
486
+ static const char *explain_step_target(const tf_ir_node *node, const cli_memory_policy *policy) {
487
+ if (policy->engine == CLI_ENGINE_DUCKDB) return "duckdb_sql";
488
+ if (policy->engine == CLI_ENGINE_NATIVE && policy->spill_dir && node_uses_native_spill(node))
489
+ return "native_spill";
490
+ return "native";
491
+ }
492
+
493
+ static const tf_ir_node *first_blocking_node_without_native_spill(const tf_ir_plan *ir) {
494
+ for (size_t i = 0; i < ir->n_nodes; i++) {
495
+ const tf_ir_node *node = &ir->nodes[i];
496
+ if (node->memory_class == TF_MEM_BLOCKING && !blocking_node_has_native_spill(node))
497
+ return node;
498
+ }
499
+ return NULL;
500
+ }
501
+
502
+ static int blocking_plan_can_use_native_spill(const tf_ir_plan *ir) {
503
+ return first_blocking_node(ir) && !first_blocking_node_without_native_spill(ir);
504
+ }
505
+
506
+ static const char *native_spill_support_summary(void) {
507
+ return "native spill currently supports sort, capped unsorted pivot "
508
+ "(categories=... or max_categories=N), unsorted unique/dedup, "
509
+ "unsorted group-agg, capped unsorted inner/left joins, unsorted "
510
+ "semi/anti filtering joins, unsorted intersect/setdiff/"
511
+ "intersect-all/setdiff-all, and duplicate-eliminating union";
512
+ }
513
+
514
+ static int set_json_item(cJSON *obj, const char *name, cJSON *item) {
515
+ if (!obj || !name || !item) {
516
+ cJSON_Delete(item);
517
+ return -1;
518
+ }
519
+ if (cJSON_GetObjectItemCaseSensitive(obj, name)) {
520
+ if (cJSON_ReplaceItemInObjectCaseSensitive(obj, name, item)) return 0;
521
+ cJSON_Delete(item);
522
+ return -1;
523
+ }
524
+ if (tf_json_add_item(obj, name, item) != TF_OK) {
525
+ cJSON_Delete(item);
526
+ return -1;
527
+ }
528
+ return 0;
529
+ }
530
+
531
+ static int set_json_string(cJSON *obj, const char *name, const char *value) {
532
+ return set_json_item(obj, name, cJSON_CreateString(value ? value : ""));
533
+ }
534
+
535
+ static int set_json_number(cJSON *obj, const char *name, double value) {
536
+ return set_json_item(obj, name, cJSON_CreateNumber(value));
537
+ }
538
+
539
+ static int apply_native_spill_policy(tf_ir_plan *ir, const cli_memory_policy *policy) {
540
+ if (!policy->spill_dir || policy->engine != CLI_ENGINE_NATIVE) return 0;
541
+ for (size_t i = 0; i < ir->n_nodes; i++) {
542
+ tf_ir_node *node = &ir->nodes[i];
543
+ if (!node_uses_native_spill(node)) continue;
544
+ if (!node->args) {
545
+ node->args = cJSON_CreateObject();
546
+ if (!node->args) return -1;
547
+ }
548
+ if (set_json_string(node->args, "spill_dir", policy->spill_dir) != 0) return -1;
549
+ if (policy->has_memory_limit &&
550
+ set_json_number(node->args, "spill_memory_bytes", (double)policy->memory_limit_bytes) != 0)
551
+ return -1;
552
+ }
553
+ return 0;
554
+ }
555
+
556
+ static int validate_memory_policy(const tf_ir_plan *ir, const cli_memory_policy *policy) {
557
+ const tf_ir_node *blocking = first_blocking_node(ir);
558
+ const tf_ir_node *key_state = first_key_state_node(ir);
559
+
560
+ if (policy->engine == CLI_ENGINE_DUCKDB) return 0;
561
+
562
+ if (policy->spill_dir) {
563
+ const tf_ir_node *unsupported_key = first_key_state_node_without_native_spill(ir);
564
+ if (unsupported_key) {
565
+ fprintf(stderr,
566
+ "error: --spill-dir was requested, but native spill is not implemented yet for key-state step '%s'\n",
567
+ unsupported_key->op);
568
+ fprintf(stderr,
569
+ "hint: %s; use op-specific caps or an external engine for this plan\n",
570
+ native_spill_support_summary());
571
+ return 1;
572
+ }
573
+ const tf_ir_node *unsupported = first_blocking_node_without_native_spill(ir);
574
+ if (unsupported) {
575
+ fprintf(stderr,
576
+ "error: --spill-dir was requested, but native spill is unavailable for blocking step '%s' with the current arguments\n",
577
+ unsupported->op);
578
+ fprintf(stderr,
579
+ "hint: %s; use --allow-blocking only for known-small data or an external engine for this plan\n",
580
+ native_spill_support_summary());
581
+ return 1;
582
+ }
583
+ }
584
+
585
+ if (policy->has_memory_limit && blocking &&
586
+ !(policy->spill_dir && blocking_plan_can_use_native_spill(ir))) {
587
+ char cap[32];
588
+ fprintf(stderr,
589
+ "error: --memory %s was requested, but native byte caps are not implemented for blocking step '%s' (state=%s)\n",
590
+ format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)),
591
+ blocking->op,
592
+ blocking->state_estimate ? blocking->state_estimate : "unknown");
593
+ fprintf(stderr,
594
+ "hint: rewrite to a bounded op, choose an external engine, or use --spill-dir when the step has a native spill mode and required caps\n");
595
+ return 1;
596
+ }
597
+
598
+ if (policy->has_memory_limit && key_state) {
599
+ size_t estimated = 0;
600
+ const tf_ir_node *failed = NULL;
601
+ char reason[192] = {0};
602
+ if (!tf_estimate_key_state_plan_bytes(ir, &estimated, &failed, reason, sizeof(reason))) {
603
+ char cap[32];
604
+ fprintf(stderr,
605
+ "error: --memory %s was requested, but %s\n",
606
+ format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)),
607
+ reason[0] ? reason : "a key-state step is not byte-bounded");
608
+ if (failed) {
609
+ fprintf(stderr,
610
+ "hint: add the op-specific cap for step '%s' or use an external/spill engine\n",
611
+ failed->op);
612
+ }
613
+ return 1;
614
+ }
615
+ if (estimated > policy->memory_limit_bytes) {
616
+ char cap[32], est[32];
617
+ fprintf(stderr,
618
+ "error: estimated native key-state memory %s exceeds --memory %s\n",
619
+ format_bytes(estimated, est, sizeof(est)),
620
+ format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)));
621
+ fprintf(stderr,
622
+ "hint: lower key/category/lookup caps or raise --memory for this known-bounded plan\n");
623
+ return 1;
624
+ }
625
+ }
626
+
627
+ if (blocking && !policy->allow_blocking &&
628
+ !(policy->spill_dir && blocking_plan_can_use_native_spill(ir))) {
629
+ fprintf(stderr,
630
+ "error: blocking step '%s' requires full input in native mode (state=%s)\n",
631
+ blocking->op,
632
+ blocking->state_estimate ? blocking->state_estimate : "unknown");
633
+ fprintf(stderr,
634
+ "hint: add --allow-blocking only for known-small inputs, rewrite to a bounded op, or choose an external engine/spill mode when available\n");
635
+ if (has_sort_head_pattern(ir)) {
636
+ fprintf(stderr,
637
+ "hint: replace sort | head N with top N column when top-k semantics are acceptable\n");
638
+ }
639
+ return 1;
640
+ }
641
+
642
+ return 0;
643
+ }
644
+
645
+ static void print_explain(const tf_ir_plan *ir, const cli_memory_policy *policy) {
646
+ tf_memory_class worst_mem = TF_MEM_ROW_LOCAL;
647
+ const char *worst_state = "O(batch_rows * columns)";
648
+ int has_flush = 0;
649
+ int has_per_batch = 0;
650
+ int has_data_schema = 0;
651
+
652
+ for (size_t i = 0; i < ir->n_nodes; i++) {
653
+ const tf_ir_node *node = &ir->nodes[i];
654
+ if (memory_rank(node->memory_class) > memory_rank(worst_mem)) {
655
+ worst_mem = node->memory_class;
656
+ worst_state = node->state_estimate ? node->state_estimate : "unknown";
657
+ }
658
+ if (node->emit_class == TF_EMIT_ON_FLUSH) has_flush = 1;
659
+ if (node->emit_class == TF_EMIT_PER_BATCH) has_per_batch = 1;
660
+ if (node->schema_class == TF_SCHEMA_DATA_DEPENDENT) has_data_schema = 1;
661
+ }
662
+
663
+ char cap[32];
664
+ printf("Tranfi execution plan\n");
665
+ char *normalized_dsl = cli_ir_to_normalized_dsl(ir);
666
+ printf("normalized_dsl: %s\n", normalized_dsl ? normalized_dsl : "null");
667
+ free(normalized_dsl);
668
+ printf("execution_target: %s\n", policy->spill_dir && policy->engine == CLI_ENGINE_NATIVE ? "native+spill" : engine_name(policy->engine));
669
+ printf("memory_policy: %s\n", policy->spill_dir ? "spill" : (policy->allow_blocking ? "allow_blocking" : "strict"));
670
+ printf("memory_limit: %s\n", policy->has_memory_limit ? format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)) : "none");
671
+ printf("spill_dir: %s\n", policy->spill_dir ? policy->spill_dir : "none");
672
+ printf("memory_class: %s\n", tf_memory_class_name(worst_mem));
673
+ printf("emit_class: %s\n", has_flush && has_per_batch ? "mixed" :
674
+ (has_flush ? "on_flush" : "per_batch"));
675
+ printf("schema_class: %s\n", has_data_schema ? "data_dependent" : "stable_or_parametric");
676
+ printf("state_estimate: %s\n", worst_state);
677
+ if (first_key_state_node(ir)) {
678
+ size_t key_bytes = 0;
679
+ const tf_ir_node *failed = NULL;
680
+ char reason[192] = {0};
681
+ if (tf_estimate_key_state_plan_bytes(ir, &key_bytes, &failed, reason, sizeof(reason))) {
682
+ char key_est[32];
683
+ printf("state_bytes_estimate: %s\n", format_bytes(key_bytes, key_est, sizeof(key_est)));
684
+ } else {
685
+ printf("state_bytes_estimate: unbounded (%s)\n", reason[0] ? reason : "missing key-state byte estimator");
686
+ }
687
+ } else {
688
+ printf("state_bytes_estimate: none\n");
689
+ }
690
+ if (first_blocking_node(ir)) {
691
+ if (policy->spill_dir && policy->engine == CLI_ENGINE_NATIVE && blocking_plan_can_use_native_spill(ir))
692
+ printf("note: blocking step will use native spill files\n");
693
+ else if (policy->engine == CLI_ENGINE_NATIVE && !policy->allow_blocking)
694
+ printf("warning: blocking native step present; default execution will reject it unless --allow-blocking is set\n");
695
+ else
696
+ printf("warning: blocking step present; selected policy must provide a bounded/external execution path\n");
697
+ if (has_sort_head_pattern(ir))
698
+ printf("hint: replace sort | head N with top N column when top-k semantics are acceptable\n");
699
+ }
700
+ printf("steps:\n");
701
+ for (size_t i = 0; i < ir->n_nodes; i++) {
702
+ const tf_ir_node *node = &ir->nodes[i];
703
+ printf(" %zu. %s\n", i, node->op);
704
+ printf(" target=%s memory=%s emit=%s schema=%s state=%s caps=",
705
+ explain_step_target(node, policy),
706
+ tf_memory_class_name(node->memory_class),
707
+ tf_emit_class_name(node->emit_class),
708
+ tf_schema_class_name(node->schema_class),
709
+ node->state_estimate ? node->state_estimate : "unknown");
710
+ print_caps(stdout, node->caps);
711
+ printf("\n");
712
+ if (node->memory_class == TF_MEM_KEY_STATE) {
713
+ size_t step_bytes = 0;
714
+ char reason[192] = {0};
715
+ if (tf_estimate_step_state_bytes(node, &step_bytes, reason, sizeof(reason))) {
716
+ char step_est[32];
717
+ printf(" state_bytes_estimate=%s\n", format_bytes(step_bytes, step_est, sizeof(step_est)));
718
+ } else {
719
+ printf(" state_bytes_estimate=unbounded (%s)\n", reason[0] ? reason : "missing key-state byte estimator");
720
+ }
721
+ }
722
+ }
723
+
724
+ char *ir_json = tf_ir_to_json(ir);
725
+ if (ir_json) {
726
+ printf("ir_json: %s\n", ir_json);
727
+ free(ir_json);
728
+ } else {
729
+ printf("ir_json: null\n");
730
+ }
731
+ }
732
+
733
+ int main(int argc, char **argv) {
734
+ const char *pipeline_file = NULL;
735
+ const char *pipeline_text = NULL;
736
+ const char *input_file = NULL;
737
+ const char *output_file = NULL;
738
+ int json_mode = 0;
739
+ int sql_mode = 0;
740
+ int explain_mode = 0;
741
+ const char *sql_dialect = "duckdb";
742
+ cli_memory_policy policy = {0};
743
+ policy.engine = CLI_ENGINE_NATIVE;
744
+ int quiet = 0;
745
+ int progress = 0;
746
+ int raw_stats = 0;
747
+ const char *stats_json_file = NULL;
748
+
749
+ /* Parse options */
750
+ int argi = 1;
751
+ while (argi < argc && argv[argi][0] == '-') {
752
+ const char *opt = argv[argi];
753
+ if (strcmp(opt, "-h") == 0 || strcmp(opt, "--help") == 0) {
754
+ usage(argv[0]);
755
+ return 0;
756
+ } else if (strcmp(opt, "-v") == 0 || strcmp(opt, "--version") == 0) {
757
+ printf("tranfi %s\n", tf_version());
758
+ return 0;
759
+ } else if (strcmp(opt, "-R") == 0 || strcmp(opt, "--recipes") == 0) {
760
+ size_t n = tf_recipe_count();
761
+ printf("Built-in recipes (%zu):\n\n", n);
762
+ for (size_t i = 0; i < n; i++) {
763
+ printf(" %-12s %s\n", tf_recipe_name(i), tf_recipe_description(i));
764
+ printf(" %-12s %s\n", "", tf_recipe_dsl(i));
765
+ printf("\n");
766
+ }
767
+ return 0;
768
+ } else if (strcmp(opt, "-j") == 0) {
769
+ json_mode = 1;
770
+ sql_mode = 0;
771
+ } else if (strcmp(opt, "--target") == 0 || strncmp(opt, "--target=", 9) == 0) {
772
+ const char *value = NULL;
773
+ if (strncmp(opt, "--target=", 9) == 0) {
774
+ value = opt + 9;
775
+ } else {
776
+ argi++;
777
+ if (argi >= argc) {
778
+ fprintf(stderr, "error: --target requires native, json, or sql\n");
779
+ return 1;
780
+ }
781
+ value = argv[argi];
782
+ }
783
+ if (strcmp(value, "native") == 0) {
784
+ json_mode = 0;
785
+ sql_mode = 0;
786
+ } else if (strcmp(value, "json") == 0 || strcmp(value, "ir") == 0) {
787
+ json_mode = 1;
788
+ sql_mode = 0;
789
+ } else if (strcmp(value, "sql") == 0) {
790
+ json_mode = 0;
791
+ sql_mode = 1;
792
+ } else {
793
+ fprintf(stderr, "error: unknown --target '%s' (expected native, json, or sql)\n", value);
794
+ return 1;
795
+ }
796
+ } else if (strcmp(opt, "--dialect") == 0 || strncmp(opt, "--dialect=", 10) == 0) {
797
+ const char *value = NULL;
798
+ if (strncmp(opt, "--dialect=", 10) == 0) {
799
+ value = opt + 10;
800
+ } else {
801
+ argi++;
802
+ if (argi >= argc) {
803
+ fprintf(stderr, "error: --dialect requires duckdb, sqlite, or postgres\n");
804
+ return 1;
805
+ }
806
+ value = argv[argi];
807
+ }
808
+ if (!is_known_sql_dialect(value)) {
809
+ fprintf(stderr, "error: unknown SQL dialect '%s' (expected duckdb, sqlite, or postgres)\n", value);
810
+ return 1;
811
+ }
812
+ sql_dialect = value;
813
+ } else if (strcmp(opt, "--explain") == 0) {
814
+ explain_mode = 1;
815
+ } else if (strcmp(opt, "--allow-blocking") == 0) {
816
+ policy.allow_blocking = 1;
817
+ } else if (strcmp(opt, "--fail-on-blocking") == 0) {
818
+ policy.fail_on_blocking_seen = 1;
819
+ } else if (strcmp(opt, "--memory") == 0 || strncmp(opt, "--memory=", 9) == 0) {
820
+ const char *value = NULL;
821
+ if (strncmp(opt, "--memory=", 9) == 0) {
822
+ value = opt + 9;
823
+ } else {
824
+ argi++;
825
+ if (argi >= argc) {
826
+ fprintf(stderr, "error: --memory requires a size such as max:64MB\n");
827
+ return 1;
828
+ }
829
+ value = argv[argi];
830
+ }
831
+ char err[160];
832
+ if (parse_size_bytes(value, &policy.memory_limit_bytes, err, sizeof(err)) != 0) {
833
+ fprintf(stderr, "error: invalid --memory value: %s\n", err);
834
+ return 1;
835
+ }
836
+ policy.has_memory_limit = 1;
837
+ } else if (strcmp(opt, "--spill-dir") == 0 || strncmp(opt, "--spill-dir=", 12) == 0) {
838
+ if (strncmp(opt, "--spill-dir=", 12) == 0) {
839
+ policy.spill_dir = opt + 12;
840
+ } else {
841
+ argi++;
842
+ if (argi >= argc) {
843
+ fprintf(stderr, "error: --spill-dir requires a directory argument\n");
844
+ return 1;
845
+ }
846
+ policy.spill_dir = argv[argi];
847
+ }
848
+ if (!policy.spill_dir || policy.spill_dir[0] == '\0') {
849
+ fprintf(stderr, "error: --spill-dir requires a non-empty directory\n");
850
+ return 1;
851
+ }
852
+ } else if (strcmp(opt, "--engine") == 0 || strncmp(opt, "--engine=", 9) == 0) {
853
+ const char *value = NULL;
854
+ if (strncmp(opt, "--engine=", 9) == 0) {
855
+ value = opt + 9;
856
+ } else {
857
+ argi++;
858
+ if (argi >= argc) {
859
+ fprintf(stderr, "error: --engine requires native or duckdb\n");
860
+ return 1;
861
+ }
862
+ value = argv[argi];
863
+ }
864
+ char err[160];
865
+ if (set_engine(&policy, value, err, sizeof(err)) != 0) {
866
+ fprintf(stderr, "error: %s\n", err);
867
+ return 1;
868
+ }
869
+ } else if (strcmp(opt, "-q") == 0) {
870
+ quiet = 1;
871
+ } else if (strcmp(opt, "-p") == 0 || strcmp(opt, "--progress") == 0) {
872
+ progress = 1;
873
+ } else if (strcmp(opt, "--raw") == 0) {
874
+ raw_stats = 1;
875
+ } else if (strcmp(opt, "--stats-json") == 0 || strncmp(opt, "--stats-json=", 13) == 0) {
876
+ if (strncmp(opt, "--stats-json=", 13) == 0) {
877
+ stats_json_file = opt + 13;
878
+ } else {
879
+ argi++;
880
+ if (argi >= argc) {
881
+ fprintf(stderr, "error: --stats-json requires a file path, or - for stderr\n");
882
+ return 1;
883
+ }
884
+ stats_json_file = argv[argi];
885
+ }
886
+ if (!stats_json_file || stats_json_file[0] == '\0') {
887
+ fprintf(stderr, "error: --stats-json requires a non-empty file path, or - for stderr\n");
888
+ return 1;
889
+ }
890
+ } else if (strcmp(opt, "-f") == 0) {
891
+ argi++;
892
+ if (argi >= argc) {
893
+ fprintf(stderr, "error: -f requires a file argument\n");
894
+ return 1;
895
+ }
896
+ pipeline_file = argv[argi];
897
+ } else if (strcmp(opt, "-i") == 0) {
898
+ argi++;
899
+ if (argi >= argc) {
900
+ fprintf(stderr, "error: -i requires a file argument\n");
901
+ return 1;
902
+ }
903
+ input_file = argv[argi];
904
+ } else if (strcmp(opt, "-o") == 0) {
905
+ argi++;
906
+ if (argi >= argc) {
907
+ fprintf(stderr, "error: -o requires a file argument\n");
908
+ return 1;
909
+ }
910
+ output_file = argv[argi];
911
+ } else {
912
+ fprintf(stderr, "error: unknown option '%s'\n", opt);
913
+ return 1;
914
+ }
915
+ argi++;
916
+ }
917
+
918
+ if (policy.allow_blocking && policy.fail_on_blocking_seen) {
919
+ fprintf(stderr, "error: --allow-blocking conflicts with --fail-on-blocking\n");
920
+ return 1;
921
+ }
922
+ if (sql_mode) {
923
+ policy.engine = CLI_ENGINE_DUCKDB;
924
+ }
925
+
926
+ /* Get pipeline text */
927
+ char *file_content = NULL;
928
+ if (pipeline_file) {
929
+ file_content = read_file(pipeline_file);
930
+ if (!file_content) {
931
+ fprintf(stderr, "error: cannot read file '%s'\n", pipeline_file);
932
+ return 1;
933
+ }
934
+ pipeline_text = file_content;
935
+ } else if (argi < argc) {
936
+ pipeline_text = argv[argi];
937
+ } else {
938
+ fprintf(stderr, "error: no pipeline specified\n\n");
939
+ usage(argv[0]);
940
+ free(file_content);
941
+ return 1;
942
+ }
943
+
944
+ /* Parse pipeline: recipe name → JSON recipe → DSL */
945
+ char *error = NULL;
946
+ tf_ir_plan *ir = NULL;
947
+ size_t pt_len = strlen(pipeline_text);
948
+ /* Skip leading whitespace for detection */
949
+ const char *pt = pipeline_text;
950
+ while (*pt == ' ' || *pt == '\t' || *pt == '\n' || *pt == '\r') pt++;
951
+
952
+ if (*pt == '{') {
953
+ /* JSON recipe */
954
+ ir = tf_ir_from_json(pipeline_text, pt_len, &error);
955
+ } else if (!strchr(pt, '|') && !strchr(pt, ' ')) {
956
+ /* Single word — try built-in recipe */
957
+ const char *recipe_dsl = tf_recipe_find_dsl(pt);
958
+ if (recipe_dsl) {
959
+ ir = tf_dsl_parse(recipe_dsl, strlen(recipe_dsl), &error);
960
+ } else {
961
+ ir = tf_dsl_parse(pipeline_text, pt_len, &error);
962
+ }
963
+ } else {
964
+ ir = tf_dsl_parse(pipeline_text, pt_len, &error);
965
+ }
966
+ free(file_content);
967
+
968
+ if (!ir) {
969
+ fprintf(stderr, "error: %s\n", error ? error : "failed to parse pipeline");
970
+ free(error);
971
+ return 1;
972
+ }
973
+
974
+ /* Validate */
975
+ if (tf_ir_validate(ir) != TF_OK) {
976
+ fprintf(stderr, "error: %s\n", ir->error ? ir->error : "validation failed");
977
+ tf_ir_plan_free(ir);
978
+ return 1;
979
+ }
980
+
981
+ /* Schema inference allows unknown runtime schemas but reports invalid known-schema plans. */
982
+ tf_set_last_error(NULL);
983
+ if (tf_ir_infer_schema(ir) != TF_OK) {
984
+ const char *detail = tf_last_error();
985
+ if (detail && detail[0])
986
+ fprintf(stderr, "error: schema inference failed: %s\n", detail);
987
+ else
988
+ fprintf(stderr, "error: schema inference failed\n");
989
+ tf_ir_plan_free(ir);
990
+ return 1;
991
+ }
992
+
993
+ if (apply_native_spill_policy(ir, &policy) != 0) {
994
+ fprintf(stderr, "error: failed to apply native spill policy\n");
995
+ tf_ir_plan_free(ir);
996
+ return 1;
997
+ }
998
+
999
+ if (tf_ir_validate(ir) != TF_OK) {
1000
+ fprintf(stderr, "error: %s\n", ir->error ? ir->error : "validation failed after applying native spill policy");
1001
+ tf_ir_plan_free(ir);
1002
+ return 1;
1003
+ }
1004
+
1005
+ if (explain_mode) {
1006
+ print_explain(ir, &policy);
1007
+ tf_ir_plan_free(ir);
1008
+ return 0;
1009
+ }
1010
+
1011
+ if (sql_mode) {
1012
+ if (!sql_dialect_supported(sql_dialect)) {
1013
+ fprintf(stderr,
1014
+ "error: SQL dialect '%s' is recognized but not implemented yet; only duckdb lowering is available\n",
1015
+ sql_dialect);
1016
+ tf_ir_plan_free(ir);
1017
+ return 1;
1018
+ }
1019
+ char *sql_error = NULL;
1020
+ char *sql = tf_ir_plan_to_sql(ir, &sql_error);
1021
+ if (!sql) {
1022
+ fprintf(stderr, "error: SQL compile failed for dialect '%s': %s\n",
1023
+ sql_dialect, sql_error ? sql_error : "unknown error");
1024
+ free(sql_error);
1025
+ tf_ir_plan_free(ir);
1026
+ return 1;
1027
+ }
1028
+ printf("%s\n", sql);
1029
+ tf_string_free(sql);
1030
+ tf_ir_plan_free(ir);
1031
+ return 0;
1032
+ }
1033
+
1034
+ /* JSON mode: print IR and exit */
1035
+ if (json_mode) {
1036
+ char *json = tf_ir_to_json(ir);
1037
+ if (json) {
1038
+ printf("%s\n", json);
1039
+ free(json);
1040
+ }
1041
+ tf_ir_plan_free(ir);
1042
+ return 0;
1043
+ }
1044
+
1045
+ if (validate_memory_policy(ir, &policy) != 0) {
1046
+ tf_ir_plan_free(ir);
1047
+ return 1;
1048
+ }
1049
+
1050
+ if (policy.engine == CLI_ENGINE_DUCKDB) {
1051
+ fprintf(stderr,
1052
+ "error: the C CLI cannot execute --engine duckdb directly; use Python/Node with engine='duckdb' or compile SQL with --target sql --dialect duckdb\n");
1053
+ tf_ir_plan_free(ir);
1054
+ return 1;
1055
+ }
1056
+
1057
+ /* Compile to native pipeline */
1058
+ tf_pipeline *p = tf_pipeline_create_from_ir(ir);
1059
+ tf_ir_plan_free(ir);
1060
+
1061
+ if (!p) {
1062
+ fprintf(stderr, "error: %s\n",
1063
+ tf_last_error() ? tf_last_error() : "failed to create pipeline");
1064
+ return 1;
1065
+ }
1066
+
1067
+ /* Open I/O files */
1068
+ FILE *fin = stdin;
1069
+ FILE *fout = stdout;
1070
+ FILE *fstats = NULL;
1071
+ int close_stats = 0;
1072
+
1073
+ if (input_file) {
1074
+ fin = fopen(input_file, "rb");
1075
+ if (!fin) {
1076
+ fprintf(stderr, "error: cannot open input file '%s'\n", input_file);
1077
+ tf_pipeline_free(p);
1078
+ return 1;
1079
+ }
1080
+ }
1081
+
1082
+ if (output_file) {
1083
+ fout = fopen(output_file, "wb");
1084
+ if (!fout) {
1085
+ fprintf(stderr, "error: cannot open output file '%s'\n", output_file);
1086
+ if (fin != stdin) fclose(fin);
1087
+ tf_pipeline_free(p);
1088
+ return 1;
1089
+ }
1090
+ }
1091
+
1092
+ if (stats_json_file) {
1093
+ if (strcmp(stats_json_file, "-") == 0) {
1094
+ fstats = stderr;
1095
+ } else {
1096
+ fstats = fopen(stats_json_file, "wb");
1097
+ if (!fstats) {
1098
+ fprintf(stderr, "error: cannot open stats JSON file '%s'\n", stats_json_file);
1099
+ if (fin != stdin) fclose(fin);
1100
+ if (fout != stdout) fclose(fout);
1101
+ tf_pipeline_free(p);
1102
+ return 1;
1103
+ }
1104
+ close_stats = 1;
1105
+ }
1106
+ }
1107
+
1108
+ /* Decide whether to buffer output for report formatting.
1109
+ * When stdout is a TTY and --raw is not set, buffer main output
1110
+ * and try to render it as a rich report. Falls back to raw CSV
1111
+ * if the output doesn't look like a stats table. */
1112
+ int try_report = !raw_stats && !output_file && isatty(STDOUT_FILENO);
1113
+ if (!try_report) {
1114
+ if (tf_pipeline_set_sink(p, TF_CHAN_MAIN, cli_file_output_sink, fout) != TF_OK) {
1115
+ fprintf(stderr, "error: failed to attach output sink\n");
1116
+ if (fin != stdin) fclose(fin);
1117
+ if (fout != stdout) fclose(fout);
1118
+ if (close_stats && fstats) fclose(fstats);
1119
+ tf_pipeline_free(p);
1120
+ return 1;
1121
+ }
1122
+ }
1123
+
1124
+ /* Stream input → pipeline → output */
1125
+ uint8_t read_buf[READ_BUF_SIZE];
1126
+ size_t nread;
1127
+ size_t total_bytes = 0;
1128
+
1129
+ /* Output buffer (used when try_report is true) */
1130
+ size_t out_cap = PULL_BUF_SIZE;
1131
+ size_t out_len = 0;
1132
+ char *out_buf = try_report ? tf_mallocarray_checked(out_cap, sizeof(char)) : NULL;
1133
+ if (try_report && !out_buf) {
1134
+ out_cap = 0;
1135
+ try_report = 0;
1136
+ }
1137
+
1138
+ while ((nread = fread(read_buf, 1, sizeof(read_buf), fin)) > 0) {
1139
+ if (tf_pipeline_push(p, read_buf, nread) != TF_OK) {
1140
+ fprintf(stderr, "error: %s\n",
1141
+ tf_pipeline_error(p) ? tf_pipeline_error(p) : "push failed");
1142
+ free(out_buf);
1143
+ if (fin != stdin) fclose(fin);
1144
+ if (fout != stdout) fclose(fout);
1145
+ if (close_stats && fstats) fclose(fstats);
1146
+ tf_pipeline_free(p);
1147
+ return 1;
1148
+ }
1149
+
1150
+ total_bytes += nread;
1151
+
1152
+ /* Pull any available output immediately (streaming) */
1153
+ uint8_t pull_buf[PULL_BUF_SIZE];
1154
+ size_t n;
1155
+ while ((n = tf_pipeline_pull(p, TF_CHAN_MAIN, pull_buf, sizeof(pull_buf))) > 0) {
1156
+ if (try_report && out_buf) {
1157
+ if (cli_report_buffer_append(&out_buf, &out_len, &out_cap, pull_buf, n) != 0) {
1158
+ cli_disable_report_buffer(fout, &out_buf, &out_len, &out_cap);
1159
+ fwrite(pull_buf, 1, n, fout);
1160
+ try_report = 0;
1161
+ }
1162
+ } else {
1163
+ fwrite(pull_buf, 1, n, fout);
1164
+ }
1165
+ }
1166
+
1167
+ /* Show progress */
1168
+ if (progress) {
1169
+ char bytes_str[32];
1170
+ format_bytes(total_bytes, bytes_str, sizeof(bytes_str));
1171
+ fprintf(stderr, "\r%s processed", bytes_str);
1172
+ }
1173
+ }
1174
+
1175
+ /* Finish */
1176
+ if (tf_pipeline_finish(p) != TF_OK) {
1177
+ fprintf(stderr, "error: %s\n",
1178
+ tf_pipeline_error(p) ? tf_pipeline_error(p) : "finish failed");
1179
+ free(out_buf);
1180
+ if (fin != stdin) fclose(fin);
1181
+ if (fout != stdout) fclose(fout);
1182
+ if (close_stats && fstats) fclose(fstats);
1183
+ tf_pipeline_free(p);
1184
+ return 1;
1185
+ }
1186
+
1187
+ /* Pull remaining output */
1188
+ uint8_t pull_buf[PULL_BUF_SIZE];
1189
+ size_t n;
1190
+ while ((n = tf_pipeline_pull(p, TF_CHAN_MAIN, pull_buf, sizeof(pull_buf))) > 0) {
1191
+ if (try_report && out_buf) {
1192
+ if (cli_report_buffer_append(&out_buf, &out_len, &out_cap, pull_buf, n) != 0) {
1193
+ cli_disable_report_buffer(fout, &out_buf, &out_len, &out_cap);
1194
+ fwrite(pull_buf, 1, n, fout);
1195
+ try_report = 0;
1196
+ }
1197
+ } else {
1198
+ fwrite(pull_buf, 1, n, fout);
1199
+ }
1200
+ }
1201
+
1202
+ /* Try report formatting, fall back to raw */
1203
+ if (try_report && out_buf && out_len > 0) {
1204
+ char *report = tf_report_format(out_buf, out_len, 1);
1205
+ if (report) {
1206
+ fwrite(report, 1, strlen(report), fout);
1207
+ free(report);
1208
+ } else {
1209
+ fwrite(out_buf, 1, out_len, fout);
1210
+ }
1211
+ }
1212
+ free(out_buf);
1213
+ fflush(fout);
1214
+
1215
+ if (progress) {
1216
+ char bytes_str[32];
1217
+ format_bytes(total_bytes, bytes_str, sizeof(bytes_str));
1218
+ fprintf(stderr, "\r%s processed (done)\n", bytes_str);
1219
+ }
1220
+
1221
+ /* Pull errors to stderr */
1222
+ while ((n = tf_pipeline_pull(p, TF_CHAN_ERRORS, pull_buf, sizeof(pull_buf))) > 0) {
1223
+ fwrite(pull_buf, 1, n, stderr);
1224
+ }
1225
+
1226
+ /* Pull stats to the requested machine-readable path, or stderr unless quiet. */
1227
+ FILE *stats_out = stats_json_file ? fstats : (!quiet ? stderr : NULL);
1228
+ if (stats_out) {
1229
+ while ((n = tf_pipeline_pull(p, TF_CHAN_STATS, pull_buf, sizeof(pull_buf))) > 0) {
1230
+ fwrite(pull_buf, 1, n, stats_out);
1231
+ }
1232
+ fflush(stats_out);
1233
+ }
1234
+
1235
+ /* Cleanup */
1236
+ if (fin != stdin) fclose(fin);
1237
+ if (fout != stdout) fclose(fout);
1238
+ if (close_stats && fstats) fclose(fstats);
1239
+ tf_pipeline_free(p);
1240
+ return 0;
1241
+ }