tranfi 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +272 -40
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +348 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +8 -11
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/main.c CHANGED
@@ -5,6 +5,7 @@
5
5
  * tranfi 'csv | filter "col(age) > 25" | select name,age | csv' < in.csv
6
6
  * tranfi -f pipeline.tf < in.csv > out.csv
7
7
  * tranfi -j 'csv | head 5 | csv' # compile only, output JSON
8
+ * tranfi --target sql --dialect duckdb 'csv | head 5 | csv'
8
9
  * tranfi -i input.csv -o output.csv 'csv | filter "col(age) > 25" | csv'
9
10
  *
10
11
  * Channels:
@@ -14,6 +15,7 @@
14
15
 
15
16
  #include "tranfi.h"
16
17
  #include "internal.h"
18
+ #include "cJSON.h"
17
19
  #include "ir.h"
18
20
  #include "dsl.h"
19
21
  #include "recipes.h"
@@ -22,6 +24,10 @@
22
24
  #include <stdlib.h>
23
25
  #include <string.h>
24
26
  #include <unistd.h>
27
+ #include <ctype.h>
28
+ #include <errno.h>
29
+ #include <stdint.h>
30
+ #include <strings.h>
25
31
 
26
32
  #define READ_BUF_SIZE (64 * 1024)
27
33
  #define PULL_BUF_SIZE (64 * 1024)
@@ -41,16 +47,26 @@ static void usage(const char *prog) {
41
47
  " %s 'csv | head 10 | csv' # first N rows\n"
42
48
  " %s 'csv | skip 5 | csv' # skip first 5 rows\n"
43
49
  " %s 'csv | derive total=col(price)*col(qty) | csv' # computed columns\n"
44
- " %s 'csv | sort age | csv' # sort by column\n"
50
+ " %s --allow-blocking 'csv | sort age | csv' # sort known-small input\n"
45
51
  " %s 'csv | unique name | csv' # deduplicate\n"
46
52
  " %s 'csv | stats | csv' # aggregate stats\n"
47
53
  " %s 'jsonl | filter \"col(x) > 0\" | jsonl' # JSONL variant\n"
54
+ " %s --target sql --dialect duckdb 'csv | head 5 | csv' # print SQL\n"
48
55
  "\n"
49
56
  "Options:\n"
50
57
  " -f FILE Read pipeline from file instead of argument\n"
51
58
  " -i FILE Read input from file instead of stdin\n"
52
59
  " -o FILE Write output to file instead of stdout\n"
53
60
  " -j Output plan as JSON (compile only, don't execute)\n"
61
+ " --target NAME Compile target: native, json, or sql\n"
62
+ " --dialect NAME SQL dialect for --target sql (duckdb; sqlite/postgres planned)\n"
63
+ " --explain Output target/memory/emit/schema/state execution plan and exit\n"
64
+ " --memory max:SIZE Set memory cap policy, e.g. max:64MB\n"
65
+ " --spill-dir DIR Request disk spill directory for spillable plans\n"
66
+ " --engine NAME Execution engine: native or duckdb\n"
67
+ " --stats-json FILE Write stats side-channel NDJSON to FILE; use - for stderr\n"
68
+ " --allow-blocking Explicitly permit full-input native blocking steps\n"
69
+ " --fail-on-blocking Refuse native execution plans with blocking steps (default)\n"
54
70
  " -p, --progress Show progress on stderr\n"
55
71
  " -q Quiet mode (suppress stats on stderr)\n"
56
72
  " --raw Force raw CSV stats output (disable report formatting)\n"
@@ -63,28 +79,64 @@ static void usage(const char *prog) {
63
79
  " distro, freq, dedup, clean, sample, head, tail, csv2json,\n"
64
80
  " json2csv, tsv2csv, csv2tsv, histogram, hash, samples\n",
65
81
  prog, prog, prog, prog, prog, prog, prog, prog,
66
- prog, prog, prog, prog, prog, prog);
82
+ prog, prog, prog, prog, prog, prog, prog);
83
+ }
84
+
85
+ static int cli_file_output_sink(int channel, const uint8_t *data, size_t len, void *user) {
86
+ (void)channel;
87
+ FILE *out = (FILE *)user;
88
+ if (!out || len == 0) return out ? TF_OK : TF_ERROR;
89
+ return fwrite(data, 1, len, out) == len ? TF_OK : TF_ERROR;
67
90
  }
68
91
 
69
92
  static char *read_file(const char *path) {
70
93
  FILE *f = fopen(path, "r");
71
94
  if (!f) return NULL;
72
95
 
73
- fseek(f, 0, SEEK_END);
96
+ if (fseek(f, 0, SEEK_END) != 0) { fclose(f); return NULL; }
74
97
  long size = ftell(f);
75
- fseek(f, 0, SEEK_SET);
98
+ if (fseek(f, 0, SEEK_SET) != 0) { fclose(f); return NULL; }
76
99
 
77
100
  if (size <= 0) { fclose(f); return NULL; }
78
101
 
79
- char *buf = malloc(size + 1);
102
+ size_t alloc_len;
103
+ if (tf_size_add((size_t)size, 1, &alloc_len) != TF_OK) { fclose(f); return NULL; }
104
+ char *buf = tf_mallocarray_checked(alloc_len, sizeof(char));
80
105
  if (!buf) { fclose(f); return NULL; }
81
106
 
82
- size_t nread = fread(buf, 1, size, f);
107
+ size_t nread = fread(buf, 1, (size_t)size, f);
108
+ if (ferror(f)) {
109
+ free(buf);
110
+ fclose(f);
111
+ return NULL;
112
+ }
83
113
  fclose(f);
84
114
  buf[nread] = '\0';
85
115
  return buf;
86
116
  }
87
117
 
118
+ typedef enum {
119
+ CLI_ENGINE_NATIVE,
120
+ CLI_ENGINE_DUCKDB,
121
+ } cli_engine;
122
+
123
+ typedef struct {
124
+ cli_engine engine;
125
+ int allow_blocking;
126
+ int fail_on_blocking_seen;
127
+ int has_memory_limit;
128
+ size_t memory_limit_bytes;
129
+ const char *spill_dir;
130
+ } cli_memory_policy;
131
+
132
+ static const char *engine_name(cli_engine engine) {
133
+ switch (engine) {
134
+ case CLI_ENGINE_NATIVE: return "native";
135
+ case CLI_ENGINE_DUCKDB: return "duckdb";
136
+ default: return "unknown";
137
+ }
138
+ }
139
+
88
140
  static const char *format_bytes(size_t bytes, char *buf, size_t buf_size) {
89
141
  if (bytes < 1024) {
90
142
  snprintf(buf, buf_size, "%zuB", bytes);
@@ -98,15 +150,601 @@ static const char *format_bytes(size_t bytes, char *buf, size_t buf_size) {
98
150
  return buf;
99
151
  }
100
152
 
153
+ static int parse_size_bytes(const char *text, size_t *out, char *err, size_t err_size) {
154
+ const char *p = text;
155
+ if (!p || !*p) {
156
+ snprintf(err, err_size, "empty size");
157
+ return -1;
158
+ }
159
+ if (strncmp(p, "max:", 4) == 0 || strncmp(p, "max=", 4) == 0) p += 4;
160
+ while (isspace((unsigned char)*p)) p++;
161
+ if (!isdigit((unsigned char)*p)) {
162
+ snprintf(err, err_size, "expected size such as 64MB or max:64MB");
163
+ return -1;
164
+ }
165
+
166
+ errno = 0;
167
+ char *end = NULL;
168
+ unsigned long long value = strtoull(p, &end, 10);
169
+ if (errno != 0 || end == p || value == 0) {
170
+ snprintf(err, err_size, "invalid positive size '%s'", text);
171
+ return -1;
172
+ }
173
+ while (isspace((unsigned char)*end)) end++;
174
+
175
+ char suffix[8] = {0};
176
+ size_t si = 0;
177
+ while (*end && si + 1 < sizeof(suffix)) {
178
+ suffix[si++] = (char)tolower((unsigned char)*end++);
179
+ }
180
+ while (isspace((unsigned char)*end)) end++;
181
+ if (*end) {
182
+ snprintf(err, err_size, "invalid size suffix in '%s'", text);
183
+ return -1;
184
+ }
185
+
186
+ unsigned long long mul = 1;
187
+ if (suffix[0] == '\0' || strcmp(suffix, "b") == 0) mul = 1;
188
+ else if (strcmp(suffix, "k") == 0 || strcmp(suffix, "kb") == 0 || strcmp(suffix, "kib") == 0) mul = 1024ULL;
189
+ else if (strcmp(suffix, "m") == 0 || strcmp(suffix, "mb") == 0 || strcmp(suffix, "mib") == 0) mul = 1024ULL * 1024ULL;
190
+ else if (strcmp(suffix, "g") == 0 || strcmp(suffix, "gb") == 0 || strcmp(suffix, "gib") == 0) mul = 1024ULL * 1024ULL * 1024ULL;
191
+ else if (strcmp(suffix, "t") == 0 || strcmp(suffix, "tb") == 0 || strcmp(suffix, "tib") == 0) mul = 1024ULL * 1024ULL * 1024ULL * 1024ULL;
192
+ else {
193
+ snprintf(err, err_size, "unsupported size suffix '%s'", suffix);
194
+ return -1;
195
+ }
196
+
197
+ if (value > (unsigned long long)SIZE_MAX / mul) {
198
+ snprintf(err, err_size, "size '%s' is too large", text);
199
+ return -1;
200
+ }
201
+ *out = (size_t)(value * mul);
202
+ return 0;
203
+ }
204
+
205
+ static int set_engine(cli_memory_policy *policy, const char *name, char *err, size_t err_size) {
206
+ if (strcmp(name, "native") == 0) {
207
+ policy->engine = CLI_ENGINE_NATIVE;
208
+ return 0;
209
+ }
210
+ if (strcmp(name, "duckdb") == 0 || strcmp(name, "sql") == 0) {
211
+ policy->engine = CLI_ENGINE_DUCKDB;
212
+ return 0;
213
+ }
214
+ snprintf(err, err_size, "unknown engine '%s' (expected native or duckdb)", name);
215
+ return -1;
216
+ }
217
+
218
+
219
+ static int is_known_sql_dialect(const char *name) {
220
+ return name && (strcmp(name, "duckdb") == 0 ||
221
+ strcmp(name, "sqlite") == 0 ||
222
+ strcmp(name, "postgres") == 0);
223
+ }
224
+
225
+ static int sql_dialect_supported(const char *name) {
226
+ return name && strcmp(name, "duckdb") == 0;
227
+ }
228
+
229
+ static int memory_rank(tf_memory_class cls) {
230
+ switch (cls) {
231
+ case TF_MEM_ROW_LOCAL: return 0;
232
+ case TF_MEM_BOUNDED_STATE: return 1;
233
+ case TF_MEM_KEY_STATE: return 2;
234
+ case TF_MEM_EXTERNAL: return 3;
235
+ case TF_MEM_BLOCKING: return 4;
236
+ default: return 5;
237
+ }
238
+ }
239
+
240
+ static void print_cap_item(FILE *f, int *first, const char *name) {
241
+ fprintf(f, "%s%s", *first ? "" : ",", name);
242
+ *first = 0;
243
+ }
244
+
245
+
246
+ typedef struct {
247
+ char *data;
248
+ size_t len;
249
+ size_t cap;
250
+ } cli_string_builder;
251
+
252
+ static int sb_reserve(cli_string_builder *sb, size_t extra) {
253
+ if (!sb) return -1;
254
+ size_t need;
255
+ if (tf_size_add(sb->len, extra, &need) != TF_OK ||
256
+ tf_size_add(need, 1, &need) != TF_OK) {
257
+ return -1;
258
+ }
259
+ if (need <= sb->cap) return 0;
260
+ size_t new_cap;
261
+ if (tf_size_grow_pow2(sb->cap, need, 256, &new_cap) != TF_OK) return -1;
262
+ char *tmp = tf_reallocarray_checked(sb->data, new_cap, sizeof(char));
263
+ if (!tmp) return -1;
264
+ sb->data = tmp;
265
+ sb->cap = new_cap;
266
+ return 0;
267
+ }
268
+
269
+ static int sb_append_n(cli_string_builder *sb, const char *s, size_t n) {
270
+ if (!s) return 0;
271
+ if (sb_reserve(sb, n) != 0) return -1;
272
+ memcpy(sb->data + sb->len, s, n);
273
+ sb->len += n;
274
+ sb->data[sb->len] = '\0';
275
+ return 0;
276
+ }
277
+
278
+ static int sb_append(cli_string_builder *sb, const char *s) {
279
+ return sb_append_n(sb, s, s ? strlen(s) : 0);
280
+ }
281
+
282
+ static int sb_append_char(cli_string_builder *sb, char ch) {
283
+ if (sb_reserve(sb, 1) != 0) return -1;
284
+ sb->data[sb->len++] = ch;
285
+ sb->data[sb->len] = '\0';
286
+ return 0;
287
+ }
288
+
289
+ static int cli_report_buffer_append(char **buf, size_t *len, size_t *cap,
290
+ const uint8_t *data, size_t n) {
291
+ if (!buf || !len || !cap || !*buf || (!data && n > 0)) return -1;
292
+ size_t need;
293
+ if (tf_size_add(*len, n, &need) != TF_OK) return -1;
294
+ if (need > *cap) {
295
+ size_t new_cap;
296
+ if (tf_size_grow_pow2(*cap, need, PULL_BUF_SIZE, &new_cap) != TF_OK) return -1;
297
+ char *tmp = tf_reallocarray_checked(*buf, new_cap, sizeof(char));
298
+ if (!tmp) return -1;
299
+ *buf = tmp;
300
+ *cap = new_cap;
301
+ }
302
+ if (n > 0) memcpy(*buf + *len, data, n);
303
+ *len = need;
304
+ return 0;
305
+ }
306
+
307
+ static void cli_disable_report_buffer(FILE *fout, char **buf, size_t *len, size_t *cap) {
308
+ if (!buf || !len || !cap) return;
309
+ if (fout && *buf && *len > 0) fwrite(*buf, 1, *len, fout);
310
+ free(*buf);
311
+ *buf = NULL;
312
+ *len = 0;
313
+ *cap = 0;
314
+ }
315
+
316
+ static int cli_is_bare_dsl_token(const char *s) {
317
+ if (!s || !*s) return 0;
318
+ for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
319
+ if (isspace(*p) || *p == '|' || *p == '=' || *p == '"' || *p == '\'' || *p == '\\') return 0;
320
+ }
321
+ return 1;
322
+ }
323
+
324
+ static const char *cli_normalized_step_name(const char *op) {
325
+ if (!op) return "unknown";
326
+ if (strcmp(op, "codec.csv.decode") == 0 || strcmp(op, "codec.csv.encode") == 0) return "csv";
327
+ if (strcmp(op, "codec.jsonl.decode") == 0 || strcmp(op, "codec.jsonl.encode") == 0) return "jsonl";
328
+ if (strcmp(op, "codec.text.decode") == 0 || strcmp(op, "codec.text.encode") == 0) return "text";
329
+ if (strcmp(op, "codec.table.encode") == 0) return "table";
330
+ return op;
331
+ }
332
+
333
+ static int cli_append_json_value_token(cli_string_builder *sb, const cJSON *value) {
334
+ if (!value) return sb_append(sb, "null");
335
+ if (cJSON_IsString(value) && value->valuestring && cli_is_bare_dsl_token(value->valuestring)) {
336
+ return sb_append(sb, value->valuestring);
337
+ }
338
+ char *json = cJSON_PrintUnformatted((cJSON *)value);
339
+ if (!json) return -1;
340
+ int rc = sb_append(sb, json);
341
+ free(json);
342
+ return rc;
343
+ }
344
+
345
+ static int cli_append_normalized_arg(cli_string_builder *sb, const cJSON *arg) {
346
+ if (!arg || !arg->string) return 0;
347
+ if (sb_append_char(sb, ' ') != 0) return -1;
348
+ if (sb_append(sb, arg->string) != 0) return -1;
349
+ if (sb_append_char(sb, '=') != 0) return -1;
350
+ return cli_append_json_value_token(sb, arg);
351
+ }
352
+
353
+ static char *cli_ir_to_normalized_dsl(const tf_ir_plan *ir) {
354
+ if (!ir) return NULL;
355
+ cli_string_builder sb = {0};
356
+ for (size_t i = 0; i < ir->n_nodes; i++) {
357
+ const tf_ir_node *node = &ir->nodes[i];
358
+ if (i > 0 && sb_append(&sb, " | ") != 0) goto fail;
359
+ if (sb_append(&sb, cli_normalized_step_name(node->op)) != 0) goto fail;
360
+ if (node->args && cJSON_IsObject(node->args)) {
361
+ for (const cJSON *arg = node->args->child; arg; arg = arg->next) {
362
+ if (arg->string &&
363
+ (strcmp(arg->string, TF_POLICY_VALIDATED_FILE_PATH_ARG) == 0 ||
364
+ strcmp(arg->string, TF_POLICY_VALIDATED_RULES_FILE_PATH_ARG) == 0)) {
365
+ continue;
366
+ }
367
+ if (cli_append_normalized_arg(&sb, arg) != 0) goto fail;
368
+ }
369
+ }
370
+ }
371
+ if (!sb.data) {
372
+ sb.data = strdup("");
373
+ if (!sb.data) return NULL;
374
+ }
375
+ return sb.data;
376
+
377
+ fail:
378
+ free(sb.data);
379
+ return NULL;
380
+ }
381
+
382
+ static void print_caps(FILE *f, uint32_t caps) {
383
+ int first = 1;
384
+ if (caps & TF_CAP_STREAMING) print_cap_item(f, &first, "streaming");
385
+ if (caps & TF_CAP_BOUNDED_MEMORY) print_cap_item(f, &first, "bounded_memory");
386
+ if (caps & TF_CAP_BROWSER_SAFE) print_cap_item(f, &first, "browser_safe");
387
+ if (caps & TF_CAP_DETERMINISTIC) print_cap_item(f, &first, "deterministic");
388
+ if (caps & TF_CAP_FS) print_cap_item(f, &first, "fs");
389
+ if (caps & TF_CAP_NET) print_cap_item(f, &first, "net");
390
+ if (first) fprintf(f, "none");
391
+ }
392
+
393
+ static const tf_ir_node *first_node_with_memory_class(const tf_ir_plan *ir, tf_memory_class cls) {
394
+ for (size_t i = 0; i < ir->n_nodes; i++) {
395
+ if (ir->nodes[i].memory_class == cls) return &ir->nodes[i];
396
+ }
397
+ return NULL;
398
+ }
399
+
400
+ static const tf_ir_node *first_blocking_node(const tf_ir_plan *ir) {
401
+ return first_node_with_memory_class(ir, TF_MEM_BLOCKING);
402
+ }
403
+
404
+ static const tf_ir_node *first_key_state_node(const tf_ir_plan *ir) {
405
+ return first_node_with_memory_class(ir, TF_MEM_KEY_STATE);
406
+ }
407
+
408
+ static int has_sort_head_pattern(const tf_ir_plan *ir) {
409
+ for (size_t i = 0; i + 1 < ir->n_nodes; i++) {
410
+ if (strcmp(ir->nodes[i].op, "sort") == 0 && strcmp(ir->nodes[i + 1].op, "head") == 0)
411
+ return 1;
412
+ }
413
+ return 0;
414
+ }
415
+
416
+ static int node_arg_true(const tf_ir_node *node, const char *name) {
417
+ cJSON *item = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, name) : NULL;
418
+ return cJSON_IsTrue(item);
419
+ }
420
+
421
+ static int node_op_is_unique(const tf_ir_node *node) {
422
+ return node && node->op &&
423
+ (strcmp(node->op, "unique") == 0 || strcmp(node->op, "dedup") == 0);
424
+ }
425
+
426
+ static int node_array_arg_nonempty_main(const tf_ir_node *node, const char *name) {
427
+ cJSON *item = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, name) : NULL;
428
+ return cJSON_IsArray(item) && cJSON_GetArraySize(item) > 0;
429
+ }
430
+
431
+ static int node_positive_arg_main(const tf_ir_node *node, const char *name) {
432
+ cJSON *item = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, name) : NULL;
433
+ return cJSON_IsNumber(item) && item->valuedouble > 0.0;
434
+ }
435
+
436
+ static int node_op_is_pivot_spillable(const tf_ir_node *node) {
437
+ return node && node->op && strcmp(node->op, "pivot") == 0 &&
438
+ !node_arg_true(node, "sorted") &&
439
+ (node_array_arg_nonempty_main(node, "categories") || node_positive_arg_main(node, "max_categories"));
440
+ }
441
+
442
+
443
+ static int node_op_is_join_spillable(const tf_ir_node *node) {
444
+ if (!node || !node->op) return 0;
445
+ if (strcmp(node->op, "semi-join") == 0 || strcmp(node->op, "anti-join") == 0) return 1;
446
+ if (strcmp(node->op, "join") != 0) return 0;
447
+ if (!node->args) return 1;
448
+ cJSON *how = cJSON_GetObjectItemCaseSensitive(node->args, "how");
449
+ if (!cJSON_IsString(how) || !how->valuestring) return 1;
450
+ return strcmp(how->valuestring, "inner") == 0 || strcmp(how->valuestring, "left") == 0 ||
451
+ strcmp(how->valuestring, "semi") == 0 || strcmp(how->valuestring, "anti") == 0;
452
+ }
453
+
454
+ static int node_op_is_row_set_spillable(const tf_ir_node *node) {
455
+ return node && node->op &&
456
+ (strcmp(node->op, "intersect") == 0 || strcmp(node->op, "setdiff") == 0 ||
457
+ strcmp(node->op, "intersect-all") == 0 || strcmp(node->op, "setdiff-all") == 0 ||
458
+ strcmp(node->op, "union") == 0);
459
+ }
460
+
461
+ static int blocking_node_has_native_spill(const tf_ir_node *node) {
462
+ return node && node->op && (strcmp(node->op, "sort") == 0 || node_op_is_pivot_spillable(node));
463
+ }
464
+
465
+ static int key_state_node_has_native_spill(const tf_ir_node *node) {
466
+ return ((node_op_is_unique(node) ||
467
+ (node && node->op && strcmp(node->op, "group-agg") == 0) ||
468
+ node_op_is_join_spillable(node) ||
469
+ node_op_is_row_set_spillable(node)) &&
470
+ !node_arg_true(node, "sorted"));
471
+ }
472
+
473
+ static const tf_ir_node *first_key_state_node_without_native_spill(const tf_ir_plan *ir) {
474
+ for (size_t i = 0; i < ir->n_nodes; i++) {
475
+ const tf_ir_node *node = &ir->nodes[i];
476
+ if (node->memory_class == TF_MEM_KEY_STATE && !key_state_node_has_native_spill(node))
477
+ return node;
478
+ }
479
+ return NULL;
480
+ }
481
+
482
+ static int node_uses_native_spill(const tf_ir_node *node) {
483
+ return blocking_node_has_native_spill(node) || key_state_node_has_native_spill(node);
484
+ }
485
+
486
+ static const char *explain_step_target(const tf_ir_node *node, const cli_memory_policy *policy) {
487
+ if (policy->engine == CLI_ENGINE_DUCKDB) return "duckdb_sql";
488
+ if (policy->engine == CLI_ENGINE_NATIVE && policy->spill_dir && node_uses_native_spill(node))
489
+ return "native_spill";
490
+ return "native";
491
+ }
492
+
493
+ static const tf_ir_node *first_blocking_node_without_native_spill(const tf_ir_plan *ir) {
494
+ for (size_t i = 0; i < ir->n_nodes; i++) {
495
+ const tf_ir_node *node = &ir->nodes[i];
496
+ if (node->memory_class == TF_MEM_BLOCKING && !blocking_node_has_native_spill(node))
497
+ return node;
498
+ }
499
+ return NULL;
500
+ }
501
+
502
+ static int blocking_plan_can_use_native_spill(const tf_ir_plan *ir) {
503
+ return first_blocking_node(ir) && !first_blocking_node_without_native_spill(ir);
504
+ }
505
+
506
+ static const char *native_spill_support_summary(void) {
507
+ return "native spill currently supports sort, capped unsorted pivot "
508
+ "(categories=... or max_categories=N), unsorted unique/dedup, "
509
+ "unsorted group-agg, capped unsorted inner/left joins, unsorted "
510
+ "semi/anti filtering joins, unsorted intersect/setdiff/"
511
+ "intersect-all/setdiff-all, and duplicate-eliminating union";
512
+ }
513
+
514
+ static int set_json_item(cJSON *obj, const char *name, cJSON *item) {
515
+ if (!obj || !name || !item) {
516
+ cJSON_Delete(item);
517
+ return -1;
518
+ }
519
+ if (cJSON_GetObjectItemCaseSensitive(obj, name)) {
520
+ if (cJSON_ReplaceItemInObjectCaseSensitive(obj, name, item)) return 0;
521
+ cJSON_Delete(item);
522
+ return -1;
523
+ }
524
+ if (tf_json_add_item(obj, name, item) != TF_OK) {
525
+ cJSON_Delete(item);
526
+ return -1;
527
+ }
528
+ return 0;
529
+ }
530
+
531
+ static int set_json_string(cJSON *obj, const char *name, const char *value) {
532
+ return set_json_item(obj, name, cJSON_CreateString(value ? value : ""));
533
+ }
534
+
535
+ static int set_json_number(cJSON *obj, const char *name, double value) {
536
+ return set_json_item(obj, name, cJSON_CreateNumber(value));
537
+ }
538
+
539
+ static int apply_native_spill_policy(tf_ir_plan *ir, const cli_memory_policy *policy) {
540
+ if (!policy->spill_dir || policy->engine != CLI_ENGINE_NATIVE) return 0;
541
+ for (size_t i = 0; i < ir->n_nodes; i++) {
542
+ tf_ir_node *node = &ir->nodes[i];
543
+ if (!node_uses_native_spill(node)) continue;
544
+ if (!node->args) {
545
+ node->args = cJSON_CreateObject();
546
+ if (!node->args) return -1;
547
+ }
548
+ if (set_json_string(node->args, "spill_dir", policy->spill_dir) != 0) return -1;
549
+ if (policy->has_memory_limit &&
550
+ set_json_number(node->args, "spill_memory_bytes", (double)policy->memory_limit_bytes) != 0)
551
+ return -1;
552
+ }
553
+ return 0;
554
+ }
555
+
556
+ static int validate_memory_policy(const tf_ir_plan *ir, const cli_memory_policy *policy) {
557
+ const tf_ir_node *blocking = first_blocking_node(ir);
558
+ const tf_ir_node *key_state = first_key_state_node(ir);
559
+
560
+ if (policy->engine == CLI_ENGINE_DUCKDB) return 0;
561
+
562
+ if (policy->spill_dir) {
563
+ const tf_ir_node *unsupported_key = first_key_state_node_without_native_spill(ir);
564
+ if (unsupported_key) {
565
+ fprintf(stderr,
566
+ "error: --spill-dir was requested, but native spill is not implemented yet for key-state step '%s'\n",
567
+ unsupported_key->op);
568
+ fprintf(stderr,
569
+ "hint: %s; use op-specific caps or an external engine for this plan\n",
570
+ native_spill_support_summary());
571
+ return 1;
572
+ }
573
+ const tf_ir_node *unsupported = first_blocking_node_without_native_spill(ir);
574
+ if (unsupported) {
575
+ fprintf(stderr,
576
+ "error: --spill-dir was requested, but native spill is unavailable for blocking step '%s' with the current arguments\n",
577
+ unsupported->op);
578
+ fprintf(stderr,
579
+ "hint: %s; use --allow-blocking only for known-small data or an external engine for this plan\n",
580
+ native_spill_support_summary());
581
+ return 1;
582
+ }
583
+ }
584
+
585
+ if (policy->has_memory_limit && blocking &&
586
+ !(policy->spill_dir && blocking_plan_can_use_native_spill(ir))) {
587
+ char cap[32];
588
+ fprintf(stderr,
589
+ "error: --memory %s was requested, but native byte caps are not implemented for blocking step '%s' (state=%s)\n",
590
+ format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)),
591
+ blocking->op,
592
+ blocking->state_estimate ? blocking->state_estimate : "unknown");
593
+ fprintf(stderr,
594
+ "hint: rewrite to a bounded op, choose an external engine, or use --spill-dir when the step has a native spill mode and required caps\n");
595
+ return 1;
596
+ }
597
+
598
+ if (policy->has_memory_limit && key_state) {
599
+ size_t estimated = 0;
600
+ const tf_ir_node *failed = NULL;
601
+ char reason[192] = {0};
602
+ if (!tf_estimate_key_state_plan_bytes(ir, &estimated, &failed, reason, sizeof(reason))) {
603
+ char cap[32];
604
+ fprintf(stderr,
605
+ "error: --memory %s was requested, but %s\n",
606
+ format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)),
607
+ reason[0] ? reason : "a key-state step is not byte-bounded");
608
+ if (failed) {
609
+ fprintf(stderr,
610
+ "hint: add the op-specific cap for step '%s' or use an external/spill engine\n",
611
+ failed->op);
612
+ }
613
+ return 1;
614
+ }
615
+ if (estimated > policy->memory_limit_bytes) {
616
+ char cap[32], est[32];
617
+ fprintf(stderr,
618
+ "error: estimated native key-state memory %s exceeds --memory %s\n",
619
+ format_bytes(estimated, est, sizeof(est)),
620
+ format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)));
621
+ fprintf(stderr,
622
+ "hint: lower key/category/lookup caps or raise --memory for this known-bounded plan\n");
623
+ return 1;
624
+ }
625
+ }
626
+
627
+ if (blocking && !policy->allow_blocking &&
628
+ !(policy->spill_dir && blocking_plan_can_use_native_spill(ir))) {
629
+ fprintf(stderr,
630
+ "error: blocking step '%s' requires full input in native mode (state=%s)\n",
631
+ blocking->op,
632
+ blocking->state_estimate ? blocking->state_estimate : "unknown");
633
+ fprintf(stderr,
634
+ "hint: add --allow-blocking only for known-small inputs, rewrite to a bounded op, or choose an external engine/spill mode when available\n");
635
+ if (has_sort_head_pattern(ir)) {
636
+ fprintf(stderr,
637
+ "hint: replace sort | head N with top N column when top-k semantics are acceptable\n");
638
+ }
639
+ return 1;
640
+ }
641
+
642
+ return 0;
643
+ }
644
+
645
+ static void print_explain(const tf_ir_plan *ir, const cli_memory_policy *policy) {
646
+ tf_memory_class worst_mem = TF_MEM_ROW_LOCAL;
647
+ const char *worst_state = "O(batch_rows * columns)";
648
+ int has_flush = 0;
649
+ int has_per_batch = 0;
650
+ int has_data_schema = 0;
651
+
652
+ for (size_t i = 0; i < ir->n_nodes; i++) {
653
+ const tf_ir_node *node = &ir->nodes[i];
654
+ if (memory_rank(node->memory_class) > memory_rank(worst_mem)) {
655
+ worst_mem = node->memory_class;
656
+ worst_state = node->state_estimate ? node->state_estimate : "unknown";
657
+ }
658
+ if (node->emit_class == TF_EMIT_ON_FLUSH) has_flush = 1;
659
+ if (node->emit_class == TF_EMIT_PER_BATCH) has_per_batch = 1;
660
+ if (node->schema_class == TF_SCHEMA_DATA_DEPENDENT) has_data_schema = 1;
661
+ }
662
+
663
+ char cap[32];
664
+ printf("Tranfi execution plan\n");
665
+ char *normalized_dsl = cli_ir_to_normalized_dsl(ir);
666
+ printf("normalized_dsl: %s\n", normalized_dsl ? normalized_dsl : "null");
667
+ free(normalized_dsl);
668
+ printf("execution_target: %s\n", policy->spill_dir && policy->engine == CLI_ENGINE_NATIVE ? "native+spill" : engine_name(policy->engine));
669
+ printf("memory_policy: %s\n", policy->spill_dir ? "spill" : (policy->allow_blocking ? "allow_blocking" : "strict"));
670
+ printf("memory_limit: %s\n", policy->has_memory_limit ? format_bytes(policy->memory_limit_bytes, cap, sizeof(cap)) : "none");
671
+ printf("spill_dir: %s\n", policy->spill_dir ? policy->spill_dir : "none");
672
+ printf("memory_class: %s\n", tf_memory_class_name(worst_mem));
673
+ printf("emit_class: %s\n", has_flush && has_per_batch ? "mixed" :
674
+ (has_flush ? "on_flush" : "per_batch"));
675
+ printf("schema_class: %s\n", has_data_schema ? "data_dependent" : "stable_or_parametric");
676
+ printf("state_estimate: %s\n", worst_state);
677
+ if (first_key_state_node(ir)) {
678
+ size_t key_bytes = 0;
679
+ const tf_ir_node *failed = NULL;
680
+ char reason[192] = {0};
681
+ if (tf_estimate_key_state_plan_bytes(ir, &key_bytes, &failed, reason, sizeof(reason))) {
682
+ char key_est[32];
683
+ printf("state_bytes_estimate: %s\n", format_bytes(key_bytes, key_est, sizeof(key_est)));
684
+ } else {
685
+ printf("state_bytes_estimate: unbounded (%s)\n", reason[0] ? reason : "missing key-state byte estimator");
686
+ }
687
+ } else {
688
+ printf("state_bytes_estimate: none\n");
689
+ }
690
+ if (first_blocking_node(ir)) {
691
+ if (policy->spill_dir && policy->engine == CLI_ENGINE_NATIVE && blocking_plan_can_use_native_spill(ir))
692
+ printf("note: blocking step will use native spill files\n");
693
+ else if (policy->engine == CLI_ENGINE_NATIVE && !policy->allow_blocking)
694
+ printf("warning: blocking native step present; default execution will reject it unless --allow-blocking is set\n");
695
+ else
696
+ printf("warning: blocking step present; selected policy must provide a bounded/external execution path\n");
697
+ if (has_sort_head_pattern(ir))
698
+ printf("hint: replace sort | head N with top N column when top-k semantics are acceptable\n");
699
+ }
700
+ printf("steps:\n");
701
+ for (size_t i = 0; i < ir->n_nodes; i++) {
702
+ const tf_ir_node *node = &ir->nodes[i];
703
+ printf(" %zu. %s\n", i, node->op);
704
+ printf(" target=%s memory=%s emit=%s schema=%s state=%s caps=",
705
+ explain_step_target(node, policy),
706
+ tf_memory_class_name(node->memory_class),
707
+ tf_emit_class_name(node->emit_class),
708
+ tf_schema_class_name(node->schema_class),
709
+ node->state_estimate ? node->state_estimate : "unknown");
710
+ print_caps(stdout, node->caps);
711
+ printf("\n");
712
+ if (node->memory_class == TF_MEM_KEY_STATE) {
713
+ size_t step_bytes = 0;
714
+ char reason[192] = {0};
715
+ if (tf_estimate_step_state_bytes(node, &step_bytes, reason, sizeof(reason))) {
716
+ char step_est[32];
717
+ printf(" state_bytes_estimate=%s\n", format_bytes(step_bytes, step_est, sizeof(step_est)));
718
+ } else {
719
+ printf(" state_bytes_estimate=unbounded (%s)\n", reason[0] ? reason : "missing key-state byte estimator");
720
+ }
721
+ }
722
+ }
723
+
724
+ char *ir_json = tf_ir_to_json(ir);
725
+ if (ir_json) {
726
+ printf("ir_json: %s\n", ir_json);
727
+ free(ir_json);
728
+ } else {
729
+ printf("ir_json: null\n");
730
+ }
731
+ }
732
+
101
733
  int main(int argc, char **argv) {
102
734
  const char *pipeline_file = NULL;
103
735
  const char *pipeline_text = NULL;
104
736
  const char *input_file = NULL;
105
737
  const char *output_file = NULL;
106
738
  int json_mode = 0;
739
+ int sql_mode = 0;
740
+ int explain_mode = 0;
741
+ const char *sql_dialect = "duckdb";
742
+ cli_memory_policy policy = {0};
743
+ policy.engine = CLI_ENGINE_NATIVE;
107
744
  int quiet = 0;
108
745
  int progress = 0;
109
746
  int raw_stats = 0;
747
+ const char *stats_json_file = NULL;
110
748
 
111
749
  /* Parse options */
112
750
  int argi = 1;
@@ -129,12 +767,126 @@ int main(int argc, char **argv) {
129
767
  return 0;
130
768
  } else if (strcmp(opt, "-j") == 0) {
131
769
  json_mode = 1;
770
+ sql_mode = 0;
771
+ } else if (strcmp(opt, "--target") == 0 || strncmp(opt, "--target=", 9) == 0) {
772
+ const char *value = NULL;
773
+ if (strncmp(opt, "--target=", 9) == 0) {
774
+ value = opt + 9;
775
+ } else {
776
+ argi++;
777
+ if (argi >= argc) {
778
+ fprintf(stderr, "error: --target requires native, json, or sql\n");
779
+ return 1;
780
+ }
781
+ value = argv[argi];
782
+ }
783
+ if (strcmp(value, "native") == 0) {
784
+ json_mode = 0;
785
+ sql_mode = 0;
786
+ } else if (strcmp(value, "json") == 0 || strcmp(value, "ir") == 0) {
787
+ json_mode = 1;
788
+ sql_mode = 0;
789
+ } else if (strcmp(value, "sql") == 0) {
790
+ json_mode = 0;
791
+ sql_mode = 1;
792
+ } else {
793
+ fprintf(stderr, "error: unknown --target '%s' (expected native, json, or sql)\n", value);
794
+ return 1;
795
+ }
796
+ } else if (strcmp(opt, "--dialect") == 0 || strncmp(opt, "--dialect=", 10) == 0) {
797
+ const char *value = NULL;
798
+ if (strncmp(opt, "--dialect=", 10) == 0) {
799
+ value = opt + 10;
800
+ } else {
801
+ argi++;
802
+ if (argi >= argc) {
803
+ fprintf(stderr, "error: --dialect requires duckdb, sqlite, or postgres\n");
804
+ return 1;
805
+ }
806
+ value = argv[argi];
807
+ }
808
+ if (!is_known_sql_dialect(value)) {
809
+ fprintf(stderr, "error: unknown SQL dialect '%s' (expected duckdb, sqlite, or postgres)\n", value);
810
+ return 1;
811
+ }
812
+ sql_dialect = value;
813
+ } else if (strcmp(opt, "--explain") == 0) {
814
+ explain_mode = 1;
815
+ } else if (strcmp(opt, "--allow-blocking") == 0) {
816
+ policy.allow_blocking = 1;
817
+ } else if (strcmp(opt, "--fail-on-blocking") == 0) {
818
+ policy.fail_on_blocking_seen = 1;
819
+ } else if (strcmp(opt, "--memory") == 0 || strncmp(opt, "--memory=", 9) == 0) {
820
+ const char *value = NULL;
821
+ if (strncmp(opt, "--memory=", 9) == 0) {
822
+ value = opt + 9;
823
+ } else {
824
+ argi++;
825
+ if (argi >= argc) {
826
+ fprintf(stderr, "error: --memory requires a size such as max:64MB\n");
827
+ return 1;
828
+ }
829
+ value = argv[argi];
830
+ }
831
+ char err[160];
832
+ if (parse_size_bytes(value, &policy.memory_limit_bytes, err, sizeof(err)) != 0) {
833
+ fprintf(stderr, "error: invalid --memory value: %s\n", err);
834
+ return 1;
835
+ }
836
+ policy.has_memory_limit = 1;
837
+ } else if (strcmp(opt, "--spill-dir") == 0 || strncmp(opt, "--spill-dir=", 12) == 0) {
838
+ if (strncmp(opt, "--spill-dir=", 12) == 0) {
839
+ policy.spill_dir = opt + 12;
840
+ } else {
841
+ argi++;
842
+ if (argi >= argc) {
843
+ fprintf(stderr, "error: --spill-dir requires a directory argument\n");
844
+ return 1;
845
+ }
846
+ policy.spill_dir = argv[argi];
847
+ }
848
+ if (!policy.spill_dir || policy.spill_dir[0] == '\0') {
849
+ fprintf(stderr, "error: --spill-dir requires a non-empty directory\n");
850
+ return 1;
851
+ }
852
+ } else if (strcmp(opt, "--engine") == 0 || strncmp(opt, "--engine=", 9) == 0) {
853
+ const char *value = NULL;
854
+ if (strncmp(opt, "--engine=", 9) == 0) {
855
+ value = opt + 9;
856
+ } else {
857
+ argi++;
858
+ if (argi >= argc) {
859
+ fprintf(stderr, "error: --engine requires native or duckdb\n");
860
+ return 1;
861
+ }
862
+ value = argv[argi];
863
+ }
864
+ char err[160];
865
+ if (set_engine(&policy, value, err, sizeof(err)) != 0) {
866
+ fprintf(stderr, "error: %s\n", err);
867
+ return 1;
868
+ }
132
869
  } else if (strcmp(opt, "-q") == 0) {
133
870
  quiet = 1;
134
871
  } else if (strcmp(opt, "-p") == 0 || strcmp(opt, "--progress") == 0) {
135
872
  progress = 1;
136
873
  } else if (strcmp(opt, "--raw") == 0) {
137
874
  raw_stats = 1;
875
+ } else if (strcmp(opt, "--stats-json") == 0 || strncmp(opt, "--stats-json=", 13) == 0) {
876
+ if (strncmp(opt, "--stats-json=", 13) == 0) {
877
+ stats_json_file = opt + 13;
878
+ } else {
879
+ argi++;
880
+ if (argi >= argc) {
881
+ fprintf(stderr, "error: --stats-json requires a file path, or - for stderr\n");
882
+ return 1;
883
+ }
884
+ stats_json_file = argv[argi];
885
+ }
886
+ if (!stats_json_file || stats_json_file[0] == '\0') {
887
+ fprintf(stderr, "error: --stats-json requires a non-empty file path, or - for stderr\n");
888
+ return 1;
889
+ }
138
890
  } else if (strcmp(opt, "-f") == 0) {
139
891
  argi++;
140
892
  if (argi >= argc) {
@@ -163,6 +915,14 @@ int main(int argc, char **argv) {
163
915
  argi++;
164
916
  }
165
917
 
918
+ if (policy.allow_blocking && policy.fail_on_blocking_seen) {
919
+ fprintf(stderr, "error: --allow-blocking conflicts with --fail-on-blocking\n");
920
+ return 1;
921
+ }
922
+ if (sql_mode) {
923
+ policy.engine = CLI_ENGINE_DUCKDB;
924
+ }
925
+
166
926
  /* Get pipeline text */
167
927
  char *file_content = NULL;
168
928
  if (pipeline_file) {
@@ -218,8 +978,58 @@ int main(int argc, char **argv) {
218
978
  return 1;
219
979
  }
220
980
 
221
- /* Schema inference (best-effort) */
222
- tf_ir_infer_schema(ir);
981
+ /* Schema inference allows unknown runtime schemas but reports invalid known-schema plans. */
982
+ tf_set_last_error(NULL);
983
+ if (tf_ir_infer_schema(ir) != TF_OK) {
984
+ const char *detail = tf_last_error();
985
+ if (detail && detail[0])
986
+ fprintf(stderr, "error: schema inference failed: %s\n", detail);
987
+ else
988
+ fprintf(stderr, "error: schema inference failed\n");
989
+ tf_ir_plan_free(ir);
990
+ return 1;
991
+ }
992
+
993
+ if (apply_native_spill_policy(ir, &policy) != 0) {
994
+ fprintf(stderr, "error: failed to apply native spill policy\n");
995
+ tf_ir_plan_free(ir);
996
+ return 1;
997
+ }
998
+
999
+ if (tf_ir_validate(ir) != TF_OK) {
1000
+ fprintf(stderr, "error: %s\n", ir->error ? ir->error : "validation failed after applying native spill policy");
1001
+ tf_ir_plan_free(ir);
1002
+ return 1;
1003
+ }
1004
+
1005
+ if (explain_mode) {
1006
+ print_explain(ir, &policy);
1007
+ tf_ir_plan_free(ir);
1008
+ return 0;
1009
+ }
1010
+
1011
+ if (sql_mode) {
1012
+ if (!sql_dialect_supported(sql_dialect)) {
1013
+ fprintf(stderr,
1014
+ "error: SQL dialect '%s' is recognized but not implemented yet; only duckdb lowering is available\n",
1015
+ sql_dialect);
1016
+ tf_ir_plan_free(ir);
1017
+ return 1;
1018
+ }
1019
+ char *sql_error = NULL;
1020
+ char *sql = tf_ir_plan_to_sql(ir, &sql_error);
1021
+ if (!sql) {
1022
+ fprintf(stderr, "error: SQL compile failed for dialect '%s': %s\n",
1023
+ sql_dialect, sql_error ? sql_error : "unknown error");
1024
+ free(sql_error);
1025
+ tf_ir_plan_free(ir);
1026
+ return 1;
1027
+ }
1028
+ printf("%s\n", sql);
1029
+ tf_string_free(sql);
1030
+ tf_ir_plan_free(ir);
1031
+ return 0;
1032
+ }
223
1033
 
224
1034
  /* JSON mode: print IR and exit */
225
1035
  if (json_mode) {
@@ -232,6 +1042,18 @@ int main(int argc, char **argv) {
232
1042
  return 0;
233
1043
  }
234
1044
 
1045
+ if (validate_memory_policy(ir, &policy) != 0) {
1046
+ tf_ir_plan_free(ir);
1047
+ return 1;
1048
+ }
1049
+
1050
+ if (policy.engine == CLI_ENGINE_DUCKDB) {
1051
+ fprintf(stderr,
1052
+ "error: the C CLI cannot execute --engine duckdb directly; use Python/Node with engine='duckdb' or compile SQL with --target sql --dialect duckdb\n");
1053
+ tf_ir_plan_free(ir);
1054
+ return 1;
1055
+ }
1056
+
235
1057
  /* Compile to native pipeline */
236
1058
  tf_pipeline *p = tf_pipeline_create_from_ir(ir);
237
1059
  tf_ir_plan_free(ir);
@@ -245,6 +1067,8 @@ int main(int argc, char **argv) {
245
1067
  /* Open I/O files */
246
1068
  FILE *fin = stdin;
247
1069
  FILE *fout = stdout;
1070
+ FILE *fstats = NULL;
1071
+ int close_stats = 0;
248
1072
 
249
1073
  if (input_file) {
250
1074
  fin = fopen(input_file, "rb");
@@ -265,11 +1089,37 @@ int main(int argc, char **argv) {
265
1089
  }
266
1090
  }
267
1091
 
1092
+ if (stats_json_file) {
1093
+ if (strcmp(stats_json_file, "-") == 0) {
1094
+ fstats = stderr;
1095
+ } else {
1096
+ fstats = fopen(stats_json_file, "wb");
1097
+ if (!fstats) {
1098
+ fprintf(stderr, "error: cannot open stats JSON file '%s'\n", stats_json_file);
1099
+ if (fin != stdin) fclose(fin);
1100
+ if (fout != stdout) fclose(fout);
1101
+ tf_pipeline_free(p);
1102
+ return 1;
1103
+ }
1104
+ close_stats = 1;
1105
+ }
1106
+ }
1107
+
268
1108
  /* Decide whether to buffer output for report formatting.
269
1109
  * When stdout is a TTY and --raw is not set, buffer main output
270
1110
  * and try to render it as a rich report. Falls back to raw CSV
271
1111
  * if the output doesn't look like a stats table. */
272
1112
  int try_report = !raw_stats && !output_file && isatty(STDOUT_FILENO);
1113
+ if (!try_report) {
1114
+ if (tf_pipeline_set_sink(p, TF_CHAN_MAIN, cli_file_output_sink, fout) != TF_OK) {
1115
+ fprintf(stderr, "error: failed to attach output sink\n");
1116
+ if (fin != stdin) fclose(fin);
1117
+ if (fout != stdout) fclose(fout);
1118
+ if (close_stats && fstats) fclose(fstats);
1119
+ tf_pipeline_free(p);
1120
+ return 1;
1121
+ }
1122
+ }
273
1123
 
274
1124
  /* Stream input → pipeline → output */
275
1125
  uint8_t read_buf[READ_BUF_SIZE];
@@ -279,7 +1129,11 @@ int main(int argc, char **argv) {
279
1129
  /* Output buffer (used when try_report is true) */
280
1130
  size_t out_cap = PULL_BUF_SIZE;
281
1131
  size_t out_len = 0;
282
- char *out_buf = try_report ? malloc(out_cap) : NULL;
1132
+ char *out_buf = try_report ? tf_mallocarray_checked(out_cap, sizeof(char)) : NULL;
1133
+ if (try_report && !out_buf) {
1134
+ out_cap = 0;
1135
+ try_report = 0;
1136
+ }
283
1137
 
284
1138
  while ((nread = fread(read_buf, 1, sizeof(read_buf), fin)) > 0) {
285
1139
  if (tf_pipeline_push(p, read_buf, nread) != TF_OK) {
@@ -288,6 +1142,7 @@ int main(int argc, char **argv) {
288
1142
  free(out_buf);
289
1143
  if (fin != stdin) fclose(fin);
290
1144
  if (fout != stdout) fclose(fout);
1145
+ if (close_stats && fstats) fclose(fstats);
291
1146
  tf_pipeline_free(p);
292
1147
  return 1;
293
1148
  }
@@ -299,15 +1154,10 @@ int main(int argc, char **argv) {
299
1154
  size_t n;
300
1155
  while ((n = tf_pipeline_pull(p, TF_CHAN_MAIN, pull_buf, sizeof(pull_buf))) > 0) {
301
1156
  if (try_report && out_buf) {
302
- while (out_len + n > out_cap) {
303
- out_cap *= 2;
304
- char *tmp = realloc(out_buf, out_cap);
305
- if (!tmp) { free(out_buf); out_buf = NULL; break; }
306
- out_buf = tmp;
307
- }
308
- if (out_buf) {
309
- memcpy(out_buf + out_len, pull_buf, n);
310
- out_len += n;
1157
+ if (cli_report_buffer_append(&out_buf, &out_len, &out_cap, pull_buf, n) != 0) {
1158
+ cli_disable_report_buffer(fout, &out_buf, &out_len, &out_cap);
1159
+ fwrite(pull_buf, 1, n, fout);
1160
+ try_report = 0;
311
1161
  }
312
1162
  } else {
313
1163
  fwrite(pull_buf, 1, n, fout);
@@ -329,6 +1179,7 @@ int main(int argc, char **argv) {
329
1179
  free(out_buf);
330
1180
  if (fin != stdin) fclose(fin);
331
1181
  if (fout != stdout) fclose(fout);
1182
+ if (close_stats && fstats) fclose(fstats);
332
1183
  tf_pipeline_free(p);
333
1184
  return 1;
334
1185
  }
@@ -338,15 +1189,10 @@ int main(int argc, char **argv) {
338
1189
  size_t n;
339
1190
  while ((n = tf_pipeline_pull(p, TF_CHAN_MAIN, pull_buf, sizeof(pull_buf))) > 0) {
340
1191
  if (try_report && out_buf) {
341
- while (out_len + n > out_cap) {
342
- out_cap *= 2;
343
- char *tmp = realloc(out_buf, out_cap);
344
- if (!tmp) { free(out_buf); out_buf = NULL; break; }
345
- out_buf = tmp;
346
- }
347
- if (out_buf) {
348
- memcpy(out_buf + out_len, pull_buf, n);
349
- out_len += n;
1192
+ if (cli_report_buffer_append(&out_buf, &out_len, &out_cap, pull_buf, n) != 0) {
1193
+ cli_disable_report_buffer(fout, &out_buf, &out_len, &out_cap);
1194
+ fwrite(pull_buf, 1, n, fout);
1195
+ try_report = 0;
350
1196
  }
351
1197
  } else {
352
1198
  fwrite(pull_buf, 1, n, fout);
@@ -377,16 +1223,19 @@ int main(int argc, char **argv) {
377
1223
  fwrite(pull_buf, 1, n, stderr);
378
1224
  }
379
1225
 
380
- /* Pull stats to stderr (unless quiet) */
381
- if (!quiet) {
1226
+ /* Pull stats to the requested machine-readable path, or stderr unless quiet. */
1227
+ FILE *stats_out = stats_json_file ? fstats : (!quiet ? stderr : NULL);
1228
+ if (stats_out) {
382
1229
  while ((n = tf_pipeline_pull(p, TF_CHAN_STATS, pull_buf, sizeof(pull_buf))) > 0) {
383
- fwrite(pull_buf, 1, n, stderr);
1230
+ fwrite(pull_buf, 1, n, stats_out);
384
1231
  }
1232
+ fflush(stats_out);
385
1233
  }
386
1234
 
387
1235
  /* Cleanup */
388
1236
  if (fin != stdin) fclose(fin);
389
1237
  if (fout != stdout) fclose(fout);
1238
+ if (close_stats && fstats) fclose(fstats);
390
1239
  tf_pipeline_free(p);
391
1240
  return 0;
392
1241
  }