tranfi 0.0.1 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/README.md +395 -0
  2. package/app/assets/index-6quYZ5Ap.css +5 -0
  3. package/app/assets/index-pDFMluyz.js +160 -0
  4. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  5. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  6. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  7. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  8. package/app/index.html +13 -0
  9. package/binding.gyp +69 -0
  10. package/csrc/arena.c +91 -0
  11. package/csrc/batch.c +229 -0
  12. package/csrc/buffer.c +78 -0
  13. package/csrc/cJSON.c +3143 -0
  14. package/csrc/cJSON.h +300 -0
  15. package/csrc/codec_csv.c +1058 -0
  16. package/csrc/codec_jsonl.c +374 -0
  17. package/csrc/codec_table.c +218 -0
  18. package/csrc/codec_text.c +229 -0
  19. package/csrc/compiler.c +102 -0
  20. package/csrc/date_utils.h +94 -0
  21. package/csrc/dsl.c +1180 -0
  22. package/csrc/dsl.h +22 -0
  23. package/csrc/expr.c +1245 -0
  24. package/csrc/expr.h +56 -0
  25. package/csrc/internal.h +250 -0
  26. package/csrc/ir.c +119 -0
  27. package/csrc/ir.h +167 -0
  28. package/csrc/ir_schema.c +60 -0
  29. package/csrc/ir_serialize.c +104 -0
  30. package/csrc/ir_sql.c +1211 -0
  31. package/csrc/ir_validate.c +120 -0
  32. package/csrc/main.c +392 -0
  33. package/csrc/op_acf.c +133 -0
  34. package/csrc/op_anomaly.c +120 -0
  35. package/csrc/op_bin.c +109 -0
  36. package/csrc/op_cast.c +195 -0
  37. package/csrc/op_clip.c +88 -0
  38. package/csrc/op_date_trunc.c +181 -0
  39. package/csrc/op_datetime.c +212 -0
  40. package/csrc/op_derive.c +248 -0
  41. package/csrc/op_diff.c +134 -0
  42. package/csrc/op_ewma.c +103 -0
  43. package/csrc/op_explode.c +108 -0
  44. package/csrc/op_fill_down.c +163 -0
  45. package/csrc/op_fill_null.c +123 -0
  46. package/csrc/op_filter.c +132 -0
  47. package/csrc/op_frequency.c +193 -0
  48. package/csrc/op_grep.c +163 -0
  49. package/csrc/op_group_agg.c +285 -0
  50. package/csrc/op_hash.c +126 -0
  51. package/csrc/op_head.c +149 -0
  52. package/csrc/op_interpolate.c +239 -0
  53. package/csrc/op_join.c +384 -0
  54. package/csrc/op_label_encode.c +144 -0
  55. package/csrc/op_lead.c +190 -0
  56. package/csrc/op_normalize.c +226 -0
  57. package/csrc/op_onehot.c +185 -0
  58. package/csrc/op_pivot.c +370 -0
  59. package/csrc/op_registry.c +1148 -0
  60. package/csrc/op_rename.c +138 -0
  61. package/csrc/op_replace.c +202 -0
  62. package/csrc/op_sample.c +101 -0
  63. package/csrc/op_select.c +140 -0
  64. package/csrc/op_skip.c +152 -0
  65. package/csrc/op_sort.c +273 -0
  66. package/csrc/op_split.c +114 -0
  67. package/csrc/op_split_data.c +87 -0
  68. package/csrc/op_stack.c +315 -0
  69. package/csrc/op_stats.c +779 -0
  70. package/csrc/op_step.c +171 -0
  71. package/csrc/op_tail.c +96 -0
  72. package/csrc/op_top.c +150 -0
  73. package/csrc/op_trim.c +109 -0
  74. package/csrc/op_unique.c +300 -0
  75. package/csrc/op_unpivot.c +159 -0
  76. package/csrc/op_validate.c +71 -0
  77. package/csrc/op_window.c +150 -0
  78. package/csrc/pipeline.c +315 -0
  79. package/csrc/plan.c +206 -0
  80. package/csrc/recipes.c +102 -0
  81. package/csrc/recipes.h +27 -0
  82. package/csrc/report.c +463 -0
  83. package/csrc/report.h +22 -0
  84. package/csrc/tranfi.h +123 -0
  85. package/csrc/wasm_api.c +157 -0
  86. package/napi_api.c +326 -0
  87. package/package.json +46 -57
  88. package/src/cli.js +193 -0
  89. package/src/engines/duckdb.js +109 -0
  90. package/src/index.js +306 -0
  91. package/src/native.js +22 -0
  92. package/src/pipeline.js +286 -0
  93. package/src/server.js +277 -0
  94. package/src/wasm.js +19 -0
  95. package/wasm/index.js +244 -0
  96. package/wasm/package.json +1 -0
  97. package/wasm/tranfi_core.js +0 -0
  98. package/LICENSE +0 -21
  99. package/dist/bundle.js +0 -1
  100. package/index.html +0 -18
  101. package/logo.png +0 -0
  102. package/src/app.css +0 -160
  103. package/src/app.js +0 -203
  104. package/src/app.vue +0 -253
  105. package/src/bulma-input.vue +0 -110
  106. package/src/common-inputs.js +0 -28
  107. package/src/main.js +0 -18
  108. package/src/transforms.js +0 -166
  109. package/webpack.config.js +0 -108
package/csrc/dsl.c ADDED
@@ -0,0 +1,1180 @@
1
+ /*
2
+ * dsl.c — Pipe-style DSL parser (L3 → L2 IR).
3
+ *
4
+ * Grammar:
5
+ * pipeline = stage ( '|' stage )*
6
+ * stage = op_name arg*
7
+ * arg = quoted_string | key=value | bare_word
8
+ *
9
+ * Positional codec resolution:
10
+ * "csv" at first position → "codec.csv.decode"
11
+ * "csv" at last position → "codec.csv.encode"
12
+ * "jsonl" at first position → "codec.jsonl.decode"
13
+ * "jsonl" at last position → "codec.jsonl.encode"
14
+ *
15
+ * Explicit forms ("csv.decode", "csv.encode", etc.) always work.
16
+ */
17
+
18
+ #include "dsl.h"
19
+ #include "ir.h"
20
+ #include "cJSON.h"
21
+ #include <stdlib.h>
22
+ #include <string.h>
23
+ #include <stdio.h>
24
+ #include <ctype.h>
25
+
26
+ #define TF_OK 0
27
+ #define TF_ERROR (-1)
28
+
29
+ /* ---- Error helpers ---- */
30
+
31
+ static void set_error(char **error, const char *msg) {
32
+ if (error) {
33
+ free(*error);
34
+ size_t len = strlen(msg) + 1;
35
+ *error = malloc(len);
36
+ if (*error) memcpy(*error, msg, len);
37
+ }
38
+ }
39
+
40
+ static void set_errorf(char **error, const char *fmt, const char *a) {
41
+ if (error) {
42
+ char buf[256];
43
+ snprintf(buf, sizeof(buf), fmt, a);
44
+ set_error(error, buf);
45
+ }
46
+ }
47
+
48
+ /* ---- Token type ---- */
49
+
50
+ typedef struct {
51
+ char **items;
52
+ size_t count;
53
+ size_t cap;
54
+ } token_list;
55
+
56
+ static void tl_init(token_list *tl) {
57
+ tl->items = NULL;
58
+ tl->count = 0;
59
+ tl->cap = 0;
60
+ }
61
+
62
+ static int tl_push(token_list *tl, const char *s, size_t len) {
63
+ if (tl->count >= tl->cap) {
64
+ size_t new_cap = tl->cap ? tl->cap * 2 : 8;
65
+ char **new_items = realloc(tl->items, new_cap * sizeof(char *));
66
+ if (!new_items) return -1;
67
+ tl->items = new_items;
68
+ tl->cap = new_cap;
69
+ }
70
+ char *dup = malloc(len + 1);
71
+ if (!dup) return -1;
72
+ memcpy(dup, s, len);
73
+ dup[len] = '\0';
74
+ tl->items[tl->count++] = dup;
75
+ return 0;
76
+ }
77
+
78
+ static void tl_free(token_list *tl) {
79
+ for (size_t i = 0; i < tl->count; i++) free(tl->items[i]);
80
+ free(tl->items);
81
+ tl->items = NULL;
82
+ tl->count = 0;
83
+ tl->cap = 0;
84
+ }
85
+
86
+ /* ---- Stage splitting ---- */
87
+
88
+ /*
89
+ * Split input on '|' while respecting double-quoted strings.
90
+ * Each stage is trimmed of leading/trailing whitespace.
91
+ */
92
+ static int split_stages(const char *text, size_t len, token_list *out) {
93
+ tl_init(out);
94
+ size_t start = 0;
95
+ int in_quote = 0;
96
+
97
+ for (size_t i = 0; i < len; i++) {
98
+ if (text[i] == '"') {
99
+ in_quote = !in_quote;
100
+ } else if (text[i] == '|' && !in_quote) {
101
+ /* Trim whitespace */
102
+ size_t s = start, e = i;
103
+ while (s < e && isspace((unsigned char)text[s])) s++;
104
+ while (e > s && isspace((unsigned char)text[e - 1])) e--;
105
+ if (s == e) return -1; /* empty stage */
106
+ tl_push(out, text + s, e - s);
107
+ start = i + 1;
108
+ }
109
+ }
110
+
111
+ /* Last stage */
112
+ size_t s = start, e = len;
113
+ while (s < e && isspace((unsigned char)text[s])) s++;
114
+ while (e > s && isspace((unsigned char)text[e - 1])) e--;
115
+ if (s < e) tl_push(out, text + s, e - s);
116
+
117
+ return (out->count > 0) ? 0 : -1;
118
+ }
119
+
120
+ /* ---- Stage tokenization ---- */
121
+
122
+ /*
123
+ * Tokenize a single stage into op name + args.
124
+ * Handles: "quoted strings", key=value, bare words.
125
+ * Comma-separated bare words are split into individual tokens.
126
+ *
127
+ * For derive, we need special handling: key="expr with spaces"
128
+ * is kept as a single token "key=expr with spaces".
129
+ */
130
+ static int tokenize_stage(const char *stage, token_list *out) {
131
+ tl_init(out);
132
+ const char *p = stage;
133
+
134
+ while (*p) {
135
+ /* Skip whitespace */
136
+ while (*p && isspace((unsigned char)*p)) p++;
137
+ if (!*p) break;
138
+
139
+ if (*p == '"') {
140
+ /* Quoted string — capture content without quotes */
141
+ p++;
142
+ const char *start = p;
143
+ while (*p && *p != '"') p++;
144
+ tl_push(out, start, p - start);
145
+ if (*p == '"') p++;
146
+ } else {
147
+ /* Bare word or key=value — delimited by whitespace */
148
+ const char *start = p;
149
+
150
+ /* Check if this is key="value with spaces" */
151
+ const char *eq = NULL;
152
+ const char *scan = p;
153
+ while (*scan && !isspace((unsigned char)*scan) && *scan != '"') {
154
+ if (*scan == '=' && !eq) eq = scan;
155
+ scan++;
156
+ }
157
+
158
+ if (eq && *scan == '"') {
159
+ /* key="quoted value" — read until closing quote */
160
+ scan++; /* skip opening quote */
161
+ while (*scan && *scan != '"') scan++;
162
+ if (*scan == '"') scan++; /* skip closing quote */
163
+ /* The full token is from start to scan, but we need to
164
+ * strip the quotes from the value part */
165
+ size_t key_len = eq - start;
166
+ const char *val_start = eq + 2; /* skip = and " */
167
+ const char *val_end = scan - 1; /* before closing " */
168
+ size_t total = key_len + 1 + (val_end - val_start);
169
+ char *tok = malloc(total + 1);
170
+ if (tok) {
171
+ memcpy(tok, start, key_len);
172
+ tok[key_len] = '=';
173
+ memcpy(tok + key_len + 1, val_start, val_end - val_start);
174
+ tok[total] = '\0';
175
+ if (out->count >= out->cap) {
176
+ size_t new_cap = out->cap ? out->cap * 2 : 8;
177
+ char **new_items = realloc(out->items, new_cap * sizeof(char *));
178
+ if (new_items) { out->items = new_items; out->cap = new_cap; }
179
+ }
180
+ out->items[out->count++] = tok;
181
+ }
182
+ p = scan;
183
+ } else {
184
+ while (*p && !isspace((unsigned char)*p) && *p != '"') p++;
185
+ size_t len = p - start;
186
+
187
+ /* Split comma-separated tokens (for "name,age" → "name", "age") */
188
+ /* But not if it contains '=' (key=value pair) */
189
+ if (memchr(start, '=', len) == NULL && memchr(start, ',', len) != NULL) {
190
+ const char *cs = start;
191
+ while (cs < start + len) {
192
+ const char *comma = memchr(cs, ',', (start + len) - cs);
193
+ size_t tok_len = comma ? (size_t)(comma - cs) : (size_t)((start + len) - cs);
194
+ if (tok_len > 0) tl_push(out, cs, tok_len);
195
+ cs += tok_len + 1;
196
+ }
197
+ } else {
198
+ tl_push(out, start, len);
199
+ }
200
+ }
201
+ }
202
+ }
203
+
204
+ return (out->count > 0) ? 0 : -1;
205
+ }
206
+
207
+ /* ---- Codec resolution ---- */
208
+
209
+ /*
210
+ * Resolve bare codec names to full op names based on position.
211
+ * Returns a malloc'd string (caller frees) or NULL if not a codec shorthand.
212
+ */
213
+ static char *resolve_codec(const char *name, int is_first, int is_last) {
214
+ /* Already explicit */
215
+ if (strcmp(name, "codec.csv.decode") == 0 ||
216
+ strcmp(name, "codec.csv.encode") == 0 ||
217
+ strcmp(name, "codec.jsonl.decode") == 0 ||
218
+ strcmp(name, "codec.jsonl.encode") == 0 ||
219
+ strcmp(name, "codec.text.decode") == 0 ||
220
+ strcmp(name, "codec.text.encode") == 0 ||
221
+ strcmp(name, "csv.decode") == 0 ||
222
+ strcmp(name, "csv.encode") == 0 ||
223
+ strcmp(name, "jsonl.decode") == 0 ||
224
+ strcmp(name, "jsonl.encode") == 0 ||
225
+ strcmp(name, "text.decode") == 0 ||
226
+ strcmp(name, "text.encode") == 0) {
227
+ /* Normalize short explicit forms */
228
+ if (strcmp(name, "csv.decode") == 0) return strdup("codec.csv.decode");
229
+ if (strcmp(name, "csv.encode") == 0) return strdup("codec.csv.encode");
230
+ if (strcmp(name, "jsonl.decode") == 0) return strdup("codec.jsonl.decode");
231
+ if (strcmp(name, "jsonl.encode") == 0) return strdup("codec.jsonl.encode");
232
+ if (strcmp(name, "text.decode") == 0) return strdup("codec.text.decode");
233
+ if (strcmp(name, "text.encode") == 0) return strdup("codec.text.encode");
234
+ return strdup(name);
235
+ }
236
+
237
+ if (strcmp(name, "csv") == 0) {
238
+ if (is_first) return strdup("codec.csv.decode");
239
+ if (is_last) return strdup("codec.csv.encode");
240
+ return NULL; /* ambiguous */
241
+ }
242
+ if (strcmp(name, "jsonl") == 0) {
243
+ if (is_first) return strdup("codec.jsonl.decode");
244
+ if (is_last) return strdup("codec.jsonl.encode");
245
+ return NULL;
246
+ }
247
+ if (strcmp(name, "text") == 0) {
248
+ if (is_first) return strdup("codec.text.decode");
249
+ if (is_last) return strdup("codec.text.encode");
250
+ return NULL;
251
+ }
252
+ if (strcmp(name, "table") == 0) {
253
+ if (is_last) return strdup("codec.table.encode");
254
+ return NULL;
255
+ }
256
+
257
+ return NULL; /* not a codec */
258
+ }
259
+
260
+ /* ---- Arg builders per op type ---- */
261
+
262
+ static cJSON *build_codec_args(const token_list *tokens) {
263
+ /* tokens[0] is op name, rest are key=value pairs */
264
+ cJSON *args = cJSON_CreateObject();
265
+ for (size_t i = 1; i < tokens->count; i++) {
266
+ char *eq = strchr(tokens->items[i], '=');
267
+ if (eq) {
268
+ *eq = '\0';
269
+ const char *key = tokens->items[i];
270
+ const char *val = eq + 1;
271
+ /* Try to detect booleans and ints */
272
+ if (strcmp(val, "true") == 0 || strcmp(val, "false") == 0) {
273
+ cJSON_AddBoolToObject(args, key, strcmp(val, "true") == 0);
274
+ } else {
275
+ /* Check if integer */
276
+ char *end;
277
+ long num = strtol(val, &end, 10);
278
+ if (*end == '\0' && end != val) {
279
+ cJSON_AddNumberToObject(args, key, num);
280
+ } else {
281
+ cJSON_AddStringToObject(args, key, val);
282
+ }
283
+ }
284
+ *eq = '='; /* restore */
285
+ }
286
+ }
287
+ return args;
288
+ }
289
+
290
+ static cJSON *build_filter_args(const token_list *tokens, char **error) {
291
+ /* filter "expr" — tokens[1] should be the expression */
292
+ if (tokens->count < 2) {
293
+ set_error(error, "filter requires an expression argument");
294
+ return NULL;
295
+ }
296
+ cJSON *args = cJSON_CreateObject();
297
+ cJSON_AddStringToObject(args, "expr", tokens->items[1]);
298
+ return args;
299
+ }
300
+
301
+ static cJSON *build_select_args(const token_list *tokens, char **error) {
302
+ /* select name,age or select name age — tokens[1..n] are column names */
303
+ if (tokens->count < 2) {
304
+ set_error(error, "select requires at least one column name");
305
+ return NULL;
306
+ }
307
+ cJSON *args = cJSON_CreateObject();
308
+ cJSON *cols = cJSON_CreateArray();
309
+ for (size_t i = 1; i < tokens->count; i++) {
310
+ cJSON_AddItemToArray(cols, cJSON_CreateString(tokens->items[i]));
311
+ }
312
+ cJSON_AddItemToObject(args, "columns", cols);
313
+ return args;
314
+ }
315
+
316
+ static cJSON *build_rename_args(const token_list *tokens, char **error) {
317
+ /* rename old=new,old2=new2 or rename old=new old2=new2 */
318
+ if (tokens->count < 2) {
319
+ set_error(error, "rename requires at least one old=new mapping");
320
+ return NULL;
321
+ }
322
+ cJSON *args = cJSON_CreateObject();
323
+ cJSON *mapping = cJSON_CreateObject();
324
+
325
+ for (size_t i = 1; i < tokens->count; i++) {
326
+ /* Each token might be "old=new" or comma-separated "old=new,old2=new2" */
327
+ char *tok = tokens->items[i];
328
+ /* Split on commas within this token */
329
+ char *saveptr = NULL;
330
+ char *copy = strdup(tok);
331
+ char *part = strtok_r(copy, ",", &saveptr);
332
+ while (part) {
333
+ char *eq = strchr(part, '=');
334
+ if (!eq) {
335
+ free(copy);
336
+ cJSON_Delete(args);
337
+ set_errorf(error, "invalid rename mapping: '%s' (expected old=new)", part);
338
+ return NULL;
339
+ }
340
+ *eq = '\0';
341
+ cJSON_AddStringToObject(mapping, part, eq + 1);
342
+ part = strtok_r(NULL, ",", &saveptr);
343
+ }
344
+ free(copy);
345
+ }
346
+
347
+ cJSON_AddItemToObject(args, "mapping", mapping);
348
+ return args;
349
+ }
350
+
351
+ static cJSON *build_head_args(const token_list *tokens, char **error) {
352
+ /* head N */
353
+ if (tokens->count < 2) {
354
+ set_error(error, "head requires a count argument");
355
+ return NULL;
356
+ }
357
+ char *end;
358
+ long n = strtol(tokens->items[1], &end, 10);
359
+ if (*end != '\0' || n <= 0) {
360
+ set_errorf(error, "head: invalid count '%s'", tokens->items[1]);
361
+ return NULL;
362
+ }
363
+ cJSON *args = cJSON_CreateObject();
364
+ cJSON_AddNumberToObject(args, "n", n);
365
+ return args;
366
+ }
367
+
368
+ static cJSON *build_skip_args(const token_list *tokens, char **error) {
369
+ /* skip N */
370
+ if (tokens->count < 2) {
371
+ set_error(error, "skip requires a count argument");
372
+ return NULL;
373
+ }
374
+ char *end;
375
+ long n = strtol(tokens->items[1], &end, 10);
376
+ if (*end != '\0' || n <= 0) {
377
+ set_errorf(error, "skip: invalid count '%s'", tokens->items[1]);
378
+ return NULL;
379
+ }
380
+ cJSON *args = cJSON_CreateObject();
381
+ cJSON_AddNumberToObject(args, "n", n);
382
+ return args;
383
+ }
384
+
385
+ static cJSON *build_derive_args(const token_list *tokens, char **error) {
386
+ /* derive name=expr name2=expr2 */
387
+ if (tokens->count < 2) {
388
+ set_error(error, "derive requires at least one name=expression mapping");
389
+ return NULL;
390
+ }
391
+ cJSON *args = cJSON_CreateObject();
392
+ cJSON *columns = cJSON_CreateArray();
393
+
394
+ for (size_t i = 1; i < tokens->count; i++) {
395
+ char *eq = strchr(tokens->items[i], '=');
396
+ if (!eq) {
397
+ cJSON_Delete(args);
398
+ set_errorf(error, "derive: invalid mapping '%s' (expected name=expr)", tokens->items[i]);
399
+ return NULL;
400
+ }
401
+ *eq = '\0';
402
+ cJSON *col = cJSON_CreateObject();
403
+ cJSON_AddStringToObject(col, "name", tokens->items[i]);
404
+ cJSON_AddStringToObject(col, "expr", eq + 1);
405
+ cJSON_AddItemToArray(columns, col);
406
+ *eq = '='; /* restore */
407
+ }
408
+
409
+ cJSON_AddItemToObject(args, "columns", columns);
410
+ return args;
411
+ }
412
+
413
+ static cJSON *build_stats_args(const token_list *tokens, char **error) {
414
+ (void)error;
415
+ /* stats [count,sum,avg,min,max] */
416
+ cJSON *args = cJSON_CreateObject();
417
+ if (tokens->count >= 2) {
418
+ /* Parse comma-separated stat names */
419
+ cJSON *stats = cJSON_CreateArray();
420
+ for (size_t i = 1; i < tokens->count; i++) {
421
+ cJSON_AddItemToArray(stats, cJSON_CreateString(tokens->items[i]));
422
+ }
423
+ cJSON_AddItemToObject(args, "stats", stats);
424
+ }
425
+ return args;
426
+ }
427
+
428
+ static cJSON *build_unique_args(const token_list *tokens, char **error) {
429
+ (void)error;
430
+ /* unique [col1,col2] */
431
+ cJSON *args = cJSON_CreateObject();
432
+ if (tokens->count >= 2) {
433
+ cJSON *cols = cJSON_CreateArray();
434
+ for (size_t i = 1; i < tokens->count; i++) {
435
+ cJSON_AddItemToArray(cols, cJSON_CreateString(tokens->items[i]));
436
+ }
437
+ cJSON_AddItemToObject(args, "columns", cols);
438
+ }
439
+ return args;
440
+ }
441
+
442
+ static cJSON *build_sort_args(const token_list *tokens, char **error) {
443
+ /* sort col1,-col2,col3 */
444
+ if (tokens->count < 2) {
445
+ set_error(error, "sort requires at least one column name");
446
+ return NULL;
447
+ }
448
+ cJSON *args = cJSON_CreateObject();
449
+ cJSON *columns = cJSON_CreateArray();
450
+
451
+ for (size_t i = 1; i < tokens->count; i++) {
452
+ const char *tok = tokens->items[i];
453
+ int desc = 0;
454
+ if (tok[0] == '-') {
455
+ desc = 1;
456
+ tok++;
457
+ }
458
+ cJSON *col = cJSON_CreateObject();
459
+ cJSON_AddStringToObject(col, "name", tok);
460
+ cJSON_AddBoolToObject(col, "desc", desc);
461
+ cJSON_AddItemToArray(columns, col);
462
+ }
463
+
464
+ cJSON_AddItemToObject(args, "columns", columns);
465
+ return args;
466
+ }
467
+
468
+ static cJSON *build_top_args(const token_list *tokens, char **error) {
469
+ /* top 10 score or top 10 -score */
470
+ if (tokens->count < 3) {
471
+ set_error(error, "top requires N and column arguments");
472
+ return NULL;
473
+ }
474
+ char *end;
475
+ long n = strtol(tokens->items[1], &end, 10);
476
+ if (*end != '\0' || n <= 0) {
477
+ set_errorf(error, "top: invalid count '%s'", tokens->items[1]);
478
+ return NULL;
479
+ }
480
+ cJSON *args = cJSON_CreateObject();
481
+ cJSON_AddNumberToObject(args, "n", n);
482
+ const char *col = tokens->items[2];
483
+ int desc = 1; /* default: highest first */
484
+ if (col[0] == '-') { col++; desc = 1; }
485
+ else if (col[0] == '+') { col++; desc = 0; }
486
+ cJSON_AddStringToObject(args, "column", col);
487
+ cJSON_AddBoolToObject(args, "desc", desc);
488
+ return args;
489
+ }
490
+
491
+ static cJSON *build_replace_args(const token_list *tokens, char **error) {
492
+ /* replace [--regex] column pattern replacement */
493
+ if (tokens->count < 4) {
494
+ set_error(error, "replace requires column, pattern, and replacement");
495
+ return NULL;
496
+ }
497
+ cJSON *args = cJSON_CreateObject();
498
+ size_t idx = 1;
499
+ if (strcmp(tokens->items[idx], "--regex") == 0 || strcmp(tokens->items[idx], "-r") == 0) {
500
+ cJSON_AddBoolToObject(args, "regex", 1);
501
+ idx++;
502
+ if (idx + 2 >= tokens->count) {
503
+ cJSON_Delete(args);
504
+ set_error(error, "replace requires column, pattern, and replacement");
505
+ return NULL;
506
+ }
507
+ }
508
+ cJSON_AddStringToObject(args, "column", tokens->items[idx]);
509
+ cJSON_AddStringToObject(args, "pattern", tokens->items[idx + 1]);
510
+ cJSON_AddStringToObject(args, "replacement", tokens->items[idx + 2]);
511
+ return args;
512
+ }
513
+
514
+ static cJSON *build_clip_args(const token_list *tokens, char **error) {
515
+ /* clip column min=0 max=100 */
516
+ if (tokens->count < 2) {
517
+ set_error(error, "clip requires a column name");
518
+ return NULL;
519
+ }
520
+ cJSON *args = cJSON_CreateObject();
521
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
522
+ for (size_t i = 2; i < tokens->count; i++) {
523
+ char *eq = strchr(tokens->items[i], '=');
524
+ if (eq) {
525
+ *eq = '\0';
526
+ double v = strtod(eq + 1, NULL);
527
+ cJSON_AddNumberToObject(args, tokens->items[i], v);
528
+ *eq = '=';
529
+ }
530
+ }
531
+ return args;
532
+ }
533
+
534
+ static cJSON *build_bin_args(const token_list *tokens, char **error) {
535
+ /* bin column 10,20,30 */
536
+ if (tokens->count < 3) {
537
+ set_error(error, "bin requires column and boundaries");
538
+ return NULL;
539
+ }
540
+ cJSON *args = cJSON_CreateObject();
541
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
542
+ cJSON *boundaries = cJSON_CreateArray();
543
+ for (size_t i = 2; i < tokens->count; i++) {
544
+ cJSON_AddItemToArray(boundaries, cJSON_CreateNumber(strtod(tokens->items[i], NULL)));
545
+ }
546
+ cJSON_AddItemToObject(args, "boundaries", boundaries);
547
+ return args;
548
+ }
549
+
550
+ static cJSON *build_datetime_args(const token_list *tokens, char **error) {
551
+ /* datetime date_col year,month,day */
552
+ if (tokens->count < 2) {
553
+ set_error(error, "datetime requires a column name");
554
+ return NULL;
555
+ }
556
+ cJSON *args = cJSON_CreateObject();
557
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
558
+ if (tokens->count >= 3) {
559
+ cJSON *extract = cJSON_CreateArray();
560
+ for (size_t i = 2; i < tokens->count; i++)
561
+ cJSON_AddItemToArray(extract, cJSON_CreateString(tokens->items[i]));
562
+ cJSON_AddItemToObject(args, "extract", extract);
563
+ }
564
+ return args;
565
+ }
566
+
567
+ static cJSON *build_explode_args(const token_list *tokens, char **error) {
568
+ /* explode column [delimiter] */
569
+ if (tokens->count < 2) {
570
+ set_error(error, "explode requires a column name");
571
+ return NULL;
572
+ }
573
+ cJSON *args = cJSON_CreateObject();
574
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
575
+ if (tokens->count >= 3)
576
+ cJSON_AddStringToObject(args, "delimiter", tokens->items[2]);
577
+ return args;
578
+ }
579
+
580
+ static cJSON *build_split_args(const token_list *tokens, char **error) {
581
+ /* split column delimiter name1,name2 */
582
+ if (tokens->count < 4) {
583
+ set_error(error, "split requires column, delimiter, and names");
584
+ return NULL;
585
+ }
586
+ cJSON *args = cJSON_CreateObject();
587
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
588
+ cJSON_AddStringToObject(args, "delimiter", tokens->items[2]);
589
+ cJSON *names = cJSON_CreateArray();
590
+ for (size_t i = 3; i < tokens->count; i++)
591
+ cJSON_AddItemToArray(names, cJSON_CreateString(tokens->items[i]));
592
+ cJSON_AddItemToObject(args, "names", names);
593
+ return args;
594
+ }
595
+
596
+ static cJSON *build_unpivot_args(const token_list *tokens, char **error) {
597
+ /* unpivot col1,col2,col3 */
598
+ if (tokens->count < 2) {
599
+ set_error(error, "unpivot requires at least one column name");
600
+ return NULL;
601
+ }
602
+ cJSON *args = cJSON_CreateObject();
603
+ cJSON *cols = cJSON_CreateArray();
604
+ for (size_t i = 1; i < tokens->count; i++)
605
+ cJSON_AddItemToArray(cols, cJSON_CreateString(tokens->items[i]));
606
+ cJSON_AddItemToObject(args, "columns", cols);
607
+ return args;
608
+ }
609
+
610
+ static cJSON *build_group_agg_args(const token_list *tokens, char **error) {
611
+ /* group-agg group_col1,group_col2 col:func[:name] ... */
612
+ if (tokens->count < 3) {
613
+ set_error(error, "group-agg requires group columns and at least one aggregation");
614
+ return NULL;
615
+ }
616
+ cJSON *args = cJSON_CreateObject();
617
+
618
+ /* First arg: group columns (comma-separated already split by tokenizer) */
619
+ cJSON *group_by = cJSON_CreateArray();
620
+ /* The first token after op name contains group columns */
621
+ cJSON_AddItemToArray(group_by, cJSON_CreateString(tokens->items[1]));
622
+ cJSON_AddItemToObject(args, "group_by", group_by);
623
+
624
+ /* Remaining args: col:func or col:func:name */
625
+ cJSON *aggs = cJSON_CreateArray();
626
+ for (size_t i = 2; i < tokens->count; i++) {
627
+ char *tok = strdup(tokens->items[i]);
628
+ char *colon1 = strchr(tok, ':');
629
+ if (!colon1) { free(tok); continue; }
630
+ *colon1 = '\0';
631
+ char *func = colon1 + 1;
632
+ char *colon2 = strchr(func, ':');
633
+ char *name = NULL;
634
+ if (colon2) { *colon2 = '\0'; name = colon2 + 1; }
635
+
636
+ cJSON *agg = cJSON_CreateObject();
637
+ cJSON_AddStringToObject(agg, "column", tok);
638
+ cJSON_AddStringToObject(agg, "func", func);
639
+ if (name) cJSON_AddStringToObject(agg, "name", name);
640
+ cJSON_AddItemToArray(aggs, agg);
641
+ free(tok);
642
+ }
643
+ cJSON_AddItemToObject(args, "aggs", aggs);
644
+ return args;
645
+ }
646
+
647
+ static cJSON *build_frequency_args(const token_list *tokens, char **error) {
648
+ (void)error;
649
+ /* frequency [col1,col2] */
650
+ cJSON *args = cJSON_CreateObject();
651
+ if (tokens->count >= 2) {
652
+ cJSON *cols = cJSON_CreateArray();
653
+ for (size_t i = 1; i < tokens->count; i++)
654
+ cJSON_AddItemToArray(cols, cJSON_CreateString(tokens->items[i]));
655
+ cJSON_AddItemToObject(args, "columns", cols);
656
+ }
657
+ return args;
658
+ }
659
+
660
+ static cJSON *build_window_args(const token_list *tokens, char **error) {
661
+ /* window column size func [result_name] */
662
+ if (tokens->count < 4) {
663
+ set_error(error, "window requires column, size, and func");
664
+ return NULL;
665
+ }
666
+ cJSON *args = cJSON_CreateObject();
667
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
668
+ cJSON_AddNumberToObject(args, "size", strtol(tokens->items[2], NULL, 10));
669
+ cJSON_AddStringToObject(args, "func", tokens->items[3]);
670
+ if (tokens->count >= 5)
671
+ cJSON_AddStringToObject(args, "result", tokens->items[4]);
672
+ return args;
673
+ }
674
+
675
+ static cJSON *build_step_args(const token_list *tokens, char **error) {
676
+ /* step column func [result_name] */
677
+ if (tokens->count < 3) {
678
+ set_error(error, "step requires column and func");
679
+ return NULL;
680
+ }
681
+ cJSON *args = cJSON_CreateObject();
682
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
683
+ cJSON_AddStringToObject(args, "func", tokens->items[2]);
684
+ if (tokens->count >= 4)
685
+ cJSON_AddStringToObject(args, "result", tokens->items[3]);
686
+ return args;
687
+ }
688
+
689
+ static cJSON *build_flatten_args(const token_list *tokens, char **error) {
690
+ (void)tokens; (void)error;
691
+ return cJSON_CreateObject();
692
+ }
693
+
694
+ static cJSON *build_grep_args(const token_list *tokens, char **error) {
695
+ /* grep [-v] [-r] pattern */
696
+ if (tokens->count < 2) {
697
+ set_error(error, "grep requires a pattern argument");
698
+ return NULL;
699
+ }
700
+ cJSON *args = cJSON_CreateObject();
701
+ size_t idx = 1;
702
+ while (idx < tokens->count && tokens->items[idx][0] == '-' && tokens->items[idx][1] != '\0') {
703
+ const char *flag = tokens->items[idx];
704
+ if (strcmp(flag, "-v") == 0) {
705
+ cJSON_AddBoolToObject(args, "invert", 1);
706
+ } else if (strcmp(flag, "-r") == 0 || strcmp(flag, "--regex") == 0) {
707
+ cJSON_AddBoolToObject(args, "regex", 1);
708
+ } else if (strcmp(flag, "-rv") == 0 || strcmp(flag, "-vr") == 0) {
709
+ cJSON_AddBoolToObject(args, "invert", 1);
710
+ cJSON_AddBoolToObject(args, "regex", 1);
711
+ } else {
712
+ break; /* not a flag, must be pattern */
713
+ }
714
+ idx++;
715
+ }
716
+ if (idx >= tokens->count) {
717
+ cJSON_Delete(args);
718
+ set_error(error, "grep requires a pattern argument");
719
+ return NULL;
720
+ }
721
+ cJSON_AddStringToObject(args, "pattern", tokens->items[idx]);
722
+ return args;
723
+ }
724
+
725
+ static cJSON *build_pivot_args(const token_list *tokens, char **error) {
726
+ /* pivot name_col value_col [agg] */
727
+ if (tokens->count < 3) {
728
+ set_error(error, "pivot requires name_column and value_column");
729
+ return NULL;
730
+ }
731
+ cJSON *args = cJSON_CreateObject();
732
+ cJSON_AddStringToObject(args, "name_column", tokens->items[1]);
733
+ cJSON_AddStringToObject(args, "value_column", tokens->items[2]);
734
+ if (tokens->count >= 4)
735
+ cJSON_AddStringToObject(args, "agg", tokens->items[3]);
736
+ return args;
737
+ }
738
+
739
+ static cJSON *build_join_args(const token_list *tokens, char **error) {
740
+ /* join lookup.csv on id [--left]
741
+ * join lookup.csv on id=lookup_id [--left] */
742
+ if (tokens->count < 4) {
743
+ set_error(error, "join requires file, 'on', and column");
744
+ return NULL;
745
+ }
746
+ if (strcmp(tokens->items[2], "on") != 0) {
747
+ set_error(error, "join: expected 'on' keyword");
748
+ return NULL;
749
+ }
750
+ cJSON *args = cJSON_CreateObject();
751
+ cJSON_AddStringToObject(args, "file", tokens->items[1]);
752
+ cJSON_AddStringToObject(args, "on", tokens->items[3]);
753
+ for (size_t i = 4; i < tokens->count; i++) {
754
+ if (strcmp(tokens->items[i], "--left") == 0)
755
+ cJSON_AddStringToObject(args, "how", "left");
756
+ }
757
+ return args;
758
+ }
759
+
760
+ static cJSON *build_stack_args(const token_list *tokens, char **error) {
761
+ /* stack file.csv [--tag source] */
762
+ if (tokens->count < 2) {
763
+ set_error(error, "stack requires a file path");
764
+ return NULL;
765
+ }
766
+ cJSON *args = cJSON_CreateObject();
767
+ cJSON_AddStringToObject(args, "file", tokens->items[1]);
768
+ for (size_t i = 2; i < tokens->count; i++) {
769
+ if (strcmp(tokens->items[i], "--tag") == 0 && i + 1 < tokens->count) {
770
+ cJSON_AddStringToObject(args, "tag", tokens->items[i + 1]);
771
+ i++;
772
+ }
773
+ }
774
+ return args;
775
+ }
776
+
777
+ static cJSON *build_lead_args(const token_list *tokens, char **error) {
778
+ /* lead column [offset] [result_name] */
779
+ if (tokens->count < 2) {
780
+ set_error(error, "lead requires a column name");
781
+ return NULL;
782
+ }
783
+ cJSON *args = cJSON_CreateObject();
784
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
785
+ if (tokens->count >= 3) {
786
+ /* Check if next arg is numeric (offset) or a name (result) */
787
+ char *end;
788
+ long off = strtol(tokens->items[2], &end, 10);
789
+ if (*end == '\0') {
790
+ cJSON_AddNumberToObject(args, "offset", off);
791
+ if (tokens->count >= 4)
792
+ cJSON_AddStringToObject(args, "result", tokens->items[3]);
793
+ } else {
794
+ cJSON_AddStringToObject(args, "result", tokens->items[2]);
795
+ }
796
+ }
797
+ return args;
798
+ }
799
+
800
+ static cJSON *build_date_trunc_args(const token_list *tokens, char **error) {
801
+ /* date-trunc column granularity [result_name] */
802
+ if (tokens->count < 3) {
803
+ set_error(error, "date-trunc requires column and granularity");
804
+ return NULL;
805
+ }
806
+ cJSON *args = cJSON_CreateObject();
807
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
808
+ cJSON_AddStringToObject(args, "trunc", tokens->items[2]);
809
+ if (tokens->count >= 4)
810
+ cJSON_AddStringToObject(args, "result", tokens->items[3]);
811
+ return args;
812
+ }
813
+
814
+ static cJSON *build_onehot_args(const token_list *tokens, char **error) {
815
+ /* onehot column [--drop] */
816
+ if (tokens->count < 2) {
817
+ set_error(error, "onehot requires a column name");
818
+ return NULL;
819
+ }
820
+ cJSON *args = cJSON_CreateObject();
821
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
822
+ for (size_t i = 2; i < tokens->count; i++) {
823
+ if (strcmp(tokens->items[i], "--drop") == 0)
824
+ cJSON_AddBoolToObject(args, "drop", 1);
825
+ }
826
+ return args;
827
+ }
828
+
829
+ static cJSON *build_label_encode_args(const token_list *tokens, char **error) {
830
+ /* label-encode column [result_name] */
831
+ if (tokens->count < 2) {
832
+ set_error(error, "label-encode requires a column name");
833
+ return NULL;
834
+ }
835
+ cJSON *args = cJSON_CreateObject();
836
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
837
+ if (tokens->count >= 3)
838
+ cJSON_AddStringToObject(args, "result", tokens->items[2]);
839
+ return args;
840
+ }
841
+
842
+ static cJSON *build_ewma_args(const token_list *tokens, char **error) {
843
+ /* ewma column alpha [result_name] */
844
+ if (tokens->count < 3) {
845
+ set_error(error, "ewma requires column and alpha");
846
+ return NULL;
847
+ }
848
+ cJSON *args = cJSON_CreateObject();
849
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
850
+ cJSON_AddNumberToObject(args, "alpha", atof(tokens->items[2]));
851
+ if (tokens->count >= 4)
852
+ cJSON_AddStringToObject(args, "result", tokens->items[3]);
853
+ return args;
854
+ }
855
+
856
+ static cJSON *build_diff_args(const token_list *tokens, char **error) {
857
+ /* diff column [order] [result_name] */
858
+ if (tokens->count < 2) {
859
+ set_error(error, "diff requires a column name");
860
+ return NULL;
861
+ }
862
+ cJSON *args = cJSON_CreateObject();
863
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
864
+ if (tokens->count >= 3) {
865
+ char *end;
866
+ long order = strtol(tokens->items[2], &end, 10);
867
+ if (*end == '\0') {
868
+ cJSON_AddNumberToObject(args, "order", order);
869
+ if (tokens->count >= 4)
870
+ cJSON_AddStringToObject(args, "result", tokens->items[3]);
871
+ } else {
872
+ cJSON_AddStringToObject(args, "result", tokens->items[2]);
873
+ }
874
+ }
875
+ return args;
876
+ }
877
+
878
+ static cJSON *build_anomaly_args(const token_list *tokens, char **error) {
879
+ /* anomaly column [threshold] [result_name] */
880
+ if (tokens->count < 2) {
881
+ set_error(error, "anomaly requires a column name");
882
+ return NULL;
883
+ }
884
+ cJSON *args = cJSON_CreateObject();
885
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
886
+ if (tokens->count >= 3) {
887
+ char *end;
888
+ double thresh = strtod(tokens->items[2], &end);
889
+ if (*end == '\0') {
890
+ cJSON_AddNumberToObject(args, "threshold", thresh);
891
+ if (tokens->count >= 4)
892
+ cJSON_AddStringToObject(args, "result", tokens->items[3]);
893
+ } else {
894
+ cJSON_AddStringToObject(args, "result", tokens->items[2]);
895
+ }
896
+ }
897
+ return args;
898
+ }
899
+
900
+ static cJSON *build_split_data_args(const token_list *tokens, char **error) {
901
+ /* split-data [ratio] [--seed N] [result_name] */
902
+ (void)error;
903
+ cJSON *args = cJSON_CreateObject();
904
+ size_t idx = 1;
905
+ if (idx < tokens->count) {
906
+ char *end;
907
+ double ratio = strtod(tokens->items[idx], &end);
908
+ if (*end == '\0') {
909
+ cJSON_AddNumberToObject(args, "ratio", ratio);
910
+ idx++;
911
+ }
912
+ }
913
+ while (idx < tokens->count) {
914
+ if (strcmp(tokens->items[idx], "--seed") == 0 && idx + 1 < tokens->count) {
915
+ cJSON_AddNumberToObject(args, "seed", atoi(tokens->items[idx + 1]));
916
+ idx += 2;
917
+ } else {
918
+ cJSON_AddStringToObject(args, "result", tokens->items[idx]);
919
+ idx++;
920
+ }
921
+ }
922
+ return args;
923
+ }
924
+
925
+ static cJSON *build_interpolate_args(const token_list *tokens, char **error) {
926
+ /* interpolate column [method] */
927
+ if (tokens->count < 2) {
928
+ set_error(error, "interpolate requires a column name");
929
+ return NULL;
930
+ }
931
+ cJSON *args = cJSON_CreateObject();
932
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
933
+ if (tokens->count >= 3)
934
+ cJSON_AddStringToObject(args, "method", tokens->items[2]);
935
+ return args;
936
+ }
937
+
938
+ static cJSON *build_normalize_args(const token_list *tokens, char **error) {
939
+ /* normalize col1,col2,... [method] */
940
+ if (tokens->count < 2) {
941
+ set_error(error, "normalize requires column names");
942
+ return NULL;
943
+ }
944
+ cJSON *args = cJSON_CreateObject();
945
+ /* Parse comma-separated columns */
946
+ cJSON *cols = cJSON_CreateArray();
947
+ char *dup = strdup(tokens->items[1]);
948
+ char *tok = strtok(dup, ",");
949
+ while (tok) {
950
+ cJSON_AddItemToArray(cols, cJSON_CreateString(tok));
951
+ tok = strtok(NULL, ",");
952
+ }
953
+ free(dup);
954
+ cJSON_AddItemToObject(args, "columns", cols);
955
+ if (tokens->count >= 3)
956
+ cJSON_AddStringToObject(args, "method", tokens->items[2]);
957
+ return args;
958
+ }
959
+
960
+ static cJSON *build_acf_args(const token_list *tokens, char **error) {
961
+ /* acf column [lags] */
962
+ if (tokens->count < 2) {
963
+ set_error(error, "acf requires a column name");
964
+ return NULL;
965
+ }
966
+ cJSON *args = cJSON_CreateObject();
967
+ cJSON_AddStringToObject(args, "column", tokens->items[1]);
968
+ if (tokens->count >= 3)
969
+ cJSON_AddNumberToObject(args, "lags", atoi(tokens->items[2]));
970
+ return args;
971
+ }
972
+
973
+ /* ---- Main parser ---- */
974
+
975
+ tf_ir_plan *tf_dsl_parse(const char *text, size_t len, char **error) {
976
+ if (error) *error = NULL;
977
+
978
+ if (!text || len == 0) {
979
+ set_error(error, "empty pipeline");
980
+ return NULL;
981
+ }
982
+
983
+ /* Split into stages */
984
+ token_list stages;
985
+ if (split_stages(text, len, &stages) != 0 || stages.count == 0) {
986
+ set_error(error, "empty pipeline");
987
+ tl_free(&stages);
988
+ return NULL;
989
+ }
990
+
991
+ tf_ir_plan *plan = tf_ir_plan_create();
992
+ if (!plan) {
993
+ set_error(error, "out of memory");
994
+ tl_free(&stages);
995
+ return NULL;
996
+ }
997
+
998
+ for (size_t i = 0; i < stages.count; i++) {
999
+ /* Tokenize this stage */
1000
+ token_list tokens;
1001
+ if (tokenize_stage(stages.items[i], &tokens) != 0 || tokens.count == 0) {
1002
+ set_errorf(error, "empty stage at position %zu", "");
1003
+ tl_free(&tokens);
1004
+ goto fail;
1005
+ }
1006
+
1007
+ const char *raw_op = tokens.items[0];
1008
+ int is_first = (i == 0);
1009
+ int is_last = (i == stages.count - 1);
1010
+
1011
+ /* Resolve op name */
1012
+ char *resolved = resolve_codec(raw_op, is_first, is_last);
1013
+ const char *op_name = resolved ? resolved : raw_op;
1014
+
1015
+ /* Build args based on op type */
1016
+ cJSON *args = NULL;
1017
+
1018
+ /* Check if it's a codec (starts with "codec." or resolved from shorthand) */
1019
+ if (strncmp(op_name, "codec.", 6) == 0) {
1020
+ args = build_codec_args(&tokens);
1021
+ } else if (strcmp(op_name, "filter") == 0) {
1022
+ args = build_filter_args(&tokens, error);
1023
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1024
+ } else if (strcmp(op_name, "select") == 0) {
1025
+ args = build_select_args(&tokens, error);
1026
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1027
+ } else if (strcmp(op_name, "rename") == 0) {
1028
+ args = build_rename_args(&tokens, error);
1029
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1030
+ } else if (strcmp(op_name, "head") == 0) {
1031
+ args = build_head_args(&tokens, error);
1032
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1033
+ } else if (strcmp(op_name, "skip") == 0) {
1034
+ args = build_skip_args(&tokens, error);
1035
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1036
+ } else if (strcmp(op_name, "derive") == 0) {
1037
+ args = build_derive_args(&tokens, error);
1038
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1039
+ } else if (strcmp(op_name, "stats") == 0) {
1040
+ args = build_stats_args(&tokens, error);
1041
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1042
+ } else if (strcmp(op_name, "unique") == 0) {
1043
+ args = build_unique_args(&tokens, error);
1044
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1045
+ } else if (strcmp(op_name, "sort") == 0) {
1046
+ args = build_sort_args(&tokens, error);
1047
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1048
+ } else if (strcmp(op_name, "reorder") == 0) {
1049
+ args = build_select_args(&tokens, error);
1050
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1051
+ } else if (strcmp(op_name, "dedup") == 0) {
1052
+ args = build_unique_args(&tokens, error);
1053
+ } else if (strcmp(op_name, "validate") == 0) {
1054
+ args = build_filter_args(&tokens, error);
1055
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1056
+ } else if (strcmp(op_name, "trim") == 0) {
1057
+ args = build_unique_args(&tokens, error);
1058
+ } else if (strcmp(op_name, "fill-null") == 0) {
1059
+ args = build_rename_args(&tokens, error);
1060
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1061
+ } else if (strcmp(op_name, "cast") == 0) {
1062
+ args = build_rename_args(&tokens, error);
1063
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1064
+ } else if (strcmp(op_name, "clip") == 0) {
1065
+ args = build_clip_args(&tokens, error);
1066
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1067
+ } else if (strcmp(op_name, "replace") == 0) {
1068
+ args = build_replace_args(&tokens, error);
1069
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1070
+ } else if (strcmp(op_name, "hash") == 0) {
1071
+ args = build_unique_args(&tokens, error);
1072
+ } else if (strcmp(op_name, "bin") == 0) {
1073
+ args = build_bin_args(&tokens, error);
1074
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1075
+ } else if (strcmp(op_name, "fill-down") == 0) {
1076
+ args = build_unique_args(&tokens, error);
1077
+ } else if (strcmp(op_name, "step") == 0) {
1078
+ args = build_step_args(&tokens, error);
1079
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1080
+ } else if (strcmp(op_name, "window") == 0) {
1081
+ args = build_window_args(&tokens, error);
1082
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1083
+ } else if (strcmp(op_name, "explode") == 0) {
1084
+ args = build_explode_args(&tokens, error);
1085
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1086
+ } else if (strcmp(op_name, "split") == 0) {
1087
+ args = build_split_args(&tokens, error);
1088
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1089
+ } else if (strcmp(op_name, "unpivot") == 0) {
1090
+ args = build_unpivot_args(&tokens, error);
1091
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1092
+ } else if (strcmp(op_name, "tail") == 0) {
1093
+ args = build_head_args(&tokens, error);
1094
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1095
+ } else if (strcmp(op_name, "top") == 0) {
1096
+ args = build_top_args(&tokens, error);
1097
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1098
+ } else if (strcmp(op_name, "sample") == 0) {
1099
+ args = build_head_args(&tokens, error);
1100
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1101
+ } else if (strcmp(op_name, "group-agg") == 0) {
1102
+ args = build_group_agg_args(&tokens, error);
1103
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1104
+ } else if (strcmp(op_name, "frequency") == 0) {
1105
+ args = build_frequency_args(&tokens, error);
1106
+ } else if (strcmp(op_name, "datetime") == 0) {
1107
+ args = build_datetime_args(&tokens, error);
1108
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1109
+ } else if (strcmp(op_name, "flatten") == 0) {
1110
+ args = build_flatten_args(&tokens, error);
1111
+ } else if (strcmp(op_name, "grep") == 0) {
1112
+ args = build_grep_args(&tokens, error);
1113
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1114
+ } else if (strcmp(op_name, "pivot") == 0) {
1115
+ args = build_pivot_args(&tokens, error);
1116
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1117
+ } else if (strcmp(op_name, "join") == 0) {
1118
+ args = build_join_args(&tokens, error);
1119
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1120
+ } else if (strcmp(op_name, "stack") == 0) {
1121
+ args = build_stack_args(&tokens, error);
1122
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1123
+ } else if (strcmp(op_name, "lead") == 0) {
1124
+ args = build_lead_args(&tokens, error);
1125
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1126
+ } else if (strcmp(op_name, "date-trunc") == 0) {
1127
+ args = build_date_trunc_args(&tokens, error);
1128
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1129
+ } else if (strcmp(op_name, "onehot") == 0) {
1130
+ args = build_onehot_args(&tokens, error);
1131
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1132
+ } else if (strcmp(op_name, "label-encode") == 0) {
1133
+ args = build_label_encode_args(&tokens, error);
1134
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1135
+ } else if (strcmp(op_name, "ewma") == 0) {
1136
+ args = build_ewma_args(&tokens, error);
1137
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1138
+ } else if (strcmp(op_name, "diff") == 0) {
1139
+ args = build_diff_args(&tokens, error);
1140
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1141
+ } else if (strcmp(op_name, "anomaly") == 0) {
1142
+ args = build_anomaly_args(&tokens, error);
1143
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1144
+ } else if (strcmp(op_name, "split-data") == 0) {
1145
+ args = build_split_data_args(&tokens, error);
1146
+ } else if (strcmp(op_name, "interpolate") == 0) {
1147
+ args = build_interpolate_args(&tokens, error);
1148
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1149
+ } else if (strcmp(op_name, "normalize") == 0) {
1150
+ args = build_normalize_args(&tokens, error);
1151
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1152
+ } else if (strcmp(op_name, "acf") == 0) {
1153
+ args = build_acf_args(&tokens, error);
1154
+ if (!args) { free(resolved); tl_free(&tokens); goto fail; }
1155
+ } else {
1156
+ /* Unknown op — pass through with codec-style args, let validation catch it */
1157
+ args = build_codec_args(&tokens);
1158
+ }
1159
+
1160
+ if (tf_ir_plan_add_node(plan, op_name, args) != 0) {
1161
+ set_error(error, "out of memory adding node");
1162
+ cJSON_Delete(args);
1163
+ free(resolved);
1164
+ tl_free(&tokens);
1165
+ goto fail;
1166
+ }
1167
+
1168
+ cJSON_Delete(args);
1169
+ free(resolved);
1170
+ tl_free(&tokens);
1171
+ }
1172
+
1173
+ tl_free(&stages);
1174
+ return plan;
1175
+
1176
+ fail:
1177
+ tf_ir_plan_free(plan);
1178
+ tl_free(&stages);
1179
+ return NULL;
1180
+ }