tranfi 0.0.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/LICENSE +177 -21
  2. package/NOTICE +8 -0
  3. package/README.md +627 -0
  4. package/app/assets/index-6quYZ5Ap.css +5 -0
  5. package/app/assets/index-BIAIKnrp.js +160 -0
  6. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  7. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  8. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  9. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  10. package/app/index.html +13 -0
  11. package/binding.gyp +121 -0
  12. package/csrc/arena.c +93 -0
  13. package/csrc/batch.c +976 -0
  14. package/csrc/buffer.c +154 -0
  15. package/csrc/cJSON.c +3386 -0
  16. package/csrc/cJSON.h +316 -0
  17. package/csrc/codec_csv.c +1951 -0
  18. package/csrc/codec_jsonl.c +1086 -0
  19. package/csrc/codec_table.c +248 -0
  20. package/csrc/codec_text.c +447 -0
  21. package/csrc/compiler.c +130 -0
  22. package/csrc/config.h +21 -0
  23. package/csrc/date_utils.h +94 -0
  24. package/csrc/dsl.c +5417 -0
  25. package/csrc/dsl.h +22 -0
  26. package/csrc/expr.c +1553 -0
  27. package/csrc/expr.h +58 -0
  28. package/csrc/internal.h +539 -0
  29. package/csrc/ir.c +166 -0
  30. package/csrc/ir.h +208 -0
  31. package/csrc/ir_schema.c +75 -0
  32. package/csrc/ir_serialize.c +166 -0
  33. package/csrc/ir_sql.c +1822 -0
  34. package/csrc/ir_validate.c +576 -0
  35. package/csrc/json_path.c +210 -0
  36. package/csrc/main.c +1241 -0
  37. package/csrc/memory_estimate.c +477 -0
  38. package/csrc/op_acf.c +283 -0
  39. package/csrc/op_across.c +477 -0
  40. package/csrc/op_anomaly.c +255 -0
  41. package/csrc/op_assert.c +761 -0
  42. package/csrc/op_bin.c +248 -0
  43. package/csrc/op_cast.c +523 -0
  44. package/csrc/op_clip.c +99 -0
  45. package/csrc/op_date_trunc.c +355 -0
  46. package/csrc/op_datetime.c +394 -0
  47. package/csrc/op_derive.c +216 -0
  48. package/csrc/op_diff.c +250 -0
  49. package/csrc/op_ewma.c +222 -0
  50. package/csrc/op_explode.c +206 -0
  51. package/csrc/op_fill_down.c +235 -0
  52. package/csrc/op_fill_null.c +268 -0
  53. package/csrc/op_filter.c +181 -0
  54. package/csrc/op_frequency.c +721 -0
  55. package/csrc/op_grep.c +181 -0
  56. package/csrc/op_group_agg.c +1956 -0
  57. package/csrc/op_hash.c +159 -0
  58. package/csrc/op_head.c +84 -0
  59. package/csrc/op_interpolate.c +445 -0
  60. package/csrc/op_join.c +2902 -0
  61. package/csrc/op_json_extract.c +227 -0
  62. package/csrc/op_json_filter.c +384 -0
  63. package/csrc/op_json_flatten.c +293 -0
  64. package/csrc/op_json_schema.c +503 -0
  65. package/csrc/op_label_encode.c +419 -0
  66. package/csrc/op_lag.c +181 -0
  67. package/csrc/op_lead.c +242 -0
  68. package/csrc/op_normalize.c +510 -0
  69. package/csrc/op_onehot.c +457 -0
  70. package/csrc/op_pivot.c +1754 -0
  71. package/csrc/op_quarantine.c +189 -0
  72. package/csrc/op_registry.c +3044 -0
  73. package/csrc/op_rename.c +129 -0
  74. package/csrc/op_replace.c +354 -0
  75. package/csrc/op_rleid.c +297 -0
  76. package/csrc/op_rowid.c +559 -0
  77. package/csrc/op_sample.c +158 -0
  78. package/csrc/op_schema.c +1341 -0
  79. package/csrc/op_schema_infer.c +252 -0
  80. package/csrc/op_select.c +340 -0
  81. package/csrc/op_set.c +3449 -0
  82. package/csrc/op_skip.c +95 -0
  83. package/csrc/op_sort.c +819 -0
  84. package/csrc/op_source_name.c +120 -0
  85. package/csrc/op_split.c +151 -0
  86. package/csrc/op_split_data.c +119 -0
  87. package/csrc/op_stack.c +271 -0
  88. package/csrc/op_stats.c +875 -0
  89. package/csrc/op_step.c +333 -0
  90. package/csrc/op_tail.c +105 -0
  91. package/csrc/op_tee.c +338 -0
  92. package/csrc/op_top.c +357 -0
  93. package/csrc/op_trim.c +138 -0
  94. package/csrc/op_unique.c +1343 -0
  95. package/csrc/op_unpivot.c +193 -0
  96. package/csrc/op_validate.c +648 -0
  97. package/csrc/op_window.c +591 -0
  98. package/csrc/path_policy.c +85 -0
  99. package/csrc/pipeline.c +1088 -0
  100. package/csrc/recipes.c +104 -0
  101. package/csrc/recipes.h +27 -0
  102. package/csrc/report.c +506 -0
  103. package/csrc/report.h +22 -0
  104. package/csrc/selector.c +1097 -0
  105. package/csrc/size_utils.c +348 -0
  106. package/csrc/spill.c +317 -0
  107. package/csrc/spill.h +21 -0
  108. package/csrc/tranfi.h +291 -0
  109. package/csrc/transform.h +209 -0
  110. package/csrc/transform_api.c +2237 -0
  111. package/csrc/transform_categorical.c +923 -0
  112. package/csrc/transform_internal.h +472 -0
  113. package/csrc/transform_json.c +3812 -0
  114. package/csrc/transform_numeric.c +1966 -0
  115. package/csrc/transform_sha256.c +154 -0
  116. package/csrc/transform_wasm.h +162 -0
  117. package/csrc/transform_wasm_api.c +1373 -0
  118. package/csrc/wasm_api.c +218 -0
  119. package/napi_api.c +534 -0
  120. package/napi_transform.c +1648 -0
  121. package/napi_transform.h +8 -0
  122. package/package.json +64 -59
  123. package/scripts/install-native.js +76 -0
  124. package/scripts/prepack.js +64 -0
  125. package/scripts/sync-csrc.js +23 -0
  126. package/src/cli.js +190 -0
  127. package/src/engines/duckdb.js +142 -0
  128. package/src/index.js +925 -0
  129. package/src/memory_policy.js +411 -0
  130. package/src/native.js +18 -0
  131. package/src/pipeline.js +709 -0
  132. package/src/recipe_json.js +80 -0
  133. package/src/server.js +279 -0
  134. package/src/transform.js +403 -0
  135. package/src/transform_error.js +10 -0
  136. package/src/wasm.js +21 -0
  137. package/wasm/index.js +732 -0
  138. package/wasm/package.json +1 -0
  139. package/wasm/tranfi_core.js +0 -0
  140. package/wasm/transform.js +1156 -0
  141. package/wasm/worker.js +786 -0
  142. package/dist/bundle.js +0 -1
  143. package/index.html +0 -18
  144. package/src/app.css +0 -169
  145. package/src/app.js +0 -203
  146. package/src/app.vue +0 -250
  147. package/src/bulma-input.vue +0 -110
  148. package/src/common-inputs.js +0 -28
  149. package/src/main.js +0 -20
  150. package/src/transforms.js +0 -166
  151. package/webpack.config.js +0 -108
@@ -0,0 +1,1088 @@
1
+ /*
2
+ * pipeline.c — Pipeline orchestrator.
3
+ *
4
+ * Creates a pipeline from a JSON plan, streams bytes through
5
+ * decode → steps → encode, and routes output to channels.
6
+ */
7
+
8
+ #include "internal.h"
9
+ #include "dsl.h"
10
+ #include "cJSON.h"
11
+ #include <stdlib.h>
12
+ #include <string.h>
13
+ #include <stdio.h>
14
+ #include <errno.h>
15
+ #ifndef _WIN32
16
+ #include <unistd.h>
17
+ #endif
18
+
19
+ #define TRANFI_VERSION "0.2.0"
20
+
21
+ enum {
22
+ TF_FINISH_PHASE_DECODER = 0,
23
+ TF_FINISH_PHASE_STEPS = 1,
24
+ TF_FINISH_PHASE_ENCODER = 2,
25
+ TF_FINISH_PHASE_STATS = 3,
26
+ TF_FINISH_PHASE_DONE = 4
27
+ };
28
+
29
+ #define TF_LAST_ERROR_CAP 1024
30
+
31
+ #if defined(_MSC_VER)
32
+ static __declspec(thread) char g_last_error[TF_LAST_ERROR_CAP];
33
+ #else
34
+ static _Thread_local char g_last_error[TF_LAST_ERROR_CAP];
35
+ #endif
36
+
37
+ void tf_set_last_error(const char *msg) {
38
+ if (!msg || !*msg) {
39
+ g_last_error[0] = '\0';
40
+ return;
41
+ }
42
+ snprintf(g_last_error, sizeof(g_last_error), "%s", msg);
43
+ }
44
+
45
+ const char *tf_last_error(void) {
46
+ return g_last_error[0] ? g_last_error : NULL;
47
+ }
48
+
49
+ static void pipeline_set_error(tf_pipeline *p, const char *fallback) {
50
+ if (!p) return;
51
+ const char *last = tf_last_error();
52
+ free(p->error);
53
+ p->error = strdup((last && *last) ? last : fallback);
54
+ }
55
+
56
+ static int pipeline_fail(tf_pipeline *p, const char *fallback) {
57
+ pipeline_set_error(p, fallback);
58
+ return TF_ERROR;
59
+ }
60
+
61
+ static void set_last_schema_inference_error(void) {
62
+ const char *detail = tf_last_error();
63
+ if (detail && detail[0]) {
64
+ char buf[TF_LAST_ERROR_CAP];
65
+ snprintf(buf, sizeof(buf), "schema inference failed: %s", detail);
66
+ tf_set_last_error(buf);
67
+ } else {
68
+ tf_set_last_error("schema inference failed");
69
+ }
70
+ }
71
+
72
+ static char *schema_inference_error_string(void) {
73
+ const char *detail = tf_last_error();
74
+ if (detail && detail[0]) {
75
+ char buf[TF_LAST_ERROR_CAP];
76
+ snprintf(buf, sizeof(buf), "schema inference failed: %s", detail);
77
+ return strdup(buf);
78
+ }
79
+ return strdup("schema inference failed");
80
+ }
81
+
82
+ static int pipeline_fail_msg(tf_pipeline *p, const char *msg) {
83
+ if (p) {
84
+ free(p->error);
85
+ p->error = strdup(msg ? msg : "pipeline error");
86
+ }
87
+ return TF_ERROR;
88
+ }
89
+
90
+ static void pipeline_clear_finish_batches(tf_pipeline *p) {
91
+ if (!p || !p->finish_batches) return;
92
+ for (size_t i = p->finish_batch_index; i < p->finish_n_batches; i++) {
93
+ if (p->finish_batches[i]) tf_batch_free(p->finish_batches[i]);
94
+ }
95
+ free(p->finish_batches);
96
+ p->finish_batches = NULL;
97
+ p->finish_n_batches = 0;
98
+ p->finish_batch_index = 0;
99
+ }
100
+
101
+ static int pipeline_drain_channel_to_sink(tf_pipeline *p, int channel,
102
+ tf_pipeline_sink_fn sink, void *user) {
103
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS || !sink) return TF_ERROR;
104
+ uint8_t buf[64 * 1024];
105
+ for (;;) {
106
+ size_t n = tf_buffer_read(&p->output[channel], buf, sizeof(buf));
107
+ if (n == 0) break;
108
+ if (sink(channel, buf, n, user) != TF_OK) {
109
+ free(p->error);
110
+ p->error = strdup("sink callback failed");
111
+ return TF_ERROR;
112
+ }
113
+ }
114
+ return TF_OK;
115
+ }
116
+
117
+ static int pipeline_auto_drain_sinks(tf_pipeline *p) {
118
+ if (!p) return TF_ERROR;
119
+ for (int channel = 0; channel < TF_NUM_CHANNELS; channel++) {
120
+ if (p->sinks[channel]) {
121
+ if (pipeline_drain_channel_to_sink(p, channel, p->sinks[channel],
122
+ p->sink_users[channel]) != TF_OK) {
123
+ return TF_ERROR;
124
+ }
125
+ }
126
+ }
127
+ return TF_OK;
128
+ }
129
+
130
+ const char *tf_version(void) {
131
+ return TRANFI_VERSION;
132
+ }
133
+
134
+ static void step_stats_free(tf_step_run_stats *stats, size_t n) {
135
+ if (!stats) return;
136
+ for (size_t i = 0; i < n; i++) {
137
+ free(stats[i].op);
138
+ free(stats[i].state_estimate);
139
+ free(stats[i].state_bytes_reason);
140
+ free(stats[i].execution_target);
141
+ }
142
+ free(stats);
143
+ }
144
+
145
+ static int node_has_spill_dir(const tf_ir_node *node) {
146
+ cJSON *spill = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, "spill_dir") : NULL;
147
+ return cJSON_IsString(spill) && spill->valuestring && spill->valuestring[0];
148
+ }
149
+
150
+
151
+ static int node_op_is_join_spillable(const tf_ir_node *node) {
152
+ if (!node || !node->op) return 0;
153
+ if (strcmp(node->op, "semi-join") == 0 || strcmp(node->op, "anti-join") == 0) return 1;
154
+ if (strcmp(node->op, "join") != 0) return 0;
155
+ if (!node->args) return 1;
156
+ cJSON *how = cJSON_GetObjectItemCaseSensitive(node->args, "how");
157
+ if (!cJSON_IsString(how) || !how->valuestring) return 1;
158
+ return strcmp(how->valuestring, "inner") == 0 || strcmp(how->valuestring, "left") == 0 ||
159
+ strcmp(how->valuestring, "semi") == 0 || strcmp(how->valuestring, "anti") == 0;
160
+ }
161
+
162
+ static int node_op_is_row_set_spillable(const tf_ir_node *node) {
163
+ return node && node->op &&
164
+ (strcmp(node->op, "intersect") == 0 || strcmp(node->op, "setdiff") == 0 ||
165
+ strcmp(node->op, "intersect-all") == 0 || strcmp(node->op, "setdiff-all") == 0 ||
166
+ strcmp(node->op, "union") == 0);
167
+ }
168
+
169
+ static int node_op_uses_native_spill(const tf_ir_node *node) {
170
+ return node && node->op && node_has_spill_dir(node) &&
171
+ (strcmp(node->op, "sort") == 0 ||
172
+ strcmp(node->op, "unique") == 0 ||
173
+ strcmp(node->op, "dedup") == 0 ||
174
+ strcmp(node->op, "group-agg") == 0 ||
175
+ strcmp(node->op, "pivot") == 0 ||
176
+ node_op_is_join_spillable(node) ||
177
+ node_op_is_row_set_spillable(node));
178
+ }
179
+
180
+ static const char *runtime_step_execution_target(const tf_ir_node *node) {
181
+ #ifdef __EMSCRIPTEN__
182
+ (void)node;
183
+ return "wasm";
184
+ #else
185
+ if (node_op_uses_native_spill(node)) return "native_spill";
186
+ return "native";
187
+ #endif
188
+ }
189
+
190
+ static int build_step_stats_from_plan(const tf_ir_plan *plan,
191
+ tf_step_run_stats **out_stats,
192
+ size_t *out_n) {
193
+ if (!out_stats || !out_n) return TF_ERROR;
194
+ *out_stats = NULL;
195
+ *out_n = 0;
196
+ if (!plan) return TF_OK;
197
+
198
+ size_t count = 0;
199
+ for (size_t i = 0; i < plan->n_nodes; i++) {
200
+ const tf_op_entry *entry = tf_op_registry_find(plan->nodes[i].op);
201
+ if (entry && entry->kind == TF_OP_TRANSFORM && entry->create_native) count++;
202
+ }
203
+ if (count == 0) return TF_OK;
204
+
205
+ tf_step_run_stats *stats = calloc(count, sizeof(tf_step_run_stats));
206
+ if (!stats) return TF_ERROR;
207
+
208
+ size_t j = 0;
209
+ for (size_t i = 0; i < plan->n_nodes; i++) {
210
+ const tf_ir_node *node = &plan->nodes[i];
211
+ const tf_op_entry *entry = tf_op_registry_find(node->op);
212
+ if (!entry || entry->kind != TF_OP_TRANSFORM || !entry->create_native) continue;
213
+
214
+ stats[j].op = strdup(node->op);
215
+ stats[j].state_estimate = strdup(node->state_estimate ? node->state_estimate : "unknown");
216
+ stats[j].execution_target = strdup(runtime_step_execution_target(node));
217
+ if (!stats[j].op || !stats[j].state_estimate || !stats[j].execution_target) {
218
+ step_stats_free(stats, count);
219
+ return TF_ERROR;
220
+ }
221
+ stats[j].node_index = node->index;
222
+ stats[j].memory_class = node->memory_class;
223
+ stats[j].emit_class = node->emit_class;
224
+ stats[j].schema_class = node->schema_class;
225
+ if (node->memory_class == TF_MEM_BLOCKING) stats[j].warnings |= TF_STEP_WARN_BLOCKING;
226
+ if (node->emit_class == TF_EMIT_ON_FLUSH) stats[j].warnings |= TF_STEP_WARN_FLUSH_LATENT;
227
+ if (node->schema_class == TF_SCHEMA_DATA_DEPENDENT) stats[j].warnings |= TF_STEP_WARN_DATA_DEPENDENT_SCHEMA;
228
+
229
+ char reason[192] = {0};
230
+ size_t estimate = 0;
231
+ if (tf_estimate_step_state_bytes(node, &estimate, reason, sizeof(reason))) {
232
+ stats[j].has_state_bytes_estimate = 1;
233
+ stats[j].state_bytes_estimate = estimate;
234
+ } else if (node->memory_class == TF_MEM_KEY_STATE || node->memory_class == TF_MEM_BLOCKING) {
235
+ stats[j].warnings |= TF_STEP_WARN_UNBOUNDED_STATE;
236
+ stats[j].state_bytes_reason = strdup(reason[0] ? reason : "no native byte estimator");
237
+ if (!stats[j].state_bytes_reason) {
238
+ step_stats_free(stats, count);
239
+ return TF_ERROR;
240
+ }
241
+ }
242
+ j++;
243
+ }
244
+
245
+ *out_stats = stats;
246
+ *out_n = count;
247
+ return TF_OK;
248
+ }
249
+
250
+ static void step_stats_record_input(tf_pipeline *p, size_t step_idx, size_t rows) {
251
+ if (!p || step_idx >= p->n_step_stats) return;
252
+ p->step_stats[step_idx].batches_in++;
253
+ p->step_stats[step_idx].rows_in += rows;
254
+ }
255
+
256
+ static void step_stats_record_output(tf_pipeline *p, size_t step_idx, const tf_batch *batch) {
257
+ if (!p || step_idx >= p->n_step_stats || !batch) return;
258
+ p->step_stats[step_idx].batches_out++;
259
+ p->step_stats[step_idx].rows_out += batch->n_rows;
260
+ }
261
+
262
+ static int step_stats_write_warning_token(tf_pipeline *p, const char *token, int *first) {
263
+ char buf[96];
264
+ snprintf(buf, sizeof(buf), "%s\"%s\"", *first ? "" : ",", token);
265
+ *first = 0;
266
+ return tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf);
267
+ }
268
+
269
+ static int step_stats_write_warnings(tf_pipeline *p, uint32_t warnings) {
270
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], "\"warnings\":[") != TF_OK)
271
+ return TF_ERROR;
272
+ int first = 1;
273
+ if ((warnings & TF_STEP_WARN_BLOCKING) &&
274
+ step_stats_write_warning_token(p, "blocking", &first) != TF_OK)
275
+ return TF_ERROR;
276
+ if ((warnings & TF_STEP_WARN_FLUSH_LATENT) &&
277
+ step_stats_write_warning_token(p, "flush_latent", &first) != TF_OK)
278
+ return TF_ERROR;
279
+ if ((warnings & TF_STEP_WARN_UNBOUNDED_STATE) &&
280
+ step_stats_write_warning_token(p, "unbounded_state", &first) != TF_OK)
281
+ return TF_ERROR;
282
+ if ((warnings & TF_STEP_WARN_DATA_DEPENDENT_SCHEMA) &&
283
+ step_stats_write_warning_token(p, "data_dependent_schema", &first) != TF_OK)
284
+ return TF_ERROR;
285
+ return tf_buffer_write_str(&p->output[TF_CHAN_STATS], "],");
286
+ }
287
+
288
+ static int emit_step_stats(tf_pipeline *p) {
289
+ if (!p) return TF_ERROR;
290
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], "{\"type\":\"step_stats\",\"steps\":[") != TF_OK)
291
+ return TF_ERROR;
292
+ for (size_t i = 0; i < p->n_step_stats; i++) {
293
+ tf_step_run_stats *st = &p->step_stats[i];
294
+ char buf[1024];
295
+ snprintf(buf, sizeof(buf),
296
+ "%s{\"index\":%zu,\"op\":\"%s\",\"execution_target\":\"%s\","
297
+ "\"memory_class\":\"%s\",\"emit_class\":\"%s\","
298
+ "\"schema_class\":\"%s\",\"state_estimate\":\"%s\",",
299
+ i ? "," : "",
300
+ st->node_index,
301
+ st->op ? st->op : "unknown",
302
+ st->execution_target ? st->execution_target : "native",
303
+ tf_memory_class_name(st->memory_class),
304
+ tf_emit_class_name(st->emit_class),
305
+ tf_schema_class_name(st->schema_class),
306
+ st->state_estimate ? st->state_estimate : "unknown");
307
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
308
+ return TF_ERROR;
309
+ if (st->has_state_bytes_estimate) {
310
+ snprintf(buf, sizeof(buf), "\"state_bytes_estimate\":%zu,", st->state_bytes_estimate);
311
+ } else {
312
+ snprintf(buf, sizeof(buf), "\"state_bytes_estimate\":null,");
313
+ }
314
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
315
+ return TF_ERROR;
316
+ if (st->state_bytes_reason) {
317
+ snprintf(buf, sizeof(buf), "\"state_bytes_reason\":\"%s\",", st->state_bytes_reason);
318
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
319
+ return TF_ERROR;
320
+ }
321
+ if (step_stats_write_warnings(p, st->warnings) != TF_OK)
322
+ return TF_ERROR;
323
+ snprintf(buf, sizeof(buf),
324
+ "\"batches_in\":%zu,\"batches_out\":%zu,\"rows_in\":%zu,\"rows_out\":%zu",
325
+ st->batches_in, st->batches_out, st->rows_in, st->rows_out);
326
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
327
+ return TF_ERROR;
328
+ if (i < p->n_steps && p->steps[i] && p->steps[i]->append_stats &&
329
+ p->steps[i]->append_stats(p->steps[i], &p->output[TF_CHAN_STATS]) != TF_OK)
330
+ return TF_ERROR;
331
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], "}") != TF_OK)
332
+ return TF_ERROR;
333
+ }
334
+ return tf_buffer_write_str(&p->output[TF_CHAN_STATS], "]}\n");
335
+ }
336
+
337
+ static tf_pipeline *assemble_pipeline(tf_decoder *decoder, tf_step **steps,
338
+ size_t n_steps,
339
+ tf_step_run_stats *step_stats,
340
+ size_t n_step_stats,
341
+ tf_encoder *encoder) {
342
+ if (!decoder || !encoder) {
343
+ if (decoder) decoder->destroy(decoder);
344
+ if (encoder) encoder->destroy(encoder);
345
+ for (size_t i = 0; i < n_steps; i++) {
346
+ if (steps && steps[i]) steps[i]->destroy(steps[i]);
347
+ }
348
+ free(steps);
349
+ step_stats_free(step_stats, n_step_stats);
350
+ tf_set_last_error(!decoder ? "plan missing decoder" : "plan missing encoder");
351
+ return NULL;
352
+ }
353
+
354
+ tf_pipeline *p = calloc(1, sizeof(tf_pipeline));
355
+ if (!p) {
356
+ if (decoder) decoder->destroy(decoder);
357
+ if (encoder) encoder->destroy(encoder);
358
+ for (size_t i = 0; i < n_steps; i++) {
359
+ if (steps && steps[i]) steps[i]->destroy(steps[i]);
360
+ }
361
+ free(steps);
362
+ step_stats_free(step_stats, n_step_stats);
363
+ tf_set_last_error("out of memory");
364
+ return NULL;
365
+ }
366
+
367
+ p->decoder = decoder;
368
+ p->steps = steps;
369
+ p->n_steps = n_steps;
370
+ p->step_stats = step_stats;
371
+ p->n_step_stats = n_step_stats;
372
+ p->encoder = encoder;
373
+ p->encode_output = 1;
374
+
375
+ for (int i = 0; i < TF_NUM_CHANNELS; i++) {
376
+ tf_buffer_init(&p->output[i]);
377
+ }
378
+
379
+ /* Wire up side channels */
380
+ p->side.errors = &p->output[TF_CHAN_ERRORS];
381
+ p->side.stats = &p->output[TF_CHAN_STATS];
382
+ p->side.samples = &p->output[TF_CHAN_SAMPLES];
383
+ p->side.source_name = p->source_name;
384
+
385
+ return p;
386
+ }
387
+
388
+ static tf_pipeline *pipeline_create_from_owned_ir(tf_ir_plan *ir, const tf_host_policy *policy) {
389
+ char *error = NULL;
390
+
391
+ if (tf_ir_validate_with_host_policy(ir, policy) != TF_OK) {
392
+ tf_set_last_error(ir->error ? ir->error : "validation failed");
393
+ tf_ir_plan_free(ir);
394
+ return NULL;
395
+ }
396
+
397
+ tf_set_last_error(NULL);
398
+ if (tf_ir_infer_schema(ir) != TF_OK) {
399
+ set_last_schema_inference_error();
400
+ tf_ir_plan_free(ir);
401
+ return NULL;
402
+ }
403
+
404
+ tf_step_run_stats *step_stats = NULL;
405
+ size_t n_step_stats = 0;
406
+ if (build_step_stats_from_plan(ir, &step_stats, &n_step_stats) != TF_OK) {
407
+ tf_set_last_error("out of memory");
408
+ tf_ir_plan_free(ir);
409
+ return NULL;
410
+ }
411
+
412
+ tf_decoder *decoder = NULL;
413
+ tf_step **steps = NULL;
414
+ size_t n_steps = 0;
415
+ tf_encoder *encoder = NULL;
416
+ if (tf_compile_native(ir, &decoder, &steps, &n_steps, &encoder, &error) != TF_OK) {
417
+ tf_set_last_error(error ? error : "compilation failed");
418
+ free(error);
419
+ step_stats_free(step_stats, n_step_stats);
420
+ tf_ir_plan_free(ir);
421
+ return NULL;
422
+ }
423
+
424
+ tf_ir_plan_free(ir);
425
+ return assemble_pipeline(decoder, steps, n_steps, step_stats, n_step_stats, encoder);
426
+ }
427
+
428
+ tf_pipeline *tf_pipeline_create_with_host_policy(const char *plan_json, size_t len,
429
+ const tf_host_policy *policy) {
430
+ if (!plan_json || len == 0) {
431
+ tf_set_last_error("empty plan");
432
+ return NULL;
433
+ }
434
+
435
+ char *error = NULL;
436
+ tf_ir_plan *ir = tf_ir_from_json(plan_json, len, &error);
437
+ if (!ir) {
438
+ tf_set_last_error(error ? error : "failed to parse plan");
439
+ free(error);
440
+ return NULL;
441
+ }
442
+ return pipeline_create_from_owned_ir(ir, policy);
443
+ }
444
+
445
+ tf_pipeline *tf_pipeline_create(const char *plan_json, size_t len) {
446
+ return tf_pipeline_create_with_host_policy(plan_json, len, NULL);
447
+ }
448
+
449
+ tf_pipeline *tf_pipeline_create_from_ir_with_host_policy(const tf_ir_plan *plan,
450
+ const tf_host_policy *policy) {
451
+ if (!plan) {
452
+ tf_set_last_error("NULL IR plan");
453
+ return NULL;
454
+ }
455
+ tf_ir_plan *copy = tf_ir_plan_clone(plan);
456
+ if (!copy) {
457
+ tf_set_last_error("out of memory");
458
+ return NULL;
459
+ }
460
+ return pipeline_create_from_owned_ir(copy, policy);
461
+ }
462
+
463
+ tf_pipeline *tf_pipeline_create_from_ir(const tf_ir_plan *plan) {
464
+ return tf_pipeline_create_from_ir_with_host_policy(plan, NULL);
465
+ }
466
+
467
+ /* Public IR wrappers (thin forwarding to ir.h functions) */
468
+ tf_ir_plan *tf_ir_plan_from_json(const char *json, size_t len, char **error) {
469
+ return tf_ir_from_json(json, len, error);
470
+ }
471
+ char *tf_ir_plan_to_json(const tf_ir_plan *plan) {
472
+ return tf_ir_to_json(plan);
473
+ }
474
+ int tf_ir_plan_validate(tf_ir_plan *plan) {
475
+ return tf_ir_validate(plan);
476
+ }
477
+ int tf_ir_plan_validate_with_host_policy(tf_ir_plan *plan, const tf_host_policy *policy) {
478
+ return tf_ir_validate_with_host_policy(plan, policy);
479
+ }
480
+ int tf_ir_plan_infer_schema(tf_ir_plan *plan) {
481
+ return tf_ir_infer_schema(plan);
482
+ }
483
+ void tf_ir_plan_destroy(tf_ir_plan *plan) {
484
+ tf_ir_plan_free(plan);
485
+ }
486
+
487
+ static const char *pipeline_phase_name(const tf_pipeline *p) {
488
+ if (!p) return "unknown";
489
+ if (p->finished || p->finish_phase == TF_FINISH_PHASE_DONE) return "done";
490
+ if (!p->finish_started) return "push";
491
+ switch (p->finish_phase) {
492
+ case TF_FINISH_PHASE_DECODER: return "decoder_flush";
493
+ case TF_FINISH_PHASE_STEPS: return "step_flush";
494
+ case TF_FINISH_PHASE_ENCODER: return "encoder_flush";
495
+ case TF_FINISH_PHASE_STATS: return "stats";
496
+ case TF_FINISH_PHASE_DONE: return "done";
497
+ default: return "unknown";
498
+ }
499
+ }
500
+
501
+ static int pipeline_report_progress(tf_pipeline *p, int force) {
502
+ if (!p || !p->progress_cb) return TF_OK;
503
+ if (!force && p->progress_interval_rows > 0) {
504
+ size_t in_delta = p->rows_in >= p->progress_last_rows_in
505
+ ? p->rows_in - p->progress_last_rows_in : 0;
506
+ size_t out_delta = p->rows_out >= p->progress_last_rows_out
507
+ ? p->rows_out - p->progress_last_rows_out : 0;
508
+ if (in_delta < p->progress_interval_rows && out_delta < p->progress_interval_rows) {
509
+ return TF_OK;
510
+ }
511
+ }
512
+
513
+ tf_pipeline_progress progress = {
514
+ p->bytes_in,
515
+ p->bytes_out,
516
+ p->rows_in,
517
+ p->rows_out,
518
+ p->batches_in,
519
+ p->batches_out,
520
+ pipeline_phase_name(p),
521
+ p->finished ? 1 : 0,
522
+ };
523
+ if (p->progress_cb(&progress, p->progress_user) != TF_OK) {
524
+ tf_set_last_error("progress callback failed");
525
+ free(p->error);
526
+ p->error = strdup("progress callback failed");
527
+ return TF_ERROR;
528
+ }
529
+ p->progress_last_rows_in = p->rows_in;
530
+ p->progress_last_rows_out = p->rows_out;
531
+ return TF_OK;
532
+ }
533
+
534
+ static int emit_output_batch(tf_pipeline *p, tf_batch *batch) {
535
+ if (!p || !batch) return TF_OK;
536
+
537
+ if (batch->n_rows > 0) {
538
+ p->rows_out += batch->n_rows;
539
+ p->batches_out++;
540
+
541
+ if (p->batch_sink && p->batch_sink(batch, p->batch_sink_user) != TF_OK) {
542
+ tf_set_last_error("batch sink callback failed");
543
+ free(p->error);
544
+ p->error = strdup("batch sink callback failed");
545
+ return TF_ERROR;
546
+ }
547
+ }
548
+
549
+ if (!p->encode_output) return TF_OK;
550
+
551
+ tf_set_last_error(NULL);
552
+ size_t before = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
553
+ int rc = p->encoder->encode(p->encoder, batch, &p->output[TF_CHAN_MAIN]);
554
+ if (rc == TF_OK) {
555
+ size_t after = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
556
+ if (after >= before) p->bytes_out += after - before;
557
+ }
558
+ return rc;
559
+ }
560
+
561
+ /*
562
+ * Process a batch through all steps, then encode.
563
+ */
564
+ static int process_batch(tf_pipeline *p, tf_batch *batch) {
565
+ tf_batch *current = batch;
566
+ int batch_owned = 0; /* 0 = still owned by caller (decoder) */
567
+
568
+ p->rows_in += current->n_rows;
569
+ p->batches_in++;
570
+
571
+ for (size_t i = 0; i < p->n_steps; i++) {
572
+ tf_batch *next = NULL;
573
+ step_stats_record_input(p, i, current->n_rows);
574
+ tf_set_last_error(NULL);
575
+ int rc = p->steps[i]->process(p->steps[i], current, &next, &p->side);
576
+
577
+ if (batch_owned) tf_batch_free(current);
578
+ batch_owned = 1;
579
+
580
+ if (rc != TF_OK) return TF_ERROR;
581
+ step_stats_record_output(p, i, next);
582
+ if (!next) return TF_OK; /* filtered away entirely */
583
+ current = next;
584
+ }
585
+
586
+ /* Emit transformed output batch. */
587
+ int rc = emit_output_batch(p, current);
588
+ if (batch_owned) tf_batch_free(current);
589
+ return rc;
590
+ }
591
+
592
+ static int route_flushed_batch(tf_pipeline *p, size_t source_step, tf_batch *flushed) {
593
+ if (!flushed) return TF_OK;
594
+ step_stats_record_output(p, source_step, flushed);
595
+
596
+ /* Run flushed batch through remaining steps. */
597
+ tf_batch *current = flushed;
598
+ int owned = 1;
599
+ int rc = TF_OK;
600
+ for (size_t j = source_step + 1; j < p->n_steps; j++) {
601
+ tf_batch *next = NULL;
602
+ step_stats_record_input(p, j, current->n_rows);
603
+ tf_set_last_error(NULL);
604
+ rc = p->steps[j]->process(p->steps[j], current, &next, &p->side);
605
+ if (owned) tf_batch_free(current);
606
+ owned = 1;
607
+ if (rc != TF_OK) return pipeline_fail(p, "processing error");
608
+ step_stats_record_output(p, j, next);
609
+ if (!next) { current = NULL; break; }
610
+ current = next;
611
+ }
612
+
613
+ if (current) {
614
+ rc = emit_output_batch(p, current);
615
+ if (rc != TF_OK) {
616
+ if (current && owned) tf_batch_free(current);
617
+ return pipeline_fail(p, "encode error");
618
+ }
619
+ }
620
+ if (current && owned) tf_batch_free(current);
621
+ if (pipeline_auto_drain_sinks(p) != TF_OK) return TF_ERROR;
622
+ if (pipeline_report_progress(p, 0) != TF_OK) return TF_ERROR;
623
+ return TF_OK;
624
+ }
625
+
626
+ int tf_pipeline_push(tf_pipeline *p, const uint8_t *data, size_t len) {
627
+ if (!p || p->finished || p->finish_started) return TF_ERROR;
628
+
629
+ p->bytes_in += len;
630
+
631
+ /* Decode bytes into batches */
632
+ tf_batch **batches = NULL;
633
+ size_t n_batches = 0;
634
+ tf_set_last_error(NULL);
635
+ int rc = p->decoder->decode(p->decoder, data, len, &batches, &n_batches, &p->side);
636
+ if (rc != TF_OK) {
637
+ tf_batch_array_free(batches, n_batches);
638
+ return pipeline_fail(p, "decode error");
639
+ }
640
+
641
+ /* Process each batch */
642
+ for (size_t i = 0; i < n_batches; i++) {
643
+ rc = process_batch(p, batches[i]);
644
+ tf_batch_free(batches[i]);
645
+ if (rc != TF_OK) {
646
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
647
+ free(batches);
648
+ return p->error ? TF_ERROR : pipeline_fail(p, "processing error");
649
+ }
650
+ if (pipeline_auto_drain_sinks(p) != TF_OK) {
651
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
652
+ free(batches);
653
+ return TF_ERROR;
654
+ }
655
+ if (pipeline_report_progress(p, 0) != TF_OK) {
656
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
657
+ free(batches);
658
+ return TF_ERROR;
659
+ }
660
+ }
661
+ free(batches);
662
+
663
+ return TF_OK;
664
+ }
665
+
666
+ int tf_pipeline_finish_step(tf_pipeline *p) {
667
+ if (!p) return TF_ERROR;
668
+ if (p->finished || p->finish_phase == TF_FINISH_PHASE_DONE) return TF_DONE;
669
+
670
+ if (!p->finish_started) {
671
+ p->finish_started = 1;
672
+ p->finish_phase = TF_FINISH_PHASE_DECODER;
673
+ p->finish_batch_index = 0;
674
+ p->finish_n_batches = 0;
675
+ p->finish_batches = NULL;
676
+ p->finish_step_index = 0;
677
+ p->finish_step_in_next = 0;
678
+
679
+ tf_set_last_error(NULL);
680
+ int rc = p->decoder->flush(p->decoder, &p->finish_batches,
681
+ &p->finish_n_batches, &p->side);
682
+ if (rc != TF_OK) {
683
+ pipeline_clear_finish_batches(p);
684
+ return pipeline_fail(p, "decode flush error");
685
+ }
686
+ }
687
+
688
+ for (;;) {
689
+ if (p->finish_phase == TF_FINISH_PHASE_DECODER) {
690
+ if (p->finish_batch_index < p->finish_n_batches) {
691
+ tf_batch *batch = p->finish_batches[p->finish_batch_index];
692
+ p->finish_batches[p->finish_batch_index] = NULL;
693
+ p->finish_batch_index++;
694
+
695
+ int rc = process_batch(p, batch);
696
+ tf_batch_free(batch);
697
+ if (rc != TF_OK) {
698
+ pipeline_clear_finish_batches(p);
699
+ return p->error ? TF_ERROR : pipeline_fail(p, "processing error");
700
+ }
701
+ if (pipeline_auto_drain_sinks(p) != TF_OK) {
702
+ pipeline_clear_finish_batches(p);
703
+ return TF_ERROR;
704
+ }
705
+ if (pipeline_report_progress(p, 0) != TF_OK) {
706
+ pipeline_clear_finish_batches(p);
707
+ return TF_ERROR;
708
+ }
709
+ return TF_OK;
710
+ }
711
+ pipeline_clear_finish_batches(p);
712
+ p->finish_phase = TF_FINISH_PHASE_STEPS;
713
+ continue;
714
+ }
715
+
716
+ if (p->finish_phase == TF_FINISH_PHASE_STEPS) {
717
+ if (p->finish_step_index >= p->n_steps) {
718
+ p->finish_phase = TF_FINISH_PHASE_ENCODER;
719
+ continue;
720
+ }
721
+
722
+ tf_step *step = p->steps[p->finish_step_index];
723
+ tf_batch *flushed = NULL;
724
+ int rc = TF_OK;
725
+
726
+ if (!p->finish_step_in_next) {
727
+ tf_set_last_error(NULL);
728
+ rc = step->flush(step, &flushed, &p->side);
729
+ if (rc != TF_OK) return pipeline_fail(p, "flush error");
730
+ p->finish_step_in_next = 1;
731
+ if (flushed) {
732
+ if (route_flushed_batch(p, p->finish_step_index, flushed) != TF_OK) return TF_ERROR;
733
+ return TF_OK;
734
+ }
735
+ if (step->flush_next) continue;
736
+ p->finish_step_index++;
737
+ p->finish_step_in_next = 0;
738
+ continue;
739
+ }
740
+
741
+ if (step->flush_next) {
742
+ tf_set_last_error(NULL);
743
+ rc = step->flush_next(step, &flushed, &p->side);
744
+ if (rc != TF_OK) return pipeline_fail(p, "flush error");
745
+ if (flushed) {
746
+ if (route_flushed_batch(p, p->finish_step_index, flushed) != TF_OK) return TF_ERROR;
747
+ return TF_OK;
748
+ }
749
+ }
750
+ p->finish_step_index++;
751
+ p->finish_step_in_next = 0;
752
+ continue;
753
+ }
754
+
755
+ if (p->finish_phase == TF_FINISH_PHASE_ENCODER) {
756
+ tf_set_last_error(NULL);
757
+ size_t flush_before = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
758
+ if (p->encoder->flush(p->encoder, &p->output[TF_CHAN_MAIN]) != TF_OK) {
759
+ return pipeline_fail(p, "encode flush error");
760
+ }
761
+ size_t flush_after = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
762
+ if (flush_after >= flush_before) p->bytes_out += flush_after - flush_before;
763
+ if (pipeline_auto_drain_sinks(p) != TF_OK) return TF_ERROR;
764
+ if (pipeline_report_progress(p, 0) != TF_OK) return TF_ERROR;
765
+ p->finish_phase = TF_FINISH_PHASE_STATS;
766
+ return TF_OK;
767
+ }
768
+
769
+ if (p->finish_phase == TF_FINISH_PHASE_STATS) {
770
+ char stats_buf[256];
771
+ snprintf(stats_buf, sizeof(stats_buf),
772
+ "{\"rows_in\":%zu,\"rows_out\":%zu,\"bytes_in\":%zu,\"bytes_out\":%zu}\n",
773
+ p->rows_in, p->rows_out, p->bytes_in, p->bytes_out);
774
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], stats_buf) != TF_OK)
775
+ return pipeline_fail_msg(p, "stats output error");
776
+ if (emit_step_stats(p) != TF_OK)
777
+ return pipeline_fail_msg(p, "stats output error");
778
+ if (pipeline_auto_drain_sinks(p) != TF_OK) return TF_ERROR;
779
+ p->finished = 1;
780
+ p->finish_phase = TF_FINISH_PHASE_DONE;
781
+ if (pipeline_report_progress(p, 1) != TF_OK) return TF_ERROR;
782
+ return TF_DONE;
783
+ }
784
+
785
+ if (p->finish_phase == TF_FINISH_PHASE_DONE) {
786
+ p->finished = 1;
787
+ return TF_DONE;
788
+ }
789
+
790
+ return pipeline_fail_msg(p, "invalid finish state");
791
+ }
792
+ }
793
+
794
+ int tf_pipeline_finish(tf_pipeline *p) {
795
+ if (!p) return TF_ERROR;
796
+ if (p->finished) return TF_ERROR;
797
+ for (;;) {
798
+ int rc = tf_pipeline_finish_step(p);
799
+ if (rc == TF_DONE) return TF_OK;
800
+ if (rc != TF_OK) return TF_ERROR;
801
+ }
802
+ }
803
+
804
+ size_t tf_pipeline_pull(tf_pipeline *p, int channel, uint8_t *buf, size_t buf_len) {
805
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS) return 0;
806
+ return tf_buffer_read(&p->output[channel], buf, buf_len);
807
+ }
808
+
809
+ int tf_pipeline_drain(tf_pipeline *p, int channel, tf_pipeline_sink_fn sink, void *user) {
810
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS || !sink) return TF_ERROR;
811
+ return pipeline_drain_channel_to_sink(p, channel, sink, user);
812
+ }
813
+
814
+ int tf_pipeline_set_sink(tf_pipeline *p, int channel, tf_pipeline_sink_fn sink, void *user) {
815
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS) return TF_ERROR;
816
+ p->sinks[channel] = sink;
817
+ p->sink_users[channel] = sink ? user : NULL;
818
+ if (sink && tf_buffer_readable(&p->output[channel]) > 0) {
819
+ return pipeline_drain_channel_to_sink(p, channel, sink, user);
820
+ }
821
+ return TF_OK;
822
+ }
823
+
824
+ int tf_pipeline_set_batch_sink(tf_pipeline *p, tf_pipeline_batch_sink_fn sink, void *user) {
825
+ if (!p) return TF_ERROR;
826
+ p->batch_sink = sink;
827
+ p->batch_sink_user = sink ? user : NULL;
828
+ return TF_OK;
829
+ }
830
+
831
+ int tf_pipeline_set_encode_output(tf_pipeline *p, int enabled) {
832
+ if (!p) return TF_ERROR;
833
+ p->encode_output = enabled ? 1 : 0;
834
+ return TF_OK;
835
+ }
836
+
837
+ int tf_pipeline_set_progress_callback(tf_pipeline *p, tf_pipeline_progress_fn cb,
838
+ void *user, size_t interval_rows) {
839
+ if (!p) return TF_ERROR;
840
+ p->progress_cb = cb;
841
+ p->progress_user = cb ? user : NULL;
842
+ p->progress_interval_rows = interval_rows;
843
+ p->progress_last_rows_in = p->rows_in;
844
+ p->progress_last_rows_out = p->rows_out;
845
+ return TF_OK;
846
+ }
847
+
848
+ int tf_pipeline_set_source_name(tf_pipeline *p, const char *name) {
849
+ if (!p || p->finished || p->finish_started) return TF_ERROR;
850
+ char *copy = NULL;
851
+ if (name && name[0]) {
852
+ copy = strdup(name);
853
+ if (!copy) return pipeline_fail_msg(p, "out of memory");
854
+ }
855
+ free(p->source_name);
856
+ p->source_name = copy;
857
+ p->side.source_name = p->source_name;
858
+ return TF_OK;
859
+ }
860
+
861
+ int tf_pipeline_flush_input(tf_pipeline *p) {
862
+ if (!p || p->finished || p->finish_started) return TF_ERROR;
863
+
864
+ tf_batch **batches = NULL;
865
+ size_t n_batches = 0;
866
+ tf_set_last_error(NULL);
867
+ int rc = p->decoder->flush(p->decoder, &batches, &n_batches, &p->side);
868
+ if (rc != TF_OK) {
869
+ tf_batch_array_free(batches, n_batches);
870
+ return pipeline_fail(p, "decode boundary flush error");
871
+ }
872
+
873
+ for (size_t i = 0; i < n_batches; i++) {
874
+ rc = process_batch(p, batches[i]);
875
+ tf_batch_free(batches[i]);
876
+ if (rc != TF_OK) {
877
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
878
+ free(batches);
879
+ return p->error ? TF_ERROR : pipeline_fail(p, "processing error");
880
+ }
881
+ if (pipeline_auto_drain_sinks(p) != TF_OK) {
882
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
883
+ free(batches);
884
+ return TF_ERROR;
885
+ }
886
+ if (pipeline_report_progress(p, 0) != TF_OK) {
887
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
888
+ free(batches);
889
+ return TF_ERROR;
890
+ }
891
+ }
892
+ free(batches);
893
+ return TF_OK;
894
+ }
895
+
896
+ static int file_output_sink(int channel, const uint8_t *data, size_t len, void *user) {
897
+ (void)channel;
898
+ FILE *out = (FILE *)user;
899
+ if (!out) return TF_ERROR;
900
+ if (len == 0) return TF_OK;
901
+ return fwrite(data, 1, len, out) == len ? TF_OK : TF_ERROR;
902
+ }
903
+
904
+ #ifndef _WIN32
905
+ static int fd_output_sink(int channel, const uint8_t *data, size_t len, void *user) {
906
+ (void)channel;
907
+ if (!user) return TF_ERROR;
908
+ int fd = *(int *)user;
909
+ if (fd < 0) return TF_ERROR;
910
+ const uint8_t *cursor = data;
911
+ size_t remaining = len;
912
+ while (remaining > 0) {
913
+ ssize_t n = write(fd, cursor, remaining);
914
+ if (n < 0) {
915
+ if (errno == EINTR) continue;
916
+ return TF_ERROR;
917
+ }
918
+ if (n == 0) return TF_ERROR;
919
+ cursor += (size_t)n;
920
+ remaining -= (size_t)n;
921
+ }
922
+ return TF_OK;
923
+ }
924
+ #endif
925
+
926
+ int tf_pipeline_run_file(tf_pipeline *p, FILE *in, FILE *out, size_t chunk_size) {
927
+ if (!p || !in || !out || p->finished || p->finish_started) return TF_ERROR;
928
+ if (chunk_size == 0) chunk_size = 64 * 1024;
929
+
930
+ uint8_t *buf = malloc(chunk_size);
931
+ if (!buf) return pipeline_fail_msg(p, "out of memory");
932
+
933
+ tf_pipeline_sink_fn prev_sink = p->sinks[TF_CHAN_MAIN];
934
+ void *prev_user = p->sink_users[TF_CHAN_MAIN];
935
+ if (tf_pipeline_set_sink(p, TF_CHAN_MAIN, file_output_sink, out) != TF_OK) {
936
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
937
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
938
+ free(buf);
939
+ return TF_ERROR;
940
+ }
941
+
942
+ int rc = TF_OK;
943
+ for (;;) {
944
+ size_t n = fread(buf, 1, chunk_size, in);
945
+ if (n > 0 && tf_pipeline_push(p, buf, n) != TF_OK) {
946
+ rc = TF_ERROR;
947
+ break;
948
+ }
949
+ if (n < chunk_size) {
950
+ if (ferror(in)) rc = pipeline_fail_msg(p, "file input read error");
951
+ break;
952
+ }
953
+ }
954
+
955
+ if (rc == TF_OK && tf_pipeline_finish(p) != TF_OK) rc = TF_ERROR;
956
+ if (rc == TF_OK && fflush(out) != 0) rc = pipeline_fail_msg(p, "file output flush error");
957
+
958
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
959
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
960
+ free(buf);
961
+ return rc;
962
+ }
963
+
964
+ int tf_pipeline_run_fd(tf_pipeline *p, int in_fd, int out_fd, size_t chunk_size) {
965
+ #ifdef _WIN32
966
+ (void)in_fd;
967
+ (void)out_fd;
968
+ (void)chunk_size;
969
+ if (p) return pipeline_fail_msg(p, "file descriptor runner is not supported on this platform");
970
+ tf_set_last_error("file descriptor runner is not supported on this platform");
971
+ return TF_ERROR;
972
+ #else
973
+ if (!p || in_fd < 0 || out_fd < 0 || p->finished || p->finish_started) return TF_ERROR;
974
+ if (chunk_size == 0) chunk_size = 64 * 1024;
975
+
976
+ uint8_t *buf = malloc(chunk_size);
977
+ if (!buf) return pipeline_fail_msg(p, "out of memory");
978
+
979
+ tf_pipeline_sink_fn prev_sink = p->sinks[TF_CHAN_MAIN];
980
+ void *prev_user = p->sink_users[TF_CHAN_MAIN];
981
+ if (tf_pipeline_set_sink(p, TF_CHAN_MAIN, fd_output_sink, &out_fd) != TF_OK) {
982
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
983
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
984
+ free(buf);
985
+ return TF_ERROR;
986
+ }
987
+
988
+ int rc = TF_OK;
989
+ for (;;) {
990
+ ssize_t n = read(in_fd, buf, chunk_size);
991
+ if (n > 0) {
992
+ if (tf_pipeline_push(p, buf, (size_t)n) != TF_OK) {
993
+ rc = TF_ERROR;
994
+ break;
995
+ }
996
+ continue;
997
+ }
998
+ if (n == 0) break;
999
+ if (errno == EINTR) continue;
1000
+ rc = pipeline_fail_msg(p, "file descriptor input read error");
1001
+ break;
1002
+ }
1003
+
1004
+ if (rc == TF_OK && tf_pipeline_finish(p) != TF_OK) rc = TF_ERROR;
1005
+
1006
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
1007
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
1008
+ free(buf);
1009
+ return rc;
1010
+ #endif
1011
+ }
1012
+
1013
+ const char *tf_pipeline_error(tf_pipeline *p) {
1014
+ return p ? p->error : NULL;
1015
+ }
1016
+
1017
+ void tf_pipeline_free(tf_pipeline *p) {
1018
+ if (!p) return;
1019
+ if (p->decoder) p->decoder->destroy(p->decoder);
1020
+ if (p->encoder) p->encoder->destroy(p->encoder);
1021
+ for (size_t i = 0; i < p->n_steps; i++) {
1022
+ if (p->steps[i]) p->steps[i]->destroy(p->steps[i]);
1023
+ }
1024
+ free(p->steps);
1025
+ pipeline_clear_finish_batches(p);
1026
+ step_stats_free(p->step_stats, p->n_step_stats);
1027
+ for (int i = 0; i < TF_NUM_CHANNELS; i++) {
1028
+ tf_buffer_free(&p->output[i]);
1029
+ }
1030
+ free(p->source_name);
1031
+ free(p->error);
1032
+ free(p);
1033
+ }
1034
+
1035
+ char *tf_compile_to_sql(const char *dsl, size_t len, char **error) {
1036
+ if (error) *error = NULL;
1037
+ tf_ir_plan *plan = tf_dsl_parse(dsl, len, error);
1038
+ if (!plan) return NULL;
1039
+ if (tf_ir_validate(plan) != TF_OK) {
1040
+ if (error) { free(*error); *error = strdup(plan->error ? plan->error : "validation failed"); }
1041
+ tf_ir_plan_destroy(plan);
1042
+ return NULL;
1043
+ }
1044
+ tf_set_last_error(NULL);
1045
+ if (tf_ir_infer_schema(plan) != TF_OK) {
1046
+ if (error) *error = schema_inference_error_string();
1047
+ tf_ir_plan_destroy(plan);
1048
+ return NULL;
1049
+ }
1050
+ char *sql = tf_ir_to_sql(plan, error);
1051
+ tf_ir_plan_destroy(plan);
1052
+ return sql;
1053
+ }
1054
+
1055
+ char *tf_ir_plan_to_sql(const tf_ir_plan *plan, char **error) {
1056
+ return tf_ir_to_sql(plan, error);
1057
+ }
1058
+
1059
+ char *tf_compile_dsl_with_host_policy(const char *dsl, size_t len,
1060
+ const tf_host_policy *policy, char **error) {
1061
+ if (error) *error = NULL;
1062
+ tf_ir_plan *plan = tf_dsl_parse(dsl, len, error);
1063
+ if (!plan) return NULL;
1064
+
1065
+ if (tf_ir_validate_with_host_policy(plan, policy) != TF_OK) {
1066
+ if (error) *error = strdup(plan->error ? plan->error : "validation failed");
1067
+ tf_ir_plan_destroy(plan);
1068
+ return NULL;
1069
+ }
1070
+ tf_set_last_error(NULL);
1071
+ if (tf_ir_infer_schema(plan) != TF_OK) {
1072
+ if (error) *error = schema_inference_error_string();
1073
+ tf_ir_plan_destroy(plan);
1074
+ return NULL;
1075
+ }
1076
+
1077
+ char *json = tf_ir_plan_to_json(plan);
1078
+ tf_ir_plan_destroy(plan);
1079
+ return json;
1080
+ }
1081
+
1082
+ char *tf_compile_dsl(const char *dsl, size_t len, char **error) {
1083
+ return tf_compile_dsl_with_host_policy(dsl, len, NULL, error);
1084
+ }
1085
+
1086
+ void tf_string_free(char *s) {
1087
+ free(s);
1088
+ }