tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/pipeline.c CHANGED
@@ -7,35 +7,359 @@
7
7
 
8
8
  #include "internal.h"
9
9
  #include "dsl.h"
10
+ #include "cJSON.h"
10
11
  #include <stdlib.h>
11
12
  #include <string.h>
12
13
  #include <stdio.h>
14
+ #include <errno.h>
15
+ #ifndef _WIN32
16
+ #include <unistd.h>
17
+ #endif
13
18
 
14
- #define TRANFI_VERSION "0.1.0"
19
+ #define TRANFI_VERSION "0.2.1"
15
20
 
16
- static char *g_last_error = NULL;
21
+ enum {
22
+ TF_FINISH_PHASE_DECODER = 0,
23
+ TF_FINISH_PHASE_STEPS = 1,
24
+ TF_FINISH_PHASE_ENCODER = 2,
25
+ TF_FINISH_PHASE_STATS = 3,
26
+ TF_FINISH_PHASE_DONE = 4
27
+ };
28
+
29
+ #define TF_LAST_ERROR_CAP 1024
30
+
31
+ #if defined(_MSC_VER)
32
+ static __declspec(thread) char g_last_error[TF_LAST_ERROR_CAP];
33
+ #else
34
+ static _Thread_local char g_last_error[TF_LAST_ERROR_CAP];
35
+ #endif
17
36
 
18
37
  void tf_set_last_error(const char *msg) {
19
- free(g_last_error);
20
- g_last_error = msg ? strdup(msg) : NULL;
38
+ if (!msg || !*msg) {
39
+ g_last_error[0] = '\0';
40
+ return;
41
+ }
42
+ snprintf(g_last_error, sizeof(g_last_error), "%s", msg);
21
43
  }
22
44
 
23
45
  const char *tf_last_error(void) {
24
- return g_last_error;
46
+ return g_last_error[0] ? g_last_error : NULL;
47
+ }
48
+
49
+ static void pipeline_set_error(tf_pipeline *p, const char *fallback) {
50
+ if (!p) return;
51
+ const char *last = tf_last_error();
52
+ free(p->error);
53
+ p->error = strdup((last && *last) ? last : fallback);
54
+ }
55
+
56
+ static int pipeline_fail(tf_pipeline *p, const char *fallback) {
57
+ pipeline_set_error(p, fallback);
58
+ return TF_ERROR;
59
+ }
60
+
61
+ static void set_last_schema_inference_error(void) {
62
+ const char *detail = tf_last_error();
63
+ if (detail && detail[0]) {
64
+ char buf[TF_LAST_ERROR_CAP];
65
+ snprintf(buf, sizeof(buf), "schema inference failed: %s", detail);
66
+ tf_set_last_error(buf);
67
+ } else {
68
+ tf_set_last_error("schema inference failed");
69
+ }
70
+ }
71
+
72
+ static char *schema_inference_error_string(void) {
73
+ const char *detail = tf_last_error();
74
+ if (detail && detail[0]) {
75
+ char buf[TF_LAST_ERROR_CAP];
76
+ snprintf(buf, sizeof(buf), "schema inference failed: %s", detail);
77
+ return strdup(buf);
78
+ }
79
+ return strdup("schema inference failed");
80
+ }
81
+
82
+ static int pipeline_fail_msg(tf_pipeline *p, const char *msg) {
83
+ if (p) {
84
+ free(p->error);
85
+ p->error = strdup(msg ? msg : "pipeline error");
86
+ }
87
+ return TF_ERROR;
88
+ }
89
+
90
+ static void pipeline_clear_finish_batches(tf_pipeline *p) {
91
+ if (!p || !p->finish_batches) return;
92
+ for (size_t i = p->finish_batch_index; i < p->finish_n_batches; i++) {
93
+ if (p->finish_batches[i]) tf_batch_free(p->finish_batches[i]);
94
+ }
95
+ free(p->finish_batches);
96
+ p->finish_batches = NULL;
97
+ p->finish_n_batches = 0;
98
+ p->finish_batch_index = 0;
99
+ }
100
+
101
+ static int pipeline_drain_channel_to_sink(tf_pipeline *p, int channel,
102
+ tf_pipeline_sink_fn sink, void *user) {
103
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS || !sink) return TF_ERROR;
104
+ uint8_t buf[64 * 1024];
105
+ for (;;) {
106
+ size_t n = tf_buffer_read(&p->output[channel], buf, sizeof(buf));
107
+ if (n == 0) break;
108
+ if (sink(channel, buf, n, user) != TF_OK) {
109
+ free(p->error);
110
+ p->error = strdup("sink callback failed");
111
+ return TF_ERROR;
112
+ }
113
+ }
114
+ return TF_OK;
115
+ }
116
+
117
+ static int pipeline_auto_drain_sinks(tf_pipeline *p) {
118
+ if (!p) return TF_ERROR;
119
+ for (int channel = 0; channel < TF_NUM_CHANNELS; channel++) {
120
+ if (p->sinks[channel]) {
121
+ if (pipeline_drain_channel_to_sink(p, channel, p->sinks[channel],
122
+ p->sink_users[channel]) != TF_OK) {
123
+ return TF_ERROR;
124
+ }
125
+ }
126
+ }
127
+ return TF_OK;
25
128
  }
26
129
 
27
130
  const char *tf_version(void) {
28
131
  return TRANFI_VERSION;
29
132
  }
30
133
 
134
+ static void step_stats_free(tf_step_run_stats *stats, size_t n) {
135
+ if (!stats) return;
136
+ for (size_t i = 0; i < n; i++) {
137
+ free(stats[i].op);
138
+ free(stats[i].state_estimate);
139
+ free(stats[i].state_bytes_reason);
140
+ free(stats[i].execution_target);
141
+ }
142
+ free(stats);
143
+ }
144
+
145
+ static int node_has_spill_dir(const tf_ir_node *node) {
146
+ cJSON *spill = node && node->args ? cJSON_GetObjectItemCaseSensitive(node->args, "spill_dir") : NULL;
147
+ return cJSON_IsString(spill) && spill->valuestring && spill->valuestring[0];
148
+ }
149
+
150
+
151
+ static int node_op_is_join_spillable(const tf_ir_node *node) {
152
+ if (!node || !node->op) return 0;
153
+ if (strcmp(node->op, "semi-join") == 0 || strcmp(node->op, "anti-join") == 0) return 1;
154
+ if (strcmp(node->op, "join") != 0) return 0;
155
+ if (!node->args) return 1;
156
+ cJSON *how = cJSON_GetObjectItemCaseSensitive(node->args, "how");
157
+ if (!cJSON_IsString(how) || !how->valuestring) return 1;
158
+ return strcmp(how->valuestring, "inner") == 0 || strcmp(how->valuestring, "left") == 0 ||
159
+ strcmp(how->valuestring, "semi") == 0 || strcmp(how->valuestring, "anti") == 0;
160
+ }
161
+
162
+ static int node_op_is_row_set_spillable(const tf_ir_node *node) {
163
+ return node && node->op &&
164
+ (strcmp(node->op, "intersect") == 0 || strcmp(node->op, "setdiff") == 0 ||
165
+ strcmp(node->op, "intersect-all") == 0 || strcmp(node->op, "setdiff-all") == 0 ||
166
+ strcmp(node->op, "union") == 0);
167
+ }
168
+
169
+ static int node_op_uses_native_spill(const tf_ir_node *node) {
170
+ return node && node->op && node_has_spill_dir(node) &&
171
+ (strcmp(node->op, "sort") == 0 ||
172
+ strcmp(node->op, "unique") == 0 ||
173
+ strcmp(node->op, "dedup") == 0 ||
174
+ strcmp(node->op, "group-agg") == 0 ||
175
+ strcmp(node->op, "pivot") == 0 ||
176
+ node_op_is_join_spillable(node) ||
177
+ node_op_is_row_set_spillable(node));
178
+ }
179
+
180
+ static const char *runtime_step_execution_target(const tf_ir_node *node) {
181
+ #ifdef __EMSCRIPTEN__
182
+ (void)node;
183
+ return "wasm";
184
+ #else
185
+ if (node_op_uses_native_spill(node)) return "native_spill";
186
+ return "native";
187
+ #endif
188
+ }
189
+
190
+ static int build_step_stats_from_plan(const tf_ir_plan *plan,
191
+ tf_step_run_stats **out_stats,
192
+ size_t *out_n) {
193
+ if (!out_stats || !out_n) return TF_ERROR;
194
+ *out_stats = NULL;
195
+ *out_n = 0;
196
+ if (!plan) return TF_OK;
197
+
198
+ size_t count = 0;
199
+ for (size_t i = 0; i < plan->n_nodes; i++) {
200
+ const tf_op_entry *entry = tf_op_registry_find(plan->nodes[i].op);
201
+ if (entry && entry->kind == TF_OP_TRANSFORM && entry->create_native) count++;
202
+ }
203
+ if (count == 0) return TF_OK;
204
+
205
+ tf_step_run_stats *stats = calloc(count, sizeof(tf_step_run_stats));
206
+ if (!stats) return TF_ERROR;
207
+
208
+ size_t j = 0;
209
+ for (size_t i = 0; i < plan->n_nodes; i++) {
210
+ const tf_ir_node *node = &plan->nodes[i];
211
+ const tf_op_entry *entry = tf_op_registry_find(node->op);
212
+ if (!entry || entry->kind != TF_OP_TRANSFORM || !entry->create_native) continue;
213
+
214
+ stats[j].op = strdup(node->op);
215
+ stats[j].state_estimate = strdup(node->state_estimate ? node->state_estimate : "unknown");
216
+ stats[j].execution_target = strdup(runtime_step_execution_target(node));
217
+ if (!stats[j].op || !stats[j].state_estimate || !stats[j].execution_target) {
218
+ step_stats_free(stats, count);
219
+ return TF_ERROR;
220
+ }
221
+ stats[j].node_index = node->index;
222
+ stats[j].memory_class = node->memory_class;
223
+ stats[j].emit_class = node->emit_class;
224
+ stats[j].schema_class = node->schema_class;
225
+ if (node->memory_class == TF_MEM_BLOCKING) stats[j].warnings |= TF_STEP_WARN_BLOCKING;
226
+ if (node->emit_class == TF_EMIT_ON_FLUSH) stats[j].warnings |= TF_STEP_WARN_FLUSH_LATENT;
227
+ if (node->schema_class == TF_SCHEMA_DATA_DEPENDENT) stats[j].warnings |= TF_STEP_WARN_DATA_DEPENDENT_SCHEMA;
228
+
229
+ char reason[192] = {0};
230
+ size_t estimate = 0;
231
+ if (tf_estimate_step_state_bytes(node, &estimate, reason, sizeof(reason))) {
232
+ stats[j].has_state_bytes_estimate = 1;
233
+ stats[j].state_bytes_estimate = estimate;
234
+ } else if (node->memory_class == TF_MEM_KEY_STATE || node->memory_class == TF_MEM_BLOCKING) {
235
+ stats[j].warnings |= TF_STEP_WARN_UNBOUNDED_STATE;
236
+ stats[j].state_bytes_reason = strdup(reason[0] ? reason : "no native byte estimator");
237
+ if (!stats[j].state_bytes_reason) {
238
+ step_stats_free(stats, count);
239
+ return TF_ERROR;
240
+ }
241
+ }
242
+ j++;
243
+ }
244
+
245
+ *out_stats = stats;
246
+ *out_n = count;
247
+ return TF_OK;
248
+ }
249
+
250
+ static void step_stats_record_input(tf_pipeline *p, size_t step_idx, size_t rows) {
251
+ if (!p || step_idx >= p->n_step_stats) return;
252
+ p->step_stats[step_idx].batches_in++;
253
+ p->step_stats[step_idx].rows_in += rows;
254
+ }
255
+
256
+ static void step_stats_record_output(tf_pipeline *p, size_t step_idx, const tf_batch *batch) {
257
+ if (!p || step_idx >= p->n_step_stats || !batch) return;
258
+ p->step_stats[step_idx].batches_out++;
259
+ p->step_stats[step_idx].rows_out += batch->n_rows;
260
+ }
261
+
262
+ static int step_stats_write_warning_token(tf_pipeline *p, const char *token, int *first) {
263
+ char buf[96];
264
+ snprintf(buf, sizeof(buf), "%s\"%s\"", *first ? "" : ",", token);
265
+ *first = 0;
266
+ return tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf);
267
+ }
268
+
269
+ static int step_stats_write_warnings(tf_pipeline *p, uint32_t warnings) {
270
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], "\"warnings\":[") != TF_OK)
271
+ return TF_ERROR;
272
+ int first = 1;
273
+ if ((warnings & TF_STEP_WARN_BLOCKING) &&
274
+ step_stats_write_warning_token(p, "blocking", &first) != TF_OK)
275
+ return TF_ERROR;
276
+ if ((warnings & TF_STEP_WARN_FLUSH_LATENT) &&
277
+ step_stats_write_warning_token(p, "flush_latent", &first) != TF_OK)
278
+ return TF_ERROR;
279
+ if ((warnings & TF_STEP_WARN_UNBOUNDED_STATE) &&
280
+ step_stats_write_warning_token(p, "unbounded_state", &first) != TF_OK)
281
+ return TF_ERROR;
282
+ if ((warnings & TF_STEP_WARN_DATA_DEPENDENT_SCHEMA) &&
283
+ step_stats_write_warning_token(p, "data_dependent_schema", &first) != TF_OK)
284
+ return TF_ERROR;
285
+ return tf_buffer_write_str(&p->output[TF_CHAN_STATS], "],");
286
+ }
287
+
288
+ static int emit_step_stats(tf_pipeline *p) {
289
+ if (!p) return TF_ERROR;
290
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], "{\"type\":\"step_stats\",\"steps\":[") != TF_OK)
291
+ return TF_ERROR;
292
+ for (size_t i = 0; i < p->n_step_stats; i++) {
293
+ tf_step_run_stats *st = &p->step_stats[i];
294
+ char buf[1024];
295
+ snprintf(buf, sizeof(buf),
296
+ "%s{\"index\":%zu,\"op\":\"%s\",\"execution_target\":\"%s\","
297
+ "\"memory_class\":\"%s\",\"emit_class\":\"%s\","
298
+ "\"schema_class\":\"%s\",\"state_estimate\":\"%s\",",
299
+ i ? "," : "",
300
+ st->node_index,
301
+ st->op ? st->op : "unknown",
302
+ st->execution_target ? st->execution_target : "native",
303
+ tf_memory_class_name(st->memory_class),
304
+ tf_emit_class_name(st->emit_class),
305
+ tf_schema_class_name(st->schema_class),
306
+ st->state_estimate ? st->state_estimate : "unknown");
307
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
308
+ return TF_ERROR;
309
+ if (st->has_state_bytes_estimate) {
310
+ snprintf(buf, sizeof(buf), "\"state_bytes_estimate\":%zu,", st->state_bytes_estimate);
311
+ } else {
312
+ snprintf(buf, sizeof(buf), "\"state_bytes_estimate\":null,");
313
+ }
314
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
315
+ return TF_ERROR;
316
+ if (st->state_bytes_reason) {
317
+ snprintf(buf, sizeof(buf), "\"state_bytes_reason\":\"%s\",", st->state_bytes_reason);
318
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
319
+ return TF_ERROR;
320
+ }
321
+ if (step_stats_write_warnings(p, st->warnings) != TF_OK)
322
+ return TF_ERROR;
323
+ snprintf(buf, sizeof(buf),
324
+ "\"batches_in\":%zu,\"batches_out\":%zu,\"rows_in\":%zu,\"rows_out\":%zu",
325
+ st->batches_in, st->batches_out, st->rows_in, st->rows_out);
326
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], buf) != TF_OK)
327
+ return TF_ERROR;
328
+ if (i < p->n_steps && p->steps[i] && p->steps[i]->append_stats &&
329
+ p->steps[i]->append_stats(p->steps[i], &p->output[TF_CHAN_STATS]) != TF_OK)
330
+ return TF_ERROR;
331
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], "}") != TF_OK)
332
+ return TF_ERROR;
333
+ }
334
+ return tf_buffer_write_str(&p->output[TF_CHAN_STATS], "]}\n");
335
+ }
336
+
31
337
  static tf_pipeline *assemble_pipeline(tf_decoder *decoder, tf_step **steps,
32
- size_t n_steps, tf_encoder *encoder) {
338
+ size_t n_steps,
339
+ tf_step_run_stats *step_stats,
340
+ size_t n_step_stats,
341
+ tf_encoder *encoder) {
342
+ if (!decoder || !encoder) {
343
+ if (decoder) decoder->destroy(decoder);
344
+ if (encoder) encoder->destroy(encoder);
345
+ for (size_t i = 0; i < n_steps; i++) {
346
+ if (steps && steps[i]) steps[i]->destroy(steps[i]);
347
+ }
348
+ free(steps);
349
+ step_stats_free(step_stats, n_step_stats);
350
+ tf_set_last_error(!decoder ? "plan missing decoder" : "plan missing encoder");
351
+ return NULL;
352
+ }
353
+
33
354
  tf_pipeline *p = calloc(1, sizeof(tf_pipeline));
34
355
  if (!p) {
35
- decoder->destroy(decoder);
36
- encoder->destroy(encoder);
37
- for (size_t i = 0; i < n_steps; i++) steps[i]->destroy(steps[i]);
356
+ if (decoder) decoder->destroy(decoder);
357
+ if (encoder) encoder->destroy(encoder);
358
+ for (size_t i = 0; i < n_steps; i++) {
359
+ if (steps && steps[i]) steps[i]->destroy(steps[i]);
360
+ }
38
361
  free(steps);
362
+ step_stats_free(step_stats, n_step_stats);
39
363
  tf_set_last_error("out of memory");
40
364
  return NULL;
41
365
  }
@@ -43,7 +367,10 @@ static tf_pipeline *assemble_pipeline(tf_decoder *decoder, tf_step **steps,
43
367
  p->decoder = decoder;
44
368
  p->steps = steps;
45
369
  p->n_steps = n_steps;
370
+ p->step_stats = step_stats;
371
+ p->n_step_stats = n_step_stats;
46
372
  p->encoder = encoder;
373
+ p->encode_output = 1;
47
374
 
48
375
  for (int i = 0; i < TF_NUM_CHANNELS; i++) {
49
376
  tf_buffer_init(&p->output[i]);
@@ -53,37 +380,35 @@ static tf_pipeline *assemble_pipeline(tf_decoder *decoder, tf_step **steps,
53
380
  p->side.errors = &p->output[TF_CHAN_ERRORS];
54
381
  p->side.stats = &p->output[TF_CHAN_STATS];
55
382
  p->side.samples = &p->output[TF_CHAN_SAMPLES];
383
+ p->side.source_name = p->source_name;
56
384
 
57
385
  return p;
58
386
  }
59
387
 
60
- tf_pipeline *tf_pipeline_create(const char *plan_json, size_t len) {
61
- if (!plan_json || len == 0) {
62
- tf_set_last_error("empty plan");
63
- return NULL;
64
- }
65
-
388
+ static tf_pipeline *pipeline_create_from_owned_ir(tf_ir_plan *ir, const tf_host_policy *policy) {
66
389
  char *error = NULL;
67
390
 
68
- /* 1. Parse JSON → IR */
69
- tf_ir_plan *ir = tf_ir_from_json(plan_json, len, &error);
70
- if (!ir) {
71
- tf_set_last_error(error ? error : "failed to parse plan");
72
- free(error);
391
+ if (tf_ir_validate_with_host_policy(ir, policy) != TF_OK) {
392
+ tf_set_last_error(ir->error ? ir->error : "validation failed");
393
+ tf_ir_plan_free(ir);
73
394
  return NULL;
74
395
  }
75
396
 
76
- /* 2. Validate */
77
- if (tf_ir_validate(ir) != TF_OK) {
78
- tf_set_last_error(ir->error ? ir->error : "validation failed");
397
+ tf_set_last_error(NULL);
398
+ if (tf_ir_infer_schema(ir) != TF_OK) {
399
+ set_last_schema_inference_error();
79
400
  tf_ir_plan_free(ir);
80
401
  return NULL;
81
402
  }
82
403
 
83
- /* 3. Schema inference (best-effort, non-fatal) */
84
- tf_ir_infer_schema(ir);
404
+ tf_step_run_stats *step_stats = NULL;
405
+ size_t n_step_stats = 0;
406
+ if (build_step_stats_from_plan(ir, &step_stats, &n_step_stats) != TF_OK) {
407
+ tf_set_last_error("out of memory");
408
+ tf_ir_plan_free(ir);
409
+ return NULL;
410
+ }
85
411
 
86
- /* 4. Compile to native target */
87
412
  tf_decoder *decoder = NULL;
88
413
  tf_step **steps = NULL;
89
414
  size_t n_steps = 0;
@@ -91,35 +416,52 @@ tf_pipeline *tf_pipeline_create(const char *plan_json, size_t len) {
91
416
  if (tf_compile_native(ir, &decoder, &steps, &n_steps, &encoder, &error) != TF_OK) {
92
417
  tf_set_last_error(error ? error : "compilation failed");
93
418
  free(error);
419
+ step_stats_free(step_stats, n_step_stats);
94
420
  tf_ir_plan_free(ir);
95
421
  return NULL;
96
422
  }
97
423
 
98
424
  tf_ir_plan_free(ir);
99
-
100
- /* 5. Assemble pipeline */
101
- return assemble_pipeline(decoder, steps, n_steps, encoder);
425
+ return assemble_pipeline(decoder, steps, n_steps, step_stats, n_step_stats, encoder);
102
426
  }
103
427
 
104
- tf_pipeline *tf_pipeline_create_from_ir(const tf_ir_plan *plan) {
105
- if (!plan) {
106
- tf_set_last_error("NULL IR plan");
428
+ tf_pipeline *tf_pipeline_create_with_host_policy(const char *plan_json, size_t len,
429
+ const tf_host_policy *policy) {
430
+ if (!plan_json || len == 0) {
431
+ tf_set_last_error("empty plan");
107
432
  return NULL;
108
433
  }
109
434
 
110
435
  char *error = NULL;
111
- tf_decoder *decoder = NULL;
112
- tf_step **steps = NULL;
113
- size_t n_steps = 0;
114
- tf_encoder *encoder = NULL;
115
-
116
- if (tf_compile_native(plan, &decoder, &steps, &n_steps, &encoder, &error) != TF_OK) {
117
- tf_set_last_error(error ? error : "compilation failed");
436
+ tf_ir_plan *ir = tf_ir_from_json(plan_json, len, &error);
437
+ if (!ir) {
438
+ tf_set_last_error(error ? error : "failed to parse plan");
118
439
  free(error);
119
440
  return NULL;
120
441
  }
442
+ return pipeline_create_from_owned_ir(ir, policy);
443
+ }
444
+
445
+ tf_pipeline *tf_pipeline_create(const char *plan_json, size_t len) {
446
+ return tf_pipeline_create_with_host_policy(plan_json, len, NULL);
447
+ }
448
+
449
+ tf_pipeline *tf_pipeline_create_from_ir_with_host_policy(const tf_ir_plan *plan,
450
+ const tf_host_policy *policy) {
451
+ if (!plan) {
452
+ tf_set_last_error("NULL IR plan");
453
+ return NULL;
454
+ }
455
+ tf_ir_plan *copy = tf_ir_plan_clone(plan);
456
+ if (!copy) {
457
+ tf_set_last_error("out of memory");
458
+ return NULL;
459
+ }
460
+ return pipeline_create_from_owned_ir(copy, policy);
461
+ }
121
462
 
122
- return assemble_pipeline(decoder, steps, n_steps, encoder);
463
+ tf_pipeline *tf_pipeline_create_from_ir(const tf_ir_plan *plan) {
464
+ return tf_pipeline_create_from_ir_with_host_policy(plan, NULL);
123
465
  }
124
466
 
125
467
  /* Public IR wrappers (thin forwarding to ir.h functions) */
@@ -132,6 +474,9 @@ char *tf_ir_plan_to_json(const tf_ir_plan *plan) {
132
474
  int tf_ir_plan_validate(tf_ir_plan *plan) {
133
475
  return tf_ir_validate(plan);
134
476
  }
477
+ int tf_ir_plan_validate_with_host_policy(tf_ir_plan *plan, const tf_host_policy *policy) {
478
+ return tf_ir_validate_with_host_policy(plan, policy);
479
+ }
135
480
  int tf_ir_plan_infer_schema(tf_ir_plan *plan) {
136
481
  return tf_ir_infer_schema(plan);
137
482
  }
@@ -139,6 +484,80 @@ void tf_ir_plan_destroy(tf_ir_plan *plan) {
139
484
  tf_ir_plan_free(plan);
140
485
  }
141
486
 
487
+ static const char *pipeline_phase_name(const tf_pipeline *p) {
488
+ if (!p) return "unknown";
489
+ if (p->finished || p->finish_phase == TF_FINISH_PHASE_DONE) return "done";
490
+ if (!p->finish_started) return "push";
491
+ switch (p->finish_phase) {
492
+ case TF_FINISH_PHASE_DECODER: return "decoder_flush";
493
+ case TF_FINISH_PHASE_STEPS: return "step_flush";
494
+ case TF_FINISH_PHASE_ENCODER: return "encoder_flush";
495
+ case TF_FINISH_PHASE_STATS: return "stats";
496
+ case TF_FINISH_PHASE_DONE: return "done";
497
+ default: return "unknown";
498
+ }
499
+ }
500
+
501
+ static int pipeline_report_progress(tf_pipeline *p, int force) {
502
+ if (!p || !p->progress_cb) return TF_OK;
503
+ if (!force && p->progress_interval_rows > 0) {
504
+ size_t in_delta = p->rows_in >= p->progress_last_rows_in
505
+ ? p->rows_in - p->progress_last_rows_in : 0;
506
+ size_t out_delta = p->rows_out >= p->progress_last_rows_out
507
+ ? p->rows_out - p->progress_last_rows_out : 0;
508
+ if (in_delta < p->progress_interval_rows && out_delta < p->progress_interval_rows) {
509
+ return TF_OK;
510
+ }
511
+ }
512
+
513
+ tf_pipeline_progress progress = {
514
+ p->bytes_in,
515
+ p->bytes_out,
516
+ p->rows_in,
517
+ p->rows_out,
518
+ p->batches_in,
519
+ p->batches_out,
520
+ pipeline_phase_name(p),
521
+ p->finished ? 1 : 0,
522
+ };
523
+ if (p->progress_cb(&progress, p->progress_user) != TF_OK) {
524
+ tf_set_last_error("progress callback failed");
525
+ free(p->error);
526
+ p->error = strdup("progress callback failed");
527
+ return TF_ERROR;
528
+ }
529
+ p->progress_last_rows_in = p->rows_in;
530
+ p->progress_last_rows_out = p->rows_out;
531
+ return TF_OK;
532
+ }
533
+
534
+ static int emit_output_batch(tf_pipeline *p, tf_batch *batch) {
535
+ if (!p || !batch) return TF_OK;
536
+
537
+ if (batch->n_rows > 0) {
538
+ p->rows_out += batch->n_rows;
539
+ p->batches_out++;
540
+
541
+ if (p->batch_sink && p->batch_sink(batch, p->batch_sink_user) != TF_OK) {
542
+ tf_set_last_error("batch sink callback failed");
543
+ free(p->error);
544
+ p->error = strdup("batch sink callback failed");
545
+ return TF_ERROR;
546
+ }
547
+ }
548
+
549
+ if (!p->encode_output) return TF_OK;
550
+
551
+ tf_set_last_error(NULL);
552
+ size_t before = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
553
+ int rc = p->encoder->encode(p->encoder, batch, &p->output[TF_CHAN_MAIN]);
554
+ if (rc == TF_OK) {
555
+ size_t after = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
556
+ if (after >= before) p->bytes_out += after - before;
557
+ }
558
+ return rc;
559
+ }
560
+
142
561
  /*
143
562
  * Process a batch through all steps, then encode.
144
563
  */
@@ -147,44 +566,76 @@ static int process_batch(tf_pipeline *p, tf_batch *batch) {
147
566
  int batch_owned = 0; /* 0 = still owned by caller (decoder) */
148
567
 
149
568
  p->rows_in += current->n_rows;
569
+ p->batches_in++;
150
570
 
151
571
  for (size_t i = 0; i < p->n_steps; i++) {
152
572
  tf_batch *next = NULL;
573
+ step_stats_record_input(p, i, current->n_rows);
574
+ tf_set_last_error(NULL);
153
575
  int rc = p->steps[i]->process(p->steps[i], current, &next, &p->side);
154
576
 
155
577
  if (batch_owned) tf_batch_free(current);
156
578
  batch_owned = 1;
157
579
 
158
580
  if (rc != TF_OK) return TF_ERROR;
581
+ step_stats_record_output(p, i, next);
159
582
  if (!next) return TF_OK; /* filtered away entirely */
160
583
  current = next;
161
584
  }
162
585
 
163
- /* Encode */
164
- if (current->n_rows > 0) {
165
- p->rows_out += current->n_rows;
166
- int rc = p->encoder->encode(p->encoder, current, &p->output[TF_CHAN_MAIN]);
167
- if (batch_owned) tf_batch_free(current);
168
- return rc;
586
+ /* Emit transformed output batch. */
587
+ int rc = emit_output_batch(p, current);
588
+ if (batch_owned) tf_batch_free(current);
589
+ return rc;
590
+ }
591
+
592
+ static int route_flushed_batch(tf_pipeline *p, size_t source_step, tf_batch *flushed) {
593
+ if (!flushed) return TF_OK;
594
+ step_stats_record_output(p, source_step, flushed);
595
+
596
+ /* Run flushed batch through remaining steps. */
597
+ tf_batch *current = flushed;
598
+ int owned = 1;
599
+ int rc = TF_OK;
600
+ for (size_t j = source_step + 1; j < p->n_steps; j++) {
601
+ tf_batch *next = NULL;
602
+ step_stats_record_input(p, j, current->n_rows);
603
+ tf_set_last_error(NULL);
604
+ rc = p->steps[j]->process(p->steps[j], current, &next, &p->side);
605
+ if (owned) tf_batch_free(current);
606
+ owned = 1;
607
+ if (rc != TF_OK) return pipeline_fail(p, "processing error");
608
+ step_stats_record_output(p, j, next);
609
+ if (!next) { current = NULL; break; }
610
+ current = next;
169
611
  }
170
612
 
171
- if (batch_owned) tf_batch_free(current);
613
+ if (current) {
614
+ rc = emit_output_batch(p, current);
615
+ if (rc != TF_OK) {
616
+ if (current && owned) tf_batch_free(current);
617
+ return pipeline_fail(p, "encode error");
618
+ }
619
+ }
620
+ if (current && owned) tf_batch_free(current);
621
+ if (pipeline_auto_drain_sinks(p) != TF_OK) return TF_ERROR;
622
+ if (pipeline_report_progress(p, 0) != TF_OK) return TF_ERROR;
172
623
  return TF_OK;
173
624
  }
174
625
 
175
626
  int tf_pipeline_push(tf_pipeline *p, const uint8_t *data, size_t len) {
176
- if (!p || p->finished) return TF_ERROR;
627
+ if (!p || p->finished || p->finish_started) return TF_ERROR;
177
628
 
178
629
  p->bytes_in += len;
179
630
 
180
631
  /* Decode bytes into batches */
181
632
  tf_batch **batches = NULL;
182
633
  size_t n_batches = 0;
183
- int rc = p->decoder->decode(p->decoder, data, len, &batches, &n_batches);
634
+ tf_set_last_error(NULL);
635
+ int rc = p->decoder->decode(p->decoder, data, len, &batches, &n_batches, &p->side);
184
636
  if (rc != TF_OK) {
185
- free(p->error);
186
- p->error = strdup("decode error");
187
- return TF_ERROR;
637
+ tf_batch_array_free(batches, n_batches);
638
+ return pipeline_fail(p, "decode error");
188
639
  }
189
640
 
190
641
  /* Process each batch */
@@ -192,10 +643,18 @@ int tf_pipeline_push(tf_pipeline *p, const uint8_t *data, size_t len) {
192
643
  rc = process_batch(p, batches[i]);
193
644
  tf_batch_free(batches[i]);
194
645
  if (rc != TF_OK) {
195
- for (size_t j = i + 1; j < n_batches; j++) tf_batch_free(batches[j]);
646
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
647
+ free(batches);
648
+ return p->error ? TF_ERROR : pipeline_fail(p, "processing error");
649
+ }
650
+ if (pipeline_auto_drain_sinks(p) != TF_OK) {
651
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
652
+ free(batches);
653
+ return TF_ERROR;
654
+ }
655
+ if (pipeline_report_progress(p, 0) != TF_OK) {
656
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
196
657
  free(batches);
197
- free(p->error);
198
- p->error = strdup("processing error");
199
658
  return TF_ERROR;
200
659
  }
201
660
  }
@@ -204,58 +663,142 @@ int tf_pipeline_push(tf_pipeline *p, const uint8_t *data, size_t len) {
204
663
  return TF_OK;
205
664
  }
206
665
 
207
- int tf_pipeline_finish(tf_pipeline *p) {
208
- if (!p || p->finished) return TF_ERROR;
209
- p->finished = 1;
666
+ int tf_pipeline_finish_step(tf_pipeline *p) {
667
+ if (!p) return TF_ERROR;
668
+ if (p->finished || p->finish_phase == TF_FINISH_PHASE_DONE) return TF_DONE;
210
669
 
211
- /* Flush decoder */
212
- tf_batch **batches = NULL;
213
- size_t n_batches = 0;
214
- int rc = p->decoder->flush(p->decoder, &batches, &n_batches);
215
- if (rc == TF_OK) {
216
- for (size_t i = 0; i < n_batches; i++) {
217
- process_batch(p, batches[i]);
218
- tf_batch_free(batches[i]);
670
+ if (!p->finish_started) {
671
+ p->finish_started = 1;
672
+ p->finish_phase = TF_FINISH_PHASE_DECODER;
673
+ p->finish_batch_index = 0;
674
+ p->finish_n_batches = 0;
675
+ p->finish_batches = NULL;
676
+ p->finish_step_index = 0;
677
+ p->finish_step_in_next = 0;
678
+
679
+ tf_set_last_error(NULL);
680
+ int rc = p->decoder->flush(p->decoder, &p->finish_batches,
681
+ &p->finish_n_batches, &p->side);
682
+ if (rc != TF_OK) {
683
+ pipeline_clear_finish_batches(p);
684
+ return pipeline_fail(p, "decode flush error");
219
685
  }
220
- free(batches);
221
686
  }
222
687
 
223
- /* Flush steps */
224
- for (size_t i = 0; i < p->n_steps; i++) {
225
- tf_batch *flushed = NULL;
226
- rc = p->steps[i]->flush(p->steps[i], &flushed, &p->side);
227
- if (rc == TF_OK && flushed) {
228
- /* Run flushed batch through remaining steps */
229
- tf_batch *current = flushed;
230
- int owned = 1;
231
- for (size_t j = i + 1; j < p->n_steps; j++) {
232
- tf_batch *next = NULL;
233
- p->steps[j]->process(p->steps[j], current, &next, &p->side);
234
- if (owned) tf_batch_free(current);
235
- owned = 1;
236
- if (!next) { current = NULL; break; }
237
- current = next;
688
+ for (;;) {
689
+ if (p->finish_phase == TF_FINISH_PHASE_DECODER) {
690
+ if (p->finish_batch_index < p->finish_n_batches) {
691
+ tf_batch *batch = p->finish_batches[p->finish_batch_index];
692
+ p->finish_batches[p->finish_batch_index] = NULL;
693
+ p->finish_batch_index++;
694
+
695
+ int rc = process_batch(p, batch);
696
+ tf_batch_free(batch);
697
+ if (rc != TF_OK) {
698
+ pipeline_clear_finish_batches(p);
699
+ return p->error ? TF_ERROR : pipeline_fail(p, "processing error");
700
+ }
701
+ if (pipeline_auto_drain_sinks(p) != TF_OK) {
702
+ pipeline_clear_finish_batches(p);
703
+ return TF_ERROR;
704
+ }
705
+ if (pipeline_report_progress(p, 0) != TF_OK) {
706
+ pipeline_clear_finish_batches(p);
707
+ return TF_ERROR;
708
+ }
709
+ return TF_OK;
710
+ }
711
+ pipeline_clear_finish_batches(p);
712
+ p->finish_phase = TF_FINISH_PHASE_STEPS;
713
+ continue;
714
+ }
715
+
716
+ if (p->finish_phase == TF_FINISH_PHASE_STEPS) {
717
+ if (p->finish_step_index >= p->n_steps) {
718
+ p->finish_phase = TF_FINISH_PHASE_ENCODER;
719
+ continue;
238
720
  }
239
- if (current && current->n_rows > 0) {
240
- p->rows_out += current->n_rows;
241
- p->encoder->encode(p->encoder, current, &p->output[TF_CHAN_MAIN]);
721
+
722
+ tf_step *step = p->steps[p->finish_step_index];
723
+ tf_batch *flushed = NULL;
724
+ int rc = TF_OK;
725
+
726
+ if (!p->finish_step_in_next) {
727
+ tf_set_last_error(NULL);
728
+ rc = step->flush(step, &flushed, &p->side);
729
+ if (rc != TF_OK) return pipeline_fail(p, "flush error");
730
+ p->finish_step_in_next = 1;
731
+ if (flushed) {
732
+ if (route_flushed_batch(p, p->finish_step_index, flushed) != TF_OK) return TF_ERROR;
733
+ return TF_OK;
734
+ }
735
+ if (step->flush_next) continue;
736
+ p->finish_step_index++;
737
+ p->finish_step_in_next = 0;
738
+ continue;
242
739
  }
243
- if (current && owned) tf_batch_free(current);
740
+
741
+ if (step->flush_next) {
742
+ tf_set_last_error(NULL);
743
+ rc = step->flush_next(step, &flushed, &p->side);
744
+ if (rc != TF_OK) return pipeline_fail(p, "flush error");
745
+ if (flushed) {
746
+ if (route_flushed_batch(p, p->finish_step_index, flushed) != TF_OK) return TF_ERROR;
747
+ return TF_OK;
748
+ }
749
+ }
750
+ p->finish_step_index++;
751
+ p->finish_step_in_next = 0;
752
+ continue;
244
753
  }
245
- }
246
754
 
247
- /* Flush encoder */
248
- p->encoder->flush(p->encoder, &p->output[TF_CHAN_MAIN]);
755
+ if (p->finish_phase == TF_FINISH_PHASE_ENCODER) {
756
+ tf_set_last_error(NULL);
757
+ size_t flush_before = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
758
+ if (p->encoder->flush(p->encoder, &p->output[TF_CHAN_MAIN]) != TF_OK) {
759
+ return pipeline_fail(p, "encode flush error");
760
+ }
761
+ size_t flush_after = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
762
+ if (flush_after >= flush_before) p->bytes_out += flush_after - flush_before;
763
+ if (pipeline_auto_drain_sinks(p) != TF_OK) return TF_ERROR;
764
+ if (pipeline_report_progress(p, 0) != TF_OK) return TF_ERROR;
765
+ p->finish_phase = TF_FINISH_PHASE_STATS;
766
+ return TF_OK;
767
+ }
249
768
 
250
- /* Emit final stats */
251
- p->bytes_out = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
252
- char stats_buf[256];
253
- snprintf(stats_buf, sizeof(stats_buf),
254
- "{\"rows_in\":%zu,\"rows_out\":%zu,\"bytes_in\":%zu,\"bytes_out\":%zu}\n",
255
- p->rows_in, p->rows_out, p->bytes_in, p->bytes_out);
256
- tf_buffer_write_str(&p->output[TF_CHAN_STATS], stats_buf);
769
+ if (p->finish_phase == TF_FINISH_PHASE_STATS) {
770
+ char stats_buf[256];
771
+ snprintf(stats_buf, sizeof(stats_buf),
772
+ "{\"rows_in\":%zu,\"rows_out\":%zu,\"bytes_in\":%zu,\"bytes_out\":%zu}\n",
773
+ p->rows_in, p->rows_out, p->bytes_in, p->bytes_out);
774
+ if (tf_buffer_write_str(&p->output[TF_CHAN_STATS], stats_buf) != TF_OK)
775
+ return pipeline_fail_msg(p, "stats output error");
776
+ if (emit_step_stats(p) != TF_OK)
777
+ return pipeline_fail_msg(p, "stats output error");
778
+ if (pipeline_auto_drain_sinks(p) != TF_OK) return TF_ERROR;
779
+ p->finished = 1;
780
+ p->finish_phase = TF_FINISH_PHASE_DONE;
781
+ if (pipeline_report_progress(p, 1) != TF_OK) return TF_ERROR;
782
+ return TF_DONE;
783
+ }
257
784
 
258
- return TF_OK;
785
+ if (p->finish_phase == TF_FINISH_PHASE_DONE) {
786
+ p->finished = 1;
787
+ return TF_DONE;
788
+ }
789
+
790
+ return pipeline_fail_msg(p, "invalid finish state");
791
+ }
792
+ }
793
+
794
+ int tf_pipeline_finish(tf_pipeline *p) {
795
+ if (!p) return TF_ERROR;
796
+ if (p->finished) return TF_ERROR;
797
+ for (;;) {
798
+ int rc = tf_pipeline_finish_step(p);
799
+ if (rc == TF_DONE) return TF_OK;
800
+ if (rc != TF_OK) return TF_ERROR;
801
+ }
259
802
  }
260
803
 
261
804
  size_t tf_pipeline_pull(tf_pipeline *p, int channel, uint8_t *buf, size_t buf_len) {
@@ -263,6 +806,210 @@ size_t tf_pipeline_pull(tf_pipeline *p, int channel, uint8_t *buf, size_t buf_le
263
806
  return tf_buffer_read(&p->output[channel], buf, buf_len);
264
807
  }
265
808
 
809
+ int tf_pipeline_drain(tf_pipeline *p, int channel, tf_pipeline_sink_fn sink, void *user) {
810
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS || !sink) return TF_ERROR;
811
+ return pipeline_drain_channel_to_sink(p, channel, sink, user);
812
+ }
813
+
814
+ int tf_pipeline_set_sink(tf_pipeline *p, int channel, tf_pipeline_sink_fn sink, void *user) {
815
+ if (!p || channel < 0 || channel >= TF_NUM_CHANNELS) return TF_ERROR;
816
+ p->sinks[channel] = sink;
817
+ p->sink_users[channel] = sink ? user : NULL;
818
+ if (sink && tf_buffer_readable(&p->output[channel]) > 0) {
819
+ return pipeline_drain_channel_to_sink(p, channel, sink, user);
820
+ }
821
+ return TF_OK;
822
+ }
823
+
824
+ int tf_pipeline_set_batch_sink(tf_pipeline *p, tf_pipeline_batch_sink_fn sink, void *user) {
825
+ if (!p) return TF_ERROR;
826
+ p->batch_sink = sink;
827
+ p->batch_sink_user = sink ? user : NULL;
828
+ return TF_OK;
829
+ }
830
+
831
+ int tf_pipeline_set_encode_output(tf_pipeline *p, int enabled) {
832
+ if (!p) return TF_ERROR;
833
+ p->encode_output = enabled ? 1 : 0;
834
+ return TF_OK;
835
+ }
836
+
837
+ int tf_pipeline_set_progress_callback(tf_pipeline *p, tf_pipeline_progress_fn cb,
838
+ void *user, size_t interval_rows) {
839
+ if (!p) return TF_ERROR;
840
+ p->progress_cb = cb;
841
+ p->progress_user = cb ? user : NULL;
842
+ p->progress_interval_rows = interval_rows;
843
+ p->progress_last_rows_in = p->rows_in;
844
+ p->progress_last_rows_out = p->rows_out;
845
+ return TF_OK;
846
+ }
847
+
848
+ int tf_pipeline_set_source_name(tf_pipeline *p, const char *name) {
849
+ if (!p || p->finished || p->finish_started) return TF_ERROR;
850
+ char *copy = NULL;
851
+ if (name && name[0]) {
852
+ copy = strdup(name);
853
+ if (!copy) return pipeline_fail_msg(p, "out of memory");
854
+ }
855
+ free(p->source_name);
856
+ p->source_name = copy;
857
+ p->side.source_name = p->source_name;
858
+ return TF_OK;
859
+ }
860
+
861
+ int tf_pipeline_flush_input(tf_pipeline *p) {
862
+ if (!p || p->finished || p->finish_started) return TF_ERROR;
863
+
864
+ tf_batch **batches = NULL;
865
+ size_t n_batches = 0;
866
+ tf_set_last_error(NULL);
867
+ int rc = p->decoder->flush(p->decoder, &batches, &n_batches, &p->side);
868
+ if (rc != TF_OK) {
869
+ tf_batch_array_free(batches, n_batches);
870
+ return pipeline_fail(p, "decode boundary flush error");
871
+ }
872
+
873
+ for (size_t i = 0; i < n_batches; i++) {
874
+ rc = process_batch(p, batches[i]);
875
+ tf_batch_free(batches[i]);
876
+ if (rc != TF_OK) {
877
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
878
+ free(batches);
879
+ return p->error ? TF_ERROR : pipeline_fail(p, "processing error");
880
+ }
881
+ if (pipeline_auto_drain_sinks(p) != TF_OK) {
882
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
883
+ free(batches);
884
+ return TF_ERROR;
885
+ }
886
+ if (pipeline_report_progress(p, 0) != TF_OK) {
887
+ tf_batch_array_free_items(batches + i + 1, n_batches - i - 1);
888
+ free(batches);
889
+ return TF_ERROR;
890
+ }
891
+ }
892
+ free(batches);
893
+ return TF_OK;
894
+ }
895
+
896
+ static int file_output_sink(int channel, const uint8_t *data, size_t len, void *user) {
897
+ (void)channel;
898
+ FILE *out = (FILE *)user;
899
+ if (!out) return TF_ERROR;
900
+ if (len == 0) return TF_OK;
901
+ return fwrite(data, 1, len, out) == len ? TF_OK : TF_ERROR;
902
+ }
903
+
904
+ #ifndef _WIN32
905
+ static int fd_output_sink(int channel, const uint8_t *data, size_t len, void *user) {
906
+ (void)channel;
907
+ if (!user) return TF_ERROR;
908
+ int fd = *(int *)user;
909
+ if (fd < 0) return TF_ERROR;
910
+ const uint8_t *cursor = data;
911
+ size_t remaining = len;
912
+ while (remaining > 0) {
913
+ ssize_t n = write(fd, cursor, remaining);
914
+ if (n < 0) {
915
+ if (errno == EINTR) continue;
916
+ return TF_ERROR;
917
+ }
918
+ if (n == 0) return TF_ERROR;
919
+ cursor += (size_t)n;
920
+ remaining -= (size_t)n;
921
+ }
922
+ return TF_OK;
923
+ }
924
+ #endif
925
+
926
+ int tf_pipeline_run_file(tf_pipeline *p, FILE *in, FILE *out, size_t chunk_size) {
927
+ if (!p || !in || !out || p->finished || p->finish_started) return TF_ERROR;
928
+ if (chunk_size == 0) chunk_size = 64 * 1024;
929
+
930
+ uint8_t *buf = malloc(chunk_size);
931
+ if (!buf) return pipeline_fail_msg(p, "out of memory");
932
+
933
+ tf_pipeline_sink_fn prev_sink = p->sinks[TF_CHAN_MAIN];
934
+ void *prev_user = p->sink_users[TF_CHAN_MAIN];
935
+ if (tf_pipeline_set_sink(p, TF_CHAN_MAIN, file_output_sink, out) != TF_OK) {
936
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
937
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
938
+ free(buf);
939
+ return TF_ERROR;
940
+ }
941
+
942
+ int rc = TF_OK;
943
+ for (;;) {
944
+ size_t n = fread(buf, 1, chunk_size, in);
945
+ if (n > 0 && tf_pipeline_push(p, buf, n) != TF_OK) {
946
+ rc = TF_ERROR;
947
+ break;
948
+ }
949
+ if (n < chunk_size) {
950
+ if (ferror(in)) rc = pipeline_fail_msg(p, "file input read error");
951
+ break;
952
+ }
953
+ }
954
+
955
+ if (rc == TF_OK && tf_pipeline_finish(p) != TF_OK) rc = TF_ERROR;
956
+ if (rc == TF_OK && fflush(out) != 0) rc = pipeline_fail_msg(p, "file output flush error");
957
+
958
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
959
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
960
+ free(buf);
961
+ return rc;
962
+ }
963
+
964
+ int tf_pipeline_run_fd(tf_pipeline *p, int in_fd, int out_fd, size_t chunk_size) {
965
+ #ifdef _WIN32
966
+ (void)in_fd;
967
+ (void)out_fd;
968
+ (void)chunk_size;
969
+ if (p) return pipeline_fail_msg(p, "file descriptor runner is not supported on this platform");
970
+ tf_set_last_error("file descriptor runner is not supported on this platform");
971
+ return TF_ERROR;
972
+ #else
973
+ if (!p || in_fd < 0 || out_fd < 0 || p->finished || p->finish_started) return TF_ERROR;
974
+ if (chunk_size == 0) chunk_size = 64 * 1024;
975
+
976
+ uint8_t *buf = malloc(chunk_size);
977
+ if (!buf) return pipeline_fail_msg(p, "out of memory");
978
+
979
+ tf_pipeline_sink_fn prev_sink = p->sinks[TF_CHAN_MAIN];
980
+ void *prev_user = p->sink_users[TF_CHAN_MAIN];
981
+ if (tf_pipeline_set_sink(p, TF_CHAN_MAIN, fd_output_sink, &out_fd) != TF_OK) {
982
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
983
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
984
+ free(buf);
985
+ return TF_ERROR;
986
+ }
987
+
988
+ int rc = TF_OK;
989
+ for (;;) {
990
+ ssize_t n = read(in_fd, buf, chunk_size);
991
+ if (n > 0) {
992
+ if (tf_pipeline_push(p, buf, (size_t)n) != TF_OK) {
993
+ rc = TF_ERROR;
994
+ break;
995
+ }
996
+ continue;
997
+ }
998
+ if (n == 0) break;
999
+ if (errno == EINTR) continue;
1000
+ rc = pipeline_fail_msg(p, "file descriptor input read error");
1001
+ break;
1002
+ }
1003
+
1004
+ if (rc == TF_OK && tf_pipeline_finish(p) != TF_OK) rc = TF_ERROR;
1005
+
1006
+ p->sinks[TF_CHAN_MAIN] = prev_sink;
1007
+ p->sink_users[TF_CHAN_MAIN] = prev_sink ? prev_user : NULL;
1008
+ free(buf);
1009
+ return rc;
1010
+ #endif
1011
+ }
1012
+
266
1013
  const char *tf_pipeline_error(tf_pipeline *p) {
267
1014
  return p ? p->error : NULL;
268
1015
  }
@@ -275,9 +1022,12 @@ void tf_pipeline_free(tf_pipeline *p) {
275
1022
  if (p->steps[i]) p->steps[i]->destroy(p->steps[i]);
276
1023
  }
277
1024
  free(p->steps);
1025
+ pipeline_clear_finish_batches(p);
1026
+ step_stats_free(p->step_stats, p->n_step_stats);
278
1027
  for (int i = 0; i < TF_NUM_CHANNELS; i++) {
279
1028
  tf_buffer_free(&p->output[i]);
280
1029
  }
1030
+ free(p->source_name);
281
1031
  free(p->error);
282
1032
  free(p);
283
1033
  }
@@ -291,7 +1041,12 @@ char *tf_compile_to_sql(const char *dsl, size_t len, char **error) {
291
1041
  tf_ir_plan_destroy(plan);
292
1042
  return NULL;
293
1043
  }
294
- tf_ir_infer_schema(plan);
1044
+ tf_set_last_error(NULL);
1045
+ if (tf_ir_infer_schema(plan) != TF_OK) {
1046
+ if (error) *error = schema_inference_error_string();
1047
+ tf_ir_plan_destroy(plan);
1048
+ return NULL;
1049
+ }
295
1050
  char *sql = tf_ir_to_sql(plan, error);
296
1051
  tf_ir_plan_destroy(plan);
297
1052
  return sql;
@@ -301,15 +1056,33 @@ char *tf_ir_plan_to_sql(const tf_ir_plan *plan, char **error) {
301
1056
  return tf_ir_to_sql(plan, error);
302
1057
  }
303
1058
 
304
- char *tf_compile_dsl(const char *dsl, size_t len, char **error) {
1059
+ char *tf_compile_dsl_with_host_policy(const char *dsl, size_t len,
1060
+ const tf_host_policy *policy, char **error) {
305
1061
  if (error) *error = NULL;
306
1062
  tf_ir_plan *plan = tf_dsl_parse(dsl, len, error);
307
1063
  if (!plan) return NULL;
1064
+
1065
+ if (tf_ir_validate_with_host_policy(plan, policy) != TF_OK) {
1066
+ if (error) *error = strdup(plan->error ? plan->error : "validation failed");
1067
+ tf_ir_plan_destroy(plan);
1068
+ return NULL;
1069
+ }
1070
+ tf_set_last_error(NULL);
1071
+ if (tf_ir_infer_schema(plan) != TF_OK) {
1072
+ if (error) *error = schema_inference_error_string();
1073
+ tf_ir_plan_destroy(plan);
1074
+ return NULL;
1075
+ }
1076
+
308
1077
  char *json = tf_ir_plan_to_json(plan);
309
1078
  tf_ir_plan_destroy(plan);
310
1079
  return json;
311
1080
  }
312
1081
 
1082
+ char *tf_compile_dsl(const char *dsl, size_t len, char **error) {
1083
+ return tf_compile_dsl_with_host_policy(dsl, len, NULL, error);
1084
+ }
1085
+
313
1086
  void tf_string_free(char *s) {
314
1087
  free(s);
315
1088
  }