tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
@@ -0,0 +1,252 @@
1
+ /*
2
+ * op_schema_infer.c -- Bounded schema inference report.
3
+ *
4
+ * schema infer [rows=N] samples decoded batches without retaining rows and
5
+ * emits one schema-report row per input column at finish(). It reports the
6
+ * decoded runtime types rather than changing parser behavior.
7
+ */
8
+
9
+ #include "internal.h"
10
+ #include "cJSON.h"
11
+ #include <stdlib.h>
12
+ #include <string.h>
13
+ #include <stdio.h>
14
+ #include <stdint.h>
15
+
16
+ #define SCHEMA_INFER_DEFAULT_ROWS 10000u
17
+
18
+ typedef struct {
19
+ char *name;
20
+ size_t missing;
21
+ size_t counts[6]; /* bool, int, float, string, date, timestamp */
22
+ } schema_infer_col;
23
+
24
+ typedef struct {
25
+ size_t rows_limit;
26
+ size_t rows_seen;
27
+ size_t rows_sampled;
28
+ int initialized;
29
+ int schema_changed;
30
+ schema_infer_col *cols;
31
+ size_t n_cols;
32
+ } schema_infer_state;
33
+
34
+ static int schema_infer_type_index(tf_type type) {
35
+ switch (type) {
36
+ case TF_TYPE_BOOL: return 0;
37
+ case TF_TYPE_INT64: return 1;
38
+ case TF_TYPE_FLOAT64: return 2;
39
+ case TF_TYPE_STRING: return 3;
40
+ case TF_TYPE_DATE: return 4;
41
+ case TF_TYPE_TIMESTAMP: return 5;
42
+ default: return -1;
43
+ }
44
+ }
45
+
46
+ static void append_token(char *buf, size_t buf_size, int *first, const char *token) {
47
+ if (!buf || buf_size == 0 || !token || !token[0]) return;
48
+ size_t used = strlen(buf);
49
+ if (used >= buf_size - 1) return;
50
+ snprintf(buf + used, buf_size - used, "%s%s", *first ? "" : ",", token);
51
+ *first = 0;
52
+ }
53
+
54
+ static void append_count(char *buf, size_t buf_size, int *first,
55
+ const char *name, size_t count) {
56
+ if (count == 0 || !buf || buf_size == 0 || !name) return;
57
+ size_t used = strlen(buf);
58
+ if (used >= buf_size - 1) return;
59
+ snprintf(buf + used, buf_size - used, "%s%s:%zu", *first ? "" : ";", name, count);
60
+ *first = 0;
61
+ }
62
+
63
+ static int schema_infer_init(schema_infer_state *st, const tf_batch *in) {
64
+ if (st->initialized) return TF_OK;
65
+ size_t n_cols = in ? in->n_cols : 0;
66
+ if (n_cols == 0) {
67
+ st->initialized = 1;
68
+ return TF_OK;
69
+ }
70
+
71
+ schema_infer_col *cols = calloc(n_cols, sizeof(schema_infer_col));
72
+ if (!cols) return TF_ERROR;
73
+ for (size_t c = 0; c < n_cols; c++) {
74
+ const char *name = in->col_names[c] ? in->col_names[c] : "";
75
+ cols[c].name = strdup(name);
76
+ if (!cols[c].name) {
77
+ for (size_t i = 0; i < c; i++) free(cols[i].name);
78
+ free(cols);
79
+ return TF_ERROR;
80
+ }
81
+ }
82
+
83
+ st->cols = cols;
84
+ st->n_cols = n_cols;
85
+ st->initialized = 1;
86
+ return TF_OK;
87
+ }
88
+
89
+ static int schema_infer_process(tf_step *self, tf_batch *in, tf_batch **out,
90
+ tf_side_channels *side) {
91
+ (void)side;
92
+ schema_infer_state *st = self->state;
93
+ *out = NULL;
94
+ if (schema_infer_init(st, in) != TF_OK) return TF_ERROR;
95
+ if (!in) return TF_OK;
96
+ if (st->initialized && in->n_cols != st->n_cols) st->schema_changed = 1;
97
+
98
+ for (size_t r = 0; r < in->n_rows; r++) {
99
+ st->rows_seen++;
100
+ if (st->rows_sampled >= st->rows_limit) continue;
101
+ st->rows_sampled++;
102
+ size_t n = in->n_cols < st->n_cols ? in->n_cols : st->n_cols;
103
+ for (size_t c = 0; c < n; c++) {
104
+ schema_infer_col *col = &st->cols[c];
105
+ if (tf_batch_is_null(in, r, c)) {
106
+ col->missing++;
107
+ continue;
108
+ }
109
+ int idx = schema_infer_type_index(in->col_types[c]);
110
+ if (idx >= 0) col->counts[idx]++;
111
+ }
112
+ }
113
+ return TF_OK;
114
+ }
115
+
116
+ static const char *schema_infer_inferred_type(const schema_infer_col *col,
117
+ const char **warning_token) {
118
+ size_t non_missing = 0;
119
+ size_t nonzero = 0;
120
+ int last = -1;
121
+ *warning_token = NULL;
122
+ for (int i = 0; i < 6; i++) {
123
+ non_missing += col->counts[i];
124
+ if (col->counts[i] > 0) { nonzero++; last = i; }
125
+ }
126
+ if (non_missing == 0) {
127
+ *warning_token = "no_non_null_sample";
128
+ return "unknown";
129
+ }
130
+ if (nonzero == 1) {
131
+ static const char *names[] = {"bool", "int", "float", "string", "date", "timestamp"};
132
+ return names[last];
133
+ }
134
+ if (col->counts[1] > 0 && col->counts[2] > 0 && nonzero == 2) {
135
+ *warning_token = "mixed_numeric";
136
+ return "float";
137
+ }
138
+ *warning_token = "mixed_types";
139
+ return "string";
140
+ }
141
+
142
+ static int schema_infer_set_output_schema(tf_batch *out) {
143
+ return tf_batch_set_schema(out, 0, "column", TF_TYPE_STRING) == TF_OK &&
144
+ tf_batch_set_schema(out, 1, "type", TF_TYPE_STRING) == TF_OK &&
145
+ tf_batch_set_schema(out, 2, "nullable", TF_TYPE_BOOL) == TF_OK &&
146
+ tf_batch_set_schema(out, 3, "non_null", TF_TYPE_BOOL) == TF_OK &&
147
+ tf_batch_set_schema(out, 4, "rows_seen", TF_TYPE_INT64) == TF_OK &&
148
+ tf_batch_set_schema(out, 5, "rows_sampled", TF_TYPE_INT64) == TF_OK &&
149
+ tf_batch_set_schema(out, 6, "missing", TF_TYPE_INT64) == TF_OK &&
150
+ tf_batch_set_schema(out, 7, "non_missing", TF_TYPE_INT64) == TF_OK &&
151
+ tf_batch_set_schema(out, 8, "observed_types", TF_TYPE_STRING) == TF_OK &&
152
+ tf_batch_set_schema(out, 9, "warning", TF_TYPE_STRING) == TF_OK ? TF_OK : TF_ERROR;
153
+ }
154
+
155
+ static int schema_infer_flush(tf_step *self, tf_batch **out,
156
+ tf_side_channels *side) {
157
+ (void)side;
158
+ schema_infer_state *st = self->state;
159
+ *out = NULL;
160
+ if (!st->initialized || st->n_cols == 0) return TF_OK;
161
+
162
+ tf_batch *ob = tf_batch_create(10, st->n_cols ? st->n_cols : 1);
163
+ if (!ob) return TF_ERROR;
164
+ int rc = TF_ERROR;
165
+ #define SCHEMA_INFER_WRITE(expr) do { if ((expr) != TF_OK) goto done; } while (0)
166
+
167
+ SCHEMA_INFER_WRITE(schema_infer_set_output_schema(ob));
168
+
169
+ static const char *type_names[] = {"bool", "int", "float", "string", "date", "timestamp"};
170
+ for (size_t c = 0; c < st->n_cols; c++) {
171
+ schema_infer_col *col = &st->cols[c];
172
+ size_t non_missing = 0;
173
+ char observed[256] = {0};
174
+ char warning[192] = {0};
175
+ int first_obs = 1;
176
+ int first_warn = 1;
177
+ for (int i = 0; i < 6; i++) {
178
+ non_missing += col->counts[i];
179
+ append_count(observed, sizeof(observed), &first_obs, type_names[i], col->counts[i]);
180
+ }
181
+ append_count(observed, sizeof(observed), &first_obs, "null", col->missing);
182
+ if (observed[0] == '\0') snprintf(observed, sizeof(observed), "none");
183
+
184
+ const char *type_warning = NULL;
185
+ const char *inferred = schema_infer_inferred_type(col, &type_warning);
186
+ if (type_warning) append_token(warning, sizeof(warning), &first_warn, type_warning);
187
+ if (st->rows_seen > st->rows_sampled) append_token(warning, sizeof(warning), &first_warn, "sample_limited");
188
+ if (st->schema_changed) append_token(warning, sizeof(warning), &first_warn, "schema_changed");
189
+ if (st->rows_sampled == 0) append_token(warning, sizeof(warning), &first_warn, "no_rows_sampled");
190
+
191
+ SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 0, col->name ? col->name : ""));
192
+ SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 1, inferred));
193
+ SCHEMA_INFER_WRITE(tf_batch_set_bool(ob, c, 2, col->missing > 0));
194
+ SCHEMA_INFER_WRITE(tf_batch_set_bool(ob, c, 3, col->missing == 0 && st->rows_sampled > 0));
195
+ SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 4, (int64_t)st->rows_seen));
196
+ SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 5, (int64_t)st->rows_sampled));
197
+ SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 6, (int64_t)col->missing));
198
+ SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 7, (int64_t)non_missing));
199
+ SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 8, observed));
200
+ SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 9, warning));
201
+ SCHEMA_INFER_WRITE(tf_batch_expose_row(ob, c));
202
+ }
203
+
204
+ *out = ob;
205
+ ob = NULL;
206
+ rc = TF_OK;
207
+
208
+ done:
209
+ if (ob) tf_batch_free(ob);
210
+ #undef SCHEMA_INFER_WRITE
211
+ return rc;
212
+ }
213
+
214
+ static void schema_infer_destroy(tf_step *self) {
215
+ if (!self) return;
216
+ schema_infer_state *st = self->state;
217
+ if (st) {
218
+ for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i].name);
219
+ free(st->cols);
220
+ free(st);
221
+ }
222
+ free(self);
223
+ }
224
+
225
+ static int parse_rows_arg(const cJSON *args, size_t *rows) {
226
+ *rows = SCHEMA_INFER_DEFAULT_ROWS;
227
+ if (!args) return TF_OK;
228
+ int has_rows = tf_json_get_size_arg(args, "rows",
229
+ 1, TF_MAX_COUNT_ARG,
230
+ rows, "schema-infer");
231
+ if (has_rows < 0) {
232
+ return TF_ERROR;
233
+ }
234
+ if (has_rows == 0) *rows = SCHEMA_INFER_DEFAULT_ROWS;
235
+ return TF_OK;
236
+ }
237
+
238
+ tf_step *tf_schema_infer_create(const cJSON *args) {
239
+ schema_infer_state *st = calloc(1, sizeof(schema_infer_state));
240
+ if (!st) return NULL;
241
+ if (parse_rows_arg(args, &st->rows_limit) != TF_OK) {
242
+ free(st);
243
+ return NULL;
244
+ }
245
+ tf_step *step = calloc(1, sizeof(tf_step));
246
+ if (!step) { free(st); return NULL; }
247
+ step->process = schema_infer_process;
248
+ step->flush = schema_infer_flush;
249
+ step->destroy = schema_infer_destroy;
250
+ step->state = st;
251
+ return step;
252
+ }
package/csrc/op_select.c CHANGED
@@ -14,76 +14,75 @@
14
14
  typedef struct {
15
15
  char **col_names;
16
16
  size_t n_cols;
17
+ int use_selector_syntax;
17
18
  } select_state;
18
19
 
20
+ static int select_write_error(tf_side_channels *side, const char *msg) {
21
+ if (!side || !side->errors || !msg) return TF_OK;
22
+ char buf[256];
23
+ snprintf(buf, sizeof(buf), "{\"op\":\"select\",\"error\":\"%s\"}", msg);
24
+ return tf_buffer_write_line(side->errors, buf);
25
+ }
26
+
19
27
  static int select_process(tf_step *self, tf_batch *in, tf_batch **out,
20
28
  tf_side_channels *side) {
21
29
  select_state *st = self->state;
22
30
  *out = NULL;
23
31
 
24
- /* Resolve column indices */
25
- int *indices = malloc(st->n_cols * sizeof(int));
26
- if (!indices) return TF_ERROR;
27
-
28
- for (size_t i = 0; i < st->n_cols; i++) {
29
- indices[i] = tf_batch_col_index(in, st->col_names[i]);
30
- if (indices[i] < 0 && side && side->errors) {
31
- char buf[128];
32
- snprintf(buf, sizeof(buf),
33
- "{\"op\":\"select\",\"error\":\"column '%s' not found\"}\n",
34
- st->col_names[i]);
35
- tf_buffer_write_str(side->errors, buf);
32
+ int *indices = NULL;
33
+ size_t n_indices = st->n_cols;
34
+
35
+ if (st->use_selector_syntax) {
36
+ char *error = NULL;
37
+ if (tf_column_selectors_resolve(st->col_names, st->n_cols,
38
+ in->col_names, in->col_types, in->n_cols,
39
+ &indices, &n_indices, &error) != TF_OK) {
40
+ int err_rc = select_write_error(side, error ? error : "selector resolution failed");
41
+ free(error);
42
+ if (err_rc != TF_OK) return TF_ERROR;
43
+ return TF_ERROR;
44
+ }
45
+ } else {
46
+ indices = malloc(st->n_cols * sizeof(int));
47
+ if (!indices) return TF_ERROR;
48
+ for (size_t i = 0; i < st->n_cols; i++) {
49
+ indices[i] = tf_batch_col_index(in, st->col_names[i]);
50
+ if (indices[i] < 0 && side && side->errors) {
51
+ char msg[128];
52
+ snprintf(msg, sizeof(msg), "column '%s' not found", st->col_names[i]);
53
+ if (select_write_error(side, msg) != TF_OK) {
54
+ free(indices);
55
+ return TF_ERROR;
56
+ }
57
+ }
36
58
  }
37
59
  }
38
60
 
39
- /* Create output batch */
40
- tf_batch *ob = tf_batch_create(st->n_cols, in->n_rows);
61
+ tf_batch *ob = tf_batch_create(n_indices, in->n_rows);
41
62
  if (!ob) { free(indices); return TF_ERROR; }
42
63
 
43
- for (size_t i = 0; i < st->n_cols; i++) {
44
- if (indices[i] >= 0) {
45
- tf_batch_set_schema(ob, i, st->col_names[i], in->col_types[indices[i]]);
64
+ size_t *selected_cols = malloc(n_indices * sizeof(size_t));
65
+ if (!selected_cols) { free(indices); tf_batch_free(ob); return TF_ERROR; }
66
+ for (size_t i = 0; i < n_indices; i++) {
67
+ int ci = indices[i];
68
+ if (ci >= 0) {
69
+ selected_cols[i] = (size_t)ci;
70
+ if (tf_batch_set_schema(ob, i, in->col_names[(size_t)ci], in->col_types[(size_t)ci]) != TF_OK) {
71
+ free(selected_cols); free(indices); tf_batch_free(ob); return TF_ERROR;
72
+ }
46
73
  } else {
47
- tf_batch_set_schema(ob, i, st->col_names[i], TF_TYPE_NULL);
74
+ selected_cols[i] = SIZE_MAX;
75
+ if (tf_batch_set_schema(ob, i, st->col_names[i], TF_TYPE_NULL) != TF_OK) {
76
+ free(selected_cols); free(indices); tf_batch_free(ob); return TF_ERROR;
77
+ }
48
78
  }
49
79
  }
50
80
 
51
- /* Copy rows */
52
- for (size_t r = 0; r < in->n_rows; r++) {
53
- tf_batch_ensure_capacity(ob, r + 1);
54
- for (size_t i = 0; i < st->n_cols; i++) {
55
- int ci = indices[i];
56
- if (ci < 0 || tf_batch_is_null(in, r, ci)) {
57
- tf_batch_set_null(ob, r, i);
58
- continue;
59
- }
60
- switch (in->col_types[ci]) {
61
- case TF_TYPE_BOOL:
62
- tf_batch_set_bool(ob, r, i, tf_batch_get_bool(in, r, ci));
63
- break;
64
- case TF_TYPE_INT64:
65
- tf_batch_set_int64(ob, r, i, tf_batch_get_int64(in, r, ci));
66
- break;
67
- case TF_TYPE_FLOAT64:
68
- tf_batch_set_float64(ob, r, i, tf_batch_get_float64(in, r, ci));
69
- break;
70
- case TF_TYPE_STRING:
71
- tf_batch_set_string(ob, r, i, tf_batch_get_string(in, r, ci));
72
- break;
73
- case TF_TYPE_DATE:
74
- tf_batch_set_date(ob, r, i, tf_batch_get_date(in, r, ci));
75
- break;
76
- case TF_TYPE_TIMESTAMP:
77
- tf_batch_set_timestamp(ob, r, i, tf_batch_get_timestamp(in, r, ci));
78
- break;
79
- default:
80
- tf_batch_set_null(ob, r, i);
81
- break;
82
- }
83
- }
84
- ob->n_rows = r + 1;
81
+ if (tf_batch_copy_selected_columns(ob, in, selected_cols, n_indices) != TF_OK) {
82
+ free(selected_cols); free(indices); tf_batch_free(ob); return TF_ERROR;
85
83
  }
86
84
 
85
+ free(selected_cols);
87
86
  free(indices);
88
87
  *out = ob;
89
88
  return TF_OK;
@@ -95,13 +94,19 @@ static int select_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
95
94
  return TF_OK;
96
95
  }
97
96
 
98
- static void select_destroy(tf_step *self) {
99
- select_state *st = self->state;
97
+ static void select_state_free(select_state *st) {
100
98
  if (st) {
101
- for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
99
+ if (st->col_names) {
100
+ for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
101
+ }
102
102
  free(st->col_names);
103
103
  free(st);
104
104
  }
105
+ }
106
+
107
+ static void select_destroy(tf_step *self) {
108
+ if (!self) return;
109
+ select_state_free(self->state);
105
110
  free(self);
106
111
  }
107
112
 
@@ -116,25 +121,220 @@ tf_step *tf_select_create(const cJSON *args) {
116
121
  select_state *st = calloc(1, sizeof(select_state));
117
122
  if (!st) return NULL;
118
123
  st->n_cols = (size_t)n;
119
- st->col_names = malloc(n * sizeof(char *));
120
- if (!st->col_names) { free(st); return NULL; }
124
+ st->col_names = calloc((size_t)n, sizeof(char *));
125
+ if (!st->col_names) { select_state_free(st); return NULL; }
121
126
 
122
127
  for (int i = 0; i < n; i++) {
123
128
  cJSON *item = cJSON_GetArrayItem(cols, i);
124
- if (!cJSON_IsString(item)) {
125
- for (int j = 0; j < i; j++) free(st->col_names[j]);
126
- free(st->col_names);
127
- free(st);
128
- return NULL;
129
- }
129
+ if (!cJSON_IsString(item)) { select_state_free(st); return NULL; }
130
130
  st->col_names[i] = strdup(item->valuestring);
131
+ if (!st->col_names[i]) { select_state_free(st); return NULL; }
132
+ if (tf_column_selector_has_syntax(item->valuestring)) st->use_selector_syntax = 1;
131
133
  }
132
134
 
133
- tf_step *step = malloc(sizeof(tf_step));
134
- if (!step) { select_destroy(&(tf_step){.state = st}); return NULL; }
135
+ tf_step *step = calloc(1, sizeof(tf_step));
136
+ if (!step) { select_state_free(st); return NULL; }
135
137
  step->process = select_process;
136
138
  step->flush = select_flush;
137
139
  step->destroy = select_destroy;
138
140
  step->state = st;
139
141
  return step;
140
142
  }
143
+
144
+
145
+ typedef struct {
146
+ char **move_names;
147
+ size_t n_move;
148
+ char *before;
149
+ char *after;
150
+ } relocate_state;
151
+
152
+ static int relocate_write_error(tf_side_channels *side, const char *msg, const char *name) {
153
+ if (!side || !side->errors) return TF_OK;
154
+ char buf[256];
155
+ snprintf(buf, sizeof(buf),
156
+ "{\"op\":\"relocate\",\"error\":\"%s '%s'\"}",
157
+ msg, name ? name : "");
158
+ return tf_buffer_write_line(side->errors, buf);
159
+ }
160
+
161
+ static int relocate_process(tf_step *self, tf_batch *in, tf_batch **out,
162
+ tf_side_channels *side) {
163
+ relocate_state *st = self->state;
164
+ *out = NULL;
165
+
166
+ size_t n_in = in->n_cols;
167
+ int *move_idx = NULL;
168
+ size_t n_move = 0;
169
+ char *selector_error = NULL;
170
+ if (tf_column_selectors_resolve(st->move_names, st->n_move,
171
+ in->col_names, in->col_types, in->n_cols,
172
+ &move_idx, &n_move, &selector_error) != TF_OK) {
173
+ int err_rc = relocate_write_error(side, selector_error ? selector_error : "selector resolution failed", "");
174
+ free(selector_error);
175
+ if (err_rc != TF_OK) return TF_ERROR;
176
+ return TF_ERROR;
177
+ }
178
+
179
+ int *is_moving = calloc(n_in ? n_in : 1, sizeof(int));
180
+ int *order = malloc(n_in ? n_in * sizeof(int) : sizeof(int));
181
+ if (!is_moving || !order) {
182
+ free(move_idx); free(is_moving); free(order);
183
+ return TF_ERROR;
184
+ }
185
+
186
+ for (size_t i = 0; i < n_move; i++) {
187
+ int ci = move_idx[i];
188
+ if (ci < 0 || (size_t)ci >= n_in || is_moving[ci]) {
189
+ int err_rc = relocate_write_error(side, "invalid relocated column", "");
190
+ free(move_idx); free(is_moving); free(order);
191
+ if (err_rc != TF_OK) return TF_ERROR;
192
+ return TF_ERROR;
193
+ }
194
+ is_moving[ci] = 1;
195
+ }
196
+
197
+ const char *anchor = st->before ? st->before : st->after;
198
+ int anchor_idx = -1;
199
+ if (anchor) {
200
+ anchor_idx = tf_batch_col_index(in, anchor);
201
+ if (anchor_idx < 0) {
202
+ int err_rc = relocate_write_error(side, "anchor column not found", anchor);
203
+ free(move_idx); free(is_moving); free(order);
204
+ if (err_rc != TF_OK) return TF_ERROR;
205
+ return TF_ERROR;
206
+ }
207
+ if (is_moving[anchor_idx]) {
208
+ int err_rc = relocate_write_error(side, "anchor column is being relocated", anchor);
209
+ free(move_idx); free(is_moving); free(order);
210
+ if (err_rc != TF_OK) return TF_ERROR;
211
+ return TF_ERROR;
212
+ }
213
+ }
214
+
215
+ size_t n_order = 0;
216
+ if (!anchor) {
217
+ for (size_t i = 0; i < n_move; i++) order[n_order++] = move_idx[i];
218
+ for (size_t i = 0; i < n_in; i++) {
219
+ if (!is_moving[i]) order[n_order++] = (int)i;
220
+ }
221
+ } else {
222
+ for (size_t i = 0; i < n_in; i++) {
223
+ if (is_moving[i]) continue;
224
+ if (st->before && (int)i == anchor_idx) {
225
+ for (size_t j = 0; j < n_move; j++) order[n_order++] = move_idx[j];
226
+ }
227
+ order[n_order++] = (int)i;
228
+ if (st->after && (int)i == anchor_idx) {
229
+ for (size_t j = 0; j < n_move; j++) order[n_order++] = move_idx[j];
230
+ }
231
+ }
232
+ }
233
+
234
+ if (n_order != n_in) {
235
+ int err_rc = relocate_write_error(side, "internal order size mismatch", "");
236
+ free(move_idx); free(is_moving); free(order);
237
+ if (err_rc != TF_OK) return TF_ERROR;
238
+ return TF_ERROR;
239
+ }
240
+
241
+ tf_batch *ob = tf_batch_create(n_in, in->n_rows);
242
+ if (!ob) {
243
+ free(move_idx); free(is_moving); free(order);
244
+ return TF_ERROR;
245
+ }
246
+
247
+ size_t *selected_cols = malloc(n_in * sizeof(size_t));
248
+ if (!selected_cols) {
249
+ free(move_idx); free(is_moving); free(order); tf_batch_free(ob);
250
+ return TF_ERROR;
251
+ }
252
+ for (size_t i = 0; i < n_in; i++) {
253
+ int ci = order[i];
254
+ selected_cols[i] = (size_t)ci;
255
+ if (tf_batch_set_schema(ob, i, in->col_names[(size_t)ci], in->col_types[(size_t)ci]) != TF_OK) {
256
+ free(selected_cols); free(move_idx); free(is_moving); free(order); tf_batch_free(ob);
257
+ return TF_ERROR;
258
+ }
259
+ }
260
+
261
+ if (tf_batch_copy_selected_columns(ob, in, selected_cols, n_in) != TF_OK) {
262
+ free(selected_cols); free(move_idx); free(is_moving); free(order); tf_batch_free(ob);
263
+ return TF_ERROR;
264
+ }
265
+
266
+ free(selected_cols);
267
+ free(move_idx);
268
+ free(is_moving);
269
+ free(order);
270
+ *out = ob;
271
+ return TF_OK;
272
+ }
273
+
274
+ static int relocate_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
275
+ (void)self; (void)side;
276
+ *out = NULL;
277
+ return TF_OK;
278
+ }
279
+
280
+ static void relocate_state_free(relocate_state *st) {
281
+ if (st) {
282
+ if (st->move_names) {
283
+ for (size_t i = 0; i < st->n_move; i++) free(st->move_names[i]);
284
+ }
285
+ free(st->move_names);
286
+ free(st->before);
287
+ free(st->after);
288
+ free(st);
289
+ }
290
+ }
291
+
292
+ static void relocate_destroy(tf_step *self) {
293
+ if (!self) return;
294
+ relocate_state_free(self->state);
295
+ free(self);
296
+ }
297
+
298
+ tf_step *tf_relocate_create(const cJSON *args) {
299
+ if (!args) return NULL;
300
+ cJSON *cols = cJSON_GetObjectItemCaseSensitive(args, "columns");
301
+ if (!cJSON_IsArray(cols)) return NULL;
302
+
303
+ cJSON *before = cJSON_GetObjectItemCaseSensitive(args, "before");
304
+ cJSON *after = cJSON_GetObjectItemCaseSensitive(args, "after");
305
+ if (before && after) return NULL;
306
+ if (before && !cJSON_IsString(before)) return NULL;
307
+ if (after && !cJSON_IsString(after)) return NULL;
308
+
309
+ int n = cJSON_GetArraySize(cols);
310
+ if (n <= 0) return NULL;
311
+
312
+ relocate_state *st = calloc(1, sizeof(relocate_state));
313
+ if (!st) return NULL;
314
+ st->n_move = (size_t)n;
315
+ st->move_names = calloc((size_t)n, sizeof(char *));
316
+ if (!st->move_names) { relocate_state_free(st); return NULL; }
317
+
318
+ for (int i = 0; i < n; i++) {
319
+ cJSON *item = cJSON_GetArrayItem(cols, i);
320
+ if (!cJSON_IsString(item)) { relocate_state_free(st); return NULL; }
321
+ st->move_names[i] = strdup(item->valuestring);
322
+ if (!st->move_names[i]) { relocate_state_free(st); return NULL; }
323
+ }
324
+ if (before) {
325
+ st->before = strdup(before->valuestring);
326
+ if (!st->before) { relocate_state_free(st); return NULL; }
327
+ }
328
+ if (after) {
329
+ st->after = strdup(after->valuestring);
330
+ if (!st->after) { relocate_state_free(st); return NULL; }
331
+ }
332
+
333
+ tf_step *step = calloc(1, sizeof(tf_step));
334
+ if (!step) { relocate_state_free(st); return NULL; }
335
+ step->process = relocate_process;
336
+ step->flush = relocate_flush;
337
+ step->destroy = relocate_destroy;
338
+ step->state = st;
339
+ return step;
340
+ }