tranfi 0.0.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/LICENSE +177 -21
  2. package/NOTICE +8 -0
  3. package/README.md +627 -0
  4. package/app/assets/index-6quYZ5Ap.css +5 -0
  5. package/app/assets/index-BIAIKnrp.js +160 -0
  6. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  7. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  8. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  9. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  10. package/app/index.html +13 -0
  11. package/binding.gyp +121 -0
  12. package/csrc/arena.c +93 -0
  13. package/csrc/batch.c +976 -0
  14. package/csrc/buffer.c +154 -0
  15. package/csrc/cJSON.c +3386 -0
  16. package/csrc/cJSON.h +316 -0
  17. package/csrc/codec_csv.c +1951 -0
  18. package/csrc/codec_jsonl.c +1086 -0
  19. package/csrc/codec_table.c +248 -0
  20. package/csrc/codec_text.c +447 -0
  21. package/csrc/compiler.c +130 -0
  22. package/csrc/config.h +21 -0
  23. package/csrc/date_utils.h +94 -0
  24. package/csrc/dsl.c +5417 -0
  25. package/csrc/dsl.h +22 -0
  26. package/csrc/expr.c +1553 -0
  27. package/csrc/expr.h +58 -0
  28. package/csrc/internal.h +539 -0
  29. package/csrc/ir.c +166 -0
  30. package/csrc/ir.h +208 -0
  31. package/csrc/ir_schema.c +75 -0
  32. package/csrc/ir_serialize.c +166 -0
  33. package/csrc/ir_sql.c +1822 -0
  34. package/csrc/ir_validate.c +576 -0
  35. package/csrc/json_path.c +210 -0
  36. package/csrc/main.c +1241 -0
  37. package/csrc/memory_estimate.c +477 -0
  38. package/csrc/op_acf.c +283 -0
  39. package/csrc/op_across.c +477 -0
  40. package/csrc/op_anomaly.c +255 -0
  41. package/csrc/op_assert.c +761 -0
  42. package/csrc/op_bin.c +248 -0
  43. package/csrc/op_cast.c +523 -0
  44. package/csrc/op_clip.c +99 -0
  45. package/csrc/op_date_trunc.c +355 -0
  46. package/csrc/op_datetime.c +394 -0
  47. package/csrc/op_derive.c +216 -0
  48. package/csrc/op_diff.c +250 -0
  49. package/csrc/op_ewma.c +222 -0
  50. package/csrc/op_explode.c +206 -0
  51. package/csrc/op_fill_down.c +235 -0
  52. package/csrc/op_fill_null.c +268 -0
  53. package/csrc/op_filter.c +181 -0
  54. package/csrc/op_frequency.c +721 -0
  55. package/csrc/op_grep.c +181 -0
  56. package/csrc/op_group_agg.c +1956 -0
  57. package/csrc/op_hash.c +159 -0
  58. package/csrc/op_head.c +84 -0
  59. package/csrc/op_interpolate.c +445 -0
  60. package/csrc/op_join.c +2902 -0
  61. package/csrc/op_json_extract.c +227 -0
  62. package/csrc/op_json_filter.c +384 -0
  63. package/csrc/op_json_flatten.c +293 -0
  64. package/csrc/op_json_schema.c +503 -0
  65. package/csrc/op_label_encode.c +419 -0
  66. package/csrc/op_lag.c +181 -0
  67. package/csrc/op_lead.c +242 -0
  68. package/csrc/op_normalize.c +510 -0
  69. package/csrc/op_onehot.c +457 -0
  70. package/csrc/op_pivot.c +1754 -0
  71. package/csrc/op_quarantine.c +189 -0
  72. package/csrc/op_registry.c +3044 -0
  73. package/csrc/op_rename.c +129 -0
  74. package/csrc/op_replace.c +354 -0
  75. package/csrc/op_rleid.c +297 -0
  76. package/csrc/op_rowid.c +559 -0
  77. package/csrc/op_sample.c +158 -0
  78. package/csrc/op_schema.c +1341 -0
  79. package/csrc/op_schema_infer.c +252 -0
  80. package/csrc/op_select.c +340 -0
  81. package/csrc/op_set.c +3449 -0
  82. package/csrc/op_skip.c +95 -0
  83. package/csrc/op_sort.c +819 -0
  84. package/csrc/op_source_name.c +120 -0
  85. package/csrc/op_split.c +151 -0
  86. package/csrc/op_split_data.c +119 -0
  87. package/csrc/op_stack.c +271 -0
  88. package/csrc/op_stats.c +875 -0
  89. package/csrc/op_step.c +333 -0
  90. package/csrc/op_tail.c +105 -0
  91. package/csrc/op_tee.c +338 -0
  92. package/csrc/op_top.c +357 -0
  93. package/csrc/op_trim.c +138 -0
  94. package/csrc/op_unique.c +1343 -0
  95. package/csrc/op_unpivot.c +193 -0
  96. package/csrc/op_validate.c +648 -0
  97. package/csrc/op_window.c +591 -0
  98. package/csrc/path_policy.c +85 -0
  99. package/csrc/pipeline.c +1088 -0
  100. package/csrc/recipes.c +104 -0
  101. package/csrc/recipes.h +27 -0
  102. package/csrc/report.c +506 -0
  103. package/csrc/report.h +22 -0
  104. package/csrc/selector.c +1097 -0
  105. package/csrc/size_utils.c +348 -0
  106. package/csrc/spill.c +317 -0
  107. package/csrc/spill.h +21 -0
  108. package/csrc/tranfi.h +291 -0
  109. package/csrc/transform.h +209 -0
  110. package/csrc/transform_api.c +2237 -0
  111. package/csrc/transform_categorical.c +923 -0
  112. package/csrc/transform_internal.h +472 -0
  113. package/csrc/transform_json.c +3812 -0
  114. package/csrc/transform_numeric.c +1966 -0
  115. package/csrc/transform_sha256.c +154 -0
  116. package/csrc/transform_wasm.h +162 -0
  117. package/csrc/transform_wasm_api.c +1373 -0
  118. package/csrc/wasm_api.c +218 -0
  119. package/napi_api.c +534 -0
  120. package/napi_transform.c +1648 -0
  121. package/napi_transform.h +8 -0
  122. package/package.json +64 -59
  123. package/scripts/install-native.js +76 -0
  124. package/scripts/prepack.js +64 -0
  125. package/scripts/sync-csrc.js +23 -0
  126. package/src/cli.js +190 -0
  127. package/src/engines/duckdb.js +142 -0
  128. package/src/index.js +925 -0
  129. package/src/memory_policy.js +411 -0
  130. package/src/native.js +18 -0
  131. package/src/pipeline.js +709 -0
  132. package/src/recipe_json.js +80 -0
  133. package/src/server.js +279 -0
  134. package/src/transform.js +403 -0
  135. package/src/transform_error.js +10 -0
  136. package/src/wasm.js +21 -0
  137. package/wasm/index.js +732 -0
  138. package/wasm/package.json +1 -0
  139. package/wasm/tranfi_core.js +0 -0
  140. package/wasm/transform.js +1156 -0
  141. package/wasm/worker.js +786 -0
  142. package/dist/bundle.js +0 -1
  143. package/index.html +0 -18
  144. package/src/app.css +0 -169
  145. package/src/app.js +0 -203
  146. package/src/app.vue +0 -250
  147. package/src/bulma-input.vue +0 -110
  148. package/src/common-inputs.js +0 -28
  149. package/src/main.js +0 -20
  150. package/src/transforms.js +0 -166
  151. package/webpack.config.js +0 -108
@@ -0,0 +1,510 @@
1
+ /*
2
+ * op_normalize.c — Min-max or z-score normalization.
3
+ * Aggregate op: buffers all rows, computes stats, then emits normalized.
4
+ *
5
+ * Config: {"columns": ["price", "score"], "method": "minmax"}
6
+ */
7
+
8
+ #include "internal.h"
9
+ #include "cJSON.h"
10
+ #include <stdlib.h>
11
+ #include <string.h>
12
+ #include <stdio.h>
13
+ #include <math.h>
14
+
15
+ typedef enum {
16
+ NORM_MINMAX,
17
+ NORM_ZSCORE
18
+ } norm_method;
19
+
20
+ typedef enum {
21
+ NORM_MISSING_ERROR,
22
+ NORM_MISSING_NULL,
23
+ NORM_MISSING_IGNORE
24
+ } norm_missing_policy;
25
+
26
+ typedef enum {
27
+ NORM_TYPE_FAIL,
28
+ NORM_TYPE_NULL
29
+ } norm_type_policy;
30
+
31
+ typedef struct {
32
+ int col_idx;
33
+ int missing_null;
34
+ int ignored;
35
+ int type_null;
36
+ size_t count;
37
+ double mean;
38
+ double m2;
39
+ double min_val;
40
+ double max_val;
41
+ } col_stats;
42
+
43
+ typedef struct {
44
+ tf_batch *batch; /* single-row batch */
45
+ } buf_row;
46
+
47
+ typedef struct {
48
+ char **columns;
49
+ size_t n_columns;
50
+ norm_method method;
51
+ norm_missing_policy missing;
52
+ norm_type_policy on_type_error;
53
+ col_stats *stats;
54
+ buf_row *rows;
55
+ size_t n_rows;
56
+ size_t cap_rows;
57
+ int has_schema;
58
+ size_t schema_n_cols;
59
+ char **schema_names;
60
+ tf_type *schema_types;
61
+ size_t audit_limit;
62
+ size_t audit_emitted;
63
+ int audit;
64
+ tf_audit_options audit_opts;
65
+ } normalize_state;
66
+
67
+ static int parse_method(const char *s, norm_method *out) {
68
+ if (!s || strcmp(s, "minmax") == 0) { *out = NORM_MINMAX; return TF_OK; }
69
+ if (strcmp(s, "zscore") == 0) { *out = NORM_ZSCORE; return TF_OK; }
70
+ tf_set_last_error("normalize: method must be minmax or zscore");
71
+ return TF_ERROR;
72
+ }
73
+
74
+ static const char *method_name(norm_method method) {
75
+ return method == NORM_ZSCORE ? "zscore" : "minmax";
76
+ }
77
+
78
+ static int normalize_is_numeric(tf_type type) {
79
+ return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
80
+ }
81
+
82
+ static int normalize_parse_missing_policy(const cJSON *args, norm_missing_policy *out) {
83
+ *out = NORM_MISSING_ERROR;
84
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
85
+ if (!j) return TF_OK;
86
+ if (!cJSON_IsString(j)) {
87
+ tf_set_last_error("normalize: missing must be error, null, or ignore");
88
+ return TF_ERROR;
89
+ }
90
+ if (strcmp(j->valuestring, "error") == 0) *out = NORM_MISSING_ERROR;
91
+ else if (strcmp(j->valuestring, "null") == 0) *out = NORM_MISSING_NULL;
92
+ else if (strcmp(j->valuestring, "ignore") == 0) *out = NORM_MISSING_IGNORE;
93
+ else {
94
+ tf_set_last_error("normalize: missing must be error, null, or ignore");
95
+ return TF_ERROR;
96
+ }
97
+ return TF_OK;
98
+ }
99
+
100
+ static int normalize_parse_type_policy(const cJSON *args, norm_type_policy *out) {
101
+ *out = NORM_TYPE_FAIL;
102
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
103
+ if (!j) return TF_OK;
104
+ if (!cJSON_IsString(j)) {
105
+ tf_set_last_error("normalize: on_type_error must be fail or null");
106
+ return TF_ERROR;
107
+ }
108
+ if (strcmp(j->valuestring, "fail") == 0) *out = NORM_TYPE_FAIL;
109
+ else if (strcmp(j->valuestring, "null") == 0) *out = NORM_TYPE_NULL;
110
+ else {
111
+ tf_set_last_error("normalize: on_type_error must be fail or null");
112
+ return TF_ERROR;
113
+ }
114
+ return TF_OK;
115
+ }
116
+
117
+ static int normalize_set_col_error(normalize_state *st, const tf_batch *in, size_t i) {
118
+ if (st->stats[i].col_idx < 0) {
119
+ char msg[256];
120
+ snprintf(msg, sizeof(msg), "normalize: column '%s' not found", st->columns[i]);
121
+ tf_set_last_error(msg);
122
+ return TF_ERROR;
123
+ }
124
+ if (!normalize_is_numeric(in->col_types[st->stats[i].col_idx])) {
125
+ char msg[256];
126
+ snprintf(msg, sizeof(msg), "normalize: column '%s' must be numeric", st->columns[i]);
127
+ tf_set_last_error(msg);
128
+ return TF_ERROR;
129
+ }
130
+ return TF_OK;
131
+ }
132
+
133
+ static double normalize_compute(norm_method method, const col_stats *cs, double val) {
134
+ if (method == NORM_MINMAX) {
135
+ double range = cs->max_val - cs->min_val;
136
+ return (range > 0) ? (val - cs->min_val) / range : 0;
137
+ }
138
+ double std = (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 1;
139
+ return (std > 0) ? (val - cs->mean) / std : 0;
140
+ }
141
+
142
+ static double normalize_stddev(const col_stats *cs) {
143
+ return (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 0;
144
+ }
145
+
146
+ static int normalize_add_audit_number(cJSON *obj, const char *key,
147
+ const tf_audit_options *opts,
148
+ const char *column, double value) {
149
+ if (opts && (tf_audit_column_is_redacted(opts, column) || tf_audit_column_is_hashed(opts, column))) {
150
+ char raw[128];
151
+ if (tf_format_float64(raw, sizeof(raw), value) != TF_OK) return TF_ERROR;
152
+ return tf_json_add_audit_string(obj, key, opts, column, raw);
153
+ }
154
+ return tf_json_add_number(obj, key, value);
155
+ }
156
+
157
+ static double get_numeric(const tf_batch *b, size_t r, int ci) {
158
+ if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
159
+ if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
160
+ return 0;
161
+ }
162
+
163
+ static int emit_normalize_audit(normalize_state *st, const tf_batch *before_b,
164
+ const tf_batch *after_b, size_t before_row,
165
+ size_t after_row, size_t col, size_t row_no,
166
+ const col_stats *cs, tf_side_channels *side) {
167
+ if (!st->audit || st->audit_emitted >= st->audit_limit || !side || !side->stats) return TF_OK;
168
+ cJSON *obj = cJSON_CreateObject();
169
+ if (!obj) return TF_ERROR;
170
+ int rc = TF_ERROR;
171
+ if (tf_json_add_string(obj, "type", "audit") != TF_OK ||
172
+ tf_json_add_string(obj, "op", "normalize") != TF_OK ||
173
+ tf_json_add_string(obj, "event", "value_changed") != TF_OK ||
174
+ tf_json_add_string(obj, "reason",
175
+ st->method == NORM_ZSCORE ? "normalize_zscore" : "normalize_minmax") != TF_OK ||
176
+ tf_json_add_string(obj, "channel", "audit") != TF_OK) {
177
+ goto done;
178
+ }
179
+ const char *column_name = after_b->col_names[col] ? after_b->col_names[col] : "";
180
+ if (tf_json_add_string(obj, "column", column_name) != TF_OK ||
181
+ tf_json_add_string(obj, "method", method_name(st->method)) != TF_OK ||
182
+ tf_json_add_number(obj, "row", (double)row_no) != TF_OK) {
183
+ goto done;
184
+ }
185
+ cJSON *before = tf_audit_cell_to_json(before_b, before_row, col, &st->audit_opts);
186
+ if (!before || tf_json_add_item(obj, "before", before) != TF_OK) goto done;
187
+ cJSON *after = tf_audit_cell_to_json(after_b, after_row, col, &st->audit_opts);
188
+ if (!after || tf_json_add_item(obj, "after", after) != TF_OK) goto done;
189
+ if (tf_json_add_number(obj, "count", (double)cs->count) != TF_OK) goto done;
190
+ if (normalize_add_audit_number(obj, "min", &st->audit_opts, column_name, cs->min_val) != TF_OK ||
191
+ normalize_add_audit_number(obj, "max", &st->audit_opts, column_name, cs->max_val) != TF_OK ||
192
+ normalize_add_audit_number(obj, "mean", &st->audit_opts, column_name, cs->mean) != TF_OK ||
193
+ normalize_add_audit_number(obj, "stddev", &st->audit_opts, column_name, normalize_stddev(cs)) != TF_OK) {
194
+ goto done;
195
+ }
196
+ cJSON *row_obj = tf_audit_row_to_json(after_b, after_row, &st->audit_opts);
197
+ if (row_obj) {
198
+ if (tf_json_add_item(obj, "data", row_obj) != TF_OK) goto done;
199
+ } else if (st->audit_opts.include_row) {
200
+ goto done;
201
+ }
202
+ rc = tf_buffer_write_json_line(side->stats, obj);
203
+ done:
204
+ cJSON_Delete(obj);
205
+ if (rc == TF_OK) st->audit_emitted++;
206
+ return rc;
207
+ }
208
+
209
+ static int add_row(normalize_state *st, tf_batch *b, size_t r) {
210
+ if (st->n_rows >= st->cap_rows) {
211
+ size_t min_cap = 0, newcap = 0;
212
+ if (tf_size_add(st->n_rows, 1, &min_cap) != TF_OK ||
213
+ tf_size_grow_pow2(st->cap_rows, min_cap, 256, &newcap) != TF_OK) {
214
+ return TF_ERROR;
215
+ }
216
+ buf_row *tmp = tf_reallocarray_checked(st->rows, newcap, sizeof(buf_row));
217
+ if (!tmp) return TF_ERROR;
218
+ st->rows = tmp;
219
+ st->cap_rows = newcap;
220
+ }
221
+ tf_batch *rb = tf_batch_create(b->n_cols, 1);
222
+ if (!rb) return TF_ERROR;
223
+ if (tf_batch_clone_schema(rb, b) != TF_OK) {
224
+ tf_batch_free(rb);
225
+ return TF_ERROR;
226
+ }
227
+ if (tf_batch_copy_row(rb, 0, b, r) != TF_OK) {
228
+ tf_batch_free(rb);
229
+ return TF_ERROR;
230
+ }
231
+ if (tf_batch_expose_row(rb, 0) != TF_OK) {
232
+ tf_batch_free(rb);
233
+ return TF_ERROR;
234
+ }
235
+ st->rows[st->n_rows].batch = rb;
236
+ st->n_rows++;
237
+ return TF_OK;
238
+ }
239
+
240
+ static int normalize_capture_schema(normalize_state *st, const tf_batch *in) {
241
+ st->schema_n_cols = in->n_cols;
242
+ st->schema_names = tf_callocarray_checked(in->n_cols, sizeof(char *));
243
+ st->schema_types = tf_callocarray_checked(in->n_cols, sizeof(tf_type));
244
+ if (!st->schema_names || !st->schema_types) return TF_ERROR;
245
+ for (size_t c = 0; c < in->n_cols; c++) {
246
+ st->schema_names[c] = strdup(in->col_names[c] ? in->col_names[c] : "");
247
+ if (!st->schema_names[c]) return TF_ERROR;
248
+ st->schema_types[c] = in->col_types[c];
249
+ }
250
+ st->has_schema = 1;
251
+ return TF_OK;
252
+ }
253
+
254
+ static int normalize_resolve_columns(normalize_state *st, const tf_batch *in) {
255
+ for (size_t i = 0; i < st->n_columns; i++) {
256
+ col_stats *cs = &st->stats[i];
257
+ cs->col_idx = tf_batch_col_index(in, st->columns[i]);
258
+ if (cs->col_idx < 0) {
259
+ if (st->missing == NORM_MISSING_ERROR) return normalize_set_col_error(st, in, i);
260
+ if (st->missing == NORM_MISSING_IGNORE) cs->ignored = 1;
261
+ else cs->missing_null = 1;
262
+ continue;
263
+ }
264
+ if (!normalize_is_numeric(in->col_types[cs->col_idx])) {
265
+ if (st->on_type_error == NORM_TYPE_FAIL) return normalize_set_col_error(st, in, i);
266
+ cs->type_null = 1;
267
+ }
268
+ }
269
+ return TF_OK;
270
+ }
271
+
272
+ static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
273
+ tf_side_channels *side) {
274
+ (void)side;
275
+ normalize_state *st = self->state;
276
+ *out = NULL;
277
+
278
+ if (!st->has_schema) {
279
+ if (normalize_capture_schema(st, in) != TF_OK) return TF_ERROR;
280
+ if (normalize_resolve_columns(st, in) != TF_OK) return TF_ERROR;
281
+ }
282
+
283
+ for (size_t r = 0; r < in->n_rows; r++) {
284
+ if (add_row(st, in, r) != TF_OK) return TF_ERROR;
285
+
286
+ for (size_t i = 0; i < st->n_columns; i++) {
287
+ col_stats *cs = &st->stats[i];
288
+ int ci = cs->col_idx;
289
+ if (cs->ignored || cs->missing_null || cs->type_null || ci < 0 || tf_batch_is_null(in, r, (size_t)ci)) continue;
290
+
291
+ double val = get_numeric(in, r, ci);
292
+ cs->count++;
293
+ double delta = val - cs->mean;
294
+ cs->mean += delta / (double)cs->count;
295
+ double delta2 = val - cs->mean;
296
+ cs->m2 += delta * delta2;
297
+
298
+ if (cs->count == 1 || val < cs->min_val) cs->min_val = val;
299
+ if (cs->count == 1 || val > cs->max_val) cs->max_val = val;
300
+ }
301
+ }
302
+
303
+ return TF_OK;
304
+ }
305
+
306
+ static size_t normalize_missing_null_count(const normalize_state *st) {
307
+ size_t n = 0;
308
+ for (size_t i = 0; i < st->n_columns; i++) {
309
+ if (st->stats[i].missing_null) n++;
310
+ }
311
+ return n;
312
+ }
313
+
314
+ static int normalize_is_output_norm_col(const normalize_state *st, size_t col) {
315
+ for (size_t i = 0; i < st->n_columns; i++) {
316
+ const col_stats *cs = &st->stats[i];
317
+ if (cs->col_idx == (int)col && !cs->ignored) return 1;
318
+ }
319
+ return 0;
320
+ }
321
+
322
+ static int normalize_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
323
+ normalize_state *st = self->state;
324
+ *out = NULL;
325
+
326
+ if (st->n_rows == 0) return TF_OK;
327
+
328
+ size_t extra_cols = normalize_missing_null_count(st);
329
+ size_t out_cols = 0;
330
+ if (tf_size_add(st->schema_n_cols, extra_cols, &out_cols) != TF_OK) return TF_ERROR;
331
+ tf_batch *ob = tf_batch_create(out_cols, st->n_rows);
332
+ if (!ob) return TF_ERROR;
333
+ for (size_t c = 0; c < st->schema_n_cols; c++) {
334
+ tf_type type = st->schema_types[c];
335
+ if (normalize_is_output_norm_col(st, c)) type = TF_TYPE_FLOAT64;
336
+ if (tf_batch_set_schema(ob, c, st->schema_names[c], type) != TF_OK) {
337
+ tf_batch_free(ob);
338
+ return TF_ERROR;
339
+ }
340
+ }
341
+ size_t extra = 0;
342
+ for (size_t i = 0; i < st->n_columns; i++) {
343
+ if (!st->stats[i].missing_null) continue;
344
+ if (tf_batch_set_schema(ob, st->schema_n_cols + extra, st->columns[i], TF_TYPE_FLOAT64) != TF_OK) {
345
+ tf_batch_free(ob);
346
+ return TF_ERROR;
347
+ }
348
+ extra++;
349
+ }
350
+
351
+ for (size_t r = 0; r < st->n_rows; r++) {
352
+ tf_batch *rb = st->rows[r].batch;
353
+ for (size_t c = 0; c < st->schema_n_cols; c++) {
354
+ if (normalize_is_output_norm_col(st, c)) continue;
355
+ if (tf_batch_copy_cell(ob, r, c, rb, 0, c) != TF_OK) {
356
+ tf_batch_free(ob);
357
+ return TF_ERROR;
358
+ }
359
+ }
360
+
361
+ for (size_t i = 0; i < st->n_columns; i++) {
362
+ col_stats *cs = &st->stats[i];
363
+ int ci = cs->col_idx;
364
+ if (cs->ignored || cs->missing_null || ci < 0) continue;
365
+ if (cs->type_null) {
366
+ if (tf_batch_set_null(ob, r, (size_t)ci) != TF_OK) {
367
+ tf_batch_free(ob);
368
+ return TF_ERROR;
369
+ }
370
+ continue;
371
+ }
372
+ if (tf_batch_is_null(rb, 0, (size_t)ci)) continue;
373
+
374
+ double val = get_numeric(rb, 0, ci);
375
+ double norm = normalize_compute(st->method, cs, val);
376
+ if (tf_batch_set_float64(ob, r, (size_t)ci, norm) != TF_OK) {
377
+ tf_batch_free(ob);
378
+ return TF_ERROR;
379
+ }
380
+ }
381
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
382
+ tf_batch_free(ob);
383
+ return TF_ERROR;
384
+ }
385
+
386
+ if (st->audit) {
387
+ for (size_t i = 0; i < st->n_columns; i++) {
388
+ col_stats *cs = &st->stats[i];
389
+ int ci = cs->col_idx;
390
+ if (cs->ignored || cs->missing_null || cs->type_null || ci < 0 || tf_batch_is_null(rb, 0, (size_t)ci)) continue;
391
+
392
+ double val = get_numeric(rb, 0, ci);
393
+ double norm = normalize_compute(st->method, cs, val);
394
+ if (val == norm) continue;
395
+ if (emit_normalize_audit(st, rb, ob, 0, r, (size_t)ci, r + 1, cs, side) != TF_OK) {
396
+ tf_batch_free(ob);
397
+ return TF_ERROR;
398
+ }
399
+ }
400
+ }
401
+ }
402
+
403
+ *out = ob;
404
+ return TF_OK;
405
+ }
406
+
407
+ static void normalize_state_free(normalize_state *st) {
408
+ if (!st) return;
409
+ tf_audit_options_free(&st->audit_opts);
410
+ if (st->columns) {
411
+ for (size_t i = 0; i < st->n_columns; i++)
412
+ free(st->columns[i]);
413
+ }
414
+ free(st->columns);
415
+ free(st->stats);
416
+ if (st->rows) {
417
+ for (size_t i = 0; i < st->n_rows; i++)
418
+ tf_batch_free(st->rows[i].batch);
419
+ }
420
+ free(st->rows);
421
+ if (st->schema_names) {
422
+ for (size_t c = 0; c < st->schema_n_cols; c++)
423
+ free(st->schema_names[c]);
424
+ free(st->schema_names);
425
+ }
426
+ free(st->schema_types);
427
+ free(st);
428
+ }
429
+
430
+ static void normalize_destroy(tf_step *self) {
431
+ if (self) normalize_state_free(self->state);
432
+ free(self);
433
+ }
434
+
435
+ tf_step *tf_normalize_create(const cJSON *args) {
436
+ if (!args) return NULL;
437
+ cJSON *cols_j = cJSON_GetObjectItemCaseSensitive(args, "columns");
438
+ if (!cols_j || !cJSON_IsArray(cols_j)) {
439
+ tf_set_last_error("normalize: columns are required");
440
+ return NULL;
441
+ }
442
+
443
+ int n = cJSON_GetArraySize(cols_j);
444
+ if (n == 0) {
445
+ tf_set_last_error("normalize: columns are required");
446
+ return NULL;
447
+ }
448
+
449
+ normalize_state *st = calloc(1, sizeof(normalize_state));
450
+ if (!st) return NULL;
451
+ tf_audit_options_init(&st->audit_opts, 1);
452
+
453
+ st->audit_limit = 1000;
454
+ st->n_columns = (size_t)n;
455
+ st->columns = tf_callocarray_checked((size_t)n, sizeof(char *));
456
+ st->stats = tf_callocarray_checked((size_t)n, sizeof(col_stats));
457
+ if (!st->columns || !st->stats) { normalize_state_free(st); return NULL; }
458
+ for (int i = 0; i < n; i++) {
459
+ cJSON *item = cJSON_GetArrayItem(cols_j, i);
460
+ if (!cJSON_IsString(item) || !item->valuestring[0]) {
461
+ tf_set_last_error("normalize: columns must be non-empty strings");
462
+ normalize_state_free(st);
463
+ return NULL;
464
+ }
465
+ st->columns[i] = strdup(item->valuestring);
466
+ if (!st->columns[i]) { normalize_state_free(st); return NULL; }
467
+ st->stats[i].col_idx = -1;
468
+ }
469
+
470
+ cJSON *method_j = cJSON_GetObjectItemCaseSensitive(args, "method");
471
+ if (method_j && !cJSON_IsString(method_j)) {
472
+ tf_set_last_error("normalize: method must be minmax or zscore");
473
+ normalize_state_free(st);
474
+ return NULL;
475
+ }
476
+ if (parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL, &st->method) != TF_OK ||
477
+ normalize_parse_missing_policy(args, &st->missing) != TF_OK ||
478
+ normalize_parse_type_policy(args, &st->on_type_error) != TF_OK) {
479
+ normalize_state_free(st);
480
+ return NULL;
481
+ }
482
+
483
+ cJSON *audit_j = cJSON_GetObjectItemCaseSensitive(args, "audit");
484
+ st->audit = cJSON_IsTrue(audit_j) ? 1 : 0;
485
+ cJSON *audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "audit_limit");
486
+ if (!audit_limit_j) audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "auditLimit");
487
+ if (audit_limit_j) {
488
+ size_t parsed_limit = 0;
489
+ if (tf_json_get_size_arg_any(args, "audit_limit", "auditLimit",
490
+ 1, TF_MAX_AUDIT_RECORDS,
491
+ &parsed_limit, "normalize") < 0) {
492
+ normalize_state_free(st);
493
+ return NULL;
494
+ }
495
+ st->audit_limit = parsed_limit;
496
+ }
497
+
498
+ if (tf_audit_options_parse(&st->audit_opts, args, "normalize") != TF_OK) {
499
+ normalize_state_free(st);
500
+ return NULL;
501
+ }
502
+
503
+ tf_step *step = calloc(1, sizeof(tf_step));
504
+ if (!step) { normalize_state_free(st); return NULL; }
505
+ step->process = normalize_process;
506
+ step->flush = normalize_flush;
507
+ step->destroy = normalize_destroy;
508
+ step->state = st;
509
+ return step;
510
+ }