tranfi 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +272 -40
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +348 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +8 -11
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
@@ -17,9 +17,22 @@ typedef enum {
17
17
  NORM_ZSCORE
18
18
  } norm_method;
19
19
 
20
+ typedef enum {
21
+ NORM_MISSING_ERROR,
22
+ NORM_MISSING_NULL,
23
+ NORM_MISSING_IGNORE
24
+ } norm_missing_policy;
25
+
26
+ typedef enum {
27
+ NORM_TYPE_FAIL,
28
+ NORM_TYPE_NULL
29
+ } norm_type_policy;
30
+
20
31
  typedef struct {
21
32
  int col_idx;
22
- /* Welford's online stats */
33
+ int missing_null;
34
+ int ignored;
35
+ int type_null;
23
36
  size_t count;
24
37
  double mean;
25
38
  double m2;
@@ -27,7 +40,6 @@ typedef struct {
27
40
  double max_val;
28
41
  } col_stats;
29
42
 
30
- /* Row buffer entry */
31
43
  typedef struct {
32
44
  tf_batch *batch; /* single-row batch */
33
45
  } buf_row;
@@ -36,6 +48,8 @@ typedef struct {
36
48
  char **columns;
37
49
  size_t n_columns;
38
50
  norm_method method;
51
+ norm_missing_policy missing;
52
+ norm_type_policy on_type_error;
39
53
  col_stats *stats;
40
54
  buf_row *rows;
41
55
  size_t n_rows;
@@ -44,11 +58,100 @@ typedef struct {
44
58
  size_t schema_n_cols;
45
59
  char **schema_names;
46
60
  tf_type *schema_types;
61
+ size_t audit_limit;
62
+ size_t audit_emitted;
63
+ int audit;
64
+ tf_audit_options audit_opts;
47
65
  } normalize_state;
48
66
 
49
- static norm_method parse_method(const char *s) {
50
- if (s && strcmp(s, "zscore") == 0) return NORM_ZSCORE;
51
- return NORM_MINMAX;
67
+ static int parse_method(const char *s, norm_method *out) {
68
+ if (!s || strcmp(s, "minmax") == 0) { *out = NORM_MINMAX; return TF_OK; }
69
+ if (strcmp(s, "zscore") == 0) { *out = NORM_ZSCORE; return TF_OK; }
70
+ tf_set_last_error("normalize: method must be minmax or zscore");
71
+ return TF_ERROR;
72
+ }
73
+
74
+ static const char *method_name(norm_method method) {
75
+ return method == NORM_ZSCORE ? "zscore" : "minmax";
76
+ }
77
+
78
+ static int normalize_is_numeric(tf_type type) {
79
+ return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
80
+ }
81
+
82
+ static int normalize_parse_missing_policy(const cJSON *args, norm_missing_policy *out) {
83
+ *out = NORM_MISSING_ERROR;
84
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
85
+ if (!j) return TF_OK;
86
+ if (!cJSON_IsString(j)) {
87
+ tf_set_last_error("normalize: missing must be error, null, or ignore");
88
+ return TF_ERROR;
89
+ }
90
+ if (strcmp(j->valuestring, "error") == 0) *out = NORM_MISSING_ERROR;
91
+ else if (strcmp(j->valuestring, "null") == 0) *out = NORM_MISSING_NULL;
92
+ else if (strcmp(j->valuestring, "ignore") == 0) *out = NORM_MISSING_IGNORE;
93
+ else {
94
+ tf_set_last_error("normalize: missing must be error, null, or ignore");
95
+ return TF_ERROR;
96
+ }
97
+ return TF_OK;
98
+ }
99
+
100
+ static int normalize_parse_type_policy(const cJSON *args, norm_type_policy *out) {
101
+ *out = NORM_TYPE_FAIL;
102
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
103
+ if (!j) return TF_OK;
104
+ if (!cJSON_IsString(j)) {
105
+ tf_set_last_error("normalize: on_type_error must be fail or null");
106
+ return TF_ERROR;
107
+ }
108
+ if (strcmp(j->valuestring, "fail") == 0) *out = NORM_TYPE_FAIL;
109
+ else if (strcmp(j->valuestring, "null") == 0) *out = NORM_TYPE_NULL;
110
+ else {
111
+ tf_set_last_error("normalize: on_type_error must be fail or null");
112
+ return TF_ERROR;
113
+ }
114
+ return TF_OK;
115
+ }
116
+
117
+ static int normalize_set_col_error(normalize_state *st, const tf_batch *in, size_t i) {
118
+ if (st->stats[i].col_idx < 0) {
119
+ char msg[256];
120
+ snprintf(msg, sizeof(msg), "normalize: column '%s' not found", st->columns[i]);
121
+ tf_set_last_error(msg);
122
+ return TF_ERROR;
123
+ }
124
+ if (!normalize_is_numeric(in->col_types[st->stats[i].col_idx])) {
125
+ char msg[256];
126
+ snprintf(msg, sizeof(msg), "normalize: column '%s' must be numeric", st->columns[i]);
127
+ tf_set_last_error(msg);
128
+ return TF_ERROR;
129
+ }
130
+ return TF_OK;
131
+ }
132
+
133
+ static double normalize_compute(norm_method method, const col_stats *cs, double val) {
134
+ if (method == NORM_MINMAX) {
135
+ double range = cs->max_val - cs->min_val;
136
+ return (range > 0) ? (val - cs->min_val) / range : 0;
137
+ }
138
+ double std = (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 1;
139
+ return (std > 0) ? (val - cs->mean) / std : 0;
140
+ }
141
+
142
+ static double normalize_stddev(const col_stats *cs) {
143
+ return (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 0;
144
+ }
145
+
146
+ static int normalize_add_audit_number(cJSON *obj, const char *key,
147
+ const tf_audit_options *opts,
148
+ const char *column, double value) {
149
+ if (opts && (tf_audit_column_is_redacted(opts, column) || tf_audit_column_is_hashed(opts, column))) {
150
+ char raw[128];
151
+ if (tf_format_float64(raw, sizeof(raw), value) != TF_OK) return TF_ERROR;
152
+ return tf_json_add_audit_string(obj, key, opts, column, raw);
153
+ }
154
+ return tf_json_add_number(obj, key, value);
52
155
  }
53
156
 
54
157
  static double get_numeric(const tf_batch *b, size_t r, int ci) {
@@ -57,23 +160,113 @@ static double get_numeric(const tf_batch *b, size_t r, int ci) {
57
160
  return 0;
58
161
  }
59
162
 
60
- static void add_row(normalize_state *st, tf_batch *b, size_t r) {
163
+ static int emit_normalize_audit(normalize_state *st, const tf_batch *before_b,
164
+ const tf_batch *after_b, size_t before_row,
165
+ size_t after_row, size_t col, size_t row_no,
166
+ const col_stats *cs, tf_side_channels *side) {
167
+ if (!st->audit || st->audit_emitted >= st->audit_limit || !side || !side->stats) return TF_OK;
168
+ cJSON *obj = cJSON_CreateObject();
169
+ if (!obj) return TF_ERROR;
170
+ int rc = TF_ERROR;
171
+ if (tf_json_add_string(obj, "type", "audit") != TF_OK ||
172
+ tf_json_add_string(obj, "op", "normalize") != TF_OK ||
173
+ tf_json_add_string(obj, "event", "value_changed") != TF_OK ||
174
+ tf_json_add_string(obj, "reason",
175
+ st->method == NORM_ZSCORE ? "normalize_zscore" : "normalize_minmax") != TF_OK ||
176
+ tf_json_add_string(obj, "channel", "audit") != TF_OK) {
177
+ goto done;
178
+ }
179
+ const char *column_name = after_b->col_names[col] ? after_b->col_names[col] : "";
180
+ if (tf_json_add_string(obj, "column", column_name) != TF_OK ||
181
+ tf_json_add_string(obj, "method", method_name(st->method)) != TF_OK ||
182
+ tf_json_add_number(obj, "row", (double)row_no) != TF_OK) {
183
+ goto done;
184
+ }
185
+ cJSON *before = tf_audit_cell_to_json(before_b, before_row, col, &st->audit_opts);
186
+ if (!before || tf_json_add_item(obj, "before", before) != TF_OK) goto done;
187
+ cJSON *after = tf_audit_cell_to_json(after_b, after_row, col, &st->audit_opts);
188
+ if (!after || tf_json_add_item(obj, "after", after) != TF_OK) goto done;
189
+ if (tf_json_add_number(obj, "count", (double)cs->count) != TF_OK) goto done;
190
+ if (normalize_add_audit_number(obj, "min", &st->audit_opts, column_name, cs->min_val) != TF_OK ||
191
+ normalize_add_audit_number(obj, "max", &st->audit_opts, column_name, cs->max_val) != TF_OK ||
192
+ normalize_add_audit_number(obj, "mean", &st->audit_opts, column_name, cs->mean) != TF_OK ||
193
+ normalize_add_audit_number(obj, "stddev", &st->audit_opts, column_name, normalize_stddev(cs)) != TF_OK) {
194
+ goto done;
195
+ }
196
+ cJSON *row_obj = tf_audit_row_to_json(after_b, after_row, &st->audit_opts);
197
+ if (row_obj) {
198
+ if (tf_json_add_item(obj, "data", row_obj) != TF_OK) goto done;
199
+ } else if (st->audit_opts.include_row) {
200
+ goto done;
201
+ }
202
+ rc = tf_buffer_write_json_line(side->stats, obj);
203
+ done:
204
+ cJSON_Delete(obj);
205
+ if (rc == TF_OK) st->audit_emitted++;
206
+ return rc;
207
+ }
208
+
209
+ static int add_row(normalize_state *st, tf_batch *b, size_t r) {
61
210
  if (st->n_rows >= st->cap_rows) {
62
- size_t newcap = st->cap_rows ? st->cap_rows * 2 : 256;
63
- buf_row *tmp = realloc(st->rows, newcap * sizeof(buf_row));
64
- if (!tmp) return;
211
+ size_t min_cap = 0, newcap = 0;
212
+ if (tf_size_add(st->n_rows, 1, &min_cap) != TF_OK ||
213
+ tf_size_grow_pow2(st->cap_rows, min_cap, 256, &newcap) != TF_OK) {
214
+ return TF_ERROR;
215
+ }
216
+ buf_row *tmp = tf_reallocarray_checked(st->rows, newcap, sizeof(buf_row));
217
+ if (!tmp) return TF_ERROR;
65
218
  st->rows = tmp;
66
219
  st->cap_rows = newcap;
67
220
  }
68
- /* Copy single row into its own batch */
69
221
  tf_batch *rb = tf_batch_create(b->n_cols, 1);
70
- if (!rb) return;
71
- for (size_t c = 0; c < b->n_cols; c++)
72
- tf_batch_set_schema(rb, c, b->col_names[c], b->col_types[c]);
73
- tf_batch_copy_row(rb, 0, b, r);
74
- rb->n_rows = 1;
222
+ if (!rb) return TF_ERROR;
223
+ if (tf_batch_clone_schema(rb, b) != TF_OK) {
224
+ tf_batch_free(rb);
225
+ return TF_ERROR;
226
+ }
227
+ if (tf_batch_copy_row(rb, 0, b, r) != TF_OK) {
228
+ tf_batch_free(rb);
229
+ return TF_ERROR;
230
+ }
231
+ if (tf_batch_expose_row(rb, 0) != TF_OK) {
232
+ tf_batch_free(rb);
233
+ return TF_ERROR;
234
+ }
75
235
  st->rows[st->n_rows].batch = rb;
76
236
  st->n_rows++;
237
+ return TF_OK;
238
+ }
239
+
240
+ static int normalize_capture_schema(normalize_state *st, const tf_batch *in) {
241
+ st->schema_n_cols = in->n_cols;
242
+ st->schema_names = tf_callocarray_checked(in->n_cols, sizeof(char *));
243
+ st->schema_types = tf_callocarray_checked(in->n_cols, sizeof(tf_type));
244
+ if (!st->schema_names || !st->schema_types) return TF_ERROR;
245
+ for (size_t c = 0; c < in->n_cols; c++) {
246
+ st->schema_names[c] = strdup(in->col_names[c] ? in->col_names[c] : "");
247
+ if (!st->schema_names[c]) return TF_ERROR;
248
+ st->schema_types[c] = in->col_types[c];
249
+ }
250
+ st->has_schema = 1;
251
+ return TF_OK;
252
+ }
253
+
254
+ static int normalize_resolve_columns(normalize_state *st, const tf_batch *in) {
255
+ for (size_t i = 0; i < st->n_columns; i++) {
256
+ col_stats *cs = &st->stats[i];
257
+ cs->col_idx = tf_batch_col_index(in, st->columns[i]);
258
+ if (cs->col_idx < 0) {
259
+ if (st->missing == NORM_MISSING_ERROR) return normalize_set_col_error(st, in, i);
260
+ if (st->missing == NORM_MISSING_IGNORE) cs->ignored = 1;
261
+ else cs->missing_null = 1;
262
+ continue;
263
+ }
264
+ if (!normalize_is_numeric(in->col_types[cs->col_idx])) {
265
+ if (st->on_type_error == NORM_TYPE_FAIL) return normalize_set_col_error(st, in, i);
266
+ cs->type_null = 1;
267
+ }
268
+ }
269
+ return TF_OK;
77
270
  }
78
271
 
79
272
  static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
@@ -82,33 +275,20 @@ static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
82
275
  normalize_state *st = self->state;
83
276
  *out = NULL;
84
277
 
85
- /* Save schema from first batch */
86
278
  if (!st->has_schema) {
87
- st->schema_n_cols = in->n_cols;
88
- st->schema_names = malloc(in->n_cols * sizeof(char *));
89
- st->schema_types = malloc(in->n_cols * sizeof(tf_type));
90
- for (size_t c = 0; c < in->n_cols; c++) {
91
- st->schema_names[c] = strdup(in->col_names[c]);
92
- st->schema_types[c] = in->col_types[c];
93
- }
94
- st->has_schema = 1;
95
-
96
- /* Resolve column indices */
97
- for (size_t i = 0; i < st->n_columns; i++) {
98
- st->stats[i].col_idx = tf_batch_col_index(in, st->columns[i]);
99
- }
279
+ if (normalize_capture_schema(st, in) != TF_OK) return TF_ERROR;
280
+ if (normalize_resolve_columns(st, in) != TF_OK) return TF_ERROR;
100
281
  }
101
282
 
102
- /* Buffer rows and update stats */
103
283
  for (size_t r = 0; r < in->n_rows; r++) {
104
- add_row(st, in, r);
284
+ if (add_row(st, in, r) != TF_OK) return TF_ERROR;
105
285
 
106
286
  for (size_t i = 0; i < st->n_columns; i++) {
107
- int ci = st->stats[i].col_idx;
108
- if (ci < 0 || tf_batch_is_null(in, r, ci)) continue;
287
+ col_stats *cs = &st->stats[i];
288
+ int ci = cs->col_idx;
289
+ if (cs->ignored || cs->missing_null || cs->type_null || ci < 0 || tf_batch_is_null(in, r, (size_t)ci)) continue;
109
290
 
110
291
  double val = get_numeric(in, r, ci);
111
- col_stats *cs = &st->stats[i];
112
292
  cs->count++;
113
293
  double delta = val - cs->mean;
114
294
  cs->mean += delta / (double)cs->count;
@@ -123,101 +303,205 @@ static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
123
303
  return TF_OK;
124
304
  }
125
305
 
306
+ static size_t normalize_missing_null_count(const normalize_state *st) {
307
+ size_t n = 0;
308
+ for (size_t i = 0; i < st->n_columns; i++) {
309
+ if (st->stats[i].missing_null) n++;
310
+ }
311
+ return n;
312
+ }
313
+
314
+ static int normalize_is_output_norm_col(const normalize_state *st, size_t col) {
315
+ for (size_t i = 0; i < st->n_columns; i++) {
316
+ const col_stats *cs = &st->stats[i];
317
+ if (cs->col_idx == (int)col && !cs->ignored) return 1;
318
+ }
319
+ return 0;
320
+ }
321
+
126
322
  static int normalize_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
127
- (void)side;
128
323
  normalize_state *st = self->state;
129
324
  *out = NULL;
130
325
 
131
326
  if (st->n_rows == 0) return TF_OK;
132
327
 
133
- tf_batch *ob = tf_batch_create(st->schema_n_cols, st->n_rows);
328
+ size_t extra_cols = normalize_missing_null_count(st);
329
+ size_t out_cols = 0;
330
+ if (tf_size_add(st->schema_n_cols, extra_cols, &out_cols) != TF_OK) return TF_ERROR;
331
+ tf_batch *ob = tf_batch_create(out_cols, st->n_rows);
134
332
  if (!ob) return TF_ERROR;
135
333
  for (size_t c = 0; c < st->schema_n_cols; c++) {
136
- /* Normalized columns become FLOAT64 */
137
334
  tf_type type = st->schema_types[c];
138
- int is_norm_col = 0;
139
- for (size_t i = 0; i < st->n_columns; i++) {
140
- if (st->stats[i].col_idx == (int)c) { is_norm_col = 1; break; }
335
+ if (normalize_is_output_norm_col(st, c)) type = TF_TYPE_FLOAT64;
336
+ if (tf_batch_set_schema(ob, c, st->schema_names[c], type) != TF_OK) {
337
+ tf_batch_free(ob);
338
+ return TF_ERROR;
141
339
  }
142
- tf_batch_set_schema(ob, c, st->schema_names[c],
143
- is_norm_col ? TF_TYPE_FLOAT64 : type);
340
+ }
341
+ size_t extra = 0;
342
+ for (size_t i = 0; i < st->n_columns; i++) {
343
+ if (!st->stats[i].missing_null) continue;
344
+ if (tf_batch_set_schema(ob, st->schema_n_cols + extra, st->columns[i], TF_TYPE_FLOAT64) != TF_OK) {
345
+ tf_batch_free(ob);
346
+ return TF_ERROR;
347
+ }
348
+ extra++;
144
349
  }
145
350
 
146
351
  for (size_t r = 0; r < st->n_rows; r++) {
147
352
  tf_batch *rb = st->rows[r].batch;
148
- tf_batch_copy_row(ob, r, rb, 0);
353
+ for (size_t c = 0; c < st->schema_n_cols; c++) {
354
+ if (normalize_is_output_norm_col(st, c)) continue;
355
+ if (tf_batch_copy_cell(ob, r, c, rb, 0, c) != TF_OK) {
356
+ tf_batch_free(ob);
357
+ return TF_ERROR;
358
+ }
359
+ }
149
360
 
150
- /* Normalize target columns */
151
361
  for (size_t i = 0; i < st->n_columns; i++) {
152
- int ci = st->stats[i].col_idx;
153
- if (ci < 0 || tf_batch_is_null(rb, 0, ci)) continue;
362
+ col_stats *cs = &st->stats[i];
363
+ int ci = cs->col_idx;
364
+ if (cs->ignored || cs->missing_null || ci < 0) continue;
365
+ if (cs->type_null) {
366
+ if (tf_batch_set_null(ob, r, (size_t)ci) != TF_OK) {
367
+ tf_batch_free(ob);
368
+ return TF_ERROR;
369
+ }
370
+ continue;
371
+ }
372
+ if (tf_batch_is_null(rb, 0, (size_t)ci)) continue;
154
373
 
155
374
  double val = get_numeric(rb, 0, ci);
156
- col_stats *cs = &st->stats[i];
157
- double norm;
158
-
159
- if (st->method == NORM_MINMAX) {
160
- double range = cs->max_val - cs->min_val;
161
- norm = (range > 0) ? (val - cs->min_val) / range : 0;
162
- } else {
163
- double std = (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 1;
164
- norm = (std > 0) ? (val - cs->mean) / std : 0;
375
+ double norm = normalize_compute(st->method, cs, val);
376
+ if (tf_batch_set_float64(ob, r, (size_t)ci, norm) != TF_OK) {
377
+ tf_batch_free(ob);
378
+ return TF_ERROR;
379
+ }
380
+ }
381
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
382
+ tf_batch_free(ob);
383
+ return TF_ERROR;
384
+ }
385
+
386
+ if (st->audit) {
387
+ for (size_t i = 0; i < st->n_columns; i++) {
388
+ col_stats *cs = &st->stats[i];
389
+ int ci = cs->col_idx;
390
+ if (cs->ignored || cs->missing_null || cs->type_null || ci < 0 || tf_batch_is_null(rb, 0, (size_t)ci)) continue;
391
+
392
+ double val = get_numeric(rb, 0, ci);
393
+ double norm = normalize_compute(st->method, cs, val);
394
+ if (val == norm) continue;
395
+ if (emit_normalize_audit(st, rb, ob, 0, r, (size_t)ci, r + 1, cs, side) != TF_OK) {
396
+ tf_batch_free(ob);
397
+ return TF_ERROR;
398
+ }
165
399
  }
166
- tf_batch_set_float64(ob, r, ci, norm);
167
400
  }
168
- ob->n_rows = r + 1;
169
401
  }
170
402
 
171
403
  *out = ob;
172
404
  return TF_OK;
173
405
  }
174
406
 
175
- static void normalize_destroy(tf_step *self) {
176
- normalize_state *st = self->state;
177
- if (st) {
407
+ static void normalize_state_free(normalize_state *st) {
408
+ if (!st) return;
409
+ tf_audit_options_free(&st->audit_opts);
410
+ if (st->columns) {
178
411
  for (size_t i = 0; i < st->n_columns; i++)
179
412
  free(st->columns[i]);
180
- free(st->columns);
181
- free(st->stats);
413
+ }
414
+ free(st->columns);
415
+ free(st->stats);
416
+ if (st->rows) {
182
417
  for (size_t i = 0; i < st->n_rows; i++)
183
418
  tf_batch_free(st->rows[i].batch);
184
- free(st->rows);
185
- if (st->schema_names) {
186
- for (size_t c = 0; c < st->schema_n_cols; c++)
187
- free(st->schema_names[c]);
188
- free(st->schema_names);
189
- }
190
- free(st->schema_types);
191
- free(st);
192
419
  }
420
+ free(st->rows);
421
+ if (st->schema_names) {
422
+ for (size_t c = 0; c < st->schema_n_cols; c++)
423
+ free(st->schema_names[c]);
424
+ free(st->schema_names);
425
+ }
426
+ free(st->schema_types);
427
+ free(st);
428
+ }
429
+
430
+ static void normalize_destroy(tf_step *self) {
431
+ if (self) normalize_state_free(self->state);
193
432
  free(self);
194
433
  }
195
434
 
196
435
  tf_step *tf_normalize_create(const cJSON *args) {
197
436
  if (!args) return NULL;
198
437
  cJSON *cols_j = cJSON_GetObjectItemCaseSensitive(args, "columns");
199
- if (!cols_j || !cJSON_IsArray(cols_j)) return NULL;
438
+ if (!cols_j || !cJSON_IsArray(cols_j)) {
439
+ tf_set_last_error("normalize: columns are required");
440
+ return NULL;
441
+ }
200
442
 
201
443
  int n = cJSON_GetArraySize(cols_j);
202
- if (n == 0) return NULL;
444
+ if (n == 0) {
445
+ tf_set_last_error("normalize: columns are required");
446
+ return NULL;
447
+ }
203
448
 
204
449
  normalize_state *st = calloc(1, sizeof(normalize_state));
205
450
  if (!st) return NULL;
451
+ tf_audit_options_init(&st->audit_opts, 1);
206
452
 
207
- st->n_columns = n;
208
- st->columns = malloc(n * sizeof(char *));
209
- st->stats = calloc(n, sizeof(col_stats));
453
+ st->audit_limit = 1000;
454
+ st->n_columns = (size_t)n;
455
+ st->columns = tf_callocarray_checked((size_t)n, sizeof(char *));
456
+ st->stats = tf_callocarray_checked((size_t)n, sizeof(col_stats));
457
+ if (!st->columns || !st->stats) { normalize_state_free(st); return NULL; }
210
458
  for (int i = 0; i < n; i++) {
211
459
  cJSON *item = cJSON_GetArrayItem(cols_j, i);
212
- st->columns[i] = strdup(cJSON_IsString(item) ? item->valuestring : "");
460
+ if (!cJSON_IsString(item) || !item->valuestring[0]) {
461
+ tf_set_last_error("normalize: columns must be non-empty strings");
462
+ normalize_state_free(st);
463
+ return NULL;
464
+ }
465
+ st->columns[i] = strdup(item->valuestring);
466
+ if (!st->columns[i]) { normalize_state_free(st); return NULL; }
213
467
  st->stats[i].col_idx = -1;
214
468
  }
215
469
 
216
470
  cJSON *method_j = cJSON_GetObjectItemCaseSensitive(args, "method");
217
- st->method = parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL);
471
+ if (method_j && !cJSON_IsString(method_j)) {
472
+ tf_set_last_error("normalize: method must be minmax or zscore");
473
+ normalize_state_free(st);
474
+ return NULL;
475
+ }
476
+ if (parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL, &st->method) != TF_OK ||
477
+ normalize_parse_missing_policy(args, &st->missing) != TF_OK ||
478
+ normalize_parse_type_policy(args, &st->on_type_error) != TF_OK) {
479
+ normalize_state_free(st);
480
+ return NULL;
481
+ }
482
+
483
+ cJSON *audit_j = cJSON_GetObjectItemCaseSensitive(args, "audit");
484
+ st->audit = cJSON_IsTrue(audit_j) ? 1 : 0;
485
+ cJSON *audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "audit_limit");
486
+ if (!audit_limit_j) audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "auditLimit");
487
+ if (audit_limit_j) {
488
+ size_t parsed_limit = 0;
489
+ if (tf_json_get_size_arg_any(args, "audit_limit", "auditLimit",
490
+ 1, TF_MAX_AUDIT_RECORDS,
491
+ &parsed_limit, "normalize") < 0) {
492
+ normalize_state_free(st);
493
+ return NULL;
494
+ }
495
+ st->audit_limit = parsed_limit;
496
+ }
497
+
498
+ if (tf_audit_options_parse(&st->audit_opts, args, "normalize") != TF_OK) {
499
+ normalize_state_free(st);
500
+ return NULL;
501
+ }
218
502
 
219
- tf_step *step = malloc(sizeof(tf_step));
220
- if (!step) { normalize_destroy(&(tf_step){.state = st}); return NULL; }
503
+ tf_step *step = calloc(1, sizeof(tf_step));
504
+ if (!step) { normalize_state_free(st); return NULL; }
221
505
  step->process = normalize_process;
222
506
  step->flush = normalize_flush;
223
507
  step->destroy = normalize_destroy;