tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/csrc/op_normalize.c
CHANGED
|
@@ -17,9 +17,22 @@ typedef enum {
|
|
|
17
17
|
NORM_ZSCORE
|
|
18
18
|
} norm_method;
|
|
19
19
|
|
|
20
|
+
typedef enum {
|
|
21
|
+
NORM_MISSING_ERROR,
|
|
22
|
+
NORM_MISSING_NULL,
|
|
23
|
+
NORM_MISSING_IGNORE
|
|
24
|
+
} norm_missing_policy;
|
|
25
|
+
|
|
26
|
+
typedef enum {
|
|
27
|
+
NORM_TYPE_FAIL,
|
|
28
|
+
NORM_TYPE_NULL
|
|
29
|
+
} norm_type_policy;
|
|
30
|
+
|
|
20
31
|
typedef struct {
|
|
21
32
|
int col_idx;
|
|
22
|
-
|
|
33
|
+
int missing_null;
|
|
34
|
+
int ignored;
|
|
35
|
+
int type_null;
|
|
23
36
|
size_t count;
|
|
24
37
|
double mean;
|
|
25
38
|
double m2;
|
|
@@ -27,7 +40,6 @@ typedef struct {
|
|
|
27
40
|
double max_val;
|
|
28
41
|
} col_stats;
|
|
29
42
|
|
|
30
|
-
/* Row buffer entry */
|
|
31
43
|
typedef struct {
|
|
32
44
|
tf_batch *batch; /* single-row batch */
|
|
33
45
|
} buf_row;
|
|
@@ -36,6 +48,8 @@ typedef struct {
|
|
|
36
48
|
char **columns;
|
|
37
49
|
size_t n_columns;
|
|
38
50
|
norm_method method;
|
|
51
|
+
norm_missing_policy missing;
|
|
52
|
+
norm_type_policy on_type_error;
|
|
39
53
|
col_stats *stats;
|
|
40
54
|
buf_row *rows;
|
|
41
55
|
size_t n_rows;
|
|
@@ -44,11 +58,100 @@ typedef struct {
|
|
|
44
58
|
size_t schema_n_cols;
|
|
45
59
|
char **schema_names;
|
|
46
60
|
tf_type *schema_types;
|
|
61
|
+
size_t audit_limit;
|
|
62
|
+
size_t audit_emitted;
|
|
63
|
+
int audit;
|
|
64
|
+
tf_audit_options audit_opts;
|
|
47
65
|
} normalize_state;
|
|
48
66
|
|
|
49
|
-
static
|
|
50
|
-
if (s
|
|
51
|
-
return
|
|
67
|
+
static int parse_method(const char *s, norm_method *out) {
|
|
68
|
+
if (!s || strcmp(s, "minmax") == 0) { *out = NORM_MINMAX; return TF_OK; }
|
|
69
|
+
if (strcmp(s, "zscore") == 0) { *out = NORM_ZSCORE; return TF_OK; }
|
|
70
|
+
tf_set_last_error("normalize: method must be minmax or zscore");
|
|
71
|
+
return TF_ERROR;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
static const char *method_name(norm_method method) {
|
|
75
|
+
return method == NORM_ZSCORE ? "zscore" : "minmax";
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
static int normalize_is_numeric(tf_type type) {
|
|
79
|
+
return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
static int normalize_parse_missing_policy(const cJSON *args, norm_missing_policy *out) {
|
|
83
|
+
*out = NORM_MISSING_ERROR;
|
|
84
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
|
|
85
|
+
if (!j) return TF_OK;
|
|
86
|
+
if (!cJSON_IsString(j)) {
|
|
87
|
+
tf_set_last_error("normalize: missing must be error, null, or ignore");
|
|
88
|
+
return TF_ERROR;
|
|
89
|
+
}
|
|
90
|
+
if (strcmp(j->valuestring, "error") == 0) *out = NORM_MISSING_ERROR;
|
|
91
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = NORM_MISSING_NULL;
|
|
92
|
+
else if (strcmp(j->valuestring, "ignore") == 0) *out = NORM_MISSING_IGNORE;
|
|
93
|
+
else {
|
|
94
|
+
tf_set_last_error("normalize: missing must be error, null, or ignore");
|
|
95
|
+
return TF_ERROR;
|
|
96
|
+
}
|
|
97
|
+
return TF_OK;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
static int normalize_parse_type_policy(const cJSON *args, norm_type_policy *out) {
|
|
101
|
+
*out = NORM_TYPE_FAIL;
|
|
102
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
|
|
103
|
+
if (!j) return TF_OK;
|
|
104
|
+
if (!cJSON_IsString(j)) {
|
|
105
|
+
tf_set_last_error("normalize: on_type_error must be fail or null");
|
|
106
|
+
return TF_ERROR;
|
|
107
|
+
}
|
|
108
|
+
if (strcmp(j->valuestring, "fail") == 0) *out = NORM_TYPE_FAIL;
|
|
109
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = NORM_TYPE_NULL;
|
|
110
|
+
else {
|
|
111
|
+
tf_set_last_error("normalize: on_type_error must be fail or null");
|
|
112
|
+
return TF_ERROR;
|
|
113
|
+
}
|
|
114
|
+
return TF_OK;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
static int normalize_set_col_error(normalize_state *st, const tf_batch *in, size_t i) {
|
|
118
|
+
if (st->stats[i].col_idx < 0) {
|
|
119
|
+
char msg[256];
|
|
120
|
+
snprintf(msg, sizeof(msg), "normalize: column '%s' not found", st->columns[i]);
|
|
121
|
+
tf_set_last_error(msg);
|
|
122
|
+
return TF_ERROR;
|
|
123
|
+
}
|
|
124
|
+
if (!normalize_is_numeric(in->col_types[st->stats[i].col_idx])) {
|
|
125
|
+
char msg[256];
|
|
126
|
+
snprintf(msg, sizeof(msg), "normalize: column '%s' must be numeric", st->columns[i]);
|
|
127
|
+
tf_set_last_error(msg);
|
|
128
|
+
return TF_ERROR;
|
|
129
|
+
}
|
|
130
|
+
return TF_OK;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
static double normalize_compute(norm_method method, const col_stats *cs, double val) {
|
|
134
|
+
if (method == NORM_MINMAX) {
|
|
135
|
+
double range = cs->max_val - cs->min_val;
|
|
136
|
+
return (range > 0) ? (val - cs->min_val) / range : 0;
|
|
137
|
+
}
|
|
138
|
+
double std = (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 1;
|
|
139
|
+
return (std > 0) ? (val - cs->mean) / std : 0;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
static double normalize_stddev(const col_stats *cs) {
|
|
143
|
+
return (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 0;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
static int normalize_add_audit_number(cJSON *obj, const char *key,
|
|
147
|
+
const tf_audit_options *opts,
|
|
148
|
+
const char *column, double value) {
|
|
149
|
+
if (opts && (tf_audit_column_is_redacted(opts, column) || tf_audit_column_is_hashed(opts, column))) {
|
|
150
|
+
char raw[128];
|
|
151
|
+
if (tf_format_float64(raw, sizeof(raw), value) != TF_OK) return TF_ERROR;
|
|
152
|
+
return tf_json_add_audit_string(obj, key, opts, column, raw);
|
|
153
|
+
}
|
|
154
|
+
return tf_json_add_number(obj, key, value);
|
|
52
155
|
}
|
|
53
156
|
|
|
54
157
|
static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
@@ -57,23 +160,113 @@ static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
|
57
160
|
return 0;
|
|
58
161
|
}
|
|
59
162
|
|
|
60
|
-
static
|
|
163
|
+
static int emit_normalize_audit(normalize_state *st, const tf_batch *before_b,
|
|
164
|
+
const tf_batch *after_b, size_t before_row,
|
|
165
|
+
size_t after_row, size_t col, size_t row_no,
|
|
166
|
+
const col_stats *cs, tf_side_channels *side) {
|
|
167
|
+
if (!st->audit || st->audit_emitted >= st->audit_limit || !side || !side->stats) return TF_OK;
|
|
168
|
+
cJSON *obj = cJSON_CreateObject();
|
|
169
|
+
if (!obj) return TF_ERROR;
|
|
170
|
+
int rc = TF_ERROR;
|
|
171
|
+
if (tf_json_add_string(obj, "type", "audit") != TF_OK ||
|
|
172
|
+
tf_json_add_string(obj, "op", "normalize") != TF_OK ||
|
|
173
|
+
tf_json_add_string(obj, "event", "value_changed") != TF_OK ||
|
|
174
|
+
tf_json_add_string(obj, "reason",
|
|
175
|
+
st->method == NORM_ZSCORE ? "normalize_zscore" : "normalize_minmax") != TF_OK ||
|
|
176
|
+
tf_json_add_string(obj, "channel", "audit") != TF_OK) {
|
|
177
|
+
goto done;
|
|
178
|
+
}
|
|
179
|
+
const char *column_name = after_b->col_names[col] ? after_b->col_names[col] : "";
|
|
180
|
+
if (tf_json_add_string(obj, "column", column_name) != TF_OK ||
|
|
181
|
+
tf_json_add_string(obj, "method", method_name(st->method)) != TF_OK ||
|
|
182
|
+
tf_json_add_number(obj, "row", (double)row_no) != TF_OK) {
|
|
183
|
+
goto done;
|
|
184
|
+
}
|
|
185
|
+
cJSON *before = tf_audit_cell_to_json(before_b, before_row, col, &st->audit_opts);
|
|
186
|
+
if (!before || tf_json_add_item(obj, "before", before) != TF_OK) goto done;
|
|
187
|
+
cJSON *after = tf_audit_cell_to_json(after_b, after_row, col, &st->audit_opts);
|
|
188
|
+
if (!after || tf_json_add_item(obj, "after", after) != TF_OK) goto done;
|
|
189
|
+
if (tf_json_add_number(obj, "count", (double)cs->count) != TF_OK) goto done;
|
|
190
|
+
if (normalize_add_audit_number(obj, "min", &st->audit_opts, column_name, cs->min_val) != TF_OK ||
|
|
191
|
+
normalize_add_audit_number(obj, "max", &st->audit_opts, column_name, cs->max_val) != TF_OK ||
|
|
192
|
+
normalize_add_audit_number(obj, "mean", &st->audit_opts, column_name, cs->mean) != TF_OK ||
|
|
193
|
+
normalize_add_audit_number(obj, "stddev", &st->audit_opts, column_name, normalize_stddev(cs)) != TF_OK) {
|
|
194
|
+
goto done;
|
|
195
|
+
}
|
|
196
|
+
cJSON *row_obj = tf_audit_row_to_json(after_b, after_row, &st->audit_opts);
|
|
197
|
+
if (row_obj) {
|
|
198
|
+
if (tf_json_add_item(obj, "data", row_obj) != TF_OK) goto done;
|
|
199
|
+
} else if (st->audit_opts.include_row) {
|
|
200
|
+
goto done;
|
|
201
|
+
}
|
|
202
|
+
rc = tf_buffer_write_json_line(side->stats, obj);
|
|
203
|
+
done:
|
|
204
|
+
cJSON_Delete(obj);
|
|
205
|
+
if (rc == TF_OK) st->audit_emitted++;
|
|
206
|
+
return rc;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
static int add_row(normalize_state *st, tf_batch *b, size_t r) {
|
|
61
210
|
if (st->n_rows >= st->cap_rows) {
|
|
62
|
-
size_t
|
|
63
|
-
|
|
64
|
-
|
|
211
|
+
size_t min_cap = 0, newcap = 0;
|
|
212
|
+
if (tf_size_add(st->n_rows, 1, &min_cap) != TF_OK ||
|
|
213
|
+
tf_size_grow_pow2(st->cap_rows, min_cap, 256, &newcap) != TF_OK) {
|
|
214
|
+
return TF_ERROR;
|
|
215
|
+
}
|
|
216
|
+
buf_row *tmp = tf_reallocarray_checked(st->rows, newcap, sizeof(buf_row));
|
|
217
|
+
if (!tmp) return TF_ERROR;
|
|
65
218
|
st->rows = tmp;
|
|
66
219
|
st->cap_rows = newcap;
|
|
67
220
|
}
|
|
68
|
-
/* Copy single row into its own batch */
|
|
69
221
|
tf_batch *rb = tf_batch_create(b->n_cols, 1);
|
|
70
|
-
if (!rb) return;
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
222
|
+
if (!rb) return TF_ERROR;
|
|
223
|
+
if (tf_batch_clone_schema(rb, b) != TF_OK) {
|
|
224
|
+
tf_batch_free(rb);
|
|
225
|
+
return TF_ERROR;
|
|
226
|
+
}
|
|
227
|
+
if (tf_batch_copy_row(rb, 0, b, r) != TF_OK) {
|
|
228
|
+
tf_batch_free(rb);
|
|
229
|
+
return TF_ERROR;
|
|
230
|
+
}
|
|
231
|
+
if (tf_batch_expose_row(rb, 0) != TF_OK) {
|
|
232
|
+
tf_batch_free(rb);
|
|
233
|
+
return TF_ERROR;
|
|
234
|
+
}
|
|
75
235
|
st->rows[st->n_rows].batch = rb;
|
|
76
236
|
st->n_rows++;
|
|
237
|
+
return TF_OK;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
static int normalize_capture_schema(normalize_state *st, const tf_batch *in) {
|
|
241
|
+
st->schema_n_cols = in->n_cols;
|
|
242
|
+
st->schema_names = tf_callocarray_checked(in->n_cols, sizeof(char *));
|
|
243
|
+
st->schema_types = tf_callocarray_checked(in->n_cols, sizeof(tf_type));
|
|
244
|
+
if (!st->schema_names || !st->schema_types) return TF_ERROR;
|
|
245
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
246
|
+
st->schema_names[c] = strdup(in->col_names[c] ? in->col_names[c] : "");
|
|
247
|
+
if (!st->schema_names[c]) return TF_ERROR;
|
|
248
|
+
st->schema_types[c] = in->col_types[c];
|
|
249
|
+
}
|
|
250
|
+
st->has_schema = 1;
|
|
251
|
+
return TF_OK;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
static int normalize_resolve_columns(normalize_state *st, const tf_batch *in) {
|
|
255
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
256
|
+
col_stats *cs = &st->stats[i];
|
|
257
|
+
cs->col_idx = tf_batch_col_index(in, st->columns[i]);
|
|
258
|
+
if (cs->col_idx < 0) {
|
|
259
|
+
if (st->missing == NORM_MISSING_ERROR) return normalize_set_col_error(st, in, i);
|
|
260
|
+
if (st->missing == NORM_MISSING_IGNORE) cs->ignored = 1;
|
|
261
|
+
else cs->missing_null = 1;
|
|
262
|
+
continue;
|
|
263
|
+
}
|
|
264
|
+
if (!normalize_is_numeric(in->col_types[cs->col_idx])) {
|
|
265
|
+
if (st->on_type_error == NORM_TYPE_FAIL) return normalize_set_col_error(st, in, i);
|
|
266
|
+
cs->type_null = 1;
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
return TF_OK;
|
|
77
270
|
}
|
|
78
271
|
|
|
79
272
|
static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
@@ -82,33 +275,20 @@ static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
|
82
275
|
normalize_state *st = self->state;
|
|
83
276
|
*out = NULL;
|
|
84
277
|
|
|
85
|
-
/* Save schema from first batch */
|
|
86
278
|
if (!st->has_schema) {
|
|
87
|
-
st
|
|
88
|
-
st
|
|
89
|
-
st->schema_types = malloc(in->n_cols * sizeof(tf_type));
|
|
90
|
-
for (size_t c = 0; c < in->n_cols; c++) {
|
|
91
|
-
st->schema_names[c] = strdup(in->col_names[c]);
|
|
92
|
-
st->schema_types[c] = in->col_types[c];
|
|
93
|
-
}
|
|
94
|
-
st->has_schema = 1;
|
|
95
|
-
|
|
96
|
-
/* Resolve column indices */
|
|
97
|
-
for (size_t i = 0; i < st->n_columns; i++) {
|
|
98
|
-
st->stats[i].col_idx = tf_batch_col_index(in, st->columns[i]);
|
|
99
|
-
}
|
|
279
|
+
if (normalize_capture_schema(st, in) != TF_OK) return TF_ERROR;
|
|
280
|
+
if (normalize_resolve_columns(st, in) != TF_OK) return TF_ERROR;
|
|
100
281
|
}
|
|
101
282
|
|
|
102
|
-
/* Buffer rows and update stats */
|
|
103
283
|
for (size_t r = 0; r < in->n_rows; r++) {
|
|
104
|
-
add_row(st, in, r);
|
|
284
|
+
if (add_row(st, in, r) != TF_OK) return TF_ERROR;
|
|
105
285
|
|
|
106
286
|
for (size_t i = 0; i < st->n_columns; i++) {
|
|
107
|
-
|
|
108
|
-
|
|
287
|
+
col_stats *cs = &st->stats[i];
|
|
288
|
+
int ci = cs->col_idx;
|
|
289
|
+
if (cs->ignored || cs->missing_null || cs->type_null || ci < 0 || tf_batch_is_null(in, r, (size_t)ci)) continue;
|
|
109
290
|
|
|
110
291
|
double val = get_numeric(in, r, ci);
|
|
111
|
-
col_stats *cs = &st->stats[i];
|
|
112
292
|
cs->count++;
|
|
113
293
|
double delta = val - cs->mean;
|
|
114
294
|
cs->mean += delta / (double)cs->count;
|
|
@@ -123,101 +303,205 @@ static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
|
123
303
|
return TF_OK;
|
|
124
304
|
}
|
|
125
305
|
|
|
306
|
+
static size_t normalize_missing_null_count(const normalize_state *st) {
|
|
307
|
+
size_t n = 0;
|
|
308
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
309
|
+
if (st->stats[i].missing_null) n++;
|
|
310
|
+
}
|
|
311
|
+
return n;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
static int normalize_is_output_norm_col(const normalize_state *st, size_t col) {
|
|
315
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
316
|
+
const col_stats *cs = &st->stats[i];
|
|
317
|
+
if (cs->col_idx == (int)col && !cs->ignored) return 1;
|
|
318
|
+
}
|
|
319
|
+
return 0;
|
|
320
|
+
}
|
|
321
|
+
|
|
126
322
|
static int normalize_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
127
|
-
(void)side;
|
|
128
323
|
normalize_state *st = self->state;
|
|
129
324
|
*out = NULL;
|
|
130
325
|
|
|
131
326
|
if (st->n_rows == 0) return TF_OK;
|
|
132
327
|
|
|
133
|
-
|
|
328
|
+
size_t extra_cols = normalize_missing_null_count(st);
|
|
329
|
+
size_t out_cols = 0;
|
|
330
|
+
if (tf_size_add(st->schema_n_cols, extra_cols, &out_cols) != TF_OK) return TF_ERROR;
|
|
331
|
+
tf_batch *ob = tf_batch_create(out_cols, st->n_rows);
|
|
134
332
|
if (!ob) return TF_ERROR;
|
|
135
333
|
for (size_t c = 0; c < st->schema_n_cols; c++) {
|
|
136
|
-
/* Normalized columns become FLOAT64 */
|
|
137
334
|
tf_type type = st->schema_types[c];
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
335
|
+
if (normalize_is_output_norm_col(st, c)) type = TF_TYPE_FLOAT64;
|
|
336
|
+
if (tf_batch_set_schema(ob, c, st->schema_names[c], type) != TF_OK) {
|
|
337
|
+
tf_batch_free(ob);
|
|
338
|
+
return TF_ERROR;
|
|
141
339
|
}
|
|
142
|
-
|
|
143
|
-
|
|
340
|
+
}
|
|
341
|
+
size_t extra = 0;
|
|
342
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
343
|
+
if (!st->stats[i].missing_null) continue;
|
|
344
|
+
if (tf_batch_set_schema(ob, st->schema_n_cols + extra, st->columns[i], TF_TYPE_FLOAT64) != TF_OK) {
|
|
345
|
+
tf_batch_free(ob);
|
|
346
|
+
return TF_ERROR;
|
|
347
|
+
}
|
|
348
|
+
extra++;
|
|
144
349
|
}
|
|
145
350
|
|
|
146
351
|
for (size_t r = 0; r < st->n_rows; r++) {
|
|
147
352
|
tf_batch *rb = st->rows[r].batch;
|
|
148
|
-
|
|
353
|
+
for (size_t c = 0; c < st->schema_n_cols; c++) {
|
|
354
|
+
if (normalize_is_output_norm_col(st, c)) continue;
|
|
355
|
+
if (tf_batch_copy_cell(ob, r, c, rb, 0, c) != TF_OK) {
|
|
356
|
+
tf_batch_free(ob);
|
|
357
|
+
return TF_ERROR;
|
|
358
|
+
}
|
|
359
|
+
}
|
|
149
360
|
|
|
150
|
-
/* Normalize target columns */
|
|
151
361
|
for (size_t i = 0; i < st->n_columns; i++) {
|
|
152
|
-
|
|
153
|
-
|
|
362
|
+
col_stats *cs = &st->stats[i];
|
|
363
|
+
int ci = cs->col_idx;
|
|
364
|
+
if (cs->ignored || cs->missing_null || ci < 0) continue;
|
|
365
|
+
if (cs->type_null) {
|
|
366
|
+
if (tf_batch_set_null(ob, r, (size_t)ci) != TF_OK) {
|
|
367
|
+
tf_batch_free(ob);
|
|
368
|
+
return TF_ERROR;
|
|
369
|
+
}
|
|
370
|
+
continue;
|
|
371
|
+
}
|
|
372
|
+
if (tf_batch_is_null(rb, 0, (size_t)ci)) continue;
|
|
154
373
|
|
|
155
374
|
double val = get_numeric(rb, 0, ci);
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
375
|
+
double norm = normalize_compute(st->method, cs, val);
|
|
376
|
+
if (tf_batch_set_float64(ob, r, (size_t)ci, norm) != TF_OK) {
|
|
377
|
+
tf_batch_free(ob);
|
|
378
|
+
return TF_ERROR;
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
382
|
+
tf_batch_free(ob);
|
|
383
|
+
return TF_ERROR;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
if (st->audit) {
|
|
387
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
388
|
+
col_stats *cs = &st->stats[i];
|
|
389
|
+
int ci = cs->col_idx;
|
|
390
|
+
if (cs->ignored || cs->missing_null || cs->type_null || ci < 0 || tf_batch_is_null(rb, 0, (size_t)ci)) continue;
|
|
391
|
+
|
|
392
|
+
double val = get_numeric(rb, 0, ci);
|
|
393
|
+
double norm = normalize_compute(st->method, cs, val);
|
|
394
|
+
if (val == norm) continue;
|
|
395
|
+
if (emit_normalize_audit(st, rb, ob, 0, r, (size_t)ci, r + 1, cs, side) != TF_OK) {
|
|
396
|
+
tf_batch_free(ob);
|
|
397
|
+
return TF_ERROR;
|
|
398
|
+
}
|
|
165
399
|
}
|
|
166
|
-
tf_batch_set_float64(ob, r, ci, norm);
|
|
167
400
|
}
|
|
168
|
-
ob->n_rows = r + 1;
|
|
169
401
|
}
|
|
170
402
|
|
|
171
403
|
*out = ob;
|
|
172
404
|
return TF_OK;
|
|
173
405
|
}
|
|
174
406
|
|
|
175
|
-
static void
|
|
176
|
-
|
|
177
|
-
|
|
407
|
+
static void normalize_state_free(normalize_state *st) {
|
|
408
|
+
if (!st) return;
|
|
409
|
+
tf_audit_options_free(&st->audit_opts);
|
|
410
|
+
if (st->columns) {
|
|
178
411
|
for (size_t i = 0; i < st->n_columns; i++)
|
|
179
412
|
free(st->columns[i]);
|
|
180
|
-
|
|
181
|
-
|
|
413
|
+
}
|
|
414
|
+
free(st->columns);
|
|
415
|
+
free(st->stats);
|
|
416
|
+
if (st->rows) {
|
|
182
417
|
for (size_t i = 0; i < st->n_rows; i++)
|
|
183
418
|
tf_batch_free(st->rows[i].batch);
|
|
184
|
-
free(st->rows);
|
|
185
|
-
if (st->schema_names) {
|
|
186
|
-
for (size_t c = 0; c < st->schema_n_cols; c++)
|
|
187
|
-
free(st->schema_names[c]);
|
|
188
|
-
free(st->schema_names);
|
|
189
|
-
}
|
|
190
|
-
free(st->schema_types);
|
|
191
|
-
free(st);
|
|
192
419
|
}
|
|
420
|
+
free(st->rows);
|
|
421
|
+
if (st->schema_names) {
|
|
422
|
+
for (size_t c = 0; c < st->schema_n_cols; c++)
|
|
423
|
+
free(st->schema_names[c]);
|
|
424
|
+
free(st->schema_names);
|
|
425
|
+
}
|
|
426
|
+
free(st->schema_types);
|
|
427
|
+
free(st);
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
static void normalize_destroy(tf_step *self) {
|
|
431
|
+
if (self) normalize_state_free(self->state);
|
|
193
432
|
free(self);
|
|
194
433
|
}
|
|
195
434
|
|
|
196
435
|
tf_step *tf_normalize_create(const cJSON *args) {
|
|
197
436
|
if (!args) return NULL;
|
|
198
437
|
cJSON *cols_j = cJSON_GetObjectItemCaseSensitive(args, "columns");
|
|
199
|
-
if (!cols_j || !cJSON_IsArray(cols_j))
|
|
438
|
+
if (!cols_j || !cJSON_IsArray(cols_j)) {
|
|
439
|
+
tf_set_last_error("normalize: columns are required");
|
|
440
|
+
return NULL;
|
|
441
|
+
}
|
|
200
442
|
|
|
201
443
|
int n = cJSON_GetArraySize(cols_j);
|
|
202
|
-
if (n == 0)
|
|
444
|
+
if (n == 0) {
|
|
445
|
+
tf_set_last_error("normalize: columns are required");
|
|
446
|
+
return NULL;
|
|
447
|
+
}
|
|
203
448
|
|
|
204
449
|
normalize_state *st = calloc(1, sizeof(normalize_state));
|
|
205
450
|
if (!st) return NULL;
|
|
451
|
+
tf_audit_options_init(&st->audit_opts, 1);
|
|
206
452
|
|
|
207
|
-
st->
|
|
208
|
-
st->
|
|
209
|
-
st->
|
|
453
|
+
st->audit_limit = 1000;
|
|
454
|
+
st->n_columns = (size_t)n;
|
|
455
|
+
st->columns = tf_callocarray_checked((size_t)n, sizeof(char *));
|
|
456
|
+
st->stats = tf_callocarray_checked((size_t)n, sizeof(col_stats));
|
|
457
|
+
if (!st->columns || !st->stats) { normalize_state_free(st); return NULL; }
|
|
210
458
|
for (int i = 0; i < n; i++) {
|
|
211
459
|
cJSON *item = cJSON_GetArrayItem(cols_j, i);
|
|
212
|
-
|
|
460
|
+
if (!cJSON_IsString(item) || !item->valuestring[0]) {
|
|
461
|
+
tf_set_last_error("normalize: columns must be non-empty strings");
|
|
462
|
+
normalize_state_free(st);
|
|
463
|
+
return NULL;
|
|
464
|
+
}
|
|
465
|
+
st->columns[i] = strdup(item->valuestring);
|
|
466
|
+
if (!st->columns[i]) { normalize_state_free(st); return NULL; }
|
|
213
467
|
st->stats[i].col_idx = -1;
|
|
214
468
|
}
|
|
215
469
|
|
|
216
470
|
cJSON *method_j = cJSON_GetObjectItemCaseSensitive(args, "method");
|
|
217
|
-
|
|
471
|
+
if (method_j && !cJSON_IsString(method_j)) {
|
|
472
|
+
tf_set_last_error("normalize: method must be minmax or zscore");
|
|
473
|
+
normalize_state_free(st);
|
|
474
|
+
return NULL;
|
|
475
|
+
}
|
|
476
|
+
if (parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL, &st->method) != TF_OK ||
|
|
477
|
+
normalize_parse_missing_policy(args, &st->missing) != TF_OK ||
|
|
478
|
+
normalize_parse_type_policy(args, &st->on_type_error) != TF_OK) {
|
|
479
|
+
normalize_state_free(st);
|
|
480
|
+
return NULL;
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
cJSON *audit_j = cJSON_GetObjectItemCaseSensitive(args, "audit");
|
|
484
|
+
st->audit = cJSON_IsTrue(audit_j) ? 1 : 0;
|
|
485
|
+
cJSON *audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "audit_limit");
|
|
486
|
+
if (!audit_limit_j) audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "auditLimit");
|
|
487
|
+
if (audit_limit_j) {
|
|
488
|
+
size_t parsed_limit = 0;
|
|
489
|
+
if (tf_json_get_size_arg_any(args, "audit_limit", "auditLimit",
|
|
490
|
+
1, TF_MAX_AUDIT_RECORDS,
|
|
491
|
+
&parsed_limit, "normalize") < 0) {
|
|
492
|
+
normalize_state_free(st);
|
|
493
|
+
return NULL;
|
|
494
|
+
}
|
|
495
|
+
st->audit_limit = parsed_limit;
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
if (tf_audit_options_parse(&st->audit_opts, args, "normalize") != TF_OK) {
|
|
499
|
+
normalize_state_free(st);
|
|
500
|
+
return NULL;
|
|
501
|
+
}
|
|
218
502
|
|
|
219
|
-
tf_step *step =
|
|
220
|
-
if (!step) {
|
|
503
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
504
|
+
if (!step) { normalize_state_free(st); return NULL; }
|
|
221
505
|
step->process = normalize_process;
|
|
222
506
|
step->flush = normalize_flush;
|
|
223
507
|
step->destroy = normalize_destroy;
|