tranfi 0.0.2 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +395 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-pDFMluyz.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +69 -0
- package/csrc/arena.c +91 -0
- package/csrc/batch.c +229 -0
- package/csrc/buffer.c +78 -0
- package/csrc/cJSON.c +3143 -0
- package/csrc/cJSON.h +300 -0
- package/csrc/codec_csv.c +1058 -0
- package/csrc/codec_jsonl.c +374 -0
- package/csrc/codec_table.c +218 -0
- package/csrc/codec_text.c +229 -0
- package/csrc/compiler.c +102 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +1180 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1245 -0
- package/csrc/expr.h +56 -0
- package/csrc/internal.h +250 -0
- package/csrc/ir.c +119 -0
- package/csrc/ir.h +167 -0
- package/csrc/ir_schema.c +60 -0
- package/csrc/ir_serialize.c +104 -0
- package/csrc/ir_sql.c +1211 -0
- package/csrc/ir_validate.c +120 -0
- package/csrc/main.c +392 -0
- package/csrc/op_acf.c +133 -0
- package/csrc/op_anomaly.c +120 -0
- package/csrc/op_bin.c +109 -0
- package/csrc/op_cast.c +195 -0
- package/csrc/op_clip.c +88 -0
- package/csrc/op_date_trunc.c +181 -0
- package/csrc/op_datetime.c +212 -0
- package/csrc/op_derive.c +248 -0
- package/csrc/op_diff.c +134 -0
- package/csrc/op_ewma.c +103 -0
- package/csrc/op_explode.c +108 -0
- package/csrc/op_fill_down.c +163 -0
- package/csrc/op_fill_null.c +123 -0
- package/csrc/op_filter.c +132 -0
- package/csrc/op_frequency.c +193 -0
- package/csrc/op_grep.c +163 -0
- package/csrc/op_group_agg.c +285 -0
- package/csrc/op_hash.c +126 -0
- package/csrc/op_head.c +149 -0
- package/csrc/op_interpolate.c +239 -0
- package/csrc/op_join.c +384 -0
- package/csrc/op_label_encode.c +144 -0
- package/csrc/op_lead.c +190 -0
- package/csrc/op_normalize.c +226 -0
- package/csrc/op_onehot.c +185 -0
- package/csrc/op_pivot.c +370 -0
- package/csrc/op_registry.c +1148 -0
- package/csrc/op_rename.c +138 -0
- package/csrc/op_replace.c +202 -0
- package/csrc/op_sample.c +101 -0
- package/csrc/op_select.c +140 -0
- package/csrc/op_skip.c +152 -0
- package/csrc/op_sort.c +273 -0
- package/csrc/op_split.c +114 -0
- package/csrc/op_split_data.c +87 -0
- package/csrc/op_stack.c +315 -0
- package/csrc/op_stats.c +779 -0
- package/csrc/op_step.c +171 -0
- package/csrc/op_tail.c +96 -0
- package/csrc/op_top.c +150 -0
- package/csrc/op_trim.c +109 -0
- package/csrc/op_unique.c +300 -0
- package/csrc/op_unpivot.c +159 -0
- package/csrc/op_validate.c +71 -0
- package/csrc/op_window.c +150 -0
- package/csrc/pipeline.c +315 -0
- package/csrc/plan.c +206 -0
- package/csrc/recipes.c +102 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +463 -0
- package/csrc/report.h +22 -0
- package/csrc/tranfi.h +123 -0
- package/csrc/wasm_api.c +157 -0
- package/napi_api.c +326 -0
- package/package.json +46 -57
- package/src/cli.js +193 -0
- package/src/engines/duckdb.js +109 -0
- package/src/index.js +306 -0
- package/src/native.js +22 -0
- package/src/pipeline.js +286 -0
- package/src/server.js +277 -0
- package/src/wasm.js +19 -0
- package/wasm/index.js +244 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/LICENSE +0 -21
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/csrc/op_lead.c
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_lead.c — Lookahead: access value N rows ahead.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"column": "price", "offset": 1, "result": "next_price"}
|
|
5
|
+
*
|
|
6
|
+
* Buffers the last `offset` rows across batch boundaries.
|
|
7
|
+
* On flush, emits remaining rows with NULL lead values.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
#include "internal.h"
|
|
11
|
+
#include "cJSON.h"
|
|
12
|
+
#include <stdlib.h>
|
|
13
|
+
#include <string.h>
|
|
14
|
+
#include <stdio.h>
|
|
15
|
+
|
|
16
|
+
typedef struct {
|
|
17
|
+
char *column;
|
|
18
|
+
char *result;
|
|
19
|
+
size_t offset;
|
|
20
|
+
/* Pending rows: stored as a batch that hasn't been emitted yet */
|
|
21
|
+
tf_batch *pending;
|
|
22
|
+
} lead_state;
|
|
23
|
+
|
|
24
|
+
static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
25
|
+
if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
|
|
26
|
+
if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
|
|
27
|
+
return 0;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
static int lead_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
31
|
+
tf_side_channels *side) {
|
|
32
|
+
(void)side;
|
|
33
|
+
lead_state *st = self->state;
|
|
34
|
+
*out = NULL;
|
|
35
|
+
|
|
36
|
+
size_t pend_count = st->pending ? st->pending->n_rows : 0;
|
|
37
|
+
size_t total = pend_count + in->n_rows;
|
|
38
|
+
|
|
39
|
+
if (total <= st->offset) {
|
|
40
|
+
/* Not enough rows yet — append all to pending */
|
|
41
|
+
tf_batch *new_pend = tf_batch_create(in->n_cols, total);
|
|
42
|
+
if (!new_pend) return TF_ERROR;
|
|
43
|
+
for (size_t c = 0; c < in->n_cols; c++)
|
|
44
|
+
tf_batch_set_schema(new_pend, c, in->col_names[c], in->col_types[c]);
|
|
45
|
+
size_t row = 0;
|
|
46
|
+
if (st->pending) {
|
|
47
|
+
for (size_t r = 0; r < pend_count; r++)
|
|
48
|
+
tf_batch_copy_row(new_pend, row++, st->pending, r);
|
|
49
|
+
tf_batch_free(st->pending);
|
|
50
|
+
}
|
|
51
|
+
for (size_t r = 0; r < in->n_rows; r++)
|
|
52
|
+
tf_batch_copy_row(new_pend, row++, in, r);
|
|
53
|
+
new_pend->n_rows = total;
|
|
54
|
+
st->pending = new_pend;
|
|
55
|
+
return TF_OK;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
size_t emit_count = total - st->offset;
|
|
59
|
+
tf_batch *ob = tf_batch_create(in->n_cols + 1, emit_count);
|
|
60
|
+
if (!ob) return TF_ERROR;
|
|
61
|
+
for (size_t c = 0; c < in->n_cols; c++)
|
|
62
|
+
tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
|
|
63
|
+
tf_batch_set_schema(ob, in->n_cols, st->result, TF_TYPE_FLOAT64);
|
|
64
|
+
|
|
65
|
+
/* Emit rows: for each emitted row i, lead value comes from row i+offset */
|
|
66
|
+
for (size_t i = 0; i < emit_count; i++) {
|
|
67
|
+
/* Source row for base data */
|
|
68
|
+
if (i < pend_count) {
|
|
69
|
+
tf_batch_copy_row(ob, i, st->pending, i);
|
|
70
|
+
} else {
|
|
71
|
+
tf_batch_copy_row(ob, i, in, i - pend_count);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/* Lead value from row i + offset */
|
|
75
|
+
size_t lead_idx = i + st->offset;
|
|
76
|
+
tf_batch *lead_src;
|
|
77
|
+
size_t lead_row;
|
|
78
|
+
if (lead_idx < pend_count) {
|
|
79
|
+
lead_src = st->pending;
|
|
80
|
+
lead_row = lead_idx;
|
|
81
|
+
} else {
|
|
82
|
+
lead_src = in;
|
|
83
|
+
lead_row = lead_idx - pend_count;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
int lead_ci = tf_batch_col_index(lead_src, st->column);
|
|
87
|
+
if (lead_ci >= 0 && !tf_batch_is_null(lead_src, lead_row, lead_ci)) {
|
|
88
|
+
tf_batch_set_float64(ob, i, in->n_cols, get_numeric(lead_src, lead_row, lead_ci));
|
|
89
|
+
} else {
|
|
90
|
+
tf_batch_set_null(ob, i, in->n_cols);
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
ob->n_rows = emit_count;
|
|
94
|
+
|
|
95
|
+
/* Store remaining rows as new pending */
|
|
96
|
+
if (st->pending) {
|
|
97
|
+
tf_batch_free(st->pending);
|
|
98
|
+
st->pending = NULL;
|
|
99
|
+
}
|
|
100
|
+
size_t new_pend_count = st->offset;
|
|
101
|
+
if (new_pend_count > 0) {
|
|
102
|
+
tf_batch *new_pend = tf_batch_create(in->n_cols, new_pend_count);
|
|
103
|
+
if (!new_pend) { tf_batch_free(ob); return TF_ERROR; }
|
|
104
|
+
for (size_t c = 0; c < in->n_cols; c++)
|
|
105
|
+
tf_batch_set_schema(new_pend, c, in->col_names[c], in->col_types[c]);
|
|
106
|
+
for (size_t i = 0; i < new_pend_count; i++) {
|
|
107
|
+
size_t src_idx = total - st->offset + i;
|
|
108
|
+
if (src_idx < pend_count) {
|
|
109
|
+
/* This shouldn't happen since we emit at least pend_count rows when total > offset */
|
|
110
|
+
tf_batch_copy_row(new_pend, i, st->pending, src_idx);
|
|
111
|
+
} else {
|
|
112
|
+
tf_batch_copy_row(new_pend, i, in, src_idx - pend_count);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
new_pend->n_rows = new_pend_count;
|
|
116
|
+
st->pending = new_pend;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
*out = ob;
|
|
120
|
+
return TF_OK;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
static int lead_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
124
|
+
(void)side;
|
|
125
|
+
lead_state *st = self->state;
|
|
126
|
+
*out = NULL;
|
|
127
|
+
|
|
128
|
+
if (!st->pending || st->pending->n_rows == 0) return TF_OK;
|
|
129
|
+
|
|
130
|
+
/* Emit remaining pending rows with NULL lead values */
|
|
131
|
+
tf_batch *pend = st->pending;
|
|
132
|
+
tf_batch *ob = tf_batch_create(pend->n_cols + 1, pend->n_rows);
|
|
133
|
+
if (!ob) return TF_ERROR;
|
|
134
|
+
for (size_t c = 0; c < pend->n_cols; c++)
|
|
135
|
+
tf_batch_set_schema(ob, c, pend->col_names[c], pend->col_types[c]);
|
|
136
|
+
tf_batch_set_schema(ob, pend->n_cols, st->result, TF_TYPE_FLOAT64);
|
|
137
|
+
|
|
138
|
+
for (size_t r = 0; r < pend->n_rows; r++) {
|
|
139
|
+
tf_batch_copy_row(ob, r, pend, r);
|
|
140
|
+
tf_batch_set_null(ob, r, pend->n_cols);
|
|
141
|
+
}
|
|
142
|
+
ob->n_rows = pend->n_rows;
|
|
143
|
+
|
|
144
|
+
tf_batch_free(st->pending);
|
|
145
|
+
st->pending = NULL;
|
|
146
|
+
|
|
147
|
+
*out = ob;
|
|
148
|
+
return TF_OK;
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
static void lead_destroy(tf_step *self) {
|
|
152
|
+
lead_state *st = self->state;
|
|
153
|
+
if (st) {
|
|
154
|
+
free(st->column);
|
|
155
|
+
free(st->result);
|
|
156
|
+
if (st->pending) tf_batch_free(st->pending);
|
|
157
|
+
free(st);
|
|
158
|
+
}
|
|
159
|
+
free(self);
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
tf_step *tf_lead_create(const cJSON *args) {
|
|
163
|
+
if (!args) return NULL;
|
|
164
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
165
|
+
if (!cJSON_IsString(col_j)) return NULL;
|
|
166
|
+
|
|
167
|
+
lead_state *st = calloc(1, sizeof(lead_state));
|
|
168
|
+
if (!st) return NULL;
|
|
169
|
+
st->column = strdup(col_j->valuestring);
|
|
170
|
+
|
|
171
|
+
cJSON *off_j = cJSON_GetObjectItemCaseSensitive(args, "offset");
|
|
172
|
+
st->offset = (cJSON_IsNumber(off_j) && off_j->valueint > 0) ? (size_t)off_j->valueint : 1;
|
|
173
|
+
|
|
174
|
+
cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
|
|
175
|
+
if (cJSON_IsString(res_j)) {
|
|
176
|
+
st->result = strdup(res_j->valuestring);
|
|
177
|
+
} else {
|
|
178
|
+
char buf[256];
|
|
179
|
+
snprintf(buf, sizeof(buf), "%s_lead", st->column);
|
|
180
|
+
st->result = strdup(buf);
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
184
|
+
if (!step) { free(st->column); free(st->result); free(st); return NULL; }
|
|
185
|
+
step->process = lead_process;
|
|
186
|
+
step->flush = lead_flush;
|
|
187
|
+
step->destroy = lead_destroy;
|
|
188
|
+
step->state = st;
|
|
189
|
+
return step;
|
|
190
|
+
}
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_normalize.c — Min-max or z-score normalization.
|
|
3
|
+
* Aggregate op: buffers all rows, computes stats, then emits normalized.
|
|
4
|
+
*
|
|
5
|
+
* Config: {"columns": ["price", "score"], "method": "minmax"}
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "internal.h"
|
|
9
|
+
#include "cJSON.h"
|
|
10
|
+
#include <stdlib.h>
|
|
11
|
+
#include <string.h>
|
|
12
|
+
#include <stdio.h>
|
|
13
|
+
#include <math.h>
|
|
14
|
+
|
|
15
|
+
typedef enum {
|
|
16
|
+
NORM_MINMAX,
|
|
17
|
+
NORM_ZSCORE
|
|
18
|
+
} norm_method;
|
|
19
|
+
|
|
20
|
+
typedef struct {
|
|
21
|
+
int col_idx;
|
|
22
|
+
/* Welford's online stats */
|
|
23
|
+
size_t count;
|
|
24
|
+
double mean;
|
|
25
|
+
double m2;
|
|
26
|
+
double min_val;
|
|
27
|
+
double max_val;
|
|
28
|
+
} col_stats;
|
|
29
|
+
|
|
30
|
+
/* Row buffer entry */
|
|
31
|
+
typedef struct {
|
|
32
|
+
tf_batch *batch; /* single-row batch */
|
|
33
|
+
} buf_row;
|
|
34
|
+
|
|
35
|
+
typedef struct {
|
|
36
|
+
char **columns;
|
|
37
|
+
size_t n_columns;
|
|
38
|
+
norm_method method;
|
|
39
|
+
col_stats *stats;
|
|
40
|
+
buf_row *rows;
|
|
41
|
+
size_t n_rows;
|
|
42
|
+
size_t cap_rows;
|
|
43
|
+
int has_schema;
|
|
44
|
+
size_t schema_n_cols;
|
|
45
|
+
char **schema_names;
|
|
46
|
+
tf_type *schema_types;
|
|
47
|
+
} normalize_state;
|
|
48
|
+
|
|
49
|
+
static norm_method parse_method(const char *s) {
|
|
50
|
+
if (s && strcmp(s, "zscore") == 0) return NORM_ZSCORE;
|
|
51
|
+
return NORM_MINMAX;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
55
|
+
if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
|
|
56
|
+
if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
|
|
57
|
+
return 0;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
static void add_row(normalize_state *st, tf_batch *b, size_t r) {
|
|
61
|
+
if (st->n_rows >= st->cap_rows) {
|
|
62
|
+
size_t newcap = st->cap_rows ? st->cap_rows * 2 : 256;
|
|
63
|
+
buf_row *tmp = realloc(st->rows, newcap * sizeof(buf_row));
|
|
64
|
+
if (!tmp) return;
|
|
65
|
+
st->rows = tmp;
|
|
66
|
+
st->cap_rows = newcap;
|
|
67
|
+
}
|
|
68
|
+
/* Copy single row into its own batch */
|
|
69
|
+
tf_batch *rb = tf_batch_create(b->n_cols, 1);
|
|
70
|
+
if (!rb) return;
|
|
71
|
+
for (size_t c = 0; c < b->n_cols; c++)
|
|
72
|
+
tf_batch_set_schema(rb, c, b->col_names[c], b->col_types[c]);
|
|
73
|
+
tf_batch_copy_row(rb, 0, b, r);
|
|
74
|
+
rb->n_rows = 1;
|
|
75
|
+
st->rows[st->n_rows].batch = rb;
|
|
76
|
+
st->n_rows++;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
static int normalize_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
80
|
+
tf_side_channels *side) {
|
|
81
|
+
(void)side;
|
|
82
|
+
normalize_state *st = self->state;
|
|
83
|
+
*out = NULL;
|
|
84
|
+
|
|
85
|
+
/* Save schema from first batch */
|
|
86
|
+
if (!st->has_schema) {
|
|
87
|
+
st->schema_n_cols = in->n_cols;
|
|
88
|
+
st->schema_names = malloc(in->n_cols * sizeof(char *));
|
|
89
|
+
st->schema_types = malloc(in->n_cols * sizeof(tf_type));
|
|
90
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
91
|
+
st->schema_names[c] = strdup(in->col_names[c]);
|
|
92
|
+
st->schema_types[c] = in->col_types[c];
|
|
93
|
+
}
|
|
94
|
+
st->has_schema = 1;
|
|
95
|
+
|
|
96
|
+
/* Resolve column indices */
|
|
97
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
98
|
+
st->stats[i].col_idx = tf_batch_col_index(in, st->columns[i]);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/* Buffer rows and update stats */
|
|
103
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
104
|
+
add_row(st, in, r);
|
|
105
|
+
|
|
106
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
107
|
+
int ci = st->stats[i].col_idx;
|
|
108
|
+
if (ci < 0 || tf_batch_is_null(in, r, ci)) continue;
|
|
109
|
+
|
|
110
|
+
double val = get_numeric(in, r, ci);
|
|
111
|
+
col_stats *cs = &st->stats[i];
|
|
112
|
+
cs->count++;
|
|
113
|
+
double delta = val - cs->mean;
|
|
114
|
+
cs->mean += delta / (double)cs->count;
|
|
115
|
+
double delta2 = val - cs->mean;
|
|
116
|
+
cs->m2 += delta * delta2;
|
|
117
|
+
|
|
118
|
+
if (cs->count == 1 || val < cs->min_val) cs->min_val = val;
|
|
119
|
+
if (cs->count == 1 || val > cs->max_val) cs->max_val = val;
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
return TF_OK;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
static int normalize_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
127
|
+
(void)side;
|
|
128
|
+
normalize_state *st = self->state;
|
|
129
|
+
*out = NULL;
|
|
130
|
+
|
|
131
|
+
if (st->n_rows == 0) return TF_OK;
|
|
132
|
+
|
|
133
|
+
tf_batch *ob = tf_batch_create(st->schema_n_cols, st->n_rows);
|
|
134
|
+
if (!ob) return TF_ERROR;
|
|
135
|
+
for (size_t c = 0; c < st->schema_n_cols; c++) {
|
|
136
|
+
/* Normalized columns become FLOAT64 */
|
|
137
|
+
tf_type type = st->schema_types[c];
|
|
138
|
+
int is_norm_col = 0;
|
|
139
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
140
|
+
if (st->stats[i].col_idx == (int)c) { is_norm_col = 1; break; }
|
|
141
|
+
}
|
|
142
|
+
tf_batch_set_schema(ob, c, st->schema_names[c],
|
|
143
|
+
is_norm_col ? TF_TYPE_FLOAT64 : type);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
for (size_t r = 0; r < st->n_rows; r++) {
|
|
147
|
+
tf_batch *rb = st->rows[r].batch;
|
|
148
|
+
tf_batch_copy_row(ob, r, rb, 0);
|
|
149
|
+
|
|
150
|
+
/* Normalize target columns */
|
|
151
|
+
for (size_t i = 0; i < st->n_columns; i++) {
|
|
152
|
+
int ci = st->stats[i].col_idx;
|
|
153
|
+
if (ci < 0 || tf_batch_is_null(rb, 0, ci)) continue;
|
|
154
|
+
|
|
155
|
+
double val = get_numeric(rb, 0, ci);
|
|
156
|
+
col_stats *cs = &st->stats[i];
|
|
157
|
+
double norm;
|
|
158
|
+
|
|
159
|
+
if (st->method == NORM_MINMAX) {
|
|
160
|
+
double range = cs->max_val - cs->min_val;
|
|
161
|
+
norm = (range > 0) ? (val - cs->min_val) / range : 0;
|
|
162
|
+
} else {
|
|
163
|
+
double std = (cs->count > 1) ? sqrt(cs->m2 / (double)(cs->count - 1)) : 1;
|
|
164
|
+
norm = (std > 0) ? (val - cs->mean) / std : 0;
|
|
165
|
+
}
|
|
166
|
+
tf_batch_set_float64(ob, r, ci, norm);
|
|
167
|
+
}
|
|
168
|
+
ob->n_rows = r + 1;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
*out = ob;
|
|
172
|
+
return TF_OK;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
static void normalize_destroy(tf_step *self) {
|
|
176
|
+
normalize_state *st = self->state;
|
|
177
|
+
if (st) {
|
|
178
|
+
for (size_t i = 0; i < st->n_columns; i++)
|
|
179
|
+
free(st->columns[i]);
|
|
180
|
+
free(st->columns);
|
|
181
|
+
free(st->stats);
|
|
182
|
+
for (size_t i = 0; i < st->n_rows; i++)
|
|
183
|
+
tf_batch_free(st->rows[i].batch);
|
|
184
|
+
free(st->rows);
|
|
185
|
+
if (st->schema_names) {
|
|
186
|
+
for (size_t c = 0; c < st->schema_n_cols; c++)
|
|
187
|
+
free(st->schema_names[c]);
|
|
188
|
+
free(st->schema_names);
|
|
189
|
+
}
|
|
190
|
+
free(st->schema_types);
|
|
191
|
+
free(st);
|
|
192
|
+
}
|
|
193
|
+
free(self);
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
tf_step *tf_normalize_create(const cJSON *args) {
|
|
197
|
+
if (!args) return NULL;
|
|
198
|
+
cJSON *cols_j = cJSON_GetObjectItemCaseSensitive(args, "columns");
|
|
199
|
+
if (!cols_j || !cJSON_IsArray(cols_j)) return NULL;
|
|
200
|
+
|
|
201
|
+
int n = cJSON_GetArraySize(cols_j);
|
|
202
|
+
if (n == 0) return NULL;
|
|
203
|
+
|
|
204
|
+
normalize_state *st = calloc(1, sizeof(normalize_state));
|
|
205
|
+
if (!st) return NULL;
|
|
206
|
+
|
|
207
|
+
st->n_columns = n;
|
|
208
|
+
st->columns = malloc(n * sizeof(char *));
|
|
209
|
+
st->stats = calloc(n, sizeof(col_stats));
|
|
210
|
+
for (int i = 0; i < n; i++) {
|
|
211
|
+
cJSON *item = cJSON_GetArrayItem(cols_j, i);
|
|
212
|
+
st->columns[i] = strdup(cJSON_IsString(item) ? item->valuestring : "");
|
|
213
|
+
st->stats[i].col_idx = -1;
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
cJSON *method_j = cJSON_GetObjectItemCaseSensitive(args, "method");
|
|
217
|
+
st->method = parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL);
|
|
218
|
+
|
|
219
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
220
|
+
if (!step) { normalize_destroy(&(tf_step){.state = st}); return NULL; }
|
|
221
|
+
step->process = normalize_process;
|
|
222
|
+
step->flush = normalize_flush;
|
|
223
|
+
step->destroy = normalize_destroy;
|
|
224
|
+
step->state = st;
|
|
225
|
+
return step;
|
|
226
|
+
}
|
package/csrc/op_onehot.c
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_onehot.c — One-hot encoding of a categorical column.
|
|
3
|
+
* Expands a single column into N binary (0/1) columns.
|
|
4
|
+
*
|
|
5
|
+
* Config: {"column": "city", "drop": false}
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "internal.h"
|
|
9
|
+
#include "cJSON.h"
|
|
10
|
+
#include <stdlib.h>
|
|
11
|
+
#include <string.h>
|
|
12
|
+
#include <stdio.h>
|
|
13
|
+
|
|
14
|
+
typedef struct {
|
|
15
|
+
char *value; /* category value string */
|
|
16
|
+
char *col_name; /* generated column name: "column_value" */
|
|
17
|
+
} onehot_category;
|
|
18
|
+
|
|
19
|
+
typedef struct {
|
|
20
|
+
char *column;
|
|
21
|
+
int drop; /* drop original column */
|
|
22
|
+
onehot_category *cats;
|
|
23
|
+
size_t n_cats;
|
|
24
|
+
size_t cap;
|
|
25
|
+
} onehot_state;
|
|
26
|
+
|
|
27
|
+
static const char *get_string_value(const tf_batch *b, size_t r, int ci, char *buf, size_t bufsz) {
|
|
28
|
+
if (tf_batch_is_null(b, r, ci)) return NULL;
|
|
29
|
+
switch (b->col_types[ci]) {
|
|
30
|
+
case TF_TYPE_STRING: return tf_batch_get_string(b, r, ci);
|
|
31
|
+
case TF_TYPE_INT64:
|
|
32
|
+
snprintf(buf, bufsz, "%lld", (long long)tf_batch_get_int64(b, r, ci));
|
|
33
|
+
return buf;
|
|
34
|
+
case TF_TYPE_FLOAT64:
|
|
35
|
+
snprintf(buf, bufsz, "%.17g", tf_batch_get_float64(b, r, ci));
|
|
36
|
+
return buf;
|
|
37
|
+
case TF_TYPE_BOOL:
|
|
38
|
+
return tf_batch_get_bool(b, r, ci) ? "true" : "false";
|
|
39
|
+
default: return NULL;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
static int find_or_add_category(onehot_state *st, const char *val) {
|
|
44
|
+
for (size_t i = 0; i < st->n_cats; i++) {
|
|
45
|
+
if (strcmp(st->cats[i].value, val) == 0)
|
|
46
|
+
return (int)i;
|
|
47
|
+
}
|
|
48
|
+
/* Add new category */
|
|
49
|
+
if (st->n_cats >= st->cap) {
|
|
50
|
+
size_t newcap = st->cap ? st->cap * 2 : 16;
|
|
51
|
+
onehot_category *tmp = realloc(st->cats, newcap * sizeof(onehot_category));
|
|
52
|
+
if (!tmp) return -1;
|
|
53
|
+
st->cats = tmp;
|
|
54
|
+
st->cap = newcap;
|
|
55
|
+
}
|
|
56
|
+
st->cats[st->n_cats].value = strdup(val);
|
|
57
|
+
char namebuf[512];
|
|
58
|
+
snprintf(namebuf, sizeof(namebuf), "%s_%s", st->column, val);
|
|
59
|
+
st->cats[st->n_cats].col_name = strdup(namebuf);
|
|
60
|
+
st->n_cats++;
|
|
61
|
+
return (int)(st->n_cats - 1);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
static int onehot_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
65
|
+
tf_side_channels *side) {
|
|
66
|
+
(void)side;
|
|
67
|
+
onehot_state *st = self->state;
|
|
68
|
+
*out = NULL;
|
|
69
|
+
|
|
70
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
71
|
+
|
|
72
|
+
/* First pass: discover any new categories in this batch */
|
|
73
|
+
size_t cats_before = st->n_cats;
|
|
74
|
+
if (ci >= 0) {
|
|
75
|
+
char buf[64];
|
|
76
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
77
|
+
const char *val = get_string_value(in, r, ci, buf, sizeof(buf));
|
|
78
|
+
if (val) find_or_add_category(st, val);
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
(void)cats_before;
|
|
82
|
+
|
|
83
|
+
/* Compute output column count */
|
|
84
|
+
size_t out_cols = st->drop ? (in->n_cols - 1 + st->n_cats) : (in->n_cols + st->n_cats);
|
|
85
|
+
|
|
86
|
+
tf_batch *ob = tf_batch_create(out_cols, in->n_rows);
|
|
87
|
+
if (!ob) return TF_ERROR;
|
|
88
|
+
|
|
89
|
+
/* Set schema: copy input cols (optionally skipping target), append onehot cols */
|
|
90
|
+
size_t oc = 0;
|
|
91
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
92
|
+
if (st->drop && ci >= 0 && c == (size_t)ci) continue;
|
|
93
|
+
tf_batch_set_schema(ob, oc, in->col_names[c], in->col_types[c]);
|
|
94
|
+
oc++;
|
|
95
|
+
}
|
|
96
|
+
for (size_t i = 0; i < st->n_cats; i++) {
|
|
97
|
+
tf_batch_set_schema(ob, oc + i, st->cats[i].col_name, TF_TYPE_INT64);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/* Fill rows */
|
|
101
|
+
char buf[64];
|
|
102
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
103
|
+
/* Copy input columns */
|
|
104
|
+
oc = 0;
|
|
105
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
106
|
+
if (st->drop && ci >= 0 && c == (size_t)ci) continue;
|
|
107
|
+
if (tf_batch_is_null(in, r, c)) {
|
|
108
|
+
tf_batch_set_null(ob, r, oc);
|
|
109
|
+
} else {
|
|
110
|
+
switch (in->col_types[c]) {
|
|
111
|
+
case TF_TYPE_STRING:
|
|
112
|
+
tf_batch_set_string(ob, r, oc, tf_batch_get_string(in, r, c)); break;
|
|
113
|
+
case TF_TYPE_INT64:
|
|
114
|
+
tf_batch_set_int64(ob, r, oc, tf_batch_get_int64(in, r, c)); break;
|
|
115
|
+
case TF_TYPE_FLOAT64:
|
|
116
|
+
tf_batch_set_float64(ob, r, oc, tf_batch_get_float64(in, r, c)); break;
|
|
117
|
+
case TF_TYPE_BOOL:
|
|
118
|
+
tf_batch_set_bool(ob, r, oc, tf_batch_get_bool(in, r, c)); break;
|
|
119
|
+
case TF_TYPE_DATE:
|
|
120
|
+
tf_batch_set_date(ob, r, oc, tf_batch_get_date(in, r, c)); break;
|
|
121
|
+
case TF_TYPE_TIMESTAMP:
|
|
122
|
+
tf_batch_set_timestamp(ob, r, oc, tf_batch_get_timestamp(in, r, c)); break;
|
|
123
|
+
default: tf_batch_set_null(ob, r, oc); break;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
oc++;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/* Set onehot columns */
|
|
130
|
+
const char *val = (ci >= 0) ? get_string_value(in, r, ci, buf, sizeof(buf)) : NULL;
|
|
131
|
+
int match = -1;
|
|
132
|
+
if (val) {
|
|
133
|
+
for (size_t i = 0; i < st->n_cats; i++) {
|
|
134
|
+
if (strcmp(st->cats[i].value, val) == 0) { match = (int)i; break; }
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
for (size_t i = 0; i < st->n_cats; i++) {
|
|
138
|
+
tf_batch_set_int64(ob, r, oc + i, (int)i == match ? 1 : 0);
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
ob->n_rows = r + 1;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
*out = ob;
|
|
145
|
+
return TF_OK;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
static int onehot_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
149
|
+
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
static void onehot_destroy(tf_step *self) {
|
|
153
|
+
onehot_state *st = self->state;
|
|
154
|
+
if (st) {
|
|
155
|
+
for (size_t i = 0; i < st->n_cats; i++) {
|
|
156
|
+
free(st->cats[i].value);
|
|
157
|
+
free(st->cats[i].col_name);
|
|
158
|
+
}
|
|
159
|
+
free(st->cats);
|
|
160
|
+
free(st->column);
|
|
161
|
+
free(st);
|
|
162
|
+
}
|
|
163
|
+
free(self);
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
tf_step *tf_onehot_create(const cJSON *args) {
|
|
167
|
+
if (!args) return NULL;
|
|
168
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
169
|
+
if (!cJSON_IsString(col_j)) return NULL;
|
|
170
|
+
|
|
171
|
+
onehot_state *st = calloc(1, sizeof(onehot_state));
|
|
172
|
+
if (!st) return NULL;
|
|
173
|
+
st->column = strdup(col_j->valuestring);
|
|
174
|
+
|
|
175
|
+
cJSON *drop_j = cJSON_GetObjectItemCaseSensitive(args, "drop");
|
|
176
|
+
st->drop = cJSON_IsBool(drop_j) && cJSON_IsTrue(drop_j) ? 1 : 0;
|
|
177
|
+
|
|
178
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
179
|
+
if (!step) { free(st->column); free(st); return NULL; }
|
|
180
|
+
step->process = onehot_process;
|
|
181
|
+
step->flush = onehot_flush;
|
|
182
|
+
step->destroy = onehot_destroy;
|
|
183
|
+
step->state = st;
|
|
184
|
+
return step;
|
|
185
|
+
}
|