tranfi 0.0.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -21
- package/NOTICE +8 -0
- package/README.md +627 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-BIAIKnrp.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +121 -0
- package/csrc/arena.c +93 -0
- package/csrc/batch.c +976 -0
- package/csrc/buffer.c +154 -0
- package/csrc/cJSON.c +3386 -0
- package/csrc/cJSON.h +316 -0
- package/csrc/codec_csv.c +1951 -0
- package/csrc/codec_jsonl.c +1086 -0
- package/csrc/codec_table.c +248 -0
- package/csrc/codec_text.c +447 -0
- package/csrc/compiler.c +130 -0
- package/csrc/config.h +21 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +5417 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1553 -0
- package/csrc/expr.h +58 -0
- package/csrc/internal.h +539 -0
- package/csrc/ir.c +166 -0
- package/csrc/ir.h +208 -0
- package/csrc/ir_schema.c +75 -0
- package/csrc/ir_serialize.c +166 -0
- package/csrc/ir_sql.c +1822 -0
- package/csrc/ir_validate.c +576 -0
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +1241 -0
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +283 -0
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +255 -0
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +248 -0
- package/csrc/op_cast.c +523 -0
- package/csrc/op_clip.c +99 -0
- package/csrc/op_date_trunc.c +355 -0
- package/csrc/op_datetime.c +394 -0
- package/csrc/op_derive.c +216 -0
- package/csrc/op_diff.c +250 -0
- package/csrc/op_ewma.c +222 -0
- package/csrc/op_explode.c +206 -0
- package/csrc/op_fill_down.c +235 -0
- package/csrc/op_fill_null.c +268 -0
- package/csrc/op_filter.c +181 -0
- package/csrc/op_frequency.c +721 -0
- package/csrc/op_grep.c +181 -0
- package/csrc/op_group_agg.c +1956 -0
- package/csrc/op_hash.c +159 -0
- package/csrc/op_head.c +84 -0
- package/csrc/op_interpolate.c +445 -0
- package/csrc/op_join.c +2902 -0
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +419 -0
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +242 -0
- package/csrc/op_normalize.c +510 -0
- package/csrc/op_onehot.c +457 -0
- package/csrc/op_pivot.c +1754 -0
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +3044 -0
- package/csrc/op_rename.c +129 -0
- package/csrc/op_replace.c +354 -0
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +158 -0
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +340 -0
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +95 -0
- package/csrc/op_sort.c +819 -0
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +151 -0
- package/csrc/op_split_data.c +119 -0
- package/csrc/op_stack.c +271 -0
- package/csrc/op_stats.c +875 -0
- package/csrc/op_step.c +333 -0
- package/csrc/op_tail.c +105 -0
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +357 -0
- package/csrc/op_trim.c +138 -0
- package/csrc/op_unique.c +1343 -0
- package/csrc/op_unpivot.c +193 -0
- package/csrc/op_validate.c +648 -0
- package/csrc/op_window.c +591 -0
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +1088 -0
- package/csrc/recipes.c +104 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +506 -0
- package/csrc/report.h +22 -0
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +291 -0
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +218 -0
- package/napi_api.c +534 -0
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +64 -59
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +190 -0
- package/src/engines/duckdb.js +142 -0
- package/src/index.js +925 -0
- package/src/memory_policy.js +411 -0
- package/src/native.js +18 -0
- package/src/pipeline.js +709 -0
- package/src/recipe_json.js +80 -0
- package/src/server.js +279 -0
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +21 -0
- package/wasm/index.js +732 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/csrc/op_diff.c
ADDED
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_diff.c — First (or higher-order) differencing.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"column": "price", "order": 1, "result": "price_diff"}
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
#include "internal.h"
|
|
8
|
+
#include "cJSON.h"
|
|
9
|
+
#include <stdlib.h>
|
|
10
|
+
#include <string.h>
|
|
11
|
+
#include <stdio.h>
|
|
12
|
+
|
|
13
|
+
#define MAX_DIFF_ORDER 8
|
|
14
|
+
|
|
15
|
+
typedef enum {
|
|
16
|
+
DIFF_MISSING_ERROR,
|
|
17
|
+
DIFF_MISSING_NULL,
|
|
18
|
+
DIFF_MISSING_IGNORE,
|
|
19
|
+
} diff_missing_policy;
|
|
20
|
+
|
|
21
|
+
typedef enum {
|
|
22
|
+
DIFF_TYPE_FAIL,
|
|
23
|
+
DIFF_TYPE_NULL,
|
|
24
|
+
} diff_type_policy;
|
|
25
|
+
|
|
26
|
+
typedef struct {
|
|
27
|
+
char *column;
|
|
28
|
+
char *result;
|
|
29
|
+
int order;
|
|
30
|
+
double prev[MAX_DIFF_ORDER]; /* circular buffer of previous values */
|
|
31
|
+
int count; /* rows seen so far */
|
|
32
|
+
diff_missing_policy missing;
|
|
33
|
+
diff_type_policy on_type_error;
|
|
34
|
+
} diff_state;
|
|
35
|
+
|
|
36
|
+
static int diff_is_numeric_type(tf_type type) {
|
|
37
|
+
return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
41
|
+
if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
|
|
42
|
+
return tf_batch_get_float64(b, r, ci);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
static void diff_set_col_error(const char *column, const char *suffix) {
|
|
46
|
+
char msg[512];
|
|
47
|
+
snprintf(msg, sizeof(msg), "diff: column '%s' %s", column ? column : "", suffix);
|
|
48
|
+
tf_set_last_error(msg);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
static int diff_parse_missing_policy(const cJSON *args, diff_missing_policy *out) {
|
|
52
|
+
*out = DIFF_MISSING_ERROR;
|
|
53
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
|
|
54
|
+
if (!j) return TF_OK;
|
|
55
|
+
if (!cJSON_IsString(j)) {
|
|
56
|
+
tf_set_last_error("diff: missing must be error, null, or ignore");
|
|
57
|
+
return TF_ERROR;
|
|
58
|
+
}
|
|
59
|
+
if (strcmp(j->valuestring, "error") == 0) *out = DIFF_MISSING_ERROR;
|
|
60
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = DIFF_MISSING_NULL;
|
|
61
|
+
else if (strcmp(j->valuestring, "ignore") == 0) *out = DIFF_MISSING_IGNORE;
|
|
62
|
+
else {
|
|
63
|
+
tf_set_last_error("diff: missing must be error, null, or ignore");
|
|
64
|
+
return TF_ERROR;
|
|
65
|
+
}
|
|
66
|
+
return TF_OK;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
static int diff_parse_type_policy(const cJSON *args, diff_type_policy *out) {
|
|
70
|
+
*out = DIFF_TYPE_FAIL;
|
|
71
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
|
|
72
|
+
if (!j) return TF_OK;
|
|
73
|
+
if (!cJSON_IsString(j)) {
|
|
74
|
+
tf_set_last_error("diff: on_type_error must be fail or null");
|
|
75
|
+
return TF_ERROR;
|
|
76
|
+
}
|
|
77
|
+
if (strcmp(j->valuestring, "fail") == 0) *out = DIFF_TYPE_FAIL;
|
|
78
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = DIFF_TYPE_NULL;
|
|
79
|
+
else {
|
|
80
|
+
tf_set_last_error("diff: on_type_error must be fail or null");
|
|
81
|
+
return TF_ERROR;
|
|
82
|
+
}
|
|
83
|
+
return TF_OK;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
static int diff_passthrough(tf_batch *in, tf_batch **out) {
|
|
87
|
+
tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
|
|
88
|
+
if (!ob) return TF_ERROR;
|
|
89
|
+
if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
90
|
+
tf_batch_free(ob);
|
|
91
|
+
return TF_ERROR;
|
|
92
|
+
}
|
|
93
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
94
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK ||
|
|
95
|
+
tf_batch_expose_row(ob, r) != TF_OK) {
|
|
96
|
+
tf_batch_free(ob);
|
|
97
|
+
return TF_ERROR;
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
*out = ob;
|
|
101
|
+
return TF_OK;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
static int diff_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
105
|
+
tf_side_channels *side) {
|
|
106
|
+
(void)side;
|
|
107
|
+
diff_state *st = self->state;
|
|
108
|
+
*out = NULL;
|
|
109
|
+
|
|
110
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
111
|
+
int force_null = 0;
|
|
112
|
+
if (ci < 0) {
|
|
113
|
+
if (st->missing == DIFF_MISSING_ERROR) {
|
|
114
|
+
diff_set_col_error(st->column, "not found");
|
|
115
|
+
return TF_ERROR;
|
|
116
|
+
}
|
|
117
|
+
if (st->missing == DIFF_MISSING_IGNORE) {
|
|
118
|
+
return diff_passthrough(in, out);
|
|
119
|
+
}
|
|
120
|
+
force_null = 1;
|
|
121
|
+
} else if (!diff_is_numeric_type(in->col_types[(size_t)ci])) {
|
|
122
|
+
if (st->on_type_error == DIFF_TYPE_FAIL) {
|
|
123
|
+
diff_set_col_error(st->column, "must be numeric");
|
|
124
|
+
return TF_ERROR;
|
|
125
|
+
}
|
|
126
|
+
force_null = 1;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
const char *extra_names[1] = {st->result};
|
|
130
|
+
tf_type extra_types[1] = {TF_TYPE_FLOAT64};
|
|
131
|
+
tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
|
|
132
|
+
if (!ob) return TF_ERROR;
|
|
133
|
+
if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
|
|
134
|
+
tf_batch_free(ob);
|
|
135
|
+
return TF_ERROR;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
139
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
140
|
+
tf_batch_free(ob);
|
|
141
|
+
return TF_ERROR;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
if (force_null || tf_batch_is_null(in, r, (size_t)ci)) {
|
|
145
|
+
if (tf_batch_set_null(ob, r, in->n_cols) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
146
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
double val = get_numeric(in, r, ci);
|
|
151
|
+
|
|
152
|
+
if (st->count < st->order) {
|
|
153
|
+
/* Not enough history yet -- shift and insert at front
|
|
154
|
+
* so prev[0] is always the most recent value */
|
|
155
|
+
if (tf_batch_set_null(ob, r, in->n_cols) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
156
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
157
|
+
for (int k = st->count; k > 0; k--)
|
|
158
|
+
st->prev[k] = st->prev[k - 1];
|
|
159
|
+
st->prev[0] = val;
|
|
160
|
+
st->count++;
|
|
161
|
+
continue;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/* Compute difference using binomial coefficients:
|
|
165
|
+
* diff(order=1): val - prev[0]
|
|
166
|
+
* diff(order=2): val - 2*prev[0] + prev[1]
|
|
167
|
+
* General: sum_{k=0}^{order} (-1)^k * C(order,k) * x_{n-k}
|
|
168
|
+
*/
|
|
169
|
+
double result = 0;
|
|
170
|
+
int binom = 1;
|
|
171
|
+
int sign = 1;
|
|
172
|
+
/* x_n (current value) */
|
|
173
|
+
result = val;
|
|
174
|
+
/* x_{n-1} ... x_{n-order} from prev buffer (most recent first) */
|
|
175
|
+
for (int k = 1; k <= st->order; k++) {
|
|
176
|
+
binom = binom * (st->order - k + 1) / k;
|
|
177
|
+
sign = -sign;
|
|
178
|
+
/* prev[0] is the most recent previous value */
|
|
179
|
+
result += sign * binom * st->prev[k - 1];
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
if (tf_batch_set_float64(ob, r, in->n_cols, result) != TF_OK) {
|
|
183
|
+
tf_batch_free(ob);
|
|
184
|
+
return TF_ERROR;
|
|
185
|
+
}
|
|
186
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
187
|
+
tf_batch_free(ob);
|
|
188
|
+
return TF_ERROR;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/* Shift prev buffer: move everything down, put val at [0] */
|
|
192
|
+
for (int k = st->order - 1; k > 0; k--)
|
|
193
|
+
st->prev[k] = st->prev[k - 1];
|
|
194
|
+
st->prev[0] = val;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
*out = ob;
|
|
198
|
+
return TF_OK;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
static int diff_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
202
|
+
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
static void diff_destroy(tf_step *self) {
|
|
206
|
+
diff_state *st = self->state;
|
|
207
|
+
if (st) { free(st->column); free(st->result); free(st); }
|
|
208
|
+
free(self);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
tf_step *tf_diff_create(const cJSON *args) {
|
|
212
|
+
if (!args) return NULL;
|
|
213
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
214
|
+
if (!cJSON_IsString(col_j) || !col_j->valuestring[0]) {
|
|
215
|
+
tf_set_last_error("diff: column is required");
|
|
216
|
+
return NULL;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
diff_state *st = tf_callocarray_checked(1, sizeof(diff_state));
|
|
220
|
+
if (!st) return NULL;
|
|
221
|
+
st->column = tf_strdup_checked(col_j->valuestring);
|
|
222
|
+
if (!st->column) { free(st); return NULL; }
|
|
223
|
+
|
|
224
|
+
size_t order = 1;
|
|
225
|
+
int has_order = tf_json_get_size_arg(args, "order", 1, MAX_DIFF_ORDER, &order, "diff");
|
|
226
|
+
if (has_order < 0) { free(st->column); free(st); return NULL; }
|
|
227
|
+
st->order = (int)order;
|
|
228
|
+
if (diff_parse_missing_policy(args, &st->missing) != TF_OK ||
|
|
229
|
+
diff_parse_type_policy(args, &st->on_type_error) != TF_OK) {
|
|
230
|
+
free(st->column);
|
|
231
|
+
free(st);
|
|
232
|
+
return NULL;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
|
|
236
|
+
if (cJSON_IsString(res_j)) {
|
|
237
|
+
st->result = tf_strdup_checked(res_j->valuestring);
|
|
238
|
+
} else {
|
|
239
|
+
st->result = tf_string_append_suffix_checked(st->column, "_diff");
|
|
240
|
+
}
|
|
241
|
+
if (!st->result) { free(st->column); free(st); return NULL; }
|
|
242
|
+
|
|
243
|
+
tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
|
|
244
|
+
if (!step) { free(st->column); free(st->result); free(st); return NULL; }
|
|
245
|
+
step->process = diff_process;
|
|
246
|
+
step->flush = diff_flush;
|
|
247
|
+
step->destroy = diff_destroy;
|
|
248
|
+
step->state = st;
|
|
249
|
+
return step;
|
|
250
|
+
}
|
package/csrc/op_ewma.c
ADDED
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_ewma.c — Exponentially weighted moving average.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"column": "price", "alpha": 0.3, "result": "price_ewma"}
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
#include "internal.h"
|
|
8
|
+
#include "cJSON.h"
|
|
9
|
+
#include <stdlib.h>
|
|
10
|
+
#include <string.h>
|
|
11
|
+
#include <stdio.h>
|
|
12
|
+
#include <math.h>
|
|
13
|
+
|
|
14
|
+
typedef enum {
|
|
15
|
+
EWMA_MISSING_ERROR,
|
|
16
|
+
EWMA_MISSING_NULL,
|
|
17
|
+
EWMA_MISSING_IGNORE,
|
|
18
|
+
} ewma_missing_policy;
|
|
19
|
+
|
|
20
|
+
typedef enum {
|
|
21
|
+
EWMA_TYPE_FAIL,
|
|
22
|
+
EWMA_TYPE_NULL,
|
|
23
|
+
} ewma_type_policy;
|
|
24
|
+
|
|
25
|
+
typedef struct {
|
|
26
|
+
char *column;
|
|
27
|
+
char *result;
|
|
28
|
+
double alpha;
|
|
29
|
+
double ewma;
|
|
30
|
+
int initialized;
|
|
31
|
+
ewma_missing_policy missing;
|
|
32
|
+
ewma_type_policy on_type_error;
|
|
33
|
+
} ewma_state;
|
|
34
|
+
|
|
35
|
+
static int ewma_is_numeric_type(tf_type type) {
|
|
36
|
+
return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
static double ewma_get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
40
|
+
if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
|
|
41
|
+
return tf_batch_get_float64(b, r, ci);
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
static void ewma_set_col_error(const char *column, const char *suffix) {
|
|
45
|
+
char msg[512];
|
|
46
|
+
snprintf(msg, sizeof(msg), "ewma: column '%s' %s", column ? column : "", suffix);
|
|
47
|
+
tf_set_last_error(msg);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
static int ewma_parse_missing_policy(const cJSON *args, ewma_missing_policy *out) {
|
|
51
|
+
*out = EWMA_MISSING_ERROR;
|
|
52
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
|
|
53
|
+
if (!j) return TF_OK;
|
|
54
|
+
if (!cJSON_IsString(j)) {
|
|
55
|
+
tf_set_last_error("ewma: missing must be error, null, or ignore");
|
|
56
|
+
return TF_ERROR;
|
|
57
|
+
}
|
|
58
|
+
if (strcmp(j->valuestring, "error") == 0) *out = EWMA_MISSING_ERROR;
|
|
59
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = EWMA_MISSING_NULL;
|
|
60
|
+
else if (strcmp(j->valuestring, "ignore") == 0) *out = EWMA_MISSING_IGNORE;
|
|
61
|
+
else {
|
|
62
|
+
tf_set_last_error("ewma: missing must be error, null, or ignore");
|
|
63
|
+
return TF_ERROR;
|
|
64
|
+
}
|
|
65
|
+
return TF_OK;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
static int ewma_parse_type_policy(const cJSON *args, ewma_type_policy *out) {
|
|
69
|
+
*out = EWMA_TYPE_FAIL;
|
|
70
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
|
|
71
|
+
if (!j) return TF_OK;
|
|
72
|
+
if (!cJSON_IsString(j)) {
|
|
73
|
+
tf_set_last_error("ewma: on_type_error must be fail or null");
|
|
74
|
+
return TF_ERROR;
|
|
75
|
+
}
|
|
76
|
+
if (strcmp(j->valuestring, "fail") == 0) *out = EWMA_TYPE_FAIL;
|
|
77
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = EWMA_TYPE_NULL;
|
|
78
|
+
else {
|
|
79
|
+
tf_set_last_error("ewma: on_type_error must be fail or null");
|
|
80
|
+
return TF_ERROR;
|
|
81
|
+
}
|
|
82
|
+
return TF_OK;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
static int ewma_passthrough(tf_batch *in, tf_batch **out) {
|
|
86
|
+
tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
|
|
87
|
+
if (!ob) return TF_ERROR;
|
|
88
|
+
if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
89
|
+
tf_batch_free(ob);
|
|
90
|
+
return TF_ERROR;
|
|
91
|
+
}
|
|
92
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
93
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
94
|
+
tf_batch_free(ob);
|
|
95
|
+
return TF_ERROR;
|
|
96
|
+
}
|
|
97
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
98
|
+
tf_batch_free(ob);
|
|
99
|
+
return TF_ERROR;
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
*out = ob;
|
|
103
|
+
return TF_OK;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
static int ewma_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
107
|
+
tf_side_channels *side) {
|
|
108
|
+
(void)side;
|
|
109
|
+
ewma_state *st = self->state;
|
|
110
|
+
*out = NULL;
|
|
111
|
+
|
|
112
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
113
|
+
int force_null = 0;
|
|
114
|
+
if (ci < 0) {
|
|
115
|
+
if (st->missing == EWMA_MISSING_ERROR) {
|
|
116
|
+
ewma_set_col_error(st->column, "not found");
|
|
117
|
+
return TF_ERROR;
|
|
118
|
+
}
|
|
119
|
+
if (st->missing == EWMA_MISSING_IGNORE) return ewma_passthrough(in, out);
|
|
120
|
+
force_null = 1;
|
|
121
|
+
} else if (!ewma_is_numeric_type(in->col_types[ci])) {
|
|
122
|
+
if (st->on_type_error == EWMA_TYPE_FAIL) {
|
|
123
|
+
ewma_set_col_error(st->column, "must be numeric");
|
|
124
|
+
return TF_ERROR;
|
|
125
|
+
}
|
|
126
|
+
force_null = 1;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
const char *extra_names[1] = {st->result};
|
|
130
|
+
tf_type extra_types[1] = {TF_TYPE_FLOAT64};
|
|
131
|
+
tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
|
|
132
|
+
if (!ob) return TF_ERROR;
|
|
133
|
+
if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
|
|
134
|
+
tf_batch_free(ob);
|
|
135
|
+
return TF_ERROR;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
139
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
140
|
+
tf_batch_free(ob);
|
|
141
|
+
return TF_ERROR;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
if (force_null || tf_batch_is_null(in, r, ci)) {
|
|
145
|
+
if (tf_batch_set_null(ob, r, in->n_cols) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
146
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
double val = ewma_get_numeric(in, r, ci);
|
|
151
|
+
double next_ewma = st->initialized
|
|
152
|
+
? st->alpha * val + (1.0 - st->alpha) * st->ewma
|
|
153
|
+
: val;
|
|
154
|
+
|
|
155
|
+
if (tf_batch_set_float64(ob, r, in->n_cols, next_ewma) != TF_OK) {
|
|
156
|
+
tf_batch_free(ob);
|
|
157
|
+
return TF_ERROR;
|
|
158
|
+
}
|
|
159
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
160
|
+
tf_batch_free(ob);
|
|
161
|
+
return TF_ERROR;
|
|
162
|
+
}
|
|
163
|
+
st->ewma = next_ewma;
|
|
164
|
+
st->initialized = 1;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
*out = ob;
|
|
168
|
+
return TF_OK;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
static int ewma_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
172
|
+
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
static void ewma_destroy(tf_step *self) {
|
|
176
|
+
ewma_state *st = self->state;
|
|
177
|
+
if (st) { free(st->column); free(st->result); free(st); }
|
|
178
|
+
free(self);
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
tf_step *tf_ewma_create(const cJSON *args) {
|
|
182
|
+
if (!args) return NULL;
|
|
183
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
184
|
+
cJSON *alpha_j = cJSON_GetObjectItemCaseSensitive(args, "alpha");
|
|
185
|
+
if (!cJSON_IsString(col_j) || !col_j->valuestring[0] || !cJSON_IsNumber(alpha_j)) {
|
|
186
|
+
tf_set_last_error("ewma: column and alpha are required");
|
|
187
|
+
return NULL;
|
|
188
|
+
}
|
|
189
|
+
double alpha = alpha_j->valuedouble;
|
|
190
|
+
if (!isfinite(alpha) || alpha < 0.0 || alpha > 1.0) {
|
|
191
|
+
tf_set_last_error("ewma: alpha must be a finite number between 0 and 1");
|
|
192
|
+
return NULL;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
ewma_state *st = tf_callocarray_checked(1, sizeof(ewma_state));
|
|
196
|
+
if (!st) return NULL;
|
|
197
|
+
st->column = tf_strdup_checked(col_j->valuestring);
|
|
198
|
+
st->alpha = alpha;
|
|
199
|
+
if (!st->column) { free(st); return NULL; }
|
|
200
|
+
if (ewma_parse_missing_policy(args, &st->missing) != TF_OK ||
|
|
201
|
+
ewma_parse_type_policy(args, &st->on_type_error) != TF_OK) {
|
|
202
|
+
free(st->column);
|
|
203
|
+
free(st);
|
|
204
|
+
return NULL;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
|
|
208
|
+
if (cJSON_IsString(res_j)) {
|
|
209
|
+
st->result = tf_strdup_checked(res_j->valuestring);
|
|
210
|
+
} else {
|
|
211
|
+
st->result = tf_string_append_suffix_checked(st->column, "_ewma");
|
|
212
|
+
}
|
|
213
|
+
if (!st->result) { free(st->column); free(st); return NULL; }
|
|
214
|
+
|
|
215
|
+
tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
|
|
216
|
+
if (!step) { free(st->column); free(st->result); free(st); return NULL; }
|
|
217
|
+
step->process = ewma_process;
|
|
218
|
+
step->flush = ewma_flush;
|
|
219
|
+
step->destroy = ewma_destroy;
|
|
220
|
+
step->state = st;
|
|
221
|
+
return step;
|
|
222
|
+
}
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_explode.c -- Split delimited string into multiple rows.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"column": "tags", "delimiter": ",", "max_tokens_per_row": 1024}
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
#include "internal.h"
|
|
8
|
+
#include "cJSON.h"
|
|
9
|
+
#include <stdio.h>
|
|
10
|
+
#include <stdlib.h>
|
|
11
|
+
#include <string.h>
|
|
12
|
+
|
|
13
|
+
typedef struct {
|
|
14
|
+
char *column;
|
|
15
|
+
char *delimiter;
|
|
16
|
+
size_t max_tokens_per_row;
|
|
17
|
+
size_t max_output_rows_per_input_row;
|
|
18
|
+
size_t max_output_rows_per_batch;
|
|
19
|
+
size_t max_token_bytes;
|
|
20
|
+
} explode_state;
|
|
21
|
+
|
|
22
|
+
static int explode_set_cap_error(const char *name, size_t limit) {
|
|
23
|
+
char msg[160];
|
|
24
|
+
snprintf(msg, sizeof(msg), "explode: %s=%zu exceeded", name, limit);
|
|
25
|
+
tf_set_last_error(msg);
|
|
26
|
+
return TF_ERROR;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
static int explode_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
30
|
+
tf_side_channels *side) {
|
|
31
|
+
(void)side;
|
|
32
|
+
explode_state *st = self->state;
|
|
33
|
+
*out = NULL;
|
|
34
|
+
|
|
35
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
36
|
+
|
|
37
|
+
size_t initial_cap = in->n_rows ? in->n_rows : 1;
|
|
38
|
+
if (initial_cap > st->max_output_rows_per_batch)
|
|
39
|
+
initial_cap = st->max_output_rows_per_batch;
|
|
40
|
+
if (initial_cap < 16 && st->max_output_rows_per_batch >= 16)
|
|
41
|
+
initial_cap = 16;
|
|
42
|
+
tf_batch *ob = tf_batch_create(in->n_cols, initial_cap);
|
|
43
|
+
if (!ob) return TF_ERROR;
|
|
44
|
+
if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
45
|
+
tf_batch_free(ob);
|
|
46
|
+
return TF_ERROR;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
size_t out_row = 0;
|
|
50
|
+
size_t delim_len = strlen(st->delimiter);
|
|
51
|
+
|
|
52
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
53
|
+
if (ci < 0 || tf_batch_is_null(in, r, ci) || in->col_types[ci] != TF_TYPE_STRING) {
|
|
54
|
+
if (out_row >= st->max_output_rows_per_batch) {
|
|
55
|
+
explode_set_cap_error("max_output_rows_per_batch", st->max_output_rows_per_batch);
|
|
56
|
+
tf_batch_free(ob);
|
|
57
|
+
return TF_ERROR;
|
|
58
|
+
}
|
|
59
|
+
if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
|
|
60
|
+
tf_batch_free(ob);
|
|
61
|
+
return TF_ERROR;
|
|
62
|
+
}
|
|
63
|
+
if (tf_batch_expose_row(ob, out_row) != TF_OK) {
|
|
64
|
+
tf_batch_free(ob);
|
|
65
|
+
return TF_ERROR;
|
|
66
|
+
}
|
|
67
|
+
out_row++;
|
|
68
|
+
continue;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const char *val = tf_batch_get_string(in, r, ci);
|
|
72
|
+
const char *p = val;
|
|
73
|
+
size_t row_outputs = 0;
|
|
74
|
+
|
|
75
|
+
while (*p) {
|
|
76
|
+
if (row_outputs >= st->max_tokens_per_row) {
|
|
77
|
+
explode_set_cap_error("max_tokens_per_row", st->max_tokens_per_row);
|
|
78
|
+
tf_batch_free(ob);
|
|
79
|
+
return TF_ERROR;
|
|
80
|
+
}
|
|
81
|
+
if (row_outputs >= st->max_output_rows_per_input_row) {
|
|
82
|
+
explode_set_cap_error("max_output_rows_per_input_row", st->max_output_rows_per_input_row);
|
|
83
|
+
tf_batch_free(ob);
|
|
84
|
+
return TF_ERROR;
|
|
85
|
+
}
|
|
86
|
+
if (out_row >= st->max_output_rows_per_batch) {
|
|
87
|
+
explode_set_cap_error("max_output_rows_per_batch", st->max_output_rows_per_batch);
|
|
88
|
+
tf_batch_free(ob);
|
|
89
|
+
return TF_ERROR;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const char *found = strstr(p, st->delimiter);
|
|
93
|
+
size_t tok_len = found ? (size_t)(found - p) : strlen(p);
|
|
94
|
+
if (tok_len > st->max_token_bytes) {
|
|
95
|
+
explode_set_cap_error("max_token_bytes", st->max_token_bytes);
|
|
96
|
+
tf_batch_free(ob);
|
|
97
|
+
return TF_ERROR;
|
|
98
|
+
}
|
|
99
|
+
if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
|
|
100
|
+
tf_batch_free(ob);
|
|
101
|
+
return TF_ERROR;
|
|
102
|
+
}
|
|
103
|
+
/* Override the exploded column */
|
|
104
|
+
size_t tok_cap = 0;
|
|
105
|
+
if (tf_size_add(tok_len, 1, &tok_cap) != TF_OK) {
|
|
106
|
+
tf_batch_free(ob);
|
|
107
|
+
return TF_ERROR;
|
|
108
|
+
}
|
|
109
|
+
char *tok = malloc(tok_cap);
|
|
110
|
+
if (!tok) {
|
|
111
|
+
tf_batch_free(ob);
|
|
112
|
+
return TF_ERROR;
|
|
113
|
+
}
|
|
114
|
+
memcpy(tok, p, tok_len);
|
|
115
|
+
tok[tok_len] = '\0';
|
|
116
|
+
/* Trim leading/trailing whitespace */
|
|
117
|
+
char *s = tok;
|
|
118
|
+
while (*s == ' ') s++;
|
|
119
|
+
char *e = s + strlen(s);
|
|
120
|
+
while (e > s && *(e - 1) == ' ') e--;
|
|
121
|
+
*e = '\0';
|
|
122
|
+
if (tf_batch_set_string(ob, out_row, (size_t)ci, s) != TF_OK) {
|
|
123
|
+
free(tok);
|
|
124
|
+
tf_batch_free(ob);
|
|
125
|
+
return TF_ERROR;
|
|
126
|
+
}
|
|
127
|
+
free(tok);
|
|
128
|
+
if (tf_batch_expose_row(ob, out_row) != TF_OK) {
|
|
129
|
+
tf_batch_free(ob);
|
|
130
|
+
return TF_ERROR;
|
|
131
|
+
}
|
|
132
|
+
out_row++;
|
|
133
|
+
row_outputs++;
|
|
134
|
+
|
|
135
|
+
if (found) p = found + delim_len;
|
|
136
|
+
else break;
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
if (out_row > 0) {
|
|
141
|
+
*out = ob;
|
|
142
|
+
} else {
|
|
143
|
+
tf_batch_free(ob);
|
|
144
|
+
}
|
|
145
|
+
return TF_OK;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
static int explode_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
149
|
+
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
static void explode_destroy(tf_step *self) {
|
|
153
|
+
explode_state *st = self->state;
|
|
154
|
+
if (st) { free(st->column); free(st->delimiter); free(st); }
|
|
155
|
+
free(self);
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
tf_step *tf_explode_create(const cJSON *args) {
|
|
159
|
+
if (!args) return NULL;
|
|
160
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
161
|
+
if (!cJSON_IsString(col_j)) return NULL;
|
|
162
|
+
|
|
163
|
+
explode_state *st = calloc(1, sizeof(explode_state));
|
|
164
|
+
if (!st) return NULL;
|
|
165
|
+
st->max_tokens_per_row = TF_MAX_EXPLODE_TOKENS_PER_ROW;
|
|
166
|
+
st->max_output_rows_per_input_row = TF_MAX_EXPANDING_ROWS_PER_INPUT;
|
|
167
|
+
st->max_output_rows_per_batch = TF_MAX_OUTPUT_ROWS_PER_BATCH;
|
|
168
|
+
st->max_token_bytes = TF_MAX_RECORD_BYTES;
|
|
169
|
+
|
|
170
|
+
size_t parsed_size = 0;
|
|
171
|
+
int has_size = tf_json_get_size_arg(args, "max_tokens_per_row", 1, TF_MAX_EXPLODE_TOKENS_PER_ROW, &parsed_size, "explode");
|
|
172
|
+
if (has_size < 0) goto fail;
|
|
173
|
+
if (has_size > 0) st->max_tokens_per_row = parsed_size;
|
|
174
|
+
has_size = tf_json_get_size_arg(args, "max_output_rows_per_input_row", 1, TF_MAX_EXPANDING_ROWS_PER_INPUT, &parsed_size, "explode");
|
|
175
|
+
if (has_size < 0) goto fail;
|
|
176
|
+
if (has_size > 0) st->max_output_rows_per_input_row = parsed_size;
|
|
177
|
+
has_size = tf_json_get_size_arg(args, "max_output_rows_per_batch", 1, TF_MAX_OUTPUT_ROWS_PER_BATCH, &parsed_size, "explode");
|
|
178
|
+
if (has_size < 0) goto fail;
|
|
179
|
+
if (has_size > 0) st->max_output_rows_per_batch = parsed_size;
|
|
180
|
+
has_size = tf_json_get_size_arg(args, "max_token_bytes", 0, TF_MAX_RECORD_BYTES, &parsed_size, "explode");
|
|
181
|
+
if (has_size < 0) goto fail;
|
|
182
|
+
if (has_size > 0) st->max_token_bytes = parsed_size;
|
|
183
|
+
|
|
184
|
+
st->column = strdup(col_j->valuestring);
|
|
185
|
+
if (!st->column) goto fail;
|
|
186
|
+
|
|
187
|
+
cJSON *delim_j = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
|
|
188
|
+
const char *delim = cJSON_IsString(delim_j) ? delim_j->valuestring : ",";
|
|
189
|
+
if (delim[0] == '\0') goto fail;
|
|
190
|
+
st->delimiter = strdup(delim);
|
|
191
|
+
if (!st->delimiter) goto fail;
|
|
192
|
+
|
|
193
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
194
|
+
if (!step) goto fail;
|
|
195
|
+
step->process = explode_process;
|
|
196
|
+
step->flush = explode_flush;
|
|
197
|
+
step->destroy = explode_destroy;
|
|
198
|
+
step->state = st;
|
|
199
|
+
return step;
|
|
200
|
+
|
|
201
|
+
fail:
|
|
202
|
+
free(st->column);
|
|
203
|
+
free(st->delimiter);
|
|
204
|
+
free(st);
|
|
205
|
+
return NULL;
|
|
206
|
+
}
|