tranfi 0.0.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -21
- package/NOTICE +8 -0
- package/README.md +627 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-BIAIKnrp.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +121 -0
- package/csrc/arena.c +93 -0
- package/csrc/batch.c +976 -0
- package/csrc/buffer.c +154 -0
- package/csrc/cJSON.c +3386 -0
- package/csrc/cJSON.h +316 -0
- package/csrc/codec_csv.c +1951 -0
- package/csrc/codec_jsonl.c +1086 -0
- package/csrc/codec_table.c +248 -0
- package/csrc/codec_text.c +447 -0
- package/csrc/compiler.c +130 -0
- package/csrc/config.h +21 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +5417 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1553 -0
- package/csrc/expr.h +58 -0
- package/csrc/internal.h +539 -0
- package/csrc/ir.c +166 -0
- package/csrc/ir.h +208 -0
- package/csrc/ir_schema.c +75 -0
- package/csrc/ir_serialize.c +166 -0
- package/csrc/ir_sql.c +1822 -0
- package/csrc/ir_validate.c +576 -0
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +1241 -0
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +283 -0
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +255 -0
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +248 -0
- package/csrc/op_cast.c +523 -0
- package/csrc/op_clip.c +99 -0
- package/csrc/op_date_trunc.c +355 -0
- package/csrc/op_datetime.c +394 -0
- package/csrc/op_derive.c +216 -0
- package/csrc/op_diff.c +250 -0
- package/csrc/op_ewma.c +222 -0
- package/csrc/op_explode.c +206 -0
- package/csrc/op_fill_down.c +235 -0
- package/csrc/op_fill_null.c +268 -0
- package/csrc/op_filter.c +181 -0
- package/csrc/op_frequency.c +721 -0
- package/csrc/op_grep.c +181 -0
- package/csrc/op_group_agg.c +1956 -0
- package/csrc/op_hash.c +159 -0
- package/csrc/op_head.c +84 -0
- package/csrc/op_interpolate.c +445 -0
- package/csrc/op_join.c +2902 -0
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +419 -0
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +242 -0
- package/csrc/op_normalize.c +510 -0
- package/csrc/op_onehot.c +457 -0
- package/csrc/op_pivot.c +1754 -0
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +3044 -0
- package/csrc/op_rename.c +129 -0
- package/csrc/op_replace.c +354 -0
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +158 -0
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +340 -0
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +95 -0
- package/csrc/op_sort.c +819 -0
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +151 -0
- package/csrc/op_split_data.c +119 -0
- package/csrc/op_stack.c +271 -0
- package/csrc/op_stats.c +875 -0
- package/csrc/op_step.c +333 -0
- package/csrc/op_tail.c +105 -0
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +357 -0
- package/csrc/op_trim.c +138 -0
- package/csrc/op_unique.c +1343 -0
- package/csrc/op_unpivot.c +193 -0
- package/csrc/op_validate.c +648 -0
- package/csrc/op_window.c +591 -0
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +1088 -0
- package/csrc/recipes.c +104 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +506 -0
- package/csrc/report.h +22 -0
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +291 -0
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +218 -0
- package/napi_api.c +534 -0
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +64 -59
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +190 -0
- package/src/engines/duckdb.js +142 -0
- package/src/index.js +925 -0
- package/src/memory_policy.js +411 -0
- package/src/native.js +18 -0
- package/src/pipeline.js +709 -0
- package/src/recipe_json.js +80 -0
- package/src/server.js +279 -0
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +21 -0
- package/wasm/index.js +732 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_datetime.c — Extract date/time components from date strings.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"column": "date", "extract": ["year", "month", "day"]}
|
|
5
|
+
* Supports: YYYY-MM-DD, YYYY-MM-DD HH:MM:SS, epoch seconds.
|
|
6
|
+
* Components: year, month, day, hour, minute, second, weekday, epoch
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
#include "internal.h"
|
|
10
|
+
#include "date_utils.h"
|
|
11
|
+
#include "cJSON.h"
|
|
12
|
+
#include <stdlib.h>
|
|
13
|
+
#include <string.h>
|
|
14
|
+
#include <stdio.h>
|
|
15
|
+
|
|
16
|
+
typedef enum {
|
|
17
|
+
DATETIME_MISSING_ERROR,
|
|
18
|
+
DATETIME_MISSING_NULL,
|
|
19
|
+
DATETIME_MISSING_IGNORE,
|
|
20
|
+
} datetime_missing_policy;
|
|
21
|
+
|
|
22
|
+
typedef enum {
|
|
23
|
+
DATETIME_TYPE_FAIL,
|
|
24
|
+
DATETIME_TYPE_NULL,
|
|
25
|
+
} datetime_type_policy;
|
|
26
|
+
|
|
27
|
+
typedef struct {
|
|
28
|
+
char *column;
|
|
29
|
+
int w_year, w_month, w_day, w_hour, w_minute, w_second, w_weekday, w_epoch;
|
|
30
|
+
datetime_missing_policy missing;
|
|
31
|
+
datetime_type_policy on_type_error;
|
|
32
|
+
} datetime_state;
|
|
33
|
+
|
|
34
|
+
/* Days in each month (non-leap) */
|
|
35
|
+
static int days_in_month(int m, int y) {
|
|
36
|
+
static const int d[] = {0,31,28,31,30,31,30,31,31,30,31,30,31};
|
|
37
|
+
if (m == 2 && ((y%4==0 && y%100!=0) || y%400==0)) return 29;
|
|
38
|
+
return (m >= 1 && m <= 12) ? d[m] : 30;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/* Zeller-like weekday: 0=Sunday..6=Saturday */
|
|
42
|
+
static int weekday(int y, int m, int d) {
|
|
43
|
+
/* Tomohiko Sakamoto's algorithm */
|
|
44
|
+
static int t[] = {0, 3, 2, 5, 0, 3, 5, 1, 4, 6, 2, 4};
|
|
45
|
+
if (m < 3) y--;
|
|
46
|
+
return (y + y/4 - y/100 + y/400 + t[m-1] + d) % 7;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/* Convert date to epoch (seconds since 1970-01-01) */
|
|
50
|
+
static int64_t date_to_epoch(int y, int mo, int d, int h, int mi, int s) {
|
|
51
|
+
/* Simplified: assume Gregorian, no timezone */
|
|
52
|
+
int64_t days = 0;
|
|
53
|
+
for (int yr = 1970; yr < y; yr++) {
|
|
54
|
+
days += 365 + ((yr%4==0 && yr%100!=0) || yr%400==0);
|
|
55
|
+
}
|
|
56
|
+
for (int m = 1; m < mo; m++) days += days_in_month(m, y);
|
|
57
|
+
days += d - 1;
|
|
58
|
+
return days * 86400 + h * 3600 + mi * 60 + s;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
static int parse_date(const char *s, int *y, int *mo, int *d, int *h, int *mi, int *se) {
|
|
62
|
+
*y = *mo = *d = *h = *mi = *se = 0;
|
|
63
|
+
|
|
64
|
+
/* Try epoch (pure number) */
|
|
65
|
+
char *end;
|
|
66
|
+
double epoch = strtod(s, &end);
|
|
67
|
+
if (*end == '\0' && end != s) {
|
|
68
|
+
/* Convert epoch to components */
|
|
69
|
+
int64_t ts = (int64_t)epoch;
|
|
70
|
+
*se = ts % 60; ts /= 60;
|
|
71
|
+
*mi = ts % 60; ts /= 60;
|
|
72
|
+
*h = ts % 24; ts /= 24;
|
|
73
|
+
/* Days since 1970-01-01 */
|
|
74
|
+
int64_t dd = ts;
|
|
75
|
+
*y = 1970;
|
|
76
|
+
while (1) {
|
|
77
|
+
int dy = 365 + ((*y%4==0 && *y%100!=0) || *y%400==0);
|
|
78
|
+
if (dd < dy) break;
|
|
79
|
+
dd -= dy;
|
|
80
|
+
(*y)++;
|
|
81
|
+
}
|
|
82
|
+
*mo = 1;
|
|
83
|
+
while (1) {
|
|
84
|
+
int dm = days_in_month(*mo, *y);
|
|
85
|
+
if (dd < dm) break;
|
|
86
|
+
dd -= dm;
|
|
87
|
+
(*mo)++;
|
|
88
|
+
}
|
|
89
|
+
*d = (int)dd + 1;
|
|
90
|
+
return 1;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/* Try YYYY-MM-DD [HH:MM:SS] */
|
|
94
|
+
int n = sscanf(s, "%d-%d-%d %d:%d:%d", y, mo, d, h, mi, se);
|
|
95
|
+
return n >= 3;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
static int datetime_is_temporal_type(tf_type type) {
|
|
99
|
+
return type == TF_TYPE_STRING || type == TF_TYPE_DATE || type == TF_TYPE_TIMESTAMP;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
static void datetime_set_col_error(const char *column, const char *suffix) {
|
|
103
|
+
char msg[512];
|
|
104
|
+
snprintf(msg, sizeof(msg), "datetime: column '%s' %s", column ? column : "", suffix);
|
|
105
|
+
tf_set_last_error(msg);
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
static int datetime_parse_missing_policy(const cJSON *args, datetime_missing_policy *out) {
|
|
109
|
+
*out = DATETIME_MISSING_ERROR;
|
|
110
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
|
|
111
|
+
if (!j) return TF_OK;
|
|
112
|
+
if (!cJSON_IsString(j)) {
|
|
113
|
+
tf_set_last_error("datetime: missing must be error, null, or ignore");
|
|
114
|
+
return TF_ERROR;
|
|
115
|
+
}
|
|
116
|
+
if (strcmp(j->valuestring, "error") == 0) *out = DATETIME_MISSING_ERROR;
|
|
117
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = DATETIME_MISSING_NULL;
|
|
118
|
+
else if (strcmp(j->valuestring, "ignore") == 0) *out = DATETIME_MISSING_IGNORE;
|
|
119
|
+
else {
|
|
120
|
+
tf_set_last_error("datetime: missing must be error, null, or ignore");
|
|
121
|
+
return TF_ERROR;
|
|
122
|
+
}
|
|
123
|
+
return TF_OK;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
static int datetime_parse_type_policy(const cJSON *args, datetime_type_policy *out) {
|
|
127
|
+
*out = DATETIME_TYPE_FAIL;
|
|
128
|
+
const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
|
|
129
|
+
if (!j) return TF_OK;
|
|
130
|
+
if (!cJSON_IsString(j)) {
|
|
131
|
+
tf_set_last_error("datetime: on_type_error must be fail or null");
|
|
132
|
+
return TF_ERROR;
|
|
133
|
+
}
|
|
134
|
+
if (strcmp(j->valuestring, "fail") == 0) *out = DATETIME_TYPE_FAIL;
|
|
135
|
+
else if (strcmp(j->valuestring, "null") == 0) *out = DATETIME_TYPE_NULL;
|
|
136
|
+
else {
|
|
137
|
+
tf_set_last_error("datetime: on_type_error must be fail or null");
|
|
138
|
+
return TF_ERROR;
|
|
139
|
+
}
|
|
140
|
+
return TF_OK;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
static int datetime_passthrough(tf_batch *in, tf_batch **out) {
|
|
144
|
+
tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
|
|
145
|
+
if (!ob) return TF_ERROR;
|
|
146
|
+
if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
147
|
+
tf_batch_free(ob);
|
|
148
|
+
return TF_ERROR;
|
|
149
|
+
}
|
|
150
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
151
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
152
|
+
tf_batch_free(ob);
|
|
153
|
+
return TF_ERROR;
|
|
154
|
+
}
|
|
155
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
156
|
+
tf_batch_free(ob);
|
|
157
|
+
return TF_ERROR;
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
*out = ob;
|
|
161
|
+
return TF_OK;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
static void datetime_free_extra_names(char **owned_names, size_t n_extra) {
|
|
165
|
+
for (size_t i = 0; i < n_extra; i++) free(owned_names[i]);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
static int datetime_add_extra_column(const datetime_state *st, const char *suffix,
|
|
169
|
+
const char **extra_names, char **owned_names,
|
|
170
|
+
tf_type *extra_types, size_t *n_extra) {
|
|
171
|
+
char *name = tf_string_append_suffix_checked(st->column, suffix);
|
|
172
|
+
if (!name) return TF_ERROR;
|
|
173
|
+
owned_names[*n_extra] = name;
|
|
174
|
+
extra_names[*n_extra] = name;
|
|
175
|
+
extra_types[*n_extra] = TF_TYPE_INT64;
|
|
176
|
+
(*n_extra)++;
|
|
177
|
+
return TF_OK;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
static int datetime_extra_columns(const datetime_state *st, const char **extra_names,
|
|
181
|
+
char **owned_names, tf_type *extra_types, size_t *n_extra) {
|
|
182
|
+
*n_extra = 0;
|
|
183
|
+
#define ADD_DATETIME_EXTRA(suffix) do { \
|
|
184
|
+
if (datetime_add_extra_column(st, suffix, extra_names, owned_names, \
|
|
185
|
+
extra_types, n_extra) != TF_OK) { \
|
|
186
|
+
datetime_free_extra_names(owned_names, *n_extra); \
|
|
187
|
+
return TF_ERROR; \
|
|
188
|
+
} \
|
|
189
|
+
} while (0)
|
|
190
|
+
if (st->w_year) ADD_DATETIME_EXTRA("_year");
|
|
191
|
+
if (st->w_month) ADD_DATETIME_EXTRA("_month");
|
|
192
|
+
if (st->w_day) ADD_DATETIME_EXTRA("_day");
|
|
193
|
+
if (st->w_hour) ADD_DATETIME_EXTRA("_hour");
|
|
194
|
+
if (st->w_minute) ADD_DATETIME_EXTRA("_minute");
|
|
195
|
+
if (st->w_second) ADD_DATETIME_EXTRA("_second");
|
|
196
|
+
if (st->w_weekday) ADD_DATETIME_EXTRA("_weekday");
|
|
197
|
+
if (st->w_epoch) ADD_DATETIME_EXTRA("_epoch");
|
|
198
|
+
#undef ADD_DATETIME_EXTRA
|
|
199
|
+
return TF_OK;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
static int datetime_set_extra_nulls(tf_batch *ob, size_t row, size_t start, size_t n_extra) {
|
|
203
|
+
for (size_t k = start; k < start + n_extra; k++) {
|
|
204
|
+
if (tf_batch_set_null(ob, row, k) != TF_OK) return TF_ERROR;
|
|
205
|
+
}
|
|
206
|
+
return TF_OK;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
static int datetime_apply_extracts(const datetime_state *st, tf_batch *ob, size_t row,
|
|
210
|
+
size_t *ei, int y, int mo, int d, int h, int mi, int se) {
|
|
211
|
+
if (st->w_year && tf_batch_set_int64(ob, row, (*ei)++, y) != TF_OK) return TF_ERROR;
|
|
212
|
+
if (st->w_month && tf_batch_set_int64(ob, row, (*ei)++, mo) != TF_OK) return TF_ERROR;
|
|
213
|
+
if (st->w_day && tf_batch_set_int64(ob, row, (*ei)++, d) != TF_OK) return TF_ERROR;
|
|
214
|
+
if (st->w_hour && tf_batch_set_int64(ob, row, (*ei)++, h) != TF_OK) return TF_ERROR;
|
|
215
|
+
if (st->w_minute && tf_batch_set_int64(ob, row, (*ei)++, mi) != TF_OK) return TF_ERROR;
|
|
216
|
+
if (st->w_second && tf_batch_set_int64(ob, row, (*ei)++, se) != TF_OK) return TF_ERROR;
|
|
217
|
+
if (st->w_weekday && tf_batch_set_int64(ob, row, (*ei)++, weekday(y, mo, d)) != TF_OK) return TF_ERROR;
|
|
218
|
+
if (st->w_epoch && tf_batch_set_int64(ob, row, (*ei)++, date_to_epoch(y, mo, d, h, mi, se)) != TF_OK) return TF_ERROR;
|
|
219
|
+
return TF_OK;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
static int datetime_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
223
|
+
tf_side_channels *side) {
|
|
224
|
+
(void)side;
|
|
225
|
+
datetime_state *st = self->state;
|
|
226
|
+
*out = NULL;
|
|
227
|
+
|
|
228
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
229
|
+
int force_null = 0;
|
|
230
|
+
if (ci < 0) {
|
|
231
|
+
if (st->missing == DATETIME_MISSING_ERROR) {
|
|
232
|
+
datetime_set_col_error(st->column, "not found");
|
|
233
|
+
return TF_ERROR;
|
|
234
|
+
}
|
|
235
|
+
if (st->missing == DATETIME_MISSING_IGNORE) return datetime_passthrough(in, out);
|
|
236
|
+
force_null = 1;
|
|
237
|
+
} else if (!datetime_is_temporal_type(in->col_types[ci])) {
|
|
238
|
+
if (st->on_type_error == DATETIME_TYPE_FAIL) {
|
|
239
|
+
datetime_set_col_error(st->column, "must be string, date, or timestamp");
|
|
240
|
+
return TF_ERROR;
|
|
241
|
+
}
|
|
242
|
+
force_null = 1;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
const char *extra_names[8];
|
|
246
|
+
char *owned_extra_names[8] = {0};
|
|
247
|
+
tf_type extra_types[8];
|
|
248
|
+
size_t n_extra = 0;
|
|
249
|
+
if (datetime_extra_columns(st, extra_names, owned_extra_names, extra_types, &n_extra) != TF_OK)
|
|
250
|
+
return TF_ERROR;
|
|
251
|
+
|
|
252
|
+
size_t out_cols = 0;
|
|
253
|
+
if (tf_size_add(in->n_cols, n_extra, &out_cols) != TF_OK) {
|
|
254
|
+
datetime_free_extra_names(owned_extra_names, n_extra);
|
|
255
|
+
return TF_ERROR;
|
|
256
|
+
}
|
|
257
|
+
tf_batch *ob = tf_batch_create(out_cols, in->n_rows);
|
|
258
|
+
if (!ob) {
|
|
259
|
+
datetime_free_extra_names(owned_extra_names, n_extra);
|
|
260
|
+
return TF_ERROR;
|
|
261
|
+
}
|
|
262
|
+
if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, n_extra) != TF_OK) {
|
|
263
|
+
datetime_free_extra_names(owned_extra_names, n_extra);
|
|
264
|
+
tf_batch_free(ob);
|
|
265
|
+
return TF_ERROR;
|
|
266
|
+
}
|
|
267
|
+
datetime_free_extra_names(owned_extra_names, n_extra);
|
|
268
|
+
|
|
269
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
270
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
271
|
+
tf_batch_free(ob);
|
|
272
|
+
return TF_ERROR;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
size_t ei = in->n_cols;
|
|
276
|
+
if (force_null || tf_batch_is_null(in, r, (size_t)ci)) {
|
|
277
|
+
if (datetime_set_extra_nulls(ob, r, in->n_cols, n_extra) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
278
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
int y = 0, mo = 0, d = 0, h = 0, mi = 0, se = 0;
|
|
283
|
+
int parsed = 0;
|
|
284
|
+
if (in->col_types[ci] == TF_TYPE_STRING) {
|
|
285
|
+
parsed = parse_date(tf_batch_get_string(in, r, ci), &y, &mo, &d, &h, &mi, &se);
|
|
286
|
+
} else if (in->col_types[ci] == TF_TYPE_DATE) {
|
|
287
|
+
tf_date_to_ymd(tf_batch_get_date(in, r, ci), &y, &mo, &d);
|
|
288
|
+
parsed = 1;
|
|
289
|
+
} else if (in->col_types[ci] == TF_TYPE_TIMESTAMP) {
|
|
290
|
+
int frac;
|
|
291
|
+
tf_timestamp_to_parts(tf_batch_get_timestamp(in, r, ci), &y, &mo, &d, &h, &mi, &se, &frac);
|
|
292
|
+
parsed = 1;
|
|
293
|
+
}
|
|
294
|
+
if (parsed) {
|
|
295
|
+
if (datetime_apply_extracts(st, ob, r, &ei, y, mo, d, h, mi, se) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
296
|
+
} else if (datetime_set_extra_nulls(ob, r, in->n_cols, n_extra) != TF_OK) {
|
|
297
|
+
tf_batch_free(ob);
|
|
298
|
+
return TF_ERROR;
|
|
299
|
+
}
|
|
300
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
301
|
+
tf_batch_free(ob);
|
|
302
|
+
return TF_ERROR;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
*out = ob;
|
|
307
|
+
return TF_OK;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
static int datetime_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
311
|
+
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
static void datetime_state_free(datetime_state *st) {
|
|
315
|
+
if (!st) return;
|
|
316
|
+
free(st->column);
|
|
317
|
+
free(st);
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
static void datetime_destroy(tf_step *self) {
|
|
321
|
+
if (self) datetime_state_free(self->state);
|
|
322
|
+
free(self);
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
static int datetime_enable_component(datetime_state *st, const char *s) {
|
|
326
|
+
if (strcmp(s, "year") == 0) st->w_year = 1;
|
|
327
|
+
else if (strcmp(s, "month") == 0) st->w_month = 1;
|
|
328
|
+
else if (strcmp(s, "day") == 0) st->w_day = 1;
|
|
329
|
+
else if (strcmp(s, "hour") == 0) st->w_hour = 1;
|
|
330
|
+
else if (strcmp(s, "minute") == 0) st->w_minute = 1;
|
|
331
|
+
else if (strcmp(s, "second") == 0) st->w_second = 1;
|
|
332
|
+
else if (strcmp(s, "weekday") == 0) st->w_weekday = 1;
|
|
333
|
+
else if (strcmp(s, "epoch") == 0) st->w_epoch = 1;
|
|
334
|
+
else return TF_ERROR;
|
|
335
|
+
return TF_OK;
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
static int datetime_has_component(const datetime_state *st) {
|
|
339
|
+
return st->w_year || st->w_month || st->w_day || st->w_hour || st->w_minute ||
|
|
340
|
+
st->w_second || st->w_weekday || st->w_epoch;
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
tf_step *tf_datetime_create(const cJSON *args) {
|
|
344
|
+
if (!args) return NULL;
|
|
345
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
346
|
+
if (!cJSON_IsString(col_j) || !col_j->valuestring[0]) {
|
|
347
|
+
tf_set_last_error("datetime: column is required");
|
|
348
|
+
return NULL;
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
datetime_state *st = tf_callocarray_checked(1, sizeof(datetime_state));
|
|
352
|
+
if (!st) return NULL;
|
|
353
|
+
st->column = tf_strdup_checked(col_j->valuestring);
|
|
354
|
+
if (!st->column) { free(st); return NULL; }
|
|
355
|
+
|
|
356
|
+
cJSON *extract = cJSON_GetObjectItemCaseSensitive(args, "extract");
|
|
357
|
+
if (extract && cJSON_IsArray(extract)) {
|
|
358
|
+
int n = cJSON_GetArraySize(extract);
|
|
359
|
+
for (int i = 0; i < n; i++) {
|
|
360
|
+
cJSON *item = cJSON_GetArrayItem(extract, i);
|
|
361
|
+
if (!cJSON_IsString(item) || !item->valuestring[0] ||
|
|
362
|
+
datetime_enable_component(st, item->valuestring) != TF_OK) {
|
|
363
|
+
tf_set_last_error("datetime: extract must contain year, month, day, hour, minute, second, weekday, or epoch");
|
|
364
|
+
datetime_state_free(st);
|
|
365
|
+
return NULL;
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
if (!datetime_has_component(st)) {
|
|
369
|
+
tf_set_last_error("datetime: extract must contain at least one component");
|
|
370
|
+
datetime_state_free(st);
|
|
371
|
+
return NULL;
|
|
372
|
+
}
|
|
373
|
+
} else {
|
|
374
|
+
/* Default: all */
|
|
375
|
+
st->w_year = st->w_month = st->w_day = 1;
|
|
376
|
+
st->w_hour = st->w_minute = st->w_second = 1;
|
|
377
|
+
st->w_weekday = st->w_epoch = 1;
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
if (datetime_parse_missing_policy(args, &st->missing) != TF_OK ||
|
|
381
|
+
datetime_parse_type_policy(args, &st->on_type_error) != TF_OK) {
|
|
382
|
+
free(st->column);
|
|
383
|
+
free(st);
|
|
384
|
+
return NULL;
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
|
|
388
|
+
if (!step) { free(st->column); free(st); return NULL; }
|
|
389
|
+
step->process = datetime_process;
|
|
390
|
+
step->flush = datetime_flush;
|
|
391
|
+
step->destroy = datetime_destroy;
|
|
392
|
+
step->state = st;
|
|
393
|
+
return step;
|
|
394
|
+
}
|
package/csrc/op_derive.c
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_derive.c — Add computed columns using arithmetic expressions.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"columns": [{"name": "total", "expr": "col(price)*col(qty)"}]}
|
|
5
|
+
* For each row, evaluates each expression and appends the result as a new column.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "internal.h"
|
|
9
|
+
#include "cJSON.h"
|
|
10
|
+
#include "date_utils.h"
|
|
11
|
+
#include <stdlib.h>
|
|
12
|
+
#include <string.h>
|
|
13
|
+
#include <stdio.h>
|
|
14
|
+
#include <limits.h>
|
|
15
|
+
#include <math.h>
|
|
16
|
+
|
|
17
|
+
typedef struct {
|
|
18
|
+
char *name;
|
|
19
|
+
tf_expr *expr;
|
|
20
|
+
} derive_col;
|
|
21
|
+
|
|
22
|
+
typedef struct {
|
|
23
|
+
derive_col *cols;
|
|
24
|
+
size_t n_cols;
|
|
25
|
+
int types_resolved; /* have we determined output types? */
|
|
26
|
+
tf_type *col_types; /* resolved type per derived column */
|
|
27
|
+
} derive_state;
|
|
28
|
+
|
|
29
|
+
/* Evaluate first row to determine derived column types. */
|
|
30
|
+
static int resolve_types(derive_state *st, const tf_batch *in) {
|
|
31
|
+
st->col_types = calloc(st->n_cols ? st->n_cols : 1, sizeof(tf_type));
|
|
32
|
+
if (!st->col_types) return TF_ERROR;
|
|
33
|
+
|
|
34
|
+
for (size_t d = 0; d < st->n_cols; d++) {
|
|
35
|
+
if (in->n_rows == 0) {
|
|
36
|
+
st->col_types[d] = TF_TYPE_FLOAT64;
|
|
37
|
+
continue;
|
|
38
|
+
}
|
|
39
|
+
tf_eval_result val;
|
|
40
|
+
if (tf_expr_eval_val(st->cols[d].expr, in, 0, &val) != TF_OK) {
|
|
41
|
+
st->col_types[d] = TF_TYPE_FLOAT64;
|
|
42
|
+
continue;
|
|
43
|
+
}
|
|
44
|
+
switch (val.type) {
|
|
45
|
+
case TF_TYPE_INT64: st->col_types[d] = TF_TYPE_INT64; break;
|
|
46
|
+
case TF_TYPE_FLOAT64: st->col_types[d] = TF_TYPE_FLOAT64; break;
|
|
47
|
+
case TF_TYPE_STRING: st->col_types[d] = TF_TYPE_STRING; break;
|
|
48
|
+
case TF_TYPE_BOOL: st->col_types[d] = TF_TYPE_BOOL; break;
|
|
49
|
+
case TF_TYPE_DATE: st->col_types[d] = TF_TYPE_DATE; break;
|
|
50
|
+
case TF_TYPE_TIMESTAMP: st->col_types[d] = TF_TYPE_TIMESTAMP; break;
|
|
51
|
+
default: st->col_types[d] = TF_TYPE_FLOAT64; break;
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
st->types_resolved = 1;
|
|
55
|
+
return TF_OK;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
static int derive_double_to_i64(double v, int64_t *out) {
|
|
59
|
+
if (!isfinite(v) || v < (double)INT64_MIN || v > (double)INT64_MAX) return TF_ERROR;
|
|
60
|
+
*out = (int64_t)v;
|
|
61
|
+
return TF_OK;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
static int set_derived_value(tf_batch *ob, size_t row, size_t col,
|
|
65
|
+
tf_type col_type, const tf_eval_result *val) {
|
|
66
|
+
if (!val || val->type == TF_TYPE_NULL) return tf_batch_set_null(ob, row, col);
|
|
67
|
+
|
|
68
|
+
switch (col_type) {
|
|
69
|
+
case TF_TYPE_INT64:
|
|
70
|
+
if (val->type == TF_TYPE_INT64) return tf_batch_set_int64(ob, row, col, val->i);
|
|
71
|
+
if (val->type == TF_TYPE_FLOAT64) {
|
|
72
|
+
int64_t iv = 0;
|
|
73
|
+
if (derive_double_to_i64(val->f, &iv) != TF_OK) return tf_batch_set_null(ob, row, col);
|
|
74
|
+
return tf_batch_set_int64(ob, row, col, iv);
|
|
75
|
+
}
|
|
76
|
+
return tf_batch_set_null(ob, row, col);
|
|
77
|
+
case TF_TYPE_FLOAT64:
|
|
78
|
+
if (val->type == TF_TYPE_FLOAT64) return tf_batch_set_float64(ob, row, col, val->f);
|
|
79
|
+
if (val->type == TF_TYPE_INT64) return tf_batch_set_float64(ob, row, col, (double)val->i);
|
|
80
|
+
return tf_batch_set_null(ob, row, col);
|
|
81
|
+
case TF_TYPE_STRING:
|
|
82
|
+
if (val->type == TF_TYPE_STRING) {
|
|
83
|
+
if (!val->s) return tf_batch_set_null(ob, row, col);
|
|
84
|
+
return tf_batch_set_string(ob, row, col, val->s);
|
|
85
|
+
}
|
|
86
|
+
return tf_batch_set_null(ob, row, col);
|
|
87
|
+
case TF_TYPE_BOOL:
|
|
88
|
+
if (val->type == TF_TYPE_BOOL) return tf_batch_set_bool(ob, row, col, val->b);
|
|
89
|
+
return tf_batch_set_null(ob, row, col);
|
|
90
|
+
case TF_TYPE_DATE:
|
|
91
|
+
if (val->type == TF_TYPE_DATE) return tf_batch_set_date(ob, row, col, val->date);
|
|
92
|
+
return tf_batch_set_null(ob, row, col);
|
|
93
|
+
case TF_TYPE_TIMESTAMP:
|
|
94
|
+
if (val->type == TF_TYPE_TIMESTAMP) return tf_batch_set_timestamp(ob, row, col, val->i);
|
|
95
|
+
return tf_batch_set_null(ob, row, col);
|
|
96
|
+
default:
|
|
97
|
+
return tf_batch_set_null(ob, row, col);
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
static int derive_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
102
|
+
tf_side_channels *side) {
|
|
103
|
+
(void)side;
|
|
104
|
+
derive_state *st = self->state;
|
|
105
|
+
*out = NULL;
|
|
106
|
+
|
|
107
|
+
if (!st->types_resolved && resolve_types(st, in) != TF_OK) return TF_ERROR;
|
|
108
|
+
|
|
109
|
+
size_t out_n_cols = 0;
|
|
110
|
+
if (tf_size_add(in->n_cols, st->n_cols, &out_n_cols) != TF_OK) return TF_ERROR;
|
|
111
|
+
|
|
112
|
+
tf_batch *ob = tf_batch_create(out_n_cols, in->n_rows > 0 ? in->n_rows : 1);
|
|
113
|
+
if (!ob) return TF_ERROR;
|
|
114
|
+
|
|
115
|
+
const char **extra_names = calloc(st->n_cols ? st->n_cols : 1, sizeof(char *));
|
|
116
|
+
if (!extra_names) {
|
|
117
|
+
tf_batch_free(ob);
|
|
118
|
+
return TF_ERROR;
|
|
119
|
+
}
|
|
120
|
+
for (size_t d = 0; d < st->n_cols; d++) extra_names[d] = st->cols[d].name;
|
|
121
|
+
if (tf_batch_clone_with_extra_cols(ob, in, extra_names, st->col_types, st->n_cols) != TF_OK) {
|
|
122
|
+
free(extra_names);
|
|
123
|
+
tf_batch_free(ob);
|
|
124
|
+
return TF_ERROR;
|
|
125
|
+
}
|
|
126
|
+
free(extra_names);
|
|
127
|
+
|
|
128
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
129
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) goto fail;
|
|
130
|
+
|
|
131
|
+
for (size_t d = 0; d < st->n_cols; d++) {
|
|
132
|
+
size_t col_idx = in->n_cols + d;
|
|
133
|
+
tf_eval_result val;
|
|
134
|
+
if (tf_expr_eval_val(st->cols[d].expr, in, r, &val) != TF_OK) {
|
|
135
|
+
if (tf_batch_set_null(ob, r, col_idx) != TF_OK) goto fail;
|
|
136
|
+
continue;
|
|
137
|
+
}
|
|
138
|
+
if (set_derived_value(ob, r, col_idx, st->col_types[d], &val) != TF_OK) goto fail;
|
|
139
|
+
}
|
|
140
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) goto fail;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
if (ob->n_rows > 0) {
|
|
144
|
+
*out = ob;
|
|
145
|
+
} else {
|
|
146
|
+
tf_batch_free(ob);
|
|
147
|
+
}
|
|
148
|
+
return TF_OK;
|
|
149
|
+
|
|
150
|
+
fail:
|
|
151
|
+
tf_batch_free(ob);
|
|
152
|
+
return TF_ERROR;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
static int derive_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
156
|
+
(void)self; (void)side;
|
|
157
|
+
*out = NULL;
|
|
158
|
+
return TF_OK;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
static void derive_destroy(tf_step *self) {
|
|
162
|
+
derive_state *st = self->state;
|
|
163
|
+
if (st) {
|
|
164
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
165
|
+
free(st->cols[i].name);
|
|
166
|
+
tf_expr_free(st->cols[i].expr);
|
|
167
|
+
}
|
|
168
|
+
free(st->cols);
|
|
169
|
+
free(st->col_types);
|
|
170
|
+
free(st);
|
|
171
|
+
}
|
|
172
|
+
free(self);
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
tf_step *tf_derive_create(const cJSON *args) {
|
|
176
|
+
if (!args) return NULL;
|
|
177
|
+
cJSON *columns = cJSON_GetObjectItemCaseSensitive(args, "columns");
|
|
178
|
+
if (!columns || !cJSON_IsArray(columns)) return NULL;
|
|
179
|
+
|
|
180
|
+
int n = cJSON_GetArraySize(columns);
|
|
181
|
+
if (n <= 0) return NULL;
|
|
182
|
+
|
|
183
|
+
derive_state *st = calloc(1, sizeof(derive_state));
|
|
184
|
+
if (!st) return NULL;
|
|
185
|
+
st->cols = calloc(n, sizeof(derive_col));
|
|
186
|
+
if (!st->cols) { free(st); return NULL; }
|
|
187
|
+
st->n_cols = n;
|
|
188
|
+
|
|
189
|
+
for (int i = 0; i < n; i++) {
|
|
190
|
+
cJSON *item = cJSON_GetArrayItem(columns, i);
|
|
191
|
+
cJSON *name_j = cJSON_GetObjectItemCaseSensitive(item, "name");
|
|
192
|
+
cJSON *expr_j = cJSON_GetObjectItemCaseSensitive(item, "expr");
|
|
193
|
+
if (!cJSON_IsString(name_j) || !cJSON_IsString(expr_j)) goto fail;
|
|
194
|
+
|
|
195
|
+
st->cols[i].name = strdup(name_j->valuestring);
|
|
196
|
+
st->cols[i].expr = tf_expr_parse(expr_j->valuestring);
|
|
197
|
+
if (!st->cols[i].name || !st->cols[i].expr) goto fail;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
201
|
+
if (!step) goto fail;
|
|
202
|
+
step->process = derive_process;
|
|
203
|
+
step->flush = derive_flush;
|
|
204
|
+
step->destroy = derive_destroy;
|
|
205
|
+
step->state = st;
|
|
206
|
+
return step;
|
|
207
|
+
|
|
208
|
+
fail:
|
|
209
|
+
for (int i = 0; i < n; i++) {
|
|
210
|
+
free(st->cols[i].name);
|
|
211
|
+
tf_expr_free(st->cols[i].expr);
|
|
212
|
+
}
|
|
213
|
+
free(st->cols);
|
|
214
|
+
free(st);
|
|
215
|
+
return NULL;
|
|
216
|
+
}
|