tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
@@ -13,9 +13,22 @@
13
13
  #include <string.h>
14
14
  #include <stdio.h>
15
15
 
16
+ typedef enum {
17
+ DATETIME_MISSING_ERROR,
18
+ DATETIME_MISSING_NULL,
19
+ DATETIME_MISSING_IGNORE,
20
+ } datetime_missing_policy;
21
+
22
+ typedef enum {
23
+ DATETIME_TYPE_FAIL,
24
+ DATETIME_TYPE_NULL,
25
+ } datetime_type_policy;
26
+
16
27
  typedef struct {
17
28
  char *column;
18
29
  int w_year, w_month, w_day, w_hour, w_minute, w_second, w_weekday, w_epoch;
30
+ datetime_missing_policy missing;
31
+ datetime_type_policy on_type_error;
19
32
  } datetime_state;
20
33
 
21
34
  /* Days in each month (non-leap) */
@@ -82,78 +95,212 @@ static int parse_date(const char *s, int *y, int *mo, int *d, int *h, int *mi, i
82
95
  return n >= 3;
83
96
  }
84
97
 
98
+ static int datetime_is_temporal_type(tf_type type) {
99
+ return type == TF_TYPE_STRING || type == TF_TYPE_DATE || type == TF_TYPE_TIMESTAMP;
100
+ }
101
+
102
+ static void datetime_set_col_error(const char *column, const char *suffix) {
103
+ char msg[512];
104
+ snprintf(msg, sizeof(msg), "datetime: column '%s' %s", column ? column : "", suffix);
105
+ tf_set_last_error(msg);
106
+ }
107
+
108
+ static int datetime_parse_missing_policy(const cJSON *args, datetime_missing_policy *out) {
109
+ *out = DATETIME_MISSING_ERROR;
110
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
111
+ if (!j) return TF_OK;
112
+ if (!cJSON_IsString(j)) {
113
+ tf_set_last_error("datetime: missing must be error, null, or ignore");
114
+ return TF_ERROR;
115
+ }
116
+ if (strcmp(j->valuestring, "error") == 0) *out = DATETIME_MISSING_ERROR;
117
+ else if (strcmp(j->valuestring, "null") == 0) *out = DATETIME_MISSING_NULL;
118
+ else if (strcmp(j->valuestring, "ignore") == 0) *out = DATETIME_MISSING_IGNORE;
119
+ else {
120
+ tf_set_last_error("datetime: missing must be error, null, or ignore");
121
+ return TF_ERROR;
122
+ }
123
+ return TF_OK;
124
+ }
125
+
126
+ static int datetime_parse_type_policy(const cJSON *args, datetime_type_policy *out) {
127
+ *out = DATETIME_TYPE_FAIL;
128
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
129
+ if (!j) return TF_OK;
130
+ if (!cJSON_IsString(j)) {
131
+ tf_set_last_error("datetime: on_type_error must be fail or null");
132
+ return TF_ERROR;
133
+ }
134
+ if (strcmp(j->valuestring, "fail") == 0) *out = DATETIME_TYPE_FAIL;
135
+ else if (strcmp(j->valuestring, "null") == 0) *out = DATETIME_TYPE_NULL;
136
+ else {
137
+ tf_set_last_error("datetime: on_type_error must be fail or null");
138
+ return TF_ERROR;
139
+ }
140
+ return TF_OK;
141
+ }
142
+
143
+ static int datetime_passthrough(tf_batch *in, tf_batch **out) {
144
+ tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
145
+ if (!ob) return TF_ERROR;
146
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
147
+ tf_batch_free(ob);
148
+ return TF_ERROR;
149
+ }
150
+ for (size_t r = 0; r < in->n_rows; r++) {
151
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
152
+ tf_batch_free(ob);
153
+ return TF_ERROR;
154
+ }
155
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
156
+ tf_batch_free(ob);
157
+ return TF_ERROR;
158
+ }
159
+ }
160
+ *out = ob;
161
+ return TF_OK;
162
+ }
163
+
164
+ static void datetime_free_extra_names(char **owned_names, size_t n_extra) {
165
+ for (size_t i = 0; i < n_extra; i++) free(owned_names[i]);
166
+ }
167
+
168
+ static int datetime_add_extra_column(const datetime_state *st, const char *suffix,
169
+ const char **extra_names, char **owned_names,
170
+ tf_type *extra_types, size_t *n_extra) {
171
+ char *name = tf_string_append_suffix_checked(st->column, suffix);
172
+ if (!name) return TF_ERROR;
173
+ owned_names[*n_extra] = name;
174
+ extra_names[*n_extra] = name;
175
+ extra_types[*n_extra] = TF_TYPE_INT64;
176
+ (*n_extra)++;
177
+ return TF_OK;
178
+ }
179
+
180
+ static int datetime_extra_columns(const datetime_state *st, const char **extra_names,
181
+ char **owned_names, tf_type *extra_types, size_t *n_extra) {
182
+ *n_extra = 0;
183
+ #define ADD_DATETIME_EXTRA(suffix) do { \
184
+ if (datetime_add_extra_column(st, suffix, extra_names, owned_names, \
185
+ extra_types, n_extra) != TF_OK) { \
186
+ datetime_free_extra_names(owned_names, *n_extra); \
187
+ return TF_ERROR; \
188
+ } \
189
+ } while (0)
190
+ if (st->w_year) ADD_DATETIME_EXTRA("_year");
191
+ if (st->w_month) ADD_DATETIME_EXTRA("_month");
192
+ if (st->w_day) ADD_DATETIME_EXTRA("_day");
193
+ if (st->w_hour) ADD_DATETIME_EXTRA("_hour");
194
+ if (st->w_minute) ADD_DATETIME_EXTRA("_minute");
195
+ if (st->w_second) ADD_DATETIME_EXTRA("_second");
196
+ if (st->w_weekday) ADD_DATETIME_EXTRA("_weekday");
197
+ if (st->w_epoch) ADD_DATETIME_EXTRA("_epoch");
198
+ #undef ADD_DATETIME_EXTRA
199
+ return TF_OK;
200
+ }
201
+
202
+ static int datetime_set_extra_nulls(tf_batch *ob, size_t row, size_t start, size_t n_extra) {
203
+ for (size_t k = start; k < start + n_extra; k++) {
204
+ if (tf_batch_set_null(ob, row, k) != TF_OK) return TF_ERROR;
205
+ }
206
+ return TF_OK;
207
+ }
208
+
209
+ static int datetime_apply_extracts(const datetime_state *st, tf_batch *ob, size_t row,
210
+ size_t *ei, int y, int mo, int d, int h, int mi, int se) {
211
+ if (st->w_year && tf_batch_set_int64(ob, row, (*ei)++, y) != TF_OK) return TF_ERROR;
212
+ if (st->w_month && tf_batch_set_int64(ob, row, (*ei)++, mo) != TF_OK) return TF_ERROR;
213
+ if (st->w_day && tf_batch_set_int64(ob, row, (*ei)++, d) != TF_OK) return TF_ERROR;
214
+ if (st->w_hour && tf_batch_set_int64(ob, row, (*ei)++, h) != TF_OK) return TF_ERROR;
215
+ if (st->w_minute && tf_batch_set_int64(ob, row, (*ei)++, mi) != TF_OK) return TF_ERROR;
216
+ if (st->w_second && tf_batch_set_int64(ob, row, (*ei)++, se) != TF_OK) return TF_ERROR;
217
+ if (st->w_weekday && tf_batch_set_int64(ob, row, (*ei)++, weekday(y, mo, d)) != TF_OK) return TF_ERROR;
218
+ if (st->w_epoch && tf_batch_set_int64(ob, row, (*ei)++, date_to_epoch(y, mo, d, h, mi, se)) != TF_OK) return TF_ERROR;
219
+ return TF_OK;
220
+ }
221
+
85
222
  static int datetime_process(tf_step *self, tf_batch *in, tf_batch **out,
86
223
  tf_side_channels *side) {
87
224
  (void)side;
88
225
  datetime_state *st = self->state;
89
226
  *out = NULL;
90
227
 
91
- /* Count extra columns */
228
+ int ci = tf_batch_col_index(in, st->column);
229
+ int force_null = 0;
230
+ if (ci < 0) {
231
+ if (st->missing == DATETIME_MISSING_ERROR) {
232
+ datetime_set_col_error(st->column, "not found");
233
+ return TF_ERROR;
234
+ }
235
+ if (st->missing == DATETIME_MISSING_IGNORE) return datetime_passthrough(in, out);
236
+ force_null = 1;
237
+ } else if (!datetime_is_temporal_type(in->col_types[ci])) {
238
+ if (st->on_type_error == DATETIME_TYPE_FAIL) {
239
+ datetime_set_col_error(st->column, "must be string, date, or timestamp");
240
+ return TF_ERROR;
241
+ }
242
+ force_null = 1;
243
+ }
244
+
245
+ const char *extra_names[8];
246
+ char *owned_extra_names[8] = {0};
247
+ tf_type extra_types[8];
92
248
  size_t n_extra = 0;
93
- if (st->w_year) n_extra++;
94
- if (st->w_month) n_extra++;
95
- if (st->w_day) n_extra++;
96
- if (st->w_hour) n_extra++;
97
- if (st->w_minute) n_extra++;
98
- if (st->w_second) n_extra++;
99
- if (st->w_weekday) n_extra++;
100
- if (st->w_epoch) n_extra++;
101
-
102
- tf_batch *ob = tf_batch_create(in->n_cols + n_extra, in->n_rows);
103
- if (!ob) return TF_ERROR;
104
- for (size_t c = 0; c < in->n_cols; c++)
105
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
106
-
107
- size_t ei = in->n_cols;
108
- char col_prefix[256];
109
- snprintf(col_prefix, sizeof(col_prefix), "%s_", st->column);
110
- char col_name[256];
111
- if (st->w_year) { snprintf(col_name, sizeof(col_name), "%syear", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
112
- if (st->w_month) { snprintf(col_name, sizeof(col_name), "%smonth", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
113
- if (st->w_day) { snprintf(col_name, sizeof(col_name), "%sday", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
114
- if (st->w_hour) { snprintf(col_name, sizeof(col_name), "%shour", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
115
- if (st->w_minute) { snprintf(col_name, sizeof(col_name), "%sminute", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
116
- if (st->w_second) { snprintf(col_name, sizeof(col_name), "%ssecond", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
117
- if (st->w_weekday) { snprintf(col_name, sizeof(col_name), "%sweekday", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
118
- if (st->w_epoch) { snprintf(col_name, sizeof(col_name), "%sepoch", col_prefix); tf_batch_set_schema(ob, ei++, col_name, TF_TYPE_INT64); }
249
+ if (datetime_extra_columns(st, extra_names, owned_extra_names, extra_types, &n_extra) != TF_OK)
250
+ return TF_ERROR;
119
251
 
120
- int ci = tf_batch_col_index(in, st->column);
252
+ size_t out_cols = 0;
253
+ if (tf_size_add(in->n_cols, n_extra, &out_cols) != TF_OK) {
254
+ datetime_free_extra_names(owned_extra_names, n_extra);
255
+ return TF_ERROR;
256
+ }
257
+ tf_batch *ob = tf_batch_create(out_cols, in->n_rows);
258
+ if (!ob) {
259
+ datetime_free_extra_names(owned_extra_names, n_extra);
260
+ return TF_ERROR;
261
+ }
262
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, n_extra) != TF_OK) {
263
+ datetime_free_extra_names(owned_extra_names, n_extra);
264
+ tf_batch_free(ob);
265
+ return TF_ERROR;
266
+ }
267
+ datetime_free_extra_names(owned_extra_names, n_extra);
121
268
 
122
269
  for (size_t r = 0; r < in->n_rows; r++) {
123
- tf_batch_copy_row(ob, r, in, r);
124
-
125
- ei = in->n_cols;
126
- if (ci >= 0 && !tf_batch_is_null(in, r, ci)) {
127
- int y = 0, mo = 0, d = 0, h = 0, mi = 0, se = 0;
128
- int parsed = 0;
129
- if (in->col_types[ci] == TF_TYPE_STRING) {
130
- parsed = parse_date(tf_batch_get_string(in, r, ci), &y, &mo, &d, &h, &mi, &se);
131
- } else if (in->col_types[ci] == TF_TYPE_DATE) {
132
- tf_date_to_ymd(tf_batch_get_date(in, r, ci), &y, &mo, &d);
133
- parsed = 1;
134
- } else if (in->col_types[ci] == TF_TYPE_TIMESTAMP) {
135
- int frac;
136
- tf_timestamp_to_parts(tf_batch_get_timestamp(in, r, ci), &y, &mo, &d, &h, &mi, &se, &frac);
137
- parsed = 1;
138
- }
139
- if (parsed) {
140
- if (st->w_year) tf_batch_set_int64(ob, r, ei++, y);
141
- if (st->w_month) tf_batch_set_int64(ob, r, ei++, mo);
142
- if (st->w_day) tf_batch_set_int64(ob, r, ei++, d);
143
- if (st->w_hour) tf_batch_set_int64(ob, r, ei++, h);
144
- if (st->w_minute) tf_batch_set_int64(ob, r, ei++, mi);
145
- if (st->w_second) tf_batch_set_int64(ob, r, ei++, se);
146
- if (st->w_weekday) tf_batch_set_int64(ob, r, ei++, weekday(y, mo, d));
147
- if (st->w_epoch) tf_batch_set_int64(ob, r, ei++, date_to_epoch(y, mo, d, h, mi, se));
148
- } else {
149
- for (size_t k = in->n_cols; k < in->n_cols + n_extra; k++)
150
- tf_batch_set_null(ob, r, k);
151
- }
152
- } else {
153
- for (size_t k = in->n_cols; k < in->n_cols + n_extra; k++)
154
- tf_batch_set_null(ob, r, k);
270
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
271
+ tf_batch_free(ob);
272
+ return TF_ERROR;
273
+ }
274
+
275
+ size_t ei = in->n_cols;
276
+ if (force_null || tf_batch_is_null(in, r, (size_t)ci)) {
277
+ if (datetime_set_extra_nulls(ob, r, in->n_cols, n_extra) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
278
+ if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
279
+ continue;
280
+ }
281
+
282
+ int y = 0, mo = 0, d = 0, h = 0, mi = 0, se = 0;
283
+ int parsed = 0;
284
+ if (in->col_types[ci] == TF_TYPE_STRING) {
285
+ parsed = parse_date(tf_batch_get_string(in, r, ci), &y, &mo, &d, &h, &mi, &se);
286
+ } else if (in->col_types[ci] == TF_TYPE_DATE) {
287
+ tf_date_to_ymd(tf_batch_get_date(in, r, ci), &y, &mo, &d);
288
+ parsed = 1;
289
+ } else if (in->col_types[ci] == TF_TYPE_TIMESTAMP) {
290
+ int frac;
291
+ tf_timestamp_to_parts(tf_batch_get_timestamp(in, r, ci), &y, &mo, &d, &h, &mi, &se, &frac);
292
+ parsed = 1;
293
+ }
294
+ if (parsed) {
295
+ if (datetime_apply_extracts(st, ob, r, &ei, y, mo, d, h, mi, se) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
296
+ } else if (datetime_set_extra_nulls(ob, r, in->n_cols, n_extra) != TF_OK) {
297
+ tf_batch_free(ob);
298
+ return TF_ERROR;
299
+ }
300
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
301
+ tf_batch_free(ob);
302
+ return TF_ERROR;
155
303
  }
156
- ob->n_rows = r + 1;
157
304
  }
158
305
 
159
306
  *out = ob;
@@ -164,36 +311,64 @@ static int datetime_flush(tf_step *self, tf_batch **out, tf_side_channels *side)
164
311
  (void)self; (void)side; *out = NULL; return TF_OK;
165
312
  }
166
313
 
314
+ static void datetime_state_free(datetime_state *st) {
315
+ if (!st) return;
316
+ free(st->column);
317
+ free(st);
318
+ }
319
+
167
320
  static void datetime_destroy(tf_step *self) {
168
- datetime_state *st = self->state;
169
- if (st) { free(st->column); free(st); }
321
+ if (self) datetime_state_free(self->state);
170
322
  free(self);
171
323
  }
172
324
 
325
+ static int datetime_enable_component(datetime_state *st, const char *s) {
326
+ if (strcmp(s, "year") == 0) st->w_year = 1;
327
+ else if (strcmp(s, "month") == 0) st->w_month = 1;
328
+ else if (strcmp(s, "day") == 0) st->w_day = 1;
329
+ else if (strcmp(s, "hour") == 0) st->w_hour = 1;
330
+ else if (strcmp(s, "minute") == 0) st->w_minute = 1;
331
+ else if (strcmp(s, "second") == 0) st->w_second = 1;
332
+ else if (strcmp(s, "weekday") == 0) st->w_weekday = 1;
333
+ else if (strcmp(s, "epoch") == 0) st->w_epoch = 1;
334
+ else return TF_ERROR;
335
+ return TF_OK;
336
+ }
337
+
338
+ static int datetime_has_component(const datetime_state *st) {
339
+ return st->w_year || st->w_month || st->w_day || st->w_hour || st->w_minute ||
340
+ st->w_second || st->w_weekday || st->w_epoch;
341
+ }
342
+
173
343
  tf_step *tf_datetime_create(const cJSON *args) {
174
344
  if (!args) return NULL;
175
345
  cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
176
- if (!cJSON_IsString(col_j)) return NULL;
346
+ if (!cJSON_IsString(col_j) || !col_j->valuestring[0]) {
347
+ tf_set_last_error("datetime: column is required");
348
+ return NULL;
349
+ }
177
350
 
178
- datetime_state *st = calloc(1, sizeof(datetime_state));
351
+ datetime_state *st = tf_callocarray_checked(1, sizeof(datetime_state));
179
352
  if (!st) return NULL;
180
- st->column = strdup(col_j->valuestring);
353
+ st->column = tf_strdup_checked(col_j->valuestring);
354
+ if (!st->column) { free(st); return NULL; }
181
355
 
182
356
  cJSON *extract = cJSON_GetObjectItemCaseSensitive(args, "extract");
183
357
  if (extract && cJSON_IsArray(extract)) {
184
358
  int n = cJSON_GetArraySize(extract);
185
359
  for (int i = 0; i < n; i++) {
186
360
  cJSON *item = cJSON_GetArrayItem(extract, i);
187
- if (!cJSON_IsString(item)) continue;
188
- const char *s = item->valuestring;
189
- if (strcmp(s, "year") == 0) st->w_year = 1;
190
- else if (strcmp(s, "month") == 0) st->w_month = 1;
191
- else if (strcmp(s, "day") == 0) st->w_day = 1;
192
- else if (strcmp(s, "hour") == 0) st->w_hour = 1;
193
- else if (strcmp(s, "minute") == 0) st->w_minute = 1;
194
- else if (strcmp(s, "second") == 0) st->w_second = 1;
195
- else if (strcmp(s, "weekday") == 0) st->w_weekday = 1;
196
- else if (strcmp(s, "epoch") == 0) st->w_epoch = 1;
361
+ if (!cJSON_IsString(item) || !item->valuestring[0] ||
362
+ datetime_enable_component(st, item->valuestring) != TF_OK) {
363
+ tf_set_last_error("datetime: extract must contain year, month, day, hour, minute, second, weekday, or epoch");
364
+ datetime_state_free(st);
365
+ return NULL;
366
+ }
367
+ }
368
+ if (!datetime_has_component(st)) {
369
+ tf_set_last_error("datetime: extract must contain at least one component");
370
+ datetime_state_free(st);
371
+ return NULL;
197
372
  }
198
373
  } else {
199
374
  /* Default: all */
@@ -202,7 +377,14 @@ tf_step *tf_datetime_create(const cJSON *args) {
202
377
  st->w_weekday = st->w_epoch = 1;
203
378
  }
204
379
 
205
- tf_step *step = malloc(sizeof(tf_step));
380
+ if (datetime_parse_missing_policy(args, &st->missing) != TF_OK ||
381
+ datetime_parse_type_policy(args, &st->on_type_error) != TF_OK) {
382
+ free(st->column);
383
+ free(st);
384
+ return NULL;
385
+ }
386
+
387
+ tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
206
388
  if (!step) { free(st->column); free(st); return NULL; }
207
389
  step->process = datetime_process;
208
390
  step->flush = datetime_flush;
package/csrc/op_derive.c CHANGED
@@ -11,6 +11,8 @@
11
11
  #include <stdlib.h>
12
12
  #include <string.h>
13
13
  #include <stdio.h>
14
+ #include <limits.h>
15
+ #include <math.h>
14
16
 
15
17
  typedef struct {
16
18
  char *name;
@@ -24,19 +26,21 @@ typedef struct {
24
26
  tf_type *col_types; /* resolved type per derived column */
25
27
  } derive_state;
26
28
 
27
- /* Evaluate first row to determine derived column types */
28
- static void resolve_types(derive_state *st, const tf_batch *in) {
29
- st->col_types = calloc(st->n_cols, sizeof(tf_type));
30
- if (!st->col_types) return;
29
+ /* Evaluate first row to determine derived column types. */
30
+ static int resolve_types(derive_state *st, const tf_batch *in) {
31
+ st->col_types = calloc(st->n_cols ? st->n_cols : 1, sizeof(tf_type));
32
+ if (!st->col_types) return TF_ERROR;
31
33
 
32
34
  for (size_t d = 0; d < st->n_cols; d++) {
33
35
  if (in->n_rows == 0) {
34
- /* No rows to sample; default to FLOAT64 for arithmetic */
35
36
  st->col_types[d] = TF_TYPE_FLOAT64;
36
37
  continue;
37
38
  }
38
39
  tf_eval_result val;
39
- tf_expr_eval_val(st->cols[d].expr, in, 0, &val);
40
+ if (tf_expr_eval_val(st->cols[d].expr, in, 0, &val) != TF_OK) {
41
+ st->col_types[d] = TF_TYPE_FLOAT64;
42
+ continue;
43
+ }
40
44
  switch (val.type) {
41
45
  case TF_TYPE_INT64: st->col_types[d] = TF_TYPE_INT64; break;
42
46
  case TF_TYPE_FLOAT64: st->col_types[d] = TF_TYPE_FLOAT64; break;
@@ -48,60 +52,49 @@ static void resolve_types(derive_state *st, const tf_batch *in) {
48
52
  }
49
53
  }
50
54
  st->types_resolved = 1;
55
+ return TF_OK;
51
56
  }
52
57
 
53
- static void set_derived_value(tf_batch *ob, size_t row, size_t col,
54
- tf_type col_type, const tf_eval_result *val) {
55
- if (val->type == TF_TYPE_NULL) {
56
- tf_batch_set_null(ob, row, col);
57
- return;
58
- }
58
+ static int derive_double_to_i64(double v, int64_t *out) {
59
+ if (!isfinite(v) || v < (double)INT64_MIN || v > (double)INT64_MAX) return TF_ERROR;
60
+ *out = (int64_t)v;
61
+ return TF_OK;
62
+ }
63
+
64
+ static int set_derived_value(tf_batch *ob, size_t row, size_t col,
65
+ tf_type col_type, const tf_eval_result *val) {
66
+ if (!val || val->type == TF_TYPE_NULL) return tf_batch_set_null(ob, row, col);
59
67
 
60
- /* Convert value to the column's type */
61
68
  switch (col_type) {
62
69
  case TF_TYPE_INT64:
63
- if (val->type == TF_TYPE_INT64)
64
- tf_batch_set_int64(ob, row, col, val->i);
65
- else if (val->type == TF_TYPE_FLOAT64)
66
- tf_batch_set_int64(ob, row, col, (int64_t)val->f);
67
- else
68
- tf_batch_set_null(ob, row, col);
69
- break;
70
+ if (val->type == TF_TYPE_INT64) return tf_batch_set_int64(ob, row, col, val->i);
71
+ if (val->type == TF_TYPE_FLOAT64) {
72
+ int64_t iv = 0;
73
+ if (derive_double_to_i64(val->f, &iv) != TF_OK) return tf_batch_set_null(ob, row, col);
74
+ return tf_batch_set_int64(ob, row, col, iv);
75
+ }
76
+ return tf_batch_set_null(ob, row, col);
70
77
  case TF_TYPE_FLOAT64:
71
- if (val->type == TF_TYPE_FLOAT64)
72
- tf_batch_set_float64(ob, row, col, val->f);
73
- else if (val->type == TF_TYPE_INT64)
74
- tf_batch_set_float64(ob, row, col, (double)val->i);
75
- else
76
- tf_batch_set_null(ob, row, col);
77
- break;
78
+ if (val->type == TF_TYPE_FLOAT64) return tf_batch_set_float64(ob, row, col, val->f);
79
+ if (val->type == TF_TYPE_INT64) return tf_batch_set_float64(ob, row, col, (double)val->i);
80
+ return tf_batch_set_null(ob, row, col);
78
81
  case TF_TYPE_STRING:
79
- if (val->type == TF_TYPE_STRING)
80
- tf_batch_set_string(ob, row, col, val->s);
81
- else
82
- tf_batch_set_null(ob, row, col);
83
- break;
82
+ if (val->type == TF_TYPE_STRING) {
83
+ if (!val->s) return tf_batch_set_null(ob, row, col);
84
+ return tf_batch_set_string(ob, row, col, val->s);
85
+ }
86
+ return tf_batch_set_null(ob, row, col);
84
87
  case TF_TYPE_BOOL:
85
- if (val->type == TF_TYPE_BOOL)
86
- tf_batch_set_bool(ob, row, col, val->b);
87
- else
88
- tf_batch_set_null(ob, row, col);
89
- break;
88
+ if (val->type == TF_TYPE_BOOL) return tf_batch_set_bool(ob, row, col, val->b);
89
+ return tf_batch_set_null(ob, row, col);
90
90
  case TF_TYPE_DATE:
91
- if (val->type == TF_TYPE_DATE)
92
- tf_batch_set_date(ob, row, col, val->date);
93
- else
94
- tf_batch_set_null(ob, row, col);
95
- break;
91
+ if (val->type == TF_TYPE_DATE) return tf_batch_set_date(ob, row, col, val->date);
92
+ return tf_batch_set_null(ob, row, col);
96
93
  case TF_TYPE_TIMESTAMP:
97
- if (val->type == TF_TYPE_TIMESTAMP)
98
- tf_batch_set_timestamp(ob, row, col, val->i);
99
- else
100
- tf_batch_set_null(ob, row, col);
101
- break;
94
+ if (val->type == TF_TYPE_TIMESTAMP) return tf_batch_set_timestamp(ob, row, col, val->i);
95
+ return tf_batch_set_null(ob, row, col);
102
96
  default:
103
- tf_batch_set_null(ob, row, col);
104
- break;
97
+ return tf_batch_set_null(ob, row, col);
105
98
  }
106
99
  }
107
100
 
@@ -111,69 +104,40 @@ static int derive_process(tf_step *self, tf_batch *in, tf_batch **out,
111
104
  derive_state *st = self->state;
112
105
  *out = NULL;
113
106
 
114
- /* Resolve types on first batch */
115
- if (!st->types_resolved) {
116
- resolve_types(st, in);
117
- }
107
+ if (!st->types_resolved && resolve_types(st, in) != TF_OK) return TF_ERROR;
108
+
109
+ size_t out_n_cols = 0;
110
+ if (tf_size_add(in->n_cols, st->n_cols, &out_n_cols) != TF_OK) return TF_ERROR;
118
111
 
119
- size_t out_n_cols = in->n_cols + st->n_cols;
120
112
  tf_batch *ob = tf_batch_create(out_n_cols, in->n_rows > 0 ? in->n_rows : 1);
121
113
  if (!ob) return TF_ERROR;
122
114
 
123
- /* Copy input schema */
124
- for (size_t c = 0; c < in->n_cols; c++) {
125
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
115
+ const char **extra_names = calloc(st->n_cols ? st->n_cols : 1, sizeof(char *));
116
+ if (!extra_names) {
117
+ tf_batch_free(ob);
118
+ return TF_ERROR;
126
119
  }
127
- /* Set derived column schemas with resolved types */
128
- for (size_t d = 0; d < st->n_cols; d++) {
129
- tf_batch_set_schema(ob, in->n_cols + d, st->cols[d].name, st->col_types[d]);
120
+ for (size_t d = 0; d < st->n_cols; d++) extra_names[d] = st->cols[d].name;
121
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, st->col_types, st->n_cols) != TF_OK) {
122
+ free(extra_names);
123
+ tf_batch_free(ob);
124
+ return TF_ERROR;
130
125
  }
126
+ free(extra_names);
131
127
 
132
128
  for (size_t r = 0; r < in->n_rows; r++) {
133
- tf_batch_ensure_capacity(ob, r + 1);
129
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) goto fail;
134
130
 
135
- /* Copy input columns */
136
- for (size_t c = 0; c < in->n_cols; c++) {
137
- if (tf_batch_is_null(in, r, c)) {
138
- tf_batch_set_null(ob, r, c);
139
- continue;
140
- }
141
- switch (in->col_types[c]) {
142
- case TF_TYPE_BOOL:
143
- tf_batch_set_bool(ob, r, c, tf_batch_get_bool(in, r, c));
144
- break;
145
- case TF_TYPE_INT64:
146
- tf_batch_set_int64(ob, r, c, tf_batch_get_int64(in, r, c));
147
- break;
148
- case TF_TYPE_FLOAT64:
149
- tf_batch_set_float64(ob, r, c, tf_batch_get_float64(in, r, c));
150
- break;
151
- case TF_TYPE_STRING:
152
- tf_batch_set_string(ob, r, c, tf_batch_get_string(in, r, c));
153
- break;
154
- case TF_TYPE_DATE:
155
- tf_batch_set_date(ob, r, c, tf_batch_get_date(in, r, c));
156
- break;
157
- case TF_TYPE_TIMESTAMP:
158
- tf_batch_set_timestamp(ob, r, c, tf_batch_get_timestamp(in, r, c));
159
- break;
160
- default:
161
- tf_batch_set_null(ob, r, c);
162
- break;
163
- }
164
- }
165
-
166
- /* Evaluate derived columns */
167
131
  for (size_t d = 0; d < st->n_cols; d++) {
168
132
  size_t col_idx = in->n_cols + d;
169
133
  tf_eval_result val;
170
134
  if (tf_expr_eval_val(st->cols[d].expr, in, r, &val) != TF_OK) {
171
- tf_batch_set_null(ob, r, col_idx);
135
+ if (tf_batch_set_null(ob, r, col_idx) != TF_OK) goto fail;
172
136
  continue;
173
137
  }
174
- set_derived_value(ob, r, col_idx, st->col_types[d], &val);
138
+ if (set_derived_value(ob, r, col_idx, st->col_types[d], &val) != TF_OK) goto fail;
175
139
  }
176
- ob->n_rows = r + 1;
140
+ if (tf_batch_expose_row(ob, r) != TF_OK) goto fail;
177
141
  }
178
142
 
179
143
  if (ob->n_rows > 0) {
@@ -182,6 +146,10 @@ static int derive_process(tf_step *self, tf_batch *in, tf_batch **out,
182
146
  tf_batch_free(ob);
183
147
  }
184
148
  return TF_OK;
149
+
150
+ fail:
151
+ tf_batch_free(ob);
152
+ return TF_ERROR;
185
153
  }
186
154
 
187
155
  static int derive_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
@@ -229,7 +197,7 @@ tf_step *tf_derive_create(const cJSON *args) {
229
197
  if (!st->cols[i].name || !st->cols[i].expr) goto fail;
230
198
  }
231
199
 
232
- tf_step *step = malloc(sizeof(tf_step));
200
+ tf_step *step = calloc(1, sizeof(tf_step));
233
201
  if (!step) goto fail;
234
202
  step->process = derive_process;
235
203
  step->flush = derive_flush;