tranfi 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +272 -40
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +348 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +8 -11
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/op_diff.c CHANGED
@@ -12,18 +12,93 @@
12
12
 
13
13
  #define MAX_DIFF_ORDER 8
14
14
 
15
+ typedef enum {
16
+ DIFF_MISSING_ERROR,
17
+ DIFF_MISSING_NULL,
18
+ DIFF_MISSING_IGNORE,
19
+ } diff_missing_policy;
20
+
21
+ typedef enum {
22
+ DIFF_TYPE_FAIL,
23
+ DIFF_TYPE_NULL,
24
+ } diff_type_policy;
25
+
15
26
  typedef struct {
16
27
  char *column;
17
28
  char *result;
18
29
  int order;
19
30
  double prev[MAX_DIFF_ORDER]; /* circular buffer of previous values */
20
31
  int count; /* rows seen so far */
32
+ diff_missing_policy missing;
33
+ diff_type_policy on_type_error;
21
34
  } diff_state;
22
35
 
36
+ static int diff_is_numeric_type(tf_type type) {
37
+ return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
38
+ }
39
+
23
40
  static double get_numeric(const tf_batch *b, size_t r, int ci) {
24
41
  if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
25
- if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
26
- return 0;
42
+ return tf_batch_get_float64(b, r, ci);
43
+ }
44
+
45
+ static void diff_set_col_error(const char *column, const char *suffix) {
46
+ char msg[512];
47
+ snprintf(msg, sizeof(msg), "diff: column '%s' %s", column ? column : "", suffix);
48
+ tf_set_last_error(msg);
49
+ }
50
+
51
+ static int diff_parse_missing_policy(const cJSON *args, diff_missing_policy *out) {
52
+ *out = DIFF_MISSING_ERROR;
53
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
54
+ if (!j) return TF_OK;
55
+ if (!cJSON_IsString(j)) {
56
+ tf_set_last_error("diff: missing must be error, null, or ignore");
57
+ return TF_ERROR;
58
+ }
59
+ if (strcmp(j->valuestring, "error") == 0) *out = DIFF_MISSING_ERROR;
60
+ else if (strcmp(j->valuestring, "null") == 0) *out = DIFF_MISSING_NULL;
61
+ else if (strcmp(j->valuestring, "ignore") == 0) *out = DIFF_MISSING_IGNORE;
62
+ else {
63
+ tf_set_last_error("diff: missing must be error, null, or ignore");
64
+ return TF_ERROR;
65
+ }
66
+ return TF_OK;
67
+ }
68
+
69
+ static int diff_parse_type_policy(const cJSON *args, diff_type_policy *out) {
70
+ *out = DIFF_TYPE_FAIL;
71
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
72
+ if (!j) return TF_OK;
73
+ if (!cJSON_IsString(j)) {
74
+ tf_set_last_error("diff: on_type_error must be fail or null");
75
+ return TF_ERROR;
76
+ }
77
+ if (strcmp(j->valuestring, "fail") == 0) *out = DIFF_TYPE_FAIL;
78
+ else if (strcmp(j->valuestring, "null") == 0) *out = DIFF_TYPE_NULL;
79
+ else {
80
+ tf_set_last_error("diff: on_type_error must be fail or null");
81
+ return TF_ERROR;
82
+ }
83
+ return TF_OK;
84
+ }
85
+
86
+ static int diff_passthrough(tf_batch *in, tf_batch **out) {
87
+ tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
88
+ if (!ob) return TF_ERROR;
89
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
90
+ tf_batch_free(ob);
91
+ return TF_ERROR;
92
+ }
93
+ for (size_t r = 0; r < in->n_rows; r++) {
94
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK ||
95
+ tf_batch_expose_row(ob, r) != TF_OK) {
96
+ tf_batch_free(ob);
97
+ return TF_ERROR;
98
+ }
99
+ }
100
+ *out = ob;
101
+ return TF_OK;
27
102
  }
28
103
 
29
104
  static int diff_process(tf_step *self, tf_batch *in, tf_batch **out,
@@ -32,31 +107,57 @@ static int diff_process(tf_step *self, tf_batch *in, tf_batch **out,
32
107
  diff_state *st = self->state;
33
108
  *out = NULL;
34
109
 
110
+ int ci = tf_batch_col_index(in, st->column);
111
+ int force_null = 0;
112
+ if (ci < 0) {
113
+ if (st->missing == DIFF_MISSING_ERROR) {
114
+ diff_set_col_error(st->column, "not found");
115
+ return TF_ERROR;
116
+ }
117
+ if (st->missing == DIFF_MISSING_IGNORE) {
118
+ return diff_passthrough(in, out);
119
+ }
120
+ force_null = 1;
121
+ } else if (!diff_is_numeric_type(in->col_types[(size_t)ci])) {
122
+ if (st->on_type_error == DIFF_TYPE_FAIL) {
123
+ diff_set_col_error(st->column, "must be numeric");
124
+ return TF_ERROR;
125
+ }
126
+ force_null = 1;
127
+ }
128
+
129
+ const char *extra_names[1] = {st->result};
130
+ tf_type extra_types[1] = {TF_TYPE_FLOAT64};
35
131
  tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
36
132
  if (!ob) return TF_ERROR;
37
- for (size_t c = 0; c < in->n_cols; c++)
38
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
39
- tf_batch_set_schema(ob, in->n_cols, st->result, TF_TYPE_FLOAT64);
40
-
41
- int ci = tf_batch_col_index(in, st->column);
133
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
134
+ tf_batch_free(ob);
135
+ return TF_ERROR;
136
+ }
42
137
 
43
138
  for (size_t r = 0; r < in->n_rows; r++) {
44
- tf_batch_copy_row(ob, r, in, r);
139
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
140
+ tf_batch_free(ob);
141
+ return TF_ERROR;
142
+ }
45
143
 
46
- if (ci < 0 || tf_batch_is_null(in, r, ci)) {
47
- tf_batch_set_null(ob, r, in->n_cols);
48
- ob->n_rows = r + 1;
144
+ if (force_null || tf_batch_is_null(in, r, (size_t)ci)) {
145
+ if (tf_batch_set_null(ob, r, in->n_cols) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
146
+ if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
49
147
  continue;
50
148
  }
51
149
 
52
150
  double val = get_numeric(in, r, ci);
53
151
 
54
152
  if (st->count < st->order) {
55
- /* Not enough history yet */
56
- st->prev[st->count] = val;
153
+ /* Not enough history yet -- shift and insert at front
154
+ * so prev[0] is always the most recent value */
155
+ if (tf_batch_set_null(ob, r, in->n_cols) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
156
+ if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
157
+ for (int k = st->count; k > 0; k--)
158
+ st->prev[k] = st->prev[k - 1];
159
+ st->prev[0] = val;
57
160
  st->count++;
58
- tf_batch_set_null(ob, r, in->n_cols);
59
- ob->n_rows = r + 1;
60
161
  continue;
61
162
  }
62
163
 
@@ -78,8 +179,14 @@ static int diff_process(tf_step *self, tf_batch *in, tf_batch **out,
78
179
  result += sign * binom * st->prev[k - 1];
79
180
  }
80
181
 
81
- tf_batch_set_float64(ob, r, in->n_cols, result);
82
- ob->n_rows = r + 1;
182
+ if (tf_batch_set_float64(ob, r, in->n_cols, result) != TF_OK) {
183
+ tf_batch_free(ob);
184
+ return TF_ERROR;
185
+ }
186
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
187
+ tf_batch_free(ob);
188
+ return TF_ERROR;
189
+ }
83
190
 
84
191
  /* Shift prev buffer: move everything down, put val at [0] */
85
192
  for (int k = st->order - 1; k > 0; k--)
@@ -104,27 +211,36 @@ static void diff_destroy(tf_step *self) {
104
211
  tf_step *tf_diff_create(const cJSON *args) {
105
212
  if (!args) return NULL;
106
213
  cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
107
- if (!cJSON_IsString(col_j)) return NULL;
214
+ if (!cJSON_IsString(col_j) || !col_j->valuestring[0]) {
215
+ tf_set_last_error("diff: column is required");
216
+ return NULL;
217
+ }
108
218
 
109
- diff_state *st = calloc(1, sizeof(diff_state));
219
+ diff_state *st = tf_callocarray_checked(1, sizeof(diff_state));
110
220
  if (!st) return NULL;
111
- st->column = strdup(col_j->valuestring);
112
-
113
- cJSON *order_j = cJSON_GetObjectItemCaseSensitive(args, "order");
114
- st->order = (cJSON_IsNumber(order_j) && order_j->valueint > 0) ?
115
- order_j->valueint : 1;
116
- if (st->order > MAX_DIFF_ORDER) st->order = MAX_DIFF_ORDER;
221
+ st->column = tf_strdup_checked(col_j->valuestring);
222
+ if (!st->column) { free(st); return NULL; }
223
+
224
+ size_t order = 1;
225
+ int has_order = tf_json_get_size_arg(args, "order", 1, MAX_DIFF_ORDER, &order, "diff");
226
+ if (has_order < 0) { free(st->column); free(st); return NULL; }
227
+ st->order = (int)order;
228
+ if (diff_parse_missing_policy(args, &st->missing) != TF_OK ||
229
+ diff_parse_type_policy(args, &st->on_type_error) != TF_OK) {
230
+ free(st->column);
231
+ free(st);
232
+ return NULL;
233
+ }
117
234
 
118
235
  cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
119
236
  if (cJSON_IsString(res_j)) {
120
- st->result = strdup(res_j->valuestring);
237
+ st->result = tf_strdup_checked(res_j->valuestring);
121
238
  } else {
122
- char buf[256];
123
- snprintf(buf, sizeof(buf), "%s_diff", st->column);
124
- st->result = strdup(buf);
239
+ st->result = tf_string_append_suffix_checked(st->column, "_diff");
125
240
  }
241
+ if (!st->result) { free(st->column); free(st); return NULL; }
126
242
 
127
- tf_step *step = malloc(sizeof(tf_step));
243
+ tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
128
244
  if (!step) { free(st->column); free(st->result); free(st); return NULL; }
129
245
  step->process = diff_process;
130
246
  step->flush = diff_flush;
package/csrc/op_ewma.c CHANGED
@@ -9,6 +9,18 @@
9
9
  #include <stdlib.h>
10
10
  #include <string.h>
11
11
  #include <stdio.h>
12
+ #include <math.h>
13
+
14
+ typedef enum {
15
+ EWMA_MISSING_ERROR,
16
+ EWMA_MISSING_NULL,
17
+ EWMA_MISSING_IGNORE,
18
+ } ewma_missing_policy;
19
+
20
+ typedef enum {
21
+ EWMA_TYPE_FAIL,
22
+ EWMA_TYPE_NULL,
23
+ } ewma_type_policy;
12
24
 
13
25
  typedef struct {
14
26
  char *column;
@@ -16,12 +28,79 @@ typedef struct {
16
28
  double alpha;
17
29
  double ewma;
18
30
  int initialized;
31
+ ewma_missing_policy missing;
32
+ ewma_type_policy on_type_error;
19
33
  } ewma_state;
20
34
 
21
- static double get_numeric(const tf_batch *b, size_t r, int ci) {
35
+ static int ewma_is_numeric_type(tf_type type) {
36
+ return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
37
+ }
38
+
39
+ static double ewma_get_numeric(const tf_batch *b, size_t r, int ci) {
22
40
  if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
23
- if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
24
- return 0;
41
+ return tf_batch_get_float64(b, r, ci);
42
+ }
43
+
44
+ static void ewma_set_col_error(const char *column, const char *suffix) {
45
+ char msg[512];
46
+ snprintf(msg, sizeof(msg), "ewma: column '%s' %s", column ? column : "", suffix);
47
+ tf_set_last_error(msg);
48
+ }
49
+
50
+ static int ewma_parse_missing_policy(const cJSON *args, ewma_missing_policy *out) {
51
+ *out = EWMA_MISSING_ERROR;
52
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
53
+ if (!j) return TF_OK;
54
+ if (!cJSON_IsString(j)) {
55
+ tf_set_last_error("ewma: missing must be error, null, or ignore");
56
+ return TF_ERROR;
57
+ }
58
+ if (strcmp(j->valuestring, "error") == 0) *out = EWMA_MISSING_ERROR;
59
+ else if (strcmp(j->valuestring, "null") == 0) *out = EWMA_MISSING_NULL;
60
+ else if (strcmp(j->valuestring, "ignore") == 0) *out = EWMA_MISSING_IGNORE;
61
+ else {
62
+ tf_set_last_error("ewma: missing must be error, null, or ignore");
63
+ return TF_ERROR;
64
+ }
65
+ return TF_OK;
66
+ }
67
+
68
+ static int ewma_parse_type_policy(const cJSON *args, ewma_type_policy *out) {
69
+ *out = EWMA_TYPE_FAIL;
70
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
71
+ if (!j) return TF_OK;
72
+ if (!cJSON_IsString(j)) {
73
+ tf_set_last_error("ewma: on_type_error must be fail or null");
74
+ return TF_ERROR;
75
+ }
76
+ if (strcmp(j->valuestring, "fail") == 0) *out = EWMA_TYPE_FAIL;
77
+ else if (strcmp(j->valuestring, "null") == 0) *out = EWMA_TYPE_NULL;
78
+ else {
79
+ tf_set_last_error("ewma: on_type_error must be fail or null");
80
+ return TF_ERROR;
81
+ }
82
+ return TF_OK;
83
+ }
84
+
85
+ static int ewma_passthrough(tf_batch *in, tf_batch **out) {
86
+ tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
87
+ if (!ob) return TF_ERROR;
88
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
89
+ tf_batch_free(ob);
90
+ return TF_ERROR;
91
+ }
92
+ for (size_t r = 0; r < in->n_rows; r++) {
93
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
94
+ tf_batch_free(ob);
95
+ return TF_ERROR;
96
+ }
97
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
98
+ tf_batch_free(ob);
99
+ return TF_ERROR;
100
+ }
101
+ }
102
+ *out = ob;
103
+ return TF_OK;
25
104
  }
26
105
 
27
106
  static int ewma_process(tf_step *self, tf_batch *in, tf_batch **out,
@@ -30,33 +109,59 @@ static int ewma_process(tf_step *self, tf_batch *in, tf_batch **out,
30
109
  ewma_state *st = self->state;
31
110
  *out = NULL;
32
111
 
112
+ int ci = tf_batch_col_index(in, st->column);
113
+ int force_null = 0;
114
+ if (ci < 0) {
115
+ if (st->missing == EWMA_MISSING_ERROR) {
116
+ ewma_set_col_error(st->column, "not found");
117
+ return TF_ERROR;
118
+ }
119
+ if (st->missing == EWMA_MISSING_IGNORE) return ewma_passthrough(in, out);
120
+ force_null = 1;
121
+ } else if (!ewma_is_numeric_type(in->col_types[ci])) {
122
+ if (st->on_type_error == EWMA_TYPE_FAIL) {
123
+ ewma_set_col_error(st->column, "must be numeric");
124
+ return TF_ERROR;
125
+ }
126
+ force_null = 1;
127
+ }
128
+
129
+ const char *extra_names[1] = {st->result};
130
+ tf_type extra_types[1] = {TF_TYPE_FLOAT64};
33
131
  tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
34
132
  if (!ob) return TF_ERROR;
35
- for (size_t c = 0; c < in->n_cols; c++)
36
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
37
- tf_batch_set_schema(ob, in->n_cols, st->result, TF_TYPE_FLOAT64);
38
-
39
- int ci = tf_batch_col_index(in, st->column);
133
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
134
+ tf_batch_free(ob);
135
+ return TF_ERROR;
136
+ }
40
137
 
41
138
  for (size_t r = 0; r < in->n_rows; r++) {
42
- tf_batch_copy_row(ob, r, in, r);
139
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
140
+ tf_batch_free(ob);
141
+ return TF_ERROR;
142
+ }
43
143
 
44
- if (ci < 0 || tf_batch_is_null(in, r, ci)) {
45
- tf_batch_set_null(ob, r, in->n_cols);
46
- ob->n_rows = r + 1;
144
+ if (force_null || tf_batch_is_null(in, r, ci)) {
145
+ if (tf_batch_set_null(ob, r, in->n_cols) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
146
+ if (tf_batch_expose_row(ob, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
47
147
  continue;
48
148
  }
49
149
 
50
- double val = get_numeric(in, r, ci);
51
- if (!st->initialized) {
52
- st->ewma = val;
53
- st->initialized = 1;
54
- } else {
55
- st->ewma = st->alpha * val + (1.0 - st->alpha) * st->ewma;
56
- }
150
+ double val = ewma_get_numeric(in, r, ci);
151
+ double next_ewma = st->initialized
152
+ ? st->alpha * val + (1.0 - st->alpha) * st->ewma
153
+ : val;
57
154
 
58
- tf_batch_set_float64(ob, r, in->n_cols, st->ewma);
59
- ob->n_rows = r + 1;
155
+ if (tf_batch_set_float64(ob, r, in->n_cols, next_ewma) != TF_OK) {
156
+ tf_batch_free(ob);
157
+ return TF_ERROR;
158
+ }
159
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
160
+ tf_batch_free(ob);
161
+ return TF_ERROR;
162
+ }
163
+ st->ewma = next_ewma;
164
+ st->initialized = 1;
60
165
  }
61
166
 
62
167
  *out = ob;
@@ -77,23 +182,37 @@ tf_step *tf_ewma_create(const cJSON *args) {
77
182
  if (!args) return NULL;
78
183
  cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
79
184
  cJSON *alpha_j = cJSON_GetObjectItemCaseSensitive(args, "alpha");
80
- if (!cJSON_IsString(col_j) || !cJSON_IsNumber(alpha_j)) return NULL;
185
+ if (!cJSON_IsString(col_j) || !col_j->valuestring[0] || !cJSON_IsNumber(alpha_j)) {
186
+ tf_set_last_error("ewma: column and alpha are required");
187
+ return NULL;
188
+ }
189
+ double alpha = alpha_j->valuedouble;
190
+ if (!isfinite(alpha) || alpha < 0.0 || alpha > 1.0) {
191
+ tf_set_last_error("ewma: alpha must be a finite number between 0 and 1");
192
+ return NULL;
193
+ }
81
194
 
82
- ewma_state *st = calloc(1, sizeof(ewma_state));
195
+ ewma_state *st = tf_callocarray_checked(1, sizeof(ewma_state));
83
196
  if (!st) return NULL;
84
- st->column = strdup(col_j->valuestring);
85
- st->alpha = alpha_j->valuedouble;
197
+ st->column = tf_strdup_checked(col_j->valuestring);
198
+ st->alpha = alpha;
199
+ if (!st->column) { free(st); return NULL; }
200
+ if (ewma_parse_missing_policy(args, &st->missing) != TF_OK ||
201
+ ewma_parse_type_policy(args, &st->on_type_error) != TF_OK) {
202
+ free(st->column);
203
+ free(st);
204
+ return NULL;
205
+ }
86
206
 
87
207
  cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
88
208
  if (cJSON_IsString(res_j)) {
89
- st->result = strdup(res_j->valuestring);
209
+ st->result = tf_strdup_checked(res_j->valuestring);
90
210
  } else {
91
- char buf[256];
92
- snprintf(buf, sizeof(buf), "%s_ewma", st->column);
93
- st->result = strdup(buf);
211
+ st->result = tf_string_append_suffix_checked(st->column, "_ewma");
94
212
  }
213
+ if (!st->result) { free(st->column); free(st); return NULL; }
95
214
 
96
- tf_step *step = malloc(sizeof(tf_step));
215
+ tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
97
216
  if (!step) { free(st->column); free(st->result); free(st); return NULL; }
98
217
  step->process = ewma_process;
99
218
  step->flush = ewma_flush;
package/csrc/op_explode.c CHANGED
@@ -1,19 +1,31 @@
1
1
  /*
2
- * op_explode.c — Split delimited string into multiple rows.
2
+ * op_explode.c -- Split delimited string into multiple rows.
3
3
  *
4
- * Config: {"column": "tags", "delimiter": ","}
4
+ * Config: {"column": "tags", "delimiter": ",", "max_tokens_per_row": 1024}
5
5
  */
6
6
 
7
7
  #include "internal.h"
8
8
  #include "cJSON.h"
9
+ #include <stdio.h>
9
10
  #include <stdlib.h>
10
11
  #include <string.h>
11
12
 
12
13
  typedef struct {
13
14
  char *column;
14
15
  char *delimiter;
16
+ size_t max_tokens_per_row;
17
+ size_t max_output_rows_per_input_row;
18
+ size_t max_output_rows_per_batch;
19
+ size_t max_token_bytes;
15
20
  } explode_state;
16
21
 
22
+ static int explode_set_cap_error(const char *name, size_t limit) {
23
+ char msg[160];
24
+ snprintf(msg, sizeof(msg), "explode: %s=%zu exceeded", name, limit);
25
+ tf_set_last_error(msg);
26
+ return TF_ERROR;
27
+ }
28
+
17
29
  static int explode_process(tf_step *self, tf_batch *in, tf_batch **out,
18
30
  tf_side_channels *side) {
19
31
  (void)side;
@@ -22,46 +34,103 @@ static int explode_process(tf_step *self, tf_batch *in, tf_batch **out,
22
34
 
23
35
  int ci = tf_batch_col_index(in, st->column);
24
36
 
25
- /* Estimate max output rows */
26
- size_t max_out = in->n_rows * 4;
27
- tf_batch *ob = tf_batch_create(in->n_cols, max_out > 0 ? max_out : 16);
37
+ size_t initial_cap = in->n_rows ? in->n_rows : 1;
38
+ if (initial_cap > st->max_output_rows_per_batch)
39
+ initial_cap = st->max_output_rows_per_batch;
40
+ if (initial_cap < 16 && st->max_output_rows_per_batch >= 16)
41
+ initial_cap = 16;
42
+ tf_batch *ob = tf_batch_create(in->n_cols, initial_cap);
28
43
  if (!ob) return TF_ERROR;
29
- for (size_t c = 0; c < in->n_cols; c++)
30
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
44
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
45
+ tf_batch_free(ob);
46
+ return TF_ERROR;
47
+ }
31
48
 
32
49
  size_t out_row = 0;
33
50
  size_t delim_len = strlen(st->delimiter);
34
51
 
35
52
  for (size_t r = 0; r < in->n_rows; r++) {
36
53
  if (ci < 0 || tf_batch_is_null(in, r, ci) || in->col_types[ci] != TF_TYPE_STRING) {
37
- tf_batch_copy_row(ob, out_row, in, r);
38
- ob->n_rows = ++out_row;
54
+ if (out_row >= st->max_output_rows_per_batch) {
55
+ explode_set_cap_error("max_output_rows_per_batch", st->max_output_rows_per_batch);
56
+ tf_batch_free(ob);
57
+ return TF_ERROR;
58
+ }
59
+ if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
60
+ tf_batch_free(ob);
61
+ return TF_ERROR;
62
+ }
63
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) {
64
+ tf_batch_free(ob);
65
+ return TF_ERROR;
66
+ }
67
+ out_row++;
39
68
  continue;
40
69
  }
41
70
 
42
71
  const char *val = tf_batch_get_string(in, r, ci);
43
72
  const char *p = val;
73
+ size_t row_outputs = 0;
44
74
 
45
75
  while (*p) {
76
+ if (row_outputs >= st->max_tokens_per_row) {
77
+ explode_set_cap_error("max_tokens_per_row", st->max_tokens_per_row);
78
+ tf_batch_free(ob);
79
+ return TF_ERROR;
80
+ }
81
+ if (row_outputs >= st->max_output_rows_per_input_row) {
82
+ explode_set_cap_error("max_output_rows_per_input_row", st->max_output_rows_per_input_row);
83
+ tf_batch_free(ob);
84
+ return TF_ERROR;
85
+ }
86
+ if (out_row >= st->max_output_rows_per_batch) {
87
+ explode_set_cap_error("max_output_rows_per_batch", st->max_output_rows_per_batch);
88
+ tf_batch_free(ob);
89
+ return TF_ERROR;
90
+ }
91
+
46
92
  const char *found = strstr(p, st->delimiter);
47
93
  size_t tok_len = found ? (size_t)(found - p) : strlen(p);
48
-
49
- tf_batch_copy_row(ob, out_row, in, r);
94
+ if (tok_len > st->max_token_bytes) {
95
+ explode_set_cap_error("max_token_bytes", st->max_token_bytes);
96
+ tf_batch_free(ob);
97
+ return TF_ERROR;
98
+ }
99
+ if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
100
+ tf_batch_free(ob);
101
+ return TF_ERROR;
102
+ }
50
103
  /* Override the exploded column */
51
- char *tok = malloc(tok_len + 1);
52
- if (tok) {
53
- memcpy(tok, p, tok_len);
54
- tok[tok_len] = '\0';
55
- /* Trim leading/trailing whitespace */
56
- char *s = tok;
57
- while (*s == ' ') s++;
58
- char *e = s + strlen(s);
59
- while (e > s && *(e - 1) == ' ') e--;
60
- *e = '\0';
61
- tf_batch_set_string(ob, out_row, ci, s);
104
+ size_t tok_cap = 0;
105
+ if (tf_size_add(tok_len, 1, &tok_cap) != TF_OK) {
106
+ tf_batch_free(ob);
107
+ return TF_ERROR;
108
+ }
109
+ char *tok = malloc(tok_cap);
110
+ if (!tok) {
111
+ tf_batch_free(ob);
112
+ return TF_ERROR;
113
+ }
114
+ memcpy(tok, p, tok_len);
115
+ tok[tok_len] = '\0';
116
+ /* Trim leading/trailing whitespace */
117
+ char *s = tok;
118
+ while (*s == ' ') s++;
119
+ char *e = s + strlen(s);
120
+ while (e > s && *(e - 1) == ' ') e--;
121
+ *e = '\0';
122
+ if (tf_batch_set_string(ob, out_row, (size_t)ci, s) != TF_OK) {
62
123
  free(tok);
124
+ tf_batch_free(ob);
125
+ return TF_ERROR;
126
+ }
127
+ free(tok);
128
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) {
129
+ tf_batch_free(ob);
130
+ return TF_ERROR;
63
131
  }
64
- ob->n_rows = ++out_row;
132
+ out_row++;
133
+ row_outputs++;
65
134
 
66
135
  if (found) p = found + delim_len;
67
136
  else break;
@@ -93,16 +162,45 @@ tf_step *tf_explode_create(const cJSON *args) {
93
162
 
94
163
  explode_state *st = calloc(1, sizeof(explode_state));
95
164
  if (!st) return NULL;
165
+ st->max_tokens_per_row = TF_MAX_EXPLODE_TOKENS_PER_ROW;
166
+ st->max_output_rows_per_input_row = TF_MAX_EXPANDING_ROWS_PER_INPUT;
167
+ st->max_output_rows_per_batch = TF_MAX_OUTPUT_ROWS_PER_BATCH;
168
+ st->max_token_bytes = TF_MAX_RECORD_BYTES;
169
+
170
+ size_t parsed_size = 0;
171
+ int has_size = tf_json_get_size_arg(args, "max_tokens_per_row", 1, TF_MAX_EXPLODE_TOKENS_PER_ROW, &parsed_size, "explode");
172
+ if (has_size < 0) goto fail;
173
+ if (has_size > 0) st->max_tokens_per_row = parsed_size;
174
+ has_size = tf_json_get_size_arg(args, "max_output_rows_per_input_row", 1, TF_MAX_EXPANDING_ROWS_PER_INPUT, &parsed_size, "explode");
175
+ if (has_size < 0) goto fail;
176
+ if (has_size > 0) st->max_output_rows_per_input_row = parsed_size;
177
+ has_size = tf_json_get_size_arg(args, "max_output_rows_per_batch", 1, TF_MAX_OUTPUT_ROWS_PER_BATCH, &parsed_size, "explode");
178
+ if (has_size < 0) goto fail;
179
+ if (has_size > 0) st->max_output_rows_per_batch = parsed_size;
180
+ has_size = tf_json_get_size_arg(args, "max_token_bytes", 0, TF_MAX_RECORD_BYTES, &parsed_size, "explode");
181
+ if (has_size < 0) goto fail;
182
+ if (has_size > 0) st->max_token_bytes = parsed_size;
183
+
96
184
  st->column = strdup(col_j->valuestring);
185
+ if (!st->column) goto fail;
97
186
 
98
187
  cJSON *delim_j = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
99
- st->delimiter = strdup(cJSON_IsString(delim_j) ? delim_j->valuestring : ",");
188
+ const char *delim = cJSON_IsString(delim_j) ? delim_j->valuestring : ",";
189
+ if (delim[0] == '\0') goto fail;
190
+ st->delimiter = strdup(delim);
191
+ if (!st->delimiter) goto fail;
100
192
 
101
- tf_step *step = malloc(sizeof(tf_step));
102
- if (!step) { free(st->column); free(st->delimiter); free(st); return NULL; }
193
+ tf_step *step = calloc(1, sizeof(tf_step));
194
+ if (!step) goto fail;
103
195
  step->process = explode_process;
104
196
  step->flush = explode_flush;
105
197
  step->destroy = explode_destroy;
106
198
  step->state = st;
107
199
  return step;
200
+
201
+ fail:
202
+ free(st->column);
203
+ free(st->delimiter);
204
+ free(st);
205
+ return NULL;
108
206
  }