tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/csrc/op_lead.c
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
/*
|
|
2
|
-
* op_lead.c
|
|
2
|
+
* op_lead.c - Bounded lookahead.
|
|
3
3
|
*
|
|
4
4
|
* Config: {"column": "price", "offset": 1, "result": "next_price"}
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
* On flush, emits remaining rows with NULL lead
|
|
6
|
+
* Delays output by `offset` rows so the appended value can come from rows ahead
|
|
7
|
+
* of the current row. On flush, emits the remaining pending rows with NULL lead
|
|
8
|
+
* values. Pending state is bounded by `offset` rows.
|
|
8
9
|
*/
|
|
9
10
|
|
|
10
11
|
#include "internal.h"
|
|
@@ -14,136 +15,178 @@
|
|
|
14
15
|
#include <stdio.h>
|
|
15
16
|
|
|
16
17
|
typedef struct {
|
|
17
|
-
char
|
|
18
|
-
char
|
|
19
|
-
size_t
|
|
20
|
-
/* Pending rows: stored as a batch that hasn't been emitted yet */
|
|
18
|
+
char *column;
|
|
19
|
+
char *result;
|
|
20
|
+
size_t offset;
|
|
21
21
|
tf_batch *pending;
|
|
22
22
|
} lead_state;
|
|
23
23
|
|
|
24
|
-
static
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
24
|
+
static int lead_set_error(tf_side_channels *side, const char *msg) {
|
|
25
|
+
return tf_side_write_error(side, msg);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
static const tf_batch *source_at(const lead_state *st, const tf_batch *in,
|
|
29
|
+
size_t idx, size_t pend_count, size_t *row) {
|
|
30
|
+
if (idx < pend_count) {
|
|
31
|
+
*row = idx;
|
|
32
|
+
return st->pending;
|
|
33
|
+
}
|
|
34
|
+
*row = idx - pend_count;
|
|
35
|
+
return in;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
static int copy_source_row(tf_batch *dst, size_t dst_row, const lead_state *st,
|
|
39
|
+
const tf_batch *in, size_t idx, size_t pend_count) {
|
|
40
|
+
size_t src_row = 0;
|
|
41
|
+
const tf_batch *src = source_at(st, in, idx, pend_count, &src_row);
|
|
42
|
+
return tf_batch_copy_row(dst, dst_row, src, src_row);
|
|
28
43
|
}
|
|
29
44
|
|
|
30
45
|
static int lead_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
31
46
|
tf_side_channels *side) {
|
|
32
|
-
(void)side;
|
|
33
47
|
lead_state *st = self->state;
|
|
34
48
|
*out = NULL;
|
|
35
49
|
|
|
50
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
51
|
+
if (ci < 0) {
|
|
52
|
+
char msg[256];
|
|
53
|
+
snprintf(msg, sizeof(msg), "lead: column '%s' not found", st->column);
|
|
54
|
+
if (lead_set_error(side, msg) != TF_OK) return TF_ERROR;
|
|
55
|
+
return TF_ERROR;
|
|
56
|
+
}
|
|
57
|
+
|
|
36
58
|
size_t pend_count = st->pending ? st->pending->n_rows : 0;
|
|
37
59
|
size_t total = pend_count + in->n_rows;
|
|
38
60
|
|
|
39
61
|
if (total <= st->offset) {
|
|
40
|
-
/* Not enough rows yet — append all to pending */
|
|
41
62
|
tf_batch *new_pend = tf_batch_create(in->n_cols, total);
|
|
42
63
|
if (!new_pend) return TF_ERROR;
|
|
43
|
-
|
|
44
|
-
|
|
64
|
+
if (tf_batch_clone_schema(new_pend, in) != TF_OK) {
|
|
65
|
+
tf_batch_free(new_pend);
|
|
66
|
+
return TF_ERROR;
|
|
67
|
+
}
|
|
45
68
|
size_t row = 0;
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
69
|
+
for (size_t r = 0; r < pend_count; r++) {
|
|
70
|
+
if (tf_batch_copy_row(new_pend, row++, st->pending, r) != TF_OK) {
|
|
71
|
+
tf_batch_free(new_pend);
|
|
72
|
+
return TF_ERROR;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
76
|
+
if (tf_batch_copy_row(new_pend, row++, in, r) != TF_OK) {
|
|
77
|
+
tf_batch_free(new_pend);
|
|
78
|
+
return TF_ERROR;
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
if (total > 0 && tf_batch_expose_row(new_pend, total - 1) != TF_OK) {
|
|
82
|
+
tf_batch_free(new_pend);
|
|
83
|
+
return TF_ERROR;
|
|
50
84
|
}
|
|
51
|
-
|
|
52
|
-
tf_batch_copy_row(new_pend, row++, in, r);
|
|
53
|
-
new_pend->n_rows = total;
|
|
85
|
+
if (st->pending) tf_batch_free(st->pending);
|
|
54
86
|
st->pending = new_pend;
|
|
55
87
|
return TF_OK;
|
|
56
88
|
}
|
|
57
89
|
|
|
58
90
|
size_t emit_count = total - st->offset;
|
|
91
|
+
const char *extra_names[1] = {st->result};
|
|
92
|
+
tf_type extra_types[1] = {in->col_types[ci]};
|
|
59
93
|
tf_batch *ob = tf_batch_create(in->n_cols + 1, emit_count);
|
|
60
94
|
if (!ob) return TF_ERROR;
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
95
|
+
if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
|
|
96
|
+
tf_batch_free(ob);
|
|
97
|
+
return TF_ERROR;
|
|
98
|
+
}
|
|
64
99
|
|
|
65
|
-
/* Emit rows: for each emitted row i, lead value comes from row i+offset */
|
|
66
100
|
for (size_t i = 0; i < emit_count; i++) {
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
} else {
|
|
71
|
-
tf_batch_copy_row(ob, i, in, i - pend_count);
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
/* Lead value from row i + offset */
|
|
75
|
-
size_t lead_idx = i + st->offset;
|
|
76
|
-
tf_batch *lead_src;
|
|
77
|
-
size_t lead_row;
|
|
78
|
-
if (lead_idx < pend_count) {
|
|
79
|
-
lead_src = st->pending;
|
|
80
|
-
lead_row = lead_idx;
|
|
81
|
-
} else {
|
|
82
|
-
lead_src = in;
|
|
83
|
-
lead_row = lead_idx - pend_count;
|
|
101
|
+
if (copy_source_row(ob, i, st, in, i, pend_count) != TF_OK) {
|
|
102
|
+
tf_batch_free(ob);
|
|
103
|
+
return TF_ERROR;
|
|
84
104
|
}
|
|
85
105
|
|
|
106
|
+
size_t lead_row = 0;
|
|
107
|
+
const tf_batch *lead_src = source_at(st, in, i + st->offset, pend_count, &lead_row);
|
|
86
108
|
int lead_ci = tf_batch_col_index(lead_src, st->column);
|
|
87
|
-
if (lead_ci
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
109
|
+
if (lead_ci < 0) {
|
|
110
|
+
tf_batch_free(ob);
|
|
111
|
+
if (lead_set_error(side, "lead: buffered schema lost selected column") != TF_OK) return TF_ERROR;
|
|
112
|
+
return TF_ERROR;
|
|
113
|
+
}
|
|
114
|
+
if (tf_batch_copy_cell(ob, i, in->n_cols, lead_src, lead_row, (size_t)lead_ci) != TF_OK) {
|
|
115
|
+
tf_batch_free(ob);
|
|
116
|
+
return TF_ERROR;
|
|
117
|
+
}
|
|
118
|
+
if (tf_batch_expose_row(ob, i) != TF_OK) {
|
|
119
|
+
tf_batch_free(ob);
|
|
120
|
+
return TF_ERROR;
|
|
91
121
|
}
|
|
92
122
|
}
|
|
93
|
-
ob->n_rows = emit_count;
|
|
94
123
|
|
|
95
|
-
|
|
96
|
-
if (
|
|
97
|
-
|
|
98
|
-
|
|
124
|
+
tf_batch *new_pend = tf_batch_create(in->n_cols, st->offset);
|
|
125
|
+
if (!new_pend) { tf_batch_free(ob); return TF_ERROR; }
|
|
126
|
+
if (tf_batch_clone_schema(new_pend, in) != TF_OK) {
|
|
127
|
+
tf_batch_free(new_pend);
|
|
128
|
+
tf_batch_free(ob);
|
|
129
|
+
return TF_ERROR;
|
|
99
130
|
}
|
|
100
|
-
size_t
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
} else {
|
|
112
|
-
tf_batch_copy_row(new_pend, i, in, src_idx - pend_count);
|
|
113
|
-
}
|
|
131
|
+
for (size_t i = 0; i < st->offset; i++) {
|
|
132
|
+
size_t src_idx = total - st->offset + i;
|
|
133
|
+
if (copy_source_row(new_pend, i, st, in, src_idx, pend_count) != TF_OK) {
|
|
134
|
+
tf_batch_free(new_pend);
|
|
135
|
+
tf_batch_free(ob);
|
|
136
|
+
return TF_ERROR;
|
|
137
|
+
}
|
|
138
|
+
if (tf_batch_expose_row(new_pend, i) != TF_OK) {
|
|
139
|
+
tf_batch_free(new_pend);
|
|
140
|
+
tf_batch_free(ob);
|
|
141
|
+
return TF_ERROR;
|
|
114
142
|
}
|
|
115
|
-
new_pend->n_rows = new_pend_count;
|
|
116
|
-
st->pending = new_pend;
|
|
117
143
|
}
|
|
118
144
|
|
|
145
|
+
if (st->pending) tf_batch_free(st->pending);
|
|
146
|
+
st->pending = new_pend;
|
|
119
147
|
*out = ob;
|
|
120
148
|
return TF_OK;
|
|
121
149
|
}
|
|
122
150
|
|
|
123
151
|
static int lead_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
124
|
-
(void)side;
|
|
125
152
|
lead_state *st = self->state;
|
|
126
153
|
*out = NULL;
|
|
127
154
|
|
|
128
155
|
if (!st->pending || st->pending->n_rows == 0) return TF_OK;
|
|
129
156
|
|
|
130
|
-
/* Emit remaining pending rows with NULL lead values */
|
|
131
157
|
tf_batch *pend = st->pending;
|
|
158
|
+
int ci = tf_batch_col_index(pend, st->column);
|
|
159
|
+
if (ci < 0) {
|
|
160
|
+
if (lead_set_error(side, "lead: buffered schema lost selected column") != TF_OK) return TF_ERROR;
|
|
161
|
+
return TF_ERROR;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
const char *extra_names[1] = {st->result};
|
|
165
|
+
tf_type extra_types[1] = {pend->col_types[ci]};
|
|
132
166
|
tf_batch *ob = tf_batch_create(pend->n_cols + 1, pend->n_rows);
|
|
133
167
|
if (!ob) return TF_ERROR;
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
168
|
+
if (tf_batch_clone_with_extra_cols(ob, pend, extra_names, extra_types, 1) != TF_OK) {
|
|
169
|
+
tf_batch_free(ob);
|
|
170
|
+
return TF_ERROR;
|
|
171
|
+
}
|
|
137
172
|
|
|
138
173
|
for (size_t r = 0; r < pend->n_rows; r++) {
|
|
139
|
-
tf_batch_copy_row(ob, r, pend, r)
|
|
140
|
-
|
|
174
|
+
if (tf_batch_copy_row(ob, r, pend, r) != TF_OK) {
|
|
175
|
+
tf_batch_free(ob);
|
|
176
|
+
return TF_ERROR;
|
|
177
|
+
}
|
|
178
|
+
if (tf_batch_set_null(ob, r, pend->n_cols) != TF_OK) {
|
|
179
|
+
tf_batch_free(ob);
|
|
180
|
+
return TF_ERROR;
|
|
181
|
+
}
|
|
182
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
183
|
+
tf_batch_free(ob);
|
|
184
|
+
return TF_ERROR;
|
|
185
|
+
}
|
|
141
186
|
}
|
|
142
|
-
ob->n_rows = pend->n_rows;
|
|
143
187
|
|
|
144
188
|
tf_batch_free(st->pending);
|
|
145
189
|
st->pending = NULL;
|
|
146
|
-
|
|
147
190
|
*out = ob;
|
|
148
191
|
return TF_OK;
|
|
149
192
|
}
|
|
@@ -162,26 +205,35 @@ static void lead_destroy(tf_step *self) {
|
|
|
162
205
|
tf_step *tf_lead_create(const cJSON *args) {
|
|
163
206
|
if (!args) return NULL;
|
|
164
207
|
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
165
|
-
if (!cJSON_IsString(col_j)) return NULL;
|
|
208
|
+
if (!cJSON_IsString(col_j) || !col_j->valuestring[0]) return NULL;
|
|
166
209
|
|
|
167
|
-
lead_state *st =
|
|
210
|
+
lead_state *st = tf_callocarray_checked(1, sizeof(lead_state));
|
|
168
211
|
if (!st) return NULL;
|
|
169
|
-
st->column =
|
|
212
|
+
st->column = tf_strdup_checked(col_j->valuestring);
|
|
213
|
+
if (!st->column) { free(st); return NULL; }
|
|
170
214
|
|
|
171
|
-
|
|
172
|
-
|
|
215
|
+
size_t offset = 1;
|
|
216
|
+
int has_offset = tf_json_get_size_arg(args, "offset",
|
|
217
|
+
1, TF_MAX_WINDOW_SIZE,
|
|
218
|
+
&offset, "lead");
|
|
219
|
+
if (has_offset < 0) { free(st->column); free(st); return NULL; }
|
|
220
|
+
st->offset = offset;
|
|
173
221
|
|
|
174
222
|
cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
|
|
175
|
-
if (cJSON_IsString(res_j)) {
|
|
176
|
-
st->result =
|
|
223
|
+
if (cJSON_IsString(res_j) && res_j->valuestring[0]) {
|
|
224
|
+
st->result = tf_strdup_checked(res_j->valuestring);
|
|
177
225
|
} else {
|
|
178
|
-
|
|
179
|
-
snprintf(buf, sizeof(buf), "%s_lead", st->column);
|
|
180
|
-
st->result = strdup(buf);
|
|
226
|
+
st->result = tf_string_append_suffix_checked(st->column, "_lead");
|
|
181
227
|
}
|
|
228
|
+
if (!st->result) { free(st->column); free(st); return NULL; }
|
|
182
229
|
|
|
183
|
-
tf_step *step =
|
|
184
|
-
if (!step) {
|
|
230
|
+
tf_step *step = tf_callocarray_checked(1, sizeof(tf_step));
|
|
231
|
+
if (!step) {
|
|
232
|
+
free(st->column);
|
|
233
|
+
free(st->result);
|
|
234
|
+
free(st);
|
|
235
|
+
return NULL;
|
|
236
|
+
}
|
|
185
237
|
step->process = lead_process;
|
|
186
238
|
step->flush = lead_flush;
|
|
187
239
|
step->destroy = lead_destroy;
|