tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_schema_infer.c -- Bounded schema inference report.
|
|
3
|
+
*
|
|
4
|
+
* schema infer [rows=N] samples decoded batches without retaining rows and
|
|
5
|
+
* emits one schema-report row per input column at finish(). It reports the
|
|
6
|
+
* decoded runtime types rather than changing parser behavior.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
#include "internal.h"
|
|
10
|
+
#include "cJSON.h"
|
|
11
|
+
#include <stdlib.h>
|
|
12
|
+
#include <string.h>
|
|
13
|
+
#include <stdio.h>
|
|
14
|
+
#include <stdint.h>
|
|
15
|
+
|
|
16
|
+
#define SCHEMA_INFER_DEFAULT_ROWS 10000u
|
|
17
|
+
|
|
18
|
+
typedef struct {
|
|
19
|
+
char *name;
|
|
20
|
+
size_t missing;
|
|
21
|
+
size_t counts[6]; /* bool, int, float, string, date, timestamp */
|
|
22
|
+
} schema_infer_col;
|
|
23
|
+
|
|
24
|
+
typedef struct {
|
|
25
|
+
size_t rows_limit;
|
|
26
|
+
size_t rows_seen;
|
|
27
|
+
size_t rows_sampled;
|
|
28
|
+
int initialized;
|
|
29
|
+
int schema_changed;
|
|
30
|
+
schema_infer_col *cols;
|
|
31
|
+
size_t n_cols;
|
|
32
|
+
} schema_infer_state;
|
|
33
|
+
|
|
34
|
+
static int schema_infer_type_index(tf_type type) {
|
|
35
|
+
switch (type) {
|
|
36
|
+
case TF_TYPE_BOOL: return 0;
|
|
37
|
+
case TF_TYPE_INT64: return 1;
|
|
38
|
+
case TF_TYPE_FLOAT64: return 2;
|
|
39
|
+
case TF_TYPE_STRING: return 3;
|
|
40
|
+
case TF_TYPE_DATE: return 4;
|
|
41
|
+
case TF_TYPE_TIMESTAMP: return 5;
|
|
42
|
+
default: return -1;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
static void append_token(char *buf, size_t buf_size, int *first, const char *token) {
|
|
47
|
+
if (!buf || buf_size == 0 || !token || !token[0]) return;
|
|
48
|
+
size_t used = strlen(buf);
|
|
49
|
+
if (used >= buf_size - 1) return;
|
|
50
|
+
snprintf(buf + used, buf_size - used, "%s%s", *first ? "" : ",", token);
|
|
51
|
+
*first = 0;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
static void append_count(char *buf, size_t buf_size, int *first,
|
|
55
|
+
const char *name, size_t count) {
|
|
56
|
+
if (count == 0 || !buf || buf_size == 0 || !name) return;
|
|
57
|
+
size_t used = strlen(buf);
|
|
58
|
+
if (used >= buf_size - 1) return;
|
|
59
|
+
snprintf(buf + used, buf_size - used, "%s%s:%zu", *first ? "" : ";", name, count);
|
|
60
|
+
*first = 0;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
static int schema_infer_init(schema_infer_state *st, const tf_batch *in) {
|
|
64
|
+
if (st->initialized) return TF_OK;
|
|
65
|
+
size_t n_cols = in ? in->n_cols : 0;
|
|
66
|
+
if (n_cols == 0) {
|
|
67
|
+
st->initialized = 1;
|
|
68
|
+
return TF_OK;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
schema_infer_col *cols = calloc(n_cols, sizeof(schema_infer_col));
|
|
72
|
+
if (!cols) return TF_ERROR;
|
|
73
|
+
for (size_t c = 0; c < n_cols; c++) {
|
|
74
|
+
const char *name = in->col_names[c] ? in->col_names[c] : "";
|
|
75
|
+
cols[c].name = strdup(name);
|
|
76
|
+
if (!cols[c].name) {
|
|
77
|
+
for (size_t i = 0; i < c; i++) free(cols[i].name);
|
|
78
|
+
free(cols);
|
|
79
|
+
return TF_ERROR;
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
st->cols = cols;
|
|
84
|
+
st->n_cols = n_cols;
|
|
85
|
+
st->initialized = 1;
|
|
86
|
+
return TF_OK;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
static int schema_infer_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
90
|
+
tf_side_channels *side) {
|
|
91
|
+
(void)side;
|
|
92
|
+
schema_infer_state *st = self->state;
|
|
93
|
+
*out = NULL;
|
|
94
|
+
if (schema_infer_init(st, in) != TF_OK) return TF_ERROR;
|
|
95
|
+
if (!in) return TF_OK;
|
|
96
|
+
if (st->initialized && in->n_cols != st->n_cols) st->schema_changed = 1;
|
|
97
|
+
|
|
98
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
99
|
+
st->rows_seen++;
|
|
100
|
+
if (st->rows_sampled >= st->rows_limit) continue;
|
|
101
|
+
st->rows_sampled++;
|
|
102
|
+
size_t n = in->n_cols < st->n_cols ? in->n_cols : st->n_cols;
|
|
103
|
+
for (size_t c = 0; c < n; c++) {
|
|
104
|
+
schema_infer_col *col = &st->cols[c];
|
|
105
|
+
if (tf_batch_is_null(in, r, c)) {
|
|
106
|
+
col->missing++;
|
|
107
|
+
continue;
|
|
108
|
+
}
|
|
109
|
+
int idx = schema_infer_type_index(in->col_types[c]);
|
|
110
|
+
if (idx >= 0) col->counts[idx]++;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
return TF_OK;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
static const char *schema_infer_inferred_type(const schema_infer_col *col,
|
|
117
|
+
const char **warning_token) {
|
|
118
|
+
size_t non_missing = 0;
|
|
119
|
+
size_t nonzero = 0;
|
|
120
|
+
int last = -1;
|
|
121
|
+
*warning_token = NULL;
|
|
122
|
+
for (int i = 0; i < 6; i++) {
|
|
123
|
+
non_missing += col->counts[i];
|
|
124
|
+
if (col->counts[i] > 0) { nonzero++; last = i; }
|
|
125
|
+
}
|
|
126
|
+
if (non_missing == 0) {
|
|
127
|
+
*warning_token = "no_non_null_sample";
|
|
128
|
+
return "unknown";
|
|
129
|
+
}
|
|
130
|
+
if (nonzero == 1) {
|
|
131
|
+
static const char *names[] = {"bool", "int", "float", "string", "date", "timestamp"};
|
|
132
|
+
return names[last];
|
|
133
|
+
}
|
|
134
|
+
if (col->counts[1] > 0 && col->counts[2] > 0 && nonzero == 2) {
|
|
135
|
+
*warning_token = "mixed_numeric";
|
|
136
|
+
return "float";
|
|
137
|
+
}
|
|
138
|
+
*warning_token = "mixed_types";
|
|
139
|
+
return "string";
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
static int schema_infer_set_output_schema(tf_batch *out) {
|
|
143
|
+
return tf_batch_set_schema(out, 0, "column", TF_TYPE_STRING) == TF_OK &&
|
|
144
|
+
tf_batch_set_schema(out, 1, "type", TF_TYPE_STRING) == TF_OK &&
|
|
145
|
+
tf_batch_set_schema(out, 2, "nullable", TF_TYPE_BOOL) == TF_OK &&
|
|
146
|
+
tf_batch_set_schema(out, 3, "non_null", TF_TYPE_BOOL) == TF_OK &&
|
|
147
|
+
tf_batch_set_schema(out, 4, "rows_seen", TF_TYPE_INT64) == TF_OK &&
|
|
148
|
+
tf_batch_set_schema(out, 5, "rows_sampled", TF_TYPE_INT64) == TF_OK &&
|
|
149
|
+
tf_batch_set_schema(out, 6, "missing", TF_TYPE_INT64) == TF_OK &&
|
|
150
|
+
tf_batch_set_schema(out, 7, "non_missing", TF_TYPE_INT64) == TF_OK &&
|
|
151
|
+
tf_batch_set_schema(out, 8, "observed_types", TF_TYPE_STRING) == TF_OK &&
|
|
152
|
+
tf_batch_set_schema(out, 9, "warning", TF_TYPE_STRING) == TF_OK ? TF_OK : TF_ERROR;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
static int schema_infer_flush(tf_step *self, tf_batch **out,
|
|
156
|
+
tf_side_channels *side) {
|
|
157
|
+
(void)side;
|
|
158
|
+
schema_infer_state *st = self->state;
|
|
159
|
+
*out = NULL;
|
|
160
|
+
if (!st->initialized || st->n_cols == 0) return TF_OK;
|
|
161
|
+
|
|
162
|
+
tf_batch *ob = tf_batch_create(10, st->n_cols ? st->n_cols : 1);
|
|
163
|
+
if (!ob) return TF_ERROR;
|
|
164
|
+
int rc = TF_ERROR;
|
|
165
|
+
#define SCHEMA_INFER_WRITE(expr) do { if ((expr) != TF_OK) goto done; } while (0)
|
|
166
|
+
|
|
167
|
+
SCHEMA_INFER_WRITE(schema_infer_set_output_schema(ob));
|
|
168
|
+
|
|
169
|
+
static const char *type_names[] = {"bool", "int", "float", "string", "date", "timestamp"};
|
|
170
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
171
|
+
schema_infer_col *col = &st->cols[c];
|
|
172
|
+
size_t non_missing = 0;
|
|
173
|
+
char observed[256] = {0};
|
|
174
|
+
char warning[192] = {0};
|
|
175
|
+
int first_obs = 1;
|
|
176
|
+
int first_warn = 1;
|
|
177
|
+
for (int i = 0; i < 6; i++) {
|
|
178
|
+
non_missing += col->counts[i];
|
|
179
|
+
append_count(observed, sizeof(observed), &first_obs, type_names[i], col->counts[i]);
|
|
180
|
+
}
|
|
181
|
+
append_count(observed, sizeof(observed), &first_obs, "null", col->missing);
|
|
182
|
+
if (observed[0] == '\0') snprintf(observed, sizeof(observed), "none");
|
|
183
|
+
|
|
184
|
+
const char *type_warning = NULL;
|
|
185
|
+
const char *inferred = schema_infer_inferred_type(col, &type_warning);
|
|
186
|
+
if (type_warning) append_token(warning, sizeof(warning), &first_warn, type_warning);
|
|
187
|
+
if (st->rows_seen > st->rows_sampled) append_token(warning, sizeof(warning), &first_warn, "sample_limited");
|
|
188
|
+
if (st->schema_changed) append_token(warning, sizeof(warning), &first_warn, "schema_changed");
|
|
189
|
+
if (st->rows_sampled == 0) append_token(warning, sizeof(warning), &first_warn, "no_rows_sampled");
|
|
190
|
+
|
|
191
|
+
SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 0, col->name ? col->name : ""));
|
|
192
|
+
SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 1, inferred));
|
|
193
|
+
SCHEMA_INFER_WRITE(tf_batch_set_bool(ob, c, 2, col->missing > 0));
|
|
194
|
+
SCHEMA_INFER_WRITE(tf_batch_set_bool(ob, c, 3, col->missing == 0 && st->rows_sampled > 0));
|
|
195
|
+
SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 4, (int64_t)st->rows_seen));
|
|
196
|
+
SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 5, (int64_t)st->rows_sampled));
|
|
197
|
+
SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 6, (int64_t)col->missing));
|
|
198
|
+
SCHEMA_INFER_WRITE(tf_batch_set_int64(ob, c, 7, (int64_t)non_missing));
|
|
199
|
+
SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 8, observed));
|
|
200
|
+
SCHEMA_INFER_WRITE(tf_batch_set_string(ob, c, 9, warning));
|
|
201
|
+
SCHEMA_INFER_WRITE(tf_batch_expose_row(ob, c));
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
*out = ob;
|
|
205
|
+
ob = NULL;
|
|
206
|
+
rc = TF_OK;
|
|
207
|
+
|
|
208
|
+
done:
|
|
209
|
+
if (ob) tf_batch_free(ob);
|
|
210
|
+
#undef SCHEMA_INFER_WRITE
|
|
211
|
+
return rc;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
static void schema_infer_destroy(tf_step *self) {
|
|
215
|
+
if (!self) return;
|
|
216
|
+
schema_infer_state *st = self->state;
|
|
217
|
+
if (st) {
|
|
218
|
+
for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i].name);
|
|
219
|
+
free(st->cols);
|
|
220
|
+
free(st);
|
|
221
|
+
}
|
|
222
|
+
free(self);
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
static int parse_rows_arg(const cJSON *args, size_t *rows) {
|
|
226
|
+
*rows = SCHEMA_INFER_DEFAULT_ROWS;
|
|
227
|
+
if (!args) return TF_OK;
|
|
228
|
+
int has_rows = tf_json_get_size_arg(args, "rows",
|
|
229
|
+
1, TF_MAX_COUNT_ARG,
|
|
230
|
+
rows, "schema-infer");
|
|
231
|
+
if (has_rows < 0) {
|
|
232
|
+
return TF_ERROR;
|
|
233
|
+
}
|
|
234
|
+
if (has_rows == 0) *rows = SCHEMA_INFER_DEFAULT_ROWS;
|
|
235
|
+
return TF_OK;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
tf_step *tf_schema_infer_create(const cJSON *args) {
|
|
239
|
+
schema_infer_state *st = calloc(1, sizeof(schema_infer_state));
|
|
240
|
+
if (!st) return NULL;
|
|
241
|
+
if (parse_rows_arg(args, &st->rows_limit) != TF_OK) {
|
|
242
|
+
free(st);
|
|
243
|
+
return NULL;
|
|
244
|
+
}
|
|
245
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
246
|
+
if (!step) { free(st); return NULL; }
|
|
247
|
+
step->process = schema_infer_process;
|
|
248
|
+
step->flush = schema_infer_flush;
|
|
249
|
+
step->destroy = schema_infer_destroy;
|
|
250
|
+
step->state = st;
|
|
251
|
+
return step;
|
|
252
|
+
}
|
package/csrc/op_select.c
CHANGED
|
@@ -14,76 +14,75 @@
|
|
|
14
14
|
typedef struct {
|
|
15
15
|
char **col_names;
|
|
16
16
|
size_t n_cols;
|
|
17
|
+
int use_selector_syntax;
|
|
17
18
|
} select_state;
|
|
18
19
|
|
|
20
|
+
static int select_write_error(tf_side_channels *side, const char *msg) {
|
|
21
|
+
if (!side || !side->errors || !msg) return TF_OK;
|
|
22
|
+
char buf[256];
|
|
23
|
+
snprintf(buf, sizeof(buf), "{\"op\":\"select\",\"error\":\"%s\"}", msg);
|
|
24
|
+
return tf_buffer_write_line(side->errors, buf);
|
|
25
|
+
}
|
|
26
|
+
|
|
19
27
|
static int select_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
20
28
|
tf_side_channels *side) {
|
|
21
29
|
select_state *st = self->state;
|
|
22
30
|
*out = NULL;
|
|
23
31
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
32
|
+
int *indices = NULL;
|
|
33
|
+
size_t n_indices = st->n_cols;
|
|
34
|
+
|
|
35
|
+
if (st->use_selector_syntax) {
|
|
36
|
+
char *error = NULL;
|
|
37
|
+
if (tf_column_selectors_resolve(st->col_names, st->n_cols,
|
|
38
|
+
in->col_names, in->col_types, in->n_cols,
|
|
39
|
+
&indices, &n_indices, &error) != TF_OK) {
|
|
40
|
+
int err_rc = select_write_error(side, error ? error : "selector resolution failed");
|
|
41
|
+
free(error);
|
|
42
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
43
|
+
return TF_ERROR;
|
|
44
|
+
}
|
|
45
|
+
} else {
|
|
46
|
+
indices = malloc(st->n_cols * sizeof(int));
|
|
47
|
+
if (!indices) return TF_ERROR;
|
|
48
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
49
|
+
indices[i] = tf_batch_col_index(in, st->col_names[i]);
|
|
50
|
+
if (indices[i] < 0 && side && side->errors) {
|
|
51
|
+
char msg[128];
|
|
52
|
+
snprintf(msg, sizeof(msg), "column '%s' not found", st->col_names[i]);
|
|
53
|
+
if (select_write_error(side, msg) != TF_OK) {
|
|
54
|
+
free(indices);
|
|
55
|
+
return TF_ERROR;
|
|
56
|
+
}
|
|
57
|
+
}
|
|
36
58
|
}
|
|
37
59
|
}
|
|
38
60
|
|
|
39
|
-
|
|
40
|
-
tf_batch *ob = tf_batch_create(st->n_cols, in->n_rows);
|
|
61
|
+
tf_batch *ob = tf_batch_create(n_indices, in->n_rows);
|
|
41
62
|
if (!ob) { free(indices); return TF_ERROR; }
|
|
42
63
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
64
|
+
size_t *selected_cols = malloc(n_indices * sizeof(size_t));
|
|
65
|
+
if (!selected_cols) { free(indices); tf_batch_free(ob); return TF_ERROR; }
|
|
66
|
+
for (size_t i = 0; i < n_indices; i++) {
|
|
67
|
+
int ci = indices[i];
|
|
68
|
+
if (ci >= 0) {
|
|
69
|
+
selected_cols[i] = (size_t)ci;
|
|
70
|
+
if (tf_batch_set_schema(ob, i, in->col_names[(size_t)ci], in->col_types[(size_t)ci]) != TF_OK) {
|
|
71
|
+
free(selected_cols); free(indices); tf_batch_free(ob); return TF_ERROR;
|
|
72
|
+
}
|
|
46
73
|
} else {
|
|
47
|
-
|
|
74
|
+
selected_cols[i] = SIZE_MAX;
|
|
75
|
+
if (tf_batch_set_schema(ob, i, st->col_names[i], TF_TYPE_NULL) != TF_OK) {
|
|
76
|
+
free(selected_cols); free(indices); tf_batch_free(ob); return TF_ERROR;
|
|
77
|
+
}
|
|
48
78
|
}
|
|
49
79
|
}
|
|
50
80
|
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
tf_batch_ensure_capacity(ob, r + 1);
|
|
54
|
-
for (size_t i = 0; i < st->n_cols; i++) {
|
|
55
|
-
int ci = indices[i];
|
|
56
|
-
if (ci < 0 || tf_batch_is_null(in, r, ci)) {
|
|
57
|
-
tf_batch_set_null(ob, r, i);
|
|
58
|
-
continue;
|
|
59
|
-
}
|
|
60
|
-
switch (in->col_types[ci]) {
|
|
61
|
-
case TF_TYPE_BOOL:
|
|
62
|
-
tf_batch_set_bool(ob, r, i, tf_batch_get_bool(in, r, ci));
|
|
63
|
-
break;
|
|
64
|
-
case TF_TYPE_INT64:
|
|
65
|
-
tf_batch_set_int64(ob, r, i, tf_batch_get_int64(in, r, ci));
|
|
66
|
-
break;
|
|
67
|
-
case TF_TYPE_FLOAT64:
|
|
68
|
-
tf_batch_set_float64(ob, r, i, tf_batch_get_float64(in, r, ci));
|
|
69
|
-
break;
|
|
70
|
-
case TF_TYPE_STRING:
|
|
71
|
-
tf_batch_set_string(ob, r, i, tf_batch_get_string(in, r, ci));
|
|
72
|
-
break;
|
|
73
|
-
case TF_TYPE_DATE:
|
|
74
|
-
tf_batch_set_date(ob, r, i, tf_batch_get_date(in, r, ci));
|
|
75
|
-
break;
|
|
76
|
-
case TF_TYPE_TIMESTAMP:
|
|
77
|
-
tf_batch_set_timestamp(ob, r, i, tf_batch_get_timestamp(in, r, ci));
|
|
78
|
-
break;
|
|
79
|
-
default:
|
|
80
|
-
tf_batch_set_null(ob, r, i);
|
|
81
|
-
break;
|
|
82
|
-
}
|
|
83
|
-
}
|
|
84
|
-
ob->n_rows = r + 1;
|
|
81
|
+
if (tf_batch_copy_selected_columns(ob, in, selected_cols, n_indices) != TF_OK) {
|
|
82
|
+
free(selected_cols); free(indices); tf_batch_free(ob); return TF_ERROR;
|
|
85
83
|
}
|
|
86
84
|
|
|
85
|
+
free(selected_cols);
|
|
87
86
|
free(indices);
|
|
88
87
|
*out = ob;
|
|
89
88
|
return TF_OK;
|
|
@@ -95,13 +94,19 @@ static int select_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
|
95
94
|
return TF_OK;
|
|
96
95
|
}
|
|
97
96
|
|
|
98
|
-
static void
|
|
99
|
-
select_state *st = self->state;
|
|
97
|
+
static void select_state_free(select_state *st) {
|
|
100
98
|
if (st) {
|
|
101
|
-
|
|
99
|
+
if (st->col_names) {
|
|
100
|
+
for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
|
|
101
|
+
}
|
|
102
102
|
free(st->col_names);
|
|
103
103
|
free(st);
|
|
104
104
|
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
static void select_destroy(tf_step *self) {
|
|
108
|
+
if (!self) return;
|
|
109
|
+
select_state_free(self->state);
|
|
105
110
|
free(self);
|
|
106
111
|
}
|
|
107
112
|
|
|
@@ -116,25 +121,220 @@ tf_step *tf_select_create(const cJSON *args) {
|
|
|
116
121
|
select_state *st = calloc(1, sizeof(select_state));
|
|
117
122
|
if (!st) return NULL;
|
|
118
123
|
st->n_cols = (size_t)n;
|
|
119
|
-
st->col_names =
|
|
120
|
-
if (!st->col_names) {
|
|
124
|
+
st->col_names = calloc((size_t)n, sizeof(char *));
|
|
125
|
+
if (!st->col_names) { select_state_free(st); return NULL; }
|
|
121
126
|
|
|
122
127
|
for (int i = 0; i < n; i++) {
|
|
123
128
|
cJSON *item = cJSON_GetArrayItem(cols, i);
|
|
124
|
-
if (!cJSON_IsString(item)) {
|
|
125
|
-
for (int j = 0; j < i; j++) free(st->col_names[j]);
|
|
126
|
-
free(st->col_names);
|
|
127
|
-
free(st);
|
|
128
|
-
return NULL;
|
|
129
|
-
}
|
|
129
|
+
if (!cJSON_IsString(item)) { select_state_free(st); return NULL; }
|
|
130
130
|
st->col_names[i] = strdup(item->valuestring);
|
|
131
|
+
if (!st->col_names[i]) { select_state_free(st); return NULL; }
|
|
132
|
+
if (tf_column_selector_has_syntax(item->valuestring)) st->use_selector_syntax = 1;
|
|
131
133
|
}
|
|
132
134
|
|
|
133
|
-
tf_step *step =
|
|
134
|
-
if (!step) {
|
|
135
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
136
|
+
if (!step) { select_state_free(st); return NULL; }
|
|
135
137
|
step->process = select_process;
|
|
136
138
|
step->flush = select_flush;
|
|
137
139
|
step->destroy = select_destroy;
|
|
138
140
|
step->state = st;
|
|
139
141
|
return step;
|
|
140
142
|
}
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
typedef struct {
|
|
146
|
+
char **move_names;
|
|
147
|
+
size_t n_move;
|
|
148
|
+
char *before;
|
|
149
|
+
char *after;
|
|
150
|
+
} relocate_state;
|
|
151
|
+
|
|
152
|
+
static int relocate_write_error(tf_side_channels *side, const char *msg, const char *name) {
|
|
153
|
+
if (!side || !side->errors) return TF_OK;
|
|
154
|
+
char buf[256];
|
|
155
|
+
snprintf(buf, sizeof(buf),
|
|
156
|
+
"{\"op\":\"relocate\",\"error\":\"%s '%s'\"}",
|
|
157
|
+
msg, name ? name : "");
|
|
158
|
+
return tf_buffer_write_line(side->errors, buf);
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
static int relocate_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
162
|
+
tf_side_channels *side) {
|
|
163
|
+
relocate_state *st = self->state;
|
|
164
|
+
*out = NULL;
|
|
165
|
+
|
|
166
|
+
size_t n_in = in->n_cols;
|
|
167
|
+
int *move_idx = NULL;
|
|
168
|
+
size_t n_move = 0;
|
|
169
|
+
char *selector_error = NULL;
|
|
170
|
+
if (tf_column_selectors_resolve(st->move_names, st->n_move,
|
|
171
|
+
in->col_names, in->col_types, in->n_cols,
|
|
172
|
+
&move_idx, &n_move, &selector_error) != TF_OK) {
|
|
173
|
+
int err_rc = relocate_write_error(side, selector_error ? selector_error : "selector resolution failed", "");
|
|
174
|
+
free(selector_error);
|
|
175
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
176
|
+
return TF_ERROR;
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
int *is_moving = calloc(n_in ? n_in : 1, sizeof(int));
|
|
180
|
+
int *order = malloc(n_in ? n_in * sizeof(int) : sizeof(int));
|
|
181
|
+
if (!is_moving || !order) {
|
|
182
|
+
free(move_idx); free(is_moving); free(order);
|
|
183
|
+
return TF_ERROR;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
for (size_t i = 0; i < n_move; i++) {
|
|
187
|
+
int ci = move_idx[i];
|
|
188
|
+
if (ci < 0 || (size_t)ci >= n_in || is_moving[ci]) {
|
|
189
|
+
int err_rc = relocate_write_error(side, "invalid relocated column", "");
|
|
190
|
+
free(move_idx); free(is_moving); free(order);
|
|
191
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
192
|
+
return TF_ERROR;
|
|
193
|
+
}
|
|
194
|
+
is_moving[ci] = 1;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
const char *anchor = st->before ? st->before : st->after;
|
|
198
|
+
int anchor_idx = -1;
|
|
199
|
+
if (anchor) {
|
|
200
|
+
anchor_idx = tf_batch_col_index(in, anchor);
|
|
201
|
+
if (anchor_idx < 0) {
|
|
202
|
+
int err_rc = relocate_write_error(side, "anchor column not found", anchor);
|
|
203
|
+
free(move_idx); free(is_moving); free(order);
|
|
204
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
205
|
+
return TF_ERROR;
|
|
206
|
+
}
|
|
207
|
+
if (is_moving[anchor_idx]) {
|
|
208
|
+
int err_rc = relocate_write_error(side, "anchor column is being relocated", anchor);
|
|
209
|
+
free(move_idx); free(is_moving); free(order);
|
|
210
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
211
|
+
return TF_ERROR;
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
size_t n_order = 0;
|
|
216
|
+
if (!anchor) {
|
|
217
|
+
for (size_t i = 0; i < n_move; i++) order[n_order++] = move_idx[i];
|
|
218
|
+
for (size_t i = 0; i < n_in; i++) {
|
|
219
|
+
if (!is_moving[i]) order[n_order++] = (int)i;
|
|
220
|
+
}
|
|
221
|
+
} else {
|
|
222
|
+
for (size_t i = 0; i < n_in; i++) {
|
|
223
|
+
if (is_moving[i]) continue;
|
|
224
|
+
if (st->before && (int)i == anchor_idx) {
|
|
225
|
+
for (size_t j = 0; j < n_move; j++) order[n_order++] = move_idx[j];
|
|
226
|
+
}
|
|
227
|
+
order[n_order++] = (int)i;
|
|
228
|
+
if (st->after && (int)i == anchor_idx) {
|
|
229
|
+
for (size_t j = 0; j < n_move; j++) order[n_order++] = move_idx[j];
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
if (n_order != n_in) {
|
|
235
|
+
int err_rc = relocate_write_error(side, "internal order size mismatch", "");
|
|
236
|
+
free(move_idx); free(is_moving); free(order);
|
|
237
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
238
|
+
return TF_ERROR;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
tf_batch *ob = tf_batch_create(n_in, in->n_rows);
|
|
242
|
+
if (!ob) {
|
|
243
|
+
free(move_idx); free(is_moving); free(order);
|
|
244
|
+
return TF_ERROR;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
size_t *selected_cols = malloc(n_in * sizeof(size_t));
|
|
248
|
+
if (!selected_cols) {
|
|
249
|
+
free(move_idx); free(is_moving); free(order); tf_batch_free(ob);
|
|
250
|
+
return TF_ERROR;
|
|
251
|
+
}
|
|
252
|
+
for (size_t i = 0; i < n_in; i++) {
|
|
253
|
+
int ci = order[i];
|
|
254
|
+
selected_cols[i] = (size_t)ci;
|
|
255
|
+
if (tf_batch_set_schema(ob, i, in->col_names[(size_t)ci], in->col_types[(size_t)ci]) != TF_OK) {
|
|
256
|
+
free(selected_cols); free(move_idx); free(is_moving); free(order); tf_batch_free(ob);
|
|
257
|
+
return TF_ERROR;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
if (tf_batch_copy_selected_columns(ob, in, selected_cols, n_in) != TF_OK) {
|
|
262
|
+
free(selected_cols); free(move_idx); free(is_moving); free(order); tf_batch_free(ob);
|
|
263
|
+
return TF_ERROR;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
free(selected_cols);
|
|
267
|
+
free(move_idx);
|
|
268
|
+
free(is_moving);
|
|
269
|
+
free(order);
|
|
270
|
+
*out = ob;
|
|
271
|
+
return TF_OK;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
static int relocate_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
275
|
+
(void)self; (void)side;
|
|
276
|
+
*out = NULL;
|
|
277
|
+
return TF_OK;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
static void relocate_state_free(relocate_state *st) {
|
|
281
|
+
if (st) {
|
|
282
|
+
if (st->move_names) {
|
|
283
|
+
for (size_t i = 0; i < st->n_move; i++) free(st->move_names[i]);
|
|
284
|
+
}
|
|
285
|
+
free(st->move_names);
|
|
286
|
+
free(st->before);
|
|
287
|
+
free(st->after);
|
|
288
|
+
free(st);
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
static void relocate_destroy(tf_step *self) {
|
|
293
|
+
if (!self) return;
|
|
294
|
+
relocate_state_free(self->state);
|
|
295
|
+
free(self);
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
tf_step *tf_relocate_create(const cJSON *args) {
|
|
299
|
+
if (!args) return NULL;
|
|
300
|
+
cJSON *cols = cJSON_GetObjectItemCaseSensitive(args, "columns");
|
|
301
|
+
if (!cJSON_IsArray(cols)) return NULL;
|
|
302
|
+
|
|
303
|
+
cJSON *before = cJSON_GetObjectItemCaseSensitive(args, "before");
|
|
304
|
+
cJSON *after = cJSON_GetObjectItemCaseSensitive(args, "after");
|
|
305
|
+
if (before && after) return NULL;
|
|
306
|
+
if (before && !cJSON_IsString(before)) return NULL;
|
|
307
|
+
if (after && !cJSON_IsString(after)) return NULL;
|
|
308
|
+
|
|
309
|
+
int n = cJSON_GetArraySize(cols);
|
|
310
|
+
if (n <= 0) return NULL;
|
|
311
|
+
|
|
312
|
+
relocate_state *st = calloc(1, sizeof(relocate_state));
|
|
313
|
+
if (!st) return NULL;
|
|
314
|
+
st->n_move = (size_t)n;
|
|
315
|
+
st->move_names = calloc((size_t)n, sizeof(char *));
|
|
316
|
+
if (!st->move_names) { relocate_state_free(st); return NULL; }
|
|
317
|
+
|
|
318
|
+
for (int i = 0; i < n; i++) {
|
|
319
|
+
cJSON *item = cJSON_GetArrayItem(cols, i);
|
|
320
|
+
if (!cJSON_IsString(item)) { relocate_state_free(st); return NULL; }
|
|
321
|
+
st->move_names[i] = strdup(item->valuestring);
|
|
322
|
+
if (!st->move_names[i]) { relocate_state_free(st); return NULL; }
|
|
323
|
+
}
|
|
324
|
+
if (before) {
|
|
325
|
+
st->before = strdup(before->valuestring);
|
|
326
|
+
if (!st->before) { relocate_state_free(st); return NULL; }
|
|
327
|
+
}
|
|
328
|
+
if (after) {
|
|
329
|
+
st->after = strdup(after->valuestring);
|
|
330
|
+
if (!st->after) { relocate_state_free(st); return NULL; }
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
334
|
+
if (!step) { relocate_state_free(st); return NULL; }
|
|
335
|
+
step->process = relocate_process;
|
|
336
|
+
step->flush = relocate_flush;
|
|
337
|
+
step->destroy = relocate_destroy;
|
|
338
|
+
step->state = st;
|
|
339
|
+
return step;
|
|
340
|
+
}
|