tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/csrc/op_stack.c
CHANGED
|
@@ -19,271 +19,216 @@
|
|
|
19
19
|
|
|
20
20
|
typedef struct {
|
|
21
21
|
char *file_path;
|
|
22
|
+
char *validated_file_path;
|
|
22
23
|
char *tag_col; /* NULL if no tag */
|
|
23
24
|
char *tag_value; /* value for appended rows */
|
|
24
25
|
char *tag_value_in; /* value for passthrough rows (filename or "input") */
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
size_t
|
|
30
|
-
int
|
|
31
|
-
int
|
|
26
|
+
FILE *file;
|
|
27
|
+
tf_decoder *decoder;
|
|
28
|
+
tf_batch **pending_batches;
|
|
29
|
+
size_t n_pending_batches;
|
|
30
|
+
size_t pending_batch_index;
|
|
31
|
+
int file_started;
|
|
32
|
+
int file_done;
|
|
32
33
|
} stack_state;
|
|
33
34
|
|
|
35
|
+
static void stack_clear_pending(stack_state *st) {
|
|
36
|
+
if (!st || !st->pending_batches) return;
|
|
37
|
+
tf_batch_array_free_items(st->pending_batches + st->pending_batch_index,
|
|
38
|
+
st->n_pending_batches - st->pending_batch_index);
|
|
39
|
+
free(st->pending_batches);
|
|
40
|
+
st->pending_batches = NULL;
|
|
41
|
+
st->n_pending_batches = 0;
|
|
42
|
+
st->pending_batch_index = 0;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
static void stack_close_file_reader(stack_state *st) {
|
|
46
|
+
if (!st) return;
|
|
47
|
+
stack_clear_pending(st);
|
|
48
|
+
if (st->decoder) {
|
|
49
|
+
st->decoder->destroy(st->decoder);
|
|
50
|
+
st->decoder = NULL;
|
|
51
|
+
}
|
|
52
|
+
if (st->file) {
|
|
53
|
+
fclose(st->file);
|
|
54
|
+
st->file = NULL;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
static void stack_state_free(stack_state *st) {
|
|
59
|
+
if (!st) return;
|
|
60
|
+
stack_close_file_reader(st);
|
|
61
|
+
free(st->file_path);
|
|
62
|
+
free(st->validated_file_path);
|
|
63
|
+
free(st->tag_col);
|
|
64
|
+
free(st->tag_value);
|
|
65
|
+
free(st->tag_value_in);
|
|
66
|
+
free(st);
|
|
67
|
+
}
|
|
68
|
+
|
|
34
69
|
static void stack_destroy(tf_step *self) {
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
70
|
+
if (self) stack_state_free(self->state);
|
|
71
|
+
free(self);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
TF_WARN_UNUSED static int stack_write_error(tf_side_channels *side, const char *msg) {
|
|
75
|
+
return tf_side_write_error(side, msg);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
static int stack_open_file_reader(stack_state *st, tf_side_channels *side) {
|
|
79
|
+
if (st->file_started) return TF_OK;
|
|
80
|
+
st->file_started = 1;
|
|
81
|
+
st->file = tf_policy_fopen_read(st->file_path, st->validated_file_path);
|
|
82
|
+
if (!st->file) {
|
|
83
|
+
if (stack_write_error(side, "stack: cannot open file") != TF_OK) return TF_ERROR;
|
|
84
|
+
return TF_ERROR;
|
|
85
|
+
}
|
|
86
|
+
st->decoder = tf_csv_decoder_create(NULL);
|
|
87
|
+
if (!st->decoder) {
|
|
88
|
+
stack_close_file_reader(st);
|
|
89
|
+
return TF_ERROR;
|
|
90
|
+
}
|
|
91
|
+
return TF_OK;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
static int stack_take_pending_batch(stack_state *st, tf_batch **out) {
|
|
95
|
+
if (!st->pending_batches || st->pending_batch_index >= st->n_pending_batches) {
|
|
96
|
+
stack_clear_pending(st);
|
|
97
|
+
return 0;
|
|
98
|
+
}
|
|
99
|
+
*out = st->pending_batches[st->pending_batch_index];
|
|
100
|
+
st->pending_batches[st->pending_batch_index] = NULL;
|
|
101
|
+
st->pending_batch_index++;
|
|
102
|
+
if (st->pending_batch_index >= st->n_pending_batches) stack_clear_pending(st);
|
|
103
|
+
return 1;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
static int stack_next_decoded_file_batch(stack_state *st, tf_batch **out,
|
|
107
|
+
tf_side_channels *side) {
|
|
108
|
+
*out = NULL;
|
|
109
|
+
if (stack_take_pending_batch(st, out)) return TF_OK;
|
|
110
|
+
if (st->file_done) return TF_OK;
|
|
111
|
+
if (stack_open_file_reader(st, side) != TF_OK) return TF_ERROR;
|
|
112
|
+
|
|
113
|
+
for (;;) {
|
|
114
|
+
uint8_t buf[64 * 1024];
|
|
115
|
+
size_t n = fread(buf, 1, sizeof(buf), st->file);
|
|
116
|
+
tf_batch **batches = NULL;
|
|
117
|
+
size_t n_batches = 0;
|
|
118
|
+
int rc = TF_OK;
|
|
119
|
+
|
|
120
|
+
if (n > 0) {
|
|
121
|
+
rc = st->decoder->decode(st->decoder, buf, n, &batches, &n_batches, side);
|
|
122
|
+
} else {
|
|
123
|
+
if (ferror(st->file)) {
|
|
124
|
+
int err_rc = stack_write_error(side, "stack: failed reading file");
|
|
125
|
+
stack_close_file_reader(st);
|
|
126
|
+
st->file_done = 1;
|
|
127
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
128
|
+
return TF_ERROR;
|
|
129
|
+
}
|
|
130
|
+
rc = st->decoder->flush(st->decoder, &batches, &n_batches, side);
|
|
131
|
+
st->file_done = 1;
|
|
132
|
+
stack_close_file_reader(st);
|
|
133
|
+
}
|
|
134
|
+
if (rc != TF_OK) {
|
|
135
|
+
tf_batch_array_free(batches, n_batches);
|
|
136
|
+
stack_close_file_reader(st);
|
|
137
|
+
st->file_done = 1;
|
|
138
|
+
return TF_ERROR;
|
|
44
139
|
}
|
|
45
|
-
|
|
46
|
-
|
|
140
|
+
st->pending_batches = batches;
|
|
141
|
+
st->n_pending_batches = n_batches;
|
|
142
|
+
st->pending_batch_index = 0;
|
|
143
|
+
if (stack_take_pending_batch(st, out)) return TF_OK;
|
|
144
|
+
if (st->file_done) return TF_OK;
|
|
47
145
|
}
|
|
48
|
-
free(self);
|
|
49
146
|
}
|
|
50
147
|
|
|
51
148
|
/*
|
|
52
149
|
* Add tag column to a batch. Returns a new batch with the tag column prepended.
|
|
53
150
|
*/
|
|
54
151
|
static tf_batch *add_tag_column(tf_batch *in, const char *tag_col, const char *tag_value) {
|
|
55
|
-
|
|
152
|
+
size_t out_cols = 0;
|
|
153
|
+
if (tf_size_add(in->n_cols, 1, &out_cols) != TF_OK) return NULL;
|
|
154
|
+
tf_batch *out = tf_batch_create(out_cols, in->n_rows);
|
|
56
155
|
if (!out) return NULL;
|
|
57
156
|
|
|
58
|
-
|
|
59
|
-
tf_batch_set_schema(out, 0, tag_col, TF_TYPE_STRING);
|
|
60
|
-
/* Copy remaining columns */
|
|
157
|
+
if (tf_batch_set_schema(out, 0, tag_col, TF_TYPE_STRING) != TF_OK) goto fail;
|
|
61
158
|
for (size_t c = 0; c < in->n_cols; c++) {
|
|
62
|
-
tf_batch_set_schema(out, c + 1, in->col_names[c], in->col_types[c]);
|
|
159
|
+
if (tf_batch_set_schema(out, c + 1, in->col_names[c], in->col_types[c]) != TF_OK) goto fail;
|
|
63
160
|
}
|
|
64
161
|
|
|
65
162
|
for (size_t r = 0; r < in->n_rows; r++) {
|
|
66
|
-
tf_batch_ensure_capacity(out, r + 1);
|
|
67
|
-
tf_batch_set_string(out, r, 0, tag_value);
|
|
163
|
+
if (tf_batch_ensure_capacity(out, r + 1) != TF_OK) goto fail;
|
|
164
|
+
if (tf_batch_set_string(out, r, 0, tag_value ? tag_value : "") != TF_OK) goto fail;
|
|
68
165
|
for (size_t c = 0; c < in->n_cols; c++) {
|
|
69
|
-
if (
|
|
70
|
-
tf_batch_set_null(out, r, c + 1);
|
|
71
|
-
} else {
|
|
72
|
-
switch (in->col_types[c]) {
|
|
73
|
-
case TF_TYPE_INT64:
|
|
74
|
-
tf_batch_set_int64(out, r, c + 1, tf_batch_get_int64(in, r, c));
|
|
75
|
-
break;
|
|
76
|
-
case TF_TYPE_FLOAT64:
|
|
77
|
-
tf_batch_set_float64(out, r, c + 1, tf_batch_get_float64(in, r, c));
|
|
78
|
-
break;
|
|
79
|
-
case TF_TYPE_STRING:
|
|
80
|
-
tf_batch_set_string(out, r, c + 1, tf_batch_get_string(in, r, c));
|
|
81
|
-
break;
|
|
82
|
-
case TF_TYPE_BOOL:
|
|
83
|
-
tf_batch_set_bool(out, r, c + 1, tf_batch_get_bool(in, r, c));
|
|
84
|
-
break;
|
|
85
|
-
case TF_TYPE_DATE:
|
|
86
|
-
tf_batch_set_date(out, r, c + 1, tf_batch_get_date(in, r, c));
|
|
87
|
-
break;
|
|
88
|
-
case TF_TYPE_TIMESTAMP:
|
|
89
|
-
tf_batch_set_timestamp(out, r, c + 1, tf_batch_get_timestamp(in, r, c));
|
|
90
|
-
break;
|
|
91
|
-
default:
|
|
92
|
-
tf_batch_set_null(out, r, c + 1);
|
|
93
|
-
break;
|
|
94
|
-
}
|
|
95
|
-
}
|
|
166
|
+
if (tf_batch_copy_cell(out, r, c + 1, in, r, c) != TF_OK) goto fail;
|
|
96
167
|
}
|
|
97
|
-
out
|
|
168
|
+
if (tf_batch_expose_row(out, r) != TF_OK) goto fail;
|
|
98
169
|
}
|
|
99
170
|
return out;
|
|
171
|
+
|
|
172
|
+
fail:
|
|
173
|
+
tf_batch_free(out);
|
|
174
|
+
return NULL;
|
|
100
175
|
}
|
|
101
176
|
|
|
102
177
|
static int stack_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
103
178
|
tf_side_channels *side) {
|
|
104
179
|
stack_state *st = self->state;
|
|
105
180
|
(void)side;
|
|
106
|
-
|
|
107
|
-
/* Capture schema from first batch */
|
|
108
|
-
if (!st->schema_captured && in->n_cols > 0) {
|
|
109
|
-
st->n_cols = in->n_cols;
|
|
110
|
-
st->col_names = malloc(in->n_cols * sizeof(char *));
|
|
111
|
-
st->col_types = malloc(in->n_cols * sizeof(tf_type));
|
|
112
|
-
for (size_t i = 0; i < in->n_cols; i++) {
|
|
113
|
-
st->col_names[i] = strdup(in->col_names[i]);
|
|
114
|
-
st->col_types[i] = in->col_types[i];
|
|
115
|
-
}
|
|
116
|
-
st->schema_captured = 1;
|
|
117
|
-
}
|
|
181
|
+
*out = NULL;
|
|
118
182
|
|
|
119
183
|
if (st->tag_col) {
|
|
120
184
|
*out = add_tag_column(in, st->tag_col, st->tag_value_in);
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
185
|
+
return *out ? TF_OK : TF_ERROR;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
|
|
189
|
+
if (!ob) return TF_ERROR;
|
|
190
|
+
if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
191
|
+
tf_batch_free(ob);
|
|
192
|
+
return TF_ERROR;
|
|
193
|
+
}
|
|
194
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
195
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
196
|
+
tf_batch_free(ob);
|
|
197
|
+
return TF_ERROR;
|
|
198
|
+
}
|
|
199
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
200
|
+
tf_batch_free(ob);
|
|
201
|
+
return TF_ERROR;
|
|
131
202
|
}
|
|
132
|
-
*out = ob;
|
|
133
203
|
}
|
|
204
|
+
*out = ob;
|
|
134
205
|
return TF_OK;
|
|
135
206
|
}
|
|
136
207
|
|
|
137
|
-
static int
|
|
208
|
+
static int stack_flush_next(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
138
209
|
stack_state *st = self->state;
|
|
139
|
-
(void)side;
|
|
140
210
|
*out = NULL;
|
|
141
211
|
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
if (fsize <= 0) { fclose(f); return TF_OK; }
|
|
150
|
-
|
|
151
|
-
char *data = malloc((size_t)fsize + 1);
|
|
152
|
-
if (!data) { fclose(f); return TF_ERROR; }
|
|
153
|
-
size_t nread = fread(data, 1, (size_t)fsize, f);
|
|
154
|
-
fclose(f);
|
|
155
|
-
data[nread] = '\0';
|
|
156
|
-
|
|
157
|
-
/* Parse CSV: use a sub-pipeline to decode the file */
|
|
158
|
-
/* Simple approach: parse line by line */
|
|
159
|
-
char **file_col_names = NULL;
|
|
160
|
-
size_t file_n_cols = 0;
|
|
161
|
-
|
|
162
|
-
/* Count rows and parse */
|
|
163
|
-
size_t n_rows = 0;
|
|
164
|
-
char *line = data;
|
|
165
|
-
char *end = data + nread;
|
|
166
|
-
|
|
167
|
-
/* First pass: count data lines */
|
|
168
|
-
char *p = data;
|
|
169
|
-
while (p < end) {
|
|
170
|
-
char *nl = memchr(p, '\n', (size_t)(end - p));
|
|
171
|
-
if (!nl) nl = end;
|
|
172
|
-
if (nl > p || p < end) n_rows++;
|
|
173
|
-
p = nl + 1;
|
|
174
|
-
}
|
|
175
|
-
if (n_rows == 0) { free(data); return TF_OK; }
|
|
176
|
-
n_rows--; /* subtract header */
|
|
177
|
-
|
|
178
|
-
/* Parse header */
|
|
179
|
-
line = data;
|
|
180
|
-
char *nl = memchr(line, '\n', (size_t)(end - line));
|
|
181
|
-
if (!nl) nl = end;
|
|
182
|
-
size_t hdr_len = (size_t)(nl - line);
|
|
183
|
-
if (hdr_len > 0 && line[hdr_len - 1] == '\r') hdr_len--;
|
|
184
|
-
|
|
185
|
-
/* Count commas to get column count */
|
|
186
|
-
file_n_cols = 1;
|
|
187
|
-
for (size_t i = 0; i < hdr_len; i++) {
|
|
188
|
-
if (line[i] == ',') file_n_cols++;
|
|
189
|
-
}
|
|
190
|
-
|
|
191
|
-
/* Parse column names */
|
|
192
|
-
file_col_names = malloc(file_n_cols * sizeof(char *));
|
|
193
|
-
size_t ci = 0;
|
|
194
|
-
size_t field_start = 0;
|
|
195
|
-
for (size_t i = 0; i <= hdr_len; i++) {
|
|
196
|
-
if (i == hdr_len || line[i] == ',') {
|
|
197
|
-
size_t flen = i - field_start;
|
|
198
|
-
/* Trim whitespace */
|
|
199
|
-
while (flen > 0 && (line[field_start] == ' ' || line[field_start] == '\t'))
|
|
200
|
-
{ field_start++; flen--; }
|
|
201
|
-
while (flen > 0 && (line[field_start + flen - 1] == ' ' || line[field_start + flen - 1] == '\t'))
|
|
202
|
-
flen--;
|
|
203
|
-
file_col_names[ci] = malloc(flen + 1);
|
|
204
|
-
memcpy(file_col_names[ci], line + field_start, flen);
|
|
205
|
-
file_col_names[ci][flen] = '\0';
|
|
206
|
-
ci++;
|
|
207
|
-
field_start = i + 1;
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
|
-
|
|
211
|
-
/* Create output batch — use schema from input or file */
|
|
212
|
-
size_t out_cols = st->tag_col ? file_n_cols + 1 : file_n_cols;
|
|
213
|
-
tf_batch *ob = tf_batch_create(out_cols, n_rows > 0 ? n_rows : 1);
|
|
214
|
-
if (!ob) {
|
|
215
|
-
for (size_t i = 0; i < file_n_cols; i++) free(file_col_names[i]);
|
|
216
|
-
free(file_col_names);
|
|
217
|
-
free(data);
|
|
218
|
-
return TF_ERROR;
|
|
219
|
-
}
|
|
220
|
-
|
|
221
|
-
size_t col_offset = 0;
|
|
222
|
-
if (st->tag_col) {
|
|
223
|
-
tf_batch_set_schema(ob, 0, st->tag_col, TF_TYPE_STRING);
|
|
224
|
-
col_offset = 1;
|
|
225
|
-
}
|
|
226
|
-
for (size_t i = 0; i < file_n_cols; i++) {
|
|
227
|
-
tf_batch_set_schema(ob, i + col_offset, file_col_names[i], TF_TYPE_STRING);
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
/* Parse data rows */
|
|
231
|
-
p = nl + 1; /* skip past header newline */
|
|
232
|
-
size_t row = 0;
|
|
233
|
-
while (p < end && row < n_rows) {
|
|
234
|
-
nl = memchr(p, '\n', (size_t)(end - p));
|
|
235
|
-
if (!nl) nl = end;
|
|
236
|
-
size_t line_len = (size_t)(nl - p);
|
|
237
|
-
if (line_len > 0 && p[line_len - 1] == '\r') line_len--;
|
|
238
|
-
if (line_len == 0) { p = nl + 1; continue; }
|
|
239
|
-
|
|
240
|
-
tf_batch_ensure_capacity(ob, row + 1);
|
|
241
|
-
|
|
242
|
-
if (st->tag_col) {
|
|
243
|
-
tf_batch_set_string(ob, row, 0, st->tag_value);
|
|
212
|
+
for (;;) {
|
|
213
|
+
tf_batch *batch = NULL;
|
|
214
|
+
if (stack_next_decoded_file_batch(st, &batch, side) != TF_OK) return TF_ERROR;
|
|
215
|
+
if (!batch) return TF_OK;
|
|
216
|
+
if (batch->n_rows == 0) {
|
|
217
|
+
tf_batch_free(batch);
|
|
218
|
+
continue;
|
|
244
219
|
}
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
size_t fs = 0;
|
|
249
|
-
for (size_t i = 0; i <= line_len; i++) {
|
|
250
|
-
if (i == line_len || p[i] == ',') {
|
|
251
|
-
if (fc < file_n_cols) {
|
|
252
|
-
size_t flen = i - fs;
|
|
253
|
-
/* Trim */
|
|
254
|
-
const char *fp = p + fs;
|
|
255
|
-
while (flen > 0 && (*fp == ' ' || *fp == '\t')) { fp++; flen--; }
|
|
256
|
-
while (flen > 0 && (fp[flen - 1] == ' ' || fp[flen - 1] == '\t')) flen--;
|
|
257
|
-
if (flen == 0) {
|
|
258
|
-
tf_batch_set_null(ob, row, fc + col_offset);
|
|
259
|
-
} else {
|
|
260
|
-
char tmp[4096];
|
|
261
|
-
size_t clen = flen < sizeof(tmp) - 1 ? flen : sizeof(tmp) - 1;
|
|
262
|
-
memcpy(tmp, fp, clen);
|
|
263
|
-
tmp[clen] = '\0';
|
|
264
|
-
tf_batch_set_string(ob, row, fc + col_offset, tmp);
|
|
265
|
-
}
|
|
266
|
-
fc++;
|
|
267
|
-
}
|
|
268
|
-
fs = i + 1;
|
|
269
|
-
}
|
|
220
|
+
if (!st->tag_col) {
|
|
221
|
+
*out = batch;
|
|
222
|
+
return TF_OK;
|
|
270
223
|
}
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
}
|
|
275
|
-
|
|
276
|
-
ob->n_rows = row + 1;
|
|
277
|
-
row++;
|
|
278
|
-
p = nl + 1;
|
|
224
|
+
*out = add_tag_column(batch, st->tag_col, st->tag_value);
|
|
225
|
+
tf_batch_free(batch);
|
|
226
|
+
return *out ? TF_OK : TF_ERROR;
|
|
279
227
|
}
|
|
228
|
+
}
|
|
280
229
|
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
free(data);
|
|
284
|
-
|
|
285
|
-
*out = ob;
|
|
286
|
-
return TF_OK;
|
|
230
|
+
static int stack_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
231
|
+
return stack_flush_next(self, out, side);
|
|
287
232
|
}
|
|
288
233
|
|
|
289
234
|
tf_step *tf_stack_create(const cJSON *args) {
|
|
@@ -296,6 +241,12 @@ tf_step *tf_stack_create(const cJSON *args) {
|
|
|
296
241
|
if (!st) return NULL;
|
|
297
242
|
|
|
298
243
|
st->file_path = strdup(file->valuestring);
|
|
244
|
+
if (!st->file_path) { stack_state_free(st); return NULL; }
|
|
245
|
+
const char *validated = tf_policy_validated_path_arg(args, "file");
|
|
246
|
+
if (validated) {
|
|
247
|
+
st->validated_file_path = strdup(validated);
|
|
248
|
+
if (!st->validated_file_path) { stack_state_free(st); return NULL; }
|
|
249
|
+
}
|
|
299
250
|
|
|
300
251
|
cJSON *tag = cJSON_GetObjectItemCaseSensitive(args, "tag");
|
|
301
252
|
if (cJSON_IsString(tag) && tag->valuestring[0]) {
|
|
@@ -303,12 +254,17 @@ tf_step *tf_stack_create(const cJSON *args) {
|
|
|
303
254
|
cJSON *tv = cJSON_GetObjectItemCaseSensitive(args, "tag_value");
|
|
304
255
|
st->tag_value = strdup(cJSON_IsString(tv) ? tv->valuestring : st->file_path);
|
|
305
256
|
st->tag_value_in = strdup("input");
|
|
257
|
+
if (!st->tag_col || !st->tag_value || !st->tag_value_in) {
|
|
258
|
+
stack_state_free(st);
|
|
259
|
+
return NULL;
|
|
260
|
+
}
|
|
306
261
|
}
|
|
307
262
|
|
|
308
|
-
tf_step *step =
|
|
309
|
-
if (!step) {
|
|
263
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
264
|
+
if (!step) { stack_state_free(st); return NULL; }
|
|
310
265
|
step->process = stack_process;
|
|
311
266
|
step->flush = stack_flush;
|
|
267
|
+
step->flush_next = stack_flush_next;
|
|
312
268
|
step->destroy = stack_destroy;
|
|
313
269
|
step->state = st;
|
|
314
270
|
return step;
|