tranfi 0.0.2 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +395 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-pDFMluyz.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +69 -0
- package/csrc/arena.c +91 -0
- package/csrc/batch.c +229 -0
- package/csrc/buffer.c +78 -0
- package/csrc/cJSON.c +3143 -0
- package/csrc/cJSON.h +300 -0
- package/csrc/codec_csv.c +1058 -0
- package/csrc/codec_jsonl.c +374 -0
- package/csrc/codec_table.c +218 -0
- package/csrc/codec_text.c +229 -0
- package/csrc/compiler.c +102 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +1180 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1245 -0
- package/csrc/expr.h +56 -0
- package/csrc/internal.h +250 -0
- package/csrc/ir.c +119 -0
- package/csrc/ir.h +167 -0
- package/csrc/ir_schema.c +60 -0
- package/csrc/ir_serialize.c +104 -0
- package/csrc/ir_sql.c +1211 -0
- package/csrc/ir_validate.c +120 -0
- package/csrc/main.c +392 -0
- package/csrc/op_acf.c +133 -0
- package/csrc/op_anomaly.c +120 -0
- package/csrc/op_bin.c +109 -0
- package/csrc/op_cast.c +195 -0
- package/csrc/op_clip.c +88 -0
- package/csrc/op_date_trunc.c +181 -0
- package/csrc/op_datetime.c +212 -0
- package/csrc/op_derive.c +248 -0
- package/csrc/op_diff.c +134 -0
- package/csrc/op_ewma.c +103 -0
- package/csrc/op_explode.c +108 -0
- package/csrc/op_fill_down.c +163 -0
- package/csrc/op_fill_null.c +123 -0
- package/csrc/op_filter.c +132 -0
- package/csrc/op_frequency.c +193 -0
- package/csrc/op_grep.c +163 -0
- package/csrc/op_group_agg.c +285 -0
- package/csrc/op_hash.c +126 -0
- package/csrc/op_head.c +149 -0
- package/csrc/op_interpolate.c +239 -0
- package/csrc/op_join.c +384 -0
- package/csrc/op_label_encode.c +144 -0
- package/csrc/op_lead.c +190 -0
- package/csrc/op_normalize.c +226 -0
- package/csrc/op_onehot.c +185 -0
- package/csrc/op_pivot.c +370 -0
- package/csrc/op_registry.c +1148 -0
- package/csrc/op_rename.c +138 -0
- package/csrc/op_replace.c +202 -0
- package/csrc/op_sample.c +101 -0
- package/csrc/op_select.c +140 -0
- package/csrc/op_skip.c +152 -0
- package/csrc/op_sort.c +273 -0
- package/csrc/op_split.c +114 -0
- package/csrc/op_split_data.c +87 -0
- package/csrc/op_stack.c +315 -0
- package/csrc/op_stats.c +779 -0
- package/csrc/op_step.c +171 -0
- package/csrc/op_tail.c +96 -0
- package/csrc/op_top.c +150 -0
- package/csrc/op_trim.c +109 -0
- package/csrc/op_unique.c +300 -0
- package/csrc/op_unpivot.c +159 -0
- package/csrc/op_validate.c +71 -0
- package/csrc/op_window.c +150 -0
- package/csrc/pipeline.c +315 -0
- package/csrc/plan.c +206 -0
- package/csrc/recipes.c +102 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +463 -0
- package/csrc/report.h +22 -0
- package/csrc/tranfi.h +123 -0
- package/csrc/wasm_api.c +157 -0
- package/napi_api.c +326 -0
- package/package.json +46 -57
- package/src/cli.js +193 -0
- package/src/engines/duckdb.js +109 -0
- package/src/index.js +306 -0
- package/src/native.js +22 -0
- package/src/pipeline.js +286 -0
- package/src/server.js +277 -0
- package/src/wasm.js +19 -0
- package/wasm/index.js +244 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/LICENSE +0 -21
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/csrc/pipeline.c
ADDED
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* pipeline.c — Pipeline orchestrator.
|
|
3
|
+
*
|
|
4
|
+
* Creates a pipeline from a JSON plan, streams bytes through
|
|
5
|
+
* decode → steps → encode, and routes output to channels.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "internal.h"
|
|
9
|
+
#include "dsl.h"
|
|
10
|
+
#include <stdlib.h>
|
|
11
|
+
#include <string.h>
|
|
12
|
+
#include <stdio.h>
|
|
13
|
+
|
|
14
|
+
#define TRANFI_VERSION "0.1.0"
|
|
15
|
+
|
|
16
|
+
static char *g_last_error = NULL;
|
|
17
|
+
|
|
18
|
+
void tf_set_last_error(const char *msg) {
|
|
19
|
+
free(g_last_error);
|
|
20
|
+
g_last_error = msg ? strdup(msg) : NULL;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
const char *tf_last_error(void) {
|
|
24
|
+
return g_last_error;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const char *tf_version(void) {
|
|
28
|
+
return TRANFI_VERSION;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
static tf_pipeline *assemble_pipeline(tf_decoder *decoder, tf_step **steps,
|
|
32
|
+
size_t n_steps, tf_encoder *encoder) {
|
|
33
|
+
tf_pipeline *p = calloc(1, sizeof(tf_pipeline));
|
|
34
|
+
if (!p) {
|
|
35
|
+
decoder->destroy(decoder);
|
|
36
|
+
encoder->destroy(encoder);
|
|
37
|
+
for (size_t i = 0; i < n_steps; i++) steps[i]->destroy(steps[i]);
|
|
38
|
+
free(steps);
|
|
39
|
+
tf_set_last_error("out of memory");
|
|
40
|
+
return NULL;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
p->decoder = decoder;
|
|
44
|
+
p->steps = steps;
|
|
45
|
+
p->n_steps = n_steps;
|
|
46
|
+
p->encoder = encoder;
|
|
47
|
+
|
|
48
|
+
for (int i = 0; i < TF_NUM_CHANNELS; i++) {
|
|
49
|
+
tf_buffer_init(&p->output[i]);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/* Wire up side channels */
|
|
53
|
+
p->side.errors = &p->output[TF_CHAN_ERRORS];
|
|
54
|
+
p->side.stats = &p->output[TF_CHAN_STATS];
|
|
55
|
+
p->side.samples = &p->output[TF_CHAN_SAMPLES];
|
|
56
|
+
|
|
57
|
+
return p;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
tf_pipeline *tf_pipeline_create(const char *plan_json, size_t len) {
|
|
61
|
+
if (!plan_json || len == 0) {
|
|
62
|
+
tf_set_last_error("empty plan");
|
|
63
|
+
return NULL;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
char *error = NULL;
|
|
67
|
+
|
|
68
|
+
/* 1. Parse JSON → IR */
|
|
69
|
+
tf_ir_plan *ir = tf_ir_from_json(plan_json, len, &error);
|
|
70
|
+
if (!ir) {
|
|
71
|
+
tf_set_last_error(error ? error : "failed to parse plan");
|
|
72
|
+
free(error);
|
|
73
|
+
return NULL;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/* 2. Validate */
|
|
77
|
+
if (tf_ir_validate(ir) != TF_OK) {
|
|
78
|
+
tf_set_last_error(ir->error ? ir->error : "validation failed");
|
|
79
|
+
tf_ir_plan_free(ir);
|
|
80
|
+
return NULL;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/* 3. Schema inference (best-effort, non-fatal) */
|
|
84
|
+
tf_ir_infer_schema(ir);
|
|
85
|
+
|
|
86
|
+
/* 4. Compile to native target */
|
|
87
|
+
tf_decoder *decoder = NULL;
|
|
88
|
+
tf_step **steps = NULL;
|
|
89
|
+
size_t n_steps = 0;
|
|
90
|
+
tf_encoder *encoder = NULL;
|
|
91
|
+
if (tf_compile_native(ir, &decoder, &steps, &n_steps, &encoder, &error) != TF_OK) {
|
|
92
|
+
tf_set_last_error(error ? error : "compilation failed");
|
|
93
|
+
free(error);
|
|
94
|
+
tf_ir_plan_free(ir);
|
|
95
|
+
return NULL;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
tf_ir_plan_free(ir);
|
|
99
|
+
|
|
100
|
+
/* 5. Assemble pipeline */
|
|
101
|
+
return assemble_pipeline(decoder, steps, n_steps, encoder);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
tf_pipeline *tf_pipeline_create_from_ir(const tf_ir_plan *plan) {
|
|
105
|
+
if (!plan) {
|
|
106
|
+
tf_set_last_error("NULL IR plan");
|
|
107
|
+
return NULL;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
char *error = NULL;
|
|
111
|
+
tf_decoder *decoder = NULL;
|
|
112
|
+
tf_step **steps = NULL;
|
|
113
|
+
size_t n_steps = 0;
|
|
114
|
+
tf_encoder *encoder = NULL;
|
|
115
|
+
|
|
116
|
+
if (tf_compile_native(plan, &decoder, &steps, &n_steps, &encoder, &error) != TF_OK) {
|
|
117
|
+
tf_set_last_error(error ? error : "compilation failed");
|
|
118
|
+
free(error);
|
|
119
|
+
return NULL;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
return assemble_pipeline(decoder, steps, n_steps, encoder);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/* Public IR wrappers (thin forwarding to ir.h functions) */
|
|
126
|
+
tf_ir_plan *tf_ir_plan_from_json(const char *json, size_t len, char **error) {
|
|
127
|
+
return tf_ir_from_json(json, len, error);
|
|
128
|
+
}
|
|
129
|
+
char *tf_ir_plan_to_json(const tf_ir_plan *plan) {
|
|
130
|
+
return tf_ir_to_json(plan);
|
|
131
|
+
}
|
|
132
|
+
int tf_ir_plan_validate(tf_ir_plan *plan) {
|
|
133
|
+
return tf_ir_validate(plan);
|
|
134
|
+
}
|
|
135
|
+
int tf_ir_plan_infer_schema(tf_ir_plan *plan) {
|
|
136
|
+
return tf_ir_infer_schema(plan);
|
|
137
|
+
}
|
|
138
|
+
void tf_ir_plan_destroy(tf_ir_plan *plan) {
|
|
139
|
+
tf_ir_plan_free(plan);
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/*
|
|
143
|
+
* Process a batch through all steps, then encode.
|
|
144
|
+
*/
|
|
145
|
+
static int process_batch(tf_pipeline *p, tf_batch *batch) {
|
|
146
|
+
tf_batch *current = batch;
|
|
147
|
+
int batch_owned = 0; /* 0 = still owned by caller (decoder) */
|
|
148
|
+
|
|
149
|
+
p->rows_in += current->n_rows;
|
|
150
|
+
|
|
151
|
+
for (size_t i = 0; i < p->n_steps; i++) {
|
|
152
|
+
tf_batch *next = NULL;
|
|
153
|
+
int rc = p->steps[i]->process(p->steps[i], current, &next, &p->side);
|
|
154
|
+
|
|
155
|
+
if (batch_owned) tf_batch_free(current);
|
|
156
|
+
batch_owned = 1;
|
|
157
|
+
|
|
158
|
+
if (rc != TF_OK) return TF_ERROR;
|
|
159
|
+
if (!next) return TF_OK; /* filtered away entirely */
|
|
160
|
+
current = next;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/* Encode */
|
|
164
|
+
if (current->n_rows > 0) {
|
|
165
|
+
p->rows_out += current->n_rows;
|
|
166
|
+
int rc = p->encoder->encode(p->encoder, current, &p->output[TF_CHAN_MAIN]);
|
|
167
|
+
if (batch_owned) tf_batch_free(current);
|
|
168
|
+
return rc;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
if (batch_owned) tf_batch_free(current);
|
|
172
|
+
return TF_OK;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
int tf_pipeline_push(tf_pipeline *p, const uint8_t *data, size_t len) {
|
|
176
|
+
if (!p || p->finished) return TF_ERROR;
|
|
177
|
+
|
|
178
|
+
p->bytes_in += len;
|
|
179
|
+
|
|
180
|
+
/* Decode bytes into batches */
|
|
181
|
+
tf_batch **batches = NULL;
|
|
182
|
+
size_t n_batches = 0;
|
|
183
|
+
int rc = p->decoder->decode(p->decoder, data, len, &batches, &n_batches);
|
|
184
|
+
if (rc != TF_OK) {
|
|
185
|
+
free(p->error);
|
|
186
|
+
p->error = strdup("decode error");
|
|
187
|
+
return TF_ERROR;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/* Process each batch */
|
|
191
|
+
for (size_t i = 0; i < n_batches; i++) {
|
|
192
|
+
rc = process_batch(p, batches[i]);
|
|
193
|
+
tf_batch_free(batches[i]);
|
|
194
|
+
if (rc != TF_OK) {
|
|
195
|
+
for (size_t j = i + 1; j < n_batches; j++) tf_batch_free(batches[j]);
|
|
196
|
+
free(batches);
|
|
197
|
+
free(p->error);
|
|
198
|
+
p->error = strdup("processing error");
|
|
199
|
+
return TF_ERROR;
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
free(batches);
|
|
203
|
+
|
|
204
|
+
return TF_OK;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
int tf_pipeline_finish(tf_pipeline *p) {
|
|
208
|
+
if (!p || p->finished) return TF_ERROR;
|
|
209
|
+
p->finished = 1;
|
|
210
|
+
|
|
211
|
+
/* Flush decoder */
|
|
212
|
+
tf_batch **batches = NULL;
|
|
213
|
+
size_t n_batches = 0;
|
|
214
|
+
int rc = p->decoder->flush(p->decoder, &batches, &n_batches);
|
|
215
|
+
if (rc == TF_OK) {
|
|
216
|
+
for (size_t i = 0; i < n_batches; i++) {
|
|
217
|
+
process_batch(p, batches[i]);
|
|
218
|
+
tf_batch_free(batches[i]);
|
|
219
|
+
}
|
|
220
|
+
free(batches);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/* Flush steps */
|
|
224
|
+
for (size_t i = 0; i < p->n_steps; i++) {
|
|
225
|
+
tf_batch *flushed = NULL;
|
|
226
|
+
rc = p->steps[i]->flush(p->steps[i], &flushed, &p->side);
|
|
227
|
+
if (rc == TF_OK && flushed) {
|
|
228
|
+
/* Run flushed batch through remaining steps */
|
|
229
|
+
tf_batch *current = flushed;
|
|
230
|
+
int owned = 1;
|
|
231
|
+
for (size_t j = i + 1; j < p->n_steps; j++) {
|
|
232
|
+
tf_batch *next = NULL;
|
|
233
|
+
p->steps[j]->process(p->steps[j], current, &next, &p->side);
|
|
234
|
+
if (owned) tf_batch_free(current);
|
|
235
|
+
owned = 1;
|
|
236
|
+
if (!next) { current = NULL; break; }
|
|
237
|
+
current = next;
|
|
238
|
+
}
|
|
239
|
+
if (current && current->n_rows > 0) {
|
|
240
|
+
p->rows_out += current->n_rows;
|
|
241
|
+
p->encoder->encode(p->encoder, current, &p->output[TF_CHAN_MAIN]);
|
|
242
|
+
}
|
|
243
|
+
if (current && owned) tf_batch_free(current);
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/* Flush encoder */
|
|
248
|
+
p->encoder->flush(p->encoder, &p->output[TF_CHAN_MAIN]);
|
|
249
|
+
|
|
250
|
+
/* Emit final stats */
|
|
251
|
+
p->bytes_out = tf_buffer_readable(&p->output[TF_CHAN_MAIN]);
|
|
252
|
+
char stats_buf[256];
|
|
253
|
+
snprintf(stats_buf, sizeof(stats_buf),
|
|
254
|
+
"{\"rows_in\":%zu,\"rows_out\":%zu,\"bytes_in\":%zu,\"bytes_out\":%zu}\n",
|
|
255
|
+
p->rows_in, p->rows_out, p->bytes_in, p->bytes_out);
|
|
256
|
+
tf_buffer_write_str(&p->output[TF_CHAN_STATS], stats_buf);
|
|
257
|
+
|
|
258
|
+
return TF_OK;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
size_t tf_pipeline_pull(tf_pipeline *p, int channel, uint8_t *buf, size_t buf_len) {
|
|
262
|
+
if (!p || channel < 0 || channel >= TF_NUM_CHANNELS) return 0;
|
|
263
|
+
return tf_buffer_read(&p->output[channel], buf, buf_len);
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const char *tf_pipeline_error(tf_pipeline *p) {
|
|
267
|
+
return p ? p->error : NULL;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
void tf_pipeline_free(tf_pipeline *p) {
|
|
271
|
+
if (!p) return;
|
|
272
|
+
if (p->decoder) p->decoder->destroy(p->decoder);
|
|
273
|
+
if (p->encoder) p->encoder->destroy(p->encoder);
|
|
274
|
+
for (size_t i = 0; i < p->n_steps; i++) {
|
|
275
|
+
if (p->steps[i]) p->steps[i]->destroy(p->steps[i]);
|
|
276
|
+
}
|
|
277
|
+
free(p->steps);
|
|
278
|
+
for (int i = 0; i < TF_NUM_CHANNELS; i++) {
|
|
279
|
+
tf_buffer_free(&p->output[i]);
|
|
280
|
+
}
|
|
281
|
+
free(p->error);
|
|
282
|
+
free(p);
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
char *tf_compile_to_sql(const char *dsl, size_t len, char **error) {
|
|
286
|
+
if (error) *error = NULL;
|
|
287
|
+
tf_ir_plan *plan = tf_dsl_parse(dsl, len, error);
|
|
288
|
+
if (!plan) return NULL;
|
|
289
|
+
if (tf_ir_validate(plan) != TF_OK) {
|
|
290
|
+
if (error) { free(*error); *error = strdup(plan->error ? plan->error : "validation failed"); }
|
|
291
|
+
tf_ir_plan_destroy(plan);
|
|
292
|
+
return NULL;
|
|
293
|
+
}
|
|
294
|
+
tf_ir_infer_schema(plan);
|
|
295
|
+
char *sql = tf_ir_to_sql(plan, error);
|
|
296
|
+
tf_ir_plan_destroy(plan);
|
|
297
|
+
return sql;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
char *tf_ir_plan_to_sql(const tf_ir_plan *plan, char **error) {
|
|
301
|
+
return tf_ir_to_sql(plan, error);
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
char *tf_compile_dsl(const char *dsl, size_t len, char **error) {
|
|
305
|
+
if (error) *error = NULL;
|
|
306
|
+
tf_ir_plan *plan = tf_dsl_parse(dsl, len, error);
|
|
307
|
+
if (!plan) return NULL;
|
|
308
|
+
char *json = tf_ir_plan_to_json(plan);
|
|
309
|
+
tf_ir_plan_destroy(plan);
|
|
310
|
+
return json;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
void tf_string_free(char *s) {
|
|
314
|
+
free(s);
|
|
315
|
+
}
|
package/csrc/plan.c
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* plan.c — Parse a JSON pipeline plan and instantiate decoder, steps, encoder.
|
|
3
|
+
*
|
|
4
|
+
* Plan format:
|
|
5
|
+
* {
|
|
6
|
+
* "steps": [
|
|
7
|
+
* {"op": "codec.csv.decode", "args": {"delimiter": ","}},
|
|
8
|
+
* {"op": "filter", "args": {"expr": "col('age') > 25"}},
|
|
9
|
+
* {"op": "codec.csv.encode", "args": {}}
|
|
10
|
+
* ]
|
|
11
|
+
* }
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
#include "internal.h"
|
|
15
|
+
#include "cJSON.h"
|
|
16
|
+
#include <stdlib.h>
|
|
17
|
+
#include <string.h>
|
|
18
|
+
#include <stdio.h>
|
|
19
|
+
|
|
20
|
+
static void set_error(char **error, const char *msg) {
|
|
21
|
+
if (error) {
|
|
22
|
+
free(*error);
|
|
23
|
+
size_t len = strlen(msg) + 1;
|
|
24
|
+
*error = malloc(len);
|
|
25
|
+
if (*error) memcpy(*error, msg, len);
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
static void set_errorf(char **error, const char *fmt, const char *detail) {
|
|
30
|
+
if (error) {
|
|
31
|
+
char buf[256];
|
|
32
|
+
snprintf(buf, sizeof(buf), fmt, detail);
|
|
33
|
+
set_error(error, buf);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
int tf_plan_parse(const char *json, size_t len,
|
|
38
|
+
tf_decoder **out_decoder, tf_step ***out_steps, size_t *out_n_steps,
|
|
39
|
+
tf_encoder **out_encoder, char **error) {
|
|
40
|
+
*out_decoder = NULL;
|
|
41
|
+
*out_steps = NULL;
|
|
42
|
+
*out_n_steps = 0;
|
|
43
|
+
*out_encoder = NULL;
|
|
44
|
+
|
|
45
|
+
/* Parse JSON */
|
|
46
|
+
cJSON *root = cJSON_ParseWithLength(json, len);
|
|
47
|
+
if (!root) {
|
|
48
|
+
set_error(error, "invalid JSON in plan");
|
|
49
|
+
return TF_ERROR;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
cJSON *steps_arr = cJSON_GetObjectItemCaseSensitive(root, "steps");
|
|
53
|
+
if (!cJSON_IsArray(steps_arr)) {
|
|
54
|
+
set_error(error, "plan must have a 'steps' array");
|
|
55
|
+
cJSON_Delete(root);
|
|
56
|
+
return TF_ERROR;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
int n_json_steps = cJSON_GetArraySize(steps_arr);
|
|
60
|
+
if (n_json_steps == 0) {
|
|
61
|
+
set_error(error, "plan has no steps");
|
|
62
|
+
cJSON_Delete(root);
|
|
63
|
+
return TF_ERROR;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/* Temporary arrays for transforms (max = n_json_steps) */
|
|
67
|
+
tf_step **steps = calloc(n_json_steps, sizeof(tf_step *));
|
|
68
|
+
size_t n_steps = 0;
|
|
69
|
+
tf_decoder *decoder = NULL;
|
|
70
|
+
tf_encoder *encoder = NULL;
|
|
71
|
+
int err = TF_OK;
|
|
72
|
+
|
|
73
|
+
for (int i = 0; i < n_json_steps; i++) {
|
|
74
|
+
cJSON *step_json = cJSON_GetArrayItem(steps_arr, i);
|
|
75
|
+
cJSON *op_json = cJSON_GetObjectItemCaseSensitive(step_json, "op");
|
|
76
|
+
cJSON *args_json = cJSON_GetObjectItemCaseSensitive(step_json, "args");
|
|
77
|
+
|
|
78
|
+
if (!cJSON_IsString(op_json)) {
|
|
79
|
+
set_errorf(error, "step %d missing 'op' string", "");
|
|
80
|
+
err = TF_ERROR;
|
|
81
|
+
break;
|
|
82
|
+
}
|
|
83
|
+
const char *op = op_json->valuestring;
|
|
84
|
+
|
|
85
|
+
/* Match op to constructor */
|
|
86
|
+
if (strcmp(op, "codec.csv.decode") == 0) {
|
|
87
|
+
if (decoder) {
|
|
88
|
+
set_error(error, "multiple decoders not supported");
|
|
89
|
+
err = TF_ERROR;
|
|
90
|
+
break;
|
|
91
|
+
}
|
|
92
|
+
decoder = tf_csv_decoder_create(args_json);
|
|
93
|
+
if (!decoder) {
|
|
94
|
+
set_error(error, "failed to create CSV decoder");
|
|
95
|
+
err = TF_ERROR;
|
|
96
|
+
break;
|
|
97
|
+
}
|
|
98
|
+
} else if (strcmp(op, "codec.csv.encode") == 0) {
|
|
99
|
+
if (encoder) {
|
|
100
|
+
set_error(error, "multiple encoders not supported");
|
|
101
|
+
err = TF_ERROR;
|
|
102
|
+
break;
|
|
103
|
+
}
|
|
104
|
+
encoder = tf_csv_encoder_create(args_json);
|
|
105
|
+
if (!encoder) {
|
|
106
|
+
set_error(error, "failed to create CSV encoder");
|
|
107
|
+
err = TF_ERROR;
|
|
108
|
+
break;
|
|
109
|
+
}
|
|
110
|
+
} else if (strcmp(op, "codec.jsonl.decode") == 0) {
|
|
111
|
+
if (decoder) {
|
|
112
|
+
set_error(error, "multiple decoders not supported");
|
|
113
|
+
err = TF_ERROR;
|
|
114
|
+
break;
|
|
115
|
+
}
|
|
116
|
+
decoder = tf_jsonl_decoder_create(args_json);
|
|
117
|
+
if (!decoder) {
|
|
118
|
+
set_error(error, "failed to create JSONL decoder");
|
|
119
|
+
err = TF_ERROR;
|
|
120
|
+
break;
|
|
121
|
+
}
|
|
122
|
+
} else if (strcmp(op, "codec.jsonl.encode") == 0) {
|
|
123
|
+
if (encoder) {
|
|
124
|
+
set_error(error, "multiple encoders not supported");
|
|
125
|
+
err = TF_ERROR;
|
|
126
|
+
break;
|
|
127
|
+
}
|
|
128
|
+
encoder = tf_jsonl_encoder_create(args_json);
|
|
129
|
+
if (!encoder) {
|
|
130
|
+
set_error(error, "failed to create JSONL encoder");
|
|
131
|
+
err = TF_ERROR;
|
|
132
|
+
break;
|
|
133
|
+
}
|
|
134
|
+
} else if (strcmp(op, "filter") == 0) {
|
|
135
|
+
tf_step *s = tf_filter_create(args_json);
|
|
136
|
+
if (!s) {
|
|
137
|
+
set_error(error, "failed to create filter step");
|
|
138
|
+
err = TF_ERROR;
|
|
139
|
+
break;
|
|
140
|
+
}
|
|
141
|
+
steps[n_steps++] = s;
|
|
142
|
+
} else if (strcmp(op, "select") == 0) {
|
|
143
|
+
tf_step *s = tf_select_create(args_json);
|
|
144
|
+
if (!s) {
|
|
145
|
+
set_error(error, "failed to create select step");
|
|
146
|
+
err = TF_ERROR;
|
|
147
|
+
break;
|
|
148
|
+
}
|
|
149
|
+
steps[n_steps++] = s;
|
|
150
|
+
} else if (strcmp(op, "rename") == 0) {
|
|
151
|
+
tf_step *s = tf_rename_create(args_json);
|
|
152
|
+
if (!s) {
|
|
153
|
+
set_error(error, "failed to create rename step");
|
|
154
|
+
err = TF_ERROR;
|
|
155
|
+
break;
|
|
156
|
+
}
|
|
157
|
+
steps[n_steps++] = s;
|
|
158
|
+
} else if (strcmp(op, "head") == 0) {
|
|
159
|
+
tf_step *s = tf_head_create(args_json);
|
|
160
|
+
if (!s) {
|
|
161
|
+
set_error(error, "failed to create head step");
|
|
162
|
+
err = TF_ERROR;
|
|
163
|
+
break;
|
|
164
|
+
}
|
|
165
|
+
steps[n_steps++] = s;
|
|
166
|
+
} else {
|
|
167
|
+
set_errorf(error, "unknown op: '%s'", op);
|
|
168
|
+
err = TF_ERROR;
|
|
169
|
+
break;
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
cJSON_Delete(root);
|
|
174
|
+
|
|
175
|
+
if (err != TF_OK) {
|
|
176
|
+
/* Cleanup on error */
|
|
177
|
+
if (decoder) decoder->destroy(decoder);
|
|
178
|
+
if (encoder) encoder->destroy(encoder);
|
|
179
|
+
for (size_t i = 0; i < n_steps; i++) {
|
|
180
|
+
if (steps[i]) steps[i]->destroy(steps[i]);
|
|
181
|
+
}
|
|
182
|
+
free(steps);
|
|
183
|
+
return TF_ERROR;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
if (!decoder) {
|
|
187
|
+
set_error(error, "plan has no decoder (need a codec.*.decode step)");
|
|
188
|
+
if (encoder) encoder->destroy(encoder);
|
|
189
|
+
for (size_t i = 0; i < n_steps; i++) steps[i]->destroy(steps[i]);
|
|
190
|
+
free(steps);
|
|
191
|
+
return TF_ERROR;
|
|
192
|
+
}
|
|
193
|
+
if (!encoder) {
|
|
194
|
+
set_error(error, "plan has no encoder (need a codec.*.encode step)");
|
|
195
|
+
decoder->destroy(decoder);
|
|
196
|
+
for (size_t i = 0; i < n_steps; i++) steps[i]->destroy(steps[i]);
|
|
197
|
+
free(steps);
|
|
198
|
+
return TF_ERROR;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
*out_decoder = decoder;
|
|
202
|
+
*out_steps = steps;
|
|
203
|
+
*out_n_steps = n_steps;
|
|
204
|
+
*out_encoder = encoder;
|
|
205
|
+
return TF_OK;
|
|
206
|
+
}
|
package/csrc/recipes.c
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* recipes.c — Built-in named recipes (20 common ETL pipelines).
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
#include "recipes.h"
|
|
6
|
+
#include <string.h>
|
|
7
|
+
#include <ctype.h>
|
|
8
|
+
|
|
9
|
+
typedef struct {
|
|
10
|
+
const char *name;
|
|
11
|
+
const char *dsl;
|
|
12
|
+
const char *description;
|
|
13
|
+
} recipe_entry;
|
|
14
|
+
|
|
15
|
+
static const recipe_entry recipes[] = {
|
|
16
|
+
/* ---- Data Exploration ---- */
|
|
17
|
+
{"profile", "csv | stats | csv",
|
|
18
|
+
"Full data profiling (all statistics per column)"},
|
|
19
|
+
{"preview", "csv | head 10 | csv",
|
|
20
|
+
"Quick preview of first 10 rows"},
|
|
21
|
+
{"schema", "csv | head 0 | csv",
|
|
22
|
+
"Show column names only"},
|
|
23
|
+
{"summary", "csv | stats count,min,max,avg,stddev | csv",
|
|
24
|
+
"Summary statistics"},
|
|
25
|
+
{"count", "csv | stats count | csv",
|
|
26
|
+
"Row count per column"},
|
|
27
|
+
{"cardinality", "csv | stats count,distinct | csv",
|
|
28
|
+
"Unique value counts per column"},
|
|
29
|
+
{"distro", "csv | stats min,p25,median,p75,max | csv",
|
|
30
|
+
"Five-number summary (quartiles)"},
|
|
31
|
+
|
|
32
|
+
/* ---- Data Quality ---- */
|
|
33
|
+
{"freq", "csv | frequency | csv",
|
|
34
|
+
"Value frequency distribution"},
|
|
35
|
+
{"dedup", "csv | dedup | csv",
|
|
36
|
+
"Remove duplicate rows"},
|
|
37
|
+
{"clean", "csv | trim | csv",
|
|
38
|
+
"Trim whitespace from all columns"},
|
|
39
|
+
|
|
40
|
+
/* ---- Data Sampling ---- */
|
|
41
|
+
{"sample", "csv | sample 100 | csv",
|
|
42
|
+
"Random sample of 100 rows"},
|
|
43
|
+
{"head", "csv | head 20 | csv",
|
|
44
|
+
"First 20 rows"},
|
|
45
|
+
{"tail", "csv | tail 20 | csv",
|
|
46
|
+
"Last 20 rows"},
|
|
47
|
+
|
|
48
|
+
/* ---- Format Conversion ---- */
|
|
49
|
+
{"csv2json", "csv | jsonl",
|
|
50
|
+
"Convert CSV to JSONL"},
|
|
51
|
+
{"json2csv", "jsonl | csv",
|
|
52
|
+
"Convert JSONL to CSV"},
|
|
53
|
+
{"tsv2csv", "csv delimiter=\"\t\" | csv",
|
|
54
|
+
"Convert TSV to CSV"},
|
|
55
|
+
{"csv2tsv", "csv | csv delimiter=\"\t\"",
|
|
56
|
+
"Convert CSV to TSV"},
|
|
57
|
+
|
|
58
|
+
/* ---- Display ---- */
|
|
59
|
+
{"look", "csv | table",
|
|
60
|
+
"Pretty-print as Markdown table"},
|
|
61
|
+
|
|
62
|
+
/* ---- Analysis ---- */
|
|
63
|
+
{"histogram", "csv | stats hist | csv",
|
|
64
|
+
"Distribution histograms"},
|
|
65
|
+
{"hash", "csv | hash | csv",
|
|
66
|
+
"Add row hash column for change detection"},
|
|
67
|
+
{"samples", "csv | stats sample | csv",
|
|
68
|
+
"Show sample values per column"},
|
|
69
|
+
};
|
|
70
|
+
|
|
71
|
+
#define RECIPE_COUNT (sizeof(recipes) / sizeof(recipes[0]))
|
|
72
|
+
|
|
73
|
+
size_t tf_recipe_count(void) {
|
|
74
|
+
return RECIPE_COUNT;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const char *tf_recipe_name(size_t index) {
|
|
78
|
+
return index < RECIPE_COUNT ? recipes[index].name : NULL;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const char *tf_recipe_dsl(size_t index) {
|
|
82
|
+
return index < RECIPE_COUNT ? recipes[index].dsl : NULL;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const char *tf_recipe_description(size_t index) {
|
|
86
|
+
return index < RECIPE_COUNT ? recipes[index].description : NULL;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
const char *tf_recipe_find_dsl(const char *name) {
|
|
90
|
+
if (!name) return NULL;
|
|
91
|
+
for (size_t i = 0; i < RECIPE_COUNT; i++) {
|
|
92
|
+
/* Case-insensitive comparison */
|
|
93
|
+
const char *a = name;
|
|
94
|
+
const char *b = recipes[i].name;
|
|
95
|
+
while (*a && *b && tolower((unsigned char)*a) == tolower((unsigned char)*b)) {
|
|
96
|
+
a++;
|
|
97
|
+
b++;
|
|
98
|
+
}
|
|
99
|
+
if (*a == '\0' && *b == '\0') return recipes[i].dsl;
|
|
100
|
+
}
|
|
101
|
+
return NULL;
|
|
102
|
+
}
|
package/csrc/recipes.h
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* recipes.h — Built-in named recipes.
|
|
3
|
+
*
|
|
4
|
+
* 20 pre-built DSL pipelines for common ETL operations:
|
|
5
|
+
* tranfi profile → full data profiling
|
|
6
|
+
* tranfi preview → first 10 rows
|
|
7
|
+
* tranfi csv2json → format conversion
|
|
8
|
+
* ...
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
#ifndef TF_RECIPES_H
|
|
12
|
+
#define TF_RECIPES_H
|
|
13
|
+
|
|
14
|
+
#include <stddef.h>
|
|
15
|
+
|
|
16
|
+
/* Number of built-in recipes. */
|
|
17
|
+
size_t tf_recipe_count(void);
|
|
18
|
+
|
|
19
|
+
/* Accessors by index (0-based). Return NULL if index out of range. */
|
|
20
|
+
const char *tf_recipe_name(size_t index);
|
|
21
|
+
const char *tf_recipe_dsl(size_t index);
|
|
22
|
+
const char *tf_recipe_description(size_t index);
|
|
23
|
+
|
|
24
|
+
/* Lookup by name (case-insensitive). Returns DSL string or NULL. */
|
|
25
|
+
const char *tf_recipe_find_dsl(const char *name);
|
|
26
|
+
|
|
27
|
+
#endif /* TF_RECIPES_H */
|