tranfi 0.0.2 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +395 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-pDFMluyz.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +69 -0
- package/csrc/arena.c +91 -0
- package/csrc/batch.c +229 -0
- package/csrc/buffer.c +78 -0
- package/csrc/cJSON.c +3143 -0
- package/csrc/cJSON.h +300 -0
- package/csrc/codec_csv.c +1058 -0
- package/csrc/codec_jsonl.c +374 -0
- package/csrc/codec_table.c +218 -0
- package/csrc/codec_text.c +229 -0
- package/csrc/compiler.c +102 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +1180 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1245 -0
- package/csrc/expr.h +56 -0
- package/csrc/internal.h +250 -0
- package/csrc/ir.c +119 -0
- package/csrc/ir.h +167 -0
- package/csrc/ir_schema.c +60 -0
- package/csrc/ir_serialize.c +104 -0
- package/csrc/ir_sql.c +1211 -0
- package/csrc/ir_validate.c +120 -0
- package/csrc/main.c +392 -0
- package/csrc/op_acf.c +133 -0
- package/csrc/op_anomaly.c +120 -0
- package/csrc/op_bin.c +109 -0
- package/csrc/op_cast.c +195 -0
- package/csrc/op_clip.c +88 -0
- package/csrc/op_date_trunc.c +181 -0
- package/csrc/op_datetime.c +212 -0
- package/csrc/op_derive.c +248 -0
- package/csrc/op_diff.c +134 -0
- package/csrc/op_ewma.c +103 -0
- package/csrc/op_explode.c +108 -0
- package/csrc/op_fill_down.c +163 -0
- package/csrc/op_fill_null.c +123 -0
- package/csrc/op_filter.c +132 -0
- package/csrc/op_frequency.c +193 -0
- package/csrc/op_grep.c +163 -0
- package/csrc/op_group_agg.c +285 -0
- package/csrc/op_hash.c +126 -0
- package/csrc/op_head.c +149 -0
- package/csrc/op_interpolate.c +239 -0
- package/csrc/op_join.c +384 -0
- package/csrc/op_label_encode.c +144 -0
- package/csrc/op_lead.c +190 -0
- package/csrc/op_normalize.c +226 -0
- package/csrc/op_onehot.c +185 -0
- package/csrc/op_pivot.c +370 -0
- package/csrc/op_registry.c +1148 -0
- package/csrc/op_rename.c +138 -0
- package/csrc/op_replace.c +202 -0
- package/csrc/op_sample.c +101 -0
- package/csrc/op_select.c +140 -0
- package/csrc/op_skip.c +152 -0
- package/csrc/op_sort.c +273 -0
- package/csrc/op_split.c +114 -0
- package/csrc/op_split_data.c +87 -0
- package/csrc/op_stack.c +315 -0
- package/csrc/op_stats.c +779 -0
- package/csrc/op_step.c +171 -0
- package/csrc/op_tail.c +96 -0
- package/csrc/op_top.c +150 -0
- package/csrc/op_trim.c +109 -0
- package/csrc/op_unique.c +300 -0
- package/csrc/op_unpivot.c +159 -0
- package/csrc/op_validate.c +71 -0
- package/csrc/op_window.c +150 -0
- package/csrc/pipeline.c +315 -0
- package/csrc/plan.c +206 -0
- package/csrc/recipes.c +102 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +463 -0
- package/csrc/report.h +22 -0
- package/csrc/tranfi.h +123 -0
- package/csrc/wasm_api.c +157 -0
- package/napi_api.c +326 -0
- package/package.json +46 -57
- package/src/cli.js +193 -0
- package/src/engines/duckdb.js +109 -0
- package/src/index.js +306 -0
- package/src/native.js +22 -0
- package/src/pipeline.js +286 -0
- package/src/server.js +277 -0
- package/src/wasm.js +19 -0
- package/wasm/index.js +244 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/LICENSE +0 -21
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* ir_validate.c — Validation pass over an IR plan.
|
|
3
|
+
*
|
|
4
|
+
* Checks:
|
|
5
|
+
* 1. At least one node
|
|
6
|
+
* 2. First node must be a decoder
|
|
7
|
+
* 3. Last node must be an encoder
|
|
8
|
+
* 4. All op names exist in the registry
|
|
9
|
+
* 5. Required args are present
|
|
10
|
+
* 6. No multiple decoders or encoders
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
#include "ir.h"
|
|
14
|
+
#include "cJSON.h"
|
|
15
|
+
#include <stdlib.h>
|
|
16
|
+
#include <string.h>
|
|
17
|
+
#include <stdio.h>
|
|
18
|
+
|
|
19
|
+
#define TF_OK 0
|
|
20
|
+
#define TF_ERROR (-1)
|
|
21
|
+
|
|
22
|
+
static void set_plan_error(tf_ir_plan *plan, const char *msg) {
|
|
23
|
+
free(plan->error);
|
|
24
|
+
plan->error = strdup(msg);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
static void set_plan_errorf(tf_ir_plan *plan, const char *fmt, const char *detail) {
|
|
28
|
+
free(plan->error);
|
|
29
|
+
char buf[256];
|
|
30
|
+
snprintf(buf, sizeof(buf), fmt, detail);
|
|
31
|
+
plan->error = strdup(buf);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
int tf_ir_validate(tf_ir_plan *plan) {
|
|
35
|
+
plan->validated = false;
|
|
36
|
+
free(plan->error);
|
|
37
|
+
plan->error = NULL;
|
|
38
|
+
|
|
39
|
+
/* 1. At least one node */
|
|
40
|
+
if (plan->n_nodes == 0) {
|
|
41
|
+
set_plan_error(plan, "plan has no steps");
|
|
42
|
+
return TF_ERROR;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
bool has_decoder = false;
|
|
46
|
+
bool has_encoder = false;
|
|
47
|
+
|
|
48
|
+
for (size_t i = 0; i < plan->n_nodes; i++) {
|
|
49
|
+
tf_ir_node *node = &plan->nodes[i];
|
|
50
|
+
|
|
51
|
+
/* 4. Op must exist in registry */
|
|
52
|
+
const tf_op_entry *entry = tf_op_registry_find(node->op);
|
|
53
|
+
if (!entry) {
|
|
54
|
+
set_plan_errorf(plan, "unknown op: '%s'", node->op);
|
|
55
|
+
return TF_ERROR;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/* Populate caps from registry */
|
|
59
|
+
node->caps = entry->caps;
|
|
60
|
+
|
|
61
|
+
/* Check decoder/encoder placement */
|
|
62
|
+
if (entry->kind == TF_OP_DECODER) {
|
|
63
|
+
if (has_decoder) {
|
|
64
|
+
set_plan_error(plan, "multiple decoders not supported");
|
|
65
|
+
return TF_ERROR;
|
|
66
|
+
}
|
|
67
|
+
/* 2. Decoder must be the first node */
|
|
68
|
+
if (i != 0) {
|
|
69
|
+
set_plan_errorf(plan, "decoder '%s' must be the first step", node->op);
|
|
70
|
+
return TF_ERROR;
|
|
71
|
+
}
|
|
72
|
+
has_decoder = true;
|
|
73
|
+
} else if (entry->kind == TF_OP_ENCODER) {
|
|
74
|
+
if (has_encoder) {
|
|
75
|
+
set_plan_error(plan, "multiple encoders not supported");
|
|
76
|
+
return TF_ERROR;
|
|
77
|
+
}
|
|
78
|
+
/* 3. Encoder must be the last node */
|
|
79
|
+
if (i != plan->n_nodes - 1) {
|
|
80
|
+
set_plan_errorf(plan, "encoder '%s' must be the last step", node->op);
|
|
81
|
+
return TF_ERROR;
|
|
82
|
+
}
|
|
83
|
+
has_encoder = true;
|
|
84
|
+
} else {
|
|
85
|
+
/* Transform must not be first or last if we expect decoder/encoder */
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/* 5. Required args present */
|
|
89
|
+
for (size_t a = 0; a < entry->n_args; a++) {
|
|
90
|
+
if (!entry->args[a].required) continue;
|
|
91
|
+
cJSON *val = cJSON_GetObjectItemCaseSensitive(node->args,
|
|
92
|
+
entry->args[a].name);
|
|
93
|
+
if (!val) {
|
|
94
|
+
char buf[256];
|
|
95
|
+
snprintf(buf, sizeof(buf), "op '%s' missing required arg '%s'",
|
|
96
|
+
node->op, entry->args[a].name);
|
|
97
|
+
set_plan_error(plan, buf);
|
|
98
|
+
return TF_ERROR;
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (!has_decoder) {
|
|
104
|
+
set_plan_error(plan, "plan has no decoder (need a codec.*.decode step)");
|
|
105
|
+
return TF_ERROR;
|
|
106
|
+
}
|
|
107
|
+
if (!has_encoder) {
|
|
108
|
+
set_plan_error(plan, "plan has no encoder (need a codec.*.encode step)");
|
|
109
|
+
return TF_ERROR;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/* Compute plan-level caps (intersection of all node caps) */
|
|
113
|
+
plan->plan_caps = ~(uint32_t)0;
|
|
114
|
+
for (size_t i = 0; i < plan->n_nodes; i++) {
|
|
115
|
+
plan->plan_caps &= plan->nodes[i].caps;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
plan->validated = true;
|
|
119
|
+
return TF_OK;
|
|
120
|
+
}
|
package/csrc/main.c
ADDED
|
@@ -0,0 +1,392 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* main.c — Tranfi CLI.
|
|
3
|
+
*
|
|
4
|
+
* Usage:
|
|
5
|
+
* tranfi 'csv | filter "col(age) > 25" | select name,age | csv' < in.csv
|
|
6
|
+
* tranfi -f pipeline.tf < in.csv > out.csv
|
|
7
|
+
* tranfi -j 'csv | head 5 | csv' # compile only, output JSON
|
|
8
|
+
* tranfi -i input.csv -o output.csv 'csv | filter "col(age) > 25" | csv'
|
|
9
|
+
*
|
|
10
|
+
* Channels:
|
|
11
|
+
* stdout — main output (encoded data)
|
|
12
|
+
* stderr — stats and errors
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
#include "tranfi.h"
|
|
16
|
+
#include "internal.h"
|
|
17
|
+
#include "ir.h"
|
|
18
|
+
#include "dsl.h"
|
|
19
|
+
#include "recipes.h"
|
|
20
|
+
#include "report.h"
|
|
21
|
+
#include <stdio.h>
|
|
22
|
+
#include <stdlib.h>
|
|
23
|
+
#include <string.h>
|
|
24
|
+
#include <unistd.h>
|
|
25
|
+
|
|
26
|
+
#define READ_BUF_SIZE (64 * 1024)
|
|
27
|
+
#define PULL_BUF_SIZE (64 * 1024)
|
|
28
|
+
|
|
29
|
+
static void usage(const char *prog) {
|
|
30
|
+
fprintf(stderr,
|
|
31
|
+
"Usage: %s [OPTIONS] PIPELINE\n"
|
|
32
|
+
" %s [OPTIONS] -f FILE\n"
|
|
33
|
+
"\n"
|
|
34
|
+
"Streaming ETL with a pipe-style DSL.\n"
|
|
35
|
+
"\n"
|
|
36
|
+
"Examples:\n"
|
|
37
|
+
" %s 'csv | csv' # passthrough\n"
|
|
38
|
+
" %s 'csv | filter \"col(age) > 25\" | csv' # filter rows\n"
|
|
39
|
+
" %s 'csv | select name,age | csv' # select columns\n"
|
|
40
|
+
" %s 'csv | rename name=full_name | csv' # rename columns\n"
|
|
41
|
+
" %s 'csv | head 10 | csv' # first N rows\n"
|
|
42
|
+
" %s 'csv | skip 5 | csv' # skip first 5 rows\n"
|
|
43
|
+
" %s 'csv | derive total=col(price)*col(qty) | csv' # computed columns\n"
|
|
44
|
+
" %s 'csv | sort age | csv' # sort by column\n"
|
|
45
|
+
" %s 'csv | unique name | csv' # deduplicate\n"
|
|
46
|
+
" %s 'csv | stats | csv' # aggregate stats\n"
|
|
47
|
+
" %s 'jsonl | filter \"col(x) > 0\" | jsonl' # JSONL variant\n"
|
|
48
|
+
"\n"
|
|
49
|
+
"Options:\n"
|
|
50
|
+
" -f FILE Read pipeline from file instead of argument\n"
|
|
51
|
+
" -i FILE Read input from file instead of stdin\n"
|
|
52
|
+
" -o FILE Write output to file instead of stdout\n"
|
|
53
|
+
" -j Output plan as JSON (compile only, don't execute)\n"
|
|
54
|
+
" -p, --progress Show progress on stderr\n"
|
|
55
|
+
" -q Quiet mode (suppress stats on stderr)\n"
|
|
56
|
+
" --raw Force raw CSV stats output (disable report formatting)\n"
|
|
57
|
+
" -v Show version\n"
|
|
58
|
+
" -R, --recipes List built-in recipes\n"
|
|
59
|
+
" -h Show this help\n"
|
|
60
|
+
"\n"
|
|
61
|
+
"Recipes (use by name, e.g. %s profile):\n"
|
|
62
|
+
" profile, preview, schema, summary, count, cardinality,\n"
|
|
63
|
+
" distro, freq, dedup, clean, sample, head, tail, csv2json,\n"
|
|
64
|
+
" json2csv, tsv2csv, csv2tsv, histogram, hash, samples\n",
|
|
65
|
+
prog, prog, prog, prog, prog, prog, prog, prog,
|
|
66
|
+
prog, prog, prog, prog, prog, prog);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
static char *read_file(const char *path) {
|
|
70
|
+
FILE *f = fopen(path, "r");
|
|
71
|
+
if (!f) return NULL;
|
|
72
|
+
|
|
73
|
+
fseek(f, 0, SEEK_END);
|
|
74
|
+
long size = ftell(f);
|
|
75
|
+
fseek(f, 0, SEEK_SET);
|
|
76
|
+
|
|
77
|
+
if (size <= 0) { fclose(f); return NULL; }
|
|
78
|
+
|
|
79
|
+
char *buf = malloc(size + 1);
|
|
80
|
+
if (!buf) { fclose(f); return NULL; }
|
|
81
|
+
|
|
82
|
+
size_t nread = fread(buf, 1, size, f);
|
|
83
|
+
fclose(f);
|
|
84
|
+
buf[nread] = '\0';
|
|
85
|
+
return buf;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
static const char *format_bytes(size_t bytes, char *buf, size_t buf_size) {
|
|
89
|
+
if (bytes < 1024) {
|
|
90
|
+
snprintf(buf, buf_size, "%zuB", bytes);
|
|
91
|
+
} else if (bytes < 1024 * 1024) {
|
|
92
|
+
snprintf(buf, buf_size, "%.1fKB", (double)bytes / 1024);
|
|
93
|
+
} else if (bytes < 1024 * 1024 * 1024) {
|
|
94
|
+
snprintf(buf, buf_size, "%.1fMB", (double)bytes / (1024 * 1024));
|
|
95
|
+
} else {
|
|
96
|
+
snprintf(buf, buf_size, "%.1fGB", (double)bytes / (1024 * 1024 * 1024));
|
|
97
|
+
}
|
|
98
|
+
return buf;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
int main(int argc, char **argv) {
|
|
102
|
+
const char *pipeline_file = NULL;
|
|
103
|
+
const char *pipeline_text = NULL;
|
|
104
|
+
const char *input_file = NULL;
|
|
105
|
+
const char *output_file = NULL;
|
|
106
|
+
int json_mode = 0;
|
|
107
|
+
int quiet = 0;
|
|
108
|
+
int progress = 0;
|
|
109
|
+
int raw_stats = 0;
|
|
110
|
+
|
|
111
|
+
/* Parse options */
|
|
112
|
+
int argi = 1;
|
|
113
|
+
while (argi < argc && argv[argi][0] == '-') {
|
|
114
|
+
const char *opt = argv[argi];
|
|
115
|
+
if (strcmp(opt, "-h") == 0 || strcmp(opt, "--help") == 0) {
|
|
116
|
+
usage(argv[0]);
|
|
117
|
+
return 0;
|
|
118
|
+
} else if (strcmp(opt, "-v") == 0 || strcmp(opt, "--version") == 0) {
|
|
119
|
+
printf("tranfi %s\n", tf_version());
|
|
120
|
+
return 0;
|
|
121
|
+
} else if (strcmp(opt, "-R") == 0 || strcmp(opt, "--recipes") == 0) {
|
|
122
|
+
size_t n = tf_recipe_count();
|
|
123
|
+
printf("Built-in recipes (%zu):\n\n", n);
|
|
124
|
+
for (size_t i = 0; i < n; i++) {
|
|
125
|
+
printf(" %-12s %s\n", tf_recipe_name(i), tf_recipe_description(i));
|
|
126
|
+
printf(" %-12s %s\n", "", tf_recipe_dsl(i));
|
|
127
|
+
printf("\n");
|
|
128
|
+
}
|
|
129
|
+
return 0;
|
|
130
|
+
} else if (strcmp(opt, "-j") == 0) {
|
|
131
|
+
json_mode = 1;
|
|
132
|
+
} else if (strcmp(opt, "-q") == 0) {
|
|
133
|
+
quiet = 1;
|
|
134
|
+
} else if (strcmp(opt, "-p") == 0 || strcmp(opt, "--progress") == 0) {
|
|
135
|
+
progress = 1;
|
|
136
|
+
} else if (strcmp(opt, "--raw") == 0) {
|
|
137
|
+
raw_stats = 1;
|
|
138
|
+
} else if (strcmp(opt, "-f") == 0) {
|
|
139
|
+
argi++;
|
|
140
|
+
if (argi >= argc) {
|
|
141
|
+
fprintf(stderr, "error: -f requires a file argument\n");
|
|
142
|
+
return 1;
|
|
143
|
+
}
|
|
144
|
+
pipeline_file = argv[argi];
|
|
145
|
+
} else if (strcmp(opt, "-i") == 0) {
|
|
146
|
+
argi++;
|
|
147
|
+
if (argi >= argc) {
|
|
148
|
+
fprintf(stderr, "error: -i requires a file argument\n");
|
|
149
|
+
return 1;
|
|
150
|
+
}
|
|
151
|
+
input_file = argv[argi];
|
|
152
|
+
} else if (strcmp(opt, "-o") == 0) {
|
|
153
|
+
argi++;
|
|
154
|
+
if (argi >= argc) {
|
|
155
|
+
fprintf(stderr, "error: -o requires a file argument\n");
|
|
156
|
+
return 1;
|
|
157
|
+
}
|
|
158
|
+
output_file = argv[argi];
|
|
159
|
+
} else {
|
|
160
|
+
fprintf(stderr, "error: unknown option '%s'\n", opt);
|
|
161
|
+
return 1;
|
|
162
|
+
}
|
|
163
|
+
argi++;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/* Get pipeline text */
|
|
167
|
+
char *file_content = NULL;
|
|
168
|
+
if (pipeline_file) {
|
|
169
|
+
file_content = read_file(pipeline_file);
|
|
170
|
+
if (!file_content) {
|
|
171
|
+
fprintf(stderr, "error: cannot read file '%s'\n", pipeline_file);
|
|
172
|
+
return 1;
|
|
173
|
+
}
|
|
174
|
+
pipeline_text = file_content;
|
|
175
|
+
} else if (argi < argc) {
|
|
176
|
+
pipeline_text = argv[argi];
|
|
177
|
+
} else {
|
|
178
|
+
fprintf(stderr, "error: no pipeline specified\n\n");
|
|
179
|
+
usage(argv[0]);
|
|
180
|
+
free(file_content);
|
|
181
|
+
return 1;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/* Parse pipeline: recipe name → JSON recipe → DSL */
|
|
185
|
+
char *error = NULL;
|
|
186
|
+
tf_ir_plan *ir = NULL;
|
|
187
|
+
size_t pt_len = strlen(pipeline_text);
|
|
188
|
+
/* Skip leading whitespace for detection */
|
|
189
|
+
const char *pt = pipeline_text;
|
|
190
|
+
while (*pt == ' ' || *pt == '\t' || *pt == '\n' || *pt == '\r') pt++;
|
|
191
|
+
|
|
192
|
+
if (*pt == '{') {
|
|
193
|
+
/* JSON recipe */
|
|
194
|
+
ir = tf_ir_from_json(pipeline_text, pt_len, &error);
|
|
195
|
+
} else if (!strchr(pt, '|') && !strchr(pt, ' ')) {
|
|
196
|
+
/* Single word — try built-in recipe */
|
|
197
|
+
const char *recipe_dsl = tf_recipe_find_dsl(pt);
|
|
198
|
+
if (recipe_dsl) {
|
|
199
|
+
ir = tf_dsl_parse(recipe_dsl, strlen(recipe_dsl), &error);
|
|
200
|
+
} else {
|
|
201
|
+
ir = tf_dsl_parse(pipeline_text, pt_len, &error);
|
|
202
|
+
}
|
|
203
|
+
} else {
|
|
204
|
+
ir = tf_dsl_parse(pipeline_text, pt_len, &error);
|
|
205
|
+
}
|
|
206
|
+
free(file_content);
|
|
207
|
+
|
|
208
|
+
if (!ir) {
|
|
209
|
+
fprintf(stderr, "error: %s\n", error ? error : "failed to parse pipeline");
|
|
210
|
+
free(error);
|
|
211
|
+
return 1;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/* Validate */
|
|
215
|
+
if (tf_ir_validate(ir) != TF_OK) {
|
|
216
|
+
fprintf(stderr, "error: %s\n", ir->error ? ir->error : "validation failed");
|
|
217
|
+
tf_ir_plan_free(ir);
|
|
218
|
+
return 1;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/* Schema inference (best-effort) */
|
|
222
|
+
tf_ir_infer_schema(ir);
|
|
223
|
+
|
|
224
|
+
/* JSON mode: print IR and exit */
|
|
225
|
+
if (json_mode) {
|
|
226
|
+
char *json = tf_ir_to_json(ir);
|
|
227
|
+
if (json) {
|
|
228
|
+
printf("%s\n", json);
|
|
229
|
+
free(json);
|
|
230
|
+
}
|
|
231
|
+
tf_ir_plan_free(ir);
|
|
232
|
+
return 0;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
/* Compile to native pipeline */
|
|
236
|
+
tf_pipeline *p = tf_pipeline_create_from_ir(ir);
|
|
237
|
+
tf_ir_plan_free(ir);
|
|
238
|
+
|
|
239
|
+
if (!p) {
|
|
240
|
+
fprintf(stderr, "error: %s\n",
|
|
241
|
+
tf_last_error() ? tf_last_error() : "failed to create pipeline");
|
|
242
|
+
return 1;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/* Open I/O files */
|
|
246
|
+
FILE *fin = stdin;
|
|
247
|
+
FILE *fout = stdout;
|
|
248
|
+
|
|
249
|
+
if (input_file) {
|
|
250
|
+
fin = fopen(input_file, "rb");
|
|
251
|
+
if (!fin) {
|
|
252
|
+
fprintf(stderr, "error: cannot open input file '%s'\n", input_file);
|
|
253
|
+
tf_pipeline_free(p);
|
|
254
|
+
return 1;
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
if (output_file) {
|
|
259
|
+
fout = fopen(output_file, "wb");
|
|
260
|
+
if (!fout) {
|
|
261
|
+
fprintf(stderr, "error: cannot open output file '%s'\n", output_file);
|
|
262
|
+
if (fin != stdin) fclose(fin);
|
|
263
|
+
tf_pipeline_free(p);
|
|
264
|
+
return 1;
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
/* Decide whether to buffer output for report formatting.
|
|
269
|
+
* When stdout is a TTY and --raw is not set, buffer main output
|
|
270
|
+
* and try to render it as a rich report. Falls back to raw CSV
|
|
271
|
+
* if the output doesn't look like a stats table. */
|
|
272
|
+
int try_report = !raw_stats && !output_file && isatty(STDOUT_FILENO);
|
|
273
|
+
|
|
274
|
+
/* Stream input → pipeline → output */
|
|
275
|
+
uint8_t read_buf[READ_BUF_SIZE];
|
|
276
|
+
size_t nread;
|
|
277
|
+
size_t total_bytes = 0;
|
|
278
|
+
|
|
279
|
+
/* Output buffer (used when try_report is true) */
|
|
280
|
+
size_t out_cap = PULL_BUF_SIZE;
|
|
281
|
+
size_t out_len = 0;
|
|
282
|
+
char *out_buf = try_report ? malloc(out_cap) : NULL;
|
|
283
|
+
|
|
284
|
+
while ((nread = fread(read_buf, 1, sizeof(read_buf), fin)) > 0) {
|
|
285
|
+
if (tf_pipeline_push(p, read_buf, nread) != TF_OK) {
|
|
286
|
+
fprintf(stderr, "error: %s\n",
|
|
287
|
+
tf_pipeline_error(p) ? tf_pipeline_error(p) : "push failed");
|
|
288
|
+
free(out_buf);
|
|
289
|
+
if (fin != stdin) fclose(fin);
|
|
290
|
+
if (fout != stdout) fclose(fout);
|
|
291
|
+
tf_pipeline_free(p);
|
|
292
|
+
return 1;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
total_bytes += nread;
|
|
296
|
+
|
|
297
|
+
/* Pull any available output immediately (streaming) */
|
|
298
|
+
uint8_t pull_buf[PULL_BUF_SIZE];
|
|
299
|
+
size_t n;
|
|
300
|
+
while ((n = tf_pipeline_pull(p, TF_CHAN_MAIN, pull_buf, sizeof(pull_buf))) > 0) {
|
|
301
|
+
if (try_report && out_buf) {
|
|
302
|
+
while (out_len + n > out_cap) {
|
|
303
|
+
out_cap *= 2;
|
|
304
|
+
char *tmp = realloc(out_buf, out_cap);
|
|
305
|
+
if (!tmp) { free(out_buf); out_buf = NULL; break; }
|
|
306
|
+
out_buf = tmp;
|
|
307
|
+
}
|
|
308
|
+
if (out_buf) {
|
|
309
|
+
memcpy(out_buf + out_len, pull_buf, n);
|
|
310
|
+
out_len += n;
|
|
311
|
+
}
|
|
312
|
+
} else {
|
|
313
|
+
fwrite(pull_buf, 1, n, fout);
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
/* Show progress */
|
|
318
|
+
if (progress) {
|
|
319
|
+
char bytes_str[32];
|
|
320
|
+
format_bytes(total_bytes, bytes_str, sizeof(bytes_str));
|
|
321
|
+
fprintf(stderr, "\r%s processed", bytes_str);
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/* Finish */
|
|
326
|
+
if (tf_pipeline_finish(p) != TF_OK) {
|
|
327
|
+
fprintf(stderr, "error: %s\n",
|
|
328
|
+
tf_pipeline_error(p) ? tf_pipeline_error(p) : "finish failed");
|
|
329
|
+
free(out_buf);
|
|
330
|
+
if (fin != stdin) fclose(fin);
|
|
331
|
+
if (fout != stdout) fclose(fout);
|
|
332
|
+
tf_pipeline_free(p);
|
|
333
|
+
return 1;
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
/* Pull remaining output */
|
|
337
|
+
uint8_t pull_buf[PULL_BUF_SIZE];
|
|
338
|
+
size_t n;
|
|
339
|
+
while ((n = tf_pipeline_pull(p, TF_CHAN_MAIN, pull_buf, sizeof(pull_buf))) > 0) {
|
|
340
|
+
if (try_report && out_buf) {
|
|
341
|
+
while (out_len + n > out_cap) {
|
|
342
|
+
out_cap *= 2;
|
|
343
|
+
char *tmp = realloc(out_buf, out_cap);
|
|
344
|
+
if (!tmp) { free(out_buf); out_buf = NULL; break; }
|
|
345
|
+
out_buf = tmp;
|
|
346
|
+
}
|
|
347
|
+
if (out_buf) {
|
|
348
|
+
memcpy(out_buf + out_len, pull_buf, n);
|
|
349
|
+
out_len += n;
|
|
350
|
+
}
|
|
351
|
+
} else {
|
|
352
|
+
fwrite(pull_buf, 1, n, fout);
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
/* Try report formatting, fall back to raw */
|
|
357
|
+
if (try_report && out_buf && out_len > 0) {
|
|
358
|
+
char *report = tf_report_format(out_buf, out_len, 1);
|
|
359
|
+
if (report) {
|
|
360
|
+
fwrite(report, 1, strlen(report), fout);
|
|
361
|
+
free(report);
|
|
362
|
+
} else {
|
|
363
|
+
fwrite(out_buf, 1, out_len, fout);
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
free(out_buf);
|
|
367
|
+
fflush(fout);
|
|
368
|
+
|
|
369
|
+
if (progress) {
|
|
370
|
+
char bytes_str[32];
|
|
371
|
+
format_bytes(total_bytes, bytes_str, sizeof(bytes_str));
|
|
372
|
+
fprintf(stderr, "\r%s processed (done)\n", bytes_str);
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/* Pull errors to stderr */
|
|
376
|
+
while ((n = tf_pipeline_pull(p, TF_CHAN_ERRORS, pull_buf, sizeof(pull_buf))) > 0) {
|
|
377
|
+
fwrite(pull_buf, 1, n, stderr);
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
/* Pull stats to stderr (unless quiet) */
|
|
381
|
+
if (!quiet) {
|
|
382
|
+
while ((n = tf_pipeline_pull(p, TF_CHAN_STATS, pull_buf, sizeof(pull_buf))) > 0) {
|
|
383
|
+
fwrite(pull_buf, 1, n, stderr);
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
/* Cleanup */
|
|
388
|
+
if (fin != stdin) fclose(fin);
|
|
389
|
+
if (fout != stdout) fclose(fout);
|
|
390
|
+
tf_pipeline_free(p);
|
|
391
|
+
return 0;
|
|
392
|
+
}
|
package/csrc/op_acf.c
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_acf.c — Autocorrelation function.
|
|
3
|
+
* Aggregate op: buffers all values, computes ACF for lags 0..N.
|
|
4
|
+
* Output: 2-column table (lag, acf).
|
|
5
|
+
*
|
|
6
|
+
* Config: {"column": "price", "lags": 20}
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
#include "internal.h"
|
|
10
|
+
#include "cJSON.h"
|
|
11
|
+
#include <stdlib.h>
|
|
12
|
+
#include <string.h>
|
|
13
|
+
#include <stdio.h>
|
|
14
|
+
#include <math.h>
|
|
15
|
+
|
|
16
|
+
typedef struct {
|
|
17
|
+
char *column;
|
|
18
|
+
int lags;
|
|
19
|
+
double *values;
|
|
20
|
+
size_t n_values;
|
|
21
|
+
size_t cap_values;
|
|
22
|
+
} acf_state;
|
|
23
|
+
|
|
24
|
+
static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
25
|
+
if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
|
|
26
|
+
if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
|
|
27
|
+
return 0;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
static int acf_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
31
|
+
tf_side_channels *side) {
|
|
32
|
+
(void)side;
|
|
33
|
+
acf_state *st = self->state;
|
|
34
|
+
*out = NULL;
|
|
35
|
+
|
|
36
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
37
|
+
if (ci < 0) return TF_OK;
|
|
38
|
+
|
|
39
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
40
|
+
if (tf_batch_is_null(in, r, ci)) continue;
|
|
41
|
+
double val = get_numeric(in, r, ci);
|
|
42
|
+
|
|
43
|
+
if (st->n_values >= st->cap_values) {
|
|
44
|
+
size_t newcap = st->cap_values ? st->cap_values * 2 : 256;
|
|
45
|
+
double *tmp = realloc(st->values, newcap * sizeof(double));
|
|
46
|
+
if (!tmp) return TF_ERROR;
|
|
47
|
+
st->values = tmp;
|
|
48
|
+
st->cap_values = newcap;
|
|
49
|
+
}
|
|
50
|
+
st->values[st->n_values++] = val;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
return TF_OK;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
static int acf_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
57
|
+
(void)side;
|
|
58
|
+
acf_state *st = self->state;
|
|
59
|
+
*out = NULL;
|
|
60
|
+
|
|
61
|
+
size_t n = st->n_values;
|
|
62
|
+
if (n < 2) return TF_OK;
|
|
63
|
+
|
|
64
|
+
int max_lag = st->lags;
|
|
65
|
+
if ((size_t)max_lag >= n) max_lag = (int)(n - 1);
|
|
66
|
+
|
|
67
|
+
/* Compute mean */
|
|
68
|
+
double mean = 0;
|
|
69
|
+
for (size_t i = 0; i < n; i++) mean += st->values[i];
|
|
70
|
+
mean /= (double)n;
|
|
71
|
+
|
|
72
|
+
/* Compute variance (denominator for ACF) */
|
|
73
|
+
double var = 0;
|
|
74
|
+
for (size_t i = 0; i < n; i++) {
|
|
75
|
+
double d = st->values[i] - mean;
|
|
76
|
+
var += d * d;
|
|
77
|
+
}
|
|
78
|
+
if (var == 0) return TF_OK;
|
|
79
|
+
|
|
80
|
+
/* Output: lag+1 rows (lag 0 to max_lag) */
|
|
81
|
+
size_t out_rows = (size_t)(max_lag + 1);
|
|
82
|
+
tf_batch *ob = tf_batch_create(2, out_rows);
|
|
83
|
+
if (!ob) return TF_ERROR;
|
|
84
|
+
tf_batch_set_schema(ob, 0, "lag", TF_TYPE_INT64);
|
|
85
|
+
tf_batch_set_schema(ob, 1, "acf", TF_TYPE_FLOAT64);
|
|
86
|
+
|
|
87
|
+
for (int k = 0; k <= max_lag; k++) {
|
|
88
|
+
double cov = 0;
|
|
89
|
+
for (size_t i = 0; i < n - (size_t)k; i++) {
|
|
90
|
+
cov += (st->values[i] - mean) * (st->values[i + k] - mean);
|
|
91
|
+
}
|
|
92
|
+
double acf_val = cov / var;
|
|
93
|
+
|
|
94
|
+
tf_batch_set_int64(ob, k, 0, k);
|
|
95
|
+
tf_batch_set_float64(ob, k, 1, acf_val);
|
|
96
|
+
ob->n_rows = k + 1;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
*out = ob;
|
|
100
|
+
return TF_OK;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
static void acf_destroy(tf_step *self) {
|
|
104
|
+
acf_state *st = self->state;
|
|
105
|
+
if (st) {
|
|
106
|
+
free(st->column);
|
|
107
|
+
free(st->values);
|
|
108
|
+
free(st);
|
|
109
|
+
}
|
|
110
|
+
free(self);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
tf_step *tf_acf_create(const cJSON *args) {
|
|
114
|
+
if (!args) return NULL;
|
|
115
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
116
|
+
if (!cJSON_IsString(col_j)) return NULL;
|
|
117
|
+
|
|
118
|
+
acf_state *st = calloc(1, sizeof(acf_state));
|
|
119
|
+
if (!st) return NULL;
|
|
120
|
+
st->column = strdup(col_j->valuestring);
|
|
121
|
+
|
|
122
|
+
cJSON *lags_j = cJSON_GetObjectItemCaseSensitive(args, "lags");
|
|
123
|
+
st->lags = cJSON_IsNumber(lags_j) ? lags_j->valueint : 20;
|
|
124
|
+
if (st->lags < 1) st->lags = 1;
|
|
125
|
+
|
|
126
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
127
|
+
if (!step) { free(st->column); free(st); return NULL; }
|
|
128
|
+
step->process = acf_process;
|
|
129
|
+
step->flush = acf_flush;
|
|
130
|
+
step->destroy = acf_destroy;
|
|
131
|
+
step->state = st;
|
|
132
|
+
return step;
|
|
133
|
+
}
|