tranfi 0.0.2 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +395 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-pDFMluyz.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +69 -0
- package/csrc/arena.c +91 -0
- package/csrc/batch.c +229 -0
- package/csrc/buffer.c +78 -0
- package/csrc/cJSON.c +3143 -0
- package/csrc/cJSON.h +300 -0
- package/csrc/codec_csv.c +1058 -0
- package/csrc/codec_jsonl.c +374 -0
- package/csrc/codec_table.c +218 -0
- package/csrc/codec_text.c +229 -0
- package/csrc/compiler.c +102 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +1180 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1245 -0
- package/csrc/expr.h +56 -0
- package/csrc/internal.h +250 -0
- package/csrc/ir.c +119 -0
- package/csrc/ir.h +167 -0
- package/csrc/ir_schema.c +60 -0
- package/csrc/ir_serialize.c +104 -0
- package/csrc/ir_sql.c +1211 -0
- package/csrc/ir_validate.c +120 -0
- package/csrc/main.c +392 -0
- package/csrc/op_acf.c +133 -0
- package/csrc/op_anomaly.c +120 -0
- package/csrc/op_bin.c +109 -0
- package/csrc/op_cast.c +195 -0
- package/csrc/op_clip.c +88 -0
- package/csrc/op_date_trunc.c +181 -0
- package/csrc/op_datetime.c +212 -0
- package/csrc/op_derive.c +248 -0
- package/csrc/op_diff.c +134 -0
- package/csrc/op_ewma.c +103 -0
- package/csrc/op_explode.c +108 -0
- package/csrc/op_fill_down.c +163 -0
- package/csrc/op_fill_null.c +123 -0
- package/csrc/op_filter.c +132 -0
- package/csrc/op_frequency.c +193 -0
- package/csrc/op_grep.c +163 -0
- package/csrc/op_group_agg.c +285 -0
- package/csrc/op_hash.c +126 -0
- package/csrc/op_head.c +149 -0
- package/csrc/op_interpolate.c +239 -0
- package/csrc/op_join.c +384 -0
- package/csrc/op_label_encode.c +144 -0
- package/csrc/op_lead.c +190 -0
- package/csrc/op_normalize.c +226 -0
- package/csrc/op_onehot.c +185 -0
- package/csrc/op_pivot.c +370 -0
- package/csrc/op_registry.c +1148 -0
- package/csrc/op_rename.c +138 -0
- package/csrc/op_replace.c +202 -0
- package/csrc/op_sample.c +101 -0
- package/csrc/op_select.c +140 -0
- package/csrc/op_skip.c +152 -0
- package/csrc/op_sort.c +273 -0
- package/csrc/op_split.c +114 -0
- package/csrc/op_split_data.c +87 -0
- package/csrc/op_stack.c +315 -0
- package/csrc/op_stats.c +779 -0
- package/csrc/op_step.c +171 -0
- package/csrc/op_tail.c +96 -0
- package/csrc/op_top.c +150 -0
- package/csrc/op_trim.c +109 -0
- package/csrc/op_unique.c +300 -0
- package/csrc/op_unpivot.c +159 -0
- package/csrc/op_validate.c +71 -0
- package/csrc/op_window.c +150 -0
- package/csrc/pipeline.c +315 -0
- package/csrc/plan.c +206 -0
- package/csrc/recipes.c +102 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +463 -0
- package/csrc/report.h +22 -0
- package/csrc/tranfi.h +123 -0
- package/csrc/wasm_api.c +157 -0
- package/napi_api.c +326 -0
- package/package.json +46 -57
- package/src/cli.js +193 -0
- package/src/engines/duckdb.js +109 -0
- package/src/index.js +306 -0
- package/src/native.js +22 -0
- package/src/pipeline.js +286 -0
- package/src/server.js +277 -0
- package/src/wasm.js +19 -0
- package/wasm/index.js +244 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/LICENSE +0 -21
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/csrc/report.c
ADDED
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* report.c — Rich ANSI terminal report for stats output.
|
|
3
|
+
*
|
|
4
|
+
* Parses the CSV text produced by the stats channel and renders a
|
|
5
|
+
* compact per-column summary with Unicode sparkline histograms.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "report.h"
|
|
9
|
+
#include <stdlib.h>
|
|
10
|
+
#include <string.h>
|
|
11
|
+
#include <stdio.h>
|
|
12
|
+
#include <stdarg.h>
|
|
13
|
+
#include <math.h>
|
|
14
|
+
|
|
15
|
+
/* ---- ANSI escape codes ---- */
|
|
16
|
+
|
|
17
|
+
#define ANSI_BOLD "\033[1m"
|
|
18
|
+
#define ANSI_DIM "\033[2m"
|
|
19
|
+
#define ANSI_CYAN "\033[36m"
|
|
20
|
+
#define ANSI_GREEN "\033[32m"
|
|
21
|
+
#define ANSI_YELLOW "\033[33m"
|
|
22
|
+
#define ANSI_RESET "\033[0m"
|
|
23
|
+
|
|
24
|
+
/* ---- Dynamic string buffer ---- */
|
|
25
|
+
|
|
26
|
+
typedef struct {
|
|
27
|
+
char *data;
|
|
28
|
+
size_t len;
|
|
29
|
+
size_t cap;
|
|
30
|
+
} strbuf;
|
|
31
|
+
|
|
32
|
+
static void sb_init(strbuf *sb) {
|
|
33
|
+
sb->cap = 4096;
|
|
34
|
+
sb->data = malloc(sb->cap);
|
|
35
|
+
sb->len = 0;
|
|
36
|
+
if (sb->data) sb->data[0] = '\0';
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
static void sb_ensure(strbuf *sb, size_t extra) {
|
|
40
|
+
if (sb->len + extra + 1 > sb->cap) {
|
|
41
|
+
while (sb->len + extra + 1 > sb->cap) sb->cap *= 2;
|
|
42
|
+
sb->data = realloc(sb->data, sb->cap);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
static void sb_append(strbuf *sb, const char *s) {
|
|
47
|
+
size_t n = strlen(s);
|
|
48
|
+
sb_ensure(sb, n);
|
|
49
|
+
memcpy(sb->data + sb->len, s, n);
|
|
50
|
+
sb->len += n;
|
|
51
|
+
sb->data[sb->len] = '\0';
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
static void sb_printf(strbuf *sb, const char *fmt, ...) {
|
|
55
|
+
va_list ap;
|
|
56
|
+
va_start(ap, fmt);
|
|
57
|
+
int needed = vsnprintf(NULL, 0, fmt, ap);
|
|
58
|
+
va_end(ap);
|
|
59
|
+
if (needed < 0) return;
|
|
60
|
+
sb_ensure(sb, (size_t)needed);
|
|
61
|
+
va_start(ap, fmt);
|
|
62
|
+
vsnprintf(sb->data + sb->len, (size_t)needed + 1, fmt, ap);
|
|
63
|
+
va_end(ap);
|
|
64
|
+
sb->len += (size_t)needed;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/* ---- Sparkline ---- */
|
|
68
|
+
|
|
69
|
+
static const char *spark_chars[] = {
|
|
70
|
+
"\xe2\x96\x81", /* ▁ U+2581 */
|
|
71
|
+
"\xe2\x96\x82", /* ▂ U+2582 */
|
|
72
|
+
"\xe2\x96\x83", /* ▃ U+2583 */
|
|
73
|
+
"\xe2\x96\x84", /* ▄ U+2584 */
|
|
74
|
+
"\xe2\x96\x85", /* ▅ U+2585 */
|
|
75
|
+
"\xe2\x96\x86", /* ▆ U+2586 */
|
|
76
|
+
"\xe2\x96\x87", /* ▇ U+2587 */
|
|
77
|
+
"\xe2\x96\x88", /* █ U+2588 */
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
/* ---- CSV parsing helpers ---- */
|
|
81
|
+
|
|
82
|
+
#define MAX_COLS 32
|
|
83
|
+
#define MAX_ROWS 256
|
|
84
|
+
#define MAX_FIELD 1024
|
|
85
|
+
|
|
86
|
+
typedef struct {
|
|
87
|
+
char *headers[MAX_COLS];
|
|
88
|
+
char *cells[MAX_ROWS][MAX_COLS];
|
|
89
|
+
size_t n_cols;
|
|
90
|
+
size_t n_rows;
|
|
91
|
+
} csv_table;
|
|
92
|
+
|
|
93
|
+
/* Find column index by name, -1 if not found */
|
|
94
|
+
static int csv_col(const csv_table *t, const char *name) {
|
|
95
|
+
for (size_t i = 0; i < t->n_cols; i++) {
|
|
96
|
+
if (t->headers[i] && strcmp(t->headers[i], name) == 0)
|
|
97
|
+
return (int)i;
|
|
98
|
+
}
|
|
99
|
+
return -1;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/* Parse a simple CSV (no quoted fields with embedded commas needed for stats) */
|
|
103
|
+
static csv_table *csv_parse(const char *csv, size_t len) {
|
|
104
|
+
csv_table *t = calloc(1, sizeof(csv_table));
|
|
105
|
+
if (!t) return NULL;
|
|
106
|
+
|
|
107
|
+
const char *p = csv;
|
|
108
|
+
const char *end = csv + len;
|
|
109
|
+
|
|
110
|
+
/* Parse header line */
|
|
111
|
+
while (p < end && *p != '\n' && *p != '\r') {
|
|
112
|
+
const char *start = p;
|
|
113
|
+
while (p < end && *p != ',' && *p != '\n' && *p != '\r') p++;
|
|
114
|
+
size_t flen = (size_t)(p - start);
|
|
115
|
+
if (flen > MAX_FIELD - 1) flen = MAX_FIELD - 1;
|
|
116
|
+
if (t->n_cols < MAX_COLS) {
|
|
117
|
+
t->headers[t->n_cols] = malloc(flen + 1);
|
|
118
|
+
if (t->headers[t->n_cols]) {
|
|
119
|
+
memcpy(t->headers[t->n_cols], start, flen);
|
|
120
|
+
t->headers[t->n_cols][flen] = '\0';
|
|
121
|
+
}
|
|
122
|
+
t->n_cols++;
|
|
123
|
+
}
|
|
124
|
+
if (p < end && *p == ',') p++;
|
|
125
|
+
}
|
|
126
|
+
/* Skip newline */
|
|
127
|
+
if (p < end && *p == '\r') p++;
|
|
128
|
+
if (p < end && *p == '\n') p++;
|
|
129
|
+
|
|
130
|
+
/* Parse data rows */
|
|
131
|
+
while (p < end && t->n_rows < MAX_ROWS) {
|
|
132
|
+
if (*p == '\n' || *p == '\r') { p++; continue; }
|
|
133
|
+
size_t ci = 0;
|
|
134
|
+
while (p < end && *p != '\n' && *p != '\r') {
|
|
135
|
+
const char *start = p;
|
|
136
|
+
/* Handle quoted fields (hist column has colons and commas inside quotes) */
|
|
137
|
+
if (*p == '"') {
|
|
138
|
+
p++; /* skip opening quote */
|
|
139
|
+
start = p;
|
|
140
|
+
while (p < end && *p != '"') p++;
|
|
141
|
+
size_t flen = (size_t)(p - start);
|
|
142
|
+
if (flen > MAX_FIELD - 1) flen = MAX_FIELD - 1;
|
|
143
|
+
if (ci < MAX_COLS) {
|
|
144
|
+
t->cells[t->n_rows][ci] = malloc(flen + 1);
|
|
145
|
+
if (t->cells[t->n_rows][ci]) {
|
|
146
|
+
memcpy(t->cells[t->n_rows][ci], start, flen);
|
|
147
|
+
t->cells[t->n_rows][ci][flen] = '\0';
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
if (p < end && *p == '"') p++; /* skip closing quote */
|
|
151
|
+
if (p < end && *p == ',') p++;
|
|
152
|
+
ci++;
|
|
153
|
+
continue;
|
|
154
|
+
}
|
|
155
|
+
while (p < end && *p != ',' && *p != '\n' && *p != '\r') p++;
|
|
156
|
+
size_t flen = (size_t)(p - start);
|
|
157
|
+
if (flen > MAX_FIELD - 1) flen = MAX_FIELD - 1;
|
|
158
|
+
if (ci < MAX_COLS) {
|
|
159
|
+
t->cells[t->n_rows][ci] = malloc(flen + 1);
|
|
160
|
+
if (t->cells[t->n_rows][ci]) {
|
|
161
|
+
memcpy(t->cells[t->n_rows][ci], start, flen);
|
|
162
|
+
t->cells[t->n_rows][ci][flen] = '\0';
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
if (p < end && *p == ',') p++;
|
|
166
|
+
ci++;
|
|
167
|
+
}
|
|
168
|
+
t->n_rows++;
|
|
169
|
+
if (p < end && *p == '\r') p++;
|
|
170
|
+
if (p < end && *p == '\n') p++;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
return t;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
static void csv_free(csv_table *t) {
|
|
177
|
+
if (!t) return;
|
|
178
|
+
for (size_t i = 0; i < t->n_cols; i++) free(t->headers[i]);
|
|
179
|
+
for (size_t r = 0; r < t->n_rows; r++)
|
|
180
|
+
for (size_t c = 0; c < t->n_cols; c++)
|
|
181
|
+
free(t->cells[r][c]);
|
|
182
|
+
free(t);
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
static const char *csv_get(const csv_table *t, size_t row, int col) {
|
|
186
|
+
if (col < 0 || (size_t)col >= t->n_cols || row >= t->n_rows) return NULL;
|
|
187
|
+
return t->cells[row][col];
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
static double csv_getf(const csv_table *t, size_t row, int col) {
|
|
191
|
+
const char *s = csv_get(t, row, col);
|
|
192
|
+
if (!s || *s == '\0') return NAN;
|
|
193
|
+
return atof(s);
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
static long long csv_getll(const csv_table *t, size_t row, int col) {
|
|
197
|
+
const char *s = csv_get(t, row, col);
|
|
198
|
+
if (!s || *s == '\0') return 0;
|
|
199
|
+
return atoll(s);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/* ---- Histogram parsing ---- */
|
|
203
|
+
|
|
204
|
+
#define HIST_BINS 32
|
|
205
|
+
|
|
206
|
+
typedef struct {
|
|
207
|
+
double lo, hi;
|
|
208
|
+
size_t counts[HIST_BINS];
|
|
209
|
+
size_t total;
|
|
210
|
+
} parsed_hist;
|
|
211
|
+
|
|
212
|
+
static int parse_hist(const char *s, parsed_hist *h) {
|
|
213
|
+
if (!s || !*s) return 0;
|
|
214
|
+
memset(h, 0, sizeof(*h));
|
|
215
|
+
|
|
216
|
+
/* Format: "lo:hi:c1,c2,...,c32" */
|
|
217
|
+
char *endp;
|
|
218
|
+
h->lo = strtod(s, &endp);
|
|
219
|
+
if (*endp != ':') return 0;
|
|
220
|
+
endp++;
|
|
221
|
+
h->hi = strtod(endp, &endp);
|
|
222
|
+
if (*endp != ':') return 0;
|
|
223
|
+
endp++;
|
|
224
|
+
|
|
225
|
+
for (int i = 0; i < HIST_BINS; i++) {
|
|
226
|
+
h->counts[i] = (size_t)strtoul(endp, &endp, 10);
|
|
227
|
+
h->total += h->counts[i];
|
|
228
|
+
if (*endp == ',') endp++;
|
|
229
|
+
}
|
|
230
|
+
return 1;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/* ---- Number formatting ---- */
|
|
234
|
+
|
|
235
|
+
static void fmt_num(char *buf, size_t bufsz, double v) {
|
|
236
|
+
if (isnan(v)) { snprintf(buf, bufsz, "-"); return; }
|
|
237
|
+
double av = fabs(v);
|
|
238
|
+
if (av == 0.0)
|
|
239
|
+
snprintf(buf, bufsz, "0");
|
|
240
|
+
else if (av >= 1e6)
|
|
241
|
+
snprintf(buf, bufsz, "%.3g", v);
|
|
242
|
+
else if (av >= 100.0)
|
|
243
|
+
snprintf(buf, bufsz, "%.1f", v);
|
|
244
|
+
else if (av >= 1.0)
|
|
245
|
+
snprintf(buf, bufsz, "%.2f", v);
|
|
246
|
+
else
|
|
247
|
+
snprintf(buf, bufsz, "%.4f", v);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
static void fmt_int(char *buf, size_t bufsz, long long v) {
|
|
251
|
+
/* Add thousand separators */
|
|
252
|
+
char raw[32];
|
|
253
|
+
snprintf(raw, sizeof(raw), "%lld", v < 0 ? -v : v);
|
|
254
|
+
size_t rlen = strlen(raw);
|
|
255
|
+
size_t pos = 0;
|
|
256
|
+
if (v < 0 && pos < bufsz - 1) buf[pos++] = '-';
|
|
257
|
+
for (size_t i = 0; i < rlen && pos < bufsz - 1; i++) {
|
|
258
|
+
if (i > 0 && (rlen - i) % 3 == 0 && pos < bufsz - 1) buf[pos++] = ',';
|
|
259
|
+
buf[pos++] = raw[i];
|
|
260
|
+
}
|
|
261
|
+
buf[pos] = '\0';
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/* ---- Main report formatter ---- */
|
|
265
|
+
|
|
266
|
+
char *tf_report_format(const char *stats_csv, size_t stats_len, int use_color) {
|
|
267
|
+
if (!stats_csv || stats_len == 0) return NULL;
|
|
268
|
+
|
|
269
|
+
csv_table *t = csv_parse(stats_csv, stats_len);
|
|
270
|
+
if (!t || t->n_rows == 0) { csv_free(t); return NULL; }
|
|
271
|
+
|
|
272
|
+
/* Find column indices */
|
|
273
|
+
int ci_name = csv_col(t, "column");
|
|
274
|
+
int ci_count = csv_col(t, "count");
|
|
275
|
+
int ci_avg = csv_col(t, "avg");
|
|
276
|
+
int ci_min = csv_col(t, "min");
|
|
277
|
+
int ci_max = csv_col(t, "max");
|
|
278
|
+
int ci_stddev = csv_col(t, "stddev");
|
|
279
|
+
int ci_median = csv_col(t, "median");
|
|
280
|
+
int ci_p25 = csv_col(t, "p25");
|
|
281
|
+
int ci_p75 = csv_col(t, "p75");
|
|
282
|
+
int ci_distinct = csv_col(t, "distinct");
|
|
283
|
+
int ci_hist = csv_col(t, "hist");
|
|
284
|
+
int ci_sample = csv_col(t, "sample");
|
|
285
|
+
|
|
286
|
+
if (ci_name < 0) { csv_free(t); return NULL; }
|
|
287
|
+
|
|
288
|
+
const char *bold = use_color ? ANSI_BOLD : "";
|
|
289
|
+
const char *dim = use_color ? ANSI_DIM : "";
|
|
290
|
+
const char *cyan = use_color ? ANSI_CYAN : "";
|
|
291
|
+
const char *green = use_color ? ANSI_GREEN : "";
|
|
292
|
+
const char *reset = use_color ? ANSI_RESET : "";
|
|
293
|
+
|
|
294
|
+
strbuf sb;
|
|
295
|
+
sb_init(&sb);
|
|
296
|
+
if (!sb.data) { csv_free(t); return NULL; }
|
|
297
|
+
|
|
298
|
+
/* Header */
|
|
299
|
+
long long total_count = 0;
|
|
300
|
+
if (ci_count >= 0 && t->n_rows > 0)
|
|
301
|
+
total_count = csv_getll(t, 0, ci_count);
|
|
302
|
+
|
|
303
|
+
char count_str[32];
|
|
304
|
+
fmt_int(count_str, sizeof(count_str), total_count);
|
|
305
|
+
|
|
306
|
+
sb_printf(&sb, "\n");
|
|
307
|
+
if (use_color) {
|
|
308
|
+
sb_printf(&sb, " %s%zu columns%s %s%s rows%s\n\n",
|
|
309
|
+
bold, t->n_rows, reset,
|
|
310
|
+
dim, count_str, reset);
|
|
311
|
+
} else {
|
|
312
|
+
sb_printf(&sb, " %zu columns %s rows\n\n", t->n_rows, count_str);
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/* Per-column stats */
|
|
316
|
+
for (size_t r = 0; r < t->n_rows; r++) {
|
|
317
|
+
const char *name = csv_get(t, r, ci_name);
|
|
318
|
+
if (!name) continue;
|
|
319
|
+
|
|
320
|
+
/* Column name */
|
|
321
|
+
sb_printf(&sb, " %s%-20s%s", bold, name, reset);
|
|
322
|
+
|
|
323
|
+
/* Detect numeric vs string: if min/max are present and not empty */
|
|
324
|
+
const char *min_s = csv_get(t, r, ci_min);
|
|
325
|
+
int is_numeric = (min_s && *min_s != '\0' && strcmp(min_s, "") != 0);
|
|
326
|
+
|
|
327
|
+
if (is_numeric) {
|
|
328
|
+
/* Numeric column: show key stats inline */
|
|
329
|
+
char buf[32];
|
|
330
|
+
|
|
331
|
+
if (ci_min >= 0) {
|
|
332
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_min));
|
|
333
|
+
sb_printf(&sb, " %smin%s %-10s", dim, reset, buf);
|
|
334
|
+
}
|
|
335
|
+
if (ci_max >= 0) {
|
|
336
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_max));
|
|
337
|
+
sb_printf(&sb, " %smax%s %-10s", dim, reset, buf);
|
|
338
|
+
}
|
|
339
|
+
if (ci_avg >= 0) {
|
|
340
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_avg));
|
|
341
|
+
sb_printf(&sb, " %savg%s %-10s", dim, reset, buf);
|
|
342
|
+
}
|
|
343
|
+
if (ci_stddev >= 0) {
|
|
344
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_stddev));
|
|
345
|
+
sb_printf(&sb, " %sstd%s %-10s", dim, reset, buf);
|
|
346
|
+
}
|
|
347
|
+
sb_append(&sb, "\n");
|
|
348
|
+
|
|
349
|
+
/* Quantiles line */
|
|
350
|
+
if (ci_median >= 0 || ci_p25 >= 0 || ci_p75 >= 0) {
|
|
351
|
+
sb_printf(&sb, " %-20s", "");
|
|
352
|
+
if (ci_p25 >= 0) {
|
|
353
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_p25));
|
|
354
|
+
sb_printf(&sb, " %sp25%s %-10s", dim, reset, buf);
|
|
355
|
+
}
|
|
356
|
+
if (ci_median >= 0) {
|
|
357
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_median));
|
|
358
|
+
sb_printf(&sb, " %smed%s %-10s", dim, reset, buf);
|
|
359
|
+
}
|
|
360
|
+
if (ci_p75 >= 0) {
|
|
361
|
+
fmt_num(buf, sizeof(buf), csv_getf(t, r, ci_p75));
|
|
362
|
+
sb_printf(&sb, " %sp75%s %-10s", dim, reset, buf);
|
|
363
|
+
}
|
|
364
|
+
sb_append(&sb, "\n");
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
/* Histogram sparkline */
|
|
368
|
+
if (ci_hist >= 0) {
|
|
369
|
+
const char *hist_str = csv_get(t, r, ci_hist);
|
|
370
|
+
parsed_hist h;
|
|
371
|
+
if (parse_hist(hist_str, &h) && h.total > 0) {
|
|
372
|
+
/* Find max count for scaling */
|
|
373
|
+
size_t max_c = 0;
|
|
374
|
+
for (int i = 0; i < HIST_BINS; i++)
|
|
375
|
+
if (h.counts[i] > max_c) max_c = h.counts[i];
|
|
376
|
+
|
|
377
|
+
sb_printf(&sb, " %-20s %s", "", cyan);
|
|
378
|
+
for (int i = 0; i < HIST_BINS; i++) {
|
|
379
|
+
if (max_c == 0) {
|
|
380
|
+
sb_append(&sb, spark_chars[0]);
|
|
381
|
+
} else {
|
|
382
|
+
int level = (int)((double)h.counts[i] / (double)max_c * 7.0);
|
|
383
|
+
if (level < 0) level = 0;
|
|
384
|
+
if (level > 7) level = 7;
|
|
385
|
+
/* Use at least level 1 for non-zero bins */
|
|
386
|
+
if (h.counts[i] > 0 && level == 0) level = 1;
|
|
387
|
+
sb_append(&sb, spark_chars[level]);
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
char lo_buf[16], hi_buf[16];
|
|
392
|
+
fmt_num(lo_buf, sizeof(lo_buf), h.lo);
|
|
393
|
+
fmt_num(hi_buf, sizeof(hi_buf), h.hi);
|
|
394
|
+
sb_printf(&sb, "%s %s%s — %s%s\n", reset, dim, lo_buf, hi_buf, reset);
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
/* Sample dots */
|
|
399
|
+
if (ci_sample >= 0) {
|
|
400
|
+
const char *samp = csv_get(t, r, ci_sample);
|
|
401
|
+
if (samp && *samp) {
|
|
402
|
+
double vmin = csv_getf(t, r, ci_min);
|
|
403
|
+
double vmax = csv_getf(t, r, ci_max);
|
|
404
|
+
double range = vmax - vmin;
|
|
405
|
+
if (range > 0) {
|
|
406
|
+
sb_printf(&sb, " %-20s %s", "", green);
|
|
407
|
+
/* Parse sample values and place dots in a 32-char strip */
|
|
408
|
+
char strip[33];
|
|
409
|
+
memset(strip, ' ', 32);
|
|
410
|
+
strip[32] = '\0';
|
|
411
|
+
|
|
412
|
+
const char *p = samp;
|
|
413
|
+
while (*p) {
|
|
414
|
+
char *ep;
|
|
415
|
+
double v = strtod(p, &ep);
|
|
416
|
+
if (ep == p) break;
|
|
417
|
+
int pos = (int)((v - vmin) / range * 31.0);
|
|
418
|
+
if (pos < 0) pos = 0;
|
|
419
|
+
if (pos > 31) pos = 31;
|
|
420
|
+
strip[pos] = '.';
|
|
421
|
+
p = ep;
|
|
422
|
+
if (*p == ',') p++;
|
|
423
|
+
}
|
|
424
|
+
/* Print using unicode dots for better visibility */
|
|
425
|
+
for (int i = 0; i < 32; i++) {
|
|
426
|
+
if (strip[i] == '.')
|
|
427
|
+
sb_append(&sb, "\xc2\xb7"); /* · U+00B7 */
|
|
428
|
+
else
|
|
429
|
+
sb_append(&sb, " ");
|
|
430
|
+
}
|
|
431
|
+
sb_printf(&sb, "%s\n", reset);
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
} else {
|
|
436
|
+
/* String/non-numeric column */
|
|
437
|
+
if (ci_count >= 0) {
|
|
438
|
+
char cnt_buf[32];
|
|
439
|
+
fmt_int(cnt_buf, sizeof(cnt_buf), csv_getll(t, r, ci_count));
|
|
440
|
+
sb_printf(&sb, " %sn%s %-10s", dim, reset, cnt_buf);
|
|
441
|
+
}
|
|
442
|
+
if (ci_distinct >= 0) {
|
|
443
|
+
long long dist = csv_getll(t, r, ci_distinct);
|
|
444
|
+
long long cnt = ci_count >= 0 ? csv_getll(t, r, ci_count) : 0;
|
|
445
|
+
char dist_buf[32];
|
|
446
|
+
fmt_int(dist_buf, sizeof(dist_buf), dist);
|
|
447
|
+
if (cnt > 0) {
|
|
448
|
+
double pct = 100.0 * (double)dist / (double)cnt;
|
|
449
|
+
sb_printf(&sb, " %suniq%s %s (%.1f%%)", dim, reset, dist_buf, pct);
|
|
450
|
+
} else {
|
|
451
|
+
sb_printf(&sb, " %suniq%s %s", dim, reset, dist_buf);
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
sb_append(&sb, "\n");
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
sb_append(&sb, "\n");
|
|
459
|
+
|
|
460
|
+
csv_free(t);
|
|
461
|
+
|
|
462
|
+
return sb.data;
|
|
463
|
+
}
|
package/csrc/report.h
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* report.h — Rich ANSI terminal report for stats output.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
#ifndef TF_REPORT_H
|
|
6
|
+
#define TF_REPORT_H
|
|
7
|
+
|
|
8
|
+
#include <stddef.h>
|
|
9
|
+
|
|
10
|
+
/*
|
|
11
|
+
* Format stats CSV into a rich terminal report with ANSI colors and
|
|
12
|
+
* Unicode sparkline histograms.
|
|
13
|
+
*
|
|
14
|
+
* stats_csv: the CSV output from the stats channel (column,count,avg,...)
|
|
15
|
+
* stats_len: byte length
|
|
16
|
+
* use_color: 1 = ANSI escape codes, 0 = plain text
|
|
17
|
+
*
|
|
18
|
+
* Returns a malloc'd string (caller frees), or NULL on error.
|
|
19
|
+
*/
|
|
20
|
+
char *tf_report_format(const char *stats_csv, size_t stats_len, int use_color);
|
|
21
|
+
|
|
22
|
+
#endif
|
package/csrc/tranfi.h
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* tranfi.h — Public C API for the Tranfi streaming ETL core.
|
|
3
|
+
*
|
|
4
|
+
* The host streams bytes in via push(), pulls output bytes from
|
|
5
|
+
* multiple channels (main, errors, stats, samples) via pull().
|
|
6
|
+
* Core does: decode → typed batches → transforms → encode.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
#ifndef TRANFI_H
|
|
10
|
+
#define TRANFI_H
|
|
11
|
+
|
|
12
|
+
#include <stddef.h>
|
|
13
|
+
#include <stdint.h>
|
|
14
|
+
|
|
15
|
+
#ifdef __cplusplus
|
|
16
|
+
extern "C" {
|
|
17
|
+
#endif
|
|
18
|
+
|
|
19
|
+
/* Channel IDs for pull() */
|
|
20
|
+
#define TF_CHAN_MAIN 0
|
|
21
|
+
#define TF_CHAN_ERRORS 1
|
|
22
|
+
#define TF_CHAN_STATS 2
|
|
23
|
+
#define TF_CHAN_SAMPLES 3
|
|
24
|
+
#define TF_NUM_CHANNELS 4
|
|
25
|
+
|
|
26
|
+
/* Return codes */
|
|
27
|
+
#define TF_OK 0
|
|
28
|
+
#define TF_ERROR (-1)
|
|
29
|
+
|
|
30
|
+
/* Opaque pipeline handle */
|
|
31
|
+
typedef struct tf_pipeline tf_pipeline;
|
|
32
|
+
|
|
33
|
+
/*
|
|
34
|
+
* Create a pipeline from a JSON plan.
|
|
35
|
+
* Returns NULL on error (call tf_pipeline_error on NULL is undefined;
|
|
36
|
+
* use tf_last_error() instead).
|
|
37
|
+
*/
|
|
38
|
+
tf_pipeline *tf_pipeline_create(const char *plan_json, size_t len);
|
|
39
|
+
|
|
40
|
+
/* Free all resources associated with a pipeline. */
|
|
41
|
+
void tf_pipeline_free(tf_pipeline *p);
|
|
42
|
+
|
|
43
|
+
/*
|
|
44
|
+
* Push input bytes into the pipeline.
|
|
45
|
+
* Returns TF_OK on success, TF_ERROR on failure.
|
|
46
|
+
*/
|
|
47
|
+
int tf_pipeline_push(tf_pipeline *p, const uint8_t *data, size_t len);
|
|
48
|
+
|
|
49
|
+
/*
|
|
50
|
+
* Signal end of input. Flushes all buffered data through the pipeline.
|
|
51
|
+
* Returns TF_OK on success, TF_ERROR on failure.
|
|
52
|
+
*/
|
|
53
|
+
int tf_pipeline_finish(tf_pipeline *p);
|
|
54
|
+
|
|
55
|
+
/*
|
|
56
|
+
* Pull output bytes from a channel.
|
|
57
|
+
* Writes up to buf_len bytes into buf.
|
|
58
|
+
* Returns the number of bytes written (0 if nothing available).
|
|
59
|
+
*/
|
|
60
|
+
size_t tf_pipeline_pull(tf_pipeline *p, int channel, uint8_t *buf, size_t buf_len);
|
|
61
|
+
|
|
62
|
+
/* Get the last error message, or NULL if no error. */
|
|
63
|
+
const char *tf_pipeline_error(tf_pipeline *p);
|
|
64
|
+
|
|
65
|
+
/* Get the library version string. */
|
|
66
|
+
const char *tf_version(void);
|
|
67
|
+
|
|
68
|
+
/* Get the last global error (for errors before pipeline creation). */
|
|
69
|
+
const char *tf_last_error(void);
|
|
70
|
+
|
|
71
|
+
/* ---- IR plan API (L2 intermediate representation) ---- */
|
|
72
|
+
|
|
73
|
+
/* Opaque IR plan handle */
|
|
74
|
+
typedef struct tf_ir_plan tf_ir_plan;
|
|
75
|
+
|
|
76
|
+
/* Parse a JSON plan string into an IR plan. Returns NULL on error. */
|
|
77
|
+
tf_ir_plan *tf_ir_plan_from_json(const char *json, size_t len, char **error);
|
|
78
|
+
|
|
79
|
+
/* Serialize an IR plan back to JSON. Caller frees the returned string. */
|
|
80
|
+
char *tf_ir_plan_to_json(const tf_ir_plan *plan);
|
|
81
|
+
|
|
82
|
+
/* Validate an IR plan. Returns TF_OK or TF_ERROR. */
|
|
83
|
+
int tf_ir_plan_validate(tf_ir_plan *plan);
|
|
84
|
+
|
|
85
|
+
/* Infer schemas through an IR plan. Best-effort, non-fatal. */
|
|
86
|
+
int tf_ir_plan_infer_schema(tf_ir_plan *plan);
|
|
87
|
+
|
|
88
|
+
/* Free an IR plan. */
|
|
89
|
+
void tf_ir_plan_destroy(tf_ir_plan *plan);
|
|
90
|
+
|
|
91
|
+
/* Create a pipeline from a pre-built IR plan. */
|
|
92
|
+
tf_pipeline *tf_pipeline_create_from_ir(const tf_ir_plan *plan);
|
|
93
|
+
|
|
94
|
+
/* Compile a DSL string to a JSON recipe. Caller frees with tf_string_free(). */
|
|
95
|
+
char *tf_compile_dsl(const char *dsl, size_t len, char **error);
|
|
96
|
+
|
|
97
|
+
/* Compile a DSL string directly to SQL. Caller frees with tf_string_free(). */
|
|
98
|
+
char *tf_compile_to_sql(const char *dsl, size_t len, char **error);
|
|
99
|
+
|
|
100
|
+
/* Convert an IR plan to SQL. Caller frees with tf_string_free(). */
|
|
101
|
+
char *tf_ir_plan_to_sql(const tf_ir_plan *plan, char **error);
|
|
102
|
+
|
|
103
|
+
/* Free a string returned by tf_compile_dsl or tf_ir_plan_to_json. */
|
|
104
|
+
void tf_string_free(char *s);
|
|
105
|
+
|
|
106
|
+
/* ---- Built-in recipes ---- */
|
|
107
|
+
|
|
108
|
+
/* Number of built-in recipes. */
|
|
109
|
+
size_t tf_recipe_count(void);
|
|
110
|
+
|
|
111
|
+
/* Accessors by index (0-based). Return NULL if index out of range. */
|
|
112
|
+
const char *tf_recipe_name(size_t index);
|
|
113
|
+
const char *tf_recipe_dsl(size_t index);
|
|
114
|
+
const char *tf_recipe_description(size_t index);
|
|
115
|
+
|
|
116
|
+
/* Lookup by name (case-insensitive). Returns DSL string or NULL. */
|
|
117
|
+
const char *tf_recipe_find_dsl(const char *name);
|
|
118
|
+
|
|
119
|
+
#ifdef __cplusplus
|
|
120
|
+
}
|
|
121
|
+
#endif
|
|
122
|
+
|
|
123
|
+
#endif /* TRANFI_H */
|