tranfi 0.0.2 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/README.md +395 -0
  2. package/app/assets/index-6quYZ5Ap.css +5 -0
  3. package/app/assets/index-pDFMluyz.js +160 -0
  4. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  5. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  6. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  7. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  8. package/app/index.html +13 -0
  9. package/binding.gyp +69 -0
  10. package/csrc/arena.c +91 -0
  11. package/csrc/batch.c +229 -0
  12. package/csrc/buffer.c +78 -0
  13. package/csrc/cJSON.c +3143 -0
  14. package/csrc/cJSON.h +300 -0
  15. package/csrc/codec_csv.c +1058 -0
  16. package/csrc/codec_jsonl.c +374 -0
  17. package/csrc/codec_table.c +218 -0
  18. package/csrc/codec_text.c +229 -0
  19. package/csrc/compiler.c +102 -0
  20. package/csrc/date_utils.h +94 -0
  21. package/csrc/dsl.c +1180 -0
  22. package/csrc/dsl.h +22 -0
  23. package/csrc/expr.c +1245 -0
  24. package/csrc/expr.h +56 -0
  25. package/csrc/internal.h +250 -0
  26. package/csrc/ir.c +119 -0
  27. package/csrc/ir.h +167 -0
  28. package/csrc/ir_schema.c +60 -0
  29. package/csrc/ir_serialize.c +104 -0
  30. package/csrc/ir_sql.c +1211 -0
  31. package/csrc/ir_validate.c +120 -0
  32. package/csrc/main.c +392 -0
  33. package/csrc/op_acf.c +133 -0
  34. package/csrc/op_anomaly.c +120 -0
  35. package/csrc/op_bin.c +109 -0
  36. package/csrc/op_cast.c +195 -0
  37. package/csrc/op_clip.c +88 -0
  38. package/csrc/op_date_trunc.c +181 -0
  39. package/csrc/op_datetime.c +212 -0
  40. package/csrc/op_derive.c +248 -0
  41. package/csrc/op_diff.c +134 -0
  42. package/csrc/op_ewma.c +103 -0
  43. package/csrc/op_explode.c +108 -0
  44. package/csrc/op_fill_down.c +163 -0
  45. package/csrc/op_fill_null.c +123 -0
  46. package/csrc/op_filter.c +132 -0
  47. package/csrc/op_frequency.c +193 -0
  48. package/csrc/op_grep.c +163 -0
  49. package/csrc/op_group_agg.c +285 -0
  50. package/csrc/op_hash.c +126 -0
  51. package/csrc/op_head.c +149 -0
  52. package/csrc/op_interpolate.c +239 -0
  53. package/csrc/op_join.c +384 -0
  54. package/csrc/op_label_encode.c +144 -0
  55. package/csrc/op_lead.c +190 -0
  56. package/csrc/op_normalize.c +226 -0
  57. package/csrc/op_onehot.c +185 -0
  58. package/csrc/op_pivot.c +370 -0
  59. package/csrc/op_registry.c +1148 -0
  60. package/csrc/op_rename.c +138 -0
  61. package/csrc/op_replace.c +202 -0
  62. package/csrc/op_sample.c +101 -0
  63. package/csrc/op_select.c +140 -0
  64. package/csrc/op_skip.c +152 -0
  65. package/csrc/op_sort.c +273 -0
  66. package/csrc/op_split.c +114 -0
  67. package/csrc/op_split_data.c +87 -0
  68. package/csrc/op_stack.c +315 -0
  69. package/csrc/op_stats.c +779 -0
  70. package/csrc/op_step.c +171 -0
  71. package/csrc/op_tail.c +96 -0
  72. package/csrc/op_top.c +150 -0
  73. package/csrc/op_trim.c +109 -0
  74. package/csrc/op_unique.c +300 -0
  75. package/csrc/op_unpivot.c +159 -0
  76. package/csrc/op_validate.c +71 -0
  77. package/csrc/op_window.c +150 -0
  78. package/csrc/pipeline.c +315 -0
  79. package/csrc/plan.c +206 -0
  80. package/csrc/recipes.c +102 -0
  81. package/csrc/recipes.h +27 -0
  82. package/csrc/report.c +463 -0
  83. package/csrc/report.h +22 -0
  84. package/csrc/tranfi.h +123 -0
  85. package/csrc/wasm_api.c +157 -0
  86. package/napi_api.c +326 -0
  87. package/package.json +46 -57
  88. package/src/cli.js +193 -0
  89. package/src/engines/duckdb.js +109 -0
  90. package/src/index.js +306 -0
  91. package/src/native.js +22 -0
  92. package/src/pipeline.js +286 -0
  93. package/src/server.js +277 -0
  94. package/src/wasm.js +19 -0
  95. package/wasm/index.js +244 -0
  96. package/wasm/package.json +1 -0
  97. package/wasm/tranfi_core.js +0 -0
  98. package/LICENSE +0 -21
  99. package/dist/bundle.js +0 -1
  100. package/index.html +0 -18
  101. package/src/app.css +0 -169
  102. package/src/app.js +0 -203
  103. package/src/app.vue +0 -250
  104. package/src/bulma-input.vue +0 -110
  105. package/src/common-inputs.js +0 -28
  106. package/src/main.js +0 -20
  107. package/src/transforms.js +0 -166
  108. package/webpack.config.js +0 -108
@@ -0,0 +1,315 @@
1
+ /*
2
+ * op_stack.c — Vertically concatenate a second CSV file into the stream.
3
+ *
4
+ * Passes through all input batches, then on flush reads and appends
5
+ * rows from a second CSV file. Optionally adds a tag column to
6
+ * identify the source.
7
+ *
8
+ * Args:
9
+ * file (string, required) — path to CSV file to append
10
+ * tag (string, optional) — name of source-identifying column
11
+ * tag_value (string, optional) — value for tag column on appended rows
12
+ */
13
+
14
+ #include "internal.h"
15
+ #include "cJSON.h"
16
+ #include <stdlib.h>
17
+ #include <string.h>
18
+ #include <stdio.h>
19
+
20
+ typedef struct {
21
+ char *file_path;
22
+ char *tag_col; /* NULL if no tag */
23
+ char *tag_value; /* value for appended rows */
24
+ char *tag_value_in; /* value for passthrough rows (filename or "input") */
25
+
26
+ /* Schema from first input batch */
27
+ char **col_names;
28
+ tf_type *col_types;
29
+ size_t n_cols;
30
+ int schema_captured;
31
+ int has_tag; /* whether tag column was added */
32
+ } stack_state;
33
+
34
+ static void stack_destroy(tf_step *self) {
35
+ stack_state *st = self->state;
36
+ if (st) {
37
+ free(st->file_path);
38
+ free(st->tag_col);
39
+ free(st->tag_value);
40
+ free(st->tag_value_in);
41
+ if (st->col_names) {
42
+ for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
43
+ free(st->col_names);
44
+ }
45
+ free(st->col_types);
46
+ free(st);
47
+ }
48
+ free(self);
49
+ }
50
+
51
+ /*
52
+ * Add tag column to a batch. Returns a new batch with the tag column prepended.
53
+ */
54
+ static tf_batch *add_tag_column(tf_batch *in, const char *tag_col, const char *tag_value) {
55
+ tf_batch *out = tf_batch_create(in->n_cols + 1, in->n_rows);
56
+ if (!out) return NULL;
57
+
58
+ /* First column is the tag */
59
+ tf_batch_set_schema(out, 0, tag_col, TF_TYPE_STRING);
60
+ /* Copy remaining columns */
61
+ for (size_t c = 0; c < in->n_cols; c++) {
62
+ tf_batch_set_schema(out, c + 1, in->col_names[c], in->col_types[c]);
63
+ }
64
+
65
+ for (size_t r = 0; r < in->n_rows; r++) {
66
+ tf_batch_ensure_capacity(out, r + 1);
67
+ tf_batch_set_string(out, r, 0, tag_value);
68
+ for (size_t c = 0; c < in->n_cols; c++) {
69
+ if (tf_batch_is_null(in, r, c)) {
70
+ tf_batch_set_null(out, r, c + 1);
71
+ } else {
72
+ switch (in->col_types[c]) {
73
+ case TF_TYPE_INT64:
74
+ tf_batch_set_int64(out, r, c + 1, tf_batch_get_int64(in, r, c));
75
+ break;
76
+ case TF_TYPE_FLOAT64:
77
+ tf_batch_set_float64(out, r, c + 1, tf_batch_get_float64(in, r, c));
78
+ break;
79
+ case TF_TYPE_STRING:
80
+ tf_batch_set_string(out, r, c + 1, tf_batch_get_string(in, r, c));
81
+ break;
82
+ case TF_TYPE_BOOL:
83
+ tf_batch_set_bool(out, r, c + 1, tf_batch_get_bool(in, r, c));
84
+ break;
85
+ case TF_TYPE_DATE:
86
+ tf_batch_set_date(out, r, c + 1, tf_batch_get_date(in, r, c));
87
+ break;
88
+ case TF_TYPE_TIMESTAMP:
89
+ tf_batch_set_timestamp(out, r, c + 1, tf_batch_get_timestamp(in, r, c));
90
+ break;
91
+ default:
92
+ tf_batch_set_null(out, r, c + 1);
93
+ break;
94
+ }
95
+ }
96
+ }
97
+ out->n_rows = r + 1;
98
+ }
99
+ return out;
100
+ }
101
+
102
+ static int stack_process(tf_step *self, tf_batch *in, tf_batch **out,
103
+ tf_side_channels *side) {
104
+ stack_state *st = self->state;
105
+ (void)side;
106
+
107
+ /* Capture schema from first batch */
108
+ if (!st->schema_captured && in->n_cols > 0) {
109
+ st->n_cols = in->n_cols;
110
+ st->col_names = malloc(in->n_cols * sizeof(char *));
111
+ st->col_types = malloc(in->n_cols * sizeof(tf_type));
112
+ for (size_t i = 0; i < in->n_cols; i++) {
113
+ st->col_names[i] = strdup(in->col_names[i]);
114
+ st->col_types[i] = in->col_types[i];
115
+ }
116
+ st->schema_captured = 1;
117
+ }
118
+
119
+ if (st->tag_col) {
120
+ *out = add_tag_column(in, st->tag_col, st->tag_value_in);
121
+ } else {
122
+ /* Clone input batch */
123
+ tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
124
+ if (!ob) { *out = NULL; return TF_ERROR; }
125
+ for (size_t c = 0; c < in->n_cols; c++)
126
+ tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
127
+ for (size_t r = 0; r < in->n_rows; r++) {
128
+ tf_batch_ensure_capacity(ob, r + 1);
129
+ tf_batch_copy_row(ob, r, in, r);
130
+ ob->n_rows = r + 1;
131
+ }
132
+ *out = ob;
133
+ }
134
+ return TF_OK;
135
+ }
136
+
137
+ static int stack_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
138
+ stack_state *st = self->state;
139
+ (void)side;
140
+ *out = NULL;
141
+
142
+ /* Read the file */
143
+ FILE *f = fopen(st->file_path, "rb");
144
+ if (!f) return TF_OK; /* silently skip if file not found */
145
+
146
+ fseek(f, 0, SEEK_END);
147
+ long fsize = ftell(f);
148
+ fseek(f, 0, SEEK_SET);
149
+ if (fsize <= 0) { fclose(f); return TF_OK; }
150
+
151
+ char *data = malloc((size_t)fsize + 1);
152
+ if (!data) { fclose(f); return TF_ERROR; }
153
+ size_t nread = fread(data, 1, (size_t)fsize, f);
154
+ fclose(f);
155
+ data[nread] = '\0';
156
+
157
+ /* Parse CSV: use a sub-pipeline to decode the file */
158
+ /* Simple approach: parse line by line */
159
+ char **file_col_names = NULL;
160
+ size_t file_n_cols = 0;
161
+
162
+ /* Count rows and parse */
163
+ size_t n_rows = 0;
164
+ char *line = data;
165
+ char *end = data + nread;
166
+
167
+ /* First pass: count data lines */
168
+ char *p = data;
169
+ while (p < end) {
170
+ char *nl = memchr(p, '\n', (size_t)(end - p));
171
+ if (!nl) nl = end;
172
+ if (nl > p || p < end) n_rows++;
173
+ p = nl + 1;
174
+ }
175
+ if (n_rows == 0) { free(data); return TF_OK; }
176
+ n_rows--; /* subtract header */
177
+
178
+ /* Parse header */
179
+ line = data;
180
+ char *nl = memchr(line, '\n', (size_t)(end - line));
181
+ if (!nl) nl = end;
182
+ size_t hdr_len = (size_t)(nl - line);
183
+ if (hdr_len > 0 && line[hdr_len - 1] == '\r') hdr_len--;
184
+
185
+ /* Count commas to get column count */
186
+ file_n_cols = 1;
187
+ for (size_t i = 0; i < hdr_len; i++) {
188
+ if (line[i] == ',') file_n_cols++;
189
+ }
190
+
191
+ /* Parse column names */
192
+ file_col_names = malloc(file_n_cols * sizeof(char *));
193
+ size_t ci = 0;
194
+ size_t field_start = 0;
195
+ for (size_t i = 0; i <= hdr_len; i++) {
196
+ if (i == hdr_len || line[i] == ',') {
197
+ size_t flen = i - field_start;
198
+ /* Trim whitespace */
199
+ while (flen > 0 && (line[field_start] == ' ' || line[field_start] == '\t'))
200
+ { field_start++; flen--; }
201
+ while (flen > 0 && (line[field_start + flen - 1] == ' ' || line[field_start + flen - 1] == '\t'))
202
+ flen--;
203
+ file_col_names[ci] = malloc(flen + 1);
204
+ memcpy(file_col_names[ci], line + field_start, flen);
205
+ file_col_names[ci][flen] = '\0';
206
+ ci++;
207
+ field_start = i + 1;
208
+ }
209
+ }
210
+
211
+ /* Create output batch — use schema from input or file */
212
+ size_t out_cols = st->tag_col ? file_n_cols + 1 : file_n_cols;
213
+ tf_batch *ob = tf_batch_create(out_cols, n_rows > 0 ? n_rows : 1);
214
+ if (!ob) {
215
+ for (size_t i = 0; i < file_n_cols; i++) free(file_col_names[i]);
216
+ free(file_col_names);
217
+ free(data);
218
+ return TF_ERROR;
219
+ }
220
+
221
+ size_t col_offset = 0;
222
+ if (st->tag_col) {
223
+ tf_batch_set_schema(ob, 0, st->tag_col, TF_TYPE_STRING);
224
+ col_offset = 1;
225
+ }
226
+ for (size_t i = 0; i < file_n_cols; i++) {
227
+ tf_batch_set_schema(ob, i + col_offset, file_col_names[i], TF_TYPE_STRING);
228
+ }
229
+
230
+ /* Parse data rows */
231
+ p = nl + 1; /* skip past header newline */
232
+ size_t row = 0;
233
+ while (p < end && row < n_rows) {
234
+ nl = memchr(p, '\n', (size_t)(end - p));
235
+ if (!nl) nl = end;
236
+ size_t line_len = (size_t)(nl - p);
237
+ if (line_len > 0 && p[line_len - 1] == '\r') line_len--;
238
+ if (line_len == 0) { p = nl + 1; continue; }
239
+
240
+ tf_batch_ensure_capacity(ob, row + 1);
241
+
242
+ if (st->tag_col) {
243
+ tf_batch_set_string(ob, row, 0, st->tag_value);
244
+ }
245
+
246
+ /* Parse fields */
247
+ size_t fc = 0;
248
+ size_t fs = 0;
249
+ for (size_t i = 0; i <= line_len; i++) {
250
+ if (i == line_len || p[i] == ',') {
251
+ if (fc < file_n_cols) {
252
+ size_t flen = i - fs;
253
+ /* Trim */
254
+ const char *fp = p + fs;
255
+ while (flen > 0 && (*fp == ' ' || *fp == '\t')) { fp++; flen--; }
256
+ while (flen > 0 && (fp[flen - 1] == ' ' || fp[flen - 1] == '\t')) flen--;
257
+ if (flen == 0) {
258
+ tf_batch_set_null(ob, row, fc + col_offset);
259
+ } else {
260
+ char tmp[4096];
261
+ size_t clen = flen < sizeof(tmp) - 1 ? flen : sizeof(tmp) - 1;
262
+ memcpy(tmp, fp, clen);
263
+ tmp[clen] = '\0';
264
+ tf_batch_set_string(ob, row, fc + col_offset, tmp);
265
+ }
266
+ fc++;
267
+ }
268
+ fs = i + 1;
269
+ }
270
+ }
271
+ /* Null-fill remaining columns */
272
+ for (size_t c = fc; c < file_n_cols; c++) {
273
+ tf_batch_set_null(ob, row, c + col_offset);
274
+ }
275
+
276
+ ob->n_rows = row + 1;
277
+ row++;
278
+ p = nl + 1;
279
+ }
280
+
281
+ for (size_t i = 0; i < file_n_cols; i++) free(file_col_names[i]);
282
+ free(file_col_names);
283
+ free(data);
284
+
285
+ *out = ob;
286
+ return TF_OK;
287
+ }
288
+
289
+ tf_step *tf_stack_create(const cJSON *args) {
290
+ if (!args) return NULL;
291
+
292
+ cJSON *file = cJSON_GetObjectItemCaseSensitive(args, "file");
293
+ if (!cJSON_IsString(file) || !file->valuestring[0]) return NULL;
294
+
295
+ stack_state *st = calloc(1, sizeof(stack_state));
296
+ if (!st) return NULL;
297
+
298
+ st->file_path = strdup(file->valuestring);
299
+
300
+ cJSON *tag = cJSON_GetObjectItemCaseSensitive(args, "tag");
301
+ if (cJSON_IsString(tag) && tag->valuestring[0]) {
302
+ st->tag_col = strdup(tag->valuestring);
303
+ cJSON *tv = cJSON_GetObjectItemCaseSensitive(args, "tag_value");
304
+ st->tag_value = strdup(cJSON_IsString(tv) ? tv->valuestring : st->file_path);
305
+ st->tag_value_in = strdup("input");
306
+ }
307
+
308
+ tf_step *step = malloc(sizeof(tf_step));
309
+ if (!step) { free(st->file_path); free(st->tag_col); free(st->tag_value); free(st->tag_value_in); free(st); return NULL; }
310
+ step->process = stack_process;
311
+ step->flush = stack_flush;
312
+ step->destroy = stack_destroy;
313
+ step->state = st;
314
+ return step;
315
+ }