tranfi 0.0.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +395 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-pDFMluyz.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +69 -0
- package/csrc/arena.c +91 -0
- package/csrc/batch.c +229 -0
- package/csrc/buffer.c +78 -0
- package/csrc/cJSON.c +3143 -0
- package/csrc/cJSON.h +300 -0
- package/csrc/codec_csv.c +1058 -0
- package/csrc/codec_jsonl.c +374 -0
- package/csrc/codec_table.c +218 -0
- package/csrc/codec_text.c +229 -0
- package/csrc/compiler.c +102 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +1180 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1245 -0
- package/csrc/expr.h +56 -0
- package/csrc/internal.h +250 -0
- package/csrc/ir.c +119 -0
- package/csrc/ir.h +167 -0
- package/csrc/ir_schema.c +60 -0
- package/csrc/ir_serialize.c +104 -0
- package/csrc/ir_sql.c +1211 -0
- package/csrc/ir_validate.c +120 -0
- package/csrc/main.c +392 -0
- package/csrc/op_acf.c +133 -0
- package/csrc/op_anomaly.c +120 -0
- package/csrc/op_bin.c +109 -0
- package/csrc/op_cast.c +195 -0
- package/csrc/op_clip.c +88 -0
- package/csrc/op_date_trunc.c +181 -0
- package/csrc/op_datetime.c +212 -0
- package/csrc/op_derive.c +248 -0
- package/csrc/op_diff.c +134 -0
- package/csrc/op_ewma.c +103 -0
- package/csrc/op_explode.c +108 -0
- package/csrc/op_fill_down.c +163 -0
- package/csrc/op_fill_null.c +123 -0
- package/csrc/op_filter.c +132 -0
- package/csrc/op_frequency.c +193 -0
- package/csrc/op_grep.c +163 -0
- package/csrc/op_group_agg.c +285 -0
- package/csrc/op_hash.c +126 -0
- package/csrc/op_head.c +149 -0
- package/csrc/op_interpolate.c +239 -0
- package/csrc/op_join.c +384 -0
- package/csrc/op_label_encode.c +144 -0
- package/csrc/op_lead.c +190 -0
- package/csrc/op_normalize.c +226 -0
- package/csrc/op_onehot.c +185 -0
- package/csrc/op_pivot.c +370 -0
- package/csrc/op_registry.c +1148 -0
- package/csrc/op_rename.c +138 -0
- package/csrc/op_replace.c +202 -0
- package/csrc/op_sample.c +101 -0
- package/csrc/op_select.c +140 -0
- package/csrc/op_skip.c +152 -0
- package/csrc/op_sort.c +273 -0
- package/csrc/op_split.c +114 -0
- package/csrc/op_split_data.c +87 -0
- package/csrc/op_stack.c +315 -0
- package/csrc/op_stats.c +779 -0
- package/csrc/op_step.c +171 -0
- package/csrc/op_tail.c +96 -0
- package/csrc/op_top.c +150 -0
- package/csrc/op_trim.c +109 -0
- package/csrc/op_unique.c +300 -0
- package/csrc/op_unpivot.c +159 -0
- package/csrc/op_validate.c +71 -0
- package/csrc/op_window.c +150 -0
- package/csrc/pipeline.c +315 -0
- package/csrc/plan.c +206 -0
- package/csrc/recipes.c +102 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +463 -0
- package/csrc/report.h +22 -0
- package/csrc/tranfi.h +123 -0
- package/csrc/wasm_api.c +157 -0
- package/napi_api.c +326 -0
- package/package.json +46 -57
- package/src/cli.js +193 -0
- package/src/engines/duckdb.js +109 -0
- package/src/index.js +306 -0
- package/src/native.js +22 -0
- package/src/pipeline.js +286 -0
- package/src/server.js +277 -0
- package/src/wasm.js +19 -0
- package/wasm/index.js +244 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/LICENSE +0 -21
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/logo.png +0 -0
- package/src/app.css +0 -160
- package/src/app.js +0 -203
- package/src/app.vue +0 -253
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -18
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/csrc/op_hash.c
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_hash.c — DJB2 hash of columns, adds _hash column.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"columns": ["name", "city"]}
|
|
5
|
+
* or {} for all columns.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "internal.h"
|
|
9
|
+
#include "cJSON.h"
|
|
10
|
+
#include "date_utils.h"
|
|
11
|
+
#include <stdlib.h>
|
|
12
|
+
#include <string.h>
|
|
13
|
+
#include <stdio.h>
|
|
14
|
+
|
|
15
|
+
typedef struct {
|
|
16
|
+
char **cols;
|
|
17
|
+
size_t n_cols;
|
|
18
|
+
} hash_state;
|
|
19
|
+
|
|
20
|
+
static int hash_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
21
|
+
tf_side_channels *side) {
|
|
22
|
+
(void)side;
|
|
23
|
+
hash_state *st = self->state;
|
|
24
|
+
*out = NULL;
|
|
25
|
+
|
|
26
|
+
tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
|
|
27
|
+
if (!ob) return TF_ERROR;
|
|
28
|
+
for (size_t c = 0; c < in->n_cols; c++)
|
|
29
|
+
tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
|
|
30
|
+
tf_batch_set_schema(ob, in->n_cols, "_hash", TF_TYPE_INT64);
|
|
31
|
+
|
|
32
|
+
/* Resolve column indices */
|
|
33
|
+
size_t n_keys;
|
|
34
|
+
int *col_indices;
|
|
35
|
+
if (st->n_cols > 0) {
|
|
36
|
+
n_keys = st->n_cols;
|
|
37
|
+
col_indices = malloc(n_keys * sizeof(int));
|
|
38
|
+
if (!col_indices) { tf_batch_free(ob); return TF_ERROR; }
|
|
39
|
+
for (size_t k = 0; k < n_keys; k++)
|
|
40
|
+
col_indices[k] = tf_batch_col_index(in, st->cols[k]);
|
|
41
|
+
} else {
|
|
42
|
+
n_keys = in->n_cols;
|
|
43
|
+
col_indices = malloc(n_keys * sizeof(int));
|
|
44
|
+
if (!col_indices) { tf_batch_free(ob); return TF_ERROR; }
|
|
45
|
+
for (size_t k = 0; k < n_keys; k++) col_indices[k] = (int)k;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
49
|
+
tf_batch_copy_row(ob, r, in, r);
|
|
50
|
+
/* Compute hash */
|
|
51
|
+
uint32_t h = 5381;
|
|
52
|
+
char val_buf[64];
|
|
53
|
+
for (size_t k = 0; k < n_keys; k++) {
|
|
54
|
+
int ci = col_indices[k];
|
|
55
|
+
if (ci < 0 || tf_batch_is_null(in, r, ci)) continue;
|
|
56
|
+
const char *val;
|
|
57
|
+
switch (in->col_types[ci]) {
|
|
58
|
+
case TF_TYPE_STRING: val = tf_batch_get_string(in, r, ci); break;
|
|
59
|
+
case TF_TYPE_INT64:
|
|
60
|
+
snprintf(val_buf, sizeof(val_buf), "%lld", (long long)tf_batch_get_int64(in, r, ci));
|
|
61
|
+
val = val_buf; break;
|
|
62
|
+
case TF_TYPE_FLOAT64:
|
|
63
|
+
snprintf(val_buf, sizeof(val_buf), "%.17g", tf_batch_get_float64(in, r, ci));
|
|
64
|
+
val = val_buf; break;
|
|
65
|
+
case TF_TYPE_BOOL:
|
|
66
|
+
val = tf_batch_get_bool(in, r, ci) ? "T" : "F"; break;
|
|
67
|
+
case TF_TYPE_DATE:
|
|
68
|
+
snprintf(val_buf, sizeof(val_buf), "%d", (int)tf_batch_get_date(in, r, ci));
|
|
69
|
+
val = val_buf; break;
|
|
70
|
+
case TF_TYPE_TIMESTAMP:
|
|
71
|
+
snprintf(val_buf, sizeof(val_buf), "%lld", (long long)tf_batch_get_timestamp(in, r, ci));
|
|
72
|
+
val = val_buf; break;
|
|
73
|
+
default: val = ""; break;
|
|
74
|
+
}
|
|
75
|
+
for (const unsigned char *p = (const unsigned char *)val; *p; p++)
|
|
76
|
+
h = ((h << 5) + h) ^ *p;
|
|
77
|
+
}
|
|
78
|
+
tf_batch_set_int64(ob, r, in->n_cols, (int64_t)h);
|
|
79
|
+
ob->n_rows = r + 1;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
free(col_indices);
|
|
83
|
+
*out = ob;
|
|
84
|
+
return TF_OK;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
static int hash_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
88
|
+
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
static void hash_destroy(tf_step *self) {
|
|
92
|
+
hash_state *st = self->state;
|
|
93
|
+
if (st) {
|
|
94
|
+
for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i]);
|
|
95
|
+
free(st->cols); free(st);
|
|
96
|
+
}
|
|
97
|
+
free(self);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
tf_step *tf_hash_create(const cJSON *args) {
|
|
101
|
+
hash_state *st = calloc(1, sizeof(hash_state));
|
|
102
|
+
if (!st) return NULL;
|
|
103
|
+
|
|
104
|
+
if (args) {
|
|
105
|
+
cJSON *columns = cJSON_GetObjectItemCaseSensitive(args, "columns");
|
|
106
|
+
if (columns && cJSON_IsArray(columns)) {
|
|
107
|
+
int n = cJSON_GetArraySize(columns);
|
|
108
|
+
if (n > 0) {
|
|
109
|
+
st->cols = calloc(n, sizeof(char *));
|
|
110
|
+
st->n_cols = n;
|
|
111
|
+
for (int i = 0; i < n; i++) {
|
|
112
|
+
cJSON *item = cJSON_GetArrayItem(columns, i);
|
|
113
|
+
if (cJSON_IsString(item)) st->cols[i] = strdup(item->valuestring);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
120
|
+
if (!step) { hash_destroy(&(tf_step){.state = st}); return NULL; }
|
|
121
|
+
step->process = hash_process;
|
|
122
|
+
step->flush = hash_flush;
|
|
123
|
+
step->destroy = hash_destroy;
|
|
124
|
+
step->state = st;
|
|
125
|
+
return step;
|
|
126
|
+
}
|
package/csrc/op_head.c
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_head.c — Take first N rows.
|
|
3
|
+
*
|
|
4
|
+
* Config: {"n": 5}
|
|
5
|
+
* Passes through rows until N have been seen, then emits empty batches.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
#include "internal.h"
|
|
9
|
+
#include "cJSON.h"
|
|
10
|
+
#include <stdlib.h>
|
|
11
|
+
#include <string.h>
|
|
12
|
+
|
|
13
|
+
typedef struct {
|
|
14
|
+
size_t limit;
|
|
15
|
+
size_t seen;
|
|
16
|
+
} head_state;
|
|
17
|
+
|
|
18
|
+
static int head_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
19
|
+
tf_side_channels *side) {
|
|
20
|
+
(void)side;
|
|
21
|
+
head_state *st = self->state;
|
|
22
|
+
*out = NULL;
|
|
23
|
+
|
|
24
|
+
if (st->seen >= st->limit) {
|
|
25
|
+
/* Already past limit, discard */
|
|
26
|
+
return TF_OK;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
size_t remaining = st->limit - st->seen;
|
|
30
|
+
size_t take = in->n_rows < remaining ? in->n_rows : remaining;
|
|
31
|
+
|
|
32
|
+
if (take == in->n_rows) {
|
|
33
|
+
/* Take all rows — create a copy */
|
|
34
|
+
tf_batch *ob = tf_batch_create(in->n_cols, take);
|
|
35
|
+
if (!ob) return TF_ERROR;
|
|
36
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
37
|
+
tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
|
|
38
|
+
}
|
|
39
|
+
for (size_t r = 0; r < take; r++) {
|
|
40
|
+
tf_batch_ensure_capacity(ob, r + 1);
|
|
41
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
42
|
+
if (tf_batch_is_null(in, r, c)) {
|
|
43
|
+
tf_batch_set_null(ob, r, c);
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
switch (in->col_types[c]) {
|
|
47
|
+
case TF_TYPE_BOOL:
|
|
48
|
+
tf_batch_set_bool(ob, r, c, tf_batch_get_bool(in, r, c));
|
|
49
|
+
break;
|
|
50
|
+
case TF_TYPE_INT64:
|
|
51
|
+
tf_batch_set_int64(ob, r, c, tf_batch_get_int64(in, r, c));
|
|
52
|
+
break;
|
|
53
|
+
case TF_TYPE_FLOAT64:
|
|
54
|
+
tf_batch_set_float64(ob, r, c, tf_batch_get_float64(in, r, c));
|
|
55
|
+
break;
|
|
56
|
+
case TF_TYPE_STRING:
|
|
57
|
+
tf_batch_set_string(ob, r, c, tf_batch_get_string(in, r, c));
|
|
58
|
+
break;
|
|
59
|
+
case TF_TYPE_DATE:
|
|
60
|
+
tf_batch_set_date(ob, r, c, tf_batch_get_date(in, r, c));
|
|
61
|
+
break;
|
|
62
|
+
case TF_TYPE_TIMESTAMP:
|
|
63
|
+
tf_batch_set_timestamp(ob, r, c, tf_batch_get_timestamp(in, r, c));
|
|
64
|
+
break;
|
|
65
|
+
default:
|
|
66
|
+
tf_batch_set_null(ob, r, c);
|
|
67
|
+
break;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
ob->n_rows = r + 1;
|
|
71
|
+
}
|
|
72
|
+
st->seen += take;
|
|
73
|
+
*out = ob;
|
|
74
|
+
} else {
|
|
75
|
+
/* Partial take — only copy first `take` rows */
|
|
76
|
+
tf_batch *ob = tf_batch_create(in->n_cols, take);
|
|
77
|
+
if (!ob) return TF_ERROR;
|
|
78
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
79
|
+
tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
|
|
80
|
+
}
|
|
81
|
+
for (size_t r = 0; r < take; r++) {
|
|
82
|
+
tf_batch_ensure_capacity(ob, r + 1);
|
|
83
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
84
|
+
if (tf_batch_is_null(in, r, c)) {
|
|
85
|
+
tf_batch_set_null(ob, r, c);
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
switch (in->col_types[c]) {
|
|
89
|
+
case TF_TYPE_BOOL:
|
|
90
|
+
tf_batch_set_bool(ob, r, c, tf_batch_get_bool(in, r, c));
|
|
91
|
+
break;
|
|
92
|
+
case TF_TYPE_INT64:
|
|
93
|
+
tf_batch_set_int64(ob, r, c, tf_batch_get_int64(in, r, c));
|
|
94
|
+
break;
|
|
95
|
+
case TF_TYPE_FLOAT64:
|
|
96
|
+
tf_batch_set_float64(ob, r, c, tf_batch_get_float64(in, r, c));
|
|
97
|
+
break;
|
|
98
|
+
case TF_TYPE_STRING:
|
|
99
|
+
tf_batch_set_string(ob, r, c, tf_batch_get_string(in, r, c));
|
|
100
|
+
break;
|
|
101
|
+
case TF_TYPE_DATE:
|
|
102
|
+
tf_batch_set_date(ob, r, c, tf_batch_get_date(in, r, c));
|
|
103
|
+
break;
|
|
104
|
+
case TF_TYPE_TIMESTAMP:
|
|
105
|
+
tf_batch_set_timestamp(ob, r, c, tf_batch_get_timestamp(in, r, c));
|
|
106
|
+
break;
|
|
107
|
+
default:
|
|
108
|
+
tf_batch_set_null(ob, r, c);
|
|
109
|
+
break;
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
ob->n_rows = r + 1;
|
|
113
|
+
}
|
|
114
|
+
st->seen += take;
|
|
115
|
+
*out = ob;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
return TF_OK;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
static int head_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
122
|
+
(void)self; (void)side;
|
|
123
|
+
*out = NULL;
|
|
124
|
+
return TF_OK;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
static void head_destroy(tf_step *self) {
|
|
128
|
+
free(self->state);
|
|
129
|
+
free(self);
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
tf_step *tf_head_create(const cJSON *args) {
|
|
133
|
+
if (!args) return NULL;
|
|
134
|
+
cJSON *n_json = cJSON_GetObjectItemCaseSensitive(args, "n");
|
|
135
|
+
if (!cJSON_IsNumber(n_json) || n_json->valueint <= 0) return NULL;
|
|
136
|
+
|
|
137
|
+
head_state *st = calloc(1, sizeof(head_state));
|
|
138
|
+
if (!st) return NULL;
|
|
139
|
+
st->limit = (size_t)n_json->valueint;
|
|
140
|
+
st->seen = 0;
|
|
141
|
+
|
|
142
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
143
|
+
if (!step) { free(st); return NULL; }
|
|
144
|
+
step->process = head_process;
|
|
145
|
+
step->flush = head_flush;
|
|
146
|
+
step->destroy = head_destroy;
|
|
147
|
+
step->state = st;
|
|
148
|
+
return step;
|
|
149
|
+
}
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_interpolate.c — Fill null values via interpolation.
|
|
3
|
+
* Methods: forward, backward, linear.
|
|
4
|
+
*
|
|
5
|
+
* For backward/linear: buffers rows with null target values and emits
|
|
6
|
+
* them when the next non-null value arrives.
|
|
7
|
+
*
|
|
8
|
+
* Config: {"column": "price", "method": "linear"}
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
#include "internal.h"
|
|
12
|
+
#include "cJSON.h"
|
|
13
|
+
#include <stdlib.h>
|
|
14
|
+
#include <string.h>
|
|
15
|
+
#include <stdio.h>
|
|
16
|
+
|
|
17
|
+
typedef enum {
|
|
18
|
+
INTERP_FORWARD,
|
|
19
|
+
INTERP_BACKWARD,
|
|
20
|
+
INTERP_LINEAR
|
|
21
|
+
} interp_method;
|
|
22
|
+
|
|
23
|
+
/* Buffered row: we store complete batches and track which rows are pending */
|
|
24
|
+
typedef struct pending_row {
|
|
25
|
+
tf_batch *batch; /* single-row batch (copy of original row) */
|
|
26
|
+
size_t target_col; /* column index in the batch */
|
|
27
|
+
} pending_row;
|
|
28
|
+
|
|
29
|
+
typedef struct {
|
|
30
|
+
char *column;
|
|
31
|
+
interp_method method;
|
|
32
|
+
double last_val;
|
|
33
|
+
int has_last;
|
|
34
|
+
/* Pending null rows (for backward/linear) */
|
|
35
|
+
pending_row *pending;
|
|
36
|
+
size_t n_pending;
|
|
37
|
+
size_t cap_pending;
|
|
38
|
+
} interpolate_state;
|
|
39
|
+
|
|
40
|
+
static interp_method parse_method(const char *s) {
|
|
41
|
+
if (!s) return INTERP_LINEAR;
|
|
42
|
+
if (strcmp(s, "forward") == 0) return INTERP_FORWARD;
|
|
43
|
+
if (strcmp(s, "backward") == 0) return INTERP_BACKWARD;
|
|
44
|
+
return INTERP_LINEAR;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
static double get_numeric(const tf_batch *b, size_t r, int ci) {
|
|
48
|
+
if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
|
|
49
|
+
if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
|
|
50
|
+
return 0;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
static void add_pending(interpolate_state *st, tf_batch *row_batch, size_t target_col) {
|
|
54
|
+
if (st->n_pending >= st->cap_pending) {
|
|
55
|
+
size_t newcap = st->cap_pending ? st->cap_pending * 2 : 16;
|
|
56
|
+
pending_row *tmp = realloc(st->pending, newcap * sizeof(pending_row));
|
|
57
|
+
if (!tmp) return;
|
|
58
|
+
st->pending = tmp;
|
|
59
|
+
st->cap_pending = newcap;
|
|
60
|
+
}
|
|
61
|
+
st->pending[st->n_pending].batch = row_batch;
|
|
62
|
+
st->pending[st->n_pending].target_col = target_col;
|
|
63
|
+
st->n_pending++;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/* Emit all pending rows into output batch, interpolating values */
|
|
67
|
+
static void flush_pending(interpolate_state *st, tf_batch *ob, size_t *out_row,
|
|
68
|
+
double end_val, size_t target_col) {
|
|
69
|
+
if (st->n_pending == 0) return;
|
|
70
|
+
|
|
71
|
+
for (size_t i = 0; i < st->n_pending; i++) {
|
|
72
|
+
tf_batch *pb = st->pending[i].batch;
|
|
73
|
+
size_t r = *out_row;
|
|
74
|
+
tf_batch_copy_row(ob, r, pb, 0);
|
|
75
|
+
|
|
76
|
+
double interp_val;
|
|
77
|
+
if (st->method == INTERP_BACKWARD) {
|
|
78
|
+
interp_val = end_val;
|
|
79
|
+
} else {
|
|
80
|
+
/* Linear: interpolate between last_val and end_val */
|
|
81
|
+
if (st->has_last) {
|
|
82
|
+
double t = (double)(i + 1) / (double)(st->n_pending + 1);
|
|
83
|
+
interp_val = st->last_val + t * (end_val - st->last_val);
|
|
84
|
+
} else {
|
|
85
|
+
interp_val = end_val;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
tf_batch_set_float64(ob, r, target_col, interp_val);
|
|
90
|
+
ob->n_rows = r + 1;
|
|
91
|
+
(*out_row)++;
|
|
92
|
+
|
|
93
|
+
tf_batch_free(pb);
|
|
94
|
+
}
|
|
95
|
+
st->n_pending = 0;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
static int interpolate_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
99
|
+
tf_side_channels *side) {
|
|
100
|
+
(void)side;
|
|
101
|
+
interpolate_state *st = self->state;
|
|
102
|
+
*out = NULL;
|
|
103
|
+
|
|
104
|
+
int ci = tf_batch_col_index(in, st->column);
|
|
105
|
+
|
|
106
|
+
/* Count total rows we might output (pending + current batch) */
|
|
107
|
+
size_t max_rows = st->n_pending + in->n_rows;
|
|
108
|
+
tf_batch *ob = tf_batch_create(in->n_cols, max_rows);
|
|
109
|
+
if (!ob) return TF_ERROR;
|
|
110
|
+
for (size_t c = 0; c < in->n_cols; c++)
|
|
111
|
+
tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
|
|
112
|
+
|
|
113
|
+
size_t out_row = 0;
|
|
114
|
+
|
|
115
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
116
|
+
if (ci < 0) {
|
|
117
|
+
/* No target column — just pass through */
|
|
118
|
+
tf_batch_copy_row(ob, out_row, in, r);
|
|
119
|
+
ob->n_rows = out_row + 1;
|
|
120
|
+
out_row++;
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
int is_null = tf_batch_is_null(in, r, ci);
|
|
125
|
+
|
|
126
|
+
if (is_null) {
|
|
127
|
+
if (st->method == INTERP_FORWARD && st->has_last) {
|
|
128
|
+
/* Forward fill: use last known value */
|
|
129
|
+
tf_batch_copy_row(ob, out_row, in, r);
|
|
130
|
+
tf_batch_set_float64(ob, out_row, ci, st->last_val);
|
|
131
|
+
ob->n_rows = out_row + 1;
|
|
132
|
+
out_row++;
|
|
133
|
+
} else if (st->method == INTERP_FORWARD) {
|
|
134
|
+
/* No previous value yet — pass null through */
|
|
135
|
+
tf_batch_copy_row(ob, out_row, in, r);
|
|
136
|
+
ob->n_rows = out_row + 1;
|
|
137
|
+
out_row++;
|
|
138
|
+
} else {
|
|
139
|
+
/* Backward/linear: buffer this row */
|
|
140
|
+
tf_batch *row_copy = tf_batch_create(in->n_cols, 1);
|
|
141
|
+
if (row_copy) {
|
|
142
|
+
for (size_t c = 0; c < in->n_cols; c++)
|
|
143
|
+
tf_batch_set_schema(row_copy, c, in->col_names[c], in->col_types[c]);
|
|
144
|
+
tf_batch_copy_row(row_copy, 0, in, r);
|
|
145
|
+
row_copy->n_rows = 1;
|
|
146
|
+
add_pending(st, row_copy, ci);
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
} else {
|
|
150
|
+
double val = get_numeric(in, r, ci);
|
|
151
|
+
|
|
152
|
+
/* Flush any pending rows */
|
|
153
|
+
if (st->n_pending > 0) {
|
|
154
|
+
/* Grow output if needed */
|
|
155
|
+
tf_batch_ensure_capacity(ob, out_row + st->n_pending + 1);
|
|
156
|
+
flush_pending(st, ob, &out_row, val, ci);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/* Output current row */
|
|
160
|
+
tf_batch_copy_row(ob, out_row, in, r);
|
|
161
|
+
ob->n_rows = out_row + 1;
|
|
162
|
+
out_row++;
|
|
163
|
+
|
|
164
|
+
st->last_val = val;
|
|
165
|
+
st->has_last = 1;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
if (ob->n_rows == 0) {
|
|
170
|
+
tf_batch_free(ob);
|
|
171
|
+
*out = NULL;
|
|
172
|
+
} else {
|
|
173
|
+
*out = ob;
|
|
174
|
+
}
|
|
175
|
+
return TF_OK;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
static int interpolate_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
179
|
+
(void)side;
|
|
180
|
+
interpolate_state *st = self->state;
|
|
181
|
+
*out = NULL;
|
|
182
|
+
|
|
183
|
+
/* Emit any remaining pending rows as nulls (or forward-fill with last known) */
|
|
184
|
+
if (st->n_pending > 0) {
|
|
185
|
+
tf_batch *first = st->pending[0].batch;
|
|
186
|
+
tf_batch *ob = tf_batch_create(first->n_cols, st->n_pending);
|
|
187
|
+
if (!ob) return TF_OK;
|
|
188
|
+
for (size_t c = 0; c < first->n_cols; c++)
|
|
189
|
+
tf_batch_set_schema(ob, c, first->col_names[c], first->col_types[c]);
|
|
190
|
+
|
|
191
|
+
for (size_t i = 0; i < st->n_pending; i++) {
|
|
192
|
+
tf_batch *pb = st->pending[i].batch;
|
|
193
|
+
tf_batch_copy_row(ob, i, pb, 0);
|
|
194
|
+
/* For linear/backward at end of stream: use last known if available */
|
|
195
|
+
if (st->has_last) {
|
|
196
|
+
tf_batch_set_float64(ob, i, st->pending[i].target_col, st->last_val);
|
|
197
|
+
}
|
|
198
|
+
ob->n_rows = i + 1;
|
|
199
|
+
tf_batch_free(pb);
|
|
200
|
+
}
|
|
201
|
+
st->n_pending = 0;
|
|
202
|
+
*out = ob;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
return TF_OK;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
static void interpolate_destroy(tf_step *self) {
|
|
209
|
+
interpolate_state *st = self->state;
|
|
210
|
+
if (st) {
|
|
211
|
+
for (size_t i = 0; i < st->n_pending; i++)
|
|
212
|
+
tf_batch_free(st->pending[i].batch);
|
|
213
|
+
free(st->pending);
|
|
214
|
+
free(st->column);
|
|
215
|
+
free(st);
|
|
216
|
+
}
|
|
217
|
+
free(self);
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
tf_step *tf_interpolate_create(const cJSON *args) {
|
|
221
|
+
if (!args) return NULL;
|
|
222
|
+
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
223
|
+
if (!cJSON_IsString(col_j)) return NULL;
|
|
224
|
+
|
|
225
|
+
interpolate_state *st = calloc(1, sizeof(interpolate_state));
|
|
226
|
+
if (!st) return NULL;
|
|
227
|
+
st->column = strdup(col_j->valuestring);
|
|
228
|
+
|
|
229
|
+
cJSON *method_j = cJSON_GetObjectItemCaseSensitive(args, "method");
|
|
230
|
+
st->method = parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL);
|
|
231
|
+
|
|
232
|
+
tf_step *step = malloc(sizeof(tf_step));
|
|
233
|
+
if (!step) { free(st->column); free(st); return NULL; }
|
|
234
|
+
step->process = interpolate_process;
|
|
235
|
+
step->flush = interpolate_flush;
|
|
236
|
+
step->destroy = interpolate_destroy;
|
|
237
|
+
step->state = st;
|
|
238
|
+
return step;
|
|
239
|
+
}
|