tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/csrc/op_top.c
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
/*
|
|
2
|
-
* op_top.c —
|
|
2
|
+
* op_top.c — Heap-backed bounded top/bottom N rows by column.
|
|
3
3
|
*
|
|
4
4
|
* Config: {"n": 10, "column": "score", "desc": true}
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
#include "internal.h"
|
|
8
8
|
#include "cJSON.h"
|
|
9
|
+
#include <stdio.h>
|
|
9
10
|
#include <stdlib.h>
|
|
10
11
|
#include <string.h>
|
|
11
12
|
|
|
@@ -14,16 +15,182 @@ typedef struct {
|
|
|
14
15
|
char *column;
|
|
15
16
|
int desc; /* 1 = highest first (default) */
|
|
16
17
|
tf_batch *buf;
|
|
18
|
+
size_t *heap; /* row indices in buf; root is worst retained row */
|
|
19
|
+
size_t heap_len;
|
|
20
|
+
size_t replacements_since_compact;
|
|
17
21
|
int has_schema;
|
|
18
22
|
int col_idx;
|
|
19
23
|
} top_state;
|
|
20
24
|
|
|
21
|
-
static
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
25
|
+
static int top_compare_cells(const tf_batch *a, size_t ra,
|
|
26
|
+
const tf_batch *b, size_t rb,
|
|
27
|
+
int ci, int desc) {
|
|
28
|
+
int null_a = tf_batch_is_null(a, ra, ci);
|
|
29
|
+
int null_b = tf_batch_is_null(b, rb, ci);
|
|
30
|
+
|
|
31
|
+
/* Match sort semantics: nulls rank last for both ascending and descending. */
|
|
32
|
+
if (null_a && null_b) return 0;
|
|
33
|
+
if (null_a) return 1;
|
|
34
|
+
if (null_b) return -1;
|
|
35
|
+
|
|
36
|
+
int cmp = 0;
|
|
37
|
+
switch (a->col_types[ci]) {
|
|
38
|
+
case TF_TYPE_BOOL: {
|
|
39
|
+
bool va = tf_batch_get_bool(a, ra, ci);
|
|
40
|
+
bool vb = tf_batch_get_bool(b, rb, ci);
|
|
41
|
+
cmp = (int)va - (int)vb;
|
|
42
|
+
break;
|
|
43
|
+
}
|
|
44
|
+
case TF_TYPE_INT64: {
|
|
45
|
+
int64_t va = tf_batch_get_int64(a, ra, ci);
|
|
46
|
+
int64_t vb = tf_batch_get_int64(b, rb, ci);
|
|
47
|
+
cmp = (va > vb) - (va < vb);
|
|
48
|
+
break;
|
|
49
|
+
}
|
|
50
|
+
case TF_TYPE_FLOAT64: {
|
|
51
|
+
double va = tf_batch_get_float64(a, ra, ci);
|
|
52
|
+
double vb = tf_batch_get_float64(b, rb, ci);
|
|
53
|
+
cmp = (va > vb) - (va < vb);
|
|
54
|
+
break;
|
|
55
|
+
}
|
|
56
|
+
case TF_TYPE_STRING:
|
|
57
|
+
cmp = strcmp(tf_batch_get_string(a, ra, ci),
|
|
58
|
+
tf_batch_get_string(b, rb, ci));
|
|
59
|
+
break;
|
|
60
|
+
case TF_TYPE_DATE: {
|
|
61
|
+
int32_t va = tf_batch_get_date(a, ra, ci);
|
|
62
|
+
int32_t vb = tf_batch_get_date(b, rb, ci);
|
|
63
|
+
cmp = (va > vb) - (va < vb);
|
|
64
|
+
break;
|
|
65
|
+
}
|
|
66
|
+
case TF_TYPE_TIMESTAMP: {
|
|
67
|
+
int64_t va = tf_batch_get_timestamp(a, ra, ci);
|
|
68
|
+
int64_t vb = tf_batch_get_timestamp(b, rb, ci);
|
|
69
|
+
cmp = (va > vb) - (va < vb);
|
|
70
|
+
break;
|
|
71
|
+
}
|
|
72
|
+
default:
|
|
73
|
+
break;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return desc ? -cmp : cmp;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
static int top_heap_worse(const top_state *st, size_t a, size_t b) {
|
|
80
|
+
return top_compare_cells(st->buf, a, st->buf, b, st->col_idx, st->desc) > 0;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
static void top_heap_swap(size_t *a, size_t *b) {
|
|
84
|
+
size_t tmp = *a;
|
|
85
|
+
*a = *b;
|
|
86
|
+
*b = tmp;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
static void top_heap_sift_up(top_state *st, size_t pos) {
|
|
90
|
+
while (pos > 0) {
|
|
91
|
+
size_t parent = (pos - 1) / 2;
|
|
92
|
+
if (!top_heap_worse(st, st->heap[pos], st->heap[parent])) break;
|
|
93
|
+
top_heap_swap(&st->heap[pos], &st->heap[parent]);
|
|
94
|
+
pos = parent;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
static void top_heap_sift_down(top_state *st, size_t pos) {
|
|
99
|
+
for (;;) {
|
|
100
|
+
size_t left = pos * 2 + 1;
|
|
101
|
+
size_t right = left + 1;
|
|
102
|
+
size_t worst = pos;
|
|
103
|
+
|
|
104
|
+
if (left < st->heap_len && top_heap_worse(st, st->heap[left], st->heap[worst])) {
|
|
105
|
+
worst = left;
|
|
106
|
+
}
|
|
107
|
+
if (right < st->heap_len && top_heap_worse(st, st->heap[right], st->heap[worst])) {
|
|
108
|
+
worst = right;
|
|
109
|
+
}
|
|
110
|
+
if (worst == pos) break;
|
|
111
|
+
top_heap_swap(&st->heap[pos], &st->heap[worst]);
|
|
112
|
+
pos = worst;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
static void top_heap_push(top_state *st, size_t row) {
|
|
117
|
+
st->heap[st->heap_len] = row;
|
|
118
|
+
top_heap_sift_up(st, st->heap_len);
|
|
119
|
+
st->heap_len++;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
static void top_heap_rebuild(top_state *st) {
|
|
123
|
+
st->heap_len = 0;
|
|
124
|
+
for (size_t r = 0; r < st->buf->n_rows; r++) {
|
|
125
|
+
top_heap_push(st, r);
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
static int top_compact_buffer(top_state *st) {
|
|
130
|
+
if (!st->buf) return TF_OK;
|
|
131
|
+
|
|
132
|
+
tf_batch *old = st->buf;
|
|
133
|
+
size_t cap = st->n > 0 ? st->n : 1;
|
|
134
|
+
tf_batch *nb = tf_batch_create(old->n_cols, cap);
|
|
135
|
+
if (!nb) return TF_ERROR;
|
|
136
|
+
|
|
137
|
+
if (tf_batch_clone_schema(nb, old) != TF_OK) {
|
|
138
|
+
tf_batch_free(nb);
|
|
139
|
+
return TF_ERROR;
|
|
140
|
+
}
|
|
141
|
+
for (size_t r = 0; r < old->n_rows; r++) {
|
|
142
|
+
if (tf_batch_copy_row(nb, r, old, r) != TF_OK) {
|
|
143
|
+
tf_batch_free(nb);
|
|
144
|
+
return TF_ERROR;
|
|
145
|
+
}
|
|
146
|
+
if (tf_batch_expose_row(nb, r) != TF_OK) {
|
|
147
|
+
tf_batch_free(nb);
|
|
148
|
+
return TF_ERROR;
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
st->buf = nb;
|
|
153
|
+
tf_batch_free(old);
|
|
154
|
+
top_heap_rebuild(st);
|
|
155
|
+
st->replacements_since_compact = 0;
|
|
156
|
+
return TF_OK;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
static int top_init_schema(top_state *st, const tf_batch *in) {
|
|
160
|
+
size_t cap = st->n > 0 ? st->n : 1;
|
|
161
|
+
tf_batch *buf = tf_batch_create(in->n_cols, cap);
|
|
162
|
+
if (!buf) return TF_ERROR;
|
|
163
|
+
|
|
164
|
+
size_t *heap = NULL;
|
|
165
|
+
if (st->n > 0) {
|
|
166
|
+
heap = calloc(st->n, sizeof(size_t));
|
|
167
|
+
if (!heap) {
|
|
168
|
+
tf_batch_free(buf);
|
|
169
|
+
return TF_ERROR;
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
int col_idx = tf_batch_col_index(in, st->column);
|
|
174
|
+
if (col_idx < 0) {
|
|
175
|
+
char msg[256];
|
|
176
|
+
snprintf(msg, sizeof(msg), "top: column '%s' not found", st->column ? st->column : "");
|
|
177
|
+
tf_set_last_error(msg);
|
|
178
|
+
free(heap);
|
|
179
|
+
tf_batch_free(buf);
|
|
180
|
+
return TF_ERROR;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
if (tf_batch_clone_schema(buf, in) != TF_OK) {
|
|
184
|
+
free(heap);
|
|
185
|
+
tf_batch_free(buf);
|
|
186
|
+
return TF_ERROR;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
st->buf = buf;
|
|
190
|
+
st->heap = heap;
|
|
191
|
+
st->col_idx = col_idx;
|
|
192
|
+
st->has_schema = 1;
|
|
193
|
+
return TF_OK;
|
|
27
194
|
}
|
|
28
195
|
|
|
29
196
|
static int top_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
@@ -33,55 +200,47 @@ static int top_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
|
33
200
|
*out = NULL;
|
|
34
201
|
|
|
35
202
|
if (!st->has_schema) {
|
|
36
|
-
|
|
37
|
-
if (!st->buf) return TF_ERROR;
|
|
38
|
-
for (size_t c = 0; c < in->n_cols; c++)
|
|
39
|
-
tf_batch_set_schema(st->buf, c, in->col_names[c], in->col_types[c]);
|
|
40
|
-
st->col_idx = tf_batch_col_index(in, st->column);
|
|
41
|
-
st->has_schema = 1;
|
|
203
|
+
if (top_init_schema(st, in) != TF_OK) return TF_ERROR;
|
|
42
204
|
}
|
|
43
205
|
|
|
44
|
-
|
|
45
|
-
double new_val = top_get_val(in, r, st->col_idx >= 0 ? st->col_idx : 0);
|
|
206
|
+
if (st->n == 0) return TF_OK;
|
|
46
207
|
|
|
208
|
+
int ci = st->col_idx;
|
|
209
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
47
210
|
if (st->buf->n_rows < st->n) {
|
|
48
|
-
/* Buffer not full, just add */
|
|
49
211
|
size_t dst = st->buf->n_rows;
|
|
50
|
-
tf_batch_copy_row(st->buf, dst, in, r);
|
|
51
|
-
st->buf
|
|
212
|
+
if (tf_batch_copy_row(st->buf, dst, in, r) != TF_OK) return TF_ERROR;
|
|
213
|
+
if (tf_batch_expose_row(st->buf, dst) != TF_OK) return TF_ERROR;
|
|
214
|
+
top_heap_push(st, dst);
|
|
52
215
|
} else {
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
216
|
+
size_t worst_idx = st->heap[0];
|
|
217
|
+
if (top_compare_cells(in, r, st->buf, worst_idx, ci, st->desc) < 0) {
|
|
218
|
+
if (tf_batch_copy_row(st->buf, worst_idx, in, r) != TF_OK) return TF_ERROR;
|
|
219
|
+
top_heap_sift_down(st, 0);
|
|
220
|
+
st->replacements_since_compact++;
|
|
221
|
+
size_t compact_every = st->n < 4096 ? st->n : 4096;
|
|
222
|
+
if (compact_every > 0 && st->replacements_since_compact >= compact_every) {
|
|
223
|
+
if (top_compact_buffer(st) != TF_OK) return TF_ERROR;
|
|
61
224
|
}
|
|
62
225
|
}
|
|
63
|
-
/* Replace worst if new value is better */
|
|
64
|
-
int replace = st->desc ? (new_val > worst_val) : (new_val < worst_val);
|
|
65
|
-
if (replace) {
|
|
66
|
-
tf_batch_copy_row(st->buf, worst_idx, in, r);
|
|
67
|
-
}
|
|
68
226
|
}
|
|
69
227
|
}
|
|
70
228
|
|
|
71
229
|
return TF_OK;
|
|
72
230
|
}
|
|
73
231
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
232
|
+
typedef struct {
|
|
233
|
+
const tf_batch *batch;
|
|
234
|
+
int col_idx;
|
|
235
|
+
int desc;
|
|
236
|
+
} top_sort_ctx;
|
|
237
|
+
|
|
238
|
+
static int top_sort_cmp(const top_sort_ctx *ctx, size_t a, size_t b) {
|
|
239
|
+
return top_compare_cells(ctx->batch, a, ctx->batch, b, ctx->col_idx, ctx->desc);
|
|
240
|
+
}
|
|
77
241
|
|
|
78
|
-
static int
|
|
79
|
-
|
|
80
|
-
size_t rb = *(const size_t *)b;
|
|
81
|
-
double va = top_get_val(g_top_ctx->batch, ra, g_top_ctx->col_idx);
|
|
82
|
-
double vb = top_get_val(g_top_ctx->batch, rb, g_top_ctx->col_idx);
|
|
83
|
-
int cmp = (va > vb) - (va < vb);
|
|
84
|
-
return g_top_ctx->desc ? -cmp : cmp;
|
|
242
|
+
static int top_sort_cmp_index(const void *ctx, size_t a, size_t b) {
|
|
243
|
+
return top_sort_cmp((const top_sort_ctx *)ctx, a, b);
|
|
85
244
|
}
|
|
86
245
|
|
|
87
246
|
static int top_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
@@ -97,18 +256,27 @@ static int top_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
|
97
256
|
if (!indices) return TF_ERROR;
|
|
98
257
|
for (size_t i = 0; i < n; i++) indices[i] = i;
|
|
99
258
|
|
|
100
|
-
top_sort_ctx ctx = { .batch = st->buf, .col_idx = st->col_idx
|
|
101
|
-
|
|
102
|
-
qsort(indices, n, sizeof(size_t), top_compare);
|
|
103
|
-
g_top_ctx = NULL;
|
|
259
|
+
top_sort_ctx ctx = { .batch = st->buf, .col_idx = st->col_idx, .desc = st->desc };
|
|
260
|
+
tf_sort_indices(indices, n, top_sort_cmp_index, &ctx);
|
|
104
261
|
|
|
105
262
|
tf_batch *ob = tf_batch_create(st->buf->n_cols, n);
|
|
106
263
|
if (!ob) { free(indices); return TF_ERROR; }
|
|
107
|
-
|
|
108
|
-
|
|
264
|
+
if (tf_batch_clone_schema(ob, st->buf) != TF_OK) {
|
|
265
|
+
free(indices);
|
|
266
|
+
tf_batch_free(ob);
|
|
267
|
+
return TF_ERROR;
|
|
268
|
+
}
|
|
109
269
|
for (size_t i = 0; i < n; i++) {
|
|
110
|
-
tf_batch_copy_row(ob, i, st->buf, indices[i])
|
|
111
|
-
|
|
270
|
+
if (tf_batch_copy_row(ob, i, st->buf, indices[i]) != TF_OK) {
|
|
271
|
+
free(indices);
|
|
272
|
+
tf_batch_free(ob);
|
|
273
|
+
return TF_ERROR;
|
|
274
|
+
}
|
|
275
|
+
if (tf_batch_expose_row(ob, i) != TF_OK) {
|
|
276
|
+
free(indices);
|
|
277
|
+
tf_batch_free(ob);
|
|
278
|
+
return TF_ERROR;
|
|
279
|
+
}
|
|
112
280
|
}
|
|
113
281
|
|
|
114
282
|
free(indices);
|
|
@@ -120,6 +288,7 @@ static void top_destroy(tf_step *self) {
|
|
|
120
288
|
top_state *st = self->state;
|
|
121
289
|
if (st) {
|
|
122
290
|
free(st->column);
|
|
291
|
+
free(st->heap);
|
|
123
292
|
if (st->buf) tf_batch_free(st->buf);
|
|
124
293
|
free(st);
|
|
125
294
|
}
|
|
@@ -128,19 +297,26 @@ static void top_destroy(tf_step *self) {
|
|
|
128
297
|
|
|
129
298
|
tf_step *tf_top_create(const cJSON *args) {
|
|
130
299
|
if (!args) return NULL;
|
|
131
|
-
cJSON *n_j = cJSON_GetObjectItemCaseSensitive(args, "n");
|
|
132
300
|
cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
|
|
133
|
-
|
|
301
|
+
size_t n = 0;
|
|
302
|
+
int has_n = tf_json_get_size_arg(args, "n", 0, TF_MAX_OUTPUT_ROWS_PER_BATCH, &n, "top");
|
|
303
|
+
if (has_n <= 0) return NULL;
|
|
304
|
+
if (!cJSON_IsString(col_j) || !col_j->valuestring || col_j->valuestring[0] == '\0') {
|
|
305
|
+
tf_set_last_error("top: column must be a non-empty string");
|
|
306
|
+
return NULL;
|
|
307
|
+
}
|
|
134
308
|
|
|
135
309
|
top_state *st = calloc(1, sizeof(top_state));
|
|
136
310
|
if (!st) return NULL;
|
|
137
|
-
st->n =
|
|
311
|
+
st->n = n;
|
|
312
|
+
st->col_idx = -1;
|
|
138
313
|
st->column = strdup(col_j->valuestring);
|
|
314
|
+
if (!st->column) { free(st); return NULL; }
|
|
139
315
|
|
|
140
316
|
cJSON *desc_j = cJSON_GetObjectItemCaseSensitive(args, "desc");
|
|
141
317
|
st->desc = (desc_j && cJSON_IsBool(desc_j)) ? cJSON_IsTrue(desc_j) : 1; /* default desc */
|
|
142
318
|
|
|
143
|
-
tf_step *step =
|
|
319
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
144
320
|
if (!step) { free(st->column); free(st); return NULL; }
|
|
145
321
|
step->process = top_process;
|
|
146
322
|
step->flush = top_flush;
|
|
@@ -148,3 +324,34 @@ tf_step *tf_top_create(const cJSON *args) {
|
|
|
148
324
|
step->state = st;
|
|
149
325
|
return step;
|
|
150
326
|
}
|
|
327
|
+
|
|
328
|
+
static tf_step *top_create_with_default_desc(const cJSON *args, int default_desc) {
|
|
329
|
+
if (!args) return NULL;
|
|
330
|
+
cJSON *copy = cJSON_Duplicate(args, 1);
|
|
331
|
+
if (!copy) return NULL;
|
|
332
|
+
if (!cJSON_GetObjectItemCaseSensitive(copy, "desc")) {
|
|
333
|
+
if (tf_json_add_bool(copy, "desc", default_desc) != TF_OK) {
|
|
334
|
+
cJSON_Delete(copy);
|
|
335
|
+
return NULL;
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
tf_step *step = tf_top_create(copy);
|
|
339
|
+
cJSON_Delete(copy);
|
|
340
|
+
return step;
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
tf_step *tf_top_k_create(const cJSON *args) {
|
|
344
|
+
return top_create_with_default_desc(args, 1);
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
tf_step *tf_bottom_k_create(const cJSON *args) {
|
|
348
|
+
return top_create_with_default_desc(args, 0);
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
tf_step *tf_slice_min_create(const cJSON *args) {
|
|
352
|
+
return top_create_with_default_desc(args, 0);
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
tf_step *tf_slice_max_create(const cJSON *args) {
|
|
356
|
+
return top_create_with_default_desc(args, 1);
|
|
357
|
+
}
|
package/csrc/op_trim.c
CHANGED
|
@@ -16,14 +16,13 @@ typedef struct {
|
|
|
16
16
|
size_t n_cols;
|
|
17
17
|
} trim_state;
|
|
18
18
|
|
|
19
|
-
static
|
|
19
|
+
static void trim_bounds(const char *s, const char **start, size_t *len_out) {
|
|
20
|
+
if (!s) s = "";
|
|
20
21
|
while (*s && isspace((unsigned char)*s)) s++;
|
|
21
22
|
size_t len = strlen(s);
|
|
22
23
|
while (len > 0 && isspace((unsigned char)s[len - 1])) len--;
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
buf[len] = '\0';
|
|
26
|
-
return buf;
|
|
24
|
+
*start = s;
|
|
25
|
+
*len_out = len;
|
|
27
26
|
}
|
|
28
27
|
|
|
29
28
|
static int trim_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
@@ -34,12 +33,20 @@ static int trim_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
|
34
33
|
|
|
35
34
|
tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
|
|
36
35
|
if (!ob) return TF_ERROR;
|
|
37
|
-
|
|
38
|
-
|
|
36
|
+
if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
37
|
+
tf_batch_free(ob);
|
|
38
|
+
return TF_ERROR;
|
|
39
|
+
}
|
|
39
40
|
|
|
40
41
|
for (size_t r = 0; r < in->n_rows; r++) {
|
|
41
|
-
tf_batch_copy_row(ob, r, in, r)
|
|
42
|
-
|
|
42
|
+
if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
|
|
43
|
+
tf_batch_free(ob);
|
|
44
|
+
return TF_ERROR;
|
|
45
|
+
}
|
|
46
|
+
if (tf_batch_expose_row(ob, r) != TF_OK) {
|
|
47
|
+
tf_batch_free(ob);
|
|
48
|
+
return TF_ERROR;
|
|
49
|
+
}
|
|
43
50
|
}
|
|
44
51
|
|
|
45
52
|
/* Trim target columns */
|
|
@@ -54,11 +61,16 @@ static int trim_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
|
54
61
|
}
|
|
55
62
|
}
|
|
56
63
|
if (!target) continue;
|
|
57
|
-
char buf[4096];
|
|
58
64
|
for (size_t r = 0; r < ob->n_rows; r++) {
|
|
59
65
|
if (tf_batch_is_null(ob, r, c)) continue;
|
|
60
66
|
const char *val = tf_batch_get_string(ob, r, c);
|
|
61
|
-
|
|
67
|
+
const char *trimmed = NULL;
|
|
68
|
+
size_t trimmed_len = 0;
|
|
69
|
+
trim_bounds(val, &trimmed, &trimmed_len);
|
|
70
|
+
if (tf_batch_set_string_len(ob, r, c, trimmed, trimmed_len) != TF_OK) {
|
|
71
|
+
tf_batch_free(ob);
|
|
72
|
+
return TF_ERROR;
|
|
73
|
+
}
|
|
62
74
|
}
|
|
63
75
|
}
|
|
64
76
|
|
|
@@ -70,13 +82,18 @@ static int trim_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
|
70
82
|
(void)self; (void)side; *out = NULL; return TF_OK;
|
|
71
83
|
}
|
|
72
84
|
|
|
73
|
-
static void
|
|
74
|
-
trim_state *st = self->state;
|
|
85
|
+
static void trim_state_free(trim_state *st) {
|
|
75
86
|
if (st) {
|
|
76
|
-
|
|
87
|
+
if (st->cols) {
|
|
88
|
+
for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i]);
|
|
89
|
+
}
|
|
77
90
|
free(st->cols);
|
|
78
91
|
free(st);
|
|
79
92
|
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
static void trim_destroy(tf_step *self) {
|
|
96
|
+
trim_state_free(self ? self->state : NULL);
|
|
80
97
|
free(self);
|
|
81
98
|
}
|
|
82
99
|
|
|
@@ -86,21 +103,33 @@ tf_step *tf_trim_create(const cJSON *args) {
|
|
|
86
103
|
|
|
87
104
|
if (args) {
|
|
88
105
|
cJSON *columns = cJSON_GetObjectItemCaseSensitive(args, "columns");
|
|
89
|
-
if (columns
|
|
106
|
+
if (columns) {
|
|
107
|
+
if (!cJSON_IsArray(columns)) {
|
|
108
|
+
tf_set_last_error("trim: columns must be an array");
|
|
109
|
+
trim_state_free(st);
|
|
110
|
+
return NULL;
|
|
111
|
+
}
|
|
90
112
|
int n = cJSON_GetArraySize(columns);
|
|
91
113
|
if (n > 0) {
|
|
92
114
|
st->cols = calloc(n, sizeof(char *));
|
|
93
|
-
st->
|
|
115
|
+
if (!st->cols) { trim_state_free(st); return NULL; }
|
|
116
|
+
st->n_cols = (size_t)n;
|
|
94
117
|
for (int i = 0; i < n; i++) {
|
|
95
118
|
cJSON *item = cJSON_GetArrayItem(columns, i);
|
|
96
|
-
if (cJSON_IsString(item)
|
|
119
|
+
if (!cJSON_IsString(item) || !item->valuestring || !item->valuestring[0]) {
|
|
120
|
+
tf_set_last_error("trim: column names must be non-empty strings");
|
|
121
|
+
trim_state_free(st);
|
|
122
|
+
return NULL;
|
|
123
|
+
}
|
|
124
|
+
st->cols[i] = strdup(item->valuestring);
|
|
125
|
+
if (!st->cols[i]) { trim_state_free(st); return NULL; }
|
|
97
126
|
}
|
|
98
127
|
}
|
|
99
128
|
}
|
|
100
129
|
}
|
|
101
130
|
|
|
102
|
-
tf_step *step =
|
|
103
|
-
if (!step) {
|
|
131
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
132
|
+
if (!step) { trim_state_free(st); return NULL; }
|
|
104
133
|
step->process = trim_process;
|
|
105
134
|
step->flush = trim_flush;
|
|
106
135
|
step->destroy = trim_destroy;
|