tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/op_sort.c CHANGED
@@ -1,94 +1,164 @@
1
1
  /*
2
- * op_sort.c — Sort all rows by column(s). Requires buffering all data.
2
+ * op_sort.c -- Sort rows by column(s).
3
3
  *
4
- * Config: {"columns": [{"name": "age", "desc": false}, {"name": "name", "desc": true}]}
5
- * Buffers all incoming batches, sorts on flush, emits as a single batch.
4
+ * Default mode preserves the historical in-memory/blocking sort. When the
5
+ * args contain spill_dir, the operator writes memory-sized sorted runs to
6
+ * temporary files and merges them at flush in bounded output batches.
6
7
  */
7
8
 
8
9
  #include "internal.h"
10
+ #include "spill.h"
9
11
  #include "cJSON.h"
12
+
13
+ #include <errno.h>
14
+ #include <stdint.h>
15
+ #include <stdio.h>
10
16
  #include <stdlib.h>
11
17
  #include <string.h>
12
- #include <stdio.h>
18
+ #include <unistd.h>
19
+
20
+ #define SORT_DEFAULT_RUN_ROWS 8192
21
+ #define SORT_DEFAULT_OUTPUT_ROWS 1024
22
+ #define SORT_MIN_RUN_ROWS 16
13
23
 
24
+ /* Sort spec. */
14
25
  typedef struct {
15
26
  char *name;
16
- int desc; /* 1 = descending */
27
+ int desc;
17
28
  } sort_col;
18
29
 
30
+ typedef tf_owned_cell_value sort_cell;
31
+
32
+ typedef struct {
33
+ uint8_t *nulls;
34
+ sort_cell *cells;
35
+ } sort_spill_row;
36
+
37
+ typedef struct {
38
+ FILE *file;
39
+ sort_spill_row row;
40
+ int has_row;
41
+ int done;
42
+ } sort_run_reader;
43
+
19
44
  typedef struct {
20
- /* Accumulated data */
21
- tf_batch *buf; /* growing buffer of all rows */
45
+ tf_batch *buf;
22
46
  int has_schema;
47
+ char **schema_names;
48
+ tf_type *schema_types;
49
+ size_t n_schema_cols;
23
50
 
24
- /* Sort spec */
25
51
  sort_col *cols;
26
52
  size_t n_cols;
53
+ int *col_indices;
54
+ int *col_desc;
55
+
56
+ int use_spill;
57
+ char *spill_dir;
58
+ tf_spill_session *spill;
59
+ size_t spill_memory_bytes;
60
+ size_t configured_run_rows;
61
+ size_t run_rows;
62
+ size_t output_batch_rows;
63
+
64
+ char **run_paths;
65
+ size_t n_runs;
66
+ size_t cap_runs;
67
+ size_t run_seq;
68
+ size_t spilled_bytes;
69
+ size_t spill_runs_created;
70
+ size_t spill_output_batches;
71
+ size_t spill_output_rows;
72
+
73
+ sort_run_reader *readers;
74
+ size_t n_readers;
75
+ int merge_started;
76
+ int merge_done;
27
77
  } sort_state;
28
78
 
29
- /* Copy a row from src batch to dst batch */
30
- static int copy_row(tf_batch *dst, size_t dst_row,
31
- const tf_batch *src, size_t src_row) {
32
- if (tf_batch_ensure_capacity(dst, dst_row + 1) != TF_OK) return TF_ERROR;
33
- for (size_t c = 0; c < src->n_cols; c++) {
34
- if (tf_batch_is_null(src, src_row, c)) {
35
- tf_batch_set_null(dst, dst_row, c);
36
- continue;
37
- }
38
- switch (src->col_types[c]) {
39
- case TF_TYPE_BOOL:
40
- tf_batch_set_bool(dst, dst_row, c, tf_batch_get_bool(src, src_row, c));
41
- break;
42
- case TF_TYPE_INT64:
43
- tf_batch_set_int64(dst, dst_row, c, tf_batch_get_int64(src, src_row, c));
44
- break;
45
- case TF_TYPE_FLOAT64:
46
- tf_batch_set_float64(dst, dst_row, c, tf_batch_get_float64(src, src_row, c));
47
- break;
48
- case TF_TYPE_STRING:
49
- tf_batch_set_string(dst, dst_row, c, tf_batch_get_string(src, src_row, c));
50
- break;
51
- case TF_TYPE_DATE:
52
- tf_batch_set_date(dst, dst_row, c, tf_batch_get_date(src, src_row, c));
53
- break;
54
- case TF_TYPE_TIMESTAMP:
55
- tf_batch_set_timestamp(dst, dst_row, c, tf_batch_get_timestamp(src, src_row, c));
56
- break;
57
- default:
58
- tf_batch_set_null(dst, dst_row, c);
59
- break;
79
+ static tf_batch *create_buffer_from_schema(const sort_state *st, size_t capacity) {
80
+ tf_batch *b = tf_batch_create(st->n_schema_cols, capacity ? capacity : 16);
81
+ if (!b) return NULL;
82
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
83
+ if (tf_batch_set_schema(b, c, st->schema_names[c], st->schema_types[c]) != TF_OK) {
84
+ tf_batch_free(b);
85
+ return NULL;
60
86
  }
61
87
  }
62
- return TF_OK;
88
+ return b;
63
89
  }
64
90
 
65
- static int sort_process(tf_step *self, tf_batch *in, tf_batch **out,
66
- tf_side_channels *side) {
67
- (void)side;
68
- sort_state *st = self->state;
69
- *out = NULL;
91
+ static size_t sort_estimated_row_bytes(const sort_state *st) {
92
+ size_t bytes = 32;
93
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
94
+ bytes += 1;
95
+ switch (st->schema_types[c]) {
96
+ case TF_TYPE_BOOL: bytes += 1; break;
97
+ case TF_TYPE_INT64: bytes += sizeof(int64_t); break;
98
+ case TF_TYPE_FLOAT64: bytes += sizeof(double); break;
99
+ case TF_TYPE_STRING: bytes += sizeof(char *) + 64; break;
100
+ case TF_TYPE_DATE: bytes += sizeof(int32_t); break;
101
+ case TF_TYPE_TIMESTAMP: bytes += sizeof(int64_t); break;
102
+ default: break;
103
+ }
104
+ }
105
+ return bytes < 64 ? 64 : bytes;
106
+ }
70
107
 
71
- /* Initialize buffer on first batch */
72
- if (!st->has_schema) {
73
- st->buf = tf_batch_create(in->n_cols, in->n_rows > 0 ? in->n_rows : 16);
74
- if (!st->buf) return TF_ERROR;
75
- for (size_t c = 0; c < in->n_cols; c++) {
76
- tf_batch_set_schema(st->buf, c, in->col_names[c], in->col_types[c]);
108
+ static int resolve_sort_columns(sort_state *st) {
109
+ free(st->col_indices);
110
+ free(st->col_desc);
111
+ st->col_indices = tf_callocarray_checked(st->n_cols ? st->n_cols : 1, sizeof(int));
112
+ st->col_desc = tf_callocarray_checked(st->n_cols ? st->n_cols : 1, sizeof(int));
113
+ if (!st->col_indices || !st->col_desc) return TF_ERROR;
114
+ for (size_t k = 0; k < st->n_cols; k++) {
115
+ int idx = tf_batch_col_index(st->buf, st->cols[k].name);
116
+ if (idx < 0) {
117
+ char msg[256];
118
+ snprintf(msg, sizeof(msg), "sort: column '%s' not found",
119
+ st->cols[k].name ? st->cols[k].name : "");
120
+ tf_set_last_error(msg);
121
+ return TF_ERROR;
77
122
  }
78
- st->has_schema = 1;
123
+ st->col_indices[k] = idx;
124
+ st->col_desc[k] = st->cols[k].desc;
79
125
  }
126
+ return TF_OK;
127
+ }
80
128
 
81
- /* Append all rows */
82
- for (size_t r = 0; r < in->n_rows; r++) {
83
- size_t dst_row = st->buf->n_rows;
84
- if (copy_row(st->buf, dst_row, in, r) != TF_OK) return TF_ERROR;
85
- st->buf->n_rows = dst_row + 1;
129
+ static int init_schema(sort_state *st, const tf_batch *in) {
130
+ if (st->has_schema) return TF_OK;
131
+
132
+ st->n_schema_cols = in->n_cols;
133
+ st->schema_names = tf_callocarray_checked(in->n_cols ? in->n_cols : 1, sizeof(char *));
134
+ st->schema_types = tf_callocarray_checked(in->n_cols ? in->n_cols : 1, sizeof(tf_type));
135
+ if (!st->schema_names || !st->schema_types) return TF_ERROR;
136
+ for (size_t c = 0; c < in->n_cols; c++) {
137
+ st->schema_names[c] = strdup(in->col_names[c] ? in->col_names[c] : "");
138
+ if (!st->schema_names[c]) return TF_ERROR;
139
+ st->schema_types[c] = in->col_types[c];
86
140
  }
87
141
 
88
- return TF_OK;
142
+ size_t initial = in->n_rows > 0 ? in->n_rows : 16;
143
+ if (st->use_spill) {
144
+ if (st->configured_run_rows > 0) {
145
+ st->run_rows = st->configured_run_rows;
146
+ } else if (st->spill_memory_bytes > 0) {
147
+ size_t row_bytes = sort_estimated_row_bytes(st);
148
+ st->run_rows = st->spill_memory_bytes / (row_bytes * 3);
149
+ if (st->run_rows < SORT_MIN_RUN_ROWS) st->run_rows = SORT_MIN_RUN_ROWS;
150
+ } else {
151
+ st->run_rows = SORT_DEFAULT_RUN_ROWS;
152
+ }
153
+ initial = st->run_rows;
154
+ }
155
+
156
+ st->buf = create_buffer_from_schema(st, initial);
157
+ if (!st->buf) return TF_ERROR;
158
+ st->has_schema = 1;
159
+ return resolve_sort_columns(st);
89
160
  }
90
161
 
91
- /* Comparator context for qsort_r / qsort */
92
162
  typedef struct {
93
163
  const tf_batch *batch;
94
164
  int *col_indices;
@@ -96,21 +166,14 @@ typedef struct {
96
166
  size_t n_sort_cols;
97
167
  } sort_ctx;
98
168
 
99
- static sort_ctx *g_sort_ctx; /* global for qsort (no qsort_r on all platforms) */
100
-
101
- static int compare_rows(const void *a, const void *b) {
102
- size_t ra = *(const size_t *)a;
103
- size_t rb = *(const size_t *)b;
104
- const tf_batch *batch = g_sort_ctx->batch;
169
+ static int sort_compare_rows(const sort_ctx *ctx, size_t ra, size_t rb) {
170
+ const tf_batch *batch = ctx->batch;
105
171
 
106
- for (size_t k = 0; k < g_sort_ctx->n_sort_cols; k++) {
107
- int ci = g_sort_ctx->col_indices[k];
108
- if (ci < 0) continue;
172
+ for (size_t k = 0; k < ctx->n_sort_cols; k++) {
173
+ int ci = ctx->col_indices[k];
109
174
 
110
175
  int null_a = tf_batch_is_null(batch, ra, ci);
111
176
  int null_b = tf_batch_is_null(batch, rb, ci);
112
-
113
- /* Nulls sort last */
114
177
  if (null_a && null_b) continue;
115
178
  if (null_a) return 1;
116
179
  if (null_b) return -1;
@@ -155,80 +218,527 @@ static int compare_rows(const void *a, const void *b) {
155
218
  break;
156
219
  }
157
220
 
158
- if (cmp != 0) {
159
- return g_sort_ctx->col_desc[k] ? -cmp : cmp;
160
- }
221
+ if (cmp != 0) return ctx->col_desc[k] ? -cmp : cmp;
161
222
  }
162
223
  return 0;
163
224
  }
164
225
 
165
- static int sort_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
166
- (void)side;
167
- sort_state *st = self->state;
168
- *out = NULL;
226
+ static int sort_compare_index(const void *ctx, size_t ra, size_t rb) {
227
+ int cmp = sort_compare_rows((const sort_ctx *)ctx, ra, rb);
228
+ if (cmp != 0) return cmp;
229
+ return (ra > rb) - (ra < rb);
230
+ }
169
231
 
232
+ static size_t *sort_batch_indices(const sort_state *st, const tf_batch *batch) {
233
+ size_t n = batch->n_rows;
234
+ size_t *indices = tf_mallocarray_checked(n ? n : 1, sizeof(size_t));
235
+ if (!indices) return NULL;
236
+ for (size_t i = 0; i < n; i++) indices[i] = i;
237
+ sort_ctx ctx = {
238
+ .batch = batch,
239
+ .col_indices = st->col_indices,
240
+ .col_desc = st->col_desc,
241
+ .n_sort_cols = st->n_cols,
242
+ };
243
+ tf_sort_indices(indices, n, sort_compare_index, &ctx);
244
+ return indices;
245
+ }
246
+
247
+ static int write_exact(FILE *f, const void *ptr, size_t len) {
248
+ return fwrite(ptr, 1, len, f) == len ? TF_OK : TF_ERROR;
249
+ }
250
+
251
+ static int read_exact(FILE *f, void *ptr, size_t len) {
252
+ return fread(ptr, 1, len, f) == len ? TF_OK : TF_ERROR;
253
+ }
254
+
255
+ static int write_cell(FILE *f, const tf_batch *b, size_t r, size_t c) {
256
+ uint8_t is_null = tf_batch_is_null(b, r, c) ? 1 : 0;
257
+ if (write_exact(f, &is_null, sizeof(is_null)) != TF_OK) return TF_ERROR;
258
+ if (is_null) return TF_OK;
259
+
260
+ switch (b->col_types[c]) {
261
+ case TF_TYPE_BOOL: {
262
+ uint8_t v = tf_batch_get_bool(b, r, c) ? 1 : 0;
263
+ return write_exact(f, &v, sizeof(v));
264
+ }
265
+ case TF_TYPE_INT64: {
266
+ int64_t v = tf_batch_get_int64(b, r, c);
267
+ return write_exact(f, &v, sizeof(v));
268
+ }
269
+ case TF_TYPE_FLOAT64: {
270
+ double v = tf_batch_get_float64(b, r, c);
271
+ return write_exact(f, &v, sizeof(v));
272
+ }
273
+ case TF_TYPE_STRING: {
274
+ const char *s = tf_batch_get_string(b, r, c);
275
+ uint64_t len = s ? (uint64_t)strlen(s) : 0;
276
+ if (write_exact(f, &len, sizeof(len)) != TF_OK) return TF_ERROR;
277
+ return len ? write_exact(f, s, (size_t)len) : TF_OK;
278
+ }
279
+ case TF_TYPE_DATE: {
280
+ int32_t v = tf_batch_get_date(b, r, c);
281
+ return write_exact(f, &v, sizeof(v));
282
+ }
283
+ case TF_TYPE_TIMESTAMP: {
284
+ int64_t v = tf_batch_get_timestamp(b, r, c);
285
+ return write_exact(f, &v, sizeof(v));
286
+ }
287
+ default:
288
+ return TF_OK;
289
+ }
290
+ }
291
+
292
+ static int append_run_path(sort_state *st, char *path) {
293
+ if (st->n_runs == st->cap_runs) {
294
+ size_t need = 0;
295
+ size_t new_cap = 0;
296
+ if (tf_size_add(st->n_runs, 1, &need) != TF_OK ||
297
+ tf_size_grow_pow2(st->cap_runs, need, 8, &new_cap) != TF_OK) {
298
+ return TF_ERROR;
299
+ }
300
+ char **tmp = tf_reallocarray_checked(st->run_paths, new_cap, sizeof(char *));
301
+ if (!tmp) return TF_ERROR;
302
+ st->run_paths = tmp;
303
+ st->cap_runs = new_cap;
304
+ }
305
+ st->run_paths[st->n_runs++] = path;
306
+ st->spill_runs_created++;
307
+ return TF_OK;
308
+ }
309
+
310
+ static int write_spill_run(sort_state *st) {
170
311
  if (!st->buf || st->buf->n_rows == 0) return TF_OK;
171
312
 
172
- size_t n = st->buf->n_rows;
313
+ size_t *indices = sort_batch_indices(st, st->buf);
314
+ if (!indices) return TF_ERROR;
315
+
316
+ char *path = NULL;
317
+ FILE *f = tf_spill_open_run_file(st->spill, "sort", &path);
318
+ if (!f) {
319
+ free(indices);
320
+ return TF_ERROR;
321
+ }
322
+
323
+ for (size_t i = 0; i < st->buf->n_rows; i++) {
324
+ size_t r = indices[i];
325
+ for (size_t c = 0; c < st->buf->n_cols; c++) {
326
+ if (write_cell(f, st->buf, r, c) != TF_OK) {
327
+ tf_set_last_error("sort spill: failed writing run file");
328
+ fclose(f);
329
+ remove(path);
330
+ free(path);
331
+ free(indices);
332
+ return TF_ERROR;
333
+ }
334
+ }
335
+ }
336
+ long pos = ftell(f);
337
+ if (pos > 0) st->spilled_bytes += (size_t)pos;
338
+ if (fclose(f) != 0) {
339
+ tf_set_last_error("sort spill: failed closing run file");
340
+ remove(path);
341
+ free(path);
342
+ free(indices);
343
+ return TF_ERROR;
344
+ }
345
+
346
+ free(indices);
347
+ if (append_run_path(st, path) != TF_OK) {
348
+ remove(path);
349
+ free(path);
350
+ return TF_ERROR;
351
+ }
173
352
 
174
- /* Resolve sort column indices */
175
- int *col_indices = malloc(st->n_cols * sizeof(int));
176
- int *col_desc = malloc(st->n_cols * sizeof(int));
177
- if (!col_indices || !col_desc) { free(col_indices); free(col_desc); return TF_ERROR; }
353
+ tf_batch_free(st->buf);
354
+ st->buf = create_buffer_from_schema(st, st->run_rows);
355
+ return st->buf ? TF_OK : TF_ERROR;
356
+ }
357
+
358
+ static void spill_row_clear(sort_spill_row *row, const tf_type *types, size_t n_cols) {
359
+ if (!row || !row->cells || !row->nulls) return;
360
+ for (size_t c = 0; c < n_cols; c++) {
361
+ if (!row->nulls[c] && types[c] == TF_TYPE_STRING) {
362
+ free(row->cells[c].str);
363
+ row->cells[c].str = NULL;
364
+ }
365
+ row->nulls[c] = 1;
366
+ }
367
+ }
368
+
369
+ static int spill_row_init(sort_spill_row *row, size_t n_cols) {
370
+ row->nulls = tf_callocarray_checked(n_cols ? n_cols : 1, sizeof(uint8_t));
371
+ row->cells = tf_callocarray_checked(n_cols ? n_cols : 1, sizeof(sort_cell));
372
+ if (!row->nulls || !row->cells) {
373
+ free(row->nulls);
374
+ free(row->cells);
375
+ row->nulls = NULL;
376
+ row->cells = NULL;
377
+ return TF_ERROR;
378
+ }
379
+ for (size_t c = 0; c < n_cols; c++) row->nulls[c] = 1;
380
+ return TF_OK;
381
+ }
382
+
383
+ static void spill_row_free(sort_spill_row *row, const tf_type *types, size_t n_cols) {
384
+ if (!row) return;
385
+ spill_row_clear(row, types, n_cols);
386
+ free(row->nulls);
387
+ free(row->cells);
388
+ row->nulls = NULL;
389
+ row->cells = NULL;
390
+ }
391
+
392
+ static int read_cell_value(FILE *f, sort_spill_row *row, const tf_type *types, size_t c) {
393
+ switch (types[c]) {
394
+ case TF_TYPE_BOOL: {
395
+ uint8_t v = 0;
396
+ if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR;
397
+ row->cells[c].b = v;
398
+ return TF_OK;
399
+ }
400
+ case TF_TYPE_INT64: {
401
+ int64_t v = 0;
402
+ if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR;
403
+ row->cells[c].i64 = v;
404
+ return TF_OK;
405
+ }
406
+ case TF_TYPE_FLOAT64: {
407
+ double v = 0.0;
408
+ if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR;
409
+ row->cells[c].f64 = v;
410
+ return TF_OK;
411
+ }
412
+ case TF_TYPE_STRING: {
413
+ uint64_t len = 0;
414
+ if (read_exact(f, &len, sizeof(len)) != TF_OK) return TF_ERROR;
415
+ if (len > (uint64_t)SIZE_MAX - 1) return TF_ERROR;
416
+ char *s = malloc((size_t)len + 1);
417
+ if (!s) return TF_ERROR;
418
+ if (len && read_exact(f, s, (size_t)len) != TF_OK) {
419
+ free(s);
420
+ return TF_ERROR;
421
+ }
422
+ s[len] = '\0';
423
+ row->cells[c].str = s;
424
+ return TF_OK;
425
+ }
426
+ case TF_TYPE_DATE: {
427
+ int32_t v = 0;
428
+ if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR;
429
+ row->cells[c].date = v;
430
+ return TF_OK;
431
+ }
432
+ case TF_TYPE_TIMESTAMP: {
433
+ int64_t v = 0;
434
+ if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR;
435
+ row->cells[c].i64 = v;
436
+ return TF_OK;
437
+ }
438
+ default:
439
+ return TF_OK;
440
+ }
441
+ }
178
442
 
443
+ static int reader_advance(sort_state *st, sort_run_reader *reader) {
444
+ if (!reader || !reader->file || reader->done) return 0;
445
+ spill_row_clear(&reader->row, st->schema_types, st->n_schema_cols);
446
+
447
+ int first = fgetc(reader->file);
448
+ if (first == EOF) {
449
+ if (ferror(reader->file)) {
450
+ tf_set_last_error("sort spill: failed reading run file");
451
+ return -1;
452
+ }
453
+ reader->done = 1;
454
+ reader->has_row = 0;
455
+ return 0;
456
+ }
457
+
458
+ reader->row.nulls[0] = first ? 1 : 0;
459
+ if (!reader->row.nulls[0] && read_cell_value(reader->file, &reader->row, st->schema_types, 0) != TF_OK) {
460
+ tf_set_last_error("sort spill: corrupt run file");
461
+ return -1;
462
+ }
463
+ for (size_t c = 1; c < st->n_schema_cols; c++) {
464
+ uint8_t is_null = 1;
465
+ if (read_exact(reader->file, &is_null, sizeof(is_null)) != TF_OK) {
466
+ tf_set_last_error("sort spill: corrupt run file");
467
+ return -1;
468
+ }
469
+ reader->row.nulls[c] = is_null ? 1 : 0;
470
+ if (!reader->row.nulls[c] && read_cell_value(reader->file, &reader->row, st->schema_types, c) != TF_OK) {
471
+ tf_set_last_error("sort spill: corrupt run file");
472
+ return -1;
473
+ }
474
+ }
475
+ reader->has_row = 1;
476
+ return 1;
477
+ }
478
+
479
+ static int compare_spill_rows(const sort_state *st,
480
+ const sort_spill_row *a,
481
+ const sort_spill_row *b) {
179
482
  for (size_t k = 0; k < st->n_cols; k++) {
180
- col_indices[k] = tf_batch_col_index(st->buf, st->cols[k].name);
181
- col_desc[k] = st->cols[k].desc;
483
+ int ci = st->col_indices[k];
484
+ if (ci < 0) continue;
485
+ size_t c = (size_t)ci;
486
+ int null_a = a->nulls[c] != 0;
487
+ int null_b = b->nulls[c] != 0;
488
+ if (null_a && null_b) continue;
489
+ if (null_a) return 1;
490
+ if (null_b) return -1;
491
+
492
+ int cmp = 0;
493
+ switch (st->schema_types[c]) {
494
+ case TF_TYPE_BOOL:
495
+ cmp = (int)a->cells[c].b - (int)b->cells[c].b;
496
+ break;
497
+ case TF_TYPE_INT64:
498
+ case TF_TYPE_TIMESTAMP:
499
+ cmp = (a->cells[c].i64 > b->cells[c].i64) -
500
+ (a->cells[c].i64 < b->cells[c].i64);
501
+ break;
502
+ case TF_TYPE_FLOAT64:
503
+ cmp = (a->cells[c].f64 > b->cells[c].f64) -
504
+ (a->cells[c].f64 < b->cells[c].f64);
505
+ break;
506
+ case TF_TYPE_STRING:
507
+ cmp = strcmp(a->cells[c].str, b->cells[c].str);
508
+ break;
509
+ case TF_TYPE_DATE:
510
+ cmp = (a->cells[c].date > b->cells[c].date) -
511
+ (a->cells[c].date < b->cells[c].date);
512
+ break;
513
+ default:
514
+ break;
515
+ }
516
+ if (cmp != 0) return st->col_desc[k] ? -cmp : cmp;
182
517
  }
518
+ return 0;
519
+ }
183
520
 
184
- /* Build index array */
185
- size_t *indices = malloc(n * sizeof(size_t));
186
- if (!indices) { free(col_indices); free(col_desc); return TF_ERROR; }
187
- for (size_t i = 0; i < n; i++) indices[i] = i;
521
+ static int spill_row_to_batch(const sort_state *st, tf_batch *out,
522
+ size_t dst_row, const sort_spill_row *row) {
523
+ if (tf_batch_ensure_capacity(out, dst_row + 1) != TF_OK) return TF_ERROR;
524
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
525
+ if (tf_batch_set_owned_cell_value(out, dst_row, c, st->schema_types[c],
526
+ row->nulls[c], &row->cells[c]) != TF_OK) {
527
+ return TF_ERROR;
528
+ }
529
+ }
530
+ return TF_OK;
531
+ }
188
532
 
189
- /* Sort */
190
- sort_ctx ctx = {
191
- .batch = st->buf,
192
- .col_indices = col_indices,
193
- .col_desc = col_desc,
194
- .n_sort_cols = st->n_cols,
195
- };
196
- g_sort_ctx = &ctx;
197
- qsort(indices, n, sizeof(size_t), compare_rows);
198
- g_sort_ctx = NULL;
533
+ static void close_readers(sort_state *st) {
534
+ if (!st->readers) return;
535
+ for (size_t i = 0; i < st->n_readers; i++) {
536
+ if (st->readers[i].file) fclose(st->readers[i].file);
537
+ spill_row_free(&st->readers[i].row, st->schema_types, st->n_schema_cols);
538
+ }
539
+ free(st->readers);
540
+ st->readers = NULL;
541
+ st->n_readers = 0;
542
+ }
199
543
 
200
- /* Build output batch in sorted order */
201
- tf_batch *ob = tf_batch_create(st->buf->n_cols, n);
202
- if (!ob) { free(indices); free(col_indices); free(col_desc); return TF_ERROR; }
203
- for (size_t c = 0; c < st->buf->n_cols; c++) {
204
- tf_batch_set_schema(ob, c, st->buf->col_names[c], st->buf->col_types[c]);
544
+ static void remove_run_files(sort_state *st) {
545
+ for (size_t i = 0; i < st->n_runs; i++) {
546
+ if (st->run_paths[i]) {
547
+ remove(st->run_paths[i]);
548
+ free(st->run_paths[i]);
549
+ st->run_paths[i] = NULL;
550
+ }
205
551
  }
552
+ free(st->run_paths);
553
+ st->run_paths = NULL;
554
+ st->n_runs = 0;
555
+ st->cap_runs = 0;
556
+ }
206
557
 
207
- for (size_t i = 0; i < n; i++) {
208
- if (copy_row(ob, i, st->buf, indices[i]) != TF_OK) {
209
- free(indices); free(col_indices); free(col_desc);
558
+ static int begin_merge(sort_state *st) {
559
+ if (st->merge_started) return TF_OK;
560
+ if (st->buf && st->buf->n_rows > 0 && write_spill_run(st) != TF_OK) return TF_ERROR;
561
+ if (st->buf) {
562
+ tf_batch_free(st->buf);
563
+ st->buf = NULL;
564
+ }
565
+
566
+ st->merge_started = 1;
567
+ if (st->n_runs == 0) {
568
+ st->merge_done = 1;
569
+ return TF_OK;
570
+ }
571
+
572
+ st->readers = tf_callocarray_checked(st->n_runs, sizeof(sort_run_reader));
573
+ if (!st->readers) return TF_ERROR;
574
+ st->n_readers = st->n_runs;
575
+ for (size_t i = 0; i < st->n_runs; i++) {
576
+ st->readers[i].file = fopen(st->run_paths[i], "rb");
577
+ if (!st->readers[i].file) {
578
+ tf_set_last_error("sort spill: cannot reopen run file");
579
+ return TF_ERROR;
580
+ }
581
+ if (spill_row_init(&st->readers[i].row, st->n_schema_cols) != TF_OK) return TF_ERROR;
582
+ int rc = reader_advance(st, &st->readers[i]);
583
+ if (rc < 0) return TF_ERROR;
584
+ }
585
+ return TF_OK;
586
+ }
587
+
588
+ static int next_best_reader(const sort_state *st) {
589
+ int best = -1;
590
+ for (size_t i = 0; i < st->n_readers; i++) {
591
+ const sort_run_reader *r = &st->readers[i];
592
+ if (!r->has_row || r->done) continue;
593
+ if (best < 0) {
594
+ best = (int)i;
595
+ continue;
596
+ }
597
+ int cmp = compare_spill_rows(st, &r->row, &st->readers[best].row);
598
+ if (cmp < 0 || (cmp == 0 && i < (size_t)best)) best = (int)i;
599
+ }
600
+ return best;
601
+ }
602
+
603
+ static int spill_next_batch(sort_state *st, tf_batch **out) {
604
+ *out = NULL;
605
+ if (begin_merge(st) != TF_OK) return TF_ERROR;
606
+ if (st->merge_done) return TF_OK;
607
+
608
+ tf_batch *ob = create_buffer_from_schema(st, st->output_batch_rows);
609
+ if (!ob) return TF_ERROR;
610
+
611
+ while (ob->n_rows < st->output_batch_rows) {
612
+ int best = next_best_reader(st);
613
+ if (best < 0) break;
614
+ size_t out_row = ob->n_rows;
615
+ if (spill_row_to_batch(st, ob, out_row, &st->readers[best].row) != TF_OK) {
616
+ tf_batch_free(ob);
617
+ return TF_ERROR;
618
+ }
619
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) {
620
+ tf_batch_free(ob);
621
+ return TF_ERROR;
622
+ }
623
+ int rc = reader_advance(st, &st->readers[best]);
624
+ if (rc < 0) {
210
625
  tf_batch_free(ob);
211
626
  return TF_ERROR;
212
627
  }
213
- ob->n_rows = i + 1;
214
628
  }
215
629
 
216
- free(indices);
217
- free(col_indices);
218
- free(col_desc);
630
+ if (ob->n_rows == 0) {
631
+ tf_batch_free(ob);
632
+ close_readers(st);
633
+ remove_run_files(st);
634
+ tf_spill_cleanup(st->spill);
635
+ st->spill = NULL;
636
+ st->merge_done = 1;
637
+ return TF_OK;
638
+ }
219
639
 
640
+ st->spill_output_batches++;
641
+ st->spill_output_rows += ob->n_rows;
220
642
  *out = ob;
221
643
  return TF_OK;
222
644
  }
223
645
 
224
- static void sort_destroy(tf_step *self) {
646
+ static int sort_process(tf_step *self, tf_batch *in, tf_batch **out,
647
+ tf_side_channels *side) {
648
+ (void)side;
649
+ sort_state *st = self->state;
650
+ *out = NULL;
651
+
652
+ if (init_schema(st, in) != TF_OK) return TF_ERROR;
653
+ for (size_t r = 0; r < in->n_rows; r++) {
654
+ size_t dst_row = st->buf->n_rows;
655
+ if (tf_batch_copy_row(st->buf, dst_row, in, r) != TF_OK) return TF_ERROR;
656
+ if (tf_batch_expose_row(st->buf, dst_row) != TF_OK) return TF_ERROR;
657
+ if (st->use_spill && st->buf->n_rows >= st->run_rows) {
658
+ if (write_spill_run(st) != TF_OK) return TF_ERROR;
659
+ }
660
+ }
661
+ return TF_OK;
662
+ }
663
+
664
+ static int sort_flush_in_memory(tf_step *self, tf_batch **out) {
225
665
  sort_state *st = self->state;
226
- if (st) {
227
- if (st->buf) tf_batch_free(st->buf);
228
- for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i].name);
229
- free(st->cols);
230
- free(st);
666
+ *out = NULL;
667
+ if (!st->buf || st->buf->n_rows == 0) return TF_OK;
668
+
669
+ size_t n = st->buf->n_rows;
670
+ size_t *indices = sort_batch_indices(st, st->buf);
671
+ if (!indices) return TF_ERROR;
672
+
673
+ tf_batch *ob = create_buffer_from_schema(st, n);
674
+ if (!ob) { free(indices); return TF_ERROR; }
675
+ for (size_t i = 0; i < n; i++) {
676
+ if (tf_batch_copy_row(ob, i, st->buf, indices[i]) != TF_OK) {
677
+ free(indices);
678
+ tf_batch_free(ob);
679
+ return TF_ERROR;
680
+ }
681
+ if (tf_batch_expose_row(ob, i) != TF_OK) {
682
+ free(indices);
683
+ tf_batch_free(ob);
684
+ return TF_ERROR;
685
+ }
231
686
  }
687
+
688
+ free(indices);
689
+ *out = ob;
690
+ return TF_OK;
691
+ }
692
+
693
+ static int sort_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
694
+ (void)side;
695
+ sort_state *st = self->state;
696
+ *out = NULL;
697
+ if (!st->use_spill) return sort_flush_in_memory(self, out);
698
+ return spill_next_batch(st, out);
699
+ }
700
+
701
+ static int sort_flush_next(tf_step *self, tf_batch **out, tf_side_channels *side) {
702
+ (void)side;
703
+ sort_state *st = self->state;
704
+ *out = NULL;
705
+ if (!st->use_spill) return TF_OK;
706
+ return spill_next_batch(st, out);
707
+ }
708
+
709
+ static int sort_append_stats(tf_step *self, tf_buffer *out) {
710
+ if (!self || !out) return TF_ERROR;
711
+ sort_state *st = self->state;
712
+ if (!st || !st->use_spill) return TF_OK;
713
+
714
+ char buf[256];
715
+ snprintf(buf, sizeof(buf),
716
+ ",\"spill_bytes\":%zu,\"spill_runs\":%zu,"
717
+ "\"spill_output_batches\":%zu,\"spill_output_rows\":%zu",
718
+ st->spilled_bytes, st->spill_runs_created,
719
+ st->spill_output_batches, st->spill_output_rows);
720
+ return tf_buffer_write_str(out, buf);
721
+ }
722
+
723
+ static void sort_state_free(sort_state *st) {
724
+ if (!st) return;
725
+ if (st->buf) tf_batch_free(st->buf);
726
+ close_readers(st);
727
+ remove_run_files(st);
728
+ for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i].name);
729
+ for (size_t i = 0; i < st->n_schema_cols; i++) free(st->schema_names ? st->schema_names[i] : NULL);
730
+ free(st->schema_names);
731
+ free(st->schema_types);
732
+ free(st->cols);
733
+ free(st->col_indices);
734
+ free(st->col_desc);
735
+ tf_spill_cleanup(st->spill);
736
+ free(st->spill_dir);
737
+ free(st);
738
+ }
739
+
740
+ static void sort_destroy(tf_step *self) {
741
+ if (self) sort_state_free(self->state);
232
742
  free(self);
233
743
  }
234
744
 
@@ -242,31 +752,67 @@ tf_step *tf_sort_create(const cJSON *args) {
242
752
 
243
753
  sort_state *st = calloc(1, sizeof(sort_state));
244
754
  if (!st) return NULL;
245
- st->cols = calloc(n, sizeof(sort_col));
755
+ st->cols = tf_callocarray_checked((size_t)n, sizeof(sort_col));
246
756
  if (!st->cols) { free(st); return NULL; }
247
- st->n_cols = n;
757
+ st->n_cols = (size_t)n;
758
+ st->output_batch_rows = SORT_DEFAULT_OUTPUT_ROWS;
248
759
 
249
760
  for (int i = 0; i < n; i++) {
250
761
  cJSON *item = cJSON_GetArrayItem(columns, i);
251
762
  cJSON *name_j = cJSON_GetObjectItemCaseSensitive(item, "name");
252
763
  cJSON *desc_j = cJSON_GetObjectItemCaseSensitive(item, "desc");
253
- if (!cJSON_IsString(name_j)) {
254
- for (int j = 0; j < i; j++) free(st->cols[j].name);
255
- free(st->cols); free(st);
764
+ if (!cJSON_IsString(name_j) || !name_j->valuestring || name_j->valuestring[0] == '\0') {
765
+ tf_set_last_error("sort: column names must be non-empty strings");
766
+ sort_state_free(st);
256
767
  return NULL;
257
768
  }
258
769
  st->cols[i].name = strdup(name_j->valuestring);
770
+ if (!st->cols[i].name) {
771
+ sort_state_free(st);
772
+ return NULL;
773
+ }
259
774
  st->cols[i].desc = (desc_j && cJSON_IsBool(desc_j)) ? cJSON_IsTrue(desc_j) : 0;
260
775
  }
261
776
 
262
- tf_step *step = malloc(sizeof(tf_step));
777
+ cJSON *spill_dir_j = cJSON_GetObjectItemCaseSensitive(args, "spill_dir");
778
+ if (cJSON_IsString(spill_dir_j) && spill_dir_j->valuestring && spill_dir_j->valuestring[0]) {
779
+ st->use_spill = 1;
780
+ st->spill_dir = strdup(spill_dir_j->valuestring);
781
+ if (!st->spill_dir) {
782
+ sort_state_free(st);
783
+ return NULL;
784
+ }
785
+ if (tf_spill_session_create(st->spill_dir, &st->spill) != TF_OK) {
786
+ sort_state_free(st);
787
+ return NULL;
788
+ }
789
+ size_t parsed_size = 0;
790
+ int has_spill_memory = tf_json_get_size_arg(args, "spill_memory_bytes",
791
+ 1, TF_MAX_SPILL_MEMORY_BYTES,
792
+ &parsed_size, "sort");
793
+ if (has_spill_memory < 0) { sort_state_free(st); return NULL; }
794
+ if (has_spill_memory > 0) st->spill_memory_bytes = parsed_size;
795
+ int has_spill_rows = tf_json_get_size_arg(args, "spill_run_rows",
796
+ 1, TF_MAX_SPILL_RUN_ROWS,
797
+ &parsed_size, "sort");
798
+ if (has_spill_rows < 0) { sort_state_free(st); return NULL; }
799
+ if (has_spill_rows > 0) st->configured_run_rows = parsed_size;
800
+ int has_output_rows = tf_json_get_size_arg(args, "spill_output_rows",
801
+ 1, TF_MAX_SPILL_OUTPUT_ROWS,
802
+ &parsed_size, "sort");
803
+ if (has_output_rows < 0) { sort_state_free(st); return NULL; }
804
+ if (has_output_rows > 0) st->output_batch_rows = parsed_size;
805
+ }
806
+
807
+ tf_step *step = calloc(1, sizeof(tf_step));
263
808
  if (!step) {
264
- for (int i = 0; i < n; i++) free(st->cols[i].name);
265
- free(st->cols); free(st);
809
+ sort_state_free(st);
266
810
  return NULL;
267
811
  }
268
812
  step->process = sort_process;
269
813
  step->flush = sort_flush;
814
+ step->flush_next = st->use_spill ? sort_flush_next : NULL;
815
+ step->append_stats = sort_append_stats;
270
816
  step->destroy = sort_destroy;
271
817
  step->state = st;
272
818
  return step;