tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/op_pivot.c CHANGED
@@ -1,20 +1,30 @@
1
1
  /*
2
- * op_pivot.c — Pivot (long to wide). Full-load: buffers all data.
2
+ * op_pivot.c - Pivot (long to wide).
3
3
  *
4
- * Config: {"name_column": "metric", "value_column": "value", "agg": "first"}
5
- * Supported aggs: first, sum, count, avg, min, max.
6
- *
7
- * Pass-through columns = all columns except name_column and value_column.
8
- * Output: pass-through columns + one column per unique value of name_column.
4
+ * Default mode buffers input because output columns are data-dependent and
5
+ * groups can reappear anywhere. Guarded sorted mode requires declared
6
+ * categories and consecutive pass-through keys, then emits completed groups
7
+ * with only current-group state.
9
8
  */
10
9
 
11
10
  #include "internal.h"
11
+ #include "spill.h"
12
12
  #include "date_utils.h"
13
13
  #include "cJSON.h"
14
14
  #include <stdlib.h>
15
15
  #include <string.h>
16
16
  #include <stdio.h>
17
17
  #include <float.h>
18
+ #include <errno.h>
19
+ #include <stdint.h>
20
+ #include <unistd.h>
21
+
22
+ #define TF_OK 0
23
+ #define TF_ERROR (-1)
24
+
25
+ #define PIVOT_DEFAULT_RUN_ROWS 8192
26
+ #define PIVOT_DEFAULT_OUTPUT_ROWS 1024
27
+ #define PIVOT_MIN_RUN_ROWS 16
18
28
 
19
29
  typedef enum {
20
30
  PIVOT_FIRST, PIVOT_SUM, PIVOT_COUNT, PIVOT_AVG, PIVOT_MIN, PIVOT_MAX,
@@ -27,63 +37,226 @@ typedef struct {
27
37
  size_t *counts;
28
38
  int *has_first;
29
39
  double *firsts;
40
+ size_t n;
30
41
  } pivot_accum;
31
42
 
32
43
  typedef struct {
33
44
  char **keys;
34
- size_t *pass_rows; /* first row index for each group (for pass-through values) */
45
+ size_t *pass_rows;
35
46
  pivot_accum *accums;
36
47
  size_t count;
37
48
  size_t cap;
38
49
  } pivot_map;
39
50
 
51
+ typedef struct {
52
+ uint64_t ordinal;
53
+ uint8_t *nulls;
54
+ tf_owned_cell_value *cells;
55
+ } pivot_spill_row;
56
+
57
+ typedef struct {
58
+ FILE *file;
59
+ pivot_spill_row row;
60
+ int has_row;
61
+ int done;
62
+ } pivot_run_reader;
63
+
40
64
  typedef struct {
41
65
  char *name_column;
42
66
  char *value_column;
43
67
  pivot_agg agg;
68
+ int sorted;
69
+ size_t max_categories;
70
+ int categories_declared;
71
+ int use_spill;
72
+
73
+ char *spill_dir;
74
+ tf_spill_session *spill;
75
+ size_t spill_memory_bytes;
76
+ size_t configured_run_rows;
77
+ size_t run_rows;
78
+ size_t output_batch_rows;
44
79
 
45
80
  tf_batch *buf;
46
81
  int has_schema;
82
+ char **schema_names;
83
+ tf_type *schema_types;
84
+ size_t n_schema_cols;
85
+ uint64_t *buf_ordinals;
86
+ size_t buf_ordinal_cap;
87
+ uint64_t next_ordinal;
47
88
 
48
89
  char **unique_names;
49
90
  size_t n_names;
50
91
  size_t names_cap;
92
+
93
+ int *pt_cols;
94
+ size_t n_pt;
95
+ int name_ci;
96
+ int val_ci;
97
+ tf_batch *current_pt;
98
+ char *current_key;
99
+ pivot_accum current_accum;
100
+ int have_current;
101
+
102
+ tf_batch *out_buf;
103
+ uint64_t *out_ordinals;
104
+ size_t out_ordinal_cap;
105
+ char **run_paths;
106
+ size_t n_runs;
107
+ size_t cap_runs;
108
+ char **out_run_paths;
109
+ size_t n_out_runs;
110
+ size_t cap_out_runs;
111
+ size_t run_seq;
112
+ size_t out_run_seq;
113
+ pivot_run_reader *readers;
114
+ size_t n_readers;
115
+ pivot_run_reader *out_readers;
116
+ size_t n_out_readers;
117
+ int key_merge_done;
118
+ int output_merge_started;
119
+ int output_merge_done;
120
+
121
+ size_t spilled_bytes;
122
+ size_t spill_runs_created;
123
+ size_t spill_output_batches;
124
+ size_t spill_output_rows;
125
+ size_t spill_distinct_groups;
126
+ size_t spill_key_bytes;
51
127
  } pivot_state;
52
128
 
129
+ static int pivot_resolve_name(pivot_state *st, const char *name,
130
+ tf_side_channels *side);
131
+
132
+ TF_WARN_UNUSED static int pivot_write_error(tf_side_channels *side, const char *msg) {
133
+ return tf_side_write_error(side, msg);
134
+ }
135
+
53
136
  static pivot_agg parse_pivot_agg(const char *s) {
54
137
  if (!s) return PIVOT_FIRST;
55
138
  if (strcmp(s, "first") == 0) return PIVOT_FIRST;
56
139
  if (strcmp(s, "sum") == 0) return PIVOT_SUM;
57
140
  if (strcmp(s, "count") == 0) return PIVOT_COUNT;
58
- if (strcmp(s, "avg") == 0) return PIVOT_AVG;
141
+ if (strcmp(s, "avg") == 0 || strcmp(s, "mean") == 0) return PIVOT_AVG;
59
142
  if (strcmp(s, "min") == 0) return PIVOT_MIN;
60
143
  if (strcmp(s, "max") == 0) return PIVOT_MAX;
61
144
  return PIVOT_FIRST;
62
145
  }
63
146
 
64
- static int find_unique_name(pivot_state *st, const char *name) {
147
+ static int pivot_accum_init(pivot_accum *a, size_t n) {
148
+ memset(a, 0, sizeof(*a));
149
+ if (n == 0) return TF_OK;
150
+
151
+ double *sums = tf_callocarray_checked(n, sizeof(double));
152
+ double *mins = tf_mallocarray_checked(n, sizeof(double));
153
+ double *maxs = tf_mallocarray_checked(n, sizeof(double));
154
+ size_t *counts = tf_callocarray_checked(n, sizeof(size_t));
155
+ int *has_first = tf_callocarray_checked(n, sizeof(int));
156
+ double *firsts = tf_callocarray_checked(n, sizeof(double));
157
+ if (!sums || !mins || !maxs || !counts || !has_first || !firsts) {
158
+ free(sums);
159
+ free(mins);
160
+ free(maxs);
161
+ free(counts);
162
+ free(has_first);
163
+ free(firsts);
164
+ return TF_ERROR;
165
+ }
166
+
167
+ a->sums = sums;
168
+ a->mins = mins;
169
+ a->maxs = maxs;
170
+ a->counts = counts;
171
+ a->has_first = has_first;
172
+ a->firsts = firsts;
173
+ a->n = n;
174
+ for (size_t i = 0; i < n; i++) {
175
+ a->mins[i] = DBL_MAX;
176
+ a->maxs[i] = -DBL_MAX;
177
+ }
178
+ return TF_OK;
179
+ }
180
+
181
+ static void pivot_accum_reset(pivot_accum *a) {
182
+ if (!a || a->n == 0) return;
183
+ memset(a->sums, 0, a->n * sizeof(double));
184
+ memset(a->counts, 0, a->n * sizeof(size_t));
185
+ memset(a->has_first, 0, a->n * sizeof(int));
186
+ memset(a->firsts, 0, a->n * sizeof(double));
187
+ for (size_t i = 0; i < a->n; i++) {
188
+ a->mins[i] = DBL_MAX;
189
+ a->maxs[i] = -DBL_MAX;
190
+ }
191
+ }
192
+
193
+ static void pivot_accum_free(pivot_accum *a) {
194
+ if (!a) return;
195
+ free(a->sums);
196
+ free(a->mins);
197
+ free(a->maxs);
198
+ free(a->counts);
199
+ free(a->has_first);
200
+ free(a->firsts);
201
+ memset(a, 0, sizeof(*a));
202
+ }
203
+
204
+ static void pivot_accum_add(pivot_accum *a, size_t idx, double v) {
205
+ if (!a || idx >= a->n) return;
206
+ a->sums[idx] += v;
207
+ if (v < a->mins[idx]) a->mins[idx] = v;
208
+ if (v > a->maxs[idx]) a->maxs[idx] = v;
209
+ a->counts[idx]++;
210
+ if (!a->has_first[idx]) {
211
+ a->firsts[idx] = v;
212
+ a->has_first[idx] = 1;
213
+ }
214
+ }
215
+
216
+ static int find_unique_name(const pivot_state *st, const char *name) {
65
217
  for (size_t i = 0; i < st->n_names; i++) {
66
218
  if (strcmp(st->unique_names[i], name) == 0) return (int)i;
67
219
  }
68
220
  return -1;
69
221
  }
70
222
 
71
- static int add_unique_name(pivot_state *st, const char *name) {
223
+ static int add_unique_name(pivot_state *st, const char *name, tf_side_channels *side) {
72
224
  int idx = find_unique_name(st, name);
73
225
  if (idx >= 0) return idx;
226
+ if (st->max_categories > 0 && st->n_names >= st->max_categories) {
227
+ char msg[256];
228
+ snprintf(msg, sizeof(msg), "pivot: max_categories=%zu exceeded while tracking category '%s'",
229
+ st->max_categories, name ? name : "");
230
+ if (pivot_write_error(side, msg) != TF_OK) return -1;
231
+ return -1;
232
+ }
74
233
  if (st->n_names >= st->names_cap) {
75
- size_t new_cap = st->names_cap ? st->names_cap * 2 : 32;
76
- st->unique_names = realloc(st->unique_names, new_cap * sizeof(char *));
77
- if (!st->unique_names) return -1;
234
+ size_t need = 0;
235
+ size_t new_cap = 0;
236
+ if (tf_size_add(st->n_names, 1, &need) != TF_OK ||
237
+ tf_size_grow_pow2(st->names_cap, need, 32, &new_cap) != TF_OK) {
238
+ return -1;
239
+ }
240
+ char **tmp = tf_reallocarray_checked(st->unique_names, new_cap, sizeof(char *));
241
+ if (!tmp) return -1;
242
+ st->unique_names = tmp;
78
243
  st->names_cap = new_cap;
79
244
  }
80
- st->unique_names[st->n_names] = strdup(name);
245
+ st->unique_names[st->n_names] = strdup(name ? name : "");
246
+ if (!st->unique_names[st->n_names]) return -1;
81
247
  return (int)st->n_names++;
82
248
  }
83
249
 
84
- /* Build group key from pass-through columns */
250
+ static int unknown_declared_category(pivot_state *st, const char *name, tf_side_channels *side) {
251
+ char msg[256];
252
+ snprintf(msg, sizeof(msg), "pivot: unknown category '%s' for column '%s'",
253
+ name ? name : "", st->name_column ? st->name_column : "");
254
+ if (pivot_write_error(side, msg) != TF_OK) return TF_ERROR;
255
+ return TF_ERROR;
256
+ }
257
+
85
258
  static char *build_pivot_key(const tf_batch *b, size_t row,
86
- int *pt_cols, size_t n_pt) {
259
+ const int *pt_cols, size_t n_pt) {
87
260
  size_t buf_cap = 256;
88
261
  char *buf = malloc(buf_cap);
89
262
  if (!buf) return NULL;
@@ -94,29 +267,62 @@ static char *build_pivot_key(const tf_batch *b, size_t row,
94
267
  char val_buf[64];
95
268
  const char *val = "";
96
269
  size_t val_len = 0;
97
- if (tf_batch_is_null(b, row, c)) {
98
- val = "\\N"; val_len = 2;
270
+ if (tf_batch_is_null(b, row, (size_t)c)) {
271
+ val = "\\N";
272
+ val_len = 2;
99
273
  } else {
100
274
  switch (b->col_types[c]) {
101
- case TF_TYPE_STRING: val = tf_batch_get_string(b, row, c); val_len = strlen(val); break;
275
+ case TF_TYPE_STRING:
276
+ val = tf_batch_get_string(b, row, (size_t)c);
277
+ val_len = strlen(val);
278
+ break;
102
279
  case TF_TYPE_INT64:
103
- val_len = snprintf(val_buf, sizeof(val_buf), "%lld", (long long)tf_batch_get_int64(b, row, c));
104
- val = val_buf; break;
280
+ val_len = snprintf(val_buf, sizeof(val_buf), "%lld",
281
+ (long long)tf_batch_get_int64(b, row, (size_t)c));
282
+ val = val_buf;
283
+ break;
105
284
  case TF_TYPE_FLOAT64:
106
- val_len = snprintf(val_buf, sizeof(val_buf), "%.17g", tf_batch_get_float64(b, row, c));
107
- val = val_buf; break;
285
+ val_len = snprintf(val_buf, sizeof(val_buf), "%.17g", tf_batch_get_float64(b, row, (size_t)c));
286
+ val = val_buf;
287
+ break;
108
288
  case TF_TYPE_BOOL:
109
- val = tf_batch_get_bool(b, row, c) ? "T" : "F"; val_len = 1; break;
289
+ val = tf_batch_get_bool(b, row, (size_t)c) ? "T" : "F";
290
+ val_len = 1;
291
+ break;
110
292
  case TF_TYPE_DATE:
111
- val_len = snprintf(val_buf, sizeof(val_buf), "%d", (int)tf_batch_get_date(b, row, c));
112
- val = val_buf; break;
293
+ val_len = snprintf(val_buf, sizeof(val_buf), "%d", (int)tf_batch_get_date(b, row, (size_t)c));
294
+ val = val_buf;
295
+ break;
113
296
  case TF_TYPE_TIMESTAMP:
114
- val_len = snprintf(val_buf, sizeof(val_buf), "%lld", (long long)tf_batch_get_timestamp(b, row, c));
115
- val = val_buf; break;
116
- default: val = "\\N"; val_len = 2; break;
297
+ val_len = snprintf(val_buf, sizeof(val_buf), "%lld",
298
+ (long long)tf_batch_get_timestamp(b, row, (size_t)c));
299
+ val = val_buf;
300
+ break;
301
+ default:
302
+ val = "\\N";
303
+ val_len = 2;
304
+ break;
305
+ }
306
+ }
307
+ size_t need = 0;
308
+ if (tf_size_add(buf_len, val_len, &need) != TF_OK ||
309
+ tf_size_add(need, 2, &need) != TF_OK) {
310
+ free(buf);
311
+ return NULL;
312
+ }
313
+ if (need >= buf_cap) {
314
+ size_t new_cap = 0;
315
+ size_t min_cap = 0;
316
+ if (tf_size_add(need, 1, &min_cap) != TF_OK ||
317
+ tf_size_grow_pow2(buf_cap, min_cap, 256, &new_cap) != TF_OK) {
318
+ free(buf);
319
+ return NULL;
117
320
  }
321
+ char *tmp = tf_reallocarray_checked(buf, new_cap, sizeof(char));
322
+ if (!tmp) { free(buf); return NULL; }
323
+ buf = tmp;
324
+ buf_cap = new_cap;
118
325
  }
119
- while (buf_len + val_len + 2 >= buf_cap) { buf_cap *= 2; buf = realloc(buf, buf_cap); }
120
326
  memcpy(buf + buf_len, val, val_len);
121
327
  buf_len += val_len;
122
328
  }
@@ -125,92 +331,1221 @@ static char *build_pivot_key(const tf_batch *b, size_t row,
125
331
  }
126
332
 
127
333
  static int find_or_add_pivot_group(pivot_map *map, const char *key,
128
- size_t n_names, size_t src_row) {
334
+ size_t n_names, size_t src_row) {
129
335
  for (size_t i = 0; i < map->count; i++) {
130
336
  if (strcmp(map->keys[i], key) == 0) return (int)i;
131
337
  }
132
338
  if (map->count >= map->cap) {
133
- size_t new_cap = map->cap ? map->cap * 2 : 64;
134
- map->keys = realloc(map->keys, new_cap * sizeof(char *));
135
- map->pass_rows = realloc(map->pass_rows, new_cap * sizeof(size_t));
136
- map->accums = realloc(map->accums, new_cap * sizeof(pivot_accum));
339
+ size_t need = 0;
340
+ size_t new_cap = 0;
341
+ if (tf_size_add(map->count, 1, &need) != TF_OK ||
342
+ tf_size_grow_pow2(map->cap, need, 64, &new_cap) != TF_OK) {
343
+ return -1;
344
+ }
345
+ size_t keys_bytes = 0;
346
+ size_t pass_rows_bytes = 0;
347
+ size_t accums_bytes = 0;
348
+ if (map->count > 0 &&
349
+ (tf_size_mul(map->count, sizeof(char *), &keys_bytes) != TF_OK ||
350
+ tf_size_mul(map->count, sizeof(size_t), &pass_rows_bytes) != TF_OK ||
351
+ tf_size_mul(map->count, sizeof(pivot_accum), &accums_bytes) != TF_OK)) {
352
+ return -1;
353
+ }
354
+ char **keys = tf_mallocarray_checked(new_cap, sizeof(char *));
355
+ size_t *pass_rows = tf_mallocarray_checked(new_cap, sizeof(size_t));
356
+ pivot_accum *accums = tf_mallocarray_checked(new_cap, sizeof(pivot_accum));
357
+ if (!keys || !pass_rows || !accums) {
358
+ free(keys);
359
+ free(pass_rows);
360
+ free(accums);
361
+ return -1;
362
+ }
363
+ if (map->count > 0) {
364
+ memcpy(keys, map->keys, keys_bytes);
365
+ memcpy(pass_rows, map->pass_rows, pass_rows_bytes);
366
+ memcpy(accums, map->accums, accums_bytes);
367
+ }
368
+ free(map->keys);
369
+ free(map->pass_rows);
370
+ free(map->accums);
371
+ map->keys = keys;
372
+ map->pass_rows = pass_rows;
373
+ map->accums = accums;
137
374
  map->cap = new_cap;
138
375
  }
139
- size_t idx = map->count++;
376
+ size_t idx = map->count;
140
377
  map->keys[idx] = strdup(key);
378
+ if (!map->keys[idx]) return -1;
141
379
  map->pass_rows[idx] = src_row;
142
- pivot_accum *a = &map->accums[idx];
143
- a->sums = calloc(n_names, sizeof(double));
144
- a->mins = malloc(n_names * sizeof(double));
145
- a->maxs = malloc(n_names * sizeof(double));
146
- a->counts = calloc(n_names, sizeof(size_t));
147
- a->has_first = calloc(n_names, sizeof(int));
148
- a->firsts = calloc(n_names, sizeof(double));
149
- for (size_t i = 0; i < n_names; i++) { a->mins[i] = DBL_MAX; a->maxs[i] = -DBL_MAX; }
380
+ if (pivot_accum_init(&map->accums[idx], n_names) != TF_OK) {
381
+ free(map->keys[idx]);
382
+ map->keys[idx] = NULL;
383
+ return -1;
384
+ }
385
+ map->count = idx + 1;
150
386
  return (int)idx;
151
387
  }
152
388
 
389
+ static void pivot_map_free(pivot_map *map) {
390
+ if (!map) return;
391
+ for (size_t i = 0; i < map->count; i++) {
392
+ free(map->keys[i]);
393
+ pivot_accum_free(&map->accums[i]);
394
+ }
395
+ free(map->keys);
396
+ free(map->pass_rows);
397
+ free(map->accums);
398
+ memset(map, 0, sizeof(*map));
399
+ }
400
+
153
401
  static double get_numeric_value(const tf_batch *b, size_t row, int col) {
154
- if (tf_batch_is_null(b, row, col)) return 0.0;
402
+ if (tf_batch_is_null(b, row, (size_t)col)) return 0.0;
155
403
  switch (b->col_types[col]) {
156
- case TF_TYPE_INT64: return (double)tf_batch_get_int64(b, row, col);
157
- case TF_TYPE_FLOAT64: return tf_batch_get_float64(b, row, col);
158
- case TF_TYPE_DATE: return (double)tf_batch_get_date(b, row, col);
159
- case TF_TYPE_TIMESTAMP: return (double)tf_batch_get_timestamp(b, row, col);
160
- case TF_TYPE_BOOL: return tf_batch_get_bool(b, row, col) ? 1.0 : 0.0;
404
+ case TF_TYPE_INT64: return (double)tf_batch_get_int64(b, row, (size_t)col);
405
+ case TF_TYPE_FLOAT64: return tf_batch_get_float64(b, row, (size_t)col);
406
+ case TF_TYPE_DATE: return (double)tf_batch_get_date(b, row, (size_t)col);
407
+ case TF_TYPE_TIMESTAMP: return (double)tf_batch_get_timestamp(b, row, (size_t)col);
408
+ case TF_TYPE_BOOL: return tf_batch_get_bool(b, row, (size_t)col) ? 1.0 : 0.0;
161
409
  default: return 0.0;
162
410
  }
163
411
  }
164
412
 
165
- /* Get name column value as string */
166
413
  static const char *get_name_str(const tf_batch *b, size_t row, int col, char *buf, size_t buf_sz) {
167
- if (tf_batch_is_null(b, row, col)) return NULL;
414
+ if (tf_batch_is_null(b, row, (size_t)col)) return NULL;
168
415
  switch (b->col_types[col]) {
169
- case TF_TYPE_STRING: return tf_batch_get_string(b, row, col);
170
- case TF_TYPE_INT64: snprintf(buf, buf_sz, "%lld", (long long)tf_batch_get_int64(b, row, col)); return buf;
171
- case TF_TYPE_FLOAT64: snprintf(buf, buf_sz, "%g", tf_batch_get_float64(b, row, col)); return buf;
172
- case TF_TYPE_BOOL: return tf_batch_get_bool(b, row, col) ? "true" : "false";
173
- case TF_TYPE_DATE: tf_date_format(tf_batch_get_date(b, row, col), buf, buf_sz); return buf;
174
- case TF_TYPE_TIMESTAMP: tf_timestamp_format(tf_batch_get_timestamp(b, row, col), buf, buf_sz); return buf;
416
+ case TF_TYPE_STRING: return tf_batch_get_string(b, row, (size_t)col);
417
+ case TF_TYPE_INT64:
418
+ snprintf(buf, buf_sz, "%lld", (long long)tf_batch_get_int64(b, row, (size_t)col));
419
+ return buf;
420
+ case TF_TYPE_FLOAT64:
421
+ snprintf(buf, buf_sz, "%.17g", tf_batch_get_float64(b, row, (size_t)col));
422
+ return buf;
423
+ case TF_TYPE_BOOL:
424
+ return tf_batch_get_bool(b, row, (size_t)col) ? "true" : "false";
425
+ case TF_TYPE_DATE:
426
+ tf_date_format(tf_batch_get_date(b, row, (size_t)col), buf, buf_sz);
427
+ return buf;
428
+ case TF_TYPE_TIMESTAMP:
429
+ tf_timestamp_format(tf_batch_get_timestamp(b, row, (size_t)col), buf, buf_sz);
430
+ return buf;
175
431
  default: return NULL;
176
432
  }
177
433
  }
178
434
 
435
+ static int pivot_set_output_schema_from_source(const pivot_state *st, tf_batch *ob,
436
+ const tf_batch *src,
437
+ const int *pt_cols, size_t n_pt) {
438
+ for (size_t k = 0; k < n_pt; k++) {
439
+ int sc = pt_cols ? pt_cols[k] : (int)k;
440
+ if (tf_batch_set_schema(ob, k, src->col_names[sc], src->col_types[sc]) != TF_OK)
441
+ return TF_ERROR;
442
+ }
443
+ tf_type pivot_type = st->agg == PIVOT_COUNT ? TF_TYPE_INT64 : TF_TYPE_FLOAT64;
444
+ for (size_t k = 0; k < st->n_names; k++) {
445
+ if (tf_batch_set_schema(ob, n_pt + k, st->unique_names[k], pivot_type) != TF_OK)
446
+ return TF_ERROR;
447
+ }
448
+ return TF_OK;
449
+ }
450
+
451
+ static int pivot_emit_row(const pivot_state *st, tf_batch *ob, size_t out_row,
452
+ const tf_batch *pt_src, size_t pt_row,
453
+ const int *pt_cols, size_t n_pt,
454
+ const pivot_accum *a) {
455
+ if (tf_batch_ensure_capacity(ob, out_row + 1) != TF_OK) return TF_ERROR;
456
+ for (size_t k = 0; k < n_pt; k++) {
457
+ int sc = pt_cols ? pt_cols[k] : (int)k;
458
+ if (tf_batch_copy_cell_index(ob, out_row, k, pt_src, pt_row, sc) != TF_OK) return TF_ERROR;
459
+ }
460
+ for (size_t k = 0; k < st->n_names; k++) {
461
+ size_t oc = n_pt + k;
462
+ if (a->counts[k] == 0) {
463
+ if (tf_batch_set_null(ob, out_row, oc) != TF_OK) return TF_ERROR;
464
+ continue;
465
+ }
466
+ double v = 0.0;
467
+ switch (st->agg) {
468
+ case PIVOT_FIRST: v = a->firsts[k]; break;
469
+ case PIVOT_SUM: v = a->sums[k]; break;
470
+ case PIVOT_COUNT:
471
+ if (tf_batch_set_int64(ob, out_row, oc, (int64_t)a->counts[k]) != TF_OK) return TF_ERROR;
472
+ continue;
473
+ case PIVOT_AVG: v = a->sums[k] / (double)a->counts[k]; break;
474
+ case PIVOT_MIN: v = a->mins[k]; break;
475
+ case PIVOT_MAX: v = a->maxs[k]; break;
476
+ }
477
+ if (tf_batch_set_float64(ob, out_row, oc, v) != TF_OK) return TF_ERROR;
478
+ }
479
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) return TF_ERROR;
480
+ return TF_OK;
481
+ }
482
+
483
+
484
+ static int pivot_set_output_schema_from_arrays(const pivot_state *st, tf_batch *ob) {
485
+ for (size_t k = 0; k < st->n_pt; k++) {
486
+ int sc = st->pt_cols ? st->pt_cols[k] : (int)k;
487
+ if (tf_batch_set_schema(ob, k, st->schema_names[sc], st->schema_types[sc]) != TF_OK)
488
+ return TF_ERROR;
489
+ }
490
+ tf_type pivot_type = st->agg == PIVOT_COUNT ? TF_TYPE_INT64 : TF_TYPE_FLOAT64;
491
+ for (size_t k = 0; k < st->n_names; k++) {
492
+ if (tf_batch_set_schema(ob, st->n_pt + k, st->unique_names[k], pivot_type) != TF_OK)
493
+ return TF_ERROR;
494
+ }
495
+ return TF_OK;
496
+ }
497
+
498
+ static int pivot_key_append(char **buf, size_t *buf_cap, size_t *buf_len,
499
+ const char *data, size_t data_len) {
500
+ size_t need = 0;
501
+ size_t min_cap = 0;
502
+ if (tf_size_add(*buf_len, data_len, &need) != TF_OK ||
503
+ tf_size_add(need, 1, &min_cap) != TF_OK) {
504
+ return TF_ERROR;
505
+ }
506
+ if (min_cap >= *buf_cap) {
507
+ size_t new_cap = 0;
508
+ if (tf_size_grow_pow2(*buf_cap, min_cap, 256, &new_cap) != TF_OK) return TF_ERROR;
509
+ char *tmp = tf_reallocarray_checked(*buf, new_cap, sizeof(char));
510
+ if (!tmp) return TF_ERROR;
511
+ *buf = tmp;
512
+ *buf_cap = new_cap;
513
+ }
514
+ memcpy(*buf + *buf_len, data, data_len);
515
+ *buf_len += data_len;
516
+ (*buf)[*buf_len] = '\0';
517
+ return TF_OK;
518
+ }
519
+
520
+ static int pivot_key_append_cstr(char **buf, size_t *buf_cap, size_t *buf_len,
521
+ const char *s) {
522
+ return pivot_key_append(buf, buf_cap, buf_len, s, strlen(s));
523
+ }
524
+
525
+ static int pivot_key_append_size(char **buf, size_t *buf_cap, size_t *buf_len,
526
+ size_t value) {
527
+ char tmp[32];
528
+ int n = snprintf(tmp, sizeof(tmp), "%zu", value);
529
+ if (n < 0 || (size_t)n >= sizeof(tmp)) return TF_ERROR;
530
+ return pivot_key_append(buf, buf_cap, buf_len, tmp, (size_t)n);
531
+ }
532
+
533
+ static int pivot_ensure_ordinals(uint64_t **ord, size_t *cap, size_t need) {
534
+ if (*cap >= need) return TF_OK;
535
+ size_t new_cap = 0;
536
+ if (tf_size_grow_pow2(*cap, need, 64, &new_cap) != TF_OK) return TF_ERROR;
537
+ uint64_t *tmp = tf_reallocarray_checked(*ord, new_cap, sizeof(uint64_t));
538
+ if (!tmp) return TF_ERROR;
539
+ *ord = tmp;
540
+ *cap = new_cap;
541
+ return TF_OK;
542
+ }
543
+
544
+ static int pivot_write_exact(FILE *f, const void *ptr, size_t len) {
545
+ return fwrite(ptr, 1, len, f) == len ? TF_OK : TF_ERROR;
546
+ }
547
+
548
+ static int pivot_read_exact(FILE *f, void *ptr, size_t len) {
549
+ return fread(ptr, 1, len, f) == len ? TF_OK : TF_ERROR;
550
+ }
551
+
552
+ static int pivot_write_cell(FILE *f, const tf_batch *b, size_t r, size_t c) {
553
+ uint8_t is_null = tf_batch_is_null(b, r, c) ? 1 : 0;
554
+ if (pivot_write_exact(f, &is_null, sizeof(is_null)) != TF_OK) return TF_ERROR;
555
+ if (is_null) return TF_OK;
556
+ switch (b->col_types[c]) {
557
+ case TF_TYPE_BOOL: { uint8_t v = tf_batch_get_bool(b, r, c) ? 1 : 0; return pivot_write_exact(f, &v, sizeof(v)); }
558
+ case TF_TYPE_INT64: { int64_t v = tf_batch_get_int64(b, r, c); return pivot_write_exact(f, &v, sizeof(v)); }
559
+ case TF_TYPE_FLOAT64: { double v = tf_batch_get_float64(b, r, c); return pivot_write_exact(f, &v, sizeof(v)); }
560
+ case TF_TYPE_STRING: {
561
+ const char *str = tf_batch_get_string(b, r, c);
562
+ uint64_t len = str ? (uint64_t)strlen(str) : 0;
563
+ if (pivot_write_exact(f, &len, sizeof(len)) != TF_OK) return TF_ERROR;
564
+ return len ? pivot_write_exact(f, str, (size_t)len) : TF_OK;
565
+ }
566
+ case TF_TYPE_DATE: { int32_t v = tf_batch_get_date(b, r, c); return pivot_write_exact(f, &v, sizeof(v)); }
567
+ case TF_TYPE_TIMESTAMP: { int64_t v = tf_batch_get_timestamp(b, r, c); return pivot_write_exact(f, &v, sizeof(v)); }
568
+ default: return TF_OK;
569
+ }
570
+ }
571
+
572
+ static int pivot_read_cell_value(FILE *f, pivot_spill_row *row, tf_type type, size_t c) {
573
+ switch (type) {
574
+ case TF_TYPE_BOOL: { uint8_t v = 0; if (pivot_read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].b = v; return TF_OK; }
575
+ case TF_TYPE_INT64: { int64_t v = 0; if (pivot_read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].i64 = v; return TF_OK; }
576
+ case TF_TYPE_FLOAT64: { double v = 0; if (pivot_read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].f64 = v; return TF_OK; }
577
+ case TF_TYPE_STRING: {
578
+ uint64_t len = 0;
579
+ if (pivot_read_exact(f, &len, sizeof(len)) != TF_OK) return TF_ERROR;
580
+ if (len > (uint64_t)SIZE_MAX - 1) return TF_ERROR;
581
+ char *str = malloc((size_t)len + 1);
582
+ if (!str) return TF_ERROR;
583
+ if (len && pivot_read_exact(f, str, (size_t)len) != TF_OK) { free(str); return TF_ERROR; }
584
+ str[len] = '\0';
585
+ row->cells[c].str = str;
586
+ return TF_OK;
587
+ }
588
+ case TF_TYPE_DATE: { int32_t v = 0; if (pivot_read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].date = v; return TF_OK; }
589
+ case TF_TYPE_TIMESTAMP: { int64_t v = 0; if (pivot_read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].i64 = v; return TF_OK; }
590
+ default: return TF_OK;
591
+ }
592
+ }
593
+
594
+ static int pivot_spill_row_init(pivot_spill_row *row, size_t n_cols) {
595
+ row->ordinal = 0;
596
+ row->nulls = tf_callocarray_checked(n_cols ? n_cols : 1, sizeof(uint8_t));
597
+ row->cells = tf_callocarray_checked(n_cols ? n_cols : 1, sizeof(*row->cells));
598
+ if (!row->nulls || !row->cells) {
599
+ free(row->nulls);
600
+ free(row->cells);
601
+ row->nulls = NULL;
602
+ row->cells = NULL;
603
+ return TF_ERROR;
604
+ }
605
+ for (size_t c = 0; c < n_cols; c++) row->nulls[c] = 1;
606
+ return TF_OK;
607
+ }
608
+
609
+ static size_t pivot_reader_cols(const pivot_state *st, int output_reader) {
610
+ return output_reader ? (st->n_pt + st->n_names) : st->n_schema_cols;
611
+ }
612
+
613
+ static tf_type pivot_output_col_type(const pivot_state *st, size_t c) {
614
+ if (c < st->n_pt) {
615
+ int sc = st->pt_cols ? st->pt_cols[c] : (int)c;
616
+ return st->schema_types[sc];
617
+ }
618
+ return st->agg == PIVOT_COUNT ? TF_TYPE_INT64 : TF_TYPE_FLOAT64;
619
+ }
620
+
621
+ static tf_type pivot_reader_col_type(const pivot_state *st, int output_reader, size_t c) {
622
+ return output_reader ? pivot_output_col_type(st, c) : st->schema_types[c];
623
+ }
624
+
625
+ static void pivot_spill_row_clear_reader(const pivot_state *st, pivot_spill_row *row,
626
+ int output_reader) {
627
+ if (!row || !row->cells || !row->nulls) return;
628
+ size_t n_cols = pivot_reader_cols(st, output_reader);
629
+ for (size_t c = 0; c < n_cols; c++) {
630
+ if (!row->nulls[c] && pivot_reader_col_type(st, output_reader, c) == TF_TYPE_STRING) {
631
+ free(row->cells[c].str);
632
+ }
633
+ row->cells[c].str = NULL;
634
+ row->nulls[c] = 1;
635
+ }
636
+ }
637
+
638
+ static void pivot_spill_row_free_reader(const pivot_state *st, pivot_spill_row *row,
639
+ int output_reader) {
640
+ if (!row) return;
641
+ pivot_spill_row_clear_reader(st, row, output_reader);
642
+ free(row->nulls);
643
+ free(row->cells);
644
+ row->nulls = NULL;
645
+ row->cells = NULL;
646
+ }
647
+
648
+ static int pivot_reader_advance(pivot_state *st, pivot_run_reader *reader, int output_reader) {
649
+ if (!reader || !reader->file || reader->done) return 0;
650
+ size_t n_cols = pivot_reader_cols(st, output_reader);
651
+ pivot_spill_row_clear_reader(st, &reader->row, output_reader);
652
+ if (fread(&reader->row.ordinal, sizeof(reader->row.ordinal), 1, reader->file) != 1) {
653
+ if (feof(reader->file)) {
654
+ reader->done = 1;
655
+ reader->has_row = 0;
656
+ return 0;
657
+ }
658
+ tf_set_last_error("pivot spill: failed reading run file");
659
+ return -1;
660
+ }
661
+ for (size_t c = 0; c < n_cols; c++) {
662
+ uint8_t is_null = 1;
663
+ if (pivot_read_exact(reader->file, &is_null, sizeof(is_null)) != TF_OK) {
664
+ tf_set_last_error("pivot spill: corrupt run file");
665
+ return -1;
666
+ }
667
+ reader->row.nulls[c] = is_null ? 1 : 0;
668
+ tf_type type = pivot_reader_col_type(st, output_reader, c);
669
+ if (!reader->row.nulls[c] && pivot_read_cell_value(reader->file, &reader->row, type, c) != TF_OK) {
670
+ tf_set_last_error("pivot spill: corrupt run file");
671
+ return -1;
672
+ }
673
+ }
674
+ reader->has_row = 1;
675
+ return 1;
676
+ }
677
+
678
+ static void pivot_close_readers(pivot_state *st, int output_readers) {
679
+ pivot_run_reader **readers = output_readers ? &st->out_readers : &st->readers;
680
+ size_t *n_readers = output_readers ? &st->n_out_readers : &st->n_readers;
681
+ if (!*readers) return;
682
+ for (size_t i = 0; i < *n_readers; i++) {
683
+ if ((*readers)[i].file) fclose((*readers)[i].file);
684
+ pivot_spill_row_free_reader(st, &(*readers)[i].row, output_readers);
685
+ }
686
+ free(*readers);
687
+ *readers = NULL;
688
+ *n_readers = 0;
689
+ }
690
+
691
+ static void pivot_remove_paths(char ***paths, size_t *n, size_t *cap) {
692
+ for (size_t i = 0; i < *n; i++) {
693
+ if ((*paths)[i]) {
694
+ remove((*paths)[i]);
695
+ free((*paths)[i]);
696
+ (*paths)[i] = NULL;
697
+ }
698
+ }
699
+ free(*paths);
700
+ *paths = NULL;
701
+ *n = 0;
702
+ *cap = 0;
703
+ }
704
+
705
+ static int pivot_append_path(char ***paths, size_t *n, size_t *cap, char *path) {
706
+ if (*n == *cap) {
707
+ size_t need = 0;
708
+ size_t new_cap = 0;
709
+ if (tf_size_add(*n, 1, &need) != TF_OK ||
710
+ tf_size_grow_pow2(*cap, need, 8, &new_cap) != TF_OK) {
711
+ return TF_ERROR;
712
+ }
713
+ char **tmp = tf_reallocarray_checked(*paths, new_cap, sizeof(char *));
714
+ if (!tmp) return TF_ERROR;
715
+ *paths = tmp;
716
+ *cap = new_cap;
717
+ }
718
+ (*paths)[(*n)++] = path;
719
+ return TF_OK;
720
+ }
721
+
722
+ static const char *pivot_spill_name_str(const pivot_state *st, const pivot_spill_row *row,
723
+ char *buf, size_t buf_sz) {
724
+ if (st->name_ci < 0 || row->nulls[(size_t)st->name_ci]) return NULL;
725
+ size_t c = (size_t)st->name_ci;
726
+ switch (st->schema_types[c]) {
727
+ case TF_TYPE_STRING: return row->cells[c].str ? row->cells[c].str : "";
728
+ case TF_TYPE_INT64: snprintf(buf, buf_sz, "%lld", (long long)row->cells[c].i64); return buf;
729
+ case TF_TYPE_FLOAT64: snprintf(buf, buf_sz, "%.17g", row->cells[c].f64); return buf;
730
+ case TF_TYPE_BOOL: return row->cells[c].b ? "true" : "false";
731
+ case TF_TYPE_DATE: tf_date_format(row->cells[c].date, buf, buf_sz); return buf;
732
+ case TF_TYPE_TIMESTAMP: tf_timestamp_format(row->cells[c].i64, buf, buf_sz); return buf;
733
+ default: return NULL;
734
+ }
735
+ }
736
+
737
+ static double pivot_spill_numeric_value(const pivot_state *st, const pivot_spill_row *row) {
738
+ if (st->val_ci < 0 || row->nulls[(size_t)st->val_ci]) return 0.0;
739
+ size_t c = (size_t)st->val_ci;
740
+ switch (st->schema_types[c]) {
741
+ case TF_TYPE_INT64: return (double)row->cells[c].i64;
742
+ case TF_TYPE_FLOAT64: return row->cells[c].f64;
743
+ case TF_TYPE_DATE: return (double)row->cells[c].date;
744
+ case TF_TYPE_TIMESTAMP: return (double)row->cells[c].i64;
745
+ case TF_TYPE_BOOL: return row->cells[c].b ? 1.0 : 0.0;
746
+ default: return 0.0;
747
+ }
748
+ }
749
+
750
+ static int pivot_compare_batch_cell(const tf_batch *b, size_t ra, size_t rb, size_t c) {
751
+ int null_a = tf_batch_is_null(b, ra, c);
752
+ int null_b = tf_batch_is_null(b, rb, c);
753
+ if (null_a && null_b) return 0;
754
+ if (null_a) return 1;
755
+ if (null_b) return -1;
756
+ switch (b->col_types[c]) {
757
+ case TF_TYPE_BOOL: return (int)tf_batch_get_bool(b, ra, c) - (int)tf_batch_get_bool(b, rb, c);
758
+ case TF_TYPE_INT64: {
759
+ int64_t a = tf_batch_get_int64(b, ra, c), v = tf_batch_get_int64(b, rb, c);
760
+ return (a > v) - (a < v);
761
+ }
762
+ case TF_TYPE_FLOAT64: {
763
+ double a = tf_batch_get_float64(b, ra, c), v = tf_batch_get_float64(b, rb, c);
764
+ return (a > v) - (a < v);
765
+ }
766
+ case TF_TYPE_STRING: return strcmp(tf_batch_get_string(b, ra, c), tf_batch_get_string(b, rb, c));
767
+ case TF_TYPE_DATE: {
768
+ int32_t a = tf_batch_get_date(b, ra, c), v = tf_batch_get_date(b, rb, c);
769
+ return (a > v) - (a < v);
770
+ }
771
+ case TF_TYPE_TIMESTAMP: {
772
+ int64_t a = tf_batch_get_timestamp(b, ra, c), v = tf_batch_get_timestamp(b, rb, c);
773
+ return (a > v) - (a < v);
774
+ }
775
+ default: return 0;
776
+ }
777
+ }
778
+
779
+ static int pivot_compare_batch_key_rows(const pivot_state *st, size_t ra, size_t rb) {
780
+ for (size_t k = 0; k < st->n_pt; k++) {
781
+ int ci = st->pt_cols[k];
782
+ if (ci < 0) continue;
783
+ int cmp = pivot_compare_batch_cell(st->buf, ra, rb, (size_t)ci);
784
+ if (cmp != 0) return cmp;
785
+ }
786
+ char abuf[64], bbuf[64];
787
+ const char *a = st->name_ci >= 0 ? get_name_str(st->buf, ra, st->name_ci, abuf, sizeof(abuf)) : NULL;
788
+ const char *b = st->name_ci >= 0 ? get_name_str(st->buf, rb, st->name_ci, bbuf, sizeof(bbuf)) : NULL;
789
+ if (!a && !b) {}
790
+ else if (!a) return 1;
791
+ else if (!b) return -1;
792
+ else {
793
+ int cmp = strcmp(a, b);
794
+ if (cmp != 0) return cmp;
795
+ }
796
+ uint64_t oa = st->buf_ordinals[ra], ob = st->buf_ordinals[rb];
797
+ return (oa > ob) - (oa < ob);
798
+ }
799
+
800
+ static int pivot_compare_spill_cell(const pivot_state *st, const pivot_spill_row *a,
801
+ const pivot_spill_row *b, size_t c) {
802
+ int null_a = a->nulls[c] != 0;
803
+ int null_b = b->nulls[c] != 0;
804
+ if (null_a && null_b) return 0;
805
+ if (null_a) return 1;
806
+ if (null_b) return -1;
807
+ switch (st->schema_types[c]) {
808
+ case TF_TYPE_BOOL: return (int)a->cells[c].b - (int)b->cells[c].b;
809
+ case TF_TYPE_INT64:
810
+ case TF_TYPE_TIMESTAMP: return (a->cells[c].i64 > b->cells[c].i64) - (a->cells[c].i64 < b->cells[c].i64);
811
+ case TF_TYPE_FLOAT64: return (a->cells[c].f64 > b->cells[c].f64) - (a->cells[c].f64 < b->cells[c].f64);
812
+ case TF_TYPE_STRING: return strcmp(a->cells[c].str, b->cells[c].str);
813
+ case TF_TYPE_DATE: return (a->cells[c].date > b->cells[c].date) - (a->cells[c].date < b->cells[c].date);
814
+ default: return 0;
815
+ }
816
+ }
817
+
818
+ static int pivot_compare_spill_key_rows(const pivot_state *st, const pivot_spill_row *a,
819
+ const pivot_spill_row *b) {
820
+ for (size_t k = 0; k < st->n_pt; k++) {
821
+ int ci = st->pt_cols[k];
822
+ if (ci < 0) continue;
823
+ int cmp = pivot_compare_spill_cell(st, a, b, (size_t)ci);
824
+ if (cmp != 0) return cmp;
825
+ }
826
+ char abuf[64], bbuf[64];
827
+ const char *an = pivot_spill_name_str(st, a, abuf, sizeof(abuf));
828
+ const char *bn = pivot_spill_name_str(st, b, bbuf, sizeof(bbuf));
829
+ if (!an && !bn) {}
830
+ else if (!an) return 1;
831
+ else if (!bn) return -1;
832
+ else {
833
+ int cmp = strcmp(an, bn);
834
+ if (cmp != 0) return cmp;
835
+ }
836
+ return (a->ordinal > b->ordinal) - (a->ordinal < b->ordinal);
837
+ }
838
+
839
+ typedef struct {
840
+ const pivot_state *st;
841
+ int output_run;
842
+ } pivot_sort_ctx;
843
+
844
+ static int pivot_compare_indices(const void *ctx, size_t ra, size_t rb) {
845
+ const pivot_sort_ctx *sort = (const pivot_sort_ctx *)ctx;
846
+ if (!sort->output_run) return pivot_compare_batch_key_rows(sort->st, ra, rb);
847
+ uint64_t oa = sort->st->out_ordinals[ra];
848
+ uint64_t ob = sort->st->out_ordinals[rb];
849
+ return (oa > ob) - (oa < ob);
850
+ }
851
+
852
+ static size_t *pivot_sorted_indices(size_t n, int output_run, const pivot_state *st) {
853
+ size_t *idx = tf_mallocarray_checked(n ? n : 1, sizeof(size_t));
854
+ if (!idx) return NULL;
855
+ for (size_t i = 0; i < n; i++) idx[i] = i;
856
+ pivot_sort_ctx ctx = { .st = st, .output_run = output_run };
857
+ tf_sort_indices(idx, n, pivot_compare_indices, &ctx);
858
+ return idx;
859
+ }
860
+
861
+ static tf_batch *pivot_create_input_buffer(const pivot_state *st, size_t capacity) {
862
+ tf_batch *b = tf_batch_create(st->n_schema_cols, capacity ? capacity : 16);
863
+ if (!b) return NULL;
864
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
865
+ if (tf_batch_set_schema(b, c, st->schema_names[c], st->schema_types[c]) != TF_OK) {
866
+ tf_batch_free(b);
867
+ return NULL;
868
+ }
869
+ }
870
+ return b;
871
+ }
872
+
873
+ static tf_batch *pivot_create_pt_batch(const pivot_state *st) {
874
+ tf_batch *b = tf_batch_create(st->n_pt, 1);
875
+ if (!b) return NULL;
876
+ for (size_t k = 0; k < st->n_pt; k++) {
877
+ int sc = st->pt_cols[k];
878
+ if (tf_batch_set_schema(b, k, st->schema_names[sc], st->schema_types[sc]) != TF_OK) {
879
+ tf_batch_free(b);
880
+ return NULL;
881
+ }
882
+ }
883
+ return b;
884
+ }
885
+
886
+ static tf_batch *pivot_create_output_buffer(const pivot_state *st, size_t capacity) {
887
+ tf_batch *b = tf_batch_create(st->n_pt + st->n_names, capacity ? capacity : 16);
888
+ if (!b) return NULL;
889
+ if (pivot_set_output_schema_from_arrays(st, b) != TF_OK) {
890
+ tf_batch_free(b);
891
+ return NULL;
892
+ }
893
+ return b;
894
+ }
895
+
896
+ static size_t pivot_estimated_spill_row_bytes(const pivot_state *st) {
897
+ size_t bytes = 48;
898
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
899
+ bytes += 1;
900
+ switch (st->schema_types[c]) {
901
+ case TF_TYPE_BOOL: bytes += 1; break;
902
+ case TF_TYPE_INT64: bytes += sizeof(int64_t); break;
903
+ case TF_TYPE_FLOAT64: bytes += sizeof(double); break;
904
+ case TF_TYPE_STRING: bytes += sizeof(char *) + 64; break;
905
+ case TF_TYPE_DATE: bytes += sizeof(int32_t); break;
906
+ case TF_TYPE_TIMESTAMP: bytes += sizeof(int64_t); break;
907
+ default: break;
908
+ }
909
+ }
910
+ bytes += st->max_categories ? st->max_categories * 64 : st->n_names * 64;
911
+ return bytes < 64 ? 64 : bytes;
912
+ }
913
+
914
+ static int pivot_init_spill_schema(pivot_state *st, const tf_batch *in, tf_side_channels *side) {
915
+ if (st->has_schema) return TF_OK;
916
+ st->n_schema_cols = in->n_cols;
917
+ st->schema_names = tf_callocarray_checked(in->n_cols ? in->n_cols : 1, sizeof(char *));
918
+ st->schema_types = tf_callocarray_checked(in->n_cols ? in->n_cols : 1, sizeof(tf_type));
919
+ if (!st->schema_names || !st->schema_types) return TF_ERROR;
920
+ for (size_t c = 0; c < in->n_cols; c++) {
921
+ st->schema_names[c] = strdup(in->col_names[c] ? in->col_names[c] : "");
922
+ if (!st->schema_names[c]) return TF_ERROR;
923
+ st->schema_types[c] = in->col_types[c];
924
+ }
925
+ st->name_ci = tf_batch_col_index(in, st->name_column);
926
+ st->val_ci = tf_batch_col_index(in, st->value_column);
927
+ if (st->name_ci < 0 || st->val_ci < 0) {
928
+ if (pivot_write_error(side, "pivot spill: name_column or value_column not found") != TF_OK)
929
+ return TF_ERROR;
930
+ return TF_ERROR;
931
+ }
932
+ st->pt_cols = tf_mallocarray_checked(in->n_cols, sizeof(int));
933
+ if (!st->pt_cols) return TF_ERROR;
934
+ st->n_pt = 0;
935
+ for (size_t c = 0; c < in->n_cols; c++) {
936
+ if ((int)c != st->name_ci && (int)c != st->val_ci) st->pt_cols[st->n_pt++] = (int)c;
937
+ }
938
+ if (st->configured_run_rows > 0) {
939
+ st->run_rows = st->configured_run_rows;
940
+ } else if (st->spill_memory_bytes > 0) {
941
+ size_t row_bytes = pivot_estimated_spill_row_bytes(st);
942
+ st->run_rows = st->spill_memory_bytes / (row_bytes * 4);
943
+ if (st->run_rows < PIVOT_MIN_RUN_ROWS) st->run_rows = PIVOT_MIN_RUN_ROWS;
944
+ } else {
945
+ st->run_rows = PIVOT_DEFAULT_RUN_ROWS;
946
+ }
947
+ if (st->run_rows == 0 || st->run_rows == SIZE_MAX) st->run_rows = PIVOT_DEFAULT_RUN_ROWS;
948
+ st->buf = pivot_create_input_buffer(st, st->run_rows);
949
+ if (!st->buf) return TF_ERROR;
950
+ st->has_schema = 1;
951
+ return TF_OK;
952
+ }
953
+
954
+ static int pivot_write_batch_run(pivot_state *st, tf_batch *batch, const uint64_t *ordinals,
955
+ size_t *indices, size_t n, int output_run) {
956
+ char *path = NULL;
957
+ FILE *f = tf_spill_open_run_file(st->spill, output_run ? "pivot-out" : "pivot-key", &path);
958
+ if (!f) return TF_ERROR;
959
+ for (size_t i = 0; i < n; i++) {
960
+ size_t r = indices[i];
961
+ uint64_t ordinal = ordinals[r];
962
+ if (pivot_write_exact(f, &ordinal, sizeof(ordinal)) != TF_OK) goto write_fail;
963
+ for (size_t c = 0; c < batch->n_cols; c++) {
964
+ if (pivot_write_cell(f, batch, r, c) != TF_OK) goto write_fail;
965
+ }
966
+ }
967
+ long pos = ftell(f);
968
+ if (pos > 0) st->spilled_bytes += (size_t)pos;
969
+ if (fclose(f) != 0) {
970
+ tf_set_last_error("pivot spill: failed closing run file");
971
+ remove(path);
972
+ free(path);
973
+ return TF_ERROR;
974
+ }
975
+ if (output_run) {
976
+ if (pivot_append_path(&st->out_run_paths, &st->n_out_runs, &st->cap_out_runs, path) != TF_OK) {
977
+ remove(path); free(path); return TF_ERROR;
978
+ }
979
+ } else {
980
+ if (pivot_append_path(&st->run_paths, &st->n_runs, &st->cap_runs, path) != TF_OK) {
981
+ remove(path); free(path); return TF_ERROR;
982
+ }
983
+ }
984
+ st->spill_runs_created++;
985
+ return TF_OK;
986
+
987
+ write_fail:
988
+ tf_set_last_error("pivot spill: failed writing run file");
989
+ fclose(f);
990
+ remove(path);
991
+ free(path);
992
+ return TF_ERROR;
993
+ }
994
+
995
+ static int pivot_write_key_run(pivot_state *st) {
996
+ if (!st->buf || st->buf->n_rows == 0) return TF_OK;
997
+ size_t *idx = pivot_sorted_indices(st->buf->n_rows, 0, st);
998
+ if (!idx) return TF_ERROR;
999
+ int rc = pivot_write_batch_run(st, st->buf, st->buf_ordinals, idx, st->buf->n_rows, 0);
1000
+ free(idx);
1001
+ if (rc != TF_OK) return TF_ERROR;
1002
+ tf_batch_free(st->buf);
1003
+ st->buf = pivot_create_input_buffer(st, st->run_rows);
1004
+ free(st->buf_ordinals);
1005
+ st->buf_ordinals = NULL;
1006
+ st->buf_ordinal_cap = 0;
1007
+ return st->buf ? TF_OK : TF_ERROR;
1008
+ }
1009
+
1010
+ static int pivot_write_output_run(pivot_state *st) {
1011
+ if (!st->out_buf || st->out_buf->n_rows == 0) return TF_OK;
1012
+ size_t *idx = pivot_sorted_indices(st->out_buf->n_rows, 1, st);
1013
+ if (!idx) return TF_ERROR;
1014
+ int rc = pivot_write_batch_run(st, st->out_buf, st->out_ordinals, idx, st->out_buf->n_rows, 1);
1015
+ free(idx);
1016
+ if (rc != TF_OK) return TF_ERROR;
1017
+ tf_batch_free(st->out_buf);
1018
+ st->out_buf = pivot_create_output_buffer(st, st->run_rows);
1019
+ free(st->out_ordinals);
1020
+ st->out_ordinals = NULL;
1021
+ st->out_ordinal_cap = 0;
1022
+ return st->out_buf ? TF_OK : TF_ERROR;
1023
+ }
1024
+
1025
+ static int pivot_open_readers(pivot_state *st, int output_readers) {
1026
+ char **paths = output_readers ? st->out_run_paths : st->run_paths;
1027
+ size_t n_paths = output_readers ? st->n_out_runs : st->n_runs;
1028
+ pivot_run_reader **readers = output_readers ? &st->out_readers : &st->readers;
1029
+ size_t *n_readers = output_readers ? &st->n_out_readers : &st->n_readers;
1030
+ if (n_paths == 0) return TF_OK;
1031
+ *readers = tf_callocarray_checked(n_paths, sizeof(pivot_run_reader));
1032
+ if (!*readers) return TF_ERROR;
1033
+ *n_readers = n_paths;
1034
+ size_t n_cols = pivot_reader_cols(st, output_readers);
1035
+ for (size_t i = 0; i < n_paths; i++) {
1036
+ (*readers)[i].file = fopen(paths[i], "rb");
1037
+ if (!(*readers)[i].file) {
1038
+ tf_set_last_error("pivot spill: cannot reopen run file");
1039
+ pivot_close_readers(st, output_readers);
1040
+ return TF_ERROR;
1041
+ }
1042
+ if (pivot_spill_row_init(&(*readers)[i].row, n_cols) != TF_OK) {
1043
+ pivot_close_readers(st, output_readers);
1044
+ return TF_ERROR;
1045
+ }
1046
+ int rc = pivot_reader_advance(st, &(*readers)[i], output_readers);
1047
+ if (rc < 0) {
1048
+ pivot_close_readers(st, output_readers);
1049
+ return TF_ERROR;
1050
+ }
1051
+ }
1052
+ return TF_OK;
1053
+ }
1054
+
1055
+ static int pivot_best_key_reader(const pivot_state *st) {
1056
+ int best = -1;
1057
+ for (size_t i = 0; i < st->n_readers; i++) {
1058
+ const pivot_run_reader *r = &st->readers[i];
1059
+ if (!r->has_row || r->done) continue;
1060
+ if (best < 0) { best = (int)i; continue; }
1061
+ int cmp = pivot_compare_spill_key_rows(st, &r->row, &st->readers[best].row);
1062
+ if (cmp < 0 || (cmp == 0 && i < (size_t)best)) best = (int)i;
1063
+ }
1064
+ return best;
1065
+ }
1066
+
1067
+ static int pivot_best_ordinal_reader(const pivot_state *st) {
1068
+ int best = -1;
1069
+ for (size_t i = 0; i < st->n_out_readers; i++) {
1070
+ const pivot_run_reader *r = &st->out_readers[i];
1071
+ if (!r->has_row || r->done) continue;
1072
+ if (best < 0) { best = (int)i; continue; }
1073
+ uint64_t a = r->row.ordinal;
1074
+ uint64_t b = st->out_readers[best].row.ordinal;
1075
+ if (a < b || (a == b && i < (size_t)best)) best = (int)i;
1076
+ }
1077
+ return best;
1078
+ }
1079
+
1080
+ static char *pivot_build_spill_group_key(const pivot_state *st, const pivot_spill_row *row) {
1081
+ size_t buf_cap = 256;
1082
+ char *buf = malloc(buf_cap);
1083
+ if (!buf) return NULL;
1084
+ size_t buf_len = 0;
1085
+ buf[0] = '\0';
1086
+ for (size_t k = 0; k < st->n_pt; k++) {
1087
+ int ci = st->pt_cols[k];
1088
+ char val_buf[64];
1089
+ const char *val = "";
1090
+ size_t val_len = 0;
1091
+ if (ci < 0 || row->nulls[(size_t)ci]) {
1092
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "N|") != TF_OK) { free(buf); return NULL; }
1093
+ continue;
1094
+ }
1095
+ size_t c = (size_t)ci;
1096
+ switch (st->schema_types[c]) {
1097
+ case TF_TYPE_STRING:
1098
+ val = row->cells[c].str ? row->cells[c].str : "";
1099
+ val_len = strlen(val);
1100
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "S") != TF_OK ||
1101
+ pivot_key_append_size(&buf, &buf_cap, &buf_len, val_len) != TF_OK ||
1102
+ pivot_key_append_cstr(&buf, &buf_cap, &buf_len, ":") != TF_OK ||
1103
+ pivot_key_append(&buf, &buf_cap, &buf_len, val, val_len) != TF_OK ||
1104
+ pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "|") != TF_OK) { free(buf); return NULL; }
1105
+ continue;
1106
+ case TF_TYPE_INT64: {
1107
+ int n = snprintf(val_buf, sizeof(val_buf), "%lld", (long long)row->cells[c].i64);
1108
+ if (n < 0 || (size_t)n >= sizeof(val_buf)) { free(buf); return NULL; }
1109
+ val = val_buf; val_len = (size_t)n;
1110
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "I") != TF_OK) { free(buf); return NULL; }
1111
+ break;
1112
+ }
1113
+ case TF_TYPE_FLOAT64: {
1114
+ int n = snprintf(val_buf, sizeof(val_buf), "%.17g", row->cells[c].f64);
1115
+ if (n < 0 || (size_t)n >= sizeof(val_buf)) { free(buf); return NULL; }
1116
+ val = val_buf; val_len = (size_t)n;
1117
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "F") != TF_OK) { free(buf); return NULL; }
1118
+ break;
1119
+ }
1120
+ case TF_TYPE_BOOL:
1121
+ val = row->cells[c].b ? "1" : "0"; val_len = 1;
1122
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "B") != TF_OK) { free(buf); return NULL; }
1123
+ break;
1124
+ case TF_TYPE_DATE: {
1125
+ int n = snprintf(val_buf, sizeof(val_buf), "%d", (int)row->cells[c].date);
1126
+ if (n < 0 || (size_t)n >= sizeof(val_buf)) { free(buf); return NULL; }
1127
+ val = val_buf; val_len = (size_t)n;
1128
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "D") != TF_OK) { free(buf); return NULL; }
1129
+ break;
1130
+ }
1131
+ case TF_TYPE_TIMESTAMP: {
1132
+ int n = snprintf(val_buf, sizeof(val_buf), "%lld", (long long)row->cells[c].i64);
1133
+ if (n < 0 || (size_t)n >= sizeof(val_buf)) { free(buf); return NULL; }
1134
+ val = val_buf; val_len = (size_t)n;
1135
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "T") != TF_OK) { free(buf); return NULL; }
1136
+ break;
1137
+ }
1138
+ default:
1139
+ if (pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "N|") != TF_OK) { free(buf); return NULL; }
1140
+ continue;
1141
+ }
1142
+ if (pivot_key_append(&buf, &buf_cap, &buf_len, val, val_len) != TF_OK ||
1143
+ pivot_key_append_cstr(&buf, &buf_cap, &buf_len, "|") != TF_OK) { free(buf); return NULL; }
1144
+ }
1145
+ return buf;
1146
+ }
1147
+
1148
+ static int pivot_copy_spill_pt_values(tf_batch *dst, const pivot_state *st, const pivot_spill_row *row) {
1149
+ if (tf_batch_ensure_capacity(dst, 1) != TF_OK) return TF_ERROR;
1150
+ for (size_t k = 0; k < st->n_pt; k++) {
1151
+ int ci = st->pt_cols[k];
1152
+ if (ci < 0 || row->nulls[(size_t)ci]) {
1153
+ if (tf_batch_set_null(dst, 0, k) != TF_OK) return TF_ERROR;
1154
+ } else {
1155
+ size_t c = (size_t)ci;
1156
+ if (tf_batch_set_owned_cell_value(dst, 0, k, dst->col_types[k],
1157
+ 0, &row->cells[c]) != TF_OK)
1158
+ return TF_ERROR;
1159
+ }
1160
+ }
1161
+ return tf_batch_expose_row(dst, 0);
1162
+ }
1163
+
1164
+ static int pivot_spill_row_to_batch(const pivot_state *st, tf_batch *out, size_t dst_row,
1165
+ const pivot_spill_row *row) {
1166
+ if (tf_batch_ensure_capacity(out, dst_row + 1) != TF_OK) return TF_ERROR;
1167
+ for (size_t c = 0; c < out->n_cols; c++) {
1168
+ if (tf_batch_set_owned_cell_value(out, dst_row, c, pivot_output_col_type(st, c),
1169
+ row->nulls[c], &row->cells[c]) != TF_OK)
1170
+ return TF_ERROR;
1171
+ }
1172
+ return TF_OK;
1173
+ }
1174
+
1175
+ static int pivot_append_spill_group(pivot_state *st, const tf_batch *pt_batch,
1176
+ const pivot_accum *accum, uint64_t first_ordinal) {
1177
+ size_t dst = st->out_buf->n_rows;
1178
+ if (pivot_emit_row(st, st->out_buf, dst, pt_batch, 0, NULL, st->n_pt, accum) != TF_OK) return TF_ERROR;
1179
+ if (pivot_ensure_ordinals(&st->out_ordinals, &st->out_ordinal_cap, dst + 1) != TF_OK) return TF_ERROR;
1180
+ st->out_ordinals[dst] = first_ordinal;
1181
+ st->spill_distinct_groups++;
1182
+ if (st->out_buf->n_rows >= st->run_rows) return pivot_write_output_run(st);
1183
+ return TF_OK;
1184
+ }
1185
+
1186
+ static int pivot_finish_spill_group(pivot_state *st, char **current_key, tf_batch **current_pt,
1187
+ pivot_accum *accum, uint64_t first_ordinal) {
1188
+ if (!*current_key || !*current_pt) return TF_OK;
1189
+ size_t key_bytes_delta = 0;
1190
+ size_t new_spill_key_bytes = 0;
1191
+ int rc = TF_ERROR;
1192
+ if (tf_size_add(strlen(*current_key), 1, &key_bytes_delta) == TF_OK &&
1193
+ tf_size_add(st->spill_key_bytes, key_bytes_delta,
1194
+ &new_spill_key_bytes) == TF_OK) {
1195
+ st->spill_key_bytes = new_spill_key_bytes;
1196
+ rc = pivot_append_spill_group(st, *current_pt, accum, first_ordinal);
1197
+ }
1198
+ pivot_accum_free(accum);
1199
+ memset(accum, 0, sizeof(*accum));
1200
+ free(*current_key);
1201
+ *current_key = NULL;
1202
+ tf_batch_free(*current_pt);
1203
+ *current_pt = NULL;
1204
+ return rc;
1205
+ }
1206
+
1207
+ static int pivot_start_spill_group(pivot_state *st, const char *key, const pivot_spill_row *row,
1208
+ char **current_key, tf_batch **current_pt,
1209
+ pivot_accum *accum, uint64_t *first_ordinal) {
1210
+ *current_key = strdup(key);
1211
+ if (!*current_key) return TF_ERROR;
1212
+ *current_pt = pivot_create_pt_batch(st);
1213
+ if (!*current_pt) return TF_ERROR;
1214
+ if (pivot_copy_spill_pt_values(*current_pt, st, row) != TF_OK) return TF_ERROR;
1215
+ if (pivot_accum_init(accum, st->n_names) != TF_OK) return TF_ERROR;
1216
+ *first_ordinal = row->ordinal;
1217
+ return TF_OK;
1218
+ }
1219
+
1220
+ static int pivot_produce_output_runs(pivot_state *st) {
1221
+ if (st->key_merge_done) return TF_OK;
1222
+ if (!st->has_schema || st->n_names == 0) {
1223
+ st->key_merge_done = 1;
1224
+ return TF_OK;
1225
+ }
1226
+ if (st->buf && st->buf->n_rows > 0 && pivot_write_key_run(st) != TF_OK) return TF_ERROR;
1227
+ if (st->buf) { tf_batch_free(st->buf); st->buf = NULL; }
1228
+ free(st->buf_ordinals); st->buf_ordinals = NULL; st->buf_ordinal_cap = 0;
1229
+
1230
+ st->out_buf = pivot_create_output_buffer(st, st->run_rows);
1231
+ if (!st->out_buf) return TF_ERROR;
1232
+ if (pivot_open_readers(st, 0) != TF_OK) return TF_ERROR;
1233
+
1234
+ char *current_key = NULL;
1235
+ tf_batch *current_pt = NULL;
1236
+ pivot_accum accum;
1237
+ memset(&accum, 0, sizeof(accum));
1238
+ int have_group = 0;
1239
+ uint64_t first_ordinal = 0;
1240
+
1241
+ for (;;) {
1242
+ int best = pivot_best_key_reader(st);
1243
+ if (best < 0) break;
1244
+ pivot_run_reader *reader = &st->readers[best];
1245
+ char *key = pivot_build_spill_group_key(st, &reader->row);
1246
+ if (!key) goto fail;
1247
+ if (!have_group) {
1248
+ if (pivot_start_spill_group(st, key, &reader->row, &current_key, &current_pt,
1249
+ &accum, &first_ordinal) != TF_OK) { free(key); goto fail; }
1250
+ have_group = 1;
1251
+ } else if (strcmp(current_key, key) != 0) {
1252
+ if (pivot_finish_spill_group(st, &current_key, &current_pt, &accum, first_ordinal) != TF_OK) {
1253
+ free(key); goto fail;
1254
+ }
1255
+ if (pivot_start_spill_group(st, key, &reader->row, &current_key, &current_pt,
1256
+ &accum, &first_ordinal) != TF_OK) { free(key); goto fail; }
1257
+ } else if (reader->row.ordinal < first_ordinal) {
1258
+ if (pivot_copy_spill_pt_values(current_pt, st, &reader->row) != TF_OK) { free(key); goto fail; }
1259
+ first_ordinal = reader->row.ordinal;
1260
+ }
1261
+ free(key);
1262
+
1263
+ char nbuf[64];
1264
+ const char *name = pivot_spill_name_str(st, &reader->row, nbuf, sizeof(nbuf));
1265
+ if (name) {
1266
+ int ni = find_unique_name(st, name);
1267
+ if (ni >= 0) pivot_accum_add(&accum, (size_t)ni, pivot_spill_numeric_value(st, &reader->row));
1268
+ }
1269
+ int rc = pivot_reader_advance(st, reader, 0);
1270
+ if (rc < 0) goto fail;
1271
+ }
1272
+
1273
+ if (have_group && pivot_finish_spill_group(st, &current_key, &current_pt, &accum, first_ordinal) != TF_OK) goto fail;
1274
+ if (st->out_buf && st->out_buf->n_rows > 0 && pivot_write_output_run(st) != TF_OK) goto fail;
1275
+ pivot_close_readers(st, 0);
1276
+ pivot_remove_paths(&st->run_paths, &st->n_runs, &st->cap_runs);
1277
+ st->key_merge_done = 1;
1278
+ return TF_OK;
1279
+
1280
+ fail:
1281
+ if (current_pt) tf_batch_free(current_pt);
1282
+ if (have_group) pivot_accum_free(&accum);
1283
+ free(current_key);
1284
+ return TF_ERROR;
1285
+ }
1286
+
1287
+ static int pivot_begin_output_merge(pivot_state *st) {
1288
+ if (st->output_merge_started) return TF_OK;
1289
+ st->output_merge_started = 1;
1290
+ if (st->out_buf) { tf_batch_free(st->out_buf); st->out_buf = NULL; }
1291
+ free(st->out_ordinals); st->out_ordinals = NULL; st->out_ordinal_cap = 0;
1292
+ if (st->n_out_runs == 0) {
1293
+ tf_spill_cleanup(st->spill);
1294
+ st->spill = NULL;
1295
+ st->output_merge_done = 1;
1296
+ return TF_OK;
1297
+ }
1298
+ return pivot_open_readers(st, 1);
1299
+ }
1300
+
1301
+ static int pivot_output_next_batch(pivot_state *st, tf_batch **out) {
1302
+ *out = NULL;
1303
+ if (pivot_produce_output_runs(st) != TF_OK) return TF_ERROR;
1304
+ if (pivot_begin_output_merge(st) != TF_OK) return TF_ERROR;
1305
+ if (st->output_merge_done) return TF_OK;
1306
+ tf_batch *ob = pivot_create_output_buffer(st, st->output_batch_rows);
1307
+ if (!ob) return TF_ERROR;
1308
+ while (ob->n_rows < st->output_batch_rows) {
1309
+ int best = pivot_best_ordinal_reader(st);
1310
+ if (best < 0) break;
1311
+ pivot_run_reader *reader = &st->out_readers[best];
1312
+ size_t out_row = ob->n_rows;
1313
+ if (pivot_spill_row_to_batch(st, ob, out_row, &reader->row) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
1314
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
1315
+ int rc = pivot_reader_advance(st, reader, 1);
1316
+ if (rc < 0) { tf_batch_free(ob); return TF_ERROR; }
1317
+ }
1318
+ if (ob->n_rows == 0) {
1319
+ tf_batch_free(ob);
1320
+ pivot_close_readers(st, 1);
1321
+ pivot_remove_paths(&st->out_run_paths, &st->n_out_runs, &st->cap_out_runs);
1322
+ tf_spill_cleanup(st->spill);
1323
+ st->spill = NULL;
1324
+ st->output_merge_done = 1;
1325
+ return TF_OK;
1326
+ }
1327
+ st->spill_output_batches++;
1328
+ st->spill_output_rows += ob->n_rows;
1329
+ *out = ob;
1330
+ return TF_OK;
1331
+ }
1332
+
1333
+ static int pivot_process_spill(pivot_state *st, tf_batch *in, tf_side_channels *side) {
1334
+ if (pivot_init_spill_schema(st, in, side) != TF_OK) return TF_ERROR;
1335
+ for (size_t r = 0; r < in->n_rows; r++) {
1336
+ if (!tf_batch_is_null(in, r, (size_t)st->name_ci)) {
1337
+ char nbuf[64];
1338
+ const char *name = get_name_str(in, r, st->name_ci, nbuf, sizeof(nbuf));
1339
+ if (name && pivot_resolve_name(st, name, side) < 0) return TF_ERROR;
1340
+ }
1341
+ size_t dst = st->buf->n_rows;
1342
+ if (tf_batch_copy_row(st->buf, dst, in, r) != TF_OK) return TF_ERROR;
1343
+ if (pivot_ensure_ordinals(&st->buf_ordinals, &st->buf_ordinal_cap, dst + 1) != TF_OK) return TF_ERROR;
1344
+ st->buf_ordinals[dst] = st->next_ordinal++;
1345
+ if (tf_batch_expose_row(st->buf, dst) != TF_OK) return TF_ERROR;
1346
+ if (st->buf->n_rows >= st->run_rows && pivot_write_key_run(st) != TF_OK) return TF_ERROR;
1347
+ }
1348
+ return TF_OK;
1349
+ }
1350
+
1351
+ static int pivot_flush_next(tf_step *self, tf_batch **out, tf_side_channels *side) {
1352
+ (void)side;
1353
+ pivot_state *st = self->state;
1354
+ *out = NULL;
1355
+ if (!st->use_spill) return TF_OK;
1356
+ return pivot_output_next_batch(st, out);
1357
+ }
1358
+
1359
+ static int pivot_append_stats(tf_step *self, tf_buffer *out) {
1360
+ if (!self || !self->state || !out) return TF_ERROR;
1361
+ pivot_state *st = self->state;
1362
+ char buf[320];
1363
+ snprintf(buf, sizeof(buf), ",\"tracked_categories\":%zu", st->n_names);
1364
+ if (tf_buffer_write_str(out, buf) != TF_OK) return TF_ERROR;
1365
+ if (st->use_spill) {
1366
+ snprintf(buf, sizeof(buf),
1367
+ ",\"spill_bytes\":%zu,\"spill_runs\":%zu,"
1368
+ "\"spill_output_batches\":%zu,\"spill_output_rows\":%zu,"
1369
+ "\"spill_distinct_groups\":%zu,\"tracked_key_bytes\":%zu",
1370
+ st->spilled_bytes, st->spill_runs_created,
1371
+ st->spill_output_batches, st->spill_output_rows,
1372
+ st->spill_distinct_groups, st->spill_key_bytes);
1373
+ return tf_buffer_write_str(out, buf);
1374
+ }
1375
+ return TF_OK;
1376
+ }
1377
+
1378
+ static int pivot_resolve_name(pivot_state *st, const char *name,
1379
+ tf_side_channels *side) {
1380
+ int ni = find_unique_name(st, name);
1381
+ if (ni >= 0) return ni;
1382
+ if (st->categories_declared) {
1383
+ unknown_declared_category(st, name, side);
1384
+ return -1;
1385
+ }
1386
+ return add_unique_name(st, name, side);
1387
+ }
1388
+
1389
+ static int pivot_prepare_sorted(pivot_state *st, const tf_batch *in,
1390
+ tf_side_channels *side) {
1391
+ if (st->has_schema) return TF_OK;
1392
+ if (!st->categories_declared || st->n_names == 0) {
1393
+ if (pivot_write_error(side, "pivot: sorted=true requires declared categories") != TF_OK)
1394
+ return TF_ERROR;
1395
+ return TF_ERROR;
1396
+ }
1397
+ st->name_ci = tf_batch_col_index(in, st->name_column);
1398
+ st->val_ci = tf_batch_col_index(in, st->value_column);
1399
+ if (st->name_ci < 0 || st->val_ci < 0) {
1400
+ if (pivot_write_error(side, "pivot: name_column or value_column not found") != TF_OK)
1401
+ return TF_ERROR;
1402
+ return TF_ERROR;
1403
+ }
1404
+ st->pt_cols = tf_mallocarray_checked(in->n_cols, sizeof(int));
1405
+ if (!st->pt_cols) return TF_ERROR;
1406
+ st->n_pt = 0;
1407
+ for (size_t c = 0; c < in->n_cols; c++) {
1408
+ if ((int)c != st->name_ci && (int)c != st->val_ci)
1409
+ st->pt_cols[st->n_pt++] = (int)c;
1410
+ }
1411
+ st->current_pt = tf_batch_create(st->n_pt, 1);
1412
+ if (!st->current_pt) return TF_ERROR;
1413
+ for (size_t k = 0; k < st->n_pt; k++) {
1414
+ int sc = st->pt_cols[k];
1415
+ if (tf_batch_set_schema(st->current_pt, k, in->col_names[sc], in->col_types[sc]) != TF_OK)
1416
+ return TF_ERROR;
1417
+ }
1418
+ if (pivot_accum_init(&st->current_accum, st->n_names) != TF_OK) return TF_ERROR;
1419
+ st->has_schema = 1;
1420
+ return TF_OK;
1421
+ }
1422
+
1423
+ static int pivot_start_sorted_group(pivot_state *st, char *key,
1424
+ const tf_batch *in, size_t row) {
1425
+ free(st->current_key);
1426
+ st->current_key = key;
1427
+ pivot_accum_reset(&st->current_accum);
1428
+ st->current_pt->n_rows = 0;
1429
+ if (tf_batch_ensure_capacity(st->current_pt, 1) != TF_OK) return TF_ERROR;
1430
+ for (size_t k = 0; k < st->n_pt; k++) {
1431
+ if (tf_batch_copy_cell_index(st->current_pt, 0, k, in, row, st->pt_cols[k]) != TF_OK) return TF_ERROR;
1432
+ }
1433
+ if (tf_batch_expose_row(st->current_pt, 0) != TF_OK) return TF_ERROR;
1434
+ st->have_current = 1;
1435
+ return TF_OK;
1436
+ }
1437
+
1438
+ static int pivot_add_sorted_row(pivot_state *st, const tf_batch *in, size_t row,
1439
+ tf_side_channels *side) {
1440
+ char nbuf[64];
1441
+ const char *name = get_name_str(in, row, st->name_ci, nbuf, sizeof(nbuf));
1442
+ if (!name) return TF_OK;
1443
+ int ni = pivot_resolve_name(st, name, side);
1444
+ if (ni < 0) return TF_ERROR;
1445
+ pivot_accum_add(&st->current_accum, (size_t)ni, get_numeric_value(in, row, st->val_ci));
1446
+ return TF_OK;
1447
+ }
1448
+
1449
+ static int pivot_process_sorted(tf_step *self, tf_batch *in, tf_batch **out,
1450
+ tf_side_channels *side) {
1451
+ pivot_state *st = self->state;
1452
+ *out = NULL;
1453
+ if (pivot_prepare_sorted(st, in, side) != TF_OK) return TF_ERROR;
1454
+
1455
+ tf_batch *ob = tf_batch_create(st->n_pt + st->n_names, in->n_rows > 0 ? in->n_rows : 1);
1456
+ if (!ob) return TF_ERROR;
1457
+ if (pivot_set_output_schema_from_source(st, ob, st->current_pt, NULL, st->n_pt) != TF_OK) {
1458
+ tf_batch_free(ob);
1459
+ return TF_ERROR;
1460
+ }
1461
+
1462
+ size_t out_rows = 0;
1463
+ for (size_t r = 0; r < in->n_rows; r++) {
1464
+ char *key = build_pivot_key(in, r, st->pt_cols, st->n_pt);
1465
+ if (!key) { tf_batch_free(ob); return TF_ERROR; }
1466
+ if (!st->have_current) {
1467
+ if (pivot_start_sorted_group(st, key, in, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
1468
+ } else if (strcmp(st->current_key, key) != 0) {
1469
+ if (pivot_emit_row(st, ob, out_rows++, st->current_pt, 0, NULL, st->n_pt, &st->current_accum) != TF_OK) {
1470
+ free(key);
1471
+ tf_batch_free(ob);
1472
+ return TF_ERROR;
1473
+ }
1474
+ if (pivot_start_sorted_group(st, key, in, r) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
1475
+ } else {
1476
+ free(key);
1477
+ }
1478
+ if (pivot_add_sorted_row(st, in, r, side) != TF_OK) {
1479
+ tf_batch_free(ob);
1480
+ return TF_ERROR;
1481
+ }
1482
+ }
1483
+
1484
+ if (out_rows > 0) *out = ob;
1485
+ else tf_batch_free(ob);
1486
+ return TF_OK;
1487
+ }
1488
+
179
1489
  static int pivot_process(tf_step *self, tf_batch *in, tf_batch **out,
180
1490
  tf_side_channels *side) {
181
- (void)side;
182
1491
  pivot_state *st = self->state;
1492
+ if (st->sorted) return pivot_process_sorted(self, in, out, side);
183
1493
  *out = NULL;
1494
+ if (st->use_spill) return pivot_process_spill(st, in, side);
184
1495
 
185
1496
  if (!st->has_schema) {
186
1497
  st->buf = tf_batch_create(in->n_cols, in->n_rows > 0 ? in->n_rows : 16);
187
1498
  if (!st->buf) return TF_ERROR;
188
- for (size_t c = 0; c < in->n_cols; c++)
189
- tf_batch_set_schema(st->buf, c, in->col_names[c], in->col_types[c]);
1499
+ for (size_t c = 0; c < in->n_cols; c++) {
1500
+ if (tf_batch_set_schema(st->buf, c, in->col_names[c], in->col_types[c]) != TF_OK)
1501
+ return TF_ERROR;
1502
+ }
190
1503
  st->has_schema = 1;
191
1504
  }
192
1505
 
193
- /* Track unique name values and buffer rows */
194
1506
  int name_ci = tf_batch_col_index(in, st->name_column);
195
1507
  for (size_t r = 0; r < in->n_rows; r++) {
196
1508
  size_t dst_row = st->buf->n_rows;
197
1509
  if (tf_batch_copy_row(st->buf, dst_row, in, r) != TF_OK) return TF_ERROR;
198
- st->buf->n_rows = dst_row + 1;
1510
+ if (tf_batch_expose_row(st->buf, dst_row) != TF_OK) return TF_ERROR;
199
1511
 
200
- if (name_ci >= 0 && !tf_batch_is_null(in, r, name_ci)) {
1512
+ if (name_ci >= 0 && !tf_batch_is_null(in, r, (size_t)name_ci)) {
201
1513
  char nbuf[64];
202
1514
  const char *name = get_name_str(in, r, name_ci, nbuf, sizeof(nbuf));
203
- if (name) add_unique_name(st, name);
1515
+ if (!name) continue;
1516
+ if (pivot_resolve_name(st, name, side) < 0) return TF_ERROR;
204
1517
  }
205
1518
  }
206
1519
 
207
1520
  return TF_OK;
208
1521
  }
209
1522
 
210
- static int pivot_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
1523
+ static int pivot_flush_sorted(tf_step *self, tf_batch **out, tf_side_channels *side) {
211
1524
  (void)side;
212
1525
  pivot_state *st = self->state;
213
1526
  *out = NULL;
1527
+ if (!st->have_current) return TF_OK;
1528
+ tf_batch *ob = tf_batch_create(st->n_pt + st->n_names, 1);
1529
+ if (!ob) return TF_ERROR;
1530
+ if (pivot_set_output_schema_from_source(st, ob, st->current_pt, NULL, st->n_pt) != TF_OK ||
1531
+ pivot_emit_row(st, ob, 0, st->current_pt, 0, NULL, st->n_pt, &st->current_accum) != TF_OK) {
1532
+ tf_batch_free(ob);
1533
+ return TF_ERROR;
1534
+ }
1535
+ free(st->current_key);
1536
+ st->current_key = NULL;
1537
+ st->have_current = 0;
1538
+ pivot_accum_reset(&st->current_accum);
1539
+ st->current_pt->n_rows = 0;
1540
+ *out = ob;
1541
+ return TF_OK;
1542
+ }
1543
+
1544
+ static int pivot_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
1545
+ pivot_state *st = self->state;
1546
+ if (st->sorted) return pivot_flush_sorted(self, out, side);
1547
+ *out = NULL;
1548
+ if (st->use_spill) return pivot_output_next_batch(st, out);
214
1549
 
215
1550
  if (!st->buf || st->buf->n_rows == 0 || st->n_names == 0) return TF_OK;
216
1551
 
@@ -219,131 +1554,115 @@ static int pivot_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
219
1554
  int val_ci = tf_batch_col_index(buf, st->value_column);
220
1555
  if (name_ci < 0 || val_ci < 0) return TF_OK;
221
1556
 
222
- /* Determine pass-through columns */
223
- size_t n_pt = 0;
224
- int *pt_cols = malloc(buf->n_cols * sizeof(int));
1557
+ int *pt_cols = tf_mallocarray_checked(buf->n_cols, sizeof(int));
225
1558
  if (!pt_cols) return TF_ERROR;
1559
+ size_t n_pt = 0;
226
1560
  for (size_t c = 0; c < buf->n_cols; c++) {
227
1561
  if ((int)c != name_ci && (int)c != val_ci)
228
1562
  pt_cols[n_pt++] = (int)c;
229
1563
  }
230
1564
 
231
- /* Build group map */
232
1565
  pivot_map map = {0};
233
1566
  for (size_t r = 0; r < buf->n_rows; r++) {
234
1567
  char *key = build_pivot_key(buf, r, pt_cols, n_pt);
235
- if (!key) { free(pt_cols); return TF_ERROR; }
1568
+ if (!key) { free(pt_cols); pivot_map_free(&map); return TF_ERROR; }
236
1569
  int gi = find_or_add_pivot_group(&map, key, st->n_names, r);
237
1570
  free(key);
238
- if (gi < 0) { free(pt_cols); return TF_ERROR; }
1571
+ if (gi < 0) { free(pt_cols); pivot_map_free(&map); return TF_ERROR; }
239
1572
 
240
- /* Get name index */
241
1573
  char nbuf[64];
242
1574
  const char *name = get_name_str(buf, r, name_ci, nbuf, sizeof(nbuf));
243
1575
  if (!name) continue;
244
1576
  int ni = find_unique_name(st, name);
245
1577
  if (ni < 0) continue;
246
-
247
- double v = get_numeric_value(buf, r, val_ci);
248
- pivot_accum *a = &map.accums[gi];
249
- a->sums[ni] += v;
250
- if (v < a->mins[ni]) a->mins[ni] = v;
251
- if (v > a->maxs[ni]) a->maxs[ni] = v;
252
- a->counts[ni]++;
253
- if (!a->has_first[ni]) { a->firsts[ni] = v; a->has_first[ni] = 1; }
1578
+ pivot_accum_add(&map.accums[gi], (size_t)ni, get_numeric_value(buf, r, val_ci));
254
1579
  }
255
1580
 
256
- /* Build output batch */
257
- size_t n_out_cols = n_pt + st->n_names;
258
- tf_batch *ob = tf_batch_create(n_out_cols, map.count);
259
- if (!ob) { free(pt_cols); return TF_ERROR; }
260
-
261
- /* Pass-through column schemas */
262
- for (size_t k = 0; k < n_pt; k++)
263
- tf_batch_set_schema(ob, k, buf->col_names[pt_cols[k]], buf->col_types[pt_cols[k]]);
264
-
265
- /* Pivot column schemas */
266
- tf_type pivot_type = TF_TYPE_FLOAT64;
267
- if (st->agg == PIVOT_COUNT) pivot_type = TF_TYPE_INT64;
268
- for (size_t k = 0; k < st->n_names; k++)
269
- tf_batch_set_schema(ob, n_pt + k, st->unique_names[k], pivot_type);
1581
+ tf_batch *ob = tf_batch_create(n_pt + st->n_names, map.count);
1582
+ if (!ob) { free(pt_cols); pivot_map_free(&map); return TF_ERROR; }
1583
+ if (pivot_set_output_schema_from_source(st, ob, buf, pt_cols, n_pt) != TF_OK) {
1584
+ free(pt_cols);
1585
+ pivot_map_free(&map);
1586
+ tf_batch_free(ob);
1587
+ return TF_ERROR;
1588
+ }
270
1589
 
271
- /* Fill output */
272
1590
  for (size_t g = 0; g < map.count; g++) {
273
- tf_batch_ensure_capacity(ob, g + 1);
274
-
275
- /* Copy pass-through values from the first row of this group */
276
- size_t src_row = map.pass_rows[g];
277
- for (size_t k = 0; k < n_pt; k++) {
278
- int sc = pt_cols[k];
279
- if (tf_batch_is_null(buf, src_row, sc)) {
280
- tf_batch_set_null(ob, g, k);
281
- } else {
282
- switch (buf->col_types[sc]) {
283
- case TF_TYPE_BOOL: tf_batch_set_bool(ob, g, k, tf_batch_get_bool(buf, src_row, sc)); break;
284
- case TF_TYPE_INT64: tf_batch_set_int64(ob, g, k, tf_batch_get_int64(buf, src_row, sc)); break;
285
- case TF_TYPE_FLOAT64: tf_batch_set_float64(ob, g, k, tf_batch_get_float64(buf, src_row, sc)); break;
286
- case TF_TYPE_STRING: tf_batch_set_string(ob, g, k, tf_batch_get_string(buf, src_row, sc)); break;
287
- case TF_TYPE_DATE: tf_batch_set_date(ob, g, k, tf_batch_get_date(buf, src_row, sc)); break;
288
- case TF_TYPE_TIMESTAMP: tf_batch_set_timestamp(ob, g, k, tf_batch_get_timestamp(buf, src_row, sc)); break;
289
- default: tf_batch_set_null(ob, g, k); break;
290
- }
291
- }
1591
+ if (pivot_emit_row(st, ob, g, buf, map.pass_rows[g], pt_cols, n_pt, &map.accums[g]) != TF_OK) {
1592
+ free(pt_cols);
1593
+ pivot_map_free(&map);
1594
+ tf_batch_free(ob);
1595
+ return TF_ERROR;
292
1596
  }
293
-
294
- /* Set pivot values */
295
- pivot_accum *a = &map.accums[g];
296
- for (size_t k = 0; k < st->n_names; k++) {
297
- size_t oc = n_pt + k;
298
- if (a->counts[k] == 0) {
299
- tf_batch_set_null(ob, g, oc);
300
- continue;
301
- }
302
- double v = 0;
303
- switch (st->agg) {
304
- case PIVOT_FIRST: v = a->firsts[k]; break;
305
- case PIVOT_SUM: v = a->sums[k]; break;
306
- case PIVOT_COUNT: tf_batch_set_int64(ob, g, oc, (int64_t)a->counts[k]); goto next_name;
307
- case PIVOT_AVG: v = a->sums[k] / (double)a->counts[k]; break;
308
- case PIVOT_MIN: v = a->mins[k]; break;
309
- case PIVOT_MAX: v = a->maxs[k]; break;
310
- }
311
- tf_batch_set_float64(ob, g, oc, v);
312
- next_name:;
313
- }
314
- ob->n_rows = g + 1;
315
1597
  }
316
1598
 
317
- /* Cleanup map */
318
- for (size_t i = 0; i < map.count; i++) {
319
- free(map.keys[i]);
320
- free(map.accums[i].sums);
321
- free(map.accums[i].mins);
322
- free(map.accums[i].maxs);
323
- free(map.accums[i].counts);
324
- free(map.accums[i].has_first);
325
- free(map.accums[i].firsts);
326
- }
327
- free(map.keys);
328
- free(map.pass_rows);
329
- free(map.accums);
1599
+ pivot_map_free(&map);
330
1600
  free(pt_cols);
331
-
332
1601
  *out = ob;
333
1602
  return TF_OK;
334
1603
  }
335
1604
 
1605
+ static void pivot_state_free(pivot_state *st) {
1606
+ if (!st) return;
1607
+ if (st->has_schema) {
1608
+ pivot_close_readers(st, 0);
1609
+ pivot_close_readers(st, 1);
1610
+ }
1611
+ free(st->name_column);
1612
+ free(st->value_column);
1613
+ if (st->buf) tf_batch_free(st->buf);
1614
+ if (st->out_buf) tf_batch_free(st->out_buf);
1615
+ for (size_t i = 0; i < st->n_names; i++) free(st->unique_names[i]);
1616
+ free(st->unique_names);
1617
+ free(st->pt_cols);
1618
+ if (st->current_pt) tf_batch_free(st->current_pt);
1619
+ free(st->current_key);
1620
+ pivot_accum_free(&st->current_accum);
1621
+ pivot_remove_paths(&st->run_paths, &st->n_runs, &st->cap_runs);
1622
+ pivot_remove_paths(&st->out_run_paths, &st->n_out_runs, &st->cap_out_runs);
1623
+ for (size_t i = 0; i < st->n_schema_cols; i++) free(st->schema_names ? st->schema_names[i] : NULL);
1624
+ free(st->schema_names);
1625
+ free(st->schema_types);
1626
+ free(st->buf_ordinals);
1627
+ free(st->out_ordinals);
1628
+ tf_spill_cleanup(st->spill);
1629
+ free(st->spill_dir);
1630
+ free(st);
1631
+ }
1632
+
336
1633
  static void pivot_destroy(tf_step *self) {
337
- pivot_state *st = self->state;
338
- if (st) {
339
- free(st->name_column);
340
- free(st->value_column);
341
- if (st->buf) tf_batch_free(st->buf);
342
- for (size_t i = 0; i < st->n_names; i++) free(st->unique_names[i]);
343
- free(st->unique_names);
344
- free(st);
1634
+ if (self) {
1635
+ pivot_state_free(self->state);
1636
+ free(self);
1637
+ }
1638
+ }
1639
+
1640
+ static int parse_positive_size(const cJSON *args, const char *name, size_t *out) {
1641
+ return tf_json_get_size_arg(args, name, 1, TF_MAX_COUNT_ARG, out, "pivot");
1642
+ }
1643
+
1644
+ static int load_declared_categories(pivot_state *st, const cJSON *args) {
1645
+ cJSON *cats = cJSON_GetObjectItemCaseSensitive(args, "categories");
1646
+ if (!cats) return TF_OK;
1647
+ if (!cJSON_IsArray(cats)) {
1648
+ tf_set_last_error("pivot: categories must be an array of non-empty strings");
1649
+ return TF_ERROR;
1650
+ }
1651
+ int n = cJSON_GetArraySize(cats);
1652
+ if (n <= 0) {
1653
+ tf_set_last_error("pivot: categories cannot be empty");
1654
+ return TF_ERROR;
1655
+ }
1656
+ st->categories_declared = 1;
1657
+ for (int i = 0; i < n; i++) {
1658
+ cJSON *item = cJSON_GetArrayItem(cats, i);
1659
+ if (!cJSON_IsString(item) || !item->valuestring || !item->valuestring[0]) {
1660
+ tf_set_last_error("pivot: categories must be non-empty strings");
1661
+ return TF_ERROR;
1662
+ }
1663
+ if (add_unique_name(st, item->valuestring, NULL) < 0) return TF_ERROR;
345
1664
  }
346
- free(self);
1665
+ return TF_OK;
347
1666
  }
348
1667
 
349
1668
  tf_step *tf_pivot_create(const cJSON *args) {
@@ -354,16 +1673,81 @@ tf_step *tf_pivot_create(const cJSON *args) {
354
1673
 
355
1674
  pivot_state *st = calloc(1, sizeof(pivot_state));
356
1675
  if (!st) return NULL;
1676
+ st->output_batch_rows = PIVOT_DEFAULT_OUTPUT_ROWS;
357
1677
  st->name_column = strdup(name_j->valuestring);
358
1678
  st->value_column = strdup(val_j->valuestring);
1679
+ if (!st->name_column || !st->value_column) { pivot_state_free(st); return NULL; }
359
1680
 
360
1681
  cJSON *agg_j = cJSON_GetObjectItemCaseSensitive(args, "agg");
361
1682
  st->agg = cJSON_IsString(agg_j) ? parse_pivot_agg(agg_j->valuestring) : PIVOT_FIRST;
362
1683
 
363
- tf_step *step = malloc(sizeof(tf_step));
364
- if (!step) { pivot_destroy(&(tf_step){.state = st}); return NULL; }
1684
+ if (parse_positive_size(args, "max_categories", &st->max_categories) < 0) {
1685
+ pivot_state_free(st);
1686
+ return NULL;
1687
+ }
1688
+ if (load_declared_categories(st, args) != TF_OK) {
1689
+ pivot_state_free(st);
1690
+ return NULL;
1691
+ }
1692
+
1693
+ cJSON *spill_dir_j = cJSON_GetObjectItemCaseSensitive(args, "spill_dir");
1694
+ if (cJSON_IsString(spill_dir_j) && spill_dir_j->valuestring && spill_dir_j->valuestring[0]) {
1695
+ st->use_spill = 1;
1696
+ st->spill_dir = strdup(spill_dir_j->valuestring);
1697
+ if (!st->spill_dir) { pivot_state_free(st); return NULL; }
1698
+ if (tf_spill_session_create(st->spill_dir, &st->spill) != TF_OK) { pivot_state_free(st); return NULL; }
1699
+ size_t parsed_size = 0;
1700
+ int has_spill_memory = tf_json_get_size_arg(args, "spill_memory_bytes",
1701
+ 1, TF_MAX_SPILL_MEMORY_BYTES,
1702
+ &parsed_size, "pivot");
1703
+ if (has_spill_memory < 0) { pivot_state_free(st); return NULL; }
1704
+ if (has_spill_memory > 0) st->spill_memory_bytes = parsed_size;
1705
+ int has_spill_rows = tf_json_get_size_arg(args, "spill_run_rows",
1706
+ 1, TF_MAX_SPILL_RUN_ROWS,
1707
+ &parsed_size, "pivot");
1708
+ if (has_spill_rows < 0) { pivot_state_free(st); return NULL; }
1709
+ if (has_spill_rows > 0) st->configured_run_rows = parsed_size;
1710
+ int has_output_rows = tf_json_get_size_arg(args, "spill_output_rows",
1711
+ 1, TF_MAX_SPILL_OUTPUT_ROWS,
1712
+ &parsed_size, "pivot");
1713
+ if (has_output_rows < 0) {
1714
+ pivot_state_free(st);
1715
+ return NULL;
1716
+ }
1717
+ if (has_output_rows > 0) st->output_batch_rows = parsed_size;
1718
+ }
1719
+
1720
+ cJSON *sorted_j = cJSON_GetObjectItemCaseSensitive(args, "sorted");
1721
+ if (sorted_j) {
1722
+ if (cJSON_IsTrue(sorted_j)) st->sorted = 1;
1723
+ else if (!cJSON_IsFalse(sorted_j)) {
1724
+ tf_set_last_error("pivot: sorted must be boolean");
1725
+ pivot_state_free(st);
1726
+ return NULL;
1727
+ }
1728
+ }
1729
+ if (st->sorted && (!st->categories_declared || st->n_names == 0)) {
1730
+ tf_set_last_error("pivot: sorted=true requires declared categories");
1731
+ pivot_state_free(st);
1732
+ return NULL;
1733
+ }
1734
+ if (st->use_spill && st->sorted) {
1735
+ tf_set_last_error("pivot: spill_dir and sorted=true are mutually exclusive");
1736
+ pivot_state_free(st);
1737
+ return NULL;
1738
+ }
1739
+ if (st->use_spill && !st->categories_declared && st->max_categories == 0) {
1740
+ tf_set_last_error("pivot: spill_dir needs declared categories or max_categories to bound output columns");
1741
+ pivot_state_free(st);
1742
+ return NULL;
1743
+ }
1744
+
1745
+ tf_step *step = calloc(1, sizeof(tf_step));
1746
+ if (!step) { pivot_state_free(st); return NULL; }
365
1747
  step->process = pivot_process;
366
1748
  step->flush = pivot_flush;
1749
+ step->flush_next = st->use_spill ? pivot_flush_next : NULL;
1750
+ step->append_stats = pivot_append_stats;
367
1751
  step->destroy = pivot_destroy;
368
1752
  step->state = st;
369
1753
  return step;