tranfi 0.0.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/LICENSE +177 -21
  2. package/NOTICE +8 -0
  3. package/README.md +627 -0
  4. package/app/assets/index-6quYZ5Ap.css +5 -0
  5. package/app/assets/index-BIAIKnrp.js +160 -0
  6. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  7. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  8. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  9. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  10. package/app/index.html +13 -0
  11. package/binding.gyp +121 -0
  12. package/csrc/arena.c +93 -0
  13. package/csrc/batch.c +976 -0
  14. package/csrc/buffer.c +154 -0
  15. package/csrc/cJSON.c +3386 -0
  16. package/csrc/cJSON.h +316 -0
  17. package/csrc/codec_csv.c +1951 -0
  18. package/csrc/codec_jsonl.c +1086 -0
  19. package/csrc/codec_table.c +248 -0
  20. package/csrc/codec_text.c +447 -0
  21. package/csrc/compiler.c +130 -0
  22. package/csrc/config.h +21 -0
  23. package/csrc/date_utils.h +94 -0
  24. package/csrc/dsl.c +5417 -0
  25. package/csrc/dsl.h +22 -0
  26. package/csrc/expr.c +1553 -0
  27. package/csrc/expr.h +58 -0
  28. package/csrc/internal.h +539 -0
  29. package/csrc/ir.c +166 -0
  30. package/csrc/ir.h +208 -0
  31. package/csrc/ir_schema.c +75 -0
  32. package/csrc/ir_serialize.c +166 -0
  33. package/csrc/ir_sql.c +1822 -0
  34. package/csrc/ir_validate.c +576 -0
  35. package/csrc/json_path.c +210 -0
  36. package/csrc/main.c +1241 -0
  37. package/csrc/memory_estimate.c +477 -0
  38. package/csrc/op_acf.c +283 -0
  39. package/csrc/op_across.c +477 -0
  40. package/csrc/op_anomaly.c +255 -0
  41. package/csrc/op_assert.c +761 -0
  42. package/csrc/op_bin.c +248 -0
  43. package/csrc/op_cast.c +523 -0
  44. package/csrc/op_clip.c +99 -0
  45. package/csrc/op_date_trunc.c +355 -0
  46. package/csrc/op_datetime.c +394 -0
  47. package/csrc/op_derive.c +216 -0
  48. package/csrc/op_diff.c +250 -0
  49. package/csrc/op_ewma.c +222 -0
  50. package/csrc/op_explode.c +206 -0
  51. package/csrc/op_fill_down.c +235 -0
  52. package/csrc/op_fill_null.c +268 -0
  53. package/csrc/op_filter.c +181 -0
  54. package/csrc/op_frequency.c +721 -0
  55. package/csrc/op_grep.c +181 -0
  56. package/csrc/op_group_agg.c +1956 -0
  57. package/csrc/op_hash.c +159 -0
  58. package/csrc/op_head.c +84 -0
  59. package/csrc/op_interpolate.c +445 -0
  60. package/csrc/op_join.c +2902 -0
  61. package/csrc/op_json_extract.c +227 -0
  62. package/csrc/op_json_filter.c +384 -0
  63. package/csrc/op_json_flatten.c +293 -0
  64. package/csrc/op_json_schema.c +503 -0
  65. package/csrc/op_label_encode.c +419 -0
  66. package/csrc/op_lag.c +181 -0
  67. package/csrc/op_lead.c +242 -0
  68. package/csrc/op_normalize.c +510 -0
  69. package/csrc/op_onehot.c +457 -0
  70. package/csrc/op_pivot.c +1754 -0
  71. package/csrc/op_quarantine.c +189 -0
  72. package/csrc/op_registry.c +3044 -0
  73. package/csrc/op_rename.c +129 -0
  74. package/csrc/op_replace.c +354 -0
  75. package/csrc/op_rleid.c +297 -0
  76. package/csrc/op_rowid.c +559 -0
  77. package/csrc/op_sample.c +158 -0
  78. package/csrc/op_schema.c +1341 -0
  79. package/csrc/op_schema_infer.c +252 -0
  80. package/csrc/op_select.c +340 -0
  81. package/csrc/op_set.c +3449 -0
  82. package/csrc/op_skip.c +95 -0
  83. package/csrc/op_sort.c +819 -0
  84. package/csrc/op_source_name.c +120 -0
  85. package/csrc/op_split.c +151 -0
  86. package/csrc/op_split_data.c +119 -0
  87. package/csrc/op_stack.c +271 -0
  88. package/csrc/op_stats.c +875 -0
  89. package/csrc/op_step.c +333 -0
  90. package/csrc/op_tail.c +105 -0
  91. package/csrc/op_tee.c +338 -0
  92. package/csrc/op_top.c +357 -0
  93. package/csrc/op_trim.c +138 -0
  94. package/csrc/op_unique.c +1343 -0
  95. package/csrc/op_unpivot.c +193 -0
  96. package/csrc/op_validate.c +648 -0
  97. package/csrc/op_window.c +591 -0
  98. package/csrc/path_policy.c +85 -0
  99. package/csrc/pipeline.c +1088 -0
  100. package/csrc/recipes.c +104 -0
  101. package/csrc/recipes.h +27 -0
  102. package/csrc/report.c +506 -0
  103. package/csrc/report.h +22 -0
  104. package/csrc/selector.c +1097 -0
  105. package/csrc/size_utils.c +348 -0
  106. package/csrc/spill.c +317 -0
  107. package/csrc/spill.h +21 -0
  108. package/csrc/tranfi.h +291 -0
  109. package/csrc/transform.h +209 -0
  110. package/csrc/transform_api.c +2237 -0
  111. package/csrc/transform_categorical.c +923 -0
  112. package/csrc/transform_internal.h +472 -0
  113. package/csrc/transform_json.c +3812 -0
  114. package/csrc/transform_numeric.c +1966 -0
  115. package/csrc/transform_sha256.c +154 -0
  116. package/csrc/transform_wasm.h +162 -0
  117. package/csrc/transform_wasm_api.c +1373 -0
  118. package/csrc/wasm_api.c +218 -0
  119. package/napi_api.c +534 -0
  120. package/napi_transform.c +1648 -0
  121. package/napi_transform.h +8 -0
  122. package/package.json +64 -59
  123. package/scripts/install-native.js +76 -0
  124. package/scripts/prepack.js +64 -0
  125. package/scripts/sync-csrc.js +23 -0
  126. package/src/cli.js +190 -0
  127. package/src/engines/duckdb.js +142 -0
  128. package/src/index.js +925 -0
  129. package/src/memory_policy.js +411 -0
  130. package/src/native.js +18 -0
  131. package/src/pipeline.js +709 -0
  132. package/src/recipe_json.js +80 -0
  133. package/src/server.js +279 -0
  134. package/src/transform.js +403 -0
  135. package/src/transform_error.js +10 -0
  136. package/src/wasm.js +21 -0
  137. package/wasm/index.js +732 -0
  138. package/wasm/package.json +1 -0
  139. package/wasm/tranfi_core.js +0 -0
  140. package/wasm/transform.js +1156 -0
  141. package/wasm/worker.js +786 -0
  142. package/dist/bundle.js +0 -1
  143. package/index.html +0 -18
  144. package/src/app.css +0 -169
  145. package/src/app.js +0 -203
  146. package/src/app.vue +0 -250
  147. package/src/bulma-input.vue +0 -110
  148. package/src/common-inputs.js +0 -28
  149. package/src/main.js +0 -20
  150. package/src/transforms.js +0 -166
  151. package/webpack.config.js +0 -108
@@ -0,0 +1,1343 @@
1
+ /*
2
+ * op_unique.c -- Deduplicate rows by key columns.
3
+ *
4
+ * Default mode is a streaming hash set with optional caps. sorted=true is an
5
+ * adjacent-key streaming mode. mode=approx uses a bounded Bloom filter and may
6
+ * drop first occurrences on false positives. spill_dir enables exact external
7
+ * dedup without retaining all keys: rows are sorted by key+ordinal to choose
8
+ * the first row per key, then selected rows are sorted by original ordinal
9
+ * before emission.
10
+ */
11
+
12
+ #include "internal.h"
13
+ #include "spill.h"
14
+ #include "cJSON.h"
15
+ #include "date_utils.h"
16
+ #include <errno.h>
17
+ #include <stdint.h>
18
+ #include <stdio.h>
19
+ #include <stdlib.h>
20
+ #include <string.h>
21
+ #include <unistd.h>
22
+
23
+ #define UNIQUE_DEFAULT_RUN_ROWS 8192
24
+ #define UNIQUE_DEFAULT_OUTPUT_ROWS 1024
25
+ #define UNIQUE_MIN_RUN_ROWS 16
26
+ #define UNIQUE_DEFAULT_BLOOM_BYTES (1024u * 1024u)
27
+ #define UNIQUE_DEFAULT_BLOOM_HASHES 7
28
+ #define UNIQUE_MAX_BLOOM_HASHES 32
29
+
30
+ typedef tf_owned_cell_value unique_cell;
31
+
32
+ typedef struct {
33
+ uint64_t ordinal;
34
+ uint8_t *nulls;
35
+ unique_cell *cells;
36
+ } unique_spill_row;
37
+
38
+ typedef struct {
39
+ FILE *file;
40
+ unique_spill_row row;
41
+ int has_row;
42
+ int done;
43
+ } unique_run_reader;
44
+
45
+ /* ---- Simple hash set for capped in-memory mode ---- */
46
+
47
+ typedef struct {
48
+ char **keys;
49
+ size_t count;
50
+ size_t cap;
51
+ size_t key_bytes;
52
+ } hash_set;
53
+
54
+ static int hs_init(hash_set *hs, size_t cap) {
55
+ hs->cap = cap;
56
+ hs->count = 0;
57
+ hs->key_bytes = 0;
58
+ hs->keys = tf_callocarray_checked(cap, sizeof(char *));
59
+ return hs->keys ? 0 : -1;
60
+ }
61
+
62
+ static uint32_t hs_hash(const char *key) {
63
+ uint32_t h = 5381;
64
+ for (const char *p = key; *p; p++)
65
+ h = ((h << 5) + h) ^ (uint32_t)(unsigned char)*p;
66
+ return h;
67
+ }
68
+
69
+ static int hs_grow(hash_set *hs) {
70
+ size_t new_cap = 0;
71
+ if (tf_size_mul(hs->cap, 2, &new_cap) != TF_OK) return -1;
72
+ char **new_keys = tf_callocarray_checked(new_cap, sizeof(char *));
73
+ if (!new_keys) return -1;
74
+
75
+ for (size_t i = 0; i < hs->cap; i++) {
76
+ if (hs->keys[i]) {
77
+ uint32_t idx = hs_hash(hs->keys[i]) % new_cap;
78
+ while (new_keys[idx]) idx = (idx + 1) % new_cap;
79
+ new_keys[idx] = hs->keys[i];
80
+ }
81
+ }
82
+ free(hs->keys);
83
+ hs->keys = new_keys;
84
+ hs->cap = new_cap;
85
+ return 0;
86
+ }
87
+
88
+ static int hs_contains(const hash_set *hs, const char *key) {
89
+ if (!hs->keys || hs->cap == 0) return 0;
90
+ uint32_t idx = hs_hash(key) % hs->cap;
91
+ while (hs->keys[idx]) {
92
+ if (strcmp(hs->keys[idx], key) == 0) return 1;
93
+ idx = (idx + 1) % hs->cap;
94
+ }
95
+ return 0;
96
+ }
97
+
98
+ static int hs_insert(hash_set *hs, const char *key) {
99
+ size_t load_count = 0;
100
+ size_t load_limit = 0;
101
+ if (tf_size_mul(hs->count, 4, &load_count) != TF_OK ||
102
+ tf_size_mul(hs->cap, 3, &load_limit) != TF_OK) {
103
+ return -1;
104
+ }
105
+ if (load_count >= load_limit) {
106
+ if (hs_grow(hs) != 0) return -1;
107
+ }
108
+ uint32_t idx = hs_hash(key) % hs->cap;
109
+ while (hs->keys[idx]) {
110
+ if (strcmp(hs->keys[idx], key) == 0) return 0;
111
+ idx = (idx + 1) % hs->cap;
112
+ }
113
+ size_t key_bytes_delta = 0;
114
+ size_t new_key_bytes = 0;
115
+ if (tf_size_add(strlen(key), 1, &key_bytes_delta) != TF_OK ||
116
+ tf_size_add(hs->key_bytes, key_bytes_delta, &new_key_bytes) != TF_OK) {
117
+ return -1;
118
+ }
119
+ hs->keys[idx] = strdup(key);
120
+ if (!hs->keys[idx]) return -1;
121
+ hs->key_bytes = new_key_bytes;
122
+ hs->count++;
123
+ return 1;
124
+ }
125
+
126
+ static void hs_free(hash_set *hs) {
127
+ if (!hs) return;
128
+ if (hs->keys) {
129
+ for (size_t i = 0; i < hs->cap; i++) free(hs->keys[i]);
130
+ }
131
+ free(hs->keys);
132
+ memset(hs, 0, sizeof(*hs));
133
+ }
134
+
135
+ /* ---- Unique transform ---- */
136
+
137
+ typedef struct {
138
+ char **key_cols;
139
+ size_t n_key_cols;
140
+ size_t max_keys;
141
+ size_t max_state_bytes;
142
+ int sorted;
143
+ int use_spill;
144
+ int approximate;
145
+ uint8_t *bloom;
146
+ size_t bloom_bytes;
147
+ size_t bloom_bits;
148
+ size_t bloom_hashes;
149
+ size_t approx_inserted;
150
+ size_t approx_filtered;
151
+
152
+ char *prev_key;
153
+ int have_prev_key;
154
+ hash_set seen;
155
+
156
+ char *spill_dir;
157
+ tf_spill_session *spill;
158
+ size_t spill_memory_bytes;
159
+ size_t configured_run_rows;
160
+ size_t run_rows;
161
+ size_t output_batch_rows;
162
+
163
+ int has_schema;
164
+ char **schema_names;
165
+ tf_type *schema_types;
166
+ size_t n_schema_cols;
167
+ int *key_indices;
168
+ size_t n_keys;
169
+
170
+ tf_batch *buf;
171
+ uint64_t *buf_ordinals;
172
+ size_t buf_ordinal_cap;
173
+ uint64_t next_ordinal;
174
+
175
+ char **run_paths;
176
+ size_t n_runs;
177
+ size_t cap_runs;
178
+ char **out_run_paths;
179
+ size_t n_out_runs;
180
+ size_t cap_out_runs;
181
+ size_t run_seq;
182
+ size_t out_run_seq;
183
+
184
+ unique_run_reader *readers;
185
+ size_t n_readers;
186
+ unique_run_reader *out_readers;
187
+ size_t n_out_readers;
188
+
189
+ tf_batch *out_buf;
190
+ uint64_t *out_ordinals;
191
+ size_t out_ordinal_cap;
192
+
193
+ int key_merge_done;
194
+ int output_merge_started;
195
+ int output_merge_done;
196
+ char *last_spill_key;
197
+
198
+ size_t spilled_bytes;
199
+ size_t spill_runs_created;
200
+ size_t spill_output_batches;
201
+ size_t spill_output_rows;
202
+ size_t spill_distinct_rows;
203
+ size_t spill_key_bytes;
204
+ } unique_state;
205
+
206
+ /* ---- Bounded approximate membership for mode=approx ---- */
207
+
208
+ static uint64_t unique_hash64(const char *key) {
209
+ uint64_t h = UINT64_C(1469598103934665603);
210
+ for (const unsigned char *p = (const unsigned char *)key; *p; p++) {
211
+ h ^= (uint64_t)*p;
212
+ h *= UINT64_C(1099511628211);
213
+ }
214
+ return h;
215
+ }
216
+
217
+ static uint64_t unique_mix64(uint64_t x) {
218
+ x ^= x >> 30;
219
+ x *= UINT64_C(0xbf58476d1ce4e5b9);
220
+ x ^= x >> 27;
221
+ x *= UINT64_C(0x94d049bb133111eb);
222
+ x ^= x >> 31;
223
+ return x;
224
+ }
225
+
226
+ static int bloom_get_bit(const unique_state *st, size_t bit) {
227
+ return (st->bloom[bit >> 3] & (uint8_t)(1u << (bit & 7u))) != 0;
228
+ }
229
+
230
+ static void bloom_set_bit(unique_state *st, size_t bit) {
231
+ st->bloom[bit >> 3] |= (uint8_t)(1u << (bit & 7u));
232
+ }
233
+
234
+ static int unique_bloom_check_add(unique_state *st, const char *key) {
235
+ uint64_t h1 = unique_mix64(unique_hash64(key));
236
+ uint64_t h2 = unique_mix64(h1 ^ UINT64_C(0x9e3779b97f4a7c15));
237
+ if ((h2 & UINT64_C(1)) == 0) h2 |= UINT64_C(1);
238
+
239
+ int maybe_seen = 1;
240
+ for (size_t i = 0; i < st->bloom_hashes; i++) {
241
+ size_t bit = (size_t)((h1 + (uint64_t)i * h2) % (uint64_t)st->bloom_bits);
242
+ if (!bloom_get_bit(st, bit)) {
243
+ maybe_seen = 0;
244
+ break;
245
+ }
246
+ }
247
+ for (size_t i = 0; i < st->bloom_hashes; i++) {
248
+ size_t bit = (size_t)((h1 + (uint64_t)i * h2) % (uint64_t)st->bloom_bits);
249
+ bloom_set_bit(st, bit);
250
+ }
251
+ return maybe_seen;
252
+ }
253
+
254
+ static int unique_write_error(tf_side_channels *side, const char *msg) {
255
+ return tf_side_write_error(side, msg);
256
+ }
257
+
258
+ static int unique_limit_error(unique_state *st, tf_side_channels *side) {
259
+ char msg[160];
260
+ snprintf(msg, sizeof(msg),
261
+ "unique: max_keys=%zu exceeded while tracking distinct keys",
262
+ st->max_keys);
263
+ return unique_write_error(side, msg);
264
+ }
265
+
266
+ static int append_bytes(char **buf, size_t *len, size_t *cap, const void *src, size_t n) {
267
+ size_t need = 0;
268
+ if (tf_size_add(*len, n, &need) != TF_OK ||
269
+ tf_size_add(need, 1, &need) != TF_OK) {
270
+ return TF_ERROR;
271
+ }
272
+ if (need > *cap) {
273
+ size_t new_cap = 0;
274
+ if (tf_size_grow_pow2(*cap, need, 128, &new_cap) != TF_OK) return TF_ERROR;
275
+ char *tmp = tf_reallocarray_checked(*buf, new_cap, sizeof(char));
276
+ if (!tmp) return TF_ERROR;
277
+ *buf = tmp;
278
+ *cap = new_cap;
279
+ }
280
+ memcpy(*buf + *len, src, n);
281
+ *len += n;
282
+ (*buf)[*len] = '\0';
283
+ return TF_OK;
284
+ }
285
+
286
+ static int append_str(char **buf, size_t *len, size_t *cap, const char *s) {
287
+ return append_bytes(buf, len, cap, s, strlen(s));
288
+ }
289
+
290
+ static int append_fmt(char **buf, size_t *len, size_t *cap, const char *fmt, long long v) {
291
+ char tmp[64];
292
+ int n = snprintf(tmp, sizeof(tmp), fmt, v);
293
+ if (n < 0) return TF_ERROR;
294
+ return append_bytes(buf, len, cap, tmp, (size_t)n);
295
+ }
296
+
297
+ static char *build_row_key(const tf_batch *b, size_t row, int *col_indices, size_t n_keys) {
298
+ char *buf = NULL;
299
+ size_t len = 0, cap = 0;
300
+ for (size_t k = 0; k < n_keys; k++) {
301
+ int c = col_indices[k];
302
+ if (c < 0) continue;
303
+ if (k > 0 && append_str(&buf, &len, &cap, "|") != TF_OK) goto fail;
304
+ if (tf_batch_is_null(b, row, c)) {
305
+ char tmp[32];
306
+ snprintf(tmp, sizeof(tmp), "N:%d", (int)b->col_types[c]);
307
+ if (append_str(&buf, &len, &cap, tmp) != TF_OK) goto fail;
308
+ continue;
309
+ }
310
+ switch (b->col_types[c]) {
311
+ case TF_TYPE_BOOL:
312
+ if (append_str(&buf, &len, &cap, tf_batch_get_bool(b, row, c) ? "B:1" : "B:0") != TF_OK) goto fail;
313
+ break;
314
+ case TF_TYPE_INT64:
315
+ if (append_fmt(&buf, &len, &cap, "I:%lld", (long long)tf_batch_get_int64(b, row, c)) != TF_OK) goto fail;
316
+ break;
317
+ case TF_TYPE_FLOAT64: {
318
+ char tmp[80];
319
+ int n = snprintf(tmp, sizeof(tmp), "F:%.17g", tf_batch_get_float64(b, row, c));
320
+ if (n < 0 || append_bytes(&buf, &len, &cap, tmp, (size_t)n) != TF_OK) goto fail;
321
+ break;
322
+ }
323
+ case TF_TYPE_STRING: {
324
+ const char *s = tf_batch_get_string(b, row, c);
325
+ char tmp[64];
326
+ int n = snprintf(tmp, sizeof(tmp), "S:%zu:", s ? strlen(s) : 0u);
327
+ if (n < 0 || append_bytes(&buf, &len, &cap, tmp, (size_t)n) != TF_OK) goto fail;
328
+ if (s && append_str(&buf, &len, &cap, s) != TF_OK) goto fail;
329
+ break;
330
+ }
331
+ case TF_TYPE_DATE:
332
+ if (append_fmt(&buf, &len, &cap, "D:%lld", (long long)tf_batch_get_date(b, row, c)) != TF_OK) goto fail;
333
+ break;
334
+ case TF_TYPE_TIMESTAMP:
335
+ if (append_fmt(&buf, &len, &cap, "T:%lld", (long long)tf_batch_get_timestamp(b, row, c)) != TF_OK) goto fail;
336
+ break;
337
+ default:
338
+ if (append_str(&buf, &len, &cap, "U") != TF_OK) goto fail;
339
+ break;
340
+ }
341
+ }
342
+ if (!buf) buf = strdup("");
343
+ return buf;
344
+ fail:
345
+ free(buf);
346
+ return NULL;
347
+ }
348
+
349
+ static tf_batch *create_buffer_from_schema(const unique_state *st, size_t capacity) {
350
+ tf_batch *b = tf_batch_create(st->n_schema_cols, capacity ? capacity : 16);
351
+ if (!b) return NULL;
352
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
353
+ if (tf_batch_set_schema(b, c, st->schema_names[c], st->schema_types[c]) != TF_OK) {
354
+ tf_batch_free(b);
355
+ return NULL;
356
+ }
357
+ }
358
+ return b;
359
+ }
360
+
361
+ static size_t unique_estimated_row_bytes(const unique_state *st) {
362
+ size_t bytes = 40;
363
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
364
+ bytes += 1;
365
+ switch (st->schema_types[c]) {
366
+ case TF_TYPE_BOOL: bytes += 1; break;
367
+ case TF_TYPE_INT64: bytes += sizeof(int64_t); break;
368
+ case TF_TYPE_FLOAT64: bytes += sizeof(double); break;
369
+ case TF_TYPE_STRING: bytes += sizeof(char *) + 64; break;
370
+ case TF_TYPE_DATE: bytes += sizeof(int32_t); break;
371
+ case TF_TYPE_TIMESTAMP: bytes += sizeof(int64_t); break;
372
+ default: break;
373
+ }
374
+ }
375
+ return bytes < 64 ? 64 : bytes;
376
+ }
377
+
378
+ static int ensure_ordinals(uint64_t **ord, size_t *cap, size_t need) {
379
+ if (*cap >= need) return TF_OK;
380
+ size_t new_cap = 0;
381
+ if (tf_size_grow_pow2(*cap, need, 16, &new_cap) != TF_OK) return TF_ERROR;
382
+ uint64_t *tmp = tf_reallocarray_checked(*ord, new_cap, sizeof(uint64_t));
383
+ if (!tmp) return TF_ERROR;
384
+ *ord = tmp;
385
+ *cap = new_cap;
386
+ return TF_OK;
387
+ }
388
+
389
+ static int init_spill_schema(unique_state *st, const tf_batch *in) {
390
+ if (st->has_schema) return TF_OK;
391
+ st->n_schema_cols = in->n_cols;
392
+ st->schema_names = tf_callocarray_checked(in->n_cols ? in->n_cols : 1, sizeof(char *));
393
+ st->schema_types = tf_callocarray_checked(in->n_cols ? in->n_cols : 1, sizeof(tf_type));
394
+ if (!st->schema_names || !st->schema_types) return TF_ERROR;
395
+ for (size_t c = 0; c < in->n_cols; c++) {
396
+ st->schema_names[c] = strdup(in->col_names[c] ? in->col_names[c] : "");
397
+ if (!st->schema_names[c]) return TF_ERROR;
398
+ st->schema_types[c] = in->col_types[c];
399
+ }
400
+
401
+ if (st->n_key_cols > 0) {
402
+ st->n_keys = st->n_key_cols;
403
+ st->key_indices = tf_callocarray_checked(st->n_keys ? st->n_keys : 1, sizeof(int));
404
+ if (!st->key_indices) return TF_ERROR;
405
+ for (size_t k = 0; k < st->n_keys; k++) {
406
+ int idx = tf_batch_col_index(in, st->key_cols[k]);
407
+ if (idx < 0) {
408
+ char msg[256];
409
+ snprintf(msg, sizeof(msg), "unique: column '%s' not found",
410
+ st->key_cols[k] ? st->key_cols[k] : "");
411
+ tf_set_last_error(msg);
412
+ return TF_ERROR;
413
+ }
414
+ st->key_indices[k] = idx;
415
+ }
416
+ } else {
417
+ st->n_keys = in->n_cols;
418
+ st->key_indices = tf_callocarray_checked(st->n_keys ? st->n_keys : 1, sizeof(int));
419
+ if (!st->key_indices) return TF_ERROR;
420
+ for (size_t k = 0; k < st->n_keys; k++) st->key_indices[k] = (int)k;
421
+ }
422
+
423
+ if (st->configured_run_rows > 0) {
424
+ st->run_rows = st->configured_run_rows;
425
+ } else if (st->spill_memory_bytes > 0) {
426
+ size_t row_bytes = unique_estimated_row_bytes(st);
427
+ st->run_rows = st->spill_memory_bytes / (row_bytes * 4);
428
+ if (st->run_rows < UNIQUE_MIN_RUN_ROWS) st->run_rows = UNIQUE_MIN_RUN_ROWS;
429
+ } else {
430
+ st->run_rows = UNIQUE_DEFAULT_RUN_ROWS;
431
+ }
432
+ st->buf = create_buffer_from_schema(st, st->run_rows);
433
+ st->out_buf = create_buffer_from_schema(st, st->run_rows);
434
+ if (!st->buf || !st->out_buf) return TF_ERROR;
435
+ st->has_schema = 1;
436
+ return TF_OK;
437
+ }
438
+
439
+ static int compare_batch_key_rows(const unique_state *st, const tf_batch *batch, size_t ra, size_t rb) {
440
+ for (size_t k = 0; k < st->n_keys; k++) {
441
+ int ci = st->key_indices[k];
442
+ if (ci < 0) continue;
443
+ int null_a = tf_batch_is_null(batch, ra, (size_t)ci);
444
+ int null_b = tf_batch_is_null(batch, rb, (size_t)ci);
445
+ if (null_a && null_b) continue;
446
+ if (null_a) return 1;
447
+ if (null_b) return -1;
448
+ int cmp = 0;
449
+ switch (batch->col_types[ci]) {
450
+ case TF_TYPE_BOOL: cmp = (int)tf_batch_get_bool(batch, ra, (size_t)ci) - (int)tf_batch_get_bool(batch, rb, (size_t)ci); break;
451
+ case TF_TYPE_INT64: {
452
+ int64_t a = tf_batch_get_int64(batch, ra, (size_t)ci), b = tf_batch_get_int64(batch, rb, (size_t)ci);
453
+ cmp = (a > b) - (a < b);
454
+ break;
455
+ }
456
+ case TF_TYPE_FLOAT64: {
457
+ double a = tf_batch_get_float64(batch, ra, (size_t)ci), b = tf_batch_get_float64(batch, rb, (size_t)ci);
458
+ cmp = (a > b) - (a < b);
459
+ break;
460
+ }
461
+ case TF_TYPE_STRING: cmp = strcmp(tf_batch_get_string(batch, ra, (size_t)ci), tf_batch_get_string(batch, rb, (size_t)ci)); break;
462
+ case TF_TYPE_DATE: {
463
+ int32_t a = tf_batch_get_date(batch, ra, (size_t)ci), b = tf_batch_get_date(batch, rb, (size_t)ci);
464
+ cmp = (a > b) - (a < b);
465
+ break;
466
+ }
467
+ case TF_TYPE_TIMESTAMP: {
468
+ int64_t a = tf_batch_get_timestamp(batch, ra, (size_t)ci), b = tf_batch_get_timestamp(batch, rb, (size_t)ci);
469
+ cmp = (a > b) - (a < b);
470
+ break;
471
+ }
472
+ default: break;
473
+ }
474
+ if (cmp != 0) return cmp;
475
+ }
476
+ uint64_t oa = st->buf_ordinals[ra], ob = st->buf_ordinals[rb];
477
+ return (oa > ob) - (oa < ob);
478
+ }
479
+
480
+ typedef struct {
481
+ const unique_state *st;
482
+ int by_ordinal;
483
+ } unique_sort_ctx;
484
+
485
+ static int unique_compare_indices(const void *ctx, size_t a, size_t b) {
486
+ const unique_sort_ctx *sort = (const unique_sort_ctx *)ctx;
487
+ if (!sort->by_ordinal) return compare_batch_key_rows(sort->st, sort->st->buf, a, b);
488
+ uint64_t oa = sort->st->out_ordinals[a];
489
+ uint64_t ob = sort->st->out_ordinals[b];
490
+ return (oa > ob) - (oa < ob);
491
+ }
492
+
493
+ static size_t *sorted_indices(size_t n, int by_ordinal, const unique_state *st) {
494
+ size_t *idx = tf_mallocarray_checked(n ? n : 1, sizeof(size_t));
495
+ if (!idx) return NULL;
496
+ for (size_t i = 0; i < n; i++) idx[i] = i;
497
+ unique_sort_ctx ctx = { .st = st, .by_ordinal = by_ordinal };
498
+ tf_sort_indices(idx, n, unique_compare_indices, &ctx);
499
+ return idx;
500
+ }
501
+
502
+ static int write_exact(FILE *f, const void *ptr, size_t len) {
503
+ return fwrite(ptr, 1, len, f) == len ? TF_OK : TF_ERROR;
504
+ }
505
+
506
+ static int read_exact(FILE *f, void *ptr, size_t len) {
507
+ return fread(ptr, 1, len, f) == len ? TF_OK : TF_ERROR;
508
+ }
509
+
510
+ static int write_cell(FILE *f, const tf_batch *b, size_t r, size_t c) {
511
+ uint8_t is_null = tf_batch_is_null(b, r, c) ? 1 : 0;
512
+ if (write_exact(f, &is_null, sizeof(is_null)) != TF_OK) return TF_ERROR;
513
+ if (is_null) return TF_OK;
514
+ switch (b->col_types[c]) {
515
+ case TF_TYPE_BOOL: { uint8_t v = tf_batch_get_bool(b, r, c) ? 1 : 0; return write_exact(f, &v, sizeof(v)); }
516
+ case TF_TYPE_INT64: { int64_t v = tf_batch_get_int64(b, r, c); return write_exact(f, &v, sizeof(v)); }
517
+ case TF_TYPE_FLOAT64: { double v = tf_batch_get_float64(b, r, c); return write_exact(f, &v, sizeof(v)); }
518
+ case TF_TYPE_STRING: {
519
+ const char *s = tf_batch_get_string(b, r, c);
520
+ uint64_t len = s ? (uint64_t)strlen(s) : 0;
521
+ if (write_exact(f, &len, sizeof(len)) != TF_OK) return TF_ERROR;
522
+ return len ? write_exact(f, s, (size_t)len) : TF_OK;
523
+ }
524
+ case TF_TYPE_DATE: { int32_t v = tf_batch_get_date(b, r, c); return write_exact(f, &v, sizeof(v)); }
525
+ case TF_TYPE_TIMESTAMP: { int64_t v = tf_batch_get_timestamp(b, r, c); return write_exact(f, &v, sizeof(v)); }
526
+ default: return TF_OK;
527
+ }
528
+ }
529
+
530
+ static int append_path(char ***paths, size_t *n, size_t *cap, char *path) {
531
+ if (*n == *cap) {
532
+ size_t need = 0;
533
+ size_t new_cap = 0;
534
+ if (tf_size_add(*n, 1, &need) != TF_OK ||
535
+ tf_size_grow_pow2(*cap, need, 8, &new_cap) != TF_OK) {
536
+ return TF_ERROR;
537
+ }
538
+ char **tmp = tf_reallocarray_checked(*paths, new_cap, sizeof(char *));
539
+ if (!tmp) return TF_ERROR;
540
+ *paths = tmp;
541
+ *cap = new_cap;
542
+ }
543
+ (*paths)[(*n)++] = path;
544
+ return TF_OK;
545
+ }
546
+
547
+ static int write_batch_run(unique_state *st, tf_batch *batch, const uint64_t *ordinals,
548
+ size_t *indices, size_t n, int output_run) {
549
+ char *path = NULL;
550
+ FILE *f = tf_spill_open_run_file(st->spill, output_run ? "unique-out" : "unique-key", &path);
551
+ if (!f) return TF_ERROR;
552
+ for (size_t i = 0; i < n; i++) {
553
+ size_t r = indices[i];
554
+ uint64_t ordinal = ordinals[r];
555
+ if (write_exact(f, &ordinal, sizeof(ordinal)) != TF_OK) goto write_fail;
556
+ for (size_t c = 0; c < batch->n_cols; c++) {
557
+ if (write_cell(f, batch, r, c) != TF_OK) goto write_fail;
558
+ }
559
+ }
560
+ long pos = ftell(f);
561
+ if (pos > 0) st->spilled_bytes += (size_t)pos;
562
+ if (fclose(f) != 0) {
563
+ tf_set_last_error("unique spill: failed closing run file");
564
+ remove(path);
565
+ free(path);
566
+ return TF_ERROR;
567
+ }
568
+ if (output_run) {
569
+ if (append_path(&st->out_run_paths, &st->n_out_runs, &st->cap_out_runs, path) != TF_OK) {
570
+ remove(path); free(path); return TF_ERROR;
571
+ }
572
+ } else {
573
+ if (append_path(&st->run_paths, &st->n_runs, &st->cap_runs, path) != TF_OK) {
574
+ remove(path); free(path); return TF_ERROR;
575
+ }
576
+ }
577
+ st->spill_runs_created++;
578
+ return TF_OK;
579
+
580
+ write_fail:
581
+ tf_set_last_error("unique spill: failed writing run file");
582
+ fclose(f);
583
+ remove(path);
584
+ free(path);
585
+ return TF_ERROR;
586
+ }
587
+
588
+ static int write_key_run(unique_state *st) {
589
+ if (!st->buf || st->buf->n_rows == 0) return TF_OK;
590
+ size_t *idx = sorted_indices(st->buf->n_rows, 0, st);
591
+ if (!idx) return TF_ERROR;
592
+ int rc = write_batch_run(st, st->buf, st->buf_ordinals, idx, st->buf->n_rows, 0);
593
+ free(idx);
594
+ if (rc != TF_OK) return TF_ERROR;
595
+ tf_batch_free(st->buf);
596
+ st->buf = create_buffer_from_schema(st, st->run_rows);
597
+ st->buf_ordinal_cap = 0;
598
+ free(st->buf_ordinals);
599
+ st->buf_ordinals = NULL;
600
+ return st->buf ? TF_OK : TF_ERROR;
601
+ }
602
+
603
+ static int write_output_run(unique_state *st) {
604
+ if (!st->out_buf || st->out_buf->n_rows == 0) return TF_OK;
605
+ size_t *idx = sorted_indices(st->out_buf->n_rows, 1, st);
606
+ if (!idx) return TF_ERROR;
607
+ int rc = write_batch_run(st, st->out_buf, st->out_ordinals, idx, st->out_buf->n_rows, 1);
608
+ free(idx);
609
+ if (rc != TF_OK) return TF_ERROR;
610
+ tf_batch_free(st->out_buf);
611
+ st->out_buf = create_buffer_from_schema(st, st->run_rows);
612
+ st->out_ordinal_cap = 0;
613
+ free(st->out_ordinals);
614
+ st->out_ordinals = NULL;
615
+ return st->out_buf ? TF_OK : TF_ERROR;
616
+ }
617
+
618
+ static void spill_row_clear(unique_spill_row *row, const tf_type *types, size_t n_cols) {
619
+ if (!row || !row->cells || !row->nulls) return;
620
+ for (size_t c = 0; c < n_cols; c++) {
621
+ if (!row->nulls[c] && types[c] == TF_TYPE_STRING) free(row->cells[c].str);
622
+ row->cells[c].str = NULL;
623
+ row->nulls[c] = 1;
624
+ }
625
+ }
626
+
627
+ static int spill_row_init(unique_spill_row *row, size_t n_cols) {
628
+ row->ordinal = 0;
629
+ row->nulls = tf_callocarray_checked(n_cols ? n_cols : 1, sizeof(uint8_t));
630
+ row->cells = tf_callocarray_checked(n_cols ? n_cols : 1, sizeof(unique_cell));
631
+ if (!row->nulls || !row->cells) {
632
+ free(row->nulls);
633
+ free(row->cells);
634
+ row->nulls = NULL;
635
+ row->cells = NULL;
636
+ return TF_ERROR;
637
+ }
638
+ for (size_t c = 0; c < n_cols; c++) row->nulls[c] = 1;
639
+ return TF_OK;
640
+ }
641
+
642
+ static void spill_row_free(unique_spill_row *row, const tf_type *types, size_t n_cols) {
643
+ if (!row) return;
644
+ spill_row_clear(row, types, n_cols);
645
+ free(row->nulls);
646
+ free(row->cells);
647
+ row->nulls = NULL;
648
+ row->cells = NULL;
649
+ }
650
+
651
+ static int read_cell_value(FILE *f, unique_spill_row *row, const tf_type *types, size_t c) {
652
+ switch (types[c]) {
653
+ case TF_TYPE_BOOL: { uint8_t v = 0; if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].b = v; return TF_OK; }
654
+ case TF_TYPE_INT64: { int64_t v = 0; if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].i64 = v; return TF_OK; }
655
+ case TF_TYPE_FLOAT64: { double v = 0; if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].f64 = v; return TF_OK; }
656
+ case TF_TYPE_STRING: {
657
+ uint64_t len = 0;
658
+ if (read_exact(f, &len, sizeof(len)) != TF_OK) return TF_ERROR;
659
+ if (len > (uint64_t)SIZE_MAX - 1) return TF_ERROR;
660
+ char *s = malloc((size_t)len + 1);
661
+ if (!s) return TF_ERROR;
662
+ if (len && read_exact(f, s, (size_t)len) != TF_OK) { free(s); return TF_ERROR; }
663
+ s[len] = '\0';
664
+ row->cells[c].str = s;
665
+ return TF_OK;
666
+ }
667
+ case TF_TYPE_DATE: { int32_t v = 0; if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].date = v; return TF_OK; }
668
+ case TF_TYPE_TIMESTAMP: { int64_t v = 0; if (read_exact(f, &v, sizeof(v)) != TF_OK) return TF_ERROR; row->cells[c].i64 = v; return TF_OK; }
669
+ default: return TF_OK;
670
+ }
671
+ }
672
+
673
+ static int reader_advance(unique_state *st, unique_run_reader *reader) {
674
+ if (!reader || !reader->file || reader->done) return 0;
675
+ spill_row_clear(&reader->row, st->schema_types, st->n_schema_cols);
676
+ if (fread(&reader->row.ordinal, sizeof(reader->row.ordinal), 1, reader->file) != 1) {
677
+ if (feof(reader->file)) {
678
+ reader->done = 1;
679
+ reader->has_row = 0;
680
+ return 0;
681
+ }
682
+ tf_set_last_error("unique spill: failed reading run file");
683
+ return -1;
684
+ }
685
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
686
+ uint8_t is_null = 1;
687
+ if (read_exact(reader->file, &is_null, sizeof(is_null)) != TF_OK) {
688
+ tf_set_last_error("unique spill: corrupt run file");
689
+ return -1;
690
+ }
691
+ reader->row.nulls[c] = is_null ? 1 : 0;
692
+ if (!reader->row.nulls[c] && read_cell_value(reader->file, &reader->row, st->schema_types, c) != TF_OK) {
693
+ tf_set_last_error("unique spill: corrupt run file");
694
+ return -1;
695
+ }
696
+ }
697
+ reader->has_row = 1;
698
+ return 1;
699
+ }
700
+
701
+ static int compare_spill_key_rows(const unique_state *st, const unique_spill_row *a, const unique_spill_row *b) {
702
+ for (size_t k = 0; k < st->n_keys; k++) {
703
+ int ci = st->key_indices[k];
704
+ if (ci < 0) continue;
705
+ size_t c = (size_t)ci;
706
+ int null_a = a->nulls[c] != 0;
707
+ int null_b = b->nulls[c] != 0;
708
+ if (null_a && null_b) continue;
709
+ if (null_a) return 1;
710
+ if (null_b) return -1;
711
+ int cmp = 0;
712
+ switch (st->schema_types[c]) {
713
+ case TF_TYPE_BOOL: cmp = (int)a->cells[c].b - (int)b->cells[c].b; break;
714
+ case TF_TYPE_INT64:
715
+ case TF_TYPE_TIMESTAMP: cmp = (a->cells[c].i64 > b->cells[c].i64) - (a->cells[c].i64 < b->cells[c].i64); break;
716
+ case TF_TYPE_FLOAT64: cmp = (a->cells[c].f64 > b->cells[c].f64) - (a->cells[c].f64 < b->cells[c].f64); break;
717
+ case TF_TYPE_STRING: cmp = strcmp(a->cells[c].str, b->cells[c].str); break;
718
+ case TF_TYPE_DATE: cmp = (a->cells[c].date > b->cells[c].date) - (a->cells[c].date < b->cells[c].date); break;
719
+ default: break;
720
+ }
721
+ if (cmp != 0) return cmp;
722
+ }
723
+ return (a->ordinal > b->ordinal) - (a->ordinal < b->ordinal);
724
+ }
725
+
726
+ static char *build_spill_row_key(const unique_state *st, const unique_spill_row *row) {
727
+ char *buf = NULL;
728
+ size_t len = 0, cap = 0;
729
+ for (size_t k = 0; k < st->n_keys; k++) {
730
+ int ci = st->key_indices[k];
731
+ if (ci < 0) continue;
732
+ size_t c = (size_t)ci;
733
+ if (k > 0 && append_str(&buf, &len, &cap, "|") != TF_OK) goto fail;
734
+ if (row->nulls[c]) {
735
+ char tmp[32];
736
+ snprintf(tmp, sizeof(tmp), "N:%d", (int)st->schema_types[c]);
737
+ if (append_str(&buf, &len, &cap, tmp) != TF_OK) goto fail;
738
+ continue;
739
+ }
740
+ switch (st->schema_types[c]) {
741
+ case TF_TYPE_BOOL:
742
+ if (append_str(&buf, &len, &cap, row->cells[c].b ? "B:1" : "B:0") != TF_OK) goto fail;
743
+ break;
744
+ case TF_TYPE_INT64:
745
+ if (append_fmt(&buf, &len, &cap, "I:%lld", (long long)row->cells[c].i64) != TF_OK) goto fail;
746
+ break;
747
+ case TF_TYPE_FLOAT64: {
748
+ char tmp[80];
749
+ int n = snprintf(tmp, sizeof(tmp), "F:%.17g", row->cells[c].f64);
750
+ if (n < 0 || append_bytes(&buf, &len, &cap, tmp, (size_t)n) != TF_OK) goto fail;
751
+ break;
752
+ }
753
+ case TF_TYPE_STRING: {
754
+ const char *s = row->cells[c].str;
755
+ char tmp[64];
756
+ int n = snprintf(tmp, sizeof(tmp), "S:%zu:", s ? strlen(s) : 0u);
757
+ if (n < 0 || append_bytes(&buf, &len, &cap, tmp, (size_t)n) != TF_OK) goto fail;
758
+ if (s && append_str(&buf, &len, &cap, s) != TF_OK) goto fail;
759
+ break;
760
+ }
761
+ case TF_TYPE_DATE:
762
+ if (append_fmt(&buf, &len, &cap, "D:%lld", (long long)row->cells[c].date) != TF_OK) goto fail;
763
+ break;
764
+ case TF_TYPE_TIMESTAMP:
765
+ if (append_fmt(&buf, &len, &cap, "T:%lld", (long long)row->cells[c].i64) != TF_OK) goto fail;
766
+ break;
767
+ default:
768
+ if (append_str(&buf, &len, &cap, "U") != TF_OK) goto fail;
769
+ break;
770
+ }
771
+ }
772
+ if (!buf) buf = strdup("");
773
+ return buf;
774
+ fail:
775
+ free(buf);
776
+ return NULL;
777
+ }
778
+
779
+ static int spill_row_to_batch(const unique_state *st, tf_batch *out,
780
+ size_t dst_row, const unique_spill_row *row) {
781
+ if (tf_batch_ensure_capacity(out, dst_row + 1) != TF_OK) return TF_ERROR;
782
+ for (size_t c = 0; c < st->n_schema_cols; c++) {
783
+ if (tf_batch_set_owned_cell_value(out, dst_row, c, st->schema_types[c],
784
+ row->nulls[c], &row->cells[c]) != TF_OK) {
785
+ return TF_ERROR;
786
+ }
787
+ }
788
+ return TF_OK;
789
+ }
790
+
791
+ static void close_readers(unique_state *st, int output_readers) {
792
+ unique_run_reader **readers = output_readers ? &st->out_readers : &st->readers;
793
+ size_t *n_readers = output_readers ? &st->n_out_readers : &st->n_readers;
794
+ if (!*readers) return;
795
+ for (size_t i = 0; i < *n_readers; i++) {
796
+ if ((*readers)[i].file) fclose((*readers)[i].file);
797
+ spill_row_free(&(*readers)[i].row, st->schema_types, st->n_schema_cols);
798
+ }
799
+ free(*readers);
800
+ *readers = NULL;
801
+ *n_readers = 0;
802
+ }
803
+
804
+ static void remove_paths(char ***paths, size_t *n, size_t *cap) {
805
+ for (size_t i = 0; i < *n; i++) {
806
+ if ((*paths)[i]) {
807
+ remove((*paths)[i]);
808
+ free((*paths)[i]);
809
+ (*paths)[i] = NULL;
810
+ }
811
+ }
812
+ free(*paths);
813
+ *paths = NULL;
814
+ *n = 0;
815
+ *cap = 0;
816
+ }
817
+
818
+ static int open_readers(unique_state *st, int output_readers) {
819
+ char **paths = output_readers ? st->out_run_paths : st->run_paths;
820
+ size_t n_paths = output_readers ? st->n_out_runs : st->n_runs;
821
+ unique_run_reader **readers = output_readers ? &st->out_readers : &st->readers;
822
+ size_t *n_readers = output_readers ? &st->n_out_readers : &st->n_readers;
823
+ if (n_paths == 0) return TF_OK;
824
+ *readers = tf_callocarray_checked(n_paths, sizeof(unique_run_reader));
825
+ if (!*readers) return TF_ERROR;
826
+ *n_readers = n_paths;
827
+ for (size_t i = 0; i < n_paths; i++) {
828
+ (*readers)[i].file = fopen(paths[i], "rb");
829
+ if (!(*readers)[i].file) { tf_set_last_error("unique spill: cannot reopen run file"); return TF_ERROR; }
830
+ if (spill_row_init(&(*readers)[i].row, st->n_schema_cols) != TF_OK) return TF_ERROR;
831
+ int rc = reader_advance(st, &(*readers)[i]);
832
+ if (rc < 0) return TF_ERROR;
833
+ }
834
+ return TF_OK;
835
+ }
836
+
837
+ static int best_key_reader(const unique_state *st) {
838
+ int best = -1;
839
+ for (size_t i = 0; i < st->n_readers; i++) {
840
+ const unique_run_reader *r = &st->readers[i];
841
+ if (!r->has_row || r->done) continue;
842
+ if (best < 0) { best = (int)i; continue; }
843
+ int cmp = compare_spill_key_rows(st, &r->row, &st->readers[best].row);
844
+ if (cmp < 0 || (cmp == 0 && i < (size_t)best)) best = (int)i;
845
+ }
846
+ return best;
847
+ }
848
+
849
+ static int best_ordinal_reader(const unique_state *st) {
850
+ int best = -1;
851
+ for (size_t i = 0; i < st->n_out_readers; i++) {
852
+ const unique_run_reader *r = &st->out_readers[i];
853
+ if (!r->has_row || r->done) continue;
854
+ if (best < 0) { best = (int)i; continue; }
855
+ uint64_t a = r->row.ordinal;
856
+ uint64_t b = st->out_readers[best].row.ordinal;
857
+ if (a < b || (a == b && i < (size_t)best)) best = (int)i;
858
+ }
859
+ return best;
860
+ }
861
+
862
+ static int append_selected_row(unique_state *st, const unique_spill_row *row) {
863
+ size_t dst = st->out_buf->n_rows;
864
+ if (ensure_ordinals(&st->out_ordinals, &st->out_ordinal_cap, dst + 1) != TF_OK) return TF_ERROR;
865
+ if (spill_row_to_batch(st, st->out_buf, dst, row) != TF_OK) return TF_ERROR;
866
+ st->out_ordinals[dst] = row->ordinal;
867
+ if (tf_batch_expose_row(st->out_buf, dst) != TF_OK) return TF_ERROR;
868
+ st->spill_distinct_rows++;
869
+ if (st->out_buf->n_rows >= st->run_rows) return write_output_run(st);
870
+ return TF_OK;
871
+ }
872
+
873
+ static int produce_output_runs(unique_state *st) {
874
+ if (st->key_merge_done) return TF_OK;
875
+ if (st->buf && st->buf->n_rows > 0 && write_key_run(st) != TF_OK) return TF_ERROR;
876
+ if (st->buf) { tf_batch_free(st->buf); st->buf = NULL; }
877
+ free(st->buf_ordinals); st->buf_ordinals = NULL; st->buf_ordinal_cap = 0;
878
+
879
+ if (open_readers(st, 0) != TF_OK) return TF_ERROR;
880
+ for (;;) {
881
+ int best = best_key_reader(st);
882
+ if (best < 0) break;
883
+ unique_run_reader *reader = &st->readers[best];
884
+ char *key = build_spill_row_key(st, &reader->row);
885
+ if (!key) return TF_ERROR;
886
+ int keep = !st->last_spill_key || strcmp(st->last_spill_key, key) != 0;
887
+ if (keep) {
888
+ size_t key_bytes_delta = 0;
889
+ size_t new_spill_key_bytes = 0;
890
+ if (tf_size_add(strlen(key), 1, &key_bytes_delta) != TF_OK ||
891
+ tf_size_add(st->spill_key_bytes, key_bytes_delta,
892
+ &new_spill_key_bytes) != TF_OK) {
893
+ free(key);
894
+ return TF_ERROR;
895
+ }
896
+ free(st->last_spill_key);
897
+ st->last_spill_key = key;
898
+ st->spill_key_bytes = new_spill_key_bytes;
899
+ key = NULL;
900
+ if (append_selected_row(st, &reader->row) != TF_OK) { free(key); return TF_ERROR; }
901
+ }
902
+ free(key);
903
+ int rc = reader_advance(st, reader);
904
+ if (rc < 0) return TF_ERROR;
905
+ }
906
+ if (st->out_buf && st->out_buf->n_rows > 0 && write_output_run(st) != TF_OK) return TF_ERROR;
907
+ close_readers(st, 0);
908
+ remove_paths(&st->run_paths, &st->n_runs, &st->cap_runs);
909
+ st->key_merge_done = 1;
910
+ return TF_OK;
911
+ }
912
+
913
+ static int begin_output_merge(unique_state *st) {
914
+ if (st->output_merge_started) return TF_OK;
915
+ st->output_merge_started = 1;
916
+ if (st->out_buf) { tf_batch_free(st->out_buf); st->out_buf = NULL; }
917
+ free(st->out_ordinals); st->out_ordinals = NULL; st->out_ordinal_cap = 0;
918
+ if (st->n_out_runs == 0) {
919
+ tf_spill_cleanup(st->spill);
920
+ st->spill = NULL;
921
+ st->output_merge_done = 1;
922
+ return TF_OK;
923
+ }
924
+ return open_readers(st, 1);
925
+ }
926
+
927
+ static int output_next_batch(unique_state *st, tf_batch **out) {
928
+ *out = NULL;
929
+ if (produce_output_runs(st) != TF_OK) return TF_ERROR;
930
+ if (begin_output_merge(st) != TF_OK) return TF_ERROR;
931
+ if (st->output_merge_done) return TF_OK;
932
+
933
+ tf_batch *ob = create_buffer_from_schema(st, st->output_batch_rows);
934
+ if (!ob) return TF_ERROR;
935
+ while (ob->n_rows < st->output_batch_rows) {
936
+ int best = best_ordinal_reader(st);
937
+ if (best < 0) break;
938
+ unique_run_reader *reader = &st->out_readers[best];
939
+ size_t out_row = ob->n_rows;
940
+ if (spill_row_to_batch(st, ob, out_row, &reader->row) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
941
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
942
+ int rc = reader_advance(st, reader);
943
+ if (rc < 0) { tf_batch_free(ob); return TF_ERROR; }
944
+ }
945
+ if (ob->n_rows == 0) {
946
+ tf_batch_free(ob);
947
+ close_readers(st, 1);
948
+ remove_paths(&st->out_run_paths, &st->n_out_runs, &st->cap_out_runs);
949
+ tf_spill_cleanup(st->spill);
950
+ st->spill = NULL;
951
+ st->output_merge_done = 1;
952
+ return TF_OK;
953
+ }
954
+ st->spill_output_batches++;
955
+ st->spill_output_rows += ob->n_rows;
956
+ *out = ob;
957
+ return TF_OK;
958
+ }
959
+
960
+ static size_t unique_prev_key_bytes(const unique_state *st) {
961
+ return (st && st->have_prev_key && st->prev_key) ? strlen(st->prev_key) + 1 : 0;
962
+ }
963
+
964
+ static size_t unique_retained_state_bytes(const unique_state *st) {
965
+ if (!st) return 0;
966
+ if (st->use_spill) {
967
+ size_t bytes = 0;
968
+ if (st->buf) bytes += st->buf->capacity * (sizeof(uint64_t) + 16);
969
+ if (st->out_buf) bytes += st->out_buf->capacity * (sizeof(uint64_t) + 16);
970
+ bytes += st->n_readers * sizeof(unique_run_reader) + st->n_out_readers * sizeof(unique_run_reader);
971
+ bytes += st->last_spill_key ? strlen(st->last_spill_key) + 1 : 0;
972
+ return bytes;
973
+ }
974
+ if (st->sorted) return unique_prev_key_bytes(st);
975
+ if (st->approximate) return st->bloom_bytes;
976
+ return st->seen.key_bytes + st->seen.cap * sizeof(char *);
977
+ }
978
+
979
+ static int unique_check_state_bytes(unique_state *st, tf_side_channels *side) {
980
+ if (!st || st->max_state_bytes == 0 || st->use_spill) return 0;
981
+ size_t retained = unique_retained_state_bytes(st);
982
+ if (retained <= st->max_state_bytes) return 0;
983
+ char msg[192];
984
+ snprintf(msg, sizeof(msg),
985
+ "unique: max_state_bytes=%zu exceeded while tracking distinct keys (%zu bytes retained)",
986
+ st->max_state_bytes, retained);
987
+ if (unique_write_error(side, msg) != TF_OK) return -1;
988
+ return -1;
989
+ }
990
+
991
+ static int unique_process_spill(unique_state *st, tf_batch *in) {
992
+ if (init_spill_schema(st, in) != TF_OK) return TF_ERROR;
993
+ for (size_t r = 0; r < in->n_rows; r++) {
994
+ size_t dst = st->buf->n_rows;
995
+ if (ensure_ordinals(&st->buf_ordinals, &st->buf_ordinal_cap, dst + 1) != TF_OK) return TF_ERROR;
996
+ if (tf_batch_copy_row(st->buf, dst, in, r) != TF_OK) return TF_ERROR;
997
+ st->buf_ordinals[dst] = st->next_ordinal++;
998
+ if (tf_batch_expose_row(st->buf, dst) != TF_OK) return TF_ERROR;
999
+ if (st->buf->n_rows >= st->run_rows && write_key_run(st) != TF_OK) return TF_ERROR;
1000
+ }
1001
+ return TF_OK;
1002
+ }
1003
+
1004
+ static int unique_process(tf_step *self, tf_batch *in, tf_batch **out, tf_side_channels *side) {
1005
+ unique_state *st = self->state;
1006
+ *out = NULL;
1007
+
1008
+ if (st->use_spill) return unique_process_spill(st, in);
1009
+
1010
+ size_t n_keys;
1011
+ int *col_indices;
1012
+ if (st->n_key_cols > 0) {
1013
+ n_keys = st->n_key_cols;
1014
+ col_indices = tf_mallocarray_checked(n_keys, sizeof(int));
1015
+ if (!col_indices) return TF_ERROR;
1016
+ for (size_t k = 0; k < n_keys; k++) {
1017
+ int idx = tf_batch_col_index(in, st->key_cols[k]);
1018
+ if (idx < 0) {
1019
+ char msg[256];
1020
+ snprintf(msg, sizeof(msg), "unique: column '%s' not found",
1021
+ st->key_cols[k] ? st->key_cols[k] : "");
1022
+ tf_set_last_error(msg);
1023
+ free(col_indices);
1024
+ return TF_ERROR;
1025
+ }
1026
+ col_indices[k] = idx;
1027
+ }
1028
+ } else {
1029
+ n_keys = in->n_cols;
1030
+ col_indices = tf_mallocarray_checked(n_keys, sizeof(int));
1031
+ if (!col_indices) return TF_ERROR;
1032
+ for (size_t k = 0; k < n_keys; k++) col_indices[k] = (int)k;
1033
+ }
1034
+
1035
+ tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
1036
+ if (!ob) { free(col_indices); return TF_ERROR; }
1037
+ if (tf_batch_clone_schema(ob, in) != TF_OK) { free(col_indices); tf_batch_free(ob); return TF_ERROR; }
1038
+
1039
+ size_t out_row = 0;
1040
+ for (size_t r = 0; r < in->n_rows; r++) {
1041
+ char *key = build_row_key(in, r, col_indices, n_keys);
1042
+ if (!key) { free(col_indices); tf_batch_free(ob); return TF_ERROR; }
1043
+ int emit_row = 0;
1044
+ if (st->sorted) {
1045
+ if (!st->have_prev_key || strcmp(st->prev_key, key) != 0) {
1046
+ free(st->prev_key);
1047
+ st->prev_key = key;
1048
+ key = NULL;
1049
+ st->have_prev_key = 1;
1050
+ if (unique_check_state_bytes(st, side) != 0) { free(col_indices); tf_batch_free(ob); return TF_ERROR; }
1051
+ emit_row = 1;
1052
+ }
1053
+ } else if (st->approximate) {
1054
+ if (unique_bloom_check_add(st, key)) {
1055
+ st->approx_filtered++;
1056
+ } else {
1057
+ st->approx_inserted++;
1058
+ emit_row = 1;
1059
+ }
1060
+ } else {
1061
+ if (!hs_contains(&st->seen, key) && st->max_keys > 0 && st->seen.count >= st->max_keys) {
1062
+ if (unique_limit_error(st, side) != TF_OK) {
1063
+ free(key);
1064
+ free(col_indices);
1065
+ tf_batch_free(ob);
1066
+ return TF_ERROR;
1067
+ }
1068
+ free(key);
1069
+ free(col_indices);
1070
+ tf_batch_free(ob);
1071
+ return TF_ERROR;
1072
+ }
1073
+ int inserted = hs_insert(&st->seen, key);
1074
+ if (inserted < 0) { free(key); free(col_indices); tf_batch_free(ob); return TF_ERROR; }
1075
+ if (inserted == 1 && unique_check_state_bytes(st, side) != 0) {
1076
+ free(key);
1077
+ free(col_indices);
1078
+ tf_batch_free(ob);
1079
+ return TF_ERROR;
1080
+ }
1081
+ emit_row = inserted == 1;
1082
+ }
1083
+ free(key);
1084
+ if (!emit_row) continue;
1085
+ if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) { free(col_indices); tf_batch_free(ob); return TF_ERROR; }
1086
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) { free(col_indices); tf_batch_free(ob); return TF_ERROR; }
1087
+ out_row++;
1088
+ }
1089
+ free(col_indices);
1090
+ if (out_row > 0) *out = ob;
1091
+ else tf_batch_free(ob);
1092
+ return TF_OK;
1093
+ }
1094
+
1095
+ static int unique_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
1096
+ (void)side;
1097
+ unique_state *st = self->state;
1098
+ *out = NULL;
1099
+ if (!st->use_spill) return TF_OK;
1100
+ return output_next_batch(st, out);
1101
+ }
1102
+
1103
+ static int unique_flush_next(tf_step *self, tf_batch **out, tf_side_channels *side) {
1104
+ (void)side;
1105
+ unique_state *st = self->state;
1106
+ *out = NULL;
1107
+ if (!st->use_spill) return TF_OK;
1108
+ return output_next_batch(st, out);
1109
+ }
1110
+
1111
+ static int unique_append_stats(tf_step *self, tf_buffer *out) {
1112
+ if (!self || !self->state || !out) return TF_ERROR;
1113
+ unique_state *st = self->state;
1114
+ size_t tracked_keys = st->use_spill ? st->spill_distinct_rows : (st->sorted ? (st->have_prev_key ? 1u : 0u) : (st->approximate ? st->approx_inserted : st->seen.count));
1115
+ size_t tracked_key_bytes = st->use_spill ? st->spill_key_bytes : (st->sorted ? unique_prev_key_bytes(st) : (st->approximate ? 0u : st->seen.key_bytes));
1116
+ size_t retained = unique_retained_state_bytes(st);
1117
+ char buf[360];
1118
+ snprintf(buf, sizeof(buf),
1119
+ ",\"tracked_keys\":%zu,\"tracked_key_bytes\":%zu,"
1120
+ "\"retained_state_bytes\":%zu,\"max_state_bytes\":%zu",
1121
+ tracked_keys, tracked_key_bytes, retained, st->max_state_bytes);
1122
+ if (tf_buffer_write_str(out, buf) != TF_OK) return TF_ERROR;
1123
+ if (st->approximate) {
1124
+ snprintf(buf, sizeof(buf),
1125
+ ",\"approximate\":true,\"bloom_bytes\":%zu,"
1126
+ "\"bloom_bits\":%zu,\"bloom_hashes\":%zu,"
1127
+ "\"approx_inserted\":%zu,\"approx_filtered\":%zu",
1128
+ st->bloom_bytes, st->bloom_bits, st->bloom_hashes,
1129
+ st->approx_inserted, st->approx_filtered);
1130
+ return tf_buffer_write_str(out, buf);
1131
+ }
1132
+ if (st->use_spill) {
1133
+ snprintf(buf, sizeof(buf),
1134
+ ",\"spill_bytes\":%zu,\"spill_runs\":%zu,"
1135
+ "\"spill_output_batches\":%zu,\"spill_output_rows\":%zu,"
1136
+ "\"spill_distinct_rows\":%zu",
1137
+ st->spilled_bytes, st->spill_runs_created,
1138
+ st->spill_output_batches, st->spill_output_rows,
1139
+ st->spill_distinct_rows);
1140
+ return tf_buffer_write_str(out, buf);
1141
+ }
1142
+ return TF_OK;
1143
+ }
1144
+
1145
+ static void unique_state_free(unique_state *st) {
1146
+ if (!st) return;
1147
+ if (st->key_cols) {
1148
+ for (size_t i = 0; i < st->n_key_cols; i++) free(st->key_cols[i]);
1149
+ }
1150
+ free(st->key_cols);
1151
+ free(st->prev_key);
1152
+ free(st->last_spill_key);
1153
+ free(st->bloom);
1154
+ hs_free(&st->seen);
1155
+ if (st->buf) tf_batch_free(st->buf);
1156
+ if (st->out_buf) tf_batch_free(st->out_buf);
1157
+ free(st->buf_ordinals);
1158
+ free(st->out_ordinals);
1159
+ close_readers(st, 0);
1160
+ close_readers(st, 1);
1161
+ remove_paths(&st->run_paths, &st->n_runs, &st->cap_runs);
1162
+ remove_paths(&st->out_run_paths, &st->n_out_runs, &st->cap_out_runs);
1163
+ for (size_t i = 0; i < st->n_schema_cols; i++) free(st->schema_names ? st->schema_names[i] : NULL);
1164
+ free(st->schema_names);
1165
+ free(st->schema_types);
1166
+ free(st->key_indices);
1167
+ tf_spill_cleanup(st->spill);
1168
+ free(st->spill_dir);
1169
+ free(st);
1170
+ }
1171
+
1172
+ static void unique_destroy(tf_step *self) {
1173
+ if (self) unique_state_free(self->state);
1174
+ free(self);
1175
+ }
1176
+
1177
+ tf_step *tf_unique_create(const cJSON *args) {
1178
+ unique_state *st = calloc(1, sizeof(unique_state));
1179
+ if (!st) return NULL;
1180
+ st->output_batch_rows = UNIQUE_DEFAULT_OUTPUT_ROWS;
1181
+
1182
+ if (args) {
1183
+ cJSON *sorted = cJSON_GetObjectItemCaseSensitive(args, "sorted");
1184
+ st->sorted = cJSON_IsBool(sorted) && cJSON_IsTrue(sorted);
1185
+
1186
+ int mode_approx = -1;
1187
+ cJSON *mode = cJSON_GetObjectItemCaseSensitive(args, "mode");
1188
+ if (mode) {
1189
+ if (!cJSON_IsString(mode) || !mode->valuestring) {
1190
+ tf_set_last_error("unique: mode must be exact or approx");
1191
+ unique_state_free(st);
1192
+ return NULL;
1193
+ }
1194
+ if (strcmp(mode->valuestring, "approx") == 0) {
1195
+ mode_approx = 1;
1196
+ } else if (strcmp(mode->valuestring, "exact") == 0) {
1197
+ mode_approx = 0;
1198
+ } else {
1199
+ tf_set_last_error("unique: mode must be exact or approx");
1200
+ unique_state_free(st);
1201
+ return NULL;
1202
+ }
1203
+ }
1204
+ cJSON *approx = cJSON_GetObjectItemCaseSensitive(args, "approx");
1205
+ if (approx) {
1206
+ if (!cJSON_IsBool(approx)) {
1207
+ tf_set_last_error("unique: approx must be true or false");
1208
+ unique_state_free(st);
1209
+ return NULL;
1210
+ }
1211
+ int approx_value = cJSON_IsTrue(approx) ? 1 : 0;
1212
+ if (mode_approx >= 0 && mode_approx != approx_value) {
1213
+ tf_set_last_error("unique: mode and approx conflict");
1214
+ unique_state_free(st);
1215
+ return NULL;
1216
+ }
1217
+ st->approximate = approx_value;
1218
+ } else if (mode_approx >= 0) {
1219
+ st->approximate = mode_approx;
1220
+ }
1221
+
1222
+ size_t parsed_size = 0;
1223
+ int has_max_keys = tf_json_get_size_arg(args, "max_keys",
1224
+ 1, TF_MAX_COUNT_ARG,
1225
+ &parsed_size, "unique");
1226
+ if (has_max_keys < 0) { unique_state_free(st); return NULL; }
1227
+ if (has_max_keys > 0) st->max_keys = parsed_size;
1228
+
1229
+ int has_max_state = tf_json_get_size_arg(args, "max_state_bytes",
1230
+ 1, TF_MAX_STATE_BYTES,
1231
+ &parsed_size, "unique");
1232
+ if (has_max_state < 0) { unique_state_free(st); return NULL; }
1233
+ if (has_max_state > 0) st->max_state_bytes = parsed_size;
1234
+
1235
+ int has_bloom_bytes = tf_json_get_size_arg(args, "bloom_bytes",
1236
+ 1, TF_MAX_STATE_BYTES,
1237
+ &parsed_size, "unique");
1238
+ if (has_bloom_bytes < 0) { unique_state_free(st); return NULL; }
1239
+ if (has_bloom_bytes > 0) st->bloom_bytes = parsed_size;
1240
+
1241
+ int has_bloom_hashes = tf_json_get_size_arg(args, "bloom_hashes",
1242
+ 1, UNIQUE_MAX_BLOOM_HASHES,
1243
+ &parsed_size, "unique");
1244
+ if (has_bloom_hashes < 0) { unique_state_free(st); return NULL; }
1245
+ if (has_bloom_hashes > 0) st->bloom_hashes = parsed_size;
1246
+
1247
+ cJSON *spill_dir_j = cJSON_GetObjectItemCaseSensitive(args, "spill_dir");
1248
+ if (cJSON_IsString(spill_dir_j) && spill_dir_j->valuestring && spill_dir_j->valuestring[0]) {
1249
+ st->use_spill = 1;
1250
+ st->spill_dir = strdup(spill_dir_j->valuestring);
1251
+ if (!st->spill_dir) { unique_state_free(st); return NULL; }
1252
+ if (tf_spill_session_create(st->spill_dir, &st->spill) != TF_OK) { unique_state_free(st); return NULL; }
1253
+ int has_spill_memory = tf_json_get_size_arg(args, "spill_memory_bytes",
1254
+ 1, TF_MAX_SPILL_MEMORY_BYTES,
1255
+ &parsed_size, "unique");
1256
+ if (has_spill_memory < 0) { unique_state_free(st); return NULL; }
1257
+ if (has_spill_memory > 0) st->spill_memory_bytes = parsed_size;
1258
+ int has_spill_rows = tf_json_get_size_arg(args, "spill_run_rows",
1259
+ 1, TF_MAX_SPILL_RUN_ROWS,
1260
+ &parsed_size, "unique");
1261
+ if (has_spill_rows < 0) { unique_state_free(st); return NULL; }
1262
+ if (has_spill_rows > 0) st->configured_run_rows = parsed_size;
1263
+ int has_output_rows = tf_json_get_size_arg(args, "spill_output_rows",
1264
+ 1, TF_MAX_SPILL_OUTPUT_ROWS,
1265
+ &parsed_size, "unique");
1266
+ if (has_output_rows < 0) { unique_state_free(st); return NULL; }
1267
+ if (has_output_rows > 0) st->output_batch_rows = parsed_size;
1268
+ }
1269
+
1270
+ cJSON *columns = cJSON_GetObjectItemCaseSensitive(args, "columns");
1271
+ if (columns) {
1272
+ if (!cJSON_IsArray(columns)) {
1273
+ tf_set_last_error("unique: columns must be an array");
1274
+ unique_state_free(st);
1275
+ return NULL;
1276
+ }
1277
+ int n = cJSON_GetArraySize(columns);
1278
+ if (n > 0) {
1279
+ st->key_cols = tf_callocarray_checked((size_t)n, sizeof(char *));
1280
+ if (!st->key_cols) { unique_state_free(st); return NULL; }
1281
+ st->n_key_cols = (size_t)n;
1282
+ for (int i = 0; i < n; i++) {
1283
+ cJSON *item = cJSON_GetArrayItem(columns, i);
1284
+ if (!cJSON_IsString(item) || !item->valuestring || item->valuestring[0] == '\0') {
1285
+ tf_set_last_error("unique: column names must be non-empty strings");
1286
+ unique_state_free(st);
1287
+ return NULL;
1288
+ }
1289
+ st->key_cols[i] = strdup(item->valuestring);
1290
+ if (!st->key_cols[i]) { unique_state_free(st); return NULL; }
1291
+ }
1292
+ }
1293
+ }
1294
+ }
1295
+
1296
+ if (st->approximate) {
1297
+ if (st->sorted) {
1298
+ tf_set_last_error("unique: mode=approx and sorted=true are mutually exclusive");
1299
+ unique_state_free(st);
1300
+ return NULL;
1301
+ }
1302
+ if (st->use_spill) {
1303
+ tf_set_last_error("unique: mode=approx and spill_dir are mutually exclusive");
1304
+ unique_state_free(st);
1305
+ return NULL;
1306
+ }
1307
+ if (st->max_keys > 0 || st->max_state_bytes > 0) {
1308
+ tf_set_last_error("unique: mode=approx uses bloom_bytes, not max_keys or max_state_bytes");
1309
+ unique_state_free(st);
1310
+ return NULL;
1311
+ }
1312
+ if (st->bloom_bytes == 0) st->bloom_bytes = UNIQUE_DEFAULT_BLOOM_BYTES;
1313
+ if (st->bloom_hashes == 0) st->bloom_hashes = UNIQUE_DEFAULT_BLOOM_HASHES;
1314
+ if (tf_size_mul(st->bloom_bytes, 8, &st->bloom_bits) != TF_OK ||
1315
+ st->bloom_bits == 0) {
1316
+ tf_set_last_error("unique: bloom_bytes overflow");
1317
+ unique_state_free(st);
1318
+ return NULL;
1319
+ }
1320
+ st->bloom = tf_callocarray_checked(st->bloom_bytes, sizeof(uint8_t));
1321
+ if (!st->bloom) { unique_state_free(st); return NULL; }
1322
+ } else if (st->bloom_bytes > 0 || st->bloom_hashes > 0) {
1323
+ tf_set_last_error("unique: bloom_bytes and bloom_hashes require mode=approx");
1324
+ unique_state_free(st);
1325
+ return NULL;
1326
+ }
1327
+ if (st->use_spill && st->sorted) {
1328
+ tf_set_last_error("unique: spill_dir and sorted=true are mutually exclusive");
1329
+ unique_state_free(st);
1330
+ return NULL;
1331
+ }
1332
+ if (!st->sorted && !st->use_spill && !st->approximate && hs_init(&st->seen, 256) != 0) { unique_state_free(st); return NULL; }
1333
+
1334
+ tf_step *step = calloc(1, sizeof(tf_step));
1335
+ if (!step) { unique_state_free(st); return NULL; }
1336
+ step->process = unique_process;
1337
+ step->flush = unique_flush;
1338
+ step->flush_next = st->use_spill ? unique_flush_next : NULL;
1339
+ step->append_stats = unique_append_stats;
1340
+ step->destroy = unique_destroy;
1341
+ step->state = st;
1342
+ return step;
1343
+ }