tranfi 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +272 -40
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +348 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +8 -11
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
@@ -0,0 +1,1966 @@
1
+ #include "transform_internal.h"
2
+
3
+ #include <float.h>
4
+ #include <limits.h>
5
+ #include <math.h>
6
+ #include <stdint.h>
7
+ #include <stdlib.h>
8
+ #include <string.h>
9
+
10
+ #if defined(__i386__) || defined(__x86_64__)
11
+ #include <xmmintrin.h>
12
+ #endif
13
+
14
+ typedef struct tf_u128 {
15
+ uint64_t high;
16
+ uint64_t low;
17
+ } tf_u128;
18
+
19
+ static int u128_compare(tf_u128 left, tf_u128 right) {
20
+ if (left.high != right.high) return left.high < right.high ? -1 : 1;
21
+ if (left.low != right.low) return left.low < right.low ? -1 : 1;
22
+ return 0;
23
+ }
24
+
25
+ static tf_u128 u128_add(tf_u128 left, tf_u128 right) {
26
+ tf_u128 result;
27
+ result.low = left.low + right.low;
28
+ result.high = left.high + right.high + (result.low < left.low ? 1u : 0u);
29
+ return result;
30
+ }
31
+
32
+ static tf_u128 u128_subtract(tf_u128 left, tf_u128 right) {
33
+ tf_u128 result;
34
+ result.high = left.high - right.high - (left.low < right.low ? 1u : 0u);
35
+ result.low = left.low - right.low;
36
+ return result;
37
+ }
38
+
39
+ static tf_u128 u128_shift_right(tf_u128 value, unsigned int amount) {
40
+ tf_u128 result = {0, 0};
41
+ if (amount == 0) return value;
42
+ if (amount < 64) {
43
+ result.low = (value.low >> amount) | (value.high << (64u - amount));
44
+ result.high = value.high >> amount;
45
+ } else if (amount < 128) {
46
+ result.low = value.high >> (amount - 64u);
47
+ }
48
+ return result;
49
+ }
50
+
51
+ static uint64_t integer_sqrt_u128(tf_u128 value, tf_u128 *remainder) {
52
+ tf_u128 result = {0, 0};
53
+ tf_u128 bit = {UINT64_C(1) << 62, 0};
54
+ while (u128_compare(bit, value) > 0) bit = u128_shift_right(bit, 2);
55
+ while (bit.high != 0 || bit.low != 0) {
56
+ tf_u128 candidate = u128_add(result, bit);
57
+ if (u128_compare(value, candidate) >= 0) {
58
+ value = u128_subtract(value, candidate);
59
+ result = u128_add(u128_shift_right(result, 1), bit);
60
+ } else {
61
+ result = u128_shift_right(result, 1);
62
+ }
63
+ bit = u128_shift_right(bit, 2);
64
+ }
65
+ if (remainder) *remainder = value;
66
+ return result.low;
67
+ }
68
+
69
+ static unsigned int highest_bit_u64(uint64_t value) {
70
+ unsigned int bit = 0;
71
+ while (value >>= 1) ++bit;
72
+ return bit;
73
+ }
74
+
75
+ double tf_transform_sqrt_f64_rne(double value) {
76
+ uint64_t bits = tf_transform_double_bits(value);
77
+ uint64_t fraction = bits & UINT64_C(0x000fffffffffffff);
78
+ unsigned int encoded_exponent = (unsigned int)((bits >> 52) & 0x7ffu);
79
+ uint64_t significand;
80
+ int exponent;
81
+ tf_u128 radicand;
82
+ tf_u128 remainder;
83
+ uint64_t root;
84
+ uint64_t result_bits;
85
+
86
+ if ((bits >> 63) != 0 || encoded_exponent == 0x7ffu) return value;
87
+ if (encoded_exponent == 0 && fraction == 0) return 0.0;
88
+ if (encoded_exponent == 0) {
89
+ unsigned int top = highest_bit_u64(fraction);
90
+ unsigned int shift = 52u - top;
91
+ significand = fraction << shift;
92
+ exponent = -1022 - (int)shift;
93
+ } else {
94
+ significand = UINT64_C(0x0010000000000000) | fraction;
95
+ exponent = (int)encoded_exponent - 1023;
96
+ }
97
+ if ((exponent & 1) != 0) {
98
+ significand <<= 1;
99
+ --exponent;
100
+ }
101
+ radicand.high = significand >> 12;
102
+ radicand.low = significand << 52;
103
+ root = integer_sqrt_u128(radicand, &remainder);
104
+ if (u128_compare(remainder, (tf_u128){0, root}) > 0) ++root;
105
+ if (root == UINT64_C(0x0020000000000000)) {
106
+ root >>= 1;
107
+ exponent += 2;
108
+ }
109
+ result_bits = ((uint64_t)(exponent / 2 + 1023) << 52)
110
+ | (root & UINT64_C(0x000fffffffffffff));
111
+ return tf_transform_double_from_bits(result_bits);
112
+ }
113
+
114
+ tf_transform_code tf_transform_check_runtime_fp(tf_transform_error **error) {
115
+ volatile double minimum = DBL_MIN;
116
+ volatile double subnormal = minimum * 0.5;
117
+ if (FLT_RADIX != 2 || FLT_MANT_DIG != 24 || DBL_MANT_DIG != 53
118
+ || FLT_EVAL_METHOD != 0 || subnormal == 0.0)
119
+ return tf_transform_set_error(
120
+ error, TF_TRANSFORM_UNSUPPORTED_RUNTIME,
121
+ "runtime does not provide prepared-transform V1 floating semantics");
122
+ #if !defined(__wasm__)
123
+ if (fegetround() != FE_TONEAREST)
124
+ return tf_transform_set_error(
125
+ error, TF_TRANSFORM_UNSUPPORTED_RUNTIME,
126
+ "prepared-transform floating environment is not round-to-nearest");
127
+ #endif
128
+ #if defined(__i386__) || defined(__x86_64__)
129
+ if ((_mm_getcsr() & ((1u << 15) | (1u << 6))) != 0)
130
+ return tf_transform_set_error(
131
+ error, TF_TRANSFORM_UNSUPPORTED_RUNTIME,
132
+ "prepared-transform floating environment flushes subnormals");
133
+ #endif
134
+ return TF_TRANSFORM_OK;
135
+ }
136
+
137
+ tf_transform_code tf_transform_fp_begin(
138
+ tf_transform_fp_guard *guard, tf_transform_error **error) {
139
+ tf_transform_code code;
140
+ if (!guard) return tf_transform_set_error(
141
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "floating guard is null");
142
+ memset(guard, 0, sizeof(*guard));
143
+ #if defined(__i386__) || defined(__x86_64__)
144
+ guard->mxcsr = _mm_getcsr();
145
+ #endif
146
+ #if !defined(__wasm__)
147
+ if (fegetenv(&guard->environment) != 0)
148
+ return tf_transform_set_error(
149
+ error, TF_TRANSFORM_UNSUPPORTED_RUNTIME,
150
+ "cannot save the caller floating environment");
151
+ if (fesetround(FE_TONEAREST) != 0) {
152
+ (void)fesetenv(&guard->environment);
153
+ #if defined(__i386__) || defined(__x86_64__)
154
+ _mm_setcsr(guard->mxcsr);
155
+ #endif
156
+ return tf_transform_set_error(
157
+ error, TF_TRANSFORM_UNSUPPORTED_RUNTIME,
158
+ "cannot select round-to-nearest floating environment");
159
+ }
160
+ #endif
161
+ #if defined(__i386__) || defined(__x86_64__)
162
+ {
163
+ unsigned int active_mxcsr = _mm_getcsr();
164
+ active_mxcsr &= ~((1u << 15) | (1u << 6));
165
+ active_mxcsr &= ~(3u << 13);
166
+ _mm_setcsr(active_mxcsr);
167
+ }
168
+ #endif
169
+ guard->active = 1;
170
+ code = tf_transform_check_runtime_fp(error);
171
+ if (code != TF_TRANSFORM_OK) tf_transform_fp_end(guard);
172
+ return code;
173
+ }
174
+
175
+ void tf_transform_fp_end(tf_transform_fp_guard *guard) {
176
+ if (!guard || !guard->active) return;
177
+ #if !defined(__wasm__)
178
+ (void)fesetenv(&guard->environment);
179
+ #endif
180
+ #if defined(__i386__) || defined(__x86_64__)
181
+ _mm_setcsr(guard->mxcsr);
182
+ #endif
183
+ guard->active = 0;
184
+ }
185
+
186
+ static tf_transform_code checked_add_u64(
187
+ uint64_t left, uint64_t right, uint64_t *out,
188
+ tf_transform_error **error, const char *message) {
189
+ if (right > UINT64_MAX - left)
190
+ return tf_transform_set_error(
191
+ error, TF_TRANSFORM_RESOURCE_LIMIT, message);
192
+ *out = left + right;
193
+ return TF_TRANSFORM_OK;
194
+ }
195
+
196
+ static tf_transform_code checked_mul_size(
197
+ size_t left, size_t right, size_t *out,
198
+ tf_transform_error **error, const char *message) {
199
+ if (left != 0 && right > SIZE_MAX / left)
200
+ return tf_transform_set_error(
201
+ error, TF_TRANSFORM_RESOURCE_LIMIT, message);
202
+ *out = left * right;
203
+ return TF_TRANSFORM_OK;
204
+ }
205
+
206
+ static tf_transform_code validate_column_view(
207
+ const tf_column_view_v1 *column, size_t rows, uint32_t dtype,
208
+ size_t *input_bytes, tf_transform_error **error) {
209
+ size_t item_size = dtype == TF_VIEW_FLOAT32 ? sizeof(float) : sizeof(double);
210
+ size_t data_span = 0;
211
+ size_t validity_span = 0;
212
+ size_t last;
213
+
214
+ if (!column || !input_bytes)
215
+ return tf_transform_set_error(
216
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "column view argument is null");
217
+ if (column->abi_version != 1 || column->struct_size != sizeof(*column))
218
+ return tf_transform_set_error(
219
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "invalid column view V1 header");
220
+ if (column->stride_bytes == 0 || column->stride_bytes % item_size != 0)
221
+ return tf_transform_set_error(
222
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "invalid numeric column stride");
223
+ if (rows == 0) {
224
+ if (column->data != NULL || column->data_bytes != 0
225
+ || column->validity != NULL || column->validity_bytes != 0
226
+ || column->validity_bit_offset != 0
227
+ || column->validity_bit_stride != 0)
228
+ return tf_transform_set_error(
229
+ error, TF_TRANSFORM_INVALID_ARGUMENT,
230
+ "zero-row column must carry empty data and validity spans");
231
+ *input_bytes = 0;
232
+ return TF_TRANSFORM_OK;
233
+ }
234
+ if (!column->data || (uintptr_t)column->data % item_size != 0)
235
+ return tf_transform_set_error(
236
+ error, TF_TRANSFORM_INVALID_ARGUMENT,
237
+ "numeric column data is null or misaligned");
238
+ if (rows - 1 > (SIZE_MAX - item_size) / column->stride_bytes)
239
+ return tf_transform_set_error(
240
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "numeric column span overflows");
241
+ data_span = (rows - 1) * column->stride_bytes + item_size;
242
+ if (data_span > column->data_bytes)
243
+ return tf_transform_set_error(
244
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "numeric column data is truncated");
245
+
246
+ if (!column->validity) {
247
+ if (column->validity_bytes != 0 || column->validity_bit_offset != 0
248
+ || column->validity_bit_stride != 0)
249
+ return tf_transform_set_error(
250
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "invalid null validity span");
251
+ } else {
252
+ if (column->validity_bit_stride == 0)
253
+ return tf_transform_set_error(
254
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "validity stride must be positive");
255
+ if (rows - 1 > (SIZE_MAX - column->validity_bit_offset)
256
+ / column->validity_bit_stride)
257
+ return tf_transform_set_error(
258
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "validity bit span overflows");
259
+ last = column->validity_bit_offset
260
+ + (rows - 1) * column->validity_bit_stride;
261
+ if (last > SIZE_MAX - 8)
262
+ return tf_transform_set_error(
263
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "validity byte span overflows");
264
+ validity_span = last / 8 + 1;
265
+ if (validity_span > column->validity_bytes)
266
+ return tf_transform_set_error(
267
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "validity bitmap is truncated");
268
+ }
269
+ if (data_span > SIZE_MAX - validity_span)
270
+ return tf_transform_set_error(
271
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "column input byte count overflows");
272
+ *input_bytes = data_span + validity_span;
273
+ return TF_TRANSFORM_OK;
274
+ }
275
+
276
+ static tf_transform_code validate_table_view(
277
+ const tf_transform_schema *schema, const tf_table_view_v1 *table,
278
+ uint64_t row_limit, uint64_t byte_limit, size_t *input_bytes,
279
+ const tf_transform_runtime_copy *runtime,
280
+ tf_transform_error **error) {
281
+ size_t descriptor_bytes = 0;
282
+ size_t total = 0;
283
+ tf_transform_code code;
284
+
285
+ if (!schema || !table || !input_bytes)
286
+ return tf_transform_set_error(
287
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "table argument is null");
288
+ if (table->abi_version != 1 || table->struct_size != sizeof(*table)
289
+ || table->column_count != schema->field_count
290
+ || table->column_count == 0)
291
+ return tf_transform_set_error(
292
+ error, TF_TRANSFORM_SCHEMA_MISMATCH, "table column count does not match schema");
293
+ code = checked_mul_size(
294
+ table->column_count, sizeof(tf_column_view_v1), &descriptor_bytes,
295
+ error, "table descriptor byte count overflows");
296
+ if (code != TF_TRANSFORM_OK) return code;
297
+ if (!table->columns || table->columns_bytes != descriptor_bytes)
298
+ return tf_transform_set_error(
299
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "invalid table descriptor span");
300
+ if ((uint64_t)table->row_count > row_limit)
301
+ return tf_transform_set_error(
302
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "table row count exceeds limit");
303
+ for (size_t i = 0; i < table->column_count; ++i) {
304
+ size_t column_bytes = 0;
305
+ if (runtime && i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
306
+ code = tf_transform_poll_cancel(runtime, error);
307
+ if (code != TF_TRANSFORM_OK) return code;
308
+ code = tf_transform_check_runtime_fp(error);
309
+ if (code != TF_TRANSFORM_OK) return code;
310
+ }
311
+ code = validate_column_view(
312
+ &table->columns[i], table->row_count, schema->fields[i].dtype,
313
+ &column_bytes, error);
314
+ if (code != TF_TRANSFORM_OK) return code;
315
+ if (column_bytes > SIZE_MAX - total)
316
+ return tf_transform_set_error(
317
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "table input byte count overflows");
318
+ total += column_bytes;
319
+ }
320
+ if ((uint64_t)total > byte_limit)
321
+ return tf_transform_set_error(
322
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "table input bytes exceed limit");
323
+ *input_bytes = total;
324
+ if (!runtime) return TF_TRANSFORM_OK;
325
+ code = tf_transform_poll_cancel(runtime, error);
326
+ if (code != TF_TRANSFORM_OK) return code;
327
+ return tf_transform_check_runtime_fp(error);
328
+ }
329
+
330
+ static int column_value_is_valid(const tf_column_view_v1 *column, size_t row) {
331
+ size_t bit;
332
+ if (!column->validity) return 1;
333
+ bit = column->validity_bit_offset + row * column->validity_bit_stride;
334
+ return (int)((column->validity[bit / 8] >> (bit % 8)) & 1u);
335
+ }
336
+
337
+ static double read_numeric_value(
338
+ const tf_column_view_v1 *column, size_t row, uint32_t dtype) {
339
+ const uint8_t *address = (const uint8_t *)column->data
340
+ + row * column->stride_bytes;
341
+ if (dtype == TF_VIEW_FLOAT32) {
342
+ float value;
343
+ memcpy(&value, address, sizeof(value));
344
+ return (double)value;
345
+ }
346
+ {
347
+ double value;
348
+ memcpy(&value, address, sizeof(value));
349
+ return value;
350
+ }
351
+ }
352
+
353
+ static tf_transform_code poll_and_recheck(
354
+ const tf_transform_runtime_copy *runtime, tf_transform_error **error) {
355
+ tf_transform_code code = tf_transform_poll_cancel(runtime, error);
356
+ if (code != TF_TRANSFORM_OK) return code;
357
+ return tf_transform_check_runtime_fp(error);
358
+ }
359
+
360
+ static tf_transform_code schema_owned_metrics(
361
+ const tf_transform_schema *schema, uint64_t *resident_bytes,
362
+ uint64_t *allocation_count, const tf_transform_runtime_copy *runtime,
363
+ int recheck_fp, tf_transform_error **error) {
364
+ uint64_t resident;
365
+ uint64_t allocations;
366
+ if (!schema || !resident_bytes || !allocation_count)
367
+ return tf_transform_set_error(
368
+ error, TF_TRANSFORM_INVALID_ARGUMENT,
369
+ "schema metric argument is null");
370
+ if (schema->field_count > UINT64_MAX / sizeof(*schema->fields))
371
+ return tf_transform_set_error(
372
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
373
+ "schema resident byte count overflows");
374
+ resident = (uint64_t)schema->field_count * sizeof(*schema->fields);
375
+ allocations = 1;
376
+ for (size_t i = 0; i < schema->field_count; ++i) {
377
+ if (runtime && i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
378
+ tf_transform_code code = tf_transform_poll_cancel(runtime, error);
379
+ if (code != TF_TRANSFORM_OK) return code;
380
+ if (recheck_fp) {
381
+ code = tf_transform_check_runtime_fp(error);
382
+ if (code != TF_TRANSFORM_OK) return code;
383
+ }
384
+ }
385
+ uint64_t strings = (uint64_t)schema->fields[i].id_len + 1;
386
+ if ((uint64_t)schema->fields[i].name_len + 1 > UINT64_MAX - strings
387
+ || strings + (uint64_t)schema->fields[i].name_len + 1
388
+ > UINT64_MAX - resident)
389
+ return tf_transform_set_error(
390
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
391
+ "schema resident string bytes overflow");
392
+ resident += strings + (uint64_t)schema->fields[i].name_len + 1;
393
+ if (schema->fields[i].source_id) {
394
+ if ((uint64_t)schema->fields[i].source_id_len + 1
395
+ > UINT64_MAX - resident)
396
+ return tf_transform_set_error(
397
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
398
+ "schema source ID bytes overflow");
399
+ resident += (uint64_t)schema->fields[i].source_id_len + 1;
400
+ }
401
+ if (allocations > UINT64_MAX - 2
402
+ - (schema->fields[i].source_id ? 1u : 0u))
403
+ return tf_transform_set_error(
404
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
405
+ "schema allocation count overflows");
406
+ allocations += 2 + (schema->fields[i].source_id ? 1u : 0u);
407
+ }
408
+ *resident_bytes = resident;
409
+ *allocation_count = allocations;
410
+ return TF_TRANSFORM_OK;
411
+ }
412
+
413
+ static tf_transform_code session_check_totals(
414
+ const tf_transform_runtime_copy *runtime,
415
+ uint64_t current_resident, uint64_t added_resident,
416
+ uint64_t current_allocations, uint64_t added_allocations,
417
+ tf_transform_error **error) {
418
+ if (!runtime || added_resident > UINT64_MAX - current_resident
419
+ || added_allocations > UINT64_MAX - current_allocations
420
+ || current_resident + added_resident
421
+ > runtime->limits.max_resident_state_bytes
422
+ || current_allocations + added_allocations
423
+ > runtime->limits.max_allocations_per_session)
424
+ return tf_transform_set_error(
425
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
426
+ "prepared-transform session resource limit exceeded");
427
+ return tf_transform_poll_cancel(runtime, error);
428
+ }
429
+
430
+ static void median_store_clear(tf_transform_median_store *store) {
431
+ if (!store) return;
432
+ for (size_t i = 0; i < store->block_count; ++i) free(store->blocks[i]);
433
+ free(store->blocks);
434
+ memset(store, 0, sizeof(*store));
435
+ }
436
+
437
+ static void analyzer_median_stores_clear(tf_transform_analyzer *analyzer) {
438
+ if (!analyzer || !analyzer->median_stores) return;
439
+ for (size_t i = 0; i < analyzer->input_schema.field_count; ++i)
440
+ median_store_clear(&analyzer->median_stores[i]);
441
+ free(analyzer->median_stores);
442
+ analyzer->median_stores = NULL;
443
+ }
444
+
445
+ tf_transform_code tf_transform_analyzer_allocate_retained(
446
+ tf_transform_analyzer *analyzer, size_t bytes, void **out,
447
+ tf_transform_error **error) {
448
+ void *allocated;
449
+ tf_transform_code code;
450
+ if (out) *out = NULL;
451
+ if (!analyzer || !out || bytes == 0)
452
+ return tf_transform_set_error(
453
+ error, TF_TRANSFORM_INTERNAL,
454
+ "invalid analyzer retained-state allocation");
455
+ if ((uint64_t)bytes > analyzer->runtime.limits.max_allocation_bytes
456
+ || analyzer->allocation_count == UINT64_MAX
457
+ || analyzer->allocation_count + 1
458
+ > analyzer->runtime.limits.max_allocations_per_session
459
+ || analyzer->resident_state_bytes > UINT64_MAX - (uint64_t)bytes
460
+ || analyzer->resident_state_bytes + (uint64_t)bytes
461
+ > analyzer->runtime.limits.max_resident_state_bytes)
462
+ return tf_transform_set_error(
463
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
464
+ "retained state exceeds analyzer limits");
465
+ code = poll_and_recheck(&analyzer->runtime, error);
466
+ if (code != TF_TRANSFORM_OK) return code;
467
+ allocated = malloc(bytes);
468
+ if (!allocated)
469
+ return tf_transform_set_error(
470
+ error, TF_TRANSFORM_ALLOCATION,
471
+ "retained-state allocation failed");
472
+ ++analyzer->allocation_count;
473
+ analyzer->resident_state_bytes += (uint64_t)bytes;
474
+ *out = allocated;
475
+ return TF_TRANSFORM_OK;
476
+ }
477
+
478
+ static tf_transform_code median_store_grow_index(
479
+ tf_transform_analyzer *analyzer, tf_transform_median_store *store,
480
+ size_t required, tf_transform_error **error) {
481
+ size_t maximum;
482
+ size_t capacity;
483
+ size_t bytes;
484
+ size_t old_bytes;
485
+ uint64_t maximum_u64;
486
+ double **blocks = NULL;
487
+ tf_transform_code code;
488
+ if (required <= store->block_capacity) return TF_TRANSFORM_OK;
489
+ maximum_u64 = analyzer->runtime.limits.max_allocation_bytes
490
+ / sizeof(*blocks);
491
+ maximum = maximum_u64 > (uint64_t)SIZE_MAX
492
+ ? SIZE_MAX : (size_t)maximum_u64;
493
+ if (required > maximum)
494
+ return tf_transform_set_error(
495
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
496
+ "median block index exceeds allocation limit");
497
+ capacity = store->block_capacity == 0 ? 4 : store->block_capacity;
498
+ if (capacity > maximum) capacity = maximum;
499
+ while (capacity < required) {
500
+ if (capacity > maximum / 2) {
501
+ capacity = maximum;
502
+ break;
503
+ }
504
+ capacity *= 2;
505
+ }
506
+ if (capacity < required || capacity > SIZE_MAX / sizeof(*blocks))
507
+ return tf_transform_set_error(
508
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
509
+ "median block index size overflows");
510
+ bytes = capacity * sizeof(*blocks);
511
+ code = tf_transform_analyzer_allocate_retained(
512
+ analyzer, bytes, (void **)&blocks, error);
513
+ if (code != TF_TRANSFORM_OK) return code;
514
+ old_bytes = store->block_capacity * sizeof(*blocks);
515
+ if (store->block_count != 0) {
516
+ code = tf_transform_copy_bytes_runtime(
517
+ blocks, store->blocks,
518
+ store->block_count * sizeof(*blocks),
519
+ &analyzer->runtime, error);
520
+ if (code != TF_TRANSFORM_OK) {
521
+ free(blocks);
522
+ analyzer->resident_state_bytes -= (uint64_t)bytes;
523
+ return code;
524
+ }
525
+ }
526
+ free(store->blocks);
527
+ analyzer->resident_state_bytes -= (uint64_t)old_bytes;
528
+ store->blocks = blocks;
529
+ store->block_capacity = capacity;
530
+ return TF_TRANSFORM_OK;
531
+ }
532
+
533
+ static tf_transform_code median_store_append(
534
+ tf_transform_analyzer *analyzer, tf_transform_median_store *store,
535
+ double value, tf_transform_error **error) {
536
+ uint64_t block_index_u64;
537
+ size_t block_index;
538
+ size_t offset;
539
+ size_t block_bytes;
540
+ double *block = NULL;
541
+ tf_transform_code code;
542
+ if (!store || store->values_per_block == 0 || store->count == UINT64_MAX)
543
+ return tf_transform_set_error(
544
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
545
+ "median retained-value count exceeds limits");
546
+ block_index_u64 = store->count / (uint64_t)store->values_per_block;
547
+ if (block_index_u64 > SIZE_MAX)
548
+ return tf_transform_set_error(
549
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
550
+ "median block index overflows");
551
+ block_index = (size_t)block_index_u64;
552
+ offset = (size_t)(store->count % (uint64_t)store->values_per_block);
553
+ if (block_index == store->block_count) {
554
+ if (store->block_count == SIZE_MAX)
555
+ return tf_transform_set_error(
556
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
557
+ "median block count overflows");
558
+ code = median_store_grow_index(
559
+ analyzer, store, store->block_count + 1, error);
560
+ if (code != TF_TRANSFORM_OK) return code;
561
+ block_bytes = store->values_per_block * sizeof(*block);
562
+ code = tf_transform_analyzer_allocate_retained(
563
+ analyzer, block_bytes, (void **)&block, error);
564
+ if (code != TF_TRANSFORM_OK) return code;
565
+ store->blocks[store->block_count++] = block;
566
+ } else if (block_index > store->block_count) {
567
+ return tf_transform_set_error(
568
+ error, TF_TRANSFORM_INTERNAL,
569
+ "median retained-value block sequence is invalid");
570
+ }
571
+ store->blocks[block_index][offset] = value;
572
+ ++store->count;
573
+ return TF_TRANSFORM_OK;
574
+ }
575
+
576
+ static double median_store_get(
577
+ const tf_transform_median_store *store, uint64_t index) {
578
+ size_t block_index = (size_t)(index / (uint64_t)store->values_per_block);
579
+ size_t offset = (size_t)(index % (uint64_t)store->values_per_block);
580
+ return store->blocks[block_index][offset];
581
+ }
582
+
583
+ static void median_store_set(
584
+ tf_transform_median_store *store, uint64_t index, double value) {
585
+ size_t block_index = (size_t)(index / (uint64_t)store->values_per_block);
586
+ size_t offset = (size_t)(index % (uint64_t)store->values_per_block);
587
+ store->blocks[block_index][offset] = value;
588
+ }
589
+
590
+ static int median_value_compare(double left, double right) {
591
+ if (left < right) return -1;
592
+ if (left > right) return 1;
593
+ return 0;
594
+ }
595
+
596
+ static tf_transform_code median_sort_tick(
597
+ const tf_transform_runtime_copy *runtime, size_t *ticks,
598
+ tf_transform_error **error) {
599
+ ++*ticks;
600
+ if (*ticks < TF_TRANSFORM_CANCEL_ITERS_V1) return TF_TRANSFORM_OK;
601
+ *ticks = 0;
602
+ return poll_and_recheck(runtime, error);
603
+ }
604
+
605
+ static tf_transform_code median_store_sift_down(
606
+ tf_transform_median_store *store, uint64_t root, uint64_t end,
607
+ const tf_transform_runtime_copy *runtime, size_t *ticks,
608
+ tf_transform_error **error) {
609
+ if (end == 0) return TF_TRANSFORM_OK;
610
+ while (root <= (end - 1) / 2) {
611
+ uint64_t child = root * 2 + 1;
612
+ uint64_t candidate = root;
613
+ tf_transform_code code = median_sort_tick(runtime, ticks, error);
614
+ if (code != TF_TRANSFORM_OK) return code;
615
+ if (median_value_compare(
616
+ median_store_get(store, candidate),
617
+ median_store_get(store, child)) < 0)
618
+ candidate = child;
619
+ if (child < end) {
620
+ code = median_sort_tick(runtime, ticks, error);
621
+ if (code != TF_TRANSFORM_OK) return code;
622
+ if (median_value_compare(
623
+ median_store_get(store, candidate),
624
+ median_store_get(store, child + 1)) < 0)
625
+ candidate = child + 1;
626
+ }
627
+ if (candidate == root) return TF_TRANSFORM_OK;
628
+ {
629
+ double root_value = median_store_get(store, root);
630
+ double candidate_value = median_store_get(store, candidate);
631
+ median_store_set(store, root, candidate_value);
632
+ median_store_set(store, candidate, root_value);
633
+ }
634
+ root = candidate;
635
+ }
636
+ return TF_TRANSFORM_OK;
637
+ }
638
+
639
+ static tf_transform_code median_store_resolve(
640
+ tf_transform_median_store *store,
641
+ const tf_transform_runtime_copy *runtime,
642
+ double *out, tf_transform_error **error) {
643
+ size_t ticks = 0;
644
+ tf_transform_code code;
645
+ if (!store || !runtime || !out || store->count == 0)
646
+ return tf_transform_set_error(
647
+ error, TF_TRANSFORM_INTERNAL,
648
+ "median retained state is unavailable");
649
+ if (store->count > 1) {
650
+ uint64_t start = store->count / 2;
651
+ uint64_t end = store->count - 1;
652
+ while (start != 0) {
653
+ --start;
654
+ code = median_store_sift_down(
655
+ store, start, end, runtime, &ticks, error);
656
+ if (code != TF_TRANSFORM_OK) return code;
657
+ }
658
+ while (end != 0) {
659
+ double first = median_store_get(store, 0);
660
+ double last = median_store_get(store, end);
661
+ median_store_set(store, 0, last);
662
+ median_store_set(store, end, first);
663
+ --end;
664
+ code = median_store_sift_down(
665
+ store, 0, end, runtime, &ticks, error);
666
+ if (code != TF_TRANSFORM_OK) return code;
667
+ }
668
+ }
669
+ code = poll_and_recheck(runtime, error);
670
+ if (code != TF_TRANSFORM_OK) return code;
671
+ if ((store->count & UINT64_C(1)) != 0) {
672
+ *out = median_store_get(store, store->count / 2);
673
+ } else {
674
+ double lower = median_store_get(store, store->count / 2 - 1);
675
+ double upper = median_store_get(store, store->count / 2);
676
+ double lower_half = lower / 2.0;
677
+ double upper_half = upper / 2.0;
678
+ *out = lower_half + upper_half;
679
+ if (!tf_transform_double_is_finite(*out))
680
+ return tf_transform_set_error(
681
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
682
+ "median imputation value became nonfinite");
683
+ }
684
+ if (*out == 0.0) *out = 0.0;
685
+ return TF_TRANSFORM_OK;
686
+ }
687
+
688
+ tf_transform_code tf_transform_analyzer_create(
689
+ const tf_transform_recipe *recipe, const tf_schema_view_v1 *input_schema,
690
+ const tf_transform_runtime_v1 *runtime,
691
+ tf_transform_analyzer **out, tf_transform_error **error) {
692
+ tf_transform_analyzer *created = NULL;
693
+ tf_transform_runtime_copy runtime_copy;
694
+ tf_transform_schema schema = {0};
695
+ tf_transform_code code;
696
+ size_t state_bytes = 0;
697
+ size_t median_store_bytes = 0;
698
+ size_t median_values_per_block = 0;
699
+ uint64_t category_store_resident = 0;
700
+ uint64_t category_store_allocations = 0;
701
+ uint64_t schema_resident = 0;
702
+ uint64_t schema_allocations = 0;
703
+ uint64_t session_resident;
704
+ uint64_t session_allocations;
705
+ uint64_t analyzer_allocations;
706
+ int uses_median = 0;
707
+ int uses_categorical = 0;
708
+
709
+ if (out) *out = NULL;
710
+ tf_transform_clear_error(error);
711
+ if (!recipe || !input_schema || !out)
712
+ return tf_transform_set_error(
713
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "analyzer create argument is null");
714
+ code = tf_transform_copy_runtime(runtime, &runtime_copy, error);
715
+ if (code != TF_TRANSFORM_OK) return code;
716
+ code = tf_transform_poll_cancel(&runtime_copy, error);
717
+ if (code != TF_TRANSFORM_OK) return code;
718
+ code = tf_transform_schema_copy_runtime(
719
+ input_schema, &runtime_copy, &schema, error);
720
+ if (code != TF_TRANSFORM_OK) return code;
721
+ if (schema.field_count != recipe->column_count) {
722
+ code = tf_transform_set_error(
723
+ error, TF_TRANSFORM_SCHEMA_MISMATCH,
724
+ "recipe columns do not match the input schema");
725
+ goto fail;
726
+ }
727
+ uint64_t fixed_categories = 0;
728
+ for (size_t i = 0; i < schema.field_count; ++i) {
729
+ const tf_transform_recipe_column *column = &recipe->columns[i];
730
+ int comparison;
731
+ code = tf_transform_compare_bytes_runtime(
732
+ column->source_id, column->source_id_len,
733
+ schema.fields[i].id, schema.fields[i].id_len,
734
+ &runtime_copy, &comparison, error);
735
+ if (code != TF_TRANSFORM_OK) goto fail;
736
+ if (comparison != 0
737
+ || (column->categorical_fixed_count
738
+ && column->categorical_fixed_dtype != schema.fields[i].dtype)
739
+ || (column->kind != TF_TRANSFORM_KIND_CATEGORICAL
740
+ && column->impute == TF_TRANSFORM_IMPUTE_CONSTANT
741
+ && column->constant_dtype != schema.fields[i].dtype)) {
742
+ code = tf_transform_set_error(
743
+ error, TF_TRANSFORM_SCHEMA_MISMATCH,
744
+ "recipe source or constant dtype does not match schema");
745
+ goto fail;
746
+ }
747
+ if ((uint64_t)column->categorical_fixed_count > runtime_copy.limits.max_categories_per_column
748
+ || (uint64_t)column->categorical_fixed_count > runtime_copy.limits.max_total_categories
749
+ || fixed_categories > runtime_copy.limits.max_total_categories - (uint64_t)column->categorical_fixed_count) {
750
+ code = tf_transform_set_error(error, TF_TRANSFORM_RESOURCE_LIMIT,
751
+ "fixed dictionary exceeds analyzer category limits");
752
+ goto fail;
753
+ }
754
+ fixed_categories += (uint64_t)column->categorical_fixed_count;
755
+ if (column->kind == TF_TRANSFORM_KIND_INFER
756
+ && (column->infer_max_categories
757
+ > runtime_copy.limits.max_categories_per_column
758
+ || column->infer_max_categories
759
+ > runtime_copy.limits.max_total_categories)) {
760
+ code = tf_transform_set_error(
761
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
762
+ "inferred category bound exceeds analyzer limits");
763
+ goto fail;
764
+ }
765
+ if (column->kind != TF_TRANSFORM_KIND_NUMERIC)
766
+ uses_categorical = 1;
767
+ if (column->kind != TF_TRANSFORM_KIND_CATEGORICAL
768
+ && column->impute == TF_TRANSFORM_IMPUTE_MEDIAN)
769
+ uses_median = 1;
770
+ }
771
+ code = checked_mul_size(
772
+ schema.field_count, sizeof(tf_transform_running_stats), &state_bytes,
773
+ error, "analyzer state byte count overflows");
774
+ if (code != TF_TRANSFORM_OK) goto fail;
775
+ if (uses_median) {
776
+ uint64_t target_block_bytes = runtime_copy.limits.max_allocation_bytes;
777
+ if (target_block_bytes > TF_TRANSFORM_CANCEL_BYTES_V1)
778
+ target_block_bytes = TF_TRANSFORM_CANCEL_BYTES_V1;
779
+ median_values_per_block = (size_t)(target_block_bytes / sizeof(double));
780
+ if (median_values_per_block == 0) {
781
+ code = tf_transform_set_error(
782
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
783
+ "median value blocks exceed the allocation limit");
784
+ goto fail;
785
+ }
786
+ code = checked_mul_size(
787
+ schema.field_count, sizeof(tf_transform_median_store),
788
+ &median_store_bytes, error,
789
+ "median store byte count overflows");
790
+ if (code != TF_TRANSFORM_OK) goto fail;
791
+ if ((uint64_t)median_store_bytes
792
+ > runtime_copy.limits.max_allocation_bytes) {
793
+ code = tf_transform_set_error(
794
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
795
+ "median stores exceed the allocation limit");
796
+ goto fail;
797
+ }
798
+ }
799
+ if (uses_categorical) {
800
+ code = tf_transform_category_stores_requirements(
801
+ schema.field_count, &category_store_resident,
802
+ &category_store_allocations, error);
803
+ if (code != TF_TRANSFORM_OK) goto fail;
804
+ if (category_store_resident > runtime_copy.limits.max_allocation_bytes) {
805
+ code = tf_transform_set_error(
806
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
807
+ "categorical stores exceed the allocation limit");
808
+ goto fail;
809
+ }
810
+ }
811
+ if ((uint64_t)state_bytes > runtime_copy.limits.max_resident_state_bytes
812
+ || (uint64_t)state_bytes > runtime_copy.limits.max_allocation_bytes) {
813
+ code = tf_transform_set_error(
814
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "analyzer state exceeds limits");
815
+ goto fail;
816
+ }
817
+ code = schema_owned_metrics(
818
+ &schema, &schema_resident, &schema_allocations,
819
+ &runtime_copy, 0, error);
820
+ if (code != TF_TRANSFORM_OK) goto fail;
821
+ if ((uint64_t)state_bytes > (UINT64_MAX - sizeof(*created)) / 2)
822
+ goto resource_overflow;
823
+ session_resident = sizeof(*created) + (uint64_t)state_bytes * 2;
824
+ if ((uint64_t)median_store_bytes > UINT64_MAX - session_resident)
825
+ goto resource_overflow;
826
+ session_resident += (uint64_t)median_store_bytes;
827
+ if (category_store_resident > UINT64_MAX - session_resident)
828
+ goto resource_overflow;
829
+ session_resident += category_store_resident;
830
+ if (schema_resident > UINT64_MAX - session_resident)
831
+ goto resource_overflow;
832
+ session_resident += schema_resident;
833
+ analyzer_allocations = 4 + (uses_median ? 1u : 0u);
834
+ if (category_store_allocations > UINT64_MAX - analyzer_allocations)
835
+ goto resource_overflow;
836
+ analyzer_allocations += category_store_allocations;
837
+ if (schema_allocations > UINT64_MAX - analyzer_allocations)
838
+ goto resource_overflow;
839
+ /* Schema copy also allocates and frees one uniqueness index. */
840
+ session_allocations = schema_allocations + analyzer_allocations;
841
+ code = session_check_totals(
842
+ &runtime_copy, 0, session_resident, 0, session_allocations, error);
843
+ if (code != TF_TRANSFORM_OK) goto fail;
844
+ created = (tf_transform_analyzer *)calloc(1, sizeof(*created));
845
+ if (!created) {
846
+ code = tf_transform_set_error(
847
+ error, TF_TRANSFORM_ALLOCATION, "analyzer allocation failed");
848
+ goto fail;
849
+ }
850
+ if (schema.field_count == 0) {
851
+ code = tf_transform_set_error(
852
+ error, TF_TRANSFORM_INTERNAL, "validated schema is unexpectedly empty");
853
+ goto fail;
854
+ }
855
+ code = tf_transform_poll_cancel(&runtime_copy, error);
856
+ if (code != TF_TRANSFORM_OK) goto fail;
857
+ created->stats = (tf_transform_running_stats *)calloc(
858
+ schema.field_count, sizeof(*created->stats));
859
+ if (!created->stats) {
860
+ code = tf_transform_set_error(
861
+ error, TF_TRANSFORM_ALLOCATION, "analyzer state allocation failed");
862
+ goto fail;
863
+ }
864
+ code = tf_transform_poll_cancel(&runtime_copy, error);
865
+ if (code != TF_TRANSFORM_OK) goto fail;
866
+ created->scratch = (tf_transform_running_stats *)calloc(
867
+ schema.field_count, sizeof(*created->scratch));
868
+ if (!created->scratch) {
869
+ code = tf_transform_set_error(
870
+ error, TF_TRANSFORM_ALLOCATION,
871
+ "analyzer scratch allocation failed");
872
+ goto fail;
873
+ }
874
+ if (uses_median) {
875
+ code = tf_transform_poll_cancel(&runtime_copy, error);
876
+ if (code != TF_TRANSFORM_OK) goto fail;
877
+ created->median_stores = (tf_transform_median_store *)calloc(
878
+ schema.field_count, sizeof(*created->median_stores));
879
+ if (!created->median_stores) {
880
+ code = tf_transform_set_error(
881
+ error, TF_TRANSFORM_ALLOCATION,
882
+ "median store allocation failed");
883
+ goto fail;
884
+ }
885
+ for (size_t i = 0; i < schema.field_count; ++i) {
886
+ if (i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
887
+ code = tf_transform_poll_cancel(&runtime_copy, error);
888
+ if (code != TF_TRANSFORM_OK) goto fail;
889
+ }
890
+ if (recipe->columns[i].impute == TF_TRANSFORM_IMPUTE_MEDIAN)
891
+ created->median_stores[i].values_per_block
892
+ = median_values_per_block;
893
+ }
894
+ }
895
+ created->recipe = (tf_transform_recipe *)recipe;
896
+ tf_transform_recipe_retain(created->recipe);
897
+ created->input_schema = schema;
898
+ memset(&schema, 0, sizeof(schema));
899
+ created->runtime = runtime_copy;
900
+ created->allocation_count = session_allocations - category_store_allocations;
901
+ created->resident_state_bytes = session_resident - category_store_resident;
902
+ if (uses_categorical) {
903
+ code = tf_transform_category_stores_init(created, error);
904
+ if (code != TF_TRANSFORM_OK) goto fail;
905
+ }
906
+ created->state = TF_ANALYZER_ACTIVE;
907
+ *out = created;
908
+ return TF_TRANSFORM_OK;
909
+ resource_overflow:
910
+ code = tf_transform_set_error(
911
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
912
+ "analyzer session resource count overflows");
913
+ fail:
914
+ tf_transform_schema_clear(&schema);
915
+ if (created) {
916
+ tf_transform_recipe_release(created->recipe);
917
+ analyzer_median_stores_clear(created);
918
+ tf_transform_category_stores_clear(created);
919
+ tf_transform_schema_clear(&created->input_schema);
920
+ free(created->stats);
921
+ free(created->scratch);
922
+ free(created);
923
+ }
924
+ return code;
925
+ }
926
+
927
+ void tf_transform_analyzer_destroy(tf_transform_analyzer **analyzer) {
928
+ if (!analyzer || !*analyzer) return;
929
+ tf_transform_recipe_release((*analyzer)->recipe);
930
+ analyzer_median_stores_clear(*analyzer);
931
+ tf_transform_category_stores_clear(*analyzer);
932
+ tf_transform_schema_clear(&(*analyzer)->input_schema);
933
+ free((*analyzer)->stats);
934
+ free((*analyzer)->scratch);
935
+ free(*analyzer);
936
+ *analyzer = NULL;
937
+ }
938
+
939
+ tf_transform_code tf_transform_analyzer_push(
940
+ tf_transform_analyzer *analyzer, const tf_table_view_v1 *table,
941
+ tf_transform_error **error) {
942
+ tf_transform_running_stats *pending;
943
+ tf_transform_fp_guard guard;
944
+ tf_transform_code code;
945
+ size_t input_bytes = 0;
946
+ size_t stats_bytes;
947
+ uint64_t new_rows = 0;
948
+ uint64_t new_input_bytes = 0;
949
+
950
+ tf_transform_clear_error(error);
951
+ if (!analyzer)
952
+ return tf_transform_set_error(
953
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "analyzer push argument is null");
954
+ if (analyzer->state != TF_ANALYZER_ACTIVE)
955
+ return tf_transform_set_error(
956
+ error, TF_TRANSFORM_INVALID_STATE, "analyzer is not active");
957
+ if (!table) {
958
+ analyzer->state = TF_ANALYZER_FAILED;
959
+ return tf_transform_set_error(
960
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "analyzer table is null");
961
+ }
962
+ code = tf_transform_fp_begin(&guard, error);
963
+ if (code != TF_TRANSFORM_OK) goto failed;
964
+ analyzer->runtime.fp_guard_active = 1;
965
+ code = poll_and_recheck(&analyzer->runtime, error);
966
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
967
+ code = validate_table_view(
968
+ &analyzer->input_schema, table, analyzer->runtime.limits.max_analyzer_rows,
969
+ analyzer->runtime.limits.max_analyzer_input_bytes, &input_bytes,
970
+ &analyzer->runtime, error);
971
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
972
+ code = checked_add_u64(
973
+ analyzer->total_rows, (uint64_t)table->row_count, &new_rows,
974
+ error, "analyzer row counter overflows");
975
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
976
+ code = checked_add_u64(
977
+ analyzer->total_input_bytes, (uint64_t)input_bytes, &new_input_bytes,
978
+ error, "analyzer input byte counter overflows");
979
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
980
+ if (new_rows > analyzer->runtime.limits.max_analyzer_rows
981
+ || new_input_bytes > analyzer->runtime.limits.max_analyzer_input_bytes) {
982
+ code = tf_transform_set_error(
983
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
984
+ "cumulative analyzer input exceeds limits");
985
+ goto guarded_failed;
986
+ }
987
+ stats_bytes = analyzer->input_schema.field_count * sizeof(*pending);
988
+ if (stats_bytes == 0) {
989
+ code = tf_transform_set_error(
990
+ error, TF_TRANSFORM_INTERNAL, "analyzer has no transactional state");
991
+ goto guarded_failed;
992
+ }
993
+ pending = analyzer->scratch;
994
+ if (!pending) {
995
+ code = tf_transform_set_error(
996
+ error, TF_TRANSFORM_INTERNAL,
997
+ "analyzer scratch state is unavailable");
998
+ goto guarded_failed;
999
+ }
1000
+ code = tf_transform_copy_bytes_runtime(
1001
+ pending, analyzer->stats, stats_bytes,
1002
+ &analyzer->runtime, error);
1003
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1004
+ code = poll_and_recheck(&analyzer->runtime, error);
1005
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1006
+ if (analyzer->median_stores) {
1007
+ for (size_t i = 0; i < analyzer->input_schema.field_count; ++i) {
1008
+ if (i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1009
+ code = poll_and_recheck(&analyzer->runtime, error);
1010
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1011
+ }
1012
+ if (analyzer->recipe->columns[i].impute
1013
+ == TF_TRANSFORM_IMPUTE_MEDIAN
1014
+ && !(analyzer->recipe->columns[i].kind
1015
+ == TF_TRANSFORM_KIND_INFER
1016
+ && analyzer->stats[i].numeric_domain_failed)
1017
+ && analyzer->median_stores[i].count
1018
+ != analyzer->stats[i].observed) {
1019
+ code = tf_transform_set_error(
1020
+ error, TF_TRANSFORM_INTERNAL,
1021
+ "median retained state does not match analyzer statistics");
1022
+ goto guarded_failed;
1023
+ }
1024
+ }
1025
+ }
1026
+ if (analyzer->category_stores) {
1027
+ for (size_t i = 0; i < analyzer->input_schema.field_count; ++i) {
1028
+ if (i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1029
+ code = poll_and_recheck(&analyzer->runtime, error);
1030
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1031
+ }
1032
+ if (analyzer->recipe->columns[i].kind
1033
+ == TF_TRANSFORM_KIND_NUMERIC)
1034
+ continue;
1035
+ code = tf_transform_category_check_observed(
1036
+ analyzer, i, analyzer->stats[i].observed, error);
1037
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1038
+ }
1039
+ }
1040
+ for (size_t row = 0; row < table->row_count; ++row) {
1041
+ if (row != 0 && row % TF_TRANSFORM_CANCEL_ROWS_V1 == 0) {
1042
+ code = poll_and_recheck(&analyzer->runtime, error);
1043
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1044
+ }
1045
+ for (size_t column_index = 0;
1046
+ column_index < analyzer->input_schema.field_count; ++column_index) {
1047
+ const tf_column_view_v1 *column = &table->columns[column_index];
1048
+ tf_transform_running_stats *stats = &pending[column_index];
1049
+ double value;
1050
+ double delta;
1051
+ double mean;
1052
+ double delta2;
1053
+ double term;
1054
+ double m2;
1055
+ uint64_t observed;
1056
+ int infer_numeric = 0;
1057
+
1058
+ if (column_index != 0
1059
+ && column_index % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1060
+ code = poll_and_recheck(&analyzer->runtime, error);
1061
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1062
+ }
1063
+ if (!column_value_is_valid(column, row)) {
1064
+ if (stats->missing == UINT64_MAX) {
1065
+ code = tf_transform_set_error(
1066
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1067
+ "missing-value counter overflows");
1068
+ goto guarded_failed;
1069
+ }
1070
+ ++stats->missing;
1071
+ continue;
1072
+ }
1073
+ value = read_numeric_value(
1074
+ column, row, analyzer->input_schema.fields[column_index].dtype);
1075
+ if (tf_transform_double_is_nan(value)) {
1076
+ if (stats->missing == UINT64_MAX) {
1077
+ code = tf_transform_set_error(
1078
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1079
+ "missing-value counter overflows");
1080
+ goto guarded_failed;
1081
+ }
1082
+ ++stats->missing;
1083
+ continue;
1084
+ }
1085
+ if (!tf_transform_double_is_finite(value)) {
1086
+ code = tf_transform_set_error(
1087
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1088
+ "infinite analyzer input is not supported");
1089
+ goto guarded_failed;
1090
+ }
1091
+ if (analyzer->recipe->columns[column_index].kind
1092
+ == TF_TRANSFORM_KIND_CATEGORICAL) {
1093
+ if (stats->observed == UINT64_MAX) {
1094
+ code = tf_transform_set_error(
1095
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1096
+ "categorical observed-value counter overflows");
1097
+ goto guarded_failed;
1098
+ }
1099
+ if (analyzer->recipe->columns[column_index].kind
1100
+ == TF_TRANSFORM_KIND_CATEGORICAL
1101
+ && analyzer->recipe->columns[column_index].categorical_impute
1102
+ == TF_TRANSFORM_CATEGORICAL_IMPUTE_NONE
1103
+ && analyzer->recipe->columns[column_index].categorical_encode
1104
+ == TF_TRANSFORM_ENCODE_NONE) {
1105
+ ++stats->observed;
1106
+ continue;
1107
+ }
1108
+ code = tf_transform_category_observe(
1109
+ analyzer, column_index, value,
1110
+ analyzer->input_schema.fields[column_index].dtype, error);
1111
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1112
+ ++stats->observed;
1113
+ continue;
1114
+ }
1115
+ if (analyzer->recipe->columns[column_index].kind
1116
+ == TF_TRANSFORM_KIND_INFER) {
1117
+ double integer_part;
1118
+ int is_integer = modf(value, &integer_part) == 0.0;
1119
+ code = tf_transform_category_infer_observe(
1120
+ analyzer, column_index, value,
1121
+ analyzer->input_schema.fields[column_index].dtype,
1122
+ is_integer,
1123
+ analyzer->recipe->columns[column_index].infer_max_categories,
1124
+ &infer_numeric, error);
1125
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1126
+ if (stats->numeric_domain_failed) {
1127
+ if (infer_numeric) {
1128
+ code = tf_transform_set_error(
1129
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1130
+ "deferred numeric analyzer state is invalid");
1131
+ goto guarded_failed;
1132
+ }
1133
+ if (stats->observed == UINT64_MAX) {
1134
+ code = tf_transform_set_error(
1135
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1136
+ "observed-value counter overflows");
1137
+ goto guarded_failed;
1138
+ }
1139
+ ++stats->observed;
1140
+ continue;
1141
+ }
1142
+ }
1143
+ if (stats->observed == UINT64_MAX) {
1144
+ code = tf_transform_set_error(
1145
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1146
+ "observed-value counter overflows");
1147
+ goto guarded_failed;
1148
+ }
1149
+ observed = stats->observed + 1;
1150
+ delta = value - stats->mean;
1151
+ mean = stats->mean + delta / (double)observed;
1152
+ delta2 = value - mean;
1153
+ term = delta * delta2;
1154
+ m2 = stats->m2 + term;
1155
+ if (!tf_transform_double_is_finite(delta)
1156
+ || !tf_transform_double_is_finite(mean)
1157
+ || !tf_transform_double_is_finite(delta2)
1158
+ || !tf_transform_double_is_finite(term)
1159
+ || !tf_transform_double_is_finite(m2)) {
1160
+ if (analyzer->recipe->columns[column_index].kind
1161
+ == TF_TRANSFORM_KIND_INFER
1162
+ && !infer_numeric) {
1163
+ stats->numeric_domain_failed = 1;
1164
+ stats->observed = observed;
1165
+ continue;
1166
+ }
1167
+ code = tf_transform_set_error(
1168
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1169
+ "numeric analyzer state became nonfinite");
1170
+ goto guarded_failed;
1171
+ }
1172
+ if (analyzer->recipe->columns[column_index].impute
1173
+ == TF_TRANSFORM_IMPUTE_MEDIAN) {
1174
+ code = median_store_append(
1175
+ analyzer, &analyzer->median_stores[column_index],
1176
+ value, error);
1177
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1178
+ }
1179
+ stats->observed = observed;
1180
+ stats->mean = mean;
1181
+ stats->m2 = m2;
1182
+ if (!stats->has_value) {
1183
+ stats->minimum = value;
1184
+ stats->maximum = value;
1185
+ stats->has_value = 1;
1186
+ } else {
1187
+ if (value < stats->minimum) stats->minimum = value;
1188
+ if (value > stats->maximum) stats->maximum = value;
1189
+ }
1190
+ }
1191
+ }
1192
+ analyzer->runtime.fp_guard_active = 0;
1193
+ tf_transform_fp_end(&guard);
1194
+ {
1195
+ tf_transform_running_stats *committed = analyzer->stats;
1196
+ analyzer->stats = analyzer->scratch;
1197
+ analyzer->scratch = committed;
1198
+ }
1199
+ analyzer->total_rows = new_rows;
1200
+ analyzer->total_input_bytes = new_input_bytes;
1201
+ return TF_TRANSFORM_OK;
1202
+ guarded_failed:
1203
+ analyzer->runtime.fp_guard_active = 0;
1204
+ tf_transform_fp_end(&guard);
1205
+ failed:
1206
+ analyzer->state = TF_ANALYZER_FAILED;
1207
+ return code;
1208
+ }
1209
+
1210
+ static tf_transform_code finite_or_domain(
1211
+ double value, tf_transform_error **error, const char *message) {
1212
+ if (!tf_transform_double_is_finite(value))
1213
+ return tf_transform_set_error(
1214
+ error, TF_TRANSFORM_NUMERIC_DOMAIN, message);
1215
+ return TF_TRANSFORM_OK;
1216
+ }
1217
+
1218
+ static tf_transform_code finalize_numeric_column(
1219
+ const tf_transform_recipe_column *recipe,
1220
+ const tf_transform_running_stats *stats,
1221
+ double median_value, int has_median_value,
1222
+ tf_transform_numeric_state *state, tf_transform_error **error) {
1223
+ uint64_t logical_count = stats->observed;
1224
+ double mean = stats->mean;
1225
+ double m2 = stats->m2;
1226
+ double minimum = stats->minimum;
1227
+ double maximum = stats->maximum;
1228
+ int has_value = stats->has_value;
1229
+ double impute_value = 0.0;
1230
+ int has_impute = recipe->impute != TF_TRANSFORM_IMPUTE_NONE;
1231
+ tf_transform_code code;
1232
+
1233
+ memset(state, 0, sizeof(*state));
1234
+ state->impute = recipe->impute;
1235
+ state->all_missing = recipe->all_missing;
1236
+ state->normalize = recipe->normalize;
1237
+ state->ddof = recipe->ddof;
1238
+ if (stats->observed == 0 && stats->missing == 0
1239
+ && (recipe->impute == TF_TRANSFORM_IMPUTE_MEAN
1240
+ || recipe->impute == TF_TRANSFORM_IMPUTE_MEDIAN))
1241
+ return tf_transform_set_error(
1242
+ error, TF_TRANSFORM_INSUFFICIENT_DATA,
1243
+ "learned imputation requires at least one analyzed row");
1244
+ if (recipe->impute == TF_TRANSFORM_IMPUTE_ZERO)
1245
+ impute_value = 0.0;
1246
+ else if (recipe->impute == TF_TRANSFORM_IMPUTE_CONSTANT)
1247
+ impute_value = recipe->constant;
1248
+ else if (recipe->impute == TF_TRANSFORM_IMPUTE_MEAN) {
1249
+ if (stats->observed != 0) impute_value = stats->mean;
1250
+ else if (recipe->all_missing == TF_TRANSFORM_ALL_MISSING_ZERO)
1251
+ impute_value = 0.0;
1252
+ else return tf_transform_set_error(
1253
+ error, TF_TRANSFORM_INSUFFICIENT_DATA,
1254
+ "mean imputation has no observed values");
1255
+ } else if (recipe->impute == TF_TRANSFORM_IMPUTE_MEDIAN) {
1256
+ if (stats->observed != 0) {
1257
+ if (!has_median_value)
1258
+ return tf_transform_set_error(
1259
+ error, TF_TRANSFORM_INTERNAL,
1260
+ "median imputation value is unavailable");
1261
+ impute_value = median_value;
1262
+ } else if (recipe->all_missing == TF_TRANSFORM_ALL_MISSING_ZERO) {
1263
+ impute_value = 0.0;
1264
+ } else {
1265
+ return tf_transform_set_error(
1266
+ error, TF_TRANSFORM_INSUFFICIENT_DATA,
1267
+ "median imputation has no observed values");
1268
+ }
1269
+ }
1270
+ if (has_impute) {
1271
+ state->impute_value = impute_value;
1272
+ state->has_impute_value = 1;
1273
+ if (stats->missing != 0) {
1274
+ if (stats->observed == 0) {
1275
+ logical_count = stats->missing;
1276
+ mean = impute_value;
1277
+ m2 = 0.0;
1278
+ minimum = impute_value;
1279
+ maximum = impute_value;
1280
+ has_value = 1;
1281
+ } else {
1282
+ uint64_t total;
1283
+ double delta;
1284
+ double nm;
1285
+ double weight;
1286
+ double delta2;
1287
+ double cross;
1288
+ double missing_weight;
1289
+ double merged_mean;
1290
+ double merged_m2;
1291
+ if (stats->missing > UINT64_MAX - stats->observed)
1292
+ return tf_transform_set_error(
1293
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1294
+ "logical value count overflows");
1295
+ total = stats->observed + stats->missing;
1296
+ delta = impute_value - stats->mean;
1297
+ nm = (double)stats->observed * (double)stats->missing;
1298
+ weight = nm / (double)total;
1299
+ delta2 = delta * delta;
1300
+ cross = delta2 * weight;
1301
+ missing_weight = (double)stats->missing / (double)total;
1302
+ merged_mean = stats->mean + delta * missing_weight;
1303
+ merged_m2 = stats->m2 + cross;
1304
+ if (!tf_transform_double_is_finite(delta)
1305
+ || !tf_transform_double_is_finite(nm)
1306
+ || !tf_transform_double_is_finite(weight)
1307
+ || !tf_transform_double_is_finite(delta2)
1308
+ || !tf_transform_double_is_finite(cross)
1309
+ || !tf_transform_double_is_finite(missing_weight)
1310
+ || !tf_transform_double_is_finite(merged_mean)
1311
+ || !tf_transform_double_is_finite(merged_m2))
1312
+ return tf_transform_set_error(
1313
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1314
+ "imputed analyzer state became nonfinite");
1315
+ logical_count = total;
1316
+ mean = merged_mean;
1317
+ m2 = merged_m2;
1318
+ if (impute_value < minimum) minimum = impute_value;
1319
+ if (impute_value > maximum) maximum = impute_value;
1320
+ }
1321
+ }
1322
+ }
1323
+ if (recipe->normalize == TF_TRANSFORM_NORMALIZE_NONE) {
1324
+ state->location = 0.0;
1325
+ state->scale = 1.0;
1326
+ return TF_TRANSFORM_OK;
1327
+ }
1328
+ if (!has_value || logical_count == 0)
1329
+ return tf_transform_set_error(
1330
+ error, TF_TRANSFORM_INSUFFICIENT_DATA,
1331
+ "normalization has no logical values");
1332
+ if (recipe->normalize == TF_TRANSFORM_NORMALIZE_MINMAX) {
1333
+ double range = maximum - minimum;
1334
+ code = finite_or_domain(
1335
+ range, error, "min-max range became nonfinite");
1336
+ if (code != TF_TRANSFORM_OK) return code;
1337
+ state->location = minimum;
1338
+ state->scale = range == 0.0 ? 1.0 : range;
1339
+ return TF_TRANSFORM_OK;
1340
+ }
1341
+ if (logical_count <= (uint64_t)recipe->ddof)
1342
+ return tf_transform_set_error(
1343
+ error, TF_TRANSFORM_INSUFFICIENT_DATA,
1344
+ "standard normalization has too few logical values");
1345
+ if (m2 < 0.0)
1346
+ return tf_transform_set_error(
1347
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1348
+ "standard normalization has negative M2");
1349
+ {
1350
+ double variance = m2 / (double)(logical_count - (uint64_t)recipe->ddof);
1351
+ double scale;
1352
+ code = finite_or_domain(
1353
+ variance, error, "standard variance became nonfinite");
1354
+ if (code != TF_TRANSFORM_OK) return code;
1355
+ scale = variance == 0.0 ? 1.0 : tf_transform_sqrt_f64_rne(variance);
1356
+ if (!tf_transform_double_is_finite(scale) || scale <= 0.0)
1357
+ return tf_transform_set_error(
1358
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1359
+ "standard scale is not positive and finite");
1360
+ state->location = mean;
1361
+ state->scale = scale;
1362
+ }
1363
+ return TF_TRANSFORM_OK;
1364
+ }
1365
+
1366
+ static tf_transform_code finalize_plan_states(
1367
+ tf_transform_analyzer *analyzer, tf_transform_plan *plan,
1368
+ size_t state_bytes, tf_transform_error **error) {
1369
+ tf_transform_code code;
1370
+ if ((uint64_t)state_bytes > analyzer->runtime.limits.max_resident_state_bytes
1371
+ || (uint64_t)state_bytes
1372
+ > analyzer->runtime.limits.max_allocation_bytes)
1373
+ return tf_transform_set_error(
1374
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "plan state exceeds limits");
1375
+ code = tf_transform_poll_cancel(&analyzer->runtime, error);
1376
+ if (code != TF_TRANSFORM_OK) return code;
1377
+ plan->states = (tf_transform_column_state *)calloc(
1378
+ plan->input_schema.field_count, sizeof(*plan->states));
1379
+ if (!plan->states)
1380
+ return tf_transform_set_error(
1381
+ error, TF_TRANSFORM_ALLOCATION, "plan state allocation failed");
1382
+ for (size_t i = 0; i < plan->input_schema.field_count; ++i) {
1383
+ double median_value = 0.0;
1384
+ int has_median_value = 0;
1385
+ tf_transform_column_kind resolved_kind;
1386
+ if (i != 0 && i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1387
+ code = poll_and_recheck(&analyzer->runtime, error);
1388
+ if (code != TF_TRANSFORM_OK) return code;
1389
+ }
1390
+ code = tf_transform_analyzer_resolve_kind(
1391
+ analyzer, i, &resolved_kind, error);
1392
+ if (code != TF_TRANSFORM_OK) return code;
1393
+ if (resolved_kind == TF_TRANSFORM_KIND_CATEGORICAL) {
1394
+ plan->states[i].kind = TF_TRANSFORM_KIND_CATEGORICAL;
1395
+ code = tf_transform_category_finalize(
1396
+ analyzer, i, &plan->states[i].value.categorical, error);
1397
+ if (code != TF_TRANSFORM_OK) return code;
1398
+ continue;
1399
+ }
1400
+ if (analyzer->stats[i].numeric_domain_failed)
1401
+ return tf_transform_set_error(
1402
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1403
+ "deferred numeric analyzer state is invalid");
1404
+ plan->states[i].kind = TF_TRANSFORM_KIND_NUMERIC;
1405
+ if (plan->recipe->columns[i].impute == TF_TRANSFORM_IMPUTE_MEDIAN
1406
+ && analyzer->stats[i].observed != 0) {
1407
+ tf_transform_median_store *store = analyzer->median_stores
1408
+ ? &analyzer->median_stores[i] : NULL;
1409
+ if (!store || store->count != analyzer->stats[i].observed)
1410
+ return tf_transform_set_error(
1411
+ error, TF_TRANSFORM_INTERNAL,
1412
+ "median retained state does not match finalized statistics");
1413
+ code = median_store_resolve(
1414
+ store, &analyzer->runtime, &median_value, error);
1415
+ if (code != TF_TRANSFORM_OK) return code;
1416
+ has_median_value = 1;
1417
+ }
1418
+ code = finalize_numeric_column(
1419
+ &plan->recipe->columns[i], &analyzer->stats[i],
1420
+ median_value, has_median_value,
1421
+ &plan->states[i].value.numeric, error);
1422
+ if (code != TF_TRANSFORM_OK) return code;
1423
+ }
1424
+ return TF_TRANSFORM_OK;
1425
+ }
1426
+
1427
+ tf_transform_code tf_transform_analyzer_finalize(
1428
+ tf_transform_analyzer *analyzer, tf_transform_plan **out,
1429
+ tf_transform_error **error) {
1430
+ tf_transform_plan *plan = NULL;
1431
+ tf_transform_fp_guard guard;
1432
+ tf_transform_code code;
1433
+ size_t state_bytes = 0;
1434
+ uint64_t schema_resident = 0;
1435
+ uint64_t schema_allocations = 0;
1436
+ uint64_t output_schema_resident = 0;
1437
+ uint64_t output_schema_allocations = 0;
1438
+ uint64_t output_collision_bytes = 0;
1439
+ uint64_t category_plan_resident = 0;
1440
+ uint64_t category_plan_allocations = 0;
1441
+ uint64_t total_plan_categories = 0;
1442
+ uint64_t plan_base_resident;
1443
+ uint64_t plan_resident;
1444
+ uint64_t plan_allocations;
1445
+ uint64_t output_field_count = 0;
1446
+ int has_onehot = 0;
1447
+
1448
+ if (out) *out = NULL;
1449
+ tf_transform_clear_error(error);
1450
+ if (!analyzer)
1451
+ return tf_transform_set_error(
1452
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "analyzer finalize argument is null");
1453
+ if (analyzer->state != TF_ANALYZER_ACTIVE)
1454
+ return tf_transform_set_error(
1455
+ error, TF_TRANSFORM_INVALID_STATE, "analyzer is not active");
1456
+ if (!out) {
1457
+ analyzer->state = TF_ANALYZER_FAILED;
1458
+ return tf_transform_set_error(
1459
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "plan output is null");
1460
+ }
1461
+ code = tf_transform_fp_begin(&guard, error);
1462
+ if (code != TF_TRANSFORM_OK) goto failed;
1463
+ analyzer->runtime.fp_guard_active = 1;
1464
+ code = poll_and_recheck(&analyzer->runtime, error);
1465
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1466
+ code = checked_mul_size(
1467
+ analyzer->input_schema.field_count, sizeof(*plan->states), &state_bytes,
1468
+ error, "plan state byte count overflows");
1469
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1470
+ code = schema_owned_metrics(
1471
+ &analyzer->input_schema, &schema_resident, &schema_allocations,
1472
+ &analyzer->runtime, 1, error);
1473
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1474
+ code = tf_transform_output_schema_requirements(
1475
+ &analyzer->input_schema, analyzer->recipe, analyzer, NULL,
1476
+ &analyzer->runtime, &output_field_count, &output_schema_resident,
1477
+ &output_schema_allocations, &output_collision_bytes, error);
1478
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1479
+ (void)output_field_count;
1480
+ for (size_t i = 0; i < analyzer->input_schema.field_count; ++i) {
1481
+ uint64_t resident = 0;
1482
+ uint64_t allocations = 0;
1483
+ uint64_t count;
1484
+ tf_transform_column_kind resolved_kind;
1485
+ if (i % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1486
+ code = poll_and_recheck(&analyzer->runtime, error);
1487
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1488
+ }
1489
+ code = tf_transform_analyzer_resolve_kind(
1490
+ analyzer, i, &resolved_kind, error);
1491
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1492
+ if (resolved_kind != TF_TRANSFORM_KIND_CATEGORICAL)
1493
+ continue;
1494
+ if (analyzer->recipe->columns[i].categorical_encode
1495
+ == TF_TRANSFORM_ENCODE_ONEHOT)
1496
+ has_onehot = 1;
1497
+ code = tf_transform_category_check_observed(
1498
+ analyzer, i, analyzer->stats[i].observed, error);
1499
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1500
+ code = tf_transform_category_plan_requirements(
1501
+ analyzer, i, &resident, &allocations, error);
1502
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1503
+ count = resident / sizeof(tf_transform_category_value);
1504
+ if (category_plan_resident > UINT64_MAX - resident
1505
+ || category_plan_allocations > UINT64_MAX - allocations
1506
+ || total_plan_categories > UINT64_MAX - count) {
1507
+ code = tf_transform_set_error(
1508
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1509
+ "categorical plan resource count overflows");
1510
+ goto guarded_failed;
1511
+ }
1512
+ category_plan_resident += resident;
1513
+ category_plan_allocations += allocations;
1514
+ total_plan_categories += count;
1515
+ }
1516
+ if (total_plan_categories
1517
+ > analyzer->runtime.limits.max_total_categories) {
1518
+ code = tf_transform_set_error(
1519
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1520
+ "categorical plan exceeds the total category limit");
1521
+ goto guarded_failed;
1522
+ }
1523
+ if (schema_resident > UINT64_MAX - sizeof(*plan)
1524
+ || output_schema_resident > UINT64_MAX - sizeof(*plan)
1525
+ - schema_resident) {
1526
+ code = tf_transform_set_error(
1527
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1528
+ "plan construction resource count overflows");
1529
+ goto guarded_failed;
1530
+ }
1531
+ plan_base_resident = sizeof(*plan)
1532
+ + schema_resident + output_schema_resident;
1533
+ if ((uint64_t)state_bytes > UINT64_MAX - plan_base_resident
1534
+ || category_plan_resident > UINT64_MAX
1535
+ - plan_base_resident - (uint64_t)state_bytes
1536
+ || output_collision_bytes > UINT64_MAX - plan_base_resident
1537
+ || category_plan_allocations > UINT64_MAX - 2
1538
+ || schema_allocations > UINT64_MAX - 2 - category_plan_allocations
1539
+ || output_schema_allocations > UINT64_MAX - 2
1540
+ - category_plan_allocations - schema_allocations) {
1541
+ code = tf_transform_set_error(
1542
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1543
+ "plan construction resource count overflows");
1544
+ goto guarded_failed;
1545
+ }
1546
+ plan_resident = plan_base_resident
1547
+ + (uint64_t)state_bytes + category_plan_resident;
1548
+ if (has_onehot) {
1549
+ if (output_collision_bytes > UINT64_MAX - plan_resident) {
1550
+ code = tf_transform_set_error(
1551
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1552
+ "one-hot plan construction resource count overflows");
1553
+ goto guarded_failed;
1554
+ }
1555
+ plan_resident += output_collision_bytes;
1556
+ } else if (plan_base_resident + output_collision_bytes > plan_resident) {
1557
+ plan_resident = plan_base_resident + output_collision_bytes;
1558
+ }
1559
+ plan_allocations = 2 + schema_allocations + output_schema_allocations
1560
+ + category_plan_allocations;
1561
+ code = session_check_totals(
1562
+ &analyzer->runtime, analyzer->resident_state_bytes, plan_resident,
1563
+ analyzer->allocation_count, plan_allocations, error);
1564
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1565
+ plan = (tf_transform_plan *)calloc(1, sizeof(*plan));
1566
+ if (!plan) {
1567
+ code = tf_transform_set_error(
1568
+ error, TF_TRANSFORM_ALLOCATION, "plan allocation failed");
1569
+ goto guarded_failed;
1570
+ }
1571
+ atomic_init(&plan->refcount, 1u);
1572
+ plan->recipe = analyzer->recipe;
1573
+ tf_transform_recipe_retain(plan->recipe);
1574
+ code = tf_transform_schema_clone_runtime(
1575
+ &analyzer->input_schema, &analyzer->runtime,
1576
+ &plan->input_schema, error);
1577
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1578
+ if (has_onehot) {
1579
+ code = finalize_plan_states(analyzer, plan, state_bytes, error);
1580
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1581
+ }
1582
+ code = tf_transform_output_schema_build(
1583
+ &analyzer->input_schema, analyzer->recipe, analyzer, plan->states,
1584
+ &analyzer->runtime, &plan->output_schema, error);
1585
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1586
+ code = tf_transform_output_schema_validate_contract(
1587
+ &plan->input_schema, plan->recipe, analyzer, plan->states,
1588
+ &plan->output_schema,
1589
+ &analyzer->runtime, NULL, TF_TRANSFORM_SCHEMA_MISMATCH, error);
1590
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1591
+ if (!has_onehot) {
1592
+ code = finalize_plan_states(analyzer, plan, state_bytes, error);
1593
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1594
+ }
1595
+ analyzer->runtime.fp_guard_active = 0;
1596
+ tf_transform_fp_end(&guard);
1597
+ analyzer->allocation_count += plan_allocations;
1598
+ analyzer->state = TF_ANALYZER_FINALIZED;
1599
+ *out = plan;
1600
+ return TF_TRANSFORM_OK;
1601
+ guarded_failed:
1602
+ analyzer->runtime.fp_guard_active = 0;
1603
+ tf_transform_fp_end(&guard);
1604
+ failed:
1605
+ tf_transform_plan_release(plan);
1606
+ analyzer->state = TF_ANALYZER_FAILED;
1607
+ return code;
1608
+ }
1609
+
1610
+ tf_transform_code tf_transform_apply_create(
1611
+ const tf_transform_plan *plan, const tf_schema_view_v1 *runtime_schema,
1612
+ const tf_transform_runtime_v1 *runtime,
1613
+ tf_transform_apply **out, tf_transform_error **error) {
1614
+ tf_transform_apply *created;
1615
+ tf_transform_runtime_copy runtime_copy;
1616
+ tf_transform_code code;
1617
+ int schema_equal = 0;
1618
+
1619
+ if (out) *out = NULL;
1620
+ tf_transform_clear_error(error);
1621
+ if (!plan || !runtime_schema || !out)
1622
+ return tf_transform_set_error(
1623
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "apply create argument is null");
1624
+ code = tf_transform_copy_runtime(runtime, &runtime_copy, error);
1625
+ if (code != TF_TRANSFORM_OK) return code;
1626
+ code = tf_transform_poll_cancel(&runtime_copy, error);
1627
+ if (code != TF_TRANSFORM_OK) return code;
1628
+ code = tf_transform_schema_equal_view_runtime(
1629
+ &plan->input_schema, runtime_schema, &runtime_copy,
1630
+ &schema_equal, error);
1631
+ if (code != TF_TRANSFORM_OK) return code;
1632
+ if (!schema_equal)
1633
+ return tf_transform_set_error(
1634
+ error, TF_TRANSFORM_SCHEMA_MISMATCH,
1635
+ "runtime schema does not match fitted input schema");
1636
+ code = session_check_totals(
1637
+ &runtime_copy, 0, sizeof(*created), 0, 1, error);
1638
+ if (code != TF_TRANSFORM_OK) return code;
1639
+ created = (tf_transform_apply *)calloc(1, sizeof(*created));
1640
+ if (!created) return tf_transform_set_error(
1641
+ error, TF_TRANSFORM_ALLOCATION, "apply session allocation failed");
1642
+ created->plan = (tf_transform_plan *)plan;
1643
+ tf_transform_plan_retain(created->plan);
1644
+ created->runtime = runtime_copy;
1645
+ created->allocation_count = 1;
1646
+ created->resident_state_bytes = sizeof(*created);
1647
+ created->state = TF_APPLY_READY;
1648
+ *out = created;
1649
+ return TF_TRANSFORM_OK;
1650
+ }
1651
+
1652
+ void tf_transform_apply_destroy(tf_transform_apply **apply) {
1653
+ if (!apply || !*apply) return;
1654
+ tf_transform_plan_release((*apply)->plan);
1655
+ free(*apply);
1656
+ *apply = NULL;
1657
+ }
1658
+
1659
+ tf_transform_code tf_transform_apply_run(
1660
+ tf_transform_apply *apply, const tf_table_view_v1 *table,
1661
+ tf_owned_dense_v1 *out, tf_transform_error **error) {
1662
+ tf_owned_dense_v1 pending;
1663
+ tf_transform_fp_guard guard;
1664
+ tf_transform_code code;
1665
+ size_t input_bytes = 0;
1666
+ size_t elements = 0;
1667
+ size_t data_bytes = 0;
1668
+ uint64_t new_rows = 0;
1669
+ uint64_t new_input_bytes = 0;
1670
+
1671
+ if (out) memset(out, 0, sizeof(*out));
1672
+ memset(&pending, 0, sizeof(pending));
1673
+ tf_transform_clear_error(error);
1674
+ if (!apply)
1675
+ return tf_transform_set_error(
1676
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "apply run argument is null");
1677
+ if (apply->state != TF_APPLY_READY)
1678
+ return tf_transform_set_error(
1679
+ error, TF_TRANSFORM_INVALID_STATE, "apply session is not ready");
1680
+ if (!table || !out) {
1681
+ apply->state = TF_APPLY_FAILED;
1682
+ return tf_transform_set_error(
1683
+ error, TF_TRANSFORM_INVALID_ARGUMENT, "apply table or output is null");
1684
+ }
1685
+ code = tf_transform_fp_begin(&guard, error);
1686
+ if (code != TF_TRANSFORM_OK) goto failed;
1687
+ apply->runtime.fp_guard_active = 1;
1688
+ code = poll_and_recheck(&apply->runtime, error);
1689
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1690
+ code = validate_table_view(
1691
+ &apply->plan->input_schema, table, apply->runtime.limits.max_apply_rows,
1692
+ apply->runtime.limits.max_apply_input_bytes, &input_bytes,
1693
+ &apply->runtime, error);
1694
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1695
+ code = checked_add_u64(
1696
+ apply->total_rows, (uint64_t)table->row_count, &new_rows,
1697
+ error, "apply row counter overflows");
1698
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1699
+ code = checked_add_u64(
1700
+ apply->total_input_bytes, (uint64_t)input_bytes, &new_input_bytes,
1701
+ error, "apply input byte counter overflows");
1702
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1703
+ if (new_rows > apply->runtime.limits.max_apply_rows
1704
+ || new_input_bytes > apply->runtime.limits.max_apply_input_bytes) {
1705
+ code = tf_transform_set_error(
1706
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1707
+ "cumulative apply input exceeds limits");
1708
+ goto guarded_failed;
1709
+ }
1710
+ code = checked_mul_size(
1711
+ table->row_count, apply->plan->output_schema.field_count, &elements,
1712
+ error, "output element count overflows");
1713
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1714
+ if ((uint64_t)elements > apply->plan->recipe->max_output_elements_per_apply
1715
+ || (uint64_t)elements
1716
+ > apply->runtime.limits.max_output_elements_per_call)
1717
+ {
1718
+ code = tf_transform_set_error(
1719
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "output element count exceeds limits");
1720
+ goto guarded_failed;
1721
+ }
1722
+ code = checked_mul_size(
1723
+ elements, sizeof(double), &data_bytes,
1724
+ error, "output byte count overflows");
1725
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1726
+ if ((uint64_t)data_bytes > apply->runtime.limits.max_allocation_bytes) {
1727
+ code = tf_transform_set_error(
1728
+ error, TF_TRANSFORM_RESOURCE_LIMIT, "output allocation exceeds limit");
1729
+ goto guarded_failed;
1730
+ }
1731
+ pending.abi_version = 1;
1732
+ pending.struct_size = (uint32_t)sizeof(pending);
1733
+ pending.dtype = TF_VIEW_FLOAT64;
1734
+ pending.rows = table->row_count;
1735
+ pending.columns = apply->plan->output_schema.field_count;
1736
+ pending.data_bytes = data_bytes;
1737
+ if (data_bytes != 0) {
1738
+ if (apply->allocation_count
1739
+ >= apply->runtime.limits.max_allocations_per_session) {
1740
+ code = tf_transform_set_error(
1741
+ error, TF_TRANSFORM_RESOURCE_LIMIT,
1742
+ "apply session allocation count exceeds limit");
1743
+ goto guarded_failed;
1744
+ }
1745
+ code = poll_and_recheck(&apply->runtime, error);
1746
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1747
+ pending.data = malloc(data_bytes);
1748
+ if (!pending.data) {
1749
+ code = tf_transform_set_error(
1750
+ error, TF_TRANSFORM_ALLOCATION, "dense output allocation failed");
1751
+ goto guarded_failed;
1752
+ }
1753
+ ++apply->allocation_count;
1754
+ }
1755
+ if (table->row_count != 0 && !pending.data) {
1756
+ code = tf_transform_set_error(
1757
+ error, TF_TRANSFORM_INTERNAL, "nonempty output has no allocation");
1758
+ goto guarded_failed;
1759
+ }
1760
+ code = poll_and_recheck(&apply->runtime, error);
1761
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1762
+ for (size_t row = 0; row < table->row_count; ++row) {
1763
+ size_t output_column = 0;
1764
+ size_t element_poll_rows = TF_TRANSFORM_CANCEL_ELEMENTS_V1
1765
+ / apply->plan->output_schema.field_count;
1766
+ if (element_poll_rows == 0) element_poll_rows = 1;
1767
+ if (row != 0 && (row % TF_TRANSFORM_CANCEL_ROWS_V1 == 0
1768
+ || row % element_poll_rows == 0)) {
1769
+ code = poll_and_recheck(&apply->runtime, error);
1770
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1771
+ }
1772
+ for (size_t column_index = 0;
1773
+ column_index < apply->plan->input_schema.field_count; ++column_index) {
1774
+ const tf_column_view_v1 *column = &table->columns[column_index];
1775
+ const tf_transform_column_state *column_state
1776
+ = &apply->plan->states[column_index];
1777
+ const tf_transform_numeric_state *state;
1778
+ double value;
1779
+ double centered;
1780
+ double transformed;
1781
+ size_t output_index = row * pending.columns + output_column;
1782
+ int missing = !column_value_is_valid(column, row);
1783
+ if (output_column >= pending.columns) {
1784
+ code = tf_transform_set_error(
1785
+ error, TF_TRANSFORM_INTERNAL,
1786
+ "apply output column offset is inconsistent");
1787
+ goto guarded_failed;
1788
+ }
1789
+ if (column_index != 0
1790
+ && column_index % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1791
+ code = poll_and_recheck(&apply->runtime, error);
1792
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1793
+ }
1794
+ if (!missing) {
1795
+ value = read_numeric_value(
1796
+ column, row, apply->plan->input_schema.fields[column_index].dtype);
1797
+ missing = tf_transform_double_is_nan(value);
1798
+ if (!missing && !tf_transform_double_is_finite(value)) {
1799
+ code = tf_transform_set_error(
1800
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1801
+ "infinite apply input is not supported");
1802
+ goto guarded_failed;
1803
+ }
1804
+ }
1805
+ if (column_state->kind == TF_TRANSFORM_KIND_CATEGORICAL) {
1806
+ const tf_transform_categorical_state *categorical
1807
+ = &column_state->value.categorical;
1808
+ size_t output_width = 1;
1809
+ uint64_t bits;
1810
+ size_t ordinal = 0;
1811
+ int known = 0;
1812
+ if (categorical->encode == TF_TRANSFORM_ENCODE_ONEHOT) {
1813
+ output_width = categorical->category_count
1814
+ + (categorical->has_other_ordinal ? 1u : 0u);
1815
+ if (output_width == 0
1816
+ || output_column > pending.columns
1817
+ || output_width > pending.columns - output_column) {
1818
+ code = tf_transform_set_error(
1819
+ error, TF_TRANSFORM_INTERNAL,
1820
+ "one-hot output width is inconsistent");
1821
+ goto guarded_failed;
1822
+ }
1823
+ for (size_t emitted = 0; emitted < output_width; ++emitted) {
1824
+ if (emitted != 0
1825
+ && emitted % TF_TRANSFORM_CANCEL_ITERS_V1 == 0) {
1826
+ code = poll_and_recheck(&apply->runtime, error);
1827
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1828
+ }
1829
+ ((double *)pending.data)[output_index + emitted] = 0.0;
1830
+ }
1831
+ }
1832
+ if (categorical->impute == TF_TRANSFORM_CATEGORICAL_IMPUTE_NONE
1833
+ && categorical->encode == TF_TRANSFORM_ENCODE_NONE) {
1834
+ ((double *)pending.data)[output_index] = missing
1835
+ ? tf_transform_double_from_bits(
1836
+ UINT64_C(0x7ff8000000000000))
1837
+ : value;
1838
+ ++output_column;
1839
+ continue;
1840
+ }
1841
+ if (missing) {
1842
+ if (categorical->has_impute_value) {
1843
+ bits = categorical->impute_bits;
1844
+ value = tf_transform_category_decode(
1845
+ categorical->impute_bits, categorical->source_dtype);
1846
+ } else {
1847
+ known = 0;
1848
+ }
1849
+ } else {
1850
+ code = tf_transform_category_key(
1851
+ value, categorical->source_dtype, &bits, error);
1852
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1853
+ }
1854
+ if (!missing || categorical->has_impute_value) {
1855
+ code = tf_transform_category_lookup(
1856
+ categorical, bits, &apply->runtime,
1857
+ &ordinal, &known, error);
1858
+ if (code != TF_TRANSFORM_OK) goto guarded_failed;
1859
+ }
1860
+ if (categorical->encode == TF_TRANSFORM_ENCODE_ONEHOT) {
1861
+ if (known) {
1862
+ if (ordinal >= categorical->category_count) {
1863
+ code = tf_transform_set_error(
1864
+ error, TF_TRANSFORM_INTERNAL,
1865
+ "one-hot category ordinal is inconsistent");
1866
+ goto guarded_failed;
1867
+ }
1868
+ ((double *)pending.data)[output_index + ordinal] = 1.0;
1869
+ } else if (categorical->unknown
1870
+ == TF_TRANSFORM_UNKNOWN_ALL_ZERO) {
1871
+ /* The output block is already canonical +0. */
1872
+ } else if (categorical->unknown
1873
+ == TF_TRANSFORM_UNKNOWN_OTHER
1874
+ && categorical->has_other_ordinal
1875
+ && categorical->other_ordinal
1876
+ < (uint64_t)output_width) {
1877
+ ((double *)pending.data)[
1878
+ output_index + (size_t)categorical->other_ordinal]
1879
+ = 1.0;
1880
+ } else {
1881
+ code = tf_transform_set_error(
1882
+ error, TF_TRANSFORM_UNKNOWN_CATEGORY,
1883
+ "apply input contains an unknown category");
1884
+ goto guarded_failed;
1885
+ }
1886
+ output_column += output_width;
1887
+ continue;
1888
+ }
1889
+ if (known && categorical->encode == TF_TRANSFORM_ENCODE_LABEL) {
1890
+ ((double *)pending.data)[output_index] = (double)ordinal;
1891
+ ++output_column;
1892
+ continue;
1893
+ }
1894
+ if (!known) {
1895
+ if (categorical->encode == TF_TRANSFORM_ENCODE_LABEL
1896
+ && categorical->unknown == TF_TRANSFORM_UNKNOWN_SENTINEL
1897
+ && categorical->has_sentinel_label) {
1898
+ ((double *)pending.data)[output_index]
1899
+ = (double)categorical->sentinel_label;
1900
+ ++output_column;
1901
+ continue;
1902
+ }
1903
+ if (categorical->encode == TF_TRANSFORM_ENCODE_LABEL
1904
+ && categorical->unknown == TF_TRANSFORM_UNKNOWN_OTHER
1905
+ && categorical->has_other_ordinal) {
1906
+ ((double *)pending.data)[output_index]
1907
+ = (double)categorical->other_ordinal;
1908
+ ++output_column;
1909
+ continue;
1910
+ }
1911
+ code = tf_transform_set_error(
1912
+ error, TF_TRANSFORM_UNKNOWN_CATEGORY,
1913
+ "apply input contains an unknown category");
1914
+ goto guarded_failed;
1915
+ }
1916
+ if (categorical->encode != TF_TRANSFORM_ENCODE_NONE) {
1917
+ code = tf_transform_set_error(
1918
+ error, TF_TRANSFORM_INTERNAL,
1919
+ "categorical encoding state is unsupported");
1920
+ goto guarded_failed;
1921
+ }
1922
+ ((double *)pending.data)[output_index] = value;
1923
+ ++output_column;
1924
+ continue;
1925
+ }
1926
+ state = &column_state->value.numeric;
1927
+ if (missing && !state->has_impute_value) {
1928
+ ((double *)pending.data)[output_index] =
1929
+ tf_transform_double_from_bits(UINT64_C(0x7ff8000000000000));
1930
+ ++output_column;
1931
+ continue;
1932
+ }
1933
+ if (missing) value = state->impute_value;
1934
+ centered = value - state->location;
1935
+ transformed = centered / state->scale;
1936
+ if (!tf_transform_double_is_finite(centered)
1937
+ || !tf_transform_double_is_finite(transformed)) {
1938
+ code = tf_transform_set_error(
1939
+ error, TF_TRANSFORM_NUMERIC_DOMAIN,
1940
+ "numeric transform output became nonfinite");
1941
+ goto guarded_failed;
1942
+ }
1943
+ ((double *)pending.data)[output_index] = transformed;
1944
+ ++output_column;
1945
+ }
1946
+ if (output_column != pending.columns) {
1947
+ code = tf_transform_set_error(
1948
+ error, TF_TRANSFORM_INTERNAL,
1949
+ "apply output width is inconsistent");
1950
+ goto guarded_failed;
1951
+ }
1952
+ }
1953
+ apply->runtime.fp_guard_active = 0;
1954
+ tf_transform_fp_end(&guard);
1955
+ apply->total_rows = new_rows;
1956
+ apply->total_input_bytes = new_input_bytes;
1957
+ *out = pending;
1958
+ return TF_TRANSFORM_OK;
1959
+ guarded_failed:
1960
+ apply->runtime.fp_guard_active = 0;
1961
+ tf_transform_fp_end(&guard);
1962
+ failed:
1963
+ free(pending.data);
1964
+ apply->state = TF_APPLY_FAILED;
1965
+ return code;
1966
+ }