tranfi 0.0.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -21
- package/NOTICE +8 -0
- package/README.md +627 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-BIAIKnrp.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +121 -0
- package/csrc/arena.c +93 -0
- package/csrc/batch.c +976 -0
- package/csrc/buffer.c +154 -0
- package/csrc/cJSON.c +3386 -0
- package/csrc/cJSON.h +316 -0
- package/csrc/codec_csv.c +1951 -0
- package/csrc/codec_jsonl.c +1086 -0
- package/csrc/codec_table.c +248 -0
- package/csrc/codec_text.c +447 -0
- package/csrc/compiler.c +130 -0
- package/csrc/config.h +21 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +5417 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1553 -0
- package/csrc/expr.h +58 -0
- package/csrc/internal.h +539 -0
- package/csrc/ir.c +166 -0
- package/csrc/ir.h +208 -0
- package/csrc/ir_schema.c +75 -0
- package/csrc/ir_serialize.c +166 -0
- package/csrc/ir_sql.c +1822 -0
- package/csrc/ir_validate.c +576 -0
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +1241 -0
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +283 -0
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +255 -0
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +248 -0
- package/csrc/op_cast.c +523 -0
- package/csrc/op_clip.c +99 -0
- package/csrc/op_date_trunc.c +355 -0
- package/csrc/op_datetime.c +394 -0
- package/csrc/op_derive.c +216 -0
- package/csrc/op_diff.c +250 -0
- package/csrc/op_ewma.c +222 -0
- package/csrc/op_explode.c +206 -0
- package/csrc/op_fill_down.c +235 -0
- package/csrc/op_fill_null.c +268 -0
- package/csrc/op_filter.c +181 -0
- package/csrc/op_frequency.c +721 -0
- package/csrc/op_grep.c +181 -0
- package/csrc/op_group_agg.c +1956 -0
- package/csrc/op_hash.c +159 -0
- package/csrc/op_head.c +84 -0
- package/csrc/op_interpolate.c +445 -0
- package/csrc/op_join.c +2902 -0
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +419 -0
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +242 -0
- package/csrc/op_normalize.c +510 -0
- package/csrc/op_onehot.c +457 -0
- package/csrc/op_pivot.c +1754 -0
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +3044 -0
- package/csrc/op_rename.c +129 -0
- package/csrc/op_replace.c +354 -0
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +158 -0
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +340 -0
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +95 -0
- package/csrc/op_sort.c +819 -0
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +151 -0
- package/csrc/op_split_data.c +119 -0
- package/csrc/op_stack.c +271 -0
- package/csrc/op_stats.c +875 -0
- package/csrc/op_step.c +333 -0
- package/csrc/op_tail.c +105 -0
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +357 -0
- package/csrc/op_trim.c +138 -0
- package/csrc/op_unique.c +1343 -0
- package/csrc/op_unpivot.c +193 -0
- package/csrc/op_validate.c +648 -0
- package/csrc/op_window.c +591 -0
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +1088 -0
- package/csrc/recipes.c +104 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +506 -0
- package/csrc/report.h +22 -0
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +291 -0
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +218 -0
- package/napi_api.c +534 -0
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +64 -59
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +190 -0
- package/src/engines/duckdb.js +142 -0
- package/src/index.js +925 -0
- package/src/memory_policy.js +411 -0
- package/src/native.js +18 -0
- package/src/pipeline.js +709 -0
- package/src/recipe_json.js +80 -0
- package/src/server.js +279 -0
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +21 -0
- package/wasm/index.js +732 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
|
@@ -0,0 +1,477 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* memory_estimate.c -- conservative retained-state byte estimates.
|
|
3
|
+
*
|
|
4
|
+
* These estimates are plan-policy metadata, not measured allocator accounting.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
#include "internal.h"
|
|
8
|
+
#include "cJSON.h"
|
|
9
|
+
#include <stdint.h>
|
|
10
|
+
#include <stdio.h>
|
|
11
|
+
#include <string.h>
|
|
12
|
+
|
|
13
|
+
#define UNIQUE_DEFAULT_BLOOM_BYTES (1024u * 1024u)
|
|
14
|
+
|
|
15
|
+
static int checked_add_size(size_t *acc, size_t value) {
|
|
16
|
+
if (*acc > SIZE_MAX - value) return -1;
|
|
17
|
+
*acc += value;
|
|
18
|
+
return 0;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
static int checked_mul_size(size_t a, size_t b, size_t *out) {
|
|
22
|
+
if (a != 0 && b > SIZE_MAX / a) return -1;
|
|
23
|
+
*out = a * b;
|
|
24
|
+
return 0;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
static int checked_add_mul_size(size_t *acc, size_t a, size_t b) {
|
|
28
|
+
size_t product = 0;
|
|
29
|
+
if (checked_mul_size(a, b, &product) != 0) return -1;
|
|
30
|
+
return checked_add_size(acc, product);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
static int json_size_arg(const cJSON *args, const char *name, size_t *out) {
|
|
34
|
+
return tf_json_get_size_arg(args, name, 1, TF_MAX_SAFE_SIZE_ARG,
|
|
35
|
+
out, "memory-estimate");
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
static size_t json_array_len_arg(const cJSON *args, const char *name) {
|
|
39
|
+
const cJSON *item = cJSON_GetObjectItemCaseSensitive(args, name);
|
|
40
|
+
if (!cJSON_IsArray(item)) return 0;
|
|
41
|
+
int n = cJSON_GetArraySize(item);
|
|
42
|
+
return n > 0 ? (size_t)n : 0;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
static size_t json_string_array_bytes_arg(const cJSON *args, const char *name) {
|
|
46
|
+
const cJSON *item = cJSON_GetObjectItemCaseSensitive(args, name);
|
|
47
|
+
if (!cJSON_IsArray(item)) return 0;
|
|
48
|
+
size_t bytes = 0;
|
|
49
|
+
const cJSON *el = NULL;
|
|
50
|
+
cJSON_ArrayForEach(el, item) {
|
|
51
|
+
if (cJSON_IsString(el) && el->valuestring) {
|
|
52
|
+
size_t len = strlen(el->valuestring);
|
|
53
|
+
if (checked_add_size(&bytes, len + 1) != 0) return SIZE_MAX;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return bytes;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
static int json_bool_arg_true(const cJSON *args, const char *name) {
|
|
60
|
+
const cJSON *item = cJSON_GetObjectItemCaseSensitive(args, name);
|
|
61
|
+
return cJSON_IsBool(item) && cJSON_IsTrue(item);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
static const char *json_string_arg(const cJSON *args, const char *name) {
|
|
65
|
+
const cJSON *item = cJSON_GetObjectItemCaseSensitive(args, name);
|
|
66
|
+
return cJSON_IsString(item) ? item->valuestring : NULL;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
static int category_cap_arg(const cJSON *args, size_t *cap, size_t *literal_bytes) {
|
|
70
|
+
size_t max_categories = 0;
|
|
71
|
+
int has_max = json_size_arg(args, "max_categories", &max_categories);
|
|
72
|
+
if (has_max < 0) return -1;
|
|
73
|
+
|
|
74
|
+
size_t n_categories = json_array_len_arg(args, "categories");
|
|
75
|
+
size_t string_bytes = json_string_array_bytes_arg(args, "categories");
|
|
76
|
+
if (string_bytes == SIZE_MAX) return -1;
|
|
77
|
+
|
|
78
|
+
const cJSON *unknown = cJSON_GetObjectItemCaseSensitive(args, "unknown");
|
|
79
|
+
if (cJSON_IsString(unknown) && unknown->valuestring && strcmp(unknown->valuestring, "other") == 0) {
|
|
80
|
+
if (checked_add_size(&n_categories, 1) != 0) return -1;
|
|
81
|
+
if (checked_add_size(&string_bytes, strlen("other") + 1) != 0) return -1;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
if (has_max > 0) {
|
|
85
|
+
*cap = max_categories;
|
|
86
|
+
} else if (n_categories > 0) {
|
|
87
|
+
*cap = n_categories;
|
|
88
|
+
} else {
|
|
89
|
+
*cap = 0;
|
|
90
|
+
}
|
|
91
|
+
*literal_bytes = string_bytes;
|
|
92
|
+
return *cap > 0 ? 1 : 0;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
int tf_estimate_step_state_bytes(const tf_ir_node *node, size_t *out,
|
|
96
|
+
char *reason, size_t reason_size) {
|
|
97
|
+
const char *op = node->op ? node->op : "unknown";
|
|
98
|
+
const cJSON *args = node->args;
|
|
99
|
+
size_t est = 0;
|
|
100
|
+
|
|
101
|
+
if (strcmp(op, "rowid") == 0) {
|
|
102
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
103
|
+
if (n_cols == 0 || json_bool_arg_true(args, "sorted")) {
|
|
104
|
+
est = 2048;
|
|
105
|
+
if (checked_add_mul_size(&est, n_cols, 128) != 0) goto overflow;
|
|
106
|
+
*out = est;
|
|
107
|
+
return 1;
|
|
108
|
+
}
|
|
109
|
+
size_t max_state_bytes = 0;
|
|
110
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
111
|
+
if (has_state_bytes < 0) goto overflow;
|
|
112
|
+
if (has_state_bytes) {
|
|
113
|
+
*out = max_state_bytes;
|
|
114
|
+
return 1;
|
|
115
|
+
}
|
|
116
|
+
size_t max_keys = 0;
|
|
117
|
+
int has_max = json_size_arg(args, "max_keys", &max_keys);
|
|
118
|
+
if (has_max < 0) goto overflow;
|
|
119
|
+
if (!has_max) {
|
|
120
|
+
snprintf(reason, reason_size, "step '%s' needs max_keys, max_state_bytes, or sorted=true", op);
|
|
121
|
+
return 0;
|
|
122
|
+
}
|
|
123
|
+
est = 1024;
|
|
124
|
+
if (checked_add_mul_size(&est, max_keys, 224) != 0) goto overflow;
|
|
125
|
+
if (checked_add_mul_size(&est, n_cols, 128) != 0) goto overflow;
|
|
126
|
+
*out = est;
|
|
127
|
+
return 1;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
if (strcmp(op, "unique") == 0 || strcmp(op, "dedup") == 0) {
|
|
131
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
132
|
+
if (json_bool_arg_true(args, "sorted")) {
|
|
133
|
+
est = 2048;
|
|
134
|
+
if (checked_add_mul_size(&est, n_cols, 128) != 0) goto overflow;
|
|
135
|
+
*out = est;
|
|
136
|
+
return 1;
|
|
137
|
+
}
|
|
138
|
+
const char *mode = json_string_arg(args, "mode");
|
|
139
|
+
if (json_bool_arg_true(args, "approx") ||
|
|
140
|
+
(mode && strcmp(mode, "approx") == 0)) {
|
|
141
|
+
size_t bloom_bytes = 0;
|
|
142
|
+
int has_bloom = json_size_arg(args, "bloom_bytes", &bloom_bytes);
|
|
143
|
+
if (has_bloom < 0) goto overflow;
|
|
144
|
+
*out = has_bloom ? bloom_bytes : UNIQUE_DEFAULT_BLOOM_BYTES;
|
|
145
|
+
return 1;
|
|
146
|
+
}
|
|
147
|
+
size_t max_state_bytes = 0;
|
|
148
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
149
|
+
if (has_state_bytes < 0) goto overflow;
|
|
150
|
+
if (has_state_bytes) {
|
|
151
|
+
*out = max_state_bytes;
|
|
152
|
+
return 1;
|
|
153
|
+
}
|
|
154
|
+
size_t max_keys = 0;
|
|
155
|
+
int has_max = json_size_arg(args, "max_keys", &max_keys);
|
|
156
|
+
if (has_max < 0) goto overflow;
|
|
157
|
+
if (!has_max) {
|
|
158
|
+
snprintf(reason, reason_size, "step '%s' needs max_keys, max_state_bytes, or sorted=true for byte-bounded native execution", op);
|
|
159
|
+
return 0;
|
|
160
|
+
}
|
|
161
|
+
est = 1024;
|
|
162
|
+
if (checked_add_mul_size(&est, max_keys, 256 + n_cols * 32) != 0) goto overflow;
|
|
163
|
+
*out = est;
|
|
164
|
+
return 1;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
if (strcmp(op, "group-agg") == 0) {
|
|
168
|
+
size_t n_group = json_array_len_arg(args, "group_by");
|
|
169
|
+
size_t n_aggs = json_array_len_arg(args, "aggs");
|
|
170
|
+
if (json_string_arg(args, "spill_dir")) {
|
|
171
|
+
size_t spill_memory_bytes = 0;
|
|
172
|
+
int has_spill_bytes = json_size_arg(args, "spill_memory_bytes", &spill_memory_bytes);
|
|
173
|
+
if (has_spill_bytes < 0) goto overflow;
|
|
174
|
+
*out = has_spill_bytes ? spill_memory_bytes : 0;
|
|
175
|
+
return has_spill_bytes ? 1 : 0;
|
|
176
|
+
}
|
|
177
|
+
size_t max_state_bytes = 0;
|
|
178
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
179
|
+
if (has_state_bytes < 0) goto overflow;
|
|
180
|
+
if (has_state_bytes) {
|
|
181
|
+
*out = max_state_bytes;
|
|
182
|
+
return 1;
|
|
183
|
+
}
|
|
184
|
+
if (json_bool_arg_true(args, "sorted")) {
|
|
185
|
+
est = 4096;
|
|
186
|
+
if (checked_add_mul_size(&est, n_group, 128) != 0) goto overflow;
|
|
187
|
+
if (checked_add_mul_size(&est, n_aggs, 128) != 0) goto overflow;
|
|
188
|
+
*out = est;
|
|
189
|
+
return 1;
|
|
190
|
+
}
|
|
191
|
+
size_t max_groups = 0;
|
|
192
|
+
int has_max = json_size_arg(args, "max_groups", &max_groups);
|
|
193
|
+
if (has_max < 0) goto overflow;
|
|
194
|
+
if (!has_max) {
|
|
195
|
+
snprintf(reason, reason_size, "step '%s' needs max_groups, max_state_bytes, or sorted=true for byte-bounded native execution", op);
|
|
196
|
+
return 0;
|
|
197
|
+
}
|
|
198
|
+
size_t per_group = 384;
|
|
199
|
+
if (checked_add_mul_size(&per_group, n_group, 64) != 0) goto overflow;
|
|
200
|
+
if (checked_add_mul_size(&per_group, n_aggs, 96) != 0) goto overflow;
|
|
201
|
+
est = 2048;
|
|
202
|
+
if (checked_add_mul_size(&est, max_groups, per_group) != 0) goto overflow;
|
|
203
|
+
*out = est;
|
|
204
|
+
return 1;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
if (strcmp(op, "frequency") == 0) {
|
|
208
|
+
size_t max_state_bytes = 0;
|
|
209
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
210
|
+
if (has_state_bytes < 0) goto overflow;
|
|
211
|
+
if (has_state_bytes) {
|
|
212
|
+
*out = max_state_bytes;
|
|
213
|
+
return 1;
|
|
214
|
+
}
|
|
215
|
+
size_t max_values = 0;
|
|
216
|
+
int has_max = json_size_arg(args, "max_values", &max_values);
|
|
217
|
+
if (has_max < 0) goto overflow;
|
|
218
|
+
if (!has_max) {
|
|
219
|
+
snprintf(reason, reason_size, "step '%s' needs max_values or max_state_bytes for byte-bounded native execution", op);
|
|
220
|
+
return 0;
|
|
221
|
+
}
|
|
222
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
223
|
+
cJSON *overflow_arg = args ? cJSON_GetObjectItemCaseSensitive(args, "overflow") : NULL;
|
|
224
|
+
if (cJSON_IsString(overflow_arg) && strcmp(overflow_arg->valuestring, "other") == 0) {
|
|
225
|
+
if (max_values == SIZE_MAX) goto overflow;
|
|
226
|
+
max_values += 1;
|
|
227
|
+
}
|
|
228
|
+
est = 1024;
|
|
229
|
+
if (checked_add_mul_size(&est, max_values, 224 + n_cols * 32) != 0) goto overflow;
|
|
230
|
+
*out = est;
|
|
231
|
+
return 1;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
if (strcmp(op, "onehot") == 0 || strcmp(op, "label-encode") == 0) {
|
|
235
|
+
size_t max_state_bytes = 0;
|
|
236
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
237
|
+
if (has_state_bytes < 0) goto overflow;
|
|
238
|
+
if (has_state_bytes) {
|
|
239
|
+
*out = max_state_bytes;
|
|
240
|
+
return 1;
|
|
241
|
+
}
|
|
242
|
+
size_t cap = 0;
|
|
243
|
+
size_t literal_bytes = 0;
|
|
244
|
+
int has_cap = category_cap_arg(args, &cap, &literal_bytes);
|
|
245
|
+
if (has_cap < 0) goto overflow;
|
|
246
|
+
if (!has_cap) {
|
|
247
|
+
snprintf(reason, reason_size, "step '%s' needs max_categories, max_state_bytes, or declared categories", op);
|
|
248
|
+
return 0;
|
|
249
|
+
}
|
|
250
|
+
est = 1024;
|
|
251
|
+
if (checked_add_size(&est, literal_bytes) != 0) goto overflow;
|
|
252
|
+
if (checked_add_mul_size(&est, cap, strcmp(op, "onehot") == 0 ? 320 : 224) != 0) goto overflow;
|
|
253
|
+
*out = est;
|
|
254
|
+
return 1;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
if (strcmp(op, "intersect") == 0 || strcmp(op, "setdiff") == 0 ||
|
|
258
|
+
strcmp(op, "intersect-all") == 0 || strcmp(op, "setdiff-all") == 0) {
|
|
259
|
+
int bag_set_op = strcmp(op, "intersect-all") == 0 || strcmp(op, "setdiff-all") == 0;
|
|
260
|
+
if (json_string_arg(args, "spill_dir")) {
|
|
261
|
+
size_t spill_memory_bytes = 0;
|
|
262
|
+
int has_spill_bytes = json_size_arg(args, "spill_memory_bytes", &spill_memory_bytes);
|
|
263
|
+
if (has_spill_bytes < 0) goto overflow;
|
|
264
|
+
*out = has_spill_bytes ? spill_memory_bytes : 0;
|
|
265
|
+
return has_spill_bytes ? 1 : 0;
|
|
266
|
+
}
|
|
267
|
+
if (json_bool_arg_true(args, "sorted")) {
|
|
268
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
269
|
+
est = 4096;
|
|
270
|
+
if (checked_add_mul_size(&est, n_cols, 256) != 0) goto overflow;
|
|
271
|
+
*out = est;
|
|
272
|
+
return 1;
|
|
273
|
+
}
|
|
274
|
+
size_t max_lookup_bytes = 0;
|
|
275
|
+
int has_bytes = json_size_arg(args, "max_lookup_bytes", &max_lookup_bytes);
|
|
276
|
+
if (has_bytes < 0) goto overflow;
|
|
277
|
+
if (!has_bytes) {
|
|
278
|
+
snprintf(reason, reason_size, "step '%s' needs max_lookup_bytes or sorted=true for byte-bounded native execution", op);
|
|
279
|
+
return 0;
|
|
280
|
+
}
|
|
281
|
+
size_t max_state_bytes = 0;
|
|
282
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
283
|
+
if (has_state_bytes < 0) goto overflow;
|
|
284
|
+
if (has_state_bytes) {
|
|
285
|
+
est = 4096;
|
|
286
|
+
if (checked_add_mul_size(&est, max_lookup_bytes, 3) != 0) goto overflow;
|
|
287
|
+
if (checked_add_size(&est, max_state_bytes) != 0) goto overflow;
|
|
288
|
+
*out = est;
|
|
289
|
+
return 1;
|
|
290
|
+
}
|
|
291
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
292
|
+
est = 4096;
|
|
293
|
+
if (checked_add_mul_size(&est, max_lookup_bytes, 3) != 0) goto overflow;
|
|
294
|
+
size_t max_lookup_keys = 0;
|
|
295
|
+
int has_lookup_keys = json_size_arg(args, "max_lookup_keys", &max_lookup_keys);
|
|
296
|
+
if (has_lookup_keys < 0) goto overflow;
|
|
297
|
+
if (bag_set_op) {
|
|
298
|
+
if (!has_lookup_keys) {
|
|
299
|
+
snprintf(reason, reason_size, "step '%s' needs max_lookup_keys or max_state_bytes for bag-count set semantics", op);
|
|
300
|
+
return 0;
|
|
301
|
+
}
|
|
302
|
+
if (checked_add_mul_size(&est, max_lookup_keys, 224 + n_cols * 32) != 0) goto overflow;
|
|
303
|
+
*out = est;
|
|
304
|
+
return 1;
|
|
305
|
+
}
|
|
306
|
+
size_t max_output_keys = 0;
|
|
307
|
+
int has_output_keys = json_size_arg(args, "max_output_keys", &max_output_keys);
|
|
308
|
+
if (has_output_keys < 0) goto overflow;
|
|
309
|
+
if (!has_output_keys) {
|
|
310
|
+
snprintf(reason, reason_size, "step '%s' needs max_output_keys or max_state_bytes for duplicate-eliminating set semantics", op);
|
|
311
|
+
return 0;
|
|
312
|
+
}
|
|
313
|
+
if (has_lookup_keys) {
|
|
314
|
+
if (checked_add_mul_size(&est, max_lookup_keys, 192 + n_cols * 32) != 0) goto overflow;
|
|
315
|
+
}
|
|
316
|
+
if (checked_add_mul_size(&est, max_output_keys, 192 + n_cols * 32) != 0) goto overflow;
|
|
317
|
+
*out = est;
|
|
318
|
+
return 1;
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
if (strcmp(op, "pivot") == 0) {
|
|
322
|
+
size_t cap = 0;
|
|
323
|
+
size_t literal_bytes = 0;
|
|
324
|
+
int has_cap = category_cap_arg(args, &cap, &literal_bytes);
|
|
325
|
+
if (has_cap < 0) goto overflow;
|
|
326
|
+
if (json_string_arg(args, "spill_dir")) {
|
|
327
|
+
if (!has_cap) {
|
|
328
|
+
snprintf(reason, reason_size,
|
|
329
|
+
"step 'pivot' spill mode needs categories or max_categories for byte-bounded output columns");
|
|
330
|
+
return 0;
|
|
331
|
+
}
|
|
332
|
+
size_t spill_memory_bytes = 0;
|
|
333
|
+
int has_spill_bytes = json_size_arg(args, "spill_memory_bytes", &spill_memory_bytes);
|
|
334
|
+
if (has_spill_bytes < 0) goto overflow;
|
|
335
|
+
*out = has_spill_bytes ? spill_memory_bytes : 0;
|
|
336
|
+
return has_spill_bytes ? 1 : 0;
|
|
337
|
+
}
|
|
338
|
+
if (json_bool_arg_true(args, "sorted") && has_cap) {
|
|
339
|
+
est = 4096 + literal_bytes;
|
|
340
|
+
if (checked_add_mul_size(&est, cap, 96) != 0) goto overflow;
|
|
341
|
+
*out = est;
|
|
342
|
+
return 1;
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
if (strcmp(op, "union") == 0) {
|
|
347
|
+
if (json_bool_arg_true(args, "sorted")) {
|
|
348
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
349
|
+
est = 4096;
|
|
350
|
+
if (checked_add_mul_size(&est, n_cols, 256) != 0) goto overflow;
|
|
351
|
+
*out = est;
|
|
352
|
+
return 1;
|
|
353
|
+
}
|
|
354
|
+
if (json_string_arg(args, "spill_dir")) {
|
|
355
|
+
size_t spill_memory_bytes = 0;
|
|
356
|
+
int has_spill_bytes = json_size_arg(args, "spill_memory_bytes", &spill_memory_bytes);
|
|
357
|
+
if (has_spill_bytes < 0) goto overflow;
|
|
358
|
+
*out = has_spill_bytes ? spill_memory_bytes : 0;
|
|
359
|
+
return has_spill_bytes ? 1 : 0;
|
|
360
|
+
}
|
|
361
|
+
size_t max_state_bytes = 0;
|
|
362
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
363
|
+
if (has_state_bytes < 0) goto overflow;
|
|
364
|
+
if (has_state_bytes) {
|
|
365
|
+
*out = max_state_bytes;
|
|
366
|
+
return 1;
|
|
367
|
+
}
|
|
368
|
+
size_t max_output_keys = 0;
|
|
369
|
+
int has_output_keys = json_size_arg(args, "max_output_keys", &max_output_keys);
|
|
370
|
+
if (has_output_keys < 0) goto overflow;
|
|
371
|
+
if (!has_output_keys) {
|
|
372
|
+
snprintf(reason, reason_size, "step 'union' needs max_output_keys or max_state_bytes for duplicate-eliminating set semantics");
|
|
373
|
+
return 0;
|
|
374
|
+
}
|
|
375
|
+
size_t n_cols = json_array_len_arg(args, "columns");
|
|
376
|
+
est = 4096;
|
|
377
|
+
if (checked_add_mul_size(&est, max_output_keys, 192 + n_cols * 32) != 0) goto overflow;
|
|
378
|
+
*out = est;
|
|
379
|
+
return 1;
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
if (strcmp(op, "join") == 0 || strcmp(op, "semi-join") == 0 || strcmp(op, "anti-join") == 0) {
|
|
383
|
+
const char *how = json_string_arg(args, "how");
|
|
384
|
+
int filtering = strcmp(op, "semi-join") == 0 || strcmp(op, "anti-join") == 0 ||
|
|
385
|
+
(how && (strcmp(how, "semi") == 0 || strcmp(how, "anti") == 0));
|
|
386
|
+
if (json_string_arg(args, "spill_dir")) {
|
|
387
|
+
if (!filtering) {
|
|
388
|
+
size_t max_matches = 0;
|
|
389
|
+
int has_matches = json_size_arg(args, "max_matches_per_row", &max_matches);
|
|
390
|
+
if (has_matches < 0) goto overflow;
|
|
391
|
+
if (!has_matches) {
|
|
392
|
+
snprintf(reason, reason_size,
|
|
393
|
+
"step '%s' spill mode needs max_matches_per_row for byte-bounded mutating join", op);
|
|
394
|
+
return 0;
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
size_t spill_memory_bytes = 0;
|
|
398
|
+
int has_spill_bytes = json_size_arg(args, "spill_memory_bytes", &spill_memory_bytes);
|
|
399
|
+
if (has_spill_bytes < 0) goto overflow;
|
|
400
|
+
*out = has_spill_bytes ? spill_memory_bytes : 0;
|
|
401
|
+
return has_spill_bytes ? 1 : 0;
|
|
402
|
+
}
|
|
403
|
+
if (json_bool_arg_true(args, "sorted")) {
|
|
404
|
+
est = 4096;
|
|
405
|
+
if (!filtering) {
|
|
406
|
+
size_t max_matches = 0;
|
|
407
|
+
int has_matches = json_size_arg(args, "max_matches_per_row", &max_matches);
|
|
408
|
+
if (has_matches < 0) goto overflow;
|
|
409
|
+
if (!has_matches) {
|
|
410
|
+
snprintf(reason, reason_size,
|
|
411
|
+
"step '%s' sorted=true needs max_matches_per_row for byte-bounded mutating join", op);
|
|
412
|
+
return 0;
|
|
413
|
+
}
|
|
414
|
+
if (checked_add_mul_size(&est, max_matches, 512) != 0) goto overflow;
|
|
415
|
+
}
|
|
416
|
+
*out = est;
|
|
417
|
+
return 1;
|
|
418
|
+
}
|
|
419
|
+
size_t max_lookup_bytes = 0;
|
|
420
|
+
int has_bytes = json_size_arg(args, "max_lookup_bytes", &max_lookup_bytes);
|
|
421
|
+
if (has_bytes < 0) goto overflow;
|
|
422
|
+
if (!has_bytes) {
|
|
423
|
+
snprintf(reason, reason_size, "step '%s' needs max_lookup_bytes or sorted=true for byte-bounded native execution", op);
|
|
424
|
+
return 0;
|
|
425
|
+
}
|
|
426
|
+
est = 4096;
|
|
427
|
+
if (checked_add_mul_size(&est, max_lookup_bytes, 4) != 0) goto overflow;
|
|
428
|
+
size_t max_state_bytes = 0;
|
|
429
|
+
int has_state_bytes = json_size_arg(args, "max_state_bytes", &max_state_bytes);
|
|
430
|
+
if (has_state_bytes < 0) goto overflow;
|
|
431
|
+
if (has_state_bytes) {
|
|
432
|
+
if (checked_add_size(&est, max_state_bytes) != 0) goto overflow;
|
|
433
|
+
*out = est;
|
|
434
|
+
return 1;
|
|
435
|
+
}
|
|
436
|
+
size_t max_lookup_rows = 0;
|
|
437
|
+
if (json_size_arg(args, "max_lookup_rows", &max_lookup_rows) > 0) {
|
|
438
|
+
if (checked_add_mul_size(&est, max_lookup_rows, 96) != 0) goto overflow;
|
|
439
|
+
}
|
|
440
|
+
size_t max_lookup_keys = 0;
|
|
441
|
+
if (json_size_arg(args, "max_lookup_keys", &max_lookup_keys) > 0) {
|
|
442
|
+
if (checked_add_mul_size(&est, max_lookup_keys, 192) != 0) goto overflow;
|
|
443
|
+
}
|
|
444
|
+
*out = est;
|
|
445
|
+
return 1;
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
snprintf(reason, reason_size, "step '%s' has no native byte estimator", op);
|
|
449
|
+
return 0;
|
|
450
|
+
|
|
451
|
+
overflow:
|
|
452
|
+
snprintf(reason, reason_size, "step '%s' byte estimate overflowed", op);
|
|
453
|
+
return 0;
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
int tf_estimate_key_state_plan_bytes(const tf_ir_plan *ir, size_t *out,
|
|
457
|
+
const tf_ir_node **failed_node,
|
|
458
|
+
char *reason, size_t reason_size) {
|
|
459
|
+
size_t total = 0;
|
|
460
|
+
*failed_node = NULL;
|
|
461
|
+
for (size_t i = 0; i < ir->n_nodes; i++) {
|
|
462
|
+
const tf_ir_node *node = &ir->nodes[i];
|
|
463
|
+
if (node->memory_class != TF_MEM_KEY_STATE) continue;
|
|
464
|
+
size_t node_est = 0;
|
|
465
|
+
if (!tf_estimate_step_state_bytes(node, &node_est, reason, reason_size)) {
|
|
466
|
+
*failed_node = node;
|
|
467
|
+
return 0;
|
|
468
|
+
}
|
|
469
|
+
if (checked_add_size(&total, node_est) != 0) {
|
|
470
|
+
snprintf(reason, reason_size, "total key-state byte estimate overflowed");
|
|
471
|
+
*failed_node = node;
|
|
472
|
+
return 0;
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
*out = total;
|
|
476
|
+
return 1;
|
|
477
|
+
}
|