tranfi 0.1.2 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +443 -51
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +352 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +81 -41
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/csrc/op_schema.c
ADDED
|
@@ -0,0 +1,1341 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* op_schema.c -- Row-local table schema/data-quality contracts.
|
|
3
|
+
*
|
|
4
|
+
* Config:
|
|
5
|
+
* {
|
|
6
|
+
* "columns": {"name":"string", "age":{"type":"int","nullable":false}},
|
|
7
|
+
* "required": ["name"], "non_null": ["age"],
|
|
8
|
+
* "values": {"city":["NY","LA"]}, "min": {"age":0}, "max": {"age":120},
|
|
9
|
+
* "regex": {"code":"^[A-Z]+$"},
|
|
10
|
+
* "mode": "fail|warn|filter|quarantine|annotate"
|
|
11
|
+
* }
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
#include "internal.h"
|
|
15
|
+
#include "cJSON.h"
|
|
16
|
+
#include <stdio.h>
|
|
17
|
+
#include <stdlib.h>
|
|
18
|
+
#include <string.h>
|
|
19
|
+
#include <stdint.h>
|
|
20
|
+
#include <regex.h>
|
|
21
|
+
#include <math.h>
|
|
22
|
+
|
|
23
|
+
typedef enum {
|
|
24
|
+
SCHEMA_EXPECT_ANY = 0,
|
|
25
|
+
SCHEMA_EXPECT_BOOL,
|
|
26
|
+
SCHEMA_EXPECT_INT,
|
|
27
|
+
SCHEMA_EXPECT_FLOAT,
|
|
28
|
+
SCHEMA_EXPECT_NUMBER,
|
|
29
|
+
SCHEMA_EXPECT_STRING,
|
|
30
|
+
SCHEMA_EXPECT_DATE,
|
|
31
|
+
SCHEMA_EXPECT_TIMESTAMP,
|
|
32
|
+
} schema_expected_type;
|
|
33
|
+
|
|
34
|
+
typedef enum {
|
|
35
|
+
SCHEMA_FAIL = 0,
|
|
36
|
+
SCHEMA_WARN,
|
|
37
|
+
SCHEMA_FILTER,
|
|
38
|
+
SCHEMA_QUARANTINE,
|
|
39
|
+
SCHEMA_ANNOTATE,
|
|
40
|
+
} schema_action;
|
|
41
|
+
|
|
42
|
+
typedef struct {
|
|
43
|
+
char *column;
|
|
44
|
+
schema_expected_type expected_type;
|
|
45
|
+
char *expected_type_name;
|
|
46
|
+
int required;
|
|
47
|
+
int nullable;
|
|
48
|
+
int check_type;
|
|
49
|
+
int has_values;
|
|
50
|
+
char **values;
|
|
51
|
+
uint8_t *values_seen;
|
|
52
|
+
size_t n_values;
|
|
53
|
+
int has_min;
|
|
54
|
+
int has_max;
|
|
55
|
+
double min_value;
|
|
56
|
+
double max_value;
|
|
57
|
+
int has_regex;
|
|
58
|
+
char *regex_pattern;
|
|
59
|
+
regex_t regex;
|
|
60
|
+
int compiled_regex;
|
|
61
|
+
int col_idx;
|
|
62
|
+
} schema_rule;
|
|
63
|
+
|
|
64
|
+
typedef struct {
|
|
65
|
+
char **selectors;
|
|
66
|
+
size_t n_selectors;
|
|
67
|
+
int type_set;
|
|
68
|
+
schema_expected_type expected_type;
|
|
69
|
+
char *expected_type_name;
|
|
70
|
+
int required_set;
|
|
71
|
+
int required;
|
|
72
|
+
int nullable_set;
|
|
73
|
+
int nullable;
|
|
74
|
+
int has_values;
|
|
75
|
+
char **values;
|
|
76
|
+
size_t n_values;
|
|
77
|
+
int has_min;
|
|
78
|
+
int has_max;
|
|
79
|
+
double min_value;
|
|
80
|
+
double max_value;
|
|
81
|
+
int has_regex;
|
|
82
|
+
char *regex_pattern;
|
|
83
|
+
} schema_selector_rule;
|
|
84
|
+
|
|
85
|
+
typedef struct {
|
|
86
|
+
schema_rule *rules;
|
|
87
|
+
size_t n_rules;
|
|
88
|
+
schema_selector_rule *selector_rules;
|
|
89
|
+
size_t n_selector_rules;
|
|
90
|
+
schema_action action;
|
|
91
|
+
char *name;
|
|
92
|
+
char *message;
|
|
93
|
+
char *result;
|
|
94
|
+
size_t row_index;
|
|
95
|
+
size_t audit_limit;
|
|
96
|
+
size_t audit_emitted;
|
|
97
|
+
size_t checked_rows;
|
|
98
|
+
size_t passed_rows;
|
|
99
|
+
size_t failed_rows;
|
|
100
|
+
size_t violation_count;
|
|
101
|
+
size_t required_failures;
|
|
102
|
+
size_t type_failures;
|
|
103
|
+
size_t nullable_failures;
|
|
104
|
+
size_t values_failures;
|
|
105
|
+
size_t min_failures;
|
|
106
|
+
size_t max_failures;
|
|
107
|
+
size_t regex_failures;
|
|
108
|
+
size_t schema_failures;
|
|
109
|
+
size_t extra_column_failures;
|
|
110
|
+
size_t missing_category_failures;
|
|
111
|
+
int audit;
|
|
112
|
+
tf_audit_options audit_opts;
|
|
113
|
+
size_t max_regex_pattern_bytes;
|
|
114
|
+
size_t max_regex_cell_bytes;
|
|
115
|
+
int baseline_mode;
|
|
116
|
+
int check_extra_columns;
|
|
117
|
+
int allow_extra_columns;
|
|
118
|
+
int require_values_seen;
|
|
119
|
+
int checked_schema;
|
|
120
|
+
int selectors_expanded;
|
|
121
|
+
int schema_ok;
|
|
122
|
+
} schema_state;
|
|
123
|
+
|
|
124
|
+
static const char *schema_action_name(schema_action action) {
|
|
125
|
+
switch (action) {
|
|
126
|
+
case SCHEMA_FAIL: return "fail";
|
|
127
|
+
case SCHEMA_WARN: return "warn";
|
|
128
|
+
case SCHEMA_FILTER: return "filter";
|
|
129
|
+
case SCHEMA_QUARANTINE: return "quarantine";
|
|
130
|
+
case SCHEMA_ANNOTATE: return "annotate";
|
|
131
|
+
default: return "fail";
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
static int parse_action(const char *s, schema_action *out) {
|
|
136
|
+
if (!s || strcmp(s, "fail") == 0 || strcmp(s, "error") == 0 || strcmp(s, "stop") == 0) {
|
|
137
|
+
*out = SCHEMA_FAIL;
|
|
138
|
+
return 1;
|
|
139
|
+
}
|
|
140
|
+
if (strcmp(s, "warn") == 0 || strcmp(s, "warning") == 0) { *out = SCHEMA_WARN; return 1; }
|
|
141
|
+
if (strcmp(s, "filter") == 0 || strcmp(s, "drop") == 0) { *out = SCHEMA_FILTER; return 1; }
|
|
142
|
+
if (strcmp(s, "quarantine") == 0) { *out = SCHEMA_QUARANTINE; return 1; }
|
|
143
|
+
if (strcmp(s, "annotate") == 0) { *out = SCHEMA_ANNOTATE; return 1; }
|
|
144
|
+
return 0;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
static int parse_expected_type(const char *s, schema_expected_type *out) {
|
|
148
|
+
if (!s || !s[0] || strcmp(s, "any") == 0 || strcmp(s, "*") == 0) {
|
|
149
|
+
*out = SCHEMA_EXPECT_ANY;
|
|
150
|
+
return 1;
|
|
151
|
+
}
|
|
152
|
+
if (strcmp(s, "bool") == 0 || strcmp(s, "boolean") == 0) { *out = SCHEMA_EXPECT_BOOL; return 1; }
|
|
153
|
+
if (strcmp(s, "int") == 0 || strcmp(s, "int64") == 0 || strcmp(s, "integer") == 0) { *out = SCHEMA_EXPECT_INT; return 1; }
|
|
154
|
+
if (strcmp(s, "float") == 0 || strcmp(s, "float64") == 0 || strcmp(s, "double") == 0) { *out = SCHEMA_EXPECT_FLOAT; return 1; }
|
|
155
|
+
if (strcmp(s, "number") == 0 || strcmp(s, "numeric") == 0) { *out = SCHEMA_EXPECT_NUMBER; return 1; }
|
|
156
|
+
if (strcmp(s, "string") == 0 || strcmp(s, "str") == 0 || strcmp(s, "text") == 0) { *out = SCHEMA_EXPECT_STRING; return 1; }
|
|
157
|
+
if (strcmp(s, "date") == 0) { *out = SCHEMA_EXPECT_DATE; return 1; }
|
|
158
|
+
if (strcmp(s, "timestamp") == 0 || strcmp(s, "datetime") == 0) { *out = SCHEMA_EXPECT_TIMESTAMP; return 1; }
|
|
159
|
+
return 0;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
static const char *type_name(tf_type type) {
|
|
163
|
+
switch (type) {
|
|
164
|
+
case TF_TYPE_BOOL: return "bool";
|
|
165
|
+
case TF_TYPE_INT64: return "int";
|
|
166
|
+
case TF_TYPE_FLOAT64: return "float";
|
|
167
|
+
case TF_TYPE_STRING: return "string";
|
|
168
|
+
case TF_TYPE_DATE: return "date";
|
|
169
|
+
case TF_TYPE_TIMESTAMP: return "timestamp";
|
|
170
|
+
case TF_TYPE_NULL: return "null";
|
|
171
|
+
default: return "unknown";
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
static int type_matches(schema_expected_type expected, tf_type actual) {
|
|
176
|
+
switch (expected) {
|
|
177
|
+
case SCHEMA_EXPECT_ANY: return 1;
|
|
178
|
+
case SCHEMA_EXPECT_BOOL: return actual == TF_TYPE_BOOL;
|
|
179
|
+
case SCHEMA_EXPECT_INT: return actual == TF_TYPE_INT64;
|
|
180
|
+
case SCHEMA_EXPECT_FLOAT: return actual == TF_TYPE_FLOAT64;
|
|
181
|
+
case SCHEMA_EXPECT_NUMBER: return actual == TF_TYPE_INT64 || actual == TF_TYPE_FLOAT64;
|
|
182
|
+
case SCHEMA_EXPECT_STRING: return actual == TF_TYPE_STRING;
|
|
183
|
+
case SCHEMA_EXPECT_DATE: return actual == TF_TYPE_DATE;
|
|
184
|
+
case SCHEMA_EXPECT_TIMESTAMP: return actual == TF_TYPE_TIMESTAMP;
|
|
185
|
+
default: return 0;
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
static schema_rule *find_rule(schema_state *st, const char *column) {
|
|
190
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
191
|
+
if (strcmp(st->rules[i].column, column) == 0) return &st->rules[i];
|
|
192
|
+
}
|
|
193
|
+
return NULL;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
static cJSON *schema_get_item_any(const cJSON *obj, const char *first, const char *second) {
|
|
197
|
+
if (!obj) return NULL;
|
|
198
|
+
cJSON *item = cJSON_GetObjectItemCaseSensitive(obj, first);
|
|
199
|
+
if (!item && second) item = cJSON_GetObjectItemCaseSensitive(obj, second);
|
|
200
|
+
return item;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
static cJSON *schema_get_item_four(const cJSON *obj, const char *a, const char *b,
|
|
204
|
+
const char *c, const char *d) {
|
|
205
|
+
cJSON *item = schema_get_item_any(obj, a, b);
|
|
206
|
+
if (!item && c) item = cJSON_GetObjectItemCaseSensitive(obj, c);
|
|
207
|
+
if (!item && d) item = cJSON_GetObjectItemCaseSensitive(obj, d);
|
|
208
|
+
return item;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
static int schema_get_bool_option(const cJSON *obj, const char *a, const char *b,
|
|
212
|
+
const char *c, const char *d, int *out, int *seen) {
|
|
213
|
+
cJSON *item = schema_get_item_four(obj, a, b, c, d);
|
|
214
|
+
if (!item) return TF_OK;
|
|
215
|
+
if (!cJSON_IsBool(item)) return TF_ERROR;
|
|
216
|
+
*out = cJSON_IsTrue(item) ? 1 : 0;
|
|
217
|
+
if (seen) *seen = 1;
|
|
218
|
+
return TF_OK;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
static void schema_record_failure(schema_state *st, const char *rule) {
|
|
222
|
+
if (!st) return;
|
|
223
|
+
st->violation_count++;
|
|
224
|
+
if (!rule) {
|
|
225
|
+
st->schema_failures++;
|
|
226
|
+
} else if (strcmp(rule, "required") == 0) {
|
|
227
|
+
st->required_failures++;
|
|
228
|
+
} else if (strcmp(rule, "type") == 0) {
|
|
229
|
+
st->type_failures++;
|
|
230
|
+
} else if (strcmp(rule, "nullable") == 0) {
|
|
231
|
+
st->nullable_failures++;
|
|
232
|
+
} else if (strcmp(rule, "values") == 0) {
|
|
233
|
+
st->values_failures++;
|
|
234
|
+
} else if (strcmp(rule, "min") == 0) {
|
|
235
|
+
st->min_failures++;
|
|
236
|
+
} else if (strcmp(rule, "max") == 0) {
|
|
237
|
+
st->max_failures++;
|
|
238
|
+
} else if (strcmp(rule, "regex") == 0) {
|
|
239
|
+
st->regex_failures++;
|
|
240
|
+
} else if (strcmp(rule, "extra_column") == 0) {
|
|
241
|
+
st->extra_column_failures++;
|
|
242
|
+
} else if (strcmp(rule, "missing_category") == 0) {
|
|
243
|
+
st->missing_category_failures++;
|
|
244
|
+
} else {
|
|
245
|
+
st->schema_failures++;
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
static schema_rule *ensure_rule(schema_state *st, const char *column) {
|
|
250
|
+
schema_rule *existing = find_rule(st, column);
|
|
251
|
+
if (existing) return existing;
|
|
252
|
+
size_t next = 0;
|
|
253
|
+
if (tf_size_add(st->n_rules, 1, &next) != TF_OK) return NULL;
|
|
254
|
+
schema_rule *nr = tf_reallocarray_checked(st->rules, next, sizeof(schema_rule));
|
|
255
|
+
if (!nr) return NULL;
|
|
256
|
+
st->rules = nr;
|
|
257
|
+
schema_rule *r = &st->rules[st->n_rules];
|
|
258
|
+
st->n_rules = next;
|
|
259
|
+
memset(r, 0, sizeof(*r));
|
|
260
|
+
r->column = strdup(column);
|
|
261
|
+
r->expected_type = SCHEMA_EXPECT_ANY;
|
|
262
|
+
r->expected_type_name = strdup("any");
|
|
263
|
+
r->required = 0;
|
|
264
|
+
r->nullable = 1;
|
|
265
|
+
r->col_idx = -1;
|
|
266
|
+
if (!r->column || !r->expected_type_name) return NULL;
|
|
267
|
+
return r;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
static int add_value(schema_rule *r, const char *value) {
|
|
271
|
+
size_t next = 0;
|
|
272
|
+
if (tf_size_add(r->n_values, 1, &next) != TF_OK) return TF_ERROR;
|
|
273
|
+
char **nv = tf_reallocarray_checked(r->values, next, sizeof(char *));
|
|
274
|
+
if (!nv) return TF_ERROR;
|
|
275
|
+
r->values = nv;
|
|
276
|
+
r->values[r->n_values] = strdup(value ? value : "");
|
|
277
|
+
if (!r->values[r->n_values]) return TF_ERROR;
|
|
278
|
+
r->n_values++;
|
|
279
|
+
r->has_values = 1;
|
|
280
|
+
return TF_OK;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
static int add_json_value(schema_rule *r, const cJSON *item) {
|
|
284
|
+
char buf[128];
|
|
285
|
+
if (cJSON_IsString(item)) return add_value(r, item->valuestring);
|
|
286
|
+
if (cJSON_IsBool(item)) return add_value(r, cJSON_IsTrue(item) ? "true" : "false");
|
|
287
|
+
if (cJSON_IsNumber(item)) {
|
|
288
|
+
snprintf(buf, sizeof(buf), "%.17g", item->valuedouble);
|
|
289
|
+
return add_value(r, buf);
|
|
290
|
+
}
|
|
291
|
+
if (cJSON_IsNull(item)) return add_value(r, "");
|
|
292
|
+
return TF_OK;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
static int rule_set_expected_type(schema_rule *r, const char *type_s) {
|
|
296
|
+
schema_expected_type parsed;
|
|
297
|
+
if (!parse_expected_type(type_s, &parsed)) return TF_ERROR;
|
|
298
|
+
char *name = strdup(type_s);
|
|
299
|
+
if (!name) return TF_ERROR;
|
|
300
|
+
free(r->expected_type_name);
|
|
301
|
+
r->expected_type_name = name;
|
|
302
|
+
r->expected_type = parsed;
|
|
303
|
+
r->check_type = parsed != SCHEMA_EXPECT_ANY;
|
|
304
|
+
return TF_OK;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
static int schema_regex_pattern_within_cap(const schema_state *st, const char *pattern) {
|
|
308
|
+
size_t len = strlen(pattern ? pattern : "");
|
|
309
|
+
return len <= st->max_regex_pattern_bytes;
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
static int rule_set_regex_pattern(schema_state *st, schema_rule *r, const char *pattern, int compile_now) {
|
|
313
|
+
if (!schema_regex_pattern_within_cap(st, pattern)) {
|
|
314
|
+
tf_set_last_error("schema: regex pattern exceeds max_regex_pattern_bytes");
|
|
315
|
+
return TF_ERROR;
|
|
316
|
+
}
|
|
317
|
+
char *copy = strdup(pattern ? pattern : "");
|
|
318
|
+
if (!copy) return TF_ERROR;
|
|
319
|
+
if (r->compiled_regex) {
|
|
320
|
+
regfree(&r->regex);
|
|
321
|
+
r->compiled_regex = 0;
|
|
322
|
+
}
|
|
323
|
+
free(r->regex_pattern);
|
|
324
|
+
r->regex_pattern = copy;
|
|
325
|
+
r->has_regex = 1;
|
|
326
|
+
if (compile_now) {
|
|
327
|
+
if (regcomp(&r->regex, r->regex_pattern, REG_EXTENDED | REG_NOSUB) != 0) return TF_ERROR;
|
|
328
|
+
r->compiled_regex = 1;
|
|
329
|
+
}
|
|
330
|
+
return TF_OK;
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
static schema_selector_rule *add_selector_rule_items(schema_state *st, char **items, size_t n_items) {
|
|
334
|
+
if (!items || n_items == 0) return NULL;
|
|
335
|
+
size_t next = 0;
|
|
336
|
+
if (tf_size_add(st->n_selector_rules, 1, &next) != TF_OK) return NULL;
|
|
337
|
+
schema_selector_rule *nr = tf_reallocarray_checked(st->selector_rules, next, sizeof(schema_selector_rule));
|
|
338
|
+
if (!nr) return NULL;
|
|
339
|
+
st->selector_rules = nr;
|
|
340
|
+
schema_selector_rule *sr = &st->selector_rules[st->n_selector_rules];
|
|
341
|
+
st->n_selector_rules = next;
|
|
342
|
+
memset(sr, 0, sizeof(*sr));
|
|
343
|
+
sr->selectors = tf_callocarray_checked(n_items, sizeof(char *));
|
|
344
|
+
if (!sr->selectors) return NULL;
|
|
345
|
+
sr->n_selectors = n_items;
|
|
346
|
+
sr->expected_type = SCHEMA_EXPECT_ANY;
|
|
347
|
+
for (size_t i = 0; i < n_items; i++) {
|
|
348
|
+
sr->selectors[i] = strdup(items[i] ? items[i] : "");
|
|
349
|
+
if (!sr->selectors[i]) return NULL;
|
|
350
|
+
}
|
|
351
|
+
return sr;
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
static schema_selector_rule *add_selector_rule_single(schema_state *st, const char *selector) {
|
|
355
|
+
char *items[1];
|
|
356
|
+
items[0] = (char *)(selector ? selector : "");
|
|
357
|
+
return add_selector_rule_items(st, items, 1);
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
static schema_selector_rule *add_selector_rule_json_array(schema_state *st, const cJSON *arr) {
|
|
361
|
+
if (!arr || !cJSON_IsArray(arr)) return NULL;
|
|
362
|
+
int n = cJSON_GetArraySize(arr);
|
|
363
|
+
if (n <= 0) return NULL;
|
|
364
|
+
char **items = tf_callocarray_checked((size_t)n, sizeof(char *));
|
|
365
|
+
if (!items) return NULL;
|
|
366
|
+
for (int i = 0; i < n; i++) {
|
|
367
|
+
cJSON *item = cJSON_GetArrayItem(arr, i);
|
|
368
|
+
if (!cJSON_IsString(item) || !item->valuestring[0]) { free(items); return NULL; }
|
|
369
|
+
items[i] = item->valuestring;
|
|
370
|
+
}
|
|
371
|
+
schema_selector_rule *sr = add_selector_rule_items(st, items, (size_t)n);
|
|
372
|
+
free(items);
|
|
373
|
+
return sr;
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
static int selector_rule_set_expected_type(schema_selector_rule *sr, const char *type_s) {
|
|
377
|
+
schema_expected_type parsed;
|
|
378
|
+
if (!parse_expected_type(type_s, &parsed)) return TF_ERROR;
|
|
379
|
+
char *name = strdup(type_s);
|
|
380
|
+
if (!name) return TF_ERROR;
|
|
381
|
+
free(sr->expected_type_name);
|
|
382
|
+
sr->expected_type_name = name;
|
|
383
|
+
sr->expected_type = parsed;
|
|
384
|
+
sr->type_set = 1;
|
|
385
|
+
return TF_OK;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
static int selector_rule_add_value(schema_selector_rule *sr, const char *value) {
|
|
389
|
+
size_t next = 0;
|
|
390
|
+
if (tf_size_add(sr->n_values, 1, &next) != TF_OK) return TF_ERROR;
|
|
391
|
+
char **nv = tf_reallocarray_checked(sr->values, next, sizeof(char *));
|
|
392
|
+
if (!nv) return TF_ERROR;
|
|
393
|
+
sr->values = nv;
|
|
394
|
+
sr->values[sr->n_values] = strdup(value ? value : "");
|
|
395
|
+
if (!sr->values[sr->n_values]) return TF_ERROR;
|
|
396
|
+
sr->n_values++;
|
|
397
|
+
sr->has_values = 1;
|
|
398
|
+
return TF_OK;
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
static int selector_rule_add_json_value(schema_selector_rule *sr, const cJSON *item) {
|
|
402
|
+
char buf[128];
|
|
403
|
+
if (cJSON_IsString(item)) return selector_rule_add_value(sr, item->valuestring);
|
|
404
|
+
if (cJSON_IsBool(item)) return selector_rule_add_value(sr, cJSON_IsTrue(item) ? "true" : "false");
|
|
405
|
+
if (cJSON_IsNumber(item)) {
|
|
406
|
+
snprintf(buf, sizeof(buf), "%.17g", item->valuedouble);
|
|
407
|
+
return selector_rule_add_value(sr, buf);
|
|
408
|
+
}
|
|
409
|
+
if (cJSON_IsNull(item)) return selector_rule_add_value(sr, "");
|
|
410
|
+
return TF_OK;
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
static int parse_string_list_into_selector_rule(schema_selector_rule *sr, const cJSON *arr) {
|
|
414
|
+
if (!cJSON_IsArray(arr)) return TF_ERROR;
|
|
415
|
+
cJSON *item = NULL;
|
|
416
|
+
cJSON_ArrayForEach(item, arr) {
|
|
417
|
+
if (selector_rule_add_json_value(sr, item) != TF_OK) return TF_ERROR;
|
|
418
|
+
}
|
|
419
|
+
return TF_OK;
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
static int selector_rule_set_regex_pattern(schema_state *st, schema_selector_rule *sr, const char *pattern) {
|
|
423
|
+
if (!schema_regex_pattern_within_cap(st, pattern)) {
|
|
424
|
+
tf_set_last_error("schema: regex pattern exceeds max_regex_pattern_bytes");
|
|
425
|
+
return TF_ERROR;
|
|
426
|
+
}
|
|
427
|
+
char *copy = strdup(pattern ? pattern : "");
|
|
428
|
+
if (!copy) return TF_ERROR;
|
|
429
|
+
free(sr->regex_pattern);
|
|
430
|
+
sr->regex_pattern = copy;
|
|
431
|
+
sr->has_regex = 1;
|
|
432
|
+
return TF_OK;
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
static int parse_string_list_into_rule(schema_rule *r, const cJSON *arr) {
|
|
436
|
+
if (!cJSON_IsArray(arr)) return TF_ERROR;
|
|
437
|
+
cJSON *item = NULL;
|
|
438
|
+
cJSON_ArrayForEach(item, arr) {
|
|
439
|
+
if (add_json_value(r, item) != TF_OK) return TF_ERROR;
|
|
440
|
+
}
|
|
441
|
+
return TF_OK;
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
static int parse_columns(schema_state *st, const cJSON *columns) {
|
|
445
|
+
if (!columns) return TF_OK;
|
|
446
|
+
if (!cJSON_IsObject(columns)) return TF_ERROR;
|
|
447
|
+
cJSON *entry = NULL;
|
|
448
|
+
cJSON_ArrayForEach(entry, columns) {
|
|
449
|
+
if (!entry->string || !entry->string[0]) return TF_ERROR;
|
|
450
|
+
int is_selector = tf_column_selector_has_syntax(entry->string);
|
|
451
|
+
schema_rule *r = NULL;
|
|
452
|
+
schema_selector_rule *sr = NULL;
|
|
453
|
+
if (is_selector) {
|
|
454
|
+
sr = add_selector_rule_single(st, entry->string);
|
|
455
|
+
if (!sr) return TF_ERROR;
|
|
456
|
+
sr->required_set = 1;
|
|
457
|
+
sr->required = 1;
|
|
458
|
+
} else {
|
|
459
|
+
r = ensure_rule(st, entry->string);
|
|
460
|
+
if (!r) return TF_ERROR;
|
|
461
|
+
r->required = 1;
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
const char *type_s = NULL;
|
|
465
|
+
if (cJSON_IsString(entry)) {
|
|
466
|
+
type_s = entry->valuestring;
|
|
467
|
+
} else if (cJSON_IsObject(entry)) {
|
|
468
|
+
cJSON *type_j = cJSON_GetObjectItemCaseSensitive(entry, "type");
|
|
469
|
+
cJSON *required_j = cJSON_GetObjectItemCaseSensitive(entry, "required");
|
|
470
|
+
cJSON *nullable_j = cJSON_GetObjectItemCaseSensitive(entry, "nullable");
|
|
471
|
+
cJSON *values_j = cJSON_GetObjectItemCaseSensitive(entry, "values");
|
|
472
|
+
cJSON *min_j = cJSON_GetObjectItemCaseSensitive(entry, "min");
|
|
473
|
+
cJSON *max_j = cJSON_GetObjectItemCaseSensitive(entry, "max");
|
|
474
|
+
cJSON *regex_j = cJSON_GetObjectItemCaseSensitive(entry, "regex");
|
|
475
|
+
if (cJSON_IsString(type_j)) type_s = type_j->valuestring;
|
|
476
|
+
if (cJSON_IsBool(required_j)) {
|
|
477
|
+
if (sr) { sr->required_set = 1; sr->required = cJSON_IsTrue(required_j) ? 1 : 0; }
|
|
478
|
+
else r->required = cJSON_IsTrue(required_j) ? 1 : 0;
|
|
479
|
+
}
|
|
480
|
+
if (cJSON_IsBool(nullable_j)) {
|
|
481
|
+
if (sr) { sr->nullable_set = 1; sr->nullable = cJSON_IsTrue(nullable_j) ? 1 : 0; }
|
|
482
|
+
else r->nullable = cJSON_IsTrue(nullable_j) ? 1 : 0;
|
|
483
|
+
}
|
|
484
|
+
if (values_j) {
|
|
485
|
+
if (sr) {
|
|
486
|
+
if (parse_string_list_into_selector_rule(sr, values_j) != TF_OK) return TF_ERROR;
|
|
487
|
+
} else if (parse_string_list_into_rule(r, values_j) != TF_OK) {
|
|
488
|
+
return TF_ERROR;
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
if (cJSON_IsNumber(min_j)) {
|
|
492
|
+
if (sr) { sr->has_min = 1; sr->min_value = min_j->valuedouble; }
|
|
493
|
+
else { r->has_min = 1; r->min_value = min_j->valuedouble; }
|
|
494
|
+
}
|
|
495
|
+
if (cJSON_IsNumber(max_j)) {
|
|
496
|
+
if (sr) { sr->has_max = 1; sr->max_value = max_j->valuedouble; }
|
|
497
|
+
else { r->has_max = 1; r->max_value = max_j->valuedouble; }
|
|
498
|
+
}
|
|
499
|
+
if (cJSON_IsString(regex_j)) {
|
|
500
|
+
if (sr) {
|
|
501
|
+
if (selector_rule_set_regex_pattern(st, sr, regex_j->valuestring) != TF_OK) return TF_ERROR;
|
|
502
|
+
} else if (rule_set_regex_pattern(st, r, regex_j->valuestring, 0) != TF_OK) {
|
|
503
|
+
return TF_ERROR;
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
} else {
|
|
507
|
+
return TF_ERROR;
|
|
508
|
+
}
|
|
509
|
+
if (type_s) {
|
|
510
|
+
if (sr) {
|
|
511
|
+
if (selector_rule_set_expected_type(sr, type_s) != TF_OK) return TF_ERROR;
|
|
512
|
+
} else if (rule_set_expected_type(r, type_s) != TF_OK) {
|
|
513
|
+
return TF_ERROR;
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
return TF_OK;
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
static int parse_string_array_rules(schema_state *st, const cJSON *arr, int required, int nullable) {
|
|
521
|
+
if (!arr) return TF_OK;
|
|
522
|
+
if (!cJSON_IsArray(arr)) return TF_ERROR;
|
|
523
|
+
if (tf_column_selectors_have_syntax_json(arr)) {
|
|
524
|
+
schema_selector_rule *sr = add_selector_rule_json_array(st, arr);
|
|
525
|
+
if (!sr) return TF_ERROR;
|
|
526
|
+
if (required >= 0) { sr->required_set = 1; sr->required = required; }
|
|
527
|
+
if (nullable >= 0) { sr->nullable_set = 1; sr->nullable = nullable; }
|
|
528
|
+
return TF_OK;
|
|
529
|
+
}
|
|
530
|
+
cJSON *item = NULL;
|
|
531
|
+
cJSON_ArrayForEach(item, arr) {
|
|
532
|
+
if (!cJSON_IsString(item) || !item->valuestring[0]) return TF_ERROR;
|
|
533
|
+
schema_rule *r = ensure_rule(st, item->valuestring);
|
|
534
|
+
if (!r) return TF_ERROR;
|
|
535
|
+
if (required >= 0) r->required = required;
|
|
536
|
+
if (nullable >= 0) r->nullable = nullable;
|
|
537
|
+
}
|
|
538
|
+
return TF_OK;
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
static int parse_values_map(schema_state *st, const cJSON *obj) {
|
|
542
|
+
if (!obj) return TF_OK;
|
|
543
|
+
if (!cJSON_IsObject(obj)) return TF_ERROR;
|
|
544
|
+
cJSON *entry = NULL;
|
|
545
|
+
cJSON_ArrayForEach(entry, obj) {
|
|
546
|
+
if (!entry->string || !entry->string[0]) return TF_ERROR;
|
|
547
|
+
if (tf_column_selector_has_syntax(entry->string)) {
|
|
548
|
+
schema_selector_rule *sr = add_selector_rule_single(st, entry->string);
|
|
549
|
+
if (!sr) return TF_ERROR;
|
|
550
|
+
sr->required_set = 1;
|
|
551
|
+
sr->required = 1;
|
|
552
|
+
if (parse_string_list_into_selector_rule(sr, entry) != TF_OK) return TF_ERROR;
|
|
553
|
+
} else {
|
|
554
|
+
schema_rule *r = ensure_rule(st, entry->string);
|
|
555
|
+
if (!r) return TF_ERROR;
|
|
556
|
+
r->required = 1;
|
|
557
|
+
if (parse_string_list_into_rule(r, entry) != TF_OK) return TF_ERROR;
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
return TF_OK;
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
static int parse_number_map(schema_state *st, const cJSON *obj, int is_min) {
|
|
564
|
+
if (!obj) return TF_OK;
|
|
565
|
+
if (!cJSON_IsObject(obj)) return TF_ERROR;
|
|
566
|
+
cJSON *entry = NULL;
|
|
567
|
+
cJSON_ArrayForEach(entry, obj) {
|
|
568
|
+
if (!entry->string || !entry->string[0] || !cJSON_IsNumber(entry)) return TF_ERROR;
|
|
569
|
+
if (tf_column_selector_has_syntax(entry->string)) {
|
|
570
|
+
schema_selector_rule *sr = add_selector_rule_single(st, entry->string);
|
|
571
|
+
if (!sr) return TF_ERROR;
|
|
572
|
+
sr->required_set = 1;
|
|
573
|
+
sr->required = 1;
|
|
574
|
+
if (is_min) { sr->has_min = 1; sr->min_value = entry->valuedouble; }
|
|
575
|
+
else { sr->has_max = 1; sr->max_value = entry->valuedouble; }
|
|
576
|
+
} else {
|
|
577
|
+
schema_rule *r = ensure_rule(st, entry->string);
|
|
578
|
+
if (!r) return TF_ERROR;
|
|
579
|
+
r->required = 1;
|
|
580
|
+
if (is_min) { r->has_min = 1; r->min_value = entry->valuedouble; }
|
|
581
|
+
else { r->has_max = 1; r->max_value = entry->valuedouble; }
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
return TF_OK;
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
static int parse_regex_map(schema_state *st, const cJSON *obj) {
|
|
588
|
+
if (!obj) return TF_OK;
|
|
589
|
+
if (!cJSON_IsObject(obj)) return TF_ERROR;
|
|
590
|
+
cJSON *entry = NULL;
|
|
591
|
+
cJSON_ArrayForEach(entry, obj) {
|
|
592
|
+
if (!entry->string || !entry->string[0] || !cJSON_IsString(entry)) return TF_ERROR;
|
|
593
|
+
if (tf_column_selector_has_syntax(entry->string)) {
|
|
594
|
+
schema_selector_rule *sr = add_selector_rule_single(st, entry->string);
|
|
595
|
+
if (!sr) return TF_ERROR;
|
|
596
|
+
sr->required_set = 1;
|
|
597
|
+
sr->required = 1;
|
|
598
|
+
if (selector_rule_set_regex_pattern(st, sr, entry->valuestring) != TF_OK) return TF_ERROR;
|
|
599
|
+
} else {
|
|
600
|
+
schema_rule *r = ensure_rule(st, entry->string);
|
|
601
|
+
if (!r) return TF_ERROR;
|
|
602
|
+
r->required = 1;
|
|
603
|
+
if (rule_set_regex_pattern(st, r, entry->valuestring, 0) != TF_OK) return TF_ERROR;
|
|
604
|
+
}
|
|
605
|
+
}
|
|
606
|
+
return TF_OK;
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
static int parse_schema_contract(schema_state *st, const cJSON *obj) {
|
|
610
|
+
if (!obj) return TF_OK;
|
|
611
|
+
if (!cJSON_IsObject(obj)) return TF_ERROR;
|
|
612
|
+
if (parse_columns(st, schema_get_item_any(obj, "columns", NULL)) != TF_OK ||
|
|
613
|
+
parse_string_array_rules(st, schema_get_item_any(obj, "required", NULL), 1, -1) != TF_OK ||
|
|
614
|
+
parse_string_array_rules(st, schema_get_item_any(obj, "non_null", "nonNull"), 1, 0) != TF_OK ||
|
|
615
|
+
parse_string_array_rules(st, schema_get_item_any(obj, "nullable", NULL), -1, 1) != TF_OK ||
|
|
616
|
+
parse_values_map(st, schema_get_item_any(obj, "values", NULL)) != TF_OK ||
|
|
617
|
+
parse_number_map(st, schema_get_item_any(obj, "min", NULL), 1) != TF_OK ||
|
|
618
|
+
parse_number_map(st, schema_get_item_any(obj, "max", NULL), 0) != TF_OK ||
|
|
619
|
+
parse_regex_map(st, schema_get_item_any(obj, "regex", NULL)) != TF_OK) {
|
|
620
|
+
return TF_ERROR;
|
|
621
|
+
}
|
|
622
|
+
return TF_OK;
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
static int parse_schema_drift_options(schema_state *st, const cJSON *obj) {
|
|
626
|
+
int seen = 0;
|
|
627
|
+
int value = 0;
|
|
628
|
+
if (schema_get_bool_option(obj, "allow_extra_columns", "allowExtraColumns",
|
|
629
|
+
NULL, NULL, &value, &seen) != TF_OK) {
|
|
630
|
+
return TF_ERROR;
|
|
631
|
+
}
|
|
632
|
+
if (seen) {
|
|
633
|
+
st->allow_extra_columns = value;
|
|
634
|
+
st->check_extra_columns = 1;
|
|
635
|
+
}
|
|
636
|
+
seen = 0;
|
|
637
|
+
value = 0;
|
|
638
|
+
if (schema_get_bool_option(obj, "require_values_seen", "requireValuesSeen",
|
|
639
|
+
"require_all_values", "requireAllValues",
|
|
640
|
+
&value, &seen) != TF_OK) {
|
|
641
|
+
return TF_ERROR;
|
|
642
|
+
}
|
|
643
|
+
if (seen) st->require_values_seen = value;
|
|
644
|
+
return TF_OK;
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
static int cell_to_string(const tf_batch *b, size_t row, size_t col, char *buf, size_t buf_size) {
|
|
648
|
+
const char *text = NULL;
|
|
649
|
+
if (tf_batch_format_cell_as_string(b, row, col, TF_CELL_STRING_NUMERIC_TIME,
|
|
650
|
+
buf, buf_size, &text) != TF_OK) {
|
|
651
|
+
if (buf_size) buf[0] = '\0';
|
|
652
|
+
return 0;
|
|
653
|
+
}
|
|
654
|
+
if (!text) return 0;
|
|
655
|
+
if (text != buf) snprintf(buf, buf_size, "%s", text);
|
|
656
|
+
return 1;
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
static int cell_to_number(const tf_batch *b, size_t row, size_t col, double *out) {
|
|
660
|
+
if (tf_batch_is_null(b, row, col)) return 0;
|
|
661
|
+
switch (b->col_types[col]) {
|
|
662
|
+
case TF_TYPE_INT64: *out = (double)tf_batch_get_int64(b, row, col); return 1;
|
|
663
|
+
case TF_TYPE_FLOAT64: *out = tf_batch_get_float64(b, row, col); return 1;
|
|
664
|
+
case TF_TYPE_DATE: *out = (double)tf_batch_get_date(b, row, col); return 1;
|
|
665
|
+
case TF_TYPE_TIMESTAMP: *out = (double)tf_batch_get_timestamp(b, row, col); return 1;
|
|
666
|
+
default: return 0;
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
static int emit_failure(schema_state *st, const char *rule, const char *column,
|
|
671
|
+
const char *expected, const char *actual,
|
|
672
|
+
const tf_batch *b, size_t source_row, size_t batch_row,
|
|
673
|
+
tf_side_channels *side, int include_row) {
|
|
674
|
+
if (!side || !side->errors) return TF_OK;
|
|
675
|
+
cJSON *obj = cJSON_CreateObject();
|
|
676
|
+
if (!obj) return TF_ERROR;
|
|
677
|
+
int rc = TF_ERROR;
|
|
678
|
+
if (tf_json_add_string(obj, "type", "schema_failure") != TF_OK ||
|
|
679
|
+
tf_json_add_string(obj, "op", "schema") != TF_OK ||
|
|
680
|
+
tf_json_add_string(obj, "action", schema_action_name(st->action)) != TF_OK ||
|
|
681
|
+
tf_json_add_string(obj, "severity", st->action == SCHEMA_WARN ? "warning" : "error") != TF_OK ||
|
|
682
|
+
tf_json_add_string(obj, "name", st->name ? st->name : "schema") != TF_OK ||
|
|
683
|
+
tf_json_add_string(obj, "rule", rule ? rule : "schema") != TF_OK) {
|
|
684
|
+
goto done;
|
|
685
|
+
}
|
|
686
|
+
if (column && tf_json_add_string(obj, "column", column) != TF_OK) goto done;
|
|
687
|
+
if (expected && tf_json_add_string(obj, "expected", expected) != TF_OK) goto done;
|
|
688
|
+
if (actual) {
|
|
689
|
+
if (tf_json_add_audit_string(obj, "actual", &st->audit_opts, column, actual) != TF_OK) {
|
|
690
|
+
goto done;
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
if (st->message && st->message[0] &&
|
|
694
|
+
tf_json_add_string(obj, "message", st->message) != TF_OK) {
|
|
695
|
+
goto done;
|
|
696
|
+
}
|
|
697
|
+
if (source_row > 0 && tf_json_add_number(obj, "row", (double)source_row) != TF_OK) goto done;
|
|
698
|
+
if (include_row && b) {
|
|
699
|
+
cJSON *row_obj = tf_audit_row_to_json(b, batch_row, &st->audit_opts);
|
|
700
|
+
if (row_obj) {
|
|
701
|
+
if (tf_json_add_item(obj, "data", row_obj) != TF_OK) goto done;
|
|
702
|
+
} else if (st->audit_opts.include_row) {
|
|
703
|
+
goto done;
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
rc = tf_buffer_write_json_line(side->errors, obj);
|
|
707
|
+
done:
|
|
708
|
+
cJSON_Delete(obj);
|
|
709
|
+
return rc;
|
|
710
|
+
}
|
|
711
|
+
|
|
712
|
+
static int emit_schema_audit(schema_state *st, const char *rule, const char *column,
|
|
713
|
+
const char *expected, const char *actual,
|
|
714
|
+
const tf_batch *b, size_t row, tf_side_channels *side) {
|
|
715
|
+
if (!st->audit || st->audit_emitted >= st->audit_limit || !side || !side->stats) return TF_OK;
|
|
716
|
+
cJSON *obj = cJSON_CreateObject();
|
|
717
|
+
if (!obj) return TF_ERROR;
|
|
718
|
+
int rc = TF_ERROR;
|
|
719
|
+
if (tf_json_add_string(obj, "type", "audit") != TF_OK ||
|
|
720
|
+
tf_json_add_string(obj, "op", "schema") != TF_OK ||
|
|
721
|
+
tf_json_add_string(obj, "event", "row_dropped") != TF_OK ||
|
|
722
|
+
tf_json_add_string(obj, "reason", "schema_failed") != TF_OK ||
|
|
723
|
+
tf_json_add_string(obj, "channel", "audit") != TF_OK ||
|
|
724
|
+
tf_json_add_string(obj, "action", schema_action_name(st->action)) != TF_OK ||
|
|
725
|
+
tf_json_add_string(obj, "name", st->name ? st->name : "schema") != TF_OK ||
|
|
726
|
+
tf_json_add_string(obj, "rule", rule ? rule : "schema") != TF_OK) {
|
|
727
|
+
goto done;
|
|
728
|
+
}
|
|
729
|
+
if (column && tf_json_add_string(obj, "column", column) != TF_OK) goto done;
|
|
730
|
+
if (expected && tf_json_add_string(obj, "expected", expected) != TF_OK) goto done;
|
|
731
|
+
if (actual) {
|
|
732
|
+
if (tf_json_add_audit_string(obj, "actual", &st->audit_opts, column, actual) != TF_OK) {
|
|
733
|
+
goto done;
|
|
734
|
+
}
|
|
735
|
+
}
|
|
736
|
+
if (st->message && st->message[0] &&
|
|
737
|
+
tf_json_add_string(obj, "message", st->message) != TF_OK) {
|
|
738
|
+
goto done;
|
|
739
|
+
}
|
|
740
|
+
if (tf_json_add_number(obj, "row", (double)st->row_index) != TF_OK) goto done;
|
|
741
|
+
cJSON *row_obj = tf_audit_row_to_json(b, row, &st->audit_opts);
|
|
742
|
+
if (row_obj) {
|
|
743
|
+
if (tf_json_add_item(obj, "data", row_obj) != TF_OK) goto done;
|
|
744
|
+
} else if (st->audit_opts.include_row) {
|
|
745
|
+
goto done;
|
|
746
|
+
}
|
|
747
|
+
rc = tf_buffer_write_json_line(side->stats, obj);
|
|
748
|
+
done:
|
|
749
|
+
cJSON_Delete(obj);
|
|
750
|
+
if (rc == TF_OK) st->audit_emitted++;
|
|
751
|
+
return rc;
|
|
752
|
+
}
|
|
753
|
+
|
|
754
|
+
static int handle_schema_level_failure(schema_state *st, const char *rule, const char *column,
|
|
755
|
+
const char *expected, const char *actual,
|
|
756
|
+
tf_side_channels *side) {
|
|
757
|
+
if (emit_failure(st, rule, column, expected, actual, NULL, 0, 0, side, 0) != TF_OK)
|
|
758
|
+
return TF_ERROR;
|
|
759
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_FILTER || st->action == SCHEMA_QUARANTINE) {
|
|
760
|
+
char msg[512];
|
|
761
|
+
snprintf(msg, sizeof(msg), "schema failed: %s column '%s' expected %s got %s",
|
|
762
|
+
rule ? rule : "schema", column ? column : "", expected ? expected : "", actual ? actual : "");
|
|
763
|
+
tf_set_last_error(msg);
|
|
764
|
+
return TF_ERROR;
|
|
765
|
+
}
|
|
766
|
+
return TF_OK;
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
static int apply_selector_rule_to_rule(schema_state *st, schema_rule *r, const schema_selector_rule *sr) {
|
|
770
|
+
if (sr->type_set) {
|
|
771
|
+
char *name = strdup(sr->expected_type_name ? sr->expected_type_name : "any");
|
|
772
|
+
if (!name) return TF_ERROR;
|
|
773
|
+
free(r->expected_type_name);
|
|
774
|
+
r->expected_type_name = name;
|
|
775
|
+
r->expected_type = sr->expected_type;
|
|
776
|
+
r->check_type = sr->expected_type != SCHEMA_EXPECT_ANY;
|
|
777
|
+
}
|
|
778
|
+
if (sr->required_set) r->required = sr->required;
|
|
779
|
+
if (sr->nullable_set) r->nullable = sr->nullable;
|
|
780
|
+
for (size_t i = 0; i < sr->n_values; i++) {
|
|
781
|
+
if (add_value(r, sr->values[i]) != TF_OK) return TF_ERROR;
|
|
782
|
+
}
|
|
783
|
+
if (sr->has_min) { r->has_min = 1; r->min_value = sr->min_value; }
|
|
784
|
+
if (sr->has_max) { r->has_max = 1; r->max_value = sr->max_value; }
|
|
785
|
+
if (sr->has_regex) {
|
|
786
|
+
if (rule_set_regex_pattern(st, r, sr->regex_pattern, 1) != TF_OK) return TF_ERROR;
|
|
787
|
+
}
|
|
788
|
+
return TF_OK;
|
|
789
|
+
}
|
|
790
|
+
|
|
791
|
+
static int expand_schema_selectors(schema_state *st, const tf_batch *in, tf_side_channels *side) {
|
|
792
|
+
if (st->selectors_expanded) return TF_OK;
|
|
793
|
+
st->selectors_expanded = 1;
|
|
794
|
+
for (size_t i = 0; i < st->n_selector_rules; i++) {
|
|
795
|
+
schema_selector_rule *sr = &st->selector_rules[i];
|
|
796
|
+
int *indices = NULL;
|
|
797
|
+
size_t n_indices = 0;
|
|
798
|
+
char *error = NULL;
|
|
799
|
+
if (tf_column_selectors_resolve(sr->selectors, sr->n_selectors,
|
|
800
|
+
in->col_names, in->col_types, in->n_cols,
|
|
801
|
+
&indices, &n_indices, &error) != TF_OK) {
|
|
802
|
+
char msg[512];
|
|
803
|
+
snprintf(msg, sizeof(msg), "schema selector failed: %s", error ? error : "invalid selector");
|
|
804
|
+
int err_rc = emit_failure(st, "selector", NULL, "matching columns",
|
|
805
|
+
error ? error : "invalid selector",
|
|
806
|
+
NULL, 0, 0, side, 0);
|
|
807
|
+
schema_record_failure(st, "schema");
|
|
808
|
+
tf_set_last_error(msg);
|
|
809
|
+
free(error);
|
|
810
|
+
if (err_rc != TF_OK) return TF_ERROR;
|
|
811
|
+
return TF_ERROR;
|
|
812
|
+
}
|
|
813
|
+
for (size_t j = 0; j < n_indices; j++) {
|
|
814
|
+
int idx = indices[j];
|
|
815
|
+
if (idx < 0 || (size_t)idx >= in->n_cols) { free(indices); return TF_ERROR; }
|
|
816
|
+
schema_rule *r = ensure_rule(st, in->col_names[(size_t)idx]);
|
|
817
|
+
if (!r) { free(indices); return TF_ERROR; }
|
|
818
|
+
if (apply_selector_rule_to_rule(st, r, sr) != TF_OK) { free(indices); return TF_ERROR; }
|
|
819
|
+
}
|
|
820
|
+
free(indices);
|
|
821
|
+
}
|
|
822
|
+
return TF_OK;
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
static int schema_rule_exists_for_column(const schema_state *st, const char *column) {
|
|
826
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
827
|
+
if (strcmp(st->rules[i].column, column) == 0) return 1;
|
|
828
|
+
}
|
|
829
|
+
return 0;
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
static int init_schema_value_seen(schema_state *st) {
|
|
833
|
+
if (!st->require_values_seen) return TF_OK;
|
|
834
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
835
|
+
schema_rule *r = &st->rules[i];
|
|
836
|
+
if (!r->has_values || r->n_values == 0 || r->values_seen || r->col_idx < 0) continue;
|
|
837
|
+
r->values_seen = tf_callocarray_checked(r->n_values, sizeof(uint8_t));
|
|
838
|
+
if (!r->values_seen) return TF_ERROR;
|
|
839
|
+
}
|
|
840
|
+
return TF_OK;
|
|
841
|
+
}
|
|
842
|
+
|
|
843
|
+
static int check_schema_once(schema_state *st, const tf_batch *in, tf_side_channels *side) {
|
|
844
|
+
if (st->checked_schema) return TF_OK;
|
|
845
|
+
st->checked_schema = 1;
|
|
846
|
+
st->schema_ok = 1;
|
|
847
|
+
if (expand_schema_selectors(st, in, side) != TF_OK) return TF_ERROR;
|
|
848
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
849
|
+
schema_rule *r = &st->rules[i];
|
|
850
|
+
r->col_idx = tf_batch_col_index(in, r->column);
|
|
851
|
+
if (r->col_idx < 0) {
|
|
852
|
+
if (r->required) {
|
|
853
|
+
st->schema_ok = 0;
|
|
854
|
+
schema_record_failure(st, "required");
|
|
855
|
+
if (handle_schema_level_failure(st, "required", r->column, "present", "missing", side) != TF_OK)
|
|
856
|
+
return TF_ERROR;
|
|
857
|
+
}
|
|
858
|
+
continue;
|
|
859
|
+
}
|
|
860
|
+
if (r->check_type && !type_matches(r->expected_type, in->col_types[(size_t)r->col_idx])) {
|
|
861
|
+
st->schema_ok = 0;
|
|
862
|
+
schema_record_failure(st, "type");
|
|
863
|
+
if (handle_schema_level_failure(st, "type", r->column,
|
|
864
|
+
r->expected_type_name ? r->expected_type_name : "type",
|
|
865
|
+
type_name(in->col_types[(size_t)r->col_idx]), side) != TF_OK)
|
|
866
|
+
return TF_ERROR;
|
|
867
|
+
}
|
|
868
|
+
}
|
|
869
|
+
if (st->check_extra_columns && !st->allow_extra_columns) {
|
|
870
|
+
for (size_t i = 0; i < in->n_cols; i++) {
|
|
871
|
+
const char *column = in->col_names[i] ? in->col_names[i] : "";
|
|
872
|
+
if (!schema_rule_exists_for_column(st, column)) {
|
|
873
|
+
st->schema_ok = 0;
|
|
874
|
+
schema_record_failure(st, "extra_column");
|
|
875
|
+
if (handle_schema_level_failure(st, "extra_column", column,
|
|
876
|
+
"absent", "present", side) != TF_OK) {
|
|
877
|
+
return TF_ERROR;
|
|
878
|
+
}
|
|
879
|
+
}
|
|
880
|
+
}
|
|
881
|
+
}
|
|
882
|
+
if (init_schema_value_seen(st) != TF_OK) return TF_ERROR;
|
|
883
|
+
return TF_OK;
|
|
884
|
+
}
|
|
885
|
+
|
|
886
|
+
static int row_rule_ok(schema_state *st, schema_rule *r, const tf_batch *in, size_t row,
|
|
887
|
+
tf_side_channels *side, const char **failed_rule,
|
|
888
|
+
const char **failed_column, const char **expected,
|
|
889
|
+
char *actual, size_t actual_size) {
|
|
890
|
+
if (r->col_idx < 0) return 1;
|
|
891
|
+
size_t col = (size_t)r->col_idx;
|
|
892
|
+
if (tf_batch_is_null(in, row, col)) {
|
|
893
|
+
if (!r->nullable) {
|
|
894
|
+
*failed_rule = "nullable";
|
|
895
|
+
*failed_column = r->column;
|
|
896
|
+
*expected = "non-null";
|
|
897
|
+
snprintf(actual, actual_size, "null");
|
|
898
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
899
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
900
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
901
|
+
return -1;
|
|
902
|
+
}
|
|
903
|
+
return 0;
|
|
904
|
+
}
|
|
905
|
+
return 1;
|
|
906
|
+
}
|
|
907
|
+
|
|
908
|
+
if (r->has_min || r->has_max) {
|
|
909
|
+
double val = 0.0;
|
|
910
|
+
if (!cell_to_number(in, row, col, &val)) {
|
|
911
|
+
*failed_rule = r->has_min ? "min" : "max";
|
|
912
|
+
*failed_column = r->column;
|
|
913
|
+
*expected = "numeric";
|
|
914
|
+
cell_to_string(in, row, col, actual, actual_size);
|
|
915
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
916
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
917
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
918
|
+
return -1;
|
|
919
|
+
}
|
|
920
|
+
return 0;
|
|
921
|
+
}
|
|
922
|
+
if (r->has_min && val < r->min_value) {
|
|
923
|
+
static char expbuf[64];
|
|
924
|
+
snprintf(expbuf, sizeof(expbuf), ">= %.17g", r->min_value);
|
|
925
|
+
*failed_rule = "min";
|
|
926
|
+
*failed_column = r->column;
|
|
927
|
+
*expected = expbuf;
|
|
928
|
+
snprintf(actual, actual_size, "%.17g", val);
|
|
929
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
930
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
931
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
932
|
+
return -1;
|
|
933
|
+
}
|
|
934
|
+
return 0;
|
|
935
|
+
}
|
|
936
|
+
if (r->has_max && val > r->max_value) {
|
|
937
|
+
static char expbuf[64];
|
|
938
|
+
snprintf(expbuf, sizeof(expbuf), "<= %.17g", r->max_value);
|
|
939
|
+
*failed_rule = "max";
|
|
940
|
+
*failed_column = r->column;
|
|
941
|
+
*expected = expbuf;
|
|
942
|
+
snprintf(actual, actual_size, "%.17g", val);
|
|
943
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
944
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
945
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
946
|
+
return -1;
|
|
947
|
+
}
|
|
948
|
+
return 0;
|
|
949
|
+
}
|
|
950
|
+
}
|
|
951
|
+
|
|
952
|
+
if (r->has_values) {
|
|
953
|
+
char valbuf[256];
|
|
954
|
+
cell_to_string(in, row, col, valbuf, sizeof(valbuf));
|
|
955
|
+
int found = 0;
|
|
956
|
+
size_t found_index = 0;
|
|
957
|
+
for (size_t i = 0; i < r->n_values; i++) {
|
|
958
|
+
if (strcmp(valbuf, r->values[i]) == 0) {
|
|
959
|
+
found = 1;
|
|
960
|
+
found_index = i;
|
|
961
|
+
break;
|
|
962
|
+
}
|
|
963
|
+
}
|
|
964
|
+
if (!found) {
|
|
965
|
+
*failed_rule = "values";
|
|
966
|
+
*failed_column = r->column;
|
|
967
|
+
*expected = "allowed value";
|
|
968
|
+
snprintf(actual, actual_size, "%s", valbuf);
|
|
969
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
970
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
971
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
972
|
+
return -1;
|
|
973
|
+
}
|
|
974
|
+
return 0;
|
|
975
|
+
}
|
|
976
|
+
if (r->values_seen && found_index < r->n_values) r->values_seen[found_index] = 1;
|
|
977
|
+
}
|
|
978
|
+
|
|
979
|
+
if (r->has_regex) {
|
|
980
|
+
if (in->col_types[col] != TF_TYPE_STRING) {
|
|
981
|
+
*failed_rule = "regex";
|
|
982
|
+
*failed_column = r->column;
|
|
983
|
+
*expected = "string";
|
|
984
|
+
snprintf(actual, actual_size, "%s", type_name(in->col_types[col]));
|
|
985
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
986
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
987
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
988
|
+
return -1;
|
|
989
|
+
}
|
|
990
|
+
return 0;
|
|
991
|
+
}
|
|
992
|
+
const char *s = tf_batch_get_string(in, row, col);
|
|
993
|
+
if (s && strlen(s) > st->max_regex_cell_bytes) {
|
|
994
|
+
*failed_rule = "regex";
|
|
995
|
+
*failed_column = r->column;
|
|
996
|
+
*expected = "regex cell within max_regex_cell_bytes";
|
|
997
|
+
snprintf(actual, actual_size, "cell exceeds %zu bytes", st->max_regex_cell_bytes);
|
|
998
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
999
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
1000
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
1001
|
+
return -1;
|
|
1002
|
+
}
|
|
1003
|
+
return 0;
|
|
1004
|
+
}
|
|
1005
|
+
if (!s || regexec(&r->regex, s, 0, NULL, 0) != 0) {
|
|
1006
|
+
*failed_rule = "regex";
|
|
1007
|
+
*failed_column = r->column;
|
|
1008
|
+
*expected = r->regex_pattern;
|
|
1009
|
+
snprintf(actual, actual_size, "%s", s ? s : "");
|
|
1010
|
+
if (st->action == SCHEMA_FAIL || st->action == SCHEMA_WARN || st->action == SCHEMA_QUARANTINE) {
|
|
1011
|
+
if (emit_failure(st, *failed_rule, r->column, *expected, actual, in, st->row_index, row, side,
|
|
1012
|
+
st->action == SCHEMA_FAIL || st->action == SCHEMA_QUARANTINE) != TF_OK)
|
|
1013
|
+
return -1;
|
|
1014
|
+
}
|
|
1015
|
+
return 0;
|
|
1016
|
+
}
|
|
1017
|
+
}
|
|
1018
|
+
|
|
1019
|
+
return 1;
|
|
1020
|
+
}
|
|
1021
|
+
|
|
1022
|
+
static int schema_process(tf_step *self, tf_batch *in, tf_batch **out,
|
|
1023
|
+
tf_side_channels *side) {
|
|
1024
|
+
schema_state *st = self->state;
|
|
1025
|
+
*out = NULL;
|
|
1026
|
+
if (check_schema_once(st, in, side) != TF_OK) return TF_ERROR;
|
|
1027
|
+
|
|
1028
|
+
size_t extra = st->action == SCHEMA_ANNOTATE ? 1u : 0u;
|
|
1029
|
+
size_t out_cols = 0;
|
|
1030
|
+
if (tf_size_add(in->n_cols, extra, &out_cols) != TF_OK) return TF_ERROR;
|
|
1031
|
+
tf_batch *ob = tf_batch_create(out_cols, in->n_rows > 0 ? in->n_rows : 1);
|
|
1032
|
+
if (!ob) return TF_ERROR;
|
|
1033
|
+
if (extra) {
|
|
1034
|
+
const char *names[] = {st->result && st->result[0] ? st->result : "_schema"};
|
|
1035
|
+
const tf_type types[] = {TF_TYPE_BOOL};
|
|
1036
|
+
if (tf_batch_clone_with_extra_cols(ob, in, names, types, 1) != TF_OK) { tf_batch_free(ob); return TF_ERROR; }
|
|
1037
|
+
} else if (tf_batch_clone_schema(ob, in) != TF_OK) {
|
|
1038
|
+
tf_batch_free(ob);
|
|
1039
|
+
return TF_ERROR;
|
|
1040
|
+
}
|
|
1041
|
+
|
|
1042
|
+
size_t out_row = 0;
|
|
1043
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
1044
|
+
st->row_index++;
|
|
1045
|
+
st->checked_rows++;
|
|
1046
|
+
int ok = st->schema_ok;
|
|
1047
|
+
const char *failed_rule = NULL;
|
|
1048
|
+
const char *failed_column = NULL;
|
|
1049
|
+
char expected_copy[256] = {0};
|
|
1050
|
+
char actual_copy[256] = {0};
|
|
1051
|
+
int run_row_rules = ok || st->action == SCHEMA_WARN || st->action == SCHEMA_ANNOTATE;
|
|
1052
|
+
if (run_row_rules) {
|
|
1053
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
1054
|
+
const char *rule = NULL;
|
|
1055
|
+
const char *column = NULL;
|
|
1056
|
+
const char *expected = NULL;
|
|
1057
|
+
char actual[256] = {0};
|
|
1058
|
+
int rule_ok = row_rule_ok(st, &st->rules[i], in, r, side,
|
|
1059
|
+
&rule, &column, &expected, actual, sizeof(actual));
|
|
1060
|
+
if (rule_ok < 0) {
|
|
1061
|
+
tf_batch_free(ob);
|
|
1062
|
+
return TF_ERROR;
|
|
1063
|
+
}
|
|
1064
|
+
if (!rule_ok) {
|
|
1065
|
+
schema_record_failure(st, rule);
|
|
1066
|
+
if (ok) {
|
|
1067
|
+
failed_rule = rule;
|
|
1068
|
+
failed_column = column;
|
|
1069
|
+
snprintf(expected_copy, sizeof(expected_copy), "%s", expected ? expected : "");
|
|
1070
|
+
snprintf(actual_copy, sizeof(actual_copy), "%s", actual);
|
|
1071
|
+
}
|
|
1072
|
+
ok = 0;
|
|
1073
|
+
if (st->action == SCHEMA_FAIL) break;
|
|
1074
|
+
}
|
|
1075
|
+
}
|
|
1076
|
+
}
|
|
1077
|
+
|
|
1078
|
+
if (!ok) {
|
|
1079
|
+
st->failed_rows++;
|
|
1080
|
+
if (st->action == SCHEMA_FAIL) {
|
|
1081
|
+
if (!failed_rule &&
|
|
1082
|
+
emit_failure(st, "schema", NULL, "valid schema", "invalid schema",
|
|
1083
|
+
in, st->row_index, r, side, 1) != TF_OK) {
|
|
1084
|
+
tf_batch_free(ob);
|
|
1085
|
+
return TF_ERROR;
|
|
1086
|
+
}
|
|
1087
|
+
char msg[512];
|
|
1088
|
+
snprintf(msg, sizeof(msg), "schema failed at row %zu%s%s",
|
|
1089
|
+
st->row_index,
|
|
1090
|
+
failed_rule ? ": " : "",
|
|
1091
|
+
failed_rule ? failed_rule : "");
|
|
1092
|
+
tf_set_last_error(msg);
|
|
1093
|
+
tf_batch_free(ob);
|
|
1094
|
+
return TF_ERROR;
|
|
1095
|
+
}
|
|
1096
|
+
if (st->action == SCHEMA_FILTER) {
|
|
1097
|
+
if (emit_schema_audit(st, failed_rule ? failed_rule : "schema", failed_column,
|
|
1098
|
+
expected_copy[0] ? expected_copy : NULL,
|
|
1099
|
+
actual_copy, in, r, side) != TF_OK) {
|
|
1100
|
+
tf_batch_free(ob);
|
|
1101
|
+
return TF_ERROR;
|
|
1102
|
+
}
|
|
1103
|
+
continue;
|
|
1104
|
+
}
|
|
1105
|
+
if (st->action == SCHEMA_QUARANTINE) continue;
|
|
1106
|
+
} else {
|
|
1107
|
+
st->passed_rows++;
|
|
1108
|
+
}
|
|
1109
|
+
|
|
1110
|
+
if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
|
|
1111
|
+
tf_batch_free(ob);
|
|
1112
|
+
return TF_ERROR;
|
|
1113
|
+
}
|
|
1114
|
+
if (st->action == SCHEMA_ANNOTATE &&
|
|
1115
|
+
tf_batch_set_bool(ob, out_row, in->n_cols, ok ? true : false) != TF_OK) {
|
|
1116
|
+
tf_batch_free(ob);
|
|
1117
|
+
return TF_ERROR;
|
|
1118
|
+
}
|
|
1119
|
+
if (tf_batch_expose_row(ob, out_row) != TF_OK) {
|
|
1120
|
+
tf_batch_free(ob);
|
|
1121
|
+
return TF_ERROR;
|
|
1122
|
+
}
|
|
1123
|
+
out_row++;
|
|
1124
|
+
}
|
|
1125
|
+
|
|
1126
|
+
if (ob->n_rows > 0 || st->action == SCHEMA_ANNOTATE || in->n_rows == 0) *out = ob;
|
|
1127
|
+
else tf_batch_free(ob);
|
|
1128
|
+
return TF_OK;
|
|
1129
|
+
}
|
|
1130
|
+
|
|
1131
|
+
static int handle_finish_failure(schema_state *st, const char *rule, const char *column,
|
|
1132
|
+
const char *expected, const char *actual,
|
|
1133
|
+
tf_side_channels *side) {
|
|
1134
|
+
if (emit_failure(st, rule, column, expected, actual, NULL, 0, 0, side, 0) != TF_OK) {
|
|
1135
|
+
return TF_ERROR;
|
|
1136
|
+
}
|
|
1137
|
+
if (st->action != SCHEMA_WARN) {
|
|
1138
|
+
char msg[512];
|
|
1139
|
+
snprintf(msg, sizeof(msg), "schema failed at finish: %s column '%s' expected %s got %s",
|
|
1140
|
+
rule ? rule : "schema", column ? column : "",
|
|
1141
|
+
expected ? expected : "", actual ? actual : "");
|
|
1142
|
+
tf_set_last_error(msg);
|
|
1143
|
+
return TF_ERROR;
|
|
1144
|
+
}
|
|
1145
|
+
return TF_OK;
|
|
1146
|
+
}
|
|
1147
|
+
|
|
1148
|
+
static int schema_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
|
|
1149
|
+
schema_state *st = self ? (schema_state *)self->state : NULL;
|
|
1150
|
+
*out = NULL;
|
|
1151
|
+
if (!st || !st->require_values_seen) return TF_OK;
|
|
1152
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
1153
|
+
schema_rule *r = &st->rules[i];
|
|
1154
|
+
if (!r->has_values || !r->values_seen || r->col_idx < 0) continue;
|
|
1155
|
+
for (size_t j = 0; j < r->n_values; j++) {
|
|
1156
|
+
if (r->values_seen[j]) continue;
|
|
1157
|
+
schema_record_failure(st, "missing_category");
|
|
1158
|
+
if (handle_finish_failure(st, "missing_category", r->column,
|
|
1159
|
+
r->values[j] ? r->values[j] : "",
|
|
1160
|
+
"missing", side) != TF_OK) {
|
|
1161
|
+
return TF_ERROR;
|
|
1162
|
+
}
|
|
1163
|
+
}
|
|
1164
|
+
}
|
|
1165
|
+
return TF_OK;
|
|
1166
|
+
}
|
|
1167
|
+
|
|
1168
|
+
static int schema_append_stats(tf_step *self, tf_buffer *out) {
|
|
1169
|
+
if (!self || !self->state || !out) return TF_ERROR;
|
|
1170
|
+
schema_state *st = self->state;
|
|
1171
|
+
char buf[1536];
|
|
1172
|
+
snprintf(buf, sizeof(buf),
|
|
1173
|
+
",\"checked_rows\":%zu,\"passed_rows\":%zu,\"failed_rows\":%zu,"
|
|
1174
|
+
"\"violation_count\":%zu,\"audit_emitted\":%zu,"
|
|
1175
|
+
"\"required_failures\":%zu,\"type_failures\":%zu,"
|
|
1176
|
+
"\"nullable_failures\":%zu,\"values_failures\":%zu,"
|
|
1177
|
+
"\"min_failures\":%zu,\"max_failures\":%zu,"
|
|
1178
|
+
"\"regex_failures\":%zu,\"schema_failures\":%zu,"
|
|
1179
|
+
"\"extra_column_failures\":%zu,\"missing_category_failures\":%zu,"
|
|
1180
|
+
"\"baseline_mode\":%s,\"allow_extra_columns\":%s,\"require_values_seen\":%s,"
|
|
1181
|
+
"\"max_regex_pattern_bytes\":%zu,\"max_regex_cell_bytes\":%zu",
|
|
1182
|
+
st->checked_rows, st->passed_rows, st->failed_rows,
|
|
1183
|
+
st->violation_count, st->audit_emitted,
|
|
1184
|
+
st->required_failures, st->type_failures,
|
|
1185
|
+
st->nullable_failures, st->values_failures,
|
|
1186
|
+
st->min_failures, st->max_failures,
|
|
1187
|
+
st->regex_failures, st->schema_failures,
|
|
1188
|
+
st->extra_column_failures, st->missing_category_failures,
|
|
1189
|
+
st->baseline_mode ? "true" : "false",
|
|
1190
|
+
st->allow_extra_columns ? "true" : "false",
|
|
1191
|
+
st->require_values_seen ? "true" : "false",
|
|
1192
|
+
st->max_regex_pattern_bytes, st->max_regex_cell_bytes);
|
|
1193
|
+
return tf_buffer_write_str(out, buf);
|
|
1194
|
+
}
|
|
1195
|
+
|
|
1196
|
+
static void schema_state_free(schema_state *st) {
|
|
1197
|
+
if (!st) return;
|
|
1198
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
1199
|
+
schema_rule *r = &st->rules[i];
|
|
1200
|
+
free(r->column);
|
|
1201
|
+
free(r->expected_type_name);
|
|
1202
|
+
for (size_t j = 0; j < r->n_values; j++) free(r->values[j]);
|
|
1203
|
+
free(r->values);
|
|
1204
|
+
free(r->values_seen);
|
|
1205
|
+
if (r->compiled_regex) regfree(&r->regex);
|
|
1206
|
+
free(r->regex_pattern);
|
|
1207
|
+
}
|
|
1208
|
+
free(st->rules);
|
|
1209
|
+
for (size_t i = 0; i < st->n_selector_rules; i++) {
|
|
1210
|
+
schema_selector_rule *sr = &st->selector_rules[i];
|
|
1211
|
+
for (size_t j = 0; j < sr->n_selectors; j++) free(sr->selectors[j]);
|
|
1212
|
+
free(sr->selectors);
|
|
1213
|
+
free(sr->expected_type_name);
|
|
1214
|
+
for (size_t j = 0; j < sr->n_values; j++) free(sr->values[j]);
|
|
1215
|
+
free(sr->values);
|
|
1216
|
+
free(sr->regex_pattern);
|
|
1217
|
+
}
|
|
1218
|
+
free(st->selector_rules);
|
|
1219
|
+
free(st->name);
|
|
1220
|
+
free(st->message);
|
|
1221
|
+
free(st->result);
|
|
1222
|
+
tf_audit_options_free(&st->audit_opts);
|
|
1223
|
+
free(st);
|
|
1224
|
+
}
|
|
1225
|
+
|
|
1226
|
+
static void schema_destroy(tf_step *self) {
|
|
1227
|
+
if (!self) return;
|
|
1228
|
+
schema_state_free((schema_state *)self->state);
|
|
1229
|
+
free(self);
|
|
1230
|
+
}
|
|
1231
|
+
|
|
1232
|
+
tf_step *tf_schema_create(const cJSON *args) {
|
|
1233
|
+
if (!args) return NULL;
|
|
1234
|
+
schema_state *st = calloc(1, sizeof(schema_state));
|
|
1235
|
+
if (!st) return NULL;
|
|
1236
|
+
st->schema_ok = 1;
|
|
1237
|
+
st->action = SCHEMA_FAIL;
|
|
1238
|
+
st->audit_limit = 1000;
|
|
1239
|
+
st->allow_extra_columns = 1;
|
|
1240
|
+
tf_audit_options_init(&st->audit_opts, 1);
|
|
1241
|
+
st->max_regex_pattern_bytes = TF_MAX_REGEX_PATTERN_BYTES;
|
|
1242
|
+
st->max_regex_cell_bytes = TF_MAX_REGEX_CELL_BYTES;
|
|
1243
|
+
|
|
1244
|
+
if (tf_json_get_size_arg(args, "max_regex_pattern_bytes", 1, TF_MAX_RECORD_BYTES,
|
|
1245
|
+
&st->max_regex_pattern_bytes, "schema") < 0 ||
|
|
1246
|
+
tf_json_get_size_arg(args, "maxRegexPatternBytes", 1, TF_MAX_RECORD_BYTES,
|
|
1247
|
+
&st->max_regex_pattern_bytes, "schema") < 0 ||
|
|
1248
|
+
tf_json_get_size_arg(args, "max_regex_cell_bytes", 1, TF_MAX_RECORD_BYTES,
|
|
1249
|
+
&st->max_regex_cell_bytes, "schema") < 0 ||
|
|
1250
|
+
tf_json_get_size_arg(args, "maxRegexCellBytes", 1, TF_MAX_RECORD_BYTES,
|
|
1251
|
+
&st->max_regex_cell_bytes, "schema") < 0) {
|
|
1252
|
+
schema_state_free(st);
|
|
1253
|
+
return NULL;
|
|
1254
|
+
}
|
|
1255
|
+
|
|
1256
|
+
cJSON *audit_j = cJSON_GetObjectItemCaseSensitive(args, "audit");
|
|
1257
|
+
st->audit = cJSON_IsTrue(audit_j) ? 1 : 0;
|
|
1258
|
+
cJSON *audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "audit_limit");
|
|
1259
|
+
if (!audit_limit_j) audit_limit_j = cJSON_GetObjectItemCaseSensitive(args, "auditLimit");
|
|
1260
|
+
if (audit_limit_j) {
|
|
1261
|
+
size_t parsed_limit = 0;
|
|
1262
|
+
if (tf_json_get_size_arg_any(args, "audit_limit", "auditLimit",
|
|
1263
|
+
1, TF_MAX_AUDIT_RECORDS,
|
|
1264
|
+
&parsed_limit, "schema") < 0) {
|
|
1265
|
+
schema_state_free(st);
|
|
1266
|
+
return NULL;
|
|
1267
|
+
}
|
|
1268
|
+
st->audit_limit = parsed_limit;
|
|
1269
|
+
}
|
|
1270
|
+
if (tf_audit_options_parse(&st->audit_opts, args, "schema") != TF_OK) {
|
|
1271
|
+
schema_state_free(st);
|
|
1272
|
+
return NULL;
|
|
1273
|
+
}
|
|
1274
|
+
|
|
1275
|
+
cJSON *mode_j = cJSON_GetObjectItemCaseSensitive(args, "mode");
|
|
1276
|
+
cJSON *action_j = cJSON_GetObjectItemCaseSensitive(args, "action");
|
|
1277
|
+
const char *action_s = cJSON_IsString(action_j) ? action_j->valuestring : (cJSON_IsString(mode_j) ? mode_j->valuestring : "fail");
|
|
1278
|
+
if (!parse_action(action_s, &st->action)) {
|
|
1279
|
+
tf_set_last_error("schema: mode must be fail, warn, filter, quarantine, or annotate");
|
|
1280
|
+
schema_state_free(st);
|
|
1281
|
+
return NULL;
|
|
1282
|
+
}
|
|
1283
|
+
|
|
1284
|
+
cJSON *name_j = cJSON_GetObjectItemCaseSensitive(args, "name");
|
|
1285
|
+
cJSON *message_j = cJSON_GetObjectItemCaseSensitive(args, "message");
|
|
1286
|
+
cJSON *result_j = cJSON_GetObjectItemCaseSensitive(args, "result");
|
|
1287
|
+
st->name = strdup(cJSON_IsString(name_j) && name_j->valuestring[0] ? name_j->valuestring : "schema");
|
|
1288
|
+
st->message = strdup(cJSON_IsString(message_j) ? message_j->valuestring : "");
|
|
1289
|
+
st->result = strdup(cJSON_IsString(result_j) && result_j->valuestring[0] ? result_j->valuestring : "_schema");
|
|
1290
|
+
if (!st->name || !st->message || !st->result) { schema_state_free(st); return NULL; }
|
|
1291
|
+
|
|
1292
|
+
tf_set_last_error(NULL);
|
|
1293
|
+
cJSON *baseline_j = cJSON_GetObjectItemCaseSensitive(args, "baseline");
|
|
1294
|
+
if (baseline_j) {
|
|
1295
|
+
if (!cJSON_IsObject(baseline_j)) {
|
|
1296
|
+
tf_set_last_error("schema: baseline must be an object");
|
|
1297
|
+
schema_state_free(st);
|
|
1298
|
+
return NULL;
|
|
1299
|
+
}
|
|
1300
|
+
st->baseline_mode = 1;
|
|
1301
|
+
st->check_extra_columns = 1;
|
|
1302
|
+
st->allow_extra_columns = 0;
|
|
1303
|
+
st->require_values_seen = 1;
|
|
1304
|
+
if (parse_schema_drift_options(st, baseline_j) != TF_OK ||
|
|
1305
|
+
parse_schema_contract(st, baseline_j) != TF_OK) {
|
|
1306
|
+
if (!tf_last_error() || !tf_last_error()[0])
|
|
1307
|
+
tf_set_last_error("schema: invalid baseline");
|
|
1308
|
+
schema_state_free(st);
|
|
1309
|
+
return NULL;
|
|
1310
|
+
}
|
|
1311
|
+
}
|
|
1312
|
+
|
|
1313
|
+
if (parse_schema_drift_options(st, args) != TF_OK ||
|
|
1314
|
+
parse_schema_contract(st, args) != TF_OK) {
|
|
1315
|
+
if (!tf_last_error() || !tf_last_error()[0])
|
|
1316
|
+
tf_set_last_error("schema: invalid schema arguments");
|
|
1317
|
+
schema_state_free(st);
|
|
1318
|
+
return NULL;
|
|
1319
|
+
}
|
|
1320
|
+
|
|
1321
|
+
for (size_t i = 0; i < st->n_rules; i++) {
|
|
1322
|
+
schema_rule *r = &st->rules[i];
|
|
1323
|
+
if (r->has_regex) {
|
|
1324
|
+
if (regcomp(&r->regex, r->regex_pattern, REG_EXTENDED | REG_NOSUB) != 0) {
|
|
1325
|
+
tf_set_last_error("schema: invalid regex");
|
|
1326
|
+
schema_state_free(st);
|
|
1327
|
+
return NULL;
|
|
1328
|
+
}
|
|
1329
|
+
r->compiled_regex = 1;
|
|
1330
|
+
}
|
|
1331
|
+
}
|
|
1332
|
+
|
|
1333
|
+
tf_step *step = calloc(1, sizeof(tf_step));
|
|
1334
|
+
if (!step) { schema_state_free(st); return NULL; }
|
|
1335
|
+
step->process = schema_process;
|
|
1336
|
+
step->flush = schema_flush;
|
|
1337
|
+
step->destroy = schema_destroy;
|
|
1338
|
+
step->append_stats = schema_append_stats;
|
|
1339
|
+
step->state = st;
|
|
1340
|
+
return step;
|
|
1341
|
+
}
|