tranfi 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -0
- package/NOTICE +8 -0
- package/README.md +272 -40
- package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
- package/app/index.html +1 -1
- package/binding.gyp +55 -3
- package/csrc/arena.c +7 -5
- package/csrc/batch.c +818 -71
- package/csrc/buffer.c +84 -8
- package/csrc/cJSON.c +262 -19
- package/csrc/cJSON.h +17 -1
- package/csrc/codec_csv.c +1074 -181
- package/csrc/codec_jsonl.c +830 -118
- package/csrc/codec_table.c +108 -78
- package/csrc/codec_text.c +286 -68
- package/csrc/compiler.c +31 -3
- package/csrc/config.h +21 -0
- package/csrc/dsl.c +4722 -485
- package/csrc/expr.c +363 -55
- package/csrc/expr.h +2 -0
- package/csrc/internal.h +316 -27
- package/csrc/ir.c +65 -18
- package/csrc/ir.h +41 -0
- package/csrc/ir_schema.c +20 -5
- package/csrc/ir_serialize.c +68 -6
- package/csrc/ir_sql.c +796 -185
- package/csrc/ir_validate.c +462 -6
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +879 -30
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +171 -21
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +167 -32
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +168 -29
- package/csrc/op_cast.c +383 -55
- package/csrc/op_clip.c +30 -19
- package/csrc/op_date_trunc.c +208 -34
- package/csrc/op_datetime.c +259 -77
- package/csrc/op_derive.c +65 -97
- package/csrc/op_diff.c +146 -30
- package/csrc/op_ewma.c +149 -30
- package/csrc/op_explode.c +124 -26
- package/csrc/op_fill_down.c +125 -53
- package/csrc/op_fill_null.c +176 -31
- package/csrc/op_filter.c +89 -40
- package/csrc/op_frequency.c +571 -43
- package/csrc/op_grep.c +36 -18
- package/csrc/op_group_agg.c +1790 -119
- package/csrc/op_hash.c +48 -15
- package/csrc/op_head.c +21 -86
- package/csrc/op_interpolate.c +268 -62
- package/csrc/op_join.c +2700 -182
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +328 -53
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +141 -89
- package/csrc/op_normalize.c +363 -79
- package/csrc/op_onehot.c +345 -73
- package/csrc/op_pivot.c +1546 -162
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +2062 -166
- package/csrc/op_rename.c +41 -50
- package/csrc/op_replace.c +270 -118
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +80 -23
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +265 -65
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +30 -87
- package/csrc/op_sort.c +670 -124
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +65 -28
- package/csrc/op_split_data.c +41 -9
- package/csrc/op_stack.c +178 -222
- package/csrc/op_stats.c +206 -110
- package/csrc/op_step.c +217 -55
- package/csrc/op_tail.c +21 -12
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +260 -53
- package/csrc/op_trim.c +48 -19
- package/csrc/op_unique.c +1193 -150
- package/csrc/op_unpivot.c +100 -66
- package/csrc/op_validate.c +601 -24
- package/csrc/op_window.c +492 -51
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +872 -99
- package/csrc/recipes.c +3 -1
- package/csrc/report.c +73 -30
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +169 -1
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +70 -9
- package/napi_api.c +219 -11
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +27 -11
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +8 -11
- package/src/engines/duckdb.js +45 -12
- package/src/index.js +661 -42
- package/src/memory_policy.js +411 -0
- package/src/native.js +1 -5
- package/src/pipeline.js +454 -31
- package/src/recipe_json.js +80 -0
- package/src/server.js +10 -8
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +6 -4
- package/wasm/index.js +498 -10
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/csrc/plan.c +0 -206
package/csrc/codec_text.c
CHANGED
|
@@ -15,6 +15,8 @@
|
|
|
15
15
|
#include <string.h>
|
|
16
16
|
|
|
17
17
|
#define DEFAULT_BATCH_SIZE 1024
|
|
18
|
+
#define DEFAULT_MAX_ERROR_BYTES 4096
|
|
19
|
+
#define DEFAULT_MAX_RECORD_BYTES (64 * 1024 * 1024)
|
|
18
20
|
|
|
19
21
|
/* ================================================================
|
|
20
22
|
* Text Decoder
|
|
@@ -22,6 +24,10 @@
|
|
|
22
24
|
|
|
23
25
|
typedef struct {
|
|
24
26
|
size_t batch_size;
|
|
27
|
+
size_t max_error_bytes;
|
|
28
|
+
size_t max_record_bytes;
|
|
29
|
+
size_t line_number;
|
|
30
|
+
size_t byte_offset;
|
|
25
31
|
tf_buffer line_buf;
|
|
26
32
|
tf_batch *batch;
|
|
27
33
|
size_t rows_buffered;
|
|
@@ -30,70 +36,262 @@ typedef struct {
|
|
|
30
36
|
static tf_batch *make_text_batch(size_t capacity) {
|
|
31
37
|
tf_batch *b = tf_batch_create(1, capacity);
|
|
32
38
|
if (!b) return NULL;
|
|
33
|
-
tf_batch_set_schema(b, 0, "_line", TF_TYPE_STRING)
|
|
39
|
+
if (tf_batch_set_schema(b, 0, "_line", TF_TYPE_STRING) != TF_OK) {
|
|
40
|
+
tf_batch_free(b);
|
|
41
|
+
return NULL;
|
|
42
|
+
}
|
|
34
43
|
return b;
|
|
35
44
|
}
|
|
36
45
|
|
|
46
|
+
static char *text_record_preview(text_decoder_state *st,
|
|
47
|
+
const uint8_t *prefix, size_t prefix_len,
|
|
48
|
+
const uint8_t *suffix, size_t suffix_len,
|
|
49
|
+
int *truncated_out) {
|
|
50
|
+
size_t total = prefix_len + suffix_len;
|
|
51
|
+
size_t keep = total;
|
|
52
|
+
int truncated = 0;
|
|
53
|
+
if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
|
|
54
|
+
keep = st->max_error_bytes;
|
|
55
|
+
truncated = 1;
|
|
56
|
+
}
|
|
57
|
+
char *raw = malloc(keep + 1);
|
|
58
|
+
if (!raw) return NULL;
|
|
59
|
+
size_t copied = 0;
|
|
60
|
+
size_t n = prefix_len < keep ? prefix_len : keep;
|
|
61
|
+
if (n > 0 && prefix) {
|
|
62
|
+
memcpy(raw, prefix, n);
|
|
63
|
+
copied += n;
|
|
64
|
+
}
|
|
65
|
+
if (copied < keep && suffix) {
|
|
66
|
+
size_t m = keep - copied;
|
|
67
|
+
if (m > suffix_len) m = suffix_len;
|
|
68
|
+
if (m > 0) {
|
|
69
|
+
memcpy(raw + copied, suffix, m);
|
|
70
|
+
copied += m;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
raw[copied] = '\0';
|
|
74
|
+
if (truncated_out) *truncated_out = truncated;
|
|
75
|
+
return raw;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
static int emit_text_record_size_diagnostic(text_decoder_state *st,
|
|
79
|
+
const uint8_t *prefix, size_t prefix_len,
|
|
80
|
+
const uint8_t *suffix, size_t suffix_len,
|
|
81
|
+
size_t line_no, size_t byte_offset,
|
|
82
|
+
size_t observed_len,
|
|
83
|
+
tf_side_channels *side) {
|
|
84
|
+
if (!side || !side->errors) return TF_OK;
|
|
85
|
+
int truncated = 0;
|
|
86
|
+
char *raw = text_record_preview(st, prefix, prefix_len, suffix, suffix_len, &truncated);
|
|
87
|
+
if (!raw) return TF_ERROR;
|
|
88
|
+
|
|
89
|
+
cJSON *obj = cJSON_CreateObject();
|
|
90
|
+
if (!obj) { free(raw); return TF_ERROR; }
|
|
91
|
+
int rc = TF_ERROR;
|
|
92
|
+
if (tf_json_add_string(obj, "type", "text_record_too_large") != TF_OK ||
|
|
93
|
+
tf_json_add_string(obj, "op", "codec.text.decode") != TF_OK ||
|
|
94
|
+
tf_json_add_string(obj, "action", "fail") != TF_OK ||
|
|
95
|
+
tf_json_add_string(obj, "severity", "error") != TF_OK ||
|
|
96
|
+
tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
|
|
97
|
+
tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
|
|
98
|
+
tf_json_add_number(obj, "max_record_bytes", (double)st->max_record_bytes) != TF_OK ||
|
|
99
|
+
tf_json_add_number(obj, "observed_bytes", (double)observed_len) != TF_OK ||
|
|
100
|
+
tf_json_add_string(obj, "message", "Text record exceeds max_record_bytes") != TF_OK ||
|
|
101
|
+
tf_json_add_number(obj, "raw_bytes", (double)observed_len) != TF_OK ||
|
|
102
|
+
tf_json_add_string(obj, "raw", raw) != TF_OK ||
|
|
103
|
+
(truncated && tf_json_add_bool(obj, "truncated", 1) != TF_OK)) {
|
|
104
|
+
goto done;
|
|
105
|
+
}
|
|
106
|
+
rc = tf_buffer_write_json_line(side->errors, obj);
|
|
107
|
+
done:
|
|
108
|
+
cJSON_Delete(obj);
|
|
109
|
+
free(raw);
|
|
110
|
+
return rc;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
static int fail_text_record_limit(text_decoder_state *st,
|
|
114
|
+
const uint8_t *prefix, size_t prefix_len,
|
|
115
|
+
const uint8_t *suffix, size_t suffix_len,
|
|
116
|
+
size_t line_no, size_t byte_offset,
|
|
117
|
+
size_t observed_len,
|
|
118
|
+
tf_side_channels *side) {
|
|
119
|
+
if (emit_text_record_size_diagnostic(st, prefix, prefix_len, suffix, suffix_len,
|
|
120
|
+
line_no, byte_offset, observed_len, side) != TF_OK)
|
|
121
|
+
return TF_ERROR;
|
|
122
|
+
char err[256];
|
|
123
|
+
snprintf(err, sizeof(err),
|
|
124
|
+
"text record exceeds max_record_bytes at line %zu: max %zu bytes, observed %zu bytes",
|
|
125
|
+
line_no, st->max_record_bytes, observed_len);
|
|
126
|
+
tf_set_last_error(err);
|
|
127
|
+
return TF_ERROR;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
static int check_text_buffer_record_limit(text_decoder_state *st, size_t observed_len,
|
|
131
|
+
size_t line_no, size_t byte_offset,
|
|
132
|
+
tf_side_channels *side) {
|
|
133
|
+
if (st->max_record_bytes == 0 || observed_len <= st->max_record_bytes) return TF_OK;
|
|
134
|
+
const uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
|
|
135
|
+
return fail_text_record_limit(st, buf, observed_len, NULL, 0,
|
|
136
|
+
line_no, byte_offset, observed_len, side);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
static int text_set_line_cell(tf_batch *batch, size_t row,
|
|
140
|
+
const uint8_t *data, size_t len) {
|
|
141
|
+
return tf_batch_set_string_len(batch, row, 0, (const char *)data, len);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
static int text_append_output_batch(tf_batch ***out, size_t *n_out,
|
|
145
|
+
size_t *out_cap, tf_batch *batch) {
|
|
146
|
+
if (*n_out >= *out_cap) {
|
|
147
|
+
size_t need = 0;
|
|
148
|
+
size_t new_cap = 0;
|
|
149
|
+
if (tf_size_add(*n_out, 1, &need) != TF_OK ||
|
|
150
|
+
tf_size_grow_pow2(*out_cap, need, 1, &new_cap) != TF_OK) {
|
|
151
|
+
return TF_ERROR;
|
|
152
|
+
}
|
|
153
|
+
tf_batch **new_out = tf_reallocarray_checked(*out, new_cap, sizeof(tf_batch *));
|
|
154
|
+
if (!new_out) return TF_ERROR;
|
|
155
|
+
*out = new_out;
|
|
156
|
+
*out_cap = new_cap;
|
|
157
|
+
}
|
|
158
|
+
(*out)[(*n_out)++] = batch;
|
|
159
|
+
return TF_OK;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
static int text_check_segment_record_limit(text_decoder_state *st,
|
|
163
|
+
const uint8_t *segment, size_t segment_len,
|
|
164
|
+
tf_side_channels *side) {
|
|
165
|
+
if (st->max_record_bytes == 0) return TF_OK;
|
|
166
|
+
|
|
167
|
+
size_t prefix_len = tf_buffer_readable(&st->line_buf);
|
|
168
|
+
size_t total = 0;
|
|
169
|
+
if (tf_size_add(prefix_len, segment_len, &total) == TF_OK &&
|
|
170
|
+
total <= st->max_record_bytes) {
|
|
171
|
+
return TF_OK;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
size_t suffix_len = segment_len;
|
|
175
|
+
size_t observed_len = total;
|
|
176
|
+
if (prefix_len >= st->max_record_bytes) {
|
|
177
|
+
suffix_len = segment_len > 0 ? 1 : 0;
|
|
178
|
+
if (tf_size_add(prefix_len, suffix_len, &observed_len) != TF_OK)
|
|
179
|
+
observed_len = st->max_record_bytes + 1;
|
|
180
|
+
} else {
|
|
181
|
+
size_t allowed = st->max_record_bytes - prefix_len;
|
|
182
|
+
suffix_len = allowed + 1;
|
|
183
|
+
if (suffix_len > segment_len) suffix_len = segment_len;
|
|
184
|
+
observed_len = prefix_len + suffix_len;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
const uint8_t *prefix = prefix_len > 0
|
|
188
|
+
? st->line_buf.data + st->line_buf.read_pos
|
|
189
|
+
: NULL;
|
|
190
|
+
return fail_text_record_limit(st, prefix, prefix_len, segment, suffix_len,
|
|
191
|
+
st->line_number + 1, st->byte_offset,
|
|
192
|
+
observed_len, side);
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
static int text_emit_line(text_decoder_state *st, const uint8_t *data, size_t len,
|
|
196
|
+
tf_batch ***out, size_t *n_out, size_t *out_cap) {
|
|
197
|
+
if (!st->batch) {
|
|
198
|
+
st->batch = make_text_batch(st->batch_size);
|
|
199
|
+
if (!st->batch) return TF_ERROR;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
size_t row = st->batch->n_rows;
|
|
203
|
+
size_t need_rows = 0;
|
|
204
|
+
if (tf_size_add(row, 1, &need_rows) != TF_OK ||
|
|
205
|
+
tf_batch_ensure_capacity(st->batch, need_rows) != TF_OK)
|
|
206
|
+
return TF_ERROR;
|
|
207
|
+
|
|
208
|
+
if (text_set_line_cell(st->batch, row, data, len) != TF_OK)
|
|
209
|
+
return TF_ERROR;
|
|
210
|
+
if (tf_batch_expose_row(st->batch, row) != TF_OK)
|
|
211
|
+
return TF_ERROR;
|
|
212
|
+
st->rows_buffered++;
|
|
213
|
+
|
|
214
|
+
if (st->rows_buffered >= st->batch_size) {
|
|
215
|
+
if (text_append_output_batch(out, n_out, out_cap, st->batch) != TF_OK)
|
|
216
|
+
return TF_ERROR;
|
|
217
|
+
st->batch = NULL;
|
|
218
|
+
st->rows_buffered = 0;
|
|
219
|
+
}
|
|
220
|
+
return TF_OK;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
static int text_emit_complete_segment(text_decoder_state *st,
|
|
224
|
+
const uint8_t *segment, size_t segment_len,
|
|
225
|
+
tf_batch ***out, size_t *n_out,
|
|
226
|
+
size_t *out_cap) {
|
|
227
|
+
size_t prefix_len = tf_buffer_readable(&st->line_buf);
|
|
228
|
+
const uint8_t *line = segment;
|
|
229
|
+
size_t line_len = segment_len;
|
|
230
|
+
|
|
231
|
+
if (prefix_len > 0) {
|
|
232
|
+
if (segment_len > 0 &&
|
|
233
|
+
tf_buffer_write(&st->line_buf, segment, segment_len) != TF_OK)
|
|
234
|
+
return TF_ERROR;
|
|
235
|
+
line = st->line_buf.data + st->line_buf.read_pos;
|
|
236
|
+
line_len = tf_buffer_readable(&st->line_buf);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
if (line_len > 0 && line[line_len - 1] == '\r')
|
|
240
|
+
line_len--;
|
|
241
|
+
|
|
242
|
+
if (text_emit_line(st, line, line_len, out, n_out, out_cap) != TF_OK)
|
|
243
|
+
return TF_ERROR;
|
|
244
|
+
|
|
245
|
+
if (prefix_len > 0) {
|
|
246
|
+
st->line_buf.read_pos = 0;
|
|
247
|
+
st->line_buf.len = 0;
|
|
248
|
+
}
|
|
249
|
+
return TF_OK;
|
|
250
|
+
}
|
|
251
|
+
|
|
37
252
|
static int text_decode(tf_decoder *self, const uint8_t *data, size_t len,
|
|
38
|
-
tf_batch ***out, size_t *n_out) {
|
|
253
|
+
tf_batch ***out, size_t *n_out, tf_side_channels *side) {
|
|
39
254
|
text_decoder_state *st = self->state;
|
|
40
255
|
*out = NULL;
|
|
41
256
|
*n_out = 0;
|
|
42
257
|
|
|
43
|
-
if (tf_buffer_write(&st->line_buf, data, len) != TF_OK) return TF_ERROR;
|
|
44
|
-
|
|
45
258
|
size_t out_cap = 0;
|
|
46
|
-
|
|
47
|
-
size_t buf_len = st->line_buf.len - st->line_buf.read_pos;
|
|
48
|
-
|
|
49
|
-
size_t line_start = 0;
|
|
50
|
-
for (size_t i = 0; i < buf_len; i++) {
|
|
51
|
-
if (buf[i] == '\n') {
|
|
52
|
-
size_t line_len = i - line_start;
|
|
53
|
-
/* Strip trailing \r for CRLF */
|
|
54
|
-
if (line_len > 0 && buf[line_start + line_len - 1] == '\r')
|
|
55
|
-
line_len--;
|
|
56
|
-
|
|
57
|
-
if (!st->batch) {
|
|
58
|
-
st->batch = make_text_batch(st->batch_size);
|
|
59
|
-
if (!st->batch) return TF_ERROR;
|
|
60
|
-
}
|
|
259
|
+
size_t pos = 0;
|
|
61
260
|
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
261
|
+
while (pos < len) {
|
|
262
|
+
const uint8_t *start = data + pos;
|
|
263
|
+
const uint8_t *nl = memchr(start, '\n', len - pos);
|
|
264
|
+
size_t segment_len = nl ? (size_t)(nl - start) : len - pos;
|
|
65
265
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
st->
|
|
86
|
-
|
|
87
|
-
|
|
266
|
+
if (text_check_segment_record_limit(st, start, segment_len, side) != TF_OK)
|
|
267
|
+
return TF_ERROR;
|
|
268
|
+
|
|
269
|
+
if (nl) {
|
|
270
|
+
size_t prefix_len = tf_buffer_readable(&st->line_buf);
|
|
271
|
+
if (text_emit_complete_segment(st, start, segment_len,
|
|
272
|
+
out, n_out, &out_cap) != TF_OK)
|
|
273
|
+
return TF_ERROR;
|
|
274
|
+
size_t advance = 0;
|
|
275
|
+
size_t new_offset = 0;
|
|
276
|
+
if (tf_size_add(prefix_len, segment_len, &advance) != TF_OK ||
|
|
277
|
+
tf_size_add(advance, 1, &advance) != TF_OK ||
|
|
278
|
+
tf_size_add(st->byte_offset, advance, &new_offset) != TF_OK)
|
|
279
|
+
return TF_ERROR;
|
|
280
|
+
st->line_number++;
|
|
281
|
+
st->byte_offset = new_offset;
|
|
282
|
+
pos += segment_len + 1;
|
|
283
|
+
} else {
|
|
284
|
+
if (segment_len > 0 &&
|
|
285
|
+
tf_buffer_write(&st->line_buf, start, segment_len) != TF_OK)
|
|
286
|
+
return TF_ERROR;
|
|
287
|
+
pos = len;
|
|
88
288
|
}
|
|
89
289
|
}
|
|
90
|
-
|
|
91
|
-
st->line_buf.read_pos += line_start;
|
|
92
|
-
tf_buffer_compact(&st->line_buf);
|
|
93
290
|
return TF_OK;
|
|
94
291
|
}
|
|
95
292
|
|
|
96
|
-
static int text_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
|
|
293
|
+
static int text_flush(tf_decoder *self, tf_batch ***out, size_t *n_out, tf_side_channels *side) {
|
|
294
|
+
(void)side;
|
|
97
295
|
text_decoder_state *st = self->state;
|
|
98
296
|
*out = NULL;
|
|
99
297
|
*n_out = 0;
|
|
@@ -103,6 +301,11 @@ static int text_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
|
|
|
103
301
|
size_t remaining = tf_buffer_readable(&st->line_buf);
|
|
104
302
|
if (remaining > 0) {
|
|
105
303
|
uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
|
|
304
|
+
size_t line_no = st->line_number + 1;
|
|
305
|
+
size_t record_offset = st->byte_offset;
|
|
306
|
+
if (check_text_buffer_record_limit(st, remaining, line_no, record_offset, side) != TF_OK)
|
|
307
|
+
return TF_ERROR;
|
|
308
|
+
st->line_number = line_no;
|
|
106
309
|
|
|
107
310
|
if (!st->batch) {
|
|
108
311
|
st->batch = make_text_batch(st->batch_size);
|
|
@@ -110,7 +313,9 @@ static int text_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
|
|
|
110
313
|
}
|
|
111
314
|
|
|
112
315
|
size_t row = st->batch->n_rows;
|
|
113
|
-
|
|
316
|
+
size_t need_rows = 0;
|
|
317
|
+
if (tf_size_add(row, 1, &need_rows) != TF_OK ||
|
|
318
|
+
tf_batch_ensure_capacity(st->batch, need_rows) != TF_OK)
|
|
114
319
|
return TF_ERROR;
|
|
115
320
|
|
|
116
321
|
/* Strip trailing \r */
|
|
@@ -118,25 +323,23 @@ static int text_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
|
|
|
118
323
|
if (line_len > 0 && buf[line_len - 1] == '\r')
|
|
119
324
|
line_len--;
|
|
120
325
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
tf_batch_set_string(st->batch, row, 0, line_str);
|
|
126
|
-
free(line_str);
|
|
127
|
-
st->batch->n_rows = row + 1;
|
|
326
|
+
if (text_set_line_cell(st->batch, row, buf, line_len) != TF_OK)
|
|
327
|
+
return TF_ERROR;
|
|
328
|
+
if (tf_batch_expose_row(st->batch, row) != TF_OK)
|
|
329
|
+
return TF_ERROR;
|
|
128
330
|
st->rows_buffered++;
|
|
129
331
|
|
|
130
332
|
st->line_buf.read_pos = st->line_buf.len;
|
|
333
|
+
size_t new_offset = 0;
|
|
334
|
+
if (tf_size_add(st->byte_offset, remaining, &new_offset) != TF_OK)
|
|
335
|
+
return TF_ERROR;
|
|
336
|
+
st->byte_offset = new_offset;
|
|
131
337
|
}
|
|
132
338
|
|
|
133
339
|
/* Emit remaining batch */
|
|
134
340
|
if (st->batch && st->rows_buffered > 0) {
|
|
135
|
-
if (
|
|
136
|
-
|
|
137
|
-
*out = realloc(*out, out_cap * sizeof(tf_batch *));
|
|
138
|
-
}
|
|
139
|
-
(*out)[(*n_out)++] = st->batch;
|
|
341
|
+
if (text_append_output_batch(out, n_out, &out_cap, st->batch) != TF_OK)
|
|
342
|
+
return TF_ERROR;
|
|
140
343
|
st->batch = NULL;
|
|
141
344
|
st->rows_buffered = 0;
|
|
142
345
|
}
|
|
@@ -159,10 +362,19 @@ tf_decoder *tf_text_decoder_create(const cJSON *args) {
|
|
|
159
362
|
if (!st) return NULL;
|
|
160
363
|
|
|
161
364
|
st->batch_size = DEFAULT_BATCH_SIZE;
|
|
365
|
+
st->max_error_bytes = DEFAULT_MAX_ERROR_BYTES;
|
|
366
|
+
st->max_record_bytes = DEFAULT_MAX_RECORD_BYTES;
|
|
162
367
|
if (args) {
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
368
|
+
size_t parsed_size = 0;
|
|
369
|
+
int has_batch_size = tf_json_get_size_arg(args, "batch_size", 1, TF_MAX_BATCH_ROWS, &parsed_size, "text");
|
|
370
|
+
if (has_batch_size < 0) { free(st); return NULL; }
|
|
371
|
+
if (has_batch_size > 0) st->batch_size = parsed_size;
|
|
372
|
+
int has_max_error = tf_json_get_size_arg(args, "max_error_bytes", 0, TF_MAX_ERROR_BYTES, &parsed_size, "text");
|
|
373
|
+
if (has_max_error < 0) { free(st); return NULL; }
|
|
374
|
+
if (has_max_error > 0) st->max_error_bytes = parsed_size;
|
|
375
|
+
int has_max_record = tf_json_get_size_arg(args, "max_record_bytes", 0, TF_MAX_RECORD_BYTES, &parsed_size, "text");
|
|
376
|
+
if (has_max_record < 0) { free(st); return NULL; }
|
|
377
|
+
if (has_max_record > 0) st->max_record_bytes = parsed_size;
|
|
166
378
|
}
|
|
167
379
|
|
|
168
380
|
tf_buffer_init(&st->line_buf);
|
|
@@ -180,6 +392,11 @@ tf_decoder *tf_text_decoder_create(const cJSON *args) {
|
|
|
180
392
|
* Text Encoder
|
|
181
393
|
* ================================================================ */
|
|
182
394
|
|
|
395
|
+
static int text_write_string(tf_buffer *out, const char *s) {
|
|
396
|
+
const char *value = s ? s : "";
|
|
397
|
+
return tf_buffer_write(out, (const uint8_t *)value, strlen(value));
|
|
398
|
+
}
|
|
399
|
+
|
|
183
400
|
static int text_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
|
|
184
401
|
(void)self;
|
|
185
402
|
|
|
@@ -191,19 +408,20 @@ static int text_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
|
|
|
191
408
|
/* Write _line column value */
|
|
192
409
|
if (!tf_batch_is_null(in, r, (size_t)line_col)) {
|
|
193
410
|
const char *s = tf_batch_get_string(in, r, (size_t)line_col);
|
|
194
|
-
|
|
411
|
+
if (text_write_string(out, s) != TF_OK) return TF_ERROR;
|
|
195
412
|
}
|
|
196
413
|
} else {
|
|
197
414
|
/* Fallback: concatenate all string columns with tab */
|
|
198
415
|
for (size_t c = 0; c < in->n_cols; c++) {
|
|
199
|
-
if (c > 0
|
|
416
|
+
if (c > 0 && tf_buffer_write(out, (const uint8_t *)"\t", 1) != TF_OK)
|
|
417
|
+
return TF_ERROR;
|
|
200
418
|
if (!tf_batch_is_null(in, r, c) && in->col_types[c] == TF_TYPE_STRING) {
|
|
201
419
|
const char *s = tf_batch_get_string(in, r, c);
|
|
202
|
-
|
|
420
|
+
if (text_write_string(out, s) != TF_OK) return TF_ERROR;
|
|
203
421
|
}
|
|
204
422
|
}
|
|
205
423
|
}
|
|
206
|
-
tf_buffer_write(out, (const uint8_t *)"\n", 1);
|
|
424
|
+
if (tf_buffer_write(out, (const uint8_t *)"\n", 1) != TF_OK) return TF_ERROR;
|
|
207
425
|
}
|
|
208
426
|
return TF_OK;
|
|
209
427
|
}
|
package/csrc/compiler.c
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Iterates IR nodes, looks up each op in the registry,
|
|
5
5
|
* and calls create_native() to build live decoder/steps/encoder structs.
|
|
6
|
-
*
|
|
6
|
+
* This is the single native construction path for JSON IR plans.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
#include "ir.h"
|
|
@@ -31,6 +31,11 @@ int tf_compile_native(const tf_ir_plan *plan,
|
|
|
31
31
|
*out_encoder = NULL;
|
|
32
32
|
if (error) *error = NULL;
|
|
33
33
|
|
|
34
|
+
if (!plan || plan->n_nodes == 0) {
|
|
35
|
+
set_error(error, "empty plan");
|
|
36
|
+
return TF_ERROR;
|
|
37
|
+
}
|
|
38
|
+
|
|
34
39
|
/* Allocate steps array (max = n_nodes, since some are decoder/encoder) */
|
|
35
40
|
tf_step **steps = calloc(plan->n_nodes, sizeof(tf_step *));
|
|
36
41
|
if (!steps) {
|
|
@@ -66,17 +71,31 @@ int tf_compile_native(const tf_ir_plan *plan,
|
|
|
66
71
|
|
|
67
72
|
void *obj = entry->create_native(node->args);
|
|
68
73
|
if (!obj) {
|
|
69
|
-
char
|
|
70
|
-
|
|
74
|
+
const char *detail = tf_last_error();
|
|
75
|
+
char buf[512];
|
|
76
|
+
if (detail && detail[0])
|
|
77
|
+
snprintf(buf, sizeof(buf), "failed to create '%s': %s", node->op, detail);
|
|
78
|
+
else
|
|
79
|
+
snprintf(buf, sizeof(buf), "failed to create '%s'", node->op);
|
|
71
80
|
set_error(error, buf);
|
|
72
81
|
goto fail;
|
|
73
82
|
}
|
|
74
83
|
|
|
75
84
|
switch (entry->kind) {
|
|
76
85
|
case TF_OP_DECODER:
|
|
86
|
+
if (decoder) {
|
|
87
|
+
((tf_decoder *)obj)->destroy((tf_decoder *)obj);
|
|
88
|
+
set_error(error, "multiple decoders");
|
|
89
|
+
goto fail;
|
|
90
|
+
}
|
|
77
91
|
decoder = (tf_decoder *)obj;
|
|
78
92
|
break;
|
|
79
93
|
case TF_OP_ENCODER:
|
|
94
|
+
if (encoder) {
|
|
95
|
+
((tf_encoder *)obj)->destroy((tf_encoder *)obj);
|
|
96
|
+
set_error(error, "multiple encoders");
|
|
97
|
+
goto fail;
|
|
98
|
+
}
|
|
80
99
|
encoder = (tf_encoder *)obj;
|
|
81
100
|
break;
|
|
82
101
|
case TF_OP_TRANSFORM:
|
|
@@ -85,6 +104,15 @@ int tf_compile_native(const tf_ir_plan *plan,
|
|
|
85
104
|
}
|
|
86
105
|
}
|
|
87
106
|
|
|
107
|
+
if (!decoder) {
|
|
108
|
+
set_error(error, "plan missing decoder");
|
|
109
|
+
goto fail;
|
|
110
|
+
}
|
|
111
|
+
if (!encoder) {
|
|
112
|
+
set_error(error, "plan missing encoder");
|
|
113
|
+
goto fail;
|
|
114
|
+
}
|
|
115
|
+
|
|
88
116
|
*out_decoder = decoder;
|
|
89
117
|
*out_steps = steps;
|
|
90
118
|
*out_n_steps = n_steps;
|
package/csrc/config.h
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* config.h -- Tranfi C feature-test configuration.
|
|
3
|
+
*
|
|
4
|
+
* Build files also define these macros at compile-command level so they are
|
|
5
|
+
* visible before any system header in every translation unit. This header
|
|
6
|
+
* documents the source contract and protects internal/public headers included
|
|
7
|
+
* first.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
#ifndef TF_CONFIG_H
|
|
11
|
+
#define TF_CONFIG_H
|
|
12
|
+
|
|
13
|
+
#ifndef _POSIX_C_SOURCE
|
|
14
|
+
#define _POSIX_C_SOURCE 200809L
|
|
15
|
+
#endif
|
|
16
|
+
|
|
17
|
+
#ifndef _XOPEN_SOURCE
|
|
18
|
+
#define _XOPEN_SOURCE 700
|
|
19
|
+
#endif
|
|
20
|
+
|
|
21
|
+
#endif /* TF_CONFIG_H */
|