tranfi 0.0.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +177 -21
- package/NOTICE +8 -0
- package/README.md +627 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-BIAIKnrp.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +121 -0
- package/csrc/arena.c +93 -0
- package/csrc/batch.c +976 -0
- package/csrc/buffer.c +154 -0
- package/csrc/cJSON.c +3386 -0
- package/csrc/cJSON.h +316 -0
- package/csrc/codec_csv.c +1951 -0
- package/csrc/codec_jsonl.c +1086 -0
- package/csrc/codec_table.c +248 -0
- package/csrc/codec_text.c +447 -0
- package/csrc/compiler.c +130 -0
- package/csrc/config.h +21 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +5417 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1553 -0
- package/csrc/expr.h +58 -0
- package/csrc/internal.h +539 -0
- package/csrc/ir.c +166 -0
- package/csrc/ir.h +208 -0
- package/csrc/ir_schema.c +75 -0
- package/csrc/ir_serialize.c +166 -0
- package/csrc/ir_sql.c +1822 -0
- package/csrc/ir_validate.c +576 -0
- package/csrc/json_path.c +210 -0
- package/csrc/main.c +1241 -0
- package/csrc/memory_estimate.c +477 -0
- package/csrc/op_acf.c +283 -0
- package/csrc/op_across.c +477 -0
- package/csrc/op_anomaly.c +255 -0
- package/csrc/op_assert.c +761 -0
- package/csrc/op_bin.c +248 -0
- package/csrc/op_cast.c +523 -0
- package/csrc/op_clip.c +99 -0
- package/csrc/op_date_trunc.c +355 -0
- package/csrc/op_datetime.c +394 -0
- package/csrc/op_derive.c +216 -0
- package/csrc/op_diff.c +250 -0
- package/csrc/op_ewma.c +222 -0
- package/csrc/op_explode.c +206 -0
- package/csrc/op_fill_down.c +235 -0
- package/csrc/op_fill_null.c +268 -0
- package/csrc/op_filter.c +181 -0
- package/csrc/op_frequency.c +721 -0
- package/csrc/op_grep.c +181 -0
- package/csrc/op_group_agg.c +1956 -0
- package/csrc/op_hash.c +159 -0
- package/csrc/op_head.c +84 -0
- package/csrc/op_interpolate.c +445 -0
- package/csrc/op_join.c +2902 -0
- package/csrc/op_json_extract.c +227 -0
- package/csrc/op_json_filter.c +384 -0
- package/csrc/op_json_flatten.c +293 -0
- package/csrc/op_json_schema.c +503 -0
- package/csrc/op_label_encode.c +419 -0
- package/csrc/op_lag.c +181 -0
- package/csrc/op_lead.c +242 -0
- package/csrc/op_normalize.c +510 -0
- package/csrc/op_onehot.c +457 -0
- package/csrc/op_pivot.c +1754 -0
- package/csrc/op_quarantine.c +189 -0
- package/csrc/op_registry.c +3044 -0
- package/csrc/op_rename.c +129 -0
- package/csrc/op_replace.c +354 -0
- package/csrc/op_rleid.c +297 -0
- package/csrc/op_rowid.c +559 -0
- package/csrc/op_sample.c +158 -0
- package/csrc/op_schema.c +1341 -0
- package/csrc/op_schema_infer.c +252 -0
- package/csrc/op_select.c +340 -0
- package/csrc/op_set.c +3449 -0
- package/csrc/op_skip.c +95 -0
- package/csrc/op_sort.c +819 -0
- package/csrc/op_source_name.c +120 -0
- package/csrc/op_split.c +151 -0
- package/csrc/op_split_data.c +119 -0
- package/csrc/op_stack.c +271 -0
- package/csrc/op_stats.c +875 -0
- package/csrc/op_step.c +333 -0
- package/csrc/op_tail.c +105 -0
- package/csrc/op_tee.c +338 -0
- package/csrc/op_top.c +357 -0
- package/csrc/op_trim.c +138 -0
- package/csrc/op_unique.c +1343 -0
- package/csrc/op_unpivot.c +193 -0
- package/csrc/op_validate.c +648 -0
- package/csrc/op_window.c +591 -0
- package/csrc/path_policy.c +85 -0
- package/csrc/pipeline.c +1088 -0
- package/csrc/recipes.c +104 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +506 -0
- package/csrc/report.h +22 -0
- package/csrc/selector.c +1097 -0
- package/csrc/size_utils.c +348 -0
- package/csrc/spill.c +317 -0
- package/csrc/spill.h +21 -0
- package/csrc/tranfi.h +291 -0
- package/csrc/transform.h +209 -0
- package/csrc/transform_api.c +2237 -0
- package/csrc/transform_categorical.c +923 -0
- package/csrc/transform_internal.h +472 -0
- package/csrc/transform_json.c +3812 -0
- package/csrc/transform_numeric.c +1966 -0
- package/csrc/transform_sha256.c +154 -0
- package/csrc/transform_wasm.h +162 -0
- package/csrc/transform_wasm_api.c +1373 -0
- package/csrc/wasm_api.c +218 -0
- package/napi_api.c +534 -0
- package/napi_transform.c +1648 -0
- package/napi_transform.h +8 -0
- package/package.json +64 -59
- package/scripts/install-native.js +76 -0
- package/scripts/prepack.js +64 -0
- package/scripts/sync-csrc.js +23 -0
- package/src/cli.js +190 -0
- package/src/engines/duckdb.js +142 -0
- package/src/index.js +925 -0
- package/src/memory_policy.js +411 -0
- package/src/native.js +18 -0
- package/src/pipeline.js +709 -0
- package/src/recipe_json.js +80 -0
- package/src/server.js +279 -0
- package/src/transform.js +403 -0
- package/src/transform_error.js +10 -0
- package/src/wasm.js +21 -0
- package/wasm/index.js +732 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/wasm/transform.js +1156 -0
- package/wasm/worker.js +786 -0
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* codec_table.c — Pretty-print Markdown-compatible table encoder.
|
|
3
|
+
*
|
|
4
|
+
* Buffers all rows to compute column widths, then on flush emits
|
|
5
|
+
* a formatted table:
|
|
6
|
+
*
|
|
7
|
+
* | name | age | city |
|
|
8
|
+
* | ------- | --- | ---- |
|
|
9
|
+
* | Alice | 30 | NY |
|
|
10
|
+
* | Bob | 25 | LA |
|
|
11
|
+
*
|
|
12
|
+
* Args:
|
|
13
|
+
* max_width (int, optional) — truncate columns wider than this (default: 40)
|
|
14
|
+
* max_rows (int, optional) — limit output rows (default: 0 = unlimited)
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
#include "internal.h"
|
|
18
|
+
#include "cJSON.h"
|
|
19
|
+
#include <stdlib.h>
|
|
20
|
+
#include <string.h>
|
|
21
|
+
|
|
22
|
+
#define TABLE_MAX_COLS 256
|
|
23
|
+
#define TABLE_DEFAULT_WIDTH 40
|
|
24
|
+
|
|
25
|
+
typedef struct {
|
|
26
|
+
char **values; /* flat array: rows × n_cols, each cell is a malloc'd string */
|
|
27
|
+
size_t n_rows;
|
|
28
|
+
size_t n_cols;
|
|
29
|
+
size_t capacity; /* allocated rows */
|
|
30
|
+
char **col_names;
|
|
31
|
+
size_t max_width;
|
|
32
|
+
size_t max_rows;
|
|
33
|
+
} table_encoder_state;
|
|
34
|
+
|
|
35
|
+
/* Format a cell value as a string. Caller must free. */
|
|
36
|
+
static char *cell_to_string(tf_batch *b, size_t row, size_t col) {
|
|
37
|
+
char buf[64];
|
|
38
|
+
const char *text = NULL;
|
|
39
|
+
if (tf_batch_format_cell_as_string(b, row, col, TF_CELL_STRING_HUMAN,
|
|
40
|
+
buf, sizeof(buf), &text) != TF_OK) {
|
|
41
|
+
return NULL;
|
|
42
|
+
}
|
|
43
|
+
return strdup(text ? text : "");
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
static int table_capture_schema(table_encoder_state *st, const tf_batch *in) {
|
|
47
|
+
size_t n_cols = in->n_cols < TABLE_MAX_COLS ? in->n_cols : TABLE_MAX_COLS;
|
|
48
|
+
if (n_cols == 0) return TF_OK;
|
|
49
|
+
char **names = tf_callocarray_checked(n_cols, sizeof(char *));
|
|
50
|
+
if (!names) return TF_ERROR;
|
|
51
|
+
for (size_t i = 0; i < n_cols; i++) {
|
|
52
|
+
names[i] = strdup(in->col_names[i] ? in->col_names[i] : "");
|
|
53
|
+
if (!names[i]) {
|
|
54
|
+
for (size_t j = 0; j < i; j++) free(names[j]);
|
|
55
|
+
free(names);
|
|
56
|
+
return TF_ERROR;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
st->col_names = names;
|
|
60
|
+
st->n_cols = n_cols;
|
|
61
|
+
return TF_OK;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
static int table_ensure_value_capacity(table_encoder_state *st, size_t min_rows) {
|
|
65
|
+
if (st->n_cols == 0 || min_rows <= st->capacity) return TF_OK;
|
|
66
|
+
size_t new_cap = 0;
|
|
67
|
+
size_t cells = 0;
|
|
68
|
+
if (tf_size_grow_pow2(st->capacity, min_rows, 64, &new_cap) != TF_OK ||
|
|
69
|
+
tf_size_mul(new_cap, st->n_cols, &cells) != TF_OK) {
|
|
70
|
+
return TF_ERROR;
|
|
71
|
+
}
|
|
72
|
+
char **new_values = tf_reallocarray_checked(st->values, cells, sizeof(char *));
|
|
73
|
+
if (!new_values) return TF_ERROR;
|
|
74
|
+
st->values = new_values;
|
|
75
|
+
st->capacity = new_cap;
|
|
76
|
+
return TF_OK;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
static int table_write_repeat(tf_buffer *out, char ch, size_t count) {
|
|
80
|
+
char pad[256];
|
|
81
|
+
memset(pad, ch, sizeof(pad));
|
|
82
|
+
while (count > 0) {
|
|
83
|
+
size_t chunk = count < sizeof(pad) ? count : sizeof(pad);
|
|
84
|
+
if (tf_buffer_write(out, (const uint8_t *)pad, chunk) != TF_OK) return TF_ERROR;
|
|
85
|
+
count -= chunk;
|
|
86
|
+
}
|
|
87
|
+
return TF_OK;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
static int table_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
|
|
91
|
+
table_encoder_state *st = self->state;
|
|
92
|
+
(void)out;
|
|
93
|
+
|
|
94
|
+
/* Capture column names on first batch */
|
|
95
|
+
if (!st->col_names && in->n_cols > 0) {
|
|
96
|
+
if (table_capture_schema(st, in) != TF_OK) return TF_ERROR;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/* Buffer all cell values as strings */
|
|
100
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
101
|
+
if (st->max_rows > 0 && st->n_rows >= st->max_rows) break;
|
|
102
|
+
size_t next_rows = 0;
|
|
103
|
+
if (tf_size_add(st->n_rows, 1, &next_rows) != TF_OK ||
|
|
104
|
+
table_ensure_value_capacity(st, next_rows) != TF_OK) {
|
|
105
|
+
return TF_ERROR;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
size_t base = 0;
|
|
109
|
+
if (tf_size_mul(st->n_rows, st->n_cols, &base) != TF_OK) return TF_ERROR;
|
|
110
|
+
size_t c = 0;
|
|
111
|
+
for (; c < st->n_cols; c++) {
|
|
112
|
+
st->values[base + c] = cell_to_string(in, r, c);
|
|
113
|
+
if (!st->values[base + c]) {
|
|
114
|
+
for (size_t j = 0; j < c; j++) {
|
|
115
|
+
free(st->values[base + j]);
|
|
116
|
+
st->values[base + j] = NULL;
|
|
117
|
+
}
|
|
118
|
+
return TF_ERROR;
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
st->n_rows = next_rows;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
return TF_OK;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
static int table_flush(tf_encoder *self, tf_buffer *out) {
|
|
128
|
+
table_encoder_state *st = self->state;
|
|
129
|
+
if (!st->col_names || st->n_cols == 0) return TF_OK;
|
|
130
|
+
|
|
131
|
+
/* Compute column widths */
|
|
132
|
+
size_t *widths = tf_callocarray_checked(st->n_cols, sizeof(size_t));
|
|
133
|
+
if (!widths) return TF_ERROR;
|
|
134
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
135
|
+
widths[c] = strlen(st->col_names[c]);
|
|
136
|
+
}
|
|
137
|
+
for (size_t r = 0; r < st->n_rows; r++) {
|
|
138
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
139
|
+
size_t len = strlen(st->values[r * st->n_cols + c]);
|
|
140
|
+
if (len > widths[c]) widths[c] = len;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
/* Apply max_width */
|
|
144
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
145
|
+
if (st->max_width > 0 && widths[c] > st->max_width)
|
|
146
|
+
widths[c] = st->max_width;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
#define TABLE_WRITE(data, len) \
|
|
150
|
+
do { \
|
|
151
|
+
if (tf_buffer_write(out, (const uint8_t *)(data), (len)) != TF_OK) goto fail; \
|
|
152
|
+
} while (0)
|
|
153
|
+
|
|
154
|
+
/* Header row */
|
|
155
|
+
TABLE_WRITE("| ", 2);
|
|
156
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
157
|
+
if (c > 0) TABLE_WRITE(" | ", 3);
|
|
158
|
+
const char *name = st->col_names[c];
|
|
159
|
+
size_t nlen = strlen(name);
|
|
160
|
+
size_t w = widths[c];
|
|
161
|
+
if (nlen > w) nlen = w;
|
|
162
|
+
TABLE_WRITE(name, nlen);
|
|
163
|
+
if (nlen < w && table_write_repeat(out, ' ', w - nlen) != TF_OK) goto fail;
|
|
164
|
+
}
|
|
165
|
+
TABLE_WRITE(" |\n", 3);
|
|
166
|
+
|
|
167
|
+
/* Separator row */
|
|
168
|
+
TABLE_WRITE("| ", 2);
|
|
169
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
170
|
+
if (c > 0) TABLE_WRITE(" | ", 3);
|
|
171
|
+
if (table_write_repeat(out, '-', widths[c]) != TF_OK) goto fail;
|
|
172
|
+
}
|
|
173
|
+
TABLE_WRITE(" |\n", 3);
|
|
174
|
+
|
|
175
|
+
/* Data rows */
|
|
176
|
+
for (size_t r = 0; r < st->n_rows; r++) {
|
|
177
|
+
TABLE_WRITE("| ", 2);
|
|
178
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
179
|
+
if (c > 0) TABLE_WRITE(" | ", 3);
|
|
180
|
+
const char *val = st->values[r * st->n_cols + c];
|
|
181
|
+
size_t vlen = strlen(val);
|
|
182
|
+
size_t w = widths[c];
|
|
183
|
+
if (vlen > w) vlen = w;
|
|
184
|
+
TABLE_WRITE(val, vlen);
|
|
185
|
+
if (vlen < w && table_write_repeat(out, ' ', w - vlen) != TF_OK) goto fail;
|
|
186
|
+
}
|
|
187
|
+
TABLE_WRITE(" |\n", 3);
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
#undef TABLE_WRITE
|
|
191
|
+
free(widths);
|
|
192
|
+
return TF_OK;
|
|
193
|
+
|
|
194
|
+
fail:
|
|
195
|
+
#undef TABLE_WRITE
|
|
196
|
+
free(widths);
|
|
197
|
+
return TF_ERROR;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
static void table_encoder_destroy(tf_encoder *self) {
|
|
201
|
+
table_encoder_state *st = self->state;
|
|
202
|
+
if (st) {
|
|
203
|
+
if (st->col_names) {
|
|
204
|
+
for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
|
|
205
|
+
free(st->col_names);
|
|
206
|
+
}
|
|
207
|
+
if (st->values) {
|
|
208
|
+
size_t n_cells = 0;
|
|
209
|
+
if (tf_size_mul(st->n_rows, st->n_cols, &n_cells) == TF_OK) {
|
|
210
|
+
for (size_t i = 0; i < n_cells; i++) free(st->values[i]);
|
|
211
|
+
}
|
|
212
|
+
free(st->values);
|
|
213
|
+
}
|
|
214
|
+
free(st);
|
|
215
|
+
}
|
|
216
|
+
free(self);
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
tf_encoder *tf_table_encoder_create(const cJSON *args) {
|
|
220
|
+
table_encoder_state *st = calloc(1, sizeof(table_encoder_state));
|
|
221
|
+
if (!st) return NULL;
|
|
222
|
+
|
|
223
|
+
st->max_width = TABLE_DEFAULT_WIDTH;
|
|
224
|
+
st->max_rows = 0;
|
|
225
|
+
|
|
226
|
+
if (args) {
|
|
227
|
+
size_t parsed_size = 0;
|
|
228
|
+
int has_max_width = tf_json_get_size_arg(args, "max_width",
|
|
229
|
+
1, TF_MAX_TABLE_WIDTH,
|
|
230
|
+
&parsed_size, "table");
|
|
231
|
+
if (has_max_width < 0) { free(st); return NULL; }
|
|
232
|
+
if (has_max_width > 0) st->max_width = parsed_size;
|
|
233
|
+
|
|
234
|
+
int has_max_rows = tf_json_get_size_arg(args, "max_rows",
|
|
235
|
+
0, TF_MAX_TABLE_ROWS,
|
|
236
|
+
&parsed_size, "table");
|
|
237
|
+
if (has_max_rows < 0) { free(st); return NULL; }
|
|
238
|
+
if (has_max_rows > 0) st->max_rows = parsed_size;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
tf_encoder *enc = malloc(sizeof(tf_encoder));
|
|
242
|
+
if (!enc) { free(st); return NULL; }
|
|
243
|
+
enc->encode = table_encode;
|
|
244
|
+
enc->flush = table_flush;
|
|
245
|
+
enc->destroy = table_encoder_destroy;
|
|
246
|
+
enc->state = st;
|
|
247
|
+
return enc;
|
|
248
|
+
}
|
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* codec_text.c — Plain text line codec (newline-split).
|
|
3
|
+
*
|
|
4
|
+
* Decoder: splits on newlines, each line → one row in single "_line" column.
|
|
5
|
+
* - No type detection, no field parsing — just memchr for newlines.
|
|
6
|
+
* - Emits batch at batch_size.
|
|
7
|
+
*
|
|
8
|
+
* Encoder: writes _line string + \n per row.
|
|
9
|
+
* - Fallback: if no _line column, concatenate all string columns with tab.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
#include "internal.h"
|
|
13
|
+
#include "cJSON.h"
|
|
14
|
+
#include <stdlib.h>
|
|
15
|
+
#include <string.h>
|
|
16
|
+
|
|
17
|
+
#define DEFAULT_BATCH_SIZE 1024
|
|
18
|
+
#define DEFAULT_MAX_ERROR_BYTES 4096
|
|
19
|
+
#define DEFAULT_MAX_RECORD_BYTES (64 * 1024 * 1024)
|
|
20
|
+
|
|
21
|
+
/* ================================================================
|
|
22
|
+
* Text Decoder
|
|
23
|
+
* ================================================================ */
|
|
24
|
+
|
|
25
|
+
typedef struct {
|
|
26
|
+
size_t batch_size;
|
|
27
|
+
size_t max_error_bytes;
|
|
28
|
+
size_t max_record_bytes;
|
|
29
|
+
size_t line_number;
|
|
30
|
+
size_t byte_offset;
|
|
31
|
+
tf_buffer line_buf;
|
|
32
|
+
tf_batch *batch;
|
|
33
|
+
size_t rows_buffered;
|
|
34
|
+
} text_decoder_state;
|
|
35
|
+
|
|
36
|
+
static tf_batch *make_text_batch(size_t capacity) {
|
|
37
|
+
tf_batch *b = tf_batch_create(1, capacity);
|
|
38
|
+
if (!b) return NULL;
|
|
39
|
+
if (tf_batch_set_schema(b, 0, "_line", TF_TYPE_STRING) != TF_OK) {
|
|
40
|
+
tf_batch_free(b);
|
|
41
|
+
return NULL;
|
|
42
|
+
}
|
|
43
|
+
return b;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
static char *text_record_preview(text_decoder_state *st,
|
|
47
|
+
const uint8_t *prefix, size_t prefix_len,
|
|
48
|
+
const uint8_t *suffix, size_t suffix_len,
|
|
49
|
+
int *truncated_out) {
|
|
50
|
+
size_t total = prefix_len + suffix_len;
|
|
51
|
+
size_t keep = total;
|
|
52
|
+
int truncated = 0;
|
|
53
|
+
if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
|
|
54
|
+
keep = st->max_error_bytes;
|
|
55
|
+
truncated = 1;
|
|
56
|
+
}
|
|
57
|
+
char *raw = malloc(keep + 1);
|
|
58
|
+
if (!raw) return NULL;
|
|
59
|
+
size_t copied = 0;
|
|
60
|
+
size_t n = prefix_len < keep ? prefix_len : keep;
|
|
61
|
+
if (n > 0 && prefix) {
|
|
62
|
+
memcpy(raw, prefix, n);
|
|
63
|
+
copied += n;
|
|
64
|
+
}
|
|
65
|
+
if (copied < keep && suffix) {
|
|
66
|
+
size_t m = keep - copied;
|
|
67
|
+
if (m > suffix_len) m = suffix_len;
|
|
68
|
+
if (m > 0) {
|
|
69
|
+
memcpy(raw + copied, suffix, m);
|
|
70
|
+
copied += m;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
raw[copied] = '\0';
|
|
74
|
+
if (truncated_out) *truncated_out = truncated;
|
|
75
|
+
return raw;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
static int emit_text_record_size_diagnostic(text_decoder_state *st,
|
|
79
|
+
const uint8_t *prefix, size_t prefix_len,
|
|
80
|
+
const uint8_t *suffix, size_t suffix_len,
|
|
81
|
+
size_t line_no, size_t byte_offset,
|
|
82
|
+
size_t observed_len,
|
|
83
|
+
tf_side_channels *side) {
|
|
84
|
+
if (!side || !side->errors) return TF_OK;
|
|
85
|
+
int truncated = 0;
|
|
86
|
+
char *raw = text_record_preview(st, prefix, prefix_len, suffix, suffix_len, &truncated);
|
|
87
|
+
if (!raw) return TF_ERROR;
|
|
88
|
+
|
|
89
|
+
cJSON *obj = cJSON_CreateObject();
|
|
90
|
+
if (!obj) { free(raw); return TF_ERROR; }
|
|
91
|
+
int rc = TF_ERROR;
|
|
92
|
+
if (tf_json_add_string(obj, "type", "text_record_too_large") != TF_OK ||
|
|
93
|
+
tf_json_add_string(obj, "op", "codec.text.decode") != TF_OK ||
|
|
94
|
+
tf_json_add_string(obj, "action", "fail") != TF_OK ||
|
|
95
|
+
tf_json_add_string(obj, "severity", "error") != TF_OK ||
|
|
96
|
+
tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
|
|
97
|
+
tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
|
|
98
|
+
tf_json_add_number(obj, "max_record_bytes", (double)st->max_record_bytes) != TF_OK ||
|
|
99
|
+
tf_json_add_number(obj, "observed_bytes", (double)observed_len) != TF_OK ||
|
|
100
|
+
tf_json_add_string(obj, "message", "Text record exceeds max_record_bytes") != TF_OK ||
|
|
101
|
+
tf_json_add_number(obj, "raw_bytes", (double)observed_len) != TF_OK ||
|
|
102
|
+
tf_json_add_string(obj, "raw", raw) != TF_OK ||
|
|
103
|
+
(truncated && tf_json_add_bool(obj, "truncated", 1) != TF_OK)) {
|
|
104
|
+
goto done;
|
|
105
|
+
}
|
|
106
|
+
rc = tf_buffer_write_json_line(side->errors, obj);
|
|
107
|
+
done:
|
|
108
|
+
cJSON_Delete(obj);
|
|
109
|
+
free(raw);
|
|
110
|
+
return rc;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
static int fail_text_record_limit(text_decoder_state *st,
|
|
114
|
+
const uint8_t *prefix, size_t prefix_len,
|
|
115
|
+
const uint8_t *suffix, size_t suffix_len,
|
|
116
|
+
size_t line_no, size_t byte_offset,
|
|
117
|
+
size_t observed_len,
|
|
118
|
+
tf_side_channels *side) {
|
|
119
|
+
if (emit_text_record_size_diagnostic(st, prefix, prefix_len, suffix, suffix_len,
|
|
120
|
+
line_no, byte_offset, observed_len, side) != TF_OK)
|
|
121
|
+
return TF_ERROR;
|
|
122
|
+
char err[256];
|
|
123
|
+
snprintf(err, sizeof(err),
|
|
124
|
+
"text record exceeds max_record_bytes at line %zu: max %zu bytes, observed %zu bytes",
|
|
125
|
+
line_no, st->max_record_bytes, observed_len);
|
|
126
|
+
tf_set_last_error(err);
|
|
127
|
+
return TF_ERROR;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
static int check_text_buffer_record_limit(text_decoder_state *st, size_t observed_len,
|
|
131
|
+
size_t line_no, size_t byte_offset,
|
|
132
|
+
tf_side_channels *side) {
|
|
133
|
+
if (st->max_record_bytes == 0 || observed_len <= st->max_record_bytes) return TF_OK;
|
|
134
|
+
const uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
|
|
135
|
+
return fail_text_record_limit(st, buf, observed_len, NULL, 0,
|
|
136
|
+
line_no, byte_offset, observed_len, side);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
static int text_set_line_cell(tf_batch *batch, size_t row,
|
|
140
|
+
const uint8_t *data, size_t len) {
|
|
141
|
+
return tf_batch_set_string_len(batch, row, 0, (const char *)data, len);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
static int text_append_output_batch(tf_batch ***out, size_t *n_out,
|
|
145
|
+
size_t *out_cap, tf_batch *batch) {
|
|
146
|
+
if (*n_out >= *out_cap) {
|
|
147
|
+
size_t need = 0;
|
|
148
|
+
size_t new_cap = 0;
|
|
149
|
+
if (tf_size_add(*n_out, 1, &need) != TF_OK ||
|
|
150
|
+
tf_size_grow_pow2(*out_cap, need, 1, &new_cap) != TF_OK) {
|
|
151
|
+
return TF_ERROR;
|
|
152
|
+
}
|
|
153
|
+
tf_batch **new_out = tf_reallocarray_checked(*out, new_cap, sizeof(tf_batch *));
|
|
154
|
+
if (!new_out) return TF_ERROR;
|
|
155
|
+
*out = new_out;
|
|
156
|
+
*out_cap = new_cap;
|
|
157
|
+
}
|
|
158
|
+
(*out)[(*n_out)++] = batch;
|
|
159
|
+
return TF_OK;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
static int text_check_segment_record_limit(text_decoder_state *st,
|
|
163
|
+
const uint8_t *segment, size_t segment_len,
|
|
164
|
+
tf_side_channels *side) {
|
|
165
|
+
if (st->max_record_bytes == 0) return TF_OK;
|
|
166
|
+
|
|
167
|
+
size_t prefix_len = tf_buffer_readable(&st->line_buf);
|
|
168
|
+
size_t total = 0;
|
|
169
|
+
if (tf_size_add(prefix_len, segment_len, &total) == TF_OK &&
|
|
170
|
+
total <= st->max_record_bytes) {
|
|
171
|
+
return TF_OK;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
size_t suffix_len = segment_len;
|
|
175
|
+
size_t observed_len = total;
|
|
176
|
+
if (prefix_len >= st->max_record_bytes) {
|
|
177
|
+
suffix_len = segment_len > 0 ? 1 : 0;
|
|
178
|
+
if (tf_size_add(prefix_len, suffix_len, &observed_len) != TF_OK)
|
|
179
|
+
observed_len = st->max_record_bytes + 1;
|
|
180
|
+
} else {
|
|
181
|
+
size_t allowed = st->max_record_bytes - prefix_len;
|
|
182
|
+
suffix_len = allowed + 1;
|
|
183
|
+
if (suffix_len > segment_len) suffix_len = segment_len;
|
|
184
|
+
observed_len = prefix_len + suffix_len;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
const uint8_t *prefix = prefix_len > 0
|
|
188
|
+
? st->line_buf.data + st->line_buf.read_pos
|
|
189
|
+
: NULL;
|
|
190
|
+
return fail_text_record_limit(st, prefix, prefix_len, segment, suffix_len,
|
|
191
|
+
st->line_number + 1, st->byte_offset,
|
|
192
|
+
observed_len, side);
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
static int text_emit_line(text_decoder_state *st, const uint8_t *data, size_t len,
|
|
196
|
+
tf_batch ***out, size_t *n_out, size_t *out_cap) {
|
|
197
|
+
if (!st->batch) {
|
|
198
|
+
st->batch = make_text_batch(st->batch_size);
|
|
199
|
+
if (!st->batch) return TF_ERROR;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
size_t row = st->batch->n_rows;
|
|
203
|
+
size_t need_rows = 0;
|
|
204
|
+
if (tf_size_add(row, 1, &need_rows) != TF_OK ||
|
|
205
|
+
tf_batch_ensure_capacity(st->batch, need_rows) != TF_OK)
|
|
206
|
+
return TF_ERROR;
|
|
207
|
+
|
|
208
|
+
if (text_set_line_cell(st->batch, row, data, len) != TF_OK)
|
|
209
|
+
return TF_ERROR;
|
|
210
|
+
if (tf_batch_expose_row(st->batch, row) != TF_OK)
|
|
211
|
+
return TF_ERROR;
|
|
212
|
+
st->rows_buffered++;
|
|
213
|
+
|
|
214
|
+
if (st->rows_buffered >= st->batch_size) {
|
|
215
|
+
if (text_append_output_batch(out, n_out, out_cap, st->batch) != TF_OK)
|
|
216
|
+
return TF_ERROR;
|
|
217
|
+
st->batch = NULL;
|
|
218
|
+
st->rows_buffered = 0;
|
|
219
|
+
}
|
|
220
|
+
return TF_OK;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
static int text_emit_complete_segment(text_decoder_state *st,
|
|
224
|
+
const uint8_t *segment, size_t segment_len,
|
|
225
|
+
tf_batch ***out, size_t *n_out,
|
|
226
|
+
size_t *out_cap) {
|
|
227
|
+
size_t prefix_len = tf_buffer_readable(&st->line_buf);
|
|
228
|
+
const uint8_t *line = segment;
|
|
229
|
+
size_t line_len = segment_len;
|
|
230
|
+
|
|
231
|
+
if (prefix_len > 0) {
|
|
232
|
+
if (segment_len > 0 &&
|
|
233
|
+
tf_buffer_write(&st->line_buf, segment, segment_len) != TF_OK)
|
|
234
|
+
return TF_ERROR;
|
|
235
|
+
line = st->line_buf.data + st->line_buf.read_pos;
|
|
236
|
+
line_len = tf_buffer_readable(&st->line_buf);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
if (line_len > 0 && line[line_len - 1] == '\r')
|
|
240
|
+
line_len--;
|
|
241
|
+
|
|
242
|
+
if (text_emit_line(st, line, line_len, out, n_out, out_cap) != TF_OK)
|
|
243
|
+
return TF_ERROR;
|
|
244
|
+
|
|
245
|
+
if (prefix_len > 0) {
|
|
246
|
+
st->line_buf.read_pos = 0;
|
|
247
|
+
st->line_buf.len = 0;
|
|
248
|
+
}
|
|
249
|
+
return TF_OK;
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
static int text_decode(tf_decoder *self, const uint8_t *data, size_t len,
|
|
253
|
+
tf_batch ***out, size_t *n_out, tf_side_channels *side) {
|
|
254
|
+
text_decoder_state *st = self->state;
|
|
255
|
+
*out = NULL;
|
|
256
|
+
*n_out = 0;
|
|
257
|
+
|
|
258
|
+
size_t out_cap = 0;
|
|
259
|
+
size_t pos = 0;
|
|
260
|
+
|
|
261
|
+
while (pos < len) {
|
|
262
|
+
const uint8_t *start = data + pos;
|
|
263
|
+
const uint8_t *nl = memchr(start, '\n', len - pos);
|
|
264
|
+
size_t segment_len = nl ? (size_t)(nl - start) : len - pos;
|
|
265
|
+
|
|
266
|
+
if (text_check_segment_record_limit(st, start, segment_len, side) != TF_OK)
|
|
267
|
+
return TF_ERROR;
|
|
268
|
+
|
|
269
|
+
if (nl) {
|
|
270
|
+
size_t prefix_len = tf_buffer_readable(&st->line_buf);
|
|
271
|
+
if (text_emit_complete_segment(st, start, segment_len,
|
|
272
|
+
out, n_out, &out_cap) != TF_OK)
|
|
273
|
+
return TF_ERROR;
|
|
274
|
+
size_t advance = 0;
|
|
275
|
+
size_t new_offset = 0;
|
|
276
|
+
if (tf_size_add(prefix_len, segment_len, &advance) != TF_OK ||
|
|
277
|
+
tf_size_add(advance, 1, &advance) != TF_OK ||
|
|
278
|
+
tf_size_add(st->byte_offset, advance, &new_offset) != TF_OK)
|
|
279
|
+
return TF_ERROR;
|
|
280
|
+
st->line_number++;
|
|
281
|
+
st->byte_offset = new_offset;
|
|
282
|
+
pos += segment_len + 1;
|
|
283
|
+
} else {
|
|
284
|
+
if (segment_len > 0 &&
|
|
285
|
+
tf_buffer_write(&st->line_buf, start, segment_len) != TF_OK)
|
|
286
|
+
return TF_ERROR;
|
|
287
|
+
pos = len;
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
return TF_OK;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
static int text_flush(tf_decoder *self, tf_batch ***out, size_t *n_out, tf_side_channels *side) {
|
|
294
|
+
(void)side;
|
|
295
|
+
text_decoder_state *st = self->state;
|
|
296
|
+
*out = NULL;
|
|
297
|
+
*n_out = 0;
|
|
298
|
+
size_t out_cap = 0;
|
|
299
|
+
|
|
300
|
+
/* Process remaining data as a final line */
|
|
301
|
+
size_t remaining = tf_buffer_readable(&st->line_buf);
|
|
302
|
+
if (remaining > 0) {
|
|
303
|
+
uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
|
|
304
|
+
size_t line_no = st->line_number + 1;
|
|
305
|
+
size_t record_offset = st->byte_offset;
|
|
306
|
+
if (check_text_buffer_record_limit(st, remaining, line_no, record_offset, side) != TF_OK)
|
|
307
|
+
return TF_ERROR;
|
|
308
|
+
st->line_number = line_no;
|
|
309
|
+
|
|
310
|
+
if (!st->batch) {
|
|
311
|
+
st->batch = make_text_batch(st->batch_size);
|
|
312
|
+
if (!st->batch) return TF_ERROR;
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
size_t row = st->batch->n_rows;
|
|
316
|
+
size_t need_rows = 0;
|
|
317
|
+
if (tf_size_add(row, 1, &need_rows) != TF_OK ||
|
|
318
|
+
tf_batch_ensure_capacity(st->batch, need_rows) != TF_OK)
|
|
319
|
+
return TF_ERROR;
|
|
320
|
+
|
|
321
|
+
/* Strip trailing \r */
|
|
322
|
+
size_t line_len = remaining;
|
|
323
|
+
if (line_len > 0 && buf[line_len - 1] == '\r')
|
|
324
|
+
line_len--;
|
|
325
|
+
|
|
326
|
+
if (text_set_line_cell(st->batch, row, buf, line_len) != TF_OK)
|
|
327
|
+
return TF_ERROR;
|
|
328
|
+
if (tf_batch_expose_row(st->batch, row) != TF_OK)
|
|
329
|
+
return TF_ERROR;
|
|
330
|
+
st->rows_buffered++;
|
|
331
|
+
|
|
332
|
+
st->line_buf.read_pos = st->line_buf.len;
|
|
333
|
+
size_t new_offset = 0;
|
|
334
|
+
if (tf_size_add(st->byte_offset, remaining, &new_offset) != TF_OK)
|
|
335
|
+
return TF_ERROR;
|
|
336
|
+
st->byte_offset = new_offset;
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/* Emit remaining batch */
|
|
340
|
+
if (st->batch && st->rows_buffered > 0) {
|
|
341
|
+
if (text_append_output_batch(out, n_out, &out_cap, st->batch) != TF_OK)
|
|
342
|
+
return TF_ERROR;
|
|
343
|
+
st->batch = NULL;
|
|
344
|
+
st->rows_buffered = 0;
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
return TF_OK;
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
static void text_decoder_destroy(tf_decoder *self) {
|
|
351
|
+
text_decoder_state *st = self->state;
|
|
352
|
+
if (st) {
|
|
353
|
+
tf_buffer_free(&st->line_buf);
|
|
354
|
+
if (st->batch) tf_batch_free(st->batch);
|
|
355
|
+
free(st);
|
|
356
|
+
}
|
|
357
|
+
free(self);
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
tf_decoder *tf_text_decoder_create(const cJSON *args) {
|
|
361
|
+
text_decoder_state *st = calloc(1, sizeof(text_decoder_state));
|
|
362
|
+
if (!st) return NULL;
|
|
363
|
+
|
|
364
|
+
st->batch_size = DEFAULT_BATCH_SIZE;
|
|
365
|
+
st->max_error_bytes = DEFAULT_MAX_ERROR_BYTES;
|
|
366
|
+
st->max_record_bytes = DEFAULT_MAX_RECORD_BYTES;
|
|
367
|
+
if (args) {
|
|
368
|
+
size_t parsed_size = 0;
|
|
369
|
+
int has_batch_size = tf_json_get_size_arg(args, "batch_size", 1, TF_MAX_BATCH_ROWS, &parsed_size, "text");
|
|
370
|
+
if (has_batch_size < 0) { free(st); return NULL; }
|
|
371
|
+
if (has_batch_size > 0) st->batch_size = parsed_size;
|
|
372
|
+
int has_max_error = tf_json_get_size_arg(args, "max_error_bytes", 0, TF_MAX_ERROR_BYTES, &parsed_size, "text");
|
|
373
|
+
if (has_max_error < 0) { free(st); return NULL; }
|
|
374
|
+
if (has_max_error > 0) st->max_error_bytes = parsed_size;
|
|
375
|
+
int has_max_record = tf_json_get_size_arg(args, "max_record_bytes", 0, TF_MAX_RECORD_BYTES, &parsed_size, "text");
|
|
376
|
+
if (has_max_record < 0) { free(st); return NULL; }
|
|
377
|
+
if (has_max_record > 0) st->max_record_bytes = parsed_size;
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
tf_buffer_init(&st->line_buf);
|
|
381
|
+
|
|
382
|
+
tf_decoder *dec = malloc(sizeof(tf_decoder));
|
|
383
|
+
if (!dec) { free(st); return NULL; }
|
|
384
|
+
dec->decode = text_decode;
|
|
385
|
+
dec->flush = text_flush;
|
|
386
|
+
dec->destroy = text_decoder_destroy;
|
|
387
|
+
dec->state = st;
|
|
388
|
+
return dec;
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/* ================================================================
|
|
392
|
+
* Text Encoder
|
|
393
|
+
* ================================================================ */
|
|
394
|
+
|
|
395
|
+
static int text_write_string(tf_buffer *out, const char *s) {
|
|
396
|
+
const char *value = s ? s : "";
|
|
397
|
+
return tf_buffer_write(out, (const uint8_t *)value, strlen(value));
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
static int text_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
|
|
401
|
+
(void)self;
|
|
402
|
+
|
|
403
|
+
/* Find _line column index */
|
|
404
|
+
int line_col = tf_batch_col_index(in, "_line");
|
|
405
|
+
|
|
406
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
407
|
+
if (line_col >= 0) {
|
|
408
|
+
/* Write _line column value */
|
|
409
|
+
if (!tf_batch_is_null(in, r, (size_t)line_col)) {
|
|
410
|
+
const char *s = tf_batch_get_string(in, r, (size_t)line_col);
|
|
411
|
+
if (text_write_string(out, s) != TF_OK) return TF_ERROR;
|
|
412
|
+
}
|
|
413
|
+
} else {
|
|
414
|
+
/* Fallback: concatenate all string columns with tab */
|
|
415
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
416
|
+
if (c > 0 && tf_buffer_write(out, (const uint8_t *)"\t", 1) != TF_OK)
|
|
417
|
+
return TF_ERROR;
|
|
418
|
+
if (!tf_batch_is_null(in, r, c) && in->col_types[c] == TF_TYPE_STRING) {
|
|
419
|
+
const char *s = tf_batch_get_string(in, r, c);
|
|
420
|
+
if (text_write_string(out, s) != TF_OK) return TF_ERROR;
|
|
421
|
+
}
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
if (tf_buffer_write(out, (const uint8_t *)"\n", 1) != TF_OK) return TF_ERROR;
|
|
425
|
+
}
|
|
426
|
+
return TF_OK;
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
static int text_encoder_flush(tf_encoder *self, tf_buffer *out) {
|
|
430
|
+
(void)self; (void)out;
|
|
431
|
+
return TF_OK;
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
static void text_encoder_destroy(tf_encoder *self) {
|
|
435
|
+
free(self);
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
tf_encoder *tf_text_encoder_create(const cJSON *args) {
|
|
439
|
+
(void)args;
|
|
440
|
+
tf_encoder *enc = malloc(sizeof(tf_encoder));
|
|
441
|
+
if (!enc) return NULL;
|
|
442
|
+
enc->encode = text_encode;
|
|
443
|
+
enc->flush = text_encoder_flush;
|
|
444
|
+
enc->destroy = text_encoder_destroy;
|
|
445
|
+
enc->state = NULL;
|
|
446
|
+
return enc;
|
|
447
|
+
}
|