tranfi 0.0.2 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +395 -0
- package/app/assets/index-6quYZ5Ap.css +5 -0
- package/app/assets/index-pDFMluyz.js +160 -0
- package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
- package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
- package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
- package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
- package/app/index.html +13 -0
- package/binding.gyp +69 -0
- package/csrc/arena.c +91 -0
- package/csrc/batch.c +229 -0
- package/csrc/buffer.c +78 -0
- package/csrc/cJSON.c +3143 -0
- package/csrc/cJSON.h +300 -0
- package/csrc/codec_csv.c +1058 -0
- package/csrc/codec_jsonl.c +374 -0
- package/csrc/codec_table.c +218 -0
- package/csrc/codec_text.c +229 -0
- package/csrc/compiler.c +102 -0
- package/csrc/date_utils.h +94 -0
- package/csrc/dsl.c +1180 -0
- package/csrc/dsl.h +22 -0
- package/csrc/expr.c +1245 -0
- package/csrc/expr.h +56 -0
- package/csrc/internal.h +250 -0
- package/csrc/ir.c +119 -0
- package/csrc/ir.h +167 -0
- package/csrc/ir_schema.c +60 -0
- package/csrc/ir_serialize.c +104 -0
- package/csrc/ir_sql.c +1211 -0
- package/csrc/ir_validate.c +120 -0
- package/csrc/main.c +392 -0
- package/csrc/op_acf.c +133 -0
- package/csrc/op_anomaly.c +120 -0
- package/csrc/op_bin.c +109 -0
- package/csrc/op_cast.c +195 -0
- package/csrc/op_clip.c +88 -0
- package/csrc/op_date_trunc.c +181 -0
- package/csrc/op_datetime.c +212 -0
- package/csrc/op_derive.c +248 -0
- package/csrc/op_diff.c +134 -0
- package/csrc/op_ewma.c +103 -0
- package/csrc/op_explode.c +108 -0
- package/csrc/op_fill_down.c +163 -0
- package/csrc/op_fill_null.c +123 -0
- package/csrc/op_filter.c +132 -0
- package/csrc/op_frequency.c +193 -0
- package/csrc/op_grep.c +163 -0
- package/csrc/op_group_agg.c +285 -0
- package/csrc/op_hash.c +126 -0
- package/csrc/op_head.c +149 -0
- package/csrc/op_interpolate.c +239 -0
- package/csrc/op_join.c +384 -0
- package/csrc/op_label_encode.c +144 -0
- package/csrc/op_lead.c +190 -0
- package/csrc/op_normalize.c +226 -0
- package/csrc/op_onehot.c +185 -0
- package/csrc/op_pivot.c +370 -0
- package/csrc/op_registry.c +1148 -0
- package/csrc/op_rename.c +138 -0
- package/csrc/op_replace.c +202 -0
- package/csrc/op_sample.c +101 -0
- package/csrc/op_select.c +140 -0
- package/csrc/op_skip.c +152 -0
- package/csrc/op_sort.c +273 -0
- package/csrc/op_split.c +114 -0
- package/csrc/op_split_data.c +87 -0
- package/csrc/op_stack.c +315 -0
- package/csrc/op_stats.c +779 -0
- package/csrc/op_step.c +171 -0
- package/csrc/op_tail.c +96 -0
- package/csrc/op_top.c +150 -0
- package/csrc/op_trim.c +109 -0
- package/csrc/op_unique.c +300 -0
- package/csrc/op_unpivot.c +159 -0
- package/csrc/op_validate.c +71 -0
- package/csrc/op_window.c +150 -0
- package/csrc/pipeline.c +315 -0
- package/csrc/plan.c +206 -0
- package/csrc/recipes.c +102 -0
- package/csrc/recipes.h +27 -0
- package/csrc/report.c +463 -0
- package/csrc/report.h +22 -0
- package/csrc/tranfi.h +123 -0
- package/csrc/wasm_api.c +157 -0
- package/napi_api.c +326 -0
- package/package.json +46 -57
- package/src/cli.js +193 -0
- package/src/engines/duckdb.js +109 -0
- package/src/index.js +306 -0
- package/src/native.js +22 -0
- package/src/pipeline.js +286 -0
- package/src/server.js +277 -0
- package/src/wasm.js +19 -0
- package/wasm/index.js +244 -0
- package/wasm/package.json +1 -0
- package/wasm/tranfi_core.js +0 -0
- package/LICENSE +0 -21
- package/dist/bundle.js +0 -1
- package/index.html +0 -18
- package/src/app.css +0 -169
- package/src/app.js +0 -203
- package/src/app.vue +0 -250
- package/src/bulma-input.vue +0 -110
- package/src/common-inputs.js +0 -28
- package/src/main.js +0 -20
- package/src/transforms.js +0 -166
- package/webpack.config.js +0 -108
package/csrc/codec_csv.c
ADDED
|
@@ -0,0 +1,1058 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* codec_csv.c — Optimized streaming CSV decoder and encoder.
|
|
3
|
+
*
|
|
4
|
+
* Decoder design (three key optimizations):
|
|
5
|
+
*
|
|
6
|
+
* 1. Zero-copy field parsing: fields are returned as (ptr, len) slices
|
|
7
|
+
* into the line buffer, avoiding per-field malloc/free. Only quoted
|
|
8
|
+
* fields with escaped quotes ("") need copying (rare in practice).
|
|
9
|
+
*
|
|
10
|
+
* 2. Type detection window: the first batch (batch_size rows) detects
|
|
11
|
+
* column types via progressive widening (NULL → INT64 → FLOAT64 →
|
|
12
|
+
* STRING). Types freeze after the first batch. This matches the
|
|
13
|
+
* behavior of Arrow CSV, DuckDB, and other production parsers.
|
|
14
|
+
*
|
|
15
|
+
* 3. Direct-to-typed parsing: after types freeze, field slices are parsed
|
|
16
|
+
* directly into typed column arrays (int64, double, string) without
|
|
17
|
+
* an intermediate STRING batch. Combined with custom fast_int64/
|
|
18
|
+
* fast_double parsers, this eliminates double parsing for >99% of rows.
|
|
19
|
+
*
|
|
20
|
+
* Encoder: writes typed batches as RFC 4180 CSV with proper quoting.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
#include "internal.h"
|
|
24
|
+
#include "date_utils.h"
|
|
25
|
+
#include "cJSON.h"
|
|
26
|
+
#include <stdlib.h>
|
|
27
|
+
#include <string.h>
|
|
28
|
+
#include <stdio.h>
|
|
29
|
+
#include <errno.h>
|
|
30
|
+
#include <limits.h>
|
|
31
|
+
#include <math.h>
|
|
32
|
+
|
|
33
|
+
#define DEFAULT_BATCH_SIZE 1024
|
|
34
|
+
#define MAX_COLS 256 /* max columns per row */
|
|
35
|
+
|
|
36
|
+
/* ================================================================
|
|
37
|
+
* Field Slice — zero-copy reference into the line buffer
|
|
38
|
+
* ================================================================ */
|
|
39
|
+
|
|
40
|
+
typedef struct {
|
|
41
|
+
const char *ptr; /* points into line buffer (or field_arena for escapes) */
|
|
42
|
+
size_t len;
|
|
43
|
+
} field_slice;
|
|
44
|
+
|
|
45
|
+
/* ================================================================
|
|
46
|
+
* Fast Numeric Parsers
|
|
47
|
+
*
|
|
48
|
+
* These avoid libc strtoll/strtod overhead for common number formats.
|
|
49
|
+
* They take (ptr, len) instead of null-terminated strings, which
|
|
50
|
+
* pairs naturally with the zero-copy field slices.
|
|
51
|
+
* ================================================================ */
|
|
52
|
+
|
|
53
|
+
/*
|
|
54
|
+
* Fast int64 parser. Handles [-+]digits format only.
|
|
55
|
+
* Rejects decimal points, exponents, leading zeros (except "0"),
|
|
56
|
+
* and values outside the int64 range.
|
|
57
|
+
*/
|
|
58
|
+
static int fast_int64(const char *s, size_t len, int64_t *out) {
|
|
59
|
+
if (len == 0) return 0;
|
|
60
|
+
|
|
61
|
+
size_t i = 0;
|
|
62
|
+
int neg = 0;
|
|
63
|
+
if (s[0] == '-') { neg = 1; i = 1; }
|
|
64
|
+
else if (s[0] == '+') { i = 1; }
|
|
65
|
+
|
|
66
|
+
/* Must have at least one digit */
|
|
67
|
+
if (i >= len || s[i] < '0' || s[i] > '9') return 0;
|
|
68
|
+
|
|
69
|
+
/* Max int64 is 19 digits; reject longer numbers to avoid overflow */
|
|
70
|
+
size_t n_digits = len - i;
|
|
71
|
+
if (n_digits > 19) return 0;
|
|
72
|
+
|
|
73
|
+
uint64_t v = 0;
|
|
74
|
+
for (; i < len; i++) {
|
|
75
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
76
|
+
v = v * 10 + (uint64_t)(s[i] - '0');
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/* Range check against int64 bounds */
|
|
80
|
+
if (neg) {
|
|
81
|
+
/* INT64_MIN magnitude is INT64_MAX + 1 */
|
|
82
|
+
if (v > (uint64_t)INT64_MAX + 1) return 0;
|
|
83
|
+
*out = (v == (uint64_t)INT64_MAX + 1) ? INT64_MIN : -(int64_t)v;
|
|
84
|
+
} else {
|
|
85
|
+
if (v > (uint64_t)INT64_MAX) return 0;
|
|
86
|
+
*out = (int64_t)v;
|
|
87
|
+
}
|
|
88
|
+
return 1;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/*
|
|
92
|
+
* Fast double parser for common decimal formats: [-+]digits[.digits]
|
|
93
|
+
*
|
|
94
|
+
* Uses integer accumulation + power-of-10 division for precision.
|
|
95
|
+
* Handles up to 18 significant digits (fits in uint64 without overflow).
|
|
96
|
+
* Falls back to strtod for exponents (e/E), special values, or
|
|
97
|
+
* very long mantissas.
|
|
98
|
+
*/
|
|
99
|
+
static int fast_double(const char *s, size_t len, double *out) {
|
|
100
|
+
if (len == 0) return 0;
|
|
101
|
+
|
|
102
|
+
size_t i = 0;
|
|
103
|
+
int neg = 0;
|
|
104
|
+
if (s[0] == '-') { neg = 1; i = 1; }
|
|
105
|
+
else if (s[0] == '+') { i = 1; }
|
|
106
|
+
if (i >= len) return 0;
|
|
107
|
+
|
|
108
|
+
/* Accumulate mantissa as integer: 123.456 → mantissa=123456, n_frac=3 */
|
|
109
|
+
uint64_t mantissa = 0;
|
|
110
|
+
int n_digits = 0;
|
|
111
|
+
int n_frac = 0;
|
|
112
|
+
|
|
113
|
+
/* Integer part */
|
|
114
|
+
while (i < len && s[i] >= '0' && s[i] <= '9') {
|
|
115
|
+
mantissa = mantissa * 10 + (uint64_t)(s[i] - '0');
|
|
116
|
+
n_digits++;
|
|
117
|
+
i++;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/* Fractional part */
|
|
121
|
+
if (i < len && s[i] == '.') {
|
|
122
|
+
i++;
|
|
123
|
+
while (i < len && s[i] >= '0' && s[i] <= '9') {
|
|
124
|
+
mantissa = mantissa * 10 + (uint64_t)(s[i] - '0');
|
|
125
|
+
n_frac++;
|
|
126
|
+
n_digits++;
|
|
127
|
+
i++;
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
if (n_digits == 0) return 0;
|
|
132
|
+
|
|
133
|
+
/* Fast path: no exponent and ≤18 digits (uint64 safe) */
|
|
134
|
+
if (i == len && n_digits <= 18) {
|
|
135
|
+
static const double pow10[] = {
|
|
136
|
+
1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9,
|
|
137
|
+
1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18
|
|
138
|
+
};
|
|
139
|
+
double result = (double)mantissa;
|
|
140
|
+
if (n_frac > 0) result /= pow10[n_frac];
|
|
141
|
+
*out = neg ? -result : result;
|
|
142
|
+
return 1;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/* Fallback to strtod for exponents, very long numbers, etc. */
|
|
146
|
+
if (len >= 64) return 0;
|
|
147
|
+
char buf[64];
|
|
148
|
+
memcpy(buf, s, len);
|
|
149
|
+
buf[len] = '\0';
|
|
150
|
+
char *end;
|
|
151
|
+
errno = 0;
|
|
152
|
+
*out = strtod(buf, &end);
|
|
153
|
+
if (errno || (size_t)(end - buf) != len) return 0;
|
|
154
|
+
return 1;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/*
|
|
158
|
+
* Fast date parser: exactly YYYY-MM-DD (10 chars) → int32_t days since epoch.
|
|
159
|
+
* Returns 1 on success, 0 on failure.
|
|
160
|
+
*/
|
|
161
|
+
static int fast_date(const char *s, size_t len, int32_t *out) {
|
|
162
|
+
if (len != 10) return 0;
|
|
163
|
+
if (s[4] != '-' || s[7] != '-') return 0;
|
|
164
|
+
/* Parse YYYY */
|
|
165
|
+
int y = 0;
|
|
166
|
+
for (int i = 0; i < 4; i++) {
|
|
167
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
168
|
+
y = y * 10 + (s[i] - '0');
|
|
169
|
+
}
|
|
170
|
+
/* Parse MM */
|
|
171
|
+
int m = 0;
|
|
172
|
+
for (int i = 5; i < 7; i++) {
|
|
173
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
174
|
+
m = m * 10 + (s[i] - '0');
|
|
175
|
+
}
|
|
176
|
+
/* Parse DD */
|
|
177
|
+
int d = 0;
|
|
178
|
+
for (int i = 8; i < 10; i++) {
|
|
179
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
180
|
+
d = d * 10 + (s[i] - '0');
|
|
181
|
+
}
|
|
182
|
+
if (m < 1 || m > 12 || d < 1 || d > 31) return 0;
|
|
183
|
+
*out = tf_date_from_ymd(y, m, d);
|
|
184
|
+
return 1;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/*
|
|
188
|
+
* Fast timestamp parser: YYYY-MM-DD[T ]HH:MM:SS[.ffffff][Z|+HH:MM|-HH:MM]
|
|
189
|
+
* Returns 1 on success, 0 on failure.
|
|
190
|
+
*/
|
|
191
|
+
static int fast_timestamp(const char *s, size_t len, int64_t *out) {
|
|
192
|
+
if (len < 19) return 0;
|
|
193
|
+
if (s[4] != '-' || s[7] != '-') return 0;
|
|
194
|
+
if (s[10] != 'T' && s[10] != ' ') return 0;
|
|
195
|
+
if (s[13] != ':' || s[16] != ':') return 0;
|
|
196
|
+
|
|
197
|
+
int y = 0, mo = 0, d = 0, h = 0, mi = 0, se = 0;
|
|
198
|
+
/* YYYY */
|
|
199
|
+
for (int i = 0; i < 4; i++) {
|
|
200
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
201
|
+
y = y * 10 + (s[i] - '0');
|
|
202
|
+
}
|
|
203
|
+
/* MM */
|
|
204
|
+
for (int i = 5; i < 7; i++) {
|
|
205
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
206
|
+
mo = mo * 10 + (s[i] - '0');
|
|
207
|
+
}
|
|
208
|
+
/* DD */
|
|
209
|
+
for (int i = 8; i < 10; i++) {
|
|
210
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
211
|
+
d = d * 10 + (s[i] - '0');
|
|
212
|
+
}
|
|
213
|
+
/* HH */
|
|
214
|
+
for (int i = 11; i < 13; i++) {
|
|
215
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
216
|
+
h = h * 10 + (s[i] - '0');
|
|
217
|
+
}
|
|
218
|
+
/* MM */
|
|
219
|
+
for (int i = 14; i < 16; i++) {
|
|
220
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
221
|
+
mi = mi * 10 + (s[i] - '0');
|
|
222
|
+
}
|
|
223
|
+
/* SS */
|
|
224
|
+
for (int i = 17; i < 19; i++) {
|
|
225
|
+
if (s[i] < '0' || s[i] > '9') return 0;
|
|
226
|
+
se = se * 10 + (s[i] - '0');
|
|
227
|
+
}
|
|
228
|
+
if (mo < 1 || mo > 12 || d < 1 || d > 31) return 0;
|
|
229
|
+
if (h > 23 || mi > 59 || se > 59) return 0;
|
|
230
|
+
|
|
231
|
+
/* Optional fractional seconds */
|
|
232
|
+
int frac_us = 0;
|
|
233
|
+
size_t pos = 19;
|
|
234
|
+
if (pos < len && s[pos] == '.') {
|
|
235
|
+
pos++;
|
|
236
|
+
int frac_digits = 0;
|
|
237
|
+
int frac_val = 0;
|
|
238
|
+
while (pos < len && s[pos] >= '0' && s[pos] <= '9' && frac_digits < 6) {
|
|
239
|
+
frac_val = frac_val * 10 + (s[pos] - '0');
|
|
240
|
+
frac_digits++;
|
|
241
|
+
pos++;
|
|
242
|
+
}
|
|
243
|
+
/* Skip remaining digits beyond 6 */
|
|
244
|
+
while (pos < len && s[pos] >= '0' && s[pos] <= '9') pos++;
|
|
245
|
+
/* Pad to 6 digits */
|
|
246
|
+
while (frac_digits < 6) { frac_val *= 10; frac_digits++; }
|
|
247
|
+
frac_us = frac_val;
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/* Optional timezone */
|
|
251
|
+
int64_t tz_offset_us = 0;
|
|
252
|
+
if (pos < len) {
|
|
253
|
+
if (s[pos] == 'Z') {
|
|
254
|
+
pos++;
|
|
255
|
+
} else if (s[pos] == '+' || s[pos] == '-') {
|
|
256
|
+
int tz_sign = (s[pos] == '-') ? -1 : 1;
|
|
257
|
+
pos++;
|
|
258
|
+
if (pos + 2 > len) return 0;
|
|
259
|
+
int tz_h = (s[pos] - '0') * 10 + (s[pos + 1] - '0');
|
|
260
|
+
pos += 2;
|
|
261
|
+
int tz_m = 0;
|
|
262
|
+
if (pos < len && s[pos] == ':') {
|
|
263
|
+
pos++;
|
|
264
|
+
if (pos + 2 > len) return 0;
|
|
265
|
+
tz_m = (s[pos] - '0') * 10 + (s[pos + 1] - '0');
|
|
266
|
+
pos += 2;
|
|
267
|
+
}
|
|
268
|
+
tz_offset_us = tz_sign * ((int64_t)tz_h * 3600000000LL + (int64_t)tz_m * 60000000LL);
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
if (pos != len) return 0;
|
|
273
|
+
|
|
274
|
+
*out = tf_timestamp_from_parts(y, mo, d, h, mi, se, frac_us) - tz_offset_us;
|
|
275
|
+
return 1;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/* Detect the type of a field slice without copying it. */
|
|
279
|
+
static tf_type detect_type_slice(const char *s, size_t len) {
|
|
280
|
+
if (len == 0) return TF_TYPE_NULL;
|
|
281
|
+
int64_t iv;
|
|
282
|
+
double fv;
|
|
283
|
+
int32_t dv;
|
|
284
|
+
if (fast_int64(s, len, &iv)) return TF_TYPE_INT64;
|
|
285
|
+
if (fast_double(s, len, &fv)) return TF_TYPE_FLOAT64;
|
|
286
|
+
if (fast_date(s, len, &dv)) return TF_TYPE_DATE;
|
|
287
|
+
if (fast_timestamp(s, len, &iv)) return TF_TYPE_TIMESTAMP;
|
|
288
|
+
return TF_TYPE_STRING;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/* Widen a column type if needed (NULL < INT64 < FLOAT64 < STRING). */
|
|
292
|
+
static tf_type widen_type(tf_type current, tf_type incoming) {
|
|
293
|
+
if (current == incoming) return current;
|
|
294
|
+
if (current == TF_TYPE_NULL) return incoming;
|
|
295
|
+
if (incoming == TF_TYPE_NULL) return current;
|
|
296
|
+
if (current == TF_TYPE_INT64 && incoming == TF_TYPE_FLOAT64) return TF_TYPE_FLOAT64;
|
|
297
|
+
if (current == TF_TYPE_FLOAT64 && incoming == TF_TYPE_INT64) return TF_TYPE_FLOAT64;
|
|
298
|
+
if ((current == TF_TYPE_DATE && incoming == TF_TYPE_TIMESTAMP) ||
|
|
299
|
+
(current == TF_TYPE_TIMESTAMP && incoming == TF_TYPE_DATE))
|
|
300
|
+
return TF_TYPE_TIMESTAMP;
|
|
301
|
+
return TF_TYPE_STRING; /* anything else → string */
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/* ================================================================
|
|
305
|
+
* Zero-Copy Field Parser
|
|
306
|
+
*
|
|
307
|
+
* Parses a CSV line into an array of field_slice structs. For most
|
|
308
|
+
* fields, the slice points directly into the line buffer (zero copy).
|
|
309
|
+
* Only quoted fields with escaped quotes ("") need allocation, which
|
|
310
|
+
* goes into field_arena and is valid until arena reset.
|
|
311
|
+
* ================================================================ */
|
|
312
|
+
|
|
313
|
+
/*
|
|
314
|
+
* Parse a CSV line into field slices. Returns the number of fields.
|
|
315
|
+
*
|
|
316
|
+
* - Unquoted fields: zero-copy slice into line buffer
|
|
317
|
+
* - Quoted fields without "": zero-copy slice (skipping quotes)
|
|
318
|
+
* - Quoted fields with "": unescaped copy allocated from field_arena
|
|
319
|
+
* - Leading/trailing whitespace is trimmed from unquoted fields
|
|
320
|
+
*/
|
|
321
|
+
static size_t parse_csv_fields(const char *line, size_t line_len, char delim,
|
|
322
|
+
field_slice *fields, size_t max_fields,
|
|
323
|
+
tf_arena *field_arena) {
|
|
324
|
+
size_t count = 0;
|
|
325
|
+
size_t i = 0;
|
|
326
|
+
|
|
327
|
+
while (i <= line_len && count < max_fields) {
|
|
328
|
+
if (i == line_len) {
|
|
329
|
+
/* Trailing delimiter → empty last field */
|
|
330
|
+
if (count > 0 && i > 0 && line[i - 1] == delim) {
|
|
331
|
+
fields[count].ptr = "";
|
|
332
|
+
fields[count].len = 0;
|
|
333
|
+
count++;
|
|
334
|
+
}
|
|
335
|
+
break;
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
if (line[i] == '"') {
|
|
339
|
+
/* --- Quoted field --- */
|
|
340
|
+
i++; /* skip opening quote */
|
|
341
|
+
size_t start = i;
|
|
342
|
+
int has_escape = 0;
|
|
343
|
+
|
|
344
|
+
/* Scan for closing quote, detecting escaped quotes ("") */
|
|
345
|
+
while (i < line_len) {
|
|
346
|
+
if (line[i] == '"') {
|
|
347
|
+
if (i + 1 < line_len && line[i + 1] == '"') {
|
|
348
|
+
has_escape = 1;
|
|
349
|
+
i += 2;
|
|
350
|
+
} else {
|
|
351
|
+
break; /* closing quote */
|
|
352
|
+
}
|
|
353
|
+
} else {
|
|
354
|
+
i++;
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
size_t field_end = i;
|
|
358
|
+
if (i < line_len) i++; /* skip closing quote */
|
|
359
|
+
if (i < line_len && line[i] == delim) i++; /* skip delimiter */
|
|
360
|
+
|
|
361
|
+
if (!has_escape) {
|
|
362
|
+
/* Zero-copy: slice directly into line buffer, past the quotes */
|
|
363
|
+
fields[count].ptr = line + start;
|
|
364
|
+
fields[count].len = field_end - start;
|
|
365
|
+
} else {
|
|
366
|
+
/* Rare path: unescape "" → " into arena-allocated buffer */
|
|
367
|
+
size_t max_len = field_end - start; /* unescaped is always shorter */
|
|
368
|
+
char *buf = tf_arena_alloc(field_arena, max_len + 1);
|
|
369
|
+
size_t out_len = 0;
|
|
370
|
+
for (size_t j = start; j < field_end; j++) {
|
|
371
|
+
if (line[j] == '"' && j + 1 < field_end && line[j + 1] == '"') {
|
|
372
|
+
buf[out_len++] = '"';
|
|
373
|
+
j++; /* skip second quote */
|
|
374
|
+
} else {
|
|
375
|
+
buf[out_len++] = line[j];
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
buf[out_len] = '\0';
|
|
379
|
+
fields[count].ptr = buf;
|
|
380
|
+
fields[count].len = out_len;
|
|
381
|
+
}
|
|
382
|
+
count++;
|
|
383
|
+
} else {
|
|
384
|
+
/* --- Unquoted field: zero-copy slice with whitespace trimming --- */
|
|
385
|
+
size_t start = i;
|
|
386
|
+
while (i < line_len && line[i] != delim) i++;
|
|
387
|
+
|
|
388
|
+
const char *fptr = line + start;
|
|
389
|
+
size_t flen = i - start;
|
|
390
|
+
|
|
391
|
+
/* Trim trailing whitespace */
|
|
392
|
+
while (flen > 0 && (fptr[flen - 1] == ' ' || fptr[flen - 1] == '\t'))
|
|
393
|
+
flen--;
|
|
394
|
+
/* Trim leading whitespace */
|
|
395
|
+
while (flen > 0 && (*fptr == ' ' || *fptr == '\t'))
|
|
396
|
+
{ fptr++; flen--; }
|
|
397
|
+
|
|
398
|
+
fields[count].ptr = fptr;
|
|
399
|
+
fields[count].len = flen;
|
|
400
|
+
count++;
|
|
401
|
+
|
|
402
|
+
if (i < line_len) i++; /* skip delimiter */
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
return count;
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
/* ================================================================
|
|
410
|
+
* CSV Decoder
|
|
411
|
+
* ================================================================ */
|
|
412
|
+
|
|
413
|
+
typedef struct {
|
|
414
|
+
char delimiter;
|
|
415
|
+
int has_header;
|
|
416
|
+
size_t batch_size;
|
|
417
|
+
int repair; /* if true, pad short rows and truncate long rows */
|
|
418
|
+
|
|
419
|
+
/* Line accumulator: incoming bytes are appended, complete lines extracted */
|
|
420
|
+
tf_buffer line_buf;
|
|
421
|
+
|
|
422
|
+
/* Schema (discovered from first row) */
|
|
423
|
+
char **col_names;
|
|
424
|
+
tf_type *col_types;
|
|
425
|
+
size_t n_cols;
|
|
426
|
+
int schema_ready;
|
|
427
|
+
|
|
428
|
+
/* After the first batch, types freeze and we parse directly to typed
|
|
429
|
+
* columns. This avoids double parsing for >99% of rows. */
|
|
430
|
+
int types_frozen;
|
|
431
|
+
|
|
432
|
+
/* Current batch being built */
|
|
433
|
+
tf_batch *batch;
|
|
434
|
+
size_t rows_buffered;
|
|
435
|
+
|
|
436
|
+
/* Reusable per-line scratch: field slices array and arena for escapes */
|
|
437
|
+
field_slice fields[MAX_COLS];
|
|
438
|
+
tf_arena *field_arena;
|
|
439
|
+
} csv_decoder_state;
|
|
440
|
+
|
|
441
|
+
/*
|
|
442
|
+
* Create a batch with all STRING columns (for type detection phase).
|
|
443
|
+
* During this phase we don't know final types yet, so everything
|
|
444
|
+
* is stored as strings and converted at emission time.
|
|
445
|
+
*/
|
|
446
|
+
static tf_batch *make_string_batch(csv_decoder_state *st) {
|
|
447
|
+
tf_batch *b = tf_batch_create(st->n_cols, st->batch_size);
|
|
448
|
+
if (!b) return NULL;
|
|
449
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
450
|
+
tf_batch_set_schema(b, i, st->col_names[i], TF_TYPE_STRING);
|
|
451
|
+
}
|
|
452
|
+
return b;
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
/*
|
|
456
|
+
* Create a batch with the final (frozen) column types.
|
|
457
|
+
* After type detection, all batches are created with correct types
|
|
458
|
+
* so values can be parsed directly into typed columns.
|
|
459
|
+
*/
|
|
460
|
+
static tf_batch *make_typed_batch(csv_decoder_state *st) {
|
|
461
|
+
tf_batch *b = tf_batch_create(st->n_cols, st->batch_size);
|
|
462
|
+
if (!b) return NULL;
|
|
463
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
464
|
+
tf_batch_set_schema(b, i, st->col_names[i], st->col_types[i]);
|
|
465
|
+
}
|
|
466
|
+
return b;
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
/*
|
|
470
|
+
* Add a row of field slices to a STRING-typed batch.
|
|
471
|
+
* Used during the type detection phase (first batch).
|
|
472
|
+
* Copies slice content into the batch's arena.
|
|
473
|
+
*/
|
|
474
|
+
static void add_row_strings(tf_batch *b, const field_slice *fields,
|
|
475
|
+
size_t n_fields, size_t n_cols) {
|
|
476
|
+
size_t row = b->n_rows;
|
|
477
|
+
size_t cols = n_cols < n_fields ? n_cols : n_fields;
|
|
478
|
+
|
|
479
|
+
for (size_t i = 0; i < cols; i++) {
|
|
480
|
+
if (fields[i].len == 0) {
|
|
481
|
+
b->nulls[i][row] = 1;
|
|
482
|
+
} else {
|
|
483
|
+
/* Copy slice into batch arena as null-terminated string */
|
|
484
|
+
char *copy = tf_arena_alloc(b->arena, fields[i].len + 1);
|
|
485
|
+
memcpy(copy, fields[i].ptr, fields[i].len);
|
|
486
|
+
copy[fields[i].len] = '\0';
|
|
487
|
+
((char **)b->columns[i])[row] = copy;
|
|
488
|
+
b->nulls[i][row] = 0;
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
/* Null-fill extra columns */
|
|
492
|
+
for (size_t i = cols; i < n_cols; i++) {
|
|
493
|
+
b->nulls[i][row] = 1;
|
|
494
|
+
}
|
|
495
|
+
b->n_rows = row + 1;
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
/*
|
|
499
|
+
* Add a row by parsing field slices directly into typed columns.
|
|
500
|
+
* Used after types are frozen (all batches after the first).
|
|
501
|
+
*
|
|
502
|
+
* Writes directly to column arrays, bypassing tf_batch_set_*()
|
|
503
|
+
* bounds/type checks for speed in this hot path.
|
|
504
|
+
*/
|
|
505
|
+
static void add_row_typed(tf_batch *b, const field_slice *fields,
|
|
506
|
+
size_t n_fields, size_t n_cols,
|
|
507
|
+
const tf_type *types) {
|
|
508
|
+
size_t row = b->n_rows;
|
|
509
|
+
size_t cols = n_cols < n_fields ? n_cols : n_fields;
|
|
510
|
+
|
|
511
|
+
for (size_t i = 0; i < cols; i++) {
|
|
512
|
+
if (fields[i].len == 0) {
|
|
513
|
+
b->nulls[i][row] = 1;
|
|
514
|
+
continue;
|
|
515
|
+
}
|
|
516
|
+
switch (types[i]) {
|
|
517
|
+
case TF_TYPE_INT64: {
|
|
518
|
+
int64_t v;
|
|
519
|
+
if (fast_int64(fields[i].ptr, fields[i].len, &v)) {
|
|
520
|
+
((int64_t *)b->columns[i])[row] = v;
|
|
521
|
+
b->nulls[i][row] = 0;
|
|
522
|
+
} else {
|
|
523
|
+
b->nulls[i][row] = 1;
|
|
524
|
+
}
|
|
525
|
+
break;
|
|
526
|
+
}
|
|
527
|
+
case TF_TYPE_FLOAT64: {
|
|
528
|
+
double v;
|
|
529
|
+
if (fast_double(fields[i].ptr, fields[i].len, &v)) {
|
|
530
|
+
((double *)b->columns[i])[row] = v;
|
|
531
|
+
b->nulls[i][row] = 0;
|
|
532
|
+
} else {
|
|
533
|
+
b->nulls[i][row] = 1;
|
|
534
|
+
}
|
|
535
|
+
break;
|
|
536
|
+
}
|
|
537
|
+
case TF_TYPE_STRING: {
|
|
538
|
+
char *copy = tf_arena_alloc(b->arena, fields[i].len + 1);
|
|
539
|
+
memcpy(copy, fields[i].ptr, fields[i].len);
|
|
540
|
+
copy[fields[i].len] = '\0';
|
|
541
|
+
((char **)b->columns[i])[row] = copy;
|
|
542
|
+
b->nulls[i][row] = 0;
|
|
543
|
+
break;
|
|
544
|
+
}
|
|
545
|
+
case TF_TYPE_DATE: {
|
|
546
|
+
int32_t v;
|
|
547
|
+
if (fast_date(fields[i].ptr, fields[i].len, &v)) {
|
|
548
|
+
((int32_t *)b->columns[i])[row] = v;
|
|
549
|
+
b->nulls[i][row] = 0;
|
|
550
|
+
} else {
|
|
551
|
+
b->nulls[i][row] = 1;
|
|
552
|
+
}
|
|
553
|
+
break;
|
|
554
|
+
}
|
|
555
|
+
case TF_TYPE_TIMESTAMP: {
|
|
556
|
+
int64_t v;
|
|
557
|
+
if (fast_timestamp(fields[i].ptr, fields[i].len, &v)) {
|
|
558
|
+
((int64_t *)b->columns[i])[row] = v;
|
|
559
|
+
b->nulls[i][row] = 0;
|
|
560
|
+
} else {
|
|
561
|
+
/* Also try parsing a date-only string as timestamp at midnight */
|
|
562
|
+
int32_t dv;
|
|
563
|
+
if (fast_date(fields[i].ptr, fields[i].len, &dv)) {
|
|
564
|
+
((int64_t *)b->columns[i])[row] = (int64_t)dv * 86400LL * 1000000LL;
|
|
565
|
+
b->nulls[i][row] = 0;
|
|
566
|
+
} else {
|
|
567
|
+
b->nulls[i][row] = 1;
|
|
568
|
+
}
|
|
569
|
+
}
|
|
570
|
+
break;
|
|
571
|
+
}
|
|
572
|
+
default:
|
|
573
|
+
b->nulls[i][row] = 1;
|
|
574
|
+
break;
|
|
575
|
+
}
|
|
576
|
+
}
|
|
577
|
+
/* Null-fill extra columns */
|
|
578
|
+
for (size_t i = cols; i < n_cols; i++) {
|
|
579
|
+
b->nulls[i][row] = 1;
|
|
580
|
+
}
|
|
581
|
+
b->n_rows = row + 1;
|
|
582
|
+
}
|
|
583
|
+
|
|
584
|
+
/*
|
|
585
|
+
* Convert a STRING batch to a typed batch using frozen column types.
|
|
586
|
+
* Called once for the first batch after type detection completes.
|
|
587
|
+
* Uses fast_int64/fast_double for numeric conversion.
|
|
588
|
+
*/
|
|
589
|
+
static tf_batch *convert_batch_types(csv_decoder_state *st) {
|
|
590
|
+
tf_batch *src = st->batch;
|
|
591
|
+
tf_batch *dst = tf_batch_create(st->n_cols, src->n_rows);
|
|
592
|
+
if (!dst) return NULL;
|
|
593
|
+
|
|
594
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
595
|
+
tf_batch_set_schema(dst, i, st->col_names[i], st->col_types[i]);
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
for (size_t r = 0; r < src->n_rows; r++) {
|
|
599
|
+
for (size_t c = 0; c < st->n_cols; c++) {
|
|
600
|
+
if (src->nulls[c][r]) {
|
|
601
|
+
dst->nulls[c][r] = 1;
|
|
602
|
+
continue;
|
|
603
|
+
}
|
|
604
|
+
const char *val = ((char **)src->columns[c])[r];
|
|
605
|
+
if (!val || !*val) {
|
|
606
|
+
dst->nulls[c][r] = 1;
|
|
607
|
+
continue;
|
|
608
|
+
}
|
|
609
|
+
size_t vlen = strlen(val);
|
|
610
|
+
switch (st->col_types[c]) {
|
|
611
|
+
case TF_TYPE_INT64: {
|
|
612
|
+
int64_t v;
|
|
613
|
+
if (fast_int64(val, vlen, &v)) {
|
|
614
|
+
((int64_t *)dst->columns[c])[r] = v;
|
|
615
|
+
dst->nulls[c][r] = 0;
|
|
616
|
+
} else {
|
|
617
|
+
dst->nulls[c][r] = 1;
|
|
618
|
+
}
|
|
619
|
+
break;
|
|
620
|
+
}
|
|
621
|
+
case TF_TYPE_FLOAT64: {
|
|
622
|
+
double v;
|
|
623
|
+
if (fast_double(val, vlen, &v)) {
|
|
624
|
+
((double *)dst->columns[c])[r] = v;
|
|
625
|
+
dst->nulls[c][r] = 0;
|
|
626
|
+
} else {
|
|
627
|
+
dst->nulls[c][r] = 1;
|
|
628
|
+
}
|
|
629
|
+
break;
|
|
630
|
+
}
|
|
631
|
+
case TF_TYPE_STRING:
|
|
632
|
+
((char **)dst->columns[c])[r] = tf_arena_strdup(dst->arena, val);
|
|
633
|
+
dst->nulls[c][r] = 0;
|
|
634
|
+
break;
|
|
635
|
+
case TF_TYPE_DATE: {
|
|
636
|
+
int32_t dv;
|
|
637
|
+
if (fast_date(val, vlen, &dv)) {
|
|
638
|
+
((int32_t *)dst->columns[c])[r] = dv;
|
|
639
|
+
dst->nulls[c][r] = 0;
|
|
640
|
+
} else {
|
|
641
|
+
dst->nulls[c][r] = 1;
|
|
642
|
+
}
|
|
643
|
+
break;
|
|
644
|
+
}
|
|
645
|
+
case TF_TYPE_TIMESTAMP: {
|
|
646
|
+
int64_t tv;
|
|
647
|
+
if (fast_timestamp(val, vlen, &tv)) {
|
|
648
|
+
((int64_t *)dst->columns[c])[r] = tv;
|
|
649
|
+
dst->nulls[c][r] = 0;
|
|
650
|
+
} else {
|
|
651
|
+
int32_t dv;
|
|
652
|
+
if (fast_date(val, vlen, &dv)) {
|
|
653
|
+
((int64_t *)dst->columns[c])[r] = (int64_t)dv * 86400LL * 1000000LL;
|
|
654
|
+
dst->nulls[c][r] = 0;
|
|
655
|
+
} else {
|
|
656
|
+
dst->nulls[c][r] = 1;
|
|
657
|
+
}
|
|
658
|
+
}
|
|
659
|
+
break;
|
|
660
|
+
}
|
|
661
|
+
default:
|
|
662
|
+
dst->nulls[c][r] = 1;
|
|
663
|
+
break;
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
dst->n_rows = r + 1;
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
return dst;
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
/*
|
|
673
|
+
* Add a completed batch to the output array.
|
|
674
|
+
*/
|
|
675
|
+
static int emit_batch(tf_batch *batch, tf_batch ***out, size_t *n_out, size_t *out_cap) {
|
|
676
|
+
if (*n_out >= *out_cap) {
|
|
677
|
+
*out_cap = (*out_cap == 0) ? 4 : *out_cap * 2;
|
|
678
|
+
*out = realloc(*out, *out_cap * sizeof(tf_batch *));
|
|
679
|
+
if (!*out) return TF_ERROR;
|
|
680
|
+
}
|
|
681
|
+
(*out)[(*n_out)++] = batch;
|
|
682
|
+
return TF_OK;
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
/*
|
|
686
|
+
* Process a single complete CSV line.
|
|
687
|
+
*
|
|
688
|
+
* First line → extract column headers.
|
|
689
|
+
* Subsequent lines → parse fields and add to current batch.
|
|
690
|
+
* When batch is full → emit it and start a new one.
|
|
691
|
+
*/
|
|
692
|
+
static int process_line(csv_decoder_state *st, const char *line, size_t line_len,
|
|
693
|
+
tf_batch ***out, size_t *n_out, size_t *out_cap) {
|
|
694
|
+
/* Reset field arena — escaped field data from previous line is discarded */
|
|
695
|
+
tf_arena_reset(st->field_arena);
|
|
696
|
+
|
|
697
|
+
/* Parse the line into zero-copy field slices */
|
|
698
|
+
size_t n_fields = parse_csv_fields(line, line_len, st->delimiter,
|
|
699
|
+
st->fields, MAX_COLS, st->field_arena);
|
|
700
|
+
|
|
701
|
+
/* --- First line: extract column headers --- */
|
|
702
|
+
if (!st->schema_ready) {
|
|
703
|
+
st->n_cols = n_fields;
|
|
704
|
+
st->col_names = malloc(n_fields * sizeof(char *));
|
|
705
|
+
st->col_types = calloc(n_fields, sizeof(tf_type));
|
|
706
|
+
if (!st->col_names || !st->col_types) return TF_ERROR;
|
|
707
|
+
|
|
708
|
+
for (size_t i = 0; i < n_fields; i++) {
|
|
709
|
+
/* Headers must outlive the line buffer, so we copy them */
|
|
710
|
+
char *name = malloc(st->fields[i].len + 1);
|
|
711
|
+
if (!name) return TF_ERROR;
|
|
712
|
+
memcpy(name, st->fields[i].ptr, st->fields[i].len);
|
|
713
|
+
name[st->fields[i].len] = '\0';
|
|
714
|
+
st->col_names[i] = name;
|
|
715
|
+
st->col_types[i] = TF_TYPE_NULL;
|
|
716
|
+
}
|
|
717
|
+
st->schema_ready = 1;
|
|
718
|
+
return TF_OK;
|
|
719
|
+
}
|
|
720
|
+
|
|
721
|
+
/* --- Repair mode: normalize field count to match header --- */
|
|
722
|
+
if (st->repair && n_fields != st->n_cols) {
|
|
723
|
+
if (n_fields < st->n_cols) {
|
|
724
|
+
/* Pad short row with empty fields */
|
|
725
|
+
for (size_t i = n_fields; i < st->n_cols && i < MAX_COLS; i++) {
|
|
726
|
+
st->fields[i].ptr = "";
|
|
727
|
+
st->fields[i].len = 0;
|
|
728
|
+
}
|
|
729
|
+
n_fields = st->n_cols;
|
|
730
|
+
} else {
|
|
731
|
+
/* Truncate long row */
|
|
732
|
+
n_fields = st->n_cols;
|
|
733
|
+
}
|
|
734
|
+
}
|
|
735
|
+
|
|
736
|
+
/* --- Ensure we have a batch --- */
|
|
737
|
+
if (!st->batch) {
|
|
738
|
+
if (st->types_frozen) {
|
|
739
|
+
st->batch = make_typed_batch(st);
|
|
740
|
+
} else {
|
|
741
|
+
st->batch = make_string_batch(st);
|
|
742
|
+
}
|
|
743
|
+
if (!st->batch) return TF_ERROR;
|
|
744
|
+
}
|
|
745
|
+
|
|
746
|
+
/* --- Add row to batch --- */
|
|
747
|
+
if (!st->types_frozen) {
|
|
748
|
+
/* Type detection phase: detect types and store as STRING */
|
|
749
|
+
for (size_t i = 0; i < n_fields && i < st->n_cols; i++) {
|
|
750
|
+
tf_type t = detect_type_slice(st->fields[i].ptr, st->fields[i].len);
|
|
751
|
+
st->col_types[i] = widen_type(st->col_types[i], t);
|
|
752
|
+
}
|
|
753
|
+
add_row_strings(st->batch, st->fields, n_fields, st->n_cols);
|
|
754
|
+
} else {
|
|
755
|
+
/* Direct parse phase: parse directly to typed columns */
|
|
756
|
+
add_row_typed(st->batch, st->fields, n_fields, st->n_cols, st->col_types);
|
|
757
|
+
}
|
|
758
|
+
st->rows_buffered++;
|
|
759
|
+
|
|
760
|
+
/* --- Emit batch if full --- */
|
|
761
|
+
if (st->rows_buffered >= st->batch_size) {
|
|
762
|
+
if (!st->types_frozen) {
|
|
763
|
+
/* First batch complete: convert STRING → typed, freeze types.
|
|
764
|
+
* Default any still-NULL columns to STRING. */
|
|
765
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
766
|
+
if (st->col_types[i] == TF_TYPE_NULL)
|
|
767
|
+
st->col_types[i] = TF_TYPE_STRING;
|
|
768
|
+
}
|
|
769
|
+
tf_batch *final = convert_batch_types(st);
|
|
770
|
+
if (!final) return TF_ERROR;
|
|
771
|
+
tf_batch_free(st->batch);
|
|
772
|
+
st->batch = NULL;
|
|
773
|
+
st->types_frozen = 1;
|
|
774
|
+
if (emit_batch(final, out, n_out, out_cap) != TF_OK) return TF_ERROR;
|
|
775
|
+
} else {
|
|
776
|
+
/* Already typed, emit directly (no conversion needed) */
|
|
777
|
+
if (emit_batch(st->batch, out, n_out, out_cap) != TF_OK) return TF_ERROR;
|
|
778
|
+
st->batch = NULL;
|
|
779
|
+
}
|
|
780
|
+
st->rows_buffered = 0;
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
return TF_OK;
|
|
784
|
+
}
|
|
785
|
+
|
|
786
|
+
/*
|
|
787
|
+
* Main decode entry point: append data, extract complete lines, process them.
|
|
788
|
+
*
|
|
789
|
+
* The line scanner respects quoted fields that may contain newlines.
|
|
790
|
+
* Complete lines are passed to process_line(); any trailing partial
|
|
791
|
+
* line remains in line_buf for the next call.
|
|
792
|
+
*/
|
|
793
|
+
static int csv_decode(tf_decoder *self, const uint8_t *data, size_t len,
|
|
794
|
+
tf_batch ***out, size_t *n_out) {
|
|
795
|
+
csv_decoder_state *st = self->state;
|
|
796
|
+
*out = NULL;
|
|
797
|
+
*n_out = 0;
|
|
798
|
+
|
|
799
|
+
/* Append incoming data to line buffer */
|
|
800
|
+
if (tf_buffer_write(&st->line_buf, data, len) != TF_OK) return TF_ERROR;
|
|
801
|
+
|
|
802
|
+
/* Scan for complete lines */
|
|
803
|
+
size_t out_cap = 0;
|
|
804
|
+
uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
|
|
805
|
+
size_t buf_len = st->line_buf.len - st->line_buf.read_pos;
|
|
806
|
+
|
|
807
|
+
size_t line_start = 0;
|
|
808
|
+
int in_quotes = 0;
|
|
809
|
+
for (size_t i = 0; i < buf_len; i++) {
|
|
810
|
+
if (buf[i] == '"') {
|
|
811
|
+
in_quotes = !in_quotes;
|
|
812
|
+
} else if (!in_quotes && (buf[i] == '\n' || buf[i] == '\r')) {
|
|
813
|
+
size_t line_len = i - line_start;
|
|
814
|
+
/* Handle \r\n */
|
|
815
|
+
if (buf[i] == '\r' && i + 1 < buf_len && buf[i + 1] == '\n') {
|
|
816
|
+
i++;
|
|
817
|
+
}
|
|
818
|
+
if (line_len > 0) {
|
|
819
|
+
if (process_line(st, (const char *)buf + line_start, line_len,
|
|
820
|
+
out, n_out, &out_cap) != TF_OK)
|
|
821
|
+
return TF_ERROR;
|
|
822
|
+
}
|
|
823
|
+
line_start = i + 1;
|
|
824
|
+
}
|
|
825
|
+
}
|
|
826
|
+
|
|
827
|
+
/* Move unconsumed data to start of buffer */
|
|
828
|
+
st->line_buf.read_pos += line_start;
|
|
829
|
+
tf_buffer_compact(&st->line_buf);
|
|
830
|
+
|
|
831
|
+
return TF_OK;
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
/*
|
|
835
|
+
* Flush: process any remaining partial line and emit the final batch.
|
|
836
|
+
*/
|
|
837
|
+
static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
|
|
838
|
+
csv_decoder_state *st = self->state;
|
|
839
|
+
*out = NULL;
|
|
840
|
+
*n_out = 0;
|
|
841
|
+
size_t out_cap = 0;
|
|
842
|
+
|
|
843
|
+
/* Process any remaining data as the last line */
|
|
844
|
+
size_t remaining = tf_buffer_readable(&st->line_buf);
|
|
845
|
+
if (remaining > 0) {
|
|
846
|
+
uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
|
|
847
|
+
if (process_line(st, (const char *)buf, remaining,
|
|
848
|
+
out, n_out, &out_cap) != TF_OK)
|
|
849
|
+
return TF_ERROR;
|
|
850
|
+
st->line_buf.read_pos = st->line_buf.len;
|
|
851
|
+
}
|
|
852
|
+
|
|
853
|
+
/* Emit any remaining partial batch */
|
|
854
|
+
if (st->batch && st->rows_buffered > 0) {
|
|
855
|
+
if (!st->types_frozen) {
|
|
856
|
+
/* Small file: fewer rows than batch_size. Convert and emit. */
|
|
857
|
+
for (size_t i = 0; i < st->n_cols; i++) {
|
|
858
|
+
if (st->col_types[i] == TF_TYPE_NULL)
|
|
859
|
+
st->col_types[i] = TF_TYPE_STRING;
|
|
860
|
+
}
|
|
861
|
+
tf_batch *final = convert_batch_types(st);
|
|
862
|
+
if (!final) return TF_ERROR;
|
|
863
|
+
tf_batch_free(st->batch);
|
|
864
|
+
st->batch = NULL;
|
|
865
|
+
if (emit_batch(final, out, n_out, &out_cap) != TF_OK) return TF_ERROR;
|
|
866
|
+
} else {
|
|
867
|
+
/* Already typed, emit directly */
|
|
868
|
+
if (emit_batch(st->batch, out, n_out, &out_cap) != TF_OK) return TF_ERROR;
|
|
869
|
+
st->batch = NULL;
|
|
870
|
+
}
|
|
871
|
+
st->rows_buffered = 0;
|
|
872
|
+
}
|
|
873
|
+
|
|
874
|
+
return TF_OK;
|
|
875
|
+
}
|
|
876
|
+
|
|
877
|
+
static void csv_decoder_destroy(tf_decoder *self) {
|
|
878
|
+
csv_decoder_state *st = self->state;
|
|
879
|
+
if (st) {
|
|
880
|
+
tf_buffer_free(&st->line_buf);
|
|
881
|
+
if (st->batch) tf_batch_free(st->batch);
|
|
882
|
+
if (st->field_arena) tf_arena_free(st->field_arena);
|
|
883
|
+
for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
|
|
884
|
+
free(st->col_names);
|
|
885
|
+
free(st->col_types);
|
|
886
|
+
free(st);
|
|
887
|
+
}
|
|
888
|
+
free(self);
|
|
889
|
+
}
|
|
890
|
+
|
|
891
|
+
tf_decoder *tf_csv_decoder_create(const cJSON *args) {
|
|
892
|
+
csv_decoder_state *st = calloc(1, sizeof(csv_decoder_state));
|
|
893
|
+
if (!st) return NULL;
|
|
894
|
+
|
|
895
|
+
st->delimiter = ',';
|
|
896
|
+
st->has_header = 1;
|
|
897
|
+
st->batch_size = DEFAULT_BATCH_SIZE;
|
|
898
|
+
st->repair = 0;
|
|
899
|
+
|
|
900
|
+
if (args) {
|
|
901
|
+
cJSON *d = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
|
|
902
|
+
if (cJSON_IsString(d) && d->valuestring[0])
|
|
903
|
+
st->delimiter = d->valuestring[0];
|
|
904
|
+
|
|
905
|
+
cJSON *h = cJSON_GetObjectItemCaseSensitive(args, "header");
|
|
906
|
+
if (cJSON_IsBool(h))
|
|
907
|
+
st->has_header = cJSON_IsTrue(h);
|
|
908
|
+
|
|
909
|
+
cJSON *bs = cJSON_GetObjectItemCaseSensitive(args, "batch_size");
|
|
910
|
+
if (cJSON_IsNumber(bs) && bs->valueint > 0)
|
|
911
|
+
st->batch_size = (size_t)bs->valueint;
|
|
912
|
+
|
|
913
|
+
cJSON *rep = cJSON_GetObjectItemCaseSensitive(args, "repair");
|
|
914
|
+
if (cJSON_IsBool(rep))
|
|
915
|
+
st->repair = cJSON_IsTrue(rep);
|
|
916
|
+
}
|
|
917
|
+
|
|
918
|
+
tf_buffer_init(&st->line_buf);
|
|
919
|
+
|
|
920
|
+
/* Arena for escaped quoted field data (reset per line, rarely used) */
|
|
921
|
+
st->field_arena = tf_arena_create(4096);
|
|
922
|
+
if (!st->field_arena) { free(st); return NULL; }
|
|
923
|
+
|
|
924
|
+
tf_decoder *dec = malloc(sizeof(tf_decoder));
|
|
925
|
+
if (!dec) { tf_arena_free(st->field_arena); free(st); return NULL; }
|
|
926
|
+
dec->decode = csv_decode;
|
|
927
|
+
dec->flush = csv_flush;
|
|
928
|
+
dec->destroy = csv_decoder_destroy;
|
|
929
|
+
dec->state = st;
|
|
930
|
+
return dec;
|
|
931
|
+
}
|
|
932
|
+
|
|
933
|
+
/* ================================================================
|
|
934
|
+
* CSV Encoder (unchanged — already efficient)
|
|
935
|
+
* ================================================================ */
|
|
936
|
+
|
|
937
|
+
typedef struct {
|
|
938
|
+
char delimiter;
|
|
939
|
+
int header_written;
|
|
940
|
+
} csv_encoder_state;
|
|
941
|
+
|
|
942
|
+
/* Check if a field needs quoting */
|
|
943
|
+
static int needs_quoting(const char *s, char delim) {
|
|
944
|
+
for (const char *p = s; *p; p++) {
|
|
945
|
+
if (*p == delim || *p == '"' || *p == '\n' || *p == '\r')
|
|
946
|
+
return 1;
|
|
947
|
+
}
|
|
948
|
+
return 0;
|
|
949
|
+
}
|
|
950
|
+
|
|
951
|
+
static int write_field(tf_buffer *out, const char *s, char delim) {
|
|
952
|
+
if (needs_quoting(s, delim)) {
|
|
953
|
+
if (tf_buffer_write(out, (const uint8_t *)"\"", 1) != TF_OK) return TF_ERROR;
|
|
954
|
+
for (const char *p = s; *p; p++) {
|
|
955
|
+
if (*p == '"') {
|
|
956
|
+
if (tf_buffer_write(out, (const uint8_t *)"\"\"", 2) != TF_OK) return TF_ERROR;
|
|
957
|
+
} else {
|
|
958
|
+
if (tf_buffer_write(out, (const uint8_t *)p, 1) != TF_OK) return TF_ERROR;
|
|
959
|
+
}
|
|
960
|
+
}
|
|
961
|
+
if (tf_buffer_write(out, (const uint8_t *)"\"", 1) != TF_OK) return TF_ERROR;
|
|
962
|
+
} else {
|
|
963
|
+
if (tf_buffer_write_str(out, s) != TF_OK) return TF_ERROR;
|
|
964
|
+
}
|
|
965
|
+
return TF_OK;
|
|
966
|
+
}
|
|
967
|
+
|
|
968
|
+
static int csv_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
|
|
969
|
+
csv_encoder_state *st = self->state;
|
|
970
|
+
char dbuf[2] = { st->delimiter, '\0' };
|
|
971
|
+
|
|
972
|
+
/* Write header */
|
|
973
|
+
if (!st->header_written) {
|
|
974
|
+
for (size_t i = 0; i < in->n_cols; i++) {
|
|
975
|
+
if (i > 0) tf_buffer_write_str(out, dbuf);
|
|
976
|
+
write_field(out, in->col_names[i], st->delimiter);
|
|
977
|
+
}
|
|
978
|
+
tf_buffer_write(out, (const uint8_t *)"\n", 1);
|
|
979
|
+
st->header_written = 1;
|
|
980
|
+
}
|
|
981
|
+
|
|
982
|
+
/* Write rows */
|
|
983
|
+
char numbuf[64];
|
|
984
|
+
for (size_t r = 0; r < in->n_rows; r++) {
|
|
985
|
+
for (size_t c = 0; c < in->n_cols; c++) {
|
|
986
|
+
if (c > 0) tf_buffer_write_str(out, dbuf);
|
|
987
|
+
if (tf_batch_is_null(in, r, c)) {
|
|
988
|
+
/* empty field for null */
|
|
989
|
+
continue;
|
|
990
|
+
}
|
|
991
|
+
switch (in->col_types[c]) {
|
|
992
|
+
case TF_TYPE_BOOL:
|
|
993
|
+
tf_buffer_write_str(out, tf_batch_get_bool(in, r, c) ? "true" : "false");
|
|
994
|
+
break;
|
|
995
|
+
case TF_TYPE_INT64:
|
|
996
|
+
snprintf(numbuf, sizeof(numbuf), "%lld", (long long)tf_batch_get_int64(in, r, c));
|
|
997
|
+
tf_buffer_write_str(out, numbuf);
|
|
998
|
+
break;
|
|
999
|
+
case TF_TYPE_FLOAT64:
|
|
1000
|
+
snprintf(numbuf, sizeof(numbuf), "%g", tf_batch_get_float64(in, r, c));
|
|
1001
|
+
tf_buffer_write_str(out, numbuf);
|
|
1002
|
+
break;
|
|
1003
|
+
case TF_TYPE_STRING:
|
|
1004
|
+
write_field(out, tf_batch_get_string(in, r, c), st->delimiter);
|
|
1005
|
+
break;
|
|
1006
|
+
case TF_TYPE_DATE: {
|
|
1007
|
+
char dbuf2[16];
|
|
1008
|
+
tf_date_format(tf_batch_get_date(in, r, c), dbuf2, sizeof(dbuf2));
|
|
1009
|
+
tf_buffer_write_str(out, dbuf2);
|
|
1010
|
+
break;
|
|
1011
|
+
}
|
|
1012
|
+
case TF_TYPE_TIMESTAMP: {
|
|
1013
|
+
char tsbuf[40];
|
|
1014
|
+
tf_timestamp_format(tf_batch_get_timestamp(in, r, c), tsbuf, sizeof(tsbuf));
|
|
1015
|
+
tf_buffer_write_str(out, tsbuf);
|
|
1016
|
+
break;
|
|
1017
|
+
}
|
|
1018
|
+
default:
|
|
1019
|
+
break;
|
|
1020
|
+
}
|
|
1021
|
+
}
|
|
1022
|
+
tf_buffer_write(out, (const uint8_t *)"\n", 1);
|
|
1023
|
+
}
|
|
1024
|
+
|
|
1025
|
+
return TF_OK;
|
|
1026
|
+
}
|
|
1027
|
+
|
|
1028
|
+
static int csv_encoder_flush(tf_encoder *self, tf_buffer *out) {
|
|
1029
|
+
(void)self; (void)out;
|
|
1030
|
+
return TF_OK;
|
|
1031
|
+
}
|
|
1032
|
+
|
|
1033
|
+
static void csv_encoder_destroy(tf_encoder *self) {
|
|
1034
|
+
free(self->state);
|
|
1035
|
+
free(self);
|
|
1036
|
+
}
|
|
1037
|
+
|
|
1038
|
+
tf_encoder *tf_csv_encoder_create(const cJSON *args) {
|
|
1039
|
+
csv_encoder_state *st = calloc(1, sizeof(csv_encoder_state));
|
|
1040
|
+
if (!st) return NULL;
|
|
1041
|
+
|
|
1042
|
+
st->delimiter = ',';
|
|
1043
|
+
st->header_written = 0;
|
|
1044
|
+
|
|
1045
|
+
if (args) {
|
|
1046
|
+
cJSON *d = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
|
|
1047
|
+
if (cJSON_IsString(d) && d->valuestring[0])
|
|
1048
|
+
st->delimiter = d->valuestring[0];
|
|
1049
|
+
}
|
|
1050
|
+
|
|
1051
|
+
tf_encoder *enc = malloc(sizeof(tf_encoder));
|
|
1052
|
+
if (!enc) { free(st); return NULL; }
|
|
1053
|
+
enc->encode = csv_encode;
|
|
1054
|
+
enc->flush = csv_encoder_flush;
|
|
1055
|
+
enc->destroy = csv_encoder_destroy;
|
|
1056
|
+
enc->state = st;
|
|
1057
|
+
return enc;
|
|
1058
|
+
}
|