tranfi 0.0.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/LICENSE +177 -21
  2. package/NOTICE +8 -0
  3. package/README.md +627 -0
  4. package/app/assets/index-6quYZ5Ap.css +5 -0
  5. package/app/assets/index-BIAIKnrp.js +160 -0
  6. package/app/assets/materialdesignicons-webfont-B7mPwVP_.ttf +0 -0
  7. package/app/assets/materialdesignicons-webfont-CSr8KVlo.eot +0 -0
  8. package/app/assets/materialdesignicons-webfont-Dp5v-WZN.woff2 +0 -0
  9. package/app/assets/materialdesignicons-webfont-PXm3-2wK.woff +0 -0
  10. package/app/index.html +13 -0
  11. package/binding.gyp +121 -0
  12. package/csrc/arena.c +93 -0
  13. package/csrc/batch.c +976 -0
  14. package/csrc/buffer.c +154 -0
  15. package/csrc/cJSON.c +3386 -0
  16. package/csrc/cJSON.h +316 -0
  17. package/csrc/codec_csv.c +1951 -0
  18. package/csrc/codec_jsonl.c +1086 -0
  19. package/csrc/codec_table.c +248 -0
  20. package/csrc/codec_text.c +447 -0
  21. package/csrc/compiler.c +130 -0
  22. package/csrc/config.h +21 -0
  23. package/csrc/date_utils.h +94 -0
  24. package/csrc/dsl.c +5417 -0
  25. package/csrc/dsl.h +22 -0
  26. package/csrc/expr.c +1553 -0
  27. package/csrc/expr.h +58 -0
  28. package/csrc/internal.h +539 -0
  29. package/csrc/ir.c +166 -0
  30. package/csrc/ir.h +208 -0
  31. package/csrc/ir_schema.c +75 -0
  32. package/csrc/ir_serialize.c +166 -0
  33. package/csrc/ir_sql.c +1822 -0
  34. package/csrc/ir_validate.c +576 -0
  35. package/csrc/json_path.c +210 -0
  36. package/csrc/main.c +1241 -0
  37. package/csrc/memory_estimate.c +477 -0
  38. package/csrc/op_acf.c +283 -0
  39. package/csrc/op_across.c +477 -0
  40. package/csrc/op_anomaly.c +255 -0
  41. package/csrc/op_assert.c +761 -0
  42. package/csrc/op_bin.c +248 -0
  43. package/csrc/op_cast.c +523 -0
  44. package/csrc/op_clip.c +99 -0
  45. package/csrc/op_date_trunc.c +355 -0
  46. package/csrc/op_datetime.c +394 -0
  47. package/csrc/op_derive.c +216 -0
  48. package/csrc/op_diff.c +250 -0
  49. package/csrc/op_ewma.c +222 -0
  50. package/csrc/op_explode.c +206 -0
  51. package/csrc/op_fill_down.c +235 -0
  52. package/csrc/op_fill_null.c +268 -0
  53. package/csrc/op_filter.c +181 -0
  54. package/csrc/op_frequency.c +721 -0
  55. package/csrc/op_grep.c +181 -0
  56. package/csrc/op_group_agg.c +1956 -0
  57. package/csrc/op_hash.c +159 -0
  58. package/csrc/op_head.c +84 -0
  59. package/csrc/op_interpolate.c +445 -0
  60. package/csrc/op_join.c +2902 -0
  61. package/csrc/op_json_extract.c +227 -0
  62. package/csrc/op_json_filter.c +384 -0
  63. package/csrc/op_json_flatten.c +293 -0
  64. package/csrc/op_json_schema.c +503 -0
  65. package/csrc/op_label_encode.c +419 -0
  66. package/csrc/op_lag.c +181 -0
  67. package/csrc/op_lead.c +242 -0
  68. package/csrc/op_normalize.c +510 -0
  69. package/csrc/op_onehot.c +457 -0
  70. package/csrc/op_pivot.c +1754 -0
  71. package/csrc/op_quarantine.c +189 -0
  72. package/csrc/op_registry.c +3044 -0
  73. package/csrc/op_rename.c +129 -0
  74. package/csrc/op_replace.c +354 -0
  75. package/csrc/op_rleid.c +297 -0
  76. package/csrc/op_rowid.c +559 -0
  77. package/csrc/op_sample.c +158 -0
  78. package/csrc/op_schema.c +1341 -0
  79. package/csrc/op_schema_infer.c +252 -0
  80. package/csrc/op_select.c +340 -0
  81. package/csrc/op_set.c +3449 -0
  82. package/csrc/op_skip.c +95 -0
  83. package/csrc/op_sort.c +819 -0
  84. package/csrc/op_source_name.c +120 -0
  85. package/csrc/op_split.c +151 -0
  86. package/csrc/op_split_data.c +119 -0
  87. package/csrc/op_stack.c +271 -0
  88. package/csrc/op_stats.c +875 -0
  89. package/csrc/op_step.c +333 -0
  90. package/csrc/op_tail.c +105 -0
  91. package/csrc/op_tee.c +338 -0
  92. package/csrc/op_top.c +357 -0
  93. package/csrc/op_trim.c +138 -0
  94. package/csrc/op_unique.c +1343 -0
  95. package/csrc/op_unpivot.c +193 -0
  96. package/csrc/op_validate.c +648 -0
  97. package/csrc/op_window.c +591 -0
  98. package/csrc/path_policy.c +85 -0
  99. package/csrc/pipeline.c +1088 -0
  100. package/csrc/recipes.c +104 -0
  101. package/csrc/recipes.h +27 -0
  102. package/csrc/report.c +506 -0
  103. package/csrc/report.h +22 -0
  104. package/csrc/selector.c +1097 -0
  105. package/csrc/size_utils.c +348 -0
  106. package/csrc/spill.c +317 -0
  107. package/csrc/spill.h +21 -0
  108. package/csrc/tranfi.h +291 -0
  109. package/csrc/transform.h +209 -0
  110. package/csrc/transform_api.c +2237 -0
  111. package/csrc/transform_categorical.c +923 -0
  112. package/csrc/transform_internal.h +472 -0
  113. package/csrc/transform_json.c +3812 -0
  114. package/csrc/transform_numeric.c +1966 -0
  115. package/csrc/transform_sha256.c +154 -0
  116. package/csrc/transform_wasm.h +162 -0
  117. package/csrc/transform_wasm_api.c +1373 -0
  118. package/csrc/wasm_api.c +218 -0
  119. package/napi_api.c +534 -0
  120. package/napi_transform.c +1648 -0
  121. package/napi_transform.h +8 -0
  122. package/package.json +64 -59
  123. package/scripts/install-native.js +76 -0
  124. package/scripts/prepack.js +64 -0
  125. package/scripts/sync-csrc.js +23 -0
  126. package/src/cli.js +190 -0
  127. package/src/engines/duckdb.js +142 -0
  128. package/src/index.js +925 -0
  129. package/src/memory_policy.js +411 -0
  130. package/src/native.js +18 -0
  131. package/src/pipeline.js +709 -0
  132. package/src/recipe_json.js +80 -0
  133. package/src/server.js +279 -0
  134. package/src/transform.js +403 -0
  135. package/src/transform_error.js +10 -0
  136. package/src/wasm.js +21 -0
  137. package/wasm/index.js +732 -0
  138. package/wasm/package.json +1 -0
  139. package/wasm/tranfi_core.js +0 -0
  140. package/wasm/transform.js +1156 -0
  141. package/wasm/worker.js +786 -0
  142. package/dist/bundle.js +0 -1
  143. package/index.html +0 -18
  144. package/src/app.css +0 -169
  145. package/src/app.js +0 -203
  146. package/src/app.vue +0 -250
  147. package/src/bulma-input.vue +0 -110
  148. package/src/common-inputs.js +0 -28
  149. package/src/main.js +0 -20
  150. package/src/transforms.js +0 -166
  151. package/webpack.config.js +0 -108
@@ -0,0 +1,1951 @@
1
+ /*
2
+ * codec_csv.c — Optimized streaming CSV decoder and encoder.
3
+ *
4
+ * Decoder design (three key optimizations):
5
+ *
6
+ * 1. Zero-copy field parsing: fields are returned as (ptr, len) slices
7
+ * into the line buffer, avoiding per-field malloc/free. Only quoted
8
+ * fields with escaped quotes ("") need copying (rare in practice).
9
+ *
10
+ * 2. Type detection window: the first batch (batch_size rows) detects
11
+ * column types via progressive widening (NULL → BOOL / INT64 →
12
+ * FLOAT64 / DATE / TIMESTAMP → STRING). Types freeze after the
13
+ * first batch. This matches Arrow CSV, DuckDB, and other production parsers.
14
+ *
15
+ * 3. Direct-to-typed parsing: after types freeze, field slices are parsed
16
+ * directly into typed column arrays (int64, double, string) without
17
+ * an intermediate STRING batch. Combined with custom fast_int64/
18
+ * fast_double parsers, this eliminates double parsing for >99% of rows.
19
+ *
20
+ * Encoder: writes typed batches as RFC 4180 CSV with proper quoting.
21
+ */
22
+
23
+ #include "internal.h"
24
+ #include "date_utils.h"
25
+ #include "cJSON.h"
26
+ #include <stdlib.h>
27
+ #include <string.h>
28
+ #include <stdio.h>
29
+ #include <errno.h>
30
+ #include <limits.h>
31
+ #include <math.h>
32
+
33
+ #define DEFAULT_BATCH_SIZE 1024
34
+ #define DEFAULT_MAX_ERROR_BYTES 4096
35
+ #define DEFAULT_MAX_RECORD_BYTES (64 * 1024 * 1024)
36
+ #define DEFAULT_MAX_COLUMNS 8192
37
+
38
+ /* ================================================================
39
+ * Field Slice — zero-copy reference into the line buffer
40
+ * ================================================================ */
41
+
42
+ typedef struct {
43
+ const char *ptr; /* points into line buffer (or field_arena for escapes) */
44
+ size_t len;
45
+ int quoted;
46
+ } field_slice;
47
+
48
+ /* ================================================================
49
+ * Fast Numeric Parsers
50
+ *
51
+ * These avoid libc strtoll/strtod overhead for common number formats.
52
+ * They take (ptr, len) instead of null-terminated strings, which
53
+ * pairs naturally with the zero-copy field slices.
54
+ * ================================================================ */
55
+
56
+ /*
57
+ * Fast int64 parser. Handles [-+]digits format only.
58
+ * Rejects decimal points, exponents, leading zeros (except "0"),
59
+ * and values outside the int64 range.
60
+ */
61
+ static int fast_int64(const char *s, size_t len, int64_t *out) {
62
+ if (len == 0) return 0;
63
+
64
+ size_t i = 0;
65
+ int neg = 0;
66
+ if (s[0] == '-') { neg = 1; i = 1; }
67
+ else if (s[0] == '+') { i = 1; }
68
+
69
+ /* Must have at least one digit */
70
+ if (i >= len || s[i] < '0' || s[i] > '9') return 0;
71
+
72
+ /* Max int64 is 19 digits; reject longer numbers to avoid overflow */
73
+ size_t n_digits = len - i;
74
+ if (n_digits > 19) return 0;
75
+
76
+ uint64_t v = 0;
77
+ for (; i < len; i++) {
78
+ if (s[i] < '0' || s[i] > '9') return 0;
79
+ v = v * 10 + (uint64_t)(s[i] - '0');
80
+ }
81
+
82
+ /* Range check against int64 bounds */
83
+ if (neg) {
84
+ /* INT64_MIN magnitude is INT64_MAX + 1 */
85
+ if (v > (uint64_t)INT64_MAX + 1) return 0;
86
+ *out = (v == (uint64_t)INT64_MAX + 1) ? INT64_MIN : -(int64_t)v;
87
+ } else {
88
+ if (v > (uint64_t)INT64_MAX) return 0;
89
+ *out = (int64_t)v;
90
+ }
91
+ return 1;
92
+ }
93
+
94
+ /*
95
+ * Fast double parser for common decimal formats: [-+]digits[.digits]
96
+ *
97
+ * Uses integer accumulation + power-of-10 division for precision.
98
+ * Uses a short-decimal fast path and falls back to strtod for values that need correct rounding.
99
+ * Falls back to strtod for exponents (e/E), special values, or
100
+ * very long mantissas.
101
+ */
102
+ static int fast_double(const char *s, size_t len, double *out) {
103
+ if (len == 0) return 0;
104
+
105
+ size_t i = 0;
106
+ int neg = 0;
107
+ if (s[0] == '-') { neg = 1; i = 1; }
108
+ else if (s[0] == '+') { i = 1; }
109
+ if (i >= len) return 0;
110
+
111
+ /* Accumulate mantissa as integer: 123.456 → mantissa=123456, n_frac=3 */
112
+ uint64_t mantissa = 0;
113
+ int n_digits = 0;
114
+ int n_frac = 0;
115
+
116
+ /* Integer part */
117
+ while (i < len && s[i] >= '0' && s[i] <= '9') {
118
+ mantissa = mantissa * 10 + (uint64_t)(s[i] - '0');
119
+ n_digits++;
120
+ i++;
121
+ }
122
+
123
+ /* Fractional part */
124
+ if (i < len && s[i] == '.') {
125
+ i++;
126
+ while (i < len && s[i] >= '0' && s[i] <= '9') {
127
+ mantissa = mantissa * 10 + (uint64_t)(s[i] - '0');
128
+ n_frac++;
129
+ n_digits++;
130
+ i++;
131
+ }
132
+ }
133
+
134
+ if (n_digits == 0) return 0;
135
+
136
+ /* Fast path: no exponent and short mantissas. Longer decimals need strtod's correct rounding. */
137
+ if (i == len && n_digits <= 15) {
138
+ static const double pow10[] = {
139
+ 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9,
140
+ 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18
141
+ };
142
+ double result = (double)mantissa;
143
+ if (n_frac > 0) result /= pow10[n_frac];
144
+ *out = neg ? -result : result;
145
+ return 1;
146
+ }
147
+
148
+ /* Fallback to strtod for exponents, very long numbers, etc. */
149
+ if (len >= 64) return 0;
150
+ char buf[64];
151
+ memcpy(buf, s, len);
152
+ buf[len] = '\0';
153
+ char *end;
154
+ errno = 0;
155
+ *out = strtod(buf, &end);
156
+ if ((size_t)(end - buf) != len) return 0;
157
+ if (errno == ERANGE) {
158
+ if (!isfinite(*out) || *out == 0.0) return 0;
159
+ } else if (errno) {
160
+ return 0;
161
+ }
162
+ return 1;
163
+ }
164
+
165
+ /*
166
+ * Fast date parser: exactly YYYY-MM-DD (10 chars) → int32_t days since epoch.
167
+ * Returns 1 on success, 0 on failure.
168
+ */
169
+ static int fast_date(const char *s, size_t len, int32_t *out) {
170
+ if (len != 10) return 0;
171
+ if (s[4] != '-' || s[7] != '-') return 0;
172
+ /* Parse YYYY */
173
+ int y = 0;
174
+ for (int i = 0; i < 4; i++) {
175
+ if (s[i] < '0' || s[i] > '9') return 0;
176
+ y = y * 10 + (s[i] - '0');
177
+ }
178
+ /* Parse MM */
179
+ int m = 0;
180
+ for (int i = 5; i < 7; i++) {
181
+ if (s[i] < '0' || s[i] > '9') return 0;
182
+ m = m * 10 + (s[i] - '0');
183
+ }
184
+ /* Parse DD */
185
+ int d = 0;
186
+ for (int i = 8; i < 10; i++) {
187
+ if (s[i] < '0' || s[i] > '9') return 0;
188
+ d = d * 10 + (s[i] - '0');
189
+ }
190
+ if (m < 1 || m > 12 || d < 1 || d > 31) return 0;
191
+ *out = tf_date_from_ymd(y, m, d);
192
+ return 1;
193
+ }
194
+
195
+ /*
196
+ * Fast timestamp parser: YYYY-MM-DD[T ]HH:MM:SS[.ffffff][Z|+HH:MM|-HH:MM]
197
+ * Returns 1 on success, 0 on failure.
198
+ */
199
+ static int fast_timestamp(const char *s, size_t len, int64_t *out) {
200
+ if (len < 19) return 0;
201
+ if (s[4] != '-' || s[7] != '-') return 0;
202
+ if (s[10] != 'T' && s[10] != ' ') return 0;
203
+ if (s[13] != ':' || s[16] != ':') return 0;
204
+
205
+ int y = 0, mo = 0, d = 0, h = 0, mi = 0, se = 0;
206
+ /* YYYY */
207
+ for (int i = 0; i < 4; i++) {
208
+ if (s[i] < '0' || s[i] > '9') return 0;
209
+ y = y * 10 + (s[i] - '0');
210
+ }
211
+ /* MM */
212
+ for (int i = 5; i < 7; i++) {
213
+ if (s[i] < '0' || s[i] > '9') return 0;
214
+ mo = mo * 10 + (s[i] - '0');
215
+ }
216
+ /* DD */
217
+ for (int i = 8; i < 10; i++) {
218
+ if (s[i] < '0' || s[i] > '9') return 0;
219
+ d = d * 10 + (s[i] - '0');
220
+ }
221
+ /* HH */
222
+ for (int i = 11; i < 13; i++) {
223
+ if (s[i] < '0' || s[i] > '9') return 0;
224
+ h = h * 10 + (s[i] - '0');
225
+ }
226
+ /* MM */
227
+ for (int i = 14; i < 16; i++) {
228
+ if (s[i] < '0' || s[i] > '9') return 0;
229
+ mi = mi * 10 + (s[i] - '0');
230
+ }
231
+ /* SS */
232
+ for (int i = 17; i < 19; i++) {
233
+ if (s[i] < '0' || s[i] > '9') return 0;
234
+ se = se * 10 + (s[i] - '0');
235
+ }
236
+ if (mo < 1 || mo > 12 || d < 1 || d > 31) return 0;
237
+ if (h > 23 || mi > 59 || se > 59) return 0;
238
+
239
+ /* Optional fractional seconds */
240
+ int frac_us = 0;
241
+ size_t pos = 19;
242
+ if (pos < len && s[pos] == '.') {
243
+ pos++;
244
+ int frac_digits = 0;
245
+ int frac_val = 0;
246
+ while (pos < len && s[pos] >= '0' && s[pos] <= '9' && frac_digits < 6) {
247
+ frac_val = frac_val * 10 + (s[pos] - '0');
248
+ frac_digits++;
249
+ pos++;
250
+ }
251
+ /* Skip remaining digits beyond 6 */
252
+ while (pos < len && s[pos] >= '0' && s[pos] <= '9') pos++;
253
+ /* Pad to 6 digits */
254
+ while (frac_digits < 6) { frac_val *= 10; frac_digits++; }
255
+ frac_us = frac_val;
256
+ }
257
+
258
+ /* Optional timezone */
259
+ int64_t tz_offset_us = 0;
260
+ if (pos < len) {
261
+ if (s[pos] == 'Z') {
262
+ pos++;
263
+ } else if (s[pos] == '+' || s[pos] == '-') {
264
+ int tz_sign = (s[pos] == '-') ? -1 : 1;
265
+ pos++;
266
+ if (pos + 2 > len) return 0;
267
+ int tz_h = (s[pos] - '0') * 10 + (s[pos + 1] - '0');
268
+ pos += 2;
269
+ int tz_m = 0;
270
+ if (pos < len && s[pos] == ':') {
271
+ pos++;
272
+ if (pos + 2 > len) return 0;
273
+ tz_m = (s[pos] - '0') * 10 + (s[pos + 1] - '0');
274
+ pos += 2;
275
+ }
276
+ tz_offset_us = tz_sign * ((int64_t)tz_h * 3600000000LL + (int64_t)tz_m * 60000000LL);
277
+ }
278
+ }
279
+
280
+ if (pos != len) return 0;
281
+
282
+ *out = tf_timestamp_from_parts(y, mo, d, h, mi, se, frac_us) - tz_offset_us;
283
+ return 1;
284
+ }
285
+
286
+ static int ascii_lower_char(int ch) {
287
+ return (ch >= 'A' && ch <= 'Z') ? ch + ('a' - 'A') : ch;
288
+ }
289
+
290
+ static int slice_ieq(const char *s, const char *lit, size_t len) {
291
+ for (size_t i = 0; i < len; i++) {
292
+ if (ascii_lower_char((unsigned char)s[i]) != (unsigned char)lit[i]) return 0;
293
+ }
294
+ return 1;
295
+ }
296
+
297
+ static int fast_bool(const char *s, size_t len, int *out) {
298
+ if (len == 4 && slice_ieq(s, "true", 4)) {
299
+ if (out) *out = 1;
300
+ return 1;
301
+ }
302
+ if (len == 5 && slice_ieq(s, "false", 5)) {
303
+ if (out) *out = 0;
304
+ return 1;
305
+ }
306
+ return 0;
307
+ }
308
+
309
+ /* Detect the type of a field slice without copying it. */
310
+ static tf_type detect_type_slice(const char *s, size_t len) {
311
+ if (len == 0) return TF_TYPE_NULL;
312
+ int64_t iv;
313
+ double fv;
314
+ int32_t dv;
315
+ if (fast_bool(s, len, NULL)) return TF_TYPE_BOOL;
316
+ if (fast_int64(s, len, &iv)) return TF_TYPE_INT64;
317
+ if (fast_double(s, len, &fv)) return TF_TYPE_FLOAT64;
318
+ if (fast_date(s, len, &dv)) return TF_TYPE_DATE;
319
+ if (fast_timestamp(s, len, &iv)) return TF_TYPE_TIMESTAMP;
320
+ return TF_TYPE_STRING;
321
+ }
322
+
323
+ /* Widen a column type if needed. */
324
+ static tf_type widen_type(tf_type current, tf_type incoming) {
325
+ if (current == incoming) return current;
326
+ if (current == TF_TYPE_NULL) return incoming;
327
+ if (incoming == TF_TYPE_NULL) return current;
328
+ if (current == TF_TYPE_INT64 && incoming == TF_TYPE_FLOAT64) return TF_TYPE_FLOAT64;
329
+ if (current == TF_TYPE_FLOAT64 && incoming == TF_TYPE_INT64) return TF_TYPE_FLOAT64;
330
+ if ((current == TF_TYPE_DATE && incoming == TF_TYPE_TIMESTAMP) ||
331
+ (current == TF_TYPE_TIMESTAMP && incoming == TF_TYPE_DATE))
332
+ return TF_TYPE_TIMESTAMP;
333
+ return TF_TYPE_STRING; /* anything else → string */
334
+ }
335
+
336
+ /* ================================================================
337
+ * Zero-Copy Field Parser
338
+ *
339
+ * Parses a CSV line into an array of field_slice structs. For most
340
+ * fields, the slice points directly into the line buffer (zero copy).
341
+ * Only quoted fields with escaped quotes ("") need allocation, which
342
+ * goes into field_arena and is valid until arena reset.
343
+ * ================================================================ */
344
+
345
+ /*
346
+ * Parse a CSV line into field slices. Returns the number of fields.
347
+ *
348
+ * - Unquoted fields: zero-copy slice into line buffer
349
+ * - Quoted fields without "": zero-copy slice (skipping quotes)
350
+ * - Quoted fields with "": unescaped copy allocated from field_arena
351
+ * - Leading/trailing whitespace is optionally trimmed from unquoted fields
352
+ */
353
+ static int parse_csv_fields(const char *line, size_t line_len, char delim,
354
+ field_slice *fields, size_t max_fields,
355
+ tf_arena *field_arena, int trim_ws,
356
+ size_t *out_count) {
357
+ if (!out_count) return TF_ERROR;
358
+ size_t count = 0;
359
+ size_t i = 0;
360
+
361
+ while (i <= line_len) {
362
+ if (i == line_len) {
363
+ /* Trailing delimiter -> empty last field */
364
+ if (count > 0 && i > 0 && line[i - 1] == delim) {
365
+ if (count < max_fields) {
366
+ fields[count].ptr = "";
367
+ fields[count].len = 0;
368
+ fields[count].quoted = 0;
369
+ }
370
+ count++;
371
+ }
372
+ break;
373
+ }
374
+
375
+ if (line[i] == '"') {
376
+ /* --- Quoted field --- */
377
+ i++; /* skip opening quote */
378
+ size_t start = i;
379
+ int has_escape = 0;
380
+
381
+ /* Scan for closing quote, detecting escaped quotes ("") */
382
+ while (i < line_len) {
383
+ if (line[i] == '"') {
384
+ if (i + 1 < line_len && line[i + 1] == '"') {
385
+ has_escape = 1;
386
+ i += 2;
387
+ } else {
388
+ break; /* closing quote */
389
+ }
390
+ } else {
391
+ i++;
392
+ }
393
+ }
394
+ size_t field_end = i;
395
+ if (i < line_len) i++; /* skip closing quote */
396
+ if (i < line_len && line[i] == delim) i++; /* skip delimiter */
397
+
398
+ if (count < max_fields) {
399
+ if (!has_escape) {
400
+ /* Zero-copy: slice directly into line buffer, past the quotes */
401
+ fields[count].ptr = line + start;
402
+ fields[count].len = field_end - start;
403
+ fields[count].quoted = 1;
404
+ } else {
405
+ /* Rare path: unescape "" -> " into arena-allocated buffer */
406
+ size_t max_len = field_end - start; /* unescaped is always shorter */
407
+ size_t bytes = 0;
408
+ if (tf_check_byte_limit(max_len, TF_MAX_CELL_BYTES,
409
+ "csv", "cell") != TF_OK ||
410
+ tf_size_add(max_len, 1, &bytes) != TF_OK)
411
+ return TF_ERROR;
412
+ char *buf = tf_arena_alloc(field_arena, bytes);
413
+ if (!buf) return TF_ERROR;
414
+ size_t out_len = 0;
415
+ for (size_t j = start; j < field_end; j++) {
416
+ if (line[j] == '"' && j + 1 < field_end && line[j + 1] == '"') {
417
+ buf[out_len++] = '"';
418
+ j++; /* skip second quote */
419
+ } else {
420
+ buf[out_len++] = line[j];
421
+ }
422
+ }
423
+ buf[out_len] = '\0';
424
+ fields[count].ptr = buf;
425
+ fields[count].len = out_len;
426
+ fields[count].quoted = 1;
427
+ }
428
+ }
429
+ count++;
430
+ } else {
431
+ /* --- Unquoted field: zero-copy slice with whitespace trimming --- */
432
+ size_t start = i;
433
+ while (i < line_len && line[i] != delim) i++;
434
+
435
+ const char *fptr = line + start;
436
+ size_t flen = i - start;
437
+
438
+ if (trim_ws) {
439
+ /* Trim trailing ASCII whitespace */
440
+ while (flen > 0 && (fptr[flen - 1] == ' ' || fptr[flen - 1] == '\t'))
441
+ flen--;
442
+ /* Trim leading ASCII whitespace */
443
+ while (flen > 0 && (*fptr == ' ' || *fptr == '\t'))
444
+ { fptr++; flen--; }
445
+ }
446
+
447
+ if (count < max_fields) {
448
+ fields[count].ptr = fptr;
449
+ fields[count].len = flen;
450
+ fields[count].quoted = 0;
451
+ }
452
+ count++;
453
+
454
+ if (i < line_len) i++; /* skip delimiter */
455
+ }
456
+ }
457
+
458
+ *out_count = count;
459
+ return TF_OK;
460
+ }
461
+
462
+ /* ================================================================
463
+ * CSV Decoder
464
+ * ================================================================ */
465
+
466
+ typedef enum {
467
+ CSV_MODE_PERMISSIVE = 0,
468
+ CSV_MODE_REPAIR,
469
+ CSV_MODE_STRICT,
470
+ } csv_decode_mode;
471
+
472
+ typedef struct {
473
+ char delimiter;
474
+ int has_header;
475
+ int skip_repeated_header;
476
+ int after_input_boundary;
477
+ size_t batch_size;
478
+ csv_decode_mode mode;
479
+ char **null_literals;
480
+ size_t n_null_literals;
481
+ int quoted_nulls;
482
+ int trim_ws;
483
+ int skip_empty_rows;
484
+ size_t skip_rows;
485
+ size_t skipped_rows;
486
+ int limit_rows;
487
+ size_t n_max;
488
+ size_t data_rows_read;
489
+ char *comment;
490
+ size_t comment_len;
491
+ size_t max_error_bytes;
492
+ size_t max_record_bytes;
493
+ size_t max_columns;
494
+ size_t audit_limit;
495
+ size_t audit_emitted;
496
+ int audit;
497
+ tf_audit_options audit_opts;
498
+ size_t line_number;
499
+ size_t byte_offset;
500
+
501
+ /* Line accumulator: incoming bytes are appended, complete lines extracted */
502
+ tf_buffer line_buf;
503
+
504
+ /* Schema (discovered from first row) */
505
+ char **col_names;
506
+ tf_type *col_types;
507
+ size_t n_cols;
508
+ int schema_ready;
509
+
510
+ /* After the first batch, types freeze and we parse directly to typed
511
+ * columns. This avoids double parsing for >99% of rows. */
512
+ int types_frozen;
513
+
514
+ /* Current batch being built */
515
+ tf_batch *batch;
516
+ size_t rows_buffered;
517
+
518
+ /* Reusable per-line scratch: field slices array and arena for escapes */
519
+ field_slice *fields;
520
+ size_t fields_cap;
521
+ tf_arena *field_arena;
522
+ } csv_decoder_state;
523
+
524
+ static const char *csv_mode_name(csv_decode_mode mode) {
525
+ switch (mode) {
526
+ case CSV_MODE_REPAIR: return "repair";
527
+ case CSV_MODE_STRICT: return "strict";
528
+ case CSV_MODE_PERMISSIVE:
529
+ default: return "permissive";
530
+ }
531
+ }
532
+
533
+ static int parse_csv_mode(const char *s, csv_decode_mode *out) {
534
+ if (!s || strcmp(s, "permissive") == 0 || strcmp(s, "lenient") == 0) {
535
+ *out = CSV_MODE_PERMISSIVE;
536
+ return 1;
537
+ }
538
+ if (strcmp(s, "repair") == 0) {
539
+ *out = CSV_MODE_REPAIR;
540
+ return 1;
541
+ }
542
+ if (strcmp(s, "strict") == 0 || strcmp(s, "fail") == 0 || strcmp(s, "error") == 0) {
543
+ *out = CSV_MODE_STRICT;
544
+ return 1;
545
+ }
546
+ return 0;
547
+ }
548
+
549
+ static int csv_line_is_blank(const char *line, size_t line_len) {
550
+ for (size_t i = 0; i < line_len; i++) {
551
+ if (line[i] != ' ' && line[i] != '\t') return 0;
552
+ }
553
+ return 1;
554
+ }
555
+
556
+ static int csv_find_comment(const csv_decoder_state *st, const char *line, size_t line_len,
557
+ size_t *comment_pos) {
558
+ if (!st->comment || st->comment_len == 0 || st->comment_len > line_len) return 0;
559
+ int in_quotes = 0;
560
+ int at_field_start = 1;
561
+ for (size_t i = 0; i < line_len; i++) {
562
+ if (in_quotes) {
563
+ if (line[i] == '"') {
564
+ if (i + 1 < line_len && line[i + 1] == '"') {
565
+ i++;
566
+ } else {
567
+ in_quotes = 0;
568
+ at_field_start = 0;
569
+ }
570
+ }
571
+ continue;
572
+ }
573
+ if (line[i] == '"' && at_field_start) {
574
+ in_quotes = 1;
575
+ at_field_start = 0;
576
+ continue;
577
+ }
578
+ if (i + st->comment_len <= line_len &&
579
+ memcmp(line + i, st->comment, st->comment_len) == 0) {
580
+ *comment_pos = i;
581
+ return 1;
582
+ }
583
+ if (line[i] == st->delimiter) at_field_start = 1;
584
+ else at_field_start = 0;
585
+ }
586
+ return 0;
587
+ }
588
+
589
+ static char *csv_raw_preview(const csv_decoder_state *st, const char *line, size_t line_len,
590
+ int *truncated_out) {
591
+ size_t keep = line_len;
592
+ int truncated = 0;
593
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
594
+ keep = st->max_error_bytes;
595
+ truncated = 1;
596
+ }
597
+ char *raw = malloc(keep + 1);
598
+ if (!raw) return NULL;
599
+ memcpy(raw, line, keep);
600
+ raw[keep] = '\0';
601
+ if (truncated_out) *truncated_out = truncated;
602
+ return raw;
603
+ }
604
+
605
+ static int csv_add_raw_payload(csv_decoder_state *st, cJSON *obj, const char *raw, int truncated) {
606
+ if (st && st->audit_opts.include_row) {
607
+ char *safe_raw = tf_audit_format_string_dup_for_column(&st->audit_opts, "raw", raw);
608
+ if (!safe_raw) return TF_ERROR;
609
+ if (st->audit_opts.max_bytes > 0 && strlen(safe_raw) > st->audit_opts.max_bytes) {
610
+ if (tf_json_add_bool(obj, "_audit_truncated", 1) != TF_OK ||
611
+ tf_json_add_number(obj, "max_bytes", (double)st->audit_opts.max_bytes) != TF_OK) {
612
+ free(safe_raw);
613
+ return TF_ERROR;
614
+ }
615
+ } else {
616
+ if (tf_json_add_string(obj, "raw", safe_raw) != TF_OK) {
617
+ free(safe_raw);
618
+ return TF_ERROR;
619
+ }
620
+ }
621
+ free(safe_raw);
622
+ }
623
+ if (truncated && tf_json_add_bool(obj, "truncated", 1) != TF_OK) return TF_ERROR;
624
+ return TF_OK;
625
+ }
626
+
627
+ static int emit_csv_repair_audit(csv_decoder_state *st, const char *line, size_t line_len,
628
+ size_t line_no, size_t byte_offset, size_t n_fields,
629
+ const char *message, tf_side_channels *side) {
630
+ if (!st->audit || st->audit_emitted >= st->audit_limit || !side || !side->stats) return TF_OK;
631
+ int truncated = 0;
632
+ char *raw = csv_raw_preview(st, line, line_len, &truncated);
633
+ if (!raw) return TF_ERROR;
634
+ cJSON *obj = cJSON_CreateObject();
635
+ if (!obj) { free(raw); return TF_ERROR; }
636
+ int rc = TF_ERROR;
637
+ if (tf_json_add_string(obj, "type", "audit") != TF_OK ||
638
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
639
+ tf_json_add_string(obj, "event", "row_repaired") != TF_OK ||
640
+ tf_json_add_string(obj, "reason", "csv_field_count") != TF_OK ||
641
+ tf_json_add_string(obj, "channel", "audit") != TF_OK ||
642
+ tf_json_add_string(obj, "mode", csv_mode_name(st->mode)) != TF_OK ||
643
+ tf_json_add_string(obj, "action", "repair") != TF_OK ||
644
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
645
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
646
+ tf_json_add_number(obj, "expected_fields", (double)st->n_cols) != TF_OK ||
647
+ tf_json_add_number(obj, "actual_fields", (double)n_fields) != TF_OK ||
648
+ tf_json_add_string(obj, "message", message ? message : "CSV row field count differs from header") != TF_OK ||
649
+ tf_json_add_number(obj, "raw_bytes", (double)line_len) != TF_OK ||
650
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
651
+ goto done;
652
+ }
653
+ rc = tf_buffer_write_json_line(side->stats, obj);
654
+ done:
655
+ cJSON_Delete(obj);
656
+ free(raw);
657
+ if (rc == TF_OK) st->audit_emitted++;
658
+ return rc;
659
+ }
660
+
661
+ static int emit_csv_field_count_diagnostic(csv_decoder_state *st, const char *line, size_t line_len,
662
+ size_t line_no, size_t byte_offset, size_t n_fields,
663
+ const char *action, const char *severity,
664
+ const char *message, tf_side_channels *side) {
665
+ if (!side || !side->errors) return TF_OK;
666
+
667
+ size_t keep = line_len;
668
+ int truncated = 0;
669
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
670
+ keep = st->max_error_bytes;
671
+ truncated = 1;
672
+ }
673
+
674
+ char *raw = malloc(keep + 1);
675
+ if (!raw) return TF_ERROR;
676
+ memcpy(raw, line, keep);
677
+ raw[keep] = '\0';
678
+
679
+ cJSON *obj = cJSON_CreateObject();
680
+ if (!obj) { free(raw); return TF_ERROR; }
681
+ int rc = TF_ERROR;
682
+ if (tf_json_add_string(obj, "type", "csv_field_count") != TF_OK ||
683
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
684
+ tf_json_add_string(obj, "mode", csv_mode_name(st->mode)) != TF_OK ||
685
+ tf_json_add_string(obj, "action", action) != TF_OK ||
686
+ tf_json_add_string(obj, "severity", severity) != TF_OK ||
687
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
688
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
689
+ tf_json_add_number(obj, "expected_fields", (double)st->n_cols) != TF_OK ||
690
+ tf_json_add_number(obj, "actual_fields", (double)n_fields) != TF_OK ||
691
+ tf_json_add_string(obj, "message", message ? message : "CSV row field count differs from header") != TF_OK ||
692
+ tf_json_add_number(obj, "raw_bytes", (double)line_len) != TF_OK ||
693
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
694
+ goto done;
695
+ }
696
+ rc = tf_buffer_write_json_line(side->errors, obj);
697
+ done:
698
+ cJSON_Delete(obj);
699
+ free(raw);
700
+ return rc;
701
+ }
702
+
703
+ static int emit_csv_column_limit_diagnostic(csv_decoder_state *st, const char *line, size_t line_len,
704
+ size_t line_no, size_t byte_offset, size_t n_fields,
705
+ tf_side_channels *side) {
706
+ if (!side || !side->errors) return TF_OK;
707
+
708
+ size_t keep = line_len;
709
+ int truncated = 0;
710
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
711
+ keep = st->max_error_bytes;
712
+ truncated = 1;
713
+ }
714
+
715
+ char *raw = malloc(keep + 1);
716
+ if (!raw) return TF_ERROR;
717
+ memcpy(raw, line, keep);
718
+ raw[keep] = '\0';
719
+
720
+ cJSON *obj = cJSON_CreateObject();
721
+ if (!obj) { free(raw); return TF_ERROR; }
722
+ int rc = TF_ERROR;
723
+ if (tf_json_add_string(obj, "type", "csv_too_many_columns") != TF_OK ||
724
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
725
+ tf_json_add_string(obj, "mode", csv_mode_name(st->mode)) != TF_OK ||
726
+ tf_json_add_string(obj, "action", "fail") != TF_OK ||
727
+ tf_json_add_string(obj, "severity", "error") != TF_OK ||
728
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
729
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
730
+ tf_json_add_number(obj, "max_columns", (double)st->max_columns) != TF_OK ||
731
+ tf_json_add_number(obj, "actual_fields", (double)n_fields) != TF_OK ||
732
+ tf_json_add_string(obj, "message", "CSV record exceeds max_columns") != TF_OK ||
733
+ tf_json_add_number(obj, "raw_bytes", (double)line_len) != TF_OK ||
734
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
735
+ goto done;
736
+ }
737
+ rc = tf_buffer_write_json_line(side->errors, obj);
738
+ done:
739
+ cJSON_Delete(obj);
740
+ free(raw);
741
+ return rc;
742
+ }
743
+
744
+ static int emit_csv_record_size_diagnostic(csv_decoder_state *st, const char *record, size_t record_len,
745
+ size_t line_no, size_t byte_offset, tf_side_channels *side) {
746
+ if (!side || !side->errors) return TF_OK;
747
+
748
+ size_t keep = record_len;
749
+ int truncated = 0;
750
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
751
+ keep = st->max_error_bytes;
752
+ truncated = 1;
753
+ }
754
+
755
+ char *raw = malloc(keep + 1);
756
+ if (!raw) return TF_ERROR;
757
+ memcpy(raw, record, keep);
758
+ raw[keep] = '\0';
759
+
760
+ cJSON *obj = cJSON_CreateObject();
761
+ if (!obj) { free(raw); return TF_ERROR; }
762
+ int rc = TF_ERROR;
763
+ if (tf_json_add_string(obj, "type", "csv_record_too_large") != TF_OK ||
764
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
765
+ tf_json_add_string(obj, "action", "fail") != TF_OK ||
766
+ tf_json_add_string(obj, "severity", "error") != TF_OK ||
767
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
768
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
769
+ tf_json_add_number(obj, "max_record_bytes", (double)st->max_record_bytes) != TF_OK ||
770
+ tf_json_add_number(obj, "observed_bytes", (double)record_len) != TF_OK ||
771
+ tf_json_add_string(obj, "message", "CSV record exceeds max_record_bytes") != TF_OK ||
772
+ tf_json_add_number(obj, "raw_bytes", (double)record_len) != TF_OK ||
773
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
774
+ goto done;
775
+ }
776
+ rc = tf_buffer_write_json_line(side->errors, obj);
777
+ done:
778
+ cJSON_Delete(obj);
779
+ free(raw);
780
+ return rc;
781
+ }
782
+
783
+ static int check_csv_record_limit(csv_decoder_state *st, const uint8_t *buf, size_t line_start,
784
+ size_t observed_len, size_t line_no, size_t byte_offset,
785
+ tf_side_channels *side) {
786
+ if (st->max_record_bytes == 0 || observed_len <= st->max_record_bytes) return TF_OK;
787
+ if (emit_csv_record_size_diagnostic(st, (const char *)buf + line_start, observed_len,
788
+ line_no, byte_offset, side) != TF_OK)
789
+ return TF_ERROR;
790
+ char err[256];
791
+ snprintf(err, sizeof(err), "csv record exceeds max_record_bytes at line %zu: max %zu bytes, observed %zu bytes",
792
+ line_no, st->max_record_bytes, observed_len);
793
+ tf_set_last_error(err);
794
+ return TF_ERROR;
795
+ }
796
+
797
+ static int csv_add_null_literal(csv_decoder_state *st, const char *ptr, size_t len) {
798
+ size_t copy_len = 0;
799
+ if (tf_size_add(len, 1, &copy_len) != TF_OK) return TF_ERROR;
800
+ char *copy = tf_mallocarray_checked(copy_len, sizeof(char));
801
+ if (!copy) return TF_ERROR;
802
+ memcpy(copy, ptr, len);
803
+ copy[len] = '\0';
804
+ size_t next_count = 0;
805
+ if (tf_size_add(st->n_null_literals, 1, &next_count) != TF_OK) {
806
+ free(copy);
807
+ return TF_ERROR;
808
+ }
809
+ char **tmp = tf_reallocarray_checked(st->null_literals, next_count, sizeof(char *));
810
+ if (!tmp) {
811
+ free(copy);
812
+ return TF_ERROR;
813
+ }
814
+ st->null_literals = tmp;
815
+ st->null_literals[st->n_null_literals++] = copy;
816
+ return TF_OK;
817
+ }
818
+
819
+ static int csv_add_null_literal_token(csv_decoder_state *st, const char *tok) {
820
+ if (!tok) return TF_OK;
821
+ return csv_add_null_literal(st, tok, strlen(tok));
822
+ }
823
+
824
+ static void csv_free_schema_arrays(char **col_names, tf_type *col_types, size_t n_names) {
825
+ if (col_names) {
826
+ for (size_t i = 0; i < n_names; i++) free(col_names[i]);
827
+ }
828
+ free(col_names);
829
+ free(col_types);
830
+ }
831
+
832
+ static int csv_init_schema(csv_decoder_state *st, const field_slice *fields,
833
+ size_t n_fields, int synthetic_names) {
834
+ char **col_names = tf_callocarray_checked(n_fields ? n_fields : 1, sizeof(char *));
835
+ tf_type *col_types = tf_callocarray_checked(n_fields ? n_fields : 1, sizeof(tf_type));
836
+ if (!col_names || !col_types) {
837
+ free(col_names);
838
+ free(col_types);
839
+ return TF_ERROR;
840
+ }
841
+
842
+ size_t schema_bytes = 0;
843
+ for (size_t i = 0; i < n_fields; i++) {
844
+ char synthetic[64];
845
+ const char *name_ptr = fields[i].ptr;
846
+ size_t name_len = fields[i].len;
847
+ if (synthetic_names) {
848
+ int n = snprintf(synthetic, sizeof(synthetic), "col%zu", i + 1);
849
+ if (n <= 0 || (size_t)n >= sizeof(synthetic)) {
850
+ csv_free_schema_arrays(col_names, col_types, i);
851
+ return TF_ERROR;
852
+ }
853
+ name_ptr = synthetic;
854
+ name_len = (size_t)n;
855
+ }
856
+
857
+ size_t name_bytes = 0, next_schema_bytes = 0;
858
+ if (tf_check_byte_limit(name_len, TF_MAX_COLUMN_NAME_BYTES,
859
+ "csv", "column name") != TF_OK ||
860
+ tf_size_add(name_len, 1, &name_bytes) != TF_OK ||
861
+ tf_size_add(schema_bytes, name_bytes, &next_schema_bytes) != TF_OK ||
862
+ tf_check_byte_limit(next_schema_bytes, TF_MAX_SCHEMA_BYTES,
863
+ "csv", "schema") != TF_OK) {
864
+ csv_free_schema_arrays(col_names, col_types, i);
865
+ return TF_ERROR;
866
+ }
867
+ char *name = malloc(name_bytes);
868
+ if (!name) {
869
+ csv_free_schema_arrays(col_names, col_types, i);
870
+ return TF_ERROR;
871
+ }
872
+ memcpy(name, name_ptr, name_len);
873
+ name[name_len] = '\0';
874
+ col_names[i] = name;
875
+ col_types[i] = TF_TYPE_NULL;
876
+ schema_bytes = next_schema_bytes;
877
+ }
878
+
879
+ st->n_cols = n_fields;
880
+ st->col_names = col_names;
881
+ st->col_types = col_types;
882
+ st->schema_ready = 1;
883
+ st->after_input_boundary = 0;
884
+ return TF_OK;
885
+ }
886
+
887
+ static int csv_add_null_literal_list(csv_decoder_state *st, const char *list) {
888
+ if (!list) return TF_OK;
889
+ const char *start = list;
890
+ for (const char *p = list; ; p++) {
891
+ if (*p == ',' || *p == '\0') {
892
+ if (csv_add_null_literal(st, start, (size_t)(p - start)) != TF_OK) return TF_ERROR;
893
+ if (*p == '\0') break;
894
+ start = p + 1;
895
+ }
896
+ }
897
+ return TF_OK;
898
+ }
899
+
900
+ static int csv_configure_null_literals(csv_decoder_state *st, const cJSON *args) {
901
+ const cJSON *nulls = cJSON_GetObjectItemCaseSensitive(args, "nulls");
902
+ if (!nulls) nulls = cJSON_GetObjectItemCaseSensitive(args, "na");
903
+ if (!nulls) return TF_OK;
904
+
905
+ if (cJSON_IsString(nulls)) {
906
+ return csv_add_null_literal_list(st, nulls->valuestring ? nulls->valuestring : "");
907
+ }
908
+ if (cJSON_IsArray(nulls)) {
909
+ const cJSON *item = NULL;
910
+ cJSON_ArrayForEach(item, nulls) {
911
+ if (cJSON_IsString(item)) {
912
+ if (csv_add_null_literal_token(st, item->valuestring) != TF_OK) return TF_ERROR;
913
+ }
914
+ }
915
+ }
916
+ return TF_OK;
917
+ }
918
+
919
+ static int csv_field_is_null(const csv_decoder_state *st, const field_slice *field) {
920
+ if (field->quoted && !st->quoted_nulls) return 0;
921
+ if (field->len == 0) return 1;
922
+ for (size_t i = 0; i < st->n_null_literals; i++) {
923
+ const char *lit = st->null_literals[i];
924
+ size_t len = strlen(lit);
925
+ if (field->len == len && memcmp(field->ptr, lit, len) == 0) return 1;
926
+ }
927
+ return 0;
928
+ }
929
+
930
+ static tf_type detect_type_field(const csv_decoder_state *st, const field_slice *field) {
931
+ if (csv_field_is_null(st, field)) return TF_TYPE_NULL;
932
+ if (field->len == 0) return TF_TYPE_STRING;
933
+ return detect_type_slice(field->ptr, field->len);
934
+ }
935
+
936
+ /*
937
+ * Create a batch with all STRING columns (for type detection phase).
938
+ * During this phase we don't know final types yet, so everything
939
+ * is stored as strings and converted at emission time.
940
+ */
941
+ static tf_batch *make_string_batch(csv_decoder_state *st) {
942
+ tf_batch *b = tf_batch_create(st->n_cols, st->batch_size);
943
+ if (!b) return NULL;
944
+ for (size_t i = 0; i < st->n_cols; i++) {
945
+ if (tf_batch_set_schema(b, i, st->col_names[i], TF_TYPE_STRING) != TF_OK) {
946
+ tf_batch_free(b);
947
+ return NULL;
948
+ }
949
+ }
950
+ return b;
951
+ }
952
+
953
+ /*
954
+ * Create a batch with the final (frozen) column types.
955
+ * After type detection, all batches are created with correct types
956
+ * so values can be parsed directly into typed columns.
957
+ */
958
+ static tf_batch *make_typed_batch(csv_decoder_state *st) {
959
+ tf_batch *b = tf_batch_create(st->n_cols, st->batch_size);
960
+ if (!b) return NULL;
961
+ for (size_t i = 0; i < st->n_cols; i++) {
962
+ if (tf_batch_set_schema(b, i, st->col_names[i], st->col_types[i]) != TF_OK) {
963
+ tf_batch_free(b);
964
+ return NULL;
965
+ }
966
+ }
967
+ return b;
968
+ }
969
+
970
+ /*
971
+ * Add a row of field slices to a STRING-typed batch.
972
+ * Used during the type detection phase (first batch).
973
+ * Copies slice content into the batch's arena.
974
+ */
975
+ static int add_row_strings(tf_batch *b, const csv_decoder_state *st,
976
+ const field_slice *fields, size_t n_fields,
977
+ size_t n_cols) {
978
+ size_t row = b->n_rows;
979
+ size_t cols = n_cols < n_fields ? n_cols : n_fields;
980
+
981
+ for (size_t i = 0; i < cols; i++) {
982
+ if (csv_field_is_null(st, &fields[i])) {
983
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
984
+ } else {
985
+ if (tf_batch_set_string_len(b, row, i, fields[i].ptr, fields[i].len) != TF_OK)
986
+ return TF_ERROR;
987
+ }
988
+ }
989
+ /* Null-fill extra columns */
990
+ for (size_t i = cols; i < n_cols; i++) {
991
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
992
+ }
993
+ if (tf_batch_expose_row(b, row) != TF_OK) return TF_ERROR;
994
+ return TF_OK;
995
+ }
996
+
997
+ /*
998
+ * Add a row by parsing field slices directly into typed columns.
999
+ * Used after types are frozen (all batches after the first).
1000
+ *
1001
+ * Uses checked batch setters so failed writes cannot expose a partial row.
1002
+ */
1003
+ static int add_row_typed(tf_batch *b, const csv_decoder_state *st,
1004
+ const field_slice *fields, size_t n_fields,
1005
+ size_t n_cols, const tf_type *types) {
1006
+ size_t row = b->n_rows;
1007
+ size_t cols = n_cols < n_fields ? n_cols : n_fields;
1008
+
1009
+ for (size_t i = 0; i < cols; i++) {
1010
+ if (csv_field_is_null(st, &fields[i])) {
1011
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1012
+ continue;
1013
+ }
1014
+ switch (types[i]) {
1015
+ case TF_TYPE_BOOL: {
1016
+ int v;
1017
+ if (fast_bool(fields[i].ptr, fields[i].len, &v)) {
1018
+ if (tf_batch_set_bool(b, row, i, v != 0) != TF_OK) return TF_ERROR;
1019
+ } else {
1020
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1021
+ }
1022
+ break;
1023
+ }
1024
+ case TF_TYPE_INT64: {
1025
+ int64_t v;
1026
+ if (fast_int64(fields[i].ptr, fields[i].len, &v)) {
1027
+ if (tf_batch_set_int64(b, row, i, v) != TF_OK) return TF_ERROR;
1028
+ } else {
1029
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1030
+ }
1031
+ break;
1032
+ }
1033
+ case TF_TYPE_FLOAT64: {
1034
+ double v;
1035
+ if (fast_double(fields[i].ptr, fields[i].len, &v)) {
1036
+ if (tf_batch_set_float64(b, row, i, v) != TF_OK) return TF_ERROR;
1037
+ } else {
1038
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1039
+ }
1040
+ break;
1041
+ }
1042
+ case TF_TYPE_STRING: {
1043
+ if (tf_batch_set_string_len(b, row, i, fields[i].ptr, fields[i].len) != TF_OK)
1044
+ return TF_ERROR;
1045
+ break;
1046
+ }
1047
+ case TF_TYPE_DATE: {
1048
+ int32_t v;
1049
+ if (fast_date(fields[i].ptr, fields[i].len, &v)) {
1050
+ if (tf_batch_set_date(b, row, i, v) != TF_OK) return TF_ERROR;
1051
+ } else {
1052
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1053
+ }
1054
+ break;
1055
+ }
1056
+ case TF_TYPE_TIMESTAMP: {
1057
+ int64_t v;
1058
+ if (fast_timestamp(fields[i].ptr, fields[i].len, &v)) {
1059
+ if (tf_batch_set_timestamp(b, row, i, v) != TF_OK) return TF_ERROR;
1060
+ } else {
1061
+ /* Also try parsing a date-only string as timestamp at midnight */
1062
+ int32_t dv;
1063
+ if (fast_date(fields[i].ptr, fields[i].len, &dv)) {
1064
+ v = (int64_t)dv * 86400LL * 1000000LL;
1065
+ if (tf_batch_set_timestamp(b, row, i, v) != TF_OK) return TF_ERROR;
1066
+ } else {
1067
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1068
+ }
1069
+ }
1070
+ break;
1071
+ }
1072
+ default:
1073
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1074
+ break;
1075
+ }
1076
+ }
1077
+ /* Null-fill extra columns */
1078
+ for (size_t i = cols; i < n_cols; i++) {
1079
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1080
+ }
1081
+ if (tf_batch_expose_row(b, row) != TF_OK) return TF_ERROR;
1082
+ return TF_OK;
1083
+ }
1084
+
1085
+ /*
1086
+ * Convert a STRING batch to a typed batch using frozen column types.
1087
+ * Called once for the first batch after type detection completes.
1088
+ * Uses fast_int64/fast_double for numeric conversion.
1089
+ */
1090
+ static tf_batch *convert_batch_types(csv_decoder_state *st) {
1091
+ tf_batch *src = st->batch;
1092
+ tf_batch *dst = tf_batch_create(st->n_cols, src->n_rows);
1093
+ if (!dst) return NULL;
1094
+
1095
+ for (size_t i = 0; i < st->n_cols; i++) {
1096
+ if (tf_batch_set_schema(dst, i, st->col_names[i], st->col_types[i]) != TF_OK) {
1097
+ tf_batch_free(dst);
1098
+ return NULL;
1099
+ }
1100
+ }
1101
+
1102
+ for (size_t r = 0; r < src->n_rows; r++) {
1103
+ for (size_t c = 0; c < st->n_cols; c++) {
1104
+ if (src->nulls[c][r]) {
1105
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1106
+ tf_batch_free(dst);
1107
+ return NULL;
1108
+ }
1109
+ continue;
1110
+ }
1111
+ const char *val = ((char **)src->columns[c])[r];
1112
+ if (!val) {
1113
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1114
+ tf_batch_free(dst);
1115
+ return NULL;
1116
+ }
1117
+ continue;
1118
+ }
1119
+ size_t vlen = strlen(val);
1120
+ switch (st->col_types[c]) {
1121
+ case TF_TYPE_BOOL: {
1122
+ int v;
1123
+ if (fast_bool(val, vlen, &v)) {
1124
+ if (tf_batch_set_bool(dst, r, c, v != 0) != TF_OK) {
1125
+ tf_batch_free(dst);
1126
+ return NULL;
1127
+ }
1128
+ } else {
1129
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1130
+ tf_batch_free(dst);
1131
+ return NULL;
1132
+ }
1133
+ }
1134
+ break;
1135
+ }
1136
+ case TF_TYPE_INT64: {
1137
+ int64_t v;
1138
+ if (fast_int64(val, vlen, &v)) {
1139
+ if (tf_batch_set_int64(dst, r, c, v) != TF_OK) {
1140
+ tf_batch_free(dst);
1141
+ return NULL;
1142
+ }
1143
+ } else {
1144
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1145
+ tf_batch_free(dst);
1146
+ return NULL;
1147
+ }
1148
+ }
1149
+ break;
1150
+ }
1151
+ case TF_TYPE_FLOAT64: {
1152
+ double v;
1153
+ if (fast_double(val, vlen, &v)) {
1154
+ if (tf_batch_set_float64(dst, r, c, v) != TF_OK) {
1155
+ tf_batch_free(dst);
1156
+ return NULL;
1157
+ }
1158
+ } else {
1159
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1160
+ tf_batch_free(dst);
1161
+ return NULL;
1162
+ }
1163
+ }
1164
+ break;
1165
+ }
1166
+ case TF_TYPE_STRING: {
1167
+ if (tf_batch_set_string(dst, r, c, val) != TF_OK) {
1168
+ tf_batch_free(dst);
1169
+ return NULL;
1170
+ }
1171
+ break;
1172
+ }
1173
+ case TF_TYPE_DATE: {
1174
+ int32_t dv;
1175
+ if (fast_date(val, vlen, &dv)) {
1176
+ if (tf_batch_set_date(dst, r, c, dv) != TF_OK) {
1177
+ tf_batch_free(dst);
1178
+ return NULL;
1179
+ }
1180
+ } else {
1181
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1182
+ tf_batch_free(dst);
1183
+ return NULL;
1184
+ }
1185
+ }
1186
+ break;
1187
+ }
1188
+ case TF_TYPE_TIMESTAMP: {
1189
+ int64_t tv;
1190
+ if (fast_timestamp(val, vlen, &tv)) {
1191
+ if (tf_batch_set_timestamp(dst, r, c, tv) != TF_OK) {
1192
+ tf_batch_free(dst);
1193
+ return NULL;
1194
+ }
1195
+ } else {
1196
+ int32_t dv;
1197
+ if (fast_date(val, vlen, &dv)) {
1198
+ tv = (int64_t)dv * 86400LL * 1000000LL;
1199
+ if (tf_batch_set_timestamp(dst, r, c, tv) != TF_OK) {
1200
+ tf_batch_free(dst);
1201
+ return NULL;
1202
+ }
1203
+ } else {
1204
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1205
+ tf_batch_free(dst);
1206
+ return NULL;
1207
+ }
1208
+ }
1209
+ }
1210
+ break;
1211
+ }
1212
+ default:
1213
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1214
+ tf_batch_free(dst);
1215
+ return NULL;
1216
+ }
1217
+ break;
1218
+ }
1219
+ }
1220
+ if (tf_batch_expose_row(dst, r) != TF_OK) {
1221
+ tf_batch_free(dst);
1222
+ return NULL;
1223
+ }
1224
+ }
1225
+
1226
+ return dst;
1227
+ }
1228
+
1229
+ /*
1230
+ * Add a completed batch to the output array.
1231
+ */
1232
+ static int emit_batch(tf_batch *batch, tf_batch ***out, size_t *n_out, size_t *out_cap) {
1233
+ if (*n_out >= *out_cap) {
1234
+ size_t need = 0;
1235
+ size_t new_cap = 0;
1236
+ if (tf_size_add(*n_out, 1, &need) != TF_OK ||
1237
+ tf_size_grow_pow2(*out_cap, need, 4, &new_cap) != TF_OK) {
1238
+ return TF_ERROR;
1239
+ }
1240
+ tf_batch **tmp = tf_reallocarray_checked(*out, new_cap, sizeof(tf_batch *));
1241
+ if (!tmp) return TF_ERROR;
1242
+ *out = tmp;
1243
+ *out_cap = new_cap;
1244
+ }
1245
+ (*out)[(*n_out)++] = batch;
1246
+ return TF_OK;
1247
+ }
1248
+
1249
+ static tf_batch *make_schema_only_batch(csv_decoder_state *st) {
1250
+ for (size_t i = 0; i < st->n_cols; i++) {
1251
+ if (st->col_types[i] == TF_TYPE_NULL) st->col_types[i] = TF_TYPE_STRING;
1252
+ }
1253
+ tf_batch *b = tf_batch_create(st->n_cols, 0);
1254
+ if (!b) return NULL;
1255
+ for (size_t i = 0; i < st->n_cols; i++) {
1256
+ if (tf_batch_set_schema(b, i, st->col_names[i], st->col_types[i]) != TF_OK) {
1257
+ tf_batch_free(b);
1258
+ return NULL;
1259
+ }
1260
+ }
1261
+ return b;
1262
+ }
1263
+
1264
+ static int csv_fields_match_header(const csv_decoder_state *st, const field_slice *fields, size_t n_fields) {
1265
+ if (!st || !st->schema_ready || n_fields != st->n_cols) return 0;
1266
+ for (size_t i = 0; i < st->n_cols; i++) {
1267
+ const char *name = st->col_names[i] ? st->col_names[i] : "";
1268
+ size_t name_len = strlen(name);
1269
+ if (fields[i].len != name_len) return 0;
1270
+ if (name_len > 0 && memcmp(fields[i].ptr, name, name_len) != 0) return 0;
1271
+ }
1272
+ return 1;
1273
+ }
1274
+
1275
+ /*
1276
+ * Process a single complete CSV line.
1277
+ *
1278
+ * First line → extract column headers.
1279
+ * Subsequent lines → parse fields and add to current batch.
1280
+ * When batch is full → emit it and start a new one.
1281
+ */
1282
+ static int process_line(csv_decoder_state *st, const char *line, size_t line_len,
1283
+ size_t line_no, size_t byte_offset, tf_batch ***out,
1284
+ size_t *n_out, size_t *out_cap, tf_side_channels *side) {
1285
+ if (st->skipped_rows < st->skip_rows) {
1286
+ st->skipped_rows++;
1287
+ return TF_OK;
1288
+ }
1289
+
1290
+ size_t comment_pos = 0;
1291
+ int had_comment = csv_find_comment(st, line, line_len, &comment_pos);
1292
+ if (had_comment) line_len = comment_pos;
1293
+ if ((had_comment && csv_line_is_blank(line, line_len)) ||
1294
+ (st->skip_empty_rows && csv_line_is_blank(line, line_len))) {
1295
+ return TF_OK;
1296
+ }
1297
+
1298
+ if (st->schema_ready && st->limit_rows && st->data_rows_read >= st->n_max) {
1299
+ return TF_OK;
1300
+ }
1301
+
1302
+ /* Reset field arena — escaped field data from previous line is discarded */
1303
+ tf_arena_reset(st->field_arena);
1304
+
1305
+ /* Parse the line into zero-copy field slices while counting every field. */
1306
+ size_t n_fields = 0;
1307
+ if (parse_csv_fields(line, line_len, st->delimiter,
1308
+ st->fields, st->fields_cap, st->field_arena,
1309
+ st->trim_ws, &n_fields) != TF_OK) {
1310
+ tf_set_last_error("csv: failed to parse fields");
1311
+ return TF_ERROR;
1312
+ }
1313
+ if (n_fields > st->max_columns) {
1314
+ if (emit_csv_column_limit_diagnostic(st, line, line_len, line_no, byte_offset,
1315
+ n_fields, side) != TF_OK)
1316
+ return TF_ERROR;
1317
+ char err[256];
1318
+ snprintf(err, sizeof(err), "csv record exceeds max_columns at line %zu: max %zu columns, observed %zu fields",
1319
+ line_no, st->max_columns, n_fields);
1320
+ tf_set_last_error(err);
1321
+ return TF_ERROR;
1322
+ }
1323
+
1324
+ if (st->schema_ready && st->has_header && st->skip_repeated_header && st->after_input_boundary) {
1325
+ st->after_input_boundary = 0;
1326
+ if (csv_fields_match_header(st, st->fields, n_fields)) {
1327
+ return TF_OK;
1328
+ }
1329
+ }
1330
+
1331
+ /* --- First record: extract column headers or synthesize headerless names. --- */
1332
+ if (!st->schema_ready) {
1333
+ if (csv_init_schema(st, st->fields, n_fields, !st->has_header) != TF_OK) return TF_ERROR;
1334
+ if (st->has_header) return TF_OK;
1335
+ if (st->limit_rows && st->data_rows_read >= st->n_max) return TF_OK;
1336
+ }
1337
+
1338
+ /* --- Empty lines: treat as all-null row --- */
1339
+ if (n_fields == 0 && st->n_cols > 0) {
1340
+ for (size_t i = 0; i < st->n_cols; i++) {
1341
+ st->fields[i].ptr = "";
1342
+ st->fields[i].len = 0;
1343
+ st->fields[i].quoted = 0;
1344
+ }
1345
+ n_fields = st->n_cols;
1346
+ }
1347
+
1348
+ /* --- Field-count policy: legacy permissive, observable repair, or strict failure. --- */
1349
+ if (n_fields != st->n_cols) {
1350
+ const char *message = n_fields < st->n_cols
1351
+ ? "CSV row has fewer fields than header"
1352
+ : "CSV row has more fields than header";
1353
+ if (st->mode == CSV_MODE_STRICT) {
1354
+ if (emit_csv_field_count_diagnostic(st, line, line_len, line_no, byte_offset, n_fields,
1355
+ "fail", "error", message, side) != TF_OK)
1356
+ return TF_ERROR;
1357
+ char err[256];
1358
+ snprintf(err, sizeof(err), "csv strict field count mismatch at line %zu: expected %zu fields, got %zu",
1359
+ line_no, st->n_cols, n_fields);
1360
+ tf_set_last_error(err);
1361
+ return TF_ERROR;
1362
+ }
1363
+ if (st->mode == CSV_MODE_REPAIR) {
1364
+ if (emit_csv_field_count_diagnostic(st, line, line_len, line_no, byte_offset, n_fields,
1365
+ "repair", "warning", message, side) != TF_OK)
1366
+ return TF_ERROR;
1367
+ if (emit_csv_repair_audit(st, line, line_len, line_no, byte_offset, n_fields,
1368
+ message, side) != TF_OK)
1369
+ return TF_ERROR;
1370
+ }
1371
+ if (n_fields < st->n_cols) {
1372
+ for (size_t i = n_fields; i < st->n_cols; i++) {
1373
+ st->fields[i].ptr = "";
1374
+ st->fields[i].len = 0;
1375
+ st->fields[i].quoted = 0;
1376
+ }
1377
+ n_fields = st->n_cols;
1378
+ } else {
1379
+ n_fields = st->n_cols;
1380
+ }
1381
+ }
1382
+
1383
+ /* --- Ensure we have a batch --- */
1384
+ if (!st->batch) {
1385
+ if (st->types_frozen) {
1386
+ st->batch = make_typed_batch(st);
1387
+ } else {
1388
+ st->batch = make_string_batch(st);
1389
+ }
1390
+ if (!st->batch) return TF_ERROR;
1391
+ }
1392
+
1393
+ /* --- Add row to batch --- */
1394
+ if (!st->types_frozen) {
1395
+ /* Type detection phase: detect types and store as STRING */
1396
+ for (size_t i = 0; i < n_fields && i < st->n_cols; i++) {
1397
+ tf_type t = detect_type_field(st, &st->fields[i]);
1398
+ st->col_types[i] = widen_type(st->col_types[i], t);
1399
+ }
1400
+ if (add_row_strings(st->batch, st, st->fields, n_fields, st->n_cols) != TF_OK)
1401
+ return TF_ERROR;
1402
+ } else {
1403
+ /* Direct parse phase: parse directly to typed columns */
1404
+ if (add_row_typed(st->batch, st, st->fields, n_fields, st->n_cols, st->col_types) != TF_OK)
1405
+ return TF_ERROR;
1406
+ }
1407
+ st->rows_buffered++;
1408
+ st->data_rows_read++;
1409
+
1410
+ /* --- Emit batch if full --- */
1411
+ if (st->rows_buffered >= st->batch_size) {
1412
+ if (!st->types_frozen) {
1413
+ /* First batch complete: convert STRING → typed, freeze types.
1414
+ * Default any still-NULL columns to STRING. */
1415
+ for (size_t i = 0; i < st->n_cols; i++) {
1416
+ if (st->col_types[i] == TF_TYPE_NULL)
1417
+ st->col_types[i] = TF_TYPE_STRING;
1418
+ }
1419
+ tf_batch *final = convert_batch_types(st);
1420
+ if (!final) return TF_ERROR;
1421
+ tf_batch_free(st->batch);
1422
+ st->batch = NULL;
1423
+ st->types_frozen = 1;
1424
+ if (emit_batch(final, out, n_out, out_cap) != TF_OK) {
1425
+ tf_batch_free(final);
1426
+ return TF_ERROR;
1427
+ }
1428
+ } else {
1429
+ /* Already typed, emit directly (no conversion needed) */
1430
+ if (emit_batch(st->batch, out, n_out, out_cap) != TF_OK) return TF_ERROR;
1431
+ st->batch = NULL;
1432
+ }
1433
+ st->rows_buffered = 0;
1434
+ }
1435
+
1436
+ return TF_OK;
1437
+ }
1438
+
1439
+ /*
1440
+ * Main decode entry point: append data, extract complete lines, process them.
1441
+ *
1442
+ * The line scanner respects quoted fields that may contain newlines.
1443
+ * Complete lines are passed to process_line(); any trailing partial
1444
+ * line remains in line_buf for the next call.
1445
+ */
1446
+ static int csv_decode(tf_decoder *self, const uint8_t *data, size_t len,
1447
+ tf_batch ***out, size_t *n_out, tf_side_channels *side) {
1448
+ csv_decoder_state *st = self->state;
1449
+ *out = NULL;
1450
+ *n_out = 0;
1451
+
1452
+ /* Append incoming data to line buffer */
1453
+ if (tf_buffer_write(&st->line_buf, data, len) != TF_OK) return TF_ERROR;
1454
+
1455
+ /* Scan for complete lines */
1456
+ size_t out_cap = 0;
1457
+ uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
1458
+ size_t buf_len = st->line_buf.len - st->line_buf.read_pos;
1459
+
1460
+ size_t base_offset = st->byte_offset;
1461
+ size_t line_start = 0;
1462
+ int in_quotes = 0;
1463
+ int at_field_start = 1;
1464
+ for (size_t i = 0; i < buf_len; i++) {
1465
+ size_t current_record_len = i - line_start;
1466
+ if (check_csv_record_limit(st, buf, line_start, current_record_len,
1467
+ st->line_number + 1, base_offset + line_start, side) != TF_OK)
1468
+ return TF_ERROR;
1469
+
1470
+ if (in_quotes) {
1471
+ if (buf[i] == '"') {
1472
+ if (i + 1 < buf_len && buf[i + 1] == '"') {
1473
+ i++; /* RFC 4180 escaped quote: doubled quote inside quoted field. */
1474
+ } else {
1475
+ in_quotes = 0;
1476
+ at_field_start = 0;
1477
+ }
1478
+ }
1479
+ continue;
1480
+ }
1481
+
1482
+ if (buf[i] == '"' && at_field_start) {
1483
+ in_quotes = 1;
1484
+ at_field_start = 0;
1485
+ } else if (buf[i] == st->delimiter) {
1486
+ at_field_start = 1;
1487
+ } else if (buf[i] == '\n' || buf[i] == '\r') {
1488
+ size_t line_len = i - line_start;
1489
+ /* Handle \r\n */
1490
+ if (buf[i] == '\r' && i + 1 < buf_len && buf[i + 1] == '\n') {
1491
+ i++;
1492
+ }
1493
+ size_t line_no = ++st->line_number;
1494
+ size_t record_offset = base_offset + line_start;
1495
+ if (line_len > 0 || st->schema_ready) {
1496
+ if (process_line(st, (const char *)buf + line_start, line_len,
1497
+ line_no, record_offset, out, n_out, &out_cap, side) != TF_OK)
1498
+ return TF_ERROR;
1499
+ }
1500
+ line_start = i + 1;
1501
+ at_field_start = 1;
1502
+ } else {
1503
+ at_field_start = 0;
1504
+ }
1505
+ }
1506
+
1507
+ /* Move unconsumed data to start of buffer */
1508
+ st->line_buf.read_pos += line_start;
1509
+ st->byte_offset += line_start;
1510
+ tf_buffer_compact(&st->line_buf);
1511
+
1512
+ return TF_OK;
1513
+ }
1514
+
1515
+ /*
1516
+ * Flush: process any remaining partial line and emit the final batch.
1517
+ */
1518
+ static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out, tf_side_channels *side) {
1519
+ csv_decoder_state *st = self->state;
1520
+ *out = NULL;
1521
+ *n_out = 0;
1522
+ size_t out_cap = 0;
1523
+
1524
+ /* Process any remaining data as the last line */
1525
+ size_t remaining = tf_buffer_readable(&st->line_buf);
1526
+ if (remaining > 0) {
1527
+ uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
1528
+ size_t line_no = ++st->line_number;
1529
+ size_t record_offset = st->byte_offset;
1530
+ if (check_csv_record_limit(st, buf, 0, remaining, line_no, record_offset, side) != TF_OK)
1531
+ return TF_ERROR;
1532
+ if (process_line(st, (const char *)buf, remaining,
1533
+ line_no, record_offset, out, n_out, &out_cap, side) != TF_OK)
1534
+ return TF_ERROR;
1535
+ st->line_buf.read_pos = st->line_buf.len;
1536
+ st->byte_offset += remaining;
1537
+ }
1538
+
1539
+ if (st->schema_ready && st->limit_rows && st->n_max == 0 && !st->batch) {
1540
+ tf_batch *schema_only = make_schema_only_batch(st);
1541
+ if (!schema_only) return TF_ERROR;
1542
+ st->types_frozen = 1;
1543
+ if (emit_batch(schema_only, out, n_out, &out_cap) != TF_OK) {
1544
+ tf_batch_free(schema_only);
1545
+ return TF_ERROR;
1546
+ }
1547
+ }
1548
+
1549
+ /* Emit any remaining partial batch */
1550
+ if (st->batch && st->rows_buffered > 0) {
1551
+ if (!st->types_frozen) {
1552
+ /* Small file: fewer rows than batch_size. Convert and emit. */
1553
+ for (size_t i = 0; i < st->n_cols; i++) {
1554
+ if (st->col_types[i] == TF_TYPE_NULL)
1555
+ st->col_types[i] = TF_TYPE_STRING;
1556
+ }
1557
+ tf_batch *final = convert_batch_types(st);
1558
+ if (!final) return TF_ERROR;
1559
+ tf_batch_free(st->batch);
1560
+ st->batch = NULL;
1561
+ if (emit_batch(final, out, n_out, &out_cap) != TF_OK) {
1562
+ tf_batch_free(final);
1563
+ return TF_ERROR;
1564
+ }
1565
+ } else {
1566
+ /* Already typed, emit directly */
1567
+ if (emit_batch(st->batch, out, n_out, &out_cap) != TF_OK) return TF_ERROR;
1568
+ st->batch = NULL;
1569
+ }
1570
+ st->rows_buffered = 0;
1571
+ }
1572
+
1573
+ st->after_input_boundary = 1;
1574
+ return TF_OK;
1575
+ }
1576
+
1577
+ static void csv_decoder_destroy(tf_decoder *self) {
1578
+ csv_decoder_state *st = self->state;
1579
+ if (st) {
1580
+ tf_buffer_free(&st->line_buf);
1581
+ if (st->batch) tf_batch_free(st->batch);
1582
+ if (st->field_arena) tf_arena_free(st->field_arena);
1583
+ free(st->fields);
1584
+ if (st->col_names) {
1585
+ for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
1586
+ }
1587
+ free(st->col_names);
1588
+ free(st->col_types);
1589
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1590
+ free(st->null_literals);
1591
+ free(st->comment);
1592
+ tf_audit_options_free(&st->audit_opts);
1593
+ free(st);
1594
+ }
1595
+ free(self);
1596
+ }
1597
+
1598
+ tf_decoder *tf_csv_decoder_create(const cJSON *args) {
1599
+ csv_decoder_state *st = calloc(1, sizeof(csv_decoder_state));
1600
+ if (!st) return NULL;
1601
+
1602
+ st->delimiter = ',';
1603
+ st->has_header = 1;
1604
+ st->skip_repeated_header = 0;
1605
+ st->after_input_boundary = 0;
1606
+ st->batch_size = DEFAULT_BATCH_SIZE;
1607
+ st->mode = CSV_MODE_PERMISSIVE;
1608
+ st->quoted_nulls = 1;
1609
+ st->trim_ws = 1;
1610
+ st->skip_empty_rows = 0;
1611
+ st->skip_rows = 0;
1612
+ st->skipped_rows = 0;
1613
+ st->limit_rows = 0;
1614
+ st->n_max = 0;
1615
+ st->data_rows_read = 0;
1616
+ st->max_error_bytes = DEFAULT_MAX_ERROR_BYTES;
1617
+ st->max_record_bytes = DEFAULT_MAX_RECORD_BYTES;
1618
+ st->max_columns = DEFAULT_MAX_COLUMNS;
1619
+ st->audit_limit = 1000;
1620
+ tf_audit_options_init(&st->audit_opts, 1);
1621
+
1622
+ if (args) {
1623
+ cJSON *d = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
1624
+ if (cJSON_IsString(d) && d->valuestring[0])
1625
+ st->delimiter = d->valuestring[0];
1626
+
1627
+ cJSON *h = cJSON_GetObjectItemCaseSensitive(args, "header");
1628
+ if (cJSON_IsBool(h))
1629
+ st->has_header = cJSON_IsTrue(h);
1630
+
1631
+ cJSON *skip_repeated_header = cJSON_GetObjectItemCaseSensitive(args, "skip_repeated_header");
1632
+ if (!skip_repeated_header) skip_repeated_header = cJSON_GetObjectItemCaseSensitive(args, "skipRepeatedHeader");
1633
+ if (cJSON_IsBool(skip_repeated_header))
1634
+ st->skip_repeated_header = cJSON_IsTrue(skip_repeated_header);
1635
+
1636
+ size_t parsed_size = 0;
1637
+ int has_batch_size = tf_json_get_size_arg(args, "batch_size", 1, TF_MAX_BATCH_ROWS, &parsed_size, "csv");
1638
+ if (has_batch_size < 0) { free(st); return NULL; }
1639
+ if (has_batch_size > 0) st->batch_size = parsed_size;
1640
+
1641
+ cJSON *rep = cJSON_GetObjectItemCaseSensitive(args, "repair");
1642
+ if (cJSON_IsBool(rep) && cJSON_IsTrue(rep))
1643
+ st->mode = CSV_MODE_REPAIR;
1644
+
1645
+ cJSON *mode = cJSON_GetObjectItemCaseSensitive(args, "mode");
1646
+ if (cJSON_IsString(mode)) {
1647
+ if (!parse_csv_mode(mode->valuestring, &st->mode)) {
1648
+ tf_set_last_error("csv: mode must be permissive, repair, or strict");
1649
+ free(st);
1650
+ return NULL;
1651
+ }
1652
+ }
1653
+
1654
+ cJSON *strict = cJSON_GetObjectItemCaseSensitive(args, "strict");
1655
+ if (cJSON_IsBool(strict) && cJSON_IsTrue(strict))
1656
+ st->mode = CSV_MODE_STRICT;
1657
+
1658
+ int has_max_error = tf_json_get_size_arg(args, "max_error_bytes", 0, TF_MAX_ERROR_BYTES, &parsed_size, "csv");
1659
+ if (has_max_error < 0) { free(st); return NULL; }
1660
+ if (has_max_error > 0) st->max_error_bytes = parsed_size;
1661
+
1662
+ int has_max_record = tf_json_get_size_arg(args, "max_record_bytes", 0, TF_MAX_RECORD_BYTES, &parsed_size, "csv");
1663
+ if (has_max_record < 0) { free(st); return NULL; }
1664
+ if (has_max_record > 0) st->max_record_bytes = parsed_size;
1665
+
1666
+ int has_max_columns = tf_json_get_size_arg(args, "max_columns", 1, TF_MAX_COLUMNS, &parsed_size, "csv");
1667
+ if (has_max_columns < 0) { free(st); return NULL; }
1668
+ if (has_max_columns == 0) {
1669
+ has_max_columns = tf_json_get_size_arg(args, "maxColumns", 1, TF_MAX_COLUMNS, &parsed_size, "csv");
1670
+ if (has_max_columns < 0) { free(st); return NULL; }
1671
+ }
1672
+ if (has_max_columns > 0) st->max_columns = parsed_size;
1673
+
1674
+ cJSON *audit = cJSON_GetObjectItemCaseSensitive(args, "audit");
1675
+ st->audit = cJSON_IsTrue(audit) ? 1 : 0;
1676
+ cJSON *audit_limit = cJSON_GetObjectItemCaseSensitive(args, "audit_limit");
1677
+ if (!audit_limit) audit_limit = cJSON_GetObjectItemCaseSensitive(args, "auditLimit");
1678
+ if (audit_limit) {
1679
+ size_t parsed_limit = 0;
1680
+ if (tf_json_get_size_arg_any(args, "audit_limit", "auditLimit",
1681
+ 1, TF_MAX_AUDIT_RECORDS,
1682
+ &parsed_limit, "csv") < 0) {
1683
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1684
+ free(st->null_literals);
1685
+ free(st->comment);
1686
+ free(st);
1687
+ return NULL;
1688
+ }
1689
+ st->audit_limit = parsed_limit;
1690
+ st->audit = 1;
1691
+ }
1692
+
1693
+ cJSON *quoted_nulls = cJSON_GetObjectItemCaseSensitive(args, "quoted_nulls");
1694
+ if (cJSON_IsBool(quoted_nulls))
1695
+ st->quoted_nulls = cJSON_IsTrue(quoted_nulls);
1696
+
1697
+ cJSON *trim_ws = cJSON_GetObjectItemCaseSensitive(args, "trim_ws");
1698
+ if (!trim_ws) trim_ws = cJSON_GetObjectItemCaseSensitive(args, "trimWs");
1699
+ if (cJSON_IsBool(trim_ws))
1700
+ st->trim_ws = cJSON_IsTrue(trim_ws);
1701
+
1702
+ cJSON *skip_empty = cJSON_GetObjectItemCaseSensitive(args, "skip_empty_rows");
1703
+ if (!skip_empty) skip_empty = cJSON_GetObjectItemCaseSensitive(args, "skipEmptyRows");
1704
+ if (cJSON_IsBool(skip_empty))
1705
+ st->skip_empty_rows = cJSON_IsTrue(skip_empty);
1706
+
1707
+ cJSON *skip_rows = cJSON_GetObjectItemCaseSensitive(args, "skip");
1708
+ if (!skip_rows) skip_rows = cJSON_GetObjectItemCaseSensitive(args, "skip_rows");
1709
+ if (!skip_rows) skip_rows = cJSON_GetObjectItemCaseSensitive(args, "skipRows");
1710
+ if (skip_rows) {
1711
+ const char *skip_name = cJSON_GetObjectItemCaseSensitive(args, "skip") ? "skip" :
1712
+ (cJSON_GetObjectItemCaseSensitive(args, "skip_rows") ? "skip_rows" : "skipRows");
1713
+ size_t parsed_skip = 0;
1714
+ if (tf_json_size_value(skip_rows, skip_name, 0, TF_MAX_COUNT_ARG,
1715
+ &parsed_skip, "csv") < 0) {
1716
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1717
+ free(st->null_literals);
1718
+ free(st->comment);
1719
+ free(st);
1720
+ return NULL;
1721
+ }
1722
+ st->skip_rows = parsed_skip;
1723
+ }
1724
+
1725
+ cJSON *n_max = cJSON_GetObjectItemCaseSensitive(args, "n_max");
1726
+ if (!n_max) n_max = cJSON_GetObjectItemCaseSensitive(args, "nMax");
1727
+ if (!n_max) n_max = cJSON_GetObjectItemCaseSensitive(args, "max_rows");
1728
+ if (!n_max) n_max = cJSON_GetObjectItemCaseSensitive(args, "maxRows");
1729
+ if (n_max) {
1730
+ const char *n_max_name = cJSON_GetObjectItemCaseSensitive(args, "n_max") ? "n_max" :
1731
+ (cJSON_GetObjectItemCaseSensitive(args, "nMax") ? "nMax" :
1732
+ (cJSON_GetObjectItemCaseSensitive(args, "max_rows") ? "max_rows" : "maxRows"));
1733
+ size_t parsed_n_max = 0;
1734
+ if (tf_json_size_value(n_max, n_max_name, 0, TF_MAX_COUNT_ARG,
1735
+ &parsed_n_max, "csv") < 0) {
1736
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1737
+ free(st->null_literals);
1738
+ free(st->comment);
1739
+ free(st);
1740
+ return NULL;
1741
+ }
1742
+ st->limit_rows = 1;
1743
+ st->n_max = parsed_n_max;
1744
+ }
1745
+
1746
+ cJSON *comment = cJSON_GetObjectItemCaseSensitive(args, "comment");
1747
+ if (cJSON_IsString(comment) && comment->valuestring && comment->valuestring[0]) {
1748
+ st->comment = strdup(comment->valuestring);
1749
+ if (!st->comment) {
1750
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1751
+ free(st->null_literals);
1752
+ free(st->comment);
1753
+ free(st);
1754
+ return NULL;
1755
+ }
1756
+ st->comment_len = strlen(st->comment);
1757
+ }
1758
+
1759
+ if (csv_configure_null_literals(st, args) != TF_OK) {
1760
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1761
+ free(st->null_literals);
1762
+ free(st->comment);
1763
+ tf_audit_options_free(&st->audit_opts);
1764
+ free(st);
1765
+ return NULL;
1766
+ }
1767
+ if (tf_audit_options_parse(&st->audit_opts, args, "csv") != TF_OK) {
1768
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1769
+ free(st->null_literals);
1770
+ free(st->comment);
1771
+ tf_audit_options_free(&st->audit_opts);
1772
+ free(st);
1773
+ return NULL;
1774
+ }
1775
+ }
1776
+
1777
+ tf_buffer_init(&st->line_buf);
1778
+
1779
+ st->fields_cap = st->max_columns;
1780
+ st->fields = tf_callocarray_checked(st->fields_cap, sizeof(field_slice));
1781
+ if (!st->fields) {
1782
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1783
+ free(st->null_literals);
1784
+ free(st->comment);
1785
+ tf_audit_options_free(&st->audit_opts);
1786
+ free(st);
1787
+ return NULL;
1788
+ }
1789
+
1790
+ /* Arena for escaped quoted field data (reset per line, rarely used) */
1791
+ st->field_arena = tf_arena_create(4096);
1792
+ if (!st->field_arena) {
1793
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1794
+ free(st->null_literals);
1795
+ free(st->comment);
1796
+ free(st->fields);
1797
+ tf_audit_options_free(&st->audit_opts);
1798
+ free(st);
1799
+ return NULL;
1800
+ }
1801
+
1802
+ tf_decoder *dec = malloc(sizeof(tf_decoder));
1803
+ if (!dec) {
1804
+ tf_arena_free(st->field_arena);
1805
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1806
+ free(st->null_literals);
1807
+ free(st->comment);
1808
+ free(st->fields);
1809
+ tf_audit_options_free(&st->audit_opts);
1810
+ free(st);
1811
+ return NULL;
1812
+ }
1813
+ dec->decode = csv_decode;
1814
+ dec->flush = csv_flush;
1815
+ dec->destroy = csv_decoder_destroy;
1816
+ dec->state = st;
1817
+ return dec;
1818
+ }
1819
+
1820
+ /* ================================================================
1821
+ * CSV Encoder
1822
+ * ================================================================ */
1823
+
1824
+ typedef struct {
1825
+ char delimiter;
1826
+ int header_written;
1827
+ } csv_encoder_state;
1828
+
1829
+ /* Check if a field needs quoting */
1830
+ static int needs_quoting(const char *s, char delim) {
1831
+ for (const char *p = s; *p; p++) {
1832
+ if (*p == delim || *p == '"' || *p == '\n' || *p == '\r')
1833
+ return 1;
1834
+ }
1835
+ return 0;
1836
+ }
1837
+
1838
+ static int csv_write_field(tf_buffer *out, const char *s, char delim) {
1839
+ if (needs_quoting(s, delim)) {
1840
+ if (tf_buffer_write(out, (const uint8_t *)"\"", 1) != TF_OK) return TF_ERROR;
1841
+ for (const char *p = s; *p; p++) {
1842
+ if (*p == '"') {
1843
+ if (tf_buffer_write(out, (const uint8_t *)"\"\"", 2) != TF_OK) return TF_ERROR;
1844
+ } else {
1845
+ if (tf_buffer_write(out, (const uint8_t *)p, 1) != TF_OK) return TF_ERROR;
1846
+ }
1847
+ }
1848
+ if (tf_buffer_write(out, (const uint8_t *)"\"", 1) != TF_OK) return TF_ERROR;
1849
+ } else {
1850
+ if (tf_buffer_write_str(out, s) != TF_OK) return TF_ERROR;
1851
+ }
1852
+ return TF_OK;
1853
+ }
1854
+
1855
+ static int csv_write_delimiter(tf_buffer *out, char delim) {
1856
+ return tf_buffer_write(out, (const uint8_t *)&delim, 1);
1857
+ }
1858
+
1859
+ static int csv_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
1860
+ csv_encoder_state *st = self->state;
1861
+
1862
+ /* Write header */
1863
+ if (!st->header_written) {
1864
+ for (size_t i = 0; i < in->n_cols; i++) {
1865
+ if (i > 0 && csv_write_delimiter(out, st->delimiter) != TF_OK) return TF_ERROR;
1866
+ if (csv_write_field(out, in->col_names[i], st->delimiter) != TF_OK) return TF_ERROR;
1867
+ }
1868
+ if (tf_buffer_write(out, (const uint8_t *)"\n", 1) != TF_OK) return TF_ERROR;
1869
+ st->header_written = 1;
1870
+ }
1871
+
1872
+ /* Write rows */
1873
+ char numbuf[64];
1874
+ for (size_t r = 0; r < in->n_rows; r++) {
1875
+ for (size_t c = 0; c < in->n_cols; c++) {
1876
+ if (c > 0 && csv_write_delimiter(out, st->delimiter) != TF_OK) return TF_ERROR;
1877
+ if (tf_batch_is_null(in, r, c)) {
1878
+ /* empty field for null */
1879
+ continue;
1880
+ }
1881
+ switch (in->col_types[c]) {
1882
+ case TF_TYPE_BOOL:
1883
+ if (tf_buffer_write_str(out, tf_batch_get_bool(in, r, c) ? "true" : "false") != TF_OK)
1884
+ return TF_ERROR;
1885
+ break;
1886
+ case TF_TYPE_INT64:
1887
+ snprintf(numbuf, sizeof(numbuf), "%lld", (long long)tf_batch_get_int64(in, r, c));
1888
+ if (tf_buffer_write_str(out, numbuf) != TF_OK) return TF_ERROR;
1889
+ break;
1890
+ case TF_TYPE_FLOAT64:
1891
+ if (tf_format_float64(numbuf, sizeof(numbuf), tf_batch_get_float64(in, r, c)) != TF_OK)
1892
+ return TF_ERROR;
1893
+ if (tf_buffer_write_str(out, numbuf) != TF_OK) return TF_ERROR;
1894
+ break;
1895
+ case TF_TYPE_STRING:
1896
+ if (csv_write_field(out, tf_batch_get_string(in, r, c), st->delimiter) != TF_OK)
1897
+ return TF_ERROR;
1898
+ break;
1899
+ case TF_TYPE_DATE: {
1900
+ char dbuf[32];
1901
+ tf_date_format(tf_batch_get_date(in, r, c), dbuf, sizeof(dbuf));
1902
+ if (tf_buffer_write_str(out, dbuf) != TF_OK) return TF_ERROR;
1903
+ break;
1904
+ }
1905
+ case TF_TYPE_TIMESTAMP: {
1906
+ char tsbuf[40];
1907
+ tf_timestamp_format(tf_batch_get_timestamp(in, r, c), tsbuf, sizeof(tsbuf));
1908
+ if (tf_buffer_write_str(out, tsbuf) != TF_OK) return TF_ERROR;
1909
+ break;
1910
+ }
1911
+ default:
1912
+ break;
1913
+ }
1914
+ }
1915
+ if (tf_buffer_write(out, (const uint8_t *)"\n", 1) != TF_OK) return TF_ERROR;
1916
+ }
1917
+
1918
+ return TF_OK;
1919
+ }
1920
+
1921
+ static int csv_encoder_flush(tf_encoder *self, tf_buffer *out) {
1922
+ (void)self; (void)out;
1923
+ return TF_OK;
1924
+ }
1925
+
1926
+ static void csv_encoder_destroy(tf_encoder *self) {
1927
+ free(self->state);
1928
+ free(self);
1929
+ }
1930
+
1931
+ tf_encoder *tf_csv_encoder_create(const cJSON *args) {
1932
+ csv_encoder_state *st = calloc(1, sizeof(csv_encoder_state));
1933
+ if (!st) return NULL;
1934
+
1935
+ st->delimiter = ',';
1936
+ st->header_written = 0;
1937
+
1938
+ if (args) {
1939
+ cJSON *d = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
1940
+ if (cJSON_IsString(d) && d->valuestring[0])
1941
+ st->delimiter = d->valuestring[0];
1942
+ }
1943
+
1944
+ tf_encoder *enc = malloc(sizeof(tf_encoder));
1945
+ if (!enc) { free(st); return NULL; }
1946
+ enc->encode = csv_encode;
1947
+ enc->flush = csv_encoder_flush;
1948
+ enc->destroy = csv_encoder_destroy;
1949
+ enc->state = st;
1950
+ return enc;
1951
+ }