tranfi 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +272 -40
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +348 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +8 -11
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/codec_csv.c CHANGED
@@ -8,9 +8,9 @@
8
8
  * fields with escaped quotes ("") need copying (rare in practice).
9
9
  *
10
10
  * 2. Type detection window: the first batch (batch_size rows) detects
11
- * column types via progressive widening (NULL → INT64 → FLOAT64 →
12
- * STRING). Types freeze after the first batch. This matches the
13
- * behavior of Arrow CSV, DuckDB, and other production parsers.
11
+ * column types via progressive widening (NULL → BOOL / INT64 →
12
+ * FLOAT64 / DATE / TIMESTAMP → STRING). Types freeze after the
13
+ * first batch. This matches Arrow CSV, DuckDB, and other production parsers.
14
14
  *
15
15
  * 3. Direct-to-typed parsing: after types freeze, field slices are parsed
16
16
  * directly into typed column arrays (int64, double, string) without
@@ -30,8 +30,10 @@
30
30
  #include <limits.h>
31
31
  #include <math.h>
32
32
 
33
- #define DEFAULT_BATCH_SIZE 1024
34
- #define MAX_COLS 256 /* max columns per row */
33
+ #define DEFAULT_BATCH_SIZE 1024
34
+ #define DEFAULT_MAX_ERROR_BYTES 4096
35
+ #define DEFAULT_MAX_RECORD_BYTES (64 * 1024 * 1024)
36
+ #define DEFAULT_MAX_COLUMNS 8192
35
37
 
36
38
  /* ================================================================
37
39
  * Field Slice — zero-copy reference into the line buffer
@@ -40,6 +42,7 @@
40
42
  typedef struct {
41
43
  const char *ptr; /* points into line buffer (or field_arena for escapes) */
42
44
  size_t len;
45
+ int quoted;
43
46
  } field_slice;
44
47
 
45
48
  /* ================================================================
@@ -92,7 +95,7 @@ static int fast_int64(const char *s, size_t len, int64_t *out) {
92
95
  * Fast double parser for common decimal formats: [-+]digits[.digits]
93
96
  *
94
97
  * Uses integer accumulation + power-of-10 division for precision.
95
- * Handles up to 18 significant digits (fits in uint64 without overflow).
98
+ * Uses a short-decimal fast path and falls back to strtod for values that need correct rounding.
96
99
  * Falls back to strtod for exponents (e/E), special values, or
97
100
  * very long mantissas.
98
101
  */
@@ -130,8 +133,8 @@ static int fast_double(const char *s, size_t len, double *out) {
130
133
 
131
134
  if (n_digits == 0) return 0;
132
135
 
133
- /* Fast path: no exponent and ≤18 digits (uint64 safe) */
134
- if (i == len && n_digits <= 18) {
136
+ /* Fast path: no exponent and short mantissas. Longer decimals need strtod's correct rounding. */
137
+ if (i == len && n_digits <= 15) {
135
138
  static const double pow10[] = {
136
139
  1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9,
137
140
  1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18
@@ -150,7 +153,12 @@ static int fast_double(const char *s, size_t len, double *out) {
150
153
  char *end;
151
154
  errno = 0;
152
155
  *out = strtod(buf, &end);
153
- if (errno || (size_t)(end - buf) != len) return 0;
156
+ if ((size_t)(end - buf) != len) return 0;
157
+ if (errno == ERANGE) {
158
+ if (!isfinite(*out) || *out == 0.0) return 0;
159
+ } else if (errno) {
160
+ return 0;
161
+ }
154
162
  return 1;
155
163
  }
156
164
 
@@ -275,12 +283,36 @@ static int fast_timestamp(const char *s, size_t len, int64_t *out) {
275
283
  return 1;
276
284
  }
277
285
 
286
+ static int ascii_lower_char(int ch) {
287
+ return (ch >= 'A' && ch <= 'Z') ? ch + ('a' - 'A') : ch;
288
+ }
289
+
290
+ static int slice_ieq(const char *s, const char *lit, size_t len) {
291
+ for (size_t i = 0; i < len; i++) {
292
+ if (ascii_lower_char((unsigned char)s[i]) != (unsigned char)lit[i]) return 0;
293
+ }
294
+ return 1;
295
+ }
296
+
297
+ static int fast_bool(const char *s, size_t len, int *out) {
298
+ if (len == 4 && slice_ieq(s, "true", 4)) {
299
+ if (out) *out = 1;
300
+ return 1;
301
+ }
302
+ if (len == 5 && slice_ieq(s, "false", 5)) {
303
+ if (out) *out = 0;
304
+ return 1;
305
+ }
306
+ return 0;
307
+ }
308
+
278
309
  /* Detect the type of a field slice without copying it. */
279
310
  static tf_type detect_type_slice(const char *s, size_t len) {
280
311
  if (len == 0) return TF_TYPE_NULL;
281
312
  int64_t iv;
282
313
  double fv;
283
314
  int32_t dv;
315
+ if (fast_bool(s, len, NULL)) return TF_TYPE_BOOL;
284
316
  if (fast_int64(s, len, &iv)) return TF_TYPE_INT64;
285
317
  if (fast_double(s, len, &fv)) return TF_TYPE_FLOAT64;
286
318
  if (fast_date(s, len, &dv)) return TF_TYPE_DATE;
@@ -288,7 +320,7 @@ static tf_type detect_type_slice(const char *s, size_t len) {
288
320
  return TF_TYPE_STRING;
289
321
  }
290
322
 
291
- /* Widen a column type if needed (NULL < INT64 < FLOAT64 < STRING). */
323
+ /* Widen a column type if needed. */
292
324
  static tf_type widen_type(tf_type current, tf_type incoming) {
293
325
  if (current == incoming) return current;
294
326
  if (current == TF_TYPE_NULL) return incoming;
@@ -316,20 +348,25 @@ static tf_type widen_type(tf_type current, tf_type incoming) {
316
348
  * - Unquoted fields: zero-copy slice into line buffer
317
349
  * - Quoted fields without "": zero-copy slice (skipping quotes)
318
350
  * - Quoted fields with "": unescaped copy allocated from field_arena
319
- * - Leading/trailing whitespace is trimmed from unquoted fields
351
+ * - Leading/trailing whitespace is optionally trimmed from unquoted fields
320
352
  */
321
- static size_t parse_csv_fields(const char *line, size_t line_len, char delim,
322
- field_slice *fields, size_t max_fields,
323
- tf_arena *field_arena) {
353
+ static int parse_csv_fields(const char *line, size_t line_len, char delim,
354
+ field_slice *fields, size_t max_fields,
355
+ tf_arena *field_arena, int trim_ws,
356
+ size_t *out_count) {
357
+ if (!out_count) return TF_ERROR;
324
358
  size_t count = 0;
325
359
  size_t i = 0;
326
360
 
327
- while (i <= line_len && count < max_fields) {
361
+ while (i <= line_len) {
328
362
  if (i == line_len) {
329
- /* Trailing delimiter → empty last field */
363
+ /* Trailing delimiter -> empty last field */
330
364
  if (count > 0 && i > 0 && line[i - 1] == delim) {
331
- fields[count].ptr = "";
332
- fields[count].len = 0;
365
+ if (count < max_fields) {
366
+ fields[count].ptr = "";
367
+ fields[count].len = 0;
368
+ fields[count].quoted = 0;
369
+ }
333
370
  count++;
334
371
  }
335
372
  break;
@@ -358,26 +395,36 @@ static size_t parse_csv_fields(const char *line, size_t line_len, char delim,
358
395
  if (i < line_len) i++; /* skip closing quote */
359
396
  if (i < line_len && line[i] == delim) i++; /* skip delimiter */
360
397
 
361
- if (!has_escape) {
362
- /* Zero-copy: slice directly into line buffer, past the quotes */
363
- fields[count].ptr = line + start;
364
- fields[count].len = field_end - start;
365
- } else {
366
- /* Rare path: unescape "" → " into arena-allocated buffer */
367
- size_t max_len = field_end - start; /* unescaped is always shorter */
368
- char *buf = tf_arena_alloc(field_arena, max_len + 1);
369
- size_t out_len = 0;
370
- for (size_t j = start; j < field_end; j++) {
371
- if (line[j] == '"' && j + 1 < field_end && line[j + 1] == '"') {
372
- buf[out_len++] = '"';
373
- j++; /* skip second quote */
374
- } else {
375
- buf[out_len++] = line[j];
398
+ if (count < max_fields) {
399
+ if (!has_escape) {
400
+ /* Zero-copy: slice directly into line buffer, past the quotes */
401
+ fields[count].ptr = line + start;
402
+ fields[count].len = field_end - start;
403
+ fields[count].quoted = 1;
404
+ } else {
405
+ /* Rare path: unescape "" -> " into arena-allocated buffer */
406
+ size_t max_len = field_end - start; /* unescaped is always shorter */
407
+ size_t bytes = 0;
408
+ if (tf_check_byte_limit(max_len, TF_MAX_CELL_BYTES,
409
+ "csv", "cell") != TF_OK ||
410
+ tf_size_add(max_len, 1, &bytes) != TF_OK)
411
+ return TF_ERROR;
412
+ char *buf = tf_arena_alloc(field_arena, bytes);
413
+ if (!buf) return TF_ERROR;
414
+ size_t out_len = 0;
415
+ for (size_t j = start; j < field_end; j++) {
416
+ if (line[j] == '"' && j + 1 < field_end && line[j + 1] == '"') {
417
+ buf[out_len++] = '"';
418
+ j++; /* skip second quote */
419
+ } else {
420
+ buf[out_len++] = line[j];
421
+ }
376
422
  }
423
+ buf[out_len] = '\0';
424
+ fields[count].ptr = buf;
425
+ fields[count].len = out_len;
426
+ fields[count].quoted = 1;
377
427
  }
378
- buf[out_len] = '\0';
379
- fields[count].ptr = buf;
380
- fields[count].len = out_len;
381
428
  }
382
429
  count++;
383
430
  } else {
@@ -388,33 +435,68 @@ static size_t parse_csv_fields(const char *line, size_t line_len, char delim,
388
435
  const char *fptr = line + start;
389
436
  size_t flen = i - start;
390
437
 
391
- /* Trim trailing whitespace */
392
- while (flen > 0 && (fptr[flen - 1] == ' ' || fptr[flen - 1] == '\t'))
393
- flen--;
394
- /* Trim leading whitespace */
395
- while (flen > 0 && (*fptr == ' ' || *fptr == '\t'))
396
- { fptr++; flen--; }
438
+ if (trim_ws) {
439
+ /* Trim trailing ASCII whitespace */
440
+ while (flen > 0 && (fptr[flen - 1] == ' ' || fptr[flen - 1] == '\t'))
441
+ flen--;
442
+ /* Trim leading ASCII whitespace */
443
+ while (flen > 0 && (*fptr == ' ' || *fptr == '\t'))
444
+ { fptr++; flen--; }
445
+ }
397
446
 
398
- fields[count].ptr = fptr;
399
- fields[count].len = flen;
447
+ if (count < max_fields) {
448
+ fields[count].ptr = fptr;
449
+ fields[count].len = flen;
450
+ fields[count].quoted = 0;
451
+ }
400
452
  count++;
401
453
 
402
454
  if (i < line_len) i++; /* skip delimiter */
403
455
  }
404
456
  }
405
457
 
406
- return count;
458
+ *out_count = count;
459
+ return TF_OK;
407
460
  }
408
461
 
409
462
  /* ================================================================
410
463
  * CSV Decoder
411
464
  * ================================================================ */
412
465
 
466
+ typedef enum {
467
+ CSV_MODE_PERMISSIVE = 0,
468
+ CSV_MODE_REPAIR,
469
+ CSV_MODE_STRICT,
470
+ } csv_decode_mode;
471
+
413
472
  typedef struct {
414
473
  char delimiter;
415
474
  int has_header;
475
+ int skip_repeated_header;
476
+ int after_input_boundary;
416
477
  size_t batch_size;
417
- int repair; /* if true, pad short rows and truncate long rows */
478
+ csv_decode_mode mode;
479
+ char **null_literals;
480
+ size_t n_null_literals;
481
+ int quoted_nulls;
482
+ int trim_ws;
483
+ int skip_empty_rows;
484
+ size_t skip_rows;
485
+ size_t skipped_rows;
486
+ int limit_rows;
487
+ size_t n_max;
488
+ size_t data_rows_read;
489
+ char *comment;
490
+ size_t comment_len;
491
+ size_t max_error_bytes;
492
+ size_t max_record_bytes;
493
+ size_t max_columns;
494
+ size_t audit_limit;
495
+ size_t audit_emitted;
496
+ int audit;
497
+ tf_audit_options audit_opts;
498
+ size_t line_number;
499
+ size_t byte_offset;
418
500
 
419
501
  /* Line accumulator: incoming bytes are appended, complete lines extracted */
420
502
  tf_buffer line_buf;
@@ -434,10 +516,423 @@ typedef struct {
434
516
  size_t rows_buffered;
435
517
 
436
518
  /* Reusable per-line scratch: field slices array and arena for escapes */
437
- field_slice fields[MAX_COLS];
438
- tf_arena *field_arena;
519
+ field_slice *fields;
520
+ size_t fields_cap;
521
+ tf_arena *field_arena;
439
522
  } csv_decoder_state;
440
523
 
524
+ static const char *csv_mode_name(csv_decode_mode mode) {
525
+ switch (mode) {
526
+ case CSV_MODE_REPAIR: return "repair";
527
+ case CSV_MODE_STRICT: return "strict";
528
+ case CSV_MODE_PERMISSIVE:
529
+ default: return "permissive";
530
+ }
531
+ }
532
+
533
+ static int parse_csv_mode(const char *s, csv_decode_mode *out) {
534
+ if (!s || strcmp(s, "permissive") == 0 || strcmp(s, "lenient") == 0) {
535
+ *out = CSV_MODE_PERMISSIVE;
536
+ return 1;
537
+ }
538
+ if (strcmp(s, "repair") == 0) {
539
+ *out = CSV_MODE_REPAIR;
540
+ return 1;
541
+ }
542
+ if (strcmp(s, "strict") == 0 || strcmp(s, "fail") == 0 || strcmp(s, "error") == 0) {
543
+ *out = CSV_MODE_STRICT;
544
+ return 1;
545
+ }
546
+ return 0;
547
+ }
548
+
549
+ static int csv_line_is_blank(const char *line, size_t line_len) {
550
+ for (size_t i = 0; i < line_len; i++) {
551
+ if (line[i] != ' ' && line[i] != '\t') return 0;
552
+ }
553
+ return 1;
554
+ }
555
+
556
+ static int csv_find_comment(const csv_decoder_state *st, const char *line, size_t line_len,
557
+ size_t *comment_pos) {
558
+ if (!st->comment || st->comment_len == 0 || st->comment_len > line_len) return 0;
559
+ int in_quotes = 0;
560
+ int at_field_start = 1;
561
+ for (size_t i = 0; i < line_len; i++) {
562
+ if (in_quotes) {
563
+ if (line[i] == '"') {
564
+ if (i + 1 < line_len && line[i + 1] == '"') {
565
+ i++;
566
+ } else {
567
+ in_quotes = 0;
568
+ at_field_start = 0;
569
+ }
570
+ }
571
+ continue;
572
+ }
573
+ if (line[i] == '"' && at_field_start) {
574
+ in_quotes = 1;
575
+ at_field_start = 0;
576
+ continue;
577
+ }
578
+ if (i + st->comment_len <= line_len &&
579
+ memcmp(line + i, st->comment, st->comment_len) == 0) {
580
+ *comment_pos = i;
581
+ return 1;
582
+ }
583
+ if (line[i] == st->delimiter) at_field_start = 1;
584
+ else at_field_start = 0;
585
+ }
586
+ return 0;
587
+ }
588
+
589
+ static char *csv_raw_preview(const csv_decoder_state *st, const char *line, size_t line_len,
590
+ int *truncated_out) {
591
+ size_t keep = line_len;
592
+ int truncated = 0;
593
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
594
+ keep = st->max_error_bytes;
595
+ truncated = 1;
596
+ }
597
+ char *raw = malloc(keep + 1);
598
+ if (!raw) return NULL;
599
+ memcpy(raw, line, keep);
600
+ raw[keep] = '\0';
601
+ if (truncated_out) *truncated_out = truncated;
602
+ return raw;
603
+ }
604
+
605
+ static int csv_add_raw_payload(csv_decoder_state *st, cJSON *obj, const char *raw, int truncated) {
606
+ if (st && st->audit_opts.include_row) {
607
+ char *safe_raw = tf_audit_format_string_dup_for_column(&st->audit_opts, "raw", raw);
608
+ if (!safe_raw) return TF_ERROR;
609
+ if (st->audit_opts.max_bytes > 0 && strlen(safe_raw) > st->audit_opts.max_bytes) {
610
+ if (tf_json_add_bool(obj, "_audit_truncated", 1) != TF_OK ||
611
+ tf_json_add_number(obj, "max_bytes", (double)st->audit_opts.max_bytes) != TF_OK) {
612
+ free(safe_raw);
613
+ return TF_ERROR;
614
+ }
615
+ } else {
616
+ if (tf_json_add_string(obj, "raw", safe_raw) != TF_OK) {
617
+ free(safe_raw);
618
+ return TF_ERROR;
619
+ }
620
+ }
621
+ free(safe_raw);
622
+ }
623
+ if (truncated && tf_json_add_bool(obj, "truncated", 1) != TF_OK) return TF_ERROR;
624
+ return TF_OK;
625
+ }
626
+
627
+ static int emit_csv_repair_audit(csv_decoder_state *st, const char *line, size_t line_len,
628
+ size_t line_no, size_t byte_offset, size_t n_fields,
629
+ const char *message, tf_side_channels *side) {
630
+ if (!st->audit || st->audit_emitted >= st->audit_limit || !side || !side->stats) return TF_OK;
631
+ int truncated = 0;
632
+ char *raw = csv_raw_preview(st, line, line_len, &truncated);
633
+ if (!raw) return TF_ERROR;
634
+ cJSON *obj = cJSON_CreateObject();
635
+ if (!obj) { free(raw); return TF_ERROR; }
636
+ int rc = TF_ERROR;
637
+ if (tf_json_add_string(obj, "type", "audit") != TF_OK ||
638
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
639
+ tf_json_add_string(obj, "event", "row_repaired") != TF_OK ||
640
+ tf_json_add_string(obj, "reason", "csv_field_count") != TF_OK ||
641
+ tf_json_add_string(obj, "channel", "audit") != TF_OK ||
642
+ tf_json_add_string(obj, "mode", csv_mode_name(st->mode)) != TF_OK ||
643
+ tf_json_add_string(obj, "action", "repair") != TF_OK ||
644
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
645
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
646
+ tf_json_add_number(obj, "expected_fields", (double)st->n_cols) != TF_OK ||
647
+ tf_json_add_number(obj, "actual_fields", (double)n_fields) != TF_OK ||
648
+ tf_json_add_string(obj, "message", message ? message : "CSV row field count differs from header") != TF_OK ||
649
+ tf_json_add_number(obj, "raw_bytes", (double)line_len) != TF_OK ||
650
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
651
+ goto done;
652
+ }
653
+ rc = tf_buffer_write_json_line(side->stats, obj);
654
+ done:
655
+ cJSON_Delete(obj);
656
+ free(raw);
657
+ if (rc == TF_OK) st->audit_emitted++;
658
+ return rc;
659
+ }
660
+
661
+ static int emit_csv_field_count_diagnostic(csv_decoder_state *st, const char *line, size_t line_len,
662
+ size_t line_no, size_t byte_offset, size_t n_fields,
663
+ const char *action, const char *severity,
664
+ const char *message, tf_side_channels *side) {
665
+ if (!side || !side->errors) return TF_OK;
666
+
667
+ size_t keep = line_len;
668
+ int truncated = 0;
669
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
670
+ keep = st->max_error_bytes;
671
+ truncated = 1;
672
+ }
673
+
674
+ char *raw = malloc(keep + 1);
675
+ if (!raw) return TF_ERROR;
676
+ memcpy(raw, line, keep);
677
+ raw[keep] = '\0';
678
+
679
+ cJSON *obj = cJSON_CreateObject();
680
+ if (!obj) { free(raw); return TF_ERROR; }
681
+ int rc = TF_ERROR;
682
+ if (tf_json_add_string(obj, "type", "csv_field_count") != TF_OK ||
683
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
684
+ tf_json_add_string(obj, "mode", csv_mode_name(st->mode)) != TF_OK ||
685
+ tf_json_add_string(obj, "action", action) != TF_OK ||
686
+ tf_json_add_string(obj, "severity", severity) != TF_OK ||
687
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
688
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
689
+ tf_json_add_number(obj, "expected_fields", (double)st->n_cols) != TF_OK ||
690
+ tf_json_add_number(obj, "actual_fields", (double)n_fields) != TF_OK ||
691
+ tf_json_add_string(obj, "message", message ? message : "CSV row field count differs from header") != TF_OK ||
692
+ tf_json_add_number(obj, "raw_bytes", (double)line_len) != TF_OK ||
693
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
694
+ goto done;
695
+ }
696
+ rc = tf_buffer_write_json_line(side->errors, obj);
697
+ done:
698
+ cJSON_Delete(obj);
699
+ free(raw);
700
+ return rc;
701
+ }
702
+
703
+ static int emit_csv_column_limit_diagnostic(csv_decoder_state *st, const char *line, size_t line_len,
704
+ size_t line_no, size_t byte_offset, size_t n_fields,
705
+ tf_side_channels *side) {
706
+ if (!side || !side->errors) return TF_OK;
707
+
708
+ size_t keep = line_len;
709
+ int truncated = 0;
710
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
711
+ keep = st->max_error_bytes;
712
+ truncated = 1;
713
+ }
714
+
715
+ char *raw = malloc(keep + 1);
716
+ if (!raw) return TF_ERROR;
717
+ memcpy(raw, line, keep);
718
+ raw[keep] = '\0';
719
+
720
+ cJSON *obj = cJSON_CreateObject();
721
+ if (!obj) { free(raw); return TF_ERROR; }
722
+ int rc = TF_ERROR;
723
+ if (tf_json_add_string(obj, "type", "csv_too_many_columns") != TF_OK ||
724
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
725
+ tf_json_add_string(obj, "mode", csv_mode_name(st->mode)) != TF_OK ||
726
+ tf_json_add_string(obj, "action", "fail") != TF_OK ||
727
+ tf_json_add_string(obj, "severity", "error") != TF_OK ||
728
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
729
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
730
+ tf_json_add_number(obj, "max_columns", (double)st->max_columns) != TF_OK ||
731
+ tf_json_add_number(obj, "actual_fields", (double)n_fields) != TF_OK ||
732
+ tf_json_add_string(obj, "message", "CSV record exceeds max_columns") != TF_OK ||
733
+ tf_json_add_number(obj, "raw_bytes", (double)line_len) != TF_OK ||
734
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
735
+ goto done;
736
+ }
737
+ rc = tf_buffer_write_json_line(side->errors, obj);
738
+ done:
739
+ cJSON_Delete(obj);
740
+ free(raw);
741
+ return rc;
742
+ }
743
+
744
+ static int emit_csv_record_size_diagnostic(csv_decoder_state *st, const char *record, size_t record_len,
745
+ size_t line_no, size_t byte_offset, tf_side_channels *side) {
746
+ if (!side || !side->errors) return TF_OK;
747
+
748
+ size_t keep = record_len;
749
+ int truncated = 0;
750
+ if (st->max_error_bytes > 0 && keep > st->max_error_bytes) {
751
+ keep = st->max_error_bytes;
752
+ truncated = 1;
753
+ }
754
+
755
+ char *raw = malloc(keep + 1);
756
+ if (!raw) return TF_ERROR;
757
+ memcpy(raw, record, keep);
758
+ raw[keep] = '\0';
759
+
760
+ cJSON *obj = cJSON_CreateObject();
761
+ if (!obj) { free(raw); return TF_ERROR; }
762
+ int rc = TF_ERROR;
763
+ if (tf_json_add_string(obj, "type", "csv_record_too_large") != TF_OK ||
764
+ tf_json_add_string(obj, "op", "codec.csv.decode") != TF_OK ||
765
+ tf_json_add_string(obj, "action", "fail") != TF_OK ||
766
+ tf_json_add_string(obj, "severity", "error") != TF_OK ||
767
+ tf_json_add_number(obj, "line", (double)line_no) != TF_OK ||
768
+ tf_json_add_number(obj, "byte_offset", (double)byte_offset) != TF_OK ||
769
+ tf_json_add_number(obj, "max_record_bytes", (double)st->max_record_bytes) != TF_OK ||
770
+ tf_json_add_number(obj, "observed_bytes", (double)record_len) != TF_OK ||
771
+ tf_json_add_string(obj, "message", "CSV record exceeds max_record_bytes") != TF_OK ||
772
+ tf_json_add_number(obj, "raw_bytes", (double)record_len) != TF_OK ||
773
+ csv_add_raw_payload(st, obj, raw, truncated) != TF_OK) {
774
+ goto done;
775
+ }
776
+ rc = tf_buffer_write_json_line(side->errors, obj);
777
+ done:
778
+ cJSON_Delete(obj);
779
+ free(raw);
780
+ return rc;
781
+ }
782
+
783
+ static int check_csv_record_limit(csv_decoder_state *st, const uint8_t *buf, size_t line_start,
784
+ size_t observed_len, size_t line_no, size_t byte_offset,
785
+ tf_side_channels *side) {
786
+ if (st->max_record_bytes == 0 || observed_len <= st->max_record_bytes) return TF_OK;
787
+ if (emit_csv_record_size_diagnostic(st, (const char *)buf + line_start, observed_len,
788
+ line_no, byte_offset, side) != TF_OK)
789
+ return TF_ERROR;
790
+ char err[256];
791
+ snprintf(err, sizeof(err), "csv record exceeds max_record_bytes at line %zu: max %zu bytes, observed %zu bytes",
792
+ line_no, st->max_record_bytes, observed_len);
793
+ tf_set_last_error(err);
794
+ return TF_ERROR;
795
+ }
796
+
797
+ static int csv_add_null_literal(csv_decoder_state *st, const char *ptr, size_t len) {
798
+ size_t copy_len = 0;
799
+ if (tf_size_add(len, 1, &copy_len) != TF_OK) return TF_ERROR;
800
+ char *copy = tf_mallocarray_checked(copy_len, sizeof(char));
801
+ if (!copy) return TF_ERROR;
802
+ memcpy(copy, ptr, len);
803
+ copy[len] = '\0';
804
+ size_t next_count = 0;
805
+ if (tf_size_add(st->n_null_literals, 1, &next_count) != TF_OK) {
806
+ free(copy);
807
+ return TF_ERROR;
808
+ }
809
+ char **tmp = tf_reallocarray_checked(st->null_literals, next_count, sizeof(char *));
810
+ if (!tmp) {
811
+ free(copy);
812
+ return TF_ERROR;
813
+ }
814
+ st->null_literals = tmp;
815
+ st->null_literals[st->n_null_literals++] = copy;
816
+ return TF_OK;
817
+ }
818
+
819
+ static int csv_add_null_literal_token(csv_decoder_state *st, const char *tok) {
820
+ if (!tok) return TF_OK;
821
+ return csv_add_null_literal(st, tok, strlen(tok));
822
+ }
823
+
824
+ static void csv_free_schema_arrays(char **col_names, tf_type *col_types, size_t n_names) {
825
+ if (col_names) {
826
+ for (size_t i = 0; i < n_names; i++) free(col_names[i]);
827
+ }
828
+ free(col_names);
829
+ free(col_types);
830
+ }
831
+
832
+ static int csv_init_schema(csv_decoder_state *st, const field_slice *fields,
833
+ size_t n_fields, int synthetic_names) {
834
+ char **col_names = tf_callocarray_checked(n_fields ? n_fields : 1, sizeof(char *));
835
+ tf_type *col_types = tf_callocarray_checked(n_fields ? n_fields : 1, sizeof(tf_type));
836
+ if (!col_names || !col_types) {
837
+ free(col_names);
838
+ free(col_types);
839
+ return TF_ERROR;
840
+ }
841
+
842
+ size_t schema_bytes = 0;
843
+ for (size_t i = 0; i < n_fields; i++) {
844
+ char synthetic[64];
845
+ const char *name_ptr = fields[i].ptr;
846
+ size_t name_len = fields[i].len;
847
+ if (synthetic_names) {
848
+ int n = snprintf(synthetic, sizeof(synthetic), "col%zu", i + 1);
849
+ if (n <= 0 || (size_t)n >= sizeof(synthetic)) {
850
+ csv_free_schema_arrays(col_names, col_types, i);
851
+ return TF_ERROR;
852
+ }
853
+ name_ptr = synthetic;
854
+ name_len = (size_t)n;
855
+ }
856
+
857
+ size_t name_bytes = 0, next_schema_bytes = 0;
858
+ if (tf_check_byte_limit(name_len, TF_MAX_COLUMN_NAME_BYTES,
859
+ "csv", "column name") != TF_OK ||
860
+ tf_size_add(name_len, 1, &name_bytes) != TF_OK ||
861
+ tf_size_add(schema_bytes, name_bytes, &next_schema_bytes) != TF_OK ||
862
+ tf_check_byte_limit(next_schema_bytes, TF_MAX_SCHEMA_BYTES,
863
+ "csv", "schema") != TF_OK) {
864
+ csv_free_schema_arrays(col_names, col_types, i);
865
+ return TF_ERROR;
866
+ }
867
+ char *name = malloc(name_bytes);
868
+ if (!name) {
869
+ csv_free_schema_arrays(col_names, col_types, i);
870
+ return TF_ERROR;
871
+ }
872
+ memcpy(name, name_ptr, name_len);
873
+ name[name_len] = '\0';
874
+ col_names[i] = name;
875
+ col_types[i] = TF_TYPE_NULL;
876
+ schema_bytes = next_schema_bytes;
877
+ }
878
+
879
+ st->n_cols = n_fields;
880
+ st->col_names = col_names;
881
+ st->col_types = col_types;
882
+ st->schema_ready = 1;
883
+ st->after_input_boundary = 0;
884
+ return TF_OK;
885
+ }
886
+
887
+ static int csv_add_null_literal_list(csv_decoder_state *st, const char *list) {
888
+ if (!list) return TF_OK;
889
+ const char *start = list;
890
+ for (const char *p = list; ; p++) {
891
+ if (*p == ',' || *p == '\0') {
892
+ if (csv_add_null_literal(st, start, (size_t)(p - start)) != TF_OK) return TF_ERROR;
893
+ if (*p == '\0') break;
894
+ start = p + 1;
895
+ }
896
+ }
897
+ return TF_OK;
898
+ }
899
+
900
+ static int csv_configure_null_literals(csv_decoder_state *st, const cJSON *args) {
901
+ const cJSON *nulls = cJSON_GetObjectItemCaseSensitive(args, "nulls");
902
+ if (!nulls) nulls = cJSON_GetObjectItemCaseSensitive(args, "na");
903
+ if (!nulls) return TF_OK;
904
+
905
+ if (cJSON_IsString(nulls)) {
906
+ return csv_add_null_literal_list(st, nulls->valuestring ? nulls->valuestring : "");
907
+ }
908
+ if (cJSON_IsArray(nulls)) {
909
+ const cJSON *item = NULL;
910
+ cJSON_ArrayForEach(item, nulls) {
911
+ if (cJSON_IsString(item)) {
912
+ if (csv_add_null_literal_token(st, item->valuestring) != TF_OK) return TF_ERROR;
913
+ }
914
+ }
915
+ }
916
+ return TF_OK;
917
+ }
918
+
919
+ static int csv_field_is_null(const csv_decoder_state *st, const field_slice *field) {
920
+ if (field->quoted && !st->quoted_nulls) return 0;
921
+ if (field->len == 0) return 1;
922
+ for (size_t i = 0; i < st->n_null_literals; i++) {
923
+ const char *lit = st->null_literals[i];
924
+ size_t len = strlen(lit);
925
+ if (field->len == len && memcmp(field->ptr, lit, len) == 0) return 1;
926
+ }
927
+ return 0;
928
+ }
929
+
930
+ static tf_type detect_type_field(const csv_decoder_state *st, const field_slice *field) {
931
+ if (csv_field_is_null(st, field)) return TF_TYPE_NULL;
932
+ if (field->len == 0) return TF_TYPE_STRING;
933
+ return detect_type_slice(field->ptr, field->len);
934
+ }
935
+
441
936
  /*
442
937
  * Create a batch with all STRING columns (for type detection phase).
443
938
  * During this phase we don't know final types yet, so everything
@@ -447,7 +942,10 @@ static tf_batch *make_string_batch(csv_decoder_state *st) {
447
942
  tf_batch *b = tf_batch_create(st->n_cols, st->batch_size);
448
943
  if (!b) return NULL;
449
944
  for (size_t i = 0; i < st->n_cols; i++) {
450
- tf_batch_set_schema(b, i, st->col_names[i], TF_TYPE_STRING);
945
+ if (tf_batch_set_schema(b, i, st->col_names[i], TF_TYPE_STRING) != TF_OK) {
946
+ tf_batch_free(b);
947
+ return NULL;
948
+ }
451
949
  }
452
950
  return b;
453
951
  }
@@ -461,7 +959,10 @@ static tf_batch *make_typed_batch(csv_decoder_state *st) {
461
959
  tf_batch *b = tf_batch_create(st->n_cols, st->batch_size);
462
960
  if (!b) return NULL;
463
961
  for (size_t i = 0; i < st->n_cols; i++) {
464
- tf_batch_set_schema(b, i, st->col_names[i], st->col_types[i]);
962
+ if (tf_batch_set_schema(b, i, st->col_names[i], st->col_types[i]) != TF_OK) {
963
+ tf_batch_free(b);
964
+ return NULL;
965
+ }
465
966
  }
466
967
  return b;
467
968
  }
@@ -471,114 +972,114 @@ static tf_batch *make_typed_batch(csv_decoder_state *st) {
471
972
  * Used during the type detection phase (first batch).
472
973
  * Copies slice content into the batch's arena.
473
974
  */
474
- static void add_row_strings(tf_batch *b, const field_slice *fields,
475
- size_t n_fields, size_t n_cols) {
975
+ static int add_row_strings(tf_batch *b, const csv_decoder_state *st,
976
+ const field_slice *fields, size_t n_fields,
977
+ size_t n_cols) {
476
978
  size_t row = b->n_rows;
477
979
  size_t cols = n_cols < n_fields ? n_cols : n_fields;
478
980
 
479
981
  for (size_t i = 0; i < cols; i++) {
480
- if (fields[i].len == 0) {
481
- b->nulls[i][row] = 1;
982
+ if (csv_field_is_null(st, &fields[i])) {
983
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
482
984
  } else {
483
- /* Copy slice into batch arena as null-terminated string */
484
- char *copy = tf_arena_alloc(b->arena, fields[i].len + 1);
485
- memcpy(copy, fields[i].ptr, fields[i].len);
486
- copy[fields[i].len] = '\0';
487
- ((char **)b->columns[i])[row] = copy;
488
- b->nulls[i][row] = 0;
985
+ if (tf_batch_set_string_len(b, row, i, fields[i].ptr, fields[i].len) != TF_OK)
986
+ return TF_ERROR;
489
987
  }
490
988
  }
491
989
  /* Null-fill extra columns */
492
990
  for (size_t i = cols; i < n_cols; i++) {
493
- b->nulls[i][row] = 1;
991
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
494
992
  }
495
- b->n_rows = row + 1;
993
+ if (tf_batch_expose_row(b, row) != TF_OK) return TF_ERROR;
994
+ return TF_OK;
496
995
  }
497
996
 
498
997
  /*
499
998
  * Add a row by parsing field slices directly into typed columns.
500
999
  * Used after types are frozen (all batches after the first).
501
1000
  *
502
- * Writes directly to column arrays, bypassing tf_batch_set_*()
503
- * bounds/type checks for speed in this hot path.
1001
+ * Uses checked batch setters so failed writes cannot expose a partial row.
504
1002
  */
505
- static void add_row_typed(tf_batch *b, const field_slice *fields,
506
- size_t n_fields, size_t n_cols,
507
- const tf_type *types) {
1003
+ static int add_row_typed(tf_batch *b, const csv_decoder_state *st,
1004
+ const field_slice *fields, size_t n_fields,
1005
+ size_t n_cols, const tf_type *types) {
508
1006
  size_t row = b->n_rows;
509
1007
  size_t cols = n_cols < n_fields ? n_cols : n_fields;
510
1008
 
511
1009
  for (size_t i = 0; i < cols; i++) {
512
- if (fields[i].len == 0) {
513
- b->nulls[i][row] = 1;
1010
+ if (csv_field_is_null(st, &fields[i])) {
1011
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
514
1012
  continue;
515
1013
  }
516
1014
  switch (types[i]) {
1015
+ case TF_TYPE_BOOL: {
1016
+ int v;
1017
+ if (fast_bool(fields[i].ptr, fields[i].len, &v)) {
1018
+ if (tf_batch_set_bool(b, row, i, v != 0) != TF_OK) return TF_ERROR;
1019
+ } else {
1020
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
1021
+ }
1022
+ break;
1023
+ }
517
1024
  case TF_TYPE_INT64: {
518
1025
  int64_t v;
519
1026
  if (fast_int64(fields[i].ptr, fields[i].len, &v)) {
520
- ((int64_t *)b->columns[i])[row] = v;
521
- b->nulls[i][row] = 0;
1027
+ if (tf_batch_set_int64(b, row, i, v) != TF_OK) return TF_ERROR;
522
1028
  } else {
523
- b->nulls[i][row] = 1;
1029
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
524
1030
  }
525
1031
  break;
526
1032
  }
527
1033
  case TF_TYPE_FLOAT64: {
528
1034
  double v;
529
1035
  if (fast_double(fields[i].ptr, fields[i].len, &v)) {
530
- ((double *)b->columns[i])[row] = v;
531
- b->nulls[i][row] = 0;
1036
+ if (tf_batch_set_float64(b, row, i, v) != TF_OK) return TF_ERROR;
532
1037
  } else {
533
- b->nulls[i][row] = 1;
1038
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
534
1039
  }
535
1040
  break;
536
1041
  }
537
1042
  case TF_TYPE_STRING: {
538
- char *copy = tf_arena_alloc(b->arena, fields[i].len + 1);
539
- memcpy(copy, fields[i].ptr, fields[i].len);
540
- copy[fields[i].len] = '\0';
541
- ((char **)b->columns[i])[row] = copy;
542
- b->nulls[i][row] = 0;
1043
+ if (tf_batch_set_string_len(b, row, i, fields[i].ptr, fields[i].len) != TF_OK)
1044
+ return TF_ERROR;
543
1045
  break;
544
1046
  }
545
1047
  case TF_TYPE_DATE: {
546
1048
  int32_t v;
547
1049
  if (fast_date(fields[i].ptr, fields[i].len, &v)) {
548
- ((int32_t *)b->columns[i])[row] = v;
549
- b->nulls[i][row] = 0;
1050
+ if (tf_batch_set_date(b, row, i, v) != TF_OK) return TF_ERROR;
550
1051
  } else {
551
- b->nulls[i][row] = 1;
1052
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
552
1053
  }
553
1054
  break;
554
1055
  }
555
1056
  case TF_TYPE_TIMESTAMP: {
556
1057
  int64_t v;
557
1058
  if (fast_timestamp(fields[i].ptr, fields[i].len, &v)) {
558
- ((int64_t *)b->columns[i])[row] = v;
559
- b->nulls[i][row] = 0;
1059
+ if (tf_batch_set_timestamp(b, row, i, v) != TF_OK) return TF_ERROR;
560
1060
  } else {
561
1061
  /* Also try parsing a date-only string as timestamp at midnight */
562
1062
  int32_t dv;
563
1063
  if (fast_date(fields[i].ptr, fields[i].len, &dv)) {
564
- ((int64_t *)b->columns[i])[row] = (int64_t)dv * 86400LL * 1000000LL;
565
- b->nulls[i][row] = 0;
1064
+ v = (int64_t)dv * 86400LL * 1000000LL;
1065
+ if (tf_batch_set_timestamp(b, row, i, v) != TF_OK) return TF_ERROR;
566
1066
  } else {
567
- b->nulls[i][row] = 1;
1067
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
568
1068
  }
569
1069
  }
570
1070
  break;
571
1071
  }
572
1072
  default:
573
- b->nulls[i][row] = 1;
1073
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
574
1074
  break;
575
1075
  }
576
1076
  }
577
1077
  /* Null-fill extra columns */
578
1078
  for (size_t i = cols; i < n_cols; i++) {
579
- b->nulls[i][row] = 1;
1079
+ if (tf_batch_set_null(b, row, i) != TF_OK) return TF_ERROR;
580
1080
  }
581
- b->n_rows = row + 1;
1081
+ if (tf_batch_expose_row(b, row) != TF_OK) return TF_ERROR;
1082
+ return TF_OK;
582
1083
  }
583
1084
 
584
1085
  /*
@@ -592,78 +1093,134 @@ static tf_batch *convert_batch_types(csv_decoder_state *st) {
592
1093
  if (!dst) return NULL;
593
1094
 
594
1095
  for (size_t i = 0; i < st->n_cols; i++) {
595
- tf_batch_set_schema(dst, i, st->col_names[i], st->col_types[i]);
1096
+ if (tf_batch_set_schema(dst, i, st->col_names[i], st->col_types[i]) != TF_OK) {
1097
+ tf_batch_free(dst);
1098
+ return NULL;
1099
+ }
596
1100
  }
597
1101
 
598
1102
  for (size_t r = 0; r < src->n_rows; r++) {
599
1103
  for (size_t c = 0; c < st->n_cols; c++) {
600
1104
  if (src->nulls[c][r]) {
601
- dst->nulls[c][r] = 1;
1105
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1106
+ tf_batch_free(dst);
1107
+ return NULL;
1108
+ }
602
1109
  continue;
603
1110
  }
604
1111
  const char *val = ((char **)src->columns[c])[r];
605
- if (!val || !*val) {
606
- dst->nulls[c][r] = 1;
1112
+ if (!val) {
1113
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1114
+ tf_batch_free(dst);
1115
+ return NULL;
1116
+ }
607
1117
  continue;
608
1118
  }
609
1119
  size_t vlen = strlen(val);
610
1120
  switch (st->col_types[c]) {
1121
+ case TF_TYPE_BOOL: {
1122
+ int v;
1123
+ if (fast_bool(val, vlen, &v)) {
1124
+ if (tf_batch_set_bool(dst, r, c, v != 0) != TF_OK) {
1125
+ tf_batch_free(dst);
1126
+ return NULL;
1127
+ }
1128
+ } else {
1129
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1130
+ tf_batch_free(dst);
1131
+ return NULL;
1132
+ }
1133
+ }
1134
+ break;
1135
+ }
611
1136
  case TF_TYPE_INT64: {
612
1137
  int64_t v;
613
1138
  if (fast_int64(val, vlen, &v)) {
614
- ((int64_t *)dst->columns[c])[r] = v;
615
- dst->nulls[c][r] = 0;
1139
+ if (tf_batch_set_int64(dst, r, c, v) != TF_OK) {
1140
+ tf_batch_free(dst);
1141
+ return NULL;
1142
+ }
616
1143
  } else {
617
- dst->nulls[c][r] = 1;
1144
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1145
+ tf_batch_free(dst);
1146
+ return NULL;
1147
+ }
618
1148
  }
619
1149
  break;
620
1150
  }
621
1151
  case TF_TYPE_FLOAT64: {
622
1152
  double v;
623
1153
  if (fast_double(val, vlen, &v)) {
624
- ((double *)dst->columns[c])[r] = v;
625
- dst->nulls[c][r] = 0;
1154
+ if (tf_batch_set_float64(dst, r, c, v) != TF_OK) {
1155
+ tf_batch_free(dst);
1156
+ return NULL;
1157
+ }
626
1158
  } else {
627
- dst->nulls[c][r] = 1;
1159
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1160
+ tf_batch_free(dst);
1161
+ return NULL;
1162
+ }
628
1163
  }
629
1164
  break;
630
1165
  }
631
- case TF_TYPE_STRING:
632
- ((char **)dst->columns[c])[r] = tf_arena_strdup(dst->arena, val);
633
- dst->nulls[c][r] = 0;
1166
+ case TF_TYPE_STRING: {
1167
+ if (tf_batch_set_string(dst, r, c, val) != TF_OK) {
1168
+ tf_batch_free(dst);
1169
+ return NULL;
1170
+ }
634
1171
  break;
1172
+ }
635
1173
  case TF_TYPE_DATE: {
636
1174
  int32_t dv;
637
1175
  if (fast_date(val, vlen, &dv)) {
638
- ((int32_t *)dst->columns[c])[r] = dv;
639
- dst->nulls[c][r] = 0;
1176
+ if (tf_batch_set_date(dst, r, c, dv) != TF_OK) {
1177
+ tf_batch_free(dst);
1178
+ return NULL;
1179
+ }
640
1180
  } else {
641
- dst->nulls[c][r] = 1;
1181
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1182
+ tf_batch_free(dst);
1183
+ return NULL;
1184
+ }
642
1185
  }
643
1186
  break;
644
1187
  }
645
1188
  case TF_TYPE_TIMESTAMP: {
646
1189
  int64_t tv;
647
1190
  if (fast_timestamp(val, vlen, &tv)) {
648
- ((int64_t *)dst->columns[c])[r] = tv;
649
- dst->nulls[c][r] = 0;
1191
+ if (tf_batch_set_timestamp(dst, r, c, tv) != TF_OK) {
1192
+ tf_batch_free(dst);
1193
+ return NULL;
1194
+ }
650
1195
  } else {
651
1196
  int32_t dv;
652
1197
  if (fast_date(val, vlen, &dv)) {
653
- ((int64_t *)dst->columns[c])[r] = (int64_t)dv * 86400LL * 1000000LL;
654
- dst->nulls[c][r] = 0;
1198
+ tv = (int64_t)dv * 86400LL * 1000000LL;
1199
+ if (tf_batch_set_timestamp(dst, r, c, tv) != TF_OK) {
1200
+ tf_batch_free(dst);
1201
+ return NULL;
1202
+ }
655
1203
  } else {
656
- dst->nulls[c][r] = 1;
1204
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1205
+ tf_batch_free(dst);
1206
+ return NULL;
1207
+ }
657
1208
  }
658
1209
  }
659
1210
  break;
660
1211
  }
661
1212
  default:
662
- dst->nulls[c][r] = 1;
1213
+ if (tf_batch_set_null(dst, r, c) != TF_OK) {
1214
+ tf_batch_free(dst);
1215
+ return NULL;
1216
+ }
663
1217
  break;
664
1218
  }
665
1219
  }
666
- dst->n_rows = r + 1;
1220
+ if (tf_batch_expose_row(dst, r) != TF_OK) {
1221
+ tf_batch_free(dst);
1222
+ return NULL;
1223
+ }
667
1224
  }
668
1225
 
669
1226
  return dst;
@@ -674,14 +1231,47 @@ static tf_batch *convert_batch_types(csv_decoder_state *st) {
674
1231
  */
675
1232
  static int emit_batch(tf_batch *batch, tf_batch ***out, size_t *n_out, size_t *out_cap) {
676
1233
  if (*n_out >= *out_cap) {
677
- *out_cap = (*out_cap == 0) ? 4 : *out_cap * 2;
678
- *out = realloc(*out, *out_cap * sizeof(tf_batch *));
679
- if (!*out) return TF_ERROR;
1234
+ size_t need = 0;
1235
+ size_t new_cap = 0;
1236
+ if (tf_size_add(*n_out, 1, &need) != TF_OK ||
1237
+ tf_size_grow_pow2(*out_cap, need, 4, &new_cap) != TF_OK) {
1238
+ return TF_ERROR;
1239
+ }
1240
+ tf_batch **tmp = tf_reallocarray_checked(*out, new_cap, sizeof(tf_batch *));
1241
+ if (!tmp) return TF_ERROR;
1242
+ *out = tmp;
1243
+ *out_cap = new_cap;
680
1244
  }
681
1245
  (*out)[(*n_out)++] = batch;
682
1246
  return TF_OK;
683
1247
  }
684
1248
 
1249
+ static tf_batch *make_schema_only_batch(csv_decoder_state *st) {
1250
+ for (size_t i = 0; i < st->n_cols; i++) {
1251
+ if (st->col_types[i] == TF_TYPE_NULL) st->col_types[i] = TF_TYPE_STRING;
1252
+ }
1253
+ tf_batch *b = tf_batch_create(st->n_cols, 0);
1254
+ if (!b) return NULL;
1255
+ for (size_t i = 0; i < st->n_cols; i++) {
1256
+ if (tf_batch_set_schema(b, i, st->col_names[i], st->col_types[i]) != TF_OK) {
1257
+ tf_batch_free(b);
1258
+ return NULL;
1259
+ }
1260
+ }
1261
+ return b;
1262
+ }
1263
+
1264
+ static int csv_fields_match_header(const csv_decoder_state *st, const field_slice *fields, size_t n_fields) {
1265
+ if (!st || !st->schema_ready || n_fields != st->n_cols) return 0;
1266
+ for (size_t i = 0; i < st->n_cols; i++) {
1267
+ const char *name = st->col_names[i] ? st->col_names[i] : "";
1268
+ size_t name_len = strlen(name);
1269
+ if (fields[i].len != name_len) return 0;
1270
+ if (name_len > 0 && memcmp(fields[i].ptr, name, name_len) != 0) return 0;
1271
+ }
1272
+ return 1;
1273
+ }
1274
+
685
1275
  /*
686
1276
  * Process a single complete CSV line.
687
1277
  *
@@ -690,45 +1280,102 @@ static int emit_batch(tf_batch *batch, tf_batch ***out, size_t *n_out, size_t *o
690
1280
  * When batch is full → emit it and start a new one.
691
1281
  */
692
1282
  static int process_line(csv_decoder_state *st, const char *line, size_t line_len,
693
- tf_batch ***out, size_t *n_out, size_t *out_cap) {
1283
+ size_t line_no, size_t byte_offset, tf_batch ***out,
1284
+ size_t *n_out, size_t *out_cap, tf_side_channels *side) {
1285
+ if (st->skipped_rows < st->skip_rows) {
1286
+ st->skipped_rows++;
1287
+ return TF_OK;
1288
+ }
1289
+
1290
+ size_t comment_pos = 0;
1291
+ int had_comment = csv_find_comment(st, line, line_len, &comment_pos);
1292
+ if (had_comment) line_len = comment_pos;
1293
+ if ((had_comment && csv_line_is_blank(line, line_len)) ||
1294
+ (st->skip_empty_rows && csv_line_is_blank(line, line_len))) {
1295
+ return TF_OK;
1296
+ }
1297
+
1298
+ if (st->schema_ready && st->limit_rows && st->data_rows_read >= st->n_max) {
1299
+ return TF_OK;
1300
+ }
1301
+
694
1302
  /* Reset field arena — escaped field data from previous line is discarded */
695
1303
  tf_arena_reset(st->field_arena);
696
1304
 
697
- /* Parse the line into zero-copy field slices */
698
- size_t n_fields = parse_csv_fields(line, line_len, st->delimiter,
699
- st->fields, MAX_COLS, st->field_arena);
1305
+ /* Parse the line into zero-copy field slices while counting every field. */
1306
+ size_t n_fields = 0;
1307
+ if (parse_csv_fields(line, line_len, st->delimiter,
1308
+ st->fields, st->fields_cap, st->field_arena,
1309
+ st->trim_ws, &n_fields) != TF_OK) {
1310
+ tf_set_last_error("csv: failed to parse fields");
1311
+ return TF_ERROR;
1312
+ }
1313
+ if (n_fields > st->max_columns) {
1314
+ if (emit_csv_column_limit_diagnostic(st, line, line_len, line_no, byte_offset,
1315
+ n_fields, side) != TF_OK)
1316
+ return TF_ERROR;
1317
+ char err[256];
1318
+ snprintf(err, sizeof(err), "csv record exceeds max_columns at line %zu: max %zu columns, observed %zu fields",
1319
+ line_no, st->max_columns, n_fields);
1320
+ tf_set_last_error(err);
1321
+ return TF_ERROR;
1322
+ }
1323
+
1324
+ if (st->schema_ready && st->has_header && st->skip_repeated_header && st->after_input_boundary) {
1325
+ st->after_input_boundary = 0;
1326
+ if (csv_fields_match_header(st, st->fields, n_fields)) {
1327
+ return TF_OK;
1328
+ }
1329
+ }
700
1330
 
701
- /* --- First line: extract column headers --- */
1331
+ /* --- First record: extract column headers or synthesize headerless names. --- */
702
1332
  if (!st->schema_ready) {
703
- st->n_cols = n_fields;
704
- st->col_names = malloc(n_fields * sizeof(char *));
705
- st->col_types = calloc(n_fields, sizeof(tf_type));
706
- if (!st->col_names || !st->col_types) return TF_ERROR;
707
-
708
- for (size_t i = 0; i < n_fields; i++) {
709
- /* Headers must outlive the line buffer, so we copy them */
710
- char *name = malloc(st->fields[i].len + 1);
711
- if (!name) return TF_ERROR;
712
- memcpy(name, st->fields[i].ptr, st->fields[i].len);
713
- name[st->fields[i].len] = '\0';
714
- st->col_names[i] = name;
715
- st->col_types[i] = TF_TYPE_NULL;
1333
+ if (csv_init_schema(st, st->fields, n_fields, !st->has_header) != TF_OK) return TF_ERROR;
1334
+ if (st->has_header) return TF_OK;
1335
+ if (st->limit_rows && st->data_rows_read >= st->n_max) return TF_OK;
1336
+ }
1337
+
1338
+ /* --- Empty lines: treat as all-null row --- */
1339
+ if (n_fields == 0 && st->n_cols > 0) {
1340
+ for (size_t i = 0; i < st->n_cols; i++) {
1341
+ st->fields[i].ptr = "";
1342
+ st->fields[i].len = 0;
1343
+ st->fields[i].quoted = 0;
716
1344
  }
717
- st->schema_ready = 1;
718
- return TF_OK;
1345
+ n_fields = st->n_cols;
719
1346
  }
720
1347
 
721
- /* --- Repair mode: normalize field count to match header --- */
722
- if (st->repair && n_fields != st->n_cols) {
1348
+ /* --- Field-count policy: legacy permissive, observable repair, or strict failure. --- */
1349
+ if (n_fields != st->n_cols) {
1350
+ const char *message = n_fields < st->n_cols
1351
+ ? "CSV row has fewer fields than header"
1352
+ : "CSV row has more fields than header";
1353
+ if (st->mode == CSV_MODE_STRICT) {
1354
+ if (emit_csv_field_count_diagnostic(st, line, line_len, line_no, byte_offset, n_fields,
1355
+ "fail", "error", message, side) != TF_OK)
1356
+ return TF_ERROR;
1357
+ char err[256];
1358
+ snprintf(err, sizeof(err), "csv strict field count mismatch at line %zu: expected %zu fields, got %zu",
1359
+ line_no, st->n_cols, n_fields);
1360
+ tf_set_last_error(err);
1361
+ return TF_ERROR;
1362
+ }
1363
+ if (st->mode == CSV_MODE_REPAIR) {
1364
+ if (emit_csv_field_count_diagnostic(st, line, line_len, line_no, byte_offset, n_fields,
1365
+ "repair", "warning", message, side) != TF_OK)
1366
+ return TF_ERROR;
1367
+ if (emit_csv_repair_audit(st, line, line_len, line_no, byte_offset, n_fields,
1368
+ message, side) != TF_OK)
1369
+ return TF_ERROR;
1370
+ }
723
1371
  if (n_fields < st->n_cols) {
724
- /* Pad short row with empty fields */
725
- for (size_t i = n_fields; i < st->n_cols && i < MAX_COLS; i++) {
1372
+ for (size_t i = n_fields; i < st->n_cols; i++) {
726
1373
  st->fields[i].ptr = "";
727
1374
  st->fields[i].len = 0;
1375
+ st->fields[i].quoted = 0;
728
1376
  }
729
1377
  n_fields = st->n_cols;
730
1378
  } else {
731
- /* Truncate long row */
732
1379
  n_fields = st->n_cols;
733
1380
  }
734
1381
  }
@@ -747,15 +1394,18 @@ static int process_line(csv_decoder_state *st, const char *line, size_t line_len
747
1394
  if (!st->types_frozen) {
748
1395
  /* Type detection phase: detect types and store as STRING */
749
1396
  for (size_t i = 0; i < n_fields && i < st->n_cols; i++) {
750
- tf_type t = detect_type_slice(st->fields[i].ptr, st->fields[i].len);
1397
+ tf_type t = detect_type_field(st, &st->fields[i]);
751
1398
  st->col_types[i] = widen_type(st->col_types[i], t);
752
1399
  }
753
- add_row_strings(st->batch, st->fields, n_fields, st->n_cols);
1400
+ if (add_row_strings(st->batch, st, st->fields, n_fields, st->n_cols) != TF_OK)
1401
+ return TF_ERROR;
754
1402
  } else {
755
1403
  /* Direct parse phase: parse directly to typed columns */
756
- add_row_typed(st->batch, st->fields, n_fields, st->n_cols, st->col_types);
1404
+ if (add_row_typed(st->batch, st, st->fields, n_fields, st->n_cols, st->col_types) != TF_OK)
1405
+ return TF_ERROR;
757
1406
  }
758
1407
  st->rows_buffered++;
1408
+ st->data_rows_read++;
759
1409
 
760
1410
  /* --- Emit batch if full --- */
761
1411
  if (st->rows_buffered >= st->batch_size) {
@@ -771,7 +1421,10 @@ static int process_line(csv_decoder_state *st, const char *line, size_t line_len
771
1421
  tf_batch_free(st->batch);
772
1422
  st->batch = NULL;
773
1423
  st->types_frozen = 1;
774
- if (emit_batch(final, out, n_out, out_cap) != TF_OK) return TF_ERROR;
1424
+ if (emit_batch(final, out, n_out, out_cap) != TF_OK) {
1425
+ tf_batch_free(final);
1426
+ return TF_ERROR;
1427
+ }
775
1428
  } else {
776
1429
  /* Already typed, emit directly (no conversion needed) */
777
1430
  if (emit_batch(st->batch, out, n_out, out_cap) != TF_OK) return TF_ERROR;
@@ -791,7 +1444,7 @@ static int process_line(csv_decoder_state *st, const char *line, size_t line_len
791
1444
  * line remains in line_buf for the next call.
792
1445
  */
793
1446
  static int csv_decode(tf_decoder *self, const uint8_t *data, size_t len,
794
- tf_batch ***out, size_t *n_out) {
1447
+ tf_batch ***out, size_t *n_out, tf_side_channels *side) {
795
1448
  csv_decoder_state *st = self->state;
796
1449
  *out = NULL;
797
1450
  *n_out = 0;
@@ -804,28 +1457,56 @@ static int csv_decode(tf_decoder *self, const uint8_t *data, size_t len,
804
1457
  uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
805
1458
  size_t buf_len = st->line_buf.len - st->line_buf.read_pos;
806
1459
 
1460
+ size_t base_offset = st->byte_offset;
807
1461
  size_t line_start = 0;
808
1462
  int in_quotes = 0;
1463
+ int at_field_start = 1;
809
1464
  for (size_t i = 0; i < buf_len; i++) {
810
- if (buf[i] == '"') {
811
- in_quotes = !in_quotes;
812
- } else if (!in_quotes && (buf[i] == '\n' || buf[i] == '\r')) {
1465
+ size_t current_record_len = i - line_start;
1466
+ if (check_csv_record_limit(st, buf, line_start, current_record_len,
1467
+ st->line_number + 1, base_offset + line_start, side) != TF_OK)
1468
+ return TF_ERROR;
1469
+
1470
+ if (in_quotes) {
1471
+ if (buf[i] == '"') {
1472
+ if (i + 1 < buf_len && buf[i + 1] == '"') {
1473
+ i++; /* RFC 4180 escaped quote: doubled quote inside quoted field. */
1474
+ } else {
1475
+ in_quotes = 0;
1476
+ at_field_start = 0;
1477
+ }
1478
+ }
1479
+ continue;
1480
+ }
1481
+
1482
+ if (buf[i] == '"' && at_field_start) {
1483
+ in_quotes = 1;
1484
+ at_field_start = 0;
1485
+ } else if (buf[i] == st->delimiter) {
1486
+ at_field_start = 1;
1487
+ } else if (buf[i] == '\n' || buf[i] == '\r') {
813
1488
  size_t line_len = i - line_start;
814
1489
  /* Handle \r\n */
815
1490
  if (buf[i] == '\r' && i + 1 < buf_len && buf[i + 1] == '\n') {
816
1491
  i++;
817
1492
  }
818
- if (line_len > 0) {
1493
+ size_t line_no = ++st->line_number;
1494
+ size_t record_offset = base_offset + line_start;
1495
+ if (line_len > 0 || st->schema_ready) {
819
1496
  if (process_line(st, (const char *)buf + line_start, line_len,
820
- out, n_out, &out_cap) != TF_OK)
1497
+ line_no, record_offset, out, n_out, &out_cap, side) != TF_OK)
821
1498
  return TF_ERROR;
822
1499
  }
823
1500
  line_start = i + 1;
1501
+ at_field_start = 1;
1502
+ } else {
1503
+ at_field_start = 0;
824
1504
  }
825
1505
  }
826
1506
 
827
1507
  /* Move unconsumed data to start of buffer */
828
1508
  st->line_buf.read_pos += line_start;
1509
+ st->byte_offset += line_start;
829
1510
  tf_buffer_compact(&st->line_buf);
830
1511
 
831
1512
  return TF_OK;
@@ -834,7 +1515,7 @@ static int csv_decode(tf_decoder *self, const uint8_t *data, size_t len,
834
1515
  /*
835
1516
  * Flush: process any remaining partial line and emit the final batch.
836
1517
  */
837
- static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
1518
+ static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out, tf_side_channels *side) {
838
1519
  csv_decoder_state *st = self->state;
839
1520
  *out = NULL;
840
1521
  *n_out = 0;
@@ -844,10 +1525,25 @@ static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
844
1525
  size_t remaining = tf_buffer_readable(&st->line_buf);
845
1526
  if (remaining > 0) {
846
1527
  uint8_t *buf = st->line_buf.data + st->line_buf.read_pos;
1528
+ size_t line_no = ++st->line_number;
1529
+ size_t record_offset = st->byte_offset;
1530
+ if (check_csv_record_limit(st, buf, 0, remaining, line_no, record_offset, side) != TF_OK)
1531
+ return TF_ERROR;
847
1532
  if (process_line(st, (const char *)buf, remaining,
848
- out, n_out, &out_cap) != TF_OK)
1533
+ line_no, record_offset, out, n_out, &out_cap, side) != TF_OK)
849
1534
  return TF_ERROR;
850
1535
  st->line_buf.read_pos = st->line_buf.len;
1536
+ st->byte_offset += remaining;
1537
+ }
1538
+
1539
+ if (st->schema_ready && st->limit_rows && st->n_max == 0 && !st->batch) {
1540
+ tf_batch *schema_only = make_schema_only_batch(st);
1541
+ if (!schema_only) return TF_ERROR;
1542
+ st->types_frozen = 1;
1543
+ if (emit_batch(schema_only, out, n_out, &out_cap) != TF_OK) {
1544
+ tf_batch_free(schema_only);
1545
+ return TF_ERROR;
1546
+ }
851
1547
  }
852
1548
 
853
1549
  /* Emit any remaining partial batch */
@@ -862,7 +1558,10 @@ static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
862
1558
  if (!final) return TF_ERROR;
863
1559
  tf_batch_free(st->batch);
864
1560
  st->batch = NULL;
865
- if (emit_batch(final, out, n_out, &out_cap) != TF_OK) return TF_ERROR;
1561
+ if (emit_batch(final, out, n_out, &out_cap) != TF_OK) {
1562
+ tf_batch_free(final);
1563
+ return TF_ERROR;
1564
+ }
866
1565
  } else {
867
1566
  /* Already typed, emit directly */
868
1567
  if (emit_batch(st->batch, out, n_out, &out_cap) != TF_OK) return TF_ERROR;
@@ -871,6 +1570,7 @@ static int csv_flush(tf_decoder *self, tf_batch ***out, size_t *n_out) {
871
1570
  st->rows_buffered = 0;
872
1571
  }
873
1572
 
1573
+ st->after_input_boundary = 1;
874
1574
  return TF_OK;
875
1575
  }
876
1576
 
@@ -880,9 +1580,16 @@ static void csv_decoder_destroy(tf_decoder *self) {
880
1580
  tf_buffer_free(&st->line_buf);
881
1581
  if (st->batch) tf_batch_free(st->batch);
882
1582
  if (st->field_arena) tf_arena_free(st->field_arena);
883
- for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
1583
+ free(st->fields);
1584
+ if (st->col_names) {
1585
+ for (size_t i = 0; i < st->n_cols; i++) free(st->col_names[i]);
1586
+ }
884
1587
  free(st->col_names);
885
1588
  free(st->col_types);
1589
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1590
+ free(st->null_literals);
1591
+ free(st->comment);
1592
+ tf_audit_options_free(&st->audit_opts);
886
1593
  free(st);
887
1594
  }
888
1595
  free(self);
@@ -894,8 +1601,23 @@ tf_decoder *tf_csv_decoder_create(const cJSON *args) {
894
1601
 
895
1602
  st->delimiter = ',';
896
1603
  st->has_header = 1;
1604
+ st->skip_repeated_header = 0;
1605
+ st->after_input_boundary = 0;
897
1606
  st->batch_size = DEFAULT_BATCH_SIZE;
898
- st->repair = 0;
1607
+ st->mode = CSV_MODE_PERMISSIVE;
1608
+ st->quoted_nulls = 1;
1609
+ st->trim_ws = 1;
1610
+ st->skip_empty_rows = 0;
1611
+ st->skip_rows = 0;
1612
+ st->skipped_rows = 0;
1613
+ st->limit_rows = 0;
1614
+ st->n_max = 0;
1615
+ st->data_rows_read = 0;
1616
+ st->max_error_bytes = DEFAULT_MAX_ERROR_BYTES;
1617
+ st->max_record_bytes = DEFAULT_MAX_RECORD_BYTES;
1618
+ st->max_columns = DEFAULT_MAX_COLUMNS;
1619
+ st->audit_limit = 1000;
1620
+ tf_audit_options_init(&st->audit_opts, 1);
899
1621
 
900
1622
  if (args) {
901
1623
  cJSON *d = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
@@ -906,23 +1628,188 @@ tf_decoder *tf_csv_decoder_create(const cJSON *args) {
906
1628
  if (cJSON_IsBool(h))
907
1629
  st->has_header = cJSON_IsTrue(h);
908
1630
 
909
- cJSON *bs = cJSON_GetObjectItemCaseSensitive(args, "batch_size");
910
- if (cJSON_IsNumber(bs) && bs->valueint > 0)
911
- st->batch_size = (size_t)bs->valueint;
1631
+ cJSON *skip_repeated_header = cJSON_GetObjectItemCaseSensitive(args, "skip_repeated_header");
1632
+ if (!skip_repeated_header) skip_repeated_header = cJSON_GetObjectItemCaseSensitive(args, "skipRepeatedHeader");
1633
+ if (cJSON_IsBool(skip_repeated_header))
1634
+ st->skip_repeated_header = cJSON_IsTrue(skip_repeated_header);
1635
+
1636
+ size_t parsed_size = 0;
1637
+ int has_batch_size = tf_json_get_size_arg(args, "batch_size", 1, TF_MAX_BATCH_ROWS, &parsed_size, "csv");
1638
+ if (has_batch_size < 0) { free(st); return NULL; }
1639
+ if (has_batch_size > 0) st->batch_size = parsed_size;
912
1640
 
913
1641
  cJSON *rep = cJSON_GetObjectItemCaseSensitive(args, "repair");
914
- if (cJSON_IsBool(rep))
915
- st->repair = cJSON_IsTrue(rep);
1642
+ if (cJSON_IsBool(rep) && cJSON_IsTrue(rep))
1643
+ st->mode = CSV_MODE_REPAIR;
1644
+
1645
+ cJSON *mode = cJSON_GetObjectItemCaseSensitive(args, "mode");
1646
+ if (cJSON_IsString(mode)) {
1647
+ if (!parse_csv_mode(mode->valuestring, &st->mode)) {
1648
+ tf_set_last_error("csv: mode must be permissive, repair, or strict");
1649
+ free(st);
1650
+ return NULL;
1651
+ }
1652
+ }
1653
+
1654
+ cJSON *strict = cJSON_GetObjectItemCaseSensitive(args, "strict");
1655
+ if (cJSON_IsBool(strict) && cJSON_IsTrue(strict))
1656
+ st->mode = CSV_MODE_STRICT;
1657
+
1658
+ int has_max_error = tf_json_get_size_arg(args, "max_error_bytes", 0, TF_MAX_ERROR_BYTES, &parsed_size, "csv");
1659
+ if (has_max_error < 0) { free(st); return NULL; }
1660
+ if (has_max_error > 0) st->max_error_bytes = parsed_size;
1661
+
1662
+ int has_max_record = tf_json_get_size_arg(args, "max_record_bytes", 0, TF_MAX_RECORD_BYTES, &parsed_size, "csv");
1663
+ if (has_max_record < 0) { free(st); return NULL; }
1664
+ if (has_max_record > 0) st->max_record_bytes = parsed_size;
1665
+
1666
+ int has_max_columns = tf_json_get_size_arg(args, "max_columns", 1, TF_MAX_COLUMNS, &parsed_size, "csv");
1667
+ if (has_max_columns < 0) { free(st); return NULL; }
1668
+ if (has_max_columns == 0) {
1669
+ has_max_columns = tf_json_get_size_arg(args, "maxColumns", 1, TF_MAX_COLUMNS, &parsed_size, "csv");
1670
+ if (has_max_columns < 0) { free(st); return NULL; }
1671
+ }
1672
+ if (has_max_columns > 0) st->max_columns = parsed_size;
1673
+
1674
+ cJSON *audit = cJSON_GetObjectItemCaseSensitive(args, "audit");
1675
+ st->audit = cJSON_IsTrue(audit) ? 1 : 0;
1676
+ cJSON *audit_limit = cJSON_GetObjectItemCaseSensitive(args, "audit_limit");
1677
+ if (!audit_limit) audit_limit = cJSON_GetObjectItemCaseSensitive(args, "auditLimit");
1678
+ if (audit_limit) {
1679
+ size_t parsed_limit = 0;
1680
+ if (tf_json_get_size_arg_any(args, "audit_limit", "auditLimit",
1681
+ 1, TF_MAX_AUDIT_RECORDS,
1682
+ &parsed_limit, "csv") < 0) {
1683
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1684
+ free(st->null_literals);
1685
+ free(st->comment);
1686
+ free(st);
1687
+ return NULL;
1688
+ }
1689
+ st->audit_limit = parsed_limit;
1690
+ st->audit = 1;
1691
+ }
1692
+
1693
+ cJSON *quoted_nulls = cJSON_GetObjectItemCaseSensitive(args, "quoted_nulls");
1694
+ if (cJSON_IsBool(quoted_nulls))
1695
+ st->quoted_nulls = cJSON_IsTrue(quoted_nulls);
1696
+
1697
+ cJSON *trim_ws = cJSON_GetObjectItemCaseSensitive(args, "trim_ws");
1698
+ if (!trim_ws) trim_ws = cJSON_GetObjectItemCaseSensitive(args, "trimWs");
1699
+ if (cJSON_IsBool(trim_ws))
1700
+ st->trim_ws = cJSON_IsTrue(trim_ws);
1701
+
1702
+ cJSON *skip_empty = cJSON_GetObjectItemCaseSensitive(args, "skip_empty_rows");
1703
+ if (!skip_empty) skip_empty = cJSON_GetObjectItemCaseSensitive(args, "skipEmptyRows");
1704
+ if (cJSON_IsBool(skip_empty))
1705
+ st->skip_empty_rows = cJSON_IsTrue(skip_empty);
1706
+
1707
+ cJSON *skip_rows = cJSON_GetObjectItemCaseSensitive(args, "skip");
1708
+ if (!skip_rows) skip_rows = cJSON_GetObjectItemCaseSensitive(args, "skip_rows");
1709
+ if (!skip_rows) skip_rows = cJSON_GetObjectItemCaseSensitive(args, "skipRows");
1710
+ if (skip_rows) {
1711
+ const char *skip_name = cJSON_GetObjectItemCaseSensitive(args, "skip") ? "skip" :
1712
+ (cJSON_GetObjectItemCaseSensitive(args, "skip_rows") ? "skip_rows" : "skipRows");
1713
+ size_t parsed_skip = 0;
1714
+ if (tf_json_size_value(skip_rows, skip_name, 0, TF_MAX_COUNT_ARG,
1715
+ &parsed_skip, "csv") < 0) {
1716
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1717
+ free(st->null_literals);
1718
+ free(st->comment);
1719
+ free(st);
1720
+ return NULL;
1721
+ }
1722
+ st->skip_rows = parsed_skip;
1723
+ }
1724
+
1725
+ cJSON *n_max = cJSON_GetObjectItemCaseSensitive(args, "n_max");
1726
+ if (!n_max) n_max = cJSON_GetObjectItemCaseSensitive(args, "nMax");
1727
+ if (!n_max) n_max = cJSON_GetObjectItemCaseSensitive(args, "max_rows");
1728
+ if (!n_max) n_max = cJSON_GetObjectItemCaseSensitive(args, "maxRows");
1729
+ if (n_max) {
1730
+ const char *n_max_name = cJSON_GetObjectItemCaseSensitive(args, "n_max") ? "n_max" :
1731
+ (cJSON_GetObjectItemCaseSensitive(args, "nMax") ? "nMax" :
1732
+ (cJSON_GetObjectItemCaseSensitive(args, "max_rows") ? "max_rows" : "maxRows"));
1733
+ size_t parsed_n_max = 0;
1734
+ if (tf_json_size_value(n_max, n_max_name, 0, TF_MAX_COUNT_ARG,
1735
+ &parsed_n_max, "csv") < 0) {
1736
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1737
+ free(st->null_literals);
1738
+ free(st->comment);
1739
+ free(st);
1740
+ return NULL;
1741
+ }
1742
+ st->limit_rows = 1;
1743
+ st->n_max = parsed_n_max;
1744
+ }
1745
+
1746
+ cJSON *comment = cJSON_GetObjectItemCaseSensitive(args, "comment");
1747
+ if (cJSON_IsString(comment) && comment->valuestring && comment->valuestring[0]) {
1748
+ st->comment = strdup(comment->valuestring);
1749
+ if (!st->comment) {
1750
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1751
+ free(st->null_literals);
1752
+ free(st->comment);
1753
+ free(st);
1754
+ return NULL;
1755
+ }
1756
+ st->comment_len = strlen(st->comment);
1757
+ }
1758
+
1759
+ if (csv_configure_null_literals(st, args) != TF_OK) {
1760
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1761
+ free(st->null_literals);
1762
+ free(st->comment);
1763
+ tf_audit_options_free(&st->audit_opts);
1764
+ free(st);
1765
+ return NULL;
1766
+ }
1767
+ if (tf_audit_options_parse(&st->audit_opts, args, "csv") != TF_OK) {
1768
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1769
+ free(st->null_literals);
1770
+ free(st->comment);
1771
+ tf_audit_options_free(&st->audit_opts);
1772
+ free(st);
1773
+ return NULL;
1774
+ }
916
1775
  }
917
1776
 
918
1777
  tf_buffer_init(&st->line_buf);
919
1778
 
1779
+ st->fields_cap = st->max_columns;
1780
+ st->fields = tf_callocarray_checked(st->fields_cap, sizeof(field_slice));
1781
+ if (!st->fields) {
1782
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1783
+ free(st->null_literals);
1784
+ free(st->comment);
1785
+ tf_audit_options_free(&st->audit_opts);
1786
+ free(st);
1787
+ return NULL;
1788
+ }
1789
+
920
1790
  /* Arena for escaped quoted field data (reset per line, rarely used) */
921
1791
  st->field_arena = tf_arena_create(4096);
922
- if (!st->field_arena) { free(st); return NULL; }
1792
+ if (!st->field_arena) {
1793
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1794
+ free(st->null_literals);
1795
+ free(st->comment);
1796
+ free(st->fields);
1797
+ tf_audit_options_free(&st->audit_opts);
1798
+ free(st);
1799
+ return NULL;
1800
+ }
923
1801
 
924
1802
  tf_decoder *dec = malloc(sizeof(tf_decoder));
925
- if (!dec) { tf_arena_free(st->field_arena); free(st); return NULL; }
1803
+ if (!dec) {
1804
+ tf_arena_free(st->field_arena);
1805
+ for (size_t i = 0; i < st->n_null_literals; i++) free(st->null_literals[i]);
1806
+ free(st->null_literals);
1807
+ free(st->comment);
1808
+ free(st->fields);
1809
+ tf_audit_options_free(&st->audit_opts);
1810
+ free(st);
1811
+ return NULL;
1812
+ }
926
1813
  dec->decode = csv_decode;
927
1814
  dec->flush = csv_flush;
928
1815
  dec->destroy = csv_decoder_destroy;
@@ -931,7 +1818,7 @@ tf_decoder *tf_csv_decoder_create(const cJSON *args) {
931
1818
  }
932
1819
 
933
1820
  /* ================================================================
934
- * CSV Encoder (unchanged — already efficient)
1821
+ * CSV Encoder
935
1822
  * ================================================================ */
936
1823
 
937
1824
  typedef struct {
@@ -948,7 +1835,7 @@ static int needs_quoting(const char *s, char delim) {
948
1835
  return 0;
949
1836
  }
950
1837
 
951
- static int write_field(tf_buffer *out, const char *s, char delim) {
1838
+ static int csv_write_field(tf_buffer *out, const char *s, char delim) {
952
1839
  if (needs_quoting(s, delim)) {
953
1840
  if (tf_buffer_write(out, (const uint8_t *)"\"", 1) != TF_OK) return TF_ERROR;
954
1841
  for (const char *p = s; *p; p++) {
@@ -965,17 +1852,20 @@ static int write_field(tf_buffer *out, const char *s, char delim) {
965
1852
  return TF_OK;
966
1853
  }
967
1854
 
1855
+ static int csv_write_delimiter(tf_buffer *out, char delim) {
1856
+ return tf_buffer_write(out, (const uint8_t *)&delim, 1);
1857
+ }
1858
+
968
1859
  static int csv_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
969
1860
  csv_encoder_state *st = self->state;
970
- char dbuf[2] = { st->delimiter, '\0' };
971
1861
 
972
1862
  /* Write header */
973
1863
  if (!st->header_written) {
974
1864
  for (size_t i = 0; i < in->n_cols; i++) {
975
- if (i > 0) tf_buffer_write_str(out, dbuf);
976
- write_field(out, in->col_names[i], st->delimiter);
1865
+ if (i > 0 && csv_write_delimiter(out, st->delimiter) != TF_OK) return TF_ERROR;
1866
+ if (csv_write_field(out, in->col_names[i], st->delimiter) != TF_OK) return TF_ERROR;
977
1867
  }
978
- tf_buffer_write(out, (const uint8_t *)"\n", 1);
1868
+ if (tf_buffer_write(out, (const uint8_t *)"\n", 1) != TF_OK) return TF_ERROR;
979
1869
  st->header_written = 1;
980
1870
  }
981
1871
 
@@ -983,43 +1873,46 @@ static int csv_encode(tf_encoder *self, tf_batch *in, tf_buffer *out) {
983
1873
  char numbuf[64];
984
1874
  for (size_t r = 0; r < in->n_rows; r++) {
985
1875
  for (size_t c = 0; c < in->n_cols; c++) {
986
- if (c > 0) tf_buffer_write_str(out, dbuf);
1876
+ if (c > 0 && csv_write_delimiter(out, st->delimiter) != TF_OK) return TF_ERROR;
987
1877
  if (tf_batch_is_null(in, r, c)) {
988
1878
  /* empty field for null */
989
1879
  continue;
990
1880
  }
991
1881
  switch (in->col_types[c]) {
992
1882
  case TF_TYPE_BOOL:
993
- tf_buffer_write_str(out, tf_batch_get_bool(in, r, c) ? "true" : "false");
1883
+ if (tf_buffer_write_str(out, tf_batch_get_bool(in, r, c) ? "true" : "false") != TF_OK)
1884
+ return TF_ERROR;
994
1885
  break;
995
1886
  case TF_TYPE_INT64:
996
1887
  snprintf(numbuf, sizeof(numbuf), "%lld", (long long)tf_batch_get_int64(in, r, c));
997
- tf_buffer_write_str(out, numbuf);
1888
+ if (tf_buffer_write_str(out, numbuf) != TF_OK) return TF_ERROR;
998
1889
  break;
999
1890
  case TF_TYPE_FLOAT64:
1000
- snprintf(numbuf, sizeof(numbuf), "%g", tf_batch_get_float64(in, r, c));
1001
- tf_buffer_write_str(out, numbuf);
1891
+ if (tf_format_float64(numbuf, sizeof(numbuf), tf_batch_get_float64(in, r, c)) != TF_OK)
1892
+ return TF_ERROR;
1893
+ if (tf_buffer_write_str(out, numbuf) != TF_OK) return TF_ERROR;
1002
1894
  break;
1003
1895
  case TF_TYPE_STRING:
1004
- write_field(out, tf_batch_get_string(in, r, c), st->delimiter);
1896
+ if (csv_write_field(out, tf_batch_get_string(in, r, c), st->delimiter) != TF_OK)
1897
+ return TF_ERROR;
1005
1898
  break;
1006
1899
  case TF_TYPE_DATE: {
1007
- char dbuf2[16];
1008
- tf_date_format(tf_batch_get_date(in, r, c), dbuf2, sizeof(dbuf2));
1009
- tf_buffer_write_str(out, dbuf2);
1900
+ char dbuf[32];
1901
+ tf_date_format(tf_batch_get_date(in, r, c), dbuf, sizeof(dbuf));
1902
+ if (tf_buffer_write_str(out, dbuf) != TF_OK) return TF_ERROR;
1010
1903
  break;
1011
1904
  }
1012
1905
  case TF_TYPE_TIMESTAMP: {
1013
1906
  char tsbuf[40];
1014
1907
  tf_timestamp_format(tf_batch_get_timestamp(in, r, c), tsbuf, sizeof(tsbuf));
1015
- tf_buffer_write_str(out, tsbuf);
1908
+ if (tf_buffer_write_str(out, tsbuf) != TF_OK) return TF_ERROR;
1016
1909
  break;
1017
1910
  }
1018
1911
  default:
1019
1912
  break;
1020
1913
  }
1021
1914
  }
1022
- tf_buffer_write(out, (const uint8_t *)"\n", 1);
1915
+ if (tf_buffer_write(out, (const uint8_t *)"\n", 1) != TF_OK) return TF_ERROR;
1023
1916
  }
1024
1917
 
1025
1918
  return TF_OK;