tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
package/csrc/op_hash.c CHANGED
@@ -25,9 +25,12 @@ static int hash_process(tf_step *self, tf_batch *in, tf_batch **out,
25
25
 
26
26
  tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
27
27
  if (!ob) return TF_ERROR;
28
- for (size_t c = 0; c < in->n_cols; c++)
29
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
30
- tf_batch_set_schema(ob, in->n_cols, "_hash", TF_TYPE_INT64);
28
+ const char *extra_names[] = {"_hash"};
29
+ const tf_type extra_types[] = {TF_TYPE_INT64};
30
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
31
+ tf_batch_free(ob);
32
+ return TF_ERROR;
33
+ }
31
34
 
32
35
  /* Resolve column indices */
33
36
  size_t n_keys;
@@ -46,7 +49,11 @@ static int hash_process(tf_step *self, tf_batch *in, tf_batch **out,
46
49
  }
47
50
 
48
51
  for (size_t r = 0; r < in->n_rows; r++) {
49
- tf_batch_copy_row(ob, r, in, r);
52
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
53
+ free(col_indices);
54
+ tf_batch_free(ob);
55
+ return TF_ERROR;
56
+ }
50
57
  /* Compute hash */
51
58
  uint32_t h = 5381;
52
59
  char val_buf[64];
@@ -75,8 +82,16 @@ static int hash_process(tf_step *self, tf_batch *in, tf_batch **out,
75
82
  for (const unsigned char *p = (const unsigned char *)val; *p; p++)
76
83
  h = ((h << 5) + h) ^ *p;
77
84
  }
78
- tf_batch_set_int64(ob, r, in->n_cols, (int64_t)h);
79
- ob->n_rows = r + 1;
85
+ if (tf_batch_set_int64(ob, r, in->n_cols, (int64_t)h) != TF_OK) {
86
+ free(col_indices);
87
+ tf_batch_free(ob);
88
+ return TF_ERROR;
89
+ }
90
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
91
+ free(col_indices);
92
+ tf_batch_free(ob);
93
+ return TF_ERROR;
94
+ }
80
95
  }
81
96
 
82
97
  free(col_indices);
@@ -88,12 +103,18 @@ static int hash_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
88
103
  (void)self; (void)side; *out = NULL; return TF_OK;
89
104
  }
90
105
 
91
- static void hash_destroy(tf_step *self) {
92
- hash_state *st = self->state;
106
+ static void hash_state_free(hash_state *st) {
93
107
  if (st) {
94
- for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i]);
95
- free(st->cols); free(st);
108
+ if (st->cols) {
109
+ for (size_t i = 0; i < st->n_cols; i++) free(st->cols[i]);
110
+ }
111
+ free(st->cols);
112
+ free(st);
96
113
  }
114
+ }
115
+
116
+ static void hash_destroy(tf_step *self) {
117
+ hash_state_free(self ? self->state : NULL);
97
118
  free(self);
98
119
  }
99
120
 
@@ -103,21 +124,33 @@ tf_step *tf_hash_create(const cJSON *args) {
103
124
 
104
125
  if (args) {
105
126
  cJSON *columns = cJSON_GetObjectItemCaseSensitive(args, "columns");
106
- if (columns && cJSON_IsArray(columns)) {
127
+ if (columns) {
128
+ if (!cJSON_IsArray(columns)) {
129
+ tf_set_last_error("hash: columns must be an array");
130
+ hash_state_free(st);
131
+ return NULL;
132
+ }
107
133
  int n = cJSON_GetArraySize(columns);
108
134
  if (n > 0) {
109
135
  st->cols = calloc(n, sizeof(char *));
110
- st->n_cols = n;
136
+ if (!st->cols) { hash_state_free(st); return NULL; }
137
+ st->n_cols = (size_t)n;
111
138
  for (int i = 0; i < n; i++) {
112
139
  cJSON *item = cJSON_GetArrayItem(columns, i);
113
- if (cJSON_IsString(item)) st->cols[i] = strdup(item->valuestring);
140
+ if (!cJSON_IsString(item) || !item->valuestring || !item->valuestring[0]) {
141
+ tf_set_last_error("hash: column names must be non-empty strings");
142
+ hash_state_free(st);
143
+ return NULL;
144
+ }
145
+ st->cols[i] = strdup(item->valuestring);
146
+ if (!st->cols[i]) { hash_state_free(st); return NULL; }
114
147
  }
115
148
  }
116
149
  }
117
150
  }
118
151
 
119
- tf_step *step = malloc(sizeof(tf_step));
120
- if (!step) { hash_destroy(&(tf_step){.state = st}); return NULL; }
152
+ tf_step *step = calloc(1, sizeof(tf_step));
153
+ if (!step) { hash_state_free(st); return NULL; }
121
154
  step->process = hash_process;
122
155
  step->flush = hash_flush;
123
156
  step->destroy = hash_destroy;
package/csrc/op_head.c CHANGED
@@ -28,93 +28,27 @@ static int head_process(tf_step *self, tf_batch *in, tf_batch **out,
28
28
 
29
29
  size_t remaining = st->limit - st->seen;
30
30
  size_t take = in->n_rows < remaining ? in->n_rows : remaining;
31
+ if (take == 0) return TF_OK;
31
32
 
32
- if (take == in->n_rows) {
33
- /* Take all rows — create a copy */
34
- tf_batch *ob = tf_batch_create(in->n_cols, take);
35
- if (!ob) return TF_ERROR;
36
- for (size_t c = 0; c < in->n_cols; c++) {
37
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
38
- }
39
- for (size_t r = 0; r < take; r++) {
40
- tf_batch_ensure_capacity(ob, r + 1);
41
- for (size_t c = 0; c < in->n_cols; c++) {
42
- if (tf_batch_is_null(in, r, c)) {
43
- tf_batch_set_null(ob, r, c);
44
- continue;
45
- }
46
- switch (in->col_types[c]) {
47
- case TF_TYPE_BOOL:
48
- tf_batch_set_bool(ob, r, c, tf_batch_get_bool(in, r, c));
49
- break;
50
- case TF_TYPE_INT64:
51
- tf_batch_set_int64(ob, r, c, tf_batch_get_int64(in, r, c));
52
- break;
53
- case TF_TYPE_FLOAT64:
54
- tf_batch_set_float64(ob, r, c, tf_batch_get_float64(in, r, c));
55
- break;
56
- case TF_TYPE_STRING:
57
- tf_batch_set_string(ob, r, c, tf_batch_get_string(in, r, c));
58
- break;
59
- case TF_TYPE_DATE:
60
- tf_batch_set_date(ob, r, c, tf_batch_get_date(in, r, c));
61
- break;
62
- case TF_TYPE_TIMESTAMP:
63
- tf_batch_set_timestamp(ob, r, c, tf_batch_get_timestamp(in, r, c));
64
- break;
65
- default:
66
- tf_batch_set_null(ob, r, c);
67
- break;
68
- }
69
- }
70
- ob->n_rows = r + 1;
71
- }
72
- st->seen += take;
73
- *out = ob;
74
- } else {
75
- /* Partial take — only copy first `take` rows */
76
- tf_batch *ob = tf_batch_create(in->n_cols, take);
77
- if (!ob) return TF_ERROR;
78
- for (size_t c = 0; c < in->n_cols; c++) {
79
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
33
+ tf_batch *ob = tf_batch_create(in->n_cols, take);
34
+ if (!ob) return TF_ERROR;
35
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
36
+ tf_batch_free(ob);
37
+ return TF_ERROR;
38
+ }
39
+ for (size_t r = 0; r < take; r++) {
40
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
41
+ tf_batch_free(ob);
42
+ return TF_ERROR;
80
43
  }
81
- for (size_t r = 0; r < take; r++) {
82
- tf_batch_ensure_capacity(ob, r + 1);
83
- for (size_t c = 0; c < in->n_cols; c++) {
84
- if (tf_batch_is_null(in, r, c)) {
85
- tf_batch_set_null(ob, r, c);
86
- continue;
87
- }
88
- switch (in->col_types[c]) {
89
- case TF_TYPE_BOOL:
90
- tf_batch_set_bool(ob, r, c, tf_batch_get_bool(in, r, c));
91
- break;
92
- case TF_TYPE_INT64:
93
- tf_batch_set_int64(ob, r, c, tf_batch_get_int64(in, r, c));
94
- break;
95
- case TF_TYPE_FLOAT64:
96
- tf_batch_set_float64(ob, r, c, tf_batch_get_float64(in, r, c));
97
- break;
98
- case TF_TYPE_STRING:
99
- tf_batch_set_string(ob, r, c, tf_batch_get_string(in, r, c));
100
- break;
101
- case TF_TYPE_DATE:
102
- tf_batch_set_date(ob, r, c, tf_batch_get_date(in, r, c));
103
- break;
104
- case TF_TYPE_TIMESTAMP:
105
- tf_batch_set_timestamp(ob, r, c, tf_batch_get_timestamp(in, r, c));
106
- break;
107
- default:
108
- tf_batch_set_null(ob, r, c);
109
- break;
110
- }
111
- }
112
- ob->n_rows = r + 1;
44
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
45
+ tf_batch_free(ob);
46
+ return TF_ERROR;
113
47
  }
114
- st->seen += take;
115
- *out = ob;
116
48
  }
117
49
 
50
+ st->seen += take;
51
+ *out = ob;
118
52
  return TF_OK;
119
53
  }
120
54
 
@@ -131,15 +65,16 @@ static void head_destroy(tf_step *self) {
131
65
 
132
66
  tf_step *tf_head_create(const cJSON *args) {
133
67
  if (!args) return NULL;
134
- cJSON *n_json = cJSON_GetObjectItemCaseSensitive(args, "n");
135
- if (!cJSON_IsNumber(n_json) || n_json->valueint <= 0) return NULL;
68
+ size_t n = 0;
69
+ int has_n = tf_json_get_size_arg(args, "n", 0, TF_MAX_COUNT_ARG, &n, "head");
70
+ if (has_n <= 0) return NULL;
136
71
 
137
72
  head_state *st = calloc(1, sizeof(head_state));
138
73
  if (!st) return NULL;
139
- st->limit = (size_t)n_json->valueint;
74
+ st->limit = n;
140
75
  st->seen = 0;
141
76
 
142
- tf_step *step = malloc(sizeof(tf_step));
77
+ tf_step *step = calloc(1, sizeof(tf_step));
143
78
  if (!step) { free(st); return NULL; }
144
79
  step->process = head_process;
145
80
  step->flush = head_flush;
@@ -20,6 +20,17 @@ typedef enum {
20
20
  INTERP_LINEAR
21
21
  } interp_method;
22
22
 
23
+ typedef enum {
24
+ INTERP_MISSING_ERROR,
25
+ INTERP_MISSING_NULL,
26
+ INTERP_MISSING_IGNORE,
27
+ } interp_missing_policy;
28
+
29
+ typedef enum {
30
+ INTERP_TYPE_FAIL,
31
+ INTERP_TYPE_NULL,
32
+ } interp_type_policy;
33
+
23
34
  /* Buffered row: we store complete batches and track which rows are pending */
24
35
  typedef struct pending_row {
25
36
  tf_batch *batch; /* single-row batch (copy of original row) */
@@ -31,47 +42,180 @@ typedef struct {
31
42
  interp_method method;
32
43
  double last_val;
33
44
  int has_last;
45
+ interp_missing_policy missing;
46
+ interp_type_policy on_type_error;
34
47
  /* Pending null rows (for backward/linear) */
35
48
  pending_row *pending;
36
49
  size_t n_pending;
37
50
  size_t cap_pending;
38
51
  } interpolate_state;
39
52
 
40
- static interp_method parse_method(const char *s) {
41
- if (!s) return INTERP_LINEAR;
42
- if (strcmp(s, "forward") == 0) return INTERP_FORWARD;
43
- if (strcmp(s, "backward") == 0) return INTERP_BACKWARD;
44
- return INTERP_LINEAR;
53
+ static int parse_method(const char *s, interp_method *out) {
54
+ if (!s) { *out = INTERP_LINEAR; return TF_OK; }
55
+ if (strcmp(s, "forward") == 0) { *out = INTERP_FORWARD; return TF_OK; }
56
+ if (strcmp(s, "backward") == 0) { *out = INTERP_BACKWARD; return TF_OK; }
57
+ if (strcmp(s, "linear") == 0) { *out = INTERP_LINEAR; return TF_OK; }
58
+ return TF_ERROR;
59
+ }
60
+
61
+ static int interpolate_is_numeric_type(tf_type type) {
62
+ return type == TF_TYPE_INT64 || type == TF_TYPE_FLOAT64;
45
63
  }
46
64
 
47
- static double get_numeric(const tf_batch *b, size_t r, int ci) {
65
+ static double interpolate_get_numeric(const tf_batch *b, size_t r, int ci) {
48
66
  if (b->col_types[ci] == TF_TYPE_INT64) return (double)tf_batch_get_int64(b, r, ci);
49
- if (b->col_types[ci] == TF_TYPE_FLOAT64) return tf_batch_get_float64(b, r, ci);
50
- return 0;
67
+ return tf_batch_get_float64(b, r, ci);
68
+ }
69
+
70
+ static int interpolate_set_numeric(tf_batch *b, size_t r, size_t ci, double val) {
71
+ if (b->col_types[ci] == TF_TYPE_INT64)
72
+ return tf_batch_set_int64(b, r, ci, (int64_t)val);
73
+ if (b->col_types[ci] == TF_TYPE_FLOAT64)
74
+ return tf_batch_set_float64(b, r, ci, val);
75
+ return TF_ERROR;
76
+ }
77
+
78
+ static void interpolate_set_col_error(const char *column, const char *suffix) {
79
+ char msg[512];
80
+ snprintf(msg, sizeof(msg), "interpolate: column '%s' %s", column ? column : "", suffix);
81
+ tf_set_last_error(msg);
82
+ }
83
+
84
+ static int interpolate_parse_missing_policy(const cJSON *args, interp_missing_policy *out) {
85
+ *out = INTERP_MISSING_ERROR;
86
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "missing");
87
+ if (!j) return TF_OK;
88
+ if (!cJSON_IsString(j)) {
89
+ tf_set_last_error("interpolate: missing must be error, null, or ignore");
90
+ return TF_ERROR;
91
+ }
92
+ if (strcmp(j->valuestring, "error") == 0) *out = INTERP_MISSING_ERROR;
93
+ else if (strcmp(j->valuestring, "null") == 0) *out = INTERP_MISSING_NULL;
94
+ else if (strcmp(j->valuestring, "ignore") == 0) *out = INTERP_MISSING_IGNORE;
95
+ else {
96
+ tf_set_last_error("interpolate: missing must be error, null, or ignore");
97
+ return TF_ERROR;
98
+ }
99
+ return TF_OK;
100
+ }
101
+
102
+ static int interpolate_parse_type_policy(const cJSON *args, interp_type_policy *out) {
103
+ *out = INTERP_TYPE_FAIL;
104
+ const cJSON *j = cJSON_GetObjectItemCaseSensitive(args, "on_type_error");
105
+ if (!j) return TF_OK;
106
+ if (!cJSON_IsString(j)) {
107
+ tf_set_last_error("interpolate: on_type_error must be fail or null");
108
+ return TF_ERROR;
109
+ }
110
+ if (strcmp(j->valuestring, "fail") == 0) *out = INTERP_TYPE_FAIL;
111
+ else if (strcmp(j->valuestring, "null") == 0) *out = INTERP_TYPE_NULL;
112
+ else {
113
+ tf_set_last_error("interpolate: on_type_error must be fail or null");
114
+ return TF_ERROR;
115
+ }
116
+ return TF_OK;
117
+ }
118
+
119
+ static int interpolate_passthrough(tf_batch *in, tf_batch **out) {
120
+ tf_batch *ob = tf_batch_create(in->n_cols, in->n_rows);
121
+ if (!ob) return TF_ERROR;
122
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
123
+ tf_batch_free(ob);
124
+ return TF_ERROR;
125
+ }
126
+ for (size_t r = 0; r < in->n_rows; r++) {
127
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
128
+ tf_batch_free(ob);
129
+ return TF_ERROR;
130
+ }
131
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
132
+ tf_batch_free(ob);
133
+ return TF_ERROR;
134
+ }
135
+ }
136
+ *out = ob;
137
+ return TF_OK;
138
+ }
139
+
140
+ static int interpolate_null_output(interpolate_state *st, tf_batch *in, int ci, tf_batch **out) {
141
+ size_t result_col = ci < 0 ? in->n_cols : (size_t)ci;
142
+ tf_batch *ob = NULL;
143
+ if (ci < 0) {
144
+ const char *extra_names[1] = {st->column};
145
+ tf_type extra_types[1] = {TF_TYPE_FLOAT64};
146
+ ob = tf_batch_create(in->n_cols + 1, in->n_rows);
147
+ if (!ob) return TF_ERROR;
148
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
149
+ tf_batch_free(ob);
150
+ return TF_ERROR;
151
+ }
152
+ } else {
153
+ ob = tf_batch_create(in->n_cols, in->n_rows);
154
+ if (!ob) return TF_ERROR;
155
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
156
+ tf_batch_free(ob);
157
+ return TF_ERROR;
158
+ }
159
+ }
160
+
161
+ for (size_t r = 0; r < in->n_rows; r++) {
162
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK ||
163
+ tf_batch_set_null(ob, r, result_col) != TF_OK) {
164
+ tf_batch_free(ob);
165
+ return TF_ERROR;
166
+ }
167
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
168
+ tf_batch_free(ob);
169
+ return TF_ERROR;
170
+ }
171
+ }
172
+ *out = ob;
173
+ return TF_OK;
51
174
  }
52
175
 
53
- static void add_pending(interpolate_state *st, tf_batch *row_batch, size_t target_col) {
54
- if (st->n_pending >= st->cap_pending) {
55
- size_t newcap = st->cap_pending ? st->cap_pending * 2 : 16;
56
- pending_row *tmp = realloc(st->pending, newcap * sizeof(pending_row));
57
- if (!tmp) return;
176
+ static int add_pending(interpolate_state *st, tf_batch *row_batch, size_t target_col) {
177
+ size_t next_pending = 0;
178
+ if (tf_size_add(st->n_pending, 1, &next_pending) != TF_OK) return TF_ERROR;
179
+ if (next_pending > st->cap_pending) {
180
+ size_t newcap = 0;
181
+ if (tf_size_grow_pow2(st->cap_pending, next_pending, 16, &newcap) != TF_OK) {
182
+ return TF_ERROR;
183
+ }
184
+ pending_row *tmp = tf_reallocarray_checked(st->pending, newcap, sizeof(pending_row));
185
+ if (!tmp) return TF_ERROR;
58
186
  st->pending = tmp;
59
187
  st->cap_pending = newcap;
60
188
  }
61
189
  st->pending[st->n_pending].batch = row_batch;
62
190
  st->pending[st->n_pending].target_col = target_col;
63
- st->n_pending++;
191
+ st->n_pending = next_pending;
192
+ return TF_OK;
193
+ }
194
+
195
+ static tf_batch *copy_single_row(const tf_batch *in, size_t row) {
196
+ tf_batch *row_copy = tf_batch_create(in->n_cols, 1);
197
+ if (!row_copy) return NULL;
198
+ if (tf_batch_clone_schema(row_copy, in) != TF_OK ||
199
+ tf_batch_copy_row(row_copy, 0, in, row) != TF_OK) {
200
+ tf_batch_free(row_copy);
201
+ return NULL;
202
+ }
203
+ if (tf_batch_expose_row(row_copy, 0) != TF_OK) {
204
+ tf_batch_free(row_copy);
205
+ return NULL;
206
+ }
207
+ return row_copy;
64
208
  }
65
209
 
66
210
  /* Emit all pending rows into output batch, interpolating values */
67
- static void flush_pending(interpolate_state *st, tf_batch *ob, size_t *out_row,
68
- double end_val, size_t target_col) {
69
- if (st->n_pending == 0) return;
211
+ static int flush_pending(interpolate_state *st, tf_batch *ob, size_t *out_row,
212
+ double end_val, size_t target_col) {
213
+ if (st->n_pending == 0) return TF_OK;
70
214
 
71
215
  for (size_t i = 0; i < st->n_pending; i++) {
72
216
  tf_batch *pb = st->pending[i].batch;
73
217
  size_t r = *out_row;
74
- tf_batch_copy_row(ob, r, pb, 0);
218
+ if (tf_batch_copy_row(ob, r, pb, 0) != TF_OK) return TF_ERROR;
75
219
 
76
220
  double interp_val;
77
221
  if (st->method == INTERP_BACKWARD) {
@@ -86,13 +230,17 @@ static void flush_pending(interpolate_state *st, tf_batch *ob, size_t *out_row,
86
230
  }
87
231
  }
88
232
 
89
- tf_batch_set_float64(ob, r, target_col, interp_val);
90
- ob->n_rows = r + 1;
233
+ if (interpolate_set_numeric(ob, r, target_col, interp_val) != TF_OK) return TF_ERROR;
234
+ if (tf_batch_expose_row(ob, r) != TF_OK) return TF_ERROR;
91
235
  (*out_row)++;
236
+ }
92
237
 
93
- tf_batch_free(pb);
238
+ for (size_t i = 0; i < st->n_pending; i++) {
239
+ tf_batch_free(st->pending[i].batch);
240
+ st->pending[i].batch = NULL;
94
241
  }
95
242
  st->n_pending = 0;
243
+ return TF_OK;
96
244
  }
97
245
 
98
246
  static int interpolate_process(tf_step *self, tf_batch *in, tf_batch **out,
@@ -102,63 +250,94 @@ static int interpolate_process(tf_step *self, tf_batch *in, tf_batch **out,
102
250
  *out = NULL;
103
251
 
104
252
  int ci = tf_batch_col_index(in, st->column);
253
+ int force_null = 0;
254
+ if (ci < 0) {
255
+ if (st->missing == INTERP_MISSING_ERROR) {
256
+ interpolate_set_col_error(st->column, "not found");
257
+ return TF_ERROR;
258
+ }
259
+ if (st->missing == INTERP_MISSING_IGNORE) return interpolate_passthrough(in, out);
260
+ force_null = 1;
261
+ } else if (!interpolate_is_numeric_type(in->col_types[ci])) {
262
+ if (st->on_type_error == INTERP_TYPE_FAIL) {
263
+ interpolate_set_col_error(st->column, "must be numeric");
264
+ return TF_ERROR;
265
+ }
266
+ force_null = 1;
267
+ }
268
+ if (force_null) return interpolate_null_output(st, in, ci, out);
105
269
 
106
- /* Count total rows we might output (pending + current batch) */
107
- size_t max_rows = st->n_pending + in->n_rows;
270
+ size_t max_rows = 0;
271
+ if (tf_size_add(st->n_pending, in->n_rows, &max_rows) != TF_OK) return TF_ERROR;
108
272
  tf_batch *ob = tf_batch_create(in->n_cols, max_rows);
109
273
  if (!ob) return TF_ERROR;
110
- for (size_t c = 0; c < in->n_cols; c++)
111
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
274
+ if (tf_batch_clone_schema(ob, in) != TF_OK) {
275
+ tf_batch_free(ob);
276
+ return TF_ERROR;
277
+ }
112
278
 
113
279
  size_t out_row = 0;
114
280
 
115
281
  for (size_t r = 0; r < in->n_rows; r++) {
116
- if (ci < 0) {
117
- /* No target column — just pass through */
118
- tf_batch_copy_row(ob, out_row, in, r);
119
- ob->n_rows = out_row + 1;
120
- out_row++;
121
- continue;
122
- }
123
-
124
282
  int is_null = tf_batch_is_null(in, r, ci);
125
283
 
126
284
  if (is_null) {
127
285
  if (st->method == INTERP_FORWARD && st->has_last) {
128
286
  /* Forward fill: use last known value */
129
- tf_batch_copy_row(ob, out_row, in, r);
130
- tf_batch_set_float64(ob, out_row, ci, st->last_val);
131
- ob->n_rows = out_row + 1;
287
+ if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK ||
288
+ interpolate_set_numeric(ob, out_row, (size_t)ci, st->last_val) != TF_OK) {
289
+ tf_batch_free(ob);
290
+ return TF_ERROR;
291
+ }
292
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) {
293
+ tf_batch_free(ob);
294
+ return TF_ERROR;
295
+ }
132
296
  out_row++;
133
297
  } else if (st->method == INTERP_FORWARD) {
134
298
  /* No previous value yet — pass null through */
135
- tf_batch_copy_row(ob, out_row, in, r);
136
- ob->n_rows = out_row + 1;
299
+ if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
300
+ tf_batch_free(ob);
301
+ return TF_ERROR;
302
+ }
303
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) {
304
+ tf_batch_free(ob);
305
+ return TF_ERROR;
306
+ }
137
307
  out_row++;
138
308
  } else {
139
309
  /* Backward/linear: buffer this row */
140
- tf_batch *row_copy = tf_batch_create(in->n_cols, 1);
141
- if (row_copy) {
142
- for (size_t c = 0; c < in->n_cols; c++)
143
- tf_batch_set_schema(row_copy, c, in->col_names[c], in->col_types[c]);
144
- tf_batch_copy_row(row_copy, 0, in, r);
145
- row_copy->n_rows = 1;
146
- add_pending(st, row_copy, ci);
310
+ tf_batch *row_copy = copy_single_row(in, r);
311
+ if (!row_copy || add_pending(st, row_copy, (size_t)ci) != TF_OK) {
312
+ tf_batch_free(row_copy);
313
+ tf_batch_free(ob);
314
+ return TF_ERROR;
147
315
  }
148
316
  }
149
317
  } else {
150
- double val = get_numeric(in, r, ci);
318
+ double val = interpolate_get_numeric(in, r, ci);
151
319
 
152
320
  /* Flush any pending rows */
153
321
  if (st->n_pending > 0) {
154
- /* Grow output if needed */
155
- tf_batch_ensure_capacity(ob, out_row + st->n_pending + 1);
156
- flush_pending(st, ob, &out_row, val, ci);
322
+ size_t needed = 0;
323
+ if (tf_size_add(out_row, st->n_pending, &needed) != TF_OK ||
324
+ tf_size_add(needed, 1, &needed) != TF_OK ||
325
+ tf_batch_ensure_capacity(ob, needed) != TF_OK ||
326
+ flush_pending(st, ob, &out_row, val, (size_t)ci) != TF_OK) {
327
+ tf_batch_free(ob);
328
+ return TF_ERROR;
329
+ }
157
330
  }
158
331
 
159
332
  /* Output current row */
160
- tf_batch_copy_row(ob, out_row, in, r);
161
- ob->n_rows = out_row + 1;
333
+ if (tf_batch_copy_row(ob, out_row, in, r) != TF_OK) {
334
+ tf_batch_free(ob);
335
+ return TF_ERROR;
336
+ }
337
+ if (tf_batch_expose_row(ob, out_row) != TF_OK) {
338
+ tf_batch_free(ob);
339
+ return TF_ERROR;
340
+ }
162
341
  out_row++;
163
342
 
164
343
  st->last_val = val;
@@ -184,19 +363,31 @@ static int interpolate_flush(tf_step *self, tf_batch **out, tf_side_channels *si
184
363
  if (st->n_pending > 0) {
185
364
  tf_batch *first = st->pending[0].batch;
186
365
  tf_batch *ob = tf_batch_create(first->n_cols, st->n_pending);
187
- if (!ob) return TF_OK;
188
- for (size_t c = 0; c < first->n_cols; c++)
189
- tf_batch_set_schema(ob, c, first->col_names[c], first->col_types[c]);
366
+ if (!ob) return TF_ERROR;
367
+ if (tf_batch_clone_schema(ob, first) != TF_OK) {
368
+ tf_batch_free(ob);
369
+ return TF_ERROR;
370
+ }
190
371
 
191
372
  for (size_t i = 0; i < st->n_pending; i++) {
192
373
  tf_batch *pb = st->pending[i].batch;
193
- tf_batch_copy_row(ob, i, pb, 0);
374
+ if (tf_batch_copy_row(ob, i, pb, 0) != TF_OK) {
375
+ tf_batch_free(ob);
376
+ return TF_ERROR;
377
+ }
194
378
  /* For linear/backward at end of stream: use last known if available */
195
- if (st->has_last) {
196
- tf_batch_set_float64(ob, i, st->pending[i].target_col, st->last_val);
379
+ if (st->has_last && interpolate_set_numeric(ob, i, st->pending[i].target_col, st->last_val) != TF_OK) {
380
+ tf_batch_free(ob);
381
+ return TF_ERROR;
382
+ }
383
+ if (tf_batch_expose_row(ob, i) != TF_OK) {
384
+ tf_batch_free(ob);
385
+ return TF_ERROR;
197
386
  }
198
- ob->n_rows = i + 1;
199
- tf_batch_free(pb);
387
+ }
388
+ for (size_t i = 0; i < st->n_pending; i++) {
389
+ tf_batch_free(st->pending[i].batch);
390
+ st->pending[i].batch = NULL;
200
391
  }
201
392
  st->n_pending = 0;
202
393
  *out = ob;
@@ -220,16 +411,31 @@ static void interpolate_destroy(tf_step *self) {
220
411
  tf_step *tf_interpolate_create(const cJSON *args) {
221
412
  if (!args) return NULL;
222
413
  cJSON *col_j = cJSON_GetObjectItemCaseSensitive(args, "column");
223
- if (!cJSON_IsString(col_j)) return NULL;
414
+ if (!cJSON_IsString(col_j) || !col_j->valuestring[0]) {
415
+ tf_set_last_error("interpolate: column is required");
416
+ return NULL;
417
+ }
224
418
 
225
419
  interpolate_state *st = calloc(1, sizeof(interpolate_state));
226
420
  if (!st) return NULL;
227
421
  st->column = strdup(col_j->valuestring);
422
+ if (!st->column) { free(st); return NULL; }
228
423
 
229
424
  cJSON *method_j = cJSON_GetObjectItemCaseSensitive(args, "method");
230
- st->method = parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL);
425
+ if (parse_method(cJSON_IsString(method_j) ? method_j->valuestring : NULL, &st->method) != TF_OK) {
426
+ tf_set_last_error("interpolate: method must be forward, backward, or linear");
427
+ free(st->column);
428
+ free(st);
429
+ return NULL;
430
+ }
431
+ if (interpolate_parse_missing_policy(args, &st->missing) != TF_OK ||
432
+ interpolate_parse_type_policy(args, &st->on_type_error) != TF_OK) {
433
+ free(st->column);
434
+ free(st);
435
+ return NULL;
436
+ }
231
437
 
232
- tf_step *step = malloc(sizeof(tf_step));
438
+ tf_step *step = calloc(1, sizeof(tf_step));
233
439
  if (!step) { free(st->column); free(st); return NULL; }
234
440
  step->process = interpolate_process;
235
441
  step->flush = interpolate_flush;