tranfi 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +177 -0
  2. package/NOTICE +8 -0
  3. package/README.md +443 -51
  4. package/app/assets/{index-pDFMluyz.js → index-BIAIKnrp.js} +1 -1
  5. package/app/index.html +1 -1
  6. package/binding.gyp +55 -3
  7. package/csrc/arena.c +7 -5
  8. package/csrc/batch.c +818 -71
  9. package/csrc/buffer.c +84 -8
  10. package/csrc/cJSON.c +262 -19
  11. package/csrc/cJSON.h +17 -1
  12. package/csrc/codec_csv.c +1074 -181
  13. package/csrc/codec_jsonl.c +830 -118
  14. package/csrc/codec_table.c +108 -78
  15. package/csrc/codec_text.c +286 -68
  16. package/csrc/compiler.c +31 -3
  17. package/csrc/config.h +21 -0
  18. package/csrc/dsl.c +4722 -485
  19. package/csrc/expr.c +363 -55
  20. package/csrc/expr.h +2 -0
  21. package/csrc/internal.h +316 -27
  22. package/csrc/ir.c +65 -18
  23. package/csrc/ir.h +41 -0
  24. package/csrc/ir_schema.c +20 -5
  25. package/csrc/ir_serialize.c +68 -6
  26. package/csrc/ir_sql.c +796 -185
  27. package/csrc/ir_validate.c +462 -6
  28. package/csrc/json_path.c +210 -0
  29. package/csrc/main.c +879 -30
  30. package/csrc/memory_estimate.c +477 -0
  31. package/csrc/op_acf.c +171 -21
  32. package/csrc/op_across.c +477 -0
  33. package/csrc/op_anomaly.c +167 -32
  34. package/csrc/op_assert.c +761 -0
  35. package/csrc/op_bin.c +168 -29
  36. package/csrc/op_cast.c +383 -55
  37. package/csrc/op_clip.c +30 -19
  38. package/csrc/op_date_trunc.c +208 -34
  39. package/csrc/op_datetime.c +259 -77
  40. package/csrc/op_derive.c +65 -97
  41. package/csrc/op_diff.c +146 -30
  42. package/csrc/op_ewma.c +149 -30
  43. package/csrc/op_explode.c +124 -26
  44. package/csrc/op_fill_down.c +125 -53
  45. package/csrc/op_fill_null.c +176 -31
  46. package/csrc/op_filter.c +89 -40
  47. package/csrc/op_frequency.c +571 -43
  48. package/csrc/op_grep.c +36 -18
  49. package/csrc/op_group_agg.c +1790 -119
  50. package/csrc/op_hash.c +48 -15
  51. package/csrc/op_head.c +21 -86
  52. package/csrc/op_interpolate.c +268 -62
  53. package/csrc/op_join.c +2700 -182
  54. package/csrc/op_json_extract.c +227 -0
  55. package/csrc/op_json_filter.c +384 -0
  56. package/csrc/op_json_flatten.c +293 -0
  57. package/csrc/op_json_schema.c +503 -0
  58. package/csrc/op_label_encode.c +328 -53
  59. package/csrc/op_lag.c +181 -0
  60. package/csrc/op_lead.c +141 -89
  61. package/csrc/op_normalize.c +363 -79
  62. package/csrc/op_onehot.c +345 -73
  63. package/csrc/op_pivot.c +1546 -162
  64. package/csrc/op_quarantine.c +189 -0
  65. package/csrc/op_registry.c +2062 -166
  66. package/csrc/op_rename.c +41 -50
  67. package/csrc/op_replace.c +270 -118
  68. package/csrc/op_rleid.c +297 -0
  69. package/csrc/op_rowid.c +559 -0
  70. package/csrc/op_sample.c +80 -23
  71. package/csrc/op_schema.c +1341 -0
  72. package/csrc/op_schema_infer.c +252 -0
  73. package/csrc/op_select.c +265 -65
  74. package/csrc/op_set.c +3449 -0
  75. package/csrc/op_skip.c +30 -87
  76. package/csrc/op_sort.c +670 -124
  77. package/csrc/op_source_name.c +120 -0
  78. package/csrc/op_split.c +65 -28
  79. package/csrc/op_split_data.c +41 -9
  80. package/csrc/op_stack.c +178 -222
  81. package/csrc/op_stats.c +206 -110
  82. package/csrc/op_step.c +217 -55
  83. package/csrc/op_tail.c +21 -12
  84. package/csrc/op_tee.c +338 -0
  85. package/csrc/op_top.c +260 -53
  86. package/csrc/op_trim.c +48 -19
  87. package/csrc/op_unique.c +1193 -150
  88. package/csrc/op_unpivot.c +100 -66
  89. package/csrc/op_validate.c +601 -24
  90. package/csrc/op_window.c +492 -51
  91. package/csrc/path_policy.c +85 -0
  92. package/csrc/pipeline.c +872 -99
  93. package/csrc/recipes.c +3 -1
  94. package/csrc/report.c +73 -30
  95. package/csrc/selector.c +1097 -0
  96. package/csrc/size_utils.c +352 -0
  97. package/csrc/spill.c +317 -0
  98. package/csrc/spill.h +21 -0
  99. package/csrc/tranfi.h +169 -1
  100. package/csrc/transform.h +209 -0
  101. package/csrc/transform_api.c +2237 -0
  102. package/csrc/transform_categorical.c +923 -0
  103. package/csrc/transform_internal.h +472 -0
  104. package/csrc/transform_json.c +3812 -0
  105. package/csrc/transform_numeric.c +1966 -0
  106. package/csrc/transform_sha256.c +154 -0
  107. package/csrc/transform_wasm.h +162 -0
  108. package/csrc/transform_wasm_api.c +1373 -0
  109. package/csrc/wasm_api.c +70 -9
  110. package/napi_api.c +219 -11
  111. package/napi_transform.c +1648 -0
  112. package/napi_transform.h +8 -0
  113. package/package.json +27 -11
  114. package/scripts/install-native.js +76 -0
  115. package/scripts/prepack.js +64 -0
  116. package/scripts/sync-csrc.js +23 -0
  117. package/src/cli.js +81 -41
  118. package/src/engines/duckdb.js +45 -12
  119. package/src/index.js +661 -42
  120. package/src/memory_policy.js +411 -0
  121. package/src/native.js +1 -5
  122. package/src/pipeline.js +454 -31
  123. package/src/recipe_json.js +80 -0
  124. package/src/server.js +10 -8
  125. package/src/transform.js +403 -0
  126. package/src/transform_error.js +10 -0
  127. package/src/wasm.js +6 -4
  128. package/wasm/index.js +498 -10
  129. package/wasm/tranfi_core.js +0 -0
  130. package/wasm/transform.js +1156 -0
  131. package/wasm/worker.js +786 -0
  132. package/csrc/plan.c +0 -206
@@ -0,0 +1,120 @@
1
+ /*
2
+ * op_source_name.c -- Append the current host-provided source name.
3
+ *
4
+ * The source name is pipeline metadata set by the embedding host before push()
5
+ * or input-boundary flushes. The op itself is row-local and never opens files.
6
+ */
7
+
8
+ #include "internal.h"
9
+ #include "cJSON.h"
10
+ #include <stdlib.h>
11
+ #include <string.h>
12
+ #include <stdio.h>
13
+
14
+ typedef struct {
15
+ char *result;
16
+ char *default_value;
17
+ } source_name_state;
18
+
19
+ static int source_name_process(tf_step *self, tf_batch *in, tf_batch **out,
20
+ tf_side_channels *side) {
21
+ source_name_state *st = self ? self->state : NULL;
22
+ if (!st || !in || !out) return TF_ERROR;
23
+ *out = NULL;
24
+
25
+ if (tf_batch_col_index(in, st->result) >= 0) {
26
+ char msg[256];
27
+ snprintf(msg, sizeof(msg), "source-name result column already exists: %s", st->result);
28
+ tf_set_last_error(msg);
29
+ return TF_ERROR;
30
+ }
31
+
32
+ const char *extra_names[1] = {st->result};
33
+ tf_type extra_types[1] = {TF_TYPE_STRING};
34
+ tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows > 0 ? in->n_rows : 1);
35
+ if (!ob) return TF_ERROR;
36
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
37
+ tf_batch_free(ob);
38
+ return TF_ERROR;
39
+ }
40
+
41
+ const char *value = st->default_value ? st->default_value : "";
42
+ if (side && side->source_name && side->source_name[0]) value = side->source_name;
43
+
44
+ for (size_t r = 0; r < in->n_rows; r++) {
45
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
46
+ tf_batch_free(ob);
47
+ return TF_ERROR;
48
+ }
49
+ if (tf_batch_set_string(ob, r, in->n_cols, value) != TF_OK) {
50
+ tf_batch_free(ob);
51
+ return TF_ERROR;
52
+ }
53
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
54
+ tf_batch_free(ob);
55
+ return TF_ERROR;
56
+ }
57
+ }
58
+
59
+ if (ob->n_rows > 0) {
60
+ *out = ob;
61
+ } else {
62
+ tf_batch_free(ob);
63
+ }
64
+ return TF_OK;
65
+ }
66
+
67
+ static int source_name_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
68
+ (void)self;
69
+ (void)side;
70
+ *out = NULL;
71
+ return TF_OK;
72
+ }
73
+
74
+ static void source_name_destroy(tf_step *self) {
75
+ if (!self) return;
76
+ source_name_state *st = self->state;
77
+ if (st) {
78
+ free(st->result);
79
+ free(st->default_value);
80
+ free(st);
81
+ }
82
+ free(self);
83
+ }
84
+
85
+ tf_step *tf_source_name_create(const cJSON *args) {
86
+ const char *result = "_source";
87
+ const char *default_value = "";
88
+ if (args) {
89
+ cJSON *res = cJSON_GetObjectItemCaseSensitive(args, "result");
90
+ if (!cJSON_IsString(res)) res = cJSON_GetObjectItemCaseSensitive(args, "column");
91
+ if (!cJSON_IsString(res)) res = cJSON_GetObjectItemCaseSensitive(args, "as");
92
+ if (cJSON_IsString(res) && res->valuestring && res->valuestring[0]) result = res->valuestring;
93
+ cJSON *def = cJSON_GetObjectItemCaseSensitive(args, "default");
94
+ if (cJSON_IsString(def) && def->valuestring) default_value = def->valuestring;
95
+ }
96
+
97
+ source_name_state *st = calloc(1, sizeof(source_name_state));
98
+ if (!st) return NULL;
99
+ st->result = strdup(result);
100
+ st->default_value = strdup(default_value);
101
+ if (!st->result || !st->default_value) {
102
+ free(st->result);
103
+ free(st->default_value);
104
+ free(st);
105
+ return NULL;
106
+ }
107
+
108
+ tf_step *step = calloc(1, sizeof(tf_step));
109
+ if (!step) {
110
+ free(st->result);
111
+ free(st->default_value);
112
+ free(st);
113
+ return NULL;
114
+ }
115
+ step->process = source_name_process;
116
+ step->flush = source_name_flush;
117
+ step->destroy = source_name_destroy;
118
+ step->state = st;
119
+ return step;
120
+ }
package/csrc/op_split.c CHANGED
@@ -22,44 +22,69 @@ static int split_process(tf_step *self, tf_batch *in, tf_batch **out,
22
22
  split_state *st = self->state;
23
23
  *out = NULL;
24
24
 
25
- size_t out_cols = in->n_cols + st->n_names;
25
+ size_t out_cols = 0;
26
+ if (tf_size_add(in->n_cols, st->n_names, &out_cols) != TF_OK) return TF_ERROR;
26
27
  tf_batch *ob = tf_batch_create(out_cols, in->n_rows);
27
28
  if (!ob) return TF_ERROR;
28
29
 
29
30
  /* Copy existing schema */
30
- for (size_t c = 0; c < in->n_cols; c++)
31
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
31
+ for (size_t c = 0; c < in->n_cols; c++) {
32
+ if (tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]) != TF_OK) {
33
+ tf_batch_free(ob);
34
+ return TF_ERROR;
35
+ }
36
+ }
32
37
  /* Add new columns */
33
- for (size_t k = 0; k < st->n_names; k++)
34
- tf_batch_set_schema(ob, in->n_cols + k, st->names[k], TF_TYPE_STRING);
38
+ for (size_t k = 0; k < st->n_names; k++) {
39
+ if (tf_batch_set_schema(ob, in->n_cols + k, st->names[k], TF_TYPE_STRING) != TF_OK) {
40
+ tf_batch_free(ob);
41
+ return TF_ERROR;
42
+ }
43
+ }
35
44
 
36
45
  int ci = tf_batch_col_index(in, st->column);
37
46
  size_t delim_len = strlen(st->delimiter);
38
47
 
39
48
  for (size_t r = 0; r < in->n_rows; r++) {
40
- tf_batch_copy_row(ob, r, in, r);
49
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
50
+ tf_batch_free(ob);
51
+ return TF_ERROR;
52
+ }
41
53
  /* Initialize new columns as null */
42
- for (size_t k = 0; k < st->n_names; k++)
43
- tf_batch_set_null(ob, r, in->n_cols + k);
54
+ for (size_t k = 0; k < st->n_names; k++) {
55
+ if (tf_batch_set_null(ob, r, in->n_cols + k) != TF_OK) {
56
+ tf_batch_free(ob);
57
+ return TF_ERROR;
58
+ }
59
+ }
44
60
 
45
61
  if (ci >= 0 && !tf_batch_is_null(in, r, ci) && in->col_types[ci] == TF_TYPE_STRING) {
46
62
  const char *val = tf_batch_get_string(in, r, ci);
47
- const char *p = val;
63
+ const char *p = val ? val : "";
48
64
  for (size_t k = 0; k < st->n_names && *p; k++) {
49
65
  const char *found = strstr(p, st->delimiter);
50
66
  size_t tok_len = found ? (size_t)(found - p) : strlen(p);
51
67
  char *tok = malloc(tok_len + 1);
52
- if (tok) {
53
- memcpy(tok, p, tok_len);
54
- tok[tok_len] = '\0';
55
- tf_batch_set_string(ob, r, in->n_cols + k, tok);
56
- free(tok);
68
+ if (!tok) {
69
+ tf_batch_free(ob);
70
+ return TF_ERROR;
71
+ }
72
+ memcpy(tok, p, tok_len);
73
+ tok[tok_len] = '\0';
74
+ int rc = tf_batch_set_string(ob, r, in->n_cols + k, tok);
75
+ free(tok);
76
+ if (rc != TF_OK) {
77
+ tf_batch_free(ob);
78
+ return TF_ERROR;
57
79
  }
58
80
  if (found) p = found + delim_len;
59
81
  else break;
60
82
  }
61
83
  }
62
- ob->n_rows = r + 1;
84
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
85
+ tf_batch_free(ob);
86
+ return TF_ERROR;
87
+ }
63
88
  }
64
89
 
65
90
  *out = ob;
@@ -70,14 +95,19 @@ static int split_flush(tf_step *self, tf_batch **out, tf_side_channels *side) {
70
95
  (void)self; (void)side; *out = NULL; return TF_OK;
71
96
  }
72
97
 
73
- static void split_destroy(tf_step *self) {
74
- split_state *st = self->state;
75
- if (st) {
76
- free(st->column); free(st->delimiter);
98
+ static void split_state_free(split_state *st) {
99
+ if (!st) return;
100
+ free(st->column);
101
+ free(st->delimiter);
102
+ if (st->names) {
77
103
  for (size_t i = 0; i < st->n_names; i++) free(st->names[i]);
78
- free(st->names);
79
- free(st);
80
104
  }
105
+ free(st->names);
106
+ free(st);
107
+ }
108
+
109
+ static void split_destroy(tf_step *self) {
110
+ split_state_free(self ? self->state : NULL);
81
111
  free(self);
82
112
  }
83
113
 
@@ -93,19 +123,26 @@ tf_step *tf_split_create(const cJSON *args) {
93
123
  split_state *st = calloc(1, sizeof(split_state));
94
124
  if (!st) return NULL;
95
125
  st->column = strdup(col_j->valuestring);
126
+ if (!st->column) { split_state_free(st); return NULL; }
96
127
 
97
128
  cJSON *delim_j = cJSON_GetObjectItemCaseSensitive(args, "delimiter");
98
- st->delimiter = strdup(cJSON_IsString(delim_j) ? delim_j->valuestring : " ");
99
-
100
- st->names = calloc(n, sizeof(char *));
101
- st->n_names = n;
129
+ const char *delim = cJSON_IsString(delim_j) ? delim_j->valuestring : " ";
130
+ if (delim[0] == '\0') { split_state_free(st); return NULL; }
131
+ st->delimiter = strdup(delim);
132
+ if (!st->delimiter) { split_state_free(st); return NULL; }
133
+
134
+ st->names = calloc((size_t)n, sizeof(char *));
135
+ if (!st->names) { split_state_free(st); return NULL; }
136
+ st->n_names = (size_t)n;
102
137
  for (int i = 0; i < n; i++) {
103
138
  cJSON *item = cJSON_GetArrayItem(names_j, i);
104
- if (cJSON_IsString(item)) st->names[i] = strdup(item->valuestring);
139
+ if (!cJSON_IsString(item)) { split_state_free(st); return NULL; }
140
+ st->names[i] = strdup(item->valuestring);
141
+ if (!st->names[i]) { split_state_free(st); return NULL; }
105
142
  }
106
143
 
107
- tf_step *step = malloc(sizeof(tf_step));
108
- if (!step) { split_destroy(&(tf_step){.state = st}); return NULL; }
144
+ tf_step *step = calloc(1, sizeof(tf_step));
145
+ if (!step) { split_state_free(st); return NULL; }
109
146
  step->process = split_process;
110
147
  step->flush = split_flush;
111
148
  step->destroy = split_destroy;
@@ -7,6 +7,7 @@
7
7
 
8
8
  #include "internal.h"
9
9
  #include "cJSON.h"
10
+ #include <math.h>
10
11
  #include <stdlib.h>
11
12
  #include <string.h>
12
13
  #include <stdio.h>
@@ -32,19 +33,31 @@ static int split_data_process(tf_step *self, tf_batch *in, tf_batch **out,
32
33
  split_data_state *st = self->state;
33
34
  *out = NULL;
34
35
 
36
+ const char *extra_names[1] = {st->result};
37
+ tf_type extra_types[1] = {TF_TYPE_STRING};
35
38
  tf_batch *ob = tf_batch_create(in->n_cols + 1, in->n_rows);
36
39
  if (!ob) return TF_ERROR;
37
- for (size_t c = 0; c < in->n_cols; c++)
38
- tf_batch_set_schema(ob, c, in->col_names[c], in->col_types[c]);
39
- tf_batch_set_schema(ob, in->n_cols, st->result, TF_TYPE_STRING);
40
+ if (tf_batch_clone_with_extra_cols(ob, in, extra_names, extra_types, 1) != TF_OK) {
41
+ tf_batch_free(ob);
42
+ return TF_ERROR;
43
+ }
40
44
 
41
45
  for (size_t r = 0; r < in->n_rows; r++) {
42
- tf_batch_copy_row(ob, r, in, r);
46
+ if (tf_batch_copy_row(ob, r, in, r) != TF_OK) {
47
+ tf_batch_free(ob);
48
+ return TF_ERROR;
49
+ }
43
50
 
44
51
  double rval = lcg_random(st->seed, st->row_index);
45
52
  const char *label = (rval < st->ratio) ? "train" : "test";
46
- tf_batch_set_string(ob, r, in->n_cols, label);
47
- ob->n_rows = r + 1;
53
+ if (tf_batch_set_string(ob, r, in->n_cols, label) != TF_OK) {
54
+ tf_batch_free(ob);
55
+ return TF_ERROR;
56
+ }
57
+ if (tf_batch_expose_row(ob, r) != TF_OK) {
58
+ tf_batch_free(ob);
59
+ return TF_ERROR;
60
+ }
48
61
  st->row_index++;
49
62
  }
50
63
 
@@ -69,15 +82,34 @@ tf_step *tf_split_data_create(const cJSON *args) {
69
82
  if (!st) return NULL;
70
83
 
71
84
  cJSON *ratio_j = cJSON_GetObjectItemCaseSensitive(args, "ratio");
72
- st->ratio = cJSON_IsNumber(ratio_j) ? ratio_j->valuedouble : 0.8;
85
+ st->ratio = 0.8;
86
+ if (ratio_j) {
87
+ if (!cJSON_IsNumber(ratio_j) || !isfinite(ratio_j->valuedouble) ||
88
+ ratio_j->valuedouble < 0.0 || ratio_j->valuedouble > 1.0) {
89
+ tf_set_last_error("split-data: ratio must be a finite number between 0 and 1");
90
+ free(st);
91
+ return NULL;
92
+ }
93
+ st->ratio = ratio_j->valuedouble;
94
+ }
73
95
 
74
96
  cJSON *seed_j = cJSON_GetObjectItemCaseSensitive(args, "seed");
75
- st->seed = cJSON_IsNumber(seed_j) ? (uint64_t)seed_j->valueint : 42;
97
+ st->seed = 42;
98
+ if (seed_j) {
99
+ size_t parsed_seed = 0;
100
+ if (tf_json_size_value(seed_j, "seed", 0, TF_MAX_SAFE_SIZE_ARG,
101
+ &parsed_seed, "split-data") < 0) {
102
+ free(st);
103
+ return NULL;
104
+ }
105
+ st->seed = (uint64_t)parsed_seed;
106
+ }
76
107
 
77
108
  cJSON *res_j = cJSON_GetObjectItemCaseSensitive(args, "result");
78
109
  st->result = strdup(cJSON_IsString(res_j) ? res_j->valuestring : "_split");
110
+ if (!st->result) { free(st); return NULL; }
79
111
 
80
- tf_step *step = malloc(sizeof(tf_step));
112
+ tf_step *step = calloc(1, sizeof(tf_step));
81
113
  if (!step) { free(st->result); free(st); return NULL; }
82
114
  step->process = split_data_process;
83
115
  step->flush = split_data_flush;