rastack 0.0.16 → 0.0.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/CHANGELOG.md +43874 -43013
  2. package/dist/.prettierrc +7 -7
  3. package/dist/compile/analyze.js +19 -11
  4. package/dist/compile/index.js +26 -17
  5. package/dist/compile/manifest.js +2 -3
  6. package/dist/compile/openapi.js +5 -8
  7. package/dist/compile/program.js +19 -10
  8. package/dist/csv-schema.js +5 -6
  9. package/dist/define/index.js +2 -2
  10. package/dist/entity-generation/delete-method.js +8 -10
  11. package/dist/entity-generation/form.js +80 -82
  12. package/dist/entity-generation/get-method.js +35 -36
  13. package/dist/entity-generation/imports.js +24 -20
  14. package/dist/entity-generation/sync-method.d.ts +1 -1
  15. package/dist/entity-generation/sync-method.js +45 -46
  16. package/dist/entity-generation/update-method.js +9 -10
  17. package/dist/rad-wasm-build.d.ts +1 -1
  18. package/dist/rad-wasm-build.js +15 -6
  19. package/dist/rad.js +9 -2
  20. package/dist/schema/camel-to-pastel.js +1 -2
  21. package/dist/schema/capitalise-first-letter.js +1 -2
  22. package/dist/schema/extract-response.js +3 -4
  23. package/dist/schema/fetch-schema.js +4 -16
  24. package/dist/schema/remove-non-model-paths.js +1 -2
  25. package/dist/schema/to-camel-case.js +1 -2
  26. package/dist/schema/to-pastel-case.js +1 -2
  27. package/dist/schema-convert.js +22 -33
  28. package/dist/schema-entities.js +177 -189
  29. package/dist/schema-fetch.js +46 -63
  30. package/dist/schema-full.js +23 -36
  31. package/dist/schema-index.js +16 -27
  32. package/dist/schema-params.js +11 -22
  33. package/dist/seed.js +95 -97
  34. package/hooks/form/form.ts +207 -207
  35. package/hooks/form/index.ts +8 -8
  36. package/hooks/form/interfaces.ts +217 -217
  37. package/hooks/form/structure.ts +39 -39
  38. package/hooks/form/validate-schema.ts +49 -49
  39. package/hooks/index.ts +3 -3
  40. package/hooks/query/api.ts +42 -42
  41. package/hooks/query/delete.ts +45 -45
  42. package/hooks/query/fetch.ts +48 -48
  43. package/hooks/query/index.ts +21 -21
  44. package/hooks/query/interfaces.ts +111 -111
  45. package/hooks/query/list.ts +286 -286
  46. package/hooks/query/update.ts +88 -88
  47. package/hooks/query/url.ts +31 -31
  48. package/hooks/real-time/index.ts +1 -1
  49. package/hooks/real-time/pusher.ts +43 -43
  50. package/jest.config.cjs +6 -6
  51. package/package.json +57 -57
  52. package/provider/index.ts +13 -13
  53. package/provider/provider.tsx +187 -187
  54. package/provider/types.ts +114 -114
  55. package/provider/warehouse.ts +167 -167
  56. package/provider/wasm.ts +124 -124
  57. package/runtime.ts +3 -3
  58. package/src/.prettierrc +7 -7
  59. package/src/compile/analyze.ts +224 -224
  60. package/src/compile/index.ts +86 -86
  61. package/src/compile/manifest.ts +10 -10
  62. package/src/compile/model.ts +69 -69
  63. package/src/compile/openapi.ts +266 -266
  64. package/src/compile/program.ts +40 -40
  65. package/src/csv-schema.ts +187 -187
  66. package/src/define/index.ts +162 -162
  67. package/src/entity-generation/delete-method.ts +37 -37
  68. package/src/entity-generation/form.ts +169 -169
  69. package/src/entity-generation/get-method.ts +80 -80
  70. package/src/entity-generation/imports.ts +110 -110
  71. package/src/entity-generation/sync-method.ts +222 -222
  72. package/src/entity-generation/update-method.ts +38 -38
  73. package/src/index.ts +77 -77
  74. package/src/rad-compile.ts +88 -88
  75. package/src/rad-wasm-build.ts +102 -102
  76. package/src/rad.ts +101 -101
  77. package/src/scan.ts +89 -89
  78. package/src/schema/camel-to-pastel.ts +3 -3
  79. package/src/schema/capitalise-first-letter.ts +3 -3
  80. package/src/schema/extract-response.ts +38 -38
  81. package/src/schema/fetch-schema.ts +12 -12
  82. package/src/schema/remove-non-model-paths.ts +13 -13
  83. package/src/schema/to-camel-case.ts +7 -7
  84. package/src/schema/to-pastel-case.ts +6 -6
  85. package/src/schema-convert.ts +61 -61
  86. package/src/schema-entities.ts +401 -401
  87. package/src/schema-fetch.ts +82 -82
  88. package/src/schema-full.ts +38 -38
  89. package/src/schema-index.ts +29 -29
  90. package/src/schema-params.ts +116 -116
  91. package/src/seed.ts +297 -297
  92. package/sync/engine.ts +392 -392
  93. package/sync/hooks.ts +450 -450
  94. package/sync/index.ts +8 -8
  95. package/sync/persistence.ts +237 -237
  96. package/sync/provider.tsx +91 -91
  97. package/sync/registry.ts +36 -36
  98. package/sync/store.ts +126 -126
  99. package/sync/transactions.ts +300 -300
  100. package/sync/types.ts +94 -94
  101. package/test/compile.spec.ts +192 -192
  102. package/test/csv-schema.spec.ts +143 -143
  103. package/test/schema-entities.spec.ts +688 -688
  104. package/tsconfig.json +24 -24
  105. package/types.ts +18 -18
package/src/seed.ts CHANGED
@@ -1,297 +1,297 @@
1
- #!/usr/bin/env node
2
-
3
- /**
4
- * rad seed
5
- *
6
- * Loads CSV files from a local directory or GCS bucket into the target
7
- * Iceberg namespace. Performs schema inference, PII check, and idempotent
8
- * table creation before writing rows.
9
- *
10
- * Usage:
11
- * rad seed [--env dev|staging] [--source gs://bucket/ | --source ./myapp-dev/]
12
- *
13
- * Environment variables (set via CI secrets or .env):
14
- * GCP_PROJECT GCP project ID
15
- * ICEBERG_WAREHOUSE gs://... path to Iceberg warehouse root
16
- * ICEBERG_CATALOG_URI sqlite:///... or http://... catalog URI
17
- */
18
-
19
- import fs from "fs";
20
- import path from "path";
21
- import { execSync } from "child_process";
22
- import {
23
- discoverCsvFiles,
24
- inferSchema,
25
- parseCsv,
26
- TableSchema,
27
- } from "./csv-schema";
28
-
29
- // ---------------------------------------------------------------------------
30
- // Args
31
- // ---------------------------------------------------------------------------
32
- const args = process.argv.slice(2);
33
-
34
- function getArg(flag: string): string | null {
35
- const i = args.indexOf(flag);
36
- return i !== -1 && args[i + 1] ? args[i + 1] : null;
37
- }
38
-
39
- const env = getArg("--env") ?? "dev";
40
- const source = getArg("--source") ?? path.join(process.cwd(), "myapp-dev");
41
-
42
- const repoRoot = process.cwd();
43
-
44
- // ---------------------------------------------------------------------------
45
- // Step 1: resolve CSV files
46
- // ---------------------------------------------------------------------------
47
- function resolveFiles(): string[] {
48
- if (source.startsWith("gs://")) {
49
- return downloadFromGcs(source);
50
- }
51
- return discoverCsvFiles(source);
52
- }
53
-
54
- function downloadFromGcs(gcsUri: string): string[] {
55
- const tmpDir = path.join(repoRoot, ".rad", "tmp", "seed");
56
- fs.mkdirSync(tmpDir, { recursive: true });
57
-
58
- console.log(`Downloading CSVs from ${gcsUri} ...`);
59
- // `rad seed` is a local/CI developer tool; `gcsUri` is an operator-supplied
60
- // --source argument (a gs:// bucket the operator controls), not untrusted
61
- // network input, so there is no command-injection surface here.
62
- try {
63
- // nosemgrep: javascript.lang.security.detect-child-process.detect-child-process
64
- execSync(`gcloud storage cp "${gcsUri}*.csv" "${tmpDir}/" --quiet`, {
65
- stdio: "inherit",
66
- });
67
- } catch {
68
- // gcloud storage cp uses gsutil-style globs — fall back
69
- // nosemgrep: javascript.lang.security.detect-child-process.detect-child-process
70
- execSync(`gsutil -m cp "${gcsUri}*.csv" "${tmpDir}/"`, {
71
- stdio: "inherit",
72
- });
73
- }
74
-
75
- return discoverCsvFiles(tmpDir);
76
- }
77
-
78
- // ---------------------------------------------------------------------------
79
- // Step 2: validate schemas (PII check — hard fail in CI)
80
- // ---------------------------------------------------------------------------
81
- function validateSchema(csvPath: string): TableSchema {
82
- const { schema, piiViolations } = inferSchema(csvPath, repoRoot);
83
-
84
- if (piiViolations.length > 0) {
85
- const list = piiViolations
86
- .map((v) => ` ${v.column}: ${v.reason}`)
87
- .join("\n");
88
- throw new Error(
89
- `PII detected in ${path.basename(csvPath)}:\n${list}\n` +
90
- `Remove PII data before seeding. Run "rad scan" to review.`,
91
- );
92
- }
93
-
94
- return schema;
95
- }
96
-
97
- // ---------------------------------------------------------------------------
98
- // Step 3: emit Iceberg DDL for a table
99
- // ---------------------------------------------------------------------------
100
- function icebergType(col: TableSchema["columns"][number]["type"]): string {
101
- switch (col) {
102
- case "integer":
103
- return "long";
104
- case "float":
105
- return "double";
106
- case "boolean":
107
- return "boolean";
108
- case "date":
109
- return "date";
110
- case "datetime":
111
- return "timestamp";
112
- default:
113
- return "string";
114
- }
115
- }
116
-
117
- function buildCreateTableSql(schema: TableSchema, namespace: string): string {
118
- const cols = schema.columns
119
- .map((c) => {
120
- const type = icebergType(c.type);
121
- const nullable = c.nullable ? "" : " NOT NULL";
122
- return ` ${c.name} ${type}${nullable}`;
123
- })
124
- .join(",\n");
125
-
126
- return (
127
- `CREATE TABLE IF NOT EXISTS ${namespace}.${schema.tableName} (\n` +
128
- `${cols}\n` +
129
- `) USING iceberg\n` +
130
- `TBLPROPERTIES (\n` +
131
- ` 'write.format.default'='parquet',\n` +
132
- ` 'write.metadata.fingerprint'='${schema.fingerprint}'\n` +
133
- `);`
134
- );
135
- }
136
-
137
- // ---------------------------------------------------------------------------
138
- // Step 4: write schema fingerprints to registry
139
- // ---------------------------------------------------------------------------
140
- function writeFingerprint(schema: TableSchema): void {
141
- const registryDir = path.join(repoRoot, ".rad", "registry");
142
- fs.mkdirSync(registryDir, { recursive: true });
143
- const registryPath = path.join(registryDir, `${schema.tableName}.json`);
144
-
145
- let existing: Record<string, unknown> = {};
146
- if (fs.existsSync(registryPath)) {
147
- existing = JSON.parse(fs.readFileSync(registryPath, "utf-8"));
148
- }
149
-
150
- const prev = (existing as { fingerprint?: string }).fingerprint;
151
- const changed = prev && prev !== schema.fingerprint;
152
-
153
- const record = {
154
- tableName: schema.tableName,
155
- fingerprint: schema.fingerprint,
156
- columns: schema.columns,
157
- rowCount: schema.rowCount,
158
- seededAt: new Date().toISOString(),
159
- env,
160
- previousFingerprint: prev ?? null,
161
- };
162
-
163
- fs.writeFileSync(registryPath, JSON.stringify(record, null, 2));
164
-
165
- if (changed) {
166
- console.warn(
167
- ` ⚠ Schema change detected for "${schema.tableName}": ` +
168
- `${prev} → ${schema.fingerprint}. A migration may be required.`,
169
- );
170
- }
171
- }
172
-
173
- // ---------------------------------------------------------------------------
174
- // Step 5: build a Python seed script and run it via django-iceberg
175
- // ---------------------------------------------------------------------------
176
- function buildPythonSeedScript(
177
- schemas: TableSchema[],
178
- csvPaths: string[],
179
- ): string {
180
- const namespace = env;
181
- const warehouse = process.env.ICEBERG_WAREHOUSE ?? "data/warehouse";
182
- const catalogUri =
183
- process.env.ICEBERG_CATALOG_URI ?? "sqlite:///data/catalog.db";
184
-
185
- const tableBlocks = schemas
186
- .map((schema, i) => {
187
- const ddl = buildCreateTableSql(schema, namespace).replace(/`/g, "\\`");
188
- const csvPathEscaped = csvPaths[i].replace(/\\/g, "/");
189
- return `
190
- # --- ${schema.tableName} ---
191
- ddl = """${ddl}"""
192
- cat.execute(ddl)
193
- df = pl.read_csv("${csvPathEscaped}", infer_schema_length=1000)
194
- tbl = cat.load_table("${namespace}.${schema.tableName}")
195
- tbl.overwrite(df.to_arrow())
196
- print(f" Seeded ${schema.tableName}: ${schema.rowCount} rows")
197
- `;
198
- })
199
- .join("\n");
200
-
201
- return `
202
- import polars as pl
203
- from pyiceberg.catalog import load_catalog
204
-
205
- cat = load_catalog(
206
- "default",
207
- **{
208
- "uri": "${catalogUri}",
209
- "warehouse": "${warehouse}",
210
- }
211
- )
212
-
213
- try:
214
- cat.create_namespace("${namespace}")
215
- except Exception:
216
- pass # namespace already exists
217
-
218
- ${tableBlocks}
219
-
220
- print("Seed complete.")
221
- `.trim();
222
- }
223
-
224
- // ---------------------------------------------------------------------------
225
- // Main
226
- // ---------------------------------------------------------------------------
227
- async function main() {
228
- console.log(`\nrad seed env=${env} source=${source}\n`);
229
-
230
- const files = resolveFiles();
231
- if (files.length === 0) {
232
- console.log("No CSV files found. Nothing to seed.");
233
- process.exit(0);
234
- }
235
-
236
- console.log(`Found ${files.length} CSV file(s):\n`);
237
-
238
- const schemas: TableSchema[] = [];
239
- for (const csvPath of files) {
240
- const rel = path.relative(repoRoot, csvPath);
241
- process.stdout.write(` Validating ${rel} ... `);
242
- try {
243
- const schema = validateSchema(csvPath);
244
- schemas.push(schema);
245
- console.log(
246
- `OK [${schema.columns.length} cols, ${schema.rowCount} rows]`,
247
- );
248
- } catch (err) {
249
- console.error(`FAIL\n${(err as Error).message}`);
250
- process.exit(1);
251
- }
252
- }
253
-
254
- // Write fingerprints before seeding so CI can detect schema drift
255
- for (const schema of schemas) {
256
- writeFingerprint(schema);
257
- }
258
-
259
- // Emit DDL summary
260
- const ddlDir = path.join(repoRoot, ".rad", "ddl");
261
- fs.mkdirSync(ddlDir, { recursive: true });
262
- for (const schema of schemas) {
263
- const ddl = buildCreateTableSql(schema, env);
264
- fs.writeFileSync(path.join(ddlDir, `${schema.tableName}.sql`), ddl + "\n");
265
- }
266
-
267
- console.log(`\nDDL written to .rad/ddl/`);
268
-
269
- // Run Python seed via pyiceberg if available
270
- const seedScript = buildPythonSeedScript(schemas, files);
271
- const scriptPath = path.join(repoRoot, ".rad", "tmp", "seed.py");
272
- fs.mkdirSync(path.dirname(scriptPath), { recursive: true });
273
- fs.writeFileSync(scriptPath, seedScript);
274
-
275
- const pythonBin = process.env.PYTHON ?? "python";
276
- try {
277
- console.log(`\nSeeding Iceberg namespace "${env}" ...\n`);
278
- // Runs an internally-generated seed script with the operator-configured
279
- // python binary; no untrusted input, so no command-injection surface.
280
- // nosemgrep: javascript.lang.security.detect-child-process.detect-child-process
281
- execSync(`${pythonBin} "${scriptPath}"`, { stdio: "inherit" });
282
- } catch {
283
- console.error(
284
- "\nPython seed step failed. Ensure pyiceberg and polars are installed:\n" +
285
- " pip install pyiceberg polars pyarrow\n\n" +
286
- `Generated seed script: ${scriptPath}`,
287
- );
288
- process.exit(1);
289
- }
290
-
291
- console.log("\nDone.");
292
- }
293
-
294
- main().catch((err) => {
295
- console.error(err);
296
- process.exit(1);
297
- });
1
+ #!/usr/bin/env node
2
+
3
+ /**
4
+ * rad seed
5
+ *
6
+ * Loads CSV files from a local directory or GCS bucket into the target
7
+ * Iceberg namespace. Performs schema inference, PII check, and idempotent
8
+ * table creation before writing rows.
9
+ *
10
+ * Usage:
11
+ * rad seed [--env dev|staging] [--source gs://bucket/ | --source ./myapp-dev/]
12
+ *
13
+ * Environment variables (set via CI secrets or .env):
14
+ * GCP_PROJECT GCP project ID
15
+ * ICEBERG_WAREHOUSE gs://... path to Iceberg warehouse root
16
+ * ICEBERG_CATALOG_URI sqlite:///... or http://... catalog URI
17
+ */
18
+
19
+ import fs from "fs";
20
+ import path from "path";
21
+ import { execSync } from "child_process";
22
+ import {
23
+ discoverCsvFiles,
24
+ inferSchema,
25
+ parseCsv,
26
+ TableSchema,
27
+ } from "./csv-schema";
28
+
29
+ // ---------------------------------------------------------------------------
30
+ // Args
31
+ // ---------------------------------------------------------------------------
32
+ const args = process.argv.slice(2);
33
+
34
+ function getArg(flag: string): string | null {
35
+ const i = args.indexOf(flag);
36
+ return i !== -1 && args[i + 1] ? args[i + 1] : null;
37
+ }
38
+
39
+ const env = getArg("--env") ?? "dev";
40
+ const source = getArg("--source") ?? path.join(process.cwd(), "myapp-dev");
41
+
42
+ const repoRoot = process.cwd();
43
+
44
+ // ---------------------------------------------------------------------------
45
+ // Step 1: resolve CSV files
46
+ // ---------------------------------------------------------------------------
47
+ function resolveFiles(): string[] {
48
+ if (source.startsWith("gs://")) {
49
+ return downloadFromGcs(source);
50
+ }
51
+ return discoverCsvFiles(source);
52
+ }
53
+
54
+ function downloadFromGcs(gcsUri: string): string[] {
55
+ const tmpDir = path.join(repoRoot, ".rad", "tmp", "seed");
56
+ fs.mkdirSync(tmpDir, { recursive: true });
57
+
58
+ console.log(`Downloading CSVs from ${gcsUri} ...`);
59
+ // `rad seed` is a local/CI developer tool; `gcsUri` is an operator-supplied
60
+ // --source argument (a gs:// bucket the operator controls), not untrusted
61
+ // network input, so there is no command-injection surface here.
62
+ try {
63
+ // nosemgrep: javascript.lang.security.detect-child-process.detect-child-process
64
+ execSync(`gcloud storage cp "${gcsUri}*.csv" "${tmpDir}/" --quiet`, {
65
+ stdio: "inherit",
66
+ });
67
+ } catch {
68
+ // gcloud storage cp uses gsutil-style globs — fall back
69
+ // nosemgrep: javascript.lang.security.detect-child-process.detect-child-process
70
+ execSync(`gsutil -m cp "${gcsUri}*.csv" "${tmpDir}/"`, {
71
+ stdio: "inherit",
72
+ });
73
+ }
74
+
75
+ return discoverCsvFiles(tmpDir);
76
+ }
77
+
78
+ // ---------------------------------------------------------------------------
79
+ // Step 2: validate schemas (PII check — hard fail in CI)
80
+ // ---------------------------------------------------------------------------
81
+ function validateSchema(csvPath: string): TableSchema {
82
+ const { schema, piiViolations } = inferSchema(csvPath, repoRoot);
83
+
84
+ if (piiViolations.length > 0) {
85
+ const list = piiViolations
86
+ .map((v) => ` ${v.column}: ${v.reason}`)
87
+ .join("\n");
88
+ throw new Error(
89
+ `PII detected in ${path.basename(csvPath)}:\n${list}\n` +
90
+ `Remove PII data before seeding. Run "rad scan" to review.`,
91
+ );
92
+ }
93
+
94
+ return schema;
95
+ }
96
+
97
+ // ---------------------------------------------------------------------------
98
+ // Step 3: emit Iceberg DDL for a table
99
+ // ---------------------------------------------------------------------------
100
+ function icebergType(col: TableSchema["columns"][number]["type"]): string {
101
+ switch (col) {
102
+ case "integer":
103
+ return "long";
104
+ case "float":
105
+ return "double";
106
+ case "boolean":
107
+ return "boolean";
108
+ case "date":
109
+ return "date";
110
+ case "datetime":
111
+ return "timestamp";
112
+ default:
113
+ return "string";
114
+ }
115
+ }
116
+
117
+ function buildCreateTableSql(schema: TableSchema, namespace: string): string {
118
+ const cols = schema.columns
119
+ .map((c) => {
120
+ const type = icebergType(c.type);
121
+ const nullable = c.nullable ? "" : " NOT NULL";
122
+ return ` ${c.name} ${type}${nullable}`;
123
+ })
124
+ .join(",\n");
125
+
126
+ return (
127
+ `CREATE TABLE IF NOT EXISTS ${namespace}.${schema.tableName} (\n` +
128
+ `${cols}\n` +
129
+ `) USING iceberg\n` +
130
+ `TBLPROPERTIES (\n` +
131
+ ` 'write.format.default'='parquet',\n` +
132
+ ` 'write.metadata.fingerprint'='${schema.fingerprint}'\n` +
133
+ `);`
134
+ );
135
+ }
136
+
137
+ // ---------------------------------------------------------------------------
138
+ // Step 4: write schema fingerprints to registry
139
+ // ---------------------------------------------------------------------------
140
+ function writeFingerprint(schema: TableSchema): void {
141
+ const registryDir = path.join(repoRoot, ".rad", "registry");
142
+ fs.mkdirSync(registryDir, { recursive: true });
143
+ const registryPath = path.join(registryDir, `${schema.tableName}.json`);
144
+
145
+ let existing: Record<string, unknown> = {};
146
+ if (fs.existsSync(registryPath)) {
147
+ existing = JSON.parse(fs.readFileSync(registryPath, "utf-8"));
148
+ }
149
+
150
+ const prev = (existing as { fingerprint?: string }).fingerprint;
151
+ const changed = prev && prev !== schema.fingerprint;
152
+
153
+ const record = {
154
+ tableName: schema.tableName,
155
+ fingerprint: schema.fingerprint,
156
+ columns: schema.columns,
157
+ rowCount: schema.rowCount,
158
+ seededAt: new Date().toISOString(),
159
+ env,
160
+ previousFingerprint: prev ?? null,
161
+ };
162
+
163
+ fs.writeFileSync(registryPath, JSON.stringify(record, null, 2));
164
+
165
+ if (changed) {
166
+ console.warn(
167
+ ` ⚠ Schema change detected for "${schema.tableName}": ` +
168
+ `${prev} → ${schema.fingerprint}. A migration may be required.`,
169
+ );
170
+ }
171
+ }
172
+
173
+ // ---------------------------------------------------------------------------
174
+ // Step 5: build a Python seed script and run it via django-iceberg
175
+ // ---------------------------------------------------------------------------
176
+ function buildPythonSeedScript(
177
+ schemas: TableSchema[],
178
+ csvPaths: string[],
179
+ ): string {
180
+ const namespace = env;
181
+ const warehouse = process.env.ICEBERG_WAREHOUSE ?? "data/warehouse";
182
+ const catalogUri =
183
+ process.env.ICEBERG_CATALOG_URI ?? "sqlite:///data/catalog.db";
184
+
185
+ const tableBlocks = schemas
186
+ .map((schema, i) => {
187
+ const ddl = buildCreateTableSql(schema, namespace).replace(/`/g, "\\`");
188
+ const csvPathEscaped = csvPaths[i].replace(/\\/g, "/");
189
+ return `
190
+ # --- ${schema.tableName} ---
191
+ ddl = """${ddl}"""
192
+ cat.execute(ddl)
193
+ df = pl.read_csv("${csvPathEscaped}", infer_schema_length=1000)
194
+ tbl = cat.load_table("${namespace}.${schema.tableName}")
195
+ tbl.overwrite(df.to_arrow())
196
+ print(f" Seeded ${schema.tableName}: ${schema.rowCount} rows")
197
+ `;
198
+ })
199
+ .join("\n");
200
+
201
+ return `
202
+ import polars as pl
203
+ from pyiceberg.catalog import load_catalog
204
+
205
+ cat = load_catalog(
206
+ "default",
207
+ **{
208
+ "uri": "${catalogUri}",
209
+ "warehouse": "${warehouse}",
210
+ }
211
+ )
212
+
213
+ try:
214
+ cat.create_namespace("${namespace}")
215
+ except Exception:
216
+ pass # namespace already exists
217
+
218
+ ${tableBlocks}
219
+
220
+ print("Seed complete.")
221
+ `.trim();
222
+ }
223
+
224
+ // ---------------------------------------------------------------------------
225
+ // Main
226
+ // ---------------------------------------------------------------------------
227
+ async function main() {
228
+ console.log(`\nrad seed env=${env} source=${source}\n`);
229
+
230
+ const files = resolveFiles();
231
+ if (files.length === 0) {
232
+ console.log("No CSV files found. Nothing to seed.");
233
+ process.exit(0);
234
+ }
235
+
236
+ console.log(`Found ${files.length} CSV file(s):\n`);
237
+
238
+ const schemas: TableSchema[] = [];
239
+ for (const csvPath of files) {
240
+ const rel = path.relative(repoRoot, csvPath);
241
+ process.stdout.write(` Validating ${rel} ... `);
242
+ try {
243
+ const schema = validateSchema(csvPath);
244
+ schemas.push(schema);
245
+ console.log(
246
+ `OK [${schema.columns.length} cols, ${schema.rowCount} rows]`,
247
+ );
248
+ } catch (err) {
249
+ console.error(`FAIL\n${(err as Error).message}`);
250
+ process.exit(1);
251
+ }
252
+ }
253
+
254
+ // Write fingerprints before seeding so CI can detect schema drift
255
+ for (const schema of schemas) {
256
+ writeFingerprint(schema);
257
+ }
258
+
259
+ // Emit DDL summary
260
+ const ddlDir = path.join(repoRoot, ".rad", "ddl");
261
+ fs.mkdirSync(ddlDir, { recursive: true });
262
+ for (const schema of schemas) {
263
+ const ddl = buildCreateTableSql(schema, env);
264
+ fs.writeFileSync(path.join(ddlDir, `${schema.tableName}.sql`), ddl + "\n");
265
+ }
266
+
267
+ console.log(`\nDDL written to .rad/ddl/`);
268
+
269
+ // Run Python seed via pyiceberg if available
270
+ const seedScript = buildPythonSeedScript(schemas, files);
271
+ const scriptPath = path.join(repoRoot, ".rad", "tmp", "seed.py");
272
+ fs.mkdirSync(path.dirname(scriptPath), { recursive: true });
273
+ fs.writeFileSync(scriptPath, seedScript);
274
+
275
+ const pythonBin = process.env.PYTHON ?? "python";
276
+ try {
277
+ console.log(`\nSeeding Iceberg namespace "${env}" ...\n`);
278
+ // Runs an internally-generated seed script with the operator-configured
279
+ // python binary; no untrusted input, so no command-injection surface.
280
+ // nosemgrep: javascript.lang.security.detect-child-process.detect-child-process
281
+ execSync(`${pythonBin} "${scriptPath}"`, { stdio: "inherit" });
282
+ } catch {
283
+ console.error(
284
+ "\nPython seed step failed. Ensure pyiceberg and polars are installed:\n" +
285
+ " pip install pyiceberg polars pyarrow\n\n" +
286
+ `Generated seed script: ${scriptPath}`,
287
+ );
288
+ process.exit(1);
289
+ }
290
+
291
+ console.log("\nDone.");
292
+ }
293
+
294
+ main().catch((err) => {
295
+ console.error(err);
296
+ process.exit(1);
297
+ });