@thinkingai/ae-cli 6.0.44 → 6.0.46
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/README.zh.md +5 -2
- package/dist/{auth-2WTQOP77.js → auth-GBMV6TEJ.js} +3 -3
- package/dist/{auth-77BUFLGC.js → auth-ROB2EDYV.js} +12 -13
- package/dist/{capability-WA37LSIR.js → capability-GQ47BCFI.js} +49 -12
- package/dist/{capability-ENHWPMTG.js → capability-IINANQJA.js} +50 -13
- package/dist/{chunk-QI524WPY.js → chunk-2MN54X6H.js} +5 -5
- package/dist/chunk-3FY3RJ26.js +293 -0
- package/dist/{chunk-PTE56QPL.js → chunk-3KWQYGYI.js} +4 -0
- package/dist/{chunk-UOUS37JQ.js → chunk-4XXOWOTA.js} +3 -3
- package/dist/{chunk-GT46FPXN.js → chunk-BYYS3ANB.js} +17 -8
- package/dist/{chunk-EE45RGOO.js → chunk-DQRPU6EE.js} +3 -3
- package/dist/{chunk-MGU2N3HW.js → chunk-GJLGIMAO.js} +3 -3
- package/dist/{chunk-4SGZG4XY.js → chunk-J2DEBMRF.js} +9 -7
- package/dist/{chunk-LYVNONC4.js → chunk-JHENBQ5B.js} +35 -0
- package/dist/{chunk-ILIU36SU.js → chunk-OMPRXM3V.js} +3 -3
- package/dist/{chunk-VPKZ7I72.js → chunk-QNOLN2LJ.js} +2 -2
- package/dist/chunk-RJDU7NYP.js +1198 -0
- package/dist/{chunk-4KVPKXFX.js → chunk-RNAALWJK.js} +2 -2
- package/dist/{chunk-C4MGVGJW.js → chunk-SERWF6G5.js} +1 -1
- package/dist/{chunk-QEFCIRNJ.js → chunk-XNVMVIUI.js} +5 -5
- package/dist/{chunk-P3FGXJTU.js → chunk-ZQ47LWTI.js} +4 -4
- package/dist/chunk-ZQKDZXDO.js +317 -0
- package/dist/{client-TKG4WBHN.js → client-L2YDMHQ6.js} +5 -4
- package/dist/{community-report-client-FI4LNVYS.js → community-report-client-C7WDGET3.js} +1 -2
- package/dist/{config-RE6CMGPK.js → config-BMYZX2UE.js} +7 -6
- package/dist/{data-integration-2MYMANJI.js → data-integration-QEKDWQDY.js} +1741 -212
- package/dist/index.js +124 -1239
- package/dist/{local-data-upload-client-BWHSUQQK.js → local-data-upload-client-4YYHSYD6.js} +1 -2
- package/dist/{memory-CHRU2F7W.js → memory-3ORCR7JH.js} +6 -6
- package/dist/{memory-YK33G4T7.js → memory-I2WXDTV2.js} +5 -5
- package/dist/{metadata-AXVHAAZG.js → metadata-A6QLH3IS.js} +9 -9
- package/dist/{metadata-GGEHHPBE.js → metadata-HC7GBTTD.js} +10 -10
- package/dist/{model-NR3JHFSJ.js → model-HLHIEFMU.js} +5 -5
- package/dist/{model-K3KLWIW6.js → model-UGRDX4MW.js} +6 -6
- package/dist/personal-semantic-preference-QZFAJWGE.js +239 -0
- package/dist/personal-semantic-preference-YXAZBFVW.js +239 -0
- package/dist/{sync-FCKOVWWS.js → sync-HKIOZXQE.js} +6 -6
- package/dist/{sync-DAVKYVMW.js → sync-TFHU2UTG.js} +7 -7
- package/dist/{te-agent-HLW4VTQK.js → te-agent-BR6VDBNX.js} +9 -8
- package/dist/{te-agent-4BKBODMF.js → te-agent-VLYOV7S4.js} +8 -7
- package/dist/{te-analysis-WX7LRVAV.js → te-analysis-JECYCV6K.js} +436 -37
- package/dist/{te-analysis-YKFDBMA6.js → te-analysis-MUKUXJL4.js} +437 -38
- package/dist/{te-community-HLC43QKH.js → te-community-5DMNKJWY.js} +5 -5
- package/dist/{te-community-6HPBWJUZ.js → te-community-ISDQWJU7.js} +6 -6
- package/dist/{te-dataops-HDRUXY4K.js → te-dataops-6P5IKWNJ.js} +8 -7
- package/dist/{te-dataops-EJP56W3K.js → te-dataops-CVULXNVB.js} +7 -6
- package/dist/{te-engage-UBHPAJCO.js → te-engage-ELA3C5BM.js} +8 -8
- package/dist/{te-engage-WLDTS5J5.js → te-engage-MC5IQZIU.js} +9 -9
- package/dist/{te-kb-APXBWBDY.js → te-kb-RCLSSH2Q.js} +251 -127
- package/dist/{te-system-Z77IKZFN.js → te-system-FXITO2JG.js} +5 -5
- package/dist/{te-system-YARIK4S5.js → te-system-K2GYMCTB.js} +6 -6
- package/dist/{te-team-EFKWYKMK.js → te-team-ADOC2ROP.js} +6 -6
- package/dist/{update-OGPSZM5A.js → update-YCYCKJOO.js} +7 -6
- package/package.json +2 -1
- package/skills/ae-agent/SKILL.md +3 -4
- package/skills/ae-agent/references/edit-skill.md +3 -0
- package/skills/ae-agent/references/get-skill-content.md +1 -1
- package/skills/ae-agent/references/rescan-skills.md +15 -13
- package/skills/ae-agent/references/upload-skill.md +7 -4
- package/skills/ae-analysis/SKILL.md +29 -4
- package/skills/ae-analysis/metadata_resolution.md +38 -4
- package/skills/ae-analysis/references/analysis_data_retrieval.md +29 -0
- package/skills/ae-analysis/references/asset_authentication_export.md +22 -0
- package/skills/ae-analysis/references/asset_authentication_list.md +18 -14
- package/skills/ae-analysis/references/asset_authentication_update.md +29 -14
- package/skills/ae-analysis/references/command_index.md +17 -9
- package/skills/ae-analysis/references/dashboard_get.md +18 -1
- package/skills/ae-analysis/references/dashboard_update.md +3 -0
- package/skills/ae-analysis/references/personal_semantic_preference_add.md +23 -0
- package/skills/ae-analysis/references/personal_semantic_preference_delete.md +17 -0
- package/skills/ae-analysis/references/personal_semantic_preference_get.md +17 -0
- package/skills/ae-analysis/references/personal_semantic_preference_list.md +19 -0
- package/skills/ae-analysis/references/personal_semantic_preference_update.md +19 -0
- package/skills/ae-data-integration/SKILL.md +23 -4
- package/skills/ae-data-integration/references/custom-layer.md +93 -0
- package/skills/ae-data-integration/references/error-handling.md +92 -0
- package/skills/ae-data-integration/references/handoff.md +77 -18
- package/skills/ae-data-integration/references/local-analysis.md +1 -1
- package/skills/ae-data-integration/references/reuse.md +9 -5
- package/skills/ae-data-integration/references/sink-upload.md +1 -1
- package/skills/ae-data-integration/references/source-inspect.md +32 -13
- package/skills/ae-data-integration/references/tracking-plan.md +7 -5
- package/skills/ae-data-integration/references/transform.md +10 -10
- package/skills/ae-data-integration/references/ue-mapping.md +33 -11
- package/skills/ae-engage/references/build-task-save-guide.md +9 -0
- package/skills/ae-engage/references/save-task.md +82 -0
- package/skills/ae-kb/SKILL.md +56 -51
- package/skills/ae-kb/references/query-workflow.md +112 -0
- package/skills/ae-kb-discovery/SKILL.md +105 -0
- package/dist/chunk-QGM4M3NI.js +0 -37
- package/dist/chunk-ZZUOD757.js +0 -598
|
@@ -1,13 +1,108 @@
|
|
|
1
|
+
import {
|
|
2
|
+
validateAndFix,
|
|
3
|
+
validateDraft
|
|
4
|
+
} from "./chunk-RJDU7NYP.js";
|
|
5
|
+
import {
|
|
6
|
+
getConfigDir
|
|
7
|
+
} from "./chunk-3FY3RJ26.js";
|
|
1
8
|
import {
|
|
2
9
|
CliValidationError,
|
|
3
10
|
LocalDataUploadError
|
|
4
11
|
} from "./chunk-UW5UN47B.js";
|
|
5
|
-
import "./chunk-
|
|
12
|
+
import "./chunk-JHENBQ5B.js";
|
|
6
13
|
|
|
7
|
-
// src/commands/data-integration/
|
|
14
|
+
// src/commands/data-integration/inspect.ts
|
|
8
15
|
import { basename as basename3 } from "path";
|
|
9
16
|
|
|
10
|
-
// src/commands/data-integration/
|
|
17
|
+
// src/commands/data-integration/estimate.ts
|
|
18
|
+
var XLS_SIZE_WARN_BYTES = 100 * 1024 * 1024;
|
|
19
|
+
var LARGE_FILE_WARN_BYTES = 1024 * 1024 * 1024;
|
|
20
|
+
var XLS_HARD_LIMIT_BYTES = 1024 * 1024 * 1024;
|
|
21
|
+
var THROUGHPUT_BYTES_PER_SECOND = {
|
|
22
|
+
jsonl: 10 * 1024 * 1024,
|
|
23
|
+
json: 8 * 1024 * 1024,
|
|
24
|
+
csv: 4 * 1024 * 1024,
|
|
25
|
+
tsv: 4 * 1024 * 1024,
|
|
26
|
+
xlsx: 2 * 1024 * 1024
|
|
27
|
+
};
|
|
28
|
+
var SLOW_ENCODINGS = /* @__PURE__ */ new Set(["gbk", "gb2312", "gb18030", "big5", "shift_jis", "euc-jp", "euc-kr"]);
|
|
29
|
+
function estimateProcessingSeconds(format, sizeBytes, encoding) {
|
|
30
|
+
if (format === "xls") return 0;
|
|
31
|
+
let throughput = THROUGHPUT_BYTES_PER_SECOND[format];
|
|
32
|
+
if (encoding && SLOW_ENCODINGS.has(encoding.toLowerCase())) throughput /= 2;
|
|
33
|
+
return sizeBytes / throughput;
|
|
34
|
+
}
|
|
35
|
+
function formatDuration(seconds) {
|
|
36
|
+
if (seconds < 60) {
|
|
37
|
+
const low2 = Math.max(1, Math.round(seconds * 0.5));
|
|
38
|
+
const high2 = Math.max(low2 + 1, Math.round(seconds * 1.5));
|
|
39
|
+
return `roughly ${low2} to ${high2} seconds`;
|
|
40
|
+
}
|
|
41
|
+
const low = Math.max(1, Math.round(seconds / 60 * 0.5));
|
|
42
|
+
const high = Math.max(low + 1, Math.round(seconds / 60 * 1.5));
|
|
43
|
+
return `roughly ${low} to ${high} minutes`;
|
|
44
|
+
}
|
|
45
|
+
function humanSize(bytes) {
|
|
46
|
+
if (bytes >= 1024 * 1024 * 1024) return `${(bytes / (1024 * 1024 * 1024)).toFixed(1)} GB`;
|
|
47
|
+
return `${Math.round(bytes / (1024 * 1024))} MB`;
|
|
48
|
+
}
|
|
49
|
+
function xlsMemoryWarning(fileName, sizeBytes) {
|
|
50
|
+
return `Warning: ${fileName} is ${humanSize(sizeBytes)} (XLS). The legacy XLS parser loads the entire workbook into memory, roughly 5-10x the file size. Prefer converting to XLSX or splitting the workbook first.`;
|
|
51
|
+
}
|
|
52
|
+
function largeFileTimeWarning(fileName, sizeBytes, format, encoding) {
|
|
53
|
+
const duration = formatDuration(estimateProcessingSeconds(format, sizeBytes, encoding));
|
|
54
|
+
return `Warning: ${fileName} is ${humanSize(sizeBytes)}; processing is estimated to take ${duration}. Consider splitting the file if that is too long.`;
|
|
55
|
+
}
|
|
56
|
+
function largeFileMemoryWarning(format) {
|
|
57
|
+
if (format === "xlsx") {
|
|
58
|
+
return "Memory risk: XLSX processing keeps the workbook's shared string table in memory, so peak memory can substantially exceed the file size.";
|
|
59
|
+
}
|
|
60
|
+
if (format === "json") {
|
|
61
|
+
return "Memory risk: a JSON root object or a very large record may be materialized in memory during inspection, so peak memory can substantially exceed the file size.";
|
|
62
|
+
}
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
function assessFileSize(fileName, format, sizeBytes, encoding) {
|
|
66
|
+
const baseAssessment = {
|
|
67
|
+
size: humanSize(sizeBytes),
|
|
68
|
+
estimatedDuration: null,
|
|
69
|
+
warning: null,
|
|
70
|
+
reason: null,
|
|
71
|
+
memoryRisk: false,
|
|
72
|
+
rejected: false
|
|
73
|
+
};
|
|
74
|
+
if (format === "xls") {
|
|
75
|
+
if (sizeBytes > XLS_HARD_LIMIT_BYTES) {
|
|
76
|
+
return {
|
|
77
|
+
...baseAssessment,
|
|
78
|
+
reason: "XLS files larger than 1 GB are not supported; convert the workbook to XLSX or split it first.",
|
|
79
|
+
memoryRisk: true,
|
|
80
|
+
rejected: true
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
if (sizeBytes > XLS_SIZE_WARN_BYTES) {
|
|
84
|
+
return {
|
|
85
|
+
...baseAssessment,
|
|
86
|
+
warning: xlsMemoryWarning(fileName, sizeBytes),
|
|
87
|
+
memoryRisk: true
|
|
88
|
+
};
|
|
89
|
+
}
|
|
90
|
+
return baseAssessment;
|
|
91
|
+
}
|
|
92
|
+
if (sizeBytes > LARGE_FILE_WARN_BYTES) {
|
|
93
|
+
const memoryWarning = largeFileMemoryWarning(format);
|
|
94
|
+
const timeWarning = largeFileTimeWarning(fileName, sizeBytes, format, encoding);
|
|
95
|
+
return {
|
|
96
|
+
...baseAssessment,
|
|
97
|
+
estimatedDuration: formatDuration(estimateProcessingSeconds(format, sizeBytes, encoding)),
|
|
98
|
+
warning: memoryWarning ? `${timeWarning} ${memoryWarning}` : timeWarning,
|
|
99
|
+
memoryRisk: Boolean(memoryWarning)
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
return baseAssessment;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// src/commands/data-integration/input.ts
|
|
11
106
|
import { createHash } from "crypto";
|
|
12
107
|
import { createReadStream as createReadStream2, statSync } from "fs";
|
|
13
108
|
import { createRequire as createRequire2 } from "module";
|
|
@@ -24,7 +119,7 @@ import StreamValues from "stream-json/streamers/StreamValues.js";
|
|
|
24
119
|
import * as unzipper from "unzipper";
|
|
25
120
|
import { SaxesParser } from "saxes";
|
|
26
121
|
|
|
27
|
-
// src/commands/data-integration/
|
|
122
|
+
// src/commands/data-integration/encoding.ts
|
|
28
123
|
import { createReadStream } from "fs";
|
|
29
124
|
import { createRequire } from "module";
|
|
30
125
|
import { openSync, readSync, closeSync } from "fs";
|
|
@@ -68,7 +163,7 @@ function readSample(filePath, maxBytes) {
|
|
|
68
163
|
}
|
|
69
164
|
}
|
|
70
165
|
|
|
71
|
-
// src/commands/data-integration/
|
|
166
|
+
// src/commands/data-integration/time.ts
|
|
72
167
|
var TIME_FORMATS = [
|
|
73
168
|
// Standard AE format
|
|
74
169
|
"yyyy-MM-dd HH:mm:ss.SSS",
|
|
@@ -312,7 +407,7 @@ function tokenizeFormat(format) {
|
|
|
312
407
|
return tokens;
|
|
313
408
|
}
|
|
314
409
|
|
|
315
|
-
// src/commands/data-integration/
|
|
410
|
+
// src/commands/data-integration/flatten.ts
|
|
316
411
|
var NDJSON_MAX_DEPTH = 1;
|
|
317
412
|
var NESTED_NODE_SAMPLE_LIMIT = 5;
|
|
318
413
|
var NESTED_SAMPLE_TRUNCATE = 40;
|
|
@@ -345,7 +440,7 @@ function flattenJSON(obj, prefix = "", depth = 0, maxDepth = NDJSON_MAX_DEPTH) {
|
|
|
345
440
|
}
|
|
346
441
|
return result;
|
|
347
442
|
}
|
|
348
|
-
function buildRowWithFlatten(obj, flattenRules) {
|
|
443
|
+
function buildRowWithFlatten(obj, flattenRules, misses) {
|
|
349
444
|
const row = {};
|
|
350
445
|
const coveredRoots = new Set(Object.values(flattenRules).map((path) => path.split(".")[0]));
|
|
351
446
|
const base = flattenJSON(obj);
|
|
@@ -354,43 +449,77 @@ function buildRowWithFlatten(obj, flattenRules) {
|
|
|
354
449
|
}
|
|
355
450
|
for (const [outColumn, sourcePath] of Object.entries(flattenRules)) {
|
|
356
451
|
const value = getNestedValue(obj, sourcePath);
|
|
357
|
-
|
|
452
|
+
if (value == null) {
|
|
453
|
+
recordFlattenMiss(misses, outColumn);
|
|
454
|
+
row[outColumn] = "";
|
|
455
|
+
} else {
|
|
456
|
+
row[outColumn] = typeof value === "object" ? JSON.stringify(value) : String(value);
|
|
457
|
+
}
|
|
358
458
|
}
|
|
359
459
|
return row;
|
|
360
460
|
}
|
|
361
|
-
function flattenLocalDataRow(value, flattenRules) {
|
|
461
|
+
function flattenLocalDataRow(value, flattenRules, misses) {
|
|
362
462
|
if (value !== null && typeof value === "object" && !Array.isArray(value)) {
|
|
363
463
|
if (flattenRules && Object.keys(flattenRules).length > 0) {
|
|
364
|
-
return buildRowWithFlatten(value, flattenRules);
|
|
464
|
+
return buildRowWithFlatten(value, flattenRules, misses);
|
|
365
465
|
}
|
|
366
466
|
return value;
|
|
367
467
|
}
|
|
368
468
|
return { value };
|
|
369
469
|
}
|
|
370
|
-
function flattenDelimitedRow(row, flattenRules) {
|
|
470
|
+
function flattenDelimitedRow(row, flattenRules, misses) {
|
|
371
471
|
const result = { ...row };
|
|
372
472
|
for (const [outColumn, path] of Object.entries(flattenRules)) {
|
|
373
473
|
const dot = path.indexOf(".");
|
|
374
|
-
if (dot <= 0)
|
|
474
|
+
if (dot <= 0) {
|
|
475
|
+
recordFlattenMiss(misses, outColumn);
|
|
476
|
+
continue;
|
|
477
|
+
}
|
|
375
478
|
const column = path.slice(0, dot);
|
|
376
479
|
const cell = row[column];
|
|
377
|
-
if (typeof cell !== "string")
|
|
480
|
+
if (typeof cell !== "string") {
|
|
481
|
+
recordFlattenMiss(misses, outColumn);
|
|
482
|
+
continue;
|
|
483
|
+
}
|
|
378
484
|
let parsed;
|
|
379
485
|
try {
|
|
380
486
|
parsed = JSON.parse(cell);
|
|
381
487
|
} catch {
|
|
488
|
+
recordFlattenMiss(misses, outColumn);
|
|
382
489
|
continue;
|
|
383
490
|
}
|
|
384
491
|
const value = getNestedValue(parsed, path.slice(dot + 1));
|
|
385
|
-
if (value === null || value === void 0)
|
|
492
|
+
if (value === null || value === void 0) {
|
|
493
|
+
recordFlattenMiss(misses, outColumn);
|
|
494
|
+
continue;
|
|
495
|
+
}
|
|
386
496
|
result[outColumn] = typeof value === "object" ? JSON.stringify(value) : String(value);
|
|
387
497
|
}
|
|
388
498
|
return result;
|
|
389
499
|
}
|
|
500
|
+
function recordFlattenMiss(misses, outColumn) {
|
|
501
|
+
if (!misses) return;
|
|
502
|
+
misses[outColumn] = (misses[outColumn] ?? 0) + 1;
|
|
503
|
+
}
|
|
390
504
|
function buildNestedTree(rows) {
|
|
391
505
|
const objects = rows.filter((row) => row !== null && typeof row === "object" && !Array.isArray(row));
|
|
392
506
|
return buildObjectChildren(objects, "");
|
|
393
507
|
}
|
|
508
|
+
function buildColumnNestedTree(cells) {
|
|
509
|
+
const objects = cells.filter((cell) => cell !== null && typeof cell === "object" && !Array.isArray(cell));
|
|
510
|
+
const arrays = cells.filter(Array.isArray);
|
|
511
|
+
if (objects.length >= arrays.length) {
|
|
512
|
+
return objects.length > 0 ? buildObjectChildren(objects, "") : [];
|
|
513
|
+
}
|
|
514
|
+
const elements = arrays.flat();
|
|
515
|
+
const elementObjects = elements.filter(
|
|
516
|
+
(value) => value !== null && typeof value === "object" && !Array.isArray(value)
|
|
517
|
+
);
|
|
518
|
+
const elementKind = elementObjects.length > 0 ? "object" : "primitive";
|
|
519
|
+
const node = { path: "", name: "", kind: "array", elementKind, nonEmpty: elements.length > 0 };
|
|
520
|
+
if (elementKind === "object") node.children = buildObjectChildren(elementObjects, "");
|
|
521
|
+
return [node];
|
|
522
|
+
}
|
|
394
523
|
function buildObjectChildren(objects, parentPath) {
|
|
395
524
|
const keys = [];
|
|
396
525
|
const seen = /* @__PURE__ */ new Set();
|
|
@@ -416,8 +545,18 @@ function buildNode(path, name, values, nonEmpty) {
|
|
|
416
545
|
if (structural > 0 && structural >= values.length - structural) {
|
|
417
546
|
if (arrays.length > objects.length) {
|
|
418
547
|
const elementValues = arrays.flat();
|
|
419
|
-
const
|
|
420
|
-
|
|
548
|
+
const elementObjects = elementValues.filter(
|
|
549
|
+
(value) => value !== null && typeof value === "object" && !Array.isArray(value)
|
|
550
|
+
);
|
|
551
|
+
const elementKind = elementObjects.length > 0 ? "object" : "primitive";
|
|
552
|
+
return {
|
|
553
|
+
path,
|
|
554
|
+
name,
|
|
555
|
+
kind: "array",
|
|
556
|
+
elementKind,
|
|
557
|
+
nonEmpty,
|
|
558
|
+
...elementKind === "object" ? { children: buildObjectChildren(elementObjects, path) } : {}
|
|
559
|
+
};
|
|
421
560
|
}
|
|
422
561
|
return { path, name, kind: "object", children: buildObjectChildren(objects, path), nonEmpty };
|
|
423
562
|
}
|
|
@@ -485,13 +624,11 @@ function isStrongDateTime(value) {
|
|
|
485
624
|
return /^\d{4}[-/]\d{1,2}[-/]\d{1,2}(?:[ T]\d{1,2}:\d{2}(?::\d{2}(?:\.\d{1,3})?)?(?:Z|[+-]\d{2}:?\d{2})?)?$/.test(value.trim());
|
|
486
625
|
}
|
|
487
626
|
|
|
488
|
-
// src/commands/data-integration/
|
|
489
|
-
var MAX_STREAMING_FILE_BYTES = 200 * 1024 * 1024;
|
|
490
|
-
var MAX_XLS_FILE_BYTES = 50 * 1024 * 1024;
|
|
627
|
+
// src/commands/data-integration/input.ts
|
|
491
628
|
var XLSX = XLSXMod.default ?? XLSXMod;
|
|
492
629
|
var require3 = createRequire2(import.meta.url);
|
|
493
630
|
var ExcelWorksheetReader = require3("exceljs/lib/stream/xlsx/worksheet-reader");
|
|
494
|
-
|
|
631
|
+
function resolveLocalDataInputMeta(filePath) {
|
|
495
632
|
let format = resolveFormat(filePath);
|
|
496
633
|
let delimiter;
|
|
497
634
|
let encoding;
|
|
@@ -512,33 +649,43 @@ async function inspectLocalDataInput(filePath) {
|
|
|
512
649
|
location: { field: "input-file" }
|
|
513
650
|
});
|
|
514
651
|
}
|
|
515
|
-
const maxBytes = format === "xls" ? MAX_XLS_FILE_BYTES : MAX_STREAMING_FILE_BYTES;
|
|
516
|
-
if (sizeBytes > maxBytes) {
|
|
517
|
-
throw new CliValidationError(
|
|
518
|
-
`${format.toUpperCase()} input exceeds the supported file size limit.`,
|
|
519
|
-
{
|
|
520
|
-
code: "LOCAL_DATA_FILE_TOO_LARGE",
|
|
521
|
-
hint: `Split the file below ${format === "xls" ? "50" : "200"} MB and retry.`,
|
|
522
|
-
location: { field: "input-file" }
|
|
523
|
-
}
|
|
524
|
-
);
|
|
525
|
-
}
|
|
526
652
|
if (format !== "xls" && format !== "xlsx" && !encoding) {
|
|
527
653
|
encoding = detectEncoding(filePath);
|
|
528
654
|
}
|
|
655
|
+
return {
|
|
656
|
+
filePath,
|
|
657
|
+
format,
|
|
658
|
+
sizeBytes,
|
|
659
|
+
...delimiter ? { delimiter } : {},
|
|
660
|
+
...encoding ? { encoding } : {}
|
|
661
|
+
};
|
|
662
|
+
}
|
|
663
|
+
async function inspectLocalDataInput(filePath) {
|
|
664
|
+
const meta = resolveLocalDataInputMeta(filePath);
|
|
665
|
+
emitSizeWarning(meta.filePath, meta.format, meta.sizeBytes, meta.encoding);
|
|
529
666
|
try {
|
|
530
667
|
return {
|
|
531
|
-
|
|
532
|
-
format,
|
|
533
|
-
sizeBytes,
|
|
668
|
+
...meta,
|
|
534
669
|
sha256: await sha256File(filePath),
|
|
535
|
-
dataSets: await discoverDataSets(filePath, format)
|
|
536
|
-
...delimiter ? { delimiter } : {},
|
|
537
|
-
...encoding ? { encoding } : {}
|
|
670
|
+
dataSets: await discoverDataSets(filePath, meta.format)
|
|
538
671
|
};
|
|
539
672
|
} catch (error) {
|
|
540
673
|
if (error instanceof CliValidationError) throw error;
|
|
541
|
-
throw localDataParseError(format);
|
|
674
|
+
throw localDataParseError(meta.format);
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
function emitSizeWarning(filePath, format, sizeBytes, encoding) {
|
|
678
|
+
const assessment = assessFileSize(basename(filePath), format, sizeBytes, encoding);
|
|
679
|
+
if (assessment.rejected) {
|
|
680
|
+
throw new CliValidationError("XLS input exceeds the supported file size limit.", {
|
|
681
|
+
code: "LOCAL_DATA_FILE_TOO_LARGE",
|
|
682
|
+
hint: "XLS files larger than 1 GB are not supported; convert the workbook to XLSX or split it first.",
|
|
683
|
+
location: { field: "input-file" }
|
|
684
|
+
});
|
|
685
|
+
}
|
|
686
|
+
if (assessment.warning) {
|
|
687
|
+
process.stderr.write(`${assessment.warning}
|
|
688
|
+
`);
|
|
542
689
|
}
|
|
543
690
|
}
|
|
544
691
|
function selectDataSet(input, requested) {
|
|
@@ -562,6 +709,24 @@ function selectDataSet(input, requested) {
|
|
|
562
709
|
}
|
|
563
710
|
return input.dataSets[0];
|
|
564
711
|
}
|
|
712
|
+
var LocalDataRowCallbackError = class extends Error {
|
|
713
|
+
constructor(cause) {
|
|
714
|
+
super(cause instanceof Error ? cause.message : String(cause));
|
|
715
|
+
this.cause = cause;
|
|
716
|
+
this.name = "LocalDataRowCallbackError";
|
|
717
|
+
}
|
|
718
|
+
cause;
|
|
719
|
+
};
|
|
720
|
+
function wrapRowCallback(onRow) {
|
|
721
|
+
return async (row, rowNumber) => {
|
|
722
|
+
try {
|
|
723
|
+
await onRow(row, rowNumber);
|
|
724
|
+
} catch (error) {
|
|
725
|
+
if (error instanceof CliValidationError) throw error;
|
|
726
|
+
throw new LocalDataRowCallbackError(error);
|
|
727
|
+
}
|
|
728
|
+
};
|
|
729
|
+
}
|
|
565
730
|
async function streamLocalDataRows(input, dataSet, onRow, options = {}) {
|
|
566
731
|
const opts = {
|
|
567
732
|
delimiter: options.delimiter ?? input.delimiter ?? (input.format === "tsv" ? " " : ","),
|
|
@@ -569,24 +734,30 @@ async function streamLocalDataRows(input, dataSet, onRow, options = {}) {
|
|
|
569
734
|
headerNames: options.headerNames,
|
|
570
735
|
noHeader: options.noHeader,
|
|
571
736
|
flattenRules: options.flattenRules,
|
|
572
|
-
|
|
737
|
+
flattenMisses: options.flattenMisses,
|
|
738
|
+
mergeSheets: options.mergeSheets,
|
|
739
|
+
warnRagged: options.warnRagged
|
|
573
740
|
};
|
|
741
|
+
const wrappedRow = wrapRowCallback(onRow);
|
|
574
742
|
try {
|
|
575
743
|
switch (input.format) {
|
|
576
744
|
case "csv":
|
|
577
745
|
case "tsv":
|
|
578
|
-
return await streamDelimited(input.filePath,
|
|
746
|
+
return await streamDelimited(input.filePath, wrappedRow, opts);
|
|
579
747
|
case "jsonl":
|
|
580
|
-
return await streamJsonLines(input.filePath,
|
|
748
|
+
return await streamJsonLines(input.filePath, wrappedRow, opts);
|
|
581
749
|
case "json":
|
|
582
|
-
return await streamJson(input.filePath, dataSet.selector ?? "$",
|
|
750
|
+
return await streamJson(input.filePath, dataSet.selector ?? "$", wrappedRow, opts);
|
|
583
751
|
case "xlsx":
|
|
584
|
-
return await streamXlsx(input.filePath, dataSet.label,
|
|
752
|
+
return await streamXlsx(input.filePath, dataSet.label, wrappedRow, opts);
|
|
585
753
|
case "xls":
|
|
586
|
-
return await streamXls(input.filePath, dataSet.label,
|
|
754
|
+
return await streamXls(input.filePath, dataSet.label, wrappedRow, opts);
|
|
587
755
|
}
|
|
588
756
|
} catch (error) {
|
|
589
757
|
if (error instanceof CliValidationError) throw error;
|
|
758
|
+
if (error instanceof LocalDataRowCallbackError) {
|
|
759
|
+
throw error.cause instanceof Error ? error.cause : error;
|
|
760
|
+
}
|
|
590
761
|
throw localDataParseError(input.format);
|
|
591
762
|
}
|
|
592
763
|
}
|
|
@@ -774,29 +945,36 @@ async function streamDelimited(filePath, onRow, options) {
|
|
|
774
945
|
quote: delimiter === " " ? null : '"',
|
|
775
946
|
relax_column_count: true,
|
|
776
947
|
skip_empty_lines: true,
|
|
777
|
-
trim: true
|
|
778
|
-
columns: headerNames ? headerNames : (headers) => dedupeHeaders(headers.map((header) => String(header).trim()))
|
|
948
|
+
trim: true
|
|
779
949
|
});
|
|
780
950
|
decodeTextStream(filePath, encoding).pipe(parser);
|
|
781
951
|
let count = 0;
|
|
782
|
-
|
|
952
|
+
let widthMismatches = 0;
|
|
953
|
+
let resolvedHeaders = headerNames;
|
|
954
|
+
for await (const raw of parser) {
|
|
955
|
+
const values = raw;
|
|
956
|
+
if (!resolvedHeaders) {
|
|
957
|
+
resolvedHeaders = dedupeHeaders(values.map((header) => String(header).trim()));
|
|
958
|
+
continue;
|
|
959
|
+
}
|
|
960
|
+
if (values.length !== resolvedHeaders.length) widthMismatches += 1;
|
|
783
961
|
count += 1;
|
|
784
|
-
const row =
|
|
962
|
+
const row = {};
|
|
963
|
+
for (let index = 0; index < resolvedHeaders.length; index += 1) {
|
|
964
|
+
row[resolvedHeaders[index]] = values[index] ?? null;
|
|
965
|
+
}
|
|
785
966
|
await onRow(
|
|
786
|
-
options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
|
|
967
|
+
options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules, options.flattenMisses) : row,
|
|
787
968
|
count
|
|
788
969
|
);
|
|
789
970
|
}
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
971
|
+
if (widthMismatches > 0 && options.warnRagged !== false) {
|
|
972
|
+
process.stderr.write(
|
|
973
|
+
`Warning: ${widthMismatches} record(s) had a column count different from the header row; extra fields were dropped and missing fields were treated as empty.
|
|
974
|
+
`
|
|
975
|
+
);
|
|
795
976
|
}
|
|
796
|
-
|
|
797
|
-
const row = {};
|
|
798
|
-
for (const header of headerNames) row[header] = source[header] ?? null;
|
|
799
|
-
return row;
|
|
977
|
+
return count;
|
|
800
978
|
}
|
|
801
979
|
async function streamJsonLines(filePath, onRow, options) {
|
|
802
980
|
let count = 0;
|
|
@@ -813,7 +991,7 @@ async function streamJsonLines(filePath, onRow, options) {
|
|
|
813
991
|
location: { record: count }
|
|
814
992
|
});
|
|
815
993
|
}
|
|
816
|
-
await onRow(flattenLocalDataRow(value, options.flattenRules), count);
|
|
994
|
+
await onRow(flattenLocalDataRow(value, options.flattenRules, options.flattenMisses), count);
|
|
817
995
|
}
|
|
818
996
|
return count;
|
|
819
997
|
}
|
|
@@ -825,7 +1003,7 @@ async function streamJson(filePath, selector, onRow, options) {
|
|
|
825
1003
|
const chain = selector === "$" || selector === "$object" ? source.pipe(parser).pipe(streamer) : source.pipe(parser).pipe(Pick.pick({ filter: selector })).pipe(streamer);
|
|
826
1004
|
for await (const item of chain) {
|
|
827
1005
|
count += 1;
|
|
828
|
-
await onRow(flattenLocalDataRow(item.value, options.flattenRules), count);
|
|
1006
|
+
await onRow(flattenLocalDataRow(item.value, options.flattenRules, options.flattenMisses), count);
|
|
829
1007
|
}
|
|
830
1008
|
return count;
|
|
831
1009
|
}
|
|
@@ -869,7 +1047,11 @@ async function streamXlsx(filePath, sheetName, onRow, options) {
|
|
|
869
1047
|
}
|
|
870
1048
|
if (values.every(isMissing)) continue;
|
|
871
1049
|
count += 1;
|
|
872
|
-
|
|
1050
|
+
const row = Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null]));
|
|
1051
|
+
await onRow(
|
|
1052
|
+
options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
|
|
1053
|
+
count
|
|
1054
|
+
);
|
|
873
1055
|
}
|
|
874
1056
|
}
|
|
875
1057
|
return count;
|
|
@@ -1021,17 +1203,15 @@ async function streamXls(filePath, sheetName, onRow, options) {
|
|
|
1021
1203
|
for (const values of rows.slice(start)) {
|
|
1022
1204
|
if (values.every(isMissing)) continue;
|
|
1023
1205
|
count += 1;
|
|
1024
|
-
|
|
1206
|
+
const row = Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null]));
|
|
1207
|
+
await onRow(
|
|
1208
|
+
options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
|
|
1209
|
+
count
|
|
1210
|
+
);
|
|
1025
1211
|
}
|
|
1026
1212
|
}
|
|
1027
1213
|
return count;
|
|
1028
1214
|
}
|
|
1029
|
-
function normalizeRow(value) {
|
|
1030
|
-
if (value !== null && typeof value === "object" && !Array.isArray(value)) {
|
|
1031
|
-
return value;
|
|
1032
|
-
}
|
|
1033
|
-
return { value };
|
|
1034
|
-
}
|
|
1035
1215
|
function normalizeExcelValue(value) {
|
|
1036
1216
|
if (value instanceof Date) return value;
|
|
1037
1217
|
if (value && typeof value === "object") {
|
|
@@ -1074,9 +1254,16 @@ async function firstNonWhitespaceCharacter(filePath, encoding) {
|
|
|
1074
1254
|
return void 0;
|
|
1075
1255
|
}
|
|
1076
1256
|
|
|
1077
|
-
// src/commands/data-integration/
|
|
1257
|
+
// src/commands/data-integration/mapping.ts
|
|
1078
1258
|
import { readFileSync } from "fs";
|
|
1259
|
+
|
|
1260
|
+
// src/commands/data-integration/types.ts
|
|
1261
|
+
var MAPPING_VERSION = "ae-data-integration-mapping/v1";
|
|
1262
|
+
|
|
1263
|
+
// src/commands/data-integration/mapping.ts
|
|
1079
1264
|
var VALID_PROPERTY_NAME = /^[a-z][a-z0-9_]{0,49}$/;
|
|
1265
|
+
var VALID_CHILD_PROPERTY_NAME = /^[a-z][a-z0-9_]{0,49}(\.[a-z][a-z0-9_]{0,49})+$/;
|
|
1266
|
+
var FLATTEN_SUPPORTED_FORMATS = ["csv", "tsv", "json", "jsonl", "xls", "xlsx"];
|
|
1080
1267
|
function readLocalDataMapping(raw, options) {
|
|
1081
1268
|
const trimmed = raw.trim();
|
|
1082
1269
|
let text;
|
|
@@ -1106,8 +1293,8 @@ function readLocalDataMapping(raw, options) {
|
|
|
1106
1293
|
return value;
|
|
1107
1294
|
}
|
|
1108
1295
|
function validateMapping(value, options) {
|
|
1109
|
-
if (!isRecord(value) || value.version !==
|
|
1110
|
-
throw mappingError(
|
|
1296
|
+
if (!isRecord(value) || value.version !== MAPPING_VERSION) {
|
|
1297
|
+
throw mappingError(`Mapping version must be ${MAPPING_VERSION}.`);
|
|
1111
1298
|
}
|
|
1112
1299
|
const sha256Valid = typeof value.source?.sha256 === "string" && (options?.sourceWildcard ? value.source.sha256 === "*" || /^[a-f0-9]{64}$/i.test(value.source.sha256) : /^[a-f0-9]{64}$/i.test(value.source.sha256));
|
|
1113
1300
|
if (!isRecord(value.source) || !sha256Valid || !["csv", "tsv", "json", "jsonl", "xls", "xlsx"].includes(String(value.source.format)) || typeof value.source.data_set !== "string" || !value.source.data_set) {
|
|
@@ -1137,12 +1324,28 @@ function validateMapping(value, options) {
|
|
|
1137
1324
|
}
|
|
1138
1325
|
if (!Array.isArray(value.properties)) throw mappingError("Mapping properties must be an array.");
|
|
1139
1326
|
const targets = /* @__PURE__ */ new Set();
|
|
1327
|
+
const propertyTypes = /* @__PURE__ */ new Map();
|
|
1140
1328
|
for (const property of value.properties) {
|
|
1141
|
-
if (!isRecord(property) || typeof property.source !== "string" || typeof property.target !== "string" || !VALID_PROPERTY_NAME.test(property.target) || !["number", "string", "boolean", "datetime", "list", "object"].includes(String(property.type)) || property.transform !== void 0 && !["stringify", "number", "boolean", "json"].includes(String(property.transform)) || property.value_mapping !== void 0 && !isStringMap(property.value_mapping) || property.time_format !== void 0 && (typeof property.time_format !== "string" || !property.time_format.trim() || property.time_format.length > 64) || property.desc !== void 0 && (typeof property.desc !== "string" || !property.desc.trim() || property.desc.length > 200)) {
|
|
1329
|
+
if (!isRecord(property) || typeof property.source !== "string" || typeof property.target !== "string" || !(VALID_PROPERTY_NAME.test(property.target) || VALID_CHILD_PROPERTY_NAME.test(property.target)) || !["number", "string", "boolean", "datetime", "list", "object", "array_row"].includes(String(property.type)) || property.transform !== void 0 && !["stringify", "number", "boolean", "json"].includes(String(property.transform)) || property.value_mapping !== void 0 && !isStringMap(property.value_mapping) || property.time_format !== void 0 && (typeof property.time_format !== "string" || !property.time_format.trim() || property.time_format.length > 64) || property.desc !== void 0 && (typeof property.desc !== "string" || !property.desc.trim() || property.desc.length > 200)) {
|
|
1142
1330
|
throw mappingError("Every property mapping needs a source, legal AE target name, and supported type.");
|
|
1143
1331
|
}
|
|
1144
1332
|
if (targets.has(property.target)) throw mappingError("Property target names must be unique.");
|
|
1145
1333
|
targets.add(property.target);
|
|
1334
|
+
propertyTypes.set(property.target, { type: String(property.type), source: property.source });
|
|
1335
|
+
}
|
|
1336
|
+
for (const property of value.properties) {
|
|
1337
|
+
if (!property.target.includes(".")) continue;
|
|
1338
|
+
const parentName = property.target.split(".")[0];
|
|
1339
|
+
const parent = propertyTypes.get(parentName);
|
|
1340
|
+
if (!parent || parent.type !== "object" && parent.type !== "array_row") {
|
|
1341
|
+
throw mappingError(`The sub-property "${property.target}" references "${parentName}", which is not an object/array_row property in this mapping.`);
|
|
1342
|
+
}
|
|
1343
|
+
if (property.type === "object" || property.type === "array_row") {
|
|
1344
|
+
throw mappingError(`The sub-property "${property.target}" must be scalar or list, not ${property.type}.`);
|
|
1345
|
+
}
|
|
1346
|
+
if (property.source !== parent.source) {
|
|
1347
|
+
throw mappingError(`The sub-property "${property.target}" must read from the same source column as its parent "${parentName}".`);
|
|
1348
|
+
}
|
|
1146
1349
|
}
|
|
1147
1350
|
if (value.time_format !== void 0 && (typeof value.time_format !== "string" || !value.time_format.trim() || value.time_format.length > 64)) {
|
|
1148
1351
|
throw mappingError("time_format must be a non-empty string of at most 64 characters.");
|
|
@@ -1184,6 +1387,9 @@ function validateMapping(value, options) {
|
|
|
1184
1387
|
}
|
|
1185
1388
|
if (value.flatten_rules !== void 0) {
|
|
1186
1389
|
if (!isRecord(value.flatten_rules)) throw mappingError("flatten_rules must be an object of { column: dot.path }.");
|
|
1390
|
+
if (!FLATTEN_SUPPORTED_FORMATS.includes(String(value.source.format))) {
|
|
1391
|
+
throw mappingError(`flatten_rules are not supported for ${String(value.source.format)} input.`);
|
|
1392
|
+
}
|
|
1187
1393
|
for (const [column, path] of Object.entries(value.flatten_rules)) {
|
|
1188
1394
|
if (!VALID_PROPERTY_NAME.test(column) || typeof path !== "string" || !path.trim()) {
|
|
1189
1395
|
throw mappingError("flatten_rules keys must be legal AE property names and values must be non-empty dot paths.");
|
|
@@ -1227,6 +1433,31 @@ function validateMapping(value, options) {
|
|
|
1227
1433
|
function isValidAeName(value) {
|
|
1228
1434
|
return VALID_PROPERTY_NAME.test(value);
|
|
1229
1435
|
}
|
|
1436
|
+
function sourceColumns(mapping) {
|
|
1437
|
+
const columns = /* @__PURE__ */ new Set();
|
|
1438
|
+
const flattenOut = new Set(Object.keys(mapping.flatten_rules ?? {}));
|
|
1439
|
+
for (const property of mapping.properties) {
|
|
1440
|
+
if (!flattenOut.has(property.source)) columns.add(property.source);
|
|
1441
|
+
}
|
|
1442
|
+
for (const path of Object.values(mapping.flatten_rules ?? {})) {
|
|
1443
|
+
const root = path.split(".")[0];
|
|
1444
|
+
if (root) columns.add(root);
|
|
1445
|
+
}
|
|
1446
|
+
const systemFields = [
|
|
1447
|
+
mapping.account_id_field,
|
|
1448
|
+
mapping.distinct_id_field,
|
|
1449
|
+
mapping.record_type_field,
|
|
1450
|
+
mapping.event_name_field,
|
|
1451
|
+
mapping.time.field,
|
|
1452
|
+
mapping.ip_field,
|
|
1453
|
+
mapping.uuid_field,
|
|
1454
|
+
mapping.zone_offset_field
|
|
1455
|
+
];
|
|
1456
|
+
for (const field of systemFields) if (field) columns.add(field);
|
|
1457
|
+
for (const column of mapping.exclude_columns ?? []) columns.add(column);
|
|
1458
|
+
for (const header of mapping.headers ?? []) columns.add(header);
|
|
1459
|
+
return [...columns].sort();
|
|
1460
|
+
}
|
|
1230
1461
|
function mappingError(message) {
|
|
1231
1462
|
return new CliValidationError(message, {
|
|
1232
1463
|
code: "LOCAL_DATA_MAPPING_INVALID",
|
|
@@ -1253,7 +1484,7 @@ function isRecord(value) {
|
|
|
1253
1484
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
1254
1485
|
}
|
|
1255
1486
|
|
|
1256
|
-
// src/commands/data-integration/
|
|
1487
|
+
// src/commands/data-integration/multi.ts
|
|
1257
1488
|
var PROPERTY_TYPES = /* @__PURE__ */ new Set([
|
|
1258
1489
|
"number",
|
|
1259
1490
|
"string",
|
|
@@ -1398,9 +1629,46 @@ function isRecord2(value) {
|
|
|
1398
1629
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
1399
1630
|
}
|
|
1400
1631
|
|
|
1401
|
-
// src/commands/data-integration/
|
|
1632
|
+
// src/commands/data-integration/profile.ts
|
|
1402
1633
|
import { createHash as createHash2, randomInt } from "crypto";
|
|
1403
1634
|
import { basename as basename2, extname as extname2 } from "path";
|
|
1635
|
+
|
|
1636
|
+
// src/commands/data-integration/field-spec.ts
|
|
1637
|
+
import { isIP } from "net";
|
|
1638
|
+
function stripQuotes(value) {
|
|
1639
|
+
if (typeof value !== "string") return value;
|
|
1640
|
+
const text = value;
|
|
1641
|
+
if (text.length >= 2 && (text.startsWith('"') && text.endsWith('"') || text.startsWith("'") && text.endsWith("'"))) {
|
|
1642
|
+
return text.slice(1, -1).trim();
|
|
1643
|
+
}
|
|
1644
|
+
return text.trim();
|
|
1645
|
+
}
|
|
1646
|
+
var UUID_PATTERN = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
|
|
1647
|
+
function isValidUuid(value) {
|
|
1648
|
+
if (typeof value !== "string") return false;
|
|
1649
|
+
return UUID_PATTERN.test(stripQuotes(value));
|
|
1650
|
+
}
|
|
1651
|
+
function isValidIp(value) {
|
|
1652
|
+
if (typeof value !== "string") return false;
|
|
1653
|
+
return isIP(stripQuotes(value)) !== 0;
|
|
1654
|
+
}
|
|
1655
|
+
function isPrivateIp(value) {
|
|
1656
|
+
if (typeof value !== "string") return false;
|
|
1657
|
+
const text = stripQuotes(value);
|
|
1658
|
+
const version = isIP(text);
|
|
1659
|
+
if (version === 6) {
|
|
1660
|
+
const lower = text.toLowerCase();
|
|
1661
|
+
if (lower === "::1") return true;
|
|
1662
|
+
const first2 = lower.split(":")[0];
|
|
1663
|
+
return first2.startsWith("fc") || first2.startsWith("fd") || /^fe[89ab]/.test(lower);
|
|
1664
|
+
}
|
|
1665
|
+
if (version !== 4) return false;
|
|
1666
|
+
const octets = text.split(".").map((part) => Number(part));
|
|
1667
|
+
const [first, second] = octets;
|
|
1668
|
+
return first === 10 || first === 172 && second >= 16 && second <= 31 || first === 192 && second === 168 || first === 127 || first === 169 && second === 254;
|
|
1669
|
+
}
|
|
1670
|
+
|
|
1671
|
+
// src/commands/data-integration/profile.ts
|
|
1404
1672
|
var UNIQUE_SAMPLE_LIMIT = 1e4;
|
|
1405
1673
|
var GLOBAL_UNIQUE_SAMPLE_LIMIT = 2e5;
|
|
1406
1674
|
var IDENTITY_MAX_LENGTH = 128;
|
|
@@ -1424,6 +1692,8 @@ var DISTINCT_NAMES = ["#distinct_id", "distinct_id", "distinctid", "device_id",
|
|
|
1424
1692
|
var TIME_NAMES = ["#time", "time", "timestamp", "event_time", "created_at", "occurred_at", "datetime", "date", "\u65F6\u95F4", "\u4E8B\u4EF6\u65F6\u95F4", "\u53D1\u751F\u65F6\u95F4", "\u521B\u5EFA\u65F6\u95F4", "\u4E0B\u5355\u65F6\u95F4", "\u8BA2\u5355\u65F6\u95F4"];
|
|
1425
1693
|
var EVENT_NAMES = ["#event_name", "event_name", "event", "action", "activity", "\u4E8B\u4EF6\u540D", "\u4E8B\u4EF6\u540D\u79F0", "\u4E8B\u4EF6", "\u884C\u4E3A", "\u52A8\u4F5C"];
|
|
1426
1694
|
var TYPE_NAMES = ["#type", "record_type", "data_type", "\u64CD\u4F5C\u7C7B\u578B"];
|
|
1695
|
+
var IP_NAMES = ["#ip", "ip", "ip_address", "ipaddress", "client_ip", "clientip", "remote_addr", "remoteaddr", "ip\u5730\u5740", "\u5BA2\u6237\u7AEFip"];
|
|
1696
|
+
var UUID_NAMES = ["#uuid", "uuid", "event_uuid", "eventuuid", "request_uuid", "requestuuid", "\u552F\u4E00\u6807\u8BC6", "\u552F\u4E00id"];
|
|
1427
1697
|
async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai", options = {}) {
|
|
1428
1698
|
const columns = /* @__PURE__ */ new Map();
|
|
1429
1699
|
const recognizedRecordTypes = /* @__PURE__ */ new Set();
|
|
@@ -1433,7 +1703,7 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1433
1703
|
options.collectNestedTree && (input.format === "json" || input.format === "jsonl")
|
|
1434
1704
|
);
|
|
1435
1705
|
const collectDelimitedTree = Boolean(
|
|
1436
|
-
options.collectNestedTree && (input.format === "csv" || input.format === "tsv")
|
|
1706
|
+
options.collectNestedTree && (input.format === "csv" || input.format === "tsv" || input.format === "xlsx" || input.format === "xls")
|
|
1437
1707
|
);
|
|
1438
1708
|
let nestedObjects = [];
|
|
1439
1709
|
let nestedSeen = 0;
|
|
@@ -1464,6 +1734,9 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1464
1734
|
uniqueOverflow: false,
|
|
1465
1735
|
timeParseCount: 0,
|
|
1466
1736
|
timeFormatCounts: /* @__PURE__ */ new Map(),
|
|
1737
|
+
uuidValidCount: 0,
|
|
1738
|
+
ipValidCount: 0,
|
|
1739
|
+
lanIpCount: 0,
|
|
1467
1740
|
samples: [],
|
|
1468
1741
|
sampleSet: /* @__PURE__ */ new Set()
|
|
1469
1742
|
};
|
|
@@ -1486,24 +1759,27 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1486
1759
|
accumulator.uniqueOverflow = true;
|
|
1487
1760
|
}
|
|
1488
1761
|
if (options.collectSamples) recordSample(accumulator, value);
|
|
1489
|
-
if (collectDelimitedTree && typeof value === "string"
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
if (
|
|
1495
|
-
sampler =
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
sampler.
|
|
1501
|
-
|
|
1502
|
-
|
|
1503
|
-
|
|
1762
|
+
if (collectDelimitedTree && typeof value === "string") {
|
|
1763
|
+
const trimmed = value.trim();
|
|
1764
|
+
if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
|
|
1765
|
+
try {
|
|
1766
|
+
const parsed = JSON.parse(value);
|
|
1767
|
+
if (parsed !== null && typeof parsed === "object") {
|
|
1768
|
+
let sampler = delimitedNested.get(name);
|
|
1769
|
+
if (!sampler) {
|
|
1770
|
+
sampler = { seen: 0, values: [] };
|
|
1771
|
+
delimitedNested.set(name, sampler);
|
|
1772
|
+
}
|
|
1773
|
+
sampler.seen += 1;
|
|
1774
|
+
if (sampler.values.length < NESTED_TREE_SAMPLE_LIMIT) {
|
|
1775
|
+
sampler.values.push(parsed);
|
|
1776
|
+
} else {
|
|
1777
|
+
const slot = randomInt(sampler.seen);
|
|
1778
|
+
if (slot < NESTED_TREE_SAMPLE_LIMIT) sampler.values[slot] = parsed;
|
|
1779
|
+
}
|
|
1504
1780
|
}
|
|
1781
|
+
} catch {
|
|
1505
1782
|
}
|
|
1506
|
-
} catch {
|
|
1507
1783
|
}
|
|
1508
1784
|
}
|
|
1509
1785
|
if (isParseableTime(value, matchesName(name, TIME_NAMES))) {
|
|
@@ -1515,6 +1791,11 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1515
1791
|
}
|
|
1516
1792
|
}
|
|
1517
1793
|
}
|
|
1794
|
+
if (matchesName(name, UUID_NAMES) && isValidUuid(value)) accumulator.uuidValidCount += 1;
|
|
1795
|
+
if (matchesName(name, IP_NAMES) && isValidIp(value)) {
|
|
1796
|
+
accumulator.ipValidCount += 1;
|
|
1797
|
+
if (isPrivateIp(value)) accumulator.lanIpCount += 1;
|
|
1798
|
+
}
|
|
1518
1799
|
if (matchesName(name, TYPE_NAMES)) {
|
|
1519
1800
|
const normalized = normalizeRecordType(value);
|
|
1520
1801
|
if (normalized) recognizedRecordTypes.add(normalized);
|
|
@@ -1527,23 +1808,25 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1527
1808
|
headerNames: options.headerNames,
|
|
1528
1809
|
noHeader: options.noHeader,
|
|
1529
1810
|
flattenRules: options.flattenRules,
|
|
1530
|
-
mergeSheets: options.mergeSheets
|
|
1811
|
+
mergeSheets: options.mergeSheets,
|
|
1812
|
+
warnRagged: options.warnRagged
|
|
1531
1813
|
}
|
|
1532
1814
|
);
|
|
1533
1815
|
const delimitedNestedTree = /* @__PURE__ */ new Map();
|
|
1534
1816
|
for (const [columnName, sampler] of delimitedNested) {
|
|
1535
|
-
if (sampler.
|
|
1817
|
+
if (sampler.values.length > 0) delimitedNestedTree.set(columnName, buildColumnNestedTree(sampler.values));
|
|
1536
1818
|
}
|
|
1537
1819
|
const columnProfiles = [];
|
|
1538
1820
|
const timeFormatByColumn = /* @__PURE__ */ new Map();
|
|
1539
1821
|
for (const column of columns.values()) {
|
|
1540
1822
|
const profile = formatColumnProfile(column, rowCount, options.collectSamples ?? false);
|
|
1541
|
-
const
|
|
1542
|
-
if (
|
|
1823
|
+
const nestedTree2 = delimitedNestedTree.get(column.name);
|
|
1824
|
+
if (nestedTree2 && nestedTree2.length > 0) profile.nested_tree = nestedTree2;
|
|
1543
1825
|
columnProfiles.push(profile);
|
|
1544
1826
|
const dominantFormat = dominantTimeFormat(column);
|
|
1545
1827
|
if (dominantFormat) timeFormatByColumn.set(column.name, dominantFormat);
|
|
1546
1828
|
}
|
|
1829
|
+
const nestedTree = collectNestedTree && nestedObjects.length > 0 ? buildNestedTree(nestedObjects) : void 0;
|
|
1547
1830
|
const identityCandidates = findIdentityCandidates(columnProfiles);
|
|
1548
1831
|
const recommendedMapping = recommendMapping({
|
|
1549
1832
|
input,
|
|
@@ -1553,9 +1836,21 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1553
1836
|
recognizedRecordTypes,
|
|
1554
1837
|
sourceTimezone,
|
|
1555
1838
|
headerNames: options.headerNames,
|
|
1556
|
-
timeFormatByColumn
|
|
1839
|
+
timeFormatByColumn,
|
|
1840
|
+
nestedTree
|
|
1557
1841
|
});
|
|
1558
1842
|
const warnings = [...recommendedMapping.warnings ?? []];
|
|
1843
|
+
for (const column of columns.values()) {
|
|
1844
|
+
if (matchesName(column.name, UUID_NAMES) && column.nonMissing > 0) {
|
|
1845
|
+
const invalid = column.nonMissing - column.uuidValidCount;
|
|
1846
|
+
if (invalid > 0) warnings.push(`${column.name}: ${invalid} non-empty value(s) are not standard 36-character UUIDs; #uuid requires the standard UUID format.`);
|
|
1847
|
+
}
|
|
1848
|
+
if (matchesName(column.name, IP_NAMES) && column.nonMissing > 0) {
|
|
1849
|
+
const invalid = column.nonMissing - column.ipValidCount;
|
|
1850
|
+
if (invalid > 0) warnings.push(`${column.name}: ${invalid} non-empty value(s) are not valid IPv4 or IPv6 addresses.`);
|
|
1851
|
+
if (column.lanIpCount > 0) warnings.push(`${column.name}: ${column.lanIpCount} value(s) are private/LAN IP addresses; AE cannot resolve geo information for them.`);
|
|
1852
|
+
}
|
|
1853
|
+
}
|
|
1559
1854
|
if (rowCount === 0) warnings.push("The selected data set contains no data rows.");
|
|
1560
1855
|
return {
|
|
1561
1856
|
version: "ae-local-data-profile/v1",
|
|
@@ -1574,7 +1869,7 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
|
|
|
1574
1869
|
(recommendedMapping.account_id_field || recommendedMapping.distinct_id_field) && recommendedMapping.time.field
|
|
1575
1870
|
),
|
|
1576
1871
|
warnings,
|
|
1577
|
-
...
|
|
1872
|
+
...nestedTree ? { nested_tree: nestedTree } : {}
|
|
1578
1873
|
};
|
|
1579
1874
|
}
|
|
1580
1875
|
function normalizeAeName(input, fallback) {
|
|
@@ -1647,24 +1942,12 @@ function recommendMapping(input) {
|
|
|
1647
1942
|
warnings.push("No event-name column was found; review the generated default event name.");
|
|
1648
1943
|
}
|
|
1649
1944
|
const reserved = new Set([account?.name, distinct?.name, time?.name, event?.name, recordType?.name].filter(Boolean));
|
|
1650
|
-
const
|
|
1651
|
-
const
|
|
1652
|
-
|
|
1653
|
-
let target = baseTarget;
|
|
1654
|
-
let suffix = 2;
|
|
1655
|
-
while (usedTargets.has(target)) target = `${baseTarget.slice(0, 46)}_${suffix++}`;
|
|
1656
|
-
usedTargets.add(target);
|
|
1657
|
-
const type = mappingType(column.inferred_type);
|
|
1658
|
-
return {
|
|
1659
|
-
source: column.name,
|
|
1660
|
-
target,
|
|
1661
|
-
type,
|
|
1662
|
-
...type === "object" || type === "list" ? { transform: "json" } : {}
|
|
1663
|
-
};
|
|
1664
|
-
});
|
|
1945
|
+
const recordRoots = /* @__PURE__ */ new Map();
|
|
1946
|
+
for (const node of input.nestedTree ?? []) recordRoots.set(node.name, node);
|
|
1947
|
+
const { properties, flattenRules } = recommendProperties(input.columns, reserved, recordRoots, warnings);
|
|
1665
1948
|
const defaultEventSource = input.dataSet.kind === "sheet" ? input.dataSet.label : basename2(input.input.filePath, extname2(input.input.filePath));
|
|
1666
1949
|
return {
|
|
1667
|
-
version:
|
|
1950
|
+
version: MAPPING_VERSION,
|
|
1668
1951
|
source: {
|
|
1669
1952
|
sha256: input.input.sha256,
|
|
1670
1953
|
format: input.input.format,
|
|
@@ -1684,6 +1967,7 @@ function recommendMapping(input) {
|
|
|
1684
1967
|
...recordType ? { record_type_field: recordType.name } : {},
|
|
1685
1968
|
...event ? { event_name_field: event.name } : {},
|
|
1686
1969
|
...mode !== "user_set" && !event ? { default_event_name: normalizeAeName(defaultEventSource, "local_event") } : {},
|
|
1970
|
+
...Object.keys(flattenRules).length > 0 ? { flatten_rules: flattenRules } : {},
|
|
1687
1971
|
properties,
|
|
1688
1972
|
...warnings.length > 0 ? { warnings } : {}
|
|
1689
1973
|
};
|
|
@@ -1767,9 +2051,120 @@ function normalizeRecordType(value) {
|
|
|
1767
2051
|
};
|
|
1768
2052
|
return aliases[normalized];
|
|
1769
2053
|
}
|
|
2054
|
+
function recommendProperties(columns, reserved, recordRoots, warnings) {
|
|
2055
|
+
const properties = [];
|
|
2056
|
+
const flattenRules = {};
|
|
2057
|
+
const usedTargets = /* @__PURE__ */ new Set();
|
|
2058
|
+
const claim = (desired) => {
|
|
2059
|
+
let target = desired;
|
|
2060
|
+
let suffix = 2;
|
|
2061
|
+
while (usedTargets.has(target)) target = `${desired.slice(0, 46)}_${suffix++}`;
|
|
2062
|
+
usedTargets.add(target);
|
|
2063
|
+
return target;
|
|
2064
|
+
};
|
|
2065
|
+
const state = { properties, flattenRules, usedTargets, warnings, claim };
|
|
2066
|
+
columns.filter((column) => !reserved.has(column.name)).forEach((column, index) => {
|
|
2067
|
+
const baseName = normalizeAeName(column.name, `field_${index + 1}`);
|
|
2068
|
+
const valueNode = columnValueNode(column, recordRoots);
|
|
2069
|
+
if (!valueNode) {
|
|
2070
|
+
const type = mappingType(column.inferred_type);
|
|
2071
|
+
const target = claim(baseName);
|
|
2072
|
+
properties.push({ source: column.name, target, type, ...isContainerType(type) ? { transform: "json" } : {} });
|
|
2073
|
+
return;
|
|
2074
|
+
}
|
|
2075
|
+
if (valueNode.kind === "primitive") {
|
|
2076
|
+
const type = mappingType(valueNode.inferredType ?? column.inferred_type);
|
|
2077
|
+
const target = claim(baseName);
|
|
2078
|
+
properties.push({ source: column.name, target, type });
|
|
2079
|
+
return;
|
|
2080
|
+
}
|
|
2081
|
+
if (valueNode.kind === "object") {
|
|
2082
|
+
recommendObject(state, baseName, column.name, valueNode.children ?? [], column.name);
|
|
2083
|
+
return;
|
|
2084
|
+
}
|
|
2085
|
+
recommendArray(state, baseName, column.name, valueNode, column.name);
|
|
2086
|
+
});
|
|
2087
|
+
return { properties, flattenRules };
|
|
2088
|
+
}
|
|
2089
|
+
function columnValueNode(column, recordRoots) {
|
|
2090
|
+
const recordNode = recordRoots.get(column.name);
|
|
2091
|
+
if (recordNode) return recordNode;
|
|
2092
|
+
const tree = column.nested_tree;
|
|
2093
|
+
if (!tree || tree.length === 0) return void 0;
|
|
2094
|
+
if (tree.length === 1 && tree[0].kind === "array") return tree[0];
|
|
2095
|
+
return { path: "", name: column.name, kind: "object", children: tree, nonEmpty: true };
|
|
2096
|
+
}
|
|
2097
|
+
function isScalarNode(node) {
|
|
2098
|
+
return node.kind === "primitive" || node.kind === "array" && node.elementKind === "primitive";
|
|
2099
|
+
}
|
|
2100
|
+
function isContainerType(type) {
|
|
2101
|
+
return type === "object" || type === "list" || type === "array_row";
|
|
2102
|
+
}
|
|
2103
|
+
function scalarPropType(node) {
|
|
2104
|
+
if (node.kind === "array") return "list";
|
|
2105
|
+
switch (node.inferredType) {
|
|
2106
|
+
case "number":
|
|
2107
|
+
return "number";
|
|
2108
|
+
case "boolean":
|
|
2109
|
+
return "boolean";
|
|
2110
|
+
case "datetime":
|
|
2111
|
+
return "datetime";
|
|
2112
|
+
default:
|
|
2113
|
+
return "string";
|
|
2114
|
+
}
|
|
2115
|
+
}
|
|
2116
|
+
function snakeSegment(name) {
|
|
2117
|
+
return normalizeAeName(name, "field");
|
|
2118
|
+
}
|
|
2119
|
+
function recommendObject(state, prefix, dotPath, children, columnName) {
|
|
2120
|
+
const hasComposite = children.some((child) => child.kind === "object" || child.kind === "array" && child.elementKind === "object");
|
|
2121
|
+
if (!hasComposite) {
|
|
2122
|
+
const parentTarget = state.claim(prefix);
|
|
2123
|
+
const source = dotPath === columnName ? columnName : parentTarget;
|
|
2124
|
+
if (source !== columnName) state.flattenRules[parentTarget] = dotPath;
|
|
2125
|
+
state.properties.push({ source, target: parentTarget, type: "object", transform: "json" });
|
|
2126
|
+
for (const child of children) {
|
|
2127
|
+
state.properties.push({ source: parentTarget, target: `${parentTarget}.${snakeSegment(child.name)}`, type: scalarPropType(child) });
|
|
2128
|
+
}
|
|
2129
|
+
return;
|
|
2130
|
+
}
|
|
2131
|
+
for (const child of children) {
|
|
2132
|
+
const childPrefix = `${prefix}_${snakeSegment(child.name)}`;
|
|
2133
|
+
const childDotPath = `${dotPath}.${child.name}`;
|
|
2134
|
+
if (isScalarNode(child)) {
|
|
2135
|
+
const target = state.claim(childPrefix);
|
|
2136
|
+
state.flattenRules[target] = childDotPath;
|
|
2137
|
+
state.properties.push({ source: target, target, type: scalarPropType(child) });
|
|
2138
|
+
} else if (child.kind === "object") {
|
|
2139
|
+
recommendObject(state, childPrefix, childDotPath, child.children ?? [], columnName);
|
|
2140
|
+
} else {
|
|
2141
|
+
recommendArray(state, childPrefix, childDotPath, child, columnName);
|
|
2142
|
+
}
|
|
2143
|
+
}
|
|
2144
|
+
}
|
|
2145
|
+
function recommendArray(state, prefix, dotPath, node, columnName) {
|
|
2146
|
+
if (node.elementKind !== "object") {
|
|
2147
|
+
const target = state.claim(prefix);
|
|
2148
|
+
const source2 = dotPath === columnName ? columnName : target;
|
|
2149
|
+
if (source2 !== columnName) state.flattenRules[target] = dotPath;
|
|
2150
|
+
state.properties.push({ source: source2, target, type: "list", transform: "json" });
|
|
2151
|
+
return;
|
|
2152
|
+
}
|
|
2153
|
+
const parentTarget = state.claim(prefix);
|
|
2154
|
+
const source = dotPath === columnName ? columnName : parentTarget;
|
|
2155
|
+
if (source !== columnName) state.flattenRules[parentTarget] = dotPath;
|
|
2156
|
+
state.properties.push({ source, target: parentTarget, type: "array_row", transform: "json" });
|
|
2157
|
+
for (const field of node.children ?? []) {
|
|
2158
|
+
if (isScalarNode(field)) {
|
|
2159
|
+
state.properties.push({ source: parentTarget, target: `${parentTarget}.${snakeSegment(field.name)}`, type: scalarPropType(field) });
|
|
2160
|
+
} else {
|
|
2161
|
+
state.warnings.push(`${dotPath} element field "${field.name}" is nested; it stays inside the ${parentTarget} array data and is not declared as a sub-property (array-element flattening is not yet supported).`);
|
|
2162
|
+
}
|
|
2163
|
+
}
|
|
2164
|
+
}
|
|
1770
2165
|
function mappingType(type) {
|
|
1771
2166
|
if (type === "datetime") return "datetime";
|
|
1772
|
-
if (type === "number" || type === "boolean" || type === "
|
|
2167
|
+
if (type === "number" || type === "boolean" || type === "object" || type === "list") return type;
|
|
1773
2168
|
return "string";
|
|
1774
2169
|
}
|
|
1775
2170
|
function compatibleTypes(types) {
|
|
@@ -1832,7 +2227,7 @@ function dominantTimeFormat(column) {
|
|
|
1832
2227
|
return best && bestCount >= total * 0.9 ? best : void 0;
|
|
1833
2228
|
}
|
|
1834
2229
|
|
|
1835
|
-
// src/commands/data-integration/
|
|
2230
|
+
// src/commands/data-integration/inspect.ts
|
|
1836
2231
|
var HEADERLESS_WARNING = "The first row appears to be data, not a header; columns were auto-named col_1..col_N. Re-run with --headers to supply explicit names.";
|
|
1837
2232
|
var dataIntegrationInspect = {
|
|
1838
2233
|
service: "data-integration",
|
|
@@ -1847,6 +2242,30 @@ var dataIntegrationInspect = {
|
|
|
1847
2242
|
{ name: "headerless", type: "boolean", default: false, desc: "Treat the first row as data and auto-generate col_1..col_N names." }
|
|
1848
2243
|
],
|
|
1849
2244
|
risk: "read",
|
|
2245
|
+
// Fast size/time pre-check (stat + format/encoding sniff only — no sha256, no profile).
|
|
2246
|
+
// Lets agents surface the estimate before committing to a multi-minute full inspection.
|
|
2247
|
+
dryRun: async (ctx) => {
|
|
2248
|
+
const files = ctx.list("input-file").map((filePath) => {
|
|
2249
|
+
const meta = resolveLocalDataInputMeta(filePath);
|
|
2250
|
+
const assessment = assessFileSize(basename3(filePath), meta.format, meta.sizeBytes, meta.encoding);
|
|
2251
|
+
return {
|
|
2252
|
+
file: basename3(filePath),
|
|
2253
|
+
format: meta.format,
|
|
2254
|
+
size_bytes: meta.sizeBytes,
|
|
2255
|
+
size: assessment.size,
|
|
2256
|
+
...assessment.estimatedDuration ? { estimated_duration: assessment.estimatedDuration } : {},
|
|
2257
|
+
...assessment.warning ? { warning: assessment.warning } : {},
|
|
2258
|
+
...assessment.reason ? { reason: assessment.reason } : {},
|
|
2259
|
+
...assessment.memoryRisk ? { memory_risk: true } : {},
|
|
2260
|
+
...assessment.rejected ? { rejected: true } : {}
|
|
2261
|
+
};
|
|
2262
|
+
});
|
|
2263
|
+
return {
|
|
2264
|
+
version: "ae-local-data-estimate/v1",
|
|
2265
|
+
files,
|
|
2266
|
+
has_large_file: files.some((file) => Boolean(file.warning) || file.rejected)
|
|
2267
|
+
};
|
|
2268
|
+
},
|
|
1850
2269
|
execute: async (ctx) => {
|
|
1851
2270
|
const inputFiles = ctx.list("input-file");
|
|
1852
2271
|
const headerNames = splitHeaders(ctx.str("headers"));
|
|
@@ -1940,7 +2359,7 @@ function splitHeaders(raw) {
|
|
|
1940
2359
|
return headers.length > 0 ? headers : void 0;
|
|
1941
2360
|
}
|
|
1942
2361
|
|
|
1943
|
-
// src/commands/data-integration/
|
|
2362
|
+
// src/commands/data-integration/plan.ts
|
|
1944
2363
|
import { writeFile } from "fs/promises";
|
|
1945
2364
|
var VALID_LOCALES = /* @__PURE__ */ new Set(["zh", "en", "ja", "ko"]);
|
|
1946
2365
|
var EVENT_NAME_RE = /^[a-z][a-z0-9_]*$/;
|
|
@@ -1956,6 +2375,8 @@ function toPropType(type) {
|
|
|
1956
2375
|
return "array_string";
|
|
1957
2376
|
case "object":
|
|
1958
2377
|
return "object";
|
|
2378
|
+
case "array_row":
|
|
2379
|
+
return "array_row";
|
|
1959
2380
|
case "string":
|
|
1960
2381
|
return "string";
|
|
1961
2382
|
}
|
|
@@ -1963,13 +2384,16 @@ function toPropType(type) {
|
|
|
1963
2384
|
function buildDraftFromMapping(options) {
|
|
1964
2385
|
const { mapping } = options;
|
|
1965
2386
|
const excluded = new Set(mapping.exclude_columns ?? []);
|
|
1966
|
-
const properties = mapping.properties.filter((property) => !excluded.has(property.source)).map((property) =>
|
|
1967
|
-
|
|
1968
|
-
|
|
1969
|
-
|
|
1970
|
-
|
|
1971
|
-
|
|
1972
|
-
|
|
2387
|
+
const properties = mapping.properties.filter((property) => !excluded.has(property.source)).map((property) => {
|
|
2388
|
+
const leaf = property.target.includes(".") ? property.target.split(".").pop() : void 0;
|
|
2389
|
+
return {
|
|
2390
|
+
name: property.target,
|
|
2391
|
+
display_name: leaf ?? property.source,
|
|
2392
|
+
desc: property.desc ?? leaf ?? property.source,
|
|
2393
|
+
type: toPropType(property.type),
|
|
2394
|
+
source: "data"
|
|
2395
|
+
};
|
|
2396
|
+
});
|
|
1973
2397
|
const propNames = properties.map((property) => property.name);
|
|
1974
2398
|
const eventNames = resolveEventNames(options);
|
|
1975
2399
|
const events = eventNames.map((eventName) => {
|
|
@@ -2047,7 +2471,7 @@ function buildPlanDraft(ctx) {
|
|
|
2047
2471
|
location: { field: "lang" }
|
|
2048
2472
|
});
|
|
2049
2473
|
}
|
|
2050
|
-
|
|
2474
|
+
const draft = buildDraftFromMapping({
|
|
2051
2475
|
mapping,
|
|
2052
2476
|
planName: ctx.str("plan-name").trim() || mapping.default_event_name || "local-data",
|
|
2053
2477
|
eventNames: ctx.list("event-name"),
|
|
@@ -2055,6 +2479,17 @@ function buildPlanDraft(ctx) {
|
|
|
2055
2479
|
lang,
|
|
2056
2480
|
projectId: ctx.optionalNum("project-id")
|
|
2057
2481
|
});
|
|
2482
|
+
validateAndFix(draft);
|
|
2483
|
+
try {
|
|
2484
|
+
validateDraft(draft);
|
|
2485
|
+
} catch (error) {
|
|
2486
|
+
throw new CliValidationError("The tracking-plan draft is invalid.", {
|
|
2487
|
+
code: "LOCAL_DATA_PLAN_INVALID_DRAFT",
|
|
2488
|
+
hint: error instanceof Error ? error.message : String(error),
|
|
2489
|
+
location: { field: "mapping" }
|
|
2490
|
+
});
|
|
2491
|
+
}
|
|
2492
|
+
return draft;
|
|
2058
2493
|
}
|
|
2059
2494
|
var dataIntegrationPlan = {
|
|
2060
2495
|
service: "data-integration",
|
|
@@ -2062,7 +2497,7 @@ var dataIntegrationPlan = {
|
|
|
2062
2497
|
usesAeHost: false,
|
|
2063
2498
|
description: "Convert a confirmed local-data mapping into a tracking-plan draft.json (source_type=data, sdk_integration_mode=none).",
|
|
2064
2499
|
flags: [
|
|
2065
|
-
{ name: "mapping", type: "string", required: true, sensitive: true, desc:
|
|
2500
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: `Confirmed ${MAPPING_VERSION} JSON, file path, or @file.` },
|
|
2066
2501
|
{ name: "event-name", type: "string", variadic: true, desc: "Concrete event name (track/mixed without default_event_name). Repeat for multiple events." },
|
|
2067
2502
|
{ name: "plan-name", type: "string", desc: 'Plan name. Default: the mapping default_event_name, else "local-data".' },
|
|
2068
2503
|
{ name: "app-type", type: "string", default: "unknown", desc: "Application type recorded in draft meta (informational)." },
|
|
@@ -2093,7 +2528,7 @@ var dataIntegrationPlan = {
|
|
|
2093
2528
|
}
|
|
2094
2529
|
};
|
|
2095
2530
|
|
|
2096
|
-
// src/commands/data-integration/
|
|
2531
|
+
// src/commands/data-integration/conversion.ts
|
|
2097
2532
|
import { randomInt as randomInt2, randomUUID } from "crypto";
|
|
2098
2533
|
import {
|
|
2099
2534
|
chmodSync,
|
|
@@ -2114,14 +2549,6 @@ import { createInterface as createInterface2 } from "readline";
|
|
|
2114
2549
|
var SORT_CHUNK_SIZE = 1e4;
|
|
2115
2550
|
var THREE_YEARS_MS = 3 * 365 * 24 * 60 * 60 * 1e3;
|
|
2116
2551
|
var THREE_DAYS_MS = 3 * 24 * 60 * 60 * 1e3;
|
|
2117
|
-
function stripQuotes(value) {
|
|
2118
|
-
if (typeof value !== "string") return value;
|
|
2119
|
-
const text = value;
|
|
2120
|
-
if (text.length >= 2 && (text.startsWith('"') && text.endsWith('"') || text.startsWith("'") && text.endsWith("'"))) {
|
|
2121
|
-
return text.slice(1, -1).trim();
|
|
2122
|
-
}
|
|
2123
|
-
return text.trim();
|
|
2124
|
-
}
|
|
2125
2552
|
async function convertLocalData(options) {
|
|
2126
2553
|
const input = await inspectLocalDataInput(options.inputFile);
|
|
2127
2554
|
if (input.format !== options.mapping.source.format) {
|
|
@@ -2141,12 +2568,15 @@ async function convertLocalData(options) {
|
|
|
2141
2568
|
const streamOptions = {
|
|
2142
2569
|
headerNames: options.mapping.headers,
|
|
2143
2570
|
flattenRules: options.mapping.flatten_rules,
|
|
2144
|
-
mergeSheets: options.mergeSheets
|
|
2571
|
+
mergeSheets: options.mergeSheets,
|
|
2572
|
+
// The profile pass inside convert is internal (it writes profile.json); the ragged-row
|
|
2573
|
+
// warning is surfaced by the conversion pass below instead, so suppress it here.
|
|
2574
|
+
warnRagged: false
|
|
2145
2575
|
};
|
|
2146
2576
|
const salvageSet = options.salvageFrom ? readSalvageRowNumbers(options.salvageFrom) : void 0;
|
|
2147
2577
|
let salvageMatched = 0;
|
|
2148
2578
|
const runId = `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`;
|
|
2149
|
-
const outputDir = resolve(options.outputDir || join(".ae-cli", "data-integration", runId));
|
|
2579
|
+
const outputDir = resolve(options.outputDir || join(".ae-cli", "data-integration", "runs", runId));
|
|
2150
2580
|
prepareOutputDirectory(outputDir);
|
|
2151
2581
|
const profile = await profileLocalData(input, dataSet, options.mapping.time.source_timezone, streamOptions);
|
|
2152
2582
|
const profilePath = join(outputDir, "profile.json");
|
|
@@ -2163,7 +2593,10 @@ async function convertLocalData(options) {
|
|
|
2163
2593
|
let validRecords = 0;
|
|
2164
2594
|
let invalidRecords = 0;
|
|
2165
2595
|
const recordTypes = {};
|
|
2166
|
-
|
|
2596
|
+
const skippedFields = {};
|
|
2597
|
+
let lanIpRecords = 0;
|
|
2598
|
+
const flattenMisses = {};
|
|
2599
|
+
const rowCount = await streamLocalDataRows(
|
|
2167
2600
|
input,
|
|
2168
2601
|
dataSet,
|
|
2169
2602
|
async (row, rowNumber) => {
|
|
@@ -2177,6 +2610,10 @@ async function convertLocalData(options) {
|
|
|
2177
2610
|
}
|
|
2178
2611
|
validRecords += 1;
|
|
2179
2612
|
recordTypes[result.recordType] = (recordTypes[result.recordType] ?? 0) + 1;
|
|
2613
|
+
if (result.lanIp) lanIpRecords += 1;
|
|
2614
|
+
for (const skip of result.skips) {
|
|
2615
|
+
skippedFields[skip.code] = (skippedFields[skip.code] ?? 0) + 1;
|
|
2616
|
+
}
|
|
2180
2617
|
const line = JSON.stringify(result.record);
|
|
2181
2618
|
if (isUserProfileType(result.recordType)) {
|
|
2182
2619
|
userSetBuffer.push({ key: result.sortKey, line });
|
|
@@ -2188,13 +2625,16 @@ async function convertLocalData(options) {
|
|
|
2188
2625
|
{
|
|
2189
2626
|
headerNames: streamOptions.headerNames,
|
|
2190
2627
|
flattenRules: streamOptions.flattenRules,
|
|
2628
|
+
flattenMisses,
|
|
2191
2629
|
mergeSheets: streamOptions.mergeSheets
|
|
2192
2630
|
}
|
|
2193
2631
|
);
|
|
2194
2632
|
if (userSetBuffer.length > 0) flushSortChunk(outputDir, userSetBuffer, sortChunks);
|
|
2195
|
-
|
|
2196
|
-
|
|
2197
|
-
|
|
2633
|
+
await Promise.all([finishStream(trackStream), finishStream(invalidStream)]);
|
|
2634
|
+
for (const [outColumn, count] of Object.entries(flattenMisses)) {
|
|
2635
|
+
process.stderr.write(`Warning: flatten rule "${outColumn}" did not materialize for ${count} row(s).
|
|
2636
|
+
`);
|
|
2637
|
+
}
|
|
2198
2638
|
if (salvageSet && salvageMatched === 0) {
|
|
2199
2639
|
throw new CliValidationError("The salvage file lists no rows from this source.", {
|
|
2200
2640
|
code: "LOCAL_DATA_SALVAGE_NO_MATCH",
|
|
@@ -2205,12 +2645,11 @@ async function convertLocalData(options) {
|
|
|
2205
2645
|
const validStream = secureWriteStream(validPath);
|
|
2206
2646
|
if (existsSync(trackTempPath)) {
|
|
2207
2647
|
for await (const chunk of createReadStream3(trackTempPath)) {
|
|
2208
|
-
|
|
2648
|
+
await writeRaw(validStream, chunk);
|
|
2209
2649
|
}
|
|
2210
2650
|
}
|
|
2211
2651
|
await mergeSortChunks(sortChunks, validStream);
|
|
2212
|
-
validStream
|
|
2213
|
-
await once(validStream, "finish");
|
|
2652
|
+
await finishStream(validStream);
|
|
2214
2653
|
if (existsSync(trackTempPath)) unlinkSync(trackTempPath);
|
|
2215
2654
|
for (const path of sortChunks) if (existsSync(path)) unlinkSync(path);
|
|
2216
2655
|
writeSecureJson(profilePath, profile);
|
|
@@ -2218,8 +2657,9 @@ async function convertLocalData(options) {
|
|
|
2218
2657
|
writeSecureText(transformPath, createTransformScript(options.inputFile, mappingPath));
|
|
2219
2658
|
const validBytes = statSize(validPath);
|
|
2220
2659
|
const blockedReasons = [
|
|
2221
|
-
...
|
|
2222
|
-
...validRecords === 0 ? ["No valid UE records were generated."] : []
|
|
2660
|
+
...rowCount === 0 ? ["The source contained no data rows."] : [],
|
|
2661
|
+
...rowCount > 0 && validRecords === 0 ? ["No valid UE records were generated."] : [],
|
|
2662
|
+
...invalidRecords > 0 ? ["Some source rows failed UE validation."] : []
|
|
2223
2663
|
];
|
|
2224
2664
|
const manifest = {
|
|
2225
2665
|
version: "ae-local-data-manifest/v1",
|
|
@@ -2240,7 +2680,10 @@ async function convertLocalData(options) {
|
|
|
2240
2680
|
valid_records: validRecords,
|
|
2241
2681
|
invalid_records: invalidRecords,
|
|
2242
2682
|
valid_bytes: validBytes,
|
|
2243
|
-
record_types: recordTypes
|
|
2683
|
+
record_types: recordTypes,
|
|
2684
|
+
...Object.keys(skippedFields).length > 0 ? { skipped_fields: skippedFields } : {},
|
|
2685
|
+
...lanIpRecords > 0 ? { lan_ip_records: lanIpRecords } : {},
|
|
2686
|
+
...Object.keys(flattenMisses).length > 0 ? { flatten_misses: flattenMisses } : {}
|
|
2244
2687
|
},
|
|
2245
2688
|
blocked_reasons: blockedReasons
|
|
2246
2689
|
};
|
|
@@ -2258,7 +2701,8 @@ async function convertLocalDataMulti(options) {
|
|
|
2258
2701
|
const profile = await profileLocalData(input, dataSet, sourceTimezone, {
|
|
2259
2702
|
collectSamples: true,
|
|
2260
2703
|
headerNames: options.mapping.headers,
|
|
2261
|
-
flattenRules: options.mapping.flatten_rules
|
|
2704
|
+
flattenRules: options.mapping.flatten_rules,
|
|
2705
|
+
warnRagged: false
|
|
2262
2706
|
});
|
|
2263
2707
|
profiled.push({ file: basename4(inputFile), profile });
|
|
2264
2708
|
}
|
|
@@ -2274,7 +2718,7 @@ async function convertLocalDataMulti(options) {
|
|
|
2274
2718
|
validateTypeResolutions(resolutions, profiled.map((entry) => entry.file));
|
|
2275
2719
|
const overrides = applyTypeResolutions(resolutions, profiled);
|
|
2276
2720
|
const parent = resolve(
|
|
2277
|
-
options.outputDir || join(".ae-cli", "data-integration", `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`)
|
|
2721
|
+
options.outputDir || join(".ae-cli", "data-integration", "runs", `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`)
|
|
2278
2722
|
);
|
|
2279
2723
|
prepareOutputDirectory(parent);
|
|
2280
2724
|
const files = [];
|
|
@@ -2332,11 +2776,32 @@ function convertRow(row, rowNumber, mapping, now) {
|
|
|
2332
2776
|
errors.push({ code: "INVALID_EVENT_NAME", field: mapping.event_name_field });
|
|
2333
2777
|
}
|
|
2334
2778
|
}
|
|
2335
|
-
const
|
|
2336
|
-
|
|
2779
|
+
const skips = [];
|
|
2780
|
+
let lanIp = false;
|
|
2781
|
+
let ip;
|
|
2782
|
+
if (recordType === "track") {
|
|
2783
|
+
const rawIp = readOptionalField(row, mapping.ip_field);
|
|
2784
|
+
if (rawIp) {
|
|
2785
|
+
if (isValidIp(rawIp)) {
|
|
2786
|
+
ip = rawIp;
|
|
2787
|
+
lanIp = isPrivateIp(rawIp);
|
|
2788
|
+
} else {
|
|
2789
|
+
skips.push({ code: "INVALID_IP", field: mapping.ip_field });
|
|
2790
|
+
}
|
|
2791
|
+
}
|
|
2792
|
+
}
|
|
2793
|
+
let uuid;
|
|
2794
|
+
{
|
|
2795
|
+
const rawUuid = readOptionalField(row, mapping.uuid_field);
|
|
2796
|
+
if (rawUuid) {
|
|
2797
|
+
if (isValidUuid(rawUuid)) uuid = rawUuid;
|
|
2798
|
+
else skips.push({ code: "INVALID_UUID", field: mapping.uuid_field });
|
|
2799
|
+
}
|
|
2800
|
+
}
|
|
2337
2801
|
const excluded = new Set(mapping.exclude_columns ?? []);
|
|
2338
2802
|
const properties = {};
|
|
2339
2803
|
for (const property of mapping.properties) {
|
|
2804
|
+
if (property.target.includes(".")) continue;
|
|
2340
2805
|
if (excluded.has(property.source)) continue;
|
|
2341
2806
|
let value = stripQuotes(row[property.source]);
|
|
2342
2807
|
if (isMissing2(value)) continue;
|
|
@@ -2350,8 +2815,8 @@ function convertRow(row, rowNumber, mapping, now) {
|
|
|
2350
2815
|
properties[property.target] = converted.value;
|
|
2351
2816
|
}
|
|
2352
2817
|
}
|
|
2353
|
-
const zoneOffset = mapping.zone_offset_value !== void 0 ? resolveZoneOffsetValue(mapping.zone_offset_value, now) : mapping.zone_offset_field ? readZoneOffset(row, mapping.zone_offset_field) : void 0;
|
|
2354
|
-
if (mapping.zone_offset_field && zoneOffset === void 0) {
|
|
2818
|
+
const zoneOffset = recordType === "track" && mapping.zone_offset_value !== void 0 ? resolveZoneOffsetValue(mapping.zone_offset_value, now) : recordType === "track" && mapping.zone_offset_field ? readZoneOffset(row, mapping.zone_offset_field) : void 0;
|
|
2819
|
+
if (recordType === "track" && mapping.zone_offset_field && zoneOffset === void 0) {
|
|
2355
2820
|
errors.push({ code: "INVALID_ZONE_OFFSET", field: mapping.zone_offset_field });
|
|
2356
2821
|
}
|
|
2357
2822
|
if (errors.length > 0 || !recordType || !normalizedTime) return { ok: false, errors };
|
|
@@ -2369,7 +2834,9 @@ function convertRow(row, rowNumber, mapping, now) {
|
|
|
2369
2834
|
ok: true,
|
|
2370
2835
|
recordType,
|
|
2371
2836
|
record,
|
|
2372
|
-
sortKey: `${accountId ?? distinctId ?? ""}\0${normalizedTime.instant.toISOString()}\0${String(rowNumber).padStart(12, "0")}
|
|
2837
|
+
sortKey: `${accountId ?? distinctId ?? ""}\0${normalizedTime.instant.toISOString()}\0${String(rowNumber).padStart(12, "0")}`,
|
|
2838
|
+
skips,
|
|
2839
|
+
lanIp
|
|
2373
2840
|
};
|
|
2374
2841
|
}
|
|
2375
2842
|
function readOptionalField(row, field) {
|
|
@@ -2477,7 +2944,7 @@ function convertProperty(value, type, transform, timeZone = "UTC", timeFormat) {
|
|
|
2477
2944
|
const normalized = normalizeTime(value, timeZone, timeFormat);
|
|
2478
2945
|
return normalized ? { ok: true, value: normalized.formatted } : { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2479
2946
|
}
|
|
2480
|
-
if (type === "list") {
|
|
2947
|
+
if (type === "list" || type === "array_row") {
|
|
2481
2948
|
if (!Array.isArray(value)) return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2482
2949
|
return isPropertyWithinLimits(value, type) ? { ok: true, value } : { ok: false, code: "PROPERTY_LIMIT_EXCEEDED" };
|
|
2483
2950
|
}
|
|
@@ -2679,11 +3146,40 @@ function prepareOutputDirectory(path) {
|
|
|
2679
3146
|
chmodSync(path, 448);
|
|
2680
3147
|
}
|
|
2681
3148
|
function secureWriteStream(path) {
|
|
2682
|
-
|
|
3149
|
+
const stream = createWriteStream(path, { encoding: "utf8", mode: 384, flags: "wx" });
|
|
3150
|
+
let fail;
|
|
3151
|
+
const errorPromise = new Promise((_, reject) => {
|
|
3152
|
+
fail = reject;
|
|
3153
|
+
});
|
|
3154
|
+
errorPromise.catch(() => {
|
|
3155
|
+
});
|
|
3156
|
+
stream.on("error", (error) => fail?.(error));
|
|
3157
|
+
return { stream, path, errorPromise };
|
|
2683
3158
|
}
|
|
2684
|
-
|
|
2685
|
-
|
|
2686
|
-
|
|
3159
|
+
function writeFailure(path, error) {
|
|
3160
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
3161
|
+
return new Error(`Failed to write "${path}": ${detail}. Check available disk space and directory permissions.`, { cause: error });
|
|
3162
|
+
}
|
|
3163
|
+
async function writeRaw(target, data) {
|
|
3164
|
+
try {
|
|
3165
|
+
if (!target.stream.write(data)) {
|
|
3166
|
+
await Promise.race([once(target.stream, "drain"), target.errorPromise]);
|
|
3167
|
+
}
|
|
3168
|
+
} catch (error) {
|
|
3169
|
+
throw writeFailure(target.path, error);
|
|
3170
|
+
}
|
|
3171
|
+
}
|
|
3172
|
+
async function writeLine(target, line) {
|
|
3173
|
+
await writeRaw(target, `${line}
|
|
3174
|
+
`);
|
|
3175
|
+
}
|
|
3176
|
+
async function finishStream(target) {
|
|
3177
|
+
try {
|
|
3178
|
+
target.stream.end();
|
|
3179
|
+
await Promise.race([once(target.stream, "finish"), target.errorPromise]);
|
|
3180
|
+
} catch (error) {
|
|
3181
|
+
throw writeFailure(target.path, error);
|
|
3182
|
+
}
|
|
2687
3183
|
}
|
|
2688
3184
|
function writeSecureJson(path, value) {
|
|
2689
3185
|
writeSecureText(path, `${JSON.stringify(value, null, 2)}
|
|
@@ -2710,7 +3206,7 @@ function formatRunTimestamp(value) {
|
|
|
2710
3206
|
return value.toISOString().replace(/[-:]/g, "").replace(/\.\d{3}Z$/, "Z");
|
|
2711
3207
|
}
|
|
2712
3208
|
|
|
2713
|
-
// src/commands/data-integration/
|
|
3209
|
+
// src/commands/data-integration/convert.ts
|
|
2714
3210
|
var dataIntegrationConvert = {
|
|
2715
3211
|
service: "data-integration",
|
|
2716
3212
|
command: "convert",
|
|
@@ -2718,8 +3214,8 @@ var dataIntegrationConvert = {
|
|
|
2718
3214
|
description: "Convert one or more local data sets into validated UE JSONL and quarantine invalid rows.",
|
|
2719
3215
|
flags: [
|
|
2720
3216
|
{ name: "input-file", type: "string", required: true, sensitive: true, variadic: true, desc: "Source local data file. Repeat for multiple files (requires a wildcard mapping). The source is never modified." },
|
|
2721
|
-
{ name: "mapping", type: "string", required: true, sensitive: true, desc:
|
|
2722
|
-
{ name: "output-dir", type: "string", sensitive: true, desc: "New or empty output directory. Default: .ae-cli/data-integration/<run-id>." },
|
|
3217
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: `${MAPPING_VERSION} JSON, file path, or @file.` },
|
|
3218
|
+
{ name: "output-dir", type: "string", sensitive: true, desc: "New or empty output directory. Default: .ae-cli/data-integration/runs/<run-id>." },
|
|
2723
3219
|
{ name: "type-resolutions", type: "json", sensitive: true, desc: "JSON object resolving cross-file column type conflicts (unify, split, or skip)." },
|
|
2724
3220
|
{ name: "merge-sheets", type: "boolean", default: false, desc: "Stream every worksheet in file order instead of a single selected sheet." },
|
|
2725
3221
|
{ name: "salvage-from", type: "string", sensitive: true, desc: "Re-process only the rows listed in a previous run's invalid.rows.jsonl, against the current (fixed) mapping. Single-file only." }
|
|
@@ -2771,7 +3267,7 @@ var dataIntegrationConvert = {
|
|
|
2771
3267
|
}
|
|
2772
3268
|
};
|
|
2773
3269
|
|
|
2774
|
-
// src/commands/data-integration/
|
|
3270
|
+
// src/commands/data-integration/upload.ts
|
|
2775
3271
|
import { createReadStream as createReadStream4, readFileSync as readFileSync3, statSync as statSync3 } from "fs";
|
|
2776
3272
|
import { basename as basename5, dirname, resolve as resolve2 } from "path";
|
|
2777
3273
|
import { createInterface as createInterface3 } from "readline";
|
|
@@ -3116,22 +3612,900 @@ function isRecord3(value) {
|
|
|
3116
3612
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3117
3613
|
}
|
|
3118
3614
|
|
|
3119
|
-
// src/commands/data-integration/
|
|
3615
|
+
// src/commands/data-integration/handoff.ts
|
|
3120
3616
|
import { createHash as createHash3 } from "crypto";
|
|
3121
|
-
import { chmodSync as chmodSync2, mkdirSync as mkdirSync2, readFileSync as readFileSync4, renameSync as renameSync2, writeFileSync as writeFileSync2 } from "fs";
|
|
3122
|
-
import { join as
|
|
3617
|
+
import { chmodSync as chmodSync2, mkdirSync as mkdirSync2, mkdtempSync, readFileSync as readFileSync4, renameSync as renameSync2, rmSync, writeFileSync as writeFileSync2 } from "fs";
|
|
3618
|
+
import { dirname as dirname2, join as join3, resolve as resolve3 } from "path";
|
|
3619
|
+
import { tmpdir } from "os";
|
|
3620
|
+
|
|
3621
|
+
// src/commands/data-integration/archive.ts
|
|
3622
|
+
import archiver from "archiver";
|
|
3623
|
+
import { createWriteStream as createWriteStream2, readdirSync as readdirSync2, statSync as statSync4 } from "fs";
|
|
3624
|
+
import { join as join2, relative, sep } from "path";
|
|
3625
|
+
async function zipPackage(dir, zipPath) {
|
|
3626
|
+
await new Promise((resolvePromise, rejectPromise) => {
|
|
3627
|
+
const output = createWriteStream2(zipPath, { mode: 384 });
|
|
3628
|
+
const zip = archiver("zip", { zlib: { level: 9 } });
|
|
3629
|
+
output.on("close", resolvePromise);
|
|
3630
|
+
output.on("error", rejectPromise);
|
|
3631
|
+
zip.on("error", rejectPromise);
|
|
3632
|
+
zip.on("warning", (error) => {
|
|
3633
|
+
if (error.code !== "ENOENT") rejectPromise(error);
|
|
3634
|
+
});
|
|
3635
|
+
zip.pipe(output);
|
|
3636
|
+
const walk = (current) => {
|
|
3637
|
+
for (const name of readdirSync2(current)) {
|
|
3638
|
+
if (name === ".DS_Store") continue;
|
|
3639
|
+
const full = join2(current, name);
|
|
3640
|
+
const stats = statSync4(full);
|
|
3641
|
+
if (stats.isDirectory()) {
|
|
3642
|
+
walk(full);
|
|
3643
|
+
} else {
|
|
3644
|
+
const rel = relative(dir, full).split(sep).join("/");
|
|
3645
|
+
zip.file(full, { name: rel, mode: stats.mode & 511 });
|
|
3646
|
+
}
|
|
3647
|
+
}
|
|
3648
|
+
};
|
|
3649
|
+
walk(dir);
|
|
3650
|
+
void zip.finalize();
|
|
3651
|
+
});
|
|
3652
|
+
}
|
|
3653
|
+
|
|
3654
|
+
// src/commands/data-integration/relay.ts
|
|
3655
|
+
var PIPELINE_VERSION = "ae-data-integration-pipeline/v1";
|
|
3656
|
+
var SHAPE_VERSION = "ae-data-integration-shape/v1";
|
|
3657
|
+
var DEFAULT_BATCH_SIZE2 = 500;
|
|
3658
|
+
var ENV_FILE = ".local/target.env";
|
|
3659
|
+
function buildPipelineDescriptor(entries, target = {}) {
|
|
3660
|
+
const first = entries[0];
|
|
3661
|
+
return {
|
|
3662
|
+
version: PIPELINE_VERSION,
|
|
3663
|
+
created_at: first?.created_at ?? (/* @__PURE__ */ new Date()).toISOString(),
|
|
3664
|
+
source: { type: "local_file", params: { format: first?.format ?? "csv" } },
|
|
3665
|
+
transform: { type: MAPPING_VERSION, refs: entries.map((entry) => entry.mapping_file) },
|
|
3666
|
+
sink: {
|
|
3667
|
+
type: "restful_sync_json",
|
|
3668
|
+
params: {
|
|
3669
|
+
batch_size: DEFAULT_BATCH_SIZE2,
|
|
3670
|
+
env_file: ENV_FILE,
|
|
3671
|
+
...target.pushurl ? { pushurl: target.pushurl } : {},
|
|
3672
|
+
...target.project_id ? { project_id: target.project_id } : {}
|
|
3673
|
+
}
|
|
3674
|
+
}
|
|
3675
|
+
};
|
|
3676
|
+
}
|
|
3677
|
+
function buildShapeBaseline(items) {
|
|
3678
|
+
return {
|
|
3679
|
+
version: SHAPE_VERSION,
|
|
3680
|
+
entries: items.map(({ mapping, fingerprint }) => ({
|
|
3681
|
+
fingerprint,
|
|
3682
|
+
mode: mapping.mode,
|
|
3683
|
+
data_set: mapping.source.data_set,
|
|
3684
|
+
format: mapping.source.format,
|
|
3685
|
+
columns: sourceColumns(mapping)
|
|
3686
|
+
}))
|
|
3687
|
+
};
|
|
3688
|
+
}
|
|
3689
|
+
function sh(...lines) {
|
|
3690
|
+
return `${lines.join("\n")}
|
|
3691
|
+
`;
|
|
3692
|
+
}
|
|
3693
|
+
function generateBinScripts() {
|
|
3694
|
+
return [
|
|
3695
|
+
{ relPath: "bin/run.sh", content: runSh(), mode: 448 },
|
|
3696
|
+
{ relPath: "bin/upload.sh", content: uploadSh(), mode: 448 },
|
|
3697
|
+
{ relPath: "bin/bind_mapping.py", content: bindMappingPy(), mode: 448 },
|
|
3698
|
+
{ relPath: "bin/summarize.py", content: summarizePy(), mode: 448 },
|
|
3699
|
+
{ relPath: "bin/plan_check.py", content: planCheckPy(), mode: 448 },
|
|
3700
|
+
{ relPath: "bin/verify.py", content: verifyPy(), mode: 448 },
|
|
3701
|
+
{ relPath: "bin/resolve_appid.py", content: resolveAppidPy(), mode: 448 }
|
|
3702
|
+
];
|
|
3703
|
+
}
|
|
3704
|
+
function runSh() {
|
|
3705
|
+
return sh(
|
|
3706
|
+
"#!/usr/bin/env bash",
|
|
3707
|
+
"# Generic pipeline executor: source -> transform -> plan. Never uploads (see upload.sh).",
|
|
3708
|
+
"# Reads pipeline.json and dispatches each stage by its `type` to ae-cli subcommands.",
|
|
3709
|
+
"set -euo pipefail",
|
|
3710
|
+
'SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"',
|
|
3711
|
+
'PKG_ROOT="$(dirname "$SCRIPT_DIR")"',
|
|
3712
|
+
'cd "$PKG_ROOT"',
|
|
3713
|
+
"",
|
|
3714
|
+
`SRC_TYPE="$(python3 -c 'import json; print(json.load(open("pipeline.json"))["source"]["type"])')"`,
|
|
3715
|
+
'case "$SRC_TYPE" in',
|
|
3716
|
+
" local_file) ;;",
|
|
3717
|
+
" *)",
|
|
3718
|
+
' echo "unsupported source type: $SRC_TYPE" >&2',
|
|
3719
|
+
' echo "re-run the full ae-data-integration pipeline (or upgrade ae-cli) for this source." >&2',
|
|
3720
|
+
" exit 64",
|
|
3721
|
+
" ;;",
|
|
3722
|
+
"esac",
|
|
3723
|
+
"",
|
|
3724
|
+
'INPUT="${1:-}"',
|
|
3725
|
+
'if [ -z "$INPUT" ]; then',
|
|
3726
|
+
" shopt -s nullglob; FILES=(inbox/*); shopt -u nullglob",
|
|
3727
|
+
' if [ "${#FILES[@]}" -ne 1 ]; then',
|
|
3728
|
+
' echo "usage: bin/run.sh <input-file>" >&2',
|
|
3729
|
+
' echo " (or put exactly one file in inbox/)" >&2',
|
|
3730
|
+
" exit 2",
|
|
3731
|
+
" fi",
|
|
3732
|
+
' INPUT="${FILES[0]}"',
|
|
3733
|
+
"fi",
|
|
3734
|
+
"",
|
|
3735
|
+
'RUN_DIR="runs/$(date +%Y%m%d-%H%M%S)"',
|
|
3736
|
+
'mkdir -p "$RUN_DIR"',
|
|
3737
|
+
'echo "run: $RUN_DIR"',
|
|
3738
|
+
"",
|
|
3739
|
+
'python3 bin/bind_mapping.py "$INPUT" "$RUN_DIR"',
|
|
3740
|
+
"",
|
|
3741
|
+
"while IFS= read -r ref; do",
|
|
3742
|
+
' ref_dir="$(dirname "$ref")"',
|
|
3743
|
+
' echo "convert: $ref_dir"',
|
|
3744
|
+
" ae-cli data-integration convert \\",
|
|
3745
|
+
' --input-file "$INPUT" \\',
|
|
3746
|
+
' --mapping "$RUN_DIR/bound/$ref_dir/mapping.json" \\',
|
|
3747
|
+
' --output-dir "$RUN_DIR/$ref_dir" || exit $?',
|
|
3748
|
+
`done < <(python3 -c 'import json,sys; print("\\n".join(json.load(open("pipeline.json"))["transform"]["refs"]))')`,
|
|
3749
|
+
"",
|
|
3750
|
+
'python3 bin/summarize.py "$RUN_DIR"',
|
|
3751
|
+
'python3 bin/plan_check.py "$RUN_DIR"',
|
|
3752
|
+
"",
|
|
3753
|
+
"# Salvage hint: quarantined rows are never silently dropped. Re-process only",
|
|
3754
|
+
"# them against the fixed mapping instead of re-uploading the whole file.",
|
|
3755
|
+
"while IFS= read -r ref; do",
|
|
3756
|
+
' ref_dir="$(dirname "$ref")"',
|
|
3757
|
+
' inv="$RUN_DIR/$ref_dir/invalid.rows.jsonl"',
|
|
3758
|
+
' if [ -s "$inv" ]; then',
|
|
3759
|
+
' echo "note: $inv has quarantined rows \u2014 salvage them with:"',
|
|
3760
|
+
' echo " ae-cli data-integration convert --input-file "$INPUT" --mapping "$RUN_DIR/bound/$ref_dir/mapping.json" --salvage-from "$inv" --output-dir "$RUN_DIR-salvage""',
|
|
3761
|
+
" fi",
|
|
3762
|
+
`done < <(python3 -c 'import json,sys; print("\\n".join(json.load(open("pipeline.json"))["transform"]["refs"]))')`,
|
|
3763
|
+
"",
|
|
3764
|
+
'echo "done. review the summary and plan check, then run: bin/upload.sh $RUN_DIR"'
|
|
3765
|
+
);
|
|
3766
|
+
}
|
|
3767
|
+
function uploadSh() {
|
|
3768
|
+
return sh(
|
|
3769
|
+
"#!/usr/bin/env bash",
|
|
3770
|
+
"# Sink executor. Dry-run by default; --confirm actually uploads.",
|
|
3771
|
+
"# Resolves the recorded target from pipeline.json (pushurl + project_id), derives",
|
|
3772
|
+
"# the APPID via `ae-cli project info get` (bin/resolve_appid.py), and falls back to",
|
|
3773
|
+
"# .local/target.env for explicit APPID / endpoint / project-id overrides.",
|
|
3774
|
+
"set -euo pipefail",
|
|
3775
|
+
'SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"',
|
|
3776
|
+
'PKG_ROOT="$(dirname "$SCRIPT_DIR")"',
|
|
3777
|
+
'cd "$PKG_ROOT"',
|
|
3778
|
+
"",
|
|
3779
|
+
`SINK_TYPE="$(python3 -c 'import json; print(json.load(open("pipeline.json"))["sink"]["type"])')"`,
|
|
3780
|
+
'case "$SINK_TYPE" in',
|
|
3781
|
+
" restful_sync_json) ;;",
|
|
3782
|
+
" *)",
|
|
3783
|
+
' echo "unsupported sink type: $SINK_TYPE" >&2',
|
|
3784
|
+
' echo "re-run the full ae-data-integration pipeline (or upgrade ae-cli) for this sink." >&2',
|
|
3785
|
+
" exit 64",
|
|
3786
|
+
" ;;",
|
|
3787
|
+
"esac",
|
|
3788
|
+
"",
|
|
3789
|
+
`SINK_PARAMS="$(python3 -c 'import json; print(json.dumps(json.load(open("pipeline.json"))["sink"]["params"]))')"`,
|
|
3790
|
+
`BATCH_SIZE="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("batch_size", 500))' "$SINK_PARAMS")"`,
|
|
3791
|
+
`ENV_FILE="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("env_file", ".local/target.env"))' "$SINK_PARAMS")"`,
|
|
3792
|
+
`PUSHURL="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("pushurl") or "")' "$SINK_PARAMS")"`,
|
|
3793
|
+
`PROJECT_ID="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("project_id") or "")' "$SINK_PARAMS")"`,
|
|
3794
|
+
"",
|
|
3795
|
+
"CONFIRM=0",
|
|
3796
|
+
'RUN_DIR=""',
|
|
3797
|
+
'for arg in "$@"; do',
|
|
3798
|
+
' case "$arg" in',
|
|
3799
|
+
" --confirm) CONFIRM=1 ;;",
|
|
3800
|
+
' -*) echo "unknown flag: $arg" >&2; exit 2 ;;',
|
|
3801
|
+
' *) RUN_DIR="$arg" ;;',
|
|
3802
|
+
" esac",
|
|
3803
|
+
"done",
|
|
3804
|
+
"",
|
|
3805
|
+
'if [ -z "$RUN_DIR" ]; then',
|
|
3806
|
+
' echo "usage: bin/upload.sh [--confirm] <runs/<run-id>>" >&2',
|
|
3807
|
+
" exit 2",
|
|
3808
|
+
"fi",
|
|
3809
|
+
"",
|
|
3810
|
+
"# .local/target.env is optional: the package may record the target itself.",
|
|
3811
|
+
"# Env values still win as explicit overrides (the documented fallback).",
|
|
3812
|
+
'if [ -f "$ENV_FILE" ]; then set -a; . "$ENV_FILE"; set +a; fi',
|
|
3813
|
+
'[ -z "$PROJECT_ID" ] && PROJECT_ID="${AE_PROJECT_ID:-}"',
|
|
3814
|
+
"",
|
|
3815
|
+
"# Endpoint: the recorded pushurl (a receiver base URL; append /sync_json), else AE_ENDPOINT.",
|
|
3816
|
+
'if [ -n "$PUSHURL" ]; then',
|
|
3817
|
+
' case "$PUSHURL" in',
|
|
3818
|
+
' */sync_json) ENDPOINT="$PUSHURL" ;;',
|
|
3819
|
+
' *) ENDPOINT="${PUSHURL%/}/sync_json" ;;',
|
|
3820
|
+
" esac",
|
|
3821
|
+
"else",
|
|
3822
|
+
' : "${AE_ENDPOINT:?set AE_ENDPOINT in .local/target.env (or record pushurl at handoff)}"',
|
|
3823
|
+
' ENDPOINT="$AE_ENDPOINT"',
|
|
3824
|
+
"fi",
|
|
3825
|
+
"",
|
|
3826
|
+
"# APPID: explicit env wins, else derive from the recorded project via project info get.",
|
|
3827
|
+
'APPID="${AE_APPID:-}"',
|
|
3828
|
+
'if [ -z "$APPID" ] && [ -n "$PROJECT_ID" ]; then',
|
|
3829
|
+
' APPID="$(python3 bin/resolve_appid.py "$PROJECT_ID")"',
|
|
3830
|
+
"fi",
|
|
3831
|
+
': "${APPID:?set AE_APPID in .local/target.env (or record project_id at handoff)}"',
|
|
3832
|
+
"",
|
|
3833
|
+
"# Mask the APPID in anything this script prints \u2014 it must never land in logs.",
|
|
3834
|
+
"mask() {",
|
|
3835
|
+
' local a="$1"',
|
|
3836
|
+
' if [ "${#a}" -le 4 ]; then printf "%s" "****"; return; fi',
|
|
3837
|
+
' printf "%s%s" "$(printf "%*s" "$(( ${#a} - 4 ))" "" | tr " " "*")" "${a: -4}"',
|
|
3838
|
+
"}",
|
|
3839
|
+
"display_args() {",
|
|
3840
|
+
' local args=("$@") out=() i',
|
|
3841
|
+
" for ((i=0; i<${#args[@]}; i++)); do",
|
|
3842
|
+
' if [ "${args[$i]}" = "--appid" ] && [ -n "${args[$((i+1))]:-}" ]; then',
|
|
3843
|
+
' out+=("--appid" "$(mask "${args[$((i+1))]}")")',
|
|
3844
|
+
" i=$((i+1))",
|
|
3845
|
+
" else",
|
|
3846
|
+
' out+=("${args[$i]}")',
|
|
3847
|
+
" fi",
|
|
3848
|
+
" done",
|
|
3849
|
+
' printf "%s\\n" "${out[*]}"',
|
|
3850
|
+
"}",
|
|
3851
|
+
"",
|
|
3852
|
+
'echo "target: project_id=${PROJECT_ID:-<unset>}"',
|
|
3853
|
+
'echo " endpoint=$ENDPOINT"',
|
|
3854
|
+
'echo " appid=$(mask "$APPID")"',
|
|
3855
|
+
'if [ "$CONFIRM" -eq 0 ]; then',
|
|
3856
|
+
' echo "dry-run \u2014 re-run with --confirm to upload to this address and project."',
|
|
3857
|
+
"else",
|
|
3858
|
+
' echo "confirmed: uploading to the address and project shown above."',
|
|
3859
|
+
"fi",
|
|
3860
|
+
"",
|
|
3861
|
+
'FLAGS=(--endpoint "$ENDPOINT" --appid "$APPID" --batch-size "$BATCH_SIZE")',
|
|
3862
|
+
'if [ "$CONFIRM" -eq 0 ]; then FLAGS+=(--dry-run); fi',
|
|
3863
|
+
"",
|
|
3864
|
+
"while IFS= read -r ref; do",
|
|
3865
|
+
' ref_dir="$(dirname "$ref")"',
|
|
3866
|
+
' ue="$RUN_DIR/$ref_dir/valid.ue.jsonl"',
|
|
3867
|
+
' manifest="$RUN_DIR/$ref_dir/manifest.json"',
|
|
3868
|
+
' if [ ! -f "$ue" ]; then',
|
|
3869
|
+
' echo "missing $ue (run bin/run.sh first)" >&2',
|
|
3870
|
+
" exit 2",
|
|
3871
|
+
" fi",
|
|
3872
|
+
` status="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["status"])' "$manifest")"`,
|
|
3873
|
+
' args=("${FLAGS[@]}" --ue-file "$ue" --manifest "$manifest")',
|
|
3874
|
+
' if [ "$status" = "blocked" ]; then',
|
|
3875
|
+
' echo "manifest $manifest is blocked (rows were quarantined)."',
|
|
3876
|
+
' echo "Uploading passes only the valid subset (--allow-clean-subset); quarantined rows stay in invalid.rows.jsonl."',
|
|
3877
|
+
' if [ "$CONFIRM" -eq 1 ]; then args+=(--allow-clean-subset); else echo " (dry-run) re-run with --confirm to accept the clean subset."; fi',
|
|
3878
|
+
" fi",
|
|
3879
|
+
' echo "> ae-cli data-integration upload $(display_args "${args[@]}")"',
|
|
3880
|
+
' ae-cli data-integration upload "${args[@]}" || exit $?',
|
|
3881
|
+
`done < <(python3 -c 'import json,sys; print("\\n".join(json.load(open("pipeline.json"))["transform"]["refs"]))')`
|
|
3882
|
+
);
|
|
3883
|
+
}
|
|
3884
|
+
function bindMappingPy() {
|
|
3885
|
+
return `#!/usr/bin/env python3
|
|
3886
|
+
"""Source stage: rebind the frozen mappings to a new same-shape file.
|
|
3887
|
+
|
|
3888
|
+
Runs \`ae-cli data-integration inspect\` once, then for every mapping the
|
|
3889
|
+
pipeline references (pipeline.json's transform.refs) re-binds the frozen
|
|
3890
|
+
mapping's \`source.sha256\` and \`source.data_set\` to the new file (identity
|
|
3891
|
+
fields only \u2014 business logic is untouched), after checking the column set
|
|
3892
|
+
against shape.json. Historical index entries the pipeline does not run are
|
|
3893
|
+
left alone \u2014 the index accumulates across handoffs in the same directory.
|
|
3894
|
+
|
|
3895
|
+
Usage: bin/bind_mapping.py <input-file> <run-dir>
|
|
3896
|
+
"""
|
|
3897
|
+
import json
|
|
3898
|
+
import os
|
|
3899
|
+
import subprocess
|
|
3900
|
+
import sys
|
|
3901
|
+
|
|
3902
|
+
|
|
3903
|
+
def fail(message):
|
|
3904
|
+
print(f"bind_mapping: {message}", file=sys.stderr)
|
|
3905
|
+
sys.exit(1)
|
|
3906
|
+
|
|
3907
|
+
|
|
3908
|
+
def pkg_root():
|
|
3909
|
+
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
3910
|
+
|
|
3911
|
+
|
|
3912
|
+
def load_json(path):
|
|
3913
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
3914
|
+
return json.load(f)
|
|
3915
|
+
|
|
3916
|
+
|
|
3917
|
+
def write_json(path, value):
|
|
3918
|
+
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
3919
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
3920
|
+
json.dump(value, f, ensure_ascii=False, indent=2)
|
|
3921
|
+
f.write("\\n")
|
|
3922
|
+
|
|
3923
|
+
|
|
3924
|
+
def dataset_key(dataset):
|
|
3925
|
+
return dataset.get("id") or dataset.get("label") or ""
|
|
3926
|
+
|
|
3927
|
+
|
|
3928
|
+
def main():
|
|
3929
|
+
if len(sys.argv) != 3:
|
|
3930
|
+
fail("usage: bind_mapping.py <input-file> <run-dir>")
|
|
3931
|
+
input_file, run_dir = sys.argv[1], sys.argv[2]
|
|
3932
|
+
root = pkg_root()
|
|
3933
|
+
|
|
3934
|
+
pipeline = load_json(os.path.join(root, "pipeline.json"))
|
|
3935
|
+
if pipeline["source"]["type"] != "local_file":
|
|
3936
|
+
fail(f"unsupported source type: {pipeline['source']['type']}")
|
|
3937
|
+
|
|
3938
|
+
index = load_json(os.path.join(root, "index.json"))
|
|
3939
|
+
shape = load_json(os.path.join(root, "shape.json"))
|
|
3940
|
+
shape_by_fp = {entry["fingerprint"]: entry for entry in shape["entries"]}
|
|
3941
|
+
index_by_ref = {entry["mapping_file"]: entry for entry in index["entries"]}
|
|
3942
|
+
|
|
3943
|
+
inspect = run_inspect(input_file)
|
|
3944
|
+
datasets = extract_datasets(inspect)
|
|
3945
|
+
headers_by_dataset = extract_headers(inspect, datasets)
|
|
3946
|
+
sha = (inspect.get("source") or {}).get("sha256")
|
|
3947
|
+
|
|
3948
|
+
# Rebind only the mappings this pipeline runs (transform.refs). The index
|
|
3949
|
+
# accumulates entries across handoffs in the same directory; earlier entries
|
|
3950
|
+
# may have no shape baseline here and are never converted by run.sh, so
|
|
3951
|
+
# walking the whole index would fail on them.
|
|
3952
|
+
for ref in pipeline["transform"]["refs"]:
|
|
3953
|
+
entry = index_by_ref.get(ref)
|
|
3954
|
+
if entry is None:
|
|
3955
|
+
fail(f"index entry missing for {ref}; re-run the full pipeline")
|
|
3956
|
+
fingerprint = entry["fingerprint"]
|
|
3957
|
+
baseline = shape_by_fp.get(fingerprint)
|
|
3958
|
+
if baseline is None:
|
|
3959
|
+
fail(f"shape baseline missing for {fingerprint}; re-run the full pipeline")
|
|
3960
|
+
frozen = load_json(os.path.join(root, ref))
|
|
3961
|
+
data_set_id = match_dataset(frozen, baseline, datasets, headers_by_dataset)
|
|
3962
|
+
validate_headers(data_set_id, baseline, headers_by_dataset)
|
|
3963
|
+
frozen["source"]["sha256"] = sha
|
|
3964
|
+
frozen["source"]["data_set"] = data_set_id
|
|
3965
|
+
out = os.path.join(root, run_dir, "bound", os.path.dirname(ref), "mapping.json")
|
|
3966
|
+
write_json(out, frozen)
|
|
3967
|
+
print(f"rebound {os.path.dirname(ref)} -> data_set {data_set_id!r}")
|
|
3968
|
+
print("shape check passed")
|
|
3969
|
+
|
|
3970
|
+
|
|
3971
|
+
def run_inspect(input_file):
|
|
3972
|
+
proc = subprocess.run(
|
|
3973
|
+
["ae-cli", "data-integration", "inspect", "--input-file", input_file],
|
|
3974
|
+
capture_output=True, text=True,
|
|
3975
|
+
)
|
|
3976
|
+
if proc.returncode != 0:
|
|
3977
|
+
fail(f"inspect failed: {proc.stderr.strip()}")
|
|
3978
|
+
try:
|
|
3979
|
+
parsed = json.loads(proc.stdout)
|
|
3980
|
+
except json.JSONDecodeError:
|
|
3981
|
+
fail("inspect returned non-JSON output")
|
|
3982
|
+
# ae-cli wraps every command result in { ok, data, error }; unwrap it.
|
|
3983
|
+
data = parsed.get("data") if isinstance(parsed, dict) else None
|
|
3984
|
+
if not isinstance(data, dict):
|
|
3985
|
+
fail("inspect returned no data payload")
|
|
3986
|
+
return data
|
|
3987
|
+
|
|
3988
|
+
|
|
3989
|
+
def extract_datasets(inspect):
|
|
3990
|
+
if inspect.get("selection_required"):
|
|
3991
|
+
return inspect.get("data_sets") or []
|
|
3992
|
+
data_set = inspect.get("data_set")
|
|
3993
|
+
return [data_set] if data_set else []
|
|
3994
|
+
|
|
3995
|
+
|
|
3996
|
+
def extract_headers(inspect, datasets):
|
|
3997
|
+
result = {}
|
|
3998
|
+
details = inspect.get("header_details")
|
|
3999
|
+
if details:
|
|
4000
|
+
for dataset in datasets:
|
|
4001
|
+
names = (dataset.get("label"), dataset.get("id"), dataset.get("selector"))
|
|
4002
|
+
for sheet in details:
|
|
4003
|
+
if sheet.get("name") in names:
|
|
4004
|
+
result[dataset_key(dataset)] = sheet.get("headers") or []
|
|
4005
|
+
break
|
|
4006
|
+
return result
|
|
4007
|
+
columns = inspect.get("columns")
|
|
4008
|
+
if columns and datasets:
|
|
4009
|
+
result[dataset_key(datasets[0])] = [column["name"] for column in columns]
|
|
4010
|
+
return result
|
|
4011
|
+
|
|
4012
|
+
|
|
4013
|
+
def match_dataset(frozen, baseline, datasets, headers_by_dataset):
|
|
4014
|
+
if not datasets:
|
|
4015
|
+
fail("inspect reported no data sets")
|
|
4016
|
+
wanted = frozen["source"]["data_set"]
|
|
4017
|
+
for dataset in datasets:
|
|
4018
|
+
if dataset.get("id") == wanted:
|
|
4019
|
+
return dataset.get("id")
|
|
4020
|
+
for dataset in datasets:
|
|
4021
|
+
if dataset.get("label") == wanted:
|
|
4022
|
+
return dataset.get("id")
|
|
4023
|
+
baseline_cols = set(baseline.get("columns") or [])
|
|
4024
|
+
if baseline_cols:
|
|
4025
|
+
for dataset in datasets:
|
|
4026
|
+
headers = headers_by_dataset.get(dataset_key(dataset))
|
|
4027
|
+
if headers and set(headers) == baseline_cols:
|
|
4028
|
+
return dataset.get("id")
|
|
4029
|
+
if len(datasets) == 1:
|
|
4030
|
+
return datasets[0].get("id")
|
|
4031
|
+
fail(f"cannot rebind data_set {wanted!r}: no exact or header match; re-run the full pipeline")
|
|
4032
|
+
|
|
4033
|
+
|
|
4034
|
+
def validate_headers(data_set_id, baseline, headers_by_dataset):
|
|
4035
|
+
baseline_cols = set(baseline.get("columns") or [])
|
|
4036
|
+
if not baseline_cols:
|
|
4037
|
+
return
|
|
4038
|
+
headers = headers_by_dataset.get(data_set_id)
|
|
4039
|
+
if headers is None:
|
|
4040
|
+
return
|
|
4041
|
+
if set(headers) != baseline_cols:
|
|
4042
|
+
missing = sorted(baseline_cols - set(headers))
|
|
4043
|
+
extra = sorted(set(headers) - baseline_cols)
|
|
4044
|
+
fail(
|
|
4045
|
+
f"shape mismatch for {data_set_id!r}: missing={missing} extra={extra} \u2014 "
|
|
4046
|
+
"re-run the full pipeline; do not edit the frozen mapping"
|
|
4047
|
+
)
|
|
4048
|
+
|
|
4049
|
+
|
|
4050
|
+
if __name__ == "__main__":
|
|
4051
|
+
main()
|
|
4052
|
+
`;
|
|
4053
|
+
}
|
|
4054
|
+
function summarizePy() {
|
|
4055
|
+
return `#!/usr/bin/env python3
|
|
4056
|
+
"""Transform stage summary: print valid/quarantined counts per data set."""
|
|
4057
|
+
import json
|
|
4058
|
+
import os
|
|
4059
|
+
import sys
|
|
4060
|
+
|
|
4061
|
+
|
|
4062
|
+
def pkg_root():
|
|
4063
|
+
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
4064
|
+
|
|
4065
|
+
|
|
4066
|
+
def main():
|
|
4067
|
+
if len(sys.argv) != 2:
|
|
4068
|
+
print("usage: summarize.py <run-dir>", file=sys.stderr)
|
|
4069
|
+
sys.exit(2)
|
|
4070
|
+
run_dir = sys.argv[1]
|
|
4071
|
+
root = pkg_root()
|
|
4072
|
+
with open(os.path.join(root, "pipeline.json"), "r", encoding="utf-8") as f:
|
|
4073
|
+
pipeline = json.load(f)
|
|
4074
|
+
total_valid = 0
|
|
4075
|
+
total_invalid = 0
|
|
4076
|
+
for ref in pipeline["transform"]["refs"]:
|
|
4077
|
+
ref_dir = os.path.dirname(ref)
|
|
4078
|
+
manifest_path = os.path.join(root, run_dir, ref_dir, "manifest.json")
|
|
4079
|
+
if not os.path.exists(manifest_path):
|
|
4080
|
+
continue
|
|
4081
|
+
with open(manifest_path, "r", encoding="utf-8") as f:
|
|
4082
|
+
manifest = json.load(f)
|
|
4083
|
+
output = manifest["output"]
|
|
4084
|
+
total_valid += output["valid_records"]
|
|
4085
|
+
total_invalid += output["invalid_records"]
|
|
4086
|
+
print(f"{ref_dir}: {output['valid_records']} valid / {output['invalid_records']} quarantined")
|
|
4087
|
+
for reason in manifest.get("blocked_reasons") or []:
|
|
4088
|
+
print(f" - {reason}")
|
|
4089
|
+
print(f"total: {total_valid} valid / {total_invalid} quarantined")
|
|
4090
|
+
|
|
4091
|
+
|
|
4092
|
+
if __name__ == "__main__":
|
|
4093
|
+
main()
|
|
4094
|
+
`;
|
|
4095
|
+
}
|
|
4096
|
+
function planCheckPy() {
|
|
4097
|
+
return `#!/usr/bin/env python3
|
|
4098
|
+
"""Plan gate: every event and property produced must already exist in the frozen tracking plan."""
|
|
4099
|
+
import json
|
|
4100
|
+
import os
|
|
4101
|
+
import sys
|
|
4102
|
+
|
|
4103
|
+
|
|
4104
|
+
def pkg_root():
|
|
4105
|
+
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
4106
|
+
|
|
4107
|
+
|
|
4108
|
+
def load_json(path):
|
|
4109
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
4110
|
+
return json.load(f)
|
|
4111
|
+
|
|
4112
|
+
|
|
4113
|
+
def main():
|
|
4114
|
+
if len(sys.argv) != 2:
|
|
4115
|
+
print("usage: plan_check.py <run-dir>", file=sys.stderr)
|
|
4116
|
+
sys.exit(2)
|
|
4117
|
+
run_dir = sys.argv[1]
|
|
4118
|
+
root = pkg_root()
|
|
4119
|
+
pipeline = load_json(os.path.join(root, "pipeline.json"))
|
|
4120
|
+
new_events = []
|
|
4121
|
+
new_properties = []
|
|
4122
|
+
missing_plans = []
|
|
4123
|
+
for ref in pipeline["transform"]["refs"]:
|
|
4124
|
+
ref_dir = os.path.dirname(ref)
|
|
4125
|
+
plan_path = os.path.join(root, ref_dir, "plan.json")
|
|
4126
|
+
ue_path = os.path.join(root, run_dir, ref_dir, "valid.ue.jsonl")
|
|
4127
|
+
if not os.path.exists(plan_path):
|
|
4128
|
+
missing_plans.append(ref_dir)
|
|
4129
|
+
continue
|
|
4130
|
+
plan = load_json(plan_path)
|
|
4131
|
+
plan_events = {event["event_name"] for event in plan.get("events", [])}
|
|
4132
|
+
plan_properties = {prop["name"] for prop in plan.get("event_properties", [])}
|
|
4133
|
+
plan_properties |= {prop["name"] for prop in plan.get("common_event_properties", [])}
|
|
4134
|
+
plan_properties |= {prop["name"] for prop in plan.get("user_properties", [])}
|
|
4135
|
+
produced_events = set()
|
|
4136
|
+
produced_properties = set()
|
|
4137
|
+
if os.path.exists(ue_path):
|
|
4138
|
+
with open(ue_path, "r", encoding="utf-8") as f:
|
|
4139
|
+
for line in f:
|
|
4140
|
+
line = line.strip()
|
|
4141
|
+
if not line:
|
|
4142
|
+
continue
|
|
4143
|
+
record = json.loads(line)
|
|
4144
|
+
if record.get("#event_name"):
|
|
4145
|
+
produced_events.add(record["#event_name"])
|
|
4146
|
+
props = record.get("properties")
|
|
4147
|
+
if isinstance(props, dict):
|
|
4148
|
+
for key in props:
|
|
4149
|
+
if key.startswith("#"):
|
|
4150
|
+
continue
|
|
4151
|
+
produced_properties.add(key)
|
|
4152
|
+
new_events.extend(sorted(produced_events - plan_events))
|
|
4153
|
+
new_properties.extend(sorted(produced_properties - plan_properties))
|
|
4154
|
+
if missing_plans:
|
|
4155
|
+
print("no plan.json in package for: " + ", ".join(missing_plans), file=sys.stderr)
|
|
4156
|
+
print("run the Tracking plan step first \u2014 the plan gate cannot be skipped", file=sys.stderr)
|
|
4157
|
+
sys.exit(3)
|
|
4158
|
+
if new_events:
|
|
4159
|
+
print("new events not in the plan: " + ", ".join(new_events), file=sys.stderr)
|
|
4160
|
+
sys.exit(3)
|
|
4161
|
+
if new_properties:
|
|
4162
|
+
print("new properties not in the plan: " + ", ".join(new_properties), file=sys.stderr)
|
|
4163
|
+
sys.exit(3)
|
|
4164
|
+
print("plan coverage ok")
|
|
4165
|
+
|
|
4166
|
+
|
|
4167
|
+
if __name__ == "__main__":
|
|
4168
|
+
main()
|
|
4169
|
+
`;
|
|
4170
|
+
}
|
|
4171
|
+
function verifyPy() {
|
|
4172
|
+
return `#!/usr/bin/env python3
|
|
4173
|
+
"""Persistence consistency check: submit-window counts vs the platform summary.
|
|
4174
|
+
|
|
4175
|
+
A soft check, not a hard gate. It computes what this run submitted from the local
|
|
4176
|
+
UE output (knowable), snapshots \`ae-cli tracking ingest summary\` over the submit
|
|
4177
|
+
window before and after upload, and prints both next to the expected counts. It
|
|
4178
|
+
does NOT parse the summary payload into per-event numbers: the capability's data
|
|
4179
|
+
shape is server-defined and not a stable CLI contract, and a shared project cannot
|
|
4180
|
+
attribute the window delta to this import alone. For a hard per-event SQL judge,
|
|
4181
|
+
overlay a project custom layer (see custom-layer.md in the ae-data-integration skill).
|
|
4182
|
+
|
|
4183
|
+
verify.py <run-dir> --baseline snapshot the summary before upload
|
|
4184
|
+
verify.py <run-dir> --check snapshot again and diff against the baseline
|
|
4185
|
+
"""
|
|
4186
|
+
import json
|
|
4187
|
+
import os
|
|
4188
|
+
import subprocess
|
|
4189
|
+
import sys
|
|
4190
|
+
|
|
4191
|
+
|
|
4192
|
+
def pkg_root():
|
|
4193
|
+
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
4194
|
+
|
|
4195
|
+
|
|
4196
|
+
def load_json(path):
|
|
4197
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
4198
|
+
return json.load(f)
|
|
4199
|
+
|
|
4200
|
+
|
|
4201
|
+
def visit_window(record, lo, hi):
|
|
4202
|
+
raw = record.get("#time")
|
|
4203
|
+
if not isinstance(raw, str) or len(raw) < 19:
|
|
4204
|
+
return lo, hi
|
|
4205
|
+
stamp = raw[:19] # YYYY-MM-DD HH:mm:ss
|
|
4206
|
+
if lo is None or stamp < lo:
|
|
4207
|
+
lo = stamp
|
|
4208
|
+
if hi is None or stamp > hi:
|
|
4209
|
+
hi = stamp
|
|
4210
|
+
return lo, hi
|
|
4211
|
+
|
|
4212
|
+
|
|
4213
|
+
def main():
|
|
4214
|
+
if len(sys.argv) < 2:
|
|
4215
|
+
print("usage: verify.py <run-dir> [--baseline|--check]", file=sys.stderr)
|
|
4216
|
+
sys.exit(2)
|
|
4217
|
+
run_dir = sys.argv[1]
|
|
4218
|
+
mode = sys.argv[2] if len(sys.argv) > 2 else "--check"
|
|
4219
|
+
if mode not in ("--baseline", "--check"):
|
|
4220
|
+
print("usage: verify.py <run-dir> [--baseline|--check]", file=sys.stderr)
|
|
4221
|
+
sys.exit(2)
|
|
4222
|
+
root = pkg_root()
|
|
4223
|
+
pipeline = load_json(os.path.join(root, "pipeline.json"))
|
|
4224
|
+
project_id = pipeline["sink"]["params"].get("project_id") or os.environ.get("AE_PROJECT_ID") or "1"
|
|
4225
|
+
|
|
4226
|
+
expected_events = {}
|
|
4227
|
+
expected_users = 0
|
|
4228
|
+
lo = hi = None
|
|
4229
|
+
for ref in pipeline["transform"]["refs"]:
|
|
4230
|
+
ue = os.path.join(root, run_dir, os.path.dirname(ref), "valid.ue.jsonl")
|
|
4231
|
+
if not os.path.exists(ue):
|
|
4232
|
+
continue
|
|
4233
|
+
with open(ue, "r", encoding="utf-8") as f:
|
|
4234
|
+
for line in f:
|
|
4235
|
+
line = line.strip()
|
|
4236
|
+
if not line:
|
|
4237
|
+
continue
|
|
4238
|
+
record = json.loads(line)
|
|
4239
|
+
if record.get("#type") == "track" and record.get("#event_name"):
|
|
4240
|
+
expected_events[record["#event_name"]] = expected_events.get(record["#event_name"], 0) + 1
|
|
4241
|
+
else:
|
|
4242
|
+
expected_users += 1
|
|
4243
|
+
lo, hi = visit_window(record, lo, hi)
|
|
4244
|
+
|
|
4245
|
+
if lo is None:
|
|
4246
|
+
print("verify: no UE records found in " + run_dir, file=sys.stderr)
|
|
4247
|
+
sys.exit(2)
|
|
4248
|
+
|
|
4249
|
+
def run_summary():
|
|
4250
|
+
proc = subprocess.run(
|
|
4251
|
+
["ae-cli", "tracking", "ingest", "summary",
|
|
4252
|
+
"-p", str(project_id), "--start-time", lo, "--end-time", hi],
|
|
4253
|
+
capture_output=True, text=True,
|
|
4254
|
+
)
|
|
4255
|
+
if proc.returncode != 0:
|
|
4256
|
+
return {"error": (proc.stderr or proc.stdout).strip()[:500]}
|
|
4257
|
+
try:
|
|
4258
|
+
return json.loads(proc.stdout)
|
|
4259
|
+
except json.JSONDecodeError:
|
|
4260
|
+
return {"raw": proc.stdout.strip()[:500]}
|
|
4261
|
+
|
|
4262
|
+
print("window " + lo + " .. " + hi + " project_id=" + str(project_id))
|
|
4263
|
+
print("expected (this run):")
|
|
4264
|
+
for event, count in sorted(expected_events.items()):
|
|
4265
|
+
print(" {:<20}{:>10,}".format(event, count))
|
|
4266
|
+
print(" {:<20}{:>10,}".format("<user rows>", expected_users))
|
|
4267
|
+
print()
|
|
4268
|
+
|
|
4269
|
+
baseline_path = os.path.join(root, run_dir, "baseline.json")
|
|
4270
|
+
if mode == "--baseline":
|
|
4271
|
+
payload = run_summary()
|
|
4272
|
+
with open(baseline_path, "w", encoding="utf-8") as f:
|
|
4273
|
+
json.dump(payload, f, ensure_ascii=False, indent=2)
|
|
4274
|
+
print("baseline recorded: " + baseline_path)
|
|
4275
|
+
print("platform summary (before):")
|
|
4276
|
+
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
4277
|
+
sys.exit(0)
|
|
4278
|
+
|
|
4279
|
+
if not os.path.exists(baseline_path):
|
|
4280
|
+
print("no baseline.json \u2014 run \`verify.py <run-dir> --baseline\` before upload.", file=sys.stderr)
|
|
4281
|
+
print("falling back to a single after-upload summary:", file=sys.stderr)
|
|
4282
|
+
print(json.dumps(run_summary(), ensure_ascii=False, indent=2))
|
|
4283
|
+
sys.exit(1)
|
|
4284
|
+
|
|
4285
|
+
with open(baseline_path, "r", encoding="utf-8") as f:
|
|
4286
|
+
baseline = json.load(f)
|
|
4287
|
+
after = run_summary()
|
|
4288
|
+
print("platform summary before vs after:")
|
|
4289
|
+
print(json.dumps(baseline, ensure_ascii=False, indent=2))
|
|
4290
|
+
print("---")
|
|
4291
|
+
print(json.dumps(after, ensure_ascii=False, indent=2))
|
|
4292
|
+
print()
|
|
4293
|
+
print("summary changed: " + ("yes" if baseline != after else "no"))
|
|
4294
|
+
print()
|
|
4295
|
+
print("Boundary: the summary payload is server-defined and may under-report even")
|
|
4296
|
+
print("landed data; this check surfaces it for comparison, it does not auto-verify")
|
|
4297
|
+
print("per-event counts, and a shared project's window delta is not attributed to")
|
|
4298
|
+
print("this import. Cross-check with:")
|
|
4299
|
+
print(" ae-cli tracking live-data list -p " + str(project_id))
|
|
4300
|
+
print("For a hard SQL judge, add a project custom layer (custom-layer.md).")
|
|
4301
|
+
sys.exit(0)
|
|
4302
|
+
|
|
4303
|
+
|
|
4304
|
+
if __name__ == "__main__":
|
|
4305
|
+
main()
|
|
4306
|
+
`;
|
|
4307
|
+
}
|
|
4308
|
+
function resolveAppidPy() {
|
|
4309
|
+
return `#!/usr/bin/env python3
|
|
4310
|
+
"""Derive the destination APPID from \`ae-cli project info get --project-id <id>\`.
|
|
4311
|
+
|
|
4312
|
+
\`project info get\` returns \`data.appid\` at the top level (verified against the
|
|
4313
|
+
AE demo host), so this helper reads that exact field and prints it to stdout. It
|
|
4314
|
+
prints nothing to stdout and reports the payload when the field is absent or not
|
|
4315
|
+
a non-empty string \u2014 the caller then falls back to AE_APPID.
|
|
4316
|
+
|
|
4317
|
+
Usage: bin/resolve_appid.py <project-id>
|
|
4318
|
+
"""
|
|
4319
|
+
import json
|
|
4320
|
+
import subprocess
|
|
4321
|
+
import sys
|
|
4322
|
+
|
|
4323
|
+
|
|
4324
|
+
def main():
|
|
4325
|
+
if len(sys.argv) != 2:
|
|
4326
|
+
print("usage: resolve_appid.py <project-id>", file=sys.stderr)
|
|
4327
|
+
sys.exit(2)
|
|
4328
|
+
project_id = sys.argv[1]
|
|
4329
|
+
proc = subprocess.run(
|
|
4330
|
+
["ae-cli", "project", "info", "get", "--project-id", project_id],
|
|
4331
|
+
capture_output=True, text=True,
|
|
4332
|
+
)
|
|
4333
|
+
if proc.returncode != 0:
|
|
4334
|
+
print("resolve_appid: project info get failed: " + (proc.stderr or proc.stdout).strip()[:300], file=sys.stderr)
|
|
4335
|
+
sys.exit(0)
|
|
4336
|
+
try:
|
|
4337
|
+
parsed = json.loads(proc.stdout)
|
|
4338
|
+
except json.JSONDecodeError:
|
|
4339
|
+
print("resolve_appid: project info get returned non-JSON output", file=sys.stderr)
|
|
4340
|
+
sys.exit(0)
|
|
4341
|
+
data = parsed.get("data") if isinstance(parsed, dict) else None
|
|
4342
|
+
appid = data.get("appid") if isinstance(data, dict) else None
|
|
4343
|
+
if not isinstance(appid, str) or not appid:
|
|
4344
|
+
print("resolve_appid: project info get returned no appid; set AE_APPID", file=sys.stderr)
|
|
4345
|
+
print(json.dumps(data, ensure_ascii=False, indent=2) if data is not None else "{}", file=sys.stderr)
|
|
4346
|
+
sys.exit(0)
|
|
4347
|
+
masked = appid if len(appid) <= 4 else ("*" * (len(appid) - 4)) + appid[-4:]
|
|
4348
|
+
print("resolve_appid: resolved APPID data.appid = " + masked, file=sys.stderr)
|
|
4349
|
+
print(appid)
|
|
4350
|
+
|
|
4351
|
+
|
|
4352
|
+
if __name__ == "__main__":
|
|
4353
|
+
main()
|
|
4354
|
+
`;
|
|
4355
|
+
}
|
|
4356
|
+
function generateReadme() {
|
|
4357
|
+
return `# AE Data Integration \u2014 handoff package
|
|
4358
|
+
|
|
4359
|
+
A frozen, reusable pipeline for importing a **same-shape** local file into AE,
|
|
4360
|
+
generated by \`ae-cli data-integration handoff\`. Source and Transform are frozen
|
|
4361
|
+
(the confirmed business logic); only the Tracking-plan and Sink gates still
|
|
4362
|
+
require human confirmation.
|
|
4363
|
+
|
|
4364
|
+
## Quick start
|
|
4365
|
+
|
|
4366
|
+
\`\`\`bash
|
|
4367
|
+
cp <today's file> inbox/
|
|
4368
|
+
bin/run.sh # shape check -> convert -> plan check (no upload)
|
|
4369
|
+
bin/upload.sh runs/<latest> # dry-run
|
|
4370
|
+
bin/upload.sh runs/<latest> --confirm
|
|
4371
|
+
\`\`\`
|
|
4372
|
+
|
|
4373
|
+
Read [RUNBOOK.md](RUNBOOK.md) for the full flow, the four confirmation gates,
|
|
4374
|
+
and how to verify persistence.
|
|
4375
|
+
|
|
4376
|
+
## Handing off to an agent
|
|
4377
|
+
|
|
4378
|
+
Import \`inbox/<today's file>\` through the ae-data-integration skill using this
|
|
4379
|
+
package. Tell it to follow RUNBOOK.md and stop for confirmation before uploading.
|
|
4380
|
+
|
|
4381
|
+
## Package layout
|
|
4382
|
+
|
|
4383
|
+
| Path | Purpose |
|
|
4384
|
+
| --- | --- |
|
|
4385
|
+
| \`pipeline.json\` | Declarative source -> transform -> sink descriptor |
|
|
4386
|
+
| \`index.json\` | Structure-fingerprint index (reuse matching) |
|
|
4387
|
+
| \`shape.json\` | Column baseline used by the shape gate |
|
|
4388
|
+
| \`<fingerprint16>/\` | Frozen mapping + tracking plan + transform wrapper |
|
|
4389
|
+
| \`bin/run.sh\` | Source + Transform + Plan executor (never uploads) |
|
|
4390
|
+
| \`bin/upload.sh\` | Sink executor (dry-run by default; resolves the recorded target) |
|
|
4391
|
+
| \`bin/verify.py\` | Soft persistence check (submit window vs ingest summary) |
|
|
4392
|
+
| \`bin/resolve_appid.py\` | APPID derivation helper (project info get) |
|
|
4393
|
+
| \`.local/target.env\` | Upload target overrides (APPID / endpoint); only the template ships |
|
|
4394
|
+
| \`inbox/\` \`runs/\` | Daily input / per-run outputs |
|
|
4395
|
+
|
|
4396
|
+
## Safety
|
|
4397
|
+
|
|
4398
|
+
The package records at most a destination \`pushurl\` and \`project_id\` (no APPID,
|
|
4399
|
+
tokens, or raw data values). \`bin/upload.sh\` always requires \`--confirm\` before
|
|
4400
|
+
sending, so the operator re-confirms the address and project on every reuse. Copy
|
|
4401
|
+
\`.local/target.env.example\` to \`.local/target.env\` for explicit overrides and
|
|
4402
|
+
never commit it.
|
|
4403
|
+
`;
|
|
4404
|
+
}
|
|
4405
|
+
function generateRunbook() {
|
|
4406
|
+
return `# RUNBOOK \u2014 same-shape file import
|
|
4407
|
+
|
|
4408
|
+
Run this when a file of the **same shape** arrives again (same sheets and
|
|
4409
|
+
headers as \`shape.json\`). If the headers changed, stop: re-run the full
|
|
4410
|
+
ae-data-integration pipeline instead of editing the frozen mapping.
|
|
4411
|
+
|
|
4412
|
+
## Gates
|
|
4413
|
+
|
|
4414
|
+
1. **Shape gate** \u2014 \`bin/run.sh\` rebinds the frozen mappings to the new file
|
|
4415
|
+
and compares the column set against \`shape.json\`. A mismatch fails fast on
|
|
4416
|
+
purpose: a different shape means the frozen business logic was never reviewed
|
|
4417
|
+
for it.
|
|
4418
|
+
2. **Transform** \u2014 each frozen mapping runs through
|
|
4419
|
+
\`ae-cli data-integration convert\`. Quarantined rows land in
|
|
4420
|
+
\`invalid.rows.jsonl\`; they are never silently dropped.
|
|
4421
|
+
3. **Tracking-plan gate** \u2014 \`bin/run.sh\` verifies every produced event and
|
|
4422
|
+
property already exists in the frozen \`plan.json\`. New events or properties
|
|
4423
|
+
make it exit with code 3: merge them into the project tracking plan first,
|
|
4424
|
+
then return here to upload.
|
|
4425
|
+
4. **Sink gate** \u2014 \`bin/upload.sh\` is dry-run by default. Read
|
|
4426
|
+
\`record_count\`, \`batch_count\`, and \`manifest_status\` before adding
|
|
4427
|
+
\`--confirm\`. A \`blocked\` manifest means rows were quarantined;
|
|
4428
|
+
\`--confirm\` then uploads only the valid subset (\`--allow-clean-subset\`).
|
|
4429
|
+
|
|
4430
|
+
## Destination
|
|
4431
|
+
|
|
4432
|
+
The package records the destination it was handed off for when \`handoff\` was run
|
|
4433
|
+
with \`--pushurl\` / \`--project-id\` (see \`pipeline.json\` \u2192 \`sink.params\`). Reuse
|
|
4434
|
+
defaults to that target, but \`bin/upload.sh\` never sends without \`--confirm\`, so
|
|
4435
|
+
the operator re-confirms the address and project every time.
|
|
4436
|
+
|
|
4437
|
+
Resolution order at upload time:
|
|
4438
|
+
|
|
4439
|
+
- endpoint: recorded \`pushurl\` (+ \`/sync_json\`), else \`AE_ENDPOINT\`.
|
|
4440
|
+
- APPID: \`AE_APPID\`, else derived via \`ae-cli project info get --project-id <id>\`
|
|
4441
|
+
(see \`bin/resolve_appid.py\`; it reads the \`data.appid\` field \u2014 set \`AE_APPID\`
|
|
4442
|
+
when that field is absent).
|
|
4443
|
+
- project id: recorded \`project_id\`, else \`AE_PROJECT_ID\`.
|
|
4444
|
+
|
|
4445
|
+
\`.local/target.env\` remains the explicit override for all three:
|
|
4446
|
+
\`AE_ENDPOINT\` (full receiver URL ending in \`/sync_json\`), \`AE_APPID\`, \`AE_PROJECT_ID\`.
|
|
4447
|
+
|
|
4448
|
+
## Verify persistence
|
|
4449
|
+
|
|
4450
|
+
\`receiver_accepted\` is not persistence. About a minute after upload, confirm
|
|
4451
|
+
the data landed with ae-cli. The package ships a soft check that automates the
|
|
4452
|
+
before/after comparison:
|
|
4453
|
+
|
|
4454
|
+
\`\`\`bash
|
|
4455
|
+
bin/verify.py runs/<run-id> --baseline # before upload
|
|
4456
|
+
bin/upload.sh runs/<run-id> --confirm
|
|
4457
|
+
bin/verify.py runs/<run-id> --check # after upload
|
|
4458
|
+
\`\`\`
|
|
4459
|
+
|
|
4460
|
+
\`bin/verify.py\` prints the submit window and expected counts, then shows the
|
|
4461
|
+
\`tracking ingest summary\` payload before and after for comparison. It does not
|
|
4462
|
+
auto-verify per-event counts \u2014 the summary shape is server-defined, and a shared
|
|
4463
|
+
project's window delta is not attributed to this import. Cross-check with:
|
|
4464
|
+
|
|
4465
|
+
\`\`\`bash
|
|
4466
|
+
ae-cli tracking live-data list -p <AE_PROJECT_ID>
|
|
4467
|
+
ae-cli tracking ingest-error list -p <AE_PROJECT_ID> --data-name <name>
|
|
4468
|
+
\`\`\`
|
|
4469
|
+
|
|
4470
|
+
For a hard per-event SQL judge, overlay a project custom layer instead of editing
|
|
4471
|
+
this package (see \`custom-layer.md\` in the ae-data-integration skill).
|
|
4472
|
+
|
|
4473
|
+
## Interrupted uploads
|
|
4474
|
+
|
|
4475
|
+
If a batch times out or loses the network, that batch's state is unknown. Stop:
|
|
4476
|
+
verify what actually landed, then follow the ae-data-integration skill to resume
|
|
4477
|
+
from the verified offset. Never re-run the whole upload blindly.
|
|
4478
|
+
|
|
4479
|
+
## Files
|
|
4480
|
+
|
|
4481
|
+
See \`README.md\` for the package layout.
|
|
4482
|
+
`;
|
|
4483
|
+
}
|
|
4484
|
+
function generateEnvTemplate() {
|
|
4485
|
+
return [
|
|
4486
|
+
"# Upload target overrides. Copy this file to .local/target.env and fill only",
|
|
4487
|
+
"# what the package does not already record (pipeline.json sink.params).",
|
|
4488
|
+
"# AE_ENDPOINT: a full receiver URL ending in /sync_json (used when no pushurl is recorded).",
|
|
4489
|
+
"# AE_APPID: the destination project APPID (overrides the project info get derivation).",
|
|
4490
|
+
"# AE_PROJECT_ID: the destination project ID, used when no project_id is recorded.",
|
|
4491
|
+
"AE_ENDPOINT=",
|
|
4492
|
+
"AE_APPID=",
|
|
4493
|
+
"AE_PROJECT_ID=",
|
|
4494
|
+
""
|
|
4495
|
+
].join("\n");
|
|
4496
|
+
}
|
|
4497
|
+
function generateGitignore() {
|
|
4498
|
+
return ["inbox/", "runs/", ".local/target.env", ""].join("\n");
|
|
4499
|
+
}
|
|
4500
|
+
|
|
4501
|
+
// src/commands/data-integration/handoff.ts
|
|
3123
4502
|
var HANDOFF_INDEX_VERSION = "ae-data-integration-index/v1";
|
|
3124
4503
|
var HANDOFF_DIR_LEN = 16;
|
|
3125
4504
|
function structureFingerprint(mapping) {
|
|
3126
4505
|
const canonical = {
|
|
3127
4506
|
mode: mapping.mode,
|
|
3128
|
-
|
|
3129
|
-
|
|
3130
|
-
distinct_id_field: mapping.distinct_id_field ?? null,
|
|
3131
|
-
record_type_field: mapping.record_type_field ?? null,
|
|
3132
|
-
event_name_field: mapping.event_name_field ?? null,
|
|
3133
|
-
columns: mapping.properties.map((property) => ({ source: property.source, type: property.type })).sort((left, right) => left.source.localeCompare(right.source)),
|
|
3134
|
-
excluded: [...mapping.exclude_columns ?? []].sort()
|
|
4507
|
+
format: mapping.source.format,
|
|
4508
|
+
columns: sourceColumns(mapping)
|
|
3135
4509
|
};
|
|
3136
4510
|
return createHash3("sha256").update(JSON.stringify(canonical)).digest("hex");
|
|
3137
4511
|
}
|
|
@@ -3142,17 +4516,17 @@ function upsertIndexEntry(index, entry) {
|
|
|
3142
4516
|
function buildHandoffPackage(outDir, mapping, planFile) {
|
|
3143
4517
|
const fingerprint = structureFingerprint(mapping);
|
|
3144
4518
|
const dirName = fingerprint.slice(0, HANDOFF_DIR_LEN);
|
|
3145
|
-
const handoffDir =
|
|
3146
|
-
const indexPath =
|
|
4519
|
+
const handoffDir = join3(outDir, dirName);
|
|
4520
|
+
const indexPath = join3(outDir, "index.json");
|
|
3147
4521
|
const index = readHandoffIndex(indexPath);
|
|
3148
4522
|
const reusedExisting = index.entries.some((item) => item.fingerprint === fingerprint);
|
|
3149
4523
|
mkdirSync2(handoffDir, { recursive: true, mode: 448 });
|
|
3150
4524
|
chmodSync2(handoffDir, 448);
|
|
3151
|
-
writeSecureJson2(
|
|
3152
|
-
writeSecureText2(
|
|
4525
|
+
writeSecureJson2(join3(handoffDir, "mapping.json"), mapping);
|
|
4526
|
+
writeSecureText2(join3(handoffDir, "transform.mjs"), createHandoffScript());
|
|
3153
4527
|
let planFileRel;
|
|
3154
4528
|
if (planFile) {
|
|
3155
|
-
writeSecureJson2(
|
|
4529
|
+
writeSecureJson2(join3(handoffDir, "plan.json"), readPlanFile(planFile));
|
|
3156
4530
|
planFileRel = `${dirName}/plan.json`;
|
|
3157
4531
|
}
|
|
3158
4532
|
const entry = {
|
|
@@ -3167,7 +4541,7 @@ function buildHandoffPackage(outDir, mapping, planFile) {
|
|
|
3167
4541
|
...planFileRel ? { plan_file: planFileRel } : {}
|
|
3168
4542
|
};
|
|
3169
4543
|
writeAtomicJson(indexPath, upsertIndexEntry(index, entry));
|
|
3170
|
-
return { outDir, handoffDir, dirName, fingerprint, reusedExisting, planFile: planFileRel };
|
|
4544
|
+
return { outDir, handoffDir, dirName, fingerprint, reusedExisting, planFile: planFileRel, entry };
|
|
3171
4545
|
}
|
|
3172
4546
|
function readHandoffIndex(path) {
|
|
3173
4547
|
let raw;
|
|
@@ -3262,56 +4636,199 @@ var dataIntegrationHandoff = {
|
|
|
3262
4636
|
service: "data-integration",
|
|
3263
4637
|
command: "handoff",
|
|
3264
4638
|
usesAeHost: false,
|
|
3265
|
-
description: "Export a reusable handoff package (frozen
|
|
4639
|
+
description: "Export a reusable handoff package (pipeline descriptor + frozen mappings + stage executors + docs) and a shareable zip.",
|
|
3266
4640
|
flags: [
|
|
3267
|
-
{ name: "mapping", type: "string", required: true, sensitive: true, desc:
|
|
4641
|
+
{ name: "mapping", type: "string", required: true, variadic: true, sensitive: true, desc: `Confirmed ${MAPPING_VERSION} JSON, file path, or @file. Repeat for multiple data sets (one sheet each).` },
|
|
3268
4642
|
{ name: "plan-file", type: "string", sensitive: true, desc: "Tracking-plan draft.json to reference inside the handoff package." },
|
|
3269
|
-
{ name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-data-integration." }
|
|
4643
|
+
{ name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-cli/data-integration (project workspace)." },
|
|
4644
|
+
{ name: "pushurl", type: "string", sensitive: true, desc: "Receiver base URL to record as the reuse upload target (sink endpoint = pushurl + /sync_json). Reuse still requires --confirm." },
|
|
4645
|
+
{ name: "project-id", type: "string", sensitive: true, desc: "Numeric destination project ID to record; upload derives the APPID from it via project info get." }
|
|
3270
4646
|
],
|
|
3271
4647
|
risk: "write",
|
|
3272
4648
|
dryRun: async (ctx) => {
|
|
3273
|
-
const
|
|
3274
|
-
const outDir = resolve3(ctx.str("out-dir").trim() || ".ae-data-integration");
|
|
4649
|
+
const mappings = ctx.list("mapping").map((source) => readLocalDataMapping(source));
|
|
4650
|
+
const outDir = resolve3(ctx.str("out-dir").trim() || join3(".ae-cli", "data-integration"));
|
|
3275
4651
|
const planFile = ctx.str("plan-file").trim() || void 0;
|
|
4652
|
+
const pushurl = ctx.str("pushurl").trim() || void 0;
|
|
4653
|
+
const projectId = ctx.str("project-id").trim() || void 0;
|
|
4654
|
+
const fingerprints = mappings.map(structureFingerprint);
|
|
3276
4655
|
return {
|
|
3277
4656
|
action: "handoff_local_data",
|
|
3278
4657
|
out_dir: outDir,
|
|
3279
|
-
|
|
3280
|
-
|
|
3281
|
-
|
|
3282
|
-
|
|
3283
|
-
|
|
3284
|
-
|
|
4658
|
+
mapping_count: mappings.length,
|
|
4659
|
+
fingerprints,
|
|
4660
|
+
target: { pushurl, project_id: projectId },
|
|
4661
|
+
files: handoffFileList(fingerprints, planFile),
|
|
4662
|
+
index_file: join3(outDir, "index.json"),
|
|
4663
|
+
pipeline_file: join3(outDir, "pipeline.json"),
|
|
4664
|
+
shape_file: join3(outDir, "shape.json"),
|
|
4665
|
+
zip_path: zipPathFor(outDir, fingerprints[0])
|
|
3285
4666
|
};
|
|
3286
4667
|
},
|
|
3287
4668
|
execute: async (ctx) => {
|
|
3288
|
-
const
|
|
3289
|
-
const outDir = resolve3(ctx.str("out-dir").trim() || ".ae-data-integration");
|
|
4669
|
+
const mappings = ctx.list("mapping").map((source) => readLocalDataMapping(source));
|
|
4670
|
+
const outDir = resolve3(ctx.str("out-dir").trim() || join3(".ae-cli", "data-integration"));
|
|
3290
4671
|
const planFile = ctx.str("plan-file").trim() || void 0;
|
|
3291
|
-
const
|
|
4672
|
+
const target = {
|
|
4673
|
+
pushurl: ctx.str("pushurl").trim() || void 0,
|
|
4674
|
+
project_id: ctx.str("project-id").trim() || void 0
|
|
4675
|
+
};
|
|
4676
|
+
const items = mappings.map((mapping) => ({
|
|
4677
|
+
mapping,
|
|
4678
|
+
build: buildHandoffPackage(outDir, mapping, planFile)
|
|
4679
|
+
}));
|
|
4680
|
+
const entries = items.map((item) => item.build.entry);
|
|
4681
|
+
const relayFiles = [
|
|
4682
|
+
{ relPath: "pipeline.json", content: `${JSON.stringify(buildPipelineDescriptor(entries, target), null, 2)}
|
|
4683
|
+
`, mode: 384 },
|
|
4684
|
+
{ relPath: "shape.json", content: `${JSON.stringify(buildShapeBaseline(items.map((item) => ({ mapping: item.mapping, fingerprint: item.build.fingerprint }))), null, 2)}
|
|
4685
|
+
`, mode: 384 },
|
|
4686
|
+
...generateBinScripts(),
|
|
4687
|
+
{ relPath: "README.md", content: generateReadme(), mode: 384 },
|
|
4688
|
+
{ relPath: "RUNBOOK.md", content: generateRunbook(), mode: 384 },
|
|
4689
|
+
{ relPath: ".local/target.env.example", content: generateEnvTemplate(), mode: 384 },
|
|
4690
|
+
{ relPath: ".gitignore", content: generateGitignore(), mode: 384 },
|
|
4691
|
+
{ relPath: "inbox/.gitkeep", content: "", mode: 384 },
|
|
4692
|
+
{ relPath: "runs/.gitkeep", content: "", mode: 384 }
|
|
4693
|
+
];
|
|
4694
|
+
for (const file of relayFiles) writeRelayFile(outDir, file);
|
|
4695
|
+
const zipPath = zipPathFor(outDir, entries[0].fingerprint);
|
|
4696
|
+
const stagingDir = stageScopedPackage(outDir, items.map((item) => item.build), entries, relayFiles);
|
|
4697
|
+
try {
|
|
4698
|
+
await zipPackage(stagingDir, zipPath);
|
|
4699
|
+
chmodSync2(zipPath, 384);
|
|
4700
|
+
} finally {
|
|
4701
|
+
rmSync(stagingDir, { recursive: true, force: true });
|
|
4702
|
+
}
|
|
3292
4703
|
return {
|
|
3293
|
-
out_dir:
|
|
3294
|
-
|
|
3295
|
-
|
|
3296
|
-
|
|
3297
|
-
|
|
3298
|
-
|
|
3299
|
-
|
|
3300
|
-
|
|
4704
|
+
out_dir: outDir,
|
|
4705
|
+
zip_path: zipPath,
|
|
4706
|
+
pipeline_file: join3(outDir, "pipeline.json"),
|
|
4707
|
+
shape_file: join3(outDir, "shape.json"),
|
|
4708
|
+
index_file: join3(outDir, "index.json"),
|
|
4709
|
+
handoff_dirs: items.map((item) => item.build.dirName),
|
|
4710
|
+
deliverables: buildDeliverables(outDir, relayFiles, items),
|
|
4711
|
+
next_steps: [
|
|
4712
|
+
`Review the pipeline descriptor and four confirmation gates: ${join3(outDir, "RUNBOOK.md")}.`,
|
|
4713
|
+
`Share the archive: ${zipPath}.`,
|
|
4714
|
+
`Next same-shape file: cd ${outDir} && bin/run.sh <new-file>, then bin/upload.sh runs/<run-id> --confirm.`
|
|
4715
|
+
],
|
|
4716
|
+
reused_existing: items.some((item) => item.build.reusedExisting)
|
|
3301
4717
|
};
|
|
3302
4718
|
}
|
|
3303
4719
|
};
|
|
4720
|
+
function zipPathFor(outDir, fingerprint) {
|
|
4721
|
+
if (!fingerprint) return void 0;
|
|
4722
|
+
return join3(dirname2(resolve3(outDir)), `ae-data-integration-handoff-${fingerprint.slice(0, 8)}.zip`);
|
|
4723
|
+
}
|
|
4724
|
+
var RELAY_FILE_PATHS = [
|
|
4725
|
+
"pipeline.json",
|
|
4726
|
+
"shape.json",
|
|
4727
|
+
"index.json",
|
|
4728
|
+
"README.md",
|
|
4729
|
+
"RUNBOOK.md",
|
|
4730
|
+
".local/target.env.example",
|
|
4731
|
+
".gitignore",
|
|
4732
|
+
"bin/run.sh",
|
|
4733
|
+
"bin/upload.sh",
|
|
4734
|
+
"bin/bind_mapping.py",
|
|
4735
|
+
"bin/summarize.py",
|
|
4736
|
+
"bin/plan_check.py",
|
|
4737
|
+
"bin/verify.py",
|
|
4738
|
+
"bin/resolve_appid.py",
|
|
4739
|
+
"inbox/.gitkeep",
|
|
4740
|
+
"runs/.gitkeep"
|
|
4741
|
+
];
|
|
4742
|
+
function mappingDirFiles(dirName, planFile) {
|
|
4743
|
+
return [
|
|
4744
|
+
`${dirName}/mapping.json`,
|
|
4745
|
+
`${dirName}/transform.mjs`,
|
|
4746
|
+
...planFile ? [`${dirName}/plan.json`] : []
|
|
4747
|
+
];
|
|
4748
|
+
}
|
|
4749
|
+
function handoffFileList(fingerprints, planFile) {
|
|
4750
|
+
const perMapping = fingerprints.flatMap(
|
|
4751
|
+
(fingerprint) => mappingDirFiles(fingerprint.slice(0, HANDOFF_DIR_LEN), planFile)
|
|
4752
|
+
);
|
|
4753
|
+
return [...RELAY_FILE_PATHS, ...perMapping];
|
|
4754
|
+
}
|
|
4755
|
+
function buildDeliverables(outDir, relayFiles, items) {
|
|
4756
|
+
const relay = [
|
|
4757
|
+
{ rel_path: "index.json", abs_path: join3(outDir, "index.json") },
|
|
4758
|
+
...relayFiles.map((file) => ({ rel_path: file.relPath, abs_path: join3(outDir, file.relPath) }))
|
|
4759
|
+
];
|
|
4760
|
+
const mappings = items.flatMap(
|
|
4761
|
+
(item) => mappingDirFiles(item.build.dirName, item.build.planFile).map((rel) => ({
|
|
4762
|
+
rel_path: rel,
|
|
4763
|
+
abs_path: join3(outDir, rel)
|
|
4764
|
+
}))
|
|
4765
|
+
);
|
|
4766
|
+
return [...relay, ...mappings];
|
|
4767
|
+
}
|
|
4768
|
+
function writeRelayFile(outDir, file) {
|
|
4769
|
+
const abs = join3(outDir, file.relPath);
|
|
4770
|
+
mkdirSync2(dirname2(abs), { recursive: true, mode: 448 });
|
|
4771
|
+
writeFileSync2(abs, file.content, { encoding: "utf8", mode: file.mode });
|
|
4772
|
+
chmodSync2(abs, file.mode);
|
|
4773
|
+
}
|
|
4774
|
+
function stageScopedPackage(outDir, mappingDirs, entries, relayFiles) {
|
|
4775
|
+
const staging = mkdtempSync(join3(tmpdir(), "ae-handoff-"));
|
|
4776
|
+
writeSecureJson2(join3(staging, "index.json"), { version: HANDOFF_INDEX_VERSION, entries });
|
|
4777
|
+
for (const file of relayFiles) writeRelayFile(staging, file);
|
|
4778
|
+
for (const { dirName, planFile } of mappingDirs) {
|
|
4779
|
+
for (const rel of mappingDirFiles(dirName, planFile)) {
|
|
4780
|
+
const dest = join3(staging, rel);
|
|
4781
|
+
mkdirSync2(dirname2(dest), { recursive: true, mode: 448 });
|
|
4782
|
+
writeFileSync2(dest, readFileSync4(join3(outDir, rel), "utf8"), { encoding: "utf8", mode: 384 });
|
|
4783
|
+
chmodSync2(dest, 384);
|
|
4784
|
+
}
|
|
4785
|
+
}
|
|
4786
|
+
return staging;
|
|
4787
|
+
}
|
|
3304
4788
|
|
|
3305
|
-
// src/commands/data-integration/
|
|
4789
|
+
// src/commands/data-integration/reuse.ts
|
|
3306
4790
|
import { readFileSync as readFileSync5 } from "fs";
|
|
3307
|
-
import { dirname as
|
|
4791
|
+
import { dirname as dirname4, join as join5, resolve as resolve5 } from "path";
|
|
4792
|
+
|
|
4793
|
+
// src/commands/data-integration/handoff-root.ts
|
|
4794
|
+
import { existsSync as existsSync2 } from "fs";
|
|
4795
|
+
import { dirname as dirname3, join as join4, resolve as resolve4 } from "path";
|
|
4796
|
+
function globalHandoffDir() {
|
|
4797
|
+
const home = process.env.HOME;
|
|
4798
|
+
if (!home) return void 0;
|
|
4799
|
+
return join4(getConfigDir(), "data-integration");
|
|
4800
|
+
}
|
|
4801
|
+
function upwardHandoffDirs(startDir) {
|
|
4802
|
+
const dirs = [];
|
|
4803
|
+
let current = resolve4(startDir);
|
|
4804
|
+
for (; ; ) {
|
|
4805
|
+
const candidate = join4(current, ".ae-cli", "data-integration");
|
|
4806
|
+
if (!dirs.includes(candidate)) dirs.push(candidate);
|
|
4807
|
+
const parent = dirname3(current);
|
|
4808
|
+
if (parent === current) break;
|
|
4809
|
+
current = parent;
|
|
4810
|
+
}
|
|
4811
|
+
return dirs;
|
|
4812
|
+
}
|
|
4813
|
+
function reuseSearchPaths(startDir = process.cwd()) {
|
|
4814
|
+
const globalDir = globalHandoffDir();
|
|
4815
|
+
return globalDir ? [...upwardHandoffDirs(startDir), globalDir] : upwardHandoffDirs(startDir);
|
|
4816
|
+
}
|
|
4817
|
+
function findReuseRoot(startDir = process.cwd()) {
|
|
4818
|
+
for (const dir of upwardHandoffDirs(startDir)) {
|
|
4819
|
+
if (existsSync2(join4(dir, "index.json"))) return dir;
|
|
4820
|
+
}
|
|
4821
|
+
return globalHandoffDir();
|
|
4822
|
+
}
|
|
4823
|
+
|
|
4824
|
+
// src/commands/data-integration/reuse.ts
|
|
3308
4825
|
function detectReuse(mapping, outDir) {
|
|
3309
4826
|
const fingerprint = structureFingerprint(mapping);
|
|
3310
|
-
const indexPath =
|
|
4827
|
+
const indexPath = join5(outDir, "index.json");
|
|
3311
4828
|
const index = readHandoffIndex(indexPath);
|
|
3312
4829
|
const entry = index.entries.find((item) => item.fingerprint === fingerprint);
|
|
3313
4830
|
if (!entry) return { matched: false, fingerprint, index_file: indexPath };
|
|
3314
|
-
const mappingPath =
|
|
4831
|
+
const mappingPath = join5(outDir, entry.mapping_file);
|
|
3315
4832
|
const match = {
|
|
3316
4833
|
fingerprint: entry.fingerprint,
|
|
3317
4834
|
created_at: entry.created_at,
|
|
@@ -3322,7 +4839,7 @@ function detectReuse(mapping, outDir) {
|
|
|
3322
4839
|
mapping_file: entry.mapping_file,
|
|
3323
4840
|
...entry.plan_file ? { plan_file: entry.plan_file } : {},
|
|
3324
4841
|
...readFrozenEventName(mappingPath),
|
|
3325
|
-
run: `node ${
|
|
4842
|
+
run: `node ${join5(outDir, dirname4(entry.mapping_file), "transform.mjs")} <new-input-file> [<output-dir>]`
|
|
3326
4843
|
};
|
|
3327
4844
|
return { matched: true, fingerprint, index_file: indexPath, match };
|
|
3328
4845
|
}
|
|
@@ -3345,29 +4862,41 @@ var dataIntegrationReuse = {
|
|
|
3345
4862
|
service: "data-integration",
|
|
3346
4863
|
command: "reuse",
|
|
3347
4864
|
usesAeHost: false,
|
|
3348
|
-
description: "Match a candidate mapping against the .ae-data-integration/ handoff index and propose a reusable package.",
|
|
4865
|
+
description: "Match a candidate mapping against the .ae-cli/data-integration/ handoff index and propose a reusable package.",
|
|
3349
4866
|
flags: [
|
|
3350
|
-
{ name: "mapping", type: "string", required: true, sensitive: true, desc:
|
|
3351
|
-
{ name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-data-integration
|
|
4867
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: `Candidate ${MAPPING_VERSION} JSON, file path, or @file (typically inspect recommended_mapping).` },
|
|
4868
|
+
{ name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: nearest .ae-cli/data-integration/ upward from cwd, then ~/.ae-cli/data-integration/." }
|
|
3352
4869
|
],
|
|
3353
4870
|
risk: "read",
|
|
3354
4871
|
dryRun: async (ctx) => {
|
|
3355
4872
|
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
3356
|
-
const outDir =
|
|
4873
|
+
const outDir = resolveReuseOutDir(ctx);
|
|
3357
4874
|
const result = detectReuse(mapping, outDir);
|
|
3358
4875
|
return {
|
|
3359
4876
|
action: "reuse_detect",
|
|
3360
4877
|
fingerprint: result.fingerprint,
|
|
3361
4878
|
matched: result.matched,
|
|
3362
|
-
index_file: result.index_file
|
|
4879
|
+
index_file: result.index_file,
|
|
4880
|
+
searched_paths: searchedIndexPaths(ctx)
|
|
3363
4881
|
};
|
|
3364
4882
|
},
|
|
3365
4883
|
execute: async (ctx) => {
|
|
3366
4884
|
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
3367
|
-
const outDir =
|
|
3368
|
-
|
|
4885
|
+
const outDir = resolveReuseOutDir(ctx);
|
|
4886
|
+
const result = detectReuse(mapping, outDir);
|
|
4887
|
+
return { ...result, searched_paths: searchedIndexPaths(ctx) };
|
|
3369
4888
|
}
|
|
3370
4889
|
};
|
|
4890
|
+
function resolveReuseOutDir(ctx) {
|
|
4891
|
+
const explicit = ctx.str("out-dir").trim();
|
|
4892
|
+
if (explicit) return resolve5(explicit);
|
|
4893
|
+
return findReuseRoot() ?? resolve5(join5(".ae-cli", "data-integration"));
|
|
4894
|
+
}
|
|
4895
|
+
function searchedIndexPaths(ctx) {
|
|
4896
|
+
const explicit = ctx.str("out-dir").trim();
|
|
4897
|
+
const dirs = explicit ? [resolve5(explicit)] : reuseSearchPaths();
|
|
4898
|
+
return dirs.map((dir) => join5(dir, "index.json"));
|
|
4899
|
+
}
|
|
3371
4900
|
|
|
3372
4901
|
// src/commands/data-integration/index.ts
|
|
3373
4902
|
var commands = [dataIntegrationInspect, dataIntegrationPlan, dataIntegrationConvert, dataIntegrationUpload, dataIntegrationHandoff, dataIntegrationReuse];
|