@thinkingai/ae-cli 6.1.16 → 6.1.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/README.md +5 -2
  2. package/README.zh.md +5 -2
  3. package/dist/{auth-2WTQOP77.js → auth-GBMV6TEJ.js} +3 -3
  4. package/dist/{auth-77BUFLGC.js → auth-ROB2EDYV.js} +12 -13
  5. package/dist/{capability-72DTW5M2.js → capability-DKMYUTLC.js} +50 -13
  6. package/dist/{capability-PJHNI4GJ.js → capability-HYVVPG25.js} +49 -12
  7. package/dist/chunk-3FY3RJ26.js +293 -0
  8. package/dist/{chunk-PTE56QPL.js → chunk-3KWQYGYI.js} +4 -0
  9. package/dist/{chunk-UOUS37JQ.js → chunk-4XXOWOTA.js} +3 -3
  10. package/dist/{chunk-GT46FPXN.js → chunk-BYYS3ANB.js} +17 -8
  11. package/dist/{chunk-YA6SMTXG.js → chunk-EFH4XWYC.js} +3 -3
  12. package/dist/{chunk-4SGZG4XY.js → chunk-J2DEBMRF.js} +9 -7
  13. package/dist/{chunk-LYVNONC4.js → chunk-JHENBQ5B.js} +35 -0
  14. package/dist/{chunk-VR3LCBHW.js → chunk-JRJY5DMJ.js} +5 -5
  15. package/dist/{chunk-ILIU36SU.js → chunk-OMPRXM3V.js} +3 -3
  16. package/dist/{chunk-VPKZ7I72.js → chunk-QNOLN2LJ.js} +2 -2
  17. package/dist/{chunk-SAU3QFIQ.js → chunk-QZ3AS4KK.js} +3 -3
  18. package/dist/chunk-RJDU7NYP.js +1198 -0
  19. package/dist/{chunk-4KVPKXFX.js → chunk-RNAALWJK.js} +2 -2
  20. package/dist/{chunk-C4MGVGJW.js → chunk-SERWF6G5.js} +1 -1
  21. package/dist/{chunk-RGKJGKT7.js → chunk-Y3LOALAV.js} +5 -5
  22. package/dist/{chunk-P3FGXJTU.js → chunk-ZQ47LWTI.js} +4 -4
  23. package/dist/chunk-ZQKDZXDO.js +317 -0
  24. package/dist/{client-TKG4WBHN.js → client-L2YDMHQ6.js} +5 -4
  25. package/dist/{community-report-client-FI4LNVYS.js → community-report-client-C7WDGET3.js} +1 -2
  26. package/dist/{config-RE6CMGPK.js → config-BMYZX2UE.js} +7 -6
  27. package/dist/{data-integration-2MYMANJI.js → data-integration-QEKDWQDY.js} +1741 -212
  28. package/dist/index.js +131 -1241
  29. package/dist/{local-data-upload-client-BWHSUQQK.js → local-data-upload-client-4YYHSYD6.js} +1 -2
  30. package/dist/{memory-CHRU2F7W.js → memory-3ORCR7JH.js} +6 -6
  31. package/dist/{memory-YK33G4T7.js → memory-I2WXDTV2.js} +5 -5
  32. package/dist/{metadata-XXR34N5P.js → metadata-I4C2EWUN.js} +10 -10
  33. package/dist/{metadata-UILXHBWF.js → metadata-VUOQJE26.js} +9 -9
  34. package/dist/{model-NR3JHFSJ.js → model-HLHIEFMU.js} +5 -5
  35. package/dist/{model-K3KLWIW6.js → model-UGRDX4MW.js} +6 -6
  36. package/dist/personal-semantic-preference-LIPACBDX.js +239 -0
  37. package/dist/personal-semantic-preference-OEISBRHM.js +239 -0
  38. package/dist/project-semantic-FFPWFPIW.js +1114 -0
  39. package/dist/project-semantic-RT3R2VQD.js +1114 -0
  40. package/dist/{sync-FCKOVWWS.js → sync-HKIOZXQE.js} +6 -6
  41. package/dist/{sync-DAVKYVMW.js → sync-TFHU2UTG.js} +7 -7
  42. package/dist/{te-agent-HLW4VTQK.js → te-agent-BR6VDBNX.js} +9 -8
  43. package/dist/{te-agent-4BKBODMF.js → te-agent-VLYOV7S4.js} +8 -7
  44. package/dist/{te-analysis-ZMNGOVNW.js → te-analysis-4YGQL5RC.js} +437 -38
  45. package/dist/{te-analysis-O6DCO6BS.js → te-analysis-7VUNUYWZ.js} +436 -37
  46. package/dist/{te-community-HLC43QKH.js → te-community-5DMNKJWY.js} +5 -5
  47. package/dist/{te-community-6HPBWJUZ.js → te-community-ISDQWJU7.js} +6 -6
  48. package/dist/{te-dataops-HDRUXY4K.js → te-dataops-6P5IKWNJ.js} +8 -7
  49. package/dist/{te-dataops-EJP56W3K.js → te-dataops-CVULXNVB.js} +7 -6
  50. package/dist/{te-engage-RAK5PESW.js → te-engage-KZPR5R22.js} +9 -9
  51. package/dist/{te-engage-FGBGQ4IY.js → te-engage-N5WI32H6.js} +8 -8
  52. package/dist/{te-experiment-SO5MPDMJ.js → te-experiment-6BITX4RD.js} +226 -8
  53. package/dist/{te-experiment-VZF7BT6G.js → te-experiment-UVR4HLND.js} +227 -9
  54. package/dist/{te-kb-APXBWBDY.js → te-kb-RCLSSH2Q.js} +251 -127
  55. package/dist/{te-system-Z77IKZFN.js → te-system-FXITO2JG.js} +5 -5
  56. package/dist/{te-system-YARIK4S5.js → te-system-K2GYMCTB.js} +6 -6
  57. package/dist/{te-team-EFKWYKMK.js → te-team-ADOC2ROP.js} +6 -6
  58. package/dist/{update-OGPSZM5A.js → update-YCYCKJOO.js} +7 -6
  59. package/package.json +2 -1
  60. package/skills/ae-agent/SKILL.md +3 -4
  61. package/skills/ae-agent/references/edit-skill.md +3 -0
  62. package/skills/ae-agent/references/get-skill-content.md +1 -1
  63. package/skills/ae-agent/references/rescan-skills.md +15 -13
  64. package/skills/ae-agent/references/upload-skill.md +7 -4
  65. package/skills/ae-analysis/SKILL.md +45 -4
  66. package/skills/ae-analysis/metadata_resolution.md +38 -4
  67. package/skills/ae-analysis/references/analysis_data_retrieval.md +29 -0
  68. package/skills/ae-analysis/references/asset_authentication_export.md +22 -0
  69. package/skills/ae-analysis/references/asset_authentication_list.md +18 -14
  70. package/skills/ae-analysis/references/asset_authentication_update.md +29 -14
  71. package/skills/ae-analysis/references/command_index.md +17 -9
  72. package/skills/ae-analysis/references/dashboard_get.md +18 -1
  73. package/skills/ae-analysis/references/dashboard_update.md +3 -0
  74. package/skills/ae-analysis/references/personal_semantic_preference_add.md +23 -0
  75. package/skills/ae-analysis/references/personal_semantic_preference_delete.md +17 -0
  76. package/skills/ae-analysis/references/personal_semantic_preference_get.md +19 -0
  77. package/skills/ae-analysis/references/personal_semantic_preference_list.md +21 -0
  78. package/skills/ae-analysis/references/personal_semantic_preference_update.md +19 -0
  79. package/skills/ae-data-integration/SKILL.md +23 -4
  80. package/skills/ae-data-integration/references/custom-layer.md +93 -0
  81. package/skills/ae-data-integration/references/error-handling.md +92 -0
  82. package/skills/ae-data-integration/references/handoff.md +77 -18
  83. package/skills/ae-data-integration/references/local-analysis.md +1 -1
  84. package/skills/ae-data-integration/references/reuse.md +9 -5
  85. package/skills/ae-data-integration/references/sink-upload.md +1 -1
  86. package/skills/ae-data-integration/references/source-inspect.md +32 -13
  87. package/skills/ae-data-integration/references/tracking-plan.md +7 -5
  88. package/skills/ae-data-integration/references/transform.md +10 -10
  89. package/skills/ae-data-integration/references/ue-mapping.md +33 -11
  90. package/skills/ae-engage/references/build-task-save-guide.md +9 -0
  91. package/skills/ae-engage/references/save-task.md +82 -0
  92. package/skills/ae-experiment/SKILL.md +8 -2
  93. package/skills/ae-experiment/references/manage_feature_whitelist.md +66 -0
  94. package/skills/ae-experiment/references/manage_guardrail_metrics.md +26 -0
  95. package/skills/ae-experiment/references/save_experiment.md +1 -1
  96. package/skills/ae-kb/SKILL.md +56 -51
  97. package/skills/ae-kb/references/query-workflow.md +112 -0
  98. package/skills/ae-kb-discovery/SKILL.md +105 -0
  99. package/skills/ae-project-semantic/SKILL.md +193 -0
  100. package/skills/ae-project-semantic/references/query-routing-v5.md +165 -0
  101. package/skills/ae-project-semantic/references/recommendation-quality.md +68 -0
  102. package/dist/chunk-QGM4M3NI.js +0 -37
  103. package/dist/chunk-ZZUOD757.js +0 -598
@@ -1,13 +1,108 @@
1
+ import {
2
+ validateAndFix,
3
+ validateDraft
4
+ } from "./chunk-RJDU7NYP.js";
5
+ import {
6
+ getConfigDir
7
+ } from "./chunk-3FY3RJ26.js";
1
8
  import {
2
9
  CliValidationError,
3
10
  LocalDataUploadError
4
11
  } from "./chunk-UW5UN47B.js";
5
- import "./chunk-QGM4M3NI.js";
12
+ import "./chunk-JHENBQ5B.js";
6
13
 
7
- // src/commands/data-integration/local-data/inspect.ts
14
+ // src/commands/data-integration/inspect.ts
8
15
  import { basename as basename3 } from "path";
9
16
 
10
- // src/commands/data-integration/local-data/input.ts
17
+ // src/commands/data-integration/estimate.ts
18
+ var XLS_SIZE_WARN_BYTES = 100 * 1024 * 1024;
19
+ var LARGE_FILE_WARN_BYTES = 1024 * 1024 * 1024;
20
+ var XLS_HARD_LIMIT_BYTES = 1024 * 1024 * 1024;
21
+ var THROUGHPUT_BYTES_PER_SECOND = {
22
+ jsonl: 10 * 1024 * 1024,
23
+ json: 8 * 1024 * 1024,
24
+ csv: 4 * 1024 * 1024,
25
+ tsv: 4 * 1024 * 1024,
26
+ xlsx: 2 * 1024 * 1024
27
+ };
28
+ var SLOW_ENCODINGS = /* @__PURE__ */ new Set(["gbk", "gb2312", "gb18030", "big5", "shift_jis", "euc-jp", "euc-kr"]);
29
+ function estimateProcessingSeconds(format, sizeBytes, encoding) {
30
+ if (format === "xls") return 0;
31
+ let throughput = THROUGHPUT_BYTES_PER_SECOND[format];
32
+ if (encoding && SLOW_ENCODINGS.has(encoding.toLowerCase())) throughput /= 2;
33
+ return sizeBytes / throughput;
34
+ }
35
+ function formatDuration(seconds) {
36
+ if (seconds < 60) {
37
+ const low2 = Math.max(1, Math.round(seconds * 0.5));
38
+ const high2 = Math.max(low2 + 1, Math.round(seconds * 1.5));
39
+ return `roughly ${low2} to ${high2} seconds`;
40
+ }
41
+ const low = Math.max(1, Math.round(seconds / 60 * 0.5));
42
+ const high = Math.max(low + 1, Math.round(seconds / 60 * 1.5));
43
+ return `roughly ${low} to ${high} minutes`;
44
+ }
45
+ function humanSize(bytes) {
46
+ if (bytes >= 1024 * 1024 * 1024) return `${(bytes / (1024 * 1024 * 1024)).toFixed(1)} GB`;
47
+ return `${Math.round(bytes / (1024 * 1024))} MB`;
48
+ }
49
+ function xlsMemoryWarning(fileName, sizeBytes) {
50
+ return `Warning: ${fileName} is ${humanSize(sizeBytes)} (XLS). The legacy XLS parser loads the entire workbook into memory, roughly 5-10x the file size. Prefer converting to XLSX or splitting the workbook first.`;
51
+ }
52
+ function largeFileTimeWarning(fileName, sizeBytes, format, encoding) {
53
+ const duration = formatDuration(estimateProcessingSeconds(format, sizeBytes, encoding));
54
+ return `Warning: ${fileName} is ${humanSize(sizeBytes)}; processing is estimated to take ${duration}. Consider splitting the file if that is too long.`;
55
+ }
56
+ function largeFileMemoryWarning(format) {
57
+ if (format === "xlsx") {
58
+ return "Memory risk: XLSX processing keeps the workbook's shared string table in memory, so peak memory can substantially exceed the file size.";
59
+ }
60
+ if (format === "json") {
61
+ return "Memory risk: a JSON root object or a very large record may be materialized in memory during inspection, so peak memory can substantially exceed the file size.";
62
+ }
63
+ return null;
64
+ }
65
+ function assessFileSize(fileName, format, sizeBytes, encoding) {
66
+ const baseAssessment = {
67
+ size: humanSize(sizeBytes),
68
+ estimatedDuration: null,
69
+ warning: null,
70
+ reason: null,
71
+ memoryRisk: false,
72
+ rejected: false
73
+ };
74
+ if (format === "xls") {
75
+ if (sizeBytes > XLS_HARD_LIMIT_BYTES) {
76
+ return {
77
+ ...baseAssessment,
78
+ reason: "XLS files larger than 1 GB are not supported; convert the workbook to XLSX or split it first.",
79
+ memoryRisk: true,
80
+ rejected: true
81
+ };
82
+ }
83
+ if (sizeBytes > XLS_SIZE_WARN_BYTES) {
84
+ return {
85
+ ...baseAssessment,
86
+ warning: xlsMemoryWarning(fileName, sizeBytes),
87
+ memoryRisk: true
88
+ };
89
+ }
90
+ return baseAssessment;
91
+ }
92
+ if (sizeBytes > LARGE_FILE_WARN_BYTES) {
93
+ const memoryWarning = largeFileMemoryWarning(format);
94
+ const timeWarning = largeFileTimeWarning(fileName, sizeBytes, format, encoding);
95
+ return {
96
+ ...baseAssessment,
97
+ estimatedDuration: formatDuration(estimateProcessingSeconds(format, sizeBytes, encoding)),
98
+ warning: memoryWarning ? `${timeWarning} ${memoryWarning}` : timeWarning,
99
+ memoryRisk: Boolean(memoryWarning)
100
+ };
101
+ }
102
+ return baseAssessment;
103
+ }
104
+
105
+ // src/commands/data-integration/input.ts
11
106
  import { createHash } from "crypto";
12
107
  import { createReadStream as createReadStream2, statSync } from "fs";
13
108
  import { createRequire as createRequire2 } from "module";
@@ -24,7 +119,7 @@ import StreamValues from "stream-json/streamers/StreamValues.js";
24
119
  import * as unzipper from "unzipper";
25
120
  import { SaxesParser } from "saxes";
26
121
 
27
- // src/commands/data-integration/local-data/encoding.ts
122
+ // src/commands/data-integration/encoding.ts
28
123
  import { createReadStream } from "fs";
29
124
  import { createRequire } from "module";
30
125
  import { openSync, readSync, closeSync } from "fs";
@@ -68,7 +163,7 @@ function readSample(filePath, maxBytes) {
68
163
  }
69
164
  }
70
165
 
71
- // src/commands/data-integration/local-data/time.ts
166
+ // src/commands/data-integration/time.ts
72
167
  var TIME_FORMATS = [
73
168
  // Standard AE format
74
169
  "yyyy-MM-dd HH:mm:ss.SSS",
@@ -312,7 +407,7 @@ function tokenizeFormat(format) {
312
407
  return tokens;
313
408
  }
314
409
 
315
- // src/commands/data-integration/local-data/flatten.ts
410
+ // src/commands/data-integration/flatten.ts
316
411
  var NDJSON_MAX_DEPTH = 1;
317
412
  var NESTED_NODE_SAMPLE_LIMIT = 5;
318
413
  var NESTED_SAMPLE_TRUNCATE = 40;
@@ -345,7 +440,7 @@ function flattenJSON(obj, prefix = "", depth = 0, maxDepth = NDJSON_MAX_DEPTH) {
345
440
  }
346
441
  return result;
347
442
  }
348
- function buildRowWithFlatten(obj, flattenRules) {
443
+ function buildRowWithFlatten(obj, flattenRules, misses) {
349
444
  const row = {};
350
445
  const coveredRoots = new Set(Object.values(flattenRules).map((path) => path.split(".")[0]));
351
446
  const base = flattenJSON(obj);
@@ -354,43 +449,77 @@ function buildRowWithFlatten(obj, flattenRules) {
354
449
  }
355
450
  for (const [outColumn, sourcePath] of Object.entries(flattenRules)) {
356
451
  const value = getNestedValue(obj, sourcePath);
357
- row[outColumn] = value == null ? "" : typeof value === "object" ? JSON.stringify(value) : String(value);
452
+ if (value == null) {
453
+ recordFlattenMiss(misses, outColumn);
454
+ row[outColumn] = "";
455
+ } else {
456
+ row[outColumn] = typeof value === "object" ? JSON.stringify(value) : String(value);
457
+ }
358
458
  }
359
459
  return row;
360
460
  }
361
- function flattenLocalDataRow(value, flattenRules) {
461
+ function flattenLocalDataRow(value, flattenRules, misses) {
362
462
  if (value !== null && typeof value === "object" && !Array.isArray(value)) {
363
463
  if (flattenRules && Object.keys(flattenRules).length > 0) {
364
- return buildRowWithFlatten(value, flattenRules);
464
+ return buildRowWithFlatten(value, flattenRules, misses);
365
465
  }
366
466
  return value;
367
467
  }
368
468
  return { value };
369
469
  }
370
- function flattenDelimitedRow(row, flattenRules) {
470
+ function flattenDelimitedRow(row, flattenRules, misses) {
371
471
  const result = { ...row };
372
472
  for (const [outColumn, path] of Object.entries(flattenRules)) {
373
473
  const dot = path.indexOf(".");
374
- if (dot <= 0) continue;
474
+ if (dot <= 0) {
475
+ recordFlattenMiss(misses, outColumn);
476
+ continue;
477
+ }
375
478
  const column = path.slice(0, dot);
376
479
  const cell = row[column];
377
- if (typeof cell !== "string") continue;
480
+ if (typeof cell !== "string") {
481
+ recordFlattenMiss(misses, outColumn);
482
+ continue;
483
+ }
378
484
  let parsed;
379
485
  try {
380
486
  parsed = JSON.parse(cell);
381
487
  } catch {
488
+ recordFlattenMiss(misses, outColumn);
382
489
  continue;
383
490
  }
384
491
  const value = getNestedValue(parsed, path.slice(dot + 1));
385
- if (value === null || value === void 0) continue;
492
+ if (value === null || value === void 0) {
493
+ recordFlattenMiss(misses, outColumn);
494
+ continue;
495
+ }
386
496
  result[outColumn] = typeof value === "object" ? JSON.stringify(value) : String(value);
387
497
  }
388
498
  return result;
389
499
  }
500
+ function recordFlattenMiss(misses, outColumn) {
501
+ if (!misses) return;
502
+ misses[outColumn] = (misses[outColumn] ?? 0) + 1;
503
+ }
390
504
  function buildNestedTree(rows) {
391
505
  const objects = rows.filter((row) => row !== null && typeof row === "object" && !Array.isArray(row));
392
506
  return buildObjectChildren(objects, "");
393
507
  }
508
+ function buildColumnNestedTree(cells) {
509
+ const objects = cells.filter((cell) => cell !== null && typeof cell === "object" && !Array.isArray(cell));
510
+ const arrays = cells.filter(Array.isArray);
511
+ if (objects.length >= arrays.length) {
512
+ return objects.length > 0 ? buildObjectChildren(objects, "") : [];
513
+ }
514
+ const elements = arrays.flat();
515
+ const elementObjects = elements.filter(
516
+ (value) => value !== null && typeof value === "object" && !Array.isArray(value)
517
+ );
518
+ const elementKind = elementObjects.length > 0 ? "object" : "primitive";
519
+ const node = { path: "", name: "", kind: "array", elementKind, nonEmpty: elements.length > 0 };
520
+ if (elementKind === "object") node.children = buildObjectChildren(elementObjects, "");
521
+ return [node];
522
+ }
394
523
  function buildObjectChildren(objects, parentPath) {
395
524
  const keys = [];
396
525
  const seen = /* @__PURE__ */ new Set();
@@ -416,8 +545,18 @@ function buildNode(path, name, values, nonEmpty) {
416
545
  if (structural > 0 && structural >= values.length - structural) {
417
546
  if (arrays.length > objects.length) {
418
547
  const elementValues = arrays.flat();
419
- const elementKind = elementValues.some((value) => value !== null && typeof value === "object") ? "object" : "primitive";
420
- return { path, name, kind: "array", elementKind, nonEmpty };
548
+ const elementObjects = elementValues.filter(
549
+ (value) => value !== null && typeof value === "object" && !Array.isArray(value)
550
+ );
551
+ const elementKind = elementObjects.length > 0 ? "object" : "primitive";
552
+ return {
553
+ path,
554
+ name,
555
+ kind: "array",
556
+ elementKind,
557
+ nonEmpty,
558
+ ...elementKind === "object" ? { children: buildObjectChildren(elementObjects, path) } : {}
559
+ };
421
560
  }
422
561
  return { path, name, kind: "object", children: buildObjectChildren(objects, path), nonEmpty };
423
562
  }
@@ -485,13 +624,11 @@ function isStrongDateTime(value) {
485
624
  return /^\d{4}[-/]\d{1,2}[-/]\d{1,2}(?:[ T]\d{1,2}:\d{2}(?::\d{2}(?:\.\d{1,3})?)?(?:Z|[+-]\d{2}:?\d{2})?)?$/.test(value.trim());
486
625
  }
487
626
 
488
- // src/commands/data-integration/local-data/input.ts
489
- var MAX_STREAMING_FILE_BYTES = 200 * 1024 * 1024;
490
- var MAX_XLS_FILE_BYTES = 50 * 1024 * 1024;
627
+ // src/commands/data-integration/input.ts
491
628
  var XLSX = XLSXMod.default ?? XLSXMod;
492
629
  var require3 = createRequire2(import.meta.url);
493
630
  var ExcelWorksheetReader = require3("exceljs/lib/stream/xlsx/worksheet-reader");
494
- async function inspectLocalDataInput(filePath) {
631
+ function resolveLocalDataInputMeta(filePath) {
495
632
  let format = resolveFormat(filePath);
496
633
  let delimiter;
497
634
  let encoding;
@@ -512,33 +649,43 @@ async function inspectLocalDataInput(filePath) {
512
649
  location: { field: "input-file" }
513
650
  });
514
651
  }
515
- const maxBytes = format === "xls" ? MAX_XLS_FILE_BYTES : MAX_STREAMING_FILE_BYTES;
516
- if (sizeBytes > maxBytes) {
517
- throw new CliValidationError(
518
- `${format.toUpperCase()} input exceeds the supported file size limit.`,
519
- {
520
- code: "LOCAL_DATA_FILE_TOO_LARGE",
521
- hint: `Split the file below ${format === "xls" ? "50" : "200"} MB and retry.`,
522
- location: { field: "input-file" }
523
- }
524
- );
525
- }
526
652
  if (format !== "xls" && format !== "xlsx" && !encoding) {
527
653
  encoding = detectEncoding(filePath);
528
654
  }
655
+ return {
656
+ filePath,
657
+ format,
658
+ sizeBytes,
659
+ ...delimiter ? { delimiter } : {},
660
+ ...encoding ? { encoding } : {}
661
+ };
662
+ }
663
+ async function inspectLocalDataInput(filePath) {
664
+ const meta = resolveLocalDataInputMeta(filePath);
665
+ emitSizeWarning(meta.filePath, meta.format, meta.sizeBytes, meta.encoding);
529
666
  try {
530
667
  return {
531
- filePath,
532
- format,
533
- sizeBytes,
668
+ ...meta,
534
669
  sha256: await sha256File(filePath),
535
- dataSets: await discoverDataSets(filePath, format),
536
- ...delimiter ? { delimiter } : {},
537
- ...encoding ? { encoding } : {}
670
+ dataSets: await discoverDataSets(filePath, meta.format)
538
671
  };
539
672
  } catch (error) {
540
673
  if (error instanceof CliValidationError) throw error;
541
- throw localDataParseError(format);
674
+ throw localDataParseError(meta.format);
675
+ }
676
+ }
677
+ function emitSizeWarning(filePath, format, sizeBytes, encoding) {
678
+ const assessment = assessFileSize(basename(filePath), format, sizeBytes, encoding);
679
+ if (assessment.rejected) {
680
+ throw new CliValidationError("XLS input exceeds the supported file size limit.", {
681
+ code: "LOCAL_DATA_FILE_TOO_LARGE",
682
+ hint: "XLS files larger than 1 GB are not supported; convert the workbook to XLSX or split it first.",
683
+ location: { field: "input-file" }
684
+ });
685
+ }
686
+ if (assessment.warning) {
687
+ process.stderr.write(`${assessment.warning}
688
+ `);
542
689
  }
543
690
  }
544
691
  function selectDataSet(input, requested) {
@@ -562,6 +709,24 @@ function selectDataSet(input, requested) {
562
709
  }
563
710
  return input.dataSets[0];
564
711
  }
712
+ var LocalDataRowCallbackError = class extends Error {
713
+ constructor(cause) {
714
+ super(cause instanceof Error ? cause.message : String(cause));
715
+ this.cause = cause;
716
+ this.name = "LocalDataRowCallbackError";
717
+ }
718
+ cause;
719
+ };
720
+ function wrapRowCallback(onRow) {
721
+ return async (row, rowNumber) => {
722
+ try {
723
+ await onRow(row, rowNumber);
724
+ } catch (error) {
725
+ if (error instanceof CliValidationError) throw error;
726
+ throw new LocalDataRowCallbackError(error);
727
+ }
728
+ };
729
+ }
565
730
  async function streamLocalDataRows(input, dataSet, onRow, options = {}) {
566
731
  const opts = {
567
732
  delimiter: options.delimiter ?? input.delimiter ?? (input.format === "tsv" ? " " : ","),
@@ -569,24 +734,30 @@ async function streamLocalDataRows(input, dataSet, onRow, options = {}) {
569
734
  headerNames: options.headerNames,
570
735
  noHeader: options.noHeader,
571
736
  flattenRules: options.flattenRules,
572
- mergeSheets: options.mergeSheets
737
+ flattenMisses: options.flattenMisses,
738
+ mergeSheets: options.mergeSheets,
739
+ warnRagged: options.warnRagged
573
740
  };
741
+ const wrappedRow = wrapRowCallback(onRow);
574
742
  try {
575
743
  switch (input.format) {
576
744
  case "csv":
577
745
  case "tsv":
578
- return await streamDelimited(input.filePath, onRow, opts);
746
+ return await streamDelimited(input.filePath, wrappedRow, opts);
579
747
  case "jsonl":
580
- return await streamJsonLines(input.filePath, onRow, opts);
748
+ return await streamJsonLines(input.filePath, wrappedRow, opts);
581
749
  case "json":
582
- return await streamJson(input.filePath, dataSet.selector ?? "$", onRow, opts);
750
+ return await streamJson(input.filePath, dataSet.selector ?? "$", wrappedRow, opts);
583
751
  case "xlsx":
584
- return await streamXlsx(input.filePath, dataSet.label, onRow, opts);
752
+ return await streamXlsx(input.filePath, dataSet.label, wrappedRow, opts);
585
753
  case "xls":
586
- return await streamXls(input.filePath, dataSet.label, onRow, opts);
754
+ return await streamXls(input.filePath, dataSet.label, wrappedRow, opts);
587
755
  }
588
756
  } catch (error) {
589
757
  if (error instanceof CliValidationError) throw error;
758
+ if (error instanceof LocalDataRowCallbackError) {
759
+ throw error.cause instanceof Error ? error.cause : error;
760
+ }
590
761
  throw localDataParseError(input.format);
591
762
  }
592
763
  }
@@ -774,29 +945,36 @@ async function streamDelimited(filePath, onRow, options) {
774
945
  quote: delimiter === " " ? null : '"',
775
946
  relax_column_count: true,
776
947
  skip_empty_lines: true,
777
- trim: true,
778
- columns: headerNames ? headerNames : (headers) => dedupeHeaders(headers.map((header) => String(header).trim()))
948
+ trim: true
779
949
  });
780
950
  decodeTextStream(filePath, encoding).pipe(parser);
781
951
  let count = 0;
782
- for await (const value of parser) {
952
+ let widthMismatches = 0;
953
+ let resolvedHeaders = headerNames;
954
+ for await (const raw of parser) {
955
+ const values = raw;
956
+ if (!resolvedHeaders) {
957
+ resolvedHeaders = dedupeHeaders(values.map((header) => String(header).trim()));
958
+ continue;
959
+ }
960
+ if (values.length !== resolvedHeaders.length) widthMismatches += 1;
783
961
  count += 1;
784
- const row = normalizeDelimitedRow(value, headerNames);
962
+ const row = {};
963
+ for (let index = 0; index < resolvedHeaders.length; index += 1) {
964
+ row[resolvedHeaders[index]] = values[index] ?? null;
965
+ }
785
966
  await onRow(
786
- options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
967
+ options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules, options.flattenMisses) : row,
787
968
  count
788
969
  );
789
970
  }
790
- return count;
791
- }
792
- function normalizeDelimitedRow(value, headerNames) {
793
- if (!headerNames || value === null || typeof value !== "object" || Array.isArray(value)) {
794
- return normalizeRow(value);
971
+ if (widthMismatches > 0 && options.warnRagged !== false) {
972
+ process.stderr.write(
973
+ `Warning: ${widthMismatches} record(s) had a column count different from the header row; extra fields were dropped and missing fields were treated as empty.
974
+ `
975
+ );
795
976
  }
796
- const source = value;
797
- const row = {};
798
- for (const header of headerNames) row[header] = source[header] ?? null;
799
- return row;
977
+ return count;
800
978
  }
801
979
  async function streamJsonLines(filePath, onRow, options) {
802
980
  let count = 0;
@@ -813,7 +991,7 @@ async function streamJsonLines(filePath, onRow, options) {
813
991
  location: { record: count }
814
992
  });
815
993
  }
816
- await onRow(flattenLocalDataRow(value, options.flattenRules), count);
994
+ await onRow(flattenLocalDataRow(value, options.flattenRules, options.flattenMisses), count);
817
995
  }
818
996
  return count;
819
997
  }
@@ -825,7 +1003,7 @@ async function streamJson(filePath, selector, onRow, options) {
825
1003
  const chain = selector === "$" || selector === "$object" ? source.pipe(parser).pipe(streamer) : source.pipe(parser).pipe(Pick.pick({ filter: selector })).pipe(streamer);
826
1004
  for await (const item of chain) {
827
1005
  count += 1;
828
- await onRow(flattenLocalDataRow(item.value, options.flattenRules), count);
1006
+ await onRow(flattenLocalDataRow(item.value, options.flattenRules, options.flattenMisses), count);
829
1007
  }
830
1008
  return count;
831
1009
  }
@@ -869,7 +1047,11 @@ async function streamXlsx(filePath, sheetName, onRow, options) {
869
1047
  }
870
1048
  if (values.every(isMissing)) continue;
871
1049
  count += 1;
872
- await onRow(Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null])), count);
1050
+ const row = Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null]));
1051
+ await onRow(
1052
+ options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
1053
+ count
1054
+ );
873
1055
  }
874
1056
  }
875
1057
  return count;
@@ -1021,17 +1203,15 @@ async function streamXls(filePath, sheetName, onRow, options) {
1021
1203
  for (const values of rows.slice(start)) {
1022
1204
  if (values.every(isMissing)) continue;
1023
1205
  count += 1;
1024
- await onRow(Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null])), count);
1206
+ const row = Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null]));
1207
+ await onRow(
1208
+ options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
1209
+ count
1210
+ );
1025
1211
  }
1026
1212
  }
1027
1213
  return count;
1028
1214
  }
1029
- function normalizeRow(value) {
1030
- if (value !== null && typeof value === "object" && !Array.isArray(value)) {
1031
- return value;
1032
- }
1033
- return { value };
1034
- }
1035
1215
  function normalizeExcelValue(value) {
1036
1216
  if (value instanceof Date) return value;
1037
1217
  if (value && typeof value === "object") {
@@ -1074,9 +1254,16 @@ async function firstNonWhitespaceCharacter(filePath, encoding) {
1074
1254
  return void 0;
1075
1255
  }
1076
1256
 
1077
- // src/commands/data-integration/local-data/mapping.ts
1257
+ // src/commands/data-integration/mapping.ts
1078
1258
  import { readFileSync } from "fs";
1259
+
1260
+ // src/commands/data-integration/types.ts
1261
+ var MAPPING_VERSION = "ae-data-integration-mapping/v1";
1262
+
1263
+ // src/commands/data-integration/mapping.ts
1079
1264
  var VALID_PROPERTY_NAME = /^[a-z][a-z0-9_]{0,49}$/;
1265
+ var VALID_CHILD_PROPERTY_NAME = /^[a-z][a-z0-9_]{0,49}(\.[a-z][a-z0-9_]{0,49})+$/;
1266
+ var FLATTEN_SUPPORTED_FORMATS = ["csv", "tsv", "json", "jsonl", "xls", "xlsx"];
1080
1267
  function readLocalDataMapping(raw, options) {
1081
1268
  const trimmed = raw.trim();
1082
1269
  let text;
@@ -1106,8 +1293,8 @@ function readLocalDataMapping(raw, options) {
1106
1293
  return value;
1107
1294
  }
1108
1295
  function validateMapping(value, options) {
1109
- if (!isRecord(value) || value.version !== "ae-local-data-mapping/v1") {
1110
- throw mappingError("Mapping version must be ae-local-data-mapping/v1.");
1296
+ if (!isRecord(value) || value.version !== MAPPING_VERSION) {
1297
+ throw mappingError(`Mapping version must be ${MAPPING_VERSION}.`);
1111
1298
  }
1112
1299
  const sha256Valid = typeof value.source?.sha256 === "string" && (options?.sourceWildcard ? value.source.sha256 === "*" || /^[a-f0-9]{64}$/i.test(value.source.sha256) : /^[a-f0-9]{64}$/i.test(value.source.sha256));
1113
1300
  if (!isRecord(value.source) || !sha256Valid || !["csv", "tsv", "json", "jsonl", "xls", "xlsx"].includes(String(value.source.format)) || typeof value.source.data_set !== "string" || !value.source.data_set) {
@@ -1137,12 +1324,28 @@ function validateMapping(value, options) {
1137
1324
  }
1138
1325
  if (!Array.isArray(value.properties)) throw mappingError("Mapping properties must be an array.");
1139
1326
  const targets = /* @__PURE__ */ new Set();
1327
+ const propertyTypes = /* @__PURE__ */ new Map();
1140
1328
  for (const property of value.properties) {
1141
- if (!isRecord(property) || typeof property.source !== "string" || typeof property.target !== "string" || !VALID_PROPERTY_NAME.test(property.target) || !["number", "string", "boolean", "datetime", "list", "object"].includes(String(property.type)) || property.transform !== void 0 && !["stringify", "number", "boolean", "json"].includes(String(property.transform)) || property.value_mapping !== void 0 && !isStringMap(property.value_mapping) || property.time_format !== void 0 && (typeof property.time_format !== "string" || !property.time_format.trim() || property.time_format.length > 64) || property.desc !== void 0 && (typeof property.desc !== "string" || !property.desc.trim() || property.desc.length > 200)) {
1329
+ if (!isRecord(property) || typeof property.source !== "string" || typeof property.target !== "string" || !(VALID_PROPERTY_NAME.test(property.target) || VALID_CHILD_PROPERTY_NAME.test(property.target)) || !["number", "string", "boolean", "datetime", "list", "object", "array_row"].includes(String(property.type)) || property.transform !== void 0 && !["stringify", "number", "boolean", "json"].includes(String(property.transform)) || property.value_mapping !== void 0 && !isStringMap(property.value_mapping) || property.time_format !== void 0 && (typeof property.time_format !== "string" || !property.time_format.trim() || property.time_format.length > 64) || property.desc !== void 0 && (typeof property.desc !== "string" || !property.desc.trim() || property.desc.length > 200)) {
1142
1330
  throw mappingError("Every property mapping needs a source, legal AE target name, and supported type.");
1143
1331
  }
1144
1332
  if (targets.has(property.target)) throw mappingError("Property target names must be unique.");
1145
1333
  targets.add(property.target);
1334
+ propertyTypes.set(property.target, { type: String(property.type), source: property.source });
1335
+ }
1336
+ for (const property of value.properties) {
1337
+ if (!property.target.includes(".")) continue;
1338
+ const parentName = property.target.split(".")[0];
1339
+ const parent = propertyTypes.get(parentName);
1340
+ if (!parent || parent.type !== "object" && parent.type !== "array_row") {
1341
+ throw mappingError(`The sub-property "${property.target}" references "${parentName}", which is not an object/array_row property in this mapping.`);
1342
+ }
1343
+ if (property.type === "object" || property.type === "array_row") {
1344
+ throw mappingError(`The sub-property "${property.target}" must be scalar or list, not ${property.type}.`);
1345
+ }
1346
+ if (property.source !== parent.source) {
1347
+ throw mappingError(`The sub-property "${property.target}" must read from the same source column as its parent "${parentName}".`);
1348
+ }
1146
1349
  }
1147
1350
  if (value.time_format !== void 0 && (typeof value.time_format !== "string" || !value.time_format.trim() || value.time_format.length > 64)) {
1148
1351
  throw mappingError("time_format must be a non-empty string of at most 64 characters.");
@@ -1184,6 +1387,9 @@ function validateMapping(value, options) {
1184
1387
  }
1185
1388
  if (value.flatten_rules !== void 0) {
1186
1389
  if (!isRecord(value.flatten_rules)) throw mappingError("flatten_rules must be an object of { column: dot.path }.");
1390
+ if (!FLATTEN_SUPPORTED_FORMATS.includes(String(value.source.format))) {
1391
+ throw mappingError(`flatten_rules are not supported for ${String(value.source.format)} input.`);
1392
+ }
1187
1393
  for (const [column, path] of Object.entries(value.flatten_rules)) {
1188
1394
  if (!VALID_PROPERTY_NAME.test(column) || typeof path !== "string" || !path.trim()) {
1189
1395
  throw mappingError("flatten_rules keys must be legal AE property names and values must be non-empty dot paths.");
@@ -1227,6 +1433,31 @@ function validateMapping(value, options) {
1227
1433
  function isValidAeName(value) {
1228
1434
  return VALID_PROPERTY_NAME.test(value);
1229
1435
  }
1436
+ function sourceColumns(mapping) {
1437
+ const columns = /* @__PURE__ */ new Set();
1438
+ const flattenOut = new Set(Object.keys(mapping.flatten_rules ?? {}));
1439
+ for (const property of mapping.properties) {
1440
+ if (!flattenOut.has(property.source)) columns.add(property.source);
1441
+ }
1442
+ for (const path of Object.values(mapping.flatten_rules ?? {})) {
1443
+ const root = path.split(".")[0];
1444
+ if (root) columns.add(root);
1445
+ }
1446
+ const systemFields = [
1447
+ mapping.account_id_field,
1448
+ mapping.distinct_id_field,
1449
+ mapping.record_type_field,
1450
+ mapping.event_name_field,
1451
+ mapping.time.field,
1452
+ mapping.ip_field,
1453
+ mapping.uuid_field,
1454
+ mapping.zone_offset_field
1455
+ ];
1456
+ for (const field of systemFields) if (field) columns.add(field);
1457
+ for (const column of mapping.exclude_columns ?? []) columns.add(column);
1458
+ for (const header of mapping.headers ?? []) columns.add(header);
1459
+ return [...columns].sort();
1460
+ }
1230
1461
  function mappingError(message) {
1231
1462
  return new CliValidationError(message, {
1232
1463
  code: "LOCAL_DATA_MAPPING_INVALID",
@@ -1253,7 +1484,7 @@ function isRecord(value) {
1253
1484
  return value !== null && typeof value === "object" && !Array.isArray(value);
1254
1485
  }
1255
1486
 
1256
- // src/commands/data-integration/local-data/multi.ts
1487
+ // src/commands/data-integration/multi.ts
1257
1488
  var PROPERTY_TYPES = /* @__PURE__ */ new Set([
1258
1489
  "number",
1259
1490
  "string",
@@ -1398,9 +1629,46 @@ function isRecord2(value) {
1398
1629
  return value !== null && typeof value === "object" && !Array.isArray(value);
1399
1630
  }
1400
1631
 
1401
- // src/commands/data-integration/local-data/profile.ts
1632
+ // src/commands/data-integration/profile.ts
1402
1633
  import { createHash as createHash2, randomInt } from "crypto";
1403
1634
  import { basename as basename2, extname as extname2 } from "path";
1635
+
1636
+ // src/commands/data-integration/field-spec.ts
1637
+ import { isIP } from "net";
1638
+ function stripQuotes(value) {
1639
+ if (typeof value !== "string") return value;
1640
+ const text = value;
1641
+ if (text.length >= 2 && (text.startsWith('"') && text.endsWith('"') || text.startsWith("'") && text.endsWith("'"))) {
1642
+ return text.slice(1, -1).trim();
1643
+ }
1644
+ return text.trim();
1645
+ }
1646
+ var UUID_PATTERN = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
1647
+ function isValidUuid(value) {
1648
+ if (typeof value !== "string") return false;
1649
+ return UUID_PATTERN.test(stripQuotes(value));
1650
+ }
1651
+ function isValidIp(value) {
1652
+ if (typeof value !== "string") return false;
1653
+ return isIP(stripQuotes(value)) !== 0;
1654
+ }
1655
+ function isPrivateIp(value) {
1656
+ if (typeof value !== "string") return false;
1657
+ const text = stripQuotes(value);
1658
+ const version = isIP(text);
1659
+ if (version === 6) {
1660
+ const lower = text.toLowerCase();
1661
+ if (lower === "::1") return true;
1662
+ const first2 = lower.split(":")[0];
1663
+ return first2.startsWith("fc") || first2.startsWith("fd") || /^fe[89ab]/.test(lower);
1664
+ }
1665
+ if (version !== 4) return false;
1666
+ const octets = text.split(".").map((part) => Number(part));
1667
+ const [first, second] = octets;
1668
+ return first === 10 || first === 172 && second >= 16 && second <= 31 || first === 192 && second === 168 || first === 127 || first === 169 && second === 254;
1669
+ }
1670
+
1671
+ // src/commands/data-integration/profile.ts
1404
1672
  var UNIQUE_SAMPLE_LIMIT = 1e4;
1405
1673
  var GLOBAL_UNIQUE_SAMPLE_LIMIT = 2e5;
1406
1674
  var IDENTITY_MAX_LENGTH = 128;
@@ -1424,6 +1692,8 @@ var DISTINCT_NAMES = ["#distinct_id", "distinct_id", "distinctid", "device_id",
1424
1692
  var TIME_NAMES = ["#time", "time", "timestamp", "event_time", "created_at", "occurred_at", "datetime", "date", "\u65F6\u95F4", "\u4E8B\u4EF6\u65F6\u95F4", "\u53D1\u751F\u65F6\u95F4", "\u521B\u5EFA\u65F6\u95F4", "\u4E0B\u5355\u65F6\u95F4", "\u8BA2\u5355\u65F6\u95F4"];
1425
1693
  var EVENT_NAMES = ["#event_name", "event_name", "event", "action", "activity", "\u4E8B\u4EF6\u540D", "\u4E8B\u4EF6\u540D\u79F0", "\u4E8B\u4EF6", "\u884C\u4E3A", "\u52A8\u4F5C"];
1426
1694
  var TYPE_NAMES = ["#type", "record_type", "data_type", "\u64CD\u4F5C\u7C7B\u578B"];
1695
+ var IP_NAMES = ["#ip", "ip", "ip_address", "ipaddress", "client_ip", "clientip", "remote_addr", "remoteaddr", "ip\u5730\u5740", "\u5BA2\u6237\u7AEFip"];
1696
+ var UUID_NAMES = ["#uuid", "uuid", "event_uuid", "eventuuid", "request_uuid", "requestuuid", "\u552F\u4E00\u6807\u8BC6", "\u552F\u4E00id"];
1427
1697
  async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai", options = {}) {
1428
1698
  const columns = /* @__PURE__ */ new Map();
1429
1699
  const recognizedRecordTypes = /* @__PURE__ */ new Set();
@@ -1433,7 +1703,7 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1433
1703
  options.collectNestedTree && (input.format === "json" || input.format === "jsonl")
1434
1704
  );
1435
1705
  const collectDelimitedTree = Boolean(
1436
- options.collectNestedTree && (input.format === "csv" || input.format === "tsv")
1706
+ options.collectNestedTree && (input.format === "csv" || input.format === "tsv" || input.format === "xlsx" || input.format === "xls")
1437
1707
  );
1438
1708
  let nestedObjects = [];
1439
1709
  let nestedSeen = 0;
@@ -1464,6 +1734,9 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1464
1734
  uniqueOverflow: false,
1465
1735
  timeParseCount: 0,
1466
1736
  timeFormatCounts: /* @__PURE__ */ new Map(),
1737
+ uuidValidCount: 0,
1738
+ ipValidCount: 0,
1739
+ lanIpCount: 0,
1467
1740
  samples: [],
1468
1741
  sampleSet: /* @__PURE__ */ new Set()
1469
1742
  };
@@ -1486,24 +1759,27 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1486
1759
  accumulator.uniqueOverflow = true;
1487
1760
  }
1488
1761
  if (options.collectSamples) recordSample(accumulator, value);
1489
- if (collectDelimitedTree && typeof value === "string" && value.trim().startsWith("{")) {
1490
- try {
1491
- const parsed = JSON.parse(value);
1492
- if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) {
1493
- let sampler = delimitedNested.get(name);
1494
- if (!sampler) {
1495
- sampler = { seen: 0, objects: [] };
1496
- delimitedNested.set(name, sampler);
1497
- }
1498
- sampler.seen += 1;
1499
- if (sampler.objects.length < NESTED_TREE_SAMPLE_LIMIT) {
1500
- sampler.objects.push(parsed);
1501
- } else {
1502
- const slot = randomInt(sampler.seen);
1503
- if (slot < NESTED_TREE_SAMPLE_LIMIT) sampler.objects[slot] = parsed;
1762
+ if (collectDelimitedTree && typeof value === "string") {
1763
+ const trimmed = value.trim();
1764
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
1765
+ try {
1766
+ const parsed = JSON.parse(value);
1767
+ if (parsed !== null && typeof parsed === "object") {
1768
+ let sampler = delimitedNested.get(name);
1769
+ if (!sampler) {
1770
+ sampler = { seen: 0, values: [] };
1771
+ delimitedNested.set(name, sampler);
1772
+ }
1773
+ sampler.seen += 1;
1774
+ if (sampler.values.length < NESTED_TREE_SAMPLE_LIMIT) {
1775
+ sampler.values.push(parsed);
1776
+ } else {
1777
+ const slot = randomInt(sampler.seen);
1778
+ if (slot < NESTED_TREE_SAMPLE_LIMIT) sampler.values[slot] = parsed;
1779
+ }
1504
1780
  }
1781
+ } catch {
1505
1782
  }
1506
- } catch {
1507
1783
  }
1508
1784
  }
1509
1785
  if (isParseableTime(value, matchesName(name, TIME_NAMES))) {
@@ -1515,6 +1791,11 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1515
1791
  }
1516
1792
  }
1517
1793
  }
1794
+ if (matchesName(name, UUID_NAMES) && isValidUuid(value)) accumulator.uuidValidCount += 1;
1795
+ if (matchesName(name, IP_NAMES) && isValidIp(value)) {
1796
+ accumulator.ipValidCount += 1;
1797
+ if (isPrivateIp(value)) accumulator.lanIpCount += 1;
1798
+ }
1518
1799
  if (matchesName(name, TYPE_NAMES)) {
1519
1800
  const normalized = normalizeRecordType(value);
1520
1801
  if (normalized) recognizedRecordTypes.add(normalized);
@@ -1527,23 +1808,25 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1527
1808
  headerNames: options.headerNames,
1528
1809
  noHeader: options.noHeader,
1529
1810
  flattenRules: options.flattenRules,
1530
- mergeSheets: options.mergeSheets
1811
+ mergeSheets: options.mergeSheets,
1812
+ warnRagged: options.warnRagged
1531
1813
  }
1532
1814
  );
1533
1815
  const delimitedNestedTree = /* @__PURE__ */ new Map();
1534
1816
  for (const [columnName, sampler] of delimitedNested) {
1535
- if (sampler.objects.length > 0) delimitedNestedTree.set(columnName, buildNestedTree(sampler.objects));
1817
+ if (sampler.values.length > 0) delimitedNestedTree.set(columnName, buildColumnNestedTree(sampler.values));
1536
1818
  }
1537
1819
  const columnProfiles = [];
1538
1820
  const timeFormatByColumn = /* @__PURE__ */ new Map();
1539
1821
  for (const column of columns.values()) {
1540
1822
  const profile = formatColumnProfile(column, rowCount, options.collectSamples ?? false);
1541
- const nestedTree = delimitedNestedTree.get(column.name);
1542
- if (nestedTree && nestedTree.length > 0) profile.nested_tree = nestedTree;
1823
+ const nestedTree2 = delimitedNestedTree.get(column.name);
1824
+ if (nestedTree2 && nestedTree2.length > 0) profile.nested_tree = nestedTree2;
1543
1825
  columnProfiles.push(profile);
1544
1826
  const dominantFormat = dominantTimeFormat(column);
1545
1827
  if (dominantFormat) timeFormatByColumn.set(column.name, dominantFormat);
1546
1828
  }
1829
+ const nestedTree = collectNestedTree && nestedObjects.length > 0 ? buildNestedTree(nestedObjects) : void 0;
1547
1830
  const identityCandidates = findIdentityCandidates(columnProfiles);
1548
1831
  const recommendedMapping = recommendMapping({
1549
1832
  input,
@@ -1553,9 +1836,21 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1553
1836
  recognizedRecordTypes,
1554
1837
  sourceTimezone,
1555
1838
  headerNames: options.headerNames,
1556
- timeFormatByColumn
1839
+ timeFormatByColumn,
1840
+ nestedTree
1557
1841
  });
1558
1842
  const warnings = [...recommendedMapping.warnings ?? []];
1843
+ for (const column of columns.values()) {
1844
+ if (matchesName(column.name, UUID_NAMES) && column.nonMissing > 0) {
1845
+ const invalid = column.nonMissing - column.uuidValidCount;
1846
+ if (invalid > 0) warnings.push(`${column.name}: ${invalid} non-empty value(s) are not standard 36-character UUIDs; #uuid requires the standard UUID format.`);
1847
+ }
1848
+ if (matchesName(column.name, IP_NAMES) && column.nonMissing > 0) {
1849
+ const invalid = column.nonMissing - column.ipValidCount;
1850
+ if (invalid > 0) warnings.push(`${column.name}: ${invalid} non-empty value(s) are not valid IPv4 or IPv6 addresses.`);
1851
+ if (column.lanIpCount > 0) warnings.push(`${column.name}: ${column.lanIpCount} value(s) are private/LAN IP addresses; AE cannot resolve geo information for them.`);
1852
+ }
1853
+ }
1559
1854
  if (rowCount === 0) warnings.push("The selected data set contains no data rows.");
1560
1855
  return {
1561
1856
  version: "ae-local-data-profile/v1",
@@ -1574,7 +1869,7 @@ async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai"
1574
1869
  (recommendedMapping.account_id_field || recommendedMapping.distinct_id_field) && recommendedMapping.time.field
1575
1870
  ),
1576
1871
  warnings,
1577
- ...collectNestedTree && nestedObjects.length > 0 ? { nested_tree: buildNestedTree(nestedObjects) } : {}
1872
+ ...nestedTree ? { nested_tree: nestedTree } : {}
1578
1873
  };
1579
1874
  }
1580
1875
  function normalizeAeName(input, fallback) {
@@ -1647,24 +1942,12 @@ function recommendMapping(input) {
1647
1942
  warnings.push("No event-name column was found; review the generated default event name.");
1648
1943
  }
1649
1944
  const reserved = new Set([account?.name, distinct?.name, time?.name, event?.name, recordType?.name].filter(Boolean));
1650
- const usedTargets = /* @__PURE__ */ new Set();
1651
- const properties = input.columns.filter((column) => !reserved.has(column.name)).map((column, index) => {
1652
- const baseTarget = normalizeAeName(column.name, `field_${index + 1}`);
1653
- let target = baseTarget;
1654
- let suffix = 2;
1655
- while (usedTargets.has(target)) target = `${baseTarget.slice(0, 46)}_${suffix++}`;
1656
- usedTargets.add(target);
1657
- const type = mappingType(column.inferred_type);
1658
- return {
1659
- source: column.name,
1660
- target,
1661
- type,
1662
- ...type === "object" || type === "list" ? { transform: "json" } : {}
1663
- };
1664
- });
1945
+ const recordRoots = /* @__PURE__ */ new Map();
1946
+ for (const node of input.nestedTree ?? []) recordRoots.set(node.name, node);
1947
+ const { properties, flattenRules } = recommendProperties(input.columns, reserved, recordRoots, warnings);
1665
1948
  const defaultEventSource = input.dataSet.kind === "sheet" ? input.dataSet.label : basename2(input.input.filePath, extname2(input.input.filePath));
1666
1949
  return {
1667
- version: "ae-local-data-mapping/v1",
1950
+ version: MAPPING_VERSION,
1668
1951
  source: {
1669
1952
  sha256: input.input.sha256,
1670
1953
  format: input.input.format,
@@ -1684,6 +1967,7 @@ function recommendMapping(input) {
1684
1967
  ...recordType ? { record_type_field: recordType.name } : {},
1685
1968
  ...event ? { event_name_field: event.name } : {},
1686
1969
  ...mode !== "user_set" && !event ? { default_event_name: normalizeAeName(defaultEventSource, "local_event") } : {},
1970
+ ...Object.keys(flattenRules).length > 0 ? { flatten_rules: flattenRules } : {},
1687
1971
  properties,
1688
1972
  ...warnings.length > 0 ? { warnings } : {}
1689
1973
  };
@@ -1767,9 +2051,120 @@ function normalizeRecordType(value) {
1767
2051
  };
1768
2052
  return aliases[normalized];
1769
2053
  }
2054
+ function recommendProperties(columns, reserved, recordRoots, warnings) {
2055
+ const properties = [];
2056
+ const flattenRules = {};
2057
+ const usedTargets = /* @__PURE__ */ new Set();
2058
+ const claim = (desired) => {
2059
+ let target = desired;
2060
+ let suffix = 2;
2061
+ while (usedTargets.has(target)) target = `${desired.slice(0, 46)}_${suffix++}`;
2062
+ usedTargets.add(target);
2063
+ return target;
2064
+ };
2065
+ const state = { properties, flattenRules, usedTargets, warnings, claim };
2066
+ columns.filter((column) => !reserved.has(column.name)).forEach((column, index) => {
2067
+ const baseName = normalizeAeName(column.name, `field_${index + 1}`);
2068
+ const valueNode = columnValueNode(column, recordRoots);
2069
+ if (!valueNode) {
2070
+ const type = mappingType(column.inferred_type);
2071
+ const target = claim(baseName);
2072
+ properties.push({ source: column.name, target, type, ...isContainerType(type) ? { transform: "json" } : {} });
2073
+ return;
2074
+ }
2075
+ if (valueNode.kind === "primitive") {
2076
+ const type = mappingType(valueNode.inferredType ?? column.inferred_type);
2077
+ const target = claim(baseName);
2078
+ properties.push({ source: column.name, target, type });
2079
+ return;
2080
+ }
2081
+ if (valueNode.kind === "object") {
2082
+ recommendObject(state, baseName, column.name, valueNode.children ?? [], column.name);
2083
+ return;
2084
+ }
2085
+ recommendArray(state, baseName, column.name, valueNode, column.name);
2086
+ });
2087
+ return { properties, flattenRules };
2088
+ }
2089
+ function columnValueNode(column, recordRoots) {
2090
+ const recordNode = recordRoots.get(column.name);
2091
+ if (recordNode) return recordNode;
2092
+ const tree = column.nested_tree;
2093
+ if (!tree || tree.length === 0) return void 0;
2094
+ if (tree.length === 1 && tree[0].kind === "array") return tree[0];
2095
+ return { path: "", name: column.name, kind: "object", children: tree, nonEmpty: true };
2096
+ }
2097
+ function isScalarNode(node) {
2098
+ return node.kind === "primitive" || node.kind === "array" && node.elementKind === "primitive";
2099
+ }
2100
+ function isContainerType(type) {
2101
+ return type === "object" || type === "list" || type === "array_row";
2102
+ }
2103
+ function scalarPropType(node) {
2104
+ if (node.kind === "array") return "list";
2105
+ switch (node.inferredType) {
2106
+ case "number":
2107
+ return "number";
2108
+ case "boolean":
2109
+ return "boolean";
2110
+ case "datetime":
2111
+ return "datetime";
2112
+ default:
2113
+ return "string";
2114
+ }
2115
+ }
2116
+ function snakeSegment(name) {
2117
+ return normalizeAeName(name, "field");
2118
+ }
2119
+ function recommendObject(state, prefix, dotPath, children, columnName) {
2120
+ const hasComposite = children.some((child) => child.kind === "object" || child.kind === "array" && child.elementKind === "object");
2121
+ if (!hasComposite) {
2122
+ const parentTarget = state.claim(prefix);
2123
+ const source = dotPath === columnName ? columnName : parentTarget;
2124
+ if (source !== columnName) state.flattenRules[parentTarget] = dotPath;
2125
+ state.properties.push({ source, target: parentTarget, type: "object", transform: "json" });
2126
+ for (const child of children) {
2127
+ state.properties.push({ source: parentTarget, target: `${parentTarget}.${snakeSegment(child.name)}`, type: scalarPropType(child) });
2128
+ }
2129
+ return;
2130
+ }
2131
+ for (const child of children) {
2132
+ const childPrefix = `${prefix}_${snakeSegment(child.name)}`;
2133
+ const childDotPath = `${dotPath}.${child.name}`;
2134
+ if (isScalarNode(child)) {
2135
+ const target = state.claim(childPrefix);
2136
+ state.flattenRules[target] = childDotPath;
2137
+ state.properties.push({ source: target, target, type: scalarPropType(child) });
2138
+ } else if (child.kind === "object") {
2139
+ recommendObject(state, childPrefix, childDotPath, child.children ?? [], columnName);
2140
+ } else {
2141
+ recommendArray(state, childPrefix, childDotPath, child, columnName);
2142
+ }
2143
+ }
2144
+ }
2145
+ function recommendArray(state, prefix, dotPath, node, columnName) {
2146
+ if (node.elementKind !== "object") {
2147
+ const target = state.claim(prefix);
2148
+ const source2 = dotPath === columnName ? columnName : target;
2149
+ if (source2 !== columnName) state.flattenRules[target] = dotPath;
2150
+ state.properties.push({ source: source2, target, type: "list", transform: "json" });
2151
+ return;
2152
+ }
2153
+ const parentTarget = state.claim(prefix);
2154
+ const source = dotPath === columnName ? columnName : parentTarget;
2155
+ if (source !== columnName) state.flattenRules[parentTarget] = dotPath;
2156
+ state.properties.push({ source, target: parentTarget, type: "array_row", transform: "json" });
2157
+ for (const field of node.children ?? []) {
2158
+ if (isScalarNode(field)) {
2159
+ state.properties.push({ source: parentTarget, target: `${parentTarget}.${snakeSegment(field.name)}`, type: scalarPropType(field) });
2160
+ } else {
2161
+ state.warnings.push(`${dotPath} element field "${field.name}" is nested; it stays inside the ${parentTarget} array data and is not declared as a sub-property (array-element flattening is not yet supported).`);
2162
+ }
2163
+ }
2164
+ }
1770
2165
  function mappingType(type) {
1771
2166
  if (type === "datetime") return "datetime";
1772
- if (type === "number" || type === "boolean" || type === "list" || type === "object") return type;
2167
+ if (type === "number" || type === "boolean" || type === "object" || type === "list") return type;
1773
2168
  return "string";
1774
2169
  }
1775
2170
  function compatibleTypes(types) {
@@ -1832,7 +2227,7 @@ function dominantTimeFormat(column) {
1832
2227
  return best && bestCount >= total * 0.9 ? best : void 0;
1833
2228
  }
1834
2229
 
1835
- // src/commands/data-integration/local-data/inspect.ts
2230
+ // src/commands/data-integration/inspect.ts
1836
2231
  var HEADERLESS_WARNING = "The first row appears to be data, not a header; columns were auto-named col_1..col_N. Re-run with --headers to supply explicit names.";
1837
2232
  var dataIntegrationInspect = {
1838
2233
  service: "data-integration",
@@ -1847,6 +2242,30 @@ var dataIntegrationInspect = {
1847
2242
  { name: "headerless", type: "boolean", default: false, desc: "Treat the first row as data and auto-generate col_1..col_N names." }
1848
2243
  ],
1849
2244
  risk: "read",
2245
+ // Fast size/time pre-check (stat + format/encoding sniff only — no sha256, no profile).
2246
+ // Lets agents surface the estimate before committing to a multi-minute full inspection.
2247
+ dryRun: async (ctx) => {
2248
+ const files = ctx.list("input-file").map((filePath) => {
2249
+ const meta = resolveLocalDataInputMeta(filePath);
2250
+ const assessment = assessFileSize(basename3(filePath), meta.format, meta.sizeBytes, meta.encoding);
2251
+ return {
2252
+ file: basename3(filePath),
2253
+ format: meta.format,
2254
+ size_bytes: meta.sizeBytes,
2255
+ size: assessment.size,
2256
+ ...assessment.estimatedDuration ? { estimated_duration: assessment.estimatedDuration } : {},
2257
+ ...assessment.warning ? { warning: assessment.warning } : {},
2258
+ ...assessment.reason ? { reason: assessment.reason } : {},
2259
+ ...assessment.memoryRisk ? { memory_risk: true } : {},
2260
+ ...assessment.rejected ? { rejected: true } : {}
2261
+ };
2262
+ });
2263
+ return {
2264
+ version: "ae-local-data-estimate/v1",
2265
+ files,
2266
+ has_large_file: files.some((file) => Boolean(file.warning) || file.rejected)
2267
+ };
2268
+ },
1850
2269
  execute: async (ctx) => {
1851
2270
  const inputFiles = ctx.list("input-file");
1852
2271
  const headerNames = splitHeaders(ctx.str("headers"));
@@ -1940,7 +2359,7 @@ function splitHeaders(raw) {
1940
2359
  return headers.length > 0 ? headers : void 0;
1941
2360
  }
1942
2361
 
1943
- // src/commands/data-integration/local-data/plan.ts
2362
+ // src/commands/data-integration/plan.ts
1944
2363
  import { writeFile } from "fs/promises";
1945
2364
  var VALID_LOCALES = /* @__PURE__ */ new Set(["zh", "en", "ja", "ko"]);
1946
2365
  var EVENT_NAME_RE = /^[a-z][a-z0-9_]*$/;
@@ -1956,6 +2375,8 @@ function toPropType(type) {
1956
2375
  return "array_string";
1957
2376
  case "object":
1958
2377
  return "object";
2378
+ case "array_row":
2379
+ return "array_row";
1959
2380
  case "string":
1960
2381
  return "string";
1961
2382
  }
@@ -1963,13 +2384,16 @@ function toPropType(type) {
1963
2384
  function buildDraftFromMapping(options) {
1964
2385
  const { mapping } = options;
1965
2386
  const excluded = new Set(mapping.exclude_columns ?? []);
1966
- const properties = mapping.properties.filter((property) => !excluded.has(property.source)).map((property) => ({
1967
- name: property.target,
1968
- display_name: property.source,
1969
- desc: property.desc ?? property.source,
1970
- type: toPropType(property.type),
1971
- source: "data"
1972
- }));
2387
+ const properties = mapping.properties.filter((property) => !excluded.has(property.source)).map((property) => {
2388
+ const leaf = property.target.includes(".") ? property.target.split(".").pop() : void 0;
2389
+ return {
2390
+ name: property.target,
2391
+ display_name: leaf ?? property.source,
2392
+ desc: property.desc ?? leaf ?? property.source,
2393
+ type: toPropType(property.type),
2394
+ source: "data"
2395
+ };
2396
+ });
1973
2397
  const propNames = properties.map((property) => property.name);
1974
2398
  const eventNames = resolveEventNames(options);
1975
2399
  const events = eventNames.map((eventName) => {
@@ -2047,7 +2471,7 @@ function buildPlanDraft(ctx) {
2047
2471
  location: { field: "lang" }
2048
2472
  });
2049
2473
  }
2050
- return buildDraftFromMapping({
2474
+ const draft = buildDraftFromMapping({
2051
2475
  mapping,
2052
2476
  planName: ctx.str("plan-name").trim() || mapping.default_event_name || "local-data",
2053
2477
  eventNames: ctx.list("event-name"),
@@ -2055,6 +2479,17 @@ function buildPlanDraft(ctx) {
2055
2479
  lang,
2056
2480
  projectId: ctx.optionalNum("project-id")
2057
2481
  });
2482
+ validateAndFix(draft);
2483
+ try {
2484
+ validateDraft(draft);
2485
+ } catch (error) {
2486
+ throw new CliValidationError("The tracking-plan draft is invalid.", {
2487
+ code: "LOCAL_DATA_PLAN_INVALID_DRAFT",
2488
+ hint: error instanceof Error ? error.message : String(error),
2489
+ location: { field: "mapping" }
2490
+ });
2491
+ }
2492
+ return draft;
2058
2493
  }
2059
2494
  var dataIntegrationPlan = {
2060
2495
  service: "data-integration",
@@ -2062,7 +2497,7 @@ var dataIntegrationPlan = {
2062
2497
  usesAeHost: false,
2063
2498
  description: "Convert a confirmed local-data mapping into a tracking-plan draft.json (source_type=data, sdk_integration_mode=none).",
2064
2499
  flags: [
2065
- { name: "mapping", type: "string", required: true, sensitive: true, desc: "Confirmed ae-local-data-mapping/v1 JSON, file path, or @file." },
2500
+ { name: "mapping", type: "string", required: true, sensitive: true, desc: `Confirmed ${MAPPING_VERSION} JSON, file path, or @file.` },
2066
2501
  { name: "event-name", type: "string", variadic: true, desc: "Concrete event name (track/mixed without default_event_name). Repeat for multiple events." },
2067
2502
  { name: "plan-name", type: "string", desc: 'Plan name. Default: the mapping default_event_name, else "local-data".' },
2068
2503
  { name: "app-type", type: "string", default: "unknown", desc: "Application type recorded in draft meta (informational)." },
@@ -2093,7 +2528,7 @@ var dataIntegrationPlan = {
2093
2528
  }
2094
2529
  };
2095
2530
 
2096
- // src/commands/data-integration/local-data/conversion.ts
2531
+ // src/commands/data-integration/conversion.ts
2097
2532
  import { randomInt as randomInt2, randomUUID } from "crypto";
2098
2533
  import {
2099
2534
  chmodSync,
@@ -2114,14 +2549,6 @@ import { createInterface as createInterface2 } from "readline";
2114
2549
  var SORT_CHUNK_SIZE = 1e4;
2115
2550
  var THREE_YEARS_MS = 3 * 365 * 24 * 60 * 60 * 1e3;
2116
2551
  var THREE_DAYS_MS = 3 * 24 * 60 * 60 * 1e3;
2117
- function stripQuotes(value) {
2118
- if (typeof value !== "string") return value;
2119
- const text = value;
2120
- if (text.length >= 2 && (text.startsWith('"') && text.endsWith('"') || text.startsWith("'") && text.endsWith("'"))) {
2121
- return text.slice(1, -1).trim();
2122
- }
2123
- return text.trim();
2124
- }
2125
2552
  async function convertLocalData(options) {
2126
2553
  const input = await inspectLocalDataInput(options.inputFile);
2127
2554
  if (input.format !== options.mapping.source.format) {
@@ -2141,12 +2568,15 @@ async function convertLocalData(options) {
2141
2568
  const streamOptions = {
2142
2569
  headerNames: options.mapping.headers,
2143
2570
  flattenRules: options.mapping.flatten_rules,
2144
- mergeSheets: options.mergeSheets
2571
+ mergeSheets: options.mergeSheets,
2572
+ // The profile pass inside convert is internal (it writes profile.json); the ragged-row
2573
+ // warning is surfaced by the conversion pass below instead, so suppress it here.
2574
+ warnRagged: false
2145
2575
  };
2146
2576
  const salvageSet = options.salvageFrom ? readSalvageRowNumbers(options.salvageFrom) : void 0;
2147
2577
  let salvageMatched = 0;
2148
2578
  const runId = `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`;
2149
- const outputDir = resolve(options.outputDir || join(".ae-cli", "data-integration", runId));
2579
+ const outputDir = resolve(options.outputDir || join(".ae-cli", "data-integration", "runs", runId));
2150
2580
  prepareOutputDirectory(outputDir);
2151
2581
  const profile = await profileLocalData(input, dataSet, options.mapping.time.source_timezone, streamOptions);
2152
2582
  const profilePath = join(outputDir, "profile.json");
@@ -2163,7 +2593,10 @@ async function convertLocalData(options) {
2163
2593
  let validRecords = 0;
2164
2594
  let invalidRecords = 0;
2165
2595
  const recordTypes = {};
2166
- await streamLocalDataRows(
2596
+ const skippedFields = {};
2597
+ let lanIpRecords = 0;
2598
+ const flattenMisses = {};
2599
+ const rowCount = await streamLocalDataRows(
2167
2600
  input,
2168
2601
  dataSet,
2169
2602
  async (row, rowNumber) => {
@@ -2177,6 +2610,10 @@ async function convertLocalData(options) {
2177
2610
  }
2178
2611
  validRecords += 1;
2179
2612
  recordTypes[result.recordType] = (recordTypes[result.recordType] ?? 0) + 1;
2613
+ if (result.lanIp) lanIpRecords += 1;
2614
+ for (const skip of result.skips) {
2615
+ skippedFields[skip.code] = (skippedFields[skip.code] ?? 0) + 1;
2616
+ }
2180
2617
  const line = JSON.stringify(result.record);
2181
2618
  if (isUserProfileType(result.recordType)) {
2182
2619
  userSetBuffer.push({ key: result.sortKey, line });
@@ -2188,13 +2625,16 @@ async function convertLocalData(options) {
2188
2625
  {
2189
2626
  headerNames: streamOptions.headerNames,
2190
2627
  flattenRules: streamOptions.flattenRules,
2628
+ flattenMisses,
2191
2629
  mergeSheets: streamOptions.mergeSheets
2192
2630
  }
2193
2631
  );
2194
2632
  if (userSetBuffer.length > 0) flushSortChunk(outputDir, userSetBuffer, sortChunks);
2195
- trackStream.end();
2196
- invalidStream.end();
2197
- await Promise.all([once(trackStream, "finish"), once(invalidStream, "finish")]);
2633
+ await Promise.all([finishStream(trackStream), finishStream(invalidStream)]);
2634
+ for (const [outColumn, count] of Object.entries(flattenMisses)) {
2635
+ process.stderr.write(`Warning: flatten rule "${outColumn}" did not materialize for ${count} row(s).
2636
+ `);
2637
+ }
2198
2638
  if (salvageSet && salvageMatched === 0) {
2199
2639
  throw new CliValidationError("The salvage file lists no rows from this source.", {
2200
2640
  code: "LOCAL_DATA_SALVAGE_NO_MATCH",
@@ -2205,12 +2645,11 @@ async function convertLocalData(options) {
2205
2645
  const validStream = secureWriteStream(validPath);
2206
2646
  if (existsSync(trackTempPath)) {
2207
2647
  for await (const chunk of createReadStream3(trackTempPath)) {
2208
- if (!validStream.write(chunk)) await once(validStream, "drain");
2648
+ await writeRaw(validStream, chunk);
2209
2649
  }
2210
2650
  }
2211
2651
  await mergeSortChunks(sortChunks, validStream);
2212
- validStream.end();
2213
- await once(validStream, "finish");
2652
+ await finishStream(validStream);
2214
2653
  if (existsSync(trackTempPath)) unlinkSync(trackTempPath);
2215
2654
  for (const path of sortChunks) if (existsSync(path)) unlinkSync(path);
2216
2655
  writeSecureJson(profilePath, profile);
@@ -2218,8 +2657,9 @@ async function convertLocalData(options) {
2218
2657
  writeSecureText(transformPath, createTransformScript(options.inputFile, mappingPath));
2219
2658
  const validBytes = statSize(validPath);
2220
2659
  const blockedReasons = [
2221
- ...invalidRecords > 0 ? ["Some source rows failed UE validation."] : [],
2222
- ...validRecords === 0 ? ["No valid UE records were generated."] : []
2660
+ ...rowCount === 0 ? ["The source contained no data rows."] : [],
2661
+ ...rowCount > 0 && validRecords === 0 ? ["No valid UE records were generated."] : [],
2662
+ ...invalidRecords > 0 ? ["Some source rows failed UE validation."] : []
2223
2663
  ];
2224
2664
  const manifest = {
2225
2665
  version: "ae-local-data-manifest/v1",
@@ -2240,7 +2680,10 @@ async function convertLocalData(options) {
2240
2680
  valid_records: validRecords,
2241
2681
  invalid_records: invalidRecords,
2242
2682
  valid_bytes: validBytes,
2243
- record_types: recordTypes
2683
+ record_types: recordTypes,
2684
+ ...Object.keys(skippedFields).length > 0 ? { skipped_fields: skippedFields } : {},
2685
+ ...lanIpRecords > 0 ? { lan_ip_records: lanIpRecords } : {},
2686
+ ...Object.keys(flattenMisses).length > 0 ? { flatten_misses: flattenMisses } : {}
2244
2687
  },
2245
2688
  blocked_reasons: blockedReasons
2246
2689
  };
@@ -2258,7 +2701,8 @@ async function convertLocalDataMulti(options) {
2258
2701
  const profile = await profileLocalData(input, dataSet, sourceTimezone, {
2259
2702
  collectSamples: true,
2260
2703
  headerNames: options.mapping.headers,
2261
- flattenRules: options.mapping.flatten_rules
2704
+ flattenRules: options.mapping.flatten_rules,
2705
+ warnRagged: false
2262
2706
  });
2263
2707
  profiled.push({ file: basename4(inputFile), profile });
2264
2708
  }
@@ -2274,7 +2718,7 @@ async function convertLocalDataMulti(options) {
2274
2718
  validateTypeResolutions(resolutions, profiled.map((entry) => entry.file));
2275
2719
  const overrides = applyTypeResolutions(resolutions, profiled);
2276
2720
  const parent = resolve(
2277
- options.outputDir || join(".ae-cli", "data-integration", `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`)
2721
+ options.outputDir || join(".ae-cli", "data-integration", "runs", `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`)
2278
2722
  );
2279
2723
  prepareOutputDirectory(parent);
2280
2724
  const files = [];
@@ -2332,11 +2776,32 @@ function convertRow(row, rowNumber, mapping, now) {
2332
2776
  errors.push({ code: "INVALID_EVENT_NAME", field: mapping.event_name_field });
2333
2777
  }
2334
2778
  }
2335
- const ip = readOptionalField(row, mapping.ip_field);
2336
- const uuid = readOptionalField(row, mapping.uuid_field);
2779
+ const skips = [];
2780
+ let lanIp = false;
2781
+ let ip;
2782
+ if (recordType === "track") {
2783
+ const rawIp = readOptionalField(row, mapping.ip_field);
2784
+ if (rawIp) {
2785
+ if (isValidIp(rawIp)) {
2786
+ ip = rawIp;
2787
+ lanIp = isPrivateIp(rawIp);
2788
+ } else {
2789
+ skips.push({ code: "INVALID_IP", field: mapping.ip_field });
2790
+ }
2791
+ }
2792
+ }
2793
+ let uuid;
2794
+ {
2795
+ const rawUuid = readOptionalField(row, mapping.uuid_field);
2796
+ if (rawUuid) {
2797
+ if (isValidUuid(rawUuid)) uuid = rawUuid;
2798
+ else skips.push({ code: "INVALID_UUID", field: mapping.uuid_field });
2799
+ }
2800
+ }
2337
2801
  const excluded = new Set(mapping.exclude_columns ?? []);
2338
2802
  const properties = {};
2339
2803
  for (const property of mapping.properties) {
2804
+ if (property.target.includes(".")) continue;
2340
2805
  if (excluded.has(property.source)) continue;
2341
2806
  let value = stripQuotes(row[property.source]);
2342
2807
  if (isMissing2(value)) continue;
@@ -2350,8 +2815,8 @@ function convertRow(row, rowNumber, mapping, now) {
2350
2815
  properties[property.target] = converted.value;
2351
2816
  }
2352
2817
  }
2353
- const zoneOffset = mapping.zone_offset_value !== void 0 ? resolveZoneOffsetValue(mapping.zone_offset_value, now) : mapping.zone_offset_field ? readZoneOffset(row, mapping.zone_offset_field) : void 0;
2354
- if (mapping.zone_offset_field && zoneOffset === void 0) {
2818
+ const zoneOffset = recordType === "track" && mapping.zone_offset_value !== void 0 ? resolveZoneOffsetValue(mapping.zone_offset_value, now) : recordType === "track" && mapping.zone_offset_field ? readZoneOffset(row, mapping.zone_offset_field) : void 0;
2819
+ if (recordType === "track" && mapping.zone_offset_field && zoneOffset === void 0) {
2355
2820
  errors.push({ code: "INVALID_ZONE_OFFSET", field: mapping.zone_offset_field });
2356
2821
  }
2357
2822
  if (errors.length > 0 || !recordType || !normalizedTime) return { ok: false, errors };
@@ -2369,7 +2834,9 @@ function convertRow(row, rowNumber, mapping, now) {
2369
2834
  ok: true,
2370
2835
  recordType,
2371
2836
  record,
2372
- sortKey: `${accountId ?? distinctId ?? ""}\0${normalizedTime.instant.toISOString()}\0${String(rowNumber).padStart(12, "0")}`
2837
+ sortKey: `${accountId ?? distinctId ?? ""}\0${normalizedTime.instant.toISOString()}\0${String(rowNumber).padStart(12, "0")}`,
2838
+ skips,
2839
+ lanIp
2373
2840
  };
2374
2841
  }
2375
2842
  function readOptionalField(row, field) {
@@ -2477,7 +2944,7 @@ function convertProperty(value, type, transform, timeZone = "UTC", timeFormat) {
2477
2944
  const normalized = normalizeTime(value, timeZone, timeFormat);
2478
2945
  return normalized ? { ok: true, value: normalized.formatted } : { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
2479
2946
  }
2480
- if (type === "list") {
2947
+ if (type === "list" || type === "array_row") {
2481
2948
  if (!Array.isArray(value)) return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
2482
2949
  return isPropertyWithinLimits(value, type) ? { ok: true, value } : { ok: false, code: "PROPERTY_LIMIT_EXCEEDED" };
2483
2950
  }
@@ -2679,11 +3146,40 @@ function prepareOutputDirectory(path) {
2679
3146
  chmodSync(path, 448);
2680
3147
  }
2681
3148
  function secureWriteStream(path) {
2682
- return createWriteStream(path, { encoding: "utf8", mode: 384, flags: "wx" });
3149
+ const stream = createWriteStream(path, { encoding: "utf8", mode: 384, flags: "wx" });
3150
+ let fail;
3151
+ const errorPromise = new Promise((_, reject) => {
3152
+ fail = reject;
3153
+ });
3154
+ errorPromise.catch(() => {
3155
+ });
3156
+ stream.on("error", (error) => fail?.(error));
3157
+ return { stream, path, errorPromise };
2683
3158
  }
2684
- async function writeLine(stream, line) {
2685
- if (!stream.write(`${line}
2686
- `)) await once(stream, "drain");
3159
+ function writeFailure(path, error) {
3160
+ const detail = error instanceof Error ? error.message : String(error);
3161
+ return new Error(`Failed to write "${path}": ${detail}. Check available disk space and directory permissions.`, { cause: error });
3162
+ }
3163
+ async function writeRaw(target, data) {
3164
+ try {
3165
+ if (!target.stream.write(data)) {
3166
+ await Promise.race([once(target.stream, "drain"), target.errorPromise]);
3167
+ }
3168
+ } catch (error) {
3169
+ throw writeFailure(target.path, error);
3170
+ }
3171
+ }
3172
+ async function writeLine(target, line) {
3173
+ await writeRaw(target, `${line}
3174
+ `);
3175
+ }
3176
+ async function finishStream(target) {
3177
+ try {
3178
+ target.stream.end();
3179
+ await Promise.race([once(target.stream, "finish"), target.errorPromise]);
3180
+ } catch (error) {
3181
+ throw writeFailure(target.path, error);
3182
+ }
2687
3183
  }
2688
3184
  function writeSecureJson(path, value) {
2689
3185
  writeSecureText(path, `${JSON.stringify(value, null, 2)}
@@ -2710,7 +3206,7 @@ function formatRunTimestamp(value) {
2710
3206
  return value.toISOString().replace(/[-:]/g, "").replace(/\.\d{3}Z$/, "Z");
2711
3207
  }
2712
3208
 
2713
- // src/commands/data-integration/local-data/convert.ts
3209
+ // src/commands/data-integration/convert.ts
2714
3210
  var dataIntegrationConvert = {
2715
3211
  service: "data-integration",
2716
3212
  command: "convert",
@@ -2718,8 +3214,8 @@ var dataIntegrationConvert = {
2718
3214
  description: "Convert one or more local data sets into validated UE JSONL and quarantine invalid rows.",
2719
3215
  flags: [
2720
3216
  { name: "input-file", type: "string", required: true, sensitive: true, variadic: true, desc: "Source local data file. Repeat for multiple files (requires a wildcard mapping). The source is never modified." },
2721
- { name: "mapping", type: "string", required: true, sensitive: true, desc: "ae-local-data-mapping/v1 JSON, file path, or @file." },
2722
- { name: "output-dir", type: "string", sensitive: true, desc: "New or empty output directory. Default: .ae-cli/data-integration/<run-id>." },
3217
+ { name: "mapping", type: "string", required: true, sensitive: true, desc: `${MAPPING_VERSION} JSON, file path, or @file.` },
3218
+ { name: "output-dir", type: "string", sensitive: true, desc: "New or empty output directory. Default: .ae-cli/data-integration/runs/<run-id>." },
2723
3219
  { name: "type-resolutions", type: "json", sensitive: true, desc: "JSON object resolving cross-file column type conflicts (unify, split, or skip)." },
2724
3220
  { name: "merge-sheets", type: "boolean", default: false, desc: "Stream every worksheet in file order instead of a single selected sheet." },
2725
3221
  { name: "salvage-from", type: "string", sensitive: true, desc: "Re-process only the rows listed in a previous run's invalid.rows.jsonl, against the current (fixed) mapping. Single-file only." }
@@ -2771,7 +3267,7 @@ var dataIntegrationConvert = {
2771
3267
  }
2772
3268
  };
2773
3269
 
2774
- // src/commands/data-integration/local-data/upload.ts
3270
+ // src/commands/data-integration/upload.ts
2775
3271
  import { createReadStream as createReadStream4, readFileSync as readFileSync3, statSync as statSync3 } from "fs";
2776
3272
  import { basename as basename5, dirname, resolve as resolve2 } from "path";
2777
3273
  import { createInterface as createInterface3 } from "readline";
@@ -3116,22 +3612,900 @@ function isRecord3(value) {
3116
3612
  return value !== null && typeof value === "object" && !Array.isArray(value);
3117
3613
  }
3118
3614
 
3119
- // src/commands/data-integration/local-data/handoff.ts
3615
+ // src/commands/data-integration/handoff.ts
3120
3616
  import { createHash as createHash3 } from "crypto";
3121
- import { chmodSync as chmodSync2, mkdirSync as mkdirSync2, readFileSync as readFileSync4, renameSync as renameSync2, writeFileSync as writeFileSync2 } from "fs";
3122
- import { join as join2, resolve as resolve3 } from "path";
3617
+ import { chmodSync as chmodSync2, mkdirSync as mkdirSync2, mkdtempSync, readFileSync as readFileSync4, renameSync as renameSync2, rmSync, writeFileSync as writeFileSync2 } from "fs";
3618
+ import { dirname as dirname2, join as join3, resolve as resolve3 } from "path";
3619
+ import { tmpdir } from "os";
3620
+
3621
+ // src/commands/data-integration/archive.ts
3622
+ import archiver from "archiver";
3623
+ import { createWriteStream as createWriteStream2, readdirSync as readdirSync2, statSync as statSync4 } from "fs";
3624
+ import { join as join2, relative, sep } from "path";
3625
+ async function zipPackage(dir, zipPath) {
3626
+ await new Promise((resolvePromise, rejectPromise) => {
3627
+ const output = createWriteStream2(zipPath, { mode: 384 });
3628
+ const zip = archiver("zip", { zlib: { level: 9 } });
3629
+ output.on("close", resolvePromise);
3630
+ output.on("error", rejectPromise);
3631
+ zip.on("error", rejectPromise);
3632
+ zip.on("warning", (error) => {
3633
+ if (error.code !== "ENOENT") rejectPromise(error);
3634
+ });
3635
+ zip.pipe(output);
3636
+ const walk = (current) => {
3637
+ for (const name of readdirSync2(current)) {
3638
+ if (name === ".DS_Store") continue;
3639
+ const full = join2(current, name);
3640
+ const stats = statSync4(full);
3641
+ if (stats.isDirectory()) {
3642
+ walk(full);
3643
+ } else {
3644
+ const rel = relative(dir, full).split(sep).join("/");
3645
+ zip.file(full, { name: rel, mode: stats.mode & 511 });
3646
+ }
3647
+ }
3648
+ };
3649
+ walk(dir);
3650
+ void zip.finalize();
3651
+ });
3652
+ }
3653
+
3654
+ // src/commands/data-integration/relay.ts
3655
+ var PIPELINE_VERSION = "ae-data-integration-pipeline/v1";
3656
+ var SHAPE_VERSION = "ae-data-integration-shape/v1";
3657
+ var DEFAULT_BATCH_SIZE2 = 500;
3658
+ var ENV_FILE = ".local/target.env";
3659
+ function buildPipelineDescriptor(entries, target = {}) {
3660
+ const first = entries[0];
3661
+ return {
3662
+ version: PIPELINE_VERSION,
3663
+ created_at: first?.created_at ?? (/* @__PURE__ */ new Date()).toISOString(),
3664
+ source: { type: "local_file", params: { format: first?.format ?? "csv" } },
3665
+ transform: { type: MAPPING_VERSION, refs: entries.map((entry) => entry.mapping_file) },
3666
+ sink: {
3667
+ type: "restful_sync_json",
3668
+ params: {
3669
+ batch_size: DEFAULT_BATCH_SIZE2,
3670
+ env_file: ENV_FILE,
3671
+ ...target.pushurl ? { pushurl: target.pushurl } : {},
3672
+ ...target.project_id ? { project_id: target.project_id } : {}
3673
+ }
3674
+ }
3675
+ };
3676
+ }
3677
+ function buildShapeBaseline(items) {
3678
+ return {
3679
+ version: SHAPE_VERSION,
3680
+ entries: items.map(({ mapping, fingerprint }) => ({
3681
+ fingerprint,
3682
+ mode: mapping.mode,
3683
+ data_set: mapping.source.data_set,
3684
+ format: mapping.source.format,
3685
+ columns: sourceColumns(mapping)
3686
+ }))
3687
+ };
3688
+ }
3689
+ function sh(...lines) {
3690
+ return `${lines.join("\n")}
3691
+ `;
3692
+ }
3693
+ function generateBinScripts() {
3694
+ return [
3695
+ { relPath: "bin/run.sh", content: runSh(), mode: 448 },
3696
+ { relPath: "bin/upload.sh", content: uploadSh(), mode: 448 },
3697
+ { relPath: "bin/bind_mapping.py", content: bindMappingPy(), mode: 448 },
3698
+ { relPath: "bin/summarize.py", content: summarizePy(), mode: 448 },
3699
+ { relPath: "bin/plan_check.py", content: planCheckPy(), mode: 448 },
3700
+ { relPath: "bin/verify.py", content: verifyPy(), mode: 448 },
3701
+ { relPath: "bin/resolve_appid.py", content: resolveAppidPy(), mode: 448 }
3702
+ ];
3703
+ }
3704
+ function runSh() {
3705
+ return sh(
3706
+ "#!/usr/bin/env bash",
3707
+ "# Generic pipeline executor: source -> transform -> plan. Never uploads (see upload.sh).",
3708
+ "# Reads pipeline.json and dispatches each stage by its `type` to ae-cli subcommands.",
3709
+ "set -euo pipefail",
3710
+ 'SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"',
3711
+ 'PKG_ROOT="$(dirname "$SCRIPT_DIR")"',
3712
+ 'cd "$PKG_ROOT"',
3713
+ "",
3714
+ `SRC_TYPE="$(python3 -c 'import json; print(json.load(open("pipeline.json"))["source"]["type"])')"`,
3715
+ 'case "$SRC_TYPE" in',
3716
+ " local_file) ;;",
3717
+ " *)",
3718
+ ' echo "unsupported source type: $SRC_TYPE" >&2',
3719
+ ' echo "re-run the full ae-data-integration pipeline (or upgrade ae-cli) for this source." >&2',
3720
+ " exit 64",
3721
+ " ;;",
3722
+ "esac",
3723
+ "",
3724
+ 'INPUT="${1:-}"',
3725
+ 'if [ -z "$INPUT" ]; then',
3726
+ " shopt -s nullglob; FILES=(inbox/*); shopt -u nullglob",
3727
+ ' if [ "${#FILES[@]}" -ne 1 ]; then',
3728
+ ' echo "usage: bin/run.sh <input-file>" >&2',
3729
+ ' echo " (or put exactly one file in inbox/)" >&2',
3730
+ " exit 2",
3731
+ " fi",
3732
+ ' INPUT="${FILES[0]}"',
3733
+ "fi",
3734
+ "",
3735
+ 'RUN_DIR="runs/$(date +%Y%m%d-%H%M%S)"',
3736
+ 'mkdir -p "$RUN_DIR"',
3737
+ 'echo "run: $RUN_DIR"',
3738
+ "",
3739
+ 'python3 bin/bind_mapping.py "$INPUT" "$RUN_DIR"',
3740
+ "",
3741
+ "while IFS= read -r ref; do",
3742
+ ' ref_dir="$(dirname "$ref")"',
3743
+ ' echo "convert: $ref_dir"',
3744
+ " ae-cli data-integration convert \\",
3745
+ ' --input-file "$INPUT" \\',
3746
+ ' --mapping "$RUN_DIR/bound/$ref_dir/mapping.json" \\',
3747
+ ' --output-dir "$RUN_DIR/$ref_dir" || exit $?',
3748
+ `done < <(python3 -c 'import json,sys; print("\\n".join(json.load(open("pipeline.json"))["transform"]["refs"]))')`,
3749
+ "",
3750
+ 'python3 bin/summarize.py "$RUN_DIR"',
3751
+ 'python3 bin/plan_check.py "$RUN_DIR"',
3752
+ "",
3753
+ "# Salvage hint: quarantined rows are never silently dropped. Re-process only",
3754
+ "# them against the fixed mapping instead of re-uploading the whole file.",
3755
+ "while IFS= read -r ref; do",
3756
+ ' ref_dir="$(dirname "$ref")"',
3757
+ ' inv="$RUN_DIR/$ref_dir/invalid.rows.jsonl"',
3758
+ ' if [ -s "$inv" ]; then',
3759
+ ' echo "note: $inv has quarantined rows \u2014 salvage them with:"',
3760
+ ' echo " ae-cli data-integration convert --input-file "$INPUT" --mapping "$RUN_DIR/bound/$ref_dir/mapping.json" --salvage-from "$inv" --output-dir "$RUN_DIR-salvage""',
3761
+ " fi",
3762
+ `done < <(python3 -c 'import json,sys; print("\\n".join(json.load(open("pipeline.json"))["transform"]["refs"]))')`,
3763
+ "",
3764
+ 'echo "done. review the summary and plan check, then run: bin/upload.sh $RUN_DIR"'
3765
+ );
3766
+ }
3767
+ function uploadSh() {
3768
+ return sh(
3769
+ "#!/usr/bin/env bash",
3770
+ "# Sink executor. Dry-run by default; --confirm actually uploads.",
3771
+ "# Resolves the recorded target from pipeline.json (pushurl + project_id), derives",
3772
+ "# the APPID via `ae-cli project info get` (bin/resolve_appid.py), and falls back to",
3773
+ "# .local/target.env for explicit APPID / endpoint / project-id overrides.",
3774
+ "set -euo pipefail",
3775
+ 'SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"',
3776
+ 'PKG_ROOT="$(dirname "$SCRIPT_DIR")"',
3777
+ 'cd "$PKG_ROOT"',
3778
+ "",
3779
+ `SINK_TYPE="$(python3 -c 'import json; print(json.load(open("pipeline.json"))["sink"]["type"])')"`,
3780
+ 'case "$SINK_TYPE" in',
3781
+ " restful_sync_json) ;;",
3782
+ " *)",
3783
+ ' echo "unsupported sink type: $SINK_TYPE" >&2',
3784
+ ' echo "re-run the full ae-data-integration pipeline (or upgrade ae-cli) for this sink." >&2',
3785
+ " exit 64",
3786
+ " ;;",
3787
+ "esac",
3788
+ "",
3789
+ `SINK_PARAMS="$(python3 -c 'import json; print(json.dumps(json.load(open("pipeline.json"))["sink"]["params"]))')"`,
3790
+ `BATCH_SIZE="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("batch_size", 500))' "$SINK_PARAMS")"`,
3791
+ `ENV_FILE="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("env_file", ".local/target.env"))' "$SINK_PARAMS")"`,
3792
+ `PUSHURL="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("pushurl") or "")' "$SINK_PARAMS")"`,
3793
+ `PROJECT_ID="$(python3 -c 'import json,sys; print(json.loads(sys.argv[1]).get("project_id") or "")' "$SINK_PARAMS")"`,
3794
+ "",
3795
+ "CONFIRM=0",
3796
+ 'RUN_DIR=""',
3797
+ 'for arg in "$@"; do',
3798
+ ' case "$arg" in',
3799
+ " --confirm) CONFIRM=1 ;;",
3800
+ ' -*) echo "unknown flag: $arg" >&2; exit 2 ;;',
3801
+ ' *) RUN_DIR="$arg" ;;',
3802
+ " esac",
3803
+ "done",
3804
+ "",
3805
+ 'if [ -z "$RUN_DIR" ]; then',
3806
+ ' echo "usage: bin/upload.sh [--confirm] <runs/<run-id>>" >&2',
3807
+ " exit 2",
3808
+ "fi",
3809
+ "",
3810
+ "# .local/target.env is optional: the package may record the target itself.",
3811
+ "# Env values still win as explicit overrides (the documented fallback).",
3812
+ 'if [ -f "$ENV_FILE" ]; then set -a; . "$ENV_FILE"; set +a; fi',
3813
+ '[ -z "$PROJECT_ID" ] && PROJECT_ID="${AE_PROJECT_ID:-}"',
3814
+ "",
3815
+ "# Endpoint: the recorded pushurl (a receiver base URL; append /sync_json), else AE_ENDPOINT.",
3816
+ 'if [ -n "$PUSHURL" ]; then',
3817
+ ' case "$PUSHURL" in',
3818
+ ' */sync_json) ENDPOINT="$PUSHURL" ;;',
3819
+ ' *) ENDPOINT="${PUSHURL%/}/sync_json" ;;',
3820
+ " esac",
3821
+ "else",
3822
+ ' : "${AE_ENDPOINT:?set AE_ENDPOINT in .local/target.env (or record pushurl at handoff)}"',
3823
+ ' ENDPOINT="$AE_ENDPOINT"',
3824
+ "fi",
3825
+ "",
3826
+ "# APPID: explicit env wins, else derive from the recorded project via project info get.",
3827
+ 'APPID="${AE_APPID:-}"',
3828
+ 'if [ -z "$APPID" ] && [ -n "$PROJECT_ID" ]; then',
3829
+ ' APPID="$(python3 bin/resolve_appid.py "$PROJECT_ID")"',
3830
+ "fi",
3831
+ ': "${APPID:?set AE_APPID in .local/target.env (or record project_id at handoff)}"',
3832
+ "",
3833
+ "# Mask the APPID in anything this script prints \u2014 it must never land in logs.",
3834
+ "mask() {",
3835
+ ' local a="$1"',
3836
+ ' if [ "${#a}" -le 4 ]; then printf "%s" "****"; return; fi',
3837
+ ' printf "%s%s" "$(printf "%*s" "$(( ${#a} - 4 ))" "" | tr " " "*")" "${a: -4}"',
3838
+ "}",
3839
+ "display_args() {",
3840
+ ' local args=("$@") out=() i',
3841
+ " for ((i=0; i<${#args[@]}; i++)); do",
3842
+ ' if [ "${args[$i]}" = "--appid" ] && [ -n "${args[$((i+1))]:-}" ]; then',
3843
+ ' out+=("--appid" "$(mask "${args[$((i+1))]}")")',
3844
+ " i=$((i+1))",
3845
+ " else",
3846
+ ' out+=("${args[$i]}")',
3847
+ " fi",
3848
+ " done",
3849
+ ' printf "%s\\n" "${out[*]}"',
3850
+ "}",
3851
+ "",
3852
+ 'echo "target: project_id=${PROJECT_ID:-<unset>}"',
3853
+ 'echo " endpoint=$ENDPOINT"',
3854
+ 'echo " appid=$(mask "$APPID")"',
3855
+ 'if [ "$CONFIRM" -eq 0 ]; then',
3856
+ ' echo "dry-run \u2014 re-run with --confirm to upload to this address and project."',
3857
+ "else",
3858
+ ' echo "confirmed: uploading to the address and project shown above."',
3859
+ "fi",
3860
+ "",
3861
+ 'FLAGS=(--endpoint "$ENDPOINT" --appid "$APPID" --batch-size "$BATCH_SIZE")',
3862
+ 'if [ "$CONFIRM" -eq 0 ]; then FLAGS+=(--dry-run); fi',
3863
+ "",
3864
+ "while IFS= read -r ref; do",
3865
+ ' ref_dir="$(dirname "$ref")"',
3866
+ ' ue="$RUN_DIR/$ref_dir/valid.ue.jsonl"',
3867
+ ' manifest="$RUN_DIR/$ref_dir/manifest.json"',
3868
+ ' if [ ! -f "$ue" ]; then',
3869
+ ' echo "missing $ue (run bin/run.sh first)" >&2',
3870
+ " exit 2",
3871
+ " fi",
3872
+ ` status="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["status"])' "$manifest")"`,
3873
+ ' args=("${FLAGS[@]}" --ue-file "$ue" --manifest "$manifest")',
3874
+ ' if [ "$status" = "blocked" ]; then',
3875
+ ' echo "manifest $manifest is blocked (rows were quarantined)."',
3876
+ ' echo "Uploading passes only the valid subset (--allow-clean-subset); quarantined rows stay in invalid.rows.jsonl."',
3877
+ ' if [ "$CONFIRM" -eq 1 ]; then args+=(--allow-clean-subset); else echo " (dry-run) re-run with --confirm to accept the clean subset."; fi',
3878
+ " fi",
3879
+ ' echo "> ae-cli data-integration upload $(display_args "${args[@]}")"',
3880
+ ' ae-cli data-integration upload "${args[@]}" || exit $?',
3881
+ `done < <(python3 -c 'import json,sys; print("\\n".join(json.load(open("pipeline.json"))["transform"]["refs"]))')`
3882
+ );
3883
+ }
3884
+ function bindMappingPy() {
3885
+ return `#!/usr/bin/env python3
3886
+ """Source stage: rebind the frozen mappings to a new same-shape file.
3887
+
3888
+ Runs \`ae-cli data-integration inspect\` once, then for every mapping the
3889
+ pipeline references (pipeline.json's transform.refs) re-binds the frozen
3890
+ mapping's \`source.sha256\` and \`source.data_set\` to the new file (identity
3891
+ fields only \u2014 business logic is untouched), after checking the column set
3892
+ against shape.json. Historical index entries the pipeline does not run are
3893
+ left alone \u2014 the index accumulates across handoffs in the same directory.
3894
+
3895
+ Usage: bin/bind_mapping.py <input-file> <run-dir>
3896
+ """
3897
+ import json
3898
+ import os
3899
+ import subprocess
3900
+ import sys
3901
+
3902
+
3903
+ def fail(message):
3904
+ print(f"bind_mapping: {message}", file=sys.stderr)
3905
+ sys.exit(1)
3906
+
3907
+
3908
+ def pkg_root():
3909
+ return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
3910
+
3911
+
3912
+ def load_json(path):
3913
+ with open(path, "r", encoding="utf-8") as f:
3914
+ return json.load(f)
3915
+
3916
+
3917
+ def write_json(path, value):
3918
+ os.makedirs(os.path.dirname(path), exist_ok=True)
3919
+ with open(path, "w", encoding="utf-8") as f:
3920
+ json.dump(value, f, ensure_ascii=False, indent=2)
3921
+ f.write("\\n")
3922
+
3923
+
3924
+ def dataset_key(dataset):
3925
+ return dataset.get("id") or dataset.get("label") or ""
3926
+
3927
+
3928
+ def main():
3929
+ if len(sys.argv) != 3:
3930
+ fail("usage: bind_mapping.py <input-file> <run-dir>")
3931
+ input_file, run_dir = sys.argv[1], sys.argv[2]
3932
+ root = pkg_root()
3933
+
3934
+ pipeline = load_json(os.path.join(root, "pipeline.json"))
3935
+ if pipeline["source"]["type"] != "local_file":
3936
+ fail(f"unsupported source type: {pipeline['source']['type']}")
3937
+
3938
+ index = load_json(os.path.join(root, "index.json"))
3939
+ shape = load_json(os.path.join(root, "shape.json"))
3940
+ shape_by_fp = {entry["fingerprint"]: entry for entry in shape["entries"]}
3941
+ index_by_ref = {entry["mapping_file"]: entry for entry in index["entries"]}
3942
+
3943
+ inspect = run_inspect(input_file)
3944
+ datasets = extract_datasets(inspect)
3945
+ headers_by_dataset = extract_headers(inspect, datasets)
3946
+ sha = (inspect.get("source") or {}).get("sha256")
3947
+
3948
+ # Rebind only the mappings this pipeline runs (transform.refs). The index
3949
+ # accumulates entries across handoffs in the same directory; earlier entries
3950
+ # may have no shape baseline here and are never converted by run.sh, so
3951
+ # walking the whole index would fail on them.
3952
+ for ref in pipeline["transform"]["refs"]:
3953
+ entry = index_by_ref.get(ref)
3954
+ if entry is None:
3955
+ fail(f"index entry missing for {ref}; re-run the full pipeline")
3956
+ fingerprint = entry["fingerprint"]
3957
+ baseline = shape_by_fp.get(fingerprint)
3958
+ if baseline is None:
3959
+ fail(f"shape baseline missing for {fingerprint}; re-run the full pipeline")
3960
+ frozen = load_json(os.path.join(root, ref))
3961
+ data_set_id = match_dataset(frozen, baseline, datasets, headers_by_dataset)
3962
+ validate_headers(data_set_id, baseline, headers_by_dataset)
3963
+ frozen["source"]["sha256"] = sha
3964
+ frozen["source"]["data_set"] = data_set_id
3965
+ out = os.path.join(root, run_dir, "bound", os.path.dirname(ref), "mapping.json")
3966
+ write_json(out, frozen)
3967
+ print(f"rebound {os.path.dirname(ref)} -> data_set {data_set_id!r}")
3968
+ print("shape check passed")
3969
+
3970
+
3971
+ def run_inspect(input_file):
3972
+ proc = subprocess.run(
3973
+ ["ae-cli", "data-integration", "inspect", "--input-file", input_file],
3974
+ capture_output=True, text=True,
3975
+ )
3976
+ if proc.returncode != 0:
3977
+ fail(f"inspect failed: {proc.stderr.strip()}")
3978
+ try:
3979
+ parsed = json.loads(proc.stdout)
3980
+ except json.JSONDecodeError:
3981
+ fail("inspect returned non-JSON output")
3982
+ # ae-cli wraps every command result in { ok, data, error }; unwrap it.
3983
+ data = parsed.get("data") if isinstance(parsed, dict) else None
3984
+ if not isinstance(data, dict):
3985
+ fail("inspect returned no data payload")
3986
+ return data
3987
+
3988
+
3989
+ def extract_datasets(inspect):
3990
+ if inspect.get("selection_required"):
3991
+ return inspect.get("data_sets") or []
3992
+ data_set = inspect.get("data_set")
3993
+ return [data_set] if data_set else []
3994
+
3995
+
3996
+ def extract_headers(inspect, datasets):
3997
+ result = {}
3998
+ details = inspect.get("header_details")
3999
+ if details:
4000
+ for dataset in datasets:
4001
+ names = (dataset.get("label"), dataset.get("id"), dataset.get("selector"))
4002
+ for sheet in details:
4003
+ if sheet.get("name") in names:
4004
+ result[dataset_key(dataset)] = sheet.get("headers") or []
4005
+ break
4006
+ return result
4007
+ columns = inspect.get("columns")
4008
+ if columns and datasets:
4009
+ result[dataset_key(datasets[0])] = [column["name"] for column in columns]
4010
+ return result
4011
+
4012
+
4013
+ def match_dataset(frozen, baseline, datasets, headers_by_dataset):
4014
+ if not datasets:
4015
+ fail("inspect reported no data sets")
4016
+ wanted = frozen["source"]["data_set"]
4017
+ for dataset in datasets:
4018
+ if dataset.get("id") == wanted:
4019
+ return dataset.get("id")
4020
+ for dataset in datasets:
4021
+ if dataset.get("label") == wanted:
4022
+ return dataset.get("id")
4023
+ baseline_cols = set(baseline.get("columns") or [])
4024
+ if baseline_cols:
4025
+ for dataset in datasets:
4026
+ headers = headers_by_dataset.get(dataset_key(dataset))
4027
+ if headers and set(headers) == baseline_cols:
4028
+ return dataset.get("id")
4029
+ if len(datasets) == 1:
4030
+ return datasets[0].get("id")
4031
+ fail(f"cannot rebind data_set {wanted!r}: no exact or header match; re-run the full pipeline")
4032
+
4033
+
4034
+ def validate_headers(data_set_id, baseline, headers_by_dataset):
4035
+ baseline_cols = set(baseline.get("columns") or [])
4036
+ if not baseline_cols:
4037
+ return
4038
+ headers = headers_by_dataset.get(data_set_id)
4039
+ if headers is None:
4040
+ return
4041
+ if set(headers) != baseline_cols:
4042
+ missing = sorted(baseline_cols - set(headers))
4043
+ extra = sorted(set(headers) - baseline_cols)
4044
+ fail(
4045
+ f"shape mismatch for {data_set_id!r}: missing={missing} extra={extra} \u2014 "
4046
+ "re-run the full pipeline; do not edit the frozen mapping"
4047
+ )
4048
+
4049
+
4050
+ if __name__ == "__main__":
4051
+ main()
4052
+ `;
4053
+ }
4054
+ function summarizePy() {
4055
+ return `#!/usr/bin/env python3
4056
+ """Transform stage summary: print valid/quarantined counts per data set."""
4057
+ import json
4058
+ import os
4059
+ import sys
4060
+
4061
+
4062
+ def pkg_root():
4063
+ return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
4064
+
4065
+
4066
+ def main():
4067
+ if len(sys.argv) != 2:
4068
+ print("usage: summarize.py <run-dir>", file=sys.stderr)
4069
+ sys.exit(2)
4070
+ run_dir = sys.argv[1]
4071
+ root = pkg_root()
4072
+ with open(os.path.join(root, "pipeline.json"), "r", encoding="utf-8") as f:
4073
+ pipeline = json.load(f)
4074
+ total_valid = 0
4075
+ total_invalid = 0
4076
+ for ref in pipeline["transform"]["refs"]:
4077
+ ref_dir = os.path.dirname(ref)
4078
+ manifest_path = os.path.join(root, run_dir, ref_dir, "manifest.json")
4079
+ if not os.path.exists(manifest_path):
4080
+ continue
4081
+ with open(manifest_path, "r", encoding="utf-8") as f:
4082
+ manifest = json.load(f)
4083
+ output = manifest["output"]
4084
+ total_valid += output["valid_records"]
4085
+ total_invalid += output["invalid_records"]
4086
+ print(f"{ref_dir}: {output['valid_records']} valid / {output['invalid_records']} quarantined")
4087
+ for reason in manifest.get("blocked_reasons") or []:
4088
+ print(f" - {reason}")
4089
+ print(f"total: {total_valid} valid / {total_invalid} quarantined")
4090
+
4091
+
4092
+ if __name__ == "__main__":
4093
+ main()
4094
+ `;
4095
+ }
4096
+ function planCheckPy() {
4097
+ return `#!/usr/bin/env python3
4098
+ """Plan gate: every event and property produced must already exist in the frozen tracking plan."""
4099
+ import json
4100
+ import os
4101
+ import sys
4102
+
4103
+
4104
+ def pkg_root():
4105
+ return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
4106
+
4107
+
4108
+ def load_json(path):
4109
+ with open(path, "r", encoding="utf-8") as f:
4110
+ return json.load(f)
4111
+
4112
+
4113
+ def main():
4114
+ if len(sys.argv) != 2:
4115
+ print("usage: plan_check.py <run-dir>", file=sys.stderr)
4116
+ sys.exit(2)
4117
+ run_dir = sys.argv[1]
4118
+ root = pkg_root()
4119
+ pipeline = load_json(os.path.join(root, "pipeline.json"))
4120
+ new_events = []
4121
+ new_properties = []
4122
+ missing_plans = []
4123
+ for ref in pipeline["transform"]["refs"]:
4124
+ ref_dir = os.path.dirname(ref)
4125
+ plan_path = os.path.join(root, ref_dir, "plan.json")
4126
+ ue_path = os.path.join(root, run_dir, ref_dir, "valid.ue.jsonl")
4127
+ if not os.path.exists(plan_path):
4128
+ missing_plans.append(ref_dir)
4129
+ continue
4130
+ plan = load_json(plan_path)
4131
+ plan_events = {event["event_name"] for event in plan.get("events", [])}
4132
+ plan_properties = {prop["name"] for prop in plan.get("event_properties", [])}
4133
+ plan_properties |= {prop["name"] for prop in plan.get("common_event_properties", [])}
4134
+ plan_properties |= {prop["name"] for prop in plan.get("user_properties", [])}
4135
+ produced_events = set()
4136
+ produced_properties = set()
4137
+ if os.path.exists(ue_path):
4138
+ with open(ue_path, "r", encoding="utf-8") as f:
4139
+ for line in f:
4140
+ line = line.strip()
4141
+ if not line:
4142
+ continue
4143
+ record = json.loads(line)
4144
+ if record.get("#event_name"):
4145
+ produced_events.add(record["#event_name"])
4146
+ props = record.get("properties")
4147
+ if isinstance(props, dict):
4148
+ for key in props:
4149
+ if key.startswith("#"):
4150
+ continue
4151
+ produced_properties.add(key)
4152
+ new_events.extend(sorted(produced_events - plan_events))
4153
+ new_properties.extend(sorted(produced_properties - plan_properties))
4154
+ if missing_plans:
4155
+ print("no plan.json in package for: " + ", ".join(missing_plans), file=sys.stderr)
4156
+ print("run the Tracking plan step first \u2014 the plan gate cannot be skipped", file=sys.stderr)
4157
+ sys.exit(3)
4158
+ if new_events:
4159
+ print("new events not in the plan: " + ", ".join(new_events), file=sys.stderr)
4160
+ sys.exit(3)
4161
+ if new_properties:
4162
+ print("new properties not in the plan: " + ", ".join(new_properties), file=sys.stderr)
4163
+ sys.exit(3)
4164
+ print("plan coverage ok")
4165
+
4166
+
4167
+ if __name__ == "__main__":
4168
+ main()
4169
+ `;
4170
+ }
4171
+ function verifyPy() {
4172
+ return `#!/usr/bin/env python3
4173
+ """Persistence consistency check: submit-window counts vs the platform summary.
4174
+
4175
+ A soft check, not a hard gate. It computes what this run submitted from the local
4176
+ UE output (knowable), snapshots \`ae-cli tracking ingest summary\` over the submit
4177
+ window before and after upload, and prints both next to the expected counts. It
4178
+ does NOT parse the summary payload into per-event numbers: the capability's data
4179
+ shape is server-defined and not a stable CLI contract, and a shared project cannot
4180
+ attribute the window delta to this import alone. For a hard per-event SQL judge,
4181
+ overlay a project custom layer (see custom-layer.md in the ae-data-integration skill).
4182
+
4183
+ verify.py <run-dir> --baseline snapshot the summary before upload
4184
+ verify.py <run-dir> --check snapshot again and diff against the baseline
4185
+ """
4186
+ import json
4187
+ import os
4188
+ import subprocess
4189
+ import sys
4190
+
4191
+
4192
+ def pkg_root():
4193
+ return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
4194
+
4195
+
4196
+ def load_json(path):
4197
+ with open(path, "r", encoding="utf-8") as f:
4198
+ return json.load(f)
4199
+
4200
+
4201
+ def visit_window(record, lo, hi):
4202
+ raw = record.get("#time")
4203
+ if not isinstance(raw, str) or len(raw) < 19:
4204
+ return lo, hi
4205
+ stamp = raw[:19] # YYYY-MM-DD HH:mm:ss
4206
+ if lo is None or stamp < lo:
4207
+ lo = stamp
4208
+ if hi is None or stamp > hi:
4209
+ hi = stamp
4210
+ return lo, hi
4211
+
4212
+
4213
+ def main():
4214
+ if len(sys.argv) < 2:
4215
+ print("usage: verify.py <run-dir> [--baseline|--check]", file=sys.stderr)
4216
+ sys.exit(2)
4217
+ run_dir = sys.argv[1]
4218
+ mode = sys.argv[2] if len(sys.argv) > 2 else "--check"
4219
+ if mode not in ("--baseline", "--check"):
4220
+ print("usage: verify.py <run-dir> [--baseline|--check]", file=sys.stderr)
4221
+ sys.exit(2)
4222
+ root = pkg_root()
4223
+ pipeline = load_json(os.path.join(root, "pipeline.json"))
4224
+ project_id = pipeline["sink"]["params"].get("project_id") or os.environ.get("AE_PROJECT_ID") or "1"
4225
+
4226
+ expected_events = {}
4227
+ expected_users = 0
4228
+ lo = hi = None
4229
+ for ref in pipeline["transform"]["refs"]:
4230
+ ue = os.path.join(root, run_dir, os.path.dirname(ref), "valid.ue.jsonl")
4231
+ if not os.path.exists(ue):
4232
+ continue
4233
+ with open(ue, "r", encoding="utf-8") as f:
4234
+ for line in f:
4235
+ line = line.strip()
4236
+ if not line:
4237
+ continue
4238
+ record = json.loads(line)
4239
+ if record.get("#type") == "track" and record.get("#event_name"):
4240
+ expected_events[record["#event_name"]] = expected_events.get(record["#event_name"], 0) + 1
4241
+ else:
4242
+ expected_users += 1
4243
+ lo, hi = visit_window(record, lo, hi)
4244
+
4245
+ if lo is None:
4246
+ print("verify: no UE records found in " + run_dir, file=sys.stderr)
4247
+ sys.exit(2)
4248
+
4249
+ def run_summary():
4250
+ proc = subprocess.run(
4251
+ ["ae-cli", "tracking", "ingest", "summary",
4252
+ "-p", str(project_id), "--start-time", lo, "--end-time", hi],
4253
+ capture_output=True, text=True,
4254
+ )
4255
+ if proc.returncode != 0:
4256
+ return {"error": (proc.stderr or proc.stdout).strip()[:500]}
4257
+ try:
4258
+ return json.loads(proc.stdout)
4259
+ except json.JSONDecodeError:
4260
+ return {"raw": proc.stdout.strip()[:500]}
4261
+
4262
+ print("window " + lo + " .. " + hi + " project_id=" + str(project_id))
4263
+ print("expected (this run):")
4264
+ for event, count in sorted(expected_events.items()):
4265
+ print(" {:<20}{:>10,}".format(event, count))
4266
+ print(" {:<20}{:>10,}".format("<user rows>", expected_users))
4267
+ print()
4268
+
4269
+ baseline_path = os.path.join(root, run_dir, "baseline.json")
4270
+ if mode == "--baseline":
4271
+ payload = run_summary()
4272
+ with open(baseline_path, "w", encoding="utf-8") as f:
4273
+ json.dump(payload, f, ensure_ascii=False, indent=2)
4274
+ print("baseline recorded: " + baseline_path)
4275
+ print("platform summary (before):")
4276
+ print(json.dumps(payload, ensure_ascii=False, indent=2))
4277
+ sys.exit(0)
4278
+
4279
+ if not os.path.exists(baseline_path):
4280
+ print("no baseline.json \u2014 run \`verify.py <run-dir> --baseline\` before upload.", file=sys.stderr)
4281
+ print("falling back to a single after-upload summary:", file=sys.stderr)
4282
+ print(json.dumps(run_summary(), ensure_ascii=False, indent=2))
4283
+ sys.exit(1)
4284
+
4285
+ with open(baseline_path, "r", encoding="utf-8") as f:
4286
+ baseline = json.load(f)
4287
+ after = run_summary()
4288
+ print("platform summary before vs after:")
4289
+ print(json.dumps(baseline, ensure_ascii=False, indent=2))
4290
+ print("---")
4291
+ print(json.dumps(after, ensure_ascii=False, indent=2))
4292
+ print()
4293
+ print("summary changed: " + ("yes" if baseline != after else "no"))
4294
+ print()
4295
+ print("Boundary: the summary payload is server-defined and may under-report even")
4296
+ print("landed data; this check surfaces it for comparison, it does not auto-verify")
4297
+ print("per-event counts, and a shared project's window delta is not attributed to")
4298
+ print("this import. Cross-check with:")
4299
+ print(" ae-cli tracking live-data list -p " + str(project_id))
4300
+ print("For a hard SQL judge, add a project custom layer (custom-layer.md).")
4301
+ sys.exit(0)
4302
+
4303
+
4304
+ if __name__ == "__main__":
4305
+ main()
4306
+ `;
4307
+ }
4308
+ function resolveAppidPy() {
4309
+ return `#!/usr/bin/env python3
4310
+ """Derive the destination APPID from \`ae-cli project info get --project-id <id>\`.
4311
+
4312
+ \`project info get\` returns \`data.appid\` at the top level (verified against the
4313
+ AE demo host), so this helper reads that exact field and prints it to stdout. It
4314
+ prints nothing to stdout and reports the payload when the field is absent or not
4315
+ a non-empty string \u2014 the caller then falls back to AE_APPID.
4316
+
4317
+ Usage: bin/resolve_appid.py <project-id>
4318
+ """
4319
+ import json
4320
+ import subprocess
4321
+ import sys
4322
+
4323
+
4324
+ def main():
4325
+ if len(sys.argv) != 2:
4326
+ print("usage: resolve_appid.py <project-id>", file=sys.stderr)
4327
+ sys.exit(2)
4328
+ project_id = sys.argv[1]
4329
+ proc = subprocess.run(
4330
+ ["ae-cli", "project", "info", "get", "--project-id", project_id],
4331
+ capture_output=True, text=True,
4332
+ )
4333
+ if proc.returncode != 0:
4334
+ print("resolve_appid: project info get failed: " + (proc.stderr or proc.stdout).strip()[:300], file=sys.stderr)
4335
+ sys.exit(0)
4336
+ try:
4337
+ parsed = json.loads(proc.stdout)
4338
+ except json.JSONDecodeError:
4339
+ print("resolve_appid: project info get returned non-JSON output", file=sys.stderr)
4340
+ sys.exit(0)
4341
+ data = parsed.get("data") if isinstance(parsed, dict) else None
4342
+ appid = data.get("appid") if isinstance(data, dict) else None
4343
+ if not isinstance(appid, str) or not appid:
4344
+ print("resolve_appid: project info get returned no appid; set AE_APPID", file=sys.stderr)
4345
+ print(json.dumps(data, ensure_ascii=False, indent=2) if data is not None else "{}", file=sys.stderr)
4346
+ sys.exit(0)
4347
+ masked = appid if len(appid) <= 4 else ("*" * (len(appid) - 4)) + appid[-4:]
4348
+ print("resolve_appid: resolved APPID data.appid = " + masked, file=sys.stderr)
4349
+ print(appid)
4350
+
4351
+
4352
+ if __name__ == "__main__":
4353
+ main()
4354
+ `;
4355
+ }
4356
+ function generateReadme() {
4357
+ return `# AE Data Integration \u2014 handoff package
4358
+
4359
+ A frozen, reusable pipeline for importing a **same-shape** local file into AE,
4360
+ generated by \`ae-cli data-integration handoff\`. Source and Transform are frozen
4361
+ (the confirmed business logic); only the Tracking-plan and Sink gates still
4362
+ require human confirmation.
4363
+
4364
+ ## Quick start
4365
+
4366
+ \`\`\`bash
4367
+ cp <today's file> inbox/
4368
+ bin/run.sh # shape check -> convert -> plan check (no upload)
4369
+ bin/upload.sh runs/<latest> # dry-run
4370
+ bin/upload.sh runs/<latest> --confirm
4371
+ \`\`\`
4372
+
4373
+ Read [RUNBOOK.md](RUNBOOK.md) for the full flow, the four confirmation gates,
4374
+ and how to verify persistence.
4375
+
4376
+ ## Handing off to an agent
4377
+
4378
+ Import \`inbox/<today's file>\` through the ae-data-integration skill using this
4379
+ package. Tell it to follow RUNBOOK.md and stop for confirmation before uploading.
4380
+
4381
+ ## Package layout
4382
+
4383
+ | Path | Purpose |
4384
+ | --- | --- |
4385
+ | \`pipeline.json\` | Declarative source -> transform -> sink descriptor |
4386
+ | \`index.json\` | Structure-fingerprint index (reuse matching) |
4387
+ | \`shape.json\` | Column baseline used by the shape gate |
4388
+ | \`<fingerprint16>/\` | Frozen mapping + tracking plan + transform wrapper |
4389
+ | \`bin/run.sh\` | Source + Transform + Plan executor (never uploads) |
4390
+ | \`bin/upload.sh\` | Sink executor (dry-run by default; resolves the recorded target) |
4391
+ | \`bin/verify.py\` | Soft persistence check (submit window vs ingest summary) |
4392
+ | \`bin/resolve_appid.py\` | APPID derivation helper (project info get) |
4393
+ | \`.local/target.env\` | Upload target overrides (APPID / endpoint); only the template ships |
4394
+ | \`inbox/\` \`runs/\` | Daily input / per-run outputs |
4395
+
4396
+ ## Safety
4397
+
4398
+ The package records at most a destination \`pushurl\` and \`project_id\` (no APPID,
4399
+ tokens, or raw data values). \`bin/upload.sh\` always requires \`--confirm\` before
4400
+ sending, so the operator re-confirms the address and project on every reuse. Copy
4401
+ \`.local/target.env.example\` to \`.local/target.env\` for explicit overrides and
4402
+ never commit it.
4403
+ `;
4404
+ }
4405
+ function generateRunbook() {
4406
+ return `# RUNBOOK \u2014 same-shape file import
4407
+
4408
+ Run this when a file of the **same shape** arrives again (same sheets and
4409
+ headers as \`shape.json\`). If the headers changed, stop: re-run the full
4410
+ ae-data-integration pipeline instead of editing the frozen mapping.
4411
+
4412
+ ## Gates
4413
+
4414
+ 1. **Shape gate** \u2014 \`bin/run.sh\` rebinds the frozen mappings to the new file
4415
+ and compares the column set against \`shape.json\`. A mismatch fails fast on
4416
+ purpose: a different shape means the frozen business logic was never reviewed
4417
+ for it.
4418
+ 2. **Transform** \u2014 each frozen mapping runs through
4419
+ \`ae-cli data-integration convert\`. Quarantined rows land in
4420
+ \`invalid.rows.jsonl\`; they are never silently dropped.
4421
+ 3. **Tracking-plan gate** \u2014 \`bin/run.sh\` verifies every produced event and
4422
+ property already exists in the frozen \`plan.json\`. New events or properties
4423
+ make it exit with code 3: merge them into the project tracking plan first,
4424
+ then return here to upload.
4425
+ 4. **Sink gate** \u2014 \`bin/upload.sh\` is dry-run by default. Read
4426
+ \`record_count\`, \`batch_count\`, and \`manifest_status\` before adding
4427
+ \`--confirm\`. A \`blocked\` manifest means rows were quarantined;
4428
+ \`--confirm\` then uploads only the valid subset (\`--allow-clean-subset\`).
4429
+
4430
+ ## Destination
4431
+
4432
+ The package records the destination it was handed off for when \`handoff\` was run
4433
+ with \`--pushurl\` / \`--project-id\` (see \`pipeline.json\` \u2192 \`sink.params\`). Reuse
4434
+ defaults to that target, but \`bin/upload.sh\` never sends without \`--confirm\`, so
4435
+ the operator re-confirms the address and project every time.
4436
+
4437
+ Resolution order at upload time:
4438
+
4439
+ - endpoint: recorded \`pushurl\` (+ \`/sync_json\`), else \`AE_ENDPOINT\`.
4440
+ - APPID: \`AE_APPID\`, else derived via \`ae-cli project info get --project-id <id>\`
4441
+ (see \`bin/resolve_appid.py\`; it reads the \`data.appid\` field \u2014 set \`AE_APPID\`
4442
+ when that field is absent).
4443
+ - project id: recorded \`project_id\`, else \`AE_PROJECT_ID\`.
4444
+
4445
+ \`.local/target.env\` remains the explicit override for all three:
4446
+ \`AE_ENDPOINT\` (full receiver URL ending in \`/sync_json\`), \`AE_APPID\`, \`AE_PROJECT_ID\`.
4447
+
4448
+ ## Verify persistence
4449
+
4450
+ \`receiver_accepted\` is not persistence. About a minute after upload, confirm
4451
+ the data landed with ae-cli. The package ships a soft check that automates the
4452
+ before/after comparison:
4453
+
4454
+ \`\`\`bash
4455
+ bin/verify.py runs/<run-id> --baseline # before upload
4456
+ bin/upload.sh runs/<run-id> --confirm
4457
+ bin/verify.py runs/<run-id> --check # after upload
4458
+ \`\`\`
4459
+
4460
+ \`bin/verify.py\` prints the submit window and expected counts, then shows the
4461
+ \`tracking ingest summary\` payload before and after for comparison. It does not
4462
+ auto-verify per-event counts \u2014 the summary shape is server-defined, and a shared
4463
+ project's window delta is not attributed to this import. Cross-check with:
4464
+
4465
+ \`\`\`bash
4466
+ ae-cli tracking live-data list -p <AE_PROJECT_ID>
4467
+ ae-cli tracking ingest-error list -p <AE_PROJECT_ID> --data-name <name>
4468
+ \`\`\`
4469
+
4470
+ For a hard per-event SQL judge, overlay a project custom layer instead of editing
4471
+ this package (see \`custom-layer.md\` in the ae-data-integration skill).
4472
+
4473
+ ## Interrupted uploads
4474
+
4475
+ If a batch times out or loses the network, that batch's state is unknown. Stop:
4476
+ verify what actually landed, then follow the ae-data-integration skill to resume
4477
+ from the verified offset. Never re-run the whole upload blindly.
4478
+
4479
+ ## Files
4480
+
4481
+ See \`README.md\` for the package layout.
4482
+ `;
4483
+ }
4484
+ function generateEnvTemplate() {
4485
+ return [
4486
+ "# Upload target overrides. Copy this file to .local/target.env and fill only",
4487
+ "# what the package does not already record (pipeline.json sink.params).",
4488
+ "# AE_ENDPOINT: a full receiver URL ending in /sync_json (used when no pushurl is recorded).",
4489
+ "# AE_APPID: the destination project APPID (overrides the project info get derivation).",
4490
+ "# AE_PROJECT_ID: the destination project ID, used when no project_id is recorded.",
4491
+ "AE_ENDPOINT=",
4492
+ "AE_APPID=",
4493
+ "AE_PROJECT_ID=",
4494
+ ""
4495
+ ].join("\n");
4496
+ }
4497
+ function generateGitignore() {
4498
+ return ["inbox/", "runs/", ".local/target.env", ""].join("\n");
4499
+ }
4500
+
4501
+ // src/commands/data-integration/handoff.ts
3123
4502
  var HANDOFF_INDEX_VERSION = "ae-data-integration-index/v1";
3124
4503
  var HANDOFF_DIR_LEN = 16;
3125
4504
  function structureFingerprint(mapping) {
3126
4505
  const canonical = {
3127
4506
  mode: mapping.mode,
3128
- time_field: mapping.time.field,
3129
- account_id_field: mapping.account_id_field ?? null,
3130
- distinct_id_field: mapping.distinct_id_field ?? null,
3131
- record_type_field: mapping.record_type_field ?? null,
3132
- event_name_field: mapping.event_name_field ?? null,
3133
- columns: mapping.properties.map((property) => ({ source: property.source, type: property.type })).sort((left, right) => left.source.localeCompare(right.source)),
3134
- excluded: [...mapping.exclude_columns ?? []].sort()
4507
+ format: mapping.source.format,
4508
+ columns: sourceColumns(mapping)
3135
4509
  };
3136
4510
  return createHash3("sha256").update(JSON.stringify(canonical)).digest("hex");
3137
4511
  }
@@ -3142,17 +4516,17 @@ function upsertIndexEntry(index, entry) {
3142
4516
  function buildHandoffPackage(outDir, mapping, planFile) {
3143
4517
  const fingerprint = structureFingerprint(mapping);
3144
4518
  const dirName = fingerprint.slice(0, HANDOFF_DIR_LEN);
3145
- const handoffDir = join2(outDir, dirName);
3146
- const indexPath = join2(outDir, "index.json");
4519
+ const handoffDir = join3(outDir, dirName);
4520
+ const indexPath = join3(outDir, "index.json");
3147
4521
  const index = readHandoffIndex(indexPath);
3148
4522
  const reusedExisting = index.entries.some((item) => item.fingerprint === fingerprint);
3149
4523
  mkdirSync2(handoffDir, { recursive: true, mode: 448 });
3150
4524
  chmodSync2(handoffDir, 448);
3151
- writeSecureJson2(join2(handoffDir, "mapping.json"), mapping);
3152
- writeSecureText2(join2(handoffDir, "transform.mjs"), createHandoffScript());
4525
+ writeSecureJson2(join3(handoffDir, "mapping.json"), mapping);
4526
+ writeSecureText2(join3(handoffDir, "transform.mjs"), createHandoffScript());
3153
4527
  let planFileRel;
3154
4528
  if (planFile) {
3155
- writeSecureJson2(join2(handoffDir, "plan.json"), readPlanFile(planFile));
4529
+ writeSecureJson2(join3(handoffDir, "plan.json"), readPlanFile(planFile));
3156
4530
  planFileRel = `${dirName}/plan.json`;
3157
4531
  }
3158
4532
  const entry = {
@@ -3167,7 +4541,7 @@ function buildHandoffPackage(outDir, mapping, planFile) {
3167
4541
  ...planFileRel ? { plan_file: planFileRel } : {}
3168
4542
  };
3169
4543
  writeAtomicJson(indexPath, upsertIndexEntry(index, entry));
3170
- return { outDir, handoffDir, dirName, fingerprint, reusedExisting, planFile: planFileRel };
4544
+ return { outDir, handoffDir, dirName, fingerprint, reusedExisting, planFile: planFileRel, entry };
3171
4545
  }
3172
4546
  function readHandoffIndex(path) {
3173
4547
  let raw;
@@ -3262,56 +4636,199 @@ var dataIntegrationHandoff = {
3262
4636
  service: "data-integration",
3263
4637
  command: "handoff",
3264
4638
  usesAeHost: false,
3265
- description: "Export a reusable handoff package (frozen mapping + transform script + plan reference) under .ae-data-integration/.",
4639
+ description: "Export a reusable handoff package (pipeline descriptor + frozen mappings + stage executors + docs) and a shareable zip.",
3266
4640
  flags: [
3267
- { name: "mapping", type: "string", required: true, sensitive: true, desc: "Confirmed ae-local-data-mapping/v1 JSON, file path, or @file." },
4641
+ { name: "mapping", type: "string", required: true, variadic: true, sensitive: true, desc: `Confirmed ${MAPPING_VERSION} JSON, file path, or @file. Repeat for multiple data sets (one sheet each).` },
3268
4642
  { name: "plan-file", type: "string", sensitive: true, desc: "Tracking-plan draft.json to reference inside the handoff package." },
3269
- { name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-data-integration." }
4643
+ { name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-cli/data-integration (project workspace)." },
4644
+ { name: "pushurl", type: "string", sensitive: true, desc: "Receiver base URL to record as the reuse upload target (sink endpoint = pushurl + /sync_json). Reuse still requires --confirm." },
4645
+ { name: "project-id", type: "string", sensitive: true, desc: "Numeric destination project ID to record; upload derives the APPID from it via project info get." }
3270
4646
  ],
3271
4647
  risk: "write",
3272
4648
  dryRun: async (ctx) => {
3273
- const mapping = readLocalDataMapping(ctx.str("mapping"));
3274
- const outDir = resolve3(ctx.str("out-dir").trim() || ".ae-data-integration");
4649
+ const mappings = ctx.list("mapping").map((source) => readLocalDataMapping(source));
4650
+ const outDir = resolve3(ctx.str("out-dir").trim() || join3(".ae-cli", "data-integration"));
3275
4651
  const planFile = ctx.str("plan-file").trim() || void 0;
4652
+ const pushurl = ctx.str("pushurl").trim() || void 0;
4653
+ const projectId = ctx.str("project-id").trim() || void 0;
4654
+ const fingerprints = mappings.map(structureFingerprint);
3276
4655
  return {
3277
4656
  action: "handoff_local_data",
3278
4657
  out_dir: outDir,
3279
- fingerprint: structureFingerprint(mapping),
3280
- source_sha256: mapping.source.sha256,
3281
- mode: mapping.mode,
3282
- property_count: mapping.properties.length,
3283
- files: ["mapping.json", "transform.mjs", ...planFile ? ["plan.json"] : []],
3284
- index_file: join2(outDir, "index.json")
4658
+ mapping_count: mappings.length,
4659
+ fingerprints,
4660
+ target: { pushurl, project_id: projectId },
4661
+ files: handoffFileList(fingerprints, planFile),
4662
+ index_file: join3(outDir, "index.json"),
4663
+ pipeline_file: join3(outDir, "pipeline.json"),
4664
+ shape_file: join3(outDir, "shape.json"),
4665
+ zip_path: zipPathFor(outDir, fingerprints[0])
3285
4666
  };
3286
4667
  },
3287
4668
  execute: async (ctx) => {
3288
- const mapping = readLocalDataMapping(ctx.str("mapping"));
3289
- const outDir = resolve3(ctx.str("out-dir").trim() || ".ae-data-integration");
4669
+ const mappings = ctx.list("mapping").map((source) => readLocalDataMapping(source));
4670
+ const outDir = resolve3(ctx.str("out-dir").trim() || join3(".ae-cli", "data-integration"));
3290
4671
  const planFile = ctx.str("plan-file").trim() || void 0;
3291
- const build = buildHandoffPackage(outDir, mapping, planFile);
4672
+ const target = {
4673
+ pushurl: ctx.str("pushurl").trim() || void 0,
4674
+ project_id: ctx.str("project-id").trim() || void 0
4675
+ };
4676
+ const items = mappings.map((mapping) => ({
4677
+ mapping,
4678
+ build: buildHandoffPackage(outDir, mapping, planFile)
4679
+ }));
4680
+ const entries = items.map((item) => item.build.entry);
4681
+ const relayFiles = [
4682
+ { relPath: "pipeline.json", content: `${JSON.stringify(buildPipelineDescriptor(entries, target), null, 2)}
4683
+ `, mode: 384 },
4684
+ { relPath: "shape.json", content: `${JSON.stringify(buildShapeBaseline(items.map((item) => ({ mapping: item.mapping, fingerprint: item.build.fingerprint }))), null, 2)}
4685
+ `, mode: 384 },
4686
+ ...generateBinScripts(),
4687
+ { relPath: "README.md", content: generateReadme(), mode: 384 },
4688
+ { relPath: "RUNBOOK.md", content: generateRunbook(), mode: 384 },
4689
+ { relPath: ".local/target.env.example", content: generateEnvTemplate(), mode: 384 },
4690
+ { relPath: ".gitignore", content: generateGitignore(), mode: 384 },
4691
+ { relPath: "inbox/.gitkeep", content: "", mode: 384 },
4692
+ { relPath: "runs/.gitkeep", content: "", mode: 384 }
4693
+ ];
4694
+ for (const file of relayFiles) writeRelayFile(outDir, file);
4695
+ const zipPath = zipPathFor(outDir, entries[0].fingerprint);
4696
+ const stagingDir = stageScopedPackage(outDir, items.map((item) => item.build), entries, relayFiles);
4697
+ try {
4698
+ await zipPackage(stagingDir, zipPath);
4699
+ chmodSync2(zipPath, 384);
4700
+ } finally {
4701
+ rmSync(stagingDir, { recursive: true, force: true });
4702
+ }
3292
4703
  return {
3293
- out_dir: build.outDir,
3294
- handoff_dir: build.handoffDir,
3295
- fingerprint: build.fingerprint,
3296
- mapping_file: `${build.dirName}/mapping.json`,
3297
- ...build.planFile ? { plan_file: build.planFile } : {},
3298
- run: `node ${join2(build.handoffDir, "transform.mjs")} <new-input-file> [<output-dir>]`,
3299
- index_file: join2(outDir, "index.json"),
3300
- reused_existing: build.reusedExisting
4704
+ out_dir: outDir,
4705
+ zip_path: zipPath,
4706
+ pipeline_file: join3(outDir, "pipeline.json"),
4707
+ shape_file: join3(outDir, "shape.json"),
4708
+ index_file: join3(outDir, "index.json"),
4709
+ handoff_dirs: items.map((item) => item.build.dirName),
4710
+ deliverables: buildDeliverables(outDir, relayFiles, items),
4711
+ next_steps: [
4712
+ `Review the pipeline descriptor and four confirmation gates: ${join3(outDir, "RUNBOOK.md")}.`,
4713
+ `Share the archive: ${zipPath}.`,
4714
+ `Next same-shape file: cd ${outDir} && bin/run.sh <new-file>, then bin/upload.sh runs/<run-id> --confirm.`
4715
+ ],
4716
+ reused_existing: items.some((item) => item.build.reusedExisting)
3301
4717
  };
3302
4718
  }
3303
4719
  };
4720
+ function zipPathFor(outDir, fingerprint) {
4721
+ if (!fingerprint) return void 0;
4722
+ return join3(dirname2(resolve3(outDir)), `ae-data-integration-handoff-${fingerprint.slice(0, 8)}.zip`);
4723
+ }
4724
+ var RELAY_FILE_PATHS = [
4725
+ "pipeline.json",
4726
+ "shape.json",
4727
+ "index.json",
4728
+ "README.md",
4729
+ "RUNBOOK.md",
4730
+ ".local/target.env.example",
4731
+ ".gitignore",
4732
+ "bin/run.sh",
4733
+ "bin/upload.sh",
4734
+ "bin/bind_mapping.py",
4735
+ "bin/summarize.py",
4736
+ "bin/plan_check.py",
4737
+ "bin/verify.py",
4738
+ "bin/resolve_appid.py",
4739
+ "inbox/.gitkeep",
4740
+ "runs/.gitkeep"
4741
+ ];
4742
+ function mappingDirFiles(dirName, planFile) {
4743
+ return [
4744
+ `${dirName}/mapping.json`,
4745
+ `${dirName}/transform.mjs`,
4746
+ ...planFile ? [`${dirName}/plan.json`] : []
4747
+ ];
4748
+ }
4749
+ function handoffFileList(fingerprints, planFile) {
4750
+ const perMapping = fingerprints.flatMap(
4751
+ (fingerprint) => mappingDirFiles(fingerprint.slice(0, HANDOFF_DIR_LEN), planFile)
4752
+ );
4753
+ return [...RELAY_FILE_PATHS, ...perMapping];
4754
+ }
4755
+ function buildDeliverables(outDir, relayFiles, items) {
4756
+ const relay = [
4757
+ { rel_path: "index.json", abs_path: join3(outDir, "index.json") },
4758
+ ...relayFiles.map((file) => ({ rel_path: file.relPath, abs_path: join3(outDir, file.relPath) }))
4759
+ ];
4760
+ const mappings = items.flatMap(
4761
+ (item) => mappingDirFiles(item.build.dirName, item.build.planFile).map((rel) => ({
4762
+ rel_path: rel,
4763
+ abs_path: join3(outDir, rel)
4764
+ }))
4765
+ );
4766
+ return [...relay, ...mappings];
4767
+ }
4768
+ function writeRelayFile(outDir, file) {
4769
+ const abs = join3(outDir, file.relPath);
4770
+ mkdirSync2(dirname2(abs), { recursive: true, mode: 448 });
4771
+ writeFileSync2(abs, file.content, { encoding: "utf8", mode: file.mode });
4772
+ chmodSync2(abs, file.mode);
4773
+ }
4774
+ function stageScopedPackage(outDir, mappingDirs, entries, relayFiles) {
4775
+ const staging = mkdtempSync(join3(tmpdir(), "ae-handoff-"));
4776
+ writeSecureJson2(join3(staging, "index.json"), { version: HANDOFF_INDEX_VERSION, entries });
4777
+ for (const file of relayFiles) writeRelayFile(staging, file);
4778
+ for (const { dirName, planFile } of mappingDirs) {
4779
+ for (const rel of mappingDirFiles(dirName, planFile)) {
4780
+ const dest = join3(staging, rel);
4781
+ mkdirSync2(dirname2(dest), { recursive: true, mode: 448 });
4782
+ writeFileSync2(dest, readFileSync4(join3(outDir, rel), "utf8"), { encoding: "utf8", mode: 384 });
4783
+ chmodSync2(dest, 384);
4784
+ }
4785
+ }
4786
+ return staging;
4787
+ }
3304
4788
 
3305
- // src/commands/data-integration/local-data/reuse.ts
4789
+ // src/commands/data-integration/reuse.ts
3306
4790
  import { readFileSync as readFileSync5 } from "fs";
3307
- import { dirname as dirname2, join as join3, resolve as resolve4 } from "path";
4791
+ import { dirname as dirname4, join as join5, resolve as resolve5 } from "path";
4792
+
4793
+ // src/commands/data-integration/handoff-root.ts
4794
+ import { existsSync as existsSync2 } from "fs";
4795
+ import { dirname as dirname3, join as join4, resolve as resolve4 } from "path";
4796
+ function globalHandoffDir() {
4797
+ const home = process.env.HOME;
4798
+ if (!home) return void 0;
4799
+ return join4(getConfigDir(), "data-integration");
4800
+ }
4801
+ function upwardHandoffDirs(startDir) {
4802
+ const dirs = [];
4803
+ let current = resolve4(startDir);
4804
+ for (; ; ) {
4805
+ const candidate = join4(current, ".ae-cli", "data-integration");
4806
+ if (!dirs.includes(candidate)) dirs.push(candidate);
4807
+ const parent = dirname3(current);
4808
+ if (parent === current) break;
4809
+ current = parent;
4810
+ }
4811
+ return dirs;
4812
+ }
4813
+ function reuseSearchPaths(startDir = process.cwd()) {
4814
+ const globalDir = globalHandoffDir();
4815
+ return globalDir ? [...upwardHandoffDirs(startDir), globalDir] : upwardHandoffDirs(startDir);
4816
+ }
4817
+ function findReuseRoot(startDir = process.cwd()) {
4818
+ for (const dir of upwardHandoffDirs(startDir)) {
4819
+ if (existsSync2(join4(dir, "index.json"))) return dir;
4820
+ }
4821
+ return globalHandoffDir();
4822
+ }
4823
+
4824
+ // src/commands/data-integration/reuse.ts
3308
4825
  function detectReuse(mapping, outDir) {
3309
4826
  const fingerprint = structureFingerprint(mapping);
3310
- const indexPath = join3(outDir, "index.json");
4827
+ const indexPath = join5(outDir, "index.json");
3311
4828
  const index = readHandoffIndex(indexPath);
3312
4829
  const entry = index.entries.find((item) => item.fingerprint === fingerprint);
3313
4830
  if (!entry) return { matched: false, fingerprint, index_file: indexPath };
3314
- const mappingPath = join3(outDir, entry.mapping_file);
4831
+ const mappingPath = join5(outDir, entry.mapping_file);
3315
4832
  const match = {
3316
4833
  fingerprint: entry.fingerprint,
3317
4834
  created_at: entry.created_at,
@@ -3322,7 +4839,7 @@ function detectReuse(mapping, outDir) {
3322
4839
  mapping_file: entry.mapping_file,
3323
4840
  ...entry.plan_file ? { plan_file: entry.plan_file } : {},
3324
4841
  ...readFrozenEventName(mappingPath),
3325
- run: `node ${join3(outDir, dirname2(entry.mapping_file), "transform.mjs")} <new-input-file> [<output-dir>]`
4842
+ run: `node ${join5(outDir, dirname4(entry.mapping_file), "transform.mjs")} <new-input-file> [<output-dir>]`
3326
4843
  };
3327
4844
  return { matched: true, fingerprint, index_file: indexPath, match };
3328
4845
  }
@@ -3345,29 +4862,41 @@ var dataIntegrationReuse = {
3345
4862
  service: "data-integration",
3346
4863
  command: "reuse",
3347
4864
  usesAeHost: false,
3348
- description: "Match a candidate mapping against the .ae-data-integration/ handoff index and propose a reusable package.",
4865
+ description: "Match a candidate mapping against the .ae-cli/data-integration/ handoff index and propose a reusable package.",
3349
4866
  flags: [
3350
- { name: "mapping", type: "string", required: true, sensitive: true, desc: "Candidate ae-local-data-mapping/v1 JSON, file path, or @file (typically inspect recommended_mapping)." },
3351
- { name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-data-integration." }
4867
+ { name: "mapping", type: "string", required: true, sensitive: true, desc: `Candidate ${MAPPING_VERSION} JSON, file path, or @file (typically inspect recommended_mapping).` },
4868
+ { name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: nearest .ae-cli/data-integration/ upward from cwd, then ~/.ae-cli/data-integration/." }
3352
4869
  ],
3353
4870
  risk: "read",
3354
4871
  dryRun: async (ctx) => {
3355
4872
  const mapping = readLocalDataMapping(ctx.str("mapping"));
3356
- const outDir = resolve4(ctx.str("out-dir").trim() || ".ae-data-integration");
4873
+ const outDir = resolveReuseOutDir(ctx);
3357
4874
  const result = detectReuse(mapping, outDir);
3358
4875
  return {
3359
4876
  action: "reuse_detect",
3360
4877
  fingerprint: result.fingerprint,
3361
4878
  matched: result.matched,
3362
- index_file: result.index_file
4879
+ index_file: result.index_file,
4880
+ searched_paths: searchedIndexPaths(ctx)
3363
4881
  };
3364
4882
  },
3365
4883
  execute: async (ctx) => {
3366
4884
  const mapping = readLocalDataMapping(ctx.str("mapping"));
3367
- const outDir = resolve4(ctx.str("out-dir").trim() || ".ae-data-integration");
3368
- return detectReuse(mapping, outDir);
4885
+ const outDir = resolveReuseOutDir(ctx);
4886
+ const result = detectReuse(mapping, outDir);
4887
+ return { ...result, searched_paths: searchedIndexPaths(ctx) };
3369
4888
  }
3370
4889
  };
4890
+ function resolveReuseOutDir(ctx) {
4891
+ const explicit = ctx.str("out-dir").trim();
4892
+ if (explicit) return resolve5(explicit);
4893
+ return findReuseRoot() ?? resolve5(join5(".ae-cli", "data-integration"));
4894
+ }
4895
+ function searchedIndexPaths(ctx) {
4896
+ const explicit = ctx.str("out-dir").trim();
4897
+ const dirs = explicit ? [resolve5(explicit)] : reuseSearchPaths();
4898
+ return dirs.map((dir) => join5(dir, "index.json"));
4899
+ }
3371
4900
 
3372
4901
  // src/commands/data-integration/index.ts
3373
4902
  var commands = [dataIntegrationInspect, dataIntegrationPlan, dataIntegrationConvert, dataIntegrationUpload, dataIntegrationHandoff, dataIntegrationReuse];