@thinkingai/ae-cli 6.1.14 → 6.1.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/README.zh.md +5 -2
- package/dist/{auth-GBMV6TEJ.js → auth-2WTQOP77.js} +3 -3
- package/dist/{auth-NDSXE54J.js → auth-77BUFLGC.js} +57 -13
- package/dist/{capability-TAMDRZYV.js → capability-72DTW5M2.js} +9 -9
- package/dist/{capability-U7TDEEEG.js → capability-PJHNI4GJ.js} +9 -9
- package/dist/{chunk-AFXA7BRK.js → chunk-4KVPKXFX.js} +5 -5
- package/dist/chunk-4SGZG4XY.js +311 -0
- package/dist/chunk-6EIJSNBD.js +8 -0
- package/dist/chunk-C4MGVGJW.js +13 -0
- package/dist/{chunk-WZRX4KOH.js → chunk-GT46FPXN.js} +4 -6
- package/dist/{chunk-5XUSIK27.js → chunk-ILIU36SU.js} +10 -7
- package/dist/{chunk-JHENBQ5B.js → chunk-LYVNONC4.js} +0 -35
- package/dist/{chunk-JUW4AJXN.js → chunk-P3FGXJTU.js} +12 -7
- package/dist/chunk-QGM4M3NI.js +37 -0
- package/dist/{chunk-753BUTNZ.js → chunk-RGKJGKT7.js} +2 -2
- package/dist/{chunk-UIHQJK5E.js → chunk-SAU3QFIQ.js} +3 -3
- package/dist/{chunk-VLWOLBGZ.js → chunk-UOUS37JQ.js} +149 -161
- package/dist/{chunk-AXDXJTPC.js → chunk-UW5UN47B.js} +19 -2
- package/dist/chunk-VPKZ7I72.js +509 -0
- package/dist/{chunk-QATA32VR.js → chunk-VR3LCBHW.js} +2 -2
- package/dist/{chunk-IBH3LDAH.js → chunk-YA6SMTXG.js} +3 -3
- package/dist/chunk-ZZUOD757.js +598 -0
- package/dist/{client-L2YDMHQ6.js → client-TKG4WBHN.js} +4 -5
- package/dist/{community-report-client-M2RW4MXD.js → community-report-client-FI4LNVYS.js} +3 -2
- package/dist/{config-OL2LWGBV.js → config-RE6CMGPK.js} +7 -7
- package/dist/data-integration-XQYB4X4F.js +3503 -0
- package/dist/index.js +68 -34
- package/dist/local-data-upload-client-BWHSUQQK.js +167 -0
- package/dist/{memory-MUP7PPL7.js → memory-CHRU2F7W.js} +7 -6
- package/dist/{memory-U4O5PMXH.js → memory-YK33G4T7.js} +7 -6
- package/dist/{metadata-5MIMNIMT.js → metadata-UILXHBWF.js} +10 -9
- package/dist/{metadata-LERKDJN6.js → metadata-XXR34N5P.js} +10 -9
- package/dist/{model-JTUEO5M4.js → model-K3KLWIW6.js} +7 -3
- package/dist/model-NR3JHFSJ.js +139 -0
- package/dist/{sync-MOSFNBVR.js → sync-DAVKYVMW.js} +8 -6
- package/dist/sync-FCKOVWWS.js +10261 -0
- package/dist/{te-agent-IFKZDHZI.js → te-agent-4BKBODMF.js} +686 -71
- package/dist/te-agent-HLW4VTQK.js +3893 -0
- package/dist/{te-analysis-FCHRTNIY.js → te-analysis-O6DCO6BS.js} +36 -13
- package/dist/{te-analysis-QG7UKGDA.js → te-analysis-ZMNGOVNW.js} +36 -13
- package/dist/{te-community-HNKVTERD.js → te-community-6HPBWJUZ.js} +24 -19
- package/dist/{te-community-IWE5B7W6.js → te-community-HLC43QKH.js} +24 -20
- package/dist/{te-dataops-KXCEB4CS.js → te-dataops-EJP56W3K.js} +57 -34
- package/dist/{te-dataops-KQPYNAE3.js → te-dataops-HDRUXY4K.js} +59 -34
- package/dist/{te-engage-D6EG3NOR.js → te-engage-FGBGQ4IY.js} +9 -8
- package/dist/{te-engage-ZSMIJLUW.js → te-engage-RAK5PESW.js} +9 -8
- package/dist/{te-experiment-K5US7RMG.js → te-experiment-SO5MPDMJ.js} +9 -8
- package/dist/{te-experiment-WA7TFMEL.js → te-experiment-VZF7BT6G.js} +9 -8
- package/dist/{te-kb-OIH3T6CS.js → te-kb-SQCLHG6X.js} +250 -126
- package/dist/{te-system-AZ3URMUO.js → te-system-YARIK4S5.js} +7 -4
- package/dist/te-system-Z77IKZFN.js +2213 -0
- package/dist/{te-team-GZPU6UWA.js → te-team-EFKWYKMK.js} +7 -7
- package/dist/{update-TOBFXF2V.js → update-OGPSZM5A.js} +7 -7
- package/package.json +14 -3
- package/skills/ae-agent/SKILL.md +21 -11
- package/skills/ae-agent/references/approval-effect.md +44 -0
- package/skills/ae-agent/references/approval-request.md +46 -0
- package/skills/ae-agent/references/approval-task.md +45 -0
- package/skills/ae-agent/references/approval-type.md +32 -0
- package/skills/ae-agent/references/command_index.md +19 -0
- package/skills/ae-analysis/SKILL.md +11 -7
- package/skills/ae-analysis/references/adhoc_run.md +2 -0
- package/skills/ae-analysis/references/ai_models.md +27 -0
- package/skills/ae-analysis/references/analysis_gateway_assets.md +3 -2
- package/skills/ae-analysis/references/command_index.md +4 -1
- package/skills/ae-analysis/references/dashboard_get.md +8 -2
- package/skills/ae-analysis/references/dashboard_report_data_run.md +3 -1
- package/skills/ae-analysis/references/dashboard_update.md +4 -1
- package/skills/ae-analysis/references/project_space_business_filter_upsert.md +17 -0
- package/skills/ae-analysis/references/user_tag_models.md +4 -4
- package/skills/ae-data-integration/SKILL.md +54 -0
- package/skills/ae-data-integration/references/handoff.md +43 -0
- package/skills/ae-data-integration/references/local-analysis.md +27 -0
- package/skills/ae-data-integration/references/reuse.md +41 -0
- package/skills/ae-data-integration/references/sink-upload.md +58 -0
- package/skills/ae-data-integration/references/source-inspect.md +68 -0
- package/skills/ae-data-integration/references/sync-json-upload.md +60 -0
- package/skills/ae-data-integration/references/tracking-plan.md +35 -0
- package/skills/ae-data-integration/references/transform.md +65 -0
- package/skills/ae-data-integration/references/ue-mapping.md +142 -0
- package/skills/ae-data-integration/references/ue-routing.md +35 -0
- package/skills/ae-data-integration-helper/SKILL.md +1 -1
- package/skills/ae-dataops/SKILL.md +1 -1
- package/skills/ae-dataops/references/dataops-query.md +4 -4
- package/skills/ae-generate-tracking-plan/SKILL.md +54 -3
- package/skills/ae-kb/SKILL.md +55 -50
- package/skills/ae-kb/references/query-workflow.md +112 -0
- package/skills/ae-kb-discovery/SKILL.md +105 -0
- package/skills/ae-metadata/SKILL.md +0 -1
- package/dist/chunk-3FY3RJ26.js +0 -293
- package/dist/chunk-S5NTSDBS.js +0 -198
- package/dist/chunk-ZQKDZXDO.js +0 -317
- package/dist/cli-token-4UPER74P.js +0 -21
|
@@ -0,0 +1,3503 @@
|
|
|
1
|
+
import {
|
|
2
|
+
CliValidationError,
|
|
3
|
+
LocalDataUploadError
|
|
4
|
+
} from "./chunk-UW5UN47B.js";
|
|
5
|
+
import "./chunk-QGM4M3NI.js";
|
|
6
|
+
|
|
7
|
+
// src/commands/data-integration/local-data/inspect.ts
|
|
8
|
+
import { basename as basename3 } from "path";
|
|
9
|
+
|
|
10
|
+
// src/commands/data-integration/local-data/estimate.ts
|
|
11
|
+
var XLS_SIZE_WARN_BYTES = 100 * 1024 * 1024;
|
|
12
|
+
var LARGE_FILE_WARN_BYTES = 1024 * 1024 * 1024;
|
|
13
|
+
var XLS_HARD_LIMIT_BYTES = 1024 * 1024 * 1024;
|
|
14
|
+
var THROUGHPUT_BYTES_PER_SECOND = {
|
|
15
|
+
jsonl: 10 * 1024 * 1024,
|
|
16
|
+
json: 8 * 1024 * 1024,
|
|
17
|
+
csv: 4 * 1024 * 1024,
|
|
18
|
+
tsv: 4 * 1024 * 1024,
|
|
19
|
+
xlsx: 2 * 1024 * 1024
|
|
20
|
+
};
|
|
21
|
+
var SLOW_ENCODINGS = /* @__PURE__ */ new Set(["gbk", "gb2312", "gb18030", "big5", "shift_jis", "euc-jp", "euc-kr"]);
|
|
22
|
+
function estimateProcessingSeconds(format, sizeBytes, encoding) {
|
|
23
|
+
if (format === "xls") return 0;
|
|
24
|
+
let throughput = THROUGHPUT_BYTES_PER_SECOND[format];
|
|
25
|
+
if (encoding && SLOW_ENCODINGS.has(encoding.toLowerCase())) throughput /= 2;
|
|
26
|
+
return sizeBytes / throughput;
|
|
27
|
+
}
|
|
28
|
+
function formatDuration(seconds) {
|
|
29
|
+
if (seconds < 60) {
|
|
30
|
+
const low2 = Math.max(1, Math.round(seconds * 0.5));
|
|
31
|
+
const high2 = Math.max(low2 + 1, Math.round(seconds * 1.5));
|
|
32
|
+
return `roughly ${low2} to ${high2} seconds`;
|
|
33
|
+
}
|
|
34
|
+
const low = Math.max(1, Math.round(seconds / 60 * 0.5));
|
|
35
|
+
const high = Math.max(low + 1, Math.round(seconds / 60 * 1.5));
|
|
36
|
+
return `roughly ${low} to ${high} minutes`;
|
|
37
|
+
}
|
|
38
|
+
function humanSize(bytes) {
|
|
39
|
+
if (bytes >= 1024 * 1024 * 1024) return `${(bytes / (1024 * 1024 * 1024)).toFixed(1)} GB`;
|
|
40
|
+
return `${Math.round(bytes / (1024 * 1024))} MB`;
|
|
41
|
+
}
|
|
42
|
+
function xlsMemoryWarning(fileName, sizeBytes) {
|
|
43
|
+
return `Warning: ${fileName} is ${humanSize(sizeBytes)} (XLS). The legacy XLS parser loads the entire workbook into memory, roughly 5-10x the file size. Prefer converting to XLSX or splitting the workbook first.`;
|
|
44
|
+
}
|
|
45
|
+
function largeFileTimeWarning(fileName, sizeBytes, format, encoding) {
|
|
46
|
+
const duration = formatDuration(estimateProcessingSeconds(format, sizeBytes, encoding));
|
|
47
|
+
return `Warning: ${fileName} is ${humanSize(sizeBytes)}; processing is estimated to take ${duration}. Consider splitting the file if that is too long.`;
|
|
48
|
+
}
|
|
49
|
+
function largeFileMemoryWarning(format) {
|
|
50
|
+
if (format === "xlsx") {
|
|
51
|
+
return "Memory risk: XLSX processing keeps the workbook's shared string table in memory, so peak memory can substantially exceed the file size.";
|
|
52
|
+
}
|
|
53
|
+
if (format === "json") {
|
|
54
|
+
return "Memory risk: a JSON root object or a very large record may be materialized in memory during inspection, so peak memory can substantially exceed the file size.";
|
|
55
|
+
}
|
|
56
|
+
return null;
|
|
57
|
+
}
|
|
58
|
+
function assessFileSize(fileName, format, sizeBytes, encoding) {
|
|
59
|
+
const baseAssessment = {
|
|
60
|
+
size: humanSize(sizeBytes),
|
|
61
|
+
estimatedDuration: null,
|
|
62
|
+
warning: null,
|
|
63
|
+
reason: null,
|
|
64
|
+
memoryRisk: false,
|
|
65
|
+
rejected: false
|
|
66
|
+
};
|
|
67
|
+
if (format === "xls") {
|
|
68
|
+
if (sizeBytes > XLS_HARD_LIMIT_BYTES) {
|
|
69
|
+
return {
|
|
70
|
+
...baseAssessment,
|
|
71
|
+
reason: "XLS files larger than 1 GB are not supported; convert the workbook to XLSX or split it first.",
|
|
72
|
+
memoryRisk: true,
|
|
73
|
+
rejected: true
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
if (sizeBytes > XLS_SIZE_WARN_BYTES) {
|
|
77
|
+
return {
|
|
78
|
+
...baseAssessment,
|
|
79
|
+
warning: xlsMemoryWarning(fileName, sizeBytes),
|
|
80
|
+
memoryRisk: true
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
return baseAssessment;
|
|
84
|
+
}
|
|
85
|
+
if (sizeBytes > LARGE_FILE_WARN_BYTES) {
|
|
86
|
+
const memoryWarning = largeFileMemoryWarning(format);
|
|
87
|
+
const timeWarning = largeFileTimeWarning(fileName, sizeBytes, format, encoding);
|
|
88
|
+
return {
|
|
89
|
+
...baseAssessment,
|
|
90
|
+
estimatedDuration: formatDuration(estimateProcessingSeconds(format, sizeBytes, encoding)),
|
|
91
|
+
warning: memoryWarning ? `${timeWarning} ${memoryWarning}` : timeWarning,
|
|
92
|
+
memoryRisk: Boolean(memoryWarning)
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
return baseAssessment;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// src/commands/data-integration/local-data/input.ts
|
|
99
|
+
import { createHash } from "crypto";
|
|
100
|
+
import { createReadStream as createReadStream2, statSync } from "fs";
|
|
101
|
+
import { createRequire as createRequire2 } from "module";
|
|
102
|
+
import { extname, basename } from "path";
|
|
103
|
+
import { createInterface } from "readline";
|
|
104
|
+
import { pipeline } from "stream/promises";
|
|
105
|
+
import XLSXMod from "xlsx";
|
|
106
|
+
import { parse as parseCsv } from "csv-parse";
|
|
107
|
+
import { parse as parseCsvSync } from "csv-parse/sync";
|
|
108
|
+
import createJsonParser from "stream-json";
|
|
109
|
+
import Pick from "stream-json/filters/Pick.js";
|
|
110
|
+
import StreamArray from "stream-json/streamers/StreamArray.js";
|
|
111
|
+
import StreamValues from "stream-json/streamers/StreamValues.js";
|
|
112
|
+
import * as unzipper from "unzipper";
|
|
113
|
+
import { SaxesParser } from "saxes";
|
|
114
|
+
|
|
115
|
+
// src/commands/data-integration/local-data/encoding.ts
|
|
116
|
+
import { createReadStream } from "fs";
|
|
117
|
+
import { createRequire } from "module";
|
|
118
|
+
import { openSync, readSync, closeSync } from "fs";
|
|
119
|
+
var require2 = createRequire(import.meta.url);
|
|
120
|
+
var jschardet = require2("jschardet");
|
|
121
|
+
var iconv = require2("iconv-lite");
|
|
122
|
+
var NODE_ENCODINGS = /* @__PURE__ */ new Set(["utf-8", "utf8", "utf16le", "ucs2", "latin1", "ascii", "base64", "hex"]);
|
|
123
|
+
var NODE_UTF8 = /* @__PURE__ */ new Set(["utf-8", "utf8"]);
|
|
124
|
+
var DETECT_SAMPLE_BYTES = 64 * 1024;
|
|
125
|
+
function detectEncoding(filePath) {
|
|
126
|
+
const raw = readSample(filePath, DETECT_SAMPLE_BYTES);
|
|
127
|
+
const detected = jschardet.detect(raw);
|
|
128
|
+
const encoding = (detected.encoding || "utf-8").toLowerCase();
|
|
129
|
+
if (NODE_ENCODINGS.has(encoding) || iconv.encodingExists(encoding)) return encoding;
|
|
130
|
+
return "utf-8";
|
|
131
|
+
}
|
|
132
|
+
function decodeBuffer(raw, encoding) {
|
|
133
|
+
const normalized = encoding.toLowerCase();
|
|
134
|
+
if (NODE_ENCODINGS.has(normalized)) return raw.toString(normalized);
|
|
135
|
+
return iconv.decode(raw, normalized);
|
|
136
|
+
}
|
|
137
|
+
function decodeTextStream(filePath, encoding) {
|
|
138
|
+
const normalized = encoding.toLowerCase();
|
|
139
|
+
if (NODE_UTF8.has(normalized)) return createReadStream(filePath, { encoding: "utf8" });
|
|
140
|
+
if (!iconv.encodingExists(normalized)) {
|
|
141
|
+
throw new Error(`Unsupported source encoding: ${encoding}`);
|
|
142
|
+
}
|
|
143
|
+
return createReadStream(filePath).pipe(iconv.decodeStream(normalized));
|
|
144
|
+
}
|
|
145
|
+
function decodeFileSample(filePath, encoding, maxBytes = DETECT_SAMPLE_BYTES) {
|
|
146
|
+
return decodeBuffer(readSample(filePath, maxBytes), encoding);
|
|
147
|
+
}
|
|
148
|
+
function readSample(filePath, maxBytes) {
|
|
149
|
+
const fd = openSync(filePath, "r");
|
|
150
|
+
try {
|
|
151
|
+
const buffer = Buffer.alloc(maxBytes);
|
|
152
|
+
const read = readSync(fd, buffer, 0, maxBytes, 0);
|
|
153
|
+
return buffer.subarray(0, read);
|
|
154
|
+
} finally {
|
|
155
|
+
closeSync(fd);
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// src/commands/data-integration/local-data/time.ts
|
|
160
|
+
var TIME_FORMATS = [
|
|
161
|
+
// Standard AE format
|
|
162
|
+
"yyyy-MM-dd HH:mm:ss.SSS",
|
|
163
|
+
"yyyy-MM-dd HH:mm:ss",
|
|
164
|
+
// ISO 8601
|
|
165
|
+
"yyyy-MM-ddTHH:mm:ss.SSS",
|
|
166
|
+
"yyyy-MM-ddTHH:mm:ss",
|
|
167
|
+
"yyyy-MM-ddTHH:mm:ssXXX",
|
|
168
|
+
"yyyy-MM-ddTHH:mm:ss.SSSXXX",
|
|
169
|
+
// Slash separated
|
|
170
|
+
"yyyy/MM/dd HH:mm:ss",
|
|
171
|
+
"yyyy/MM/dd",
|
|
172
|
+
// Dot separated
|
|
173
|
+
"yyyy.MM.dd HH:mm:ss",
|
|
174
|
+
"yyyy.MM.dd",
|
|
175
|
+
// Compact digits
|
|
176
|
+
"yyyyMMddHHmmss",
|
|
177
|
+
"yyyyMMddHHmm",
|
|
178
|
+
"yyyyMMdd",
|
|
179
|
+
// English month names
|
|
180
|
+
"dd MMM yyyy HH:mm:ss",
|
|
181
|
+
"MMMM dd yyyy HH:mm:ss",
|
|
182
|
+
"MMM dd yyyy HH:mm:ss",
|
|
183
|
+
// Chinese date
|
|
184
|
+
"yyyy\u5E74M\u6708d\u65E5",
|
|
185
|
+
"yyyy\u5E74M\u6708d\u65E5 HH:mm:ss",
|
|
186
|
+
"yyyy\u5E74M\u6708d\u65E5 HH\u65F6mm\u5206ss\u79D2",
|
|
187
|
+
// Date only (require an explicit time_format)
|
|
188
|
+
"MM/dd/yyyy HH:mm:ss",
|
|
189
|
+
"dd/MM/yyyy HH:mm:ss"
|
|
190
|
+
];
|
|
191
|
+
var MONTH_NAMES = [
|
|
192
|
+
"january",
|
|
193
|
+
"february",
|
|
194
|
+
"march",
|
|
195
|
+
"april",
|
|
196
|
+
"may",
|
|
197
|
+
"june",
|
|
198
|
+
"july",
|
|
199
|
+
"august",
|
|
200
|
+
"september",
|
|
201
|
+
"october",
|
|
202
|
+
"november",
|
|
203
|
+
"december"
|
|
204
|
+
];
|
|
205
|
+
var MONTH_ABBR = [
|
|
206
|
+
"jan",
|
|
207
|
+
"feb",
|
|
208
|
+
"mar",
|
|
209
|
+
"apr",
|
|
210
|
+
"may",
|
|
211
|
+
"jun",
|
|
212
|
+
"jul",
|
|
213
|
+
"aug",
|
|
214
|
+
"sep",
|
|
215
|
+
"oct",
|
|
216
|
+
"nov",
|
|
217
|
+
"dec"
|
|
218
|
+
];
|
|
219
|
+
function tryStrptime(value, format) {
|
|
220
|
+
return tryStrptimeTokens(value.trim(), tokenizeFormat(format));
|
|
221
|
+
}
|
|
222
|
+
function isParseableByAnyFormat(value) {
|
|
223
|
+
if (typeof value !== "string") return false;
|
|
224
|
+
return findFirstTimeFormat(value) !== null;
|
|
225
|
+
}
|
|
226
|
+
var TIME_PARSE_CACHE_MAX = 1e5;
|
|
227
|
+
var timeParseCache = /* @__PURE__ */ new Map();
|
|
228
|
+
var tokenizedFormats = null;
|
|
229
|
+
function getTokenizedFormats() {
|
|
230
|
+
if (!tokenizedFormats) tokenizedFormats = TIME_FORMATS.map(tokenizeFormat);
|
|
231
|
+
return tokenizedFormats;
|
|
232
|
+
}
|
|
233
|
+
function findFirstTimeFormat(value) {
|
|
234
|
+
const text = value.trim();
|
|
235
|
+
if (!text) return null;
|
|
236
|
+
const cached = timeParseCache.get(text);
|
|
237
|
+
if (cached !== void 0) return cached;
|
|
238
|
+
const formats = getTokenizedFormats();
|
|
239
|
+
let match = null;
|
|
240
|
+
for (let index = 0; index < formats.length; index += 1) {
|
|
241
|
+
if (tryStrptimeTokens(text, formats[index])) {
|
|
242
|
+
match = TIME_FORMATS[index];
|
|
243
|
+
break;
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
if (timeParseCache.size < TIME_PARSE_CACHE_MAX) timeParseCache.set(text, match);
|
|
247
|
+
return match;
|
|
248
|
+
}
|
|
249
|
+
function parseTimeByAnyFormat(value) {
|
|
250
|
+
const text = value.trim();
|
|
251
|
+
if (!text) return null;
|
|
252
|
+
for (const tokens of getTokenizedFormats()) {
|
|
253
|
+
const parts = tryStrptimeTokens(text, tokens);
|
|
254
|
+
if (parts) return parts;
|
|
255
|
+
}
|
|
256
|
+
return null;
|
|
257
|
+
}
|
|
258
|
+
function tryStrptimeTokens(v, tokens) {
|
|
259
|
+
const dateParts = {};
|
|
260
|
+
let vi = 0;
|
|
261
|
+
for (const tok of tokens) {
|
|
262
|
+
if (tok.type === "literal") {
|
|
263
|
+
while (vi < v.length && /[\s\-T:./,年日月时分秒]/.test(v[vi])) vi += 1;
|
|
264
|
+
continue;
|
|
265
|
+
}
|
|
266
|
+
if (tok.type === "tz") {
|
|
267
|
+
const match2 = v.slice(vi).match(/^([+-]\d{2}):?(\d{2})/);
|
|
268
|
+
if (match2) vi += match2[0].length;
|
|
269
|
+
continue;
|
|
270
|
+
}
|
|
271
|
+
if (tok.type === "MMMM") {
|
|
272
|
+
const match2 = v.slice(vi).match(/^(January|February|March|April|May|June|July|August|September|October|November|December)/i);
|
|
273
|
+
if (!match2) return null;
|
|
274
|
+
dateParts.M = MONTH_NAMES.indexOf(match2[1].toLowerCase()) + 1;
|
|
275
|
+
vi += match2[0].length;
|
|
276
|
+
continue;
|
|
277
|
+
}
|
|
278
|
+
if (tok.type === "MMM") {
|
|
279
|
+
const match2 = v.slice(vi).match(/^(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)/i);
|
|
280
|
+
if (!match2) return null;
|
|
281
|
+
dateParts.M = MONTH_ABBR.indexOf(match2[1].toLowerCase()) + 1;
|
|
282
|
+
vi += match2[0].length;
|
|
283
|
+
continue;
|
|
284
|
+
}
|
|
285
|
+
const match = tok.type === "S" ? v.slice(vi).match(/^(\d{1,3})/) : v.slice(vi).match(tok.size === 1 ? /^(\d+)/ : new RegExp(`^(\\d{${tok.size}})`));
|
|
286
|
+
if (!match) return null;
|
|
287
|
+
const num = parseInt(match[1], 10);
|
|
288
|
+
switch (tok.type) {
|
|
289
|
+
case "y":
|
|
290
|
+
dateParts.y = tok.size === 2 ? 2e3 + num : num;
|
|
291
|
+
break;
|
|
292
|
+
case "M":
|
|
293
|
+
if (!dateParts.M) dateParts.M = num;
|
|
294
|
+
break;
|
|
295
|
+
case "d":
|
|
296
|
+
dateParts.d = num;
|
|
297
|
+
break;
|
|
298
|
+
case "H":
|
|
299
|
+
case "h":
|
|
300
|
+
dateParts.H = num;
|
|
301
|
+
break;
|
|
302
|
+
case "m":
|
|
303
|
+
dateParts.m = num;
|
|
304
|
+
break;
|
|
305
|
+
case "s":
|
|
306
|
+
dateParts.s = num;
|
|
307
|
+
break;
|
|
308
|
+
case "S":
|
|
309
|
+
dateParts.S = num * Math.pow(10, 3 - match[1].length);
|
|
310
|
+
break;
|
|
311
|
+
default:
|
|
312
|
+
break;
|
|
313
|
+
}
|
|
314
|
+
vi += match[0].length;
|
|
315
|
+
}
|
|
316
|
+
const parts = {
|
|
317
|
+
year: dateParts.y ?? 2e3,
|
|
318
|
+
month: dateParts.M ?? 1,
|
|
319
|
+
day: dateParts.d ?? 1,
|
|
320
|
+
hour: dateParts.H ?? 0,
|
|
321
|
+
minute: dateParts.m ?? 0,
|
|
322
|
+
second: dateParts.s ?? 0,
|
|
323
|
+
millisecond: dateParts.S ?? 0
|
|
324
|
+
};
|
|
325
|
+
return isValidWallTimeParts(parts) ? parts : null;
|
|
326
|
+
}
|
|
327
|
+
function isValidWallTimeParts(parts) {
|
|
328
|
+
if (parts.hour < 0 || parts.hour > 23 || parts.minute < 0 || parts.minute > 59 || parts.second < 0 || parts.second > 59 || parts.millisecond < 0 || parts.millisecond > 999) {
|
|
329
|
+
return false;
|
|
330
|
+
}
|
|
331
|
+
const date = new Date(Date.UTC(parts.year, parts.month - 1, parts.day));
|
|
332
|
+
return date.getUTCFullYear() === parts.year && date.getUTCMonth() === parts.month - 1 && date.getUTCDate() === parts.day;
|
|
333
|
+
}
|
|
334
|
+
function tokenizeFormat(format) {
|
|
335
|
+
const tokens = [];
|
|
336
|
+
let i = 0;
|
|
337
|
+
while (i < format.length) {
|
|
338
|
+
const ch = format[i];
|
|
339
|
+
if (ch === "y") {
|
|
340
|
+
let n = 0;
|
|
341
|
+
while (i + n < format.length && format[i + n] === "y") n += 1;
|
|
342
|
+
tokens.push({ type: "y", size: n });
|
|
343
|
+
i += n;
|
|
344
|
+
} else if (ch === "M") {
|
|
345
|
+
if (format.slice(i, i + 4) === "MMMM") {
|
|
346
|
+
tokens.push({ type: "MMMM", size: 0 });
|
|
347
|
+
i += 4;
|
|
348
|
+
} else if (format.slice(i, i + 3) === "MMM") {
|
|
349
|
+
tokens.push({ type: "MMM", size: 0 });
|
|
350
|
+
i += 3;
|
|
351
|
+
} else {
|
|
352
|
+
let n = 0;
|
|
353
|
+
while (i + n < format.length && format[i + n] === "M") n += 1;
|
|
354
|
+
tokens.push({ type: "M", size: n });
|
|
355
|
+
i += n;
|
|
356
|
+
}
|
|
357
|
+
} else if (ch === "d") {
|
|
358
|
+
let n = 0;
|
|
359
|
+
while (i + n < format.length && format[i + n] === "d") n += 1;
|
|
360
|
+
tokens.push({ type: "d", size: n });
|
|
361
|
+
i += n;
|
|
362
|
+
} else if (ch === "H") {
|
|
363
|
+
let n = 0;
|
|
364
|
+
while (i + n < format.length && format[i + n] === "H") n += 1;
|
|
365
|
+
tokens.push({ type: "H", size: n });
|
|
366
|
+
i += n;
|
|
367
|
+
} else if (ch === "h") {
|
|
368
|
+
let n = 0;
|
|
369
|
+
while (i + n < format.length && format[i + n] === "h") n += 1;
|
|
370
|
+
tokens.push({ type: "h", size: n });
|
|
371
|
+
i += n;
|
|
372
|
+
} else if (ch === "m") {
|
|
373
|
+
let n = 0;
|
|
374
|
+
while (i + n < format.length && format[i + n] === "m") n += 1;
|
|
375
|
+
tokens.push({ type: "m", size: n });
|
|
376
|
+
i += n;
|
|
377
|
+
} else if (ch === "s") {
|
|
378
|
+
let n = 0;
|
|
379
|
+
while (i + n < format.length && format[i + n] === "s") n += 1;
|
|
380
|
+
tokens.push({ type: "s", size: n });
|
|
381
|
+
i += n;
|
|
382
|
+
} else if (ch === "S") {
|
|
383
|
+
let n = 0;
|
|
384
|
+
while (i + n < format.length && format[i + n] === "S") n += 1;
|
|
385
|
+
tokens.push({ type: "S", size: n });
|
|
386
|
+
i += n;
|
|
387
|
+
} else if (ch === "T") {
|
|
388
|
+
tokens.push({ type: "literal", size: 0 });
|
|
389
|
+
i += 1;
|
|
390
|
+
} else if (ch === "X") {
|
|
391
|
+
let n = 0;
|
|
392
|
+
while (i + n < format.length && format[i + n] === "X") n += 1;
|
|
393
|
+
tokens.push({ type: "tz", size: n });
|
|
394
|
+
i += n;
|
|
395
|
+
} else {
|
|
396
|
+
tokens.push({ type: "literal", size: 0 });
|
|
397
|
+
i += 1;
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
return tokens;
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
// src/commands/data-integration/local-data/flatten.ts
|
|
404
|
+
var NDJSON_MAX_DEPTH = 1;
|
|
405
|
+
var NESTED_NODE_SAMPLE_LIMIT = 5;
|
|
406
|
+
var NESTED_SAMPLE_TRUNCATE = 40;
|
|
407
|
+
function getNestedValue(obj, path) {
|
|
408
|
+
let current = obj;
|
|
409
|
+
for (const segment of path.split(".")) {
|
|
410
|
+
if (current == null || typeof current !== "object") return void 0;
|
|
411
|
+
current = current[segment];
|
|
412
|
+
}
|
|
413
|
+
return current;
|
|
414
|
+
}
|
|
415
|
+
function flattenJSON(obj, prefix = "", depth = 0, maxDepth = NDJSON_MAX_DEPTH) {
|
|
416
|
+
if (obj == null) {
|
|
417
|
+
if (!prefix) return {};
|
|
418
|
+
return { [prefix]: "" };
|
|
419
|
+
}
|
|
420
|
+
if (typeof obj !== "object") {
|
|
421
|
+
return { [prefix]: String(obj) };
|
|
422
|
+
}
|
|
423
|
+
if (Array.isArray(obj)) {
|
|
424
|
+
return { [prefix]: JSON.stringify(obj) };
|
|
425
|
+
}
|
|
426
|
+
if (depth >= maxDepth) {
|
|
427
|
+
return { [prefix]: JSON.stringify(obj) };
|
|
428
|
+
}
|
|
429
|
+
const result = {};
|
|
430
|
+
for (const [key, value] of Object.entries(obj)) {
|
|
431
|
+
const childKey = prefix ? `${prefix}.${key}` : key;
|
|
432
|
+
Object.assign(result, flattenJSON(value, childKey, depth + 1, maxDepth));
|
|
433
|
+
}
|
|
434
|
+
return result;
|
|
435
|
+
}
|
|
436
|
+
function buildRowWithFlatten(obj, flattenRules) {
|
|
437
|
+
const row = {};
|
|
438
|
+
const coveredRoots = new Set(Object.values(flattenRules).map((path) => path.split(".")[0]));
|
|
439
|
+
const base = flattenJSON(obj);
|
|
440
|
+
for (const [key, value] of Object.entries(base)) {
|
|
441
|
+
if (!coveredRoots.has(key)) row[key] = value;
|
|
442
|
+
}
|
|
443
|
+
for (const [outColumn, sourcePath] of Object.entries(flattenRules)) {
|
|
444
|
+
const value = getNestedValue(obj, sourcePath);
|
|
445
|
+
row[outColumn] = value == null ? "" : typeof value === "object" ? JSON.stringify(value) : String(value);
|
|
446
|
+
}
|
|
447
|
+
return row;
|
|
448
|
+
}
|
|
449
|
+
function flattenLocalDataRow(value, flattenRules) {
|
|
450
|
+
if (value !== null && typeof value === "object" && !Array.isArray(value)) {
|
|
451
|
+
if (flattenRules && Object.keys(flattenRules).length > 0) {
|
|
452
|
+
return buildRowWithFlatten(value, flattenRules);
|
|
453
|
+
}
|
|
454
|
+
return value;
|
|
455
|
+
}
|
|
456
|
+
return { value };
|
|
457
|
+
}
|
|
458
|
+
function flattenDelimitedRow(row, flattenRules) {
|
|
459
|
+
const result = { ...row };
|
|
460
|
+
for (const [outColumn, path] of Object.entries(flattenRules)) {
|
|
461
|
+
const dot = path.indexOf(".");
|
|
462
|
+
if (dot <= 0) continue;
|
|
463
|
+
const column = path.slice(0, dot);
|
|
464
|
+
const cell = row[column];
|
|
465
|
+
if (typeof cell !== "string") continue;
|
|
466
|
+
let parsed;
|
|
467
|
+
try {
|
|
468
|
+
parsed = JSON.parse(cell);
|
|
469
|
+
} catch {
|
|
470
|
+
continue;
|
|
471
|
+
}
|
|
472
|
+
const value = getNestedValue(parsed, path.slice(dot + 1));
|
|
473
|
+
if (value === null || value === void 0) continue;
|
|
474
|
+
result[outColumn] = typeof value === "object" ? JSON.stringify(value) : String(value);
|
|
475
|
+
}
|
|
476
|
+
return result;
|
|
477
|
+
}
|
|
478
|
+
function buildNestedTree(rows) {
|
|
479
|
+
const objects = rows.filter((row) => row !== null && typeof row === "object" && !Array.isArray(row));
|
|
480
|
+
return buildObjectChildren(objects, "");
|
|
481
|
+
}
|
|
482
|
+
function buildObjectChildren(objects, parentPath) {
|
|
483
|
+
const keys = [];
|
|
484
|
+
const seen = /* @__PURE__ */ new Set();
|
|
485
|
+
for (const object of objects) {
|
|
486
|
+
for (const key of Object.keys(object)) {
|
|
487
|
+
if (!seen.has(key)) {
|
|
488
|
+
seen.add(key);
|
|
489
|
+
keys.push(key);
|
|
490
|
+
}
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
return keys.map((key) => {
|
|
494
|
+
const path = parentPath ? `${parentPath}.${key}` : key;
|
|
495
|
+
const values = objects.map((object) => object[key]).filter((value) => value !== void 0);
|
|
496
|
+
const nonEmpty = values.some((value) => !isEmpty(value));
|
|
497
|
+
return buildNode(path, key, values, nonEmpty);
|
|
498
|
+
});
|
|
499
|
+
}
|
|
500
|
+
function buildNode(path, name, values, nonEmpty) {
|
|
501
|
+
const arrays = values.filter(Array.isArray);
|
|
502
|
+
const objects = values.filter((value) => value !== null && typeof value === "object" && !Array.isArray(value));
|
|
503
|
+
const structural = arrays.length + objects.length;
|
|
504
|
+
if (structural > 0 && structural >= values.length - structural) {
|
|
505
|
+
if (arrays.length > objects.length) {
|
|
506
|
+
const elementValues = arrays.flat();
|
|
507
|
+
const elementKind = elementValues.some((value) => value !== null && typeof value === "object") ? "object" : "primitive";
|
|
508
|
+
return { path, name, kind: "array", elementKind, nonEmpty };
|
|
509
|
+
}
|
|
510
|
+
return { path, name, kind: "object", children: buildObjectChildren(objects, path), nonEmpty };
|
|
511
|
+
}
|
|
512
|
+
return { path, name, kind: "primitive", ...inferPrimitive(values), nonEmpty };
|
|
513
|
+
}
|
|
514
|
+
function inferPrimitive(values) {
|
|
515
|
+
const nonNull = values.filter((value) => value !== null && value !== void 0 && value !== "");
|
|
516
|
+
const samples = boundedSamples(nonNull);
|
|
517
|
+
if (nonNull.length === 0) return { samples };
|
|
518
|
+
const typeCounts = /* @__PURE__ */ new Map();
|
|
519
|
+
const kindSet = /* @__PURE__ */ new Set();
|
|
520
|
+
for (const value of nonNull) {
|
|
521
|
+
typeCounts.set(classifyPrimitive(value), (typeCounts.get(classifyPrimitive(value)) ?? 0) + 1);
|
|
522
|
+
kindSet.add(rawValueKind(value));
|
|
523
|
+
}
|
|
524
|
+
let inferredType = "string";
|
|
525
|
+
let best = -1;
|
|
526
|
+
for (const [type, count] of typeCounts) {
|
|
527
|
+
if (count > best) {
|
|
528
|
+
best = count;
|
|
529
|
+
inferredType = type;
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
const valueKind = kindSet.size === 1 ? [...kindSet][0] : "mixed";
|
|
533
|
+
return { inferredType, valueKind, samples };
|
|
534
|
+
}
|
|
535
|
+
function classifyPrimitive(value) {
|
|
536
|
+
if (value instanceof Date) return "datetime";
|
|
537
|
+
if (typeof value === "boolean") return "boolean";
|
|
538
|
+
if (typeof value === "number") return "number";
|
|
539
|
+
if (typeof value === "string") {
|
|
540
|
+
const normalized = value.trim().toLowerCase();
|
|
541
|
+
if (normalized === "true" || normalized === "false") return "boolean";
|
|
542
|
+
if (isStrongDateTime(value) || isParseableByAnyFormat(value)) return "datetime";
|
|
543
|
+
if (/^[+-]?(?:\d+\.?\d*|\.\d+)(?:e[+-]?\d+)?$/i.test(normalized)) return "number";
|
|
544
|
+
}
|
|
545
|
+
return "string";
|
|
546
|
+
}
|
|
547
|
+
function rawValueKind(value) {
|
|
548
|
+
if (typeof value === "boolean") return "boolean";
|
|
549
|
+
if (typeof value === "number") return "number";
|
|
550
|
+
return "string";
|
|
551
|
+
}
|
|
552
|
+
function boundedSamples(values) {
|
|
553
|
+
const samples = [];
|
|
554
|
+
const seen = /* @__PURE__ */ new Set();
|
|
555
|
+
for (const value of values) {
|
|
556
|
+
if (samples.length >= NESTED_NODE_SAMPLE_LIMIT) break;
|
|
557
|
+
const text = truncateSample(value);
|
|
558
|
+
if (seen.has(text)) continue;
|
|
559
|
+
seen.add(text);
|
|
560
|
+
samples.push(text);
|
|
561
|
+
}
|
|
562
|
+
return samples;
|
|
563
|
+
}
|
|
564
|
+
function truncateSample(value) {
|
|
565
|
+
const text = value instanceof Date ? value.toISOString() : String(value);
|
|
566
|
+
if (text.length <= NESTED_SAMPLE_TRUNCATE) return text;
|
|
567
|
+
return `${Array.from(text).slice(0, NESTED_SAMPLE_TRUNCATE).join("")}\u2026`;
|
|
568
|
+
}
|
|
569
|
+
function isEmpty(value) {
|
|
570
|
+
return value === null || value === void 0 || typeof value === "string" && value.trim() === "" || Array.isArray(value) && value.length === 0;
|
|
571
|
+
}
|
|
572
|
+
function isStrongDateTime(value) {
|
|
573
|
+
return /^\d{4}[-/]\d{1,2}[-/]\d{1,2}(?:[ T]\d{1,2}:\d{2}(?::\d{2}(?:\.\d{1,3})?)?(?:Z|[+-]\d{2}:?\d{2})?)?$/.test(value.trim());
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
// src/commands/data-integration/local-data/input.ts
|
|
577
|
+
var XLSX = XLSXMod.default ?? XLSXMod;
|
|
578
|
+
var require3 = createRequire2(import.meta.url);
|
|
579
|
+
var ExcelWorksheetReader = require3("exceljs/lib/stream/xlsx/worksheet-reader");
|
|
580
|
+
function resolveLocalDataInputMeta(filePath) {
|
|
581
|
+
let format = resolveFormat(filePath);
|
|
582
|
+
let delimiter;
|
|
583
|
+
let encoding;
|
|
584
|
+
if (format === "txt") {
|
|
585
|
+
encoding = detectEncoding(filePath);
|
|
586
|
+
const sniffed = sniffFormat(filePath, encoding);
|
|
587
|
+
format = sniffed.format;
|
|
588
|
+
delimiter = sniffed.delimiter;
|
|
589
|
+
}
|
|
590
|
+
let sizeBytes;
|
|
591
|
+
try {
|
|
592
|
+
const stat = statSync(filePath);
|
|
593
|
+
if (!stat.isFile()) throw new Error("not a file");
|
|
594
|
+
sizeBytes = stat.size;
|
|
595
|
+
} catch {
|
|
596
|
+
throw new CliValidationError("The local data input must be a readable file.", {
|
|
597
|
+
code: "LOCAL_DATA_INPUT_NOT_FOUND",
|
|
598
|
+
location: { field: "input-file" }
|
|
599
|
+
});
|
|
600
|
+
}
|
|
601
|
+
if (format !== "xls" && format !== "xlsx" && !encoding) {
|
|
602
|
+
encoding = detectEncoding(filePath);
|
|
603
|
+
}
|
|
604
|
+
return {
|
|
605
|
+
filePath,
|
|
606
|
+
format,
|
|
607
|
+
sizeBytes,
|
|
608
|
+
...delimiter ? { delimiter } : {},
|
|
609
|
+
...encoding ? { encoding } : {}
|
|
610
|
+
};
|
|
611
|
+
}
|
|
612
|
+
async function inspectLocalDataInput(filePath) {
|
|
613
|
+
const meta = resolveLocalDataInputMeta(filePath);
|
|
614
|
+
emitSizeWarning(meta.filePath, meta.format, meta.sizeBytes, meta.encoding);
|
|
615
|
+
try {
|
|
616
|
+
return {
|
|
617
|
+
...meta,
|
|
618
|
+
sha256: await sha256File(filePath),
|
|
619
|
+
dataSets: await discoverDataSets(filePath, meta.format)
|
|
620
|
+
};
|
|
621
|
+
} catch (error) {
|
|
622
|
+
if (error instanceof CliValidationError) throw error;
|
|
623
|
+
throw localDataParseError(meta.format);
|
|
624
|
+
}
|
|
625
|
+
}
|
|
626
|
+
function emitSizeWarning(filePath, format, sizeBytes, encoding) {
|
|
627
|
+
const assessment = assessFileSize(basename(filePath), format, sizeBytes, encoding);
|
|
628
|
+
if (assessment.rejected) {
|
|
629
|
+
throw new CliValidationError("XLS input exceeds the supported file size limit.", {
|
|
630
|
+
code: "LOCAL_DATA_FILE_TOO_LARGE",
|
|
631
|
+
hint: "XLS files larger than 1 GB are not supported; convert the workbook to XLSX or split it first.",
|
|
632
|
+
location: { field: "input-file" }
|
|
633
|
+
});
|
|
634
|
+
}
|
|
635
|
+
if (assessment.warning) {
|
|
636
|
+
process.stderr.write(`${assessment.warning}
|
|
637
|
+
`);
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
function selectDataSet(input, requested) {
|
|
641
|
+
if (requested) {
|
|
642
|
+
const selected = input.dataSets.find((candidate) => candidate.id === requested || candidate.selector === requested || candidate.label === requested);
|
|
643
|
+
if (!selected) {
|
|
644
|
+
throw new CliValidationError("The requested Sheet or JSON Path was not found.", {
|
|
645
|
+
code: "LOCAL_DATA_SET_NOT_FOUND",
|
|
646
|
+
hint: `Choose one of: ${input.dataSets.map((item) => item.id).join(", ")}`,
|
|
647
|
+
location: { field: "data-set" }
|
|
648
|
+
});
|
|
649
|
+
}
|
|
650
|
+
return selected;
|
|
651
|
+
}
|
|
652
|
+
if (input.dataSets.length !== 1) {
|
|
653
|
+
throw new CliValidationError("This file contains multiple data sets.", {
|
|
654
|
+
code: "LOCAL_DATA_SET_REQUIRED",
|
|
655
|
+
hint: `Pass --data-set with one of: ${input.dataSets.map((item) => item.id).join(", ")}`,
|
|
656
|
+
location: { field: "data-set" }
|
|
657
|
+
});
|
|
658
|
+
}
|
|
659
|
+
return input.dataSets[0];
|
|
660
|
+
}
|
|
661
|
+
async function streamLocalDataRows(input, dataSet, onRow, options = {}) {
|
|
662
|
+
const opts = {
|
|
663
|
+
delimiter: options.delimiter ?? input.delimiter ?? (input.format === "tsv" ? " " : ","),
|
|
664
|
+
encoding: options.encoding ?? input.encoding ?? "utf-8",
|
|
665
|
+
headerNames: options.headerNames,
|
|
666
|
+
noHeader: options.noHeader,
|
|
667
|
+
flattenRules: options.flattenRules,
|
|
668
|
+
mergeSheets: options.mergeSheets
|
|
669
|
+
};
|
|
670
|
+
try {
|
|
671
|
+
switch (input.format) {
|
|
672
|
+
case "csv":
|
|
673
|
+
case "tsv":
|
|
674
|
+
return await streamDelimited(input.filePath, onRow, opts);
|
|
675
|
+
case "jsonl":
|
|
676
|
+
return await streamJsonLines(input.filePath, onRow, opts);
|
|
677
|
+
case "json":
|
|
678
|
+
return await streamJson(input.filePath, dataSet.selector ?? "$", onRow, opts);
|
|
679
|
+
case "xlsx":
|
|
680
|
+
return await streamXlsx(input.filePath, dataSet.label, onRow, opts);
|
|
681
|
+
case "xls":
|
|
682
|
+
return await streamXls(input.filePath, dataSet.label, onRow, opts);
|
|
683
|
+
}
|
|
684
|
+
} catch (error) {
|
|
685
|
+
if (error instanceof CliValidationError) throw error;
|
|
686
|
+
throw localDataParseError(input.format);
|
|
687
|
+
}
|
|
688
|
+
}
|
|
689
|
+
async function sha256File(filePath) {
|
|
690
|
+
const hash = createHash("sha256");
|
|
691
|
+
for await (const chunk of createReadStream2(filePath)) hash.update(chunk);
|
|
692
|
+
return hash.digest("hex");
|
|
693
|
+
}
|
|
694
|
+
function resolveFormat(filePath) {
|
|
695
|
+
const extension = extname(filePath).slice(1).toLowerCase();
|
|
696
|
+
if (extension === "csv") return "csv";
|
|
697
|
+
if (extension === "tsv" || extension === "tab") return "tsv";
|
|
698
|
+
if (extension === "json" || extension === "jsonl") return extension;
|
|
699
|
+
if (extension === "ndjson") return "jsonl";
|
|
700
|
+
if (extension === "xls") return "xls";
|
|
701
|
+
if (extension === "xlsx" || extension === "xlsm") return "xlsx";
|
|
702
|
+
return "txt";
|
|
703
|
+
}
|
|
704
|
+
function sniffDelimiter(filePath, encoding) {
|
|
705
|
+
const content = decodeFileSample(filePath, encoding ?? detectEncoding(filePath));
|
|
706
|
+
const lines = content.split(/\r?\n/).filter((line) => line.trim()).slice(0, 10);
|
|
707
|
+
if (lines.length === 0) return ",";
|
|
708
|
+
let best = { delimiter: ",", score: Number.NEGATIVE_INFINITY };
|
|
709
|
+
for (const char of [",", " "]) {
|
|
710
|
+
const regex = new RegExp(escapeRegex(char), "g");
|
|
711
|
+
const counts = lines.map((line) => {
|
|
712
|
+
const stripped = line.replace(/"[^"]*"/g, "").replace(/'[^']*'/g, "");
|
|
713
|
+
return (stripped.match(regex) || []).length;
|
|
714
|
+
});
|
|
715
|
+
const nonZero = counts.filter((count) => count > 0);
|
|
716
|
+
if (nonZero.length < lines.length * 0.5) continue;
|
|
717
|
+
const average = nonZero.reduce((sum, count) => sum + count, 0) / nonZero.length;
|
|
718
|
+
if (average < 1) continue;
|
|
719
|
+
const variance = nonZero.reduce((sum, count) => sum + (count - average) ** 2, 0) / nonZero.length;
|
|
720
|
+
const score = nonZero.length - variance;
|
|
721
|
+
if (score > best.score) best = { delimiter: char, score };
|
|
722
|
+
}
|
|
723
|
+
return best.delimiter;
|
|
724
|
+
}
|
|
725
|
+
function sniffFormat(filePath, encoding) {
|
|
726
|
+
const content = decodeFileSample(filePath, encoding ?? detectEncoding(filePath));
|
|
727
|
+
const lines = content.split(/\r?\n/).filter((line) => line.trim()).slice(0, 10);
|
|
728
|
+
if (lines.length > 0) {
|
|
729
|
+
let jsonCount = 0;
|
|
730
|
+
for (const line of lines) {
|
|
731
|
+
try {
|
|
732
|
+
const parsed = JSON.parse(line);
|
|
733
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) jsonCount += 1;
|
|
734
|
+
} catch {
|
|
735
|
+
}
|
|
736
|
+
}
|
|
737
|
+
if (jsonCount >= lines.length * 0.8 && jsonCount > 0) return { format: "jsonl" };
|
|
738
|
+
}
|
|
739
|
+
const delimiter = sniffDelimiter(filePath, encoding);
|
|
740
|
+
return { format: delimiter === " " ? "tsv" : "csv", delimiter };
|
|
741
|
+
}
|
|
742
|
+
function detectHeaderRow(rows, numSample = 10) {
|
|
743
|
+
if (rows.length === 0) return { hasHeaders: true, confidence: "high", reason: "empty file" };
|
|
744
|
+
if (rows.length === 1) {
|
|
745
|
+
return { hasHeaders: true, confidence: "low", reason: "only one row; treating it as a header by default" };
|
|
746
|
+
}
|
|
747
|
+
const first = rows[0].map((value) => String(value).trim());
|
|
748
|
+
const columnCount = first.length;
|
|
749
|
+
if (columnCount === 0) return { hasHeaders: true, confidence: "high", reason: "no column data" };
|
|
750
|
+
const firstNumeric = first.filter(isNumeric).length;
|
|
751
|
+
const firstRatio = firstNumeric / columnCount;
|
|
752
|
+
const sample = rows.slice(1, numSample + 1);
|
|
753
|
+
if (new Set(first).size === first.length && firstNumeric === 0) {
|
|
754
|
+
return { hasHeaders: true, confidence: "high", reason: "all first-row values are unique and non-numeric, appears to be headers" };
|
|
755
|
+
}
|
|
756
|
+
const dataRatios = [];
|
|
757
|
+
for (const row of sample) {
|
|
758
|
+
const values = row.map((value) => String(value).trim());
|
|
759
|
+
if (values.length !== columnCount) continue;
|
|
760
|
+
dataRatios.push(values.filter(isNumeric).length / columnCount);
|
|
761
|
+
}
|
|
762
|
+
if (dataRatios.length === 0) {
|
|
763
|
+
return { hasHeaders: true, confidence: "low", reason: "cannot sample enough data rows; treating the first row as a header by default" };
|
|
764
|
+
}
|
|
765
|
+
const averageRatio = dataRatios.reduce((sum, ratio2) => sum + ratio2, 0) / dataRatios.length;
|
|
766
|
+
if (Math.abs(firstRatio - averageRatio) > 0.5) {
|
|
767
|
+
return {
|
|
768
|
+
hasHeaders: true,
|
|
769
|
+
confidence: "medium",
|
|
770
|
+
reason: `first row pattern differs significantly from data rows (numeric ratio ${Math.round(firstRatio * 100)}% vs ${Math.round(averageRatio * 100)}%), appears to be headers`
|
|
771
|
+
};
|
|
772
|
+
}
|
|
773
|
+
return {
|
|
774
|
+
hasHeaders: false,
|
|
775
|
+
confidence: "medium",
|
|
776
|
+
reason: `first row pattern matches data rows (numeric ratio ${Math.round(firstRatio * 100)}% vs ${Math.round(averageRatio * 100)}%), appears to lack headers`
|
|
777
|
+
};
|
|
778
|
+
}
|
|
779
|
+
function peekDelimitedRecords(filePath, options = {}) {
|
|
780
|
+
const delimiter = options.delimiter ?? ",";
|
|
781
|
+
const encoding = options.encoding ?? detectEncoding(filePath);
|
|
782
|
+
const content = decodeFileSample(filePath, encoding, 256 * 1024);
|
|
783
|
+
const records = parseCsvSync(content, {
|
|
784
|
+
delimiter,
|
|
785
|
+
quote: delimiter === " " ? null : '"',
|
|
786
|
+
relax_column_count: true,
|
|
787
|
+
skip_empty_lines: true,
|
|
788
|
+
trim: true
|
|
789
|
+
});
|
|
790
|
+
return records.slice(0, options.limit ?? 10);
|
|
791
|
+
}
|
|
792
|
+
function escapeRegex(source) {
|
|
793
|
+
return source.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
794
|
+
}
|
|
795
|
+
function isNumeric(value) {
|
|
796
|
+
const trimmed = value.trim();
|
|
797
|
+
if (!trimmed) return false;
|
|
798
|
+
return !Number.isNaN(Number(trimmed));
|
|
799
|
+
}
|
|
800
|
+
async function discoverDataSets(filePath, format) {
|
|
801
|
+
if (format === "csv" || format === "tsv" || format === "jsonl") {
|
|
802
|
+
return [{ id: "$", kind: "file", label: basename(filePath), selector: "$" }];
|
|
803
|
+
}
|
|
804
|
+
if (format === "xls") {
|
|
805
|
+
const workbook = XLSX.readFile(filePath, { dense: true });
|
|
806
|
+
return workbook.SheetNames.map((name) => ({ id: `sheet:${name}`, kind: "sheet", label: name, selector: name }));
|
|
807
|
+
}
|
|
808
|
+
if (format === "xlsx") {
|
|
809
|
+
return (await readXlsxSheetDefinitions(filePath)).map((sheet) => ({
|
|
810
|
+
id: `sheet:${sheet.name}`,
|
|
811
|
+
kind: "sheet",
|
|
812
|
+
label: sheet.name,
|
|
813
|
+
selector: sheet.name
|
|
814
|
+
}));
|
|
815
|
+
}
|
|
816
|
+
return discoverJsonDataSets(filePath);
|
|
817
|
+
}
|
|
818
|
+
async function discoverJsonDataSets(filePath) {
|
|
819
|
+
const encoding = detectEncoding(filePath);
|
|
820
|
+
const first = await firstNonWhitespaceCharacter(filePath, encoding);
|
|
821
|
+
if (first === "[") return [{ id: "$", kind: "json-path", label: "$", selector: "$" }];
|
|
822
|
+
if (first !== "{") {
|
|
823
|
+
throw new CliValidationError("JSON input must contain an object or array.", {
|
|
824
|
+
code: "LOCAL_DATA_JSON_ROOT_INVALID",
|
|
825
|
+
location: { field: "input-file" }
|
|
826
|
+
});
|
|
827
|
+
}
|
|
828
|
+
const candidates = /* @__PURE__ */ new Map();
|
|
829
|
+
const stack = [];
|
|
830
|
+
const tokenParser = createJsonParser();
|
|
831
|
+
tokenParser.on("data", (token) => {
|
|
832
|
+
const parent = stack.at(-1);
|
|
833
|
+
if (token.name === "keyValue" && parent?.type === "object") {
|
|
834
|
+
parent.pendingKey = String(token.value ?? "");
|
|
835
|
+
} else if (token.name === "startObject") {
|
|
836
|
+
stack.push({ type: "object", path: childPath(parent) });
|
|
837
|
+
} else if (token.name === "endObject") {
|
|
838
|
+
stack.pop();
|
|
839
|
+
} else if (token.name === "startArray") {
|
|
840
|
+
const path = childPath(parent);
|
|
841
|
+
if (path.length > 0) {
|
|
842
|
+
const selector = path.join(".");
|
|
843
|
+
const label = `$.${selector}`;
|
|
844
|
+
candidates.set(selector, { id: `json-path:${label}`, kind: "json-path", label, selector });
|
|
845
|
+
}
|
|
846
|
+
stack.push({ type: "array", path });
|
|
847
|
+
} else if (token.name === "endArray") {
|
|
848
|
+
stack.pop();
|
|
849
|
+
}
|
|
850
|
+
});
|
|
851
|
+
await pipeline(decodeTextStream(filePath, encoding), tokenParser);
|
|
852
|
+
return candidates.size > 0 ? [...candidates.values()] : [{ id: "$", kind: "json-path", label: "$", selector: "$object" }];
|
|
853
|
+
}
|
|
854
|
+
function childPath(parent) {
|
|
855
|
+
if (!parent) return [];
|
|
856
|
+
if (parent.type === "object" && parent.pendingKey) return [...parent.path, parent.pendingKey];
|
|
857
|
+
return [...parent.path];
|
|
858
|
+
}
|
|
859
|
+
async function streamDelimited(filePath, onRow, options) {
|
|
860
|
+
const delimiter = options.delimiter ?? ",";
|
|
861
|
+
const encoding = options.encoding ?? "utf-8";
|
|
862
|
+
let headerNames = options.headerNames;
|
|
863
|
+
if (!headerNames && options.noHeader) {
|
|
864
|
+
const firstRecord = peekDelimitedRecords(filePath, { delimiter, encoding, limit: 1 })[0] ?? [];
|
|
865
|
+
headerNames = firstRecord.map((_, index) => `col_${index + 1}`);
|
|
866
|
+
}
|
|
867
|
+
const parser = parseCsv({
|
|
868
|
+
bom: true,
|
|
869
|
+
delimiter,
|
|
870
|
+
quote: delimiter === " " ? null : '"',
|
|
871
|
+
relax_column_count: true,
|
|
872
|
+
skip_empty_lines: true,
|
|
873
|
+
trim: true,
|
|
874
|
+
columns: headerNames ? headerNames : (headers) => dedupeHeaders(headers.map((header) => String(header).trim()))
|
|
875
|
+
});
|
|
876
|
+
decodeTextStream(filePath, encoding).pipe(parser);
|
|
877
|
+
let count = 0;
|
|
878
|
+
for await (const value of parser) {
|
|
879
|
+
count += 1;
|
|
880
|
+
const row = normalizeDelimitedRow(value, headerNames);
|
|
881
|
+
await onRow(
|
|
882
|
+
options.flattenRules && Object.keys(options.flattenRules).length > 0 ? flattenDelimitedRow(row, options.flattenRules) : row,
|
|
883
|
+
count
|
|
884
|
+
);
|
|
885
|
+
}
|
|
886
|
+
return count;
|
|
887
|
+
}
|
|
888
|
+
function normalizeDelimitedRow(value, headerNames) {
|
|
889
|
+
if (!headerNames || value === null || typeof value !== "object" || Array.isArray(value)) {
|
|
890
|
+
return normalizeRow(value);
|
|
891
|
+
}
|
|
892
|
+
const source = value;
|
|
893
|
+
const row = {};
|
|
894
|
+
for (const header of headerNames) row[header] = source[header] ?? null;
|
|
895
|
+
return row;
|
|
896
|
+
}
|
|
897
|
+
async function streamJsonLines(filePath, onRow, options) {
|
|
898
|
+
let count = 0;
|
|
899
|
+
const lines = createInterface({ input: decodeTextStream(filePath, options.encoding ?? "utf-8"), crlfDelay: Infinity });
|
|
900
|
+
for await (const line of lines) {
|
|
901
|
+
if (!line.trim()) continue;
|
|
902
|
+
count += 1;
|
|
903
|
+
let value;
|
|
904
|
+
try {
|
|
905
|
+
value = JSON.parse(line);
|
|
906
|
+
} catch {
|
|
907
|
+
throw new CliValidationError("JSONL contains an invalid JSON record.", {
|
|
908
|
+
code: "LOCAL_DATA_JSONL_INVALID",
|
|
909
|
+
location: { record: count }
|
|
910
|
+
});
|
|
911
|
+
}
|
|
912
|
+
await onRow(flattenLocalDataRow(value, options.flattenRules), count);
|
|
913
|
+
}
|
|
914
|
+
return count;
|
|
915
|
+
}
|
|
916
|
+
async function streamJson(filePath, selector, onRow, options) {
|
|
917
|
+
let count = 0;
|
|
918
|
+
const parser = createJsonParser();
|
|
919
|
+
const streamer = selector === "$object" ? StreamValues.streamValues() : StreamArray.streamArray();
|
|
920
|
+
const source = decodeTextStream(filePath, options.encoding ?? "utf-8");
|
|
921
|
+
const chain = selector === "$" || selector === "$object" ? source.pipe(parser).pipe(streamer) : source.pipe(parser).pipe(Pick.pick({ filter: selector })).pipe(streamer);
|
|
922
|
+
for await (const item of chain) {
|
|
923
|
+
count += 1;
|
|
924
|
+
await onRow(flattenLocalDataRow(item.value, options.flattenRules), count);
|
|
925
|
+
}
|
|
926
|
+
return count;
|
|
927
|
+
}
|
|
928
|
+
async function streamXlsx(filePath, sheetName, onRow, options) {
|
|
929
|
+
const sheetDefinitions = await readXlsxSheetDefinitions(filePath);
|
|
930
|
+
const targets = options.mergeSheets ? sheetDefinitions : sheetDefinitions.filter((sheet) => sheet.name === sheetName);
|
|
931
|
+
if (targets.length === 0) {
|
|
932
|
+
throw new CliValidationError("The requested Excel Sheet was not found.", { code: "LOCAL_DATA_SET_NOT_FOUND" });
|
|
933
|
+
}
|
|
934
|
+
const archive = await unzipper.Open.file(filePath);
|
|
935
|
+
const sharedStringsEntry = archive.files.find((entry) => entry.path === "xl/sharedStrings.xml");
|
|
936
|
+
const sharedStrings = sharedStringsEntry ? await readXlsxSharedStrings(sharedStringsEntry) : [];
|
|
937
|
+
let count = 0;
|
|
938
|
+
for (const definition of targets) {
|
|
939
|
+
const worksheetEntry = archive.files.find((entry) => entry.path === definition.entryPath);
|
|
940
|
+
if (!worksheetEntry) {
|
|
941
|
+
throw new CliValidationError("The selected XLSX worksheet entry is missing.", {
|
|
942
|
+
code: "LOCAL_DATA_XLSX_INVALID",
|
|
943
|
+
location: { field: "input-file" }
|
|
944
|
+
});
|
|
945
|
+
}
|
|
946
|
+
const worksheet = new ExcelWorksheetReader({
|
|
947
|
+
workbook: {
|
|
948
|
+
sharedStrings,
|
|
949
|
+
styles: { getStyleModel: () => null },
|
|
950
|
+
properties: { model: {} }
|
|
951
|
+
},
|
|
952
|
+
id: definition.id,
|
|
953
|
+
iterator: worksheetEntry.stream(),
|
|
954
|
+
options: { worksheets: "emit", hyperlinks: "ignore" }
|
|
955
|
+
});
|
|
956
|
+
let headers = options.headerNames;
|
|
957
|
+
for await (const excelRow of worksheet) {
|
|
958
|
+
const values = Array.from({ length: Math.max(0, excelRow.cellCount) }, (_, index) => normalizeExcelValue(excelRow.getCell(index + 1).value));
|
|
959
|
+
if (options.noHeader && !headers) {
|
|
960
|
+
headers = values.map((_, index) => `col_${index + 1}`);
|
|
961
|
+
}
|
|
962
|
+
if (!headers) {
|
|
963
|
+
headers = dedupeHeaders(values.map((value, index) => String(value ?? `column_${index + 1}`).trim()));
|
|
964
|
+
continue;
|
|
965
|
+
}
|
|
966
|
+
if (values.every(isMissing)) continue;
|
|
967
|
+
count += 1;
|
|
968
|
+
await onRow(Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null])), count);
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
return count;
|
|
972
|
+
}
|
|
973
|
+
async function readXlsxSheetDefinitions(filePath) {
|
|
974
|
+
const archive = await unzipper.Open.file(filePath);
|
|
975
|
+
const workbookEntry = archive.files.find((entry) => entry.path === "xl/workbook.xml");
|
|
976
|
+
const relationshipsEntry = archive.files.find((entry) => entry.path === "xl/_rels/workbook.xml.rels");
|
|
977
|
+
if (!workbookEntry || !relationshipsEntry) {
|
|
978
|
+
throw new CliValidationError("XLSX workbook metadata is missing.", {
|
|
979
|
+
code: "LOCAL_DATA_XLSX_INVALID",
|
|
980
|
+
location: { field: "input-file" }
|
|
981
|
+
});
|
|
982
|
+
}
|
|
983
|
+
const [workbookXml, relationshipsXml] = await Promise.all([
|
|
984
|
+
workbookEntry.buffer().then((value) => value.toString("utf8")),
|
|
985
|
+
relationshipsEntry.buffer().then((value) => value.toString("utf8"))
|
|
986
|
+
]);
|
|
987
|
+
const targets = /* @__PURE__ */ new Map();
|
|
988
|
+
for (const match of relationshipsXml.matchAll(/<Relationship\b([^>]*)\/?\s*>/g)) {
|
|
989
|
+
const attributes = parseXmlAttributes(match[1]);
|
|
990
|
+
if (attributes.Id && attributes.Target) targets.set(attributes.Id, attributes.Target);
|
|
991
|
+
}
|
|
992
|
+
const sheets = [];
|
|
993
|
+
for (const match of workbookXml.matchAll(/<sheet\b([^>]*)\/?\s*>/g)) {
|
|
994
|
+
const attributes = parseXmlAttributes(match[1]);
|
|
995
|
+
const relationshipId = attributes["r:id"];
|
|
996
|
+
const target = relationshipId ? targets.get(relationshipId) : void 0;
|
|
997
|
+
const fileNumber = target?.match(/worksheets\/sheet(\d+)\.xml$/)?.[1];
|
|
998
|
+
if (!attributes.name || !relationshipId || !fileNumber) continue;
|
|
999
|
+
sheets.push({
|
|
1000
|
+
id: Number(attributes.sheetId ?? fileNumber),
|
|
1001
|
+
name: decodeXml(attributes.name),
|
|
1002
|
+
entryPath: normalizeXlsxEntryPath(target)
|
|
1003
|
+
});
|
|
1004
|
+
}
|
|
1005
|
+
if (sheets.length === 0) {
|
|
1006
|
+
throw new CliValidationError("XLSX workbook contains no readable Sheets.", {
|
|
1007
|
+
code: "LOCAL_DATA_XLSX_INVALID",
|
|
1008
|
+
location: { field: "input-file" }
|
|
1009
|
+
});
|
|
1010
|
+
}
|
|
1011
|
+
return sheets;
|
|
1012
|
+
}
|
|
1013
|
+
async function readExcelSheetHeaders(filePath, format) {
|
|
1014
|
+
return format === "xls" ? readXlsSheetHeaders(filePath) : readXlsxSheetHeaders(filePath);
|
|
1015
|
+
}
|
|
1016
|
+
function readXlsSheetHeaders(filePath) {
|
|
1017
|
+
const workbook = XLSX.readFile(filePath, { dense: true });
|
|
1018
|
+
return workbook.SheetNames.map((name) => {
|
|
1019
|
+
const sheet = workbook.Sheets[name];
|
|
1020
|
+
const rows = sheet ? XLSX.utils.sheet_to_json(sheet, { header: 1, raw: true, defval: null }) : [];
|
|
1021
|
+
return { name, headers: firstRowHeaders(rows[0] ?? []) };
|
|
1022
|
+
});
|
|
1023
|
+
}
|
|
1024
|
+
async function readXlsxSheetHeaders(filePath) {
|
|
1025
|
+
const sheetDefinitions = await readXlsxSheetDefinitions(filePath);
|
|
1026
|
+
const archive = await unzipper.Open.file(filePath);
|
|
1027
|
+
const sharedStringsEntry = archive.files.find((entry) => entry.path === "xl/sharedStrings.xml");
|
|
1028
|
+
const sharedStrings = sharedStringsEntry ? await readXlsxSharedStrings(sharedStringsEntry) : [];
|
|
1029
|
+
const sheets = [];
|
|
1030
|
+
for (const definition of sheetDefinitions) {
|
|
1031
|
+
const worksheetEntry = archive.files.find((entry) => entry.path === definition.entryPath);
|
|
1032
|
+
if (!worksheetEntry) continue;
|
|
1033
|
+
const worksheet = new ExcelWorksheetReader({
|
|
1034
|
+
workbook: {
|
|
1035
|
+
sharedStrings,
|
|
1036
|
+
styles: { getStyleModel: () => null },
|
|
1037
|
+
properties: { model: {} }
|
|
1038
|
+
},
|
|
1039
|
+
id: definition.id,
|
|
1040
|
+
iterator: worksheetEntry.stream(),
|
|
1041
|
+
options: { worksheets: "emit", hyperlinks: "ignore" }
|
|
1042
|
+
});
|
|
1043
|
+
let headers = [];
|
|
1044
|
+
for await (const excelRow of worksheet) {
|
|
1045
|
+
const values = Array.from({ length: Math.max(0, excelRow.cellCount) }, (_, index) => normalizeExcelValue(excelRow.getCell(index + 1).value));
|
|
1046
|
+
headers = firstRowHeaders(values);
|
|
1047
|
+
break;
|
|
1048
|
+
}
|
|
1049
|
+
sheets.push({ name: definition.name, headers });
|
|
1050
|
+
}
|
|
1051
|
+
return sheets;
|
|
1052
|
+
}
|
|
1053
|
+
function firstRowHeaders(values) {
|
|
1054
|
+
return dedupeHeaders(values.map((value, index) => String(value ?? `column_${index + 1}`).trim()));
|
|
1055
|
+
}
|
|
1056
|
+
async function readXlsxSharedStrings(entry) {
|
|
1057
|
+
const values = [];
|
|
1058
|
+
let inItem = false;
|
|
1059
|
+
let current = "";
|
|
1060
|
+
const parser = new SaxesParser();
|
|
1061
|
+
parser.on("opentag", (tag) => {
|
|
1062
|
+
if (tag.name === "si") {
|
|
1063
|
+
inItem = true;
|
|
1064
|
+
current = "";
|
|
1065
|
+
}
|
|
1066
|
+
});
|
|
1067
|
+
parser.on("text", (text) => {
|
|
1068
|
+
if (inItem) current += text;
|
|
1069
|
+
});
|
|
1070
|
+
parser.on("closetag", (tag) => {
|
|
1071
|
+
if (tag.name === "si") {
|
|
1072
|
+
values.push(current);
|
|
1073
|
+
inItem = false;
|
|
1074
|
+
current = "";
|
|
1075
|
+
}
|
|
1076
|
+
});
|
|
1077
|
+
for await (const chunk of entry.stream()) {
|
|
1078
|
+
parser.write(Buffer.from(chunk).toString("utf8"));
|
|
1079
|
+
}
|
|
1080
|
+
parser.close();
|
|
1081
|
+
return values;
|
|
1082
|
+
}
|
|
1083
|
+
function normalizeXlsxEntryPath(target) {
|
|
1084
|
+
const normalized = target.replace(/^\//, "");
|
|
1085
|
+
return normalized.startsWith("xl/") ? normalized : `xl/${normalized.replace(/^\.\//, "")}`;
|
|
1086
|
+
}
|
|
1087
|
+
function parseXmlAttributes(source) {
|
|
1088
|
+
const attributes = {};
|
|
1089
|
+
for (const match of source.matchAll(/([\w:-]+)="([^"]*)"/g)) attributes[match[1]] = match[2];
|
|
1090
|
+
return attributes;
|
|
1091
|
+
}
|
|
1092
|
+
function decodeXml(source) {
|
|
1093
|
+
return source.replace(/"/g, '"').replace(/'/g, "'").replace(/</g, "<").replace(/>/g, ">").replace(/&/g, "&");
|
|
1094
|
+
}
|
|
1095
|
+
async function streamXls(filePath, sheetName, onRow, options) {
|
|
1096
|
+
const workbook = XLSX.readFile(filePath, { cellDates: true, dense: true });
|
|
1097
|
+
const names = options.mergeSheets ? workbook.SheetNames : workbook.SheetNames.includes(sheetName) ? [sheetName] : [];
|
|
1098
|
+
if (names.length === 0) {
|
|
1099
|
+
throw new CliValidationError("The requested Excel Sheet was not found.", { code: "LOCAL_DATA_SET_NOT_FOUND" });
|
|
1100
|
+
}
|
|
1101
|
+
let count = 0;
|
|
1102
|
+
for (const name of names) {
|
|
1103
|
+
const sheet = workbook.Sheets[name];
|
|
1104
|
+
if (!sheet) continue;
|
|
1105
|
+
const rows = XLSX.utils.sheet_to_json(sheet, { header: 1, raw: true, defval: null });
|
|
1106
|
+
if (rows.length === 0) continue;
|
|
1107
|
+
let headers = options.headerNames;
|
|
1108
|
+
let start = 0;
|
|
1109
|
+
if (!headers) {
|
|
1110
|
+
if (options.noHeader) {
|
|
1111
|
+
headers = rows[0].map((_, index) => `col_${index + 1}`);
|
|
1112
|
+
} else {
|
|
1113
|
+
headers = dedupeHeaders(rows[0].map((value, index) => String(value ?? `column_${index + 1}`).trim()));
|
|
1114
|
+
start = 1;
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
for (const values of rows.slice(start)) {
|
|
1118
|
+
if (values.every(isMissing)) continue;
|
|
1119
|
+
count += 1;
|
|
1120
|
+
await onRow(Object.fromEntries(headers.map((header, index) => [header, values[index] ?? null])), count);
|
|
1121
|
+
}
|
|
1122
|
+
}
|
|
1123
|
+
return count;
|
|
1124
|
+
}
|
|
1125
|
+
function normalizeRow(value) {
|
|
1126
|
+
if (value !== null && typeof value === "object" && !Array.isArray(value)) {
|
|
1127
|
+
return value;
|
|
1128
|
+
}
|
|
1129
|
+
return { value };
|
|
1130
|
+
}
|
|
1131
|
+
function normalizeExcelValue(value) {
|
|
1132
|
+
if (value instanceof Date) return value;
|
|
1133
|
+
if (value && typeof value === "object") {
|
|
1134
|
+
if ("result" in value) return value.result ?? null;
|
|
1135
|
+
if ("text" in value) return String(value.text);
|
|
1136
|
+
if ("richText" in value) {
|
|
1137
|
+
return value.richText.map((part) => part.text).join("");
|
|
1138
|
+
}
|
|
1139
|
+
}
|
|
1140
|
+
return value ?? null;
|
|
1141
|
+
}
|
|
1142
|
+
function dedupeHeaders(headers) {
|
|
1143
|
+
const counts = /* @__PURE__ */ new Map();
|
|
1144
|
+
return headers.map((header, index) => {
|
|
1145
|
+
const base = header || `column_${index + 1}`;
|
|
1146
|
+
const count = (counts.get(base) ?? 0) + 1;
|
|
1147
|
+
counts.set(base, count);
|
|
1148
|
+
return count === 1 ? base : `${base}_${count}`;
|
|
1149
|
+
});
|
|
1150
|
+
}
|
|
1151
|
+
function isMissing(value) {
|
|
1152
|
+
return value === null || value === void 0 || typeof value === "string" && value.trim() === "";
|
|
1153
|
+
}
|
|
1154
|
+
function localDataParseError(format) {
|
|
1155
|
+
return new CliValidationError(`${format.toUpperCase()} input could not be parsed.`, {
|
|
1156
|
+
code: "LOCAL_DATA_INPUT_INVALID",
|
|
1157
|
+
hint: "Verify the file encoding and structure, then retry without changing the source file.",
|
|
1158
|
+
location: { field: "input-file" }
|
|
1159
|
+
});
|
|
1160
|
+
}
|
|
1161
|
+
async function firstNonWhitespaceCharacter(filePath, encoding) {
|
|
1162
|
+
const stream = decodeTextStream(filePath, encoding);
|
|
1163
|
+
for await (const chunk of stream) {
|
|
1164
|
+
const match = String(chunk).match(/\S/);
|
|
1165
|
+
if (match) {
|
|
1166
|
+
stream.destroy();
|
|
1167
|
+
return match[0];
|
|
1168
|
+
}
|
|
1169
|
+
}
|
|
1170
|
+
return void 0;
|
|
1171
|
+
}
|
|
1172
|
+
|
|
1173
|
+
// src/commands/data-integration/local-data/mapping.ts
|
|
1174
|
+
import { readFileSync } from "fs";
|
|
1175
|
+
var VALID_PROPERTY_NAME = /^[a-z][a-z0-9_]{0,49}$/;
|
|
1176
|
+
function readLocalDataMapping(raw, options) {
|
|
1177
|
+
const trimmed = raw.trim();
|
|
1178
|
+
let text;
|
|
1179
|
+
try {
|
|
1180
|
+
if (trimmed.startsWith("{")) {
|
|
1181
|
+
text = trimmed;
|
|
1182
|
+
} else {
|
|
1183
|
+
const filePath = trimmed.startsWith("@") ? trimmed.slice(1) : trimmed;
|
|
1184
|
+
text = readFileSync(filePath, "utf8");
|
|
1185
|
+
}
|
|
1186
|
+
} catch {
|
|
1187
|
+
throw new CliValidationError("The local-data mapping could not be read.", {
|
|
1188
|
+
code: "LOCAL_DATA_MAPPING_NOT_FOUND",
|
|
1189
|
+
location: { field: "mapping" }
|
|
1190
|
+
});
|
|
1191
|
+
}
|
|
1192
|
+
let value;
|
|
1193
|
+
try {
|
|
1194
|
+
value = JSON.parse(text);
|
|
1195
|
+
} catch {
|
|
1196
|
+
throw new CliValidationError("The local-data mapping is not valid JSON.", {
|
|
1197
|
+
code: "LOCAL_DATA_MAPPING_INVALID_JSON",
|
|
1198
|
+
location: { field: "mapping" }
|
|
1199
|
+
});
|
|
1200
|
+
}
|
|
1201
|
+
validateMapping(value, options);
|
|
1202
|
+
return value;
|
|
1203
|
+
}
|
|
1204
|
+
function validateMapping(value, options) {
|
|
1205
|
+
if (!isRecord(value) || value.version !== "ae-local-data-mapping/v1") {
|
|
1206
|
+
throw mappingError("Mapping version must be ae-local-data-mapping/v1.");
|
|
1207
|
+
}
|
|
1208
|
+
const sha256Valid = typeof value.source?.sha256 === "string" && (options?.sourceWildcard ? value.source.sha256 === "*" || /^[a-f0-9]{64}$/i.test(value.source.sha256) : /^[a-f0-9]{64}$/i.test(value.source.sha256));
|
|
1209
|
+
if (!isRecord(value.source) || !sha256Valid || !["csv", "tsv", "json", "jsonl", "xls", "xlsx"].includes(String(value.source.format)) || typeof value.source.data_set !== "string" || !value.source.data_set) {
|
|
1210
|
+
throw mappingError("Mapping source fingerprint and data set are required.");
|
|
1211
|
+
}
|
|
1212
|
+
if (value.mode !== "track" && value.mode !== "user_set" && value.mode !== "mixed") {
|
|
1213
|
+
throw mappingError("Mapping mode must be track, user_set, or mixed.");
|
|
1214
|
+
}
|
|
1215
|
+
if (value.confidence !== "high" && value.confidence !== "medium" && value.confidence !== "low") {
|
|
1216
|
+
throw mappingError("Mapping confidence must be high, medium, or low.");
|
|
1217
|
+
}
|
|
1218
|
+
const hasRandomIdentity = isRecord(value.random_pool) && (Array.isArray(value.random_pool.account_ids) && value.random_pool.account_ids.length > 0 || Array.isArray(value.random_pool.distinct_ids) && value.random_pool.distinct_ids.length > 0);
|
|
1219
|
+
if (!value.account_id_field && !value.distinct_id_field && !value.account_id_value && !value.distinct_id_value && !hasRandomIdentity) {
|
|
1220
|
+
throw mappingError("Mapping requires a real account or distinct ID field, a fixed identity value, or a random pool.");
|
|
1221
|
+
}
|
|
1222
|
+
if (!isRecord(value.time) || typeof value.time.field !== "string" || !value.time.field || value.time.format !== "auto" || typeof value.time.source_timezone !== "string" || !isValidTimeZone(value.time.source_timezone)) {
|
|
1223
|
+
throw mappingError("Mapping requires a real time field and valid IANA source timezone.");
|
|
1224
|
+
}
|
|
1225
|
+
if (value.mode === "mixed" && !value.record_type_field) {
|
|
1226
|
+
throw mappingError("Mixed mappings require record_type_field.");
|
|
1227
|
+
}
|
|
1228
|
+
if (value.mode !== "user_set" && !value.event_name_field && !value.default_event_name) {
|
|
1229
|
+
throw mappingError("Track mappings require an event field or default event name.");
|
|
1230
|
+
}
|
|
1231
|
+
if (value.default_event_name && !VALID_PROPERTY_NAME.test(value.default_event_name)) {
|
|
1232
|
+
throw mappingError("The default event name is not a legal AE event name.");
|
|
1233
|
+
}
|
|
1234
|
+
if (!Array.isArray(value.properties)) throw mappingError("Mapping properties must be an array.");
|
|
1235
|
+
const targets = /* @__PURE__ */ new Set();
|
|
1236
|
+
for (const property of value.properties) {
|
|
1237
|
+
if (!isRecord(property) || typeof property.source !== "string" || typeof property.target !== "string" || !VALID_PROPERTY_NAME.test(property.target) || !["number", "string", "boolean", "datetime", "list", "object"].includes(String(property.type)) || property.transform !== void 0 && !["stringify", "number", "boolean", "json"].includes(String(property.transform)) || property.value_mapping !== void 0 && !isStringMap(property.value_mapping) || property.time_format !== void 0 && (typeof property.time_format !== "string" || !property.time_format.trim() || property.time_format.length > 64) || property.desc !== void 0 && (typeof property.desc !== "string" || !property.desc.trim() || property.desc.length > 200)) {
|
|
1238
|
+
throw mappingError("Every property mapping needs a source, legal AE target name, and supported type.");
|
|
1239
|
+
}
|
|
1240
|
+
if (targets.has(property.target)) throw mappingError("Property target names must be unique.");
|
|
1241
|
+
targets.add(property.target);
|
|
1242
|
+
}
|
|
1243
|
+
if (value.time_format !== void 0 && (typeof value.time_format !== "string" || !value.time_format.trim() || value.time_format.length > 64)) {
|
|
1244
|
+
throw mappingError("time_format must be a non-empty string of at most 64 characters.");
|
|
1245
|
+
}
|
|
1246
|
+
if (value.event_meta !== void 0) {
|
|
1247
|
+
if (!isRecord(value.event_meta)) throw mappingError("event_meta must be an object keyed by AE event name.");
|
|
1248
|
+
for (const [name, meta] of Object.entries(value.event_meta)) {
|
|
1249
|
+
if (!VALID_PROPERTY_NAME.test(name) || !isRecord(meta) || meta.desc !== void 0 && (typeof meta.desc !== "string" || !meta.desc.trim() || meta.desc.length > 200) || meta.tag !== void 0 && (typeof meta.tag !== "string" || !meta.tag.trim() || meta.tag.length > 64)) {
|
|
1250
|
+
throw mappingError("event_meta entries need a legal AE event name and non-empty desc/tag strings.");
|
|
1251
|
+
}
|
|
1252
|
+
}
|
|
1253
|
+
}
|
|
1254
|
+
if (value.value_mapping !== void 0) {
|
|
1255
|
+
if (!isRecord(value.value_mapping)) throw mappingError("value_mapping must be an object.");
|
|
1256
|
+
for (const key of ["account_id", "distinct_id", "event_name", "record_type"]) {
|
|
1257
|
+
const map = value.value_mapping[key];
|
|
1258
|
+
if (map === void 0) continue;
|
|
1259
|
+
if (!isStringMap(map)) throw mappingError(`value_mapping.${key} must map strings to strings.`);
|
|
1260
|
+
if (key === "event_name") {
|
|
1261
|
+
for (const target of Object.values(map)) {
|
|
1262
|
+
if (!VALID_PROPERTY_NAME.test(target)) {
|
|
1263
|
+
throw mappingError("value_mapping.event_name values must be legal AE event names.");
|
|
1264
|
+
}
|
|
1265
|
+
}
|
|
1266
|
+
}
|
|
1267
|
+
}
|
|
1268
|
+
}
|
|
1269
|
+
if (value.random_pool !== void 0) {
|
|
1270
|
+
if (!isRecord(value.random_pool)) throw mappingError("random_pool must be an object.");
|
|
1271
|
+
for (const key of ["account_ids", "distinct_ids"]) {
|
|
1272
|
+
const pool = value.random_pool[key];
|
|
1273
|
+
if (pool !== void 0 && !isNonEmptyStringArray(pool, true)) {
|
|
1274
|
+
throw mappingError(`random_pool.${key} must be a non-empty array of unique strings.`);
|
|
1275
|
+
}
|
|
1276
|
+
}
|
|
1277
|
+
}
|
|
1278
|
+
if (value.exclude_columns !== void 0 && !isNonEmptyStringArray(value.exclude_columns, true)) {
|
|
1279
|
+
throw mappingError("exclude_columns must be a non-empty array of unique strings.");
|
|
1280
|
+
}
|
|
1281
|
+
if (value.flatten_rules !== void 0) {
|
|
1282
|
+
if (!isRecord(value.flatten_rules)) throw mappingError("flatten_rules must be an object of { column: dot.path }.");
|
|
1283
|
+
for (const [column, path] of Object.entries(value.flatten_rules)) {
|
|
1284
|
+
if (!VALID_PROPERTY_NAME.test(column) || typeof path !== "string" || !path.trim()) {
|
|
1285
|
+
throw mappingError("flatten_rules keys must be legal AE property names and values must be non-empty dot paths.");
|
|
1286
|
+
}
|
|
1287
|
+
}
|
|
1288
|
+
}
|
|
1289
|
+
if (value.headers !== void 0 && !isNonEmptyStringArray(value.headers, true)) {
|
|
1290
|
+
throw mappingError("headers must be a non-empty array of unique strings.");
|
|
1291
|
+
}
|
|
1292
|
+
if (value.missing_time !== void 0 && value.missing_time !== "now") {
|
|
1293
|
+
throw mappingError('missing_time must be "now" when provided.');
|
|
1294
|
+
}
|
|
1295
|
+
if (value.ip_field !== void 0 && (typeof value.ip_field !== "string" || !value.ip_field.trim())) {
|
|
1296
|
+
throw mappingError("ip_field must be a non-empty column name.");
|
|
1297
|
+
}
|
|
1298
|
+
if (value.uuid_field !== void 0 && (typeof value.uuid_field !== "string" || !value.uuid_field.trim())) {
|
|
1299
|
+
throw mappingError("uuid_field must be a non-empty column name.");
|
|
1300
|
+
}
|
|
1301
|
+
if (value.zone_offset_value !== void 0 && value.zone_offset_field !== void 0) {
|
|
1302
|
+
throw mappingError("zone_offset_value and zone_offset_field are mutually exclusive.");
|
|
1303
|
+
}
|
|
1304
|
+
if (value.zone_offset_value !== void 0) {
|
|
1305
|
+
if (typeof value.zone_offset_value === "number") {
|
|
1306
|
+
if (!Number.isInteger(value.zone_offset_value) || value.zone_offset_value < -12 || value.zone_offset_value > 14) {
|
|
1307
|
+
throw mappingError("zone_offset_value must be an integer between -12 and 14.");
|
|
1308
|
+
}
|
|
1309
|
+
} else if (typeof value.zone_offset_value !== "string" || !isValidTimeZone(value.zone_offset_value)) {
|
|
1310
|
+
throw mappingError("zone_offset_value must be an integer or a valid IANA timezone name.");
|
|
1311
|
+
}
|
|
1312
|
+
}
|
|
1313
|
+
if (value.zone_offset_field !== void 0 && (typeof value.zone_offset_field !== "string" || !value.zone_offset_field.trim())) {
|
|
1314
|
+
throw mappingError("zone_offset_field must be a non-empty column name.");
|
|
1315
|
+
}
|
|
1316
|
+
if (value.account_id_value !== void 0 && (typeof value.account_id_value !== "string" || !value.account_id_value.trim() || value.account_id_value.length > 128)) {
|
|
1317
|
+
throw mappingError("account_id_value must be a non-empty string of at most 128 characters.");
|
|
1318
|
+
}
|
|
1319
|
+
if (value.distinct_id_value !== void 0 && (typeof value.distinct_id_value !== "string" || !value.distinct_id_value.trim() || value.distinct_id_value.length > 128)) {
|
|
1320
|
+
throw mappingError("distinct_id_value must be a non-empty string of at most 128 characters.");
|
|
1321
|
+
}
|
|
1322
|
+
}
|
|
1323
|
+
function isValidAeName(value) {
|
|
1324
|
+
return VALID_PROPERTY_NAME.test(value);
|
|
1325
|
+
}
|
|
1326
|
+
function mappingError(message) {
|
|
1327
|
+
return new CliValidationError(message, {
|
|
1328
|
+
code: "LOCAL_DATA_MAPPING_INVALID",
|
|
1329
|
+
location: { field: "mapping" }
|
|
1330
|
+
});
|
|
1331
|
+
}
|
|
1332
|
+
function isValidTimeZone(value) {
|
|
1333
|
+
try {
|
|
1334
|
+
new Intl.DateTimeFormat("en-US", { timeZone: value }).format(/* @__PURE__ */ new Date());
|
|
1335
|
+
return true;
|
|
1336
|
+
} catch {
|
|
1337
|
+
return false;
|
|
1338
|
+
}
|
|
1339
|
+
}
|
|
1340
|
+
function isStringMap(value) {
|
|
1341
|
+
return isRecord(value) && Object.keys(value).length > 0 && Object.entries(value).every(([key, entry]) => key.length > 0 && typeof entry === "string" && entry.length > 0);
|
|
1342
|
+
}
|
|
1343
|
+
function isNonEmptyStringArray(value, unique = false) {
|
|
1344
|
+
if (!Array.isArray(value) || value.length === 0) return false;
|
|
1345
|
+
if (!value.every((entry) => typeof entry === "string" && entry.length > 0)) return false;
|
|
1346
|
+
return !unique || new Set(value).size === value.length;
|
|
1347
|
+
}
|
|
1348
|
+
function isRecord(value) {
|
|
1349
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
// src/commands/data-integration/local-data/multi.ts
|
|
1353
|
+
var PROPERTY_TYPES = /* @__PURE__ */ new Set([
|
|
1354
|
+
"number",
|
|
1355
|
+
"string",
|
|
1356
|
+
"boolean",
|
|
1357
|
+
"datetime",
|
|
1358
|
+
"list",
|
|
1359
|
+
"object"
|
|
1360
|
+
]);
|
|
1361
|
+
function detectColumnTypeConflicts(files) {
|
|
1362
|
+
const columnMap = /* @__PURE__ */ new Map();
|
|
1363
|
+
for (const entry of files) {
|
|
1364
|
+
for (const column of entry.profile.columns) {
|
|
1365
|
+
const sources = columnMap.get(column.name) ?? [];
|
|
1366
|
+
sources.push({ file: entry.file, type: column.inferred_type, samples: column.samples ?? [] });
|
|
1367
|
+
columnMap.set(column.name, sources);
|
|
1368
|
+
}
|
|
1369
|
+
}
|
|
1370
|
+
const conflicts = [];
|
|
1371
|
+
for (const [column, sources] of columnMap) {
|
|
1372
|
+
if (sources.length < 2) continue;
|
|
1373
|
+
if (new Set(sources.map((source) => source.type)).size > 1) {
|
|
1374
|
+
conflicts.push({ column, sources });
|
|
1375
|
+
}
|
|
1376
|
+
}
|
|
1377
|
+
return conflicts;
|
|
1378
|
+
}
|
|
1379
|
+
function buildColumnUnion(files) {
|
|
1380
|
+
const columnMap = /* @__PURE__ */ new Map();
|
|
1381
|
+
for (const entry of files) {
|
|
1382
|
+
for (const column of entry.profile.columns) {
|
|
1383
|
+
const perFileTypes = columnMap.get(column.name) ?? {};
|
|
1384
|
+
perFileTypes[entry.file] = column.inferred_type;
|
|
1385
|
+
columnMap.set(column.name, perFileTypes);
|
|
1386
|
+
}
|
|
1387
|
+
}
|
|
1388
|
+
const entries = [];
|
|
1389
|
+
for (const [column, per_file_types] of columnMap) {
|
|
1390
|
+
const types = Object.values(per_file_types);
|
|
1391
|
+
entries.push({ column, per_file_types, has_conflict: new Set(types).size > 1 });
|
|
1392
|
+
}
|
|
1393
|
+
return entries;
|
|
1394
|
+
}
|
|
1395
|
+
function validateTypeResolutions(value, fileKeys) {
|
|
1396
|
+
if (!isRecord2(value)) {
|
|
1397
|
+
throw resolutionError("type-resolutions must be a JSON object mapping column names to resolutions.");
|
|
1398
|
+
}
|
|
1399
|
+
const knownFiles = new Set(fileKeys);
|
|
1400
|
+
for (const [column, resolution] of Object.entries(value)) {
|
|
1401
|
+
if (!column.trim()) throw resolutionError("type-resolutions keys must be non-empty column names.");
|
|
1402
|
+
if (!isRecord2(resolution)) throw resolutionError("Each type resolution must be an object.");
|
|
1403
|
+
if (resolution.action === "unify") {
|
|
1404
|
+
if (!isPropertyType(resolution.unifiedType)) {
|
|
1405
|
+
throw resolutionError("unify resolutions require a valid unifiedType (number, string, boolean, datetime, list, or object).");
|
|
1406
|
+
}
|
|
1407
|
+
} else if (resolution.action === "split") {
|
|
1408
|
+
if (!isRecord2(resolution.fileMappings) || Object.keys(resolution.fileMappings).length === 0) {
|
|
1409
|
+
throw resolutionError("split resolutions require a non-empty fileMappings object.");
|
|
1410
|
+
}
|
|
1411
|
+
for (const [file, mapping] of Object.entries(resolution.fileMappings)) {
|
|
1412
|
+
if (!knownFiles.has(file)) throw resolutionError("split resolutions reference an unknown file.");
|
|
1413
|
+
if (!isRecord2(mapping) || typeof mapping.ae_name !== "string" || !isValidAeName(mapping.ae_name) || !isPropertyType(mapping.type)) {
|
|
1414
|
+
throw resolutionError("split fileMappings entries require a legal ae_name and a valid type.");
|
|
1415
|
+
}
|
|
1416
|
+
}
|
|
1417
|
+
} else if (resolution.action === "skip") {
|
|
1418
|
+
if (!isNonEmptyStringArray2(resolution.skipFiles) || resolution.skipFiles.some((file) => !knownFiles.has(file))) {
|
|
1419
|
+
throw resolutionError("skip resolutions require a non-empty skipFiles array of known files.");
|
|
1420
|
+
}
|
|
1421
|
+
} else {
|
|
1422
|
+
throw resolutionError("Resolution action must be unify, split, or skip.");
|
|
1423
|
+
}
|
|
1424
|
+
}
|
|
1425
|
+
}
|
|
1426
|
+
function applyTypeResolutions(resolutions, files) {
|
|
1427
|
+
const contexts = /* @__PURE__ */ new Map();
|
|
1428
|
+
for (const entry of files) {
|
|
1429
|
+
const columnTypes = {};
|
|
1430
|
+
for (const column of entry.profile.columns) {
|
|
1431
|
+
columnTypes[column.name] = toPropertyType(column.inferred_type);
|
|
1432
|
+
}
|
|
1433
|
+
contexts.set(entry.file, { columnTypes, targets: {}, skipColumns: /* @__PURE__ */ new Set() });
|
|
1434
|
+
}
|
|
1435
|
+
for (const [column, resolution] of Object.entries(resolutions)) {
|
|
1436
|
+
if (resolution.action === "unify" && resolution.unifiedType) {
|
|
1437
|
+
for (const context of contexts.values()) {
|
|
1438
|
+
context.columnTypes[column] = resolution.unifiedType;
|
|
1439
|
+
}
|
|
1440
|
+
} else if (resolution.action === "skip" && resolution.skipFiles) {
|
|
1441
|
+
for (const file of resolution.skipFiles) {
|
|
1442
|
+
contexts.get(file)?.skipColumns.add(column);
|
|
1443
|
+
}
|
|
1444
|
+
} else if (resolution.action === "split" && resolution.fileMappings) {
|
|
1445
|
+
for (const [file, mapping] of Object.entries(resolution.fileMappings)) {
|
|
1446
|
+
const context = contexts.get(file);
|
|
1447
|
+
if (!context) continue;
|
|
1448
|
+
context.columnTypes[column] = mapping.type;
|
|
1449
|
+
context.targets[column] = mapping.ae_name;
|
|
1450
|
+
}
|
|
1451
|
+
}
|
|
1452
|
+
}
|
|
1453
|
+
return contexts;
|
|
1454
|
+
}
|
|
1455
|
+
function buildPerFileMapping(baseMapping, profile, overrides) {
|
|
1456
|
+
const excludeColumns = new Set(baseMapping.exclude_columns ?? []);
|
|
1457
|
+
for (const column of overrides.skipColumns) excludeColumns.add(column);
|
|
1458
|
+
const properties = baseMapping.properties.map((property) => {
|
|
1459
|
+
const overrideType = overrides.columnTypes[property.source];
|
|
1460
|
+
const target = overrides.targets[property.source] ?? property.target;
|
|
1461
|
+
return {
|
|
1462
|
+
...property,
|
|
1463
|
+
...overrideType !== void 0 ? { type: overrideType } : {},
|
|
1464
|
+
target
|
|
1465
|
+
};
|
|
1466
|
+
});
|
|
1467
|
+
return {
|
|
1468
|
+
...baseMapping,
|
|
1469
|
+
source: {
|
|
1470
|
+
sha256: profile.source.sha256,
|
|
1471
|
+
format: profile.source.format,
|
|
1472
|
+
data_set: profile.data_set.id
|
|
1473
|
+
},
|
|
1474
|
+
properties,
|
|
1475
|
+
...excludeColumns.size > 0 ? { exclude_columns: [...excludeColumns] } : {}
|
|
1476
|
+
};
|
|
1477
|
+
}
|
|
1478
|
+
function toPropertyType(inferredType) {
|
|
1479
|
+
return isPropertyType(inferredType) ? inferredType : "string";
|
|
1480
|
+
}
|
|
1481
|
+
function isPropertyType(value) {
|
|
1482
|
+
return typeof value === "string" && PROPERTY_TYPES.has(value);
|
|
1483
|
+
}
|
|
1484
|
+
function isNonEmptyStringArray2(value) {
|
|
1485
|
+
return Array.isArray(value) && value.length > 0 && value.every((entry) => typeof entry === "string" && entry.length > 0);
|
|
1486
|
+
}
|
|
1487
|
+
function resolutionError(message) {
|
|
1488
|
+
return new CliValidationError(message, {
|
|
1489
|
+
code: "LOCAL_DATA_TYPE_RESOLUTIONS_INVALID",
|
|
1490
|
+
location: { field: "type-resolutions" }
|
|
1491
|
+
});
|
|
1492
|
+
}
|
|
1493
|
+
function isRecord2(value) {
|
|
1494
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
1495
|
+
}
|
|
1496
|
+
|
|
1497
|
+
// src/commands/data-integration/local-data/profile.ts
|
|
1498
|
+
import { createHash as createHash2, randomInt } from "crypto";
|
|
1499
|
+
import { basename as basename2, extname as extname2 } from "path";
|
|
1500
|
+
var UNIQUE_SAMPLE_LIMIT = 1e4;
|
|
1501
|
+
var GLOBAL_UNIQUE_SAMPLE_LIMIT = 2e5;
|
|
1502
|
+
var IDENTITY_MAX_LENGTH = 128;
|
|
1503
|
+
var COLUMN_SAMPLE_LIMIT = 5;
|
|
1504
|
+
var SAMPLE_TRUNCATE_LENGTH = 40;
|
|
1505
|
+
var NESTED_TREE_SAMPLE_LIMIT = 1e3;
|
|
1506
|
+
var USER_PROFILE_TYPES = /* @__PURE__ */ new Set([
|
|
1507
|
+
"user_set",
|
|
1508
|
+
"user_setOnce",
|
|
1509
|
+
"user_add",
|
|
1510
|
+
"user_unset",
|
|
1511
|
+
"user_del",
|
|
1512
|
+
"user_append",
|
|
1513
|
+
"user_uniq_append"
|
|
1514
|
+
]);
|
|
1515
|
+
function isUserProfileType(recordType) {
|
|
1516
|
+
return USER_PROFILE_TYPES.has(recordType);
|
|
1517
|
+
}
|
|
1518
|
+
var ACCOUNT_NAMES = ["#account_id", "account_id", "accountid", "user_id", "userid", "uid", "member_id", "\u8D26\u53F7", "\u7528\u6237id", "\u8D26\u6237id"];
|
|
1519
|
+
var DISTINCT_NAMES = ["#distinct_id", "distinct_id", "distinctid", "device_id", "deviceid", "visitor_id", "anonymous_id", "\u8BBE\u5907id", "\u8BBF\u5BA2id"];
|
|
1520
|
+
var TIME_NAMES = ["#time", "time", "timestamp", "event_time", "created_at", "occurred_at", "datetime", "date", "\u65F6\u95F4", "\u4E8B\u4EF6\u65F6\u95F4", "\u53D1\u751F\u65F6\u95F4", "\u521B\u5EFA\u65F6\u95F4", "\u4E0B\u5355\u65F6\u95F4", "\u8BA2\u5355\u65F6\u95F4"];
|
|
1521
|
+
var EVENT_NAMES = ["#event_name", "event_name", "event", "action", "activity", "\u4E8B\u4EF6\u540D", "\u4E8B\u4EF6\u540D\u79F0", "\u4E8B\u4EF6", "\u884C\u4E3A", "\u52A8\u4F5C"];
|
|
1522
|
+
var TYPE_NAMES = ["#type", "record_type", "data_type", "\u64CD\u4F5C\u7C7B\u578B"];
|
|
1523
|
+
async function profileLocalData(input, dataSet, sourceTimezone = "Asia/Shanghai", options = {}) {
|
|
1524
|
+
const columns = /* @__PURE__ */ new Map();
|
|
1525
|
+
const recognizedRecordTypes = /* @__PURE__ */ new Set();
|
|
1526
|
+
let uniqueSamples = 0;
|
|
1527
|
+
let rowCount = 0;
|
|
1528
|
+
const collectNestedTree = Boolean(
|
|
1529
|
+
options.collectNestedTree && (input.format === "json" || input.format === "jsonl")
|
|
1530
|
+
);
|
|
1531
|
+
const collectDelimitedTree = Boolean(
|
|
1532
|
+
options.collectNestedTree && (input.format === "csv" || input.format === "tsv")
|
|
1533
|
+
);
|
|
1534
|
+
let nestedObjects = [];
|
|
1535
|
+
let nestedSeen = 0;
|
|
1536
|
+
const delimitedNested = /* @__PURE__ */ new Map();
|
|
1537
|
+
await streamLocalDataRows(
|
|
1538
|
+
input,
|
|
1539
|
+
dataSet,
|
|
1540
|
+
(row) => {
|
|
1541
|
+
rowCount += 1;
|
|
1542
|
+
if (collectNestedTree && row !== null && typeof row === "object" && !Array.isArray(row)) {
|
|
1543
|
+
nestedSeen += 1;
|
|
1544
|
+
if (nestedObjects.length < NESTED_TREE_SAMPLE_LIMIT) {
|
|
1545
|
+
nestedObjects.push(row);
|
|
1546
|
+
} else {
|
|
1547
|
+
const slot = randomInt(nestedSeen);
|
|
1548
|
+
if (slot < NESTED_TREE_SAMPLE_LIMIT) nestedObjects[slot] = row;
|
|
1549
|
+
}
|
|
1550
|
+
}
|
|
1551
|
+
for (const name of /* @__PURE__ */ new Set([...columns.keys(), ...Object.keys(row)])) {
|
|
1552
|
+
let accumulator = columns.get(name);
|
|
1553
|
+
if (!accumulator) {
|
|
1554
|
+
accumulator = {
|
|
1555
|
+
name,
|
|
1556
|
+
missing: rowCount - 1,
|
|
1557
|
+
nonMissing: 0,
|
|
1558
|
+
types: /* @__PURE__ */ new Map(),
|
|
1559
|
+
uniqueHashes: /* @__PURE__ */ new Set(),
|
|
1560
|
+
uniqueOverflow: false,
|
|
1561
|
+
timeParseCount: 0,
|
|
1562
|
+
timeFormatCounts: /* @__PURE__ */ new Map(),
|
|
1563
|
+
samples: [],
|
|
1564
|
+
sampleSet: /* @__PURE__ */ new Set()
|
|
1565
|
+
};
|
|
1566
|
+
columns.set(name, accumulator);
|
|
1567
|
+
}
|
|
1568
|
+
const value = row[name];
|
|
1569
|
+
if (isMissing2(value)) {
|
|
1570
|
+
accumulator.missing += 1;
|
|
1571
|
+
continue;
|
|
1572
|
+
}
|
|
1573
|
+
accumulator.nonMissing += 1;
|
|
1574
|
+
const type = inferValueType(value);
|
|
1575
|
+
accumulator.types.set(type, (accumulator.types.get(type) ?? 0) + 1);
|
|
1576
|
+
const valueHash = hashValue(value);
|
|
1577
|
+
if (accumulator.uniqueHashes.has(valueHash)) {
|
|
1578
|
+
} else if (accumulator.uniqueHashes.size < UNIQUE_SAMPLE_LIMIT && uniqueSamples < GLOBAL_UNIQUE_SAMPLE_LIMIT) {
|
|
1579
|
+
accumulator.uniqueHashes.add(valueHash);
|
|
1580
|
+
uniqueSamples += 1;
|
|
1581
|
+
} else {
|
|
1582
|
+
accumulator.uniqueOverflow = true;
|
|
1583
|
+
}
|
|
1584
|
+
if (options.collectSamples) recordSample(accumulator, value);
|
|
1585
|
+
if (collectDelimitedTree && typeof value === "string" && value.trim().startsWith("{")) {
|
|
1586
|
+
try {
|
|
1587
|
+
const parsed = JSON.parse(value);
|
|
1588
|
+
if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
1589
|
+
let sampler = delimitedNested.get(name);
|
|
1590
|
+
if (!sampler) {
|
|
1591
|
+
sampler = { seen: 0, objects: [] };
|
|
1592
|
+
delimitedNested.set(name, sampler);
|
|
1593
|
+
}
|
|
1594
|
+
sampler.seen += 1;
|
|
1595
|
+
if (sampler.objects.length < NESTED_TREE_SAMPLE_LIMIT) {
|
|
1596
|
+
sampler.objects.push(parsed);
|
|
1597
|
+
} else {
|
|
1598
|
+
const slot = randomInt(sampler.seen);
|
|
1599
|
+
if (slot < NESTED_TREE_SAMPLE_LIMIT) sampler.objects[slot] = parsed;
|
|
1600
|
+
}
|
|
1601
|
+
}
|
|
1602
|
+
} catch {
|
|
1603
|
+
}
|
|
1604
|
+
}
|
|
1605
|
+
if (isParseableTime(value, matchesName(name, TIME_NAMES))) {
|
|
1606
|
+
accumulator.timeParseCount += 1;
|
|
1607
|
+
if (typeof value === "string") {
|
|
1608
|
+
const format = findFirstTimeFormat(value);
|
|
1609
|
+
if (format) {
|
|
1610
|
+
accumulator.timeFormatCounts.set(format, (accumulator.timeFormatCounts.get(format) ?? 0) + 1);
|
|
1611
|
+
}
|
|
1612
|
+
}
|
|
1613
|
+
}
|
|
1614
|
+
if (matchesName(name, TYPE_NAMES)) {
|
|
1615
|
+
const normalized = normalizeRecordType(value);
|
|
1616
|
+
if (normalized) recognizedRecordTypes.add(normalized);
|
|
1617
|
+
}
|
|
1618
|
+
}
|
|
1619
|
+
},
|
|
1620
|
+
{
|
|
1621
|
+
delimiter: options.delimiter,
|
|
1622
|
+
encoding: options.encoding,
|
|
1623
|
+
headerNames: options.headerNames,
|
|
1624
|
+
noHeader: options.noHeader,
|
|
1625
|
+
flattenRules: options.flattenRules,
|
|
1626
|
+
mergeSheets: options.mergeSheets
|
|
1627
|
+
}
|
|
1628
|
+
);
|
|
1629
|
+
const delimitedNestedTree = /* @__PURE__ */ new Map();
|
|
1630
|
+
for (const [columnName, sampler] of delimitedNested) {
|
|
1631
|
+
if (sampler.objects.length > 0) delimitedNestedTree.set(columnName, buildNestedTree(sampler.objects));
|
|
1632
|
+
}
|
|
1633
|
+
const columnProfiles = [];
|
|
1634
|
+
const timeFormatByColumn = /* @__PURE__ */ new Map();
|
|
1635
|
+
for (const column of columns.values()) {
|
|
1636
|
+
const profile = formatColumnProfile(column, rowCount, options.collectSamples ?? false);
|
|
1637
|
+
const nestedTree = delimitedNestedTree.get(column.name);
|
|
1638
|
+
if (nestedTree && nestedTree.length > 0) profile.nested_tree = nestedTree;
|
|
1639
|
+
columnProfiles.push(profile);
|
|
1640
|
+
const dominantFormat = dominantTimeFormat(column);
|
|
1641
|
+
if (dominantFormat) timeFormatByColumn.set(column.name, dominantFormat);
|
|
1642
|
+
}
|
|
1643
|
+
const identityCandidates = findIdentityCandidates(columnProfiles);
|
|
1644
|
+
const recommendedMapping = recommendMapping({
|
|
1645
|
+
input,
|
|
1646
|
+
dataSet,
|
|
1647
|
+
rowCount,
|
|
1648
|
+
columns: columnProfiles,
|
|
1649
|
+
recognizedRecordTypes,
|
|
1650
|
+
sourceTimezone,
|
|
1651
|
+
headerNames: options.headerNames,
|
|
1652
|
+
timeFormatByColumn
|
|
1653
|
+
});
|
|
1654
|
+
const warnings = [...recommendedMapping.warnings ?? []];
|
|
1655
|
+
if (rowCount === 0) warnings.push("The selected data set contains no data rows.");
|
|
1656
|
+
return {
|
|
1657
|
+
version: "ae-local-data-profile/v1",
|
|
1658
|
+
source: {
|
|
1659
|
+
format: input.format,
|
|
1660
|
+
size_bytes: input.sizeBytes,
|
|
1661
|
+
sha256: input.sha256
|
|
1662
|
+
},
|
|
1663
|
+
data_set: dataSet,
|
|
1664
|
+
row_count: rowCount,
|
|
1665
|
+
column_count: columnProfiles.length,
|
|
1666
|
+
columns: columnProfiles,
|
|
1667
|
+
identity_candidates: identityCandidates,
|
|
1668
|
+
recommended_mapping: recommendedMapping,
|
|
1669
|
+
ue_eligible: Boolean(
|
|
1670
|
+
(recommendedMapping.account_id_field || recommendedMapping.distinct_id_field) && recommendedMapping.time.field
|
|
1671
|
+
),
|
|
1672
|
+
warnings,
|
|
1673
|
+
...collectNestedTree && nestedObjects.length > 0 ? { nested_tree: buildNestedTree(nestedObjects) } : {}
|
|
1674
|
+
};
|
|
1675
|
+
}
|
|
1676
|
+
function normalizeAeName(input, fallback) {
|
|
1677
|
+
let normalized = input.normalize("NFKD").replace(/[\u0300-\u036f]/g, "").replace(/([a-z0-9])([A-Z])/g, "$1_$2").toLowerCase().replace(/[^a-z0-9_]+/g, "_").replace(/_+/g, "_").replace(/^_+|_+$/g, "");
|
|
1678
|
+
if (!normalized) normalized = fallback;
|
|
1679
|
+
if (!/^[a-z]/.test(normalized)) normalized = `field_${normalized}`;
|
|
1680
|
+
return normalized.slice(0, 50).replace(/_+$/g, "") || fallback;
|
|
1681
|
+
}
|
|
1682
|
+
function inferValueType(value) {
|
|
1683
|
+
if (isMissing2(value)) return "null";
|
|
1684
|
+
if (value instanceof Date) return "datetime";
|
|
1685
|
+
if (Array.isArray(value)) return "list";
|
|
1686
|
+
if (typeof value === "boolean") return "boolean";
|
|
1687
|
+
if (typeof value === "number") return "number";
|
|
1688
|
+
if (value !== null && typeof value === "object") return "object";
|
|
1689
|
+
if (typeof value === "string") {
|
|
1690
|
+
const normalized = value.trim().toLowerCase();
|
|
1691
|
+
if (isStrongDateTime2(value)) return "datetime";
|
|
1692
|
+
if (normalized === "true" || normalized === "false") return "boolean";
|
|
1693
|
+
if (/^[+-]?(?:\d+\.?\d*|\.\d+)(?:e[+-]?\d+)?$/i.test(normalized)) return "number";
|
|
1694
|
+
if (looksLikeJsonContainer(value)) {
|
|
1695
|
+
if (looksLikeArray(value)) return "list";
|
|
1696
|
+
if (looksLikeObject(value)) return "object";
|
|
1697
|
+
}
|
|
1698
|
+
}
|
|
1699
|
+
return "string";
|
|
1700
|
+
}
|
|
1701
|
+
function isMissing2(value) {
|
|
1702
|
+
return value === null || value === void 0 || typeof value === "string" && value.trim() === "";
|
|
1703
|
+
}
|
|
1704
|
+
function recommendMapping(input) {
|
|
1705
|
+
const account = findCandidate(input.columns, ACCOUNT_NAMES);
|
|
1706
|
+
const distinct = findCandidate(input.columns, DISTINCT_NAMES);
|
|
1707
|
+
const time = findTimeCandidate(input.columns);
|
|
1708
|
+
const event = findCandidate(input.columns, EVENT_NAMES);
|
|
1709
|
+
const recordType = findCandidate(input.columns, TYPE_NAMES);
|
|
1710
|
+
const identity = account ?? distinct;
|
|
1711
|
+
const recognized = [...input.recognizedRecordTypes];
|
|
1712
|
+
const hasTrack = recognized.includes("track");
|
|
1713
|
+
const hasProfile = recognized.some(isUserProfileType);
|
|
1714
|
+
let mode;
|
|
1715
|
+
let confidence;
|
|
1716
|
+
if (recognized.length > 0) {
|
|
1717
|
+
if (hasTrack && hasProfile) {
|
|
1718
|
+
mode = "mixed";
|
|
1719
|
+
} else if (hasProfile) {
|
|
1720
|
+
mode = "user_set";
|
|
1721
|
+
} else {
|
|
1722
|
+
mode = "track";
|
|
1723
|
+
}
|
|
1724
|
+
confidence = "high";
|
|
1725
|
+
} else if (event) {
|
|
1726
|
+
mode = "track";
|
|
1727
|
+
confidence = "high";
|
|
1728
|
+
} else if (identity && identity.unique_ratio < 0.9 && input.rowCount > 1) {
|
|
1729
|
+
mode = "track";
|
|
1730
|
+
confidence = "medium";
|
|
1731
|
+
} else {
|
|
1732
|
+
mode = "user_set";
|
|
1733
|
+
confidence = identity ? "medium" : "low";
|
|
1734
|
+
}
|
|
1735
|
+
const warnings = [];
|
|
1736
|
+
if (!account && !distinct) warnings.push("No real account or distinct ID field was identified.");
|
|
1737
|
+
if (!time) warnings.push("No real time field with parseable values was identified.");
|
|
1738
|
+
if (confidence === "low") warnings.push("The UE classification has low confidence and requires review.");
|
|
1739
|
+
if (recordType && input.recognizedRecordTypes.size === 0) {
|
|
1740
|
+
warnings.push("A record-type column exists, but its values are not recognized as track or a user profile type.");
|
|
1741
|
+
}
|
|
1742
|
+
if (mode === "track" && !event) {
|
|
1743
|
+
warnings.push("No event-name column was found; review the generated default event name.");
|
|
1744
|
+
}
|
|
1745
|
+
const reserved = new Set([account?.name, distinct?.name, time?.name, event?.name, recordType?.name].filter(Boolean));
|
|
1746
|
+
const usedTargets = /* @__PURE__ */ new Set();
|
|
1747
|
+
const properties = input.columns.filter((column) => !reserved.has(column.name)).map((column, index) => {
|
|
1748
|
+
const baseTarget = normalizeAeName(column.name, `field_${index + 1}`);
|
|
1749
|
+
let target = baseTarget;
|
|
1750
|
+
let suffix = 2;
|
|
1751
|
+
while (usedTargets.has(target)) target = `${baseTarget.slice(0, 46)}_${suffix++}`;
|
|
1752
|
+
usedTargets.add(target);
|
|
1753
|
+
const type = mappingType(column.inferred_type);
|
|
1754
|
+
return {
|
|
1755
|
+
source: column.name,
|
|
1756
|
+
target,
|
|
1757
|
+
type,
|
|
1758
|
+
...type === "object" || type === "list" ? { transform: "json" } : {}
|
|
1759
|
+
};
|
|
1760
|
+
});
|
|
1761
|
+
const defaultEventSource = input.dataSet.kind === "sheet" ? input.dataSet.label : basename2(input.input.filePath, extname2(input.input.filePath));
|
|
1762
|
+
return {
|
|
1763
|
+
version: "ae-local-data-mapping/v1",
|
|
1764
|
+
source: {
|
|
1765
|
+
sha256: input.input.sha256,
|
|
1766
|
+
format: input.input.format,
|
|
1767
|
+
data_set: input.dataSet.id
|
|
1768
|
+
},
|
|
1769
|
+
mode,
|
|
1770
|
+
confidence,
|
|
1771
|
+
...account ? { account_id_field: account.name } : {},
|
|
1772
|
+
...distinct ? { distinct_id_field: distinct.name } : {},
|
|
1773
|
+
time: {
|
|
1774
|
+
field: time?.name ?? "",
|
|
1775
|
+
format: "auto",
|
|
1776
|
+
source_timezone: input.sourceTimezone
|
|
1777
|
+
},
|
|
1778
|
+
...time ? { time_format: input.timeFormatByColumn.get(time.name) } : {},
|
|
1779
|
+
...input.headerNames && input.headerNames.length > 0 ? { headers: input.headerNames } : {},
|
|
1780
|
+
...recordType ? { record_type_field: recordType.name } : {},
|
|
1781
|
+
...event ? { event_name_field: event.name } : {},
|
|
1782
|
+
...mode !== "user_set" && !event ? { default_event_name: normalizeAeName(defaultEventSource, "local_event") } : {},
|
|
1783
|
+
properties,
|
|
1784
|
+
...warnings.length > 0 ? { warnings } : {}
|
|
1785
|
+
};
|
|
1786
|
+
}
|
|
1787
|
+
function formatColumnProfile(column, rowCount, includeSamples) {
|
|
1788
|
+
const nonNullTypes = [...column.types.entries()].filter(([type]) => type !== "null").sort((left, right) => right[1] - left[1]);
|
|
1789
|
+
const inferred = nonNullTypes.length === 0 ? "unknown" : nonNullTypes.length === 1 ? nonNullTypes[0][0] : compatibleTypes(nonNullTypes.map(([type]) => type)) ? nonNullTypes[0][0] : "mixed";
|
|
1790
|
+
const finalType = inferred === "number" && idLikeColumn(column.name) ? "string" : inferred;
|
|
1791
|
+
return {
|
|
1792
|
+
name: column.name,
|
|
1793
|
+
inferred_type: finalType,
|
|
1794
|
+
missing_count: column.missing,
|
|
1795
|
+
missing_ratio: ratio(column.missing, rowCount),
|
|
1796
|
+
unique_count: column.uniqueHashes.size,
|
|
1797
|
+
unique_ratio: ratio(column.uniqueHashes.size, column.nonMissing),
|
|
1798
|
+
unique_count_approximate: column.uniqueOverflow,
|
|
1799
|
+
time_parse_count: column.timeParseCount,
|
|
1800
|
+
time_parse_ratio: ratio(column.timeParseCount, column.nonMissing),
|
|
1801
|
+
...includeSamples ? { samples: column.samples } : {}
|
|
1802
|
+
};
|
|
1803
|
+
}
|
|
1804
|
+
function findCandidate(columns, names) {
|
|
1805
|
+
return columns.find((column) => matchesName(column.name, names));
|
|
1806
|
+
}
|
|
1807
|
+
function findIdentityCandidates(columns) {
|
|
1808
|
+
return columns.filter((column) => matchesName(column.name, ACCOUNT_NAMES) || matchesName(column.name, DISTINCT_NAMES)).map((column) => ({
|
|
1809
|
+
name: column.name,
|
|
1810
|
+
kind: matchesName(column.name, DISTINCT_NAMES) ? "distinct" : "account",
|
|
1811
|
+
unique_ratio: column.unique_ratio,
|
|
1812
|
+
missing_ratio: column.missing_ratio
|
|
1813
|
+
}));
|
|
1814
|
+
}
|
|
1815
|
+
function findTimeCandidate(columns) {
|
|
1816
|
+
const named = findCandidate(columns, TIME_NAMES);
|
|
1817
|
+
if (named && named.time_parse_ratio >= 0.8) return named;
|
|
1818
|
+
return columns.filter((column) => column.time_parse_ratio >= 0.95).sort((left, right) => right.time_parse_ratio - left.time_parse_ratio)[0];
|
|
1819
|
+
}
|
|
1820
|
+
function matchesName(input, names) {
|
|
1821
|
+
const normalized = input.trim().toLowerCase().replace(/[\s-]+/g, "_");
|
|
1822
|
+
return names.some((name) => normalized === name.toLowerCase());
|
|
1823
|
+
}
|
|
1824
|
+
function idLikeColumn(name) {
|
|
1825
|
+
const low = name.toLowerCase().trim();
|
|
1826
|
+
return /_(id|key|code|no|num)$/i.test(low) || low === "id" || low.endsWith("id");
|
|
1827
|
+
}
|
|
1828
|
+
function looksLikeJsonContainer(value) {
|
|
1829
|
+
const trimmed = value.trim();
|
|
1830
|
+
if (trimmed.length < 2) return false;
|
|
1831
|
+
const first = trimmed[0];
|
|
1832
|
+
const last = trimmed[trimmed.length - 1];
|
|
1833
|
+
return first === "{" && last === "}" || first === "[" && last === "]";
|
|
1834
|
+
}
|
|
1835
|
+
function looksLikeObject(value) {
|
|
1836
|
+
try {
|
|
1837
|
+
const parsed = JSON.parse(value);
|
|
1838
|
+
return parsed !== null && typeof parsed === "object" && !Array.isArray(parsed);
|
|
1839
|
+
} catch {
|
|
1840
|
+
return false;
|
|
1841
|
+
}
|
|
1842
|
+
}
|
|
1843
|
+
function looksLikeArray(value) {
|
|
1844
|
+
try {
|
|
1845
|
+
return Array.isArray(JSON.parse(value));
|
|
1846
|
+
} catch {
|
|
1847
|
+
return false;
|
|
1848
|
+
}
|
|
1849
|
+
}
|
|
1850
|
+
function normalizeRecordType(value) {
|
|
1851
|
+
const normalized = String(value ?? "").trim().toLowerCase().replace(/^#/, "").replace(/[-_\s]+/g, "");
|
|
1852
|
+
const aliases = {
|
|
1853
|
+
track: "track",
|
|
1854
|
+
event: "track",
|
|
1855
|
+
userset: "user_set",
|
|
1856
|
+
user: "user_set",
|
|
1857
|
+
usersetonce: "user_setOnce",
|
|
1858
|
+
useradd: "user_add",
|
|
1859
|
+
userunset: "user_unset",
|
|
1860
|
+
userdel: "user_del",
|
|
1861
|
+
userappend: "user_append",
|
|
1862
|
+
useruniqappend: "user_uniq_append"
|
|
1863
|
+
};
|
|
1864
|
+
return aliases[normalized];
|
|
1865
|
+
}
|
|
1866
|
+
function mappingType(type) {
|
|
1867
|
+
if (type === "datetime") return "datetime";
|
|
1868
|
+
if (type === "number" || type === "boolean" || type === "list" || type === "object") return type;
|
|
1869
|
+
return "string";
|
|
1870
|
+
}
|
|
1871
|
+
function compatibleTypes(types) {
|
|
1872
|
+
const set = new Set(types);
|
|
1873
|
+
return set.size <= 1 || set.size === 2 && set.has("string") && set.has("datetime");
|
|
1874
|
+
}
|
|
1875
|
+
function hashValue(value) {
|
|
1876
|
+
let serialized;
|
|
1877
|
+
try {
|
|
1878
|
+
serialized = value instanceof Date ? value.toISOString() : JSON.stringify(value);
|
|
1879
|
+
} catch {
|
|
1880
|
+
serialized = String(value);
|
|
1881
|
+
}
|
|
1882
|
+
return createHash2("sha256").update(serialized).digest("hex");
|
|
1883
|
+
}
|
|
1884
|
+
function isParseableTime(value, allowNumeric) {
|
|
1885
|
+
if (value instanceof Date) return Number.isFinite(value.getTime());
|
|
1886
|
+
if (typeof value === "number") {
|
|
1887
|
+
return allowNumeric && Number.isFinite(value) && (value > 1e9 || value >= 1 && value <= 2958465);
|
|
1888
|
+
}
|
|
1889
|
+
if (typeof value !== "string") return false;
|
|
1890
|
+
if (/^\d{10}(?:\d{3})?$/.test(value.trim())) return true;
|
|
1891
|
+
return isStrongDateTime2(value) || isParseableByAnyFormat(value);
|
|
1892
|
+
}
|
|
1893
|
+
function isStrongDateTime2(value) {
|
|
1894
|
+
return /^\d{4}[-/]\d{1,2}[-/]\d{1,2}(?:[ T]\d{1,2}:\d{2}(?::\d{2}(?:\.\d{1,3})?)?(?:Z|[+-]\d{2}:?\d{2})?)?$/.test(value.trim());
|
|
1895
|
+
}
|
|
1896
|
+
function ratio(numerator, denominator) {
|
|
1897
|
+
return denominator === 0 ? 0 : Number((numerator / denominator).toFixed(6));
|
|
1898
|
+
}
|
|
1899
|
+
function recordSample(accumulator, value) {
|
|
1900
|
+
if (accumulator.samples.length >= COLUMN_SAMPLE_LIMIT) return;
|
|
1901
|
+
const text = truncateSample2(value);
|
|
1902
|
+
if (accumulator.sampleSet.has(text)) return;
|
|
1903
|
+
accumulator.sampleSet.add(text);
|
|
1904
|
+
accumulator.samples.push(text);
|
|
1905
|
+
}
|
|
1906
|
+
function truncateSample2(value) {
|
|
1907
|
+
const text = sampleText(value);
|
|
1908
|
+
if (text.length <= SAMPLE_TRUNCATE_LENGTH) return text;
|
|
1909
|
+
return `${Array.from(text).slice(0, SAMPLE_TRUNCATE_LENGTH).join("")}\u2026`;
|
|
1910
|
+
}
|
|
1911
|
+
function sampleText(value) {
|
|
1912
|
+
if (value instanceof Date) return value.toISOString();
|
|
1913
|
+
if (value === null) return "null";
|
|
1914
|
+
if (typeof value === "object") return JSON.stringify(value);
|
|
1915
|
+
return String(value);
|
|
1916
|
+
}
|
|
1917
|
+
function dominantTimeFormat(column) {
|
|
1918
|
+
if (column.timeFormatCounts.size === 0) return void 0;
|
|
1919
|
+
const total = [...column.timeFormatCounts.values()].reduce((sum, count) => sum + count, 0);
|
|
1920
|
+
let best;
|
|
1921
|
+
let bestCount = 0;
|
|
1922
|
+
for (const [format, count] of column.timeFormatCounts) {
|
|
1923
|
+
if (count > bestCount) {
|
|
1924
|
+
best = format;
|
|
1925
|
+
bestCount = count;
|
|
1926
|
+
}
|
|
1927
|
+
}
|
|
1928
|
+
return best && bestCount >= total * 0.9 ? best : void 0;
|
|
1929
|
+
}
|
|
1930
|
+
|
|
1931
|
+
// src/commands/data-integration/local-data/inspect.ts
|
|
1932
|
+
var HEADERLESS_WARNING = "The first row appears to be data, not a header; columns were auto-named col_1..col_N. Re-run with --headers to supply explicit names.";
|
|
1933
|
+
var dataIntegrationInspect = {
|
|
1934
|
+
service: "data-integration",
|
|
1935
|
+
command: "inspect",
|
|
1936
|
+
usesAeHost: false,
|
|
1937
|
+
description: "Stream one or more local CSV, TSV, TXT, JSON, JSONL, XLS, or XLSX files and recommend a UE mapping.",
|
|
1938
|
+
flags: [
|
|
1939
|
+
{ name: "input-file", type: "string", required: true, sensitive: true, variadic: true, desc: "Local CSV, TSV, TXT, JSON, JSONL, XLS, or XLSX file. Repeat for multiple files." },
|
|
1940
|
+
{ name: "data-set", type: "string", sensitive: true, desc: "Sheet or JSON Path ID returned by a discovery-only inspection." },
|
|
1941
|
+
{ name: "source-timezone", type: "string", default: "Asia/Shanghai", desc: "IANA timezone used to interpret source times without an offset." },
|
|
1942
|
+
{ name: "headers", type: "string", sensitive: true, desc: "Comma-separated explicit column names for a headerless file; the first row is data." },
|
|
1943
|
+
{ name: "headerless", type: "boolean", default: false, desc: "Treat the first row as data and auto-generate col_1..col_N names." }
|
|
1944
|
+
],
|
|
1945
|
+
risk: "read",
|
|
1946
|
+
// Fast size/time pre-check (stat + format/encoding sniff only — no sha256, no profile).
|
|
1947
|
+
// Lets agents surface the estimate before committing to a multi-minute full inspection.
|
|
1948
|
+
dryRun: async (ctx) => {
|
|
1949
|
+
const files = ctx.list("input-file").map((filePath) => {
|
|
1950
|
+
const meta = resolveLocalDataInputMeta(filePath);
|
|
1951
|
+
const assessment = assessFileSize(basename3(filePath), meta.format, meta.sizeBytes, meta.encoding);
|
|
1952
|
+
return {
|
|
1953
|
+
file: basename3(filePath),
|
|
1954
|
+
format: meta.format,
|
|
1955
|
+
size_bytes: meta.sizeBytes,
|
|
1956
|
+
size: assessment.size,
|
|
1957
|
+
...assessment.estimatedDuration ? { estimated_duration: assessment.estimatedDuration } : {},
|
|
1958
|
+
...assessment.warning ? { warning: assessment.warning } : {},
|
|
1959
|
+
...assessment.reason ? { reason: assessment.reason } : {},
|
|
1960
|
+
...assessment.memoryRisk ? { memory_risk: true } : {},
|
|
1961
|
+
...assessment.rejected ? { rejected: true } : {}
|
|
1962
|
+
};
|
|
1963
|
+
});
|
|
1964
|
+
return {
|
|
1965
|
+
version: "ae-local-data-estimate/v1",
|
|
1966
|
+
files,
|
|
1967
|
+
has_large_file: files.some((file) => Boolean(file.warning) || file.rejected)
|
|
1968
|
+
};
|
|
1969
|
+
},
|
|
1970
|
+
execute: async (ctx) => {
|
|
1971
|
+
const inputFiles = ctx.list("input-file");
|
|
1972
|
+
const headerNames = splitHeaders(ctx.str("headers"));
|
|
1973
|
+
const noHeader = ctx.bool("headerless");
|
|
1974
|
+
const sourceTimezone = ctx.str("source-timezone");
|
|
1975
|
+
const requested = ctx.str("data-set").trim() || void 0;
|
|
1976
|
+
if (inputFiles.length === 1) {
|
|
1977
|
+
const input = await inspectLocalDataInput(inputFiles[0]);
|
|
1978
|
+
const headerConsistency = await readExcelHeaderConsistency(input);
|
|
1979
|
+
if (!requested && input.dataSets.length > 1) {
|
|
1980
|
+
return {
|
|
1981
|
+
version: "ae-local-data-profile/v1",
|
|
1982
|
+
selection_required: true,
|
|
1983
|
+
source: { format: input.format, size_bytes: input.sizeBytes, sha256: input.sha256 },
|
|
1984
|
+
data_sets: input.dataSets,
|
|
1985
|
+
...headerConsistency ?? {},
|
|
1986
|
+
next_step: "Run inspect again with --data-set, then review the recommended mapping."
|
|
1987
|
+
};
|
|
1988
|
+
}
|
|
1989
|
+
const dataSet = selectDataSet(input, requested);
|
|
1990
|
+
const headerPresence = headerNames || noHeader ? void 0 : detectHeaderPresence(input);
|
|
1991
|
+
const profile = await profileLocalData(input, dataSet, sourceTimezone, {
|
|
1992
|
+
collectSamples: true,
|
|
1993
|
+
collectNestedTree: true,
|
|
1994
|
+
headerNames,
|
|
1995
|
+
noHeader: noHeader || Boolean(headerPresence)
|
|
1996
|
+
});
|
|
1997
|
+
const annotated = headerPresence ? annotateHeaderless(profile, headerPresence) : profile;
|
|
1998
|
+
return headerConsistency ? { ...annotated, ...headerConsistency } : annotated;
|
|
1999
|
+
}
|
|
2000
|
+
const files = [];
|
|
2001
|
+
for (const inputFile of inputFiles) {
|
|
2002
|
+
const input = await inspectLocalDataInput(inputFile);
|
|
2003
|
+
const dataSet = selectDataSet(input);
|
|
2004
|
+
const headerPresence = headerNames || noHeader ? void 0 : detectHeaderPresence(input);
|
|
2005
|
+
const profile = await profileLocalData(input, dataSet, sourceTimezone, {
|
|
2006
|
+
collectSamples: true,
|
|
2007
|
+
collectNestedTree: true,
|
|
2008
|
+
headerNames,
|
|
2009
|
+
noHeader: noHeader || Boolean(headerPresence)
|
|
2010
|
+
});
|
|
2011
|
+
const annotated = headerPresence ? annotateHeaderless(profile, headerPresence) : profile;
|
|
2012
|
+
const headerConsistency = await readExcelHeaderConsistency(input);
|
|
2013
|
+
files.push({
|
|
2014
|
+
file: basename3(inputFile),
|
|
2015
|
+
profile: headerConsistency ? { ...annotated, ...headerConsistency } : annotated
|
|
2016
|
+
});
|
|
2017
|
+
}
|
|
2018
|
+
return {
|
|
2019
|
+
version: "ae-local-data-profile/v1",
|
|
2020
|
+
files: files.map((entry) => entry.profile),
|
|
2021
|
+
conflicts: detectColumnTypeConflicts(files),
|
|
2022
|
+
column_union: buildColumnUnion(files)
|
|
2023
|
+
};
|
|
2024
|
+
}
|
|
2025
|
+
};
|
|
2026
|
+
async function readExcelHeaderConsistency(input) {
|
|
2027
|
+
if (input.format !== "xls" && input.format !== "xlsx") return void 0;
|
|
2028
|
+
return summarizeHeaderConsistency(await readExcelSheetHeaders(input.filePath, input.format));
|
|
2029
|
+
}
|
|
2030
|
+
function summarizeHeaderConsistency(sheets) {
|
|
2031
|
+
if (sheets.length <= 1) return void 0;
|
|
2032
|
+
const first = JSON.stringify(sheets[0].headers);
|
|
2033
|
+
const allSame = sheets.every((sheet) => JSON.stringify(sheet.headers) === first);
|
|
2034
|
+
return allSame ? { header_consistency: "all_same" } : { header_consistency: "different", header_details: sheets.map((sheet) => ({ name: sheet.name, headers: sheet.headers })) };
|
|
2035
|
+
}
|
|
2036
|
+
function detectHeaderPresence(input) {
|
|
2037
|
+
if (input.format !== "csv" && input.format !== "tsv") return void 0;
|
|
2038
|
+
const delimiter = input.delimiter ?? (input.format === "tsv" ? " " : ",");
|
|
2039
|
+
const records = peekDelimitedRecords(input.filePath, { delimiter, encoding: input.encoding, limit: 10 });
|
|
2040
|
+
const detection = detectHeaderRow(records);
|
|
2041
|
+
if (detection.hasHeaders) return void 0;
|
|
2042
|
+
return {
|
|
2043
|
+
detection,
|
|
2044
|
+
autoHeaders: (records[0] ?? []).map((_, index) => `col_${index + 1}`)
|
|
2045
|
+
};
|
|
2046
|
+
}
|
|
2047
|
+
function annotateHeaderless(profile, presence) {
|
|
2048
|
+
return {
|
|
2049
|
+
...profile,
|
|
2050
|
+
no_headers: true,
|
|
2051
|
+
header_detection: presence.detection,
|
|
2052
|
+
auto_headers: presence.autoHeaders,
|
|
2053
|
+
warnings: [...profile.warnings, HEADERLESS_WARNING]
|
|
2054
|
+
};
|
|
2055
|
+
}
|
|
2056
|
+
function splitHeaders(raw) {
|
|
2057
|
+
const value = raw.trim();
|
|
2058
|
+
if (!value) return void 0;
|
|
2059
|
+
const headers = value.split(",").map((header) => header.trim()).filter((header) => header.length > 0);
|
|
2060
|
+
return headers.length > 0 ? headers : void 0;
|
|
2061
|
+
}
|
|
2062
|
+
|
|
2063
|
+
// src/commands/data-integration/local-data/plan.ts
|
|
2064
|
+
import { writeFile } from "fs/promises";
|
|
2065
|
+
var VALID_LOCALES = /* @__PURE__ */ new Set(["zh", "en", "ja", "ko"]);
|
|
2066
|
+
var EVENT_NAME_RE = /^[a-z][a-z0-9_]*$/;
|
|
2067
|
+
function toPropType(type) {
|
|
2068
|
+
switch (type) {
|
|
2069
|
+
case "number":
|
|
2070
|
+
return "number";
|
|
2071
|
+
case "boolean":
|
|
2072
|
+
return "bool";
|
|
2073
|
+
case "datetime":
|
|
2074
|
+
return "datetime";
|
|
2075
|
+
case "list":
|
|
2076
|
+
return "array_string";
|
|
2077
|
+
case "object":
|
|
2078
|
+
return "object";
|
|
2079
|
+
case "string":
|
|
2080
|
+
return "string";
|
|
2081
|
+
}
|
|
2082
|
+
}
|
|
2083
|
+
function buildDraftFromMapping(options) {
|
|
2084
|
+
const { mapping } = options;
|
|
2085
|
+
const excluded = new Set(mapping.exclude_columns ?? []);
|
|
2086
|
+
const properties = mapping.properties.filter((property) => !excluded.has(property.source)).map((property) => ({
|
|
2087
|
+
name: property.target,
|
|
2088
|
+
display_name: property.source,
|
|
2089
|
+
desc: property.desc ?? property.source,
|
|
2090
|
+
type: toPropType(property.type),
|
|
2091
|
+
source: "data"
|
|
2092
|
+
}));
|
|
2093
|
+
const propNames = properties.map((property) => property.name);
|
|
2094
|
+
const eventNames = resolveEventNames(options);
|
|
2095
|
+
const events = eventNames.map((eventName) => {
|
|
2096
|
+
const meta = mapping.event_meta?.[eventName];
|
|
2097
|
+
const sourceName = reverseEventSourceName(mapping, eventName) ?? eventName;
|
|
2098
|
+
return {
|
|
2099
|
+
event_name: eventName,
|
|
2100
|
+
display_name: eventName,
|
|
2101
|
+
event_desc: meta?.desc ?? sourceName,
|
|
2102
|
+
event_tag: meta?.tag ?? sourceName,
|
|
2103
|
+
source: "data",
|
|
2104
|
+
prop_names: propNames
|
|
2105
|
+
};
|
|
2106
|
+
});
|
|
2107
|
+
const isUserSet = mapping.mode === "user_set";
|
|
2108
|
+
const isMixed = mapping.mode === "mixed";
|
|
2109
|
+
return {
|
|
2110
|
+
meta: {
|
|
2111
|
+
app_type: options.appType,
|
|
2112
|
+
sdk_integration_mode: "none",
|
|
2113
|
+
plan_name: options.planName,
|
|
2114
|
+
source_type: "data",
|
|
2115
|
+
lang: options.lang,
|
|
2116
|
+
...options.projectId !== void 0 ? { project_id: options.projectId } : {}
|
|
2117
|
+
},
|
|
2118
|
+
events: isUserSet ? [] : events,
|
|
2119
|
+
event_properties: isUserSet ? [] : properties,
|
|
2120
|
+
common_event_properties: [],
|
|
2121
|
+
user_properties: isUserSet || isMixed ? properties.map((property) => ({ ...property, update_type: "user_set" })) : []
|
|
2122
|
+
};
|
|
2123
|
+
}
|
|
2124
|
+
function reverseEventSourceName(mapping, eventName) {
|
|
2125
|
+
const eventMap = mapping.value_mapping?.event_name;
|
|
2126
|
+
if (!eventMap) return void 0;
|
|
2127
|
+
return Object.keys(eventMap).find((source) => eventMap[source] === eventName);
|
|
2128
|
+
}
|
|
2129
|
+
function resolveEventNames(options) {
|
|
2130
|
+
const { mapping } = options;
|
|
2131
|
+
if (mapping.mode === "user_set") return [];
|
|
2132
|
+
if (options.eventNames.length > 0) {
|
|
2133
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2134
|
+
for (const name of options.eventNames) {
|
|
2135
|
+
if (!EVENT_NAME_RE.test(name)) {
|
|
2136
|
+
throw new CliValidationError(`The event name "${name}" is not a legal AE event name.`, {
|
|
2137
|
+
code: "LOCAL_DATA_PLAN_INVALID_EVENT_NAME",
|
|
2138
|
+
hint: "Event names must match ^[a-z][a-z0-9_]*$.",
|
|
2139
|
+
location: { field: "event-name" }
|
|
2140
|
+
});
|
|
2141
|
+
}
|
|
2142
|
+
seen.add(name);
|
|
2143
|
+
}
|
|
2144
|
+
return [...seen];
|
|
2145
|
+
}
|
|
2146
|
+
if (mapping.default_event_name) return [mapping.default_event_name];
|
|
2147
|
+
throw new CliValidationError("The mapping derives event names from a per-row column.", {
|
|
2148
|
+
code: "LOCAL_DATA_PLAN_EVENT_NAMES_REQUIRED",
|
|
2149
|
+
hint: "Pass --event-name for each concrete event name, or set default_event_name in the mapping.",
|
|
2150
|
+
location: { field: "event-name" }
|
|
2151
|
+
});
|
|
2152
|
+
}
|
|
2153
|
+
function summarize(draft) {
|
|
2154
|
+
return {
|
|
2155
|
+
plan_name: draft.meta.plan_name,
|
|
2156
|
+
events: draft.events.length,
|
|
2157
|
+
event_properties: draft.event_properties.length,
|
|
2158
|
+
user_properties: draft.user_properties.length
|
|
2159
|
+
};
|
|
2160
|
+
}
|
|
2161
|
+
function buildPlanDraft(ctx) {
|
|
2162
|
+
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
2163
|
+
const lang = ctx.str("lang");
|
|
2164
|
+
if (!VALID_LOCALES.has(lang)) {
|
|
2165
|
+
throw new CliValidationError("lang must be one of zh, en, ja, ko.", {
|
|
2166
|
+
code: "LOCAL_DATA_PLAN_INVALID_LANG",
|
|
2167
|
+
location: { field: "lang" }
|
|
2168
|
+
});
|
|
2169
|
+
}
|
|
2170
|
+
return buildDraftFromMapping({
|
|
2171
|
+
mapping,
|
|
2172
|
+
planName: ctx.str("plan-name").trim() || mapping.default_event_name || "local-data",
|
|
2173
|
+
eventNames: ctx.list("event-name"),
|
|
2174
|
+
appType: ctx.str("app-type").trim() || "unknown",
|
|
2175
|
+
lang,
|
|
2176
|
+
projectId: ctx.optionalNum("project-id")
|
|
2177
|
+
});
|
|
2178
|
+
}
|
|
2179
|
+
var dataIntegrationPlan = {
|
|
2180
|
+
service: "data-integration",
|
|
2181
|
+
command: "plan",
|
|
2182
|
+
usesAeHost: false,
|
|
2183
|
+
description: "Convert a confirmed local-data mapping into a tracking-plan draft.json (source_type=data, sdk_integration_mode=none).",
|
|
2184
|
+
flags: [
|
|
2185
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: "Confirmed ae-local-data-mapping/v1 JSON, file path, or @file." },
|
|
2186
|
+
{ name: "event-name", type: "string", variadic: true, desc: "Concrete event name (track/mixed without default_event_name). Repeat for multiple events." },
|
|
2187
|
+
{ name: "plan-name", type: "string", desc: 'Plan name. Default: the mapping default_event_name, else "local-data".' },
|
|
2188
|
+
{ name: "app-type", type: "string", default: "unknown", desc: "Application type recorded in draft meta (informational)." },
|
|
2189
|
+
{ name: "lang", type: "string", default: "zh", desc: "xlsx output language: zh / en / ja / ko." },
|
|
2190
|
+
{ name: "project-id", type: "number", desc: "AE project ID recorded in draft meta (informational)." },
|
|
2191
|
+
{ name: "out", type: "string", sensitive: true, desc: "Write draft.json to this file instead of stdout." }
|
|
2192
|
+
],
|
|
2193
|
+
risk: "write",
|
|
2194
|
+
dryRun: async (ctx) => {
|
|
2195
|
+
const draft = buildPlanDraft(ctx);
|
|
2196
|
+
return {
|
|
2197
|
+
action: "plan_local_data",
|
|
2198
|
+
plan_name: draft.meta.plan_name,
|
|
2199
|
+
sdk_integration_mode: draft.meta.sdk_integration_mode,
|
|
2200
|
+
source_type: draft.meta.source_type,
|
|
2201
|
+
events: draft.events.map((event) => event.event_name),
|
|
2202
|
+
event_properties: draft.events[0]?.prop_names ?? [],
|
|
2203
|
+
user_properties: draft.user_properties.map((property) => property.name)
|
|
2204
|
+
};
|
|
2205
|
+
},
|
|
2206
|
+
execute: async (ctx) => {
|
|
2207
|
+
const draft = buildPlanDraft(ctx);
|
|
2208
|
+
const outPath = ctx.str("out").trim() || void 0;
|
|
2209
|
+
if (!outPath) return draft;
|
|
2210
|
+
await writeFile(outPath, `${JSON.stringify(draft, null, 2)}
|
|
2211
|
+
`, "utf8");
|
|
2212
|
+
return { draft_path: outPath, ...summarize(draft) };
|
|
2213
|
+
}
|
|
2214
|
+
};
|
|
2215
|
+
|
|
2216
|
+
// src/commands/data-integration/local-data/conversion.ts
|
|
2217
|
+
import { randomInt as randomInt2, randomUUID } from "crypto";
|
|
2218
|
+
import {
|
|
2219
|
+
chmodSync,
|
|
2220
|
+
createReadStream as createReadStream3,
|
|
2221
|
+
createWriteStream,
|
|
2222
|
+
existsSync,
|
|
2223
|
+
mkdirSync,
|
|
2224
|
+
readdirSync,
|
|
2225
|
+
readFileSync as readFileSync2,
|
|
2226
|
+
renameSync,
|
|
2227
|
+
statSync as statSync2,
|
|
2228
|
+
unlinkSync,
|
|
2229
|
+
writeFileSync
|
|
2230
|
+
} from "fs";
|
|
2231
|
+
import { once } from "events";
|
|
2232
|
+
import { basename as basename4, join, resolve } from "path";
|
|
2233
|
+
import { createInterface as createInterface2 } from "readline";
|
|
2234
|
+
var SORT_CHUNK_SIZE = 1e4;
|
|
2235
|
+
var THREE_YEARS_MS = 3 * 365 * 24 * 60 * 60 * 1e3;
|
|
2236
|
+
var THREE_DAYS_MS = 3 * 24 * 60 * 60 * 1e3;
|
|
2237
|
+
function stripQuotes(value) {
|
|
2238
|
+
if (typeof value !== "string") return value;
|
|
2239
|
+
const text = value;
|
|
2240
|
+
if (text.length >= 2 && (text.startsWith('"') && text.endsWith('"') || text.startsWith("'") && text.endsWith("'"))) {
|
|
2241
|
+
return text.slice(1, -1).trim();
|
|
2242
|
+
}
|
|
2243
|
+
return text.trim();
|
|
2244
|
+
}
|
|
2245
|
+
async function convertLocalData(options) {
|
|
2246
|
+
const input = await inspectLocalDataInput(options.inputFile);
|
|
2247
|
+
if (input.format !== options.mapping.source.format) {
|
|
2248
|
+
throw new CliValidationError("The source file format does not match the mapping.", {
|
|
2249
|
+
code: "LOCAL_DATA_SOURCE_FORMAT_CHANGED",
|
|
2250
|
+
location: { field: "input-file" }
|
|
2251
|
+
});
|
|
2252
|
+
}
|
|
2253
|
+
if (input.sha256 !== options.mapping.source.sha256) {
|
|
2254
|
+
throw new CliValidationError("The source file does not match the mapping fingerprint.", {
|
|
2255
|
+
code: "LOCAL_DATA_SOURCE_CHANGED",
|
|
2256
|
+
hint: "Run ae-cli data-integration inspect again and review the new mapping before conversion.",
|
|
2257
|
+
location: { field: "input-file" }
|
|
2258
|
+
});
|
|
2259
|
+
}
|
|
2260
|
+
const dataSet = options.mergeSheets ? { id: "all-sheets", kind: "sheet", label: "all-sheets", selector: "all-sheets" } : selectDataSet(input, options.mapping.source.data_set);
|
|
2261
|
+
const streamOptions = {
|
|
2262
|
+
headerNames: options.mapping.headers,
|
|
2263
|
+
flattenRules: options.mapping.flatten_rules,
|
|
2264
|
+
mergeSheets: options.mergeSheets
|
|
2265
|
+
};
|
|
2266
|
+
const salvageSet = options.salvageFrom ? readSalvageRowNumbers(options.salvageFrom) : void 0;
|
|
2267
|
+
let salvageMatched = 0;
|
|
2268
|
+
const runId = `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`;
|
|
2269
|
+
const outputDir = resolve(options.outputDir || join(".ae-cli", "data-integration", runId));
|
|
2270
|
+
prepareOutputDirectory(outputDir);
|
|
2271
|
+
const profile = await profileLocalData(input, dataSet, options.mapping.time.source_timezone, streamOptions);
|
|
2272
|
+
const profilePath = join(outputDir, "profile.json");
|
|
2273
|
+
const mappingPath = join(outputDir, "mapping.json");
|
|
2274
|
+
const validPath = join(outputDir, "valid.ue.jsonl");
|
|
2275
|
+
const invalidPath = join(outputDir, "invalid.rows.jsonl");
|
|
2276
|
+
const manifestPath = join(outputDir, "manifest.json");
|
|
2277
|
+
const transformPath = join(outputDir, "transform.mjs");
|
|
2278
|
+
const trackTempPath = join(outputDir, ".track.tmp.jsonl");
|
|
2279
|
+
const invalidStream = secureWriteStream(invalidPath);
|
|
2280
|
+
const trackStream = secureWriteStream(trackTempPath);
|
|
2281
|
+
const userSetBuffer = [];
|
|
2282
|
+
const sortChunks = [];
|
|
2283
|
+
let validRecords = 0;
|
|
2284
|
+
let invalidRecords = 0;
|
|
2285
|
+
const recordTypes = {};
|
|
2286
|
+
await streamLocalDataRows(
|
|
2287
|
+
input,
|
|
2288
|
+
dataSet,
|
|
2289
|
+
async (row, rowNumber) => {
|
|
2290
|
+
if (salvageSet && !salvageSet.has(rowNumber)) return;
|
|
2291
|
+
if (salvageSet) salvageMatched += 1;
|
|
2292
|
+
const result = convertRow(row, rowNumber, options.mapping, options.now ?? /* @__PURE__ */ new Date());
|
|
2293
|
+
if (!result.ok) {
|
|
2294
|
+
invalidRecords += 1;
|
|
2295
|
+
await writeLine(invalidStream, JSON.stringify({ row_number: rowNumber, errors: result.errors, row }));
|
|
2296
|
+
return;
|
|
2297
|
+
}
|
|
2298
|
+
validRecords += 1;
|
|
2299
|
+
recordTypes[result.recordType] = (recordTypes[result.recordType] ?? 0) + 1;
|
|
2300
|
+
const line = JSON.stringify(result.record);
|
|
2301
|
+
if (isUserProfileType(result.recordType)) {
|
|
2302
|
+
userSetBuffer.push({ key: result.sortKey, line });
|
|
2303
|
+
if (userSetBuffer.length >= SORT_CHUNK_SIZE) flushSortChunk(outputDir, userSetBuffer, sortChunks);
|
|
2304
|
+
} else {
|
|
2305
|
+
await writeLine(trackStream, line);
|
|
2306
|
+
}
|
|
2307
|
+
},
|
|
2308
|
+
{
|
|
2309
|
+
headerNames: streamOptions.headerNames,
|
|
2310
|
+
flattenRules: streamOptions.flattenRules,
|
|
2311
|
+
mergeSheets: streamOptions.mergeSheets
|
|
2312
|
+
}
|
|
2313
|
+
);
|
|
2314
|
+
if (userSetBuffer.length > 0) flushSortChunk(outputDir, userSetBuffer, sortChunks);
|
|
2315
|
+
trackStream.end();
|
|
2316
|
+
invalidStream.end();
|
|
2317
|
+
await Promise.all([once(trackStream, "finish"), once(invalidStream, "finish")]);
|
|
2318
|
+
if (salvageSet && salvageMatched === 0) {
|
|
2319
|
+
throw new CliValidationError("The salvage file lists no rows from this source.", {
|
|
2320
|
+
code: "LOCAL_DATA_SALVAGE_NO_MATCH",
|
|
2321
|
+
hint: "Pass the invalid.rows.jsonl produced by converting the same source file.",
|
|
2322
|
+
location: { field: "salvage-from" }
|
|
2323
|
+
});
|
|
2324
|
+
}
|
|
2325
|
+
const validStream = secureWriteStream(validPath);
|
|
2326
|
+
if (existsSync(trackTempPath)) {
|
|
2327
|
+
for await (const chunk of createReadStream3(trackTempPath)) {
|
|
2328
|
+
if (!validStream.write(chunk)) await once(validStream, "drain");
|
|
2329
|
+
}
|
|
2330
|
+
}
|
|
2331
|
+
await mergeSortChunks(sortChunks, validStream);
|
|
2332
|
+
validStream.end();
|
|
2333
|
+
await once(validStream, "finish");
|
|
2334
|
+
if (existsSync(trackTempPath)) unlinkSync(trackTempPath);
|
|
2335
|
+
for (const path of sortChunks) if (existsSync(path)) unlinkSync(path);
|
|
2336
|
+
writeSecureJson(profilePath, profile);
|
|
2337
|
+
writeSecureJson(mappingPath, options.mapping);
|
|
2338
|
+
writeSecureText(transformPath, createTransformScript(options.inputFile, mappingPath));
|
|
2339
|
+
const validBytes = statSize(validPath);
|
|
2340
|
+
const blockedReasons = [
|
|
2341
|
+
...invalidRecords > 0 ? ["Some source rows failed UE validation."] : [],
|
|
2342
|
+
...validRecords === 0 ? ["No valid UE records were generated."] : []
|
|
2343
|
+
];
|
|
2344
|
+
const manifest = {
|
|
2345
|
+
version: "ae-local-data-manifest/v1",
|
|
2346
|
+
run_id: runId,
|
|
2347
|
+
created_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
2348
|
+
status: blockedReasons.length > 0 ? "blocked" : "ready",
|
|
2349
|
+
...options.salvageFrom ? { salvage_from: basename4(options.salvageFrom) } : {},
|
|
2350
|
+
source: {
|
|
2351
|
+
sha256: input.sha256,
|
|
2352
|
+
format: input.format,
|
|
2353
|
+
data_set: dataSet.id,
|
|
2354
|
+
size_bytes: input.sizeBytes
|
|
2355
|
+
},
|
|
2356
|
+
output: {
|
|
2357
|
+
valid_file: basename4(validPath),
|
|
2358
|
+
valid_sha256: await sha256File(validPath),
|
|
2359
|
+
invalid_file: basename4(invalidPath),
|
|
2360
|
+
valid_records: validRecords,
|
|
2361
|
+
invalid_records: invalidRecords,
|
|
2362
|
+
valid_bytes: validBytes,
|
|
2363
|
+
record_types: recordTypes
|
|
2364
|
+
},
|
|
2365
|
+
blocked_reasons: blockedReasons
|
|
2366
|
+
};
|
|
2367
|
+
writeSecureJson(`${manifestPath}.tmp`, manifest);
|
|
2368
|
+
renameSync(`${manifestPath}.tmp`, manifestPath);
|
|
2369
|
+
chmodSync(manifestPath, 384);
|
|
2370
|
+
return { status: manifest.status, output_dir: outputDir, manifest };
|
|
2371
|
+
}
|
|
2372
|
+
async function convertLocalDataMulti(options) {
|
|
2373
|
+
const sourceTimezone = options.mapping.time.source_timezone;
|
|
2374
|
+
const profiled = [];
|
|
2375
|
+
for (const inputFile of options.inputFiles) {
|
|
2376
|
+
const input = await inspectLocalDataInput(inputFile);
|
|
2377
|
+
const dataSet = selectDataSet(input);
|
|
2378
|
+
const profile = await profileLocalData(input, dataSet, sourceTimezone, {
|
|
2379
|
+
collectSamples: true,
|
|
2380
|
+
headerNames: options.mapping.headers,
|
|
2381
|
+
flattenRules: options.mapping.flatten_rules
|
|
2382
|
+
});
|
|
2383
|
+
profiled.push({ file: basename4(inputFile), profile });
|
|
2384
|
+
}
|
|
2385
|
+
const conflicts = detectColumnTypeConflicts(profiled);
|
|
2386
|
+
const resolutions = options.typeResolutions ?? {};
|
|
2387
|
+
if (conflicts.length > 0 && Object.keys(resolutions).length === 0) {
|
|
2388
|
+
throw new CliValidationError("Cross-file column type conflicts require explicit resolutions.", {
|
|
2389
|
+
code: "LOCAL_DATA_TYPE_CONFLICTS_UNRESOLVED",
|
|
2390
|
+
hint: `Provide --type-resolutions for: ${conflicts.map((conflict) => conflict.column).join(", ")}`,
|
|
2391
|
+
location: { field: "type-resolutions" }
|
|
2392
|
+
});
|
|
2393
|
+
}
|
|
2394
|
+
validateTypeResolutions(resolutions, profiled.map((entry) => entry.file));
|
|
2395
|
+
const overrides = applyTypeResolutions(resolutions, profiled);
|
|
2396
|
+
const parent = resolve(
|
|
2397
|
+
options.outputDir || join(".ae-cli", "data-integration", `${formatRunTimestamp(/* @__PURE__ */ new Date())}-${randomUUID().slice(0, 8)}`)
|
|
2398
|
+
);
|
|
2399
|
+
prepareOutputDirectory(parent);
|
|
2400
|
+
const files = [];
|
|
2401
|
+
let blocked = false;
|
|
2402
|
+
for (let index = 0; index < profiled.length; index += 1) {
|
|
2403
|
+
const entry = profiled[index];
|
|
2404
|
+
const perFileMapping = buildPerFileMapping(options.mapping, entry.profile, overrides.get(entry.file));
|
|
2405
|
+
const result = await convertLocalData({
|
|
2406
|
+
inputFile: options.inputFiles[index],
|
|
2407
|
+
mapping: perFileMapping,
|
|
2408
|
+
outputDir: join(parent, `${String(index).padStart(2, "0")}-${entry.file}`),
|
|
2409
|
+
now: options.now
|
|
2410
|
+
});
|
|
2411
|
+
if (result.status === "blocked") blocked = true;
|
|
2412
|
+
files.push({
|
|
2413
|
+
file: entry.file,
|
|
2414
|
+
status: result.status,
|
|
2415
|
+
output_dir: result.output_dir,
|
|
2416
|
+
manifest: result.manifest
|
|
2417
|
+
});
|
|
2418
|
+
}
|
|
2419
|
+
return { status: blocked ? "blocked" : "ready", output_dir: parent, files };
|
|
2420
|
+
}
|
|
2421
|
+
function convertRow(row, rowNumber, mapping, now) {
|
|
2422
|
+
const errors = [];
|
|
2423
|
+
const accountId = resolveIdentity(row, mapping.account_id_field, mapping.value_mapping?.account_id, mapping.account_id_value, mapping.random_pool?.account_ids, errors);
|
|
2424
|
+
const distinctId = resolveIdentity(row, mapping.distinct_id_field, mapping.value_mapping?.distinct_id, mapping.distinct_id_value, mapping.random_pool?.distinct_ids, errors);
|
|
2425
|
+
if (!accountId && !distinctId) errors.push({ code: "MISSING_USER_ID" });
|
|
2426
|
+
let recordType;
|
|
2427
|
+
if (mapping.mode === "mixed") {
|
|
2428
|
+
const rawType = stripQuotes(row[mapping.record_type_field]);
|
|
2429
|
+
const mappedType = mapping.value_mapping?.record_type && typeof rawType === "string" ? mapping.value_mapping.record_type[rawType] ?? rawType : rawType;
|
|
2430
|
+
recordType = normalizeRecordType(mappedType);
|
|
2431
|
+
if (!recordType) errors.push({ code: "INVALID_RECORD_TYPE", field: mapping.record_type_field });
|
|
2432
|
+
} else {
|
|
2433
|
+
recordType = mapping.mode;
|
|
2434
|
+
}
|
|
2435
|
+
const timeValue = row[mapping.time.field];
|
|
2436
|
+
let normalizedTime;
|
|
2437
|
+
if (isMissing2(timeValue) && mapping.missing_time === "now" && recordType !== void 0 && isUserProfileType(recordType)) {
|
|
2438
|
+
normalizedTime = { instant: now, formatted: formatInTimeZone(now, mapping.time.source_timezone) };
|
|
2439
|
+
} else {
|
|
2440
|
+
normalizedTime = normalizeTime(timeValue, mapping.time.source_timezone, mapping.time_format);
|
|
2441
|
+
if (!normalizedTime) {
|
|
2442
|
+
errors.push({ code: "INVALID_TIME", field: mapping.time.field });
|
|
2443
|
+
} else if (normalizedTime.instant.getTime() < now.getTime() - THREE_YEARS_MS || normalizedTime.instant.getTime() > now.getTime() + THREE_DAYS_MS) {
|
|
2444
|
+
errors.push({ code: "TIME_OUT_OF_RANGE", field: mapping.time.field });
|
|
2445
|
+
}
|
|
2446
|
+
}
|
|
2447
|
+
let eventName;
|
|
2448
|
+
if (recordType === "track") {
|
|
2449
|
+
const rawEvent = String(stripQuotes(mapping.event_name_field ? row[mapping.event_name_field] ?? "" : mapping.default_event_name ?? ""));
|
|
2450
|
+
eventName = mapping.value_mapping?.event_name && mapping.value_mapping.event_name[rawEvent] !== void 0 ? mapping.value_mapping.event_name[rawEvent] : rawEvent;
|
|
2451
|
+
if (!isValidAeName(eventName)) {
|
|
2452
|
+
errors.push({ code: "INVALID_EVENT_NAME", field: mapping.event_name_field });
|
|
2453
|
+
}
|
|
2454
|
+
}
|
|
2455
|
+
const ip = readOptionalField(row, mapping.ip_field);
|
|
2456
|
+
const uuid = readOptionalField(row, mapping.uuid_field);
|
|
2457
|
+
const excluded = new Set(mapping.exclude_columns ?? []);
|
|
2458
|
+
const properties = {};
|
|
2459
|
+
for (const property of mapping.properties) {
|
|
2460
|
+
if (excluded.has(property.source)) continue;
|
|
2461
|
+
let value = stripQuotes(row[property.source]);
|
|
2462
|
+
if (isMissing2(value)) continue;
|
|
2463
|
+
if (property.value_mapping && typeof value === "string" && property.value_mapping[value] !== void 0) {
|
|
2464
|
+
value = property.value_mapping[value];
|
|
2465
|
+
}
|
|
2466
|
+
const converted = convertProperty(value, property.type, property.transform, mapping.time.source_timezone, property.time_format);
|
|
2467
|
+
if (!converted.ok) {
|
|
2468
|
+
errors.push({ code: converted.code, field: property.source });
|
|
2469
|
+
} else {
|
|
2470
|
+
properties[property.target] = converted.value;
|
|
2471
|
+
}
|
|
2472
|
+
}
|
|
2473
|
+
const zoneOffset = mapping.zone_offset_value !== void 0 ? resolveZoneOffsetValue(mapping.zone_offset_value, now) : mapping.zone_offset_field ? readZoneOffset(row, mapping.zone_offset_field) : void 0;
|
|
2474
|
+
if (mapping.zone_offset_field && zoneOffset === void 0) {
|
|
2475
|
+
errors.push({ code: "INVALID_ZONE_OFFSET", field: mapping.zone_offset_field });
|
|
2476
|
+
}
|
|
2477
|
+
if (errors.length > 0 || !recordType || !normalizedTime) return { ok: false, errors };
|
|
2478
|
+
const record = {
|
|
2479
|
+
"#type": recordType,
|
|
2480
|
+
"#time": normalizedTime.formatted,
|
|
2481
|
+
...accountId ? { "#account_id": accountId } : {},
|
|
2482
|
+
...distinctId ? { "#distinct_id": distinctId } : {},
|
|
2483
|
+
...eventName ? { "#event_name": eventName } : {},
|
|
2484
|
+
...ip ? { "#ip": ip } : {},
|
|
2485
|
+
...uuid ? { "#uuid": uuid } : {},
|
|
2486
|
+
properties: { ...properties, ...zoneOffset !== void 0 ? { "#zone_offset": zoneOffset } : {} }
|
|
2487
|
+
};
|
|
2488
|
+
return {
|
|
2489
|
+
ok: true,
|
|
2490
|
+
recordType,
|
|
2491
|
+
record,
|
|
2492
|
+
sortKey: `${accountId ?? distinctId ?? ""}\0${normalizedTime.instant.toISOString()}\0${String(rowNumber).padStart(12, "0")}`
|
|
2493
|
+
};
|
|
2494
|
+
}
|
|
2495
|
+
function readOptionalField(row, field) {
|
|
2496
|
+
if (!field) return void 0;
|
|
2497
|
+
const value = stripQuotes(row[field]);
|
|
2498
|
+
if (value === null || value === void 0) return void 0;
|
|
2499
|
+
const text = String(value).trim();
|
|
2500
|
+
return text.length > 0 ? text : void 0;
|
|
2501
|
+
}
|
|
2502
|
+
function resolveZoneOffsetValue(value, now) {
|
|
2503
|
+
if (typeof value === "number") return value;
|
|
2504
|
+
const dtf = new Intl.DateTimeFormat("en-US", { timeZone: value, timeZoneName: "longOffset" });
|
|
2505
|
+
const offsetName = dtf.formatToParts(now).find((part) => part.type === "timeZoneName")?.value;
|
|
2506
|
+
if (!offsetName || offsetName === "GMT") return 0;
|
|
2507
|
+
const match = /^GMT([+-])(\d{1,2}):?(\d{2})?$/.exec(offsetName);
|
|
2508
|
+
if (!match) return 0;
|
|
2509
|
+
const sign = match[1] === "-" ? -1 : 1;
|
|
2510
|
+
return Math.round(sign * (Number(match[2]) + Number(match[3] ?? 0) / 60));
|
|
2511
|
+
}
|
|
2512
|
+
function readZoneOffset(row, field) {
|
|
2513
|
+
const value = stripQuotes(row[field]);
|
|
2514
|
+
if (value === null || value === void 0) return void 0;
|
|
2515
|
+
const text = String(value).trim();
|
|
2516
|
+
if (!text) return void 0;
|
|
2517
|
+
const offset = Number(text);
|
|
2518
|
+
return Number.isInteger(offset) && offset >= -12 && offset <= 14 ? offset : void 0;
|
|
2519
|
+
}
|
|
2520
|
+
function readSalvageRowNumbers(path) {
|
|
2521
|
+
let text;
|
|
2522
|
+
try {
|
|
2523
|
+
text = readFileSync2(path, "utf8");
|
|
2524
|
+
} catch {
|
|
2525
|
+
throw new CliValidationError("The salvage file could not be read.", {
|
|
2526
|
+
code: "LOCAL_DATA_SALVAGE_INVALID",
|
|
2527
|
+
location: { field: "salvage-from" }
|
|
2528
|
+
});
|
|
2529
|
+
}
|
|
2530
|
+
const rows = /* @__PURE__ */ new Set();
|
|
2531
|
+
let lineNumber = 0;
|
|
2532
|
+
for (const line of text.split("\n")) {
|
|
2533
|
+
lineNumber += 1;
|
|
2534
|
+
const trimmed = line.trim();
|
|
2535
|
+
if (!trimmed) continue;
|
|
2536
|
+
let value;
|
|
2537
|
+
try {
|
|
2538
|
+
value = JSON.parse(trimmed);
|
|
2539
|
+
} catch {
|
|
2540
|
+
throw new CliValidationError("The salvage file contains a line that is not valid JSON.", {
|
|
2541
|
+
code: "LOCAL_DATA_SALVAGE_INVALID",
|
|
2542
|
+
location: { field: "salvage-from", record: lineNumber }
|
|
2543
|
+
});
|
|
2544
|
+
}
|
|
2545
|
+
if (typeof value !== "object" || value === null || Array.isArray(value) || !Number.isInteger(value.row_number) || value.row_number <= 0) {
|
|
2546
|
+
throw new CliValidationError("The salvage file is not an invalid.rows.jsonl quarantine file.", {
|
|
2547
|
+
code: "LOCAL_DATA_SALVAGE_INVALID",
|
|
2548
|
+
location: { field: "salvage-from", record: lineNumber }
|
|
2549
|
+
});
|
|
2550
|
+
}
|
|
2551
|
+
rows.add(value.row_number);
|
|
2552
|
+
}
|
|
2553
|
+
if (rows.size === 0) {
|
|
2554
|
+
throw new CliValidationError("The salvage file contains no rows.", {
|
|
2555
|
+
code: "LOCAL_DATA_SALVAGE_EMPTY",
|
|
2556
|
+
location: { field: "salvage-from" }
|
|
2557
|
+
});
|
|
2558
|
+
}
|
|
2559
|
+
return rows;
|
|
2560
|
+
}
|
|
2561
|
+
function readIdentity(row, field, errors) {
|
|
2562
|
+
if (!field || isMissing2(row[field])) return void 0;
|
|
2563
|
+
const value = String(stripQuotes(row[field]));
|
|
2564
|
+
if (value.length > IDENTITY_MAX_LENGTH) {
|
|
2565
|
+
errors.push({ code: "USER_ID_TOO_LONG", field });
|
|
2566
|
+
return void 0;
|
|
2567
|
+
}
|
|
2568
|
+
return value;
|
|
2569
|
+
}
|
|
2570
|
+
function resolveIdentity(row, field, valueMapping, fixedValue, pool, errors) {
|
|
2571
|
+
const raw = readIdentity(row, field, errors);
|
|
2572
|
+
if (raw !== void 0) return valueMapping?.[raw] ?? raw;
|
|
2573
|
+
if (fixedValue) return fixedValue;
|
|
2574
|
+
if (pool && pool.length > 0) return pool[randomInt2(pool.length)];
|
|
2575
|
+
return void 0;
|
|
2576
|
+
}
|
|
2577
|
+
function convertProperty(value, type, transform, timeZone = "UTC", timeFormat) {
|
|
2578
|
+
try {
|
|
2579
|
+
if (transform === "stringify") value = typeof value === "string" ? value : JSON.stringify(value);
|
|
2580
|
+
if (transform === "json" && typeof value === "string") value = JSON.parse(value);
|
|
2581
|
+
if (transform === "number") value = Number(value);
|
|
2582
|
+
if (transform === "boolean") value = parseBoolean(value);
|
|
2583
|
+
if (type === "string") {
|
|
2584
|
+
const text = value instanceof Date ? value.toISOString() : String(value);
|
|
2585
|
+
return isPropertyWithinLimits(text, type) ? { ok: true, value: text } : { ok: false, code: "PROPERTY_LIMIT_EXCEEDED" };
|
|
2586
|
+
}
|
|
2587
|
+
if (type === "number") {
|
|
2588
|
+
const number = typeof value === "number" ? value : Number(value);
|
|
2589
|
+
if (!Number.isFinite(number)) return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2590
|
+
return isPropertyWithinLimits(number, type) ? { ok: true, value: number } : { ok: false, code: "PROPERTY_LIMIT_EXCEEDED" };
|
|
2591
|
+
}
|
|
2592
|
+
if (type === "boolean") {
|
|
2593
|
+
const boolean = typeof value === "boolean" ? value : parseBoolean(value);
|
|
2594
|
+
return typeof boolean === "boolean" ? { ok: true, value: boolean } : { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2595
|
+
}
|
|
2596
|
+
if (type === "datetime") {
|
|
2597
|
+
const normalized = normalizeTime(value, timeZone, timeFormat);
|
|
2598
|
+
return normalized ? { ok: true, value: normalized.formatted } : { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2599
|
+
}
|
|
2600
|
+
if (type === "list") {
|
|
2601
|
+
if (!Array.isArray(value)) return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2602
|
+
return isPropertyWithinLimits(value, type) ? { ok: true, value } : { ok: false, code: "PROPERTY_LIMIT_EXCEEDED" };
|
|
2603
|
+
}
|
|
2604
|
+
if (type === "object") {
|
|
2605
|
+
if (value === null || typeof value !== "object" || Array.isArray(value)) {
|
|
2606
|
+
return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2607
|
+
}
|
|
2608
|
+
return isPropertyWithinLimits(value, type) ? { ok: true, value } : { ok: false, code: "PROPERTY_LIMIT_EXCEEDED" };
|
|
2609
|
+
}
|
|
2610
|
+
} catch {
|
|
2611
|
+
return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2612
|
+
}
|
|
2613
|
+
return { ok: false, code: "PROPERTY_TYPE_CONFLICT" };
|
|
2614
|
+
}
|
|
2615
|
+
function isPropertyWithinLimits(value, expectedType) {
|
|
2616
|
+
if (expectedType === "string") return typeof value === "string" && Buffer.byteLength(value, "utf8") <= 2 * 1024;
|
|
2617
|
+
if (expectedType === "number") return typeof value === "number" && Number.isFinite(value) && value >= -9e15 && value <= 9e15;
|
|
2618
|
+
if (expectedType === "boolean") return typeof value === "boolean";
|
|
2619
|
+
if (expectedType === "datetime") return typeof value === "string";
|
|
2620
|
+
if (expectedType === "object") return isValidObjectProperty(value);
|
|
2621
|
+
if (!Array.isArray(value) || value.length > 500) return false;
|
|
2622
|
+
if (value.every((item) => typeof item === "string")) {
|
|
2623
|
+
return value.every((item) => Buffer.byteLength(item, "utf8") <= 255);
|
|
2624
|
+
}
|
|
2625
|
+
if (value.every((item) => item !== null && typeof item === "object" && !Array.isArray(item))) {
|
|
2626
|
+
return value.every(isValidObjectProperty);
|
|
2627
|
+
}
|
|
2628
|
+
return false;
|
|
2629
|
+
}
|
|
2630
|
+
function isValidObjectProperty(value) {
|
|
2631
|
+
if (value === null || typeof value !== "object" || Array.isArray(value)) return false;
|
|
2632
|
+
const entries = Object.entries(value);
|
|
2633
|
+
if (entries.length > 100) return false;
|
|
2634
|
+
return entries.every(([name, nested]) => isValidAeName(name) && isValidNestedProperty(nested));
|
|
2635
|
+
}
|
|
2636
|
+
function isValidNestedProperty(value) {
|
|
2637
|
+
if (value === null || value === void 0) return true;
|
|
2638
|
+
if (typeof value === "string") return Buffer.byteLength(value, "utf8") <= 2 * 1024;
|
|
2639
|
+
if (typeof value === "number") return Number.isFinite(value) && value >= -9e15 && value <= 9e15;
|
|
2640
|
+
if (typeof value === "boolean") return true;
|
|
2641
|
+
if (Array.isArray(value)) {
|
|
2642
|
+
if (value.length > 500) return false;
|
|
2643
|
+
if (value.every((item) => typeof item === "string")) {
|
|
2644
|
+
return value.every((item) => Buffer.byteLength(item, "utf8") <= 255);
|
|
2645
|
+
}
|
|
2646
|
+
return value.every(isValidObjectProperty);
|
|
2647
|
+
}
|
|
2648
|
+
return isValidObjectProperty(value);
|
|
2649
|
+
}
|
|
2650
|
+
function parseBoolean(value) {
|
|
2651
|
+
if (typeof value === "boolean") return value;
|
|
2652
|
+
const normalized = String(value).trim();
|
|
2653
|
+
const lowered = normalized.toLowerCase();
|
|
2654
|
+
if (["true", "1", "yes", "y", "\u662F"].includes(lowered)) return true;
|
|
2655
|
+
if (["false", "0", "no", "n", "\u5426"].includes(lowered)) return false;
|
|
2656
|
+
return void 0;
|
|
2657
|
+
}
|
|
2658
|
+
function normalizeTime(value, timeZone, timeFormat) {
|
|
2659
|
+
let instant;
|
|
2660
|
+
if (value instanceof Date) {
|
|
2661
|
+
instant = value;
|
|
2662
|
+
} else if (typeof value === "number" && Number.isFinite(value)) {
|
|
2663
|
+
if (value >= 1 && value <= 2958465) {
|
|
2664
|
+
const excelWallTime = new Date(Date.UTC(1899, 11, 30) + value * 24 * 60 * 60 * 1e3);
|
|
2665
|
+
instant = zonedWallTimeToDate({
|
|
2666
|
+
year: excelWallTime.getUTCFullYear(),
|
|
2667
|
+
month: excelWallTime.getUTCMonth() + 1,
|
|
2668
|
+
day: excelWallTime.getUTCDate(),
|
|
2669
|
+
hour: excelWallTime.getUTCHours(),
|
|
2670
|
+
minute: excelWallTime.getUTCMinutes(),
|
|
2671
|
+
second: excelWallTime.getUTCSeconds(),
|
|
2672
|
+
millisecond: excelWallTime.getUTCMilliseconds()
|
|
2673
|
+
}, timeZone);
|
|
2674
|
+
} else {
|
|
2675
|
+
const millis = value > 1e12 ? value : value > 1e9 ? value * 1e3 : Number.NaN;
|
|
2676
|
+
instant = new Date(millis);
|
|
2677
|
+
}
|
|
2678
|
+
} else if (typeof value === "string") {
|
|
2679
|
+
const text = stripQuotes(value);
|
|
2680
|
+
if (/^\d{10}(?:\d{3})?$/.test(text)) {
|
|
2681
|
+
const numeric = Number(text);
|
|
2682
|
+
instant = new Date(text.length === 10 ? numeric * 1e3 : numeric);
|
|
2683
|
+
if (!Number.isFinite(instant.getTime())) return void 0;
|
|
2684
|
+
return { instant, formatted: formatInTimeZone(instant, timeZone) };
|
|
2685
|
+
}
|
|
2686
|
+
if (hasExplicitOffset(text)) {
|
|
2687
|
+
instant = new Date(text.replace(/\s/, "T"));
|
|
2688
|
+
if (!Number.isFinite(instant.getTime())) return void 0;
|
|
2689
|
+
return { instant, formatted: formatInTimeZone(instant, timeZone) };
|
|
2690
|
+
}
|
|
2691
|
+
if (timeFormat) {
|
|
2692
|
+
const parts = tryStrptime(text, timeFormat);
|
|
2693
|
+
if (!parts) return void 0;
|
|
2694
|
+
instant = zonedWallTimeToDate(parts, timeZone);
|
|
2695
|
+
} else {
|
|
2696
|
+
const dashed = text.replace(/\//g, "-");
|
|
2697
|
+
const naive = dashed.match(/^(\d{4})-(\d{1,2})-(\d{1,2})(?:[ T](\d{1,2}):(\d{2})(?::(\d{2})(?:\.(\d{1,3}))?)?)?$/);
|
|
2698
|
+
if (naive) {
|
|
2699
|
+
const parts = {
|
|
2700
|
+
year: Number(naive[1]),
|
|
2701
|
+
month: Number(naive[2]),
|
|
2702
|
+
day: Number(naive[3]),
|
|
2703
|
+
hour: Number(naive[4] ?? 0),
|
|
2704
|
+
minute: Number(naive[5] ?? 0),
|
|
2705
|
+
second: Number(naive[6] ?? 0),
|
|
2706
|
+
millisecond: Number((naive[7] ?? "0").padEnd(3, "0"))
|
|
2707
|
+
};
|
|
2708
|
+
if (!isValidCalendarParts(parts)) return void 0;
|
|
2709
|
+
instant = zonedWallTimeToDate(parts, timeZone);
|
|
2710
|
+
} else {
|
|
2711
|
+
const wallParts = parseTimeByAnyFormat(text);
|
|
2712
|
+
if (wallParts) {
|
|
2713
|
+
instant = zonedWallTimeToDate(wallParts, timeZone);
|
|
2714
|
+
} else {
|
|
2715
|
+
instant = new Date(text);
|
|
2716
|
+
}
|
|
2717
|
+
}
|
|
2718
|
+
}
|
|
2719
|
+
} else {
|
|
2720
|
+
return void 0;
|
|
2721
|
+
}
|
|
2722
|
+
if (!Number.isFinite(instant.getTime())) return void 0;
|
|
2723
|
+
return { instant, formatted: formatInTimeZone(instant, timeZone) };
|
|
2724
|
+
}
|
|
2725
|
+
function hasExplicitOffset(text) {
|
|
2726
|
+
return /(?:Z|[+-]\d{2}:?\d{2})$/.test(text);
|
|
2727
|
+
}
|
|
2728
|
+
function zonedWallTimeToDate(parts, timeZone) {
|
|
2729
|
+
const desired = Date.UTC(parts.year, parts.month - 1, parts.day, parts.hour, parts.minute, parts.second, parts.millisecond);
|
|
2730
|
+
let guess = desired;
|
|
2731
|
+
for (let attempt = 0; attempt < 3; attempt += 1) {
|
|
2732
|
+
const current = datePartsInZone(new Date(guess), timeZone);
|
|
2733
|
+
const represented = Date.UTC(current.year, current.month - 1, current.day, current.hour, current.minute, current.second, parts.millisecond);
|
|
2734
|
+
guess += desired - represented;
|
|
2735
|
+
}
|
|
2736
|
+
return new Date(guess);
|
|
2737
|
+
}
|
|
2738
|
+
function formatInTimeZone(value, timeZone) {
|
|
2739
|
+
const parts = datePartsInZone(value, timeZone);
|
|
2740
|
+
const pad = (number, length = 2) => String(number).padStart(length, "0");
|
|
2741
|
+
return `${pad(parts.year, 4)}-${pad(parts.month)}-${pad(parts.day)} ${pad(parts.hour)}:${pad(parts.minute)}:${pad(parts.second)}.${pad(value.getUTCMilliseconds(), 3)}`;
|
|
2742
|
+
}
|
|
2743
|
+
function datePartsInZone(value, timeZone) {
|
|
2744
|
+
const parts = new Intl.DateTimeFormat("en-US", {
|
|
2745
|
+
timeZone,
|
|
2746
|
+
year: "numeric",
|
|
2747
|
+
month: "2-digit",
|
|
2748
|
+
day: "2-digit",
|
|
2749
|
+
hour: "2-digit",
|
|
2750
|
+
minute: "2-digit",
|
|
2751
|
+
second: "2-digit",
|
|
2752
|
+
hourCycle: "h23"
|
|
2753
|
+
}).formatToParts(value);
|
|
2754
|
+
const read = (type) => Number(parts.find((part) => part.type === type)?.value ?? 0);
|
|
2755
|
+
return { year: read("year"), month: read("month"), day: read("day"), hour: read("hour"), minute: read("minute"), second: read("second") };
|
|
2756
|
+
}
|
|
2757
|
+
function isValidCalendarParts(parts) {
|
|
2758
|
+
if (parts.hour < 0 || parts.hour > 23 || parts.minute < 0 || parts.minute > 59 || parts.second < 0 || parts.second > 59 || parts.millisecond < 0 || parts.millisecond > 999) return false;
|
|
2759
|
+
const date = new Date(Date.UTC(parts.year, parts.month - 1, parts.day));
|
|
2760
|
+
return date.getUTCFullYear() === parts.year && date.getUTCMonth() === parts.month - 1 && date.getUTCDate() === parts.day;
|
|
2761
|
+
}
|
|
2762
|
+
function flushSortChunk(outputDir, buffer, chunks) {
|
|
2763
|
+
buffer.sort((left, right) => left.key.localeCompare(right.key));
|
|
2764
|
+
const path = join(outputDir, `.user-set-${chunks.length}.tmp.jsonl`);
|
|
2765
|
+
writeSecureText(path, `${buffer.map((item) => JSON.stringify(item)).join("\n")}
|
|
2766
|
+
`);
|
|
2767
|
+
chunks.push(path);
|
|
2768
|
+
buffer.length = 0;
|
|
2769
|
+
}
|
|
2770
|
+
async function mergeSortChunks(paths, output) {
|
|
2771
|
+
const cursors = await Promise.all(paths.map(async (path) => {
|
|
2772
|
+
const iterator = createInterface2({ input: createReadStream3(path), crlfDelay: Infinity })[Symbol.asyncIterator]();
|
|
2773
|
+
return { iterator, current: await readSortItem(iterator) };
|
|
2774
|
+
}));
|
|
2775
|
+
while (true) {
|
|
2776
|
+
let selected = -1;
|
|
2777
|
+
for (let index = 0; index < cursors.length; index += 1) {
|
|
2778
|
+
if (!cursors[index].current) continue;
|
|
2779
|
+
if (selected < 0 || cursors[index].current.key < cursors[selected].current.key) selected = index;
|
|
2780
|
+
}
|
|
2781
|
+
if (selected < 0) break;
|
|
2782
|
+
await writeLine(output, cursors[selected].current.line);
|
|
2783
|
+
cursors[selected].current = await readSortItem(cursors[selected].iterator);
|
|
2784
|
+
}
|
|
2785
|
+
}
|
|
2786
|
+
async function readSortItem(iterator) {
|
|
2787
|
+
const next = await iterator.next();
|
|
2788
|
+
if (next.done) return void 0;
|
|
2789
|
+
return JSON.parse(next.value);
|
|
2790
|
+
}
|
|
2791
|
+
function prepareOutputDirectory(path) {
|
|
2792
|
+
if (existsSync(path) && readdirSync(path).length > 0) {
|
|
2793
|
+
throw new CliValidationError("The output directory must be new or empty.", {
|
|
2794
|
+
code: "LOCAL_DATA_OUTPUT_NOT_EMPTY",
|
|
2795
|
+
location: { field: "output-dir" }
|
|
2796
|
+
});
|
|
2797
|
+
}
|
|
2798
|
+
mkdirSync(path, { recursive: true, mode: 448 });
|
|
2799
|
+
chmodSync(path, 448);
|
|
2800
|
+
}
|
|
2801
|
+
function secureWriteStream(path) {
|
|
2802
|
+
return createWriteStream(path, { encoding: "utf8", mode: 384, flags: "wx" });
|
|
2803
|
+
}
|
|
2804
|
+
async function writeLine(stream, line) {
|
|
2805
|
+
if (!stream.write(`${line}
|
|
2806
|
+
`)) await once(stream, "drain");
|
|
2807
|
+
}
|
|
2808
|
+
function writeSecureJson(path, value) {
|
|
2809
|
+
writeSecureText(path, `${JSON.stringify(value, null, 2)}
|
|
2810
|
+
`);
|
|
2811
|
+
}
|
|
2812
|
+
function writeSecureText(path, value) {
|
|
2813
|
+
writeFileSync(path, value, { encoding: "utf8", mode: 384, flag: "wx" });
|
|
2814
|
+
chmodSync(path, 384);
|
|
2815
|
+
}
|
|
2816
|
+
function createTransformScript(inputFile, mappingPath) {
|
|
2817
|
+
return [
|
|
2818
|
+
"import { spawnSync } from 'node:child_process';",
|
|
2819
|
+
"const outputDir = process.argv[2];",
|
|
2820
|
+
"if (!outputDir) { console.error('Usage: node transform.mjs <new-output-directory>'); process.exit(2); }",
|
|
2821
|
+
`const result = spawnSync('ae-cli', ['data-integration', 'convert', '--input-file', ${JSON.stringify(resolve(inputFile))}, '--mapping', ${JSON.stringify(mappingPath)}, '--output-dir', outputDir], { stdio: 'inherit' });`,
|
|
2822
|
+
"process.exit(result.status ?? 1);",
|
|
2823
|
+
""
|
|
2824
|
+
].join("\n");
|
|
2825
|
+
}
|
|
2826
|
+
function statSize(path) {
|
|
2827
|
+
return statSync2(path).size;
|
|
2828
|
+
}
|
|
2829
|
+
function formatRunTimestamp(value) {
|
|
2830
|
+
return value.toISOString().replace(/[-:]/g, "").replace(/\.\d{3}Z$/, "Z");
|
|
2831
|
+
}
|
|
2832
|
+
|
|
2833
|
+
// src/commands/data-integration/local-data/convert.ts
|
|
2834
|
+
var dataIntegrationConvert = {
|
|
2835
|
+
service: "data-integration",
|
|
2836
|
+
command: "convert",
|
|
2837
|
+
usesAeHost: false,
|
|
2838
|
+
description: "Convert one or more local data sets into validated UE JSONL and quarantine invalid rows.",
|
|
2839
|
+
flags: [
|
|
2840
|
+
{ name: "input-file", type: "string", required: true, sensitive: true, variadic: true, desc: "Source local data file. Repeat for multiple files (requires a wildcard mapping). The source is never modified." },
|
|
2841
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: "ae-local-data-mapping/v1 JSON, file path, or @file." },
|
|
2842
|
+
{ name: "output-dir", type: "string", sensitive: true, desc: "New or empty output directory. Default: .ae-cli/data-integration/<run-id>." },
|
|
2843
|
+
{ name: "type-resolutions", type: "json", sensitive: true, desc: "JSON object resolving cross-file column type conflicts (unify, split, or skip)." },
|
|
2844
|
+
{ name: "merge-sheets", type: "boolean", default: false, desc: "Stream every worksheet in file order instead of a single selected sheet." },
|
|
2845
|
+
{ name: "salvage-from", type: "string", sensitive: true, desc: "Re-process only the rows listed in a previous run's invalid.rows.jsonl, against the current (fixed) mapping. Single-file only." }
|
|
2846
|
+
],
|
|
2847
|
+
risk: "write",
|
|
2848
|
+
dryRun: async (ctx) => {
|
|
2849
|
+
const inputFiles = ctx.list("input-file");
|
|
2850
|
+
const multi = inputFiles.length > 1;
|
|
2851
|
+
const mapping = readLocalDataMapping(ctx.str("mapping"), { sourceWildcard: multi });
|
|
2852
|
+
if (multi) {
|
|
2853
|
+
return {
|
|
2854
|
+
action: "convert_local_data_multi",
|
|
2855
|
+
file_count: inputFiles.length,
|
|
2856
|
+
source_sha256: mapping.source.sha256,
|
|
2857
|
+
mode: mapping.mode,
|
|
2858
|
+
property_count: mapping.properties.length,
|
|
2859
|
+
type_resolutions: ctx.json("type-resolutions") ?? {},
|
|
2860
|
+
source_files_unchanged: true
|
|
2861
|
+
};
|
|
2862
|
+
}
|
|
2863
|
+
return {
|
|
2864
|
+
action: "convert_local_data",
|
|
2865
|
+
source_sha256: mapping.source.sha256,
|
|
2866
|
+
data_set: mapping.source.data_set,
|
|
2867
|
+
mode: mapping.mode,
|
|
2868
|
+
property_count: mapping.properties.length,
|
|
2869
|
+
salvage_from: ctx.str("salvage-from").trim() || void 0,
|
|
2870
|
+
source_file_unchanged: true
|
|
2871
|
+
};
|
|
2872
|
+
},
|
|
2873
|
+
execute: async (ctx) => {
|
|
2874
|
+
const inputFiles = ctx.list("input-file");
|
|
2875
|
+
const outputDir = ctx.str("output-dir").trim() || void 0;
|
|
2876
|
+
if (inputFiles.length === 1) {
|
|
2877
|
+
return convertLocalData({
|
|
2878
|
+
inputFile: inputFiles[0],
|
|
2879
|
+
mapping: readLocalDataMapping(ctx.str("mapping")),
|
|
2880
|
+
outputDir,
|
|
2881
|
+
mergeSheets: ctx.bool("merge-sheets"),
|
|
2882
|
+
salvageFrom: ctx.str("salvage-from").trim() || void 0
|
|
2883
|
+
});
|
|
2884
|
+
}
|
|
2885
|
+
return convertLocalDataMulti({
|
|
2886
|
+
inputFiles,
|
|
2887
|
+
mapping: readLocalDataMapping(ctx.str("mapping"), { sourceWildcard: true }),
|
|
2888
|
+
typeResolutions: ctx.json("type-resolutions") ?? {},
|
|
2889
|
+
outputDir
|
|
2890
|
+
});
|
|
2891
|
+
}
|
|
2892
|
+
};
|
|
2893
|
+
|
|
2894
|
+
// src/commands/data-integration/local-data/upload.ts
|
|
2895
|
+
import { createReadStream as createReadStream4, readFileSync as readFileSync3, statSync as statSync3 } from "fs";
|
|
2896
|
+
import { basename as basename5, dirname, resolve as resolve2 } from "path";
|
|
2897
|
+
import { createInterface as createInterface3 } from "readline";
|
|
2898
|
+
var DEFAULT_BATCH_SIZE = 500;
|
|
2899
|
+
var MAX_BATCH_SIZE = 1e3;
|
|
2900
|
+
var dataIntegrationUpload = {
|
|
2901
|
+
service: "data-integration",
|
|
2902
|
+
command: "upload",
|
|
2903
|
+
usesAeHost: false,
|
|
2904
|
+
description: "Sequentially submit validated UE JSONL to an explicit receiver /sync_json endpoint.",
|
|
2905
|
+
flags: [
|
|
2906
|
+
{ name: "ue-file", type: "string", required: true, sensitive: true, desc: "Validated valid.ue.jsonl generated by ae-cli data-integration convert." },
|
|
2907
|
+
{ name: "manifest", type: "string", required: true, sensitive: true, desc: "Matching ae-local-data-manifest/v1 file." },
|
|
2908
|
+
{ name: "endpoint", type: "string", required: true, sensitive: true, desc: "Complete receiver URL ending in /sync_json. Unrelated to --host." },
|
|
2909
|
+
{ name: "appid", type: "string", required: true, sensitive: true, desc: "Destination project APPID. Never stored in generated artifacts or logs." },
|
|
2910
|
+
{ name: "batch-size", type: "number", default: DEFAULT_BATCH_SIZE, min: 1, max: MAX_BATCH_SIZE, desc: `Sequential records per request. Maximum: ${MAX_BATCH_SIZE}.` },
|
|
2911
|
+
{ name: "resume-from", type: "number", default: 0, min: 0, desc: "Zero-based record offset selected by the user after verifying an interrupted upload." },
|
|
2912
|
+
{ name: "allow-clean-subset", type: "boolean", default: false, desc: "Explicitly allow valid rows from a blocked manifest after reviewing quarantine statistics." },
|
|
2913
|
+
{ name: "retry", type: "boolean", default: false, desc: "After the default 3 transport retries, split a still-uncertain batch into single-record retries and report per-record failures." },
|
|
2914
|
+
{ name: "compress", type: "string", default: "none", desc: "Request body compression: gzip or none." },
|
|
2915
|
+
{ name: "debug", type: "boolean", default: false, desc: "Mark every sync_json record debug=1 so AE returns per-record validation details. Testing only; not for production." }
|
|
2916
|
+
],
|
|
2917
|
+
risk: "write",
|
|
2918
|
+
validate: (ctx) => {
|
|
2919
|
+
resolveSyncJsonEndpoint(ctx.str("endpoint"));
|
|
2920
|
+
const batchSize = ctx.num("batch-size");
|
|
2921
|
+
const resumeFrom = ctx.num("resume-from");
|
|
2922
|
+
const compress = ctx.str("compress").trim().toLowerCase();
|
|
2923
|
+
if (compress !== "gzip" && compress !== "none") {
|
|
2924
|
+
throw new CliValidationError("Compression must be gzip or none.", {
|
|
2925
|
+
code: "LOCAL_DATA_COMPRESS_INVALID",
|
|
2926
|
+
location: { field: "compress" }
|
|
2927
|
+
});
|
|
2928
|
+
}
|
|
2929
|
+
if (!Number.isInteger(batchSize) || batchSize < 1 || batchSize > MAX_BATCH_SIZE) {
|
|
2930
|
+
throw new CliValidationError(`Batch size must be an integer between 1 and ${MAX_BATCH_SIZE}.`, {
|
|
2931
|
+
code: "LOCAL_DATA_BATCH_SIZE_INVALID",
|
|
2932
|
+
location: { field: "batch-size" }
|
|
2933
|
+
});
|
|
2934
|
+
}
|
|
2935
|
+
if (!Number.isInteger(resumeFrom) || resumeFrom < 0) {
|
|
2936
|
+
throw new CliValidationError("Resume position must be a non-negative integer.", {
|
|
2937
|
+
code: "LOCAL_DATA_RESUME_INVALID",
|
|
2938
|
+
location: { field: "resume-from" }
|
|
2939
|
+
});
|
|
2940
|
+
}
|
|
2941
|
+
},
|
|
2942
|
+
dryRun: async (ctx) => {
|
|
2943
|
+
const prepared = await prepareUpload(ctx, false);
|
|
2944
|
+
return {
|
|
2945
|
+
action: "upload_local_ue_data",
|
|
2946
|
+
target: maskEndpoint(prepared.endpoint),
|
|
2947
|
+
appid: maskAppid(prepared.appid),
|
|
2948
|
+
file_sha256: prepared.manifest.output.valid_sha256,
|
|
2949
|
+
record_count: prepared.recordCount,
|
|
2950
|
+
resume_from: prepared.resumeFrom,
|
|
2951
|
+
remaining_records: Math.max(0, prepared.recordCount - prepared.resumeFrom),
|
|
2952
|
+
batch_size: prepared.batchSize,
|
|
2953
|
+
batch_count: prepared.batchCount,
|
|
2954
|
+
bytes: prepared.fileBytes,
|
|
2955
|
+
manifest_status: prepared.manifest.status,
|
|
2956
|
+
upload_allowed: prepared.manifest.status === "ready" || ctx.bool("allow-clean-subset"),
|
|
2957
|
+
authentication_headers: false,
|
|
2958
|
+
transport_retries: 3,
|
|
2959
|
+
split_uncertain_batches: ctx.bool("retry"),
|
|
2960
|
+
compression: ctx.str("compress").trim().toLowerCase() === "gzip" ? "gzip" : "none",
|
|
2961
|
+
debug: ctx.bool("debug"),
|
|
2962
|
+
persistence_verified: false
|
|
2963
|
+
};
|
|
2964
|
+
},
|
|
2965
|
+
execute: async (ctx) => {
|
|
2966
|
+
const prepared = await prepareUpload(ctx, true);
|
|
2967
|
+
const salvageUncertain = ctx.bool("retry");
|
|
2968
|
+
const compress = ctx.str("compress").trim().toLowerCase() === "gzip" ? "gzip" : void 0;
|
|
2969
|
+
const debug = ctx.bool("debug");
|
|
2970
|
+
const transportOptions = { retries: 3, retryDelayMs: 2e3, compress };
|
|
2971
|
+
let submittedRecords = 0;
|
|
2972
|
+
let submittedBatches = 0;
|
|
2973
|
+
let processedRecords = 0;
|
|
2974
|
+
let requestBytes = 0;
|
|
2975
|
+
let responseBytes = 0;
|
|
2976
|
+
const failedRecords = [];
|
|
2977
|
+
for await (const batch of readUeBatches(prepared.ueFile, prepared.resumeFrom, prepared.batchSize)) {
|
|
2978
|
+
const batchStart = prepared.resumeFrom + processedRecords;
|
|
2979
|
+
const body = assembleSyncJsonBody(batch, prepared.appid, debug);
|
|
2980
|
+
try {
|
|
2981
|
+
const transport = await ctx.localDataUpload(prepared.endpoint, body, transportOptions);
|
|
2982
|
+
submittedRecords += batch.length;
|
|
2983
|
+
submittedBatches += 1;
|
|
2984
|
+
requestBytes += transport.request_bytes;
|
|
2985
|
+
responseBytes += transport.response_bytes;
|
|
2986
|
+
} catch (error) {
|
|
2987
|
+
if (!salvageUncertain || !(error instanceof LocalDataUploadError) || error.meta?.delivery_state !== "unknown") {
|
|
2988
|
+
throw decorateBatchError(error, submittedRecords, submittedBatches, batchStart, batch.length);
|
|
2989
|
+
}
|
|
2990
|
+
for (let index = 0; index < batch.length; index += 1) {
|
|
2991
|
+
const offset = batchStart + index;
|
|
2992
|
+
try {
|
|
2993
|
+
const single = await ctx.localDataUpload(
|
|
2994
|
+
prepared.endpoint,
|
|
2995
|
+
assembleSyncJsonBody([batch[index]], prepared.appid, debug),
|
|
2996
|
+
transportOptions
|
|
2997
|
+
);
|
|
2998
|
+
submittedRecords += 1;
|
|
2999
|
+
submittedBatches += 1;
|
|
3000
|
+
requestBytes += single.request_bytes;
|
|
3001
|
+
responseBytes += single.response_bytes;
|
|
3002
|
+
} catch {
|
|
3003
|
+
failedRecords.push(offset);
|
|
3004
|
+
}
|
|
3005
|
+
}
|
|
3006
|
+
}
|
|
3007
|
+
processedRecords += batch.length;
|
|
3008
|
+
}
|
|
3009
|
+
if (failedRecords.length > 0) {
|
|
3010
|
+
return {
|
|
3011
|
+
status: "partially_delivered",
|
|
3012
|
+
code: 0,
|
|
3013
|
+
delivery_state: "partially_delivered",
|
|
3014
|
+
submitted_records: submittedRecords,
|
|
3015
|
+
submitted_batches: submittedBatches,
|
|
3016
|
+
resume_from: prepared.resumeFrom,
|
|
3017
|
+
failed_records: failedRecords,
|
|
3018
|
+
failed_count: failedRecords.length,
|
|
3019
|
+
request_bytes: requestBytes,
|
|
3020
|
+
response_bytes: responseBytes,
|
|
3021
|
+
persistence_verified: false,
|
|
3022
|
+
retry_attempted: true,
|
|
3023
|
+
next_step: "Verify received data in AE before treating the upload as persisted."
|
|
3024
|
+
};
|
|
3025
|
+
}
|
|
3026
|
+
return {
|
|
3027
|
+
status: "receiver_accepted",
|
|
3028
|
+
code: 0,
|
|
3029
|
+
delivery_state: "receiver_accepted",
|
|
3030
|
+
submitted_records: submittedRecords,
|
|
3031
|
+
submitted_batches: submittedBatches,
|
|
3032
|
+
resume_from: prepared.resumeFrom,
|
|
3033
|
+
request_bytes: requestBytes,
|
|
3034
|
+
response_bytes: responseBytes,
|
|
3035
|
+
persistence_verified: false,
|
|
3036
|
+
retry_attempted: false,
|
|
3037
|
+
next_step: "Verify received data in AE before treating the upload as persisted."
|
|
3038
|
+
};
|
|
3039
|
+
}
|
|
3040
|
+
};
|
|
3041
|
+
function resolveSyncJsonEndpoint(raw) {
|
|
3042
|
+
let url;
|
|
3043
|
+
try {
|
|
3044
|
+
url = new URL(raw.trim());
|
|
3045
|
+
} catch {
|
|
3046
|
+
throw endpointError();
|
|
3047
|
+
}
|
|
3048
|
+
const validProtocol = url.protocol === "http:" || url.protocol === "https:";
|
|
3049
|
+
const forbidden = Boolean(url.username || url.password || url.search || url.hash);
|
|
3050
|
+
if (!validProtocol || forbidden || !url.pathname.endsWith("/sync_json")) throw endpointError();
|
|
3051
|
+
return url.toString();
|
|
3052
|
+
}
|
|
3053
|
+
function assembleSyncJsonBody(lines, appid, debug = false) {
|
|
3054
|
+
const appidJson = JSON.stringify(appid);
|
|
3055
|
+
const debugValue = debug ? 1 : 0;
|
|
3056
|
+
return `[${lines.map((line) => `{"appid":${appidJson},"debug":${debugValue},"data":${line}}`).join(",")}]`;
|
|
3057
|
+
}
|
|
3058
|
+
async function prepareUpload(ctx, enforceReady) {
|
|
3059
|
+
const endpoint = resolveSyncJsonEndpoint(ctx.str("endpoint"));
|
|
3060
|
+
const appid = ctx.str("appid").trim();
|
|
3061
|
+
if (!appid || appid.length > 256) {
|
|
3062
|
+
throw new CliValidationError("APPID must be a non-empty value of at most 256 characters.", {
|
|
3063
|
+
code: "LOCAL_DATA_APPID_INVALID",
|
|
3064
|
+
location: { field: "appid" }
|
|
3065
|
+
});
|
|
3066
|
+
}
|
|
3067
|
+
const manifestPath = resolve2(ctx.str("manifest"));
|
|
3068
|
+
const manifest = readManifest(manifestPath);
|
|
3069
|
+
const ueFile = resolve2(ctx.str("ue-file"));
|
|
3070
|
+
const expectedFile = resolve2(dirname(manifestPath), manifest.output.valid_file);
|
|
3071
|
+
if (ueFile !== expectedFile) {
|
|
3072
|
+
throw new CliValidationError("The UE file does not match the manifest output path.", {
|
|
3073
|
+
code: "LOCAL_DATA_MANIFEST_FILE_MISMATCH",
|
|
3074
|
+
location: { field: "ue-file" }
|
|
3075
|
+
});
|
|
3076
|
+
}
|
|
3077
|
+
let fileBytes;
|
|
3078
|
+
try {
|
|
3079
|
+
fileBytes = statSync3(ueFile).size;
|
|
3080
|
+
} catch {
|
|
3081
|
+
throw new CliValidationError("The UE JSONL file could not be read.", {
|
|
3082
|
+
code: "LOCAL_DATA_UE_FILE_NOT_FOUND",
|
|
3083
|
+
location: { field: "ue-file" }
|
|
3084
|
+
});
|
|
3085
|
+
}
|
|
3086
|
+
if (await sha256File(ueFile) !== manifest.output.valid_sha256) {
|
|
3087
|
+
throw new CliValidationError("The UE file fingerprint does not match the manifest.", {
|
|
3088
|
+
code: "LOCAL_DATA_UE_FILE_CHANGED",
|
|
3089
|
+
location: { field: "ue-file" }
|
|
3090
|
+
});
|
|
3091
|
+
}
|
|
3092
|
+
const recordCount = await validateUeFile(ueFile);
|
|
3093
|
+
if (recordCount !== manifest.output.valid_records) {
|
|
3094
|
+
throw new CliValidationError("The UE record count does not match the manifest.", {
|
|
3095
|
+
code: "LOCAL_DATA_UE_COUNT_MISMATCH",
|
|
3096
|
+
location: { field: "ue-file" }
|
|
3097
|
+
});
|
|
3098
|
+
}
|
|
3099
|
+
if (enforceReady && manifest.status === "blocked" && !ctx.bool("allow-clean-subset")) {
|
|
3100
|
+
throw new CliValidationError("The manifest is blocked because rows were quarantined.", {
|
|
3101
|
+
code: "LOCAL_DATA_CLEAN_SUBSET_CONFIRMATION_REQUIRED",
|
|
3102
|
+
hint: "Review manifest and invalid.rows.jsonl statistics, then explicitly pass --allow-clean-subset if the valid subset may be uploaded.",
|
|
3103
|
+
location: { field: "allow-clean-subset" }
|
|
3104
|
+
});
|
|
3105
|
+
}
|
|
3106
|
+
const resumeFrom = ctx.num("resume-from");
|
|
3107
|
+
if (resumeFrom > recordCount) {
|
|
3108
|
+
throw new CliValidationError("Resume position exceeds the UE record count.", {
|
|
3109
|
+
code: "LOCAL_DATA_RESUME_OUT_OF_RANGE",
|
|
3110
|
+
location: { field: "resume-from" }
|
|
3111
|
+
});
|
|
3112
|
+
}
|
|
3113
|
+
const batchSize = ctx.num("batch-size");
|
|
3114
|
+
return {
|
|
3115
|
+
endpoint,
|
|
3116
|
+
appid,
|
|
3117
|
+
ueFile,
|
|
3118
|
+
manifest,
|
|
3119
|
+
batchSize,
|
|
3120
|
+
resumeFrom,
|
|
3121
|
+
recordCount,
|
|
3122
|
+
fileBytes,
|
|
3123
|
+
batchCount: Math.ceil(Math.max(0, recordCount - resumeFrom) / batchSize)
|
|
3124
|
+
};
|
|
3125
|
+
}
|
|
3126
|
+
function readManifest(path) {
|
|
3127
|
+
let value;
|
|
3128
|
+
try {
|
|
3129
|
+
value = JSON.parse(readFileSync3(path, "utf8"));
|
|
3130
|
+
} catch {
|
|
3131
|
+
throw new CliValidationError("The local-data manifest could not be read as JSON.", {
|
|
3132
|
+
code: "LOCAL_DATA_MANIFEST_INVALID",
|
|
3133
|
+
location: { field: "manifest" }
|
|
3134
|
+
});
|
|
3135
|
+
}
|
|
3136
|
+
if (!isRecord3(value) || value.version !== "ae-local-data-manifest/v1" || !isRecord3(value.output) || typeof value.output.valid_file !== "string" || basename5(value.output.valid_file) !== value.output.valid_file || typeof value.output.valid_sha256 !== "string" || !/^[a-f0-9]{64}$/i.test(value.output.valid_sha256) || !Number.isInteger(value.output.valid_records) || value.output.valid_records < 0 || value.status !== "ready" && value.status !== "blocked") {
|
|
3137
|
+
throw new CliValidationError("The manifest is not ae-local-data-manifest/v1.", {
|
|
3138
|
+
code: "LOCAL_DATA_MANIFEST_INVALID",
|
|
3139
|
+
location: { field: "manifest" }
|
|
3140
|
+
});
|
|
3141
|
+
}
|
|
3142
|
+
return value;
|
|
3143
|
+
}
|
|
3144
|
+
async function validateUeFile(path) {
|
|
3145
|
+
let count = 0;
|
|
3146
|
+
const lines = createInterface3({ input: createReadStream4(path), crlfDelay: Infinity });
|
|
3147
|
+
for await (const line of lines) {
|
|
3148
|
+
if (!line.trim()) continue;
|
|
3149
|
+
count += 1;
|
|
3150
|
+
let value;
|
|
3151
|
+
try {
|
|
3152
|
+
value = JSON.parse(line);
|
|
3153
|
+
} catch {
|
|
3154
|
+
throw invalidUeFile(count);
|
|
3155
|
+
}
|
|
3156
|
+
if (!isRecord3(value) || !isUeRecordType(value["#type"])) {
|
|
3157
|
+
throw invalidUeFile(count);
|
|
3158
|
+
}
|
|
3159
|
+
}
|
|
3160
|
+
return count;
|
|
3161
|
+
}
|
|
3162
|
+
var UE_RECORD_TYPES = /* @__PURE__ */ new Set([
|
|
3163
|
+
"track",
|
|
3164
|
+
"user_set",
|
|
3165
|
+
"user_setOnce",
|
|
3166
|
+
"user_add",
|
|
3167
|
+
"user_unset",
|
|
3168
|
+
"user_del",
|
|
3169
|
+
"user_append",
|
|
3170
|
+
"user_uniq_append"
|
|
3171
|
+
]);
|
|
3172
|
+
function isUeRecordType(value) {
|
|
3173
|
+
return typeof value === "string" && UE_RECORD_TYPES.has(value);
|
|
3174
|
+
}
|
|
3175
|
+
function decorateBatchError(error, completedRecords, completedBatches, batchStart, batchSize) {
|
|
3176
|
+
if (!(error instanceof LocalDataUploadError)) return error;
|
|
3177
|
+
const deliveryUnknown = error.meta?.delivery_state === "unknown";
|
|
3178
|
+
return new LocalDataUploadError(error.message, {
|
|
3179
|
+
code: error.code,
|
|
3180
|
+
httpStatus: error.httpStatus,
|
|
3181
|
+
hint: error.hint,
|
|
3182
|
+
meta: {
|
|
3183
|
+
...error.meta,
|
|
3184
|
+
completed_records: completedRecords,
|
|
3185
|
+
completed_batches: completedBatches,
|
|
3186
|
+
...deliveryUnknown ? {
|
|
3187
|
+
uncertain_batch_start: batchStart,
|
|
3188
|
+
uncertain_batch_size: batchSize,
|
|
3189
|
+
resume_from_after_verification: batchStart
|
|
3190
|
+
} : {
|
|
3191
|
+
failed_batch_start: batchStart,
|
|
3192
|
+
failed_batch_size: batchSize
|
|
3193
|
+
}
|
|
3194
|
+
},
|
|
3195
|
+
cause: error
|
|
3196
|
+
});
|
|
3197
|
+
}
|
|
3198
|
+
async function* readUeBatches(path, offset, batchSize) {
|
|
3199
|
+
let index = 0;
|
|
3200
|
+
let batch = [];
|
|
3201
|
+
const lines = createInterface3({ input: createReadStream4(path), crlfDelay: Infinity });
|
|
3202
|
+
for await (const line of lines) {
|
|
3203
|
+
if (!line.trim()) continue;
|
|
3204
|
+
if (index++ < offset) continue;
|
|
3205
|
+
batch.push(line);
|
|
3206
|
+
if (batch.length >= batchSize) {
|
|
3207
|
+
yield batch;
|
|
3208
|
+
batch = [];
|
|
3209
|
+
}
|
|
3210
|
+
}
|
|
3211
|
+
if (batch.length > 0) yield batch;
|
|
3212
|
+
}
|
|
3213
|
+
function invalidUeFile(record) {
|
|
3214
|
+
return new CliValidationError("The UE JSONL file contains an invalid record.", {
|
|
3215
|
+
code: "LOCAL_DATA_UE_FILE_INVALID",
|
|
3216
|
+
location: { record }
|
|
3217
|
+
});
|
|
3218
|
+
}
|
|
3219
|
+
function endpointError() {
|
|
3220
|
+
return new CliValidationError("A complete http(s) receiver endpoint ending in /sync_json is required.", {
|
|
3221
|
+
code: "LOCAL_DATA_ENDPOINT_INVALID",
|
|
3222
|
+
location: { field: "endpoint" }
|
|
3223
|
+
});
|
|
3224
|
+
}
|
|
3225
|
+
function maskEndpoint(endpoint) {
|
|
3226
|
+
const url = new URL(endpoint);
|
|
3227
|
+
const parts = url.hostname.split(".");
|
|
3228
|
+
const first = parts[0];
|
|
3229
|
+
parts[0] = first.length <= 2 ? "**" : `${first[0]}***${first.at(-1)}`;
|
|
3230
|
+
return `${url.protocol}//${parts.join(".")}${url.port ? `:${url.port}` : ""}/sync_json`;
|
|
3231
|
+
}
|
|
3232
|
+
function maskAppid(appid) {
|
|
3233
|
+
return appid.length <= 4 ? "****" : `${"*".repeat(Math.min(8, appid.length - 4))}${appid.slice(-4)}`;
|
|
3234
|
+
}
|
|
3235
|
+
function isRecord3(value) {
|
|
3236
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3237
|
+
}
|
|
3238
|
+
|
|
3239
|
+
// src/commands/data-integration/local-data/handoff.ts
|
|
3240
|
+
import { createHash as createHash3 } from "crypto";
|
|
3241
|
+
import { chmodSync as chmodSync2, mkdirSync as mkdirSync2, readFileSync as readFileSync4, renameSync as renameSync2, writeFileSync as writeFileSync2 } from "fs";
|
|
3242
|
+
import { join as join2, resolve as resolve3 } from "path";
|
|
3243
|
+
var HANDOFF_INDEX_VERSION = "ae-data-integration-index/v1";
|
|
3244
|
+
var HANDOFF_DIR_LEN = 16;
|
|
3245
|
+
function structureFingerprint(mapping) {
|
|
3246
|
+
const canonical = {
|
|
3247
|
+
mode: mapping.mode,
|
|
3248
|
+
time_field: mapping.time.field,
|
|
3249
|
+
account_id_field: mapping.account_id_field ?? null,
|
|
3250
|
+
distinct_id_field: mapping.distinct_id_field ?? null,
|
|
3251
|
+
record_type_field: mapping.record_type_field ?? null,
|
|
3252
|
+
event_name_field: mapping.event_name_field ?? null,
|
|
3253
|
+
columns: mapping.properties.map((property) => ({ source: property.source, type: property.type })).sort((left, right) => left.source.localeCompare(right.source)),
|
|
3254
|
+
excluded: [...mapping.exclude_columns ?? []].sort()
|
|
3255
|
+
};
|
|
3256
|
+
return createHash3("sha256").update(JSON.stringify(canonical)).digest("hex");
|
|
3257
|
+
}
|
|
3258
|
+
function upsertIndexEntry(index, entry) {
|
|
3259
|
+
const entries = index.entries.filter((item) => item.fingerprint !== entry.fingerprint);
|
|
3260
|
+
return { version: index.version, entries: [...entries, entry] };
|
|
3261
|
+
}
|
|
3262
|
+
function buildHandoffPackage(outDir, mapping, planFile) {
|
|
3263
|
+
const fingerprint = structureFingerprint(mapping);
|
|
3264
|
+
const dirName = fingerprint.slice(0, HANDOFF_DIR_LEN);
|
|
3265
|
+
const handoffDir = join2(outDir, dirName);
|
|
3266
|
+
const indexPath = join2(outDir, "index.json");
|
|
3267
|
+
const index = readHandoffIndex(indexPath);
|
|
3268
|
+
const reusedExisting = index.entries.some((item) => item.fingerprint === fingerprint);
|
|
3269
|
+
mkdirSync2(handoffDir, { recursive: true, mode: 448 });
|
|
3270
|
+
chmodSync2(handoffDir, 448);
|
|
3271
|
+
writeSecureJson2(join2(handoffDir, "mapping.json"), mapping);
|
|
3272
|
+
writeSecureText2(join2(handoffDir, "transform.mjs"), createHandoffScript());
|
|
3273
|
+
let planFileRel;
|
|
3274
|
+
if (planFile) {
|
|
3275
|
+
writeSecureJson2(join2(handoffDir, "plan.json"), readPlanFile(planFile));
|
|
3276
|
+
planFileRel = `${dirName}/plan.json`;
|
|
3277
|
+
}
|
|
3278
|
+
const entry = {
|
|
3279
|
+
fingerprint,
|
|
3280
|
+
created_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
3281
|
+
source_sha256: mapping.source.sha256,
|
|
3282
|
+
format: mapping.source.format,
|
|
3283
|
+
data_set: mapping.source.data_set,
|
|
3284
|
+
mode: mapping.mode,
|
|
3285
|
+
property_count: mapping.properties.length,
|
|
3286
|
+
mapping_file: `${dirName}/mapping.json`,
|
|
3287
|
+
...planFileRel ? { plan_file: planFileRel } : {}
|
|
3288
|
+
};
|
|
3289
|
+
writeAtomicJson(indexPath, upsertIndexEntry(index, entry));
|
|
3290
|
+
return { outDir, handoffDir, dirName, fingerprint, reusedExisting, planFile: planFileRel };
|
|
3291
|
+
}
|
|
3292
|
+
function readHandoffIndex(path) {
|
|
3293
|
+
let raw;
|
|
3294
|
+
try {
|
|
3295
|
+
raw = readFileSync4(path, "utf8");
|
|
3296
|
+
} catch {
|
|
3297
|
+
return { version: HANDOFF_INDEX_VERSION, entries: [] };
|
|
3298
|
+
}
|
|
3299
|
+
let value;
|
|
3300
|
+
try {
|
|
3301
|
+
value = JSON.parse(raw);
|
|
3302
|
+
} catch {
|
|
3303
|
+
throw new CliValidationError("The handoff index is not valid JSON.", {
|
|
3304
|
+
code: "LOCAL_DATA_HANDOFF_INDEX_INVALID",
|
|
3305
|
+
location: { field: "out-dir" }
|
|
3306
|
+
});
|
|
3307
|
+
}
|
|
3308
|
+
if (!isRecord4(value) || value.version !== HANDOFF_INDEX_VERSION || !Array.isArray(value.entries)) {
|
|
3309
|
+
throw new CliValidationError("The handoff index is not ae-data-integration-index/v1.", {
|
|
3310
|
+
code: "LOCAL_DATA_HANDOFF_INDEX_INVALID",
|
|
3311
|
+
location: { field: "out-dir" }
|
|
3312
|
+
});
|
|
3313
|
+
}
|
|
3314
|
+
return value;
|
|
3315
|
+
}
|
|
3316
|
+
function readPlanFile(path) {
|
|
3317
|
+
let raw;
|
|
3318
|
+
try {
|
|
3319
|
+
raw = readFileSync4(path, "utf8");
|
|
3320
|
+
} catch {
|
|
3321
|
+
throw new CliValidationError("The tracking-plan file could not be read.", {
|
|
3322
|
+
code: "LOCAL_DATA_HANDOFF_PLAN_NOT_FOUND",
|
|
3323
|
+
location: { field: "plan-file" }
|
|
3324
|
+
});
|
|
3325
|
+
}
|
|
3326
|
+
try {
|
|
3327
|
+
return JSON.parse(raw);
|
|
3328
|
+
} catch {
|
|
3329
|
+
throw new CliValidationError("The tracking-plan file is not valid JSON.", {
|
|
3330
|
+
code: "LOCAL_DATA_HANDOFF_PLAN_INVALID",
|
|
3331
|
+
location: { field: "plan-file" }
|
|
3332
|
+
});
|
|
3333
|
+
}
|
|
3334
|
+
}
|
|
3335
|
+
function createHandoffScript() {
|
|
3336
|
+
return [
|
|
3337
|
+
"import { createHash } from 'node:crypto';",
|
|
3338
|
+
"import { createReadStream, readFileSync } from 'node:fs';",
|
|
3339
|
+
"import { spawnSync } from 'node:child_process';",
|
|
3340
|
+
"import { fileURLToPath } from 'node:url';",
|
|
3341
|
+
"import { dirname, join } from 'node:path';",
|
|
3342
|
+
"",
|
|
3343
|
+
"const [inputFile, outputDir] = process.argv.slice(2);",
|
|
3344
|
+
"if (!inputFile) { console.error('Usage: node transform.mjs <new-input-file> [<output-dir>]'); process.exit(2); }",
|
|
3345
|
+
"",
|
|
3346
|
+
"async function sha256Of(file) {",
|
|
3347
|
+
" const hash = createHash('sha256');",
|
|
3348
|
+
" for await (const chunk of createReadStream(file)) hash.update(chunk);",
|
|
3349
|
+
" return hash.digest('hex');",
|
|
3350
|
+
"}",
|
|
3351
|
+
"",
|
|
3352
|
+
"const here = dirname(fileURLToPath(import.meta.url));",
|
|
3353
|
+
"const mapping = JSON.parse(readFileSync(join(here, 'mapping.json'), 'utf8'));",
|
|
3354
|
+
"// Re-stamp the content fingerprint for this specific file; the transform logic is reused unchanged.",
|
|
3355
|
+
"mapping.source.sha256 = await sha256Of(inputFile);",
|
|
3356
|
+
"",
|
|
3357
|
+
"const args = ['data-integration', 'convert', '--input-file', inputFile, '--mapping', JSON.stringify(mapping)];",
|
|
3358
|
+
"if (outputDir) args.push('--output-dir', outputDir);",
|
|
3359
|
+
"const result = spawnSync('ae-cli', args, { stdio: 'inherit' });",
|
|
3360
|
+
"process.exit(result.status ?? 1);",
|
|
3361
|
+
""
|
|
3362
|
+
].join("\n");
|
|
3363
|
+
}
|
|
3364
|
+
function writeSecureJson2(path, value) {
|
|
3365
|
+
writeSecureText2(path, `${JSON.stringify(value, null, 2)}
|
|
3366
|
+
`);
|
|
3367
|
+
}
|
|
3368
|
+
function writeSecureText2(path, content) {
|
|
3369
|
+
writeFileSync2(path, content, { encoding: "utf8", mode: 384 });
|
|
3370
|
+
chmodSync2(path, 384);
|
|
3371
|
+
}
|
|
3372
|
+
function writeAtomicJson(path, value) {
|
|
3373
|
+
writeSecureText2(`${path}.tmp`, `${JSON.stringify(value, null, 2)}
|
|
3374
|
+
`);
|
|
3375
|
+
renameSync2(`${path}.tmp`, path);
|
|
3376
|
+
chmodSync2(path, 384);
|
|
3377
|
+
}
|
|
3378
|
+
function isRecord4(value) {
|
|
3379
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3380
|
+
}
|
|
3381
|
+
var dataIntegrationHandoff = {
|
|
3382
|
+
service: "data-integration",
|
|
3383
|
+
command: "handoff",
|
|
3384
|
+
usesAeHost: false,
|
|
3385
|
+
description: "Export a reusable handoff package (frozen mapping + transform script + plan reference) under .ae-data-integration/.",
|
|
3386
|
+
flags: [
|
|
3387
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: "Confirmed ae-local-data-mapping/v1 JSON, file path, or @file." },
|
|
3388
|
+
{ name: "plan-file", type: "string", sensitive: true, desc: "Tracking-plan draft.json to reference inside the handoff package." },
|
|
3389
|
+
{ name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-data-integration." }
|
|
3390
|
+
],
|
|
3391
|
+
risk: "write",
|
|
3392
|
+
dryRun: async (ctx) => {
|
|
3393
|
+
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
3394
|
+
const outDir = resolve3(ctx.str("out-dir").trim() || ".ae-data-integration");
|
|
3395
|
+
const planFile = ctx.str("plan-file").trim() || void 0;
|
|
3396
|
+
return {
|
|
3397
|
+
action: "handoff_local_data",
|
|
3398
|
+
out_dir: outDir,
|
|
3399
|
+
fingerprint: structureFingerprint(mapping),
|
|
3400
|
+
source_sha256: mapping.source.sha256,
|
|
3401
|
+
mode: mapping.mode,
|
|
3402
|
+
property_count: mapping.properties.length,
|
|
3403
|
+
files: ["mapping.json", "transform.mjs", ...planFile ? ["plan.json"] : []],
|
|
3404
|
+
index_file: join2(outDir, "index.json")
|
|
3405
|
+
};
|
|
3406
|
+
},
|
|
3407
|
+
execute: async (ctx) => {
|
|
3408
|
+
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
3409
|
+
const outDir = resolve3(ctx.str("out-dir").trim() || ".ae-data-integration");
|
|
3410
|
+
const planFile = ctx.str("plan-file").trim() || void 0;
|
|
3411
|
+
const build = buildHandoffPackage(outDir, mapping, planFile);
|
|
3412
|
+
return {
|
|
3413
|
+
out_dir: build.outDir,
|
|
3414
|
+
handoff_dir: build.handoffDir,
|
|
3415
|
+
fingerprint: build.fingerprint,
|
|
3416
|
+
mapping_file: `${build.dirName}/mapping.json`,
|
|
3417
|
+
...build.planFile ? { plan_file: build.planFile } : {},
|
|
3418
|
+
run: `node ${join2(build.handoffDir, "transform.mjs")} <new-input-file> [<output-dir>]`,
|
|
3419
|
+
index_file: join2(outDir, "index.json"),
|
|
3420
|
+
reused_existing: build.reusedExisting
|
|
3421
|
+
};
|
|
3422
|
+
}
|
|
3423
|
+
};
|
|
3424
|
+
|
|
3425
|
+
// src/commands/data-integration/local-data/reuse.ts
|
|
3426
|
+
import { readFileSync as readFileSync5 } from "fs";
|
|
3427
|
+
import { dirname as dirname2, join as join3, resolve as resolve4 } from "path";
|
|
3428
|
+
function detectReuse(mapping, outDir) {
|
|
3429
|
+
const fingerprint = structureFingerprint(mapping);
|
|
3430
|
+
const indexPath = join3(outDir, "index.json");
|
|
3431
|
+
const index = readHandoffIndex(indexPath);
|
|
3432
|
+
const entry = index.entries.find((item) => item.fingerprint === fingerprint);
|
|
3433
|
+
if (!entry) return { matched: false, fingerprint, index_file: indexPath };
|
|
3434
|
+
const mappingPath = join3(outDir, entry.mapping_file);
|
|
3435
|
+
const match = {
|
|
3436
|
+
fingerprint: entry.fingerprint,
|
|
3437
|
+
created_at: entry.created_at,
|
|
3438
|
+
format: entry.format,
|
|
3439
|
+
data_set: entry.data_set,
|
|
3440
|
+
mode: entry.mode,
|
|
3441
|
+
property_count: entry.property_count,
|
|
3442
|
+
mapping_file: entry.mapping_file,
|
|
3443
|
+
...entry.plan_file ? { plan_file: entry.plan_file } : {},
|
|
3444
|
+
...readFrozenEventName(mappingPath),
|
|
3445
|
+
run: `node ${join3(outDir, dirname2(entry.mapping_file), "transform.mjs")} <new-input-file> [<output-dir>]`
|
|
3446
|
+
};
|
|
3447
|
+
return { matched: true, fingerprint, index_file: indexPath, match };
|
|
3448
|
+
}
|
|
3449
|
+
function readFrozenEventName(path) {
|
|
3450
|
+
let value;
|
|
3451
|
+
try {
|
|
3452
|
+
value = JSON.parse(readFileSync5(path, "utf8"));
|
|
3453
|
+
} catch {
|
|
3454
|
+
return {};
|
|
3455
|
+
}
|
|
3456
|
+
if (isRecord5(value) && typeof value.default_event_name === "string" && value.default_event_name) {
|
|
3457
|
+
return { default_event_name: value.default_event_name };
|
|
3458
|
+
}
|
|
3459
|
+
return {};
|
|
3460
|
+
}
|
|
3461
|
+
function isRecord5(value) {
|
|
3462
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
3463
|
+
}
|
|
3464
|
+
var dataIntegrationReuse = {
|
|
3465
|
+
service: "data-integration",
|
|
3466
|
+
command: "reuse",
|
|
3467
|
+
usesAeHost: false,
|
|
3468
|
+
description: "Match a candidate mapping against the .ae-data-integration/ handoff index and propose a reusable package.",
|
|
3469
|
+
flags: [
|
|
3470
|
+
{ name: "mapping", type: "string", required: true, sensitive: true, desc: "Candidate ae-local-data-mapping/v1 JSON, file path, or @file (typically inspect recommended_mapping)." },
|
|
3471
|
+
{ name: "out-dir", type: "string", sensitive: true, desc: "Handoff root directory. Default: .ae-data-integration." }
|
|
3472
|
+
],
|
|
3473
|
+
risk: "read",
|
|
3474
|
+
dryRun: async (ctx) => {
|
|
3475
|
+
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
3476
|
+
const outDir = resolve4(ctx.str("out-dir").trim() || ".ae-data-integration");
|
|
3477
|
+
const result = detectReuse(mapping, outDir);
|
|
3478
|
+
return {
|
|
3479
|
+
action: "reuse_detect",
|
|
3480
|
+
fingerprint: result.fingerprint,
|
|
3481
|
+
matched: result.matched,
|
|
3482
|
+
index_file: result.index_file
|
|
3483
|
+
};
|
|
3484
|
+
},
|
|
3485
|
+
execute: async (ctx) => {
|
|
3486
|
+
const mapping = readLocalDataMapping(ctx.str("mapping"));
|
|
3487
|
+
const outDir = resolve4(ctx.str("out-dir").trim() || ".ae-data-integration");
|
|
3488
|
+
return detectReuse(mapping, outDir);
|
|
3489
|
+
}
|
|
3490
|
+
};
|
|
3491
|
+
|
|
3492
|
+
// src/commands/data-integration/index.ts
|
|
3493
|
+
var commands = [dataIntegrationInspect, dataIntegrationPlan, dataIntegrationConvert, dataIntegrationUpload, dataIntegrationHandoff, dataIntegrationReuse];
|
|
3494
|
+
var data_integration_default = commands;
|
|
3495
|
+
export {
|
|
3496
|
+
dataIntegrationConvert,
|
|
3497
|
+
dataIntegrationHandoff,
|
|
3498
|
+
dataIntegrationInspect,
|
|
3499
|
+
dataIntegrationPlan,
|
|
3500
|
+
dataIntegrationReuse,
|
|
3501
|
+
dataIntegrationUpload,
|
|
3502
|
+
data_integration_default as default
|
|
3503
|
+
};
|