@thinkingai/ae-cli 6.1.16 → 6.1.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/README.md +5 -2
  2. package/README.zh.md +5 -2
  3. package/dist/{auth-2WTQOP77.js → auth-GBMV6TEJ.js} +3 -3
  4. package/dist/{auth-77BUFLGC.js → auth-ROB2EDYV.js} +12 -13
  5. package/dist/{capability-72DTW5M2.js → capability-DKMYUTLC.js} +50 -13
  6. package/dist/{capability-PJHNI4GJ.js → capability-HYVVPG25.js} +49 -12
  7. package/dist/chunk-3FY3RJ26.js +293 -0
  8. package/dist/{chunk-PTE56QPL.js → chunk-3KWQYGYI.js} +4 -0
  9. package/dist/{chunk-UOUS37JQ.js → chunk-4XXOWOTA.js} +3 -3
  10. package/dist/{chunk-GT46FPXN.js → chunk-BYYS3ANB.js} +17 -8
  11. package/dist/{chunk-YA6SMTXG.js → chunk-EFH4XWYC.js} +3 -3
  12. package/dist/{chunk-4SGZG4XY.js → chunk-J2DEBMRF.js} +9 -7
  13. package/dist/{chunk-LYVNONC4.js → chunk-JHENBQ5B.js} +35 -0
  14. package/dist/{chunk-VR3LCBHW.js → chunk-JRJY5DMJ.js} +5 -5
  15. package/dist/{chunk-ILIU36SU.js → chunk-OMPRXM3V.js} +3 -3
  16. package/dist/{chunk-VPKZ7I72.js → chunk-QNOLN2LJ.js} +2 -2
  17. package/dist/{chunk-SAU3QFIQ.js → chunk-QZ3AS4KK.js} +3 -3
  18. package/dist/chunk-RJDU7NYP.js +1198 -0
  19. package/dist/{chunk-4KVPKXFX.js → chunk-RNAALWJK.js} +2 -2
  20. package/dist/{chunk-C4MGVGJW.js → chunk-SERWF6G5.js} +1 -1
  21. package/dist/{chunk-RGKJGKT7.js → chunk-Y3LOALAV.js} +5 -5
  22. package/dist/{chunk-P3FGXJTU.js → chunk-ZQ47LWTI.js} +4 -4
  23. package/dist/chunk-ZQKDZXDO.js +317 -0
  24. package/dist/{client-TKG4WBHN.js → client-L2YDMHQ6.js} +5 -4
  25. package/dist/{community-report-client-FI4LNVYS.js → community-report-client-C7WDGET3.js} +1 -2
  26. package/dist/{config-RE6CMGPK.js → config-BMYZX2UE.js} +7 -6
  27. package/dist/{data-integration-2MYMANJI.js → data-integration-QEKDWQDY.js} +1741 -212
  28. package/dist/index.js +131 -1241
  29. package/dist/{local-data-upload-client-BWHSUQQK.js → local-data-upload-client-4YYHSYD6.js} +1 -2
  30. package/dist/{memory-CHRU2F7W.js → memory-3ORCR7JH.js} +6 -6
  31. package/dist/{memory-YK33G4T7.js → memory-I2WXDTV2.js} +5 -5
  32. package/dist/{metadata-XXR34N5P.js → metadata-I4C2EWUN.js} +10 -10
  33. package/dist/{metadata-UILXHBWF.js → metadata-VUOQJE26.js} +9 -9
  34. package/dist/{model-NR3JHFSJ.js → model-HLHIEFMU.js} +5 -5
  35. package/dist/{model-K3KLWIW6.js → model-UGRDX4MW.js} +6 -6
  36. package/dist/personal-semantic-preference-LIPACBDX.js +239 -0
  37. package/dist/personal-semantic-preference-OEISBRHM.js +239 -0
  38. package/dist/project-semantic-FFPWFPIW.js +1114 -0
  39. package/dist/project-semantic-RT3R2VQD.js +1114 -0
  40. package/dist/{sync-FCKOVWWS.js → sync-HKIOZXQE.js} +6 -6
  41. package/dist/{sync-DAVKYVMW.js → sync-TFHU2UTG.js} +7 -7
  42. package/dist/{te-agent-HLW4VTQK.js → te-agent-BR6VDBNX.js} +9 -8
  43. package/dist/{te-agent-4BKBODMF.js → te-agent-VLYOV7S4.js} +8 -7
  44. package/dist/{te-analysis-ZMNGOVNW.js → te-analysis-4YGQL5RC.js} +437 -38
  45. package/dist/{te-analysis-O6DCO6BS.js → te-analysis-7VUNUYWZ.js} +436 -37
  46. package/dist/{te-community-HLC43QKH.js → te-community-5DMNKJWY.js} +5 -5
  47. package/dist/{te-community-6HPBWJUZ.js → te-community-ISDQWJU7.js} +6 -6
  48. package/dist/{te-dataops-HDRUXY4K.js → te-dataops-6P5IKWNJ.js} +8 -7
  49. package/dist/{te-dataops-EJP56W3K.js → te-dataops-CVULXNVB.js} +7 -6
  50. package/dist/{te-engage-RAK5PESW.js → te-engage-KZPR5R22.js} +9 -9
  51. package/dist/{te-engage-FGBGQ4IY.js → te-engage-N5WI32H6.js} +8 -8
  52. package/dist/{te-experiment-SO5MPDMJ.js → te-experiment-6BITX4RD.js} +226 -8
  53. package/dist/{te-experiment-VZF7BT6G.js → te-experiment-UVR4HLND.js} +227 -9
  54. package/dist/{te-kb-APXBWBDY.js → te-kb-RCLSSH2Q.js} +251 -127
  55. package/dist/{te-system-Z77IKZFN.js → te-system-FXITO2JG.js} +5 -5
  56. package/dist/{te-system-YARIK4S5.js → te-system-K2GYMCTB.js} +6 -6
  57. package/dist/{te-team-EFKWYKMK.js → te-team-ADOC2ROP.js} +6 -6
  58. package/dist/{update-OGPSZM5A.js → update-YCYCKJOO.js} +7 -6
  59. package/package.json +2 -1
  60. package/skills/ae-agent/SKILL.md +3 -4
  61. package/skills/ae-agent/references/edit-skill.md +3 -0
  62. package/skills/ae-agent/references/get-skill-content.md +1 -1
  63. package/skills/ae-agent/references/rescan-skills.md +15 -13
  64. package/skills/ae-agent/references/upload-skill.md +7 -4
  65. package/skills/ae-analysis/SKILL.md +45 -4
  66. package/skills/ae-analysis/metadata_resolution.md +38 -4
  67. package/skills/ae-analysis/references/analysis_data_retrieval.md +29 -0
  68. package/skills/ae-analysis/references/asset_authentication_export.md +22 -0
  69. package/skills/ae-analysis/references/asset_authentication_list.md +18 -14
  70. package/skills/ae-analysis/references/asset_authentication_update.md +29 -14
  71. package/skills/ae-analysis/references/command_index.md +17 -9
  72. package/skills/ae-analysis/references/dashboard_get.md +18 -1
  73. package/skills/ae-analysis/references/dashboard_update.md +3 -0
  74. package/skills/ae-analysis/references/personal_semantic_preference_add.md +23 -0
  75. package/skills/ae-analysis/references/personal_semantic_preference_delete.md +17 -0
  76. package/skills/ae-analysis/references/personal_semantic_preference_get.md +19 -0
  77. package/skills/ae-analysis/references/personal_semantic_preference_list.md +21 -0
  78. package/skills/ae-analysis/references/personal_semantic_preference_update.md +19 -0
  79. package/skills/ae-data-integration/SKILL.md +23 -4
  80. package/skills/ae-data-integration/references/custom-layer.md +93 -0
  81. package/skills/ae-data-integration/references/error-handling.md +92 -0
  82. package/skills/ae-data-integration/references/handoff.md +77 -18
  83. package/skills/ae-data-integration/references/local-analysis.md +1 -1
  84. package/skills/ae-data-integration/references/reuse.md +9 -5
  85. package/skills/ae-data-integration/references/sink-upload.md +1 -1
  86. package/skills/ae-data-integration/references/source-inspect.md +32 -13
  87. package/skills/ae-data-integration/references/tracking-plan.md +7 -5
  88. package/skills/ae-data-integration/references/transform.md +10 -10
  89. package/skills/ae-data-integration/references/ue-mapping.md +33 -11
  90. package/skills/ae-engage/references/build-task-save-guide.md +9 -0
  91. package/skills/ae-engage/references/save-task.md +82 -0
  92. package/skills/ae-experiment/SKILL.md +8 -2
  93. package/skills/ae-experiment/references/manage_feature_whitelist.md +66 -0
  94. package/skills/ae-experiment/references/manage_guardrail_metrics.md +26 -0
  95. package/skills/ae-experiment/references/save_experiment.md +1 -1
  96. package/skills/ae-kb/SKILL.md +56 -51
  97. package/skills/ae-kb/references/query-workflow.md +112 -0
  98. package/skills/ae-kb-discovery/SKILL.md +105 -0
  99. package/skills/ae-project-semantic/SKILL.md +193 -0
  100. package/skills/ae-project-semantic/references/query-routing-v5.md +165 -0
  101. package/skills/ae-project-semantic/references/recommendation-quality.md +68 -0
  102. package/dist/chunk-QGM4M3NI.js +0 -37
  103. package/dist/chunk-ZZUOD757.js +0 -598
@@ -1,43 +1,102 @@
1
1
  # Handoff (reusable package)
2
2
 
3
- After a successful run, export a **handoff package** so the next file of the same shape skips the full pipeline. The package freezes the transform logic (the confirmed mapping), the runnable transform script, and the tracking-plan reference. It does **not** re-freeze the raw data or any upload secrets.
3
+ After a successful run, export a **handoff package** so the next file of the same shape skips the full pipeline. The package is a DataX-style pipeline — a declarative `source → transform → sink` descriptor plus generic stage executors that dispatch to `ae-cli data-integration` subcommands. It freezes the transform logic (the confirmed mappings) and the tracking-plan reference, and ships a `bin/` directory that a human or agent can run directly. It never re-freezes raw data or upload secrets.
4
4
 
5
5
  ## When
6
6
 
7
- Run handoff after Transform — or after Sink — once the mapping is confirmed. It is local-only and idempotent: re-running it for the same table refreshes the package in place.
7
+ Run handoff after Transform — or after Sink — once the mappings are confirmed. It is local-only and idempotent: re-running it for the same table refreshes the package in place.
8
8
 
9
9
  ## CLI
10
10
 
11
11
  ```
12
- ae-cli data-integration handoff --mapping <mapping> [--plan-file <draft.json>] [--out-dir .ae-data-integration]
12
+ ae-cli data-integration handoff --mapping <mapping> [--mapping <mapping2> ...] [--plan-file <draft.json>] [--pushurl <url>] [--project-id <id>] [--out-dir .ae-cli/data-integration]
13
13
  ```
14
14
 
15
- - `--mapping` — the confirmed `ae-local-data-mapping/v1` mapping (the frozen transform logic, including `value_mapping` and `flatten_rules`).
16
- - `--plan-file` — optional tracking-plan `draft.json` to reference inside the package (`plan.json`).
17
- - `--out-dir` — handoff root. Default `.ae-data-integration/` (travels with the project).
18
- - `--dry-run` previews the fingerprint, files, and index path without writing.
15
+ - `--mapping` — one or more confirmed `ae-data-integration-mapping/v1` mappings (the frozen transform logic, including `value_mapping` and `flatten_rules`). Repeat it for multi-sheet workbooks: one mapping per sheet.
16
+ - `--plan-file` — optional tracking-plan `draft.json` to reference inside each mapping directory (`plan.json`).
17
+ - `--pushurl` — optional receiver base URL to record as the reuse upload target (the sink endpoint is `pushurl` + `/sync_json`). Record it when the next same-shape file will most likely land at the same receiver.
18
+ - `--project-id` optional numeric destination project ID to record; `bin/upload.sh` derives the APPID from it via `project info get` at upload time.
19
+ - `--out-dir` — handoff root. Default `.ae-cli/data-integration/` (project workspace; travels with the project).
20
+ - `--dry-run` previews the fingerprints, target, file list, and zip path without writing.
19
21
 
20
22
  ## Package layout
21
23
 
22
24
  ```
23
- .ae-data-integration/
24
- index.json shared fingerprint index (reuse detection)
25
- <fingerprint[:16]>/
26
- mapping.json frozen mapping (re-runnable by convert)
27
- transform.mjs node transform.mjs <new-file> [<output-dir>]
28
- plan.json optional tracking-plan reference
25
+ .ae-cli/data-integration/ ← handoff root (= out-dir, project workspace)
26
+ pipeline.json source transform → sink descriptor (ae-data-integration-pipeline/v1)
27
+ index.json ← shared structure-fingerprint index (reuse detection)
28
+ shape.json per-mapping column baseline (shape gate)
29
+ <fingerprint[:16]>/ one directory per mapping
30
+ mapping.json frozen mapping
31
+ transform.mjs ← node transform.mjs <new-file> [<output-dir>]
32
+ plan.json ← optional tracking-plan reference
33
+ bin/ ← generic stage executors (read pipeline.json)
34
+ run.sh ← source → transform → plan (never uploads)
35
+ upload.sh ← sink (dry-run by default; --confirm uploads; resolves recorded target)
36
+ bind_mapping.py ← shape check + rebind sha256/data_set to the new file
37
+ summarize.py ← valid/quarantined counts
38
+ plan_check.py ← tracking-plan coverage gate (exit 3 on new events/properties)
39
+ verify.py ← soft persistence check (submit window vs ingest summary)
40
+ resolve_appid.py ← APPID derivation helper (project info get)
41
+ README.md / RUNBOOK.md ← how to run, the four gates, persistence verification
42
+ .local/target.env.example ← destination template (no real secrets)
43
+ .gitignore ← inbox/ runs/ .local/target.env
44
+ inbox/ runs/ ← daily input / per-run outputs
29
45
  ```
30
46
 
31
- The `transform.mjs` wrapper shells out to `ae-cli data-integration convert`; it takes the new file path as its first argument, never bakes the source path, and re-stamps `source.sha256` with the new file's fingerprint before converting. Reuse it only for a file of the **same shape** (same header/schema and format) the transform logic is frozen, but the content guard is re-bound to each specific file.
47
+ A shareable archive is written next to the package root: `<parent>/ae-data-integration-handoff-<fingerprint[:8]>.zip`. The zip carries only this handoff round the frozen mappings just written, a scoped `index.json` (just this round's entries), and the generic executors/docs not the accumulated history, which stays in `.ae-cli/data-integration/` for reuse matching.
48
+
49
+ ## Pipeline descriptor
50
+
51
+ `pipeline.json` declares the three stages and their types. Only `source: local_file` and `sink: restful_sync_json` are implemented today; the `type` fields reserve logbus / datax / mysql for later phases. `bin/run.sh` and `bin/upload.sh` read the descriptor and dispatch each stage by its `type` to `ae-cli data-integration inspect / convert / upload` — they are executors, not a second runtime engine.
52
+
53
+ ## Recorded destination
54
+
55
+ `pipeline.json` → `sink.params` records `pushurl` and `project_id` when the handoff
56
+ was run with those flags. Reuse defaults to that target, but **`bin/upload.sh`
57
+ never sends without `--confirm`**, so the operator re-confirms the address and
58
+ project on every reuse. Resolution order at upload time:
59
+
60
+ - endpoint: recorded `pushurl` (+ `/sync_json`), else `AE_ENDPOINT`.
61
+ - APPID: `AE_APPID`, else derived via `ae-cli project info get --project-id <id>`
62
+ (see `bin/resolve_appid.py`; set `AE_APPID` when that payload lacks `appid`).
63
+ - project id: recorded `project_id`, else `AE_PROJECT_ID`.
64
+
65
+ `project info get` returns `data.appid` at the top level (verified against the AE
66
+ demo host); `bin/resolve_appid.py` reads that exact field and prints it, falling
67
+ back to `AE_APPID` when the field is absent or not a non-empty string.
68
+
69
+ ## Project custom layers
70
+
71
+ For projects that need a hard per-event SQL judge or a multi-round salvage loop
72
+ (the reference package's environment-specific workarounds), overlay a custom layer
73
+ next to the package instead of editing it — see [custom-layer.md](custom-layer.md).
74
+
75
+ ## Four confirmation gates
76
+
77
+ The RUNBOOK and the scripts enforce four gates. The first two run automatically; the last two always need human confirmation:
78
+
79
+ 1. **Shape gate** — `bind_mapping.py` compares the new file's column set against `shape.json` and fails fast on a mismatch. A changed shape means the frozen logic was never reviewed for it: re-run the full pipeline.
80
+ 2. **Transform** — `ae-cli data-integration convert` per mapping; quarantined rows go to `invalid.rows.jsonl`, never silently dropped.
81
+ 3. **Tracking-plan gate** — `plan_check.py` verifies every produced event/property already exists in `plan.json`; new ones exit 3 and must be merged into the project plan first.
82
+ 4. **Sink gate** — `upload.sh` is dry-run by default; `--confirm` is the explicit upload decision.
32
83
 
33
84
  ## Structure fingerprint and index
34
85
 
35
- Every handoff records one entry in `index.json` (`ae-data-integration-index/v1`) keyed by a **structure fingerprint** — a SHA-256 over the table shape: columns (source name + type), event model (`mode`, `event_name_field`, `record_type_field`), identity fields, and excluded columns. Business logic (`value_mapping`, transforms, `time_format`, and the fixed `default_event_name`) is excluded, so re-handing off the same table with new business rules refreshes the existing entry instead of forking a new one.
86
+ Every handoff records one entry per mapping in `index.json` (`ae-data-integration-index/v1`) keyed by a **structure fingerprint** — a SHA-256 over the table shape: the raw source columns (by name, reconstructed so flatten, exclude, and account-vs-distinct decisions don't move it), the format, and the event model (`mode`). Business logic (`value_mapping`, transforms, `time_format`, `flatten_rules`, `exclude_columns`, the fixed `default_event_name`, and the system-field assignments) is excluded, so re-handing off the same table with new business rules refreshes the existing entry instead of forking a new one.
87
+
88
+ Reuse matching is its own step — see [references/reuse.md](reuse.md).
36
89
 
37
- Reuse matching is its own step — see [references/reuse.md](reuse.md). It compares a new file's profile structure against this index and proposes the matching package, which the user confirms before skipping the full pipeline.
90
+ ## Completion response
91
+
92
+ After handoff succeeds, state the **absolute zip path**, the package directory, and the one-line way to run the next same-shape file:
93
+
94
+ ```
95
+ cd <out-dir> && bin/run.sh <new-file> # then bin/upload.sh runs/<run-id> --confirm
96
+ ```
38
97
 
39
98
  ## Safety rules
40
99
 
41
- - Treat the mapping, plan, generated artifacts, and the `.ae-data-integration/` directory as sensitive.
42
- - Never write APPID, endpoints, tokens, or raw data values into a handoff package. The package references the plan and the mapping; uploads still require an explicit, confirmed `upload` call.
100
+ - Treat the mappings, plans, generated artifacts, the `.ae-cli/data-integration/` directory, and the zip as sensitive.
101
+ - Never write APPID, tokens, or raw data values into a handoff package. The package records at most a destination `pushurl` and `project_id`; uploads still require an explicit, confirmed `upload` call, so the operator re-confirms the address and project each time.
43
102
  - Do not invent a mapping or plan. Handoff only packages what the user already confirmed.
@@ -1,6 +1,6 @@
1
1
  # Local analysis
2
2
 
3
- Keep the source on the local machine. Generated scripts and reports belong under `.ae-cli/data-integration/<run-id>/` with restrictive permissions.
3
+ Keep the source on the local machine. Generated scripts and reports belong under `.ae-cli/data-integration/runs/<run-id>/` with restrictive permissions.
4
4
  Set the directory to `0700` and generated scripts/reports to `0600`.
5
5
 
6
6
  ## Default report when no question is supplied
@@ -4,21 +4,25 @@ Before walking the full pipeline for a new file, check whether a **handoff packa
4
4
 
5
5
  ## When
6
6
 
7
- Run reuse right after `inspect`, only when the profile is `ue_eligible` and a `.ae-data-integration/index.json` exists. It is read-only and makes no writes.
7
+ Run reuse right after `inspect`, only when the profile is `ue_eligible` and a `.ae-cli/data-integration/index.json` exists. It is read-only and makes no writes.
8
8
 
9
9
  ## CLI
10
10
 
11
11
  ```
12
- ae-cli data-integration reuse --mapping <recommended_mapping> [--out-dir .ae-data-integration]
12
+ ae-cli data-integration reuse --mapping <recommended_mapping> [--out-dir .ae-cli/data-integration]
13
13
  ```
14
14
 
15
15
  - `--mapping` — the candidate mapping, typically `inspect`'s `recommended_mapping` (the new file's structure, inferred exactly as Source would infer it).
16
- - `--out-dir` — handoff root. Default `.ae-data-integration/`.
16
+ - `--out-dir` — handoff root. Optional. Without it, `reuse` searches in order: the current directory's `.ae-cli/data-integration/`, then each parent directory upward, then `~/.ae-cli/data-integration/` as a global fallback — so a package written in another directory or another session is still reachable.
17
17
  - `--dry-run` previews the fingerprint and match verdict without reading frozen packages.
18
18
 
19
+ ## Search paths
20
+
21
+ When `--out-dir` is omitted, the result carries `searched_paths` — the exact `index.json` paths probed, in order. A `matched: false` result with a non-empty `searched_paths` means all of them were checked; a missing global fallback (no `$HOME`) simply omits it from the list.
22
+
19
23
  ## Matching
20
24
 
21
- The command computes the same **structure fingerprint** the handoff index is keyed on — columns (source name + type), event model, identity fields, and excluded columns — and looks it up in `index.json`. Business logic (the frozen event name, `value_mapping`, transforms, `time_format`) is not part of the key, so a same-shape file with different content or a different file name still matches.
25
+ The command computes the same **structure fingerprint** the handoff index is keyed on — the raw source columns (by name, reconstructed so flatten, exclude, and account-vs-distinct decisions don't move it), the format, and the event model — and looks it up in `index.json`. Business logic (the frozen event name, `value_mapping`, transforms, `time_format`, `flatten_rules`, `exclude_columns`, and the system-field assignments) is not part of the key, so a same-shape file with different content or a different file name still matches.
22
26
 
23
27
  ## Result
24
28
 
@@ -26,7 +30,7 @@ The command computes the same **structure fingerprint** the handoff index is key
26
30
  - **Match** (`matched: true`): the result carries the matched package — `mapping_file`, optional `plan_file`, the frozen `default_event_name` (so the user sees which event name will be reused), and a `run` command:
27
31
 
28
32
  ```
29
- node .ae-data-integration/<fingerprint[:16]>/transform.mjs <new-input-file> [<output-dir>]
33
+ node .ae-cli/data-integration/<fingerprint[:16]>/transform.mjs <new-input-file> [<output-dir>]
30
34
  ```
31
35
 
32
36
  ## Confirmation gate
@@ -31,7 +31,7 @@ ae-cli data-integration upload \
31
31
  --dry-run
32
32
  ```
33
33
 
34
- Show the masked target, project, file fingerprint, record count, quarantined count, batch count, and persistence limitation. Re-state the system-field mapping first (`#type`/mode, `#account_id`, `#distinct_id`, `#time` + source timezone, `#event_name`), then re-list the final property mapping for every file or sheet being uploaded — source column → target AE name → type, grouped into event properties (`track`) and user properties (profile modes) — never a counts-only summary. Wait for explicit confirmation. Execute the same command without `--dry-run` only after confirmation. For a blocked manifest, first show the quarantine statistics and separately ask whether the user accepts uploading only valid rows; add `--allow-clean-subset` only after a clear yes.
34
+ Show the masked target, project, file fingerprint, record count, quarantined count, batch count, and persistence limitation. Re-state the system-field mapping first (`#type`/mode, `#account_id`, `#distinct_id`, `#time` + source timezone, `#event_name`), then re-list the final property mapping for every file or sheet being uploaded — source column → target AE name → type (+ `display_name`/`desc` when set), grouped into event properties (`track`) and user properties (profile modes), with each event's attached properties listed — never a counts-only summary. For a multi-sheet workbook, group by sheet. Wait for explicit confirmation. Execute the same command without `--dry-run` only after confirmation. For a blocked manifest, first show the quarantine statistics and separately ask whether the user accepts uploading only valid rows; add `--allow-clean-subset` only after a clear yes.
35
35
 
36
36
  `status=receiver_accepted` means receiver acceptance only, not durable storage. Say that persistence remains unverified. Never report success on this status alone.
37
37
 
@@ -8,10 +8,23 @@ Confirm:
8
8
  2. The desired outcome: AE ingestion, local analysis, or help deciding.
9
9
  3. For ingestion, the target environment/project if already known.
10
10
 
11
- Accept one or more CSV, TSV, TXT, JSON, JSONL (NDJSON), XLS, or XLSX files. CSV/TSV/TXT/JSON/JSONL/XLSX may be at most 200 MB each; XLS may be at most 50 MB each.
11
+ Accept one or more CSV, TSV, TXT, JSON, JSONL (NDJSON), XLS, or XLSX files. CSV/TSV/TXT/JSON/JSONL/XLSX have no hard size limit; files over 1 GB print a stderr warning with an estimated processing time and suggest splitting. XLS over 100 MB prints a memory-risk warning (the legacy parser loads the whole workbook, roughly 5-10x file size); XLS over 1 GB is still rejected — convert it to XLSX or split it first.
12
12
 
13
13
  ## Inspect without exposing raw values
14
14
 
15
+ Before the full inspection (which streams and profiles the entire file and can take
16
+ minutes on large files), pre-check the size and time estimate:
17
+
18
+ ```bash
19
+ ae-cli data-integration inspect --input-file '<path>' --dry-run
20
+ ```
21
+
22
+ For each file that reports a `warning`, `memory_risk: true`, or `rejected: true`,
23
+ surface the `estimated_duration` / `reason` / `warning` to the user verbatim and
24
+ confirm before running the real `inspect`. Never start a multi-minute or memory-risk
25
+ job without the user seeing the estimate and risk first. A rejected file cannot
26
+ proceed; follow its `reason` instead of asking for confirmation.
27
+
15
28
  Run:
16
29
 
17
30
  ```bash
@@ -26,7 +39,9 @@ If `selection_required=true`, show only the Sheet/JSON Path candidates and ask t
26
39
  ae-cli data-integration inspect --input-file '<path>' --data-set '<candidate-id>' --source-timezone '<iana-timezone>'
27
40
  ```
28
41
 
29
- Report row/column counts, field types, missing/unique/time-parse ratios, UE eligibility, mapping confidence, and warnings. Samples are bounded (up to 5 distinct, truncated) — summarize, never paste them. ID-like columns (`id`, `*_id`, `*_key`, `*_code`, `*_no`, `*_num`) stay `string` even when every value is numeric; JSON-encoded object/array values inside CSV cells are recognized as `object`/`list`, not `string`. Read [UE routing](ue-routing.md) before choosing a branch.
42
+ Report row/column counts, field types, missing/unique/time-parse ratios, UE eligibility, mapping confidence, and warnings. Samples are bounded (up to 5 distinct, truncated) — summarize, never paste them. ID-like columns (`id`, `*_id`, `*_key`, `*_code`, `*_no`, `*_num`) stay `string` even when every value is numeric; JSON-encoded object/array values inside CSV cells are recognized as `object`/`list`, not `string`. IP- and UUID-named columns are additionally checked against their value specs: inspect warns how many non-empty values are invalid IPv4/IPv6, private/LAN IPs, or non-UUID strings, so the user can decide whether to map them as `ip_field`/`uuid_field`. Read [UE routing](ue-routing.md) before choosing a branch.
43
+
44
+ A stderr `Warning: … column count different from the header row …` means the CSV/TSV has ragged rows (extra fields dropped, missing fields treated as empty); report it as a data-quality signal. For how every pipeline failure — abnormal data, parse errors, and program errors — is classified and handled, see [error handling](error-handling.md).
30
45
 
31
46
  ## Advanced input
32
47
 
@@ -36,20 +51,24 @@ Report row/column counts, field types, missing/unique/time-parse ratios, UE elig
36
51
  - **Excel sheets** — `--merge-sheets` streams every worksheet in file order instead of a single selected sheet; otherwise ask which sheet/`--data-set` to use. Inspect also reports `header_consistency` (`all_same` or `different`) across a workbook's sheets, with `header_details` listing each sheet's header row when they differ; prefer `--merge-sheets` only when headers match.
37
52
  - **Multi-file type conflicts** — when the same column has different inferred types across files, present each conflict and resolve with `--type-resolutions` on `convert` (see [transform](transform.md)).
38
53
 
39
- ## Nested flattening (NDJSON/JSON records and JSON-encoded CSV/TSV cells)
54
+ ## Nested flattening (NDJSON/JSON records and JSON-encoded CSV/TSV/Excel cells)
40
55
 
41
- Nested data is flattened one level per user decision. This flow is agent-driven and non-interactive: never pipe pre-filled answers into any prompt, and never silently pick `Flatten` for a node.
56
+ Nested data is analyzed per field, and the recommended mapping encodes one decision per container: keep whole, flatten one level, or flatten to the leaves. This flow is agent-driven and non-interactive: never pipe pre-filled answers into any prompt, and never silently pick a depth for a node.
42
57
 
43
58
  1. **Locate the nested structure.**
44
59
  - NDJSON/JSON: read the top-level `nested_tree` from the inspect result (record-root paths).
45
- - CSV/TSV: a JSON-encoded object column carries its own tree at `columns[].nested_tree` (paths relative to that cell). Array cells (`items`-style) are `list` columns and carry no tree.
60
+ - CSV/TSV/Excel: a JSON-encoded column carries its own tree at `columns[].nested_tree` (paths relative to that cell). Object cells list child keys; array cells (`items`-style) expose an `elementKind` and, for object arrays, the union of the element fields.
46
61
  Object nodes list child keys, array nodes carry an `elementKind`, primitive leaves carry an inferred type and bounded samples. Summarize node kinds, never paste samples.
47
- 2. **Judge first, then confirm.** Read the node names and inferred types and propose, per container, whether to flatten (and to what depth), keep as JSON, or keep as a list. Present the proposal as a table and default to it; ask the user to confirm or adjust. Only ask node-by-node for the nodes you cannot judge from the names.
48
- - Business-entity names with stable scalar children (`user_info`, `address`, `order`) propose flattening to the leaves.
49
- - Generic container names (`payload`, `data`, `config`, `meta`, `extra`, `attributes`) propose keeping as JSON, unless the user wants a specific sub-field.
50
- - Arrays are always kept as a list property and never split.
51
- - Nesting deeper than one level with no clear intent propose keeping as JSON or flattening only the first level.
52
- 3. **Record the answers in `flatten_rules` (`{ "out_column": "dot.path" }`).**
62
+ 2. **Review the recommended mapping's per-field decision, then confirm adjustments.**
63
+ The recommended mapping already encodes, per container, a data-driven depth decision the depth is not uniform across fields:
64
+ - **Keep whole** a single-level object (every child scalar or a scalar array) is declared `type: 'object'` with `transform: 'json'`, plus one `parent.child` sub-property per child; a single-level object array is declared `type: 'array_row'` with one `parent.child` sub-property per element field. The parent entry carries the native value and conversion emits it once; each child is a plan-only declaration (scalar or `list`) that never reads a column of its own.
65
+ - **Flatten one level / to the leaves** — an object that itself contains an object or object array cannot be kept whole; it collapses one level and each child is decided independently: scalar children become flat properties materialized through `flatten_rules`, and nested objects/arrays recurse until a single-level container is reached (which is then kept whole).
66
+ - **Scalar array** `["a","b"]` stays `list` (AE `array_string`), a leaf with no children.
67
+ - **Nested element field** — an element field that is itself an object/array stays inside the array data and is not declared as a sub-property (array-element flattening is not supported); the mapping warns so it is never silently dropped.
68
+ Present the resulting property list as a table and default to it; ask the user to confirm or adjust only where you disagree with a node's inferred decision (business-entity names with stable scalar children vs generic containers like `payload`/`data`/`config` vs nesting deeper than one level with no clear intent).
69
+ 3. **Record overrides in `flatten_rules` (`{ "out_column": "dot.path" }`) and `exclude_columns`.**
70
+ The recommended mapping already carries the flatten rules for its collapsed levels. When you override it:
53
71
  - NDJSON/JSON: the path is from the record root (`user_info.name`).
54
- - CSV/TSV: the path is `<column>.<cell-relative path>` (`user_profile.name`), and add the source column to `exclude_columns` so the whole object is not also mapped.
55
- A leaf path becomes a snake_case out column when not explicitly named (`user_info.address.geo.lat` → `user_info_address_geo_lat`; CSV prefixes the column: `user_profile.level` `user_profile_level`). A string that looks like a number (phone, zip, ID) is kept as a string unless the user says otherwise. Containers kept whole are declared `type: 'object'`/`'list'` **with `transform: 'json'`** so conversion restores the native structure.
72
+ - CSV/TSV/Excel: the path is `<column>.<cell-relative path>` (`user_profile.name`), and add the source column to `exclude_columns` so the whole object is not also mapped.
73
+ - **`flatten_rules` only materializes the out column in the row it does not emit it.** For every out column, also add a `properties` entry with `source` set to that out-column name (plus `target`/`type`/`desc`); otherwise the flattened value is silently dropped from the output record.
74
+ A leaf path becomes a snake_case out column when not explicitly named (`user_info.address.geo.lat` → `user_info_address_geo_lat`; cell-relative rules prefix the column: `user_profile.level` → `user_profile_level`). A string that looks like a number (phone, zip, ID) is kept as a string unless the user says otherwise. Containers kept whole are declared `type: 'object'`/`'array_row'`/`'list'` **with `transform: 'json'`** so conversion restores the native structure; a kept whole `object`/`array_row` also declares its scalar children as dotted `parent.child` sub-properties in `properties`.
@@ -10,15 +10,15 @@ Generate and confirm the event/property plan **before** any transform or upload.
10
10
  ## Sub-steps
11
11
 
12
12
  1. **Event-model decision** — reuse the UE routing result: single-table single-event `track`, single-table multi-event (event-name column), `user_set`, or `mixed`. The agent may propose splitting one table into several events (for example an ad table into `ad_show`/`ad_click` by `campaign_type`); that proposal must be confirmed by the user in the confirmation gate.
13
- 2. **Column → property draft** — identify system columns (time, `distinct_id`/`account_id`, event-name column, user-property-name column); map the remaining columns to event / user / common properties. Name events and properties in snake_case and fill **every** `display_name`, `desc`, and `event_tag` (events also carry `event_desc`). Infer all three from field names, value distribution, samples, and business-doc / prompt priors — never leave them empty: `desc`/`event_desc` state what the item means in plain language (language follows the user), and `event_tag` picks the closest category from the canonical tag list (see the `event_tag` appendix in `../../ae-generate-tracking-plan/references/business-dimension-mapping.md`). When you cannot infer a `desc` or `event_tag`, mark it pending and ask the user for it inside the confirmation gate. Infer types (`number` / `bool` / `datetime` / enum) the same way; CSV defaults to `string`. Columns that stay uncertain or conflicting are marked pending and asked only inside the confirmation gate.
14
- 3. **Confirmation gate (single, one pass)** — present one summary table: event list + property list (uncertain types highlighted) + field scope (default: plan fields only, with a full-import switch) + unrecognized/dirty-data handling. The user answers once with ok or edits (renames, types, add/drop columns).
13
+ 2. **Column → property draft** — confirm the recommended mapping's key system fields with the user **before** drafting: `mode` (`#type`), `#account_id`/`#distinct_id` (ask together; at least one is required — a `user_id` column can be either an anonymous or a login ID and only the user knows), `#time` + source timezone + `#zone_offset`, `#event_name` (track only; the event column or a reviewed `default_event_name`), and `#ip`/`#uuid` when the data has such a column. The exact questions and the never-infer-from-a-column-name-alone rule are [transform.md](transform.md) steps 1–5; run them here. The plan is generated from this mapping, so never draft from an unconfirmed mapping. With the system fields settled, map the remaining columns to event and/or user properties (the mapping `mode` decides). Common event properties — project-level super properties attached to every event — are defined by `ae-generate-tracking-plan`, not this import path. Name events and properties in snake_case and fill **every** `display_name`, `desc`, and `event_tag` (events also carry `event_desc`). Infer all three from field names, value distribution, samples, and business-doc / prompt priors — never leave them empty: `desc`/`event_desc` state what the item means in plain language (language follows the user), and `event_tag` picks the closest category from the canonical tag list (see the `event_tag` appendix in `../../ae-generate-tracking-plan/references/business-dimension-mapping.md`). When you cannot infer a `desc` or `event_tag`, mark it pending and ask the user for it inside the confirmation gate. Infer types (`number` / `bool` / `datetime` / enum) the same way; CSV defaults to `string`. Columns that stay uncertain or conflicting are marked pending and asked only inside the confirmation gate.
14
+ 3. **Confirmation gate (single, one pass)** — present the concrete plan, never a counts-only summary: the confirmed key system-field mapping (`mode`, `#account_id`/`#distinct_id`, `#time` + source timezone, `#event_name`; `#ip`/`#uuid` when present), then a full event table (one row per event: `event_name` + `event_tag`/`event_desc` + the properties attached to it), then a full property table (one row per property: source column → target AE name → type → `display_name`/`desc`, uncertain types highlighted; a kept-whole `object`/`array_row` lists its `parent.child` sub-properties next to the parent), plus field scope (default: plan fields only, with a full-import switch) and unrecognized/dirty-data handling. For a multi-sheet workbook, group the property table by sheet so each sheet's source columns are visible. The user answers once with ok or edits (renames, types, identity/time/event columns, add/drop columns).
15
15
  4. **Merge with the existing plan** — fetch the project's current tracking plan; same-name property type conflicts are severe, same-name events are advisory; decide append vs replace. This runs for **every** file, not just the first: when the project already has a plan (an earlier file or run), diff this file's events and properties against it and put the additions — new events, new properties, new object sub-properties from flattening — in the confirmation gate. An existing plan is never a reason to skip this step; only when every addition is already present may you skip the merge, and even then state and confirm that fact with the user.
16
- 5. **Persist the plan** — draft.json → xlsx → upload with `sdk_integration_mode=none`.
16
+ 5. **Persist the plan** — `.ae-cli/data-integration/draft.json``.ae-cli/data-integration/draft.xlsx` → upload with `sdk_integration_mode=none`.
17
17
  6. **Hand off the field mapping** — the confirmed plan plus column→property mapping, `value_mapping`, and `flatten_rules` feed the Transform submodule.
18
18
 
19
19
  ## CLI
20
20
 
21
- `ae-cli data-integration plan --mapping <mapping> [--event-name <name>...] [--plan-name <name>] [--out <draft.json>] [--dry-run]` converts the confirmed `ae-local-data-mapping/v1` mapping into a tracking-plan `draft.json` with `sdk_integration_mode=none` and `source_type=data`.
21
+ `ae-cli data-integration plan --mapping <mapping> [--event-name <name>...] [--plan-name <name>] [--out .ae-cli/data-integration/draft.json] [--dry-run]` converts the confirmed `ae-data-integration-mapping/v1` mapping into a tracking-plan `draft.json` with `sdk_integration_mode=none` and `source_type=data`.
22
22
 
23
23
  - `user_set` mode → no events; every mapping property becomes a user property.
24
24
  - `track` mode → one event per `--event-name` (or the mapping `default_event_name`); every property becomes an event property.
@@ -26,9 +26,11 @@ Generate and confirm the event/property plan **before** any transform or upload.
26
26
  - `exclude_columns` are dropped from the draft.
27
27
  - `desc` and `event_tag` flow from the mapping: each property's `desc` (falling back to its source column name) and each event's `event_meta.<name>.desc` / `event_meta.<name>.tag` (falling back to the source event name) are written into the draft. Fill them in the mapping so the plan is never empty.
28
28
  - The per-row event-name column cannot be enumerated without a full scan, so when `default_event_name` is absent the CLI requires `--event-name` for each concrete event name.
29
+ - Duplicate `display_name`s across the property pool are auto-deduplicated (the property name is appended) before validation, so same-named source columns under different object paths never collide. Dotted `target`s default each sub-property's `display_name`/`desc` to its leaf segment, which also avoids collisions.
30
+ - The draft is validated locally before any upload: an `object`/`array_row` property with no `parent.child` sub-property, or a sub-property whose type is `object`/`array_row`, fails fast with `LOCAL_DATA_PLAN_INVALID_DRAFT` (`--dry-run` fails too). Fix it in the mapping before running `plan`: a mapping `properties` entry with a dotted `target` (`user_info.name`) declares a `parent.child` sub-property whose parent is the `object`/`array_row` entry with the same first segment — so a kept whole container must carry one dotted entry per child (each scalar or `list`) next to the parent entry. A nested object/object-array that cannot stay nested must instead be flattened to scalar leaves: record one `flatten_rules` entry per leaf and one flat scalar (or `list`) `properties` entry per leaf (see [source-inspect.md](source-inspect.md)).
29
31
  - `--dry-run` previews events and properties without writing `draft.json`.
30
32
 
31
- Afterward reuse `ae-cli tracking plan draft --in draft.json --out draft.xlsx` and `ae-cli tracking plan validate / upload` for xlsx generation and ingestion.
33
+ Afterward reuse `ae-cli tracking plan draft --in .ae-cli/data-integration/draft.json --out .ae-cli/data-integration/draft.xlsx` and `ae-cli tracking plan validate / upload` for xlsx generation and ingestion.
32
34
 
33
35
  ## Reuse
34
36
 
@@ -1,6 +1,6 @@
1
1
  # Transform — column → UE mapping
2
2
 
3
- Precondition: the tracking plan ([tracking-plan.md](tracking-plan.md)) has been generated and confirmed. A field mapping is not a tracking plan; if the plan is missing, return to tracking-plan.md first.
3
+ Precondition: the tracking plan ([tracking-plan.md](tracking-plan.md)) has been generated and confirmed. A field mapping is not a tracking plan; if the plan is missing, return to tracking-plan.md first. The key system fields (steps 1–5 below) were already confirmed during the tracking-plan step (tracking-plan.md step 2); if nothing changed, state the confirmed values once and move to the property set — never skip a confirmation the user has not actually given.
4
4
 
5
5
  Read [UE mapping](ue-mapping.md) and apply these gates:
6
6
 
@@ -12,10 +12,10 @@ Read [UE mapping](ue-mapping.md) and apply these gates:
12
12
  Confirm the system fields with the user before touching properties. These are the top-level `#` fields, and users may not understand them, so explain each in plain language before asking; never infer a decision from a column name alone.
13
13
 
14
14
  1. **`#type` / mode.** Confirm the recommendation's `mode` (`track` / `user_set` / `mixed`). If `record_type_field` is set, confirm each distinct value normalizes to one of the eight record types. Ask whether the data reports events (→ `track`) or modifies user profiles (→ `user_set` or another profile type).
15
- 2. **`#account_id` and `#distinct_id` — ask together; at least one is required.** Explain both: `#account_id` identifies logged-in users (database `user_id`, phone, member ID); `#distinct_id` identifies anonymous visitors (device/cookie/visitor ID). Present every `identity_candidates` entry from the inspect result alongside the recommendation's `account_id_field`/`distinct_id_field` picks, and ask which column maps to `#account_id`, which to `#distinct_id`, and whether both apply. A column named `user_id` can be either an anonymous or a login ID — only the user knows. If a required identity column is missing, offer an explicit `account_id_value`/`distinct_id_value` placeholder or `random_pool`; never invent one.
15
+ 2. **`#account_id` and `#distinct_id` — ask together; at least one is required.** Explain both: `#account_id` identifies logged-in users (database `user_id`, phone, member ID); `#distinct_id` identifies anonymous visitors (device/cookie/visitor ID). Present every `identity_candidates` entry from the inspect result alongside the recommendation's `account_id_field`/`distinct_id_field` picks, and ask which column maps to `#account_id`, which to `#distinct_id`, and whether both apply. A column named `user_id` can be either an anonymous or a login ID — only the user knows. If a required identity column is missing, offer an explicit `account_id_value`/`distinct_id_value` placeholder or `random_pool`; never invent one. `identity_candidates` is name-matched only, so it misses identity columns with arbitrary names (`玩家ID`, `用户账号`, `player_name`); surface any column whose shape is identifier-like (high `unique_ratio`, low `missing_ratio`, a string or numeric key, not a time column) and ask the user what it is. When one file has multiple sheets/data-sets, cross-check identity columns across them before finalizing: a column that appears (or near-matches, e.g. `设备ID` vs `主设备ID`) in more than one sheet likely has the same business meaning, so present a per-sheet `#account_id`/`#distinct_id` view and ask whether the same column maps to the same system field everywhere — a device ID used as `#distinct_id` in the event sheet but left as a differently-named plain property in the profile sheet silently breaks anonymous-user linkage, so surface the mismatch and let the user decide rather than letting it pass silently.
16
16
  3. **`#time`.** Confirm the time column and the source timezone (an IANA name from the user/project context). Leave `time_format` unset unless auto-detection failed (US/EU ambiguous dates). Also confirm the data's `#zone_offset` — the whole-hour UTC offset AE applies to those times, emitted inside `properties` (not top-level). The property's value is a number, so ask the user in numeric terms ("is the offset 8, i.e. UTC+8?") and set `zone_offset_value` to that integer; when a column carries the offset per row, set `zone_offset_field` instead. The two are mutually exclusive.
17
17
  4. **`#event_name` (track only).** Confirm the event-name column, or a reviewed `default_event_name` when no column exists.
18
- 5. **`#ip` / `#uuid` (optional).** Ask only when the data has an IP- or UUID-like column; map it via `ip_field`/`uuid_field`, otherwise skip.
18
+ 5. **`#ip` / `#uuid` (optional).** Ask only when the data has an IP- or UUID-like column; map it via `ip_field`/`uuid_field`, otherwise skip. `#ip` is event data only and must be a valid IPv4/IPv6 address (a private/LAN IP is kept but reported — AE cannot geolocate it); `#uuid` must be a standard 36-character UUID. A value that violates the spec is dropped from that row only (`INVALID_IP` / `INVALID_UUID`) — the row itself is kept. The program never auto-generates a `#uuid`.
19
19
 
20
20
  When an `#event_name`, `#account_id`, or `#distinct_id` column's values do not satisfy AE naming rules (pure Chinese, uppercase, spaces), do not stop — scan the distinct values, list them to the user, and ask for one AE-name replacement each; record the pairs in `value_mapping` (see [UE mapping](ue-mapping.md)). A value with no matching key keeps its original text and fails validation, so confirm every distinct value is covered or excluded. The same mechanism applies to a property column via that entry's own `value_mapping`.
21
21
 
@@ -24,18 +24,18 @@ Then confirm the property set with the user before saving the mapping. Present e
24
24
  Walk the table item by item:
25
25
 
26
26
  1. **Auto-renamed targets.** Recommendation sanitizes column names (camelCase split, illegal chars → `_`, digit-leading → `field_`, nothing recognizable → `field_N`). Show each source → target; ask for a manual name for every `field_N` fallback and rewrite the `target`.
27
- 2. **Types.** Confirm each inferred type (`number`/`string`/`boolean`/`datetime`/`list`/`object`). ID-like columns stay `string` even when numeric; `object`/`list` columns carry `transform: 'json'`. A type is locked on first receipt, so review it before committing. If the user changes a type, warn how many rows would fail coercion — `convert` reports each failing row with its error code, so run it with the tentative mapping and confirm the quarantined count before finalizing.
27
+ 2. **Types.** Confirm each inferred type (`number`/`string`/`boolean`/`datetime`/`list`/`object`/`array_row`). ID-like columns stay `string` even when numeric; `object`/`array_row`/`list` columns carry `transform: 'json'`. A scalar array is `list` (→ AE `array_string`); an object array is `array_row` (→ AE `array_row`). A type is locked on first receipt, so review it before committing. If the user changes a type, warn how many rows would fail coercion — `convert` reports each failing row with its error code, so run it with the tentative mapping and confirm the quarantined count before finalizing.
28
28
  3. **Exclusions.** Ask which columns to drop and record them as `exclude_columns`.
29
29
 
30
30
  Before saving, present the complete mapping for a final sign-off on one grouped page — system fields (`#type`/mode, `#account_id`, `#distinct_id`, `#time` + source timezone, `#zone_offset`, `#event_name`, `#ip`/`#uuid`), then the property table (source column → target AE name → type), then excluded columns — and wait for an explicit yes. Never dump the raw mapping JSON at the user.
31
31
 
32
- Save the reviewed `ae-local-data-mapping/v1` JSON locally. Convert into a new output directory:
32
+ Save the reviewed `ae-data-integration-mapping/v1` JSON to `.ae-cli/data-integration/mapping.json` (project workspace). Convert into a new output directory:
33
33
 
34
34
  ```bash
35
35
  ae-cli data-integration convert \
36
36
  --input-file '<path>' \
37
- --mapping '<mapping.json>' \
38
- --output-dir '.ae-cli/data-integration/<run-id>'
37
+ --mapping '.ae-cli/data-integration/mapping.json' \
38
+ --output-dir '.ae-cli/data-integration/runs/<run-id>'
39
39
  ```
40
40
 
41
41
  For multiple files, repeat `--input-file` and pass a wildcard mapping plus `--type-resolutions` when inspect reported conflicts:
@@ -45,10 +45,10 @@ ae-cli data-integration convert \
45
45
  --input-file '<a.csv>' --input-file '<b.csv>' \
46
46
  --mapping '<wildcard-mapping.json>' \
47
47
  --type-resolutions '<resolutions.json>' \
48
- --output-dir '.ae-cli/data-integration/<run-id>'
48
+ --output-dir '.ae-cli/data-integration/runs/<run-id>'
49
49
  ```
50
50
 
51
- The command never modifies the source. Inspect `manifest.json`; summarize valid and quarantined counts and the block reason. Do not expose rows from `invalid.rows.jsonl` unless the user specifically asks to inspect the local quarantine.
51
+ The command never modifies the source. Inspect `manifest.json`; summarize valid and quarantined counts and the block reason. If `manifest.output.skipped_fields` is present, tell the user how many `#ip`/`#uuid` values were invalid and dropped (the rows were otherwise kept); if `manifest.output.lan_ip_records` is present, tell the user that many `#ip` values are private/LAN addresses that AE cannot geolocate. Neither blocks the manifest. If `manifest.output.flatten_misses` is present — or a stderr `Warning: flatten rule "X" did not materialize for N row(s).` fires — a `flatten_rules` path missed some rows: re-check the dot path against the source shape (or confirm the column is legitimately optional); the rows are otherwise kept. A stderr `Warning: … column count different from the header row …` means ragged CSV/TSV rows were tolerated (extra fields dropped, missing fields treated as empty) — surface it as a data-quality note. A blocked manifest whose reason is `The source contained no data rows.` means the file had zero data rows; do not re-run the same command on it. Do not expose rows from `invalid.rows.jsonl` unless the user specifically asks to inspect the local quarantine. For the full failure taxonomy and how to respond, see [error handling](error-handling.md).
52
52
 
53
53
  ## Re-report only the failed rows (salvage loop)
54
54
 
@@ -59,7 +59,7 @@ ae-cli data-integration convert \
59
59
  --input-file '<same-source>' \
60
60
  --mapping '<fixed-mapping.json>' \
61
61
  --salvage-from '<run-dir>/invalid.rows.jsonl' \
62
- --output-dir '.ae-cli/data-integration/<run-id>-salvage'
62
+ --output-dir '.ae-cli/data-integration/runs/<run-id>-salvage'
63
63
  ```
64
64
 
65
65
  This emits a `valid.ue.jsonl` containing only the rows that now pass, plus a new `invalid.rows.jsonl` with whatever still fails. Fixes are rarely one-shot, so repeat: feed each round's `invalid.rows.jsonl` into the next `--salvage-from` until no rows fail or the user stops. Each round's `valid.ue.jsonl` is disjoint from earlier rounds', so upload each round independently (same confirmation and `--allow-clean-subset` gates as a normal upload). `--salvage-from` is single-file only and the source must be the same file that produced the quarantine.
@@ -1,12 +1,12 @@
1
1
  # UE mapping contract
2
2
 
3
- The mapping version is `ae-local-data-mapping/v1`.
3
+ The mapping version is `ae-data-integration-mapping/v1`.
4
4
 
5
5
  Required structure:
6
6
 
7
7
  ```json
8
8
  {
9
- "version": "ae-local-data-mapping/v1",
9
+ "version": "ae-data-integration-mapping/v1",
10
10
  "source": {
11
11
  "sha256": "<source-sha256>",
12
12
  "format": "csv",
@@ -54,10 +54,25 @@ Top-level `#` fields map from named source columns. The user confirms each mappi
54
54
  | `#account_id` | Login/account ID (database `user_id`, phone, member ID) — identifies authenticated users | At least one of `#distinct_id` / `#account_id` |
55
55
  | `#time` | Event/profile occurrence time; AE bins data by it | Yes |
56
56
  | `#event_name` | What the user did (`purchase`, `login`) | Only for `track` |
57
- | `#ip` | Client IP; AE resolves geo from it | No |
58
- | `#uuid` | Short-window deduplication ID | No |
57
+ | `#ip` | Client IP; AE resolves geo from it. Event data only | No |
58
+ | `#uuid` | Standard 36-character UUID uniquely identifying the record (dedup). Both event and user data | No |
59
59
 
60
- `#zone_offset` is a preset property that tells AE the data's UTC offset (whole hours, -12..14) so `#time` is interpreted correctly. Unlike the fields above it lives **inside `properties`**, not at the top level. Provide it via `zone_offset_value` (a whole-hour integer such as `8` for UTC+8; an IANA name is also accepted and resolved to its offset at conversion time — sub-hour zones round to the nearest whole hour and DST zones reflect the offset then in effect, so historical data crossing a DST boundary should use `zone_offset_field`) or `zone_offset_field` (a source column carrying the offset per row); the two are mutually exclusive. Rows whose `zone_offset_field` value is missing or not an integer in -12..14 are quarantined.
60
+ `#zone_offset` is a preset property that tells AE the data's UTC offset (whole hours, -12..14) so `#time` is interpreted correctly. Unlike the fields above it lives **inside `properties`**, not at the top level. Provide it via `zone_offset_value` (a whole-hour integer such as `8` for UTC+8; an IANA name is also accepted and resolved to its offset at conversion time — sub-hour zones round to the nearest whole hour and DST zones reflect the offset then in effect, so historical data crossing a DST boundary should use `zone_offset_field`) or `zone_offset_field` (a source column carrying the offset per row); the two are mutually exclusive. Rows whose `zone_offset_field` value is missing or not an integer in -12..14 are quarantined. `#zone_offset` is event data only — user profile records never carry it.
61
+
62
+ ## Value specifications
63
+
64
+ Each mapped system field carries a value spec, enforced at both inspect (warnings) and convert (row handling). A row error quarantines the whole row; a field skip drops only that field and keeps the row.
65
+
66
+ | Field | Value spec | On violation (convert) |
67
+ | --- | --- | --- |
68
+ | `#account_id` / `#distinct_id` | Non-empty string, at most 128 characters | Row error `MISSING_USER_ID` (absent) / `USER_ID_TOO_LONG` (>128) |
69
+ | `#event_name` | `^[a-z][a-z0-9_]{0,49}$` (lowercase snake_case, letter-leading, ≤50 chars) | Row error `INVALID_EVENT_NAME` |
70
+ | `#time` | One of the supported formats, within 3 years back / 3 days forward | Row error `INVALID_TIME` / `TIME_OUT_OF_RANGE` |
71
+ | `#ip` | Valid IPv4 or IPv6. Event data only | Field skip `INVALID_IP`; a private/LAN IP is kept and reported — AE cannot geolocate it |
72
+ | `#zone_offset` | Integer -12..14 (or an IANA name for `zone_offset_value`). Event data only | Row error `INVALID_ZONE_OFFSET` |
73
+ | `#uuid` | Standard 36-character UUID (`xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx`). Both data kinds | Field skip `INVALID_UUID` |
74
+
75
+ A field skip is not a row failure: the row is kept with its other fields, and the count is reported in `manifest.output.skipped_fields` so the agent tells the user at the end. `#uuid` is never auto-generated — it comes only from a mapped source column.
61
76
 
62
77
  Mapping keys: `account_id_field`, `distinct_id_field`, `time.field`, `event_name_field` (or a reviewed `default_event_name`), `ip_field`, `uuid_field`. Inspect surfaces every identity-shaped column in `identity_candidates` (name, `account`/`distinct` kind, unique and missing ratios) so the agent can present all candidates for confirmation. When the required identity column is absent, use an explicit `account_id_value`/`distinct_id_value` placeholder or `random_pool` — a user decision, never invented.
63
78
 
@@ -86,16 +101,20 @@ All fields below are optional and are explicit user decisions — never invent t
86
101
  | `account_id_value` / `distinct_id_value` | `string` | Fixed placeholder identity (≤128 chars). Applies to every row when the corresponding `*_field` is absent; when the field exists, fills only the rows whose column is empty |
87
102
  | `random_pool` | `{ account_ids?: string[], distinct_ids?: string[] }` | Synthesizes a random identity when the source field is absent |
88
103
  | `exclude_columns` | `string[]` | Source columns skipped when building properties |
89
- | `flatten_rules` | `{ outColumn: 'dot.path' }` | Nested flatten map. NDJSON/JSON paths are from the record root (`user_info.name`); CSV/TSV paths are `<column>.<cell-relative path>` into a JSON-encoded object cell (`user_profile.name`) — add the source column to `exclude_columns` when flattening it |
104
+ | `flatten_rules` | `{ outColumn: 'dot.path' }` | Nested flatten map. NDJSON/JSON paths are from the record root (`user_info.name`); CSV/TSV/Excel paths are `<column>.<cell-relative path>` into a JSON-encoded object cell (`user_profile.name`) — add the source column to `exclude_columns` when flattening it. `flatten_rules` only materializes the out column in the row; it does **not** emit it — add a `properties` entry with `source` = the out column to actually produce the property. An out column that fails to materialize for some rows (missing path, non-JSON cell, or a path without a `<column>.` prefix) is counted per out-column in `manifest.output.flatten_misses` and named in a convert stderr warning — fix the path or confirm the column is legitimately optional |
90
105
  | `headers` | `string[]` | User-confirmed column names for a headerless file; presence means the first row is data. Never use inspect's `col_1..col_N` placeholders — infer names from each column's values, confirm them with the user, then write them here |
91
106
  | `missing_time` | `'now'` | Fill a missing/empty `#time` with the current time, for user-profile rows only (explicit user decision; track rows are never filled) |
92
- | `ip_field` | `string` | Source column emitted as the top-level `#ip` system field (client IP; AE resolves geo) |
93
- | `uuid_field` | `string` | Source column emitted as the top-level `#uuid` system field (short-window deduplication ID) |
107
+ | `ip_field` | `string` | Source column emitted as the top-level `#ip` system field (client IP; AE resolves geo). Event data only; a value that is not valid IPv4/IPv6 is dropped from that row, and a private/LAN IP is kept but reported |
108
+ | `uuid_field` | `string` | Source column emitted as the top-level `#uuid` system field (standard 36-character UUID, both data kinds). A non-UUID value is dropped from that row |
94
109
  | `zone_offset_value` | `number` (integer -12..14) or IANA `string` | Emits the fixed `#zone_offset` preset property inside `properties`. An IANA name resolves to its integer UTC offset |
95
110
  | `zone_offset_field` | `string` | Source column whose per-row integer value (-12..14) is emitted as `#zone_offset`; missing/non-integer rows are quarantined. Mutually exclusive with `zone_offset_value` |
96
111
  | `event_meta` | `{ <event-name>: { desc?: string, tag?: string } }` | Per-event business description and `event_tag` for the tracking plan, keyed by AE event name. Inferred from the data and user context; the user supplies anything not inferable — never leave them empty |
97
112
 
98
- Each `properties` entry may also carry `value_mapping` (per-property exact-key replacement), `transform` (one of `stringify`, `number`, `boolean`, `json`), `time_format` (only meaningful for `type: 'datetime'`), and `desc` (business description for the tracking plan; inferred, or user-provided when not inferable). Container columns (`object`/`list`) whose values arrive as JSON text — JSON-encoded CSV cells, or flattened NDJSON leaves — must set `transform: 'json'` so conversion parses the text back into a native object/array; it is a safe no-op when the value already arrived native.
113
+ Each `properties` entry may also carry `value_mapping` (per-property exact-key replacement), `transform` (one of `stringify`, `number`, `boolean`, `json`), `time_format` (only meaningful for `type: 'datetime'`), and `desc` (business description for the tracking plan; inferred, or user-provided when not inferable). Container columns (`object`/`array_row`/`list`) whose values arrive as JSON text — JSON-encoded CSV/TSV/Excel cells, or flattened NDJSON leaves — must set `transform: 'json'` so conversion parses the text back into a native object/array; it is a safe no-op when the value already arrived native.
114
+
115
+ A `properties` entry whose `target` is dotted (`parent.child`) is a plan-only sub-property declaration: `parent` must name an `object`/`array_row` entry in the same mapping, the child must be scalar or `list`, and the child's `source` must equal the parent's `source` — the nested value lives inside the parent, which conversion emits once and never reads the child's column. A kept whole `object`/`array_row` therefore carries one dotted entry per child alongside the parent entry.
116
+
117
+ Property types are `string`, `number`, `boolean`, `datetime`, `list`, `object`, and `array_row`. `list` is a scalar array (`["a","b"]`, → AE `array_string`); `object` is a one-to-one object (`{...}`); `array_row` is an object array (`[{...}]`, one-to-many). A mapping expresses an object's/object array's sub-properties as dotted `target`s (`user_info.name`, `orders.sku`); `data-integration plan` maps them one-to-one to `parent.child` plan properties (each child's `display_name`/`desc` default to its leaf segment). In the tracking plan, `object` and `array_row` properties must each have at least one `parent.child` sub-property, and every sub-property must itself be scalar or `array_string` — a nested object/object-array cannot stay nested and must be flattened further to scalar leaves. `list` (→ `array_string`) is a leaf and takes no children. `ae-cli data-integration plan` enforces this locally: a composite property with no children, or a child that is `object`/`array_row`, fails fast with `LOCAL_DATA_PLAN_INVALID_DRAFT` before any upload.
99
118
 
100
119
  ## Time formats
101
120
 
@@ -130,13 +149,16 @@ Values with an explicit offset or `Z` are parsed by the JavaScript `Date` constr
130
149
  - `#zone_offset`, when set, is a whole-hour integer in -12..14 (or an IANA name/column that resolves to one) and is emitted inside `properties`, never at the top level.
131
150
  - Event/property names are lowercase snake_case, begin with a letter, and are at most 50 characters.
132
151
  - Target property names are unique and do not collide with UE system fields.
133
- - Types are one of `string`, `number`, `boolean`, `datetime`, `list`, or `object`.
152
+ - Types are one of `string`, `number`, `boolean`, `datetime`, `list`, `object`, or `array_row`.
134
153
  - Text is at most 2 KB; numbers stay within -9E15..9E15.
135
154
  - Lists contain at most 500 strings (255 bytes each) or 500 objects.
136
155
  - Objects contain at most 100 legal sub-properties; nested values follow the same type limits.
156
+ - No `object`/`array_row` survives into the tracking plan: AE requires each to have a `parent.child` sub-property (scalar or `array_string` only), and a mapping cannot express one, so nested objects/object-arrays are flattened to scalar leaves first (a kept-whole container fails `data-integration plan` fast).
137
157
  - A conversion rule does not hide a real type conflict.
138
158
  - Event/profile times fall within the receiver window: previous 3 years through next 3 days.
139
- - Do not fabricate UUIDs, identities, times, or events; `#ip`/`#uuid` map from named source columns only (`ip_field`/`uuid_field`).
159
+ - Do not fabricate UUIDs, identities, times, or events; `#ip`/`#uuid` map from named source columns only (`ip_field`/`uuid_field`). `#uuid` is never auto-generated.
160
+ - `#uuid` values are standard 36-character UUIDs and `#ip` values are valid IPv4/IPv6; invalid ones are dropped from the row only (`skipped_fields`), never quarantined.
161
+ - `#ip` and `#zone_offset` appear on `track` records only; user data carries neither.
140
162
  - `value_mapping`, `random_pool`, and fixed `account_id_value`/`distinct_id_value` came from an explicit user decision and match the actual distinct values/columns.
141
163
 
142
164
  `user_set` output for the same identity is ordered by time so receiver application order is deterministic. Conversion applies whole-row quarantine: any error on a row — identity, time, event, record type, or a single property (type coercion, size/limit) — drops the entire row. The row is written to `invalid.rows.jsonl` with its error codes, counted in `manifest.output.invalid_records`, and the manifest is blocked until reviewed. The failed rows are re-reportable without re-sending valid rows: fix the mapping, then run `convert --input-file <same-source> --mapping <fixed-mapping> --salvage-from <invalid.rows.jsonl>`. The salvage run re-processes only the listed row numbers against the same source, emits a `valid.ue.jsonl` containing only the newly fixed rows, and writes a new `invalid.rows.jsonl` with whatever still fails — so the loop repeats (feeding each round's `invalid.rows.jsonl` into the next `--salvage-from`) until no rows fail or the user stops.
@@ -229,6 +229,8 @@ When experiment mode is enabled (`context.enableExp=true` or draft `expConfig.en
229
229
  - capability `engage-task.task.build-save-guide` enriches `handoff.reqTemplate.channelConfig.groupContentList`
230
230
  so each entry carries `expGroupName`, `expGroupType`, `percentageInExperiment`, `order`, and `contentList`
231
231
  - do not drop those association fields when filling content; they must stay aligned with `expConfig.expGroupList`
232
+ - copy the complete group tuple (`expGroupName`, `expGroupType`, `percentageInExperiment`, `order`)
233
+ into both lists; matching only by list position is not sufficient
232
234
 
233
235
  ### 4.8 `fieldRules`
234
236
 
@@ -300,6 +302,11 @@ Apply this rule only to
300
302
  `completionIndicatorDef.completionIndicators[].eventDefinition.filters`. Trigger-event filters have
301
303
  their own scenario rules and are not subject to this completion-filter restriction.
302
304
 
305
+ For `completionIndicatorType=0`, also read and preserve `requiredMainGoalFields` and
306
+ `touchCycleRule`. A valid main goal includes `touch_cycle_num` and `touch_cycle_num_unit`; use `1`
307
+ and `day` when no custom completion window is requested. Do not rely on static `--validate` alone
308
+ because the save service performs this additional business validation.
309
+
303
310
  ### 4.9 `handoff`
304
311
 
305
312
  This is the final section before `save_task`.
@@ -336,6 +343,8 @@ Recommended usage pattern:
336
343
  definition directly
337
344
  7. omit `clientConfig.clientQp`; partial updates preserve the server-authored value
338
345
  8. call `engage-task task save`
346
+ 9. after an experiment save succeeds, call `engage-task task get` and verify that both
347
+ `exp_config.exp_group_list` and `group_content_list` contain the expected group tuples
339
348
 
340
349
  ---
341
350