@alvera-ai/platform-sdk 0.17.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/.agent/AGENTS.md +82 -144
  2. package/.agent/account_management.md +2 -2
  3. package/.agent/action_logs.md +4 -4
  4. package/.agent/ai_agents.md +28 -21
  5. package/.agent/ai_sandbox.md +49 -39
  6. package/.agent/connected_apps.md +3 -3
  7. package/.agent/cookbook/_fixtures/README.md +1 -1
  8. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_generic_table.liquid +1 -1
  9. package/.agent/cookbook/_fixtures/organic-marketing/_lead_submissions_foundation_legal_entity.liquid +80 -0
  10. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_mdm.liquid +2 -1
  11. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_generic_table.liquid +57 -0
  12. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_legal_entity.liquid +30 -0
  13. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_mdm.liquid +44 -0
  14. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_generic_table.liquid +57 -0
  15. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_legal_entity.liquid +36 -0
  16. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_mdm.liquid +41 -0
  17. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_generic_table.liquid +70 -0
  18. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_legal_entity.liquid +52 -0
  19. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_mdm.liquid +42 -0
  20. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_generic_table.liquid +38 -0
  21. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_legal_entity.liquid +48 -0
  22. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_mdm.liquid +49 -0
  23. package/.agent/cookbook/organic-marketing.md +2801 -0
  24. package/.agent/cookbook/payments-compliance.md +2180 -0
  25. package/.agent/cookbook/primary-care.md +2175 -0
  26. package/.agent/cookbook/subscription-saas.md +2403 -0
  27. package/.agent/data_activation_clients.md +65 -52
  28. package/.agent/datalakes.md +338 -171
  29. package/.agent/errors.md +3 -3
  30. package/.agent/generic_tables.md +151 -62
  31. package/.agent/interoperability_contracts.md +57 -22
  32. package/.agent/mdm.md +136 -153
  33. package/.agent/messages.md +36 -34
  34. package/.agent/mock-services.md +1 -1
  35. package/.agent/mutations.md +2 -2
  36. package/.agent/templates.md +14 -13
  37. package/.agent/tools.md +63 -21
  38. package/.agent/type_naming.md +13 -13
  39. package/.agent/workflows.md +99 -53
  40. package/README.md +2 -2
  41. package/dist/bin/platform-sdk.mjs +33 -47
  42. package/dist/bin/platform-sdk.mjs.map +1 -1
  43. package/dist/index.d.mts +565 -379
  44. package/dist/index.d.mts.map +1 -1
  45. package/dist/index.mjs +494 -59
  46. package/dist/index.mjs.map +1 -1
  47. package/package.json +4 -3
  48. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +0 -88
  49. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +0 -47
  50. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +0 -24
  51. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +0 -38
  52. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_compliance_screening.liquid +0 -59
  53. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_mdm.liquid +0 -36
  54. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_mdm.liquid +0 -30
  55. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_payment_account.liquid +0 -55
  56. package/.agent/cookbook/_fixtures/subscription/_customers_subscription_mdm.liquid +0 -20
  57. package/.agent/cookbook/_setup/foundation.md +0 -359
  58. package/.agent/cookbook/_setup/healthcare.md +0 -361
  59. package/.agent/cookbook/_setup/payments.md +0 -365
  60. package/.agent/cookbook/_setup/subscription.md +0 -364
  61. package/.agent/cookbook/action-status-updaters.md +0 -278
  62. package/.agent/cookbook/ai-agent-invoke.md +0 -279
  63. package/.agent/cookbook/appointment-review-sms-workflow.md +0 -801
  64. package/.agent/cookbook/birthday-greeting-sms-trigger.md +0 -696
  65. package/.agent/cookbook/bulk-ingest.md +0 -302
  66. package/.agent/cookbook/contact-us-triage-with-llm.md +0 -663
  67. package/.agent/cookbook/dunning-sms-for-delinquent.md +0 -659
  68. package/.agent/cookbook/generic-tables.md +0 -244
  69. package/.agent/cookbook/invite-team.md +0 -200
  70. package/.agent/cookbook/kyc-notification-on-account-activation.md +0 -661
  71. package/.agent/cookbook/marketing-campaign-send.md +0 -1044
  72. package/.agent/cookbook/paginated-restapi-poller.md +0 -383
  73. package/.agent/cookbook/rest-fetch.md +0 -273
  74. package/.agent/cookbook/sanctions-screening-with-agent-review.md +0 -773
  75. package/.agent/cookbook/score-leads-with-llm-categorization.md +0 -665
  76. package/.agent/cookbook/system-templates.md +0 -165
  77. package/.agent/cookbook/talk-to-data.md +0 -178
  78. package/.agent/cookbook/triage-prospects-by-priority.md +0 -571
  79. package/.agent/cookbook/welcome-sms-for-customers.md +0 -647
  80. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/memorandum-of-association-01.png +0 -0
  81. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/sample_two_page.pdf +0 -0
  82. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/_customers_subscription_customer.liquid +0 -0
  83. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/stripe_customers_batch1.csv +0 -0
package/.agent/mdm.md CHANGED
@@ -2,8 +2,7 @@
2
2
 
3
3
  Master Data Management is the platform's entity-resolution layer.
4
4
  It binds every inbound row to a canonical **subject** record (the
5
- domain's primary entity — e.g. `patient` for healthcare-domain
6
- datalakes, `legal_entity` for foundation-domain datalakes) so
5
+ platform's one identity dataset, the legal entity) so
7
6
  that downstream datasets, workflows, and agent prompts reference
8
7
  a single, deduplicated identity rather than the raw inbound
9
8
  identifiers each upstream system carries.
@@ -22,7 +21,7 @@ see §5 Lifecycle.
22
21
  MDM is **Datalake-DB-resident** — the parent datalake must be
23
22
  `status: 'ready'` before any MDM verb is called, since
24
23
  resolution / verification both look up against the datalake's
25
- regulated master records.
24
+ master records.
26
25
 
27
26
  ## 1. Wire shape — `.verify`
28
27
 
@@ -33,72 +32,64 @@ recipient) and at runtime (connected apps confirming the form
33
32
  submitter before posting tracking updates).
34
33
 
35
34
  Every MDM subject is ultimately a **person** or a **company**,
36
- and the request type is the superset of every data domain's
37
- identity vocabulary organized along that ontology. You send
38
- `subject_id` plus the identity fields **in your datalake's own
39
- domain vocabulary** — the platform validates them with that
40
- domain's verifier, which knows only its own fields.
35
+ and the request type is that ontology's identity vocabulary. You
36
+ send `legal_entity_id` plus the identity fields you hold.
37
+
38
+ There is ONE field set. GH-859 removed data domains, and with
39
+ them the per-domain verifiers that each knew only their own
40
+ vocabulary — `given_name`, `family_name`, `birth_date` and
41
+ `gender` are gone, not renamed variants to pick between.
41
42
 
42
43
  ```
43
- data_domain identity fields the verifier matches
44
- ─────────── ────────────────────────────────────
45
- healthcare given_name family_name birth_date gender
46
- phone email identifiers
47
- foundation legal_entity_type first_name middle_name
48
- last_name preferred_name date_of_birth
44
+ MDMVerifyRequest every field the verifier matches
45
+ ──────────────── ────────────────────────────────
46
+ required legal_entity_id
47
+ individual first_name middle_name last_name
48
+ preferred_name date_of_birth
49
49
  citizenship_country nationality
50
- business_name doing_business_as_names
50
+ business business_name doing_business_as_names
51
51
  date_formed jurisdiction_country
52
- phone email identifiers
53
- subscription customer_type name tax_id phone email
54
- identifiers
55
- service_commerce consumer_type name phone email identifiers
56
- core_banking party_type given_name family_name
57
- company_name birth_date phone email
58
- identifiers
59
- payments account_holder_type given_name family_name
60
- company_name birth_date phone email
61
- identifiers (each entry additionally
62
- requires id_type)
52
+ either legal_entity_type phone email
53
+ identifiers addresses
63
54
  ```
64
55
 
65
- A subscription example — the customer's own vocabulary:
56
+ An individual example:
66
57
 
67
58
  ```typescript
68
59
  import type {
69
- MDMVerifyRequest,
70
- MDMVerifyResponse,
60
+ MdmVerifyRequest,
61
+ MdmVerifyResponse,
71
62
  } from '@alvera-ai/platform-sdk'
72
63
 
73
64
  const { data: result } = await api.mdm.verify(
74
65
  tenantSlug,
75
66
  datalakeSlug,
76
67
  {
77
- subject_id: subjectUuid, // required, every domain
78
- name: 'Test Priya Anand', // subscription: fuzzy full-name match
79
- tax_id: '123-45-6789', // subscription: exact match
68
+ legal_entity_id: legalEntityUuid, // required
69
+ first_name: 'Priya', // fuzzy
70
+ last_name: 'Anand', // fuzzy
80
71
  },
81
72
  )
82
73
  ```
83
74
 
84
- And a healthcare one — FHIR-shaped demographics:
75
+ And one keyed on an external identifier:
85
76
 
86
77
  ```typescript
87
78
  const { data: result } = await api.mdm.verify(
88
79
  tenantSlug,
89
80
  datalakeSlug,
90
81
  {
91
- subject_id: subjectUuid,
92
- family_name: 'Garcia', // fuzzy
93
- given_name: 'Maria', // fuzzy
94
- birth_date: '1978-11-03', // component-fuzzy
82
+ legal_entity_id: legalEntityUuid,
83
+ last_name: 'Garcia', // fuzzy
84
+ first_name: 'Maria', // fuzzy
85
+ date_of_birth: '1978-11-03', // component-fuzzy
95
86
  identifiers: [ // exact
96
87
  { system: 'http://hospital.org/mrn', value: 'MRN12345' },
97
88
  ],
98
89
  },
99
90
  )
100
91
  // result.status === 'verified' | 'not_verified'
101
- // result.subject_id === body.subject_id
92
+ // result.legal_entity_id === body.legal_entity_id
102
93
  // result.verified_at — ISO 8601 timestamp
103
94
  // when status === 'not_verified':
104
95
  // result.errors — { <field_name>: [reason, ...] } per-field failure detail
@@ -107,8 +98,8 @@ const { data: result } = await api.mdm.verify(
107
98
  The response is **polymorphic on `status`**:
108
99
 
109
100
  ```
110
- status === 'verified' → 200 { subject_id, verified_at }
111
- status === 'not_verified' → 422 { subject_id, verified_at, errors }
101
+ status === 'verified' → 200 { legal_entity_id, verified_at }
102
+ status === 'not_verified' → 422 { legal_entity_id, verified_at, errors }
112
103
  ```
113
104
 
114
105
  A `not_verified` response is a successful identity check that
@@ -116,89 +107,80 @@ returned a negative result — it arrives as a 422 whose body
116
107
  still carries the full envelope. Distinguish it from the other
117
108
  rejections (404 for missing subject, 401 for missing auth).
118
109
 
119
- ### Subject per data domain
120
-
121
- Every datalake has a single canonical **subject entity** — the
122
- domain's primary identity that all other rows attribute to. The
123
- verify call resolves against that subject. The subject's wire
124
- name follows the dataset's `resource_type` (see
125
- `interoperability_contracts.md` §2 `resource_type`):
110
+ ### One subject: the legal entity
126
111
 
127
- ```
128
- data_domain subject resource_type
129
- ─────────── ─────────────────────
130
- healthcare patient
131
- core_banking party
132
- payments legal_entity
133
- subscription customer
134
- service_commerce consumer
135
- trading trading_account
136
- foundation legal_entity
137
- ```
112
+ Every datalake resolves against the same canonical subject — the
113
+ **legal entity**. `legal_entity` and `beneficial_owner` are the
114
+ only identity datasets the platform ships; everything else a
115
+ build needs is a generic table it authors itself.
138
116
 
139
- (`legal_entity` is the subject for both `payments` and
140
- `foundation` — different domain context, same canonical entity
141
- shape.)
117
+ That is the whole of it now. There is no per-domain subject to
118
+ look up, no `patient` / `party` / `customer` / `consumer` /
119
+ `trading_account` to map onto, and no table of which domain
120
+ resolves to which entity. GH-859 removed the domains and the
121
+ subjects with them.
142
122
 
143
- Generic tables in a datalake automatically carry a foreign-key
144
- column to their subject (e.g. `patient_id` in a healthcare-
145
- domain generic table). Querying that column scopes any custom
146
- dataset to subjects the verify call has confirmed.
123
+ Generic-table rows carry `legal_entity_id`, populated by MDM
124
+ resolution when the row's contract supplies an `mdm_input_config`
125
+ — not a per-domain foreign key like `patient_id`. Querying that
126
+ column scopes any custom dataset to resolved entities. You do not
127
+ declare the column; every generic-table row struct already has it.
147
128
 
148
129
  ## 2. Rules the type cannot encode
149
130
 
150
- ### `subject_id` is required; identity fields per YOUR domain
131
+ ### `legal_entity_id` is required; at least one identity field with it
151
132
 
152
- `subject_id` (the master record to verify against) is the only
153
- required field. The TypeScript type carries every domain's
154
- identity fields as optional properties, but the datalake's
155
- verifier casts **only the fields its domain knows** (§1 table)
156
- and requires **at least one** of them at runtime:
133
+ `legal_entity_id` (the master record to verify against) is the
134
+ only required field. Every other property is optional, and the
135
+ verifier requires **at least one** of them at runtime:
157
136
 
158
- - a body with only `subject_id` → 422 with
137
+ - a body with only `legal_entity_id` → 422 with
159
138
  `errors.base: ["at least one verification field is required"]`
160
- - a body whose identity fields all belong to a *different*
161
- domain's vocabulary (e.g. healthcare's `family_name` sent to an
162
- subscription datalake) is treated the same way — the
163
- foreign fields are **ignored**, not rejected, so the request
164
- falls into the same at-least-one 422.
165
139
 
166
- ### Verification is per-field, with domain-scoped match rules
140
+ A body carrying the retired vocabulary — `given_name`,
141
+ `family_name`, `birth_date`, `gender` — falls into that same 422,
142
+ and this is the trap worth knowing: unknown keys are **ignored,
143
+ not rejected**. Nothing tells you the field was dropped. Send
144
+ only those and you get the at-least-one error for a request that
145
+ looks full; send one of them alongside a known field and it
146
+ verifies on the known field alone, silently ignoring the rest.
147
+
148
+ ### Verification is per-field
167
149
 
168
- Each supplied identity field the domain knows is checked
169
- independently against the subject's regulated master record.
150
+ Each supplied identity field is checked independently against
151
+ the legal entity's master record.
170
152
  String names are fuzzy-matched, dates component-fuzzy-matched,
171
153
  and structured identifiers exact-matched:
172
154
 
173
155
  ```
174
- name fields Jaro-Winkler fuzzy match (thresholds ~0.80-0.85;
175
- core_banking company_name is EXACT)
156
+ name fields Jaro-Winkler fuzzy match (thresholds ~0.80-0.85)
176
157
  date fields component-wise fuzzy match ≥ 0.80
177
158
  identifiers exact (system, value) match
178
- tax_id exact match (subscription)
159
+ addresses per-component exact, case- and whitespace-
160
+ insensitive; at least one address must match
179
161
  ```
180
162
 
181
- All supplied known fields must pass for the overall result to be
163
+ All supplied fields must pass for the overall result to be
182
164
  `'verified'`. A single failing field flips the result to
183
165
  `'not_verified'` and surfaces the field name in `errors`.
184
166
 
185
- ### `subject_id` is the UNREGULATED subject row's id
167
+ ### `legal_entity_id` is the legal entity's own id
168
+
169
+ `legal_entity_id` is `legal_entities.id` — the same value every
170
+ generic-table row carries as its own `legal_entity_id`.
186
171
 
187
- `subject_id` is the subject's id in the datalake's
188
- **unregulated** schema — the same value every other unregulated
189
- row carries as `mdm_subject_id`. For a `foundation` or
190
- `payments` datalake that is `legal_entities.id`. The
191
- identity FIELDS are matched against the regulated master record,
192
- but the HANDLE you pass is the unregulated id: supplying the
193
- regulated twin's id returns a 404 "Subject not found", never a
194
- `'not_verified'`.
172
+ There is one lake row per entity now, so there is no public
173
+ handle and regulated twin to choose between: GH-859 collapsed the
174
+ two-tier lake, and masking is infrastructure — the tokenized and
175
+ redacted lakes — rather than a second row with its own id. The
176
+ old failure mode where passing the regulated twin's id returned a
177
+ 404 cannot arise.
195
178
 
196
- If the supplied `subject_id` doesn't resolve to an unregulated
197
- subject row, the response is a 404, not a `'not_verified'`. The
198
- distinction matters:
179
+ If the supplied `legal_entity_id` doesn't resolve, the response
180
+ is a 404, not a `'not_verified'`. The distinction matters:
199
181
 
200
- - `'not_verified'` → "subject exists, identity fields didn't match"
201
- - 404 → "subject doesn't exist at all"
182
+ - `'not_verified'` -> "entity exists, identity fields didn't match"
183
+ - 404 -> "entity doesn't exist at all"
202
184
 
203
185
  A consumer that conflates the two will mis-route an identity
204
186
  challenge as a missing-record error.
@@ -217,13 +199,16 @@ errors object — present only when status === 'not_verified';
217
199
  **Caller-supplied (round-trip on the verified branch).**
218
200
 
219
201
  ```
220
- subject_id required UUID — the master record to verify against
221
- <identity> optional — YOUR domain's fields from the §1 table;
222
- person fields (given/family/first/last
223
- names, birth dates), company fields
224
- (company_name/business_name/name/tax_id),
225
- and universal fields (identifiers, phone,
226
- email, the domain's subject-type axis)
202
+ legal_entity_id required UUID — the master record to verify against
203
+ <identity> optional — any field from the §1 set: person fields
204
+ (first/middle/last/preferred names,
205
+ date_of_birth, citizenship_country,
206
+ nationality), business fields
207
+ (business_name, doing_business_as_names,
208
+ date_formed, jurisdiction_country), and
209
+ fields either kind carries
210
+ (legal_entity_type, phone, email,
211
+ identifiers, addresses)
227
212
  ```
228
213
 
229
214
  **Write-only (Request-only).** None.
@@ -234,8 +219,8 @@ Standard JSON:API envelopes per `errors.md`. Common rejections:
234
219
 
235
220
  | `source.pointer` | Cause |
236
221
  |-------------------------------|--------------------------------------------------------|
237
- | `/subject_id` | missing, or no UNREGULATED subject row has this id (passing the regulated twin's id is the classic cause) |
238
- | `errors.base` | no identity field the domain recognizes was supplied — 422 with `status: "not_verified"` in the body |
222
+ | `/legal_entity_id` | missing, or no legal entity has this id |
223
+ | `errors.base` | no identity field was supplied, or only retired ones (`given_name`, `family_name`, `birth_date`, `gender`) which are ignored — 422 with `status: "not_verified"` in the body |
239
224
  | `/identifiers/0/system` | missing system on an identifier entry |
240
225
  | `/identifiers/0/value` | missing value on an identifier entry |
241
226
 
@@ -243,19 +228,18 @@ The `not_verified` outcome is always a **422 response** carrying
243
228
  the full body envelope; the two shapes differ only in `errors`:
244
229
 
245
230
  - **Identity mismatch (most common)** — `errors` keyed by the
246
- failing field(s) in the domain's vocabulary (e.g.
247
- `errors.family_name` on healthcare, `errors.base` with
248
- "Verification match failed" on subscription). The
249
- subject exists; the supplied identity didn't match.
231
+ failing field(s), e.g. `errors.last_name`, or `errors.base`
232
+ with "Verification match failed". The entity exists; the
233
+ supplied identity didn't match.
250
234
  - **No usable identity fields** — `errors.base` set to
251
235
  `["at least one verification field is required"]`. Verification
252
- was structurally impossible (nothing supplied, or only fields
253
- foreign to the domain).
236
+ was structurally impossible: nothing supplied, or only retired
237
+ fields, which are ignored.
254
238
 
255
- Malformed identifier entries (e.g. payments's required
256
- `id_type` missing) surface the embed failure flattened into the
257
- field's error list — `errors.identifiers:
258
- ["id_type can't be blank"]` — still a 422, never a 500.
239
+ Malformed identifier entries surface the embed failure flattened
240
+ into the field's error list — `errors.identifiers: ["value can't
241
+ be blank"]` — still a 422, never a 500. `system` and `value` are
242
+ the required pair; `id_type` and `country` are optional.
259
243
 
260
244
  Pattern-match on `status` first, then on the HTTP status code.
261
245
  The 404 case (subject doesn't exist) is a separate envelope —
@@ -267,7 +251,7 @@ see §6 gotcha 2.
267
251
 
268
252
  The verify call returns the resolution outcome in the response —
269
253
  no async polling, no job id. The platform performs the fuzzy
270
- match against the regulated master record inline.
254
+ match against the master record inline.
271
255
 
272
256
  ### Other MDM operations are platform-internal
273
257
 
@@ -280,7 +264,7 @@ SDK surface:
280
264
  pipeline before the dataset upsert. The pipeline matches the
281
265
  rendered MDM input against existing subjects (or creates one
282
266
  if no match) and produces an `mdm_output` carried forward to
283
- the regulated upsert. Consumers configure this implicitly via
267
+ the legal-entity upsert. Consumers configure this implicitly via
284
268
  the contract's `mdm_input_config` — see
285
269
  `interoperability_contracts.md` §2 and `data_activation_clients.md`
286
270
  §6.
@@ -311,27 +295,28 @@ through `api.mdm`.
311
295
  explains which fields failed. Gate the downstream "act on
312
296
  this person" path on `result.status === 'verified'`.
313
297
 
314
- 2. **`subject_id` missing → 404, identity mismatch → 422 with a
315
- body.** A verify call against a deleted or never-existed
316
- subject id returns a 404 envelope, NOT a `'not_verified'`
298
+ 2. **`legal_entity_id` missing → 404, identity mismatch → 422
299
+ with a body.** A verify call against a deleted or never-existed
300
+ entity id returns a 404 envelope, NOT a `'not_verified'`
317
301
  response. Consumers that handle only the polymorphic
318
302
  verify body will surface the 404 as an unexpected exception.
319
303
  Wrap the verify call in standard 4xx handling per `errors.md`.
320
304
 
321
- 3. **At least one identity field the domain recognizes is
322
- required.** A body with only `subject_id` — or only fields
323
- from a different domain's vocabulary — returns a 422 with
305
+ 3. **At least one identity field is required.** A body with only
306
+ `legal_entity_id` — or only retired fields like `given_name`
307
+ or `birth_date`, which are ignored — returns a 422 with
324
308
  `errors.base` set to `["at least one verification field is
325
309
  required"]` and `status: "not_verified"` in the body. This is
326
310
  enforced at runtime rather than in the request schema, so the
327
311
  TypeScript type permits the body but the wire rejects it.
328
312
 
329
- 4. **The field set is domain-scoped — consult the §1 table for
330
- YOUR datalake's domain.** The request type is the superset of
331
- every domain's vocabulary; fields outside your domain's set
332
- are silently ignored by its verifier (they don't error, they
333
- just don't count). Sending only ignored fields lands you in
334
- gotcha 3's at-least-one 422.
313
+ 4. **Unknown fields are ignored, not rejected.** There is one
314
+ field set (§1) and no domain scoping, but anything outside it
315
+ — including the retired `given_name` / `family_name` /
316
+ `birth_date` / `gender` — is silently dropped rather than
317
+ erroring. Sending only those lands you in gotcha 3's
318
+ at-least-one 422; sending one alongside a known field verifies
319
+ on the known field alone and tells you nothing.
335
320
 
336
321
  5. **MDM also runs at ingestion — and you don't call it
337
322
  directly.** The `mdm_input_config` on each interoperability
@@ -347,25 +332,23 @@ through `api.mdm`.
347
332
  character-for-character; a `'http://hospital.org/mrn'` vs
348
333
  `'http://hospital.org/mrn/'` mismatch will flip the verifier
349
334
  to `'not_verified'` even when the value is correct.
350
- payments additionally REQUIRES `id_type` on every
351
- identifier entry (e.g. `'digital_identifier'`, `'us_ssn'`,
352
- `'lei'`) — omitting it is a 422 on `errors.identifiers`.
353
-
354
- 7. **Send your domain's NATIVE vocabulary — there is no wire
355
- translation.** Foundation stores and matches `first_name` /
356
- `last_name` / `date_of_birth` (matching `LegalEntity`), so a
357
- foundation verify body must carry exactly those names;
358
- healthcare's `given_name` / `family_name` / `birth_date` are
359
- foreign to foundation and fall into gotcha 3's 422. The same
360
- applies in every direction — the field names in §1's table
361
- are the wire contract per domain, verbatim.
362
-
363
- 8. **`subject_id` is the unregulated legal-entity id (`==
364
- mdm_subject_id`) — the regulated twin's id 404s.** The most
365
- common source of gotcha 2's 404 is a hook that reads the
366
- REGULATED subject row and passes ITS id to `mdm.verify` —
367
- "Subject not found", even though the person plainly exists.
368
- The verify handle is the unregulated row's id, the same value
369
- every other unregulated row carries as `mdm_subject_id` (the
370
- subject row is its own MDM subject). §1's table names each
371
- domain's subject dataset; §2 states the tier rule.
335
+ `system` and `value` are the required pair. `id_type`
336
+ (`'digital_identifier'`, `'us_ssn'`, `'lei'`) and `country`
337
+ are optional — the per-domain rule that once made `id_type`
338
+ mandatory went with the domains.
339
+
340
+ 7. **The field names ARE the wire contract — there is no
341
+ translation.** The platform stores and matches `first_name` /
342
+ `last_name` / `date_of_birth`, matching `LegalEntity`, so a
343
+ verify body must carry exactly those. The retired
344
+ `given_name` / `family_name` / `birth_date` / `gender` are
345
+ ignored and fall into gotcha 3's 422. §1's set is the contract,
346
+ verbatim.
347
+
348
+ 8. **There is no second id to confuse it with.** This gotcha used
349
+ to warn that passing the REGULATED twin's id 404s while the
350
+ person plainly exists. GH-859 collapsed the two-tier lake:
351
+ there is one `legal_entities` row per entity, and masking is
352
+ infrastructure — the tokenized and redacted lakes — not a
353
+ second row with its own id. Kept as an entry because the old
354
+ advice is still in circulation.
@@ -31,7 +31,7 @@ Read it like any dataset: `api.datasets.createUserSearch` +
31
31
  the plural `messages` is the SQL *table* name (used in
32
32
  `executeSql` queries and as this page's title), and passing it as
33
33
  the dataset type 422s with `/resource_type is not valid`. Base SQL
34
- alias: `rm`.
34
+ alias: `m`.
35
35
 
36
36
  ## 1. The status lifecycle — and who writes each hop
37
37
 
@@ -106,11 +106,11 @@ Identity + routing:
106
106
 
107
107
  ```
108
108
  id, direction ('outbound'|'inbound'), channel ('sms'|'email'|
109
- 'voice'|'web_form'|'push'), sender_type, mdm_subject_id,
109
+ 'voice'|'web_form'|'push'), sender_type, legal_entity_id,
110
110
  thread_id, in_reply_to_id
111
111
  ```
112
112
 
113
- Content (tokenized in unregulated mode):
113
+ Content (masked in the tokenized and redacted lakes):
114
114
 
115
115
  ```
116
116
  subject, body, attachments, metadata, sender_id
@@ -134,7 +134,7 @@ Provenance (which workflow produced this):
134
134
 
135
135
  ```
136
136
  workflow_id, action_id, decision_key, context_key,
137
- idempotency_key — unique per (mdm_subject_id,
137
+ idempotency_key — unique per (legal_entity_id,
138
138
  idempotency_key); a repeat fire on the
139
139
  same key lands as status 'invalidated'
140
140
  (nothing sent) — NOT status 'duplicate',
@@ -149,58 +149,60 @@ goes through `action_logs` (which has `batch_id` AND `external_id`)
149
149
  ## 3. Reading messages
150
150
 
151
151
  ```typescript
152
- // (1) materialise a search — WHERE fragment against alias `rm`
152
+ // (1) materialise a search — WHERE fragment against alias `m`
153
153
  // dataset type is SINGULAR 'message' (the table is plural)
154
154
  const { data: search } = await api.datasets.createUserSearch(
155
155
  tenantSlug, datalakeSlug, 'message',
156
- { search_query: `rm.status = 'failed' AND rm.channel = 'sms'` },
156
+ { search_query: `m.status = 'failed' AND m.channel = 'sms'` },
157
157
  )
158
158
  // (2) page through it
159
159
  const { data: page } = await api.datasets.search(
160
160
  tenantSlug, datalakeSlug, 'message',
161
- { userSearchId: search.id, dataAccessMode: 'unregulated' },
161
+ { userSearchId: search.id, dataAccessMode: 'tokenized' },
162
162
  )
163
163
  ```
164
164
 
165
165
  Filterable columns (Flop, the full set): `id`, `inserted_at`,
166
166
  `direction`, `channel`, `status`, `sender_type`, `body`, `subject`,
167
- `mdm_subject_id`, `sender_id`, `sent_at`, `queued_at`, `delivered_at`,
167
+ `legal_entity_id`, `sender_id`, `sent_at`, `queued_at`, `delivered_at`,
168
168
  `thread_id`, `external_id`, `global_search`.
169
169
 
170
170
  **`decision_key` is NOT in the Flop-filterable set** — you cannot
171
171
  `?filter[decision_key]=…` (nor can the workflow-run-log endpoints; see
172
172
  §4). But you CAN still scope a search by it two other ways, both
173
- cookbook-proven: put `rm.decision_key = '…'` in the `createUserSearch`
173
+ cookbook-proven: put `m.decision_key = '…'` in the `createUserSearch`
174
174
  **`search_query`** SQL WHERE fragment (as the SMS-workflow cookbooks
175
- do), or `GROUP BY rm.decision_key` in `executeSql` for analytics (§4).
175
+ do), or `GROUP BY m.decision_key` in `executeSql` for analytics (§4).
176
176
  The Flop bracket-filter and the raw-SQL surfaces are different lanes —
177
177
  `decision_key` lives only on the SQL lane.
178
178
 
179
- `dataAccessMode: 'regulated'` returns raw content; `'unregulated'`
180
- returns tokenized objects for content columns (see
181
- `data_activation_clients.md` §6.5 for the shape).
179
+ `dataAccessMode` selects which lake answers: `'raw'` returns the real
180
+ content, `'tokenized'` returns a stable `tok_<hash>` per masked column,
181
+ and `'redacted'` returns an unjoinable placeholder. All three return
182
+ **plain values of the column's own type** — a masked column is a string,
183
+ not a wrapper object.
182
184
 
183
185
  ## 4. Campaign math (the dashboard recipe)
184
186
 
185
187
  "How did campaign X do?" — where X is one run's `batch_id`:
186
188
 
187
189
  ```
188
- action_logs (ral) — has batch_id + external_id
190
+ action_logs (al) — has batch_id + external_id
189
191
  │ join on external_id
190
192
  ▼
191
- messages (rm) — has status + engagement timestamps
193
+ messages (m) — has status + engagement timestamps
192
194
  ```
193
195
 
194
196
  ```sql
195
197
  SELECT
196
198
  count(*) AS sent,
197
- count(*) FILTER (WHERE rm.status = 'delivered') AS delivered,
198
- count(*) FILTER (WHERE rm.status = 'failed') AS undelivered,
199
- count(*) FILTER (WHERE rm.opened_at IS NOT NULL) AS opened,
200
- count(*) FILTER (WHERE rm.status = 'clicked') AS clicked
201
- FROM action_logs ral
202
- JOIN messages rm ON rm.external_id = ral.external_id
203
- WHERE ral.batch_id = '<batch_id>'
199
+ count(*) FILTER (WHERE m.status = 'delivered') AS delivered,
200
+ count(*) FILTER (WHERE m.status = 'failed') AS undelivered,
201
+ count(*) FILTER (WHERE m.opened_at IS NOT NULL) AS opened,
202
+ count(*) FILTER (WHERE m.status = 'clicked') AS clicked
203
+ FROM action_logs al
204
+ JOIN messages m ON m.external_id = al.external_id
205
+ WHERE al.batch_id = '<batch_id>'
204
206
  ```
205
207
 
206
208
  Run it with `api.datalakes.executeSql` (read-only tier). Delivered
@@ -211,8 +213,8 @@ merged NDJSON artifact (see `action_logs.md` §5).
211
213
 
212
214
  ### The message log IS your per-variant analytics table
213
215
 
214
- Every message row (both the regulated and unregulated tier) carries
215
- the full provenance of the send:
216
+ Every message row, in all three lakes, carries the full provenance of
217
+ the send:
216
218
 
217
219
  ```
218
220
  decision_key which action fired — the A/B variant key
@@ -220,7 +222,7 @@ channel sms | email | voice | web_form | push
220
222
  status delivery state
221
223
  delivered_at when the carrier confirmed
222
224
  opened_at engagement (from updateMessageTracking)
223
- workflow_id, action_id, sender_tool_id, mdm_subject_id
225
+ workflow_id, action_id, sender_tool_id, legal_entity_id
224
226
  ```
225
227
 
226
228
  So "how did variant A do vs variant B?" or "SMS vs email open rate?"
@@ -229,15 +231,15 @@ is a plain `GROUP BY` over the datalake `messages` table via
229
231
 
230
232
  ```sql
231
233
  SELECT
232
- rm.decision_key,
233
- rm.channel,
234
+ m.decision_key,
235
+ m.channel,
234
236
  count(*) AS sent,
235
- count(*) FILTER (WHERE rm.status = 'delivered') AS delivered,
236
- count(*) FILTER (WHERE rm.status = 'failed') AS failed,
237
- count(*) FILTER (WHERE rm.opened_at IS NOT NULL) AS opened
238
- FROM messages rm
239
- GROUP BY rm.decision_key, rm.channel
240
- ORDER BY rm.decision_key, rm.channel
237
+ count(*) FILTER (WHERE m.status = 'delivered') AS delivered,
238
+ count(*) FILTER (WHERE m.status = 'failed') AS failed,
239
+ count(*) FILTER (WHERE m.opened_at IS NOT NULL) AS opened
240
+ FROM messages m
241
+ GROUP BY m.decision_key, m.channel
242
+ ORDER BY m.decision_key, m.channel
241
243
  ```
242
244
 
243
245
  Scope it to one campaign run by joining `action_logs` (which carries
@@ -282,7 +284,7 @@ per-channel reporting.
282
284
  `opened_at` from an earlier attempt in the thread — read both.
283
285
 
284
286
  4. **`status: 'invalidated'` is the idempotency guard firing** —
285
- the same `(mdm_subject_id, idempotency_key)` already sent. Not
287
+ the same `(legal_entity_id, idempotency_key)` already sent. Not
286
288
  an error; usually exactly what you want.
287
289
 
288
290
  5. **Delivery events that don't match are dropped silently.** The
@@ -71,7 +71,7 @@ after one:
71
71
 
72
72
  ```bash
73
73
  AWS_ACCESS_KEY_ID=test AWS_SECRET_ACCESS_KEY=test AWS_DEFAULT_REGION=us-east-1 \
74
- aws s3 mb s3://healthcare-lake-regulated --endpoint-url http://localhost:4566
74
+ aws s3 mb s3://healthcare-lake --endpoint-url http://localhost:4566
75
75
  ```
76
76
 
77
77
  ## The hosted mock — `wiremock.alvera.ai`
@@ -37,9 +37,9 @@ const { data: current } = await api.tools.get(
37
37
  // 2. change in memory
38
38
  const next = { ...current, status: 'inactive' }
39
39
 
40
- // 3. send the whole body back
40
+ // 3. send the whole body back. The cast is required — see below.
41
41
  const { data: updated } = await api.tools.update(
42
- tenantSlug, datalakeSlug, toolId, next,
42
+ tenantSlug, datalakeSlug, toolId, next as unknown as ToolRequestWritable,
43
43
  )
44
44
  ```
45
45