@alvera-ai/platform-sdk 0.17.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/.agent/AGENTS.md +82 -144
  2. package/.agent/account_management.md +2 -2
  3. package/.agent/action_logs.md +4 -4
  4. package/.agent/ai_agents.md +28 -21
  5. package/.agent/ai_sandbox.md +49 -39
  6. package/.agent/connected_apps.md +3 -3
  7. package/.agent/cookbook/_fixtures/README.md +1 -1
  8. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_generic_table.liquid +1 -1
  9. package/.agent/cookbook/_fixtures/organic-marketing/_lead_submissions_foundation_legal_entity.liquid +80 -0
  10. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_mdm.liquid +2 -1
  11. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_generic_table.liquid +57 -0
  12. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_legal_entity.liquid +30 -0
  13. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_mdm.liquid +44 -0
  14. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_generic_table.liquid +57 -0
  15. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_legal_entity.liquid +36 -0
  16. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_mdm.liquid +41 -0
  17. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_generic_table.liquid +70 -0
  18. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_legal_entity.liquid +52 -0
  19. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_mdm.liquid +42 -0
  20. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_generic_table.liquid +38 -0
  21. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_legal_entity.liquid +48 -0
  22. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_mdm.liquid +49 -0
  23. package/.agent/cookbook/organic-marketing.md +2801 -0
  24. package/.agent/cookbook/payments-compliance.md +2180 -0
  25. package/.agent/cookbook/primary-care.md +2175 -0
  26. package/.agent/cookbook/subscription-saas.md +2403 -0
  27. package/.agent/data_activation_clients.md +65 -52
  28. package/.agent/datalakes.md +338 -171
  29. package/.agent/errors.md +3 -3
  30. package/.agent/generic_tables.md +151 -62
  31. package/.agent/interoperability_contracts.md +57 -22
  32. package/.agent/mdm.md +136 -153
  33. package/.agent/messages.md +36 -34
  34. package/.agent/mock-services.md +1 -1
  35. package/.agent/mutations.md +2 -2
  36. package/.agent/templates.md +14 -13
  37. package/.agent/tools.md +63 -21
  38. package/.agent/type_naming.md +13 -13
  39. package/.agent/workflows.md +99 -53
  40. package/README.md +2 -2
  41. package/dist/bin/platform-sdk.mjs +33 -47
  42. package/dist/bin/platform-sdk.mjs.map +1 -1
  43. package/dist/index.d.mts +565 -379
  44. package/dist/index.d.mts.map +1 -1
  45. package/dist/index.mjs +494 -59
  46. package/dist/index.mjs.map +1 -1
  47. package/package.json +4 -3
  48. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +0 -88
  49. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +0 -47
  50. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +0 -24
  51. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +0 -38
  52. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_compliance_screening.liquid +0 -59
  53. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_mdm.liquid +0 -36
  54. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_mdm.liquid +0 -30
  55. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_payment_account.liquid +0 -55
  56. package/.agent/cookbook/_fixtures/subscription/_customers_subscription_mdm.liquid +0 -20
  57. package/.agent/cookbook/_setup/foundation.md +0 -359
  58. package/.agent/cookbook/_setup/healthcare.md +0 -361
  59. package/.agent/cookbook/_setup/payments.md +0 -365
  60. package/.agent/cookbook/_setup/subscription.md +0 -364
  61. package/.agent/cookbook/action-status-updaters.md +0 -278
  62. package/.agent/cookbook/ai-agent-invoke.md +0 -279
  63. package/.agent/cookbook/appointment-review-sms-workflow.md +0 -801
  64. package/.agent/cookbook/birthday-greeting-sms-trigger.md +0 -696
  65. package/.agent/cookbook/bulk-ingest.md +0 -302
  66. package/.agent/cookbook/contact-us-triage-with-llm.md +0 -663
  67. package/.agent/cookbook/dunning-sms-for-delinquent.md +0 -659
  68. package/.agent/cookbook/generic-tables.md +0 -244
  69. package/.agent/cookbook/invite-team.md +0 -200
  70. package/.agent/cookbook/kyc-notification-on-account-activation.md +0 -661
  71. package/.agent/cookbook/marketing-campaign-send.md +0 -1044
  72. package/.agent/cookbook/paginated-restapi-poller.md +0 -383
  73. package/.agent/cookbook/rest-fetch.md +0 -273
  74. package/.agent/cookbook/sanctions-screening-with-agent-review.md +0 -773
  75. package/.agent/cookbook/score-leads-with-llm-categorization.md +0 -665
  76. package/.agent/cookbook/system-templates.md +0 -165
  77. package/.agent/cookbook/talk-to-data.md +0 -178
  78. package/.agent/cookbook/triage-prospects-by-priority.md +0 -571
  79. package/.agent/cookbook/welcome-sms-for-customers.md +0 -647
  80. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/memorandum-of-association-01.png +0 -0
  81. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/sample_two_page.pdf +0 -0
  82. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/_customers_subscription_customer.liquid +0 -0
  83. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/stripe_customers_batch1.csv +0 -0
package/.agent/errors.md CHANGED
@@ -202,10 +202,10 @@ body** — the failure won't clear on its own.
202
202
 
203
203
  ## Mapping `source.pointer` back to your TypeScript type
204
204
 
205
- A `source.pointer` like `/regulated_data_db_reader_user` maps directly to the
206
- `regulated_data_db_reader_user` field on `DatalakeRequestWritable`. For
205
+ A `source.pointer` like `/db_reader_user` maps directly to the
206
+ `db_reader_user` field on `DatalakeRequestWritable`. For
207
207
  nested embeds, the pointer carries the embed path:
208
- `/unregulated_cloud_storage/access_key_id`.
208
+ `/cloud_storage/access_key_id`.
209
209
 
210
210
  For polymorphic embeds, check the discriminator field's pointer first — if
211
211
  `/tool_body_type` (or `/cloud_storage_type`, etc.) is in the errors array,
@@ -38,7 +38,7 @@ in the datalake's schema. The parent datalake MUST be
38
38
 
39
39
  ```typescript
40
40
  import type {
41
- GenericTableColumn,
41
+ GenericTableColumnRequest,
42
42
  GenericTableResponse,
43
43
  } from '@alvera-ai/platform-sdk'
44
44
 
@@ -121,7 +121,7 @@ type required enum — see "Column types" below
121
121
  is_array optional bool — column holds an array of `type` values
122
122
  description required string — narrative; surfaces in catalogs
123
123
  is_unique required bool — Postgres UNIQUE constraint at migration time
124
- privacy_requirement required enum — 'none' | 'tokenize' | 'redact_only'
124
+ privacy_requirement required enum — 'none' | 'tokenize' | 'redact'
125
125
  ```
126
126
 
127
127
  #### Column types
@@ -129,7 +129,8 @@ privacy_requirement required enum — 'none' | 'tokenize' | 'redact_only'
129
129
  ```
130
130
  string free-form text
131
131
  integer whole numbers
132
- float decimal numbers
132
+ float approximate numeric (IEEE 754)
133
+ decimal exact numeric, stored as `numeric`
133
134
  boolean true / false
134
135
  date calendar date (no time component)
135
136
  datetime timestamp with timezone
@@ -137,54 +138,118 @@ privacy_requirement required enum — 'none' | 'tokenize' | 'redact_only'
137
138
  jsonb arbitrary JSON document
138
139
  ```
139
140
 
141
+ `decimal` and `float` are distinct on purpose: `decimal` emits a
142
+ Postgres `numeric` and comes back as an exact Decimal, while `float`
143
+ is binary floating point and cannot load one. Money and anything
144
+ else that must not drift belongs in `decimal`.
145
+
140
146
  Each type can be paired with `is_array: true` to model an array
141
- column (`string[]`, `integer[]`, etc.). The wire enum lives in
142
- the generated SDK as `GenericTableColumnTypeEnum` — consult it for
143
- the live set since new types land independently of corpus
144
- revisions.
147
+ column (`string[]`, `integer[]`, etc.) — with one exception that has
148
+ its own section below, because it is the single most expensive
149
+ mistake available on this page. See **`jsonb` + `is_array`**.
150
+
151
+ The wire enum lives in the generated SDK as the `type` field of
152
+ `GenericTableColumnRequest` — consult it for the live set, since new
153
+ types land independently of corpus revisions.
145
154
 
146
155
  #### Privacy × type interactions
147
156
 
148
- `privacy_requirement` selects how the column flows between the
149
- regulated and unregulated storage tiers (see `datalakes.md` §1
150
- for the tier split):
151
-
152
- - `none` — keeps the raw value in the regulated tier only;
153
- the unregulated tier has no column.
154
- - `tokenize` — stores the raw value in the regulated tier and an
155
- opaque stable token in the unregulated tier. AI tools working
156
- off the unregulated tier see the token, not the underlying
157
- identity.
158
- - `redact_only` — keeps the raw value in the regulated tier and
159
- strips it from the unregulated tier entirely.
160
-
161
- Two type-level restrictions the typed validator does not encode:
162
-
163
- - **`type: 'jsonb'` REQUIRES `privacy_requirement: 'redact_only'` —
164
- `'none'` is rejected too, not just `'tokenize'`.** The platform
165
- can't tokenize an arbitrary JSON document (no canonical
166
- "identity" to tokenize), and it won't let an unredacted JSON
167
- blob flow into the unregulated tier either. `'redact_only'` is
168
- the ONLY valid privacy value for a jsonb column — submitting
169
- `{ type: 'jsonb', privacy_requirement: 'tokenize' }` OR
170
- `{ type: 'jsonb', privacy_requirement: 'none' }` both return
171
- 422 on `/columns/N/privacy_requirement`.
157
+ `privacy_requirement` decides what each of the three lakes hands back
158
+ for this column. **All three lakes carry every column** — the copies
159
+ are full copies, not projections — so the question is never whether
160
+ the column is there but what value it holds. (See `datalakes.md` §1
161
+ for the lake split, and note that the tokenized and redacted copies
162
+ exist only after an explicit `provisionTokenization` call.)
163
+
164
+ Measured against a live three-lake tenant, one row read three ways:
165
+
166
+ | `privacy_requirement` | raw | tokenized | redacted |
167
+ |---|---|---|---|
168
+ | `none` | `Family Medicine` | `Family Medicine` | `Family Medicine` |
169
+ | `tokenize` | `Rosalind Achterberg` | `tok_7a1de41f5aba…` | `xxxx-berg` |
170
+ | `redact` | the real value | masked | masked |
171
+
172
+ Three things that table is trying to make unmissable:
173
+
174
+ - **`none` is byte-identical everywhere.** It is not "raw only". A
175
+ masking policy that altered a `none` column would silently break
176
+ every join in the system, so these are the columns you filter and
177
+ join on in every lake.
178
+ - **`tokenize` and `redact` differ in the TOKENIZED lake, not the
179
+ redacted one.** A tokenized column yields a stable `tok_<hash>` —
180
+ the same input gives the same token, so a machine can join across
181
+ tables on it without ever resolving it. A redacted column yields an
182
+ unjoinable placeholder. That is the whole distinction: **`tokenize`
183
+ preserves referential identity without disclosing it; `redact`
184
+ destroys both.**
185
+ - **The redacted lake masks both.** `xxxx-<tail>` is for a person
186
+ glancing at a support screen, and it is deliberately useless as a
187
+ key.
188
+
189
+ There is **no restriction tying `jsonb` to `redact`.** A jsonb column
190
+ may be `none`, `tokenize` or `redact` like any other. An older rule
191
+ did refuse anything but `redact`, because tokenizing a JSON document
192
+ once produced a value the column could not hold; that rule was
193
+ **removed rather than relaxed**, because masking a JSON column is now
194
+ well-defined (see the next section). If you are reading an older copy
195
+ of this corpus that says otherwise, it is describing a 422 that no
196
+ longer happens.
197
+
198
+ One type-level quirk the typed validator does not encode:
199
+
172
200
  - **Tokenize widens primitive types at the schema layer.** A
173
201
  `{ type: 'integer', privacy_requirement: 'tokenize' }` column
174
- surfaces in the unregulated tier as a string-typed token, not
202
+ surfaces in the tokenized tier as a string-typed token, not
175
203
  an integer. This is a quirk of how the platform stores tokens
176
204
  (one opaque string format across all source types) — your
177
- unregulated-tier queries must treat the column as a string.
205
+ queries against the tokenized lake must treat the column as a string.
178
206
  This affects `integer`, `float`, `boolean`, `is_array string`,
179
207
  `date`, `datetime`, and `time`.
180
208
 
209
+ #### `jsonb` + `is_array` — the rule that is not what it looks like
210
+
211
+ Declaring `{ type: 'jsonb', is_array: true }` is how you model a
212
+ column holding a **JSON array** — a list of objects, say. It is
213
+ required alongside `type: 'jsonb'` when the template renders a list:
214
+ without it the column is a single JSON document, and each row
215
+ carrying an array is refused with
216
+ `upsert_row_failed %{<column>: ["is invalid"]}` **while the batch
217
+ reports itself green**.
218
+
219
+ **`is_array: true` describes the JSON document. It is NOT an
220
+ instruction to make the Postgres column an array type.** That
221
+ misreading is the most expensive one available here, because of what
222
+ it implies for a masked column.
223
+
224
+ A masked `jsonb` column is `jsonb` in the raw lake and **`text` in
225
+ both copies**, and those types are *supposed* to differ. Logical
226
+ replication ships the publisher's text rendering and feeds it to the
227
+ subscriber column's input function:
228
+
229
+ ```
230
+ jsonb -> text always succeeds (any jsonb renders to a string,
231
+ and any string is valid text)
232
+ jsonb -> text[] always fails ([{"code":"A1C"}] is JSON, and
233
+ NOT a Postgres array literal)
234
+ ```
235
+
236
+ So the invariant is **scalar-ness, not type parity**: a jsonb column
237
+ is never a Postgres array type, in any lake. Get this wrong and the
238
+ apply worker stops on the first row carrying an array — and the
239
+ failure has no error surface a client can see. You get derived lakes
240
+ that exist, report `status: 'ready'`, and never fill.
241
+
242
+ Because the copies hold text, a masked jsonb column comes back as a
243
+ masked *string* rather than as an array, and it will not parse as
244
+ JSON. That is expected: on that side the column really is text.
245
+
181
246
  The privacy choice is **immutable post-create** — `update()` exists
182
247
  but its migration is additive-only and never re-privacy's an existing
183
248
  column (see §2 update note), so the only way to change a column's
184
249
  `privacy_requirement` is `delete()` + `create()`.
185
250
 
186
251
  **Privacy is operator-decided per column, not implied by name.**
187
- A field called `message` can be `redact_only` in one tenant and
252
+ A field called `message` can be `redact` in one tenant and
188
253
  `none` in another; the SDK doesn't pattern-match column names
189
254
  to default privacy levels. Pick deliberately at design time —
190
255
  recreating the table is the only way to flip a column's
@@ -203,24 +268,35 @@ data activation clients use as the upsert key.
203
268
  The platform reserves certain column names you cannot redeclare
204
269
  in `columns[N].name`:
205
270
 
206
- - **Infrastructure columns** (always reserved, every industry):
207
- `id`, `inserted_at`, `updated_at`, `datalake_id`, plus the
208
- unregulated/regulated tier-link fields the platform manages.
209
- - **Industry-specific MDM FK columns** (reserved only inside
210
- their owning industry's datalake):
211
-
212
- ```
213
- healthcare patient_id
214
- core_banking party_id
215
- subscription customer_id
216
- (other domains as they land)
217
- ```
218
-
219
- A `patient_id` column is REJECTED in a healthcare datalake's
220
- generic table — the platform's MDM machinery owns that name —
221
- but is ACCEPTED in a `core_banking` datalake's generic table
222
- (it's just an opaque user-defined column there). Reserved-name
223
- rejections surface as 422 on `/columns/N/name`.
271
+ The platform creates every generic table with the same base columns,
272
+ and they are yours to read but never to declare:
273
+
274
+ ```
275
+ id uuid primary key
276
+ inserted_at timestamp
277
+ updated_at timestamp
278
+ batch_id the ingest run that wrote the row
279
+ client_id the activation client that wrote it
280
+ legal_entity_id the subject the contract's MDM resolve stamped
281
+ ```
282
+
283
+ **`legal_entity_id` is the one that catches people.** It is not a
284
+ column you declare and fill — a contract's `mdm_input_config`
285
+ resolves the subject and the platform stamps it. Declaring a column
286
+ of that name does not shadow the stamp silently: the create is
287
+ REJECTED with 422, and the message names the offending column
288
+ (`contain reserved system column names: legal_entity_id`). You lose
289
+ the create, not the linkage — so read the 422 as the platform telling
290
+ you the column is already there, and drop it from your definition.
291
+ There is correspondingly **nothing to guard against in a contract
292
+ hook**: a table that declares the column is never created, so an
293
+ `assert` on its absence can never fire.
294
+
295
+ There is no longer a per-domain reserved list. GH-859 removed the
296
+ industry datalakes and their MDM foreign keys, so names like
297
+ `patient_id`, `party_id` and `customer_id` are now ordinary
298
+ user-defined columns you may declare freely. Reserved-name rejections
299
+ surface as 422 on `/columns/N/name`.
224
300
 
225
301
  ### Title cannot start with `alvera_system`
226
302
 
@@ -240,7 +316,7 @@ reconcile drift. On update the platform re-runs the schema migration
240
316
  for the new column set — but the migration is **strictly additive**:
241
317
 
242
318
  - **New columns in the body are added** to the physical table
243
- (test-pinned) and to the regulated mirror.
319
+ (test-pinned) and to each lake's copy of it.
244
320
  - **Columns you omit are NOT dropped.** The migration never
245
321
  auto-drops a column ("keep down as noop to avoid destructive
246
322
  change"), so a shrunken `columns` array does not remove storage.
@@ -285,11 +361,25 @@ Standard JSON:API envelopes per `errors.md`. Common rejections:
285
361
  | `source.pointer` | Cause |
286
362
  |----------------------------------|----------------------------------------------------|
287
363
  | `/title` | empty; generates a name colliding with existing; or starts with `alvera_system` prefix |
288
- | `/columns` | empty array; duplicate `column.name`; or zero columns marked `is_unique: true` |
289
- | `/columns/0/name` | reserved column name (infrastructure column or the industry's MDM FK; see §2) |
364
+ | `/columns` | empty array; duplicate `column.name`; zero columns marked `is_unique: true`; or a RESERVED column name (see §2) |
365
+ | `/columns/0/name` | too short/long, or not matching `[a-z][a-z0-9_]*` |
290
366
  | `/columns/0/type` | value not in supported type enum (see §2 column types) |
291
367
  | `/columns/0/privacy_requirement` | value not in enum, OR `tokenize` on a `jsonb` column |
292
368
 
369
+ **Reserved names point at `/columns`, not at the offending index** —
370
+ measured, not inferred. The check runs on the parent changeset
371
+ (`add_error(changeset, :columns, …)`), so it names *every* offending
372
+ column at once rather than the first it meets:
373
+
374
+ ```
375
+ alvera: plan halted at generic_table "lead-submissions": server rejected request:
376
+ /columns: contain reserved system column names: legal_entity_id
377
+ ```
378
+
379
+ Note where that halted: at **plan**, not apply — the server answering
380
+ during plan's refresh, before any mutation. The per-column rows above
381
+ are indexed because those validations run on the column embed itself.
382
+
293
383
  All of these come back as a 422 `AlveraApiError` thrown from the
294
384
  SDK on submit — structural cases (Layer 1) and semantic conflicts
295
385
  like name collision / deployed-name reuse (Layer 2). The SDK does
@@ -318,7 +408,7 @@ const { data: created } = await api.genericTables.create(
318
408
  const deadline = Date.now() + 60_000
319
409
  while (Date.now() < deadline) {
320
410
  const { data: row } = await api.genericTables.get(
321
- tenantSlug, datalakeSlug, created.id, // UUID id, NOT created.name
411
+ tenantSlug, datalakeSlug, created.id!, // UUID id, NOT created.name
322
412
  )
323
413
  if (row.status === 'deployed') break
324
414
  await new Promise((r) => setTimeout(r, 1_000))
@@ -331,7 +421,7 @@ definition drifted; never a fresh-create state). A migration that
331
421
  fails does not flip to a `failed` status; the row simply never reaches
332
422
  `deployed`, so bound the poll with a **timeout** (as above) rather than
333
423
  watching for a failure value. (This is the same timeout-bounded idiom
334
- the validated `cookbook/generic-tables.md` runs, passing the UUID
424
+ every validated cookbook runs, passing the UUID
335
425
  `id`.) `.get(id)` distinguishes "row never
336
426
  created" (404) from "deployment in progress" (200 with a
337
427
  sub-`deployed` status) — two distinct signals on the same endpoint.
@@ -409,11 +499,10 @@ dependency. Detach dependents first.
409
499
 
410
500
  3. **Column `privacy_requirement` is immutable post-create.**
411
501
  `tokenize` columns surface as an opaque stable token in the
412
- unregulated tier; `redact_only` strips the value from that
413
- tier entirely; `none` keeps the raw value only in the
414
- regulated tier. `update()`'s additive migration never re-privacy's
415
- an existing column, so recreate the table to change a column's
416
- privacy.
502
+ tokenized lake as a stable token; `redact` masks it in both
503
+ copies; `none` passes the real value through to all three
504
+ unchanged. `update()`'s additive migration never re-privacy's an
505
+ existing column, so recreate the table to change one.
417
506
 
418
507
  4. **System-dataset enumeration EXCLUDES generic tables.**
419
508
  `api.datalakes.systemDatasets(...)` returns only the
@@ -32,12 +32,16 @@ const { data: created } = await api.interoperabilityContracts.create(
32
32
  tenantSlug,
33
33
  datalakeSlug,
34
34
  {
35
- name: 'Inbound CRM → Customer',
36
- description: 'Vendor CRM rows → Subscription Customer',
37
- resource_type: 'customer',
35
+ name: 'Inbound CRM → Legal Entity',
36
+ description: 'Vendor CRM rows → the legal entity they identify',
37
+ resource_type: 'legal_entity',
38
38
  filter_template: undefined, // optional
39
39
  template_config: { type: 'custom', body: liquidBody },
40
- mdm_input_config: { type: 'custom', body: mdmBody },
40
+ // The template IS the subject here, so no entity resolution
41
+ // runs. A contract that writes a generic table instead pairs
42
+ // with this one and carries a real mdm_input_config emitting
43
+ // the SAME (uri, external_id) — see §2 "The contract pair".
44
+ mdm_input_config: { type: 'null' },
41
45
  generic_table_id: null, // see §2
42
46
  },
43
47
  )
@@ -95,16 +99,46 @@ every other `resource_type`, the field must be `null`. The
95
99
  TypeScript type permits either value; the validator and server
96
100
  enforce the conditional rule.
97
101
 
98
- ### `resource_type` is industry-scoped
102
+ ### `resource_type` is a closed platform set
103
+
104
+ GH-859 removed the per-industry vocabularies. Every datalake now
105
+ accepts the same set:
106
+
107
+ ```
108
+ action_log beneficial_owner document
109
+ generic_table legal_entity message
110
+ ```
111
+
112
+ The subject-specific types an industry used to bring — `patient`,
113
+ `customer`, `payment_account`, `compliance_screening` — are gone.
114
+ Model those as a **generic table** and pair it with a
115
+ `legal_entity` contract (see "The contract pair" below).
116
+
117
+ On the wire `resource_type` is a plain `string`; the set is
118
+ enforced server-side, so an invalid value is a `/resource_type`
119
+ 422 at create rather than a compile error. `datalakes.md` §5
120
+ `.systemDatasets` remains the runtime authority.
121
+
122
+ ### The contract pair
123
+
124
+ One inbound row usually becomes two things, so it needs two
125
+ contracts:
126
+
127
+ | | `resource_type` | `mdm_input_config` | writes |
128
+ |---|---|---|---|
129
+ | **A** | `legal_entity` | `{ type: 'null' }` | the subject |
130
+ | **B** | `generic_table` (+ `generic_table_id`) | a real template | the row |
131
+
132
+ Both emit the **same** `(uri, external_id)` identifier pair, so
133
+ they converge on one legal entity and the platform stamps
134
+ `legal_entity_id` onto the generic-table row.
99
135
 
100
- The valid set of `resource_type` values is determined by the
101
- parent datalake's `data_domain`. Generic concepts like
102
- `'generic_table'` are accepted everywhere; domain-anchored
103
- concepts (`'patient'`, `'customer'`, `'payment_account'`,
104
- `'legal_entity'`, ...) only resolve within their owning industry.
105
- Don't hardcode a fixed enum across industries — discover the
106
- valid set at runtime via the platform's domain catalog (see
107
- `datalakes.md` §5 `.systemDatasets`).
136
+ **Never declare `legal_entity_id` as a column on that table.**
137
+ The name is reserved, so declaring it does not shadow the stamp —
138
+ the create is REJECTED with 422 (`contain reserved system column
139
+ names: legal_entity_id`). You lose the table, not the linkage.
140
+ Drop it from your definition; the platform adds the column itself
141
+ and you read the stamp back off the row.
108
142
 
109
143
  ### `filter_template` has INVERTED semantics
110
144
 
@@ -158,7 +192,7 @@ type string — mirrors template_config.type at create time
158
192
  ```
159
193
  name required string — agent-facing label
160
194
  description optional string — narrative
161
- resource_type required string — target shape (industry-scoped)
195
+ resource_type required string — target shape (closed set, §2)
162
196
  filter_template optional string — Liquid pre-filter (inverted semantics)
163
197
  template_config required embed — { type, body?, path? }
164
198
  mdm_input_config optional embed — { type, body?, path? }; omit equals { type: 'null' }
@@ -407,8 +441,8 @@ reject with 422 on `/base` and a literal message:
407
441
  System-created contracts include the auto-generated identity
408
442
  contracts the platform provisions when a generic table is
409
443
  created (see §5 "Generic-table contracts: auto-created identity
410
- contracts") and any industry-built templates the platform ships
411
- with the datalake's data_domain. User-created contracts of any
444
+ contracts") and the templates the platform ships with every
445
+ datalake. User-created contracts of any
412
446
  `type` (including caller-created `type: 'identity'` rows bound
413
447
  to a generic table) ARE editable and deletable — the gate is
414
448
  `system_created`, not `type`.
@@ -452,7 +486,7 @@ const { data: contract } = await api.interoperabilityContracts.create(
452
486
  body: JSON.stringify({ name: '{{ msg.company_name }}' }),
453
487
  // A contract join AUTHORS output_schema — unlike a workflow
454
488
  // join, the server does NOT pin it here.
455
- output_schema: '{"type":"object","properties":{"lead":{"type":"string"}}}',
489
+ output_schema: { type: 'object', properties: { lead: { type: 'string' } } },
456
490
  },
457
491
  },
458
492
  ],
@@ -503,11 +537,12 @@ surface, and `workflows.md` for the workflow side of the same join.
503
537
  runs". Declaring `{ type: 'null' }` explicitly is allowed
504
538
  for readability but not required.
505
539
 
506
- 3. **`resource_type` is industry-scoped.** A datalake with
507
- `data_domain: 'foundation'` accepts `'legal_entity'` but
508
- rejects `'patient'`. The error surfaces as
509
- `/resource_type` 422 at create time. Don't ship a fixed
510
- enum across industries.
540
+ 3. **`resource_type` is a closed platform set.** Every datalake
541
+ accepts the same six values; the retired per-industry types
542
+ (`'patient'`, `'customer'`, `'payment_account'`, ...) now
543
+ reject with a `/resource_type` 422 at create time. Model
544
+ those as a generic table paired with a `legal_entity`
545
+ contract instead.
511
546
 
512
547
  4. **Sandbox-run is not optional in practice.** A template that
513
548
  compiles but renders structurally-invalid JSON, or that