@alvera-ai/platform-sdk 0.10.0-rc.2 → 0.10.0-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/.agent/AGENTS.md +440 -0
  2. package/.agent/account_management.md +455 -0
  3. package/.agent/action_status_updaters.md +262 -0
  4. package/.agent/ai_agents.md +423 -0
  5. package/.agent/ai_sandbox.md +265 -0
  6. package/.agent/async.md +111 -0
  7. package/.agent/connected_apps.md +407 -0
  8. package/.agent/cookbook/_fixtures/README.md +99 -0
  9. package/.agent/cookbook/_fixtures/accounts_receivable/_customers_accounts_receivable_customer.liquid +32 -0
  10. package/.agent/cookbook/_fixtures/accounts_receivable/_customers_accounts_receivable_mdm.liquid +20 -0
  11. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_generic_table.liquid +33 -0
  12. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +88 -0
  13. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_mdm.liquid +48 -0
  14. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +47 -0
  15. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +24 -0
  16. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +38 -0
  17. package/.agent/cookbook/_fixtures/payment_risk/_compliance_screenings_payment_risk_compliance_screening.liquid +59 -0
  18. package/.agent/cookbook/_fixtures/payment_risk/_compliance_screenings_payment_risk_mdm.liquid +36 -0
  19. package/.agent/cookbook/_fixtures/payment_risk/_payment_accounts_payment_risk_mdm.liquid +30 -0
  20. package/.agent/cookbook/_fixtures/payment_risk/_payment_accounts_payment_risk_payment_account.liquid +55 -0
  21. package/.agent/cookbook/_setup/accounts_receivable.md +282 -0
  22. package/.agent/cookbook/_setup/foundation.md +277 -0
  23. package/.agent/cookbook/_setup/healthcare.md +279 -0
  24. package/.agent/cookbook/_setup/payment_risk.md +283 -0
  25. package/.agent/cookbook/appointment-review-sms-workflow.md +761 -0
  26. package/.agent/cookbook/birthday-greeting-sms-trigger.md +656 -0
  27. package/.agent/cookbook/contact-us-triage-with-llm.md +603 -0
  28. package/.agent/cookbook/dunning-sms-for-delinquent.md +619 -0
  29. package/.agent/cookbook/kyc-notification-on-account-activation.md +619 -0
  30. package/.agent/cookbook/sanctions-screening-with-agent-review.md +711 -0
  31. package/.agent/cookbook/score-leads-with-llm-categorization.md +602 -0
  32. package/.agent/cookbook/welcome-sms-for-customers.md +607 -0
  33. package/.agent/data_activation_clients.md +557 -0
  34. package/.agent/data_sources.md +234 -0
  35. package/.agent/datalakes.md +712 -0
  36. package/.agent/debugging.md +137 -0
  37. package/.agent/errors.md +196 -0
  38. package/.agent/generic_tables.md +351 -0
  39. package/.agent/interoperability_contracts.md +351 -0
  40. package/.agent/mdm.md +293 -0
  41. package/.agent/mutations.md +152 -0
  42. package/.agent/templates.md +98 -0
  43. package/.agent/tool-call-configs.md +90 -0
  44. package/.agent/tools.md +546 -0
  45. package/.agent/type_naming.md +131 -0
  46. package/.agent/workflows.md +601 -0
  47. package/README.md +46 -0
  48. package/dist/bin/platform-sdk.d.mts +1 -0
  49. package/dist/bin/platform-sdk.mjs +106 -0
  50. package/dist/bin/platform-sdk.mjs.map +1 -0
  51. package/dist/index.d.mts +1200 -43201
  52. package/dist/index.d.mts.map +1 -1
  53. package/dist/index.mjs +1859 -7319
  54. package/dist/index.mjs.map +1 -1
  55. package/package.json +19 -10
@@ -0,0 +1,712 @@
1
+ # Datalakes
2
+
3
+ A **datalake** is the storage layer for one tenant slice. It holds
4
+ two databases under one roof:
5
+
6
+ - an **unregulated** database for tokenized data, exposed to AI
7
+ workflows and downstream agents
8
+ - a **regulated** database for raw values that never leave the
9
+ trust boundary
10
+
11
+ Every other resource in the platform (data sources, tools, AI
12
+ agents, workflows, etc.) scopes under a datalake. A tenant can own
13
+ many datalakes; the canonical pattern is one per data domain (e.g.
14
+ `healthcare`, `foundation`, `payment_risk`).
15
+
16
+ ```typescript
17
+ import type { PlatformApi } from '@alvera-ai/platform-sdk'
18
+
19
+ const api: PlatformApi = /* see AGENTS.md */
20
+ const { data: created } = await api.datalakes.create(tenantSlug, body)
21
+ const datalakeSlug = created.slug // canonical — use for downstream scoping
22
+ ```
23
+
24
+ SDK namespace: `api.datalakes`.
25
+
26
+ ## 1. Wire shape
27
+
28
+ The request type splits across two layers: top-level database +
29
+ identity fields, plus two polymorphic cloud-storage embeds.
30
+
31
+ ```typescript
32
+ import type {
33
+ DatalakeRequestWritable,
34
+ DatalakeResponse,
35
+ } from '@alvera-ai/platform-sdk'
36
+ ```
37
+
38
+ See `type_naming.md` for the `Writable` convention and the
39
+ universal server-derived field set. (The SDK exports types only —
40
+ there is no runtime validator; validation is server-authoritative.)
41
+
42
+ ### Polymorphic cloud-storage embed
43
+
44
+ `unregulated_cloud_storage` and `regulated_cloud_storage` are
45
+ polymorphic. The discriminator field is **`cloud_storage_type`**
46
+ (see `mutations.md` "Polymorphic discriminator naming") with three
47
+ branches:
48
+
49
+ | `cloud_storage_type` | TypeScript branch |
50
+ |----------------------|-----------------------------------------|
51
+ | `"aws"` | `DatalakeCloudStorageAwsRequest` |
52
+ | `"r2"` | `DatalakeCloudStorageR2Request` |
53
+ | `"custom"` | `DatalakeCloudStorageCustomRequest` |
54
+
55
+ Each branch carries its own field set; the shared keys are
56
+ `bucket`, `region`, `access_key_id`, `secret_access_key`. `custom`
57
+ adds `endpoint` (used for LocalStack and self-hosted S3-compatible
58
+ stores).
59
+
60
+ The `aws` branch also carries an `auth_method` field —
61
+ `"access_key"` (default) or `"iam_role"` — that pairs with the
62
+ credentials the same way the DB-profile `auth_method` does. See
63
+ §2 "AWS cloud-storage `auth_method = iam_role` forbids
64
+ credentials" for the conditional the type cannot encode.
65
+
66
+ The two embeds are independent — a datalake can pair an `aws`
67
+ unregulated bucket with a `custom` regulated bucket. Both embeds
68
+ ship in the same POST/PUT body.
69
+
70
+ ### Cloud storage is also the cold-archive destination
71
+
72
+ Beyond being the platform's primary blob store for inline payloads
73
+ (image uploads, large context payloads, presigned-URL ingest), the
74
+ configured `regulated_cloud_storage` and `unregulated_cloud_storage`
75
+ buckets are the **cold-archive destination** for the datalake. As
76
+ data flows through the pipeline, two NDJSON streams land under
77
+ well-defined key prefixes inside the bucket the credentials point at:
78
+
79
+ ```
80
+ {datalake_slug}/data-activations/{client_slug}/{batch_id}/{dataset_table}.ndjson
81
+ {datalake_slug}/workflows/{wf_slug}/{batch_id}/merged.ndjson
82
+ {datalake_slug}/workflows/{wf_slug}/{batch_id}/{decision_key}/merged.ndjson
83
+ ```
84
+
85
+ The merged NDJSONs are produced by background batch-merge workers
86
+ after each Data Activation Client batch and each workflow run
87
+ completes. The platform never streams the merged bytes through its
88
+ own process — the merge happens via DuckDB's `httpfs` extension
89
+ operating directly against the bucket.
90
+
91
+ The same DuckDB-over-`httpfs` pattern is available to **consumers**
92
+ for cold-data analytics. With the datalake's cloud-storage
93
+ credentials (or a separately-provisioned reader key on the same
94
+ bucket), an external DuckDB process can read the archived NDJSONs
95
+ directly:
96
+
97
+ ```sql
98
+ INSTALL httpfs; LOAD httpfs;
99
+ CREATE OR REPLACE SECRET (TYPE S3, KEY_ID '...', SECRET '...', ENDPOINT '...');
100
+ SELECT * FROM read_ndjson_auto(
101
+ 's3://{bucket}/{datalake_slug}/data-activations/{client_slug}/*/{dataset_table}.ndjson'
102
+ );
103
+ ```
104
+
105
+ The platform does NOT expose a query API over the archive itself —
106
+ cold reads happen externally so the platform's processes aren't
107
+ responsible for analytical workloads. Live data lives in the
108
+ datalake's Postgres halves (queryable via the dataset search
109
+ two-step pattern in `data_activation_clients.md` §6.5); historical
110
+ data lives in the cloud-storage buckets configured here.
111
+
112
+ ### Canonical create body
113
+
114
+ Paste-ready TypeScript body. The four database-role groups
115
+ (unregulated writer + reader, regulated writer + reader) share
116
+ an identical eight-field shape; the type lists all 32 explicitly
117
+ with no role-level union:
118
+
119
+ ```typescript
120
+ import type { DatalakeRequestWritable } from '@alvera-ai/platform-sdk'
121
+
122
+ // Pick one of four supported industries. The body shape below is
123
+ // identical across them; only `data_domain` and the names it
124
+ // interpolates differ.
125
+ const dataDomain = 'healthcare'
126
+ // ^ swap to 'foundation' | 'accounts_receivable' | 'payment_risk'
127
+
128
+ // Secrets resolved upstream by your code (secret store, config loader, etc.)
129
+ const pgHost = '...'
130
+ const pgUser = '...'
131
+ const pgPass = '...'
132
+ const pgReaderHost = '...'
133
+ const pgReaderUser = '...'
134
+ const pgReaderPass = '...'
135
+ const awsAccessKeyId = '...'
136
+ const awsSecretAccessKey = '...'
137
+
138
+ const body: DatalakeRequestWritable = {
139
+ name: 'Production Lake',
140
+ description: `Primary ${dataDomain} datalake`,
141
+ data_domain: dataDomain,
142
+ timezone: 'America/New_York',
143
+ pool_size: 5,
144
+
145
+ unregulated_db_writer_host: pgHost,
146
+ unregulated_db_writer_port: 5432,
147
+ unregulated_db_writer_name: `alvera_${dataDomain}`,
148
+ unregulated_db_writer_schema: 'tenant_acme_unreg',
149
+ unregulated_db_writer_auth_method: 'password',
150
+ unregulated_db_writer_user: pgUser,
151
+ unregulated_db_writer_pass: pgPass,
152
+ unregulated_db_writer_enable_ssl: true,
153
+ unregulated_db_reader_host: pgReaderHost,
154
+ unregulated_db_reader_port: 5432,
155
+ unregulated_db_reader_name: `alvera_${dataDomain}`,
156
+ unregulated_db_reader_schema: 'tenant_acme_unreg',
157
+ unregulated_db_reader_auth_method: 'password',
158
+ unregulated_db_reader_user: pgReaderUser,
159
+ unregulated_db_reader_pass: pgReaderPass,
160
+ unregulated_db_reader_enable_ssl: true,
161
+
162
+ regulated_data_db_writer_host: pgHost,
163
+ regulated_data_db_writer_port: 5432,
164
+ regulated_data_db_writer_name: `alvera_${dataDomain}`,
165
+ regulated_data_db_writer_schema: 'tenant_acme_reg',
166
+ regulated_data_db_writer_auth_method: 'password',
167
+ regulated_data_db_writer_user: pgUser,
168
+ regulated_data_db_writer_pass: pgPass,
169
+ regulated_data_db_writer_enable_ssl: true,
170
+ regulated_data_db_reader_host: pgReaderHost,
171
+ regulated_data_db_reader_port: 5432,
172
+ regulated_data_db_reader_name: `alvera_${dataDomain}`,
173
+ regulated_data_db_reader_schema: 'tenant_acme_reg',
174
+ regulated_data_db_reader_auth_method: 'password',
175
+ regulated_data_db_reader_user: pgReaderUser,
176
+ regulated_data_db_reader_pass: pgReaderPass,
177
+ regulated_data_db_reader_enable_ssl: true,
178
+
179
+ unregulated_cloud_storage: {
180
+ cloud_storage_type: 'aws',
181
+ region: 'us-east-1',
182
+ bucket: `acme-${dataDomain}-unregulated`,
183
+ access_key_id: awsAccessKeyId,
184
+ secret_access_key: awsSecretAccessKey,
185
+ },
186
+ regulated_cloud_storage: {
187
+ cloud_storage_type: 'aws',
188
+ region: 'us-east-1',
189
+ bucket: `acme-${dataDomain}-regulated`,
190
+ access_key_id: awsAccessKeyId,
191
+ secret_access_key: awsSecretAccessKey,
192
+ },
193
+ }
194
+
195
+ const { data: created } = await api.datalakes.create(tenantSlug, body)
196
+ // created.id, created.slug — server-derived; use slug for downstream scoping
197
+ // created.status === 'new' — migration NOT yet started; see §5 Lifecycle
198
+ ```
199
+
200
+ ## 2. Rules the type cannot encode
201
+
202
+ ### `auth_method = "iam_role"` relaxes the password requirement
203
+
204
+ All eight `*_pass` fields are typed as required strings. When a
205
+ role's `auth_method` is `"iam_role"`, its matching `*_pass` value
206
+ is **server-cleared at write time** — the value submitted is
207
+ ignored. The typed validator does not encode this conditional, so
208
+ the field is still structurally required.
209
+
210
+ The convention is to supply a sentinel literal (e.g.
211
+ `"unused-iam-role"`) so a reviewer sees intent. **Real credentials
212
+ do not belong in this slot** — the value never leaves the
213
+ validator.
214
+
215
+ ### AWS cloud-storage `auth_method = iam_role` forbids credentials
216
+
217
+ Mirrors the DB `auth_method` rule above but with the **opposite
218
+ disposition** on the credential fields. Inside an AWS cloud-storage
219
+ embed:
220
+
221
+ - `auth_method: "access_key"` (default) — `access_key_id` and
222
+ `secret_access_key` are **required**.
223
+ - `auth_method: "iam_role"` — `access_key_id` and
224
+ `secret_access_key` MUST be omitted. Supplying either yields a
225
+ field-level rejection at create. Unlike the DB case, the values
226
+ are NOT silently server-cleared — the platform treats their
227
+ presence as a configuration conflict.
228
+
229
+ The two embeds (`unregulated_cloud_storage` and
230
+ `regulated_cloud_storage`) are validated independently, so a body
231
+ can pair `iam_role` on one side with `access_key` on the other.
232
+
233
+ ### Database identifiers follow strict format rules
234
+
235
+ The four `*_db_*_schema` and four `*_db_*_name` fields must be
236
+ lowercase identifiers (letters, digits, underscore) — uppercase
237
+ letters and dashes are rejected with a `/<role>_schema` or
238
+ `/<role>_name` pointer. The eight `*_db_*_host` fields must match
239
+ a hostname pattern; values like `"invalid_host!"` (with
240
+ non-hostname punctuation) are rejected.
241
+
242
+ This is the only field-format constraint surfaced before the
243
+ synchronous reachability probe runs (see §6 Gotcha 9) — a
244
+ malformed identifier never gets as far as a connection attempt.
245
+
246
+ ### Schema names must be unique within a database
247
+
248
+ `*_db_*_schema` fields name the PostgreSQL schema the platform
249
+ creates inside the target database. Sibling datalakes sharing a
250
+ `*_db_*_name` (same physical database) **must** use distinct
251
+ schema values. Collision produces a 422 at create time.
252
+
253
+ The reader and writer for the same role typically share a schema
254
+ (they're two connection profiles into the same logical schema);
255
+ unregulated and regulated must not (different trust boundaries).
256
+
257
+ ### Discriminator on PUT must match
258
+
259
+ Per `mutations.md`, `cloud_storage_type` on update must match the
260
+ existing row's discriminator. Changing the embed type (e.g. `aws`
261
+ → `r2`) requires DELETE + recreate, not PUT.
262
+
263
+ ### `data_domain` is a closed enum
264
+
265
+ Industry values are enumerated in the OpenAPI spec
266
+ (`DatalakeDataDomain`); the SDK exports them as a const-style
267
+ value (see `type_naming.md` "Enum types"). The enum drifts as
268
+ new industries land — consult the SDK's `DatalakeDataDomain`
269
+ const for the live set rather than hardcoding values.
270
+
271
+ ## 3. Field ownership
272
+
273
+ Datalake's fields split across three POV-relevant categories —
274
+ distinguishing them by whether they round-trip determines how a
275
+ consumer constructs the PUT body (see §5 Update).
276
+
277
+ **Server-derived (Response-only).** The universal set (`id`,
278
+ `slug`, `created_at`, `updated_at`, `status`, `status_reason`)
279
+ is documented in `type_naming.md`. Datalake adds no extras here
280
+ — but the `status` field has a richer lifecycle than most
281
+ resources (see §5).
282
+
283
+ **Caller-supplied (round-trip).** Present on both
284
+ `DatalakeRequestWritable` and `DatalakeResponse`:
285
+
286
+ ```
287
+ identification name, description, data_domain
288
+ global settings timezone, pool_size
289
+ db profiles (×4) <role>_{host, port, name, schema, user,
290
+ auth_method, enable_ssl}
291
+ cloud storage (×2) <role>_cloud_storage.{cloud_storage_type,
292
+ bucket, region,
293
+ + branch-specific
294
+ fields per §1}
295
+ ```
296
+
297
+ **Write-only (Request-only).** Present on
298
+ `DatalakeRequestWritable`, absent from `DatalakeResponse`. The
299
+ consumer must re-supply these on every PUT from their secret
300
+ store (see §5 Update for the canonical pattern):
301
+
302
+ ```
303
+ db credentials (×4) unregulated_db_writer_pass
304
+ unregulated_db_reader_pass
305
+ regulated_data_db_writer_pass
306
+ regulated_data_db_reader_pass
307
+
308
+ cloud storage unregulated_cloud_storage.access_key_id
309
+ credentials (×4) unregulated_cloud_storage.secret_access_key
310
+ regulated_cloud_storage.access_key_id
311
+ regulated_cloud_storage.secret_access_key
312
+ ```
313
+
314
+ The `slug` returned in the Response is the canonical reference
315
+ for all downstream scoping. Never pre-compute it client-side —
316
+ see `type_naming.md` "Never pre-compute the slug client-side".
317
+
318
+ ## 4. Error envelopes
319
+
320
+ The platform returns a standardized JSON:API envelope — see
321
+ `errors.md` for the canonical shape (`{ errors: [{ source:
322
+ { pointer }, title, detail }, ...] }`) and the two-layer
323
+ validation model (structural cast → server-side semantic check).
324
+
325
+ All validation errors come back from the server as a thrown 422
326
+ `AlveraApiError` (the SDK does not validate client-side). See
327
+ `errors.md` for the typed narrowing pattern.
328
+
329
+ Datalake-specific common rejections:
330
+
331
+ | `source.pointer` | Typical cause |
332
+ |-------------------------------------------------|------------------------------------------------------|
333
+ | `/name` | name collision within tenant (uniqueness) |
334
+ | `/data_domain` | value not in the live enum |
335
+ | `/unregulated_cloud_storage/cloud_storage_type` | discriminator value doesn't match any oneOf branch |
336
+ | `/unregulated_db_writer_schema` | schema-name collision within the target DB; or identifier format violation (uppercase / dash) |
337
+ | `/unregulated_db_writer_name` | identifier format violation |
338
+ | `/unregulated_db_writer_host` | hostname format violation, OR synchronous probe could not reach the host |
339
+ | `/unregulated_cloud_storage` | synchronous probe could not reach the bucket with the supplied credentials |
340
+ | `/unregulated_cloud_storage/access_key_id` | supplied under `auth_method: "iam_role"` (must be omitted) |
341
+ | `/pool_size` | out of the inclusive range 1–50 |
342
+ | `/timezone` | not a valid IANA timezone identifier |
343
+
344
+ For polymorphic embeds, the `cloud_storage_type` value is the
345
+ first thing to inspect — it routes downstream field validation to
346
+ one of the three branches.
347
+
348
+ ## 5. Lifecycle
349
+
350
+ Datalakes have the richest lifecycle of any platform resource.
351
+ The status enum (wire-level values):
352
+
353
+ ```
354
+ new POST returned; migration not yet started
355
+ processing platform provisioning the per-datalake schemas and
356
+ supporting infrastructure
357
+ ready both DBs migrated; downstream Datalake-DB-resident
358
+ resources can be created
359
+ ```
360
+
361
+ The platform console UI renders `processing` as the label
362
+ "Migrating..." — the wire-level enum value is `processing`.
363
+ Code that switches on status MUST use the wire values, not the
364
+ UI labels.
365
+
366
+ The status enum has NO terminal failure value. If a migration
367
+ fails, the datalake remains in `processing` (or `new` if it
368
+ never started); failures surface in the admin console's Migration
369
+ Logs view, which is not exposed over the SDK. Consumers polling
370
+ for readiness must therefore rely on a sensible timeout — see §5.
371
+
372
+ ### Create
373
+
374
+ ```typescript
375
+ const { data: created } = await api.datalakes.create(tenantSlug, body)
376
+ // created.status === 'new'
377
+ // created.slug === '<server-derived>'
378
+ ```
379
+
380
+ `POST` returns immediately with `status: "new"`. **Migration is
381
+ not auto-enqueued** — the caller explicitly invokes it.
382
+
383
+ ### Migrate (explicit, async)
384
+
385
+ ```typescript
386
+ const { data: enqueued } = await api.datalakes.migrate(
387
+ tenantSlug,
388
+ created.slug,
389
+ )
390
+ // enqueued.status === 'enqueued'
391
+ // enqueued.job_id === <number>
392
+ // enqueued.datalake_id === created.id
393
+ ```
394
+
395
+ `api.datalakes.migrate(...)` enqueues the schema-creation job. The
396
+ response is the *job acknowledgement* (`status: "enqueued"`), NOT
397
+ the datalake row — it carries `job_id` + `datalake_id` +
398
+ `enqueued_at` only. To learn the datalake's live status, GET
399
+ separately:
400
+
401
+ ```typescript
402
+ const { data: dl } = await api.datalakes.get(tenantSlug, created.slug)
403
+ // dl.status === 'new' | 'processing' | 'ready'
404
+ ```
405
+
406
+ Use the canonical `waitUntilReady` from `async.md`:
407
+
408
+ ```typescript
409
+ const dl = await waitUntilReady(
410
+ () => api.datalakes.get(tenantSlug, created.slug),
411
+ (status) => status === 'ready' ? 'ready' : 'pending',
412
+ { intervalMs: 15_000, timeoutMs: 5 * 60_000 },
413
+ )
414
+ ```
415
+
416
+ Local-dev migration takes seconds to a minute; remote deployments
417
+ can run several minutes. **Most** downstream resource creates
418
+ must wait for the datalake to reach `ready` — but the rule has
419
+ one exception worth understanding.
420
+
421
+ Downstream resources split into two categories by storage
422
+ location:
423
+
424
+ | Where the resource's state lives | Examples | Wait for `ready`? |
425
+ |----------------------------------|------------------------------------------------|-------------------|
426
+ | Platform DB (tenant-wide) | Data sources | No |
427
+ | Per-datalake DB (tenant + datalake schemas) | Tools, AI agents, generic tables, interoperability contracts, data activation clients, workflows | **Yes** |
428
+
429
+ Data sources can be created against a `new`-status datalake
430
+ because their metadata writes to the platform DB, not the
431
+ per-datalake schemas. Everything else needs the per-datalake
432
+ schemas in place — POSTing them against an unready datalake
433
+ produces server-side errors (table-not-found, capability-gate,
434
+ etc.) that are recoverable but indicate a missing wait.
435
+
436
+ In practice: a sibling datalake left unmigrated is still a valid
437
+ target for data-source creation, since data sources are
438
+ platform-DB-resident and don't depend on per-datalake readiness.
439
+
440
+ ### Update
441
+
442
+ `PUT` replays the full body per `mutations.md`. Common pattern:
443
+ GET current → re-supply write-only fields from your secret store →
444
+ mutate target fields → PUT the whole shape. Changing the
445
+ cloud-storage discriminator is rejected — DELETE + recreate
446
+ instead.
447
+
448
+ **`DatalakeResponse` omits write-only fields.** Eight fields on
449
+ `DatalakeRequestWritable` — the four `*_pass` fields plus both
450
+ cloud-storage embeds' `access_key_id` / `secret_access_key` — are
451
+ write-only: present on the request type, absent from the response
452
+ type. A naive spread of `current` into the PUT body is therefore
453
+ missing those 8 required fields. The server rejects this with a 422
454
+ `AlveraApiError` pointing at the missing fields (see `errors.md`).
455
+ Re-supply them explicitly from your secret store:
456
+
457
+ ```typescript
458
+ const { data: current } = await api.datalakes.get(tenantSlug, datalakeId)
459
+ // `current` is a DatalakeResponse — write-only fields (DB passwords +
460
+ // cloud-storage secrets) are not present on the response type.
461
+
462
+ const next = {
463
+ ...current,
464
+
465
+ // Re-supply every write-only field from your secret store:
466
+ unregulated_db_writer_pass: pgPass,
467
+ unregulated_db_reader_pass: pgReaderPass,
468
+ regulated_data_db_writer_pass: pgPass,
469
+ regulated_data_db_reader_pass: pgReaderPass,
470
+ unregulated_cloud_storage: {
471
+ ...current.unregulated_cloud_storage,
472
+ access_key_id: awsAccessKeyId,
473
+ secret_access_key: awsSecretAccessKey,
474
+ },
475
+ regulated_cloud_storage: {
476
+ ...current.regulated_cloud_storage,
477
+ access_key_id: awsAccessKeyId,
478
+ secret_access_key: awsSecretAccessKey,
479
+ },
480
+
481
+ // Mutate your target fields:
482
+ description: 'Now with revised description',
483
+ }
484
+
485
+ const { data: updated } = await api.datalakes.update(
486
+ tenantSlug,
487
+ datalakeId,
488
+ next,
489
+ )
490
+ // updated.id === datalakeId — path-key stable
491
+ // updated.description === 'Now with revised description'
492
+ ```
493
+
494
+ The canonical pattern: caller re-supplies the write-only DB +
495
+ cloud-storage fields verbatim, mutates only what changes (e.g.
496
+ `description`). The eight write-only fields are enumerated in
497
+ §3 Field ownership.
498
+
499
+ **PUT fires server-side connectivity probes before persisting.**
500
+ Submitting a body produces a successful 200 only after two
501
+ probes pass for each role profile:
502
+
503
+ - **Cloud-storage probe** — verifies the bucket exists + the
504
+ supplied `access_key_id` / `secret_access_key` (or IAM role)
505
+ can list it.
506
+ - **DB-connection probe** — verifies the platform can connect
507
+ to the supplied host:port with the supplied user + auth
508
+ method.
509
+
510
+ A body that's structurally valid but carries stale or wrong
511
+ credentials therefore fails at PUT time, not at later write
512
+ time. The 422 envelope (see `errors.md`) names the failing probe
513
+ via `source.pointer` (e.g. `/unregulated_cloud_storage` or
514
+ `/regulated_data_db_writer_host`). This makes drift-then-PUT
515
+ workflows fail fast and locally rather than mid-pipeline.
516
+
517
+ Most field changes do NOT re-trigger migration. Schema-name or
518
+ auth-method changes DO — the status reverts to `processing` and
519
+ downstream operations must wait again.
520
+
521
+ `api.datalakes.update(...)` accepts the datalake's `id` OR `slug`
522
+ as the path key (same as `.get`); both are stable across migrate
523
+ status changes.
524
+
525
+ ### Delete
526
+
527
+ `DELETE` requires the datalake to be empty of dependent resources
528
+ (no data sources, tools, AI agents, workflows scoped under it). A
529
+ non-empty datalake DELETE returns a 422 with a `source.pointer`
530
+ naming the blocking dependency type. Cleanup is operator-driven in
531
+ most deployments.
532
+
533
+ ### Read shapes
534
+
535
+ `api.datalakes` exposes five distinct read methods, each
536
+ returning a different shape:
537
+
538
+ | Method | Returns |
539
+ |--------------------------------------|-------------------------------------------------------------|
540
+ | `.list(tenantSlug)` | Paged list: `{ data: DatalakeResponse[], meta }` |
541
+ | `.get(tenantSlug, idOrSlug)` | One row: `DatalakeResponse` |
542
+ | `.metadata(tenantSlug, slug)` | **Markdown string** describing the lake |
543
+ | `.indexMetadata(tenantSlug)` | **Markdown string** listing every lake |
544
+ | `.systemDatasets(tenantSlug, slug)` | `{ datasets: string[] }` — industry-built dataset names |
545
+
546
+ The two `metadata` methods return `string`, not a structured
547
+ object — they're agent-facing summaries (`data: string`, not `data:
548
+ {...}`). Test the shape with `typeof data === 'string'`.
549
+
550
+ `.indexMetadata(tenantSlug)` body begins with `# Datalakes on
551
+ tenant` and lists every datalake by slug — useful for agent-side
552
+ discovery without paging through `.list`.
553
+
554
+ `.systemDatasets(...)` enumerates the dataset names the platform
555
+ auto-creates for the datalake's `data_domain`. Each industry's
556
+ catalog mixes **domain-anchored names** (the primary subject of
557
+ the industry plus related entities) with **cross-cutting concepts**
558
+ (audit logs,
559
+ messages, documents, etc.) that the platform composes per
560
+ industry — agents iterating the catalog should not hardcode a
561
+ closed set of expected names; discover them at runtime via
562
+ this call. The array is sorted, unique, and **excludes
563
+ operator-defined generic tables** — those live under
564
+ `api.genericTables.list(...)`. The
565
+ literal placeholder `"generic_table"` is also absent. Each name
566
+ can be drilled into via
567
+ `api.datasets.metadataDetails(tenantSlug, datalakeSlug, name)` for
568
+ agent-facing markdown, or the whole-domain catalog can be fetched
569
+ in one shot via `api.datasets.metadata(tenantSlug, datalakeSlug)`. Pair `.systemDatasets()` with a
570
+ `page_size: 1` probe per name as a post-migration smoke (proves
571
+ each industry-built table is queryable; the post-ready probe
572
+ code is in §6 Gotcha 3).
573
+
574
+ ```typescript
575
+ const { data: catalog } = await api.datalakes.systemDatasets(
576
+ tenantSlug,
577
+ datalakeSlug,
578
+ )
579
+ // catalog.datasets: sorted, unique array of industry-built dataset names
580
+
581
+ for (const name of catalog.datasets) {
582
+ const { data: md } = await api.datasets.metadataDetails(tenantSlug, datalakeSlug, name)
583
+ // md is a string — agent-facing markdown describing the dataset
584
+ }
585
+ ```
586
+
587
+ ## 6. Gotchas
588
+
589
+ 1. **POST does NOT auto-enqueue migration.** The create response
590
+ returns immediately with `status: 'new'`; Datalake-DB-resident
591
+ downstream resources (tools, AI agents, generic tables,
592
+ interoperability contracts, data activation clients,
593
+ workflows) MUST wait for `ready` (see §5 Lifecycle for the
594
+ storage-location split). The canonical sequence:
595
+
596
+ ```typescript
597
+ const { data: created } = await api.datalakes.create(tenantSlug, body)
598
+ // created.status === 'new'
599
+
600
+ await api.datalakes.migrate(tenantSlug, created.slug)
601
+
602
+ const ready = await waitUntilReady(
603
+ () => api.datalakes.get(tenantSlug, created.slug),
604
+ (s) => s === 'ready' ? 'ready' : 'pending',
605
+ { intervalMs: 15_000, timeoutMs: 5 * 60_000 },
606
+ )
607
+ // ready.status === 'ready' — safe to POST tools, AI agents, etc.
608
+ ```
609
+
610
+ 2. **`migrate` returns the job ack, not the datalake.** The
611
+ `{ status: "enqueued", job_id, ... }` shape can be mistaken for
612
+ the datalake row. To learn live status, `.get` separately and
613
+ poll per `async.md`.
614
+
615
+ 3. **`status: "ready"` is necessary but not sufficient.** A
616
+ partial-migration failure can leave a datalake in `ready`
617
+ with some tables silently missing — a failure mode that
618
+ otherwise only surfaces mid-pipeline (during downstream
619
+ resource creation or live data ingestion), far from its
620
+ root cause.
621
+ Pair the status check with a post-ready probe:
622
+
623
+ ```typescript
624
+ const { data: catalog } = await api.datalakes.systemDatasets(
625
+ tenantSlug, datalakeSlug,
626
+ )
627
+ // catalog.datasets is the industry-built name list —
628
+ // discovered at runtime; don't hardcode names.
629
+
630
+ for (const dataset of catalog.datasets) {
631
+ const { data } = await api.datasets.createUserSearch(
632
+ tenantSlug,
633
+ datalakeSlug,
634
+ dataset,
635
+ { search_query: '1 = 1' },
636
+ )
637
+ if (data.status !== 'completed') {
638
+ throw new Error(
639
+ `post-ready probe failed: dataset=${dataset} status=${data.status}`,
640
+ )
641
+ }
642
+ }
643
+ ```
644
+
645
+ 4. **`id` and `slug` are interchangeable as path keys.**
646
+ `.get`, `.update`, `.migrate`, `.metadata` all accept either.
647
+ The `slug` is human-readable; the `id` is a UUID. Pick one
648
+ convention per code path to avoid mixed references.
649
+
650
+ 5. **Schema-name collisions surface at create, not migrate.** A
651
+ sibling datalake reusing a `*_schema` value gets a 422
652
+ immediately — well before the asynchronous schema-creation job
653
+ would have hit the conflict.
654
+
655
+ 6. **`iam_role` auth still requires a `*_pass` value structurally.**
656
+ The TypeScript type (and the server's schema) treat the field as
657
+ a required string. Supply a sentinel literal (e.g.
658
+ `"unused-iam-role"`); the platform clears it server-side.
659
+
660
+ 7. **`cloud_storage_type` lives inside each embed, not at the
661
+ top level.** A single body has two discriminators
662
+ (`unregulated_cloud_storage.cloud_storage_type` and
663
+ `regulated_cloud_storage.cloud_storage_type`); they're
664
+ independently validated and can carry different values.
665
+
666
+ 8. **DELETE is rare in practice.** Datalakes accrete dependents
667
+ quickly; routine cleanup is via operator-driven console
668
+ actions, not SDK calls. Treat `.delete` as an integration-test
669
+ teardown method, not a production operation.
670
+
671
+ 9. **Create and update run synchronous reachability probes.** Before
672
+ the row is written, the platform opens connections to every
673
+ declared endpoint — all four DB profiles (unregulated reader +
674
+ writer, regulated reader + writer) AND both cloud-storage
675
+ buckets — and only proceeds if every probe succeeds. A
676
+ structurally valid body can still fail with field-level
677
+ rejections on `*_db_*_host` (unreachable, wrong credentials, or
678
+ SSL-mismatch) or `*_cloud_storage` (bucket inaccessible). The
679
+ same probe runs on update; an update that only changes the
680
+ datalake's `name` still re-probes if the body re-supplies any
681
+ `*_host` or `*_cloud_storage` field.
682
+
683
+ Two consumer-relevant consequences:
684
+
685
+ - **Multiple probe failures are surfaced in a single envelope.**
686
+ Two bad reader hosts produce two field-level entries —
687
+ consumer agents iterating on a single response can correct
688
+ several fields per round-trip.
689
+ - **`enable_ssl: true` against a non-SSL database fails the
690
+ probe.** The flag looks client-side but is validated by the
691
+ probe attempting an SSL handshake. Match it to what the
692
+ target database actually supports.
693
+
694
+ 10. **`pool_size` is clamped to the inclusive range 1–50.** The
695
+ typed field accepts any integer; the server rejects values
696
+ outside `[1, 50]` with a `/pool_size` pointer. The error message
697
+ mentions a `greater_than: 0` or `less_than_or_equal_to: 50`
698
+ constraint depending on which side was crossed.
699
+
700
+ 11. **`timezone` is a closed IANA enum.** Submit a non-IANA string
701
+ (e.g. `"Bad/Timezone"` or `"EST"`) and the platform rejects
702
+ with `is not a valid IANA timezone` on `/timezone`. The set of
703
+ valid identifiers is not exposed over the SDK as a discovery
704
+ surface — consumers must source IANA tz database names
705
+ externally (e.g. JavaScript's `Intl.supportedValuesOf('timeZone')`,
706
+ or the IANA tz data directly).
707
+
708
+ 12. **`auth_method` defaults to `"password"` on every DB profile.**
709
+ Omitting `*_auth_method` or supplying `null` is equivalent to
710
+ `"password"` — and consequently requires both `*_user` and
711
+ `*_pass` to be supplied. The IAM-role relaxation in §2 applies
712
+ only when `auth_method` is explicitly `"iam_role"`.