@alvera-ai/platform-sdk 0.17.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/.agent/AGENTS.md +82 -144
  2. package/.agent/account_management.md +2 -2
  3. package/.agent/action_logs.md +4 -4
  4. package/.agent/ai_agents.md +28 -21
  5. package/.agent/ai_sandbox.md +49 -39
  6. package/.agent/connected_apps.md +3 -3
  7. package/.agent/cookbook/_fixtures/README.md +1 -1
  8. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_generic_table.liquid +1 -1
  9. package/.agent/cookbook/_fixtures/organic-marketing/_lead_submissions_foundation_legal_entity.liquid +80 -0
  10. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_mdm.liquid +2 -1
  11. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_generic_table.liquid +57 -0
  12. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_legal_entity.liquid +30 -0
  13. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_mdm.liquid +44 -0
  14. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_generic_table.liquid +57 -0
  15. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_legal_entity.liquid +36 -0
  16. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_mdm.liquid +41 -0
  17. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_generic_table.liquid +70 -0
  18. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_legal_entity.liquid +52 -0
  19. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_mdm.liquid +42 -0
  20. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_generic_table.liquid +38 -0
  21. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_legal_entity.liquid +48 -0
  22. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_mdm.liquid +49 -0
  23. package/.agent/cookbook/organic-marketing.md +2801 -0
  24. package/.agent/cookbook/payments-compliance.md +2180 -0
  25. package/.agent/cookbook/primary-care.md +2175 -0
  26. package/.agent/cookbook/subscription-saas.md +2403 -0
  27. package/.agent/data_activation_clients.md +65 -52
  28. package/.agent/datalakes.md +338 -171
  29. package/.agent/errors.md +3 -3
  30. package/.agent/generic_tables.md +151 -62
  31. package/.agent/interoperability_contracts.md +57 -22
  32. package/.agent/mdm.md +136 -153
  33. package/.agent/messages.md +36 -34
  34. package/.agent/mock-services.md +1 -1
  35. package/.agent/mutations.md +2 -2
  36. package/.agent/templates.md +14 -13
  37. package/.agent/tools.md +63 -21
  38. package/.agent/type_naming.md +13 -13
  39. package/.agent/workflows.md +99 -53
  40. package/README.md +2 -2
  41. package/dist/bin/platform-sdk.mjs +33 -47
  42. package/dist/bin/platform-sdk.mjs.map +1 -1
  43. package/dist/index.d.mts +565 -379
  44. package/dist/index.d.mts.map +1 -1
  45. package/dist/index.mjs +494 -59
  46. package/dist/index.mjs.map +1 -1
  47. package/package.json +4 -3
  48. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +0 -88
  49. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +0 -47
  50. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +0 -24
  51. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +0 -38
  52. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_compliance_screening.liquid +0 -59
  53. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_mdm.liquid +0 -36
  54. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_mdm.liquid +0 -30
  55. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_payment_account.liquid +0 -55
  56. package/.agent/cookbook/_fixtures/subscription/_customers_subscription_mdm.liquid +0 -20
  57. package/.agent/cookbook/_setup/foundation.md +0 -359
  58. package/.agent/cookbook/_setup/healthcare.md +0 -361
  59. package/.agent/cookbook/_setup/payments.md +0 -365
  60. package/.agent/cookbook/_setup/subscription.md +0 -364
  61. package/.agent/cookbook/action-status-updaters.md +0 -278
  62. package/.agent/cookbook/ai-agent-invoke.md +0 -279
  63. package/.agent/cookbook/appointment-review-sms-workflow.md +0 -801
  64. package/.agent/cookbook/birthday-greeting-sms-trigger.md +0 -696
  65. package/.agent/cookbook/bulk-ingest.md +0 -302
  66. package/.agent/cookbook/contact-us-triage-with-llm.md +0 -663
  67. package/.agent/cookbook/dunning-sms-for-delinquent.md +0 -659
  68. package/.agent/cookbook/generic-tables.md +0 -244
  69. package/.agent/cookbook/invite-team.md +0 -200
  70. package/.agent/cookbook/kyc-notification-on-account-activation.md +0 -661
  71. package/.agent/cookbook/marketing-campaign-send.md +0 -1044
  72. package/.agent/cookbook/paginated-restapi-poller.md +0 -383
  73. package/.agent/cookbook/rest-fetch.md +0 -273
  74. package/.agent/cookbook/sanctions-screening-with-agent-review.md +0 -773
  75. package/.agent/cookbook/score-leads-with-llm-categorization.md +0 -665
  76. package/.agent/cookbook/system-templates.md +0 -165
  77. package/.agent/cookbook/talk-to-data.md +0 -178
  78. package/.agent/cookbook/triage-prospects-by-priority.md +0 -571
  79. package/.agent/cookbook/welcome-sms-for-customers.md +0 -647
  80. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/memorandum-of-association-01.png +0 -0
  81. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/sample_two_page.pdf +0 -0
  82. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/_customers_subscription_customer.liquid +0 -0
  83. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/stripe_customers_batch1.csv +0 -0
@@ -0,0 +1,2180 @@
1
+ ---
2
+ title: "Payments compliance: screen, decide, notify — on a tokenized lake"
3
+ summary: The whole payments-compliance surface in one walk. Stand up a tenant and a raw datalake, turn tokenization on because a model is going to read this data, then run two scenarios end to end — an LLM agent disambiguating gray-zone sanctions screenings into a verdict-keyed SMS fan-out, and a KYC notification fired by an account activation. Ends by reading one row back through raw, tokenized and redacted keys to show what each reader actually sees, then inspecting how far the tokenized copy has got and repairing one of its tables.
4
+ use_case: payments-compliance
5
+ slug: payments-compliance
6
+ vitest_source:
7
+ - integration-tests/tests/payments-compliance/sanctions-review-workflow.test.ts
8
+ - integration-tests/tests/payments-compliance/kyc-upload-workflow.test.ts
9
+ - integration-tests/tests/payments-compliance/interoperability-contracts.test.ts
10
+ - integration-tests/tests/payments-compliance/run-dac-single.test.ts
11
+ - integration-tests/tests/payments-compliance/create-dac.test.ts
12
+ - integration-tests/tests/payments-compliance/system-templates.test.ts
13
+ - integration-tests/tests/payments-compliance/manual-upload-tool.test.ts
14
+ - integration-tests/tests/payments-compliance/data-sources.test.ts
15
+ - integration-tests/tests/workspace/bootstrap.test.ts
16
+ status: draft
17
+ ---
18
+
19
+ # Problem
20
+
21
+ A payments compliance team has two jobs that look like one. Screenings
22
+ arrive from a provider and most of them answer themselves — a clean pass,
23
+ an obvious hit. The ones in the middle are the work, and they are the ones
24
+ a human should not be reading first. Separately, an account activation has
25
+ to reach the holder as a notification, promptly, and not at all if the
26
+ account was suspended instead.
27
+
28
+ Both are the same shape: rows land, a filter decides which ones matter, and
29
+ something is sent about the ones that do. This walk builds both on one
30
+ tenant, because a compliance team does not run two platforms.
31
+
32
+ It also turns **tokenization** on, and that is the point of doing this
33
+ surface rather than another one. A gray-zone screening is disambiguated by
34
+ a language model. The model needs the shape of the record — the score, the
35
+ match count, the type — and it does not need the name of the person. The
36
+ tokenized copy of the lake is what makes that distinction enforceable
37
+ rather than a promise, so this walk provisions it in §007 and reads
38
+ through it in §037.
39
+
40
+ # Composition
41
+
42
+ | Resource | Why it exists here |
43
+ |-------------------------------------|---------------------------------------------|
44
+ | Tenant + raw datalake | The compliance team's own world |
45
+ | Tokenized + redacted datalakes | A model reads this data — see §007 |
46
+ | SMS tool, LLM tool, Manual Upload | Shared by both scenarios |
47
+ | Atomic FI data source | Where both feeds claim to come from |
48
+ | Sanctions Review AI agent | Disambiguates the gray zone |
49
+ | Two generic tables + contract pairs | Screenings, and payment accounts |
50
+ | Four Data Activation Clients | Identity then data, per pair — see §020 |
51
+ | Two agentic workflows | Verdict fan-out, and activation notice |
52
+
53
+ # Walkthrough
54
+
55
+ This cookbook is **self-reliant**: it stands up everything it uses and
56
+ depends on no other file. It is also **idempotent** — every creating step
57
+ looks first and creates only what is missing, so running it twice costs
58
+ what running it once cost. That is not a nicety. Nothing in the platform
59
+ reclaims an abandoned datalake, and a walk that mints a fresh tenant per
60
+ run leaves every previous run's lakes behind forever.
61
+
62
+ ## 001 — sign in as root, and make sure the admin user exists
63
+
64
+ Authenticate as the platform's root admin (`admin@dev.local` /
65
+ `devpassword` in local dev) via the tenantless bootstrap login — keyless by
66
+ structural necessity, since no tenant exists yet to scope a key to, and a
67
+ dev/test-only surface.
68
+
69
+ Then make sure the admin user this walk runs as exists. **The email is
70
+ stable, not per-run**, which is what makes the step idempotent — and it
71
+ means the second run finds the user already signed up. A duplicate signup
72
+ is refused, so the refusal is caught and inspected: if it says the email is
73
+ taken, that is the idempotent path and the walk continues. Any other
74
+ failure is re-raised, because swallowing it would turn a real auth problem
75
+ into a confusing failure three steps later.
76
+
77
+ ```typescript
78
+ ctx.rootSession = await createBootstrapSession({
79
+ baseUrl: process.env.ALVERA_BASE_URL!,
80
+ email: process.env.ALVERA_ROOT_EMAIL!,
81
+ password: process.env.ALVERA_ROOT_PASSWORD!,
82
+ })
83
+ ctx.rootApi = createIsolatedPlatformApi({
84
+ baseUrl: process.env.ALVERA_BASE_URL!,
85
+ sessionToken: ctx.rootSession.sessionToken,
86
+ apiKey: '',
87
+ })
88
+
89
+ ctx.sarahEmail = 'cookbook-payments-compliance@dev.local'
90
+ ctx.sarahPassword = 'CookbookPass1!'
91
+
92
+ try {
93
+ const signUpResp = await ctx.rootApi.admin.signUp({
94
+ email: ctx.sarahEmail,
95
+ password: ctx.sarahPassword,
96
+ first_name: 'Cookbook',
97
+ last_name: 'Compliance',
98
+ })
99
+ await ctx.rootApi.admin.confirmUser(signUpResp.data.id!)
100
+ } catch (err) {
101
+ // Already provisioned by a previous run. Confirm that is what happened
102
+ // rather than assuming it — a genuine signup failure must not be read as
103
+ // "already there".
104
+ const detail = JSON.stringify((err as { errors?: unknown }).errors ?? err)
105
+ if (!/taken|already|exist/i.test(detail)) throw err
106
+ }
107
+ ```
108
+
109
+ ## 002 — the find-or-create helper every later step uses
110
+
111
+ Idempotence is one question asked over and over: *is this already here?*
112
+ Rather than answer it eleven different ways, the walk defines it once.
113
+
114
+ `ensure` takes a label, a lookup and a create. It runs the lookup, returns
115
+ what it finds, and only creates when the lookup comes back empty. The label
116
+ is not decoration — when a run reuses something you did not expect it to,
117
+ the log line naming it is how you find out.
118
+
119
+ The helper lives on `ctx` rather than as a bare function because each
120
+ numbered step compiles into its own `it()` block, so a plain `function`
121
+ here would not be in scope for the steps that call it.
122
+
123
+ ```typescript
124
+ ctx.ensure = async <T>(
125
+ label: string,
126
+ find: () => Promise<T | undefined>,
127
+ create: () => Promise<T>,
128
+ ): Promise<T> => {
129
+ const existing = await find()
130
+ if (existing !== undefined) {
131
+ console.log(` ↻ reusing ${label}`)
132
+ return existing
133
+ }
134
+ console.log(` + creating ${label}`)
135
+ return await create()
136
+ }
137
+ ```
138
+
139
+ ## 003 — the tenant
140
+
141
+ Sarah signs in without a tenant scope — she may not belong to one yet — and
142
+ the walk finds or creates the compliance tenant by its stable name. The
143
+ server derives the slug; capture it, because every later call is addressed
144
+ by it.
145
+
146
+ ```typescript
147
+ ctx.sarahTenantlessSession = await createBootstrapSession({
148
+ baseUrl: process.env.ALVERA_BASE_URL!,
149
+ email: ctx.sarahEmail,
150
+ password: ctx.sarahPassword,
151
+ })
152
+ ctx.sarahTenantlessApi = createIsolatedPlatformApi({
153
+ baseUrl: process.env.ALVERA_BASE_URL!,
154
+ sessionToken: ctx.sarahTenantlessSession.sessionToken,
155
+ apiKey: '',
156
+ })
157
+
158
+ const TENANT_NAME = 'Cookbook Payments Compliance'
159
+
160
+ const tenant = await ctx.ensure(
161
+ `tenant ${TENANT_NAME}`,
162
+ async () => {
163
+ const { data } = await ctx.sarahTenantlessApi.tenants.list()
164
+ return (data.data ?? []).find((t: { name?: string }) => t.name === TENANT_NAME)
165
+ },
166
+ async () => {
167
+ const { data } = await ctx.sarahTenantlessApi.tenants.create({ name: TENANT_NAME })
168
+ return data
169
+ },
170
+ )
171
+ tenantSlug = tenant.slug!
172
+ ```
173
+
174
+ ## 004 — the tenant-scoped client
175
+
176
+ A tenant-scoped login requires `X-API-Key`, so a key has to exist before
177
+ sarah can sign in against the tenant. Mint one through the platform-admin
178
+ side door with the root bearer; in the web console this is
179
+ *Settings → API Keys*.
180
+
181
+ `data_access_mode: 'raw'` is deliberate. Raw is the superset, and every
182
+ read-back in this walk is a question about **this walk's own data** — not
183
+ about whether replication has caught up. The tokenized and redacted reads
184
+ are a separate subject with their own step and their own keys (§037).
185
+
186
+ **This is the one step in the walk that is not idempotent**, and it is
187
+ worth knowing why rather than discovering it. There is no endpoint that
188
+ lists a tenant's API keys, so there is nothing to look the existing one up
189
+ with — `ensure` has no lookup to run. Each run therefore mints another key.
190
+ A key row is cheap where a datalake is not, so the walk accepts it; if you
191
+ are counting rows in a shared environment, this is the one to count.
192
+
193
+ ```typescript
194
+ const { data: mintedKey } = await ctx.rootApi.admin.createTenantApiKey(tenantSlug, {
195
+ name: 'Cookbook Payments Compliance Key',
196
+ data_access_mode: 'raw',
197
+ })
198
+ ctx.tenantApiKey = mintedKey.api_key
199
+
200
+ const sarahTenantSession = await createSession({
201
+ baseUrl: process.env.ALVERA_BASE_URL!,
202
+ email: ctx.sarahEmail,
203
+ password: ctx.sarahPassword,
204
+ tenantSlug,
205
+ apiKey: ctx.tenantApiKey,
206
+ })
207
+ api = createIsolatedPlatformApi({
208
+ baseUrl: process.env.ALVERA_BASE_URL!,
209
+ sessionToken: sarahTenantSession.sessionToken,
210
+ apiKey: ctx.tenantApiKey,
211
+ })
212
+ ```
213
+
214
+ ## 005 — the datalake
215
+
216
+ One datalake, created `raw`. A datalake is one database and one bucket; the
217
+ tokenized and redacted copies are separate rows provisioned in §007 and
218
+ filled by replication, never written to directly.
219
+
220
+ The lookup filters on `type === 'raw'`, and that matters from the second
221
+ run onwards: once §007 has run, `datalakes.list` returns three rows, and
222
+ two of them are derived copies that must never be mistaken for the primary.
223
+
224
+ Local-dev defaults match the seeded `dev.exs` setup — `postgres` on
225
+ `localhost:5432`, database `alvera_dev_payments`, LocalStack S3 on
226
+ `localhost:4566` — and the schema name is stable, because a fresh schema
227
+ per run is the same leak as a fresh tenant per run.
228
+
229
+ ```typescript
230
+ const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_payments' }
231
+ const DB_SCHEMA = 'cookbook_payments_compliance'
232
+ const LAKE_NAME = 'Cookbook Payments Compliance Datalake'
233
+
234
+ const S3 = {
235
+ cloud_storage_type: 'aws' as const,
236
+ region: 'us-east-1',
237
+ access_key_id: 'test',
238
+ secret_access_key: 'test',
239
+ endpoint: 'http://localhost:4566',
240
+ }
241
+
242
+ const datalake = await ctx.ensure(
243
+ `datalake ${LAKE_NAME}`,
244
+ async () => {
245
+ const { data } = await api.datalakes.list(tenantSlug)
246
+ return (data.data ?? []).find(
247
+ (l: { name?: string; type?: string }) => l.name === LAKE_NAME && l.type === 'raw',
248
+ )
249
+ },
250
+ async () => {
251
+ const { data } = await api.datalakes.create(tenantSlug, {
252
+ name: LAKE_NAME,
253
+ description: 'Payments compliance datalake provisioned by the cookbook doctest.',
254
+ timezone: 'America/New_York',
255
+ pool_size: 3,
256
+ type: 'raw',
257
+
258
+ db_writer_host: DB.host,
259
+ db_writer_port: DB.port,
260
+ db_writer_name: DB.name,
261
+ db_writer_schema: DB_SCHEMA,
262
+ db_writer_auth_method: 'password',
263
+ db_writer_user: DB.user,
264
+ db_writer_pass: DB.pass,
265
+ db_writer_enable_ssl: false,
266
+ db_reader_host: DB.host,
267
+ db_reader_port: DB.port,
268
+ db_reader_name: DB.name,
269
+ db_reader_schema: DB_SCHEMA,
270
+ db_reader_auth_method: 'password',
271
+ db_reader_user: DB.user,
272
+ db_reader_pass: DB.pass,
273
+ db_reader_enable_ssl: false,
274
+
275
+ cloud_storage: { ...S3, bucket: 'alvera-platform-dev', base_path: 'cookbook/payments-compliance' },
276
+ })
277
+ return data
278
+ },
279
+ )
280
+ datalakeSlug = datalake.slug!
281
+ ctx.datalakeId = datalake.id!
282
+ ```
283
+
284
+ ## 006 — run the migrations, and wait for ready
285
+
286
+ `datalakes.create` persists the row at `status: 'new'`; it does not run the
287
+ schema DDL. Migration is triggered separately so the operator decides when
288
+ the potentially-slow part happens. `datalakes.migrate` enqueues the job and
289
+ returns immediately with `status: 'enqueued'`; the poll after it is what
290
+ waits for the worker.
291
+
292
+ Migrating is safe to repeat, which is what lets this step stay unguarded on
293
+ a second run. The wait exits the moment the lake reports `ready`, so the
294
+ five-minute ceiling is only ever paid in failure.
295
+
296
+ ```typescript
297
+ const migrateResp = await api.datalakes.migrate(tenantSlug, datalakeSlug)
298
+ if (migrateResp.data.status !== 'enqueued') {
299
+ throw new Error(`datalake migration not enqueued (status: ${migrateResp.data.status})`)
300
+ }
301
+
302
+ const READY_TIMEOUT_MS = 5 * 60_000
303
+ const deadline = Date.now() + READY_TIMEOUT_MS
304
+ let datalakeStatus: string | undefined
305
+ while (Date.now() < deadline) {
306
+ const { data } = await api.datalakes.get(tenantSlug, ctx.datalakeId)
307
+ datalakeStatus = data.status
308
+ if (datalakeStatus === 'ready') break
309
+ await new Promise((r) => setTimeout(r, 5_000))
310
+ }
311
+ if (datalakeStatus !== 'ready') {
312
+ throw new Error(`datalake did not reach :ready within ${READY_TIMEOUT_MS}ms (last: ${datalakeStatus})`)
313
+ }
314
+ ```
315
+
316
+ ## 007 — turn tokenization on, because a model is going to read this
317
+
318
+ A datalake is created raw and stays raw. Tokenization is one separate call,
319
+ and it is a decision, not a default: *the tokenized copy is for a machine,
320
+ the redacted copy is for a person*, and you turn them on when one of those
321
+ readers is about to look. Here one is — the agent in §016 reads screenings
322
+ to disambiguate them, and it has no business seeing who was screened.
323
+
324
+ `provisionTokenization` creates both derived lakes and enqueues a migration
325
+ for each. It is **both-or-neither**: a run that provisions the tokenized
326
+ lake and cannot provision the redacted one reports the failure rather than
327
+ leaving half a pair. It is **idempotent**, so this step needs no `ensure`
328
+ around it — already-provisioned lakes come back as they are.
329
+
330
+ **Order matters.** The raw lake must be `ready` before this runs. A derived
331
+ lake is provisioned empty and its own migration is what creates its tables
332
+ and starts the copy; provision too early and you get lakes that exist and
333
+ stay empty forever. That is why §006 waits.
334
+
335
+ ```typescript
336
+ const { data: provisioned } = await api.datalakes.provisionTokenization(tenantSlug, datalakeSlug)
337
+ const derivedLakes = provisioned.derived_datalakes ?? []
338
+ if (derivedLakes.length === 0) {
339
+ throw new Error('provisionTokenization returned no derived datalakes')
340
+ }
341
+
342
+ // Each derived lake migrates on its own job. Poll the DERIVED ids — the raw
343
+ // lake's own status says nothing about whether its copies are ready.
344
+ const DERIVED_TIMEOUT_MS = 5 * 60_000
345
+ let pendingLakes = derivedLakes.map((d) => d.id!)
346
+ const derivedDeadline = Date.now() + DERIVED_TIMEOUT_MS
347
+ while (Date.now() < derivedDeadline && pendingLakes.length > 0) {
348
+ const stillPending: string[] = []
349
+ for (const id of pendingLakes) {
350
+ const { data } = await api.datalakes.get(tenantSlug, id)
351
+ if (data.status !== 'ready') stillPending.push(id)
352
+ }
353
+ pendingLakes = stillPending
354
+ if (pendingLakes.length > 0) await new Promise((r) => setTimeout(r, 5_000))
355
+ }
356
+ if (pendingLakes.length > 0) {
357
+ throw new Error(`derived datalakes did not reach :ready: ${pendingLakes.join(', ')}`)
358
+ }
359
+ ```
360
+
361
+ ## 008 — three helpers the scenarios below share
362
+
363
+ **`waitForFiredRun`.** `workflows.run` only *schedules* a run. It returns
364
+ immediately with a `workflow_run_id`, and the `workflow_run_log_id` and
365
+ `batch_id` a scenario needs are written later, when the run actually fires.
366
+ Two traps live in that gap:
367
+
368
+ - **Poll until `workflow_run_log_id` is a string — not until `status`
369
+ leaves `'scheduled'`.** Those are different moments; the run reaches
370
+ `processing` first and writes the log id a beat later. A predicate on
371
+ status alone releases you to read a `null`, and because
372
+ `typeof null === 'object'` the symptom is a baffling *"expected string,
373
+ got object"* rather than an obvious nil.
374
+ - **Raise on `failed` carrying `failure_reason`** rather than polling to
375
+ the deadline. A scenario blocked on a run that will never fire should say
376
+ why on the first read, not thirty seconds later behind a generic timeout.
377
+
378
+ **`waitForBatches`.** Ingestion is async: `ingest` returns a `batch_id` the
379
+ instant the rows are accepted, and the per-row jobs drain into the lake
380
+ afterwards. This polls a DAC's activation logs until every named batch has
381
+ actually written.
382
+
383
+ **Gate on `dataset_updated`, never on `rows_ingested`.** They answer
384
+ different questions — received versus written — and a batch whose row is
385
+ refused reports `rows_ingested: 1, dataset_updated: 0, status: 'partial'`
386
+ with the reason in `error`. A gate on `rows_ingested` calls that green, the
387
+ next step then runs against a table with nothing in it, and the failure
388
+ surfaces three steps later as *"expected 2 execution logs, got 0"* — which
389
+ reads like a broken workflow rather than a row that never landed. Raising
390
+ here, where the reason is still attached, costs one line.
391
+
392
+ **`deployGenericTable`.** Creating a generic table does not deploy it; it
393
+ rests at `status: 'new'` until a migration runs, and `datalakes.migrate`
394
+ enqueues one job per lake so a single call builds the table in all three.
395
+
396
+ The wait is the part worth getting right, because the obvious one is a
397
+ false green. Polling the derived lakes until they report `ready` returns in
398
+ milliseconds and proves nothing — a derived lake is `ready` whether or not
399
+ any particular table exists in it. Waiting on the table's own
400
+ `status: 'deployed'` is no better; that is the raw lake's answer, and it
401
+ goes true before the copies exist. The only wait that means anything asks
402
+ for the table the same way the next step is about to, and retries until it
403
+ stops erroring.
404
+
405
+ ```typescript
406
+ ctx.waitForFiredRun = async (
407
+ runDatalakeSlug: string,
408
+ runId: string,
409
+ timeoutMs = 120_000,
410
+ ): Promise<{ workflowRunLogId: string; batchId: string | null }> => {
411
+ const deadline = Date.now() + timeoutMs
412
+ let lastStatus: string | undefined
413
+ while (Date.now() < deadline) {
414
+ const { data } = await api.workflowRuns.get(tenantSlug, runDatalakeSlug, runId)
415
+ lastStatus = data.status
416
+ if (data.status === 'failed') {
417
+ throw new Error(`workflow run ${runId} failed: ${data.failure_reason ?? 'no failure_reason given'}`)
418
+ }
419
+ if (typeof data.workflow_run_log_id === 'string') {
420
+ return { workflowRunLogId: data.workflow_run_log_id, batchId: data.batch_id ?? null }
421
+ }
422
+ await new Promise((r) => setTimeout(r, 1_000))
423
+ }
424
+ throw new Error(`workflow run ${runId} never fired within ${timeoutMs}ms (last status: ${lastStatus})`)
425
+ }
426
+
427
+ ctx.waitForBatches = async (
428
+ dacSlug: string,
429
+ batchIds: readonly string[],
430
+ timeoutMs = 90_000,
431
+ ): Promise<void> => {
432
+ const targets = new Set(batchIds)
433
+ const deadline = Date.now() + timeoutMs
434
+ let greenCount = 0
435
+ while (Date.now() < deadline) {
436
+ const { data } = await api.dataActivationClients.logs.list(tenantSlug, datalakeSlug, dacSlug)
437
+ const green = new Set<string>()
438
+ for (const row of (data.data ?? []) as Array<Record<string, unknown>>) {
439
+ const b = row.batch_id
440
+ if (typeof b !== 'string' || !targets.has(b)) continue
441
+ if (row.status === 'partial' || row.status === 'failed') {
442
+ throw new Error(
443
+ `batch ${b} on ${dacSlug} did not persist its rows (status: ${row.status}): ` +
444
+ `${String(row.error ?? 'no reason given')}`,
445
+ )
446
+ }
447
+ if (typeof row.dataset_updated !== 'number' || row.dataset_updated < 1) continue
448
+ const files = row.output_files
449
+ if (!Array.isArray(files) || files.length === 0) continue
450
+ green.add(b)
451
+ }
452
+ greenCount = green.size
453
+ if (greenCount === targets.size) return
454
+ await new Promise((r) => setTimeout(r, 1_000))
455
+ }
456
+ throw new Error(`only ${greenCount}/${targets.size} batches on ${dacSlug} persisted within ${timeoutMs}ms`)
457
+ }
458
+
459
+ ctx.deployGenericTable = async (tableName: string, timeoutMs = 120_000): Promise<void> => {
460
+ await api.datalakes.migrate(tenantSlug, datalakeSlug)
461
+ const deadline = Date.now() + timeoutMs
462
+ let lastError: unknown
463
+ while (Date.now() < deadline) {
464
+ try {
465
+ await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
466
+ sql: `SELECT 1 FROM ${tableName} LIMIT 1`,
467
+ mode: 'raw',
468
+ })
469
+ return
470
+ } catch (err) {
471
+ lastError = err
472
+ await new Promise((r) => setTimeout(r, 1_000))
473
+ }
474
+ }
475
+ throw new Error(
476
+ `generic table ${tableName} was not readable within ${timeoutMs}ms ` +
477
+ `(last error: ${lastError instanceof Error ? lastError.message : String(lastError)})`,
478
+ )
479
+ }
480
+ ```
481
+
482
+ ## 009 — the SMS tool
483
+
484
+ Both scenarios send SMS, so the tool is created once and shared.
485
+ `body.tool_body_type: 'sns'` routes via AWS SNS; local dev points it at
486
+ LocalStack on `http://localhost:4566` through `endpoint_url`, so no real AWS
487
+ credentials are needed. `intent: 'sms'` tags the tool for workflow actions
488
+ that send SMS — as opposed to `data_exchange`, which is what an ingestion
489
+ tool carries.
490
+
491
+ ```typescript
492
+ const smsTool = await ctx.ensure(
493
+ 'SMS tool',
494
+ async () => {
495
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
496
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Compliance SMS Tool')
497
+ },
498
+ async () => {
499
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
500
+ name: 'Cookbook Compliance SMS Tool',
501
+ description: 'SNS-backed SMS dispatcher for both compliance workflows, wired to LocalStack.',
502
+ intent: 'sms',
503
+ status: 'active',
504
+ datalake_id: ctx.datalakeId,
505
+ body: {
506
+ tool_body_type: 'sns',
507
+ auth_method: 'access_key',
508
+ region: 'us-east-1',
509
+ phone_number: '+15551234567',
510
+ endpoint_url: 'http://localhost:4566',
511
+ access_key_id: 'test',
512
+ secret_access_key: 'test',
513
+ },
514
+ })
515
+ return data
516
+ },
517
+ )
518
+ toolId = smsTool.id!
519
+ ```
520
+
521
+ ## 010 — the LLM tool
522
+
523
+ The sanctions agent calls a chat-completion endpoint to disambiguate each
524
+ gray-zone screening. `intent: 'llm_enrichment'` distinguishes it from the
525
+ SMS tool.
526
+
527
+ It is a **provider adapter**, and both halves matter. `base_body` authors
528
+ the provider's request — here Ollama's native `/api/chat` shape, with
529
+ `think: false` and a `format` schema so the model returns clean,
530
+ schema-constrained JSON. `response_extractor` maps the provider's envelope
531
+ back to the canonical `{ output_json, … }` the platform reads, and its
532
+ `output_schema` is **required** for an `llm_enrichment` tool. The
533
+ `api_key`/`auth_method` pair satisfies the REST tool's schema even though
534
+ Ollama ignores the header.
535
+
536
+ ```typescript
537
+ const ENRICHMENT_OUTPUT_SCHEMA = {
538
+ type: 'object',
539
+ properties: {
540
+ output_json: {},
541
+ input_tokens: { type: ['integer', 'null'] },
542
+ output_tokens: { type: ['integer', 'null'] },
543
+ total_tokens: { type: ['integer', 'null'] },
544
+ explanation: { type: ['string', 'null'] },
545
+ },
546
+ required: ['output_json'],
547
+ }
548
+
549
+ const OLLAMA_BASE_BODY =
550
+ '{"model": "{{ model }}", "messages": [{"role": "user", "content": "{{ rendered_prompt | json_escape }}", "images": [{% for img in images %}{% unless forloop.first %}, {% endunless %}"{{ img.data }}"{% endfor %}]}], "stream": false, "think": false, "options": {"temperature": {{ temperature }}, "num_predict": {{ max_tokens }}, "num_ctx": 40960}, "format": {{ schema | to_json }}}'
551
+
552
+ const OLLAMA_EXTRACTOR =
553
+ '{"output_json": "{{ msg.message.content | json_escape }}", "explanation": "{{ msg.message.thinking | json_escape }}", "input_tokens": {{ msg.prompt_eval_count | default: 0 }}, "output_tokens": {{ msg.eval_count | default: 0 }}, "total_tokens": {{ msg.prompt_eval_count | default: 0 | plus: msg.eval_count }}}'
554
+
555
+ const llmTool = await ctx.ensure(
556
+ 'LLM tool',
557
+ async () => {
558
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
559
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Compliance LLM Tool')
560
+ },
561
+ async () => {
562
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
563
+ name: 'Cookbook Compliance LLM Tool',
564
+ description: 'Ollama-backed chat-completion adapter for sanctions-screening disambiguation.',
565
+ intent: 'llm_enrichment',
566
+ status: 'active',
567
+ datalake_id: ctx.datalakeId,
568
+ response_extractor: { type: 'custom', body: OLLAMA_EXTRACTOR, output_schema: ENRICHMENT_OUTPUT_SCHEMA },
569
+ body: {
570
+ tool_body_type: 'rest_api',
571
+ base_url: 'http://localhost:11434',
572
+ base_path: { type: 'custom', body: '/api/chat' },
573
+ auth_method: 'api_key',
574
+ api_key: 'stub-key',
575
+ api_key_name: 'Authorization',
576
+ api_key_location: 'header',
577
+ request_type: 'json',
578
+ response_type: 'json',
579
+ timeout_ms: 60_000,
580
+ base_body: { type: 'custom', body: OLLAMA_BASE_BODY },
581
+ },
582
+ })
583
+ return data
584
+ },
585
+ )
586
+ ctx.llmToolId = llmTool.id!
587
+ ```
588
+
589
+ ## 011 — the Atomic FI data source
590
+
591
+ A workflow runs on rows, and rows arrive through the data-activation chain.
592
+ The chain's first link is a `DataSource` — a registration of where the rows
593
+ originate. The `uri` is the system-of-record address, and it flows into each
594
+ ingested row as `source_uri`, which is also what the MDM template in §019
595
+ reads to build its identifier. Both feeds in this walk claim the same
596
+ origin, so they share one source and converge on one subject.
597
+
598
+ ```typescript
599
+ const dataSource = await ctx.ensure(
600
+ 'Atomic FI data source',
601
+ async () => {
602
+ const { data } = await api.dataSources.list(tenantSlug, datalakeSlug)
603
+ return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Atomic FI Source')
604
+ },
605
+ async () => {
606
+ const { data } = await api.dataSources.create(tenantSlug, datalakeSlug, {
607
+ name: 'Cookbook Atomic FI Source',
608
+ uri: 'api.atomic.fi/compliance',
609
+ description: 'Atomic FI compliance API — origin of the screening and payment-account rows.',
610
+ status: 'active',
611
+ is_default: false,
612
+ })
613
+ return data
614
+ },
615
+ )
616
+ dataSourceId = dataSource.id!
617
+ ```
618
+
619
+ ## 012 — the Manual Upload tool
620
+
621
+ A Data Activation Client needs a tool. For inline-JSON ingest a
622
+ `manual_upload` tool is the minimal choice: `tool_body_type:
623
+ 'manual_upload'` needs no endpoint or credential wiring, because the rows
624
+ arrive in the ingest call's body rather than by the tool fetching them.
625
+
626
+ ```typescript
627
+ const manualUploadTool = await ctx.ensure(
628
+ 'Manual Upload tool',
629
+ async () => {
630
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
631
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Manual Upload Tool')
632
+ },
633
+ async () => {
634
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
635
+ name: 'Cookbook Manual Upload Tool',
636
+ description: 'Manual-upload data-exchange tool — backs both DACs in this walk.',
637
+ intent: 'data_exchange',
638
+ status: 'active',
639
+ datalake_id: ctx.datalakeId,
640
+ data_source_id: dataSourceId,
641
+ body: { tool_body_type: 'manual_upload' },
642
+ })
643
+ return data
644
+ },
645
+ )
646
+ ctx.manualUploadToolId = manualUploadTool.id!
647
+ ```
648
+
649
+ ## 013 — what the platform already ships: the system-template catalog
650
+
651
+ Before hand-writing Liquid, look at what is already there. `systemTemplates`
652
+ returns every platform-shipped template visible to this datalake — each with
653
+ a `path`, its Liquid `content`, and an `output_schema` (`null` when the
654
+ template ships no companion schema).
655
+
656
+ **The catalog is flat.** It used to be filtered by the datalake's data
657
+ domain, so a payments template never appeared in a subscription lake;
658
+ GH-859 removed domains and with them the filter. What one datalake can see,
659
+ every datalake can see. The assertion below is the positive form of that
660
+ fact: no path carries a retired vertical segment.
661
+
662
+ ```typescript
663
+ const { data: templates } = await api.templates.systemTemplates(tenantSlug, datalakeSlug)
664
+
665
+ const paths = (templates.data ?? []).map((t) => t.path)
666
+ if (paths.length === 0) {
667
+ throw new Error('expected the platform to ship at least one system template')
668
+ }
669
+
670
+ const anchor = 'data_activation/interoperability/stripe/customers_stripe_legal_entity'
671
+ if (!paths.some((p) => p.includes(anchor))) {
672
+ throw new Error(`expected the Stripe customer template (${anchor}) in the list`)
673
+ }
674
+
675
+ const RETIRED_DOMAINS = ['healthcare', 'payments', 'foundation', 'core_banking', 'service_commerce', 'trading']
676
+ const stale = paths.filter((p) => RETIRED_DOMAINS.some((d) => p.split('/').includes(d)))
677
+ if (stale.length > 0) {
678
+ throw new Error(`template paths still carry a retired data-domain segment: ${stale.join(', ')}`)
679
+ }
680
+ ```
681
+
682
+ ## 014 — the same catalog as markdown, and the page you will forget to read
683
+
684
+ `metadata` returns the catalog as a markdown document — the form an agent
685
+ reads to choose a template. It inlines each template's full Liquid source,
686
+ so it is long.
687
+
688
+ **It paginates, and page 1 is not the catalog.** The document opens with
689
+ `Showing page 1 of N — M templates total`, and the template you want is as
690
+ likely to be on page 3. Reading page 1 and concluding a template does not
691
+ exist is the trap, and it used to be hidden: the per-domain filter made the
692
+ list short enough to fit one page, and GH-859 removed that filter.
693
+
694
+ ```typescript
695
+ let found = false
696
+ let page = 1
697
+ let pages = 1
698
+ do {
699
+ const { data: catalog } = await api.templates.metadata(tenantSlug, datalakeSlug, { page })
700
+ if (typeof catalog !== 'string' || catalog.length === 0) {
701
+ throw new Error(`expected a non-empty markdown template catalog on page ${page}`)
702
+ }
703
+ const header = catalog.match(/Showing page \d+ of (\d+)/)
704
+ if (header) pages = Number(header[1])
705
+ if (catalog.includes('customers_stripe_legal_entity')) found = true
706
+ page += 1
707
+ } while (!found && page <= pages)
708
+
709
+ if (!found) {
710
+ throw new Error(`expected the Stripe customer template in the catalog markdown (searched ${pages} page(s))`)
711
+ }
712
+ ```
713
+
714
+ ## 015 — one template's detail, by basename and intent
715
+
716
+ `metadataDetails` pulls a single template. Address it by its **basename**
717
+ (the template name without the path) and its **intent** (the path prefix
718
+ joined with underscores — here `data_activation_interoperability`), not by
719
+ the full path. The result is a markdown body to show an operator, or to feed
720
+ an agent, before copying the template into a contract's `template_config`.
721
+
722
+ ```typescript
723
+ const { data: detail } = await api.templates.metadataDetails(
724
+ tenantSlug,
725
+ datalakeSlug,
726
+ 'customers_stripe_legal_entity',
727
+ 'data_activation_interoperability',
728
+ )
729
+ if (typeof detail !== 'string' || detail.length === 0) {
730
+ throw new Error('expected a non-empty markdown detail for the Stripe customer template')
731
+ }
732
+ ```
733
+ ## 016 — the Sanctions Review agent
734
+
735
+ The agent binds three things: the model, an input schema the workflow's
736
+ context-mapping must satisfy, and a response schema its output must match.
737
+ The response schema's `enum: ['verified','blocked']` is a guard — it stops
738
+ the agent emitting a verdict the workflow has no action for.
739
+ `temperature: 0.0` removes sampling noise so the walk is deterministic.
740
+
741
+ `data_access: 'raw'` is the interesting decision, and it is deliberately
742
+ the opposite of what §007 might lead you to expect. `screened_entity_name`
743
+ is a `tokenize` column, and this agent's whole job is to judge whether a
744
+ name is a true match or a coincidence — a token tells it nothing. So this
745
+ agent reads raw. §037 shows the reader for whom the tokenized copy is the
746
+ right answer. **Tokenization is a per-reader decision, not a lake-wide
747
+ setting**, and an agent that does not need plaintext should not ask for it.
748
+
749
+ ```typescript
750
+ const AGENT_INPUT_SCHEMA = {
751
+ type: 'object',
752
+ properties: {
753
+ screening_id: { type: 'string' },
754
+ screened_entity_name: { type: 'string' },
755
+ match_score: { type: 'number' },
756
+ },
757
+ required: ['screening_id', 'screened_entity_name', 'match_score'],
758
+ }
759
+
760
+ const AGENT_RESPONSE_SCHEMA = {
761
+ type: 'object',
762
+ properties: {
763
+ verdict: { type: 'string', enum: ['verified', 'blocked'] },
764
+ comments: { type: 'string' },
765
+ },
766
+ required: ['verdict', 'comments'],
767
+ }
768
+
769
+ const AGENT_PROMPT_BODY = `You are a sanctions-review disambiguation assistant. A name-match against a sanctions list has come in with a gray-zone score (close enough to be plausible, ambiguous enough to need a human-style judgement call). Decide whether this is a true match ("blocked") or a false positive on a similar name ("verified").
770
+
771
+ Screening id: {{ screening_id }}
772
+ Screened entity name: {{ screened_entity_name }}
773
+ Match score: {{ match_score }}
774
+
775
+ Respond with JSON: {"verdict": "verified" | "blocked", "comments": "<one-sentence rationale>"}
776
+
777
+ For this cookbook fixture, the entity name is a deliberately disambiguating non-sanctioned identity — respond with "verified".`
778
+
779
+ const agent = await ctx.ensure(
780
+ 'Sanctions Review agent',
781
+ async () => {
782
+ const { data } = await api.aiAgents.list(tenantSlug, datalakeSlug)
783
+ return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Sanctions Review Agent')
784
+ },
785
+ async () => {
786
+ const { data } = await api.aiAgents.create(tenantSlug, datalakeSlug, {
787
+ name: 'Cookbook Sanctions Review Agent',
788
+ tool_id: ctx.llmToolId,
789
+ model: 'qwen3-vl:8b-instruct',
790
+ data_access: 'raw',
791
+ temperature: 0.0,
792
+ max_tokens: 256,
793
+ enabled: true,
794
+ input_schema: AGENT_INPUT_SCHEMA,
795
+ llm_response_schema: AGENT_RESPONSE_SCHEMA,
796
+ prompt_config: { type: 'custom', body: AGENT_PROMPT_BODY },
797
+ })
798
+ return data
799
+ },
800
+ )
801
+ aiAgentId = agent.id!
802
+ ctx.agentSlug = agent.slug!
803
+ ```
804
+
805
+ ## 017 — the compliance-screenings table
806
+
807
+ The workflow needs an event source. GH-859 removed the
808
+ `compliance_screening` dataset along with every other per-domain resource,
809
+ so a screening is a table you declare.
810
+
811
+ Four column decisions carry this walk.
812
+
813
+ `screening_score` is a **float**, not a string. The gray-zone filter
814
+ compares it with `>=` and `<`; a string column compares lexically and the
815
+ band silently stops meaning anything.
816
+
817
+ `screened_entity_name` is `tokenize` and `review_notes` is `redact`. That
818
+ declaration **is** the masking policy — there is no changeset doing it on
819
+ the way past any more, and no second schema to write into. `privacy_requirement`
820
+ is a floor, not an instruction: a `tokenize` column is tokenized in the
821
+ tokenized copy and redacted in the redacted one.
822
+
823
+ `sanctions_matches` is `jsonb` **and `is_array: true`**. The second half is
824
+ not optional and its absence does not look like a type error. `is_array`
825
+ defaults to `false`, which compiles the column's schema as a single map, so
826
+ ingesting a JSON array into it is refused per row —
827
+ `upsert_row_failed %{sanctions_matches: ["is invalid"]}` — while the batch
828
+ still reports a row received.
829
+
830
+ Do **not** declare a `legal_entity_id` column. The platform stamps it from
831
+ Contract B's MDM resolve in §019, and the name is reserved — declaring it
832
+ does not shadow the stamp silently, it 422s the create.
833
+
834
+ ```typescript
835
+ const screenings = await ctx.ensure(
836
+ 'compliance-screenings table',
837
+ async () => {
838
+ const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
839
+ return (data.data ?? []).find((t: { title?: string }) => t.title === 'Compliance Screenings')
840
+ },
841
+ async () => {
842
+ const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
843
+ title: 'Compliance Screenings',
844
+ description: 'Sanctions/AML screenings loaded from the Atomic FI API',
845
+ columns: [
846
+ { name: 'compliance_screening_number', title: 'Screening Number', type: 'string', description: "The provider's screening id — unique per screening", is_unique: true, privacy_requirement: 'none' },
847
+ { name: 'account_holder_ref', title: 'Account Holder Ref', type: 'string', description: "The provider's account-holder id — the identifier MDM resolves on", is_unique: false, privacy_requirement: 'none' },
848
+ { name: 'scope', title: 'Scope', type: 'string', description: 'What the screening was run against', is_unique: false, privacy_requirement: 'none' },
849
+ { name: 'screening_type', title: 'Screening Type', type: 'string', description: 'sanctions / pep / adverse_media', is_unique: false, privacy_requirement: 'none' },
850
+ { name: 'screening_status', title: 'Screening Status', type: 'string', description: 'pending / pass / fail', is_unique: false, privacy_requirement: 'none' },
851
+ { name: 'screened_entity_type', title: 'Screened Entity Type', type: 'string', description: 'individual / business', is_unique: false, privacy_requirement: 'none' },
852
+ { name: 'screening_score', title: 'Screening Score', type: 'float', description: 'Match confidence 0-1 — the gray-zone band compares on it numerically', is_unique: false, privacy_requirement: 'none' },
853
+ { name: 'match_count', title: 'Match Count', type: 'integer', description: 'How many list entries matched', is_unique: false, privacy_requirement: 'none' },
854
+ { name: 'sanctions_screening_status', title: 'Sanctions Status', type: 'string', description: 'match / cleared', is_unique: false, privacy_requirement: 'none' },
855
+ { name: 'screened_entity_name', title: 'Screened Entity Name', type: 'string', description: 'Name that was screened — the agent disambiguates on it', is_unique: false, privacy_requirement: 'tokenize' },
856
+ { name: 'review_notes', title: 'Review Notes', type: 'string', description: 'Free-text reviewer narrative', is_unique: false, privacy_requirement: 'redact' },
857
+ { name: 'pep_list_name', title: 'PEP List Name', type: 'string', description: 'Politically-exposed-person list the entity appears on', is_unique: false, privacy_requirement: 'tokenize' },
858
+ { name: 'aml_risk_score', title: 'AML Risk Score', type: 'float', description: 'Analytic AML risk score', is_unique: false, privacy_requirement: 'none' },
859
+ { name: 'aml_velocity_count', title: 'AML Velocity Count', type: 'integer', description: 'Transaction-velocity signal', is_unique: false, privacy_requirement: 'none' },
860
+ { name: 'aml_high_risk_country', title: 'AML High Risk Country', type: 'string', description: 'High-risk jurisdiction flag', is_unique: false, privacy_requirement: 'none' },
861
+ { name: 'sanctions_matches', title: 'Sanctions Matches', type: 'jsonb', description: 'The matched list entries, as a JSON array', is_unique: false, is_array: true, privacy_requirement: 'redact' },
862
+ ],
863
+ })
864
+ return data
865
+ },
866
+ )
867
+ ctx.screeningsTableId = screenings.id!
868
+ ctx.screeningsTableName = screenings.name!
869
+
870
+ // Creating a table records it; deploying it is a migration. Called
871
+ // unconditionally rather than only on the create branch, so a run that
872
+ // crashed between the two still converges.
873
+ await ctx.deployGenericTable(ctx.screeningsTableName)
874
+ ```
875
+
876
+ ## 018 — the Sanctions Review workflow
877
+
878
+ Filter, decision, actions — plus the agent **nested** in the create body
879
+ via `workflow_ai_agents` (a `cast_assoc`, not a separate attach call).
880
+ Three pieces are worth stopping on.
881
+
882
+ The **filter** is the gray-zone band: it passes only screenings scoring at
883
+ least `0.80` and below `0.95`. Below `0.80` is an auto-clear and at or
884
+ above `0.95` is an auto-block — neither needs a model, so the filter
885
+ rejects them before any LLM call is paid for.
886
+
887
+ The **context_mapping** projects each screening into the agent's input
888
+ schema. `match_score` is interpolated *without* surrounding quotes so it
889
+ renders as a raw JSON number; the input schema declares it `number`, and a
890
+ quoted `"0.88"` fails validation.
891
+
892
+ The **decision_config** interpolates the agent's `verdict` — read from
893
+ `additional_context["<agent-slug>"]` with bracket access, because the slug
894
+ contains hyphens — into a single-element decision array. The two actions
895
+ are keyed `decision_key: verified|blocked`; whichever verdict the agent
896
+ emits picks the action that fires and leaves the other `:skipped`.
897
+
898
+ ```typescript
899
+ const VERDICTS = ['verified', 'blocked'] as const
900
+
901
+ const GRAY_ZONE_FILTER =
902
+ '{% if event_dataset.screening_score >= 0.80 ' +
903
+ 'and event_dataset.screening_score < 0.95 %}true{% endif %}'
904
+
905
+ const CONTEXT_MAPPING_BODY =
906
+ '{' +
907
+ '"screening_id": "{{ event_dataset.compliance_screening_number }}",' +
908
+ '"screened_entity_name": "{{ event_dataset.screened_entity_name }}",' +
909
+ '"match_score": {{ event_dataset.screening_score }}' +
910
+ '}'
911
+
912
+ const DECISION_CONFIG_BODY = `["{{ additional_context["${ctx.agentSlug}"].verdict }}"]`
913
+ const DECISION_OUTPUT_SCHEMA = { type: 'array', items: { type: 'string' } }
914
+
915
+ const workflow = await ctx.ensure(
916
+ 'Sanctions Review workflow',
917
+ async () => {
918
+ const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
919
+ return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Sanctions Review Workflow')
920
+ },
921
+ async () => {
922
+ const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
923
+ name: 'Cookbook Sanctions Review Workflow',
924
+ description: 'Runs an LLM disambiguation pass on gray-zone sanctions matches and fires a verdict-keyed SMS.',
925
+ dataset_type: 'generic_table',
926
+ generic_table_id: ctx.screeningsTableId,
927
+ status: 'live',
928
+ tags: ['compliance', 'sanctions'],
929
+ skip_mdm_resolution: false,
930
+ filter_config: { type: 'custom', body: GRAY_ZONE_FILTER },
931
+ decision_config: {
932
+ type: 'custom',
933
+ body: DECISION_CONFIG_BODY,
934
+ output_schema: DECISION_OUTPUT_SCHEMA,
935
+ },
936
+ actions: VERDICTS.map((verdict) => ({
937
+ decision_key: verdict,
938
+ action_type: 'sms',
939
+ tool_id: toolId,
940
+ position: 0,
941
+ trigger_template: 'now',
942
+ idempotency_template: `{{ event_dataset.compliance_screening_number }}-${verdict}`,
943
+ tool_call: {
944
+ tool_call_type: 'sms_request',
945
+ to: { type: 'custom', body: '+15550000000' },
946
+ body: {
947
+ type: 'custom',
948
+ body: `[${verdict}] {{ additional_context["${ctx.agentSlug}"].comments }}`,
949
+ },
950
+ sms_type: 'transactional',
951
+ },
952
+ })),
953
+ // Workflow joins omit output_schema — the server pins it from the
954
+ // agent's own input_schema.
955
+ workflow_ai_agents: [
956
+ {
957
+ ai_agent_id: aiAgentId,
958
+ position: 0,
959
+ context_mapping_config: { type: 'custom', body: CONTEXT_MAPPING_BODY },
960
+ },
961
+ ],
962
+ })
963
+ return data
964
+ },
965
+ )
966
+ workflowId = workflow.id!
967
+ ctx.workflowSlug = workflow.slug!
968
+ ```
969
+
970
+ ## 019 — the Atomic FI contract pair
971
+
972
+ An inbound screening row becomes two things, so it takes two contracts.
973
+ They are defined together here and deliberately **not** bound to one client
974
+ — §020 is where that matters, and why.
975
+
976
+ **Contract A** (`resource_type: 'legal_entity'`) writes the account holder
977
+ the screening was run against, keyed on Atomic FI's `account_holder_id`.
978
+ Its `mdm_input_config` is `{ type: 'null' }` — a template that already *is*
979
+ the subject has nothing to resolve.
980
+
981
+ **Contract B** (`resource_type: 'generic_table'`, pinned to §017's table)
982
+ writes the screening row, and its `mdm_input_config` emits the same
983
+ `(uri, account_holder_id)` pair — collapsing both onto one legal entity and
984
+ stamping `legal_entity_id` onto the screening.
985
+
986
+ **The identifier shape is where this goes wrong, and there are three of
987
+ them on this platform.** `Platform.MDMInput` — what a contract's
988
+ `mdm_input_config` renders — takes `uri` / `type` / `value`. A LegalEntity
989
+ identification, which is Contract A's own body, takes `id_type` /
990
+ `id_number` / `uri`. And `POST /mdm/verify` takes `system` / `id_type` /
991
+ `value`.
992
+
993
+ The overlap between them is the trap, and it is not symmetric. `system` is a
994
+ real alias: `MDMInput` casts it and normalises it onto `uri`, so that one
995
+ name is safe in both places. `id_type` and `id_number` are not — an MDM
996
+ identifier casts neither. Unknown keys are dropped rather than refused, so a
997
+ template that reaches for the LegalEntity names on the MDM side resolves
998
+ nothing and fails as a bare `mdm_dispatcher_error` with no field named.
999
+
1000
+ And the reason the two look interchangeable is that **one becomes the
1001
+ other**. `MDMInput.to_identification/1` takes `(uri, type, value)` and
1002
+ returns `(uri, id_type, id_number)` — the rename happens inside the platform,
1003
+ on the way through. So the identifier reads back under names it will not
1004
+ accept on the way in, and that asymmetry is invisible from the stored shape
1005
+ alone. Do not reach for `/mdm/verify` to check the shape
1006
+ either: it requires a `legal_entity_id` and verifies attributes against an
1007
+ entity that already exists — it is not a resolver and not a linter.
1008
+
1009
+ The templates are loaded from the vendored fixtures rather than inlined.
1010
+
1011
+ ```typescript
1012
+ const { readFileSync } = await import('node:fs')
1013
+ const { join } = await import('node:path')
1014
+ const read = (f: string) =>
1015
+ readFileSync(join(process.env.COOKBOOK_FIXTURES_DIR!, 'payments-compliance', f), 'utf8')
1016
+
1017
+ const leContract = await ctx.ensure(
1018
+ 'screening holder LE contract',
1019
+ async () => {
1020
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1021
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Screening Holder LE Contract')
1022
+ },
1023
+ async () => {
1024
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1025
+ name: 'Cookbook Screening Holder LE Contract',
1026
+ description: 'Atomic FI screening → the screened account holder as a LegalEntity.',
1027
+ resource_type: 'legal_entity',
1028
+ type: 'identity',
1029
+ generic_table_id: null,
1030
+ template_config: { type: 'custom', body: read('_compliance_screenings_legal_entity.liquid') },
1031
+ mdm_input_config: { type: 'null' },
1032
+ })
1033
+ return data
1034
+ },
1035
+ )
1036
+ ctx.leContractId = leContract.id!
1037
+
1038
+ const gtContract = await ctx.ensure(
1039
+ 'compliance screening GT contract',
1040
+ async () => {
1041
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1042
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Compliance Screening GT Contract')
1043
+ },
1044
+ async () => {
1045
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1046
+ name: 'Cookbook Compliance Screening GT Contract',
1047
+ description: 'Atomic FI screening → compliance_screenings row, stamped with its holder subject.',
1048
+ resource_type: 'generic_table',
1049
+ type: 'identity',
1050
+ generic_table_id: ctx.screeningsTableId,
1051
+ template_config: { type: 'custom', body: read('_compliance_screenings_generic_table.liquid') },
1052
+ mdm_input_config: { type: 'custom', body: read('_compliance_screenings_mdm.liquid') },
1053
+ })
1054
+ return data
1055
+ },
1056
+ )
1057
+ interopContractId = gtContract.id!
1058
+ ```
1059
+
1060
+ ## 020 — two DACs, because the pair must not run concurrently
1061
+
1062
+ The obvious move is one Data Activation Client carrying both contracts.
1063
+ **It works on every run but the first, which is the worst way for something
1064
+ to be wrong.**
1065
+
1066
+ A DAC fans out per `(row, contract)` at enqueue time — one job each — and
1067
+ the job key is `"<r2_key>-<index>-<client_id>-<contract_id>"`. The contract
1068
+ id is *in* the key, so the pair's two jobs sit in different chains and run
1069
+ **at the same instant**: the scheduler staggers by row index, not by
1070
+ contract. Contract A writes the holder as a legal entity. Contract B's
1071
+ `mdm_input_config` resolves the *same* holder, and resolution is
1072
+ find-or-create, so it writes one too. Both miss the dedupe read, both
1073
+ insert, and `legal_entity_identifications_uri_type_value_uk` —
1074
+ `(uri, id_type, id_number)`, with the legal entity deliberately left out of
1075
+ the key — refuses the loser:
1076
+
1077
+ ```
1078
+ upsert_legal_entity_failed %{identifications: [%{uri: ["has already been taken"]}]}
1079
+ ```
1080
+
1081
+ The batch reports `partial`, the row never lands, and the failure surfaces
1082
+ several steps later as a missing execution log.
1083
+
1084
+ **That index is not a bug you are hitting; it is the guard working.** It
1085
+ omits the legal entity from the key precisely so that concurrent workers
1086
+ racing to create one subject cannot each succeed and leave duplicates that
1087
+ are indistinguishable from two people who genuinely share a handle. One
1088
+ writer per subject is the contract, and the second run only looks green
1089
+ because the first run's Contract A already created the entity — so nothing
1090
+ inserts and nothing collides. Ingest a holder the platform has never seen
1091
+ and it fails again.
1092
+
1093
+ So the pair is split across two clients over the same tool and data source:
1094
+ an **identity** DAC that writes the subject, and a **data** DAC that writes
1095
+ the row and resolves the subject that now exists. §021 ingests through them
1096
+ in that order, waiting for the first to persist before starting the second.
1097
+ That is not a workaround for the index — it is the honest shape of the
1098
+ dependency, and it reads as one: an identity feed, then a data feed.
1099
+
1100
+ ```typescript
1101
+ const identityDac = await ctx.ensure(
1102
+ 'compliance screening identity DAC',
1103
+ async () => {
1104
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
1105
+ return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Compliance Screening Identity DAC')
1106
+ },
1107
+ async () => {
1108
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1109
+ name: 'Cookbook Compliance Screening Identity DAC',
1110
+ description: 'Manual-upload DAC — writes the screened holder as a legal entity, ahead of the screening row.',
1111
+ tool_id: ctx.manualUploadToolId,
1112
+ data_source_id: dataSourceId,
1113
+ tool_call: { tool_call_type: 'manual_upload' },
1114
+ interoperability_contract_ids: [ctx.leContractId],
1115
+ })
1116
+ return data
1117
+ },
1118
+ )
1119
+ ctx.identityDacSlug = identityDac.slug!
1120
+
1121
+ const dac = await ctx.ensure(
1122
+ 'compliance screening data DAC',
1123
+ async () => {
1124
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
1125
+ return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Compliance Screening Data DAC')
1126
+ },
1127
+ async () => {
1128
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1129
+ name: 'Cookbook Compliance Screening Data DAC',
1130
+ description: 'Manual-upload DAC — ingests compliance-screening rows, resolving the holder written by the identity DAC.',
1131
+ tool_id: ctx.manualUploadToolId,
1132
+ data_source_id: dataSourceId,
1133
+ tool_call: { tool_call_type: 'manual_upload' },
1134
+ interoperability_contract_ids: [interopContractId],
1135
+ })
1136
+ return data
1137
+ },
1138
+ )
1139
+ dacId = dac.id!
1140
+ ctx.dacSlug = dac.slug!
1141
+ ```
1142
+
1143
+ ## 021 — ingest two screenings, holders first
1144
+
1145
+ Two rows, inline JSON, differing in the one field the filter reads. The
1146
+ first scores `0.88` — squarely inside `[0.80, 0.95)`, so the filter passes
1147
+ it to the agent. The second scores `0.20`, an auto-clear the filter rejects
1148
+ before any agent call.
1149
+
1150
+ **The same two rows are ingested twice: through the identity DAC, then
1151
+ through the data DAC.** That is the ordering §020 exists for — the holder
1152
+ has to be committed before a screening row can resolve it. The first stage
1153
+ is waited on before the second starts, and that wait is the whole point;
1154
+ firing both stages together reproduces the race the split was made to
1155
+ avoid.
1156
+
1157
+ Each row carries its own `account_holder_id`, so the two resolve to
1158
+ different legal entities and never contend with *each other* — within a
1159
+ stage they are independent, and the two `ingest` calls are separate only so
1160
+ each row gets its own activation-log row for the gate to key on.
1161
+
1162
+ The screening numbers are **stable across runs**, and
1163
+ `compliance_screening_number` is the table's unique column — so a second
1164
+ run upserts these same two rows rather than adding two more. `source_uri`
1165
+ matches the data source's `uri` from §011, which is what both contracts key
1166
+ their identifier on.
1167
+
1168
+ ```typescript
1169
+ ctx.grayZoneNumber = 'CS-COOKBOOK-0101'
1170
+ ctx.autoClearNumber = 'CS-COOKBOOK-0102'
1171
+ // §037 reads this name back through all three lakes.
1172
+ ctx.grayZoneName = 'Jane Cookbook-Disambiguating-Doe'
1173
+
1174
+ const grayZoneRow = {
1175
+ compliance_screening_number: ctx.grayZoneNumber,
1176
+ account_holder_id: 'AH-COOKBOOK-0101',
1177
+ scope: 'account_holder',
1178
+ screening_type: 'sanctions',
1179
+ screening_status: 'pending',
1180
+ screened_entity_type: 'individual',
1181
+ sanctions_screening_status: 'match',
1182
+ screening_score: 0.88,
1183
+ match_count: 1,
1184
+ screened_entity_name: ctx.grayZoneName,
1185
+ source_uri: 'api.atomic.fi/compliance',
1186
+ }
1187
+ const autoClearRow = {
1188
+ compliance_screening_number: ctx.autoClearNumber,
1189
+ account_holder_id: 'AH-COOKBOOK-0102',
1190
+ scope: 'account_holder',
1191
+ screening_type: 'sanctions',
1192
+ screening_status: 'pass',
1193
+ screened_entity_type: 'individual',
1194
+ sanctions_screening_status: 'cleared',
1195
+ screening_score: 0.2,
1196
+ match_count: 0,
1197
+ screened_entity_name: 'John Cookbook-Clear-Smith',
1198
+ source_uri: 'api.atomic.fi/compliance',
1199
+ }
1200
+
1201
+ // STAGE 1 — the holders. Contract A writes each screened entity as a legal
1202
+ // entity. Nothing resolves anything here, so nothing can collide.
1203
+ const holderGray = await api.dataActivationClients.ingest(
1204
+ tenantSlug, datalakeSlug, ctx.identityDacSlug, { data: grayZoneRow },
1205
+ )
1206
+ const holderClear = await api.dataActivationClients.ingest(
1207
+ tenantSlug, datalakeSlug, ctx.identityDacSlug, { data: autoClearRow },
1208
+ )
1209
+
1210
+ // The wait that makes the split work. Skip it and stage 2 resolves a
1211
+ // holder that is still being written, which is the race in a slower shape.
1212
+ await ctx.waitForBatches(ctx.identityDacSlug, [
1213
+ holderGray.data.batch_id!,
1214
+ holderClear.data.batch_id!,
1215
+ ])
1216
+
1217
+ // STAGE 2 — the screenings. Contract B resolves the holder that now
1218
+ // exists, and the platform stamps `legal_entity_id` onto the row.
1219
+ const grayZoneIngest = await api.dataActivationClients.ingest(
1220
+ tenantSlug, datalakeSlug, ctx.dacSlug, { data: grayZoneRow },
1221
+ )
1222
+ const autoClearIngest = await api.dataActivationClients.ingest(
1223
+ tenantSlug, datalakeSlug, ctx.dacSlug, { data: autoClearRow },
1224
+ )
1225
+ ctx.batchGrayZone = grayZoneIngest.data.batch_id!
1226
+ ctx.batchAutoClear = autoClearIngest.data.batch_id!
1227
+ ```
1228
+
1229
+ ## 022 — wait for the screening rows to PERSIST, not merely to arrive
1230
+
1231
+ The same `waitForBatches` from §008, now on the data DAC. Stage 1 was
1232
+ already waited on inside §021; this is the gate on stage 2, and it is what
1233
+ stands between a refused row and a workflow that runs against an empty
1234
+ table.
1235
+
1236
+ ```typescript
1237
+ await ctx.waitForBatches(ctx.dacSlug, [ctx.batchGrayZone, ctx.batchAutoClear])
1238
+ ```
1239
+
1240
+ ## 023 — run the workflow against the two rows
1241
+
1242
+ `workflows.run` schedules; it does not fire. The helper from §008 waits for
1243
+ the run to actually fire and hands back the log id and batch id the
1244
+ verification needs.
1245
+
1246
+ ```typescript
1247
+ const runResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.workflowSlug, {
1248
+ sql_where_clause: `compliance_screening_number IN ('${ctx.grayZoneNumber}', '${ctx.autoClearNumber}')`,
1249
+ mode: 'live',
1250
+ manual_override: false,
1251
+ })
1252
+ const fired = await ctx.waitForFiredRun(datalakeSlug, runResp.data.workflow_run_id)
1253
+ ctx.runLogId = fired.workflowRunLogId
1254
+ ctx.runBatchId = fired.batchId!
1255
+
1256
+ const deadline = Date.now() + 240_000
1257
+ let status: string | null = null
1258
+ while (Date.now() < deadline) {
1259
+ const { data: log } = await api.workflows.batchLogs.refresh(tenantSlug, datalakeSlug, ctx.workflowSlug, ctx.runLogId)
1260
+ status = log.status ?? null
1261
+ if (status && status !== 'pending') break
1262
+ await new Promise((r) => setTimeout(r, 2_000))
1263
+ }
1264
+ if (status === 'failed') throw new Error('agent-driven workflow run reached :failed')
1265
+ if (!status || status === 'pending') {
1266
+ throw new Error('workflow run did not leave :pending within 240s')
1267
+ }
1268
+ ```
1269
+
1270
+ ## 024 — verify the band filtered, and the agent steered the fan-out
1271
+
1272
+ Each row produced a Workflow Execution Log. The gray-zone screening passes
1273
+ the filter, reaches the agent, and its WEL is `:executing` or `:completed`.
1274
+ The auto-clear screening fails the filter, so its WEL is `:filtered` — it
1275
+ never reached the agent. A run where both passed, or both were filtered,
1276
+ would mean the band is not actually evaluating the score.
1277
+
1278
+ The pass-branch WEL carries exactly **two** action execution logs, one per
1279
+ verdict. On the first run exactly one is matched and the other `:skipped`,
1280
+ which is the proof the verdict steered the decision rather than every
1281
+ action firing.
1282
+
1283
+ **On a second run both are `:skipped`, and that is correct.** This is where
1284
+ idempotence stops being a property of the walk and becomes a property of
1285
+ the product. `idempotency_template` renders
1286
+ `{{ event_dataset.compliance_screening_number }}-<verdict>`, the screening
1287
+ number is stable, so the key is the key that already fired — and the
1288
+ platform refuses to send the same SMS about the same screening twice. That
1289
+ is exactly what you want in production, and it means **an assertion on "an
1290
+ action fired just now" is wrong**: it tests a transition that only ever
1291
+ happens once. Assert the end state instead — the verdict SMS exists for
1292
+ this screening — which is what §025 does, and which is true on every run.
1293
+
1294
+ ```typescript
1295
+ const welDeadline = Date.now() + 120_000
1296
+ let ourWels: Array<Record<string, unknown>> = []
1297
+ let byStatus: Record<string, number> = {}
1298
+ while (Date.now() < welDeadline) {
1299
+ const { data: wfLogs } = await api.workflows.workflowLogs.list(tenantSlug, datalakeSlug, ctx.workflowSlug)
1300
+ ourWels = (wfLogs.data ?? [])
1301
+ .map((w) => w as Record<string, unknown>)
1302
+ .filter((w) => w.batch_id === ctx.runBatchId)
1303
+ byStatus = {}
1304
+ for (const w of ourWels) {
1305
+ const st = (w.status as string | undefined) ?? 'unknown'
1306
+ byStatus[st] = (byStatus[st] ?? 0) + 1
1307
+ }
1308
+ const settled = (byStatus.completed ?? 0) + (byStatus.executing ?? 0)
1309
+ if (ourWels.length === 2 && settled === 1 && (byStatus.filtered ?? 0) === 1) break
1310
+ await new Promise((r) => setTimeout(r, 2_000))
1311
+ }
1312
+ if (ourWels.length !== 2) {
1313
+ throw new Error(`expected 2 WELs for the run, got ${ourWels.length}`)
1314
+ }
1315
+ if ((byStatus.filtered ?? 0) !== 1) {
1316
+ throw new Error(`expected 1 :filtered WEL (the auto-clear) — distribution ${JSON.stringify(byStatus)}`)
1317
+ }
1318
+ const passWel = ourWels.find((w) => {
1319
+ const st = (w as { status?: string }).status
1320
+ return st === 'executing' || st === 'completed'
1321
+ })
1322
+ if (!passWel) {
1323
+ throw new Error(`no pass-branch WEL — distribution ${JSON.stringify(byStatus)}`)
1324
+ }
1325
+
1326
+ const aels =
1327
+ (passWel as { action_execution_logs?: Array<{ status?: string; decision_key?: string }> })
1328
+ .action_execution_logs ?? []
1329
+ if (aels.length !== 2) {
1330
+ throw new Error(`expected 2 AELs (one per verdict) on the gray-zone WEL, got ${aels.length}`)
1331
+ }
1332
+ const aelByStatus: Record<string, number> = {}
1333
+ for (const ael of aels) {
1334
+ const st = ael.status ?? 'unknown'
1335
+ aelByStatus[st] = (aelByStatus[st] ?? 0) + 1
1336
+ }
1337
+ const matched = (aelByStatus.pending ?? 0) + (aelByStatus.completed ?? 0)
1338
+ if (matched === 1) {
1339
+ // First run for this screening: the verdict picked one action and left
1340
+ // the other alone.
1341
+ if ((aelByStatus.skipped ?? 0) !== 1) {
1342
+ throw new Error(`expected 1 :skipped AEL alongside the matched one — ${JSON.stringify(aelByStatus)}`)
1343
+ }
1344
+ const routed = aels.find((a) => a.status === 'pending' || a.status === 'completed')
1345
+ if (!routed?.decision_key || !/^(verified|blocked)$/.test(routed.decision_key)) {
1346
+ throw new Error(`matched AEL has unexpected decision_key ${routed?.decision_key}`)
1347
+ }
1348
+ } else if (matched === 0 && (aelByStatus.skipped ?? 0) === aels.length) {
1349
+ // Re-run: this screening's SMS already went out, so the idempotency key
1350
+ // suppressed both actions. §025 is the gate that still has to hold.
1351
+ } else {
1352
+ throw new Error(
1353
+ `expected either 1 matched AEL, or all skipped on a repeat run — got ${JSON.stringify(aelByStatus)}`,
1354
+ )
1355
+ }
1356
+ ```
1357
+
1358
+ ## 025 — confirm the verdict-keyed SMS actually rendered
1359
+
1360
+ The matched action fired immediately (`trigger_template: 'now'`) and
1361
+ persisted a row in the `message` dataset carrying the fully-rendered body,
1362
+ prefixed with the verdict that selected it. For this fixture's entity name
1363
+ the agent is prompted toward `verified`, so the persisted SMS reads
1364
+ `[verified]`.
1365
+
1366
+ **This is the gate, not §024.** It asserts the end state — a verdict SMS
1367
+ exists for this screening — which is true on the first run because the
1368
+ action just fired, and true on every run after because it fired once and
1369
+ the idempotency key has protected it ever since.
1370
+
1371
+ **Search on the idempotency key, not on `workflow_id`.** They look
1372
+ interchangeable and are not, and the difference only shows up on the run
1373
+ you care about most. The key is `<screening number>-<verdict>`, and the
1374
+ screening number is stable — so the key names the same SMS forever. A
1375
+ workflow *id* is stable only as long as the workflow resource is. Rebuild
1376
+ the tenant against a lake that still holds its rows — a reset control
1377
+ plane, a restored database, a re-run after `destroy` — and the workflow
1378
+ comes back with a new id while the action logs keep the old key. The
1379
+ action then correctly refuses to fire, because that key already fired, and
1380
+ a search for the new `workflow_id` finds nothing. Green becomes red with
1381
+ nothing wrong: the SMS is right there, filed under the identity that did
1382
+ not change.
1383
+
1384
+ The read answers from **raw**. That is the right tier for this question: it
1385
+ asks what this walk wrote, not whether replication has caught up.
1386
+
1387
+ ```typescript
1388
+ const { data: userSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
1389
+ search_query: `m.idempotency_key LIKE '${ctx.grayZoneNumber}-%'`,
1390
+ })
1391
+ if (userSearch.status !== 'completed') {
1392
+ throw new Error(`message user-search status=${userSearch.status} error=${userSearch.error_message ?? '(none)'}`)
1393
+ }
1394
+
1395
+ const msgDeadline = Date.now() + 45_000
1396
+ let messages: Array<Record<string, unknown>> = []
1397
+ while (Date.now() < msgDeadline && messages.length === 0) {
1398
+ const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
1399
+ userSearchId: userSearch.id!,
1400
+ dataAccessMode: 'raw',
1401
+ })
1402
+ messages = (data.data ?? []) as Array<Record<string, unknown>>
1403
+ if (messages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
1404
+ }
1405
+ if (messages.length === 0) {
1406
+ throw new Error(
1407
+ `no verdict SMS for screening ${ctx.grayZoneNumber} within 45s — ` +
1408
+ `the action never fired, or it fired under a different idempotency key`,
1409
+ )
1410
+ }
1411
+
1412
+ const verdictBody = messages
1413
+ .map((m) => String(m.body ?? ''))
1414
+ .find((body) => body.includes('[verified]') || body.includes('[blocked]'))
1415
+ if (!verdictBody) {
1416
+ throw new Error(`no verdict-prefixed SMS body — bodies: ${JSON.stringify(messages.map((m) => m.body))}`)
1417
+ }
1418
+ if (!verdictBody.includes('[verified]')) {
1419
+ throw new Error(`expected the agent verdict to render [verified] — got: ${verdictBody}`)
1420
+ }
1421
+ ```
1422
+
1423
+ ## 026 — the KYC portal, as a connected app
1424
+
1425
+ The second scenario's SMS carries a deep link to a self-serve KYC portal.
1426
+ The connected app is a thin registration of that portal's URL and mode —
1427
+ the page itself is hosted outside the platform (`mode: 'self_hosted'`). The
1428
+ server-derived slug is what §036 resolves the link against.
1429
+
1430
+ ```typescript
1431
+ const connectedApp = await ctx.ensure(
1432
+ 'KYC portal connected app',
1433
+ async () => {
1434
+ const { data } = await api.connectedApps.list(tenantSlug, datalakeSlug)
1435
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook KYC Portal')
1436
+ },
1437
+ async () => {
1438
+ const { data } = await api.connectedApps.create(tenantSlug, datalakeSlug, {
1439
+ name: 'Cookbook KYC Portal',
1440
+ description: 'Self-serve KYC portal linked from the outbound notification SMS.',
1441
+ mode: 'self_hosted',
1442
+ urls: [{ url: 'https://kyc.example.local', is_primary: true, label: 'production' }],
1443
+ })
1444
+ return data
1445
+ },
1446
+ )
1447
+ connectedAppId = connectedApp.id!
1448
+ ctx.connectedAppSlug = connectedApp.slug!
1449
+ ```
1450
+
1451
+ ## 027 — the payment-accounts table
1452
+
1453
+ `payment_account_external_id` is the unique column. `status` is what the
1454
+ filter gates on, and it stays `none` so the workflow can read it in any
1455
+ tier.
1456
+
1457
+ **`privacy_requirement` is where PCI lives now.** `account_number` and
1458
+ `iban` are `tokenize`; the free-text payment narrative is `redact`. That
1459
+ declaration is the whole masking policy — there is no second schema to
1460
+ write sensitive values into and no changeset quietly tokenizing on the way
1461
+ past. The raw value goes in, and the derived lakes decide who sees what.
1462
+
1463
+ Again: no `legal_entity_id` column. The platform stamps it from Contract B.
1464
+
1465
+ ```typescript
1466
+ const accounts = await ctx.ensure(
1467
+ 'payment-accounts table',
1468
+ async () => {
1469
+ const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
1470
+ return (data.data ?? []).find((t: { title?: string }) => t.title === 'Payment Accounts')
1471
+ },
1472
+ async () => {
1473
+ const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
1474
+ title: 'Payment Accounts',
1475
+ description: 'Payment accounts loaded from the Atomic FI API',
1476
+ columns: [
1477
+ { name: 'payment_account_external_id', title: 'External ID', type: 'string', description: "The provider's account id — unique per account", is_unique: true, privacy_requirement: 'none' },
1478
+ { name: 'payment_account_number', title: 'Account Ref', type: 'string', description: 'Internal account reference', is_unique: false, privacy_requirement: 'none' },
1479
+ { name: 'account_holder_ref', title: 'Account Holder Ref', type: 'string', description: "The provider's account-holder id — the identifier MDM resolves on", is_unique: false, privacy_requirement: 'none' },
1480
+ { name: 'status', title: 'Status', type: 'string', description: 'active / suspended / pending — the filter gate', is_unique: false, privacy_requirement: 'none' },
1481
+ { name: 'account_type', title: 'Account Type', type: 'string', description: 'bank_account / card / wallet', is_unique: false, privacy_requirement: 'none' },
1482
+ { name: 'currency', title: 'Currency', type: 'string', description: 'Settlement currency', is_unique: false, privacy_requirement: 'none' },
1483
+ { name: 'routing_number', title: 'Routing Number', type: 'string', description: 'Domestic routing number', is_unique: false, privacy_requirement: 'none' },
1484
+ { name: 'swift_bic', title: 'SWIFT/BIC', type: 'string', description: 'ISO 9362 bank identifier', is_unique: false, privacy_requirement: 'none' },
1485
+ { name: 'bank_name', title: 'Bank Name', type: 'string', description: 'Holding institution', is_unique: false, privacy_requirement: 'none' },
1486
+ { name: 'account_number', title: 'Account Number', type: 'string', description: 'PCI-sensitive account number', is_unique: false, privacy_requirement: 'tokenize' },
1487
+ { name: 'iban', title: 'IBAN', type: 'string', description: 'PCI-sensitive international account number', is_unique: false, privacy_requirement: 'tokenize' },
1488
+ { name: 'destination_country', title: 'Destination Country', type: 'string', description: 'ISO country derived from the BIC or IBAN, so nothing downstream re-parses it', is_unique: false, privacy_requirement: 'none' },
1489
+ { name: 'purpose_of_payment_detail', title: 'Purpose of Payment', type: 'string', description: 'Free-text payment narrative', is_unique: false, privacy_requirement: 'redact' },
1490
+ { name: 'enabled_regimes', title: 'Enabled Regimes', type: 'string', description: 'Comma-delimited regimes this account may settle under', is_unique: false, privacy_requirement: 'none' },
1491
+ ],
1492
+ })
1493
+ return data
1494
+ },
1495
+ )
1496
+ ctx.accountsTableId = accounts.id!
1497
+ ctx.accountsTableName = accounts.name!
1498
+
1499
+ await ctx.deployGenericTable(ctx.accountsTableName)
1500
+ ```
1501
+
1502
+ ## 028 — the KYC-notification workflow
1503
+
1504
+ Standard shape: a filter passing accounts whose `status` is `active`, a
1505
+ decision naming one key, and one SMS action. The body references the
1506
+ account's external id so the recipient knows which account activated, and
1507
+ `{{ connected_app_form_url }}` so the platform bakes in the `/t/<token>`
1508
+ portal link.
1509
+
1510
+ `skip_mdm_resolution: false` is set explicitly and **must be**. The account
1511
+ holder is the subject the portal token is minted against, and it reaches
1512
+ the workflow only through the `legal_entity_id` stamped on the row.
1513
+
1514
+ A `context_datasets` entry queries the `message` dataset for any
1515
+ KYC notification already sent to the same legal entity in the last six
1516
+ months — the built-in guard against re-notifying. The join key is
1517
+ `legal_entity_id`, and `{{ legal_entity_id }}` is the binding the context
1518
+ builder exposes for the resolved subject.
1519
+
1520
+ The SMS `to` is a fixed compliance inbox. The holder's own contact fields
1521
+ are tokenized, and this workflow deliberately notifies the operator rather
1522
+ than the customer.
1523
+
1524
+ ```typescript
1525
+ const FILTER_BODY = '{% if event_dataset.status == "active" %}true{% endif %}'
1526
+ const DECISION_KEY = 'send_pr_notification_sms'
1527
+ const DECISION_BODY = `["${DECISION_KEY}"]`
1528
+ const KYC_DECISION_OUTPUT_SCHEMA = { type: 'array', items: { type: 'string' } }
1529
+ const SMS_BODY_TEMPLATE =
1530
+ 'Alvera PR notification — payment_account ' +
1531
+ '{{ event_dataset.payment_account_external_id }} has been activated. ' +
1532
+ 'Manage KYC: {{ connected_app_form_url }}'
1533
+
1534
+ const kycWorkflow = await ctx.ensure(
1535
+ 'KYC notification workflow',
1536
+ async () => {
1537
+ const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
1538
+ return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook KYC Notification Workflow')
1539
+ },
1540
+ async () => {
1541
+ const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
1542
+ name: 'Cookbook KYC Notification Workflow',
1543
+ description: 'Sends a KYC-notification SMS for newly activated payment accounts with a self-serve portal link.',
1544
+ dataset_type: 'generic_table',
1545
+ generic_table_id: ctx.accountsTableId,
1546
+ status: 'live',
1547
+ tags: ['compliance', 'kyc'],
1548
+ skip_mdm_resolution: false,
1549
+ filter_config: { type: 'custom', body: FILTER_BODY },
1550
+ decision_config: {
1551
+ type: 'custom',
1552
+ body: DECISION_BODY,
1553
+ output_schema: KYC_DECISION_OUTPUT_SCHEMA,
1554
+ },
1555
+ context_datasets: [
1556
+ {
1557
+ dataset_type: 'message',
1558
+ where_clause:
1559
+ `m.legal_entity_id = '{{ legal_entity_id }}' ` +
1560
+ `AND m.decision_key = '${DECISION_KEY}' ` +
1561
+ "AND m.sent_at > NOW() - INTERVAL '6 months'",
1562
+ limit: 1,
1563
+ position: 0,
1564
+ },
1565
+ ],
1566
+ actions: [
1567
+ {
1568
+ action_type: 'sms',
1569
+ tool_id: toolId,
1570
+ decision_key: DECISION_KEY,
1571
+ position: 0,
1572
+ trigger_template: 'now',
1573
+ idempotency_template: '{{ event_dataset.payment_account_external_id }}-{{ decision_key }}',
1574
+ connected_app_id: connectedAppId,
1575
+ connected_app_route: '/portal/kyc',
1576
+ connected_app_metadata_template:
1577
+ '{"payment_account_external_id":"{{ event_dataset.payment_account_external_id }}"}',
1578
+ tool_call: {
1579
+ tool_call_type: 'sms_request',
1580
+ to: { type: 'custom', body: '+15551234567' },
1581
+ body: { type: 'custom', body: SMS_BODY_TEMPLATE },
1582
+ sms_type: 'transactional',
1583
+ },
1584
+ },
1585
+ ],
1586
+ })
1587
+ return data
1588
+ },
1589
+ )
1590
+ ctx.kycWorkflowId = kycWorkflow.id!
1591
+ ctx.kycWorkflowSlug = kycWorkflow.slug!
1592
+
1593
+ if (kycWorkflow.skip_mdm_resolution !== false) {
1594
+ throw new Error('a generic-table workflow that mints a per-holder link must keep MDM resolution ON')
1595
+ }
1596
+ ```
1597
+
1598
+ ## 029 — the payment-account contract pair
1599
+
1600
+ Same two-contract shape as §019, over a different table. Contract A writes
1601
+ the holder as the subject; Contract B writes the account row and emits the
1602
+ same `(uri, account_holder_id)` identifier so both converge and the row is
1603
+ stamped.
1604
+
1605
+ ```typescript
1606
+ const { readFileSync: readFileSync2 } = await import('node:fs')
1607
+ const { join: join2 } = await import('node:path')
1608
+ const read2 = (f: string) =>
1609
+ readFileSync2(join2(process.env.COOKBOOK_FIXTURES_DIR!, 'payments-compliance', f), 'utf8')
1610
+
1611
+ const accountLeContract = await ctx.ensure(
1612
+ 'payment-account holder LE contract',
1613
+ async () => {
1614
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1615
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Atomic FI Holder LE Contract')
1616
+ },
1617
+ async () => {
1618
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1619
+ name: 'Cookbook Atomic FI Holder LE Contract',
1620
+ description: 'Atomic FI payment-account → the account holder LegalEntity.',
1621
+ resource_type: 'legal_entity',
1622
+ type: 'identity',
1623
+ generic_table_id: null,
1624
+ template_config: { type: 'custom', body: read2('_payment_accounts_legal_entity.liquid') },
1625
+ mdm_input_config: { type: 'null' },
1626
+ })
1627
+ return data
1628
+ },
1629
+ )
1630
+ ctx.accountLeContractId = accountLeContract.id!
1631
+
1632
+ const accountGtContract = await ctx.ensure(
1633
+ 'payment-account GT contract',
1634
+ async () => {
1635
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1636
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Atomic FI Payment Account GT Contract')
1637
+ },
1638
+ async () => {
1639
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1640
+ name: 'Cookbook Atomic FI Payment Account GT Contract',
1641
+ description: 'Atomic FI payment-account → payment_accounts row, stamped with its holder subject.',
1642
+ resource_type: 'generic_table',
1643
+ type: 'identity',
1644
+ generic_table_id: ctx.accountsTableId,
1645
+ template_config: { type: 'custom', body: read2('_payment_accounts_generic_table.liquid') },
1646
+ mdm_input_config: { type: 'custom', body: read2('_payment_accounts_mdm.liquid') },
1647
+ })
1648
+ return data
1649
+ },
1650
+ )
1651
+ ctx.accountGtContractId = accountGtContract.id!
1652
+ ```
1653
+
1654
+ ## 030 — the payment-account DACs, split the same way
1655
+
1656
+ The second contract pair gets the same treatment as the first, for the same
1657
+ reason — §020 has the full account of why. Both clients reuse the same
1658
+ manual-upload tool and the same data source; what differs between any two
1659
+ of these ingestion paths is only the contracts, which is the level the
1660
+ difference actually lives at.
1661
+
1662
+ ```typescript
1663
+ const accountIdentityDac = await ctx.ensure(
1664
+ 'payment-account identity DAC',
1665
+ async () => {
1666
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
1667
+ return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Atomic FI Payment Account Identity DAC')
1668
+ },
1669
+ async () => {
1670
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1671
+ name: 'Cookbook Atomic FI Payment Account Identity DAC',
1672
+ description: 'Manual-upload DAC — writes the account holder as a legal entity, ahead of the account row.',
1673
+ tool_id: ctx.manualUploadToolId,
1674
+ data_source_id: dataSourceId,
1675
+ tool_call: { tool_call_type: 'manual_upload' },
1676
+ interoperability_contract_ids: [ctx.accountLeContractId],
1677
+ })
1678
+ return data
1679
+ },
1680
+ )
1681
+ ctx.accountIdentityDacSlug = accountIdentityDac.slug!
1682
+
1683
+ const accountDac = await ctx.ensure(
1684
+ 'payment-account data DAC',
1685
+ async () => {
1686
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
1687
+ return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Atomic FI Payment Account Data DAC')
1688
+ },
1689
+ async () => {
1690
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1691
+ name: 'Cookbook Atomic FI Payment Account Data DAC',
1692
+ description: 'Manual-upload DAC — ingests payment-account rows, resolving the holder written by the identity DAC.',
1693
+ tool_id: ctx.manualUploadToolId,
1694
+ data_source_id: dataSourceId,
1695
+ tool_call: { tool_call_type: 'manual_upload' },
1696
+ interoperability_contract_ids: [ctx.accountGtContractId],
1697
+ })
1698
+ return data
1699
+ },
1700
+ )
1701
+ ctx.accountDacSlug = accountDac.slug!
1702
+ ```
1703
+
1704
+ ## 031 — ingest one active account and one suspended one
1705
+
1706
+ Two rows differing in the field the filter reads. `active` passes;
1707
+ `suspended` must be rejected.
1708
+
1709
+ Two stages again, holders before accounts, exactly as in §021.
1710
+
1711
+ ```typescript
1712
+ ctx.activeExternalId = 'PA-COOKBOOK-0101'
1713
+ ctx.suspendedExternalId = 'PA-COOKBOOK-0102'
1714
+
1715
+ const activeRow = {
1716
+ payment_account_external_id: ctx.activeExternalId,
1717
+ payment_account_number: 'PA-NUM-COOKBOOK-0101',
1718
+ account_holder_id: 'AH-COOKBOOK-KYC-0101',
1719
+ account_holder_name: 'Ada Cookbook-Active-Lovelace',
1720
+ status: 'active',
1721
+ account_type: 'bank_account',
1722
+ currency: 'USD',
1723
+ routing_number: '121000358',
1724
+ swift_bic: 'BOFAUS3N',
1725
+ bank_name: 'Bank of America',
1726
+ enabled_regimes: ['us_domestic'],
1727
+ source_uri: 'api.atomic.fi/compliance',
1728
+ }
1729
+ const suspendedRow = {
1730
+ payment_account_external_id: ctx.suspendedExternalId,
1731
+ payment_account_number: 'PA-NUM-COOKBOOK-0102',
1732
+ account_holder_id: 'AH-COOKBOOK-KYC-0102',
1733
+ account_holder_name: 'Grace Cookbook-Suspended-Hopper',
1734
+ status: 'suspended',
1735
+ account_type: 'bank_account',
1736
+ currency: 'USD',
1737
+ routing_number: '121000358',
1738
+ bank_name: 'Bank of America',
1739
+ enabled_regimes: ['us_domestic'],
1740
+ source_uri: 'api.atomic.fi/compliance',
1741
+ }
1742
+
1743
+ // STAGE 1 — the account holders.
1744
+ const holderActive = await api.dataActivationClients.ingest(
1745
+ tenantSlug, datalakeSlug, ctx.accountIdentityDacSlug, { data: activeRow },
1746
+ )
1747
+ const holderSuspended = await api.dataActivationClients.ingest(
1748
+ tenantSlug, datalakeSlug, ctx.accountIdentityDacSlug, { data: suspendedRow },
1749
+ )
1750
+ await ctx.waitForBatches(ctx.accountIdentityDacSlug, [
1751
+ holderActive.data.batch_id!,
1752
+ holderSuspended.data.batch_id!,
1753
+ ])
1754
+
1755
+ // STAGE 2 — the accounts themselves.
1756
+ const activeIngest = await api.dataActivationClients.ingest(
1757
+ tenantSlug, datalakeSlug, ctx.accountDacSlug, { data: activeRow },
1758
+ )
1759
+ const suspendedIngest = await api.dataActivationClients.ingest(
1760
+ tenantSlug, datalakeSlug, ctx.accountDacSlug, { data: suspendedRow },
1761
+ )
1762
+ ctx.batchActive = activeIngest.data.batch_id!
1763
+ ctx.batchSuspended = suspendedIngest.data.batch_id!
1764
+ ```
1765
+
1766
+ ## 032 — wait for both account batches to persist
1767
+
1768
+ Same gate as §022, on the account data DAC's logs.
1769
+
1770
+ ```typescript
1771
+ await ctx.waitForBatches(ctx.accountDacSlug, [ctx.batchActive, ctx.batchSuspended])
1772
+ ```
1773
+
1774
+ ## 033 — run the KYC workflow
1775
+
1776
+ ```typescript
1777
+ const kycRunResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.kycWorkflowSlug, {
1778
+ sql_where_clause: `payment_account_external_id IN ('${ctx.activeExternalId}', '${ctx.suspendedExternalId}')`,
1779
+ mode: 'live',
1780
+ manual_override: false,
1781
+ })
1782
+ const kycFired = await ctx.waitForFiredRun(datalakeSlug, kycRunResp.data.workflow_run_id)
1783
+ ctx.kycRunLogId = kycFired.workflowRunLogId
1784
+ ctx.kycRunBatchId = kycFired.batchId!
1785
+
1786
+ const kycDeadline = Date.now() + 240_000
1787
+ let kycStatus: string | null = null
1788
+ while (Date.now() < kycDeadline) {
1789
+ const { data: log } = await api.workflows.batchLogs.refresh(
1790
+ tenantSlug, datalakeSlug, ctx.kycWorkflowSlug, ctx.kycRunLogId,
1791
+ )
1792
+ kycStatus = log.status ?? null
1793
+ if (kycStatus && kycStatus !== 'pending') break
1794
+ await new Promise((r) => setTimeout(r, 2_000))
1795
+ }
1796
+ if (kycStatus === 'failed') throw new Error('KYC workflow run reached :failed')
1797
+ if (!kycStatus || kycStatus === 'pending') {
1798
+ throw new Error('KYC workflow run did not leave :pending within 240s')
1799
+ }
1800
+ ```
1801
+
1802
+ ## 034 — verify the filter routed: active passes, suspended is filtered
1803
+
1804
+ The active account passes the filter, so its WEL is `:executing` or
1805
+ `:completed`. The suspended one fails it, so its WEL is `:filtered`. Both
1806
+ passing — or both filtered — would mean the filter is not reading
1807
+ `event_dataset.status` at all.
1808
+
1809
+ ```typescript
1810
+ const kycWelDeadline = Date.now() + 120_000
1811
+ let kycWels: Array<Record<string, unknown>> = []
1812
+ let kycByStatus: Record<string, number> = {}
1813
+ while (Date.now() < kycWelDeadline) {
1814
+ const { data: wfLogs } = await api.workflows.workflowLogs.list(tenantSlug, datalakeSlug, ctx.kycWorkflowSlug)
1815
+ kycWels = (wfLogs.data ?? [])
1816
+ .map((w) => w as Record<string, unknown>)
1817
+ .filter((w) => w.batch_id === ctx.kycRunBatchId)
1818
+ kycByStatus = {}
1819
+ for (const w of kycWels) {
1820
+ const st = (w.status as string | undefined) ?? 'unknown'
1821
+ kycByStatus[st] = (kycByStatus[st] ?? 0) + 1
1822
+ }
1823
+ const settled = (kycByStatus.completed ?? 0) + (kycByStatus.executing ?? 0)
1824
+ if (kycWels.length === 2 && settled === 1 && (kycByStatus.filtered ?? 0) === 1) break
1825
+ await new Promise((r) => setTimeout(r, 2_000))
1826
+ }
1827
+ if (kycWels.length !== 2) {
1828
+ throw new Error(`expected 2 KYC WELs for the run, got ${kycWels.length}`)
1829
+ }
1830
+ const kycPassCount = (kycByStatus.executing ?? 0) + (kycByStatus.completed ?? 0)
1831
+ if (kycPassCount !== 1) {
1832
+ throw new Error(`expected 1 pass-branch WEL, got ${kycPassCount} — ${JSON.stringify(kycByStatus)}`)
1833
+ }
1834
+ if ((kycByStatus.filtered ?? 0) !== 1) {
1835
+ throw new Error(`expected 1 :filtered WEL (the suspended account) — ${JSON.stringify(kycByStatus)}`)
1836
+ }
1837
+ ```
1838
+
1839
+ ## 035 — read the notification SMS back, and pull its portal token
1840
+
1841
+ The passing action fired immediately and persisted a `message` row with
1842
+ the fully-rendered body. Extract the `/t/<token>` shortlink the platform
1843
+ baked in when it minted the connected-app page token.
1844
+
1845
+ End-state assertion again, keyed the same way §025 is and for the same
1846
+ reason: on a repeat run the SMS was not re-sent, and the row from the first
1847
+ run is still the correct answer to "does this account have a notification
1848
+ with a portal link". The account's external id is what makes that row
1849
+ findable forever; the workflow's id is not.
1850
+
1851
+ ```typescript
1852
+ const { data: kycSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
1853
+ search_query: `m.idempotency_key LIKE '${ctx.activeExternalId}-%'`,
1854
+ })
1855
+ if (kycSearch.status !== 'completed') {
1856
+ throw new Error(`message user-search status=${kycSearch.status} error=${kycSearch.error_message ?? '(none)'}`)
1857
+ }
1858
+
1859
+ const kycMsgDeadline = Date.now() + 45_000
1860
+ let kycMessages: Array<Record<string, unknown>> = []
1861
+ while (Date.now() < kycMsgDeadline && kycMessages.length === 0) {
1862
+ const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
1863
+ userSearchId: kycSearch.id!,
1864
+ dataAccessMode: 'raw',
1865
+ })
1866
+ kycMessages = (data.data ?? []) as Array<Record<string, unknown>>
1867
+ if (kycMessages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
1868
+ }
1869
+ if (kycMessages.length === 0) {
1870
+ throw new Error(
1871
+ `no KYC-notification SMS for account ${ctx.activeExternalId} within 45s — ` +
1872
+ `the action never fired, or it fired under a different idempotency key`,
1873
+ )
1874
+ }
1875
+
1876
+ const withLink = kycMessages
1877
+ .map((m) => String(m.body ?? ''))
1878
+ .find((body) => body.includes('/t/') && body.includes('Alvera PR notification'))
1879
+ if (!withLink) {
1880
+ throw new Error(`no rendered body with a /t/ link — bodies: ${JSON.stringify(kycMessages.map((m) => m.body))}`)
1881
+ }
1882
+ const tokenMatch = withLink.match(/\/t\/([A-Za-z0-9_-]+)/)
1883
+ if (!tokenMatch) {
1884
+ throw new Error(`no /t/<token> in rendered body: ${withLink}`)
1885
+ }
1886
+ ctx.kycShortPath = tokenMatch[1]!
1887
+ ```
1888
+
1889
+ ## 036 — resolve the deep link, and close the loop with tracking
1890
+
1891
+ The `/t/<token>` shortlink resolves through `connectedApps.resolvePage`,
1892
+ and the `route_path` must match the action's `connected_app_route`.
1893
+ Posting `opened_at` and `form_submitted_at` through
1894
+ `updateMessageTracking` then mirrors exactly what the portal frontend does
1895
+ when the holder opens it — closing the SMS-to-reply loop end to end.
1896
+
1897
+ ```typescript
1898
+ const { data: resolved } = await api.connectedApps.resolvePage(
1899
+ tenantSlug, datalakeSlug, ctx.connectedAppSlug,
1900
+ { short_path: ctx.kycShortPath, user_agent: 'cookbook-doctest/kyc-notification' },
1901
+ )
1902
+ if (resolved.route_path !== '/portal/kyc') {
1903
+ throw new Error(`resolvePage route_path mismatch: ${resolved.route_path}`)
1904
+ }
1905
+ if (!(resolved.message?.body ?? '').includes('Alvera PR notification')) {
1906
+ throw new Error('resolved page message body missing the notification copy')
1907
+ }
1908
+
1909
+ const now = new Date().toISOString()
1910
+ const { data: tracked } = await api.connectedApps.updateMessageTracking(
1911
+ tenantSlug, datalakeSlug, ctx.connectedAppSlug,
1912
+ { short_path: ctx.kycShortPath, opened_at: now, form_submitted_at: now },
1913
+ )
1914
+ if (!tracked.message?.opened_at || !tracked.message?.form_submitted_at) {
1915
+ throw new Error('message tracking did not persist opened_at + form_submitted_at')
1916
+ }
1917
+ ```
1918
+
1919
+
1920
+ ## 037 — read one row back through all three lakes
1921
+
1922
+ This is what §007 was for. The same row, the same SQL, three modes — and
1923
+ the difference between them is the whole argument for turning tokenization
1924
+ on.
1925
+
1926
+ `screened_entity_name` is declared `tokenize`, so the raw lake holds the
1927
+ name and neither copy does. `screening_score` and `match_count` are
1928
+ `privacy_requirement: 'none'`, so they are **identical in all three** —
1929
+ which is the point rather than an aside. The agent in §016 disambiguates on
1930
+ score, match count and type; it never needed to know who was screened. A
1931
+ masking policy that also destroyed the signal would have made the agent
1932
+ useless, and one that preserved the name would have made it a liability.
1933
+
1934
+ Note that the mode is a parameter of the read, not a different slug: one
1935
+ lake is addressed, and `mode` selects which copy answers. The session's
1936
+ `data_access_mode` is a **ceiling**, so this step only works because §004
1937
+ built its client at `raw`; a tokenized-ceiling session asking for `raw` is
1938
+ refused rather than quietly downgraded.
1939
+
1940
+ ```typescript
1941
+ const maskedReadSql =
1942
+ `SELECT compliance_screening_number, screened_entity_name, screening_score, match_count ` +
1943
+ `FROM ${ctx.screeningsTableName} ` +
1944
+ `WHERE compliance_screening_number = '${ctx.grayZoneNumber}'`
1945
+
1946
+ const readAs = async (mode: 'raw' | 'tokenized' | 'redacted') => {
1947
+ const { data } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
1948
+ sql: maskedReadSql,
1949
+ mode,
1950
+ })
1951
+ const row = (data.data ?? [])[0]
1952
+ if (!row) {
1953
+ throw new Error(`the gray-zone screening was not readable through the ${mode} lake`)
1954
+ }
1955
+ const cell = (column: string): unknown => row[data.meta.columns.indexOf(column)]
1956
+ return {
1957
+ name: cell('screened_entity_name'),
1958
+ score: Number(cell('screening_score')),
1959
+ matchCount: Number(cell('match_count')),
1960
+ }
1961
+ }
1962
+
1963
+ const rawRead = await readAs('raw')
1964
+ const tokenizedRead = await readAs('tokenized')
1965
+ const redactedRead = await readAs('redacted')
1966
+
1967
+ console.log(' raw →', JSON.stringify(rawRead))
1968
+ console.log(' tokenized →', JSON.stringify(tokenizedRead))
1969
+ console.log(' redacted →', JSON.stringify(redactedRead))
1970
+
1971
+ // The identity does NOT survive the copy.
1972
+ if (rawRead.name !== ctx.grayZoneName) {
1973
+ throw new Error(`the raw lake should hold the real name, got ${JSON.stringify(rawRead.name)}`)
1974
+ }
1975
+ if (tokenizedRead.name === ctx.grayZoneName) {
1976
+ throw new Error('the tokenized lake handed back the real name — tokenization is not in force')
1977
+ }
1978
+ if (redactedRead.name === ctx.grayZoneName) {
1979
+ throw new Error('the redacted lake handed back the real name — redaction is not in force')
1980
+ }
1981
+
1982
+ // The signal DOES. A masking policy that broke this would break the agent.
1983
+ for (const [mode, read] of [['tokenized', tokenizedRead], ['redacted', redactedRead]] as const) {
1984
+ if (read.score !== rawRead.score || read.matchCount !== rawRead.matchCount) {
1985
+ throw new Error(
1986
+ `the ${mode} lake changed a 'none' column: score ${read.score} vs ${rawRead.score}, ` +
1987
+ `match_count ${read.matchCount} vs ${rawRead.matchCount}`,
1988
+ )
1989
+ }
1990
+ }
1991
+ ```
1992
+
1993
+ ## 038 — how far the tokenized copy actually got
1994
+
1995
+ Every masked read above trusted the copy. This asks it directly, and on a
1996
+ compliance surface the question is not academic: an agent reading a copy that
1997
+ has not caught up is an agent deciding on a partial view of who was screened.
1998
+
1999
+ The addressing is the first trap. You name the **primary** lake plus a mode.
2000
+ A derived lake does have a slug, but that slug is not an address — the lookup
2001
+ resolves primaries only — so the walk asks with the tokenized lake's own slug
2002
+ and proves the refusal instead of assuming it.
2003
+
2004
+ The second trap is the state names. `finished_copy` has **not** finished: the
2005
+ first pass is done, but the changes made during that pass are still
2006
+ unapplied. Read `caught_up`, which exists precisely so nobody has to
2007
+ pattern-match the word. `label` is the same state in words and is what the
2008
+ platform's own Tokenization screen renders — reuse it rather than inventing a
2009
+ second vocabulary for the same five states.
2010
+
2011
+ `resync_empties` carries a rule worth asserting: `null` when a table is
2012
+ already caught up, because the platform will not spend a query per table
2013
+ answering a question about tables nobody will repair — and a non-empty array
2014
+ otherwise, naming the table itself first. `null` and `[]` are different
2015
+ answers here.
2016
+
2017
+ ```typescript
2018
+ const { data: pair } = await api.datalakes.provisionTokenization(tenantSlug, datalakeSlug)
2019
+ const tokenizedLake = (pair.derived_datalakes ?? []).find((d) => d.type === 'tokenized')
2020
+ if (!tokenizedLake) throw new Error('no tokenized copy on the lake — §007 should have provisioned it')
2021
+
2022
+ const { data: tables } = await api.datalakes.derivedLakeTables(
2023
+ tenantSlug, datalakeSlug, 'tokenized',
2024
+ )
2025
+ if (tables.data.length === 0) throw new Error('the tokenized copy reported no tables at all')
2026
+ console.log(` ${tables.data.length} tables in the tokenized copy`)
2027
+
2028
+ for (const t of tables.data) {
2029
+ if (typeof t.caught_up !== 'boolean') {
2030
+ throw new Error(`${t.table}: caught_up is not a boolean, got ${JSON.stringify(t.caught_up)}`)
2031
+ }
2032
+ if (!t.label || t.label.length === 0) {
2033
+ throw new Error(`${t.table}: label is empty — it is what a screen renders`)
2034
+ }
2035
+ // The agreement that makes the boolean trustworthy: `caught_up` is true for
2036
+ // exactly one of the five states. `finished_copy` reading true here would be
2037
+ // the defect the boolean exists to prevent. This cannot be asserted against a
2038
+ // mock — the schema cannot express a cross-field constraint, so Prism draws
2039
+ // the two independently — which is exactly why it is asserted here.
2040
+ if (t.caught_up !== (t.state === 'caught_up')) {
2041
+ throw new Error(
2042
+ `${t.table}: state '${t.state}' disagrees with caught_up=${t.caught_up} — ` +
2043
+ `only 'caught_up' means arrived`,
2044
+ )
2045
+ }
2046
+ if (t.caught_up && t.resync_empties !== null && t.resync_empties !== undefined) {
2047
+ throw new Error(
2048
+ `${t.table} is caught up, so resync_empties should not be computed — got ` +
2049
+ JSON.stringify(t.resync_empties),
2050
+ )
2051
+ }
2052
+ if (!t.caught_up) {
2053
+ if (!Array.isArray(t.resync_empties) || t.resync_empties.length === 0) {
2054
+ throw new Error(`${t.table} is behind, so resync_empties should be a non-empty array`)
2055
+ }
2056
+ if (t.resync_empties[0] !== t.table) {
2057
+ throw new Error(`${t.table}: resync_empties should name the table itself first`)
2058
+ }
2059
+ }
2060
+ }
2061
+ ctx.copyTables = tables.data.map((t) => t.table)
2062
+
2063
+ // The derived lake's own slug is not a second way to name the same thing.
2064
+ let refusedDerivedSlug = false
2065
+ try {
2066
+ await api.datalakes.derivedLakeTables(tenantSlug, tokenizedLake.slug!, 'tokenized')
2067
+ } catch (err) {
2068
+ const status = (err as { _httpStatus?: number })._httpStatus
2069
+ if (status !== 404) throw err
2070
+ refusedDerivedSlug = true
2071
+ }
2072
+ if (!refusedDerivedSlug) {
2073
+ throw new Error("the derived lake's own slug was accepted as an address — it should 404")
2074
+ }
2075
+ ```
2076
+
2077
+ ## 039 — empty one table and wait for it back
2078
+
2079
+ `resyncTable` empties a table in the copy and copies it again from the start.
2080
+ It is the repair for a table that is stuck or wrong, and two of its properties
2081
+ are the kind that bite in production, so both are asserted rather than
2082
+ described.
2083
+
2084
+ **The blast radius is bigger than the table you name.** Postgres will not
2085
+ truncate a table with an inbound foreign key, regardless of whether the
2086
+ referencing tables hold any rows, so repairing a parent empties its whole
2087
+ dependent closure, transitively, in a single statement — and all of them are
2088
+ re-copied by the same run. `legal_entities` is the target here on purpose:
2089
+ §029's contract pair resolved every screened subject through it, and
2090
+ `legal_entity_identifications` points at it, so `emptied` genuinely names more
2091
+ than one table. Any surface offering this button should read `resync_empties`
2092
+ and say the count before it is pressed.
2093
+
2094
+ **`202` means started.** There is no completion callback, so the walk polls
2095
+ `derivedLakeTables` until every emptied table reads `caught_up` again. That
2096
+ poll is also what makes this a proof rather than a dispatch.
2097
+
2098
+ The gate goes first. `table` is checked against what the copy carries or the
2099
+ original can send, and anything else is a `422` with nothing emptied — the
2100
+ one thing standing between a typo and an emptied table. Proving it *after*
2101
+ running the real repair would be checking the lock from inside the house.
2102
+
2103
+ ```typescript
2104
+ const TARGET = 'legal_entities'
2105
+ if (!(ctx.copyTables as string[]).includes(TARGET)) {
2106
+ throw new Error(`${TARGET} is not in the copy; it carries: ${(ctx.copyTables as string[]).join(', ')}`)
2107
+ }
2108
+
2109
+ let refusedBadTable = false
2110
+ try {
2111
+ await api.datalakes.resyncTable(tenantSlug, datalakeSlug, 'tokenized', {
2112
+ table: 'no_such_table_here',
2113
+ })
2114
+ } catch (err) {
2115
+ const status = (err as { _httpStatus?: number })._httpStatus
2116
+ if (status !== 422) throw err
2117
+ refusedBadTable = true
2118
+ }
2119
+ if (!refusedBadTable) throw new Error('an unknown table name was accepted — the gate is not closed')
2120
+
2121
+ const { data: repair } = await api.datalakes.resyncTable(
2122
+ tenantSlug, datalakeSlug, 'tokenized', { table: TARGET },
2123
+ )
2124
+ if (repair.table !== TARGET) throw new Error(`repaired ${repair.table}, asked for ${TARGET}`)
2125
+ if (repair.mode !== 'tokenized') throw new Error(`repaired the ${repair.mode} copy, asked for tokenized`)
2126
+ if (repair.status !== 'copying') throw new Error(`expected status 'copying', got ${repair.status}`)
2127
+ if (!repair.emptied.includes(TARGET)) {
2128
+ throw new Error(`emptied ${JSON.stringify(repair.emptied)} — the named table is not among them`)
2129
+ }
2130
+ console.log(` emptied ${repair.emptied.length} table(s): ${repair.emptied.join(', ')}`)
2131
+
2132
+ // Identifications point at legal entities, so the closure cannot be just one.
2133
+ if (repair.emptied.length < 2) {
2134
+ throw new Error(
2135
+ `${TARGET} has dependents, so a repair should empty more than itself — got ` +
2136
+ JSON.stringify(repair.emptied),
2137
+ )
2138
+ }
2139
+
2140
+ const RESYNC_TIMEOUT_MS = 5 * 60_000
2141
+ const resyncDeadline = Date.now() + RESYNC_TIMEOUT_MS
2142
+ let backOnline = false
2143
+ while (Date.now() < resyncDeadline && !backOnline) {
2144
+ const { data: after } = await api.datalakes.derivedLakeTables(
2145
+ tenantSlug, datalakeSlug, 'tokenized',
2146
+ )
2147
+ backOnline = (repair.emptied as string[]).every(
2148
+ (name) => after.data.find((t) => t.table === name)?.caught_up === true,
2149
+ )
2150
+ if (!backOnline) await new Promise((r) => setTimeout(r, 5_000))
2151
+ }
2152
+ if (!backOnline) {
2153
+ throw new Error(`the repaired tables did not come back within ${RESYNC_TIMEOUT_MS / 1000}s`)
2154
+ }
2155
+ console.log(' every emptied table is caught up again')
2156
+ ```
2157
+
2158
+ # Outcome
2159
+
2160
+ One tenant carries the whole payments-compliance surface: a raw datalake
2161
+ with its tokenized and redacted copies, two generic tables, two contract
2162
+ pairs split across four Data Activation Clients, and two agentic workflows
2163
+ that between them screen, decide, notify and track — plus the copy's own
2164
+ arrival state, read back per table and repaired.
2165
+
2166
+ Two things are worth carrying out of this walk. **The pair must not run
2167
+ concurrently** (§020) — one writer per subject, or the unique index that
2168
+ guards against duplicate entities refuses your row on the only run that
2169
+ matters, the first. And **assert end states, not transitions** (§024) — an
2170
+ idempotency key that has already fired will not fire again, which is the
2171
+ product working, not the walk failing.
2172
+
2173
+ Running this walk again reuses all of it.
2174
+
2175
+ # See also
2176
+
2177
+ - `datalakes.md` — the create body, and what the derived pair is for
2178
+ - `workflows.md` — scheduling versus firing, and the accessor tables
2179
+ - `interoperability_contracts.md` — the contract pair, and the
2180
+ reserved `legal_entity_id` stamp you must not declare