@alvera-ai/platform-sdk 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/AGENTS.md +82 -144
- package/.agent/account_management.md +2 -2
- package/.agent/action_logs.md +4 -4
- package/.agent/ai_agents.md +28 -21
- package/.agent/ai_sandbox.md +49 -39
- package/.agent/connected_apps.md +3 -3
- package/.agent/cookbook/_fixtures/README.md +1 -1
- package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_generic_table.liquid +1 -1
- package/.agent/cookbook/_fixtures/organic-marketing/_lead_submissions_foundation_legal_entity.liquid +80 -0
- package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_mdm.liquid +2 -1
- package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_generic_table.liquid +57 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_legal_entity.liquid +30 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_mdm.liquid +44 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_generic_table.liquid +57 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_legal_entity.liquid +36 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_mdm.liquid +41 -0
- package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_generic_table.liquid +70 -0
- package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_legal_entity.liquid +52 -0
- package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_mdm.liquid +42 -0
- package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_generic_table.liquid +38 -0
- package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_legal_entity.liquid +48 -0
- package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_mdm.liquid +49 -0
- package/.agent/cookbook/organic-marketing.md +2801 -0
- package/.agent/cookbook/payments-compliance.md +2180 -0
- package/.agent/cookbook/primary-care.md +2175 -0
- package/.agent/cookbook/subscription-saas.md +2403 -0
- package/.agent/data_activation_clients.md +65 -52
- package/.agent/datalakes.md +338 -171
- package/.agent/errors.md +3 -3
- package/.agent/generic_tables.md +151 -62
- package/.agent/interoperability_contracts.md +57 -22
- package/.agent/mdm.md +136 -153
- package/.agent/messages.md +36 -34
- package/.agent/mock-services.md +1 -1
- package/.agent/mutations.md +2 -2
- package/.agent/templates.md +14 -13
- package/.agent/tools.md +63 -21
- package/.agent/type_naming.md +13 -13
- package/.agent/workflows.md +99 -53
- package/README.md +2 -2
- package/dist/bin/platform-sdk.mjs +33 -47
- package/dist/bin/platform-sdk.mjs.map +1 -1
- package/dist/index.d.mts +565 -379
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +494 -59
- package/dist/index.mjs.map +1 -1
- package/package.json +4 -3
- package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +0 -88
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +0 -47
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +0 -24
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +0 -38
- package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_compliance_screening.liquid +0 -59
- package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_mdm.liquid +0 -36
- package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_mdm.liquid +0 -30
- package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_payment_account.liquid +0 -55
- package/.agent/cookbook/_fixtures/subscription/_customers_subscription_mdm.liquid +0 -20
- package/.agent/cookbook/_setup/foundation.md +0 -359
- package/.agent/cookbook/_setup/healthcare.md +0 -361
- package/.agent/cookbook/_setup/payments.md +0 -365
- package/.agent/cookbook/_setup/subscription.md +0 -364
- package/.agent/cookbook/action-status-updaters.md +0 -278
- package/.agent/cookbook/ai-agent-invoke.md +0 -279
- package/.agent/cookbook/appointment-review-sms-workflow.md +0 -801
- package/.agent/cookbook/birthday-greeting-sms-trigger.md +0 -696
- package/.agent/cookbook/bulk-ingest.md +0 -302
- package/.agent/cookbook/contact-us-triage-with-llm.md +0 -663
- package/.agent/cookbook/dunning-sms-for-delinquent.md +0 -659
- package/.agent/cookbook/generic-tables.md +0 -244
- package/.agent/cookbook/invite-team.md +0 -200
- package/.agent/cookbook/kyc-notification-on-account-activation.md +0 -661
- package/.agent/cookbook/marketing-campaign-send.md +0 -1044
- package/.agent/cookbook/paginated-restapi-poller.md +0 -383
- package/.agent/cookbook/rest-fetch.md +0 -273
- package/.agent/cookbook/sanctions-screening-with-agent-review.md +0 -773
- package/.agent/cookbook/score-leads-with-llm-categorization.md +0 -665
- package/.agent/cookbook/system-templates.md +0 -165
- package/.agent/cookbook/talk-to-data.md +0 -178
- package/.agent/cookbook/triage-prospects-by-priority.md +0 -571
- package/.agent/cookbook/welcome-sms-for-customers.md +0 -647
- /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/memorandum-of-association-01.png +0 -0
- /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/sample_two_page.pdf +0 -0
- /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/_customers_subscription_customer.liquid +0 -0
- /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/stripe_customers_batch1.csv +0 -0
|
@@ -0,0 +1,2180 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: "Payments compliance: screen, decide, notify — on a tokenized lake"
|
|
3
|
+
summary: The whole payments-compliance surface in one walk. Stand up a tenant and a raw datalake, turn tokenization on because a model is going to read this data, then run two scenarios end to end — an LLM agent disambiguating gray-zone sanctions screenings into a verdict-keyed SMS fan-out, and a KYC notification fired by an account activation. Ends by reading one row back through raw, tokenized and redacted keys to show what each reader actually sees, then inspecting how far the tokenized copy has got and repairing one of its tables.
|
|
4
|
+
use_case: payments-compliance
|
|
5
|
+
slug: payments-compliance
|
|
6
|
+
vitest_source:
|
|
7
|
+
- integration-tests/tests/payments-compliance/sanctions-review-workflow.test.ts
|
|
8
|
+
- integration-tests/tests/payments-compliance/kyc-upload-workflow.test.ts
|
|
9
|
+
- integration-tests/tests/payments-compliance/interoperability-contracts.test.ts
|
|
10
|
+
- integration-tests/tests/payments-compliance/run-dac-single.test.ts
|
|
11
|
+
- integration-tests/tests/payments-compliance/create-dac.test.ts
|
|
12
|
+
- integration-tests/tests/payments-compliance/system-templates.test.ts
|
|
13
|
+
- integration-tests/tests/payments-compliance/manual-upload-tool.test.ts
|
|
14
|
+
- integration-tests/tests/payments-compliance/data-sources.test.ts
|
|
15
|
+
- integration-tests/tests/workspace/bootstrap.test.ts
|
|
16
|
+
status: draft
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
# Problem
|
|
20
|
+
|
|
21
|
+
A payments compliance team has two jobs that look like one. Screenings
|
|
22
|
+
arrive from a provider and most of them answer themselves — a clean pass,
|
|
23
|
+
an obvious hit. The ones in the middle are the work, and they are the ones
|
|
24
|
+
a human should not be reading first. Separately, an account activation has
|
|
25
|
+
to reach the holder as a notification, promptly, and not at all if the
|
|
26
|
+
account was suspended instead.
|
|
27
|
+
|
|
28
|
+
Both are the same shape: rows land, a filter decides which ones matter, and
|
|
29
|
+
something is sent about the ones that do. This walk builds both on one
|
|
30
|
+
tenant, because a compliance team does not run two platforms.
|
|
31
|
+
|
|
32
|
+
It also turns **tokenization** on, and that is the point of doing this
|
|
33
|
+
surface rather than another one. A gray-zone screening is disambiguated by
|
|
34
|
+
a language model. The model needs the shape of the record — the score, the
|
|
35
|
+
match count, the type — and it does not need the name of the person. The
|
|
36
|
+
tokenized copy of the lake is what makes that distinction enforceable
|
|
37
|
+
rather than a promise, so this walk provisions it in §007 and reads
|
|
38
|
+
through it in §037.
|
|
39
|
+
|
|
40
|
+
# Composition
|
|
41
|
+
|
|
42
|
+
| Resource | Why it exists here |
|
|
43
|
+
|-------------------------------------|---------------------------------------------|
|
|
44
|
+
| Tenant + raw datalake | The compliance team's own world |
|
|
45
|
+
| Tokenized + redacted datalakes | A model reads this data — see §007 |
|
|
46
|
+
| SMS tool, LLM tool, Manual Upload | Shared by both scenarios |
|
|
47
|
+
| Atomic FI data source | Where both feeds claim to come from |
|
|
48
|
+
| Sanctions Review AI agent | Disambiguates the gray zone |
|
|
49
|
+
| Two generic tables + contract pairs | Screenings, and payment accounts |
|
|
50
|
+
| Four Data Activation Clients | Identity then data, per pair — see §020 |
|
|
51
|
+
| Two agentic workflows | Verdict fan-out, and activation notice |
|
|
52
|
+
|
|
53
|
+
# Walkthrough
|
|
54
|
+
|
|
55
|
+
This cookbook is **self-reliant**: it stands up everything it uses and
|
|
56
|
+
depends on no other file. It is also **idempotent** — every creating step
|
|
57
|
+
looks first and creates only what is missing, so running it twice costs
|
|
58
|
+
what running it once cost. That is not a nicety. Nothing in the platform
|
|
59
|
+
reclaims an abandoned datalake, and a walk that mints a fresh tenant per
|
|
60
|
+
run leaves every previous run's lakes behind forever.
|
|
61
|
+
|
|
62
|
+
## 001 — sign in as root, and make sure the admin user exists
|
|
63
|
+
|
|
64
|
+
Authenticate as the platform's root admin (`admin@dev.local` /
|
|
65
|
+
`devpassword` in local dev) via the tenantless bootstrap login — keyless by
|
|
66
|
+
structural necessity, since no tenant exists yet to scope a key to, and a
|
|
67
|
+
dev/test-only surface.
|
|
68
|
+
|
|
69
|
+
Then make sure the admin user this walk runs as exists. **The email is
|
|
70
|
+
stable, not per-run**, which is what makes the step idempotent — and it
|
|
71
|
+
means the second run finds the user already signed up. A duplicate signup
|
|
72
|
+
is refused, so the refusal is caught and inspected: if it says the email is
|
|
73
|
+
taken, that is the idempotent path and the walk continues. Any other
|
|
74
|
+
failure is re-raised, because swallowing it would turn a real auth problem
|
|
75
|
+
into a confusing failure three steps later.
|
|
76
|
+
|
|
77
|
+
```typescript
|
|
78
|
+
ctx.rootSession = await createBootstrapSession({
|
|
79
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
80
|
+
email: process.env.ALVERA_ROOT_EMAIL!,
|
|
81
|
+
password: process.env.ALVERA_ROOT_PASSWORD!,
|
|
82
|
+
})
|
|
83
|
+
ctx.rootApi = createIsolatedPlatformApi({
|
|
84
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
85
|
+
sessionToken: ctx.rootSession.sessionToken,
|
|
86
|
+
apiKey: '',
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
ctx.sarahEmail = 'cookbook-payments-compliance@dev.local'
|
|
90
|
+
ctx.sarahPassword = 'CookbookPass1!'
|
|
91
|
+
|
|
92
|
+
try {
|
|
93
|
+
const signUpResp = await ctx.rootApi.admin.signUp({
|
|
94
|
+
email: ctx.sarahEmail,
|
|
95
|
+
password: ctx.sarahPassword,
|
|
96
|
+
first_name: 'Cookbook',
|
|
97
|
+
last_name: 'Compliance',
|
|
98
|
+
})
|
|
99
|
+
await ctx.rootApi.admin.confirmUser(signUpResp.data.id!)
|
|
100
|
+
} catch (err) {
|
|
101
|
+
// Already provisioned by a previous run. Confirm that is what happened
|
|
102
|
+
// rather than assuming it — a genuine signup failure must not be read as
|
|
103
|
+
// "already there".
|
|
104
|
+
const detail = JSON.stringify((err as { errors?: unknown }).errors ?? err)
|
|
105
|
+
if (!/taken|already|exist/i.test(detail)) throw err
|
|
106
|
+
}
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## 002 — the find-or-create helper every later step uses
|
|
110
|
+
|
|
111
|
+
Idempotence is one question asked over and over: *is this already here?*
|
|
112
|
+
Rather than answer it eleven different ways, the walk defines it once.
|
|
113
|
+
|
|
114
|
+
`ensure` takes a label, a lookup and a create. It runs the lookup, returns
|
|
115
|
+
what it finds, and only creates when the lookup comes back empty. The label
|
|
116
|
+
is not decoration — when a run reuses something you did not expect it to,
|
|
117
|
+
the log line naming it is how you find out.
|
|
118
|
+
|
|
119
|
+
The helper lives on `ctx` rather than as a bare function because each
|
|
120
|
+
numbered step compiles into its own `it()` block, so a plain `function`
|
|
121
|
+
here would not be in scope for the steps that call it.
|
|
122
|
+
|
|
123
|
+
```typescript
|
|
124
|
+
ctx.ensure = async <T>(
|
|
125
|
+
label: string,
|
|
126
|
+
find: () => Promise<T | undefined>,
|
|
127
|
+
create: () => Promise<T>,
|
|
128
|
+
): Promise<T> => {
|
|
129
|
+
const existing = await find()
|
|
130
|
+
if (existing !== undefined) {
|
|
131
|
+
console.log(` ↻ reusing ${label}`)
|
|
132
|
+
return existing
|
|
133
|
+
}
|
|
134
|
+
console.log(` + creating ${label}`)
|
|
135
|
+
return await create()
|
|
136
|
+
}
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## 003 — the tenant
|
|
140
|
+
|
|
141
|
+
Sarah signs in without a tenant scope — she may not belong to one yet — and
|
|
142
|
+
the walk finds or creates the compliance tenant by its stable name. The
|
|
143
|
+
server derives the slug; capture it, because every later call is addressed
|
|
144
|
+
by it.
|
|
145
|
+
|
|
146
|
+
```typescript
|
|
147
|
+
ctx.sarahTenantlessSession = await createBootstrapSession({
|
|
148
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
149
|
+
email: ctx.sarahEmail,
|
|
150
|
+
password: ctx.sarahPassword,
|
|
151
|
+
})
|
|
152
|
+
ctx.sarahTenantlessApi = createIsolatedPlatformApi({
|
|
153
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
154
|
+
sessionToken: ctx.sarahTenantlessSession.sessionToken,
|
|
155
|
+
apiKey: '',
|
|
156
|
+
})
|
|
157
|
+
|
|
158
|
+
const TENANT_NAME = 'Cookbook Payments Compliance'
|
|
159
|
+
|
|
160
|
+
const tenant = await ctx.ensure(
|
|
161
|
+
`tenant ${TENANT_NAME}`,
|
|
162
|
+
async () => {
|
|
163
|
+
const { data } = await ctx.sarahTenantlessApi.tenants.list()
|
|
164
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === TENANT_NAME)
|
|
165
|
+
},
|
|
166
|
+
async () => {
|
|
167
|
+
const { data } = await ctx.sarahTenantlessApi.tenants.create({ name: TENANT_NAME })
|
|
168
|
+
return data
|
|
169
|
+
},
|
|
170
|
+
)
|
|
171
|
+
tenantSlug = tenant.slug!
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
## 004 — the tenant-scoped client
|
|
175
|
+
|
|
176
|
+
A tenant-scoped login requires `X-API-Key`, so a key has to exist before
|
|
177
|
+
sarah can sign in against the tenant. Mint one through the platform-admin
|
|
178
|
+
side door with the root bearer; in the web console this is
|
|
179
|
+
*Settings → API Keys*.
|
|
180
|
+
|
|
181
|
+
`data_access_mode: 'raw'` is deliberate. Raw is the superset, and every
|
|
182
|
+
read-back in this walk is a question about **this walk's own data** — not
|
|
183
|
+
about whether replication has caught up. The tokenized and redacted reads
|
|
184
|
+
are a separate subject with their own step and their own keys (§037).
|
|
185
|
+
|
|
186
|
+
**This is the one step in the walk that is not idempotent**, and it is
|
|
187
|
+
worth knowing why rather than discovering it. There is no endpoint that
|
|
188
|
+
lists a tenant's API keys, so there is nothing to look the existing one up
|
|
189
|
+
with — `ensure` has no lookup to run. Each run therefore mints another key.
|
|
190
|
+
A key row is cheap where a datalake is not, so the walk accepts it; if you
|
|
191
|
+
are counting rows in a shared environment, this is the one to count.
|
|
192
|
+
|
|
193
|
+
```typescript
|
|
194
|
+
const { data: mintedKey } = await ctx.rootApi.admin.createTenantApiKey(tenantSlug, {
|
|
195
|
+
name: 'Cookbook Payments Compliance Key',
|
|
196
|
+
data_access_mode: 'raw',
|
|
197
|
+
})
|
|
198
|
+
ctx.tenantApiKey = mintedKey.api_key
|
|
199
|
+
|
|
200
|
+
const sarahTenantSession = await createSession({
|
|
201
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
202
|
+
email: ctx.sarahEmail,
|
|
203
|
+
password: ctx.sarahPassword,
|
|
204
|
+
tenantSlug,
|
|
205
|
+
apiKey: ctx.tenantApiKey,
|
|
206
|
+
})
|
|
207
|
+
api = createIsolatedPlatformApi({
|
|
208
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
209
|
+
sessionToken: sarahTenantSession.sessionToken,
|
|
210
|
+
apiKey: ctx.tenantApiKey,
|
|
211
|
+
})
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
## 005 — the datalake
|
|
215
|
+
|
|
216
|
+
One datalake, created `raw`. A datalake is one database and one bucket; the
|
|
217
|
+
tokenized and redacted copies are separate rows provisioned in §007 and
|
|
218
|
+
filled by replication, never written to directly.
|
|
219
|
+
|
|
220
|
+
The lookup filters on `type === 'raw'`, and that matters from the second
|
|
221
|
+
run onwards: once §007 has run, `datalakes.list` returns three rows, and
|
|
222
|
+
two of them are derived copies that must never be mistaken for the primary.
|
|
223
|
+
|
|
224
|
+
Local-dev defaults match the seeded `dev.exs` setup — `postgres` on
|
|
225
|
+
`localhost:5432`, database `alvera_dev_payments`, LocalStack S3 on
|
|
226
|
+
`localhost:4566` — and the schema name is stable, because a fresh schema
|
|
227
|
+
per run is the same leak as a fresh tenant per run.
|
|
228
|
+
|
|
229
|
+
```typescript
|
|
230
|
+
const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_payments' }
|
|
231
|
+
const DB_SCHEMA = 'cookbook_payments_compliance'
|
|
232
|
+
const LAKE_NAME = 'Cookbook Payments Compliance Datalake'
|
|
233
|
+
|
|
234
|
+
const S3 = {
|
|
235
|
+
cloud_storage_type: 'aws' as const,
|
|
236
|
+
region: 'us-east-1',
|
|
237
|
+
access_key_id: 'test',
|
|
238
|
+
secret_access_key: 'test',
|
|
239
|
+
endpoint: 'http://localhost:4566',
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
const datalake = await ctx.ensure(
|
|
243
|
+
`datalake ${LAKE_NAME}`,
|
|
244
|
+
async () => {
|
|
245
|
+
const { data } = await api.datalakes.list(tenantSlug)
|
|
246
|
+
return (data.data ?? []).find(
|
|
247
|
+
(l: { name?: string; type?: string }) => l.name === LAKE_NAME && l.type === 'raw',
|
|
248
|
+
)
|
|
249
|
+
},
|
|
250
|
+
async () => {
|
|
251
|
+
const { data } = await api.datalakes.create(tenantSlug, {
|
|
252
|
+
name: LAKE_NAME,
|
|
253
|
+
description: 'Payments compliance datalake provisioned by the cookbook doctest.',
|
|
254
|
+
timezone: 'America/New_York',
|
|
255
|
+
pool_size: 3,
|
|
256
|
+
type: 'raw',
|
|
257
|
+
|
|
258
|
+
db_writer_host: DB.host,
|
|
259
|
+
db_writer_port: DB.port,
|
|
260
|
+
db_writer_name: DB.name,
|
|
261
|
+
db_writer_schema: DB_SCHEMA,
|
|
262
|
+
db_writer_auth_method: 'password',
|
|
263
|
+
db_writer_user: DB.user,
|
|
264
|
+
db_writer_pass: DB.pass,
|
|
265
|
+
db_writer_enable_ssl: false,
|
|
266
|
+
db_reader_host: DB.host,
|
|
267
|
+
db_reader_port: DB.port,
|
|
268
|
+
db_reader_name: DB.name,
|
|
269
|
+
db_reader_schema: DB_SCHEMA,
|
|
270
|
+
db_reader_auth_method: 'password',
|
|
271
|
+
db_reader_user: DB.user,
|
|
272
|
+
db_reader_pass: DB.pass,
|
|
273
|
+
db_reader_enable_ssl: false,
|
|
274
|
+
|
|
275
|
+
cloud_storage: { ...S3, bucket: 'alvera-platform-dev', base_path: 'cookbook/payments-compliance' },
|
|
276
|
+
})
|
|
277
|
+
return data
|
|
278
|
+
},
|
|
279
|
+
)
|
|
280
|
+
datalakeSlug = datalake.slug!
|
|
281
|
+
ctx.datalakeId = datalake.id!
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
## 006 — run the migrations, and wait for ready
|
|
285
|
+
|
|
286
|
+
`datalakes.create` persists the row at `status: 'new'`; it does not run the
|
|
287
|
+
schema DDL. Migration is triggered separately so the operator decides when
|
|
288
|
+
the potentially-slow part happens. `datalakes.migrate` enqueues the job and
|
|
289
|
+
returns immediately with `status: 'enqueued'`; the poll after it is what
|
|
290
|
+
waits for the worker.
|
|
291
|
+
|
|
292
|
+
Migrating is safe to repeat, which is what lets this step stay unguarded on
|
|
293
|
+
a second run. The wait exits the moment the lake reports `ready`, so the
|
|
294
|
+
five-minute ceiling is only ever paid in failure.
|
|
295
|
+
|
|
296
|
+
```typescript
|
|
297
|
+
const migrateResp = await api.datalakes.migrate(tenantSlug, datalakeSlug)
|
|
298
|
+
if (migrateResp.data.status !== 'enqueued') {
|
|
299
|
+
throw new Error(`datalake migration not enqueued (status: ${migrateResp.data.status})`)
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
const READY_TIMEOUT_MS = 5 * 60_000
|
|
303
|
+
const deadline = Date.now() + READY_TIMEOUT_MS
|
|
304
|
+
let datalakeStatus: string | undefined
|
|
305
|
+
while (Date.now() < deadline) {
|
|
306
|
+
const { data } = await api.datalakes.get(tenantSlug, ctx.datalakeId)
|
|
307
|
+
datalakeStatus = data.status
|
|
308
|
+
if (datalakeStatus === 'ready') break
|
|
309
|
+
await new Promise((r) => setTimeout(r, 5_000))
|
|
310
|
+
}
|
|
311
|
+
if (datalakeStatus !== 'ready') {
|
|
312
|
+
throw new Error(`datalake did not reach :ready within ${READY_TIMEOUT_MS}ms (last: ${datalakeStatus})`)
|
|
313
|
+
}
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
## 007 — turn tokenization on, because a model is going to read this
|
|
317
|
+
|
|
318
|
+
A datalake is created raw and stays raw. Tokenization is one separate call,
|
|
319
|
+
and it is a decision, not a default: *the tokenized copy is for a machine,
|
|
320
|
+
the redacted copy is for a person*, and you turn them on when one of those
|
|
321
|
+
readers is about to look. Here one is — the agent in §016 reads screenings
|
|
322
|
+
to disambiguate them, and it has no business seeing who was screened.
|
|
323
|
+
|
|
324
|
+
`provisionTokenization` creates both derived lakes and enqueues a migration
|
|
325
|
+
for each. It is **both-or-neither**: a run that provisions the tokenized
|
|
326
|
+
lake and cannot provision the redacted one reports the failure rather than
|
|
327
|
+
leaving half a pair. It is **idempotent**, so this step needs no `ensure`
|
|
328
|
+
around it — already-provisioned lakes come back as they are.
|
|
329
|
+
|
|
330
|
+
**Order matters.** The raw lake must be `ready` before this runs. A derived
|
|
331
|
+
lake is provisioned empty and its own migration is what creates its tables
|
|
332
|
+
and starts the copy; provision too early and you get lakes that exist and
|
|
333
|
+
stay empty forever. That is why §006 waits.
|
|
334
|
+
|
|
335
|
+
```typescript
|
|
336
|
+
const { data: provisioned } = await api.datalakes.provisionTokenization(tenantSlug, datalakeSlug)
|
|
337
|
+
const derivedLakes = provisioned.derived_datalakes ?? []
|
|
338
|
+
if (derivedLakes.length === 0) {
|
|
339
|
+
throw new Error('provisionTokenization returned no derived datalakes')
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
// Each derived lake migrates on its own job. Poll the DERIVED ids — the raw
|
|
343
|
+
// lake's own status says nothing about whether its copies are ready.
|
|
344
|
+
const DERIVED_TIMEOUT_MS = 5 * 60_000
|
|
345
|
+
let pendingLakes = derivedLakes.map((d) => d.id!)
|
|
346
|
+
const derivedDeadline = Date.now() + DERIVED_TIMEOUT_MS
|
|
347
|
+
while (Date.now() < derivedDeadline && pendingLakes.length > 0) {
|
|
348
|
+
const stillPending: string[] = []
|
|
349
|
+
for (const id of pendingLakes) {
|
|
350
|
+
const { data } = await api.datalakes.get(tenantSlug, id)
|
|
351
|
+
if (data.status !== 'ready') stillPending.push(id)
|
|
352
|
+
}
|
|
353
|
+
pendingLakes = stillPending
|
|
354
|
+
if (pendingLakes.length > 0) await new Promise((r) => setTimeout(r, 5_000))
|
|
355
|
+
}
|
|
356
|
+
if (pendingLakes.length > 0) {
|
|
357
|
+
throw new Error(`derived datalakes did not reach :ready: ${pendingLakes.join(', ')}`)
|
|
358
|
+
}
|
|
359
|
+
```
|
|
360
|
+
|
|
361
|
+
## 008 — three helpers the scenarios below share
|
|
362
|
+
|
|
363
|
+
**`waitForFiredRun`.** `workflows.run` only *schedules* a run. It returns
|
|
364
|
+
immediately with a `workflow_run_id`, and the `workflow_run_log_id` and
|
|
365
|
+
`batch_id` a scenario needs are written later, when the run actually fires.
|
|
366
|
+
Two traps live in that gap:
|
|
367
|
+
|
|
368
|
+
- **Poll until `workflow_run_log_id` is a string — not until `status`
|
|
369
|
+
leaves `'scheduled'`.** Those are different moments; the run reaches
|
|
370
|
+
`processing` first and writes the log id a beat later. A predicate on
|
|
371
|
+
status alone releases you to read a `null`, and because
|
|
372
|
+
`typeof null === 'object'` the symptom is a baffling *"expected string,
|
|
373
|
+
got object"* rather than an obvious nil.
|
|
374
|
+
- **Raise on `failed` carrying `failure_reason`** rather than polling to
|
|
375
|
+
the deadline. A scenario blocked on a run that will never fire should say
|
|
376
|
+
why on the first read, not thirty seconds later behind a generic timeout.
|
|
377
|
+
|
|
378
|
+
**`waitForBatches`.** Ingestion is async: `ingest` returns a `batch_id` the
|
|
379
|
+
instant the rows are accepted, and the per-row jobs drain into the lake
|
|
380
|
+
afterwards. This polls a DAC's activation logs until every named batch has
|
|
381
|
+
actually written.
|
|
382
|
+
|
|
383
|
+
**Gate on `dataset_updated`, never on `rows_ingested`.** They answer
|
|
384
|
+
different questions — received versus written — and a batch whose row is
|
|
385
|
+
refused reports `rows_ingested: 1, dataset_updated: 0, status: 'partial'`
|
|
386
|
+
with the reason in `error`. A gate on `rows_ingested` calls that green, the
|
|
387
|
+
next step then runs against a table with nothing in it, and the failure
|
|
388
|
+
surfaces three steps later as *"expected 2 execution logs, got 0"* — which
|
|
389
|
+
reads like a broken workflow rather than a row that never landed. Raising
|
|
390
|
+
here, where the reason is still attached, costs one line.
|
|
391
|
+
|
|
392
|
+
**`deployGenericTable`.** Creating a generic table does not deploy it; it
|
|
393
|
+
rests at `status: 'new'` until a migration runs, and `datalakes.migrate`
|
|
394
|
+
enqueues one job per lake so a single call builds the table in all three.
|
|
395
|
+
|
|
396
|
+
The wait is the part worth getting right, because the obvious one is a
|
|
397
|
+
false green. Polling the derived lakes until they report `ready` returns in
|
|
398
|
+
milliseconds and proves nothing — a derived lake is `ready` whether or not
|
|
399
|
+
any particular table exists in it. Waiting on the table's own
|
|
400
|
+
`status: 'deployed'` is no better; that is the raw lake's answer, and it
|
|
401
|
+
goes true before the copies exist. The only wait that means anything asks
|
|
402
|
+
for the table the same way the next step is about to, and retries until it
|
|
403
|
+
stops erroring.
|
|
404
|
+
|
|
405
|
+
```typescript
|
|
406
|
+
ctx.waitForFiredRun = async (
|
|
407
|
+
runDatalakeSlug: string,
|
|
408
|
+
runId: string,
|
|
409
|
+
timeoutMs = 120_000,
|
|
410
|
+
): Promise<{ workflowRunLogId: string; batchId: string | null }> => {
|
|
411
|
+
const deadline = Date.now() + timeoutMs
|
|
412
|
+
let lastStatus: string | undefined
|
|
413
|
+
while (Date.now() < deadline) {
|
|
414
|
+
const { data } = await api.workflowRuns.get(tenantSlug, runDatalakeSlug, runId)
|
|
415
|
+
lastStatus = data.status
|
|
416
|
+
if (data.status === 'failed') {
|
|
417
|
+
throw new Error(`workflow run ${runId} failed: ${data.failure_reason ?? 'no failure_reason given'}`)
|
|
418
|
+
}
|
|
419
|
+
if (typeof data.workflow_run_log_id === 'string') {
|
|
420
|
+
return { workflowRunLogId: data.workflow_run_log_id, batchId: data.batch_id ?? null }
|
|
421
|
+
}
|
|
422
|
+
await new Promise((r) => setTimeout(r, 1_000))
|
|
423
|
+
}
|
|
424
|
+
throw new Error(`workflow run ${runId} never fired within ${timeoutMs}ms (last status: ${lastStatus})`)
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
ctx.waitForBatches = async (
|
|
428
|
+
dacSlug: string,
|
|
429
|
+
batchIds: readonly string[],
|
|
430
|
+
timeoutMs = 90_000,
|
|
431
|
+
): Promise<void> => {
|
|
432
|
+
const targets = new Set(batchIds)
|
|
433
|
+
const deadline = Date.now() + timeoutMs
|
|
434
|
+
let greenCount = 0
|
|
435
|
+
while (Date.now() < deadline) {
|
|
436
|
+
const { data } = await api.dataActivationClients.logs.list(tenantSlug, datalakeSlug, dacSlug)
|
|
437
|
+
const green = new Set<string>()
|
|
438
|
+
for (const row of (data.data ?? []) as Array<Record<string, unknown>>) {
|
|
439
|
+
const b = row.batch_id
|
|
440
|
+
if (typeof b !== 'string' || !targets.has(b)) continue
|
|
441
|
+
if (row.status === 'partial' || row.status === 'failed') {
|
|
442
|
+
throw new Error(
|
|
443
|
+
`batch ${b} on ${dacSlug} did not persist its rows (status: ${row.status}): ` +
|
|
444
|
+
`${String(row.error ?? 'no reason given')}`,
|
|
445
|
+
)
|
|
446
|
+
}
|
|
447
|
+
if (typeof row.dataset_updated !== 'number' || row.dataset_updated < 1) continue
|
|
448
|
+
const files = row.output_files
|
|
449
|
+
if (!Array.isArray(files) || files.length === 0) continue
|
|
450
|
+
green.add(b)
|
|
451
|
+
}
|
|
452
|
+
greenCount = green.size
|
|
453
|
+
if (greenCount === targets.size) return
|
|
454
|
+
await new Promise((r) => setTimeout(r, 1_000))
|
|
455
|
+
}
|
|
456
|
+
throw new Error(`only ${greenCount}/${targets.size} batches on ${dacSlug} persisted within ${timeoutMs}ms`)
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
ctx.deployGenericTable = async (tableName: string, timeoutMs = 120_000): Promise<void> => {
|
|
460
|
+
await api.datalakes.migrate(tenantSlug, datalakeSlug)
|
|
461
|
+
const deadline = Date.now() + timeoutMs
|
|
462
|
+
let lastError: unknown
|
|
463
|
+
while (Date.now() < deadline) {
|
|
464
|
+
try {
|
|
465
|
+
await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
|
|
466
|
+
sql: `SELECT 1 FROM ${tableName} LIMIT 1`,
|
|
467
|
+
mode: 'raw',
|
|
468
|
+
})
|
|
469
|
+
return
|
|
470
|
+
} catch (err) {
|
|
471
|
+
lastError = err
|
|
472
|
+
await new Promise((r) => setTimeout(r, 1_000))
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
throw new Error(
|
|
476
|
+
`generic table ${tableName} was not readable within ${timeoutMs}ms ` +
|
|
477
|
+
`(last error: ${lastError instanceof Error ? lastError.message : String(lastError)})`,
|
|
478
|
+
)
|
|
479
|
+
}
|
|
480
|
+
```
|
|
481
|
+
|
|
482
|
+
## 009 — the SMS tool
|
|
483
|
+
|
|
484
|
+
Both scenarios send SMS, so the tool is created once and shared.
|
|
485
|
+
`body.tool_body_type: 'sns'` routes via AWS SNS; local dev points it at
|
|
486
|
+
LocalStack on `http://localhost:4566` through `endpoint_url`, so no real AWS
|
|
487
|
+
credentials are needed. `intent: 'sms'` tags the tool for workflow actions
|
|
488
|
+
that send SMS — as opposed to `data_exchange`, which is what an ingestion
|
|
489
|
+
tool carries.
|
|
490
|
+
|
|
491
|
+
```typescript
|
|
492
|
+
const smsTool = await ctx.ensure(
|
|
493
|
+
'SMS tool',
|
|
494
|
+
async () => {
|
|
495
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
496
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Compliance SMS Tool')
|
|
497
|
+
},
|
|
498
|
+
async () => {
|
|
499
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
500
|
+
name: 'Cookbook Compliance SMS Tool',
|
|
501
|
+
description: 'SNS-backed SMS dispatcher for both compliance workflows, wired to LocalStack.',
|
|
502
|
+
intent: 'sms',
|
|
503
|
+
status: 'active',
|
|
504
|
+
datalake_id: ctx.datalakeId,
|
|
505
|
+
body: {
|
|
506
|
+
tool_body_type: 'sns',
|
|
507
|
+
auth_method: 'access_key',
|
|
508
|
+
region: 'us-east-1',
|
|
509
|
+
phone_number: '+15551234567',
|
|
510
|
+
endpoint_url: 'http://localhost:4566',
|
|
511
|
+
access_key_id: 'test',
|
|
512
|
+
secret_access_key: 'test',
|
|
513
|
+
},
|
|
514
|
+
})
|
|
515
|
+
return data
|
|
516
|
+
},
|
|
517
|
+
)
|
|
518
|
+
toolId = smsTool.id!
|
|
519
|
+
```
|
|
520
|
+
|
|
521
|
+
## 010 — the LLM tool
|
|
522
|
+
|
|
523
|
+
The sanctions agent calls a chat-completion endpoint to disambiguate each
|
|
524
|
+
gray-zone screening. `intent: 'llm_enrichment'` distinguishes it from the
|
|
525
|
+
SMS tool.
|
|
526
|
+
|
|
527
|
+
It is a **provider adapter**, and both halves matter. `base_body` authors
|
|
528
|
+
the provider's request — here Ollama's native `/api/chat` shape, with
|
|
529
|
+
`think: false` and a `format` schema so the model returns clean,
|
|
530
|
+
schema-constrained JSON. `response_extractor` maps the provider's envelope
|
|
531
|
+
back to the canonical `{ output_json, … }` the platform reads, and its
|
|
532
|
+
`output_schema` is **required** for an `llm_enrichment` tool. The
|
|
533
|
+
`api_key`/`auth_method` pair satisfies the REST tool's schema even though
|
|
534
|
+
Ollama ignores the header.
|
|
535
|
+
|
|
536
|
+
```typescript
|
|
537
|
+
const ENRICHMENT_OUTPUT_SCHEMA = {
|
|
538
|
+
type: 'object',
|
|
539
|
+
properties: {
|
|
540
|
+
output_json: {},
|
|
541
|
+
input_tokens: { type: ['integer', 'null'] },
|
|
542
|
+
output_tokens: { type: ['integer', 'null'] },
|
|
543
|
+
total_tokens: { type: ['integer', 'null'] },
|
|
544
|
+
explanation: { type: ['string', 'null'] },
|
|
545
|
+
},
|
|
546
|
+
required: ['output_json'],
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
const OLLAMA_BASE_BODY =
|
|
550
|
+
'{"model": "{{ model }}", "messages": [{"role": "user", "content": "{{ rendered_prompt | json_escape }}", "images": [{% for img in images %}{% unless forloop.first %}, {% endunless %}"{{ img.data }}"{% endfor %}]}], "stream": false, "think": false, "options": {"temperature": {{ temperature }}, "num_predict": {{ max_tokens }}, "num_ctx": 40960}, "format": {{ schema | to_json }}}'
|
|
551
|
+
|
|
552
|
+
const OLLAMA_EXTRACTOR =
|
|
553
|
+
'{"output_json": "{{ msg.message.content | json_escape }}", "explanation": "{{ msg.message.thinking | json_escape }}", "input_tokens": {{ msg.prompt_eval_count | default: 0 }}, "output_tokens": {{ msg.eval_count | default: 0 }}, "total_tokens": {{ msg.prompt_eval_count | default: 0 | plus: msg.eval_count }}}'
|
|
554
|
+
|
|
555
|
+
const llmTool = await ctx.ensure(
|
|
556
|
+
'LLM tool',
|
|
557
|
+
async () => {
|
|
558
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
559
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Compliance LLM Tool')
|
|
560
|
+
},
|
|
561
|
+
async () => {
|
|
562
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
563
|
+
name: 'Cookbook Compliance LLM Tool',
|
|
564
|
+
description: 'Ollama-backed chat-completion adapter for sanctions-screening disambiguation.',
|
|
565
|
+
intent: 'llm_enrichment',
|
|
566
|
+
status: 'active',
|
|
567
|
+
datalake_id: ctx.datalakeId,
|
|
568
|
+
response_extractor: { type: 'custom', body: OLLAMA_EXTRACTOR, output_schema: ENRICHMENT_OUTPUT_SCHEMA },
|
|
569
|
+
body: {
|
|
570
|
+
tool_body_type: 'rest_api',
|
|
571
|
+
base_url: 'http://localhost:11434',
|
|
572
|
+
base_path: { type: 'custom', body: '/api/chat' },
|
|
573
|
+
auth_method: 'api_key',
|
|
574
|
+
api_key: 'stub-key',
|
|
575
|
+
api_key_name: 'Authorization',
|
|
576
|
+
api_key_location: 'header',
|
|
577
|
+
request_type: 'json',
|
|
578
|
+
response_type: 'json',
|
|
579
|
+
timeout_ms: 60_000,
|
|
580
|
+
base_body: { type: 'custom', body: OLLAMA_BASE_BODY },
|
|
581
|
+
},
|
|
582
|
+
})
|
|
583
|
+
return data
|
|
584
|
+
},
|
|
585
|
+
)
|
|
586
|
+
ctx.llmToolId = llmTool.id!
|
|
587
|
+
```
|
|
588
|
+
|
|
589
|
+
## 011 — the Atomic FI data source
|
|
590
|
+
|
|
591
|
+
A workflow runs on rows, and rows arrive through the data-activation chain.
|
|
592
|
+
The chain's first link is a `DataSource` — a registration of where the rows
|
|
593
|
+
originate. The `uri` is the system-of-record address, and it flows into each
|
|
594
|
+
ingested row as `source_uri`, which is also what the MDM template in §019
|
|
595
|
+
reads to build its identifier. Both feeds in this walk claim the same
|
|
596
|
+
origin, so they share one source and converge on one subject.
|
|
597
|
+
|
|
598
|
+
```typescript
|
|
599
|
+
const dataSource = await ctx.ensure(
|
|
600
|
+
'Atomic FI data source',
|
|
601
|
+
async () => {
|
|
602
|
+
const { data } = await api.dataSources.list(tenantSlug, datalakeSlug)
|
|
603
|
+
return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Atomic FI Source')
|
|
604
|
+
},
|
|
605
|
+
async () => {
|
|
606
|
+
const { data } = await api.dataSources.create(tenantSlug, datalakeSlug, {
|
|
607
|
+
name: 'Cookbook Atomic FI Source',
|
|
608
|
+
uri: 'api.atomic.fi/compliance',
|
|
609
|
+
description: 'Atomic FI compliance API — origin of the screening and payment-account rows.',
|
|
610
|
+
status: 'active',
|
|
611
|
+
is_default: false,
|
|
612
|
+
})
|
|
613
|
+
return data
|
|
614
|
+
},
|
|
615
|
+
)
|
|
616
|
+
dataSourceId = dataSource.id!
|
|
617
|
+
```
|
|
618
|
+
|
|
619
|
+
## 012 — the Manual Upload tool
|
|
620
|
+
|
|
621
|
+
A Data Activation Client needs a tool. For inline-JSON ingest a
|
|
622
|
+
`manual_upload` tool is the minimal choice: `tool_body_type:
|
|
623
|
+
'manual_upload'` needs no endpoint or credential wiring, because the rows
|
|
624
|
+
arrive in the ingest call's body rather than by the tool fetching them.
|
|
625
|
+
|
|
626
|
+
```typescript
|
|
627
|
+
const manualUploadTool = await ctx.ensure(
|
|
628
|
+
'Manual Upload tool',
|
|
629
|
+
async () => {
|
|
630
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
631
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Manual Upload Tool')
|
|
632
|
+
},
|
|
633
|
+
async () => {
|
|
634
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
635
|
+
name: 'Cookbook Manual Upload Tool',
|
|
636
|
+
description: 'Manual-upload data-exchange tool — backs both DACs in this walk.',
|
|
637
|
+
intent: 'data_exchange',
|
|
638
|
+
status: 'active',
|
|
639
|
+
datalake_id: ctx.datalakeId,
|
|
640
|
+
data_source_id: dataSourceId,
|
|
641
|
+
body: { tool_body_type: 'manual_upload' },
|
|
642
|
+
})
|
|
643
|
+
return data
|
|
644
|
+
},
|
|
645
|
+
)
|
|
646
|
+
ctx.manualUploadToolId = manualUploadTool.id!
|
|
647
|
+
```
|
|
648
|
+
|
|
649
|
+
## 013 — what the platform already ships: the system-template catalog
|
|
650
|
+
|
|
651
|
+
Before hand-writing Liquid, look at what is already there. `systemTemplates`
|
|
652
|
+
returns every platform-shipped template visible to this datalake — each with
|
|
653
|
+
a `path`, its Liquid `content`, and an `output_schema` (`null` when the
|
|
654
|
+
template ships no companion schema).
|
|
655
|
+
|
|
656
|
+
**The catalog is flat.** It used to be filtered by the datalake's data
|
|
657
|
+
domain, so a payments template never appeared in a subscription lake;
|
|
658
|
+
GH-859 removed domains and with them the filter. What one datalake can see,
|
|
659
|
+
every datalake can see. The assertion below is the positive form of that
|
|
660
|
+
fact: no path carries a retired vertical segment.
|
|
661
|
+
|
|
662
|
+
```typescript
|
|
663
|
+
const { data: templates } = await api.templates.systemTemplates(tenantSlug, datalakeSlug)
|
|
664
|
+
|
|
665
|
+
const paths = (templates.data ?? []).map((t) => t.path)
|
|
666
|
+
if (paths.length === 0) {
|
|
667
|
+
throw new Error('expected the platform to ship at least one system template')
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
const anchor = 'data_activation/interoperability/stripe/customers_stripe_legal_entity'
|
|
671
|
+
if (!paths.some((p) => p.includes(anchor))) {
|
|
672
|
+
throw new Error(`expected the Stripe customer template (${anchor}) in the list`)
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
const RETIRED_DOMAINS = ['healthcare', 'payments', 'foundation', 'core_banking', 'service_commerce', 'trading']
|
|
676
|
+
const stale = paths.filter((p) => RETIRED_DOMAINS.some((d) => p.split('/').includes(d)))
|
|
677
|
+
if (stale.length > 0) {
|
|
678
|
+
throw new Error(`template paths still carry a retired data-domain segment: ${stale.join(', ')}`)
|
|
679
|
+
}
|
|
680
|
+
```
|
|
681
|
+
|
|
682
|
+
## 014 — the same catalog as markdown, and the page you will forget to read
|
|
683
|
+
|
|
684
|
+
`metadata` returns the catalog as a markdown document — the form an agent
|
|
685
|
+
reads to choose a template. It inlines each template's full Liquid source,
|
|
686
|
+
so it is long.
|
|
687
|
+
|
|
688
|
+
**It paginates, and page 1 is not the catalog.** The document opens with
|
|
689
|
+
`Showing page 1 of N — M templates total`, and the template you want is as
|
|
690
|
+
likely to be on page 3. Reading page 1 and concluding a template does not
|
|
691
|
+
exist is the trap, and it used to be hidden: the per-domain filter made the
|
|
692
|
+
list short enough to fit one page, and GH-859 removed that filter.
|
|
693
|
+
|
|
694
|
+
```typescript
|
|
695
|
+
let found = false
|
|
696
|
+
let page = 1
|
|
697
|
+
let pages = 1
|
|
698
|
+
do {
|
|
699
|
+
const { data: catalog } = await api.templates.metadata(tenantSlug, datalakeSlug, { page })
|
|
700
|
+
if (typeof catalog !== 'string' || catalog.length === 0) {
|
|
701
|
+
throw new Error(`expected a non-empty markdown template catalog on page ${page}`)
|
|
702
|
+
}
|
|
703
|
+
const header = catalog.match(/Showing page \d+ of (\d+)/)
|
|
704
|
+
if (header) pages = Number(header[1])
|
|
705
|
+
if (catalog.includes('customers_stripe_legal_entity')) found = true
|
|
706
|
+
page += 1
|
|
707
|
+
} while (!found && page <= pages)
|
|
708
|
+
|
|
709
|
+
if (!found) {
|
|
710
|
+
throw new Error(`expected the Stripe customer template in the catalog markdown (searched ${pages} page(s))`)
|
|
711
|
+
}
|
|
712
|
+
```
|
|
713
|
+
|
|
714
|
+
## 015 — one template's detail, by basename and intent
|
|
715
|
+
|
|
716
|
+
`metadataDetails` pulls a single template. Address it by its **basename**
|
|
717
|
+
(the template name without the path) and its **intent** (the path prefix
|
|
718
|
+
joined with underscores — here `data_activation_interoperability`), not by
|
|
719
|
+
the full path. The result is a markdown body to show an operator, or to feed
|
|
720
|
+
an agent, before copying the template into a contract's `template_config`.
|
|
721
|
+
|
|
722
|
+
```typescript
|
|
723
|
+
const { data: detail } = await api.templates.metadataDetails(
|
|
724
|
+
tenantSlug,
|
|
725
|
+
datalakeSlug,
|
|
726
|
+
'customers_stripe_legal_entity',
|
|
727
|
+
'data_activation_interoperability',
|
|
728
|
+
)
|
|
729
|
+
if (typeof detail !== 'string' || detail.length === 0) {
|
|
730
|
+
throw new Error('expected a non-empty markdown detail for the Stripe customer template')
|
|
731
|
+
}
|
|
732
|
+
```
|
|
733
|
+
## 016 — the Sanctions Review agent
|
|
734
|
+
|
|
735
|
+
The agent binds three things: the model, an input schema the workflow's
|
|
736
|
+
context-mapping must satisfy, and a response schema its output must match.
|
|
737
|
+
The response schema's `enum: ['verified','blocked']` is a guard — it stops
|
|
738
|
+
the agent emitting a verdict the workflow has no action for.
|
|
739
|
+
`temperature: 0.0` removes sampling noise so the walk is deterministic.
|
|
740
|
+
|
|
741
|
+
`data_access: 'raw'` is the interesting decision, and it is deliberately
|
|
742
|
+
the opposite of what §007 might lead you to expect. `screened_entity_name`
|
|
743
|
+
is a `tokenize` column, and this agent's whole job is to judge whether a
|
|
744
|
+
name is a true match or a coincidence — a token tells it nothing. So this
|
|
745
|
+
agent reads raw. §037 shows the reader for whom the tokenized copy is the
|
|
746
|
+
right answer. **Tokenization is a per-reader decision, not a lake-wide
|
|
747
|
+
setting**, and an agent that does not need plaintext should not ask for it.
|
|
748
|
+
|
|
749
|
+
```typescript
|
|
750
|
+
const AGENT_INPUT_SCHEMA = {
|
|
751
|
+
type: 'object',
|
|
752
|
+
properties: {
|
|
753
|
+
screening_id: { type: 'string' },
|
|
754
|
+
screened_entity_name: { type: 'string' },
|
|
755
|
+
match_score: { type: 'number' },
|
|
756
|
+
},
|
|
757
|
+
required: ['screening_id', 'screened_entity_name', 'match_score'],
|
|
758
|
+
}
|
|
759
|
+
|
|
760
|
+
const AGENT_RESPONSE_SCHEMA = {
|
|
761
|
+
type: 'object',
|
|
762
|
+
properties: {
|
|
763
|
+
verdict: { type: 'string', enum: ['verified', 'blocked'] },
|
|
764
|
+
comments: { type: 'string' },
|
|
765
|
+
},
|
|
766
|
+
required: ['verdict', 'comments'],
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
const AGENT_PROMPT_BODY = `You are a sanctions-review disambiguation assistant. A name-match against a sanctions list has come in with a gray-zone score (close enough to be plausible, ambiguous enough to need a human-style judgement call). Decide whether this is a true match ("blocked") or a false positive on a similar name ("verified").
|
|
770
|
+
|
|
771
|
+
Screening id: {{ screening_id }}
|
|
772
|
+
Screened entity name: {{ screened_entity_name }}
|
|
773
|
+
Match score: {{ match_score }}
|
|
774
|
+
|
|
775
|
+
Respond with JSON: {"verdict": "verified" | "blocked", "comments": "<one-sentence rationale>"}
|
|
776
|
+
|
|
777
|
+
For this cookbook fixture, the entity name is a deliberately disambiguating non-sanctioned identity — respond with "verified".`
|
|
778
|
+
|
|
779
|
+
const agent = await ctx.ensure(
|
|
780
|
+
'Sanctions Review agent',
|
|
781
|
+
async () => {
|
|
782
|
+
const { data } = await api.aiAgents.list(tenantSlug, datalakeSlug)
|
|
783
|
+
return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Sanctions Review Agent')
|
|
784
|
+
},
|
|
785
|
+
async () => {
|
|
786
|
+
const { data } = await api.aiAgents.create(tenantSlug, datalakeSlug, {
|
|
787
|
+
name: 'Cookbook Sanctions Review Agent',
|
|
788
|
+
tool_id: ctx.llmToolId,
|
|
789
|
+
model: 'qwen3-vl:8b-instruct',
|
|
790
|
+
data_access: 'raw',
|
|
791
|
+
temperature: 0.0,
|
|
792
|
+
max_tokens: 256,
|
|
793
|
+
enabled: true,
|
|
794
|
+
input_schema: AGENT_INPUT_SCHEMA,
|
|
795
|
+
llm_response_schema: AGENT_RESPONSE_SCHEMA,
|
|
796
|
+
prompt_config: { type: 'custom', body: AGENT_PROMPT_BODY },
|
|
797
|
+
})
|
|
798
|
+
return data
|
|
799
|
+
},
|
|
800
|
+
)
|
|
801
|
+
aiAgentId = agent.id!
|
|
802
|
+
ctx.agentSlug = agent.slug!
|
|
803
|
+
```
|
|
804
|
+
|
|
805
|
+
## 017 — the compliance-screenings table
|
|
806
|
+
|
|
807
|
+
The workflow needs an event source. GH-859 removed the
|
|
808
|
+
`compliance_screening` dataset along with every other per-domain resource,
|
|
809
|
+
so a screening is a table you declare.
|
|
810
|
+
|
|
811
|
+
Four column decisions carry this walk.
|
|
812
|
+
|
|
813
|
+
`screening_score` is a **float**, not a string. The gray-zone filter
|
|
814
|
+
compares it with `>=` and `<`; a string column compares lexically and the
|
|
815
|
+
band silently stops meaning anything.
|
|
816
|
+
|
|
817
|
+
`screened_entity_name` is `tokenize` and `review_notes` is `redact`. That
|
|
818
|
+
declaration **is** the masking policy — there is no changeset doing it on
|
|
819
|
+
the way past any more, and no second schema to write into. `privacy_requirement`
|
|
820
|
+
is a floor, not an instruction: a `tokenize` column is tokenized in the
|
|
821
|
+
tokenized copy and redacted in the redacted one.
|
|
822
|
+
|
|
823
|
+
`sanctions_matches` is `jsonb` **and `is_array: true`**. The second half is
|
|
824
|
+
not optional and its absence does not look like a type error. `is_array`
|
|
825
|
+
defaults to `false`, which compiles the column's schema as a single map, so
|
|
826
|
+
ingesting a JSON array into it is refused per row —
|
|
827
|
+
`upsert_row_failed %{sanctions_matches: ["is invalid"]}` — while the batch
|
|
828
|
+
still reports a row received.
|
|
829
|
+
|
|
830
|
+
Do **not** declare a `legal_entity_id` column. The platform stamps it from
|
|
831
|
+
Contract B's MDM resolve in §019, and the name is reserved — declaring it
|
|
832
|
+
does not shadow the stamp silently, it 422s the create.
|
|
833
|
+
|
|
834
|
+
```typescript
|
|
835
|
+
const screenings = await ctx.ensure(
|
|
836
|
+
'compliance-screenings table',
|
|
837
|
+
async () => {
|
|
838
|
+
const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
|
|
839
|
+
return (data.data ?? []).find((t: { title?: string }) => t.title === 'Compliance Screenings')
|
|
840
|
+
},
|
|
841
|
+
async () => {
|
|
842
|
+
const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
|
|
843
|
+
title: 'Compliance Screenings',
|
|
844
|
+
description: 'Sanctions/AML screenings loaded from the Atomic FI API',
|
|
845
|
+
columns: [
|
|
846
|
+
{ name: 'compliance_screening_number', title: 'Screening Number', type: 'string', description: "The provider's screening id — unique per screening", is_unique: true, privacy_requirement: 'none' },
|
|
847
|
+
{ name: 'account_holder_ref', title: 'Account Holder Ref', type: 'string', description: "The provider's account-holder id — the identifier MDM resolves on", is_unique: false, privacy_requirement: 'none' },
|
|
848
|
+
{ name: 'scope', title: 'Scope', type: 'string', description: 'What the screening was run against', is_unique: false, privacy_requirement: 'none' },
|
|
849
|
+
{ name: 'screening_type', title: 'Screening Type', type: 'string', description: 'sanctions / pep / adverse_media', is_unique: false, privacy_requirement: 'none' },
|
|
850
|
+
{ name: 'screening_status', title: 'Screening Status', type: 'string', description: 'pending / pass / fail', is_unique: false, privacy_requirement: 'none' },
|
|
851
|
+
{ name: 'screened_entity_type', title: 'Screened Entity Type', type: 'string', description: 'individual / business', is_unique: false, privacy_requirement: 'none' },
|
|
852
|
+
{ name: 'screening_score', title: 'Screening Score', type: 'float', description: 'Match confidence 0-1 — the gray-zone band compares on it numerically', is_unique: false, privacy_requirement: 'none' },
|
|
853
|
+
{ name: 'match_count', title: 'Match Count', type: 'integer', description: 'How many list entries matched', is_unique: false, privacy_requirement: 'none' },
|
|
854
|
+
{ name: 'sanctions_screening_status', title: 'Sanctions Status', type: 'string', description: 'match / cleared', is_unique: false, privacy_requirement: 'none' },
|
|
855
|
+
{ name: 'screened_entity_name', title: 'Screened Entity Name', type: 'string', description: 'Name that was screened — the agent disambiguates on it', is_unique: false, privacy_requirement: 'tokenize' },
|
|
856
|
+
{ name: 'review_notes', title: 'Review Notes', type: 'string', description: 'Free-text reviewer narrative', is_unique: false, privacy_requirement: 'redact' },
|
|
857
|
+
{ name: 'pep_list_name', title: 'PEP List Name', type: 'string', description: 'Politically-exposed-person list the entity appears on', is_unique: false, privacy_requirement: 'tokenize' },
|
|
858
|
+
{ name: 'aml_risk_score', title: 'AML Risk Score', type: 'float', description: 'Analytic AML risk score', is_unique: false, privacy_requirement: 'none' },
|
|
859
|
+
{ name: 'aml_velocity_count', title: 'AML Velocity Count', type: 'integer', description: 'Transaction-velocity signal', is_unique: false, privacy_requirement: 'none' },
|
|
860
|
+
{ name: 'aml_high_risk_country', title: 'AML High Risk Country', type: 'string', description: 'High-risk jurisdiction flag', is_unique: false, privacy_requirement: 'none' },
|
|
861
|
+
{ name: 'sanctions_matches', title: 'Sanctions Matches', type: 'jsonb', description: 'The matched list entries, as a JSON array', is_unique: false, is_array: true, privacy_requirement: 'redact' },
|
|
862
|
+
],
|
|
863
|
+
})
|
|
864
|
+
return data
|
|
865
|
+
},
|
|
866
|
+
)
|
|
867
|
+
ctx.screeningsTableId = screenings.id!
|
|
868
|
+
ctx.screeningsTableName = screenings.name!
|
|
869
|
+
|
|
870
|
+
// Creating a table records it; deploying it is a migration. Called
|
|
871
|
+
// unconditionally rather than only on the create branch, so a run that
|
|
872
|
+
// crashed between the two still converges.
|
|
873
|
+
await ctx.deployGenericTable(ctx.screeningsTableName)
|
|
874
|
+
```
|
|
875
|
+
|
|
876
|
+
## 018 — the Sanctions Review workflow
|
|
877
|
+
|
|
878
|
+
Filter, decision, actions — plus the agent **nested** in the create body
|
|
879
|
+
via `workflow_ai_agents` (a `cast_assoc`, not a separate attach call).
|
|
880
|
+
Three pieces are worth stopping on.
|
|
881
|
+
|
|
882
|
+
The **filter** is the gray-zone band: it passes only screenings scoring at
|
|
883
|
+
least `0.80` and below `0.95`. Below `0.80` is an auto-clear and at or
|
|
884
|
+
above `0.95` is an auto-block — neither needs a model, so the filter
|
|
885
|
+
rejects them before any LLM call is paid for.
|
|
886
|
+
|
|
887
|
+
The **context_mapping** projects each screening into the agent's input
|
|
888
|
+
schema. `match_score` is interpolated *without* surrounding quotes so it
|
|
889
|
+
renders as a raw JSON number; the input schema declares it `number`, and a
|
|
890
|
+
quoted `"0.88"` fails validation.
|
|
891
|
+
|
|
892
|
+
The **decision_config** interpolates the agent's `verdict` — read from
|
|
893
|
+
`additional_context["<agent-slug>"]` with bracket access, because the slug
|
|
894
|
+
contains hyphens — into a single-element decision array. The two actions
|
|
895
|
+
are keyed `decision_key: verified|blocked`; whichever verdict the agent
|
|
896
|
+
emits picks the action that fires and leaves the other `:skipped`.
|
|
897
|
+
|
|
898
|
+
```typescript
|
|
899
|
+
const VERDICTS = ['verified', 'blocked'] as const
|
|
900
|
+
|
|
901
|
+
const GRAY_ZONE_FILTER =
|
|
902
|
+
'{% if event_dataset.screening_score >= 0.80 ' +
|
|
903
|
+
'and event_dataset.screening_score < 0.95 %}true{% endif %}'
|
|
904
|
+
|
|
905
|
+
const CONTEXT_MAPPING_BODY =
|
|
906
|
+
'{' +
|
|
907
|
+
'"screening_id": "{{ event_dataset.compliance_screening_number }}",' +
|
|
908
|
+
'"screened_entity_name": "{{ event_dataset.screened_entity_name }}",' +
|
|
909
|
+
'"match_score": {{ event_dataset.screening_score }}' +
|
|
910
|
+
'}'
|
|
911
|
+
|
|
912
|
+
const DECISION_CONFIG_BODY = `["{{ additional_context["${ctx.agentSlug}"].verdict }}"]`
|
|
913
|
+
const DECISION_OUTPUT_SCHEMA = { type: 'array', items: { type: 'string' } }
|
|
914
|
+
|
|
915
|
+
const workflow = await ctx.ensure(
|
|
916
|
+
'Sanctions Review workflow',
|
|
917
|
+
async () => {
|
|
918
|
+
const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
|
|
919
|
+
return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Sanctions Review Workflow')
|
|
920
|
+
},
|
|
921
|
+
async () => {
|
|
922
|
+
const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
|
|
923
|
+
name: 'Cookbook Sanctions Review Workflow',
|
|
924
|
+
description: 'Runs an LLM disambiguation pass on gray-zone sanctions matches and fires a verdict-keyed SMS.',
|
|
925
|
+
dataset_type: 'generic_table',
|
|
926
|
+
generic_table_id: ctx.screeningsTableId,
|
|
927
|
+
status: 'live',
|
|
928
|
+
tags: ['compliance', 'sanctions'],
|
|
929
|
+
skip_mdm_resolution: false,
|
|
930
|
+
filter_config: { type: 'custom', body: GRAY_ZONE_FILTER },
|
|
931
|
+
decision_config: {
|
|
932
|
+
type: 'custom',
|
|
933
|
+
body: DECISION_CONFIG_BODY,
|
|
934
|
+
output_schema: DECISION_OUTPUT_SCHEMA,
|
|
935
|
+
},
|
|
936
|
+
actions: VERDICTS.map((verdict) => ({
|
|
937
|
+
decision_key: verdict,
|
|
938
|
+
action_type: 'sms',
|
|
939
|
+
tool_id: toolId,
|
|
940
|
+
position: 0,
|
|
941
|
+
trigger_template: 'now',
|
|
942
|
+
idempotency_template: `{{ event_dataset.compliance_screening_number }}-${verdict}`,
|
|
943
|
+
tool_call: {
|
|
944
|
+
tool_call_type: 'sms_request',
|
|
945
|
+
to: { type: 'custom', body: '+15550000000' },
|
|
946
|
+
body: {
|
|
947
|
+
type: 'custom',
|
|
948
|
+
body: `[${verdict}] {{ additional_context["${ctx.agentSlug}"].comments }}`,
|
|
949
|
+
},
|
|
950
|
+
sms_type: 'transactional',
|
|
951
|
+
},
|
|
952
|
+
})),
|
|
953
|
+
// Workflow joins omit output_schema — the server pins it from the
|
|
954
|
+
// agent's own input_schema.
|
|
955
|
+
workflow_ai_agents: [
|
|
956
|
+
{
|
|
957
|
+
ai_agent_id: aiAgentId,
|
|
958
|
+
position: 0,
|
|
959
|
+
context_mapping_config: { type: 'custom', body: CONTEXT_MAPPING_BODY },
|
|
960
|
+
},
|
|
961
|
+
],
|
|
962
|
+
})
|
|
963
|
+
return data
|
|
964
|
+
},
|
|
965
|
+
)
|
|
966
|
+
workflowId = workflow.id!
|
|
967
|
+
ctx.workflowSlug = workflow.slug!
|
|
968
|
+
```
|
|
969
|
+
|
|
970
|
+
## 019 — the Atomic FI contract pair
|
|
971
|
+
|
|
972
|
+
An inbound screening row becomes two things, so it takes two contracts.
|
|
973
|
+
They are defined together here and deliberately **not** bound to one client
|
|
974
|
+
— §020 is where that matters, and why.
|
|
975
|
+
|
|
976
|
+
**Contract A** (`resource_type: 'legal_entity'`) writes the account holder
|
|
977
|
+
the screening was run against, keyed on Atomic FI's `account_holder_id`.
|
|
978
|
+
Its `mdm_input_config` is `{ type: 'null' }` — a template that already *is*
|
|
979
|
+
the subject has nothing to resolve.
|
|
980
|
+
|
|
981
|
+
**Contract B** (`resource_type: 'generic_table'`, pinned to §017's table)
|
|
982
|
+
writes the screening row, and its `mdm_input_config` emits the same
|
|
983
|
+
`(uri, account_holder_id)` pair — collapsing both onto one legal entity and
|
|
984
|
+
stamping `legal_entity_id` onto the screening.
|
|
985
|
+
|
|
986
|
+
**The identifier shape is where this goes wrong, and there are three of
|
|
987
|
+
them on this platform.** `Platform.MDMInput` — what a contract's
|
|
988
|
+
`mdm_input_config` renders — takes `uri` / `type` / `value`. A LegalEntity
|
|
989
|
+
identification, which is Contract A's own body, takes `id_type` /
|
|
990
|
+
`id_number` / `uri`. And `POST /mdm/verify` takes `system` / `id_type` /
|
|
991
|
+
`value`.
|
|
992
|
+
|
|
993
|
+
The overlap between them is the trap, and it is not symmetric. `system` is a
|
|
994
|
+
real alias: `MDMInput` casts it and normalises it onto `uri`, so that one
|
|
995
|
+
name is safe in both places. `id_type` and `id_number` are not — an MDM
|
|
996
|
+
identifier casts neither. Unknown keys are dropped rather than refused, so a
|
|
997
|
+
template that reaches for the LegalEntity names on the MDM side resolves
|
|
998
|
+
nothing and fails as a bare `mdm_dispatcher_error` with no field named.
|
|
999
|
+
|
|
1000
|
+
And the reason the two look interchangeable is that **one becomes the
|
|
1001
|
+
other**. `MDMInput.to_identification/1` takes `(uri, type, value)` and
|
|
1002
|
+
returns `(uri, id_type, id_number)` — the rename happens inside the platform,
|
|
1003
|
+
on the way through. So the identifier reads back under names it will not
|
|
1004
|
+
accept on the way in, and that asymmetry is invisible from the stored shape
|
|
1005
|
+
alone. Do not reach for `/mdm/verify` to check the shape
|
|
1006
|
+
either: it requires a `legal_entity_id` and verifies attributes against an
|
|
1007
|
+
entity that already exists — it is not a resolver and not a linter.
|
|
1008
|
+
|
|
1009
|
+
The templates are loaded from the vendored fixtures rather than inlined.
|
|
1010
|
+
|
|
1011
|
+
```typescript
|
|
1012
|
+
const { readFileSync } = await import('node:fs')
|
|
1013
|
+
const { join } = await import('node:path')
|
|
1014
|
+
const read = (f: string) =>
|
|
1015
|
+
readFileSync(join(process.env.COOKBOOK_FIXTURES_DIR!, 'payments-compliance', f), 'utf8')
|
|
1016
|
+
|
|
1017
|
+
const leContract = await ctx.ensure(
|
|
1018
|
+
'screening holder LE contract',
|
|
1019
|
+
async () => {
|
|
1020
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1021
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Screening Holder LE Contract')
|
|
1022
|
+
},
|
|
1023
|
+
async () => {
|
|
1024
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1025
|
+
name: 'Cookbook Screening Holder LE Contract',
|
|
1026
|
+
description: 'Atomic FI screening → the screened account holder as a LegalEntity.',
|
|
1027
|
+
resource_type: 'legal_entity',
|
|
1028
|
+
type: 'identity',
|
|
1029
|
+
generic_table_id: null,
|
|
1030
|
+
template_config: { type: 'custom', body: read('_compliance_screenings_legal_entity.liquid') },
|
|
1031
|
+
mdm_input_config: { type: 'null' },
|
|
1032
|
+
})
|
|
1033
|
+
return data
|
|
1034
|
+
},
|
|
1035
|
+
)
|
|
1036
|
+
ctx.leContractId = leContract.id!
|
|
1037
|
+
|
|
1038
|
+
const gtContract = await ctx.ensure(
|
|
1039
|
+
'compliance screening GT contract',
|
|
1040
|
+
async () => {
|
|
1041
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1042
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Compliance Screening GT Contract')
|
|
1043
|
+
},
|
|
1044
|
+
async () => {
|
|
1045
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1046
|
+
name: 'Cookbook Compliance Screening GT Contract',
|
|
1047
|
+
description: 'Atomic FI screening → compliance_screenings row, stamped with its holder subject.',
|
|
1048
|
+
resource_type: 'generic_table',
|
|
1049
|
+
type: 'identity',
|
|
1050
|
+
generic_table_id: ctx.screeningsTableId,
|
|
1051
|
+
template_config: { type: 'custom', body: read('_compliance_screenings_generic_table.liquid') },
|
|
1052
|
+
mdm_input_config: { type: 'custom', body: read('_compliance_screenings_mdm.liquid') },
|
|
1053
|
+
})
|
|
1054
|
+
return data
|
|
1055
|
+
},
|
|
1056
|
+
)
|
|
1057
|
+
interopContractId = gtContract.id!
|
|
1058
|
+
```
|
|
1059
|
+
|
|
1060
|
+
## 020 — two DACs, because the pair must not run concurrently
|
|
1061
|
+
|
|
1062
|
+
The obvious move is one Data Activation Client carrying both contracts.
|
|
1063
|
+
**It works on every run but the first, which is the worst way for something
|
|
1064
|
+
to be wrong.**
|
|
1065
|
+
|
|
1066
|
+
A DAC fans out per `(row, contract)` at enqueue time — one job each — and
|
|
1067
|
+
the job key is `"<r2_key>-<index>-<client_id>-<contract_id>"`. The contract
|
|
1068
|
+
id is *in* the key, so the pair's two jobs sit in different chains and run
|
|
1069
|
+
**at the same instant**: the scheduler staggers by row index, not by
|
|
1070
|
+
contract. Contract A writes the holder as a legal entity. Contract B's
|
|
1071
|
+
`mdm_input_config` resolves the *same* holder, and resolution is
|
|
1072
|
+
find-or-create, so it writes one too. Both miss the dedupe read, both
|
|
1073
|
+
insert, and `legal_entity_identifications_uri_type_value_uk` —
|
|
1074
|
+
`(uri, id_type, id_number)`, with the legal entity deliberately left out of
|
|
1075
|
+
the key — refuses the loser:
|
|
1076
|
+
|
|
1077
|
+
```
|
|
1078
|
+
upsert_legal_entity_failed %{identifications: [%{uri: ["has already been taken"]}]}
|
|
1079
|
+
```
|
|
1080
|
+
|
|
1081
|
+
The batch reports `partial`, the row never lands, and the failure surfaces
|
|
1082
|
+
several steps later as a missing execution log.
|
|
1083
|
+
|
|
1084
|
+
**That index is not a bug you are hitting; it is the guard working.** It
|
|
1085
|
+
omits the legal entity from the key precisely so that concurrent workers
|
|
1086
|
+
racing to create one subject cannot each succeed and leave duplicates that
|
|
1087
|
+
are indistinguishable from two people who genuinely share a handle. One
|
|
1088
|
+
writer per subject is the contract, and the second run only looks green
|
|
1089
|
+
because the first run's Contract A already created the entity — so nothing
|
|
1090
|
+
inserts and nothing collides. Ingest a holder the platform has never seen
|
|
1091
|
+
and it fails again.
|
|
1092
|
+
|
|
1093
|
+
So the pair is split across two clients over the same tool and data source:
|
|
1094
|
+
an **identity** DAC that writes the subject, and a **data** DAC that writes
|
|
1095
|
+
the row and resolves the subject that now exists. §021 ingests through them
|
|
1096
|
+
in that order, waiting for the first to persist before starting the second.
|
|
1097
|
+
That is not a workaround for the index — it is the honest shape of the
|
|
1098
|
+
dependency, and it reads as one: an identity feed, then a data feed.
|
|
1099
|
+
|
|
1100
|
+
```typescript
|
|
1101
|
+
const identityDac = await ctx.ensure(
|
|
1102
|
+
'compliance screening identity DAC',
|
|
1103
|
+
async () => {
|
|
1104
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
|
|
1105
|
+
return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Compliance Screening Identity DAC')
|
|
1106
|
+
},
|
|
1107
|
+
async () => {
|
|
1108
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1109
|
+
name: 'Cookbook Compliance Screening Identity DAC',
|
|
1110
|
+
description: 'Manual-upload DAC — writes the screened holder as a legal entity, ahead of the screening row.',
|
|
1111
|
+
tool_id: ctx.manualUploadToolId,
|
|
1112
|
+
data_source_id: dataSourceId,
|
|
1113
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1114
|
+
interoperability_contract_ids: [ctx.leContractId],
|
|
1115
|
+
})
|
|
1116
|
+
return data
|
|
1117
|
+
},
|
|
1118
|
+
)
|
|
1119
|
+
ctx.identityDacSlug = identityDac.slug!
|
|
1120
|
+
|
|
1121
|
+
const dac = await ctx.ensure(
|
|
1122
|
+
'compliance screening data DAC',
|
|
1123
|
+
async () => {
|
|
1124
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
|
|
1125
|
+
return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Compliance Screening Data DAC')
|
|
1126
|
+
},
|
|
1127
|
+
async () => {
|
|
1128
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1129
|
+
name: 'Cookbook Compliance Screening Data DAC',
|
|
1130
|
+
description: 'Manual-upload DAC — ingests compliance-screening rows, resolving the holder written by the identity DAC.',
|
|
1131
|
+
tool_id: ctx.manualUploadToolId,
|
|
1132
|
+
data_source_id: dataSourceId,
|
|
1133
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1134
|
+
interoperability_contract_ids: [interopContractId],
|
|
1135
|
+
})
|
|
1136
|
+
return data
|
|
1137
|
+
},
|
|
1138
|
+
)
|
|
1139
|
+
dacId = dac.id!
|
|
1140
|
+
ctx.dacSlug = dac.slug!
|
|
1141
|
+
```
|
|
1142
|
+
|
|
1143
|
+
## 021 — ingest two screenings, holders first
|
|
1144
|
+
|
|
1145
|
+
Two rows, inline JSON, differing in the one field the filter reads. The
|
|
1146
|
+
first scores `0.88` — squarely inside `[0.80, 0.95)`, so the filter passes
|
|
1147
|
+
it to the agent. The second scores `0.20`, an auto-clear the filter rejects
|
|
1148
|
+
before any agent call.
|
|
1149
|
+
|
|
1150
|
+
**The same two rows are ingested twice: through the identity DAC, then
|
|
1151
|
+
through the data DAC.** That is the ordering §020 exists for — the holder
|
|
1152
|
+
has to be committed before a screening row can resolve it. The first stage
|
|
1153
|
+
is waited on before the second starts, and that wait is the whole point;
|
|
1154
|
+
firing both stages together reproduces the race the split was made to
|
|
1155
|
+
avoid.
|
|
1156
|
+
|
|
1157
|
+
Each row carries its own `account_holder_id`, so the two resolve to
|
|
1158
|
+
different legal entities and never contend with *each other* — within a
|
|
1159
|
+
stage they are independent, and the two `ingest` calls are separate only so
|
|
1160
|
+
each row gets its own activation-log row for the gate to key on.
|
|
1161
|
+
|
|
1162
|
+
The screening numbers are **stable across runs**, and
|
|
1163
|
+
`compliance_screening_number` is the table's unique column — so a second
|
|
1164
|
+
run upserts these same two rows rather than adding two more. `source_uri`
|
|
1165
|
+
matches the data source's `uri` from §011, which is what both contracts key
|
|
1166
|
+
their identifier on.
|
|
1167
|
+
|
|
1168
|
+
```typescript
|
|
1169
|
+
ctx.grayZoneNumber = 'CS-COOKBOOK-0101'
|
|
1170
|
+
ctx.autoClearNumber = 'CS-COOKBOOK-0102'
|
|
1171
|
+
// §037 reads this name back through all three lakes.
|
|
1172
|
+
ctx.grayZoneName = 'Jane Cookbook-Disambiguating-Doe'
|
|
1173
|
+
|
|
1174
|
+
const grayZoneRow = {
|
|
1175
|
+
compliance_screening_number: ctx.grayZoneNumber,
|
|
1176
|
+
account_holder_id: 'AH-COOKBOOK-0101',
|
|
1177
|
+
scope: 'account_holder',
|
|
1178
|
+
screening_type: 'sanctions',
|
|
1179
|
+
screening_status: 'pending',
|
|
1180
|
+
screened_entity_type: 'individual',
|
|
1181
|
+
sanctions_screening_status: 'match',
|
|
1182
|
+
screening_score: 0.88,
|
|
1183
|
+
match_count: 1,
|
|
1184
|
+
screened_entity_name: ctx.grayZoneName,
|
|
1185
|
+
source_uri: 'api.atomic.fi/compliance',
|
|
1186
|
+
}
|
|
1187
|
+
const autoClearRow = {
|
|
1188
|
+
compliance_screening_number: ctx.autoClearNumber,
|
|
1189
|
+
account_holder_id: 'AH-COOKBOOK-0102',
|
|
1190
|
+
scope: 'account_holder',
|
|
1191
|
+
screening_type: 'sanctions',
|
|
1192
|
+
screening_status: 'pass',
|
|
1193
|
+
screened_entity_type: 'individual',
|
|
1194
|
+
sanctions_screening_status: 'cleared',
|
|
1195
|
+
screening_score: 0.2,
|
|
1196
|
+
match_count: 0,
|
|
1197
|
+
screened_entity_name: 'John Cookbook-Clear-Smith',
|
|
1198
|
+
source_uri: 'api.atomic.fi/compliance',
|
|
1199
|
+
}
|
|
1200
|
+
|
|
1201
|
+
// STAGE 1 — the holders. Contract A writes each screened entity as a legal
|
|
1202
|
+
// entity. Nothing resolves anything here, so nothing can collide.
|
|
1203
|
+
const holderGray = await api.dataActivationClients.ingest(
|
|
1204
|
+
tenantSlug, datalakeSlug, ctx.identityDacSlug, { data: grayZoneRow },
|
|
1205
|
+
)
|
|
1206
|
+
const holderClear = await api.dataActivationClients.ingest(
|
|
1207
|
+
tenantSlug, datalakeSlug, ctx.identityDacSlug, { data: autoClearRow },
|
|
1208
|
+
)
|
|
1209
|
+
|
|
1210
|
+
// The wait that makes the split work. Skip it and stage 2 resolves a
|
|
1211
|
+
// holder that is still being written, which is the race in a slower shape.
|
|
1212
|
+
await ctx.waitForBatches(ctx.identityDacSlug, [
|
|
1213
|
+
holderGray.data.batch_id!,
|
|
1214
|
+
holderClear.data.batch_id!,
|
|
1215
|
+
])
|
|
1216
|
+
|
|
1217
|
+
// STAGE 2 — the screenings. Contract B resolves the holder that now
|
|
1218
|
+
// exists, and the platform stamps `legal_entity_id` onto the row.
|
|
1219
|
+
const grayZoneIngest = await api.dataActivationClients.ingest(
|
|
1220
|
+
tenantSlug, datalakeSlug, ctx.dacSlug, { data: grayZoneRow },
|
|
1221
|
+
)
|
|
1222
|
+
const autoClearIngest = await api.dataActivationClients.ingest(
|
|
1223
|
+
tenantSlug, datalakeSlug, ctx.dacSlug, { data: autoClearRow },
|
|
1224
|
+
)
|
|
1225
|
+
ctx.batchGrayZone = grayZoneIngest.data.batch_id!
|
|
1226
|
+
ctx.batchAutoClear = autoClearIngest.data.batch_id!
|
|
1227
|
+
```
|
|
1228
|
+
|
|
1229
|
+
## 022 — wait for the screening rows to PERSIST, not merely to arrive
|
|
1230
|
+
|
|
1231
|
+
The same `waitForBatches` from §008, now on the data DAC. Stage 1 was
|
|
1232
|
+
already waited on inside §021; this is the gate on stage 2, and it is what
|
|
1233
|
+
stands between a refused row and a workflow that runs against an empty
|
|
1234
|
+
table.
|
|
1235
|
+
|
|
1236
|
+
```typescript
|
|
1237
|
+
await ctx.waitForBatches(ctx.dacSlug, [ctx.batchGrayZone, ctx.batchAutoClear])
|
|
1238
|
+
```
|
|
1239
|
+
|
|
1240
|
+
## 023 — run the workflow against the two rows
|
|
1241
|
+
|
|
1242
|
+
`workflows.run` schedules; it does not fire. The helper from §008 waits for
|
|
1243
|
+
the run to actually fire and hands back the log id and batch id the
|
|
1244
|
+
verification needs.
|
|
1245
|
+
|
|
1246
|
+
```typescript
|
|
1247
|
+
const runResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.workflowSlug, {
|
|
1248
|
+
sql_where_clause: `compliance_screening_number IN ('${ctx.grayZoneNumber}', '${ctx.autoClearNumber}')`,
|
|
1249
|
+
mode: 'live',
|
|
1250
|
+
manual_override: false,
|
|
1251
|
+
})
|
|
1252
|
+
const fired = await ctx.waitForFiredRun(datalakeSlug, runResp.data.workflow_run_id)
|
|
1253
|
+
ctx.runLogId = fired.workflowRunLogId
|
|
1254
|
+
ctx.runBatchId = fired.batchId!
|
|
1255
|
+
|
|
1256
|
+
const deadline = Date.now() + 240_000
|
|
1257
|
+
let status: string | null = null
|
|
1258
|
+
while (Date.now() < deadline) {
|
|
1259
|
+
const { data: log } = await api.workflows.batchLogs.refresh(tenantSlug, datalakeSlug, ctx.workflowSlug, ctx.runLogId)
|
|
1260
|
+
status = log.status ?? null
|
|
1261
|
+
if (status && status !== 'pending') break
|
|
1262
|
+
await new Promise((r) => setTimeout(r, 2_000))
|
|
1263
|
+
}
|
|
1264
|
+
if (status === 'failed') throw new Error('agent-driven workflow run reached :failed')
|
|
1265
|
+
if (!status || status === 'pending') {
|
|
1266
|
+
throw new Error('workflow run did not leave :pending within 240s')
|
|
1267
|
+
}
|
|
1268
|
+
```
|
|
1269
|
+
|
|
1270
|
+
## 024 — verify the band filtered, and the agent steered the fan-out
|
|
1271
|
+
|
|
1272
|
+
Each row produced a Workflow Execution Log. The gray-zone screening passes
|
|
1273
|
+
the filter, reaches the agent, and its WEL is `:executing` or `:completed`.
|
|
1274
|
+
The auto-clear screening fails the filter, so its WEL is `:filtered` — it
|
|
1275
|
+
never reached the agent. A run where both passed, or both were filtered,
|
|
1276
|
+
would mean the band is not actually evaluating the score.
|
|
1277
|
+
|
|
1278
|
+
The pass-branch WEL carries exactly **two** action execution logs, one per
|
|
1279
|
+
verdict. On the first run exactly one is matched and the other `:skipped`,
|
|
1280
|
+
which is the proof the verdict steered the decision rather than every
|
|
1281
|
+
action firing.
|
|
1282
|
+
|
|
1283
|
+
**On a second run both are `:skipped`, and that is correct.** This is where
|
|
1284
|
+
idempotence stops being a property of the walk and becomes a property of
|
|
1285
|
+
the product. `idempotency_template` renders
|
|
1286
|
+
`{{ event_dataset.compliance_screening_number }}-<verdict>`, the screening
|
|
1287
|
+
number is stable, so the key is the key that already fired — and the
|
|
1288
|
+
platform refuses to send the same SMS about the same screening twice. That
|
|
1289
|
+
is exactly what you want in production, and it means **an assertion on "an
|
|
1290
|
+
action fired just now" is wrong**: it tests a transition that only ever
|
|
1291
|
+
happens once. Assert the end state instead — the verdict SMS exists for
|
|
1292
|
+
this screening — which is what §025 does, and which is true on every run.
|
|
1293
|
+
|
|
1294
|
+
```typescript
|
|
1295
|
+
const welDeadline = Date.now() + 120_000
|
|
1296
|
+
let ourWels: Array<Record<string, unknown>> = []
|
|
1297
|
+
let byStatus: Record<string, number> = {}
|
|
1298
|
+
while (Date.now() < welDeadline) {
|
|
1299
|
+
const { data: wfLogs } = await api.workflows.workflowLogs.list(tenantSlug, datalakeSlug, ctx.workflowSlug)
|
|
1300
|
+
ourWels = (wfLogs.data ?? [])
|
|
1301
|
+
.map((w) => w as Record<string, unknown>)
|
|
1302
|
+
.filter((w) => w.batch_id === ctx.runBatchId)
|
|
1303
|
+
byStatus = {}
|
|
1304
|
+
for (const w of ourWels) {
|
|
1305
|
+
const st = (w.status as string | undefined) ?? 'unknown'
|
|
1306
|
+
byStatus[st] = (byStatus[st] ?? 0) + 1
|
|
1307
|
+
}
|
|
1308
|
+
const settled = (byStatus.completed ?? 0) + (byStatus.executing ?? 0)
|
|
1309
|
+
if (ourWels.length === 2 && settled === 1 && (byStatus.filtered ?? 0) === 1) break
|
|
1310
|
+
await new Promise((r) => setTimeout(r, 2_000))
|
|
1311
|
+
}
|
|
1312
|
+
if (ourWels.length !== 2) {
|
|
1313
|
+
throw new Error(`expected 2 WELs for the run, got ${ourWels.length}`)
|
|
1314
|
+
}
|
|
1315
|
+
if ((byStatus.filtered ?? 0) !== 1) {
|
|
1316
|
+
throw new Error(`expected 1 :filtered WEL (the auto-clear) — distribution ${JSON.stringify(byStatus)}`)
|
|
1317
|
+
}
|
|
1318
|
+
const passWel = ourWels.find((w) => {
|
|
1319
|
+
const st = (w as { status?: string }).status
|
|
1320
|
+
return st === 'executing' || st === 'completed'
|
|
1321
|
+
})
|
|
1322
|
+
if (!passWel) {
|
|
1323
|
+
throw new Error(`no pass-branch WEL — distribution ${JSON.stringify(byStatus)}`)
|
|
1324
|
+
}
|
|
1325
|
+
|
|
1326
|
+
const aels =
|
|
1327
|
+
(passWel as { action_execution_logs?: Array<{ status?: string; decision_key?: string }> })
|
|
1328
|
+
.action_execution_logs ?? []
|
|
1329
|
+
if (aels.length !== 2) {
|
|
1330
|
+
throw new Error(`expected 2 AELs (one per verdict) on the gray-zone WEL, got ${aels.length}`)
|
|
1331
|
+
}
|
|
1332
|
+
const aelByStatus: Record<string, number> = {}
|
|
1333
|
+
for (const ael of aels) {
|
|
1334
|
+
const st = ael.status ?? 'unknown'
|
|
1335
|
+
aelByStatus[st] = (aelByStatus[st] ?? 0) + 1
|
|
1336
|
+
}
|
|
1337
|
+
const matched = (aelByStatus.pending ?? 0) + (aelByStatus.completed ?? 0)
|
|
1338
|
+
if (matched === 1) {
|
|
1339
|
+
// First run for this screening: the verdict picked one action and left
|
|
1340
|
+
// the other alone.
|
|
1341
|
+
if ((aelByStatus.skipped ?? 0) !== 1) {
|
|
1342
|
+
throw new Error(`expected 1 :skipped AEL alongside the matched one — ${JSON.stringify(aelByStatus)}`)
|
|
1343
|
+
}
|
|
1344
|
+
const routed = aels.find((a) => a.status === 'pending' || a.status === 'completed')
|
|
1345
|
+
if (!routed?.decision_key || !/^(verified|blocked)$/.test(routed.decision_key)) {
|
|
1346
|
+
throw new Error(`matched AEL has unexpected decision_key ${routed?.decision_key}`)
|
|
1347
|
+
}
|
|
1348
|
+
} else if (matched === 0 && (aelByStatus.skipped ?? 0) === aels.length) {
|
|
1349
|
+
// Re-run: this screening's SMS already went out, so the idempotency key
|
|
1350
|
+
// suppressed both actions. §025 is the gate that still has to hold.
|
|
1351
|
+
} else {
|
|
1352
|
+
throw new Error(
|
|
1353
|
+
`expected either 1 matched AEL, or all skipped on a repeat run — got ${JSON.stringify(aelByStatus)}`,
|
|
1354
|
+
)
|
|
1355
|
+
}
|
|
1356
|
+
```
|
|
1357
|
+
|
|
1358
|
+
## 025 — confirm the verdict-keyed SMS actually rendered
|
|
1359
|
+
|
|
1360
|
+
The matched action fired immediately (`trigger_template: 'now'`) and
|
|
1361
|
+
persisted a row in the `message` dataset carrying the fully-rendered body,
|
|
1362
|
+
prefixed with the verdict that selected it. For this fixture's entity name
|
|
1363
|
+
the agent is prompted toward `verified`, so the persisted SMS reads
|
|
1364
|
+
`[verified]`.
|
|
1365
|
+
|
|
1366
|
+
**This is the gate, not §024.** It asserts the end state — a verdict SMS
|
|
1367
|
+
exists for this screening — which is true on the first run because the
|
|
1368
|
+
action just fired, and true on every run after because it fired once and
|
|
1369
|
+
the idempotency key has protected it ever since.
|
|
1370
|
+
|
|
1371
|
+
**Search on the idempotency key, not on `workflow_id`.** They look
|
|
1372
|
+
interchangeable and are not, and the difference only shows up on the run
|
|
1373
|
+
you care about most. The key is `<screening number>-<verdict>`, and the
|
|
1374
|
+
screening number is stable — so the key names the same SMS forever. A
|
|
1375
|
+
workflow *id* is stable only as long as the workflow resource is. Rebuild
|
|
1376
|
+
the tenant against a lake that still holds its rows — a reset control
|
|
1377
|
+
plane, a restored database, a re-run after `destroy` — and the workflow
|
|
1378
|
+
comes back with a new id while the action logs keep the old key. The
|
|
1379
|
+
action then correctly refuses to fire, because that key already fired, and
|
|
1380
|
+
a search for the new `workflow_id` finds nothing. Green becomes red with
|
|
1381
|
+
nothing wrong: the SMS is right there, filed under the identity that did
|
|
1382
|
+
not change.
|
|
1383
|
+
|
|
1384
|
+
The read answers from **raw**. That is the right tier for this question: it
|
|
1385
|
+
asks what this walk wrote, not whether replication has caught up.
|
|
1386
|
+
|
|
1387
|
+
```typescript
|
|
1388
|
+
const { data: userSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
|
|
1389
|
+
search_query: `m.idempotency_key LIKE '${ctx.grayZoneNumber}-%'`,
|
|
1390
|
+
})
|
|
1391
|
+
if (userSearch.status !== 'completed') {
|
|
1392
|
+
throw new Error(`message user-search status=${userSearch.status} error=${userSearch.error_message ?? '(none)'}`)
|
|
1393
|
+
}
|
|
1394
|
+
|
|
1395
|
+
const msgDeadline = Date.now() + 45_000
|
|
1396
|
+
let messages: Array<Record<string, unknown>> = []
|
|
1397
|
+
while (Date.now() < msgDeadline && messages.length === 0) {
|
|
1398
|
+
const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
|
|
1399
|
+
userSearchId: userSearch.id!,
|
|
1400
|
+
dataAccessMode: 'raw',
|
|
1401
|
+
})
|
|
1402
|
+
messages = (data.data ?? []) as Array<Record<string, unknown>>
|
|
1403
|
+
if (messages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
|
|
1404
|
+
}
|
|
1405
|
+
if (messages.length === 0) {
|
|
1406
|
+
throw new Error(
|
|
1407
|
+
`no verdict SMS for screening ${ctx.grayZoneNumber} within 45s — ` +
|
|
1408
|
+
`the action never fired, or it fired under a different idempotency key`,
|
|
1409
|
+
)
|
|
1410
|
+
}
|
|
1411
|
+
|
|
1412
|
+
const verdictBody = messages
|
|
1413
|
+
.map((m) => String(m.body ?? ''))
|
|
1414
|
+
.find((body) => body.includes('[verified]') || body.includes('[blocked]'))
|
|
1415
|
+
if (!verdictBody) {
|
|
1416
|
+
throw new Error(`no verdict-prefixed SMS body — bodies: ${JSON.stringify(messages.map((m) => m.body))}`)
|
|
1417
|
+
}
|
|
1418
|
+
if (!verdictBody.includes('[verified]')) {
|
|
1419
|
+
throw new Error(`expected the agent verdict to render [verified] — got: ${verdictBody}`)
|
|
1420
|
+
}
|
|
1421
|
+
```
|
|
1422
|
+
|
|
1423
|
+
## 026 — the KYC portal, as a connected app
|
|
1424
|
+
|
|
1425
|
+
The second scenario's SMS carries a deep link to a self-serve KYC portal.
|
|
1426
|
+
The connected app is a thin registration of that portal's URL and mode —
|
|
1427
|
+
the page itself is hosted outside the platform (`mode: 'self_hosted'`). The
|
|
1428
|
+
server-derived slug is what §036 resolves the link against.
|
|
1429
|
+
|
|
1430
|
+
```typescript
|
|
1431
|
+
const connectedApp = await ctx.ensure(
|
|
1432
|
+
'KYC portal connected app',
|
|
1433
|
+
async () => {
|
|
1434
|
+
const { data } = await api.connectedApps.list(tenantSlug, datalakeSlug)
|
|
1435
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook KYC Portal')
|
|
1436
|
+
},
|
|
1437
|
+
async () => {
|
|
1438
|
+
const { data } = await api.connectedApps.create(tenantSlug, datalakeSlug, {
|
|
1439
|
+
name: 'Cookbook KYC Portal',
|
|
1440
|
+
description: 'Self-serve KYC portal linked from the outbound notification SMS.',
|
|
1441
|
+
mode: 'self_hosted',
|
|
1442
|
+
urls: [{ url: 'https://kyc.example.local', is_primary: true, label: 'production' }],
|
|
1443
|
+
})
|
|
1444
|
+
return data
|
|
1445
|
+
},
|
|
1446
|
+
)
|
|
1447
|
+
connectedAppId = connectedApp.id!
|
|
1448
|
+
ctx.connectedAppSlug = connectedApp.slug!
|
|
1449
|
+
```
|
|
1450
|
+
|
|
1451
|
+
## 027 — the payment-accounts table
|
|
1452
|
+
|
|
1453
|
+
`payment_account_external_id` is the unique column. `status` is what the
|
|
1454
|
+
filter gates on, and it stays `none` so the workflow can read it in any
|
|
1455
|
+
tier.
|
|
1456
|
+
|
|
1457
|
+
**`privacy_requirement` is where PCI lives now.** `account_number` and
|
|
1458
|
+
`iban` are `tokenize`; the free-text payment narrative is `redact`. That
|
|
1459
|
+
declaration is the whole masking policy — there is no second schema to
|
|
1460
|
+
write sensitive values into and no changeset quietly tokenizing on the way
|
|
1461
|
+
past. The raw value goes in, and the derived lakes decide who sees what.
|
|
1462
|
+
|
|
1463
|
+
Again: no `legal_entity_id` column. The platform stamps it from Contract B.
|
|
1464
|
+
|
|
1465
|
+
```typescript
|
|
1466
|
+
const accounts = await ctx.ensure(
|
|
1467
|
+
'payment-accounts table',
|
|
1468
|
+
async () => {
|
|
1469
|
+
const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
|
|
1470
|
+
return (data.data ?? []).find((t: { title?: string }) => t.title === 'Payment Accounts')
|
|
1471
|
+
},
|
|
1472
|
+
async () => {
|
|
1473
|
+
const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
|
|
1474
|
+
title: 'Payment Accounts',
|
|
1475
|
+
description: 'Payment accounts loaded from the Atomic FI API',
|
|
1476
|
+
columns: [
|
|
1477
|
+
{ name: 'payment_account_external_id', title: 'External ID', type: 'string', description: "The provider's account id — unique per account", is_unique: true, privacy_requirement: 'none' },
|
|
1478
|
+
{ name: 'payment_account_number', title: 'Account Ref', type: 'string', description: 'Internal account reference', is_unique: false, privacy_requirement: 'none' },
|
|
1479
|
+
{ name: 'account_holder_ref', title: 'Account Holder Ref', type: 'string', description: "The provider's account-holder id — the identifier MDM resolves on", is_unique: false, privacy_requirement: 'none' },
|
|
1480
|
+
{ name: 'status', title: 'Status', type: 'string', description: 'active / suspended / pending — the filter gate', is_unique: false, privacy_requirement: 'none' },
|
|
1481
|
+
{ name: 'account_type', title: 'Account Type', type: 'string', description: 'bank_account / card / wallet', is_unique: false, privacy_requirement: 'none' },
|
|
1482
|
+
{ name: 'currency', title: 'Currency', type: 'string', description: 'Settlement currency', is_unique: false, privacy_requirement: 'none' },
|
|
1483
|
+
{ name: 'routing_number', title: 'Routing Number', type: 'string', description: 'Domestic routing number', is_unique: false, privacy_requirement: 'none' },
|
|
1484
|
+
{ name: 'swift_bic', title: 'SWIFT/BIC', type: 'string', description: 'ISO 9362 bank identifier', is_unique: false, privacy_requirement: 'none' },
|
|
1485
|
+
{ name: 'bank_name', title: 'Bank Name', type: 'string', description: 'Holding institution', is_unique: false, privacy_requirement: 'none' },
|
|
1486
|
+
{ name: 'account_number', title: 'Account Number', type: 'string', description: 'PCI-sensitive account number', is_unique: false, privacy_requirement: 'tokenize' },
|
|
1487
|
+
{ name: 'iban', title: 'IBAN', type: 'string', description: 'PCI-sensitive international account number', is_unique: false, privacy_requirement: 'tokenize' },
|
|
1488
|
+
{ name: 'destination_country', title: 'Destination Country', type: 'string', description: 'ISO country derived from the BIC or IBAN, so nothing downstream re-parses it', is_unique: false, privacy_requirement: 'none' },
|
|
1489
|
+
{ name: 'purpose_of_payment_detail', title: 'Purpose of Payment', type: 'string', description: 'Free-text payment narrative', is_unique: false, privacy_requirement: 'redact' },
|
|
1490
|
+
{ name: 'enabled_regimes', title: 'Enabled Regimes', type: 'string', description: 'Comma-delimited regimes this account may settle under', is_unique: false, privacy_requirement: 'none' },
|
|
1491
|
+
],
|
|
1492
|
+
})
|
|
1493
|
+
return data
|
|
1494
|
+
},
|
|
1495
|
+
)
|
|
1496
|
+
ctx.accountsTableId = accounts.id!
|
|
1497
|
+
ctx.accountsTableName = accounts.name!
|
|
1498
|
+
|
|
1499
|
+
await ctx.deployGenericTable(ctx.accountsTableName)
|
|
1500
|
+
```
|
|
1501
|
+
|
|
1502
|
+
## 028 — the KYC-notification workflow
|
|
1503
|
+
|
|
1504
|
+
Standard shape: a filter passing accounts whose `status` is `active`, a
|
|
1505
|
+
decision naming one key, and one SMS action. The body references the
|
|
1506
|
+
account's external id so the recipient knows which account activated, and
|
|
1507
|
+
`{{ connected_app_form_url }}` so the platform bakes in the `/t/<token>`
|
|
1508
|
+
portal link.
|
|
1509
|
+
|
|
1510
|
+
`skip_mdm_resolution: false` is set explicitly and **must be**. The account
|
|
1511
|
+
holder is the subject the portal token is minted against, and it reaches
|
|
1512
|
+
the workflow only through the `legal_entity_id` stamped on the row.
|
|
1513
|
+
|
|
1514
|
+
A `context_datasets` entry queries the `message` dataset for any
|
|
1515
|
+
KYC notification already sent to the same legal entity in the last six
|
|
1516
|
+
months — the built-in guard against re-notifying. The join key is
|
|
1517
|
+
`legal_entity_id`, and `{{ legal_entity_id }}` is the binding the context
|
|
1518
|
+
builder exposes for the resolved subject.
|
|
1519
|
+
|
|
1520
|
+
The SMS `to` is a fixed compliance inbox. The holder's own contact fields
|
|
1521
|
+
are tokenized, and this workflow deliberately notifies the operator rather
|
|
1522
|
+
than the customer.
|
|
1523
|
+
|
|
1524
|
+
```typescript
|
|
1525
|
+
const FILTER_BODY = '{% if event_dataset.status == "active" %}true{% endif %}'
|
|
1526
|
+
const DECISION_KEY = 'send_pr_notification_sms'
|
|
1527
|
+
const DECISION_BODY = `["${DECISION_KEY}"]`
|
|
1528
|
+
const KYC_DECISION_OUTPUT_SCHEMA = { type: 'array', items: { type: 'string' } }
|
|
1529
|
+
const SMS_BODY_TEMPLATE =
|
|
1530
|
+
'Alvera PR notification — payment_account ' +
|
|
1531
|
+
'{{ event_dataset.payment_account_external_id }} has been activated. ' +
|
|
1532
|
+
'Manage KYC: {{ connected_app_form_url }}'
|
|
1533
|
+
|
|
1534
|
+
const kycWorkflow = await ctx.ensure(
|
|
1535
|
+
'KYC notification workflow',
|
|
1536
|
+
async () => {
|
|
1537
|
+
const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
|
|
1538
|
+
return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook KYC Notification Workflow')
|
|
1539
|
+
},
|
|
1540
|
+
async () => {
|
|
1541
|
+
const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
|
|
1542
|
+
name: 'Cookbook KYC Notification Workflow',
|
|
1543
|
+
description: 'Sends a KYC-notification SMS for newly activated payment accounts with a self-serve portal link.',
|
|
1544
|
+
dataset_type: 'generic_table',
|
|
1545
|
+
generic_table_id: ctx.accountsTableId,
|
|
1546
|
+
status: 'live',
|
|
1547
|
+
tags: ['compliance', 'kyc'],
|
|
1548
|
+
skip_mdm_resolution: false,
|
|
1549
|
+
filter_config: { type: 'custom', body: FILTER_BODY },
|
|
1550
|
+
decision_config: {
|
|
1551
|
+
type: 'custom',
|
|
1552
|
+
body: DECISION_BODY,
|
|
1553
|
+
output_schema: KYC_DECISION_OUTPUT_SCHEMA,
|
|
1554
|
+
},
|
|
1555
|
+
context_datasets: [
|
|
1556
|
+
{
|
|
1557
|
+
dataset_type: 'message',
|
|
1558
|
+
where_clause:
|
|
1559
|
+
`m.legal_entity_id = '{{ legal_entity_id }}' ` +
|
|
1560
|
+
`AND m.decision_key = '${DECISION_KEY}' ` +
|
|
1561
|
+
"AND m.sent_at > NOW() - INTERVAL '6 months'",
|
|
1562
|
+
limit: 1,
|
|
1563
|
+
position: 0,
|
|
1564
|
+
},
|
|
1565
|
+
],
|
|
1566
|
+
actions: [
|
|
1567
|
+
{
|
|
1568
|
+
action_type: 'sms',
|
|
1569
|
+
tool_id: toolId,
|
|
1570
|
+
decision_key: DECISION_KEY,
|
|
1571
|
+
position: 0,
|
|
1572
|
+
trigger_template: 'now',
|
|
1573
|
+
idempotency_template: '{{ event_dataset.payment_account_external_id }}-{{ decision_key }}',
|
|
1574
|
+
connected_app_id: connectedAppId,
|
|
1575
|
+
connected_app_route: '/portal/kyc',
|
|
1576
|
+
connected_app_metadata_template:
|
|
1577
|
+
'{"payment_account_external_id":"{{ event_dataset.payment_account_external_id }}"}',
|
|
1578
|
+
tool_call: {
|
|
1579
|
+
tool_call_type: 'sms_request',
|
|
1580
|
+
to: { type: 'custom', body: '+15551234567' },
|
|
1581
|
+
body: { type: 'custom', body: SMS_BODY_TEMPLATE },
|
|
1582
|
+
sms_type: 'transactional',
|
|
1583
|
+
},
|
|
1584
|
+
},
|
|
1585
|
+
],
|
|
1586
|
+
})
|
|
1587
|
+
return data
|
|
1588
|
+
},
|
|
1589
|
+
)
|
|
1590
|
+
ctx.kycWorkflowId = kycWorkflow.id!
|
|
1591
|
+
ctx.kycWorkflowSlug = kycWorkflow.slug!
|
|
1592
|
+
|
|
1593
|
+
if (kycWorkflow.skip_mdm_resolution !== false) {
|
|
1594
|
+
throw new Error('a generic-table workflow that mints a per-holder link must keep MDM resolution ON')
|
|
1595
|
+
}
|
|
1596
|
+
```
|
|
1597
|
+
|
|
1598
|
+
## 029 — the payment-account contract pair
|
|
1599
|
+
|
|
1600
|
+
Same two-contract shape as §019, over a different table. Contract A writes
|
|
1601
|
+
the holder as the subject; Contract B writes the account row and emits the
|
|
1602
|
+
same `(uri, account_holder_id)` identifier so both converge and the row is
|
|
1603
|
+
stamped.
|
|
1604
|
+
|
|
1605
|
+
```typescript
|
|
1606
|
+
const { readFileSync: readFileSync2 } = await import('node:fs')
|
|
1607
|
+
const { join: join2 } = await import('node:path')
|
|
1608
|
+
const read2 = (f: string) =>
|
|
1609
|
+
readFileSync2(join2(process.env.COOKBOOK_FIXTURES_DIR!, 'payments-compliance', f), 'utf8')
|
|
1610
|
+
|
|
1611
|
+
const accountLeContract = await ctx.ensure(
|
|
1612
|
+
'payment-account holder LE contract',
|
|
1613
|
+
async () => {
|
|
1614
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1615
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Atomic FI Holder LE Contract')
|
|
1616
|
+
},
|
|
1617
|
+
async () => {
|
|
1618
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1619
|
+
name: 'Cookbook Atomic FI Holder LE Contract',
|
|
1620
|
+
description: 'Atomic FI payment-account → the account holder LegalEntity.',
|
|
1621
|
+
resource_type: 'legal_entity',
|
|
1622
|
+
type: 'identity',
|
|
1623
|
+
generic_table_id: null,
|
|
1624
|
+
template_config: { type: 'custom', body: read2('_payment_accounts_legal_entity.liquid') },
|
|
1625
|
+
mdm_input_config: { type: 'null' },
|
|
1626
|
+
})
|
|
1627
|
+
return data
|
|
1628
|
+
},
|
|
1629
|
+
)
|
|
1630
|
+
ctx.accountLeContractId = accountLeContract.id!
|
|
1631
|
+
|
|
1632
|
+
const accountGtContract = await ctx.ensure(
|
|
1633
|
+
'payment-account GT contract',
|
|
1634
|
+
async () => {
|
|
1635
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1636
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Atomic FI Payment Account GT Contract')
|
|
1637
|
+
},
|
|
1638
|
+
async () => {
|
|
1639
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1640
|
+
name: 'Cookbook Atomic FI Payment Account GT Contract',
|
|
1641
|
+
description: 'Atomic FI payment-account → payment_accounts row, stamped with its holder subject.',
|
|
1642
|
+
resource_type: 'generic_table',
|
|
1643
|
+
type: 'identity',
|
|
1644
|
+
generic_table_id: ctx.accountsTableId,
|
|
1645
|
+
template_config: { type: 'custom', body: read2('_payment_accounts_generic_table.liquid') },
|
|
1646
|
+
mdm_input_config: { type: 'custom', body: read2('_payment_accounts_mdm.liquid') },
|
|
1647
|
+
})
|
|
1648
|
+
return data
|
|
1649
|
+
},
|
|
1650
|
+
)
|
|
1651
|
+
ctx.accountGtContractId = accountGtContract.id!
|
|
1652
|
+
```
|
|
1653
|
+
|
|
1654
|
+
## 030 — the payment-account DACs, split the same way
|
|
1655
|
+
|
|
1656
|
+
The second contract pair gets the same treatment as the first, for the same
|
|
1657
|
+
reason — §020 has the full account of why. Both clients reuse the same
|
|
1658
|
+
manual-upload tool and the same data source; what differs between any two
|
|
1659
|
+
of these ingestion paths is only the contracts, which is the level the
|
|
1660
|
+
difference actually lives at.
|
|
1661
|
+
|
|
1662
|
+
```typescript
|
|
1663
|
+
const accountIdentityDac = await ctx.ensure(
|
|
1664
|
+
'payment-account identity DAC',
|
|
1665
|
+
async () => {
|
|
1666
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
|
|
1667
|
+
return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Atomic FI Payment Account Identity DAC')
|
|
1668
|
+
},
|
|
1669
|
+
async () => {
|
|
1670
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1671
|
+
name: 'Cookbook Atomic FI Payment Account Identity DAC',
|
|
1672
|
+
description: 'Manual-upload DAC — writes the account holder as a legal entity, ahead of the account row.',
|
|
1673
|
+
tool_id: ctx.manualUploadToolId,
|
|
1674
|
+
data_source_id: dataSourceId,
|
|
1675
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1676
|
+
interoperability_contract_ids: [ctx.accountLeContractId],
|
|
1677
|
+
})
|
|
1678
|
+
return data
|
|
1679
|
+
},
|
|
1680
|
+
)
|
|
1681
|
+
ctx.accountIdentityDacSlug = accountIdentityDac.slug!
|
|
1682
|
+
|
|
1683
|
+
const accountDac = await ctx.ensure(
|
|
1684
|
+
'payment-account data DAC',
|
|
1685
|
+
async () => {
|
|
1686
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug)
|
|
1687
|
+
return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Atomic FI Payment Account Data DAC')
|
|
1688
|
+
},
|
|
1689
|
+
async () => {
|
|
1690
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1691
|
+
name: 'Cookbook Atomic FI Payment Account Data DAC',
|
|
1692
|
+
description: 'Manual-upload DAC — ingests payment-account rows, resolving the holder written by the identity DAC.',
|
|
1693
|
+
tool_id: ctx.manualUploadToolId,
|
|
1694
|
+
data_source_id: dataSourceId,
|
|
1695
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1696
|
+
interoperability_contract_ids: [ctx.accountGtContractId],
|
|
1697
|
+
})
|
|
1698
|
+
return data
|
|
1699
|
+
},
|
|
1700
|
+
)
|
|
1701
|
+
ctx.accountDacSlug = accountDac.slug!
|
|
1702
|
+
```
|
|
1703
|
+
|
|
1704
|
+
## 031 — ingest one active account and one suspended one
|
|
1705
|
+
|
|
1706
|
+
Two rows differing in the field the filter reads. `active` passes;
|
|
1707
|
+
`suspended` must be rejected.
|
|
1708
|
+
|
|
1709
|
+
Two stages again, holders before accounts, exactly as in §021.
|
|
1710
|
+
|
|
1711
|
+
```typescript
|
|
1712
|
+
ctx.activeExternalId = 'PA-COOKBOOK-0101'
|
|
1713
|
+
ctx.suspendedExternalId = 'PA-COOKBOOK-0102'
|
|
1714
|
+
|
|
1715
|
+
const activeRow = {
|
|
1716
|
+
payment_account_external_id: ctx.activeExternalId,
|
|
1717
|
+
payment_account_number: 'PA-NUM-COOKBOOK-0101',
|
|
1718
|
+
account_holder_id: 'AH-COOKBOOK-KYC-0101',
|
|
1719
|
+
account_holder_name: 'Ada Cookbook-Active-Lovelace',
|
|
1720
|
+
status: 'active',
|
|
1721
|
+
account_type: 'bank_account',
|
|
1722
|
+
currency: 'USD',
|
|
1723
|
+
routing_number: '121000358',
|
|
1724
|
+
swift_bic: 'BOFAUS3N',
|
|
1725
|
+
bank_name: 'Bank of America',
|
|
1726
|
+
enabled_regimes: ['us_domestic'],
|
|
1727
|
+
source_uri: 'api.atomic.fi/compliance',
|
|
1728
|
+
}
|
|
1729
|
+
const suspendedRow = {
|
|
1730
|
+
payment_account_external_id: ctx.suspendedExternalId,
|
|
1731
|
+
payment_account_number: 'PA-NUM-COOKBOOK-0102',
|
|
1732
|
+
account_holder_id: 'AH-COOKBOOK-KYC-0102',
|
|
1733
|
+
account_holder_name: 'Grace Cookbook-Suspended-Hopper',
|
|
1734
|
+
status: 'suspended',
|
|
1735
|
+
account_type: 'bank_account',
|
|
1736
|
+
currency: 'USD',
|
|
1737
|
+
routing_number: '121000358',
|
|
1738
|
+
bank_name: 'Bank of America',
|
|
1739
|
+
enabled_regimes: ['us_domestic'],
|
|
1740
|
+
source_uri: 'api.atomic.fi/compliance',
|
|
1741
|
+
}
|
|
1742
|
+
|
|
1743
|
+
// STAGE 1 — the account holders.
|
|
1744
|
+
const holderActive = await api.dataActivationClients.ingest(
|
|
1745
|
+
tenantSlug, datalakeSlug, ctx.accountIdentityDacSlug, { data: activeRow },
|
|
1746
|
+
)
|
|
1747
|
+
const holderSuspended = await api.dataActivationClients.ingest(
|
|
1748
|
+
tenantSlug, datalakeSlug, ctx.accountIdentityDacSlug, { data: suspendedRow },
|
|
1749
|
+
)
|
|
1750
|
+
await ctx.waitForBatches(ctx.accountIdentityDacSlug, [
|
|
1751
|
+
holderActive.data.batch_id!,
|
|
1752
|
+
holderSuspended.data.batch_id!,
|
|
1753
|
+
])
|
|
1754
|
+
|
|
1755
|
+
// STAGE 2 — the accounts themselves.
|
|
1756
|
+
const activeIngest = await api.dataActivationClients.ingest(
|
|
1757
|
+
tenantSlug, datalakeSlug, ctx.accountDacSlug, { data: activeRow },
|
|
1758
|
+
)
|
|
1759
|
+
const suspendedIngest = await api.dataActivationClients.ingest(
|
|
1760
|
+
tenantSlug, datalakeSlug, ctx.accountDacSlug, { data: suspendedRow },
|
|
1761
|
+
)
|
|
1762
|
+
ctx.batchActive = activeIngest.data.batch_id!
|
|
1763
|
+
ctx.batchSuspended = suspendedIngest.data.batch_id!
|
|
1764
|
+
```
|
|
1765
|
+
|
|
1766
|
+
## 032 — wait for both account batches to persist
|
|
1767
|
+
|
|
1768
|
+
Same gate as §022, on the account data DAC's logs.
|
|
1769
|
+
|
|
1770
|
+
```typescript
|
|
1771
|
+
await ctx.waitForBatches(ctx.accountDacSlug, [ctx.batchActive, ctx.batchSuspended])
|
|
1772
|
+
```
|
|
1773
|
+
|
|
1774
|
+
## 033 — run the KYC workflow
|
|
1775
|
+
|
|
1776
|
+
```typescript
|
|
1777
|
+
const kycRunResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.kycWorkflowSlug, {
|
|
1778
|
+
sql_where_clause: `payment_account_external_id IN ('${ctx.activeExternalId}', '${ctx.suspendedExternalId}')`,
|
|
1779
|
+
mode: 'live',
|
|
1780
|
+
manual_override: false,
|
|
1781
|
+
})
|
|
1782
|
+
const kycFired = await ctx.waitForFiredRun(datalakeSlug, kycRunResp.data.workflow_run_id)
|
|
1783
|
+
ctx.kycRunLogId = kycFired.workflowRunLogId
|
|
1784
|
+
ctx.kycRunBatchId = kycFired.batchId!
|
|
1785
|
+
|
|
1786
|
+
const kycDeadline = Date.now() + 240_000
|
|
1787
|
+
let kycStatus: string | null = null
|
|
1788
|
+
while (Date.now() < kycDeadline) {
|
|
1789
|
+
const { data: log } = await api.workflows.batchLogs.refresh(
|
|
1790
|
+
tenantSlug, datalakeSlug, ctx.kycWorkflowSlug, ctx.kycRunLogId,
|
|
1791
|
+
)
|
|
1792
|
+
kycStatus = log.status ?? null
|
|
1793
|
+
if (kycStatus && kycStatus !== 'pending') break
|
|
1794
|
+
await new Promise((r) => setTimeout(r, 2_000))
|
|
1795
|
+
}
|
|
1796
|
+
if (kycStatus === 'failed') throw new Error('KYC workflow run reached :failed')
|
|
1797
|
+
if (!kycStatus || kycStatus === 'pending') {
|
|
1798
|
+
throw new Error('KYC workflow run did not leave :pending within 240s')
|
|
1799
|
+
}
|
|
1800
|
+
```
|
|
1801
|
+
|
|
1802
|
+
## 034 — verify the filter routed: active passes, suspended is filtered
|
|
1803
|
+
|
|
1804
|
+
The active account passes the filter, so its WEL is `:executing` or
|
|
1805
|
+
`:completed`. The suspended one fails it, so its WEL is `:filtered`. Both
|
|
1806
|
+
passing — or both filtered — would mean the filter is not reading
|
|
1807
|
+
`event_dataset.status` at all.
|
|
1808
|
+
|
|
1809
|
+
```typescript
|
|
1810
|
+
const kycWelDeadline = Date.now() + 120_000
|
|
1811
|
+
let kycWels: Array<Record<string, unknown>> = []
|
|
1812
|
+
let kycByStatus: Record<string, number> = {}
|
|
1813
|
+
while (Date.now() < kycWelDeadline) {
|
|
1814
|
+
const { data: wfLogs } = await api.workflows.workflowLogs.list(tenantSlug, datalakeSlug, ctx.kycWorkflowSlug)
|
|
1815
|
+
kycWels = (wfLogs.data ?? [])
|
|
1816
|
+
.map((w) => w as Record<string, unknown>)
|
|
1817
|
+
.filter((w) => w.batch_id === ctx.kycRunBatchId)
|
|
1818
|
+
kycByStatus = {}
|
|
1819
|
+
for (const w of kycWels) {
|
|
1820
|
+
const st = (w.status as string | undefined) ?? 'unknown'
|
|
1821
|
+
kycByStatus[st] = (kycByStatus[st] ?? 0) + 1
|
|
1822
|
+
}
|
|
1823
|
+
const settled = (kycByStatus.completed ?? 0) + (kycByStatus.executing ?? 0)
|
|
1824
|
+
if (kycWels.length === 2 && settled === 1 && (kycByStatus.filtered ?? 0) === 1) break
|
|
1825
|
+
await new Promise((r) => setTimeout(r, 2_000))
|
|
1826
|
+
}
|
|
1827
|
+
if (kycWels.length !== 2) {
|
|
1828
|
+
throw new Error(`expected 2 KYC WELs for the run, got ${kycWels.length}`)
|
|
1829
|
+
}
|
|
1830
|
+
const kycPassCount = (kycByStatus.executing ?? 0) + (kycByStatus.completed ?? 0)
|
|
1831
|
+
if (kycPassCount !== 1) {
|
|
1832
|
+
throw new Error(`expected 1 pass-branch WEL, got ${kycPassCount} — ${JSON.stringify(kycByStatus)}`)
|
|
1833
|
+
}
|
|
1834
|
+
if ((kycByStatus.filtered ?? 0) !== 1) {
|
|
1835
|
+
throw new Error(`expected 1 :filtered WEL (the suspended account) — ${JSON.stringify(kycByStatus)}`)
|
|
1836
|
+
}
|
|
1837
|
+
```
|
|
1838
|
+
|
|
1839
|
+
## 035 — read the notification SMS back, and pull its portal token
|
|
1840
|
+
|
|
1841
|
+
The passing action fired immediately and persisted a `message` row with
|
|
1842
|
+
the fully-rendered body. Extract the `/t/<token>` shortlink the platform
|
|
1843
|
+
baked in when it minted the connected-app page token.
|
|
1844
|
+
|
|
1845
|
+
End-state assertion again, keyed the same way §025 is and for the same
|
|
1846
|
+
reason: on a repeat run the SMS was not re-sent, and the row from the first
|
|
1847
|
+
run is still the correct answer to "does this account have a notification
|
|
1848
|
+
with a portal link". The account's external id is what makes that row
|
|
1849
|
+
findable forever; the workflow's id is not.
|
|
1850
|
+
|
|
1851
|
+
```typescript
|
|
1852
|
+
const { data: kycSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
|
|
1853
|
+
search_query: `m.idempotency_key LIKE '${ctx.activeExternalId}-%'`,
|
|
1854
|
+
})
|
|
1855
|
+
if (kycSearch.status !== 'completed') {
|
|
1856
|
+
throw new Error(`message user-search status=${kycSearch.status} error=${kycSearch.error_message ?? '(none)'}`)
|
|
1857
|
+
}
|
|
1858
|
+
|
|
1859
|
+
const kycMsgDeadline = Date.now() + 45_000
|
|
1860
|
+
let kycMessages: Array<Record<string, unknown>> = []
|
|
1861
|
+
while (Date.now() < kycMsgDeadline && kycMessages.length === 0) {
|
|
1862
|
+
const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
|
|
1863
|
+
userSearchId: kycSearch.id!,
|
|
1864
|
+
dataAccessMode: 'raw',
|
|
1865
|
+
})
|
|
1866
|
+
kycMessages = (data.data ?? []) as Array<Record<string, unknown>>
|
|
1867
|
+
if (kycMessages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
|
|
1868
|
+
}
|
|
1869
|
+
if (kycMessages.length === 0) {
|
|
1870
|
+
throw new Error(
|
|
1871
|
+
`no KYC-notification SMS for account ${ctx.activeExternalId} within 45s — ` +
|
|
1872
|
+
`the action never fired, or it fired under a different idempotency key`,
|
|
1873
|
+
)
|
|
1874
|
+
}
|
|
1875
|
+
|
|
1876
|
+
const withLink = kycMessages
|
|
1877
|
+
.map((m) => String(m.body ?? ''))
|
|
1878
|
+
.find((body) => body.includes('/t/') && body.includes('Alvera PR notification'))
|
|
1879
|
+
if (!withLink) {
|
|
1880
|
+
throw new Error(`no rendered body with a /t/ link — bodies: ${JSON.stringify(kycMessages.map((m) => m.body))}`)
|
|
1881
|
+
}
|
|
1882
|
+
const tokenMatch = withLink.match(/\/t\/([A-Za-z0-9_-]+)/)
|
|
1883
|
+
if (!tokenMatch) {
|
|
1884
|
+
throw new Error(`no /t/<token> in rendered body: ${withLink}`)
|
|
1885
|
+
}
|
|
1886
|
+
ctx.kycShortPath = tokenMatch[1]!
|
|
1887
|
+
```
|
|
1888
|
+
|
|
1889
|
+
## 036 — resolve the deep link, and close the loop with tracking
|
|
1890
|
+
|
|
1891
|
+
The `/t/<token>` shortlink resolves through `connectedApps.resolvePage`,
|
|
1892
|
+
and the `route_path` must match the action's `connected_app_route`.
|
|
1893
|
+
Posting `opened_at` and `form_submitted_at` through
|
|
1894
|
+
`updateMessageTracking` then mirrors exactly what the portal frontend does
|
|
1895
|
+
when the holder opens it — closing the SMS-to-reply loop end to end.
|
|
1896
|
+
|
|
1897
|
+
```typescript
|
|
1898
|
+
const { data: resolved } = await api.connectedApps.resolvePage(
|
|
1899
|
+
tenantSlug, datalakeSlug, ctx.connectedAppSlug,
|
|
1900
|
+
{ short_path: ctx.kycShortPath, user_agent: 'cookbook-doctest/kyc-notification' },
|
|
1901
|
+
)
|
|
1902
|
+
if (resolved.route_path !== '/portal/kyc') {
|
|
1903
|
+
throw new Error(`resolvePage route_path mismatch: ${resolved.route_path}`)
|
|
1904
|
+
}
|
|
1905
|
+
if (!(resolved.message?.body ?? '').includes('Alvera PR notification')) {
|
|
1906
|
+
throw new Error('resolved page message body missing the notification copy')
|
|
1907
|
+
}
|
|
1908
|
+
|
|
1909
|
+
const now = new Date().toISOString()
|
|
1910
|
+
const { data: tracked } = await api.connectedApps.updateMessageTracking(
|
|
1911
|
+
tenantSlug, datalakeSlug, ctx.connectedAppSlug,
|
|
1912
|
+
{ short_path: ctx.kycShortPath, opened_at: now, form_submitted_at: now },
|
|
1913
|
+
)
|
|
1914
|
+
if (!tracked.message?.opened_at || !tracked.message?.form_submitted_at) {
|
|
1915
|
+
throw new Error('message tracking did not persist opened_at + form_submitted_at')
|
|
1916
|
+
}
|
|
1917
|
+
```
|
|
1918
|
+
|
|
1919
|
+
|
|
1920
|
+
## 037 — read one row back through all three lakes
|
|
1921
|
+
|
|
1922
|
+
This is what §007 was for. The same row, the same SQL, three modes — and
|
|
1923
|
+
the difference between them is the whole argument for turning tokenization
|
|
1924
|
+
on.
|
|
1925
|
+
|
|
1926
|
+
`screened_entity_name` is declared `tokenize`, so the raw lake holds the
|
|
1927
|
+
name and neither copy does. `screening_score` and `match_count` are
|
|
1928
|
+
`privacy_requirement: 'none'`, so they are **identical in all three** —
|
|
1929
|
+
which is the point rather than an aside. The agent in §016 disambiguates on
|
|
1930
|
+
score, match count and type; it never needed to know who was screened. A
|
|
1931
|
+
masking policy that also destroyed the signal would have made the agent
|
|
1932
|
+
useless, and one that preserved the name would have made it a liability.
|
|
1933
|
+
|
|
1934
|
+
Note that the mode is a parameter of the read, not a different slug: one
|
|
1935
|
+
lake is addressed, and `mode` selects which copy answers. The session's
|
|
1936
|
+
`data_access_mode` is a **ceiling**, so this step only works because §004
|
|
1937
|
+
built its client at `raw`; a tokenized-ceiling session asking for `raw` is
|
|
1938
|
+
refused rather than quietly downgraded.
|
|
1939
|
+
|
|
1940
|
+
```typescript
|
|
1941
|
+
const maskedReadSql =
|
|
1942
|
+
`SELECT compliance_screening_number, screened_entity_name, screening_score, match_count ` +
|
|
1943
|
+
`FROM ${ctx.screeningsTableName} ` +
|
|
1944
|
+
`WHERE compliance_screening_number = '${ctx.grayZoneNumber}'`
|
|
1945
|
+
|
|
1946
|
+
const readAs = async (mode: 'raw' | 'tokenized' | 'redacted') => {
|
|
1947
|
+
const { data } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
|
|
1948
|
+
sql: maskedReadSql,
|
|
1949
|
+
mode,
|
|
1950
|
+
})
|
|
1951
|
+
const row = (data.data ?? [])[0]
|
|
1952
|
+
if (!row) {
|
|
1953
|
+
throw new Error(`the gray-zone screening was not readable through the ${mode} lake`)
|
|
1954
|
+
}
|
|
1955
|
+
const cell = (column: string): unknown => row[data.meta.columns.indexOf(column)]
|
|
1956
|
+
return {
|
|
1957
|
+
name: cell('screened_entity_name'),
|
|
1958
|
+
score: Number(cell('screening_score')),
|
|
1959
|
+
matchCount: Number(cell('match_count')),
|
|
1960
|
+
}
|
|
1961
|
+
}
|
|
1962
|
+
|
|
1963
|
+
const rawRead = await readAs('raw')
|
|
1964
|
+
const tokenizedRead = await readAs('tokenized')
|
|
1965
|
+
const redactedRead = await readAs('redacted')
|
|
1966
|
+
|
|
1967
|
+
console.log(' raw →', JSON.stringify(rawRead))
|
|
1968
|
+
console.log(' tokenized →', JSON.stringify(tokenizedRead))
|
|
1969
|
+
console.log(' redacted →', JSON.stringify(redactedRead))
|
|
1970
|
+
|
|
1971
|
+
// The identity does NOT survive the copy.
|
|
1972
|
+
if (rawRead.name !== ctx.grayZoneName) {
|
|
1973
|
+
throw new Error(`the raw lake should hold the real name, got ${JSON.stringify(rawRead.name)}`)
|
|
1974
|
+
}
|
|
1975
|
+
if (tokenizedRead.name === ctx.grayZoneName) {
|
|
1976
|
+
throw new Error('the tokenized lake handed back the real name — tokenization is not in force')
|
|
1977
|
+
}
|
|
1978
|
+
if (redactedRead.name === ctx.grayZoneName) {
|
|
1979
|
+
throw new Error('the redacted lake handed back the real name — redaction is not in force')
|
|
1980
|
+
}
|
|
1981
|
+
|
|
1982
|
+
// The signal DOES. A masking policy that broke this would break the agent.
|
|
1983
|
+
for (const [mode, read] of [['tokenized', tokenizedRead], ['redacted', redactedRead]] as const) {
|
|
1984
|
+
if (read.score !== rawRead.score || read.matchCount !== rawRead.matchCount) {
|
|
1985
|
+
throw new Error(
|
|
1986
|
+
`the ${mode} lake changed a 'none' column: score ${read.score} vs ${rawRead.score}, ` +
|
|
1987
|
+
`match_count ${read.matchCount} vs ${rawRead.matchCount}`,
|
|
1988
|
+
)
|
|
1989
|
+
}
|
|
1990
|
+
}
|
|
1991
|
+
```
|
|
1992
|
+
|
|
1993
|
+
## 038 — how far the tokenized copy actually got
|
|
1994
|
+
|
|
1995
|
+
Every masked read above trusted the copy. This asks it directly, and on a
|
|
1996
|
+
compliance surface the question is not academic: an agent reading a copy that
|
|
1997
|
+
has not caught up is an agent deciding on a partial view of who was screened.
|
|
1998
|
+
|
|
1999
|
+
The addressing is the first trap. You name the **primary** lake plus a mode.
|
|
2000
|
+
A derived lake does have a slug, but that slug is not an address — the lookup
|
|
2001
|
+
resolves primaries only — so the walk asks with the tokenized lake's own slug
|
|
2002
|
+
and proves the refusal instead of assuming it.
|
|
2003
|
+
|
|
2004
|
+
The second trap is the state names. `finished_copy` has **not** finished: the
|
|
2005
|
+
first pass is done, but the changes made during that pass are still
|
|
2006
|
+
unapplied. Read `caught_up`, which exists precisely so nobody has to
|
|
2007
|
+
pattern-match the word. `label` is the same state in words and is what the
|
|
2008
|
+
platform's own Tokenization screen renders — reuse it rather than inventing a
|
|
2009
|
+
second vocabulary for the same five states.
|
|
2010
|
+
|
|
2011
|
+
`resync_empties` carries a rule worth asserting: `null` when a table is
|
|
2012
|
+
already caught up, because the platform will not spend a query per table
|
|
2013
|
+
answering a question about tables nobody will repair — and a non-empty array
|
|
2014
|
+
otherwise, naming the table itself first. `null` and `[]` are different
|
|
2015
|
+
answers here.
|
|
2016
|
+
|
|
2017
|
+
```typescript
|
|
2018
|
+
const { data: pair } = await api.datalakes.provisionTokenization(tenantSlug, datalakeSlug)
|
|
2019
|
+
const tokenizedLake = (pair.derived_datalakes ?? []).find((d) => d.type === 'tokenized')
|
|
2020
|
+
if (!tokenizedLake) throw new Error('no tokenized copy on the lake — §007 should have provisioned it')
|
|
2021
|
+
|
|
2022
|
+
const { data: tables } = await api.datalakes.derivedLakeTables(
|
|
2023
|
+
tenantSlug, datalakeSlug, 'tokenized',
|
|
2024
|
+
)
|
|
2025
|
+
if (tables.data.length === 0) throw new Error('the tokenized copy reported no tables at all')
|
|
2026
|
+
console.log(` ${tables.data.length} tables in the tokenized copy`)
|
|
2027
|
+
|
|
2028
|
+
for (const t of tables.data) {
|
|
2029
|
+
if (typeof t.caught_up !== 'boolean') {
|
|
2030
|
+
throw new Error(`${t.table}: caught_up is not a boolean, got ${JSON.stringify(t.caught_up)}`)
|
|
2031
|
+
}
|
|
2032
|
+
if (!t.label || t.label.length === 0) {
|
|
2033
|
+
throw new Error(`${t.table}: label is empty — it is what a screen renders`)
|
|
2034
|
+
}
|
|
2035
|
+
// The agreement that makes the boolean trustworthy: `caught_up` is true for
|
|
2036
|
+
// exactly one of the five states. `finished_copy` reading true here would be
|
|
2037
|
+
// the defect the boolean exists to prevent. This cannot be asserted against a
|
|
2038
|
+
// mock — the schema cannot express a cross-field constraint, so Prism draws
|
|
2039
|
+
// the two independently — which is exactly why it is asserted here.
|
|
2040
|
+
if (t.caught_up !== (t.state === 'caught_up')) {
|
|
2041
|
+
throw new Error(
|
|
2042
|
+
`${t.table}: state '${t.state}' disagrees with caught_up=${t.caught_up} — ` +
|
|
2043
|
+
`only 'caught_up' means arrived`,
|
|
2044
|
+
)
|
|
2045
|
+
}
|
|
2046
|
+
if (t.caught_up && t.resync_empties !== null && t.resync_empties !== undefined) {
|
|
2047
|
+
throw new Error(
|
|
2048
|
+
`${t.table} is caught up, so resync_empties should not be computed — got ` +
|
|
2049
|
+
JSON.stringify(t.resync_empties),
|
|
2050
|
+
)
|
|
2051
|
+
}
|
|
2052
|
+
if (!t.caught_up) {
|
|
2053
|
+
if (!Array.isArray(t.resync_empties) || t.resync_empties.length === 0) {
|
|
2054
|
+
throw new Error(`${t.table} is behind, so resync_empties should be a non-empty array`)
|
|
2055
|
+
}
|
|
2056
|
+
if (t.resync_empties[0] !== t.table) {
|
|
2057
|
+
throw new Error(`${t.table}: resync_empties should name the table itself first`)
|
|
2058
|
+
}
|
|
2059
|
+
}
|
|
2060
|
+
}
|
|
2061
|
+
ctx.copyTables = tables.data.map((t) => t.table)
|
|
2062
|
+
|
|
2063
|
+
// The derived lake's own slug is not a second way to name the same thing.
|
|
2064
|
+
let refusedDerivedSlug = false
|
|
2065
|
+
try {
|
|
2066
|
+
await api.datalakes.derivedLakeTables(tenantSlug, tokenizedLake.slug!, 'tokenized')
|
|
2067
|
+
} catch (err) {
|
|
2068
|
+
const status = (err as { _httpStatus?: number })._httpStatus
|
|
2069
|
+
if (status !== 404) throw err
|
|
2070
|
+
refusedDerivedSlug = true
|
|
2071
|
+
}
|
|
2072
|
+
if (!refusedDerivedSlug) {
|
|
2073
|
+
throw new Error("the derived lake's own slug was accepted as an address — it should 404")
|
|
2074
|
+
}
|
|
2075
|
+
```
|
|
2076
|
+
|
|
2077
|
+
## 039 — empty one table and wait for it back
|
|
2078
|
+
|
|
2079
|
+
`resyncTable` empties a table in the copy and copies it again from the start.
|
|
2080
|
+
It is the repair for a table that is stuck or wrong, and two of its properties
|
|
2081
|
+
are the kind that bite in production, so both are asserted rather than
|
|
2082
|
+
described.
|
|
2083
|
+
|
|
2084
|
+
**The blast radius is bigger than the table you name.** Postgres will not
|
|
2085
|
+
truncate a table with an inbound foreign key, regardless of whether the
|
|
2086
|
+
referencing tables hold any rows, so repairing a parent empties its whole
|
|
2087
|
+
dependent closure, transitively, in a single statement — and all of them are
|
|
2088
|
+
re-copied by the same run. `legal_entities` is the target here on purpose:
|
|
2089
|
+
§029's contract pair resolved every screened subject through it, and
|
|
2090
|
+
`legal_entity_identifications` points at it, so `emptied` genuinely names more
|
|
2091
|
+
than one table. Any surface offering this button should read `resync_empties`
|
|
2092
|
+
and say the count before it is pressed.
|
|
2093
|
+
|
|
2094
|
+
**`202` means started.** There is no completion callback, so the walk polls
|
|
2095
|
+
`derivedLakeTables` until every emptied table reads `caught_up` again. That
|
|
2096
|
+
poll is also what makes this a proof rather than a dispatch.
|
|
2097
|
+
|
|
2098
|
+
The gate goes first. `table` is checked against what the copy carries or the
|
|
2099
|
+
original can send, and anything else is a `422` with nothing emptied — the
|
|
2100
|
+
one thing standing between a typo and an emptied table. Proving it *after*
|
|
2101
|
+
running the real repair would be checking the lock from inside the house.
|
|
2102
|
+
|
|
2103
|
+
```typescript
|
|
2104
|
+
const TARGET = 'legal_entities'
|
|
2105
|
+
if (!(ctx.copyTables as string[]).includes(TARGET)) {
|
|
2106
|
+
throw new Error(`${TARGET} is not in the copy; it carries: ${(ctx.copyTables as string[]).join(', ')}`)
|
|
2107
|
+
}
|
|
2108
|
+
|
|
2109
|
+
let refusedBadTable = false
|
|
2110
|
+
try {
|
|
2111
|
+
await api.datalakes.resyncTable(tenantSlug, datalakeSlug, 'tokenized', {
|
|
2112
|
+
table: 'no_such_table_here',
|
|
2113
|
+
})
|
|
2114
|
+
} catch (err) {
|
|
2115
|
+
const status = (err as { _httpStatus?: number })._httpStatus
|
|
2116
|
+
if (status !== 422) throw err
|
|
2117
|
+
refusedBadTable = true
|
|
2118
|
+
}
|
|
2119
|
+
if (!refusedBadTable) throw new Error('an unknown table name was accepted — the gate is not closed')
|
|
2120
|
+
|
|
2121
|
+
const { data: repair } = await api.datalakes.resyncTable(
|
|
2122
|
+
tenantSlug, datalakeSlug, 'tokenized', { table: TARGET },
|
|
2123
|
+
)
|
|
2124
|
+
if (repair.table !== TARGET) throw new Error(`repaired ${repair.table}, asked for ${TARGET}`)
|
|
2125
|
+
if (repair.mode !== 'tokenized') throw new Error(`repaired the ${repair.mode} copy, asked for tokenized`)
|
|
2126
|
+
if (repair.status !== 'copying') throw new Error(`expected status 'copying', got ${repair.status}`)
|
|
2127
|
+
if (!repair.emptied.includes(TARGET)) {
|
|
2128
|
+
throw new Error(`emptied ${JSON.stringify(repair.emptied)} — the named table is not among them`)
|
|
2129
|
+
}
|
|
2130
|
+
console.log(` emptied ${repair.emptied.length} table(s): ${repair.emptied.join(', ')}`)
|
|
2131
|
+
|
|
2132
|
+
// Identifications point at legal entities, so the closure cannot be just one.
|
|
2133
|
+
if (repair.emptied.length < 2) {
|
|
2134
|
+
throw new Error(
|
|
2135
|
+
`${TARGET} has dependents, so a repair should empty more than itself — got ` +
|
|
2136
|
+
JSON.stringify(repair.emptied),
|
|
2137
|
+
)
|
|
2138
|
+
}
|
|
2139
|
+
|
|
2140
|
+
const RESYNC_TIMEOUT_MS = 5 * 60_000
|
|
2141
|
+
const resyncDeadline = Date.now() + RESYNC_TIMEOUT_MS
|
|
2142
|
+
let backOnline = false
|
|
2143
|
+
while (Date.now() < resyncDeadline && !backOnline) {
|
|
2144
|
+
const { data: after } = await api.datalakes.derivedLakeTables(
|
|
2145
|
+
tenantSlug, datalakeSlug, 'tokenized',
|
|
2146
|
+
)
|
|
2147
|
+
backOnline = (repair.emptied as string[]).every(
|
|
2148
|
+
(name) => after.data.find((t) => t.table === name)?.caught_up === true,
|
|
2149
|
+
)
|
|
2150
|
+
if (!backOnline) await new Promise((r) => setTimeout(r, 5_000))
|
|
2151
|
+
}
|
|
2152
|
+
if (!backOnline) {
|
|
2153
|
+
throw new Error(`the repaired tables did not come back within ${RESYNC_TIMEOUT_MS / 1000}s`)
|
|
2154
|
+
}
|
|
2155
|
+
console.log(' every emptied table is caught up again')
|
|
2156
|
+
```
|
|
2157
|
+
|
|
2158
|
+
# Outcome
|
|
2159
|
+
|
|
2160
|
+
One tenant carries the whole payments-compliance surface: a raw datalake
|
|
2161
|
+
with its tokenized and redacted copies, two generic tables, two contract
|
|
2162
|
+
pairs split across four Data Activation Clients, and two agentic workflows
|
|
2163
|
+
that between them screen, decide, notify and track — plus the copy's own
|
|
2164
|
+
arrival state, read back per table and repaired.
|
|
2165
|
+
|
|
2166
|
+
Two things are worth carrying out of this walk. **The pair must not run
|
|
2167
|
+
concurrently** (§020) — one writer per subject, or the unique index that
|
|
2168
|
+
guards against duplicate entities refuses your row on the only run that
|
|
2169
|
+
matters, the first. And **assert end states, not transitions** (§024) — an
|
|
2170
|
+
idempotency key that has already fired will not fire again, which is the
|
|
2171
|
+
product working, not the walk failing.
|
|
2172
|
+
|
|
2173
|
+
Running this walk again reuses all of it.
|
|
2174
|
+
|
|
2175
|
+
# See also
|
|
2176
|
+
|
|
2177
|
+
- `datalakes.md` — the create body, and what the derived pair is for
|
|
2178
|
+
- `workflows.md` — scheduling versus firing, and the accessor tables
|
|
2179
|
+
- `interoperability_contracts.md` — the contract pair, and the
|
|
2180
|
+
reserved `legal_entity_id` stamp you must not declare
|