@alvera-ai/platform-sdk 0.17.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/.agent/AGENTS.md +82 -144
  2. package/.agent/account_management.md +2 -2
  3. package/.agent/action_logs.md +4 -4
  4. package/.agent/ai_agents.md +28 -21
  5. package/.agent/ai_sandbox.md +49 -39
  6. package/.agent/connected_apps.md +3 -3
  7. package/.agent/cookbook/_fixtures/README.md +1 -1
  8. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_generic_table.liquid +1 -1
  9. package/.agent/cookbook/_fixtures/organic-marketing/_lead_submissions_foundation_legal_entity.liquid +80 -0
  10. package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_mdm.liquid +2 -1
  11. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_generic_table.liquid +57 -0
  12. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_legal_entity.liquid +30 -0
  13. package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_mdm.liquid +44 -0
  14. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_generic_table.liquid +57 -0
  15. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_legal_entity.liquid +36 -0
  16. package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_mdm.liquid +41 -0
  17. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_generic_table.liquid +70 -0
  18. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_legal_entity.liquid +52 -0
  19. package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_mdm.liquid +42 -0
  20. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_generic_table.liquid +38 -0
  21. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_legal_entity.liquid +48 -0
  22. package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_mdm.liquid +49 -0
  23. package/.agent/cookbook/organic-marketing.md +2801 -0
  24. package/.agent/cookbook/payments-compliance.md +2180 -0
  25. package/.agent/cookbook/primary-care.md +2175 -0
  26. package/.agent/cookbook/subscription-saas.md +2403 -0
  27. package/.agent/data_activation_clients.md +65 -52
  28. package/.agent/datalakes.md +338 -171
  29. package/.agent/errors.md +3 -3
  30. package/.agent/generic_tables.md +151 -62
  31. package/.agent/interoperability_contracts.md +57 -22
  32. package/.agent/mdm.md +136 -153
  33. package/.agent/messages.md +36 -34
  34. package/.agent/mock-services.md +1 -1
  35. package/.agent/mutations.md +2 -2
  36. package/.agent/templates.md +14 -13
  37. package/.agent/tools.md +63 -21
  38. package/.agent/type_naming.md +13 -13
  39. package/.agent/workflows.md +99 -53
  40. package/README.md +2 -2
  41. package/dist/bin/platform-sdk.mjs +33 -47
  42. package/dist/bin/platform-sdk.mjs.map +1 -1
  43. package/dist/index.d.mts +565 -379
  44. package/dist/index.d.mts.map +1 -1
  45. package/dist/index.mjs +494 -59
  46. package/dist/index.mjs.map +1 -1
  47. package/package.json +4 -3
  48. package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +0 -88
  49. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +0 -47
  50. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +0 -24
  51. package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +0 -38
  52. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_compliance_screening.liquid +0 -59
  53. package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_mdm.liquid +0 -36
  54. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_mdm.liquid +0 -30
  55. package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_payment_account.liquid +0 -55
  56. package/.agent/cookbook/_fixtures/subscription/_customers_subscription_mdm.liquid +0 -20
  57. package/.agent/cookbook/_setup/foundation.md +0 -359
  58. package/.agent/cookbook/_setup/healthcare.md +0 -361
  59. package/.agent/cookbook/_setup/payments.md +0 -365
  60. package/.agent/cookbook/_setup/subscription.md +0 -364
  61. package/.agent/cookbook/action-status-updaters.md +0 -278
  62. package/.agent/cookbook/ai-agent-invoke.md +0 -279
  63. package/.agent/cookbook/appointment-review-sms-workflow.md +0 -801
  64. package/.agent/cookbook/birthday-greeting-sms-trigger.md +0 -696
  65. package/.agent/cookbook/bulk-ingest.md +0 -302
  66. package/.agent/cookbook/contact-us-triage-with-llm.md +0 -663
  67. package/.agent/cookbook/dunning-sms-for-delinquent.md +0 -659
  68. package/.agent/cookbook/generic-tables.md +0 -244
  69. package/.agent/cookbook/invite-team.md +0 -200
  70. package/.agent/cookbook/kyc-notification-on-account-activation.md +0 -661
  71. package/.agent/cookbook/marketing-campaign-send.md +0 -1044
  72. package/.agent/cookbook/paginated-restapi-poller.md +0 -383
  73. package/.agent/cookbook/rest-fetch.md +0 -273
  74. package/.agent/cookbook/sanctions-screening-with-agent-review.md +0 -773
  75. package/.agent/cookbook/score-leads-with-llm-categorization.md +0 -665
  76. package/.agent/cookbook/system-templates.md +0 -165
  77. package/.agent/cookbook/talk-to-data.md +0 -178
  78. package/.agent/cookbook/triage-prospects-by-priority.md +0 -571
  79. package/.agent/cookbook/welcome-sms-for-customers.md +0 -647
  80. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/memorandum-of-association-01.png +0 -0
  81. /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/sample_two_page.pdf +0 -0
  82. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/_customers_subscription_customer.liquid +0 -0
  83. /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/stripe_customers_batch1.csv +0 -0
@@ -0,0 +1,2801 @@
1
+ ---
2
+ title: "Organic marketing: score, greet, campaign — on one raw lake"
3
+ summary: The whole organic-marketing surface in one walk. Stand up a tenant and a raw datalake, then run the scenarios end to end — an LLM agent banding inbound leads into a four-way SMS fan-out, a pure-Liquid birthday trigger that schedules a greeting a year out and closes the reply loop through a connected app, and an A/B campaign that gates on suppression, splits inside the workflow and dispatches on SMS and email in one run. Ends with delivery reconciliation and a natural-language read of the lake.
4
+ use_case: organic-marketing
5
+ slug: organic-marketing
6
+ vitest_source:
7
+ - integration-tests/tests/organic-marketing/agent-driven-workflow.test.ts
8
+ - integration-tests/tests/organic-marketing/generic-tables.test.ts
9
+ - integration-tests/tests/organic-marketing/manual-upload-tool.test.ts
10
+ - integration-tests/tests/workspace/bootstrap.test.ts
11
+ status: draft
12
+ ---
13
+
14
+ # Problem
15
+
16
+ A marketing team's work is three jobs that share one audience. Leads
17
+ arrive from forms and most of them are noise, so something has to read each
18
+ one and decide whether a human should. Contacts have birthdays, and a
19
+ greeting has to be scheduled once and fire a year later without anyone
20
+ remembering. And campaigns go out to a roster of businesses, split A/B, on
21
+ whichever channel the recipient can actually be reached on.
22
+
23
+ All three are the same shape — rows land, a filter decides which ones
24
+ matter, and something is sent about the ones that do. This walk builds all
25
+ three on one tenant, because a marketing team does not run three platforms.
26
+
27
+ **This surface stays raw.** No tokenized or redacted copy is provisioned,
28
+ and that is a decision rather than an omission: everything read back here
29
+ is a question about this walk's own rows, and the lead's own message — not
30
+ the lead's name — is what the classifier reads. If you want a model held
31
+ away from identities, that is what `payments-compliance` §007 shows, and it
32
+ is one call away from here.
33
+
34
+ # Composition
35
+
36
+ | Resource | Why it exists here |
37
+ |-------------------------------------|---------------------------------------------|
38
+ | Tenant + raw datalake | The marketing team's own world |
39
+ | SMS tool, email tool, LLM tool | Shared by every scenario below |
40
+ | Score Lead AI agent | Bands inbound leads |
41
+ | Three generic tables | Leads, the roster, and the audience |
42
+ | Three workflows | Band fan-out, birthday greeting, campaign |
43
+ | Connected app | The reply loop, twice |
44
+ | Action status updater | Delivery outcomes, reconciled on a cron |
45
+
46
+ # Walkthrough
47
+
48
+ This cookbook is **self-reliant**: it stands up everything it uses and
49
+ depends on no other file. It is also **idempotent** — every creating step
50
+ looks first and creates only what is missing, so running it twice costs
51
+ what running it once cost. That is not a nicety. Nothing in the platform
52
+ reclaims an abandoned datalake, and a walk that mints a fresh tenant per
53
+ run leaves every previous run's lakes behind forever.
54
+
55
+ ## 001 — sign in as root, and make sure the admin user exists
56
+
57
+ Authenticate as the platform's root admin (`admin@dev.local` /
58
+ `devpassword` in local dev) via the tenantless bootstrap login — keyless by
59
+ structural necessity, since no tenant exists yet to scope a key to, and a
60
+ dev/test-only surface.
61
+
62
+ Then make sure the admin user this walk runs as exists. **The email is
63
+ stable, not per-run**, which is what makes the step idempotent — and it
64
+ means the second run finds the user already signed up. A duplicate signup
65
+ is refused, so the refusal is caught and inspected: if it says the email is
66
+ taken, that is the idempotent path and the walk continues. Any other
67
+ failure is re-raised, because swallowing it would turn a real auth problem
68
+ into a confusing failure three steps later.
69
+
70
+ ```typescript
71
+ ctx.rootSession = await createBootstrapSession({
72
+ baseUrl: process.env.ALVERA_BASE_URL!,
73
+ email: process.env.ALVERA_ROOT_EMAIL!,
74
+ password: process.env.ALVERA_ROOT_PASSWORD!,
75
+ })
76
+ ctx.rootApi = createIsolatedPlatformApi({
77
+ baseUrl: process.env.ALVERA_BASE_URL!,
78
+ sessionToken: ctx.rootSession.sessionToken,
79
+ apiKey: '',
80
+ })
81
+
82
+ ctx.marketerEmail = 'cookbook-organic-marketing@dev.local'
83
+ ctx.marketerPassword = 'CookbookPass1!'
84
+
85
+ try {
86
+ const signUpResp = await ctx.rootApi.admin.signUp({
87
+ email: ctx.marketerEmail,
88
+ password: ctx.marketerPassword,
89
+ first_name: 'Cookbook',
90
+ last_name: 'Marketing',
91
+ })
92
+ await ctx.rootApi.admin.confirmUser(signUpResp.data.id!)
93
+ } catch (err) {
94
+ // Already provisioned by a previous run. Confirm that is what happened
95
+ // rather than assuming it — a genuine signup failure must not be read as
96
+ // "already there".
97
+ const detail = JSON.stringify((err as { errors?: unknown }).errors ?? err)
98
+ if (!/taken|already|exist/i.test(detail)) throw err
99
+ }
100
+ ```
101
+
102
+ ## 002 — the find-or-create helper every later step uses
103
+
104
+ Idempotence is one question asked over and over: *is this already here?*
105
+ Rather than answer it a dozen different ways, the walk defines it once.
106
+
107
+ `ensure` takes a label, a lookup and a create. It runs the lookup, returns
108
+ what it finds, and only creates when the lookup comes back empty. The label
109
+ is not decoration — when a run reuses something you did not expect it to,
110
+ the log line naming it is how you find out.
111
+
112
+ The helper lives on `ctx` rather than as a bare function because each
113
+ numbered step compiles into its own `it()` block, so a plain `function`
114
+ here would not be in scope for the steps that call it.
115
+
116
+ ```typescript
117
+ ctx.ensure = async <T>(
118
+ label: string,
119
+ find: () => Promise<T | undefined>,
120
+ create: () => Promise<T>,
121
+ ): Promise<T> => {
122
+ const existing = await find()
123
+ if (existing !== undefined) {
124
+ console.log(` ↻ reusing ${label}`)
125
+ return existing
126
+ }
127
+ console.log(` + creating ${label}`)
128
+ return await create()
129
+ }
130
+ ```
131
+
132
+ ## 003 — the tenant
133
+
134
+ The marketer signs in without a tenant scope — they may not belong to one
135
+ yet — and the walk finds or creates the marketing tenant by its stable
136
+ name. The server derives the slug; capture it, because every later call is
137
+ addressed by it.
138
+
139
+ ```typescript
140
+ ctx.marketerTenantlessSession = await createBootstrapSession({
141
+ baseUrl: process.env.ALVERA_BASE_URL!,
142
+ email: ctx.marketerEmail,
143
+ password: ctx.marketerPassword,
144
+ })
145
+ ctx.marketerTenantlessApi = createIsolatedPlatformApi({
146
+ baseUrl: process.env.ALVERA_BASE_URL!,
147
+ sessionToken: ctx.marketerTenantlessSession.sessionToken,
148
+ apiKey: '',
149
+ })
150
+
151
+ const TENANT_NAME = 'Cookbook Organic Marketing'
152
+
153
+ const tenant = await ctx.ensure(
154
+ `tenant ${TENANT_NAME}`,
155
+ async () => {
156
+ const { data } = await ctx.marketerTenantlessApi.tenants.list()
157
+ return (data.data ?? []).find((t: { name?: string }) => t.name === TENANT_NAME)
158
+ },
159
+ async () => {
160
+ const { data } = await ctx.marketerTenantlessApi.tenants.create({ name: TENANT_NAME })
161
+ return data
162
+ },
163
+ )
164
+ tenantSlug = tenant.slug!
165
+ ```
166
+
167
+ ## 004 — the tenant-scoped client
168
+
169
+ A tenant-scoped login requires `X-API-Key`, so a key has to exist before
170
+ the marketer can sign in against the tenant. Mint one through the
171
+ platform-admin side door with the root bearer; in the web console this is
172
+ *Settings → API Keys*.
173
+
174
+ `data_access_mode: 'raw'` is the ceiling, and on this surface it is the
175
+ only tier there is — no derived lake is provisioned, so raw is what every
176
+ read answers from.
177
+
178
+ **This is the one step in the walk that is not idempotent**, and it is
179
+ worth knowing why rather than discovering it. There is no endpoint that
180
+ lists a tenant's API keys, so there is nothing to look the existing one up
181
+ with — `ensure` has no lookup to run. Each run therefore mints another key.
182
+ A key row is cheap where a datalake is not, so the walk accepts it; if you
183
+ are counting rows in a shared environment, this is the one to count.
184
+
185
+ ```typescript
186
+ const { data: mintedKey } = await ctx.rootApi.admin.createTenantApiKey(tenantSlug, {
187
+ name: 'Cookbook Organic Marketing Key',
188
+ data_access_mode: 'raw',
189
+ })
190
+ ctx.tenantApiKey = mintedKey.api_key
191
+
192
+ const marketerTenantSession = await createSession({
193
+ baseUrl: process.env.ALVERA_BASE_URL!,
194
+ email: ctx.marketerEmail,
195
+ password: ctx.marketerPassword,
196
+ tenantSlug,
197
+ apiKey: ctx.tenantApiKey,
198
+ })
199
+ api = createIsolatedPlatformApi({
200
+ baseUrl: process.env.ALVERA_BASE_URL!,
201
+ sessionToken: marketerTenantSession.sessionToken,
202
+ apiKey: ctx.tenantApiKey,
203
+ })
204
+ ```
205
+
206
+ ## 005 — the datalake
207
+
208
+ One datalake, created `raw`, and on this surface that is the whole story —
209
+ there is no derived pair to provision and nothing to wait for beyond the
210
+ migration in §006.
211
+
212
+ The lookup still filters on `type === 'raw'`. That costs nothing here and
213
+ keeps the step correct if anyone ever turns tokenization on: from that
214
+ moment `datalakes.list` returns three rows, two of which are copies that
215
+ must never be mistaken for the primary.
216
+
217
+ Local-dev defaults match the seeded `dev.exs` setup — `postgres` on
218
+ `localhost:5432`, database `alvera_dev_foundation`, LocalStack S3 on
219
+ `localhost:4566` — and the schema name is stable, because a fresh schema
220
+ per run is the same leak as a fresh tenant per run.
221
+
222
+ ```typescript
223
+ const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
224
+ const DB_SCHEMA = 'cookbook_organic_marketing'
225
+ const LAKE_NAME = 'Cookbook Organic Marketing Datalake'
226
+
227
+ const S3 = {
228
+ cloud_storage_type: 'aws' as const,
229
+ region: 'us-east-1',
230
+ access_key_id: 'test',
231
+ secret_access_key: 'test',
232
+ endpoint: 'http://localhost:4566',
233
+ }
234
+
235
+ const datalake = await ctx.ensure(
236
+ `datalake ${LAKE_NAME}`,
237
+ async () => {
238
+ const { data } = await api.datalakes.list(tenantSlug)
239
+ return (data.data ?? []).find(
240
+ (l: { name?: string; type?: string }) => l.name === LAKE_NAME && l.type === 'raw',
241
+ )
242
+ },
243
+ async () => {
244
+ const { data } = await api.datalakes.create(tenantSlug, {
245
+ name: LAKE_NAME,
246
+ description: 'Organic marketing datalake provisioned by the cookbook doctest.',
247
+ timezone: 'America/New_York',
248
+ pool_size: 3,
249
+ type: 'raw',
250
+
251
+ db_writer_host: DB.host,
252
+ db_writer_port: DB.port,
253
+ db_writer_name: DB.name,
254
+ db_writer_schema: DB_SCHEMA,
255
+ db_writer_auth_method: 'password',
256
+ db_writer_user: DB.user,
257
+ db_writer_pass: DB.pass,
258
+ db_writer_enable_ssl: false,
259
+ db_reader_host: DB.host,
260
+ db_reader_port: DB.port,
261
+ db_reader_name: DB.name,
262
+ db_reader_schema: DB_SCHEMA,
263
+ db_reader_auth_method: 'password',
264
+ db_reader_user: DB.user,
265
+ db_reader_pass: DB.pass,
266
+ db_reader_enable_ssl: false,
267
+
268
+ cloud_storage: { ...S3, bucket: 'alvera-platform-dev', base_path: 'cookbook/organic-marketing' },
269
+ })
270
+ return data
271
+ },
272
+ )
273
+ datalakeSlug = datalake.slug!
274
+ ctx.datalakeId = datalake.id!
275
+ ```
276
+
277
+ ## 006 — run the migrations, and wait for ready
278
+
279
+ `datalakes.create` persists the row at `status: 'new'`; it does not run the
280
+ schema DDL. Migration is triggered separately so the operator decides when
281
+ the potentially-slow part happens. `datalakes.migrate` enqueues the job and
282
+ returns immediately with `status: 'enqueued'`; the poll after it is what
283
+ waits for the worker.
284
+
285
+ Migrating is safe to repeat, which is what lets this step stay unguarded on
286
+ a second run. The wait exits the moment the lake reports `ready`, so the
287
+ five-minute ceiling is only ever paid in failure.
288
+
289
+ ```typescript
290
+ const migrateResp = await api.datalakes.migrate(tenantSlug, datalakeSlug)
291
+ if (migrateResp.data.status !== 'enqueued') {
292
+ throw new Error(`datalake migration not enqueued (status: ${migrateResp.data.status})`)
293
+ }
294
+
295
+ const READY_TIMEOUT_MS = 5 * 60_000
296
+ const readyDeadline = Date.now() + READY_TIMEOUT_MS
297
+ let datalakeStatus: string | undefined
298
+ while (Date.now() < readyDeadline) {
299
+ const { data } = await api.datalakes.get(tenantSlug, ctx.datalakeId)
300
+ datalakeStatus = data.status
301
+ if (datalakeStatus === 'ready') break
302
+ await new Promise((r) => setTimeout(r, 5_000))
303
+ }
304
+ if (datalakeStatus !== 'ready') {
305
+ throw new Error(`datalake did not reach :ready within ${READY_TIMEOUT_MS}ms (last: ${datalakeStatus})`)
306
+ }
307
+ ```
308
+
309
+ ## 007 — three helpers the scenarios below share
310
+
311
+ **`waitForFiredRun`.** `workflows.run` only *schedules* a run. It returns
312
+ immediately with a `workflow_run_id`, and the `workflow_run_log_id` and
313
+ `batch_id` a scenario needs are written later, when the run actually fires.
314
+ Two traps live in that gap:
315
+
316
+ - **Poll until `workflow_run_log_id` is a string — not until `status`
317
+ leaves `'scheduled'`.** Those are different moments; the run reaches
318
+ `processing` first and writes the log id a beat later. A predicate on
319
+ status alone releases you to read a `null`, and because
320
+ `typeof null === 'object'` the symptom is a baffling *"expected string,
321
+ got object"* rather than an obvious nil.
322
+ - **Raise on `failed` carrying `failure_reason`** rather than polling to
323
+ the deadline. A scenario blocked on a run that will never fire should say
324
+ why on the first read, not thirty seconds later behind a generic timeout.
325
+
326
+ **`waitForBatches`.** Ingestion is async: `ingest` returns a `batch_id` the
327
+ instant the rows are accepted, and the per-row jobs drain afterwards. This
328
+ polls a client's activation logs until every named batch has actually
329
+ written.
330
+
331
+ **Gate on `dataset_updated`, never on `rows_ingested`.** They answer
332
+ different questions — received versus written — and a batch whose row is
333
+ refused reports `rows_ingested: 1, dataset_updated: 0, status: 'partial'`
334
+ with the reason in `error`. A gate on `rows_ingested` calls that green, the
335
+ next step then runs against a table with nothing in it, and the failure
336
+ surfaces three steps later as *"expected 3 execution logs, got 0"* — which
337
+ reads like a broken workflow rather than a row that never landed.
338
+
339
+ **`deployGenericTable`.** Creating a generic table does not deploy it; it
340
+ rests at `status: 'new'` until a migration runs. The wait that means
341
+ anything asks for the table the same way the next step is about to, and
342
+ retries until it stops erroring — polling the table's own
343
+ `status: 'deployed'` goes true earlier and proves less.
344
+
345
+ ```typescript
346
+ ctx.waitForFiredRun = async (
347
+ runDatalakeSlug: string,
348
+ runId: string,
349
+ timeoutMs = 120_000,
350
+ ): Promise<{ workflowRunLogId: string; batchId: string | null }> => {
351
+ const deadline = Date.now() + timeoutMs
352
+ let lastStatus: string | undefined
353
+ while (Date.now() < deadline) {
354
+ const { data } = await api.workflowRuns.get(tenantSlug, runDatalakeSlug, runId)
355
+ lastStatus = data.status
356
+ if (data.status === 'failed') {
357
+ throw new Error(`workflow run ${runId} failed: ${data.failure_reason ?? 'no failure_reason given'}`)
358
+ }
359
+ if (typeof data.workflow_run_log_id === 'string') {
360
+ return { workflowRunLogId: data.workflow_run_log_id, batchId: data.batch_id ?? null }
361
+ }
362
+ await new Promise((r) => setTimeout(r, 1_000))
363
+ }
364
+ throw new Error(`workflow run ${runId} never fired within ${timeoutMs}ms (last status: ${lastStatus})`)
365
+ }
366
+
367
+ ctx.waitForBatches = async (
368
+ dacSlug: string,
369
+ batchIds: readonly string[],
370
+ timeoutMs = 90_000,
371
+ ): Promise<void> => {
372
+ const targets = new Set(batchIds)
373
+ const deadline = Date.now() + timeoutMs
374
+ let greenCount = 0
375
+ while (Date.now() < deadline) {
376
+ const { data } = await api.dataActivationClients.logs.list(tenantSlug, datalakeSlug, dacSlug)
377
+ const green = new Set<string>()
378
+ for (const row of (data.data ?? []) as Array<Record<string, unknown>>) {
379
+ const b = row.batch_id
380
+ if (typeof b !== 'string' || !targets.has(b)) continue
381
+ if (row.status === 'partial' || row.status === 'failed') {
382
+ throw new Error(
383
+ `batch ${b} on ${dacSlug} did not persist its rows (status: ${row.status}): ` +
384
+ `${String(row.error ?? 'no reason given')}`,
385
+ )
386
+ }
387
+ if (typeof row.dataset_updated !== 'number' || row.dataset_updated < 1) continue
388
+ const files = row.output_files
389
+ if (!Array.isArray(files) || files.length === 0) continue
390
+ green.add(b)
391
+ }
392
+ greenCount = green.size
393
+ if (greenCount === targets.size) return
394
+ await new Promise((r) => setTimeout(r, 1_000))
395
+ }
396
+ throw new Error(`only ${greenCount}/${targets.size} batches on ${dacSlug} persisted within ${timeoutMs}ms`)
397
+ }
398
+
399
+ ctx.deployGenericTable = async (tableName: string, timeoutMs = 120_000): Promise<void> => {
400
+ await api.datalakes.migrate(tenantSlug, datalakeSlug)
401
+ const deadline = Date.now() + timeoutMs
402
+ let lastError: unknown
403
+ while (Date.now() < deadline) {
404
+ try {
405
+ await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
406
+ sql: `SELECT 1 FROM ${tableName} LIMIT 1`,
407
+ mode: 'raw',
408
+ })
409
+ return
410
+ } catch (err) {
411
+ lastError = err
412
+ await new Promise((r) => setTimeout(r, 1_000))
413
+ }
414
+ }
415
+ throw new Error(
416
+ `generic table ${tableName} was not readable within ${timeoutMs}ms ` +
417
+ `(last error: ${lastError instanceof Error ? lastError.message : String(lastError)})`,
418
+ )
419
+ }
420
+ ```
421
+
422
+ ## 008 — the SMS tool
423
+
424
+ One SMS tool dispatches every outbound message in this walk — four lead
425
+ bands, a birthday greeting and a campaign variant. **The per-message body
426
+ lives on the workflow action, never on the tool**, which is what lets one
427
+ tool serve all of them. `intent: 'sms'` tags it for workflow-action use.
428
+
429
+ Local dev points at LocalStack's SNS on `:4566`, so nothing leaves the
430
+ machine.
431
+
432
+ ```typescript
433
+ const smsTool = await ctx.ensure(
434
+ 'SMS tool',
435
+ async () => {
436
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
437
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Marketing SMS Tool')
438
+ },
439
+ async () => {
440
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
441
+ name: 'Cookbook Marketing SMS Tool',
442
+ description: 'SNS-backed SMS dispatcher for every marketing scenario, wired to LocalStack.',
443
+ intent: 'sms',
444
+ status: 'active',
445
+ datalake_id: ctx.datalakeId,
446
+ body: {
447
+ tool_body_type: 'sns',
448
+ auth_method: 'access_key',
449
+ region: 'us-east-1',
450
+ phone_number: '+15551234567',
451
+ endpoint_url: 'http://localhost:4566',
452
+ access_key_id: 'test',
453
+ secret_access_key: 'test',
454
+ },
455
+ })
456
+ return data
457
+ },
458
+ )
459
+ toolId = smsTool.id!
460
+ ctx.smsToolId = smsTool.id!
461
+ ```
462
+
463
+ ## 009 — the LLM tool
464
+
465
+ The Score Lead agent calls a chat-completion endpoint to classify each row.
466
+ `intent: 'llm_enrichment'` distinguishes it from the SMS tool above.
467
+
468
+ It is a **provider adapter**, and the two halves are the thing to read.
469
+ `base_body` authors the provider's request — here Ollama's native
470
+ `/api/chat` shape with `think: false` and a `format` schema so the model
471
+ returns clean, schema-constrained JSON. `response_extractor` maps the
472
+ provider's envelope back to the canonical `{ output_json, … }` the platform
473
+ reads, and its `output_schema` is **required** for an `llm_enrichment`
474
+ tool. The `api_key` / `auth_method` pair satisfies the REST tool's schema
475
+ even though Ollama ignores the header.
476
+
477
+ ```typescript
478
+ const ENRICHMENT_OUTPUT_SCHEMA = {
479
+ type: 'object',
480
+ properties: {
481
+ output_json: {},
482
+ input_tokens: { type: ['integer', 'null'] },
483
+ output_tokens: { type: ['integer', 'null'] },
484
+ total_tokens: { type: ['integer', 'null'] },
485
+ explanation: { type: ['string', 'null'] },
486
+ },
487
+ required: ['output_json'],
488
+ }
489
+
490
+ const OLLAMA_BASE_BODY =
491
+ '{"model": "{{ model }}", "messages": [{"role": "user", "content": "{{ rendered_prompt | json_escape }}", "images": [{% for img in images %}{% unless forloop.first %}, {% endunless %}"{{ img.data }}"{% endfor %}]}], "stream": false, "think": false, "options": {"temperature": {{ temperature }}, "num_predict": {{ max_tokens }}, "num_ctx": 40960}, "format": {{ schema | to_json }}}'
492
+
493
+ const OLLAMA_EXTRACTOR =
494
+ '{"output_json": "{{ msg.message.content | json_escape }}", "explanation": "{{ msg.message.thinking | json_escape }}", "input_tokens": {{ msg.prompt_eval_count | default: 0 }}, "output_tokens": {{ msg.eval_count | default: 0 }}, "total_tokens": {{ msg.prompt_eval_count | default: 0 | plus: msg.eval_count }}}'
495
+
496
+ const llmTool = await ctx.ensure(
497
+ 'LLM tool',
498
+ async () => {
499
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
500
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Marketing LLM Tool')
501
+ },
502
+ async () => {
503
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
504
+ name: 'Cookbook Marketing LLM Tool',
505
+ description: 'Ollama-backed chat-completion adapter for Score Lead classification.',
506
+ intent: 'llm_enrichment',
507
+ status: 'active',
508
+ datalake_id: ctx.datalakeId,
509
+ response_extractor: {
510
+ type: 'custom',
511
+ body: OLLAMA_EXTRACTOR,
512
+ output_schema: ENRICHMENT_OUTPUT_SCHEMA,
513
+ },
514
+ body: {
515
+ tool_body_type: 'rest_api',
516
+ base_url: 'http://localhost:11434',
517
+ base_path: { type: 'custom', body: '/api/chat' },
518
+ auth_method: 'api_key',
519
+ api_key: 'stub-key',
520
+ api_key_name: 'Authorization',
521
+ api_key_location: 'header',
522
+ request_type: 'json',
523
+ response_type: 'json',
524
+ timeout_ms: 60_000,
525
+ base_body: { type: 'custom', body: OLLAMA_BASE_BODY },
526
+ },
527
+ })
528
+ return data
529
+ },
530
+ )
531
+ ctx.llmToolId = llmTool.id!
532
+ ```
533
+
534
+ ## 010 — the Lead Submissions table
535
+
536
+ Lead-capture rows do not fit any canonical entity, so they are a generic
537
+ table you declare. The columns mirror what a form export or a spreadsheet
538
+ tab actually produces. `message` and `lead_source` are what the agent
539
+ reads; `band` is what a production system writes back.
540
+
541
+ `name` and `email` are declared `tokenize`. **That declaration is the
542
+ masking policy, and it is correct whether or not a tokenized lake exists
543
+ today** — this walk provisions none, so nothing masks here, and the
544
+ declaration goes live the moment someone calls `provisionTokenization`.
545
+ Declaring it now is how the policy survives that call rather than being
546
+ remembered afterwards.
547
+
548
+ ```typescript
549
+ const leads = await ctx.ensure(
550
+ 'lead-submissions table',
551
+ async () => {
552
+ const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
553
+ return (data.data ?? []).find((t: { title?: string }) => t.title === 'Cookbook Lead Submissions')
554
+ },
555
+ async () => {
556
+ const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
557
+ title: 'Cookbook Lead Submissions',
558
+ description: 'Inbound lead-capture form rows the Score Lead agent classifies.',
559
+ columns: [
560
+ { name: 'submission_id', title: 'Submission ID', type: 'string', description: 'Vendor-supplied unique submission id', is_unique: true, privacy_requirement: 'none' },
561
+ { name: 'name', title: 'Name', type: 'string', description: 'Submitter full name', is_unique: false, privacy_requirement: 'tokenize' },
562
+ { name: 'email', title: 'Email', type: 'string', description: 'Submitter email', is_unique: false, privacy_requirement: 'tokenize' },
563
+ { name: 'message', title: 'Message', type: 'string', description: 'Free-text lead message — the agent classifies this', is_unique: false, privacy_requirement: 'none' },
564
+ { name: 'lead_source', title: 'Lead Source', type: 'string', description: 'utm_source / sheet tab id', is_unique: false, privacy_requirement: 'none' },
565
+ { name: 'band', title: 'Band', type: 'string', description: 'Lead-scoring band assigned by the Score Lead agent', is_unique: false, privacy_requirement: 'none' },
566
+ ],
567
+ })
568
+ return data
569
+ },
570
+ )
571
+ genericTableId = leads.id!
572
+ ctx.leadsTableName = leads.name!
573
+
574
+ // Called unconditionally rather than only on the create branch, so a run
575
+ // that crashed between recording the table and deploying it still converges.
576
+ await ctx.deployGenericTable(ctx.leadsTableName)
577
+ ```
578
+
579
+ ## 011 — the Score Lead agent
580
+
581
+ The agent binds three things: the model name sent on every inference call,
582
+ an input schema the workflow's context mapping must satisfy, and a response
583
+ schema the output must match.
584
+
585
+ The response schema's `enum: ['hot','warm','cold','spam']` is the guard that
586
+ matters. It stops the model emitting a band the workflow has no action for
587
+ — without it a creative answer becomes a decision key nothing matches, and
588
+ the row silently fans out to nothing. `temperature: 0.0` removes sampling
589
+ noise so identical inputs classify identically, which is what makes the
590
+ assertion in §016 stable.
591
+
592
+ `data_access: 'raw'` because this lake has one tier. On a surface that
593
+ provisions the derived pair you would set `'tokenized'` here and the agent
594
+ would never see the name at all.
595
+
596
+ ```typescript
597
+ const SCORE_INPUT_SCHEMA = {
598
+ type: 'object',
599
+ properties: {
600
+ name: { type: 'string' },
601
+ message: { type: 'string' },
602
+ lead_source: { type: 'string' },
603
+ },
604
+ required: ['name', 'message'],
605
+ }
606
+
607
+ const SCORE_RESPONSE_SCHEMA = {
608
+ type: 'object',
609
+ properties: { band: { type: 'string', enum: ['hot', 'warm', 'cold', 'spam'] } },
610
+ required: ['band'],
611
+ }
612
+
613
+ const AGENT_PROMPT_BODY = `You are a B2B lead-scoring assistant. Classify the following lead's intent into EXACTLY ONE of these bands:
614
+
615
+ - "hot" — clear buying intent, decision maker, ready to engage
616
+ - "warm" — interested but exploratory, requires nurture
617
+ - "cold" — generic/lukewarm interest, low conversion signal
618
+ - "spam" — promotional, irrelevant, or low-quality submission
619
+
620
+ Lead name: {{ name }}
621
+ Lead message: {{ message }}
622
+ Lead source: {{ lead_source }}
623
+
624
+ Respond with a JSON object: {"band": "<one of hot|warm|cold|spam>"}`
625
+
626
+ const agent = await ctx.ensure(
627
+ 'Score Lead agent',
628
+ async () => {
629
+ const { data } = await api.aiAgents.list(tenantSlug, datalakeSlug)
630
+ return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Score Lead Agent')
631
+ },
632
+ async () => {
633
+ const { data } = await api.aiAgents.create(tenantSlug, datalakeSlug, {
634
+ name: 'Cookbook Score Lead Agent',
635
+ tool_id: ctx.llmToolId,
636
+ model: 'qwen3-vl:8b-instruct',
637
+ data_access: 'raw',
638
+ temperature: 0.0,
639
+ max_tokens: 1024,
640
+ enabled: true,
641
+ input_schema: SCORE_INPUT_SCHEMA,
642
+ llm_response_schema: SCORE_RESPONSE_SCHEMA,
643
+ prompt_config: { type: 'custom', body: AGENT_PROMPT_BODY },
644
+ })
645
+ return data
646
+ },
647
+ )
648
+ aiAgentId = agent.id!
649
+ ctx.agentSlug = agent.slug!
650
+ ```
651
+
652
+ ## 012 — the Score Lead workflow
653
+
654
+ Standard shape — filter, decision, actions — with two things worth reading
655
+ closely.
656
+
657
+ The agent is **nested in the create body** (`workflow_ai_agents`, a
658
+ `cast_assoc` rather than a separate attach call), which binds it into the
659
+ workflow's enrichment phase along with a Liquid `context_mapping_config`
660
+ that projects each row into the agent's input schema. A workflow join omits
661
+ `output_schema`; the server pins it from the agent's own `input_schema`.
662
+
663
+ The `decision_config` body interpolates the agent's `band` into a
664
+ single-element decision array, and the four actions are keyed
665
+ `decision_key: hot|warm|cold|spam`. Whichever band comes back picks one
666
+ action and leaves the other three `:skipped`.
667
+
668
+ **Bracket access is required**, not stylistic: the agent's slug contains
669
+ hyphens, and Liquid's dot parser would read `additional_context.score-lead`
670
+ as a subtraction.
671
+
672
+ ```typescript
673
+ const BANDS = ['hot', 'warm', 'cold', 'spam'] as const
674
+
675
+ const CONTEXT_MAPPING_BODY = JSON.stringify({
676
+ name: '{{ event_dataset.name }}',
677
+ message: '{{ event_dataset.message }}',
678
+ lead_source: '{{ event_dataset.lead_source }}',
679
+ })
680
+
681
+ const DECISION_CONFIG_BODY = `["{{ additional_context["${ctx.agentSlug}"].band }}"]`
682
+
683
+ const scoreWorkflow = await ctx.ensure(
684
+ 'Score Lead workflow',
685
+ async () => {
686
+ const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
687
+ return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Score Lead Workflow')
688
+ },
689
+ async () => {
690
+ const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
691
+ name: 'Cookbook Score Lead Workflow',
692
+ description: 'Classifies lead submissions into hot/warm/cold/spam via an LLM agent; one SMS action per band.',
693
+ dataset_type: 'generic_table',
694
+ generic_table_id: genericTableId,
695
+ skip_mdm_resolution: true,
696
+ status: 'live',
697
+ tags: ['leads', 'llm'],
698
+ filter_config: { type: 'custom', body: 'true' },
699
+ decision_config: {
700
+ type: 'custom',
701
+ body: DECISION_CONFIG_BODY,
702
+ output_schema: { type: 'array', items: { type: 'string' } },
703
+ },
704
+ actions: BANDS.map((band) => ({
705
+ decision_key: band,
706
+ action_type: 'sms',
707
+ tool_id: ctx.smsToolId,
708
+ position: 0,
709
+ trigger_template: 'now',
710
+ idempotency_template: `{{ event_dataset.submission_id }}-${band}`,
711
+ tool_call: {
712
+ tool_call_type: 'sms_request',
713
+ to: { type: 'custom', body: '+15551234567' },
714
+ body: { type: 'custom', body: `Lead band [${band}]: {{ event_dataset.message }}` },
715
+ sms_type: 'transactional',
716
+ },
717
+ })),
718
+ workflow_ai_agents: [
719
+ {
720
+ ai_agent_id: aiAgentId,
721
+ position: 0,
722
+ context_mapping_config: { type: 'custom', body: CONTEXT_MAPPING_BODY },
723
+ },
724
+ ],
725
+ })
726
+ return data
727
+ },
728
+ )
729
+ workflowId = scoreWorkflow.id!
730
+ ctx.scoreWorkflowSlug = scoreWorkflow.slug!
731
+ ```
732
+
733
+ ## 013 — the table's own Data Activation Client, which you did not create
734
+
735
+ Creating a generic table provisions two things for you: an identity
736
+ interoperability contract, and a **default Data Activation Client** bound to
737
+ it, named `<datalake> <table> DataActivationClient` with
738
+ `tool_call: manual_upload`. For plain row ingestion into a table you
739
+ declared, there is nothing to build — no data source, no tool, no contract,
740
+ no client.
741
+
742
+ **Scope the lookup server-side.** The listing pages at twenty and is not
743
+ newest-first, so a `.find()` over page one starts missing the client you
744
+ want as soon as the lake has a few tables — and the failure looks like the
745
+ client was never provisioned.
746
+
747
+ ```typescript
748
+ const dacDeadline = Date.now() + 120_000
749
+ let defaultDac: { slug?: string | null } | undefined
750
+ while (Date.now() < dacDeadline && !defaultDac) {
751
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
752
+ filters: [{ field: 'name', op: 'ilike', value: ctx.leadsTableName }],
753
+ })
754
+ defaultDac = (data.data ?? [])[0]
755
+ if (!defaultDac) await new Promise((r) => setTimeout(r, 1_000))
756
+ }
757
+ if (!defaultDac?.slug) {
758
+ throw new Error(`no default DAC found for table ${ctx.leadsTableName} within 120s`)
759
+ }
760
+ ctx.leadsDacSlug = defaultDac.slug
761
+ ```
762
+
763
+ ## 014 — ingest three leads, one obviously each way
764
+
765
+ Three rows through the default client. Each carries a deliberately
766
+ unambiguous `message` — approved budget and a deadline, an idle browse, and
767
+ obvious promotional spam — so the classification is not a coin toss and the
768
+ assertion in §016 means something. `band` is left unset on ingest; assigning
769
+ it is the agent's job.
770
+
771
+ The submission ids are **stable across runs**, and `submission_id` is the
772
+ table's unique column, so a second run upserts these same three rows rather
773
+ than adding three more.
774
+
775
+ ```typescript
776
+ ctx.hotLead = 'LEAD-COOKBOOK-0101'
777
+ ctx.coldLead = 'LEAD-COOKBOOK-0102'
778
+ ctx.spamLead = 'LEAD-COOKBOOK-0103'
779
+
780
+ const leadRows = [
781
+ {
782
+ submission_id: ctx.hotLead,
783
+ name: 'Dana Decisive',
784
+ email: 'dana@example.com',
785
+ message:
786
+ 'We have budget approved and need to roll out to 500 seats this quarter — can we start onboarding next week?',
787
+ lead_source: 'cookbook_score_leads',
788
+ },
789
+ {
790
+ submission_id: ctx.coldLead,
791
+ name: 'Sam Browsing',
792
+ email: 'sam@example.com',
793
+ message: 'Just browsing — found your site via a blog post. No particular need right now.',
794
+ lead_source: 'cookbook_score_leads',
795
+ },
796
+ {
797
+ submission_id: ctx.spamLead,
798
+ name: 'Promo Bot',
799
+ email: 'promo@example.com',
800
+ message: 'BUY CHEAP FOLLOWERS NOW!!! 90% off SEO backlinks, crypto giveaways, click here click here!!!',
801
+ lead_source: 'cookbook_score_leads',
802
+ },
803
+ ]
804
+
805
+ const ingests: string[] = []
806
+ for (const row of leadRows) {
807
+ const resp = await api.dataActivationClients.ingest(
808
+ tenantSlug, datalakeSlug, ctx.leadsDacSlug, { data: row },
809
+ )
810
+ ingests.push(resp.data.batch_id!)
811
+ }
812
+ ctx.leadBatchIds = ingests
813
+ ```
814
+
815
+ ## 015 — wait for all three to persist
816
+
817
+ ```typescript
818
+ await ctx.waitForBatches(ctx.leadsDacSlug, ctx.leadBatchIds)
819
+ ```
820
+
821
+ ## 016 — run the workflow across the three leads
822
+
823
+ `workflows.run` triggers the whole pipeline per row: filter, then agent
824
+ enrichment — a live inference call per row — then the decision, then the
825
+ action fan-out. The where clause scopes the run to exactly the three
826
+ submissions §014 ingested, so rows left by any other walk are untouched.
827
+
828
+ The window is generous because three real model calls happen inside it.
829
+
830
+ ```typescript
831
+ const submissionList = [ctx.hotLead, ctx.coldLead, ctx.spamLead].map((s) => `'${s}'`).join(', ')
832
+
833
+ const runResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.scoreWorkflowSlug, {
834
+ sql_where_clause: `submission_id IN (${submissionList})`,
835
+ mode: 'live',
836
+ manual_override: true,
837
+ })
838
+ const fired = await ctx.waitForFiredRun(datalakeSlug, runResp.data.workflow_run_id)
839
+ ctx.scoreRunLogId = fired.workflowRunLogId
840
+ ctx.scoreRunBatchId = fired.batchId!
841
+
842
+ const runDeadline = Date.now() + 240_000
843
+ let runStatus: string | null = null
844
+ while (Date.now() < runDeadline) {
845
+ const { data: log } = await api.workflows.batchLogs.refresh(
846
+ tenantSlug, datalakeSlug, ctx.scoreWorkflowSlug, ctx.scoreRunLogId,
847
+ )
848
+ runStatus = log.status ?? null
849
+ if (runStatus && runStatus !== 'pending') break
850
+ await new Promise((r) => setTimeout(r, 2_000))
851
+ }
852
+ if (runStatus === 'failed') throw new Error('Score Lead run reached :failed')
853
+ if (!runStatus || runStatus === 'pending') {
854
+ throw new Error('Score Lead run did not leave :pending within 240s')
855
+ }
856
+ ```
857
+
858
+ ## 017 — verify the agent's band actually steered the fan-out
859
+
860
+ Each lead produced a Workflow Execution Log carrying **four** action
861
+ execution logs, one per band.
862
+
863
+ On the first run for a given submission, exactly **one** is matched
864
+ (`:pending` or `:completed` — the band the agent chose) and **three** are
865
+ `:skipped`. That one-of-four is the proof the agent's output drove the
866
+ decision rather than every action firing.
867
+
868
+ **On a later run all four are `:skipped`, and that is correct.** The
869
+ idempotency key is `{{ submission_id }}-<band>`, the submission ids are
870
+ stable, so the key that already fired refuses to fire again — the platform
871
+ will not send the same SMS about the same lead twice. Both shapes are
872
+ accepted here, and anything else is a real failure: two matched actions
873
+ means the decision matched more than one band, and zero matched with zero
874
+ skipped means the fan-out never happened.
875
+
876
+ ```typescript
877
+ const { data: wfLogs } = await api.workflows.workflowLogs.list(
878
+ tenantSlug, datalakeSlug, ctx.scoreWorkflowSlug,
879
+ )
880
+ const ourWels = (wfLogs.data ?? []).filter(
881
+ (w) => (w as { batch_id?: string }).batch_id === ctx.scoreRunBatchId,
882
+ )
883
+ if (ourWels.length !== 3) {
884
+ throw new Error(`expected 3 WELs for the run, got ${ourWels.length}`)
885
+ }
886
+
887
+ for (const wel of ourWels) {
888
+ const welId = (wel as { id?: string }).id
889
+ const aels = (wel as { action_execution_logs?: Array<{ status?: string }> }).action_execution_logs ?? []
890
+ if (aels.length !== 4) {
891
+ throw new Error(`WEL ${welId}: expected 4 AELs (one per band), got ${aels.length}`)
892
+ }
893
+ const byStatus: Record<string, number> = {}
894
+ for (const ael of aels) {
895
+ const st = ael.status ?? 'unknown'
896
+ byStatus[st] = (byStatus[st] ?? 0) + 1
897
+ }
898
+ const matched = (byStatus.pending ?? 0) + (byStatus.completed ?? 0)
899
+ const skipped = byStatus.skipped ?? 0
900
+
901
+ if (matched === 1 && skipped === 3) continue // first run for this lead
902
+ if (matched === 0 && skipped === 4) continue // already fired, correctly refusing
903
+ throw new Error(
904
+ `WEL ${welId}: expected 1 matched + 3 skipped, or 4 skipped on a repeat run — got ${JSON.stringify(byStatus)}`,
905
+ )
906
+ }
907
+ ```
908
+
909
+ ## 018 — confirm a band-prefixed SMS actually rendered
910
+
911
+ §017 proves the fan-out chose one action. This proves something was
912
+ actually *sent*, and it is the assertion that survives every rerun.
913
+
914
+ Search on the **idempotency key**, never on `workflow_id`. The key is
915
+ `<submission id>-<band>` and the submission id is stable, so it names the
916
+ same SMS forever. A workflow id is stable only as long as the workflow
917
+ resource is: recreate it against a lake that still holds its rows and the
918
+ id changes while the key does not, so the action correctly refuses to fire
919
+ and a search for the new id finds nothing at all.
920
+
921
+ The hot lead is the one asserted, because it is the only one whose band is
922
+ genuinely unambiguous — approved budget, a seat count and a deadline is
923
+ `hot` to any classifier worth running.
924
+
925
+ ```typescript
926
+ const { data: leadSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
927
+ search_query: `m.idempotency_key LIKE '${ctx.hotLead}-%'`,
928
+ })
929
+ if (leadSearch.status !== 'completed') {
930
+ throw new Error(`message user-search status=${leadSearch.status} error=${leadSearch.error_message ?? '(none)'}`)
931
+ }
932
+
933
+ const leadMsgDeadline = Date.now() + 45_000
934
+ let leadMessages: Array<Record<string, unknown>> = []
935
+ while (Date.now() < leadMsgDeadline && leadMessages.length === 0) {
936
+ const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
937
+ userSearchId: leadSearch.id!,
938
+ dataAccessMode: 'raw',
939
+ })
940
+ leadMessages = (data.data ?? []) as Array<Record<string, unknown>>
941
+ if (leadMessages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
942
+ }
943
+ if (leadMessages.length === 0) {
944
+ throw new Error(
945
+ `no band SMS for lead ${ctx.hotLead} within 45s — ` +
946
+ `the action never fired, or it fired under a different idempotency key`,
947
+ )
948
+ }
949
+
950
+ const bandBody = leadMessages.map((m) => String(m.body ?? '')).find((b) => b.includes('Lead band ['))
951
+ if (!bandBody) {
952
+ throw new Error(`no band-prefixed SMS body — bodies: ${JSON.stringify(leadMessages.map((m) => m.body))}`)
953
+ }
954
+ if (!bandBody.includes('[hot]')) {
955
+ throw new Error(`expected the classifier to band this lead [hot] — got: ${bandBody}`)
956
+ }
957
+ ```
958
+
959
+ ## 019 — the Birthday Greeting connected app
960
+
961
+ The greeting carries a deep link so the recipient can do something with it
962
+ — upload a photo, RSVP, whatever the operator wires up. A connected app is
963
+ a thin registration of that page's URL and mode; the page itself is hosted
964
+ outside the platform (`mode: 'self_hosted'`). Capture the server-derived
965
+ `slug` — §030 resolves a token against it.
966
+
967
+ ```typescript
968
+ const birthdayApp = await ctx.ensure(
969
+ 'Birthday Greeting connected app',
970
+ async () => {
971
+ const { data } = await api.connectedApps.list(tenantSlug, datalakeSlug)
972
+ return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Birthday Greeting Form')
973
+ },
974
+ async () => {
975
+ const { data } = await api.connectedApps.create(tenantSlug, datalakeSlug, {
976
+ name: 'Cookbook Birthday Greeting Form',
977
+ description: 'Birthday-greeting form linked from the outbound SMS.',
978
+ mode: 'self_hosted',
979
+ urls: [{ url: 'https://birthday.example.local', is_primary: true, label: 'production' }],
980
+ })
981
+ return data
982
+ },
983
+ )
984
+ connectedAppId = birthdayApp.id!
985
+ ctx.birthdayAppSlug = birthdayApp.slug!
986
+ ```
987
+
988
+ ## 020 — the Happy Birthday workflow, and the year-roll trigger
989
+
990
+ No agent here, and that is the point of putting it beside §012: the same
991
+ workflow shape does this job with nothing but Liquid.
992
+
993
+ **The trigger template is the whole scenario.** It compares today's `MM-DD`
994
+ against the contact's date of birth `MM-DD` and renders either this year's
995
+ birthday or next year's. String comparison works because both sides are
996
+ zero-padded `MM-DD` — the one detail that makes a one-line year-roll
997
+ correct instead of nearly correct.
998
+
999
+ Two things that look optional and are not. The filter reads
1000
+ `mdm_output.legal_entity.date_of_birth`, which requires
1001
+ `skip_mdm_resolution: false` — the resolution is what puts the subject in
1002
+ scope for the filter at all. And the SMS body must reference
1003
+ `{{ connected_app_form_url }}`: the platform mints the page token at action
1004
+ time and injects that variable, so a body that never mentions it gets a
1005
+ token minted and thrown away, and §031 has no link to resolve.
1006
+
1007
+ ```typescript
1008
+ const DECISION_KEY = 'send_happy_birthday_sms'
1009
+ const TRIGGER_TEMPLATE =
1010
+ '{% assign year = "" | now | date: "%Y" %}' +
1011
+ '{% assign today_mmdd = "" | now | date: "%m-%d" %}' +
1012
+ '{% assign dob_mmdd = mdm_output.legal_entity.date_of_birth | date: "%m-%d" %}' +
1013
+ '{% if dob_mmdd >= today_mmdd %}{{ year }}-{{ dob_mmdd }} 00:00:00' +
1014
+ '{% else %}{{ year | plus: 1 }}-{{ dob_mmdd }} 00:00:00{% endif %}'
1015
+
1016
+ const birthdayWorkflow = await ctx.ensure(
1017
+ 'Happy Birthday workflow',
1018
+ async () => {
1019
+ const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
1020
+ return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Happy Birthday Workflow')
1021
+ },
1022
+ async () => {
1023
+ const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
1024
+ name: 'Cookbook Happy Birthday Workflow',
1025
+ description: "Sends a birthday SMS on each contact's next birthday — pure-Liquid trigger does the year-roll math.",
1026
+ dataset_type: 'legal_entity',
1027
+ status: 'live',
1028
+ tags: ['lifecycle', 'birthday'],
1029
+ skip_mdm_resolution: false,
1030
+ filter_config: { type: 'custom', body: '{% if mdm_output.legal_entity.date_of_birth %}true{% endif %}' },
1031
+ decision_config: {
1032
+ type: 'custom',
1033
+ body: `["${DECISION_KEY}"]`,
1034
+ output_schema: { type: 'array', items: { type: 'string' } },
1035
+ },
1036
+ actions: [
1037
+ {
1038
+ action_type: 'sms',
1039
+ tool_id: ctx.smsToolId,
1040
+ decision_key: DECISION_KEY,
1041
+ position: 0,
1042
+ trigger_template: TRIGGER_TEMPLATE,
1043
+ idempotency_template: '{{ legal_entity.id }}-birthday-greeting',
1044
+ connected_app_id: connectedAppId,
1045
+ connected_app_route: '/forms/birthday-greeting',
1046
+ connected_app_metadata_template: '{"legal_entity_id":"{{ legal_entity.id }}"}',
1047
+ tool_call: {
1048
+ tool_call_type: 'sms_request',
1049
+ to: { type: 'custom', body: '{{ mdm_output.legal_entity.phone_numbers | first | map: "phone_number" }}' },
1050
+ body: {
1051
+ type: 'custom',
1052
+ body: 'Happy birthday {{ mdm_output.legal_entity.first_name }}! \u{1F382} Share a photo: {{ connected_app_form_url }}',
1053
+ },
1054
+ sms_type: 'transactional',
1055
+ },
1056
+ },
1057
+ ],
1058
+ })
1059
+ return data
1060
+ },
1061
+ )
1062
+ ctx.birthdayWorkflowSlug = birthdayWorkflow.slug!
1063
+ ```
1064
+
1065
+ ## 021 — the data source, the upload tool, and the contract
1066
+
1067
+ Three small pieces, one step, because none of them is interesting alone.
1068
+
1069
+ The **data source** registers where rows come from. Its `uri` matters more
1070
+ than it looks: the Data Activation Client injects it into every row as
1071
+ `source_uri`, and the contract template writes it as the identification's
1072
+ `uri` — so it is half of the key the legal-entity dedupe uses. Change it
1073
+ and yesterday's contacts stop matching today's.
1074
+
1075
+ The **manual-upload tool** is the minimal tool for inline JSON; it needs no
1076
+ endpoint or credential because the rows arrive in the ingest call rather
1077
+ than being fetched.
1078
+
1079
+ The **contract** shapes a row into a legal-entity upsert. Its
1080
+ `mdm_input_config` is `{ type: 'null' }` — the row *is* the subject, so
1081
+ there is nothing to resolve. The template comes from the vendored fixture
1082
+ rather than several hundred inlined lines of Liquid.
1083
+
1084
+ ```typescript
1085
+ const { readFileSync } = await import('node:fs')
1086
+ const { join } = await import('node:path')
1087
+ const leTemplate = readFileSync(
1088
+ join(process.env.COOKBOOK_FIXTURES_DIR!, 'organic-marketing', '_lead_submissions_foundation_legal_entity.liquid'),
1089
+ 'utf8',
1090
+ )
1091
+
1092
+ ctx.leadSourceUri = 'https://app.alvera.ai/cookbook-lead-form'
1093
+
1094
+ const leadSource = await ctx.ensure(
1095
+ 'lead-form data source',
1096
+ async () => {
1097
+ const { data } = await api.dataSources.list(tenantSlug, datalakeSlug)
1098
+ return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Lead Form Source')
1099
+ },
1100
+ async () => {
1101
+ const { data } = await api.dataSources.create(tenantSlug, datalakeSlug, {
1102
+ name: 'Cookbook Lead Form Source',
1103
+ uri: ctx.leadSourceUri,
1104
+ description: 'Inbound lead-capture form — origin of the contact rows the birthday workflow runs on.',
1105
+ status: 'active',
1106
+ is_default: false,
1107
+ })
1108
+ return data
1109
+ },
1110
+ )
1111
+ dataSourceId = leadSource.id!
1112
+
1113
+ const uploadTool = await ctx.ensure(
1114
+ 'manual-upload tool',
1115
+ async () => {
1116
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
1117
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Manual Upload Tool')
1118
+ },
1119
+ async () => {
1120
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
1121
+ name: 'Cookbook Manual Upload Tool',
1122
+ description: 'Manual-upload data-exchange tool — backs the client that ingests contact rows.',
1123
+ intent: 'data_exchange',
1124
+ status: 'active',
1125
+ datalake_id: ctx.datalakeId,
1126
+ data_source_id: dataSourceId,
1127
+ body: { tool_body_type: 'manual_upload' },
1128
+ })
1129
+ return data
1130
+ },
1131
+ )
1132
+ ctx.manualUploadToolId = uploadTool.id!
1133
+
1134
+ const leContract = await ctx.ensure(
1135
+ 'lead-form legal-entity contract',
1136
+ async () => {
1137
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1138
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Lead Form to LegalEntity')
1139
+ },
1140
+ async () => {
1141
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1142
+ name: 'Cookbook Lead Form to LegalEntity',
1143
+ description: 'Lead-form row → LegalEntity (custom Liquid, no MDM input).',
1144
+ resource_type: 'legal_entity',
1145
+ template_config: { type: 'custom', body: leTemplate },
1146
+ mdm_input_config: { type: 'null' },
1147
+ generic_table_id: null,
1148
+ })
1149
+ return data
1150
+ },
1151
+ )
1152
+ interopContractId = leContract.id!
1153
+ ```
1154
+
1155
+ ## 022 — the lead-form activation client
1156
+
1157
+ One contract on one client, so nothing to sequence here — the trap that
1158
+ forces two clients apart only appears when a pair resolves the same
1159
+ subject at once.
1160
+
1161
+ ```typescript
1162
+ const leadDac = await ctx.ensure(
1163
+ 'lead-form activation client',
1164
+ async () => {
1165
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
1166
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Lead Form DAC' }],
1167
+ })
1168
+ return (data.data ?? [])[0]
1169
+ },
1170
+ async () => {
1171
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1172
+ name: 'Cookbook Lead Form DAC',
1173
+ description: 'Manual-upload client — ingests lead-form rows as legal entities.',
1174
+ tool_id: ctx.manualUploadToolId,
1175
+ data_source_id: dataSourceId,
1176
+ tool_call: { tool_call_type: 'manual_upload' },
1177
+ interoperability_contract_ids: [interopContractId],
1178
+ })
1179
+ return data
1180
+ },
1181
+ )
1182
+ dacId = leadDac.id!
1183
+ ctx.leadFormDacSlug = leadDac.slug!
1184
+ ```
1185
+
1186
+ ## 023 — ingest two contacts, one with a birthday and one without
1187
+
1188
+ Two rows. One carries a `date_of_birth` so the filter passes it; the other
1189
+ omits it entirely so the filter must reject it. A run where both passed, or
1190
+ both were filtered, would mean the filter is not evaluating anything — and
1191
+ that is exactly the failure a walk that never runs its workflow cannot see.
1192
+
1193
+ **The date of birth is tomorrow's month-day, stamped thirty years back.**
1194
+ Both halves are deliberate. The past year satisfies the changeset's
1195
+ not-in-the-future guard on a date of birth. Tomorrow's month-day makes the
1196
+ greeting schedule for the *future*, which is what makes §029's
1197
+ fast-forward observable at all — a birthday today renders a time already
1198
+ passed and would dispatch on its own, proving nothing about the override.
1199
+
1200
+ The emails are stable, and the email is what the identification dedupes on,
1201
+ so a second run updates these same two contacts rather than adding two more.
1202
+
1203
+ ```typescript
1204
+ const today = new Date()
1205
+ const tomorrow = new Date(today.getTime() + 86_400_000)
1206
+ const tomorrowMM = String(tomorrow.getUTCMonth() + 1).padStart(2, '0')
1207
+ const tomorrowDD = String(tomorrow.getUTCDate()).padStart(2, '0')
1208
+ const dobYear = today.getUTCFullYear() - 30
1209
+
1210
+ ctx.birthdayEmail = 'maria-birthday@cookbook.example.com'
1211
+ ctx.birthdayPhone = '+15550100101'
1212
+
1213
+ const birthdayRow = {
1214
+ submission_id: 'BD-COOKBOOK-0101',
1215
+ name: 'Maria Birthday',
1216
+ email: ctx.birthdayEmail,
1217
+ phone: ctx.birthdayPhone,
1218
+ company: '',
1219
+ message: 'My birthday is tomorrow',
1220
+ lead_source: 'cookbook_birthday',
1221
+ source_uri: ctx.leadSourceUri,
1222
+ date_of_birth: `${dobYear}-${tomorrowMM}-${tomorrowDD}`,
1223
+ }
1224
+ const noDobRow = {
1225
+ submission_id: 'BD-COOKBOOK-0102',
1226
+ name: 'Priya NoDob',
1227
+ email: 'priya-nodob@cookbook.example.com',
1228
+ phone: '+15550100102',
1229
+ company: '',
1230
+ message: 'I have no date of birth on file',
1231
+ lead_source: 'cookbook_birthday',
1232
+ source_uri: ctx.leadSourceUri,
1233
+ // date_of_birth deliberately absent — the filter must reject this row
1234
+ }
1235
+
1236
+ const bdIngest = await api.dataActivationClients.ingest(
1237
+ tenantSlug, datalakeSlug, ctx.leadFormDacSlug, { data: birthdayRow },
1238
+ )
1239
+ const noDobIngest = await api.dataActivationClients.ingest(
1240
+ tenantSlug, datalakeSlug, ctx.leadFormDacSlug, { data: noDobRow },
1241
+ )
1242
+ ctx.batchBirthday = bdIngest.data.batch_id!
1243
+ ctx.batchNoDob = noDobIngest.data.batch_id!
1244
+ ```
1245
+
1246
+ ## 024 — wait for both contacts to persist
1247
+
1248
+ ```typescript
1249
+ await ctx.waitForBatches(ctx.leadFormDacSlug, [ctx.batchBirthday, ctx.batchNoDob])
1250
+ ```
1251
+
1252
+ ## 025 — run the workflow, and wait for the right thing
1253
+
1254
+ `manual_override: false` so the filter genuinely evaluates rather than
1255
+ being bypassed.
1256
+
1257
+ **This run never reaches a terminal status, and waiting for one would hang
1258
+ until the deadline.** The birthday row's action is scheduled for a date in
1259
+ the future, so its execution log sits at `:executing` for the life of the
1260
+ run — correctly. So the wait is on `total_wels`: every row has produced an
1261
+ execution log, which is the moment the next step can read them.
1262
+
1263
+ ```typescript
1264
+ const bdRunResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, {
1265
+ sql_where_clause: `le.batch_id IN ('${ctx.batchBirthday}', '${ctx.batchNoDob}')`,
1266
+ mode: 'live',
1267
+ manual_override: false,
1268
+ })
1269
+ const bdFired = await ctx.waitForFiredRun(datalakeSlug, bdRunResp.data.workflow_run_id)
1270
+ ctx.bdRunLogId = bdFired.workflowRunLogId
1271
+ ctx.bdRunBatchId = bdFired.batchId!
1272
+
1273
+ const welDeadline = Date.now() + 120_000
1274
+ let totalWels = 0
1275
+ while (Date.now() < welDeadline) {
1276
+ const { data: log } = await api.workflows.batchLogs.refresh(
1277
+ tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, ctx.bdRunLogId,
1278
+ )
1279
+ if (log.status === 'failed') throw new Error('birthday workflow run reached :failed')
1280
+ totalWels = typeof log.total_wels === 'number' ? log.total_wels : 0
1281
+ if (totalWels >= 2) break
1282
+ await new Promise((r) => setTimeout(r, 1_500))
1283
+ }
1284
+ if (totalWels < 2) throw new Error(`birthday run produced only ${totalWels}/2 execution logs within 120s`)
1285
+ ```
1286
+
1287
+ ## 026 — verify the filter routed
1288
+
1289
+ The contact with a date of birth passes and its log is `:executing` (a
1290
+ future action is scheduled) or `:completed`. The contact without one is
1291
+ `:filtered` — it never reached the action at all.
1292
+
1293
+ ```typescript
1294
+ const { data: bdLogs } = await api.workflows.workflowLogs.list(
1295
+ tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug,
1296
+ )
1297
+ const bdWels = (bdLogs.data ?? []).filter(
1298
+ (w) => (w as { batch_id?: string }).batch_id === ctx.bdRunBatchId,
1299
+ )
1300
+ if (bdWels.length !== 2) {
1301
+ throw new Error(`expected 2 execution logs for the birthday run, got ${bdWels.length}`)
1302
+ }
1303
+
1304
+ const bdByStatus: Record<string, number> = {}
1305
+ for (const w of bdWels) {
1306
+ const st = (w as { status?: string }).status ?? 'unknown'
1307
+ bdByStatus[st] = (bdByStatus[st] ?? 0) + 1
1308
+ }
1309
+ const bdPassed = (bdByStatus.executing ?? 0) + (bdByStatus.completed ?? 0)
1310
+ if (bdPassed !== 1) {
1311
+ throw new Error(`expected 1 pass-branch log, got ${bdPassed} — ${JSON.stringify(bdByStatus)}`)
1312
+ }
1313
+ if ((bdByStatus.filtered ?? 0) !== 1) {
1314
+ throw new Error(`expected 1 :filtered log (the contact with no date of birth) — ${JSON.stringify(bdByStatus)}`)
1315
+ }
1316
+ ```
1317
+
1318
+ ## 027 — find the contact the greeting is for
1319
+
1320
+ `workflows.execute` is addressed by dataset id, so the birthday contact's
1321
+ legal-entity id has to be resolved first. Search scoped to its own batch.
1322
+
1323
+ ```typescript
1324
+ const { data: bdSearchDef } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'legal_entity', {
1325
+ search_query: `le.batch_id = '${ctx.batchBirthday}'`,
1326
+ })
1327
+ if (bdSearchDef.status !== 'completed') {
1328
+ throw new Error(`legal_entity user-search status=${bdSearchDef.status} error=${bdSearchDef.error_message ?? '(none)'}`)
1329
+ }
1330
+ const { data: bdSearch } = await api.datasets.search(tenantSlug, datalakeSlug, 'legal_entity', {
1331
+ userSearchId: bdSearchDef.id!,
1332
+ dataAccessMode: 'raw',
1333
+ })
1334
+ const birthdayLegalEntityId = (bdSearch.data?.[0] as { id?: string } | undefined)?.id
1335
+ if (!birthdayLegalEntityId) throw new Error('the birthday contact was not found by search')
1336
+ ctx.birthdayLegalEntityId = birthdayLegalEntityId
1337
+ ```
1338
+
1339
+ ## 028 — fast-forward the scheduled greeting
1340
+
1341
+ The greeting is scheduled for a date that has not arrived, so it will not
1342
+ dispatch during this sitting. `trigger_override: true` promotes the queued
1343
+ job so it runs now.
1344
+
1345
+ **What moves is the job, and only the job.** The execution log still
1346
+ records the year-roll-rendered `scheduled_at` it was always going to have,
1347
+ which is what keeps the override's blast radius to this one dispatch
1348
+ instead of rewriting the schedule.
1349
+
1350
+ On a repeat run the greeting has already been sent for this contact and the
1351
+ idempotency key refuses to send it again, so the forced log completes with
1352
+ nothing dispatched. Both are `:completed`; the assertion that the greeting
1353
+ *exists* belongs to §029, where it is true either way.
1354
+
1355
+ ```typescript
1356
+ const { data: execResp } = await api.workflows.execute(tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, {
1357
+ dataset_id: ctx.birthdayLegalEntityId,
1358
+ decision_key: 'send_happy_birthday_sms',
1359
+ manual_override: true,
1360
+ trigger_override: true,
1361
+ })
1362
+ ctx.forcedWelId = execResp.workflow_execution_log_id!
1363
+
1364
+ const forcedDeadline = Date.now() + 120_000
1365
+ let forcedStatus: string | null = null
1366
+ while (Date.now() < forcedDeadline) {
1367
+ const { data: wel } = await api.workflows.workflowLogs.get(
1368
+ tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, ctx.forcedWelId,
1369
+ )
1370
+ forcedStatus = (wel as { status?: string }).status ?? null
1371
+ if (forcedStatus && forcedStatus !== 'pending' && forcedStatus !== 'executing') break
1372
+ await new Promise((r) => setTimeout(r, 2_000))
1373
+ }
1374
+ if (forcedStatus !== 'completed') {
1375
+ throw new Error(`the fast-forwarded log did not reach :completed (last status: ${forcedStatus})`)
1376
+ }
1377
+ ```
1378
+
1379
+ ## 029 — the greeting exists, and it carries a deep link
1380
+
1381
+ The end state: a birthday greeting exists for this contact, with a
1382
+ `/t/<token>` link in its body. True on the first run because the action
1383
+ just dispatched, and true on every run after because it dispatched once and
1384
+ the idempotency key has protected it since.
1385
+
1386
+ **Read the durable record, not the run you just made.** The obvious move is
1387
+ to pull the rendered body off the execution log from §028 — and that works
1388
+ exactly once. On a repeat run that log's action is skipped and carries no
1389
+ body at all, so the walk goes red having proved nothing except that
1390
+ idempotency works. The `message` row persists in the lake and is the
1391
+ answer to "does this contact have a greeting", which is the question worth
1392
+ asking. In local dev the publish itself lands in LocalStack's SNS records
1393
+ at `/_aws/sns/sms-messages`, keyed by phone number — useful to eyeball, too
1394
+ volatile to assert on.
1395
+
1396
+ ```typescript
1397
+ const { data: bdMsgSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
1398
+ search_query: `m.idempotency_key LIKE '${ctx.birthdayLegalEntityId}-%'`,
1399
+ })
1400
+ if (bdMsgSearch.status !== 'completed') {
1401
+ throw new Error(`message user-search status=${bdMsgSearch.status} error=${bdMsgSearch.error_message ?? '(none)'}`)
1402
+ }
1403
+
1404
+ const bdMsgDeadline = Date.now() + 45_000
1405
+ let bdMessages: Array<Record<string, unknown>> = []
1406
+ while (Date.now() < bdMsgDeadline && bdMessages.length === 0) {
1407
+ const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
1408
+ userSearchId: bdMsgSearch.id!,
1409
+ dataAccessMode: 'raw',
1410
+ })
1411
+ bdMessages = (data.data ?? []) as Array<Record<string, unknown>>
1412
+ if (bdMessages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
1413
+ }
1414
+ if (bdMessages.length === 0) {
1415
+ throw new Error(
1416
+ `no birthday greeting for contact ${ctx.birthdayLegalEntityId} within 45s — ` +
1417
+ `the action never fired, or it fired under a different idempotency key`,
1418
+ )
1419
+ }
1420
+
1421
+ const greeting = bdMessages
1422
+ .map((m) => String(m.body ?? ''))
1423
+ .find((b) => b.includes('Happy birthday') && b.includes('/t/'))
1424
+ if (!greeting) {
1425
+ throw new Error(`no greeting body with a /t/ link — bodies: ${JSON.stringify(bdMessages.map((m) => m.body))}`)
1426
+ }
1427
+ const bdToken = greeting.match(/\/t\/([A-Za-z0-9_-]+)/)
1428
+ if (!bdToken) throw new Error(`no /t/<token> in the rendered greeting: ${greeting}`)
1429
+ ctx.birthdayShortPath = bdToken[1]!
1430
+ ```
1431
+
1432
+ ## 030 — resolve the link, and close the loop
1433
+
1434
+ The shortlink resolves through `connectedApps.resolvePage`, and the
1435
+ `route_path` that comes back must match the action's `connected_app_route`.
1436
+ Posting `opened_at` and `form_submitted_at` then mirrors exactly what the
1437
+ form's frontend does when the recipient opens it — which closes the loop
1438
+ from a scheduled greeting to a tracked reply.
1439
+
1440
+ ```typescript
1441
+ const { data: bdResolved } = await api.connectedApps.resolvePage(
1442
+ tenantSlug, datalakeSlug, ctx.birthdayAppSlug,
1443
+ { short_path: ctx.birthdayShortPath, user_agent: 'cookbook-doctest/birthday-greeting' },
1444
+ )
1445
+ if (bdResolved.route_path !== '/forms/birthday-greeting') {
1446
+ throw new Error(`resolvePage route_path mismatch: ${bdResolved.route_path}`)
1447
+ }
1448
+
1449
+ const bdNow = new Date().toISOString()
1450
+ const { data: bdTracked } = await api.connectedApps.updateMessageTracking(
1451
+ tenantSlug, datalakeSlug, ctx.birthdayAppSlug,
1452
+ { short_path: ctx.birthdayShortPath, opened_at: bdNow, form_submitted_at: bdNow },
1453
+ )
1454
+ if (!bdTracked.message?.opened_at || !bdTracked.message?.form_submitted_at) {
1455
+ throw new Error('message tracking did not persist opened_at + form_submitted_at')
1456
+ }
1457
+ ```
1458
+
1459
+ ## 031 — the roster and the audience, as two tables
1460
+
1461
+ A campaign needs two rosters and they are not the same thing. **Direct
1462
+ customers** are the businesses running campaigns; **end customers** are the
1463
+ people a campaign may contact. The platform models neither, because which
1464
+ businesses you serve is domain data.
1465
+
1466
+ Three columns on the audience carry the entire business rule.
1467
+ `suppressed` is do-not-contact — the workflow *filter* reads it, so a
1468
+ suppressed person is unselectable by any campaign rather than merely hidden
1469
+ in some screen. `phone` and `email` are the two channel destinations, and
1470
+ their presence is the reachability gate. `bucket` is the A/B assignment,
1471
+ stamped at ingest and split on by the decision node.
1472
+
1473
+ Note which columns are masked and which are not: the three PII columns are
1474
+ `tokenize`, while `suppressed` and `bucket` are `none`. Campaign mechanics
1475
+ are not personal data, and the workflow has to read them.
1476
+
1477
+ ```typescript
1478
+ const rosterTable = await ctx.ensure(
1479
+ 'Direct Customers table',
1480
+ async () => {
1481
+ const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
1482
+ return (data.data ?? []).find((t: { title?: string }) => t.title === 'Cookbook Direct Customers')
1483
+ },
1484
+ async () => {
1485
+ const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
1486
+ title: 'Cookbook Direct Customers',
1487
+ description: 'The businesses running marketing campaigns.',
1488
+ columns: [
1489
+ { name: 'direct_customer_id', title: 'Direct Customer ID', type: 'string', description: 'Vendor-supplied unique business id', is_unique: true, privacy_requirement: 'none' },
1490
+ { name: 'business_name', title: 'Business Name', type: 'string', description: 'Business display name (public record, not PII)', is_unique: false, privacy_requirement: 'none' },
1491
+ { name: 'sender_phone', title: 'Sender Phone', type: 'string', description: 'Shared number campaigns send SMS from', is_unique: false, privacy_requirement: 'none' },
1492
+ { name: 'sender_domain', title: 'Sender Domain', type: 'string', description: 'Shared domain campaigns send email from', is_unique: false, privacy_requirement: 'none' },
1493
+ ],
1494
+ })
1495
+ return data
1496
+ },
1497
+ )
1498
+ ctx.rosterTableId = rosterTable.id!
1499
+ ctx.rosterTableName = rosterTable.name!
1500
+
1501
+ const audienceTable = await ctx.ensure(
1502
+ 'End Customers table',
1503
+ async () => {
1504
+ const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
1505
+ return (data.data ?? []).find((t: { title?: string }) => t.title === 'Cookbook End Customers')
1506
+ },
1507
+ async () => {
1508
+ const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
1509
+ title: 'Cookbook End Customers',
1510
+ description: 'The people a campaign may contact.',
1511
+ columns: [
1512
+ { name: 'end_customer_id', title: 'End Customer ID', type: 'string', description: 'Vendor-supplied unique end-customer id', is_unique: true, privacy_requirement: 'none' },
1513
+ { name: 'direct_customer_id', title: 'Direct Customer ID', type: 'string', description: 'The business this end customer belongs to', is_unique: false, privacy_requirement: 'none' },
1514
+ { name: 'name', title: 'Name', type: 'string', description: 'Recipient full name', is_unique: false, privacy_requirement: 'tokenize' },
1515
+ { name: 'email', title: 'Email', type: 'string', description: 'Recipient email — the email-channel destination', is_unique: false, privacy_requirement: 'tokenize' },
1516
+ { name: 'phone', title: 'Phone', type: 'string', description: 'Recipient phone — the SMS-channel destination', is_unique: false, privacy_requirement: 'tokenize' },
1517
+ { name: 'suppressed', title: 'Suppressed', type: 'boolean', description: 'Do-not-contact flag — a suppressed recipient is never selected', is_unique: false, privacy_requirement: 'none' },
1518
+ { name: 'bucket', title: 'Bucket', type: 'integer', description: 'Stable 0-99 A/B bucket assigned at ingest; the decision node splits on it', is_unique: false, privacy_requirement: 'none' },
1519
+ ],
1520
+ })
1521
+ return data
1522
+ },
1523
+ )
1524
+ ctx.audienceTableId = audienceTable.id!
1525
+ ctx.audienceTableName = audienceTable.name!
1526
+
1527
+ await ctx.deployGenericTable(ctx.rosterTableName)
1528
+ await ctx.deployGenericTable(ctx.audienceTableName)
1529
+ ```
1530
+
1531
+ ## 032 — the inbound-messages table, which you must not create
1532
+
1533
+ Replies land in `inbound_messages`. It is a **system** table: it ships with
1534
+ every datalake, its `type` is `'system'` rather than `'custom'`, and nobody
1535
+ creates it. You find it by name.
1536
+
1537
+ Trying to create it is the mistake this step exists to prevent — the name is
1538
+ reserved and the physical table is already there, so the attempt fails in a
1539
+ way that reads like a permissions problem.
1540
+
1541
+ ```typescript
1542
+ const { data: allTables } = await api.genericTables.list(tenantSlug, datalakeSlug)
1543
+ const inbound = (allTables.data ?? []).find((t: { name?: string }) => t.name === 'inbound_messages')
1544
+ if (!inbound) {
1545
+ throw new Error('inbound_messages must ship with every datalake')
1546
+ }
1547
+ if (inbound.type !== 'system') {
1548
+ throw new Error(`inbound_messages must be a system table, got type=${inbound.type}`)
1549
+ }
1550
+ ctx.inboundTableId = inbound.id!
1551
+ ctx.inboundTableName = inbound.name!
1552
+ ```
1553
+
1554
+ ## 033 — the roster contracts, on two clients
1555
+
1556
+ Each roster row becomes two things: a row in the Direct Customers table,
1557
+ and a **business** legal entity. That is the contract pair — and the pair
1558
+ goes on **two clients**, for the reason §020 of `payments-compliance`
1559
+ sets out in full and this step restates because it is the single most
1560
+ expensive thing to learn twice.
1561
+
1562
+ A client fans out per row × contract and the job key includes the contract
1563
+ id, so a pair bound to one client runs both contracts **at the same
1564
+ instant**. Both find-or-create the same subject, both miss the dedupe read,
1565
+ both insert, and the unique index on `(uri, id_type, id_number)` refuses the
1566
+ loser. It works on every run except the first.
1567
+
1568
+ **And note the identifier shape in the MDM template**: `uri` / `type` /
1569
+ `value` — not `id_type` / `id_number` / `uri`, which is what the legal-entity
1570
+ body above takes. Three shapes name one idea on this platform, and they
1571
+ overlap unevenly. `system` — what `POST /mdm/verify` takes — is a genuine
1572
+ alias here: `MDMInput` casts it and normalises it onto `uri`, so a template
1573
+ using it resolves correctly. `id_type` and `id_number` are not cast at all,
1574
+ and unknown keys are dropped rather than refused — so borrowing the
1575
+ legal-entity names for the MDM side resolves nothing and fails as a bare
1576
+ `mdm_dispatcher_error` naming no field.
1577
+
1578
+ And the reason the two look interchangeable is that **one becomes the
1579
+ other**. `MDMInput.to_identification/1` takes `(uri, type, value)` and
1580
+ returns `(uri, id_type, id_number)` — the rename happens inside the platform,
1581
+ on the way through. So the identifier reads back under names it will not
1582
+ accept on the way in, and that asymmetry is invisible from the stored shape
1583
+ alone.
1584
+
1585
+ ```typescript
1586
+ const ROSTER_LE = `{% assign p = msg %}
1587
+ {
1588
+ "legal_entity_type": "business",
1589
+ "role": "direct",
1590
+ "business_name": "{{ p.business_name | json_escape }}",
1591
+ "identifications": [
1592
+ {"id_type": "digital_identifier", "uri": "{{ p.source_uri | json_escape }}", "id_number": "{{ p.direct_customer_id | json_escape }}"}
1593
+ ]
1594
+ }`
1595
+
1596
+ const ROSTER_GT = `{% assign p = msg %}
1597
+ {
1598
+ "direct_customer_id": "{{ p.direct_customer_id | json_escape }}",
1599
+ "business_name": "{{ p.business_name | json_escape }}",
1600
+ "sender_phone": "{{ p.sender_phone | json_escape }}",
1601
+ "sender_domain": "{{ p.sender_domain | json_escape }}"
1602
+ }`
1603
+
1604
+ const ROSTER_MDM = `{% assign p = msg %}
1605
+ {
1606
+ "legal_entity_type": "business",
1607
+ "business_name": "{{ p.business_name | json_escape }}",
1608
+ "identifiers": [
1609
+ {"uri": "{{ p.source_uri | json_escape }}", "type": "digital_identifier", "value": "{{ p.direct_customer_id | json_escape }}"}
1610
+ ]
1611
+ }`
1612
+
1613
+ const rosterLe = await ctx.ensure(
1614
+ 'roster business LE contract',
1615
+ async () => {
1616
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1617
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Roster Business LE')
1618
+ },
1619
+ async () => {
1620
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1621
+ name: 'Cookbook Roster Business LE',
1622
+ description: 'Roster row → business LegalEntity.',
1623
+ resource_type: 'legal_entity',
1624
+ type: 'identity',
1625
+ generic_table_id: null,
1626
+ template_config: { type: 'custom', body: ROSTER_LE },
1627
+ mdm_input_config: { type: 'null' },
1628
+ })
1629
+ return data
1630
+ },
1631
+ )
1632
+
1633
+ const rosterGt = await ctx.ensure(
1634
+ 'roster generic-table contract',
1635
+ async () => {
1636
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1637
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Roster GT')
1638
+ },
1639
+ async () => {
1640
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1641
+ name: 'Cookbook Roster GT',
1642
+ description: 'Roster row → Direct Customers row, stamped with its business subject.',
1643
+ resource_type: 'generic_table',
1644
+ type: 'identity',
1645
+ generic_table_id: ctx.rosterTableId,
1646
+ template_config: { type: 'custom', body: ROSTER_GT },
1647
+ mdm_input_config: { type: 'custom', body: ROSTER_MDM },
1648
+ })
1649
+ return data
1650
+ },
1651
+ )
1652
+
1653
+ const rosterIdentityDac = await ctx.ensure(
1654
+ 'roster identity client',
1655
+ async () => {
1656
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
1657
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Roster Identity Client' }],
1658
+ })
1659
+ return (data.data ?? [])[0]
1660
+ },
1661
+ async () => {
1662
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1663
+ name: 'Cookbook Roster Identity Client',
1664
+ description: 'Writes each business as a legal entity, ahead of its roster row.',
1665
+ tool_id: ctx.manualUploadToolId,
1666
+ data_source_id: dataSourceId,
1667
+ tool_call: { tool_call_type: 'manual_upload' },
1668
+ interoperability_contract_ids: [rosterLe.id!],
1669
+ })
1670
+ return data
1671
+ },
1672
+ )
1673
+ ctx.rosterIdentityDacSlug = rosterIdentityDac.slug!
1674
+
1675
+ const rosterDataDac = await ctx.ensure(
1676
+ 'roster data client',
1677
+ async () => {
1678
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
1679
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Roster Data Client' }],
1680
+ })
1681
+ return (data.data ?? [])[0]
1682
+ },
1683
+ async () => {
1684
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1685
+ name: 'Cookbook Roster Data Client',
1686
+ description: 'Writes the roster row, resolving the business the identity client wrote.',
1687
+ tool_id: ctx.manualUploadToolId,
1688
+ data_source_id: dataSourceId,
1689
+ tool_call: { tool_call_type: 'manual_upload' },
1690
+ interoperability_contract_ids: [rosterGt.id!],
1691
+ })
1692
+ return data
1693
+ },
1694
+ )
1695
+ ctx.rosterDataDacSlug = rosterDataDac.slug!
1696
+ ```
1697
+
1698
+ ## 034 — the audience contracts, on two clients
1699
+
1700
+ Same shape, two differences that carry the campaign.
1701
+
1702
+ Each recipient resolves to an **individual** legal entity keyed on their
1703
+ contact handle — phone when there is one, email otherwise. That subject is
1704
+ what the per-recipient short link in §039 is minted against, which is what
1705
+ makes a click attributable to one person rather than to the campaign.
1706
+
1707
+ And the generic-table contract assigns the **bucket**, derived from the last
1708
+ two digits of the vendor's own id rather than from a random draw. Re-ingest
1709
+ the same customer tomorrow and they land in the same bucket. An A/B split
1710
+ that reshuffles on every ingest is not an A/B split — it is noise with a
1711
+ report attached.
1712
+
1713
+ ```typescript
1714
+ const AUDIENCE_LE = `{% assign p = msg %}
1715
+ {% assign name_parts = p.name | default: "" | split: " " %}
1716
+ {% assign first_name = name_parts[0] | default: "" %}
1717
+ {% assign last_name_parts = name_parts | slice: 1, 9 %}
1718
+ {% assign last_name = last_name_parts | join: " " %}
1719
+ {% if p.phone and p.phone != "" %}{% assign handle = p.phone %}{% else %}{% assign handle = p.email %}{% endif %}
1720
+ {
1721
+ "legal_entity_type": "individual",
1722
+ "role": "direct",
1723
+ {% if first_name != "" %}"first_name": "{{ first_name | json_escape }}",{% endif %}
1724
+ {% if last_name != "" %}"last_name": "{{ last_name | json_escape }}",{% endif %}
1725
+ {% if p.phone and p.phone != "" %}"phone_numbers": [{ "phone_number": "{{ p.phone | json_escape }}" }],{% endif %}
1726
+ "identifications": [
1727
+ {"id_type": "digital_identifier", "uri": "{{ p.source_uri | json_escape }}", "id_number": "{{ handle | json_escape }}"}
1728
+ ]
1729
+ }`
1730
+
1731
+ const AUDIENCE_GT = `{% assign p = msg %}
1732
+ {% assign bucket = p.end_customer_id | slice: -2, 2 | plus: 0 %}
1733
+ {
1734
+ "end_customer_id": "{{ p.end_customer_id | json_escape }}",
1735
+ "direct_customer_id": "{{ p.direct_customer_id | json_escape }}",
1736
+ "name": "{{ p.name | default: "" | json_escape }}",
1737
+ "email": "{{ p.email | default: "" | json_escape }}",
1738
+ "phone": "{{ p.phone | default: "" | json_escape }}",
1739
+ "suppressed": {% if p.suppressed %}true{% else %}false{% endif %},
1740
+ "bucket": {{ bucket }}
1741
+ }`
1742
+
1743
+ const AUDIENCE_MDM = `{% assign p = msg %}
1744
+ {% assign name_parts = p.name | default: "" | split: " " %}
1745
+ {% assign first_name = name_parts[0] | default: "" %}
1746
+ {% assign last_name_parts = name_parts | slice: 1, 9 %}
1747
+ {% assign last_name = last_name_parts | join: " " %}
1748
+ {% if p.phone and p.phone != "" %}{% assign handle = p.phone %}{% else %}{% assign handle = p.email %}{% endif %}
1749
+ {
1750
+ "legal_entity_type": "individual",
1751
+ {% if first_name != "" %}"first_name": "{{ first_name | json_escape }}",{% endif %}
1752
+ {% if last_name != "" %}"last_name": "{{ last_name | json_escape }}",{% endif %}
1753
+ "identifiers": [
1754
+ {"uri": "{{ p.source_uri | json_escape }}", "type": "digital_identifier", "value": "{{ handle | json_escape }}"}
1755
+ ]
1756
+ }`
1757
+
1758
+ const audienceLe = await ctx.ensure(
1759
+ 'audience LE contract',
1760
+ async () => {
1761
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1762
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Audience LE')
1763
+ },
1764
+ async () => {
1765
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1766
+ name: 'Cookbook Audience LE',
1767
+ description: 'Audience row → end-customer LegalEntity, keyed on the contact handle.',
1768
+ resource_type: 'legal_entity',
1769
+ type: 'identity',
1770
+ generic_table_id: null,
1771
+ template_config: { type: 'custom', body: AUDIENCE_LE },
1772
+ mdm_input_config: { type: 'null' },
1773
+ })
1774
+ return data
1775
+ },
1776
+ )
1777
+
1778
+ const audienceGt = await ctx.ensure(
1779
+ 'audience generic-table contract',
1780
+ async () => {
1781
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
1782
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Audience GT')
1783
+ },
1784
+ async () => {
1785
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
1786
+ name: 'Cookbook Audience GT',
1787
+ description: 'Audience row → End Customers row, stamped with its subject and its stable A/B bucket.',
1788
+ resource_type: 'generic_table',
1789
+ type: 'identity',
1790
+ generic_table_id: ctx.audienceTableId,
1791
+ template_config: { type: 'custom', body: AUDIENCE_GT },
1792
+ mdm_input_config: { type: 'custom', body: AUDIENCE_MDM },
1793
+ })
1794
+ return data
1795
+ },
1796
+ )
1797
+
1798
+ const audienceIdentityDac = await ctx.ensure(
1799
+ 'audience identity client',
1800
+ async () => {
1801
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
1802
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Audience Identity Client' }],
1803
+ })
1804
+ return (data.data ?? [])[0]
1805
+ },
1806
+ async () => {
1807
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1808
+ name: 'Cookbook Audience Identity Client',
1809
+ description: 'Writes each recipient as a legal entity, ahead of their audience row.',
1810
+ tool_id: ctx.manualUploadToolId,
1811
+ data_source_id: dataSourceId,
1812
+ tool_call: { tool_call_type: 'manual_upload' },
1813
+ interoperability_contract_ids: [audienceLe.id!],
1814
+ })
1815
+ return data
1816
+ },
1817
+ )
1818
+ ctx.audienceIdentityDacSlug = audienceIdentityDac.slug!
1819
+
1820
+ const audienceDataDac = await ctx.ensure(
1821
+ 'audience data client',
1822
+ async () => {
1823
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
1824
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Audience Data Client' }],
1825
+ })
1826
+ return (data.data ?? [])[0]
1827
+ },
1828
+ async () => {
1829
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
1830
+ name: 'Cookbook Audience Data Client',
1831
+ description: 'Writes the audience row, resolving the recipient the identity client wrote.',
1832
+ tool_id: ctx.manualUploadToolId,
1833
+ data_source_id: dataSourceId,
1834
+ tool_call: { tool_call_type: 'manual_upload' },
1835
+ interoperability_contract_ids: [audienceGt.id!],
1836
+ })
1837
+ return data
1838
+ },
1839
+ )
1840
+ ctx.audienceDataDacSlug = audienceDataDac.slug!
1841
+ ```
1842
+
1843
+ ## 035 — ingest one business and four recipients, identities first
1844
+
1845
+ Four recipients chosen so a single run exercises every branch at once:
1846
+
1847
+ | Recipient | phone | email | suppressed | bucket | expected |
1848
+ |-----------|-------|-------|------------|--------|----------|
1849
+ | Ada | yes | yes | no | 10 | sent, variant A |
1850
+ | Grace | yes | yes | no | 80 | sent, variant B |
1851
+ | Sup | yes | yes | **yes** | 31 | filtered — suppressed |
1852
+ | Nophone | — | yes | no | 42 | filtered — unreachable |
1853
+
1854
+ The buckets are not set by hand. They fall out of the ids ending `-10`,
1855
+ `-80`, `-31`, `-42`, which is §034's Liquid doing the assignment.
1856
+
1857
+ Identities go first and are waited on, then the rows. Within a stage the
1858
+ four recipients have four different handles, so they never contend with each
1859
+ other — the ordering that matters is between the stages, not inside them.
1860
+
1861
+ ```typescript
1862
+ ctx.rosterId = 'DC-COOKBOOK-01'
1863
+ ctx.adaId = 'EC-COOKBOOK-10'
1864
+ ctx.graceId = 'EC-COOKBOOK-80'
1865
+ ctx.supId = 'EC-COOKBOOK-31'
1866
+ ctx.nophoneId = 'EC-COOKBOOK-42'
1867
+ ctx.adaPhone = '+15550200010'
1868
+
1869
+ const rosterRow = {
1870
+ direct_customer_id: ctx.rosterId,
1871
+ business_name: 'Bloom Salon',
1872
+ sender_phone: '+15550001111',
1873
+ sender_domain: 'bloom-salon.example.com',
1874
+ }
1875
+ const audienceRows = [
1876
+ { end_customer_id: ctx.adaId, direct_customer_id: ctx.rosterId, name: 'Ada Lovelace', email: 'ada@cookbook.example.com', phone: ctx.adaPhone, suppressed: false },
1877
+ { end_customer_id: ctx.graceId, direct_customer_id: ctx.rosterId, name: 'Grace Hopper', email: 'grace@cookbook.example.com', phone: '+15550200080', suppressed: false },
1878
+ { end_customer_id: ctx.supId, direct_customer_id: ctx.rosterId, name: 'Sup Pressed', email: 'sup@cookbook.example.com', phone: '+15550200031', suppressed: true },
1879
+ { end_customer_id: ctx.nophoneId, direct_customer_id: ctx.rosterId, name: 'No Phone', email: 'nophone@cookbook.example.com', phone: '', suppressed: false },
1880
+ ]
1881
+
1882
+ // Stage 1 — the business, then the people.
1883
+ const rosterIdent = await api.dataActivationClients.ingest(
1884
+ tenantSlug, datalakeSlug, ctx.rosterIdentityDacSlug, { data: rosterRow },
1885
+ )
1886
+ await ctx.waitForBatches(ctx.rosterIdentityDacSlug, [rosterIdent.data.batch_id!])
1887
+
1888
+ const audienceIdentBatches: string[] = []
1889
+ for (const row of audienceRows) {
1890
+ const r = await api.dataActivationClients.ingest(
1891
+ tenantSlug, datalakeSlug, ctx.audienceIdentityDacSlug, { data: row },
1892
+ )
1893
+ audienceIdentBatches.push(r.data.batch_id!)
1894
+ }
1895
+ await ctx.waitForBatches(ctx.audienceIdentityDacSlug, audienceIdentBatches)
1896
+
1897
+ // Stage 2 — the rows, which resolve the subjects that now exist.
1898
+ const rosterData = await api.dataActivationClients.ingest(
1899
+ tenantSlug, datalakeSlug, ctx.rosterDataDacSlug, { data: rosterRow },
1900
+ )
1901
+ await ctx.waitForBatches(ctx.rosterDataDacSlug, [rosterData.data.batch_id!])
1902
+
1903
+ const audienceDataBatches: string[] = []
1904
+ for (const row of audienceRows) {
1905
+ const r = await api.dataActivationClients.ingest(
1906
+ tenantSlug, datalakeSlug, ctx.audienceDataDacSlug, { data: row },
1907
+ )
1908
+ audienceDataBatches.push(r.data.batch_id!)
1909
+ }
1910
+ await ctx.waitForBatches(ctx.audienceDataDacSlug, audienceDataBatches)
1911
+ ```
1912
+
1913
+ ## 036 — the stamp is the proof that resolution ran
1914
+
1915
+ Read the four rows back. Every one carries a `legal_entity_id`, and the four
1916
+ are **distinct** — four recipients, four subjects. A shared subject would
1917
+ mean two people collapsed into one, and a short link pointing at the wrong
1918
+ customer.
1919
+
1920
+ The buckets are checked here too, because a wrong bucket is the kind of
1921
+ thing that stays invisible until someone asks why variant A outperformed B
1922
+ by a suspiciously round margin.
1923
+
1924
+ ```typescript
1925
+ const rowsDeadline = Date.now() + 120_000
1926
+ let audienceRowsRead: Array<Record<string, unknown>> = []
1927
+ while (Date.now() < rowsDeadline && audienceRowsRead.length < 4) {
1928
+ const { data: result } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
1929
+ sql: `SELECT end_customer_id, bucket, legal_entity_id
1930
+ FROM ${ctx.audienceTableName}
1931
+ WHERE end_customer_id LIKE 'EC-COOKBOOK-%'`,
1932
+ mode: 'raw',
1933
+ })
1934
+ audienceRowsRead = (result.data ?? []).map((r) =>
1935
+ Object.fromEntries(result.meta.columns.map((c, i) => [c, r[i]])),
1936
+ )
1937
+ if (audienceRowsRead.length < 4) await new Promise((r) => setTimeout(r, 2_000))
1938
+ }
1939
+ if (audienceRowsRead.length !== 4) {
1940
+ throw new Error(`expected 4 audience rows within 120s, got ${audienceRowsRead.length}`)
1941
+ }
1942
+
1943
+ const byCustomerId = new Map(audienceRowsRead.map((r) => [String(r.end_customer_id), r]))
1944
+ if (Number(byCustomerId.get(ctx.adaId)!.bucket) !== 10) throw new Error('Ada must land in bucket 10')
1945
+ if (Number(byCustomerId.get(ctx.graceId)!.bucket) !== 80) throw new Error('Grace must land in bucket 80')
1946
+
1947
+ for (const row of audienceRowsRead) {
1948
+ if (!row.legal_entity_id) {
1949
+ throw new Error(`${row.end_customer_id} has no stamped subject — resolution did not run`)
1950
+ }
1951
+ }
1952
+ const subjects = new Set(audienceRowsRead.map((r) => String(r.legal_entity_id)))
1953
+ if (subjects.size !== 4) {
1954
+ throw new Error(`four recipients must be four distinct subjects, got ${subjects.size}`)
1955
+ }
1956
+ ctx.adaSubjectId = String(byCustomerId.get(ctx.adaId)!.legal_entity_id)
1957
+ ```
1958
+
1959
+ ## 037 — two sender tools, and the booking form
1960
+
1961
+ **Two sender tools, not one.** An SMS number and an email identity are
1962
+ different assets, provisioned and billed separately, so §038 binds each
1963
+ channel to its own.
1964
+
1965
+ The email tool uses `provider: 'mock'`, which delivers in-process to the dev
1966
+ mailbox. Note *where* that setting lives: it is a field on the **tool body**
1967
+ — data on a row — not an environment variable. So the same manifest runs
1968
+ against any server without asking the deployment to be in "test mode", and
1969
+ swapping `mock` for a real provider sends real email through the same
1970
+ workflow.
1971
+
1972
+ ```typescript
1973
+ const campaignSms = await ctx.ensure(
1974
+ 'campaign SMS sender',
1975
+ async () => {
1976
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
1977
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Campaign SMS Sender')
1978
+ },
1979
+ async () => {
1980
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
1981
+ name: 'Cookbook Campaign SMS Sender',
1982
+ description: 'SNS-backed SMS dispatcher for the campaign, wired to LocalStack.',
1983
+ intent: 'sms',
1984
+ status: 'active',
1985
+ datalake_id: ctx.datalakeId,
1986
+ body: {
1987
+ tool_body_type: 'sns',
1988
+ auth_method: 'access_key',
1989
+ region: 'us-east-1',
1990
+ phone_number: '+15550001111',
1991
+ endpoint_url: 'http://localhost:4566',
1992
+ access_key_id: 'test',
1993
+ secret_access_key: 'test',
1994
+ },
1995
+ })
1996
+ return data
1997
+ },
1998
+ )
1999
+ ctx.campaignSmsToolId = campaignSms.id!
2000
+
2001
+ const campaignEmail = await ctx.ensure(
2002
+ 'campaign email sender',
2003
+ async () => {
2004
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
2005
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Campaign Email Sender')
2006
+ },
2007
+ async () => {
2008
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
2009
+ name: 'Cookbook Campaign Email Sender',
2010
+ description: 'Campaign email dispatcher — the mock provider delivers to the dev mailbox.',
2011
+ intent: 'email',
2012
+ status: 'active',
2013
+ datalake_id: ctx.datalakeId,
2014
+ body: {
2015
+ tool_body_type: 'email',
2016
+ provider: 'mock',
2017
+ from_email: 'campaigns@bloom-salon.example.com',
2018
+ from_name: 'Bloom Salon',
2019
+ },
2020
+ })
2021
+ return data
2022
+ },
2023
+ )
2024
+ ctx.campaignEmailToolId = campaignEmail.id!
2025
+
2026
+ const bookingApp = await ctx.ensure(
2027
+ 'campaign booking form',
2028
+ async () => {
2029
+ const { data } = await api.connectedApps.list(tenantSlug, datalakeSlug)
2030
+ return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Campaign Booking Form')
2031
+ },
2032
+ async () => {
2033
+ const { data } = await api.connectedApps.create(tenantSlug, datalakeSlug, {
2034
+ name: 'Cookbook Campaign Booking Form',
2035
+ description: 'The booking form the campaign links to.',
2036
+ mode: 'self_hosted',
2037
+ urls: [{ url: 'https://campaign.example.com', is_primary: true, label: 'production' }],
2038
+ })
2039
+ return data
2040
+ },
2041
+ )
2042
+ ctx.bookingAppId = bookingApp.id!
2043
+ ctx.bookingAppSlug = bookingApp.slug!
2044
+ ```
2045
+
2046
+ ## 038 — the campaign workflow: the gates, the split, both channels
2047
+
2048
+ All four rules live here, and none of them lives in application code.
2049
+
2050
+ **The gates are the filter.** `{% unless event_dataset.suppressed %}` is
2051
+ suppression; the two non-empty checks are reachability. A row failing either
2052
+ is `:filtered` — looked at, and deliberately not contacted. There is no code
2053
+ path around it, which is exactly what you want to be able to say about a
2054
+ do-not-contact list.
2055
+
2056
+ **The split is the decision node**, and it returns an *array* — both channel
2057
+ keys of the chosen variant. Bucket under fifty gets the A pair, otherwise
2058
+ the B pair. Four actions declare a `decision_key` each; the platform runs
2059
+ the two whose keys came back and skips the other two. That is one recipient,
2060
+ one variant, two channels, from one run — and the pairing is data rather
2061
+ than two workflows kept in step by hand.
2062
+
2063
+ **`skip_mdm_resolution: false` is set explicitly, and must be.** A
2064
+ generic-table row is never its own subject; the subject is the one §034
2065
+ stamped onto it. With resolution on, the platform loads that subject and
2066
+ mints a page token against it, which is what makes
2067
+ `{{ connected_app_form_url }}` a *per-recipient* link. Skip it and the
2068
+ subject is nil, no token is minted, and the link renders empty — with
2069
+ nothing anywhere reporting a problem.
2070
+
2071
+ `output_schema` is a JSON-Schema **object**. The string form is refused 422.
2072
+
2073
+ ```typescript
2074
+ const CAMPAIGN_FILTER =
2075
+ '{% unless event_dataset.suppressed %}' +
2076
+ '{% if event_dataset.phone and event_dataset.phone != "" %}' +
2077
+ '{% if event_dataset.email and event_dataset.email != "" %}true{% endif %}' +
2078
+ '{% endif %}{% endunless %}'
2079
+ const AB_DECISION =
2080
+ '{% if event_dataset.bucket < 50 %}["sms_variant_a","email_variant_a"]' +
2081
+ '{% else %}["sms_variant_b","email_variant_b"]{% endif %}'
2082
+ const VARIANT_A = 'Hi {{ event_dataset.name }} — 20% off your next visit this week only. Book: {{ connected_app_form_url }}'
2083
+ const VARIANT_B = 'Hi {{ event_dataset.name }} — your loyalty reward is waiting. Claim: {{ connected_app_form_url }}'
2084
+
2085
+ const linkage = {
2086
+ trigger_template: 'now',
2087
+ idempotency_template: '{{ event_dataset.end_customer_id }}/{{ action_id }}',
2088
+ connected_app_id: ctx.bookingAppId,
2089
+ connected_app_route: '/forms/campaign',
2090
+ connected_app_metadata_template:
2091
+ '{"end_customer_id":"{{ event_dataset.end_customer_id }}","name":"{{ event_dataset.name }}"}',
2092
+ }
2093
+
2094
+ const campaign = await ctx.ensure(
2095
+ 'loyalty campaign workflow',
2096
+ async () => {
2097
+ const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
2098
+ return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Loyalty Campaign')
2099
+ },
2100
+ async () => {
2101
+ const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
2102
+ name: 'Cookbook Loyalty Campaign',
2103
+ description: 'Loyalty campaign with A/B content variants across SMS and email.',
2104
+ dataset_type: 'generic_table',
2105
+ generic_table_id: ctx.audienceTableId,
2106
+ status: 'live',
2107
+ tags: ['marketing', 'loyalty'],
2108
+ skip_mdm_resolution: false,
2109
+ filter_config: { type: 'custom', body: CAMPAIGN_FILTER },
2110
+ decision_config: {
2111
+ type: 'custom',
2112
+ body: AB_DECISION,
2113
+ output_schema: { type: 'array', items: { type: 'string' } },
2114
+ },
2115
+ actions: [
2116
+ {
2117
+ ...linkage,
2118
+ action_type: 'sms', tool_id: ctx.campaignSmsToolId, decision_key: 'sms_variant_a', position: 0,
2119
+ tool_call: {
2120
+ tool_call_type: 'sms_request',
2121
+ to: { type: 'custom', body: '{{ event_dataset.phone }}' },
2122
+ body: { type: 'custom', body: VARIANT_A },
2123
+ sms_type: 'transactional',
2124
+ },
2125
+ },
2126
+ {
2127
+ ...linkage,
2128
+ action_type: 'sms', tool_id: ctx.campaignSmsToolId, decision_key: 'sms_variant_b', position: 1,
2129
+ tool_call: {
2130
+ tool_call_type: 'sms_request',
2131
+ to: { type: 'custom', body: '{{ event_dataset.phone }}' },
2132
+ body: { type: 'custom', body: VARIANT_B },
2133
+ sms_type: 'transactional',
2134
+ },
2135
+ },
2136
+ {
2137
+ ...linkage,
2138
+ action_type: 'email', tool_id: ctx.campaignEmailToolId, decision_key: 'email_variant_a', position: 2,
2139
+ tool_call: {
2140
+ tool_call_type: 'email_request',
2141
+ to: { type: 'custom', body: '{{ event_dataset.email }}' },
2142
+ subject: { type: 'custom', body: 'Your 20% off is here' },
2143
+ body: { type: 'custom', body: VARIANT_A },
2144
+ },
2145
+ },
2146
+ {
2147
+ ...linkage,
2148
+ action_type: 'email', tool_id: ctx.campaignEmailToolId, decision_key: 'email_variant_b', position: 3,
2149
+ tool_call: {
2150
+ tool_call_type: 'email_request',
2151
+ to: { type: 'custom', body: '{{ event_dataset.email }}' },
2152
+ subject: { type: 'custom', body: 'Your loyalty reward is waiting' },
2153
+ body: { type: 'custom', body: VARIANT_B },
2154
+ },
2155
+ },
2156
+ ],
2157
+ })
2158
+ return data
2159
+ },
2160
+ )
2161
+ ctx.campaignWorkflowSlug = campaign.slug!
2162
+
2163
+ if (campaign.skip_mdm_resolution !== false) {
2164
+ throw new Error('a generic-table campaign workflow must keep MDM resolution ON')
2165
+ }
2166
+ if ((campaign.actions ?? []).length !== 4) {
2167
+ throw new Error(`expected 4 actions, got ${(campaign.actions ?? []).length}`)
2168
+ }
2169
+ ```
2170
+
2171
+ ## 039 — run the campaign, and read the split off the logs
2172
+
2173
+ One pass over all four recipients: both gates and the split in the same run.
2174
+
2175
+ The per-recipient action logs are the review surface. Each sending
2176
+ recipient's execution log carries four action logs, and the two that ran are
2177
+ the **same variant on both channels**. Group the completed ones by
2178
+ `decision_key` and you have per-variant performance without building a
2179
+ reporting pipeline.
2180
+
2181
+ **On a repeat run all four are skipped**, because the idempotency key fired
2182
+ the first time — so the assertion accepts either shape, and rejects anything
2183
+ else. Two completed from *different* variants would mean the split leaked;
2184
+ that is the failure worth catching here.
2185
+
2186
+ ```typescript
2187
+ const { data: campaignRun } = await api.workflows.run(tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug, {
2188
+ sql_where_clause: `end_customer_id LIKE 'EC-COOKBOOK-%'`,
2189
+ mode: 'live',
2190
+ manual_override: false,
2191
+ })
2192
+ const campaignFired = await ctx.waitForFiredRun(datalakeSlug, campaignRun.workflow_run_id)
2193
+ ctx.campaignRunLogId = campaignFired.workflowRunLogId
2194
+ ctx.campaignRunBatchId = campaignFired.batchId!
2195
+
2196
+ const campaignDeadline = Date.now() + 180_000
2197
+ let campaignByStatus: Record<string, number> = {}
2198
+ while (Date.now() < campaignDeadline) {
2199
+ const { data: log } = await api.workflows.batchLogs.refresh(
2200
+ tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug, ctx.campaignRunLogId,
2201
+ )
2202
+ if (log.status === 'failed') throw new Error('campaign run reached :failed')
2203
+
2204
+ const { data: wfLogs } = await api.workflows.workflowLogs.list(tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug)
2205
+ const ours = (wfLogs.data ?? []).filter(
2206
+ (w) => (w as { batch_id?: string }).batch_id === ctx.campaignRunBatchId,
2207
+ )
2208
+ campaignByStatus = {}
2209
+ for (const wel of ours) {
2210
+ const st = (wel as { status?: string }).status ?? 'unknown'
2211
+ campaignByStatus[st] = (campaignByStatus[st] ?? 0) + 1
2212
+ }
2213
+ if ((campaignByStatus.completed ?? 0) >= 2 && (campaignByStatus.filtered ?? 0) >= 2) break
2214
+ await new Promise((r) => setTimeout(r, 3_000))
2215
+ }
2216
+ if ((campaignByStatus.completed ?? 0) !== 2 || (campaignByStatus.filtered ?? 0) !== 2) {
2217
+ throw new Error(`expected 2 completed + 2 filtered, got ${JSON.stringify(campaignByStatus)}`)
2218
+ }
2219
+
2220
+ const { data: campaignLogs } = await api.workflows.workflowLogs.list(
2221
+ tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug,
2222
+ )
2223
+ const sentWels = (campaignLogs.data ?? []).filter((w) => {
2224
+ const wel = w as { batch_id?: string; status?: string }
2225
+ return wel.batch_id === ctx.campaignRunBatchId && wel.status === 'completed'
2226
+ })
2227
+
2228
+ for (const wel of sentWels) {
2229
+ const aels = (wel as { action_execution_logs?: Array<{ status?: string; decision_key?: string }> })
2230
+ .action_execution_logs ?? []
2231
+ if (aels.length !== 4) throw new Error(`four actions → four action logs, got ${aels.length}`)
2232
+
2233
+ const completed = aels.filter((a) => a.status === 'completed').map((a) => a.decision_key ?? '').sort()
2234
+ const skipped = aels.filter((a) => a.status === 'skipped')
2235
+
2236
+ if (completed.length === 0 && skipped.length === 4) continue // already sent on an earlier run
2237
+
2238
+ if (completed.length !== 2 || skipped.length !== 2) {
2239
+ throw new Error(`expected 2 completed + 2 skipped action logs, got ${JSON.stringify(aels)}`)
2240
+ }
2241
+ const isA = completed[0] === 'email_variant_a' && completed[1] === 'sms_variant_a'
2242
+ const isB = completed[0] === 'email_variant_b' && completed[1] === 'sms_variant_b'
2243
+ if (!isA && !isB) {
2244
+ throw new Error(`the completed pair must be one variant on both channels, got ${completed.join(', ')}`)
2245
+ }
2246
+ }
2247
+ ```
2248
+
2249
+ ## 040 — the short link belongs to one recipient
2250
+
2251
+ Every message the campaign rendered carries a `/t/<hash>`. It is **not** one
2252
+ campaign link shared by everyone: the token was minted against this
2253
+ recipient's own subject — the one §034 stamped and §038 resolved — so a
2254
+ click is attributable to a person rather than to a send.
2255
+
2256
+ Keyed on the idempotency key again, so this reads the durable record and
2257
+ answers on every run rather than only the one that dispatched.
2258
+
2259
+ ```typescript
2260
+ const { data: campaignMsgSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
2261
+ search_query: `m.idempotency_key LIKE '${ctx.adaId}/%'`,
2262
+ })
2263
+ if (campaignMsgSearch.status !== 'completed') {
2264
+ throw new Error(`message user-search status=${campaignMsgSearch.status} error=${campaignMsgSearch.error_message ?? '(none)'}`)
2265
+ }
2266
+
2267
+ const linkDeadline = Date.now() + 120_000
2268
+ let campaignShortPath: string | undefined
2269
+ while (Date.now() < linkDeadline && !campaignShortPath) {
2270
+ const { data: found } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
2271
+ userSearchId: campaignMsgSearch.id!,
2272
+ dataAccessMode: 'raw',
2273
+ })
2274
+ const bodies = ((found.data ?? []) as Array<Record<string, unknown>>).map((m) => String(m.body ?? ''))
2275
+ campaignShortPath = bodies.find((b) => b.includes('/t/'))?.match(/\/t\/([A-Za-z0-9_-]+)/)?.[1]
2276
+ if (!campaignShortPath) await new Promise((r) => setTimeout(r, 2_000))
2277
+ }
2278
+ if (!campaignShortPath) {
2279
+ throw new Error(`no campaign message for ${ctx.adaId} carried a /t/ short link within 120s`)
2280
+ }
2281
+
2282
+ const { data: campaignResolved } = await api.connectedApps.resolvePage(
2283
+ tenantSlug, datalakeSlug, ctx.bookingAppSlug,
2284
+ { short_path: campaignShortPath, user_agent: 'cookbook-doctest/marketing-campaign-send' },
2285
+ )
2286
+ if (campaignResolved.route_path !== '/forms/campaign') {
2287
+ throw new Error(`resolvePage route_path mismatch: ${campaignResolved.route_path}`)
2288
+ }
2289
+
2290
+ const { data: campaignTracked } = await api.connectedApps.updateMessageTracking(
2291
+ tenantSlug, datalakeSlug, ctx.bookingAppSlug,
2292
+ { short_path: campaignShortPath, opened_at: new Date().toISOString() },
2293
+ )
2294
+ if (!campaignTracked.message?.opened_at) {
2295
+ throw new Error('message tracking did not persist opened_at')
2296
+ }
2297
+ ```
2298
+
2299
+ ## 041 — the reply comes back, and attaches to whoever sent it
2300
+
2301
+ A reply arrives on the same handle the campaign sent to. It ingests into the
2302
+ **system** inbound table through a client whose `mdm_input_config` keys on
2303
+ the same `(uri, value)` namespace §034 wrote — so the platform resolves the
2304
+ handle straight back to the customer and stamps their subject on the row.
2305
+
2306
+ **This client carries only a generic-table contract**, and that is not an
2307
+ oversight: a reply does not create a person, it finds one. Adding a
2308
+ legal-entity contract here would mint a second subject for someone who
2309
+ already exists — and would put two writers on one subject, which §033
2310
+ explains at length.
2311
+
2312
+ **When a handle is ambiguous the reply still attaches.** If two capture
2313
+ systems each registered a customer under the same phone, that handle maps to
2314
+ two entities. The row lands anyway, attached to one of them, and is marked
2315
+ `potential_duplicate` for a human to adjudicate. Flag, don't block: a dropped
2316
+ reply is a lost customer, a flagged one is a two-minute review.
2317
+
2318
+ Below, the reply lands and attaches to Ada; then the same `message_id` is
2319
+ re-ingested with the flag set. `message_id` is the table's unique column, so
2320
+ the second ingest **upserts** — one row, now flagged, not a second copy.
2321
+
2322
+ ```typescript
2323
+ const REPLY_GT = `{% assign p = msg %}
2324
+ {
2325
+ "message_id": "{{ p.message_id | json_escape }}",
2326
+ "handle": "{{ p.handle | json_escape }}",
2327
+ "channel": "{{ p.channel | json_escape }}",
2328
+ "body": "{{ p.body | json_escape }}",
2329
+ "potential_duplicate": {% if p.potential_duplicate %}true{% else %}false{% endif %}
2330
+ }`
2331
+
2332
+ const REPLY_MDM = `{% assign p = msg %}
2333
+ {
2334
+ "legal_entity_type": "individual",
2335
+ "identifiers": [
2336
+ {"uri": "{{ p.source_uri | json_escape }}", "type": "digital_identifier", "value": "{{ p.handle | json_escape }}"}
2337
+ ]
2338
+ }`
2339
+
2340
+ const replyGt = await ctx.ensure(
2341
+ 'inbound reply contract',
2342
+ async () => {
2343
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
2344
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Reply GT')
2345
+ },
2346
+ async () => {
2347
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
2348
+ name: 'Cookbook Reply GT',
2349
+ description: 'Inbound reply → Inbound Messages row, re-attached to the customer who sent it.',
2350
+ resource_type: 'generic_table',
2351
+ type: 'identity',
2352
+ generic_table_id: ctx.inboundTableId,
2353
+ template_config: { type: 'custom', body: REPLY_GT },
2354
+ mdm_input_config: { type: 'custom', body: REPLY_MDM },
2355
+ })
2356
+ return data
2357
+ },
2358
+ )
2359
+
2360
+ const replyDac = await ctx.ensure(
2361
+ 'inbound reply client',
2362
+ async () => {
2363
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
2364
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Reply Client' }],
2365
+ })
2366
+ return (data.data ?? [])[0]
2367
+ },
2368
+ async () => {
2369
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
2370
+ name: 'Cookbook Reply Client',
2371
+ description: 'Inbound reply ingest — re-attaches the sender handle to its subject.',
2372
+ tool_id: ctx.manualUploadToolId,
2373
+ data_source_id: dataSourceId,
2374
+ tool_call: { tool_call_type: 'manual_upload' },
2375
+ interoperability_contract_ids: [replyGt.id!],
2376
+ })
2377
+ return data
2378
+ },
2379
+ )
2380
+
2381
+ const replyMessageId = 'IN-COOKBOOK-0101'
2382
+
2383
+ const readReply = async (): Promise<Record<string, unknown> | undefined> => {
2384
+ const { data: result } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
2385
+ sql: `SELECT message_id, legal_entity_id, potential_duplicate
2386
+ FROM ${ctx.inboundTableName}
2387
+ WHERE message_id = '${replyMessageId}'`,
2388
+ mode: 'raw',
2389
+ })
2390
+ if ((result.data ?? []).length === 0) return undefined
2391
+ return Object.fromEntries(result.meta.columns.map((c, i) => [c, result.data[0][i]]))
2392
+ }
2393
+
2394
+ const firstReply = await api.dataActivationClients.ingest(
2395
+ tenantSlug, datalakeSlug, replyDac.slug!,
2396
+ { data: { message_id: replyMessageId, handle: ctx.adaPhone, channel: 'sms', body: 'Yes! Book me for Friday.' } },
2397
+ )
2398
+ await ctx.waitForBatches(replyDac.slug!, [firstReply.data.batch_id!])
2399
+
2400
+ const reply = await readReply()
2401
+ if (!reply) throw new Error('the reply did not land in the inbound messages table')
2402
+ if (String(reply.legal_entity_id) !== ctx.adaSubjectId) {
2403
+ throw new Error('the reply must re-attach to the customer who sent it')
2404
+ }
2405
+
2406
+ // Same message_id, flag set — the unique column makes this an upsert.
2407
+ const flagReply = await api.dataActivationClients.ingest(
2408
+ tenantSlug, datalakeSlug, replyDac.slug!,
2409
+ {
2410
+ data: {
2411
+ message_id: replyMessageId,
2412
+ handle: ctx.adaPhone,
2413
+ channel: 'sms',
2414
+ body: 'Yes! Book me for Friday.',
2415
+ potential_duplicate: true,
2416
+ },
2417
+ },
2418
+ )
2419
+ await ctx.waitForBatches(replyDac.slug!, [flagReply.data.batch_id!])
2420
+
2421
+ const flagged = await readReply()
2422
+ if (!flagged || flagged.potential_duplicate !== true) {
2423
+ throw new Error('the re-ingest did not flag the reply potential_duplicate')
2424
+ }
2425
+ ```
2426
+
2427
+ ## 042 — reconciling what actually got delivered
2428
+
2429
+ Everything above ends at *sent*. Sent is not delivered, and the gap between
2430
+ them is where a campaign quietly stops working — a number that has been
2431
+ disconnected for months keeps reporting `sent` forever.
2432
+
2433
+ An **action status updater** closes that gap on a schedule. It names a
2434
+ poller tool that reads a delivery log, the sender tools it reconciles for,
2435
+ and two Liquid templates that turn each provider event into an update.
2436
+
2437
+ Three details that are each a 422 or a silent nothing if you get them wrong.
2438
+
2439
+ **`start_time` and `end_time` use `now_msec`, not `now`.** `now_msec` is the
2440
+ injected variable holding unix milliseconds, which is what
2441
+ `minutes_ago` expects; the bare `now` is a DateTime and raises.
2442
+
2443
+ **`message_config` renders once per event, not once per batch.** The event
2444
+ itself is the assigns — there is no `events` list to loop over — and it must
2445
+ emit a *flat* object: the `external_id` of the message to update, plus the
2446
+ fields to set, at the top level.
2447
+
2448
+ **`action_log_config` is required alongside it.** Same per-event assigns,
2449
+ rendered into the action-log write shape. Omitting it is a 422 that names
2450
+ the field: `Missing field: action_log_config`.
2451
+
2452
+ ```typescript
2453
+ const cwTool = await ctx.ensure(
2454
+ 'CloudWatch poller tool',
2455
+ async () => {
2456
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
2457
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook CloudWatch Poller')
2458
+ },
2459
+ async () => {
2460
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
2461
+ name: 'Cookbook CloudWatch Poller',
2462
+ description: 'CloudWatch log-group poller — reads SNS delivery events for reconciliation.',
2463
+ intent: 'status_poller',
2464
+ status: 'active',
2465
+ datalake_id: ctx.datalakeId,
2466
+ data_source_id: dataSourceId,
2467
+ body: {
2468
+ tool_body_type: 'cloud_watch_log_group',
2469
+ auth_method: 'access_key',
2470
+ region: 'us-east-1',
2471
+ endpoint_url: 'http://localhost:4566',
2472
+ access_key_id: 'test',
2473
+ secret_access_key: 'test',
2474
+ },
2475
+ })
2476
+ return data
2477
+ },
2478
+ )
2479
+ ctx.cwToolId = cwTool.id!
2480
+
2481
+ const deliveryUpdater = await ctx.ensure(
2482
+ 'SMS delivery reconciler',
2483
+ async () => {
2484
+ const { data } = await api.actionStatusUpdaters.list(tenantSlug, datalakeSlug)
2485
+ return (data.data ?? []).find((u: { name?: string }) => u.name === 'Cookbook SMS Delivery Updater')
2486
+ },
2487
+ async () => {
2488
+ const { data } = await api.actionStatusUpdaters.create(tenantSlug, datalakeSlug, {
2489
+ name: 'Cookbook SMS Delivery Updater',
2490
+ cron_expression: '*/30 * * * *',
2491
+ updater_type: 'cloud_watch',
2492
+ updater_tool_id: ctx.cwToolId,
2493
+ sender_tool_ids: [ctx.smsToolId],
2494
+ datalake_id: ctx.datalakeId,
2495
+ updater_body: {
2496
+ updater_body_type: 'cloud_watch_request',
2497
+ log_group_name: 'sns/us-east-1/000000000000/DirectPublishToPhoneNumber',
2498
+ start_time: '{{ now_msec | minutes_ago: 45 }}',
2499
+ end_time: '{{ now_msec }}',
2500
+ },
2501
+ message_config: {
2502
+ type: 'custom',
2503
+ body: '{"external_id": "{{ notification.messageId }}", "status": "delivered"}',
2504
+ },
2505
+ action_log_config: {
2506
+ type: 'custom',
2507
+ body: '{"external_id": "{{ notification.messageId }}", "status": "delivered"}',
2508
+ },
2509
+ })
2510
+ return data
2511
+ },
2512
+ )
2513
+ actionStatusUpdaterId = deliveryUpdater.id!
2514
+ ```
2515
+
2516
+ ## 043 — read it back, and check the checksum
2517
+
2518
+ `get` returns the stored row and its fields echo what you sent. `checksum`
2519
+ recomputes the fingerprint from a body **without persisting it** — so
2520
+ comparing the two is how you detect that a reconciler on the server has
2521
+ drifted from the one in your manifest, before it silently reconciles
2522
+ against the wrong window.
2523
+
2524
+ The catalog reads are worth knowing about for a different reason: they are
2525
+ what an agent reads to find out which reconcilers exist at all.
2526
+
2527
+ ```typescript
2528
+ const { data: storedUpdater } = await api.actionStatusUpdaters.get(
2529
+ tenantSlug, datalakeSlug, actionStatusUpdaterId,
2530
+ )
2531
+ if (storedUpdater.cron_expression !== '*/30 * * * *') {
2532
+ throw new Error(`cron mismatch on read-back: ${storedUpdater.cron_expression}`)
2533
+ }
2534
+
2535
+ const { data: updaterList } = await api.actionStatusUpdaters.list(tenantSlug, datalakeSlug)
2536
+ if (!(updaterList.data ?? []).some((u) => u.id === actionStatusUpdaterId)) {
2537
+ throw new Error('the created reconciler is not in the list')
2538
+ }
2539
+
2540
+ const { data: updaterCatalog } = await api.actionStatusUpdaters.metadata(tenantSlug, datalakeSlug)
2541
+ if (typeof updaterCatalog !== 'string' || updaterCatalog.length === 0) {
2542
+ throw new Error('expected a non-empty reconciler catalog')
2543
+ }
2544
+
2545
+ const { data: updaterDetail } = await api.actionStatusUpdaters.metadataDetails(
2546
+ tenantSlug, datalakeSlug, actionStatusUpdaterId,
2547
+ )
2548
+ if (typeof updaterDetail !== 'string' || updaterDetail.length === 0) {
2549
+ throw new Error('expected non-empty reconciler detail')
2550
+ }
2551
+ ```
2552
+
2553
+ ## 044 — a poller that cannot advance is refused at create
2554
+
2555
+ Not every delivery log is a log group. Pulling outcomes from a provider's
2556
+ REST events API means paging, and paging is where this goes wrong in a way
2557
+ that looks like it is working.
2558
+
2559
+ **The platform refuses a poller whose request can never move.** Below, the
2560
+ first create renders a path and params that read neither
2561
+ `msg.pagination_context` nor `msg.page` — so page five hundred's request is
2562
+ byte-identical to page one's, `has_next` never goes false, and the run
2563
+ re-applies the same events forever. That is a 422 at create, and the walk
2564
+ catches it deliberately rather than mentioning it.
2565
+
2566
+ Then the shape that is accepted: page one renders the plain collection path
2567
+ with bounded params, and every later page rides the cursor the pagination
2568
+ template captured — dropping the params, because the provider's `next` is a
2569
+ complete URL that already carries them.
2570
+
2571
+ Two more floors worth stating. `events_output_schema` must be an array of
2572
+ **objects that require `external_id`**; a bare `{ type: 'array' }` is
2573
+ refused, because an event that does not name the message it reconciles
2574
+ cannot reconcile anything. And the provider's own id becomes `external_id`
2575
+ in the **events template** — the apply path pops that key to find the row it
2576
+ updates, so the render projects each event rather than passing the
2577
+ provider's body through untouched.
2578
+
2579
+ ```typescript
2580
+ const restTool = await ctx.ensure(
2581
+ 'REST events poller tool',
2582
+ async () => {
2583
+ const { data } = await api.tools.list(tenantSlug, datalakeSlug)
2584
+ return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook REST Events Poller')
2585
+ },
2586
+ async () => {
2587
+ const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
2588
+ name: 'Cookbook REST Events Poller',
2589
+ description: 'REST poller — supplies auth for a provider events API.',
2590
+ intent: 'status_poller',
2591
+ status: 'active',
2592
+ datalake_id: ctx.datalakeId,
2593
+ data_source_id: dataSourceId,
2594
+ body: {
2595
+ tool_body_type: 'rest_api',
2596
+ base_url: 'http://localhost:8080/mailgun/v3',
2597
+ auth_method: 'basic',
2598
+ username: 'api',
2599
+ password: 'key-test',
2600
+ request_type: 'json',
2601
+ response_type: 'json',
2602
+ timeout_ms: 30_000,
2603
+ },
2604
+ })
2605
+ return data
2606
+ },
2607
+ )
2608
+ ctx.restToolId = restTool.id!
2609
+
2610
+ const EVENTS_TEMPLATE =
2611
+ '[{% for item in response.items %}' +
2612
+ '{"external_id": "{{ item.message.headers[\'message-id\'] }}", "event": "{{ item.event }}"}' +
2613
+ '{% unless forloop.last %},{% endunless %}{% endfor %}]'
2614
+ const PAGINATION_TEMPLATE =
2615
+ '{"has_next": {% if response.items.size > 0 %}true{% else %}false{% endif %}, ' +
2616
+ '"next": "{{ response.paging.next }}"}'
2617
+ const EVENTS_OUTPUT_SCHEMA = {
2618
+ type: 'array',
2619
+ items: {
2620
+ type: 'object',
2621
+ required: ['external_id'],
2622
+ properties: { external_id: { type: 'string' } },
2623
+ },
2624
+ }
2625
+ const PAGINATION_OUTPUT_SCHEMA = {
2626
+ type: 'object',
2627
+ required: ['has_next'],
2628
+ properties: { has_next: { type: 'boolean' } },
2629
+ }
2630
+
2631
+ // The guard. A static request is refused — walked, not asserted from docs.
2632
+ let refused = false
2633
+ try {
2634
+ await api.actionStatusUpdaters.create(tenantSlug, datalakeSlug, {
2635
+ name: 'Cookbook Non-Advancing Poller',
2636
+ cron_expression: '*/30 * * * *',
2637
+ updater_type: 'restapi',
2638
+ updater_tool_id: ctx.restToolId,
2639
+ sender_tool_ids: [ctx.smsToolId],
2640
+ datalake_id: ctx.datalakeId,
2641
+ events_output_schema: EVENTS_OUTPUT_SCHEMA,
2642
+ pagination_context_output_schema: PAGINATION_OUTPUT_SCHEMA,
2643
+ updater_body: {
2644
+ updater_body_type: 'restapi_request',
2645
+ method: 'get',
2646
+ // Static on both — the cursor is captured below and never read back.
2647
+ path: { type: 'custom', body: '/wiremock.domain/events' },
2648
+ params: { type: 'custom', body: '{"event": "delivered"}' },
2649
+ events_template: { type: 'custom', body: EVENTS_TEMPLATE },
2650
+ pagination_context_template: { type: 'custom', body: PAGINATION_TEMPLATE },
2651
+ },
2652
+ message_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
2653
+ action_log_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
2654
+ })
2655
+ } catch (err) {
2656
+ const status = (err as { _httpStatus?: number })._httpStatus
2657
+ if (status !== 422) throw err
2658
+ refused = true
2659
+ }
2660
+ if (!refused) {
2661
+ throw new Error('expected a 422 for a poller whose request never consumes the cursor')
2662
+ }
2663
+
2664
+ // The accepted shape: the path follows the cursor.
2665
+ const advancingPoller = await ctx.ensure(
2666
+ 'advancing REST delivery poller',
2667
+ async () => {
2668
+ const { data } = await api.actionStatusUpdaters.list(tenantSlug, datalakeSlug)
2669
+ return (data.data ?? []).find((u: { name?: string }) => u.name === 'Cookbook Advancing Delivery Poller')
2670
+ },
2671
+ async () => {
2672
+ const { data } = await api.actionStatusUpdaters.create(tenantSlug, datalakeSlug, {
2673
+ name: 'Cookbook Advancing Delivery Poller',
2674
+ cron_expression: '*/30 * * * *',
2675
+ updater_type: 'restapi',
2676
+ updater_tool_id: ctx.restToolId,
2677
+ sender_tool_ids: [ctx.smsToolId],
2678
+ datalake_id: ctx.datalakeId,
2679
+ events_output_schema: EVENTS_OUTPUT_SCHEMA,
2680
+ pagination_context_output_schema: PAGINATION_OUTPUT_SCHEMA,
2681
+ updater_body: {
2682
+ updater_body_type: 'restapi_request',
2683
+ method: 'get',
2684
+ path: {
2685
+ type: 'custom',
2686
+ body:
2687
+ '{% if msg.pagination_context %}{{ msg.pagination_context.next }}' +
2688
+ '{% else %}/wiremock.domain/events{% endif %}',
2689
+ },
2690
+ params: {
2691
+ type: 'custom',
2692
+ body: '{% unless msg.pagination_context %}{"event": "delivered"}{% endunless %}',
2693
+ },
2694
+ events_template: { type: 'custom', body: EVENTS_TEMPLATE },
2695
+ pagination_context_template: { type: 'custom', body: PAGINATION_TEMPLATE },
2696
+ },
2697
+ message_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
2698
+ action_log_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
2699
+ })
2700
+ return data
2701
+ },
2702
+ )
2703
+ if (advancingPoller.status !== 'active') {
2704
+ throw new Error(`a newly created poller must be free to poll — got status ${advancingPoller.status}`)
2705
+ }
2706
+ ```
2707
+
2708
+ ## 045 — ask the lake a question in plain language
2709
+
2710
+ The last thing a marketing team needs is a way to look at any of this
2711
+ without writing SQL.
2712
+
2713
+ `textToSql` runs ordered multi-provider failover and hands back the
2714
+ generated `sql` along with which model and provider produced it, and a
2715
+ best-effort `explanation` (`null` when the explainer is unavailable).
2716
+
2717
+ **It does not execute anything.** It returns SQL for a human to read, which
2718
+ is the whole design: only the prompt and the datalake's *schema* cross the
2719
+ model boundary. No rows leave the trust boundary to answer a question about
2720
+ them.
2721
+
2722
+ ```typescript
2723
+ const { data: generated } = await api.datalakes.textToSql(tenantSlug, datalakeSlug, {
2724
+ prompt: 'how many end customers are suppressed?',
2725
+ mode: 'raw',
2726
+ })
2727
+ if (typeof generated.sql !== 'string' || generated.sql.trim() === '') {
2728
+ throw new Error(`textToSql returned no SQL (got: ${JSON.stringify(generated.sql)})`)
2729
+ }
2730
+ if (typeof generated.model !== 'string' || typeof generated.provider !== 'string') {
2731
+ throw new Error('textToSql response is missing its model/provider attribution')
2732
+ }
2733
+ console.log(` ↳ ${generated.provider}/${generated.model} proposed: ${generated.sql.trim().slice(0, 120)}`)
2734
+ ```
2735
+
2736
+ ## 046 — run it read-only, then take the same page as CSV
2737
+
2738
+ `executeSql` runs read-only — inserts, updates, deletes and DDL are all
2739
+ refused — and returns `{ data, meta }`. **`data` is an array of arrays**,
2740
+ positional and aligned to `meta.columns`, because arbitrary SQL can produce
2741
+ duplicate or expression column names that object keys would collapse.
2742
+
2743
+ The deterministic count below keeps the step independent of whatever the
2744
+ model proposed a moment ago, which is what you want in a walk: §045 proves
2745
+ the generation, this proves the execution, and neither can mask a failure in
2746
+ the other.
2747
+
2748
+ Passing `{ format: 'csv' }` returns the same page as a string instead of the
2749
+ JSON envelope — the same read, content-negotiated for download. The curated
2750
+ return type is a union, so narrow it before reading either shape.
2751
+
2752
+ ```typescript
2753
+ const suppressedCount = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
2754
+ sql: `SELECT count(*) AS n FROM ${ctx.audienceTableName} WHERE suppressed = true`,
2755
+ mode: 'raw',
2756
+ })
2757
+ if (typeof suppressedCount.data === 'string') {
2758
+ throw new Error('expected the JSON envelope, got a CSV string')
2759
+ }
2760
+ const countPage = suppressedCount.data
2761
+ if (!Array.isArray(countPage.data) || !Array.isArray(countPage.data[0])) {
2762
+ throw new Error('executeSql data is not an array-of-arrays')
2763
+ }
2764
+ if (countPage.meta.columns[0] !== 'n') {
2765
+ throw new Error(`unexpected column list: ${JSON.stringify(countPage.meta.columns)}`)
2766
+ }
2767
+ if (Number(countPage.data[0][0]) < 1) {
2768
+ throw new Error('at least one end customer is suppressed — §035 ingested one')
2769
+ }
2770
+
2771
+ const csvExport = await api.datalakes.executeSql(
2772
+ tenantSlug,
2773
+ datalakeSlug,
2774
+ {
2775
+ sql: `SELECT end_customer_id, bucket, suppressed FROM ${ctx.audienceTableName} ORDER BY end_customer_id`,
2776
+ mode: 'raw',
2777
+ },
2778
+ { format: 'csv' },
2779
+ )
2780
+ if (typeof csvExport.data !== 'string') {
2781
+ throw new Error('expected a CSV string for { format: "csv" }')
2782
+ }
2783
+ if (!csvExport.data.includes('end_customer_id') || !csvExport.data.includes(ctx.adaId)) {
2784
+ throw new Error(`unexpected CSV body: ${JSON.stringify(csvExport.data.slice(0, 200))}`)
2785
+ }
2786
+ ```
2787
+
2788
+ # Outcome
2789
+
2790
+ The marketing tenant exists with one raw datalake, an SMS tool, an LLM
2791
+ tool, a lead-submissions table, and an agent-driven workflow that bands
2792
+ every inbound lead and fans out to exactly one of four SMS actions.
2793
+
2794
+ Running this walk again reuses all of it.
2795
+
2796
+ # See also
2797
+
2798
+ - `datalakes.md` — the create body, and what a derived pair would add
2799
+ - `workflows.md` — scheduling versus firing, and the accessor tables
2800
+ - `ai_agents.md` — response schemas, and why the enum is the real guard
2801
+ - `generic_tables.md` — the default client you never have to create