@alvera-ai/platform-sdk 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/AGENTS.md +82 -144
- package/.agent/account_management.md +2 -2
- package/.agent/action_logs.md +4 -4
- package/.agent/ai_agents.md +28 -21
- package/.agent/ai_sandbox.md +49 -39
- package/.agent/connected_apps.md +3 -3
- package/.agent/cookbook/_fixtures/README.md +1 -1
- package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_generic_table.liquid +1 -1
- package/.agent/cookbook/_fixtures/organic-marketing/_lead_submissions_foundation_legal_entity.liquid +80 -0
- package/.agent/cookbook/_fixtures/{foundation → organic-marketing}/_lead_submissions_foundation_mdm.liquid +2 -1
- package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_generic_table.liquid +57 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_legal_entity.liquid +30 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_compliance_screenings_mdm.liquid +44 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_generic_table.liquid +57 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_legal_entity.liquid +36 -0
- package/.agent/cookbook/_fixtures/payments-compliance/_payment_accounts_mdm.liquid +41 -0
- package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_generic_table.liquid +70 -0
- package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_legal_entity.liquid +52 -0
- package/.agent/cookbook/_fixtures/primary-care-feedback/_cahps_appointments_mdm.liquid +42 -0
- package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_generic_table.liquid +38 -0
- package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_legal_entity.liquid +48 -0
- package/.agent/cookbook/_fixtures/subscription-saas/_customers_subscription_mdm.liquid +49 -0
- package/.agent/cookbook/organic-marketing.md +2801 -0
- package/.agent/cookbook/payments-compliance.md +2180 -0
- package/.agent/cookbook/primary-care.md +2175 -0
- package/.agent/cookbook/subscription-saas.md +2403 -0
- package/.agent/data_activation_clients.md +65 -52
- package/.agent/datalakes.md +338 -171
- package/.agent/errors.md +3 -3
- package/.agent/generic_tables.md +151 -62
- package/.agent/interoperability_contracts.md +57 -22
- package/.agent/mdm.md +136 -153
- package/.agent/messages.md +36 -34
- package/.agent/mock-services.md +1 -1
- package/.agent/mutations.md +2 -2
- package/.agent/templates.md +14 -13
- package/.agent/tools.md +63 -21
- package/.agent/type_naming.md +13 -13
- package/.agent/workflows.md +99 -53
- package/README.md +2 -2
- package/dist/bin/platform-sdk.mjs +33 -47
- package/dist/bin/platform-sdk.mjs.map +1 -1
- package/dist/index.d.mts +565 -379
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +494 -59
- package/dist/index.mjs.map +1 -1
- package/package.json +4 -3
- package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +0 -88
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +0 -47
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +0 -24
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +0 -38
- package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_compliance_screening.liquid +0 -59
- package/.agent/cookbook/_fixtures/payments/_compliance_screenings_payments_mdm.liquid +0 -36
- package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_mdm.liquid +0 -30
- package/.agent/cookbook/_fixtures/payments/_payment_accounts_payments_payment_account.liquid +0 -55
- package/.agent/cookbook/_fixtures/subscription/_customers_subscription_mdm.liquid +0 -20
- package/.agent/cookbook/_setup/foundation.md +0 -359
- package/.agent/cookbook/_setup/healthcare.md +0 -361
- package/.agent/cookbook/_setup/payments.md +0 -365
- package/.agent/cookbook/_setup/subscription.md +0 -364
- package/.agent/cookbook/action-status-updaters.md +0 -278
- package/.agent/cookbook/ai-agent-invoke.md +0 -279
- package/.agent/cookbook/appointment-review-sms-workflow.md +0 -801
- package/.agent/cookbook/birthday-greeting-sms-trigger.md +0 -696
- package/.agent/cookbook/bulk-ingest.md +0 -302
- package/.agent/cookbook/contact-us-triage-with-llm.md +0 -663
- package/.agent/cookbook/dunning-sms-for-delinquent.md +0 -659
- package/.agent/cookbook/generic-tables.md +0 -244
- package/.agent/cookbook/invite-team.md +0 -200
- package/.agent/cookbook/kyc-notification-on-account-activation.md +0 -661
- package/.agent/cookbook/marketing-campaign-send.md +0 -1044
- package/.agent/cookbook/paginated-restapi-poller.md +0 -383
- package/.agent/cookbook/rest-fetch.md +0 -273
- package/.agent/cookbook/sanctions-screening-with-agent-review.md +0 -773
- package/.agent/cookbook/score-leads-with-llm-categorization.md +0 -665
- package/.agent/cookbook/system-templates.md +0 -165
- package/.agent/cookbook/talk-to-data.md +0 -178
- package/.agent/cookbook/triage-prospects-by-priority.md +0 -571
- package/.agent/cookbook/welcome-sms-for-customers.md +0 -647
- /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/memorandum-of-association-01.png +0 -0
- /package/.agent/cookbook/_fixtures/{healthcare → primary-care-feedback}/sample_two_page.pdf +0 -0
- /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/_customers_subscription_customer.liquid +0 -0
- /package/.agent/cookbook/_fixtures/{subscription → subscription-saas}/stripe_customers_batch1.csv +0 -0
|
@@ -0,0 +1,2801 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: "Organic marketing: score, greet, campaign — on one raw lake"
|
|
3
|
+
summary: The whole organic-marketing surface in one walk. Stand up a tenant and a raw datalake, then run the scenarios end to end — an LLM agent banding inbound leads into a four-way SMS fan-out, a pure-Liquid birthday trigger that schedules a greeting a year out and closes the reply loop through a connected app, and an A/B campaign that gates on suppression, splits inside the workflow and dispatches on SMS and email in one run. Ends with delivery reconciliation and a natural-language read of the lake.
|
|
4
|
+
use_case: organic-marketing
|
|
5
|
+
slug: organic-marketing
|
|
6
|
+
vitest_source:
|
|
7
|
+
- integration-tests/tests/organic-marketing/agent-driven-workflow.test.ts
|
|
8
|
+
- integration-tests/tests/organic-marketing/generic-tables.test.ts
|
|
9
|
+
- integration-tests/tests/organic-marketing/manual-upload-tool.test.ts
|
|
10
|
+
- integration-tests/tests/workspace/bootstrap.test.ts
|
|
11
|
+
status: draft
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
# Problem
|
|
15
|
+
|
|
16
|
+
A marketing team's work is three jobs that share one audience. Leads
|
|
17
|
+
arrive from forms and most of them are noise, so something has to read each
|
|
18
|
+
one and decide whether a human should. Contacts have birthdays, and a
|
|
19
|
+
greeting has to be scheduled once and fire a year later without anyone
|
|
20
|
+
remembering. And campaigns go out to a roster of businesses, split A/B, on
|
|
21
|
+
whichever channel the recipient can actually be reached on.
|
|
22
|
+
|
|
23
|
+
All three are the same shape — rows land, a filter decides which ones
|
|
24
|
+
matter, and something is sent about the ones that do. This walk builds all
|
|
25
|
+
three on one tenant, because a marketing team does not run three platforms.
|
|
26
|
+
|
|
27
|
+
**This surface stays raw.** No tokenized or redacted copy is provisioned,
|
|
28
|
+
and that is a decision rather than an omission: everything read back here
|
|
29
|
+
is a question about this walk's own rows, and the lead's own message — not
|
|
30
|
+
the lead's name — is what the classifier reads. If you want a model held
|
|
31
|
+
away from identities, that is what `payments-compliance` §007 shows, and it
|
|
32
|
+
is one call away from here.
|
|
33
|
+
|
|
34
|
+
# Composition
|
|
35
|
+
|
|
36
|
+
| Resource | Why it exists here |
|
|
37
|
+
|-------------------------------------|---------------------------------------------|
|
|
38
|
+
| Tenant + raw datalake | The marketing team's own world |
|
|
39
|
+
| SMS tool, email tool, LLM tool | Shared by every scenario below |
|
|
40
|
+
| Score Lead AI agent | Bands inbound leads |
|
|
41
|
+
| Three generic tables | Leads, the roster, and the audience |
|
|
42
|
+
| Three workflows | Band fan-out, birthday greeting, campaign |
|
|
43
|
+
| Connected app | The reply loop, twice |
|
|
44
|
+
| Action status updater | Delivery outcomes, reconciled on a cron |
|
|
45
|
+
|
|
46
|
+
# Walkthrough
|
|
47
|
+
|
|
48
|
+
This cookbook is **self-reliant**: it stands up everything it uses and
|
|
49
|
+
depends on no other file. It is also **idempotent** — every creating step
|
|
50
|
+
looks first and creates only what is missing, so running it twice costs
|
|
51
|
+
what running it once cost. That is not a nicety. Nothing in the platform
|
|
52
|
+
reclaims an abandoned datalake, and a walk that mints a fresh tenant per
|
|
53
|
+
run leaves every previous run's lakes behind forever.
|
|
54
|
+
|
|
55
|
+
## 001 — sign in as root, and make sure the admin user exists
|
|
56
|
+
|
|
57
|
+
Authenticate as the platform's root admin (`admin@dev.local` /
|
|
58
|
+
`devpassword` in local dev) via the tenantless bootstrap login — keyless by
|
|
59
|
+
structural necessity, since no tenant exists yet to scope a key to, and a
|
|
60
|
+
dev/test-only surface.
|
|
61
|
+
|
|
62
|
+
Then make sure the admin user this walk runs as exists. **The email is
|
|
63
|
+
stable, not per-run**, which is what makes the step idempotent — and it
|
|
64
|
+
means the second run finds the user already signed up. A duplicate signup
|
|
65
|
+
is refused, so the refusal is caught and inspected: if it says the email is
|
|
66
|
+
taken, that is the idempotent path and the walk continues. Any other
|
|
67
|
+
failure is re-raised, because swallowing it would turn a real auth problem
|
|
68
|
+
into a confusing failure three steps later.
|
|
69
|
+
|
|
70
|
+
```typescript
|
|
71
|
+
ctx.rootSession = await createBootstrapSession({
|
|
72
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
73
|
+
email: process.env.ALVERA_ROOT_EMAIL!,
|
|
74
|
+
password: process.env.ALVERA_ROOT_PASSWORD!,
|
|
75
|
+
})
|
|
76
|
+
ctx.rootApi = createIsolatedPlatformApi({
|
|
77
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
78
|
+
sessionToken: ctx.rootSession.sessionToken,
|
|
79
|
+
apiKey: '',
|
|
80
|
+
})
|
|
81
|
+
|
|
82
|
+
ctx.marketerEmail = 'cookbook-organic-marketing@dev.local'
|
|
83
|
+
ctx.marketerPassword = 'CookbookPass1!'
|
|
84
|
+
|
|
85
|
+
try {
|
|
86
|
+
const signUpResp = await ctx.rootApi.admin.signUp({
|
|
87
|
+
email: ctx.marketerEmail,
|
|
88
|
+
password: ctx.marketerPassword,
|
|
89
|
+
first_name: 'Cookbook',
|
|
90
|
+
last_name: 'Marketing',
|
|
91
|
+
})
|
|
92
|
+
await ctx.rootApi.admin.confirmUser(signUpResp.data.id!)
|
|
93
|
+
} catch (err) {
|
|
94
|
+
// Already provisioned by a previous run. Confirm that is what happened
|
|
95
|
+
// rather than assuming it — a genuine signup failure must not be read as
|
|
96
|
+
// "already there".
|
|
97
|
+
const detail = JSON.stringify((err as { errors?: unknown }).errors ?? err)
|
|
98
|
+
if (!/taken|already|exist/i.test(detail)) throw err
|
|
99
|
+
}
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## 002 — the find-or-create helper every later step uses
|
|
103
|
+
|
|
104
|
+
Idempotence is one question asked over and over: *is this already here?*
|
|
105
|
+
Rather than answer it a dozen different ways, the walk defines it once.
|
|
106
|
+
|
|
107
|
+
`ensure` takes a label, a lookup and a create. It runs the lookup, returns
|
|
108
|
+
what it finds, and only creates when the lookup comes back empty. The label
|
|
109
|
+
is not decoration — when a run reuses something you did not expect it to,
|
|
110
|
+
the log line naming it is how you find out.
|
|
111
|
+
|
|
112
|
+
The helper lives on `ctx` rather than as a bare function because each
|
|
113
|
+
numbered step compiles into its own `it()` block, so a plain `function`
|
|
114
|
+
here would not be in scope for the steps that call it.
|
|
115
|
+
|
|
116
|
+
```typescript
|
|
117
|
+
ctx.ensure = async <T>(
|
|
118
|
+
label: string,
|
|
119
|
+
find: () => Promise<T | undefined>,
|
|
120
|
+
create: () => Promise<T>,
|
|
121
|
+
): Promise<T> => {
|
|
122
|
+
const existing = await find()
|
|
123
|
+
if (existing !== undefined) {
|
|
124
|
+
console.log(` ↻ reusing ${label}`)
|
|
125
|
+
return existing
|
|
126
|
+
}
|
|
127
|
+
console.log(` + creating ${label}`)
|
|
128
|
+
return await create()
|
|
129
|
+
}
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## 003 — the tenant
|
|
133
|
+
|
|
134
|
+
The marketer signs in without a tenant scope — they may not belong to one
|
|
135
|
+
yet — and the walk finds or creates the marketing tenant by its stable
|
|
136
|
+
name. The server derives the slug; capture it, because every later call is
|
|
137
|
+
addressed by it.
|
|
138
|
+
|
|
139
|
+
```typescript
|
|
140
|
+
ctx.marketerTenantlessSession = await createBootstrapSession({
|
|
141
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
142
|
+
email: ctx.marketerEmail,
|
|
143
|
+
password: ctx.marketerPassword,
|
|
144
|
+
})
|
|
145
|
+
ctx.marketerTenantlessApi = createIsolatedPlatformApi({
|
|
146
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
147
|
+
sessionToken: ctx.marketerTenantlessSession.sessionToken,
|
|
148
|
+
apiKey: '',
|
|
149
|
+
})
|
|
150
|
+
|
|
151
|
+
const TENANT_NAME = 'Cookbook Organic Marketing'
|
|
152
|
+
|
|
153
|
+
const tenant = await ctx.ensure(
|
|
154
|
+
`tenant ${TENANT_NAME}`,
|
|
155
|
+
async () => {
|
|
156
|
+
const { data } = await ctx.marketerTenantlessApi.tenants.list()
|
|
157
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === TENANT_NAME)
|
|
158
|
+
},
|
|
159
|
+
async () => {
|
|
160
|
+
const { data } = await ctx.marketerTenantlessApi.tenants.create({ name: TENANT_NAME })
|
|
161
|
+
return data
|
|
162
|
+
},
|
|
163
|
+
)
|
|
164
|
+
tenantSlug = tenant.slug!
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
## 004 — the tenant-scoped client
|
|
168
|
+
|
|
169
|
+
A tenant-scoped login requires `X-API-Key`, so a key has to exist before
|
|
170
|
+
the marketer can sign in against the tenant. Mint one through the
|
|
171
|
+
platform-admin side door with the root bearer; in the web console this is
|
|
172
|
+
*Settings → API Keys*.
|
|
173
|
+
|
|
174
|
+
`data_access_mode: 'raw'` is the ceiling, and on this surface it is the
|
|
175
|
+
only tier there is — no derived lake is provisioned, so raw is what every
|
|
176
|
+
read answers from.
|
|
177
|
+
|
|
178
|
+
**This is the one step in the walk that is not idempotent**, and it is
|
|
179
|
+
worth knowing why rather than discovering it. There is no endpoint that
|
|
180
|
+
lists a tenant's API keys, so there is nothing to look the existing one up
|
|
181
|
+
with — `ensure` has no lookup to run. Each run therefore mints another key.
|
|
182
|
+
A key row is cheap where a datalake is not, so the walk accepts it; if you
|
|
183
|
+
are counting rows in a shared environment, this is the one to count.
|
|
184
|
+
|
|
185
|
+
```typescript
|
|
186
|
+
const { data: mintedKey } = await ctx.rootApi.admin.createTenantApiKey(tenantSlug, {
|
|
187
|
+
name: 'Cookbook Organic Marketing Key',
|
|
188
|
+
data_access_mode: 'raw',
|
|
189
|
+
})
|
|
190
|
+
ctx.tenantApiKey = mintedKey.api_key
|
|
191
|
+
|
|
192
|
+
const marketerTenantSession = await createSession({
|
|
193
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
194
|
+
email: ctx.marketerEmail,
|
|
195
|
+
password: ctx.marketerPassword,
|
|
196
|
+
tenantSlug,
|
|
197
|
+
apiKey: ctx.tenantApiKey,
|
|
198
|
+
})
|
|
199
|
+
api = createIsolatedPlatformApi({
|
|
200
|
+
baseUrl: process.env.ALVERA_BASE_URL!,
|
|
201
|
+
sessionToken: marketerTenantSession.sessionToken,
|
|
202
|
+
apiKey: ctx.tenantApiKey,
|
|
203
|
+
})
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## 005 — the datalake
|
|
207
|
+
|
|
208
|
+
One datalake, created `raw`, and on this surface that is the whole story —
|
|
209
|
+
there is no derived pair to provision and nothing to wait for beyond the
|
|
210
|
+
migration in §006.
|
|
211
|
+
|
|
212
|
+
The lookup still filters on `type === 'raw'`. That costs nothing here and
|
|
213
|
+
keeps the step correct if anyone ever turns tokenization on: from that
|
|
214
|
+
moment `datalakes.list` returns three rows, two of which are copies that
|
|
215
|
+
must never be mistaken for the primary.
|
|
216
|
+
|
|
217
|
+
Local-dev defaults match the seeded `dev.exs` setup — `postgres` on
|
|
218
|
+
`localhost:5432`, database `alvera_dev_foundation`, LocalStack S3 on
|
|
219
|
+
`localhost:4566` — and the schema name is stable, because a fresh schema
|
|
220
|
+
per run is the same leak as a fresh tenant per run.
|
|
221
|
+
|
|
222
|
+
```typescript
|
|
223
|
+
const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
|
|
224
|
+
const DB_SCHEMA = 'cookbook_organic_marketing'
|
|
225
|
+
const LAKE_NAME = 'Cookbook Organic Marketing Datalake'
|
|
226
|
+
|
|
227
|
+
const S3 = {
|
|
228
|
+
cloud_storage_type: 'aws' as const,
|
|
229
|
+
region: 'us-east-1',
|
|
230
|
+
access_key_id: 'test',
|
|
231
|
+
secret_access_key: 'test',
|
|
232
|
+
endpoint: 'http://localhost:4566',
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
const datalake = await ctx.ensure(
|
|
236
|
+
`datalake ${LAKE_NAME}`,
|
|
237
|
+
async () => {
|
|
238
|
+
const { data } = await api.datalakes.list(tenantSlug)
|
|
239
|
+
return (data.data ?? []).find(
|
|
240
|
+
(l: { name?: string; type?: string }) => l.name === LAKE_NAME && l.type === 'raw',
|
|
241
|
+
)
|
|
242
|
+
},
|
|
243
|
+
async () => {
|
|
244
|
+
const { data } = await api.datalakes.create(tenantSlug, {
|
|
245
|
+
name: LAKE_NAME,
|
|
246
|
+
description: 'Organic marketing datalake provisioned by the cookbook doctest.',
|
|
247
|
+
timezone: 'America/New_York',
|
|
248
|
+
pool_size: 3,
|
|
249
|
+
type: 'raw',
|
|
250
|
+
|
|
251
|
+
db_writer_host: DB.host,
|
|
252
|
+
db_writer_port: DB.port,
|
|
253
|
+
db_writer_name: DB.name,
|
|
254
|
+
db_writer_schema: DB_SCHEMA,
|
|
255
|
+
db_writer_auth_method: 'password',
|
|
256
|
+
db_writer_user: DB.user,
|
|
257
|
+
db_writer_pass: DB.pass,
|
|
258
|
+
db_writer_enable_ssl: false,
|
|
259
|
+
db_reader_host: DB.host,
|
|
260
|
+
db_reader_port: DB.port,
|
|
261
|
+
db_reader_name: DB.name,
|
|
262
|
+
db_reader_schema: DB_SCHEMA,
|
|
263
|
+
db_reader_auth_method: 'password',
|
|
264
|
+
db_reader_user: DB.user,
|
|
265
|
+
db_reader_pass: DB.pass,
|
|
266
|
+
db_reader_enable_ssl: false,
|
|
267
|
+
|
|
268
|
+
cloud_storage: { ...S3, bucket: 'alvera-platform-dev', base_path: 'cookbook/organic-marketing' },
|
|
269
|
+
})
|
|
270
|
+
return data
|
|
271
|
+
},
|
|
272
|
+
)
|
|
273
|
+
datalakeSlug = datalake.slug!
|
|
274
|
+
ctx.datalakeId = datalake.id!
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
## 006 — run the migrations, and wait for ready
|
|
278
|
+
|
|
279
|
+
`datalakes.create` persists the row at `status: 'new'`; it does not run the
|
|
280
|
+
schema DDL. Migration is triggered separately so the operator decides when
|
|
281
|
+
the potentially-slow part happens. `datalakes.migrate` enqueues the job and
|
|
282
|
+
returns immediately with `status: 'enqueued'`; the poll after it is what
|
|
283
|
+
waits for the worker.
|
|
284
|
+
|
|
285
|
+
Migrating is safe to repeat, which is what lets this step stay unguarded on
|
|
286
|
+
a second run. The wait exits the moment the lake reports `ready`, so the
|
|
287
|
+
five-minute ceiling is only ever paid in failure.
|
|
288
|
+
|
|
289
|
+
```typescript
|
|
290
|
+
const migrateResp = await api.datalakes.migrate(tenantSlug, datalakeSlug)
|
|
291
|
+
if (migrateResp.data.status !== 'enqueued') {
|
|
292
|
+
throw new Error(`datalake migration not enqueued (status: ${migrateResp.data.status})`)
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
const READY_TIMEOUT_MS = 5 * 60_000
|
|
296
|
+
const readyDeadline = Date.now() + READY_TIMEOUT_MS
|
|
297
|
+
let datalakeStatus: string | undefined
|
|
298
|
+
while (Date.now() < readyDeadline) {
|
|
299
|
+
const { data } = await api.datalakes.get(tenantSlug, ctx.datalakeId)
|
|
300
|
+
datalakeStatus = data.status
|
|
301
|
+
if (datalakeStatus === 'ready') break
|
|
302
|
+
await new Promise((r) => setTimeout(r, 5_000))
|
|
303
|
+
}
|
|
304
|
+
if (datalakeStatus !== 'ready') {
|
|
305
|
+
throw new Error(`datalake did not reach :ready within ${READY_TIMEOUT_MS}ms (last: ${datalakeStatus})`)
|
|
306
|
+
}
|
|
307
|
+
```
|
|
308
|
+
|
|
309
|
+
## 007 — three helpers the scenarios below share
|
|
310
|
+
|
|
311
|
+
**`waitForFiredRun`.** `workflows.run` only *schedules* a run. It returns
|
|
312
|
+
immediately with a `workflow_run_id`, and the `workflow_run_log_id` and
|
|
313
|
+
`batch_id` a scenario needs are written later, when the run actually fires.
|
|
314
|
+
Two traps live in that gap:
|
|
315
|
+
|
|
316
|
+
- **Poll until `workflow_run_log_id` is a string — not until `status`
|
|
317
|
+
leaves `'scheduled'`.** Those are different moments; the run reaches
|
|
318
|
+
`processing` first and writes the log id a beat later. A predicate on
|
|
319
|
+
status alone releases you to read a `null`, and because
|
|
320
|
+
`typeof null === 'object'` the symptom is a baffling *"expected string,
|
|
321
|
+
got object"* rather than an obvious nil.
|
|
322
|
+
- **Raise on `failed` carrying `failure_reason`** rather than polling to
|
|
323
|
+
the deadline. A scenario blocked on a run that will never fire should say
|
|
324
|
+
why on the first read, not thirty seconds later behind a generic timeout.
|
|
325
|
+
|
|
326
|
+
**`waitForBatches`.** Ingestion is async: `ingest` returns a `batch_id` the
|
|
327
|
+
instant the rows are accepted, and the per-row jobs drain afterwards. This
|
|
328
|
+
polls a client's activation logs until every named batch has actually
|
|
329
|
+
written.
|
|
330
|
+
|
|
331
|
+
**Gate on `dataset_updated`, never on `rows_ingested`.** They answer
|
|
332
|
+
different questions — received versus written — and a batch whose row is
|
|
333
|
+
refused reports `rows_ingested: 1, dataset_updated: 0, status: 'partial'`
|
|
334
|
+
with the reason in `error`. A gate on `rows_ingested` calls that green, the
|
|
335
|
+
next step then runs against a table with nothing in it, and the failure
|
|
336
|
+
surfaces three steps later as *"expected 3 execution logs, got 0"* — which
|
|
337
|
+
reads like a broken workflow rather than a row that never landed.
|
|
338
|
+
|
|
339
|
+
**`deployGenericTable`.** Creating a generic table does not deploy it; it
|
|
340
|
+
rests at `status: 'new'` until a migration runs. The wait that means
|
|
341
|
+
anything asks for the table the same way the next step is about to, and
|
|
342
|
+
retries until it stops erroring — polling the table's own
|
|
343
|
+
`status: 'deployed'` goes true earlier and proves less.
|
|
344
|
+
|
|
345
|
+
```typescript
|
|
346
|
+
ctx.waitForFiredRun = async (
|
|
347
|
+
runDatalakeSlug: string,
|
|
348
|
+
runId: string,
|
|
349
|
+
timeoutMs = 120_000,
|
|
350
|
+
): Promise<{ workflowRunLogId: string; batchId: string | null }> => {
|
|
351
|
+
const deadline = Date.now() + timeoutMs
|
|
352
|
+
let lastStatus: string | undefined
|
|
353
|
+
while (Date.now() < deadline) {
|
|
354
|
+
const { data } = await api.workflowRuns.get(tenantSlug, runDatalakeSlug, runId)
|
|
355
|
+
lastStatus = data.status
|
|
356
|
+
if (data.status === 'failed') {
|
|
357
|
+
throw new Error(`workflow run ${runId} failed: ${data.failure_reason ?? 'no failure_reason given'}`)
|
|
358
|
+
}
|
|
359
|
+
if (typeof data.workflow_run_log_id === 'string') {
|
|
360
|
+
return { workflowRunLogId: data.workflow_run_log_id, batchId: data.batch_id ?? null }
|
|
361
|
+
}
|
|
362
|
+
await new Promise((r) => setTimeout(r, 1_000))
|
|
363
|
+
}
|
|
364
|
+
throw new Error(`workflow run ${runId} never fired within ${timeoutMs}ms (last status: ${lastStatus})`)
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
ctx.waitForBatches = async (
|
|
368
|
+
dacSlug: string,
|
|
369
|
+
batchIds: readonly string[],
|
|
370
|
+
timeoutMs = 90_000,
|
|
371
|
+
): Promise<void> => {
|
|
372
|
+
const targets = new Set(batchIds)
|
|
373
|
+
const deadline = Date.now() + timeoutMs
|
|
374
|
+
let greenCount = 0
|
|
375
|
+
while (Date.now() < deadline) {
|
|
376
|
+
const { data } = await api.dataActivationClients.logs.list(tenantSlug, datalakeSlug, dacSlug)
|
|
377
|
+
const green = new Set<string>()
|
|
378
|
+
for (const row of (data.data ?? []) as Array<Record<string, unknown>>) {
|
|
379
|
+
const b = row.batch_id
|
|
380
|
+
if (typeof b !== 'string' || !targets.has(b)) continue
|
|
381
|
+
if (row.status === 'partial' || row.status === 'failed') {
|
|
382
|
+
throw new Error(
|
|
383
|
+
`batch ${b} on ${dacSlug} did not persist its rows (status: ${row.status}): ` +
|
|
384
|
+
`${String(row.error ?? 'no reason given')}`,
|
|
385
|
+
)
|
|
386
|
+
}
|
|
387
|
+
if (typeof row.dataset_updated !== 'number' || row.dataset_updated < 1) continue
|
|
388
|
+
const files = row.output_files
|
|
389
|
+
if (!Array.isArray(files) || files.length === 0) continue
|
|
390
|
+
green.add(b)
|
|
391
|
+
}
|
|
392
|
+
greenCount = green.size
|
|
393
|
+
if (greenCount === targets.size) return
|
|
394
|
+
await new Promise((r) => setTimeout(r, 1_000))
|
|
395
|
+
}
|
|
396
|
+
throw new Error(`only ${greenCount}/${targets.size} batches on ${dacSlug} persisted within ${timeoutMs}ms`)
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
ctx.deployGenericTable = async (tableName: string, timeoutMs = 120_000): Promise<void> => {
|
|
400
|
+
await api.datalakes.migrate(tenantSlug, datalakeSlug)
|
|
401
|
+
const deadline = Date.now() + timeoutMs
|
|
402
|
+
let lastError: unknown
|
|
403
|
+
while (Date.now() < deadline) {
|
|
404
|
+
try {
|
|
405
|
+
await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
|
|
406
|
+
sql: `SELECT 1 FROM ${tableName} LIMIT 1`,
|
|
407
|
+
mode: 'raw',
|
|
408
|
+
})
|
|
409
|
+
return
|
|
410
|
+
} catch (err) {
|
|
411
|
+
lastError = err
|
|
412
|
+
await new Promise((r) => setTimeout(r, 1_000))
|
|
413
|
+
}
|
|
414
|
+
}
|
|
415
|
+
throw new Error(
|
|
416
|
+
`generic table ${tableName} was not readable within ${timeoutMs}ms ` +
|
|
417
|
+
`(last error: ${lastError instanceof Error ? lastError.message : String(lastError)})`,
|
|
418
|
+
)
|
|
419
|
+
}
|
|
420
|
+
```
|
|
421
|
+
|
|
422
|
+
## 008 — the SMS tool
|
|
423
|
+
|
|
424
|
+
One SMS tool dispatches every outbound message in this walk — four lead
|
|
425
|
+
bands, a birthday greeting and a campaign variant. **The per-message body
|
|
426
|
+
lives on the workflow action, never on the tool**, which is what lets one
|
|
427
|
+
tool serve all of them. `intent: 'sms'` tags it for workflow-action use.
|
|
428
|
+
|
|
429
|
+
Local dev points at LocalStack's SNS on `:4566`, so nothing leaves the
|
|
430
|
+
machine.
|
|
431
|
+
|
|
432
|
+
```typescript
|
|
433
|
+
const smsTool = await ctx.ensure(
|
|
434
|
+
'SMS tool',
|
|
435
|
+
async () => {
|
|
436
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
437
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Marketing SMS Tool')
|
|
438
|
+
},
|
|
439
|
+
async () => {
|
|
440
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
441
|
+
name: 'Cookbook Marketing SMS Tool',
|
|
442
|
+
description: 'SNS-backed SMS dispatcher for every marketing scenario, wired to LocalStack.',
|
|
443
|
+
intent: 'sms',
|
|
444
|
+
status: 'active',
|
|
445
|
+
datalake_id: ctx.datalakeId,
|
|
446
|
+
body: {
|
|
447
|
+
tool_body_type: 'sns',
|
|
448
|
+
auth_method: 'access_key',
|
|
449
|
+
region: 'us-east-1',
|
|
450
|
+
phone_number: '+15551234567',
|
|
451
|
+
endpoint_url: 'http://localhost:4566',
|
|
452
|
+
access_key_id: 'test',
|
|
453
|
+
secret_access_key: 'test',
|
|
454
|
+
},
|
|
455
|
+
})
|
|
456
|
+
return data
|
|
457
|
+
},
|
|
458
|
+
)
|
|
459
|
+
toolId = smsTool.id!
|
|
460
|
+
ctx.smsToolId = smsTool.id!
|
|
461
|
+
```
|
|
462
|
+
|
|
463
|
+
## 009 — the LLM tool
|
|
464
|
+
|
|
465
|
+
The Score Lead agent calls a chat-completion endpoint to classify each row.
|
|
466
|
+
`intent: 'llm_enrichment'` distinguishes it from the SMS tool above.
|
|
467
|
+
|
|
468
|
+
It is a **provider adapter**, and the two halves are the thing to read.
|
|
469
|
+
`base_body` authors the provider's request — here Ollama's native
|
|
470
|
+
`/api/chat` shape with `think: false` and a `format` schema so the model
|
|
471
|
+
returns clean, schema-constrained JSON. `response_extractor` maps the
|
|
472
|
+
provider's envelope back to the canonical `{ output_json, … }` the platform
|
|
473
|
+
reads, and its `output_schema` is **required** for an `llm_enrichment`
|
|
474
|
+
tool. The `api_key` / `auth_method` pair satisfies the REST tool's schema
|
|
475
|
+
even though Ollama ignores the header.
|
|
476
|
+
|
|
477
|
+
```typescript
|
|
478
|
+
const ENRICHMENT_OUTPUT_SCHEMA = {
|
|
479
|
+
type: 'object',
|
|
480
|
+
properties: {
|
|
481
|
+
output_json: {},
|
|
482
|
+
input_tokens: { type: ['integer', 'null'] },
|
|
483
|
+
output_tokens: { type: ['integer', 'null'] },
|
|
484
|
+
total_tokens: { type: ['integer', 'null'] },
|
|
485
|
+
explanation: { type: ['string', 'null'] },
|
|
486
|
+
},
|
|
487
|
+
required: ['output_json'],
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
const OLLAMA_BASE_BODY =
|
|
491
|
+
'{"model": "{{ model }}", "messages": [{"role": "user", "content": "{{ rendered_prompt | json_escape }}", "images": [{% for img in images %}{% unless forloop.first %}, {% endunless %}"{{ img.data }}"{% endfor %}]}], "stream": false, "think": false, "options": {"temperature": {{ temperature }}, "num_predict": {{ max_tokens }}, "num_ctx": 40960}, "format": {{ schema | to_json }}}'
|
|
492
|
+
|
|
493
|
+
const OLLAMA_EXTRACTOR =
|
|
494
|
+
'{"output_json": "{{ msg.message.content | json_escape }}", "explanation": "{{ msg.message.thinking | json_escape }}", "input_tokens": {{ msg.prompt_eval_count | default: 0 }}, "output_tokens": {{ msg.eval_count | default: 0 }}, "total_tokens": {{ msg.prompt_eval_count | default: 0 | plus: msg.eval_count }}}'
|
|
495
|
+
|
|
496
|
+
const llmTool = await ctx.ensure(
|
|
497
|
+
'LLM tool',
|
|
498
|
+
async () => {
|
|
499
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
500
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Marketing LLM Tool')
|
|
501
|
+
},
|
|
502
|
+
async () => {
|
|
503
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
504
|
+
name: 'Cookbook Marketing LLM Tool',
|
|
505
|
+
description: 'Ollama-backed chat-completion adapter for Score Lead classification.',
|
|
506
|
+
intent: 'llm_enrichment',
|
|
507
|
+
status: 'active',
|
|
508
|
+
datalake_id: ctx.datalakeId,
|
|
509
|
+
response_extractor: {
|
|
510
|
+
type: 'custom',
|
|
511
|
+
body: OLLAMA_EXTRACTOR,
|
|
512
|
+
output_schema: ENRICHMENT_OUTPUT_SCHEMA,
|
|
513
|
+
},
|
|
514
|
+
body: {
|
|
515
|
+
tool_body_type: 'rest_api',
|
|
516
|
+
base_url: 'http://localhost:11434',
|
|
517
|
+
base_path: { type: 'custom', body: '/api/chat' },
|
|
518
|
+
auth_method: 'api_key',
|
|
519
|
+
api_key: 'stub-key',
|
|
520
|
+
api_key_name: 'Authorization',
|
|
521
|
+
api_key_location: 'header',
|
|
522
|
+
request_type: 'json',
|
|
523
|
+
response_type: 'json',
|
|
524
|
+
timeout_ms: 60_000,
|
|
525
|
+
base_body: { type: 'custom', body: OLLAMA_BASE_BODY },
|
|
526
|
+
},
|
|
527
|
+
})
|
|
528
|
+
return data
|
|
529
|
+
},
|
|
530
|
+
)
|
|
531
|
+
ctx.llmToolId = llmTool.id!
|
|
532
|
+
```
|
|
533
|
+
|
|
534
|
+
## 010 — the Lead Submissions table
|
|
535
|
+
|
|
536
|
+
Lead-capture rows do not fit any canonical entity, so they are a generic
|
|
537
|
+
table you declare. The columns mirror what a form export or a spreadsheet
|
|
538
|
+
tab actually produces. `message` and `lead_source` are what the agent
|
|
539
|
+
reads; `band` is what a production system writes back.
|
|
540
|
+
|
|
541
|
+
`name` and `email` are declared `tokenize`. **That declaration is the
|
|
542
|
+
masking policy, and it is correct whether or not a tokenized lake exists
|
|
543
|
+
today** — this walk provisions none, so nothing masks here, and the
|
|
544
|
+
declaration goes live the moment someone calls `provisionTokenization`.
|
|
545
|
+
Declaring it now is how the policy survives that call rather than being
|
|
546
|
+
remembered afterwards.
|
|
547
|
+
|
|
548
|
+
```typescript
|
|
549
|
+
const leads = await ctx.ensure(
|
|
550
|
+
'lead-submissions table',
|
|
551
|
+
async () => {
|
|
552
|
+
const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
|
|
553
|
+
return (data.data ?? []).find((t: { title?: string }) => t.title === 'Cookbook Lead Submissions')
|
|
554
|
+
},
|
|
555
|
+
async () => {
|
|
556
|
+
const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
|
|
557
|
+
title: 'Cookbook Lead Submissions',
|
|
558
|
+
description: 'Inbound lead-capture form rows the Score Lead agent classifies.',
|
|
559
|
+
columns: [
|
|
560
|
+
{ name: 'submission_id', title: 'Submission ID', type: 'string', description: 'Vendor-supplied unique submission id', is_unique: true, privacy_requirement: 'none' },
|
|
561
|
+
{ name: 'name', title: 'Name', type: 'string', description: 'Submitter full name', is_unique: false, privacy_requirement: 'tokenize' },
|
|
562
|
+
{ name: 'email', title: 'Email', type: 'string', description: 'Submitter email', is_unique: false, privacy_requirement: 'tokenize' },
|
|
563
|
+
{ name: 'message', title: 'Message', type: 'string', description: 'Free-text lead message — the agent classifies this', is_unique: false, privacy_requirement: 'none' },
|
|
564
|
+
{ name: 'lead_source', title: 'Lead Source', type: 'string', description: 'utm_source / sheet tab id', is_unique: false, privacy_requirement: 'none' },
|
|
565
|
+
{ name: 'band', title: 'Band', type: 'string', description: 'Lead-scoring band assigned by the Score Lead agent', is_unique: false, privacy_requirement: 'none' },
|
|
566
|
+
],
|
|
567
|
+
})
|
|
568
|
+
return data
|
|
569
|
+
},
|
|
570
|
+
)
|
|
571
|
+
genericTableId = leads.id!
|
|
572
|
+
ctx.leadsTableName = leads.name!
|
|
573
|
+
|
|
574
|
+
// Called unconditionally rather than only on the create branch, so a run
|
|
575
|
+
// that crashed between recording the table and deploying it still converges.
|
|
576
|
+
await ctx.deployGenericTable(ctx.leadsTableName)
|
|
577
|
+
```
|
|
578
|
+
|
|
579
|
+
## 011 — the Score Lead agent
|
|
580
|
+
|
|
581
|
+
The agent binds three things: the model name sent on every inference call,
|
|
582
|
+
an input schema the workflow's context mapping must satisfy, and a response
|
|
583
|
+
schema the output must match.
|
|
584
|
+
|
|
585
|
+
The response schema's `enum: ['hot','warm','cold','spam']` is the guard that
|
|
586
|
+
matters. It stops the model emitting a band the workflow has no action for
|
|
587
|
+
— without it a creative answer becomes a decision key nothing matches, and
|
|
588
|
+
the row silently fans out to nothing. `temperature: 0.0` removes sampling
|
|
589
|
+
noise so identical inputs classify identically, which is what makes the
|
|
590
|
+
assertion in §016 stable.
|
|
591
|
+
|
|
592
|
+
`data_access: 'raw'` because this lake has one tier. On a surface that
|
|
593
|
+
provisions the derived pair you would set `'tokenized'` here and the agent
|
|
594
|
+
would never see the name at all.
|
|
595
|
+
|
|
596
|
+
```typescript
|
|
597
|
+
const SCORE_INPUT_SCHEMA = {
|
|
598
|
+
type: 'object',
|
|
599
|
+
properties: {
|
|
600
|
+
name: { type: 'string' },
|
|
601
|
+
message: { type: 'string' },
|
|
602
|
+
lead_source: { type: 'string' },
|
|
603
|
+
},
|
|
604
|
+
required: ['name', 'message'],
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
const SCORE_RESPONSE_SCHEMA = {
|
|
608
|
+
type: 'object',
|
|
609
|
+
properties: { band: { type: 'string', enum: ['hot', 'warm', 'cold', 'spam'] } },
|
|
610
|
+
required: ['band'],
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
const AGENT_PROMPT_BODY = `You are a B2B lead-scoring assistant. Classify the following lead's intent into EXACTLY ONE of these bands:
|
|
614
|
+
|
|
615
|
+
- "hot" — clear buying intent, decision maker, ready to engage
|
|
616
|
+
- "warm" — interested but exploratory, requires nurture
|
|
617
|
+
- "cold" — generic/lukewarm interest, low conversion signal
|
|
618
|
+
- "spam" — promotional, irrelevant, or low-quality submission
|
|
619
|
+
|
|
620
|
+
Lead name: {{ name }}
|
|
621
|
+
Lead message: {{ message }}
|
|
622
|
+
Lead source: {{ lead_source }}
|
|
623
|
+
|
|
624
|
+
Respond with a JSON object: {"band": "<one of hot|warm|cold|spam>"}`
|
|
625
|
+
|
|
626
|
+
const agent = await ctx.ensure(
|
|
627
|
+
'Score Lead agent',
|
|
628
|
+
async () => {
|
|
629
|
+
const { data } = await api.aiAgents.list(tenantSlug, datalakeSlug)
|
|
630
|
+
return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Score Lead Agent')
|
|
631
|
+
},
|
|
632
|
+
async () => {
|
|
633
|
+
const { data } = await api.aiAgents.create(tenantSlug, datalakeSlug, {
|
|
634
|
+
name: 'Cookbook Score Lead Agent',
|
|
635
|
+
tool_id: ctx.llmToolId,
|
|
636
|
+
model: 'qwen3-vl:8b-instruct',
|
|
637
|
+
data_access: 'raw',
|
|
638
|
+
temperature: 0.0,
|
|
639
|
+
max_tokens: 1024,
|
|
640
|
+
enabled: true,
|
|
641
|
+
input_schema: SCORE_INPUT_SCHEMA,
|
|
642
|
+
llm_response_schema: SCORE_RESPONSE_SCHEMA,
|
|
643
|
+
prompt_config: { type: 'custom', body: AGENT_PROMPT_BODY },
|
|
644
|
+
})
|
|
645
|
+
return data
|
|
646
|
+
},
|
|
647
|
+
)
|
|
648
|
+
aiAgentId = agent.id!
|
|
649
|
+
ctx.agentSlug = agent.slug!
|
|
650
|
+
```
|
|
651
|
+
|
|
652
|
+
## 012 — the Score Lead workflow
|
|
653
|
+
|
|
654
|
+
Standard shape — filter, decision, actions — with two things worth reading
|
|
655
|
+
closely.
|
|
656
|
+
|
|
657
|
+
The agent is **nested in the create body** (`workflow_ai_agents`, a
|
|
658
|
+
`cast_assoc` rather than a separate attach call), which binds it into the
|
|
659
|
+
workflow's enrichment phase along with a Liquid `context_mapping_config`
|
|
660
|
+
that projects each row into the agent's input schema. A workflow join omits
|
|
661
|
+
`output_schema`; the server pins it from the agent's own `input_schema`.
|
|
662
|
+
|
|
663
|
+
The `decision_config` body interpolates the agent's `band` into a
|
|
664
|
+
single-element decision array, and the four actions are keyed
|
|
665
|
+
`decision_key: hot|warm|cold|spam`. Whichever band comes back picks one
|
|
666
|
+
action and leaves the other three `:skipped`.
|
|
667
|
+
|
|
668
|
+
**Bracket access is required**, not stylistic: the agent's slug contains
|
|
669
|
+
hyphens, and Liquid's dot parser would read `additional_context.score-lead`
|
|
670
|
+
as a subtraction.
|
|
671
|
+
|
|
672
|
+
```typescript
|
|
673
|
+
const BANDS = ['hot', 'warm', 'cold', 'spam'] as const
|
|
674
|
+
|
|
675
|
+
const CONTEXT_MAPPING_BODY = JSON.stringify({
|
|
676
|
+
name: '{{ event_dataset.name }}',
|
|
677
|
+
message: '{{ event_dataset.message }}',
|
|
678
|
+
lead_source: '{{ event_dataset.lead_source }}',
|
|
679
|
+
})
|
|
680
|
+
|
|
681
|
+
const DECISION_CONFIG_BODY = `["{{ additional_context["${ctx.agentSlug}"].band }}"]`
|
|
682
|
+
|
|
683
|
+
const scoreWorkflow = await ctx.ensure(
|
|
684
|
+
'Score Lead workflow',
|
|
685
|
+
async () => {
|
|
686
|
+
const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
|
|
687
|
+
return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Score Lead Workflow')
|
|
688
|
+
},
|
|
689
|
+
async () => {
|
|
690
|
+
const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
|
|
691
|
+
name: 'Cookbook Score Lead Workflow',
|
|
692
|
+
description: 'Classifies lead submissions into hot/warm/cold/spam via an LLM agent; one SMS action per band.',
|
|
693
|
+
dataset_type: 'generic_table',
|
|
694
|
+
generic_table_id: genericTableId,
|
|
695
|
+
skip_mdm_resolution: true,
|
|
696
|
+
status: 'live',
|
|
697
|
+
tags: ['leads', 'llm'],
|
|
698
|
+
filter_config: { type: 'custom', body: 'true' },
|
|
699
|
+
decision_config: {
|
|
700
|
+
type: 'custom',
|
|
701
|
+
body: DECISION_CONFIG_BODY,
|
|
702
|
+
output_schema: { type: 'array', items: { type: 'string' } },
|
|
703
|
+
},
|
|
704
|
+
actions: BANDS.map((band) => ({
|
|
705
|
+
decision_key: band,
|
|
706
|
+
action_type: 'sms',
|
|
707
|
+
tool_id: ctx.smsToolId,
|
|
708
|
+
position: 0,
|
|
709
|
+
trigger_template: 'now',
|
|
710
|
+
idempotency_template: `{{ event_dataset.submission_id }}-${band}`,
|
|
711
|
+
tool_call: {
|
|
712
|
+
tool_call_type: 'sms_request',
|
|
713
|
+
to: { type: 'custom', body: '+15551234567' },
|
|
714
|
+
body: { type: 'custom', body: `Lead band [${band}]: {{ event_dataset.message }}` },
|
|
715
|
+
sms_type: 'transactional',
|
|
716
|
+
},
|
|
717
|
+
})),
|
|
718
|
+
workflow_ai_agents: [
|
|
719
|
+
{
|
|
720
|
+
ai_agent_id: aiAgentId,
|
|
721
|
+
position: 0,
|
|
722
|
+
context_mapping_config: { type: 'custom', body: CONTEXT_MAPPING_BODY },
|
|
723
|
+
},
|
|
724
|
+
],
|
|
725
|
+
})
|
|
726
|
+
return data
|
|
727
|
+
},
|
|
728
|
+
)
|
|
729
|
+
workflowId = scoreWorkflow.id!
|
|
730
|
+
ctx.scoreWorkflowSlug = scoreWorkflow.slug!
|
|
731
|
+
```
|
|
732
|
+
|
|
733
|
+
## 013 — the table's own Data Activation Client, which you did not create
|
|
734
|
+
|
|
735
|
+
Creating a generic table provisions two things for you: an identity
|
|
736
|
+
interoperability contract, and a **default Data Activation Client** bound to
|
|
737
|
+
it, named `<datalake> <table> DataActivationClient` with
|
|
738
|
+
`tool_call: manual_upload`. For plain row ingestion into a table you
|
|
739
|
+
declared, there is nothing to build — no data source, no tool, no contract,
|
|
740
|
+
no client.
|
|
741
|
+
|
|
742
|
+
**Scope the lookup server-side.** The listing pages at twenty and is not
|
|
743
|
+
newest-first, so a `.find()` over page one starts missing the client you
|
|
744
|
+
want as soon as the lake has a few tables — and the failure looks like the
|
|
745
|
+
client was never provisioned.
|
|
746
|
+
|
|
747
|
+
```typescript
|
|
748
|
+
const dacDeadline = Date.now() + 120_000
|
|
749
|
+
let defaultDac: { slug?: string | null } | undefined
|
|
750
|
+
while (Date.now() < dacDeadline && !defaultDac) {
|
|
751
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
752
|
+
filters: [{ field: 'name', op: 'ilike', value: ctx.leadsTableName }],
|
|
753
|
+
})
|
|
754
|
+
defaultDac = (data.data ?? [])[0]
|
|
755
|
+
if (!defaultDac) await new Promise((r) => setTimeout(r, 1_000))
|
|
756
|
+
}
|
|
757
|
+
if (!defaultDac?.slug) {
|
|
758
|
+
throw new Error(`no default DAC found for table ${ctx.leadsTableName} within 120s`)
|
|
759
|
+
}
|
|
760
|
+
ctx.leadsDacSlug = defaultDac.slug
|
|
761
|
+
```
|
|
762
|
+
|
|
763
|
+
## 014 — ingest three leads, one obviously each way
|
|
764
|
+
|
|
765
|
+
Three rows through the default client. Each carries a deliberately
|
|
766
|
+
unambiguous `message` — approved budget and a deadline, an idle browse, and
|
|
767
|
+
obvious promotional spam — so the classification is not a coin toss and the
|
|
768
|
+
assertion in §016 means something. `band` is left unset on ingest; assigning
|
|
769
|
+
it is the agent's job.
|
|
770
|
+
|
|
771
|
+
The submission ids are **stable across runs**, and `submission_id` is the
|
|
772
|
+
table's unique column, so a second run upserts these same three rows rather
|
|
773
|
+
than adding three more.
|
|
774
|
+
|
|
775
|
+
```typescript
|
|
776
|
+
ctx.hotLead = 'LEAD-COOKBOOK-0101'
|
|
777
|
+
ctx.coldLead = 'LEAD-COOKBOOK-0102'
|
|
778
|
+
ctx.spamLead = 'LEAD-COOKBOOK-0103'
|
|
779
|
+
|
|
780
|
+
const leadRows = [
|
|
781
|
+
{
|
|
782
|
+
submission_id: ctx.hotLead,
|
|
783
|
+
name: 'Dana Decisive',
|
|
784
|
+
email: 'dana@example.com',
|
|
785
|
+
message:
|
|
786
|
+
'We have budget approved and need to roll out to 500 seats this quarter — can we start onboarding next week?',
|
|
787
|
+
lead_source: 'cookbook_score_leads',
|
|
788
|
+
},
|
|
789
|
+
{
|
|
790
|
+
submission_id: ctx.coldLead,
|
|
791
|
+
name: 'Sam Browsing',
|
|
792
|
+
email: 'sam@example.com',
|
|
793
|
+
message: 'Just browsing — found your site via a blog post. No particular need right now.',
|
|
794
|
+
lead_source: 'cookbook_score_leads',
|
|
795
|
+
},
|
|
796
|
+
{
|
|
797
|
+
submission_id: ctx.spamLead,
|
|
798
|
+
name: 'Promo Bot',
|
|
799
|
+
email: 'promo@example.com',
|
|
800
|
+
message: 'BUY CHEAP FOLLOWERS NOW!!! 90% off SEO backlinks, crypto giveaways, click here click here!!!',
|
|
801
|
+
lead_source: 'cookbook_score_leads',
|
|
802
|
+
},
|
|
803
|
+
]
|
|
804
|
+
|
|
805
|
+
const ingests: string[] = []
|
|
806
|
+
for (const row of leadRows) {
|
|
807
|
+
const resp = await api.dataActivationClients.ingest(
|
|
808
|
+
tenantSlug, datalakeSlug, ctx.leadsDacSlug, { data: row },
|
|
809
|
+
)
|
|
810
|
+
ingests.push(resp.data.batch_id!)
|
|
811
|
+
}
|
|
812
|
+
ctx.leadBatchIds = ingests
|
|
813
|
+
```
|
|
814
|
+
|
|
815
|
+
## 015 — wait for all three to persist
|
|
816
|
+
|
|
817
|
+
```typescript
|
|
818
|
+
await ctx.waitForBatches(ctx.leadsDacSlug, ctx.leadBatchIds)
|
|
819
|
+
```
|
|
820
|
+
|
|
821
|
+
## 016 — run the workflow across the three leads
|
|
822
|
+
|
|
823
|
+
`workflows.run` triggers the whole pipeline per row: filter, then agent
|
|
824
|
+
enrichment — a live inference call per row — then the decision, then the
|
|
825
|
+
action fan-out. The where clause scopes the run to exactly the three
|
|
826
|
+
submissions §014 ingested, so rows left by any other walk are untouched.
|
|
827
|
+
|
|
828
|
+
The window is generous because three real model calls happen inside it.
|
|
829
|
+
|
|
830
|
+
```typescript
|
|
831
|
+
const submissionList = [ctx.hotLead, ctx.coldLead, ctx.spamLead].map((s) => `'${s}'`).join(', ')
|
|
832
|
+
|
|
833
|
+
const runResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.scoreWorkflowSlug, {
|
|
834
|
+
sql_where_clause: `submission_id IN (${submissionList})`,
|
|
835
|
+
mode: 'live',
|
|
836
|
+
manual_override: true,
|
|
837
|
+
})
|
|
838
|
+
const fired = await ctx.waitForFiredRun(datalakeSlug, runResp.data.workflow_run_id)
|
|
839
|
+
ctx.scoreRunLogId = fired.workflowRunLogId
|
|
840
|
+
ctx.scoreRunBatchId = fired.batchId!
|
|
841
|
+
|
|
842
|
+
const runDeadline = Date.now() + 240_000
|
|
843
|
+
let runStatus: string | null = null
|
|
844
|
+
while (Date.now() < runDeadline) {
|
|
845
|
+
const { data: log } = await api.workflows.batchLogs.refresh(
|
|
846
|
+
tenantSlug, datalakeSlug, ctx.scoreWorkflowSlug, ctx.scoreRunLogId,
|
|
847
|
+
)
|
|
848
|
+
runStatus = log.status ?? null
|
|
849
|
+
if (runStatus && runStatus !== 'pending') break
|
|
850
|
+
await new Promise((r) => setTimeout(r, 2_000))
|
|
851
|
+
}
|
|
852
|
+
if (runStatus === 'failed') throw new Error('Score Lead run reached :failed')
|
|
853
|
+
if (!runStatus || runStatus === 'pending') {
|
|
854
|
+
throw new Error('Score Lead run did not leave :pending within 240s')
|
|
855
|
+
}
|
|
856
|
+
```
|
|
857
|
+
|
|
858
|
+
## 017 — verify the agent's band actually steered the fan-out
|
|
859
|
+
|
|
860
|
+
Each lead produced a Workflow Execution Log carrying **four** action
|
|
861
|
+
execution logs, one per band.
|
|
862
|
+
|
|
863
|
+
On the first run for a given submission, exactly **one** is matched
|
|
864
|
+
(`:pending` or `:completed` — the band the agent chose) and **three** are
|
|
865
|
+
`:skipped`. That one-of-four is the proof the agent's output drove the
|
|
866
|
+
decision rather than every action firing.
|
|
867
|
+
|
|
868
|
+
**On a later run all four are `:skipped`, and that is correct.** The
|
|
869
|
+
idempotency key is `{{ submission_id }}-<band>`, the submission ids are
|
|
870
|
+
stable, so the key that already fired refuses to fire again — the platform
|
|
871
|
+
will not send the same SMS about the same lead twice. Both shapes are
|
|
872
|
+
accepted here, and anything else is a real failure: two matched actions
|
|
873
|
+
means the decision matched more than one band, and zero matched with zero
|
|
874
|
+
skipped means the fan-out never happened.
|
|
875
|
+
|
|
876
|
+
```typescript
|
|
877
|
+
const { data: wfLogs } = await api.workflows.workflowLogs.list(
|
|
878
|
+
tenantSlug, datalakeSlug, ctx.scoreWorkflowSlug,
|
|
879
|
+
)
|
|
880
|
+
const ourWels = (wfLogs.data ?? []).filter(
|
|
881
|
+
(w) => (w as { batch_id?: string }).batch_id === ctx.scoreRunBatchId,
|
|
882
|
+
)
|
|
883
|
+
if (ourWels.length !== 3) {
|
|
884
|
+
throw new Error(`expected 3 WELs for the run, got ${ourWels.length}`)
|
|
885
|
+
}
|
|
886
|
+
|
|
887
|
+
for (const wel of ourWels) {
|
|
888
|
+
const welId = (wel as { id?: string }).id
|
|
889
|
+
const aels = (wel as { action_execution_logs?: Array<{ status?: string }> }).action_execution_logs ?? []
|
|
890
|
+
if (aels.length !== 4) {
|
|
891
|
+
throw new Error(`WEL ${welId}: expected 4 AELs (one per band), got ${aels.length}`)
|
|
892
|
+
}
|
|
893
|
+
const byStatus: Record<string, number> = {}
|
|
894
|
+
for (const ael of aels) {
|
|
895
|
+
const st = ael.status ?? 'unknown'
|
|
896
|
+
byStatus[st] = (byStatus[st] ?? 0) + 1
|
|
897
|
+
}
|
|
898
|
+
const matched = (byStatus.pending ?? 0) + (byStatus.completed ?? 0)
|
|
899
|
+
const skipped = byStatus.skipped ?? 0
|
|
900
|
+
|
|
901
|
+
if (matched === 1 && skipped === 3) continue // first run for this lead
|
|
902
|
+
if (matched === 0 && skipped === 4) continue // already fired, correctly refusing
|
|
903
|
+
throw new Error(
|
|
904
|
+
`WEL ${welId}: expected 1 matched + 3 skipped, or 4 skipped on a repeat run — got ${JSON.stringify(byStatus)}`,
|
|
905
|
+
)
|
|
906
|
+
}
|
|
907
|
+
```
|
|
908
|
+
|
|
909
|
+
## 018 — confirm a band-prefixed SMS actually rendered
|
|
910
|
+
|
|
911
|
+
§017 proves the fan-out chose one action. This proves something was
|
|
912
|
+
actually *sent*, and it is the assertion that survives every rerun.
|
|
913
|
+
|
|
914
|
+
Search on the **idempotency key**, never on `workflow_id`. The key is
|
|
915
|
+
`<submission id>-<band>` and the submission id is stable, so it names the
|
|
916
|
+
same SMS forever. A workflow id is stable only as long as the workflow
|
|
917
|
+
resource is: recreate it against a lake that still holds its rows and the
|
|
918
|
+
id changes while the key does not, so the action correctly refuses to fire
|
|
919
|
+
and a search for the new id finds nothing at all.
|
|
920
|
+
|
|
921
|
+
The hot lead is the one asserted, because it is the only one whose band is
|
|
922
|
+
genuinely unambiguous — approved budget, a seat count and a deadline is
|
|
923
|
+
`hot` to any classifier worth running.
|
|
924
|
+
|
|
925
|
+
```typescript
|
|
926
|
+
const { data: leadSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
|
|
927
|
+
search_query: `m.idempotency_key LIKE '${ctx.hotLead}-%'`,
|
|
928
|
+
})
|
|
929
|
+
if (leadSearch.status !== 'completed') {
|
|
930
|
+
throw new Error(`message user-search status=${leadSearch.status} error=${leadSearch.error_message ?? '(none)'}`)
|
|
931
|
+
}
|
|
932
|
+
|
|
933
|
+
const leadMsgDeadline = Date.now() + 45_000
|
|
934
|
+
let leadMessages: Array<Record<string, unknown>> = []
|
|
935
|
+
while (Date.now() < leadMsgDeadline && leadMessages.length === 0) {
|
|
936
|
+
const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
|
|
937
|
+
userSearchId: leadSearch.id!,
|
|
938
|
+
dataAccessMode: 'raw',
|
|
939
|
+
})
|
|
940
|
+
leadMessages = (data.data ?? []) as Array<Record<string, unknown>>
|
|
941
|
+
if (leadMessages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
|
|
942
|
+
}
|
|
943
|
+
if (leadMessages.length === 0) {
|
|
944
|
+
throw new Error(
|
|
945
|
+
`no band SMS for lead ${ctx.hotLead} within 45s — ` +
|
|
946
|
+
`the action never fired, or it fired under a different idempotency key`,
|
|
947
|
+
)
|
|
948
|
+
}
|
|
949
|
+
|
|
950
|
+
const bandBody = leadMessages.map((m) => String(m.body ?? '')).find((b) => b.includes('Lead band ['))
|
|
951
|
+
if (!bandBody) {
|
|
952
|
+
throw new Error(`no band-prefixed SMS body — bodies: ${JSON.stringify(leadMessages.map((m) => m.body))}`)
|
|
953
|
+
}
|
|
954
|
+
if (!bandBody.includes('[hot]')) {
|
|
955
|
+
throw new Error(`expected the classifier to band this lead [hot] — got: ${bandBody}`)
|
|
956
|
+
}
|
|
957
|
+
```
|
|
958
|
+
|
|
959
|
+
## 019 — the Birthday Greeting connected app
|
|
960
|
+
|
|
961
|
+
The greeting carries a deep link so the recipient can do something with it
|
|
962
|
+
— upload a photo, RSVP, whatever the operator wires up. A connected app is
|
|
963
|
+
a thin registration of that page's URL and mode; the page itself is hosted
|
|
964
|
+
outside the platform (`mode: 'self_hosted'`). Capture the server-derived
|
|
965
|
+
`slug` — §030 resolves a token against it.
|
|
966
|
+
|
|
967
|
+
```typescript
|
|
968
|
+
const birthdayApp = await ctx.ensure(
|
|
969
|
+
'Birthday Greeting connected app',
|
|
970
|
+
async () => {
|
|
971
|
+
const { data } = await api.connectedApps.list(tenantSlug, datalakeSlug)
|
|
972
|
+
return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Birthday Greeting Form')
|
|
973
|
+
},
|
|
974
|
+
async () => {
|
|
975
|
+
const { data } = await api.connectedApps.create(tenantSlug, datalakeSlug, {
|
|
976
|
+
name: 'Cookbook Birthday Greeting Form',
|
|
977
|
+
description: 'Birthday-greeting form linked from the outbound SMS.',
|
|
978
|
+
mode: 'self_hosted',
|
|
979
|
+
urls: [{ url: 'https://birthday.example.local', is_primary: true, label: 'production' }],
|
|
980
|
+
})
|
|
981
|
+
return data
|
|
982
|
+
},
|
|
983
|
+
)
|
|
984
|
+
connectedAppId = birthdayApp.id!
|
|
985
|
+
ctx.birthdayAppSlug = birthdayApp.slug!
|
|
986
|
+
```
|
|
987
|
+
|
|
988
|
+
## 020 — the Happy Birthday workflow, and the year-roll trigger
|
|
989
|
+
|
|
990
|
+
No agent here, and that is the point of putting it beside §012: the same
|
|
991
|
+
workflow shape does this job with nothing but Liquid.
|
|
992
|
+
|
|
993
|
+
**The trigger template is the whole scenario.** It compares today's `MM-DD`
|
|
994
|
+
against the contact's date of birth `MM-DD` and renders either this year's
|
|
995
|
+
birthday or next year's. String comparison works because both sides are
|
|
996
|
+
zero-padded `MM-DD` — the one detail that makes a one-line year-roll
|
|
997
|
+
correct instead of nearly correct.
|
|
998
|
+
|
|
999
|
+
Two things that look optional and are not. The filter reads
|
|
1000
|
+
`mdm_output.legal_entity.date_of_birth`, which requires
|
|
1001
|
+
`skip_mdm_resolution: false` — the resolution is what puts the subject in
|
|
1002
|
+
scope for the filter at all. And the SMS body must reference
|
|
1003
|
+
`{{ connected_app_form_url }}`: the platform mints the page token at action
|
|
1004
|
+
time and injects that variable, so a body that never mentions it gets a
|
|
1005
|
+
token minted and thrown away, and §031 has no link to resolve.
|
|
1006
|
+
|
|
1007
|
+
```typescript
|
|
1008
|
+
const DECISION_KEY = 'send_happy_birthday_sms'
|
|
1009
|
+
const TRIGGER_TEMPLATE =
|
|
1010
|
+
'{% assign year = "" | now | date: "%Y" %}' +
|
|
1011
|
+
'{% assign today_mmdd = "" | now | date: "%m-%d" %}' +
|
|
1012
|
+
'{% assign dob_mmdd = mdm_output.legal_entity.date_of_birth | date: "%m-%d" %}' +
|
|
1013
|
+
'{% if dob_mmdd >= today_mmdd %}{{ year }}-{{ dob_mmdd }} 00:00:00' +
|
|
1014
|
+
'{% else %}{{ year | plus: 1 }}-{{ dob_mmdd }} 00:00:00{% endif %}'
|
|
1015
|
+
|
|
1016
|
+
const birthdayWorkflow = await ctx.ensure(
|
|
1017
|
+
'Happy Birthday workflow',
|
|
1018
|
+
async () => {
|
|
1019
|
+
const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
|
|
1020
|
+
return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Happy Birthday Workflow')
|
|
1021
|
+
},
|
|
1022
|
+
async () => {
|
|
1023
|
+
const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
|
|
1024
|
+
name: 'Cookbook Happy Birthday Workflow',
|
|
1025
|
+
description: "Sends a birthday SMS on each contact's next birthday — pure-Liquid trigger does the year-roll math.",
|
|
1026
|
+
dataset_type: 'legal_entity',
|
|
1027
|
+
status: 'live',
|
|
1028
|
+
tags: ['lifecycle', 'birthday'],
|
|
1029
|
+
skip_mdm_resolution: false,
|
|
1030
|
+
filter_config: { type: 'custom', body: '{% if mdm_output.legal_entity.date_of_birth %}true{% endif %}' },
|
|
1031
|
+
decision_config: {
|
|
1032
|
+
type: 'custom',
|
|
1033
|
+
body: `["${DECISION_KEY}"]`,
|
|
1034
|
+
output_schema: { type: 'array', items: { type: 'string' } },
|
|
1035
|
+
},
|
|
1036
|
+
actions: [
|
|
1037
|
+
{
|
|
1038
|
+
action_type: 'sms',
|
|
1039
|
+
tool_id: ctx.smsToolId,
|
|
1040
|
+
decision_key: DECISION_KEY,
|
|
1041
|
+
position: 0,
|
|
1042
|
+
trigger_template: TRIGGER_TEMPLATE,
|
|
1043
|
+
idempotency_template: '{{ legal_entity.id }}-birthday-greeting',
|
|
1044
|
+
connected_app_id: connectedAppId,
|
|
1045
|
+
connected_app_route: '/forms/birthday-greeting',
|
|
1046
|
+
connected_app_metadata_template: '{"legal_entity_id":"{{ legal_entity.id }}"}',
|
|
1047
|
+
tool_call: {
|
|
1048
|
+
tool_call_type: 'sms_request',
|
|
1049
|
+
to: { type: 'custom', body: '{{ mdm_output.legal_entity.phone_numbers | first | map: "phone_number" }}' },
|
|
1050
|
+
body: {
|
|
1051
|
+
type: 'custom',
|
|
1052
|
+
body: 'Happy birthday {{ mdm_output.legal_entity.first_name }}! \u{1F382} Share a photo: {{ connected_app_form_url }}',
|
|
1053
|
+
},
|
|
1054
|
+
sms_type: 'transactional',
|
|
1055
|
+
},
|
|
1056
|
+
},
|
|
1057
|
+
],
|
|
1058
|
+
})
|
|
1059
|
+
return data
|
|
1060
|
+
},
|
|
1061
|
+
)
|
|
1062
|
+
ctx.birthdayWorkflowSlug = birthdayWorkflow.slug!
|
|
1063
|
+
```
|
|
1064
|
+
|
|
1065
|
+
## 021 — the data source, the upload tool, and the contract
|
|
1066
|
+
|
|
1067
|
+
Three small pieces, one step, because none of them is interesting alone.
|
|
1068
|
+
|
|
1069
|
+
The **data source** registers where rows come from. Its `uri` matters more
|
|
1070
|
+
than it looks: the Data Activation Client injects it into every row as
|
|
1071
|
+
`source_uri`, and the contract template writes it as the identification's
|
|
1072
|
+
`uri` — so it is half of the key the legal-entity dedupe uses. Change it
|
|
1073
|
+
and yesterday's contacts stop matching today's.
|
|
1074
|
+
|
|
1075
|
+
The **manual-upload tool** is the minimal tool for inline JSON; it needs no
|
|
1076
|
+
endpoint or credential because the rows arrive in the ingest call rather
|
|
1077
|
+
than being fetched.
|
|
1078
|
+
|
|
1079
|
+
The **contract** shapes a row into a legal-entity upsert. Its
|
|
1080
|
+
`mdm_input_config` is `{ type: 'null' }` — the row *is* the subject, so
|
|
1081
|
+
there is nothing to resolve. The template comes from the vendored fixture
|
|
1082
|
+
rather than several hundred inlined lines of Liquid.
|
|
1083
|
+
|
|
1084
|
+
```typescript
|
|
1085
|
+
const { readFileSync } = await import('node:fs')
|
|
1086
|
+
const { join } = await import('node:path')
|
|
1087
|
+
const leTemplate = readFileSync(
|
|
1088
|
+
join(process.env.COOKBOOK_FIXTURES_DIR!, 'organic-marketing', '_lead_submissions_foundation_legal_entity.liquid'),
|
|
1089
|
+
'utf8',
|
|
1090
|
+
)
|
|
1091
|
+
|
|
1092
|
+
ctx.leadSourceUri = 'https://app.alvera.ai/cookbook-lead-form'
|
|
1093
|
+
|
|
1094
|
+
const leadSource = await ctx.ensure(
|
|
1095
|
+
'lead-form data source',
|
|
1096
|
+
async () => {
|
|
1097
|
+
const { data } = await api.dataSources.list(tenantSlug, datalakeSlug)
|
|
1098
|
+
return (data.data ?? []).find((d: { name?: string }) => d.name === 'Cookbook Lead Form Source')
|
|
1099
|
+
},
|
|
1100
|
+
async () => {
|
|
1101
|
+
const { data } = await api.dataSources.create(tenantSlug, datalakeSlug, {
|
|
1102
|
+
name: 'Cookbook Lead Form Source',
|
|
1103
|
+
uri: ctx.leadSourceUri,
|
|
1104
|
+
description: 'Inbound lead-capture form — origin of the contact rows the birthday workflow runs on.',
|
|
1105
|
+
status: 'active',
|
|
1106
|
+
is_default: false,
|
|
1107
|
+
})
|
|
1108
|
+
return data
|
|
1109
|
+
},
|
|
1110
|
+
)
|
|
1111
|
+
dataSourceId = leadSource.id!
|
|
1112
|
+
|
|
1113
|
+
const uploadTool = await ctx.ensure(
|
|
1114
|
+
'manual-upload tool',
|
|
1115
|
+
async () => {
|
|
1116
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
1117
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Manual Upload Tool')
|
|
1118
|
+
},
|
|
1119
|
+
async () => {
|
|
1120
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
1121
|
+
name: 'Cookbook Manual Upload Tool',
|
|
1122
|
+
description: 'Manual-upload data-exchange tool — backs the client that ingests contact rows.',
|
|
1123
|
+
intent: 'data_exchange',
|
|
1124
|
+
status: 'active',
|
|
1125
|
+
datalake_id: ctx.datalakeId,
|
|
1126
|
+
data_source_id: dataSourceId,
|
|
1127
|
+
body: { tool_body_type: 'manual_upload' },
|
|
1128
|
+
})
|
|
1129
|
+
return data
|
|
1130
|
+
},
|
|
1131
|
+
)
|
|
1132
|
+
ctx.manualUploadToolId = uploadTool.id!
|
|
1133
|
+
|
|
1134
|
+
const leContract = await ctx.ensure(
|
|
1135
|
+
'lead-form legal-entity contract',
|
|
1136
|
+
async () => {
|
|
1137
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1138
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Lead Form to LegalEntity')
|
|
1139
|
+
},
|
|
1140
|
+
async () => {
|
|
1141
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1142
|
+
name: 'Cookbook Lead Form to LegalEntity',
|
|
1143
|
+
description: 'Lead-form row → LegalEntity (custom Liquid, no MDM input).',
|
|
1144
|
+
resource_type: 'legal_entity',
|
|
1145
|
+
template_config: { type: 'custom', body: leTemplate },
|
|
1146
|
+
mdm_input_config: { type: 'null' },
|
|
1147
|
+
generic_table_id: null,
|
|
1148
|
+
})
|
|
1149
|
+
return data
|
|
1150
|
+
},
|
|
1151
|
+
)
|
|
1152
|
+
interopContractId = leContract.id!
|
|
1153
|
+
```
|
|
1154
|
+
|
|
1155
|
+
## 022 — the lead-form activation client
|
|
1156
|
+
|
|
1157
|
+
One contract on one client, so nothing to sequence here — the trap that
|
|
1158
|
+
forces two clients apart only appears when a pair resolves the same
|
|
1159
|
+
subject at once.
|
|
1160
|
+
|
|
1161
|
+
```typescript
|
|
1162
|
+
const leadDac = await ctx.ensure(
|
|
1163
|
+
'lead-form activation client',
|
|
1164
|
+
async () => {
|
|
1165
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
1166
|
+
filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Lead Form DAC' }],
|
|
1167
|
+
})
|
|
1168
|
+
return (data.data ?? [])[0]
|
|
1169
|
+
},
|
|
1170
|
+
async () => {
|
|
1171
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1172
|
+
name: 'Cookbook Lead Form DAC',
|
|
1173
|
+
description: 'Manual-upload client — ingests lead-form rows as legal entities.',
|
|
1174
|
+
tool_id: ctx.manualUploadToolId,
|
|
1175
|
+
data_source_id: dataSourceId,
|
|
1176
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1177
|
+
interoperability_contract_ids: [interopContractId],
|
|
1178
|
+
})
|
|
1179
|
+
return data
|
|
1180
|
+
},
|
|
1181
|
+
)
|
|
1182
|
+
dacId = leadDac.id!
|
|
1183
|
+
ctx.leadFormDacSlug = leadDac.slug!
|
|
1184
|
+
```
|
|
1185
|
+
|
|
1186
|
+
## 023 — ingest two contacts, one with a birthday and one without
|
|
1187
|
+
|
|
1188
|
+
Two rows. One carries a `date_of_birth` so the filter passes it; the other
|
|
1189
|
+
omits it entirely so the filter must reject it. A run where both passed, or
|
|
1190
|
+
both were filtered, would mean the filter is not evaluating anything — and
|
|
1191
|
+
that is exactly the failure a walk that never runs its workflow cannot see.
|
|
1192
|
+
|
|
1193
|
+
**The date of birth is tomorrow's month-day, stamped thirty years back.**
|
|
1194
|
+
Both halves are deliberate. The past year satisfies the changeset's
|
|
1195
|
+
not-in-the-future guard on a date of birth. Tomorrow's month-day makes the
|
|
1196
|
+
greeting schedule for the *future*, which is what makes §029's
|
|
1197
|
+
fast-forward observable at all — a birthday today renders a time already
|
|
1198
|
+
passed and would dispatch on its own, proving nothing about the override.
|
|
1199
|
+
|
|
1200
|
+
The emails are stable, and the email is what the identification dedupes on,
|
|
1201
|
+
so a second run updates these same two contacts rather than adding two more.
|
|
1202
|
+
|
|
1203
|
+
```typescript
|
|
1204
|
+
const today = new Date()
|
|
1205
|
+
const tomorrow = new Date(today.getTime() + 86_400_000)
|
|
1206
|
+
const tomorrowMM = String(tomorrow.getUTCMonth() + 1).padStart(2, '0')
|
|
1207
|
+
const tomorrowDD = String(tomorrow.getUTCDate()).padStart(2, '0')
|
|
1208
|
+
const dobYear = today.getUTCFullYear() - 30
|
|
1209
|
+
|
|
1210
|
+
ctx.birthdayEmail = 'maria-birthday@cookbook.example.com'
|
|
1211
|
+
ctx.birthdayPhone = '+15550100101'
|
|
1212
|
+
|
|
1213
|
+
const birthdayRow = {
|
|
1214
|
+
submission_id: 'BD-COOKBOOK-0101',
|
|
1215
|
+
name: 'Maria Birthday',
|
|
1216
|
+
email: ctx.birthdayEmail,
|
|
1217
|
+
phone: ctx.birthdayPhone,
|
|
1218
|
+
company: '',
|
|
1219
|
+
message: 'My birthday is tomorrow',
|
|
1220
|
+
lead_source: 'cookbook_birthday',
|
|
1221
|
+
source_uri: ctx.leadSourceUri,
|
|
1222
|
+
date_of_birth: `${dobYear}-${tomorrowMM}-${tomorrowDD}`,
|
|
1223
|
+
}
|
|
1224
|
+
const noDobRow = {
|
|
1225
|
+
submission_id: 'BD-COOKBOOK-0102',
|
|
1226
|
+
name: 'Priya NoDob',
|
|
1227
|
+
email: 'priya-nodob@cookbook.example.com',
|
|
1228
|
+
phone: '+15550100102',
|
|
1229
|
+
company: '',
|
|
1230
|
+
message: 'I have no date of birth on file',
|
|
1231
|
+
lead_source: 'cookbook_birthday',
|
|
1232
|
+
source_uri: ctx.leadSourceUri,
|
|
1233
|
+
// date_of_birth deliberately absent — the filter must reject this row
|
|
1234
|
+
}
|
|
1235
|
+
|
|
1236
|
+
const bdIngest = await api.dataActivationClients.ingest(
|
|
1237
|
+
tenantSlug, datalakeSlug, ctx.leadFormDacSlug, { data: birthdayRow },
|
|
1238
|
+
)
|
|
1239
|
+
const noDobIngest = await api.dataActivationClients.ingest(
|
|
1240
|
+
tenantSlug, datalakeSlug, ctx.leadFormDacSlug, { data: noDobRow },
|
|
1241
|
+
)
|
|
1242
|
+
ctx.batchBirthday = bdIngest.data.batch_id!
|
|
1243
|
+
ctx.batchNoDob = noDobIngest.data.batch_id!
|
|
1244
|
+
```
|
|
1245
|
+
|
|
1246
|
+
## 024 — wait for both contacts to persist
|
|
1247
|
+
|
|
1248
|
+
```typescript
|
|
1249
|
+
await ctx.waitForBatches(ctx.leadFormDacSlug, [ctx.batchBirthday, ctx.batchNoDob])
|
|
1250
|
+
```
|
|
1251
|
+
|
|
1252
|
+
## 025 — run the workflow, and wait for the right thing
|
|
1253
|
+
|
|
1254
|
+
`manual_override: false` so the filter genuinely evaluates rather than
|
|
1255
|
+
being bypassed.
|
|
1256
|
+
|
|
1257
|
+
**This run never reaches a terminal status, and waiting for one would hang
|
|
1258
|
+
until the deadline.** The birthday row's action is scheduled for a date in
|
|
1259
|
+
the future, so its execution log sits at `:executing` for the life of the
|
|
1260
|
+
run — correctly. So the wait is on `total_wels`: every row has produced an
|
|
1261
|
+
execution log, which is the moment the next step can read them.
|
|
1262
|
+
|
|
1263
|
+
```typescript
|
|
1264
|
+
const bdRunResp = await api.workflows.run(tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, {
|
|
1265
|
+
sql_where_clause: `le.batch_id IN ('${ctx.batchBirthday}', '${ctx.batchNoDob}')`,
|
|
1266
|
+
mode: 'live',
|
|
1267
|
+
manual_override: false,
|
|
1268
|
+
})
|
|
1269
|
+
const bdFired = await ctx.waitForFiredRun(datalakeSlug, bdRunResp.data.workflow_run_id)
|
|
1270
|
+
ctx.bdRunLogId = bdFired.workflowRunLogId
|
|
1271
|
+
ctx.bdRunBatchId = bdFired.batchId!
|
|
1272
|
+
|
|
1273
|
+
const welDeadline = Date.now() + 120_000
|
|
1274
|
+
let totalWels = 0
|
|
1275
|
+
while (Date.now() < welDeadline) {
|
|
1276
|
+
const { data: log } = await api.workflows.batchLogs.refresh(
|
|
1277
|
+
tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, ctx.bdRunLogId,
|
|
1278
|
+
)
|
|
1279
|
+
if (log.status === 'failed') throw new Error('birthday workflow run reached :failed')
|
|
1280
|
+
totalWels = typeof log.total_wels === 'number' ? log.total_wels : 0
|
|
1281
|
+
if (totalWels >= 2) break
|
|
1282
|
+
await new Promise((r) => setTimeout(r, 1_500))
|
|
1283
|
+
}
|
|
1284
|
+
if (totalWels < 2) throw new Error(`birthday run produced only ${totalWels}/2 execution logs within 120s`)
|
|
1285
|
+
```
|
|
1286
|
+
|
|
1287
|
+
## 026 — verify the filter routed
|
|
1288
|
+
|
|
1289
|
+
The contact with a date of birth passes and its log is `:executing` (a
|
|
1290
|
+
future action is scheduled) or `:completed`. The contact without one is
|
|
1291
|
+
`:filtered` — it never reached the action at all.
|
|
1292
|
+
|
|
1293
|
+
```typescript
|
|
1294
|
+
const { data: bdLogs } = await api.workflows.workflowLogs.list(
|
|
1295
|
+
tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug,
|
|
1296
|
+
)
|
|
1297
|
+
const bdWels = (bdLogs.data ?? []).filter(
|
|
1298
|
+
(w) => (w as { batch_id?: string }).batch_id === ctx.bdRunBatchId,
|
|
1299
|
+
)
|
|
1300
|
+
if (bdWels.length !== 2) {
|
|
1301
|
+
throw new Error(`expected 2 execution logs for the birthday run, got ${bdWels.length}`)
|
|
1302
|
+
}
|
|
1303
|
+
|
|
1304
|
+
const bdByStatus: Record<string, number> = {}
|
|
1305
|
+
for (const w of bdWels) {
|
|
1306
|
+
const st = (w as { status?: string }).status ?? 'unknown'
|
|
1307
|
+
bdByStatus[st] = (bdByStatus[st] ?? 0) + 1
|
|
1308
|
+
}
|
|
1309
|
+
const bdPassed = (bdByStatus.executing ?? 0) + (bdByStatus.completed ?? 0)
|
|
1310
|
+
if (bdPassed !== 1) {
|
|
1311
|
+
throw new Error(`expected 1 pass-branch log, got ${bdPassed} — ${JSON.stringify(bdByStatus)}`)
|
|
1312
|
+
}
|
|
1313
|
+
if ((bdByStatus.filtered ?? 0) !== 1) {
|
|
1314
|
+
throw new Error(`expected 1 :filtered log (the contact with no date of birth) — ${JSON.stringify(bdByStatus)}`)
|
|
1315
|
+
}
|
|
1316
|
+
```
|
|
1317
|
+
|
|
1318
|
+
## 027 — find the contact the greeting is for
|
|
1319
|
+
|
|
1320
|
+
`workflows.execute` is addressed by dataset id, so the birthday contact's
|
|
1321
|
+
legal-entity id has to be resolved first. Search scoped to its own batch.
|
|
1322
|
+
|
|
1323
|
+
```typescript
|
|
1324
|
+
const { data: bdSearchDef } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'legal_entity', {
|
|
1325
|
+
search_query: `le.batch_id = '${ctx.batchBirthday}'`,
|
|
1326
|
+
})
|
|
1327
|
+
if (bdSearchDef.status !== 'completed') {
|
|
1328
|
+
throw new Error(`legal_entity user-search status=${bdSearchDef.status} error=${bdSearchDef.error_message ?? '(none)'}`)
|
|
1329
|
+
}
|
|
1330
|
+
const { data: bdSearch } = await api.datasets.search(tenantSlug, datalakeSlug, 'legal_entity', {
|
|
1331
|
+
userSearchId: bdSearchDef.id!,
|
|
1332
|
+
dataAccessMode: 'raw',
|
|
1333
|
+
})
|
|
1334
|
+
const birthdayLegalEntityId = (bdSearch.data?.[0] as { id?: string } | undefined)?.id
|
|
1335
|
+
if (!birthdayLegalEntityId) throw new Error('the birthday contact was not found by search')
|
|
1336
|
+
ctx.birthdayLegalEntityId = birthdayLegalEntityId
|
|
1337
|
+
```
|
|
1338
|
+
|
|
1339
|
+
## 028 — fast-forward the scheduled greeting
|
|
1340
|
+
|
|
1341
|
+
The greeting is scheduled for a date that has not arrived, so it will not
|
|
1342
|
+
dispatch during this sitting. `trigger_override: true` promotes the queued
|
|
1343
|
+
job so it runs now.
|
|
1344
|
+
|
|
1345
|
+
**What moves is the job, and only the job.** The execution log still
|
|
1346
|
+
records the year-roll-rendered `scheduled_at` it was always going to have,
|
|
1347
|
+
which is what keeps the override's blast radius to this one dispatch
|
|
1348
|
+
instead of rewriting the schedule.
|
|
1349
|
+
|
|
1350
|
+
On a repeat run the greeting has already been sent for this contact and the
|
|
1351
|
+
idempotency key refuses to send it again, so the forced log completes with
|
|
1352
|
+
nothing dispatched. Both are `:completed`; the assertion that the greeting
|
|
1353
|
+
*exists* belongs to §029, where it is true either way.
|
|
1354
|
+
|
|
1355
|
+
```typescript
|
|
1356
|
+
const { data: execResp } = await api.workflows.execute(tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, {
|
|
1357
|
+
dataset_id: ctx.birthdayLegalEntityId,
|
|
1358
|
+
decision_key: 'send_happy_birthday_sms',
|
|
1359
|
+
manual_override: true,
|
|
1360
|
+
trigger_override: true,
|
|
1361
|
+
})
|
|
1362
|
+
ctx.forcedWelId = execResp.workflow_execution_log_id!
|
|
1363
|
+
|
|
1364
|
+
const forcedDeadline = Date.now() + 120_000
|
|
1365
|
+
let forcedStatus: string | null = null
|
|
1366
|
+
while (Date.now() < forcedDeadline) {
|
|
1367
|
+
const { data: wel } = await api.workflows.workflowLogs.get(
|
|
1368
|
+
tenantSlug, datalakeSlug, ctx.birthdayWorkflowSlug, ctx.forcedWelId,
|
|
1369
|
+
)
|
|
1370
|
+
forcedStatus = (wel as { status?: string }).status ?? null
|
|
1371
|
+
if (forcedStatus && forcedStatus !== 'pending' && forcedStatus !== 'executing') break
|
|
1372
|
+
await new Promise((r) => setTimeout(r, 2_000))
|
|
1373
|
+
}
|
|
1374
|
+
if (forcedStatus !== 'completed') {
|
|
1375
|
+
throw new Error(`the fast-forwarded log did not reach :completed (last status: ${forcedStatus})`)
|
|
1376
|
+
}
|
|
1377
|
+
```
|
|
1378
|
+
|
|
1379
|
+
## 029 — the greeting exists, and it carries a deep link
|
|
1380
|
+
|
|
1381
|
+
The end state: a birthday greeting exists for this contact, with a
|
|
1382
|
+
`/t/<token>` link in its body. True on the first run because the action
|
|
1383
|
+
just dispatched, and true on every run after because it dispatched once and
|
|
1384
|
+
the idempotency key has protected it since.
|
|
1385
|
+
|
|
1386
|
+
**Read the durable record, not the run you just made.** The obvious move is
|
|
1387
|
+
to pull the rendered body off the execution log from §028 — and that works
|
|
1388
|
+
exactly once. On a repeat run that log's action is skipped and carries no
|
|
1389
|
+
body at all, so the walk goes red having proved nothing except that
|
|
1390
|
+
idempotency works. The `message` row persists in the lake and is the
|
|
1391
|
+
answer to "does this contact have a greeting", which is the question worth
|
|
1392
|
+
asking. In local dev the publish itself lands in LocalStack's SNS records
|
|
1393
|
+
at `/_aws/sns/sms-messages`, keyed by phone number — useful to eyeball, too
|
|
1394
|
+
volatile to assert on.
|
|
1395
|
+
|
|
1396
|
+
```typescript
|
|
1397
|
+
const { data: bdMsgSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
|
|
1398
|
+
search_query: `m.idempotency_key LIKE '${ctx.birthdayLegalEntityId}-%'`,
|
|
1399
|
+
})
|
|
1400
|
+
if (bdMsgSearch.status !== 'completed') {
|
|
1401
|
+
throw new Error(`message user-search status=${bdMsgSearch.status} error=${bdMsgSearch.error_message ?? '(none)'}`)
|
|
1402
|
+
}
|
|
1403
|
+
|
|
1404
|
+
const bdMsgDeadline = Date.now() + 45_000
|
|
1405
|
+
let bdMessages: Array<Record<string, unknown>> = []
|
|
1406
|
+
while (Date.now() < bdMsgDeadline && bdMessages.length === 0) {
|
|
1407
|
+
const { data } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
|
|
1408
|
+
userSearchId: bdMsgSearch.id!,
|
|
1409
|
+
dataAccessMode: 'raw',
|
|
1410
|
+
})
|
|
1411
|
+
bdMessages = (data.data ?? []) as Array<Record<string, unknown>>
|
|
1412
|
+
if (bdMessages.length === 0) await new Promise((r) => setTimeout(r, 1_000))
|
|
1413
|
+
}
|
|
1414
|
+
if (bdMessages.length === 0) {
|
|
1415
|
+
throw new Error(
|
|
1416
|
+
`no birthday greeting for contact ${ctx.birthdayLegalEntityId} within 45s — ` +
|
|
1417
|
+
`the action never fired, or it fired under a different idempotency key`,
|
|
1418
|
+
)
|
|
1419
|
+
}
|
|
1420
|
+
|
|
1421
|
+
const greeting = bdMessages
|
|
1422
|
+
.map((m) => String(m.body ?? ''))
|
|
1423
|
+
.find((b) => b.includes('Happy birthday') && b.includes('/t/'))
|
|
1424
|
+
if (!greeting) {
|
|
1425
|
+
throw new Error(`no greeting body with a /t/ link — bodies: ${JSON.stringify(bdMessages.map((m) => m.body))}`)
|
|
1426
|
+
}
|
|
1427
|
+
const bdToken = greeting.match(/\/t\/([A-Za-z0-9_-]+)/)
|
|
1428
|
+
if (!bdToken) throw new Error(`no /t/<token> in the rendered greeting: ${greeting}`)
|
|
1429
|
+
ctx.birthdayShortPath = bdToken[1]!
|
|
1430
|
+
```
|
|
1431
|
+
|
|
1432
|
+
## 030 — resolve the link, and close the loop
|
|
1433
|
+
|
|
1434
|
+
The shortlink resolves through `connectedApps.resolvePage`, and the
|
|
1435
|
+
`route_path` that comes back must match the action's `connected_app_route`.
|
|
1436
|
+
Posting `opened_at` and `form_submitted_at` then mirrors exactly what the
|
|
1437
|
+
form's frontend does when the recipient opens it — which closes the loop
|
|
1438
|
+
from a scheduled greeting to a tracked reply.
|
|
1439
|
+
|
|
1440
|
+
```typescript
|
|
1441
|
+
const { data: bdResolved } = await api.connectedApps.resolvePage(
|
|
1442
|
+
tenantSlug, datalakeSlug, ctx.birthdayAppSlug,
|
|
1443
|
+
{ short_path: ctx.birthdayShortPath, user_agent: 'cookbook-doctest/birthday-greeting' },
|
|
1444
|
+
)
|
|
1445
|
+
if (bdResolved.route_path !== '/forms/birthday-greeting') {
|
|
1446
|
+
throw new Error(`resolvePage route_path mismatch: ${bdResolved.route_path}`)
|
|
1447
|
+
}
|
|
1448
|
+
|
|
1449
|
+
const bdNow = new Date().toISOString()
|
|
1450
|
+
const { data: bdTracked } = await api.connectedApps.updateMessageTracking(
|
|
1451
|
+
tenantSlug, datalakeSlug, ctx.birthdayAppSlug,
|
|
1452
|
+
{ short_path: ctx.birthdayShortPath, opened_at: bdNow, form_submitted_at: bdNow },
|
|
1453
|
+
)
|
|
1454
|
+
if (!bdTracked.message?.opened_at || !bdTracked.message?.form_submitted_at) {
|
|
1455
|
+
throw new Error('message tracking did not persist opened_at + form_submitted_at')
|
|
1456
|
+
}
|
|
1457
|
+
```
|
|
1458
|
+
|
|
1459
|
+
## 031 — the roster and the audience, as two tables
|
|
1460
|
+
|
|
1461
|
+
A campaign needs two rosters and they are not the same thing. **Direct
|
|
1462
|
+
customers** are the businesses running campaigns; **end customers** are the
|
|
1463
|
+
people a campaign may contact. The platform models neither, because which
|
|
1464
|
+
businesses you serve is domain data.
|
|
1465
|
+
|
|
1466
|
+
Three columns on the audience carry the entire business rule.
|
|
1467
|
+
`suppressed` is do-not-contact — the workflow *filter* reads it, so a
|
|
1468
|
+
suppressed person is unselectable by any campaign rather than merely hidden
|
|
1469
|
+
in some screen. `phone` and `email` are the two channel destinations, and
|
|
1470
|
+
their presence is the reachability gate. `bucket` is the A/B assignment,
|
|
1471
|
+
stamped at ingest and split on by the decision node.
|
|
1472
|
+
|
|
1473
|
+
Note which columns are masked and which are not: the three PII columns are
|
|
1474
|
+
`tokenize`, while `suppressed` and `bucket` are `none`. Campaign mechanics
|
|
1475
|
+
are not personal data, and the workflow has to read them.
|
|
1476
|
+
|
|
1477
|
+
```typescript
|
|
1478
|
+
const rosterTable = await ctx.ensure(
|
|
1479
|
+
'Direct Customers table',
|
|
1480
|
+
async () => {
|
|
1481
|
+
const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
|
|
1482
|
+
return (data.data ?? []).find((t: { title?: string }) => t.title === 'Cookbook Direct Customers')
|
|
1483
|
+
},
|
|
1484
|
+
async () => {
|
|
1485
|
+
const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
|
|
1486
|
+
title: 'Cookbook Direct Customers',
|
|
1487
|
+
description: 'The businesses running marketing campaigns.',
|
|
1488
|
+
columns: [
|
|
1489
|
+
{ name: 'direct_customer_id', title: 'Direct Customer ID', type: 'string', description: 'Vendor-supplied unique business id', is_unique: true, privacy_requirement: 'none' },
|
|
1490
|
+
{ name: 'business_name', title: 'Business Name', type: 'string', description: 'Business display name (public record, not PII)', is_unique: false, privacy_requirement: 'none' },
|
|
1491
|
+
{ name: 'sender_phone', title: 'Sender Phone', type: 'string', description: 'Shared number campaigns send SMS from', is_unique: false, privacy_requirement: 'none' },
|
|
1492
|
+
{ name: 'sender_domain', title: 'Sender Domain', type: 'string', description: 'Shared domain campaigns send email from', is_unique: false, privacy_requirement: 'none' },
|
|
1493
|
+
],
|
|
1494
|
+
})
|
|
1495
|
+
return data
|
|
1496
|
+
},
|
|
1497
|
+
)
|
|
1498
|
+
ctx.rosterTableId = rosterTable.id!
|
|
1499
|
+
ctx.rosterTableName = rosterTable.name!
|
|
1500
|
+
|
|
1501
|
+
const audienceTable = await ctx.ensure(
|
|
1502
|
+
'End Customers table',
|
|
1503
|
+
async () => {
|
|
1504
|
+
const { data } = await api.genericTables.list(tenantSlug, datalakeSlug)
|
|
1505
|
+
return (data.data ?? []).find((t: { title?: string }) => t.title === 'Cookbook End Customers')
|
|
1506
|
+
},
|
|
1507
|
+
async () => {
|
|
1508
|
+
const { data } = await api.genericTables.create(tenantSlug, datalakeSlug, {
|
|
1509
|
+
title: 'Cookbook End Customers',
|
|
1510
|
+
description: 'The people a campaign may contact.',
|
|
1511
|
+
columns: [
|
|
1512
|
+
{ name: 'end_customer_id', title: 'End Customer ID', type: 'string', description: 'Vendor-supplied unique end-customer id', is_unique: true, privacy_requirement: 'none' },
|
|
1513
|
+
{ name: 'direct_customer_id', title: 'Direct Customer ID', type: 'string', description: 'The business this end customer belongs to', is_unique: false, privacy_requirement: 'none' },
|
|
1514
|
+
{ name: 'name', title: 'Name', type: 'string', description: 'Recipient full name', is_unique: false, privacy_requirement: 'tokenize' },
|
|
1515
|
+
{ name: 'email', title: 'Email', type: 'string', description: 'Recipient email — the email-channel destination', is_unique: false, privacy_requirement: 'tokenize' },
|
|
1516
|
+
{ name: 'phone', title: 'Phone', type: 'string', description: 'Recipient phone — the SMS-channel destination', is_unique: false, privacy_requirement: 'tokenize' },
|
|
1517
|
+
{ name: 'suppressed', title: 'Suppressed', type: 'boolean', description: 'Do-not-contact flag — a suppressed recipient is never selected', is_unique: false, privacy_requirement: 'none' },
|
|
1518
|
+
{ name: 'bucket', title: 'Bucket', type: 'integer', description: 'Stable 0-99 A/B bucket assigned at ingest; the decision node splits on it', is_unique: false, privacy_requirement: 'none' },
|
|
1519
|
+
],
|
|
1520
|
+
})
|
|
1521
|
+
return data
|
|
1522
|
+
},
|
|
1523
|
+
)
|
|
1524
|
+
ctx.audienceTableId = audienceTable.id!
|
|
1525
|
+
ctx.audienceTableName = audienceTable.name!
|
|
1526
|
+
|
|
1527
|
+
await ctx.deployGenericTable(ctx.rosterTableName)
|
|
1528
|
+
await ctx.deployGenericTable(ctx.audienceTableName)
|
|
1529
|
+
```
|
|
1530
|
+
|
|
1531
|
+
## 032 — the inbound-messages table, which you must not create
|
|
1532
|
+
|
|
1533
|
+
Replies land in `inbound_messages`. It is a **system** table: it ships with
|
|
1534
|
+
every datalake, its `type` is `'system'` rather than `'custom'`, and nobody
|
|
1535
|
+
creates it. You find it by name.
|
|
1536
|
+
|
|
1537
|
+
Trying to create it is the mistake this step exists to prevent — the name is
|
|
1538
|
+
reserved and the physical table is already there, so the attempt fails in a
|
|
1539
|
+
way that reads like a permissions problem.
|
|
1540
|
+
|
|
1541
|
+
```typescript
|
|
1542
|
+
const { data: allTables } = await api.genericTables.list(tenantSlug, datalakeSlug)
|
|
1543
|
+
const inbound = (allTables.data ?? []).find((t: { name?: string }) => t.name === 'inbound_messages')
|
|
1544
|
+
if (!inbound) {
|
|
1545
|
+
throw new Error('inbound_messages must ship with every datalake')
|
|
1546
|
+
}
|
|
1547
|
+
if (inbound.type !== 'system') {
|
|
1548
|
+
throw new Error(`inbound_messages must be a system table, got type=${inbound.type}`)
|
|
1549
|
+
}
|
|
1550
|
+
ctx.inboundTableId = inbound.id!
|
|
1551
|
+
ctx.inboundTableName = inbound.name!
|
|
1552
|
+
```
|
|
1553
|
+
|
|
1554
|
+
## 033 — the roster contracts, on two clients
|
|
1555
|
+
|
|
1556
|
+
Each roster row becomes two things: a row in the Direct Customers table,
|
|
1557
|
+
and a **business** legal entity. That is the contract pair — and the pair
|
|
1558
|
+
goes on **two clients**, for the reason §020 of `payments-compliance`
|
|
1559
|
+
sets out in full and this step restates because it is the single most
|
|
1560
|
+
expensive thing to learn twice.
|
|
1561
|
+
|
|
1562
|
+
A client fans out per row × contract and the job key includes the contract
|
|
1563
|
+
id, so a pair bound to one client runs both contracts **at the same
|
|
1564
|
+
instant**. Both find-or-create the same subject, both miss the dedupe read,
|
|
1565
|
+
both insert, and the unique index on `(uri, id_type, id_number)` refuses the
|
|
1566
|
+
loser. It works on every run except the first.
|
|
1567
|
+
|
|
1568
|
+
**And note the identifier shape in the MDM template**: `uri` / `type` /
|
|
1569
|
+
`value` — not `id_type` / `id_number` / `uri`, which is what the legal-entity
|
|
1570
|
+
body above takes. Three shapes name one idea on this platform, and they
|
|
1571
|
+
overlap unevenly. `system` — what `POST /mdm/verify` takes — is a genuine
|
|
1572
|
+
alias here: `MDMInput` casts it and normalises it onto `uri`, so a template
|
|
1573
|
+
using it resolves correctly. `id_type` and `id_number` are not cast at all,
|
|
1574
|
+
and unknown keys are dropped rather than refused — so borrowing the
|
|
1575
|
+
legal-entity names for the MDM side resolves nothing and fails as a bare
|
|
1576
|
+
`mdm_dispatcher_error` naming no field.
|
|
1577
|
+
|
|
1578
|
+
And the reason the two look interchangeable is that **one becomes the
|
|
1579
|
+
other**. `MDMInput.to_identification/1` takes `(uri, type, value)` and
|
|
1580
|
+
returns `(uri, id_type, id_number)` — the rename happens inside the platform,
|
|
1581
|
+
on the way through. So the identifier reads back under names it will not
|
|
1582
|
+
accept on the way in, and that asymmetry is invisible from the stored shape
|
|
1583
|
+
alone.
|
|
1584
|
+
|
|
1585
|
+
```typescript
|
|
1586
|
+
const ROSTER_LE = `{% assign p = msg %}
|
|
1587
|
+
{
|
|
1588
|
+
"legal_entity_type": "business",
|
|
1589
|
+
"role": "direct",
|
|
1590
|
+
"business_name": "{{ p.business_name | json_escape }}",
|
|
1591
|
+
"identifications": [
|
|
1592
|
+
{"id_type": "digital_identifier", "uri": "{{ p.source_uri | json_escape }}", "id_number": "{{ p.direct_customer_id | json_escape }}"}
|
|
1593
|
+
]
|
|
1594
|
+
}`
|
|
1595
|
+
|
|
1596
|
+
const ROSTER_GT = `{% assign p = msg %}
|
|
1597
|
+
{
|
|
1598
|
+
"direct_customer_id": "{{ p.direct_customer_id | json_escape }}",
|
|
1599
|
+
"business_name": "{{ p.business_name | json_escape }}",
|
|
1600
|
+
"sender_phone": "{{ p.sender_phone | json_escape }}",
|
|
1601
|
+
"sender_domain": "{{ p.sender_domain | json_escape }}"
|
|
1602
|
+
}`
|
|
1603
|
+
|
|
1604
|
+
const ROSTER_MDM = `{% assign p = msg %}
|
|
1605
|
+
{
|
|
1606
|
+
"legal_entity_type": "business",
|
|
1607
|
+
"business_name": "{{ p.business_name | json_escape }}",
|
|
1608
|
+
"identifiers": [
|
|
1609
|
+
{"uri": "{{ p.source_uri | json_escape }}", "type": "digital_identifier", "value": "{{ p.direct_customer_id | json_escape }}"}
|
|
1610
|
+
]
|
|
1611
|
+
}`
|
|
1612
|
+
|
|
1613
|
+
const rosterLe = await ctx.ensure(
|
|
1614
|
+
'roster business LE contract',
|
|
1615
|
+
async () => {
|
|
1616
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1617
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Roster Business LE')
|
|
1618
|
+
},
|
|
1619
|
+
async () => {
|
|
1620
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1621
|
+
name: 'Cookbook Roster Business LE',
|
|
1622
|
+
description: 'Roster row → business LegalEntity.',
|
|
1623
|
+
resource_type: 'legal_entity',
|
|
1624
|
+
type: 'identity',
|
|
1625
|
+
generic_table_id: null,
|
|
1626
|
+
template_config: { type: 'custom', body: ROSTER_LE },
|
|
1627
|
+
mdm_input_config: { type: 'null' },
|
|
1628
|
+
})
|
|
1629
|
+
return data
|
|
1630
|
+
},
|
|
1631
|
+
)
|
|
1632
|
+
|
|
1633
|
+
const rosterGt = await ctx.ensure(
|
|
1634
|
+
'roster generic-table contract',
|
|
1635
|
+
async () => {
|
|
1636
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1637
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Roster GT')
|
|
1638
|
+
},
|
|
1639
|
+
async () => {
|
|
1640
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1641
|
+
name: 'Cookbook Roster GT',
|
|
1642
|
+
description: 'Roster row → Direct Customers row, stamped with its business subject.',
|
|
1643
|
+
resource_type: 'generic_table',
|
|
1644
|
+
type: 'identity',
|
|
1645
|
+
generic_table_id: ctx.rosterTableId,
|
|
1646
|
+
template_config: { type: 'custom', body: ROSTER_GT },
|
|
1647
|
+
mdm_input_config: { type: 'custom', body: ROSTER_MDM },
|
|
1648
|
+
})
|
|
1649
|
+
return data
|
|
1650
|
+
},
|
|
1651
|
+
)
|
|
1652
|
+
|
|
1653
|
+
const rosterIdentityDac = await ctx.ensure(
|
|
1654
|
+
'roster identity client',
|
|
1655
|
+
async () => {
|
|
1656
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
1657
|
+
filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Roster Identity Client' }],
|
|
1658
|
+
})
|
|
1659
|
+
return (data.data ?? [])[0]
|
|
1660
|
+
},
|
|
1661
|
+
async () => {
|
|
1662
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1663
|
+
name: 'Cookbook Roster Identity Client',
|
|
1664
|
+
description: 'Writes each business as a legal entity, ahead of its roster row.',
|
|
1665
|
+
tool_id: ctx.manualUploadToolId,
|
|
1666
|
+
data_source_id: dataSourceId,
|
|
1667
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1668
|
+
interoperability_contract_ids: [rosterLe.id!],
|
|
1669
|
+
})
|
|
1670
|
+
return data
|
|
1671
|
+
},
|
|
1672
|
+
)
|
|
1673
|
+
ctx.rosterIdentityDacSlug = rosterIdentityDac.slug!
|
|
1674
|
+
|
|
1675
|
+
const rosterDataDac = await ctx.ensure(
|
|
1676
|
+
'roster data client',
|
|
1677
|
+
async () => {
|
|
1678
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
1679
|
+
filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Roster Data Client' }],
|
|
1680
|
+
})
|
|
1681
|
+
return (data.data ?? [])[0]
|
|
1682
|
+
},
|
|
1683
|
+
async () => {
|
|
1684
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1685
|
+
name: 'Cookbook Roster Data Client',
|
|
1686
|
+
description: 'Writes the roster row, resolving the business the identity client wrote.',
|
|
1687
|
+
tool_id: ctx.manualUploadToolId,
|
|
1688
|
+
data_source_id: dataSourceId,
|
|
1689
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1690
|
+
interoperability_contract_ids: [rosterGt.id!],
|
|
1691
|
+
})
|
|
1692
|
+
return data
|
|
1693
|
+
},
|
|
1694
|
+
)
|
|
1695
|
+
ctx.rosterDataDacSlug = rosterDataDac.slug!
|
|
1696
|
+
```
|
|
1697
|
+
|
|
1698
|
+
## 034 — the audience contracts, on two clients
|
|
1699
|
+
|
|
1700
|
+
Same shape, two differences that carry the campaign.
|
|
1701
|
+
|
|
1702
|
+
Each recipient resolves to an **individual** legal entity keyed on their
|
|
1703
|
+
contact handle — phone when there is one, email otherwise. That subject is
|
|
1704
|
+
what the per-recipient short link in §039 is minted against, which is what
|
|
1705
|
+
makes a click attributable to one person rather than to the campaign.
|
|
1706
|
+
|
|
1707
|
+
And the generic-table contract assigns the **bucket**, derived from the last
|
|
1708
|
+
two digits of the vendor's own id rather than from a random draw. Re-ingest
|
|
1709
|
+
the same customer tomorrow and they land in the same bucket. An A/B split
|
|
1710
|
+
that reshuffles on every ingest is not an A/B split — it is noise with a
|
|
1711
|
+
report attached.
|
|
1712
|
+
|
|
1713
|
+
```typescript
|
|
1714
|
+
const AUDIENCE_LE = `{% assign p = msg %}
|
|
1715
|
+
{% assign name_parts = p.name | default: "" | split: " " %}
|
|
1716
|
+
{% assign first_name = name_parts[0] | default: "" %}
|
|
1717
|
+
{% assign last_name_parts = name_parts | slice: 1, 9 %}
|
|
1718
|
+
{% assign last_name = last_name_parts | join: " " %}
|
|
1719
|
+
{% if p.phone and p.phone != "" %}{% assign handle = p.phone %}{% else %}{% assign handle = p.email %}{% endif %}
|
|
1720
|
+
{
|
|
1721
|
+
"legal_entity_type": "individual",
|
|
1722
|
+
"role": "direct",
|
|
1723
|
+
{% if first_name != "" %}"first_name": "{{ first_name | json_escape }}",{% endif %}
|
|
1724
|
+
{% if last_name != "" %}"last_name": "{{ last_name | json_escape }}",{% endif %}
|
|
1725
|
+
{% if p.phone and p.phone != "" %}"phone_numbers": [{ "phone_number": "{{ p.phone | json_escape }}" }],{% endif %}
|
|
1726
|
+
"identifications": [
|
|
1727
|
+
{"id_type": "digital_identifier", "uri": "{{ p.source_uri | json_escape }}", "id_number": "{{ handle | json_escape }}"}
|
|
1728
|
+
]
|
|
1729
|
+
}`
|
|
1730
|
+
|
|
1731
|
+
const AUDIENCE_GT = `{% assign p = msg %}
|
|
1732
|
+
{% assign bucket = p.end_customer_id | slice: -2, 2 | plus: 0 %}
|
|
1733
|
+
{
|
|
1734
|
+
"end_customer_id": "{{ p.end_customer_id | json_escape }}",
|
|
1735
|
+
"direct_customer_id": "{{ p.direct_customer_id | json_escape }}",
|
|
1736
|
+
"name": "{{ p.name | default: "" | json_escape }}",
|
|
1737
|
+
"email": "{{ p.email | default: "" | json_escape }}",
|
|
1738
|
+
"phone": "{{ p.phone | default: "" | json_escape }}",
|
|
1739
|
+
"suppressed": {% if p.suppressed %}true{% else %}false{% endif %},
|
|
1740
|
+
"bucket": {{ bucket }}
|
|
1741
|
+
}`
|
|
1742
|
+
|
|
1743
|
+
const AUDIENCE_MDM = `{% assign p = msg %}
|
|
1744
|
+
{% assign name_parts = p.name | default: "" | split: " " %}
|
|
1745
|
+
{% assign first_name = name_parts[0] | default: "" %}
|
|
1746
|
+
{% assign last_name_parts = name_parts | slice: 1, 9 %}
|
|
1747
|
+
{% assign last_name = last_name_parts | join: " " %}
|
|
1748
|
+
{% if p.phone and p.phone != "" %}{% assign handle = p.phone %}{% else %}{% assign handle = p.email %}{% endif %}
|
|
1749
|
+
{
|
|
1750
|
+
"legal_entity_type": "individual",
|
|
1751
|
+
{% if first_name != "" %}"first_name": "{{ first_name | json_escape }}",{% endif %}
|
|
1752
|
+
{% if last_name != "" %}"last_name": "{{ last_name | json_escape }}",{% endif %}
|
|
1753
|
+
"identifiers": [
|
|
1754
|
+
{"uri": "{{ p.source_uri | json_escape }}", "type": "digital_identifier", "value": "{{ handle | json_escape }}"}
|
|
1755
|
+
]
|
|
1756
|
+
}`
|
|
1757
|
+
|
|
1758
|
+
const audienceLe = await ctx.ensure(
|
|
1759
|
+
'audience LE contract',
|
|
1760
|
+
async () => {
|
|
1761
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1762
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Audience LE')
|
|
1763
|
+
},
|
|
1764
|
+
async () => {
|
|
1765
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1766
|
+
name: 'Cookbook Audience LE',
|
|
1767
|
+
description: 'Audience row → end-customer LegalEntity, keyed on the contact handle.',
|
|
1768
|
+
resource_type: 'legal_entity',
|
|
1769
|
+
type: 'identity',
|
|
1770
|
+
generic_table_id: null,
|
|
1771
|
+
template_config: { type: 'custom', body: AUDIENCE_LE },
|
|
1772
|
+
mdm_input_config: { type: 'null' },
|
|
1773
|
+
})
|
|
1774
|
+
return data
|
|
1775
|
+
},
|
|
1776
|
+
)
|
|
1777
|
+
|
|
1778
|
+
const audienceGt = await ctx.ensure(
|
|
1779
|
+
'audience generic-table contract',
|
|
1780
|
+
async () => {
|
|
1781
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
1782
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Audience GT')
|
|
1783
|
+
},
|
|
1784
|
+
async () => {
|
|
1785
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
1786
|
+
name: 'Cookbook Audience GT',
|
|
1787
|
+
description: 'Audience row → End Customers row, stamped with its subject and its stable A/B bucket.',
|
|
1788
|
+
resource_type: 'generic_table',
|
|
1789
|
+
type: 'identity',
|
|
1790
|
+
generic_table_id: ctx.audienceTableId,
|
|
1791
|
+
template_config: { type: 'custom', body: AUDIENCE_GT },
|
|
1792
|
+
mdm_input_config: { type: 'custom', body: AUDIENCE_MDM },
|
|
1793
|
+
})
|
|
1794
|
+
return data
|
|
1795
|
+
},
|
|
1796
|
+
)
|
|
1797
|
+
|
|
1798
|
+
const audienceIdentityDac = await ctx.ensure(
|
|
1799
|
+
'audience identity client',
|
|
1800
|
+
async () => {
|
|
1801
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
1802
|
+
filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Audience Identity Client' }],
|
|
1803
|
+
})
|
|
1804
|
+
return (data.data ?? [])[0]
|
|
1805
|
+
},
|
|
1806
|
+
async () => {
|
|
1807
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1808
|
+
name: 'Cookbook Audience Identity Client',
|
|
1809
|
+
description: 'Writes each recipient as a legal entity, ahead of their audience row.',
|
|
1810
|
+
tool_id: ctx.manualUploadToolId,
|
|
1811
|
+
data_source_id: dataSourceId,
|
|
1812
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1813
|
+
interoperability_contract_ids: [audienceLe.id!],
|
|
1814
|
+
})
|
|
1815
|
+
return data
|
|
1816
|
+
},
|
|
1817
|
+
)
|
|
1818
|
+
ctx.audienceIdentityDacSlug = audienceIdentityDac.slug!
|
|
1819
|
+
|
|
1820
|
+
const audienceDataDac = await ctx.ensure(
|
|
1821
|
+
'audience data client',
|
|
1822
|
+
async () => {
|
|
1823
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
1824
|
+
filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Audience Data Client' }],
|
|
1825
|
+
})
|
|
1826
|
+
return (data.data ?? [])[0]
|
|
1827
|
+
},
|
|
1828
|
+
async () => {
|
|
1829
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
1830
|
+
name: 'Cookbook Audience Data Client',
|
|
1831
|
+
description: 'Writes the audience row, resolving the recipient the identity client wrote.',
|
|
1832
|
+
tool_id: ctx.manualUploadToolId,
|
|
1833
|
+
data_source_id: dataSourceId,
|
|
1834
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
1835
|
+
interoperability_contract_ids: [audienceGt.id!],
|
|
1836
|
+
})
|
|
1837
|
+
return data
|
|
1838
|
+
},
|
|
1839
|
+
)
|
|
1840
|
+
ctx.audienceDataDacSlug = audienceDataDac.slug!
|
|
1841
|
+
```
|
|
1842
|
+
|
|
1843
|
+
## 035 — ingest one business and four recipients, identities first
|
|
1844
|
+
|
|
1845
|
+
Four recipients chosen so a single run exercises every branch at once:
|
|
1846
|
+
|
|
1847
|
+
| Recipient | phone | email | suppressed | bucket | expected |
|
|
1848
|
+
|-----------|-------|-------|------------|--------|----------|
|
|
1849
|
+
| Ada | yes | yes | no | 10 | sent, variant A |
|
|
1850
|
+
| Grace | yes | yes | no | 80 | sent, variant B |
|
|
1851
|
+
| Sup | yes | yes | **yes** | 31 | filtered — suppressed |
|
|
1852
|
+
| Nophone | — | yes | no | 42 | filtered — unreachable |
|
|
1853
|
+
|
|
1854
|
+
The buckets are not set by hand. They fall out of the ids ending `-10`,
|
|
1855
|
+
`-80`, `-31`, `-42`, which is §034's Liquid doing the assignment.
|
|
1856
|
+
|
|
1857
|
+
Identities go first and are waited on, then the rows. Within a stage the
|
|
1858
|
+
four recipients have four different handles, so they never contend with each
|
|
1859
|
+
other — the ordering that matters is between the stages, not inside them.
|
|
1860
|
+
|
|
1861
|
+
```typescript
|
|
1862
|
+
ctx.rosterId = 'DC-COOKBOOK-01'
|
|
1863
|
+
ctx.adaId = 'EC-COOKBOOK-10'
|
|
1864
|
+
ctx.graceId = 'EC-COOKBOOK-80'
|
|
1865
|
+
ctx.supId = 'EC-COOKBOOK-31'
|
|
1866
|
+
ctx.nophoneId = 'EC-COOKBOOK-42'
|
|
1867
|
+
ctx.adaPhone = '+15550200010'
|
|
1868
|
+
|
|
1869
|
+
const rosterRow = {
|
|
1870
|
+
direct_customer_id: ctx.rosterId,
|
|
1871
|
+
business_name: 'Bloom Salon',
|
|
1872
|
+
sender_phone: '+15550001111',
|
|
1873
|
+
sender_domain: 'bloom-salon.example.com',
|
|
1874
|
+
}
|
|
1875
|
+
const audienceRows = [
|
|
1876
|
+
{ end_customer_id: ctx.adaId, direct_customer_id: ctx.rosterId, name: 'Ada Lovelace', email: 'ada@cookbook.example.com', phone: ctx.adaPhone, suppressed: false },
|
|
1877
|
+
{ end_customer_id: ctx.graceId, direct_customer_id: ctx.rosterId, name: 'Grace Hopper', email: 'grace@cookbook.example.com', phone: '+15550200080', suppressed: false },
|
|
1878
|
+
{ end_customer_id: ctx.supId, direct_customer_id: ctx.rosterId, name: 'Sup Pressed', email: 'sup@cookbook.example.com', phone: '+15550200031', suppressed: true },
|
|
1879
|
+
{ end_customer_id: ctx.nophoneId, direct_customer_id: ctx.rosterId, name: 'No Phone', email: 'nophone@cookbook.example.com', phone: '', suppressed: false },
|
|
1880
|
+
]
|
|
1881
|
+
|
|
1882
|
+
// Stage 1 — the business, then the people.
|
|
1883
|
+
const rosterIdent = await api.dataActivationClients.ingest(
|
|
1884
|
+
tenantSlug, datalakeSlug, ctx.rosterIdentityDacSlug, { data: rosterRow },
|
|
1885
|
+
)
|
|
1886
|
+
await ctx.waitForBatches(ctx.rosterIdentityDacSlug, [rosterIdent.data.batch_id!])
|
|
1887
|
+
|
|
1888
|
+
const audienceIdentBatches: string[] = []
|
|
1889
|
+
for (const row of audienceRows) {
|
|
1890
|
+
const r = await api.dataActivationClients.ingest(
|
|
1891
|
+
tenantSlug, datalakeSlug, ctx.audienceIdentityDacSlug, { data: row },
|
|
1892
|
+
)
|
|
1893
|
+
audienceIdentBatches.push(r.data.batch_id!)
|
|
1894
|
+
}
|
|
1895
|
+
await ctx.waitForBatches(ctx.audienceIdentityDacSlug, audienceIdentBatches)
|
|
1896
|
+
|
|
1897
|
+
// Stage 2 — the rows, which resolve the subjects that now exist.
|
|
1898
|
+
const rosterData = await api.dataActivationClients.ingest(
|
|
1899
|
+
tenantSlug, datalakeSlug, ctx.rosterDataDacSlug, { data: rosterRow },
|
|
1900
|
+
)
|
|
1901
|
+
await ctx.waitForBatches(ctx.rosterDataDacSlug, [rosterData.data.batch_id!])
|
|
1902
|
+
|
|
1903
|
+
const audienceDataBatches: string[] = []
|
|
1904
|
+
for (const row of audienceRows) {
|
|
1905
|
+
const r = await api.dataActivationClients.ingest(
|
|
1906
|
+
tenantSlug, datalakeSlug, ctx.audienceDataDacSlug, { data: row },
|
|
1907
|
+
)
|
|
1908
|
+
audienceDataBatches.push(r.data.batch_id!)
|
|
1909
|
+
}
|
|
1910
|
+
await ctx.waitForBatches(ctx.audienceDataDacSlug, audienceDataBatches)
|
|
1911
|
+
```
|
|
1912
|
+
|
|
1913
|
+
## 036 — the stamp is the proof that resolution ran
|
|
1914
|
+
|
|
1915
|
+
Read the four rows back. Every one carries a `legal_entity_id`, and the four
|
|
1916
|
+
are **distinct** — four recipients, four subjects. A shared subject would
|
|
1917
|
+
mean two people collapsed into one, and a short link pointing at the wrong
|
|
1918
|
+
customer.
|
|
1919
|
+
|
|
1920
|
+
The buckets are checked here too, because a wrong bucket is the kind of
|
|
1921
|
+
thing that stays invisible until someone asks why variant A outperformed B
|
|
1922
|
+
by a suspiciously round margin.
|
|
1923
|
+
|
|
1924
|
+
```typescript
|
|
1925
|
+
const rowsDeadline = Date.now() + 120_000
|
|
1926
|
+
let audienceRowsRead: Array<Record<string, unknown>> = []
|
|
1927
|
+
while (Date.now() < rowsDeadline && audienceRowsRead.length < 4) {
|
|
1928
|
+
const { data: result } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
|
|
1929
|
+
sql: `SELECT end_customer_id, bucket, legal_entity_id
|
|
1930
|
+
FROM ${ctx.audienceTableName}
|
|
1931
|
+
WHERE end_customer_id LIKE 'EC-COOKBOOK-%'`,
|
|
1932
|
+
mode: 'raw',
|
|
1933
|
+
})
|
|
1934
|
+
audienceRowsRead = (result.data ?? []).map((r) =>
|
|
1935
|
+
Object.fromEntries(result.meta.columns.map((c, i) => [c, r[i]])),
|
|
1936
|
+
)
|
|
1937
|
+
if (audienceRowsRead.length < 4) await new Promise((r) => setTimeout(r, 2_000))
|
|
1938
|
+
}
|
|
1939
|
+
if (audienceRowsRead.length !== 4) {
|
|
1940
|
+
throw new Error(`expected 4 audience rows within 120s, got ${audienceRowsRead.length}`)
|
|
1941
|
+
}
|
|
1942
|
+
|
|
1943
|
+
const byCustomerId = new Map(audienceRowsRead.map((r) => [String(r.end_customer_id), r]))
|
|
1944
|
+
if (Number(byCustomerId.get(ctx.adaId)!.bucket) !== 10) throw new Error('Ada must land in bucket 10')
|
|
1945
|
+
if (Number(byCustomerId.get(ctx.graceId)!.bucket) !== 80) throw new Error('Grace must land in bucket 80')
|
|
1946
|
+
|
|
1947
|
+
for (const row of audienceRowsRead) {
|
|
1948
|
+
if (!row.legal_entity_id) {
|
|
1949
|
+
throw new Error(`${row.end_customer_id} has no stamped subject — resolution did not run`)
|
|
1950
|
+
}
|
|
1951
|
+
}
|
|
1952
|
+
const subjects = new Set(audienceRowsRead.map((r) => String(r.legal_entity_id)))
|
|
1953
|
+
if (subjects.size !== 4) {
|
|
1954
|
+
throw new Error(`four recipients must be four distinct subjects, got ${subjects.size}`)
|
|
1955
|
+
}
|
|
1956
|
+
ctx.adaSubjectId = String(byCustomerId.get(ctx.adaId)!.legal_entity_id)
|
|
1957
|
+
```
|
|
1958
|
+
|
|
1959
|
+
## 037 — two sender tools, and the booking form
|
|
1960
|
+
|
|
1961
|
+
**Two sender tools, not one.** An SMS number and an email identity are
|
|
1962
|
+
different assets, provisioned and billed separately, so §038 binds each
|
|
1963
|
+
channel to its own.
|
|
1964
|
+
|
|
1965
|
+
The email tool uses `provider: 'mock'`, which delivers in-process to the dev
|
|
1966
|
+
mailbox. Note *where* that setting lives: it is a field on the **tool body**
|
|
1967
|
+
— data on a row — not an environment variable. So the same manifest runs
|
|
1968
|
+
against any server without asking the deployment to be in "test mode", and
|
|
1969
|
+
swapping `mock` for a real provider sends real email through the same
|
|
1970
|
+
workflow.
|
|
1971
|
+
|
|
1972
|
+
```typescript
|
|
1973
|
+
const campaignSms = await ctx.ensure(
|
|
1974
|
+
'campaign SMS sender',
|
|
1975
|
+
async () => {
|
|
1976
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
1977
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Campaign SMS Sender')
|
|
1978
|
+
},
|
|
1979
|
+
async () => {
|
|
1980
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
1981
|
+
name: 'Cookbook Campaign SMS Sender',
|
|
1982
|
+
description: 'SNS-backed SMS dispatcher for the campaign, wired to LocalStack.',
|
|
1983
|
+
intent: 'sms',
|
|
1984
|
+
status: 'active',
|
|
1985
|
+
datalake_id: ctx.datalakeId,
|
|
1986
|
+
body: {
|
|
1987
|
+
tool_body_type: 'sns',
|
|
1988
|
+
auth_method: 'access_key',
|
|
1989
|
+
region: 'us-east-1',
|
|
1990
|
+
phone_number: '+15550001111',
|
|
1991
|
+
endpoint_url: 'http://localhost:4566',
|
|
1992
|
+
access_key_id: 'test',
|
|
1993
|
+
secret_access_key: 'test',
|
|
1994
|
+
},
|
|
1995
|
+
})
|
|
1996
|
+
return data
|
|
1997
|
+
},
|
|
1998
|
+
)
|
|
1999
|
+
ctx.campaignSmsToolId = campaignSms.id!
|
|
2000
|
+
|
|
2001
|
+
const campaignEmail = await ctx.ensure(
|
|
2002
|
+
'campaign email sender',
|
|
2003
|
+
async () => {
|
|
2004
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
2005
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook Campaign Email Sender')
|
|
2006
|
+
},
|
|
2007
|
+
async () => {
|
|
2008
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
2009
|
+
name: 'Cookbook Campaign Email Sender',
|
|
2010
|
+
description: 'Campaign email dispatcher — the mock provider delivers to the dev mailbox.',
|
|
2011
|
+
intent: 'email',
|
|
2012
|
+
status: 'active',
|
|
2013
|
+
datalake_id: ctx.datalakeId,
|
|
2014
|
+
body: {
|
|
2015
|
+
tool_body_type: 'email',
|
|
2016
|
+
provider: 'mock',
|
|
2017
|
+
from_email: 'campaigns@bloom-salon.example.com',
|
|
2018
|
+
from_name: 'Bloom Salon',
|
|
2019
|
+
},
|
|
2020
|
+
})
|
|
2021
|
+
return data
|
|
2022
|
+
},
|
|
2023
|
+
)
|
|
2024
|
+
ctx.campaignEmailToolId = campaignEmail.id!
|
|
2025
|
+
|
|
2026
|
+
const bookingApp = await ctx.ensure(
|
|
2027
|
+
'campaign booking form',
|
|
2028
|
+
async () => {
|
|
2029
|
+
const { data } = await api.connectedApps.list(tenantSlug, datalakeSlug)
|
|
2030
|
+
return (data.data ?? []).find((a: { name?: string }) => a.name === 'Cookbook Campaign Booking Form')
|
|
2031
|
+
},
|
|
2032
|
+
async () => {
|
|
2033
|
+
const { data } = await api.connectedApps.create(tenantSlug, datalakeSlug, {
|
|
2034
|
+
name: 'Cookbook Campaign Booking Form',
|
|
2035
|
+
description: 'The booking form the campaign links to.',
|
|
2036
|
+
mode: 'self_hosted',
|
|
2037
|
+
urls: [{ url: 'https://campaign.example.com', is_primary: true, label: 'production' }],
|
|
2038
|
+
})
|
|
2039
|
+
return data
|
|
2040
|
+
},
|
|
2041
|
+
)
|
|
2042
|
+
ctx.bookingAppId = bookingApp.id!
|
|
2043
|
+
ctx.bookingAppSlug = bookingApp.slug!
|
|
2044
|
+
```
|
|
2045
|
+
|
|
2046
|
+
## 038 — the campaign workflow: the gates, the split, both channels
|
|
2047
|
+
|
|
2048
|
+
All four rules live here, and none of them lives in application code.
|
|
2049
|
+
|
|
2050
|
+
**The gates are the filter.** `{% unless event_dataset.suppressed %}` is
|
|
2051
|
+
suppression; the two non-empty checks are reachability. A row failing either
|
|
2052
|
+
is `:filtered` — looked at, and deliberately not contacted. There is no code
|
|
2053
|
+
path around it, which is exactly what you want to be able to say about a
|
|
2054
|
+
do-not-contact list.
|
|
2055
|
+
|
|
2056
|
+
**The split is the decision node**, and it returns an *array* — both channel
|
|
2057
|
+
keys of the chosen variant. Bucket under fifty gets the A pair, otherwise
|
|
2058
|
+
the B pair. Four actions declare a `decision_key` each; the platform runs
|
|
2059
|
+
the two whose keys came back and skips the other two. That is one recipient,
|
|
2060
|
+
one variant, two channels, from one run — and the pairing is data rather
|
|
2061
|
+
than two workflows kept in step by hand.
|
|
2062
|
+
|
|
2063
|
+
**`skip_mdm_resolution: false` is set explicitly, and must be.** A
|
|
2064
|
+
generic-table row is never its own subject; the subject is the one §034
|
|
2065
|
+
stamped onto it. With resolution on, the platform loads that subject and
|
|
2066
|
+
mints a page token against it, which is what makes
|
|
2067
|
+
`{{ connected_app_form_url }}` a *per-recipient* link. Skip it and the
|
|
2068
|
+
subject is nil, no token is minted, and the link renders empty — with
|
|
2069
|
+
nothing anywhere reporting a problem.
|
|
2070
|
+
|
|
2071
|
+
`output_schema` is a JSON-Schema **object**. The string form is refused 422.
|
|
2072
|
+
|
|
2073
|
+
```typescript
|
|
2074
|
+
const CAMPAIGN_FILTER =
|
|
2075
|
+
'{% unless event_dataset.suppressed %}' +
|
|
2076
|
+
'{% if event_dataset.phone and event_dataset.phone != "" %}' +
|
|
2077
|
+
'{% if event_dataset.email and event_dataset.email != "" %}true{% endif %}' +
|
|
2078
|
+
'{% endif %}{% endunless %}'
|
|
2079
|
+
const AB_DECISION =
|
|
2080
|
+
'{% if event_dataset.bucket < 50 %}["sms_variant_a","email_variant_a"]' +
|
|
2081
|
+
'{% else %}["sms_variant_b","email_variant_b"]{% endif %}'
|
|
2082
|
+
const VARIANT_A = 'Hi {{ event_dataset.name }} — 20% off your next visit this week only. Book: {{ connected_app_form_url }}'
|
|
2083
|
+
const VARIANT_B = 'Hi {{ event_dataset.name }} — your loyalty reward is waiting. Claim: {{ connected_app_form_url }}'
|
|
2084
|
+
|
|
2085
|
+
const linkage = {
|
|
2086
|
+
trigger_template: 'now',
|
|
2087
|
+
idempotency_template: '{{ event_dataset.end_customer_id }}/{{ action_id }}',
|
|
2088
|
+
connected_app_id: ctx.bookingAppId,
|
|
2089
|
+
connected_app_route: '/forms/campaign',
|
|
2090
|
+
connected_app_metadata_template:
|
|
2091
|
+
'{"end_customer_id":"{{ event_dataset.end_customer_id }}","name":"{{ event_dataset.name }}"}',
|
|
2092
|
+
}
|
|
2093
|
+
|
|
2094
|
+
const campaign = await ctx.ensure(
|
|
2095
|
+
'loyalty campaign workflow',
|
|
2096
|
+
async () => {
|
|
2097
|
+
const { data } = await api.workflows.list(tenantSlug, datalakeSlug)
|
|
2098
|
+
return (data.data ?? []).find((w: { name?: string }) => w.name === 'Cookbook Loyalty Campaign')
|
|
2099
|
+
},
|
|
2100
|
+
async () => {
|
|
2101
|
+
const { data } = await api.workflows.create(tenantSlug, datalakeSlug, {
|
|
2102
|
+
name: 'Cookbook Loyalty Campaign',
|
|
2103
|
+
description: 'Loyalty campaign with A/B content variants across SMS and email.',
|
|
2104
|
+
dataset_type: 'generic_table',
|
|
2105
|
+
generic_table_id: ctx.audienceTableId,
|
|
2106
|
+
status: 'live',
|
|
2107
|
+
tags: ['marketing', 'loyalty'],
|
|
2108
|
+
skip_mdm_resolution: false,
|
|
2109
|
+
filter_config: { type: 'custom', body: CAMPAIGN_FILTER },
|
|
2110
|
+
decision_config: {
|
|
2111
|
+
type: 'custom',
|
|
2112
|
+
body: AB_DECISION,
|
|
2113
|
+
output_schema: { type: 'array', items: { type: 'string' } },
|
|
2114
|
+
},
|
|
2115
|
+
actions: [
|
|
2116
|
+
{
|
|
2117
|
+
...linkage,
|
|
2118
|
+
action_type: 'sms', tool_id: ctx.campaignSmsToolId, decision_key: 'sms_variant_a', position: 0,
|
|
2119
|
+
tool_call: {
|
|
2120
|
+
tool_call_type: 'sms_request',
|
|
2121
|
+
to: { type: 'custom', body: '{{ event_dataset.phone }}' },
|
|
2122
|
+
body: { type: 'custom', body: VARIANT_A },
|
|
2123
|
+
sms_type: 'transactional',
|
|
2124
|
+
},
|
|
2125
|
+
},
|
|
2126
|
+
{
|
|
2127
|
+
...linkage,
|
|
2128
|
+
action_type: 'sms', tool_id: ctx.campaignSmsToolId, decision_key: 'sms_variant_b', position: 1,
|
|
2129
|
+
tool_call: {
|
|
2130
|
+
tool_call_type: 'sms_request',
|
|
2131
|
+
to: { type: 'custom', body: '{{ event_dataset.phone }}' },
|
|
2132
|
+
body: { type: 'custom', body: VARIANT_B },
|
|
2133
|
+
sms_type: 'transactional',
|
|
2134
|
+
},
|
|
2135
|
+
},
|
|
2136
|
+
{
|
|
2137
|
+
...linkage,
|
|
2138
|
+
action_type: 'email', tool_id: ctx.campaignEmailToolId, decision_key: 'email_variant_a', position: 2,
|
|
2139
|
+
tool_call: {
|
|
2140
|
+
tool_call_type: 'email_request',
|
|
2141
|
+
to: { type: 'custom', body: '{{ event_dataset.email }}' },
|
|
2142
|
+
subject: { type: 'custom', body: 'Your 20% off is here' },
|
|
2143
|
+
body: { type: 'custom', body: VARIANT_A },
|
|
2144
|
+
},
|
|
2145
|
+
},
|
|
2146
|
+
{
|
|
2147
|
+
...linkage,
|
|
2148
|
+
action_type: 'email', tool_id: ctx.campaignEmailToolId, decision_key: 'email_variant_b', position: 3,
|
|
2149
|
+
tool_call: {
|
|
2150
|
+
tool_call_type: 'email_request',
|
|
2151
|
+
to: { type: 'custom', body: '{{ event_dataset.email }}' },
|
|
2152
|
+
subject: { type: 'custom', body: 'Your loyalty reward is waiting' },
|
|
2153
|
+
body: { type: 'custom', body: VARIANT_B },
|
|
2154
|
+
},
|
|
2155
|
+
},
|
|
2156
|
+
],
|
|
2157
|
+
})
|
|
2158
|
+
return data
|
|
2159
|
+
},
|
|
2160
|
+
)
|
|
2161
|
+
ctx.campaignWorkflowSlug = campaign.slug!
|
|
2162
|
+
|
|
2163
|
+
if (campaign.skip_mdm_resolution !== false) {
|
|
2164
|
+
throw new Error('a generic-table campaign workflow must keep MDM resolution ON')
|
|
2165
|
+
}
|
|
2166
|
+
if ((campaign.actions ?? []).length !== 4) {
|
|
2167
|
+
throw new Error(`expected 4 actions, got ${(campaign.actions ?? []).length}`)
|
|
2168
|
+
}
|
|
2169
|
+
```
|
|
2170
|
+
|
|
2171
|
+
## 039 — run the campaign, and read the split off the logs
|
|
2172
|
+
|
|
2173
|
+
One pass over all four recipients: both gates and the split in the same run.
|
|
2174
|
+
|
|
2175
|
+
The per-recipient action logs are the review surface. Each sending
|
|
2176
|
+
recipient's execution log carries four action logs, and the two that ran are
|
|
2177
|
+
the **same variant on both channels**. Group the completed ones by
|
|
2178
|
+
`decision_key` and you have per-variant performance without building a
|
|
2179
|
+
reporting pipeline.
|
|
2180
|
+
|
|
2181
|
+
**On a repeat run all four are skipped**, because the idempotency key fired
|
|
2182
|
+
the first time — so the assertion accepts either shape, and rejects anything
|
|
2183
|
+
else. Two completed from *different* variants would mean the split leaked;
|
|
2184
|
+
that is the failure worth catching here.
|
|
2185
|
+
|
|
2186
|
+
```typescript
|
|
2187
|
+
const { data: campaignRun } = await api.workflows.run(tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug, {
|
|
2188
|
+
sql_where_clause: `end_customer_id LIKE 'EC-COOKBOOK-%'`,
|
|
2189
|
+
mode: 'live',
|
|
2190
|
+
manual_override: false,
|
|
2191
|
+
})
|
|
2192
|
+
const campaignFired = await ctx.waitForFiredRun(datalakeSlug, campaignRun.workflow_run_id)
|
|
2193
|
+
ctx.campaignRunLogId = campaignFired.workflowRunLogId
|
|
2194
|
+
ctx.campaignRunBatchId = campaignFired.batchId!
|
|
2195
|
+
|
|
2196
|
+
const campaignDeadline = Date.now() + 180_000
|
|
2197
|
+
let campaignByStatus: Record<string, number> = {}
|
|
2198
|
+
while (Date.now() < campaignDeadline) {
|
|
2199
|
+
const { data: log } = await api.workflows.batchLogs.refresh(
|
|
2200
|
+
tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug, ctx.campaignRunLogId,
|
|
2201
|
+
)
|
|
2202
|
+
if (log.status === 'failed') throw new Error('campaign run reached :failed')
|
|
2203
|
+
|
|
2204
|
+
const { data: wfLogs } = await api.workflows.workflowLogs.list(tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug)
|
|
2205
|
+
const ours = (wfLogs.data ?? []).filter(
|
|
2206
|
+
(w) => (w as { batch_id?: string }).batch_id === ctx.campaignRunBatchId,
|
|
2207
|
+
)
|
|
2208
|
+
campaignByStatus = {}
|
|
2209
|
+
for (const wel of ours) {
|
|
2210
|
+
const st = (wel as { status?: string }).status ?? 'unknown'
|
|
2211
|
+
campaignByStatus[st] = (campaignByStatus[st] ?? 0) + 1
|
|
2212
|
+
}
|
|
2213
|
+
if ((campaignByStatus.completed ?? 0) >= 2 && (campaignByStatus.filtered ?? 0) >= 2) break
|
|
2214
|
+
await new Promise((r) => setTimeout(r, 3_000))
|
|
2215
|
+
}
|
|
2216
|
+
if ((campaignByStatus.completed ?? 0) !== 2 || (campaignByStatus.filtered ?? 0) !== 2) {
|
|
2217
|
+
throw new Error(`expected 2 completed + 2 filtered, got ${JSON.stringify(campaignByStatus)}`)
|
|
2218
|
+
}
|
|
2219
|
+
|
|
2220
|
+
const { data: campaignLogs } = await api.workflows.workflowLogs.list(
|
|
2221
|
+
tenantSlug, datalakeSlug, ctx.campaignWorkflowSlug,
|
|
2222
|
+
)
|
|
2223
|
+
const sentWels = (campaignLogs.data ?? []).filter((w) => {
|
|
2224
|
+
const wel = w as { batch_id?: string; status?: string }
|
|
2225
|
+
return wel.batch_id === ctx.campaignRunBatchId && wel.status === 'completed'
|
|
2226
|
+
})
|
|
2227
|
+
|
|
2228
|
+
for (const wel of sentWels) {
|
|
2229
|
+
const aels = (wel as { action_execution_logs?: Array<{ status?: string; decision_key?: string }> })
|
|
2230
|
+
.action_execution_logs ?? []
|
|
2231
|
+
if (aels.length !== 4) throw new Error(`four actions → four action logs, got ${aels.length}`)
|
|
2232
|
+
|
|
2233
|
+
const completed = aels.filter((a) => a.status === 'completed').map((a) => a.decision_key ?? '').sort()
|
|
2234
|
+
const skipped = aels.filter((a) => a.status === 'skipped')
|
|
2235
|
+
|
|
2236
|
+
if (completed.length === 0 && skipped.length === 4) continue // already sent on an earlier run
|
|
2237
|
+
|
|
2238
|
+
if (completed.length !== 2 || skipped.length !== 2) {
|
|
2239
|
+
throw new Error(`expected 2 completed + 2 skipped action logs, got ${JSON.stringify(aels)}`)
|
|
2240
|
+
}
|
|
2241
|
+
const isA = completed[0] === 'email_variant_a' && completed[1] === 'sms_variant_a'
|
|
2242
|
+
const isB = completed[0] === 'email_variant_b' && completed[1] === 'sms_variant_b'
|
|
2243
|
+
if (!isA && !isB) {
|
|
2244
|
+
throw new Error(`the completed pair must be one variant on both channels, got ${completed.join(', ')}`)
|
|
2245
|
+
}
|
|
2246
|
+
}
|
|
2247
|
+
```
|
|
2248
|
+
|
|
2249
|
+
## 040 — the short link belongs to one recipient
|
|
2250
|
+
|
|
2251
|
+
Every message the campaign rendered carries a `/t/<hash>`. It is **not** one
|
|
2252
|
+
campaign link shared by everyone: the token was minted against this
|
|
2253
|
+
recipient's own subject — the one §034 stamped and §038 resolved — so a
|
|
2254
|
+
click is attributable to a person rather than to a send.
|
|
2255
|
+
|
|
2256
|
+
Keyed on the idempotency key again, so this reads the durable record and
|
|
2257
|
+
answers on every run rather than only the one that dispatched.
|
|
2258
|
+
|
|
2259
|
+
```typescript
|
|
2260
|
+
const { data: campaignMsgSearch } = await api.datasets.createUserSearch(tenantSlug, datalakeSlug, 'message', {
|
|
2261
|
+
search_query: `m.idempotency_key LIKE '${ctx.adaId}/%'`,
|
|
2262
|
+
})
|
|
2263
|
+
if (campaignMsgSearch.status !== 'completed') {
|
|
2264
|
+
throw new Error(`message user-search status=${campaignMsgSearch.status} error=${campaignMsgSearch.error_message ?? '(none)'}`)
|
|
2265
|
+
}
|
|
2266
|
+
|
|
2267
|
+
const linkDeadline = Date.now() + 120_000
|
|
2268
|
+
let campaignShortPath: string | undefined
|
|
2269
|
+
while (Date.now() < linkDeadline && !campaignShortPath) {
|
|
2270
|
+
const { data: found } = await api.datasets.search(tenantSlug, datalakeSlug, 'message', {
|
|
2271
|
+
userSearchId: campaignMsgSearch.id!,
|
|
2272
|
+
dataAccessMode: 'raw',
|
|
2273
|
+
})
|
|
2274
|
+
const bodies = ((found.data ?? []) as Array<Record<string, unknown>>).map((m) => String(m.body ?? ''))
|
|
2275
|
+
campaignShortPath = bodies.find((b) => b.includes('/t/'))?.match(/\/t\/([A-Za-z0-9_-]+)/)?.[1]
|
|
2276
|
+
if (!campaignShortPath) await new Promise((r) => setTimeout(r, 2_000))
|
|
2277
|
+
}
|
|
2278
|
+
if (!campaignShortPath) {
|
|
2279
|
+
throw new Error(`no campaign message for ${ctx.adaId} carried a /t/ short link within 120s`)
|
|
2280
|
+
}
|
|
2281
|
+
|
|
2282
|
+
const { data: campaignResolved } = await api.connectedApps.resolvePage(
|
|
2283
|
+
tenantSlug, datalakeSlug, ctx.bookingAppSlug,
|
|
2284
|
+
{ short_path: campaignShortPath, user_agent: 'cookbook-doctest/marketing-campaign-send' },
|
|
2285
|
+
)
|
|
2286
|
+
if (campaignResolved.route_path !== '/forms/campaign') {
|
|
2287
|
+
throw new Error(`resolvePage route_path mismatch: ${campaignResolved.route_path}`)
|
|
2288
|
+
}
|
|
2289
|
+
|
|
2290
|
+
const { data: campaignTracked } = await api.connectedApps.updateMessageTracking(
|
|
2291
|
+
tenantSlug, datalakeSlug, ctx.bookingAppSlug,
|
|
2292
|
+
{ short_path: campaignShortPath, opened_at: new Date().toISOString() },
|
|
2293
|
+
)
|
|
2294
|
+
if (!campaignTracked.message?.opened_at) {
|
|
2295
|
+
throw new Error('message tracking did not persist opened_at')
|
|
2296
|
+
}
|
|
2297
|
+
```
|
|
2298
|
+
|
|
2299
|
+
## 041 — the reply comes back, and attaches to whoever sent it
|
|
2300
|
+
|
|
2301
|
+
A reply arrives on the same handle the campaign sent to. It ingests into the
|
|
2302
|
+
**system** inbound table through a client whose `mdm_input_config` keys on
|
|
2303
|
+
the same `(uri, value)` namespace §034 wrote — so the platform resolves the
|
|
2304
|
+
handle straight back to the customer and stamps their subject on the row.
|
|
2305
|
+
|
|
2306
|
+
**This client carries only a generic-table contract**, and that is not an
|
|
2307
|
+
oversight: a reply does not create a person, it finds one. Adding a
|
|
2308
|
+
legal-entity contract here would mint a second subject for someone who
|
|
2309
|
+
already exists — and would put two writers on one subject, which §033
|
|
2310
|
+
explains at length.
|
|
2311
|
+
|
|
2312
|
+
**When a handle is ambiguous the reply still attaches.** If two capture
|
|
2313
|
+
systems each registered a customer under the same phone, that handle maps to
|
|
2314
|
+
two entities. The row lands anyway, attached to one of them, and is marked
|
|
2315
|
+
`potential_duplicate` for a human to adjudicate. Flag, don't block: a dropped
|
|
2316
|
+
reply is a lost customer, a flagged one is a two-minute review.
|
|
2317
|
+
|
|
2318
|
+
Below, the reply lands and attaches to Ada; then the same `message_id` is
|
|
2319
|
+
re-ingested with the flag set. `message_id` is the table's unique column, so
|
|
2320
|
+
the second ingest **upserts** — one row, now flagged, not a second copy.
|
|
2321
|
+
|
|
2322
|
+
```typescript
|
|
2323
|
+
const REPLY_GT = `{% assign p = msg %}
|
|
2324
|
+
{
|
|
2325
|
+
"message_id": "{{ p.message_id | json_escape }}",
|
|
2326
|
+
"handle": "{{ p.handle | json_escape }}",
|
|
2327
|
+
"channel": "{{ p.channel | json_escape }}",
|
|
2328
|
+
"body": "{{ p.body | json_escape }}",
|
|
2329
|
+
"potential_duplicate": {% if p.potential_duplicate %}true{% else %}false{% endif %}
|
|
2330
|
+
}`
|
|
2331
|
+
|
|
2332
|
+
const REPLY_MDM = `{% assign p = msg %}
|
|
2333
|
+
{
|
|
2334
|
+
"legal_entity_type": "individual",
|
|
2335
|
+
"identifiers": [
|
|
2336
|
+
{"uri": "{{ p.source_uri | json_escape }}", "type": "digital_identifier", "value": "{{ p.handle | json_escape }}"}
|
|
2337
|
+
]
|
|
2338
|
+
}`
|
|
2339
|
+
|
|
2340
|
+
const replyGt = await ctx.ensure(
|
|
2341
|
+
'inbound reply contract',
|
|
2342
|
+
async () => {
|
|
2343
|
+
const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
|
|
2344
|
+
return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Reply GT')
|
|
2345
|
+
},
|
|
2346
|
+
async () => {
|
|
2347
|
+
const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
|
|
2348
|
+
name: 'Cookbook Reply GT',
|
|
2349
|
+
description: 'Inbound reply → Inbound Messages row, re-attached to the customer who sent it.',
|
|
2350
|
+
resource_type: 'generic_table',
|
|
2351
|
+
type: 'identity',
|
|
2352
|
+
generic_table_id: ctx.inboundTableId,
|
|
2353
|
+
template_config: { type: 'custom', body: REPLY_GT },
|
|
2354
|
+
mdm_input_config: { type: 'custom', body: REPLY_MDM },
|
|
2355
|
+
})
|
|
2356
|
+
return data
|
|
2357
|
+
},
|
|
2358
|
+
)
|
|
2359
|
+
|
|
2360
|
+
const replyDac = await ctx.ensure(
|
|
2361
|
+
'inbound reply client',
|
|
2362
|
+
async () => {
|
|
2363
|
+
const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
|
|
2364
|
+
filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Reply Client' }],
|
|
2365
|
+
})
|
|
2366
|
+
return (data.data ?? [])[0]
|
|
2367
|
+
},
|
|
2368
|
+
async () => {
|
|
2369
|
+
const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
|
|
2370
|
+
name: 'Cookbook Reply Client',
|
|
2371
|
+
description: 'Inbound reply ingest — re-attaches the sender handle to its subject.',
|
|
2372
|
+
tool_id: ctx.manualUploadToolId,
|
|
2373
|
+
data_source_id: dataSourceId,
|
|
2374
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
2375
|
+
interoperability_contract_ids: [replyGt.id!],
|
|
2376
|
+
})
|
|
2377
|
+
return data
|
|
2378
|
+
},
|
|
2379
|
+
)
|
|
2380
|
+
|
|
2381
|
+
const replyMessageId = 'IN-COOKBOOK-0101'
|
|
2382
|
+
|
|
2383
|
+
const readReply = async (): Promise<Record<string, unknown> | undefined> => {
|
|
2384
|
+
const { data: result } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
|
|
2385
|
+
sql: `SELECT message_id, legal_entity_id, potential_duplicate
|
|
2386
|
+
FROM ${ctx.inboundTableName}
|
|
2387
|
+
WHERE message_id = '${replyMessageId}'`,
|
|
2388
|
+
mode: 'raw',
|
|
2389
|
+
})
|
|
2390
|
+
if ((result.data ?? []).length === 0) return undefined
|
|
2391
|
+
return Object.fromEntries(result.meta.columns.map((c, i) => [c, result.data[0][i]]))
|
|
2392
|
+
}
|
|
2393
|
+
|
|
2394
|
+
const firstReply = await api.dataActivationClients.ingest(
|
|
2395
|
+
tenantSlug, datalakeSlug, replyDac.slug!,
|
|
2396
|
+
{ data: { message_id: replyMessageId, handle: ctx.adaPhone, channel: 'sms', body: 'Yes! Book me for Friday.' } },
|
|
2397
|
+
)
|
|
2398
|
+
await ctx.waitForBatches(replyDac.slug!, [firstReply.data.batch_id!])
|
|
2399
|
+
|
|
2400
|
+
const reply = await readReply()
|
|
2401
|
+
if (!reply) throw new Error('the reply did not land in the inbound messages table')
|
|
2402
|
+
if (String(reply.legal_entity_id) !== ctx.adaSubjectId) {
|
|
2403
|
+
throw new Error('the reply must re-attach to the customer who sent it')
|
|
2404
|
+
}
|
|
2405
|
+
|
|
2406
|
+
// Same message_id, flag set — the unique column makes this an upsert.
|
|
2407
|
+
const flagReply = await api.dataActivationClients.ingest(
|
|
2408
|
+
tenantSlug, datalakeSlug, replyDac.slug!,
|
|
2409
|
+
{
|
|
2410
|
+
data: {
|
|
2411
|
+
message_id: replyMessageId,
|
|
2412
|
+
handle: ctx.adaPhone,
|
|
2413
|
+
channel: 'sms',
|
|
2414
|
+
body: 'Yes! Book me for Friday.',
|
|
2415
|
+
potential_duplicate: true,
|
|
2416
|
+
},
|
|
2417
|
+
},
|
|
2418
|
+
)
|
|
2419
|
+
await ctx.waitForBatches(replyDac.slug!, [flagReply.data.batch_id!])
|
|
2420
|
+
|
|
2421
|
+
const flagged = await readReply()
|
|
2422
|
+
if (!flagged || flagged.potential_duplicate !== true) {
|
|
2423
|
+
throw new Error('the re-ingest did not flag the reply potential_duplicate')
|
|
2424
|
+
}
|
|
2425
|
+
```
|
|
2426
|
+
|
|
2427
|
+
## 042 — reconciling what actually got delivered
|
|
2428
|
+
|
|
2429
|
+
Everything above ends at *sent*. Sent is not delivered, and the gap between
|
|
2430
|
+
them is where a campaign quietly stops working — a number that has been
|
|
2431
|
+
disconnected for months keeps reporting `sent` forever.
|
|
2432
|
+
|
|
2433
|
+
An **action status updater** closes that gap on a schedule. It names a
|
|
2434
|
+
poller tool that reads a delivery log, the sender tools it reconciles for,
|
|
2435
|
+
and two Liquid templates that turn each provider event into an update.
|
|
2436
|
+
|
|
2437
|
+
Three details that are each a 422 or a silent nothing if you get them wrong.
|
|
2438
|
+
|
|
2439
|
+
**`start_time` and `end_time` use `now_msec`, not `now`.** `now_msec` is the
|
|
2440
|
+
injected variable holding unix milliseconds, which is what
|
|
2441
|
+
`minutes_ago` expects; the bare `now` is a DateTime and raises.
|
|
2442
|
+
|
|
2443
|
+
**`message_config` renders once per event, not once per batch.** The event
|
|
2444
|
+
itself is the assigns — there is no `events` list to loop over — and it must
|
|
2445
|
+
emit a *flat* object: the `external_id` of the message to update, plus the
|
|
2446
|
+
fields to set, at the top level.
|
|
2447
|
+
|
|
2448
|
+
**`action_log_config` is required alongside it.** Same per-event assigns,
|
|
2449
|
+
rendered into the action-log write shape. Omitting it is a 422 that names
|
|
2450
|
+
the field: `Missing field: action_log_config`.
|
|
2451
|
+
|
|
2452
|
+
```typescript
|
|
2453
|
+
const cwTool = await ctx.ensure(
|
|
2454
|
+
'CloudWatch poller tool',
|
|
2455
|
+
async () => {
|
|
2456
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
2457
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook CloudWatch Poller')
|
|
2458
|
+
},
|
|
2459
|
+
async () => {
|
|
2460
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
2461
|
+
name: 'Cookbook CloudWatch Poller',
|
|
2462
|
+
description: 'CloudWatch log-group poller — reads SNS delivery events for reconciliation.',
|
|
2463
|
+
intent: 'status_poller',
|
|
2464
|
+
status: 'active',
|
|
2465
|
+
datalake_id: ctx.datalakeId,
|
|
2466
|
+
data_source_id: dataSourceId,
|
|
2467
|
+
body: {
|
|
2468
|
+
tool_body_type: 'cloud_watch_log_group',
|
|
2469
|
+
auth_method: 'access_key',
|
|
2470
|
+
region: 'us-east-1',
|
|
2471
|
+
endpoint_url: 'http://localhost:4566',
|
|
2472
|
+
access_key_id: 'test',
|
|
2473
|
+
secret_access_key: 'test',
|
|
2474
|
+
},
|
|
2475
|
+
})
|
|
2476
|
+
return data
|
|
2477
|
+
},
|
|
2478
|
+
)
|
|
2479
|
+
ctx.cwToolId = cwTool.id!
|
|
2480
|
+
|
|
2481
|
+
const deliveryUpdater = await ctx.ensure(
|
|
2482
|
+
'SMS delivery reconciler',
|
|
2483
|
+
async () => {
|
|
2484
|
+
const { data } = await api.actionStatusUpdaters.list(tenantSlug, datalakeSlug)
|
|
2485
|
+
return (data.data ?? []).find((u: { name?: string }) => u.name === 'Cookbook SMS Delivery Updater')
|
|
2486
|
+
},
|
|
2487
|
+
async () => {
|
|
2488
|
+
const { data } = await api.actionStatusUpdaters.create(tenantSlug, datalakeSlug, {
|
|
2489
|
+
name: 'Cookbook SMS Delivery Updater',
|
|
2490
|
+
cron_expression: '*/30 * * * *',
|
|
2491
|
+
updater_type: 'cloud_watch',
|
|
2492
|
+
updater_tool_id: ctx.cwToolId,
|
|
2493
|
+
sender_tool_ids: [ctx.smsToolId],
|
|
2494
|
+
datalake_id: ctx.datalakeId,
|
|
2495
|
+
updater_body: {
|
|
2496
|
+
updater_body_type: 'cloud_watch_request',
|
|
2497
|
+
log_group_name: 'sns/us-east-1/000000000000/DirectPublishToPhoneNumber',
|
|
2498
|
+
start_time: '{{ now_msec | minutes_ago: 45 }}',
|
|
2499
|
+
end_time: '{{ now_msec }}',
|
|
2500
|
+
},
|
|
2501
|
+
message_config: {
|
|
2502
|
+
type: 'custom',
|
|
2503
|
+
body: '{"external_id": "{{ notification.messageId }}", "status": "delivered"}',
|
|
2504
|
+
},
|
|
2505
|
+
action_log_config: {
|
|
2506
|
+
type: 'custom',
|
|
2507
|
+
body: '{"external_id": "{{ notification.messageId }}", "status": "delivered"}',
|
|
2508
|
+
},
|
|
2509
|
+
})
|
|
2510
|
+
return data
|
|
2511
|
+
},
|
|
2512
|
+
)
|
|
2513
|
+
actionStatusUpdaterId = deliveryUpdater.id!
|
|
2514
|
+
```
|
|
2515
|
+
|
|
2516
|
+
## 043 — read it back, and check the checksum
|
|
2517
|
+
|
|
2518
|
+
`get` returns the stored row and its fields echo what you sent. `checksum`
|
|
2519
|
+
recomputes the fingerprint from a body **without persisting it** — so
|
|
2520
|
+
comparing the two is how you detect that a reconciler on the server has
|
|
2521
|
+
drifted from the one in your manifest, before it silently reconciles
|
|
2522
|
+
against the wrong window.
|
|
2523
|
+
|
|
2524
|
+
The catalog reads are worth knowing about for a different reason: they are
|
|
2525
|
+
what an agent reads to find out which reconcilers exist at all.
|
|
2526
|
+
|
|
2527
|
+
```typescript
|
|
2528
|
+
const { data: storedUpdater } = await api.actionStatusUpdaters.get(
|
|
2529
|
+
tenantSlug, datalakeSlug, actionStatusUpdaterId,
|
|
2530
|
+
)
|
|
2531
|
+
if (storedUpdater.cron_expression !== '*/30 * * * *') {
|
|
2532
|
+
throw new Error(`cron mismatch on read-back: ${storedUpdater.cron_expression}`)
|
|
2533
|
+
}
|
|
2534
|
+
|
|
2535
|
+
const { data: updaterList } = await api.actionStatusUpdaters.list(tenantSlug, datalakeSlug)
|
|
2536
|
+
if (!(updaterList.data ?? []).some((u) => u.id === actionStatusUpdaterId)) {
|
|
2537
|
+
throw new Error('the created reconciler is not in the list')
|
|
2538
|
+
}
|
|
2539
|
+
|
|
2540
|
+
const { data: updaterCatalog } = await api.actionStatusUpdaters.metadata(tenantSlug, datalakeSlug)
|
|
2541
|
+
if (typeof updaterCatalog !== 'string' || updaterCatalog.length === 0) {
|
|
2542
|
+
throw new Error('expected a non-empty reconciler catalog')
|
|
2543
|
+
}
|
|
2544
|
+
|
|
2545
|
+
const { data: updaterDetail } = await api.actionStatusUpdaters.metadataDetails(
|
|
2546
|
+
tenantSlug, datalakeSlug, actionStatusUpdaterId,
|
|
2547
|
+
)
|
|
2548
|
+
if (typeof updaterDetail !== 'string' || updaterDetail.length === 0) {
|
|
2549
|
+
throw new Error('expected non-empty reconciler detail')
|
|
2550
|
+
}
|
|
2551
|
+
```
|
|
2552
|
+
|
|
2553
|
+
## 044 — a poller that cannot advance is refused at create
|
|
2554
|
+
|
|
2555
|
+
Not every delivery log is a log group. Pulling outcomes from a provider's
|
|
2556
|
+
REST events API means paging, and paging is where this goes wrong in a way
|
|
2557
|
+
that looks like it is working.
|
|
2558
|
+
|
|
2559
|
+
**The platform refuses a poller whose request can never move.** Below, the
|
|
2560
|
+
first create renders a path and params that read neither
|
|
2561
|
+
`msg.pagination_context` nor `msg.page` — so page five hundred's request is
|
|
2562
|
+
byte-identical to page one's, `has_next` never goes false, and the run
|
|
2563
|
+
re-applies the same events forever. That is a 422 at create, and the walk
|
|
2564
|
+
catches it deliberately rather than mentioning it.
|
|
2565
|
+
|
|
2566
|
+
Then the shape that is accepted: page one renders the plain collection path
|
|
2567
|
+
with bounded params, and every later page rides the cursor the pagination
|
|
2568
|
+
template captured — dropping the params, because the provider's `next` is a
|
|
2569
|
+
complete URL that already carries them.
|
|
2570
|
+
|
|
2571
|
+
Two more floors worth stating. `events_output_schema` must be an array of
|
|
2572
|
+
**objects that require `external_id`**; a bare `{ type: 'array' }` is
|
|
2573
|
+
refused, because an event that does not name the message it reconciles
|
|
2574
|
+
cannot reconcile anything. And the provider's own id becomes `external_id`
|
|
2575
|
+
in the **events template** — the apply path pops that key to find the row it
|
|
2576
|
+
updates, so the render projects each event rather than passing the
|
|
2577
|
+
provider's body through untouched.
|
|
2578
|
+
|
|
2579
|
+
```typescript
|
|
2580
|
+
const restTool = await ctx.ensure(
|
|
2581
|
+
'REST events poller tool',
|
|
2582
|
+
async () => {
|
|
2583
|
+
const { data } = await api.tools.list(tenantSlug, datalakeSlug)
|
|
2584
|
+
return (data.data ?? []).find((t: { name?: string }) => t.name === 'Cookbook REST Events Poller')
|
|
2585
|
+
},
|
|
2586
|
+
async () => {
|
|
2587
|
+
const { data } = await api.tools.create(tenantSlug, datalakeSlug, {
|
|
2588
|
+
name: 'Cookbook REST Events Poller',
|
|
2589
|
+
description: 'REST poller — supplies auth for a provider events API.',
|
|
2590
|
+
intent: 'status_poller',
|
|
2591
|
+
status: 'active',
|
|
2592
|
+
datalake_id: ctx.datalakeId,
|
|
2593
|
+
data_source_id: dataSourceId,
|
|
2594
|
+
body: {
|
|
2595
|
+
tool_body_type: 'rest_api',
|
|
2596
|
+
base_url: 'http://localhost:8080/mailgun/v3',
|
|
2597
|
+
auth_method: 'basic',
|
|
2598
|
+
username: 'api',
|
|
2599
|
+
password: 'key-test',
|
|
2600
|
+
request_type: 'json',
|
|
2601
|
+
response_type: 'json',
|
|
2602
|
+
timeout_ms: 30_000,
|
|
2603
|
+
},
|
|
2604
|
+
})
|
|
2605
|
+
return data
|
|
2606
|
+
},
|
|
2607
|
+
)
|
|
2608
|
+
ctx.restToolId = restTool.id!
|
|
2609
|
+
|
|
2610
|
+
const EVENTS_TEMPLATE =
|
|
2611
|
+
'[{% for item in response.items %}' +
|
|
2612
|
+
'{"external_id": "{{ item.message.headers[\'message-id\'] }}", "event": "{{ item.event }}"}' +
|
|
2613
|
+
'{% unless forloop.last %},{% endunless %}{% endfor %}]'
|
|
2614
|
+
const PAGINATION_TEMPLATE =
|
|
2615
|
+
'{"has_next": {% if response.items.size > 0 %}true{% else %}false{% endif %}, ' +
|
|
2616
|
+
'"next": "{{ response.paging.next }}"}'
|
|
2617
|
+
const EVENTS_OUTPUT_SCHEMA = {
|
|
2618
|
+
type: 'array',
|
|
2619
|
+
items: {
|
|
2620
|
+
type: 'object',
|
|
2621
|
+
required: ['external_id'],
|
|
2622
|
+
properties: { external_id: { type: 'string' } },
|
|
2623
|
+
},
|
|
2624
|
+
}
|
|
2625
|
+
const PAGINATION_OUTPUT_SCHEMA = {
|
|
2626
|
+
type: 'object',
|
|
2627
|
+
required: ['has_next'],
|
|
2628
|
+
properties: { has_next: { type: 'boolean' } },
|
|
2629
|
+
}
|
|
2630
|
+
|
|
2631
|
+
// The guard. A static request is refused — walked, not asserted from docs.
|
|
2632
|
+
let refused = false
|
|
2633
|
+
try {
|
|
2634
|
+
await api.actionStatusUpdaters.create(tenantSlug, datalakeSlug, {
|
|
2635
|
+
name: 'Cookbook Non-Advancing Poller',
|
|
2636
|
+
cron_expression: '*/30 * * * *',
|
|
2637
|
+
updater_type: 'restapi',
|
|
2638
|
+
updater_tool_id: ctx.restToolId,
|
|
2639
|
+
sender_tool_ids: [ctx.smsToolId],
|
|
2640
|
+
datalake_id: ctx.datalakeId,
|
|
2641
|
+
events_output_schema: EVENTS_OUTPUT_SCHEMA,
|
|
2642
|
+
pagination_context_output_schema: PAGINATION_OUTPUT_SCHEMA,
|
|
2643
|
+
updater_body: {
|
|
2644
|
+
updater_body_type: 'restapi_request',
|
|
2645
|
+
method: 'get',
|
|
2646
|
+
// Static on both — the cursor is captured below and never read back.
|
|
2647
|
+
path: { type: 'custom', body: '/wiremock.domain/events' },
|
|
2648
|
+
params: { type: 'custom', body: '{"event": "delivered"}' },
|
|
2649
|
+
events_template: { type: 'custom', body: EVENTS_TEMPLATE },
|
|
2650
|
+
pagination_context_template: { type: 'custom', body: PAGINATION_TEMPLATE },
|
|
2651
|
+
},
|
|
2652
|
+
message_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
|
|
2653
|
+
action_log_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
|
|
2654
|
+
})
|
|
2655
|
+
} catch (err) {
|
|
2656
|
+
const status = (err as { _httpStatus?: number })._httpStatus
|
|
2657
|
+
if (status !== 422) throw err
|
|
2658
|
+
refused = true
|
|
2659
|
+
}
|
|
2660
|
+
if (!refused) {
|
|
2661
|
+
throw new Error('expected a 422 for a poller whose request never consumes the cursor')
|
|
2662
|
+
}
|
|
2663
|
+
|
|
2664
|
+
// The accepted shape: the path follows the cursor.
|
|
2665
|
+
const advancingPoller = await ctx.ensure(
|
|
2666
|
+
'advancing REST delivery poller',
|
|
2667
|
+
async () => {
|
|
2668
|
+
const { data } = await api.actionStatusUpdaters.list(tenantSlug, datalakeSlug)
|
|
2669
|
+
return (data.data ?? []).find((u: { name?: string }) => u.name === 'Cookbook Advancing Delivery Poller')
|
|
2670
|
+
},
|
|
2671
|
+
async () => {
|
|
2672
|
+
const { data } = await api.actionStatusUpdaters.create(tenantSlug, datalakeSlug, {
|
|
2673
|
+
name: 'Cookbook Advancing Delivery Poller',
|
|
2674
|
+
cron_expression: '*/30 * * * *',
|
|
2675
|
+
updater_type: 'restapi',
|
|
2676
|
+
updater_tool_id: ctx.restToolId,
|
|
2677
|
+
sender_tool_ids: [ctx.smsToolId],
|
|
2678
|
+
datalake_id: ctx.datalakeId,
|
|
2679
|
+
events_output_schema: EVENTS_OUTPUT_SCHEMA,
|
|
2680
|
+
pagination_context_output_schema: PAGINATION_OUTPUT_SCHEMA,
|
|
2681
|
+
updater_body: {
|
|
2682
|
+
updater_body_type: 'restapi_request',
|
|
2683
|
+
method: 'get',
|
|
2684
|
+
path: {
|
|
2685
|
+
type: 'custom',
|
|
2686
|
+
body:
|
|
2687
|
+
'{% if msg.pagination_context %}{{ msg.pagination_context.next }}' +
|
|
2688
|
+
'{% else %}/wiremock.domain/events{% endif %}',
|
|
2689
|
+
},
|
|
2690
|
+
params: {
|
|
2691
|
+
type: 'custom',
|
|
2692
|
+
body: '{% unless msg.pagination_context %}{"event": "delivered"}{% endunless %}',
|
|
2693
|
+
},
|
|
2694
|
+
events_template: { type: 'custom', body: EVENTS_TEMPLATE },
|
|
2695
|
+
pagination_context_template: { type: 'custom', body: PAGINATION_TEMPLATE },
|
|
2696
|
+
},
|
|
2697
|
+
message_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
|
|
2698
|
+
action_log_config: { type: 'custom', body: '{"external_id": "{{ external_id }}", "status": "delivered"}' },
|
|
2699
|
+
})
|
|
2700
|
+
return data
|
|
2701
|
+
},
|
|
2702
|
+
)
|
|
2703
|
+
if (advancingPoller.status !== 'active') {
|
|
2704
|
+
throw new Error(`a newly created poller must be free to poll — got status ${advancingPoller.status}`)
|
|
2705
|
+
}
|
|
2706
|
+
```
|
|
2707
|
+
|
|
2708
|
+
## 045 — ask the lake a question in plain language
|
|
2709
|
+
|
|
2710
|
+
The last thing a marketing team needs is a way to look at any of this
|
|
2711
|
+
without writing SQL.
|
|
2712
|
+
|
|
2713
|
+
`textToSql` runs ordered multi-provider failover and hands back the
|
|
2714
|
+
generated `sql` along with which model and provider produced it, and a
|
|
2715
|
+
best-effort `explanation` (`null` when the explainer is unavailable).
|
|
2716
|
+
|
|
2717
|
+
**It does not execute anything.** It returns SQL for a human to read, which
|
|
2718
|
+
is the whole design: only the prompt and the datalake's *schema* cross the
|
|
2719
|
+
model boundary. No rows leave the trust boundary to answer a question about
|
|
2720
|
+
them.
|
|
2721
|
+
|
|
2722
|
+
```typescript
|
|
2723
|
+
const { data: generated } = await api.datalakes.textToSql(tenantSlug, datalakeSlug, {
|
|
2724
|
+
prompt: 'how many end customers are suppressed?',
|
|
2725
|
+
mode: 'raw',
|
|
2726
|
+
})
|
|
2727
|
+
if (typeof generated.sql !== 'string' || generated.sql.trim() === '') {
|
|
2728
|
+
throw new Error(`textToSql returned no SQL (got: ${JSON.stringify(generated.sql)})`)
|
|
2729
|
+
}
|
|
2730
|
+
if (typeof generated.model !== 'string' || typeof generated.provider !== 'string') {
|
|
2731
|
+
throw new Error('textToSql response is missing its model/provider attribution')
|
|
2732
|
+
}
|
|
2733
|
+
console.log(` ↳ ${generated.provider}/${generated.model} proposed: ${generated.sql.trim().slice(0, 120)}`)
|
|
2734
|
+
```
|
|
2735
|
+
|
|
2736
|
+
## 046 — run it read-only, then take the same page as CSV
|
|
2737
|
+
|
|
2738
|
+
`executeSql` runs read-only — inserts, updates, deletes and DDL are all
|
|
2739
|
+
refused — and returns `{ data, meta }`. **`data` is an array of arrays**,
|
|
2740
|
+
positional and aligned to `meta.columns`, because arbitrary SQL can produce
|
|
2741
|
+
duplicate or expression column names that object keys would collapse.
|
|
2742
|
+
|
|
2743
|
+
The deterministic count below keeps the step independent of whatever the
|
|
2744
|
+
model proposed a moment ago, which is what you want in a walk: §045 proves
|
|
2745
|
+
the generation, this proves the execution, and neither can mask a failure in
|
|
2746
|
+
the other.
|
|
2747
|
+
|
|
2748
|
+
Passing `{ format: 'csv' }` returns the same page as a string instead of the
|
|
2749
|
+
JSON envelope — the same read, content-negotiated for download. The curated
|
|
2750
|
+
return type is a union, so narrow it before reading either shape.
|
|
2751
|
+
|
|
2752
|
+
```typescript
|
|
2753
|
+
const suppressedCount = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
|
|
2754
|
+
sql: `SELECT count(*) AS n FROM ${ctx.audienceTableName} WHERE suppressed = true`,
|
|
2755
|
+
mode: 'raw',
|
|
2756
|
+
})
|
|
2757
|
+
if (typeof suppressedCount.data === 'string') {
|
|
2758
|
+
throw new Error('expected the JSON envelope, got a CSV string')
|
|
2759
|
+
}
|
|
2760
|
+
const countPage = suppressedCount.data
|
|
2761
|
+
if (!Array.isArray(countPage.data) || !Array.isArray(countPage.data[0])) {
|
|
2762
|
+
throw new Error('executeSql data is not an array-of-arrays')
|
|
2763
|
+
}
|
|
2764
|
+
if (countPage.meta.columns[0] !== 'n') {
|
|
2765
|
+
throw new Error(`unexpected column list: ${JSON.stringify(countPage.meta.columns)}`)
|
|
2766
|
+
}
|
|
2767
|
+
if (Number(countPage.data[0][0]) < 1) {
|
|
2768
|
+
throw new Error('at least one end customer is suppressed — §035 ingested one')
|
|
2769
|
+
}
|
|
2770
|
+
|
|
2771
|
+
const csvExport = await api.datalakes.executeSql(
|
|
2772
|
+
tenantSlug,
|
|
2773
|
+
datalakeSlug,
|
|
2774
|
+
{
|
|
2775
|
+
sql: `SELECT end_customer_id, bucket, suppressed FROM ${ctx.audienceTableName} ORDER BY end_customer_id`,
|
|
2776
|
+
mode: 'raw',
|
|
2777
|
+
},
|
|
2778
|
+
{ format: 'csv' },
|
|
2779
|
+
)
|
|
2780
|
+
if (typeof csvExport.data !== 'string') {
|
|
2781
|
+
throw new Error('expected a CSV string for { format: "csv" }')
|
|
2782
|
+
}
|
|
2783
|
+
if (!csvExport.data.includes('end_customer_id') || !csvExport.data.includes(ctx.adaId)) {
|
|
2784
|
+
throw new Error(`unexpected CSV body: ${JSON.stringify(csvExport.data.slice(0, 200))}`)
|
|
2785
|
+
}
|
|
2786
|
+
```
|
|
2787
|
+
|
|
2788
|
+
# Outcome
|
|
2789
|
+
|
|
2790
|
+
The marketing tenant exists with one raw datalake, an SMS tool, an LLM
|
|
2791
|
+
tool, a lead-submissions table, and an agent-driven workflow that bands
|
|
2792
|
+
every inbound lead and fans out to exactly one of four SMS actions.
|
|
2793
|
+
|
|
2794
|
+
Running this walk again reuses all of it.
|
|
2795
|
+
|
|
2796
|
+
# See also
|
|
2797
|
+
|
|
2798
|
+
- `datalakes.md` — the create body, and what a derived pair would add
|
|
2799
|
+
- `workflows.md` — scheduling versus firing, and the accessor tables
|
|
2800
|
+
- `ai_agents.md` — response schemas, and why the enum is the real guard
|
|
2801
|
+
- `generic_tables.md` — the default client you never have to create
|