@alvera-ai/platform-sdk 0.10.0-rc.2 → 0.10.0-rc.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/AGENTS.md +440 -0
- package/.agent/account_management.md +455 -0
- package/.agent/action_status_updaters.md +262 -0
- package/.agent/ai_agents.md +423 -0
- package/.agent/ai_sandbox.md +265 -0
- package/.agent/async.md +111 -0
- package/.agent/connected_apps.md +407 -0
- package/.agent/cookbook/_fixtures/README.md +99 -0
- package/.agent/cookbook/_fixtures/accounts_receivable/_customers_accounts_receivable_customer.liquid +32 -0
- package/.agent/cookbook/_fixtures/accounts_receivable/_customers_accounts_receivable_mdm.liquid +20 -0
- package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_generic_table.liquid +33 -0
- package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_legal_entity.liquid +88 -0
- package/.agent/cookbook/_fixtures/foundation/_lead_submissions_foundation_mdm.liquid +48 -0
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_appointment.liquid +47 -0
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_mdm.liquid +24 -0
- package/.agent/cookbook/_fixtures/healthcare/_cahps_appointments_healthcare_patient.liquid +38 -0
- package/.agent/cookbook/_fixtures/payment_risk/_compliance_screenings_payment_risk_compliance_screening.liquid +59 -0
- package/.agent/cookbook/_fixtures/payment_risk/_compliance_screenings_payment_risk_mdm.liquid +36 -0
- package/.agent/cookbook/_fixtures/payment_risk/_payment_accounts_payment_risk_mdm.liquid +30 -0
- package/.agent/cookbook/_fixtures/payment_risk/_payment_accounts_payment_risk_payment_account.liquid +55 -0
- package/.agent/cookbook/_setup/accounts_receivable.md +282 -0
- package/.agent/cookbook/_setup/foundation.md +277 -0
- package/.agent/cookbook/_setup/healthcare.md +279 -0
- package/.agent/cookbook/_setup/payment_risk.md +283 -0
- package/.agent/cookbook/appointment-review-sms-workflow.md +761 -0
- package/.agent/cookbook/birthday-greeting-sms-trigger.md +656 -0
- package/.agent/cookbook/contact-us-triage-with-llm.md +603 -0
- package/.agent/cookbook/dunning-sms-for-delinquent.md +619 -0
- package/.agent/cookbook/kyc-notification-on-account-activation.md +619 -0
- package/.agent/cookbook/sanctions-screening-with-agent-review.md +711 -0
- package/.agent/cookbook/score-leads-with-llm-categorization.md +602 -0
- package/.agent/cookbook/welcome-sms-for-customers.md +607 -0
- package/.agent/data_activation_clients.md +557 -0
- package/.agent/data_sources.md +234 -0
- package/.agent/datalakes.md +712 -0
- package/.agent/debugging.md +137 -0
- package/.agent/errors.md +196 -0
- package/.agent/generic_tables.md +351 -0
- package/.agent/interoperability_contracts.md +351 -0
- package/.agent/mdm.md +293 -0
- package/.agent/mutations.md +152 -0
- package/.agent/templates.md +98 -0
- package/.agent/tool-call-configs.md +90 -0
- package/.agent/tools.md +546 -0
- package/.agent/type_naming.md +131 -0
- package/.agent/workflows.md +601 -0
- package/README.md +46 -0
- package/dist/bin/platform-sdk.d.mts +1 -0
- package/dist/bin/platform-sdk.mjs +106 -0
- package/dist/bin/platform-sdk.mjs.map +1 -0
- package/dist/index.d.mts +1200 -43201
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +1859 -7319
- package/dist/index.mjs.map +1 -1
- package/package.json +19 -10
|
@@ -0,0 +1,557 @@
|
|
|
1
|
+
# Data activation clients
|
|
2
|
+
|
|
3
|
+
A **data activation client** is the pipeline join-point that
|
|
4
|
+
binds a **tool** (the transport — see `tools.md`), a **data
|
|
5
|
+
source** (the endpoint — see `data_sources.md`), and one or
|
|
6
|
+
more **interoperability contracts** (the row-to-resource
|
|
7
|
+
transforms — see `interoperability_contracts.md`) into a single
|
|
8
|
+
runnable ingestion pipeline.
|
|
9
|
+
|
|
10
|
+
SDK namespace: `api.dataActivationClients`. The same namespace
|
|
11
|
+
carries both the management surface (CRUD on the binding row,
|
|
12
|
+
§§1–5) and the runtime surface (the verbs that actually move
|
|
13
|
+
data through the configured pipeline, §6). There is no separate
|
|
14
|
+
"data activation runs" resource — runs are verbs on this
|
|
15
|
+
namespace, not a sibling entity.
|
|
16
|
+
|
|
17
|
+
Data activation clients are **Datalake-DB-resident** — the
|
|
18
|
+
parent datalake must be `status: 'ready'` and all referenced
|
|
19
|
+
resources (tool, data source, contracts) must exist before
|
|
20
|
+
POST.
|
|
21
|
+
|
|
22
|
+
## 1. Wire shape
|
|
23
|
+
|
|
24
|
+
```typescript
|
|
25
|
+
import type {
|
|
26
|
+
DataActivationClientRequestWritable,
|
|
27
|
+
DataActivationClientResponse,
|
|
28
|
+
} from '@alvera-ai/platform-sdk'
|
|
29
|
+
|
|
30
|
+
const { data: created } = await api.dataActivationClients.create(
|
|
31
|
+
tenantSlug,
|
|
32
|
+
datalakeSlug,
|
|
33
|
+
{
|
|
34
|
+
name: 'Inbound CSV pipeline',
|
|
35
|
+
description: 'Manual CSV upload feeding the patient and appointment contracts',
|
|
36
|
+
tool_id: manualUploadToolId,
|
|
37
|
+
data_source_id: inboundDataSourceId,
|
|
38
|
+
tool_call: { tool_call_type: 'manual_upload' },
|
|
39
|
+
interoperability_contract_ids: [patientContractId, appointmentContractId],
|
|
40
|
+
cron_expressions: [], // omit for on-demand; e.g. ['0 7,19 * * *']
|
|
41
|
+
loop_over: [], // 'services' | 'locations' | 'providers'
|
|
42
|
+
row_filter: undefined, // Liquid pre-filter; see §2
|
|
43
|
+
downstream_connection_ids: [], // DAC chaining; cycle-checked
|
|
44
|
+
},
|
|
45
|
+
)
|
|
46
|
+
// created.id, created.slug — server-derived
|
|
47
|
+
// created.tool_id === body.tool_id
|
|
48
|
+
// created.interoperability_contracts — embedded list of the bound contracts
|
|
49
|
+
// created.is_default — true for platform-auto-created defaults
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## 2. Rules the type cannot encode
|
|
53
|
+
|
|
54
|
+
### `tool_call.tool_call_type` must align with the bound tool's `tool_body_type`
|
|
55
|
+
|
|
56
|
+
The activation client's `tool_call` discriminator selects which
|
|
57
|
+
call shape will be sent to the tool at run time. The valid
|
|
58
|
+
`tool_call_type` is determined by the bound tool's
|
|
59
|
+
`tool_body_type`:
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
tool_body_type: 'manual_upload' → tool_call_type: 'manual_upload'
|
|
63
|
+
tool_body_type: 'sftp' → tool_call_type: 'sftp_fetch' | ...
|
|
64
|
+
tool_body_type: 'rest_api' → tool_call_type: 'rest_api_request'
|
|
65
|
+
tool_body_type: 'sns' → tool_call_type: 'sms_request'
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
(See `tools.md` §5 for the polymorphic call shape per tool body
|
|
69
|
+
type.) A mismatch — e.g. a `manual_upload` tool bound with an
|
|
70
|
+
`sftp_fetch` call — is rejected at create time as a 422 on
|
|
71
|
+
`/tool_call/tool_call_type`.
|
|
72
|
+
|
|
73
|
+
### `interoperability_contract_ids` is a virtual many-to-many array
|
|
74
|
+
|
|
75
|
+
The field is **caller-supplied** on `Request[Writable]` as a
|
|
76
|
+
`string[]` of contract UUIDs, but the **response** embeds the
|
|
77
|
+
bound contracts as an `interoperability_contracts` array of
|
|
78
|
+
full contract response objects. On PUT, re-supply the
|
|
79
|
+
`interoperability_contract_ids` array verbatim — omitting it
|
|
80
|
+
disassociates ALL bound contracts.
|
|
81
|
+
|
|
82
|
+
Contracts and the activation client's tool / data source must
|
|
83
|
+
all live under the same `datalake_slug`; cross-datalake
|
|
84
|
+
references reject with a 404 on the offending id field.
|
|
85
|
+
|
|
86
|
+
### `interoperability_contract_ids` composition is consumer-decided
|
|
87
|
+
|
|
88
|
+
A single activation client can bind:
|
|
89
|
+
|
|
90
|
+
- one or more `custom` / `system` contracts (e.g. patient +
|
|
91
|
+
appointment for a multi-resource ingestion), OR
|
|
92
|
+
- a single `identity` contract bound to a generic table
|
|
93
|
+
(the auto-created identity contract from
|
|
94
|
+
`interoperability_contracts.md` §5).
|
|
95
|
+
|
|
96
|
+
Mixing `identity` and non-identity contracts on one activation
|
|
97
|
+
client is semantically unusual — fan-out treats each contract
|
|
98
|
+
as a separate transformation lane, and identity contracts pass
|
|
99
|
+
rows through unchanged. The platform doesn't reject the mix,
|
|
100
|
+
but consumers typically split into two activation clients
|
|
101
|
+
(one per output table).
|
|
102
|
+
|
|
103
|
+
### `row_filter` has INVERTED semantics
|
|
104
|
+
|
|
105
|
+
A `row_filter` is a Liquid template that decides whether a row
|
|
106
|
+
enters the activation pipeline AFTER the contract-level
|
|
107
|
+
`filter_template` runs. Like `filter_template`, the semantics
|
|
108
|
+
are inverted from most filter DSLs:
|
|
109
|
+
|
|
110
|
+
- **Empty render → PASS** (row enters the pipeline).
|
|
111
|
+
- **Non-empty render → SKIP** (row is dropped; the rendered
|
|
112
|
+
string is captured as the skip reason).
|
|
113
|
+
|
|
114
|
+
Liquid context includes `msg.row` (the inbound row) and
|
|
115
|
+
`msg.client` (the activation client's template vars). If you
|
|
116
|
+
omit `row_filter`, every row passes through.
|
|
117
|
+
|
|
118
|
+
Anchor on the empty-render-passes rule, not on the Liquid
|
|
119
|
+
keyword — `{% if ... %}...{% endif %}` and
|
|
120
|
+
`{% unless ... %}...{% endunless %}` are equivalent.
|
|
121
|
+
|
|
122
|
+
## 3. Field ownership
|
|
123
|
+
|
|
124
|
+
**Server-derived (Response-only).** Universal set from
|
|
125
|
+
`type_naming.md`, plus:
|
|
126
|
+
|
|
127
|
+
```
|
|
128
|
+
interoperability_contracts array of bound contract response
|
|
129
|
+
objects (mirror of the virtual
|
|
130
|
+
*_ids array on the request)
|
|
131
|
+
is_default boolean — true for platform-auto-
|
|
132
|
+
created default DACs (one per resource
|
|
133
|
+
type at datalake migration); cannot
|
|
134
|
+
be deleted, can be PUT-updated
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
**Caller-supplied (round-trip).**
|
|
138
|
+
|
|
139
|
+
```
|
|
140
|
+
name required string
|
|
141
|
+
description optional string
|
|
142
|
+
tool_id required UUID — bound tool
|
|
143
|
+
data_source_id required UUID — bound data source
|
|
144
|
+
tool_call required embed — { tool_call_type, ... }
|
|
145
|
+
interoperability_contract_ids required string[] — virtual many-to-many
|
|
146
|
+
cron_expressions optional string[] — Crontab syntax (e.g.
|
|
147
|
+
['0 */6 * * *']); empty array = on-demand
|
|
148
|
+
loop_over optional enum[] — 'services' | 'locations' |
|
|
149
|
+
'providers'; fan-out dimensions per run
|
|
150
|
+
row_filter optional string — Liquid pre-filter; see §2
|
|
151
|
+
for inverted semantics
|
|
152
|
+
downstream_connection_ids optional UUID[] — DACs triggered after this
|
|
153
|
+
one completes; cycles rejected at create
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
**Write-only (Request-only).** None — every caller-supplied
|
|
157
|
+
field round-trips on the response (the virtual array
|
|
158
|
+
round-trips under `interoperability_contracts` on response).
|
|
159
|
+
|
|
160
|
+
## 4. Error envelopes
|
|
161
|
+
|
|
162
|
+
Standard JSON:API envelopes per `errors.md`. Common rejections:
|
|
163
|
+
|
|
164
|
+
| `source.pointer` | Cause |
|
|
165
|
+
|-------------------------------------------|----------------------------------------------------|
|
|
166
|
+
| `/tool_id` / `/data_source_id` | resource doesn't exist (or wrong datalake) |
|
|
167
|
+
| `/tool_call/tool_call_type` | mismatch with the tool's `tool_body_type` |
|
|
168
|
+
| `/interoperability_contract_ids/0` | contract id doesn't exist or wrong datalake |
|
|
169
|
+
| `/name` | uniqueness within datalake (POST and PUT — rename can collide) |
|
|
170
|
+
| `/cron_expressions/0` | invalid crontab syntax |
|
|
171
|
+
| `/cron_expressions` | every-minute cron is rejected — **409 Conflict**, not 422; minimum interval enforced platform-side |
|
|
172
|
+
|
|
173
|
+
## 5. Lifecycle
|
|
174
|
+
|
|
175
|
+
### Create
|
|
176
|
+
|
|
177
|
+
Synchronous. The activation client is immediately runnable —
|
|
178
|
+
there's no async "deploy" phase. The next step is invoking one
|
|
179
|
+
of the runtime verbs in §6.
|
|
180
|
+
|
|
181
|
+
### Read shapes
|
|
182
|
+
|
|
183
|
+
```
|
|
184
|
+
.list(tenantSlug, datalakeSlug) Paged: { data, meta }
|
|
185
|
+
.get(tenantSlug, datalakeSlug, slug) One row, by slug
|
|
186
|
+
.metadata(tenantSlug, datalakeSlug, slug) Agent-facing metadata
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
### Update
|
|
190
|
+
|
|
191
|
+
`PUT` replays the full body — no PATCH. Pointer fields
|
|
192
|
+
(`tool_id`, `data_source_id`, `interoperability_contract_ids`)
|
|
193
|
+
can be re-pointed.
|
|
194
|
+
|
|
195
|
+
**The `slug` regenerates from `name` on every PUT.** The URL
|
|
196
|
+
path uses the OLD slug for lookup (so the PUT URL is valid),
|
|
197
|
+
but the response body carries the NEW slug derived from the
|
|
198
|
+
updated `name`. Examples:
|
|
199
|
+
|
|
200
|
+
```
|
|
201
|
+
pre-update: name: 'Original', slug: 'original'
|
|
202
|
+
PUT body: name: 'Renamed', ...
|
|
203
|
+
PUT URL: .../data-activation-clients/original ← old slug works for lookup
|
|
204
|
+
PUT response: name: 'Renamed', slug: 'renamed' ← new slug derived
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
**Consequence**: any downstream resource that references the
|
|
208
|
+
DAC by slug (workflow templates, action bodies referencing
|
|
209
|
+
`{{ ... }}` paths derived from the DAC slug, monitoring
|
|
210
|
+
dashboards, etc.) breaks the moment the DAC is renamed.
|
|
211
|
+
Capture the post-PUT slug from the response, NOT the URL the
|
|
212
|
+
PUT was issued against. If you rename a DAC mid-deployment,
|
|
213
|
+
update every binding that references it.
|
|
214
|
+
|
|
215
|
+
A PUT that keeps `name` unchanged regenerates to the same
|
|
216
|
+
slug (slugification is deterministic), so cosmetic-only
|
|
217
|
+
updates don't break bindings.
|
|
218
|
+
|
|
219
|
+
### Delete
|
|
220
|
+
|
|
221
|
+
`DELETE` removes the activation client. In-flight runs
|
|
222
|
+
decouple their lifecycle from the activation client row;
|
|
223
|
+
deleting one mid-run is generally rejected with a 422.
|
|
224
|
+
|
|
225
|
+
## 6. Runtime
|
|
226
|
+
|
|
227
|
+
Three run shapes exist, selected by the bound tool's
|
|
228
|
+
`tool_body_type`:
|
|
229
|
+
|
|
230
|
+
```
|
|
231
|
+
inline JSON → .ingest(slug, { data }) (small payloads)
|
|
232
|
+
uploaded file → .ingestFile(slug, { key }) (bulk; async)
|
|
233
|
+
pull-based → .runManually(slug) (e.g. SFTP, REST poll)
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
All three produce **log rows** observable through
|
|
237
|
+
`api.dataActivationClients.logs.*` (§6.4), and the canonical way
|
|
238
|
+
to **verify** that ingestion worked is the two-step
|
|
239
|
+
dataset-search pattern in §6.5.
|
|
240
|
+
|
|
241
|
+
### 6.1 Inline JSON ingestion — `.ingest`
|
|
242
|
+
|
|
243
|
+
For activation clients bound to a `manual_upload` tool, the
|
|
244
|
+
smallest run shape is a single JSON row submitted inline:
|
|
245
|
+
|
|
246
|
+
```typescript
|
|
247
|
+
const { data } = await api.dataActivationClients.ingest(
|
|
248
|
+
tenantSlug, datalakeSlug, activationClientSlug,
|
|
249
|
+
{ data: rowAsJson }, // a single row, shaped per the CSV/JSON the contracts expect
|
|
250
|
+
)
|
|
251
|
+
// data.batch_id — UUID; the row's batch identifier (use to find log rows)
|
|
252
|
+
// data.key — storage key for the row's archived ndjson
|
|
253
|
+
// data.jobs_count — number of downstream jobs the platform enqueued
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
The response returns **immediately after enqueueing**, not
|
|
257
|
+
after ingestion completes. To know when the row has actually
|
|
258
|
+
been written, poll the logs (§6.4) or use the verification
|
|
259
|
+
pattern (§6.5).
|
|
260
|
+
|
|
261
|
+
The `batch_id` is the most important field on the response —
|
|
262
|
+
every downstream log row and dataset row carries the same
|
|
263
|
+
`batch_id`, so it's the join key for run-specific verification.
|
|
264
|
+
|
|
265
|
+
### 6.2 Bulk file ingestion — `.ingestFile`
|
|
266
|
+
|
|
267
|
+
For larger payloads, upload the file via a presigned URL first,
|
|
268
|
+
then trigger ingestion by the returned key:
|
|
269
|
+
|
|
270
|
+
```typescript
|
|
271
|
+
// 1. Mint a presigned upload URL (see datalakes.md).
|
|
272
|
+
const { data: link } = await api.datalakes.createUploadLink(
|
|
273
|
+
tenantSlug, datalakeSlug,
|
|
274
|
+
{ content_type: 'text/csv', filename: 'inbound.csv' },
|
|
275
|
+
)
|
|
276
|
+
// link.url — presigned PUT URL (signature + expiry baked in)
|
|
277
|
+
// link.key — storage key the platform will reference
|
|
278
|
+
|
|
279
|
+
// 2. PUT the file body to the presigned URL (raw HTTP, NOT via the SDK).
|
|
280
|
+
await fetch(link.url, {
|
|
281
|
+
method: 'PUT',
|
|
282
|
+
headers: { 'Content-Type': 'text/csv' },
|
|
283
|
+
body: csvBytes,
|
|
284
|
+
})
|
|
285
|
+
|
|
286
|
+
// 3. Enqueue the ingest-file job by key.
|
|
287
|
+
const { data: job } = await api.dataActivationClients.ingestFile(
|
|
288
|
+
tenantSlug, datalakeSlug, activationClientSlug,
|
|
289
|
+
{ key: link.key },
|
|
290
|
+
)
|
|
291
|
+
// job.key — echoes link.key
|
|
292
|
+
// job.job_id — numeric id of the enqueued background job
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
This path is **fully async** — the response acknowledges that
|
|
296
|
+
the worker has been enqueued, but no row has been written yet.
|
|
297
|
+
The worker reads the file, allocates a fresh `batch_id`, and
|
|
298
|
+
fans out per-row jobs. Discover the assigned `batch_id` by
|
|
299
|
+
diffing the activation client's log rows before vs after the
|
|
300
|
+
call (§6.4).
|
|
301
|
+
|
|
302
|
+
### 6.3 Manual run — `.runManually`
|
|
303
|
+
|
|
304
|
+
For activation clients bound to pull-based tools (`sftp`,
|
|
305
|
+
`rest_api`, etc.), there is no caller-supplied payload — the
|
|
306
|
+
tool itself fetches data from the external system. The SDK
|
|
307
|
+
verb:
|
|
308
|
+
|
|
309
|
+
```typescript
|
|
310
|
+
const { data } = await api.dataActivationClients.runManually(
|
|
311
|
+
tenantSlug, datalakeSlug, activationClientSlug,
|
|
312
|
+
)
|
|
313
|
+
```
|
|
314
|
+
|
|
315
|
+
Returns an acknowledgement; the actual fetch runs in a
|
|
316
|
+
background worker. As with `.ingestFile`, discover the assigned
|
|
317
|
+
`batch_id` by polling the logs.
|
|
318
|
+
|
|
319
|
+
#### One-shot `tool_call` override
|
|
320
|
+
|
|
321
|
+
`.runManually` accepts an optional `tool_call` body that
|
|
322
|
+
overrides the DAC's persisted `tool_call` FOR THIS RUN ONLY.
|
|
323
|
+
Omitting the body (or omitting just the `tool_call` field) runs
|
|
324
|
+
with the persisted `tool_call`. Supplying a `tool_call` map
|
|
325
|
+
does NOT mutate the DAC row — it's a transient per-invocation
|
|
326
|
+
override, useful for one-off backfills with a different path,
|
|
327
|
+
payload, or fetch parameters.
|
|
328
|
+
|
|
329
|
+
```typescript
|
|
330
|
+
await api.dataActivationClients.runManually(
|
|
331
|
+
tenantSlug, datalakeSlug, activationClientSlug,
|
|
332
|
+
{ tool_call: { tool_call_type: 'rest_api_request', path: '/v1/backfill', ... } },
|
|
333
|
+
)
|
|
334
|
+
```
|
|
335
|
+
|
|
336
|
+
### 6.4 Logs subresource — `.logs.list` / `.logs.get`
|
|
337
|
+
|
|
338
|
+
Every ingestion run produces one or more **log rows** — one per
|
|
339
|
+
top-level interoperability contract that fired, plus one per
|
|
340
|
+
cascade target (a nested resource produced by a template's
|
|
341
|
+
upserts). The log row carries:
|
|
342
|
+
|
|
343
|
+
```
|
|
344
|
+
batch_id UUID — the run's batch identifier
|
|
345
|
+
dataset_table string — the target dataset/table name
|
|
346
|
+
rows_ingested number — count of top-level rows the contract emitted
|
|
347
|
+
(cascade rows carry 0)
|
|
348
|
+
dataset_updated number — count of dataset rows whose batch_id this run wrote
|
|
349
|
+
(load-bearing: see Gotcha 7)
|
|
350
|
+
output_files string[] — merged ndjson archive paths
|
|
351
|
+
(empty until the post-batch merge worker commits)
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
```typescript
|
|
355
|
+
// Paginated list of log rows for an activation client.
|
|
356
|
+
const { data } = await api.dataActivationClients.logs.list(
|
|
357
|
+
tenantSlug, datalakeSlug, activationClientSlug,
|
|
358
|
+
)
|
|
359
|
+
// data.data — array of log rows; filter by batch_id to find this run
|
|
360
|
+
|
|
361
|
+
// Fetch one log row by id.
|
|
362
|
+
const { data: entry } = await api.dataActivationClients.logs.get(
|
|
363
|
+
tenantSlug, datalakeSlug, activationClientSlug, logId,
|
|
364
|
+
)
|
|
365
|
+
```
|
|
366
|
+
|
|
367
|
+
#### Top-level rows vs cascade rows
|
|
368
|
+
|
|
369
|
+
A multi-resource contract (e.g. an inbound row that produces a
|
|
370
|
+
patient AND an appointment AND a related person) fires one
|
|
371
|
+
log row **per dataset_table** the contract touched. The
|
|
372
|
+
top-level contracts carry `rows_ingested > 0`; cascade rows
|
|
373
|
+
(nested upserts inside a template) carry `rows_ingested = 0`.
|
|
374
|
+
Filter on `rows_ingested > 0` when counting "how many top-level
|
|
375
|
+
contracts fired".
|
|
376
|
+
|
|
377
|
+
#### `output_files` as the completion signal
|
|
378
|
+
|
|
379
|
+
A log row appears as soon as the per-row job acknowledges the
|
|
380
|
+
write, but `output_files` stays empty until the post-batch
|
|
381
|
+
merge worker has committed the ndjson archive. Use
|
|
382
|
+
`output_files.length > 0` as the "this batch's rows are fully
|
|
383
|
+
durable" signal — not just the presence of the log row.
|
|
384
|
+
|
|
385
|
+
### 6.5 Verifying ingestion: the two-step dataset search
|
|
386
|
+
|
|
387
|
+
The canonical way to **prove a row landed in the datalake** is
|
|
388
|
+
to query the dataset directly. This is a two-step pattern:
|
|
389
|
+
|
|
390
|
+
The dataset endpoints are datalake-slug-scoped: every call
|
|
391
|
+
takes `(tenantSlug, datalakeSlug, dataset, …)` as leading args.
|
|
392
|
+
There is no `datalake_id` query param and no tenant-derived
|
|
393
|
+
default — the datalake is mandatory and explicit, mirroring
|
|
394
|
+
`tools`, `workflows`, `dataActivationClients`, etc.
|
|
395
|
+
|
|
396
|
+
```typescript
|
|
397
|
+
// (1) POST creates a user search and runs the underlying SQL
|
|
398
|
+
// synchronously against the regulated schema. The result
|
|
399
|
+
// set is materialised server-side and identified by the
|
|
400
|
+
// returned UUID.
|
|
401
|
+
const { data: search } = await api.datasets.createUserSearch(
|
|
402
|
+
tenantSlug,
|
|
403
|
+
datalakeSlug,
|
|
404
|
+
'patient', // dataset name
|
|
405
|
+
{
|
|
406
|
+
search_query: `ri.value = '${externalPatientId}'` // SQL WHERE-clause fragment
|
|
407
|
+
+ ` AND rp.batch_id = '${batchId}'`,
|
|
408
|
+
},
|
|
409
|
+
)
|
|
410
|
+
// search.id — UUID of the materialised result set
|
|
411
|
+
// search.status — 'completed' on success; 'failed' surfaces error_message
|
|
412
|
+
// search.results_count — ALWAYS null on the POST response (see Gotcha 5)
|
|
413
|
+
|
|
414
|
+
// (2) GET paginates rows from the materialised result set.
|
|
415
|
+
// dataAccessMode selects the storage tier:
|
|
416
|
+
// 'regulated' — raw values verbatim (PHI/PII intact)
|
|
417
|
+
// 'unregulated' — tokenized columns return as
|
|
418
|
+
// { redacted_data, token, type } objects;
|
|
419
|
+
// plain columns return as their primitive type
|
|
420
|
+
const { data: page } = await api.datasets.search(
|
|
421
|
+
tenantSlug,
|
|
422
|
+
datalakeSlug,
|
|
423
|
+
'patient',
|
|
424
|
+
{ userSearchId: search.id, dataAccessMode: 'unregulated' },
|
|
425
|
+
)
|
|
426
|
+
// page.data — array of rows
|
|
427
|
+
// page.meta — pagination
|
|
428
|
+
```
|
|
429
|
+
|
|
430
|
+
#### SQL aliases
|
|
431
|
+
|
|
432
|
+
`createUserSearch` accepts a WHERE-clause fragment against a
|
|
433
|
+
fixed set of table aliases. Each dataset has a canonical base
|
|
434
|
+
alias for its regulated table, plus `ri` for the
|
|
435
|
+
regulated_identifiers join:
|
|
436
|
+
|
|
437
|
+
```
|
|
438
|
+
patient → rp (regulated_patients) + ri
|
|
439
|
+
appointment → ra (regulated_appointments) + ri
|
|
440
|
+
customer → rc (regulated_customers) + ri
|
|
441
|
+
... (one alias per regulated dataset)
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
The `ri.value` predicate filters on external identifiers (the
|
|
445
|
+
values your inbound rows carried); the `<base_alias>.batch_id`
|
|
446
|
+
predicate scopes the search to one run.
|
|
447
|
+
|
|
448
|
+
#### Tokenized vs plain column shape
|
|
449
|
+
|
|
450
|
+
In `dataAccessMode: 'unregulated'`, tokenized columns surface
|
|
451
|
+
as objects, NOT primitives:
|
|
452
|
+
|
|
453
|
+
```typescript
|
|
454
|
+
{
|
|
455
|
+
redacted_data: '*****', // server-redacted preview
|
|
456
|
+
token: 'tk_...', // stable opaque token
|
|
457
|
+
type: 'string' // original column's logical type — one of:
|
|
458
|
+
| 'text' // 'string' / 'text': free-form text content
|
|
459
|
+
| 'email' // 'email': RFC 5322 email
|
|
460
|
+
| 'phone' // 'phone': E.164 / display phone number
|
|
461
|
+
| 'date' // 'date': ISO 8601 date
|
|
462
|
+
| 'image', // 'image': image file reference
|
|
463
|
+
}
|
|
464
|
+
```
|
|
465
|
+
|
|
466
|
+
Plain columns (non-tokenized) surface as the same primitive
|
|
467
|
+
the regulated schema stores. A consumer code path that
|
|
468
|
+
unconditionally treats every field as a string will crash on
|
|
469
|
+
tokenized columns — type-check `typeof row.field === 'object'`
|
|
470
|
+
first.
|
|
471
|
+
|
|
472
|
+
## 7. Gotchas
|
|
473
|
+
|
|
474
|
+
1. **`tool_call_type` must match the tool's `tool_body_type`.**
|
|
475
|
+
An activation client is a contract between the tool and the
|
|
476
|
+
pipeline; sending a call shape the tool doesn't accept is
|
|
477
|
+
rejected at create. See `tools.md` §5 for the per-body
|
|
478
|
+
call-shape table.
|
|
479
|
+
|
|
480
|
+
2. **`interoperability_contract_ids` on PUT replaces the bound
|
|
481
|
+
set entirely.** Omitting the field disassociates ALL
|
|
482
|
+
contracts; passing a subset removes the others. Always
|
|
483
|
+
re-supply the full id array on update, even if you're only
|
|
484
|
+
changing `description`.
|
|
485
|
+
|
|
486
|
+
3. **All referenced resources live under one datalake.** The
|
|
487
|
+
activation client, its tool, its data source, and every
|
|
488
|
+
bound contract must share the same parent datalake.
|
|
489
|
+
Cross-datalake references return 404 on the offending id.
|
|
490
|
+
|
|
491
|
+
4. **Custom-dataset identity contracts get their own
|
|
492
|
+
activation client.** When you ingest into a generic table,
|
|
493
|
+
bind the table's auto-created identity contract (see
|
|
494
|
+
`interoperability_contracts.md` §5) to a dedicated activation
|
|
495
|
+
client. Don't mix it with non-identity contracts on the
|
|
496
|
+
same client unless you specifically want the fan-out
|
|
497
|
+
semantics.
|
|
498
|
+
|
|
499
|
+
5. **`createUserSearch.results_count` is always null.** The
|
|
500
|
+
POST response confirms the search materialised but does
|
|
501
|
+
NOT report row counts. The actual rows live in the
|
|
502
|
+
server-side result set and only become visible via the
|
|
503
|
+
paginated `.search` GET. A common mistake is reading
|
|
504
|
+
`results_count` and concluding "no rows matched" when the
|
|
505
|
+
POST simply doesn't populate that field.
|
|
506
|
+
|
|
507
|
+
6. **`.ingest` returns BEFORE the row is durable.** The
|
|
508
|
+
response carries `batch_id` + `jobs_count`, signalling that
|
|
509
|
+
downstream jobs have been enqueued — not that they've
|
|
510
|
+
completed. Poll logs (or use the search verification
|
|
511
|
+
pattern in §6.5) before treating the row as ingested.
|
|
512
|
+
|
|
513
|
+
7. **`dataset_updated > 0` is the load-bearing freshness
|
|
514
|
+
gate.** `rows_ingested` flips as soon as the per-row job
|
|
515
|
+
completes, but the dataset row's `batch_id` column is
|
|
516
|
+
updated by a downstream trigger that can lag. Filtering a
|
|
517
|
+
`.createUserSearch` by `<base_alias>.batch_id = '<this>'`
|
|
518
|
+
BEFORE that trigger fires returns an empty result set,
|
|
519
|
+
and there's no way to refresh the materialised result —
|
|
520
|
+
you'd have to POST a new search. Poll for
|
|
521
|
+
`dataset_updated > 0` before issuing the search.
|
|
522
|
+
|
|
523
|
+
8. **`output_files` empty ≠ batch failed.** A log row with
|
|
524
|
+
`rows_ingested > 0` but `output_files: []` is mid-merge,
|
|
525
|
+
not broken. The merge worker commits the ndjson archives
|
|
526
|
+
asynchronously; wait for `output_files.length > 0` if your
|
|
527
|
+
downstream code reads the archives directly.
|
|
528
|
+
|
|
529
|
+
9. **Bulk uploads PUT to a presigned URL, NOT through the
|
|
530
|
+
SDK.** Step 2 of the bulk flow uses raw `fetch()` against
|
|
531
|
+
the presigned URL the platform returns; the SDK is NOT
|
|
532
|
+
involved in the upload itself. The SDK round-trip only
|
|
533
|
+
covers minting the URL (`createUploadLink`) and
|
|
534
|
+
triggering ingestion by key (`ingestFile`).
|
|
535
|
+
|
|
536
|
+
10. **`runManually` and `.ingest` produce structurally
|
|
537
|
+
identical log rows.** Once a `batch_id` is allocated,
|
|
538
|
+
every run flavor merges into the same log/dataset row
|
|
539
|
+
shape. Consumers that diff or aggregate runs don't need to
|
|
540
|
+
branch on which verb produced the batch.
|
|
541
|
+
|
|
542
|
+
11. **`row_filter` is empty-passes, non-empty-skips.** Same
|
|
543
|
+
inverted semantics as InteroperabilityContract.filter_template.
|
|
544
|
+
A correctly-shaped filter renders to a non-empty string when
|
|
545
|
+
the row should be SKIPPED. Always anchor on the
|
|
546
|
+
empty-render-passes rule, not on the Liquid keyword choice.
|
|
547
|
+
|
|
548
|
+
12. **`downstream_connection_ids` is cycle-checked at create.**
|
|
549
|
+
The platform rejects DAC chains that would form a cycle. Make
|
|
550
|
+
sure both endpoints of the chain exist before adding the
|
|
551
|
+
pointer; cross-datalake chaining is also rejected.
|
|
552
|
+
|
|
553
|
+
13. **`is_default: true` DACs cannot be deleted.** Platform-
|
|
554
|
+
auto-created default DACs (one per resource type at datalake
|
|
555
|
+
migration time) are protected. Filter them out
|
|
556
|
+
(`!d.is_default`) when listing for management purposes; you
|
|
557
|
+
can still PUT-update them (e.g. to bind additional contracts).
|