@alvera-ai/platform-sdk 0.18.1 → 0.18.2-next.g6192a5c

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,7 @@
5
5
  to the compliance-screenings table).
6
6
 
7
7
  Source: GET /api/compliance-screenings on Atomic FI (paginated; the
8
- fetch pipeline extracts the `data` array via the response_extractor and
8
+ fetch pipeline pulls the `data` array out with `rows_from_file_template` and
9
9
  streams one row per `msg`).
10
10
 
11
11
  GH-859 removed the `compliance_screening` resource, and with it the
@@ -5,7 +5,7 @@
5
5
  to the payment-accounts table). One JSON field per GT column.
6
6
 
7
7
  Source: GET /api/payment-accounts on Atomic FI (paginated; the fetch
8
- pipeline extracts the `data` array via the response_extractor and
8
+ pipeline pulls the `data` array out with `rows_from_file_template` and
9
9
  streams one row per `msg`).
10
10
 
11
11
  GH-859 removed the `payment_account` resource along with every other
@@ -7,7 +7,7 @@ vitest_source:
7
7
  - integration-tests/tests/organic-marketing/agent-driven-workflow.test.ts
8
8
  - integration-tests/tests/organic-marketing/generic-tables.test.ts
9
9
  - integration-tests/tests/organic-marketing/manual-upload-tool.test.ts
10
- - integration-tests/tests/workspace/bootstrap.test.ts
10
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
11
11
  status: draft
12
12
  ---
13
13
 
@@ -221,8 +221,8 @@ per run is the same leak as a fresh tenant per run.
221
221
 
222
222
  ```typescript
223
223
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
224
- const DB_SCHEMA = 'cookbook_organic_marketing'
225
- const LAKE_NAME = 'Cookbook Organic Marketing Datalake'
224
+ const DB_SCHEMA = `cookbook_organic_marketing_${runSuffix}`
225
+ const LAKE_NAME = `Cookbook Organic Marketing Datalake ${runSuffix}`
226
226
 
227
227
  const S3 = {
228
228
  cloud_storage_type: 'aws' as const,
@@ -12,7 +12,7 @@ vitest_source:
12
12
  - integration-tests/tests/payments-compliance/system-templates.test.ts
13
13
  - integration-tests/tests/payments-compliance/manual-upload-tool.test.ts
14
14
  - integration-tests/tests/payments-compliance/data-sources.test.ts
15
- - integration-tests/tests/workspace/bootstrap.test.ts
15
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
16
16
  status: draft
17
17
  ---
18
18
 
@@ -228,8 +228,8 @@ per run is the same leak as a fresh tenant per run.
228
228
 
229
229
  ```typescript
230
230
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_payments' }
231
- const DB_SCHEMA = 'cookbook_payments_compliance'
232
- const LAKE_NAME = 'Cookbook Payments Compliance Datalake'
231
+ const DB_SCHEMA = `cookbook_payments_compliance_${runSuffix}`
232
+ const LAKE_NAME = `Cookbook Payments Compliance Datalake ${runSuffix}`
233
233
 
234
234
  const S3 = {
235
235
  cloud_storage_type: 'aws' as const,
@@ -7,7 +7,7 @@ vitest_source:
7
7
  - integration-tests/tests/primary-care-feedback/standard-workflow.test.ts
8
8
  - integration-tests/tests/primary-care-feedback/agent-driven-workflow.test.ts
9
9
  - integration-tests/tests/primary-care-feedback/ai-agent-invoke.test.ts
10
- - integration-tests/tests/workspace/bootstrap.test.ts
10
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
11
11
  status: draft
12
12
  ---
13
13
 
@@ -234,8 +234,8 @@ run is the same leak as a fresh tenant per run.
234
234
 
235
235
  ```typescript
236
236
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
237
- const DB_SCHEMA = 'cookbook_primary_care'
238
- const LAKE_NAME = 'Cookbook Primary Care Datalake'
237
+ const DB_SCHEMA = `cookbook_primary_care_${runSuffix}`
238
+ const LAKE_NAME = `Cookbook Primary Care Datalake ${runSuffix}`
239
239
 
240
240
  const S3 = {
241
241
  cloud_storage_type: 'aws' as const,
@@ -10,7 +10,7 @@ vitest_source:
10
10
  - integration-tests/tests/subscription-saas/run-dac-bulk.test.ts
11
11
  - integration-tests/tests/subscription-saas/run-dac-fetch.test.ts
12
12
  - integration-tests/tests/subscription-saas/invite-team.test.ts
13
- - integration-tests/tests/workspace/bootstrap.test.ts
13
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
14
14
  status: draft
15
15
  ---
16
16
 
@@ -249,8 +249,8 @@ run is the same leak as a fresh tenant per run.
249
249
 
250
250
  ```typescript
251
251
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
252
- const DB_SCHEMA = 'cookbook_subscription_saas'
253
- const LAKE_NAME = 'Cookbook Subscription SaaS Datalake'
252
+ const DB_SCHEMA = `cookbook_subscription_saas_${runSuffix}`
253
+ const LAKE_NAME = `Cookbook Subscription SaaS Datalake ${runSuffix}`
254
254
 
255
255
  const S3 = {
256
256
  cloud_storage_type: 'aws' as const,
@@ -2002,9 +2002,10 @@ The client's `tool_call` is what turns a REST tool into a fetch: the HTTP
2002
2002
  `pagination_context_template` declaring whether there is more to get. This
2003
2003
  endpoint answers in one page, so it declares `has_next: false`.
2004
2004
 
2005
- The `response_extractor` unwraps the API's envelope. Stripe answers
2006
- `{ object: "list", data: [...] }`, and each element of that array has to
2007
- become one row — `{{ msg.data | to_json }}` is the whole of it.
2005
+ `rows_from_file_template` says where the rows are. Stripe answers
2006
+ `{ object: "list", data: [...] }` — one document that has to become three
2007
+ rows — so the template binds `msg` to that whole document and renders the
2008
+ array inside it: `{{ msg.data | to_json }}` is the whole of it.
2008
2009
 
2009
2010
  And the pair is split across two clients again. Nothing about the fetch
2010
2011
  changes the reason: two contracts on one client resolve the same subject at
@@ -2018,7 +2019,7 @@ const restFetchCall = {
2018
2019
  path: { type: 'custom' as const, body: '/v1/customers' },
2019
2020
  pagination_context_template: { type: 'custom' as const, body: '{"has_next": false}' },
2020
2021
  }
2021
- const restExtractor = { type: 'custom' as const, body: '{{ msg.data | to_json }}' }
2022
+ const restRowsTemplate = { type: 'custom' as const, body: '{{ msg.data | to_json }}' }
2022
2023
 
2023
2024
  const restIdentityDac = await ctx.ensure(
2024
2025
  'REST identity client',
@@ -2033,7 +2034,7 @@ const restIdentityDac = await ctx.ensure(
2033
2034
  tool_id: ctx.restToolId,
2034
2035
  data_source_id: ctx.dataSourceId,
2035
2036
  tool_call: restFetchCall,
2036
- response_extractor: restExtractor,
2037
+ rows_from_file_template: restRowsTemplate,
2037
2038
  interoperability_contract_ids: [ctx.leContractId],
2038
2039
  })
2039
2040
  return data
@@ -2054,7 +2055,7 @@ const restDataDac = await ctx.ensure(
2054
2055
  tool_id: ctx.restToolId,
2055
2056
  data_source_id: ctx.dataSourceId,
2056
2057
  tool_call: restFetchCall,
2057
- response_extractor: restExtractor,
2058
+ rows_from_file_template: restRowsTemplate,
2058
2059
  interoperability_contract_ids: [ctx.gtContractId],
2059
2060
  })
2060
2061
  return data
@@ -164,6 +164,70 @@ Anchor on the empty-render-passes rule, not on the Liquid
164
164
  keyword — `{% if ... %}...{% endif %}` and
165
165
  `{% unless ... %}...{% endunless %}` are equivalent.
166
166
 
167
+ ### `rows_from_file_template` says HOW MANY rows, never what is in them
168
+
169
+ A fetch lands a file in storage. `rows_from_file_template` is a
170
+ Liquid template that turns that file into the list of rows to
171
+ ingest. It may be set on **any** file — the content type decides
172
+ only how the file is DECODED, never whether a template is
173
+ allowed — and a run is never refused for declaring one.
174
+
175
+ `msg` binds to the decoded **whole file**: one value per file,
176
+ never a row. What it holds depends on how the fetch stored the
177
+ file:
178
+
179
+ | stored as | `msg` is |
180
+ |------------------------|-------------------------------------------|
181
+ | `application/json` | the parsed document |
182
+ | `application/x-ndjson` | the list of every parsed line |
183
+ | `text/csv` | the list of row maps, keyed by header |
184
+ | PDF, PNG, JPEG, WebP | `{ r2_key, content_type }` — a pointer, bytes undecoded |
185
+
186
+ The template must render a **JSON array**. Liquid renders text,
187
+ so any value has to be serialised on the way out; `to_json` is
188
+ the filter that does it, and there is no inverse — nothing parses
189
+ a JSON string back into a value inside the template.
190
+
191
+ Three shapes cover nearly everything:
192
+
193
+ - **one document becomes many rows** —
194
+ `{{ msg.data | to_json }}`
195
+ - **many rows become fewer** —
196
+ `{% assign kept = msg | where: "status", "active" %}{{ kept | to_json }}`
197
+ - **a whole file becomes one row** — `[{{ msg | to_json }}]`.
198
+ The brackets are what make it a list; this is the usual shape
199
+ for a PDF or an image, where `msg` is already the pointer the
200
+ contract needs.
201
+
202
+ **Leave it unset when the file is already the rows.** A
203
+ `sql_query` export, a CSV, and most `sftp` / `s3` bodies land as
204
+ rows already. Unset means take them as they are — the common
205
+ case, and it streams a chunk at a time instead of holding the
206
+ file in memory, which a template cannot do because it has to see
207
+ the whole file at once.
208
+
209
+ **The template decides how many rows there are; the contract
210
+ decides what is in each one.** A contract runs per row and can
211
+ never change the count, which is the one thing only this
212
+ template can do. Renaming or reshaping fields belongs in the
213
+ contract (`interoperability_contracts.md`), not here.
214
+
215
+ It runs on `.runManually`, on `.ingestFile`, and on every cron
216
+ fire — every path that fetches a file. It does **not** run on
217
+ `.ingest`, which takes rows inline and has no file to decode.
218
+
219
+ Extraction happens in a background job, so a template that fails
220
+ never reaches the HTTP response — `.runManually` still returns
221
+ its `batch_id`. The failure surfaces as a `data_activation_logs`
222
+ row you read back through `.logs.list` (§6.4), and in the
223
+ platform's error log under `error_code: rows_template_failed`.
224
+
225
+ **Renamed.** This field was `response_extractor` until
226
+ 2026-09-14; manifests and contracts written against the old name
227
+ need updating. The tool-level `response_extractor` (`tools.md`)
228
+ is a different field doing a different job — it maps a model
229
+ provider's envelope into a canonical result — and keeps its name.
230
+
167
231
  ## 3. Field ownership
168
232
 
169
233
  **Server-derived (Response-only).** Universal set from
@@ -199,18 +263,17 @@ downstream_connection_ids optional UUID[] — DACs triggered after this
199
263
  filter_config optional embed — TemplateConfig; server-pinned
200
264
  output_schema (enum ["true","false"]). A row-
201
265
  level filter evaluated per fetched row
202
- response_extractor optional embed — TemplateConfig ({ type, body });
203
- a Liquid template that unwraps a nested API
204
- envelope into the row array (e.g.
205
- `{{ msg.data | to_json }}`). Load-bearing for
206
- `rest_api` fetch clients; see
207
- `cookbook/subscription-saas.md` §032–035
266
+ rows_from_file_template optional embed — TemplateConfig ({ type, body });
267
+ a Liquid template that says where the rows are
268
+ inside a fetched file. It may be set on ANY
269
+ file and is never refused — see §2 for what
270
+ `msg` binds to and when to leave it unset
208
271
  ```
209
272
 
210
273
  **Write-only (Request-only).** None — every caller-supplied
211
274
  field round-trips on the response (the virtual array
212
275
  round-trips under `interoperability_contracts` on response).
213
- `filter_config` / `response_extractor` come back under the read-only
276
+ `filter_config` / `rows_from_file_template` come back under the read-only
214
277
  `SimpleTemplateConfigResponse` shape (their `output_schema` is
215
278
  server-pinned and not echoed).
216
279
 
package/.agent/mdm.md CHANGED
@@ -185,6 +185,61 @@ is a 404, not a `'not_verified'`. The distinction matters:
185
185
  A consumer that conflates the two will mis-route an identity
186
186
  challenge as a missing-record error.
187
187
 
188
+ ### One label per kind of identifier a subject holds
189
+
190
+ An identification attaches to a legal entity under a label — its `uri` —
191
+ alongside an `id_type`. The pair `(uri, id_type)` is a **slot** on that
192
+ subject, and a subject holds each slot once. Two identifications under the
193
+ same pair are refused, not merged and not resolved to a winner.
194
+
195
+ So a subject that carries more than one identifier needs a label per kind.
196
+ Jane, from a Shopify store, with an email and a phone:
197
+
198
+ ```json
199
+ "identifications": [
200
+ { "id_type": "digital_identifier",
201
+ "uri": "shopify",
202
+ "id_number": "jane@example.com" },
203
+ { "id_type": "digital_identifier",
204
+ "uri": "shopify::phone",
205
+ "id_number": "+15551234" }
206
+ ]
207
+ ```
208
+
209
+ Written with `"uri": "shopify"` on both, the second is refused — the database
210
+ enforces it as `legal_entity_identifications_entity_uri_type_uk`.
211
+
212
+ The convention: the bare source identifies the **subject**; anything else the
213
+ subject carries gets its own label beneath it.
214
+
215
+ **Do not re-label identifications that already exist.** Changing the `uri` on a
216
+ recipe that has already run sends the next ingest to a DIFFERENT slot, which
217
+ adds a second identification rather than updating the first — the original is
218
+ orphaned, silently, and both then answer to the same person.
219
+
220
+ ### What a resolve does to the slots it does not name
221
+
222
+ A resolve speaks only about the slots it names:
223
+
224
+ | the slot you send | what happens |
225
+ |---|---|
226
+ | named, already filled | **overwritten** |
227
+ | named, empty | **added** |
228
+ | **not mentioned** | **kept, untouched** |
229
+
230
+ The third row is the one worth designing around, and the one nothing else
231
+ states. Several lanes can each resolve the same subject knowing only their own
232
+ identifiers, without destroying each other's work: a marketing lane sending
233
+ only an email and a billing lane sending only a phone both survive, in either
234
+ order.
235
+
236
+ Two limits that follow from the same rule:
237
+
238
+ - **A slot cannot be emptied through a resolve.** Omitting it means "leave it";
239
+ there is no value that means "remove it".
240
+ - **Two values under one `(uri, id_type)` in a single payload are refused**
241
+ rather than resolved to a winner — the platform does not pick.
242
+
188
243
  ## 3. Field ownership
189
244
 
190
245
  **Server-derived (Response-only).**
@@ -492,6 +492,73 @@ actions required array — inline; replace-on-PUT
492
492
  workflow_ai_agents optional array — inline; replace-on-PUT (see §3)
493
493
  ```
494
494
 
495
+ Caller-supplied fields **inside each `context_datasets[*]`** row:
496
+
497
+ ```
498
+ dataset_type required string — CLOSED ENUM, not free text. An
499
+ arbitrary value 422s at apply and
500
+ the error enumerates the allowed
501
+ types
502
+ generic_table_id required UUID when dataset_type == 'generic_table';
503
+ omit otherwise. NOT the same field
504
+ as the workflow's own
505
+ `generic_table_id` one level up
506
+ where_clause optional string — Liquid, max 10_000 chars (see
507
+ below for what it can reach, and
508
+ which tier it runs against)
509
+ limit optional int — must be positive
510
+ position optional int — default 0
511
+ ```
512
+
513
+ ### Reading what a context query returned
514
+
515
+ `additional_context.<dataset_type>` — the `dataset_type` string **literally**,
516
+ not the table name and not a slug. A `document` context is read at
517
+ `additional_context.document`.
518
+
519
+ Do not confuse this with the AI-agent form documented above
520
+ (`additional_context.<agent_slug>.<field>`): an agent is keyed by its slug, a
521
+ context dataset by its type.
522
+
523
+ ### ⚠ Two context datasets of the same `dataset_type` silently overwrite
524
+
525
+ The loader accumulates them into a map keyed by `dataset_type`, so declaring
526
+ two of the same type leaves **only the last one**. There is no error and no
527
+ warning — the workflow runs happily against the wrong context. If you need two
528
+ slices of one type, express it as one query with a wider `where_clause`.
529
+
530
+ ### The `where_clause` reaches the row differently than the filter does
531
+
532
+ This asymmetry is the easiest thing here to get wrong:
533
+
534
+ ```
535
+ where clause {{ <dataset_type>.field }} e.g. {{ generic_table.external_client_id }}
536
+ filter {{ event_dataset.field }} the SAME row, different accessor
537
+ ```
538
+
539
+ That makes three conventions for "the current row" across this corpus — the
540
+ third being `msg.row.x` in an interoperability filter versus `msg.x` in its
541
+ transform. Nothing about a name tells you which one you are in; the surface you
542
+ are authoring does.
543
+
544
+ ### Context queries run against the REGULATED tier
545
+
546
+ The loader reads context in regulated mode, so a `where_clause` must name the
547
+ **regulated** table alias (e.g. `regulated_alvera_custom_unsubscribe_list`),
548
+ not the unregulated name. This changes every where clause you write, and a
549
+ clause written against the unregulated name finds nothing rather than failing
550
+ loudly.
551
+
552
+ ### `context_datasets` cannot be cleared by omission
553
+
554
+ It is a **required, replace-on-PUT array**. Dropping the block from a manifest
555
+ does not clear it: the server retains what was last deployed, the rendered and
556
+ deployed checksums never converge, and every subsequent `plan` reports drift on
557
+ a config that was just applied. Send an explicit `[]` to mean "no context".
558
+
559
+ Same class as the `assume_role_external_id` convergence note — an omission that
560
+ reads as "leave it alone" on the wire and as "remove it" in the author's head.
561
+
495
562
  Caller-supplied fields **inside each `actions[*]`** row:
496
563
 
497
564
  ```
package/README.md CHANGED
@@ -111,7 +111,7 @@ block of the committed OpenAPI spec and exported as `ENVIRONMENTS`:
111
111
  import { ENVIRONMENTS, DEFAULT_ENVIRONMENT } from '@alvera-ai/platform-sdk'
112
112
 
113
113
  ENVIRONMENTS.local.base_url // http://localhost:4000
114
- ENVIRONMENTS.demo.base_url // https://platform-hh.alvera.ai
114
+ ENVIRONMENTS.demo.base_url // https://demo.alvera.ai
115
115
  ENVIRONMENTS.prod.base_url // https://app.alvera.ai
116
116
 
117
117
  DEFAULT_ENVIRONMENT // 'prod'
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@alvera-ai/platform-sdk",
3
- "version": "0.18.1",
3
+ "version": "0.18.2-next.g6192a5c",
4
4
  "description": "Typed SDK for the Alvera platform API — manage data sources, tools, generic tables, AI agents, and action status updaters.",
5
5
  "type": "module",
6
6
  "sideEffects": false,