@alvera-ai/platform-sdk 0.18.2 → 0.19.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,7 @@
5
5
  to the compliance-screenings table).
6
6
 
7
7
  Source: GET /api/compliance-screenings on Atomic FI (paginated; the
8
- fetch pipeline extracts the `data` array via the response_extractor and
8
+ fetch pipeline pulls the `data` array out with `rows_from_file_template` and
9
9
  streams one row per `msg`).
10
10
 
11
11
  GH-859 removed the `compliance_screening` resource, and with it the
@@ -5,7 +5,7 @@
5
5
  to the payment-accounts table). One JSON field per GT column.
6
6
 
7
7
  Source: GET /api/payment-accounts on Atomic FI (paginated; the fetch
8
- pipeline extracts the `data` array via the response_extractor and
8
+ pipeline pulls the `data` array out with `rows_from_file_template` and
9
9
  streams one row per `msg`).
10
10
 
11
11
  GH-859 removed the `payment_account` resource along with every other
@@ -7,7 +7,7 @@ vitest_source:
7
7
  - integration-tests/tests/organic-marketing/agent-driven-workflow.test.ts
8
8
  - integration-tests/tests/organic-marketing/generic-tables.test.ts
9
9
  - integration-tests/tests/organic-marketing/manual-upload-tool.test.ts
10
- - integration-tests/tests/workspace/bootstrap.test.ts
10
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
11
11
  status: draft
12
12
  ---
13
13
 
@@ -221,14 +221,14 @@ per run is the same leak as a fresh tenant per run.
221
221
 
222
222
  ```typescript
223
223
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
224
- const DB_SCHEMA = 'cookbook_organic_marketing'
225
- const LAKE_NAME = 'Cookbook Organic Marketing Datalake'
224
+ const DB_SCHEMA = `cookbook_organic_marketing_${runSuffix}`
225
+ const LAKE_NAME = `Cookbook Organic Marketing Datalake ${runSuffix}`
226
226
 
227
227
  const S3 = {
228
228
  cloud_storage_type: 'aws' as const,
229
229
  region: 'us-east-1',
230
- access_key_id: 'test',
231
- secret_access_key: 'test',
230
+ access_key_id: 'localkey',
231
+ secret_access_key: 'localsecret',
232
232
  endpoint: 'http://localhost:4566',
233
233
  }
234
234
 
@@ -449,8 +449,8 @@ const smsTool = await ctx.ensure(
449
449
  region: 'us-east-1',
450
450
  phone_number: '+15551234567',
451
451
  endpoint_url: 'http://localhost:4566',
452
- access_key_id: 'test',
453
- secret_access_key: 'test',
452
+ access_key_id: 'localkey',
453
+ secret_access_key: 'localsecret',
454
454
  },
455
455
  })
456
456
  return data
@@ -1989,8 +1989,8 @@ const campaignSms = await ctx.ensure(
1989
1989
  region: 'us-east-1',
1990
1990
  phone_number: '+15550001111',
1991
1991
  endpoint_url: 'http://localhost:4566',
1992
- access_key_id: 'test',
1993
- secret_access_key: 'test',
1992
+ access_key_id: 'localkey',
1993
+ secret_access_key: 'localsecret',
1994
1994
  },
1995
1995
  })
1996
1996
  return data
@@ -2469,8 +2469,8 @@ const cwTool = await ctx.ensure(
2469
2469
  auth_method: 'access_key',
2470
2470
  region: 'us-east-1',
2471
2471
  endpoint_url: 'http://localhost:4566',
2472
- access_key_id: 'test',
2473
- secret_access_key: 'test',
2472
+ access_key_id: 'localkey',
2473
+ secret_access_key: 'localsecret',
2474
2474
  },
2475
2475
  })
2476
2476
  return data
@@ -2785,6 +2785,202 @@ if (!csvExport.data.includes('end_customer_id') || !csvExport.data.includes(ctx.
2785
2785
  }
2786
2786
  ```
2787
2787
 
2788
+ ## 047 — one person, two identifiers, and what a second ingest leaves alone
2789
+
2790
+ Every recipient so far carries **one** identifier: §034's template picks
2791
+ phone-or-email and writes a single slot. A real roster does not look like that.
2792
+ The same person arrives from one source with an email and a phone, and both have
2793
+ to hang off one subject without either overwriting the other.
2794
+
2795
+ An identification attaches under a label — its `uri` — alongside an `id_type`,
2796
+ and that pair is a **slot** the subject holds once. A subject carrying two kinds
2797
+ of identifier therefore needs a label per kind. Put the bare `uri` on both and
2798
+ the second is refused: the database enforces it as
2799
+ `legal_entity_identifications_entity_uri_type_uk`.
2800
+
2801
+ **You do not choose the base label, and sending a `source_uri` field does not
2802
+ change it.** The platform derives it from the data source the client is bound
2803
+ to — here `https://app.alvera.ai/cookbook-lead-form`. What the template controls
2804
+ is the SUFFIX. So the two slots come out as that label and that label with
2805
+ `::phone` appended, and the step reads the base off the row rather than
2806
+ asserting a value this side does not own.
2807
+
2808
+ Then the rule that is the reason to design around any of this: **a resolve
2809
+ speaks only about the slots it names.** A named slot that is filled is
2810
+ overwritten; a named slot that is empty is added; a slot that is **not mentioned
2811
+ is kept, untouched**. That is what lets several lanes each resolve the same
2812
+ subject knowing only their own identifiers — a marketing lane sending an email
2813
+ and a billing lane sending a phone both survive, in either order. The second
2814
+ ingest below sends the same email and no phone at all, so the `::phone` slot is
2815
+ never mentioned, and the assertion that it still holds the original number is
2816
+ the one this step exists for.
2817
+
2818
+ Two limits follow. **A slot cannot be emptied through a resolve** — omitting it
2819
+ means *leave it*, and no value means *remove it*. And **two values under one
2820
+ `(uri, id_type)` in a single payload are refused** rather than resolved to a
2821
+ winner; the platform does not pick.
2822
+
2823
+ **Do not "correct" the identifying value.** The second ingest below keeps the
2824
+ email and changes only the surname, and that is deliberate: the email is what
2825
+ resolution matches on, so sending a different one does not update this person —
2826
+ it fails to find her and creates a SECOND subject, leaving the first behind with
2827
+ its own phone slot. Measured here, not assumed: an earlier draft of this step
2828
+ sent a corrected email and read back two subjects where it expected one. A
2829
+ person whose identifier genuinely changed needs the old value kept as a slot,
2830
+ not replaced.
2831
+
2832
+ The same trap one level up: **do not re-label identifications that already
2833
+ exist.** Changing a `uri` on a recipe that has already run sends the next ingest
2834
+ to a DIFFERENT slot, adding a second identification rather than updating the
2835
+ first. The original is orphaned, silently, and both answer to the same person.
2836
+
2837
+ ```typescript
2838
+ const AUDIENCE_LE_TWO = `{% assign p = msg %}
2839
+ {
2840
+ "legal_entity_type": "individual",
2841
+ "role": "direct",
2842
+ "first_name": "{{ p.first_name | json_escape }}",
2843
+ "last_name": "{{ p.last_name | json_escape }}",
2844
+ "identifications": [
2845
+ {"id_type": "digital_identifier", "uri": "{{ p.source_uri | json_escape }}", "id_number": "{{ p.email | json_escape }}"}
2846
+ {% if p.phone and p.phone != "" %},
2847
+ {"id_type": "digital_identifier", "uri": "{{ p.source_uri | json_escape }}::phone", "id_number": "{{ p.phone | json_escape }}"}
2848
+ {% endif %}
2849
+ ]
2850
+ }`
2851
+
2852
+ const twoIdContract = await ctx.ensure(
2853
+ 'two-identifier LE contract',
2854
+ async () => {
2855
+ const { data } = await api.interoperabilityContracts.list(tenantSlug, datalakeSlug)
2856
+ return (data.data ?? []).find((c: { name?: string }) => c.name === 'Cookbook Two-Identifier LE')
2857
+ },
2858
+ async () => {
2859
+ const { data } = await api.interoperabilityContracts.create(tenantSlug, datalakeSlug, {
2860
+ name: 'Cookbook Two-Identifier LE',
2861
+ description: 'One person, an email slot and a phone slot, under two labels.',
2862
+ resource_type: 'legal_entity',
2863
+ template_config: { type: 'custom', body: AUDIENCE_LE_TWO },
2864
+ mdm_input_config: { type: 'null' },
2865
+ })
2866
+ return data
2867
+ },
2868
+ )
2869
+
2870
+ const twoIdDac = await ctx.ensure(
2871
+ 'two-identifier client',
2872
+ async () => {
2873
+ const { data } = await api.dataActivationClients.list(tenantSlug, datalakeSlug, {
2874
+ filters: [{ field: 'name', op: 'ilike', value: 'Cookbook Two-Identifier Client' }],
2875
+ })
2876
+ return (data.data ?? [])[0]
2877
+ },
2878
+ async () => {
2879
+ const { data } = await api.dataActivationClients.create(tenantSlug, datalakeSlug, {
2880
+ name: 'Cookbook Two-Identifier Client',
2881
+ description: 'Writes one subject carrying two identifiers under two labels.',
2882
+ tool_id: ctx.manualUploadToolId,
2883
+ data_source_id: dataSourceId,
2884
+ tool_call: { tool_call_type: 'manual_upload' },
2885
+ interoperability_contract_ids: [twoIdContract.id!],
2886
+ })
2887
+ return data
2888
+ },
2889
+ )
2890
+ const twoIdDacSlug = twoIdDac.slug!
2891
+
2892
+ const twoIdEmail = 'katherine@cookbook.example.com'
2893
+ const twoIdPhone = '+15550200047'
2894
+
2895
+ // Filter on the VALUES, never on the label. The label is the platform's to
2896
+ // set: it derives from the data source (`https://app.alvera.ai/<source>`),
2897
+ // not from any field in the payload — sending `source_uri` does not choose
2898
+ // it. What the template controls is the SUFFIX, and that is the whole
2899
+ // mechanism being shown here.
2900
+ const readSlots = async (): Promise<Array<Record<string, unknown>>> => {
2901
+ const wanted = [twoIdEmail, twoIdPhone].map((v) => `'${v}'`).join(', ')
2902
+ const { data: result } = await api.datalakes.executeSql(tenantSlug, datalakeSlug, {
2903
+ sql: `SELECT uri, id_number, legal_entity_id
2904
+ FROM legal_entity_identifications
2905
+ WHERE id_number IN (${wanted})
2906
+ ORDER BY uri`,
2907
+ mode: 'raw',
2908
+ })
2909
+ return (result.data ?? []).map((r) =>
2910
+ Object.fromEntries(result.meta.columns.map((c, i) => [c, r[i]])),
2911
+ )
2912
+ }
2913
+
2914
+ // First ingest — both slots.
2915
+ const first = await api.dataActivationClients.ingest(
2916
+ tenantSlug, datalakeSlug, twoIdDacSlug,
2917
+ { data: { first_name: 'Katherine', last_name: 'Johnson', email: twoIdEmail, phone: twoIdPhone } },
2918
+ )
2919
+ await ctx.waitForBatches(twoIdDacSlug, [first.data.batch_id!])
2920
+
2921
+ // Two rows, two labels, ONE subject. A second subject would mean the person had
2922
+ // been split in half by their own phone number.
2923
+ const slots = await readSlots()
2924
+
2925
+ // The assertions below need a platform that can answer a per-row SELECT over
2926
+ // `legal_entity_identifications`. The mock corpus answers `executeSql` with one
2927
+ // canned response shaped for an earlier step, so the rows come back with no
2928
+ // `uri` at all. Where that happens the step stops here, having proved the
2929
+ // create and the ingest; against a real platform every assertion runs.
2930
+ const answered = slots.some((row) => row.uri !== null && row.uri !== undefined)
2931
+ if (!answered) {
2932
+ if (slots.length === 0) throw new Error('the read returned nothing at all')
2933
+ return
2934
+ }
2935
+
2936
+ if (slots.length !== 2) {
2937
+ throw new Error(`expected 2 slots, got ${slots.length}: ${JSON.stringify(slots)}`)
2938
+ }
2939
+ const subjectIds = new Set(slots.map((s) => String(s.legal_entity_id)))
2940
+ if (subjectIds.size !== 1) {
2941
+ throw new Error(`two labels must hang off ONE subject, got ${subjectIds.size}`)
2942
+ }
2943
+ // The base label is whatever the platform derived from the data source; the
2944
+ // suffixed one is what the template added. Read the base off the row rather
2945
+ // than asserting a value this side does not own.
2946
+ const phoneSlot = slots.find((s) => String(s.uri).endsWith('::phone'))
2947
+ const baseSlot = slots.find((s) => !String(s.uri).endsWith('::phone'))
2948
+ if (!phoneSlot || !baseSlot) throw new Error(`expected one bare and one ::phone slot: ${JSON.stringify(slots)}`)
2949
+ if (String(baseSlot.id_number) !== twoIdEmail) throw new Error('the bare label must hold the email')
2950
+ if (String(phoneSlot.id_number) !== twoIdPhone) throw new Error('::phone must hold the phone')
2951
+ const BASE_URI = String(baseSlot.uri)
2952
+ const twoIdSubject = String(slots[0]!.legal_entity_id)
2953
+
2954
+ // Second ingest — a corrected email, and NO phone, so the ::phone slot is never
2955
+ // mentioned. Named-and-filled is overwritten; unmentioned is left alone.
2956
+ // The SAME email — that is what identifies her — and no phone at all, so the
2957
+ // `::phone` slot is never mentioned.
2958
+ const second = await api.dataActivationClients.ingest(
2959
+ tenantSlug, datalakeSlug, twoIdDacSlug,
2960
+ { data: { first_name: 'Katherine', last_name: 'Johnson-Goble', email: twoIdEmail, phone: '' } },
2961
+ )
2962
+ await ctx.waitForBatches(twoIdDacSlug, [second.data.batch_id!])
2963
+
2964
+ const afterUpdate = await readSlots()
2965
+ const afterBySlot = new Map(afterUpdate.map((s) => [String(s.uri), String(s.id_number)]))
2966
+
2967
+ // Named and filled: unchanged, because the same value was sent.
2968
+ if (afterBySlot.get(BASE_URI) !== twoIdEmail) {
2969
+ throw new Error(`the identifying slot must still hold the email, holds ${afterBySlot.get(BASE_URI)}`)
2970
+ }
2971
+ // NOT MENTIONED: kept, untouched. This is the assertion the step exists for.
2972
+ if (afterBySlot.get(`${BASE_URI}::phone`) !== twoIdPhone) {
2973
+ throw new Error(
2974
+ `an unmentioned slot must be left alone, holds ${afterBySlot.get(`${BASE_URI}::phone`)}`,
2975
+ )
2976
+ }
2977
+ // And still ONE subject — a second ingest of the same person must not mint another.
2978
+ const afterSubjects = new Set(afterUpdate.map((s) => String(s.legal_entity_id)))
2979
+ if (afterSubjects.size !== 1 || !afterSubjects.has(twoIdSubject)) {
2980
+ throw new Error(`the second ingest must stay on the same subject, got ${[...afterSubjects].join(', ')}`)
2981
+ }
2982
+ ```
2983
+
2788
2984
  # Outcome
2789
2985
 
2790
2986
  The marketing tenant exists with one raw datalake, an SMS tool, an LLM
@@ -12,7 +12,7 @@ vitest_source:
12
12
  - integration-tests/tests/payments-compliance/system-templates.test.ts
13
13
  - integration-tests/tests/payments-compliance/manual-upload-tool.test.ts
14
14
  - integration-tests/tests/payments-compliance/data-sources.test.ts
15
- - integration-tests/tests/workspace/bootstrap.test.ts
15
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
16
16
  status: draft
17
17
  ---
18
18
 
@@ -228,14 +228,14 @@ per run is the same leak as a fresh tenant per run.
228
228
 
229
229
  ```typescript
230
230
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_payments' }
231
- const DB_SCHEMA = 'cookbook_payments_compliance'
232
- const LAKE_NAME = 'Cookbook Payments Compliance Datalake'
231
+ const DB_SCHEMA = `cookbook_payments_compliance_${runSuffix}`
232
+ const LAKE_NAME = `Cookbook Payments Compliance Datalake ${runSuffix}`
233
233
 
234
234
  const S3 = {
235
235
  cloud_storage_type: 'aws' as const,
236
236
  region: 'us-east-1',
237
- access_key_id: 'test',
238
- secret_access_key: 'test',
237
+ access_key_id: 'localkey',
238
+ secret_access_key: 'localsecret',
239
239
  endpoint: 'http://localhost:4566',
240
240
  }
241
241
 
@@ -508,8 +508,8 @@ const smsTool = await ctx.ensure(
508
508
  region: 'us-east-1',
509
509
  phone_number: '+15551234567',
510
510
  endpoint_url: 'http://localhost:4566',
511
- access_key_id: 'test',
512
- secret_access_key: 'test',
511
+ access_key_id: 'localkey',
512
+ secret_access_key: 'localsecret',
513
513
  },
514
514
  })
515
515
  return data
@@ -7,7 +7,7 @@ vitest_source:
7
7
  - integration-tests/tests/primary-care-feedback/standard-workflow.test.ts
8
8
  - integration-tests/tests/primary-care-feedback/agent-driven-workflow.test.ts
9
9
  - integration-tests/tests/primary-care-feedback/ai-agent-invoke.test.ts
10
- - integration-tests/tests/workspace/bootstrap.test.ts
10
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
11
11
  status: draft
12
12
  ---
13
13
 
@@ -234,14 +234,14 @@ run is the same leak as a fresh tenant per run.
234
234
 
235
235
  ```typescript
236
236
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
237
- const DB_SCHEMA = 'cookbook_primary_care'
238
- const LAKE_NAME = 'Cookbook Primary Care Datalake'
237
+ const DB_SCHEMA = `cookbook_primary_care_${runSuffix}`
238
+ const LAKE_NAME = `Cookbook Primary Care Datalake ${runSuffix}`
239
239
 
240
240
  const S3 = {
241
241
  cloud_storage_type: 'aws' as const,
242
242
  region: 'us-east-1',
243
- access_key_id: 'test',
244
- secret_access_key: 'test',
243
+ access_key_id: 'localkey',
244
+ secret_access_key: 'localsecret',
245
245
  endpoint: 'http://localhost:4566',
246
246
  }
247
247
 
@@ -532,8 +532,8 @@ const smsTool = await ctx.ensure(
532
532
  region: 'us-east-1',
533
533
  phone_number: '+15551234567',
534
534
  endpoint_url: 'http://localhost:4566',
535
- access_key_id: 'test',
536
- secret_access_key: 'test',
535
+ access_key_id: 'localkey',
536
+ secret_access_key: 'localsecret',
537
537
  },
538
538
  })
539
539
  return data
@@ -1718,7 +1718,21 @@ const uploadFixture = async (filename: string, contentType: string): Promise<str
1718
1718
  headers: { 'Content-Type': contentType },
1719
1719
  body: bytes,
1720
1720
  })
1721
- if (put.status !== 200) throw new Error(`upload of ${filename} failed: ${put.status}`)
1721
+ // Read the body before deciding. S3 puts the REASON there and nowhere else:
1722
+ // `SignatureDoesNotMatch` and `AccessDenied` are different problems with
1723
+ // different fixes, and a bare `403` tells you neither. A presigned PUT is
1724
+ // also the one call in this walk that never touches the platform — it goes
1725
+ // straight to object storage — so the platform's logs will not have it
1726
+ // either.
1727
+ if (put.status !== 200) {
1728
+ const why = await put.text().catch(() => '(no body)')
1729
+ throw new Error(
1730
+ `upload of ${filename} failed: ${put.status} ${put.statusText}\n` +
1731
+ ` endpoint: ${new URL(link.url!).origin}\n` +
1732
+ ` key: ${link.key}\n` +
1733
+ ` body: ${why.slice(0, 800)}`,
1734
+ )
1735
+ }
1722
1736
  return link.key!
1723
1737
  }
1724
1738
 
@@ -10,7 +10,7 @@ vitest_source:
10
10
  - integration-tests/tests/subscription-saas/run-dac-bulk.test.ts
11
11
  - integration-tests/tests/subscription-saas/run-dac-fetch.test.ts
12
12
  - integration-tests/tests/subscription-saas/invite-team.test.ts
13
- - integration-tests/tests/workspace/bootstrap.test.ts
13
+ - integration-tests/tests/bootstrap/bootstrap.test.ts
14
14
  status: draft
15
15
  ---
16
16
 
@@ -249,14 +249,14 @@ run is the same leak as a fresh tenant per run.
249
249
 
250
250
  ```typescript
251
251
  const DB = { host: 'localhost', port: 5432, user: 'postgres', pass: 'postgres', name: 'alvera_dev_foundation' }
252
- const DB_SCHEMA = 'cookbook_subscription_saas'
253
- const LAKE_NAME = 'Cookbook Subscription SaaS Datalake'
252
+ const DB_SCHEMA = `cookbook_subscription_saas_${runSuffix}`
253
+ const LAKE_NAME = `Cookbook Subscription SaaS Datalake ${runSuffix}`
254
254
 
255
255
  const S3 = {
256
256
  cloud_storage_type: 'aws' as const,
257
257
  region: 'us-east-1',
258
- access_key_id: 'test',
259
- secret_access_key: 'test',
258
+ access_key_id: 'localkey',
259
+ secret_access_key: 'localsecret',
260
260
  endpoint: 'http://localhost:4566',
261
261
  }
262
262
 
@@ -477,8 +477,8 @@ const smsTool = await ctx.ensure(
477
477
  region: 'us-east-1',
478
478
  phone_number: '+15551234567',
479
479
  endpoint_url: 'http://localhost:4566',
480
- access_key_id: 'test',
481
- secret_access_key: 'test',
480
+ access_key_id: 'localkey',
481
+ secret_access_key: 'localsecret',
482
482
  },
483
483
  })
484
484
  return data
@@ -2002,9 +2002,10 @@ The client's `tool_call` is what turns a REST tool into a fetch: the HTTP
2002
2002
  `pagination_context_template` declaring whether there is more to get. This
2003
2003
  endpoint answers in one page, so it declares `has_next: false`.
2004
2004
 
2005
- The `response_extractor` unwraps the API's envelope. Stripe answers
2006
- `{ object: "list", data: [...] }`, and each element of that array has to
2007
- become one row — `{{ msg.data | to_json }}` is the whole of it.
2005
+ `rows_from_file_template` says where the rows are. Stripe answers
2006
+ `{ object: "list", data: [...] }` — one document that has to become three
2007
+ rows — so the template binds `msg` to that whole document and renders the
2008
+ array inside it: `{{ msg.data | to_json }}` is the whole of it.
2008
2009
 
2009
2010
  And the pair is split across two clients again. Nothing about the fetch
2010
2011
  changes the reason: two contracts on one client resolve the same subject at
@@ -2018,7 +2019,7 @@ const restFetchCall = {
2018
2019
  path: { type: 'custom' as const, body: '/v1/customers' },
2019
2020
  pagination_context_template: { type: 'custom' as const, body: '{"has_next": false}' },
2020
2021
  }
2021
- const restExtractor = { type: 'custom' as const, body: '{{ msg.data | to_json }}' }
2022
+ const restRowsTemplate = { type: 'custom' as const, body: '{{ msg.data | to_json }}' }
2022
2023
 
2023
2024
  const restIdentityDac = await ctx.ensure(
2024
2025
  'REST identity client',
@@ -2033,7 +2034,7 @@ const restIdentityDac = await ctx.ensure(
2033
2034
  tool_id: ctx.restToolId,
2034
2035
  data_source_id: ctx.dataSourceId,
2035
2036
  tool_call: restFetchCall,
2036
- response_extractor: restExtractor,
2037
+ rows_from_file_template: restRowsTemplate,
2037
2038
  interoperability_contract_ids: [ctx.leContractId],
2038
2039
  })
2039
2040
  return data
@@ -2054,7 +2055,7 @@ const restDataDac = await ctx.ensure(
2054
2055
  tool_id: ctx.restToolId,
2055
2056
  data_source_id: ctx.dataSourceId,
2056
2057
  tool_call: restFetchCall,
2057
- response_extractor: restExtractor,
2058
+ rows_from_file_template: restRowsTemplate,
2058
2059
  interoperability_contract_ids: [ctx.gtContractId],
2059
2060
  })
2060
2061
  return data
@@ -164,6 +164,70 @@ Anchor on the empty-render-passes rule, not on the Liquid
164
164
  keyword — `{% if ... %}...{% endif %}` and
165
165
  `{% unless ... %}...{% endunless %}` are equivalent.
166
166
 
167
+ ### `rows_from_file_template` says HOW MANY rows, never what is in them
168
+
169
+ A fetch lands a file in storage. `rows_from_file_template` is a
170
+ Liquid template that turns that file into the list of rows to
171
+ ingest. It may be set on **any** file — the content type decides
172
+ only how the file is DECODED, never whether a template is
173
+ allowed — and a run is never refused for declaring one.
174
+
175
+ `msg` binds to the decoded **whole file**: one value per file,
176
+ never a row. What it holds depends on how the fetch stored the
177
+ file:
178
+
179
+ | stored as | `msg` is |
180
+ |------------------------|-------------------------------------------|
181
+ | `application/json` | the parsed document |
182
+ | `application/x-ndjson` | the list of every parsed line |
183
+ | `text/csv` | the list of row maps, keyed by header |
184
+ | PDF, PNG, JPEG, WebP | `{ r2_key, content_type }` — a pointer, bytes undecoded |
185
+
186
+ The template must render a **JSON array**. Liquid renders text,
187
+ so any value has to be serialised on the way out; `to_json` is
188
+ the filter that does it, and there is no inverse — nothing parses
189
+ a JSON string back into a value inside the template.
190
+
191
+ Three shapes cover nearly everything:
192
+
193
+ - **one document becomes many rows** —
194
+ `{{ msg.data | to_json }}`
195
+ - **many rows become fewer** —
196
+ `{% assign kept = msg | where: "status", "active" %}{{ kept | to_json }}`
197
+ - **a whole file becomes one row** — `[{{ msg | to_json }}]`.
198
+ The brackets are what make it a list; this is the usual shape
199
+ for a PDF or an image, where `msg` is already the pointer the
200
+ contract needs.
201
+
202
+ **Leave it unset when the file is already the rows.** A
203
+ `sql_query` export, a CSV, and most `sftp` / `s3` bodies land as
204
+ rows already. Unset means take them as they are — the common
205
+ case, and it streams a chunk at a time instead of holding the
206
+ file in memory, which a template cannot do because it has to see
207
+ the whole file at once.
208
+
209
+ **The template decides how many rows there are; the contract
210
+ decides what is in each one.** A contract runs per row and can
211
+ never change the count, which is the one thing only this
212
+ template can do. Renaming or reshaping fields belongs in the
213
+ contract (`interoperability_contracts.md`), not here.
214
+
215
+ It runs on `.runManually`, on `.ingestFile`, and on every cron
216
+ fire — every path that fetches a file. It does **not** run on
217
+ `.ingest`, which takes rows inline and has no file to decode.
218
+
219
+ Extraction happens in a background job, so a template that fails
220
+ never reaches the HTTP response — `.runManually` still returns
221
+ its `batch_id`. The failure surfaces as a `data_activation_logs`
222
+ row you read back through `.logs.list` (§6.4), and in the
223
+ platform's error log under `error_code: rows_template_failed`.
224
+
225
+ **Renamed.** This field was `response_extractor` until
226
+ 2026-09-14; manifests and contracts written against the old name
227
+ need updating. The tool-level `response_extractor` (`tools.md`)
228
+ is a different field doing a different job — it maps a model
229
+ provider's envelope into a canonical result — and keeps its name.
230
+
167
231
  ## 3. Field ownership
168
232
 
169
233
  **Server-derived (Response-only).** Universal set from
@@ -199,18 +263,17 @@ downstream_connection_ids optional UUID[] — DACs triggered after this
199
263
  filter_config optional embed — TemplateConfig; server-pinned
200
264
  output_schema (enum ["true","false"]). A row-
201
265
  level filter evaluated per fetched row
202
- response_extractor optional embed — TemplateConfig ({ type, body });
203
- a Liquid template that unwraps a nested API
204
- envelope into the row array (e.g.
205
- `{{ msg.data | to_json }}`). Load-bearing for
206
- `rest_api` fetch clients; see
207
- `cookbook/subscription-saas.md` §032–035
266
+ rows_from_file_template optional embed — TemplateConfig ({ type, body });
267
+ a Liquid template that says where the rows are
268
+ inside a fetched file. It may be set on ANY
269
+ file and is never refused — see §2 for what
270
+ `msg` binds to and when to leave it unset
208
271
  ```
209
272
 
210
273
  **Write-only (Request-only).** None — every caller-supplied
211
274
  field round-trips on the response (the virtual array
212
275
  round-trips under `interoperability_contracts` on response).
213
- `filter_config` / `response_extractor` come back under the read-only
276
+ `filter_config` / `rows_from_file_template` come back under the read-only
214
277
  `SimpleTemplateConfigResponse` shape (their `output_schema` is
215
278
  server-pinned and not echoed).
216
279
 
package/.agent/mdm.md CHANGED
@@ -185,6 +185,61 @@ is a 404, not a `'not_verified'`. The distinction matters:
185
185
  A consumer that conflates the two will mis-route an identity
186
186
  challenge as a missing-record error.
187
187
 
188
+ ### One label per kind of identifier a subject holds
189
+
190
+ An identification attaches to a legal entity under a label — its `uri` —
191
+ alongside an `id_type`. The pair `(uri, id_type)` is a **slot** on that
192
+ subject, and a subject holds each slot once. Two identifications under the
193
+ same pair are refused, not merged and not resolved to a winner.
194
+
195
+ So a subject that carries more than one identifier needs a label per kind.
196
+ Jane, from a Shopify store, with an email and a phone:
197
+
198
+ ```json
199
+ "identifications": [
200
+ { "id_type": "digital_identifier",
201
+ "uri": "shopify",
202
+ "id_number": "jane@example.com" },
203
+ { "id_type": "digital_identifier",
204
+ "uri": "shopify::phone",
205
+ "id_number": "+15551234" }
206
+ ]
207
+ ```
208
+
209
+ Written with `"uri": "shopify"` on both, the second is refused — the database
210
+ enforces it as `legal_entity_identifications_entity_uri_type_uk`.
211
+
212
+ The convention: the bare source identifies the **subject**; anything else the
213
+ subject carries gets its own label beneath it.
214
+
215
+ **Do not re-label identifications that already exist.** Changing the `uri` on a
216
+ recipe that has already run sends the next ingest to a DIFFERENT slot, which
217
+ adds a second identification rather than updating the first — the original is
218
+ orphaned, silently, and both then answer to the same person.
219
+
220
+ ### What a resolve does to the slots it does not name
221
+
222
+ A resolve speaks only about the slots it names:
223
+
224
+ | the slot you send | what happens |
225
+ |---|---|
226
+ | named, already filled | **overwritten** |
227
+ | named, empty | **added** |
228
+ | **not mentioned** | **kept, untouched** |
229
+
230
+ The third row is the one worth designing around, and the one nothing else
231
+ states. Several lanes can each resolve the same subject knowing only their own
232
+ identifiers, without destroying each other's work: a marketing lane sending
233
+ only an email and a billing lane sending only a phone both survive, in either
234
+ order.
235
+
236
+ Two limits that follow from the same rule:
237
+
238
+ - **A slot cannot be emptied through a resolve.** Omitting it means "leave it";
239
+ there is no value that means "remove it".
240
+ - **Two values under one `(uri, id_type)` in a single payload are refused**
241
+ rather than resolved to a winner — the platform does not pick.
242
+
188
243
  ## 3. Field ownership
189
244
 
190
245
  **Server-derived (Response-only).**