salesforce-data-customcode 6.1.0.dev1__py3-none-any.whl → 6.1.0.dev2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datacustomcode/__init__.py +0 -5
- datacustomcode/cli.py +4 -29
- datacustomcode/client.py +192 -211
- datacustomcode/config.py +0 -5
- datacustomcode/config.yaml +6 -0
- datacustomcode/constants.py +0 -8
- datacustomcode/deploy.py +23 -57
- datacustomcode/function/runtime.py +16 -0
- datacustomcode/io/reader/base.py +0 -42
- datacustomcode/io/writer/base.py +0 -33
- datacustomcode/named_credential/__init__.py +26 -0
- datacustomcode/named_credential/base.py +54 -0
- datacustomcode/named_credential/default.py +93 -0
- datacustomcode/named_credential/direct/__init__.py +19 -0
- datacustomcode/named_credential/direct/auth.py +63 -0
- datacustomcode/named_credential/direct/credentials.py +121 -0
- datacustomcode/named_credential/direct/transport.py +110 -0
- datacustomcode/named_credential/direct/url_resolver.py +112 -0
- datacustomcode/named_credential/spark_base.py +93 -0
- datacustomcode/named_credential/spark_default.py +154 -0
- datacustomcode/named_credential/types/__init__.py +14 -0
- datacustomcode/named_credential/types/http_method.py +29 -0
- datacustomcode/named_credential/types/http_request.py +63 -0
- datacustomcode/named_credential/types/http_request_builder.py +55 -0
- datacustomcode/named_credential/types/http_response.py +43 -0
- datacustomcode/named_credential/types/http_response_builder.py +24 -0
- datacustomcode/named_credential_config.py +105 -0
- datacustomcode/run.py +7 -11
- datacustomcode/scan.py +29 -164
- datacustomcode/template.py +1 -13
- datacustomcode/templates/function/example/chunking_with_external_callout/README.md +119 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/config.json +3 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +161 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +11 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +16 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/METADATA +3 -40
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/RECORD +40 -19
- datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +0 -49
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/WHEEL +0 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/entry_points.txt +0 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
3
|
+
# SPDX-License-Identifier: Apache-2
|
|
4
|
+
|
|
5
|
+
"""
|
|
6
|
+
Document Chunking with a Gemini Named Credential Callout
|
|
7
|
+
|
|
8
|
+
Splits each input document into paragraph-sized chunks and classifies every
|
|
9
|
+
chunk via Google's Gemini ``generateContent`` API, reached through a Named
|
|
10
|
+
Credential (``callout:gemini``) so the endpoint URL and API key are resolved
|
|
11
|
+
outside this code. The classification is attached to each chunk as citations.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import logging
|
|
16
|
+
|
|
17
|
+
from datacustomcode.function import Runtime
|
|
18
|
+
from datacustomcode.function.feature_types.chunking import (
|
|
19
|
+
ChunkType,
|
|
20
|
+
SearchIndexChunkingV1Output,
|
|
21
|
+
SearchIndexChunkingV1Request,
|
|
22
|
+
SearchIndexChunkingV1Response,
|
|
23
|
+
)
|
|
24
|
+
from datacustomcode.named_credential.types.http_method import HTTPMethod
|
|
25
|
+
from datacustomcode.named_credential.types.http_request_builder import (
|
|
26
|
+
HTTPRequestBuilder,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
logging.basicConfig(level=logging.INFO)
|
|
31
|
+
|
|
32
|
+
CALLOUT_URL = "callout:gemini"
|
|
33
|
+
|
|
34
|
+
_ANALYSIS_FIELDS = ("summary", "category", "sentiment")
|
|
35
|
+
|
|
36
|
+
_PROMPT = (
|
|
37
|
+
"Analyze the following document chunk and classify it. Respond with its "
|
|
38
|
+
"one-sentence summary, a single-word category, overall sentiment "
|
|
39
|
+
"(positive, negative, or neutral), and up to five key topics.\n\nChunk:\n"
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
# Force Gemini to return the classification as JSON in a fixed shape.
|
|
43
|
+
_GENERATION_CONFIG = {
|
|
44
|
+
"responseMimeType": "application/json",
|
|
45
|
+
"responseSchema": {
|
|
46
|
+
"type": "object",
|
|
47
|
+
"properties": {
|
|
48
|
+
"summary": {"type": "string"},
|
|
49
|
+
"category": {"type": "string"},
|
|
50
|
+
"sentiment": {"type": "string"},
|
|
51
|
+
"topics": {"type": "array", "items": {"type": "string"}},
|
|
52
|
+
},
|
|
53
|
+
"required": ["summary", "category", "sentiment", "topics"],
|
|
54
|
+
},
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _chunk_text(text: str, max_words: int = 80) -> list[str]:
|
|
59
|
+
"""Split text into paragraph-aligned chunks of at most ``max_words`` words."""
|
|
60
|
+
paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()]
|
|
61
|
+
|
|
62
|
+
chunks: list[str] = []
|
|
63
|
+
current: list[str] = []
|
|
64
|
+
current_words = 0
|
|
65
|
+
|
|
66
|
+
for paragraph in paragraphs:
|
|
67
|
+
paragraph_words = len(paragraph.split())
|
|
68
|
+
if current and current_words + paragraph_words > max_words:
|
|
69
|
+
chunks.append("\n\n".join(current))
|
|
70
|
+
current = []
|
|
71
|
+
current_words = 0
|
|
72
|
+
current.append(paragraph)
|
|
73
|
+
current_words += paragraph_words
|
|
74
|
+
|
|
75
|
+
if current:
|
|
76
|
+
chunks.append("\n\n".join(current))
|
|
77
|
+
|
|
78
|
+
return chunks
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _extract_model_json(body: str) -> dict:
|
|
82
|
+
"""Decode the model's JSON classification from a Gemini response.
|
|
83
|
+
|
|
84
|
+
The generated text sits at ``candidates[0].content.parts[0].text`` and is
|
|
85
|
+
itself a JSON string, so decode twice. Any malformed layer yields ``{}``.
|
|
86
|
+
"""
|
|
87
|
+
try:
|
|
88
|
+
envelope = json.loads(body) if body else {}
|
|
89
|
+
except json.JSONDecodeError:
|
|
90
|
+
return {}
|
|
91
|
+
|
|
92
|
+
try:
|
|
93
|
+
text = envelope["candidates"][0]["content"]["parts"][0]["text"]
|
|
94
|
+
except (KeyError, IndexError, TypeError):
|
|
95
|
+
return {}
|
|
96
|
+
|
|
97
|
+
try:
|
|
98
|
+
payload = json.loads(text)
|
|
99
|
+
except json.JSONDecodeError:
|
|
100
|
+
return {}
|
|
101
|
+
return payload if isinstance(payload, dict) else {}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _analyze_chunk(chunk_text: str, runtime: Runtime) -> dict[str, str]:
|
|
105
|
+
"""Classify one chunk via the Gemini callout and return it as citations."""
|
|
106
|
+
request = (
|
|
107
|
+
HTTPRequestBuilder()
|
|
108
|
+
.set_url(CALLOUT_URL)
|
|
109
|
+
.set_method(HTTPMethod.POST)
|
|
110
|
+
.set_headers({"Content-Type": "application/json", "Accept": "application/json"})
|
|
111
|
+
.build()
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
payload = {
|
|
115
|
+
"contents": [{"parts": [{"text": _PROMPT + chunk_text}]}],
|
|
116
|
+
"generationConfig": _GENERATION_CONFIG,
|
|
117
|
+
}
|
|
118
|
+
response = runtime.named_credential.request(request, json.dumps(payload))
|
|
119
|
+
|
|
120
|
+
# Don't raise: a single failed callout shouldn't abort the whole job.
|
|
121
|
+
if not response.is_success:
|
|
122
|
+
logger.error(f"Gemini callout failed with status {response.status_code}")
|
|
123
|
+
return {"analysis_status": "failed", "http_status": str(response.status_code)}
|
|
124
|
+
|
|
125
|
+
data = _extract_model_json(response.body)
|
|
126
|
+
citations = {"analysis_status": "success"}
|
|
127
|
+
for field in _ANALYSIS_FIELDS:
|
|
128
|
+
value = data.get(field)
|
|
129
|
+
citations[field] = str(value) if value is not None else "unavailable"
|
|
130
|
+
|
|
131
|
+
topics = data.get("topics")
|
|
132
|
+
if isinstance(topics, list):
|
|
133
|
+
citations["topics"] = ", ".join(str(topic) for topic in topics)
|
|
134
|
+
|
|
135
|
+
return citations
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def function(
|
|
139
|
+
request: SearchIndexChunkingV1Request, runtime: Runtime
|
|
140
|
+
) -> SearchIndexChunkingV1Response:
|
|
141
|
+
"""Chunk each input document and classify every chunk via the Gemini API."""
|
|
142
|
+
logger.info(f"Received {len(request.input)} documents to chunk")
|
|
143
|
+
|
|
144
|
+
chunks = []
|
|
145
|
+
chunk_id = 1
|
|
146
|
+
|
|
147
|
+
for doc in request.input:
|
|
148
|
+
for chunk_text in _chunk_text(doc.text):
|
|
149
|
+
citations = _analyze_chunk(chunk_text, runtime)
|
|
150
|
+
|
|
151
|
+
chunk = SearchIndexChunkingV1Output(
|
|
152
|
+
text=chunk_text,
|
|
153
|
+
seq_no=chunk_id,
|
|
154
|
+
chunk_type=ChunkType.TEXT,
|
|
155
|
+
citations=citations,
|
|
156
|
+
)
|
|
157
|
+
chunks.append(chunk)
|
|
158
|
+
chunk_id += 1
|
|
159
|
+
|
|
160
|
+
logger.info(f"Produced {len(chunks)} classified chunks")
|
|
161
|
+
return SearchIndexChunkingV1Response(output=chunks)
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": [
|
|
3
|
+
{
|
|
4
|
+
"text": "Product Review: Northstar Analytics\n\nWe rolled Northstar out to our whole revenue team last quarter and the difference has been night and day. Dashboards that used to take our analysts a full day to assemble now refresh in seconds, and the natural-language query box means our account executives can answer their own questions without filing a ticket.\n\nOnboarding was smoother than any tool we have adopted in years. The guided setup imported our Salesforce data on the first try and the sample templates gave us something useful on day one. Support answered our two questions within the hour. Easily the best purchase decision we made this year."
|
|
5
|
+
},
|
|
6
|
+
{
|
|
7
|
+
"text": "Support Ticket #48210: Repeated timeouts on scheduled exports\n\nFor the third week running our nightly export to the data warehouse has failed silently. There is no alert, no email, nothing in the activity log, and we only find out when the morning report is empty and the leadership meeting has no numbers.\n\nI have raised this twice already and both times the ticket was closed as resolved without anyone actually contacting me. This is costing us real credibility internally and I am extremely frustrated. If the connector cannot handle our volume we need to know now so we can plan a migration, because right now the product is not doing the one job we bought it for."
|
|
8
|
+
},
|
|
9
|
+
{
|
|
10
|
+
"text": "Renewal Feedback: mixed feelings heading into year two\n\nThe core product is genuinely good. The reporting engine is fast, the permissions model is granular enough for our compliance team, and our analysts like working in it. On the functionality alone I would renew without hesitation.\n\nWhat gives me pause is the pricing. The per-seat cost jumped noticeably at renewal and several add-ons that used to be included are now separate line items. The value is still there, but the conversation with my finance team was harder than it should have been, and I would like more transparency before the next cycle."
|
|
11
|
+
},
|
|
12
|
+
{
|
|
13
|
+
"text": "Feature Request: scheduled report subscriptions\n\nWe would like the ability to subscribe internal stakeholders to a report on a recurring schedule so a PDF lands in their inbox every Monday morning. Today we export manually and forward it, which is workable but easy to forget.\n\nA few teams have asked whether subscriptions could support filtered views per recipient, for example each regional manager receiving only their own territory. Not urgent for us, but it would remove a recurring bit of manual work and is something a couple of competing tools already offer."
|
|
14
|
+
}
|
|
15
|
+
]
|
|
16
|
+
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: salesforce-data-customcode
|
|
3
|
-
Version: 6.1.0.
|
|
3
|
+
Version: 6.1.0.dev2
|
|
4
4
|
Summary: Data Cloud Custom Code SDK
|
|
5
5
|
License-Expression: Apache-2.0
|
|
6
6
|
License-File: LICENSE.txt
|
|
@@ -170,22 +170,15 @@ Your Python dependencies can be packaged as .py files, .zip archives (containing
|
|
|
170
170
|
|
|
171
171
|
## API
|
|
172
172
|
|
|
173
|
-
Your entry point script will define logic using the `Client` object
|
|
173
|
+
Your entry point script will define logic using the `Client` object which wraps data access layers.
|
|
174
174
|
|
|
175
|
-
|
|
175
|
+
You should only need the following methods:
|
|
176
176
|
* `find_file_path(file_name)` – Resolve a bundled file (placed under `payload/files/`) to a `pathlib.Path` that exists. Works the same locally and inside Data Cloud — see [Bundled file resolution](#bundled-file-resolution) below for the full lookup order. Raises `FileNotFoundError` if the file isn't found.
|
|
177
177
|
* `read_dlo(name)` – Read from a Data Lake Object by name
|
|
178
178
|
* `read_dmo(name)` – Read from a Data Model Object by name
|
|
179
179
|
* `write_to_dlo(name, spark_dataframe, write_mode)` – Write to a Data Model Object by name with a Spark dataframe
|
|
180
180
|
* `write_to_dmo(name, spark_dataframe, write_mode)` – Write to a Data Lake Object by name with a Spark dataframe
|
|
181
181
|
|
|
182
|
-
For a streaming (delta) transform, use `StreamingClient`, which exposes the streaming counterparts:
|
|
183
|
-
* `read_dlo_deltas()` – Read the streaming change feed (deltas) of a Data Lake Object as a streaming DataFrame.
|
|
184
|
-
* `read_dmo_deltas()` – Read the streaming change feed (deltas) of a Data Model Object as a streaming DataFrame.
|
|
185
|
-
* `write_dlo_deltas(name, spark_dataframe)` – Write a streaming DataFrame of deltas to a Data Lake Object; returns the started `StreamingQuery`
|
|
186
|
-
|
|
187
|
-
`find_file_path`, `llm_gateway_generate_text`, and `einstein_predict` are available on both clients.
|
|
188
|
-
|
|
189
182
|
For example:
|
|
190
183
|
```python
|
|
191
184
|
from datacustomcode import Client
|
|
@@ -201,36 +194,6 @@ client.write_to_dlo('output_DLO')
|
|
|
201
194
|
> [!WARNING]
|
|
202
195
|
> Currently we only support reading from DMOs and writing to DMOs or reading from DLOs and writing to DLOs, but they cannot mix.
|
|
203
196
|
|
|
204
|
-
### Streaming (delta) transforms
|
|
205
|
-
|
|
206
|
-
Streaming BYOC transforms process a Data Lake Object's Change Data Feed continuously instead of reading a bounded snapshot. Use a `StreamingClient` and its `*_deltas` methods in place of the batch `Client` read/write methods:
|
|
207
|
-
|
|
208
|
-
```python
|
|
209
|
-
from pyspark.sql.functions import col, upper
|
|
210
|
-
|
|
211
|
-
from datacustomcode import StreamingClient
|
|
212
|
-
|
|
213
|
-
client = StreamingClient()
|
|
214
|
-
|
|
215
|
-
# read_dlo_deltas returns a *streaming* DataFrame over the change feed.
|
|
216
|
-
# The runtime resolves the single streaming source, so no name is passed.
|
|
217
|
-
deltas = client.read_dlo_deltas()
|
|
218
|
-
|
|
219
|
-
# Ordinary PySpark transform.
|
|
220
|
-
transformed = deltas.withColumn("description__c", upper(col("description__c")))
|
|
221
|
-
|
|
222
|
-
# write_dlo_deltas starts a streaming query and returns the StreamingQuery.
|
|
223
|
-
# The runtime owns the trigger and checkpoint location; you
|
|
224
|
-
# choose only the target table.
|
|
225
|
-
query = client.write_dlo_deltas("Output__dll", transformed)
|
|
226
|
-
query.awaitTermination()
|
|
227
|
-
```
|
|
228
|
-
|
|
229
|
-
Notes:
|
|
230
|
-
|
|
231
|
-
- These methods only run inside the Data Cloud streaming (`DELTA_SYNC`) runtime. Locally (`datacustomcode run`) they raise `NotImplementedError`, since there is no change feed to stream.
|
|
232
|
-
- A complete runnable entry point is provided in [`examples/streaming_deltas/entrypoint.py`](src/datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py).
|
|
233
|
-
|
|
234
197
|
### Bundled file resolution
|
|
235
198
|
|
|
236
199
|
Place bundled files (CSVs, prompt files, etc.) under `payload/files/`. The same `client.find_file_path("data.csv")` call resolves consistently across all three runtimes:
|
|
@@ -1,14 +1,14 @@
|
|
|
1
|
-
datacustomcode/__init__.py,sha256=
|
|
1
|
+
datacustomcode/__init__.py,sha256=SGsDnWDTihWkLNnWdqvH08zklQgxykar73JAcPL2plI,2666
|
|
2
2
|
datacustomcode/auth.py,sha256=fpSjhIBdv9trC8yq2vuljAix_Euu-4Ah7HDCGhYjOxI,8309
|
|
3
|
-
datacustomcode/cli.py,sha256=
|
|
4
|
-
datacustomcode/client.py,sha256=
|
|
3
|
+
datacustomcode/cli.py,sha256=i7hPoQqJskTmx3Odz2wVgkxBnf_PFaIawI12FXhaRUE,13264
|
|
4
|
+
datacustomcode/client.py,sha256=lbOseiGaDru-vUtdRXgAORhrJa5V5WrFkbcEIfzBkaU,24325
|
|
5
5
|
datacustomcode/cmd.py,sha256=ZMs46aydJw2EaU26JgCtZmnqESQFHvvaJz10hnjZTBk,3537
|
|
6
6
|
datacustomcode/common_config.py,sha256=SAUnxj3kqmOeWwPmFoYq4tuxokMgURVm4QwcGi-avL4,1928
|
|
7
|
-
datacustomcode/config.py,sha256=
|
|
8
|
-
datacustomcode/config.yaml,sha256=
|
|
9
|
-
datacustomcode/constants.py,sha256=
|
|
7
|
+
datacustomcode/config.py,sha256=2Pk61ieQsEWSxKDxlS66_rliKE8iX0j-9LweEm0aaGo,4072
|
|
8
|
+
datacustomcode/config.yaml,sha256=LqpbkZhzN0nZOPEPnaz2qOCbchp8fxC4GHamT_b56D0,1082
|
|
9
|
+
datacustomcode/constants.py,sha256=SskQpUPnQE7LjKh4Z3AdZ2p191_0Kf73jV8gdBXDPLU,1436
|
|
10
10
|
datacustomcode/credentials.py,sha256=D-7Zd3Nh_wStdj8_wUy5cnC8M_mdbfBijhIAi-2EbXs,9467
|
|
11
|
-
datacustomcode/deploy.py,sha256=
|
|
11
|
+
datacustomcode/deploy.py,sha256=T-UzXf_5RWuyJHOrfFieD7X5SKJ-2qXuNqQ3LlBM7JQ,22143
|
|
12
12
|
datacustomcode/einstein_platform_client.py,sha256=ON2B_m--vdbCVVMVcLc81T-BRZ5-UuxVCer8E6OEQPA,4083
|
|
13
13
|
datacustomcode/einstein_platform_config.py,sha256=6lb_FRbEzdt5a8-2cR15f13ZKw674uvRJ4Xq0o_pQXE,1490
|
|
14
14
|
datacustomcode/einstein_predictions/__init__.py,sha256=_RIDTqcEHDRu6AsCsj9lRBykNkHI8Ch8JBILL1ioPtA,1237
|
|
@@ -27,17 +27,17 @@ datacustomcode/function/__init__.py,sha256=oag8FevDxbx1vR0tFhtxPd5SvmUEYDj8esRo4
|
|
|
27
27
|
datacustomcode/function/base.py,sha256=H_i80GcsFScvHJJ6bYLry5phTpsEU1KubpR8D3qnsjo,691
|
|
28
28
|
datacustomcode/function/feature_types/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
|
|
29
29
|
datacustomcode/function/feature_types/chunking.py,sha256=_tCCHj7slD59-aLLEZUFRPOX37q_MEKkL4thi9CNwMw,7180
|
|
30
|
-
datacustomcode/function/runtime.py,sha256=
|
|
30
|
+
datacustomcode/function/runtime.py,sha256=Q6h-vcW1TNhOh3q1rlmQ24zs-OloNkJexdtbFyIxt4o,4437
|
|
31
31
|
datacustomcode/function_utils.py,sha256=y6vaoSSaJ9CeHhjLCNyqzJT4vx6yCePWRxNLz38ddjQ,12920
|
|
32
32
|
datacustomcode/io/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
|
|
33
33
|
datacustomcode/io/base.py,sha256=gbwZWWVUbCbGR4jIg_4h4qOz8tOMjE4RDTleD23WFKo,973
|
|
34
34
|
datacustomcode/io/reader/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
|
|
35
|
-
datacustomcode/io/reader/base.py,sha256=
|
|
35
|
+
datacustomcode/io/reader/base.py,sha256=JRKg8KfEPJGR58wqicsEdpAPiXTl2K1eV3eHQyxYWaE,1390
|
|
36
36
|
datacustomcode/io/reader/query_api.py,sha256=eVrohrcnTnhSMsGfPRHu5XltjFtGG0hyHt1o4q1hp2w,9340
|
|
37
37
|
datacustomcode/io/reader/sf_cli.py,sha256=x5QacVqRZaZSyph1_wwxo67s8r29wsOcUsOaiF9cDWE,6324
|
|
38
38
|
datacustomcode/io/reader/utils.py,sha256=HlHhPZoHfmWA3mF8kTnd3m4Nd2exaz4NsOJJfQY7Pew,1656
|
|
39
39
|
datacustomcode/io/writer/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
|
|
40
|
-
datacustomcode/io/writer/base.py,sha256=
|
|
40
|
+
datacustomcode/io/writer/base.py,sha256=6e2Yszb6BiuN1zCV2rjvZ5CQ0YikQdUgdYAkutoy9pE,1787
|
|
41
41
|
datacustomcode/io/writer/csv.py,sha256=asty5teBpNQ1fMGHZ7wA3suLhq0sk0lVQPN5x7TOSRk,1555
|
|
42
42
|
datacustomcode/io/writer/print.py,sha256=0g2sP_1Wb95UIyEELWJt3dqM18Iv_CWhDW3mDxcIhns,5105
|
|
43
43
|
datacustomcode/llm_gateway/__init__.py,sha256=rcoGpSkv36tZuAOcTxKkhNpM0qV03qgmer_ej07Vmfc,1088
|
|
@@ -53,13 +53,30 @@ datacustomcode/llm_gateway/types/generate_text_response.py,sha256=NSvj6nvzR4cZ0c
|
|
|
53
53
|
datacustomcode/llm_gateway/types/generate_text_response_builder.py,sha256=6MV6JZuTyHUId3Slw9k7m0KBgjj-5S9HI4qtd94Jl2A,1272
|
|
54
54
|
datacustomcode/llm_gateway_config.py,sha256=ECm9CFqOFOaLHwenvVFAjiRSzG9Eg9fY4pb8nHbQ34c,3312
|
|
55
55
|
datacustomcode/mixin.py,sha256=ZtROqO1W1_D7MV8lw4cw6zzcchJLoZS0aAiXrolYR4k,4914
|
|
56
|
+
datacustomcode/named_credential/__init__.py,sha256=JhQWoboiZDczgXdHh5HLcv4K4lUI-sczhh-ceCBBfbA,1055
|
|
57
|
+
datacustomcode/named_credential/base.py,sha256=-LMskZkKNoaFg-Zc80M_4IPLF7ru5pfnbkfLI7YXrFc,1816
|
|
58
|
+
datacustomcode/named_credential/default.py,sha256=z03B0daZvTPiTmJWGAtDCICB1srelESg7QmPH2iRK5o,3339
|
|
59
|
+
datacustomcode/named_credential/direct/__init__.py,sha256=zqErMI1GqEhseLnYkrWtq3WipRi_dP153c8wwLvgI-o,799
|
|
60
|
+
datacustomcode/named_credential/direct/auth.py,sha256=U09jO0Pn3OEOEudmvcBmtkwoXqE0Ae_Mv6TAMqb3eG4,2201
|
|
61
|
+
datacustomcode/named_credential/direct/credentials.py,sha256=fibIZTyBT1IbAGFjXwAXqYcQt9o_sB8zFmuixGmibE4,4534
|
|
62
|
+
datacustomcode/named_credential/direct/transport.py,sha256=Lzrww-0E5XVrZEHwOPqdJID71XcQRRD5DX0nF7a4eiI,3984
|
|
63
|
+
datacustomcode/named_credential/direct/url_resolver.py,sha256=KItv12Az0qoxnVacCTcM9t9AY7qB9_tqPWMgiP-7Y6s,3823
|
|
64
|
+
datacustomcode/named_credential/spark_base.py,sha256=w34B-Pg6m_hcUqUbjmB2b6zUme_MSgg6VeghbMMYYN4,3288
|
|
65
|
+
datacustomcode/named_credential/spark_default.py,sha256=DY_STX9-1le8cS5M_vGqB3VUtQ2XCZZYDAmKId7YekA,5219
|
|
66
|
+
datacustomcode/named_credential/types/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
|
|
67
|
+
datacustomcode/named_credential/types/http_method.py,sha256=NFf8emmW_Y73q8q2pseWHqOQ-1SoFwKp2vCMr1OdmEU,952
|
|
68
|
+
datacustomcode/named_credential/types/http_request.py,sha256=LXW4EcvUfyOJHf8-VtFSRWy6JsTJJsd0N4-X6WxxFnA,2170
|
|
69
|
+
datacustomcode/named_credential/types/http_request_builder.py,sha256=qdToSlX2xw3FzuLF6Kas3VzCp1ubdXZb4_WkjQ5fBY0,1862
|
|
70
|
+
datacustomcode/named_credential/types/http_response.py,sha256=ZqX52RMjSdn8OpHSDkKJo9730QRx8o4_ZT-1noSMRe8,1480
|
|
71
|
+
datacustomcode/named_credential/types/http_response_builder.py,sha256=MoUjpT-P5W2uT1-h1rIfKpBQHpMj_ZA_nUmn-FyBdbQ,896
|
|
72
|
+
datacustomcode/named_credential_config.py,sha256=M07-oWzYj0KTrfA0fqvk3BQvaaDaw7C6nxbR0s3lOUk,3645
|
|
56
73
|
datacustomcode/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
57
|
-
datacustomcode/run.py,sha256=
|
|
58
|
-
datacustomcode/scan.py,sha256=
|
|
74
|
+
datacustomcode/run.py,sha256=PXfpIwcgqyW6KLawTUuzGSt4FeAcxtafg-FgBHGXOLg,8091
|
|
75
|
+
datacustomcode/scan.py,sha256=Zb9mE733H_xaqyDI-LZjjnIcctMC9j1AaD_kcr9qTh4,14325
|
|
59
76
|
datacustomcode/spark/__init__.py,sha256=12drVVlRiczCxOQw-EzuGtLsikCM8baBXvDEgwvelCI,860
|
|
60
77
|
datacustomcode/spark/base.py,sha256=tlGqM4LxuLoDa7OxJF5nVeb7phO5uvD1DHBQeH7NMMs,1036
|
|
61
78
|
datacustomcode/spark/default.py,sha256=aMB8CaTPYwHbQE_7XqBLQUZPGRdDJcRL72qTVh0aG7M,1433
|
|
62
|
-
datacustomcode/template.py,sha256=
|
|
79
|
+
datacustomcode/template.py,sha256=FNNi8YPfd-JB-hUt1DDWSFK5AqwKnTnRSy_KpqYluaU,3282
|
|
63
80
|
datacustomcode/templates/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
64
81
|
datacustomcode/templates/function/.devcontainer/devcontainer.json,sha256=21RNTadhC-rynON2-VOtr5U174Hxnr_MA4RQ4592rsg,166
|
|
65
82
|
datacustomcode/templates/function/Dockerfile.dependencies,sha256=AfHRddm5l3ujv4vdrf0d-SMB-qxPCOa0XQm6ptP2Euw,174
|
|
@@ -68,6 +85,11 @@ datacustomcode/templates/function/build_native_dependencies.sh,sha256=dk9vhzlObd
|
|
|
68
85
|
datacustomcode/templates/function/chunking/payload/config.json,sha256=RBNvo1WzZ4oRRq0W9-hknpT7T8If536DEMBg9hyq_4o,2
|
|
69
86
|
datacustomcode/templates/function/chunking/payload/entrypoint.py,sha256=tzpQJjODulwzLilOoKdOo8GjpdIlCDByjnawpHIQYQA,4517
|
|
70
87
|
datacustomcode/templates/function/chunking/requirements.txt,sha256=Ih0KsVdiTdo7KuIsqFB_AJMRT2r0ao_xDEb3hiDmf48,46
|
|
88
|
+
datacustomcode/templates/function/example/chunking_with_external_callout/README.md,sha256=RIgoUysFniAE4lC5LiZl5u55Rro1w5gmM2mMnsQF_HI,4716
|
|
89
|
+
datacustomcode/templates/function/example/chunking_with_external_callout/config.json,sha256=EvjuOVssRg2SjcUb5F50RRMGDqag_f69bv8AOflaK1g,38
|
|
90
|
+
datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py,sha256=QqJLeJHgtW4RkDwTaUIsb2ng_ODwwClXpn5F7XyXBes,5296
|
|
91
|
+
datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json,sha256=PUCmw_humbVoCyswFneuGYx0INbqSTPOumDqtGOztiM,320
|
|
92
|
+
datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json,sha256=ZqUthCT9F9HoyT493tXg4nFDSbEAeaFHY1iHo-9Qx7g,2703
|
|
71
93
|
datacustomcode/templates/function/example/chunking_with_llm/config.json,sha256=EvjuOVssRg2SjcUb5F50RRMGDqag_f69bv8AOflaK1g,38
|
|
72
94
|
datacustomcode/templates/function/example/chunking_with_llm/entrypoint.py,sha256=6AMVf_busCCDALktG8vkCtjnzJheMgM-OOLML5iLKOg,3437
|
|
73
95
|
datacustomcode/templates/function/example/chunking_with_llm/files/chunking_prompt.txt,sha256=w_gSYNXwZmbMTMjZ_YYwm1IpMsVpI3hyC3ew60zqs_0,446
|
|
@@ -88,7 +110,6 @@ datacustomcode/templates/script/account.ipynb,sha256=LIbxgiVxflNASdspF2lfpMKkKAT
|
|
|
88
110
|
datacustomcode/templates/script/build_native_dependencies.sh,sha256=ICRrp4f1ATBwPaUiuVGjY-MwiubFswNJLf8gMGU6YNg,197
|
|
89
111
|
datacustomcode/templates/script/examples/employee_hierarchy/employee_data.csv,sha256=C7ggLBfoyi3M2BdMLNyOeKqF-5OO-Da76lkwgWg2cVQ,302
|
|
90
112
|
datacustomcode/templates/script/examples/employee_hierarchy/entrypoint.py,sha256=Mfm3iQtEHTQRW6cmZTwRKdTh3IeGWR-teXrd6x-YLSY,2223
|
|
91
|
-
datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py,sha256=1PpYyZck2j9IRTnoI-kfHbMf21vN8gA49_-qJr3rDjk,1984
|
|
92
113
|
datacustomcode/templates/script/jupyterlab.sh,sha256=IHR3YQ8d_busuyvesByhJgzCKULIFPI-Hqogs0NPhCs,2432
|
|
93
114
|
datacustomcode/templates/script/payload/config.json,sha256=0d2mEMt4NHIeOUTsgiPMuRKdOLIQxmh-Wq-IfEqU-gc,28
|
|
94
115
|
datacustomcode/templates/script/payload/entrypoint.py,sha256=4Uph0ILa5ukk_6C90paK3uzQn0zZvNLr9ho3o_ZwErQ,2474
|
|
@@ -96,8 +117,8 @@ datacustomcode/templates/script/requirements-dev.txt,sha256=OWwuy1awesqOZOS3Zizm
|
|
|
96
117
|
datacustomcode/templates/script/requirements.txt,sha256=qJbqzs5z0Hrx590U3T7dHq_m8UQEx2TdhgrJSz-QHIQ,40
|
|
97
118
|
datacustomcode/token_provider.py,sha256=qA_e4vqSXnxn0MccFFPULWO8ZeUOEAu5QpBnoYCg-fc,6839
|
|
98
119
|
datacustomcode/version.py,sha256=9LlbVrzwBvut1L308QbwNBgxdQk5CrghH39JARVSxms,989
|
|
99
|
-
salesforce_data_customcode-6.1.0.
|
|
100
|
-
salesforce_data_customcode-6.1.0.
|
|
101
|
-
salesforce_data_customcode-6.1.0.
|
|
102
|
-
salesforce_data_customcode-6.1.0.
|
|
103
|
-
salesforce_data_customcode-6.1.0.
|
|
120
|
+
salesforce_data_customcode-6.1.0.dev2.dist-info/METADATA,sha256=tR2bLdFaIF7VlDb_imF7RPwGsNDI7HpiTDNaGjOcmnw,25115
|
|
121
|
+
salesforce_data_customcode-6.1.0.dev2.dist-info/WHEEL,sha256=eY7nduwzv-ldUxpzbRlxwvC693Hg6PX8bWDjEHjZ_dk,88
|
|
122
|
+
salesforce_data_customcode-6.1.0.dev2.dist-info/entry_points.txt,sha256=WpQ94UB7UuRCYGOLtJV3vAgm1tl7iVf43VYRWIuNu5Y,74
|
|
123
|
+
salesforce_data_customcode-6.1.0.dev2.dist-info/licenses/LICENSE.txt,sha256=iOi8EmQpfkFhMENi7VYtkl2EqS14pEOccoXHiW2dyPU,11443
|
|
124
|
+
salesforce_data_customcode-6.1.0.dev2.dist-info/RECORD,,
|
|
@@ -1,49 +0,0 @@
|
|
|
1
|
-
"""Streaming BYOC transform: read a DLO change feed and write the deltas back.
|
|
2
|
-
|
|
3
|
-
This example is the streaming counterpart to a normal batch entrypoint. Instead
|
|
4
|
-
of a batch ``Client`` with ``read_dlo`` / ``write_to_dlo`` (which read and write
|
|
5
|
-
a bounded snapshot), it uses a :class:`StreamingClient` and its streaming delta
|
|
6
|
-
methods:
|
|
7
|
-
|
|
8
|
-
* ``client.read_dlo_deltas()`` returns a *streaming* DataFrame over the
|
|
9
|
-
Change Data Feed of the source DLO. Each row carries the source columns plus
|
|
10
|
-
change-feed metadata columns (``_record_type``, ``_commit_*``).
|
|
11
|
-
* ``client.write_dlo_deltas(name, df)`` starts a streaming query that writes
|
|
12
|
-
each micro-batch to the target DLO and returns the ``StreamingQuery`` handle.
|
|
13
|
-
The runtime owns the trigger, and checkpoint location — the caller only
|
|
14
|
-
chooses the table.
|
|
15
|
-
|
|
16
|
-
The transform in between is ordinary PySpark. Because the source is a change
|
|
17
|
-
feed, keep the metadata columns on the DataFrame you hand to
|
|
18
|
-
``write_dlo_deltas`` — the sink relies on them to merge changes correctly.
|
|
19
|
-
|
|
20
|
-
This entrypoint only runs inside the Data Cloud streaming (``DELTA_SYNC``)
|
|
21
|
-
runtime; the local ``datacustomcode run`` readers/writers raise
|
|
22
|
-
``NotImplementedError`` for the delta methods.
|
|
23
|
-
"""
|
|
24
|
-
|
|
25
|
-
from pyspark.sql.functions import col, upper
|
|
26
|
-
|
|
27
|
-
from datacustomcode.client import StreamingClient
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
def main():
|
|
31
|
-
client = StreamingClient()
|
|
32
|
-
|
|
33
|
-
# Streaming DataFrame over the source DLO's change feed.
|
|
34
|
-
deltas = client.read_dlo_deltas()
|
|
35
|
-
|
|
36
|
-
# Ordinary PySpark transform.
|
|
37
|
-
transformed = deltas.withColumn("description__c", upper(col("description__c")))
|
|
38
|
-
|
|
39
|
-
# Start the streaming write. write_dlo_deltas returns the StreamingQuery;
|
|
40
|
-
# the trigger and checkpoint location are provided by the runtime.
|
|
41
|
-
query = client.write_dlo_deltas("Account_std_copy__dll", transformed)
|
|
42
|
-
|
|
43
|
-
# Drive the query's lifecycle. In the streaming runtime this blocks until
|
|
44
|
-
# the job is stopped by the platform.
|
|
45
|
-
query.awaitTermination()
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
if __name__ == "__main__":
|
|
49
|
-
main()
|
|
File without changes
|
|
File without changes
|