salesforce-data-customcode 6.1.0.dev3__py3-none-any.whl → 6.1.0.dev5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datacustomcode/__init__.py +5 -0
- datacustomcode/cli.py +30 -5
- datacustomcode/client.py +267 -192
- datacustomcode/config.py +5 -0
- datacustomcode/config.yaml +0 -6
- datacustomcode/constants.py +9 -1
- datacustomcode/deploy.py +58 -24
- datacustomcode/function/runtime.py +0 -16
- datacustomcode/io/reader/base.py +42 -0
- datacustomcode/io/writer/base.py +45 -0
- datacustomcode/io/writer/csv.py +8 -0
- datacustomcode/io/writer/print.py +7 -0
- datacustomcode/run.py +11 -7
- datacustomcode/scan.py +164 -29
- datacustomcode/template.py +13 -1
- datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +62 -0
- datacustomcode/templates/script/jupyterlab.sh +18 -4
- {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/METADATA +42 -5
- {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/RECORD +22 -44
- datacustomcode/named_credential/__init__.py +0 -28
- datacustomcode/named_credential/base.py +0 -54
- datacustomcode/named_credential/default.py +0 -93
- datacustomcode/named_credential/direct/__init__.py +0 -19
- datacustomcode/named_credential/direct/auth.py +0 -63
- datacustomcode/named_credential/direct/credentials.py +0 -121
- datacustomcode/named_credential/direct/transport.py +0 -110
- datacustomcode/named_credential/direct/url_resolver.py +0 -112
- datacustomcode/named_credential/errors.py +0 -36
- datacustomcode/named_credential/spark_base.py +0 -93
- datacustomcode/named_credential/spark_default.py +0 -154
- datacustomcode/named_credential/types/__init__.py +0 -14
- datacustomcode/named_credential/types/http_method.py +0 -29
- datacustomcode/named_credential/types/http_request.py +0 -63
- datacustomcode/named_credential/types/http_request_builder.py +0 -55
- datacustomcode/named_credential/types/http_response.py +0 -43
- datacustomcode/named_credential/types/http_response_builder.py +0 -24
- datacustomcode/named_credential_config.py +0 -105
- datacustomcode/templates/function/example/chunking_with_external_callout/README.md +0 -119
- datacustomcode/templates/function/example/chunking_with_external_callout/config.json +0 -3
- datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +0 -161
- datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +0 -11
- datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +0 -16
- {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/WHEEL +0 -0
- {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/entry_points.txt +0 -0
- {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/licenses/LICENSE.txt +0 -0
datacustomcode/__init__.py
CHANGED
|
@@ -23,6 +23,7 @@ __all__ = [
|
|
|
23
23
|
"QueryAPIDataCloudReader",
|
|
24
24
|
"SparkEinsteinPredictions",
|
|
25
25
|
"SparkLLMGateway",
|
|
26
|
+
"StreamingClient",
|
|
26
27
|
"einstein_predict_col",
|
|
27
28
|
"llm_gateway_generate_text_col",
|
|
28
29
|
]
|
|
@@ -34,6 +35,10 @@ def __getattr__(name: str):
|
|
|
34
35
|
from datacustomcode.client import Client
|
|
35
36
|
|
|
36
37
|
return Client
|
|
38
|
+
elif name == "StreamingClient":
|
|
39
|
+
from datacustomcode.client import StreamingClient
|
|
40
|
+
|
|
41
|
+
return StreamingClient
|
|
37
42
|
elif name == "AuthType":
|
|
38
43
|
from datacustomcode.credentials import AuthType
|
|
39
44
|
|
datacustomcode/cli.py
CHANGED
|
@@ -262,7 +262,7 @@ def deploy(
|
|
|
262
262
|
mapped_feature = USE_IN_FEATURE_MAPPING_FOR_CONNECT_API.get(
|
|
263
263
|
use_in_feature, use_in_feature
|
|
264
264
|
)
|
|
265
|
-
metadata.
|
|
265
|
+
metadata.functionInvokeOptions = [mapped_feature]
|
|
266
266
|
|
|
267
267
|
try:
|
|
268
268
|
if sf_cli_org:
|
|
@@ -283,10 +283,19 @@ def deploy(
|
|
|
283
283
|
)
|
|
284
284
|
@click.option(
|
|
285
285
|
"--use-in-feature",
|
|
286
|
-
|
|
287
|
-
|
|
286
|
+
"-u",
|
|
287
|
+
default=None,
|
|
288
|
+
help=(
|
|
289
|
+
"Invoke option for this package. For scripts: 'BatchTransform' "
|
|
290
|
+
"(default) or 'StreamingTransform'. For functions: 'SearchIndexChunking'."
|
|
291
|
+
),
|
|
288
292
|
)
|
|
289
293
|
def init(directory: str, code_type: str, use_in_feature: Optional[str]):
|
|
294
|
+
from datacustomcode.constants import (
|
|
295
|
+
SCRIPT_USE_IN_FEATURE_BATCH,
|
|
296
|
+
SCRIPT_USE_IN_FEATURE_OPTIONS,
|
|
297
|
+
SCRIPT_USE_IN_FEATURE_STREAMING,
|
|
298
|
+
)
|
|
290
299
|
from datacustomcode.scan import (
|
|
291
300
|
dc_config_json_from_file,
|
|
292
301
|
update_config,
|
|
@@ -294,9 +303,23 @@ def init(directory: str, code_type: str, use_in_feature: Optional[str]):
|
|
|
294
303
|
)
|
|
295
304
|
from datacustomcode.template import copy_function_template, copy_script_template
|
|
296
305
|
|
|
306
|
+
streaming = False
|
|
307
|
+
if code_type == "script":
|
|
308
|
+
use_in_feature = use_in_feature or SCRIPT_USE_IN_FEATURE_BATCH
|
|
309
|
+
if use_in_feature not in SCRIPT_USE_IN_FEATURE_OPTIONS:
|
|
310
|
+
click.secho(
|
|
311
|
+
f"Error: Invalid --use-in-feature '{use_in_feature}' for a "
|
|
312
|
+
f"script. Valid options: {', '.join(SCRIPT_USE_IN_FEATURE_OPTIONS)}.",
|
|
313
|
+
fg="red",
|
|
314
|
+
)
|
|
315
|
+
raise click.Abort()
|
|
316
|
+
streaming = use_in_feature == SCRIPT_USE_IN_FEATURE_STREAMING
|
|
317
|
+
else:
|
|
318
|
+
use_in_feature = use_in_feature or "SearchIndexChunking"
|
|
319
|
+
|
|
297
320
|
click.echo("Copying template to " + click.style(directory, fg="blue", bold=True))
|
|
298
321
|
if code_type == "script":
|
|
299
|
-
copy_script_template(directory)
|
|
322
|
+
copy_script_template(directory, streaming=streaming)
|
|
300
323
|
elif code_type == "function":
|
|
301
324
|
copy_function_template(directory, use_in_feature)
|
|
302
325
|
entrypoint_path = os.path.join(directory, PAYLOAD_DIR, ENTRYPOINT_FILE)
|
|
@@ -306,7 +329,9 @@ def init(directory: str, code_type: str, use_in_feature: Optional[str]):
|
|
|
306
329
|
sdk_config = {"type": code_type}
|
|
307
330
|
write_sdk_config(directory, sdk_config)
|
|
308
331
|
|
|
309
|
-
config_json = dc_config_json_from_file(
|
|
332
|
+
config_json = dc_config_json_from_file(
|
|
333
|
+
entrypoint_path, code_type, streaming=streaming
|
|
334
|
+
)
|
|
310
335
|
with open(config_location, "w") as f:
|
|
311
336
|
json.dump(config_json, f, indent=2)
|
|
312
337
|
|
datacustomcode/client.py
CHANGED
|
@@ -15,13 +15,16 @@
|
|
|
15
15
|
from __future__ import annotations
|
|
16
16
|
|
|
17
17
|
from enum import Enum
|
|
18
|
+
import os
|
|
18
19
|
from typing import (
|
|
19
20
|
TYPE_CHECKING,
|
|
20
21
|
Any,
|
|
21
22
|
ClassVar,
|
|
22
23
|
Dict,
|
|
23
24
|
Optional,
|
|
25
|
+
TypeVar,
|
|
24
26
|
Union,
|
|
27
|
+
cast,
|
|
25
28
|
)
|
|
26
29
|
|
|
27
30
|
from datacustomcode.config import config
|
|
@@ -29,25 +32,60 @@ from datacustomcode.einstein_predictions_config import spark_einstein_prediction
|
|
|
29
32
|
from datacustomcode.file.path.default import DefaultFindFilePath
|
|
30
33
|
from datacustomcode.io.reader.base import BaseDataCloudReader
|
|
31
34
|
from datacustomcode.llm_gateway_config import spark_llm_gateway_config
|
|
32
|
-
from datacustomcode.named_credential_config import spark_named_credential_config
|
|
33
35
|
from datacustomcode.spark.default import DefaultSparkSessionProvider
|
|
34
36
|
|
|
35
37
|
if TYPE_CHECKING:
|
|
36
38
|
from pathlib import Path
|
|
37
39
|
|
|
38
|
-
from pyspark.sql import
|
|
40
|
+
from pyspark.sql import (
|
|
41
|
+
Column,
|
|
42
|
+
DataFrame as PySparkDataFrame,
|
|
43
|
+
SparkSession,
|
|
44
|
+
)
|
|
45
|
+
from pyspark.sql.streaming import StreamingQuery
|
|
39
46
|
|
|
40
47
|
from datacustomcode.einstein_predictions.spark_base import SparkEinsteinPredictions
|
|
41
48
|
from datacustomcode.einstein_predictions.types import PredictionType
|
|
42
49
|
from datacustomcode.io.reader.base import BaseDataCloudReader
|
|
43
50
|
from datacustomcode.io.writer.base import BaseDataCloudWriter, WriteMode
|
|
44
51
|
from datacustomcode.llm_gateway.spark_base import SparkLLMGateway
|
|
45
|
-
from datacustomcode.named_credential.spark_base import SparkNamedCredential
|
|
46
|
-
from datacustomcode.named_credential.types.http_request import HTTPRequest
|
|
47
|
-
from datacustomcode.named_credential.types.http_response import HTTPResponse
|
|
48
52
|
from datacustomcode.spark.base import BaseSparkSessionProvider
|
|
49
53
|
|
|
50
54
|
|
|
55
|
+
def _streaming_source_name() -> str:
|
|
56
|
+
"""Return the streaming transform's read-source name.
|
|
57
|
+
|
|
58
|
+
Resolved from ``config.streaming_source``, which ``run_entrypoint``
|
|
59
|
+
populates from config.json's ``streamingSource`` field.
|
|
60
|
+
|
|
61
|
+
Raises:
|
|
62
|
+
RuntimeError: If no ``streaming_source`` has been configured (e.g. the
|
|
63
|
+
transform's config.json has no ``streamingSource`` field).
|
|
64
|
+
"""
|
|
65
|
+
source = config.streaming_source
|
|
66
|
+
if not source:
|
|
67
|
+
raise RuntimeError(
|
|
68
|
+
"No streaming source configured. A streaming transform must declare "
|
|
69
|
+
"its read source in config.json under 'streamingSource'."
|
|
70
|
+
)
|
|
71
|
+
return source
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _active_client() -> "_BaseClient":
|
|
75
|
+
"""Return the client backing the module-level Spark column helpers.
|
|
76
|
+
|
|
77
|
+
Prefers an already-initialized singleton so a streaming job reuses its
|
|
78
|
+
:class:`StreamingClient` (and a batch job its :class:`Client`) rather than
|
|
79
|
+
forcing an unrelated client into existence. Falls back to building the
|
|
80
|
+
batch :class:`Client` when neither has been created yet.
|
|
81
|
+
"""
|
|
82
|
+
if Client._instance is not None:
|
|
83
|
+
return Client._instance
|
|
84
|
+
if StreamingClient._instance is not None:
|
|
85
|
+
return StreamingClient._instance
|
|
86
|
+
return Client()
|
|
87
|
+
|
|
88
|
+
|
|
51
89
|
def _build_spark_llm_gateway() -> "SparkLLMGateway":
|
|
52
90
|
"""Instantiate the SDK-configured :class:`SparkLLMGateway`.
|
|
53
91
|
|
|
@@ -103,7 +141,7 @@ def llm_gateway_generate_text_col(
|
|
|
103
141
|
the generated text; on failure, ``status == "ERROR"`` and the
|
|
104
142
|
``error_*`` fields carry diagnostic detail.
|
|
105
143
|
"""
|
|
106
|
-
gateway =
|
|
144
|
+
gateway = _active_client()._get_spark_llm_gateway()
|
|
107
145
|
return gateway.llm_gateway_generate_text_col(template, values, model_id=model_id)
|
|
108
146
|
|
|
109
147
|
|
|
@@ -122,21 +160,6 @@ def _build_spark_einstein_predictions() -> "SparkEinsteinPredictions":
|
|
|
122
160
|
return cfg.to_object()
|
|
123
161
|
|
|
124
162
|
|
|
125
|
-
def _build_spark_named_credential() -> "SparkNamedCredential":
|
|
126
|
-
"""Instantiate the SDK-configured :class:`SparkNamedCredential`.
|
|
127
|
-
|
|
128
|
-
Raises:
|
|
129
|
-
RuntimeError: If no ``spark_named_credential_config`` has been loaded.
|
|
130
|
-
"""
|
|
131
|
-
cfg = spark_named_credential_config.spark_named_credential_config
|
|
132
|
-
if cfg is None:
|
|
133
|
-
raise RuntimeError(
|
|
134
|
-
"spark_named_credential_config is not configured. Add a "
|
|
135
|
-
"'spark_named_credential_config' section to config.yaml."
|
|
136
|
-
)
|
|
137
|
-
return cfg.to_object()
|
|
138
|
-
|
|
139
|
-
|
|
140
163
|
def einstein_predict_col(
|
|
141
164
|
model_api_name: str,
|
|
142
165
|
prediction_type: "PredictionType",
|
|
@@ -180,46 +203,12 @@ def einstein_predict_col(
|
|
|
180
203
|
the JSON-serialized prediction payload; on failure, ``status ==
|
|
181
204
|
"ERROR"`` and the ``error_*`` fields carry diagnostic detail.
|
|
182
205
|
"""
|
|
183
|
-
predictions =
|
|
206
|
+
predictions = _active_client()._get_spark_einstein_predictions()
|
|
184
207
|
return predictions.einstein_predict_col(
|
|
185
208
|
model_api_name, prediction_type, features, settings=settings
|
|
186
209
|
)
|
|
187
210
|
|
|
188
211
|
|
|
189
|
-
def named_credential_request_col(
|
|
190
|
-
request: "HTTPRequest",
|
|
191
|
-
body: Optional["Column"] = None,
|
|
192
|
-
) -> "Column":
|
|
193
|
-
"""Build a Spark Column that makes one Named Credential callout per row.
|
|
194
|
-
|
|
195
|
-
The endpoint, method, and headers are fixed for the call (taken from
|
|
196
|
-
``request``); only ``body`` varies per row. Use this instead of
|
|
197
|
-
:meth:`Client.named_credential_request` when the callout runs across a
|
|
198
|
-
DataFrame so each row is dispatched independently rather than one-shot on
|
|
199
|
-
the driver.
|
|
200
|
-
|
|
201
|
-
The returned Column yields a struct ``{status, response, error_code,
|
|
202
|
-
error_message}`` for each row. ``response`` is itself a struct
|
|
203
|
-
``{status_code, body, headers}``. Use ``[...]`` to pick a field, e.g.
|
|
204
|
-
``named_credential_request_col(...)["response"]["status_code"]``. A transport
|
|
205
|
-
failure sets ``status`` to ``ERROR`` and populates ``error_message`` (a non-2xx
|
|
206
|
-
HTTP response is still ``SUCCESS`` with its code in ``response.status_code``),
|
|
207
|
-
so a single bad row does not abort the whole Spark job.
|
|
208
|
-
|
|
209
|
-
Args:
|
|
210
|
-
request: The callout template — its symbolic reference, method, and
|
|
211
|
-
headers are applied to every row.
|
|
212
|
-
body: Optional per-row ``Column`` holding the request body as a
|
|
213
|
-
string (or null for no body).
|
|
214
|
-
|
|
215
|
-
Returns:
|
|
216
|
-
A Spark ``Column`` of ``StructType`` with fields ``status``,
|
|
217
|
-
``response``, ``error_code``, and ``error_message``.
|
|
218
|
-
"""
|
|
219
|
-
named_credential = Client()._get_spark_named_credential()
|
|
220
|
-
return named_credential.request_col(request, body=body)
|
|
221
|
-
|
|
222
|
-
|
|
223
212
|
class DataCloudObjectType(Enum):
|
|
224
213
|
DLO = "dlo"
|
|
225
214
|
DMO = "dmo"
|
|
@@ -258,83 +247,86 @@ class DataCloudAccessLayerException(Exception):
|
|
|
258
247
|
return msg
|
|
259
248
|
|
|
260
249
|
|
|
261
|
-
|
|
262
|
-
|
|
250
|
+
_ClientT = TypeVar("_ClientT", bound="_BaseClient")
|
|
251
|
+
|
|
263
252
|
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
can read from DLOs and write to DLOs, read from DMOs and write to DMOs, but you
|
|
267
|
-
cannot read from DLOs and write to DMOs or read from DMOs and write to DLOs.
|
|
268
|
-
Furthermore you cannot mix during merging tables. This class is a singleton to
|
|
269
|
-
prevent accidental mixing of DLOs and DMOs.
|
|
253
|
+
class _BaseClient:
|
|
254
|
+
"""Shared machinery for the Data Cloud client singletons.
|
|
270
255
|
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
256
|
+
Holds the wiring common to :class:`Client` (batch) and
|
|
257
|
+
:class:`StreamingClient`
|
|
258
|
+
|
|
259
|
+
This base class is not meant to be instantiated directly; use
|
|
260
|
+
:class:`Client` or :class:`StreamingClient`.
|
|
276
261
|
|
|
277
262
|
Args:
|
|
278
|
-
finder: Find a file path
|
|
279
263
|
reader: A custom reader to use for reading Data Cloud objects.
|
|
280
264
|
writer: A custom writer to use for writing Data Cloud objects.
|
|
265
|
+
spark_provider: Optional custom :class:`BaseSparkSessionProvider`.
|
|
281
266
|
spark_llm_gateway: Optional custom :class:`SparkLLMGateway`.
|
|
282
267
|
spark_einstein_predictions: Optional custom
|
|
283
268
|
:class:`SparkEinsteinPredictions`.
|
|
284
|
-
spark_named_credential: Optional custom :class:`SparkNamedCredential`.
|
|
285
|
-
|
|
286
|
-
Example:
|
|
287
|
-
>>> client = Client()
|
|
288
|
-
>>> file_path = client.find_file_path("data.csv")
|
|
289
|
-
>>> dlo = client.read_dlo("my_dlo")
|
|
290
|
-
>>> client.write_to_dmo("my_dmo", dlo)
|
|
291
|
-
>>> answer = client.llm_gateway_generate_text("Generate a greeting message")
|
|
292
269
|
"""
|
|
293
270
|
|
|
294
|
-
_instance:
|
|
271
|
+
# Each concrete subclass gets its own ``_instance`` slot: reads fall through
|
|
272
|
+
# to this base default of ``None``, but ``cls._instance = ...`` in __new__
|
|
273
|
+
# always writes to the subclass, so ``Client`` and ``StreamingClient`` never
|
|
274
|
+
# share an instance.
|
|
275
|
+
_instance: ClassVar[Optional[_BaseClient]] = None
|
|
276
|
+
# Process-wide Spark session shared across BOTH client types. Unlike
|
|
277
|
+
# ``_instance``, this is written via ``_BaseClient._shared_spark`` (never
|
|
278
|
+
# ``cls._shared_spark``), so the slot lives on the base class and a
|
|
279
|
+
# ``Client`` and a ``StreamingClient`` in the same process reuse one session
|
|
280
|
+
# — and therefore one underlying connection — instead of opening two
|
|
281
|
+
# containing differing state
|
|
282
|
+
_shared_spark: ClassVar[Optional[SparkSession]] = None
|
|
295
283
|
_reader: BaseDataCloudReader
|
|
296
284
|
_writer: BaseDataCloudWriter
|
|
297
285
|
_file: DefaultFindFilePath
|
|
298
286
|
_spark_llm_gateway: Optional[SparkLLMGateway]
|
|
299
287
|
_spark_einstein_predictions: Optional[SparkEinsteinPredictions]
|
|
300
|
-
_spark_named_credential: Optional[SparkNamedCredential]
|
|
301
288
|
_data_layer_history: dict[DataCloudObjectType, set[str]]
|
|
302
289
|
_code_type: str
|
|
303
290
|
|
|
304
291
|
def __new__(
|
|
305
|
-
cls,
|
|
292
|
+
cls: type[_ClientT],
|
|
306
293
|
reader: Optional[BaseDataCloudReader] = None,
|
|
307
294
|
writer: Optional[BaseDataCloudWriter] = None,
|
|
308
295
|
spark_provider: Optional[BaseSparkSessionProvider] = None,
|
|
309
296
|
spark_llm_gateway: Optional[SparkLLMGateway] = None,
|
|
310
297
|
spark_einstein_predictions: Optional[SparkEinsteinPredictions] = None,
|
|
311
|
-
spark_named_credential: Optional[SparkNamedCredential] = None,
|
|
312
298
|
code_type: str = "script",
|
|
313
|
-
) ->
|
|
299
|
+
) -> _ClientT:
|
|
314
300
|
|
|
315
301
|
if cls._instance is None:
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
cls._instance._spark_named_credential = spark_named_credential
|
|
302
|
+
instance = super().__new__(cls)
|
|
303
|
+
instance._spark_llm_gateway = spark_llm_gateway
|
|
304
|
+
instance._spark_einstein_predictions = spark_einstein_predictions
|
|
320
305
|
# Initialize Readers and Writers from config
|
|
321
306
|
# and/or provided reader and writer
|
|
322
307
|
if reader is None or writer is None:
|
|
323
|
-
# We need a spark because we will initialize readers and writers
|
|
324
|
-
if
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
provider: BaseSparkSessionProvider
|
|
330
|
-
if spark_provider is not None:
|
|
331
|
-
provider = spark_provider
|
|
332
|
-
elif config.spark_provider_config is not None:
|
|
333
|
-
provider = config.spark_provider_config.to_object()
|
|
308
|
+
# We need a spark because we will initialize readers and writers.
|
|
309
|
+
# Reuse the process-wide session if one client already built it,
|
|
310
|
+
# so a Client and a StreamingClient share a single connection.
|
|
311
|
+
if _BaseClient._shared_spark is not None:
|
|
312
|
+
spark = _BaseClient._shared_spark
|
|
334
313
|
else:
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
314
|
+
if config.spark_config is None:
|
|
315
|
+
raise ValueError(
|
|
316
|
+
"Spark config is required when reader/writer is not "
|
|
317
|
+
"provided"
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
provider: BaseSparkSessionProvider
|
|
321
|
+
if spark_provider is not None:
|
|
322
|
+
provider = spark_provider
|
|
323
|
+
elif config.spark_provider_config is not None:
|
|
324
|
+
provider = config.spark_provider_config.to_object()
|
|
325
|
+
else:
|
|
326
|
+
provider = DefaultSparkSessionProvider()
|
|
327
|
+
|
|
328
|
+
spark = provider.get_session(config.spark_config)
|
|
329
|
+
_BaseClient._shared_spark = spark
|
|
338
330
|
|
|
339
331
|
if config.reader_config is None and reader is None:
|
|
340
332
|
raise ValueError(
|
|
@@ -357,66 +349,17 @@ class Client:
|
|
|
357
349
|
else:
|
|
358
350
|
writer_init = writer
|
|
359
351
|
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
352
|
+
instance._reader = reader_init
|
|
353
|
+
instance._writer = writer_init
|
|
354
|
+
instance._file = DefaultFindFilePath()
|
|
355
|
+
instance._data_layer_history = {
|
|
364
356
|
DataCloudObjectType.DLO: set(),
|
|
365
357
|
DataCloudObjectType.DMO: set(),
|
|
366
358
|
}
|
|
367
|
-
|
|
359
|
+
cls._instance = instance
|
|
360
|
+
elif reader is not None or writer is not None:
|
|
368
361
|
raise ValueError("Cannot set reader or writer after client is initialized")
|
|
369
|
-
return cls._instance
|
|
370
|
-
|
|
371
|
-
def read_dlo(self, name: str) -> PySparkDataFrame:
|
|
372
|
-
"""Read a DLO from Data Cloud.
|
|
373
|
-
|
|
374
|
-
Args:
|
|
375
|
-
name: The name of the DLO to read.
|
|
376
|
-
|
|
377
|
-
Returns:
|
|
378
|
-
A PySpark DataFrame containing the DLO data.
|
|
379
|
-
"""
|
|
380
|
-
self._record_dlo_access(name)
|
|
381
|
-
return self._reader.read_dlo(name) # type: ignore[no-any-return]
|
|
382
|
-
|
|
383
|
-
def read_dmo(self, name: str) -> PySparkDataFrame:
|
|
384
|
-
"""Read a DMO from Data Cloud.
|
|
385
|
-
|
|
386
|
-
Args:
|
|
387
|
-
name: The name of the DMO to read.
|
|
388
|
-
|
|
389
|
-
Returns:
|
|
390
|
-
A PySpark DataFrame containing the DMO data.
|
|
391
|
-
"""
|
|
392
|
-
self._record_dmo_access(name)
|
|
393
|
-
return self._reader.read_dmo(name) # type: ignore[no-any-return]
|
|
394
|
-
|
|
395
|
-
def write_to_dlo(
|
|
396
|
-
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
397
|
-
) -> None:
|
|
398
|
-
"""Write a PySpark DataFrame to a DLO in Data Cloud.
|
|
399
|
-
|
|
400
|
-
Args:
|
|
401
|
-
name: The name of the DLO to write to.
|
|
402
|
-
dataframe: The PySpark DataFrame to write.
|
|
403
|
-
write_mode: The write mode to use for writing to the DLO.
|
|
404
|
-
"""
|
|
405
|
-
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
406
|
-
return self._writer.write_to_dlo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
407
|
-
|
|
408
|
-
def write_to_dmo(
|
|
409
|
-
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
410
|
-
) -> None:
|
|
411
|
-
"""Write a PySpark DataFrame to a DMO in Data Cloud.
|
|
412
|
-
|
|
413
|
-
Args:
|
|
414
|
-
name: The name of the DMO to write to.
|
|
415
|
-
dataframe: The PySpark DataFrame to write.
|
|
416
|
-
write_mode: The write mode to use for writing to the DMO.
|
|
417
|
-
"""
|
|
418
|
-
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
|
|
419
|
-
return self._writer.write_to_dmo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
362
|
+
return cast(_ClientT, cls._instance)
|
|
420
363
|
|
|
421
364
|
def find_file_path(self, file_name: str) -> Path:
|
|
422
365
|
"""Resolve a bundled file shipped in the package to an absolute path.
|
|
@@ -531,41 +474,6 @@ class Client:
|
|
|
531
474
|
self._spark_einstein_predictions = _build_spark_einstein_predictions()
|
|
532
475
|
return self._spark_einstein_predictions
|
|
533
476
|
|
|
534
|
-
def named_credential_request(
|
|
535
|
-
self,
|
|
536
|
-
request: "HTTPRequest",
|
|
537
|
-
body: Optional[str] = None,
|
|
538
|
-
) -> "HTTPResponse":
|
|
539
|
-
"""Issue a one-shot Named Credential external callout. This is the
|
|
540
|
-
scalar counterpart to :func:`named_credential_request_col`: it runs
|
|
541
|
-
**once** on the driver — not per row. Use the column helper method
|
|
542
|
-
instead when you want to fan a callout out across every row of a
|
|
543
|
-
DataFrame.
|
|
544
|
-
|
|
545
|
-
Example:
|
|
546
|
-
|
|
547
|
-
>>> from datacustomcode.named_credential.types.http_request_builder \\
|
|
548
|
-
... import HTTPRequestBuilder
|
|
549
|
-
>>> request = (
|
|
550
|
-
... HTTPRequestBuilder().set_url("callout:NC/search").build()
|
|
551
|
-
... )
|
|
552
|
-
>>> response = Client().named_credential_request(request)
|
|
553
|
-
|
|
554
|
-
Args:
|
|
555
|
-
request: The callout request
|
|
556
|
-
body: Optional request body. Set the ``Content-Type`` header to
|
|
557
|
-
match the format; the SDK does not assume or inject one.
|
|
558
|
-
|
|
559
|
-
Returns:
|
|
560
|
-
The external service's response.
|
|
561
|
-
"""
|
|
562
|
-
return self._get_spark_named_credential().request(request, body=body)
|
|
563
|
-
|
|
564
|
-
def _get_spark_named_credential(self) -> SparkNamedCredential:
|
|
565
|
-
if self._spark_named_credential is None:
|
|
566
|
-
self._spark_named_credential = _build_spark_named_credential()
|
|
567
|
-
return self._spark_named_credential
|
|
568
|
-
|
|
569
477
|
def _validate_data_layer_history_does_not_contain(
|
|
570
478
|
self, data_cloud_object_type: DataCloudObjectType
|
|
571
479
|
) -> None:
|
|
@@ -579,3 +487,170 @@ class Client:
|
|
|
579
487
|
|
|
580
488
|
def _record_dmo_access(self, name: str) -> None:
|
|
581
489
|
self._data_layer_history[DataCloudObjectType.DMO].add(name)
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
class Client(_BaseClient):
|
|
493
|
+
"""Entrypoint for batch access to Data Cloud objects.
|
|
494
|
+
|
|
495
|
+
This is the object used to read and write bounded snapshots of Data Cloud
|
|
496
|
+
DLOs and DMOs.
|
|
497
|
+
"""
|
|
498
|
+
|
|
499
|
+
_instance: ClassVar[Optional[Client]] = None
|
|
500
|
+
|
|
501
|
+
def read_dlo(self, name: str) -> PySparkDataFrame:
|
|
502
|
+
"""Read a DLO from Data Cloud.
|
|
503
|
+
|
|
504
|
+
Args:
|
|
505
|
+
name: The name of the DLO to read.
|
|
506
|
+
|
|
507
|
+
Returns:
|
|
508
|
+
A PySpark DataFrame containing the DLO data.
|
|
509
|
+
"""
|
|
510
|
+
self._record_dlo_access(name)
|
|
511
|
+
return self._reader.read_dlo(name) # type: ignore[no-any-return]
|
|
512
|
+
|
|
513
|
+
def read_dmo(self, name: str) -> PySparkDataFrame:
|
|
514
|
+
"""Read a DMO from Data Cloud.
|
|
515
|
+
|
|
516
|
+
Args:
|
|
517
|
+
name: The name of the DMO to read.
|
|
518
|
+
|
|
519
|
+
Returns:
|
|
520
|
+
A PySpark DataFrame containing the DMO data.
|
|
521
|
+
"""
|
|
522
|
+
self._record_dmo_access(name)
|
|
523
|
+
return self._reader.read_dmo(name) # type: ignore[no-any-return]
|
|
524
|
+
|
|
525
|
+
def write_to_dlo(
|
|
526
|
+
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
527
|
+
) -> None:
|
|
528
|
+
"""Write a PySpark DataFrame to a DLO in Data Cloud.
|
|
529
|
+
|
|
530
|
+
Args:
|
|
531
|
+
name: The name of the DLO to write to.
|
|
532
|
+
dataframe: The PySpark DataFrame to write.
|
|
533
|
+
write_mode: The write mode to use for writing to the DLO.
|
|
534
|
+
"""
|
|
535
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
536
|
+
return self._writer.write_to_dlo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
537
|
+
|
|
538
|
+
def write_to_dmo(
|
|
539
|
+
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
540
|
+
) -> None:
|
|
541
|
+
"""Write a PySpark DataFrame to a DMO in Data Cloud.
|
|
542
|
+
|
|
543
|
+
Args:
|
|
544
|
+
name: The name of the DMO to write to.
|
|
545
|
+
dataframe: The PySpark DataFrame to write.
|
|
546
|
+
write_mode: The write mode to use for writing to the DMO.
|
|
547
|
+
"""
|
|
548
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
|
|
549
|
+
return self._writer.write_to_dmo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
class StreamingClient(_BaseClient):
|
|
553
|
+
"""Entrypoint for streaming (``DELTA_SYNC``) access to Data Cloud objects.
|
|
554
|
+
|
|
555
|
+
This is the streaming counterpart to :class:`Client`. Instead of reading and
|
|
556
|
+
writing bounded snapshots, it reads a DLO/DMO change feed as a streaming
|
|
557
|
+
DataFrame and writes the transformed stream back via a ``StreamingQuery``.
|
|
558
|
+
"""
|
|
559
|
+
|
|
560
|
+
_instance: ClassVar[Optional[StreamingClient]] = None
|
|
561
|
+
|
|
562
|
+
def read_dlo(self) -> PySparkDataFrame:
|
|
563
|
+
"""Read the streamingSource
|
|
564
|
+
|
|
565
|
+
Returns:
|
|
566
|
+
A standard PySpark DataFrame from the streaming source DLO
|
|
567
|
+
"""
|
|
568
|
+
self._record_dlo_access(_streaming_source_name())
|
|
569
|
+
return self._reader.read_dlo(_streaming_source_name())
|
|
570
|
+
|
|
571
|
+
def read_dlo_deltas(self) -> PySparkDataFrame:
|
|
572
|
+
"""Read the streaming change feed (deltas) for a DLO from Data Cloud.
|
|
573
|
+
|
|
574
|
+
For use in a streaming (``DELTA_SYNC``) BYOC transform. Returns a
|
|
575
|
+
streaming DataFrame whose rows carry the change-feed metadata columns
|
|
576
|
+
(``_record_type``, ``_commit_*``) alongside the source columns.
|
|
577
|
+
|
|
578
|
+
Returns:
|
|
579
|
+
A streaming PySpark DataFrame over the DLO change feed.
|
|
580
|
+
"""
|
|
581
|
+
self._record_dlo_access(_streaming_source_name())
|
|
582
|
+
return self._reader.read_dlo_deltas() # type: ignore[no-any-return]
|
|
583
|
+
|
|
584
|
+
def read_dmo(self) -> PySparkDataFrame:
|
|
585
|
+
"""Read the streamingSource
|
|
586
|
+
|
|
587
|
+
Returns a standard PySpark DataFrame from the streaming source DMO
|
|
588
|
+
"""
|
|
589
|
+
self._record_dmo_access(_streaming_source_name())
|
|
590
|
+
return self._reader.read_dmo(_streaming_source_name())
|
|
591
|
+
|
|
592
|
+
def read_dmo_deltas(self) -> PySparkDataFrame:
|
|
593
|
+
"""Read the streaming change feed (deltas) for a DMO from Data Cloud.
|
|
594
|
+
|
|
595
|
+
Returns:
|
|
596
|
+
A streaming PySpark DataFrame over the DMO change feed.
|
|
597
|
+
"""
|
|
598
|
+
self._record_dmo_access(_streaming_source_name())
|
|
599
|
+
return self._reader.read_dmo_deltas() # type: ignore[no-any-return]
|
|
600
|
+
|
|
601
|
+
def write_dlo_deltas(
|
|
602
|
+
self, name: str, dataframe: PySparkDataFrame, **kwargs
|
|
603
|
+
) -> StreamingQuery:
|
|
604
|
+
"""Write a streaming DataFrame of deltas to a DLO in Data Cloud.
|
|
605
|
+
|
|
606
|
+
Starts a streaming query that writes each micro-batch to the
|
|
607
|
+
target DLO and returns the ``StreamingQuery`` handle; the caller
|
|
608
|
+
typically calls ``query.awaitTermination()``.
|
|
609
|
+
|
|
610
|
+
Args:
|
|
611
|
+
name: The name of the DLO to write to.
|
|
612
|
+
dataframe: The streaming PySpark DataFrame to write.
|
|
613
|
+
|
|
614
|
+
Returns:
|
|
615
|
+
The started ``StreamingQuery``.
|
|
616
|
+
"""
|
|
617
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
618
|
+
return self._writer.write_dlo_deltas(name, dataframe, **kwargs) # type: ignore[no-any-return]
|
|
619
|
+
|
|
620
|
+
def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
|
|
621
|
+
"""Write a PySpark DataFrame to a DLO in Data Cloud automatically picking
|
|
622
|
+
the WriteMode.
|
|
623
|
+
For use with streaming transforms when running in rebuild or initial sync mode.
|
|
624
|
+
Args:
|
|
625
|
+
name: The name of the DLO to write to.
|
|
626
|
+
dataframe: The PySpark DataFrame to write.
|
|
627
|
+
"""
|
|
628
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
629
|
+
return self._writer.auto_write_to_dlo(name, dataframe)
|
|
630
|
+
|
|
631
|
+
def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
|
|
632
|
+
"""Write a PySpark DataFrame to a DMO in Data Cloud automatically picking
|
|
633
|
+
the WriteMode.
|
|
634
|
+
For use with streaming transforms when running in rebuild or initial sync mode.
|
|
635
|
+
Args:
|
|
636
|
+
name: The name of the DMO to write to.
|
|
637
|
+
dataframe: The PySpark DataFrame to write.
|
|
638
|
+
"""
|
|
639
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
|
|
640
|
+
return self._writer.auto_write_to_dmo(name, dataframe)
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
class RunMode(Enum):
|
|
644
|
+
BATCH = "BATCH"
|
|
645
|
+
INITIAL_SYNC = "INITIAL_SYNC"
|
|
646
|
+
REBUILD = "REBUILD"
|
|
647
|
+
DELTA_SYNC = "DELTA_SYNC"
|
|
648
|
+
|
|
649
|
+
|
|
650
|
+
def get_run_mode() -> RunMode:
|
|
651
|
+
"""Read and validate the BYOC_RUN_MODE env var; default to BATCH when unset."""
|
|
652
|
+
run_mode = os.getenv("BYOC_RUN_MODE", "BATCH").upper()
|
|
653
|
+
try:
|
|
654
|
+
return RunMode(run_mode)
|
|
655
|
+
except ValueError as exc:
|
|
656
|
+
raise ValueError("Set BYOC_RUN_MODE to a valid value") from exc
|
datacustomcode/config.py
CHANGED
|
@@ -89,6 +89,9 @@ class ClientConfig(BaseConfig):
|
|
|
89
89
|
spark_provider_config: Union[
|
|
90
90
|
SparkProviderConfig[BaseSparkSessionProvider], None
|
|
91
91
|
] = None
|
|
92
|
+
# Source object name for a streaming (DELTA_SYNC) transform, populated by
|
|
93
|
+
# ``run_entrypoint`` from config.json's ``streamingSource`` field
|
|
94
|
+
streaming_source: Union[str, None] = None
|
|
92
95
|
|
|
93
96
|
def update(self, other: ClientConfig) -> ClientConfig:
|
|
94
97
|
"""Merge this ClientConfig with another, respecting force flags.
|
|
@@ -116,6 +119,8 @@ class ClientConfig(BaseConfig):
|
|
|
116
119
|
self.spark_provider_config = merge(
|
|
117
120
|
self.spark_provider_config, other.spark_provider_config
|
|
118
121
|
)
|
|
122
|
+
if other.streaming_source is not None:
|
|
123
|
+
self.streaming_source = other.streaming_source
|
|
119
124
|
return self
|
|
120
125
|
|
|
121
126
|
|