salesforce-data-customcode 6.1.0.dev1__py3-none-any.whl → 6.1.0.dev3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datacustomcode/__init__.py +0 -5
- datacustomcode/cli.py +5 -30
- datacustomcode/client.py +192 -211
- datacustomcode/config.py +0 -5
- datacustomcode/config.yaml +6 -0
- datacustomcode/constants.py +1 -9
- datacustomcode/deploy.py +24 -58
- datacustomcode/function/runtime.py +16 -0
- datacustomcode/io/reader/base.py +0 -42
- datacustomcode/io/writer/base.py +0 -33
- datacustomcode/named_credential/__init__.py +28 -0
- datacustomcode/named_credential/base.py +54 -0
- datacustomcode/named_credential/default.py +93 -0
- datacustomcode/named_credential/direct/__init__.py +19 -0
- datacustomcode/named_credential/direct/auth.py +63 -0
- datacustomcode/named_credential/direct/credentials.py +121 -0
- datacustomcode/named_credential/direct/transport.py +110 -0
- datacustomcode/named_credential/direct/url_resolver.py +112 -0
- datacustomcode/named_credential/errors.py +36 -0
- datacustomcode/named_credential/spark_base.py +93 -0
- datacustomcode/named_credential/spark_default.py +154 -0
- datacustomcode/named_credential/types/__init__.py +14 -0
- datacustomcode/named_credential/types/http_method.py +29 -0
- datacustomcode/named_credential/types/http_request.py +63 -0
- datacustomcode/named_credential/types/http_request_builder.py +55 -0
- datacustomcode/named_credential/types/http_response.py +43 -0
- datacustomcode/named_credential/types/http_response_builder.py +24 -0
- datacustomcode/named_credential_config.py +105 -0
- datacustomcode/run.py +7 -11
- datacustomcode/scan.py +29 -164
- datacustomcode/template.py +1 -13
- datacustomcode/templates/function/example/chunking_with_external_callout/README.md +119 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/config.json +3 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +161 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +11 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +16 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev3.dist-info}/METADATA +3 -40
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev3.dist-info}/RECORD +41 -19
- datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +0 -49
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev3.dist-info}/WHEEL +0 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev3.dist-info}/entry_points.txt +0 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev3.dist-info}/licenses/LICENSE.txt +0 -0
datacustomcode/__init__.py
CHANGED
|
@@ -23,7 +23,6 @@ __all__ = [
|
|
|
23
23
|
"QueryAPIDataCloudReader",
|
|
24
24
|
"SparkEinsteinPredictions",
|
|
25
25
|
"SparkLLMGateway",
|
|
26
|
-
"StreamingClient",
|
|
27
26
|
"einstein_predict_col",
|
|
28
27
|
"llm_gateway_generate_text_col",
|
|
29
28
|
]
|
|
@@ -35,10 +34,6 @@ def __getattr__(name: str):
|
|
|
35
34
|
from datacustomcode.client import Client
|
|
36
35
|
|
|
37
36
|
return Client
|
|
38
|
-
elif name == "StreamingClient":
|
|
39
|
-
from datacustomcode.client import StreamingClient
|
|
40
|
-
|
|
41
|
-
return StreamingClient
|
|
42
37
|
elif name == "AuthType":
|
|
43
38
|
from datacustomcode.credentials import AuthType
|
|
44
39
|
|
datacustomcode/cli.py
CHANGED
|
@@ -262,7 +262,7 @@ def deploy(
|
|
|
262
262
|
mapped_feature = USE_IN_FEATURE_MAPPING_FOR_CONNECT_API.get(
|
|
263
263
|
use_in_feature, use_in_feature
|
|
264
264
|
)
|
|
265
|
-
metadata.
|
|
265
|
+
metadata.invokeOptions = [mapped_feature]
|
|
266
266
|
|
|
267
267
|
try:
|
|
268
268
|
if sf_cli_org:
|
|
@@ -283,19 +283,10 @@ def deploy(
|
|
|
283
283
|
)
|
|
284
284
|
@click.option(
|
|
285
285
|
"--use-in-feature",
|
|
286
|
-
"
|
|
287
|
-
|
|
288
|
-
help=(
|
|
289
|
-
"Invoke option for this package. For scripts: 'BatchTransform' "
|
|
290
|
-
"(default) or 'StreamingTransform'. For functions: 'SearchIndexChunking'."
|
|
291
|
-
),
|
|
286
|
+
default="SearchIndexChunking",
|
|
287
|
+
help="Feature where this function will be used (only applicable for function).",
|
|
292
288
|
)
|
|
293
289
|
def init(directory: str, code_type: str, use_in_feature: Optional[str]):
|
|
294
|
-
from datacustomcode.constants import (
|
|
295
|
-
SCRIPT_USE_IN_FEATURE_BATCH,
|
|
296
|
-
SCRIPT_USE_IN_FEATURE_OPTIONS,
|
|
297
|
-
SCRIPT_USE_IN_FEATURE_STREAMING,
|
|
298
|
-
)
|
|
299
290
|
from datacustomcode.scan import (
|
|
300
291
|
dc_config_json_from_file,
|
|
301
292
|
update_config,
|
|
@@ -303,23 +294,9 @@ def init(directory: str, code_type: str, use_in_feature: Optional[str]):
|
|
|
303
294
|
)
|
|
304
295
|
from datacustomcode.template import copy_function_template, copy_script_template
|
|
305
296
|
|
|
306
|
-
streaming = False
|
|
307
|
-
if code_type == "script":
|
|
308
|
-
use_in_feature = use_in_feature or SCRIPT_USE_IN_FEATURE_BATCH
|
|
309
|
-
if use_in_feature not in SCRIPT_USE_IN_FEATURE_OPTIONS:
|
|
310
|
-
click.secho(
|
|
311
|
-
f"Error: Invalid --use-in-feature '{use_in_feature}' for a "
|
|
312
|
-
f"script. Valid options: {', '.join(SCRIPT_USE_IN_FEATURE_OPTIONS)}.",
|
|
313
|
-
fg="red",
|
|
314
|
-
)
|
|
315
|
-
raise click.Abort()
|
|
316
|
-
streaming = use_in_feature == SCRIPT_USE_IN_FEATURE_STREAMING
|
|
317
|
-
else:
|
|
318
|
-
use_in_feature = use_in_feature or "SearchIndexChunking"
|
|
319
|
-
|
|
320
297
|
click.echo("Copying template to " + click.style(directory, fg="blue", bold=True))
|
|
321
298
|
if code_type == "script":
|
|
322
|
-
copy_script_template(directory
|
|
299
|
+
copy_script_template(directory)
|
|
323
300
|
elif code_type == "function":
|
|
324
301
|
copy_function_template(directory, use_in_feature)
|
|
325
302
|
entrypoint_path = os.path.join(directory, PAYLOAD_DIR, ENTRYPOINT_FILE)
|
|
@@ -329,9 +306,7 @@ def init(directory: str, code_type: str, use_in_feature: Optional[str]):
|
|
|
329
306
|
sdk_config = {"type": code_type}
|
|
330
307
|
write_sdk_config(directory, sdk_config)
|
|
331
308
|
|
|
332
|
-
config_json = dc_config_json_from_file(
|
|
333
|
-
entrypoint_path, code_type, streaming=streaming
|
|
334
|
-
)
|
|
309
|
+
config_json = dc_config_json_from_file(entrypoint_path, code_type)
|
|
335
310
|
with open(config_location, "w") as f:
|
|
336
311
|
json.dump(config_json, f, indent=2)
|
|
337
312
|
|
datacustomcode/client.py
CHANGED
|
@@ -21,9 +21,7 @@ from typing import (
|
|
|
21
21
|
ClassVar,
|
|
22
22
|
Dict,
|
|
23
23
|
Optional,
|
|
24
|
-
TypeVar,
|
|
25
24
|
Union,
|
|
26
|
-
cast,
|
|
27
25
|
)
|
|
28
26
|
|
|
29
27
|
from datacustomcode.config import config
|
|
@@ -31,60 +29,25 @@ from datacustomcode.einstein_predictions_config import spark_einstein_prediction
|
|
|
31
29
|
from datacustomcode.file.path.default import DefaultFindFilePath
|
|
32
30
|
from datacustomcode.io.reader.base import BaseDataCloudReader
|
|
33
31
|
from datacustomcode.llm_gateway_config import spark_llm_gateway_config
|
|
32
|
+
from datacustomcode.named_credential_config import spark_named_credential_config
|
|
34
33
|
from datacustomcode.spark.default import DefaultSparkSessionProvider
|
|
35
34
|
|
|
36
35
|
if TYPE_CHECKING:
|
|
37
36
|
from pathlib import Path
|
|
38
37
|
|
|
39
|
-
from pyspark.sql import
|
|
40
|
-
Column,
|
|
41
|
-
DataFrame as PySparkDataFrame,
|
|
42
|
-
SparkSession,
|
|
43
|
-
)
|
|
44
|
-
from pyspark.sql.streaming import StreamingQuery
|
|
38
|
+
from pyspark.sql import Column, DataFrame as PySparkDataFrame
|
|
45
39
|
|
|
46
40
|
from datacustomcode.einstein_predictions.spark_base import SparkEinsteinPredictions
|
|
47
41
|
from datacustomcode.einstein_predictions.types import PredictionType
|
|
48
42
|
from datacustomcode.io.reader.base import BaseDataCloudReader
|
|
49
43
|
from datacustomcode.io.writer.base import BaseDataCloudWriter, WriteMode
|
|
50
44
|
from datacustomcode.llm_gateway.spark_base import SparkLLMGateway
|
|
45
|
+
from datacustomcode.named_credential.spark_base import SparkNamedCredential
|
|
46
|
+
from datacustomcode.named_credential.types.http_request import HTTPRequest
|
|
47
|
+
from datacustomcode.named_credential.types.http_response import HTTPResponse
|
|
51
48
|
from datacustomcode.spark.base import BaseSparkSessionProvider
|
|
52
49
|
|
|
53
50
|
|
|
54
|
-
def _streaming_source_name() -> str:
|
|
55
|
-
"""Return the streaming transform's read-source name.
|
|
56
|
-
|
|
57
|
-
Resolved from ``config.streaming_source``, which ``run_entrypoint``
|
|
58
|
-
populates from config.json's ``streamingSource`` field.
|
|
59
|
-
|
|
60
|
-
Raises:
|
|
61
|
-
RuntimeError: If no ``streaming_source`` has been configured (e.g. the
|
|
62
|
-
transform's config.json has no ``streamingSource`` field).
|
|
63
|
-
"""
|
|
64
|
-
source = config.streaming_source
|
|
65
|
-
if not source:
|
|
66
|
-
raise RuntimeError(
|
|
67
|
-
"No streaming source configured. A streaming transform must declare "
|
|
68
|
-
"its read source in config.json under 'streamingSource'."
|
|
69
|
-
)
|
|
70
|
-
return source
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
def _active_client() -> "_BaseClient":
|
|
74
|
-
"""Return the client backing the module-level Spark column helpers.
|
|
75
|
-
|
|
76
|
-
Prefers an already-initialized singleton so a streaming job reuses its
|
|
77
|
-
:class:`StreamingClient` (and a batch job its :class:`Client`) rather than
|
|
78
|
-
forcing an unrelated client into existence. Falls back to building the
|
|
79
|
-
batch :class:`Client` when neither has been created yet.
|
|
80
|
-
"""
|
|
81
|
-
if Client._instance is not None:
|
|
82
|
-
return Client._instance
|
|
83
|
-
if StreamingClient._instance is not None:
|
|
84
|
-
return StreamingClient._instance
|
|
85
|
-
return Client()
|
|
86
|
-
|
|
87
|
-
|
|
88
51
|
def _build_spark_llm_gateway() -> "SparkLLMGateway":
|
|
89
52
|
"""Instantiate the SDK-configured :class:`SparkLLMGateway`.
|
|
90
53
|
|
|
@@ -140,7 +103,7 @@ def llm_gateway_generate_text_col(
|
|
|
140
103
|
the generated text; on failure, ``status == "ERROR"`` and the
|
|
141
104
|
``error_*`` fields carry diagnostic detail.
|
|
142
105
|
"""
|
|
143
|
-
gateway =
|
|
106
|
+
gateway = Client()._get_spark_llm_gateway()
|
|
144
107
|
return gateway.llm_gateway_generate_text_col(template, values, model_id=model_id)
|
|
145
108
|
|
|
146
109
|
|
|
@@ -159,6 +122,21 @@ def _build_spark_einstein_predictions() -> "SparkEinsteinPredictions":
|
|
|
159
122
|
return cfg.to_object()
|
|
160
123
|
|
|
161
124
|
|
|
125
|
+
def _build_spark_named_credential() -> "SparkNamedCredential":
|
|
126
|
+
"""Instantiate the SDK-configured :class:`SparkNamedCredential`.
|
|
127
|
+
|
|
128
|
+
Raises:
|
|
129
|
+
RuntimeError: If no ``spark_named_credential_config`` has been loaded.
|
|
130
|
+
"""
|
|
131
|
+
cfg = spark_named_credential_config.spark_named_credential_config
|
|
132
|
+
if cfg is None:
|
|
133
|
+
raise RuntimeError(
|
|
134
|
+
"spark_named_credential_config is not configured. Add a "
|
|
135
|
+
"'spark_named_credential_config' section to config.yaml."
|
|
136
|
+
)
|
|
137
|
+
return cfg.to_object()
|
|
138
|
+
|
|
139
|
+
|
|
162
140
|
def einstein_predict_col(
|
|
163
141
|
model_api_name: str,
|
|
164
142
|
prediction_type: "PredictionType",
|
|
@@ -202,12 +180,46 @@ def einstein_predict_col(
|
|
|
202
180
|
the JSON-serialized prediction payload; on failure, ``status ==
|
|
203
181
|
"ERROR"`` and the ``error_*`` fields carry diagnostic detail.
|
|
204
182
|
"""
|
|
205
|
-
predictions =
|
|
183
|
+
predictions = Client()._get_spark_einstein_predictions()
|
|
206
184
|
return predictions.einstein_predict_col(
|
|
207
185
|
model_api_name, prediction_type, features, settings=settings
|
|
208
186
|
)
|
|
209
187
|
|
|
210
188
|
|
|
189
|
+
def named_credential_request_col(
|
|
190
|
+
request: "HTTPRequest",
|
|
191
|
+
body: Optional["Column"] = None,
|
|
192
|
+
) -> "Column":
|
|
193
|
+
"""Build a Spark Column that makes one Named Credential callout per row.
|
|
194
|
+
|
|
195
|
+
The endpoint, method, and headers are fixed for the call (taken from
|
|
196
|
+
``request``); only ``body`` varies per row. Use this instead of
|
|
197
|
+
:meth:`Client.named_credential_request` when the callout runs across a
|
|
198
|
+
DataFrame so each row is dispatched independently rather than one-shot on
|
|
199
|
+
the driver.
|
|
200
|
+
|
|
201
|
+
The returned Column yields a struct ``{status, response, error_code,
|
|
202
|
+
error_message}`` for each row. ``response`` is itself a struct
|
|
203
|
+
``{status_code, body, headers}``. Use ``[...]`` to pick a field, e.g.
|
|
204
|
+
``named_credential_request_col(...)["response"]["status_code"]``. A transport
|
|
205
|
+
failure sets ``status`` to ``ERROR`` and populates ``error_message`` (a non-2xx
|
|
206
|
+
HTTP response is still ``SUCCESS`` with its code in ``response.status_code``),
|
|
207
|
+
so a single bad row does not abort the whole Spark job.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
request: The callout template — its symbolic reference, method, and
|
|
211
|
+
headers are applied to every row.
|
|
212
|
+
body: Optional per-row ``Column`` holding the request body as a
|
|
213
|
+
string (or null for no body).
|
|
214
|
+
|
|
215
|
+
Returns:
|
|
216
|
+
A Spark ``Column`` of ``StructType`` with fields ``status``,
|
|
217
|
+
``response``, ``error_code``, and ``error_message``.
|
|
218
|
+
"""
|
|
219
|
+
named_credential = Client()._get_spark_named_credential()
|
|
220
|
+
return named_credential.request_col(request, body=body)
|
|
221
|
+
|
|
222
|
+
|
|
211
223
|
class DataCloudObjectType(Enum):
|
|
212
224
|
DLO = "dlo"
|
|
213
225
|
DMO = "dmo"
|
|
@@ -246,86 +258,83 @@ class DataCloudAccessLayerException(Exception):
|
|
|
246
258
|
return msg
|
|
247
259
|
|
|
248
260
|
|
|
249
|
-
|
|
261
|
+
class Client:
|
|
262
|
+
"""Entrypoint for accessing DataCloud objects.
|
|
250
263
|
|
|
264
|
+
This is the object used to access Data Cloud DLOs and DMOs. Accessing DLOs/DMOs
|
|
265
|
+
are tracked and will throw an exception if they are mixed. In other words, you
|
|
266
|
+
can read from DLOs and write to DLOs, read from DMOs and write to DMOs, but you
|
|
267
|
+
cannot read from DLOs and write to DMOs or read from DMOs and write to DLOs.
|
|
268
|
+
Furthermore you cannot mix during merging tables. This class is a singleton to
|
|
269
|
+
prevent accidental mixing of DLOs and DMOs.
|
|
251
270
|
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
This base class is not meant to be instantiated directly; use
|
|
259
|
-
:class:`Client` or :class:`StreamingClient`.
|
|
271
|
+
You can provide custom readers and writers to the client for advanced use
|
|
272
|
+
cases, but this is not recommended for testing as they may result in unexpected
|
|
273
|
+
behavior once deployed to Data Cloud. By default, the client intercepts all
|
|
274
|
+
read/write operations and mocks access to Data Cloud. For example, during
|
|
275
|
+
writing, we print to the console instead of writing to Data Cloud.
|
|
260
276
|
|
|
261
277
|
Args:
|
|
278
|
+
finder: Find a file path
|
|
262
279
|
reader: A custom reader to use for reading Data Cloud objects.
|
|
263
280
|
writer: A custom writer to use for writing Data Cloud objects.
|
|
264
|
-
spark_provider: Optional custom :class:`BaseSparkSessionProvider`.
|
|
265
281
|
spark_llm_gateway: Optional custom :class:`SparkLLMGateway`.
|
|
266
282
|
spark_einstein_predictions: Optional custom
|
|
267
283
|
:class:`SparkEinsteinPredictions`.
|
|
284
|
+
spark_named_credential: Optional custom :class:`SparkNamedCredential`.
|
|
285
|
+
|
|
286
|
+
Example:
|
|
287
|
+
>>> client = Client()
|
|
288
|
+
>>> file_path = client.find_file_path("data.csv")
|
|
289
|
+
>>> dlo = client.read_dlo("my_dlo")
|
|
290
|
+
>>> client.write_to_dmo("my_dmo", dlo)
|
|
291
|
+
>>> answer = client.llm_gateway_generate_text("Generate a greeting message")
|
|
268
292
|
"""
|
|
269
293
|
|
|
270
|
-
|
|
271
|
-
# to this base default of ``None``, but ``cls._instance = ...`` in __new__
|
|
272
|
-
# always writes to the subclass, so ``Client`` and ``StreamingClient`` never
|
|
273
|
-
# share an instance.
|
|
274
|
-
_instance: ClassVar[Optional[_BaseClient]] = None
|
|
275
|
-
# Process-wide Spark session shared across BOTH client types. Unlike
|
|
276
|
-
# ``_instance``, this is written via ``_BaseClient._shared_spark`` (never
|
|
277
|
-
# ``cls._shared_spark``), so the slot lives on the base class and a
|
|
278
|
-
# ``Client`` and a ``StreamingClient`` in the same process reuse one session
|
|
279
|
-
# — and therefore one underlying connection — instead of opening two
|
|
280
|
-
# containing differing state
|
|
281
|
-
_shared_spark: ClassVar[Optional[SparkSession]] = None
|
|
294
|
+
_instance: ClassVar[Optional[Client]] = None
|
|
282
295
|
_reader: BaseDataCloudReader
|
|
283
296
|
_writer: BaseDataCloudWriter
|
|
284
297
|
_file: DefaultFindFilePath
|
|
285
298
|
_spark_llm_gateway: Optional[SparkLLMGateway]
|
|
286
299
|
_spark_einstein_predictions: Optional[SparkEinsteinPredictions]
|
|
300
|
+
_spark_named_credential: Optional[SparkNamedCredential]
|
|
287
301
|
_data_layer_history: dict[DataCloudObjectType, set[str]]
|
|
288
302
|
_code_type: str
|
|
289
303
|
|
|
290
304
|
def __new__(
|
|
291
|
-
cls
|
|
305
|
+
cls,
|
|
292
306
|
reader: Optional[BaseDataCloudReader] = None,
|
|
293
307
|
writer: Optional[BaseDataCloudWriter] = None,
|
|
294
308
|
spark_provider: Optional[BaseSparkSessionProvider] = None,
|
|
295
309
|
spark_llm_gateway: Optional[SparkLLMGateway] = None,
|
|
296
310
|
spark_einstein_predictions: Optional[SparkEinsteinPredictions] = None,
|
|
311
|
+
spark_named_credential: Optional[SparkNamedCredential] = None,
|
|
297
312
|
code_type: str = "script",
|
|
298
|
-
) ->
|
|
313
|
+
) -> Client:
|
|
299
314
|
|
|
300
315
|
if cls._instance is None:
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
316
|
+
cls._instance = super().__new__(cls)
|
|
317
|
+
cls._instance._spark_llm_gateway = spark_llm_gateway
|
|
318
|
+
cls._instance._spark_einstein_predictions = spark_einstein_predictions
|
|
319
|
+
cls._instance._spark_named_credential = spark_named_credential
|
|
304
320
|
# Initialize Readers and Writers from config
|
|
305
321
|
# and/or provided reader and writer
|
|
306
322
|
if reader is None or writer is None:
|
|
307
|
-
# We need a spark because we will initialize readers and writers
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
323
|
+
# We need a spark because we will initialize readers and writers
|
|
324
|
+
if config.spark_config is None:
|
|
325
|
+
raise ValueError(
|
|
326
|
+
"Spark config is required when reader/writer is not provided"
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
provider: BaseSparkSessionProvider
|
|
330
|
+
if spark_provider is not None:
|
|
331
|
+
provider = spark_provider
|
|
332
|
+
elif config.spark_provider_config is not None:
|
|
333
|
+
provider = config.spark_provider_config.to_object()
|
|
312
334
|
else:
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
"provided"
|
|
317
|
-
)
|
|
318
|
-
|
|
319
|
-
provider: BaseSparkSessionProvider
|
|
320
|
-
if spark_provider is not None:
|
|
321
|
-
provider = spark_provider
|
|
322
|
-
elif config.spark_provider_config is not None:
|
|
323
|
-
provider = config.spark_provider_config.to_object()
|
|
324
|
-
else:
|
|
325
|
-
provider = DefaultSparkSessionProvider()
|
|
326
|
-
|
|
327
|
-
spark = provider.get_session(config.spark_config)
|
|
328
|
-
_BaseClient._shared_spark = spark
|
|
335
|
+
provider = DefaultSparkSessionProvider()
|
|
336
|
+
|
|
337
|
+
spark = provider.get_session(config.spark_config)
|
|
329
338
|
|
|
330
339
|
if config.reader_config is None and reader is None:
|
|
331
340
|
raise ValueError(
|
|
@@ -348,17 +357,66 @@ class _BaseClient:
|
|
|
348
357
|
else:
|
|
349
358
|
writer_init = writer
|
|
350
359
|
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
360
|
+
cls._instance._reader = reader_init
|
|
361
|
+
cls._instance._writer = writer_init
|
|
362
|
+
cls._instance._file = DefaultFindFilePath()
|
|
363
|
+
cls._instance._data_layer_history = {
|
|
355
364
|
DataCloudObjectType.DLO: set(),
|
|
356
365
|
DataCloudObjectType.DMO: set(),
|
|
357
366
|
}
|
|
358
|
-
|
|
359
|
-
elif reader is not None or writer is not None:
|
|
367
|
+
elif (reader is not None or writer is not None) and cls._instance is not None:
|
|
360
368
|
raise ValueError("Cannot set reader or writer after client is initialized")
|
|
361
|
-
return
|
|
369
|
+
return cls._instance
|
|
370
|
+
|
|
371
|
+
def read_dlo(self, name: str) -> PySparkDataFrame:
|
|
372
|
+
"""Read a DLO from Data Cloud.
|
|
373
|
+
|
|
374
|
+
Args:
|
|
375
|
+
name: The name of the DLO to read.
|
|
376
|
+
|
|
377
|
+
Returns:
|
|
378
|
+
A PySpark DataFrame containing the DLO data.
|
|
379
|
+
"""
|
|
380
|
+
self._record_dlo_access(name)
|
|
381
|
+
return self._reader.read_dlo(name) # type: ignore[no-any-return]
|
|
382
|
+
|
|
383
|
+
def read_dmo(self, name: str) -> PySparkDataFrame:
|
|
384
|
+
"""Read a DMO from Data Cloud.
|
|
385
|
+
|
|
386
|
+
Args:
|
|
387
|
+
name: The name of the DMO to read.
|
|
388
|
+
|
|
389
|
+
Returns:
|
|
390
|
+
A PySpark DataFrame containing the DMO data.
|
|
391
|
+
"""
|
|
392
|
+
self._record_dmo_access(name)
|
|
393
|
+
return self._reader.read_dmo(name) # type: ignore[no-any-return]
|
|
394
|
+
|
|
395
|
+
def write_to_dlo(
|
|
396
|
+
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
397
|
+
) -> None:
|
|
398
|
+
"""Write a PySpark DataFrame to a DLO in Data Cloud.
|
|
399
|
+
|
|
400
|
+
Args:
|
|
401
|
+
name: The name of the DLO to write to.
|
|
402
|
+
dataframe: The PySpark DataFrame to write.
|
|
403
|
+
write_mode: The write mode to use for writing to the DLO.
|
|
404
|
+
"""
|
|
405
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
406
|
+
return self._writer.write_to_dlo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
407
|
+
|
|
408
|
+
def write_to_dmo(
|
|
409
|
+
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
410
|
+
) -> None:
|
|
411
|
+
"""Write a PySpark DataFrame to a DMO in Data Cloud.
|
|
412
|
+
|
|
413
|
+
Args:
|
|
414
|
+
name: The name of the DMO to write to.
|
|
415
|
+
dataframe: The PySpark DataFrame to write.
|
|
416
|
+
write_mode: The write mode to use for writing to the DMO.
|
|
417
|
+
"""
|
|
418
|
+
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
|
|
419
|
+
return self._writer.write_to_dmo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
362
420
|
|
|
363
421
|
def find_file_path(self, file_name: str) -> Path:
|
|
364
422
|
"""Resolve a bundled file shipped in the package to an absolute path.
|
|
@@ -473,6 +531,41 @@ class _BaseClient:
|
|
|
473
531
|
self._spark_einstein_predictions = _build_spark_einstein_predictions()
|
|
474
532
|
return self._spark_einstein_predictions
|
|
475
533
|
|
|
534
|
+
def named_credential_request(
|
|
535
|
+
self,
|
|
536
|
+
request: "HTTPRequest",
|
|
537
|
+
body: Optional[str] = None,
|
|
538
|
+
) -> "HTTPResponse":
|
|
539
|
+
"""Issue a one-shot Named Credential external callout. This is the
|
|
540
|
+
scalar counterpart to :func:`named_credential_request_col`: it runs
|
|
541
|
+
**once** on the driver — not per row. Use the column helper method
|
|
542
|
+
instead when you want to fan a callout out across every row of a
|
|
543
|
+
DataFrame.
|
|
544
|
+
|
|
545
|
+
Example:
|
|
546
|
+
|
|
547
|
+
>>> from datacustomcode.named_credential.types.http_request_builder \\
|
|
548
|
+
... import HTTPRequestBuilder
|
|
549
|
+
>>> request = (
|
|
550
|
+
... HTTPRequestBuilder().set_url("callout:NC/search").build()
|
|
551
|
+
... )
|
|
552
|
+
>>> response = Client().named_credential_request(request)
|
|
553
|
+
|
|
554
|
+
Args:
|
|
555
|
+
request: The callout request
|
|
556
|
+
body: Optional request body. Set the ``Content-Type`` header to
|
|
557
|
+
match the format; the SDK does not assume or inject one.
|
|
558
|
+
|
|
559
|
+
Returns:
|
|
560
|
+
The external service's response.
|
|
561
|
+
"""
|
|
562
|
+
return self._get_spark_named_credential().request(request, body=body)
|
|
563
|
+
|
|
564
|
+
def _get_spark_named_credential(self) -> SparkNamedCredential:
|
|
565
|
+
if self._spark_named_credential is None:
|
|
566
|
+
self._spark_named_credential = _build_spark_named_credential()
|
|
567
|
+
return self._spark_named_credential
|
|
568
|
+
|
|
476
569
|
def _validate_data_layer_history_does_not_contain(
|
|
477
570
|
self, data_cloud_object_type: DataCloudObjectType
|
|
478
571
|
) -> None:
|
|
@@ -486,115 +579,3 @@ class _BaseClient:
|
|
|
486
579
|
|
|
487
580
|
def _record_dmo_access(self, name: str) -> None:
|
|
488
581
|
self._data_layer_history[DataCloudObjectType.DMO].add(name)
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
class Client(_BaseClient):
|
|
492
|
-
"""Entrypoint for batch access to Data Cloud objects.
|
|
493
|
-
|
|
494
|
-
This is the object used to read and write bounded snapshots of Data Cloud
|
|
495
|
-
DLOs and DMOs.
|
|
496
|
-
"""
|
|
497
|
-
|
|
498
|
-
_instance: ClassVar[Optional[Client]] = None
|
|
499
|
-
|
|
500
|
-
def read_dlo(self, name: str) -> PySparkDataFrame:
|
|
501
|
-
"""Read a DLO from Data Cloud.
|
|
502
|
-
|
|
503
|
-
Args:
|
|
504
|
-
name: The name of the DLO to read.
|
|
505
|
-
|
|
506
|
-
Returns:
|
|
507
|
-
A PySpark DataFrame containing the DLO data.
|
|
508
|
-
"""
|
|
509
|
-
self._record_dlo_access(name)
|
|
510
|
-
return self._reader.read_dlo(name) # type: ignore[no-any-return]
|
|
511
|
-
|
|
512
|
-
def read_dmo(self, name: str) -> PySparkDataFrame:
|
|
513
|
-
"""Read a DMO from Data Cloud.
|
|
514
|
-
|
|
515
|
-
Args:
|
|
516
|
-
name: The name of the DMO to read.
|
|
517
|
-
|
|
518
|
-
Returns:
|
|
519
|
-
A PySpark DataFrame containing the DMO data.
|
|
520
|
-
"""
|
|
521
|
-
self._record_dmo_access(name)
|
|
522
|
-
return self._reader.read_dmo(name) # type: ignore[no-any-return]
|
|
523
|
-
|
|
524
|
-
def write_to_dlo(
|
|
525
|
-
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
526
|
-
) -> None:
|
|
527
|
-
"""Write a PySpark DataFrame to a DLO in Data Cloud.
|
|
528
|
-
|
|
529
|
-
Args:
|
|
530
|
-
name: The name of the DLO to write to.
|
|
531
|
-
dataframe: The PySpark DataFrame to write.
|
|
532
|
-
write_mode: The write mode to use for writing to the DLO.
|
|
533
|
-
"""
|
|
534
|
-
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
535
|
-
return self._writer.write_to_dlo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
536
|
-
|
|
537
|
-
def write_to_dmo(
|
|
538
|
-
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
|
|
539
|
-
) -> None:
|
|
540
|
-
"""Write a PySpark DataFrame to a DMO in Data Cloud.
|
|
541
|
-
|
|
542
|
-
Args:
|
|
543
|
-
name: The name of the DMO to write to.
|
|
544
|
-
dataframe: The PySpark DataFrame to write.
|
|
545
|
-
write_mode: The write mode to use for writing to the DMO.
|
|
546
|
-
"""
|
|
547
|
-
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
|
|
548
|
-
return self._writer.write_to_dmo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
class StreamingClient(_BaseClient):
|
|
552
|
-
"""Entrypoint for streaming (``DELTA_SYNC``) access to Data Cloud objects.
|
|
553
|
-
|
|
554
|
-
This is the streaming counterpart to :class:`Client`. Instead of reading and
|
|
555
|
-
writing bounded snapshots, it reads a DLO/DMO change feed as a streaming
|
|
556
|
-
DataFrame and writes the transformed stream back via a ``StreamingQuery``.
|
|
557
|
-
"""
|
|
558
|
-
|
|
559
|
-
_instance: ClassVar[Optional[StreamingClient]] = None
|
|
560
|
-
|
|
561
|
-
def read_dlo_deltas(self) -> PySparkDataFrame:
|
|
562
|
-
"""Read the streaming change feed (deltas) for a DLO from Data Cloud.
|
|
563
|
-
|
|
564
|
-
For use in a streaming (``DELTA_SYNC``) BYOC transform. Returns a
|
|
565
|
-
streaming DataFrame whose rows carry the change-feed metadata columns
|
|
566
|
-
(``_record_type``, ``_commit_*``) alongside the source columns.
|
|
567
|
-
|
|
568
|
-
Returns:
|
|
569
|
-
A streaming PySpark DataFrame over the DLO change feed.
|
|
570
|
-
"""
|
|
571
|
-
self._record_dlo_access(_streaming_source_name())
|
|
572
|
-
return self._reader.read_dlo_deltas() # type: ignore[no-any-return]
|
|
573
|
-
|
|
574
|
-
def read_dmo_deltas(self) -> PySparkDataFrame:
|
|
575
|
-
"""Read the streaming change feed (deltas) for a DMO from Data Cloud.
|
|
576
|
-
|
|
577
|
-
Returns:
|
|
578
|
-
A streaming PySpark DataFrame over the DMO change feed.
|
|
579
|
-
"""
|
|
580
|
-
self._record_dmo_access(_streaming_source_name())
|
|
581
|
-
return self._reader.read_dmo_deltas() # type: ignore[no-any-return]
|
|
582
|
-
|
|
583
|
-
def write_dlo_deltas(
|
|
584
|
-
self, name: str, dataframe: PySparkDataFrame, **kwargs
|
|
585
|
-
) -> StreamingQuery:
|
|
586
|
-
"""Write a streaming DataFrame of deltas to a DLO in Data Cloud.
|
|
587
|
-
|
|
588
|
-
Starts a streaming query that writes each micro-batch to the
|
|
589
|
-
target DLO and returns the ``StreamingQuery`` handle; the caller
|
|
590
|
-
typically calls ``query.awaitTermination()``.
|
|
591
|
-
|
|
592
|
-
Args:
|
|
593
|
-
name: The name of the DLO to write to.
|
|
594
|
-
dataframe: The streaming PySpark DataFrame to write.
|
|
595
|
-
|
|
596
|
-
Returns:
|
|
597
|
-
The started ``StreamingQuery``.
|
|
598
|
-
"""
|
|
599
|
-
self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
|
|
600
|
-
return self._writer.write_dlo_deltas(name, dataframe, **kwargs) # type: ignore[no-any-return]
|
datacustomcode/config.py
CHANGED
|
@@ -89,9 +89,6 @@ class ClientConfig(BaseConfig):
|
|
|
89
89
|
spark_provider_config: Union[
|
|
90
90
|
SparkProviderConfig[BaseSparkSessionProvider], None
|
|
91
91
|
] = None
|
|
92
|
-
# Source object name for a streaming (DELTA_SYNC) transform, populated by
|
|
93
|
-
# ``run_entrypoint`` from config.json's ``streamingSource`` field
|
|
94
|
-
streaming_source: Union[str, None] = None
|
|
95
92
|
|
|
96
93
|
def update(self, other: ClientConfig) -> ClientConfig:
|
|
97
94
|
"""Merge this ClientConfig with another, respecting force flags.
|
|
@@ -119,8 +116,6 @@ class ClientConfig(BaseConfig):
|
|
|
119
116
|
self.spark_provider_config = merge(
|
|
120
117
|
self.spark_provider_config, other.spark_provider_config
|
|
121
118
|
)
|
|
122
|
-
if other.streaming_source is not None:
|
|
123
|
-
self.streaming_source = other.streaming_source
|
|
124
119
|
return self
|
|
125
120
|
|
|
126
121
|
|
datacustomcode/config.yaml
CHANGED
datacustomcode/constants.py
CHANGED
|
@@ -35,17 +35,9 @@ FEATURE_TEMPLATE_MAPPING = {
|
|
|
35
35
|
|
|
36
36
|
# Feature name to Connect API name mapping
|
|
37
37
|
USE_IN_FEATURE_MAPPING_FOR_CONNECT_API = {
|
|
38
|
-
"SearchIndexChunking": "
|
|
38
|
+
"SearchIndexChunking": "SearchIndexChunking",
|
|
39
39
|
}
|
|
40
40
|
|
|
41
|
-
# Script (data transform) invoke options
|
|
42
|
-
SCRIPT_USE_IN_FEATURE_BATCH = "BatchTransform"
|
|
43
|
-
SCRIPT_USE_IN_FEATURE_STREAMING = "StreamingTransform"
|
|
44
|
-
SCRIPT_USE_IN_FEATURE_OPTIONS = [
|
|
45
|
-
SCRIPT_USE_IN_FEATURE_BATCH,
|
|
46
|
-
SCRIPT_USE_IN_FEATURE_STREAMING,
|
|
47
|
-
]
|
|
48
|
-
|
|
49
41
|
# Pydantic request/response type names to feature names
|
|
50
42
|
REQUEST_TYPE_TO_FEATURE = {
|
|
51
43
|
"SearchIndexChunkingV1Request": "SearchIndexChunking",
|