salesforce-data-customcode 6.1.0.dev3__py3-none-any.whl → 6.1.0.dev4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -23,6 +23,7 @@ __all__ = [
23
23
  "QueryAPIDataCloudReader",
24
24
  "SparkEinsteinPredictions",
25
25
  "SparkLLMGateway",
26
+ "StreamingClient",
26
27
  "einstein_predict_col",
27
28
  "llm_gateway_generate_text_col",
28
29
  ]
@@ -34,6 +35,10 @@ def __getattr__(name: str):
34
35
  from datacustomcode.client import Client
35
36
 
36
37
  return Client
38
+ elif name == "StreamingClient":
39
+ from datacustomcode.client import StreamingClient
40
+
41
+ return StreamingClient
37
42
  elif name == "AuthType":
38
43
  from datacustomcode.credentials import AuthType
39
44
 
datacustomcode/cli.py CHANGED
@@ -262,7 +262,7 @@ def deploy(
262
262
  mapped_feature = USE_IN_FEATURE_MAPPING_FOR_CONNECT_API.get(
263
263
  use_in_feature, use_in_feature
264
264
  )
265
- metadata.invokeOptions = [mapped_feature]
265
+ metadata.functionInvokeOptions = [mapped_feature]
266
266
 
267
267
  try:
268
268
  if sf_cli_org:
@@ -283,10 +283,19 @@ def deploy(
283
283
  )
284
284
  @click.option(
285
285
  "--use-in-feature",
286
- default="SearchIndexChunking",
287
- help="Feature where this function will be used (only applicable for function).",
286
+ "-u",
287
+ default=None,
288
+ help=(
289
+ "Invoke option for this package. For scripts: 'BatchTransform' "
290
+ "(default) or 'StreamingTransform'. For functions: 'SearchIndexChunking'."
291
+ ),
288
292
  )
289
293
  def init(directory: str, code_type: str, use_in_feature: Optional[str]):
294
+ from datacustomcode.constants import (
295
+ SCRIPT_USE_IN_FEATURE_BATCH,
296
+ SCRIPT_USE_IN_FEATURE_OPTIONS,
297
+ SCRIPT_USE_IN_FEATURE_STREAMING,
298
+ )
290
299
  from datacustomcode.scan import (
291
300
  dc_config_json_from_file,
292
301
  update_config,
@@ -294,9 +303,23 @@ def init(directory: str, code_type: str, use_in_feature: Optional[str]):
294
303
  )
295
304
  from datacustomcode.template import copy_function_template, copy_script_template
296
305
 
306
+ streaming = False
307
+ if code_type == "script":
308
+ use_in_feature = use_in_feature or SCRIPT_USE_IN_FEATURE_BATCH
309
+ if use_in_feature not in SCRIPT_USE_IN_FEATURE_OPTIONS:
310
+ click.secho(
311
+ f"Error: Invalid --use-in-feature '{use_in_feature}' for a "
312
+ f"script. Valid options: {', '.join(SCRIPT_USE_IN_FEATURE_OPTIONS)}.",
313
+ fg="red",
314
+ )
315
+ raise click.Abort()
316
+ streaming = use_in_feature == SCRIPT_USE_IN_FEATURE_STREAMING
317
+ else:
318
+ use_in_feature = use_in_feature or "SearchIndexChunking"
319
+
297
320
  click.echo("Copying template to " + click.style(directory, fg="blue", bold=True))
298
321
  if code_type == "script":
299
- copy_script_template(directory)
322
+ copy_script_template(directory, streaming=streaming)
300
323
  elif code_type == "function":
301
324
  copy_function_template(directory, use_in_feature)
302
325
  entrypoint_path = os.path.join(directory, PAYLOAD_DIR, ENTRYPOINT_FILE)
@@ -306,7 +329,9 @@ def init(directory: str, code_type: str, use_in_feature: Optional[str]):
306
329
  sdk_config = {"type": code_type}
307
330
  write_sdk_config(directory, sdk_config)
308
331
 
309
- config_json = dc_config_json_from_file(entrypoint_path, code_type)
332
+ config_json = dc_config_json_from_file(
333
+ entrypoint_path, code_type, streaming=streaming
334
+ )
310
335
  with open(config_location, "w") as f:
311
336
  json.dump(config_json, f, indent=2)
312
337
 
datacustomcode/client.py CHANGED
@@ -21,7 +21,9 @@ from typing import (
21
21
  ClassVar,
22
22
  Dict,
23
23
  Optional,
24
+ TypeVar,
24
25
  Union,
26
+ cast,
25
27
  )
26
28
 
27
29
  from datacustomcode.config import config
@@ -35,7 +37,12 @@ from datacustomcode.spark.default import DefaultSparkSessionProvider
35
37
  if TYPE_CHECKING:
36
38
  from pathlib import Path
37
39
 
38
- from pyspark.sql import Column, DataFrame as PySparkDataFrame
40
+ from pyspark.sql import (
41
+ Column,
42
+ DataFrame as PySparkDataFrame,
43
+ SparkSession,
44
+ )
45
+ from pyspark.sql.streaming import StreamingQuery
39
46
 
40
47
  from datacustomcode.einstein_predictions.spark_base import SparkEinsteinPredictions
41
48
  from datacustomcode.einstein_predictions.types import PredictionType
@@ -48,6 +55,40 @@ if TYPE_CHECKING:
48
55
  from datacustomcode.spark.base import BaseSparkSessionProvider
49
56
 
50
57
 
58
+ def _streaming_source_name() -> str:
59
+ """Return the streaming transform's read-source name.
60
+
61
+ Resolved from ``config.streaming_source``, which ``run_entrypoint``
62
+ populates from config.json's ``streamingSource`` field.
63
+
64
+ Raises:
65
+ RuntimeError: If no ``streaming_source`` has been configured (e.g. the
66
+ transform's config.json has no ``streamingSource`` field).
67
+ """
68
+ source = config.streaming_source
69
+ if not source:
70
+ raise RuntimeError(
71
+ "No streaming source configured. A streaming transform must declare "
72
+ "its read source in config.json under 'streamingSource'."
73
+ )
74
+ return source
75
+
76
+
77
+ def _active_client() -> "_BaseClient":
78
+ """Return the client backing the module-level Spark column helpers.
79
+
80
+ Prefers an already-initialized singleton so a streaming job reuses its
81
+ :class:`StreamingClient` (and a batch job its :class:`Client`) rather than
82
+ forcing an unrelated client into existence. Falls back to building the
83
+ batch :class:`Client` when neither has been created yet.
84
+ """
85
+ if Client._instance is not None:
86
+ return Client._instance
87
+ if StreamingClient._instance is not None:
88
+ return StreamingClient._instance
89
+ return Client()
90
+
91
+
51
92
  def _build_spark_llm_gateway() -> "SparkLLMGateway":
52
93
  """Instantiate the SDK-configured :class:`SparkLLMGateway`.
53
94
 
@@ -103,7 +144,7 @@ def llm_gateway_generate_text_col(
103
144
  the generated text; on failure, ``status == "ERROR"`` and the
104
145
  ``error_*`` fields carry diagnostic detail.
105
146
  """
106
- gateway = Client()._get_spark_llm_gateway()
147
+ gateway = _active_client()._get_spark_llm_gateway()
107
148
  return gateway.llm_gateway_generate_text_col(template, values, model_id=model_id)
108
149
 
109
150
 
@@ -180,7 +221,7 @@ def einstein_predict_col(
180
221
  the JSON-serialized prediction payload; on failure, ``status ==
181
222
  "ERROR"`` and the ``error_*`` fields carry diagnostic detail.
182
223
  """
183
- predictions = Client()._get_spark_einstein_predictions()
224
+ predictions = _active_client()._get_spark_einstein_predictions()
184
225
  return predictions.einstein_predict_col(
185
226
  model_api_name, prediction_type, features, settings=settings
186
227
  )
@@ -258,26 +299,22 @@ class DataCloudAccessLayerException(Exception):
258
299
  return msg
259
300
 
260
301
 
261
- class Client:
262
- """Entrypoint for accessing DataCloud objects.
302
+ _ClientT = TypeVar("_ClientT", bound="_BaseClient")
303
+
304
+
305
+ class _BaseClient:
306
+ """Shared machinery for the Data Cloud client singletons.
263
307
 
264
- This is the object used to access Data Cloud DLOs and DMOs. Accessing DLOs/DMOs
265
- are tracked and will throw an exception if they are mixed. In other words, you
266
- can read from DLOs and write to DLOs, read from DMOs and write to DMOs, but you
267
- cannot read from DLOs and write to DMOs or read from DMOs and write to DLOs.
268
- Furthermore you cannot mix during merging tables. This class is a singleton to
269
- prevent accidental mixing of DLOs and DMOs.
308
+ Holds the wiring common to :class:`Client` (batch) and
309
+ :class:`StreamingClient`
270
310
 
271
- You can provide custom readers and writers to the client for advanced use
272
- cases, but this is not recommended for testing as they may result in unexpected
273
- behavior once deployed to Data Cloud. By default, the client intercepts all
274
- read/write operations and mocks access to Data Cloud. For example, during
275
- writing, we print to the console instead of writing to Data Cloud.
311
+ This base class is not meant to be instantiated directly; use
312
+ :class:`Client` or :class:`StreamingClient`.
276
313
 
277
314
  Args:
278
- finder: Find a file path
279
315
  reader: A custom reader to use for reading Data Cloud objects.
280
316
  writer: A custom writer to use for writing Data Cloud objects.
317
+ spark_provider: Optional custom :class:`BaseSparkSessionProvider`.
281
318
  spark_llm_gateway: Optional custom :class:`SparkLLMGateway`.
282
319
  spark_einstein_predictions: Optional custom
283
320
  :class:`SparkEinsteinPredictions`.
@@ -291,7 +328,18 @@ class Client:
291
328
  >>> answer = client.llm_gateway_generate_text("Generate a greeting message")
292
329
  """
293
330
 
294
- _instance: ClassVar[Optional[Client]] = None
331
+ # Each concrete subclass gets its own ``_instance`` slot: reads fall through
332
+ # to this base default of ``None``, but ``cls._instance = ...`` in __new__
333
+ # always writes to the subclass, so ``Client`` and ``StreamingClient`` never
334
+ # share an instance.
335
+ _instance: ClassVar[Optional[_BaseClient]] = None
336
+ # Process-wide Spark session shared across BOTH client types. Unlike
337
+ # ``_instance``, this is written via ``_BaseClient._shared_spark`` (never
338
+ # ``cls._shared_spark``), so the slot lives on the base class and a
339
+ # ``Client`` and a ``StreamingClient`` in the same process reuse one session
340
+ # — and therefore one underlying connection — instead of opening two
341
+ # containing differing state
342
+ _shared_spark: ClassVar[Optional[SparkSession]] = None
295
343
  _reader: BaseDataCloudReader
296
344
  _writer: BaseDataCloudWriter
297
345
  _file: DefaultFindFilePath
@@ -302,7 +350,7 @@ class Client:
302
350
  _code_type: str
303
351
 
304
352
  def __new__(
305
- cls,
353
+ cls: type[_ClientT],
306
354
  reader: Optional[BaseDataCloudReader] = None,
307
355
  writer: Optional[BaseDataCloudWriter] = None,
308
356
  spark_provider: Optional[BaseSparkSessionProvider] = None,
@@ -310,31 +358,38 @@ class Client:
310
358
  spark_einstein_predictions: Optional[SparkEinsteinPredictions] = None,
311
359
  spark_named_credential: Optional[SparkNamedCredential] = None,
312
360
  code_type: str = "script",
313
- ) -> Client:
361
+ ) -> _ClientT:
314
362
 
315
363
  if cls._instance is None:
316
- cls._instance = super().__new__(cls)
317
- cls._instance._spark_llm_gateway = spark_llm_gateway
318
- cls._instance._spark_einstein_predictions = spark_einstein_predictions
319
- cls._instance._spark_named_credential = spark_named_credential
364
+ instance = super().__new__(cls)
365
+ instance._spark_llm_gateway = spark_llm_gateway
366
+ instance._spark_einstein_predictions = spark_einstein_predictions
367
+ instance._spark_named_credential = spark_named_credential
320
368
  # Initialize Readers and Writers from config
321
369
  # and/or provided reader and writer
322
370
  if reader is None or writer is None:
323
- # We need a spark because we will initialize readers and writers
324
- if config.spark_config is None:
325
- raise ValueError(
326
- "Spark config is required when reader/writer is not provided"
327
- )
328
-
329
- provider: BaseSparkSessionProvider
330
- if spark_provider is not None:
331
- provider = spark_provider
332
- elif config.spark_provider_config is not None:
333
- provider = config.spark_provider_config.to_object()
371
+ # We need a spark because we will initialize readers and writers.
372
+ # Reuse the process-wide session if one client already built it,
373
+ # so a Client and a StreamingClient share a single connection.
374
+ if _BaseClient._shared_spark is not None:
375
+ spark = _BaseClient._shared_spark
334
376
  else:
335
- provider = DefaultSparkSessionProvider()
336
-
337
- spark = provider.get_session(config.spark_config)
377
+ if config.spark_config is None:
378
+ raise ValueError(
379
+ "Spark config is required when reader/writer is not "
380
+ "provided"
381
+ )
382
+
383
+ provider: BaseSparkSessionProvider
384
+ if spark_provider is not None:
385
+ provider = spark_provider
386
+ elif config.spark_provider_config is not None:
387
+ provider = config.spark_provider_config.to_object()
388
+ else:
389
+ provider = DefaultSparkSessionProvider()
390
+
391
+ spark = provider.get_session(config.spark_config)
392
+ _BaseClient._shared_spark = spark
338
393
 
339
394
  if config.reader_config is None and reader is None:
340
395
  raise ValueError(
@@ -357,66 +412,17 @@ class Client:
357
412
  else:
358
413
  writer_init = writer
359
414
 
360
- cls._instance._reader = reader_init
361
- cls._instance._writer = writer_init
362
- cls._instance._file = DefaultFindFilePath()
363
- cls._instance._data_layer_history = {
415
+ instance._reader = reader_init
416
+ instance._writer = writer_init
417
+ instance._file = DefaultFindFilePath()
418
+ instance._data_layer_history = {
364
419
  DataCloudObjectType.DLO: set(),
365
420
  DataCloudObjectType.DMO: set(),
366
421
  }
367
- elif (reader is not None or writer is not None) and cls._instance is not None:
422
+ cls._instance = instance
423
+ elif reader is not None or writer is not None:
368
424
  raise ValueError("Cannot set reader or writer after client is initialized")
369
- return cls._instance
370
-
371
- def read_dlo(self, name: str) -> PySparkDataFrame:
372
- """Read a DLO from Data Cloud.
373
-
374
- Args:
375
- name: The name of the DLO to read.
376
-
377
- Returns:
378
- A PySpark DataFrame containing the DLO data.
379
- """
380
- self._record_dlo_access(name)
381
- return self._reader.read_dlo(name) # type: ignore[no-any-return]
382
-
383
- def read_dmo(self, name: str) -> PySparkDataFrame:
384
- """Read a DMO from Data Cloud.
385
-
386
- Args:
387
- name: The name of the DMO to read.
388
-
389
- Returns:
390
- A PySpark DataFrame containing the DMO data.
391
- """
392
- self._record_dmo_access(name)
393
- return self._reader.read_dmo(name) # type: ignore[no-any-return]
394
-
395
- def write_to_dlo(
396
- self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
397
- ) -> None:
398
- """Write a PySpark DataFrame to a DLO in Data Cloud.
399
-
400
- Args:
401
- name: The name of the DLO to write to.
402
- dataframe: The PySpark DataFrame to write.
403
- write_mode: The write mode to use for writing to the DLO.
404
- """
405
- self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
406
- return self._writer.write_to_dlo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
407
-
408
- def write_to_dmo(
409
- self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
410
- ) -> None:
411
- """Write a PySpark DataFrame to a DMO in Data Cloud.
412
-
413
- Args:
414
- name: The name of the DMO to write to.
415
- dataframe: The PySpark DataFrame to write.
416
- write_mode: The write mode to use for writing to the DMO.
417
- """
418
- self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
419
- return self._writer.write_to_dmo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
425
+ return cast(_ClientT, cls._instance)
420
426
 
421
427
  def find_file_path(self, file_name: str) -> Path:
422
428
  """Resolve a bundled file shipped in the package to an absolute path.
@@ -579,3 +585,115 @@ class Client:
579
585
 
580
586
  def _record_dmo_access(self, name: str) -> None:
581
587
  self._data_layer_history[DataCloudObjectType.DMO].add(name)
588
+
589
+
590
+ class Client(_BaseClient):
591
+ """Entrypoint for batch access to Data Cloud objects.
592
+
593
+ This is the object used to read and write bounded snapshots of Data Cloud
594
+ DLOs and DMOs.
595
+ """
596
+
597
+ _instance: ClassVar[Optional[Client]] = None
598
+
599
+ def read_dlo(self, name: str) -> PySparkDataFrame:
600
+ """Read a DLO from Data Cloud.
601
+
602
+ Args:
603
+ name: The name of the DLO to read.
604
+
605
+ Returns:
606
+ A PySpark DataFrame containing the DLO data.
607
+ """
608
+ self._record_dlo_access(name)
609
+ return self._reader.read_dlo(name) # type: ignore[no-any-return]
610
+
611
+ def read_dmo(self, name: str) -> PySparkDataFrame:
612
+ """Read a DMO from Data Cloud.
613
+
614
+ Args:
615
+ name: The name of the DMO to read.
616
+
617
+ Returns:
618
+ A PySpark DataFrame containing the DMO data.
619
+ """
620
+ self._record_dmo_access(name)
621
+ return self._reader.read_dmo(name) # type: ignore[no-any-return]
622
+
623
+ def write_to_dlo(
624
+ self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
625
+ ) -> None:
626
+ """Write a PySpark DataFrame to a DLO in Data Cloud.
627
+
628
+ Args:
629
+ name: The name of the DLO to write to.
630
+ dataframe: The PySpark DataFrame to write.
631
+ write_mode: The write mode to use for writing to the DLO.
632
+ """
633
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
634
+ return self._writer.write_to_dlo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
635
+
636
+ def write_to_dmo(
637
+ self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode, **kwargs
638
+ ) -> None:
639
+ """Write a PySpark DataFrame to a DMO in Data Cloud.
640
+
641
+ Args:
642
+ name: The name of the DMO to write to.
643
+ dataframe: The PySpark DataFrame to write.
644
+ write_mode: The write mode to use for writing to the DMO.
645
+ """
646
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
647
+ return self._writer.write_to_dmo(name, dataframe, write_mode, **kwargs) # type: ignore[no-any-return]
648
+
649
+
650
+ class StreamingClient(_BaseClient):
651
+ """Entrypoint for streaming (``DELTA_SYNC``) access to Data Cloud objects.
652
+
653
+ This is the streaming counterpart to :class:`Client`. Instead of reading and
654
+ writing bounded snapshots, it reads a DLO/DMO change feed as a streaming
655
+ DataFrame and writes the transformed stream back via a ``StreamingQuery``.
656
+ """
657
+
658
+ _instance: ClassVar[Optional[StreamingClient]] = None
659
+
660
+ def read_dlo_deltas(self) -> PySparkDataFrame:
661
+ """Read the streaming change feed (deltas) for a DLO from Data Cloud.
662
+
663
+ For use in a streaming (``DELTA_SYNC``) BYOC transform. Returns a
664
+ streaming DataFrame whose rows carry the change-feed metadata columns
665
+ (``_record_type``, ``_commit_*``) alongside the source columns.
666
+
667
+ Returns:
668
+ A streaming PySpark DataFrame over the DLO change feed.
669
+ """
670
+ self._record_dlo_access(_streaming_source_name())
671
+ return self._reader.read_dlo_deltas() # type: ignore[no-any-return]
672
+
673
+ def read_dmo_deltas(self) -> PySparkDataFrame:
674
+ """Read the streaming change feed (deltas) for a DMO from Data Cloud.
675
+
676
+ Returns:
677
+ A streaming PySpark DataFrame over the DMO change feed.
678
+ """
679
+ self._record_dmo_access(_streaming_source_name())
680
+ return self._reader.read_dmo_deltas() # type: ignore[no-any-return]
681
+
682
+ def write_dlo_deltas(
683
+ self, name: str, dataframe: PySparkDataFrame, **kwargs
684
+ ) -> StreamingQuery:
685
+ """Write a streaming DataFrame of deltas to a DLO in Data Cloud.
686
+
687
+ Starts a streaming query that writes each micro-batch to the
688
+ target DLO and returns the ``StreamingQuery`` handle; the caller
689
+ typically calls ``query.awaitTermination()``.
690
+
691
+ Args:
692
+ name: The name of the DLO to write to.
693
+ dataframe: The streaming PySpark DataFrame to write.
694
+
695
+ Returns:
696
+ The started ``StreamingQuery``.
697
+ """
698
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
699
+ return self._writer.write_dlo_deltas(name, dataframe, **kwargs) # type: ignore[no-any-return]
datacustomcode/config.py CHANGED
@@ -89,6 +89,9 @@ class ClientConfig(BaseConfig):
89
89
  spark_provider_config: Union[
90
90
  SparkProviderConfig[BaseSparkSessionProvider], None
91
91
  ] = None
92
+ # Source object name for a streaming (DELTA_SYNC) transform, populated by
93
+ # ``run_entrypoint`` from config.json's ``streamingSource`` field
94
+ streaming_source: Union[str, None] = None
92
95
 
93
96
  def update(self, other: ClientConfig) -> ClientConfig:
94
97
  """Merge this ClientConfig with another, respecting force flags.
@@ -116,6 +119,8 @@ class ClientConfig(BaseConfig):
116
119
  self.spark_provider_config = merge(
117
120
  self.spark_provider_config, other.spark_provider_config
118
121
  )
122
+ if other.streaming_source is not None:
123
+ self.streaming_source = other.streaming_source
119
124
  return self
120
125
 
121
126
 
@@ -35,9 +35,17 @@ FEATURE_TEMPLATE_MAPPING = {
35
35
 
36
36
  # Feature name to Connect API name mapping
37
37
  USE_IN_FEATURE_MAPPING_FOR_CONNECT_API = {
38
- "SearchIndexChunking": "SearchIndexChunking",
38
+ "SearchIndexChunking": "UnstructuredChunking",
39
39
  }
40
40
 
41
+ # Script (data transform) invoke options
42
+ SCRIPT_USE_IN_FEATURE_BATCH = "BatchTransform"
43
+ SCRIPT_USE_IN_FEATURE_STREAMING = "StreamingTransform"
44
+ SCRIPT_USE_IN_FEATURE_OPTIONS = [
45
+ SCRIPT_USE_IN_FEATURE_BATCH,
46
+ SCRIPT_USE_IN_FEATURE_STREAMING,
47
+ ]
48
+
41
49
  # Pydantic request/response type names to feature names
42
50
  REQUEST_TYPE_TO_FEATURE = {
43
51
  "SearchIndexChunkingV1Request": "SearchIndexChunking",
datacustomcode/deploy.py CHANGED
@@ -14,6 +14,7 @@
14
14
  # limitations under the License.
15
15
  from __future__ import annotations
16
16
 
17
+ import copy
17
18
  from html import unescape
18
19
  import json
19
20
  import os
@@ -42,8 +43,9 @@ from datacustomcode.named_credential.direct.credentials import (
42
43
  )
43
44
  from datacustomcode.scan import find_base_directory, get_package_type
44
45
 
45
- DATA_CUSTOM_CODE_PATH = "services/data/v67.0/ssot/data-custom-code"
46
+ DATA_CUSTOM_CODE_PATH = "services/data/v63.0/ssot/data-custom-code"
46
47
  DATA_TRANSFORMS_PATH = "services/data/v63.0/ssot/data-transforms"
48
+ DATA_CUSTOM_CODE_INVOKE_OPTIONS_PATH = "services/data/v67.0/ssot/data-custom-code"
47
49
  WAIT_FOR_DEPLOYMENT_TIMEOUT = 3000
48
50
 
49
51
  # Available compute types for Data Cloud deployments.
@@ -110,6 +112,7 @@ class CodeExtensionMetadata(BaseModel):
110
112
  description: str
111
113
  computeType: str
112
114
  codeType: str
115
+ functionInvokeOptions: Union[list[str], None] = None
113
116
  invokeOptions: Union[list[str], None] = None
114
117
 
115
118
  def __init__(self, **data):
@@ -203,7 +206,14 @@ def create_deployment(
203
206
  access_token: AccessTokenResponse, metadata: CodeExtensionMetadata
204
207
  ) -> CreateDeploymentResponse:
205
208
  """Create a custom code deployment in the DataCloud."""
206
- url = _join_strip_url(access_token.instance_url, DATA_CUSTOM_CODE_PATH)
209
+ # invokeOptions only binds at v67.0; route there when it is set so the
210
+ # option isn't silently dropped. Everything else stays on v63.0.
211
+ code_custom_code_path = (
212
+ DATA_CUSTOM_CODE_INVOKE_OPTIONS_PATH
213
+ if metadata.invokeOptions
214
+ else DATA_CUSTOM_CODE_PATH
215
+ )
216
+ url = _join_strip_url(access_token.instance_url, code_custom_code_path)
207
217
  body = dict[str, Any](
208
218
  {
209
219
  "label": metadata.name,
@@ -214,6 +224,8 @@ def create_deployment(
214
224
  "codeType": metadata.codeType,
215
225
  }
216
226
  )
227
+ if metadata.functionInvokeOptions:
228
+ body["functionInvokeOptions"] = metadata.functionInvokeOptions
217
229
  if metadata.invokeOptions:
218
230
  body["invokeOptions"] = metadata.invokeOptions
219
231
  logger.debug(f"Creating deployment {metadata.name}...")
@@ -393,6 +405,30 @@ class DataTransformConfig(BaseConfig):
393
405
  dataspace: str
394
406
  permissions: Permissions
395
407
  dataObjects: Optional[list[DataObject]] = None
408
+ streamingSource: Optional[StreamingSource] = None
409
+
410
+ @property
411
+ def is_streaming(self) -> bool:
412
+ return self.streamingSource is not None
413
+
414
+ @model_validator(mode="after")
415
+ def _validate_layers(self) -> "DataTransformConfig":
416
+ read_is_dlo = isinstance(self.permissions.read, DloPermission)
417
+ write_is_dlo = isinstance(self.permissions.write, DloPermission)
418
+ if self.is_streaming:
419
+ if not write_is_dlo:
420
+ raise ValueError(
421
+ "A streaming transform must write to a DLO "
422
+ "(permissions.write must be a 'dlo' entry)."
423
+ )
424
+ elif read_is_dlo != write_is_dlo:
425
+ raise ValueError(
426
+ "permissions.read and permissions.write must both reference "
427
+ "DLOs or both reference DMOs (got "
428
+ f"read={type(self.permissions.read).__name__}, "
429
+ f"write={type(self.permissions.write).__name__})"
430
+ )
431
+ return self
396
432
 
397
433
 
398
434
  class FunctionConfig(BaseConfig):
@@ -407,23 +443,15 @@ class DmoPermission(BaseModel):
407
443
  dmo: list[str]
408
444
 
409
445
 
446
+ class StreamingSource(BaseModel):
447
+ type: str
448
+ name: str
449
+
450
+
410
451
  class Permissions(BaseModel):
411
452
  read: Union[DloPermission, DmoPermission]
412
453
  write: Union[DloPermission, DmoPermission]
413
454
 
414
- @model_validator(mode="after")
415
- def _no_mixed_layers(self) -> "Permissions":
416
- read_is_dlo = isinstance(self.read, DloPermission)
417
- write_is_dlo = isinstance(self.write, DloPermission)
418
- if read_is_dlo != write_is_dlo:
419
- raise ValueError(
420
- "permissions.read and permissions.write must both reference "
421
- "DLOs or both reference DMOs (got "
422
- f"read={type(self.read).__name__}, "
423
- f"write={type(self.write).__name__})"
424
- )
425
- return self
426
-
427
455
 
428
456
  def _permission_entries(perm: Union[DloPermission, DmoPermission]) -> list[str]:
429
457
  """Return the list of object names regardless of layer (DLO or DMO)."""
@@ -495,7 +523,9 @@ def create_data_transform(
495
523
  ) -> dict:
496
524
  """Create a data transform in the DataCloud."""
497
525
  script_name = metadata.name
498
- request_hydrated = DATA_TRANSFORM_REQUEST_TEMPLATE.copy()
526
+ # Deep copy: the template's nested nodes/sources/macros dicts would
527
+ # otherwise be shared across calls and accumulate entries between deploys.
528
+ request_hydrated = copy.deepcopy(DATA_TRANSFORM_REQUEST_TEMPLATE)
499
529
 
500
530
  # Add nodes for each write entry (DLO or DMO)
501
531
  for i, name in enumerate(
@@ -538,7 +568,7 @@ def create_data_transform(
538
568
  "definition": definition,
539
569
  "label": f"{metadata.name}",
540
570
  "name": f"{metadata.name}",
541
- "type": "BATCH",
571
+ "type": "STREAMING" if data_transform_config.is_streaming else "BATCH",
542
572
  "dataSpaceName": data_transform_config.dataspace,
543
573
  }
544
574
 
@@ -621,9 +651,18 @@ def deploy_full(
621
651
  callback=None,
622
652
  ) -> AccessTokenResponse:
623
653
  """Deploy a data transform in the DataCloud."""
654
+ from datacustomcode.constants import SCRIPT_USE_IN_FEATURE_STREAMING
655
+
624
656
  # prepare payload
625
657
  config = get_config(directory)
626
658
 
659
+ if (
660
+ isinstance(config, DataTransformConfig)
661
+ and config.is_streaming
662
+ and not metadata.invokeOptions
663
+ ):
664
+ metadata.invokeOptions = [SCRIPT_USE_IN_FEATURE_STREAMING]
665
+
627
666
  # create deployment and upload payload
628
667
  deployment = create_deployment(access_token, metadata)
629
668
  zip(directory, docker_network, metadata.codeType)
@@ -41,3 +41,45 @@ class BaseDataCloudReader(BaseDataAccessLayer):
41
41
  name: str,
42
42
  schema: Union[AtomicType, StructType, str, None] = None,
43
43
  ) -> PySparkDataFrame: ...
44
+
45
+ def read_dlo_deltas(self) -> PySparkDataFrame:
46
+ """Read the streaming change feed (deltas) for a Data Lake Object.
47
+
48
+ This is the streaming counterpart to :meth:`read_dlo`. It returns a
49
+ streaming DataFrame over the change feed the Data Cloud runtime
50
+ publishes for a streaming (``DELTA_SYNC``) transform. Concrete
51
+ streaming behavior is provided by the deployed Data Cloud runtime; the
52
+ base implementation raises :class:`NotImplementedError` so local
53
+ readers that do not support streaming fail clearly.
54
+
55
+ Returns:
56
+ A streaming PySpark DataFrame over the DLO change feed.
57
+
58
+ Raises:
59
+ NotImplementedError: If the active reader does not support streaming
60
+ deltas (e.g. the local development readers).
61
+ """
62
+ raise NotImplementedError(
63
+ "read_dlo_deltas is only supported when running in the Data Cloud "
64
+ "streaming runtime; the local reader does not support streaming "
65
+ "deltas."
66
+ )
67
+
68
+ def read_dmo_deltas(self) -> PySparkDataFrame:
69
+ """Read the streaming change feed (deltas) for a Data Model Object.
70
+
71
+ Streaming counterpart to :meth:`read_dmo`. See :meth:`read_dlo_deltas`
72
+ for behavior and the local-development caveat.
73
+
74
+ Returns:
75
+ A streaming PySpark DataFrame over the DMO change feed.
76
+
77
+ Raises:
78
+ NotImplementedError: If the active reader does not support streaming
79
+ deltas (e.g. the local development readers).
80
+ """
81
+ raise NotImplementedError(
82
+ "read_dmo_deltas is only supported when running in the Data Cloud "
83
+ "streaming runtime; the local reader does not support streaming "
84
+ "deltas."
85
+ )
@@ -22,6 +22,7 @@ from datacustomcode.io.base import BaseDataAccessLayer
22
22
 
23
23
  if TYPE_CHECKING:
24
24
  from pyspark.sql import DataFrame as PySparkDataFrame, SparkSession
25
+ from pyspark.sql.streaming import StreamingQuery
25
26
 
26
27
 
27
28
  class WriteMode(str, Enum):
@@ -57,3 +58,35 @@ class BaseDataCloudWriter(BaseDataAccessLayer):
57
58
  def write_to_dmo(
58
59
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
59
60
  ) -> None: ...
61
+
62
+ def write_dlo_deltas(
63
+ self, name: str, dataframe: PySparkDataFrame
64
+ ) -> StreamingQuery:
65
+ """Write a streaming DataFrame of deltas to a Data Lake Object.
66
+
67
+ Streaming counterpart to :meth:`write_to_dlo`. Starts a streaming query
68
+ that writes each micro-batch to the target DLO via the Data Cloud
69
+ streaming sink and returns the resulting ``StreamingQuery`` handle. The
70
+ runtime owns the trigger and checkpoint location; callers pass only the
71
+ table name. Concrete streaming behavior is provided by the deployed
72
+ Data Cloud runtime; the base implementation raises
73
+ :class:`NotImplementedError`.
74
+
75
+ Args:
76
+ name: Target Data Lake Object name.
77
+ dataframe: Streaming PySpark DataFrame produced from a
78
+ ``read_dlo_deltas`` / ``read_dmo_deltas`` source.
79
+
80
+ Returns:
81
+ The started ``StreamingQuery``; the caller drives its lifecycle
82
+ (typically ``query.awaitTermination()``).
83
+
84
+ Raises:
85
+ NotImplementedError: If the active writer does not support streaming
86
+ deltas (e.g. the local development writers).
87
+ """
88
+ raise NotImplementedError(
89
+ "write_dlo_deltas is only supported when running in the Data Cloud "
90
+ "streaming runtime; the local writer does not support streaming "
91
+ "deltas."
92
+ )
datacustomcode/run.py CHANGED
@@ -43,6 +43,15 @@ def _set_config_option(config_obj, key: str, value: Optional[str]) -> None:
43
43
  config_obj.options[key] = value
44
44
 
45
45
 
46
+ def _read_streaming_source(config_json: dict) -> Optional[str]:
47
+ """Return the streaming source name from config.json's ``streamingSource``."""
48
+ source = config_json.get("streamingSource")
49
+ if not isinstance(source, dict):
50
+ return None
51
+ name = source.get("name")
52
+ return str(name) if name else None
53
+
54
+
46
55
  def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
47
56
  if sf_cli_org:
48
57
  config_key = "sf_cli_org"
@@ -132,6 +141,8 @@ def run_entrypoint(
132
141
  _set_config_option(config.reader_config, "dataspace", dataspace)
133
142
  _set_config_option(config.writer_config, "dataspace", dataspace)
134
143
 
144
+ config.streaming_source = _read_streaming_source(config_json)
145
+
135
146
  _update_config_options(profile, sf_cli_org)
136
147
 
137
148
  for dependency in dependencies:
datacustomcode/scan.py CHANGED
@@ -15,6 +15,7 @@
15
15
  from __future__ import annotations
16
16
 
17
17
  import ast
18
+ import copy
18
19
  import json
19
20
  import os
20
21
  import sys
@@ -32,6 +33,8 @@ import pydantic
32
33
  from datacustomcode.version import get_version
33
34
 
34
35
  DATA_ACCESS_METHODS = ["read_dlo", "read_dmo", "write_to_dlo", "write_to_dmo"]
36
+ STREAMING_READ_METHODS = ["read_dlo_deltas", "read_dmo_deltas"]
37
+ STREAMING_WRITE_METHODS = ["write_dlo_deltas"]
35
38
 
36
39
  DATA_TRANSFORM_CONFIG_TEMPLATE = {
37
40
  "sdkVersion": get_version(),
@@ -43,6 +46,20 @@ DATA_TRANSFORM_CONFIG_TEMPLATE = {
43
46
  },
44
47
  }
45
48
 
49
+ STREAMING_TRANSFORM_CONFIG_TEMPLATE = {
50
+ "sdkVersion": get_version(),
51
+ "entryPoint": "",
52
+ "dataspace": "default",
53
+ "streamingSource": {
54
+ "type": "dlo",
55
+ "name": "",
56
+ },
57
+ "permissions": {
58
+ "read": {},
59
+ "write": {},
60
+ },
61
+ }
62
+
46
63
  FUNCTION_CONFIG_TEMPLATE = {
47
64
  "entryPoint": "",
48
65
  }
@@ -160,6 +177,35 @@ class DataAccessLayerCalls(pydantic.BaseModel):
160
177
  return next(iter(self.write_to_dmo))
161
178
 
162
179
 
180
+ class StreamingDataAccessLayerCalls(pydantic.BaseModel):
181
+ read_dlo_deltas: bool
182
+ read_dmo_deltas: bool
183
+ write_dlo_deltas: frozenset[str]
184
+
185
+ @pydantic.model_validator(mode="after")
186
+ def validate_access_layer(self) -> StreamingDataAccessLayerCalls:
187
+ if self.read_dlo_deltas and self.read_dmo_deltas:
188
+ raise ValueError(
189
+ "Cannot read DLO and DMO deltas in the same streaming transform."
190
+ )
191
+ if not self.read_dlo_deltas and not self.read_dmo_deltas:
192
+ raise ValueError(
193
+ "A streaming transform must read from at least one DLO or DMO "
194
+ "delta stream (read_dlo_deltas / read_dmo_deltas)."
195
+ )
196
+ if not self.write_dlo_deltas:
197
+ raise ValueError(
198
+ "A streaming transform must write to at least one DLO via "
199
+ "write_dlo_deltas."
200
+ )
201
+ return self
202
+
203
+ @property
204
+ def read_layer(self) -> str:
205
+ """Return the read source layer, ``"dlo"`` or ``"dmo"``."""
206
+ return "dlo" if self.read_dlo_deltas else "dmo"
207
+
208
+
163
209
  class ClientMethodVisitor(ast.NodeVisitor):
164
210
  """AST Visitor that finds all instances of Client read/write method calls."""
165
211
 
@@ -168,6 +214,9 @@ class ClientMethodVisitor(ast.NodeVisitor):
168
214
  self._read_dmo_instances: set[str] = set()
169
215
  self._write_to_dlo_instances: set[str] = set()
170
216
  self._write_to_dmo_instances: set[str] = set()
217
+ self._read_dlo_deltas: bool = False
218
+ self._read_dmo_deltas: bool = False
219
+ self._write_dlo_deltas_instances: set[str] = set()
171
220
  self.variable_values: Dict[str, Union[str, None]] = {}
172
221
 
173
222
  def visit_Assign(self, node: ast.Assign) -> None:
@@ -189,14 +238,15 @@ class ClientMethodVisitor(ast.NodeVisitor):
189
238
  node.func.value, ast.Name
190
239
  ):
191
240
  method_name = node.func.attr
241
+
242
+ if method_name == "read_dlo_deltas":
243
+ self._read_dlo_deltas = True
244
+ elif method_name == "read_dmo_deltas":
245
+ self._read_dmo_deltas = True
246
+
192
247
  if method_name in DATA_ACCESS_METHODS and node.args:
193
248
  arg = node.args[0]
194
- name = None
195
-
196
- if isinstance(arg, ast.Constant) and isinstance(arg.value, str):
197
- name = arg.value
198
- elif isinstance(arg, ast.Name) and arg.id in self.variable_values:
199
- name = self.variable_values[arg.id]
249
+ name = self._resolve_name_arg(arg)
200
250
 
201
251
  if name:
202
252
  if method_name == "read_dlo":
@@ -207,8 +257,29 @@ class ClientMethodVisitor(ast.NodeVisitor):
207
257
  self._write_to_dlo_instances.add(name)
208
258
  elif method_name == "write_to_dmo":
209
259
  self._write_to_dmo_instances.add(name)
260
+ elif method_name in STREAMING_WRITE_METHODS and node.args:
261
+ name = self._resolve_name_arg(node.args[0])
262
+ if name and method_name == "write_dlo_deltas":
263
+ self._write_dlo_deltas_instances.add(name)
210
264
  self.generic_visit(node)
211
265
 
266
+ def _resolve_name_arg(self, arg: ast.expr) -> Union[str, None]:
267
+ """Resolve a string-literal or tracked-variable first argument."""
268
+ if isinstance(arg, ast.Constant) and isinstance(arg.value, str):
269
+ return arg.value
270
+ if isinstance(arg, ast.Name) and arg.id in self.variable_values:
271
+ return self.variable_values[arg.id]
272
+ return None
273
+
274
+ @property
275
+ def is_streaming(self) -> bool:
276
+ """Whether any streaming (delta) access method was found."""
277
+ return (
278
+ self._read_dlo_deltas
279
+ or self._read_dmo_deltas
280
+ or bool(self._write_dlo_deltas_instances)
281
+ )
282
+
212
283
  def found(self) -> DataAccessLayerCalls:
213
284
  return DataAccessLayerCalls(
214
285
  read_dlo=frozenset(self._read_dlo_instances),
@@ -217,6 +288,13 @@ class ClientMethodVisitor(ast.NodeVisitor):
217
288
  write_to_dmo=frozenset(self._write_to_dmo_instances),
218
289
  )
219
290
 
291
+ def found_streaming(self) -> StreamingDataAccessLayerCalls:
292
+ return StreamingDataAccessLayerCalls(
293
+ read_dlo_deltas=self._read_dlo_deltas,
294
+ read_dmo_deltas=self._read_dmo_deltas,
295
+ write_dlo_deltas=frozenset(self._write_dlo_deltas_instances),
296
+ )
297
+
220
298
 
221
299
  class ImportVisitor(ast.NodeVisitor):
222
300
  """AST Visitor that extracts external package imports from Python code."""
@@ -301,23 +379,51 @@ def write_requirements_file(file_path: str) -> str:
301
379
  return requirements_path
302
380
 
303
381
 
304
- def scan_file(file_path: str) -> DataAccessLayerCalls:
305
- """Scan a single Python file for Client read/write method calls."""
382
+ def _visit_file(file_path: str) -> ClientMethodVisitor:
383
+ """Parse a Python file and return the populated method visitor."""
306
384
  with open(file_path, "r") as f:
307
- code = f.read()
308
- tree = ast.parse(code)
309
- visitor = ClientMethodVisitor()
310
- visitor.visit(tree)
311
- return visitor.found()
385
+ tree = ast.parse(f.read())
386
+ visitor = ClientMethodVisitor()
387
+ visitor.visit(tree)
388
+ return visitor
389
+
390
+
391
+ def scan_file(file_path: str) -> DataAccessLayerCalls:
392
+ """Scan a single Python file for batch Client read/write method calls."""
393
+ return _visit_file(file_path).found()
394
+
395
+
396
+ def scan_file_streaming(file_path: str) -> StreamingDataAccessLayerCalls:
397
+ """Scan a single Python file for StreamingClient delta method calls."""
398
+ return _visit_file(file_path).found_streaming()
399
+
312
400
 
401
+ def file_is_streaming(file_path: str) -> bool:
402
+ """Return whether the entrypoint uses streaming (delta) access methods."""
403
+ return _visit_file(file_path).is_streaming
313
404
 
314
- def dc_config_json_from_file(file_path: str, type: str) -> dict[str, Any]:
315
- """Create a Data Cloud Custom Code config JSON from a script."""
405
+
406
+ def dc_config_json_from_file(
407
+ file_path: str, type: str, streaming: bool = False
408
+ ) -> dict[str, Any]:
409
+ """Create a Data Cloud Custom Code config JSON from a script.
410
+
411
+ Args:
412
+ file_path: Path to the entrypoint.
413
+ type: Package type, ``"script"`` or ``"function"``.
414
+ streaming: For scripts, a streaming
415
+ (``streamingSource``) config instead of a batch one.
416
+ """
316
417
  config: dict[str, Any]
317
418
  if type == "script":
318
- config = DATA_TRANSFORM_CONFIG_TEMPLATE.copy()
419
+ template = (
420
+ STREAMING_TRANSFORM_CONFIG_TEMPLATE
421
+ if streaming
422
+ else DATA_TRANSFORM_CONFIG_TEMPLATE
423
+ )
424
+ config = copy.deepcopy(template)
319
425
  elif type == "function":
320
- config = FUNCTION_CONFIG_TEMPLATE.copy()
426
+ config = copy.deepcopy(FUNCTION_CONFIG_TEMPLATE)
321
427
  config["entryPoint"] = os.path.basename(file_path)
322
428
  return config
323
429
 
@@ -372,22 +478,51 @@ def update_config(file_path: str) -> dict[str, Any]:
372
478
 
373
479
  if package_type == "script":
374
480
  existing_config["dataspace"] = get_dataspace(existing_config)
375
- output = scan_file(file_path)
376
- read: dict[str, list[str]] = {}
377
- if output.read_dlo:
378
- read["dlo"] = list(output.read_dlo)
379
- else:
380
- read["dmo"] = list(output.read_dmo)
381
- write: dict[str, list[str]] = {}
382
- if output.write_to_dlo:
383
- write["dlo"] = list(output.write_to_dlo)
481
+ if file_is_streaming(file_path):
482
+ _update_streaming_config(existing_config, file_path)
384
483
  else:
385
- write["dmo"] = list(output.write_to_dmo)
386
-
387
- existing_config["permissions"] = {"read": read, "write": write}
484
+ existing_config.pop("streamingSource", None)
485
+ output = scan_file(file_path)
486
+ read: dict[str, list[str]] = {}
487
+ if output.read_dlo:
488
+ read["dlo"] = list(output.read_dlo)
489
+ else:
490
+ read["dmo"] = list(output.read_dmo)
491
+ write: dict[str, list[str]] = {}
492
+ if output.write_to_dlo:
493
+ write["dlo"] = list(output.write_to_dlo)
494
+ else:
495
+ write["dmo"] = list(output.write_to_dmo)
496
+
497
+ existing_config["permissions"] = {"read": read, "write": write}
388
498
  return existing_config
389
499
 
390
500
 
501
+ def _update_streaming_config(existing_config: dict[str, Any], file_path: str) -> None:
502
+ output = scan_file_streaming(file_path)
503
+ read_layer = output.read_layer
504
+
505
+ source = existing_config.get("streamingSource")
506
+ if not isinstance(source, dict):
507
+ source = {}
508
+ source_name = source.get("name", "")
509
+ existing_config["streamingSource"] = {"type": read_layer, "name": source_name}
510
+
511
+ if not source_name:
512
+ logger.warning(
513
+ "streamingSource.name is empty in config.json. A streaming "
514
+ "transform must declare its read source; set streamingSource.name "
515
+ "to the DLO/DMO the transform reads from."
516
+ )
517
+
518
+ read_names = [source_name] if source_name else []
519
+ write_names = list(output.write_dlo_deltas)
520
+ existing_config["permissions"] = {
521
+ "read": {read_layer: read_names},
522
+ "write": {"dlo": write_names},
523
+ }
524
+
525
+
391
526
  def get_dataspace(existing_config: dict[str, str]) -> str:
392
527
  if "dataspace" in existing_config:
393
528
  dataspace_value = existing_config["dataspace"]
@@ -23,8 +23,12 @@ from datacustomcode.constants import FEATURE_TEMPLATE_MAPPING
23
23
  script_template_dir = os.path.join(os.path.dirname(__file__), "templates", "script")
24
24
  function_template_dir = os.path.join(os.path.dirname(__file__), "templates", "function")
25
25
 
26
+ STREAMING_EXAMPLE_ENTRYPOINT = os.path.join(
27
+ script_template_dir, "examples", "streaming_deltas", "entrypoint.py"
28
+ )
26
29
 
27
- def copy_script_template(target_dir: str) -> None:
30
+
31
+ def copy_script_template(target_dir: str, streaming: bool = False) -> None:
28
32
  """Copy the template to the target directory."""
29
33
  os.makedirs(target_dir, exist_ok=True)
30
34
 
@@ -39,6 +43,14 @@ def copy_script_template(target_dir: str) -> None:
39
43
  logger.debug(f"Copying file {source} to {destination}...")
40
44
  shutil.copy2(source, destination)
41
45
 
46
+ if streaming:
47
+ destination = os.path.join(target_dir, "payload", "entrypoint.py")
48
+ logger.debug(
49
+ f"Copying streaming example {STREAMING_EXAMPLE_ENTRYPOINT} to "
50
+ f"{destination}..."
51
+ )
52
+ shutil.copy2(STREAMING_EXAMPLE_ENTRYPOINT, destination)
53
+
42
54
 
43
55
  def copy_function_template(target_dir: str, use_in_feature: Optional[str]) -> None:
44
56
  os.makedirs(target_dir, exist_ok=True)
@@ -0,0 +1,49 @@
1
+ """Streaming BYOC transform: read a DLO change feed and write the deltas back.
2
+
3
+ This example is the streaming counterpart to a normal batch entrypoint. Instead
4
+ of a batch ``Client`` with ``read_dlo`` / ``write_to_dlo`` (which read and write
5
+ a bounded snapshot), it uses a :class:`StreamingClient` and its streaming delta
6
+ methods:
7
+
8
+ * ``client.read_dlo_deltas()`` returns a *streaming* DataFrame over the
9
+ Change Data Feed of the source DLO. Each row carries the source columns plus
10
+ change-feed metadata columns (``_record_type``, ``_commit_*``).
11
+ * ``client.write_dlo_deltas(name, df)`` starts a streaming query that writes
12
+ each micro-batch to the target DLO and returns the ``StreamingQuery`` handle.
13
+ The runtime owns the trigger, and checkpoint location — the caller only
14
+ chooses the table.
15
+
16
+ The transform in between is ordinary PySpark. Because the source is a change
17
+ feed, keep the metadata columns on the DataFrame you hand to
18
+ ``write_dlo_deltas`` — the sink relies on them to merge changes correctly.
19
+
20
+ This entrypoint only runs inside the Data Cloud streaming (``DELTA_SYNC``)
21
+ runtime; the local ``datacustomcode run`` readers/writers raise
22
+ ``NotImplementedError`` for the delta methods.
23
+ """
24
+
25
+ from pyspark.sql.functions import col, upper
26
+
27
+ from datacustomcode.client import StreamingClient
28
+
29
+
30
+ def main():
31
+ client = StreamingClient()
32
+
33
+ # Streaming DataFrame over the source DLO's change feed.
34
+ deltas = client.read_dlo_deltas()
35
+
36
+ # Ordinary PySpark transform.
37
+ transformed = deltas.withColumn("description__c", upper(col("description__c")))
38
+
39
+ # Start the streaming write. write_dlo_deltas returns the StreamingQuery;
40
+ # the trigger and checkpoint location are provided by the runtime.
41
+ query = client.write_dlo_deltas("Account_std_copy__dll", transformed)
42
+
43
+ # Drive the query's lifecycle. In the streaming runtime this blocks until
44
+ # the job is stopped by the platform.
45
+ query.awaitTermination()
46
+
47
+
48
+ if __name__ == "__main__":
49
+ main()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: salesforce-data-customcode
3
- Version: 6.1.0.dev3
3
+ Version: 6.1.0.dev4
4
4
  Summary: Data Cloud Custom Code SDK
5
5
  License-Expression: Apache-2.0
6
6
  License-File: LICENSE.txt
@@ -170,15 +170,22 @@ Your Python dependencies can be packaged as .py files, .zip archives (containing
170
170
 
171
171
  ## API
172
172
 
173
- Your entry point script will define logic using the `Client` object which wraps data access layers.
173
+ Your entry point script will define logic using the `Client` object (for batch transforms) or the `StreamingClient` object (for streaming delta transforms), which wrap the data access layers. Both are singletons; a single transform should use one or the other, not both.
174
174
 
175
- You should only need the following methods:
175
+ For a batch transform, use `Client`. You should only need the following methods:
176
176
  * `find_file_path(file_name)` – Resolve a bundled file (placed under `payload/files/`) to a `pathlib.Path` that exists. Works the same locally and inside Data Cloud — see [Bundled file resolution](#bundled-file-resolution) below for the full lookup order. Raises `FileNotFoundError` if the file isn't found.
177
177
  * `read_dlo(name)` – Read from a Data Lake Object by name
178
178
  * `read_dmo(name)` – Read from a Data Model Object by name
179
179
  * `write_to_dlo(name, spark_dataframe, write_mode)` – Write to a Data Model Object by name with a Spark dataframe
180
180
  * `write_to_dmo(name, spark_dataframe, write_mode)` – Write to a Data Lake Object by name with a Spark dataframe
181
181
 
182
+ For a streaming (delta) transform, use `StreamingClient`, which exposes the streaming counterparts:
183
+ * `read_dlo_deltas()` – Read the streaming change feed (deltas) of a Data Lake Object as a streaming DataFrame.
184
+ * `read_dmo_deltas()` – Read the streaming change feed (deltas) of a Data Model Object as a streaming DataFrame.
185
+ * `write_dlo_deltas(name, spark_dataframe)` – Write a streaming DataFrame of deltas to a Data Lake Object; returns the started `StreamingQuery`
186
+
187
+ `find_file_path`, `llm_gateway_generate_text`, and `einstein_predict` are available on both clients.
188
+
182
189
  For example:
183
190
  ```python
184
191
  from datacustomcode import Client
@@ -194,6 +201,36 @@ client.write_to_dlo('output_DLO')
194
201
  > [!WARNING]
195
202
  > Currently we only support reading from DMOs and writing to DMOs or reading from DLOs and writing to DLOs, but they cannot mix.
196
203
 
204
+ ### Streaming (delta) transforms
205
+
206
+ Streaming BYOC transforms process a Data Lake Object's Change Data Feed continuously instead of reading a bounded snapshot. Use a `StreamingClient` and its `*_deltas` methods in place of the batch `Client` read/write methods:
207
+
208
+ ```python
209
+ from pyspark.sql.functions import col, upper
210
+
211
+ from datacustomcode import StreamingClient
212
+
213
+ client = StreamingClient()
214
+
215
+ # read_dlo_deltas returns a *streaming* DataFrame over the change feed.
216
+ # The runtime resolves the single streaming source, so no name is passed.
217
+ deltas = client.read_dlo_deltas()
218
+
219
+ # Ordinary PySpark transform.
220
+ transformed = deltas.withColumn("description__c", upper(col("description__c")))
221
+
222
+ # write_dlo_deltas starts a streaming query and returns the StreamingQuery.
223
+ # The runtime owns the trigger and checkpoint location; you
224
+ # choose only the target table.
225
+ query = client.write_dlo_deltas("Output__dll", transformed)
226
+ query.awaitTermination()
227
+ ```
228
+
229
+ Notes:
230
+
231
+ - These methods only run inside the Data Cloud streaming (`DELTA_SYNC`) runtime. Locally (`datacustomcode run`) they raise `NotImplementedError`, since there is no change feed to stream.
232
+ - A complete runnable entry point is provided in [`examples/streaming_deltas/entrypoint.py`](src/datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py).
233
+
197
234
  ### Bundled file resolution
198
235
 
199
236
  Place bundled files (CSVs, prompt files, etc.) under `payload/files/`. The same `client.find_file_path("data.csv")` call resolves consistently across all three runtimes:
@@ -1,14 +1,14 @@
1
- datacustomcode/__init__.py,sha256=SGsDnWDTihWkLNnWdqvH08zklQgxykar73JAcPL2plI,2666
1
+ datacustomcode/__init__.py,sha256=5vvWRk5gb2mZIaVZ9BSPRL45uKSUwmSerxhLp6A4Q-4,2815
2
2
  datacustomcode/auth.py,sha256=fpSjhIBdv9trC8yq2vuljAix_Euu-4Ah7HDCGhYjOxI,8309
3
- datacustomcode/cli.py,sha256=qAUEFztKjYaG6Ws8xuE8rUpUr06snmFbUEuNRggjdBY,13256
4
- datacustomcode/client.py,sha256=lbOseiGaDru-vUtdRXgAORhrJa5V5WrFkbcEIfzBkaU,24325
3
+ datacustomcode/cli.py,sha256=K4LHL5xrb7-erXbjYz7pT4Kg4jG6Wf81c9wl_8tLabE,14162
4
+ datacustomcode/client.py,sha256=0uw9qptultCYT-fZ3cXwaPvIFW5J9E79nD46tnu4NUE,28743
5
5
  datacustomcode/cmd.py,sha256=ZMs46aydJw2EaU26JgCtZmnqESQFHvvaJz10hnjZTBk,3537
6
6
  datacustomcode/common_config.py,sha256=SAUnxj3kqmOeWwPmFoYq4tuxokMgURVm4QwcGi-avL4,1928
7
- datacustomcode/config.py,sha256=2Pk61ieQsEWSxKDxlS66_rliKE8iX0j-9LweEm0aaGo,4072
7
+ datacustomcode/config.py,sha256=lqed3jcWfoIweikSFZelaAoyBa0cXXScL8O6vaqdeRU,4372
8
8
  datacustomcode/config.yaml,sha256=LqpbkZhzN0nZOPEPnaz2qOCbchp8fxC4GHamT_b56D0,1082
9
- datacustomcode/constants.py,sha256=wcWy82JAZgMCzIrDLb6k0nl-qtNHawGedGBxY4atL8E,1435
9
+ datacustomcode/constants.py,sha256=NCGptcql30En5N7WAH96TYD_DTzHFYdh1sZZ5f327cs,1686
10
10
  datacustomcode/credentials.py,sha256=D-7Zd3Nh_wStdj8_wUy5cnC8M_mdbfBijhIAi-2EbXs,9467
11
- datacustomcode/deploy.py,sha256=eN5k7Rbxe-Yf8c_CeeHICZJqzJqMuvIVNu11agNW8NA,22111
11
+ datacustomcode/deploy.py,sha256=kyd8QqvWrw_thHvGWEnjbeNVgAyjpoqR8TKqjcIriGA,23678
12
12
  datacustomcode/einstein_platform_client.py,sha256=ON2B_m--vdbCVVMVcLc81T-BRZ5-UuxVCer8E6OEQPA,4083
13
13
  datacustomcode/einstein_platform_config.py,sha256=6lb_FRbEzdt5a8-2cR15f13ZKw674uvRJ4Xq0o_pQXE,1490
14
14
  datacustomcode/einstein_predictions/__init__.py,sha256=_RIDTqcEHDRu6AsCsj9lRBykNkHI8Ch8JBILL1ioPtA,1237
@@ -32,12 +32,12 @@ datacustomcode/function_utils.py,sha256=y6vaoSSaJ9CeHhjLCNyqzJT4vx6yCePWRxNLz38d
32
32
  datacustomcode/io/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
33
33
  datacustomcode/io/base.py,sha256=gbwZWWVUbCbGR4jIg_4h4qOz8tOMjE4RDTleD23WFKo,973
34
34
  datacustomcode/io/reader/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
35
- datacustomcode/io/reader/base.py,sha256=JRKg8KfEPJGR58wqicsEdpAPiXTl2K1eV3eHQyxYWaE,1390
35
+ datacustomcode/io/reader/base.py,sha256=2hAqaZvzHvvr5KBEgQSjUrHpidcHkgUXXTbGQG-b8io,3176
36
36
  datacustomcode/io/reader/query_api.py,sha256=eVrohrcnTnhSMsGfPRHu5XltjFtGG0hyHt1o4q1hp2w,9340
37
37
  datacustomcode/io/reader/sf_cli.py,sha256=x5QacVqRZaZSyph1_wwxo67s8r29wsOcUsOaiF9cDWE,6324
38
38
  datacustomcode/io/reader/utils.py,sha256=HlHhPZoHfmWA3mF8kTnd3m4Nd2exaz4NsOJJfQY7Pew,1656
39
39
  datacustomcode/io/writer/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
40
- datacustomcode/io/writer/base.py,sha256=6e2Yszb6BiuN1zCV2rjvZ5CQ0YikQdUgdYAkutoy9pE,1787
40
+ datacustomcode/io/writer/base.py,sha256=LtyOtkcEsX3oQZ-2zWFeNutVae1T3ialbGbvwl-7spU,3246
41
41
  datacustomcode/io/writer/csv.py,sha256=asty5teBpNQ1fMGHZ7wA3suLhq0sk0lVQPN5x7TOSRk,1555
42
42
  datacustomcode/io/writer/print.py,sha256=0g2sP_1Wb95UIyEELWJt3dqM18Iv_CWhDW3mDxcIhns,5105
43
43
  datacustomcode/llm_gateway/__init__.py,sha256=rcoGpSkv36tZuAOcTxKkhNpM0qV03qgmer_ej07Vmfc,1088
@@ -72,12 +72,12 @@ datacustomcode/named_credential/types/http_response.py,sha256=ZqX52RMjSdn8OpHSDk
72
72
  datacustomcode/named_credential/types/http_response_builder.py,sha256=MoUjpT-P5W2uT1-h1rIfKpBQHpMj_ZA_nUmn-FyBdbQ,896
73
73
  datacustomcode/named_credential_config.py,sha256=M07-oWzYj0KTrfA0fqvk3BQvaaDaw7C6nxbR0s3lOUk,3645
74
74
  datacustomcode/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
75
- datacustomcode/run.py,sha256=PXfpIwcgqyW6KLawTUuzGSt4FeAcxtafg-FgBHGXOLg,8091
76
- datacustomcode/scan.py,sha256=Zb9mE733H_xaqyDI-LZjjnIcctMC9j1AaD_kcr9qTh4,14325
75
+ datacustomcode/run.py,sha256=G0b5PCU0nucOFGX578LlYjmCpmgAASADnf2rh8NmccI,8485
76
+ datacustomcode/scan.py,sha256=9rq1Wae_tYACJykaOmBOv1covYvOSdky8w13exnmyJ0,19170
77
77
  datacustomcode/spark/__init__.py,sha256=12drVVlRiczCxOQw-EzuGtLsikCM8baBXvDEgwvelCI,860
78
78
  datacustomcode/spark/base.py,sha256=tlGqM4LxuLoDa7OxJF5nVeb7phO5uvD1DHBQeH7NMMs,1036
79
79
  datacustomcode/spark/default.py,sha256=aMB8CaTPYwHbQE_7XqBLQUZPGRdDJcRL72qTVh0aG7M,1433
80
- datacustomcode/template.py,sha256=FNNi8YPfd-JB-hUt1DDWSFK5AqwKnTnRSy_KpqYluaU,3282
80
+ datacustomcode/template.py,sha256=1u0U86coPX7-8g0ZixCbFGDBOMNcOqMhTXdJDAN9TaU,3726
81
81
  datacustomcode/templates/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
82
82
  datacustomcode/templates/function/.devcontainer/devcontainer.json,sha256=21RNTadhC-rynON2-VOtr5U174Hxnr_MA4RQ4592rsg,166
83
83
  datacustomcode/templates/function/Dockerfile.dependencies,sha256=AfHRddm5l3ujv4vdrf0d-SMB-qxPCOa0XQm6ptP2Euw,174
@@ -111,6 +111,7 @@ datacustomcode/templates/script/account.ipynb,sha256=LIbxgiVxflNASdspF2lfpMKkKAT
111
111
  datacustomcode/templates/script/build_native_dependencies.sh,sha256=ICRrp4f1ATBwPaUiuVGjY-MwiubFswNJLf8gMGU6YNg,197
112
112
  datacustomcode/templates/script/examples/employee_hierarchy/employee_data.csv,sha256=C7ggLBfoyi3M2BdMLNyOeKqF-5OO-Da76lkwgWg2cVQ,302
113
113
  datacustomcode/templates/script/examples/employee_hierarchy/entrypoint.py,sha256=Mfm3iQtEHTQRW6cmZTwRKdTh3IeGWR-teXrd6x-YLSY,2223
114
+ datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py,sha256=1PpYyZck2j9IRTnoI-kfHbMf21vN8gA49_-qJr3rDjk,1984
114
115
  datacustomcode/templates/script/jupyterlab.sh,sha256=IHR3YQ8d_busuyvesByhJgzCKULIFPI-Hqogs0NPhCs,2432
115
116
  datacustomcode/templates/script/payload/config.json,sha256=0d2mEMt4NHIeOUTsgiPMuRKdOLIQxmh-Wq-IfEqU-gc,28
116
117
  datacustomcode/templates/script/payload/entrypoint.py,sha256=4Uph0ILa5ukk_6C90paK3uzQn0zZvNLr9ho3o_ZwErQ,2474
@@ -118,8 +119,8 @@ datacustomcode/templates/script/requirements-dev.txt,sha256=OWwuy1awesqOZOS3Zizm
118
119
  datacustomcode/templates/script/requirements.txt,sha256=qJbqzs5z0Hrx590U3T7dHq_m8UQEx2TdhgrJSz-QHIQ,40
119
120
  datacustomcode/token_provider.py,sha256=qA_e4vqSXnxn0MccFFPULWO8ZeUOEAu5QpBnoYCg-fc,6839
120
121
  datacustomcode/version.py,sha256=9LlbVrzwBvut1L308QbwNBgxdQk5CrghH39JARVSxms,989
121
- salesforce_data_customcode-6.1.0.dev3.dist-info/METADATA,sha256=FDlrw8kdEPaFPCw8PJ0knkNqmS3vIuntw2oznOqA-A0,25115
122
- salesforce_data_customcode-6.1.0.dev3.dist-info/WHEEL,sha256=eY7nduwzv-ldUxpzbRlxwvC693Hg6PX8bWDjEHjZ_dk,88
123
- salesforce_data_customcode-6.1.0.dev3.dist-info/entry_points.txt,sha256=WpQ94UB7UuRCYGOLtJV3vAgm1tl7iVf43VYRWIuNu5Y,74
124
- salesforce_data_customcode-6.1.0.dev3.dist-info/licenses/LICENSE.txt,sha256=iOi8EmQpfkFhMENi7VYtkl2EqS14pEOccoXHiW2dyPU,11443
125
- salesforce_data_customcode-6.1.0.dev3.dist-info/RECORD,,
122
+ salesforce_data_customcode-6.1.0.dev4.dist-info/METADATA,sha256=Z3pzLUNQb2iRxImuljMvwdpumcV5x8wJEVtySG545pc,27207
123
+ salesforce_data_customcode-6.1.0.dev4.dist-info/WHEEL,sha256=eY7nduwzv-ldUxpzbRlxwvC693Hg6PX8bWDjEHjZ_dk,88
124
+ salesforce_data_customcode-6.1.0.dev4.dist-info/entry_points.txt,sha256=WpQ94UB7UuRCYGOLtJV3vAgm1tl7iVf43VYRWIuNu5Y,74
125
+ salesforce_data_customcode-6.1.0.dev4.dist-info/licenses/LICENSE.txt,sha256=iOi8EmQpfkFhMENi7VYtkl2EqS14pEOccoXHiW2dyPU,11443
126
+ salesforce_data_customcode-6.1.0.dev4.dist-info/RECORD,,