salesforce-data-customcode 6.1.0.dev3__py3-none-any.whl → 6.1.0.dev5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. datacustomcode/__init__.py +5 -0
  2. datacustomcode/cli.py +30 -5
  3. datacustomcode/client.py +267 -192
  4. datacustomcode/config.py +5 -0
  5. datacustomcode/config.yaml +0 -6
  6. datacustomcode/constants.py +9 -1
  7. datacustomcode/deploy.py +58 -24
  8. datacustomcode/function/runtime.py +0 -16
  9. datacustomcode/io/reader/base.py +42 -0
  10. datacustomcode/io/writer/base.py +45 -0
  11. datacustomcode/io/writer/csv.py +8 -0
  12. datacustomcode/io/writer/print.py +7 -0
  13. datacustomcode/run.py +11 -7
  14. datacustomcode/scan.py +164 -29
  15. datacustomcode/template.py +13 -1
  16. datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +62 -0
  17. datacustomcode/templates/script/jupyterlab.sh +18 -4
  18. {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/METADATA +42 -5
  19. {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/RECORD +22 -44
  20. datacustomcode/named_credential/__init__.py +0 -28
  21. datacustomcode/named_credential/base.py +0 -54
  22. datacustomcode/named_credential/default.py +0 -93
  23. datacustomcode/named_credential/direct/__init__.py +0 -19
  24. datacustomcode/named_credential/direct/auth.py +0 -63
  25. datacustomcode/named_credential/direct/credentials.py +0 -121
  26. datacustomcode/named_credential/direct/transport.py +0 -110
  27. datacustomcode/named_credential/direct/url_resolver.py +0 -112
  28. datacustomcode/named_credential/errors.py +0 -36
  29. datacustomcode/named_credential/spark_base.py +0 -93
  30. datacustomcode/named_credential/spark_default.py +0 -154
  31. datacustomcode/named_credential/types/__init__.py +0 -14
  32. datacustomcode/named_credential/types/http_method.py +0 -29
  33. datacustomcode/named_credential/types/http_request.py +0 -63
  34. datacustomcode/named_credential/types/http_request_builder.py +0 -55
  35. datacustomcode/named_credential/types/http_response.py +0 -43
  36. datacustomcode/named_credential/types/http_response_builder.py +0 -24
  37. datacustomcode/named_credential_config.py +0 -105
  38. datacustomcode/templates/function/example/chunking_with_external_callout/README.md +0 -119
  39. datacustomcode/templates/function/example/chunking_with_external_callout/config.json +0 -3
  40. datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +0 -161
  41. datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +0 -11
  42. datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +0 -16
  43. {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/WHEEL +0 -0
  44. {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/entry_points.txt +0 -0
  45. {salesforce_data_customcode-6.1.0.dev3.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/licenses/LICENSE.txt +0 -0
@@ -34,9 +34,3 @@ llm_gateway_config:
34
34
 
35
35
  spark_llm_gateway_config:
36
36
  type_config_name: DefaultSparkLLMGateway
37
-
38
- named_credential_config:
39
- type_config_name: DefaultNamedCredential
40
-
41
- spark_named_credential_config:
42
- type_config_name: DefaultSparkNamedCredential
@@ -35,9 +35,17 @@ FEATURE_TEMPLATE_MAPPING = {
35
35
 
36
36
  # Feature name to Connect API name mapping
37
37
  USE_IN_FEATURE_MAPPING_FOR_CONNECT_API = {
38
- "SearchIndexChunking": "SearchIndexChunking",
38
+ "SearchIndexChunking": "UnstructuredChunking",
39
39
  }
40
40
 
41
+ # Script (data transform) invoke options
42
+ SCRIPT_USE_IN_FEATURE_BATCH = "BatchTransform"
43
+ SCRIPT_USE_IN_FEATURE_STREAMING = "StreamingTransform"
44
+ SCRIPT_USE_IN_FEATURE_OPTIONS = [
45
+ SCRIPT_USE_IN_FEATURE_BATCH,
46
+ SCRIPT_USE_IN_FEATURE_STREAMING,
47
+ ]
48
+
41
49
  # Pydantic request/response type names to feature names
42
50
  REQUEST_TYPE_TO_FEATURE = {
43
51
  "SearchIndexChunkingV1Request": "SearchIndexChunking",
datacustomcode/deploy.py CHANGED
@@ -14,6 +14,7 @@
14
14
  # limitations under the License.
15
15
  from __future__ import annotations
16
16
 
17
+ import copy
17
18
  from html import unescape
18
19
  import json
19
20
  import os
@@ -37,13 +38,11 @@ import requests
37
38
 
38
39
  from datacustomcode.cmd import cmd_output
39
40
  from datacustomcode.constants import REQUEST_TYPE_TO_FEATURE
40
- from datacustomcode.named_credential.direct.credentials import (
41
- EXTERNAL_CALLOUT_CREDENTIAL,
42
- )
43
41
  from datacustomcode.scan import find_base_directory, get_package_type
44
42
 
45
- DATA_CUSTOM_CODE_PATH = "services/data/v67.0/ssot/data-custom-code"
43
+ DATA_CUSTOM_CODE_PATH = "services/data/v63.0/ssot/data-custom-code"
46
44
  DATA_TRANSFORMS_PATH = "services/data/v63.0/ssot/data-transforms"
45
+ DATA_CUSTOM_CODE_INVOKE_OPTIONS_PATH = "services/data/v67.0/ssot/data-custom-code"
47
46
  WAIT_FOR_DEPLOYMENT_TIMEOUT = 3000
48
47
 
49
48
  # Available compute types for Data Cloud deployments.
@@ -110,6 +109,7 @@ class CodeExtensionMetadata(BaseModel):
110
109
  description: str
111
110
  computeType: str
112
111
  codeType: str
112
+ functionInvokeOptions: Union[list[str], None] = None
113
113
  invokeOptions: Union[list[str], None] = None
114
114
 
115
115
  def __init__(self, **data):
@@ -203,7 +203,14 @@ def create_deployment(
203
203
  access_token: AccessTokenResponse, metadata: CodeExtensionMetadata
204
204
  ) -> CreateDeploymentResponse:
205
205
  """Create a custom code deployment in the DataCloud."""
206
- url = _join_strip_url(access_token.instance_url, DATA_CUSTOM_CODE_PATH)
206
+ # invokeOptions only binds at v67.0; route there when it is set so the
207
+ # option isn't silently dropped. Everything else stays on v63.0.
208
+ code_custom_code_path = (
209
+ DATA_CUSTOM_CODE_INVOKE_OPTIONS_PATH
210
+ if metadata.invokeOptions
211
+ else DATA_CUSTOM_CODE_PATH
212
+ )
213
+ url = _join_strip_url(access_token.instance_url, code_custom_code_path)
207
214
  body = dict[str, Any](
208
215
  {
209
216
  "label": metadata.name,
@@ -214,6 +221,8 @@ def create_deployment(
214
221
  "codeType": metadata.codeType,
215
222
  }
216
223
  )
224
+ if metadata.functionInvokeOptions:
225
+ body["functionInvokeOptions"] = metadata.functionInvokeOptions
217
226
  if metadata.invokeOptions:
218
227
  body["invokeOptions"] = metadata.invokeOptions
219
228
  logger.debug(f"Creating deployment {metadata.name}...")
@@ -239,8 +248,6 @@ DEPENDENCIES_ARCHIVE_PATH = os.path.join(
239
248
  )
240
249
  PY_FILES_PATH = os.path.join("payload", "py-files")
241
250
  ZIP_FILE_NAME = "deployment.zip"
242
- # Local-only files that must never be packaged into the deployment zip.
243
- EXCLUDED_FILES = (".DS_Store", EXTERNAL_CALLOUT_CREDENTIAL)
244
251
 
245
252
 
246
253
  def prepare_dependency_archive(
@@ -393,6 +400,30 @@ class DataTransformConfig(BaseConfig):
393
400
  dataspace: str
394
401
  permissions: Permissions
395
402
  dataObjects: Optional[list[DataObject]] = None
403
+ streamingSource: Optional[StreamingSource] = None
404
+
405
+ @property
406
+ def is_streaming(self) -> bool:
407
+ return self.streamingSource is not None
408
+
409
+ @model_validator(mode="after")
410
+ def _validate_layers(self) -> "DataTransformConfig":
411
+ read_is_dlo = isinstance(self.permissions.read, DloPermission)
412
+ write_is_dlo = isinstance(self.permissions.write, DloPermission)
413
+ if self.is_streaming:
414
+ if not write_is_dlo:
415
+ raise ValueError(
416
+ "A streaming transform must write to a DLO "
417
+ "(permissions.write must be a 'dlo' entry)."
418
+ )
419
+ elif read_is_dlo != write_is_dlo:
420
+ raise ValueError(
421
+ "permissions.read and permissions.write must both reference "
422
+ "DLOs or both reference DMOs (got "
423
+ f"read={type(self.permissions.read).__name__}, "
424
+ f"write={type(self.permissions.write).__name__})"
425
+ )
426
+ return self
396
427
 
397
428
 
398
429
  class FunctionConfig(BaseConfig):
@@ -407,23 +438,15 @@ class DmoPermission(BaseModel):
407
438
  dmo: list[str]
408
439
 
409
440
 
441
+ class StreamingSource(BaseModel):
442
+ type: str
443
+ name: str
444
+
445
+
410
446
  class Permissions(BaseModel):
411
447
  read: Union[DloPermission, DmoPermission]
412
448
  write: Union[DloPermission, DmoPermission]
413
449
 
414
- @model_validator(mode="after")
415
- def _no_mixed_layers(self) -> "Permissions":
416
- read_is_dlo = isinstance(self.read, DloPermission)
417
- write_is_dlo = isinstance(self.write, DloPermission)
418
- if read_is_dlo != write_is_dlo:
419
- raise ValueError(
420
- "permissions.read and permissions.write must both reference "
421
- "DLOs or both reference DMOs (got "
422
- f"read={type(self.read).__name__}, "
423
- f"write={type(self.write).__name__})"
424
- )
425
- return self
426
-
427
450
 
428
451
  def _permission_entries(perm: Union[DloPermission, DmoPermission]) -> list[str]:
429
452
  """Return the list of object names regardless of layer (DLO or DMO)."""
@@ -495,7 +518,9 @@ def create_data_transform(
495
518
  ) -> dict:
496
519
  """Create a data transform in the DataCloud."""
497
520
  script_name = metadata.name
498
- request_hydrated = DATA_TRANSFORM_REQUEST_TEMPLATE.copy()
521
+ # Deep copy: the template's nested nodes/sources/macros dicts would
522
+ # otherwise be shared across calls and accumulate entries between deploys.
523
+ request_hydrated = copy.deepcopy(DATA_TRANSFORM_REQUEST_TEMPLATE)
499
524
 
500
525
  # Add nodes for each write entry (DLO or DMO)
501
526
  for i, name in enumerate(
@@ -538,7 +563,7 @@ def create_data_transform(
538
563
  "definition": definition,
539
564
  "label": f"{metadata.name}",
540
565
  "name": f"{metadata.name}",
541
- "type": "BATCH",
566
+ "type": "STREAMING" if data_transform_config.is_streaming else "BATCH",
542
567
  "dataSpaceName": data_transform_config.dataspace,
543
568
  }
544
569
 
@@ -603,9 +628,9 @@ def zip(
603
628
 
604
629
  with zipfile.ZipFile(ZIP_FILE_NAME, "w", zipfile.ZIP_DEFLATED) as zipf:
605
630
  for root, dirs, files in os.walk(directory):
606
- # Skip .DS_Store and local credentials.
631
+ # Skip .DS_Store files when adding to zip
607
632
  for file in files:
608
- if file not in EXCLUDED_FILES:
633
+ if file != ".DS_Store":
609
634
  abs_path = os.path.join(root, file)
610
635
  arcname = os.path.relpath(abs_path, directory)
611
636
  zipf.write(abs_path, arcname)
@@ -621,9 +646,18 @@ def deploy_full(
621
646
  callback=None,
622
647
  ) -> AccessTokenResponse:
623
648
  """Deploy a data transform in the DataCloud."""
649
+ from datacustomcode.constants import SCRIPT_USE_IN_FEATURE_STREAMING
650
+
624
651
  # prepare payload
625
652
  config = get_config(directory)
626
653
 
654
+ if (
655
+ isinstance(config, DataTransformConfig)
656
+ and config.is_streaming
657
+ and not metadata.invokeOptions
658
+ ):
659
+ metadata.invokeOptions = [SCRIPT_USE_IN_FEATURE_STREAMING]
660
+
627
661
  # create deployment and upload payload
628
662
  deployment = create_deployment(access_token, metadata)
629
663
  zip(directory, docker_network, metadata.codeType)
@@ -23,8 +23,6 @@ from datacustomcode.file.path.default import DefaultFindFilePath
23
23
  from datacustomcode.function.base import BaseRuntime
24
24
  from datacustomcode.llm_gateway.base import LLMGateway
25
25
  from datacustomcode.llm_gateway_config import llm_gateway_config
26
- from datacustomcode.named_credential.base import NamedCredential
27
- from datacustomcode.named_credential_config import named_credential_config
28
26
 
29
27
 
30
28
  class Runtime(BaseRuntime):
@@ -71,7 +69,6 @@ class Runtime(BaseRuntime):
71
69
  self._llm_gateway: Optional[LLMGateway] = None
72
70
  self._file = DefaultFindFilePath()
73
71
  self._einstein_predictions: Optional[EinsteinPredictions] = None
74
- self._named_credential: Optional[NamedCredential] = None
75
72
 
76
73
  @property
77
74
  def llm_gateway(self) -> LLMGateway:
@@ -101,16 +98,3 @@ class Runtime(BaseRuntime):
101
98
  einstein_predictions_config.einstein_predictions_config.to_object()
102
99
  )
103
100
  return self._einstein_predictions
104
-
105
- @property
106
- def named_credential(self) -> NamedCredential:
107
- if self._named_credential is None:
108
- if named_credential_config.named_credential_config is None:
109
- raise RuntimeError(
110
- "Named Credential is not configured. Add "
111
- "'named_credential_config' section to config.yaml"
112
- )
113
- self._named_credential = (
114
- named_credential_config.named_credential_config.to_object()
115
- )
116
- return self._named_credential
@@ -41,3 +41,45 @@ class BaseDataCloudReader(BaseDataAccessLayer):
41
41
  name: str,
42
42
  schema: Union[AtomicType, StructType, str, None] = None,
43
43
  ) -> PySparkDataFrame: ...
44
+
45
+ def read_dlo_deltas(self) -> PySparkDataFrame:
46
+ """Read the streaming change feed (deltas) for a Data Lake Object.
47
+
48
+ This is the streaming counterpart to :meth:`read_dlo`. It returns a
49
+ streaming DataFrame over the change feed the Data Cloud runtime
50
+ publishes for a streaming (``DELTA_SYNC``) transform. Concrete
51
+ streaming behavior is provided by the deployed Data Cloud runtime; the
52
+ base implementation raises :class:`NotImplementedError` so local
53
+ readers that do not support streaming fail clearly.
54
+
55
+ Returns:
56
+ A streaming PySpark DataFrame over the DLO change feed.
57
+
58
+ Raises:
59
+ NotImplementedError: If the active reader does not support streaming
60
+ deltas (e.g. the local development readers).
61
+ """
62
+ raise NotImplementedError(
63
+ "read_dlo_deltas is only supported when running in the Data Cloud "
64
+ "streaming runtime; the local reader does not support streaming "
65
+ "deltas."
66
+ )
67
+
68
+ def read_dmo_deltas(self) -> PySparkDataFrame:
69
+ """Read the streaming change feed (deltas) for a Data Model Object.
70
+
71
+ Streaming counterpart to :meth:`read_dmo`. See :meth:`read_dlo_deltas`
72
+ for behavior and the local-development caveat.
73
+
74
+ Returns:
75
+ A streaming PySpark DataFrame over the DMO change feed.
76
+
77
+ Raises:
78
+ NotImplementedError: If the active reader does not support streaming
79
+ deltas (e.g. the local development readers).
80
+ """
81
+ raise NotImplementedError(
82
+ "read_dmo_deltas is only supported when running in the Data Cloud "
83
+ "streaming runtime; the local reader does not support streaming "
84
+ "deltas."
85
+ )
@@ -22,6 +22,7 @@ from datacustomcode.io.base import BaseDataAccessLayer
22
22
 
23
23
  if TYPE_CHECKING:
24
24
  from pyspark.sql import DataFrame as PySparkDataFrame, SparkSession
25
+ from pyspark.sql.streaming import StreamingQuery
25
26
 
26
27
 
27
28
  class WriteMode(str, Enum):
@@ -57,3 +58,47 @@ class BaseDataCloudWriter(BaseDataAccessLayer):
57
58
  def write_to_dmo(
58
59
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
59
60
  ) -> None: ...
61
+
62
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
63
+ """Write to a DLO automatically picking the write mode.
64
+ For use with streaming transforms when running in rebuild or initial sync mode.
65
+ """
66
+ raise NotImplementedError
67
+
68
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
69
+ """Write to a DMO automatically picking the write mode.
70
+ For use with streaming transforms when running in rebuild or initial sync mode.
71
+ """
72
+ raise NotImplementedError
73
+
74
+ def write_dlo_deltas(
75
+ self, name: str, dataframe: PySparkDataFrame
76
+ ) -> StreamingQuery:
77
+ """Write a streaming DataFrame of deltas to a Data Lake Object.
78
+
79
+ Streaming counterpart to :meth:`write_to_dlo`. Starts a streaming query
80
+ that writes each micro-batch to the target DLO via the Data Cloud
81
+ streaming sink and returns the resulting ``StreamingQuery`` handle. The
82
+ runtime owns the trigger and checkpoint location; callers pass only the
83
+ table name. Concrete streaming behavior is provided by the deployed
84
+ Data Cloud runtime; the base implementation raises
85
+ :class:`NotImplementedError`.
86
+
87
+ Args:
88
+ name: Target Data Lake Object name.
89
+ dataframe: Streaming PySpark DataFrame produced from a
90
+ ``read_dlo_deltas`` / ``read_dmo_deltas`` source.
91
+
92
+ Returns:
93
+ The started ``StreamingQuery``; the caller drives its lifecycle
94
+ (typically ``query.awaitTermination()``).
95
+
96
+ Raises:
97
+ NotImplementedError: If the active writer does not support streaming
98
+ deltas (e.g. the local development writers).
99
+ """
100
+ raise NotImplementedError(
101
+ "write_dlo_deltas is only supported when running in the Data Cloud "
102
+ "streaming runtime; the local writer does not support streaming "
103
+ "deltas."
104
+ )
@@ -36,6 +36,10 @@ class CSVDataCloudWriter(BaseDataCloudWriter):
36
36
  name = f"{name}{SUFFIX}"
37
37
  dataframe.write.csv(name, mode=write_mode)
38
38
 
39
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
40
+ # use overwrite since this is a local only writer
41
+ self.write_to_dlo(name, dataframe, WriteMode.OVERWRITE)
42
+
39
43
  def write_to_dmo(
40
44
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
41
45
  ) -> None:
@@ -43,3 +47,7 @@ class CSVDataCloudWriter(BaseDataCloudWriter):
43
47
  if not name.lower().endswith(SUFFIX):
44
48
  name = f"{name}{SUFFIX}"
45
49
  dataframe.write.csv(name, mode=write_mode)
50
+
51
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
52
+ # use overwrite since this is a local only writer
53
+ self.write_to_dmo(name, dataframe, WriteMode.OVERWRITE)
@@ -122,6 +122,10 @@ class PrintDataCloudWriter(BaseDataCloudWriter):
122
122
 
123
123
  dataframe.show()
124
124
 
125
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
126
+ self.validate_dataframe_columns_against_dlo(dataframe, name)
127
+ dataframe.show()
128
+
125
129
  def write_to_dmo(
126
130
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
127
131
  ) -> None:
@@ -130,3 +134,6 @@ class PrintDataCloudWriter(BaseDataCloudWriter):
130
134
  # so just show the dataframe.
131
135
 
132
136
  dataframe.show()
137
+
138
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
139
+ dataframe.show()
datacustomcode/run.py CHANGED
@@ -27,7 +27,6 @@ from typing import (
27
27
  from datacustomcode.config import config
28
28
  from datacustomcode.einstein_predictions_config import einstein_predictions_config
29
29
  from datacustomcode.llm_gateway_config import llm_gateway_config
30
- from datacustomcode.named_credential_config import named_credential_config
31
30
  from datacustomcode.scan import find_base_directory, get_package_type
32
31
 
33
32
 
@@ -43,6 +42,15 @@ def _set_config_option(config_obj, key: str, value: Optional[str]) -> None:
43
42
  config_obj.options[key] = value
44
43
 
45
44
 
45
+ def _read_streaming_source(config_json: dict) -> Optional[str]:
46
+ """Return the streaming source name from config.json's ``streamingSource``."""
47
+ source = config_json.get("streamingSource")
48
+ if not isinstance(source, dict):
49
+ return None
50
+ name = source.get("name")
51
+ return str(name) if name else None
52
+
53
+
46
54
  def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
47
55
  if sf_cli_org:
48
56
  config_key = "sf_cli_org"
@@ -56,9 +64,6 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
56
64
  _set_config_option(
57
65
  llm_gateway_config.llm_gateway_config, config_key, sf_cli_org
58
66
  )
59
- _set_config_option(
60
- named_credential_config.named_credential_config, config_key, sf_cli_org
61
- )
62
67
  elif profile != "default":
63
68
  config_key = "credentials_profile"
64
69
  _set_config_option(config.reader_config, config_key, profile)
@@ -67,9 +72,6 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
67
72
  einstein_predictions_config.einstein_predictions_config, config_key, profile
68
73
  )
69
74
  _set_config_option(llm_gateway_config.llm_gateway_config, config_key, profile)
70
- _set_config_option(
71
- named_credential_config.named_credential_config, config_key, profile
72
- )
73
75
 
74
76
 
75
77
  def run_entrypoint(
@@ -132,6 +134,8 @@ def run_entrypoint(
132
134
  _set_config_option(config.reader_config, "dataspace", dataspace)
133
135
  _set_config_option(config.writer_config, "dataspace", dataspace)
134
136
 
137
+ config.streaming_source = _read_streaming_source(config_json)
138
+
135
139
  _update_config_options(profile, sf_cli_org)
136
140
 
137
141
  for dependency in dependencies: