salesforce-data-customcode 6.1.0.dev4__py3-none-any.whl → 6.1.0.dev5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. datacustomcode/client.py +56 -99
  2. datacustomcode/config.yaml +0 -6
  3. datacustomcode/deploy.py +2 -7
  4. datacustomcode/function/runtime.py +0 -16
  5. datacustomcode/io/writer/base.py +12 -0
  6. datacustomcode/io/writer/csv.py +8 -0
  7. datacustomcode/io/writer/print.py +7 -0
  8. datacustomcode/run.py +0 -7
  9. datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +37 -24
  10. datacustomcode/templates/script/jupyterlab.sh +18 -4
  11. {salesforce_data_customcode-6.1.0.dev4.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/METADATA +3 -3
  12. {salesforce_data_customcode-6.1.0.dev4.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/RECORD +15 -38
  13. datacustomcode/named_credential/__init__.py +0 -28
  14. datacustomcode/named_credential/base.py +0 -54
  15. datacustomcode/named_credential/default.py +0 -93
  16. datacustomcode/named_credential/direct/__init__.py +0 -19
  17. datacustomcode/named_credential/direct/auth.py +0 -63
  18. datacustomcode/named_credential/direct/credentials.py +0 -121
  19. datacustomcode/named_credential/direct/transport.py +0 -110
  20. datacustomcode/named_credential/direct/url_resolver.py +0 -112
  21. datacustomcode/named_credential/errors.py +0 -36
  22. datacustomcode/named_credential/spark_base.py +0 -93
  23. datacustomcode/named_credential/spark_default.py +0 -154
  24. datacustomcode/named_credential/types/__init__.py +0 -14
  25. datacustomcode/named_credential/types/http_method.py +0 -29
  26. datacustomcode/named_credential/types/http_request.py +0 -63
  27. datacustomcode/named_credential/types/http_request_builder.py +0 -55
  28. datacustomcode/named_credential/types/http_response.py +0 -43
  29. datacustomcode/named_credential/types/http_response_builder.py +0 -24
  30. datacustomcode/named_credential_config.py +0 -105
  31. datacustomcode/templates/function/example/chunking_with_external_callout/README.md +0 -119
  32. datacustomcode/templates/function/example/chunking_with_external_callout/config.json +0 -3
  33. datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +0 -161
  34. datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +0 -11
  35. datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +0 -16
  36. {salesforce_data_customcode-6.1.0.dev4.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/WHEEL +0 -0
  37. {salesforce_data_customcode-6.1.0.dev4.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/entry_points.txt +0 -0
  38. {salesforce_data_customcode-6.1.0.dev4.dist-info → salesforce_data_customcode-6.1.0.dev5.dist-info}/licenses/LICENSE.txt +0 -0
datacustomcode/client.py CHANGED
@@ -15,6 +15,7 @@
15
15
  from __future__ import annotations
16
16
 
17
17
  from enum import Enum
18
+ import os
18
19
  from typing import (
19
20
  TYPE_CHECKING,
20
21
  Any,
@@ -31,7 +32,6 @@ from datacustomcode.einstein_predictions_config import spark_einstein_prediction
31
32
  from datacustomcode.file.path.default import DefaultFindFilePath
32
33
  from datacustomcode.io.reader.base import BaseDataCloudReader
33
34
  from datacustomcode.llm_gateway_config import spark_llm_gateway_config
34
- from datacustomcode.named_credential_config import spark_named_credential_config
35
35
  from datacustomcode.spark.default import DefaultSparkSessionProvider
36
36
 
37
37
  if TYPE_CHECKING:
@@ -49,9 +49,6 @@ if TYPE_CHECKING:
49
49
  from datacustomcode.io.reader.base import BaseDataCloudReader
50
50
  from datacustomcode.io.writer.base import BaseDataCloudWriter, WriteMode
51
51
  from datacustomcode.llm_gateway.spark_base import SparkLLMGateway
52
- from datacustomcode.named_credential.spark_base import SparkNamedCredential
53
- from datacustomcode.named_credential.types.http_request import HTTPRequest
54
- from datacustomcode.named_credential.types.http_response import HTTPResponse
55
52
  from datacustomcode.spark.base import BaseSparkSessionProvider
56
53
 
57
54
 
@@ -163,21 +160,6 @@ def _build_spark_einstein_predictions() -> "SparkEinsteinPredictions":
163
160
  return cfg.to_object()
164
161
 
165
162
 
166
- def _build_spark_named_credential() -> "SparkNamedCredential":
167
- """Instantiate the SDK-configured :class:`SparkNamedCredential`.
168
-
169
- Raises:
170
- RuntimeError: If no ``spark_named_credential_config`` has been loaded.
171
- """
172
- cfg = spark_named_credential_config.spark_named_credential_config
173
- if cfg is None:
174
- raise RuntimeError(
175
- "spark_named_credential_config is not configured. Add a "
176
- "'spark_named_credential_config' section to config.yaml."
177
- )
178
- return cfg.to_object()
179
-
180
-
181
163
  def einstein_predict_col(
182
164
  model_api_name: str,
183
165
  prediction_type: "PredictionType",
@@ -227,40 +209,6 @@ def einstein_predict_col(
227
209
  )
228
210
 
229
211
 
230
- def named_credential_request_col(
231
- request: "HTTPRequest",
232
- body: Optional["Column"] = None,
233
- ) -> "Column":
234
- """Build a Spark Column that makes one Named Credential callout per row.
235
-
236
- The endpoint, method, and headers are fixed for the call (taken from
237
- ``request``); only ``body`` varies per row. Use this instead of
238
- :meth:`Client.named_credential_request` when the callout runs across a
239
- DataFrame so each row is dispatched independently rather than one-shot on
240
- the driver.
241
-
242
- The returned Column yields a struct ``{status, response, error_code,
243
- error_message}`` for each row. ``response`` is itself a struct
244
- ``{status_code, body, headers}``. Use ``[...]`` to pick a field, e.g.
245
- ``named_credential_request_col(...)["response"]["status_code"]``. A transport
246
- failure sets ``status`` to ``ERROR`` and populates ``error_message`` (a non-2xx
247
- HTTP response is still ``SUCCESS`` with its code in ``response.status_code``),
248
- so a single bad row does not abort the whole Spark job.
249
-
250
- Args:
251
- request: The callout template — its symbolic reference, method, and
252
- headers are applied to every row.
253
- body: Optional per-row ``Column`` holding the request body as a
254
- string (or null for no body).
255
-
256
- Returns:
257
- A Spark ``Column`` of ``StructType`` with fields ``status``,
258
- ``response``, ``error_code``, and ``error_message``.
259
- """
260
- named_credential = Client()._get_spark_named_credential()
261
- return named_credential.request_col(request, body=body)
262
-
263
-
264
212
  class DataCloudObjectType(Enum):
265
213
  DLO = "dlo"
266
214
  DMO = "dmo"
@@ -318,14 +266,6 @@ class _BaseClient:
318
266
  spark_llm_gateway: Optional custom :class:`SparkLLMGateway`.
319
267
  spark_einstein_predictions: Optional custom
320
268
  :class:`SparkEinsteinPredictions`.
321
- spark_named_credential: Optional custom :class:`SparkNamedCredential`.
322
-
323
- Example:
324
- >>> client = Client()
325
- >>> file_path = client.find_file_path("data.csv")
326
- >>> dlo = client.read_dlo("my_dlo")
327
- >>> client.write_to_dmo("my_dmo", dlo)
328
- >>> answer = client.llm_gateway_generate_text("Generate a greeting message")
329
269
  """
330
270
 
331
271
  # Each concrete subclass gets its own ``_instance`` slot: reads fall through
@@ -345,7 +285,6 @@ class _BaseClient:
345
285
  _file: DefaultFindFilePath
346
286
  _spark_llm_gateway: Optional[SparkLLMGateway]
347
287
  _spark_einstein_predictions: Optional[SparkEinsteinPredictions]
348
- _spark_named_credential: Optional[SparkNamedCredential]
349
288
  _data_layer_history: dict[DataCloudObjectType, set[str]]
350
289
  _code_type: str
351
290
 
@@ -356,7 +295,6 @@ class _BaseClient:
356
295
  spark_provider: Optional[BaseSparkSessionProvider] = None,
357
296
  spark_llm_gateway: Optional[SparkLLMGateway] = None,
358
297
  spark_einstein_predictions: Optional[SparkEinsteinPredictions] = None,
359
- spark_named_credential: Optional[SparkNamedCredential] = None,
360
298
  code_type: str = "script",
361
299
  ) -> _ClientT:
362
300
 
@@ -364,7 +302,6 @@ class _BaseClient:
364
302
  instance = super().__new__(cls)
365
303
  instance._spark_llm_gateway = spark_llm_gateway
366
304
  instance._spark_einstein_predictions = spark_einstein_predictions
367
- instance._spark_named_credential = spark_named_credential
368
305
  # Initialize Readers and Writers from config
369
306
  # and/or provided reader and writer
370
307
  if reader is None or writer is None:
@@ -537,41 +474,6 @@ class _BaseClient:
537
474
  self._spark_einstein_predictions = _build_spark_einstein_predictions()
538
475
  return self._spark_einstein_predictions
539
476
 
540
- def named_credential_request(
541
- self,
542
- request: "HTTPRequest",
543
- body: Optional[str] = None,
544
- ) -> "HTTPResponse":
545
- """Issue a one-shot Named Credential external callout. This is the
546
- scalar counterpart to :func:`named_credential_request_col`: it runs
547
- **once** on the driver — not per row. Use the column helper method
548
- instead when you want to fan a callout out across every row of a
549
- DataFrame.
550
-
551
- Example:
552
-
553
- >>> from datacustomcode.named_credential.types.http_request_builder \\
554
- ... import HTTPRequestBuilder
555
- >>> request = (
556
- ... HTTPRequestBuilder().set_url("callout:NC/search").build()
557
- ... )
558
- >>> response = Client().named_credential_request(request)
559
-
560
- Args:
561
- request: The callout request
562
- body: Optional request body. Set the ``Content-Type`` header to
563
- match the format; the SDK does not assume or inject one.
564
-
565
- Returns:
566
- The external service's response.
567
- """
568
- return self._get_spark_named_credential().request(request, body=body)
569
-
570
- def _get_spark_named_credential(self) -> SparkNamedCredential:
571
- if self._spark_named_credential is None:
572
- self._spark_named_credential = _build_spark_named_credential()
573
- return self._spark_named_credential
574
-
575
477
  def _validate_data_layer_history_does_not_contain(
576
478
  self, data_cloud_object_type: DataCloudObjectType
577
479
  ) -> None:
@@ -657,6 +559,15 @@ class StreamingClient(_BaseClient):
657
559
 
658
560
  _instance: ClassVar[Optional[StreamingClient]] = None
659
561
 
562
+ def read_dlo(self) -> PySparkDataFrame:
563
+ """Read the streamingSource
564
+
565
+ Returns:
566
+ A standard PySpark DataFrame from the streaming source DLO
567
+ """
568
+ self._record_dlo_access(_streaming_source_name())
569
+ return self._reader.read_dlo(_streaming_source_name())
570
+
660
571
  def read_dlo_deltas(self) -> PySparkDataFrame:
661
572
  """Read the streaming change feed (deltas) for a DLO from Data Cloud.
662
573
 
@@ -670,6 +581,14 @@ class StreamingClient(_BaseClient):
670
581
  self._record_dlo_access(_streaming_source_name())
671
582
  return self._reader.read_dlo_deltas() # type: ignore[no-any-return]
672
583
 
584
+ def read_dmo(self) -> PySparkDataFrame:
585
+ """Read the streamingSource
586
+
587
+ Returns a standard PySpark DataFrame from the streaming source DMO
588
+ """
589
+ self._record_dmo_access(_streaming_source_name())
590
+ return self._reader.read_dmo(_streaming_source_name())
591
+
673
592
  def read_dmo_deltas(self) -> PySparkDataFrame:
674
593
  """Read the streaming change feed (deltas) for a DMO from Data Cloud.
675
594
 
@@ -697,3 +616,41 @@ class StreamingClient(_BaseClient):
697
616
  """
698
617
  self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
699
618
  return self._writer.write_dlo_deltas(name, dataframe, **kwargs) # type: ignore[no-any-return]
619
+
620
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
621
+ """Write a PySpark DataFrame to a DLO in Data Cloud automatically picking
622
+ the WriteMode.
623
+ For use with streaming transforms when running in rebuild or initial sync mode.
624
+ Args:
625
+ name: The name of the DLO to write to.
626
+ dataframe: The PySpark DataFrame to write.
627
+ """
628
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
629
+ return self._writer.auto_write_to_dlo(name, dataframe)
630
+
631
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
632
+ """Write a PySpark DataFrame to a DMO in Data Cloud automatically picking
633
+ the WriteMode.
634
+ For use with streaming transforms when running in rebuild or initial sync mode.
635
+ Args:
636
+ name: The name of the DMO to write to.
637
+ dataframe: The PySpark DataFrame to write.
638
+ """
639
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
640
+ return self._writer.auto_write_to_dmo(name, dataframe)
641
+
642
+
643
+ class RunMode(Enum):
644
+ BATCH = "BATCH"
645
+ INITIAL_SYNC = "INITIAL_SYNC"
646
+ REBUILD = "REBUILD"
647
+ DELTA_SYNC = "DELTA_SYNC"
648
+
649
+
650
+ def get_run_mode() -> RunMode:
651
+ """Read and validate the BYOC_RUN_MODE env var; default to BATCH when unset."""
652
+ run_mode = os.getenv("BYOC_RUN_MODE", "BATCH").upper()
653
+ try:
654
+ return RunMode(run_mode)
655
+ except ValueError as exc:
656
+ raise ValueError("Set BYOC_RUN_MODE to a valid value") from exc
@@ -34,9 +34,3 @@ llm_gateway_config:
34
34
 
35
35
  spark_llm_gateway_config:
36
36
  type_config_name: DefaultSparkLLMGateway
37
-
38
- named_credential_config:
39
- type_config_name: DefaultNamedCredential
40
-
41
- spark_named_credential_config:
42
- type_config_name: DefaultSparkNamedCredential
datacustomcode/deploy.py CHANGED
@@ -38,9 +38,6 @@ import requests
38
38
 
39
39
  from datacustomcode.cmd import cmd_output
40
40
  from datacustomcode.constants import REQUEST_TYPE_TO_FEATURE
41
- from datacustomcode.named_credential.direct.credentials import (
42
- EXTERNAL_CALLOUT_CREDENTIAL,
43
- )
44
41
  from datacustomcode.scan import find_base_directory, get_package_type
45
42
 
46
43
  DATA_CUSTOM_CODE_PATH = "services/data/v63.0/ssot/data-custom-code"
@@ -251,8 +248,6 @@ DEPENDENCIES_ARCHIVE_PATH = os.path.join(
251
248
  )
252
249
  PY_FILES_PATH = os.path.join("payload", "py-files")
253
250
  ZIP_FILE_NAME = "deployment.zip"
254
- # Local-only files that must never be packaged into the deployment zip.
255
- EXCLUDED_FILES = (".DS_Store", EXTERNAL_CALLOUT_CREDENTIAL)
256
251
 
257
252
 
258
253
  def prepare_dependency_archive(
@@ -633,9 +628,9 @@ def zip(
633
628
 
634
629
  with zipfile.ZipFile(ZIP_FILE_NAME, "w", zipfile.ZIP_DEFLATED) as zipf:
635
630
  for root, dirs, files in os.walk(directory):
636
- # Skip .DS_Store and local credentials.
631
+ # Skip .DS_Store files when adding to zip
637
632
  for file in files:
638
- if file not in EXCLUDED_FILES:
633
+ if file != ".DS_Store":
639
634
  abs_path = os.path.join(root, file)
640
635
  arcname = os.path.relpath(abs_path, directory)
641
636
  zipf.write(abs_path, arcname)
@@ -23,8 +23,6 @@ from datacustomcode.file.path.default import DefaultFindFilePath
23
23
  from datacustomcode.function.base import BaseRuntime
24
24
  from datacustomcode.llm_gateway.base import LLMGateway
25
25
  from datacustomcode.llm_gateway_config import llm_gateway_config
26
- from datacustomcode.named_credential.base import NamedCredential
27
- from datacustomcode.named_credential_config import named_credential_config
28
26
 
29
27
 
30
28
  class Runtime(BaseRuntime):
@@ -71,7 +69,6 @@ class Runtime(BaseRuntime):
71
69
  self._llm_gateway: Optional[LLMGateway] = None
72
70
  self._file = DefaultFindFilePath()
73
71
  self._einstein_predictions: Optional[EinsteinPredictions] = None
74
- self._named_credential: Optional[NamedCredential] = None
75
72
 
76
73
  @property
77
74
  def llm_gateway(self) -> LLMGateway:
@@ -101,16 +98,3 @@ class Runtime(BaseRuntime):
101
98
  einstein_predictions_config.einstein_predictions_config.to_object()
102
99
  )
103
100
  return self._einstein_predictions
104
-
105
- @property
106
- def named_credential(self) -> NamedCredential:
107
- if self._named_credential is None:
108
- if named_credential_config.named_credential_config is None:
109
- raise RuntimeError(
110
- "Named Credential is not configured. Add "
111
- "'named_credential_config' section to config.yaml"
112
- )
113
- self._named_credential = (
114
- named_credential_config.named_credential_config.to_object()
115
- )
116
- return self._named_credential
@@ -59,6 +59,18 @@ class BaseDataCloudWriter(BaseDataAccessLayer):
59
59
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
60
60
  ) -> None: ...
61
61
 
62
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
63
+ """Write to a DLO automatically picking the write mode.
64
+ For use with streaming transforms when running in rebuild or initial sync mode.
65
+ """
66
+ raise NotImplementedError
67
+
68
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
69
+ """Write to a DMO automatically picking the write mode.
70
+ For use with streaming transforms when running in rebuild or initial sync mode.
71
+ """
72
+ raise NotImplementedError
73
+
62
74
  def write_dlo_deltas(
63
75
  self, name: str, dataframe: PySparkDataFrame
64
76
  ) -> StreamingQuery:
@@ -36,6 +36,10 @@ class CSVDataCloudWriter(BaseDataCloudWriter):
36
36
  name = f"{name}{SUFFIX}"
37
37
  dataframe.write.csv(name, mode=write_mode)
38
38
 
39
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
40
+ # use overwrite since this is a local only writer
41
+ self.write_to_dlo(name, dataframe, WriteMode.OVERWRITE)
42
+
39
43
  def write_to_dmo(
40
44
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
41
45
  ) -> None:
@@ -43,3 +47,7 @@ class CSVDataCloudWriter(BaseDataCloudWriter):
43
47
  if not name.lower().endswith(SUFFIX):
44
48
  name = f"{name}{SUFFIX}"
45
49
  dataframe.write.csv(name, mode=write_mode)
50
+
51
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
52
+ # use overwrite since this is a local only writer
53
+ self.write_to_dmo(name, dataframe, WriteMode.OVERWRITE)
@@ -122,6 +122,10 @@ class PrintDataCloudWriter(BaseDataCloudWriter):
122
122
 
123
123
  dataframe.show()
124
124
 
125
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
126
+ self.validate_dataframe_columns_against_dlo(dataframe, name)
127
+ dataframe.show()
128
+
125
129
  def write_to_dmo(
126
130
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
127
131
  ) -> None:
@@ -130,3 +134,6 @@ class PrintDataCloudWriter(BaseDataCloudWriter):
130
134
  # so just show the dataframe.
131
135
 
132
136
  dataframe.show()
137
+
138
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
139
+ dataframe.show()
datacustomcode/run.py CHANGED
@@ -27,7 +27,6 @@ from typing import (
27
27
  from datacustomcode.config import config
28
28
  from datacustomcode.einstein_predictions_config import einstein_predictions_config
29
29
  from datacustomcode.llm_gateway_config import llm_gateway_config
30
- from datacustomcode.named_credential_config import named_credential_config
31
30
  from datacustomcode.scan import find_base_directory, get_package_type
32
31
 
33
32
 
@@ -65,9 +64,6 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
65
64
  _set_config_option(
66
65
  llm_gateway_config.llm_gateway_config, config_key, sf_cli_org
67
66
  )
68
- _set_config_option(
69
- named_credential_config.named_credential_config, config_key, sf_cli_org
70
- )
71
67
  elif profile != "default":
72
68
  config_key = "credentials_profile"
73
69
  _set_config_option(config.reader_config, config_key, profile)
@@ -76,9 +72,6 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
76
72
  einstein_predictions_config.einstein_predictions_config, config_key, profile
77
73
  )
78
74
  _set_config_option(llm_gateway_config.llm_gateway_config, config_key, profile)
79
- _set_config_option(
80
- named_credential_config.named_credential_config, config_key, profile
81
- )
82
75
 
83
76
 
84
77
  def run_entrypoint(
@@ -3,46 +3,59 @@
3
3
  This example is the streaming counterpart to a normal batch entrypoint. Instead
4
4
  of a batch ``Client`` with ``read_dlo`` / ``write_to_dlo`` (which read and write
5
5
  a bounded snapshot), it uses a :class:`StreamingClient` and its streaming delta
6
- methods:
6
+ methods.
7
7
 
8
- * ``client.read_dlo_deltas()`` returns a *streaming* DataFrame over the
9
- Change Data Feed of the source DLO. Each row carries the source columns plus
10
- change-feed metadata columns (``_record_type``, ``_commit_*``).
11
- * ``client.write_dlo_deltas(name, df)`` starts a streaming query that writes
12
- each micro-batch to the target DLO and returns the ``StreamingQuery`` handle.
13
- The runtime owns the trigger, and checkpoint location — the caller only
14
- chooses the table.
8
+ The first run of a streaming job will use the run mode INITIAL_SYNC which behaves
9
+ like a batch run on the streaming source. A streaming transform can also use run
10
+ mode REBUILD to do the same thing on demand. Note that these will process all
11
+ source rows and overwrite the target.
15
12
 
16
13
  The transform in between is ordinary PySpark. Because the source is a change
17
14
  feed, keep the metadata columns on the DataFrame you hand to
18
15
  ``write_dlo_deltas`` — the sink relies on them to merge changes correctly.
19
16
 
20
- This entrypoint only runs inside the Data Cloud streaming (``DELTA_SYNC``)
21
- runtime; the local ``datacustomcode run`` readers/writers raise
17
+ This entrypoint only runs inside the Data Cloud runtime;
18
+ the local ``datacustomcode run`` readers/writers raise
22
19
  ``NotImplementedError`` for the delta methods.
23
20
  """
24
21
 
22
+ from pyspark.sql import DataFrame
25
23
  from pyspark.sql.functions import col, upper
26
24
 
27
- from datacustomcode.client import StreamingClient
25
+ from datacustomcode.client import (
26
+ RunMode,
27
+ StreamingClient,
28
+ get_run_mode,
29
+ )
28
30
 
29
31
 
30
32
  def main():
33
+ target_dlo = "Account_std_copy__dll"
31
34
  client = StreamingClient()
32
35
 
33
- # Streaming DataFrame over the source DLO's change feed.
34
- deltas = client.read_dlo_deltas()
35
-
36
- # Ordinary PySpark transform.
37
- transformed = deltas.withColumn("description__c", upper(col("description__c")))
38
-
39
- # Start the streaming write. write_dlo_deltas returns the StreamingQuery;
40
- # the trigger and checkpoint location are provided by the runtime.
41
- query = client.write_dlo_deltas("Account_std_copy__dll", transformed)
42
-
43
- # Drive the query's lifecycle. In the streaming runtime this blocks until
44
- # the job is stopped by the platform.
45
- query.awaitTermination()
36
+ if get_run_mode() == RunMode.DELTA_SYNC:
37
+ # Streaming DataFrame over the source DLO's change feed.
38
+ dataframe = client.read_dlo_deltas()
39
+ # Ordinary PySpark transform.
40
+ transformed = transform(dataframe)
41
+
42
+ # Start the streaming write. write_dlo_deltas returns the StreamingQuery;
43
+ # the trigger and checkpoint location are provided by the runtime.
44
+ query = client.write_dlo_deltas(target_dlo, transformed)
45
+
46
+ # Drive the query's lifecycle. In the streaming runtime this blocks until
47
+ # the job is stopped by the platform.
48
+ query.awaitTermination()
49
+ else:
50
+ # initial sync and rebuild read the entire streaming source DLO and
51
+ # write using a server-decided mode based on the run mode
52
+ dataframe = client.read_dlo()
53
+ transformed = transform(dataframe)
54
+ client.auto_write_to_dlo(target_dlo, transformed)
55
+
56
+
57
+ def transform(dataframe: DataFrame) -> DataFrame:
58
+ return dataframe.withColumn("description__c", upper(col("description__c")))
46
59
 
47
60
 
48
61
  if __name__ == "__main__":
@@ -45,13 +45,24 @@ check_docker() {
45
45
  echo "Docker daemon is running"
46
46
  }
47
47
 
48
+ # Function to check if openssl is installed
49
+ check_openssl() {
50
+ if ! command -v openssl &> /dev/null; then
51
+ echo "openssl is not installed. It is required to generate a secure JupyterLab access token."
52
+ exit 1
53
+ fi
54
+ }
55
+
48
56
  # Function to start Jupyter server
49
57
  start_jupyter() {
50
58
  echo "Building the docker image"
51
59
  docker build -t datacloud-customcode .
52
60
 
61
+ local TOKEN
62
+ TOKEN=$(openssl rand -hex 32)
63
+
53
64
  echo "Running the docker container"
54
- docker run -d --rm -p 8888:8888 \
65
+ docker run -d --rm -p 127.0.0.1:8888:8888 \
55
66
  -v $(pwd):/workspace \
56
67
  --name jupyter-server \
57
68
  datacloud-customcode jupyter lab \
@@ -59,12 +70,14 @@ start_jupyter() {
59
70
  --port=8888 \
60
71
  --no-browser \
61
72
  --allow-root \
62
- --NotebookApp.token='' \
63
- --NotebookApp.password='' \
73
+ --NotebookApp.token="$TOKEN" \
64
74
  --notebook-dir=/workspace
65
75
 
66
76
  sleep 3 # Wait for server to start
67
- open_browser "http://localhost:8888"
77
+ local URL
78
+ URL="http://localhost:8888/?token=$TOKEN"
79
+ echo "Opening $URL"
80
+ open_browser $URL
68
81
  }
69
82
 
70
83
  # Function to stop Jupyter server
@@ -82,6 +95,7 @@ stop_jupyter() {
82
95
  case "$1" in
83
96
  "start")
84
97
  check_docker
98
+ check_openssl
85
99
  start_jupyter
86
100
  ;;
87
101
  "stop")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: salesforce-data-customcode
3
- Version: 6.1.0.dev4
3
+ Version: 6.1.0.dev5
4
4
  Summary: Data Cloud Custom Code SDK
5
5
  License-Expression: Apache-2.0
6
6
  License-File: LICENSE.txt
@@ -534,7 +534,7 @@ exploration. Instead of running an entire script, one can run one code cell at
534
534
 
535
535
  You can read more about Jupyter Notebooks here: https://jupyter.org/
536
536
 
537
- 1. Within the root project of your package folder, run `./jupyterlab.sh start`
537
+ 1. Within the root project of your package folder, run `./jupyterlab.sh start`. This prints an access token and opens an already-authenticated JupyterLab session in your browser. If the browser doesn't open automatically, copy the printed `http://localhost:8888/?token=...` URL into your browser.
538
538
  1. Double-click on "account.ipynb" file, which provides a starting point for a notebook
539
539
  1. Use shift+enter to execute each cell within the notebook. Add/edit/delete cells of code as needed for your data exploration.
540
540
  1. Don't forget to run `./jupyterlab.sh stop` to stop the docker container
@@ -625,5 +625,5 @@ If you're using OAuth Tokens authentication, the initial configure will retrieve
625
625
  ## Other docs
626
626
 
627
627
  - [Troubleshooting](./docs/troubleshooting.md)
628
- - [For Contributors](./FOR_CONTRIBUTORS.md)
628
+ - [Contributing](./CONTRIBUTING.md)
629
629