salesforce-data-customcode 6.1.0.dev4__py3-none-any.whl → 6.1.0.dev6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
datacustomcode/client.py CHANGED
@@ -15,6 +15,7 @@
15
15
  from __future__ import annotations
16
16
 
17
17
  from enum import Enum
18
+ import os
18
19
  from typing import (
19
20
  TYPE_CHECKING,
20
21
  Any,
@@ -657,6 +658,15 @@ class StreamingClient(_BaseClient):
657
658
 
658
659
  _instance: ClassVar[Optional[StreamingClient]] = None
659
660
 
661
+ def read_dlo(self) -> PySparkDataFrame:
662
+ """Read the streamingSource
663
+
664
+ Returns:
665
+ A standard PySpark DataFrame from the streaming source DLO
666
+ """
667
+ self._record_dlo_access(_streaming_source_name())
668
+ return self._reader.read_dlo(_streaming_source_name())
669
+
660
670
  def read_dlo_deltas(self) -> PySparkDataFrame:
661
671
  """Read the streaming change feed (deltas) for a DLO from Data Cloud.
662
672
 
@@ -670,6 +680,14 @@ class StreamingClient(_BaseClient):
670
680
  self._record_dlo_access(_streaming_source_name())
671
681
  return self._reader.read_dlo_deltas() # type: ignore[no-any-return]
672
682
 
683
+ def read_dmo(self) -> PySparkDataFrame:
684
+ """Read the streamingSource
685
+
686
+ Returns a standard PySpark DataFrame from the streaming source DMO
687
+ """
688
+ self._record_dmo_access(_streaming_source_name())
689
+ return self._reader.read_dmo(_streaming_source_name())
690
+
673
691
  def read_dmo_deltas(self) -> PySparkDataFrame:
674
692
  """Read the streaming change feed (deltas) for a DMO from Data Cloud.
675
693
 
@@ -697,3 +715,41 @@ class StreamingClient(_BaseClient):
697
715
  """
698
716
  self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
699
717
  return self._writer.write_dlo_deltas(name, dataframe, **kwargs) # type: ignore[no-any-return]
718
+
719
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
720
+ """Write a PySpark DataFrame to a DLO in Data Cloud automatically picking
721
+ the WriteMode.
722
+ For use with streaming transforms when running in rebuild or initial sync mode.
723
+ Args:
724
+ name: The name of the DLO to write to.
725
+ dataframe: The PySpark DataFrame to write.
726
+ """
727
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DMO)
728
+ return self._writer.auto_write_to_dlo(name, dataframe)
729
+
730
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
731
+ """Write a PySpark DataFrame to a DMO in Data Cloud automatically picking
732
+ the WriteMode.
733
+ For use with streaming transforms when running in rebuild or initial sync mode.
734
+ Args:
735
+ name: The name of the DMO to write to.
736
+ dataframe: The PySpark DataFrame to write.
737
+ """
738
+ self._validate_data_layer_history_does_not_contain(DataCloudObjectType.DLO)
739
+ return self._writer.auto_write_to_dmo(name, dataframe)
740
+
741
+
742
+ class RunMode(Enum):
743
+ BATCH = "BATCH"
744
+ INITIAL_SYNC = "INITIAL_SYNC"
745
+ REBUILD = "REBUILD"
746
+ DELTA_SYNC = "DELTA_SYNC"
747
+
748
+
749
+ def get_run_mode() -> RunMode:
750
+ """Read and validate the BYOC_RUN_MODE env var; default to BATCH when unset."""
751
+ run_mode = os.getenv("BYOC_RUN_MODE", "BATCH").upper()
752
+ try:
753
+ return RunMode(run_mode)
754
+ except ValueError as exc:
755
+ raise ValueError("Set BYOC_RUN_MODE to a valid value") from exc
@@ -59,6 +59,18 @@ class BaseDataCloudWriter(BaseDataAccessLayer):
59
59
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
60
60
  ) -> None: ...
61
61
 
62
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
63
+ """Write to a DLO automatically picking the write mode.
64
+ For use with streaming transforms when running in rebuild or initial sync mode.
65
+ """
66
+ raise NotImplementedError
67
+
68
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
69
+ """Write to a DMO automatically picking the write mode.
70
+ For use with streaming transforms when running in rebuild or initial sync mode.
71
+ """
72
+ raise NotImplementedError
73
+
62
74
  def write_dlo_deltas(
63
75
  self, name: str, dataframe: PySparkDataFrame
64
76
  ) -> StreamingQuery:
@@ -36,6 +36,10 @@ class CSVDataCloudWriter(BaseDataCloudWriter):
36
36
  name = f"{name}{SUFFIX}"
37
37
  dataframe.write.csv(name, mode=write_mode)
38
38
 
39
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
40
+ # use overwrite since this is a local only writer
41
+ self.write_to_dlo(name, dataframe, WriteMode.OVERWRITE)
42
+
39
43
  def write_to_dmo(
40
44
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
41
45
  ) -> None:
@@ -43,3 +47,7 @@ class CSVDataCloudWriter(BaseDataCloudWriter):
43
47
  if not name.lower().endswith(SUFFIX):
44
48
  name = f"{name}{SUFFIX}"
45
49
  dataframe.write.csv(name, mode=write_mode)
50
+
51
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
52
+ # use overwrite since this is a local only writer
53
+ self.write_to_dmo(name, dataframe, WriteMode.OVERWRITE)
@@ -122,6 +122,10 @@ class PrintDataCloudWriter(BaseDataCloudWriter):
122
122
 
123
123
  dataframe.show()
124
124
 
125
+ def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
126
+ self.validate_dataframe_columns_against_dlo(dataframe, name)
127
+ dataframe.show()
128
+
125
129
  def write_to_dmo(
126
130
  self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
127
131
  ) -> None:
@@ -130,3 +134,6 @@ class PrintDataCloudWriter(BaseDataCloudWriter):
130
134
  # so just show the dataframe.
131
135
 
132
136
  dataframe.show()
137
+
138
+ def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
139
+ dataframe.show()
@@ -13,17 +13,25 @@
13
13
  # See the License for the specific language governing permissions and
14
14
  # limitations under the License.
15
15
 
16
- from typing import Dict, Union
16
+ from typing import (
17
+ Dict,
18
+ Optional,
19
+ Union,
20
+ )
17
21
 
18
22
  from datacustomcode.named_credential.types.http_method import HTTPMethod
19
23
  from datacustomcode.named_credential.types.http_request import HTTPRequest
20
24
 
25
+ # Control header carrying the per-request callout response timeout (seconds).
26
+ CALLOUT_RESPONSE_TIMEOUT_HEADER = "ctx-callout-response-timeout-seconds"
27
+
21
28
 
22
29
  class HTTPRequestBuilder:
23
30
  def __init__(self) -> None:
24
31
  self._url = ""
25
32
  self._method: Union[str, HTTPMethod] = HTTPMethod.GET
26
33
  self._headers: Dict[str, str] = {}
34
+ self._response_timeout_seconds: Optional[int] = None
27
35
 
28
36
  def set_url(self, url: str) -> "HTTPRequestBuilder":
29
37
  """Set the symbolic Named Credential reference.
@@ -44,12 +52,37 @@ class HTTPRequestBuilder:
44
52
  return self
45
53
 
46
54
  def set_headers(self, headers: Dict[str, str]) -> "HTTPRequestBuilder":
55
+ """Set the HTTP headers.
56
+
57
+ Args:
58
+ headers: a dictionary of HTTP headers
59
+ """
47
60
  self._headers = headers
48
61
  return self
49
62
 
63
+ def set_response_timeout_seconds(self, seconds: int) -> "HTTPRequestBuilder":
64
+ """Set the per-request callout response timeout, in seconds.
65
+
66
+ The ``ctx-callout-response-timeout-seconds`` control header allows to
67
+ configure the timeout for the callout response. This is optional.
68
+ If not set, the platform will use the default timeout.
69
+ """
70
+ if isinstance(seconds, bool) or not isinstance(seconds, int) or seconds <= 0:
71
+ raise ValueError(
72
+ "response timeout must be a positive integer number of seconds, "
73
+ f"got {seconds!r}."
74
+ )
75
+ self._response_timeout_seconds = seconds
76
+ return self
77
+
50
78
  def build(self) -> HTTPRequest:
79
+ headers = dict(self._headers)
80
+ if self._response_timeout_seconds is not None:
81
+ headers[CALLOUT_RESPONSE_TIMEOUT_HEADER] = str(
82
+ self._response_timeout_seconds
83
+ )
51
84
  return HTTPRequest(
52
85
  url=self._url,
53
86
  method=self._method, # type: ignore[arg-type]
54
- headers=self._headers,
87
+ headers=headers,
55
88
  )
@@ -16,6 +16,7 @@ request = (
16
16
  .set_url(CALLOUT_URL)
17
17
  .set_method(HTTPMethod.POST)
18
18
  .set_headers({"Content-Type": "application/json"})
19
+ .set_response_timeout_seconds(60) # optional: per-callout timeout (seconds)
19
20
  .build()
20
21
  )
21
22
  # Body is sent verbatim (serialize it yourself); the response body is a raw string.
@@ -108,6 +108,7 @@ def _analyze_chunk(chunk_text: str, runtime: Runtime) -> dict[str, str]:
108
108
  .set_url(CALLOUT_URL)
109
109
  .set_method(HTTPMethod.POST)
110
110
  .set_headers({"Content-Type": "application/json", "Accept": "application/json"})
111
+ .set_response_timeout_seconds(60)
111
112
  .build()
112
113
  )
113
114
 
@@ -0,0 +1,136 @@
1
+ # Transform with a Gemini Named Credential Callout
2
+
3
+ The **transform** calls Google's **Gemini** `generateContent` API to summarize text, and
4
+ write the result back to a DLO. Gemini is reached through a **Named Credential**
5
+ (`callout:gemini`), so this code never handles the endpoint URL or the API key.
6
+
7
+ It shows **both** callout paths against the same Named Credential.
8
+
9
+ ## Shared request template
10
+
11
+ ```python
12
+ from datacustomcode.client import Client, named_credential_request_col
13
+
14
+ # URL, method and headers apply to every callout on this template.
15
+ request = (
16
+ HTTPRequestBuilder()
17
+ .set_url("callout:gemini") # callout:<NC name>[/<path>]
18
+ .set_method(HTTPMethod.POST)
19
+ .set_headers({"Content-Type": "application/json"})
20
+ .set_response_timeout_seconds(60) # optional: per-callout timeout (seconds)
21
+ .build()
22
+ )
23
+ ```
24
+
25
+ ## Driver path — one-shot on the driver
26
+
27
+ `Client.named_credential_request` runs the callout **once** on the driver and
28
+ returns an `HTTPResponse` (`.status_code`, `.body`, `.headers`, `.is_success`).
29
+ Use it for a lookup or a shared value you reuse across the job.
30
+
31
+ ```python
32
+ response = client.named_credential_request(request, body=json.dumps(payload))
33
+ if response.is_success:
34
+ envelope = json.loads(response.body)
35
+ text = envelope["candidates"][0]["content"]["parts"][0]["text"]
36
+ ```
37
+
38
+ ## Per-row path — fan out across the DataFrame
39
+
40
+ `named_credential_request_col` dispatches one callout per row; only the body
41
+ Column varies.
42
+
43
+ ```python
44
+ # One callout per row; body is a Column built from the row's data.
45
+ df = df.withColumn("_callout", named_credential_request_col(request, body=body_col))
46
+ ```
47
+
48
+ It returns a struct Column:
49
+
50
+ ```
51
+ {status, response: {status_code, body, headers}, error_code, error_message}
52
+ ```
53
+
54
+ - A non-2xx response is still `status = "SUCCESS"` with the HTTP code in
55
+ `response.status_code` — extracting the model text just yields null for that row.
56
+ - A transport failure sets `status = "ERROR"`; the row survives, the job does not
57
+ abort. Pull the model text out of `response.body` with `get_json_object(...)`.
58
+
59
+ Both paths resolve the same Named Credential and read auth from the same local
60
+ `external_callout_config.json`.
61
+
62
+ The `gemini` Named Credential's URL already includes the full
63
+ `/v1beta/models/<model>:generateContent` path, so the callout is just
64
+ `callout:gemini` with **no path suffix** (anything after the name is appended to
65
+ the credential's URL).
66
+
67
+ ## Configure the Named Credential
68
+
69
+ 1. Create an **External Credential** (e.g. `google_api_key`) that injects your
70
+ Gemini API key as the `X-goog-api-key` header.
71
+ 2. Create a **Named Credential** named `gemini`:
72
+ - **URL**: `https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent`
73
+ - **Enabled for Callouts** + **Generate Authorization Header**: on
74
+ - **External Credential**: `google_api_key`
75
+
76
+ ## Test locally
77
+
78
+ Copy `entrypoint.py` into your `payload/` folder (or point the run at it), then:
79
+
80
+ ```bash
81
+ DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG=/abs/path/to/external_callout_config.json \
82
+ sf data-code-extension script run --entrypoint entrypoint.py --target-org <your-org-alias>
83
+ ```
84
+
85
+ With `----target-org` the SDK fetches only the **URL** from the org's Named
86
+ Credential; **auth is always taken from `external_callout_config.json`** locally
87
+ (the org's External Credential is used only in the Data Cloud runtime). So the
88
+ `X-goog-api-key` must be in the local config for a local test. Omit
89
+ `--target-org` to run fully offline using `target_url`.
90
+
91
+ ```json
92
+ {
93
+ "credentials": {
94
+ "callout:gemini": {
95
+ "auth_type": "Custom",
96
+ "custom_headers": { "X-goog-api-key": "YOUR_GEMINI_API_KEY" },
97
+ "target_url": "https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent"
98
+ }
99
+ }
100
+ }
101
+ ```
102
+
103
+ Place `external_callout_config.json` in the **parent of your payload folder** (or
104
+ point `DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG` at it). It is never packaged into
105
+ the deployment zip. Get a key from [Google AI Studio](https://aistudio.google.com/apikey);
106
+ **do not commit it.**
107
+
108
+ ## What it reads / writes
109
+
110
+ `config.json` declares the DLO permissions for deployment:
111
+
112
+ | | DLO | Notes |
113
+ | ------ | ------------------- | --------------------------------------------- |
114
+ | read | `Account_std__dll` | source rows; `description__c` is summarized |
115
+ | write | `Account_std_copy__dll` | adds `summary__c`, `callout_status__c`, `callout_http_code__c` |
116
+
117
+ Adjust `_TEXT_COLUMN`, `_SOURCE_DLO` and `_TARGET_DLO` in `entrypoint.py` (and the
118
+ matching entries in `config.json`) to point at your own DLOs.
119
+
120
+ ## Auth types
121
+
122
+ `auth_type` selects how auth is injected for local testing. It should mirror the
123
+ External Credential your Named Credential uses in the org, so local and deployed
124
+ runs behave the same. This example uses `Custom` (Gemini's `X-goog-api-key`);
125
+ all four supported types:
126
+
127
+ | `auth_type` | Fields read | Header sent |
128
+ | ----------- | ------------------------------- | --------------------------------------- |
129
+ | `Basic` | `username`, `password` | `Authorization: Basic <base64 user:pw>` |
130
+ | `Custom` | `custom_headers` (sent verbatim)| the headers you list |
131
+ | `OAuth` | `access_token` or `token` | `Authorization: Bearer <token>` |
132
+ | `Jwt` | `access_token` or `token` | `Authorization: Bearer <token>` |
133
+
134
+ `OAuth`/`Jwt` take a token you supply for the local run — the SDK does not fetch
135
+ or refresh it. In the Data Cloud runtime the Named Credential handles token
136
+ acquisition; this local config only stands in for that during testing.
@@ -0,0 +1,144 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2025, Salesforce, Inc.
3
+ # SPDX-License-Identifier: Apache-2
4
+
5
+ """
6
+ Data Transform with a Gemini Named Credential Callout
7
+
8
+ This is a batch transform: read a DLO, enrich it via Google's Gemini ``generateContent``
9
+ API, and write the result back to a DLO.
10
+
11
+ It shows **both** callout paths against the same Named Credential
12
+ (``callout:gemini``); the request template (URL, method, headers) is shared and
13
+ only the body differs:
14
+
15
+ - **Driver path** — :meth:`Client.named_credential_request` runs **once** on the
16
+ driver and returns an :class:`HTTPResponse`. Use it for a single job-level
17
+ callout whose result you reuse across the job.
18
+ - **Per-row path** — :func:`datacustomcode.client.named_credential_request_col`
19
+ fans the callout out across the DataFrame, one call per row, returning a struct
20
+ Column ``{status, response, error_code, error_message}`` where ``response`` is
21
+ itself ``{status_code, body, headers}``. A non-2xx response is still
22
+ ``SUCCESS`` with its code in ``response.status_code``; a transport failure sets
23
+ ``status`` to ``ERROR`` without aborting the whole Spark job.
24
+ """
25
+
26
+ import json
27
+ import logging
28
+
29
+ from pyspark.sql import Column
30
+ from pyspark.sql.functions import (
31
+ array,
32
+ col,
33
+ concat,
34
+ get_json_object,
35
+ lit,
36
+ struct,
37
+ to_json,
38
+ )
39
+
40
+ from datacustomcode.client import Client, named_credential_request_col
41
+ from datacustomcode.io.writer.base import WriteMode
42
+ from datacustomcode.named_credential.types.http_method import HTTPMethod
43
+ from datacustomcode.named_credential.types.http_request_builder import (
44
+ HTTPRequestBuilder,
45
+ )
46
+
47
+ logger = logging.getLogger(__name__)
48
+ logging.basicConfig(level=logging.INFO)
49
+
50
+ CALLOUT_URL = "callout:gemini"
51
+
52
+ _TEXT_COLUMN = "Description__c"
53
+ _SOURCE_DLO = "Account_std__dll"
54
+ _TARGET_DLO = "Account_std_copy__dll"
55
+
56
+ _PROMPT = (
57
+ "Summarize the following text in one sentence. Respond with the summary "
58
+ "only, no preamble.\n\nText:\n"
59
+ )
60
+
61
+ _REQUEST = (
62
+ HTTPRequestBuilder()
63
+ .set_url(CALLOUT_URL)
64
+ .set_method(HTTPMethod.POST)
65
+ .set_headers({"Content-Type": "application/json", "Accept": "application/json"})
66
+ .set_response_timeout_seconds(60)
67
+ .build()
68
+ )
69
+
70
+
71
+ def _gemini_body_col(text_col: Column) -> Column:
72
+ """Build a per-row Gemini ``generateContent`` request body as a JSON string.
73
+
74
+ Using ``to_json(struct(...))`` keeps the row text properly escaped inside the
75
+ JSON payload rather than string-concatenating it.
76
+ """
77
+ prompt = concat(lit(_PROMPT), text_col)
78
+ contents = array(struct(array(struct(prompt.alias("text"))).alias("parts")))
79
+ return to_json(struct(contents.alias("contents")))
80
+
81
+
82
+ def _gemini_body(text: str) -> str:
83
+ """Build a Gemini ``generateContent`` request body as a JSON string (driver)."""
84
+ return json.dumps({"contents": [{"parts": [{"text": _PROMPT + text}]}]})
85
+
86
+
87
+ def _summarize_on_driver(client: Client, text: str) -> str:
88
+ """One-shot driver callout: summarize a single string once, not per row.
89
+
90
+ The scalar counterpart to the per-row column path — same request template,
91
+ but dispatched once on the driver and returning an ``HTTPResponse``.
92
+ """
93
+ response = client.named_credential_request(_REQUEST, body=_gemini_body(text))
94
+
95
+ # Don't raise: a failed driver callout shouldn't abort the whole job.
96
+ if not response.is_success:
97
+ logger.error(f"Driver Gemini callout failed: HTTP {response.status_code}")
98
+ return ""
99
+
100
+ envelope = json.loads(response.body) if response.body else {}
101
+ try:
102
+ return str(envelope["candidates"][0]["content"]["parts"][0]["text"])
103
+ except (KeyError, IndexError, TypeError):
104
+ return ""
105
+
106
+
107
+ def main():
108
+ client = Client()
109
+
110
+ df = client.read_dlo(_SOURCE_DLO)
111
+
112
+ # Driver path: one callout on the driver over a single representative row.
113
+ sample = df.select(_TEXT_COLUMN).first()
114
+ if sample and sample[0]:
115
+ driver_summary = _summarize_on_driver(client, sample[0])
116
+ logger.info(f"Driver-path sample summary: {driver_summary}")
117
+
118
+ # Per-row path: one Gemini callout per row; the result struct is a column.
119
+ callout = named_credential_request_col(
120
+ _REQUEST, body=_gemini_body_col(col(_TEXT_COLUMN))
121
+ )
122
+ df = df.withColumn("_callout", callout)
123
+
124
+ # Pull the model's text out of the response body. A row whose callout failed
125
+ # (non-2xx or transport error) yields null here rather than failing the job.
126
+ summary = get_json_object(
127
+ col("_callout")["response"]["body"],
128
+ "$.candidates[0].content.parts[0].text",
129
+ )
130
+
131
+ df = df.select(
132
+ col("id__c").alias("id__c"),
133
+ col("Description__c").alias("description__c"),
134
+ col("kq_id__c").alias("kq_id__c"),
135
+ summary.alias("summary__c"),
136
+ col("_callout")["status"].alias("callout_status__c"),
137
+ col("_callout")["response"]["status_code"].alias("callout_http_code__c"),
138
+ )
139
+
140
+ client.write_to_dlo(_TARGET_DLO, df, write_mode=WriteMode.APPEND)
141
+
142
+
143
+ if __name__ == "__main__":
144
+ main()
@@ -0,0 +1,11 @@
1
+ {
2
+ "credentials": {
3
+ "callout:gemini": {
4
+ "auth_type": "Custom",
5
+ "custom_headers": {
6
+ "X-goog-api-key": "YOUR_API_KEY"
7
+ },
8
+ "target_url": "https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent"
9
+ }
10
+ }
11
+ }
@@ -3,46 +3,59 @@
3
3
  This example is the streaming counterpart to a normal batch entrypoint. Instead
4
4
  of a batch ``Client`` with ``read_dlo`` / ``write_to_dlo`` (which read and write
5
5
  a bounded snapshot), it uses a :class:`StreamingClient` and its streaming delta
6
- methods:
6
+ methods.
7
7
 
8
- * ``client.read_dlo_deltas()`` returns a *streaming* DataFrame over the
9
- Change Data Feed of the source DLO. Each row carries the source columns plus
10
- change-feed metadata columns (``_record_type``, ``_commit_*``).
11
- * ``client.write_dlo_deltas(name, df)`` starts a streaming query that writes
12
- each micro-batch to the target DLO and returns the ``StreamingQuery`` handle.
13
- The runtime owns the trigger, and checkpoint location — the caller only
14
- chooses the table.
8
+ The first run of a streaming job will use the run mode INITIAL_SYNC which behaves
9
+ like a batch run on the streaming source. A streaming transform can also use run
10
+ mode REBUILD to do the same thing on demand. Note that these will process all
11
+ source rows and overwrite the target.
15
12
 
16
13
  The transform in between is ordinary PySpark. Because the source is a change
17
14
  feed, keep the metadata columns on the DataFrame you hand to
18
15
  ``write_dlo_deltas`` — the sink relies on them to merge changes correctly.
19
16
 
20
- This entrypoint only runs inside the Data Cloud streaming (``DELTA_SYNC``)
21
- runtime; the local ``datacustomcode run`` readers/writers raise
17
+ This entrypoint only runs inside the Data Cloud runtime;
18
+ the local ``datacustomcode run`` readers/writers raise
22
19
  ``NotImplementedError`` for the delta methods.
23
20
  """
24
21
 
22
+ from pyspark.sql import DataFrame
25
23
  from pyspark.sql.functions import col, upper
26
24
 
27
- from datacustomcode.client import StreamingClient
25
+ from datacustomcode.client import (
26
+ RunMode,
27
+ StreamingClient,
28
+ get_run_mode,
29
+ )
28
30
 
29
31
 
30
32
  def main():
33
+ target_dlo = "Account_std_copy__dll"
31
34
  client = StreamingClient()
32
35
 
33
- # Streaming DataFrame over the source DLO's change feed.
34
- deltas = client.read_dlo_deltas()
35
-
36
- # Ordinary PySpark transform.
37
- transformed = deltas.withColumn("description__c", upper(col("description__c")))
38
-
39
- # Start the streaming write. write_dlo_deltas returns the StreamingQuery;
40
- # the trigger and checkpoint location are provided by the runtime.
41
- query = client.write_dlo_deltas("Account_std_copy__dll", transformed)
42
-
43
- # Drive the query's lifecycle. In the streaming runtime this blocks until
44
- # the job is stopped by the platform.
45
- query.awaitTermination()
36
+ if get_run_mode() == RunMode.DELTA_SYNC:
37
+ # Streaming DataFrame over the source DLO's change feed.
38
+ dataframe = client.read_dlo_deltas()
39
+ # Ordinary PySpark transform.
40
+ transformed = transform(dataframe)
41
+
42
+ # Start the streaming write. write_dlo_deltas returns the StreamingQuery;
43
+ # the trigger and checkpoint location are provided by the runtime.
44
+ query = client.write_dlo_deltas(target_dlo, transformed)
45
+
46
+ # Drive the query's lifecycle. In the streaming runtime this blocks until
47
+ # the job is stopped by the platform.
48
+ query.awaitTermination()
49
+ else:
50
+ # initial sync and rebuild read the entire streaming source DLO and
51
+ # write using a server-decided mode based on the run mode
52
+ dataframe = client.read_dlo()
53
+ transformed = transform(dataframe)
54
+ client.auto_write_to_dlo(target_dlo, transformed)
55
+
56
+
57
+ def transform(dataframe: DataFrame) -> DataFrame:
58
+ return dataframe.withColumn("description__c", upper(col("description__c")))
46
59
 
47
60
 
48
61
  if __name__ == "__main__":
@@ -45,13 +45,24 @@ check_docker() {
45
45
  echo "Docker daemon is running"
46
46
  }
47
47
 
48
+ # Function to check if openssl is installed
49
+ check_openssl() {
50
+ if ! command -v openssl &> /dev/null; then
51
+ echo "openssl is not installed. It is required to generate a secure JupyterLab access token."
52
+ exit 1
53
+ fi
54
+ }
55
+
48
56
  # Function to start Jupyter server
49
57
  start_jupyter() {
50
58
  echo "Building the docker image"
51
59
  docker build -t datacloud-customcode .
52
60
 
61
+ local TOKEN
62
+ TOKEN=$(openssl rand -hex 32)
63
+
53
64
  echo "Running the docker container"
54
- docker run -d --rm -p 8888:8888 \
65
+ docker run -d --rm -p 127.0.0.1:8888:8888 \
55
66
  -v $(pwd):/workspace \
56
67
  --name jupyter-server \
57
68
  datacloud-customcode jupyter lab \
@@ -59,12 +70,14 @@ start_jupyter() {
59
70
  --port=8888 \
60
71
  --no-browser \
61
72
  --allow-root \
62
- --NotebookApp.token='' \
63
- --NotebookApp.password='' \
73
+ --NotebookApp.token="$TOKEN" \
64
74
  --notebook-dir=/workspace
65
75
 
66
76
  sleep 3 # Wait for server to start
67
- open_browser "http://localhost:8888"
77
+ local URL
78
+ URL="http://localhost:8888/?token=$TOKEN"
79
+ echo "Opening $URL"
80
+ open_browser $URL
68
81
  }
69
82
 
70
83
  # Function to stop Jupyter server
@@ -82,6 +95,7 @@ stop_jupyter() {
82
95
  case "$1" in
83
96
  "start")
84
97
  check_docker
98
+ check_openssl
85
99
  start_jupyter
86
100
  ;;
87
101
  "stop")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: salesforce-data-customcode
3
- Version: 6.1.0.dev4
3
+ Version: 6.1.0.dev6
4
4
  Summary: Data Cloud Custom Code SDK
5
5
  License-Expression: Apache-2.0
6
6
  License-File: LICENSE.txt
@@ -534,7 +534,7 @@ exploration. Instead of running an entire script, one can run one code cell at
534
534
 
535
535
  You can read more about Jupyter Notebooks here: https://jupyter.org/
536
536
 
537
- 1. Within the root project of your package folder, run `./jupyterlab.sh start`
537
+ 1. Within the root project of your package folder, run `./jupyterlab.sh start`. This prints an access token and opens an already-authenticated JupyterLab session in your browser. If the browser doesn't open automatically, copy the printed `http://localhost:8888/?token=...` URL into your browser.
538
538
  1. Double-click on "account.ipynb" file, which provides a starting point for a notebook
539
539
  1. Use shift+enter to execute each cell within the notebook. Add/edit/delete cells of code as needed for your data exploration.
540
540
  1. Don't forget to run `./jupyterlab.sh stop` to stop the docker container
@@ -625,5 +625,5 @@ If you're using OAuth Tokens authentication, the initial configure will retrieve
625
625
  ## Other docs
626
626
 
627
627
  - [Troubleshooting](./docs/troubleshooting.md)
628
- - [For Contributors](./FOR_CONTRIBUTORS.md)
628
+ - [Contributing](./CONTRIBUTING.md)
629
629
 
@@ -1,7 +1,7 @@
1
1
  datacustomcode/__init__.py,sha256=5vvWRk5gb2mZIaVZ9BSPRL45uKSUwmSerxhLp6A4Q-4,2815
2
2
  datacustomcode/auth.py,sha256=fpSjhIBdv9trC8yq2vuljAix_Euu-4Ah7HDCGhYjOxI,8309
3
3
  datacustomcode/cli.py,sha256=K4LHL5xrb7-erXbjYz7pT4Kg4jG6Wf81c9wl_8tLabE,14162
4
- datacustomcode/client.py,sha256=0uw9qptultCYT-fZ3cXwaPvIFW5J9E79nD46tnu4NUE,28743
4
+ datacustomcode/client.py,sha256=UoOqBBLGVBhznTp5eZR8aqA8bDkYTuZz04n_A0R9j_A,30905
5
5
  datacustomcode/cmd.py,sha256=ZMs46aydJw2EaU26JgCtZmnqESQFHvvaJz10hnjZTBk,3537
6
6
  datacustomcode/common_config.py,sha256=SAUnxj3kqmOeWwPmFoYq4tuxokMgURVm4QwcGi-avL4,1928
7
7
  datacustomcode/config.py,sha256=lqed3jcWfoIweikSFZelaAoyBa0cXXScL8O6vaqdeRU,4372
@@ -37,9 +37,9 @@ datacustomcode/io/reader/query_api.py,sha256=eVrohrcnTnhSMsGfPRHu5XltjFtGG0hyHt1
37
37
  datacustomcode/io/reader/sf_cli.py,sha256=x5QacVqRZaZSyph1_wwxo67s8r29wsOcUsOaiF9cDWE,6324
38
38
  datacustomcode/io/reader/utils.py,sha256=HlHhPZoHfmWA3mF8kTnd3m4Nd2exaz4NsOJJfQY7Pew,1656
39
39
  datacustomcode/io/writer/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
40
- datacustomcode/io/writer/base.py,sha256=LtyOtkcEsX3oQZ-2zWFeNutVae1T3ialbGbvwl-7spU,3246
41
- datacustomcode/io/writer/csv.py,sha256=asty5teBpNQ1fMGHZ7wA3suLhq0sk0lVQPN5x7TOSRk,1555
42
- datacustomcode/io/writer/print.py,sha256=0g2sP_1Wb95UIyEELWJt3dqM18Iv_CWhDW3mDxcIhns,5105
40
+ datacustomcode/io/writer/base.py,sha256=W0QmsmszCjOIYPlEqbAjld0O855CAFvZjm_FT7szyb4,3806
41
+ datacustomcode/io/writer/csv.py,sha256=LYQpTZlIQdn0RRx_dRah22V2Vgm-InKP4KHRKfWGiRY,1963
42
+ datacustomcode/io/writer/print.py,sha256=Tpy5-0NmJ_renC6uah5TdHQZFzDbx23ltm1tIPK0978,5388
43
43
  datacustomcode/llm_gateway/__init__.py,sha256=rcoGpSkv36tZuAOcTxKkhNpM0qV03qgmer_ej07Vmfc,1088
44
44
  datacustomcode/llm_gateway/base.py,sha256=CTUhZiZ0JFlmWj6kQpTR4_-mUvqkBKl_Fkdvv9VwxY8,1262
45
45
  datacustomcode/llm_gateway/default.py,sha256=QiwrDaUvrh00I-fYYhmogTtqC5r0fmm-JrXxNQLvTc0,1854
@@ -67,7 +67,7 @@ datacustomcode/named_credential/spark_default.py,sha256=DY_STX9-1le8cS5M_vGqB3VU
67
67
  datacustomcode/named_credential/types/__init__.py,sha256=gamfOD1VnAtEslRBpqh-yKiVjkG_wYWYdSatGIfsN-w,621
68
68
  datacustomcode/named_credential/types/http_method.py,sha256=NFf8emmW_Y73q8q2pseWHqOQ-1SoFwKp2vCMr1OdmEU,952
69
69
  datacustomcode/named_credential/types/http_request.py,sha256=LXW4EcvUfyOJHf8-VtFSRWy6JsTJJsd0N4-X6WxxFnA,2170
70
- datacustomcode/named_credential/types/http_request_builder.py,sha256=qdToSlX2xw3FzuLF6Kas3VzCp1ubdXZb4_WkjQ5fBY0,1862
70
+ datacustomcode/named_credential/types/http_request_builder.py,sha256=O2rcqHiiKwaqQQRWC3BQ054WMwVfRmGTnX4lOfc7xeA,3117
71
71
  datacustomcode/named_credential/types/http_response.py,sha256=ZqX52RMjSdn8OpHSDkKJo9730QRx8o4_ZT-1noSMRe8,1480
72
72
  datacustomcode/named_credential/types/http_response_builder.py,sha256=MoUjpT-P5W2uT1-h1rIfKpBQHpMj_ZA_nUmn-FyBdbQ,896
73
73
  datacustomcode/named_credential_config.py,sha256=M07-oWzYj0KTrfA0fqvk3BQvaaDaw7C6nxbR0s3lOUk,3645
@@ -86,9 +86,9 @@ datacustomcode/templates/function/build_native_dependencies.sh,sha256=dk9vhzlObd
86
86
  datacustomcode/templates/function/chunking/payload/config.json,sha256=RBNvo1WzZ4oRRq0W9-hknpT7T8If536DEMBg9hyq_4o,2
87
87
  datacustomcode/templates/function/chunking/payload/entrypoint.py,sha256=tzpQJjODulwzLilOoKdOo8GjpdIlCDByjnawpHIQYQA,4517
88
88
  datacustomcode/templates/function/chunking/requirements.txt,sha256=Ih0KsVdiTdo7KuIsqFB_AJMRT2r0ao_xDEb3hiDmf48,46
89
- datacustomcode/templates/function/example/chunking_with_external_callout/README.md,sha256=RIgoUysFniAE4lC5LiZl5u55Rro1w5gmM2mMnsQF_HI,4716
89
+ datacustomcode/templates/function/example/chunking_with_external_callout/README.md,sha256=iXegln2W0FlGrgWHi1StBSIAxzQHwKv_l4DL7PmoHdA,4796
90
90
  datacustomcode/templates/function/example/chunking_with_external_callout/config.json,sha256=EvjuOVssRg2SjcUb5F50RRMGDqag_f69bv8AOflaK1g,38
91
- datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py,sha256=QqJLeJHgtW4RkDwTaUIsb2ng_ODwwClXpn5F7XyXBes,5296
91
+ datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py,sha256=TwT7TKMV7Shg5inkN-B4U2wSXGXAJYZZh8wrXyePRAo,5338
92
92
  datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json,sha256=PUCmw_humbVoCyswFneuGYx0INbqSTPOumDqtGOztiM,320
93
93
  datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json,sha256=ZqUthCT9F9HoyT493tXg4nFDSbEAeaFHY1iHo-9Qx7g,2703
94
94
  datacustomcode/templates/function/example/chunking_with_llm/config.json,sha256=EvjuOVssRg2SjcUb5F50RRMGDqag_f69bv8AOflaK1g,38
@@ -111,16 +111,19 @@ datacustomcode/templates/script/account.ipynb,sha256=LIbxgiVxflNASdspF2lfpMKkKAT
111
111
  datacustomcode/templates/script/build_native_dependencies.sh,sha256=ICRrp4f1ATBwPaUiuVGjY-MwiubFswNJLf8gMGU6YNg,197
112
112
  datacustomcode/templates/script/examples/employee_hierarchy/employee_data.csv,sha256=C7ggLBfoyi3M2BdMLNyOeKqF-5OO-Da76lkwgWg2cVQ,302
113
113
  datacustomcode/templates/script/examples/employee_hierarchy/entrypoint.py,sha256=Mfm3iQtEHTQRW6cmZTwRKdTh3IeGWR-teXrd6x-YLSY,2223
114
- datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py,sha256=1PpYyZck2j9IRTnoI-kfHbMf21vN8gA49_-qJr3rDjk,1984
115
- datacustomcode/templates/script/jupyterlab.sh,sha256=IHR3YQ8d_busuyvesByhJgzCKULIFPI-Hqogs0NPhCs,2432
114
+ datacustomcode/templates/script/examples/external_callout/README.md,sha256=9pcepHizQxsTduNUjDF4GkkB8nM4dke9353mJxCZzjM,5646
115
+ datacustomcode/templates/script/examples/external_callout/entrypoint.py,sha256=KgSAmkpJ1On7R3Lr4_Slf9rTHgLr3TjC43QSFbaIzUY,5005
116
+ datacustomcode/templates/script/examples/external_callout/external_callout_config.json,sha256=PUCmw_humbVoCyswFneuGYx0INbqSTPOumDqtGOztiM,320
117
+ datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py,sha256=UQr1EhJURFFGI5626efvv-MxGHltzcpJcEJVmz5ZQdo,2333
118
+ datacustomcode/templates/script/jupyterlab.sh,sha256=0O_rPZ8w2DM1hxPo9Wig9j-ymhS2dRzVO2PlFClWdlA,2786
116
119
  datacustomcode/templates/script/payload/config.json,sha256=0d2mEMt4NHIeOUTsgiPMuRKdOLIQxmh-Wq-IfEqU-gc,28
117
120
  datacustomcode/templates/script/payload/entrypoint.py,sha256=4Uph0ILa5ukk_6C90paK3uzQn0zZvNLr9ho3o_ZwErQ,2474
118
121
  datacustomcode/templates/script/requirements-dev.txt,sha256=OWwuy1awesqOZOS3Zizmdd6BDnSePpGFN5bq5IrALvc,176
119
122
  datacustomcode/templates/script/requirements.txt,sha256=qJbqzs5z0Hrx590U3T7dHq_m8UQEx2TdhgrJSz-QHIQ,40
120
123
  datacustomcode/token_provider.py,sha256=qA_e4vqSXnxn0MccFFPULWO8ZeUOEAu5QpBnoYCg-fc,6839
121
124
  datacustomcode/version.py,sha256=9LlbVrzwBvut1L308QbwNBgxdQk5CrghH39JARVSxms,989
122
- salesforce_data_customcode-6.1.0.dev4.dist-info/METADATA,sha256=Z3pzLUNQb2iRxImuljMvwdpumcV5x8wJEVtySG545pc,27207
123
- salesforce_data_customcode-6.1.0.dev4.dist-info/WHEEL,sha256=eY7nduwzv-ldUxpzbRlxwvC693Hg6PX8bWDjEHjZ_dk,88
124
- salesforce_data_customcode-6.1.0.dev4.dist-info/entry_points.txt,sha256=WpQ94UB7UuRCYGOLtJV3vAgm1tl7iVf43VYRWIuNu5Y,74
125
- salesforce_data_customcode-6.1.0.dev4.dist-info/licenses/LICENSE.txt,sha256=iOi8EmQpfkFhMENi7VYtkl2EqS14pEOccoXHiW2dyPU,11443
126
- salesforce_data_customcode-6.1.0.dev4.dist-info/RECORD,,
125
+ salesforce_data_customcode-6.1.0.dev6.dist-info/METADATA,sha256=-7Xy6AdRhwyLbyCMCZqHQ8gbnuvdXUz744Ql0ZCE2Tw,27417
126
+ salesforce_data_customcode-6.1.0.dev6.dist-info/WHEEL,sha256=eY7nduwzv-ldUxpzbRlxwvC693Hg6PX8bWDjEHjZ_dk,88
127
+ salesforce_data_customcode-6.1.0.dev6.dist-info/entry_points.txt,sha256=WpQ94UB7UuRCYGOLtJV3vAgm1tl7iVf43VYRWIuNu5Y,74
128
+ salesforce_data_customcode-6.1.0.dev6.dist-info/licenses/LICENSE.txt,sha256=iOi8EmQpfkFhMENi7VYtkl2EqS14pEOccoXHiW2dyPU,11443
129
+ salesforce_data_customcode-6.1.0.dev6.dist-info/RECORD,,