salesforce-data-customcode 6.1.0.dev1__py3-none-any.whl → 6.1.0.dev2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datacustomcode/__init__.py +0 -5
- datacustomcode/cli.py +4 -29
- datacustomcode/client.py +192 -211
- datacustomcode/config.py +0 -5
- datacustomcode/config.yaml +6 -0
- datacustomcode/constants.py +0 -8
- datacustomcode/deploy.py +23 -57
- datacustomcode/function/runtime.py +16 -0
- datacustomcode/io/reader/base.py +0 -42
- datacustomcode/io/writer/base.py +0 -33
- datacustomcode/named_credential/__init__.py +26 -0
- datacustomcode/named_credential/base.py +54 -0
- datacustomcode/named_credential/default.py +93 -0
- datacustomcode/named_credential/direct/__init__.py +19 -0
- datacustomcode/named_credential/direct/auth.py +63 -0
- datacustomcode/named_credential/direct/credentials.py +121 -0
- datacustomcode/named_credential/direct/transport.py +110 -0
- datacustomcode/named_credential/direct/url_resolver.py +112 -0
- datacustomcode/named_credential/spark_base.py +93 -0
- datacustomcode/named_credential/spark_default.py +154 -0
- datacustomcode/named_credential/types/__init__.py +14 -0
- datacustomcode/named_credential/types/http_method.py +29 -0
- datacustomcode/named_credential/types/http_request.py +63 -0
- datacustomcode/named_credential/types/http_request_builder.py +55 -0
- datacustomcode/named_credential/types/http_response.py +43 -0
- datacustomcode/named_credential/types/http_response_builder.py +24 -0
- datacustomcode/named_credential_config.py +105 -0
- datacustomcode/run.py +7 -11
- datacustomcode/scan.py +29 -164
- datacustomcode/template.py +1 -13
- datacustomcode/templates/function/example/chunking_with_external_callout/README.md +119 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/config.json +3 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +161 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +11 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +16 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/METADATA +3 -40
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/RECORD +40 -19
- datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +0 -49
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/WHEEL +0 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/entry_points.txt +0 -0
- {salesforce_data_customcode-6.1.0.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from typing import Any, Dict
|
|
17
|
+
|
|
18
|
+
from datacustomcode.named_credential.types.http_response import HTTPResponse
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class HTTPResponseBuilder:
|
|
22
|
+
@staticmethod
|
|
23
|
+
def build(response_dict: Dict[str, Any]) -> HTTPResponse:
|
|
24
|
+
return HTTPResponse.model_validate(response_dict)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from typing import (
|
|
17
|
+
ClassVar,
|
|
18
|
+
Generic,
|
|
19
|
+
Type,
|
|
20
|
+
TypeVar,
|
|
21
|
+
Union,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
from datacustomcode.common_config import (
|
|
25
|
+
BaseConfig,
|
|
26
|
+
BaseObjectConfig,
|
|
27
|
+
default_config_file,
|
|
28
|
+
)
|
|
29
|
+
from datacustomcode.named_credential.base import NamedCredential
|
|
30
|
+
from datacustomcode.named_credential.spark_base import SparkNamedCredential
|
|
31
|
+
|
|
32
|
+
_N = TypeVar("_N", bound=NamedCredential)
|
|
33
|
+
_S = TypeVar("_S", bound=SparkNamedCredential)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class NamedCredentialObjectConfig(BaseObjectConfig, Generic[_N]):
|
|
37
|
+
type_to_create: ClassVar[Type[NamedCredential]] = NamedCredential # type: ignore[type-abstract]
|
|
38
|
+
|
|
39
|
+
def to_object(self) -> NamedCredential:
|
|
40
|
+
type_ = self.type_to_create.subclass_from_config_name(self.type_config_name)
|
|
41
|
+
return type_(**self.options)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class NamedCredentialConfig(BaseConfig):
|
|
45
|
+
named_credential_config: Union[
|
|
46
|
+
NamedCredentialObjectConfig[NamedCredential], None
|
|
47
|
+
] = None
|
|
48
|
+
|
|
49
|
+
def update(self, other: "NamedCredentialConfig") -> "NamedCredentialConfig":
|
|
50
|
+
def merge(
|
|
51
|
+
config_a: Union[NamedCredentialObjectConfig, None],
|
|
52
|
+
config_b: Union[NamedCredentialObjectConfig, None],
|
|
53
|
+
) -> Union[NamedCredentialObjectConfig, None]:
|
|
54
|
+
if config_a is not None and config_a.force:
|
|
55
|
+
return config_a
|
|
56
|
+
if config_b:
|
|
57
|
+
return config_b
|
|
58
|
+
return config_a
|
|
59
|
+
|
|
60
|
+
self.named_credential_config = merge(
|
|
61
|
+
self.named_credential_config, other.named_credential_config
|
|
62
|
+
)
|
|
63
|
+
return self
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class SparkNamedCredentialObjectConfig(BaseObjectConfig, Generic[_S]):
|
|
67
|
+
type_to_create: ClassVar[Type[SparkNamedCredential]] = SparkNamedCredential # type: ignore[type-abstract]
|
|
68
|
+
|
|
69
|
+
def to_object(self) -> SparkNamedCredential:
|
|
70
|
+
type_ = self.type_to_create.subclass_from_config_name(self.type_config_name)
|
|
71
|
+
return type_(**self.options)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class SparkNamedCredentialConfig(BaseConfig):
|
|
75
|
+
spark_named_credential_config: Union[
|
|
76
|
+
SparkNamedCredentialObjectConfig[SparkNamedCredential], None
|
|
77
|
+
] = None
|
|
78
|
+
|
|
79
|
+
def update(
|
|
80
|
+
self, other: "SparkNamedCredentialConfig"
|
|
81
|
+
) -> "SparkNamedCredentialConfig":
|
|
82
|
+
def merge(
|
|
83
|
+
config_a: Union[SparkNamedCredentialObjectConfig, None],
|
|
84
|
+
config_b: Union[SparkNamedCredentialObjectConfig, None],
|
|
85
|
+
) -> Union[SparkNamedCredentialObjectConfig, None]:
|
|
86
|
+
if config_a is not None and config_a.force:
|
|
87
|
+
return config_a
|
|
88
|
+
if config_b:
|
|
89
|
+
return config_b
|
|
90
|
+
return config_a
|
|
91
|
+
|
|
92
|
+
self.spark_named_credential_config = merge(
|
|
93
|
+
self.spark_named_credential_config, other.spark_named_credential_config
|
|
94
|
+
)
|
|
95
|
+
return self
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# Global Named Credential config instance
|
|
99
|
+
named_credential_config = NamedCredentialConfig()
|
|
100
|
+
named_credential_config.load(default_config_file())
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
# Global Spark Named Credential config instance
|
|
104
|
+
spark_named_credential_config = SparkNamedCredentialConfig()
|
|
105
|
+
spark_named_credential_config.load(default_config_file())
|
datacustomcode/run.py
CHANGED
|
@@ -27,6 +27,7 @@ from typing import (
|
|
|
27
27
|
from datacustomcode.config import config
|
|
28
28
|
from datacustomcode.einstein_predictions_config import einstein_predictions_config
|
|
29
29
|
from datacustomcode.llm_gateway_config import llm_gateway_config
|
|
30
|
+
from datacustomcode.named_credential_config import named_credential_config
|
|
30
31
|
from datacustomcode.scan import find_base_directory, get_package_type
|
|
31
32
|
|
|
32
33
|
|
|
@@ -42,15 +43,6 @@ def _set_config_option(config_obj, key: str, value: Optional[str]) -> None:
|
|
|
42
43
|
config_obj.options[key] = value
|
|
43
44
|
|
|
44
45
|
|
|
45
|
-
def _read_streaming_source(config_json: dict) -> Optional[str]:
|
|
46
|
-
"""Return the streaming source name from config.json's ``streamingSource``."""
|
|
47
|
-
source = config_json.get("streamingSource")
|
|
48
|
-
if not isinstance(source, dict):
|
|
49
|
-
return None
|
|
50
|
-
name = source.get("name")
|
|
51
|
-
return str(name) if name else None
|
|
52
|
-
|
|
53
|
-
|
|
54
46
|
def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
|
|
55
47
|
if sf_cli_org:
|
|
56
48
|
config_key = "sf_cli_org"
|
|
@@ -64,6 +56,9 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
|
|
|
64
56
|
_set_config_option(
|
|
65
57
|
llm_gateway_config.llm_gateway_config, config_key, sf_cli_org
|
|
66
58
|
)
|
|
59
|
+
_set_config_option(
|
|
60
|
+
named_credential_config.named_credential_config, config_key, sf_cli_org
|
|
61
|
+
)
|
|
67
62
|
elif profile != "default":
|
|
68
63
|
config_key = "credentials_profile"
|
|
69
64
|
_set_config_option(config.reader_config, config_key, profile)
|
|
@@ -72,6 +67,9 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
|
|
|
72
67
|
einstein_predictions_config.einstein_predictions_config, config_key, profile
|
|
73
68
|
)
|
|
74
69
|
_set_config_option(llm_gateway_config.llm_gateway_config, config_key, profile)
|
|
70
|
+
_set_config_option(
|
|
71
|
+
named_credential_config.named_credential_config, config_key, profile
|
|
72
|
+
)
|
|
75
73
|
|
|
76
74
|
|
|
77
75
|
def run_entrypoint(
|
|
@@ -134,8 +132,6 @@ def run_entrypoint(
|
|
|
134
132
|
_set_config_option(config.reader_config, "dataspace", dataspace)
|
|
135
133
|
_set_config_option(config.writer_config, "dataspace", dataspace)
|
|
136
134
|
|
|
137
|
-
config.streaming_source = _read_streaming_source(config_json)
|
|
138
|
-
|
|
139
135
|
_update_config_options(profile, sf_cli_org)
|
|
140
136
|
|
|
141
137
|
for dependency in dependencies:
|
datacustomcode/scan.py
CHANGED
|
@@ -15,7 +15,6 @@
|
|
|
15
15
|
from __future__ import annotations
|
|
16
16
|
|
|
17
17
|
import ast
|
|
18
|
-
import copy
|
|
19
18
|
import json
|
|
20
19
|
import os
|
|
21
20
|
import sys
|
|
@@ -33,8 +32,6 @@ import pydantic
|
|
|
33
32
|
from datacustomcode.version import get_version
|
|
34
33
|
|
|
35
34
|
DATA_ACCESS_METHODS = ["read_dlo", "read_dmo", "write_to_dlo", "write_to_dmo"]
|
|
36
|
-
STREAMING_READ_METHODS = ["read_dlo_deltas", "read_dmo_deltas"]
|
|
37
|
-
STREAMING_WRITE_METHODS = ["write_dlo_deltas"]
|
|
38
35
|
|
|
39
36
|
DATA_TRANSFORM_CONFIG_TEMPLATE = {
|
|
40
37
|
"sdkVersion": get_version(),
|
|
@@ -46,20 +43,6 @@ DATA_TRANSFORM_CONFIG_TEMPLATE = {
|
|
|
46
43
|
},
|
|
47
44
|
}
|
|
48
45
|
|
|
49
|
-
STREAMING_TRANSFORM_CONFIG_TEMPLATE = {
|
|
50
|
-
"sdkVersion": get_version(),
|
|
51
|
-
"entryPoint": "",
|
|
52
|
-
"dataspace": "default",
|
|
53
|
-
"streamingSource": {
|
|
54
|
-
"type": "dlo",
|
|
55
|
-
"name": "",
|
|
56
|
-
},
|
|
57
|
-
"permissions": {
|
|
58
|
-
"read": {},
|
|
59
|
-
"write": {},
|
|
60
|
-
},
|
|
61
|
-
}
|
|
62
|
-
|
|
63
46
|
FUNCTION_CONFIG_TEMPLATE = {
|
|
64
47
|
"entryPoint": "",
|
|
65
48
|
}
|
|
@@ -177,35 +160,6 @@ class DataAccessLayerCalls(pydantic.BaseModel):
|
|
|
177
160
|
return next(iter(self.write_to_dmo))
|
|
178
161
|
|
|
179
162
|
|
|
180
|
-
class StreamingDataAccessLayerCalls(pydantic.BaseModel):
|
|
181
|
-
read_dlo_deltas: bool
|
|
182
|
-
read_dmo_deltas: bool
|
|
183
|
-
write_dlo_deltas: frozenset[str]
|
|
184
|
-
|
|
185
|
-
@pydantic.model_validator(mode="after")
|
|
186
|
-
def validate_access_layer(self) -> StreamingDataAccessLayerCalls:
|
|
187
|
-
if self.read_dlo_deltas and self.read_dmo_deltas:
|
|
188
|
-
raise ValueError(
|
|
189
|
-
"Cannot read DLO and DMO deltas in the same streaming transform."
|
|
190
|
-
)
|
|
191
|
-
if not self.read_dlo_deltas and not self.read_dmo_deltas:
|
|
192
|
-
raise ValueError(
|
|
193
|
-
"A streaming transform must read from at least one DLO or DMO "
|
|
194
|
-
"delta stream (read_dlo_deltas / read_dmo_deltas)."
|
|
195
|
-
)
|
|
196
|
-
if not self.write_dlo_deltas:
|
|
197
|
-
raise ValueError(
|
|
198
|
-
"A streaming transform must write to at least one DLO via "
|
|
199
|
-
"write_dlo_deltas."
|
|
200
|
-
)
|
|
201
|
-
return self
|
|
202
|
-
|
|
203
|
-
@property
|
|
204
|
-
def read_layer(self) -> str:
|
|
205
|
-
"""Return the read source layer, ``"dlo"`` or ``"dmo"``."""
|
|
206
|
-
return "dlo" if self.read_dlo_deltas else "dmo"
|
|
207
|
-
|
|
208
|
-
|
|
209
163
|
class ClientMethodVisitor(ast.NodeVisitor):
|
|
210
164
|
"""AST Visitor that finds all instances of Client read/write method calls."""
|
|
211
165
|
|
|
@@ -214,9 +168,6 @@ class ClientMethodVisitor(ast.NodeVisitor):
|
|
|
214
168
|
self._read_dmo_instances: set[str] = set()
|
|
215
169
|
self._write_to_dlo_instances: set[str] = set()
|
|
216
170
|
self._write_to_dmo_instances: set[str] = set()
|
|
217
|
-
self._read_dlo_deltas: bool = False
|
|
218
|
-
self._read_dmo_deltas: bool = False
|
|
219
|
-
self._write_dlo_deltas_instances: set[str] = set()
|
|
220
171
|
self.variable_values: Dict[str, Union[str, None]] = {}
|
|
221
172
|
|
|
222
173
|
def visit_Assign(self, node: ast.Assign) -> None:
|
|
@@ -238,15 +189,14 @@ class ClientMethodVisitor(ast.NodeVisitor):
|
|
|
238
189
|
node.func.value, ast.Name
|
|
239
190
|
):
|
|
240
191
|
method_name = node.func.attr
|
|
241
|
-
|
|
242
|
-
if method_name == "read_dlo_deltas":
|
|
243
|
-
self._read_dlo_deltas = True
|
|
244
|
-
elif method_name == "read_dmo_deltas":
|
|
245
|
-
self._read_dmo_deltas = True
|
|
246
|
-
|
|
247
192
|
if method_name in DATA_ACCESS_METHODS and node.args:
|
|
248
193
|
arg = node.args[0]
|
|
249
|
-
name =
|
|
194
|
+
name = None
|
|
195
|
+
|
|
196
|
+
if isinstance(arg, ast.Constant) and isinstance(arg.value, str):
|
|
197
|
+
name = arg.value
|
|
198
|
+
elif isinstance(arg, ast.Name) and arg.id in self.variable_values:
|
|
199
|
+
name = self.variable_values[arg.id]
|
|
250
200
|
|
|
251
201
|
if name:
|
|
252
202
|
if method_name == "read_dlo":
|
|
@@ -257,29 +207,8 @@ class ClientMethodVisitor(ast.NodeVisitor):
|
|
|
257
207
|
self._write_to_dlo_instances.add(name)
|
|
258
208
|
elif method_name == "write_to_dmo":
|
|
259
209
|
self._write_to_dmo_instances.add(name)
|
|
260
|
-
elif method_name in STREAMING_WRITE_METHODS and node.args:
|
|
261
|
-
name = self._resolve_name_arg(node.args[0])
|
|
262
|
-
if name and method_name == "write_dlo_deltas":
|
|
263
|
-
self._write_dlo_deltas_instances.add(name)
|
|
264
210
|
self.generic_visit(node)
|
|
265
211
|
|
|
266
|
-
def _resolve_name_arg(self, arg: ast.expr) -> Union[str, None]:
|
|
267
|
-
"""Resolve a string-literal or tracked-variable first argument."""
|
|
268
|
-
if isinstance(arg, ast.Constant) and isinstance(arg.value, str):
|
|
269
|
-
return arg.value
|
|
270
|
-
if isinstance(arg, ast.Name) and arg.id in self.variable_values:
|
|
271
|
-
return self.variable_values[arg.id]
|
|
272
|
-
return None
|
|
273
|
-
|
|
274
|
-
@property
|
|
275
|
-
def is_streaming(self) -> bool:
|
|
276
|
-
"""Whether any streaming (delta) access method was found."""
|
|
277
|
-
return (
|
|
278
|
-
self._read_dlo_deltas
|
|
279
|
-
or self._read_dmo_deltas
|
|
280
|
-
or bool(self._write_dlo_deltas_instances)
|
|
281
|
-
)
|
|
282
|
-
|
|
283
212
|
def found(self) -> DataAccessLayerCalls:
|
|
284
213
|
return DataAccessLayerCalls(
|
|
285
214
|
read_dlo=frozenset(self._read_dlo_instances),
|
|
@@ -288,13 +217,6 @@ class ClientMethodVisitor(ast.NodeVisitor):
|
|
|
288
217
|
write_to_dmo=frozenset(self._write_to_dmo_instances),
|
|
289
218
|
)
|
|
290
219
|
|
|
291
|
-
def found_streaming(self) -> StreamingDataAccessLayerCalls:
|
|
292
|
-
return StreamingDataAccessLayerCalls(
|
|
293
|
-
read_dlo_deltas=self._read_dlo_deltas,
|
|
294
|
-
read_dmo_deltas=self._read_dmo_deltas,
|
|
295
|
-
write_dlo_deltas=frozenset(self._write_dlo_deltas_instances),
|
|
296
|
-
)
|
|
297
|
-
|
|
298
220
|
|
|
299
221
|
class ImportVisitor(ast.NodeVisitor):
|
|
300
222
|
"""AST Visitor that extracts external package imports from Python code."""
|
|
@@ -379,51 +301,23 @@ def write_requirements_file(file_path: str) -> str:
|
|
|
379
301
|
return requirements_path
|
|
380
302
|
|
|
381
303
|
|
|
382
|
-
def _visit_file(file_path: str) -> ClientMethodVisitor:
|
|
383
|
-
"""Parse a Python file and return the populated method visitor."""
|
|
384
|
-
with open(file_path, "r") as f:
|
|
385
|
-
tree = ast.parse(f.read())
|
|
386
|
-
visitor = ClientMethodVisitor()
|
|
387
|
-
visitor.visit(tree)
|
|
388
|
-
return visitor
|
|
389
|
-
|
|
390
|
-
|
|
391
304
|
def scan_file(file_path: str) -> DataAccessLayerCalls:
|
|
392
|
-
"""Scan a single Python file for
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
def file_is_streaming(file_path: str) -> bool:
|
|
402
|
-
"""Return whether the entrypoint uses streaming (delta) access methods."""
|
|
403
|
-
return _visit_file(file_path).is_streaming
|
|
404
|
-
|
|
305
|
+
"""Scan a single Python file for Client read/write method calls."""
|
|
306
|
+
with open(file_path, "r") as f:
|
|
307
|
+
code = f.read()
|
|
308
|
+
tree = ast.parse(code)
|
|
309
|
+
visitor = ClientMethodVisitor()
|
|
310
|
+
visitor.visit(tree)
|
|
311
|
+
return visitor.found()
|
|
405
312
|
|
|
406
|
-
def dc_config_json_from_file(
|
|
407
|
-
file_path: str, type: str, streaming: bool = False
|
|
408
|
-
) -> dict[str, Any]:
|
|
409
|
-
"""Create a Data Cloud Custom Code config JSON from a script.
|
|
410
313
|
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
type: Package type, ``"script"`` or ``"function"``.
|
|
414
|
-
streaming: For scripts, a streaming
|
|
415
|
-
(``streamingSource``) config instead of a batch one.
|
|
416
|
-
"""
|
|
314
|
+
def dc_config_json_from_file(file_path: str, type: str) -> dict[str, Any]:
|
|
315
|
+
"""Create a Data Cloud Custom Code config JSON from a script."""
|
|
417
316
|
config: dict[str, Any]
|
|
418
317
|
if type == "script":
|
|
419
|
-
|
|
420
|
-
STREAMING_TRANSFORM_CONFIG_TEMPLATE
|
|
421
|
-
if streaming
|
|
422
|
-
else DATA_TRANSFORM_CONFIG_TEMPLATE
|
|
423
|
-
)
|
|
424
|
-
config = copy.deepcopy(template)
|
|
318
|
+
config = DATA_TRANSFORM_CONFIG_TEMPLATE.copy()
|
|
425
319
|
elif type == "function":
|
|
426
|
-
config = copy
|
|
320
|
+
config = FUNCTION_CONFIG_TEMPLATE.copy()
|
|
427
321
|
config["entryPoint"] = os.path.basename(file_path)
|
|
428
322
|
return config
|
|
429
323
|
|
|
@@ -478,49 +372,20 @@ def update_config(file_path: str) -> dict[str, Any]:
|
|
|
478
372
|
|
|
479
373
|
if package_type == "script":
|
|
480
374
|
existing_config["dataspace"] = get_dataspace(existing_config)
|
|
481
|
-
|
|
482
|
-
|
|
375
|
+
output = scan_file(file_path)
|
|
376
|
+
read: dict[str, list[str]] = {}
|
|
377
|
+
if output.read_dlo:
|
|
378
|
+
read["dlo"] = list(output.read_dlo)
|
|
483
379
|
else:
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
read["dmo"] = list(output.read_dmo)
|
|
491
|
-
write: dict[str, list[str]] = {}
|
|
492
|
-
if output.write_to_dlo:
|
|
493
|
-
write["dlo"] = list(output.write_to_dlo)
|
|
494
|
-
else:
|
|
495
|
-
write["dmo"] = list(output.write_to_dmo)
|
|
496
|
-
|
|
497
|
-
existing_config["permissions"] = {"read": read, "write": write}
|
|
498
|
-
return existing_config
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
def _update_streaming_config(existing_config: dict[str, Any], file_path: str) -> None:
|
|
502
|
-
output = scan_file_streaming(file_path)
|
|
503
|
-
read_layer = output.read_layer
|
|
504
|
-
|
|
505
|
-
source = existing_config.get("streamingSource")
|
|
506
|
-
if not isinstance(source, dict):
|
|
507
|
-
source = {}
|
|
508
|
-
source_name = source.get("name", "")
|
|
509
|
-
existing_config["streamingSource"] = {"type": read_layer, "name": source_name}
|
|
510
|
-
|
|
511
|
-
if not source_name:
|
|
512
|
-
logger.warning(
|
|
513
|
-
"streamingSource.name is empty in config.json. A streaming "
|
|
514
|
-
"transform must declare its read source; set streamingSource.name "
|
|
515
|
-
"to the DLO/DMO the transform reads from."
|
|
516
|
-
)
|
|
380
|
+
read["dmo"] = list(output.read_dmo)
|
|
381
|
+
write: dict[str, list[str]] = {}
|
|
382
|
+
if output.write_to_dlo:
|
|
383
|
+
write["dlo"] = list(output.write_to_dlo)
|
|
384
|
+
else:
|
|
385
|
+
write["dmo"] = list(output.write_to_dmo)
|
|
517
386
|
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
existing_config["permissions"] = {
|
|
521
|
-
"read": {read_layer: read_names},
|
|
522
|
-
"write": {"dlo": write_names},
|
|
523
|
-
}
|
|
387
|
+
existing_config["permissions"] = {"read": read, "write": write}
|
|
388
|
+
return existing_config
|
|
524
389
|
|
|
525
390
|
|
|
526
391
|
def get_dataspace(existing_config: dict[str, str]) -> str:
|
datacustomcode/template.py
CHANGED
|
@@ -23,12 +23,8 @@ from datacustomcode.constants import FEATURE_TEMPLATE_MAPPING
|
|
|
23
23
|
script_template_dir = os.path.join(os.path.dirname(__file__), "templates", "script")
|
|
24
24
|
function_template_dir = os.path.join(os.path.dirname(__file__), "templates", "function")
|
|
25
25
|
|
|
26
|
-
STREAMING_EXAMPLE_ENTRYPOINT = os.path.join(
|
|
27
|
-
script_template_dir, "examples", "streaming_deltas", "entrypoint.py"
|
|
28
|
-
)
|
|
29
26
|
|
|
30
|
-
|
|
31
|
-
def copy_script_template(target_dir: str, streaming: bool = False) -> None:
|
|
27
|
+
def copy_script_template(target_dir: str) -> None:
|
|
32
28
|
"""Copy the template to the target directory."""
|
|
33
29
|
os.makedirs(target_dir, exist_ok=True)
|
|
34
30
|
|
|
@@ -43,14 +39,6 @@ def copy_script_template(target_dir: str, streaming: bool = False) -> None:
|
|
|
43
39
|
logger.debug(f"Copying file {source} to {destination}...")
|
|
44
40
|
shutil.copy2(source, destination)
|
|
45
41
|
|
|
46
|
-
if streaming:
|
|
47
|
-
destination = os.path.join(target_dir, "payload", "entrypoint.py")
|
|
48
|
-
logger.debug(
|
|
49
|
-
f"Copying streaming example {STREAMING_EXAMPLE_ENTRYPOINT} to "
|
|
50
|
-
f"{destination}..."
|
|
51
|
-
)
|
|
52
|
-
shutil.copy2(STREAMING_EXAMPLE_ENTRYPOINT, destination)
|
|
53
|
-
|
|
54
42
|
|
|
55
43
|
def copy_function_template(target_dir: str, use_in_feature: Optional[str]) -> None:
|
|
56
44
|
os.makedirs(target_dir, exist_ok=True)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# Chunking with a Gemini Named Credential Callout
|
|
2
|
+
|
|
3
|
+
Splits each input document into paragraph-sized chunks and calls Google's
|
|
4
|
+
**Gemini** `generateContent` API for every chunk. The model returns a summary,
|
|
5
|
+
category, sentiment, and topics, which are attached to the chunk as citations so
|
|
6
|
+
the search index can filter and rank on them. Gemini is reached through a
|
|
7
|
+
**Named Credential**, so this code never handles the endpoint URL or the API key.
|
|
8
|
+
|
|
9
|
+
## How the callout works
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
CALLOUT_URL = "callout:gemini" # callout:<NC name>[/<path>]
|
|
13
|
+
|
|
14
|
+
request = (
|
|
15
|
+
HTTPRequestBuilder()
|
|
16
|
+
.set_url(CALLOUT_URL)
|
|
17
|
+
.set_method(HTTPMethod.POST)
|
|
18
|
+
.set_headers({"Content-Type": "application/json"})
|
|
19
|
+
.build()
|
|
20
|
+
)
|
|
21
|
+
# Body is sent verbatim (serialize it yourself); the response body is a raw string.
|
|
22
|
+
response = runtime.named_credential.request(request, json.dumps(payload))
|
|
23
|
+
if response.is_success:
|
|
24
|
+
envelope = json.loads(response.body)
|
|
25
|
+
text = envelope["candidates"][0]["content"]["parts"][0]["text"]
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
The request asks for `responseMimeType: application/json` with a `responseSchema`,
|
|
29
|
+
so Gemini returns the classification as a JSON string in
|
|
30
|
+
`candidates[0].content.parts[0].text` — decode it, then decode that text again.
|
|
31
|
+
|
|
32
|
+
The `gemini` Named Credential's URL already includes the full
|
|
33
|
+
`/v1beta/models/<model>:generateContent` path, so the callout is just
|
|
34
|
+
`callout:gemini` with **no path suffix** (anything after the name is appended to
|
|
35
|
+
the credential's URL).
|
|
36
|
+
|
|
37
|
+
## Configure the Named Credential
|
|
38
|
+
|
|
39
|
+
1. Create an **External Credential** (e.g. `google_api_key`) that injects your
|
|
40
|
+
Gemini API key as the `X-goog-api-key` header.
|
|
41
|
+
2. Create a **Named Credential** named `gemini`:
|
|
42
|
+
- **URL**: `https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent`
|
|
43
|
+
- **Enabled for Callouts** + **Generate Authorization Header**: on
|
|
44
|
+
- **External Credential**: `google_api_key`
|
|
45
|
+
|
|
46
|
+
## Test locally
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG=/abs/path/to/external_callout_config.json \
|
|
50
|
+
sf data-code-extension function run \
|
|
51
|
+
--entrypoint payload/entrypoint.py \
|
|
52
|
+
--test-with payload/tests/test.json \
|
|
53
|
+
--target-org <your-org-alias>
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
With `--target-org` the SDK fetches only the **URL** from the org's Named
|
|
57
|
+
Credential; **auth is always taken from `external_callout_config.json`** locally
|
|
58
|
+
(the org's External Credential is used only in the Data Cloud runtime). So the
|
|
59
|
+
`X-goog-api-key` must be in the local config for a local test. Omit
|
|
60
|
+
`--target-org` to run fully offline using `target_url`.
|
|
61
|
+
|
|
62
|
+
```json
|
|
63
|
+
{
|
|
64
|
+
"credentials": {
|
|
65
|
+
"callout:gemini": {
|
|
66
|
+
"auth_type": "Custom",
|
|
67
|
+
"custom_headers": { "X-goog-api-key": "YOUR_GEMINI_API_KEY" },
|
|
68
|
+
"target_url": "https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent"
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Place `external_callout_config.json` in the **parent of your payload folder** (or
|
|
75
|
+
point `DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG` at it). It is never packaged into
|
|
76
|
+
the deployment zip. Get a key from [Google AI Studio](https://aistudio.google.com/apikey);
|
|
77
|
+
**do not commit it.**
|
|
78
|
+
|
|
79
|
+
## Auth types
|
|
80
|
+
|
|
81
|
+
`auth_type` selects how auth is injected for local testing. It should mirror the
|
|
82
|
+
External Credential your Named Credential uses in the org, so local and deployed
|
|
83
|
+
runs behave the same. This example uses `Custom` (Gemini's `X-goog-api-key`);
|
|
84
|
+
all four supported types:
|
|
85
|
+
|
|
86
|
+
```json
|
|
87
|
+
{
|
|
88
|
+
"credentials": {
|
|
89
|
+
"callout:my_custom_api": {
|
|
90
|
+
"auth_type": "Custom",
|
|
91
|
+
"custom_headers": { "X-goog-api-key": "YOUR_API_KEY" }
|
|
92
|
+
},
|
|
93
|
+
"callout:my_basic_api": {
|
|
94
|
+
"auth_type": "Basic",
|
|
95
|
+
"username": "svc_user",
|
|
96
|
+
"password": "YOUR_PASSWORD"
|
|
97
|
+
},
|
|
98
|
+
"callout:my_oauth_api": {
|
|
99
|
+
"auth_type": "OAuth",
|
|
100
|
+
"access_token": "YOUR_ACCESS_TOKEN"
|
|
101
|
+
},
|
|
102
|
+
"callout:my_jwt_api": {
|
|
103
|
+
"auth_type": "Jwt",
|
|
104
|
+
"token": "YOUR_JWT"
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
| `auth_type` | Fields read | Header sent |
|
|
111
|
+
| ----------- | ------------------------------- | --------------------------------------- |
|
|
112
|
+
| `Basic` | `username`, `password` | `Authorization: Basic <base64 user:pw>` |
|
|
113
|
+
| `Custom` | `custom_headers` (sent verbatim)| the headers you list |
|
|
114
|
+
| `OAuth` | `access_token` or `token` | `Authorization: Bearer <token>` |
|
|
115
|
+
| `Jwt` | `access_token` or `token` | `Authorization: Bearer <token>` |
|
|
116
|
+
|
|
117
|
+
`OAuth`/`Jwt` take a token you supply for the local run — the SDK does not fetch
|
|
118
|
+
or refresh it. In the Data Cloud runtime the Named Credential handles token
|
|
119
|
+
acquisition; this local config only stands in for that during testing.
|