salesforce-data-customcode 6.0.8.dev1__py3-none-any.whl → 6.1.0.dev2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datacustomcode/client.py +92 -0
- datacustomcode/config.yaml +6 -0
- datacustomcode/deploy.py +7 -2
- datacustomcode/einstein_predictions/spark_default.py +88 -27
- datacustomcode/function/runtime.py +16 -0
- datacustomcode/named_credential/__init__.py +26 -0
- datacustomcode/named_credential/base.py +54 -0
- datacustomcode/named_credential/default.py +93 -0
- datacustomcode/named_credential/direct/__init__.py +19 -0
- datacustomcode/named_credential/direct/auth.py +63 -0
- datacustomcode/named_credential/direct/credentials.py +121 -0
- datacustomcode/named_credential/direct/transport.py +110 -0
- datacustomcode/named_credential/direct/url_resolver.py +112 -0
- datacustomcode/named_credential/spark_base.py +93 -0
- datacustomcode/named_credential/spark_default.py +154 -0
- datacustomcode/named_credential/types/__init__.py +14 -0
- datacustomcode/named_credential/types/http_method.py +29 -0
- datacustomcode/named_credential/types/http_request.py +63 -0
- datacustomcode/named_credential/types/http_request_builder.py +55 -0
- datacustomcode/named_credential/types/http_response.py +43 -0
- datacustomcode/named_credential/types/http_response_builder.py +24 -0
- datacustomcode/named_credential_config.py +105 -0
- datacustomcode/run.py +7 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/README.md +119 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/config.json +3 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +161 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +11 -0
- datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +16 -0
- {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/METADATA +1 -1
- {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/RECORD +33 -11
- {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/WHEEL +0 -0
- {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/entry_points.txt +0 -0
- {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/licenses/LICENSE.txt +0 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from enum import Enum
|
|
17
|
+
from typing import Dict
|
|
18
|
+
|
|
19
|
+
from pydantic import (
|
|
20
|
+
BaseModel,
|
|
21
|
+
ConfigDict,
|
|
22
|
+
Field,
|
|
23
|
+
field_validator,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
from datacustomcode.named_credential.types.http_method import HTTPMethod
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class HTTPRequest(BaseModel):
|
|
30
|
+
"""External callout request. The endpoint and its auth are resolved
|
|
31
|
+
server-side from the Named Credential referenced by ``url``, which uses
|
|
32
|
+
``callout:<NamedCredential>/<path>`` syntax."""
|
|
33
|
+
|
|
34
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
35
|
+
|
|
36
|
+
url: str = Field(
|
|
37
|
+
...,
|
|
38
|
+
min_length=1,
|
|
39
|
+
description="Symbolic Named Credential reference, "
|
|
40
|
+
"e.g. 'callout:<NamedCredential>/<path>'",
|
|
41
|
+
)
|
|
42
|
+
method: str = Field(
|
|
43
|
+
default="GET",
|
|
44
|
+
description="HTTP method (GET, POST, PUT, DELETE, PATCH, HEAD, OPTIONS)",
|
|
45
|
+
)
|
|
46
|
+
headers: Dict[str, str] = Field(default_factory=dict, description="Request headers")
|
|
47
|
+
|
|
48
|
+
@field_validator("method", mode="before")
|
|
49
|
+
@classmethod
|
|
50
|
+
def _normalize_method(cls, value: object) -> str:
|
|
51
|
+
# Accept str, this module's HTTPMethod, or http.HTTPMethod (3.11+).
|
|
52
|
+
if isinstance(value, Enum):
|
|
53
|
+
method = str(value.value)
|
|
54
|
+
else:
|
|
55
|
+
method = str(value)
|
|
56
|
+
method = method.upper()
|
|
57
|
+
if method not in {m.value for m in HTTPMethod}:
|
|
58
|
+
supported = ", ".join(m.value for m in HTTPMethod)
|
|
59
|
+
raise ValueError(
|
|
60
|
+
f"Unsupported HTTP method '{method}'. "
|
|
61
|
+
f"Named Credential callouts support {supported}."
|
|
62
|
+
)
|
|
63
|
+
return method
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from typing import Dict, Union
|
|
17
|
+
|
|
18
|
+
from datacustomcode.named_credential.types.http_method import HTTPMethod
|
|
19
|
+
from datacustomcode.named_credential.types.http_request import HTTPRequest
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class HTTPRequestBuilder:
|
|
23
|
+
def __init__(self) -> None:
|
|
24
|
+
self._url = ""
|
|
25
|
+
self._method: Union[str, HTTPMethod] = HTTPMethod.GET
|
|
26
|
+
self._headers: Dict[str, str] = {}
|
|
27
|
+
|
|
28
|
+
def set_url(self, url: str) -> "HTTPRequestBuilder":
|
|
29
|
+
"""Set the symbolic Named Credential reference.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
url: e.g. 'callout:<NamedCredential>/<path>'
|
|
33
|
+
"""
|
|
34
|
+
self._url = url
|
|
35
|
+
return self
|
|
36
|
+
|
|
37
|
+
def set_method(self, method: Union[str, HTTPMethod]) -> "HTTPRequestBuilder":
|
|
38
|
+
"""Set the HTTP method.
|
|
39
|
+
|
|
40
|
+
Accepts this module's ``HTTPMethod``, ``http.HTTPMethod`` (Python 3.11+),
|
|
41
|
+
or a plain string such as ``"GET"``.
|
|
42
|
+
"""
|
|
43
|
+
self._method = method
|
|
44
|
+
return self
|
|
45
|
+
|
|
46
|
+
def set_headers(self, headers: Dict[str, str]) -> "HTTPRequestBuilder":
|
|
47
|
+
self._headers = headers
|
|
48
|
+
return self
|
|
49
|
+
|
|
50
|
+
def build(self) -> HTTPRequest:
|
|
51
|
+
return HTTPRequest(
|
|
52
|
+
url=self._url,
|
|
53
|
+
method=self._method, # type: ignore[arg-type]
|
|
54
|
+
headers=self._headers,
|
|
55
|
+
)
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from typing import Dict
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel, Field
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class HTTPResponse(BaseModel):
|
|
22
|
+
"""Response from a Named Credential external callout."""
|
|
23
|
+
|
|
24
|
+
status_code: int = Field(..., description="HTTP status code", ge=0)
|
|
25
|
+
headers: Dict[str, str] = Field(
|
|
26
|
+
default_factory=dict, description="Response headers"
|
|
27
|
+
)
|
|
28
|
+
body: str = Field(
|
|
29
|
+
default="",
|
|
30
|
+
description="Raw response body, verbatim (any format). The SDK does not "
|
|
31
|
+
"parse it; the caller decodes as needed (e.g. json.loads). Empty string "
|
|
32
|
+
"if the response had no body.",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def is_success(self) -> bool:
|
|
37
|
+
"""Check if the request succeeded (2xx)."""
|
|
38
|
+
return 200 <= self.status_code < 300
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def is_error(self) -> bool:
|
|
42
|
+
"""Check if the request failed."""
|
|
43
|
+
return not self.is_success
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from typing import Any, Dict
|
|
17
|
+
|
|
18
|
+
from datacustomcode.named_credential.types.http_response import HTTPResponse
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class HTTPResponseBuilder:
|
|
22
|
+
@staticmethod
|
|
23
|
+
def build(response_dict: Dict[str, Any]) -> HTTPResponse:
|
|
24
|
+
return HTTPResponse.model_validate(response_dict)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from typing import (
|
|
17
|
+
ClassVar,
|
|
18
|
+
Generic,
|
|
19
|
+
Type,
|
|
20
|
+
TypeVar,
|
|
21
|
+
Union,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
from datacustomcode.common_config import (
|
|
25
|
+
BaseConfig,
|
|
26
|
+
BaseObjectConfig,
|
|
27
|
+
default_config_file,
|
|
28
|
+
)
|
|
29
|
+
from datacustomcode.named_credential.base import NamedCredential
|
|
30
|
+
from datacustomcode.named_credential.spark_base import SparkNamedCredential
|
|
31
|
+
|
|
32
|
+
_N = TypeVar("_N", bound=NamedCredential)
|
|
33
|
+
_S = TypeVar("_S", bound=SparkNamedCredential)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class NamedCredentialObjectConfig(BaseObjectConfig, Generic[_N]):
|
|
37
|
+
type_to_create: ClassVar[Type[NamedCredential]] = NamedCredential # type: ignore[type-abstract]
|
|
38
|
+
|
|
39
|
+
def to_object(self) -> NamedCredential:
|
|
40
|
+
type_ = self.type_to_create.subclass_from_config_name(self.type_config_name)
|
|
41
|
+
return type_(**self.options)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class NamedCredentialConfig(BaseConfig):
|
|
45
|
+
named_credential_config: Union[
|
|
46
|
+
NamedCredentialObjectConfig[NamedCredential], None
|
|
47
|
+
] = None
|
|
48
|
+
|
|
49
|
+
def update(self, other: "NamedCredentialConfig") -> "NamedCredentialConfig":
|
|
50
|
+
def merge(
|
|
51
|
+
config_a: Union[NamedCredentialObjectConfig, None],
|
|
52
|
+
config_b: Union[NamedCredentialObjectConfig, None],
|
|
53
|
+
) -> Union[NamedCredentialObjectConfig, None]:
|
|
54
|
+
if config_a is not None and config_a.force:
|
|
55
|
+
return config_a
|
|
56
|
+
if config_b:
|
|
57
|
+
return config_b
|
|
58
|
+
return config_a
|
|
59
|
+
|
|
60
|
+
self.named_credential_config = merge(
|
|
61
|
+
self.named_credential_config, other.named_credential_config
|
|
62
|
+
)
|
|
63
|
+
return self
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class SparkNamedCredentialObjectConfig(BaseObjectConfig, Generic[_S]):
|
|
67
|
+
type_to_create: ClassVar[Type[SparkNamedCredential]] = SparkNamedCredential # type: ignore[type-abstract]
|
|
68
|
+
|
|
69
|
+
def to_object(self) -> SparkNamedCredential:
|
|
70
|
+
type_ = self.type_to_create.subclass_from_config_name(self.type_config_name)
|
|
71
|
+
return type_(**self.options)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class SparkNamedCredentialConfig(BaseConfig):
|
|
75
|
+
spark_named_credential_config: Union[
|
|
76
|
+
SparkNamedCredentialObjectConfig[SparkNamedCredential], None
|
|
77
|
+
] = None
|
|
78
|
+
|
|
79
|
+
def update(
|
|
80
|
+
self, other: "SparkNamedCredentialConfig"
|
|
81
|
+
) -> "SparkNamedCredentialConfig":
|
|
82
|
+
def merge(
|
|
83
|
+
config_a: Union[SparkNamedCredentialObjectConfig, None],
|
|
84
|
+
config_b: Union[SparkNamedCredentialObjectConfig, None],
|
|
85
|
+
) -> Union[SparkNamedCredentialObjectConfig, None]:
|
|
86
|
+
if config_a is not None and config_a.force:
|
|
87
|
+
return config_a
|
|
88
|
+
if config_b:
|
|
89
|
+
return config_b
|
|
90
|
+
return config_a
|
|
91
|
+
|
|
92
|
+
self.spark_named_credential_config = merge(
|
|
93
|
+
self.spark_named_credential_config, other.spark_named_credential_config
|
|
94
|
+
)
|
|
95
|
+
return self
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# Global Named Credential config instance
|
|
99
|
+
named_credential_config = NamedCredentialConfig()
|
|
100
|
+
named_credential_config.load(default_config_file())
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
# Global Spark Named Credential config instance
|
|
104
|
+
spark_named_credential_config = SparkNamedCredentialConfig()
|
|
105
|
+
spark_named_credential_config.load(default_config_file())
|
datacustomcode/run.py
CHANGED
|
@@ -27,6 +27,7 @@ from typing import (
|
|
|
27
27
|
from datacustomcode.config import config
|
|
28
28
|
from datacustomcode.einstein_predictions_config import einstein_predictions_config
|
|
29
29
|
from datacustomcode.llm_gateway_config import llm_gateway_config
|
|
30
|
+
from datacustomcode.named_credential_config import named_credential_config
|
|
30
31
|
from datacustomcode.scan import find_base_directory, get_package_type
|
|
31
32
|
|
|
32
33
|
|
|
@@ -55,6 +56,9 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
|
|
|
55
56
|
_set_config_option(
|
|
56
57
|
llm_gateway_config.llm_gateway_config, config_key, sf_cli_org
|
|
57
58
|
)
|
|
59
|
+
_set_config_option(
|
|
60
|
+
named_credential_config.named_credential_config, config_key, sf_cli_org
|
|
61
|
+
)
|
|
58
62
|
elif profile != "default":
|
|
59
63
|
config_key = "credentials_profile"
|
|
60
64
|
_set_config_option(config.reader_config, config_key, profile)
|
|
@@ -63,6 +67,9 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
|
|
|
63
67
|
einstein_predictions_config.einstein_predictions_config, config_key, profile
|
|
64
68
|
)
|
|
65
69
|
_set_config_option(llm_gateway_config.llm_gateway_config, config_key, profile)
|
|
70
|
+
_set_config_option(
|
|
71
|
+
named_credential_config.named_credential_config, config_key, profile
|
|
72
|
+
)
|
|
66
73
|
|
|
67
74
|
|
|
68
75
|
def run_entrypoint(
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# Chunking with a Gemini Named Credential Callout
|
|
2
|
+
|
|
3
|
+
Splits each input document into paragraph-sized chunks and calls Google's
|
|
4
|
+
**Gemini** `generateContent` API for every chunk. The model returns a summary,
|
|
5
|
+
category, sentiment, and topics, which are attached to the chunk as citations so
|
|
6
|
+
the search index can filter and rank on them. Gemini is reached through a
|
|
7
|
+
**Named Credential**, so this code never handles the endpoint URL or the API key.
|
|
8
|
+
|
|
9
|
+
## How the callout works
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
CALLOUT_URL = "callout:gemini" # callout:<NC name>[/<path>]
|
|
13
|
+
|
|
14
|
+
request = (
|
|
15
|
+
HTTPRequestBuilder()
|
|
16
|
+
.set_url(CALLOUT_URL)
|
|
17
|
+
.set_method(HTTPMethod.POST)
|
|
18
|
+
.set_headers({"Content-Type": "application/json"})
|
|
19
|
+
.build()
|
|
20
|
+
)
|
|
21
|
+
# Body is sent verbatim (serialize it yourself); the response body is a raw string.
|
|
22
|
+
response = runtime.named_credential.request(request, json.dumps(payload))
|
|
23
|
+
if response.is_success:
|
|
24
|
+
envelope = json.loads(response.body)
|
|
25
|
+
text = envelope["candidates"][0]["content"]["parts"][0]["text"]
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
The request asks for `responseMimeType: application/json` with a `responseSchema`,
|
|
29
|
+
so Gemini returns the classification as a JSON string in
|
|
30
|
+
`candidates[0].content.parts[0].text` — decode it, then decode that text again.
|
|
31
|
+
|
|
32
|
+
The `gemini` Named Credential's URL already includes the full
|
|
33
|
+
`/v1beta/models/<model>:generateContent` path, so the callout is just
|
|
34
|
+
`callout:gemini` with **no path suffix** (anything after the name is appended to
|
|
35
|
+
the credential's URL).
|
|
36
|
+
|
|
37
|
+
## Configure the Named Credential
|
|
38
|
+
|
|
39
|
+
1. Create an **External Credential** (e.g. `google_api_key`) that injects your
|
|
40
|
+
Gemini API key as the `X-goog-api-key` header.
|
|
41
|
+
2. Create a **Named Credential** named `gemini`:
|
|
42
|
+
- **URL**: `https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent`
|
|
43
|
+
- **Enabled for Callouts** + **Generate Authorization Header**: on
|
|
44
|
+
- **External Credential**: `google_api_key`
|
|
45
|
+
|
|
46
|
+
## Test locally
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG=/abs/path/to/external_callout_config.json \
|
|
50
|
+
sf data-code-extension function run \
|
|
51
|
+
--entrypoint payload/entrypoint.py \
|
|
52
|
+
--test-with payload/tests/test.json \
|
|
53
|
+
--target-org <your-org-alias>
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
With `--target-org` the SDK fetches only the **URL** from the org's Named
|
|
57
|
+
Credential; **auth is always taken from `external_callout_config.json`** locally
|
|
58
|
+
(the org's External Credential is used only in the Data Cloud runtime). So the
|
|
59
|
+
`X-goog-api-key` must be in the local config for a local test. Omit
|
|
60
|
+
`--target-org` to run fully offline using `target_url`.
|
|
61
|
+
|
|
62
|
+
```json
|
|
63
|
+
{
|
|
64
|
+
"credentials": {
|
|
65
|
+
"callout:gemini": {
|
|
66
|
+
"auth_type": "Custom",
|
|
67
|
+
"custom_headers": { "X-goog-api-key": "YOUR_GEMINI_API_KEY" },
|
|
68
|
+
"target_url": "https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent"
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Place `external_callout_config.json` in the **parent of your payload folder** (or
|
|
75
|
+
point `DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG` at it). It is never packaged into
|
|
76
|
+
the deployment zip. Get a key from [Google AI Studio](https://aistudio.google.com/apikey);
|
|
77
|
+
**do not commit it.**
|
|
78
|
+
|
|
79
|
+
## Auth types
|
|
80
|
+
|
|
81
|
+
`auth_type` selects how auth is injected for local testing. It should mirror the
|
|
82
|
+
External Credential your Named Credential uses in the org, so local and deployed
|
|
83
|
+
runs behave the same. This example uses `Custom` (Gemini's `X-goog-api-key`);
|
|
84
|
+
all four supported types:
|
|
85
|
+
|
|
86
|
+
```json
|
|
87
|
+
{
|
|
88
|
+
"credentials": {
|
|
89
|
+
"callout:my_custom_api": {
|
|
90
|
+
"auth_type": "Custom",
|
|
91
|
+
"custom_headers": { "X-goog-api-key": "YOUR_API_KEY" }
|
|
92
|
+
},
|
|
93
|
+
"callout:my_basic_api": {
|
|
94
|
+
"auth_type": "Basic",
|
|
95
|
+
"username": "svc_user",
|
|
96
|
+
"password": "YOUR_PASSWORD"
|
|
97
|
+
},
|
|
98
|
+
"callout:my_oauth_api": {
|
|
99
|
+
"auth_type": "OAuth",
|
|
100
|
+
"access_token": "YOUR_ACCESS_TOKEN"
|
|
101
|
+
},
|
|
102
|
+
"callout:my_jwt_api": {
|
|
103
|
+
"auth_type": "Jwt",
|
|
104
|
+
"token": "YOUR_JWT"
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
| `auth_type` | Fields read | Header sent |
|
|
111
|
+
| ----------- | ------------------------------- | --------------------------------------- |
|
|
112
|
+
| `Basic` | `username`, `password` | `Authorization: Basic <base64 user:pw>` |
|
|
113
|
+
| `Custom` | `custom_headers` (sent verbatim)| the headers you list |
|
|
114
|
+
| `OAuth` | `access_token` or `token` | `Authorization: Bearer <token>` |
|
|
115
|
+
| `Jwt` | `access_token` or `token` | `Authorization: Bearer <token>` |
|
|
116
|
+
|
|
117
|
+
`OAuth`/`Jwt` take a token you supply for the local run — the SDK does not fetch
|
|
118
|
+
or refresh it. In the Data Cloud runtime the Named Credential handles token
|
|
119
|
+
acquisition; this local config only stands in for that during testing.
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
3
|
+
# SPDX-License-Identifier: Apache-2
|
|
4
|
+
|
|
5
|
+
"""
|
|
6
|
+
Document Chunking with a Gemini Named Credential Callout
|
|
7
|
+
|
|
8
|
+
Splits each input document into paragraph-sized chunks and classifies every
|
|
9
|
+
chunk via Google's Gemini ``generateContent`` API, reached through a Named
|
|
10
|
+
Credential (``callout:gemini``) so the endpoint URL and API key are resolved
|
|
11
|
+
outside this code. The classification is attached to each chunk as citations.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import logging
|
|
16
|
+
|
|
17
|
+
from datacustomcode.function import Runtime
|
|
18
|
+
from datacustomcode.function.feature_types.chunking import (
|
|
19
|
+
ChunkType,
|
|
20
|
+
SearchIndexChunkingV1Output,
|
|
21
|
+
SearchIndexChunkingV1Request,
|
|
22
|
+
SearchIndexChunkingV1Response,
|
|
23
|
+
)
|
|
24
|
+
from datacustomcode.named_credential.types.http_method import HTTPMethod
|
|
25
|
+
from datacustomcode.named_credential.types.http_request_builder import (
|
|
26
|
+
HTTPRequestBuilder,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
logging.basicConfig(level=logging.INFO)
|
|
31
|
+
|
|
32
|
+
CALLOUT_URL = "callout:gemini"
|
|
33
|
+
|
|
34
|
+
_ANALYSIS_FIELDS = ("summary", "category", "sentiment")
|
|
35
|
+
|
|
36
|
+
_PROMPT = (
|
|
37
|
+
"Analyze the following document chunk and classify it. Respond with its "
|
|
38
|
+
"one-sentence summary, a single-word category, overall sentiment "
|
|
39
|
+
"(positive, negative, or neutral), and up to five key topics.\n\nChunk:\n"
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
# Force Gemini to return the classification as JSON in a fixed shape.
|
|
43
|
+
_GENERATION_CONFIG = {
|
|
44
|
+
"responseMimeType": "application/json",
|
|
45
|
+
"responseSchema": {
|
|
46
|
+
"type": "object",
|
|
47
|
+
"properties": {
|
|
48
|
+
"summary": {"type": "string"},
|
|
49
|
+
"category": {"type": "string"},
|
|
50
|
+
"sentiment": {"type": "string"},
|
|
51
|
+
"topics": {"type": "array", "items": {"type": "string"}},
|
|
52
|
+
},
|
|
53
|
+
"required": ["summary", "category", "sentiment", "topics"],
|
|
54
|
+
},
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _chunk_text(text: str, max_words: int = 80) -> list[str]:
|
|
59
|
+
"""Split text into paragraph-aligned chunks of at most ``max_words`` words."""
|
|
60
|
+
paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()]
|
|
61
|
+
|
|
62
|
+
chunks: list[str] = []
|
|
63
|
+
current: list[str] = []
|
|
64
|
+
current_words = 0
|
|
65
|
+
|
|
66
|
+
for paragraph in paragraphs:
|
|
67
|
+
paragraph_words = len(paragraph.split())
|
|
68
|
+
if current and current_words + paragraph_words > max_words:
|
|
69
|
+
chunks.append("\n\n".join(current))
|
|
70
|
+
current = []
|
|
71
|
+
current_words = 0
|
|
72
|
+
current.append(paragraph)
|
|
73
|
+
current_words += paragraph_words
|
|
74
|
+
|
|
75
|
+
if current:
|
|
76
|
+
chunks.append("\n\n".join(current))
|
|
77
|
+
|
|
78
|
+
return chunks
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _extract_model_json(body: str) -> dict:
|
|
82
|
+
"""Decode the model's JSON classification from a Gemini response.
|
|
83
|
+
|
|
84
|
+
The generated text sits at ``candidates[0].content.parts[0].text`` and is
|
|
85
|
+
itself a JSON string, so decode twice. Any malformed layer yields ``{}``.
|
|
86
|
+
"""
|
|
87
|
+
try:
|
|
88
|
+
envelope = json.loads(body) if body else {}
|
|
89
|
+
except json.JSONDecodeError:
|
|
90
|
+
return {}
|
|
91
|
+
|
|
92
|
+
try:
|
|
93
|
+
text = envelope["candidates"][0]["content"]["parts"][0]["text"]
|
|
94
|
+
except (KeyError, IndexError, TypeError):
|
|
95
|
+
return {}
|
|
96
|
+
|
|
97
|
+
try:
|
|
98
|
+
payload = json.loads(text)
|
|
99
|
+
except json.JSONDecodeError:
|
|
100
|
+
return {}
|
|
101
|
+
return payload if isinstance(payload, dict) else {}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _analyze_chunk(chunk_text: str, runtime: Runtime) -> dict[str, str]:
|
|
105
|
+
"""Classify one chunk via the Gemini callout and return it as citations."""
|
|
106
|
+
request = (
|
|
107
|
+
HTTPRequestBuilder()
|
|
108
|
+
.set_url(CALLOUT_URL)
|
|
109
|
+
.set_method(HTTPMethod.POST)
|
|
110
|
+
.set_headers({"Content-Type": "application/json", "Accept": "application/json"})
|
|
111
|
+
.build()
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
payload = {
|
|
115
|
+
"contents": [{"parts": [{"text": _PROMPT + chunk_text}]}],
|
|
116
|
+
"generationConfig": _GENERATION_CONFIG,
|
|
117
|
+
}
|
|
118
|
+
response = runtime.named_credential.request(request, json.dumps(payload))
|
|
119
|
+
|
|
120
|
+
# Don't raise: a single failed callout shouldn't abort the whole job.
|
|
121
|
+
if not response.is_success:
|
|
122
|
+
logger.error(f"Gemini callout failed with status {response.status_code}")
|
|
123
|
+
return {"analysis_status": "failed", "http_status": str(response.status_code)}
|
|
124
|
+
|
|
125
|
+
data = _extract_model_json(response.body)
|
|
126
|
+
citations = {"analysis_status": "success"}
|
|
127
|
+
for field in _ANALYSIS_FIELDS:
|
|
128
|
+
value = data.get(field)
|
|
129
|
+
citations[field] = str(value) if value is not None else "unavailable"
|
|
130
|
+
|
|
131
|
+
topics = data.get("topics")
|
|
132
|
+
if isinstance(topics, list):
|
|
133
|
+
citations["topics"] = ", ".join(str(topic) for topic in topics)
|
|
134
|
+
|
|
135
|
+
return citations
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def function(
|
|
139
|
+
request: SearchIndexChunkingV1Request, runtime: Runtime
|
|
140
|
+
) -> SearchIndexChunkingV1Response:
|
|
141
|
+
"""Chunk each input document and classify every chunk via the Gemini API."""
|
|
142
|
+
logger.info(f"Received {len(request.input)} documents to chunk")
|
|
143
|
+
|
|
144
|
+
chunks = []
|
|
145
|
+
chunk_id = 1
|
|
146
|
+
|
|
147
|
+
for doc in request.input:
|
|
148
|
+
for chunk_text in _chunk_text(doc.text):
|
|
149
|
+
citations = _analyze_chunk(chunk_text, runtime)
|
|
150
|
+
|
|
151
|
+
chunk = SearchIndexChunkingV1Output(
|
|
152
|
+
text=chunk_text,
|
|
153
|
+
seq_no=chunk_id,
|
|
154
|
+
chunk_type=ChunkType.TEXT,
|
|
155
|
+
citations=citations,
|
|
156
|
+
)
|
|
157
|
+
chunks.append(chunk)
|
|
158
|
+
chunk_id += 1
|
|
159
|
+
|
|
160
|
+
logger.info(f"Produced {len(chunks)} classified chunks")
|
|
161
|
+
return SearchIndexChunkingV1Response(output=chunks)
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": [
|
|
3
|
+
{
|
|
4
|
+
"text": "Product Review: Northstar Analytics\n\nWe rolled Northstar out to our whole revenue team last quarter and the difference has been night and day. Dashboards that used to take our analysts a full day to assemble now refresh in seconds, and the natural-language query box means our account executives can answer their own questions without filing a ticket.\n\nOnboarding was smoother than any tool we have adopted in years. The guided setup imported our Salesforce data on the first try and the sample templates gave us something useful on day one. Support answered our two questions within the hour. Easily the best purchase decision we made this year."
|
|
5
|
+
},
|
|
6
|
+
{
|
|
7
|
+
"text": "Support Ticket #48210: Repeated timeouts on scheduled exports\n\nFor the third week running our nightly export to the data warehouse has failed silently. There is no alert, no email, nothing in the activity log, and we only find out when the morning report is empty and the leadership meeting has no numbers.\n\nI have raised this twice already and both times the ticket was closed as resolved without anyone actually contacting me. This is costing us real credibility internally and I am extremely frustrated. If the connector cannot handle our volume we need to know now so we can plan a migration, because right now the product is not doing the one job we bought it for."
|
|
8
|
+
},
|
|
9
|
+
{
|
|
10
|
+
"text": "Renewal Feedback: mixed feelings heading into year two\n\nThe core product is genuinely good. The reporting engine is fast, the permissions model is granular enough for our compliance team, and our analysts like working in it. On the functionality alone I would renew without hesitation.\n\nWhat gives me pause is the pricing. The per-seat cost jumped noticeably at renewal and several add-ons that used to be included are now separate line items. The value is still there, but the conversation with my finance team was harder than it should have been, and I would like more transparency before the next cycle."
|
|
11
|
+
},
|
|
12
|
+
{
|
|
13
|
+
"text": "Feature Request: scheduled report subscriptions\n\nWe would like the ability to subscribe internal stakeholders to a report on a recurring schedule so a PDF lands in their inbox every Monday morning. Today we export manually and forward it, which is workable but easy to forget.\n\nA few teams have asked whether subscriptions could support filtered views per recipient, for example each regional manager receiving only their own territory. Not urgent for us, but it would remove a recurring bit of manual work and is something a couple of competing tools already offer."
|
|
14
|
+
}
|
|
15
|
+
]
|
|
16
|
+
}
|