salesforce-data-customcode 6.0.8.dev1__py3-none-any.whl → 6.1.0.dev2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. datacustomcode/client.py +92 -0
  2. datacustomcode/config.yaml +6 -0
  3. datacustomcode/deploy.py +7 -2
  4. datacustomcode/einstein_predictions/spark_default.py +88 -27
  5. datacustomcode/function/runtime.py +16 -0
  6. datacustomcode/named_credential/__init__.py +26 -0
  7. datacustomcode/named_credential/base.py +54 -0
  8. datacustomcode/named_credential/default.py +93 -0
  9. datacustomcode/named_credential/direct/__init__.py +19 -0
  10. datacustomcode/named_credential/direct/auth.py +63 -0
  11. datacustomcode/named_credential/direct/credentials.py +121 -0
  12. datacustomcode/named_credential/direct/transport.py +110 -0
  13. datacustomcode/named_credential/direct/url_resolver.py +112 -0
  14. datacustomcode/named_credential/spark_base.py +93 -0
  15. datacustomcode/named_credential/spark_default.py +154 -0
  16. datacustomcode/named_credential/types/__init__.py +14 -0
  17. datacustomcode/named_credential/types/http_method.py +29 -0
  18. datacustomcode/named_credential/types/http_request.py +63 -0
  19. datacustomcode/named_credential/types/http_request_builder.py +55 -0
  20. datacustomcode/named_credential/types/http_response.py +43 -0
  21. datacustomcode/named_credential/types/http_response_builder.py +24 -0
  22. datacustomcode/named_credential_config.py +105 -0
  23. datacustomcode/run.py +7 -0
  24. datacustomcode/templates/function/example/chunking_with_external_callout/README.md +119 -0
  25. datacustomcode/templates/function/example/chunking_with_external_callout/config.json +3 -0
  26. datacustomcode/templates/function/example/chunking_with_external_callout/entrypoint.py +161 -0
  27. datacustomcode/templates/function/example/chunking_with_external_callout/external_callout_config.json +11 -0
  28. datacustomcode/templates/function/example/chunking_with_external_callout/tests/test.json +16 -0
  29. {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/METADATA +1 -1
  30. {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/RECORD +33 -11
  31. {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/WHEEL +0 -0
  32. {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/entry_points.txt +0 -0
  33. {salesforce_data_customcode-6.0.8.dev1.dist-info → salesforce_data_customcode-6.1.0.dev2.dist-info}/licenses/LICENSE.txt +0 -0
@@ -0,0 +1,63 @@
1
+ # Copyright (c) 2025, Salesforce, Inc.
2
+ # SPDX-License-Identifier: Apache-2
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ from enum import Enum
17
+ from typing import Dict
18
+
19
+ from pydantic import (
20
+ BaseModel,
21
+ ConfigDict,
22
+ Field,
23
+ field_validator,
24
+ )
25
+
26
+ from datacustomcode.named_credential.types.http_method import HTTPMethod
27
+
28
+
29
+ class HTTPRequest(BaseModel):
30
+ """External callout request. The endpoint and its auth are resolved
31
+ server-side from the Named Credential referenced by ``url``, which uses
32
+ ``callout:<NamedCredential>/<path>`` syntax."""
33
+
34
+ model_config = ConfigDict(populate_by_name=True)
35
+
36
+ url: str = Field(
37
+ ...,
38
+ min_length=1,
39
+ description="Symbolic Named Credential reference, "
40
+ "e.g. 'callout:<NamedCredential>/<path>'",
41
+ )
42
+ method: str = Field(
43
+ default="GET",
44
+ description="HTTP method (GET, POST, PUT, DELETE, PATCH, HEAD, OPTIONS)",
45
+ )
46
+ headers: Dict[str, str] = Field(default_factory=dict, description="Request headers")
47
+
48
+ @field_validator("method", mode="before")
49
+ @classmethod
50
+ def _normalize_method(cls, value: object) -> str:
51
+ # Accept str, this module's HTTPMethod, or http.HTTPMethod (3.11+).
52
+ if isinstance(value, Enum):
53
+ method = str(value.value)
54
+ else:
55
+ method = str(value)
56
+ method = method.upper()
57
+ if method not in {m.value for m in HTTPMethod}:
58
+ supported = ", ".join(m.value for m in HTTPMethod)
59
+ raise ValueError(
60
+ f"Unsupported HTTP method '{method}'. "
61
+ f"Named Credential callouts support {supported}."
62
+ )
63
+ return method
@@ -0,0 +1,55 @@
1
+ # Copyright (c) 2025, Salesforce, Inc.
2
+ # SPDX-License-Identifier: Apache-2
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ from typing import Dict, Union
17
+
18
+ from datacustomcode.named_credential.types.http_method import HTTPMethod
19
+ from datacustomcode.named_credential.types.http_request import HTTPRequest
20
+
21
+
22
+ class HTTPRequestBuilder:
23
+ def __init__(self) -> None:
24
+ self._url = ""
25
+ self._method: Union[str, HTTPMethod] = HTTPMethod.GET
26
+ self._headers: Dict[str, str] = {}
27
+
28
+ def set_url(self, url: str) -> "HTTPRequestBuilder":
29
+ """Set the symbolic Named Credential reference.
30
+
31
+ Args:
32
+ url: e.g. 'callout:<NamedCredential>/<path>'
33
+ """
34
+ self._url = url
35
+ return self
36
+
37
+ def set_method(self, method: Union[str, HTTPMethod]) -> "HTTPRequestBuilder":
38
+ """Set the HTTP method.
39
+
40
+ Accepts this module's ``HTTPMethod``, ``http.HTTPMethod`` (Python 3.11+),
41
+ or a plain string such as ``"GET"``.
42
+ """
43
+ self._method = method
44
+ return self
45
+
46
+ def set_headers(self, headers: Dict[str, str]) -> "HTTPRequestBuilder":
47
+ self._headers = headers
48
+ return self
49
+
50
+ def build(self) -> HTTPRequest:
51
+ return HTTPRequest(
52
+ url=self._url,
53
+ method=self._method, # type: ignore[arg-type]
54
+ headers=self._headers,
55
+ )
@@ -0,0 +1,43 @@
1
+ # Copyright (c) 2025, Salesforce, Inc.
2
+ # SPDX-License-Identifier: Apache-2
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ from typing import Dict
17
+
18
+ from pydantic import BaseModel, Field
19
+
20
+
21
+ class HTTPResponse(BaseModel):
22
+ """Response from a Named Credential external callout."""
23
+
24
+ status_code: int = Field(..., description="HTTP status code", ge=0)
25
+ headers: Dict[str, str] = Field(
26
+ default_factory=dict, description="Response headers"
27
+ )
28
+ body: str = Field(
29
+ default="",
30
+ description="Raw response body, verbatim (any format). The SDK does not "
31
+ "parse it; the caller decodes as needed (e.g. json.loads). Empty string "
32
+ "if the response had no body.",
33
+ )
34
+
35
+ @property
36
+ def is_success(self) -> bool:
37
+ """Check if the request succeeded (2xx)."""
38
+ return 200 <= self.status_code < 300
39
+
40
+ @property
41
+ def is_error(self) -> bool:
42
+ """Check if the request failed."""
43
+ return not self.is_success
@@ -0,0 +1,24 @@
1
+ # Copyright (c) 2025, Salesforce, Inc.
2
+ # SPDX-License-Identifier: Apache-2
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ from typing import Any, Dict
17
+
18
+ from datacustomcode.named_credential.types.http_response import HTTPResponse
19
+
20
+
21
+ class HTTPResponseBuilder:
22
+ @staticmethod
23
+ def build(response_dict: Dict[str, Any]) -> HTTPResponse:
24
+ return HTTPResponse.model_validate(response_dict)
@@ -0,0 +1,105 @@
1
+ # Copyright (c) 2025, Salesforce, Inc.
2
+ # SPDX-License-Identifier: Apache-2
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ from typing import (
17
+ ClassVar,
18
+ Generic,
19
+ Type,
20
+ TypeVar,
21
+ Union,
22
+ )
23
+
24
+ from datacustomcode.common_config import (
25
+ BaseConfig,
26
+ BaseObjectConfig,
27
+ default_config_file,
28
+ )
29
+ from datacustomcode.named_credential.base import NamedCredential
30
+ from datacustomcode.named_credential.spark_base import SparkNamedCredential
31
+
32
+ _N = TypeVar("_N", bound=NamedCredential)
33
+ _S = TypeVar("_S", bound=SparkNamedCredential)
34
+
35
+
36
+ class NamedCredentialObjectConfig(BaseObjectConfig, Generic[_N]):
37
+ type_to_create: ClassVar[Type[NamedCredential]] = NamedCredential # type: ignore[type-abstract]
38
+
39
+ def to_object(self) -> NamedCredential:
40
+ type_ = self.type_to_create.subclass_from_config_name(self.type_config_name)
41
+ return type_(**self.options)
42
+
43
+
44
+ class NamedCredentialConfig(BaseConfig):
45
+ named_credential_config: Union[
46
+ NamedCredentialObjectConfig[NamedCredential], None
47
+ ] = None
48
+
49
+ def update(self, other: "NamedCredentialConfig") -> "NamedCredentialConfig":
50
+ def merge(
51
+ config_a: Union[NamedCredentialObjectConfig, None],
52
+ config_b: Union[NamedCredentialObjectConfig, None],
53
+ ) -> Union[NamedCredentialObjectConfig, None]:
54
+ if config_a is not None and config_a.force:
55
+ return config_a
56
+ if config_b:
57
+ return config_b
58
+ return config_a
59
+
60
+ self.named_credential_config = merge(
61
+ self.named_credential_config, other.named_credential_config
62
+ )
63
+ return self
64
+
65
+
66
+ class SparkNamedCredentialObjectConfig(BaseObjectConfig, Generic[_S]):
67
+ type_to_create: ClassVar[Type[SparkNamedCredential]] = SparkNamedCredential # type: ignore[type-abstract]
68
+
69
+ def to_object(self) -> SparkNamedCredential:
70
+ type_ = self.type_to_create.subclass_from_config_name(self.type_config_name)
71
+ return type_(**self.options)
72
+
73
+
74
+ class SparkNamedCredentialConfig(BaseConfig):
75
+ spark_named_credential_config: Union[
76
+ SparkNamedCredentialObjectConfig[SparkNamedCredential], None
77
+ ] = None
78
+
79
+ def update(
80
+ self, other: "SparkNamedCredentialConfig"
81
+ ) -> "SparkNamedCredentialConfig":
82
+ def merge(
83
+ config_a: Union[SparkNamedCredentialObjectConfig, None],
84
+ config_b: Union[SparkNamedCredentialObjectConfig, None],
85
+ ) -> Union[SparkNamedCredentialObjectConfig, None]:
86
+ if config_a is not None and config_a.force:
87
+ return config_a
88
+ if config_b:
89
+ return config_b
90
+ return config_a
91
+
92
+ self.spark_named_credential_config = merge(
93
+ self.spark_named_credential_config, other.spark_named_credential_config
94
+ )
95
+ return self
96
+
97
+
98
+ # Global Named Credential config instance
99
+ named_credential_config = NamedCredentialConfig()
100
+ named_credential_config.load(default_config_file())
101
+
102
+
103
+ # Global Spark Named Credential config instance
104
+ spark_named_credential_config = SparkNamedCredentialConfig()
105
+ spark_named_credential_config.load(default_config_file())
datacustomcode/run.py CHANGED
@@ -27,6 +27,7 @@ from typing import (
27
27
  from datacustomcode.config import config
28
28
  from datacustomcode.einstein_predictions_config import einstein_predictions_config
29
29
  from datacustomcode.llm_gateway_config import llm_gateway_config
30
+ from datacustomcode.named_credential_config import named_credential_config
30
31
  from datacustomcode.scan import find_base_directory, get_package_type
31
32
 
32
33
 
@@ -55,6 +56,9 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
55
56
  _set_config_option(
56
57
  llm_gateway_config.llm_gateway_config, config_key, sf_cli_org
57
58
  )
59
+ _set_config_option(
60
+ named_credential_config.named_credential_config, config_key, sf_cli_org
61
+ )
58
62
  elif profile != "default":
59
63
  config_key = "credentials_profile"
60
64
  _set_config_option(config.reader_config, config_key, profile)
@@ -63,6 +67,9 @@ def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
63
67
  einstein_predictions_config.einstein_predictions_config, config_key, profile
64
68
  )
65
69
  _set_config_option(llm_gateway_config.llm_gateway_config, config_key, profile)
70
+ _set_config_option(
71
+ named_credential_config.named_credential_config, config_key, profile
72
+ )
66
73
 
67
74
 
68
75
  def run_entrypoint(
@@ -0,0 +1,119 @@
1
+ # Chunking with a Gemini Named Credential Callout
2
+
3
+ Splits each input document into paragraph-sized chunks and calls Google's
4
+ **Gemini** `generateContent` API for every chunk. The model returns a summary,
5
+ category, sentiment, and topics, which are attached to the chunk as citations so
6
+ the search index can filter and rank on them. Gemini is reached through a
7
+ **Named Credential**, so this code never handles the endpoint URL or the API key.
8
+
9
+ ## How the callout works
10
+
11
+ ```python
12
+ CALLOUT_URL = "callout:gemini" # callout:<NC name>[/<path>]
13
+
14
+ request = (
15
+ HTTPRequestBuilder()
16
+ .set_url(CALLOUT_URL)
17
+ .set_method(HTTPMethod.POST)
18
+ .set_headers({"Content-Type": "application/json"})
19
+ .build()
20
+ )
21
+ # Body is sent verbatim (serialize it yourself); the response body is a raw string.
22
+ response = runtime.named_credential.request(request, json.dumps(payload))
23
+ if response.is_success:
24
+ envelope = json.loads(response.body)
25
+ text = envelope["candidates"][0]["content"]["parts"][0]["text"]
26
+ ```
27
+
28
+ The request asks for `responseMimeType: application/json` with a `responseSchema`,
29
+ so Gemini returns the classification as a JSON string in
30
+ `candidates[0].content.parts[0].text` — decode it, then decode that text again.
31
+
32
+ The `gemini` Named Credential's URL already includes the full
33
+ `/v1beta/models/<model>:generateContent` path, so the callout is just
34
+ `callout:gemini` with **no path suffix** (anything after the name is appended to
35
+ the credential's URL).
36
+
37
+ ## Configure the Named Credential
38
+
39
+ 1. Create an **External Credential** (e.g. `google_api_key`) that injects your
40
+ Gemini API key as the `X-goog-api-key` header.
41
+ 2. Create a **Named Credential** named `gemini`:
42
+ - **URL**: `https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent`
43
+ - **Enabled for Callouts** + **Generate Authorization Header**: on
44
+ - **External Credential**: `google_api_key`
45
+
46
+ ## Test locally
47
+
48
+ ```bash
49
+ DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG=/abs/path/to/external_callout_config.json \
50
+ sf data-code-extension function run \
51
+ --entrypoint payload/entrypoint.py \
52
+ --test-with payload/tests/test.json \
53
+ --target-org <your-org-alias>
54
+ ```
55
+
56
+ With `--target-org` the SDK fetches only the **URL** from the org's Named
57
+ Credential; **auth is always taken from `external_callout_config.json`** locally
58
+ (the org's External Credential is used only in the Data Cloud runtime). So the
59
+ `X-goog-api-key` must be in the local config for a local test. Omit
60
+ `--target-org` to run fully offline using `target_url`.
61
+
62
+ ```json
63
+ {
64
+ "credentials": {
65
+ "callout:gemini": {
66
+ "auth_type": "Custom",
67
+ "custom_headers": { "X-goog-api-key": "YOUR_GEMINI_API_KEY" },
68
+ "target_url": "https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent"
69
+ }
70
+ }
71
+ }
72
+ ```
73
+
74
+ Place `external_callout_config.json` in the **parent of your payload folder** (or
75
+ point `DATACUSTOMCODE_EXTERNAL_CALLOUT_CONFIG` at it). It is never packaged into
76
+ the deployment zip. Get a key from [Google AI Studio](https://aistudio.google.com/apikey);
77
+ **do not commit it.**
78
+
79
+ ## Auth types
80
+
81
+ `auth_type` selects how auth is injected for local testing. It should mirror the
82
+ External Credential your Named Credential uses in the org, so local and deployed
83
+ runs behave the same. This example uses `Custom` (Gemini's `X-goog-api-key`);
84
+ all four supported types:
85
+
86
+ ```json
87
+ {
88
+ "credentials": {
89
+ "callout:my_custom_api": {
90
+ "auth_type": "Custom",
91
+ "custom_headers": { "X-goog-api-key": "YOUR_API_KEY" }
92
+ },
93
+ "callout:my_basic_api": {
94
+ "auth_type": "Basic",
95
+ "username": "svc_user",
96
+ "password": "YOUR_PASSWORD"
97
+ },
98
+ "callout:my_oauth_api": {
99
+ "auth_type": "OAuth",
100
+ "access_token": "YOUR_ACCESS_TOKEN"
101
+ },
102
+ "callout:my_jwt_api": {
103
+ "auth_type": "Jwt",
104
+ "token": "YOUR_JWT"
105
+ }
106
+ }
107
+ }
108
+ ```
109
+
110
+ | `auth_type` | Fields read | Header sent |
111
+ | ----------- | ------------------------------- | --------------------------------------- |
112
+ | `Basic` | `username`, `password` | `Authorization: Basic <base64 user:pw>` |
113
+ | `Custom` | `custom_headers` (sent verbatim)| the headers you list |
114
+ | `OAuth` | `access_token` or `token` | `Authorization: Bearer <token>` |
115
+ | `Jwt` | `access_token` or `token` | `Authorization: Bearer <token>` |
116
+
117
+ `OAuth`/`Jwt` take a token you supply for the local run — the SDK does not fetch
118
+ or refresh it. In the Data Cloud runtime the Named Credential handles token
119
+ acquisition; this local config only stands in for that during testing.
@@ -0,0 +1,3 @@
1
+ {
2
+ "entryPoint": "entrypoint.py"
3
+ }
@@ -0,0 +1,161 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2025, Salesforce, Inc.
3
+ # SPDX-License-Identifier: Apache-2
4
+
5
+ """
6
+ Document Chunking with a Gemini Named Credential Callout
7
+
8
+ Splits each input document into paragraph-sized chunks and classifies every
9
+ chunk via Google's Gemini ``generateContent`` API, reached through a Named
10
+ Credential (``callout:gemini``) so the endpoint URL and API key are resolved
11
+ outside this code. The classification is attached to each chunk as citations.
12
+ """
13
+
14
+ import json
15
+ import logging
16
+
17
+ from datacustomcode.function import Runtime
18
+ from datacustomcode.function.feature_types.chunking import (
19
+ ChunkType,
20
+ SearchIndexChunkingV1Output,
21
+ SearchIndexChunkingV1Request,
22
+ SearchIndexChunkingV1Response,
23
+ )
24
+ from datacustomcode.named_credential.types.http_method import HTTPMethod
25
+ from datacustomcode.named_credential.types.http_request_builder import (
26
+ HTTPRequestBuilder,
27
+ )
28
+
29
+ logger = logging.getLogger(__name__)
30
+ logging.basicConfig(level=logging.INFO)
31
+
32
+ CALLOUT_URL = "callout:gemini"
33
+
34
+ _ANALYSIS_FIELDS = ("summary", "category", "sentiment")
35
+
36
+ _PROMPT = (
37
+ "Analyze the following document chunk and classify it. Respond with its "
38
+ "one-sentence summary, a single-word category, overall sentiment "
39
+ "(positive, negative, or neutral), and up to five key topics.\n\nChunk:\n"
40
+ )
41
+
42
+ # Force Gemini to return the classification as JSON in a fixed shape.
43
+ _GENERATION_CONFIG = {
44
+ "responseMimeType": "application/json",
45
+ "responseSchema": {
46
+ "type": "object",
47
+ "properties": {
48
+ "summary": {"type": "string"},
49
+ "category": {"type": "string"},
50
+ "sentiment": {"type": "string"},
51
+ "topics": {"type": "array", "items": {"type": "string"}},
52
+ },
53
+ "required": ["summary", "category", "sentiment", "topics"],
54
+ },
55
+ }
56
+
57
+
58
+ def _chunk_text(text: str, max_words: int = 80) -> list[str]:
59
+ """Split text into paragraph-aligned chunks of at most ``max_words`` words."""
60
+ paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()]
61
+
62
+ chunks: list[str] = []
63
+ current: list[str] = []
64
+ current_words = 0
65
+
66
+ for paragraph in paragraphs:
67
+ paragraph_words = len(paragraph.split())
68
+ if current and current_words + paragraph_words > max_words:
69
+ chunks.append("\n\n".join(current))
70
+ current = []
71
+ current_words = 0
72
+ current.append(paragraph)
73
+ current_words += paragraph_words
74
+
75
+ if current:
76
+ chunks.append("\n\n".join(current))
77
+
78
+ return chunks
79
+
80
+
81
+ def _extract_model_json(body: str) -> dict:
82
+ """Decode the model's JSON classification from a Gemini response.
83
+
84
+ The generated text sits at ``candidates[0].content.parts[0].text`` and is
85
+ itself a JSON string, so decode twice. Any malformed layer yields ``{}``.
86
+ """
87
+ try:
88
+ envelope = json.loads(body) if body else {}
89
+ except json.JSONDecodeError:
90
+ return {}
91
+
92
+ try:
93
+ text = envelope["candidates"][0]["content"]["parts"][0]["text"]
94
+ except (KeyError, IndexError, TypeError):
95
+ return {}
96
+
97
+ try:
98
+ payload = json.loads(text)
99
+ except json.JSONDecodeError:
100
+ return {}
101
+ return payload if isinstance(payload, dict) else {}
102
+
103
+
104
+ def _analyze_chunk(chunk_text: str, runtime: Runtime) -> dict[str, str]:
105
+ """Classify one chunk via the Gemini callout and return it as citations."""
106
+ request = (
107
+ HTTPRequestBuilder()
108
+ .set_url(CALLOUT_URL)
109
+ .set_method(HTTPMethod.POST)
110
+ .set_headers({"Content-Type": "application/json", "Accept": "application/json"})
111
+ .build()
112
+ )
113
+
114
+ payload = {
115
+ "contents": [{"parts": [{"text": _PROMPT + chunk_text}]}],
116
+ "generationConfig": _GENERATION_CONFIG,
117
+ }
118
+ response = runtime.named_credential.request(request, json.dumps(payload))
119
+
120
+ # Don't raise: a single failed callout shouldn't abort the whole job.
121
+ if not response.is_success:
122
+ logger.error(f"Gemini callout failed with status {response.status_code}")
123
+ return {"analysis_status": "failed", "http_status": str(response.status_code)}
124
+
125
+ data = _extract_model_json(response.body)
126
+ citations = {"analysis_status": "success"}
127
+ for field in _ANALYSIS_FIELDS:
128
+ value = data.get(field)
129
+ citations[field] = str(value) if value is not None else "unavailable"
130
+
131
+ topics = data.get("topics")
132
+ if isinstance(topics, list):
133
+ citations["topics"] = ", ".join(str(topic) for topic in topics)
134
+
135
+ return citations
136
+
137
+
138
+ def function(
139
+ request: SearchIndexChunkingV1Request, runtime: Runtime
140
+ ) -> SearchIndexChunkingV1Response:
141
+ """Chunk each input document and classify every chunk via the Gemini API."""
142
+ logger.info(f"Received {len(request.input)} documents to chunk")
143
+
144
+ chunks = []
145
+ chunk_id = 1
146
+
147
+ for doc in request.input:
148
+ for chunk_text in _chunk_text(doc.text):
149
+ citations = _analyze_chunk(chunk_text, runtime)
150
+
151
+ chunk = SearchIndexChunkingV1Output(
152
+ text=chunk_text,
153
+ seq_no=chunk_id,
154
+ chunk_type=ChunkType.TEXT,
155
+ citations=citations,
156
+ )
157
+ chunks.append(chunk)
158
+ chunk_id += 1
159
+
160
+ logger.info(f"Produced {len(chunks)} classified chunks")
161
+ return SearchIndexChunkingV1Response(output=chunks)
@@ -0,0 +1,11 @@
1
+ {
2
+ "credentials": {
3
+ "callout:gemini": {
4
+ "auth_type": "Custom",
5
+ "custom_headers": {
6
+ "X-goog-api-key": "YOUR_API_KEY"
7
+ },
8
+ "target_url": "https://generativelanguage.googleapis.com/v1beta/models/gemini-flash-latest:generateContent"
9
+ }
10
+ }
11
+ }
@@ -0,0 +1,16 @@
1
+ {
2
+ "input": [
3
+ {
4
+ "text": "Product Review: Northstar Analytics\n\nWe rolled Northstar out to our whole revenue team last quarter and the difference has been night and day. Dashboards that used to take our analysts a full day to assemble now refresh in seconds, and the natural-language query box means our account executives can answer their own questions without filing a ticket.\n\nOnboarding was smoother than any tool we have adopted in years. The guided setup imported our Salesforce data on the first try and the sample templates gave us something useful on day one. Support answered our two questions within the hour. Easily the best purchase decision we made this year."
5
+ },
6
+ {
7
+ "text": "Support Ticket #48210: Repeated timeouts on scheduled exports\n\nFor the third week running our nightly export to the data warehouse has failed silently. There is no alert, no email, nothing in the activity log, and we only find out when the morning report is empty and the leadership meeting has no numbers.\n\nI have raised this twice already and both times the ticket was closed as resolved without anyone actually contacting me. This is costing us real credibility internally and I am extremely frustrated. If the connector cannot handle our volume we need to know now so we can plan a migration, because right now the product is not doing the one job we bought it for."
8
+ },
9
+ {
10
+ "text": "Renewal Feedback: mixed feelings heading into year two\n\nThe core product is genuinely good. The reporting engine is fast, the permissions model is granular enough for our compliance team, and our analysts like working in it. On the functionality alone I would renew without hesitation.\n\nWhat gives me pause is the pricing. The per-seat cost jumped noticeably at renewal and several add-ons that used to be included are now separate line items. The value is still there, but the conversation with my finance team was harder than it should have been, and I would like more transparency before the next cycle."
11
+ },
12
+ {
13
+ "text": "Feature Request: scheduled report subscriptions\n\nWe would like the ability to subscribe internal stakeholders to a report on a recurring schedule so a PDF lands in their inbox every Monday morning. Today we export manually and forward it, which is workable but easy to forget.\n\nA few teams have asked whether subscriptions could support filtered views per recipient, for example each regional manager receiving only their own territory. Not urgent for us, but it would remove a recurring bit of manual work and is something a couple of competing tools already offer."
14
+ }
15
+ ]
16
+ }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: salesforce-data-customcode
3
- Version: 6.0.8.dev1
3
+ Version: 6.1.0.dev2
4
4
  Summary: Data Cloud Custom Code SDK
5
5
  License-Expression: Apache-2.0
6
6
  License-File: LICENSE.txt