salesforce-data-customcode 7.0.0rc1__tar.gz → 7.0.0rc3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/PKG-INFO +1 -1
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/pyproject.toml +1 -1
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/__init__.py +10 -0
- salesforce_data_customcode-7.0.0rc3/src/datacustomcode/io/cdf.py +34 -0
- salesforce_data_customcode-7.0.0rc3/src/datacustomcode/io/reader/local_deltas.py +164 -0
- salesforce_data_customcode-7.0.0rc3/src/datacustomcode/io/reader/streaming_seeder.py +114 -0
- salesforce_data_customcode-7.0.0rc3/src/datacustomcode/io/writer/local_deltas.py +133 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/run.py +25 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/spark/column_hints.py +61 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/template.py +20 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/LICENSE.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/README.md +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/auth.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/cli.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/client.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/cmd.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/common_config.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/config.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/config.yaml +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/constants.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/credentials.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/deploy.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_platform_client.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_platform_config.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/errors.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/impl/default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/spark_base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/spark_default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions/types.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/einstein_predictions_config.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/file/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/file/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/file/path/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/file/path/default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/function/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/function/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/function/feature_types/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/function/feature_types/chunking.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/function/runtime.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/function_utils.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/reader/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/reader/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/reader/query_api.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/reader/sf_cli.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/reader/utils.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/writer/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/writer/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/writer/csv.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/io/writer/print.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/errors.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/spark_base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/spark_default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/types/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/types/generate_text_request.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/types/generate_text_request_builder.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/types/generate_text_response.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway/types/generate_text_response_builder.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/llm_gateway_config.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/mixin.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/direct/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/direct/auth.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/direct/credentials.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/direct/transport.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/direct/url_resolver.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/errors.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/spark_base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/spark_default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/types/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/types/http_method.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/types/http_request.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/types/http_request_builder.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/types/http_response.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential/types/http_response_builder.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/named_credential_config.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/py.typed +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/scan.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/spark/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/spark/base.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/spark/default.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/__init__.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/.devcontainer/devcontainer.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/Dockerfile.dependencies +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/README.md +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/build_native_dependencies.sh +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/chunking/payload/config.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/chunking/payload/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/chunking/requirements.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_llm/config.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_llm/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_llm/files/chunking_prompt.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_llm/tests/test.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_prediction/config.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_prediction/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/example/chunking_with_prediction/tests/test.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/payload/config.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/payload/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/payload/utility.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/requirements-dev.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/function/requirements.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/.devcontainer/devcontainer.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/Dockerfile +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/Dockerfile.dependencies +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/README.md +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/account.ipynb +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/build_native_dependencies.sh +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/examples/employee_hierarchy/employee_data.csv +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/examples/employee_hierarchy/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/examples/streaming_deltas/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/jupyterlab.sh +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/payload/config.json +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/payload/entrypoint.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/requirements-dev.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/templates/script/requirements.txt +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/token_provider.py +0 -0
- {salesforce_data_customcode-7.0.0rc1 → salesforce_data_customcode-7.0.0rc3}/src/datacustomcode/version.py +0 -0
|
@@ -24,6 +24,8 @@ __all__ = [
|
|
|
24
24
|
"Credentials",
|
|
25
25
|
"DefaultSparkEinsteinPredictions",
|
|
26
26
|
"DefaultSparkLLMGateway",
|
|
27
|
+
"LocalDeltasReader",
|
|
28
|
+
"LocalDeltasWriter",
|
|
27
29
|
"PrintDataCloudWriter",
|
|
28
30
|
"QueryAPIDataCloudReader",
|
|
29
31
|
"SparkEinsteinPredictions",
|
|
@@ -60,6 +62,14 @@ def __getattr__(name: str):
|
|
|
60
62
|
from datacustomcode.io.reader.query_api import QueryAPIDataCloudReader
|
|
61
63
|
|
|
62
64
|
return QueryAPIDataCloudReader
|
|
65
|
+
elif name == "LocalDeltasReader":
|
|
66
|
+
from datacustomcode.io.reader.local_deltas import LocalDeltasReader
|
|
67
|
+
|
|
68
|
+
return LocalDeltasReader
|
|
69
|
+
elif name == "LocalDeltasWriter":
|
|
70
|
+
from datacustomcode.io.writer.local_deltas import LocalDeltasWriter
|
|
71
|
+
|
|
72
|
+
return LocalDeltasWriter
|
|
63
73
|
elif name == "SparkLLMGateway":
|
|
64
74
|
from datacustomcode.llm_gateway import SparkLLMGateway
|
|
65
75
|
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from typing import Final
|
|
19
|
+
|
|
20
|
+
from datacustomcode.io.writer.base import MERGE_RECORD_TYPE_COLUMN, MergeRecordType
|
|
21
|
+
|
|
22
|
+
COMMIT_VERSION: Final = "_commit_version"
|
|
23
|
+
COMMIT_TIMESTAMP: Final = "_commit_timestamp"
|
|
24
|
+
MERGE_RECORD_TYPE: Final = MERGE_RECORD_TYPE_COLUMN
|
|
25
|
+
|
|
26
|
+
CDF_METADATA_COLUMNS: Final = (COMMIT_VERSION, COMMIT_TIMESTAMP, MERGE_RECORD_TYPE)
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"COMMIT_VERSION",
|
|
30
|
+
"COMMIT_TIMESTAMP",
|
|
31
|
+
"MERGE_RECORD_TYPE",
|
|
32
|
+
"CDF_METADATA_COLUMNS",
|
|
33
|
+
"MergeRecordType",
|
|
34
|
+
]
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import logging
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
import sys
|
|
22
|
+
from typing import (
|
|
23
|
+
TYPE_CHECKING,
|
|
24
|
+
Optional,
|
|
25
|
+
Union,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
from datacustomcode.config import config
|
|
29
|
+
from datacustomcode.io import cdf
|
|
30
|
+
from datacustomcode.io.reader.base import BaseDataCloudReader
|
|
31
|
+
from datacustomcode.io.reader.query_api import QueryAPIDataCloudReader
|
|
32
|
+
|
|
33
|
+
if TYPE_CHECKING:
|
|
34
|
+
from pyspark.sql import DataFrame as PySparkDataFrame, SparkSession
|
|
35
|
+
from pyspark.sql.types import AtomicType, StructType
|
|
36
|
+
|
|
37
|
+
logger = logging.getLogger(__name__)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class LocalDeltasReader(BaseDataCloudReader):
|
|
41
|
+
"""Local reader for streaming transforms"""
|
|
42
|
+
|
|
43
|
+
CONFIG_NAME = "LocalDeltasReader"
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
spark: SparkSession,
|
|
48
|
+
credentials_profile: str = "default",
|
|
49
|
+
dataspace: Optional[str] = None,
|
|
50
|
+
sf_cli_org: Optional[str] = None,
|
|
51
|
+
default_row_limit: Optional[int] = None,
|
|
52
|
+
fixtures_root: str = "payload/streaming_fixtures",
|
|
53
|
+
) -> None:
|
|
54
|
+
super().__init__(spark)
|
|
55
|
+
self._fixtures_root = Path(fixtures_root)
|
|
56
|
+
self._credentials_profile = credentials_profile
|
|
57
|
+
self._dataspace = dataspace
|
|
58
|
+
self._sf_cli_org = sf_cli_org
|
|
59
|
+
# Reuse the batch reader for INITIAL_SYNC / REBUILD reads.
|
|
60
|
+
self._batch = QueryAPIDataCloudReader(
|
|
61
|
+
spark=spark,
|
|
62
|
+
credentials_profile=credentials_profile,
|
|
63
|
+
dataspace=dataspace,
|
|
64
|
+
sf_cli_org=sf_cli_org,
|
|
65
|
+
default_row_limit=default_row_limit,
|
|
66
|
+
)
|
|
67
|
+
self._current_layer = "dlo"
|
|
68
|
+
|
|
69
|
+
def read_dlo(
|
|
70
|
+
self,
|
|
71
|
+
name: str,
|
|
72
|
+
schema: Union[AtomicType, StructType, str, None] = None,
|
|
73
|
+
) -> PySparkDataFrame:
|
|
74
|
+
return self._batch.read_dlo(name, schema)
|
|
75
|
+
|
|
76
|
+
def read_dmo(
|
|
77
|
+
self,
|
|
78
|
+
name: str,
|
|
79
|
+
schema: Union[AtomicType, StructType, str, None] = None,
|
|
80
|
+
) -> PySparkDataFrame:
|
|
81
|
+
return self._batch.read_dmo(name, schema)
|
|
82
|
+
|
|
83
|
+
def read_dlo_deltas(self) -> PySparkDataFrame:
|
|
84
|
+
self._current_layer = "dlo"
|
|
85
|
+
return self._open_stream(self._streaming_source())
|
|
86
|
+
|
|
87
|
+
def read_dmo_deltas(self) -> PySparkDataFrame:
|
|
88
|
+
self._current_layer = "dmo"
|
|
89
|
+
return self._open_stream(self._streaming_source())
|
|
90
|
+
|
|
91
|
+
def _streaming_source(self) -> str:
|
|
92
|
+
source = config.streaming_source
|
|
93
|
+
if not source:
|
|
94
|
+
raise RuntimeError(
|
|
95
|
+
"No streaming source configured. Set streamingSource.name in "
|
|
96
|
+
"config.json."
|
|
97
|
+
)
|
|
98
|
+
return source
|
|
99
|
+
|
|
100
|
+
def _open_stream(self, name: str) -> PySparkDataFrame:
|
|
101
|
+
drop_dir = self._fixtures_root / name
|
|
102
|
+
|
|
103
|
+
if not (drop_dir / "_schema.json").exists():
|
|
104
|
+
seeded = self._try_seed(name)
|
|
105
|
+
if not seeded:
|
|
106
|
+
print(
|
|
107
|
+
f"\nStreaming source {self._current_layer}='{name}' is "
|
|
108
|
+
f"empty.\n"
|
|
109
|
+
f" Populate the stream source with at least one row in "
|
|
110
|
+
f"Data Cloud, then re-run `datacustomcode run`.\n"
|
|
111
|
+
)
|
|
112
|
+
sys.exit(0)
|
|
113
|
+
|
|
114
|
+
print(
|
|
115
|
+
f"\nWrote sample streaming fixture in: {drop_dir} "
|
|
116
|
+
f"based on the contents of the streaming source {name}.\n"
|
|
117
|
+
f" You may add more JSON files alongside it to simulate "
|
|
118
|
+
f"additional changes.\n"
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
schema = self._build_stream_schema(name)
|
|
122
|
+
return (
|
|
123
|
+
self.spark.readStream.format("json")
|
|
124
|
+
.schema(schema)
|
|
125
|
+
.option("maxFilesPerTrigger", 1) # one file = one batch
|
|
126
|
+
.option("latestFirst", "false") # oldest mtime first
|
|
127
|
+
.load(str(drop_dir))
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
def _build_stream_schema(self, name: str) -> "StructType":
|
|
131
|
+
"""Compose (source schema + CDF metadata columns) for readStream."""
|
|
132
|
+
from pyspark.sql.types import (
|
|
133
|
+
LongType,
|
|
134
|
+
StringType,
|
|
135
|
+
StructField,
|
|
136
|
+
StructType,
|
|
137
|
+
TimestampType,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
source_schema = self._resolve_source_schema(name)
|
|
141
|
+
cdf_fields = [
|
|
142
|
+
StructField(cdf.COMMIT_VERSION, LongType(), True),
|
|
143
|
+
StructField(cdf.COMMIT_TIMESTAMP, TimestampType(), True),
|
|
144
|
+
StructField(cdf.MERGE_RECORD_TYPE, StringType(), True),
|
|
145
|
+
]
|
|
146
|
+
return StructType(list(source_schema.fields) + cdf_fields)
|
|
147
|
+
|
|
148
|
+
def _resolve_source_schema(self, name: str) -> "StructType":
|
|
149
|
+
from pyspark.sql.types import StructType
|
|
150
|
+
|
|
151
|
+
schema_file = self._fixtures_root / name / "_schema.json"
|
|
152
|
+
return StructType.fromJson(json.loads(schema_file.read_text()))
|
|
153
|
+
|
|
154
|
+
def _try_seed(self, name: str) -> bool:
|
|
155
|
+
"""Creates source schema and sample change file."""
|
|
156
|
+
from datacustomcode.io.reader.streaming_seeder import StreamingSourceSeeder
|
|
157
|
+
|
|
158
|
+
seeder = StreamingSourceSeeder(
|
|
159
|
+
spark=self.spark,
|
|
160
|
+
credentials_profile=self._credentials_profile,
|
|
161
|
+
dataspace=self._dataspace,
|
|
162
|
+
sf_cli_org=self._sf_cli_org,
|
|
163
|
+
)
|
|
164
|
+
return seeder.seed_source(name, self._current_layer, str(self._fixtures_root))
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
import json
|
|
20
|
+
import math
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import (
|
|
23
|
+
TYPE_CHECKING,
|
|
24
|
+
Any,
|
|
25
|
+
Optional,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
from datacustomcode.io import cdf
|
|
29
|
+
from datacustomcode.io.reader.query_api import QueryAPIDataCloudReader
|
|
30
|
+
|
|
31
|
+
if TYPE_CHECKING:
|
|
32
|
+
import pandas as pd
|
|
33
|
+
from pyspark.sql import SparkSession
|
|
34
|
+
|
|
35
|
+
SEED_LIMIT = 10
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _clean_for_json(value: Any) -> Any:
|
|
39
|
+
if isinstance(value, float):
|
|
40
|
+
if math.isnan(value):
|
|
41
|
+
return None
|
|
42
|
+
if value.is_integer():
|
|
43
|
+
return int(value)
|
|
44
|
+
return value
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class StreamingSourceSeeder:
|
|
48
|
+
def __init__(
|
|
49
|
+
self,
|
|
50
|
+
spark: SparkSession,
|
|
51
|
+
credentials_profile: str = "default",
|
|
52
|
+
dataspace: Optional[str] = None,
|
|
53
|
+
sf_cli_org: Optional[str] = None,
|
|
54
|
+
) -> None:
|
|
55
|
+
self.spark = spark
|
|
56
|
+
self.reader = QueryAPIDataCloudReader(
|
|
57
|
+
spark=spark,
|
|
58
|
+
credentials_profile=credentials_profile,
|
|
59
|
+
dataspace=dataspace,
|
|
60
|
+
sf_cli_org=sf_cli_org,
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
def seed_source(self, name: str, layer: str, fixtures_root: str) -> bool:
|
|
64
|
+
"""Seed streaming fixtures and source schema.
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
``True`` when schema + fixtures are written
|
|
68
|
+
``False`` when the source is empty.
|
|
69
|
+
Query API errors (missing creds, missing
|
|
70
|
+
source) propagate as ``RuntimeError``.
|
|
71
|
+
"""
|
|
72
|
+
read_fn = self.reader.read_dlo if layer == "dlo" else self.reader.read_dmo
|
|
73
|
+
|
|
74
|
+
try:
|
|
75
|
+
head_df = read_fn(name).limit(1)
|
|
76
|
+
except Exception as exc:
|
|
77
|
+
raise RuntimeError(f"Failed to read {layer}='{name}': {exc}.") from exc
|
|
78
|
+
|
|
79
|
+
head_pandas = head_df.toPandas()
|
|
80
|
+
if len(head_pandas) == 0:
|
|
81
|
+
return False
|
|
82
|
+
|
|
83
|
+
schema = head_df.schema
|
|
84
|
+
out_dir = Path(fixtures_root) / name
|
|
85
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
86
|
+
schema_file_path = out_dir / "_schema.json"
|
|
87
|
+
schema_file_path.write_text(json.dumps(schema.jsonValue()))
|
|
88
|
+
print(
|
|
89
|
+
f"\nStreaming source schema written to {schema_file_path} "
|
|
90
|
+
f"({len(schema.fields)} fields)"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# Mix UPSERT and DELETE in the seed so the customer sees both
|
|
94
|
+
# operation types in the starter fixture. Annotated because
|
|
95
|
+
# pyspark's ``.toPandas()`` return type is ``PandasDataFrameLike``,
|
|
96
|
+
# which mypy doesn't recognize as having ``to_dict``.
|
|
97
|
+
snapshot: pd.DataFrame = read_fn(name).limit(SEED_LIMIT).toPandas()
|
|
98
|
+
seed_rows = []
|
|
99
|
+
for i, record in enumerate(snapshot.to_dict("records")):
|
|
100
|
+
op = cdf.MergeRecordType.DELETE if i % 2 else cdf.MergeRecordType.UPSERT
|
|
101
|
+
|
|
102
|
+
cleaned = {k: _clean_for_json(v) for k, v in record.items()}
|
|
103
|
+
seed_rows.append(
|
|
104
|
+
{
|
|
105
|
+
**cleaned,
|
|
106
|
+
cdf.COMMIT_VERSION: i + 1,
|
|
107
|
+
cdf.COMMIT_TIMESTAMP: datetime.now(timezone.utc).isoformat(),
|
|
108
|
+
cdf.MERGE_RECORD_TYPE: op.value,
|
|
109
|
+
}
|
|
110
|
+
)
|
|
111
|
+
(out_dir / "000_seed.json").write_text(
|
|
112
|
+
"\n".join(json.dumps(r, default=str) for r in seed_rows)
|
|
113
|
+
)
|
|
114
|
+
return True
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
# Copyright (c) 2025, Salesforce, Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import os
|
|
19
|
+
import tempfile
|
|
20
|
+
from typing import TYPE_CHECKING, Optional
|
|
21
|
+
from urllib.parse import urlparse
|
|
22
|
+
|
|
23
|
+
from datacustomcode.io import cdf
|
|
24
|
+
from datacustomcode.io.writer.base import BaseDataCloudWriter, WriteMode
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from pyspark.sql import DataFrame as PySparkDataFrame, SparkSession
|
|
28
|
+
from pyspark.sql.streaming import StreamingQuery
|
|
29
|
+
|
|
30
|
+
_HEADER_WARNING = (
|
|
31
|
+
" NOTE: The output below is based on example streaming changes "
|
|
32
|
+
"auto-generated by this CLI or generated by the user.\n"
|
|
33
|
+
" Once this code extension is deployed, Data Cloud will persist "
|
|
34
|
+
"only the final state per primary key.\n"
|
|
35
|
+
" This local preview shows every emitted row uncollapsed."
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class LocalDeltasWriter(BaseDataCloudWriter):
|
|
40
|
+
"""Local writer for streaming transforms."""
|
|
41
|
+
|
|
42
|
+
CONFIG_NAME = "LocalDeltasWriter"
|
|
43
|
+
|
|
44
|
+
def __init__(
|
|
45
|
+
self,
|
|
46
|
+
spark: SparkSession,
|
|
47
|
+
credentials_profile: str = "default",
|
|
48
|
+
dataspace: Optional[str] = None,
|
|
49
|
+
sf_cli_org: Optional[str] = None,
|
|
50
|
+
) -> None:
|
|
51
|
+
super().__init__(spark)
|
|
52
|
+
|
|
53
|
+
from datacustomcode.io.writer.print import PrintDataCloudWriter
|
|
54
|
+
|
|
55
|
+
self._batch_writer = PrintDataCloudWriter(
|
|
56
|
+
spark=spark,
|
|
57
|
+
credentials_profile=credentials_profile,
|
|
58
|
+
dataspace=dataspace,
|
|
59
|
+
sf_cli_org=sf_cli_org,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
def write_to_dlo(
|
|
63
|
+
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
|
|
64
|
+
) -> None:
|
|
65
|
+
return self._batch_writer.write_to_dlo(name, dataframe, write_mode)
|
|
66
|
+
|
|
67
|
+
def write_to_dmo(
|
|
68
|
+
self, name: str, dataframe: PySparkDataFrame, write_mode: WriteMode
|
|
69
|
+
) -> None:
|
|
70
|
+
return self._batch_writer.write_to_dmo(name, dataframe, write_mode)
|
|
71
|
+
|
|
72
|
+
def auto_write_to_dlo(self, name: str, dataframe: PySparkDataFrame) -> None:
|
|
73
|
+
return self._batch_writer.auto_write_to_dlo(name, dataframe)
|
|
74
|
+
|
|
75
|
+
def auto_write_to_dmo(self, name: str, dataframe: PySparkDataFrame) -> None:
|
|
76
|
+
return self._batch_writer.auto_write_to_dmo(name, dataframe)
|
|
77
|
+
|
|
78
|
+
@staticmethod
|
|
79
|
+
def _batch_source_file(batch_df: "PySparkDataFrame") -> str:
|
|
80
|
+
"""Return the fixture file backing this micro-batch, relative to CWD"""
|
|
81
|
+
from pyspark.sql import functions as F
|
|
82
|
+
|
|
83
|
+
file_row = batch_df.select(F.input_file_name().alias("_src")).limit(1).collect()
|
|
84
|
+
if not file_row:
|
|
85
|
+
return "?"
|
|
86
|
+
abs_path = str(urlparse(str(file_row[0]["_src"])).path)
|
|
87
|
+
try:
|
|
88
|
+
return os.path.relpath(abs_path)
|
|
89
|
+
except Exception:
|
|
90
|
+
return abs_path
|
|
91
|
+
|
|
92
|
+
def write_dlo_deltas(
|
|
93
|
+
self, name: str, dataframe: PySparkDataFrame, **kwargs
|
|
94
|
+
) -> StreamingQuery:
|
|
95
|
+
if not name:
|
|
96
|
+
raise ValueError("DLO name must be provided.")
|
|
97
|
+
|
|
98
|
+
def _preview(batch_df, batch_id):
|
|
99
|
+
from pyspark.sql import functions as F
|
|
100
|
+
|
|
101
|
+
source_file = self._batch_source_file(batch_df)
|
|
102
|
+
|
|
103
|
+
output_batch = batch_df.withColumn(
|
|
104
|
+
"_operation",
|
|
105
|
+
F.when(
|
|
106
|
+
F.col(cdf.MERGE_RECORD_TYPE) == cdf.MergeRecordType.DELETE.value,
|
|
107
|
+
F.lit("DELETE"),
|
|
108
|
+
).otherwise(F.lit("UPSERT")),
|
|
109
|
+
).drop(
|
|
110
|
+
cdf.COMMIT_VERSION,
|
|
111
|
+
cdf.COMMIT_TIMESTAMP,
|
|
112
|
+
cdf.MERGE_RECORD_TYPE,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
row_count = output_batch.count()
|
|
116
|
+
print(
|
|
117
|
+
f"\nTarget={name} batch_id={batch_id} "
|
|
118
|
+
f"file={source_file} rows={row_count}"
|
|
119
|
+
)
|
|
120
|
+
print(_HEADER_WARNING)
|
|
121
|
+
output_batch.show(truncate=False)
|
|
122
|
+
|
|
123
|
+
checkpoint_dir = tempfile.mkdtemp(prefix=f"local-deltas-ckpt-{name}-")
|
|
124
|
+
return (
|
|
125
|
+
dataframe.writeStream.foreachBatch(_preview)
|
|
126
|
+
.option("checkpointLocation", checkpoint_dir)
|
|
127
|
+
# AvailableNow: drain every fixture file then
|
|
128
|
+
# terminate. The user's `query.awaitTermination()`
|
|
129
|
+
# returns without needing a timeout, so
|
|
130
|
+
# `datacustomcode run` exits deterministically.
|
|
131
|
+
.trigger(availableNow=True)
|
|
132
|
+
.start()
|
|
133
|
+
)
|
|
@@ -52,6 +52,27 @@ def _read_streaming_source(config_json: dict) -> Optional[str]:
|
|
|
52
52
|
return str(name) if name else None
|
|
53
53
|
|
|
54
54
|
|
|
55
|
+
def _project_config_yaml(entrypoint: str) -> Optional[str]:
|
|
56
|
+
"""Locate a project-local ``config.yaml`` next to ``payload/``.
|
|
57
|
+
|
|
58
|
+
A streaming project scaffolded by
|
|
59
|
+
``datacustomcode init --use-in-feature StreamingTransform`` has
|
|
60
|
+
``config.yaml`` at the project root.
|
|
61
|
+
|
|
62
|
+
Layout::
|
|
63
|
+
|
|
64
|
+
<project_root>/
|
|
65
|
+
config.yaml <-- this file (streaming profile)
|
|
66
|
+
payload/
|
|
67
|
+
entrypoint.py <-- passed as `entrypoint`
|
|
68
|
+
config.json
|
|
69
|
+
"""
|
|
70
|
+
entrypoint_dir = os.path.dirname(os.path.abspath(entrypoint))
|
|
71
|
+
project_root = os.path.dirname(entrypoint_dir)
|
|
72
|
+
candidate = os.path.join(project_root, "config.yaml")
|
|
73
|
+
return candidate if os.path.exists(candidate) else None
|
|
74
|
+
|
|
75
|
+
|
|
55
76
|
def _update_config_options(profile: Optional[str], sf_cli_org: Optional[str]):
|
|
56
77
|
if sf_cli_org:
|
|
57
78
|
config_key = "sf_cli_org"
|
|
@@ -136,6 +157,10 @@ def run_entrypoint(
|
|
|
136
157
|
# Load config file first
|
|
137
158
|
if config_file:
|
|
138
159
|
config.load(config_file)
|
|
160
|
+
else:
|
|
161
|
+
project_config = _project_config_yaml(entrypoint)
|
|
162
|
+
if project_config:
|
|
163
|
+
config.load(project_config)
|
|
139
164
|
|
|
140
165
|
# Add dataspace to reader and writer config options
|
|
141
166
|
_set_config_option(config.reader_config, "dataspace", dataspace)
|
|
@@ -28,7 +28,14 @@ error that was already going to be raised, and only for a pure casing mismatch.
|
|
|
28
28
|
"""
|
|
29
29
|
from __future__ import annotations
|
|
30
30
|
|
|
31
|
+
import importlib.abc
|
|
31
32
|
import logging
|
|
33
|
+
import sys
|
|
34
|
+
from typing import TYPE_CHECKING
|
|
35
|
+
|
|
36
|
+
if TYPE_CHECKING:
|
|
37
|
+
from importlib.machinery import ModuleSpec
|
|
38
|
+
from types import ModuleType
|
|
32
39
|
|
|
33
40
|
logger = logging.getLogger(__name__)
|
|
34
41
|
|
|
@@ -125,9 +132,63 @@ def _install_getattr_hook() -> None:
|
|
|
125
132
|
|
|
126
133
|
|
|
127
134
|
def install_column_casing_hints() -> None:
|
|
135
|
+
"""Arrange for the column-casing hint hooks to be installed.
|
|
136
|
+
|
|
137
|
+
Never imports pyspark directly. If pyspark is already loaded, the hooks
|
|
138
|
+
are installed immediately; otherwise, installation is deferred until
|
|
139
|
+
``pyspark.sql`` is actually imported by someone. Idempotent; never raises.
|
|
140
|
+
"""
|
|
141
|
+
if "pyspark.sql" in sys.modules:
|
|
142
|
+
_install_hooks_now()
|
|
143
|
+
return
|
|
144
|
+
if any(isinstance(finder, _PySparkImportHook) for finder in sys.meta_path):
|
|
145
|
+
return
|
|
146
|
+
try:
|
|
147
|
+
sys.meta_path.insert(0, _PySparkImportHook())
|
|
148
|
+
except Exception as exc: # pragma: no cover - defensive
|
|
149
|
+
logger.debug(f"Could not defer column-casing hint installation: {exc}")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _install_hooks_now() -> None:
|
|
128
153
|
"""Install the column-casing hint hooks. Idempotent; never raises."""
|
|
129
154
|
try:
|
|
130
155
|
_install_analysis_exception_hook()
|
|
131
156
|
_install_getattr_hook()
|
|
132
157
|
except Exception as exc: # pragma: no cover - defensive
|
|
133
158
|
logger.debug(f"Could not install column-casing hint hooks: {exc}")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class _PySparkImportHook(importlib.abc.MetaPathFinder):
|
|
162
|
+
"""Installs the hint hooks right after ``pyspark.sql`` finishes loading.
|
|
163
|
+
|
|
164
|
+
``install_column_casing_hints`` must not import pyspark itself: this
|
|
165
|
+
package is also used in the pyspark-free Function code path that must
|
|
166
|
+
not pull pyspark in as a side effect of importing ``datacustomcode``.
|
|
167
|
+
Wrapping the real loader lets us defer the pyspark import until whoever
|
|
168
|
+
actually needs it (the SDK's own lazy accessors, or user code) imports
|
|
169
|
+
``pyspark.sql`` on their own.
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
def find_spec(
|
|
173
|
+
self, fullname: str, path: object, target: ModuleType | None = None
|
|
174
|
+
) -> ModuleSpec | None:
|
|
175
|
+
if fullname != "pyspark.sql":
|
|
176
|
+
return None
|
|
177
|
+
|
|
178
|
+
for finder in sys.meta_path:
|
|
179
|
+
if finder is self:
|
|
180
|
+
continue
|
|
181
|
+
find_spec = getattr(finder, "find_spec", None)
|
|
182
|
+
if find_spec is None:
|
|
183
|
+
continue
|
|
184
|
+
spec = find_spec(fullname, path, target)
|
|
185
|
+
if spec is not None and spec.loader is not None:
|
|
186
|
+
original_exec_module = spec.loader.exec_module
|
|
187
|
+
|
|
188
|
+
def exec_module(module: ModuleType, _orig=original_exec_module) -> None:
|
|
189
|
+
_orig(module)
|
|
190
|
+
_install_hooks_now()
|
|
191
|
+
|
|
192
|
+
spec.loader.exec_module = exec_module # type: ignore[method-assign]
|
|
193
|
+
return spec # type: ignore[no-any-return]
|
|
194
|
+
return None
|
|
@@ -27,6 +27,8 @@ STREAMING_EXAMPLE_ENTRYPOINT = os.path.join(
|
|
|
27
27
|
script_template_dir, "examples", "streaming_deltas", "entrypoint.py"
|
|
28
28
|
)
|
|
29
29
|
|
|
30
|
+
_SDK_CONFIG_YAML = os.path.join(os.path.dirname(__file__), "config.yaml")
|
|
31
|
+
|
|
30
32
|
|
|
31
33
|
def copy_script_template(target_dir: str, streaming: bool = False) -> None:
|
|
32
34
|
"""Copy the template to the target directory."""
|
|
@@ -51,6 +53,24 @@ def copy_script_template(target_dir: str, streaming: bool = False) -> None:
|
|
|
51
53
|
)
|
|
52
54
|
shutil.copy2(STREAMING_EXAMPLE_ENTRYPOINT, destination)
|
|
53
55
|
|
|
56
|
+
_write_streaming_config_yaml(target_dir)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _write_streaming_config_yaml(target_dir: str) -> None:
|
|
60
|
+
"""Writes the sdk config file with Streaming overrdies"""
|
|
61
|
+
import yaml
|
|
62
|
+
|
|
63
|
+
with open(_SDK_CONFIG_YAML) as f:
|
|
64
|
+
config_data = yaml.safe_load(f)
|
|
65
|
+
|
|
66
|
+
config_data["reader_config"]["type_config_name"] = "LocalDeltasReader"
|
|
67
|
+
config_data["writer_config"]["type_config_name"] = "LocalDeltasWriter"
|
|
68
|
+
|
|
69
|
+
destination = os.path.join(target_dir, "config.yaml")
|
|
70
|
+
logger.debug(f"Writing streaming config.yaml to {destination}...")
|
|
71
|
+
with open(destination, "w") as f:
|
|
72
|
+
yaml.safe_dump(config_data, f, sort_keys=False)
|
|
73
|
+
|
|
54
74
|
|
|
55
75
|
def copy_function_template(target_dir: str, use_in_feature: Optional[str]) -> None:
|
|
56
76
|
os.makedirs(target_dir, exist_ok=True)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|