dscompanion 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dscompanion-0.1.0/LICENSE +21 -0
- dscompanion-0.1.0/PKG-INFO +193 -0
- dscompanion-0.1.0/README.md +134 -0
- dscompanion-0.1.0/pyproject.toml +113 -0
- dscompanion-0.1.0/setup.cfg +4 -0
- dscompanion-0.1.0/src/dscompanion/__init__.py +67 -0
- dscompanion-0.1.0/src/dscompanion/api/__init__.py +6 -0
- dscompanion-0.1.0/src/dscompanion/api/auth/__init__.py +65 -0
- dscompanion-0.1.0/src/dscompanion/api/auth/api_key.py +40 -0
- dscompanion-0.1.0/src/dscompanion/api/auth/azure_entra.py +31 -0
- dscompanion-0.1.0/src/dscompanion/api/auth/databricks_oauth.py +36 -0
- dscompanion-0.1.0/src/dscompanion/api/config.py +71 -0
- dscompanion-0.1.0/src/dscompanion/api/deps.py +81 -0
- dscompanion-0.1.0/src/dscompanion/api/integrations/__init__.py +11 -0
- dscompanion-0.1.0/src/dscompanion/api/main.py +59 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/__init__.py +47 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/calibration.py +73 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/data_files.py +32 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/eda.py +93 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/evaluate.py +63 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/feature_processing.py +101 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/feature_selection.py +74 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/imbalance.py +69 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/jobs.py +50 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/load_data.py +78 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/runs.py +94 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/shap.py +88 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/split.py +71 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/steps_stub.py +65 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/task_target.py +74 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/train.py +97 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/transformer_registry.py +37 -0
- dscompanion-0.1.0/src/dscompanion/api/routers/tuning.py +98 -0
- dscompanion-0.1.0/src/dscompanion/api/schemas.py +1126 -0
- dscompanion-0.1.0/src/dscompanion/api/services/__init__.py +14 -0
- dscompanion-0.1.0/src/dscompanion/api/services/calibration.py +235 -0
- dscompanion-0.1.0/src/dscompanion/api/services/data_files.py +41 -0
- dscompanion-0.1.0/src/dscompanion/api/services/eda.py +267 -0
- dscompanion-0.1.0/src/dscompanion/api/services/evaluate.py +374 -0
- dscompanion-0.1.0/src/dscompanion/api/services/feature_processing.py +282 -0
- dscompanion-0.1.0/src/dscompanion/api/services/feature_selection.py +216 -0
- dscompanion-0.1.0/src/dscompanion/api/services/imbalance.py +150 -0
- dscompanion-0.1.0/src/dscompanion/api/services/load_data.py +247 -0
- dscompanion-0.1.0/src/dscompanion/api/services/shap.py +191 -0
- dscompanion-0.1.0/src/dscompanion/api/services/split.py +141 -0
- dscompanion-0.1.0/src/dscompanion/api/services/task_target.py +154 -0
- dscompanion-0.1.0/src/dscompanion/api/services/train.py +507 -0
- dscompanion-0.1.0/src/dscompanion/api/services/tuning.py +373 -0
- dscompanion-0.1.0/src/dscompanion/api/state.py +243 -0
- dscompanion-0.1.0/src/dscompanion/api/steps.py +193 -0
- dscompanion-0.1.0/src/dscompanion/calibration/__init__.py +5 -0
- dscompanion-0.1.0/src/dscompanion/calibration/calibrator.py +307 -0
- dscompanion-0.1.0/src/dscompanion/config.py +583 -0
- dscompanion-0.1.0/src/dscompanion/docs/__init__.py +5 -0
- dscompanion-0.1.0/src/dscompanion/docs/html_eda.py +630 -0
- dscompanion-0.1.0/src/dscompanion/docs/html_model.py +351 -0
- dscompanion-0.1.0/src/dscompanion/docs/html_widgets.py +639 -0
- dscompanion-0.1.0/src/dscompanion/docs/model_card.py +2264 -0
- dscompanion-0.1.0/src/dscompanion/eda/__init__.py +15 -0
- dscompanion-0.1.0/src/dscompanion/eda/bivariate.py +473 -0
- dscompanion-0.1.0/src/dscompanion/eda/missingness.py +173 -0
- dscompanion-0.1.0/src/dscompanion/eda/multivariate.py +397 -0
- dscompanion-0.1.0/src/dscompanion/eda/report.py +1278 -0
- dscompanion-0.1.0/src/dscompanion/eda/univariate.py +953 -0
- dscompanion-0.1.0/src/dscompanion/explain/__init__.py +14 -0
- dscompanion-0.1.0/src/dscompanion/explain/lime_explainer.py +188 -0
- dscompanion-0.1.0/src/dscompanion/explain/pdp.py +368 -0
- dscompanion-0.1.0/src/dscompanion/explain/permutation_importance.py +199 -0
- dscompanion-0.1.0/src/dscompanion/explain/shap_explainer.py +928 -0
- dscompanion-0.1.0/src/dscompanion/features/__init__.py +73 -0
- dscompanion-0.1.0/src/dscompanion/features/binner.py +230 -0
- dscompanion-0.1.0/src/dscompanion/features/combiner.py +314 -0
- dscompanion-0.1.0/src/dscompanion/features/compression.py +188 -0
- dscompanion-0.1.0/src/dscompanion/features/date_features.py +218 -0
- dscompanion-0.1.0/src/dscompanion/features/distribution.py +247 -0
- dscompanion-0.1.0/src/dscompanion/features/encoder.py +1112 -0
- dscompanion-0.1.0/src/dscompanion/features/imputer.py +390 -0
- dscompanion-0.1.0/src/dscompanion/features/leakage_guard.py +201 -0
- dscompanion-0.1.0/src/dscompanion/features/noise.py +284 -0
- dscompanion-0.1.0/src/dscompanion/features/pipeline.py +551 -0
- dscompanion-0.1.0/src/dscompanion/features/polynomial.py +167 -0
- dscompanion-0.1.0/src/dscompanion/features/rank.py +169 -0
- dscompanion-0.1.0/src/dscompanion/features/recommend.py +170 -0
- dscompanion-0.1.0/src/dscompanion/features/registry.py +271 -0
- dscompanion-0.1.0/src/dscompanion/features/relative.py +256 -0
- dscompanion-0.1.0/src/dscompanion/features/scaler.py +558 -0
- dscompanion-0.1.0/src/dscompanion/features/spline.py +159 -0
- dscompanion-0.1.0/src/dscompanion/features/transform_chain.py +289 -0
- dscompanion-0.1.0/src/dscompanion/leaderboard/__init__.py +5 -0
- dscompanion-0.1.0/src/dscompanion/leaderboard/leaderboard.py +255 -0
- dscompanion-0.1.0/src/dscompanion/models/__init__.py +15 -0
- dscompanion-0.1.0/src/dscompanion/models/base.py +292 -0
- dscompanion-0.1.0/src/dscompanion/models/classification.py +248 -0
- dscompanion-0.1.0/src/dscompanion/models/clustering.py +232 -0
- dscompanion-0.1.0/src/dscompanion/models/factory.py +417 -0
- dscompanion-0.1.0/src/dscompanion/models/regression.py +150 -0
- dscompanion-0.1.0/src/dscompanion/pipeline/__init__.py +42 -0
- dscompanion-0.1.0/src/dscompanion/pipeline/config.py +1084 -0
- dscompanion-0.1.0/src/dscompanion/pipeline/runner.py +1409 -0
- dscompanion-0.1.0/src/dscompanion/selection/__init__.py +25 -0
- dscompanion-0.1.0/src/dscompanion/selection/feature_selectors.py +645 -0
- dscompanion-0.1.0/src/dscompanion/selection/selection_pipeline.py +200 -0
- dscompanion-0.1.0/src/dscompanion/split/__init__.py +5 -0
- dscompanion-0.1.0/src/dscompanion/split/splitter.py +661 -0
- dscompanion-0.1.0/src/dscompanion/targets/__init__.py +9 -0
- dscompanion-0.1.0/src/dscompanion/targets/binariser.py +155 -0
- dscompanion-0.1.0/src/dscompanion/targets/imbalance.py +408 -0
- dscompanion-0.1.0/src/dscompanion/tracking/__init__.py +7 -0
- dscompanion-0.1.0/src/dscompanion/tracking/run_context.py +191 -0
- dscompanion-0.1.0/src/dscompanion/tuning/__init__.py +10 -0
- dscompanion-0.1.0/src/dscompanion/tuning/backends/__init__.py +1 -0
- dscompanion-0.1.0/src/dscompanion/tuning/backends/hyperopt_backend.py +256 -0
- dscompanion-0.1.0/src/dscompanion/tuning/backends/optuna_backend.py +309 -0
- dscompanion-0.1.0/src/dscompanion/tuning/search_spaces.py +361 -0
- dscompanion-0.1.0/src/dscompanion/tuning/tuner.py +288 -0
- dscompanion-0.1.0/src/dscompanion/utils/__init__.py +34 -0
- dscompanion-0.1.0/src/dscompanion/utils/metrics.py +414 -0
- dscompanion-0.1.0/src/dscompanion/utils/plotting.py +130 -0
- dscompanion-0.1.0/src/dscompanion/utils/synthetic.py +266 -0
- dscompanion-0.1.0/src/dscompanion/utils/validators.py +128 -0
- dscompanion-0.1.0/src/dscompanion.egg-info/PKG-INFO +193 -0
- dscompanion-0.1.0/src/dscompanion.egg-info/SOURCES.txt +149 -0
- dscompanion-0.1.0/src/dscompanion.egg-info/dependency_links.txt +1 -0
- dscompanion-0.1.0/src/dscompanion.egg-info/requires.txt +46 -0
- dscompanion-0.1.0/src/dscompanion.egg-info/top_level.txt +1 -0
- dscompanion-0.1.0/tests/test_api.py +815 -0
- dscompanion-0.1.0/tests/test_calibration.py +130 -0
- dscompanion-0.1.0/tests/test_docs.py +1163 -0
- dscompanion-0.1.0/tests/test_eda.py +1176 -0
- dscompanion-0.1.0/tests/test_explain.py +682 -0
- dscompanion-0.1.0/tests/test_features.py +2906 -0
- dscompanion-0.1.0/tests/test_glossary.py +51 -0
- dscompanion-0.1.0/tests/test_html_eda.py +146 -0
- dscompanion-0.1.0/tests/test_html_model.py +226 -0
- dscompanion-0.1.0/tests/test_html_widgets.py +141 -0
- dscompanion-0.1.0/tests/test_interactive_step_calibration.py +88 -0
- dscompanion-0.1.0/tests/test_interactive_step_eda.py +64 -0
- dscompanion-0.1.0/tests/test_interactive_step_feature_processing.py +225 -0
- dscompanion-0.1.0/tests/test_interactive_step_split.py +117 -0
- dscompanion-0.1.0/tests/test_interactive_step_train.py +188 -0
- dscompanion-0.1.0/tests/test_leaderboard.py +154 -0
- dscompanion-0.1.0/tests/test_models.py +607 -0
- dscompanion-0.1.0/tests/test_pipeline.py +1646 -0
- dscompanion-0.1.0/tests/test_selection.py +403 -0
- dscompanion-0.1.0/tests/test_split.py +252 -0
- dscompanion-0.1.0/tests/test_steps.py +65 -0
- dscompanion-0.1.0/tests/test_streamlit_app.py +57 -0
- dscompanion-0.1.0/tests/test_targets.py +282 -0
- dscompanion-0.1.0/tests/test_tracking.py +109 -0
- dscompanion-0.1.0/tests/test_tuning.py +659 -0
- dscompanion-0.1.0/tests/test_utils.py +389 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 dscompanion contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dscompanion
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A production-grade ML pipeline toolkit: EDA, feature engineering, model training/tuning/calibration, and governance-ready model cards.
|
|
5
|
+
Author-email: dsnaveen <mailtonaveenmittal@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/dsnaveen/dscompanion
|
|
8
|
+
Project-URL: Repository, https://github.com/dsnaveen/dscompanion
|
|
9
|
+
Project-URL: Issues, https://github.com/dsnaveen/dscompanion/issues
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.12
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: pandas<3,>=2.2
|
|
19
|
+
Requires-Dist: numpy<2.4,>=2.1
|
|
20
|
+
Requires-Dist: scikit-learn<2,>=1.7
|
|
21
|
+
Requires-Dist: scipy<2,>=1.16
|
|
22
|
+
Requires-Dist: statsmodels<0.15,>=0.14
|
|
23
|
+
Requires-Dist: joblib<2,>=1.4
|
|
24
|
+
Requires-Dist: plotly<6,>=5.20
|
|
25
|
+
Requires-Dist: jinja2<4,>=3.1
|
|
26
|
+
Requires-Dist: pyyaml<7,>=6.0
|
|
27
|
+
Requires-Dist: pydantic<2.11,>=2.10
|
|
28
|
+
Requires-Dist: pydantic-settings<3,>=2.10
|
|
29
|
+
Requires-Dist: matplotlib<4,>=3.9
|
|
30
|
+
Requires-Dist: seaborn<0.14,>=0.13
|
|
31
|
+
Requires-Dist: xgboost<4,>=3.0
|
|
32
|
+
Requires-Dist: optuna<4,>=3.6
|
|
33
|
+
Requires-Dist: shap<0.52,>=0.44
|
|
34
|
+
Requires-Dist: python-docx<2,>=1.1
|
|
35
|
+
Requires-Dist: xlsxwriter<4,>=3.2
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
38
|
+
Requires-Dist: pytest-cov>=4.1; extra == "dev"
|
|
39
|
+
Requires-Dist: black==24.10.0; extra == "dev"
|
|
40
|
+
Requires-Dist: ruff==0.15.16; extra == "dev"
|
|
41
|
+
Requires-Dist: httpx>=0.27; extra == "dev"
|
|
42
|
+
Provides-Extra: databricks
|
|
43
|
+
Requires-Dist: pyspark==4.0.0; extra == "databricks"
|
|
44
|
+
Requires-Dist: databricks-sdk>=0.20; extra == "databricks"
|
|
45
|
+
Provides-Extra: docs
|
|
46
|
+
Requires-Dist: sphinx>=9.0; extra == "docs"
|
|
47
|
+
Requires-Dist: furo>=2025.0; extra == "docs"
|
|
48
|
+
Requires-Dist: sphinx-autodoc-typehints>=3.0; extra == "docs"
|
|
49
|
+
Requires-Dist: myst-parser>=5.0; extra == "docs"
|
|
50
|
+
Provides-Extra: app
|
|
51
|
+
Requires-Dist: streamlit==1.59.1; extra == "app"
|
|
52
|
+
Requires-Dist: openpyxl>=3.1; extra == "app"
|
|
53
|
+
Provides-Extra: api
|
|
54
|
+
Requires-Dist: fastapi==0.117.1; extra == "api"
|
|
55
|
+
Requires-Dist: uvicorn==0.37.0; extra == "api"
|
|
56
|
+
Provides-Extra: models
|
|
57
|
+
Requires-Dist: lightgbm==4.6.0; extra == "models"
|
|
58
|
+
Dynamic: license-file
|
|
59
|
+
|
|
60
|
+
# dscompanion
|
|
61
|
+
|
|
62
|
+
A production-grade ML pipeline toolkit for tabular data: EDA, feature engineering, model
|
|
63
|
+
training/tuning/calibration, and governance-ready model cards — with sensible defaults
|
|
64
|
+
everywhere and full override capability. Designed to work well in regulated or restricted
|
|
65
|
+
environments (see [Architecture](docs/architecture.rst) for the constraints that shaped it),
|
|
66
|
+
but useful for any tabular ML project.
|
|
67
|
+
|
|
68
|
+
## Installation
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
pip install dscompanion
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
For local development:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
git clone https://github.com/dsnaveen/dscompanion.git
|
|
78
|
+
cd dscompanion
|
|
79
|
+
pip install -e ".[dev]"
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
**One thing worth knowing about SHAP:** `ExplainConfig.shap_enabled` defaults to `False` —
|
|
83
|
+
SHAP computation is relatively expensive, so it's opt-in rather than run automatically. See
|
|
84
|
+
`explain.permutation_enabled` for a lighter-weight, model-agnostic alternative.
|
|
85
|
+
|
|
86
|
+
**Run tracking via stdlib logging** — dscompanion logs pipeline execution (run start/finish,
|
|
87
|
+
training metrics, parameters, artifacts) via Python's `logging` module, not a dedicated
|
|
88
|
+
tracking server. A `dscompanion.tracking.run_context.tracking_run()` context manager handles
|
|
89
|
+
this transparently during pipeline execution. If you need queryable run history or a tracking
|
|
90
|
+
server UI, you'll need a separate mechanism outside dscompanion — it provides only local
|
|
91
|
+
structured logs by default.
|
|
92
|
+
|
|
93
|
+
## Quickstart
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
from dscompanion.pipeline import PipelineConfig, PipelineRunner
|
|
97
|
+
|
|
98
|
+
cfg = PipelineConfig(
|
|
99
|
+
name="my_first_model",
|
|
100
|
+
data={"path": "data.parquet", "target": "target_column"},
|
|
101
|
+
split={"method": "stratified", "test_size": 0.2, "val_size": 0.1},
|
|
102
|
+
model={"task": "classification", "algorithm": "xgboost"},
|
|
103
|
+
reporting={"output_dir": "outputs", "html_report": True},
|
|
104
|
+
)
|
|
105
|
+
result = PipelineRunner(cfg).run()
|
|
106
|
+
|
|
107
|
+
print(result.metrics) # per-split evaluation metrics
|
|
108
|
+
print(result.model_card.to_excel(...)) # governance-ready model card
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
`PipelineRunner` runs the full pipeline end-to-end — load, split, EDA, feature processing,
|
|
112
|
+
selection, imbalance handling, train (or leaderboard comparison), tune, calibrate, explain,
|
|
113
|
+
and report — with every stage overridable through `PipelineConfig`. See
|
|
114
|
+
[`docs/pipeline.rst`](docs/pipeline.rst) for the full stage-by-stage reference and
|
|
115
|
+
[`docs/quickstart.rst`](docs/quickstart.rst) for more examples.
|
|
116
|
+
|
|
117
|
+
## Package Structure
|
|
118
|
+
|
|
119
|
+
```
|
|
120
|
+
src/dscompanion/
|
|
121
|
+
├── config.py # DSCompanionConfig settings (pydantic-settings, env_prefix=DSCOMPANION_)
|
|
122
|
+
├── pipeline/ # PipelineConfig, PipelineRunner — the end-to-end orchestrator
|
|
123
|
+
├── split/ # DataSplitter — temporal, stratified, grouped, random
|
|
124
|
+
├── eda/ # UnivariateAnalyser, BivariateAnalyser, MultivariateAnalyser, EDAReport
|
|
125
|
+
├── features/ # SmartImputer, encoders, SmartScaler, AutoBinner,
|
|
126
|
+
│ # DateFeatureExtractor, LeakageGuard, FeatureProcessingPipeline
|
|
127
|
+
├── targets/ # ImbalanceHandler (SMOTE/class_weight), TargetBinariser
|
|
128
|
+
├── selection/ # ConstantSelector, CorrelationSelector, IVSelector,
|
|
129
|
+
│ # FeatureSelectionPipeline
|
|
130
|
+
├── models/ # ModelFactory, ClassificationModel, RegressionModel, ClusteringModel
|
|
131
|
+
├── tuning/ # Tuner, OptunaBackend, HyperoptBackend, search_spaces
|
|
132
|
+
├── leaderboard/ # Leaderboard — multi-algorithm comparison
|
|
133
|
+
├── explain/ # SHAPExplainer, PermutationImportanceAnalyser
|
|
134
|
+
├── calibration/ # Calibrator (isotonic, Platt, beta)
|
|
135
|
+
├── docs/ # ModelCard, HTML report generators (EDA/model cards, widgets)
|
|
136
|
+
├── tracking/ # tracking_run context manager, log_metrics/params/artifact via stdlib logging
|
|
137
|
+
├── api/ # optional FastAPI service (pip install dscompanion[api])
|
|
138
|
+
└── utils/ # metrics (KS, Gini, PSI, IV, WoE, ECE), validators, plotting
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
A Streamlit UI lives at [`app/`](app/) (`pip install dscompanion[app]`, then
|
|
142
|
+
`streamlit run app/streamlit_app.py`) for interactively driving a pipeline run without
|
|
143
|
+
writing config by hand.
|
|
144
|
+
|
|
145
|
+
## Configuration
|
|
146
|
+
|
|
147
|
+
All thresholds and defaults are controlled via `dscompanion.config.settings`. Override with environment
|
|
148
|
+
variables prefixed `DSCOMPANION_`:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
export DSCOMPANION_TARGET_LEAKAGE_CORRELATION_THRESHOLD=0.90
|
|
152
|
+
export DSCOMPANION_HIGH_CARDINALITY_THRESHOLD=30
|
|
153
|
+
export DSCOMPANION_RANDOM_STATE=0
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Or in code:
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
from dscompanion.config import settings
|
|
160
|
+
settings.high_cardinality_threshold = 30
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
## Running Tests
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
git clone https://github.com/dsnaveen/dscompanion.git
|
|
167
|
+
cd dscompanion
|
|
168
|
+
pip install -e ".[dev]"
|
|
169
|
+
pytest tests/ -v
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
## Key Design Principles
|
|
173
|
+
|
|
174
|
+
- **sklearn-compatible** — all transformers implement `fit` / `transform` / `get_feature_names_out`
|
|
175
|
+
- **joblib-serialisable** — every fitted object can be pickled with `joblib.dump`
|
|
176
|
+
- **Structured logging via stdlib** — all pipeline execution logs through Python's `logging` module
|
|
177
|
+
- **pydantic v2** — config and report models use `BaseModel` / `BaseSettings`
|
|
178
|
+
- **No hidden network calls** — works fully offline once installed, suited to restricted environments
|
|
179
|
+
|
|
180
|
+
## Dependencies
|
|
181
|
+
|
|
182
|
+
Core: `pandas`, `numpy`, `scikit-learn`, `xgboost`, `pydantic-settings`, `joblib`, `plotly`, `jinja2`
|
|
183
|
+
|
|
184
|
+
Optional: `optuna`, `shap`, `python-docx` (core extras); `streamlit` (`[app]`); `fastapi`,
|
|
185
|
+
`uvicorn` (`[api]`); `sphinx` (`[docs]`)
|
|
186
|
+
|
|
187
|
+
## Contributing
|
|
188
|
+
|
|
189
|
+
See [`CONTRIBUTING.md`](CONTRIBUTING.md).
|
|
190
|
+
|
|
191
|
+
## License
|
|
192
|
+
|
|
193
|
+
MIT — see [`LICENSE`](LICENSE).
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# dscompanion
|
|
2
|
+
|
|
3
|
+
A production-grade ML pipeline toolkit for tabular data: EDA, feature engineering, model
|
|
4
|
+
training/tuning/calibration, and governance-ready model cards — with sensible defaults
|
|
5
|
+
everywhere and full override capability. Designed to work well in regulated or restricted
|
|
6
|
+
environments (see [Architecture](docs/architecture.rst) for the constraints that shaped it),
|
|
7
|
+
but useful for any tabular ML project.
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install dscompanion
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
For local development:
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
git clone https://github.com/dsnaveen/dscompanion.git
|
|
19
|
+
cd dscompanion
|
|
20
|
+
pip install -e ".[dev]"
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
**One thing worth knowing about SHAP:** `ExplainConfig.shap_enabled` defaults to `False` —
|
|
24
|
+
SHAP computation is relatively expensive, so it's opt-in rather than run automatically. See
|
|
25
|
+
`explain.permutation_enabled` for a lighter-weight, model-agnostic alternative.
|
|
26
|
+
|
|
27
|
+
**Run tracking via stdlib logging** — dscompanion logs pipeline execution (run start/finish,
|
|
28
|
+
training metrics, parameters, artifacts) via Python's `logging` module, not a dedicated
|
|
29
|
+
tracking server. A `dscompanion.tracking.run_context.tracking_run()` context manager handles
|
|
30
|
+
this transparently during pipeline execution. If you need queryable run history or a tracking
|
|
31
|
+
server UI, you'll need a separate mechanism outside dscompanion — it provides only local
|
|
32
|
+
structured logs by default.
|
|
33
|
+
|
|
34
|
+
## Quickstart
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from dscompanion.pipeline import PipelineConfig, PipelineRunner
|
|
38
|
+
|
|
39
|
+
cfg = PipelineConfig(
|
|
40
|
+
name="my_first_model",
|
|
41
|
+
data={"path": "data.parquet", "target": "target_column"},
|
|
42
|
+
split={"method": "stratified", "test_size": 0.2, "val_size": 0.1},
|
|
43
|
+
model={"task": "classification", "algorithm": "xgboost"},
|
|
44
|
+
reporting={"output_dir": "outputs", "html_report": True},
|
|
45
|
+
)
|
|
46
|
+
result = PipelineRunner(cfg).run()
|
|
47
|
+
|
|
48
|
+
print(result.metrics) # per-split evaluation metrics
|
|
49
|
+
print(result.model_card.to_excel(...)) # governance-ready model card
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
`PipelineRunner` runs the full pipeline end-to-end — load, split, EDA, feature processing,
|
|
53
|
+
selection, imbalance handling, train (or leaderboard comparison), tune, calibrate, explain,
|
|
54
|
+
and report — with every stage overridable through `PipelineConfig`. See
|
|
55
|
+
[`docs/pipeline.rst`](docs/pipeline.rst) for the full stage-by-stage reference and
|
|
56
|
+
[`docs/quickstart.rst`](docs/quickstart.rst) for more examples.
|
|
57
|
+
|
|
58
|
+
## Package Structure
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
src/dscompanion/
|
|
62
|
+
├── config.py # DSCompanionConfig settings (pydantic-settings, env_prefix=DSCOMPANION_)
|
|
63
|
+
├── pipeline/ # PipelineConfig, PipelineRunner — the end-to-end orchestrator
|
|
64
|
+
├── split/ # DataSplitter — temporal, stratified, grouped, random
|
|
65
|
+
├── eda/ # UnivariateAnalyser, BivariateAnalyser, MultivariateAnalyser, EDAReport
|
|
66
|
+
├── features/ # SmartImputer, encoders, SmartScaler, AutoBinner,
|
|
67
|
+
│ # DateFeatureExtractor, LeakageGuard, FeatureProcessingPipeline
|
|
68
|
+
├── targets/ # ImbalanceHandler (SMOTE/class_weight), TargetBinariser
|
|
69
|
+
├── selection/ # ConstantSelector, CorrelationSelector, IVSelector,
|
|
70
|
+
│ # FeatureSelectionPipeline
|
|
71
|
+
├── models/ # ModelFactory, ClassificationModel, RegressionModel, ClusteringModel
|
|
72
|
+
├── tuning/ # Tuner, OptunaBackend, HyperoptBackend, search_spaces
|
|
73
|
+
├── leaderboard/ # Leaderboard — multi-algorithm comparison
|
|
74
|
+
├── explain/ # SHAPExplainer, PermutationImportanceAnalyser
|
|
75
|
+
├── calibration/ # Calibrator (isotonic, Platt, beta)
|
|
76
|
+
├── docs/ # ModelCard, HTML report generators (EDA/model cards, widgets)
|
|
77
|
+
├── tracking/ # tracking_run context manager, log_metrics/params/artifact via stdlib logging
|
|
78
|
+
├── api/ # optional FastAPI service (pip install dscompanion[api])
|
|
79
|
+
└── utils/ # metrics (KS, Gini, PSI, IV, WoE, ECE), validators, plotting
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
A Streamlit UI lives at [`app/`](app/) (`pip install dscompanion[app]`, then
|
|
83
|
+
`streamlit run app/streamlit_app.py`) for interactively driving a pipeline run without
|
|
84
|
+
writing config by hand.
|
|
85
|
+
|
|
86
|
+
## Configuration
|
|
87
|
+
|
|
88
|
+
All thresholds and defaults are controlled via `dscompanion.config.settings`. Override with environment
|
|
89
|
+
variables prefixed `DSCOMPANION_`:
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
export DSCOMPANION_TARGET_LEAKAGE_CORRELATION_THRESHOLD=0.90
|
|
93
|
+
export DSCOMPANION_HIGH_CARDINALITY_THRESHOLD=30
|
|
94
|
+
export DSCOMPANION_RANDOM_STATE=0
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Or in code:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
from dscompanion.config import settings
|
|
101
|
+
settings.high_cardinality_threshold = 30
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## Running Tests
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
git clone https://github.com/dsnaveen/dscompanion.git
|
|
108
|
+
cd dscompanion
|
|
109
|
+
pip install -e ".[dev]"
|
|
110
|
+
pytest tests/ -v
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
## Key Design Principles
|
|
114
|
+
|
|
115
|
+
- **sklearn-compatible** — all transformers implement `fit` / `transform` / `get_feature_names_out`
|
|
116
|
+
- **joblib-serialisable** — every fitted object can be pickled with `joblib.dump`
|
|
117
|
+
- **Structured logging via stdlib** — all pipeline execution logs through Python's `logging` module
|
|
118
|
+
- **pydantic v2** — config and report models use `BaseModel` / `BaseSettings`
|
|
119
|
+
- **No hidden network calls** — works fully offline once installed, suited to restricted environments
|
|
120
|
+
|
|
121
|
+
## Dependencies
|
|
122
|
+
|
|
123
|
+
Core: `pandas`, `numpy`, `scikit-learn`, `xgboost`, `pydantic-settings`, `joblib`, `plotly`, `jinja2`
|
|
124
|
+
|
|
125
|
+
Optional: `optuna`, `shap`, `python-docx` (core extras); `streamlit` (`[app]`); `fastapi`,
|
|
126
|
+
`uvicorn` (`[api]`); `sphinx` (`[docs]`)
|
|
127
|
+
|
|
128
|
+
## Contributing
|
|
129
|
+
|
|
130
|
+
See [`CONTRIBUTING.md`](CONTRIBUTING.md).
|
|
131
|
+
|
|
132
|
+
## License
|
|
133
|
+
|
|
134
|
+
MIT — see [`LICENSE`](LICENSE).
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=42", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dscompanion"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "A production-grade ML pipeline toolkit: EDA, feature engineering, model training/tuning/calibration, and governance-ready model cards."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
authors = [{ name = "dsnaveen", email = "mailtonaveenmittal@gmail.com" }]
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
requires-python = ">=3.12"
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 3 - Alpha",
|
|
15
|
+
"Intended Audience :: Science/Research",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
"pandas>=2.2,<3", # pandas 2.x — does not default to Copy-on-Write
|
|
22
|
+
"numpy>=2.1,<2.4", # upper-bounded for numba (a shap dependency) compatibility
|
|
23
|
+
"scikit-learn>=1.7,<2",
|
|
24
|
+
"scipy>=1.16,<2",
|
|
25
|
+
"statsmodels>=0.14,<0.15",
|
|
26
|
+
"joblib>=1.4,<2",
|
|
27
|
+
"plotly>=5.20,<6",
|
|
28
|
+
"jinja2>=3.1,<4",
|
|
29
|
+
"pyyaml>=6.0,<7",
|
|
30
|
+
"pydantic>=2.10,<2.11", # v2 native — BaseSettings lives in pydantic-settings.
|
|
31
|
+
# Narrowly bounded (not just <3): pydantic's generated JSON
|
|
32
|
+
# schema shape has changed across minor versions before, and
|
|
33
|
+
# api/openapi.json is committed and snapshot-tested against it.
|
|
34
|
+
"pydantic-settings>=2.10,<3",
|
|
35
|
+
"matplotlib>=3.9,<4",
|
|
36
|
+
"seaborn>=0.13,<0.14",
|
|
37
|
+
"xgboost>=3.0,<4", # macOS dev requires Homebrew libomp (conda-forge llvm-openmp)
|
|
38
|
+
"optuna>=3.6,<4",
|
|
39
|
+
"shap>=0.44,<0.52", # pre-0.52 API — shap_explainer.py returns list-of-arrays branch;
|
|
40
|
+
# shap 0.52.0+ returns a 3-D ndarray instead
|
|
41
|
+
# imbalanced-learn is intentionally not a dependency — ImbalanceHandler's smote/undersample
|
|
42
|
+
# strategies are implemented from scratch on scikit-learn's NearestNeighbors alone, since
|
|
43
|
+
# imbalanced-learn has a history of version conflicts with recent scikit-learn releases.
|
|
44
|
+
"python-docx>=1.1,<2",
|
|
45
|
+
"xlsxwriter>=3.2,<4", # engine for ModelCard.to_excel()
|
|
46
|
+
]
|
|
47
|
+
|
|
48
|
+
[project.urls]
|
|
49
|
+
Homepage = "https://github.com/dsnaveen/dscompanion"
|
|
50
|
+
Repository = "https://github.com/dsnaveen/dscompanion"
|
|
51
|
+
Issues = "https://github.com/dsnaveen/dscompanion/issues"
|
|
52
|
+
|
|
53
|
+
[project.optional-dependencies]
|
|
54
|
+
dev = ["pytest>=8.0", "pytest-cov>=4.1", "black==24.10.0", "ruff==0.15.16", "httpx>=0.27"]
|
|
55
|
+
databricks = ["pyspark==4.0.0", "databricks-sdk>=0.20"]
|
|
56
|
+
# Doc build tooling only — never imported by dscompanion's runtime code.
|
|
57
|
+
docs = ["sphinx>=9.0", "furo>=2025.0", "sphinx-autodoc-typehints>=3.0", "myst-parser>=5.0"]
|
|
58
|
+
# Local-dev UI only — drives PipelineRunner from a browser on a dev machine against
|
|
59
|
+
# local data. app/ is never imported by dscompanion's runtime code, so this
|
|
60
|
+
# extra is optional even for full local development.
|
|
61
|
+
app = ["streamlit==1.59.1", "openpyxl>=3.1"]
|
|
62
|
+
# Optional FastAPI service exposing PipelineRunner over HTTP, for a web-based
|
|
63
|
+
# frontend. Not required for using dscompanion as a library.
|
|
64
|
+
api = ["fastapi==0.117.1", "uvicorn==0.37.0"]
|
|
65
|
+
# Optional gradient-boosting backend for ModelFactory's "lightgbm" algorithm choice —
|
|
66
|
+
# not required unless that specific algorithm is selected.
|
|
67
|
+
models = ["lightgbm==4.6.0"]
|
|
68
|
+
|
|
69
|
+
[tool.setuptools.packages.find]
|
|
70
|
+
where = ["src"]
|
|
71
|
+
include = ["dscompanion*"]
|
|
72
|
+
|
|
73
|
+
[tool.black]
|
|
74
|
+
line-length = 100
|
|
75
|
+
|
|
76
|
+
[tool.ruff]
|
|
77
|
+
line-length = 100
|
|
78
|
+
|
|
79
|
+
[tool.ruff.lint]
|
|
80
|
+
select = [
|
|
81
|
+
"E", # pycodestyle errors
|
|
82
|
+
"F", # pyflakes (F401 = unused imports, F403 = wildcard imports)
|
|
83
|
+
"W", # pycodestyle warnings
|
|
84
|
+
"I", # isort import order
|
|
85
|
+
"UP006", # List[X] → list[X]
|
|
86
|
+
"UP007", # Optional[X] → X | None
|
|
87
|
+
"UP035", # deprecated typing imports
|
|
88
|
+
"T201", # print() call
|
|
89
|
+
"F403", # wildcard imports
|
|
90
|
+
"RUF100", # unused noqa directives
|
|
91
|
+
]
|
|
92
|
+
ignore = [
|
|
93
|
+
# This project requires `logger = logging.getLogger(__name__)` immediately
|
|
94
|
+
# after `import logging`, before all third-party imports — E402 flags exactly
|
|
95
|
+
# that placement as "import not at top of file". The convention is correct
|
|
96
|
+
# and deliberate, so this rule is disabled project-wide.
|
|
97
|
+
"E402",
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
[tool.ruff.lint.per-file-ignores]
|
|
101
|
+
# Not pytest-collected — a manual, human-run CLI script (see its own docstring).
|
|
102
|
+
# print() is the actual intended output mechanism here, not a logging omission.
|
|
103
|
+
"tests/manual_verify_interactive_flow.py" = ["T201"]
|
|
104
|
+
"tests/manual_verify_leaderboard_tuning.py" = ["T201"]
|
|
105
|
+
"tests/manual_verify_environment.py" = ["T201"]
|
|
106
|
+
|
|
107
|
+
[tool.mypy]
|
|
108
|
+
# Placeholder — strict type-checking disabled for now; enable incrementally
|
|
109
|
+
ignore_errors = true
|
|
110
|
+
|
|
111
|
+
[tool.pytest.ini_options]
|
|
112
|
+
testpaths = ["tests"]
|
|
113
|
+
addopts = "--cov=dscompanion --cov-report=term-missing --cov-fail-under=80"
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""dscompanion — production-grade ML toolkit for financial modelling."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
|
|
5
|
+
logging.basicConfig(
|
|
6
|
+
level=logging.INFO,
|
|
7
|
+
format="%(asctime)s %(levelname)-8s %(name)s %(message)s",
|
|
8
|
+
datefmt="%Y-%m-%d %H:%M:%S",
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
from dscompanion.calibration import Calibrator
|
|
12
|
+
from dscompanion.config import settings
|
|
13
|
+
from dscompanion.docs import ModelCard
|
|
14
|
+
from dscompanion.eda import EDAReport
|
|
15
|
+
from dscompanion.explain import BootstrapSHAPExplainer, LIMEExplainer, SHAPExplainer
|
|
16
|
+
from dscompanion.features import FeatureProcessingPipeline
|
|
17
|
+
from dscompanion.leaderboard import Leaderboard
|
|
18
|
+
from dscompanion.models import ModelFactory
|
|
19
|
+
from dscompanion.pipeline import PipelineConfig, PipelineRunner, PipelineRunResult
|
|
20
|
+
from dscompanion.selection import FeatureSelectionPipeline
|
|
21
|
+
from dscompanion.split import DataSplit, DataSplitter
|
|
22
|
+
from dscompanion.targets import ImbalanceHandler, TargetBinariser
|
|
23
|
+
from dscompanion.tracking import tracking_run
|
|
24
|
+
from dscompanion.tuning import Tuner
|
|
25
|
+
from dscompanion.utils.synthetic import SyntheticDataGenerator
|
|
26
|
+
|
|
27
|
+
__version__ = "0.1.0"
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"__version__",
|
|
31
|
+
# Split
|
|
32
|
+
"DataSplitter",
|
|
33
|
+
"DataSplit",
|
|
34
|
+
# EDA
|
|
35
|
+
"EDAReport",
|
|
36
|
+
# Features
|
|
37
|
+
"FeatureProcessingPipeline",
|
|
38
|
+
# Targets
|
|
39
|
+
"ImbalanceHandler",
|
|
40
|
+
"TargetBinariser",
|
|
41
|
+
# Selection
|
|
42
|
+
"FeatureSelectionPipeline",
|
|
43
|
+
# Models
|
|
44
|
+
"ModelFactory",
|
|
45
|
+
# Leaderboard
|
|
46
|
+
"Leaderboard",
|
|
47
|
+
# Tuning
|
|
48
|
+
"Tuner",
|
|
49
|
+
# Explainability
|
|
50
|
+
"SHAPExplainer",
|
|
51
|
+
"BootstrapSHAPExplainer",
|
|
52
|
+
"LIMEExplainer",
|
|
53
|
+
# Calibration
|
|
54
|
+
"Calibrator",
|
|
55
|
+
# Docs
|
|
56
|
+
"ModelCard",
|
|
57
|
+
# Tracking
|
|
58
|
+
"tracking_run",
|
|
59
|
+
# Config
|
|
60
|
+
"settings",
|
|
61
|
+
# Pipeline
|
|
62
|
+
"PipelineConfig",
|
|
63
|
+
"PipelineRunner",
|
|
64
|
+
"PipelineRunResult",
|
|
65
|
+
# Utils
|
|
66
|
+
"SyntheticDataGenerator",
|
|
67
|
+
]
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""FastAPI service wrapping dscompanion for a web frontend.
|
|
2
|
+
|
|
3
|
+
An optional extra (``pip install dscompanion[api]``), kept separate from the
|
|
4
|
+
core library so the core has no FastAPI/web-framework dependency — mirrors
|
|
5
|
+
``app/``'s relationship to the core package.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""dscompanion.api.auth — pluggable auth backends, dispatched by APISettings.auth_backend."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
logger = logging.getLogger(__name__)
|
|
8
|
+
|
|
9
|
+
__all__ = ["get_auth_backend", "AuthResult"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class AuthResult:
|
|
13
|
+
"""Outcome of an auth check.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
authenticated (bool): Whether the request is allowed through.
|
|
17
|
+
principal (str | None): Identifier for who/what made the request (e.g. the
|
|
18
|
+
API key's label, or an OAuth subject claim). ``None`` when
|
|
19
|
+
``authenticated=False``.
|
|
20
|
+
|
|
21
|
+
Returns:
|
|
22
|
+
AuthResult: A simple, backend-agnostic result every auth module returns,
|
|
23
|
+
so ``deps.py`` never has to know which backend produced it.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
def __init__(self, authenticated: bool, principal: str | None = None) -> None:
|
|
27
|
+
self.authenticated = authenticated
|
|
28
|
+
self.principal = principal
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def get_auth_backend():
|
|
32
|
+
"""Returns the auth-check callable for the currently configured backend,
|
|
33
|
+
dispatched by ``api_settings.auth_backend`` exactly like ``Tuner.backend``
|
|
34
|
+
dispatches to ``OptunaBackend``/``HyperoptBackend`` — a plain string setting
|
|
35
|
+
plus a lazy import per branch, no abstract base class.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
None
|
|
39
|
+
|
|
40
|
+
Returns:
|
|
41
|
+
Callable[[str | None], AuthResult]: A function taking the raw credential
|
|
42
|
+
(e.g. the ``X-API-Key`` header value) and returning an ``AuthResult``.
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
ValueError: If ``api_settings.auth_backend`` is not a recognised value.
|
|
46
|
+
"""
|
|
47
|
+
from dscompanion.api.config import api_settings
|
|
48
|
+
|
|
49
|
+
if api_settings.auth_backend == "api_key":
|
|
50
|
+
from dscompanion.api.auth.api_key import check_api_key
|
|
51
|
+
|
|
52
|
+
return check_api_key
|
|
53
|
+
elif api_settings.auth_backend == "databricks_oauth":
|
|
54
|
+
from dscompanion.api.auth.databricks_oauth import check_databricks_oauth
|
|
55
|
+
|
|
56
|
+
return check_databricks_oauth
|
|
57
|
+
elif api_settings.auth_backend == "azure_entra":
|
|
58
|
+
from dscompanion.api.auth.azure_entra import check_azure_entra
|
|
59
|
+
|
|
60
|
+
return check_azure_entra
|
|
61
|
+
else:
|
|
62
|
+
raise ValueError(
|
|
63
|
+
f"Unknown auth_backend: {api_settings.auth_backend!r}. "
|
|
64
|
+
"Choose api_key/databricks_oauth/azure_entra."
|
|
65
|
+
)
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""api_key auth backend — local-dev stub, checked against APISettings.api_key."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
from dscompanion.api.auth import AuthResult
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
__all__ = ["check_api_key"]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def check_api_key(credential: str | None) -> AuthResult:
|
|
15
|
+
"""Checks ``credential`` against ``api_settings.api_key``.
|
|
16
|
+
|
|
17
|
+
Local-dev only — needed regardless of eventual deployment target, easy to
|
|
18
|
+
upgrade later. Real auth (SSO/Entra, Databricks OAuth) is deferred until
|
|
19
|
+
that deployment channel is actually pursued.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
credential (str | None): The raw ``X-API-Key`` header value, or ``None``
|
|
23
|
+
if the header was absent.
|
|
24
|
+
|
|
25
|
+
Returns:
|
|
26
|
+
AuthResult: ``authenticated=True`` when ``api_settings.api_key`` is
|
|
27
|
+
``None`` (auth disabled — local dev only) or when ``credential`` matches
|
|
28
|
+
it exactly; ``authenticated=False`` otherwise.
|
|
29
|
+
"""
|
|
30
|
+
from dscompanion.api.config import api_settings
|
|
31
|
+
|
|
32
|
+
if api_settings.api_key is None:
|
|
33
|
+
logger.debug("api_key auth: no key configured, allowing all requests (local dev only)")
|
|
34
|
+
return AuthResult(authenticated=True, principal="local-dev")
|
|
35
|
+
|
|
36
|
+
if credential == api_settings.api_key:
|
|
37
|
+
return AuthResult(authenticated=True, principal="api-key-user")
|
|
38
|
+
|
|
39
|
+
logger.warning("api_key auth: rejected request with invalid or missing credential")
|
|
40
|
+
return AuthResult(authenticated=False, principal=None)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""azure_entra auth backend — stub only, not built.
|
|
2
|
+
|
|
3
|
+
Same treatment as databricks_oauth.py: this file exists so the seam is ready,
|
|
4
|
+
not implemented until the Azure channel is
|
|
5
|
+
actually pursued. Would lazy-import azure-identity
|
|
6
|
+
inside the function body when implemented, never at module level.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dscompanion.api.auth import AuthResult
|
|
12
|
+
|
|
13
|
+
__all__ = ["check_azure_entra"]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def check_azure_entra(credential: str | None) -> AuthResult:
|
|
17
|
+
"""Not implemented — see module docstring.
|
|
18
|
+
|
|
19
|
+
Args:
|
|
20
|
+
credential (str | None): The raw bearer token, once implemented.
|
|
21
|
+
|
|
22
|
+
Returns:
|
|
23
|
+
AuthResult: Never returns.
|
|
24
|
+
|
|
25
|
+
Raises:
|
|
26
|
+
NotImplementedError: Always — this backend has no implementation yet.
|
|
27
|
+
"""
|
|
28
|
+
raise NotImplementedError(
|
|
29
|
+
"auth_backend='azure_entra' is not built yet. "
|
|
30
|
+
"Use 'api_key' (the default) until the Azure channel is actually pursued."
|
|
31
|
+
)
|