betise 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
betise-0.2.0/LICENSE ADDED
@@ -0,0 +1,22 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 [Your Name]
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
22
+
betise-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,240 @@
1
+ Metadata-Version: 2.4
2
+ Name: betise
3
+ Version: 0.2.0
4
+ Summary: BeTiSe — Benchmark Time Series Generator for synthetic dataset creation
5
+ Author: Pınar Cemre Yazıcı, Pelin Erkaya, Yağmur Türkmen
6
+ Author-email: İsmail Güzel <ismailgzel@gmail.com>
7
+ License-Expression: MIT
8
+ Project-URL: Homepage, https://github.com/ismailguzel/betise
9
+ Project-URL: Repository, https://github.com/ismailguzel/betise
10
+ Project-URL: Bug Tracker, https://github.com/ismailguzel/betise/issues
11
+ Project-URL: Dataset, https://doi.org/10.5281/zenodo.18513505
12
+ Keywords: time-series,synthetic-data,dataset-generation,benchmark,arima,garch,anomaly-detection,machine-learning,forecasting,stationarity
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.8
20
+ Classifier: Programming Language :: Python :: 3.9
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Programming Language :: Python :: 3.12
24
+ Requires-Python: >=3.8
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ Requires-Dist: numpy>=1.21
28
+ Requires-Dist: pandas>=1.3
29
+ Requires-Dist: statsmodels>=0.13
30
+ Requires-Dist: arch>=5.0
31
+ Requires-Dist: pyarrow>=7.0
32
+ Requires-Dist: matplotlib>=3.4
33
+ Requires-Dist: tqdm>=4.60
34
+ Provides-Extra: dev
35
+ Requires-Dist: pytest>=7.0; extra == "dev"
36
+ Requires-Dist: pytest-cov>=4.0; extra == "dev"
37
+ Requires-Dist: black>=23.0; extra == "dev"
38
+ Requires-Dist: ruff>=0.1; extra == "dev"
39
+ Requires-Dist: mypy>=1.0; extra == "dev"
40
+ Provides-Extra: docs
41
+ Requires-Dist: mkdocs>=1.5; extra == "docs"
42
+ Requires-Dist: mkdocs-material>=9.0; extra == "docs"
43
+ Requires-Dist: mkdocstrings[python]>=0.24; extra == "docs"
44
+ Dynamic: license-file
45
+
46
+ # BeTiSe — Benchmark Time Series Generator
47
+
48
+ A modular Python library for generating synthetic time series datasets with rich, reproducible metadata.
49
+
50
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
51
+ [![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/)
52
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.18513505.svg)](https://doi.org/10.5281/zenodo.18513505)
53
+
54
+ ## Overview
55
+
56
+ BeTiSe provides a comprehensive toolkit for generating synthetic time series data with configurable statistical properties. It is designed for researchers, data scientists, and ML practitioners who need reproducible, well-documented time series datasets for benchmarking, model training, or educational purposes.
57
+
58
+ ## Published Dataset
59
+
60
+ A large-scale benchmark dataset generated with this library has been published on Zenodo.
61
+
62
+ - **Dataset Name**: BeTiSe: A Benchmark Time Series Dataset for Stationarity and Structural Analysis
63
+ - **DOI**: [10.5281/zenodo.18513505](https://doi.org/10.5281/zenodo.18513505)
64
+ - **Conference**: Submitted to [ITISE 2026](https://itise.ugr.es/)
65
+
66
+ Access: [https://zenodo.org/records/18513505](https://zenodo.org/records/18513505)
67
+
68
+ ## Installation
69
+
70
+ ```bash
71
+ pip install betise
72
+ ```
73
+
74
+ Or install from source:
75
+
76
+ ```bash
77
+ git clone https://github.com/ismailguzel/betise.git
78
+ cd betise
79
+ pip install -e .
80
+ ```
81
+
82
+ ## Quick Start
83
+
84
+ ```python
85
+ from betise import generate_dataframe, load_config
86
+
87
+ # In-memory — no file written
88
+ cfg = load_config(dataset={"base_series": "arma", "num_series": 5, "length_range": [300, 500]})
89
+ df, ctx = generate_dataframe(cfg)
90
+
91
+ # Save to parquet
92
+ from betise import run
93
+
94
+ cfg = load_config(dataset={
95
+ "base_series": "ar",
96
+ "num_series": 10,
97
+ "length_range": [200, 500],
98
+ "output_dir": "output",
99
+ "output_name": "ar_demo.parquet",
100
+ "features": {
101
+ "linear_trend": {"enabled": True, "direction": "upward"},
102
+ },
103
+ })
104
+ run(cfg)
105
+ ```
106
+
107
+ ### Load generated data
108
+
109
+ ```python
110
+ import pandas as pd
111
+
112
+ df = pd.read_parquet("output/ar_demo.parquet")
113
+ print(df[["series_id", "time", "data", "primary_category", "sub_category"]].head())
114
+ ```
115
+
116
+ For full loading examples (numpy, sklearn, PyTorch) see `examples/06_load_and_use.py`.
117
+
118
+ ## Series Types
119
+
120
+ | Category | Base types |
121
+ |---|---|
122
+ | Stationary | `ar`, `ma`, `arma`, `white_noise` |
123
+ | Stochastic | `random_walk`, `random_walk_drift`, `ari`, `ima`, `arima` |
124
+ | Seasonal | `sarma`, `sarima` |
125
+ | Volatility | `arch`, `garch`, `egarch`, `aparch` |
126
+
127
+ Feature overlays (trend, seasonality, anomaly, structural break) can be combined on top of any base type. See [USAGE.md](USAGE.md) for the full feature reference.
128
+
129
+ ## Examples
130
+
131
+ ```
132
+ examples/
133
+ ├── 00_introduction.ipynb # Interactive getting-started notebook
134
+ ├── 01_quickstart.py # In-memory generation, save to disk, feature combinations
135
+ ├── 02_benchmark_dataset.py # All base types × 3 length buckets (~495 series)
136
+ ├── 03_feature_suite.py # All base types × all feature types, phased (~4,200 series)
137
+ ├── 04_pretraining_dataset.py # Large-scale fixed-length dataset (default 75k, scalable)
138
+ ├── 05_classification_dataset.py # Balanced 7-class ML dataset (14,000 series)
139
+ ├── 06_load_and_use.py # Load parquet → numpy / sklearn / PyTorch
140
+ ├── 07_feature_gallery.py # PDF gallery: all 15 base types + all 12 features
141
+ ├── 08_combinations_gallery.py # PDF gallery: every base × feature combination (545 plots)
142
+ ├── configs/
143
+ │ └── classification_config.json # Class / sub-type config for script 05
144
+ └── data/
145
+ └── combinations.csv # Combination definitions for script 08
146
+ ```
147
+
148
+ Run any example:
149
+
150
+ ```bash
151
+ python examples/01_quickstart.py
152
+ python examples/07_feature_gallery.py # produces feature_gallery.pdf
153
+ python examples/08_combinations_gallery.py # produces combinations_gallery.pdf
154
+ ```
155
+
156
+ ## Project Structure
157
+
158
+ ```
159
+ betise/
160
+ ├── betise/
161
+ │ ├── __init__.py # Public API: run, generate_dataframe, load_config
162
+ │ ├── dataset_generation.py # generate_dataframe() / run() pipeline
163
+ │ ├── config/
164
+ │ │ ├── __init__.py # load_config() with deep merge
165
+ │ │ ├── dataset.json # Default dataset settings
166
+ │ │ └── params.json # Default process parameters
167
+ │ ├── core/
168
+ │ │ ├── generator.py # TimeSeriesGenerator
169
+ │ │ └── metadata.py # create_metadata_record()
170
+ │ └── utils/
171
+ │ └── helpers.py # Internal helpers
172
+ ├── examples/ # Ready-to-run scenarios (see above)
173
+ ├── tests/ # Test suite
174
+ ├── USAGE.md # Full feature & config reference
175
+ ├── pyproject.toml
176
+ └── requirements.txt
177
+ ```
178
+
179
+ ## Reproducibility
180
+
181
+ Default seed is 42. ARCH/GARCH models may show minor non-determinism (~1–2%) due to upstream library behaviour.
182
+
183
+ ## Dependencies
184
+
185
+ | Package | Min version | Purpose |
186
+ |---|---|---|
187
+ | `numpy` | 1.21 | Array operations |
188
+ | `pandas` | 1.3 | DataFrame output |
189
+ | `statsmodels` | 0.13 | ARIMA/SARIMA generation |
190
+ | `arch` | 5.0 | ARCH/GARCH generation |
191
+ | `pyarrow` | 7.0 | Parquet I/O |
192
+
193
+ ## Citation
194
+
195
+ If you use BeTiSe or the published dataset in your research, please cite:
196
+
197
+ ```bibtex
198
+ @dataset{betise2026,
199
+ author = {Gür, Kerem and Yazıcı, Pınar Cemre and Erkaya, Pelin and Türkmen, Yağmur and Baytak, Berke and Güzel, İsmail and Karagöz, Pınar and Yozgatlıgil, Ceylan}},
200
+ title = {{BeTiSe: A Benchmark Time Series Dataset for Stationarity
201
+ and Structural Analysis}},
202
+ year = {2026},
203
+ publisher = {Zenodo},
204
+ doi = {10.5281/zenodo.18513505},
205
+ url = {https://doi.org/10.5281/zenodo.18513505}
206
+ }
207
+ ```
208
+
209
+ ## Funding
210
+
211
+ - **TÜBİTAK** — Grant No. 124F095
212
+ - **METU** Scientific Research Projects — Grant No. GAP-109-2023-11361
213
+
214
+ ## Contributors
215
+
216
+ | Name | Role |
217
+ |---|---|
218
+ | İsmail Güzel | Library design, implementation & maintenance |
219
+ | Pınar Cemre Yazıcı | Core development |
220
+ | Pelin Erkaya | Core development |
221
+ | Yağmur Türkmen | Core development |
222
+
223
+ The broader research team (Kerem Gür, Berke Baytak, Pınar Karagöz, Ceylan Yozgatlıgil) contributed to the research project and are credited in the dataset publication.
224
+
225
+ ## Contact
226
+
227
+ For questions, bug reports, or collaboration inquiries:
228
+ **İsmail Güzel** — ismailgzel@gmail.com
229
+
230
+ ## Contributing
231
+
232
+ Issues and pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
233
+
234
+ ## License
235
+
236
+ MIT — see [LICENSE](LICENSE).
237
+
238
+ ---
239
+
240
+ **Version**: 0.2.0 | **License**: MIT
betise-0.2.0/README.md ADDED
@@ -0,0 +1,195 @@
1
+ # BeTiSe — Benchmark Time Series Generator
2
+
3
+ A modular Python library for generating synthetic time series datasets with rich, reproducible metadata.
4
+
5
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
6
+ [![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/)
7
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.18513505.svg)](https://doi.org/10.5281/zenodo.18513505)
8
+
9
+ ## Overview
10
+
11
+ BeTiSe provides a comprehensive toolkit for generating synthetic time series data with configurable statistical properties. It is designed for researchers, data scientists, and ML practitioners who need reproducible, well-documented time series datasets for benchmarking, model training, or educational purposes.
12
+
13
+ ## Published Dataset
14
+
15
+ A large-scale benchmark dataset generated with this library has been published on Zenodo.
16
+
17
+ - **Dataset Name**: BeTiSe: A Benchmark Time Series Dataset for Stationarity and Structural Analysis
18
+ - **DOI**: [10.5281/zenodo.18513505](https://doi.org/10.5281/zenodo.18513505)
19
+ - **Conference**: Submitted to [ITISE 2026](https://itise.ugr.es/)
20
+
21
+ Access: [https://zenodo.org/records/18513505](https://zenodo.org/records/18513505)
22
+
23
+ ## Installation
24
+
25
+ ```bash
26
+ pip install betise
27
+ ```
28
+
29
+ Or install from source:
30
+
31
+ ```bash
32
+ git clone https://github.com/ismailguzel/betise.git
33
+ cd betise
34
+ pip install -e .
35
+ ```
36
+
37
+ ## Quick Start
38
+
39
+ ```python
40
+ from betise import generate_dataframe, load_config
41
+
42
+ # In-memory — no file written
43
+ cfg = load_config(dataset={"base_series": "arma", "num_series": 5, "length_range": [300, 500]})
44
+ df, ctx = generate_dataframe(cfg)
45
+
46
+ # Save to parquet
47
+ from betise import run
48
+
49
+ cfg = load_config(dataset={
50
+ "base_series": "ar",
51
+ "num_series": 10,
52
+ "length_range": [200, 500],
53
+ "output_dir": "output",
54
+ "output_name": "ar_demo.parquet",
55
+ "features": {
56
+ "linear_trend": {"enabled": True, "direction": "upward"},
57
+ },
58
+ })
59
+ run(cfg)
60
+ ```
61
+
62
+ ### Load generated data
63
+
64
+ ```python
65
+ import pandas as pd
66
+
67
+ df = pd.read_parquet("output/ar_demo.parquet")
68
+ print(df[["series_id", "time", "data", "primary_category", "sub_category"]].head())
69
+ ```
70
+
71
+ For full loading examples (numpy, sklearn, PyTorch) see `examples/06_load_and_use.py`.
72
+
73
+ ## Series Types
74
+
75
+ | Category | Base types |
76
+ |---|---|
77
+ | Stationary | `ar`, `ma`, `arma`, `white_noise` |
78
+ | Stochastic | `random_walk`, `random_walk_drift`, `ari`, `ima`, `arima` |
79
+ | Seasonal | `sarma`, `sarima` |
80
+ | Volatility | `arch`, `garch`, `egarch`, `aparch` |
81
+
82
+ Feature overlays (trend, seasonality, anomaly, structural break) can be combined on top of any base type. See [USAGE.md](USAGE.md) for the full feature reference.
83
+
84
+ ## Examples
85
+
86
+ ```
87
+ examples/
88
+ ├── 00_introduction.ipynb # Interactive getting-started notebook
89
+ ├── 01_quickstart.py # In-memory generation, save to disk, feature combinations
90
+ ├── 02_benchmark_dataset.py # All base types × 3 length buckets (~495 series)
91
+ ├── 03_feature_suite.py # All base types × all feature types, phased (~4,200 series)
92
+ ├── 04_pretraining_dataset.py # Large-scale fixed-length dataset (default 75k, scalable)
93
+ ├── 05_classification_dataset.py # Balanced 7-class ML dataset (14,000 series)
94
+ ├── 06_load_and_use.py # Load parquet → numpy / sklearn / PyTorch
95
+ ├── 07_feature_gallery.py # PDF gallery: all 15 base types + all 12 features
96
+ ├── 08_combinations_gallery.py # PDF gallery: every base × feature combination (545 plots)
97
+ ├── configs/
98
+ │ └── classification_config.json # Class / sub-type config for script 05
99
+ └── data/
100
+ └── combinations.csv # Combination definitions for script 08
101
+ ```
102
+
103
+ Run any example:
104
+
105
+ ```bash
106
+ python examples/01_quickstart.py
107
+ python examples/07_feature_gallery.py # produces feature_gallery.pdf
108
+ python examples/08_combinations_gallery.py # produces combinations_gallery.pdf
109
+ ```
110
+
111
+ ## Project Structure
112
+
113
+ ```
114
+ betise/
115
+ ├── betise/
116
+ │ ├── __init__.py # Public API: run, generate_dataframe, load_config
117
+ │ ├── dataset_generation.py # generate_dataframe() / run() pipeline
118
+ │ ├── config/
119
+ │ │ ├── __init__.py # load_config() with deep merge
120
+ │ │ ├── dataset.json # Default dataset settings
121
+ │ │ └── params.json # Default process parameters
122
+ │ ├── core/
123
+ │ │ ├── generator.py # TimeSeriesGenerator
124
+ │ │ └── metadata.py # create_metadata_record()
125
+ │ └── utils/
126
+ │ └── helpers.py # Internal helpers
127
+ ├── examples/ # Ready-to-run scenarios (see above)
128
+ ├── tests/ # Test suite
129
+ ├── USAGE.md # Full feature & config reference
130
+ ├── pyproject.toml
131
+ └── requirements.txt
132
+ ```
133
+
134
+ ## Reproducibility
135
+
136
+ Default seed is 42. ARCH/GARCH models may show minor non-determinism (~1–2%) due to upstream library behaviour.
137
+
138
+ ## Dependencies
139
+
140
+ | Package | Min version | Purpose |
141
+ |---|---|---|
142
+ | `numpy` | 1.21 | Array operations |
143
+ | `pandas` | 1.3 | DataFrame output |
144
+ | `statsmodels` | 0.13 | ARIMA/SARIMA generation |
145
+ | `arch` | 5.0 | ARCH/GARCH generation |
146
+ | `pyarrow` | 7.0 | Parquet I/O |
147
+
148
+ ## Citation
149
+
150
+ If you use BeTiSe or the published dataset in your research, please cite:
151
+
152
+ ```bibtex
153
+ @dataset{betise2026,
154
+ author = {Gür, Kerem and Yazıcı, Pınar Cemre and Erkaya, Pelin and Türkmen, Yağmur and Baytak, Berke and Güzel, İsmail and Karagöz, Pınar and Yozgatlıgil, Ceylan}},
155
+ title = {{BeTiSe: A Benchmark Time Series Dataset for Stationarity
156
+ and Structural Analysis}},
157
+ year = {2026},
158
+ publisher = {Zenodo},
159
+ doi = {10.5281/zenodo.18513505},
160
+ url = {https://doi.org/10.5281/zenodo.18513505}
161
+ }
162
+ ```
163
+
164
+ ## Funding
165
+
166
+ - **TÜBİTAK** — Grant No. 124F095
167
+ - **METU** Scientific Research Projects — Grant No. GAP-109-2023-11361
168
+
169
+ ## Contributors
170
+
171
+ | Name | Role |
172
+ |---|---|
173
+ | İsmail Güzel | Library design, implementation & maintenance |
174
+ | Pınar Cemre Yazıcı | Core development |
175
+ | Pelin Erkaya | Core development |
176
+ | Yağmur Türkmen | Core development |
177
+
178
+ The broader research team (Kerem Gür, Berke Baytak, Pınar Karagöz, Ceylan Yozgatlıgil) contributed to the research project and are credited in the dataset publication.
179
+
180
+ ## Contact
181
+
182
+ For questions, bug reports, or collaboration inquiries:
183
+ **İsmail Güzel** — ismailgzel@gmail.com
184
+
185
+ ## Contributing
186
+
187
+ Issues and pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
188
+
189
+ ## License
190
+
191
+ MIT — see [LICENSE](LICENSE).
192
+
193
+ ---
194
+
195
+ **Version**: 0.2.0 | **License**: MIT
@@ -0,0 +1,65 @@
1
+ """BeTiSe — Benchmark Time Series Generator.
2
+
3
+ Generate synthetic time series with configurable statistical properties:
4
+ stationary processes (AR, MA, ARMA), stochastic-trend processes (ARIMA family),
5
+ seasonal models (SARMA, SARIMA), and volatility models (ARCH/GARCH family).
6
+ Feature overlays — trend, seasonality, anomaly, structural break — can be
7
+ stacked on top of any base process.
8
+
9
+ Quick start
10
+ -----------
11
+ >>> from betise import generate_dataframe
12
+ >>> from betise.config import load_config
13
+ >>>
14
+ >>> cfg = load_config(dataset={"base_series": "arma", "num_series": 5})
15
+ >>> df, ctx = generate_dataframe(cfg)
16
+
17
+ Save to disk
18
+ ------------
19
+ >>> from betise import run
20
+ >>> cfg = load_config(dataset={
21
+ ... "base_series": "ar",
22
+ ... "num_series": 100,
23
+ ... "output_dir": "output",
24
+ ... "output_name": "ar_dataset.parquet",
25
+ ... })
26
+ >>> run(cfg)
27
+ """
28
+
29
+ from importlib.metadata import version, PackageNotFoundError
30
+
31
+ try:
32
+ __version__ = version("betise")
33
+ except PackageNotFoundError: # running from source without install
34
+ __version__ = "0.2.0"
35
+
36
+ __author__ = "Pınar Cemre Yazıcı, Pelin Erkaya, Yağmur Türkmen, İsmail Güzel"
37
+ __license__ = "MIT"
38
+
39
+ from .core.generator import TimeSeriesGenerator
40
+ from .core.metadata import (
41
+ create_metadata_record,
42
+ attach_metadata_columns_to_df,
43
+ get_metadata_columns_defaults,
44
+ make_json_serializable,
45
+ )
46
+ from .dataset_generation import run, generate_dataframe
47
+ from .config import load_config
48
+
49
+ __all__ = [
50
+ # Primary API
51
+ "run",
52
+ "generate_dataframe",
53
+ "load_config",
54
+ # Core class
55
+ "TimeSeriesGenerator",
56
+ # Metadata helpers
57
+ "create_metadata_record",
58
+ "attach_metadata_columns_to_df",
59
+ "get_metadata_columns_defaults",
60
+ "make_json_serializable",
61
+ # Package info
62
+ "__version__",
63
+ "__author__",
64
+ "__license__",
65
+ ]
@@ -0,0 +1,83 @@
1
+ """Configuration loader for the generation pipeline."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import copy
6
+ import json
7
+ from pathlib import Path
8
+ from typing import Any, Dict, Optional
9
+
10
+
11
+ def _load_json_config(path: Path) -> Dict[str, Any]:
12
+ if not path.exists():
13
+ raise FileNotFoundError(f"Missing config file: {path}")
14
+ with path.open("r", encoding="utf-8") as f:
15
+ return json.load(f)
16
+
17
+
18
+ def _deep_merge(base: Dict[str, Any], override: Dict[str, Any]) -> Dict[str, Any]:
19
+ """Recursively merge override into base (override wins on scalar conflicts)."""
20
+ result = copy.deepcopy(base)
21
+ for key, value in override.items():
22
+ if key in result and isinstance(result[key], dict) and isinstance(value, dict):
23
+ result[key] = _deep_merge(result[key], value)
24
+ else:
25
+ result[key] = copy.deepcopy(value)
26
+ return result
27
+
28
+
29
+ def load_config(
30
+ config_dir: Optional[str] = None,
31
+ *,
32
+ dataset: Optional[Dict[str, Any]] = None,
33
+ params: Optional[Dict[str, Any]] = None,
34
+ ) -> Dict[str, Any]:
35
+ """Load generation configuration, with optional in-memory overrides.
36
+
37
+ Parameters
38
+ ----------
39
+ config_dir : str, optional
40
+ Directory containing ``params.json`` and ``dataset.json``.
41
+ Defaults to the built-in config directory shipped with the package.
42
+ dataset : dict, optional
43
+ In-memory overrides for the dataset config (deep-merged over the
44
+ file defaults). Useful for scripted generation without editing files.
45
+ params : dict, optional
46
+ In-memory overrides for the params config.
47
+
48
+ Returns
49
+ -------
50
+ dict
51
+ ``{"params": {...}, "dataset": {...}}`` ready for generate_dataframe().
52
+
53
+ Examples
54
+ --------
55
+ # Default config only
56
+ cfg = load_config()
57
+
58
+ # In-memory override (no file changes needed)
59
+ cfg = load_config(dataset={
60
+ "base_series": "arima",
61
+ "num_series": 50,
62
+ "length_range": [300, 600],
63
+ "random_seed": 7,
64
+ "features": {
65
+ "linear_trend": {"enabled": True, "direction": "upward"},
66
+ },
67
+ })
68
+
69
+ # Load from a custom directory then override
70
+ cfg = load_config("path/to/my_configs/", dataset={"num_series": 5})
71
+ """
72
+ base_dir = Path(config_dir) if config_dir else Path(__file__).resolve().parent
73
+
74
+ cfg_params = _load_json_config(base_dir / "params.json")
75
+ cfg_dataset = _load_json_config(base_dir / "dataset.json")
76
+
77
+ if params is not None:
78
+ cfg_params = _deep_merge(cfg_params, params)
79
+
80
+ if dataset is not None:
81
+ cfg_dataset = _deep_merge(cfg_dataset, dataset)
82
+
83
+ return {"params": cfg_params, "dataset": cfg_dataset}
@@ -0,0 +1,94 @@
1
+ {
2
+ "random_seed": 42,
3
+ "output_dir": "generated-dataset",
4
+ "output_name": "dataset.parquet",
5
+ "num_series": 1,
6
+ "length_range": [300, 500],
7
+ "include_indices": true,
8
+ "base_series": "ar",
9
+ "features": {
10
+ "linear_trend": {
11
+ "enabled": true,
12
+ "direction": "upward"
13
+ },
14
+ "quadratic_trend": {
15
+ "enabled": false,
16
+ "direction": "upward",
17
+ "location": "center"
18
+ },
19
+ "cubic_trend": {
20
+ "enabled": false,
21
+ "direction": "upward",
22
+ "location": "center"
23
+ },
24
+ "exponential_trend": {
25
+ "enabled": false,
26
+ "direction": "upward"
27
+ },
28
+ "arch": {
29
+ "enabled": false
30
+ },
31
+ "garch": {
32
+ "enabled": false
33
+ },
34
+ "egarch": {
35
+ "enabled": false
36
+ },
37
+ "aparch": {
38
+ "enabled": false
39
+ },
40
+ "single_seasonality": {
41
+ "enabled": false
42
+ },
43
+ "multiple_seasonality": {
44
+ "enabled": false,
45
+ "num_components": 2
46
+ },
47
+ "sarma": {
48
+ "enabled": false
49
+ },
50
+ "sarima": {
51
+ "enabled": false
52
+ },
53
+ "mean_shift": {
54
+ "enabled": false,
55
+ "mode": "single",
56
+ "location": "middle",
57
+ "direction": "down",
58
+ "num_breaks": 1
59
+ },
60
+ "variance_shift": {
61
+ "enabled": false,
62
+ "mode": "single",
63
+ "location": "middle",
64
+ "direction": "up",
65
+ "num_breaks": 1
66
+ },
67
+ "trend_shift": {
68
+ "enabled": false,
69
+ "mode": "single",
70
+ "location": "middle",
71
+ "direction": "up",
72
+ "num_breaks": 1,
73
+ "change_type": "direction_change"
74
+ },
75
+ "point_anomaly": {
76
+ "enabled": false,
77
+ "mode": "single",
78
+ "location": "middle",
79
+ "num_anomalies": 1
80
+ },
81
+ "collective_anomaly": {
82
+ "enabled": false,
83
+ "mode": "single",
84
+ "location": "middle",
85
+ "num_anomalies": 1
86
+ },
87
+ "contextual_anomaly": {
88
+ "enabled": false,
89
+ "mode": "single",
90
+ "location": "middle",
91
+ "num_anomalies": 1
92
+ }
93
+ }
94
+ }