autoforge-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- autoforge_engine-0.1.0.dist-info/METADATA +105 -0
- autoforge_engine-0.1.0.dist-info/RECORD +32 -0
- autoforge_engine-0.1.0.dist-info/WHEEL +5 -0
- autoforge_engine-0.1.0.dist-info/entry_points.txt +2 -0
- autoforge_engine-0.1.0.dist-info/licenses/LICENSE +0 -0
- autoforge_engine-0.1.0.dist-info/top_level.txt +1 -0
- modelforge/artifact_manager.py +485 -0
- modelforge/automl.py +1472 -0
- modelforge/cli.py +1258 -0
- modelforge/column_intelligence.py +404 -0
- modelforge/config.py +580 -0
- modelforge/cross_validation.py +749 -0
- modelforge/data_audit.py +392 -0
- modelforge/data_loader.py +76 -0
- modelforge/evaluation.py +397 -0
- modelforge/experiment_tracker.py +490 -0
- modelforge/explainability.py +346 -0
- modelforge/feature_engineering.py +393 -0
- modelforge/feature_selection.py +528 -0
- modelforge/hyperparameter_optimization.py +593 -0
- modelforge/model_registry.py +684 -0
- modelforge/model_screening.py +531 -0
- modelforge/persistence.py +456 -0
- modelforge/pipeline_generator.py +278 -0
- modelforge/prediction_validator.py +316 -0
- modelforge/preprocessing.py +179 -0
- modelforge/profiler.py +85 -0
- modelforge/ranking.py +351 -0
- modelforge/reproducibility.py +295 -0
- modelforge/reproducibility_integration.py +192 -0
- modelforge/run_manager.py +200 -0
- modelforge/target_selector.py +108 -0
modelforge/ranking.py
ADDED
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class RankingEngine:
|
|
8
|
+
"""
|
|
9
|
+
Rank ModelForge models using multiple objectives.
|
|
10
|
+
|
|
11
|
+
The engine does not train models.
|
|
12
|
+
It only consumes model evaluation results.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
DEFAULT_WEIGHTS = {
|
|
16
|
+
"balanced": {
|
|
17
|
+
"primary": 0.45,
|
|
18
|
+
"error": 0.25,
|
|
19
|
+
"stability": 0.20,
|
|
20
|
+
"speed": 0.10,
|
|
21
|
+
},
|
|
22
|
+
"performance": {
|
|
23
|
+
"primary": 0.70,
|
|
24
|
+
"error": 0.15,
|
|
25
|
+
"stability": 0.10,
|
|
26
|
+
"speed": 0.05,
|
|
27
|
+
},
|
|
28
|
+
"error": {
|
|
29
|
+
"primary": 0.20,
|
|
30
|
+
"error": 0.60,
|
|
31
|
+
"stability": 0.15,
|
|
32
|
+
"speed": 0.05,
|
|
33
|
+
},
|
|
34
|
+
"speed": {
|
|
35
|
+
"primary": 0.20,
|
|
36
|
+
"error": 0.15,
|
|
37
|
+
"stability": 0.15,
|
|
38
|
+
"speed": 0.50,
|
|
39
|
+
},
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
def rank(
|
|
43
|
+
self,
|
|
44
|
+
results: pd.DataFrame,
|
|
45
|
+
task_type: str,
|
|
46
|
+
objective: str = "balanced",
|
|
47
|
+
) -> pd.DataFrame:
|
|
48
|
+
"""
|
|
49
|
+
Rank models using multiple objectives.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
self._validate_inputs(
|
|
53
|
+
results=results,
|
|
54
|
+
task_type=task_type,
|
|
55
|
+
objective=objective,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
ranked = results.copy()
|
|
59
|
+
|
|
60
|
+
weights = self.DEFAULT_WEIGHTS[
|
|
61
|
+
objective
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
primary_metric = (
|
|
65
|
+
self._primary_metric(
|
|
66
|
+
ranked,
|
|
67
|
+
task_type,
|
|
68
|
+
)
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
error_metric = (
|
|
72
|
+
self._error_metric(
|
|
73
|
+
ranked,
|
|
74
|
+
task_type,
|
|
75
|
+
)
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
stability_metric = (
|
|
79
|
+
self._stability_metric(
|
|
80
|
+
ranked,
|
|
81
|
+
task_type,
|
|
82
|
+
)
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
speed_metric = self._speed_metric(ranked)
|
|
86
|
+
|
|
87
|
+
ranked["primary_score"] = (
|
|
88
|
+
self._normalize_metric(
|
|
89
|
+
ranked[primary_metric],
|
|
90
|
+
maximize=True,
|
|
91
|
+
)
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
ranked["error_score"] = (
|
|
95
|
+
self._normalize_metric(
|
|
96
|
+
ranked[error_metric],
|
|
97
|
+
maximize=False,
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
ranked["stability_score"] = (
|
|
102
|
+
self._normalize_metric(
|
|
103
|
+
ranked[stability_metric],
|
|
104
|
+
maximize=False,
|
|
105
|
+
)
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
if speed_metric is None:
|
|
109
|
+
ranked["speed_score"] = 1.0
|
|
110
|
+
else:
|
|
111
|
+
ranked["speed_score"] = self._normalize_metric(
|
|
112
|
+
ranked[speed_metric],
|
|
113
|
+
maximize=False,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
ranked["overall_score"] = (
|
|
117
|
+
ranked["primary_score"]
|
|
118
|
+
* weights["primary"]
|
|
119
|
+
+ ranked["error_score"]
|
|
120
|
+
* weights["error"]
|
|
121
|
+
+ ranked["stability_score"]
|
|
122
|
+
* weights["stability"]
|
|
123
|
+
+ ranked["speed_score"]
|
|
124
|
+
* weights["speed"]
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
ranked["rank"] = (
|
|
128
|
+
ranked["overall_score"]
|
|
129
|
+
.rank(
|
|
130
|
+
ascending=False,
|
|
131
|
+
method="min",
|
|
132
|
+
)
|
|
133
|
+
.astype(int)
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
return (
|
|
137
|
+
ranked.sort_values(
|
|
138
|
+
by=[
|
|
139
|
+
"rank",
|
|
140
|
+
"overall_score",
|
|
141
|
+
],
|
|
142
|
+
ascending=[
|
|
143
|
+
True,
|
|
144
|
+
False,
|
|
145
|
+
],
|
|
146
|
+
)
|
|
147
|
+
.reset_index(drop=True)
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
@staticmethod
|
|
151
|
+
def _primary_metric(
|
|
152
|
+
results: pd.DataFrame,
|
|
153
|
+
task_type: str,
|
|
154
|
+
) -> str:
|
|
155
|
+
"""Return the primary predictive metric."""
|
|
156
|
+
|
|
157
|
+
if task_type == "regression":
|
|
158
|
+
if "cv_mean_r2" in results.columns:
|
|
159
|
+
return "cv_mean_r2"
|
|
160
|
+
|
|
161
|
+
return "r2"
|
|
162
|
+
|
|
163
|
+
if "cv_mean_f1" in results.columns:
|
|
164
|
+
return "cv_mean_f1"
|
|
165
|
+
|
|
166
|
+
return "f1"
|
|
167
|
+
|
|
168
|
+
@staticmethod
|
|
169
|
+
def _error_metric(
|
|
170
|
+
results: pd.DataFrame,
|
|
171
|
+
task_type: str,
|
|
172
|
+
) -> str:
|
|
173
|
+
"""Return the primary error metric."""
|
|
174
|
+
|
|
175
|
+
if task_type == "regression":
|
|
176
|
+
if "cv_mean_rmse" in results.columns:
|
|
177
|
+
return "cv_mean_rmse"
|
|
178
|
+
|
|
179
|
+
if "mae" in results.columns:
|
|
180
|
+
return "mae"
|
|
181
|
+
|
|
182
|
+
return "rmse"
|
|
183
|
+
|
|
184
|
+
if "cv_mean_log_loss" in results.columns:
|
|
185
|
+
return "cv_mean_log_loss"
|
|
186
|
+
|
|
187
|
+
if "log_loss" in results.columns:
|
|
188
|
+
return "log_loss"
|
|
189
|
+
|
|
190
|
+
return "f1"
|
|
191
|
+
|
|
192
|
+
@staticmethod
|
|
193
|
+
def _stability_metric(
|
|
194
|
+
results: pd.DataFrame,
|
|
195
|
+
task_type: str,
|
|
196
|
+
) -> str:
|
|
197
|
+
"""Return the CV stability metric."""
|
|
198
|
+
|
|
199
|
+
if task_type == "regression":
|
|
200
|
+
if "cv_std_r2" in results.columns:
|
|
201
|
+
return "cv_std_r2"
|
|
202
|
+
|
|
203
|
+
return "r2"
|
|
204
|
+
|
|
205
|
+
if "cv_std_f1" in results.columns:
|
|
206
|
+
return "cv_std_f1"
|
|
207
|
+
|
|
208
|
+
return "f1"
|
|
209
|
+
|
|
210
|
+
@staticmethod
|
|
211
|
+
def _speed_metric(
|
|
212
|
+
results: pd.DataFrame,
|
|
213
|
+
) -> str | None:
|
|
214
|
+
"""Return the available training-time metric."""
|
|
215
|
+
|
|
216
|
+
if (
|
|
217
|
+
"total_time_seconds"
|
|
218
|
+
in results.columns
|
|
219
|
+
):
|
|
220
|
+
return "total_time_seconds"
|
|
221
|
+
|
|
222
|
+
if (
|
|
223
|
+
"training_time_seconds"
|
|
224
|
+
in results.columns
|
|
225
|
+
):
|
|
226
|
+
return "training_time_seconds"
|
|
227
|
+
|
|
228
|
+
return None
|
|
229
|
+
|
|
230
|
+
@staticmethod
|
|
231
|
+
def _normalize_metric(
|
|
232
|
+
values: pd.Series,
|
|
233
|
+
maximize: bool,
|
|
234
|
+
) -> pd.Series:
|
|
235
|
+
"""
|
|
236
|
+
Normalize values into [0, 1].
|
|
237
|
+
|
|
238
|
+
Higher values are better when maximize=True.
|
|
239
|
+
Lower values are better when maximize=False.
|
|
240
|
+
"""
|
|
241
|
+
|
|
242
|
+
numeric = pd.to_numeric(
|
|
243
|
+
values,
|
|
244
|
+
errors="coerce",
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
if numeric.isna().all():
|
|
248
|
+
return pd.Series(
|
|
249
|
+
0.0,
|
|
250
|
+
index=values.index,
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
minimum = numeric.min()
|
|
254
|
+
maximum = numeric.max()
|
|
255
|
+
|
|
256
|
+
if np.isclose(
|
|
257
|
+
minimum,
|
|
258
|
+
maximum,
|
|
259
|
+
):
|
|
260
|
+
return pd.Series(
|
|
261
|
+
1.0,
|
|
262
|
+
index=values.index,
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
if maximize:
|
|
266
|
+
normalized = (
|
|
267
|
+
numeric - minimum
|
|
268
|
+
) / (
|
|
269
|
+
maximum - minimum
|
|
270
|
+
)
|
|
271
|
+
else:
|
|
272
|
+
normalized = (
|
|
273
|
+
maximum - numeric
|
|
274
|
+
) / (
|
|
275
|
+
maximum - minimum
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
return normalized.fillna(0.0)
|
|
279
|
+
|
|
280
|
+
@classmethod
|
|
281
|
+
def available_objectives(
|
|
282
|
+
cls,
|
|
283
|
+
) -> list[str]:
|
|
284
|
+
"""Return supported ranking objectives."""
|
|
285
|
+
|
|
286
|
+
return list(
|
|
287
|
+
cls.DEFAULT_WEIGHTS.keys()
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
@classmethod
|
|
291
|
+
def objective_weights(
|
|
292
|
+
cls,
|
|
293
|
+
objective: str,
|
|
294
|
+
) -> dict[str, float]:
|
|
295
|
+
"""Return weights for an objective."""
|
|
296
|
+
|
|
297
|
+
if objective not in cls.DEFAULT_WEIGHTS:
|
|
298
|
+
raise KeyError(
|
|
299
|
+
f"Unknown objective: {objective}"
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
return cls.DEFAULT_WEIGHTS[
|
|
303
|
+
objective
|
|
304
|
+
].copy()
|
|
305
|
+
|
|
306
|
+
@staticmethod
|
|
307
|
+
def _validate_inputs(
|
|
308
|
+
results: pd.DataFrame,
|
|
309
|
+
task_type: str,
|
|
310
|
+
objective: str,
|
|
311
|
+
) -> None:
|
|
312
|
+
"""Validate ranking inputs."""
|
|
313
|
+
|
|
314
|
+
if not isinstance(
|
|
315
|
+
results,
|
|
316
|
+
pd.DataFrame,
|
|
317
|
+
):
|
|
318
|
+
raise TypeError(
|
|
319
|
+
"results must be a pandas DataFrame."
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
if results.empty:
|
|
323
|
+
raise ValueError(
|
|
324
|
+
"Cannot rank an empty results dataset."
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
if task_type not in {
|
|
328
|
+
"regression",
|
|
329
|
+
"classification",
|
|
330
|
+
}:
|
|
331
|
+
raise ValueError(
|
|
332
|
+
"task_type must be 'regression' "
|
|
333
|
+
"or 'classification'."
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
if objective not in {
|
|
337
|
+
"balanced",
|
|
338
|
+
"performance",
|
|
339
|
+
"error",
|
|
340
|
+
"speed",
|
|
341
|
+
}:
|
|
342
|
+
raise ValueError(
|
|
343
|
+
"objective must be one of: "
|
|
344
|
+
"balanced, performance, "
|
|
345
|
+
"error, speed."
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
if "model" not in results.columns:
|
|
349
|
+
raise ValueError(
|
|
350
|
+
"Results must contain a 'model' column."
|
|
351
|
+
)
|
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import platform
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pandas as pd
|
|
12
|
+
import sklearn
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ReproducibilityManager:
|
|
16
|
+
"""
|
|
17
|
+
Creates deterministic fingerprints and reproducibility metadata
|
|
18
|
+
for ModelForge experiments.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, random_state: int = 42) -> None:
|
|
22
|
+
if not isinstance(random_state, int):
|
|
23
|
+
raise TypeError("random_state must be an integer.")
|
|
24
|
+
|
|
25
|
+
self.random_state = random_state
|
|
26
|
+
|
|
27
|
+
def dataset_fingerprint(
|
|
28
|
+
self,
|
|
29
|
+
data: pd.DataFrame,
|
|
30
|
+
) -> str:
|
|
31
|
+
"""
|
|
32
|
+
Generate a deterministic SHA-256 fingerprint for a DataFrame.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
if not isinstance(data, pd.DataFrame):
|
|
36
|
+
raise TypeError("data must be a pandas DataFrame.")
|
|
37
|
+
|
|
38
|
+
if data.empty:
|
|
39
|
+
raise ValueError("data must not be empty.")
|
|
40
|
+
|
|
41
|
+
normalized = data.copy()
|
|
42
|
+
|
|
43
|
+
# Normalize column ordering only through explicit representation.
|
|
44
|
+
# Original column order is preserved because it can affect a pipeline.
|
|
45
|
+
normalized.columns = [str(column) for column in normalized.columns]
|
|
46
|
+
|
|
47
|
+
payload = pd.util.hash_pandas_object(
|
|
48
|
+
normalized,
|
|
49
|
+
index=True,
|
|
50
|
+
).values.tobytes()
|
|
51
|
+
|
|
52
|
+
metadata = {
|
|
53
|
+
"columns": normalized.columns.tolist(),
|
|
54
|
+
"dtypes": {
|
|
55
|
+
column: str(dtype)
|
|
56
|
+
for column, dtype in normalized.dtypes.items()
|
|
57
|
+
},
|
|
58
|
+
"shape": list(normalized.shape),
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
metadata_bytes = json.dumps(
|
|
62
|
+
metadata,
|
|
63
|
+
sort_keys=True,
|
|
64
|
+
default=str,
|
|
65
|
+
).encode("utf-8")
|
|
66
|
+
|
|
67
|
+
digest = hashlib.sha256()
|
|
68
|
+
digest.update(payload)
|
|
69
|
+
digest.update(metadata_bytes)
|
|
70
|
+
|
|
71
|
+
return digest.hexdigest()
|
|
72
|
+
|
|
73
|
+
def configuration_fingerprint(
|
|
74
|
+
self,
|
|
75
|
+
configuration: dict[str, Any],
|
|
76
|
+
) -> str:
|
|
77
|
+
"""
|
|
78
|
+
Generate a deterministic SHA-256 fingerprint for configuration.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
if not isinstance(configuration, dict):
|
|
82
|
+
raise TypeError("configuration must be a dictionary.")
|
|
83
|
+
|
|
84
|
+
serialized = json.dumps(
|
|
85
|
+
configuration,
|
|
86
|
+
sort_keys=True,
|
|
87
|
+
default=str,
|
|
88
|
+
separators=(",", ":"),
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
return hashlib.sha256(
|
|
92
|
+
serialized.encode("utf-8")
|
|
93
|
+
).hexdigest()
|
|
94
|
+
|
|
95
|
+
def artifact_fingerprint(
|
|
96
|
+
self,
|
|
97
|
+
path: str | Path,
|
|
98
|
+
) -> str:
|
|
99
|
+
"""
|
|
100
|
+
Generate a SHA-256 fingerprint for a file artifact.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
file_path = Path(path)
|
|
104
|
+
|
|
105
|
+
if not file_path.exists():
|
|
106
|
+
raise FileNotFoundError(
|
|
107
|
+
f"Artifact does not exist: {file_path}"
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
if not file_path.is_file():
|
|
111
|
+
raise ValueError(
|
|
112
|
+
f"Artifact path is not a file: {file_path}"
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
digest = hashlib.sha256()
|
|
116
|
+
|
|
117
|
+
with file_path.open("rb") as file:
|
|
118
|
+
for chunk in iter(lambda: file.read(1024 * 1024), b""):
|
|
119
|
+
digest.update(chunk)
|
|
120
|
+
|
|
121
|
+
return digest.hexdigest()
|
|
122
|
+
|
|
123
|
+
def environment_metadata(self) -> dict[str, Any]:
|
|
124
|
+
"""
|
|
125
|
+
Capture environment information required to reproduce an experiment.
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
return {
|
|
129
|
+
"python_version": platform.python_version(),
|
|
130
|
+
"python_implementation": platform.python_implementation(),
|
|
131
|
+
"platform": platform.platform(),
|
|
132
|
+
"machine": platform.machine(),
|
|
133
|
+
"numpy_version": np.__version__,
|
|
134
|
+
"pandas_version": pd.__version__,
|
|
135
|
+
"scikit_learn_version": sklearn.__version__,
|
|
136
|
+
"random_state": self.random_state,
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
def create_snapshot(
|
|
140
|
+
self,
|
|
141
|
+
data: pd.DataFrame,
|
|
142
|
+
configuration: dict[str, Any] | None = None,
|
|
143
|
+
target: str | None = None,
|
|
144
|
+
task_type: str | None = None,
|
|
145
|
+
extra_metadata: dict[str, Any] | None = None,
|
|
146
|
+
) -> dict[str, Any]:
|
|
147
|
+
"""
|
|
148
|
+
Create a complete reproducibility snapshot for an experiment.
|
|
149
|
+
"""
|
|
150
|
+
|
|
151
|
+
if not isinstance(data, pd.DataFrame):
|
|
152
|
+
raise TypeError("data must be a pandas DataFrame.")
|
|
153
|
+
|
|
154
|
+
if data.empty:
|
|
155
|
+
raise ValueError("data must not be empty.")
|
|
156
|
+
|
|
157
|
+
configuration = configuration or {}
|
|
158
|
+
dataset_fingerprint = self.dataset_fingerprint(data)
|
|
159
|
+
configuration_fingerprint = (
|
|
160
|
+
self.configuration_fingerprint(configuration)
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
snapshot: dict[str, Any] = {
|
|
164
|
+
"dataset_fingerprint": dataset_fingerprint,
|
|
165
|
+
"configuration_fingerprint": configuration_fingerprint,
|
|
166
|
+
"dataset": {
|
|
167
|
+
"fingerprint": dataset_fingerprint,
|
|
168
|
+
"rows": int(data.shape[0]),
|
|
169
|
+
"columns": int(data.shape[1]),
|
|
170
|
+
"column_names": [
|
|
171
|
+
str(column)
|
|
172
|
+
for column in data.columns
|
|
173
|
+
],
|
|
174
|
+
},
|
|
175
|
+
"configuration": {
|
|
176
|
+
"fingerprint": configuration_fingerprint,
|
|
177
|
+
"values": configuration,
|
|
178
|
+
},
|
|
179
|
+
"target": target,
|
|
180
|
+
"task_type": task_type,
|
|
181
|
+
"environment": self.environment_metadata(),
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
if extra_metadata:
|
|
185
|
+
snapshot["extra_metadata"] = extra_metadata
|
|
186
|
+
|
|
187
|
+
return snapshot
|
|
188
|
+
|
|
189
|
+
def save_snapshot(
|
|
190
|
+
self,
|
|
191
|
+
snapshot: dict[str, Any],
|
|
192
|
+
path: str | Path,
|
|
193
|
+
overwrite: bool = False,
|
|
194
|
+
) -> str:
|
|
195
|
+
"""
|
|
196
|
+
Save a reproducibility snapshot as JSON.
|
|
197
|
+
"""
|
|
198
|
+
|
|
199
|
+
if not isinstance(snapshot, dict):
|
|
200
|
+
raise TypeError("snapshot must be a dictionary.")
|
|
201
|
+
|
|
202
|
+
output_path = Path(path)
|
|
203
|
+
|
|
204
|
+
if output_path.exists() and not overwrite:
|
|
205
|
+
raise FileExistsError(
|
|
206
|
+
f"Snapshot already exists: {output_path}"
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
output_path.parent.mkdir(
|
|
210
|
+
parents=True,
|
|
211
|
+
exist_ok=True,
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
with output_path.open(
|
|
215
|
+
"w",
|
|
216
|
+
encoding="utf-8",
|
|
217
|
+
) as file:
|
|
218
|
+
json.dump(
|
|
219
|
+
snapshot,
|
|
220
|
+
file,
|
|
221
|
+
indent=2,
|
|
222
|
+
sort_keys=True,
|
|
223
|
+
default=str,
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
return str(output_path)
|
|
227
|
+
|
|
228
|
+
def load_snapshot(
|
|
229
|
+
self,
|
|
230
|
+
path: str | Path,
|
|
231
|
+
) -> dict[str, Any]:
|
|
232
|
+
"""
|
|
233
|
+
Load a reproducibility snapshot from JSON.
|
|
234
|
+
"""
|
|
235
|
+
|
|
236
|
+
input_path = Path(path)
|
|
237
|
+
|
|
238
|
+
if not input_path.exists():
|
|
239
|
+
raise FileNotFoundError(
|
|
240
|
+
f"Snapshot does not exist: {input_path}"
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
if not input_path.is_file():
|
|
244
|
+
raise ValueError(
|
|
245
|
+
f"Snapshot path is not a file: {input_path}"
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
with input_path.open(
|
|
249
|
+
"r",
|
|
250
|
+
encoding="utf-8",
|
|
251
|
+
) as file:
|
|
252
|
+
snapshot = json.load(file)
|
|
253
|
+
|
|
254
|
+
if not isinstance(snapshot, dict):
|
|
255
|
+
raise ValueError(
|
|
256
|
+
"Invalid reproducibility snapshot."
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
return snapshot
|
|
260
|
+
|
|
261
|
+
def verify_dataset(
|
|
262
|
+
self,
|
|
263
|
+
data: pd.DataFrame,
|
|
264
|
+
expected_fingerprint: str,
|
|
265
|
+
) -> bool:
|
|
266
|
+
"""
|
|
267
|
+
Verify that a DataFrame matches an expected fingerprint.
|
|
268
|
+
"""
|
|
269
|
+
|
|
270
|
+
if not isinstance(expected_fingerprint, str):
|
|
271
|
+
raise TypeError(
|
|
272
|
+
"expected_fingerprint must be a string."
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
actual_fingerprint = self.dataset_fingerprint(data)
|
|
276
|
+
|
|
277
|
+
return actual_fingerprint == expected_fingerprint
|
|
278
|
+
|
|
279
|
+
def verify_artifact(
|
|
280
|
+
self,
|
|
281
|
+
path: str | Path,
|
|
282
|
+
expected_fingerprint: str,
|
|
283
|
+
) -> bool:
|
|
284
|
+
"""
|
|
285
|
+
Verify that a file matches an expected SHA-256 fingerprint.
|
|
286
|
+
"""
|
|
287
|
+
|
|
288
|
+
if not isinstance(expected_fingerprint, str):
|
|
289
|
+
raise TypeError(
|
|
290
|
+
"expected_fingerprint must be a string."
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
actual_fingerprint = self.artifact_fingerprint(path)
|
|
294
|
+
|
|
295
|
+
return actual_fingerprint == expected_fingerprint
|