silver-diagnostics 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ from .dataset import diagnose_dataset
2
+ from .metrics import diagnose_metrics
3
+ from .models import Diagnostic, DiagnosticReport
4
+
5
+ __all__ = ["Diagnostic", "DiagnosticReport", "diagnose_dataset", "diagnose_metrics"]
@@ -0,0 +1,76 @@
1
+ from typing import Any, List
2
+
3
+ from .models import Diagnostic, DiagnosticReport
4
+ from .utils import is_finite
5
+
6
+
7
+ def diagnose_dataset(
8
+ features: List[List[float]], labels: List[float]
9
+ ) -> DiagnosticReport:
10
+ diagnostics: List[Diagnostic] = []
11
+ if len(features) == 0:
12
+ diagnostics.append(
13
+ Diagnostic(
14
+ code="empty_dataset",
15
+ severity="error",
16
+ message="The dataset contains no feature rows.",
17
+ details={},
18
+ )
19
+ )
20
+ if len(features) != len(labels):
21
+ diagnostics.append(
22
+ Diagnostic(
23
+ code="length_mismatch",
24
+ severity="error",
25
+ message="Features and labels must contain the same number of rows.",
26
+ details={"features": len(features), "labels": len(labels)},
27
+ )
28
+ )
29
+ if features:
30
+ expected_width = len(features[0])
31
+ inconsistent_widths = [
32
+ (index, len(row))
33
+ for index, row in enumerate(features)
34
+ if len(row) != expected_width
35
+ ]
36
+ if inconsistent_widths:
37
+ diagnostics.append(
38
+ Diagnostic(
39
+ code="feature_width_mismatch",
40
+ severity="error",
41
+ message="Feature rows do not have a consistent width.",
42
+ details={
43
+ "expected": expected_width,
44
+ "mismatches": inconsistent_widths[:5],
45
+ },
46
+ )
47
+ )
48
+ for row_index, row in enumerate(features):
49
+ for column_index, value in enumerate(row):
50
+ if not isinstance(value, (int, float)) or not is_finite(value):
51
+ diagnostics.append(
52
+ Diagnostic(
53
+ code="non_finite_feature",
54
+ severity="error",
55
+ message="A feature is not finite.",
56
+ details={
57
+ "row": row_index,
58
+ "column": column_index,
59
+ "value": value,
60
+ },
61
+ )
62
+ )
63
+ for label_index, label in enumerate(labels):
64
+ if not isinstance(label, (int, float)) or not is_finite(label):
65
+ diagnostics.append(
66
+ Diagnostic(
67
+ code="non_finite_label",
68
+ severity="error",
69
+ message="A label is not finite.",
70
+ details={"index": label_index, "value": label},
71
+ )
72
+ )
73
+ return DiagnosticReport(
74
+ valid=not any(diagnostic.severity == "error" for diagnostic in diagnostics),
75
+ diagnostics=diagnostics,
76
+ )
@@ -0,0 +1,36 @@
1
+ from typing import Dict
2
+
3
+ from .models import Diagnostic, DiagnosticReport
4
+ from .utils import is_finite
5
+
6
+
7
+ def diagnose_metrics(metrics: Dict[str, float]) -> DiagnosticReport:
8
+ diagnostics = []
9
+ for name, value in metrics.items():
10
+ if not isinstance(value, (int, float)) or not is_finite(value):
11
+ diagnostics.append(
12
+ Diagnostic(
13
+ code="non_finite_metric",
14
+ severity="error",
15
+ message=f"{name} is not finite.",
16
+ details={"name": name, "value": value},
17
+ )
18
+ )
19
+ if (
20
+ "gradient" in name.lower()
21
+ and isinstance(value, (int, float))
22
+ and is_finite(value)
23
+ and abs(value) > 1e4
24
+ ):
25
+ diagnostics.append(
26
+ Diagnostic(
27
+ code="exploding_gradient",
28
+ severity="warning",
29
+ message=f"{name} is unusually large and may indicate exploding gradients.",
30
+ details={"name": name, "value": value},
31
+ )
32
+ )
33
+ return DiagnosticReport(
34
+ valid=not any(diagnostic.severity == "error" for diagnostic in diagnostics),
35
+ diagnostics=diagnostics,
36
+ )
@@ -0,0 +1,16 @@
1
+ from dataclasses import dataclass
2
+ from typing import Any, Dict, List, Literal
3
+
4
+
5
+ @dataclass(frozen=True)
6
+ class Diagnostic:
7
+ code: str
8
+ severity: Literal["error", "warning", "info"]
9
+ message: str
10
+ details: Dict[str, Any]
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class DiagnosticReport:
15
+ valid: bool
16
+ diagnostics: List[Diagnostic]
@@ -0,0 +1,8 @@
1
+ from typing import Any
2
+
3
+
4
+ def is_finite(value: Any) -> bool:
5
+ try:
6
+ return value == value and abs(value) != float("inf")
7
+ except (TypeError, ValueError):
8
+ return False
@@ -0,0 +1,294 @@
1
+ Metadata-Version: 2.4
2
+ Name: silver-diagnostics
3
+ Version: 0.1.0
4
+ Summary: Framework-neutral ML data and training diagnostics for Silver.
5
+ License-Expression: Apache-2.0
6
+ Project-URL: Homepage, https://github.com/adfgdartec/silver-diagnostics
7
+ Project-URL: Repository, https://github.com/adfgdartec/silver-diagnostics
8
+ Project-URL: Issues, https://github.com/adfgdartec/silver-diagnostics/issues
9
+ Keywords: machine-learning,diagnostics,debugging,data-quality,python
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.8
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Requires-Python: >=3.8
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Provides-Extra: dev
24
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
25
+ Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
26
+ Requires-Dist: flake8>=6.0.0; extra == "dev"
27
+ Requires-Dist: mypy>=1.0.0; extra == "dev"
28
+ Requires-Dist: build>=0.10.0; extra == "dev"
29
+ Requires-Dist: twine>=4.0.0; extra == "dev"
30
+ Dynamic: license-file
31
+
32
+ # silver-diagnostics
33
+
34
+ [![Python Version](https://img.shields.io/badge/python-3.8%2B-blue.svg)](https://www.python.org/downloads/)
35
+ [![License](https://img.shields.io/badge/license-Apache%202.0-green.svg)](LICENSE)
36
+ [![Tests](https://img.shields.io/badge/tests-passing-brightgreen.svg)](tests/)
37
+ [![Code Style](https://img.shields.io/badge/code%20style-flake8-blue.svg)](https://flake8.pycqa.org/)
38
+
39
+ Framework-neutral ML data and training diagnostics for Silver. A Python package designed for ML researchers who need robust data validation and training stability checks across different frameworks.
40
+
41
+ ## Installation
42
+
43
+ ```bash
44
+ pip install silver-diagnostics
45
+ ```
46
+
47
+ ## Quick Start
48
+
49
+ ```python
50
+ from silver_diagnostics import diagnose_dataset, diagnose_metrics
51
+
52
+ # Diagnose dataset issues
53
+ features = [[1.0, 2.0], [3.0, 4.0], [5.0, 6.0]]
54
+ labels = [0.0, 1.0, 0.0]
55
+
56
+ report = diagnose_dataset(features, labels)
57
+ if not report.valid:
58
+ print("Dataset issues found:")
59
+ for diagnostic in report.diagnostics:
60
+ print(f" [{diagnostic.severity}] {diagnostic.message}")
61
+
62
+ # Diagnose metrics issues
63
+ metrics = {"loss": 0.5, "accuracy": 0.9, "gradient_norm": 1e6}
64
+ report = diagnose_metrics(metrics)
65
+ for diagnostic in report.diagnostics:
66
+ print(f"[{diagnostic.severity}] {diagnostic.message}")
67
+ ```
68
+
69
+ ## Features
70
+
71
+ - **Dataset Validation**: Comprehensive checks for empty data, length mismatches, and structural issues
72
+ - **Non-Finite Detection**: Automatic detection of NaN and infinity values in features and labels
73
+ - **Metrics Diagnostics**: Training metrics validation for numerical stability
74
+ - **Exploding Gradient Detection**: Specialized checks for gradient explosion during training
75
+ - **Framework-Agnostic**: Works with PyTorch, TensorFlow, JAX, or any numeric data
76
+ - **Detailed Reporting**: Structured diagnostic information with severity levels and context
77
+ - **Type Safety**: Full type hints for better IDE support and fewer bugs
78
+
79
+ ## Use Cases
80
+
81
+ ### Training Pipeline Validation
82
+
83
+ ```python
84
+ from silver_diagnostics import diagnose_dataset, diagnose_metrics
85
+ import torch
86
+
87
+ # Validate training data before training
88
+ train_features = torch.randn(1000, 10).numpy()
89
+ train_labels = torch.randint(0, 2, (1000,)).numpy()
90
+
91
+ report = diagnose_dataset(train_features.tolist(), train_labels.tolist())
92
+ if not report.valid:
93
+ print("Cannot train with invalid dataset:")
94
+ for diagnostic in report.diagnostics:
95
+ print(f" {diagnostic.code}: {diagnostic.message}")
96
+ else:
97
+ print("Dataset is valid for training")
98
+ ```
99
+
100
+ ### Training Stability Monitoring
101
+
102
+ ```python
103
+ from silver_diagnostics import diagnose_metrics
104
+
105
+ # Monitor training metrics for stability
106
+ def check_training_stability(metrics):
107
+ report = diagnose_metrics(metrics)
108
+
109
+ # Check for errors
110
+ errors = [d for d in report.diagnostics if d.severity == "error"]
111
+ if errors:
112
+ print("Training stability issues:")
113
+ for error in errors:
114
+ print(f" {error.code}: {error.message}")
115
+ return False
116
+
117
+ # Check for warnings
118
+ warnings = [d for d in report.diagnostics if d.severity == "warning"]
119
+ if warnings:
120
+ print("Training stability warnings:")
121
+ for warning in warnings:
122
+ print(f" {warning.code}: {warning.message}")
123
+
124
+ return True
125
+
126
+ # During training loop
127
+ for epoch in range(10):
128
+ loss = train_epoch()
129
+ metrics = {
130
+ "loss": loss,
131
+ "gradient_norm": compute_gradient_norm(),
132
+ "accuracy": evaluate()
133
+ }
134
+
135
+ if not check_training_stability(metrics):
136
+ print("Training unstable - stopping")
137
+ break
138
+ ```
139
+
140
+ ### Data Quality Assurance
141
+
142
+ ```python
143
+ from silver_diagnostics import diagnose_dataset
144
+
145
+ def validate_ml_pipeline_data(X_train, y_train, X_val, y_val):
146
+ """Validate all datasets in ML pipeline"""
147
+ datasets = {
148
+ "training": (X_train, y_train),
149
+ "validation": (X_val, y_val)
150
+ }
151
+
152
+ all_valid = True
153
+ for name, (features, labels) in datasets.items():
154
+ report = diagnose_dataset(features.tolist(), labels.tolist())
155
+
156
+ print(f"\n{name} dataset:")
157
+ if report.valid:
158
+ print(f" ✓ Valid ({len(features)} samples)")
159
+ else:
160
+ print(f" ✗ Invalid")
161
+ for diagnostic in report.diagnostics:
162
+ print(f" {diagnostic.message}")
163
+ all_valid = False
164
+
165
+ return all_valid
166
+ ```
167
+
168
+ ### Framework Integration
169
+
170
+ ```python
171
+ from silver_diagnostics import diagnose_dataset, diagnose_metrics
172
+ import tensorflow as tf
173
+ import torch
174
+
175
+ # Works with TensorFlow tensors
176
+ tf_features = tf.random.normal((100, 10))
177
+ tf_labels = tf.random.uniform((100,), maxval=2, dtype=tf.int32)
178
+
179
+ report = diagnose_dataset(
180
+ tf_features.numpy().tolist(),
181
+ tf_labels.numpy().tolist()
182
+ )
183
+
184
+ # Works with PyTorch tensors
185
+ torch_features = torch.randn(100, 10)
186
+ torch_labels = torch.randint(0, 2, (100,))
187
+
188
+ report = diagnose_dataset(
189
+ torch_features.tolist(),
190
+ torch_labels.tolist()
191
+ )
192
+ ```
193
+
194
+ ## Advanced Usage
195
+
196
+ ### Custom Diagnostic Processing
197
+
198
+ ```python
199
+ from silver_diagnostics import diagnose_dataset, Diagnostic
200
+
201
+ def categorize_diagnostics(report):
202
+ """Categorize diagnostics by type"""
203
+ categories = {
204
+ "structural": [],
205
+ "data_quality": [],
206
+ "numerical": []
207
+ }
208
+
209
+ for diagnostic in report.diagnostics:
210
+ if diagnostic.code in ["empty_dataset", "length_mismatch", "feature_width_mismatch"]:
211
+ categories["structural"].append(diagnostic)
212
+ elif diagnostic.code in ["non_finite_feature", "non_finite_label"]:
213
+ categories["numerical"].append(diagnostic)
214
+ else:
215
+ categories["data_quality"].append(diagnostic)
216
+
217
+ return categories
218
+
219
+ report = diagnose_dataset(features, labels)
220
+ categories = categorize_diagnostics(report)
221
+
222
+ for category, diagnostics in categories.items():
223
+ if diagnostics:
224
+ print(f"{category.upper()} ({len(diagnostics)}):")
225
+ for diag in diagnostics:
226
+ print(f" - {diag.message}")
227
+ ```
228
+
229
+ ### Batch Validation
230
+
231
+ ```python
232
+ from silver_diagnostics import diagnose_dataset
233
+
234
+ def validate_multiple_datasets(dataset_dict):
235
+ """Validate multiple datasets at once"""
236
+ results = {}
237
+
238
+ for name, (features, labels) in dataset_dict.items():
239
+ report = diagnose_dataset(features, labels)
240
+ results[name] = {
241
+ "valid": report.valid,
242
+ "error_count": sum(1 for d in report.diagnostics if d.severity == "error"),
243
+ "warning_count": sum(1 for d in report.diagnostics if d.severity == "warning"),
244
+ "diagnostics": report.diagnostics
245
+ }
246
+
247
+ return results
248
+
249
+ datasets = {
250
+ "train": (X_train.tolist(), y_train.tolist()),
251
+ "val": (X_val.tolist(), y_val.tolist()),
252
+ "test": (X_test.tolist(), y_test.tolist())
253
+ }
254
+
255
+ validation_results = validate_multiple_datasets(datasets)
256
+ for name, result in validation_results.items():
257
+ status = "✓" if result["valid"] else "✗"
258
+ print(f"{status} {name}: {result['error_count']} errors, {result['warning_count']} warnings")
259
+ ```
260
+
261
+ ## Requirements
262
+
263
+ - Python 3.8+
264
+
265
+ ## Development
266
+
267
+ ```bash
268
+ # Install development dependencies
269
+ pip install -e ".[dev]"
270
+
271
+ # Run tests
272
+ pytest
273
+
274
+ # Run tests with coverage
275
+ pytest --cov=silver_diagnostics --cov-report=html
276
+
277
+ # Run linting
278
+ flake8 src/ tests/
279
+ mypy src/
280
+ ```
281
+
282
+ ## Contributing
283
+
284
+ Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines.
285
+
286
+ ## License
287
+
288
+ Apache-2.0 - see [LICENSE](LICENSE) file for details.
289
+
290
+ ## Related Packages
291
+
292
+ - [silver-data](https://github.com/adfgdartec/silver-data) - Dataset handling
293
+ - [silver-run](https://github.com/adfgdartec/silver-run) - Training lifecycle
294
+ - [silver-adapters](https://github.com/adfgdartec/silver-adapters) - Framework adapters
@@ -0,0 +1,10 @@
1
+ silver_diagnostics/__init__.py,sha256=IpfWo06ghsPkb3y-wwrxGDfi2H16wSJ13NO1DPvmi5Q,211
2
+ silver_diagnostics/dataset.py,sha256=hFeE4cqoz4jJ5BQNdEUqFhguBmyPuwjoE1lIgH6icZU,2812
3
+ silver_diagnostics/metrics.py,sha256=gwisKFEkU_YOjCwLyoXEd1GFvd5IvKm3nSP5teGQ8PQ,1262
4
+ silver_diagnostics/models.py,sha256=dPUzlOrYik26Aa32u4u8PyPmAySclBVf-A_kJsmR2N0,331
5
+ silver_diagnostics/utils.py,sha256=moXFkGpck4rhkc8-VRWJc4KYGRTP1pSd1kqdz9x0njY,187
6
+ silver_diagnostics-0.1.0.dist-info/licenses/LICENSE,sha256=9uEgZddcCAZ2jYIKny4rlbkQMZFhKRvSGbxctAg4wjI,517
7
+ silver_diagnostics-0.1.0.dist-info/METADATA,sha256=CA-ki2uzHB_S875rXHfw30F20Htys-nSdSaMeuvBrk8,9071
8
+ silver_diagnostics-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
9
+ silver_diagnostics-0.1.0.dist-info/top_level.txt,sha256=GAdKKkhCBVR1rBV_Hv-Y0c7hfrXJR13C1tKCRcgr8Ng,19
10
+ silver_diagnostics-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,12 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+
4
+ Copyright 2026 Silver Contributors
5
+
6
+ Licensed under the Apache License, Version 2.0. You may obtain a copy of the
7
+ License at https://www.apache.org/licenses/LICENSE-2.0
8
+
9
+ Unless required by applicable law or agreed to in writing, software distributed
10
+ under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
11
+ CONDITIONS OF ANY KIND, either express or implied. See the License for the
12
+ specific language governing permissions and limitations under the License.
@@ -0,0 +1 @@
1
+ silver_diagnostics