lims-data-quality 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lims_data_quality-0.1.0.dist-info/METADATA +132 -0
- lims_data_quality-0.1.0.dist-info/RECORD +12 -0
- lims_data_quality-0.1.0.dist-info/WHEEL +5 -0
- lims_data_quality-0.1.0.dist-info/entry_points.txt +2 -0
- lims_data_quality-0.1.0.dist-info/licenses/LICENSE +21 -0
- lims_data_quality-0.1.0.dist-info/top_level.txt +1 -0
- lims_dq/__init__.py +21 -0
- lims_dq/audit.py +88 -0
- lims_dq/cli.py +131 -0
- lims_dq/report.py +88 -0
- lims_dq/schema.py +135 -0
- lims_dq/validators.py +145 -0
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: lims-data-quality
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Validate lab/LIMS data files before they hit your pipeline
|
|
5
|
+
Author: Sri Gorantla
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/sriranga13/lims-data-quality
|
|
8
|
+
Keywords: lims,lab,data-quality,validation
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Requires-Dist: pandas>=2.0
|
|
17
|
+
Requires-Dist: openpyxl>=3.1
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# lims-data-quality
|
|
21
|
+
|
|
22
|
+
Validate lab/LIMS data files before they hit your pipeline — schema checks, row-level error reports, audit trails.
|
|
23
|
+
|
|
24
|
+
`lims-dq` checks CSV/Excel exports against a JSON schema (required columns, dtypes, ranges, regex ID patterns, allowed values) and tells you exactly which row, which column, and why it failed. Every run can append a tamper-evident audit entry (timestamp, file hash, rule set version, pass/fail counts) — a nod to 21 CFR Part 11 style traceability.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install lims-data-quality
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Or from source:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
git clone https://github.com/sriranga13/lims-data-quality.git
|
|
36
|
+
cd lims-data-quality
|
|
37
|
+
pip install -e .
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Requires Python 3.10+.
|
|
41
|
+
|
|
42
|
+
## Quickstart
|
|
43
|
+
|
|
44
|
+
1. Describe your file with a schema (`schema.json`):
|
|
45
|
+
|
|
46
|
+
```json
|
|
47
|
+
{
|
|
48
|
+
"name": "lims-export",
|
|
49
|
+
"version": "1.0.0",
|
|
50
|
+
"columns": {
|
|
51
|
+
"sample_id": { "dtype": "string", "required": true, "pattern": "^SMP-[0-9]{6}$" },
|
|
52
|
+
"concentration": { "dtype": "float", "required": true, "min": 0.0, "max": 100.0 },
|
|
53
|
+
"unit": { "dtype": "string", "required": true, "allowed": ["mg/L", "ug/mL", "ng/uL"] },
|
|
54
|
+
"analyzed_at": { "dtype": "date", "required": false },
|
|
55
|
+
"replicates": { "dtype": "int", "required": false, "min": 1, "max": 12 }
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
2. Validate:
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
lims-dq validate samples.csv --schema schema.json
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Sample output:
|
|
67
|
+
|
|
68
|
+
```
|
|
69
|
+
lims-dq report: samples.csv
|
|
70
|
+
ruleset: lims-export v1.0.0
|
|
71
|
+
rows checked: 5 | passed: 3 | failed: 2 | errors: 7
|
|
72
|
+
|
|
73
|
+
row column rule message
|
|
74
|
+
------------------------------------------------------------------------
|
|
75
|
+
4 analyzed_at dtype 'not-a-date' is not a recognizable date
|
|
76
|
+
4 concentration constraint '150.0' is above maximum 100.0
|
|
77
|
+
4 replicates constraint '0' is below minimum 1
|
|
78
|
+
4 sample_id constraint 'BAD-ID' does not match pattern '^SMP-[0-9]{6}$'
|
|
79
|
+
4 unit constraint 'kg' is not one of ['mg/L', 'ug/mL', 'ng/uL']
|
|
80
|
+
5 concentration required required value is missing
|
|
81
|
+
5 replicates constraint '13' is above maximum 12
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
3. Machine-readable report and audit trail:
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
lims-dq validate samples.csv --schema schema.json \
|
|
88
|
+
--report report.json \
|
|
89
|
+
--audit-log audit.jsonl
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
- `--report` writes the full report (summary counts + every error) as JSON.
|
|
93
|
+
- `--audit-log` appends one JSON-lines entry per run: UTC timestamp, SHA-256 of the input file, rule set name/version, rows checked/passed/failed, error count. Re-running against a modified file produces a different hash, so entries are tamper-evident.
|
|
94
|
+
|
|
95
|
+
Exit codes: `0` = all rows pass, `1` = validation failures, `2` = usage/file errors.
|
|
96
|
+
|
|
97
|
+
## Schema reference
|
|
98
|
+
|
|
99
|
+
| Key | Applies to | Meaning |
|
|
100
|
+
|------------|-------------------|--------------------------------------------------|
|
|
101
|
+
| `dtype` | all | `string`, `int`, `float`, or `date` |
|
|
102
|
+
| `required` | all | missing values fail (missing column fails too) |
|
|
103
|
+
| `pattern` | all | regex the full value must match |
|
|
104
|
+
| `min`/`max`| `int`, `float` | inclusive numeric bounds |
|
|
105
|
+
| `allowed` | all | value must be one of the listed strings |
|
|
106
|
+
|
|
107
|
+
Columns present in the file but absent from the schema are flagged as `unexpected_column` errors unless you pass `--no-strict`.
|
|
108
|
+
|
|
109
|
+
## Use as a library
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
from lims_dq import load_schema, validate_dataframe, ErrorReport
|
|
113
|
+
|
|
114
|
+
ruleset = load_schema("schema.json")
|
|
115
|
+
import pandas as pd
|
|
116
|
+
df = pd.read_csv("samples.csv", dtype=str)
|
|
117
|
+
errors = validate_dataframe(df, ruleset)
|
|
118
|
+
report = ErrorReport("samples.csv", ruleset.name, ruleset.version, len(df), errors)
|
|
119
|
+
print(report.to_text()) # or report.to_json()
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Development
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
python -m venv .venv && source .venv/bin/activate
|
|
126
|
+
pip install -e . && pip install pytest
|
|
127
|
+
pytest
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
## License
|
|
131
|
+
|
|
132
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
lims_data_quality-0.1.0.dist-info/licenses/LICENSE,sha256=4J9t3O8u6I-R7Iit0ZN6GP_3B7FiCHqA0EjHF9psjZQ,1069
|
|
2
|
+
lims_dq/__init__.py,sha256=UHE7KQZYlGG8vMIriyuiM661EGphK0sgwTxRoVn2qF8,527
|
|
3
|
+
lims_dq/audit.py,sha256=JhXekTK8HnqWBiYFg19WDORpiFy5fj7l6rJnkMpEt2Q,2533
|
|
4
|
+
lims_dq/cli.py,sha256=oC-3ykihNwTulhoUNjC0ONz5YIv7SA9MjB1aw-C0xC0,3993
|
|
5
|
+
lims_dq/report.py,sha256=hEl9sVuQX5FzMMx5WnkJ9Z9wpayoj72oRkH9mk0vPv0,2704
|
|
6
|
+
lims_dq/schema.py,sha256=p6t31egUJ7_qQmCsAxX6g2erVL-3nf2HhHXIA7SSwqk,4616
|
|
7
|
+
lims_dq/validators.py,sha256=ghP9zMjvk94Fijlqlh8EH9HNAThgjh4usCx5ARwuUgg,5284
|
|
8
|
+
lims_data_quality-0.1.0.dist-info/METADATA,sha256=3yQVNwyTDEdA7U-YMOmszW3epWZUB1X4Dh2WO0u34PE,4683
|
|
9
|
+
lims_data_quality-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
10
|
+
lims_data_quality-0.1.0.dist-info/entry_points.txt,sha256=Eje_pmk5RNACBYdtHsq1SsiMI0GIBEoST8K2OrJKMGQ,45
|
|
11
|
+
lims_data_quality-0.1.0.dist-info/top_level.txt,sha256=MGHywZMUvDpSpvKF21_nI5ya8sBkrjznRwwaJFaf9Oc,8
|
|
12
|
+
lims_data_quality-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sri Gorantla
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
lims_dq
|
lims_dq/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""lims-dq: validate lab/LIMS data files before they enter your pipeline."""
|
|
2
|
+
|
|
3
|
+
from .audit import AuditEntry, read_audit_log, write_audit_log
|
|
4
|
+
from .report import ErrorReport, ValidationError
|
|
5
|
+
from .schema import Rule, RuleSet, SchemaError, load_schema
|
|
6
|
+
from .validators import validate_dataframe
|
|
7
|
+
|
|
8
|
+
__version__ = "0.1.0"
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"AuditEntry",
|
|
12
|
+
"ErrorReport",
|
|
13
|
+
"Rule",
|
|
14
|
+
"RuleSet",
|
|
15
|
+
"SchemaError",
|
|
16
|
+
"ValidationError",
|
|
17
|
+
"load_schema",
|
|
18
|
+
"read_audit_log",
|
|
19
|
+
"validate_dataframe",
|
|
20
|
+
"write_audit_log",
|
|
21
|
+
]
|
lims_dq/audit.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Minimal audit trail: who validated what, when, against which rules.
|
|
2
|
+
|
|
3
|
+
Each validation run appends one JSON-lines entry recording the UTC
|
|
4
|
+
timestamp, the SHA-256 hash of the input file, the rule set name and
|
|
5
|
+
version, and the pass/fail counts. The file hash makes the entry
|
|
6
|
+
tamper-evident: re-running against a modified file produces a different
|
|
7
|
+
hash. A nod to 21 CFR Part 11 style traceability — not a compliance
|
|
8
|
+
certification.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
from dataclasses import asdict, dataclass
|
|
16
|
+
from datetime import datetime, timezone
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class AuditEntry:
|
|
23
|
+
timestamp_utc: str
|
|
24
|
+
filename: str
|
|
25
|
+
file_sha256: str
|
|
26
|
+
ruleset_name: str
|
|
27
|
+
ruleset_version: str
|
|
28
|
+
rows_checked: int
|
|
29
|
+
rows_passed: int
|
|
30
|
+
rows_failed: int
|
|
31
|
+
error_count: int
|
|
32
|
+
passed: bool
|
|
33
|
+
|
|
34
|
+
def to_dict(self) -> dict[str, Any]:
|
|
35
|
+
return asdict(self)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def sha256_of_file(path: str | Path) -> str:
|
|
39
|
+
digest = hashlib.sha256()
|
|
40
|
+
with open(path, "rb") as fh:
|
|
41
|
+
for chunk in iter(lambda: fh.read(65536), b""):
|
|
42
|
+
digest.update(chunk)
|
|
43
|
+
return digest.hexdigest()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def build_entry(
|
|
47
|
+
*,
|
|
48
|
+
filename: str | Path,
|
|
49
|
+
ruleset_name: str,
|
|
50
|
+
ruleset_version: str,
|
|
51
|
+
rows_checked: int,
|
|
52
|
+
rows_passed: int,
|
|
53
|
+
rows_failed: int,
|
|
54
|
+
error_count: int,
|
|
55
|
+
) -> AuditEntry:
|
|
56
|
+
path = Path(filename)
|
|
57
|
+
return AuditEntry(
|
|
58
|
+
timestamp_utc=datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
59
|
+
filename=path.name,
|
|
60
|
+
file_sha256=sha256_of_file(path),
|
|
61
|
+
ruleset_name=ruleset_name,
|
|
62
|
+
ruleset_version=ruleset_version,
|
|
63
|
+
rows_checked=rows_checked,
|
|
64
|
+
rows_passed=rows_passed,
|
|
65
|
+
rows_failed=rows_failed,
|
|
66
|
+
error_count=error_count,
|
|
67
|
+
passed=error_count == 0,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def write_audit_log(entry: AuditEntry, log_path: str | Path) -> None:
|
|
72
|
+
"""Append one entry (as a single JSON line) to the audit log."""
|
|
73
|
+
log_path = Path(log_path)
|
|
74
|
+
if log_path.parent != Path("."):
|
|
75
|
+
log_path.parent.mkdir(parents=True, exist_ok=True)
|
|
76
|
+
with open(log_path, "a", encoding="utf-8") as fh:
|
|
77
|
+
fh.write(json.dumps(entry.to_dict()) + "\n")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def read_audit_log(log_path: str | Path) -> list[dict[str, Any]]:
|
|
81
|
+
"""Read all entries from a JSON-lines audit log."""
|
|
82
|
+
entries = []
|
|
83
|
+
with open(log_path, encoding="utf-8") as fh:
|
|
84
|
+
for line in fh:
|
|
85
|
+
line = line.strip()
|
|
86
|
+
if line:
|
|
87
|
+
entries.append(json.loads(line))
|
|
88
|
+
return entries
|
lims_dq/cli.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Command-line interface: ``lims-dq validate data.csv --schema schema.json``.
|
|
2
|
+
|
|
3
|
+
Exit codes: 0 = all rows pass, 1 = validation failures found,
|
|
4
|
+
2 = usage or file errors (missing file, bad schema, unreadable data).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
|
|
15
|
+
from . import __version__
|
|
16
|
+
from .audit import build_entry, write_audit_log
|
|
17
|
+
from .report import ErrorReport
|
|
18
|
+
from .schema import SchemaError, load_schema
|
|
19
|
+
from .validators import validate_dataframe
|
|
20
|
+
|
|
21
|
+
SUPPORTED_SUFFIXES = {".csv", ".xlsx", ".xls"}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def load_data_file(path: Path) -> pd.DataFrame:
|
|
25
|
+
suffix = path.suffix.lower()
|
|
26
|
+
if suffix == ".csv":
|
|
27
|
+
return pd.read_csv(path, dtype=str)
|
|
28
|
+
if suffix in (".xlsx", ".xls"):
|
|
29
|
+
return pd.read_excel(path, dtype=str)
|
|
30
|
+
raise ValueError(
|
|
31
|
+
f"unsupported file type {path.suffix!r} "
|
|
32
|
+
f"(expected one of {sorted(SUPPORTED_SUFFIXES)})"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
37
|
+
parser = argparse.ArgumentParser(
|
|
38
|
+
prog="lims-dq",
|
|
39
|
+
description="Validate lab/LIMS data files against a schema before they "
|
|
40
|
+
"enter your pipeline.",
|
|
41
|
+
)
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
"--version", action="version", version=f"%(prog)s {__version__}"
|
|
44
|
+
)
|
|
45
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
46
|
+
|
|
47
|
+
validate = sub.add_parser(
|
|
48
|
+
"validate", help="Validate a data file against a schema."
|
|
49
|
+
)
|
|
50
|
+
validate.add_argument("data_file", help="CSV or Excel file to validate")
|
|
51
|
+
validate.add_argument(
|
|
52
|
+
"--schema", required=True, help="Path to the schema JSON file"
|
|
53
|
+
)
|
|
54
|
+
validate.add_argument(
|
|
55
|
+
"--report",
|
|
56
|
+
default=None,
|
|
57
|
+
help="Write the full error report as JSON to this path",
|
|
58
|
+
)
|
|
59
|
+
validate.add_argument(
|
|
60
|
+
"--audit-log",
|
|
61
|
+
default=None,
|
|
62
|
+
help="Append a JSON-lines audit entry to this log file",
|
|
63
|
+
)
|
|
64
|
+
validate.add_argument(
|
|
65
|
+
"--no-strict",
|
|
66
|
+
action="store_true",
|
|
67
|
+
help="Do not flag columns missing from the schema",
|
|
68
|
+
)
|
|
69
|
+
return parser
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def cmd_validate(args: argparse.Namespace) -> int:
|
|
73
|
+
data_path = Path(args.data_file)
|
|
74
|
+
if not data_path.is_file():
|
|
75
|
+
print(f"error: data file not found: {data_path}", file=sys.stderr)
|
|
76
|
+
return 2
|
|
77
|
+
try:
|
|
78
|
+
ruleset = load_schema(args.schema)
|
|
79
|
+
except SchemaError as exc:
|
|
80
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
81
|
+
return 2
|
|
82
|
+
try:
|
|
83
|
+
df = load_data_file(data_path)
|
|
84
|
+
except ValueError as exc:
|
|
85
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
86
|
+
return 2
|
|
87
|
+
except Exception as exc: # unreadable file, bad encoding, ...
|
|
88
|
+
print(f"error: could not read {data_path}: {exc}", file=sys.stderr)
|
|
89
|
+
return 2
|
|
90
|
+
|
|
91
|
+
errors = validate_dataframe(df, ruleset, strict=not args.no_strict)
|
|
92
|
+
report = ErrorReport(
|
|
93
|
+
filename=data_path.name,
|
|
94
|
+
ruleset_name=ruleset.name,
|
|
95
|
+
ruleset_version=ruleset.version,
|
|
96
|
+
rows_checked=len(df),
|
|
97
|
+
errors=errors,
|
|
98
|
+
)
|
|
99
|
+
print(report.to_text())
|
|
100
|
+
|
|
101
|
+
if args.report:
|
|
102
|
+
Path(args.report).write_text(report.to_json() + "\n", encoding="utf-8")
|
|
103
|
+
print(f"\nJSON report written to {args.report}")
|
|
104
|
+
|
|
105
|
+
if args.audit_log:
|
|
106
|
+
entry = build_entry(
|
|
107
|
+
filename=data_path,
|
|
108
|
+
ruleset_name=ruleset.name,
|
|
109
|
+
ruleset_version=ruleset.version,
|
|
110
|
+
rows_checked=report.rows_checked,
|
|
111
|
+
rows_passed=report.rows_passed,
|
|
112
|
+
rows_failed=report.rows_failed,
|
|
113
|
+
error_count=len(errors),
|
|
114
|
+
)
|
|
115
|
+
write_audit_log(entry, args.audit_log)
|
|
116
|
+
print(f"audit entry appended to {args.audit_log}")
|
|
117
|
+
|
|
118
|
+
return 0 if report.passed else 1
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def main(argv: list[str] | None = None) -> int:
|
|
122
|
+
parser = build_parser()
|
|
123
|
+
args = parser.parse_args(argv)
|
|
124
|
+
if args.command == "validate":
|
|
125
|
+
return cmd_validate(args)
|
|
126
|
+
parser.error(f"unknown command {args.command!r}")
|
|
127
|
+
return 2 # unreachable
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
if __name__ == "__main__":
|
|
131
|
+
sys.exit(main())
|
lims_dq/report.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Row-level error reports: console text and JSON."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from dataclasses import asdict, dataclass
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class ValidationError:
|
|
13
|
+
"""A single failed check.
|
|
14
|
+
|
|
15
|
+
``row`` is the 1-based data-row number (row 1 = first row under the
|
|
16
|
+
header); ``row=0`` means the problem is file-level (e.g. a missing
|
|
17
|
+
column) rather than tied to a data row.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
row: int
|
|
21
|
+
column: str
|
|
22
|
+
value: Any
|
|
23
|
+
rule: str
|
|
24
|
+
message: str
|
|
25
|
+
|
|
26
|
+
def to_dict(self) -> dict[str, Any]:
|
|
27
|
+
return asdict(self)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class ErrorReport:
|
|
32
|
+
"""Aggregated result of validating one file."""
|
|
33
|
+
|
|
34
|
+
filename: str
|
|
35
|
+
ruleset_name: str
|
|
36
|
+
ruleset_version: str
|
|
37
|
+
rows_checked: int
|
|
38
|
+
errors: list[ValidationError]
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def passed(self) -> bool:
|
|
42
|
+
return not self.errors
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def rows_failed(self) -> int:
|
|
46
|
+
return len({e.row for e in self.errors if e.row > 0})
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def rows_passed(self) -> int:
|
|
50
|
+
return self.rows_checked - self.rows_failed
|
|
51
|
+
|
|
52
|
+
def to_dict(self) -> dict[str, Any]:
|
|
53
|
+
return {
|
|
54
|
+
"filename": self.filename,
|
|
55
|
+
"ruleset": {"name": self.ruleset_name, "version": self.ruleset_version},
|
|
56
|
+
"summary": {
|
|
57
|
+
"passed": self.passed,
|
|
58
|
+
"rows_checked": self.rows_checked,
|
|
59
|
+
"rows_passed": self.rows_passed,
|
|
60
|
+
"rows_failed": self.rows_failed,
|
|
61
|
+
"error_count": len(self.errors),
|
|
62
|
+
"errors_by_rule": dict(Counter(e.rule for e in self.errors)),
|
|
63
|
+
"errors_by_column": dict(Counter(e.column for e in self.errors)),
|
|
64
|
+
},
|
|
65
|
+
"errors": [e.to_dict() for e in self.errors],
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
69
|
+
return json.dumps(self.to_dict(), indent=indent, default=str)
|
|
70
|
+
|
|
71
|
+
def to_text(self) -> str:
|
|
72
|
+
lines = [
|
|
73
|
+
f"lims-dq report: {self.filename}",
|
|
74
|
+
f"ruleset: {self.ruleset_name} v{self.ruleset_version}",
|
|
75
|
+
f"rows checked: {self.rows_checked} | "
|
|
76
|
+
f"passed: {self.rows_passed} | failed: {self.rows_failed} | "
|
|
77
|
+
f"errors: {len(self.errors)}",
|
|
78
|
+
]
|
|
79
|
+
if self.passed:
|
|
80
|
+
lines.append("PASS: all rows satisfy the schema.")
|
|
81
|
+
return "\n".join(lines)
|
|
82
|
+
lines.append("")
|
|
83
|
+
lines.append(f"{'row':>5} {'column':<18} {'rule':<18} message")
|
|
84
|
+
lines.append("-" * 72)
|
|
85
|
+
for e in self.errors:
|
|
86
|
+
row = "file" if e.row == 0 else str(e.row)
|
|
87
|
+
lines.append(f"{row:>5} {e.column:<18} {e.rule:<18} {e.message}")
|
|
88
|
+
return "\n".join(lines)
|
lims_dq/schema.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Schema loading and validation for lims-dq.
|
|
2
|
+
|
|
3
|
+
A schema is a JSON document describing the columns a data file must have::
|
|
4
|
+
|
|
5
|
+
{
|
|
6
|
+
"name": "lims-export",
|
|
7
|
+
"version": "1.0.0",
|
|
8
|
+
"columns": {
|
|
9
|
+
"sample_id": {"dtype": "string", "required": true,
|
|
10
|
+
"pattern": "^SMP-[0-9]{6}$"},
|
|
11
|
+
"concentration": {"dtype": "float", "required": true,
|
|
12
|
+
"min": 0.0, "max": 100.0},
|
|
13
|
+
"unit": {"dtype": "string", "required": true,
|
|
14
|
+
"allowed": ["mg/L", "ug/mL", "ng/uL"]},
|
|
15
|
+
"analyzed_at": {"dtype": "date", "required": false}
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
Supported dtypes: ``string``, ``int``, ``float``, ``date``.
|
|
20
|
+
Supported constraints: ``required``, ``pattern`` (regex, full match),
|
|
21
|
+
``min``/``max`` (numeric dtypes only), ``allowed`` (enumerated values).
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import json
|
|
27
|
+
import re
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
from typing import Any
|
|
31
|
+
|
|
32
|
+
SUPPORTED_DTYPES = ("string", "int", "float", "date")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class SchemaError(ValueError):
|
|
36
|
+
"""Raised when a schema document is malformed."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class Rule:
|
|
41
|
+
"""Validation rule for a single column."""
|
|
42
|
+
|
|
43
|
+
name: str
|
|
44
|
+
dtype: str = "string"
|
|
45
|
+
required: bool = False
|
|
46
|
+
pattern: str | None = None
|
|
47
|
+
min: float | None = None
|
|
48
|
+
max: float | None = None
|
|
49
|
+
allowed: tuple[str, ...] | None = None
|
|
50
|
+
_regex: re.Pattern | None = field(default=None, repr=False, compare=False)
|
|
51
|
+
|
|
52
|
+
def compile(self) -> "Rule":
|
|
53
|
+
if self.pattern is not None:
|
|
54
|
+
try:
|
|
55
|
+
self._regex = re.compile(self.pattern)
|
|
56
|
+
except re.error as exc:
|
|
57
|
+
raise SchemaError(
|
|
58
|
+
f"column {self.name!r}: invalid regex {self.pattern!r}: {exc}"
|
|
59
|
+
) from exc
|
|
60
|
+
return self
|
|
61
|
+
|
|
62
|
+
def matches_pattern(self, value: str) -> bool:
|
|
63
|
+
assert self._regex is not None
|
|
64
|
+
return self._regex.fullmatch(value) is not None
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass
|
|
68
|
+
class RuleSet:
|
|
69
|
+
"""A named, versioned collection of column rules."""
|
|
70
|
+
|
|
71
|
+
name: str
|
|
72
|
+
version: str
|
|
73
|
+
rules: dict[str, Rule]
|
|
74
|
+
|
|
75
|
+
def required_columns(self) -> list[str]:
|
|
76
|
+
return [n for n, r in self.rules.items() if r.required]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _coerce_rule(name: str, spec: Any) -> Rule:
|
|
80
|
+
if not isinstance(spec, dict):
|
|
81
|
+
raise SchemaError(f"column {name!r}: rule must be an object, got {type(spec).__name__}")
|
|
82
|
+
dtype = spec.get("dtype", "string")
|
|
83
|
+
if dtype not in SUPPORTED_DTYPES:
|
|
84
|
+
raise SchemaError(
|
|
85
|
+
f"column {name!r}: unsupported dtype {dtype!r} "
|
|
86
|
+
f"(expected one of {SUPPORTED_DTYPES})"
|
|
87
|
+
)
|
|
88
|
+
unknown = set(spec) - {"dtype", "required", "pattern", "min", "max", "allowed"}
|
|
89
|
+
if unknown:
|
|
90
|
+
raise SchemaError(f"column {name!r}: unknown keys {sorted(unknown)}")
|
|
91
|
+
|
|
92
|
+
min_v = spec.get("min")
|
|
93
|
+
max_v = spec.get("max")
|
|
94
|
+
if (min_v is not None or max_v is not None) and dtype not in ("int", "float"):
|
|
95
|
+
raise SchemaError(f"column {name!r}: min/max only apply to int/float dtypes")
|
|
96
|
+
if min_v is not None and max_v is not None and min_v > max_v:
|
|
97
|
+
raise SchemaError(f"column {name!r}: min ({min_v}) > max ({max_v})")
|
|
98
|
+
|
|
99
|
+
allowed = spec.get("allowed")
|
|
100
|
+
if allowed is not None:
|
|
101
|
+
if not isinstance(allowed, list) or not allowed:
|
|
102
|
+
raise SchemaError(f"column {name!r}: 'allowed' must be a non-empty list")
|
|
103
|
+
allowed = tuple(str(v) for v in allowed)
|
|
104
|
+
|
|
105
|
+
return Rule(
|
|
106
|
+
name=name,
|
|
107
|
+
dtype=dtype,
|
|
108
|
+
required=bool(spec.get("required", False)),
|
|
109
|
+
pattern=spec.get("pattern"),
|
|
110
|
+
min=min_v,
|
|
111
|
+
max=max_v,
|
|
112
|
+
allowed=allowed,
|
|
113
|
+
).compile()
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def load_schema(path: str | Path) -> RuleSet:
|
|
117
|
+
"""Load and validate a schema JSON file, returning a RuleSet."""
|
|
118
|
+
path = Path(path)
|
|
119
|
+
try:
|
|
120
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
121
|
+
except FileNotFoundError:
|
|
122
|
+
raise SchemaError(f"schema file not found: {path}") from None
|
|
123
|
+
except json.JSONDecodeError as exc:
|
|
124
|
+
raise SchemaError(f"schema file {path} is not valid JSON: {exc}") from exc
|
|
125
|
+
if not isinstance(raw, dict):
|
|
126
|
+
raise SchemaError("schema must be a JSON object")
|
|
127
|
+
columns = raw.get("columns")
|
|
128
|
+
if not isinstance(columns, dict) or not columns:
|
|
129
|
+
raise SchemaError("schema must define a non-empty 'columns' object")
|
|
130
|
+
rules = {name: _coerce_rule(name, spec) for name, spec in columns.items()}
|
|
131
|
+
return RuleSet(
|
|
132
|
+
name=str(raw.get("name", path.stem)),
|
|
133
|
+
version=str(raw.get("version", "0.0.0")),
|
|
134
|
+
rules=rules,
|
|
135
|
+
)
|
lims_dq/validators.py
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Row-level validators: check a DataFrame against a RuleSet.
|
|
2
|
+
|
|
3
|
+
Every problem found becomes a :class:`ValidationError` carrying the
|
|
4
|
+
1-based data-row number (row 1 = first row beneath the header), the column
|
|
5
|
+
name, the offending value, the rule that failed, and a human message.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
|
|
14
|
+
from .schema import Rule, RuleSet
|
|
15
|
+
from .report import ValidationError
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _is_missing(value: Any) -> bool:
|
|
19
|
+
if value is None:
|
|
20
|
+
return True
|
|
21
|
+
if isinstance(value, float) and pd.isna(value):
|
|
22
|
+
return True
|
|
23
|
+
if isinstance(value, str) and value.strip() == "":
|
|
24
|
+
return True
|
|
25
|
+
return False
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _check_dtype(rule: Rule, value: Any) -> str | None:
|
|
29
|
+
"""Return an error message if value violates rule.dtype, else None."""
|
|
30
|
+
text = str(value).strip()
|
|
31
|
+
if rule.dtype == "string":
|
|
32
|
+
return None
|
|
33
|
+
if rule.dtype == "int":
|
|
34
|
+
try:
|
|
35
|
+
if not float(text).is_integer():
|
|
36
|
+
return f"{value!r} is not an integer"
|
|
37
|
+
except (ValueError, OverflowError):
|
|
38
|
+
return f"{value!r} is not an integer"
|
|
39
|
+
return None
|
|
40
|
+
if rule.dtype == "float":
|
|
41
|
+
try:
|
|
42
|
+
float(text)
|
|
43
|
+
except (ValueError, OverflowError):
|
|
44
|
+
return f"{value!r} is not a number"
|
|
45
|
+
return None
|
|
46
|
+
if rule.dtype == "date":
|
|
47
|
+
if pd.to_datetime(text, errors="coerce") is pd.NaT: # noqa: E711
|
|
48
|
+
return f"{value!r} is not a recognizable date"
|
|
49
|
+
return None
|
|
50
|
+
return None # unreachable: schema layer guards dtypes
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _check_constraints(rule: Rule, value: Any) -> list[str]:
|
|
54
|
+
"""Check pattern/min/max/allowed constraints; return error messages."""
|
|
55
|
+
problems: list[str] = []
|
|
56
|
+
text = str(value).strip()
|
|
57
|
+
if rule.pattern is not None and not rule.matches_pattern(text):
|
|
58
|
+
problems.append(f"{value!r} does not match pattern {rule.pattern!r}")
|
|
59
|
+
if rule.dtype in ("int", "float"):
|
|
60
|
+
number = float(text) # safe: dtype check ran first
|
|
61
|
+
if rule.min is not None and number < rule.min:
|
|
62
|
+
problems.append(f"{value!r} is below minimum {rule.min}")
|
|
63
|
+
if rule.max is not None and number > rule.max:
|
|
64
|
+
problems.append(f"{value!r} is above maximum {rule.max}")
|
|
65
|
+
if rule.allowed is not None and text not in rule.allowed:
|
|
66
|
+
problems.append(f"{value!r} is not one of {list(rule.allowed)}")
|
|
67
|
+
return problems
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def validate_dataframe(
|
|
71
|
+
df: pd.DataFrame, ruleset: RuleSet, *, strict: bool = True
|
|
72
|
+
) -> list[ValidationError]:
|
|
73
|
+
"""Validate every cell of ``df`` against ``ruleset``.
|
|
74
|
+
|
|
75
|
+
``strict=True`` (default) also flags columns present in the file but
|
|
76
|
+
absent from the schema as ``unexpected_column`` errors.
|
|
77
|
+
"""
|
|
78
|
+
errors: list[ValidationError] = []
|
|
79
|
+
|
|
80
|
+
for name, rule in ruleset.rules.items():
|
|
81
|
+
if rule.required and name not in df.columns:
|
|
82
|
+
errors.append(
|
|
83
|
+
ValidationError(
|
|
84
|
+
row=0,
|
|
85
|
+
column=name,
|
|
86
|
+
value=None,
|
|
87
|
+
rule="required_column",
|
|
88
|
+
message=f"required column {name!r} is missing from the file",
|
|
89
|
+
)
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
if strict:
|
|
93
|
+
for column in df.columns:
|
|
94
|
+
if column not in ruleset.rules:
|
|
95
|
+
errors.append(
|
|
96
|
+
ValidationError(
|
|
97
|
+
row=0,
|
|
98
|
+
column=str(column),
|
|
99
|
+
value=None,
|
|
100
|
+
rule="unexpected_column",
|
|
101
|
+
message=f"column {str(column)!r} is not defined in the schema",
|
|
102
|
+
)
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
for idx, (_, record) in enumerate(df.iterrows()):
|
|
106
|
+
row_num = idx + 1 # 1-based data row
|
|
107
|
+
for name, rule in ruleset.rules.items():
|
|
108
|
+
if name not in df.columns:
|
|
109
|
+
continue # already reported as missing column
|
|
110
|
+
value = record[name]
|
|
111
|
+
if _is_missing(value):
|
|
112
|
+
if rule.required:
|
|
113
|
+
errors.append(
|
|
114
|
+
ValidationError(
|
|
115
|
+
row=row_num,
|
|
116
|
+
column=name,
|
|
117
|
+
value=None,
|
|
118
|
+
rule="required",
|
|
119
|
+
message="required value is missing",
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
continue
|
|
123
|
+
dtype_problem = _check_dtype(rule, value)
|
|
124
|
+
if dtype_problem is not None:
|
|
125
|
+
errors.append(
|
|
126
|
+
ValidationError(
|
|
127
|
+
row=row_num,
|
|
128
|
+
column=name,
|
|
129
|
+
value=value if not pd.isna(value) else None,
|
|
130
|
+
rule="dtype",
|
|
131
|
+
message=dtype_problem,
|
|
132
|
+
)
|
|
133
|
+
)
|
|
134
|
+
continue # skip constraint checks on wrongly-typed values
|
|
135
|
+
for problem in _check_constraints(rule, value):
|
|
136
|
+
errors.append(
|
|
137
|
+
ValidationError(
|
|
138
|
+
row=row_num,
|
|
139
|
+
column=name,
|
|
140
|
+
value=value if not pd.isna(value) else None,
|
|
141
|
+
rule="constraint",
|
|
142
|
+
message=problem,
|
|
143
|
+
)
|
|
144
|
+
)
|
|
145
|
+
return errors
|