irw-validate 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- irw_validate/__init__.py +28 -0
- irw_validate/_checks.py +325 -0
- irw_validate/cli.py +143 -0
- irw_validate/compat.py +17 -0
- irw_validate/contract.py +214 -0
- irw_validate/core.py +343 -0
- irw_validate/extra.py +241 -0
- irw_validate/live_copies.py +200 -0
- irw_validate/live_cov_range.py +127 -0
- irw_validate/live_dup.py +206 -0
- irw_validate/model.py +113 -0
- irw_validate/repair_cov_age.py +126 -0
- irw_validate/repair_dedupe.py +145 -0
- irw_validate/repair_id_collision.py +169 -0
- irw_validate/repair_occasion.py +122 -0
- irw_validate/sweep_legacy.py +148 -0
- irw_validate-1.0.0.dist-info/METADATA +161 -0
- irw_validate-1.0.0.dist-info/RECORD +21 -0
- irw_validate-1.0.0.dist-info/WHEEL +5 -0
- irw_validate-1.0.0.dist-info/entry_points.txt +2 -0
- irw_validate-1.0.0.dist-info/top_level.txt +1 -0
irw_validate/__init__.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""One IRW format validator, with an exit code (ben-domingue/irw#1703, 1.3).
|
|
2
|
+
|
|
3
|
+
from irw_validate import validate_file, validate_frame, exit_code
|
|
4
|
+
|
|
5
|
+
report = validate_file("out/mytable.csv")
|
|
6
|
+
print(report.ok, [f.check for f in report.errors])
|
|
7
|
+
|
|
8
|
+
Command line:
|
|
9
|
+
|
|
10
|
+
irw-validate out/*.csv # exit 1 if anything blocks
|
|
11
|
+
irw-validate out/x.csv --profile core # the validate_irw.R subset
|
|
12
|
+
irw-validate out/x.csv --strict # warnings block too
|
|
13
|
+
|
|
14
|
+
Why this exists: the checks were forked between `misc/validate_irw.R` (5 checks,
|
|
15
|
+
called by nothing) and `automated_finding/irw_triage_updated.py::run_qc` (~20
|
|
16
|
+
checks, called by fifty scripts but only ever advisorily -- its __main__ exits 0
|
|
17
|
+
however many fail). Neither had an exit code, so neither could gate anything.
|
|
18
|
+
"""
|
|
19
|
+
from .core import (MAX_BYTES, format_report, validate_file, validate_frame,
|
|
20
|
+
validate_paths)
|
|
21
|
+
from .model import (CORE_CHECKS, GATE_ERRORS, PROFILES, Finding, Report,
|
|
22
|
+
exit_code, severity_for)
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"validate_file", "validate_frame", "validate_paths", "format_report",
|
|
26
|
+
"Finding", "Report", "exit_code", "severity_for",
|
|
27
|
+
"CORE_CHECKS", "GATE_ERRORS", "PROFILES", "MAX_BYTES",
|
|
28
|
+
]
|
irw_validate/_checks.py
ADDED
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
"""The IRW format checks themselves, moved verbatim from
|
|
2
|
+
`automated_finding/irw_triage_updated.py::run_qc` (#1703 sub-item 1.3).
|
|
3
|
+
|
|
4
|
+
**Moved, not rewritten.** Fifty scripts in `data/` call `run_qc` and read
|
|
5
|
+
`c.name` / `c.status` / `c.detail` off what it returns, so the check bodies and
|
|
6
|
+
above all their *emission order* are preserved exactly. `irw_validate.compat`
|
|
7
|
+
re-exports this as `run_qc`, and `automated_finding/irw_triage_updated.py`
|
|
8
|
+
re-exports that, which is why none of the fifty needed an edit.
|
|
9
|
+
|
|
10
|
+
`irw_validate.core` layers severity profiles, extra checks and an exit code on
|
|
11
|
+
top of this; nothing here knows about any of that. The golden test in
|
|
12
|
+
`tests/test_validate.py` pins the (name, status) sequence for eight fixtures so
|
|
13
|
+
the move is provably behaviour-preserving.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import collections
|
|
18
|
+
import re
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from math import sqrt
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
#: Columns that say WHEN or UNDER WHAT a measurement was taken, per
|
|
25
|
+
#: `datastandard.md`. `group`, `study` and `treat` are deliberately absent: they
|
|
26
|
+
#: describe the person or the arm, so a person appearing twice under one is a
|
|
27
|
+
#: question, not an answer (#1835).
|
|
28
|
+
OCCASION = ("rt", "rater", "wave", "timepoint", "date", "trialnum", "trial",
|
|
29
|
+
"order", "session", "occasion", "period", "block", "subtest")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class Check:
|
|
34
|
+
name: str
|
|
35
|
+
status: str # "pass" | "warn" | "fail"
|
|
36
|
+
detail: str
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
IRW_REQUIRED = ["id", "item", "resp"]
|
|
40
|
+
ITEM_LEVEL_PREFIXES = ("itemcov_", "qmatrix", "item_family", "rater")
|
|
41
|
+
|
|
42
|
+
_COMPOSITE_TOKENS = {
|
|
43
|
+
"total", "totals", "composite", "subscale", "subscales", "overall",
|
|
44
|
+
"average", "averages", "avg", "mean", "sum", "index", "score", "scores",
|
|
45
|
+
}
|
|
46
|
+
# Whole-label pre/post markers (optionally with a short subscale suffix, e.g.
|
|
47
|
+
# "pre-A", "post_F"). Matched only against the ENTIRE label: a genuine raw
|
|
48
|
+
# item at a pre-wave is usually "pre_anxiety_3", which must not trip this.
|
|
49
|
+
_PREPOST_LABEL = re.compile(
|
|
50
|
+
r"^(pre|post|baseline|follow[-_ ]?up)[-_ ]?[a-z0-9]{0,2}$", re.I)
|
|
51
|
+
|
|
52
|
+
def _looks_composite(label) -> bool:
|
|
53
|
+
"""Does this item label name a computed score rather than a question?"""
|
|
54
|
+
s = str(label).strip()
|
|
55
|
+
if not s:
|
|
56
|
+
return False
|
|
57
|
+
if _PREPOST_LABEL.match(s):
|
|
58
|
+
return True
|
|
59
|
+
# Token-wise, so "meaning_1" doesn't match on "mean" and "scoreboard_2"
|
|
60
|
+
# doesn't match on "score".
|
|
61
|
+
tokens = {t.lower() for t in re.split(r"[^A-Za-z0-9]+", s) if t}
|
|
62
|
+
return bool(tokens & _COMPOSITE_TOKENS)
|
|
63
|
+
|
|
64
|
+
def irw_metadata(df: pd.DataFrame) -> dict:
|
|
65
|
+
"""The IRW's own metadata/density computation, ported from their R/Python."""
|
|
66
|
+
d = df.loc[~df["resp"].isna()].copy()
|
|
67
|
+
d["resp"] = pd.to_numeric(d["resp"], errors="coerce")
|
|
68
|
+
n_resp = len(d)
|
|
69
|
+
n_part = d["id"].nunique()
|
|
70
|
+
n_item = d["item"].nunique()
|
|
71
|
+
# response frequency distribution — the professor's table(df$resp)
|
|
72
|
+
resp_counts = d["resp"].value_counts().sort_index()
|
|
73
|
+
resp_table = {str(k): int(v) for k, v in resp_counts.head(20).items()}
|
|
74
|
+
return {
|
|
75
|
+
"n_responses": n_resp,
|
|
76
|
+
"n_categories": int(d["resp"].nunique()),
|
|
77
|
+
"n_participants": n_part,
|
|
78
|
+
"n_items": n_item,
|
|
79
|
+
"responses_per_participant": round(n_resp / n_part, 2) if n_part else 0,
|
|
80
|
+
"responses_per_item": round(n_resp / n_item, 2) if n_item else 0,
|
|
81
|
+
"density": round((sqrt(n_resp) / n_part) * (sqrt(n_resp) / n_item), 4)
|
|
82
|
+
if n_part and n_item else 0,
|
|
83
|
+
"resp_distribution": resp_table,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def run_qc(df: pd.DataFrame, coercion_method: str = "",
|
|
88
|
+
original_cols: list = None) -> list:
|
|
89
|
+
"""QC checks. The first block is ported directly from the IRW's official
|
|
90
|
+
validate_irw.R (statuses: pass=OK, warn=NOTE, fail=ERROR). The second block
|
|
91
|
+
is extra heuristics we add on top, clearly labelled."""
|
|
92
|
+
checks = []
|
|
93
|
+
original_cols = original_cols or []
|
|
94
|
+
|
|
95
|
+
# ===== ported from validate_irw.R =====================================
|
|
96
|
+
|
|
97
|
+
# required columns (ERROR if missing)
|
|
98
|
+
missing = [c for c in IRW_REQUIRED if c not in df.columns]
|
|
99
|
+
if missing:
|
|
100
|
+
checks.append(Check("required_columns", "fail",
|
|
101
|
+
f"missing required columns: {', '.join(missing)}"))
|
|
102
|
+
return checks # nothing else is meaningful without these
|
|
103
|
+
checks.append(Check("required_columns", "pass", "id/item/resp present"))
|
|
104
|
+
|
|
105
|
+
# NAs in required columns: all-NA = ERROR, some-NA = NOTE
|
|
106
|
+
for col in IRW_REQUIRED:
|
|
107
|
+
n_na = df[col].isna().sum()
|
|
108
|
+
if n_na == len(df):
|
|
109
|
+
checks.append(Check(f"{col}_na", "fail", f"{col} is entirely NA"))
|
|
110
|
+
elif n_na > 0:
|
|
111
|
+
checks.append(Check(f"{col}_na", "warn", f"{col} has {n_na} NAs"))
|
|
112
|
+
|
|
113
|
+
# resp must be numeric (ERROR)
|
|
114
|
+
resp_num = pd.to_numeric(df["resp"], errors="coerce")
|
|
115
|
+
if resp_num.notna().mean() < 0.99:
|
|
116
|
+
checks.append(Check("resp_numeric", "fail",
|
|
117
|
+
f"resp is not numeric (only "
|
|
118
|
+
f"{resp_num.notna().mean():.0%} parse as numbers)"))
|
|
119
|
+
else:
|
|
120
|
+
checks.append(Check("resp_numeric", "pass", "resp is numeric"))
|
|
121
|
+
|
|
122
|
+
# duplicate id+item: ERROR if no longitudinal column, else NOTE
|
|
123
|
+
longitudinal = [c for c in ("wave", "timepoint", "date") if c in df.columns]
|
|
124
|
+
dups = df.duplicated(subset=["id", "item"]).sum()
|
|
125
|
+
if dups > 0 and not longitudinal:
|
|
126
|
+
checks.append(Check("dup_id_item", "fail",
|
|
127
|
+
f"{dups} duplicate id+item rows with no "
|
|
128
|
+
"wave/timepoint/date column"))
|
|
129
|
+
elif dups > 0:
|
|
130
|
+
checks.append(Check("dup_id_item", "warn",
|
|
131
|
+
f"{dups} duplicate id+item rows "
|
|
132
|
+
f"(longitudinal column {longitudinal} present — likely ok)"))
|
|
133
|
+
else:
|
|
134
|
+
checks.append(Check("dup_id_item", "pass", "id+item rows unique"))
|
|
135
|
+
|
|
136
|
+
# covariate naming: extra columns without a recognized name/prefix = NOTE.
|
|
137
|
+
# (Broadened from validate_irw.R's narrow list to the full documented
|
|
138
|
+
# standard, so legitimate columns like item_family/treat aren't flagged.)
|
|
139
|
+
#
|
|
140
|
+
# OCCASION belongs in this set, and leaving it out made the validator
|
|
141
|
+
# contradict itself: `dup_id_item` accepts `trialnum` as the column that
|
|
142
|
+
# explains a repeat, and `cov_prefix` then told you to rename it `cov_`,
|
|
143
|
+
# which would both misdescribe it -- a covariate is invariant to the person,
|
|
144
|
+
# a trial index is the opposite -- and stop `dup_id_item` from seeing it,
|
|
145
|
+
# re-breaking the table the rename had just fixed. Seen on `motion` and
|
|
146
|
+
# `rr98_accuracy` (irw#1842 block J). One list, so the two cannot drift.
|
|
147
|
+
known = {"id", "item", "resp", "date", "treat", "item_family"} | set(OCCASION)
|
|
148
|
+
known_prefix = ("cov_", "itemcov_", "qmatrix", "trial_")
|
|
149
|
+
unprefixed = [c for c in df.columns
|
|
150
|
+
if c not in known and not c.startswith(known_prefix)]
|
|
151
|
+
if unprefixed:
|
|
152
|
+
checks.append(Check("cov_prefix", "warn",
|
|
153
|
+
f"unrecognized columns (prefix with cov_ if "
|
|
154
|
+
f"covariates): {', '.join(unprefixed)}"))
|
|
155
|
+
|
|
156
|
+
# ===== extra heuristics (beyond the official validator) ===============
|
|
157
|
+
|
|
158
|
+
# resp scale sanity — flag a resp that looks continuous/mis-parsed
|
|
159
|
+
ncat = resp_num.nunique()
|
|
160
|
+
if ncat <= 1:
|
|
161
|
+
checks.append(Check("resp_variation*", "fail",
|
|
162
|
+
"resp has no variation (1 unique value)"))
|
|
163
|
+
elif ncat > 50:
|
|
164
|
+
checks.append(Check("resp_ordinal*", "warn",
|
|
165
|
+
f"{ncat} distinct resp values — confirm continuous, "
|
|
166
|
+
"not mis-parsed"))
|
|
167
|
+
|
|
168
|
+
# P1 #3: resp coding direction — can't auto-verify; always warn after melt.
|
|
169
|
+
if coercion_method == "wide-to-long":
|
|
170
|
+
checks.append(Check(
|
|
171
|
+
"resp_direction*", "warn",
|
|
172
|
+
"Cannot auto-verify: within each item, higher resp values must "
|
|
173
|
+
"indicate more of the construct (IRW standard). Confirm no "
|
|
174
|
+
"unreversed items."
|
|
175
|
+
))
|
|
176
|
+
|
|
177
|
+
# P1 #4: imputed values — column name signals and mean-imputation signature.
|
|
178
|
+
if original_cols:
|
|
179
|
+
imputed_signals = [c for c in original_cols
|
|
180
|
+
if re.search(r"_imp(?:uted)?$|_filled$|_flag$", c,
|
|
181
|
+
re.I)]
|
|
182
|
+
if imputed_signals:
|
|
183
|
+
checks.append(Check("imputed_values*", "warn",
|
|
184
|
+
f"Columns suggest imputed values may be present: "
|
|
185
|
+
f"{imputed_signals}. IRW requires their removal."))
|
|
186
|
+
# Mean-imputation signature: any item where one value accounts for >60% of rows.
|
|
187
|
+
if resp_num.notna().any():
|
|
188
|
+
by_item = df.groupby("item")["resp"]
|
|
189
|
+
for item_name, grp in by_item:
|
|
190
|
+
vc = grp.value_counts(normalize=True)
|
|
191
|
+
if not vc.empty and vc.iloc[0] > 0.60:
|
|
192
|
+
checks.append(Check("imputed_values*", "warn",
|
|
193
|
+
f"Item '{item_name}' has one resp value "
|
|
194
|
+
f"accounting for {vc.iloc[0]:.0%} of responses "
|
|
195
|
+
"— possible mean imputation."))
|
|
196
|
+
break # one warning is enough
|
|
197
|
+
|
|
198
|
+
# P1 #5: date column validation.
|
|
199
|
+
if "date" in df.columns:
|
|
200
|
+
d = pd.to_numeric(df["date"], errors="coerce")
|
|
201
|
+
if d.isna().mean() > 0.1:
|
|
202
|
+
checks.append(Check("date_numeric*", "warn",
|
|
203
|
+
"date column is not numeric — IRW requires Unix "
|
|
204
|
+
"seconds (or seconds since first observation)"))
|
|
205
|
+
elif d.notna().any() and d.max() < 1e8:
|
|
206
|
+
checks.append(Check("date_range*", "warn",
|
|
207
|
+
f"date max={d.max():.0f} — looks too small for "
|
|
208
|
+
"Unix seconds; verify units"))
|
|
209
|
+
|
|
210
|
+
# P1 #6: rt column validation.
|
|
211
|
+
if "rt" in df.columns:
|
|
212
|
+
rt = pd.to_numeric(df["rt"], errors="coerce")
|
|
213
|
+
if rt.isna().mean() > 0.1:
|
|
214
|
+
checks.append(Check("rt_numeric*", "warn",
|
|
215
|
+
"rt column is not numeric"))
|
|
216
|
+
elif rt.notna().any():
|
|
217
|
+
if rt.median() > 60000:
|
|
218
|
+
checks.append(Check("rt_units*", "warn",
|
|
219
|
+
f"rt median={rt.median():.0f} — likely "
|
|
220
|
+
"milliseconds, not seconds (IRW requires "
|
|
221
|
+
"seconds)"))
|
|
222
|
+
if (rt < 0).any():
|
|
223
|
+
checks.append(Check("rt_negative*", "warn",
|
|
224
|
+
"rt has negative values"))
|
|
225
|
+
|
|
226
|
+
# treat column should be 0/1 if present
|
|
227
|
+
if "treat" in df.columns:
|
|
228
|
+
bad = set(pd.unique(df["treat"].dropna())) - {0, 1}
|
|
229
|
+
if bad:
|
|
230
|
+
checks.append(Check("treat_binary*", "warn",
|
|
231
|
+
f"treat has non-0/1 values {sorted(bad)[:5]}"))
|
|
232
|
+
|
|
233
|
+
# P2 #7: item-level columns dropped during melt — remind user to verify.
|
|
234
|
+
if original_cols and coercion_method == "wide-to-long":
|
|
235
|
+
item_level_found = [c for c in original_cols
|
|
236
|
+
if any(c.startswith(p) for p in ITEM_LEVEL_PREFIXES)]
|
|
237
|
+
if item_level_found:
|
|
238
|
+
checks.append(Check("item_level_cols*", "warn",
|
|
239
|
+
f"Item-level columns {item_level_found} were "
|
|
240
|
+
"excluded from the melt — verify they are "
|
|
241
|
+
"correctly aligned after conversion."))
|
|
242
|
+
|
|
243
|
+
# P2 #7: multi-scale detection — distinct item-name prefixes suggest separate
|
|
244
|
+
# constructs that must be split into separate tables.
|
|
245
|
+
if "item" in df.columns:
|
|
246
|
+
prefixes = [re.split(r"[\d_]", str(i))[0].lower()
|
|
247
|
+
for i in df["item"].unique() if str(i)]
|
|
248
|
+
prefix_counts = pd.Series(prefixes).value_counts()
|
|
249
|
+
dominant = prefix_counts[prefix_counts >= 3]
|
|
250
|
+
if len(dominant) >= 2:
|
|
251
|
+
checks.append(Check("multi_scale*", "warn",
|
|
252
|
+
f"Item names suggest {len(dominant)} subscales "
|
|
253
|
+
f"({list(dominant.index)[:4]}) — IRW requires "
|
|
254
|
+
"separate tables per construct."))
|
|
255
|
+
|
|
256
|
+
# Response-scale homogeneity. The existing multi_scale* check reads item
|
|
257
|
+
# *names*; this one reads the responses themselves, which is what actually
|
|
258
|
+
# catches a mailing that bundled several instruments. Two distinct
|
|
259
|
+
# failures fall out of the same per-item range profile:
|
|
260
|
+
# * a substantial minority of items on a different range -> two scales
|
|
261
|
+
# in one table, which breaks "one table per construct" and leaves `resp`
|
|
262
|
+
# meaning different things in different rows;
|
|
263
|
+
# * one or two isolated items off the modal range -> almost always not
|
|
264
|
+
# an item at all (an administrative or count column swept in).
|
|
265
|
+
# Both were live defects in the 2026-08-26 Eugene-Springfield build:
|
|
266
|
+
# `sdv` spanned 1-5, 1-7, 1-8 and 1-9 at once, and `submiss` -- a
|
|
267
|
+
# missing-response count, 94.8% zero -- was the only column in the HPQ
|
|
268
|
+
# outside its 1-5 scale.
|
|
269
|
+
# A mix of numbers and invalid text has no comparable response-scale range.
|
|
270
|
+
# Skip this heuristic so the numeric finding can be returned (#2029).
|
|
271
|
+
# This also avoids int/string comparisons for in-memory inputs, whether
|
|
272
|
+
# the incompatible values occur within one item or across several items.
|
|
273
|
+
mixed_numeric_text = (resp_num.notna().any()
|
|
274
|
+
and (df["resp"].notna() & resp_num.isna()).any())
|
|
275
|
+
if {"item", "resp"}.issubset(df.columns) and not mixed_numeric_text:
|
|
276
|
+
rng = df.dropna(subset=["resp"]).groupby("item")["resp"].agg(["min", "max"])
|
|
277
|
+
if len(rng) >= 3:
|
|
278
|
+
profile = collections.Counter(zip(rng["min"], rng["max"]))
|
|
279
|
+
(modal, modal_n), = profile.most_common(1)
|
|
280
|
+
off = rng[(rng["min"] != modal[0]) | (rng["max"] != modal[1])]
|
|
281
|
+
# Only a range that *exceeds* the modal one is evidence of a
|
|
282
|
+
# different scale; an item nobody answered at the ceiling simply
|
|
283
|
+
# has a lower observed max.
|
|
284
|
+
over = off[(off["max"] > modal[1]) | (off["min"] < modal[0])]
|
|
285
|
+
share = len(over) / len(rng)
|
|
286
|
+
if share >= 0.15:
|
|
287
|
+
other = collections.Counter(zip(over["min"], over["max"]))
|
|
288
|
+
checks.append(Check("resp_scale_mixed", "fail",
|
|
289
|
+
f"items span more than one response scale: "
|
|
290
|
+
f"{modal_n} on {modal[0]}-{modal[1]} and {len(over)} on "
|
|
291
|
+
f"{[f'{a}-{b}' for a, b in list(other)[:3]]}. IRW requires "
|
|
292
|
+
"one table per construct; split before submitting."))
|
|
293
|
+
elif len(over):
|
|
294
|
+
checks.append(Check("item_scale_outlier", "warn",
|
|
295
|
+
f"{len(over)} item(s) fall outside the table's "
|
|
296
|
+
f"{modal[0]}-{modal[1]} scale: {list(over.index)[:4]}. An "
|
|
297
|
+
"isolated out-of-range column is usually not an item -- "
|
|
298
|
+
"check for an administrative or count field."))
|
|
299
|
+
|
|
300
|
+
# Composite columns masquerading as items. A summary table melts into a
|
|
301
|
+
# perfectly well-formed id/item/resp frame and passes every structural
|
|
302
|
+
# check above -- the only tell is what the items are NAMED.
|
|
303
|
+
if "item" in df.columns:
|
|
304
|
+
labels = [i for i in df["item"].unique() if str(i).strip()]
|
|
305
|
+
comp = [i for i in labels if _looks_composite(i)]
|
|
306
|
+
if labels and len(comp) == len(labels):
|
|
307
|
+
checks.append(Check("composite_items*", "fail",
|
|
308
|
+
f"every item label names a computed score "
|
|
309
|
+
f"({[str(c) for c in comp[:4]]}) — this looks "
|
|
310
|
+
"like a summary/aggregate table, not raw "
|
|
311
|
+
"item-level responses"))
|
|
312
|
+
elif comp:
|
|
313
|
+
checks.append(Check("composite_items*", "warn",
|
|
314
|
+
f"{len(comp)}/{len(labels)} item labels name "
|
|
315
|
+
f"computed scores ({[str(c) for c in comp[:4]]}) "
|
|
316
|
+
"— drop them, or confirm they are real items"))
|
|
317
|
+
|
|
318
|
+
# IRW's own density signal — very sparse data is worth a look
|
|
319
|
+
meta = irw_metadata(df)
|
|
320
|
+
if meta["density"] < 0.01:
|
|
321
|
+
checks.append(Check("density*", "warn",
|
|
322
|
+
f"very sparse (density={meta['density']}); fine for "
|
|
323
|
+
"adaptive/booklet designs, else verify"))
|
|
324
|
+
|
|
325
|
+
return checks
|
irw_validate/cli.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""`irw-validate` -- check IRW tables and exit non-zero if anything blocks.
|
|
2
|
+
|
|
3
|
+
irw-validate out/*.csv
|
|
4
|
+
irw-validate out/x.csv --profile core # the validate_irw.R subset
|
|
5
|
+
irw-validate out/x.csv --strict # warnings block too
|
|
6
|
+
irw-validate out/x.csv --json # machine-readable, for CI
|
|
7
|
+
irw-validate out/x.csv --override-check resp_scale_mixed \\
|
|
8
|
+
--override "two response formats, one construct; author confirmed 2026-09-02"
|
|
9
|
+
|
|
10
|
+
Exit codes, matching red_up's contract: 0 ok - 1 something blocks - 2 bad input.
|
|
11
|
+
|
|
12
|
+
**On the override.** The reason is the flag's *argument*, so overriding without
|
|
13
|
+
saying why is structurally impossible. The flag is not called `--force` or
|
|
14
|
+
`--no-verify`: those names invite reflex use, and the point is to make a waiver
|
|
15
|
+
a decision someone signed. Overridden findings are reprinted under OVERRIDDEN
|
|
16
|
+
rather than suppressed, and the reason is appended to
|
|
17
|
+
`processing_notes/validator_overrides.csv` so a waiver leaves a trail even when
|
|
18
|
+
nobody keeps the terminal output.
|
|
19
|
+
|
|
20
|
+
This formalises something that already happens informally: `data/cao_2026_cdss.py`
|
|
21
|
+
waives `resp_scale_mixed` in a prose comment -- correct judgment, recorded where
|
|
22
|
+
no tool can read it.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import argparse
|
|
27
|
+
import csv
|
|
28
|
+
import datetime as dt
|
|
29
|
+
import getpass
|
|
30
|
+
import json
|
|
31
|
+
import sys
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
|
|
34
|
+
from .core import format_report, validate_file
|
|
35
|
+
from .model import PROFILES, exit_code
|
|
36
|
+
|
|
37
|
+
MIN_REASON = 20
|
|
38
|
+
#: Where waivers are recorded. Overridable so a test run -- or anyone working
|
|
39
|
+
#: outside a checkout -- does not append to the repository's ledger.
|
|
40
|
+
LEDGER_ENV = "IRW_VALIDATE_LEDGER"
|
|
41
|
+
_DEFAULT_LEDGER = (Path(__file__).resolve().parent.parent
|
|
42
|
+
/ "processing_notes" / "validator_overrides.csv")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def ledger_path() -> Path:
|
|
46
|
+
import os
|
|
47
|
+
return Path(os.environ.get(LEDGER_ENV) or _DEFAULT_LEDGER)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _record(reasons_path: Path, rows: list) -> None:
|
|
51
|
+
"""Append waivers to the ledger. Never fatal -- a gate that fails because
|
|
52
|
+
it could not write its own audit file would be worse than the gap."""
|
|
53
|
+
try:
|
|
54
|
+
reasons_path.parent.mkdir(parents=True, exist_ok=True)
|
|
55
|
+
new = not reasons_path.exists()
|
|
56
|
+
with reasons_path.open("a", newline="") as fh:
|
|
57
|
+
w = csv.writer(fh)
|
|
58
|
+
if new:
|
|
59
|
+
w.writerow(["date", "tool", "table", "checks", "reason", "user"])
|
|
60
|
+
w.writerows(rows)
|
|
61
|
+
except OSError as exc:
|
|
62
|
+
print(f"warning: could not write {reasons_path}: {exc}", file=sys.stderr)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def main(argv: list | None = None) -> int:
|
|
66
|
+
ap = argparse.ArgumentParser(
|
|
67
|
+
prog="irw-validate",
|
|
68
|
+
description="Validate IRW tables against the data standard.")
|
|
69
|
+
ap.add_argument("paths", nargs="+", help="CSV (or any format load_table reads)")
|
|
70
|
+
ap.add_argument("--profile", default="upload", choices=PROFILES,
|
|
71
|
+
help="core = the validate_irw.R subset; triage = today's "
|
|
72
|
+
"run_qc behaviour; upload = the gate (default); "
|
|
73
|
+
"legacy = upload minus rules that postdate the table")
|
|
74
|
+
ap.add_argument("--strict", action="store_true",
|
|
75
|
+
help="warnings block too")
|
|
76
|
+
ap.add_argument("--json", action="store_true", help="machine-readable output")
|
|
77
|
+
ap.add_argument("--override", metavar="REASON",
|
|
78
|
+
help=f"waive blocking findings, giving a reason of at least "
|
|
79
|
+
f"{MIN_REASON} characters")
|
|
80
|
+
ap.add_argument("--override-check", action="append", metavar="NAME", default=[],
|
|
81
|
+
help="limit --override to this check (repeatable); without "
|
|
82
|
+
"it, --override waives every error")
|
|
83
|
+
args = ap.parse_args(argv)
|
|
84
|
+
|
|
85
|
+
if args.override_check and not args.override:
|
|
86
|
+
print("irw-validate: --override-check needs --override REASON",
|
|
87
|
+
file=sys.stderr)
|
|
88
|
+
return 2
|
|
89
|
+
if args.override is not None and len(args.override.strip()) < MIN_REASON:
|
|
90
|
+
print(f"irw-validate: that is not a reason -- give at least "
|
|
91
|
+
f"{MIN_REASON} characters saying why this table is an exception",
|
|
92
|
+
file=sys.stderr)
|
|
93
|
+
return 2
|
|
94
|
+
|
|
95
|
+
reports = []
|
|
96
|
+
for path in args.paths:
|
|
97
|
+
try:
|
|
98
|
+
reports.append(validate_file(path, profile=args.profile))
|
|
99
|
+
except FileNotFoundError:
|
|
100
|
+
print(f"irw-validate: no such file: {path}", file=sys.stderr)
|
|
101
|
+
return 2
|
|
102
|
+
except Exception as exc:
|
|
103
|
+
print(f"irw-validate: cannot read {path}: {exc}", file=sys.stderr)
|
|
104
|
+
return 2
|
|
105
|
+
|
|
106
|
+
ledger_rows = []
|
|
107
|
+
if args.override:
|
|
108
|
+
wanted = set(args.override_check)
|
|
109
|
+
for report in reports:
|
|
110
|
+
waived = [f for f in report.errors
|
|
111
|
+
if not wanted or f.check in wanted]
|
|
112
|
+
if not waived:
|
|
113
|
+
continue
|
|
114
|
+
report.findings = [f for f in report.findings if f not in waived]
|
|
115
|
+
report.overridden.extend(waived)
|
|
116
|
+
report.override_reason = args.override
|
|
117
|
+
ledger_rows.append([
|
|
118
|
+
dt.date.today().isoformat(), "irw-validate", report.label,
|
|
119
|
+
";".join(sorted({f.check for f in waived})), args.override,
|
|
120
|
+
getpass.getuser(),
|
|
121
|
+
])
|
|
122
|
+
if ledger_rows:
|
|
123
|
+
_record(ledger_path(), ledger_rows)
|
|
124
|
+
|
|
125
|
+
if args.json:
|
|
126
|
+
print(json.dumps([r.to_dict() for r in reports], indent=2))
|
|
127
|
+
else:
|
|
128
|
+
for report in reports:
|
|
129
|
+
print(format_report(report))
|
|
130
|
+
|
|
131
|
+
code = exit_code(reports, strict=args.strict)
|
|
132
|
+
if code and not args.json:
|
|
133
|
+
blocking = sum(len(r.errors) for r in reports)
|
|
134
|
+
if blocking:
|
|
135
|
+
print(f"\n{blocking} blocking finding(s). Fix them, or waive with "
|
|
136
|
+
f"--override \"<why>\".", file=sys.stderr)
|
|
137
|
+
else:
|
|
138
|
+
print("\n--strict: warnings are blocking.", file=sys.stderr)
|
|
139
|
+
return code
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
if __name__ == "__main__":
|
|
143
|
+
sys.exit(main())
|
irw_validate/compat.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Backwards-compatible `run_qc` for the fifty callers in `data/`.
|
|
2
|
+
|
|
3
|
+
Fifty conversion scripts do `from irw_triage_updated import run_qc` and read
|
|
4
|
+
`c.name` / `c.status` / `c.detail`. None of them imports `Check` or anything
|
|
5
|
+
else, so this is the entire compatibility surface -- but fifty files break at
|
|
6
|
+
someone else's runtime if it moves, which is why the checks were *moved*
|
|
7
|
+
verbatim rather than rewritten, and why the golden test pins their emission
|
|
8
|
+
order.
|
|
9
|
+
|
|
10
|
+
`run_qc` here is the moved implementation itself, not a translation of it: the
|
|
11
|
+
severity profiles in `irw_validate.model` are layered on top by
|
|
12
|
+
`irw_validate.core`, never underneath. So a caller of `run_qc` sees exactly what
|
|
13
|
+
it saw before this package existed.
|
|
14
|
+
"""
|
|
15
|
+
from ._checks import Check, irw_metadata, run_qc
|
|
16
|
+
|
|
17
|
+
__all__ = ["Check", "run_qc", "irw_metadata"]
|