statskeptic 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- statskeptic/__init__.py +43 -0
- statskeptic/agent.py +208 -0
- statskeptic/cli.py +82 -0
- statskeptic/critique/__init__.py +25 -0
- statskeptic/critique/engine.py +418 -0
- statskeptic/critique/models.py +66 -0
- statskeptic/critique/revise.py +168 -0
- statskeptic/errors.py +20 -0
- statskeptic/execution.py +137 -0
- statskeptic/loader.py +33 -0
- statskeptic/plan/__init__.py +11 -0
- statskeptic/plan/models.py +65 -0
- statskeptic/plan/planner.py +512 -0
- statskeptic/profile/__init__.py +10 -0
- statskeptic/profile/models.py +83 -0
- statskeptic/profile/profiler.py +152 -0
- statskeptic/py.typed +0 -0
- statskeptic/report/__init__.py +3 -0
- statskeptic/report/models.py +209 -0
- statskeptic/stats/__init__.py +45 -0
- statskeptic/stats/_support.py +78 -0
- statskeptic/stats/association.py +261 -0
- statskeptic/stats/assumptions.py +260 -0
- statskeptic/stats/k_group.py +163 -0
- statskeptic/stats/regression.py +231 -0
- statskeptic/stats/results.py +185 -0
- statskeptic/stats/two_group.py +252 -0
- statskeptic-0.1.0.dist-info/METADATA +201 -0
- statskeptic-0.1.0.dist-info/RECORD +32 -0
- statskeptic-0.1.0.dist-info/WHEEL +4 -0
- statskeptic-0.1.0.dist-info/entry_points.txt +2 -0
- statskeptic-0.1.0.dist-info/licenses/LICENSE +21 -0
statskeptic/__init__.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""statskeptic - a data-analysis agent that distrusts its own conclusions.
|
|
2
|
+
|
|
3
|
+
Point it at a dataset and a question. It plans an analysis, runs it with real,
|
|
4
|
+
vetted statistical code (never a number invented by a language model), then attacks
|
|
5
|
+
its own result against a methodological rubric - assumption violations, multiple
|
|
6
|
+
comparisons, confounding, underpowered samples, leakage - and reports plainly what
|
|
7
|
+
it found and, just as importantly, what it cannot conclude.
|
|
8
|
+
|
|
9
|
+
The public API grows as each capability lands; for now this exposes the package
|
|
10
|
+
version so an editable install resolves cleanly.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
14
|
+
|
|
15
|
+
from .agent import analyze
|
|
16
|
+
from .critique.models import Critique, CritiqueCategory, Verdict
|
|
17
|
+
from .errors import AnalysisError, StatskepticError
|
|
18
|
+
from .plan.models import AnalysisPlan, Decline, Method, PlanHints, QuestionType
|
|
19
|
+
from .profile.models import DataProfile
|
|
20
|
+
from .report.models import Analysis, Report
|
|
21
|
+
|
|
22
|
+
try:
|
|
23
|
+
__version__ = version("statskeptic")
|
|
24
|
+
except PackageNotFoundError: # pragma: no cover - only during local source runs
|
|
25
|
+
__version__ = "0.0.0"
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"Analysis",
|
|
29
|
+
"AnalysisError",
|
|
30
|
+
"AnalysisPlan",
|
|
31
|
+
"Critique",
|
|
32
|
+
"CritiqueCategory",
|
|
33
|
+
"DataProfile",
|
|
34
|
+
"Decline",
|
|
35
|
+
"Method",
|
|
36
|
+
"PlanHints",
|
|
37
|
+
"QuestionType",
|
|
38
|
+
"Report",
|
|
39
|
+
"StatskepticError",
|
|
40
|
+
"Verdict",
|
|
41
|
+
"__version__",
|
|
42
|
+
"analyze",
|
|
43
|
+
]
|
statskeptic/agent.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""The end-to-end loop: profile -> plan -> execute -> critique -> revise -> report.
|
|
2
|
+
|
|
3
|
+
`analyze` is the one public entry point. It owns the orchestration and the honesty
|
|
4
|
+
decisions: when the planner declines, when a screen needs a multiplicity correction,
|
|
5
|
+
and when surviving objections mean the only honest verdict is "cannot conclude".
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import Iterable
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
from pandas.api import types as pdt
|
|
16
|
+
|
|
17
|
+
from .critique.engine import CritiqueContext, run_critique
|
|
18
|
+
from .critique.models import Critique, Verdict
|
|
19
|
+
from .critique.revise import (
|
|
20
|
+
apply_holm,
|
|
21
|
+
decide_verdict,
|
|
22
|
+
multiple_comparisons_critique,
|
|
23
|
+
revise,
|
|
24
|
+
)
|
|
25
|
+
from .execution import execute
|
|
26
|
+
from .loader import read_table
|
|
27
|
+
from .plan.models import AnalysisPlan, Decline, Method, PlanHints, QuestionType
|
|
28
|
+
from .plan.planner import plan as make_plan
|
|
29
|
+
from .profile.models import DataProfile
|
|
30
|
+
from .profile.profiler import build_profile
|
|
31
|
+
from .report.models import Analysis, Report
|
|
32
|
+
from .stats.results import Severity, StatResult
|
|
33
|
+
|
|
34
|
+
_VERDICT_RANK = {
|
|
35
|
+
Verdict.defensible: 0,
|
|
36
|
+
Verdict.defensible_with_caveats: 1,
|
|
37
|
+
Verdict.cannot_conclude: 2,
|
|
38
|
+
Verdict.declined: 2,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def analyze(
|
|
43
|
+
data: pd.DataFrame | str | Path,
|
|
44
|
+
question: str,
|
|
45
|
+
*,
|
|
46
|
+
hints: PlanHints | None = None,
|
|
47
|
+
alpha: float = 0.05,
|
|
48
|
+
) -> Report:
|
|
49
|
+
if not 0.0 < alpha < 1.0:
|
|
50
|
+
raise ValueError(f"alpha must be between 0 and 1 (exclusive); got {alpha}")
|
|
51
|
+
df = data if isinstance(data, pd.DataFrame) else read_table(data)
|
|
52
|
+
df = _coerce_dates(df)
|
|
53
|
+
# Treat infinities as missing for the whole pipeline: the profiler then counts them
|
|
54
|
+
# as missing data and every downstream routine drops them with the NaNs.
|
|
55
|
+
df = df.replace([np.inf, -np.inf], np.nan)
|
|
56
|
+
profile = build_profile(df)
|
|
57
|
+
planned = make_plan(question, profile, hints)
|
|
58
|
+
|
|
59
|
+
if isinstance(planned, Decline):
|
|
60
|
+
return Report(
|
|
61
|
+
question=question,
|
|
62
|
+
n_rows=profile.n_rows,
|
|
63
|
+
n_cols=profile.n_cols,
|
|
64
|
+
verdict=Verdict.declined,
|
|
65
|
+
declined=planned,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
if planned.question_type == QuestionType.screen:
|
|
69
|
+
return _run_screen(planned, df, profile, question, alpha)
|
|
70
|
+
return _run_single(planned, df, profile, question, alpha)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# Explicit formats only. Inferring dates from arbitrary strings is where pandas misreads
|
|
74
|
+
# categorical codes as dates and emits warnings; an exact format either matches every
|
|
75
|
+
# value in a column or that column is left exactly as it was.
|
|
76
|
+
_DATE_FORMATS = ("%Y-%m-%d", "%Y-%m-%dT%H:%M:%S", "%Y-%m-%d %H:%M:%S", "%Y/%m/%d")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _coerce_dates(df: pd.DataFrame) -> pd.DataFrame:
|
|
80
|
+
out = df
|
|
81
|
+
for col in df.columns:
|
|
82
|
+
series = df[col]
|
|
83
|
+
# Only string columns are candidates (pandas 3 reads CSV text as the str dtype,
|
|
84
|
+
# not object). Numeric, datetime, and boolean columns are left untouched.
|
|
85
|
+
if not (pdt.is_object_dtype(series) or pdt.is_string_dtype(series)):
|
|
86
|
+
continue
|
|
87
|
+
non_null = series.dropna()
|
|
88
|
+
if non_null.empty:
|
|
89
|
+
continue
|
|
90
|
+
for fmt in _DATE_FORMATS:
|
|
91
|
+
try:
|
|
92
|
+
pd.to_datetime(non_null, format=fmt)
|
|
93
|
+
except (ValueError, TypeError):
|
|
94
|
+
continue
|
|
95
|
+
if out is df:
|
|
96
|
+
out = df.copy()
|
|
97
|
+
out[col] = pd.to_datetime(series, format=fmt, errors="coerce")
|
|
98
|
+
break
|
|
99
|
+
return out
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _run_single(
|
|
103
|
+
planned: AnalysisPlan,
|
|
104
|
+
df: pd.DataFrame,
|
|
105
|
+
profile: DataProfile,
|
|
106
|
+
question: str,
|
|
107
|
+
alpha: float,
|
|
108
|
+
) -> Report:
|
|
109
|
+
result = execute(planned, df, alpha)
|
|
110
|
+
outcome = revise(planned, result, profile, df, alpha)
|
|
111
|
+
verdict = decide_verdict(outcome.residual)
|
|
112
|
+
analysis = Analysis(
|
|
113
|
+
plan=outcome.plan,
|
|
114
|
+
result=outcome.result,
|
|
115
|
+
revision_steps=outcome.steps,
|
|
116
|
+
critiques=outcome.residual,
|
|
117
|
+
verdict=verdict,
|
|
118
|
+
)
|
|
119
|
+
return Report(
|
|
120
|
+
question=question,
|
|
121
|
+
n_rows=profile.n_rows,
|
|
122
|
+
n_cols=profile.n_cols,
|
|
123
|
+
analyses=[analysis],
|
|
124
|
+
verdict=verdict,
|
|
125
|
+
cannot_conclude=_cannot_conclude(outcome.residual, outcome.result),
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _run_screen(
|
|
130
|
+
planned: AnalysisPlan,
|
|
131
|
+
df: pd.DataFrame,
|
|
132
|
+
profile: DataProfile,
|
|
133
|
+
question: str,
|
|
134
|
+
alpha: float,
|
|
135
|
+
) -> Report:
|
|
136
|
+
assert planned.outcome
|
|
137
|
+
subplans = [
|
|
138
|
+
AnalysisPlan(
|
|
139
|
+
question=question,
|
|
140
|
+
question_type=QuestionType.association,
|
|
141
|
+
method=Method.pearson,
|
|
142
|
+
outcome=planned.outcome,
|
|
143
|
+
predictors=[cand],
|
|
144
|
+
)
|
|
145
|
+
for cand in planned.candidates
|
|
146
|
+
]
|
|
147
|
+
results = [execute(sp, df, alpha) for sp in subplans]
|
|
148
|
+
per_critiques = [
|
|
149
|
+
run_critique(CritiqueContext(sp, r, profile, df))
|
|
150
|
+
for sp, r in zip(subplans, results, strict=True)
|
|
151
|
+
]
|
|
152
|
+
|
|
153
|
+
mc = multiple_comparisons_critique(results, planned.outcome, alpha)
|
|
154
|
+
session: list[Critique] = []
|
|
155
|
+
if mc is not None:
|
|
156
|
+
results = apply_holm(results, alpha)
|
|
157
|
+
session = [mc]
|
|
158
|
+
|
|
159
|
+
analyses = [
|
|
160
|
+
Analysis(plan=sp, result=r, critiques=crits, verdict=decide_verdict(crits))
|
|
161
|
+
for sp, r, crits in zip(subplans, results, per_critiques, strict=True)
|
|
162
|
+
]
|
|
163
|
+
verdict = _worst(a.verdict for a in analyses)
|
|
164
|
+
if mc is not None:
|
|
165
|
+
verdict = _worse(verdict, Verdict.defensible_with_caveats)
|
|
166
|
+
|
|
167
|
+
cannot = [line for crits in per_critiques for line in _cannot_conclude(crits, None)]
|
|
168
|
+
return Report(
|
|
169
|
+
question=question,
|
|
170
|
+
n_rows=profile.n_rows,
|
|
171
|
+
n_cols=profile.n_cols,
|
|
172
|
+
analyses=analyses,
|
|
173
|
+
session_critiques=session,
|
|
174
|
+
verdict=verdict,
|
|
175
|
+
cannot_conclude=cannot,
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _worse(a: Verdict, b: Verdict) -> Verdict:
|
|
180
|
+
return a if _VERDICT_RANK[a] >= _VERDICT_RANK[b] else b
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _worst(verdicts: Iterable[Verdict]) -> Verdict:
|
|
184
|
+
worst = Verdict.defensible
|
|
185
|
+
for v in verdicts:
|
|
186
|
+
worst = _worse(worst, v)
|
|
187
|
+
return worst
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
_CANNOT_TEMPLATES = {
|
|
191
|
+
"confounding.causal_language": "Cannot conclude causation: {evidence} {remedy}",
|
|
192
|
+
"power.underpowered": "Cannot conclude there is no effect: {evidence}",
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _cannot_conclude(residual: list[Critique], result: StatResult | None) -> list[str]:
|
|
197
|
+
lines = []
|
|
198
|
+
for c in residual:
|
|
199
|
+
if c.severity not in (Severity.high, Severity.critical):
|
|
200
|
+
continue
|
|
201
|
+
template = _CANNOT_TEMPLATES.get(c.check_id)
|
|
202
|
+
if template:
|
|
203
|
+
lines.append(template.format(evidence=c.evidence, remedy=c.remedy))
|
|
204
|
+
else:
|
|
205
|
+
lines.append(
|
|
206
|
+
f"Cannot rely on this result yet: {c.title.lower()} ({c.evidence})"
|
|
207
|
+
)
|
|
208
|
+
return lines
|
statskeptic/cli.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Command-line surface. A thin adapter over `analyze`: it reads the CSV, formats the
|
|
2
|
+
report, and turns the verdict into an exit code so the tool is scriptable as a gate.
|
|
3
|
+
|
|
4
|
+
Exit codes follow the verdict, not just success/failure, because "the data cannot say"
|
|
5
|
+
is a distinct, useful signal from "here is a defensible answer" and from "I could not
|
|
6
|
+
even map your question".
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from .agent import analyze
|
|
16
|
+
from .critique.models import Verdict
|
|
17
|
+
from .errors import AnalysisError
|
|
18
|
+
from .plan.models import PlanHints
|
|
19
|
+
|
|
20
|
+
# BSD sysexits: 64 is "you called me wrong", 70 is "I broke trying".
|
|
21
|
+
EXIT_USAGE = 64
|
|
22
|
+
EXIT_INTERNAL = 70
|
|
23
|
+
|
|
24
|
+
_VERDICT_EXIT = {
|
|
25
|
+
Verdict.defensible: 0,
|
|
26
|
+
Verdict.defensible_with_caveats: 0,
|
|
27
|
+
Verdict.cannot_conclude: 2,
|
|
28
|
+
Verdict.declined: 3,
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
33
|
+
parser = argparse.ArgumentParser(
|
|
34
|
+
prog="statskeptic",
|
|
35
|
+
description="Analyze a dataset and red-team the conclusion.",
|
|
36
|
+
)
|
|
37
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
38
|
+
|
|
39
|
+
a = sub.add_parser("analyze", help="analyze a CSV against a question")
|
|
40
|
+
a.add_argument("data", type=Path, help="path to a CSV file")
|
|
41
|
+
a.add_argument("-q", "--question", required=True, help="the question to answer")
|
|
42
|
+
a.add_argument("--json", action="store_true", help="emit the report as JSON")
|
|
43
|
+
a.add_argument("--outcome", help="name the outcome column (overrides matching)")
|
|
44
|
+
a.add_argument("--group", "--by", dest="group", help="name the grouping column")
|
|
45
|
+
a.add_argument(
|
|
46
|
+
"--predictors", help="comma-separated predictor columns for regression"
|
|
47
|
+
)
|
|
48
|
+
a.add_argument("--alpha", type=float, default=0.05, help="significance level")
|
|
49
|
+
a.add_argument(
|
|
50
|
+
"--quiet", action="store_true", help="suppress the report body; exit code only"
|
|
51
|
+
)
|
|
52
|
+
return parser
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def main(argv: list[str] | None = None) -> int:
|
|
56
|
+
parser = _build_parser()
|
|
57
|
+
args = parser.parse_args(argv)
|
|
58
|
+
|
|
59
|
+
if not args.data.exists():
|
|
60
|
+
print(f"error: file not found: {args.data}", file=sys.stderr)
|
|
61
|
+
return EXIT_USAGE
|
|
62
|
+
|
|
63
|
+
predictors = args.predictors.split(",") if args.predictors else None
|
|
64
|
+
hints = PlanHints(outcome=args.outcome, group=args.group, predictors=predictors)
|
|
65
|
+
|
|
66
|
+
try:
|
|
67
|
+
report = analyze(args.data, args.question, hints=hints, alpha=args.alpha)
|
|
68
|
+
except AnalysisError as exc:
|
|
69
|
+
print(f"analysis failed: {exc}", file=sys.stderr)
|
|
70
|
+
return EXIT_INTERNAL
|
|
71
|
+
except (ValueError, KeyError) as exc:
|
|
72
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
73
|
+
return EXIT_USAGE
|
|
74
|
+
|
|
75
|
+
if not args.quiet:
|
|
76
|
+
print(report.to_json() if args.json else report.explain())
|
|
77
|
+
|
|
78
|
+
return _VERDICT_EXIT[report.verdict]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
if __name__ == "__main__":
|
|
82
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from .engine import CritiqueContext, register_check, run_critique
|
|
2
|
+
from .models import Critique, CritiqueCategory, FixAction, RevisionStep, Verdict
|
|
3
|
+
from .revise import (
|
|
4
|
+
RevisionOutcome,
|
|
5
|
+
apply_holm,
|
|
6
|
+
decide_verdict,
|
|
7
|
+
multiple_comparisons_critique,
|
|
8
|
+
revise,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"Critique",
|
|
13
|
+
"CritiqueCategory",
|
|
14
|
+
"CritiqueContext",
|
|
15
|
+
"FixAction",
|
|
16
|
+
"RevisionOutcome",
|
|
17
|
+
"RevisionStep",
|
|
18
|
+
"Verdict",
|
|
19
|
+
"apply_holm",
|
|
20
|
+
"decide_verdict",
|
|
21
|
+
"multiple_comparisons_critique",
|
|
22
|
+
"register_check",
|
|
23
|
+
"revise",
|
|
24
|
+
"run_critique",
|
|
25
|
+
]
|