statskeptic 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,43 @@
1
+ """statskeptic - a data-analysis agent that distrusts its own conclusions.
2
+
3
+ Point it at a dataset and a question. It plans an analysis, runs it with real,
4
+ vetted statistical code (never a number invented by a language model), then attacks
5
+ its own result against a methodological rubric - assumption violations, multiple
6
+ comparisons, confounding, underpowered samples, leakage - and reports plainly what
7
+ it found and, just as importantly, what it cannot conclude.
8
+
9
+ The public API grows as each capability lands; for now this exposes the package
10
+ version so an editable install resolves cleanly.
11
+ """
12
+
13
+ from importlib.metadata import PackageNotFoundError, version
14
+
15
+ from .agent import analyze
16
+ from .critique.models import Critique, CritiqueCategory, Verdict
17
+ from .errors import AnalysisError, StatskepticError
18
+ from .plan.models import AnalysisPlan, Decline, Method, PlanHints, QuestionType
19
+ from .profile.models import DataProfile
20
+ from .report.models import Analysis, Report
21
+
22
+ try:
23
+ __version__ = version("statskeptic")
24
+ except PackageNotFoundError: # pragma: no cover - only during local source runs
25
+ __version__ = "0.0.0"
26
+
27
+ __all__ = [
28
+ "Analysis",
29
+ "AnalysisError",
30
+ "AnalysisPlan",
31
+ "Critique",
32
+ "CritiqueCategory",
33
+ "DataProfile",
34
+ "Decline",
35
+ "Method",
36
+ "PlanHints",
37
+ "QuestionType",
38
+ "Report",
39
+ "StatskepticError",
40
+ "Verdict",
41
+ "__version__",
42
+ "analyze",
43
+ ]
statskeptic/agent.py ADDED
@@ -0,0 +1,208 @@
1
+ """The end-to-end loop: profile -> plan -> execute -> critique -> revise -> report.
2
+
3
+ `analyze` is the one public entry point. It owns the orchestration and the honesty
4
+ decisions: when the planner declines, when a screen needs a multiplicity correction,
5
+ and when surviving objections mean the only honest verdict is "cannot conclude".
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Iterable
11
+ from pathlib import Path
12
+
13
+ import numpy as np
14
+ import pandas as pd
15
+ from pandas.api import types as pdt
16
+
17
+ from .critique.engine import CritiqueContext, run_critique
18
+ from .critique.models import Critique, Verdict
19
+ from .critique.revise import (
20
+ apply_holm,
21
+ decide_verdict,
22
+ multiple_comparisons_critique,
23
+ revise,
24
+ )
25
+ from .execution import execute
26
+ from .loader import read_table
27
+ from .plan.models import AnalysisPlan, Decline, Method, PlanHints, QuestionType
28
+ from .plan.planner import plan as make_plan
29
+ from .profile.models import DataProfile
30
+ from .profile.profiler import build_profile
31
+ from .report.models import Analysis, Report
32
+ from .stats.results import Severity, StatResult
33
+
34
+ _VERDICT_RANK = {
35
+ Verdict.defensible: 0,
36
+ Verdict.defensible_with_caveats: 1,
37
+ Verdict.cannot_conclude: 2,
38
+ Verdict.declined: 2,
39
+ }
40
+
41
+
42
+ def analyze(
43
+ data: pd.DataFrame | str | Path,
44
+ question: str,
45
+ *,
46
+ hints: PlanHints | None = None,
47
+ alpha: float = 0.05,
48
+ ) -> Report:
49
+ if not 0.0 < alpha < 1.0:
50
+ raise ValueError(f"alpha must be between 0 and 1 (exclusive); got {alpha}")
51
+ df = data if isinstance(data, pd.DataFrame) else read_table(data)
52
+ df = _coerce_dates(df)
53
+ # Treat infinities as missing for the whole pipeline: the profiler then counts them
54
+ # as missing data and every downstream routine drops them with the NaNs.
55
+ df = df.replace([np.inf, -np.inf], np.nan)
56
+ profile = build_profile(df)
57
+ planned = make_plan(question, profile, hints)
58
+
59
+ if isinstance(planned, Decline):
60
+ return Report(
61
+ question=question,
62
+ n_rows=profile.n_rows,
63
+ n_cols=profile.n_cols,
64
+ verdict=Verdict.declined,
65
+ declined=planned,
66
+ )
67
+
68
+ if planned.question_type == QuestionType.screen:
69
+ return _run_screen(planned, df, profile, question, alpha)
70
+ return _run_single(planned, df, profile, question, alpha)
71
+
72
+
73
+ # Explicit formats only. Inferring dates from arbitrary strings is where pandas misreads
74
+ # categorical codes as dates and emits warnings; an exact format either matches every
75
+ # value in a column or that column is left exactly as it was.
76
+ _DATE_FORMATS = ("%Y-%m-%d", "%Y-%m-%dT%H:%M:%S", "%Y-%m-%d %H:%M:%S", "%Y/%m/%d")
77
+
78
+
79
+ def _coerce_dates(df: pd.DataFrame) -> pd.DataFrame:
80
+ out = df
81
+ for col in df.columns:
82
+ series = df[col]
83
+ # Only string columns are candidates (pandas 3 reads CSV text as the str dtype,
84
+ # not object). Numeric, datetime, and boolean columns are left untouched.
85
+ if not (pdt.is_object_dtype(series) or pdt.is_string_dtype(series)):
86
+ continue
87
+ non_null = series.dropna()
88
+ if non_null.empty:
89
+ continue
90
+ for fmt in _DATE_FORMATS:
91
+ try:
92
+ pd.to_datetime(non_null, format=fmt)
93
+ except (ValueError, TypeError):
94
+ continue
95
+ if out is df:
96
+ out = df.copy()
97
+ out[col] = pd.to_datetime(series, format=fmt, errors="coerce")
98
+ break
99
+ return out
100
+
101
+
102
+ def _run_single(
103
+ planned: AnalysisPlan,
104
+ df: pd.DataFrame,
105
+ profile: DataProfile,
106
+ question: str,
107
+ alpha: float,
108
+ ) -> Report:
109
+ result = execute(planned, df, alpha)
110
+ outcome = revise(planned, result, profile, df, alpha)
111
+ verdict = decide_verdict(outcome.residual)
112
+ analysis = Analysis(
113
+ plan=outcome.plan,
114
+ result=outcome.result,
115
+ revision_steps=outcome.steps,
116
+ critiques=outcome.residual,
117
+ verdict=verdict,
118
+ )
119
+ return Report(
120
+ question=question,
121
+ n_rows=profile.n_rows,
122
+ n_cols=profile.n_cols,
123
+ analyses=[analysis],
124
+ verdict=verdict,
125
+ cannot_conclude=_cannot_conclude(outcome.residual, outcome.result),
126
+ )
127
+
128
+
129
+ def _run_screen(
130
+ planned: AnalysisPlan,
131
+ df: pd.DataFrame,
132
+ profile: DataProfile,
133
+ question: str,
134
+ alpha: float,
135
+ ) -> Report:
136
+ assert planned.outcome
137
+ subplans = [
138
+ AnalysisPlan(
139
+ question=question,
140
+ question_type=QuestionType.association,
141
+ method=Method.pearson,
142
+ outcome=planned.outcome,
143
+ predictors=[cand],
144
+ )
145
+ for cand in planned.candidates
146
+ ]
147
+ results = [execute(sp, df, alpha) for sp in subplans]
148
+ per_critiques = [
149
+ run_critique(CritiqueContext(sp, r, profile, df))
150
+ for sp, r in zip(subplans, results, strict=True)
151
+ ]
152
+
153
+ mc = multiple_comparisons_critique(results, planned.outcome, alpha)
154
+ session: list[Critique] = []
155
+ if mc is not None:
156
+ results = apply_holm(results, alpha)
157
+ session = [mc]
158
+
159
+ analyses = [
160
+ Analysis(plan=sp, result=r, critiques=crits, verdict=decide_verdict(crits))
161
+ for sp, r, crits in zip(subplans, results, per_critiques, strict=True)
162
+ ]
163
+ verdict = _worst(a.verdict for a in analyses)
164
+ if mc is not None:
165
+ verdict = _worse(verdict, Verdict.defensible_with_caveats)
166
+
167
+ cannot = [line for crits in per_critiques for line in _cannot_conclude(crits, None)]
168
+ return Report(
169
+ question=question,
170
+ n_rows=profile.n_rows,
171
+ n_cols=profile.n_cols,
172
+ analyses=analyses,
173
+ session_critiques=session,
174
+ verdict=verdict,
175
+ cannot_conclude=cannot,
176
+ )
177
+
178
+
179
+ def _worse(a: Verdict, b: Verdict) -> Verdict:
180
+ return a if _VERDICT_RANK[a] >= _VERDICT_RANK[b] else b
181
+
182
+
183
+ def _worst(verdicts: Iterable[Verdict]) -> Verdict:
184
+ worst = Verdict.defensible
185
+ for v in verdicts:
186
+ worst = _worse(worst, v)
187
+ return worst
188
+
189
+
190
+ _CANNOT_TEMPLATES = {
191
+ "confounding.causal_language": "Cannot conclude causation: {evidence} {remedy}",
192
+ "power.underpowered": "Cannot conclude there is no effect: {evidence}",
193
+ }
194
+
195
+
196
+ def _cannot_conclude(residual: list[Critique], result: StatResult | None) -> list[str]:
197
+ lines = []
198
+ for c in residual:
199
+ if c.severity not in (Severity.high, Severity.critical):
200
+ continue
201
+ template = _CANNOT_TEMPLATES.get(c.check_id)
202
+ if template:
203
+ lines.append(template.format(evidence=c.evidence, remedy=c.remedy))
204
+ else:
205
+ lines.append(
206
+ f"Cannot rely on this result yet: {c.title.lower()} ({c.evidence})"
207
+ )
208
+ return lines
statskeptic/cli.py ADDED
@@ -0,0 +1,82 @@
1
+ """Command-line surface. A thin adapter over `analyze`: it reads the CSV, formats the
2
+ report, and turns the verdict into an exit code so the tool is scriptable as a gate.
3
+
4
+ Exit codes follow the verdict, not just success/failure, because "the data cannot say"
5
+ is a distinct, useful signal from "here is a defensible answer" and from "I could not
6
+ even map your question".
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import sys
13
+ from pathlib import Path
14
+
15
+ from .agent import analyze
16
+ from .critique.models import Verdict
17
+ from .errors import AnalysisError
18
+ from .plan.models import PlanHints
19
+
20
+ # BSD sysexits: 64 is "you called me wrong", 70 is "I broke trying".
21
+ EXIT_USAGE = 64
22
+ EXIT_INTERNAL = 70
23
+
24
+ _VERDICT_EXIT = {
25
+ Verdict.defensible: 0,
26
+ Verdict.defensible_with_caveats: 0,
27
+ Verdict.cannot_conclude: 2,
28
+ Verdict.declined: 3,
29
+ }
30
+
31
+
32
+ def _build_parser() -> argparse.ArgumentParser:
33
+ parser = argparse.ArgumentParser(
34
+ prog="statskeptic",
35
+ description="Analyze a dataset and red-team the conclusion.",
36
+ )
37
+ sub = parser.add_subparsers(dest="command", required=True)
38
+
39
+ a = sub.add_parser("analyze", help="analyze a CSV against a question")
40
+ a.add_argument("data", type=Path, help="path to a CSV file")
41
+ a.add_argument("-q", "--question", required=True, help="the question to answer")
42
+ a.add_argument("--json", action="store_true", help="emit the report as JSON")
43
+ a.add_argument("--outcome", help="name the outcome column (overrides matching)")
44
+ a.add_argument("--group", "--by", dest="group", help="name the grouping column")
45
+ a.add_argument(
46
+ "--predictors", help="comma-separated predictor columns for regression"
47
+ )
48
+ a.add_argument("--alpha", type=float, default=0.05, help="significance level")
49
+ a.add_argument(
50
+ "--quiet", action="store_true", help="suppress the report body; exit code only"
51
+ )
52
+ return parser
53
+
54
+
55
+ def main(argv: list[str] | None = None) -> int:
56
+ parser = _build_parser()
57
+ args = parser.parse_args(argv)
58
+
59
+ if not args.data.exists():
60
+ print(f"error: file not found: {args.data}", file=sys.stderr)
61
+ return EXIT_USAGE
62
+
63
+ predictors = args.predictors.split(",") if args.predictors else None
64
+ hints = PlanHints(outcome=args.outcome, group=args.group, predictors=predictors)
65
+
66
+ try:
67
+ report = analyze(args.data, args.question, hints=hints, alpha=args.alpha)
68
+ except AnalysisError as exc:
69
+ print(f"analysis failed: {exc}", file=sys.stderr)
70
+ return EXIT_INTERNAL
71
+ except (ValueError, KeyError) as exc:
72
+ print(f"error: {exc}", file=sys.stderr)
73
+ return EXIT_USAGE
74
+
75
+ if not args.quiet:
76
+ print(report.to_json() if args.json else report.explain())
77
+
78
+ return _VERDICT_EXIT[report.verdict]
79
+
80
+
81
+ if __name__ == "__main__":
82
+ raise SystemExit(main())
@@ -0,0 +1,25 @@
1
+ from .engine import CritiqueContext, register_check, run_critique
2
+ from .models import Critique, CritiqueCategory, FixAction, RevisionStep, Verdict
3
+ from .revise import (
4
+ RevisionOutcome,
5
+ apply_holm,
6
+ decide_verdict,
7
+ multiple_comparisons_critique,
8
+ revise,
9
+ )
10
+
11
+ __all__ = [
12
+ "Critique",
13
+ "CritiqueCategory",
14
+ "CritiqueContext",
15
+ "FixAction",
16
+ "RevisionOutcome",
17
+ "RevisionStep",
18
+ "Verdict",
19
+ "apply_holm",
20
+ "decide_verdict",
21
+ "multiple_comparisons_critique",
22
+ "register_check",
23
+ "revise",
24
+ "run_critique",
25
+ ]