goad-toolkit 0.2.5__tar.gz → 0.2.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511/.claude/settings.local.json +7 -0
  2. goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511/CHANGELOG.md +152 -0
  3. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/README.md +30 -0
  4. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/docs/04-five-families.md +16 -3
  5. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/docs/06-models-and-residuals.md +44 -0
  6. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/docs/08-api-reference.md +47 -0
  7. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/docs/09-analysis-method.md +3 -1
  8. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/goad_mcp.py +2 -2
  9. goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511/img/null-distribution.png +0 -0
  10. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/pyproject.toml +1 -1
  11. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/src/goad_toolkit/analytics.py +159 -4
  12. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/src/goad_toolkit/visualizer.py +299 -3
  13. goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511/tests/test_nulldistribution.py +156 -0
  14. goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511/tests/test_visualizer.py +383 -0
  15. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/uv.lock +1 -1
  16. goad_toolkit-0.2.8/.remember/archive.md +3 -0
  17. goad_toolkit-0.2.8/.remember/logs/memory-2026-08-12.log +28 -0
  18. goad_toolkit-0.2.8/.remember/recent.md +7 -0
  19. goad_toolkit-0.2.8/.remember/tmp/capture-alive +1 -0
  20. goad_toolkit-0.2.8/.remember/tmp/capture-alive.d/90574e63-97cd-4c11-87e5-23b656c56fd9 +0 -0
  21. goad_toolkit-0.2.8/.remember/tmp/last-save-ts +1 -0
  22. goad_toolkit-0.2.8/.remember/tmp/post-tool-ran +0 -0
  23. goad_toolkit-0.2.8/.remember/tmp/save-session.pid +1 -0
  24. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/tmp/session-slug +1 -1
  25. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/MCP_SERVER.md +7 -7
  26. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/PKG-INFO +31 -1
  27. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/README.md +30 -0
  28. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/docs/02-pipelines.md +31 -32
  29. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/docs/03-plot-composition.md +43 -3
  30. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/docs/04-five-families.md +40 -11
  31. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/docs/05-distributions.md +52 -7
  32. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/docs/06-models-and-residuals.md +44 -0
  33. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/docs/08-api-reference.md +115 -5
  34. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/docs/09-analysis-method.md +52 -1
  35. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/goad_mcp.py +77 -12
  36. goad_toolkit-0.2.8/img/null-distribution.png +0 -0
  37. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/pyproject.toml +2 -1
  38. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/src/goad_toolkit/analytics.py +230 -34
  39. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/src/goad_toolkit/datatransforms.py +87 -1
  40. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/src/goad_toolkit/distributions.py +16 -0
  41. goad_toolkit-0.2.8/src/goad_toolkit/visualizer.py +1439 -0
  42. goad_toolkit-0.2.8/tests/test_datatransforms.py +167 -0
  43. goad_toolkit-0.2.8/tests/test_distributions.py +195 -0
  44. goad_toolkit-0.2.8/tests/test_nulldistribution.py +156 -0
  45. goad_toolkit-0.2.8/tests/test_visualizer.py +666 -0
  46. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/uv.lock +48 -1
  47. goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/visualizer.py +0 -545
  48. goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511/tests/test_distributions.py +0 -72
  49. goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511/tests/test_visualizer.py +0 -113
  50. goad_toolkit-0.2.5/.remember/tmp/capture-alive +0 -1
  51. goad_toolkit-0.2.5/.remember/tmp/last-save-ts +0 -1
  52. goad_toolkit-0.2.5/.remember/tmp/save-session.pid +0 -1
  53. goad_toolkit-0.2.5/CHANGELOG.md +0 -37
  54. goad_toolkit-0.2.5/tests/test_visualizer.py +0 -200
  55. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/settings.local.json +0 -0
  56. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/.gitignore +0 -0
  57. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/.python-version +0 -0
  58. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/MCP_SERVER.md +0 -0
  59. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/demo/linear.py +0 -0
  60. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/docs/01-goal-oriented-analysis.md +0 -0
  61. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/docs/02-pipelines.md +0 -0
  62. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/docs/03-plot-composition.md +0 -0
  63. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/docs/05-distributions.md +0 -0
  64. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/docs/07-visual-critique.md +0 -0
  65. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/docs/10-teaching-path.md +0 -0
  66. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/docs/README.md +0 -0
  67. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/img/distribution_fit.png +0 -0
  68. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/img/goaded.png +0 -0
  69. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/img/linear_results.png +0 -0
  70. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/img/residuals.png +0 -0
  71. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/img/zscores.png +0 -0
  72. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/__init__.py +0 -0
  73. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/cli.py +0 -0
  74. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/config.py +0 -0
  75. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/dataprocessor.py +0 -0
  76. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/src/goad_toolkit/datatransforms.py +0 -0
  77. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/src/goad_toolkit/distributions.py +0 -0
  78. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/filehandler.py +0 -0
  79. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/models.py +0 -0
  80. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/tests/test_cli.py +0 -0
  81. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/tests/test_datatransforms.py +0 -0
  82. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8/.claude/worktrees/quizzical-solomon-b62511}/tests/test_distributions.py +0 -0
  83. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.claude/worktrees/quizzical-solomon-b62511/tests/test_filehandler.py +0 -0
  84. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.gitignore +0 -0
  85. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.python-version +0 -0
  86. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/.gitignore +0 -0
  87. /goad_toolkit-0.2.5/.remember/logs/autonomous/save-195516.log → /goad_toolkit-0.2.8/.remember/logs/autonomous/save-104759.log +0 -0
  88. /goad_toolkit-0.2.5/.remember/logs/autonomous/save-195720.log → /goad_toolkit-0.2.8/.remember/logs/autonomous/save-195516.log +0 -0
  89. /goad_toolkit-0.2.5/.remember/logs/autonomous/save-195933.log → /goad_toolkit-0.2.8/.remember/logs/autonomous/save-195720.log +0 -0
  90. /goad_toolkit-0.2.5/.remember/logs/autonomous/save-200831.log → /goad_toolkit-0.2.8/.remember/logs/autonomous/save-195933.log +0 -0
  91. /goad_toolkit-0.2.5/.remember/logs/autonomous/save-201034.log → /goad_toolkit-0.2.8/.remember/logs/autonomous/save-200831.log +0 -0
  92. /goad_toolkit-0.2.5/.remember/logs/hook-errors.log → /goad_toolkit-0.2.8/.remember/logs/autonomous/save-201034.log +0 -0
  93. /goad_toolkit-0.2.5/.remember/tmp/capture-alive.d/90574e63-97cd-4c11-87e5-23b656c56fd9 → /goad_toolkit-0.2.8/.remember/logs/hook-errors.log +0 -0
  94. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/logs/memory-2026-08-10.log +0 -0
  95. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/now.md +0 -0
  96. /goad_toolkit-0.2.5/.remember/tmp/post-tool-ran → /goad_toolkit-0.2.8/.remember/tmp/capture-alive.d/16b4788a-5507-4216-98ec-06bac13a9aba +0 -0
  97. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/tmp/case-divergence +0 -0
  98. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/tmp/last-ndc.ts +0 -0
  99. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/tmp/last-save.json +0 -0
  100. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/.remember/tmp/now-day +0 -0
  101. /goad_toolkit-0.2.5/.remember/today-2026-08-10.md → /goad_toolkit-0.2.8/.remember/today-2026-08-10.done.md +0 -0
  102. {goad_toolkit-0.2.5/.claude/worktrees/quizzical-solomon-b62511 → goad_toolkit-0.2.8}/CHANGELOG.md +0 -0
  103. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/demo/linear.py +0 -0
  104. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/docs/01-goal-oriented-analysis.md +0 -0
  105. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/docs/07-visual-critique.md +0 -0
  106. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/docs/10-teaching-path.md +0 -0
  107. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/docs/README.md +0 -0
  108. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/img/distribution_fit.png +0 -0
  109. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/img/goaded.png +0 -0
  110. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/img/linear_results.png +0 -0
  111. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/img/residuals.png +0 -0
  112. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/img/zscores.png +0 -0
  113. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/src/goad_toolkit/__init__.py +0 -0
  114. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/src/goad_toolkit/cli.py +0 -0
  115. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/src/goad_toolkit/config.py +0 -0
  116. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/src/goad_toolkit/dataprocessor.py +0 -0
  117. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/src/goad_toolkit/filehandler.py +0 -0
  118. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/src/goad_toolkit/models.py +0 -0
  119. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/tests/test_cli.py +0 -0
  120. {goad_toolkit-0.2.5 → goad_toolkit-0.2.8}/tests/test_filehandler.py +0 -0
@@ -0,0 +1,7 @@
1
+ {
2
+ "permissions": {
3
+ "allow": [
4
+ "Bash(cd /Users/rgrouls/code/courses/goad_toolkit && cat pyproject.toml)"
5
+ ]
6
+ }
7
+ }
@@ -0,0 +1,152 @@
1
+ # Changelog
2
+
3
+ ## 0.2.7
4
+
5
+ ### Added
6
+
7
+ - `ScatterPlot`, `RegPlot` and `CorrelationHeatmap`: the relation plots for two-variable
8
+ questions, in their order of use — a scatter first, a fit once the scatter has shown which
9
+ kind, a correlation matrix when there are too many pairs to plot by hand.
10
+ - `ScatterPlot` defaults `alpha` below 1, since overplotting is the standard way a scatter
11
+ misreports its own density.
12
+ - `RegPlot` wraps `sns.regplot`, which is axis-level and so can draw onto a layer's axis.
13
+ `scatter=False` lays a fit over an existing `ScatterPlot`; `lowess` together with
14
+ `order != 1` raises, since both answer "what shape" and only one can apply.
15
+ - `CorrelationHeatmap` pins the colour scale to `[-1, 1]` centred on 0 and defaults to a
16
+ diverging colormap, so the sign of a correlation stays readable. Non-numeric columns are
17
+ dropped, and a frame with nothing numeric left raises rather than drawing an empty grid.
18
+ - docs/04 §4.4 covers the three and their layering; docs/08 §8.10 documents the seaborn
19
+ axis-label override — a seaborn call with `x=`/`y=` writes the column names over
20
+ `PlotSettings`.
21
+
22
+ ## 0.2.6
23
+
24
+ ### Added
25
+
26
+ - `NullDistribution` and `NullResult`: shuffle one column, re-measure, and compare the
27
+ observed value against the cloud that produces. The statistic stays the caller's —
28
+ `NullDistribution` takes a callable and ships no ready-made statistics.
29
+ - `NullResult.p_value` counts rather than tests. Both one-sided counts include the observed
30
+ arrangement, so the smallest value it can return is `1 / (n_iter + 1)`.
31
+ - `NullPlot`: the null as a grey `HistogramPlot` with the observed value marked, its value
32
+ and p-value in the legend, stated as a bound (`p < …`) once the count floor is reached,
33
+ since an exact number there would overstate what a thousand shuffles support.
34
+ - `VerticalLine`, the numeric counterpart to `VerticalDate` and the layer `NullPlot`
35
+ composes.
36
+ - docs/06 §6.7 carries the prose, including the two things the shuffle test does not defend
37
+ against: the garden of forking paths, and a confounder that shuffling breaks along with
38
+ the association.
39
+
40
+ ## 0.2.5
41
+
42
+ ### Added
43
+
44
+ - `RegexFeature` transform: one new column derived from a regular expression over a text
45
+ column, in `count` (occurrences per row), `has` (bool) or `extract` (first capture group,
46
+ NaN where nothing matched) mode. The output column is named by `feature`, because
47
+ `Pipeline.add(name=...)` already claims `name` for the step.
48
+ - `mode="extract"` reports its match rate through `loguru`: it is the one mode that fails
49
+ silently, filling with NaN where a wrong pattern in the other two gives a visibly zero or
50
+ all-`False` column.
51
+ - docs/02 §2.5 documents the shipped transform with its mode table and coverage log; §2.4
52
+ keeps `IsWeekend` as the write-your-own example.
53
+
54
+ ## 0.2.4
55
+
56
+ ### Added
57
+
58
+ - `bernoulli`, `binomial`, `nbinom` and `beta` are default registry families. `pareto` is
59
+ deliberately not among them: lesson 4 has the student register it as the extensibility
60
+ exercise, and the class docstring shows the call.
61
+ - `QQPlot`: sample quantiles against a frozen distribution's theoretical quantiles with a
62
+ `y = x` reference line, for the tail mismatches a `PlotFits` histogram is too
63
+ bulk-dominated to show.
64
+ - `ECDFPlot`: one or two empirical CDFs, bin-free.
65
+ - `fit_table`: turns a `fit()` result list into a dataframe ranked by log-likelihood — one
66
+ row per family with params, log-likelihood, KS statistic and p-value, and which criteria
67
+ it won. `FailedFit` entries carry their message and sort last, since a missing likelihood
68
+ is not a small one.
69
+
70
+ ### Fixed
71
+
72
+ - Log-likelihood uses `logpmf` for discrete families and `logpdf` for continuous ones, so
73
+ every discrete family reports a real likelihood and the comparison between them means
74
+ something.
75
+ - The lower bound for a single-parameter discrete family is `1e-3`, low enough for a
76
+ probability (bernoulli's `p`) as well as a count-scale rate (poisson's lambda).
77
+ - `best_likelihood` and `best_ks` are marked independently: a KS p-value that never clears
78
+ the `p > 0` threshold — routine for a discrete family on a large, tied sample — leaves
79
+ the likelihood winner marked. An unknown `criterion` raises `ValueError` whether or not
80
+ a winner was found.
81
+
82
+ ## 0.2.3
83
+
84
+ ### Added
85
+
86
+ - `TimeFeatures` transform: derives `date`, `hour`, `day_name`, `isoweek` and `year_week`
87
+ from a timestamp column, with a `features=[...]` selector. The default is the three
88
+ lesson 1 relies on, and an unknown feature name raises `ValueError`.
89
+ - `DecomposePlot`: `seasonal_decompose` as the four-panel observed/trend/seasonal/residual
90
+ view, composing a `LinePlot` per panel.
91
+ - `ACFPlot`: autocorrelations as bars with a Bartlett confidence band. NaN input raises,
92
+ since `statsmodels.acf` returns an all-NaN result instead of raising.
93
+
94
+ ### Changed
95
+
96
+ - `statsmodels` is a dependency.
97
+
98
+ ## 0.2.2
99
+
100
+ ### Added
101
+
102
+ - Analysis-method rows on exploratory versus confirmatory questions and on mechanism: which
103
+ of the two a question is decides what would count as evidence, and a significant result
104
+ with no nameable mechanism points at a confounder rather than a discovery.
105
+ - Analysis-method rows on the unit of analysis — the unit a claim is about against the unit
106
+ of a row, and the number of independent units behind each group being compared — with
107
+ §9.3 naming pseudoreplication and showing the n=4000/n=5 bar chart.
108
+ - An analysis-method row on how many comparisons were considered, with §9.6 on the garden
109
+ of forking paths: the arithmetic of fifteen tries at a one-in-twenty threshold, and the
110
+ three defences (write the question down, split the data, report the count).
111
+ - A visual-critique row separating "A differs from B" from "A is the highest", since every
112
+ set of numbers has a maximum and a ranking claim needs the ranking to be stable.
113
+ - A counting step at the head of the distribution method: how many observations, and how
114
+ many independent ones.
115
+
116
+ These land in docs/09-analysis-method.md and in the matching `goad_mcp` checklists.
117
+
118
+ ## 0.2.1
119
+
120
+ ### Breaking
121
+
122
+ - **`FileConfig` gained `date_column` and `date_index`.** `FileHandler.load` reads csv and
123
+ parquet (reader picked from the suffix), parses `date_column` as dates and, when
124
+ `date_index` is true, moves it into the index. The defaults (`"date"`, `True`) keep the
125
+ covid pipeline working unchanged, but a `FileConfig` subclass that overrode `load` to
126
+ support another shape can now express it as configuration instead. A file whose suffix has
127
+ no reader, or a `date_column` that is not in the frame, raises `ValueError`. Parquet needs
128
+ `pyarrow`, available as the `parquet` extra.
129
+ - **`DistributionRegistry` is a normal class.** Constructing one no longer returns a shared
130
+ process-wide instance, so a registration affects only the registry it was made on. Code
131
+ that registered a family on one registry and relied on a separately constructed
132
+ `DistributionFitter()` picking it up must now pass it: `DistributionFitter(registry)`.
133
+
134
+ ### Added
135
+
136
+ - `goad` console script: fits every family in the registry to one column of a csv or parquet
137
+ file and prints the fits ranked by log-likelihood (`goad data/residuals.csv residual`).
138
+ - `DistributionFitter` accepts an optional `DistributionRegistry`.
139
+ - `filehandler.reader_for`, `filehandler.write_frame` and `filehandler.parse_dates` as
140
+ reusable pieces; `FileHandler.save` writes parquet as well as csv.
141
+ - A pytest suite under `tests/`.
142
+
143
+ ### Fixed
144
+
145
+ - `BarWithDates.build` and `VerticalDate.build` return `(fig, ax)` like every other `build`.
146
+ - Plots draw on `self.ax` / `self.fig` instead of module-level `plt` calls, so a layer placed
147
+ with `plot_on_axes` lands on the axis it was handed rather than on matplotlib's current one.
148
+ - The `goad` console script points at a module that exists.
149
+
150
+ ### Changed
151
+
152
+ - `__version__` is read from the installed package metadata.
@@ -183,6 +183,36 @@ For the [kstest](https://docs.scipy.org/doc/scipy/reference/generated/scipy.stat
183
183
  The plots are sorted by log-likelihood, which means there is no good fit with a distribution in this case.
184
184
  ![residuals](img/distribution_fit.png)
185
185
 
186
+ ### 🎲 Is it real? The shuffle test
187
+
188
+ Shuffle the labels, measure again, and see where the real number lands. No test, no
189
+ assumptions, no table of critical values — just a picture of what your measurement does when
190
+ the label means nothing:
191
+
192
+ ```python
193
+ from goad_toolkit.analytics import NullDistribution
194
+ from goad_toolkit.visualizer import NullPlot, PlotSettings
195
+
196
+ def gap(frame):
197
+ means = frame.groupby("is_bot")["length"].mean()
198
+ return means[True] - means[False]
199
+
200
+ result = NullDistribution(gap, n_iter=2000, seed=1).run(data, label="is_bot")
201
+ settings = PlotSettings(
202
+ title="Mean message length: bots minus humans",
203
+ xlabel="difference in mean length (characters)",
204
+ ylabel="density under shuffled labels",
205
+ )
206
+ NullPlot(settings).plot(result=result)
207
+ ```
208
+
209
+ ![null distribution](img/null-distribution.png)
210
+
211
+ The grey cloud is the statistic under shuffled labels; the line is what the real data did.
212
+ Writing the statistic is your job — that is the claim. See
213
+ [Models and residuals](docs/06-models-and-residuals.md) §6.7 for what the p-value can and
214
+ cannot carry.
215
+
186
216
  ### 🧩 Extending with Custom Distributions
187
217
 
188
218
  You can easily register new distributions:
@@ -149,9 +149,22 @@ to look.
149
149
  are not independent evidence. This is exactly why regularisation is the better tool for
150
150
  "which variables matter" — see [Models and residuals](06-models-and-residuals.md).
151
151
 
152
- **In GOAD:** the scatter and the fit are `BasePlot` subclasses; the interesting work is in
153
- [Models and residuals](06-models-and-residuals.md), because a relation you can state as a
154
- model is a relation whose failures you can inspect.
152
+ **In GOAD:** `ScatterPlot`, `RegPlot` and `CorrelationHeatmap`, in that order of use.
153
+
154
+ ```python
155
+ scatter = ScatterPlot(settings)
156
+ scatter.plot(data=data, x="messages", y="length", hue="is_bot")
157
+ scatter.plot_on(RegPlot(settings), data=data, x="messages", y="length", scatter=False)
158
+ ```
159
+
160
+ The layering is the point: the scatter is drawn first and the fit goes on top of it, so the
161
+ choice of `fit_reg`, `order` or `lowess` is made after you have seen the shape rather than
162
+ before. `CorrelationHeatmap` is the entry point when there are too many pairs to plot — it
163
+ tells you which scatter to make next, and nothing more than that.
164
+
165
+ The interesting work is still in [Models and residuals](06-models-and-residuals.md), because
166
+ a relation you can state as a model is a relation whose failures you can inspect — and
167
+ §6.7 is how you check that the relation is not something a shuffle would have produced.
155
168
 
156
169
  ---
157
170
 
@@ -158,6 +158,50 @@ They compose: the residual of a functional model is what you fit a distribution
158
158
  composition is the loop from [Goal oriented analysis](01-goal-oriented-analysis.md) §1.3, and
159
159
  it is where the two halves of this library meet.
160
160
 
161
+ ## 6.7 The shuffle test
162
+
163
+ §6.1 said a model is a statement about what is boring. The most boring statement available is
164
+ *this label explains nothing*, and you can build that model without writing one down: shuffle
165
+ the label column and measure again. Everything else stays as it was — the labels keep their
166
+ proportions, the values keep their shape — and only the pairing between them is destroyed.
167
+ Do it a thousand times and you have a picture of what your measurement does in a world where
168
+ there is nothing to find.
169
+
170
+ ```python
171
+ from goad_toolkit.analytics import NullDistribution
172
+ from goad_toolkit.visualizer import NullPlot, PlotSettings
173
+
174
+ def gap(frame):
175
+ means = frame.groupby("is_bot")["length"].mean()
176
+ return means[True] - means[False]
177
+
178
+ result = NullDistribution(gap, n_iter=2000, seed=1).run(data, label="is_bot")
179
+ NullPlot(PlotSettings(xlabel="difference in mean length")).plot(result=result)
180
+ ```
181
+
182
+ The statistic is yours. `NullDistribution` runs the loop; deciding *what* to measure and
183
+ *which* column to shuffle is the part that cannot be delegated, because it is the claim.
184
+
185
+ Read the picture, not the number. Either the line stands clear of the grey cloud or it sits
186
+ inside it, and that judgement needs no test, no distributional assumption, and no table of
187
+ critical values. A finding sitting in the middle of its own null is not a weak finding; it is
188
+ the null, and the honest write-up says so.
189
+
190
+ Three things worth knowing before you quote a number from it:
191
+
192
+ - **`p_value()` counts, it does not test.** It is the fraction of shuffles that matched or
193
+ beat what you saw, defaulting to two-sided. The observed arrangement is itself one of the
194
+ arrangements, so the count starts at one and the smallest reportable value is
195
+ `1 / (n_iter + 1)`. With 2000 shuffles, "p ≤ 0.001" is the strongest sentence available —
196
+ which is why `NullPlot` writes `≤` on the plot when you are at that floor.
197
+ - **It answers one question.** Run it on the fifteenth thing you measured and it will happily
198
+ report the fifteenth thing as striking. The shuffle test defends against noise, not against
199
+ the garden of forking paths ([Analysis method](09-analysis-method.md) §9.6) — only writing
200
+ the question down first, or confirming on a slice you did not hunt on, does that.
201
+ - **Shuffling the label tests the label.** If a confounder drives both the label and the
202
+ value, shuffling breaks the confounding too, and the test will confirm an association that
203
+ is real but not the mechanism you have in mind.
204
+
161
205
  ---
162
206
 
163
207
  Next: [Visual critique](07-visual-critique.md) — judging the picture you just made.
@@ -218,6 +218,33 @@ winner — it just has no `best_ks` winner.
218
218
  `log_likelihood` descending, with `FailedFit` rows (their `message` kept, metrics `NaN`)
219
219
  sorted last.
220
220
 
221
+ The shuffle test — see [Models and residuals](06-models-and-residuals.md) §6.7:
222
+
223
+ ```python
224
+ @dataclass
225
+ class NullResult:
226
+ observed: float
227
+ null: np.ndarray # the statistic under n_iter shuffles
228
+ n_iter: int
229
+ seed: Optional[int] = None
230
+
231
+ def p_value(self, alternative: str = "two-sided") -> float
232
+ quantile: float # property, where observed sits in the null
233
+
234
+ class NullDistribution:
235
+ def __init__(self, statistic: Callable[[pd.DataFrame], float],
236
+ n_iter: int = 1000, seed: Optional[int] = None)
237
+
238
+ def run(self, data: pd.DataFrame, label: str) -> NullResult
239
+ ```
240
+
241
+ `statistic` takes a frame and returns one number; `run` calls it once on the data as given,
242
+ then once per shuffle of the `label` column. The caller's frame is never modified.
243
+ `alternative` is one of `"two-sided"`, `"greater"`, `"less"`. Both one-sided counts include
244
+ the observed arrangement, so `p_value` bottoms out at `1 / (n_iter + 1)` rather than 0. A
245
+ `label` that is not a column raises `ValueError`, as does a statistic that is not finite on
246
+ the unshuffled data; non-finite values among the shuffles are logged, not raised.
247
+
221
248
  ---
222
249
 
223
250
  ## 8.7 `goad_toolkit.models`
@@ -288,14 +315,19 @@ class BasePlot(ABC):
288
315
  | `LinePlot` | `data, **kwargs` → `sns.lineplot` |
289
316
  | `ComparePlot` | `data, x, y1, y2, **kwargs` |
290
317
  | `VerticalDate` | `date: str, label: str` |
318
+ | `VerticalLine` | `x: float, label=None, color="crimson", linestyle="--", linewidth=2, **kwargs` |
291
319
  | `ComparePlotDate` | `data, x, y1, y2, date, datelabel, **kwargs` |
292
320
  | `BarWithDates` | `data, x, y, interval: int = 1, **kwargs` |
293
321
  | `ResidualPlot` | `data, x, y, date, datelabel, interval: int = 1` |
322
+ | `ScatterPlot` | `data, x, y, alpha=0.6, **kwargs` → `sns.scatterplot` (`hue`, `size`, `style`) |
323
+ | `RegPlot` | `data, x, y, fit_reg=True, lowess=False, order=1, scatter=True, color="crimson", **kwargs` → `sns.regplot` |
324
+ | `CorrelationHeatmap` | `data, columns=None, method="pearson", cmap="vlag", annot=True, fmt=".2f", **kwargs` |
294
325
  | `HistogramPlot` | `data: np.ndarray, bins=None, kde=False, color="skyblue", alpha=0.7, **kwargs` |
295
326
  | `DistPlot` | `distribution, x_range=None, samples=1000, color="crimson", linewidth=2, label=None, **kwargs` |
296
327
  | `ACFPlot` | `data, nlags=40, alpha=0.05, color="steelblue", band_color="gray", **kwargs` |
297
328
  | `QQPlot` | `data, distribution, color="steelblue", line_color="black", alpha=0.6, **kwargs` |
298
329
  | `ECDFPlot` | `data, compare=None, label="data", compare_label="compare", color="steelblue", compare_color="crimson", **kwargs` |
330
+ | `NullPlot` | `result: NullResult, bins=None, alternative="two-sided", color="lightgrey", line_color="crimson", label=None, **kwargs` |
299
331
 
300
332
  ```python
301
333
  class PlotFits(BasePlot):
@@ -316,6 +348,17 @@ axis it was handed.
316
348
  results. `HistogramPlot` uses `stat="density"` so a pdf overlays correctly; `DistPlot` takes a
317
349
  **frozen** distribution and falls back to `pmf` when there is no `pdf`.
318
350
 
351
+ `NullPlot` composes `HistogramPlot` and `VerticalLine`: grey cloud, coloured line. It writes
352
+ the observed value and its p-value into the legend, using `≤` when the p-value is at the
353
+ floor set by `n_iter`. Pass `label` to write your own.
354
+
355
+ `RegPlot` wraps `sns.regplot`, the axes-level counterpart to `sns.lmplot` — `lmplot` builds
356
+ its own figure and cannot draw onto a `BasePlot` axis. `lowess=True` with `order != 1` raises:
357
+ both are answers to "what shape", and only one can apply. Pass `scatter=False` to layer a fit
358
+ over an existing `ScatterPlot`. `CorrelationHeatmap` drops non-numeric columns, raises when
359
+ nothing numeric is left, and pins the colour scale to [-1, 1] centred on 0 — override with
360
+ `vmin` / `vmax` / `center` in `**kwargs` if you must.
361
+
319
362
  ---
320
363
 
321
364
  ## 8.9 `goad_toolkit.cli`
@@ -352,6 +395,10 @@ of `likelihood`, `ks`, `combined`; `--method` is passed to `scipy.stats.fit`.
352
395
  - `ACFPlot` raises `ValueError` if the data contains NaNs, rather than passing them through.
353
396
  `statsmodels.tsa.stattools.acf` does not raise on missing values — it silently returns an
354
397
  all-NaN result — so a decomposition's residual needs `.dropna()` before it reaches `ACFPlot`.
398
+ - A seaborn function called with `x=` and `y=` as column names writes those names onto the
399
+ axes, overwriting the `xlabel` / `ylabel` from `PlotSettings`. `create_figure` runs first,
400
+ seaborn runs second, and second wins. Set the labels on the axis after plotting when the
401
+ column name is not the label you want a reader to see.
355
402
  - `binomial` and `nbinom` fit two shape parameters (`n`, `p`) from the sample mean and
356
403
  variance alone. If `n` is not known independently, it is not identifiable from the data —
357
404
  many `(n, p)` pairs with the same `n·p` fit about as well, so a fitted `n` should be treated
@@ -141,7 +141,9 @@ See [Visual critique](07-visual-critique.md).
141
141
  | Does it survive changing the parameter from stage 4? | If it does not, the finding is about the parameter | the sweep you ran |
142
142
  | If you modelled it, what does the residual look like? | Structure in the residual is an unmodelled mechanism, not noise | the residual's shape over time and its distribution |
143
143
 
144
- See [Models and residuals](06-models-and-residuals.md) §6.3.
144
+ See [Models and residuals](06-models-and-residuals.md) §6.3 for the residual loop, and §6.7
145
+ for the shuffle test — `NullDistribution` runs the resampling once you have said what to
146
+ shuffle.
145
147
 
146
148
  ### The garden of forking paths
147
149
 
@@ -697,8 +697,8 @@ STAGE_ROWS: dict[AnalysisStage, list[StageRow]] = {
697
697
  "did not suggest it",
698
698
  ),
699
699
  (
700
- "If you shuffled the labels and re-measured, how often would you see something "
701
- "this strong?",
700
+ f"If you shuffled the labels and re-measured, how often would you see something "
701
+ f"this strong? ({_models()} §6.7)",
702
702
  "The cheapest honest check available, and it needs no distributional theory at "
703
703
  "all.",
704
704
  "what gets shuffled, and a rough frequency",
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "goad-toolkit"
3
- version = "0.2.5"
3
+ version = "0.2.7"
4
4
  description = "An extensible toolkit for Goal Oriented Analysis of Data"
5
5
  readme = "README.md"
6
6
  authors = [
@@ -1,5 +1,5 @@
1
1
  from dataclasses import dataclass
2
- from typing import Any, List, Optional, Tuple, Union
2
+ from typing import Any, Callable, List, Optional, Tuple, Union
3
3
 
4
4
  import numpy as np
5
5
  import pandas as pd
@@ -148,7 +148,9 @@ class DistributionFitter:
148
148
  We will prefer distributions with higher log-likelihood values.
149
149
  """
150
150
  try:
151
- log_density = dist_obj.dist.logpmf if dist_obj.is_discrete else dist_obj.dist.logpdf
151
+ log_density = (
152
+ dist_obj.dist.logpmf if dist_obj.is_discrete else dist_obj.dist.logpdf
153
+ )
152
154
  return np.sum(log_density(data, *params))
153
155
  except Exception as e:
154
156
  logger.warning(f"Log-likelihood calculation failed: {str(e)}")
@@ -327,6 +329,155 @@ class DistributionFitter:
327
329
  raise ValueError(f"Unknown criterion '{criterion}'")
328
330
 
329
331
 
332
+ @dataclass
333
+ class NullResult:
334
+ """A statistic, and the same statistic measured under shuffled labels.
335
+
336
+ `observed` is the number you found. `null` holds the same measurement taken
337
+ `n_iter` times on data where the labels were shuffled — data in which, by
338
+ construction, the label explains nothing. If `observed` sits inside that
339
+ cloud, a world where nothing is happening produces your finding routinely.
340
+ """
341
+
342
+ observed: float
343
+ null: np.ndarray
344
+ n_iter: int
345
+ seed: Optional[int] = None
346
+
347
+ def p_value(self, alternative: str = "two-sided") -> float:
348
+ """How often a shuffle matched or beat the observed value.
349
+
350
+ Args:
351
+ alternative: `"greater"` counts shuffles at or above `observed`,
352
+ `"less"` counts shuffles at or below it, and `"two-sided"`
353
+ (the default) is twice the smaller of the two, capped at 1.
354
+
355
+ Both one-sided counts include the observed arrangement itself — it is
356
+ one of the arrangements the labels could take — so the smallest value
357
+ this can return is `1 / (n_iter + 1)` rather than zero. With 1000
358
+ shuffles, "p < 0.001" is the strongest statement available; report it
359
+ that way rather than as an exact number.
360
+ """
361
+ greater = (int(np.sum(self.null >= self.observed)) + 1) / (self.n_iter + 1)
362
+ less = (int(np.sum(self.null <= self.observed)) + 1) / (self.n_iter + 1)
363
+
364
+ if alternative == "greater":
365
+ return greater
366
+ if alternative == "less":
367
+ return less
368
+ if alternative == "two-sided":
369
+ return min(1.0, 2 * min(greater, less))
370
+ raise ValueError(
371
+ f"Unknown alternative '{alternative}', "
372
+ "expected 'two-sided', 'greater' or 'less'"
373
+ )
374
+
375
+ @property
376
+ def quantile(self) -> float:
377
+ """Where `observed` sits in the null, as a fraction between 0 and 1."""
378
+ return float(np.mean(self.null < self.observed))
379
+
380
+ def __repr__(self) -> str:
381
+ return (
382
+ f"NullResult(observed={self.observed:.4g}, "
383
+ f"null_mean={float(np.mean(self.null)):.4g}, "
384
+ f"quantile={self.quantile:.3f}, "
385
+ f"p={self.p_value():.4f}, n_iter={self.n_iter})"
386
+ )
387
+
388
+
389
+ class NullDistribution:
390
+ """Shuffle the labels, measure again, and see where the real number lands.
391
+
392
+ The cheapest honest check in the method (§9.6): it assumes nothing about the
393
+ shape of the data, needs no table of critical values, and answers in a
394
+ picture rather than a test. Pair it with `NullPlot` — a student who has seen
395
+ their finding sitting in the middle of the null does not need a p-value
396
+ explained.
397
+
398
+ The statistic is yours to write. Naming what you are measuring is the
399
+ thinking part; running it a thousand times is the part worth delegating.
400
+
401
+ ```python
402
+ def gap(frame):
403
+ means = frame.groupby("is_bot")["length"].mean()
404
+ return means[True] - means[False]
405
+
406
+ result = NullDistribution(gap, n_iter=1000, seed=42).run(data, label="is_bot")
407
+ ```
408
+ """
409
+
410
+ def __init__(
411
+ self,
412
+ statistic: Callable[[pd.DataFrame], float],
413
+ n_iter: int = 1000,
414
+ seed: Optional[int] = None,
415
+ ) -> None:
416
+ """
417
+ Args:
418
+ statistic: Takes a dataframe, returns one number. Called once on the
419
+ data as given, then once per shuffle.
420
+ n_iter: Number of shuffles. 1000 is enough to see the shape; go to
421
+ 10_000 when the observed value sits far out in the tail.
422
+ seed: Passed to `numpy.random.default_rng`. Set it, or the picture
423
+ changes every time the cell is run.
424
+ """
425
+ if n_iter < 1:
426
+ raise ValueError(f"n_iter must be at least 1, got {n_iter}")
427
+ self.statistic = statistic
428
+ self.n_iter = n_iter
429
+ self.seed = seed
430
+
431
+ def run(self, data: pd.DataFrame, label: str) -> NullResult:
432
+ """Measure `statistic` on `data`, then on `n_iter` shuffles of `label`.
433
+
434
+ Shuffling one column breaks the link between it and every other column
435
+ while leaving both distributions exactly as they were: the labels keep
436
+ their proportions, the values keep their shape, and only the pairing is
437
+ destroyed. That is the "nothing is happening" world to compare against.
438
+
439
+ Args:
440
+ data: The frame to measure. It is never modified.
441
+ label: Column to shuffle — the one whose association with the rest
442
+ of the frame is in question.
443
+
444
+ Returns:
445
+ A `NullResult` holding the observed value and the shuffled cloud.
446
+ """
447
+ if label not in data.columns:
448
+ raise ValueError(
449
+ f"Column '{label}' is not in the data; got {list(data.columns)}"
450
+ )
451
+
452
+ observed = float(self.statistic(data))
453
+ if not np.isfinite(observed):
454
+ raise ValueError(
455
+ f"The statistic returned {observed} on the unshuffled data; "
456
+ "there is nothing to compare a null against."
457
+ )
458
+
459
+ rng = np.random.default_rng(self.seed)
460
+ values = data[label].to_numpy()
461
+ shuffled = data.copy()
462
+
463
+ null = np.empty(self.n_iter)
464
+ for i in range(self.n_iter):
465
+ # A plain array, not a Series: assigning a permuted Series realigns
466
+ # on the index and puts every label straight back where it started.
467
+ shuffled[label] = rng.permutation(values)
468
+ null[i] = float(self.statistic(shuffled))
469
+
470
+ if not np.all(np.isfinite(null)):
471
+ logger.warning(
472
+ f"{int(np.sum(~np.isfinite(null)))} of {self.n_iter} shuffles gave a "
473
+ "non-finite statistic; a small group may be emptying under shuffling."
474
+ )
475
+
476
+ return NullResult(
477
+ observed=observed, null=null, n_iter=self.n_iter, seed=self.seed
478
+ )
479
+
480
+
330
481
  def fit_table(results: List[Result]) -> pd.DataFrame:
331
482
  """Turn a list of fit results into a ranked dataframe: what a student hands in.
332
483
 
@@ -343,8 +494,12 @@ def fit_table(results: List[Result]) -> pd.DataFrame:
343
494
  "distribution": result.distribution,
344
495
  "params": result.params,
345
496
  "log_likelihood": result.log_likelihood,
346
- "ks_stat": result.kstest.statistic if result.kstest else float("nan"),
347
- "ks_pvalue": result.kstest.p_value if result.kstest else float("nan"),
497
+ "ks_stat": result.kstest.statistic
498
+ if result.kstest
499
+ else float("nan"),
500
+ "ks_pvalue": result.kstest.p_value
501
+ if result.kstest
502
+ else float("nan"),
348
503
  "best_likelihood": bool(result.best_likelihood),
349
504
  "best_ks": bool(result.best_ks),
350
505
  "message": result.message,