goad-toolkit 0.2.8__tar.gz → 0.2.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/PKG-INFO +1 -1
  2. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/02-pipelines.md +25 -3
  3. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/03-plot-composition.md +18 -0
  4. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/04-five-families.md +8 -3
  5. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/08-api-reference.md +22 -1
  6. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/09-analysis-method.md +1 -1
  7. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/pyproject.toml +1 -1
  8. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/datatransforms.py +59 -2
  9. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/visualizer.py +196 -0
  10. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/tests/test_datatransforms.py +91 -1
  11. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/tests/test_visualizer.py +116 -0
  12. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/uv.lock +1 -1
  13. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/settings.local.json +0 -0
  14. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/.claude/settings.local.json +0 -0
  15. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/.gitignore +0 -0
  16. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/.python-version +0 -0
  17. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/CHANGELOG.md +0 -0
  18. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/MCP_SERVER.md +0 -0
  19. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/README.md +0 -0
  20. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/demo/linear.py +0 -0
  21. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/01-goal-oriented-analysis.md +0 -0
  22. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/02-pipelines.md +0 -0
  23. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/03-plot-composition.md +0 -0
  24. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/04-five-families.md +0 -0
  25. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/05-distributions.md +0 -0
  26. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/06-models-and-residuals.md +0 -0
  27. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/07-visual-critique.md +0 -0
  28. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/08-api-reference.md +0 -0
  29. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/09-analysis-method.md +0 -0
  30. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/10-teaching-path.md +0 -0
  31. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/docs/README.md +0 -0
  32. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/goad_mcp.py +0 -0
  33. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/img/distribution_fit.png +0 -0
  34. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/img/goaded.png +0 -0
  35. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/img/linear_results.png +0 -0
  36. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/img/null-distribution.png +0 -0
  37. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/img/residuals.png +0 -0
  38. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/img/zscores.png +0 -0
  39. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/pyproject.toml +0 -0
  40. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/__init__.py +0 -0
  41. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/analytics.py +0 -0
  42. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/cli.py +0 -0
  43. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/config.py +0 -0
  44. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/dataprocessor.py +0 -0
  45. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/datatransforms.py +0 -0
  46. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/distributions.py +0 -0
  47. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/filehandler.py +0 -0
  48. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/models.py +0 -0
  49. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/src/goad_toolkit/visualizer.py +0 -0
  50. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/tests/test_cli.py +0 -0
  51. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/tests/test_datatransforms.py +0 -0
  52. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/tests/test_distributions.py +0 -0
  53. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/tests/test_filehandler.py +0 -0
  54. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/tests/test_nulldistribution.py +0 -0
  55. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/tests/test_visualizer.py +0 -0
  56. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.claude/worktrees/quizzical-solomon-b62511/uv.lock +0 -0
  57. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.gitignore +0 -0
  58. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.python-version +0 -0
  59. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/.gitignore +0 -0
  60. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/archive.md +0 -0
  61. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/autonomous/save-104759.log +0 -0
  62. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/autonomous/save-195516.log +0 -0
  63. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/autonomous/save-195720.log +0 -0
  64. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/autonomous/save-195933.log +0 -0
  65. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/autonomous/save-200831.log +0 -0
  66. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/autonomous/save-201034.log +0 -0
  67. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/hook-errors.log +0 -0
  68. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/memory-2026-08-10.log +0 -0
  69. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/logs/memory-2026-08-12.log +0 -0
  70. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/now.md +0 -0
  71. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/recent.md +0 -0
  72. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/capture-alive +0 -0
  73. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/capture-alive.d/16b4788a-5507-4216-98ec-06bac13a9aba +0 -0
  74. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/capture-alive.d/90574e63-97cd-4c11-87e5-23b656c56fd9 +0 -0
  75. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/case-divergence +0 -0
  76. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/last-ndc.ts +0 -0
  77. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/last-save-ts +0 -0
  78. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/last-save.json +0 -0
  79. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/now-day +0 -0
  80. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/post-tool-ran +0 -0
  81. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/save-session.pid +0 -0
  82. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/tmp/session-slug +0 -0
  83. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/.remember/today-2026-08-10.done.md +0 -0
  84. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/CHANGELOG.md +0 -0
  85. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/MCP_SERVER.md +0 -0
  86. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/README.md +0 -0
  87. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/demo/linear.py +0 -0
  88. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/01-goal-oriented-analysis.md +0 -0
  89. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/05-distributions.md +0 -0
  90. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/06-models-and-residuals.md +0 -0
  91. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/07-visual-critique.md +0 -0
  92. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/10-teaching-path.md +0 -0
  93. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/docs/README.md +0 -0
  94. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/goad_mcp.py +0 -0
  95. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/img/distribution_fit.png +0 -0
  96. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/img/goaded.png +0 -0
  97. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/img/linear_results.png +0 -0
  98. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/img/null-distribution.png +0 -0
  99. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/img/residuals.png +0 -0
  100. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/img/zscores.png +0 -0
  101. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/__init__.py +0 -0
  102. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/analytics.py +0 -0
  103. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/cli.py +0 -0
  104. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/config.py +0 -0
  105. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/dataprocessor.py +0 -0
  106. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/distributions.py +0 -0
  107. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/filehandler.py +0 -0
  108. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/src/goad_toolkit/models.py +0 -0
  109. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/tests/test_cli.py +0 -0
  110. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/tests/test_distributions.py +0 -0
  111. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/tests/test_filehandler.py +0 -0
  112. {goad_toolkit-0.2.8 → goad_toolkit-0.2.9}/tests/test_nulldistribution.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: goad-toolkit
3
- Version: 0.2.8
3
+ Version: 0.2.9
4
4
  Summary: An extensible toolkit for Goal Oriented Analysis of Data
5
5
  Project-URL: Github, https://github.com/raoulg/goad_toolkit
6
6
  Author-email: raoul grouls <Raoul.Grouls@han.nl>
@@ -199,9 +199,31 @@ will not warn you. Write the steps in the order you would do them by hand.
199
199
  you would want to run again on new data. A one-off `df[df.author == "Alice"]` while you are
200
200
  poking around is a notebook cell, and should stay one.
201
201
 
202
- **Debugging.** When a pipeline produces something surprising, add the steps back one at a
203
- time rather than dismantling it into cells — the moment it becomes cells again you have lost
204
- the record of what you did, which is the thing you were trying to keep.
202
+ **Debugging.** When a pipeline produces something surprising, do not dismantle it into cells:
203
+ the moment it becomes cells again you have lost the record of what you did, which is the
204
+ thing you were trying to keep. Ask it what it did instead.
205
+
206
+ ```python
207
+ result = pipeline.apply(data, keep_intermediate=True)
208
+ pipeline.report()
209
+ ```
210
+
211
+ ```
212
+ step rows columns added removed
213
+ calendar 4083 9 hour, day_name
214
+ weekend 4083 10 is_weekend
215
+ window 3961 10
216
+ ```
217
+
218
+ `report()` gives one row per step: how many rows survived it, how many columns the frame had
219
+ afterwards, and which columns that step added or removed. The step where the row count falls
220
+ off a cliff is the one filtering more than you meant it to, and a step with an empty `added`
221
+ is a step that silently did nothing.
222
+
223
+ `pipeline.intermediate` holds the frame as it stood after each step, keyed by step name, so
224
+ you can look at the rows themselves once `report()` has told you where to look. Both are off
225
+ by default — keeping them costs one full copy of the frame per step — and turning them on
226
+ does not change what `apply` returns.
205
227
 
206
228
  ---
207
229
 
@@ -155,6 +155,7 @@ step 3.
155
155
  | `HeatmapPlot` | `sns.heatmap`, pivoting the frame if asked | when both variables have many levels |
156
156
  | `BarbellPlot` | before and after per category, joined | the change is the message |
157
157
  | `HighlightCategory` | recolours bars: grey, plus the one that matters | a layer, and it needs no help from what drew them |
158
+ | `Annotate` | text on the plot, with an arrow to the point | a layer; the finding, written where it is read |
158
159
  | `ScatterPlot` | `sns.scatterplot` | the first plot of any relation question |
159
160
  | `RegPlot` | `sns.regplot` — `fit_reg`, `lowess`, `order` | `scatter=False` layers it over a `ScatterPlot` |
160
161
  | `CorrelationHeatmap` | `.corr()` as a heatmap, scale pinned to [-1, 1] | where to look next, not a ranking |
@@ -168,6 +169,8 @@ step 3.
168
169
  | `QQPlot` | sample quantiles vs. a distribution's theoretical quantiles | the tails, not the bulk — see [Distributions §5.5](05-distributions.md) |
169
170
  | `ECDFPlot` | one or two empirical CDFs, bin-free | the right tool for comparing two samples |
170
171
  | `NullPlot` | a shuffled statistic, with the observed value marked | `HistogramPlot` + `VerticalLine`, see [Models and residuals §6.7](06-models-and-residuals.md) |
172
+ | `ProjectionPlot` | already-fitted 2-D coordinates | takes an array, not a model — see [Five families §4.5](04-five-families.md) |
173
+ | `ScreePlot` | variance explained per component, plus the running total | the figure a projection has to come with |
171
174
 
172
175
  `HistogramPlot` defaults to `sqrt(n)` bins capped at 50. That default is a starting point,
173
176
  not an answer — bin count changes what a histogram appears to say, and choosing it is part of
@@ -202,6 +205,21 @@ bars.plot_on(HighlightCategory(settings))
202
205
  about the data, but they are only defaults — pass `categories=` at the call to override.
203
206
  Naming a category that is not on the axis raises, and lists the ones that are.
204
207
 
208
+ `Annotate` is the other layer worth reaching for by habit. A reader with thirty seconds reads
209
+ the annotation and not the axes, so the sentence you would have said out loud belongs on the
210
+ plot:
211
+
212
+ ```python
213
+ line.plot_on(
214
+ Annotate(settings),
215
+ text="six-day gap: the logging host was down, not the channel",
216
+ xy=(gap_date, gap_value), # the point being explained
217
+ xytext=(label_date, label_value), # where the text sits; an arrow joins them
218
+ )
219
+ ```
220
+
221
+ Give it `xy` alone and the text lands on the point with no arrow.
222
+
205
223
  ## 3.7 Writing your own plot
206
224
 
207
225
  Two questions decide the shape:
@@ -210,9 +210,14 @@ kept; a distance matrix when the items are few enough.
210
210
  space; use the projection to *look* at the result. Clustering the 2D coordinates is
211
211
  clustering an artefact.
212
212
 
213
- **In GOAD:** projections are computed elsewhere (sklearn) and plotted through a `BasePlot`
214
- subclass, so the variance-explained ends up in the `PlotSettings` title where a reader will
215
- see it.
213
+ **In GOAD:** `ProjectionPlot` and `ScreePlot`. Both take arrays, not models — fitting a PCA,
214
+ t-SNE or UMAP is sklearn's job and stays outside this library, so goad never has an opinion
215
+ about how you got the coordinates.
216
+
217
+ `ProjectionPlot` hides the tick values by default: the units of an embedding mean nothing, and
218
+ showing them invites reading a distance off them. `ScreePlot` is the figure that has to
219
+ accompany it — the variance-explained belongs in the title where a reader will see it, and
220
+ two components carrying 12% make a picture of 12% of the data.
216
221
 
217
222
  ---
218
223
 
@@ -102,14 +102,26 @@ for the step.
102
102
  ```python
103
103
  class Pipeline:
104
104
  def add(self, transform_class: Type[T], name: Optional[str] = None, **kwargs) -> "Pipeline"
105
- def apply(self, data: pd.DataFrame) -> pd.DataFrame
105
+ def apply(self, data: pd.DataFrame, keep_intermediate: bool = False) -> pd.DataFrame
106
+ def report(self) -> pd.DataFrame
106
107
  def __getitem__(self, key: str) -> Dict[str, Any]
107
108
  def __setitem__(self, key: str, params: Dict[str, Any]) -> None
109
+
110
+ intermediate: Dict[str, pd.DataFrame] # the frame after each step, when kept
111
+ input_columns: List[str] # the columns apply was handed
108
112
  ```
109
113
 
110
114
  `add` takes the **class**, not an instance, and returns `self` so calls chain. `apply` copies
111
115
  the frame once, then instantiates and runs each step in insertion order.
112
116
 
117
+ `keep_intermediate=True` fills `intermediate` with a **copy** of the frame after every step —
118
+ a copy is required, since the steps mutate one frame in place and stored references would all
119
+ show the finished columns. It does not change what `apply` returns. Both `intermediate` and
120
+ `input_columns` are reset at the start of every `apply`, so a later plain run clears them.
121
+
122
+ `report()` turns those into one row per step — `step`, `rows`, `columns`, `added`, `removed` —
123
+ and raises if no intermediates were kept. See [Pipelines §2.7](02-pipelines.md).
124
+
113
125
  ---
114
126
 
115
127
  ## 8.4 `goad_toolkit.dataprocessor`
@@ -325,6 +337,9 @@ class BasePlot(ABC):
325
337
  | `HeatmapPlot` | `data, index=None, columns=None, values=None, aggfunc="mean", cmap="rocket_r", annot=True, fmt=".1f", **kwargs` |
326
338
  | `BarbellPlot` | `data, category, start, end, start_label="before", end_label="after", start_color="lightgrey", end_color="crimson", line_color="lightgrey", markersize=80, **kwargs` |
327
339
  | `HighlightCategory` | `categories=None, color=None, base_color=None, axis="x"` |
340
+ | `Annotate` | `text, xy, xytext=None, arrow=True, color="black", fontsize=11, **kwargs` |
341
+ | `ProjectionPlot` | `coordinates, labels=None, alpha=0.7, hide_axes=True, **kwargs` |
342
+ | `ScreePlot` | `explained_variance_ratio, n_components=None, cumulative=True, color="steelblue", line_color="crimson", **kwargs` |
328
343
  | `BarWithDates` | `data, x, y, interval: int = 1, **kwargs` |
329
344
  | `ResidualPlot` | `data, x, y, date, datelabel, interval: int = 1` |
330
345
  | `ScatterPlot` | `data, x, y, alpha=0.6, **kwargs` → `sns.scatterplot` (`hue`, `size`, `style`) |
@@ -367,6 +382,12 @@ category that is not on the axis raises and lists the ones that are; so does app
367
382
  axis with no bars. Zero-size patches are skipped — seaborn keeps its legend proxies in
368
383
  `ax.patches`, and recolouring those would rewrite the legend's swatches.
369
384
 
385
+ `Annotate` draws an arrow only when `xytext` is given — an arrow from a point to itself is not
386
+ a thing — so `xy` alone places the text on the point. `ProjectionPlot` takes coordinates of
387
+ shape `(n, 2)` and raises on anything else, or on a label count that does not match; it hides
388
+ the tick values unless told not to. `ScreePlot` takes the ratios, not a fitted model, so
389
+ sklearn stays out of goad.
390
+
370
391
  `HeatmapPlot` pivots when given both `index` and `columns` (and `values`), and draws `data`
371
392
  as a matrix when given neither; one without the other raises. Its scale is sequential,
372
393
  unlike `CorrelationHeatmap`'s. `BarbellPlot` draws rows in the frame's own order — sort by the
@@ -121,7 +121,7 @@ See [The five families](04-five-families.md).
121
121
  | What is grouped visually, and does it match what is grouped conceptually? | The gestalt principles group things for the reader without asking you | the groupings, intended and accidental |
122
122
  | Which single element carries the message, and is it the only coloured thing? | Grey first, colour as a pointer, not decoration | the one element |
123
123
  | What can you delete without losing meaning? | Everything remaining should be earning its place | the list you deleted |
124
- | Is the finding written on the plot? | A reader with thirty seconds reads the annotation, not the axes | the annotation, and where it sits |
124
+ | Is the finding written on the plot? (`Annotate`) | A reader with thirty seconds reads the annotation, not the axes | the annotation, and where it sits |
125
125
 
126
126
  See [Visual critique](07-visual-critique.md).
127
127
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "goad-toolkit"
3
- version = "0.2.8"
3
+ version = "0.2.9"
4
4
  description = "An extensible toolkit for Goal Oriented Analysis of Data"
5
5
  readme = "README.md"
6
6
  authors = [
@@ -213,6 +213,8 @@ class Pipeline:
213
213
 
214
214
  def __init__(self):
215
215
  self.transforms: Dict[str, Dict[str, Any]] = {}
216
+ self.intermediate: Dict[str, pd.DataFrame] = {}
217
+ self.input_columns: List[str] = []
216
218
 
217
219
  def add(
218
220
  self, transform_class: Type[T], name: Optional[str] = None, **kwargs
@@ -229,8 +231,25 @@ class Pipeline:
229
231
  self.transforms[name] = {"class": transform_class, "params": kwargs}
230
232
  return self # Allow method chaining
231
233
 
232
- def apply(self, data: pd.DataFrame) -> pd.DataFrame:
233
- """Apply all transformations in the pipeline."""
234
+ def apply(
235
+ self, data: pd.DataFrame, keep_intermediate: bool = False
236
+ ) -> pd.DataFrame:
237
+ """Apply all transformations in the pipeline.
238
+
239
+ Args:
240
+ data: Frame to transform. It is never modified.
241
+ keep_intermediate: Keep the frame as it stood after every step in
242
+ `self.intermediate`, and the input's columns in
243
+ `self.input_columns`, so `report()` can say which step did what.
244
+ Off by default: it holds one full copy of the frame per step.
245
+
246
+ Returns:
247
+ The frame after every transform, whether or not intermediates were
248
+ kept — turning introspection on does not change what you get back.
249
+ """
250
+ self.intermediate = {}
251
+ self.input_columns = list(data.columns)
252
+
234
253
  # Make a single copy at the pipeline level
235
254
  result = data.copy() if len(self.transforms) > 0 else data
236
255
  for name, transform_config in tqdm(
@@ -240,8 +259,46 @@ class Pipeline:
240
259
  params = transform_config["params"]
241
260
  transform = transform_class(name=name, **params)
242
261
  result = transform(result)
262
+ if keep_intermediate:
263
+ # A copy, not a reference: the steps mutate one frame in place,
264
+ # so a stored reference would show every step holding the final
265
+ # columns and nothing would look like it had done anything.
266
+ self.intermediate[name] = result.copy()
243
267
  return result
244
268
 
269
+ def report(self) -> pd.DataFrame:
270
+ """One row per step: how many rows survived it, and which columns it added.
271
+
272
+ The answer to "which step broke it", without dismantling the pipeline
273
+ back into loose cells. Needs `apply(..., keep_intermediate=True)` first.
274
+
275
+ Returns:
276
+ A frame with one row per step: `step`, `rows`, `columns`, `added`
277
+ and `removed`, in the order the steps ran.
278
+ """
279
+ if not self.intermediate:
280
+ raise ValueError(
281
+ "No intermediate frames to report on. Run "
282
+ "apply(data, keep_intermediate=True) first."
283
+ )
284
+
285
+ rows = []
286
+ previous = list(self.input_columns)
287
+ for name, frame in self.intermediate.items():
288
+ current = list(frame.columns)
289
+ rows.append(
290
+ {
291
+ "step": name,
292
+ "rows": len(frame),
293
+ "columns": len(current),
294
+ "added": ", ".join(c for c in current if c not in previous),
295
+ "removed": ", ".join(c for c in previous if c not in current),
296
+ }
297
+ )
298
+ previous = current
299
+
300
+ return pd.DataFrame(rows)
301
+
245
302
  def __getitem__(self, key: str) -> Dict[str, Any]:
246
303
  """Get a specific transform configuration by name."""
247
304
  if key in self.transforms:
@@ -365,6 +365,68 @@ class HighlightCategory(BasePlot):
365
365
  return self.fig, self.ax
366
366
 
367
367
 
368
+ class Annotate(BasePlot):
369
+ """Write the finding on the plot, where a reader with thirty seconds looks.
370
+
371
+ The last question in the visual critique is whether the finding is written
372
+ down. Axis labels say what the numbers are; the annotation says what you
373
+ concluded, and it is the only part most readers will read.
374
+ """
375
+
376
+ def build(
377
+ self,
378
+ text: str,
379
+ xy: Tuple[float, float],
380
+ xytext: Optional[Tuple[float, float]] = None,
381
+ arrow: bool = True,
382
+ color: str = "black",
383
+ fontsize: float = 11,
384
+ **kwargs,
385
+ ):
386
+ """
387
+ Place `text` on the axis, optionally with an arrow to the point.
388
+
389
+ Parameters:
390
+ -----------
391
+ text : str
392
+ What you concluded. A sentence, not a label.
393
+ xy : Tuple[float, float]
394
+ Point in data coordinates the text is about.
395
+ xytext : Optional[Tuple[float, float]]
396
+ Where the text itself sits, in data coordinates. Defaults to `xy`,
397
+ which places the text on the point and draws no arrow.
398
+ arrow : bool
399
+ Draw an arrow from the text to `xy`. Ignored when `xytext` is None,
400
+ since an arrow from a point to itself is not a thing.
401
+ color : str
402
+ Text colour.
403
+ fontsize : float
404
+ Text size.
405
+ **kwargs : Additional keyword arguments passed to `ax.annotate`.
406
+
407
+ Returns:
408
+ --------
409
+ fig, ax : The created figure and axes.
410
+ """
411
+ if self.ax is None:
412
+ raise ValueError("No axes available for plotting")
413
+
414
+ arrowprops = None
415
+ if xytext is not None and arrow:
416
+ arrowprops = {"arrowstyle": "->", "color": color, "linewidth": 1}
417
+
418
+ self.ax.annotate(
419
+ text,
420
+ xy=xy,
421
+ xytext=xytext if xytext is not None else xy,
422
+ color=color,
423
+ fontsize=fontsize,
424
+ arrowprops=arrowprops,
425
+ **kwargs,
426
+ )
427
+ return self.fig, self.ax
428
+
429
+
368
430
  class ComparePlotDate(BasePlot):
369
431
  def build(
370
432
  self,
@@ -942,6 +1004,140 @@ class CorrelationHeatmap(BasePlot):
942
1004
  return self.fig, self.ax
943
1005
 
944
1006
 
1007
+ class ProjectionPlot(BasePlot):
1008
+ """Points in two dimensions after a projection has already been fitted.
1009
+
1010
+ Takes coordinates, not a model: fitting PCA, t-SNE or UMAP is sklearn's job
1011
+ and stays outside this library. Pass the array it gave you.
1012
+
1013
+ The axes of a t-SNE or UMAP plot mean nothing — not distance, not direction,
1014
+ not scale — so they are hidden by default. What survives is which points sit
1015
+ together, and even that is a claim to check rather than a finding.
1016
+ """
1017
+
1018
+ def build(
1019
+ self,
1020
+ coordinates: Any,
1021
+ labels: Optional[Any] = None,
1022
+ alpha: float = 0.7,
1023
+ hide_axes: bool = True,
1024
+ **kwargs,
1025
+ ):
1026
+ """
1027
+ Scatter a set of two-dimensional coordinates.
1028
+
1029
+ Parameters:
1030
+ -----------
1031
+ coordinates : array-like
1032
+ Shape (n, 2): the output of a fitted projection.
1033
+ labels : Optional[array-like]
1034
+ One label per point, used as colour. Length must match.
1035
+ alpha : float
1036
+ Point transparency; projections overplot heavily in dense regions.
1037
+ hide_axes : bool
1038
+ Hide the tick marks and values. The units of an embedding are not
1039
+ interpretable, and showing them invites reading a distance off them.
1040
+ **kwargs : Additional keyword arguments passed to `sns.scatterplot`.
1041
+
1042
+ Returns:
1043
+ --------
1044
+ fig, ax : The created figure and axes.
1045
+ """
1046
+ if self.ax is None:
1047
+ raise ValueError("No axes available for plotting")
1048
+
1049
+ points = np.asarray(coordinates)
1050
+ if points.ndim != 2 or points.shape[1] != 2:
1051
+ raise ValueError(
1052
+ f"ProjectionPlot needs coordinates of shape (n, 2), got {points.shape}"
1053
+ )
1054
+ if labels is not None and len(labels) != len(points):
1055
+ raise ValueError(f"Got {len(labels)} labels for {len(points)} points")
1056
+
1057
+ sns.scatterplot(
1058
+ x=points[:, 0],
1059
+ y=points[:, 1],
1060
+ hue=labels,
1061
+ alpha=alpha,
1062
+ ax=self.ax,
1063
+ **kwargs,
1064
+ )
1065
+
1066
+ if hide_axes:
1067
+ self.ax.set_xticks([])
1068
+ self.ax.set_yticks([])
1069
+
1070
+ return self.fig, self.ax
1071
+
1072
+
1073
+ class ScreePlot(BasePlot):
1074
+ """How much of the variance the components you kept actually carry.
1075
+
1076
+ The figure that has to accompany a projection. Two components that explain
1077
+ 12% of the variance make a picture of 12% of the data, and a shape in that
1078
+ picture is not a shape in the dataset.
1079
+ """
1080
+
1081
+ def build(
1082
+ self,
1083
+ explained_variance_ratio: Any,
1084
+ n_components: Optional[int] = None,
1085
+ cumulative: bool = True,
1086
+ color: str = "steelblue",
1087
+ line_color: str = "crimson",
1088
+ **kwargs,
1089
+ ):
1090
+ """
1091
+ Draw the variance explained per component, and the running total.
1092
+
1093
+ Parameters:
1094
+ -----------
1095
+ explained_variance_ratio : array-like
1096
+ One fraction per component, as `explained_variance_ratio_` gives.
1097
+ n_components : Optional[int]
1098
+ Show only the first this many components.
1099
+ cumulative : bool
1100
+ Overlay the running total, which is what "how much did I keep"
1101
+ actually asks.
1102
+ color : str
1103
+ Bar colour.
1104
+ line_color : str
1105
+ Cumulative line colour.
1106
+ **kwargs : Additional keyword arguments passed to `ax.bar`.
1107
+
1108
+ Returns:
1109
+ --------
1110
+ fig, ax : The created figure and axes.
1111
+ """
1112
+ if self.ax is None:
1113
+ raise ValueError("No axes available for plotting")
1114
+
1115
+ ratios = np.asarray(explained_variance_ratio, dtype=float)
1116
+ if ratios.ndim != 1 or len(ratios) == 0:
1117
+ raise ValueError(
1118
+ f"ScreePlot needs a one-dimensional, non-empty array of ratios, "
1119
+ f"got shape {ratios.shape}"
1120
+ )
1121
+ if n_components is not None:
1122
+ ratios = ratios[:n_components]
1123
+
1124
+ components = np.arange(1, len(ratios) + 1)
1125
+ self.ax.bar(components, ratios, color=color, **kwargs)
1126
+
1127
+ if cumulative:
1128
+ self.ax.plot(
1129
+ components,
1130
+ np.cumsum(ratios),
1131
+ color=line_color,
1132
+ marker="o",
1133
+ label="cumulative",
1134
+ )
1135
+ self.ax.legend()
1136
+
1137
+ self.ax.set_xticks(components)
1138
+ return self.fig, self.ax
1139
+
1140
+
945
1141
  class HistogramPlot(BasePlot):
946
1142
  """Plot a histogram using seaborn."""
947
1143
 
@@ -1,7 +1,12 @@
1
1
  import pandas as pd
2
2
  import pytest
3
3
 
4
- from goad_toolkit.datatransforms import Pipeline, RegexFeature, TimeFeatures
4
+ from goad_toolkit.datatransforms import (
5
+ Pipeline,
6
+ RegexFeature,
7
+ TimeFeatures,
8
+ TransformBase,
9
+ )
5
10
 
6
11
 
7
12
  @pytest.fixture
@@ -165,3 +170,88 @@ def test_feature_is_the_column_name_and_name_is_the_step(messages):
165
170
  assert "addressed_to" in result.columns
166
171
  assert "mentions" not in result.columns
167
172
  assert "mentions" in repr(pipeline)
173
+
174
+
175
+ class KeepFirst(TransformBase):
176
+ """A transform that drops rows, so a report has something to report."""
177
+
178
+ def transform(self, data: pd.DataFrame, n: int) -> pd.DataFrame:
179
+ return data.head(n)
180
+
181
+
182
+ def test_apply_keeps_nothing_by_default(data):
183
+ pipeline = Pipeline()
184
+ pipeline.add(TimeFeatures, column="timestamp", features=["hour"])
185
+
186
+ pipeline.apply(data)
187
+
188
+ assert pipeline.intermediate == {}
189
+
190
+
191
+ def test_apply_keeps_a_frame_per_step_when_asked(data):
192
+ pipeline = Pipeline()
193
+ pipeline.add(TimeFeatures, name="calendar", column="timestamp", features=["hour"])
194
+ pipeline.add(TimeFeatures, name="naming", column="timestamp", features=["day_name"])
195
+
196
+ result = pipeline.apply(data, keep_intermediate=True)
197
+
198
+ assert list(pipeline.intermediate) == ["calendar", "naming"]
199
+ # Each step is kept as it stood, not as a view of the finished frame.
200
+ assert "hour" in pipeline.intermediate["calendar"].columns
201
+ assert "day_name" not in pipeline.intermediate["calendar"].columns
202
+ assert "day_name" in pipeline.intermediate["naming"].columns
203
+ assert "day_name" in result.columns
204
+
205
+
206
+ def test_keeping_intermediates_does_not_change_the_result(data):
207
+ kept = Pipeline()
208
+ kept.add(TimeFeatures, column="timestamp", features=["hour"])
209
+
210
+ plain = Pipeline()
211
+ plain.add(TimeFeatures, column="timestamp", features=["hour"])
212
+
213
+ pd.testing.assert_frame_equal(
214
+ kept.apply(data, keep_intermediate=True), plain.apply(data)
215
+ )
216
+
217
+
218
+ def test_a_second_apply_replaces_the_kept_frames(data):
219
+ pipeline = Pipeline()
220
+ pipeline.add(TimeFeatures, column="timestamp", features=["hour"])
221
+
222
+ pipeline.apply(data, keep_intermediate=True)
223
+ pipeline.apply(data)
224
+
225
+ assert pipeline.intermediate == {}
226
+
227
+
228
+ def test_report_names_the_columns_each_step_added(data):
229
+ pipeline = Pipeline()
230
+ pipeline.add(TimeFeatures, name="calendar", column="timestamp", features=["hour"])
231
+ pipeline.add(TimeFeatures, name="naming", column="timestamp", features=["day_name"])
232
+
233
+ pipeline.apply(data, keep_intermediate=True)
234
+ report = pipeline.report()
235
+
236
+ assert list(report["step"]) == ["calendar", "naming"]
237
+ assert list(report["added"]) == ["hour", "day_name"]
238
+ assert list(report["removed"]) == ["", ""]
239
+ assert list(report["rows"]) == [len(data), len(data)]
240
+
241
+
242
+ def test_report_shows_where_the_rows_went(data):
243
+ pipeline = Pipeline()
244
+ pipeline.add(TimeFeatures, name="calendar", column="timestamp", features=["hour"])
245
+ pipeline.add(KeepFirst, name="first_two", n=2)
246
+
247
+ pipeline.apply(data, keep_intermediate=True)
248
+ report = pipeline.report()
249
+
250
+ assert list(report["rows"]) == [len(data), 2]
251
+
252
+
253
+ def test_report_without_intermediates_says_what_to_do():
254
+ pipeline = Pipeline()
255
+
256
+ with pytest.raises(ValueError, match="keep_intermediate=True"):
257
+ pipeline.report()
@@ -13,6 +13,7 @@ from scipy import stats # noqa: E402
13
13
  from goad_toolkit.analytics import NullResult # noqa: E402
14
14
  from goad_toolkit.visualizer import ( # noqa: E402
15
15
  ACFPlot,
16
+ Annotate,
16
17
  BarbellPlot,
17
18
  BarPlot,
18
19
  BarWithDates,
@@ -27,10 +28,12 @@ from goad_toolkit.visualizer import ( # noqa: E402
27
28
  LinePlot,
28
29
  NullPlot,
29
30
  PlotSettings,
31
+ ProjectionPlot,
30
32
  QQPlot,
31
33
  RegPlot,
32
34
  ResidualPlot,
33
35
  ScatterPlot,
36
+ ScreePlot,
34
37
  VerticalDate,
35
38
  VerticalLine,
36
39
  )
@@ -278,6 +281,119 @@ def facecolours(ax) -> list:
278
281
  return [patch.get_facecolor() for patch in ax.patches]
279
282
 
280
283
 
284
+ def test_annotate_writes_the_text_on_the_axis():
285
+ host = LinePlot(PlotSettings())
286
+ host.create_figure()
287
+
288
+ host.plot_on(Annotate(PlotSettings()), text="the dip is the outage", xy=(1.0, 2.0))
289
+
290
+ assert [t.get_text() for t in host.ax.texts] == ["the dip is the outage"]
291
+
292
+
293
+ def test_annotate_draws_an_arrow_when_the_text_sits_elsewhere():
294
+ host = LinePlot(PlotSettings())
295
+ host.create_figure()
296
+
297
+ host.plot_on(
298
+ Annotate(PlotSettings()), text="here", xy=(1.0, 2.0), xytext=(3.0, 4.0)
299
+ )
300
+
301
+ annotation = host.ax.texts[0]
302
+ assert annotation.arrow_patch is not None
303
+ assert annotation.get_position() == (3.0, 4.0)
304
+
305
+
306
+ def test_annotate_without_a_second_point_draws_no_arrow():
307
+ host = LinePlot(PlotSettings())
308
+ host.create_figure()
309
+
310
+ host.plot_on(Annotate(PlotSettings()), text="here", xy=(1.0, 2.0), arrow=True)
311
+
312
+ assert host.ax.texts[0].arrow_patch is None
313
+
314
+
315
+ def test_projection_plot_scatters_the_coordinates():
316
+ rng = np.random.default_rng(1)
317
+ coordinates = rng.normal(size=(30, 2))
318
+
319
+ fig, ax = ProjectionPlot(PlotSettings()).plot(coordinates=coordinates)
320
+
321
+ assert fig is not None
322
+ assert len(ax.collections[0].get_offsets()) == 30
323
+
324
+
325
+ def test_projection_plot_hides_the_meaningless_axes():
326
+ coordinates = np.random.default_rng(1).normal(size=(10, 2))
327
+
328
+ fig, ax = ProjectionPlot(PlotSettings()).plot(coordinates=coordinates)
329
+
330
+ assert list(ax.get_xticks()) == []
331
+ assert list(ax.get_yticks()) == []
332
+
333
+
334
+ def test_projection_plot_colours_by_label():
335
+ coordinates = np.random.default_rng(1).normal(size=(6, 2))
336
+ labels = ["a", "b"] * 3
337
+
338
+ fig, ax = ProjectionPlot(PlotSettings()).plot(
339
+ coordinates=coordinates, labels=labels
340
+ )
341
+
342
+ legend_labels = [t.get_text() for t in ax.get_legend().get_texts()]
343
+ assert legend_labels == ["a", "b"]
344
+
345
+
346
+ def test_projection_plot_rejects_more_than_two_dimensions():
347
+ coordinates = np.random.default_rng(1).normal(size=(10, 3))
348
+
349
+ with pytest.raises(ValueError, match=r"shape \(n, 2\)"):
350
+ ProjectionPlot(PlotSettings()).plot(coordinates=coordinates)
351
+
352
+
353
+ def test_projection_plot_rejects_a_label_count_mismatch():
354
+ coordinates = np.random.default_rng(1).normal(size=(10, 2))
355
+
356
+ with pytest.raises(ValueError, match="3 labels for 10 points"):
357
+ ProjectionPlot(PlotSettings()).plot(
358
+ coordinates=coordinates, labels=["a", "b", "c"]
359
+ )
360
+
361
+
362
+ def test_scree_plot_draws_a_bar_per_component_and_the_running_total():
363
+ ratios = [0.5, 0.25, 0.15, 0.1]
364
+
365
+ fig, ax = ScreePlot(PlotSettings()).plot(explained_variance_ratio=ratios)
366
+
367
+ assert len(ax.patches) == 4
368
+ cumulative = ax.lines[0].get_ydata()
369
+ assert cumulative[-1] == pytest.approx(1.0)
370
+ assert list(cumulative) == pytest.approx([0.5, 0.75, 0.9, 1.0])
371
+
372
+
373
+ def test_scree_plot_can_show_only_the_first_components():
374
+ ratios = [0.5, 0.25, 0.15, 0.1]
375
+
376
+ fig, ax = ScreePlot(PlotSettings()).plot(
377
+ explained_variance_ratio=ratios, n_components=2
378
+ )
379
+
380
+ assert len(ax.patches) == 2
381
+ assert ax.lines[0].get_ydata()[-1] == pytest.approx(0.75)
382
+
383
+
384
+ def test_scree_plot_without_the_cumulative_line():
385
+ fig, ax = ScreePlot(PlotSettings()).plot(
386
+ explained_variance_ratio=[0.6, 0.4], cumulative=False
387
+ )
388
+
389
+ assert len(ax.lines) == 0
390
+
391
+
392
+ def test_scree_plot_rejects_an_empty_array():
393
+ with pytest.raises(ValueError, match="non-empty"):
394
+ ScreePlot(PlotSettings()).plot(explained_variance_ratio=[])
395
+
396
+
281
397
  def test_bar_plot_draws_one_bar_per_category(categories):
282
398
  fig, ax = BarPlot(PlotSettings()).plot(data=categories, x="species", y="mass")
283
399
 
@@ -383,7 +383,7 @@ wheels = [
383
383
 
384
384
  [[package]]
385
385
  name = "goad-toolkit"
386
- version = "0.2.8"
386
+ version = "0.2.9"
387
387
  source = { editable = "." }
388
388
  dependencies = [
389
389
  { name = "loguru" },
File without changes
File without changes
File without changes
File without changes
File without changes