scistackplot 0.1.28__tar.gz → 0.1.29__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {scistackplot-0.1.28 → scistackplot-0.1.29}/.gitignore +1 -0
  2. {scistackplot-0.1.28 → scistackplot-0.1.29}/PKG-INFO +1 -1
  3. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/capability.py +38 -12
  4. scistackplot-0.1.29/src/scistackplot/dedup.py +93 -0
  5. scistackplot-0.1.29/src/scistackplot/framesize.py +105 -0
  6. scistackplot-0.1.29/src/scistackplot/numeric.py +61 -0
  7. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/reduce.py +233 -47
  8. scistackplot-0.1.29/src/scistackplot/reducer.py +506 -0
  9. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/roles.py +30 -12
  10. scistackplot-0.1.29/src/scistackplot/series_stats.py +204 -0
  11. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/sources/base.py +89 -13
  12. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/table.py +11 -0
  13. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/ylimits.py +3 -2
  14. {scistackplot-0.1.28 → scistackplot-0.1.29}/README.md +0 -0
  15. {scistackplot-0.1.28 → scistackplot-0.1.29}/pyproject.toml +0 -0
  16. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/__init__.py +0 -0
  17. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/codegen.py +0 -0
  18. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/groups.py +0 -0
  19. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/render/__init__.py +0 -0
  20. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/render/base.py +0 -0
  21. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/render/mpl.py +0 -0
  22. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/render/plotly_.py +0 -0
  23. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/resolved.py +0 -0
  24. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/shape.py +0 -0
  25. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/sources/__init__.py +0 -0
  26. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/sources/csv.py +0 -0
  27. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/sources/frame.py +0 -0
  28. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/spec.py +0 -0
  29. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/variants.py +0 -0
  30. {scistackplot-0.1.28 → scistackplot-0.1.29}/src/scistackplot/xaxis.py +0 -0
@@ -30,3 +30,4 @@ scistack-gui/frontend/dist/
30
30
  # per-run files, not source.
31
31
  /exports/
32
32
  *.log
33
+ output.txt
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: scistackplot
3
- Version: 0.1.28
3
+ Version: 0.1.29
4
4
  Summary: Spec-driven plotting for long-format scientific data — standalone, GUI-friendly
5
5
  Author: SciStack Contributors
6
6
  License-Expression: MIT
@@ -24,17 +24,18 @@ DISTRIBUTION_KINDS = (PlotKind.BOX, PlotKind.VIOLIN, PlotKind.BAR, PlotKind.BAND
24
24
 
25
25
  #: What each role is CALLED, per measure shape.
26
26
  #:
27
- #: The role names are the library's vocabulary; these are the user's. Two of
28
- #: them read as nonsense on 1-D data under the generic wording — "Average over"
29
- #: and "Replicates" describe what happens to a table, and what the user is
30
- #: looking at is a set of traces. Reported by the backend rather than hardcoded
31
- #: in the panel so the words and the behaviour stay together (CLAUDE.md NOTE 3).
27
+ #: The role names are the library's vocabulary; these are the user's — and the
28
+ #: two are kept the SAME wherever a user has to search for one (see the FREE
29
+ #: entry in _ROLE_LABELS). What varies by shape is what a role DOES: "Average
30
+ #: over" describes a table, and a user looking at 1-D data is looking at
31
+ #: traces, so AGGREGATE is renamed for them. Reported by the backend rather
32
+ #: than hardcoded in the panel so the words and the behaviour stay together
33
+ #: (CLAUDE.md NOTE 3); the per-shape explanation is _ROLE_HINTS_BY_SHAPE.
32
34
  #:
33
35
  #: Only the entries that differ from :data:`_ROLE_LABELS` need listing.
34
36
  _ROLE_LABELS_BY_SHAPE: dict[Shape, dict[Role, str]] = {
35
37
  Shape.SERIES_1D: {
36
38
  Role.AGGREGATE: "Average into one line",
37
- Role.FREE: "One line each",
38
39
  },
39
40
  }
40
41
 
@@ -67,7 +68,14 @@ _ROLE_LABELS: dict[Role, str] = {
67
68
  Role.FACET: "Facet",
68
69
  Role.ITERATE: "Separate figures",
69
70
  Role.AGGREGATE: "Average over",
70
- Role.FREE: "Replicates",
71
+ # "Free", not "Replicates" / "One line each". The role's own name, kept the
72
+ # same for every shape, because a user hunting for it in the dropdown has
73
+ # read it in the docs, in a saved spec's TOML and in an exported `roles=`
74
+ # argument — and found nothing matching (user, 2026-09-13). It is also the
75
+ # role whose meaning a noun cannot carry on its own: what a FREE factor
76
+ # does depends on the plot kind (one line each, a bar's error bars, a box's
77
+ # distribution), which is what the HINT is for, per shape.
78
+ Role.FREE: "Free",
71
79
  }
72
80
 
73
81
  _ROLE_HINTS: dict[Role, str] = {
@@ -75,15 +83,33 @@ _ROLE_HINTS: dict[Role, str] = {
75
83
  Role.COLOR: "One coloured series per level",
76
84
  Role.FACET: "One subplot per level — arrange them under Layout",
77
85
  Role.ITERATE: "One whole figure per level",
78
- Role.AGGREGATE: "Collapse this factor to its mean",
79
- Role.FREE: "Keep as repeated observations",
86
+ Role.AGGREGATE: "Collapse to the mean first, so this factor does NOT "
87
+ "widen the error bars",
88
+ Role.FREE: "Keep each level as its own observation, so this factor DOES "
89
+ "widen the error bars",
80
90
  }
81
91
 
92
+ #: Stated as a CONTRAST, and deliberately in the same terms on both sides.
93
+ #:
94
+ #: These are the two roles a user cannot tell apart from the labels alone —
95
+ #: "Average over" and "Free" both sound like "not on an axis", and the earlier
96
+ #: hints described each one on its own, both using the word "average" (user,
97
+ #: 2026-09-13). The thing that actually distinguishes them is what they do to
98
+ #: the ERROR BARS: with schema [subject, trial] and a band, trial=AGGREGATE
99
+ #: averages each subject's trials into one trace and the band is then the
100
+ #: spread across SUBJECTS; trial=FREE pools every subject-trial trace, so the
101
+ #: band mixes within- and between-subject variability and a subject with more
102
+ #: trials weighs more. Same plot kind, same data, different published numbers —
103
+ #: which is why the plot-kind control cannot express this and both roles exist
104
+ #: (`reduce._collapse_aggregates` runs before `_summarize`; `has_replicates`
105
+ #: is true only when something is FREE, which is the part the kind list DOES
106
+ #: express — collapse everything and the summary kinds disappear).
82
107
  _ROLE_HINTS_BY_SHAPE: dict[Shape, dict[Role, str]] = {
83
108
  Shape.SERIES_1D: {
84
- Role.AGGREGATE: "Average these traces together, sample by sample",
85
- Role.FREE: "Draw one trace per level — and what a mean ± error band "
86
- "is computed from",
109
+ Role.AGGREGATE: "Average these traces together sample by sample "
110
+ "first, so this factor does NOT widen the error band",
111
+ Role.FREE: "One trace per level, each its own observation — so this "
112
+ "factor DOES widen the error band",
87
113
  },
88
114
  }
89
115
 
@@ -0,0 +1,93 @@
1
+ """Run a keyed build ONCE even when several threads ask for it at the same time.
2
+
3
+ Memoizing a build only helps the second caller if the first one has *finished*.
4
+ The panel does not work that way: opening it fires ``plot_describe``,
5
+ ``plot_resolve``, ``plot_capabilities`` and ``plot_location_tree`` together, the
6
+ server runs a thread per request, and every one of them asks for the same table.
7
+
8
+ Measured on 2026-09-13: two requests both logged ``table cache MISS`` and both
9
+ built the identical 5.2 GB table, concurrently — 188.8 s and 104.1 s, finishing
10
+ within 100 ms of each other (.claude/plot-at-scale-plan.md §7.1). The memo was
11
+ written after ``_build_table`` returned, so the second arrival saw an empty cache
12
+ and repeated all of it. A classic cache stampede; the 2026-09-11 memoization work
13
+ fixed sequential repeat cost and never addressed concurrent arrival.
14
+
15
+ :class:`SingleFlight` closes that: the first caller for a key builds, everyone
16
+ else waits on the same result and pays nothing.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import threading
22
+ from concurrent.futures import Future
23
+ from typing import Any, Callable
24
+
25
+
26
+ class SingleFlight:
27
+ """Deduplicate concurrent builds by key.
28
+
29
+ One instance per cache. Keys must be hashable and must mean the same thing
30
+ the cache's own key means — a narrower key here would merge builds that are
31
+ not interchangeable.
32
+
33
+ Thread-safe. The internal lock is held only while claiming or releasing a
34
+ key, never while building, so two DIFFERENT keys still build in parallel.
35
+ """
36
+
37
+ def __init__(self) -> None:
38
+ self._lock = threading.Lock()
39
+ self._inflight: dict[Any, Future] = {}
40
+
41
+ def run(
42
+ self,
43
+ key: Any,
44
+ build: Callable[[], Any],
45
+ *,
46
+ on_wait: Callable[[], None] | None = None,
47
+ ) -> Any:
48
+ """``build()``'s result, computing it once across concurrent callers.
49
+
50
+ The first caller for ``key`` runs ``build``; any caller arriving while
51
+ that is in flight blocks until it finishes and gets the same object.
52
+
53
+ ``on_wait`` is called (once, by each waiter) instead of building, for
54
+ logging — a waiter that silently returns the right answer is
55
+ indistinguishable in a log from a cache hit, and those have very
56
+ different meanings when reading a slow request.
57
+
58
+ A failing ``build`` raises in the owner AND in every waiter: the waiters
59
+ asked for a value that could not be produced, and swallowing it here
60
+ would hand them a wrong answer or a hang. The key is released either
61
+ way, so a later attempt is free to retry.
62
+ """
63
+ with self._lock:
64
+ future = self._inflight.get(key)
65
+ owner = future is None
66
+ if owner:
67
+ future = Future()
68
+ self._inflight[key] = future
69
+
70
+ if not owner:
71
+ if on_wait is not None:
72
+ on_wait()
73
+ # Raises here if the owner's build failed — by design, see above.
74
+ return future.result()
75
+
76
+ try:
77
+ result = build()
78
+ except BaseException as exc:
79
+ # Release BEFORE publishing the failure so a waiter that wakes and
80
+ # immediately retries finds a clear slot rather than this dead one.
81
+ with self._lock:
82
+ self._inflight.pop(key, None)
83
+ future.set_exception(exc)
84
+ raise
85
+ with self._lock:
86
+ self._inflight.pop(key, None)
87
+ future.set_result(result)
88
+ return result
89
+
90
+ def in_flight(self) -> int:
91
+ """How many keys are building right now. For tests and diagnostics."""
92
+ with self._lock:
93
+ return len(self._inflight)
@@ -0,0 +1,105 @@
1
+ """How big is this frame, really — in cells AND in samples.
2
+
3
+ A row count is the wrong unit for a table whose cells hold signals. 4190 rows of
4
+ a 200-sample array and 4190 rows of a 250,000-sample array are the same row
5
+ count and four orders of magnitude apart in work, and every load/reduce log line
6
+ in this package reported only rows. On 2026-09-13 that made a plot that never
7
+ returned indistinguishable, from the log alone, from a plot over a small table
8
+ (.claude/plot-at-scale-plan.md §1).
9
+
10
+ So one helper, used by every layer that reports frame size, so `load_variable`
11
+ in ``scistackplotdb`` and the reduce path here cannot describe the same frame in
12
+ different units.
13
+
14
+ **Cost.** ``len()`` on a list or ndarray is O(1), so measuring is one pass over
15
+ the cells of the measure columns — cheap enough to run unconditionally on a load
16
+ that is already touching every one of those cells. It deliberately does NOT sum
17
+ ``nbytes`` or walk into the values: the point is to size the work, not to audit
18
+ memory exactly.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import TYPE_CHECKING, Any
24
+
25
+ if TYPE_CHECKING: # pragma: no cover - typing only
26
+ import pandas as pd
27
+
28
+ #: Rough bytes per element for a Python ``list[float]`` cell: the float object
29
+ #: (24 on CPython 64-bit) plus the list's pointer slot (8). Used only to turn a
30
+ #: sample count into an order-of-magnitude memory figure in a log line, which is
31
+ #: what distinguishes "big" from "hopeless" at a glance.
32
+ PYLIST_BYTES_PER_SAMPLE = 32
33
+
34
+ #: Bytes per element once a cell is a float64 ndarray. The gap between this and
35
+ #: :data:`PYLIST_BYTES_PER_SAMPLE` is the prize for fetching via Arrow.
36
+ NUMPY_BYTES_PER_SAMPLE = 8
37
+
38
+
39
+ def cell_samples(value: Any) -> int:
40
+ """Number of samples in one cell: its length, or 1 for a scalar.
41
+
42
+ ``None`` and strings count as 0 — neither is a sample, and a string's length
43
+ would be a character count masquerading as one.
44
+ """
45
+ if value is None or isinstance(value, (str, bytes)):
46
+ return 0
47
+ try:
48
+ return len(value)
49
+ except TypeError:
50
+ return 1 # a scalar is one sample
51
+
52
+
53
+ def frame_extent(frame: "pd.DataFrame", columns: list[str] | None = None) -> dict:
54
+ """``{rows, cells, samples, est_bytes, boxed}`` for ``frame``.
55
+
56
+ ``columns`` limits the scan to the measure columns worth measuring; omitted,
57
+ every column is scanned, which is only appropriate for a small frame.
58
+
59
+ ``boxed`` is True when the first sampled cell is a Python ``list``/``tuple``
60
+ rather than an ndarray — i.e. the payload is carried as boxed Python floats.
61
+ That single flag is what says whether an Arrow fetch path would help, so it
62
+ is worth a word in the log even though it is only a sample.
63
+ """
64
+ rows = len(frame)
65
+ names = [c for c in (columns or list(frame.columns)) if c in frame.columns]
66
+ cells = 0
67
+ samples = 0
68
+ boxed = False
69
+ seen_a_cell = False
70
+ for name in names:
71
+ series = frame[name]
72
+ for value in series.to_numpy():
73
+ if value is None:
74
+ continue
75
+ n = cell_samples(value)
76
+ if n == 0:
77
+ continue
78
+ cells += 1
79
+ samples += n
80
+ if not seen_a_cell:
81
+ boxed = isinstance(value, (list, tuple))
82
+ seen_a_cell = True
83
+ per_sample = PYLIST_BYTES_PER_SAMPLE if boxed else NUMPY_BYTES_PER_SAMPLE
84
+ return {
85
+ "rows": rows,
86
+ "cells": cells,
87
+ "samples": samples,
88
+ "est_bytes": samples * per_sample,
89
+ "boxed": boxed,
90
+ }
91
+
92
+
93
+ def format_extent(extent: dict) -> str:
94
+ """One log-line fragment: ``rows=419 cells=4190 samples=1.1e+09 ~34.0GB boxed``.
95
+
96
+ Samples go in scientific notation because the interesting range spans six
97
+ orders of magnitude and the exact digits never matter.
98
+ """
99
+ gb = extent["est_bytes"] / 1024**3
100
+ size = f"~{gb:.1f}GB" if gb >= 0.1 else f"~{extent['est_bytes'] / 1024**2:.0f}MB"
101
+ return (
102
+ f"rows={extent['rows']} cells={extent['cells']} "
103
+ f"samples={extent['samples']:.3g} {size}"
104
+ + (" boxed" if extent["boxed"] else " ndarray")
105
+ )
@@ -0,0 +1,61 @@
1
+ """``pd.to_numeric(errors="coerce")`` that survives numpy scalars.
2
+
3
+ Once a source hands a 1-D cell over as an ``np.ndarray`` rather than a list,
4
+ ``DataFrame.explode`` yields one ``np.float64`` OBJECT per sample instead of a
5
+ Python float. pandas' ``maybe_convert_numeric`` (pandas 3.x, numpy 2.x) calls
6
+ ``len()`` on any value that has ``__len__`` — numpy scalars do, and raise
7
+ ``TypeError: len() of unsized object``. Found 2026-09-13 the moment the DuckDB
8
+ fetch stopped boxing (``scistackplotdb`` ``test_load_fetch.py``); every
9
+ ``to_numeric`` downstream of an explode or over a cell column goes through here.
10
+
11
+ The fast path is also the point: a column of numpy scalars casts to float64 in
12
+ one C-level pass, where ``to_numeric`` was inspecting each object.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import Any
18
+
19
+ import numpy as np
20
+ import pandas as pd
21
+
22
+
23
+ def coerce_numeric(values: pd.Series) -> pd.Series:
24
+ """``values`` as float64, non-numeric entries NaN — never raising on a
25
+ numpy scalar, and never stacking a column of array cells into a matrix."""
26
+ if pd.api.types.is_numeric_dtype(values):
27
+ return values
28
+ raw = values.to_numpy()
29
+ sample = _first_present(raw)
30
+ if isinstance(sample, (np.generic, int, float)) and not isinstance(sample, bool):
31
+ # Scalars (numpy or Python): one vectorised cast. Anything that is not
32
+ # a number in the column raises here and falls through to the coercing
33
+ # path below, which is what `errors="coerce"` promised.
34
+ try:
35
+ return pd.Series(raw.astype("float64"), index=values.index)
36
+ except (TypeError, ValueError):
37
+ pass
38
+ return pd.to_numeric(values.map(_unboxed), errors="coerce")
39
+
40
+
41
+ def _first_present(raw: np.ndarray) -> Any:
42
+ for value in raw:
43
+ if value is None:
44
+ continue
45
+ if isinstance(value, float) and value != value:
46
+ continue
47
+ return value
48
+ return None
49
+
50
+
51
+ def _unboxed(value: Any) -> Any:
52
+ """A numpy scalar as its Python value; a list-like cell as None.
53
+
54
+ ``to_numeric(errors="coerce")`` turned a list cell into NaN; an ndarray
55
+ cell must land in the same place rather than in ``len()``.
56
+ """
57
+ if isinstance(value, np.generic):
58
+ return value.item()
59
+ if isinstance(value, (list, tuple, np.ndarray)):
60
+ return None
61
+ return value