scistackplot 0.1.26__tar.gz → 0.1.29__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scistackplot-0.1.26 → scistackplot-0.1.29}/.gitignore +5 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/PKG-INFO +2 -1
- {scistackplot-0.1.26 → scistackplot-0.1.29}/pyproject.toml +1 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/__init__.py +2 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/capability.py +43 -14
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/codegen.py +57 -0
- scistackplot-0.1.29/src/scistackplot/dedup.py +93 -0
- scistackplot-0.1.29/src/scistackplot/framesize.py +105 -0
- scistackplot-0.1.29/src/scistackplot/numeric.py +61 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/reduce.py +290 -49
- scistackplot-0.1.29/src/scistackplot/reducer.py +506 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/roles.py +30 -12
- scistackplot-0.1.29/src/scistackplot/series_stats.py +204 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/base.py +89 -13
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/spec.py +90 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/table.py +11 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/ylimits.py +3 -2
- {scistackplot-0.1.26 → scistackplot-0.1.29}/README.md +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/groups.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/__init__.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/base.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/mpl.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/plotly_.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/resolved.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/shape.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/__init__.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/csv.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/frame.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/variants.py +0 -0
- {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/xaxis.py +0 -0
|
@@ -20,9 +20,14 @@ __pycache__/
|
|
|
20
20
|
*.duckdb.wal
|
|
21
21
|
*.wal
|
|
22
22
|
scistack-gui/frontend/node_modules/
|
|
23
|
+
# Compiled output of the frontend's React-free unit tests (npm test in
|
|
24
|
+
# scistack-gui/frontend). Regenerated by `tsc -p tsconfig.test.json`; unlike
|
|
25
|
+
# extension/dist/, nothing loads it at runtime.
|
|
26
|
+
scistack-gui/frontend/dist/
|
|
23
27
|
# mkdocs build output (regenerated by `mkdocs build`)
|
|
24
28
|
/site/
|
|
25
29
|
# Runtime output of code_export_service (pipeline-to-code export) — timestamped
|
|
26
30
|
# per-run files, not source.
|
|
27
31
|
/exports/
|
|
28
32
|
*.log
|
|
33
|
+
output.txt
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: scistackplot
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.29
|
|
4
4
|
Summary: Spec-driven plotting for long-format scientific data — standalone, GUI-friendly
|
|
5
5
|
Author: SciStack Contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -13,6 +13,7 @@ Classifier: Programming Language :: Python :: 3
|
|
|
13
13
|
Classifier: Programming Language :: Python :: 3.10
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.11
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
17
|
Classifier: Topic :: Scientific/Engineering :: Visualization
|
|
17
18
|
Classifier: Typing :: Typed
|
|
18
19
|
Requires-Python: >=3.10
|
|
@@ -32,6 +32,7 @@ classifiers = [
|
|
|
32
32
|
"Programming Language :: Python :: 3.10",
|
|
33
33
|
"Programming Language :: Python :: 3.11",
|
|
34
34
|
"Programming Language :: Python :: 3.12",
|
|
35
|
+
"Programming Language :: Python :: 3.13",
|
|
35
36
|
"Topic :: Scientific/Engineering :: Visualization",
|
|
36
37
|
"Typing :: Typed",
|
|
37
38
|
]
|
|
@@ -68,6 +68,7 @@ from .spec import (
|
|
|
68
68
|
FacetOptions,
|
|
69
69
|
Filter,
|
|
70
70
|
LevelGroup,
|
|
71
|
+
LocationFilter,
|
|
71
72
|
MatchOp,
|
|
72
73
|
Matcher,
|
|
73
74
|
PlotKind,
|
|
@@ -122,6 +123,7 @@ __all__ = [
|
|
|
122
123
|
"StyleOptions",
|
|
123
124
|
"Filter",
|
|
124
125
|
"LevelGroup",
|
|
126
|
+
"LocationFilter",
|
|
125
127
|
"VariantSet",
|
|
126
128
|
# data
|
|
127
129
|
"LongTable",
|
|
@@ -24,17 +24,18 @@ DISTRIBUTION_KINDS = (PlotKind.BOX, PlotKind.VIOLIN, PlotKind.BAR, PlotKind.BAND
|
|
|
24
24
|
|
|
25
25
|
#: What each role is CALLED, per measure shape.
|
|
26
26
|
#:
|
|
27
|
-
#: The role names are the library's vocabulary; these are the user's
|
|
28
|
-
#:
|
|
29
|
-
#:
|
|
30
|
-
#:
|
|
31
|
-
#:
|
|
27
|
+
#: The role names are the library's vocabulary; these are the user's — and the
|
|
28
|
+
#: two are kept the SAME wherever a user has to search for one (see the FREE
|
|
29
|
+
#: entry in _ROLE_LABELS). What varies by shape is what a role DOES: "Average
|
|
30
|
+
#: over" describes a table, and a user looking at 1-D data is looking at
|
|
31
|
+
#: traces, so AGGREGATE is renamed for them. Reported by the backend rather
|
|
32
|
+
#: than hardcoded in the panel so the words and the behaviour stay together
|
|
33
|
+
#: (CLAUDE.md NOTE 3); the per-shape explanation is _ROLE_HINTS_BY_SHAPE.
|
|
32
34
|
#:
|
|
33
35
|
#: Only the entries that differ from :data:`_ROLE_LABELS` need listing.
|
|
34
36
|
_ROLE_LABELS_BY_SHAPE: dict[Shape, dict[Role, str]] = {
|
|
35
37
|
Shape.SERIES_1D: {
|
|
36
38
|
Role.AGGREGATE: "Average into one line",
|
|
37
|
-
Role.FREE: "One line each",
|
|
38
39
|
},
|
|
39
40
|
}
|
|
40
41
|
|
|
@@ -67,7 +68,14 @@ _ROLE_LABELS: dict[Role, str] = {
|
|
|
67
68
|
Role.FACET: "Facet",
|
|
68
69
|
Role.ITERATE: "Separate figures",
|
|
69
70
|
Role.AGGREGATE: "Average over",
|
|
70
|
-
|
|
71
|
+
# "Free", not "Replicates" / "One line each". The role's own name, kept the
|
|
72
|
+
# same for every shape, because a user hunting for it in the dropdown has
|
|
73
|
+
# read it in the docs, in a saved spec's TOML and in an exported `roles=`
|
|
74
|
+
# argument — and found nothing matching (user, 2026-09-13). It is also the
|
|
75
|
+
# role whose meaning a noun cannot carry on its own: what a FREE factor
|
|
76
|
+
# does depends on the plot kind (one line each, a bar's error bars, a box's
|
|
77
|
+
# distribution), which is what the HINT is for, per shape.
|
|
78
|
+
Role.FREE: "Free",
|
|
71
79
|
}
|
|
72
80
|
|
|
73
81
|
_ROLE_HINTS: dict[Role, str] = {
|
|
@@ -75,15 +83,33 @@ _ROLE_HINTS: dict[Role, str] = {
|
|
|
75
83
|
Role.COLOR: "One coloured series per level",
|
|
76
84
|
Role.FACET: "One subplot per level — arrange them under Layout",
|
|
77
85
|
Role.ITERATE: "One whole figure per level",
|
|
78
|
-
Role.AGGREGATE: "Collapse this factor
|
|
79
|
-
|
|
86
|
+
Role.AGGREGATE: "Collapse to the mean first, so this factor does NOT "
|
|
87
|
+
"widen the error bars",
|
|
88
|
+
Role.FREE: "Keep each level as its own observation, so this factor DOES "
|
|
89
|
+
"widen the error bars",
|
|
80
90
|
}
|
|
81
91
|
|
|
92
|
+
#: Stated as a CONTRAST, and deliberately in the same terms on both sides.
|
|
93
|
+
#:
|
|
94
|
+
#: These are the two roles a user cannot tell apart from the labels alone —
|
|
95
|
+
#: "Average over" and "Free" both sound like "not on an axis", and the earlier
|
|
96
|
+
#: hints described each one on its own, both using the word "average" (user,
|
|
97
|
+
#: 2026-09-13). The thing that actually distinguishes them is what they do to
|
|
98
|
+
#: the ERROR BARS: with schema [subject, trial] and a band, trial=AGGREGATE
|
|
99
|
+
#: averages each subject's trials into one trace and the band is then the
|
|
100
|
+
#: spread across SUBJECTS; trial=FREE pools every subject-trial trace, so the
|
|
101
|
+
#: band mixes within- and between-subject variability and a subject with more
|
|
102
|
+
#: trials weighs more. Same plot kind, same data, different published numbers —
|
|
103
|
+
#: which is why the plot-kind control cannot express this and both roles exist
|
|
104
|
+
#: (`reduce._collapse_aggregates` runs before `_summarize`; `has_replicates`
|
|
105
|
+
#: is true only when something is FREE, which is the part the kind list DOES
|
|
106
|
+
#: express — collapse everything and the summary kinds disappear).
|
|
82
107
|
_ROLE_HINTS_BY_SHAPE: dict[Shape, dict[Role, str]] = {
|
|
83
108
|
Shape.SERIES_1D: {
|
|
84
|
-
Role.AGGREGATE: "Average these traces together
|
|
85
|
-
|
|
86
|
-
"
|
|
109
|
+
Role.AGGREGATE: "Average these traces together sample by sample "
|
|
110
|
+
"first, so this factor does NOT widen the error band",
|
|
111
|
+
Role.FREE: "One trace per level, each its own observation — so this "
|
|
112
|
+
"factor DOES widen the error band",
|
|
87
113
|
},
|
|
88
114
|
}
|
|
89
115
|
|
|
@@ -402,9 +428,12 @@ def factor_summary(spec: PlotSpec, derived: LongTable) -> list[dict]:
|
|
|
402
428
|
entry["x_available"] = on_x["available"]
|
|
403
429
|
entry["x_reason"] = on_x["reason"]
|
|
404
430
|
|
|
405
|
-
if not spec.filters:
|
|
431
|
+
if not spec.filters and spec.location_filter.is_empty():
|
|
406
432
|
# Nothing filtered: everything is selected, and no frame scan is needed
|
|
407
|
-
# on the common path.
|
|
433
|
+
# on the common path. The location filter has to be checked here too —
|
|
434
|
+
# it narrows rows exactly as a Filter does, and a fast path that only
|
|
435
|
+
# knew about one of them would report "all 12 selected" beside a figure
|
|
436
|
+
# drawing 3, which is the precise failure this readout exists to avoid.
|
|
408
437
|
for entry in factors:
|
|
409
438
|
entry["selected"] = list(entry["levels"])
|
|
410
439
|
return factors
|
|
@@ -352,6 +352,14 @@ def _preamble(spec, table, roles, shape) -> list[str]:
|
|
|
352
352
|
]
|
|
353
353
|
)
|
|
354
354
|
|
|
355
|
+
# Schema locations. Emitted as one vectorised comparison per (prefix, key),
|
|
356
|
+
# which is the SAME rule reduce._location_mask applies — a ragged selection
|
|
357
|
+
# cannot be expressed as `for_each(subject=[...], trial=[...])`, whose keys
|
|
358
|
+
# cross-product, so it has to be a mask inside the function body. The
|
|
359
|
+
# `if _k in df.columns` guard is not defensive clutter: it is how a key the
|
|
360
|
+
# frame lacks goes unconstrained, matching the interactive path exactly.
|
|
361
|
+
lines.extend(_location_lines(spec))
|
|
362
|
+
|
|
355
363
|
filter_lines: list[str] = []
|
|
356
364
|
for flt in spec.filters:
|
|
357
365
|
if flt.include is not None:
|
|
@@ -467,6 +475,32 @@ def _preamble(spec, table, roles, shape) -> list[str]:
|
|
|
467
475
|
return lines
|
|
468
476
|
|
|
469
477
|
|
|
478
|
+
def _location_lines(spec: PlotSpec) -> list[str]:
|
|
479
|
+
"""Generated pandas for ``spec.location_filter`` — empty when it is inert.
|
|
480
|
+
|
|
481
|
+
Kept beside the other preamble emitters rather than inlined so the one test
|
|
482
|
+
that matters can compare its output against
|
|
483
|
+
:func:`scistackplot.reduce._location_mask` on the same frame.
|
|
484
|
+
"""
|
|
485
|
+
prefixes = spec.location_filter.prefixes()
|
|
486
|
+
if not prefixes:
|
|
487
|
+
return []
|
|
488
|
+
plural = "" if len(prefixes) == 1 else "s"
|
|
489
|
+
return [
|
|
490
|
+
f"# schema locations: {len(prefixes)} selection{plural}",
|
|
491
|
+
f"_loc_prefixes = {[[list(pair) for pair in p] for p in prefixes]!r}",
|
|
492
|
+
"_loc_mask = pd.Series(False, index=df.index)",
|
|
493
|
+
"for _p in _loc_prefixes:",
|
|
494
|
+
" _m = pd.Series(True, index=df.index)",
|
|
495
|
+
" for _k, _v in _p:",
|
|
496
|
+
" if _k in df.columns:",
|
|
497
|
+
" _m &= df[_k].astype(str) == _v",
|
|
498
|
+
" _loc_mask |= _m",
|
|
499
|
+
"df = df[_loc_mask]",
|
|
500
|
+
"",
|
|
501
|
+
]
|
|
502
|
+
|
|
503
|
+
|
|
470
504
|
def _facet_layout_args(spec, table: LongTable, facets: list[str]) -> list[str]:
|
|
471
505
|
"""
|
|
472
506
|
seaborn arguments that reproduce the interactive facet arrangement.
|
|
@@ -754,6 +788,7 @@ def _color_level_count(spec: PlotSpec, table: LongTable, color: str) -> int:
|
|
|
754
788
|
levels = [str(level) for level in table.factor(color).levels]
|
|
755
789
|
except KeyError:
|
|
756
790
|
return 2
|
|
791
|
+
levels = _levels_after_location(spec, color, levels)
|
|
757
792
|
for flt in spec.filters:
|
|
758
793
|
if flt.column != color:
|
|
759
794
|
continue
|
|
@@ -766,6 +801,28 @@ def _color_level_count(spec: PlotSpec, table: LongTable, color: str) -> int:
|
|
|
766
801
|
return len(levels)
|
|
767
802
|
|
|
768
803
|
|
|
804
|
+
def _levels_after_location(spec: PlotSpec, column: str, levels: list[str]) -> list[str]:
|
|
805
|
+
"""``levels`` narrowed by ``spec.location_filter``, for one schema key.
|
|
806
|
+
|
|
807
|
+
The rule follows from prefixes constraining only the keys they name: a
|
|
808
|
+
prefix that never mentions this column says nothing about it, so if any
|
|
809
|
+
such prefix is selected the column keeps every level. Only when *every*
|
|
810
|
+
prefix names it do the values they give become the surviving set.
|
|
811
|
+
|
|
812
|
+
Concretely, selecting all of subject 01 plus subject 02's trial 3 leaves
|
|
813
|
+
``trial`` with all its levels — because "all of subject 01" did not name a
|
|
814
|
+
trial — which is what the figure will actually draw.
|
|
815
|
+
"""
|
|
816
|
+
prefixes = spec.location_filter.prefixes()
|
|
817
|
+
if not prefixes:
|
|
818
|
+
return levels
|
|
819
|
+
named = [dict(prefix) for prefix in prefixes]
|
|
820
|
+
if any(column not in prefix for prefix in named):
|
|
821
|
+
return levels
|
|
822
|
+
keep = {prefix[column] for prefix in named}
|
|
823
|
+
return [level for level in levels if level in keep]
|
|
824
|
+
|
|
825
|
+
|
|
769
826
|
#: Column the generated code builds for a nested x axis.
|
|
770
827
|
_X_NESTED = "_x"
|
|
771
828
|
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Run a keyed build ONCE even when several threads ask for it at the same time.
|
|
2
|
+
|
|
3
|
+
Memoizing a build only helps the second caller if the first one has *finished*.
|
|
4
|
+
The panel does not work that way: opening it fires ``plot_describe``,
|
|
5
|
+
``plot_resolve``, ``plot_capabilities`` and ``plot_location_tree`` together, the
|
|
6
|
+
server runs a thread per request, and every one of them asks for the same table.
|
|
7
|
+
|
|
8
|
+
Measured on 2026-09-13: two requests both logged ``table cache MISS`` and both
|
|
9
|
+
built the identical 5.2 GB table, concurrently — 188.8 s and 104.1 s, finishing
|
|
10
|
+
within 100 ms of each other (.claude/plot-at-scale-plan.md §7.1). The memo was
|
|
11
|
+
written after ``_build_table`` returned, so the second arrival saw an empty cache
|
|
12
|
+
and repeated all of it. A classic cache stampede; the 2026-09-11 memoization work
|
|
13
|
+
fixed sequential repeat cost and never addressed concurrent arrival.
|
|
14
|
+
|
|
15
|
+
:class:`SingleFlight` closes that: the first caller for a key builds, everyone
|
|
16
|
+
else waits on the same result and pays nothing.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import threading
|
|
22
|
+
from concurrent.futures import Future
|
|
23
|
+
from typing import Any, Callable
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class SingleFlight:
|
|
27
|
+
"""Deduplicate concurrent builds by key.
|
|
28
|
+
|
|
29
|
+
One instance per cache. Keys must be hashable and must mean the same thing
|
|
30
|
+
the cache's own key means — a narrower key here would merge builds that are
|
|
31
|
+
not interchangeable.
|
|
32
|
+
|
|
33
|
+
Thread-safe. The internal lock is held only while claiming or releasing a
|
|
34
|
+
key, never while building, so two DIFFERENT keys still build in parallel.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
def __init__(self) -> None:
|
|
38
|
+
self._lock = threading.Lock()
|
|
39
|
+
self._inflight: dict[Any, Future] = {}
|
|
40
|
+
|
|
41
|
+
def run(
|
|
42
|
+
self,
|
|
43
|
+
key: Any,
|
|
44
|
+
build: Callable[[], Any],
|
|
45
|
+
*,
|
|
46
|
+
on_wait: Callable[[], None] | None = None,
|
|
47
|
+
) -> Any:
|
|
48
|
+
"""``build()``'s result, computing it once across concurrent callers.
|
|
49
|
+
|
|
50
|
+
The first caller for ``key`` runs ``build``; any caller arriving while
|
|
51
|
+
that is in flight blocks until it finishes and gets the same object.
|
|
52
|
+
|
|
53
|
+
``on_wait`` is called (once, by each waiter) instead of building, for
|
|
54
|
+
logging — a waiter that silently returns the right answer is
|
|
55
|
+
indistinguishable in a log from a cache hit, and those have very
|
|
56
|
+
different meanings when reading a slow request.
|
|
57
|
+
|
|
58
|
+
A failing ``build`` raises in the owner AND in every waiter: the waiters
|
|
59
|
+
asked for a value that could not be produced, and swallowing it here
|
|
60
|
+
would hand them a wrong answer or a hang. The key is released either
|
|
61
|
+
way, so a later attempt is free to retry.
|
|
62
|
+
"""
|
|
63
|
+
with self._lock:
|
|
64
|
+
future = self._inflight.get(key)
|
|
65
|
+
owner = future is None
|
|
66
|
+
if owner:
|
|
67
|
+
future = Future()
|
|
68
|
+
self._inflight[key] = future
|
|
69
|
+
|
|
70
|
+
if not owner:
|
|
71
|
+
if on_wait is not None:
|
|
72
|
+
on_wait()
|
|
73
|
+
# Raises here if the owner's build failed — by design, see above.
|
|
74
|
+
return future.result()
|
|
75
|
+
|
|
76
|
+
try:
|
|
77
|
+
result = build()
|
|
78
|
+
except BaseException as exc:
|
|
79
|
+
# Release BEFORE publishing the failure so a waiter that wakes and
|
|
80
|
+
# immediately retries finds a clear slot rather than this dead one.
|
|
81
|
+
with self._lock:
|
|
82
|
+
self._inflight.pop(key, None)
|
|
83
|
+
future.set_exception(exc)
|
|
84
|
+
raise
|
|
85
|
+
with self._lock:
|
|
86
|
+
self._inflight.pop(key, None)
|
|
87
|
+
future.set_result(result)
|
|
88
|
+
return result
|
|
89
|
+
|
|
90
|
+
def in_flight(self) -> int:
|
|
91
|
+
"""How many keys are building right now. For tests and diagnostics."""
|
|
92
|
+
with self._lock:
|
|
93
|
+
return len(self._inflight)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""How big is this frame, really — in cells AND in samples.
|
|
2
|
+
|
|
3
|
+
A row count is the wrong unit for a table whose cells hold signals. 4190 rows of
|
|
4
|
+
a 200-sample array and 4190 rows of a 250,000-sample array are the same row
|
|
5
|
+
count and four orders of magnitude apart in work, and every load/reduce log line
|
|
6
|
+
in this package reported only rows. On 2026-09-13 that made a plot that never
|
|
7
|
+
returned indistinguishable, from the log alone, from a plot over a small table
|
|
8
|
+
(.claude/plot-at-scale-plan.md §1).
|
|
9
|
+
|
|
10
|
+
So one helper, used by every layer that reports frame size, so `load_variable`
|
|
11
|
+
in ``scistackplotdb`` and the reduce path here cannot describe the same frame in
|
|
12
|
+
different units.
|
|
13
|
+
|
|
14
|
+
**Cost.** ``len()`` on a list or ndarray is O(1), so measuring is one pass over
|
|
15
|
+
the cells of the measure columns — cheap enough to run unconditionally on a load
|
|
16
|
+
that is already touching every one of those cells. It deliberately does NOT sum
|
|
17
|
+
``nbytes`` or walk into the values: the point is to size the work, not to audit
|
|
18
|
+
memory exactly.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from typing import TYPE_CHECKING, Any
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING: # pragma: no cover - typing only
|
|
26
|
+
import pandas as pd
|
|
27
|
+
|
|
28
|
+
#: Rough bytes per element for a Python ``list[float]`` cell: the float object
|
|
29
|
+
#: (24 on CPython 64-bit) plus the list's pointer slot (8). Used only to turn a
|
|
30
|
+
#: sample count into an order-of-magnitude memory figure in a log line, which is
|
|
31
|
+
#: what distinguishes "big" from "hopeless" at a glance.
|
|
32
|
+
PYLIST_BYTES_PER_SAMPLE = 32
|
|
33
|
+
|
|
34
|
+
#: Bytes per element once a cell is a float64 ndarray. The gap between this and
|
|
35
|
+
#: :data:`PYLIST_BYTES_PER_SAMPLE` is the prize for fetching via Arrow.
|
|
36
|
+
NUMPY_BYTES_PER_SAMPLE = 8
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def cell_samples(value: Any) -> int:
|
|
40
|
+
"""Number of samples in one cell: its length, or 1 for a scalar.
|
|
41
|
+
|
|
42
|
+
``None`` and strings count as 0 — neither is a sample, and a string's length
|
|
43
|
+
would be a character count masquerading as one.
|
|
44
|
+
"""
|
|
45
|
+
if value is None or isinstance(value, (str, bytes)):
|
|
46
|
+
return 0
|
|
47
|
+
try:
|
|
48
|
+
return len(value)
|
|
49
|
+
except TypeError:
|
|
50
|
+
return 1 # a scalar is one sample
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def frame_extent(frame: "pd.DataFrame", columns: list[str] | None = None) -> dict:
|
|
54
|
+
"""``{rows, cells, samples, est_bytes, boxed}`` for ``frame``.
|
|
55
|
+
|
|
56
|
+
``columns`` limits the scan to the measure columns worth measuring; omitted,
|
|
57
|
+
every column is scanned, which is only appropriate for a small frame.
|
|
58
|
+
|
|
59
|
+
``boxed`` is True when the first sampled cell is a Python ``list``/``tuple``
|
|
60
|
+
rather than an ndarray — i.e. the payload is carried as boxed Python floats.
|
|
61
|
+
That single flag is what says whether an Arrow fetch path would help, so it
|
|
62
|
+
is worth a word in the log even though it is only a sample.
|
|
63
|
+
"""
|
|
64
|
+
rows = len(frame)
|
|
65
|
+
names = [c for c in (columns or list(frame.columns)) if c in frame.columns]
|
|
66
|
+
cells = 0
|
|
67
|
+
samples = 0
|
|
68
|
+
boxed = False
|
|
69
|
+
seen_a_cell = False
|
|
70
|
+
for name in names:
|
|
71
|
+
series = frame[name]
|
|
72
|
+
for value in series.to_numpy():
|
|
73
|
+
if value is None:
|
|
74
|
+
continue
|
|
75
|
+
n = cell_samples(value)
|
|
76
|
+
if n == 0:
|
|
77
|
+
continue
|
|
78
|
+
cells += 1
|
|
79
|
+
samples += n
|
|
80
|
+
if not seen_a_cell:
|
|
81
|
+
boxed = isinstance(value, (list, tuple))
|
|
82
|
+
seen_a_cell = True
|
|
83
|
+
per_sample = PYLIST_BYTES_PER_SAMPLE if boxed else NUMPY_BYTES_PER_SAMPLE
|
|
84
|
+
return {
|
|
85
|
+
"rows": rows,
|
|
86
|
+
"cells": cells,
|
|
87
|
+
"samples": samples,
|
|
88
|
+
"est_bytes": samples * per_sample,
|
|
89
|
+
"boxed": boxed,
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def format_extent(extent: dict) -> str:
|
|
94
|
+
"""One log-line fragment: ``rows=419 cells=4190 samples=1.1e+09 ~34.0GB boxed``.
|
|
95
|
+
|
|
96
|
+
Samples go in scientific notation because the interesting range spans six
|
|
97
|
+
orders of magnitude and the exact digits never matter.
|
|
98
|
+
"""
|
|
99
|
+
gb = extent["est_bytes"] / 1024**3
|
|
100
|
+
size = f"~{gb:.1f}GB" if gb >= 0.1 else f"~{extent['est_bytes'] / 1024**2:.0f}MB"
|
|
101
|
+
return (
|
|
102
|
+
f"rows={extent['rows']} cells={extent['cells']} "
|
|
103
|
+
f"samples={extent['samples']:.3g} {size}"
|
|
104
|
+
+ (" boxed" if extent["boxed"] else " ndarray")
|
|
105
|
+
)
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""``pd.to_numeric(errors="coerce")`` that survives numpy scalars.
|
|
2
|
+
|
|
3
|
+
Once a source hands a 1-D cell over as an ``np.ndarray`` rather than a list,
|
|
4
|
+
``DataFrame.explode`` yields one ``np.float64`` OBJECT per sample instead of a
|
|
5
|
+
Python float. pandas' ``maybe_convert_numeric`` (pandas 3.x, numpy 2.x) calls
|
|
6
|
+
``len()`` on any value that has ``__len__`` — numpy scalars do, and raise
|
|
7
|
+
``TypeError: len() of unsized object``. Found 2026-09-13 the moment the DuckDB
|
|
8
|
+
fetch stopped boxing (``scistackplotdb`` ``test_load_fetch.py``); every
|
|
9
|
+
``to_numeric`` downstream of an explode or over a cell column goes through here.
|
|
10
|
+
|
|
11
|
+
The fast path is also the point: a column of numpy scalars casts to float64 in
|
|
12
|
+
one C-level pass, where ``to_numeric`` was inspecting each object.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import pandas as pd
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def coerce_numeric(values: pd.Series) -> pd.Series:
|
|
24
|
+
"""``values`` as float64, non-numeric entries NaN — never raising on a
|
|
25
|
+
numpy scalar, and never stacking a column of array cells into a matrix."""
|
|
26
|
+
if pd.api.types.is_numeric_dtype(values):
|
|
27
|
+
return values
|
|
28
|
+
raw = values.to_numpy()
|
|
29
|
+
sample = _first_present(raw)
|
|
30
|
+
if isinstance(sample, (np.generic, int, float)) and not isinstance(sample, bool):
|
|
31
|
+
# Scalars (numpy or Python): one vectorised cast. Anything that is not
|
|
32
|
+
# a number in the column raises here and falls through to the coercing
|
|
33
|
+
# path below, which is what `errors="coerce"` promised.
|
|
34
|
+
try:
|
|
35
|
+
return pd.Series(raw.astype("float64"), index=values.index)
|
|
36
|
+
except (TypeError, ValueError):
|
|
37
|
+
pass
|
|
38
|
+
return pd.to_numeric(values.map(_unboxed), errors="coerce")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _first_present(raw: np.ndarray) -> Any:
|
|
42
|
+
for value in raw:
|
|
43
|
+
if value is None:
|
|
44
|
+
continue
|
|
45
|
+
if isinstance(value, float) and value != value:
|
|
46
|
+
continue
|
|
47
|
+
return value
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _unboxed(value: Any) -> Any:
|
|
52
|
+
"""A numpy scalar as its Python value; a list-like cell as None.
|
|
53
|
+
|
|
54
|
+
``to_numeric(errors="coerce")`` turned a list cell into NaN; an ndarray
|
|
55
|
+
cell must land in the same place rather than in ``len()``.
|
|
56
|
+
"""
|
|
57
|
+
if isinstance(value, np.generic):
|
|
58
|
+
return value.item()
|
|
59
|
+
if isinstance(value, (list, tuple, np.ndarray)):
|
|
60
|
+
return None
|
|
61
|
+
return value
|