scistackplot 0.1.26__tar.gz → 0.1.29__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {scistackplot-0.1.26 → scistackplot-0.1.29}/.gitignore +5 -0
  2. {scistackplot-0.1.26 → scistackplot-0.1.29}/PKG-INFO +2 -1
  3. {scistackplot-0.1.26 → scistackplot-0.1.29}/pyproject.toml +1 -0
  4. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/__init__.py +2 -0
  5. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/capability.py +43 -14
  6. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/codegen.py +57 -0
  7. scistackplot-0.1.29/src/scistackplot/dedup.py +93 -0
  8. scistackplot-0.1.29/src/scistackplot/framesize.py +105 -0
  9. scistackplot-0.1.29/src/scistackplot/numeric.py +61 -0
  10. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/reduce.py +290 -49
  11. scistackplot-0.1.29/src/scistackplot/reducer.py +506 -0
  12. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/roles.py +30 -12
  13. scistackplot-0.1.29/src/scistackplot/series_stats.py +204 -0
  14. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/base.py +89 -13
  15. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/spec.py +90 -0
  16. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/table.py +11 -0
  17. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/ylimits.py +3 -2
  18. {scistackplot-0.1.26 → scistackplot-0.1.29}/README.md +0 -0
  19. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/groups.py +0 -0
  20. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/__init__.py +0 -0
  21. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/base.py +0 -0
  22. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/mpl.py +0 -0
  23. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/render/plotly_.py +0 -0
  24. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/resolved.py +0 -0
  25. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/shape.py +0 -0
  26. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/__init__.py +0 -0
  27. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/csv.py +0 -0
  28. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/sources/frame.py +0 -0
  29. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/variants.py +0 -0
  30. {scistackplot-0.1.26 → scistackplot-0.1.29}/src/scistackplot/xaxis.py +0 -0
@@ -20,9 +20,14 @@ __pycache__/
20
20
  *.duckdb.wal
21
21
  *.wal
22
22
  scistack-gui/frontend/node_modules/
23
+ # Compiled output of the frontend's React-free unit tests (npm test in
24
+ # scistack-gui/frontend). Regenerated by `tsc -p tsconfig.test.json`; unlike
25
+ # extension/dist/, nothing loads it at runtime.
26
+ scistack-gui/frontend/dist/
23
27
  # mkdocs build output (regenerated by `mkdocs build`)
24
28
  /site/
25
29
  # Runtime output of code_export_service (pipeline-to-code export) — timestamped
26
30
  # per-run files, not source.
27
31
  /exports/
28
32
  *.log
33
+ output.txt
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: scistackplot
3
- Version: 0.1.26
3
+ Version: 0.1.29
4
4
  Summary: Spec-driven plotting for long-format scientific data — standalone, GUI-friendly
5
5
  Author: SciStack Contributors
6
6
  License-Expression: MIT
@@ -13,6 +13,7 @@ Classifier: Programming Language :: Python :: 3
13
13
  Classifier: Programming Language :: Python :: 3.10
14
14
  Classifier: Programming Language :: Python :: 3.11
15
15
  Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
16
17
  Classifier: Topic :: Scientific/Engineering :: Visualization
17
18
  Classifier: Typing :: Typed
18
19
  Requires-Python: >=3.10
@@ -32,6 +32,7 @@ classifiers = [
32
32
  "Programming Language :: Python :: 3.10",
33
33
  "Programming Language :: Python :: 3.11",
34
34
  "Programming Language :: Python :: 3.12",
35
+ "Programming Language :: Python :: 3.13",
35
36
  "Topic :: Scientific/Engineering :: Visualization",
36
37
  "Typing :: Typed",
37
38
  ]
@@ -68,6 +68,7 @@ from .spec import (
68
68
  FacetOptions,
69
69
  Filter,
70
70
  LevelGroup,
71
+ LocationFilter,
71
72
  MatchOp,
72
73
  Matcher,
73
74
  PlotKind,
@@ -122,6 +123,7 @@ __all__ = [
122
123
  "StyleOptions",
123
124
  "Filter",
124
125
  "LevelGroup",
126
+ "LocationFilter",
125
127
  "VariantSet",
126
128
  # data
127
129
  "LongTable",
@@ -24,17 +24,18 @@ DISTRIBUTION_KINDS = (PlotKind.BOX, PlotKind.VIOLIN, PlotKind.BAR, PlotKind.BAND
24
24
 
25
25
  #: What each role is CALLED, per measure shape.
26
26
  #:
27
- #: The role names are the library's vocabulary; these are the user's. Two of
28
- #: them read as nonsense on 1-D data under the generic wording — "Average over"
29
- #: and "Replicates" describe what happens to a table, and what the user is
30
- #: looking at is a set of traces. Reported by the backend rather than hardcoded
31
- #: in the panel so the words and the behaviour stay together (CLAUDE.md NOTE 3).
27
+ #: The role names are the library's vocabulary; these are the user's — and the
28
+ #: two are kept the SAME wherever a user has to search for one (see the FREE
29
+ #: entry in _ROLE_LABELS). What varies by shape is what a role DOES: "Average
30
+ #: over" describes a table, and a user looking at 1-D data is looking at
31
+ #: traces, so AGGREGATE is renamed for them. Reported by the backend rather
32
+ #: than hardcoded in the panel so the words and the behaviour stay together
33
+ #: (CLAUDE.md NOTE 3); the per-shape explanation is _ROLE_HINTS_BY_SHAPE.
32
34
  #:
33
35
  #: Only the entries that differ from :data:`_ROLE_LABELS` need listing.
34
36
  _ROLE_LABELS_BY_SHAPE: dict[Shape, dict[Role, str]] = {
35
37
  Shape.SERIES_1D: {
36
38
  Role.AGGREGATE: "Average into one line",
37
- Role.FREE: "One line each",
38
39
  },
39
40
  }
40
41
 
@@ -67,7 +68,14 @@ _ROLE_LABELS: dict[Role, str] = {
67
68
  Role.FACET: "Facet",
68
69
  Role.ITERATE: "Separate figures",
69
70
  Role.AGGREGATE: "Average over",
70
- Role.FREE: "Replicates",
71
+ # "Free", not "Replicates" / "One line each". The role's own name, kept the
72
+ # same for every shape, because a user hunting for it in the dropdown has
73
+ # read it in the docs, in a saved spec's TOML and in an exported `roles=`
74
+ # argument — and found nothing matching (user, 2026-09-13). It is also the
75
+ # role whose meaning a noun cannot carry on its own: what a FREE factor
76
+ # does depends on the plot kind (one line each, a bar's error bars, a box's
77
+ # distribution), which is what the HINT is for, per shape.
78
+ Role.FREE: "Free",
71
79
  }
72
80
 
73
81
  _ROLE_HINTS: dict[Role, str] = {
@@ -75,15 +83,33 @@ _ROLE_HINTS: dict[Role, str] = {
75
83
  Role.COLOR: "One coloured series per level",
76
84
  Role.FACET: "One subplot per level — arrange them under Layout",
77
85
  Role.ITERATE: "One whole figure per level",
78
- Role.AGGREGATE: "Collapse this factor to its mean",
79
- Role.FREE: "Keep as repeated observations",
86
+ Role.AGGREGATE: "Collapse to the mean first, so this factor does NOT "
87
+ "widen the error bars",
88
+ Role.FREE: "Keep each level as its own observation, so this factor DOES "
89
+ "widen the error bars",
80
90
  }
81
91
 
92
+ #: Stated as a CONTRAST, and deliberately in the same terms on both sides.
93
+ #:
94
+ #: These are the two roles a user cannot tell apart from the labels alone —
95
+ #: "Average over" and "Free" both sound like "not on an axis", and the earlier
96
+ #: hints described each one on its own, both using the word "average" (user,
97
+ #: 2026-09-13). The thing that actually distinguishes them is what they do to
98
+ #: the ERROR BARS: with schema [subject, trial] and a band, trial=AGGREGATE
99
+ #: averages each subject's trials into one trace and the band is then the
100
+ #: spread across SUBJECTS; trial=FREE pools every subject-trial trace, so the
101
+ #: band mixes within- and between-subject variability and a subject with more
102
+ #: trials weighs more. Same plot kind, same data, different published numbers —
103
+ #: which is why the plot-kind control cannot express this and both roles exist
104
+ #: (`reduce._collapse_aggregates` runs before `_summarize`; `has_replicates`
105
+ #: is true only when something is FREE, which is the part the kind list DOES
106
+ #: express — collapse everything and the summary kinds disappear).
82
107
  _ROLE_HINTS_BY_SHAPE: dict[Shape, dict[Role, str]] = {
83
108
  Shape.SERIES_1D: {
84
- Role.AGGREGATE: "Average these traces together, sample by sample",
85
- Role.FREE: "Draw one trace per level — and what a mean ± error band "
86
- "is computed from",
109
+ Role.AGGREGATE: "Average these traces together sample by sample "
110
+ "first, so this factor does NOT widen the error band",
111
+ Role.FREE: "One trace per level, each its own observation — so this "
112
+ "factor DOES widen the error band",
87
113
  },
88
114
  }
89
115
 
@@ -402,9 +428,12 @@ def factor_summary(spec: PlotSpec, derived: LongTable) -> list[dict]:
402
428
  entry["x_available"] = on_x["available"]
403
429
  entry["x_reason"] = on_x["reason"]
404
430
 
405
- if not spec.filters:
431
+ if not spec.filters and spec.location_filter.is_empty():
406
432
  # Nothing filtered: everything is selected, and no frame scan is needed
407
- # on the common path.
433
+ # on the common path. The location filter has to be checked here too —
434
+ # it narrows rows exactly as a Filter does, and a fast path that only
435
+ # knew about one of them would report "all 12 selected" beside a figure
436
+ # drawing 3, which is the precise failure this readout exists to avoid.
408
437
  for entry in factors:
409
438
  entry["selected"] = list(entry["levels"])
410
439
  return factors
@@ -352,6 +352,14 @@ def _preamble(spec, table, roles, shape) -> list[str]:
352
352
  ]
353
353
  )
354
354
 
355
+ # Schema locations. Emitted as one vectorised comparison per (prefix, key),
356
+ # which is the SAME rule reduce._location_mask applies — a ragged selection
357
+ # cannot be expressed as `for_each(subject=[...], trial=[...])`, whose keys
358
+ # cross-product, so it has to be a mask inside the function body. The
359
+ # `if _k in df.columns` guard is not defensive clutter: it is how a key the
360
+ # frame lacks goes unconstrained, matching the interactive path exactly.
361
+ lines.extend(_location_lines(spec))
362
+
355
363
  filter_lines: list[str] = []
356
364
  for flt in spec.filters:
357
365
  if flt.include is not None:
@@ -467,6 +475,32 @@ def _preamble(spec, table, roles, shape) -> list[str]:
467
475
  return lines
468
476
 
469
477
 
478
+ def _location_lines(spec: PlotSpec) -> list[str]:
479
+ """Generated pandas for ``spec.location_filter`` — empty when it is inert.
480
+
481
+ Kept beside the other preamble emitters rather than inlined so the one test
482
+ that matters can compare its output against
483
+ :func:`scistackplot.reduce._location_mask` on the same frame.
484
+ """
485
+ prefixes = spec.location_filter.prefixes()
486
+ if not prefixes:
487
+ return []
488
+ plural = "" if len(prefixes) == 1 else "s"
489
+ return [
490
+ f"# schema locations: {len(prefixes)} selection{plural}",
491
+ f"_loc_prefixes = {[[list(pair) for pair in p] for p in prefixes]!r}",
492
+ "_loc_mask = pd.Series(False, index=df.index)",
493
+ "for _p in _loc_prefixes:",
494
+ " _m = pd.Series(True, index=df.index)",
495
+ " for _k, _v in _p:",
496
+ " if _k in df.columns:",
497
+ " _m &= df[_k].astype(str) == _v",
498
+ " _loc_mask |= _m",
499
+ "df = df[_loc_mask]",
500
+ "",
501
+ ]
502
+
503
+
470
504
  def _facet_layout_args(spec, table: LongTable, facets: list[str]) -> list[str]:
471
505
  """
472
506
  seaborn arguments that reproduce the interactive facet arrangement.
@@ -754,6 +788,7 @@ def _color_level_count(spec: PlotSpec, table: LongTable, color: str) -> int:
754
788
  levels = [str(level) for level in table.factor(color).levels]
755
789
  except KeyError:
756
790
  return 2
791
+ levels = _levels_after_location(spec, color, levels)
757
792
  for flt in spec.filters:
758
793
  if flt.column != color:
759
794
  continue
@@ -766,6 +801,28 @@ def _color_level_count(spec: PlotSpec, table: LongTable, color: str) -> int:
766
801
  return len(levels)
767
802
 
768
803
 
804
+ def _levels_after_location(spec: PlotSpec, column: str, levels: list[str]) -> list[str]:
805
+ """``levels`` narrowed by ``spec.location_filter``, for one schema key.
806
+
807
+ The rule follows from prefixes constraining only the keys they name: a
808
+ prefix that never mentions this column says nothing about it, so if any
809
+ such prefix is selected the column keeps every level. Only when *every*
810
+ prefix names it do the values they give become the surviving set.
811
+
812
+ Concretely, selecting all of subject 01 plus subject 02's trial 3 leaves
813
+ ``trial`` with all its levels — because "all of subject 01" did not name a
814
+ trial — which is what the figure will actually draw.
815
+ """
816
+ prefixes = spec.location_filter.prefixes()
817
+ if not prefixes:
818
+ return levels
819
+ named = [dict(prefix) for prefix in prefixes]
820
+ if any(column not in prefix for prefix in named):
821
+ return levels
822
+ keep = {prefix[column] for prefix in named}
823
+ return [level for level in levels if level in keep]
824
+
825
+
769
826
  #: Column the generated code builds for a nested x axis.
770
827
  _X_NESTED = "_x"
771
828
 
@@ -0,0 +1,93 @@
1
+ """Run a keyed build ONCE even when several threads ask for it at the same time.
2
+
3
+ Memoizing a build only helps the second caller if the first one has *finished*.
4
+ The panel does not work that way: opening it fires ``plot_describe``,
5
+ ``plot_resolve``, ``plot_capabilities`` and ``plot_location_tree`` together, the
6
+ server runs a thread per request, and every one of them asks for the same table.
7
+
8
+ Measured on 2026-09-13: two requests both logged ``table cache MISS`` and both
9
+ built the identical 5.2 GB table, concurrently — 188.8 s and 104.1 s, finishing
10
+ within 100 ms of each other (.claude/plot-at-scale-plan.md §7.1). The memo was
11
+ written after ``_build_table`` returned, so the second arrival saw an empty cache
12
+ and repeated all of it. A classic cache stampede; the 2026-09-11 memoization work
13
+ fixed sequential repeat cost and never addressed concurrent arrival.
14
+
15
+ :class:`SingleFlight` closes that: the first caller for a key builds, everyone
16
+ else waits on the same result and pays nothing.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import threading
22
+ from concurrent.futures import Future
23
+ from typing import Any, Callable
24
+
25
+
26
+ class SingleFlight:
27
+ """Deduplicate concurrent builds by key.
28
+
29
+ One instance per cache. Keys must be hashable and must mean the same thing
30
+ the cache's own key means — a narrower key here would merge builds that are
31
+ not interchangeable.
32
+
33
+ Thread-safe. The internal lock is held only while claiming or releasing a
34
+ key, never while building, so two DIFFERENT keys still build in parallel.
35
+ """
36
+
37
+ def __init__(self) -> None:
38
+ self._lock = threading.Lock()
39
+ self._inflight: dict[Any, Future] = {}
40
+
41
+ def run(
42
+ self,
43
+ key: Any,
44
+ build: Callable[[], Any],
45
+ *,
46
+ on_wait: Callable[[], None] | None = None,
47
+ ) -> Any:
48
+ """``build()``'s result, computing it once across concurrent callers.
49
+
50
+ The first caller for ``key`` runs ``build``; any caller arriving while
51
+ that is in flight blocks until it finishes and gets the same object.
52
+
53
+ ``on_wait`` is called (once, by each waiter) instead of building, for
54
+ logging — a waiter that silently returns the right answer is
55
+ indistinguishable in a log from a cache hit, and those have very
56
+ different meanings when reading a slow request.
57
+
58
+ A failing ``build`` raises in the owner AND in every waiter: the waiters
59
+ asked for a value that could not be produced, and swallowing it here
60
+ would hand them a wrong answer or a hang. The key is released either
61
+ way, so a later attempt is free to retry.
62
+ """
63
+ with self._lock:
64
+ future = self._inflight.get(key)
65
+ owner = future is None
66
+ if owner:
67
+ future = Future()
68
+ self._inflight[key] = future
69
+
70
+ if not owner:
71
+ if on_wait is not None:
72
+ on_wait()
73
+ # Raises here if the owner's build failed — by design, see above.
74
+ return future.result()
75
+
76
+ try:
77
+ result = build()
78
+ except BaseException as exc:
79
+ # Release BEFORE publishing the failure so a waiter that wakes and
80
+ # immediately retries finds a clear slot rather than this dead one.
81
+ with self._lock:
82
+ self._inflight.pop(key, None)
83
+ future.set_exception(exc)
84
+ raise
85
+ with self._lock:
86
+ self._inflight.pop(key, None)
87
+ future.set_result(result)
88
+ return result
89
+
90
+ def in_flight(self) -> int:
91
+ """How many keys are building right now. For tests and diagnostics."""
92
+ with self._lock:
93
+ return len(self._inflight)
@@ -0,0 +1,105 @@
1
+ """How big is this frame, really — in cells AND in samples.
2
+
3
+ A row count is the wrong unit for a table whose cells hold signals. 4190 rows of
4
+ a 200-sample array and 4190 rows of a 250,000-sample array are the same row
5
+ count and four orders of magnitude apart in work, and every load/reduce log line
6
+ in this package reported only rows. On 2026-09-13 that made a plot that never
7
+ returned indistinguishable, from the log alone, from a plot over a small table
8
+ (.claude/plot-at-scale-plan.md §1).
9
+
10
+ So one helper, used by every layer that reports frame size, so `load_variable`
11
+ in ``scistackplotdb`` and the reduce path here cannot describe the same frame in
12
+ different units.
13
+
14
+ **Cost.** ``len()`` on a list or ndarray is O(1), so measuring is one pass over
15
+ the cells of the measure columns — cheap enough to run unconditionally on a load
16
+ that is already touching every one of those cells. It deliberately does NOT sum
17
+ ``nbytes`` or walk into the values: the point is to size the work, not to audit
18
+ memory exactly.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import TYPE_CHECKING, Any
24
+
25
+ if TYPE_CHECKING: # pragma: no cover - typing only
26
+ import pandas as pd
27
+
28
+ #: Rough bytes per element for a Python ``list[float]`` cell: the float object
29
+ #: (24 on CPython 64-bit) plus the list's pointer slot (8). Used only to turn a
30
+ #: sample count into an order-of-magnitude memory figure in a log line, which is
31
+ #: what distinguishes "big" from "hopeless" at a glance.
32
+ PYLIST_BYTES_PER_SAMPLE = 32
33
+
34
+ #: Bytes per element once a cell is a float64 ndarray. The gap between this and
35
+ #: :data:`PYLIST_BYTES_PER_SAMPLE` is the prize for fetching via Arrow.
36
+ NUMPY_BYTES_PER_SAMPLE = 8
37
+
38
+
39
+ def cell_samples(value: Any) -> int:
40
+ """Number of samples in one cell: its length, or 1 for a scalar.
41
+
42
+ ``None`` and strings count as 0 — neither is a sample, and a string's length
43
+ would be a character count masquerading as one.
44
+ """
45
+ if value is None or isinstance(value, (str, bytes)):
46
+ return 0
47
+ try:
48
+ return len(value)
49
+ except TypeError:
50
+ return 1 # a scalar is one sample
51
+
52
+
53
+ def frame_extent(frame: "pd.DataFrame", columns: list[str] | None = None) -> dict:
54
+ """``{rows, cells, samples, est_bytes, boxed}`` for ``frame``.
55
+
56
+ ``columns`` limits the scan to the measure columns worth measuring; omitted,
57
+ every column is scanned, which is only appropriate for a small frame.
58
+
59
+ ``boxed`` is True when the first sampled cell is a Python ``list``/``tuple``
60
+ rather than an ndarray — i.e. the payload is carried as boxed Python floats.
61
+ That single flag is what says whether an Arrow fetch path would help, so it
62
+ is worth a word in the log even though it is only a sample.
63
+ """
64
+ rows = len(frame)
65
+ names = [c for c in (columns or list(frame.columns)) if c in frame.columns]
66
+ cells = 0
67
+ samples = 0
68
+ boxed = False
69
+ seen_a_cell = False
70
+ for name in names:
71
+ series = frame[name]
72
+ for value in series.to_numpy():
73
+ if value is None:
74
+ continue
75
+ n = cell_samples(value)
76
+ if n == 0:
77
+ continue
78
+ cells += 1
79
+ samples += n
80
+ if not seen_a_cell:
81
+ boxed = isinstance(value, (list, tuple))
82
+ seen_a_cell = True
83
+ per_sample = PYLIST_BYTES_PER_SAMPLE if boxed else NUMPY_BYTES_PER_SAMPLE
84
+ return {
85
+ "rows": rows,
86
+ "cells": cells,
87
+ "samples": samples,
88
+ "est_bytes": samples * per_sample,
89
+ "boxed": boxed,
90
+ }
91
+
92
+
93
+ def format_extent(extent: dict) -> str:
94
+ """One log-line fragment: ``rows=419 cells=4190 samples=1.1e+09 ~34.0GB boxed``.
95
+
96
+ Samples go in scientific notation because the interesting range spans six
97
+ orders of magnitude and the exact digits never matter.
98
+ """
99
+ gb = extent["est_bytes"] / 1024**3
100
+ size = f"~{gb:.1f}GB" if gb >= 0.1 else f"~{extent['est_bytes'] / 1024**2:.0f}MB"
101
+ return (
102
+ f"rows={extent['rows']} cells={extent['cells']} "
103
+ f"samples={extent['samples']:.3g} {size}"
104
+ + (" boxed" if extent["boxed"] else " ndarray")
105
+ )
@@ -0,0 +1,61 @@
1
+ """``pd.to_numeric(errors="coerce")`` that survives numpy scalars.
2
+
3
+ Once a source hands a 1-D cell over as an ``np.ndarray`` rather than a list,
4
+ ``DataFrame.explode`` yields one ``np.float64`` OBJECT per sample instead of a
5
+ Python float. pandas' ``maybe_convert_numeric`` (pandas 3.x, numpy 2.x) calls
6
+ ``len()`` on any value that has ``__len__`` — numpy scalars do, and raise
7
+ ``TypeError: len() of unsized object``. Found 2026-09-13 the moment the DuckDB
8
+ fetch stopped boxing (``scistackplotdb`` ``test_load_fetch.py``); every
9
+ ``to_numeric`` downstream of an explode or over a cell column goes through here.
10
+
11
+ The fast path is also the point: a column of numpy scalars casts to float64 in
12
+ one C-level pass, where ``to_numeric`` was inspecting each object.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import Any
18
+
19
+ import numpy as np
20
+ import pandas as pd
21
+
22
+
23
+ def coerce_numeric(values: pd.Series) -> pd.Series:
24
+ """``values`` as float64, non-numeric entries NaN — never raising on a
25
+ numpy scalar, and never stacking a column of array cells into a matrix."""
26
+ if pd.api.types.is_numeric_dtype(values):
27
+ return values
28
+ raw = values.to_numpy()
29
+ sample = _first_present(raw)
30
+ if isinstance(sample, (np.generic, int, float)) and not isinstance(sample, bool):
31
+ # Scalars (numpy or Python): one vectorised cast. Anything that is not
32
+ # a number in the column raises here and falls through to the coercing
33
+ # path below, which is what `errors="coerce"` promised.
34
+ try:
35
+ return pd.Series(raw.astype("float64"), index=values.index)
36
+ except (TypeError, ValueError):
37
+ pass
38
+ return pd.to_numeric(values.map(_unboxed), errors="coerce")
39
+
40
+
41
+ def _first_present(raw: np.ndarray) -> Any:
42
+ for value in raw:
43
+ if value is None:
44
+ continue
45
+ if isinstance(value, float) and value != value:
46
+ continue
47
+ return value
48
+ return None
49
+
50
+
51
+ def _unboxed(value: Any) -> Any:
52
+ """A numpy scalar as its Python value; a list-like cell as None.
53
+
54
+ ``to_numeric(errors="coerce")`` turned a list cell into NaN; an ndarray
55
+ cell must land in the same place rather than in ``len()``.
56
+ """
57
+ if isinstance(value, np.generic):
58
+ return value.item()
59
+ if isinstance(value, (list, tuple, np.ndarray)):
60
+ return None
61
+ return value