scistackplotdb 0.1.26__tar.gz → 0.1.29__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/.gitignore +5 -0
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/PKG-INFO +3 -1
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/pyproject.toml +2 -0
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/src/scistackplotdb/__init__.py +7 -1
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/src/scistackplotdb/load.py +162 -20
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/src/scistackplotdb/source.py +163 -11
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/src/scistackplotdb/variants.py +49 -0
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/README.md +0 -0
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/src/scistackplotdb/endpoint.py +0 -0
- {scistackplotdb-0.1.26 → scistackplotdb-0.1.29}/src/scistackplotdb/hierarchy.py +0 -0
|
@@ -20,9 +20,14 @@ __pycache__/
|
|
|
20
20
|
*.duckdb.wal
|
|
21
21
|
*.wal
|
|
22
22
|
scistack-gui/frontend/node_modules/
|
|
23
|
+
# Compiled output of the frontend's React-free unit tests (npm test in
|
|
24
|
+
# scistack-gui/frontend). Regenerated by `tsc -p tsconfig.test.json`; unlike
|
|
25
|
+
# extension/dist/, nothing loads it at runtime.
|
|
26
|
+
scistack-gui/frontend/dist/
|
|
23
27
|
# mkdocs build output (regenerated by `mkdocs build`)
|
|
24
28
|
/site/
|
|
25
29
|
# Runtime output of code_export_service (pipeline-to-code export) — timestamped
|
|
26
30
|
# per-run files, not source.
|
|
27
31
|
/exports/
|
|
28
32
|
*.log
|
|
33
|
+
output.txt
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: scistackplotdb
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.29
|
|
4
4
|
Summary: scidb-backed plotting: load variables into long tables and plot them with scistackplot
|
|
5
5
|
Author: SciStack Contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -10,8 +10,10 @@ Classifier: Intended Audience :: Science/Research
|
|
|
10
10
|
Classifier: License :: OSI Approved :: MIT License
|
|
11
11
|
Classifier: Operating System :: OS Independent
|
|
12
12
|
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
14
|
Classifier: Programming Language :: Python :: 3.11
|
|
14
15
|
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
17
|
Classifier: Topic :: Scientific/Engineering :: Visualization
|
|
16
18
|
Classifier: Typing :: Typed
|
|
17
19
|
Requires-Python: >=3.10
|
|
@@ -29,8 +29,10 @@ classifiers = [
|
|
|
29
29
|
"License :: OSI Approved :: MIT License",
|
|
30
30
|
"Operating System :: OS Independent",
|
|
31
31
|
"Programming Language :: Python :: 3",
|
|
32
|
+
"Programming Language :: Python :: 3.10",
|
|
32
33
|
"Programming Language :: Python :: 3.11",
|
|
33
34
|
"Programming Language :: Python :: 3.12",
|
|
35
|
+
"Programming Language :: Python :: 3.13",
|
|
34
36
|
"Topic :: Scientific/Engineering :: Visualization",
|
|
35
37
|
"Typing :: Typed",
|
|
36
38
|
]
|
|
@@ -50,13 +50,19 @@ from .load import (
|
|
|
50
50
|
schema_keys,
|
|
51
51
|
)
|
|
52
52
|
from .source import ScidbSource
|
|
53
|
-
from .variants import
|
|
53
|
+
from .variants import (
|
|
54
|
+
branch_params_for,
|
|
55
|
+
selection_for,
|
|
56
|
+
variant_graph,
|
|
57
|
+
variant_set,
|
|
58
|
+
)
|
|
54
59
|
|
|
55
60
|
__all__ = [
|
|
56
61
|
"ScidbSource",
|
|
57
62
|
"variant_set",
|
|
58
63
|
"variant_graph",
|
|
59
64
|
"selection_for",
|
|
65
|
+
"branch_params_for",
|
|
60
66
|
"VariableFrame",
|
|
61
67
|
"VERSION_FACTOR_PREFIX",
|
|
62
68
|
"MISSING_VERSION_LEVEL",
|
|
@@ -7,8 +7,10 @@ columns once a variable is joined to ``_schema``, which is the same shape
|
|
|
7
7
|
part a flat CSV never needed: attaching branch params as columns, and knowing
|
|
8
8
|
which schema keys a given variable actually occupies.
|
|
9
9
|
|
|
10
|
-
Queries go through ``_fetchall``/``_fetchone`` (never
|
|
11
|
-
— see docs/claude on DuckDB fetch locking)
|
|
10
|
+
Queries go through ``_fetchall``/``_fetchone``/``_fetchdf`` (never
|
|
11
|
+
``_execute(...).fetchall()`` — see docs/claude on DuckDB fetch locking); the
|
|
12
|
+
payload itself only ever through ``_fetchdf``, which is what keeps a sample from
|
|
13
|
+
becoming a Python float (see ``load_variable``) and batch the branch-params walk
|
|
12
14
|
rather than asking per record (the N+1 trap).
|
|
13
15
|
"""
|
|
14
16
|
|
|
@@ -17,9 +19,11 @@ from __future__ import annotations
|
|
|
17
19
|
from dataclasses import dataclass, field
|
|
18
20
|
from typing import Any
|
|
19
21
|
|
|
22
|
+
import numpy as np
|
|
20
23
|
import pandas as pd
|
|
21
24
|
from scistacklog import Log
|
|
22
25
|
from scistackplot import CODE_FACTOR_PREFIX
|
|
26
|
+
from scistackplot.framesize import format_extent, frame_extent
|
|
23
27
|
|
|
24
28
|
LAYER = "scistackplotdb"
|
|
25
29
|
|
|
@@ -179,17 +183,125 @@ def variable_levels(db, variable: str) -> list[str]:
|
|
|
179
183
|
return [key for key, count in zip(keys, row, strict=True) if count]
|
|
180
184
|
|
|
181
185
|
|
|
182
|
-
def
|
|
183
|
-
"""
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
186
|
+
def _normalize_cell(value: Any) -> Any:
|
|
187
|
+
"""One data cell as the shape downstream expects, converted PER CELL only.
|
|
188
|
+
|
|
189
|
+
A DataFrame fetch already delivers a ``DOUBLE[]`` cell as an ndarray; this
|
|
190
|
+
leaves it alone. What it handles:
|
|
191
|
+
|
|
192
|
+
* a ``DOUBLE[][]`` cell, which DuckDB hands over as an object ndarray of
|
|
193
|
+
row ndarrays — stacked into one 2-D float array when the rows are
|
|
194
|
+
rectangular, so ``np.asarray(cell).shape`` is ``(rows, cols)`` as the
|
|
195
|
+
heatmap path expects; ragged rows become a list of row arrays;
|
|
196
|
+
* a Python ``list`` cell, from a DuckDB build that still boxes LIST columns
|
|
197
|
+
on the pandas path — converted once, so the rest of the stack sees one
|
|
198
|
+
contract. The boxing has already been paid by then; the "loaded" log line
|
|
199
|
+
says so (``boxed``), which is the signal to look at the DuckDB version.
|
|
200
|
+
|
|
201
|
+
A NULL cell, which the DataFrame fetch spells ``pd.NA``, becomes None — the
|
|
202
|
+
spelling every downstream check (``value is None``) already knows. Scalars,
|
|
203
|
+
None and NaN pass through.
|
|
204
|
+
"""
|
|
205
|
+
if value is pd.NA:
|
|
206
|
+
return None
|
|
207
|
+
if isinstance(value, np.ndarray):
|
|
208
|
+
if (
|
|
209
|
+
value.dtype == object
|
|
210
|
+
and value.size
|
|
211
|
+
and isinstance(value.flat[0], (np.ndarray, list))
|
|
212
|
+
):
|
|
213
|
+
try:
|
|
214
|
+
return np.stack([_float_row(row) for row in value])
|
|
215
|
+
except ValueError:
|
|
216
|
+
# Ragged rows: a Python list OF row arrays (one object per row,
|
|
217
|
+
# not per sample), which `shape.classify_value` still reads
|
|
218
|
+
# as MATRIX_2D — an object ndarray of ndim 1 would read as a
|
|
219
|
+
# 1-D series and the heatmap would silently become a line.
|
|
220
|
+
return list(value)
|
|
221
|
+
if isinstance(value, np.ma.MaskedArray):
|
|
222
|
+
# A NULL ELEMENT inside a list — which is what scidb's single-record
|
|
223
|
+
# INSERT binding stores a NaN as — comes back masked, and the first
|
|
224
|
+
# `np.asarray` downstream silently drops the mask and exposes the
|
|
225
|
+
# fill value: a NaN sample became a number, and every "NaN rows are
|
|
226
|
+
# dropped" comparison in the parity suite failed (2026-09-13).
|
|
227
|
+
# Filled here, once, with the NaN every consumer already handles.
|
|
228
|
+
return _float_row(value)
|
|
229
|
+
return value
|
|
230
|
+
if isinstance(value, list):
|
|
231
|
+
try:
|
|
232
|
+
return np.asarray(value, dtype=float)
|
|
233
|
+
except (TypeError, ValueError):
|
|
234
|
+
return np.asarray(value, dtype=object)
|
|
235
|
+
return value
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _float_row(row: Any) -> np.ndarray:
|
|
239
|
+
"""One 1-D float64 array from a row/cell, a masked element becoming NaN."""
|
|
240
|
+
masked = np.ma.asarray(row, dtype="float64")
|
|
241
|
+
return np.ma.filled(masked, np.nan)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _object_column(cells: list, index) -> pd.Series:
|
|
245
|
+
"""A Series of exactly these cells, dtype object, no inference.
|
|
246
|
+
|
|
247
|
+
Not ``Series.map`` and not ``pd.Series(cells)``: both run dtype inference
|
|
248
|
+
over the values, and a column of same-length float arrays is precisely the
|
|
249
|
+
input that inference reinterprets (a 2-D block, a scalar per cell, a string
|
|
250
|
+
dtype for the keys). Filling a preallocated object array one cell at a time
|
|
251
|
+
is the one construction every pandas version leaves alone.
|
|
252
|
+
"""
|
|
253
|
+
out = np.empty(len(cells), dtype=object)
|
|
254
|
+
for i, cell in enumerate(cells):
|
|
255
|
+
out[i] = cell
|
|
256
|
+
return pd.Series(out, index=index, dtype=object)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _key_text(value: Any) -> "str | None":
|
|
260
|
+
"""A schema key as text, or None for a NULL — whether pandas spelled that
|
|
261
|
+
NULL as None or as NaN."""
|
|
262
|
+
if value is None:
|
|
263
|
+
return None
|
|
264
|
+
if isinstance(value, float) and value != value:
|
|
265
|
+
return None
|
|
266
|
+
return str(value)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def load_variable(
|
|
270
|
+
db, variable: str, *, with_variants: bool = True, include_data: bool = True
|
|
271
|
+
) -> VariableFrame:
|
|
272
|
+
"""Load every non-excluded record of ``variable`` as a long frame.
|
|
273
|
+
|
|
274
|
+
``include_data=False`` selects the record ids, schema keys and variant
|
|
275
|
+
columns but **no data columns** — everything needed to answer a question
|
|
276
|
+
about a variable's identity and variants, and none of the payload. It exists
|
|
277
|
+
because the payload is where all the cost is: on 2026-09-13 the data columns
|
|
278
|
+
of one variable were 174 million samples / ~5.2 GB, and the surfaces that
|
|
279
|
+
loaded them to read variant metadata simply never returned
|
|
280
|
+
(.claude/plot-at-scale-plan.md §7). The returned frame reports
|
|
281
|
+
``data_columns=[]``, so it is NOT a plottable table and ``_build_table`` will
|
|
282
|
+
(correctly) refuse it.
|
|
283
|
+
"""
|
|
284
|
+
with Log.timer(
|
|
285
|
+
"load_variable",
|
|
286
|
+
layer=LAYER,
|
|
287
|
+
extra=variable if include_data else f"{variable} (metadata only)",
|
|
288
|
+
) as timer:
|
|
289
|
+
with timer.phase("column_metadata"):
|
|
290
|
+
keys = schema_keys(db)
|
|
291
|
+
columns = data_columns_for(db, variable)
|
|
187
292
|
if not columns:
|
|
188
293
|
Log.warn("variable %r has no data columns", variable, layer=LAYER)
|
|
189
294
|
return VariableFrame(name=variable, frame=pd.DataFrame())
|
|
295
|
+
if not include_data:
|
|
296
|
+
columns = []
|
|
190
297
|
|
|
191
298
|
table = table_name_for(db, variable)
|
|
192
|
-
|
|
299
|
+
# Schema keys as VARCHAR in the query, not stringified after: a
|
|
300
|
+
# DataFrame fetch types each column, and an integer key with one NULL
|
|
301
|
+
# among its rows would arrive as float64 and stringify as "1.0".
|
|
302
|
+
schema_select = "".join(
|
|
303
|
+
f', CAST(s."{key}" AS VARCHAR) AS "{key}"' for key in keys
|
|
304
|
+
)
|
|
193
305
|
data_select = "".join(f', t."{column}"' for column in columns)
|
|
194
306
|
query = (
|
|
195
307
|
f"SELECT t.record_id{data_select}{schema_select} "
|
|
@@ -198,27 +310,57 @@ def load_variable(db, variable: str, *, with_variants: bool = True) -> VariableF
|
|
|
198
310
|
f"LEFT JOIN _schema s ON r.schema_id = s.schema_id "
|
|
199
311
|
f"WHERE r.type = ? AND r.excluded IS DISTINCT FROM TRUE"
|
|
200
312
|
)
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
for
|
|
207
|
-
|
|
313
|
+
# NOTE: no schema filter, no LIMIT, no column projection — this reads
|
|
314
|
+
# EVERY non-excluded record of the variable with EVERY data column,
|
|
315
|
+
# whatever the caller intends to plot (pushdown is a later stage of
|
|
316
|
+
# .claude/plan-plot-minimal-load-examples.md).
|
|
317
|
+
#
|
|
318
|
+
# `_fetchdf`, never `_fetchall`, for the data columns. Measured on the
|
|
319
|
+
# real database 2026-09-13 (plan §7): one DOUBLE[] column of 17.4 M
|
|
320
|
+
# samples took 4.7 s through `fetchall` — a Python float per sample —
|
|
321
|
+
# and 0.26 s through `.df()`, which hands each cell over as one numpy
|
|
322
|
+
# buffer. That 18x was 86 of the 91 s `plot_describe` spent before it
|
|
323
|
+
# timed out. The "loaded" line below says `ndarray` or `boxed` so a
|
|
324
|
+
# regression to per-sample boxing is visible in the log, not inferred.
|
|
325
|
+
with timer.phase("fetch"):
|
|
326
|
+
frame = db._duck._fetchdf(query, [variable])
|
|
327
|
+
|
|
328
|
+
with timer.phase("dataframe"):
|
|
329
|
+
frame = frame[["record_id", *columns, *keys]].copy()
|
|
330
|
+
for column in columns:
|
|
331
|
+
if pd.api.types.is_numeric_dtype(frame[column]):
|
|
332
|
+
continue # a scalar column: already one float64 buffer
|
|
333
|
+
frame[column] = _object_column(
|
|
334
|
+
[_normalize_cell(v) for v in frame[column].to_numpy()],
|
|
335
|
+
frame.index,
|
|
336
|
+
)
|
|
337
|
+
with timer.phase("stringify_keys"):
|
|
338
|
+
for key in keys:
|
|
339
|
+
frame[key] = _object_column(
|
|
340
|
+
[_key_text(v) for v in frame[key].to_numpy()], frame.index
|
|
341
|
+
)
|
|
208
342
|
|
|
209
343
|
levels = [key for key in keys if frame[key].notna().any()]
|
|
210
344
|
variant_columns: list[str] = []
|
|
211
345
|
variant_axes: list[dict] = []
|
|
212
346
|
latest_column: str | None = None
|
|
213
347
|
if with_variants and len(frame):
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
348
|
+
with timer.phase("attach_variants"):
|
|
349
|
+
frame, variant_columns, latest_column, variant_axes = attach_variants(
|
|
350
|
+
db, frame
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
# Cells and SAMPLES alongside the record count: 419 records is the same
|
|
354
|
+
# number whether each holds 200 samples or 250,000, and only the second
|
|
355
|
+
# explains a plot that never returns. Measured over the data columns
|
|
356
|
+
# only, one O(cells) pass (scistackplot.framesize).
|
|
357
|
+
with timer.phase("measure_extent"):
|
|
358
|
+
extent = frame_extent(frame, columns)
|
|
218
359
|
Log.info(
|
|
219
|
-
"loaded %s: %d record(s), levels=%s, variants=%s",
|
|
360
|
+
"loaded %s: %d record(s), %s, levels=%s, variants=%s",
|
|
220
361
|
variable,
|
|
221
362
|
len(frame),
|
|
363
|
+
format_extent(extent),
|
|
222
364
|
levels,
|
|
223
365
|
variant_columns or "none",
|
|
224
366
|
layer=LAYER,
|
|
@@ -22,6 +22,8 @@ from scistackplot import (
|
|
|
22
22
|
is_plottable,
|
|
23
23
|
natural_sort_key,
|
|
24
24
|
)
|
|
25
|
+
from scistackplot.dedup import SingleFlight
|
|
26
|
+
from scistackplot.framesize import format_extent, frame_extent
|
|
25
27
|
from scistackplot.sources import BaseSource
|
|
26
28
|
from scistackplot.variants import VARIABLE_COLUMN
|
|
27
29
|
|
|
@@ -56,7 +58,9 @@ class ScidbSource(BaseSource):
|
|
|
56
58
|
run-ownership work resolved).
|
|
57
59
|
"""
|
|
58
60
|
|
|
59
|
-
def __init__(
|
|
61
|
+
def __init__(
|
|
62
|
+
self, db, *, name: str | None = None, fast: bool = True
|
|
63
|
+
) -> None:
|
|
60
64
|
self._db = db
|
|
61
65
|
# `dataset_db_path`, not `db_path` — DatabaseManager has never had the
|
|
62
66
|
# latter, so this silently fell through to "scidb" for every project.
|
|
@@ -64,6 +68,30 @@ class ScidbSource(BaseSource):
|
|
|
64
68
|
self._frames: dict[str, Any] = {}
|
|
65
69
|
self._shapes: dict[str, Shape] = {}
|
|
66
70
|
self._levels: dict[str, list[str]] = {}
|
|
71
|
+
# Whether tables from this source carry the numpy reducer
|
|
72
|
+
# (``scistackplot.NumpyReducer``, over the ndarray cells `load_variable`
|
|
73
|
+
# delivers) or the pandas reference. On by default — the reference is
|
|
74
|
+
# the path that spent 500 s on 174 M samples — and switchable so the
|
|
75
|
+
# two can be run side by side on one database, which is how the parity
|
|
76
|
+
# suite works and how a suspected disagreement gets bisected.
|
|
77
|
+
#
|
|
78
|
+
# A DuckDB-SQL reducer sat here for one day (2026-09-13) and lost every
|
|
79
|
+
# measurement to numpy once the fetch stopped boxing samples
|
|
80
|
+
# (.claude/plan-plot-minimal-load-examples.md §8). DuckDB selects rows;
|
|
81
|
+
# numpy reduces them — so `resolve` no longer touches the database and
|
|
82
|
+
# the GUI's connection hold ends when the frames are loaded.
|
|
83
|
+
self._fast = fast
|
|
84
|
+
self._reducer_instance = None
|
|
85
|
+
|
|
86
|
+
def _reducer(self):
|
|
87
|
+
"""The reducer every table from this source carries (one per source)."""
|
|
88
|
+
if not self._fast:
|
|
89
|
+
return None # -> reducer_for() supplies the pandas reference
|
|
90
|
+
if self._reducer_instance is None:
|
|
91
|
+
from scistackplot.reducer import NumpyReducer
|
|
92
|
+
|
|
93
|
+
self._reducer_instance = NumpyReducer()
|
|
94
|
+
return self._reducer_instance
|
|
67
95
|
|
|
68
96
|
# ---- description -----------------------------------------------------
|
|
69
97
|
|
|
@@ -164,12 +192,124 @@ class ScidbSource(BaseSource):
|
|
|
164
192
|
return sorted(unique, key=numeric_key)
|
|
165
193
|
return sorted(unique, key=natural_sort_key)
|
|
166
194
|
|
|
195
|
+
# ---- metadata --------------------------------------------------------
|
|
196
|
+
|
|
197
|
+
def variant_table(self, variable: str) -> LongTable:
|
|
198
|
+
"""A table carrying this variable's VARIANT structure and no payload.
|
|
199
|
+
|
|
200
|
+
Same variant columns, levels, ``default_pin`` and latest-flag a full
|
|
201
|
+
``get_table`` would produce, built from a query that selects no data
|
|
202
|
+
columns at all. ``measures`` is empty, so this is not plottable and must
|
|
203
|
+
not be handed to ``resolve`` — it answers metadata questions only.
|
|
204
|
+
|
|
205
|
+
Why it exists: :func:`scistackplot.default_selection` reads
|
|
206
|
+
``default_pin``, ``latest_column`` and the variant factors' levels, and
|
|
207
|
+
nothing else — it never touches a measure column. Answering it through
|
|
208
|
+
``get_table`` meant loading 174 million samples / ~5.2 GB to read a
|
|
209
|
+
handful of variant levels, which is why the schema-location picker timed
|
|
210
|
+
out on a 419-location variable (.claude/plot-at-scale-plan.md §7).
|
|
211
|
+
|
|
212
|
+
Cached and deduplicated like any other table, under its own key, so the
|
|
213
|
+
picker opening twice costs one query.
|
|
214
|
+
"""
|
|
215
|
+
key = ("__variants__", variable)
|
|
216
|
+
memo = self._table_cache()
|
|
217
|
+
label = f"variant_table({variable})"
|
|
218
|
+
if key in memo:
|
|
219
|
+
Log.info("%s: table cache HIT", label, layer=LAYER)
|
|
220
|
+
return memo[key]
|
|
221
|
+
|
|
222
|
+
def _build():
|
|
223
|
+
if key in memo:
|
|
224
|
+
Log.info("%s: table cache HIT (filled while waiting)", label, layer=LAYER)
|
|
225
|
+
return memo[key]
|
|
226
|
+
Log.info("%s: table cache MISS — building (no data columns)", label, layer=LAYER)
|
|
227
|
+
variable_frame = load_variable(self._db, variable, include_data=False)
|
|
228
|
+
frame = variable_frame.frame
|
|
229
|
+
latest = variable_frame.latest_column
|
|
230
|
+
# Variant columns first, then schema keys — the same order and the
|
|
231
|
+
# same `_ordered` level sorting `_build_table` uses, so a selection
|
|
232
|
+
# derived here and one derived from the full table cannot disagree
|
|
233
|
+
# about which level is "first".
|
|
234
|
+
variant_columns = [
|
|
235
|
+
c for c in variable_frame.variant_columns if c in frame.columns
|
|
236
|
+
]
|
|
237
|
+
factors = list(variant_columns)
|
|
238
|
+
factors.extend(key for key in variable_frame.levels if key in frame.columns)
|
|
239
|
+
level_order = {
|
|
240
|
+
name: self._ordered(
|
|
241
|
+
name, [str(v) for v in frame[name].dropna().unique()]
|
|
242
|
+
)
|
|
243
|
+
for name in factors
|
|
244
|
+
}
|
|
245
|
+
table = LongTable.from_frame(
|
|
246
|
+
frame,
|
|
247
|
+
factors=factors,
|
|
248
|
+
measures=[],
|
|
249
|
+
level_order=level_order,
|
|
250
|
+
variant_factors=variant_columns,
|
|
251
|
+
name=variable,
|
|
252
|
+
default_pin={latest: True} if latest else None,
|
|
253
|
+
latest_column=latest,
|
|
254
|
+
factor_origins={
|
|
255
|
+
axis["column"]: axis for axis in variable_frame.variant_axes
|
|
256
|
+
},
|
|
257
|
+
schema_levels=variable_frame.levels,
|
|
258
|
+
)
|
|
259
|
+
memo[key] = table
|
|
260
|
+
return table
|
|
261
|
+
|
|
262
|
+
return self._table_single_flight().run(
|
|
263
|
+
key,
|
|
264
|
+
_build,
|
|
265
|
+
on_wait=lambda: Log.info(
|
|
266
|
+
"%s: build already in flight — waiting for it", label, layer=LAYER
|
|
267
|
+
),
|
|
268
|
+
)
|
|
269
|
+
|
|
167
270
|
# ---- data ------------------------------------------------------------
|
|
168
271
|
|
|
169
272
|
def _variable_frame(self, variable: str):
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
273
|
+
# Hit/miss at INFO: this cache is the difference between a request that
|
|
274
|
+
# reads the whole variable and one that reads nothing, and a plot request
|
|
275
|
+
# that returned in milliseconds is otherwise unattributable
|
|
276
|
+
# (.claude/plot-at-scale-plan.md §1). The layer below (load_variable)
|
|
277
|
+
# reports what a miss actually cost.
|
|
278
|
+
if variable in self._frames:
|
|
279
|
+
Log.info("variable frame cache HIT: %s", variable, layer=LAYER)
|
|
280
|
+
return self._frames[variable]
|
|
281
|
+
|
|
282
|
+
def _load():
|
|
283
|
+
if variable in self._frames:
|
|
284
|
+
Log.info(
|
|
285
|
+
"variable frame cache HIT: %s (filled while waiting)",
|
|
286
|
+
variable,
|
|
287
|
+
layer=LAYER,
|
|
288
|
+
)
|
|
289
|
+
return self._frames[variable]
|
|
290
|
+
Log.info("variable frame cache MISS: %s — loading", variable, layer=LAYER)
|
|
291
|
+
frame = load_variable(self._db, variable)
|
|
292
|
+
self._frames[variable] = frame
|
|
293
|
+
return frame
|
|
294
|
+
|
|
295
|
+
# Deduped for the same reason get_table is, one layer down: two DIFFERENT
|
|
296
|
+
# table keys over the same variable (a plot's measure and a factor join,
|
|
297
|
+
# say) both land here, and each would otherwise read the whole variable.
|
|
298
|
+
return self._frame_single_flight().run(
|
|
299
|
+
variable,
|
|
300
|
+
_load,
|
|
301
|
+
on_wait=lambda: Log.info(
|
|
302
|
+
"variable frame load already in flight: %s — waiting for it",
|
|
303
|
+
variable,
|
|
304
|
+
layer=LAYER,
|
|
305
|
+
),
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
def _frame_single_flight(self):
|
|
309
|
+
"""The in-flight map for :meth:`_variable_frame`, created on first use."""
|
|
310
|
+
from scistackplot.sources.base import _lazy_attr
|
|
311
|
+
|
|
312
|
+
return _lazy_attr(self, "_frame_inflight", SingleFlight)
|
|
173
313
|
|
|
174
314
|
def _build_table(
|
|
175
315
|
self,
|
|
@@ -330,6 +470,7 @@ class ScidbSource(BaseSource):
|
|
|
330
470
|
factors,
|
|
331
471
|
layer=LAYER,
|
|
332
472
|
)
|
|
473
|
+
table.reducer = self._reducer()
|
|
333
474
|
return table
|
|
334
475
|
|
|
335
476
|
def _attach_factor_variables(
|
|
@@ -546,6 +687,7 @@ class ScidbSource(BaseSource):
|
|
|
546
687
|
levels,
|
|
547
688
|
layer=LAYER,
|
|
548
689
|
)
|
|
690
|
+
table.reducer = self._reducer()
|
|
549
691
|
return table
|
|
550
692
|
|
|
551
693
|
def variant_graph(self, variable: str, functions: list[str] | None = None) -> dict:
|
|
@@ -607,17 +749,27 @@ class ScidbSource(BaseSource):
|
|
|
607
749
|
while field_factor in id_vars: # never shadow a schema key
|
|
608
750
|
field_factor += "_"
|
|
609
751
|
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
752
|
+
# Timed because this is where the row count multiplies by the field
|
|
753
|
+
# count — a 10-field record becomes 10 rows, each still holding a whole
|
|
754
|
+
# signal — and every one of those cells is carried through the rest of
|
|
755
|
+
# the pipeline whether or not the plot asks for that field.
|
|
756
|
+
with Log.timer(
|
|
757
|
+
f"melt_fields({variable_frame.name})",
|
|
758
|
+
layer=LAYER,
|
|
759
|
+
extra=f"{len(usable)} field(s)",
|
|
760
|
+
):
|
|
761
|
+
melted = frame.melt(
|
|
762
|
+
id_vars=id_vars,
|
|
763
|
+
value_vars=usable,
|
|
764
|
+
var_name=field_factor,
|
|
765
|
+
value_name=measure,
|
|
766
|
+
)
|
|
616
767
|
Log.info(
|
|
617
|
-
"melted %r: %d field(s) -> %d row(s), field factor %r",
|
|
768
|
+
"melted %r: %d field(s) -> %d row(s), %s, field factor %r",
|
|
618
769
|
variable_frame.name,
|
|
619
770
|
len(usable),
|
|
620
771
|
len(melted),
|
|
772
|
+
format_extent(frame_extent(melted, [measure])),
|
|
621
773
|
field_factor,
|
|
622
774
|
layer=LAYER,
|
|
623
775
|
)
|
|
@@ -95,6 +95,55 @@ def selection_for(
|
|
|
95
95
|
return selection
|
|
96
96
|
|
|
97
97
|
|
|
98
|
+
def branch_params_for(selection: dict[str, Any]) -> dict[str, Any]:
|
|
99
|
+
"""The inverse of :func:`selection_for`: a column-keyed selection →
|
|
100
|
+
``Variant.branch_params``.
|
|
101
|
+
|
|
102
|
+
Needed because the schema location picker asks **scidb** a question about a
|
|
103
|
+
variant the plotting layer is holding: "which locations have this one?"
|
|
104
|
+
``scidb.locations.location_states`` takes a ``branch_params_filter``, the
|
|
105
|
+
same dict ``Variant`` builds and ``find_record_id`` matches on, so the
|
|
106
|
+
picker's dots and a ``Variant(...).load()`` cannot disagree about which
|
|
107
|
+
records a selection names.
|
|
108
|
+
|
|
109
|
+
Three mappings, and the third is the one worth reading twice:
|
|
110
|
+
|
|
111
|
+
* ``Code:<fn>`` → ``__code__.<fn>``;
|
|
112
|
+
* anything else is already ``fn.param``, scidb's own namespacing, and
|
|
113
|
+
passes through — including a **list** value, which scidb already reads as
|
|
114
|
+
membership (``database.py``'s list-valued branch params), so the picker's
|
|
115
|
+
multi-checkbox needs no new scidb work;
|
|
116
|
+
* the :data:`~scistackplotdb.load.LATEST_COLUMN` flag → ``__code__ =
|
|
117
|
+
"latest"``. Both spell the same **per-location** rule (each schema
|
|
118
|
+
location contributes its own newest record, rather than the global highest
|
|
119
|
+
ordinal), which is exactly the meaning this picker needs: a location never
|
|
120
|
+
re-run under the newest code must show up as the record it actually has,
|
|
121
|
+
not as red.
|
|
122
|
+
|
|
123
|
+
``CodeIsLatest: False`` has no scidb spelling — "not the latest" is not a
|
|
124
|
+
pin — so it is dropped with a warning rather than silently inverted.
|
|
125
|
+
"""
|
|
126
|
+
from scidb.variant import CODE_PIN_PREFIX, LATEST_VERSION
|
|
127
|
+
|
|
128
|
+
out: dict[str, Any] = {}
|
|
129
|
+
for key, value in (selection or {}).items():
|
|
130
|
+
if key == LATEST_COLUMN:
|
|
131
|
+
if value is False:
|
|
132
|
+
Log.warn(
|
|
133
|
+
"selection pins %s=False, which scidb cannot express "
|
|
134
|
+
"(there is no 'not the latest' pin) — ignoring it",
|
|
135
|
+
LATEST_COLUMN,
|
|
136
|
+
layer=LAYER,
|
|
137
|
+
)
|
|
138
|
+
continue
|
|
139
|
+
out[CODE_PIN_PREFIX] = LATEST_VERSION
|
|
140
|
+
elif key.startswith(CODE_FACTOR_PREFIX):
|
|
141
|
+
out[f"{CODE_PIN_PREFIX}.{key[len(CODE_FACTOR_PREFIX):]}"] = value
|
|
142
|
+
else:
|
|
143
|
+
out[key] = value
|
|
144
|
+
return out
|
|
145
|
+
|
|
146
|
+
|
|
98
147
|
def _axes_of(table: LongTable | None) -> list[dict]:
|
|
99
148
|
if table is None:
|
|
100
149
|
return []
|
|
File without changes
|
|
File without changes
|
|
File without changes
|