bts-pivot 0.20.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bts_pivot/__init__.py +464 -0
- bts_pivot/__main__.py +4 -0
- bts_pivot/_binning.py +614 -0
- bts_pivot/_chains.py +153 -0
- bts_pivot/_cluster.py +477 -0
- bts_pivot/_compare.py +1152 -0
- bts_pivot/_density.py +320 -0
- bts_pivot/_deps.py +110 -0
- bts_pivot/_duckdb_engine.py +265 -0
- bts_pivot/_explain.py +339 -0
- bts_pivot/_fit.py +1110 -0
- bts_pivot/_hmm.py +230 -0
- bts_pivot/_html.py +752 -0
- bts_pivot/_insights.py +410 -0
- bts_pivot/_io.py +207 -0
- bts_pivot/_llm.py +129 -0
- bts_pivot/_log.py +185 -0
- bts_pivot/_mixture.py +190 -0
- bts_pivot/_novelty.py +226 -0
- bts_pivot/_profile.py +349 -0
- bts_pivot/_prompt.py +216 -0
- bts_pivot/_render.py +188 -0
- bts_pivot/_semantic.py +200 -0
- bts_pivot/_sparkline.py +98 -0
- bts_pivot/_spikes.py +293 -0
- bts_pivot/_survey.py +642 -0
- bts_pivot/_view.py +1511 -0
- bts_pivot/agent.py +749 -0
- bts_pivot/cli.py +198 -0
- bts_pivot/mcp_server.py +423 -0
- bts_pivot/sample.py +115 -0
- bts_pivot/ui.py +1768 -0
- bts_pivot/ui_fields.py +571 -0
- bts_pivot/ui_output.py +122 -0
- bts_pivot-0.20.0.dist-info/METADATA +1250 -0
- bts_pivot-0.20.0.dist-info/RECORD +40 -0
- bts_pivot-0.20.0.dist-info/WHEEL +5 -0
- bts_pivot-0.20.0.dist-info/entry_points.txt +3 -0
- bts_pivot-0.20.0.dist-info/licenses/LICENSE +21 -0
- bts_pivot-0.20.0.dist-info/top_level.txt +1 -0
bts_pivot/__init__.py
ADDED
|
@@ -0,0 +1,464 @@
|
|
|
1
|
+
"""bts-pivot: auto-fitted pivot tables that toggle to histograms and back, with slicing.
|
|
2
|
+
|
|
3
|
+
Quick start::
|
|
4
|
+
|
|
5
|
+
import bts_pivot as bp
|
|
6
|
+
|
|
7
|
+
v = bp.fit(df) # auto-chooses rows/cols/measure to fit a 40x12 box
|
|
8
|
+
print(v) # pivot table
|
|
9
|
+
print(v.toggle()) # the same data as a histogram
|
|
10
|
+
print(v.slice(action="deny")) # sliced pivot
|
|
11
|
+
print(v.histogram("bytes")) # histogram of a specific column
|
|
12
|
+
v.suggest() # alternative layouts
|
|
13
|
+
v.cluster(4) # group similar rows
|
|
14
|
+
bp.explore(df) # Jupyter menus (ipywidgets)
|
|
15
|
+
bp.fit("huge.parquet") # surveyed, fitted on a sample, aggregated page by page
|
|
16
|
+
bp.verbose(); bp.stats(7) # scrolling step log; the seven costliest steps
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from typing import Any, Optional, Sequence, Union
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
from . import agent, sample
|
|
26
|
+
from ._binning import RULES, bin_count, bin_edges, bin_labels, kde
|
|
27
|
+
from ._chains import sequences, steady_state, transition_matrix, transitions
|
|
28
|
+
from ._cluster import COMETHODS, METHODS, cluster_frame, cluster_rows, cocluster, dbscan, kmeans
|
|
29
|
+
from ._density import DistFit, fit_distribution, rank_distributions
|
|
30
|
+
from ._density import FAMILIES as DIST_FAMILIES
|
|
31
|
+
from ._mixture import GMMFit, choose_gmm_k, fit_gmm, mixture_cutpoints
|
|
32
|
+
from ._hmm import HMMFit, choose_hmm_states, decode_regimes, fit_hmm
|
|
33
|
+
from ._deps import dependency_pairs, mutual_info_matrix
|
|
34
|
+
from ._fit import AGGS, DEFAULT_WEIGHTS, OBJECTIVES, Dim, DimSpec, FitOptions, Layout, build_table, fit_layout, suggest_layouts
|
|
35
|
+
from ._io import load
|
|
36
|
+
from ._log import log, stats, verbose
|
|
37
|
+
from ._profile import ColumnProfile, Profile
|
|
38
|
+
from ._profile import profile as _profile
|
|
39
|
+
from ._semantic import HIERARCHY, infer_semantic
|
|
40
|
+
from ._survey import Machine, PagedSource, Plan, Survey, downcast, load_planned, survey
|
|
41
|
+
from ._view import HIST, PIVOT, Derived, Filter, View
|
|
42
|
+
from ._compare import ADDITIVE, METRICS, METRIC_HELP, Comparison, Facets
|
|
43
|
+
from ._explain import Explanation
|
|
44
|
+
from ._prompt import DEFAULT_QUESTION, Prompt
|
|
45
|
+
from ._sparkline import sparkline_table
|
|
46
|
+
from ._spikes import BASELINES as SPIKE_BASELINES
|
|
47
|
+
from ._novelty import KINDS as NOVELTY_KINDS
|
|
48
|
+
|
|
49
|
+
__version__ = "0.20.0"
|
|
50
|
+
|
|
51
|
+
_PLANNED_KEYS = ("memory_budget_mb", "mode", "columns", "query", "table", "sample_rows", "page_rows")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _needs_plan(data: Any) -> bool:
|
|
55
|
+
"""Paths and DuckDB sources go through the survey; frames and records load directly."""
|
|
56
|
+
if isinstance(data, (str, bytes)) or hasattr(data, "__fspath__"):
|
|
57
|
+
return True
|
|
58
|
+
return type(data).__name__ == "DuckDBPyConnection"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _load_for_fit(data: Any, opts: dict):
|
|
62
|
+
"""Split fit() kwargs into loading knobs and FitOptions; return (frame_or_paged, survey)."""
|
|
63
|
+
planned = {k: opts.pop(k) for k in _PLANNED_KEYS if k in opts}
|
|
64
|
+
if _needs_plan(data) or planned:
|
|
65
|
+
if not _needs_plan(data) and planned.get("mode", "auto") == "auto" and "memory_budget_mb" not in planned:
|
|
66
|
+
frame = load(data)
|
|
67
|
+
cols = planned.get("columns")
|
|
68
|
+
if cols:
|
|
69
|
+
missing = [c for c in cols if c not in frame.columns]
|
|
70
|
+
if missing:
|
|
71
|
+
raise KeyError(f"unknown column(s) {missing}; available: {list(frame.columns)[:20]}")
|
|
72
|
+
frame = frame[list(cols)]
|
|
73
|
+
return frame, None
|
|
74
|
+
frame, sv = load_planned(data, **planned)
|
|
75
|
+
return frame, sv
|
|
76
|
+
return load(data), None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def fit(
|
|
80
|
+
data: Any,
|
|
81
|
+
*,
|
|
82
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
83
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
84
|
+
values: Optional[str] = None,
|
|
85
|
+
agg: Optional[str] = None,
|
|
86
|
+
options: Optional[FitOptions] = None,
|
|
87
|
+
**opts: Any,
|
|
88
|
+
) -> View:
|
|
89
|
+
"""Auto-fit ``data`` (frame, path, records, ``duckdb://`` URL ...) into a pivot :class:`View`.
|
|
90
|
+
|
|
91
|
+
Fix any of ``rows``/``cols``/``values``/``agg`` and the rest is chosen for you.
|
|
92
|
+
Keyword options (``max_rows``, ``max_cols``, ``layers``, ``aspect``, ``bins``,
|
|
93
|
+
``scale``, ``engine``, ``objective`` ...) are :class:`FitOptions` fields -
|
|
94
|
+
``engine="duckdb"`` runs the table build as SQL against DuckDB instead of pandas
|
|
95
|
+
(needs the ``duckdb`` package); ``objective="bic"`` scores candidate layouts by a
|
|
96
|
+
BIC model-selection comparison (is the row/column association worth the table's own
|
|
97
|
+
complexity?) instead of the default entropy + mutual-info heuristic.
|
|
98
|
+
|
|
99
|
+
Files and DuckDB sources are surveyed first (rows, size on disk, estimated memory
|
|
100
|
+
against the machine's RAM); too-big data is fitted on a sample and aggregated page by
|
|
101
|
+
page. Loading knobs: ``memory_budget_mb`` (default: half the free RAM), ``mode``
|
|
102
|
+
(``"auto"`` | ``"full"`` | ``"downcast"`` | ``"sample"`` | ``"paged"``), ``columns``,
|
|
103
|
+
``query`` / ``table`` for DuckDB, ``sample_rows``, ``page_rows``.
|
|
104
|
+
"""
|
|
105
|
+
frame, sv = _load_for_fit(data, opts)
|
|
106
|
+
v = View.fit(frame, rows=rows, cols=cols, values=values, agg=agg, options=options, **opts)
|
|
107
|
+
return v if sv is None else View(frame, v.layout, options=v.options, spec=v._spec, survey=sv)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
pivot = fit
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def histogram(
|
|
114
|
+
data: Any,
|
|
115
|
+
on: Optional[str] = None,
|
|
116
|
+
by: Optional[Union[str, Sequence[DimSpec]]] = None,
|
|
117
|
+
*,
|
|
118
|
+
bins: Optional[Union[str, int]] = None,
|
|
119
|
+
values: Optional[str] = None,
|
|
120
|
+
agg: Optional[str] = None,
|
|
121
|
+
scale: Optional[str] = None,
|
|
122
|
+
options: Optional[FitOptions] = None,
|
|
123
|
+
**opts: Any,
|
|
124
|
+
) -> View:
|
|
125
|
+
"""A histogram :class:`View` of ``data`` (``toggle()`` gives the matching pivot)."""
|
|
126
|
+
frame, sv = _load_for_fit(data, opts)
|
|
127
|
+
base = View.fit(frame, options=options, **opts)
|
|
128
|
+
if sv is not None:
|
|
129
|
+
base = View(frame, base.layout, options=base.options, spec=base._spec, survey=sv)
|
|
130
|
+
if on is None and by is None and values is None and bins is None and scale is None:
|
|
131
|
+
return base.toggle()
|
|
132
|
+
return base.histogram(on, by, bins=bins, values=values, agg=agg, scale=scale)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def profile(data: Any, **kw: Any) -> Profile:
|
|
136
|
+
"""Profile the columns of ``data`` (kinds, semantic types, cardinality, nulls, time series)."""
|
|
137
|
+
return _profile(load(data), **kw)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def suggest(data: Any, n: int = 5, **opts: Any) -> list:
|
|
141
|
+
"""The ``n`` best distinct layouts for ``data``, best first (the auto-guess menu)."""
|
|
142
|
+
return [lay for _, lay in suggest_layouts(load(data), FitOptions().replace(**opts) if opts else None, n)]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def distribution(data: Any, column: str, **kw: Any):
|
|
146
|
+
"""Best-fitting probability distribution for one numeric column of ``data`` (BIC over
|
|
147
|
+
normal/lognormal/exponential/gamma/uniform/poisson/geometric/bernoulli/discrete-uniform).
|
|
148
|
+
|
|
149
|
+
``kw`` forwards to :func:`fit_distribution` (``families=``, ``discrete_max=``, ``min_n=``).
|
|
150
|
+
"""
|
|
151
|
+
df = load(data)
|
|
152
|
+
if column not in df.columns:
|
|
153
|
+
raise KeyError(f"unknown column {column!r}")
|
|
154
|
+
return fit_distribution(df[column], **kw)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def modes(data: Any, column: str, k: Optional[int] = None, **kw: Any) -> Optional[list]:
|
|
158
|
+
"""How many peaks does this numeric column have, and where? A Gaussian mixture fit
|
|
159
|
+
(component count chosen by BIC unless ``k`` is given), as a list of
|
|
160
|
+
``{"weight", "mean", "std"}`` dicts sorted by mean, or ``None`` if there isn't enough
|
|
161
|
+
data. No scipy/sklearn: EM from scratch, see :mod:`bts_pivot._mixture`.
|
|
162
|
+
"""
|
|
163
|
+
df = load(data)
|
|
164
|
+
if column not in df.columns:
|
|
165
|
+
raise KeyError(f"unknown column {column!r}")
|
|
166
|
+
import numpy as _np
|
|
167
|
+
|
|
168
|
+
x = _np.asarray(df[column], dtype=float)
|
|
169
|
+
x = x[_np.isfinite(x)]
|
|
170
|
+
if x.size < 8:
|
|
171
|
+
return None
|
|
172
|
+
kk = k if k is not None else choose_gmm_k(x.reshape(-1, 1), **kw)
|
|
173
|
+
fit = fit_gmm(x.reshape(-1, 1), max(1, kk), **{k2: v for k2, v in kw.items() if k2 != "k_max"})
|
|
174
|
+
return fit.components()
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def llm_context(
|
|
178
|
+
data: Any,
|
|
179
|
+
*,
|
|
180
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
181
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
182
|
+
values: Optional[str] = None,
|
|
183
|
+
agg: Optional[str] = None,
|
|
184
|
+
max_rows: int = 30,
|
|
185
|
+
max_cols: int = 12,
|
|
186
|
+
notes: bool = True,
|
|
187
|
+
**opts: Any,
|
|
188
|
+
) -> dict:
|
|
189
|
+
"""``data`` auto-fitted (or laid out as given) and packaged for an LLM:
|
|
190
|
+
``{"description", "metadata", "table"}`` - a short natural-language summary, compact
|
|
191
|
+
structured facts (schema, shape, slices, measure), and the table itself as a
|
|
192
|
+
GitHub-flavored markdown string, truncated to ``max_rows`` x ``max_cols``.
|
|
193
|
+
|
|
194
|
+
Equivalent to ``bp.fit(data, ...).llm_context(...)``; see :meth:`View.llm_context`
|
|
195
|
+
for what each field means, and :func:`bts_pivot.agent.llm_context` for the same
|
|
196
|
+
thing from the plain-JSON agent surface (a source path/records instead of a frame,
|
|
197
|
+
optional ``filters``).
|
|
198
|
+
"""
|
|
199
|
+
v = fit(data, rows=rows, cols=cols, values=values, agg=agg, **opts)
|
|
200
|
+
return v.llm_context(max_rows=max_rows, max_cols=max_cols, notes=notes)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def insights(
|
|
204
|
+
data: Any,
|
|
205
|
+
*,
|
|
206
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
207
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
208
|
+
values: Optional[str] = None,
|
|
209
|
+
agg: Optional[str] = None,
|
|
210
|
+
sensitivity: float = 0.5,
|
|
211
|
+
max_findings: int = 15,
|
|
212
|
+
max_pairs: int = 5,
|
|
213
|
+
**opts: Any,
|
|
214
|
+
) -> dict:
|
|
215
|
+
"""A rich, local, non-LLM analysis of ``data`` (auto-fitted first, so a 2-D layout is
|
|
216
|
+
available for the surprising-cell check): column summaries, distributions, a
|
|
217
|
+
Gaussian-mixture modality check, skew, concentration, outliers, correlated column
|
|
218
|
+
pairs, and the most surprising pivot cells - ranked findings, not a raw dump.
|
|
219
|
+
|
|
220
|
+
Equivalent to ``bp.fit(data, ...).insights(...)``; see :meth:`View.insights` for
|
|
221
|
+
what ``sensitivity`` controls and why it isn't called "temperature".
|
|
222
|
+
"""
|
|
223
|
+
v = fit(data, rows=rows, cols=cols, values=values, agg=agg, **opts)
|
|
224
|
+
return v.insights(sensitivity=sensitivity, max_findings=max_findings, max_pairs=max_pairs)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def spikes(
|
|
228
|
+
data: Any,
|
|
229
|
+
column: Optional[str] = None,
|
|
230
|
+
*,
|
|
231
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
232
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
233
|
+
values: Optional[str] = None,
|
|
234
|
+
agg: Optional[str] = None,
|
|
235
|
+
n: int = 10,
|
|
236
|
+
z: float = 3.0,
|
|
237
|
+
min_support: int = 5,
|
|
238
|
+
baseline: str = "auto",
|
|
239
|
+
shifts: bool = True,
|
|
240
|
+
**opts: Any,
|
|
241
|
+
) -> pd.DataFrame:
|
|
242
|
+
"""Which rows of the (auto-fitted) table of ``data`` moved over time, when, and by how
|
|
243
|
+
much against their own history: spikes, drops and step changes per row, scored
|
|
244
|
+
against a seasonal, share-of-total or plain robust baseline of the row's other time
|
|
245
|
+
buckets. Equivalent to ``bp.fit(data, ...).spikes(column, ...)``; see
|
|
246
|
+
:meth:`View.spikes` for the columns returned and how the baseline is chosen.
|
|
247
|
+
"""
|
|
248
|
+
v = fit(data, rows=rows, cols=cols, values=values, agg=agg, **opts)
|
|
249
|
+
return v.spikes(column, n=n, z=z, min_support=min_support, baseline=baseline, shifts=shifts)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def novel(
|
|
253
|
+
data: Any,
|
|
254
|
+
entity: Optional[str] = None,
|
|
255
|
+
attr: Optional[str] = None,
|
|
256
|
+
*,
|
|
257
|
+
since: Any = 0.25,
|
|
258
|
+
time: Optional[str] = None,
|
|
259
|
+
n: int = 10,
|
|
260
|
+
min_support: int = 3,
|
|
261
|
+
**opts: Any,
|
|
262
|
+
) -> pd.DataFrame:
|
|
263
|
+
"""What is new in the recent part of ``data``, per entity: entities never seen before
|
|
264
|
+
the split, pairs an entity never made before, values nobody had used, and entities
|
|
265
|
+
whose fan-out jumped. Equivalent to ``bp.fit(data, ...).novel(entity, attr, ...)``;
|
|
266
|
+
see :meth:`View.novel` for ``since`` and the columns returned.
|
|
267
|
+
"""
|
|
268
|
+
v = fit(data, **opts)
|
|
269
|
+
return v.novel(entity, attr, since=since, time=time, n=n, min_support=min_support)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _split_fit_kwargs(opts: dict) -> dict:
|
|
273
|
+
"""Pop the keywords that are neither :class:`FitOptions` fields nor loading knobs
|
|
274
|
+
(they name columns to split on)."""
|
|
275
|
+
fit_keys = set(FitOptions.__dataclass_fields__) | set(_PLANNED_KEYS)
|
|
276
|
+
return {k: opts.pop(k) for k in list(opts) if k not in fit_keys}
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def compare(
|
|
280
|
+
data: Any,
|
|
281
|
+
*args: Any,
|
|
282
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
283
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
284
|
+
values: Optional[str] = None,
|
|
285
|
+
agg: Optional[str] = None,
|
|
286
|
+
metric: Optional[str] = None,
|
|
287
|
+
names: Optional[Sequence[str]] = None,
|
|
288
|
+
**opts: Any,
|
|
289
|
+
) -> Comparison:
|
|
290
|
+
"""Two sides of ``data`` on one shared, auto-fitted layout, cell by cell::
|
|
291
|
+
|
|
292
|
+
bp.compare(df, action="deny") # deny vs the rest, as lift
|
|
293
|
+
bp.compare(df, "action", "deny", "allow") # deny vs allow
|
|
294
|
+
bp.compare(df, "bytes > 1000", metric="delta") # a query vs its complement
|
|
295
|
+
bp.compare("events.parquet", {"timestamp": "2026-03-02"}, {"timestamp": "2026-03-01"})
|
|
296
|
+
|
|
297
|
+
Equivalent to ``bp.fit(data, rows=..., ...).compare(...)``: keyword arguments that are
|
|
298
|
+
fit options (``max_rows``, ``layers``, ``memory_budget_mb`` ...) go to the fit, any
|
|
299
|
+
other ``column=value`` keyword is the split. See :meth:`View.compare` for the forms,
|
|
300
|
+
the metrics, and why the layout is frozen across the two sides.
|
|
301
|
+
"""
|
|
302
|
+
split = _split_fit_kwargs(opts)
|
|
303
|
+
v = fit(data, rows=rows, cols=cols, values=values, agg=agg, **opts)
|
|
304
|
+
return v.compare(*args, metric=metric, names=names, **split)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def facet(
|
|
308
|
+
data: Any,
|
|
309
|
+
column: str,
|
|
310
|
+
n: int = 6,
|
|
311
|
+
*,
|
|
312
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
313
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
314
|
+
values: Optional[str] = None,
|
|
315
|
+
agg: Optional[str] = None,
|
|
316
|
+
levels: Optional[Sequence[Any]] = None,
|
|
317
|
+
**opts: Any,
|
|
318
|
+
) -> Facets:
|
|
319
|
+
"""Small multiples of ``data``: the auto-fitted pivot once per value of ``column``
|
|
320
|
+
(the ``n`` most frequent, or ``levels``), all on one layout and one colour scale.
|
|
321
|
+
Equivalent to ``bp.fit(data, ...).facet(column, n, levels=levels)``; see
|
|
322
|
+
:meth:`View.facet`.
|
|
323
|
+
"""
|
|
324
|
+
v = fit(data, rows=rows, cols=cols, values=values, agg=agg, **opts)
|
|
325
|
+
return v.facet(column, n, levels=levels)
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def prompt(
|
|
329
|
+
data: Any,
|
|
330
|
+
question: Optional[str] = None,
|
|
331
|
+
*,
|
|
332
|
+
rows: Optional[Sequence[DimSpec]] = None,
|
|
333
|
+
cols: Optional[Sequence[DimSpec]] = None,
|
|
334
|
+
values: Optional[str] = None,
|
|
335
|
+
agg: Optional[str] = None,
|
|
336
|
+
table_max_rows: int = 30,
|
|
337
|
+
table_max_cols: int = 12,
|
|
338
|
+
**opts: Any,
|
|
339
|
+
) -> Prompt:
|
|
340
|
+
"""``data`` auto-fitted (or laid out as given) and packaged as one self-contained LLM
|
|
341
|
+
prompt: the dataset's columns, the table, the surprising cells, optionally insights /
|
|
342
|
+
a comparison / an explained cell, and ``question``. Equivalent to
|
|
343
|
+
``bp.fit(data, ...).prompt(question, ...)``; keyword arguments that are fit options
|
|
344
|
+
go to the fit, the rest (``insights=``, ``compare=``, ``explain=``, ``anomalies=`` ...)
|
|
345
|
+
to :meth:`View.prompt`. See there for what each section holds.
|
|
346
|
+
"""
|
|
347
|
+
prompt_kw = _split_fit_kwargs(opts)
|
|
348
|
+
v = fit(data, rows=rows, cols=cols, values=values, agg=agg, **opts)
|
|
349
|
+
return v.prompt(question, max_rows=table_max_rows, max_cols=table_max_cols, **prompt_kw)
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def explore(data: Any, **kw: Any) -> Any:
|
|
353
|
+
"""Interactive Jupyter explorer (needs ``ipywidgets``): menus to alter, slice, best-fit,
|
|
354
|
+
reduce and cluster, with heatmap pivots and SVG histograms."""
|
|
355
|
+
from .ui import explore as _explore
|
|
356
|
+
|
|
357
|
+
return _explore(data, **kw)
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def cluster(data: Any, columns: Optional[Sequence[str]] = None, k: Optional[int] = None, *, method: str = "kmeans", name: str = "cluster") -> pd.DataFrame:
|
|
361
|
+
"""``data`` with an extra ``cluster`` column over numeric ``columns`` (default: all).
|
|
362
|
+
|
|
363
|
+
``method``: ``"kmeans"`` (auto k), ``"dbscan"`` or ``"hdbscan"`` (outliers -> ``noise``)."""
|
|
364
|
+
df = load(data)
|
|
365
|
+
out = df.copy()
|
|
366
|
+
out[name] = cluster_frame(df, columns, k, method=method, name=name)
|
|
367
|
+
return out
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def chains(
|
|
371
|
+
data: Any,
|
|
372
|
+
state: str,
|
|
373
|
+
*,
|
|
374
|
+
by: Optional[Union[str, Sequence[str]]] = None,
|
|
375
|
+
time: Optional[str] = None,
|
|
376
|
+
order: int = 1,
|
|
377
|
+
normalize: bool = False,
|
|
378
|
+
**opts: Any,
|
|
379
|
+
) -> View:
|
|
380
|
+
"""Markov transition matrix of ``state`` as a :class:`View` (rows = from, columns = to).
|
|
381
|
+
|
|
382
|
+
``by`` keeps sequences inside an entity (user, source IP); ``time`` orders them;
|
|
383
|
+
``order=2`` conditions on the previous two states; ``normalize`` shows row
|
|
384
|
+
probabilities instead of counts. Everything else (toggle, slice, cluster, cocluster,
|
|
385
|
+
style) works as on any pivot. See :func:`sequences` for the most frequent chains.
|
|
386
|
+
"""
|
|
387
|
+
df = load(data)
|
|
388
|
+
long = transitions(df, state, by=by, time=time, order=order)
|
|
389
|
+
if long.empty:
|
|
390
|
+
raise ValueError("no transitions found (need at least two consecutive states per group)")
|
|
391
|
+
values, agg = ("prob", "sum") if normalize else ("count", "sum")
|
|
392
|
+
max_states = max(opts.pop("max_rows", 40), opts.pop("max_cols", 12))
|
|
393
|
+
v = View.fit(long, rows=[{"column": "from", "top": max_states - 1}] if long["from"].nunique() > max_states else ["from"],
|
|
394
|
+
cols=[{"column": "to", "top": max_states - 1}] if long["to"].nunique() > max_states else ["to"],
|
|
395
|
+
values=values, agg=agg, max_rows=max_states, max_cols=max_states, **opts)
|
|
396
|
+
return v.style(heat="row" if normalize else "table")
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def regimes(
|
|
400
|
+
data: Any,
|
|
401
|
+
state: str,
|
|
402
|
+
*,
|
|
403
|
+
by: Optional[Union[str, Sequence[str]]] = None,
|
|
404
|
+
time: Optional[str] = None,
|
|
405
|
+
n_states: Optional[int] = None,
|
|
406
|
+
k_max: int = 4,
|
|
407
|
+
seed: int = 0,
|
|
408
|
+
**opts: Any,
|
|
409
|
+
) -> View:
|
|
410
|
+
"""Hidden Markov regimes over ``state`` as a :class:`View` (rows = regime, columns =
|
|
411
|
+
``state``), so you can see what each regime looks like and ``toggle()`` to a
|
|
412
|
+
histogram of it.
|
|
413
|
+
|
|
414
|
+
A Baum-Welch fit (no hmmlearn/scipy, see :mod:`bts_pivot._hmm`) decodes each row
|
|
415
|
+
into one of a small number of hidden regimes from the sequence of ``state`` values,
|
|
416
|
+
e.g. a user's logins drifting from a "normal" regime into a "credential-stuffing"
|
|
417
|
+
regime. ``by`` keeps sequences inside an entity (user, source IP); ``time`` orders
|
|
418
|
+
them; ``n_states`` fixes the regime count (default: chosen by BIC, up to ``k_max``).
|
|
419
|
+
Regimes are numbered by how common they are (``"regime 1"`` = most common).
|
|
420
|
+
"""
|
|
421
|
+
df = load(data)
|
|
422
|
+
regime = decode_regimes(df, state, by=by, time=time, n_states=n_states, k_max=k_max, seed=seed)
|
|
423
|
+
out = df.copy()
|
|
424
|
+
out["regime"] = regime.values
|
|
425
|
+
max_cols = opts.pop("max_cols", 12)
|
|
426
|
+
cols = [{"column": state, "top": max_cols - 1}] if df[state].nunique() > max_cols else [state]
|
|
427
|
+
v = View.fit(out, rows=["regime"], cols=cols, values=None, agg="count", max_cols=max_cols, **opts)
|
|
428
|
+
return v.style(heat="row")
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def dependencies(data: Any, columns: Optional[Sequence[str]] = None, *, bins: int = 10, max_cols: int = 30, **opts: Any) -> View:
|
|
432
|
+
"""Which columns of ``data`` move together, as a square :class:`View` (rows = cols =
|
|
433
|
+
column names, cells = normalized mutual information, 0..1).
|
|
434
|
+
|
|
435
|
+
No correlation-matrix assumption of linearity or numeric-only columns: every column
|
|
436
|
+
is discretized (numeric/datetime into quantile bins, categorical/boolean by top-N)
|
|
437
|
+
and scored by bias-corrected mutual information, so a categorical/numeric pair (e.g.
|
|
438
|
+
``protocol`` and ``dst_port``) shows up just as well as two numeric ones. ``.toggle()``
|
|
439
|
+
turns it into a histogram of each column's total association with everything else.
|
|
440
|
+
See :func:`bts_pivot.mutual_info_matrix` for the plain matrix.
|
|
441
|
+
"""
|
|
442
|
+
df = load(data)
|
|
443
|
+
long = dependency_pairs(df, columns, bins=bins, max_cols=max_cols)
|
|
444
|
+
v = View.fit(long, rows=["column_a"], cols=["column_b"], values="association", agg="max",
|
|
445
|
+
max_rows=max_cols, max_cols=max_cols, **opts)
|
|
446
|
+
return v.style(heat="table")
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
__all__ = [
|
|
450
|
+
"fit", "pivot", "histogram", "profile", "load", "suggest", "explore", "cluster", "chains", "regimes", "dependencies",
|
|
451
|
+
"llm_context", "insights", "compare", "facet", "Comparison", "Facets", "METRICS", "METRIC_HELP", "ADDITIVE",
|
|
452
|
+
"Explanation", "prompt", "Prompt", "DEFAULT_QUESTION", "sparkline_table", "spikes", "SPIKE_BASELINES", "novel", "NOVELTY_KINDS",
|
|
453
|
+
"survey", "load_planned", "downcast", "stats", "verbose", "log", "distribution",
|
|
454
|
+
"sequences", "transitions", "transition_matrix", "steady_state",
|
|
455
|
+
"View", "Layout", "Dim", "FitOptions", "Filter", "Derived", "Profile", "ColumnProfile",
|
|
456
|
+
"Survey", "Plan", "Machine", "PagedSource", "DistFit",
|
|
457
|
+
"build_table", "fit_layout", "suggest_layouts", "bin_edges", "bin_count", "bin_labels", "kde",
|
|
458
|
+
"cluster_frame", "cluster_rows", "cocluster", "kmeans", "dbscan", "METHODS", "COMETHODS",
|
|
459
|
+
"infer_semantic", "HIERARCHY", "DEFAULT_WEIGHTS", "fit_distribution", "rank_distributions", "DIST_FAMILIES",
|
|
460
|
+
"modes", "GMMFit", "fit_gmm", "choose_gmm_k", "mixture_cutpoints",
|
|
461
|
+
"HMMFit", "fit_hmm", "choose_hmm_states", "decode_regimes",
|
|
462
|
+
"mutual_info_matrix", "dependency_pairs",
|
|
463
|
+
"RULES", "AGGS", "OBJECTIVES", "PIVOT", "HIST", "sample", "agent", "__version__",
|
|
464
|
+
]
|
bts_pivot/__main__.py
ADDED