aggregate_api 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aggregate_api/__init__.py +41 -0
- aggregate_api/__main__.py +154 -0
- aggregate_api/app.py +206 -0
- aggregate_api/audit.py +395 -0
- aggregate_api/bounds.py +331 -0
- aggregate_api/cache.py +319 -0
- aggregate_api/capability.py +823 -0
- aggregate_api/completion.py +219 -0
- aggregate_api/config.py +363 -0
- aggregate_api/cors.py +61 -0
- aggregate_api/examples.py +620 -0
- aggregate_api/layer_pricing.py +840 -0
- aggregate_api/library.py +94 -0
- aggregate_api/library_notes.py +96 -0
- aggregate_api/models.py +1407 -0
- aggregate_api/net.py +281 -0
- aggregate_api/pnl.py +101 -0
- aggregate_api/pricing.py +778 -0
- aggregate_api/resources.py +257 -0
- aggregate_api/routes/__init__.py +8 -0
- aggregate_api/routes/decl.py +327 -0
- aggregate_api/routes/examples.py +82 -0
- aggregate_api/routes/meta.py +282 -0
- aggregate_api/routes/objects.py +4119 -0
- aggregate_api/routes/status.py +466 -0
- aggregate_api/serializers.py +565 -0
- aggregate_api/sessions.py +353 -0
- aggregate_api/static/aggregate-api-logo-512.png +0 -0
- aggregate_api/static/aggregate-api-logo.png +0 -0
- aggregate_api/static/aggregate-api-trim.png +0 -0
- aggregate_api/static/android-chrome-192x192.png +0 -0
- aggregate_api/static/android-chrome-512x512.png +0 -0
- aggregate_api/static/apple-touch-icon.png +0 -0
- aggregate_api/static/assets/bootstrap-icons-BeopsB42.woff +0 -0
- aggregate_api/static/assets/bootstrap-icons-mSm7cUeB.woff2 +0 -0
- aggregate_api/static/assets/bootstrap-ohb1VZ53.js +5 -0
- aggregate_api/static/assets/codemirror-h62DHGGa.js +14 -0
- aggregate_api/static/assets/csv-grid.worker-DKzHGXac.js +4 -0
- aggregate_api/static/assets/echarts-B7o9sc00.js +40 -0
- aggregate_api/static/assets/echarts-gl-DG1Uf6wE.js +4282 -0
- aggregate_api/static/assets/lite-CUlcD8p4.css +1 -0
- aggregate_api/static/assets/lite-Dd2TnT4M.js +1 -0
- aggregate_api/static/assets/main-Bxhxa55v.css +9 -0
- aggregate_api/static/assets/main-CmoEiPit.js +9 -0
- aggregate_api/static/assets/tables-BHCF7qIF.js +8 -0
- aggregate_api/static/assets/tables-CxvajLr7.css +1 -0
- aggregate_api/static/favicon-16x16.png +0 -0
- aggregate_api/static/favicon-32x32.png +0 -0
- aggregate_api/static/favicon.ico +0 -0
- aggregate_api/static/index.html +912 -0
- aggregate_api/static/lite.html +83 -0
- aggregate_api/static/logo.png +0 -0
- aggregate_api/static/site.webmanifest +14 -0
- aggregate_api/static/sw.js +78 -0
- aggregate_api/status.py +536 -0
- aggregate_api/status_page.html +546 -0
- aggregate_api/tables.py +316 -0
- aggregate_api-1.0.0.dist-info/METADATA +187 -0
- aggregate_api-1.0.0.dist-info/RECORD +63 -0
- aggregate_api-1.0.0.dist-info/WHEEL +5 -0
- aggregate_api-1.0.0.dist-info/entry_points.txt +2 -0
- aggregate_api-1.0.0.dist-info/licenses/LICENSE +28 -0
- aggregate_api-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,565 @@
|
|
|
1
|
+
"""DataFrame and object → JSON-friendly helpers.
|
|
2
|
+
|
|
3
|
+
Centralized so route handlers stay thin: one ``frame_to_payload``
|
|
4
|
+
call converts a pandas DataFrame to the ``(columns, rows)`` shape
|
|
5
|
+
expected by :class:`aggregate_api.models.FrameResponse`.
|
|
6
|
+
|
|
7
|
+
Numeric/string coercion notes
|
|
8
|
+
-----------------------------
|
|
9
|
+
|
|
10
|
+
* NaN / Inf serialize to None -- strict JSON forbids them and
|
|
11
|
+
``json.dumps(float("nan"))`` produces non-parseable output that
|
|
12
|
+
some clients reject.
|
|
13
|
+
* MultiIndex columns are flattened with ``.`` joins so the wire
|
|
14
|
+
format stays a plain list-of-strings (``"freq.mean"``,
|
|
15
|
+
``"sev.cv"``). The client can split back on ``.`` if needed.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import math
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
import pandas as pd
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _safe(value: Any) -> Any:
|
|
28
|
+
"""Coerce one cell to a JSON-friendly value.
|
|
29
|
+
|
|
30
|
+
* Non-finite floats → None (strict JSON).
|
|
31
|
+
* numpy scalars → native Python (avoids ``numpy.int64`` objects
|
|
32
|
+
breaking ``json.dumps``).
|
|
33
|
+
* numpy arrays / list-likes → nested lists (recursive coerce).
|
|
34
|
+
* ``pd.NA`` / ``pd.NaT`` → None.
|
|
35
|
+
* Everything else falls back to ``str()`` so a single weird
|
|
36
|
+
cell can't 500 the whole response.
|
|
37
|
+
"""
|
|
38
|
+
if value is None:
|
|
39
|
+
return None
|
|
40
|
+
# pandas NA / NaT sentinels don't compare cleanly via ``is None``.
|
|
41
|
+
if value is pd.NA or value is pd.NaT:
|
|
42
|
+
return None
|
|
43
|
+
if isinstance(value, (np.integer, np.floating)):
|
|
44
|
+
value = value.item()
|
|
45
|
+
if isinstance(value, float):
|
|
46
|
+
# math.isfinite covers NaN, +Inf, -Inf.
|
|
47
|
+
if not math.isfinite(value):
|
|
48
|
+
return None
|
|
49
|
+
if isinstance(value, np.ndarray):
|
|
50
|
+
# 0-d arrays ``np.array(3.)``: .tolist() returns a scalar
|
|
51
|
+
# (not iterable). Higher-d: returns a (possibly nested) list.
|
|
52
|
+
as_list = value.tolist()
|
|
53
|
+
if isinstance(as_list, list):
|
|
54
|
+
return [_safe(v) for v in as_list]
|
|
55
|
+
return _safe(as_list)
|
|
56
|
+
if isinstance(value, (list, tuple)):
|
|
57
|
+
return [_safe(v) for v in value]
|
|
58
|
+
# ``str``, ``int``, ``bool``, ``float`` are JSON-native; let
|
|
59
|
+
# anything else through as its str() form so unexpected dtypes
|
|
60
|
+
# don't blow up serialization.
|
|
61
|
+
if isinstance(value, (str, int, bool, float)):
|
|
62
|
+
return value
|
|
63
|
+
return str(value)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _spec_value(value: Any) -> Any:
|
|
67
|
+
"""Coerce one spec value to a JSON-native one, recursively.
|
|
68
|
+
|
|
69
|
+
Parameters
|
|
70
|
+
----------
|
|
71
|
+
value : Any
|
|
72
|
+
A value out of a parser spec, or any value nested inside one.
|
|
73
|
+
|
|
74
|
+
Returns
|
|
75
|
+
-------
|
|
76
|
+
Any
|
|
77
|
+
``None``, a ``str``, an ``int``, a finite ``float``, a ``list``, or a
|
|
78
|
+
``dict`` with string keys. A non-finite float becomes the string
|
|
79
|
+
``"inf"``, ``"-inf"`` or ``"nan"``.
|
|
80
|
+
|
|
81
|
+
Notes
|
|
82
|
+
-----
|
|
83
|
+
See :func:`spec_to_payload` for why the non-finite cases are strings rather
|
|
84
|
+
than ``None``. The recursion has to cover dicts as well as lists: a ``port``
|
|
85
|
+
spec carries its unit list on a nested ``spec`` key as
|
|
86
|
+
``[("agg", "A", {...}), ...]``, so a dict appears three levels down.
|
|
87
|
+
"""
|
|
88
|
+
if value is None:
|
|
89
|
+
return None
|
|
90
|
+
if isinstance(value, np.generic):
|
|
91
|
+
# numpy scalar, including ``np.bool_`` and ``np.str_``, to its native
|
|
92
|
+
# Python twin before the type tests below.
|
|
93
|
+
value = value.item()
|
|
94
|
+
if isinstance(value, float):
|
|
95
|
+
if math.isnan(value):
|
|
96
|
+
return "nan"
|
|
97
|
+
if math.isinf(value):
|
|
98
|
+
return "inf" if value > 0 else "-inf"
|
|
99
|
+
# ``aggregate.parser._PercentNumber`` is a float subclass carrying the
|
|
100
|
+
# memory of a trailing ``%``. Hand back the plain float so nothing
|
|
101
|
+
# downstream has to serialize a type it does not know.
|
|
102
|
+
return float(value)
|
|
103
|
+
if isinstance(value, np.ndarray):
|
|
104
|
+
# 0-d arrays return a scalar from ``.tolist()``, not a list.
|
|
105
|
+
as_list = value.tolist()
|
|
106
|
+
if isinstance(as_list, list):
|
|
107
|
+
return [_spec_value(v) for v in as_list]
|
|
108
|
+
return _spec_value(as_list)
|
|
109
|
+
if isinstance(value, (list, tuple)):
|
|
110
|
+
# Tuples become arrays, which is the right wire form: a reinsurance
|
|
111
|
+
# layer reads as a three-element array, not as an object.
|
|
112
|
+
return [_spec_value(v) for v in value]
|
|
113
|
+
if isinstance(value, dict):
|
|
114
|
+
return {str(k): _spec_value(v) for k, v in value.items()}
|
|
115
|
+
if isinstance(value, (str, bool, int)):
|
|
116
|
+
return value
|
|
117
|
+
return str(value)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def spec_to_payload(spec: dict) -> dict:
|
|
121
|
+
"""Coerce a parser spec to a JSON-safe dict, keeping an unlimited limit readable.
|
|
122
|
+
|
|
123
|
+
Parameters
|
|
124
|
+
----------
|
|
125
|
+
spec : dict
|
|
126
|
+
The third member of the ``(kind, name, spec)`` triple
|
|
127
|
+
``aggregate.build.parser.parse`` returns: the transformer's own output,
|
|
128
|
+
whose values are floats, strings, ``None``, numpy arrays, and nested
|
|
129
|
+
lists, tuples and dicts of those.
|
|
130
|
+
|
|
131
|
+
Returns
|
|
132
|
+
-------
|
|
133
|
+
dict
|
|
134
|
+
The same keys, every value JSON-native, with a non-finite float as the
|
|
135
|
+
string ``"inf"``, ``"-inf"`` or ``"nan"``.
|
|
136
|
+
|
|
137
|
+
Notes
|
|
138
|
+
-----
|
|
139
|
+
**Why this is not :func:`_safe`, and must not be "fixed" into a call to
|
|
140
|
+
it.** ``_safe`` sends a non-finite float to ``None``, which is right for a
|
|
141
|
+
DataFrame cell, where a non-finite number is a missing one. In a spec it
|
|
142
|
+
destroys meaning. Position 1 of a reinsurance triple is a limit and a limit
|
|
143
|
+
is never absent, so ``agg_reins: [[0.2, null, 0.0]]`` cannot be told apart
|
|
144
|
+
from a limit the program failed to state, and ``None`` loses the sign that
|
|
145
|
+
separates ``inf`` from ``-inf``. The string form is also the spelling the
|
|
146
|
+
grammar itself accepts for an unlimited layer, so a caller round tripping a
|
|
147
|
+
limit back into DecL writes exactly what it read.
|
|
148
|
+
|
|
149
|
+
The coercion has to happen before the response is assembled rather than be
|
|
150
|
+
left to the serializer. Starlette's ``JSONResponse`` runs ``json.dumps``
|
|
151
|
+
with ``allow_nan=False``, so an unhandled ``inf`` does not ship bad JSON,
|
|
152
|
+
it **raises**, and the route answers 500. An ``inf`` is not an edge case
|
|
153
|
+
here: ``sev_ub`` is ``inf`` on the plainest one-line ``agg``.
|
|
154
|
+
"""
|
|
155
|
+
return {str(key): _spec_value(value) for key, value in spec.items()}
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def reset_index_safe(df: pd.DataFrame) -> pd.DataFrame:
|
|
159
|
+
"""``df.reset_index()`` that tolerates duplicate names.
|
|
160
|
+
|
|
161
|
+
``Aggregate.density_df`` has its index named ``'loss'`` and a
|
|
162
|
+
column named ``'loss'`` -- a plain ``reset_index()`` raises
|
|
163
|
+
``ValueError: cannot insert loss, already exists``. Detect that
|
|
164
|
+
case and return the frame unchanged (the column already carries
|
|
165
|
+
the index values).
|
|
166
|
+
|
|
167
|
+
For a MultiIndex with one or more names colliding with existing
|
|
168
|
+
columns, we let pandas raise -- that case is rare and signals
|
|
169
|
+
a real upstream issue.
|
|
170
|
+
"""
|
|
171
|
+
idx = df.index
|
|
172
|
+
if idx.name is None and not isinstance(idx, pd.MultiIndex):
|
|
173
|
+
# Anonymous single-level index -- nothing to surface.
|
|
174
|
+
return df
|
|
175
|
+
if isinstance(idx, pd.MultiIndex):
|
|
176
|
+
return df.reset_index()
|
|
177
|
+
if idx.name in df.columns:
|
|
178
|
+
# Column already carries the index values; reset would
|
|
179
|
+
# collide. Leave the frame as-is.
|
|
180
|
+
return df
|
|
181
|
+
return df.reset_index()
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _flatten_columns(columns: pd.Index) -> list[str]:
|
|
185
|
+
"""MultiIndex → list of ``"a.b.c"`` strings; plain Index → list of str."""
|
|
186
|
+
if isinstance(columns, pd.MultiIndex):
|
|
187
|
+
return [".".join(str(p) for p in tup) for tup in columns.values]
|
|
188
|
+
return [str(c) for c in columns]
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def frame_to_payload(
|
|
192
|
+
df: pd.DataFrame,
|
|
193
|
+
*,
|
|
194
|
+
cols: list[str] | None = None,
|
|
195
|
+
start: int | None = None,
|
|
196
|
+
stop: int | None = None,
|
|
197
|
+
downsample: int | None = None,
|
|
198
|
+
) -> dict:
|
|
199
|
+
"""Convert a DataFrame to the ``FrameResponse`` payload.
|
|
200
|
+
|
|
201
|
+
Parameters
|
|
202
|
+
----------
|
|
203
|
+
df : pandas.DataFrame
|
|
204
|
+
Source frame. The index is *not* included in the output
|
|
205
|
+
unless explicitly named -- callers can ``df.reset_index()``
|
|
206
|
+
first if they want it.
|
|
207
|
+
cols : list[str] | None
|
|
208
|
+
Subset of columns to return. Names absent from the frame are
|
|
209
|
+
silently dropped (callers can request a generic set and let
|
|
210
|
+
the api filter).
|
|
211
|
+
start, stop : int | None
|
|
212
|
+
Positional row slice (NOT label slice). ``None`` means
|
|
213
|
+
unbounded on that side.
|
|
214
|
+
downsample : int | None
|
|
215
|
+
If set, return at most this many rows -- evenly spaced
|
|
216
|
+
across whatever slice survived the start/stop. Used by
|
|
217
|
+
the SPA to render a 2**16-row density_df at sensible
|
|
218
|
+
display resolution.
|
|
219
|
+
"""
|
|
220
|
+
if cols:
|
|
221
|
+
# Tolerate caller passing column names that aren't on this
|
|
222
|
+
# frame -- the SPA might ask for "exeqa_total" on an
|
|
223
|
+
# Aggregate (no such column) without failing the round-trip.
|
|
224
|
+
existing = [c for c in cols if c in df.columns]
|
|
225
|
+
df = df[existing]
|
|
226
|
+
|
|
227
|
+
# Positional slicing first, then downsample. Downsampling
|
|
228
|
+
# *after* slicing means a request like ``start=0, stop=1000,
|
|
229
|
+
# downsample=100`` returns 100 rows from the first 1000, not
|
|
230
|
+
# 100 rows spread across the whole 2**N grid.
|
|
231
|
+
sliced = df.iloc[slice(start, stop)]
|
|
232
|
+
|
|
233
|
+
if downsample is not None and len(sliced) > downsample > 0:
|
|
234
|
+
# Even-spaced index sampling. linspace + round + unique
|
|
235
|
+
# avoids duplicate row picks when downsample is close to
|
|
236
|
+
# len(sliced).
|
|
237
|
+
idx = np.unique(
|
|
238
|
+
np.round(np.linspace(0, len(sliced) - 1, downsample)).astype(int)
|
|
239
|
+
)
|
|
240
|
+
sliced = sliced.iloc[idx]
|
|
241
|
+
|
|
242
|
+
columns = _flatten_columns(sliced.columns)
|
|
243
|
+
# ``.to_numpy()`` is faster than .values and preserves dtype.
|
|
244
|
+
rows = [
|
|
245
|
+
[_safe(v) for v in row]
|
|
246
|
+
for row in sliced.to_numpy().tolist()
|
|
247
|
+
]
|
|
248
|
+
return {"columns": columns, "rows": rows}
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
# Default display resolution for the binned density: 2**13 = 8192 rows. A
|
|
252
|
+
# density built at log2 = N is reduced to this many grid-aligned super-buckets,
|
|
253
|
+
# i.e. shown as if built at a coarser ``bs`` while the fine build is kept.
|
|
254
|
+
#
|
|
255
|
+
# Why 13 and not the 11 this shipped with: a discretized aggregate is routinely
|
|
256
|
+
# **atomic**, not merely spiky. A book written over layer limits, or with an
|
|
257
|
+
# occurrence cession, puts point masses in the severity and the aggregate
|
|
258
|
+
# inherits them at every multiple. Measured on one such program (limits
|
|
259
|
+
# ``250 500 1000 2000 xs 0``, ``750 xs 750`` occurrence cession, ``log2=16``,
|
|
260
|
+
# ``bs=1``): single buckets at 0 / 250 / 500 / 750 hold 8.6% / 12.9% / 10.0% /
|
|
261
|
+
# 5.7% of the mass against a continuum of 0.07% per bucket. Binning 32 fine
|
|
262
|
+
# buckets into one (which 2**11 does at ``log2=16``) merged each atom with 31
|
|
263
|
+
# neighbours and located it only to within half a super-bucket, so the plot drew
|
|
264
|
+
# a triangle 64 loss units wide where the truth is a spine one unit wide.
|
|
265
|
+
DENSITY_DISPLAY_LOG2 = 13
|
|
266
|
+
|
|
267
|
+
# Cell budget for one density payload, roughly 2**16 numbers. The row target
|
|
268
|
+
# above is right for the four-column ``loss / p_total / F / S`` case; a
|
|
269
|
+
# Portfolio's per-unit frame is 2 * units + 3 columns wide and would ship several
|
|
270
|
+
# megabytes of JSON at the same row count. Trading rows for columns keeps the
|
|
271
|
+
# payload flat instead of scaling with the unit count.
|
|
272
|
+
DENSITY_DISPLAY_CELLS = 1 << 16
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def display_log2_for(n_cols: int, cap: int = DENSITY_DISPLAY_LOG2) -> int:
|
|
276
|
+
"""Display ``log2`` for a frame ``n_cols`` wide, under the cell budget.
|
|
277
|
+
|
|
278
|
+
Parameters
|
|
279
|
+
----------
|
|
280
|
+
n_cols : int
|
|
281
|
+
Number of columns the payload will carry.
|
|
282
|
+
cap : int, optional
|
|
283
|
+
Upper bound, :data:`DENSITY_DISPLAY_LOG2` by default.
|
|
284
|
+
|
|
285
|
+
Returns
|
|
286
|
+
-------
|
|
287
|
+
int
|
|
288
|
+
``log2`` of the row target: ``cap`` for a narrow frame, reduced by whole
|
|
289
|
+
powers of two until ``rows * n_cols`` fits :data:`DENSITY_DISPLAY_CELLS`.
|
|
290
|
+
Never below 11, which is the resolution this shipped with and the floor
|
|
291
|
+
at which a density is still worth drawing.
|
|
292
|
+
|
|
293
|
+
Examples
|
|
294
|
+
--------
|
|
295
|
+
Four columns keep the full grid; an eleven-column portfolio frame steps down
|
|
296
|
+
one notch.
|
|
297
|
+
|
|
298
|
+
>>> display_log2_for(4)
|
|
299
|
+
13
|
|
300
|
+
>>> display_log2_for(11)
|
|
301
|
+
12
|
|
302
|
+
"""
|
|
303
|
+
rows = DENSITY_DISPLAY_CELLS // max(1, int(n_cols))
|
|
304
|
+
return max(11, min(cap, rows.bit_length() - 1))
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def bin_density(
|
|
308
|
+
df: pd.DataFrame,
|
|
309
|
+
source_log2: int,
|
|
310
|
+
*,
|
|
311
|
+
sum_cols: set[str],
|
|
312
|
+
label_col: str = "loss",
|
|
313
|
+
display_log2: int = DENSITY_DISPLAY_LOG2,
|
|
314
|
+
) -> pd.DataFrame:
|
|
315
|
+
"""Aggregate a density frame onto a coarser power-of-two display grid.
|
|
316
|
+
|
|
317
|
+
The density grid has ``2**source_log2`` rows. We reduce it to exactly
|
|
318
|
+
``2**display_log2`` rows binned by a factor ``k = 2**j``
|
|
319
|
+
(``j = source_log2 - display_log2``) -- the density "as if built at a
|
|
320
|
+
coarser ``bs' = k * bs``", with no loss of severity detail in the fine
|
|
321
|
+
build. This keeps the power-of-two paradigm.
|
|
322
|
+
|
|
323
|
+
Buckets are **centered on the coarse grid nodes** (the "around x_i"
|
|
324
|
+
convention). Node ``i`` sits at ``loss = i * bs'`` (a clean multiple:
|
|
325
|
+
0, bs', 2*bs', …) and owns the fine buckets in the half-open window
|
|
326
|
+
``(i*bs' - bs'/2, i*bs' + bs'/2]``. So with ``bs' = 320`` the first row is
|
|
327
|
+
labeled ``0`` and covers ``loss <= 160``; the second is labeled ``320`` and
|
|
328
|
+
covers ``160 < loss <= 480``; and so on. The first bucket is a left
|
|
329
|
+
half-window (nothing below 0) and the final node absorbs the short tail, so
|
|
330
|
+
the partition is exact and the row count stays ``2**display_log2``.
|
|
331
|
+
|
|
332
|
+
Column reductions:
|
|
333
|
+
|
|
334
|
+
* **mass columns** (``sum_cols`` -- ``p_total`` / ``p_sev`` / any ``p_*``)
|
|
335
|
+
are **summed** over the window (probability-conserving);
|
|
336
|
+
* the **label column** (``label_col``, default ``loss``) takes the node
|
|
337
|
+
center ``i * bs'`` -- the clean coarse-grid label;
|
|
338
|
+
* **every other column** (``F`` / ``S`` / the ``ex***`` series) takes the
|
|
339
|
+
window's **right edge** (``last``). For ``F`` / ``S`` that makes the
|
|
340
|
+
surfaced value the running cumulative through the bucket, so
|
|
341
|
+
``F[i] - F[i-1]`` equals the summed mass ``p_total[i]`` -- the same
|
|
342
|
+
convention a native coarse build uses (``F`` = cumsum of the masses).
|
|
343
|
+
|
|
344
|
+
Parameters
|
|
345
|
+
----------
|
|
346
|
+
df : pandas.DataFrame
|
|
347
|
+
Density frame (already index-reset, ``loss`` a column).
|
|
348
|
+
source_log2 : int
|
|
349
|
+
``log2`` of the underlying build grid.
|
|
350
|
+
sum_cols : set[str]
|
|
351
|
+
Column names whose values are masses and should be summed.
|
|
352
|
+
label_col : str, default ``"loss"``
|
|
353
|
+
The coarse-grid label column, sampled at each node center. Absent from
|
|
354
|
+
the frame is fine (then no column is treated as the label).
|
|
355
|
+
display_log2 : int, default ``DENSITY_DISPLAY_LOG2``
|
|
356
|
+
Target ``log2`` of the displayed grid. ``source_log2 <= display_log2``
|
|
357
|
+
means no binning (``k == 1``), the frame is returned unchanged. Callers
|
|
358
|
+
serving a wide frame should pass :func:`display_log2_for` rather than the
|
|
359
|
+
bare default, so the payload stays inside the cell budget.
|
|
360
|
+
|
|
361
|
+
Returns
|
|
362
|
+
-------
|
|
363
|
+
pandas.DataFrame
|
|
364
|
+
The binned frame, column order preserved, with a fresh 0..2**m index.
|
|
365
|
+
|
|
366
|
+
Notes
|
|
367
|
+
-----
|
|
368
|
+
Grouping is positional, not value-based, so it does not depend on the index
|
|
369
|
+
being clean or monotone. ``(arange(n) + k//2 - 1) // k`` (clamped to the last
|
|
370
|
+
node) assigns each fine index to the nearest node under the upper-inclusive
|
|
371
|
+
``(node - bs'/2, node + bs'/2]`` rule; the node centers themselves are the
|
|
372
|
+
fine rows ``0, k, 2k, …``.
|
|
373
|
+
"""
|
|
374
|
+
j = max(0, int(source_log2) - int(display_log2))
|
|
375
|
+
k = 1 << j
|
|
376
|
+
if k == 1:
|
|
377
|
+
return df
|
|
378
|
+
n = len(df)
|
|
379
|
+
half = k // 2
|
|
380
|
+
num_groups = (n - 1) // k + 1
|
|
381
|
+
# Nearest-node assignment under the (node - bs'/2, node + bs'/2] rule, with
|
|
382
|
+
# the short tail folded into the final node so the partition is exact.
|
|
383
|
+
groups = np.minimum((np.arange(n) + half - 1) // k, num_groups - 1)
|
|
384
|
+
# Node centers are the fine rows 0, k, 2k, … (capped at the last row).
|
|
385
|
+
center_idx = np.minimum(np.arange(num_groups) * k, n - 1)
|
|
386
|
+
|
|
387
|
+
cols = list(df.columns)
|
|
388
|
+
out: dict[str, np.ndarray] = {}
|
|
389
|
+
for col in cols:
|
|
390
|
+
series = df[col]
|
|
391
|
+
if col in sum_cols:
|
|
392
|
+
out[col] = series.groupby(groups, sort=True).sum().to_numpy()
|
|
393
|
+
elif col == label_col:
|
|
394
|
+
out[col] = series.to_numpy()[center_idx]
|
|
395
|
+
else:
|
|
396
|
+
# Window right edge -> F/S read as the cumulative through the bucket.
|
|
397
|
+
out[col] = series.groupby(groups, sort=True).last().to_numpy()
|
|
398
|
+
return pd.DataFrame(out, columns=cols)
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def pnl_density_frame(obj: Any) -> pd.DataFrame:
|
|
402
|
+
"""Synthesize a ``loss / p_total / F / S`` density frame from a PnL result.
|
|
403
|
+
|
|
404
|
+
A :class:`PnL` object's ``density_df`` is an ``OrderedDict`` of per-leg
|
|
405
|
+
``GridDistribution``s, not a single DataFrame, so it can't flow through the
|
|
406
|
+
generic density path. Its ``result`` attribute is the grand-result
|
|
407
|
+
``GridDistribution`` (the consolidated P&L outcome), carrying the outcome
|
|
408
|
+
grid ``x`` and its probability masses ``p``. We surface it in the same
|
|
409
|
+
``loss / p_total / F / S`` shape the SPA density plot expects for an
|
|
410
|
+
aggregate, so the Overview and Density views render unchanged.
|
|
411
|
+
|
|
412
|
+
Parameters
|
|
413
|
+
----------
|
|
414
|
+
obj : Any
|
|
415
|
+
A built ``PnL`` object exposing ``result.x`` / ``result.p``.
|
|
416
|
+
|
|
417
|
+
Returns
|
|
418
|
+
-------
|
|
419
|
+
pandas.DataFrame
|
|
420
|
+
Columns ``loss`` (outcome), ``p_total`` (mass), ``F`` (cdf), ``S``
|
|
421
|
+
(survival), one row per grid node, ascending in ``loss``.
|
|
422
|
+
|
|
423
|
+
Notes
|
|
424
|
+
-----
|
|
425
|
+
The P&L outcome axis is *signed* -- losses are negative -- unlike an
|
|
426
|
+
aggregate's non-negative loss grid. The downstream reduction
|
|
427
|
+
(:func:`bin_density`) is positional and samples the real ``loss`` value at
|
|
428
|
+
each node, so it makes no non-negativity assumption and bins the signed grid
|
|
429
|
+
faithfully.
|
|
430
|
+
"""
|
|
431
|
+
gd = obj.result
|
|
432
|
+
x = np.asarray(gd.x, dtype=float)
|
|
433
|
+
p = np.asarray(gd.p, dtype=float)
|
|
434
|
+
cdf = np.cumsum(p)
|
|
435
|
+
return pd.DataFrame({"loss": x, "p_total": p, "F": cdf, "S": 1.0 - cdf})
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def severity_density_frame(obj: Any, n: int = 512) -> pd.DataFrame:
|
|
439
|
+
"""Synthesize a ``loss / pdf / F / S`` curve from a frozen severity.
|
|
440
|
+
|
|
441
|
+
A :class:`Severity` is a look-through onto a frozen scipy random variable,
|
|
442
|
+
not a compute result, so it carries no ``density_df``: upstream lists it in
|
|
443
|
+
``NEAR_FIRST_CLASS`` and exempts it from the DataFrame quartet for exactly
|
|
444
|
+
that reason. The api still needs *something* to draw, so it samples the
|
|
445
|
+
frozen variable here, the same presentation-layer move
|
|
446
|
+
:func:`pnl_density_frame` makes for a ``PnL``.
|
|
447
|
+
|
|
448
|
+
Parameters
|
|
449
|
+
----------
|
|
450
|
+
obj : Any
|
|
451
|
+
A built ``Severity`` exposing the scipy surface (``isf`` / ``pdf`` /
|
|
452
|
+
``cdf`` / ``sf``).
|
|
453
|
+
n : int, optional
|
|
454
|
+
Number of grid points. 512 is smooth at any plot width and trivial to
|
|
455
|
+
serialize.
|
|
456
|
+
|
|
457
|
+
Returns
|
|
458
|
+
-------
|
|
459
|
+
pandas.DataFrame
|
|
460
|
+
Columns ``loss``, ``pdf``, ``F``, ``S``, one row per grid point,
|
|
461
|
+
ascending in ``loss``.
|
|
462
|
+
|
|
463
|
+
Notes
|
|
464
|
+
-----
|
|
465
|
+
The column is ``pdf``, **not** ``p_total``, and the distinction is load
|
|
466
|
+
bearing: an aggregate's ``p_total`` is a probability *mass* per bucket that
|
|
467
|
+
sums to one, while this is a density *ordinate* that does not. Reusing the
|
|
468
|
+
aggregate's column name would invite summing a column that has no business
|
|
469
|
+
being summed.
|
|
470
|
+
|
|
471
|
+
The grid is built by inverting the survival function over log-spaced
|
|
472
|
+
exceedance probabilities rather than by walking loss linearly. A severity is
|
|
473
|
+
routinely heavy-tailed and its support often unbounded, so a linear grid
|
|
474
|
+
either truncates the tail or wastes nearly every point on it. Quantile
|
|
475
|
+
spacing puts points where the probability is.
|
|
476
|
+
"""
|
|
477
|
+
# Log-spaced exceedance probabilities: dense near the median, and still
|
|
478
|
+
# resolving the 1-in-100,000 tail without a huge grid.
|
|
479
|
+
ps = np.concatenate([
|
|
480
|
+
np.logspace(np.log10(1 - 1e-5), np.log10(0.5), n // 2, endpoint=False),
|
|
481
|
+
np.logspace(np.log10(0.5), np.log10(1e-5), n - n // 2),
|
|
482
|
+
])
|
|
483
|
+
loss = np.asarray(obj.isf(ps), dtype=float)
|
|
484
|
+
# A bounded or discrete severity can repeat or invert; keep it monotone and
|
|
485
|
+
# finite so the client never has to defend against a bad axis.
|
|
486
|
+
ok = np.isfinite(loss)
|
|
487
|
+
loss = np.unique(loss[ok])
|
|
488
|
+
with np.errstate(divide="ignore", invalid="ignore"):
|
|
489
|
+
pdf = np.asarray(obj.pdf(loss), dtype=float)
|
|
490
|
+
cdf = np.asarray(obj.cdf(loss), dtype=float)
|
|
491
|
+
sf = np.asarray(obj.sf(loss), dtype=float)
|
|
492
|
+
return pd.DataFrame({"loss": loss, "pdf": pdf, "F": cdf, "S": sf})
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def bivariate_marginal_frame(obj: Any) -> pd.DataFrame:
|
|
496
|
+
"""The two component marginals of a bivariate, as one long frame.
|
|
497
|
+
|
|
498
|
+
A :class:`BivariateAggregate`'s ``density_df`` is the **joint** matrix: the
|
|
499
|
+
axis-0 grid as the index and the axis-1 grid as the columns, so 2**16 cells
|
|
500
|
+
or more. That is a picture, not a table, and serving it to a grid produces
|
|
501
|
+
something no reader can use and a payload nobody wants. The marginals are
|
|
502
|
+
what a table of a bivariate should say.
|
|
503
|
+
|
|
504
|
+
Parameters
|
|
505
|
+
----------
|
|
506
|
+
obj : Any
|
|
507
|
+
A built ``BivariateAggregate`` exposing ``marginals`` (the exact pass-3
|
|
508
|
+
fold, precomputed), ``axis_xs`` and ``unit_names``.
|
|
509
|
+
|
|
510
|
+
Returns
|
|
511
|
+
-------
|
|
512
|
+
pandas.DataFrame
|
|
513
|
+
Columns ``unit``, ``loss``, ``p``, ``F``, ``S``; the two components
|
|
514
|
+
stacked, each ascending in ``loss``.
|
|
515
|
+
|
|
516
|
+
Notes
|
|
517
|
+
-----
|
|
518
|
+
Long rather than wide because the two axes have **different grids** (and
|
|
519
|
+
routinely different lengths: 2048 and 512 on one measured build). Aligning
|
|
520
|
+
them side by side would mean padding one with nulls and inviting a reader to
|
|
521
|
+
compare row `i` of one against row `i` of the other, which means nothing.
|
|
522
|
+
``unit`` is a real column, so the grid's own filter narrows to one component.
|
|
523
|
+
|
|
524
|
+
``marginals`` returns plain arrays that each sum to 1, so ``F`` is their
|
|
525
|
+
cumulative sum and ``S`` its complement, exactly as for an aggregate.
|
|
526
|
+
"""
|
|
527
|
+
marginals = obj.marginals
|
|
528
|
+
grids = obj.axis_xs
|
|
529
|
+
names = list(getattr(obj, "unit_names", None) or ["axis 0", "axis 1"])
|
|
530
|
+
|
|
531
|
+
parts = []
|
|
532
|
+
for i, (mass, loss) in enumerate(zip(marginals, grids)):
|
|
533
|
+
mass = np.asarray(mass, dtype=float)
|
|
534
|
+
loss = np.asarray(loss, dtype=float)
|
|
535
|
+
cdf = np.cumsum(mass)
|
|
536
|
+
parts.append(pd.DataFrame({
|
|
537
|
+
"unit": names[i] if i < len(names) else f"axis {i}",
|
|
538
|
+
"loss": loss,
|
|
539
|
+
"p": mass,
|
|
540
|
+
"F": cdf,
|
|
541
|
+
# Clamped: accumulating thousands of floats to 1 overshoots by a few
|
|
542
|
+
# parts in 1e15, and a negative survival is not a thing to serve.
|
|
543
|
+
"S": np.maximum(0.0, 1.0 - cdf),
|
|
544
|
+
}))
|
|
545
|
+
return pd.concat(parts, ignore_index=True)
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
def info_to_payload(obj: Any) -> dict:
|
|
549
|
+
"""Return ``{"info": "..."}``.
|
|
550
|
+
|
|
551
|
+
:attr:`Aggregate.info` and :attr:`Portfolio.info` are
|
|
552
|
+
multi-line strings (formatted summaries). We expose them
|
|
553
|
+
verbatim; clients display in a monospaced block.
|
|
554
|
+
|
|
555
|
+
A ``PnL`` object has no ``info`` string; it carries the equivalent narrative
|
|
556
|
+
on ``construction_explanation``. We fall back to that so the Info tab isn't
|
|
557
|
+
blank for a P&L build. Both accesses are getattr-gated, so an object kind
|
|
558
|
+
with neither simply reports an empty string.
|
|
559
|
+
"""
|
|
560
|
+
info = getattr(obj, "info", "") or getattr(obj, "construction_explanation", "")
|
|
561
|
+
if not isinstance(info, str):
|
|
562
|
+
# Fallback for objects that override .info as something
|
|
563
|
+
# else -- str() coerces to a usable rendering.
|
|
564
|
+
info = str(info)
|
|
565
|
+
return {"info": info}
|