scelo 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scelo/__init__.py +47 -0
- scelo/_alias.py +121 -0
- scelo/_audit.py +137 -0
- scelo/_table.py +148 -0
- scelo/_version.py +7 -0
- scelo/accessor.py +59 -0
- scelo/clean.py +1303 -0
- scelo/cli.py +86 -0
- scelo/climate.py +104 -0
- scelo/combine.py +227 -0
- scelo/data/claims.csv +80 -0
- scelo/data/climate.csv +31 -0
- scelo/data/dirty.csv +54 -0
- scelo/data/lifelib-mp.csv +101 -0
- scelo/data/wmtr-scenarios.csv +13 -0
- scelo/data/workspace-demo.csv +2001 -0
- scelo/fairness.py +131 -0
- scelo/finance.py +407 -0
- scelo/hard.py +214 -0
- scelo/io.py +308 -0
- scelo/life.py +950 -0
- scelo/lifelib.py +191 -0
- scelo/pipelines.py +128 -0
- scelo/pricing.py +440 -0
- scelo/profile.py +406 -0
- scelo/reserving.py +617 -0
- scelo/risk.py +561 -0
- scelo/swarm.py +311 -0
- scelo/viz.py +425 -0
- scelo/wmtr.py +501 -0
- scelo/workspace.py +166 -0
- scelo-0.1.0.dist-info/METADATA +908 -0
- scelo-0.1.0.dist-info/RECORD +36 -0
- scelo-0.1.0.dist-info/WHEEL +4 -0
- scelo-0.1.0.dist-info/entry_points.txt +2 -0
- scelo-0.1.0.dist-info/licenses/LICENSE +662 -0
scelo/__init__.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""scelo: soft data → tools → hard data, for actuaries who write code.
|
|
2
|
+
|
|
3
|
+
import scelo as sc
|
|
4
|
+
df = sc.load("claims.csv") # soft: typed the way Scelo IDE types it
|
|
5
|
+
df = sc.clean(df) # the IDE's safe cleaning ops, audited
|
|
6
|
+
res = sc.reserve(df) # tools: chain ladder, Mack, BF, bootstrap
|
|
7
|
+
sc.report(res, to="pack.html") # hard: numbers that travel with their basis
|
|
8
|
+
|
|
9
|
+
Everything is one ``sc.`` away and every table-shaped result is a
|
|
10
|
+
:class:`Table`, a pandas DataFrame that carries its title, basis, notes and
|
|
11
|
+
provenance. ``sc.cheatsheet()`` prints the one-screen map.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from ._version import __version__
|
|
17
|
+
from ._table import Table, as_table, notes
|
|
18
|
+
from ._alias import COLUMN_ALIASES, find_column, infer
|
|
19
|
+
from ._audit import audit, clear_audit, enable_audit, content_hash
|
|
20
|
+
from .io import * # noqa: F401,F403
|
|
21
|
+
from .profile import * # noqa: F401,F403
|
|
22
|
+
from .clean import * # noqa: F401,F403
|
|
23
|
+
from .combine import * # noqa: F401,F403
|
|
24
|
+
from .life import * # noqa: F401,F403
|
|
25
|
+
from .reserving import * # noqa: F401,F403
|
|
26
|
+
from .finance import * # noqa: F401,F403
|
|
27
|
+
from .risk import * # noqa: F401,F403
|
|
28
|
+
from .pricing import * # noqa: F401,F403
|
|
29
|
+
from .fairness import * # noqa: F401,F403
|
|
30
|
+
from .climate import * # noqa: F401,F403
|
|
31
|
+
from .wmtr import * # noqa: F401,F403
|
|
32
|
+
from .swarm import * # noqa: F401,F403
|
|
33
|
+
from .workspace import * # noqa: F401,F403
|
|
34
|
+
from .lifelib import * # noqa: F401,F403
|
|
35
|
+
from .hard import * # noqa: F401,F403
|
|
36
|
+
from .pipelines import * # noqa: F401,F403
|
|
37
|
+
from .viz import * # noqa: F401,F403
|
|
38
|
+
import importlib as _importlib # noqa: E402
|
|
39
|
+
|
|
40
|
+
_importlib.import_module("scelo.accessor") # registers df.sc
|
|
41
|
+
|
|
42
|
+
_MODULES = ("io", "profile", "clean", "combine", "life", "reserving", "finance", "risk", "pricing", "fairness", "climate", "wmtr", "swarm",
|
|
43
|
+
"workspace", "lifelib", "hard", "pipelines", "viz")
|
|
44
|
+
__all__ = sorted(
|
|
45
|
+
{n for m in _MODULES for n in getattr(_importlib.import_module(f"scelo.{m}"), "__all__", [])}
|
|
46
|
+
| {"Table", "as_table", "notes", "COLUMN_ALIASES", "find_column", "infer", "audit", "clear_audit", "enable_audit", "content_hash", "__version__"}
|
|
47
|
+
)
|
scelo/_alias.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Column inference: the reason ``sc.triangle(df)`` needs no arguments.
|
|
2
|
+
|
|
3
|
+
Every tools function accepts explicit column names, but when you leave them
|
|
4
|
+
out it looks the columns up here: case- and punctuation-insensitive matches
|
|
5
|
+
against the alias lists Scelo IDE uses for its own table suggestions, so
|
|
6
|
+
``accident_year`` / ``AY`` / ``Origin Year`` all resolve to the origin column.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from typing import Dict, Iterable, List, Optional, Sequence
|
|
13
|
+
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
_NON_ALNUM = re.compile(r"[^a-z0-9]")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _lc(s: str) -> str:
|
|
20
|
+
return _NON_ALNUM.sub("", str(s).lower())
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
# Mirrors packages/scelo-core/src/actuarialTables.ts COLUMN_ALIASES, with the
|
|
24
|
+
# extra spellings that show up on real extracts. Order matters: the first alias
|
|
25
|
+
# that matches wins, so the canonical name leads each list.
|
|
26
|
+
COLUMN_ALIASES: Dict[str, List[str]] = {
|
|
27
|
+
"age": [
|
|
28
|
+
"age", "age_at_entry", "ageatentry", "issue_age", "attained_age", "x", "age_x",
|
|
29
|
+
"age_band", "ageband", "age_last", "age_nearest", "entry_age", "age_years",
|
|
30
|
+
],
|
|
31
|
+
"qx": ["qx", "q_x", "mortality", "mortality_rate", "death_rate", "q", "prob_death", "rate", "qx_ult"],
|
|
32
|
+
"mx": ["mx", "m_x", "central_rate", "hazard", "mu", "mu_x", "force_of_mortality"],
|
|
33
|
+
"lx": ["lx", "l_x", "lives", "survivors", "l"],
|
|
34
|
+
"deaths": ["deaths", "death", "d", "dx", "actual_deaths", "claims_count", "n_deaths", "died", "events", "actual"],
|
|
35
|
+
"exposure": [
|
|
36
|
+
"exposure", "exposures", "exposed", "exposed_to_risk", "etr", "lives_exposed", "person_years",
|
|
37
|
+
"policy_years", "central_exposure", "initial_exposure", "expo", "e", "ex", "time_at_risk",
|
|
38
|
+
],
|
|
39
|
+
"expected": ["expected", "expected_deaths", "exp_deaths", "e_deaths", "expected_claims"],
|
|
40
|
+
"origin": [
|
|
41
|
+
"origin", "origin_year", "accident_year", "accidentyear", "ay", "uw_year", "underwriting_year",
|
|
42
|
+
"occurrence_year", "loss_year", "year_of_origin", "cohort", "origin_period", "accident_period",
|
|
43
|
+
"acc_year", "accyear", "uwy", "policy_year",
|
|
44
|
+
],
|
|
45
|
+
"development": [
|
|
46
|
+
"development", "dev", "development_period", "dev_period", "development_year", "dev_year", "lag",
|
|
47
|
+
"delay", "age_months", "development_lag", "dev_lag", "period", "devyear", "development_months",
|
|
48
|
+
],
|
|
49
|
+
"payment": [
|
|
50
|
+
"payment_year", "calendar_year", "paid_year", "settlement_year", "transaction_year", "report_year",
|
|
51
|
+
"valuation_year", "cal_year", "calendar_period", "payment_period", "cy",
|
|
52
|
+
],
|
|
53
|
+
"value": [
|
|
54
|
+
"paid", "incurred", "paid_amount", "incurred_amount", "claims", "claim_amount", "amount", "loss",
|
|
55
|
+
"losses", "payments", "value", "cumulative", "reported", "paid_claims", "incurred_claims",
|
|
56
|
+
"claim", "cost", "severity",
|
|
57
|
+
],
|
|
58
|
+
"premium": [
|
|
59
|
+
"premium_pp", "premium", "annual_premium", "monthly_premium", "prem", "earned_premium",
|
|
60
|
+
"written_premium", "gwp", "gep", "premiums", "ep",
|
|
61
|
+
],
|
|
62
|
+
"tenor": ["tenor", "maturity", "term", "years", "year", "t", "maturity_years", "tenor_years"],
|
|
63
|
+
"rate": [
|
|
64
|
+
"rate", "zero_rate", "spot", "spot_rate", "yield", "zero", "swap_rate", "par_rate",
|
|
65
|
+
"interest_rate", "zero_coupon", "spot_yield", "r",
|
|
66
|
+
],
|
|
67
|
+
"sex": ["sex", "gender", "male_female", "m_f"],
|
|
68
|
+
"policy_term": ["policy_term", "policyterm", "term", "term_years", "policy_term_years", "duration_years"],
|
|
69
|
+
"sum_assured": ["sum_assured", "sumassured", "sa", "face_amount", "face", "benefit", "sum_insured", "si", "coverage"],
|
|
70
|
+
"count": ["count", "policy_count", "policycount", "n", "number", "claim_count", "frequency", "num_claims", "nclaims", "claims_count", "policies"],
|
|
71
|
+
"year": ["year", "calendar_year", "cal_year", "period", "yr"],
|
|
72
|
+
"date": ["date", "as_at", "valuation_date", "effective_date", "start_date", "issue_date"],
|
|
73
|
+
"duration": ["duration", "time", "t", "survival_time", "policy_duration", "tenure"],
|
|
74
|
+
"event": ["event", "status", "died", "death", "claimed", "lapsed", "censored"],
|
|
75
|
+
"group": ["group", "segment", "class", "cohort", "risk_class", "region", "band"],
|
|
76
|
+
"actual": ["actual", "observed", "actual_claims", "actual_deaths", "deaths", "claims"],
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def find_column(columns: Iterable[str], aliases: Sequence[str]) -> Optional[str]:
|
|
81
|
+
"""First column whose normalised name matches one of the aliases, else None."""
|
|
82
|
+
cols = list(columns)
|
|
83
|
+
table = {}
|
|
84
|
+
for c in cols:
|
|
85
|
+
table.setdefault(_lc(c), c)
|
|
86
|
+
for a in aliases:
|
|
87
|
+
hit = table.get(_lc(a))
|
|
88
|
+
if hit is not None:
|
|
89
|
+
return hit
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def infer(df: pd.DataFrame, role: str, explicit: Optional[str] = None, *, required: bool = True,
|
|
94
|
+
exclude: Iterable[str] = ()) -> Optional[str]:
|
|
95
|
+
"""Resolve a column for ``role`` ("age", "origin", …).
|
|
96
|
+
|
|
97
|
+
``explicit`` wins when given (and is validated). Otherwise the alias list
|
|
98
|
+
for the role is searched, skipping anything in ``exclude`` (so the same
|
|
99
|
+
column is not picked for two roles). Raises a ``KeyError`` naming the
|
|
100
|
+
role, the aliases tried and the columns available when ``required``.
|
|
101
|
+
"""
|
|
102
|
+
if explicit is not None:
|
|
103
|
+
if explicit not in df.columns:
|
|
104
|
+
raise KeyError(f'column "{explicit}" is not in the data (have: {", ".join(map(str, df.columns))})')
|
|
105
|
+
return explicit
|
|
106
|
+
aliases = COLUMN_ALIASES.get(role)
|
|
107
|
+
if aliases is None:
|
|
108
|
+
raise KeyError(f"unknown column role {role!r}")
|
|
109
|
+
ex = set(exclude)
|
|
110
|
+
hit = find_column([c for c in df.columns if c not in ex], aliases)
|
|
111
|
+
if hit is None and required:
|
|
112
|
+
raise KeyError(
|
|
113
|
+
f"could not infer the {role} column: pass {role}=<name>. "
|
|
114
|
+
f"Tried {', '.join(aliases[:6])}…; columns: {', '.join(map(str, df.columns))}"
|
|
115
|
+
)
|
|
116
|
+
return hit
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def numeric_columns(df: pd.DataFrame) -> List[str]:
|
|
120
|
+
"""Names of the numeric (non-boolean) columns."""
|
|
121
|
+
return [c for c in df.columns if pd.api.types.is_numeric_dtype(df[c]) and not pd.api.types.is_bool_dtype(df[c])]
|
scelo/_audit.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""The audit trail: what the tools layer did, in order.
|
|
2
|
+
|
|
3
|
+
Scelo's pipeline rule is that hard data never travels without the trail that
|
|
4
|
+
produced it. Every tools function records one entry here (function, the
|
|
5
|
+
arguments that matter, input/output shapes and content hashes, wall time).
|
|
6
|
+
``scelo.audit()`` returns the trail as a DataFrame; ``scelo.hard()`` copies
|
|
7
|
+
the relevant entries onto the table it stamps.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import functools
|
|
13
|
+
import hashlib
|
|
14
|
+
import inspect
|
|
15
|
+
import time
|
|
16
|
+
from datetime import datetime, timezone
|
|
17
|
+
from typing import Any, Callable, Dict, List, Optional
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import pandas as pd
|
|
21
|
+
|
|
22
|
+
from ._version import __version__
|
|
23
|
+
|
|
24
|
+
_TRAIL: List[Dict[str, Any]] = []
|
|
25
|
+
_ENABLED = True
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def content_hash(obj: Any) -> str:
|
|
29
|
+
"""sha256 of a DataFrame / Series / array / scalar's content (not its identity)."""
|
|
30
|
+
h = hashlib.sha256()
|
|
31
|
+
if isinstance(obj, pd.DataFrame):
|
|
32
|
+
h.update(",".join(map(str, obj.columns)).encode())
|
|
33
|
+
try:
|
|
34
|
+
h.update(pd.util.hash_pandas_object(obj, index=False).values.tobytes())
|
|
35
|
+
except TypeError: # unhashable cells (lists, dicts)
|
|
36
|
+
h.update(obj.to_csv(index=False).encode())
|
|
37
|
+
elif isinstance(obj, pd.Series):
|
|
38
|
+
try:
|
|
39
|
+
h.update(pd.util.hash_pandas_object(obj, index=False).values.tobytes())
|
|
40
|
+
except TypeError:
|
|
41
|
+
h.update(obj.to_csv(index=False).encode())
|
|
42
|
+
elif isinstance(obj, np.ndarray):
|
|
43
|
+
h.update(np.ascontiguousarray(obj).tobytes())
|
|
44
|
+
elif obj is None:
|
|
45
|
+
h.update(b"None")
|
|
46
|
+
else:
|
|
47
|
+
h.update(repr(obj).encode())
|
|
48
|
+
return h.hexdigest()
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _shape(obj: Any) -> Optional[str]:
|
|
52
|
+
if isinstance(obj, pd.DataFrame):
|
|
53
|
+
return f"{len(obj)}×{obj.shape[1]}"
|
|
54
|
+
if isinstance(obj, (pd.Series, np.ndarray, list, tuple)):
|
|
55
|
+
return f"{len(obj)}"
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _summarise_arg(v: Any) -> Any:
|
|
60
|
+
if isinstance(v, (pd.DataFrame, pd.Series, np.ndarray)):
|
|
61
|
+
return f"<{type(v).__name__} {_shape(v)} {content_hash(v)[:8]}>"
|
|
62
|
+
if isinstance(v, (int, float, str, bool)) or v is None:
|
|
63
|
+
return v
|
|
64
|
+
if isinstance(v, (list, tuple)) and len(v) <= 12:
|
|
65
|
+
return [_summarise_arg(x) for x in v]
|
|
66
|
+
if isinstance(v, dict) and len(v) <= 12:
|
|
67
|
+
return {k: _summarise_arg(x) for k, x in v.items()}
|
|
68
|
+
return f"<{type(v).__name__}>"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def record(fn: str, args: Dict[str, Any], inputs: Any, output: Any, ms: float, note: str = "") -> Dict[str, Any]:
|
|
72
|
+
entry = {
|
|
73
|
+
"at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
74
|
+
"fn": fn,
|
|
75
|
+
"args": {k: _summarise_arg(v) for k, v in args.items()},
|
|
76
|
+
"in": content_hash(inputs)[:16] if inputs is not None else None,
|
|
77
|
+
"in_shape": _shape(inputs),
|
|
78
|
+
"out": content_hash(output)[:16] if output is not None else None,
|
|
79
|
+
"out_shape": _shape(output),
|
|
80
|
+
"ms": round(ms, 2),
|
|
81
|
+
"scelo": __version__,
|
|
82
|
+
"note": note,
|
|
83
|
+
}
|
|
84
|
+
if _ENABLED:
|
|
85
|
+
_TRAIL.append(entry)
|
|
86
|
+
return entry
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def tool(fn: Callable) -> Callable:
|
|
90
|
+
"""Decorator: time the call, hash its first DataFrame-like input and its output, log it."""
|
|
91
|
+
sig = inspect.signature(fn)
|
|
92
|
+
|
|
93
|
+
@functools.wraps(fn)
|
|
94
|
+
def wrapper(*args: Any, **kwargs: Any):
|
|
95
|
+
t0 = time.perf_counter()
|
|
96
|
+
out = fn(*args, **kwargs)
|
|
97
|
+
ms = (time.perf_counter() - t0) * 1000
|
|
98
|
+
if _ENABLED:
|
|
99
|
+
try:
|
|
100
|
+
bound = sig.bind_partial(*args, **kwargs)
|
|
101
|
+
argmap = dict(bound.arguments)
|
|
102
|
+
except TypeError:
|
|
103
|
+
argmap = {"args": args, "kwargs": kwargs}
|
|
104
|
+
first = next((v for v in argmap.values() if isinstance(v, (pd.DataFrame, pd.Series, np.ndarray))), None)
|
|
105
|
+
out_for_hash = out
|
|
106
|
+
if isinstance(out, tuple) and out and isinstance(out[0], (pd.DataFrame, pd.Series)):
|
|
107
|
+
out_for_hash = out[0]
|
|
108
|
+
elif not isinstance(out, (pd.DataFrame, pd.Series, np.ndarray)):
|
|
109
|
+
out_for_hash = getattr(out, "table", None)
|
|
110
|
+
record(fn.__name__, argmap, first, out_for_hash, ms)
|
|
111
|
+
return out
|
|
112
|
+
|
|
113
|
+
wrapper.__scelo_tool__ = True # type: ignore[attr-defined]
|
|
114
|
+
return wrapper
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def audit(last: Optional[int] = None) -> pd.DataFrame:
|
|
118
|
+
"""The audit trail as a DataFrame (most recent last). ``last=n`` for the tail."""
|
|
119
|
+
rows = _TRAIL[-last:] if last else list(_TRAIL)
|
|
120
|
+
if not rows:
|
|
121
|
+
return pd.DataFrame(columns=["at", "fn", "args", "in", "in_shape", "out", "out_shape", "ms", "scelo", "note"])
|
|
122
|
+
return pd.DataFrame(rows)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def clear_audit() -> None:
|
|
126
|
+
"""Forget the trail (a new session / a new deliverable)."""
|
|
127
|
+
_TRAIL.clear()
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def enable_audit(on: bool = True) -> None:
|
|
131
|
+
"""Switch recording on/off (it is on by default; off saves a little time in tight loops)."""
|
|
132
|
+
global _ENABLED
|
|
133
|
+
_ENABLED = bool(on)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def entries() -> List[Dict[str, Any]]:
|
|
137
|
+
return list(_TRAIL)
|
scelo/_table.py
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""Table: a pandas DataFrame that carries its own caveats.
|
|
2
|
+
|
|
3
|
+
Every Scelo output that is table-shaped is a :class:`Table`: a real
|
|
4
|
+
``pandas.DataFrame`` (slice it, merge it, plot it, ``.to_csv`` it) with four
|
|
5
|
+
extra attributes that survive most pandas operations:
|
|
6
|
+
|
|
7
|
+
``title`` one line naming the table,
|
|
8
|
+
``notes`` the things an actuary should know before trusting it,
|
|
9
|
+
``basis`` a one-line provenance label ("Gompertz–Makeham (illustrative) · i = 4 %"),
|
|
10
|
+
``provenance`` a dict stamped by :func:`scelo.hard`: content hash, time,
|
|
11
|
+
scelo version, the audit entries that produced it.
|
|
12
|
+
|
|
13
|
+
Printing a Table prints the frame and then its notes, so the caveat is on
|
|
14
|
+
screen at the moment the number is, the one-way pipeline rule in practice:
|
|
15
|
+
a number never travels without its basis.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from typing import Any, Dict, List, Optional
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
_NOTE_PREFIX = " · "
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Table(pd.DataFrame):
|
|
28
|
+
"""A DataFrame with ``title``, ``notes``, ``basis`` and ``provenance``."""
|
|
29
|
+
|
|
30
|
+
_metadata = ["title", "notes", "basis", "provenance", "stage"]
|
|
31
|
+
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
data=None,
|
|
35
|
+
*args: Any,
|
|
36
|
+
title: Optional[str] = None,
|
|
37
|
+
notes: Optional[List[str]] = None,
|
|
38
|
+
basis: Optional[str] = None,
|
|
39
|
+
provenance: Optional[Dict[str, Any]] = None,
|
|
40
|
+
stage: Optional[str] = None,
|
|
41
|
+
**kwargs: Any,
|
|
42
|
+
) -> None:
|
|
43
|
+
super().__init__(data, *args, **kwargs)
|
|
44
|
+
# Preserve metadata when constructed from another Table.
|
|
45
|
+
src = data if isinstance(data, Table) else None
|
|
46
|
+
has = src is not None
|
|
47
|
+
object.__setattr__(self, "title", title if title is not None else (src.title if has else None))
|
|
48
|
+
object.__setattr__(self, "notes", list(notes) if notes is not None else (list(src.notes) if has else []))
|
|
49
|
+
object.__setattr__(self, "basis", basis if basis is not None else (src.basis if has else None))
|
|
50
|
+
object.__setattr__(
|
|
51
|
+
self, "provenance", dict(provenance) if provenance is not None else (dict(src.provenance) if has else {})
|
|
52
|
+
)
|
|
53
|
+
object.__setattr__(self, "stage", stage if stage is not None else (src.stage if has else "soft"))
|
|
54
|
+
|
|
55
|
+
# pandas subclassing contract ------------------------------------------------
|
|
56
|
+
@property
|
|
57
|
+
def _constructor(self):
|
|
58
|
+
return Table
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def _constructor_sliced(self):
|
|
62
|
+
return pd.Series
|
|
63
|
+
|
|
64
|
+
# presentation ----------------------------------------------------------------
|
|
65
|
+
def _footer(self) -> str:
|
|
66
|
+
lines: List[str] = []
|
|
67
|
+
if self.title:
|
|
68
|
+
lines.append(f"— {self.title}")
|
|
69
|
+
if self.basis:
|
|
70
|
+
lines.append(f" basis: {self.basis}")
|
|
71
|
+
for n in self.notes or []:
|
|
72
|
+
lines.append(_NOTE_PREFIX + n)
|
|
73
|
+
if self.provenance:
|
|
74
|
+
h = self.provenance.get("sha256", "")
|
|
75
|
+
if h:
|
|
76
|
+
lines.append(f" hard · {h[:12]} · scelo {self.provenance.get('scelo', '?')} · {self.provenance.get('at', '')}")
|
|
77
|
+
return "\n".join(lines)
|
|
78
|
+
|
|
79
|
+
def __repr__(self) -> str: # pragma: no cover - presentation
|
|
80
|
+
body = super().__repr__()
|
|
81
|
+
foot = self._footer()
|
|
82
|
+
return body + ("\n" + foot if foot else "")
|
|
83
|
+
|
|
84
|
+
def _repr_html_(self) -> Optional[str]: # pragma: no cover - presentation
|
|
85
|
+
body = super()._repr_html_()
|
|
86
|
+
foot = self._footer()
|
|
87
|
+
if not foot or body is None:
|
|
88
|
+
return body
|
|
89
|
+
esc = foot.replace("&", "&").replace("<", "<").replace(">", ">")
|
|
90
|
+
return body + f"<pre style='font-size:0.85em;opacity:0.8'>{esc}</pre>"
|
|
91
|
+
|
|
92
|
+
# convenience -------------------------------------------------------------------
|
|
93
|
+
def note(self, text: str) -> "Table":
|
|
94
|
+
"""Append a note (returns self, for chaining)."""
|
|
95
|
+
self.notes.append(text)
|
|
96
|
+
return self
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def df(self) -> pd.DataFrame:
|
|
100
|
+
"""A plain ``pandas.DataFrame`` copy (drops the Scelo metadata)."""
|
|
101
|
+
return pd.DataFrame(self)
|
|
102
|
+
|
|
103
|
+
def to_markdown_report(self) -> str:
|
|
104
|
+
"""Title, basis, the table (markdown) and the notes, one block."""
|
|
105
|
+
out: List[str] = []
|
|
106
|
+
if self.title:
|
|
107
|
+
out.append(f"### {self.title}")
|
|
108
|
+
if self.basis:
|
|
109
|
+
out.append(f"*Basis:* {self.basis}")
|
|
110
|
+
out.append("")
|
|
111
|
+
try:
|
|
112
|
+
out.append(pd.DataFrame(self).to_markdown(index=False))
|
|
113
|
+
except ImportError: # tabulate not installed
|
|
114
|
+
out.append("```\n" + pd.DataFrame(self).to_string(index=False) + "\n```")
|
|
115
|
+
if self.notes:
|
|
116
|
+
out.append("")
|
|
117
|
+
out.extend(f"- {n}" for n in self.notes)
|
|
118
|
+
if self.provenance:
|
|
119
|
+
out.append("")
|
|
120
|
+
out.append(
|
|
121
|
+
f"<sub>hard · sha256 {self.provenance.get('sha256', '')[:16]} · "
|
|
122
|
+
f"scelo {self.provenance.get('scelo', '')} · {self.provenance.get('at', '')}</sub>"
|
|
123
|
+
)
|
|
124
|
+
return "\n".join(out)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def as_table(
|
|
128
|
+
df: pd.DataFrame,
|
|
129
|
+
title: Optional[str] = None,
|
|
130
|
+
notes: Optional[List[str]] = None,
|
|
131
|
+
basis: Optional[str] = None,
|
|
132
|
+
stage: str = "hard",
|
|
133
|
+
) -> Table:
|
|
134
|
+
"""Wrap a DataFrame as a :class:`Table` (copying metadata if it already is one)."""
|
|
135
|
+
t = Table(df)
|
|
136
|
+
if title is not None:
|
|
137
|
+
object.__setattr__(t, "title", title)
|
|
138
|
+
if notes is not None:
|
|
139
|
+
object.__setattr__(t, "notes", list(notes))
|
|
140
|
+
if basis is not None:
|
|
141
|
+
object.__setattr__(t, "basis", basis)
|
|
142
|
+
object.__setattr__(t, "stage", stage)
|
|
143
|
+
return t
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def notes(x: Any) -> List[str]:
|
|
147
|
+
"""The notes attached to a Scelo result (empty list for plain objects)."""
|
|
148
|
+
return list(getattr(x, "notes", []) or [])
|
scelo/_version.py
ADDED
scelo/accessor.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""``df.sc``: the pandas accessor, for people who chain.
|
|
2
|
+
|
|
3
|
+
df.sc.profile() df.sc.clean("all") df.sc.triangle().sc.mack()
|
|
4
|
+
|
|
5
|
+
Every Scelo function that takes a frame first is available as a method;
|
|
6
|
+
the frame is passed as the first argument.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
|
|
15
|
+
import importlib
|
|
16
|
+
|
|
17
|
+
_clean = importlib.import_module("scelo.clean")
|
|
18
|
+
_combine = importlib.import_module("scelo.combine")
|
|
19
|
+
_fairness = importlib.import_module("scelo.fairness")
|
|
20
|
+
_life = importlib.import_module("scelo.life")
|
|
21
|
+
_pricing = importlib.import_module("scelo.pricing")
|
|
22
|
+
_profile = importlib.import_module("scelo.profile")
|
|
23
|
+
_reserving = importlib.import_module("scelo.reserving")
|
|
24
|
+
_workspace = importlib.import_module("scelo.workspace")
|
|
25
|
+
_hard = importlib.import_module("scelo.hard").hard
|
|
26
|
+
|
|
27
|
+
_METHODS = {
|
|
28
|
+
"profile": _profile.profile, "describe": _profile.describe, "types": _profile.types, "missing": _profile.missing, "tab": _profile.tab,
|
|
29
|
+
"corr": _profile.corr, "outliers": _profile.outliers, "inliers": _profile.inliers,
|
|
30
|
+
"suggest": _clean.suggest, "clean": _clean.clean, "dedupe": _clean.dedupe, "snake_names": _clean.snake_names, "impute": _clean.impute,
|
|
31
|
+
"cap_outliers": _clean.cap_outliers, "parse_dates": _clean.parse_dates, "parse_numbers": _clean.parse_numbers,
|
|
32
|
+
"triangle": _reserving.triangle, "chain_ladder": _reserving.chain_ladder, "mack": _reserving.mack, "bf": _reserving.bf,
|
|
33
|
+
"bootstrap": _reserving.bootstrap, "reserve": _reserving.reserve, "ldf": _reserving.ldf, "cdf": _reserving.cdf, "ata": _reserving.ata,
|
|
34
|
+
"life_table": lambda df, **kw: _life.life_table(None, df, **kw), "commutation": lambda df, **kw: _life.commutation(None, df, **kw),
|
|
35
|
+
"factors": lambda df, **kw: _life.factors(None, df, **kw), "ae": _life.ae, "model_points": _life.model_points, "graduate": _life.graduate,
|
|
36
|
+
"lee_carter": _life.lee_carter, "kaplan_meier": _life.kaplan_meier, "basicterm": _life.basicterm,
|
|
37
|
+
"glm": _pricing.glm, "freq_sev": _pricing.freq_sev, "loss_ratio": _pricing.loss_ratio,
|
|
38
|
+
"fairness": _fairness.fairness, "fairness_audit": _fairness.fairness_audit,
|
|
39
|
+
"join": _combine.join, "append": _combine.append, "combine": _combine.combine, "diff": _combine.diff,
|
|
40
|
+
"bottleneck": _workspace.bottleneck, "active_subspace": _workspace.active_subspace,
|
|
41
|
+
"hard": _hard,
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@pd.api.extensions.register_dataframe_accessor("sc")
|
|
46
|
+
class SceloAccessor:
|
|
47
|
+
"""``df.sc.<function>(...)`` for every frame-first Scelo function."""
|
|
48
|
+
|
|
49
|
+
def __init__(self, obj: pd.DataFrame) -> None:
|
|
50
|
+
self._obj = obj
|
|
51
|
+
|
|
52
|
+
def __getattr__(self, name: str) -> Any:
|
|
53
|
+
fn = _METHODS.get(name)
|
|
54
|
+
if fn is None:
|
|
55
|
+
raise AttributeError(f"df.sc has no method {name!r}; available: {', '.join(sorted(_METHODS))}")
|
|
56
|
+
return lambda *a, **kw: fn(self._obj, *a, **kw)
|
|
57
|
+
|
|
58
|
+
def __dir__(self): # pragma: no cover
|
|
59
|
+
return sorted(_METHODS)
|