mclovin 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mclovin/__init__.py +118 -0
- mclovin/basketball/__init__.py +65 -0
- mclovin/basketball/box_metrics.py +916 -0
- mclovin/basketball/coefficients.py +30 -0
- mclovin/basketball/draft.py +954 -0
- mclovin/basketball/four_factors.py +527 -0
- mclovin/basketball/identities.py +55 -0
- mclovin/basketball/possessions.py +247 -0
- mclovin/basketball/reconstruct.py +201 -0
- mclovin/core.py +177 -0
- mclovin/evaluation.py +125 -0
- mclovin/panel/__init__.py +29 -0
- mclovin/panel/features.py +204 -0
- mclovin/panel/splits.py +93 -0
- mclovin/py.typed +0 -0
- mclovin/quality.py +309 -0
- mclovin/schema.py +167 -0
- mclovin/shrinkage.py +615 -0
- mclovin/similarity.py +625 -0
- mclovin/survival.py +221 -0
- mclovin/units.py +253 -0
- mclovin/vocab/__init__.py +128 -0
- mclovin/vocab/basketball.py +175 -0
- mclovin-1.0.0.dist-info/METADATA +319 -0
- mclovin-1.0.0.dist-info/RECORD +28 -0
- mclovin-1.0.0.dist-info/WHEEL +5 -0
- mclovin-1.0.0.dist-info/licenses/LICENSE +21 -0
- mclovin-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,916 @@
|
|
|
1
|
+
"""
|
|
2
|
+
box_metrics.py
|
|
3
|
+
==============
|
|
4
|
+
|
|
5
|
+
All-in-one box-score metrics and the usage/rate family.
|
|
6
|
+
|
|
7
|
+
Contents
|
|
8
|
+
--------
|
|
9
|
+
`add_box_plus_minus_linear` Daniel Myers' simplified linear BPM
|
|
10
|
+
`add_player_efficiency_rating` John Hollinger's PER (needs LeagueContext)
|
|
11
|
+
`add_game_score` Hollinger's Game Score
|
|
12
|
+
`add_performance_impact_estimator` / `add_team_impact_estimator` PIE / TIE
|
|
13
|
+
`add_true_shooting_pct` TS%
|
|
14
|
+
`add_usage_rate` Usage% (team context)
|
|
15
|
+
`add_assist_pct` AST% (team context)
|
|
16
|
+
`add_rebound_pct` TRB%/ORB%/DRB% (team context)
|
|
17
|
+
`add_steal_pct` / `add_block_pct` STL%/BLK% (team context)
|
|
18
|
+
|
|
19
|
+
A note on what these are for
|
|
20
|
+
----------------------------
|
|
21
|
+
BPM, PER and Win Shares are *descriptive* summaries of what a player did.
|
|
22
|
+
The draft-prediction literature (see `draft.py`) consistently finds that the
|
|
23
|
+
component rates (steal rate, block rate, assist-to-turnover, free-throw
|
|
24
|
+
percentage) carry more predictive signal about NBA outcomes than the
|
|
25
|
+
all-in-one composites do. Use the composites as features and as scouting
|
|
26
|
+
context, not as the whole model.
|
|
27
|
+
|
|
28
|
+
Defence caveat, in Myers' own words: box-score defensive estimates are weak,
|
|
29
|
+
so treat DBPM as a guide rather than ground truth, and discount it for
|
|
30
|
+
players known to be much better or worse defenders than their box line.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
from typing import Optional
|
|
37
|
+
|
|
38
|
+
import numpy as np
|
|
39
|
+
import pandas as pd
|
|
40
|
+
|
|
41
|
+
from ..core import (
|
|
42
|
+
ComputationReport,
|
|
43
|
+
MetricSpec,
|
|
44
|
+
Reliability,
|
|
45
|
+
require_columns,
|
|
46
|
+
safe_divide,
|
|
47
|
+
)
|
|
48
|
+
from .coefficients import fta_coefficient
|
|
49
|
+
from ..schema import SchemaResolver
|
|
50
|
+
|
|
51
|
+
# ---------------------------------------------------------------------------
|
|
52
|
+
# Box Plus/Minus (Daniel Myers)
|
|
53
|
+
# ---------------------------------------------------------------------------
|
|
54
|
+
|
|
55
|
+
#: Myers' published **simplified linear** BPM coefficients, per 100 team
|
|
56
|
+
#: possessions. Source: Basketball-Reference "About Box Plus/Minus (BPM)"
|
|
57
|
+
#: (Myers, Feb 2020) and godismyjudgeok.com/DStats/box-plusminus/.
|
|
58
|
+
#:
|
|
59
|
+
#: IMPORTANT: the full BPM 2.0 uses position- and role-dependent coefficients
|
|
60
|
+
#: with interaction terms that are only partially published. This linear
|
|
61
|
+
#: version is the one Myers explicitly offers for small samples and for
|
|
62
|
+
#: reimplementation. Results approximate but do not reproduce
|
|
63
|
+
#: Basketball-Reference's published BPM.
|
|
64
|
+
#:
|
|
65
|
+
#: Note the steal coefficient (1.3571) is by far the largest positive weight,
|
|
66
|
+
#: which is mechanically consistent with the independent draft-research
|
|
67
|
+
#: finding that steal rate is an unusually strong predictor of NBA success.
|
|
68
|
+
BPM_LINEAR_COEFFICIENTS: dict[str, float] = {
|
|
69
|
+
"PTS": 0.7008,
|
|
70
|
+
"TPA": -0.4155,
|
|
71
|
+
"TWOPA": -0.5424,
|
|
72
|
+
"FTA": -0.2589,
|
|
73
|
+
"OREB": 0.1398,
|
|
74
|
+
"DREB": 0.3444,
|
|
75
|
+
"AST": 0.3846,
|
|
76
|
+
"STL": 1.3571,
|
|
77
|
+
"BLK": 0.6475,
|
|
78
|
+
"TOV": -0.9347,
|
|
79
|
+
"PF": -0.5153,
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
#: Intercept term applied per 100 possessions.
|
|
83
|
+
BPM_LINEAR_INTERCEPT: float = -0.0650
|
|
84
|
+
|
|
85
|
+
BPM_LINEAR_SPEC = MetricSpec(
|
|
86
|
+
name="BPM_LINEAR",
|
|
87
|
+
requires=("PTS", "TPA", "TWOPA", "FTA", "OREB", "DREB", "AST", "STL", "BLK", "TOV"),
|
|
88
|
+
source="Myers, D. (2020), 'About Box Plus/Minus (BPM)', Basketball-Reference",
|
|
89
|
+
description="Points above league average per 100 possessions (simplified linear BPM).",
|
|
90
|
+
notes=(
|
|
91
|
+
"Approximation of BPM 2.0. The full model uses position/role-dependent "
|
|
92
|
+
"coefficients that are only partially published. Scale: +8 MVP-level, "
|
|
93
|
+
"+4 all-star consideration, 0 solid starter, -2 replacement level."
|
|
94
|
+
),
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
#: Replacement level on the BPM scale, per Myers. VORP is measured against it.
|
|
98
|
+
BPM_REPLACEMENT_LEVEL: float = -2.0
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
#: Default team possessions per 40 minutes for NCAA play, used to estimate
|
|
102
|
+
#: team possessions while a player was on the floor when no pace data exists.
|
|
103
|
+
#: Division I pace has hovered around 67-70 possessions per 40 minutes in the
|
|
104
|
+
#: shot-clock era; KenPom publishes the exact figure by season and team.
|
|
105
|
+
DEFAULT_NCAA_PACE: float = 68.0
|
|
106
|
+
DEFAULT_NBA_PACE: float = 100.0
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def add_box_plus_minus_linear(
|
|
110
|
+
df: pd.DataFrame,
|
|
111
|
+
resolver: Optional[SchemaResolver] = None,
|
|
112
|
+
poss_col: Optional[str] = None,
|
|
113
|
+
minutes_col: Optional[str] = None,
|
|
114
|
+
pace: Optional[float] = None,
|
|
115
|
+
level: str = "ncaa",
|
|
116
|
+
center: bool = True,
|
|
117
|
+
new_col: str = "BPM_LINEAR",
|
|
118
|
+
report: Optional[ComputationReport] = None,
|
|
119
|
+
) -> pd.DataFrame:
|
|
120
|
+
"""Myers' simplified linear Box Plus/Minus.
|
|
121
|
+
|
|
122
|
+
BPM estimates a player's contribution in **points above league average per
|
|
123
|
+
100 team possessions**, with replacement level at -2.0. It was built by
|
|
124
|
+
regressing box-score inputs against long-run RAPM, so it is best
|
|
125
|
+
understood as a box-score approximation of RAPM (which itself needs
|
|
126
|
+
play-by-play lineup data and is not computable from box scores at all).
|
|
127
|
+
|
|
128
|
+
The denominator matters
|
|
129
|
+
-----------------------
|
|
130
|
+
Myers' coefficients are calibrated against **team possessions while the
|
|
131
|
+
player was on the floor**, not the possessions the player personally
|
|
132
|
+
used. A player uses roughly a fifth of his team's possessions, so
|
|
133
|
+
substituting individual possessions inflates every per-100 rate about
|
|
134
|
+
fivefold and (because the shot-attempt coefficients are negative) drives
|
|
135
|
+
BPM violently negative. This function therefore estimates::
|
|
136
|
+
|
|
137
|
+
team_poss_on_floor = minutes_played * (pace / game_minutes)
|
|
138
|
+
|
|
139
|
+
Parameters
|
|
140
|
+
----------
|
|
141
|
+
poss_col : explicit team-possessions-on-floor column. Use this when you
|
|
142
|
+
have real team pace data, since it is strictly better than the estimate.
|
|
143
|
+
minutes_col : minutes column. Defaults to the canonical MIN.
|
|
144
|
+
pace : team possessions per regulation game. Defaults to a league-average
|
|
145
|
+
assumption (68 NCAA, 100 NBA); supply the team's real pace when known.
|
|
146
|
+
center : re-centre so the minutes-weighted mean is 0.0 (default True).
|
|
147
|
+
BPM is *defined* as points above **league average**, so a zero mean is
|
|
148
|
+
part of the specification, not a cosmetic adjustment. Myers' published
|
|
149
|
+
coefficients are NBA-calibrated; applied to NCAA data they produce a
|
|
150
|
+
constant offset that centring removes. Note this centres against
|
|
151
|
+
whatever rows you pass, so it equals the true league average only if
|
|
152
|
+
the DataFrame is the full league. Pass the whole season, not a
|
|
153
|
+
subset, and re-use the same offset for train and test.
|
|
154
|
+
|
|
155
|
+
Warning
|
|
156
|
+
-------
|
|
157
|
+
If your dataset already provides a BPM column (BartTorvik's `bpm`), prefer
|
|
158
|
+
it, since it is computed with the full position/role model against proper
|
|
159
|
+
team context. Use this function when you need BPM on data that lacks one,
|
|
160
|
+
and label the result as an approximation.
|
|
161
|
+
"""
|
|
162
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
163
|
+
out = df.copy()
|
|
164
|
+
game_minutes = 40.0 if level.lower() == "ncaa" else 48.0
|
|
165
|
+
assumed_pace = pace if pace is not None else (
|
|
166
|
+
DEFAULT_NCAA_PACE if level.lower() == "ncaa" else DEFAULT_NBA_PACE
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
required = list(BPM_LINEAR_COEFFICIENTS.keys())
|
|
170
|
+
hard_required = [k for k in required if k != "PF"]
|
|
171
|
+
require_columns(r, hard_required, "add_box_plus_minus_linear")
|
|
172
|
+
|
|
173
|
+
pace_source = "supplied possessions column"
|
|
174
|
+
if poss_col is not None and poss_col in out.columns:
|
|
175
|
+
possessions = pd.to_numeric(out[poss_col], errors="coerce")
|
|
176
|
+
else:
|
|
177
|
+
if minutes_col is not None and minutes_col in out.columns:
|
|
178
|
+
minutes = pd.to_numeric(out[minutes_col], errors="coerce")
|
|
179
|
+
elif r.has("MIN"):
|
|
180
|
+
minutes = pd.to_numeric(r.get(out, "MIN"), errors="coerce")
|
|
181
|
+
else:
|
|
182
|
+
raise KeyError(
|
|
183
|
+
"add_box_plus_minus_linear needs minutes (canonical MIN) to "
|
|
184
|
+
"estimate team possessions on the floor, or an explicit "
|
|
185
|
+
"poss_col. Supply one of them."
|
|
186
|
+
)
|
|
187
|
+
possessions = minutes * (assumed_pace / game_minutes)
|
|
188
|
+
pace_source = f"estimated from minutes at pace={assumed_pace}"
|
|
189
|
+
|
|
190
|
+
contribution = pd.Series(0.0, index=out.index)
|
|
191
|
+
used, skipped = [], []
|
|
192
|
+
for canonical, coefficient in BPM_LINEAR_COEFFICIENTS.items():
|
|
193
|
+
if not r.has(canonical):
|
|
194
|
+
skipped.append(canonical)
|
|
195
|
+
continue
|
|
196
|
+
contribution = contribution + coefficient * pd.to_numeric(
|
|
197
|
+
r.get(out, canonical), errors="coerce"
|
|
198
|
+
).fillna(0.0)
|
|
199
|
+
used.append(canonical)
|
|
200
|
+
|
|
201
|
+
per_100 = safe_divide(contribution, possessions) * 100.0
|
|
202
|
+
raw = per_100 + BPM_LINEAR_INTERCEPT
|
|
203
|
+
|
|
204
|
+
offset = 0.0
|
|
205
|
+
if center:
|
|
206
|
+
weights = pd.to_numeric(
|
|
207
|
+
r.get(out, "MIN") if r.has("MIN") else pd.Series(1.0, index=out.index),
|
|
208
|
+
errors="coerce",
|
|
209
|
+
).fillna(0.0)
|
|
210
|
+
finite = np.isfinite(raw) & (weights > 0)
|
|
211
|
+
if finite.any() and weights[finite].sum() > 0:
|
|
212
|
+
offset = float(np.average(raw[finite], weights=weights[finite]))
|
|
213
|
+
out[new_col] = raw - offset
|
|
214
|
+
|
|
215
|
+
if report is not None:
|
|
216
|
+
detail = (
|
|
217
|
+
f"Linear approximation of BPM 2.0. Terms used: {len(used)}. "
|
|
218
|
+
f"Possessions {pace_source}."
|
|
219
|
+
)
|
|
220
|
+
if center:
|
|
221
|
+
detail += f" Centred to zero mean (offset {offset:.2f})."
|
|
222
|
+
if skipped:
|
|
223
|
+
detail += f" Missing (treated as zero): {skipped}."
|
|
224
|
+
report.record(BPM_LINEAR_SPEC, Reliability.APPROXIMATED, detail)
|
|
225
|
+
return out
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def add_vorp(
|
|
229
|
+
df: pd.DataFrame,
|
|
230
|
+
resolver: Optional[SchemaResolver] = None,
|
|
231
|
+
bpm_col: str = "BPM",
|
|
232
|
+
minutes_share_col: Optional[str] = None,
|
|
233
|
+
team_games: Optional[float] = None,
|
|
234
|
+
season_games: float = 32.0,
|
|
235
|
+
new_col: str = "VORP",
|
|
236
|
+
report: Optional[ComputationReport] = None,
|
|
237
|
+
) -> pd.DataFrame:
|
|
238
|
+
"""Value Over Replacement Player.
|
|
239
|
+
|
|
240
|
+
VORP = (BPM - (-2.0)) * (share of team minutes played) * (team games / season games)
|
|
241
|
+
|
|
242
|
+
Replacement level is -2.0 BPM (Myers); a full team of replacement players
|
|
243
|
+
wins roughly 14 NBA games. The minutes-share term converts a rate into a
|
|
244
|
+
volume statistic: a +4 BPM player who plays 10% of minutes contributes
|
|
245
|
+
far less than one who plays 80%.
|
|
246
|
+
|
|
247
|
+
Parameters
|
|
248
|
+
----------
|
|
249
|
+
minutes_share_col : fraction (0-1) or percentage (0-100) of team minutes.
|
|
250
|
+
BartTorvik's `Min_per` (canonical MIN_PCT) is exactly this; if values
|
|
251
|
+
exceed 1.0 they are assumed to be percentages and divided by 100.
|
|
252
|
+
season_games : games in a full season (NCAA ~32, NBA 82).
|
|
253
|
+
"""
|
|
254
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
255
|
+
out = df.copy()
|
|
256
|
+
|
|
257
|
+
if bpm_col not in out.columns:
|
|
258
|
+
if r.has("BPM"):
|
|
259
|
+
bpm = r.get(out, "BPM")
|
|
260
|
+
else:
|
|
261
|
+
raise KeyError(
|
|
262
|
+
f"add_vorp needs a BPM column ('{bpm_col}' not found and canonical "
|
|
263
|
+
f"BPM unresolvable). Compute add_box_plus_minus_linear first."
|
|
264
|
+
)
|
|
265
|
+
else:
|
|
266
|
+
bpm = pd.to_numeric(out[bpm_col], errors="coerce")
|
|
267
|
+
|
|
268
|
+
if minutes_share_col is not None and minutes_share_col in out.columns:
|
|
269
|
+
share = pd.to_numeric(out[minutes_share_col], errors="coerce")
|
|
270
|
+
elif r.has("MIN_PCT"):
|
|
271
|
+
share = pd.to_numeric(r.get(out, "MIN_PCT"), errors="coerce")
|
|
272
|
+
else:
|
|
273
|
+
raise KeyError(
|
|
274
|
+
"add_vorp needs a minutes-share column (canonical MIN_PCT, e.g. "
|
|
275
|
+
"BartTorvik's 'Min_per')."
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
if share.max(skipna=True) is not np.nan and share.max(skipna=True) > 1.0:
|
|
279
|
+
share = share / 100.0
|
|
280
|
+
|
|
281
|
+
games_factor = 1.0 if team_games is None else (team_games / season_games)
|
|
282
|
+
out[new_col] = (bpm - BPM_REPLACEMENT_LEVEL) * share * games_factor
|
|
283
|
+
|
|
284
|
+
if report is not None:
|
|
285
|
+
spec = MetricSpec(
|
|
286
|
+
name=new_col,
|
|
287
|
+
requires=("BPM", "MIN_PCT"),
|
|
288
|
+
source="Basketball-Reference, 'NBA Win Shares' / VORP definition",
|
|
289
|
+
description="Points above a -2.0 BPM replacement player, scaled by playing time.",
|
|
290
|
+
notes="Replacement level -2.0 BPM per Myers.",
|
|
291
|
+
)
|
|
292
|
+
report.record(spec, Reliability.APPROXIMATED if team_games is None else Reliability.EXACT)
|
|
293
|
+
return out
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# ---------------------------------------------------------------------------
|
|
297
|
+
# Player Efficiency Rating (John Hollinger)
|
|
298
|
+
# ---------------------------------------------------------------------------
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
@dataclass(frozen=True)
|
|
302
|
+
class LeagueContext:
|
|
303
|
+
"""League-wide totals required by PER's unadjusted stage.
|
|
304
|
+
|
|
305
|
+
PER is defined relative to league averages, so it simply cannot be
|
|
306
|
+
computed from player rows alone. Supply these from a season-level
|
|
307
|
+
aggregate (Sports-Reference publishes them per season).
|
|
308
|
+
|
|
309
|
+
Attributes are league TOTALS for the season, not per-game averages.
|
|
310
|
+
"""
|
|
311
|
+
|
|
312
|
+
lg_pts: float
|
|
313
|
+
lg_fga: float
|
|
314
|
+
lg_fgm: float
|
|
315
|
+
lg_fta: float
|
|
316
|
+
lg_ftm: float
|
|
317
|
+
lg_ast: float
|
|
318
|
+
lg_orb: float
|
|
319
|
+
lg_trb: float
|
|
320
|
+
lg_tov: float
|
|
321
|
+
lg_pf: float
|
|
322
|
+
lg_pace: Optional[float] = None
|
|
323
|
+
label: str = ""
|
|
324
|
+
|
|
325
|
+
@property
|
|
326
|
+
def value_of_possession(self) -> float:
|
|
327
|
+
"""VOP = lgPTS / (lgFGA - lgORB + lgTOV + 0.44*lgFTA)."""
|
|
328
|
+
denom = self.lg_fga - self.lg_orb + self.lg_tov + 0.44 * self.lg_fta
|
|
329
|
+
return self.lg_pts / denom if denom else 0.0
|
|
330
|
+
|
|
331
|
+
@property
|
|
332
|
+
def defensive_rebound_pct(self) -> float:
|
|
333
|
+
"""DRB% = (lgTRB - lgORB) / lgTRB."""
|
|
334
|
+
return (self.lg_trb - self.lg_orb) / self.lg_trb if self.lg_trb else 0.7
|
|
335
|
+
|
|
336
|
+
@property
|
|
337
|
+
def factor(self) -> float:
|
|
338
|
+
"""factor = 2/3 - (0.5 * (lgAST/lgFG)) / (2 * (lgFG/lgFT))."""
|
|
339
|
+
if not self.lg_fgm or not self.lg_ftm:
|
|
340
|
+
return 2.0 / 3.0
|
|
341
|
+
return (2.0 / 3.0) - (0.5 * (self.lg_ast / self.lg_fgm)) / (
|
|
342
|
+
2.0 * (self.lg_fgm / self.lg_ftm)
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
PER_SPEC = MetricSpec(
|
|
347
|
+
name="PER",
|
|
348
|
+
requires=("MIN", "TPM", "AST", "FGM", "FGA", "FTM", "FTA", "TOV", "REB", "OREB", "STL", "BLK", "PF"),
|
|
349
|
+
requires_team=("TM_AST", "TM_FG"),
|
|
350
|
+
source="Hollinger, J.; formula per Basketball-Reference",
|
|
351
|
+
description="Per-minute production, pace-adjusted, normalised so league average = 15.00.",
|
|
352
|
+
notes=(
|
|
353
|
+
"Requires league totals (LeagueContext) AND team AST/FG. Widely "
|
|
354
|
+
"criticised as usage- and offence-biased with weak defensive content. "
|
|
355
|
+
"Useful as a feature, poor as a target."
|
|
356
|
+
),
|
|
357
|
+
)
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def add_player_efficiency_rating(
|
|
361
|
+
df: pd.DataFrame,
|
|
362
|
+
league: LeagueContext,
|
|
363
|
+
resolver: Optional[SchemaResolver] = None,
|
|
364
|
+
team_pace_col: Optional[str] = None,
|
|
365
|
+
normalise: bool = True,
|
|
366
|
+
new_col: str = "PER",
|
|
367
|
+
report: Optional[ComputationReport] = None,
|
|
368
|
+
) -> pd.DataFrame:
|
|
369
|
+
"""Hollinger's Player Efficiency Rating.
|
|
370
|
+
|
|
371
|
+
Three stages: unadjusted PER (uPER) -> pace-adjusted (aPER) -> normalised
|
|
372
|
+
so the minutes-weighted league average is exactly 15.00.
|
|
373
|
+
|
|
374
|
+
uPER (per Basketball-Reference)::
|
|
375
|
+
|
|
376
|
+
uPER = (1/MP) * [ 3P
|
|
377
|
+
+ (2/3)*AST
|
|
378
|
+
+ (2 - factor*(TmAST/TmFG))*FG
|
|
379
|
+
+ FT*0.5*(1 + (1 - TmAST/TmFG) + (2/3)*(TmAST/TmFG))
|
|
380
|
+
- VOP*TOV
|
|
381
|
+
- VOP*DRB%*(FGA - FG)
|
|
382
|
+
- VOP*0.44*(0.44 + 0.56*DRB%)*(FTA - FT)
|
|
383
|
+
+ VOP*(1 - DRB%)*(TRB - ORB)
|
|
384
|
+
+ VOP*DRB%*ORB
|
|
385
|
+
+ VOP*STL
|
|
386
|
+
+ VOP*DRB%*BLK
|
|
387
|
+
- PF*((lgFT/lgPF) - 0.44*(lgFTA/lgPF)*VOP) ]
|
|
388
|
+
|
|
389
|
+
Parameters
|
|
390
|
+
----------
|
|
391
|
+
league : LeagueContext with season-wide totals.
|
|
392
|
+
team_pace_col : if supplied (and league.lg_pace set), applies the pace
|
|
393
|
+
adjustment aPER = (lgPace/TmPace) * uPER. Omitted -> uPER is used and
|
|
394
|
+
the result is flagged APPROXIMATED.
|
|
395
|
+
normalise : rescale so the minutes-weighted mean is 15.00. Note this
|
|
396
|
+
normalises against *this dataset*, which equals the true league
|
|
397
|
+
average only if the dataset is the full league.
|
|
398
|
+
"""
|
|
399
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
400
|
+
out = df.copy()
|
|
401
|
+
|
|
402
|
+
require_columns(
|
|
403
|
+
r,
|
|
404
|
+
["MIN", "TPM", "AST", "FGM", "FGA", "FTM", "FTA", "TOV", "REB", "OREB", "STL", "BLK"],
|
|
405
|
+
"add_player_efficiency_rating",
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
vop = league.value_of_possession
|
|
409
|
+
drb_pct = league.defensive_rebound_pct
|
|
410
|
+
factor = league.factor
|
|
411
|
+
|
|
412
|
+
mp = pd.to_numeric(r.get(out, "MIN"), errors="coerce")
|
|
413
|
+
tpm = pd.to_numeric(r.get(out, "TPM"), errors="coerce").fillna(0.0)
|
|
414
|
+
ast = pd.to_numeric(r.get(out, "AST"), errors="coerce").fillna(0.0)
|
|
415
|
+
fgm = pd.to_numeric(r.get(out, "FGM"), errors="coerce").fillna(0.0)
|
|
416
|
+
fga = pd.to_numeric(r.get(out, "FGA"), errors="coerce").fillna(0.0)
|
|
417
|
+
ftm = pd.to_numeric(r.get(out, "FTM"), errors="coerce").fillna(0.0)
|
|
418
|
+
fta = pd.to_numeric(r.get(out, "FTA"), errors="coerce").fillna(0.0)
|
|
419
|
+
tov = pd.to_numeric(r.get(out, "TOV"), errors="coerce").fillna(0.0)
|
|
420
|
+
trb = pd.to_numeric(r.get(out, "REB"), errors="coerce").fillna(0.0)
|
|
421
|
+
orb = pd.to_numeric(r.get(out, "OREB"), errors="coerce").fillna(0.0)
|
|
422
|
+
stl = pd.to_numeric(r.get(out, "STL"), errors="coerce").fillna(0.0)
|
|
423
|
+
blk = pd.to_numeric(r.get(out, "BLK"), errors="coerce").fillna(0.0)
|
|
424
|
+
pf = pd.to_numeric(r.get(out, "PF"), errors="coerce").fillna(0.0) if r.has("PF") else pd.Series(0.0, index=out.index)
|
|
425
|
+
|
|
426
|
+
# Team assist ratio: exact if team totals present, else league average.
|
|
427
|
+
team_context_exact = not r.missing(["TM_AST", "TM_FG"])
|
|
428
|
+
if team_context_exact:
|
|
429
|
+
tm_ast_ratio = safe_divide(r.get(out, "TM_AST"), r.get(out, "TM_FG"))
|
|
430
|
+
else:
|
|
431
|
+
lg_ratio = league.lg_ast / league.lg_fgm if league.lg_fgm else 0.0
|
|
432
|
+
tm_ast_ratio = pd.Series(lg_ratio, index=out.index)
|
|
433
|
+
|
|
434
|
+
foul_term = (
|
|
435
|
+
(league.lg_ftm / league.lg_pf) - 0.44 * (league.lg_fta / league.lg_pf) * vop
|
|
436
|
+
if league.lg_pf
|
|
437
|
+
else 0.0
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
raw = (
|
|
441
|
+
tpm
|
|
442
|
+
+ (2.0 / 3.0) * ast
|
|
443
|
+
+ (2.0 - factor * tm_ast_ratio) * fgm
|
|
444
|
+
+ ftm * 0.5 * (1.0 + (1.0 - tm_ast_ratio) + (2.0 / 3.0) * tm_ast_ratio)
|
|
445
|
+
- vop * tov
|
|
446
|
+
- vop * drb_pct * (fga - fgm)
|
|
447
|
+
- vop * 0.44 * (0.44 + 0.56 * drb_pct) * (fta - ftm)
|
|
448
|
+
+ vop * (1.0 - drb_pct) * (trb - orb)
|
|
449
|
+
+ vop * drb_pct * orb
|
|
450
|
+
+ vop * stl
|
|
451
|
+
+ vop * drb_pct * blk
|
|
452
|
+
- pf * foul_term
|
|
453
|
+
)
|
|
454
|
+
uper = safe_divide(raw, mp, fill=np.nan)
|
|
455
|
+
|
|
456
|
+
paced = uper
|
|
457
|
+
pace_applied = False
|
|
458
|
+
if team_pace_col is not None and team_pace_col in out.columns and league.lg_pace:
|
|
459
|
+
paced = uper * safe_divide(
|
|
460
|
+
pd.Series(league.lg_pace, index=out.index),
|
|
461
|
+
pd.to_numeric(out[team_pace_col], errors="coerce"),
|
|
462
|
+
fill=1.0,
|
|
463
|
+
)
|
|
464
|
+
pace_applied = True
|
|
465
|
+
|
|
466
|
+
if normalise:
|
|
467
|
+
weights = mp.fillna(0.0)
|
|
468
|
+
weighted_mean = (
|
|
469
|
+
np.average(paced.fillna(0.0), weights=weights) if weights.sum() > 0 else np.nan
|
|
470
|
+
)
|
|
471
|
+
out[new_col] = paced * (15.0 / weighted_mean) if weighted_mean else paced
|
|
472
|
+
else:
|
|
473
|
+
out[new_col] = paced
|
|
474
|
+
|
|
475
|
+
if report is not None:
|
|
476
|
+
reliability = (
|
|
477
|
+
Reliability.EXACT
|
|
478
|
+
if (team_context_exact and pace_applied)
|
|
479
|
+
else Reliability.APPROXIMATED
|
|
480
|
+
)
|
|
481
|
+
detail_bits = []
|
|
482
|
+
if not team_context_exact:
|
|
483
|
+
detail_bits.append("team AST/FG substituted with league average")
|
|
484
|
+
if not pace_applied:
|
|
485
|
+
detail_bits.append("no pace adjustment applied")
|
|
486
|
+
if normalise:
|
|
487
|
+
detail_bits.append("normalised against this dataset, not the true league")
|
|
488
|
+
report.record(PER_SPEC, reliability, "; ".join(detail_bits))
|
|
489
|
+
return out
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
GAME_SCORE_SPEC = MetricSpec(
|
|
493
|
+
name="GAME_SCORE",
|
|
494
|
+
requires=("PTS", "FGM", "FGA", "FTA", "FTM", "OREB", "DREB", "STL", "AST", "BLK", "TOV"),
|
|
495
|
+
source="Hollinger, J.; Basketball-Reference glossary",
|
|
496
|
+
description="Single-number box-score summary scaled roughly like points (10 = solid).",
|
|
497
|
+
notes="Simple and transparent; no team or league context required.",
|
|
498
|
+
)
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def add_game_score(
|
|
502
|
+
df: pd.DataFrame,
|
|
503
|
+
resolver: Optional[SchemaResolver] = None,
|
|
504
|
+
new_col: str = "GAME_SCORE",
|
|
505
|
+
report: Optional[ComputationReport] = None,
|
|
506
|
+
) -> pd.DataFrame:
|
|
507
|
+
"""Hollinger's Game Score::
|
|
508
|
+
|
|
509
|
+
GmSc = PTS + 0.4*FG - 0.7*FGA - 0.4*(FTA - FT) + 0.7*ORB + 0.3*DRB
|
|
510
|
+
+ STL + 0.7*AST + 0.7*BLK - 0.4*PF - TOV
|
|
511
|
+
|
|
512
|
+
Calibrated so ~10 is a solid outing and ~40 is outstanding. Fully
|
|
513
|
+
computable from an individual box score, which makes it a useful,
|
|
514
|
+
honest baseline when team context is unavailable.
|
|
515
|
+
"""
|
|
516
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
517
|
+
out = df.copy()
|
|
518
|
+
require_columns(
|
|
519
|
+
r,
|
|
520
|
+
["PTS", "FGM", "FGA", "FTA", "FTM", "OREB", "DREB", "STL", "AST", "BLK", "TOV"],
|
|
521
|
+
"add_game_score",
|
|
522
|
+
)
|
|
523
|
+
|
|
524
|
+
g = lambda name: pd.to_numeric(r.get(out, name), errors="coerce").fillna(0.0)
|
|
525
|
+
pf = g("PF") if r.has("PF") else pd.Series(0.0, index=out.index)
|
|
526
|
+
|
|
527
|
+
out[new_col] = (
|
|
528
|
+
g("PTS")
|
|
529
|
+
+ 0.4 * g("FGM")
|
|
530
|
+
- 0.7 * g("FGA")
|
|
531
|
+
- 0.4 * (g("FTA") - g("FTM"))
|
|
532
|
+
+ 0.7 * g("OREB")
|
|
533
|
+
+ 0.3 * g("DREB")
|
|
534
|
+
+ g("STL")
|
|
535
|
+
+ 0.7 * g("AST")
|
|
536
|
+
+ 0.7 * g("BLK")
|
|
537
|
+
- 0.4 * pf
|
|
538
|
+
- g("TOV")
|
|
539
|
+
)
|
|
540
|
+
|
|
541
|
+
if report is not None:
|
|
542
|
+
report.record(GAME_SCORE_SPEC, Reliability.EXACT)
|
|
543
|
+
return out
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
# ---------------------------------------------------------------------------
|
|
547
|
+
# Performance/Team Impact Estimator (NBA.com)
|
|
548
|
+
# ---------------------------------------------------------------------------
|
|
549
|
+
|
|
550
|
+
PIE_SPEC = MetricSpec(
|
|
551
|
+
name="PIE",
|
|
552
|
+
requires=("PTS", "FGM", "FGA", "FTM", "FTA", "DREB", "OREB", "AST", "STL", "BLK", "TOV"),
|
|
553
|
+
source="NBA.com Advanced Stats glossary, 'Performance Impact Estimator'",
|
|
554
|
+
description="Single-number estimate of a player's or team's overall statistical output in a game.",
|
|
555
|
+
notes=(
|
|
556
|
+
"PIE = PTS + FGM + FTM - FGA - FTA + DREB + 0.5*OREB + AST + STL "
|
|
557
|
+
"+ 0.5*BLK - PF - TOV. Unlike Game Score, it has no tuned "
|
|
558
|
+
"coefficients: it is a plain sum of positive box-score contributions "
|
|
559
|
+
"minus negative ones (missed shots, missed free throws, fouls, "
|
|
560
|
+
"turnovers). Mainly useful as the input to add_team_impact_estimator."
|
|
561
|
+
),
|
|
562
|
+
)
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def add_performance_impact_estimator(
|
|
566
|
+
df: pd.DataFrame,
|
|
567
|
+
resolver: Optional[SchemaResolver] = None,
|
|
568
|
+
new_col: str = "PIE",
|
|
569
|
+
report: Optional[ComputationReport] = None,
|
|
570
|
+
) -> pd.DataFrame:
|
|
571
|
+
"""Performance Impact Estimator (PIE)::
|
|
572
|
+
|
|
573
|
+
PIE = PTS + FGM + FTM - FGA - FTA + DREB + 0.5*OREB
|
|
574
|
+
+ AST + STL + 0.5*BLK - PF - TOV
|
|
575
|
+
|
|
576
|
+
Fully computable from a single row's own box score (player or team); no
|
|
577
|
+
team or opponent context required. On its own PIE is not very meaningful,
|
|
578
|
+
since it has no fixed scale; it exists as the building block for
|
|
579
|
+
`add_team_impact_estimator`, which expresses it as a share of the total
|
|
580
|
+
production in the game.
|
|
581
|
+
"""
|
|
582
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
583
|
+
out = df.copy()
|
|
584
|
+
require_columns(
|
|
585
|
+
r,
|
|
586
|
+
["PTS", "FGM", "FGA", "FTM", "FTA", "DREB", "OREB", "AST", "STL", "BLK", "TOV"],
|
|
587
|
+
"add_performance_impact_estimator",
|
|
588
|
+
)
|
|
589
|
+
|
|
590
|
+
g = lambda name: pd.to_numeric(r.get(out, name), errors="coerce").fillna(0.0)
|
|
591
|
+
pf = g("PF") if r.has("PF") else pd.Series(0.0, index=out.index)
|
|
592
|
+
|
|
593
|
+
out[new_col] = (
|
|
594
|
+
g("PTS")
|
|
595
|
+
+ g("FGM")
|
|
596
|
+
+ g("FTM")
|
|
597
|
+
- g("FGA")
|
|
598
|
+
- g("FTA")
|
|
599
|
+
+ g("DREB")
|
|
600
|
+
+ 0.5 * g("OREB")
|
|
601
|
+
+ g("AST")
|
|
602
|
+
+ g("STL")
|
|
603
|
+
+ 0.5 * g("BLK")
|
|
604
|
+
- pf
|
|
605
|
+
- g("TOV")
|
|
606
|
+
)
|
|
607
|
+
|
|
608
|
+
if report is not None:
|
|
609
|
+
report.record(PIE_SPEC, Reliability.EXACT)
|
|
610
|
+
return out
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
TIE_SPEC = MetricSpec(
|
|
614
|
+
name="TIE",
|
|
615
|
+
requires=("PTS", "FGM", "FGA", "FTM", "FTA", "DREB", "OREB", "AST", "STL", "BLK", "TOV"),
|
|
616
|
+
requires_team=(
|
|
617
|
+
"OPP_PTS", "OPP_FG", "OPP_FGA", "OPP_FTM", "OPP_FTA",
|
|
618
|
+
"OPP_DRB", "OPP_ORB", "OPP_AST", "OPP_STL", "OPP_BLK", "OPP_TOV",
|
|
619
|
+
),
|
|
620
|
+
source="NBA.com Advanced Stats glossary, 'Team Impact Estimator'",
|
|
621
|
+
description="A player's or team's PIE as a share of the combined PIE produced by both sides.",
|
|
622
|
+
notes="TIE = PIE / (PIE + OppPIE). 0.50 is an even statistical split.",
|
|
623
|
+
)
|
|
624
|
+
|
|
625
|
+
|
|
626
|
+
def add_team_impact_estimator(
|
|
627
|
+
df: pd.DataFrame,
|
|
628
|
+
resolver: Optional[SchemaResolver] = None,
|
|
629
|
+
pie_col: Optional[str] = None,
|
|
630
|
+
new_col: str = "TIE",
|
|
631
|
+
report: Optional[ComputationReport] = None,
|
|
632
|
+
) -> pd.DataFrame:
|
|
633
|
+
"""Team Impact Estimator: TIE = PIE / (PIE + OppPIE).
|
|
634
|
+
|
|
635
|
+
Rescales PIE into a share (0-1) of the combined production in the game:
|
|
636
|
+
0.50 is an even split, above it means winning the statistical battle.
|
|
637
|
+
**Requires the opponent's full box score**, since the opponent's PIE is
|
|
638
|
+
computed directly from OPP_PTS, OPP_FG, OPP_FGA, OPP_FTM, OPP_FTA,
|
|
639
|
+
OPP_DRB, OPP_ORB, OPP_AST, OPP_STL, OPP_BLK, OPP_TOV (and OPP_PF if
|
|
640
|
+
fouls are tracked), the same "eight factors"-style opponent context
|
|
641
|
+
`four_factors.py` uses for the defensive Four Factors.
|
|
642
|
+
|
|
643
|
+
Parameters
|
|
644
|
+
----------
|
|
645
|
+
pie_col : an existing PIE column to reuse instead of recomputing it.
|
|
646
|
+
Defaults to computing PIE fresh via `add_performance_impact_estimator`.
|
|
647
|
+
"""
|
|
648
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
649
|
+
out = df.copy()
|
|
650
|
+
|
|
651
|
+
if pie_col is not None and pie_col in out.columns:
|
|
652
|
+
pie = pd.to_numeric(out[pie_col], errors="coerce")
|
|
653
|
+
else:
|
|
654
|
+
out = add_performance_impact_estimator(out, resolver=r)
|
|
655
|
+
pie = out["PIE"]
|
|
656
|
+
|
|
657
|
+
opp_required = [
|
|
658
|
+
"OPP_PTS", "OPP_FG", "OPP_FGA", "OPP_FTM", "OPP_FTA",
|
|
659
|
+
"OPP_DRB", "OPP_ORB", "OPP_AST", "OPP_STL", "OPP_BLK", "OPP_TOV",
|
|
660
|
+
]
|
|
661
|
+
require_columns(r, opp_required, "add_team_impact_estimator")
|
|
662
|
+
|
|
663
|
+
g = lambda name: pd.to_numeric(r.get(out, name), errors="coerce").fillna(0.0)
|
|
664
|
+
opp_pf = g("OPP_PF") if r.has("OPP_PF") else pd.Series(0.0, index=out.index)
|
|
665
|
+
|
|
666
|
+
opp_pie = (
|
|
667
|
+
g("OPP_PTS")
|
|
668
|
+
+ g("OPP_FG")
|
|
669
|
+
+ g("OPP_FTM")
|
|
670
|
+
- g("OPP_FGA")
|
|
671
|
+
- g("OPP_FTA")
|
|
672
|
+
+ g("OPP_DRB")
|
|
673
|
+
+ 0.5 * g("OPP_ORB")
|
|
674
|
+
+ g("OPP_AST")
|
|
675
|
+
+ g("OPP_STL")
|
|
676
|
+
+ 0.5 * g("OPP_BLK")
|
|
677
|
+
- opp_pf
|
|
678
|
+
- g("OPP_TOV")
|
|
679
|
+
)
|
|
680
|
+
|
|
681
|
+
out[new_col] = safe_divide(pie, pie + opp_pie)
|
|
682
|
+
|
|
683
|
+
if report is not None:
|
|
684
|
+
report.record(TIE_SPEC, Reliability.EXACT)
|
|
685
|
+
return out
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
# ---------------------------------------------------------------------------
|
|
689
|
+
# Shooting efficiency and the usage/rate family
|
|
690
|
+
# ---------------------------------------------------------------------------
|
|
691
|
+
|
|
692
|
+
TS_SPEC = MetricSpec(
|
|
693
|
+
name="TS_PCT_CALC",
|
|
694
|
+
requires=("PTS", "FGA", "FTA"),
|
|
695
|
+
source="Basketball-Reference glossary; Oliver (2004)",
|
|
696
|
+
description="True shooting %: scoring efficiency including threes and free throws.",
|
|
697
|
+
notes="TS% = PTS / (2 * (FGA + c*FTA)). c = 0.44 NBA, 0.475 NCAA.",
|
|
698
|
+
)
|
|
699
|
+
|
|
700
|
+
|
|
701
|
+
def add_true_shooting_pct(
|
|
702
|
+
df: pd.DataFrame,
|
|
703
|
+
resolver: Optional[SchemaResolver] = None,
|
|
704
|
+
level: str = "ncaa",
|
|
705
|
+
new_col: str = "TS_PCT_CALC",
|
|
706
|
+
report: Optional[ComputationReport] = None,
|
|
707
|
+
) -> pd.DataFrame:
|
|
708
|
+
"""TS% = PTS / (2 * (FGA + c*FTA)).
|
|
709
|
+
|
|
710
|
+
The most complete single measure of scoring efficiency, since it accounts
|
|
711
|
+
for two-pointers, threes and free throws in one number. Prefer the
|
|
712
|
+
dataset's own TS% column when present.
|
|
713
|
+
"""
|
|
714
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
715
|
+
out = df.copy()
|
|
716
|
+
c = fta_coefficient(level)
|
|
717
|
+
require_columns(r, ["PTS", "FGA", "FTA"], "add_true_shooting_pct")
|
|
718
|
+
|
|
719
|
+
pts, fga, fta = r.get(out, "PTS"), r.get(out, "FGA"), r.get(out, "FTA")
|
|
720
|
+
out[new_col] = safe_divide(pts, 2.0 * (fga + c * fta))
|
|
721
|
+
|
|
722
|
+
if report is not None:
|
|
723
|
+
report.record(TS_SPEC, Reliability.EXACT)
|
|
724
|
+
return out
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
USAGE_SPEC = MetricSpec(
|
|
728
|
+
name="USG_PCT_CALC",
|
|
729
|
+
requires=("FGA", "FTA", "TOV", "MIN"),
|
|
730
|
+
requires_team=("TM_MP", "TM_FGA", "TM_FTA", "TM_TOV"),
|
|
731
|
+
source="Basketball-Reference glossary",
|
|
732
|
+
description="Share of team possessions a player used while on the floor.",
|
|
733
|
+
notes="Requires team totals. Without them, use possessions-used per minute instead.",
|
|
734
|
+
)
|
|
735
|
+
|
|
736
|
+
|
|
737
|
+
def add_usage_rate(
|
|
738
|
+
df: pd.DataFrame,
|
|
739
|
+
resolver: Optional[SchemaResolver] = None,
|
|
740
|
+
level: str = "ncaa",
|
|
741
|
+
new_col: str = "USG_PCT_CALC",
|
|
742
|
+
report: Optional[ComputationReport] = None,
|
|
743
|
+
) -> pd.DataFrame:
|
|
744
|
+
"""Usage% = 100 * ((FGA + c*FTA + TOV) * (TmMP/5)) / (MP * (TmFGA + c*TmFTA + TmTOV)).
|
|
745
|
+
|
|
746
|
+
**Requires team totals.** BartTorvik supplies `usg` precomputed, so prefer
|
|
747
|
+
that. This exists for datasets that carry team aggregates.
|
|
748
|
+
"""
|
|
749
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
750
|
+
out = df.copy()
|
|
751
|
+
c = fta_coefficient(level)
|
|
752
|
+
require_columns(
|
|
753
|
+
r, ["FGA", "FTA", "TOV", "MIN", "TM_MP", "TM_FGA", "TM_FTA", "TM_TOV"], "add_usage_rate"
|
|
754
|
+
)
|
|
755
|
+
|
|
756
|
+
player_poss = r.get(out, "FGA") + c * r.get(out, "FTA") + r.get(out, "TOV")
|
|
757
|
+
team_poss = r.get(out, "TM_FGA") + c * r.get(out, "TM_FTA") + r.get(out, "TM_TOV")
|
|
758
|
+
out[new_col] = 100.0 * safe_divide(
|
|
759
|
+
player_poss * (r.get(out, "TM_MP") / 5.0), r.get(out, "MIN") * team_poss
|
|
760
|
+
)
|
|
761
|
+
|
|
762
|
+
if report is not None:
|
|
763
|
+
report.record(USAGE_SPEC, Reliability.EXACT)
|
|
764
|
+
return out
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
def add_assist_pct(
|
|
768
|
+
df: pd.DataFrame,
|
|
769
|
+
resolver: Optional[SchemaResolver] = None,
|
|
770
|
+
new_col: str = "AST_PCT_CALC",
|
|
771
|
+
report: Optional[ComputationReport] = None,
|
|
772
|
+
) -> pd.DataFrame:
|
|
773
|
+
"""AST% = 100 * AST / (((MP / (TmMP/5)) * TmFG) - FG).
|
|
774
|
+
|
|
775
|
+
Share of teammate field goals a player assisted while on the floor.
|
|
776
|
+
**Requires team totals.**
|
|
777
|
+
"""
|
|
778
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
779
|
+
out = df.copy()
|
|
780
|
+
require_columns(r, ["AST", "MIN", "FGM", "TM_MP", "TM_FG"], "add_assist_pct")
|
|
781
|
+
|
|
782
|
+
mp, tm_mp, tm_fg = r.get(out, "MIN"), r.get(out, "TM_MP"), r.get(out, "TM_FG")
|
|
783
|
+
denom = safe_divide(mp, tm_mp / 5.0) * tm_fg - r.get(out, "FGM")
|
|
784
|
+
out[new_col] = 100.0 * safe_divide(r.get(out, "AST"), denom)
|
|
785
|
+
|
|
786
|
+
if report is not None:
|
|
787
|
+
spec = MetricSpec(
|
|
788
|
+
name=new_col,
|
|
789
|
+
requires=("AST", "MIN", "FGM"),
|
|
790
|
+
requires_team=("TM_MP", "TM_FG"),
|
|
791
|
+
source="Basketball-Reference glossary",
|
|
792
|
+
description="Share of teammate field goals assisted while on court.",
|
|
793
|
+
)
|
|
794
|
+
report.record(spec, Reliability.EXACT)
|
|
795
|
+
return out
|
|
796
|
+
|
|
797
|
+
|
|
798
|
+
def add_rebound_pct(
|
|
799
|
+
df: pd.DataFrame,
|
|
800
|
+
resolver: Optional[SchemaResolver] = None,
|
|
801
|
+
which: str = "total",
|
|
802
|
+
new_col: Optional[str] = None,
|
|
803
|
+
report: Optional[ComputationReport] = None,
|
|
804
|
+
) -> pd.DataFrame:
|
|
805
|
+
"""Rebound percentage: share of available rebounds captured on court.
|
|
806
|
+
|
|
807
|
+
TRB% = 100 * (TRB * (TmMP/5)) / (MP * (TmTRB + OppTRB))
|
|
808
|
+
ORB% = 100 * (ORB * (TmMP/5)) / (MP * (TmORB + OppDRB))
|
|
809
|
+
DRB% = 100 * (DRB * (TmMP/5)) / (MP * (TmDRB + OppORB))
|
|
810
|
+
|
|
811
|
+
**Requires team and opponent totals.**
|
|
812
|
+
|
|
813
|
+
Parameters
|
|
814
|
+
----------
|
|
815
|
+
which : {"total", "offensive", "defensive"}
|
|
816
|
+
"""
|
|
817
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
818
|
+
out = df.copy()
|
|
819
|
+
|
|
820
|
+
config = {
|
|
821
|
+
"total": ("REB", ["TM_TRB", "OPP_TRB"], "TRB_PCT_CALC"),
|
|
822
|
+
"offensive": ("OREB", ["TM_ORB", "OPP_DRB"], "ORB_PCT_CALC"),
|
|
823
|
+
"defensive": ("DREB", ["TM_DRB", "OPP_ORB"], "DRB_PCT_CALC"),
|
|
824
|
+
}
|
|
825
|
+
if which not in config:
|
|
826
|
+
raise ValueError(f"which must be one of {sorted(config)}, got {which!r}")
|
|
827
|
+
|
|
828
|
+
stat, team_terms, default_name = config[which]
|
|
829
|
+
target = new_col or default_name
|
|
830
|
+
require_columns(r, [stat, "MIN", "TM_MP"] + team_terms, f"add_rebound_pct({which})")
|
|
831
|
+
|
|
832
|
+
available = r.get(out, team_terms[0]) + r.get(out, team_terms[1])
|
|
833
|
+
out[target] = 100.0 * safe_divide(
|
|
834
|
+
r.get(out, stat) * (r.get(out, "TM_MP") / 5.0), r.get(out, "MIN") * available
|
|
835
|
+
)
|
|
836
|
+
|
|
837
|
+
if report is not None:
|
|
838
|
+
spec = MetricSpec(
|
|
839
|
+
name=target,
|
|
840
|
+
requires=(stat, "MIN"),
|
|
841
|
+
requires_team=tuple(["TM_MP"] + team_terms),
|
|
842
|
+
source="Basketball-Reference glossary; Oliver (2004)",
|
|
843
|
+
description=f"Share of available {which} rebounds captured while on court.",
|
|
844
|
+
)
|
|
845
|
+
report.record(spec, Reliability.EXACT)
|
|
846
|
+
return out
|
|
847
|
+
|
|
848
|
+
|
|
849
|
+
def add_steal_pct(
|
|
850
|
+
df: pd.DataFrame,
|
|
851
|
+
resolver: Optional[SchemaResolver] = None,
|
|
852
|
+
new_col: str = "STL_PCT_CALC",
|
|
853
|
+
report: Optional[ComputationReport] = None,
|
|
854
|
+
) -> pd.DataFrame:
|
|
855
|
+
"""STL% = 100 * (STL * (TmMP/5)) / (MP * OppPoss).
|
|
856
|
+
|
|
857
|
+
Share of opponent possessions ending in a steal by this player.
|
|
858
|
+
**Requires team minutes and opponent possessions.**
|
|
859
|
+
|
|
860
|
+
Worth computing carefully: steal rate is one of the strongest known
|
|
861
|
+
college-to-NBA predictors (see `draft.py` and Vashro's work), and Myers'
|
|
862
|
+
BPM assigns it the largest positive coefficient of any box-score term.
|
|
863
|
+
"""
|
|
864
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
865
|
+
out = df.copy()
|
|
866
|
+
require_columns(r, ["STL", "MIN", "TM_MP", "OPP_POSS"], "add_steal_pct")
|
|
867
|
+
|
|
868
|
+
out[new_col] = 100.0 * safe_divide(
|
|
869
|
+
r.get(out, "STL") * (r.get(out, "TM_MP") / 5.0),
|
|
870
|
+
r.get(out, "MIN") * r.get(out, "OPP_POSS"),
|
|
871
|
+
)
|
|
872
|
+
|
|
873
|
+
if report is not None:
|
|
874
|
+
spec = MetricSpec(
|
|
875
|
+
name=new_col,
|
|
876
|
+
requires=("STL", "MIN"),
|
|
877
|
+
requires_team=("TM_MP", "OPP_POSS"),
|
|
878
|
+
source="Basketball-Reference glossary",
|
|
879
|
+
description="Share of opponent possessions ended by a steal.",
|
|
880
|
+
notes="Strong NBA-success predictor in the draft literature.",
|
|
881
|
+
)
|
|
882
|
+
report.record(spec, Reliability.EXACT)
|
|
883
|
+
return out
|
|
884
|
+
|
|
885
|
+
|
|
886
|
+
def add_block_pct(
|
|
887
|
+
df: pd.DataFrame,
|
|
888
|
+
resolver: Optional[SchemaResolver] = None,
|
|
889
|
+
new_col: str = "BLK_PCT_CALC",
|
|
890
|
+
report: Optional[ComputationReport] = None,
|
|
891
|
+
) -> pd.DataFrame:
|
|
892
|
+
"""BLK% = 100 * (BLK * (TmMP/5)) / (MP * (OppFGA - Opp3PA)).
|
|
893
|
+
|
|
894
|
+
Share of opponent two-point attempts blocked. **Requires team minutes and
|
|
895
|
+
opponent shooting totals.** Block rate carries much of height's predictive
|
|
896
|
+
value for interior prospects.
|
|
897
|
+
"""
|
|
898
|
+
r = resolver if resolver is not None else SchemaResolver(df.columns)
|
|
899
|
+
out = df.copy()
|
|
900
|
+
require_columns(r, ["BLK", "MIN", "TM_MP", "OPP_FGA", "OPP_TPA"], "add_block_pct")
|
|
901
|
+
|
|
902
|
+
two_pt_attempts = r.get(out, "OPP_FGA") - r.get(out, "OPP_TPA")
|
|
903
|
+
out[new_col] = 100.0 * safe_divide(
|
|
904
|
+
r.get(out, "BLK") * (r.get(out, "TM_MP") / 5.0), r.get(out, "MIN") * two_pt_attempts
|
|
905
|
+
)
|
|
906
|
+
|
|
907
|
+
if report is not None:
|
|
908
|
+
spec = MetricSpec(
|
|
909
|
+
name=new_col,
|
|
910
|
+
requires=("BLK", "MIN"),
|
|
911
|
+
requires_team=("TM_MP", "OPP_FGA", "OPP_TPA"),
|
|
912
|
+
source="Basketball-Reference glossary",
|
|
913
|
+
description="Share of opponent two-point attempts blocked.",
|
|
914
|
+
)
|
|
915
|
+
report.record(spec, Reliability.EXACT)
|
|
916
|
+
return out
|