mclovin 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,916 @@
1
+ """
2
+ box_metrics.py
3
+ ==============
4
+
5
+ All-in-one box-score metrics and the usage/rate family.
6
+
7
+ Contents
8
+ --------
9
+ `add_box_plus_minus_linear` Daniel Myers' simplified linear BPM
10
+ `add_player_efficiency_rating` John Hollinger's PER (needs LeagueContext)
11
+ `add_game_score` Hollinger's Game Score
12
+ `add_performance_impact_estimator` / `add_team_impact_estimator` PIE / TIE
13
+ `add_true_shooting_pct` TS%
14
+ `add_usage_rate` Usage% (team context)
15
+ `add_assist_pct` AST% (team context)
16
+ `add_rebound_pct` TRB%/ORB%/DRB% (team context)
17
+ `add_steal_pct` / `add_block_pct` STL%/BLK% (team context)
18
+
19
+ A note on what these are for
20
+ ----------------------------
21
+ BPM, PER and Win Shares are *descriptive* summaries of what a player did.
22
+ The draft-prediction literature (see `draft.py`) consistently finds that the
23
+ component rates (steal rate, block rate, assist-to-turnover, free-throw
24
+ percentage) carry more predictive signal about NBA outcomes than the
25
+ all-in-one composites do. Use the composites as features and as scouting
26
+ context, not as the whole model.
27
+
28
+ Defence caveat, in Myers' own words: box-score defensive estimates are weak,
29
+ so treat DBPM as a guide rather than ground truth, and discount it for
30
+ players known to be much better or worse defenders than their box line.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ from dataclasses import dataclass
36
+ from typing import Optional
37
+
38
+ import numpy as np
39
+ import pandas as pd
40
+
41
+ from ..core import (
42
+ ComputationReport,
43
+ MetricSpec,
44
+ Reliability,
45
+ require_columns,
46
+ safe_divide,
47
+ )
48
+ from .coefficients import fta_coefficient
49
+ from ..schema import SchemaResolver
50
+
51
+ # ---------------------------------------------------------------------------
52
+ # Box Plus/Minus (Daniel Myers)
53
+ # ---------------------------------------------------------------------------
54
+
55
+ #: Myers' published **simplified linear** BPM coefficients, per 100 team
56
+ #: possessions. Source: Basketball-Reference "About Box Plus/Minus (BPM)"
57
+ #: (Myers, Feb 2020) and godismyjudgeok.com/DStats/box-plusminus/.
58
+ #:
59
+ #: IMPORTANT: the full BPM 2.0 uses position- and role-dependent coefficients
60
+ #: with interaction terms that are only partially published. This linear
61
+ #: version is the one Myers explicitly offers for small samples and for
62
+ #: reimplementation. Results approximate but do not reproduce
63
+ #: Basketball-Reference's published BPM.
64
+ #:
65
+ #: Note the steal coefficient (1.3571) is by far the largest positive weight,
66
+ #: which is mechanically consistent with the independent draft-research
67
+ #: finding that steal rate is an unusually strong predictor of NBA success.
68
+ BPM_LINEAR_COEFFICIENTS: dict[str, float] = {
69
+ "PTS": 0.7008,
70
+ "TPA": -0.4155,
71
+ "TWOPA": -0.5424,
72
+ "FTA": -0.2589,
73
+ "OREB": 0.1398,
74
+ "DREB": 0.3444,
75
+ "AST": 0.3846,
76
+ "STL": 1.3571,
77
+ "BLK": 0.6475,
78
+ "TOV": -0.9347,
79
+ "PF": -0.5153,
80
+ }
81
+
82
+ #: Intercept term applied per 100 possessions.
83
+ BPM_LINEAR_INTERCEPT: float = -0.0650
84
+
85
+ BPM_LINEAR_SPEC = MetricSpec(
86
+ name="BPM_LINEAR",
87
+ requires=("PTS", "TPA", "TWOPA", "FTA", "OREB", "DREB", "AST", "STL", "BLK", "TOV"),
88
+ source="Myers, D. (2020), 'About Box Plus/Minus (BPM)', Basketball-Reference",
89
+ description="Points above league average per 100 possessions (simplified linear BPM).",
90
+ notes=(
91
+ "Approximation of BPM 2.0. The full model uses position/role-dependent "
92
+ "coefficients that are only partially published. Scale: +8 MVP-level, "
93
+ "+4 all-star consideration, 0 solid starter, -2 replacement level."
94
+ ),
95
+ )
96
+
97
+ #: Replacement level on the BPM scale, per Myers. VORP is measured against it.
98
+ BPM_REPLACEMENT_LEVEL: float = -2.0
99
+
100
+
101
+ #: Default team possessions per 40 minutes for NCAA play, used to estimate
102
+ #: team possessions while a player was on the floor when no pace data exists.
103
+ #: Division I pace has hovered around 67-70 possessions per 40 minutes in the
104
+ #: shot-clock era; KenPom publishes the exact figure by season and team.
105
+ DEFAULT_NCAA_PACE: float = 68.0
106
+ DEFAULT_NBA_PACE: float = 100.0
107
+
108
+
109
+ def add_box_plus_minus_linear(
110
+ df: pd.DataFrame,
111
+ resolver: Optional[SchemaResolver] = None,
112
+ poss_col: Optional[str] = None,
113
+ minutes_col: Optional[str] = None,
114
+ pace: Optional[float] = None,
115
+ level: str = "ncaa",
116
+ center: bool = True,
117
+ new_col: str = "BPM_LINEAR",
118
+ report: Optional[ComputationReport] = None,
119
+ ) -> pd.DataFrame:
120
+ """Myers' simplified linear Box Plus/Minus.
121
+
122
+ BPM estimates a player's contribution in **points above league average per
123
+ 100 team possessions**, with replacement level at -2.0. It was built by
124
+ regressing box-score inputs against long-run RAPM, so it is best
125
+ understood as a box-score approximation of RAPM (which itself needs
126
+ play-by-play lineup data and is not computable from box scores at all).
127
+
128
+ The denominator matters
129
+ -----------------------
130
+ Myers' coefficients are calibrated against **team possessions while the
131
+ player was on the floor**, not the possessions the player personally
132
+ used. A player uses roughly a fifth of his team's possessions, so
133
+ substituting individual possessions inflates every per-100 rate about
134
+ fivefold and (because the shot-attempt coefficients are negative) drives
135
+ BPM violently negative. This function therefore estimates::
136
+
137
+ team_poss_on_floor = minutes_played * (pace / game_minutes)
138
+
139
+ Parameters
140
+ ----------
141
+ poss_col : explicit team-possessions-on-floor column. Use this when you
142
+ have real team pace data, since it is strictly better than the estimate.
143
+ minutes_col : minutes column. Defaults to the canonical MIN.
144
+ pace : team possessions per regulation game. Defaults to a league-average
145
+ assumption (68 NCAA, 100 NBA); supply the team's real pace when known.
146
+ center : re-centre so the minutes-weighted mean is 0.0 (default True).
147
+ BPM is *defined* as points above **league average**, so a zero mean is
148
+ part of the specification, not a cosmetic adjustment. Myers' published
149
+ coefficients are NBA-calibrated; applied to NCAA data they produce a
150
+ constant offset that centring removes. Note this centres against
151
+ whatever rows you pass, so it equals the true league average only if
152
+ the DataFrame is the full league. Pass the whole season, not a
153
+ subset, and re-use the same offset for train and test.
154
+
155
+ Warning
156
+ -------
157
+ If your dataset already provides a BPM column (BartTorvik's `bpm`), prefer
158
+ it, since it is computed with the full position/role model against proper
159
+ team context. Use this function when you need BPM on data that lacks one,
160
+ and label the result as an approximation.
161
+ """
162
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
163
+ out = df.copy()
164
+ game_minutes = 40.0 if level.lower() == "ncaa" else 48.0
165
+ assumed_pace = pace if pace is not None else (
166
+ DEFAULT_NCAA_PACE if level.lower() == "ncaa" else DEFAULT_NBA_PACE
167
+ )
168
+
169
+ required = list(BPM_LINEAR_COEFFICIENTS.keys())
170
+ hard_required = [k for k in required if k != "PF"]
171
+ require_columns(r, hard_required, "add_box_plus_minus_linear")
172
+
173
+ pace_source = "supplied possessions column"
174
+ if poss_col is not None and poss_col in out.columns:
175
+ possessions = pd.to_numeric(out[poss_col], errors="coerce")
176
+ else:
177
+ if minutes_col is not None and minutes_col in out.columns:
178
+ minutes = pd.to_numeric(out[minutes_col], errors="coerce")
179
+ elif r.has("MIN"):
180
+ minutes = pd.to_numeric(r.get(out, "MIN"), errors="coerce")
181
+ else:
182
+ raise KeyError(
183
+ "add_box_plus_minus_linear needs minutes (canonical MIN) to "
184
+ "estimate team possessions on the floor, or an explicit "
185
+ "poss_col. Supply one of them."
186
+ )
187
+ possessions = minutes * (assumed_pace / game_minutes)
188
+ pace_source = f"estimated from minutes at pace={assumed_pace}"
189
+
190
+ contribution = pd.Series(0.0, index=out.index)
191
+ used, skipped = [], []
192
+ for canonical, coefficient in BPM_LINEAR_COEFFICIENTS.items():
193
+ if not r.has(canonical):
194
+ skipped.append(canonical)
195
+ continue
196
+ contribution = contribution + coefficient * pd.to_numeric(
197
+ r.get(out, canonical), errors="coerce"
198
+ ).fillna(0.0)
199
+ used.append(canonical)
200
+
201
+ per_100 = safe_divide(contribution, possessions) * 100.0
202
+ raw = per_100 + BPM_LINEAR_INTERCEPT
203
+
204
+ offset = 0.0
205
+ if center:
206
+ weights = pd.to_numeric(
207
+ r.get(out, "MIN") if r.has("MIN") else pd.Series(1.0, index=out.index),
208
+ errors="coerce",
209
+ ).fillna(0.0)
210
+ finite = np.isfinite(raw) & (weights > 0)
211
+ if finite.any() and weights[finite].sum() > 0:
212
+ offset = float(np.average(raw[finite], weights=weights[finite]))
213
+ out[new_col] = raw - offset
214
+
215
+ if report is not None:
216
+ detail = (
217
+ f"Linear approximation of BPM 2.0. Terms used: {len(used)}. "
218
+ f"Possessions {pace_source}."
219
+ )
220
+ if center:
221
+ detail += f" Centred to zero mean (offset {offset:.2f})."
222
+ if skipped:
223
+ detail += f" Missing (treated as zero): {skipped}."
224
+ report.record(BPM_LINEAR_SPEC, Reliability.APPROXIMATED, detail)
225
+ return out
226
+
227
+
228
+ def add_vorp(
229
+ df: pd.DataFrame,
230
+ resolver: Optional[SchemaResolver] = None,
231
+ bpm_col: str = "BPM",
232
+ minutes_share_col: Optional[str] = None,
233
+ team_games: Optional[float] = None,
234
+ season_games: float = 32.0,
235
+ new_col: str = "VORP",
236
+ report: Optional[ComputationReport] = None,
237
+ ) -> pd.DataFrame:
238
+ """Value Over Replacement Player.
239
+
240
+ VORP = (BPM - (-2.0)) * (share of team minutes played) * (team games / season games)
241
+
242
+ Replacement level is -2.0 BPM (Myers); a full team of replacement players
243
+ wins roughly 14 NBA games. The minutes-share term converts a rate into a
244
+ volume statistic: a +4 BPM player who plays 10% of minutes contributes
245
+ far less than one who plays 80%.
246
+
247
+ Parameters
248
+ ----------
249
+ minutes_share_col : fraction (0-1) or percentage (0-100) of team minutes.
250
+ BartTorvik's `Min_per` (canonical MIN_PCT) is exactly this; if values
251
+ exceed 1.0 they are assumed to be percentages and divided by 100.
252
+ season_games : games in a full season (NCAA ~32, NBA 82).
253
+ """
254
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
255
+ out = df.copy()
256
+
257
+ if bpm_col not in out.columns:
258
+ if r.has("BPM"):
259
+ bpm = r.get(out, "BPM")
260
+ else:
261
+ raise KeyError(
262
+ f"add_vorp needs a BPM column ('{bpm_col}' not found and canonical "
263
+ f"BPM unresolvable). Compute add_box_plus_minus_linear first."
264
+ )
265
+ else:
266
+ bpm = pd.to_numeric(out[bpm_col], errors="coerce")
267
+
268
+ if minutes_share_col is not None and minutes_share_col in out.columns:
269
+ share = pd.to_numeric(out[minutes_share_col], errors="coerce")
270
+ elif r.has("MIN_PCT"):
271
+ share = pd.to_numeric(r.get(out, "MIN_PCT"), errors="coerce")
272
+ else:
273
+ raise KeyError(
274
+ "add_vorp needs a minutes-share column (canonical MIN_PCT, e.g. "
275
+ "BartTorvik's 'Min_per')."
276
+ )
277
+
278
+ if share.max(skipna=True) is not np.nan and share.max(skipna=True) > 1.0:
279
+ share = share / 100.0
280
+
281
+ games_factor = 1.0 if team_games is None else (team_games / season_games)
282
+ out[new_col] = (bpm - BPM_REPLACEMENT_LEVEL) * share * games_factor
283
+
284
+ if report is not None:
285
+ spec = MetricSpec(
286
+ name=new_col,
287
+ requires=("BPM", "MIN_PCT"),
288
+ source="Basketball-Reference, 'NBA Win Shares' / VORP definition",
289
+ description="Points above a -2.0 BPM replacement player, scaled by playing time.",
290
+ notes="Replacement level -2.0 BPM per Myers.",
291
+ )
292
+ report.record(spec, Reliability.APPROXIMATED if team_games is None else Reliability.EXACT)
293
+ return out
294
+
295
+
296
+ # ---------------------------------------------------------------------------
297
+ # Player Efficiency Rating (John Hollinger)
298
+ # ---------------------------------------------------------------------------
299
+
300
+
301
+ @dataclass(frozen=True)
302
+ class LeagueContext:
303
+ """League-wide totals required by PER's unadjusted stage.
304
+
305
+ PER is defined relative to league averages, so it simply cannot be
306
+ computed from player rows alone. Supply these from a season-level
307
+ aggregate (Sports-Reference publishes them per season).
308
+
309
+ Attributes are league TOTALS for the season, not per-game averages.
310
+ """
311
+
312
+ lg_pts: float
313
+ lg_fga: float
314
+ lg_fgm: float
315
+ lg_fta: float
316
+ lg_ftm: float
317
+ lg_ast: float
318
+ lg_orb: float
319
+ lg_trb: float
320
+ lg_tov: float
321
+ lg_pf: float
322
+ lg_pace: Optional[float] = None
323
+ label: str = ""
324
+
325
+ @property
326
+ def value_of_possession(self) -> float:
327
+ """VOP = lgPTS / (lgFGA - lgORB + lgTOV + 0.44*lgFTA)."""
328
+ denom = self.lg_fga - self.lg_orb + self.lg_tov + 0.44 * self.lg_fta
329
+ return self.lg_pts / denom if denom else 0.0
330
+
331
+ @property
332
+ def defensive_rebound_pct(self) -> float:
333
+ """DRB% = (lgTRB - lgORB) / lgTRB."""
334
+ return (self.lg_trb - self.lg_orb) / self.lg_trb if self.lg_trb else 0.7
335
+
336
+ @property
337
+ def factor(self) -> float:
338
+ """factor = 2/3 - (0.5 * (lgAST/lgFG)) / (2 * (lgFG/lgFT))."""
339
+ if not self.lg_fgm or not self.lg_ftm:
340
+ return 2.0 / 3.0
341
+ return (2.0 / 3.0) - (0.5 * (self.lg_ast / self.lg_fgm)) / (
342
+ 2.0 * (self.lg_fgm / self.lg_ftm)
343
+ )
344
+
345
+
346
+ PER_SPEC = MetricSpec(
347
+ name="PER",
348
+ requires=("MIN", "TPM", "AST", "FGM", "FGA", "FTM", "FTA", "TOV", "REB", "OREB", "STL", "BLK", "PF"),
349
+ requires_team=("TM_AST", "TM_FG"),
350
+ source="Hollinger, J.; formula per Basketball-Reference",
351
+ description="Per-minute production, pace-adjusted, normalised so league average = 15.00.",
352
+ notes=(
353
+ "Requires league totals (LeagueContext) AND team AST/FG. Widely "
354
+ "criticised as usage- and offence-biased with weak defensive content. "
355
+ "Useful as a feature, poor as a target."
356
+ ),
357
+ )
358
+
359
+
360
+ def add_player_efficiency_rating(
361
+ df: pd.DataFrame,
362
+ league: LeagueContext,
363
+ resolver: Optional[SchemaResolver] = None,
364
+ team_pace_col: Optional[str] = None,
365
+ normalise: bool = True,
366
+ new_col: str = "PER",
367
+ report: Optional[ComputationReport] = None,
368
+ ) -> pd.DataFrame:
369
+ """Hollinger's Player Efficiency Rating.
370
+
371
+ Three stages: unadjusted PER (uPER) -> pace-adjusted (aPER) -> normalised
372
+ so the minutes-weighted league average is exactly 15.00.
373
+
374
+ uPER (per Basketball-Reference)::
375
+
376
+ uPER = (1/MP) * [ 3P
377
+ + (2/3)*AST
378
+ + (2 - factor*(TmAST/TmFG))*FG
379
+ + FT*0.5*(1 + (1 - TmAST/TmFG) + (2/3)*(TmAST/TmFG))
380
+ - VOP*TOV
381
+ - VOP*DRB%*(FGA - FG)
382
+ - VOP*0.44*(0.44 + 0.56*DRB%)*(FTA - FT)
383
+ + VOP*(1 - DRB%)*(TRB - ORB)
384
+ + VOP*DRB%*ORB
385
+ + VOP*STL
386
+ + VOP*DRB%*BLK
387
+ - PF*((lgFT/lgPF) - 0.44*(lgFTA/lgPF)*VOP) ]
388
+
389
+ Parameters
390
+ ----------
391
+ league : LeagueContext with season-wide totals.
392
+ team_pace_col : if supplied (and league.lg_pace set), applies the pace
393
+ adjustment aPER = (lgPace/TmPace) * uPER. Omitted -> uPER is used and
394
+ the result is flagged APPROXIMATED.
395
+ normalise : rescale so the minutes-weighted mean is 15.00. Note this
396
+ normalises against *this dataset*, which equals the true league
397
+ average only if the dataset is the full league.
398
+ """
399
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
400
+ out = df.copy()
401
+
402
+ require_columns(
403
+ r,
404
+ ["MIN", "TPM", "AST", "FGM", "FGA", "FTM", "FTA", "TOV", "REB", "OREB", "STL", "BLK"],
405
+ "add_player_efficiency_rating",
406
+ )
407
+
408
+ vop = league.value_of_possession
409
+ drb_pct = league.defensive_rebound_pct
410
+ factor = league.factor
411
+
412
+ mp = pd.to_numeric(r.get(out, "MIN"), errors="coerce")
413
+ tpm = pd.to_numeric(r.get(out, "TPM"), errors="coerce").fillna(0.0)
414
+ ast = pd.to_numeric(r.get(out, "AST"), errors="coerce").fillna(0.0)
415
+ fgm = pd.to_numeric(r.get(out, "FGM"), errors="coerce").fillna(0.0)
416
+ fga = pd.to_numeric(r.get(out, "FGA"), errors="coerce").fillna(0.0)
417
+ ftm = pd.to_numeric(r.get(out, "FTM"), errors="coerce").fillna(0.0)
418
+ fta = pd.to_numeric(r.get(out, "FTA"), errors="coerce").fillna(0.0)
419
+ tov = pd.to_numeric(r.get(out, "TOV"), errors="coerce").fillna(0.0)
420
+ trb = pd.to_numeric(r.get(out, "REB"), errors="coerce").fillna(0.0)
421
+ orb = pd.to_numeric(r.get(out, "OREB"), errors="coerce").fillna(0.0)
422
+ stl = pd.to_numeric(r.get(out, "STL"), errors="coerce").fillna(0.0)
423
+ blk = pd.to_numeric(r.get(out, "BLK"), errors="coerce").fillna(0.0)
424
+ pf = pd.to_numeric(r.get(out, "PF"), errors="coerce").fillna(0.0) if r.has("PF") else pd.Series(0.0, index=out.index)
425
+
426
+ # Team assist ratio: exact if team totals present, else league average.
427
+ team_context_exact = not r.missing(["TM_AST", "TM_FG"])
428
+ if team_context_exact:
429
+ tm_ast_ratio = safe_divide(r.get(out, "TM_AST"), r.get(out, "TM_FG"))
430
+ else:
431
+ lg_ratio = league.lg_ast / league.lg_fgm if league.lg_fgm else 0.0
432
+ tm_ast_ratio = pd.Series(lg_ratio, index=out.index)
433
+
434
+ foul_term = (
435
+ (league.lg_ftm / league.lg_pf) - 0.44 * (league.lg_fta / league.lg_pf) * vop
436
+ if league.lg_pf
437
+ else 0.0
438
+ )
439
+
440
+ raw = (
441
+ tpm
442
+ + (2.0 / 3.0) * ast
443
+ + (2.0 - factor * tm_ast_ratio) * fgm
444
+ + ftm * 0.5 * (1.0 + (1.0 - tm_ast_ratio) + (2.0 / 3.0) * tm_ast_ratio)
445
+ - vop * tov
446
+ - vop * drb_pct * (fga - fgm)
447
+ - vop * 0.44 * (0.44 + 0.56 * drb_pct) * (fta - ftm)
448
+ + vop * (1.0 - drb_pct) * (trb - orb)
449
+ + vop * drb_pct * orb
450
+ + vop * stl
451
+ + vop * drb_pct * blk
452
+ - pf * foul_term
453
+ )
454
+ uper = safe_divide(raw, mp, fill=np.nan)
455
+
456
+ paced = uper
457
+ pace_applied = False
458
+ if team_pace_col is not None and team_pace_col in out.columns and league.lg_pace:
459
+ paced = uper * safe_divide(
460
+ pd.Series(league.lg_pace, index=out.index),
461
+ pd.to_numeric(out[team_pace_col], errors="coerce"),
462
+ fill=1.0,
463
+ )
464
+ pace_applied = True
465
+
466
+ if normalise:
467
+ weights = mp.fillna(0.0)
468
+ weighted_mean = (
469
+ np.average(paced.fillna(0.0), weights=weights) if weights.sum() > 0 else np.nan
470
+ )
471
+ out[new_col] = paced * (15.0 / weighted_mean) if weighted_mean else paced
472
+ else:
473
+ out[new_col] = paced
474
+
475
+ if report is not None:
476
+ reliability = (
477
+ Reliability.EXACT
478
+ if (team_context_exact and pace_applied)
479
+ else Reliability.APPROXIMATED
480
+ )
481
+ detail_bits = []
482
+ if not team_context_exact:
483
+ detail_bits.append("team AST/FG substituted with league average")
484
+ if not pace_applied:
485
+ detail_bits.append("no pace adjustment applied")
486
+ if normalise:
487
+ detail_bits.append("normalised against this dataset, not the true league")
488
+ report.record(PER_SPEC, reliability, "; ".join(detail_bits))
489
+ return out
490
+
491
+
492
+ GAME_SCORE_SPEC = MetricSpec(
493
+ name="GAME_SCORE",
494
+ requires=("PTS", "FGM", "FGA", "FTA", "FTM", "OREB", "DREB", "STL", "AST", "BLK", "TOV"),
495
+ source="Hollinger, J.; Basketball-Reference glossary",
496
+ description="Single-number box-score summary scaled roughly like points (10 = solid).",
497
+ notes="Simple and transparent; no team or league context required.",
498
+ )
499
+
500
+
501
+ def add_game_score(
502
+ df: pd.DataFrame,
503
+ resolver: Optional[SchemaResolver] = None,
504
+ new_col: str = "GAME_SCORE",
505
+ report: Optional[ComputationReport] = None,
506
+ ) -> pd.DataFrame:
507
+ """Hollinger's Game Score::
508
+
509
+ GmSc = PTS + 0.4*FG - 0.7*FGA - 0.4*(FTA - FT) + 0.7*ORB + 0.3*DRB
510
+ + STL + 0.7*AST + 0.7*BLK - 0.4*PF - TOV
511
+
512
+ Calibrated so ~10 is a solid outing and ~40 is outstanding. Fully
513
+ computable from an individual box score, which makes it a useful,
514
+ honest baseline when team context is unavailable.
515
+ """
516
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
517
+ out = df.copy()
518
+ require_columns(
519
+ r,
520
+ ["PTS", "FGM", "FGA", "FTA", "FTM", "OREB", "DREB", "STL", "AST", "BLK", "TOV"],
521
+ "add_game_score",
522
+ )
523
+
524
+ g = lambda name: pd.to_numeric(r.get(out, name), errors="coerce").fillna(0.0)
525
+ pf = g("PF") if r.has("PF") else pd.Series(0.0, index=out.index)
526
+
527
+ out[new_col] = (
528
+ g("PTS")
529
+ + 0.4 * g("FGM")
530
+ - 0.7 * g("FGA")
531
+ - 0.4 * (g("FTA") - g("FTM"))
532
+ + 0.7 * g("OREB")
533
+ + 0.3 * g("DREB")
534
+ + g("STL")
535
+ + 0.7 * g("AST")
536
+ + 0.7 * g("BLK")
537
+ - 0.4 * pf
538
+ - g("TOV")
539
+ )
540
+
541
+ if report is not None:
542
+ report.record(GAME_SCORE_SPEC, Reliability.EXACT)
543
+ return out
544
+
545
+
546
+ # ---------------------------------------------------------------------------
547
+ # Performance/Team Impact Estimator (NBA.com)
548
+ # ---------------------------------------------------------------------------
549
+
550
+ PIE_SPEC = MetricSpec(
551
+ name="PIE",
552
+ requires=("PTS", "FGM", "FGA", "FTM", "FTA", "DREB", "OREB", "AST", "STL", "BLK", "TOV"),
553
+ source="NBA.com Advanced Stats glossary, 'Performance Impact Estimator'",
554
+ description="Single-number estimate of a player's or team's overall statistical output in a game.",
555
+ notes=(
556
+ "PIE = PTS + FGM + FTM - FGA - FTA + DREB + 0.5*OREB + AST + STL "
557
+ "+ 0.5*BLK - PF - TOV. Unlike Game Score, it has no tuned "
558
+ "coefficients: it is a plain sum of positive box-score contributions "
559
+ "minus negative ones (missed shots, missed free throws, fouls, "
560
+ "turnovers). Mainly useful as the input to add_team_impact_estimator."
561
+ ),
562
+ )
563
+
564
+
565
+ def add_performance_impact_estimator(
566
+ df: pd.DataFrame,
567
+ resolver: Optional[SchemaResolver] = None,
568
+ new_col: str = "PIE",
569
+ report: Optional[ComputationReport] = None,
570
+ ) -> pd.DataFrame:
571
+ """Performance Impact Estimator (PIE)::
572
+
573
+ PIE = PTS + FGM + FTM - FGA - FTA + DREB + 0.5*OREB
574
+ + AST + STL + 0.5*BLK - PF - TOV
575
+
576
+ Fully computable from a single row's own box score (player or team); no
577
+ team or opponent context required. On its own PIE is not very meaningful,
578
+ since it has no fixed scale; it exists as the building block for
579
+ `add_team_impact_estimator`, which expresses it as a share of the total
580
+ production in the game.
581
+ """
582
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
583
+ out = df.copy()
584
+ require_columns(
585
+ r,
586
+ ["PTS", "FGM", "FGA", "FTM", "FTA", "DREB", "OREB", "AST", "STL", "BLK", "TOV"],
587
+ "add_performance_impact_estimator",
588
+ )
589
+
590
+ g = lambda name: pd.to_numeric(r.get(out, name), errors="coerce").fillna(0.0)
591
+ pf = g("PF") if r.has("PF") else pd.Series(0.0, index=out.index)
592
+
593
+ out[new_col] = (
594
+ g("PTS")
595
+ + g("FGM")
596
+ + g("FTM")
597
+ - g("FGA")
598
+ - g("FTA")
599
+ + g("DREB")
600
+ + 0.5 * g("OREB")
601
+ + g("AST")
602
+ + g("STL")
603
+ + 0.5 * g("BLK")
604
+ - pf
605
+ - g("TOV")
606
+ )
607
+
608
+ if report is not None:
609
+ report.record(PIE_SPEC, Reliability.EXACT)
610
+ return out
611
+
612
+
613
+ TIE_SPEC = MetricSpec(
614
+ name="TIE",
615
+ requires=("PTS", "FGM", "FGA", "FTM", "FTA", "DREB", "OREB", "AST", "STL", "BLK", "TOV"),
616
+ requires_team=(
617
+ "OPP_PTS", "OPP_FG", "OPP_FGA", "OPP_FTM", "OPP_FTA",
618
+ "OPP_DRB", "OPP_ORB", "OPP_AST", "OPP_STL", "OPP_BLK", "OPP_TOV",
619
+ ),
620
+ source="NBA.com Advanced Stats glossary, 'Team Impact Estimator'",
621
+ description="A player's or team's PIE as a share of the combined PIE produced by both sides.",
622
+ notes="TIE = PIE / (PIE + OppPIE). 0.50 is an even statistical split.",
623
+ )
624
+
625
+
626
+ def add_team_impact_estimator(
627
+ df: pd.DataFrame,
628
+ resolver: Optional[SchemaResolver] = None,
629
+ pie_col: Optional[str] = None,
630
+ new_col: str = "TIE",
631
+ report: Optional[ComputationReport] = None,
632
+ ) -> pd.DataFrame:
633
+ """Team Impact Estimator: TIE = PIE / (PIE + OppPIE).
634
+
635
+ Rescales PIE into a share (0-1) of the combined production in the game:
636
+ 0.50 is an even split, above it means winning the statistical battle.
637
+ **Requires the opponent's full box score**, since the opponent's PIE is
638
+ computed directly from OPP_PTS, OPP_FG, OPP_FGA, OPP_FTM, OPP_FTA,
639
+ OPP_DRB, OPP_ORB, OPP_AST, OPP_STL, OPP_BLK, OPP_TOV (and OPP_PF if
640
+ fouls are tracked), the same "eight factors"-style opponent context
641
+ `four_factors.py` uses for the defensive Four Factors.
642
+
643
+ Parameters
644
+ ----------
645
+ pie_col : an existing PIE column to reuse instead of recomputing it.
646
+ Defaults to computing PIE fresh via `add_performance_impact_estimator`.
647
+ """
648
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
649
+ out = df.copy()
650
+
651
+ if pie_col is not None and pie_col in out.columns:
652
+ pie = pd.to_numeric(out[pie_col], errors="coerce")
653
+ else:
654
+ out = add_performance_impact_estimator(out, resolver=r)
655
+ pie = out["PIE"]
656
+
657
+ opp_required = [
658
+ "OPP_PTS", "OPP_FG", "OPP_FGA", "OPP_FTM", "OPP_FTA",
659
+ "OPP_DRB", "OPP_ORB", "OPP_AST", "OPP_STL", "OPP_BLK", "OPP_TOV",
660
+ ]
661
+ require_columns(r, opp_required, "add_team_impact_estimator")
662
+
663
+ g = lambda name: pd.to_numeric(r.get(out, name), errors="coerce").fillna(0.0)
664
+ opp_pf = g("OPP_PF") if r.has("OPP_PF") else pd.Series(0.0, index=out.index)
665
+
666
+ opp_pie = (
667
+ g("OPP_PTS")
668
+ + g("OPP_FG")
669
+ + g("OPP_FTM")
670
+ - g("OPP_FGA")
671
+ - g("OPP_FTA")
672
+ + g("OPP_DRB")
673
+ + 0.5 * g("OPP_ORB")
674
+ + g("OPP_AST")
675
+ + g("OPP_STL")
676
+ + 0.5 * g("OPP_BLK")
677
+ - opp_pf
678
+ - g("OPP_TOV")
679
+ )
680
+
681
+ out[new_col] = safe_divide(pie, pie + opp_pie)
682
+
683
+ if report is not None:
684
+ report.record(TIE_SPEC, Reliability.EXACT)
685
+ return out
686
+
687
+
688
+ # ---------------------------------------------------------------------------
689
+ # Shooting efficiency and the usage/rate family
690
+ # ---------------------------------------------------------------------------
691
+
692
+ TS_SPEC = MetricSpec(
693
+ name="TS_PCT_CALC",
694
+ requires=("PTS", "FGA", "FTA"),
695
+ source="Basketball-Reference glossary; Oliver (2004)",
696
+ description="True shooting %: scoring efficiency including threes and free throws.",
697
+ notes="TS% = PTS / (2 * (FGA + c*FTA)). c = 0.44 NBA, 0.475 NCAA.",
698
+ )
699
+
700
+
701
+ def add_true_shooting_pct(
702
+ df: pd.DataFrame,
703
+ resolver: Optional[SchemaResolver] = None,
704
+ level: str = "ncaa",
705
+ new_col: str = "TS_PCT_CALC",
706
+ report: Optional[ComputationReport] = None,
707
+ ) -> pd.DataFrame:
708
+ """TS% = PTS / (2 * (FGA + c*FTA)).
709
+
710
+ The most complete single measure of scoring efficiency, since it accounts
711
+ for two-pointers, threes and free throws in one number. Prefer the
712
+ dataset's own TS% column when present.
713
+ """
714
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
715
+ out = df.copy()
716
+ c = fta_coefficient(level)
717
+ require_columns(r, ["PTS", "FGA", "FTA"], "add_true_shooting_pct")
718
+
719
+ pts, fga, fta = r.get(out, "PTS"), r.get(out, "FGA"), r.get(out, "FTA")
720
+ out[new_col] = safe_divide(pts, 2.0 * (fga + c * fta))
721
+
722
+ if report is not None:
723
+ report.record(TS_SPEC, Reliability.EXACT)
724
+ return out
725
+
726
+
727
+ USAGE_SPEC = MetricSpec(
728
+ name="USG_PCT_CALC",
729
+ requires=("FGA", "FTA", "TOV", "MIN"),
730
+ requires_team=("TM_MP", "TM_FGA", "TM_FTA", "TM_TOV"),
731
+ source="Basketball-Reference glossary",
732
+ description="Share of team possessions a player used while on the floor.",
733
+ notes="Requires team totals. Without them, use possessions-used per minute instead.",
734
+ )
735
+
736
+
737
+ def add_usage_rate(
738
+ df: pd.DataFrame,
739
+ resolver: Optional[SchemaResolver] = None,
740
+ level: str = "ncaa",
741
+ new_col: str = "USG_PCT_CALC",
742
+ report: Optional[ComputationReport] = None,
743
+ ) -> pd.DataFrame:
744
+ """Usage% = 100 * ((FGA + c*FTA + TOV) * (TmMP/5)) / (MP * (TmFGA + c*TmFTA + TmTOV)).
745
+
746
+ **Requires team totals.** BartTorvik supplies `usg` precomputed, so prefer
747
+ that. This exists for datasets that carry team aggregates.
748
+ """
749
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
750
+ out = df.copy()
751
+ c = fta_coefficient(level)
752
+ require_columns(
753
+ r, ["FGA", "FTA", "TOV", "MIN", "TM_MP", "TM_FGA", "TM_FTA", "TM_TOV"], "add_usage_rate"
754
+ )
755
+
756
+ player_poss = r.get(out, "FGA") + c * r.get(out, "FTA") + r.get(out, "TOV")
757
+ team_poss = r.get(out, "TM_FGA") + c * r.get(out, "TM_FTA") + r.get(out, "TM_TOV")
758
+ out[new_col] = 100.0 * safe_divide(
759
+ player_poss * (r.get(out, "TM_MP") / 5.0), r.get(out, "MIN") * team_poss
760
+ )
761
+
762
+ if report is not None:
763
+ report.record(USAGE_SPEC, Reliability.EXACT)
764
+ return out
765
+
766
+
767
+ def add_assist_pct(
768
+ df: pd.DataFrame,
769
+ resolver: Optional[SchemaResolver] = None,
770
+ new_col: str = "AST_PCT_CALC",
771
+ report: Optional[ComputationReport] = None,
772
+ ) -> pd.DataFrame:
773
+ """AST% = 100 * AST / (((MP / (TmMP/5)) * TmFG) - FG).
774
+
775
+ Share of teammate field goals a player assisted while on the floor.
776
+ **Requires team totals.**
777
+ """
778
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
779
+ out = df.copy()
780
+ require_columns(r, ["AST", "MIN", "FGM", "TM_MP", "TM_FG"], "add_assist_pct")
781
+
782
+ mp, tm_mp, tm_fg = r.get(out, "MIN"), r.get(out, "TM_MP"), r.get(out, "TM_FG")
783
+ denom = safe_divide(mp, tm_mp / 5.0) * tm_fg - r.get(out, "FGM")
784
+ out[new_col] = 100.0 * safe_divide(r.get(out, "AST"), denom)
785
+
786
+ if report is not None:
787
+ spec = MetricSpec(
788
+ name=new_col,
789
+ requires=("AST", "MIN", "FGM"),
790
+ requires_team=("TM_MP", "TM_FG"),
791
+ source="Basketball-Reference glossary",
792
+ description="Share of teammate field goals assisted while on court.",
793
+ )
794
+ report.record(spec, Reliability.EXACT)
795
+ return out
796
+
797
+
798
+ def add_rebound_pct(
799
+ df: pd.DataFrame,
800
+ resolver: Optional[SchemaResolver] = None,
801
+ which: str = "total",
802
+ new_col: Optional[str] = None,
803
+ report: Optional[ComputationReport] = None,
804
+ ) -> pd.DataFrame:
805
+ """Rebound percentage: share of available rebounds captured on court.
806
+
807
+ TRB% = 100 * (TRB * (TmMP/5)) / (MP * (TmTRB + OppTRB))
808
+ ORB% = 100 * (ORB * (TmMP/5)) / (MP * (TmORB + OppDRB))
809
+ DRB% = 100 * (DRB * (TmMP/5)) / (MP * (TmDRB + OppORB))
810
+
811
+ **Requires team and opponent totals.**
812
+
813
+ Parameters
814
+ ----------
815
+ which : {"total", "offensive", "defensive"}
816
+ """
817
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
818
+ out = df.copy()
819
+
820
+ config = {
821
+ "total": ("REB", ["TM_TRB", "OPP_TRB"], "TRB_PCT_CALC"),
822
+ "offensive": ("OREB", ["TM_ORB", "OPP_DRB"], "ORB_PCT_CALC"),
823
+ "defensive": ("DREB", ["TM_DRB", "OPP_ORB"], "DRB_PCT_CALC"),
824
+ }
825
+ if which not in config:
826
+ raise ValueError(f"which must be one of {sorted(config)}, got {which!r}")
827
+
828
+ stat, team_terms, default_name = config[which]
829
+ target = new_col or default_name
830
+ require_columns(r, [stat, "MIN", "TM_MP"] + team_terms, f"add_rebound_pct({which})")
831
+
832
+ available = r.get(out, team_terms[0]) + r.get(out, team_terms[1])
833
+ out[target] = 100.0 * safe_divide(
834
+ r.get(out, stat) * (r.get(out, "TM_MP") / 5.0), r.get(out, "MIN") * available
835
+ )
836
+
837
+ if report is not None:
838
+ spec = MetricSpec(
839
+ name=target,
840
+ requires=(stat, "MIN"),
841
+ requires_team=tuple(["TM_MP"] + team_terms),
842
+ source="Basketball-Reference glossary; Oliver (2004)",
843
+ description=f"Share of available {which} rebounds captured while on court.",
844
+ )
845
+ report.record(spec, Reliability.EXACT)
846
+ return out
847
+
848
+
849
+ def add_steal_pct(
850
+ df: pd.DataFrame,
851
+ resolver: Optional[SchemaResolver] = None,
852
+ new_col: str = "STL_PCT_CALC",
853
+ report: Optional[ComputationReport] = None,
854
+ ) -> pd.DataFrame:
855
+ """STL% = 100 * (STL * (TmMP/5)) / (MP * OppPoss).
856
+
857
+ Share of opponent possessions ending in a steal by this player.
858
+ **Requires team minutes and opponent possessions.**
859
+
860
+ Worth computing carefully: steal rate is one of the strongest known
861
+ college-to-NBA predictors (see `draft.py` and Vashro's work), and Myers'
862
+ BPM assigns it the largest positive coefficient of any box-score term.
863
+ """
864
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
865
+ out = df.copy()
866
+ require_columns(r, ["STL", "MIN", "TM_MP", "OPP_POSS"], "add_steal_pct")
867
+
868
+ out[new_col] = 100.0 * safe_divide(
869
+ r.get(out, "STL") * (r.get(out, "TM_MP") / 5.0),
870
+ r.get(out, "MIN") * r.get(out, "OPP_POSS"),
871
+ )
872
+
873
+ if report is not None:
874
+ spec = MetricSpec(
875
+ name=new_col,
876
+ requires=("STL", "MIN"),
877
+ requires_team=("TM_MP", "OPP_POSS"),
878
+ source="Basketball-Reference glossary",
879
+ description="Share of opponent possessions ended by a steal.",
880
+ notes="Strong NBA-success predictor in the draft literature.",
881
+ )
882
+ report.record(spec, Reliability.EXACT)
883
+ return out
884
+
885
+
886
+ def add_block_pct(
887
+ df: pd.DataFrame,
888
+ resolver: Optional[SchemaResolver] = None,
889
+ new_col: str = "BLK_PCT_CALC",
890
+ report: Optional[ComputationReport] = None,
891
+ ) -> pd.DataFrame:
892
+ """BLK% = 100 * (BLK * (TmMP/5)) / (MP * (OppFGA - Opp3PA)).
893
+
894
+ Share of opponent two-point attempts blocked. **Requires team minutes and
895
+ opponent shooting totals.** Block rate carries much of height's predictive
896
+ value for interior prospects.
897
+ """
898
+ r = resolver if resolver is not None else SchemaResolver(df.columns)
899
+ out = df.copy()
900
+ require_columns(r, ["BLK", "MIN", "TM_MP", "OPP_FGA", "OPP_TPA"], "add_block_pct")
901
+
902
+ two_pt_attempts = r.get(out, "OPP_FGA") - r.get(out, "OPP_TPA")
903
+ out[new_col] = 100.0 * safe_divide(
904
+ r.get(out, "BLK") * (r.get(out, "TM_MP") / 5.0), r.get(out, "MIN") * two_pt_attempts
905
+ )
906
+
907
+ if report is not None:
908
+ spec = MetricSpec(
909
+ name=new_col,
910
+ requires=("BLK", "MIN"),
911
+ requires_team=("TM_MP", "OPP_FGA", "OPP_TPA"),
912
+ source="Basketball-Reference glossary",
913
+ description="Share of opponent two-point attempts blocked.",
914
+ )
915
+ report.record(spec, Reliability.EXACT)
916
+ return out