UFCData 0.1.3__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ufcdata-0.1.3 → ufcdata-0.3.0}/PKG-INFO +2 -1
- ufcdata-0.3.0/README.md +216 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/UFCData.egg-info/PKG-INFO +2 -1
- {ufcdata-0.1.3 → ufcdata-0.3.0}/UFCData.egg-info/SOURCES.txt +3 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/UFCData.egg-info/requires.txt +1 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/pyproject.toml +2 -1
- ufcdata-0.3.0/ufcdata/__init__.py +4 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/ufcdata/data.py +14 -0
- ufcdata-0.3.0/ufcdata/fighter.py +701 -0
- ufcdata-0.3.0/ufcdata/ratings.py +130 -0
- ufcdata-0.3.0/ufcdata/search.py +39 -0
- ufcdata-0.1.3/README.md +0 -0
- ufcdata-0.1.3/ufcdata/__init__.py +0 -1
- {ufcdata-0.1.3 → ufcdata-0.3.0}/LICENSE.md +0 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/UFCData.egg-info/dependency_links.txt +0 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/UFCData.egg-info/top_level.txt +0 -0
- {ufcdata-0.1.3 → ufcdata-0.3.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: UFCData
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Obtain UFC data and functions to manipulate it
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
License-File: LICENSE.md
|
|
@@ -8,4 +8,5 @@ Requires-Dist: huggingface-hub
|
|
|
8
8
|
Requires-Dist: pandas
|
|
9
9
|
Requires-Dist: numpy
|
|
10
10
|
Requires-Dist: glicko2
|
|
11
|
+
Requires-Dist: rapidfuzz
|
|
11
12
|
Dynamic: license-file
|
ufcdata-0.3.0/README.md
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
# UFCData
|
|
2
|
+
|
|
3
|
+
An open-source Python package for accessing, manipulating, and analyzing UFC data.
|
|
4
|
+
|
|
5
|
+
https://pypi.org/project/UFCData/
|
|
6
|
+
|
|
7
|
+
add helper functions like convert odds
|
|
8
|
+
|
|
9
|
+
## Table of Contents
|
|
10
|
+
|
|
11
|
+
- [Installation](#installation)
|
|
12
|
+
|
|
13
|
+
Accessing Data
|
|
14
|
+
- [Loading Data](#loading-data)
|
|
15
|
+
- [Search](#search)
|
|
16
|
+
|
|
17
|
+
Rating Function
|
|
18
|
+
- [Elo](#elo)
|
|
19
|
+
|
|
20
|
+
Fighter Functions
|
|
21
|
+
- [Fighter History](#elo)
|
|
22
|
+
- [Fighter Statistic](#elo)
|
|
23
|
+
|
|
24
|
+
Helper Functions
|
|
25
|
+
- [Odds Convert](#elo)
|
|
26
|
+
- [Weight Convert](#elo)
|
|
27
|
+
- [Gender Convert](#elo)
|
|
28
|
+
- [Mirror](#elo)
|
|
29
|
+
|
|
30
|
+
Model Evaluation
|
|
31
|
+
- [Odds Baseline Example](#elo)
|
|
32
|
+
|
|
33
|
+
Data Sources
|
|
34
|
+
- [Online Sources](#online-sources)
|
|
35
|
+
- [Discrepancies in Data](#discrepancies-in-data)
|
|
36
|
+
- [Update Frequency](#update-frequency)
|
|
37
|
+
|
|
38
|
+
## Installation
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
pip install ufcdata
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Loading Data
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
import ufcdata as ufc
|
|
48
|
+
data = ufc.get_data()
|
|
49
|
+
|
|
50
|
+
## Obtain fighter_bio dataframe
|
|
51
|
+
fighter_bio = data["fighter_bio"].copy()
|
|
52
|
+
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Note that data in here and other functions below reference data that is available prior to the data.
|
|
56
|
+
|
|
57
|
+
## Search
|
|
58
|
+
|
|
59
|
+
Due to various event naming conventions and fighters sharing the same name, the primary keys for the dataframes are links, which can be difficult for humans to interpret. The `search()` function uses fuzzy string matching to make it easier to find fighters, events, and other records.
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
search(data, column, query, matches=1)
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
### Parameters
|
|
66
|
+
|
|
67
|
+
* `data` — The dataframe you are searching.
|
|
68
|
+
* `column` — The name of the column to search.
|
|
69
|
+
* `query` — The name or text you are searching for.
|
|
70
|
+
* `matches` — The number of closest matches to return. Defaults to `1`.
|
|
71
|
+
|
|
72
|
+
### Example
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
john_jones = ufc.search(fighter_bio, "Name", "Jon Jones", 3)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
This searches the `"Name"` column of the `fighter_bio` dataframe for the three closest matches to `"Jon Jones"`.
|
|
79
|
+
|
|
80
|
+
The results are returned as a dataframe, with the closest match appearing first.
|
|
81
|
+
|
|
82
|
+
### Returns
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
```text
|
|
86
|
+
Name Nickname Height Weight Reach Stance Total W Total L Fighter Link
|
|
87
|
+
Jon Jones Bones 76.0 248.0 84.0 Orthodox 28 1 ufcstats.com/...
|
|
88
|
+
Roshaun Jones NaN 68.0 135.0 NaN NaN 2 6 ufcstats.com/...
|
|
89
|
+
Mason Jones The Dragon 70.0 155.0 74.0 Orthodox 18 2 ufcstats.com/...
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
This is useful when the exact value stored in the dataset is unknown or when there are multiple similar names. The same function can be used with any dataframe and column containing searchable text.
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
## Elo
|
|
96
|
+
|
|
97
|
+
UFCData provides an Elo rating system for calculating fighter ratings based on their previous fight results. Ratings are calculated chronologically, with each fighter's rating recorded **immediately before each fight**.
|
|
98
|
+
|
|
99
|
+
This allows Elo ratings to be used as features for predictive modeling without incorporating information from the fight being predicted.
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
ufc.get_elo(fight_df, fighter_df, r=1500, k=30, s=400)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### Parameters
|
|
107
|
+
|
|
108
|
+
* `fight_df` — The UFC fight dataframe, sorted from most recent to least recent.
|
|
109
|
+
* `fighter_df` — The fighter information dataframe containing a `"Fighter Link"` column.
|
|
110
|
+
* `r` — Initial Elo rating for each fighter. Defaults to `1500`.
|
|
111
|
+
* `k` — K-factor controlling how much ratings change after each fight. Defaults to `30`.
|
|
112
|
+
* `s` — Scaling factor used when calculating expected scores. Defaults to `400`.
|
|
113
|
+
|
|
114
|
+
### Returns
|
|
115
|
+
|
|
116
|
+
The function returns a tuple containing:
|
|
117
|
+
|
|
118
|
+
1. `fighter_elo` — A dictionary mapping each fighter's UFCStats link to their final Elo rating.
|
|
119
|
+
2. `elo` — The original `fights_df` dataframe with `"Fighter 1 Elo"` and `"Fighter 2 Elo"` columns added. These columns contain each fighter's Elo rating immediately prior to the corresponding fight.
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
### Example
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
import ufcdata as ufc
|
|
126
|
+
|
|
127
|
+
data = ufc.get_data()
|
|
128
|
+
|
|
129
|
+
past_fights = data["past_fights"].copy()
|
|
130
|
+
fighter_bio = data["fighter_bio"].copy()
|
|
131
|
+
|
|
132
|
+
current_elo, past_elo = ufc.get_elo(
|
|
133
|
+
past_fights,
|
|
134
|
+
fighter_bio
|
|
135
|
+
)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
The resulting `fight_elo` dataframe can then be used to examine or incorporate pre-fight Elo ratings into analysis and predictive models.
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
### Important
|
|
142
|
+
|
|
143
|
+
`fight_df` should be sorted from **most recent to least recent** before being passed to `get_elo()`. The function reverses the dataframe internally to process fights chronologically and returns the resulting data in the original order.
|
|
144
|
+
|
|
145
|
+
## Fighter History
|
|
146
|
+
|
|
147
|
+
The `get_fighter_history()` function retrieves all UFC fights for a specific fighter and formats the results from the fighter's perspective. The requested fighter is always represented as `"Fighter 1"`, regardless of which side of the original fight dataframe they appeared on.
|
|
148
|
+
|
|
149
|
+
The function also calculates the age of both fighters at the time of each fight and assigns a `"UFC Fight"` number to each fight.
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
ufc.get_fighter_history(fighter_link, fighter_bio, fights_df)
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### Parameters
|
|
157
|
+
|
|
158
|
+
* `fighter_link` — The UFCStats link identifying the fighter.
|
|
159
|
+
* `fighter_bio` — The fighter information dataframe returned by `get_data()`.
|
|
160
|
+
* `fights_df` — The fight dataframe containing UFC fight history.
|
|
161
|
+
|
|
162
|
+
### Returns
|
|
163
|
+
|
|
164
|
+
A dataframe containing the fighter's fight history from most recent to least recent. The requested fighter is always `"Fighter 1"`, with fighter ages and `"UFC Fight"` numbers included.
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
### Example
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
import ufcdata as ufc
|
|
172
|
+
|
|
173
|
+
data = ufc.get_data()
|
|
174
|
+
|
|
175
|
+
fighter_bio = data["fighter_bio"].copy()
|
|
176
|
+
fights_df = data["past_fights"].copy()
|
|
177
|
+
|
|
178
|
+
# Obtain Jon Jones' UFCStats link
|
|
179
|
+
jon_jones = ufc.search(
|
|
180
|
+
fighter_bio,
|
|
181
|
+
"Name",
|
|
182
|
+
"Jon Jones"
|
|
183
|
+
).iloc[0]["Fighter Link"]
|
|
184
|
+
|
|
185
|
+
# Get Jon Jones' fight history
|
|
186
|
+
jon_jones_history = ufc.get_fighter_history(
|
|
187
|
+
jon_jones,
|
|
188
|
+
fighter_bio,
|
|
189
|
+
fights_df
|
|
190
|
+
)
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
This function is useful for analyzing an individual fighter's career, constructing fighter-level features, or preparing historical data for predictive modeling.
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
## Online Sources
|
|
197
|
+
|
|
198
|
+
Odds and birthplace data was obtained from https://www.tapology.com
|
|
199
|
+
|
|
200
|
+
Venue and attendance data was obtained from https://en.wikipedia.org/wiki/List_of_UFC_events
|
|
201
|
+
|
|
202
|
+
All other data was obtained from http://ufcstats.com
|
|
203
|
+
|
|
204
|
+
## Discrepancies in Data
|
|
205
|
+
|
|
206
|
+
UFCStats is treated as the authoritative source for UFCData. When discrepancies exist between UFCStats and other sources, such as Wikipedia or Tapology, the UFCStats data is used.
|
|
207
|
+
|
|
208
|
+
The UFCStats completed events page serves as the authoritative source for event, fight, and round data. Individual fighter profiles may contain fights from organizations or events that are not included in the completed events database, including WEC, Strikeforce, and PRIDE. These events are therefore excluded from UFCData.
|
|
209
|
+
|
|
210
|
+
## Update Frequency
|
|
211
|
+
|
|
212
|
+
The dataset is updated at the start of the scheduled broadcast time for each event. This update captures changes to betting odds, as well as any cancelled, postponed, or otherwise modified fights.
|
|
213
|
+
|
|
214
|
+
A second update occurs 24 hours after the start of the broadcast. This update captures the finalized event, fight, and round data, as well as information on upcoming events and fights.
|
|
215
|
+
|
|
216
|
+
Changes occurring between these scheduled updates are not automatically captured. Users are responsible for manually updating the dataset if they require the most current data for analysis or prediction.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: UFCData
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Obtain UFC data and functions to manipulate it
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
License-File: LICENSE.md
|
|
@@ -8,4 +8,5 @@ Requires-Dist: huggingface-hub
|
|
|
8
8
|
Requires-Dist: pandas
|
|
9
9
|
Requires-Dist: numpy
|
|
10
10
|
Requires-Dist: glicko2
|
|
11
|
+
Requires-Dist: rapidfuzz
|
|
11
12
|
Dynamic: license-file
|
|
@@ -8,6 +8,9 @@ UFCData.egg-info/requires.txt
|
|
|
8
8
|
UFCData.egg-info/top_level.txt
|
|
9
9
|
ufcdata/__init__.py
|
|
10
10
|
ufcdata/data.py
|
|
11
|
+
ufcdata/fighter.py
|
|
12
|
+
ufcdata/ratings.py
|
|
13
|
+
ufcdata/search.py
|
|
11
14
|
ufcdata.egg-info/PKG-INFO
|
|
12
15
|
ufcdata.egg-info/SOURCES.txt
|
|
13
16
|
ufcdata.egg-info/dependency_links.txt
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "UFCData"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.3.0"
|
|
4
4
|
description = "Obtain UFC data and functions to manipulate it"
|
|
5
5
|
requires-python = ">=3.10"
|
|
6
6
|
|
|
@@ -9,4 +9,5 @@ dependencies = [
|
|
|
9
9
|
"pandas",
|
|
10
10
|
"numpy",
|
|
11
11
|
"glicko2",
|
|
12
|
+
"rapidfuzz",
|
|
12
13
|
]
|
|
@@ -3,6 +3,20 @@ import pickle
|
|
|
3
3
|
import logging
|
|
4
4
|
|
|
5
5
|
def get_data():
|
|
6
|
+
"""
|
|
7
|
+
Obtains all UFC related data
|
|
8
|
+
|
|
9
|
+
Returns
|
|
10
|
+
dict
|
|
11
|
+
Dictionary containing UFC data as pandas DataFrames.
|
|
12
|
+
Individual DataFrames can be accessed using their
|
|
13
|
+
corresponding dictionary keys.
|
|
14
|
+
|
|
15
|
+
Examples
|
|
16
|
+
--------
|
|
17
|
+
>>> data = get_data()
|
|
18
|
+
>>> past_events = data["past_events"].copy()
|
|
19
|
+
"""
|
|
6
20
|
|
|
7
21
|
logging.getLogger("huggingface_hub").setLevel(logging.ERROR)
|
|
8
22
|
|
|
@@ -0,0 +1,701 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
import numpy as np
|
|
3
|
+
from datetime import date
|
|
4
|
+
from .ratings import get_elo, expected_score, update_rating
|
|
5
|
+
|
|
6
|
+
def _mirror_helper(df, cols_1, cols_2, in_order=True):
|
|
7
|
+
df = df.copy()
|
|
8
|
+
df_original = df.copy()
|
|
9
|
+
|
|
10
|
+
# Swap each pair of columns
|
|
11
|
+
for c1, c2 in zip(cols_1, cols_2):
|
|
12
|
+
temp = df[c1].copy()
|
|
13
|
+
df[c1] = df[c2]
|
|
14
|
+
df[c2] = temp
|
|
15
|
+
|
|
16
|
+
if not in_order:
|
|
17
|
+
return pd.concat([df_original, df], ignore_index=True)
|
|
18
|
+
|
|
19
|
+
rows = []
|
|
20
|
+
for r1, r2 in zip(df_original.itertuples(index=False), df.itertuples(index=False)):
|
|
21
|
+
rows.append(r1)
|
|
22
|
+
rows.append(r2)
|
|
23
|
+
|
|
24
|
+
return pd.DataFrame(rows, columns=df.columns)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def get_fighter_history(fighter_link, fighter_bio, fights_df):
|
|
28
|
+
"""
|
|
29
|
+
Retrieves the fight history of a UFC fighter.
|
|
30
|
+
|
|
31
|
+
The returned DataFrame contains all fights involving the specified
|
|
32
|
+
fighter, with the fighter consistently represented as "Fighter 1".
|
|
33
|
+
The function also calculates the age of both fighters at the time
|
|
34
|
+
of each fight and assigns a chronological UFC fight number.
|
|
35
|
+
|
|
36
|
+
Parameters
|
|
37
|
+
----------
|
|
38
|
+
fighter_link : str
|
|
39
|
+
UFCStats link identifying the fighter whose history is being
|
|
40
|
+
retrieved.
|
|
41
|
+
fighter_bio : pandas.DataFrame
|
|
42
|
+
DataFrame containing fighter information. Must contain
|
|
43
|
+
"Fighter Link" and "DOB" columns.
|
|
44
|
+
fights_df : pandas.DataFrame
|
|
45
|
+
DataFrame containing UFC fight data. Must contain fighter links,
|
|
46
|
+
fight dates, outcomes, and the other fight information required
|
|
47
|
+
by the function.
|
|
48
|
+
|
|
49
|
+
Returns
|
|
50
|
+
-------
|
|
51
|
+
pandas.DataFrame
|
|
52
|
+
DataFrame containing the fighter's complete fight history.
|
|
53
|
+
The requested fighter is represented as "Fighter 1" in every
|
|
54
|
+
row. The DataFrame includes the ages of both fighters at the
|
|
55
|
+
time of each fight and a "UFC Fight" column numbering the
|
|
56
|
+
fighter's fights chronologically.
|
|
57
|
+
|
|
58
|
+
Notes
|
|
59
|
+
-----
|
|
60
|
+
Fighter ages are calculated using 365.25 days per year. The
|
|
61
|
+
"UFC Fight" column counts fights from the most recent fight
|
|
62
|
+
backwards, with the most recent fight numbered 1.
|
|
63
|
+
|
|
64
|
+
Examples
|
|
65
|
+
--------
|
|
66
|
+
>>> fighter_history = get_fighter_history(
|
|
67
|
+
... fighter_link,
|
|
68
|
+
... fighter_bio,
|
|
69
|
+
... fights_df
|
|
70
|
+
... )
|
|
71
|
+
>>> fighter_history[["Date", "UFC Fight", "Fighter 1",
|
|
72
|
+
... "Fighter 1 Age"]].head()
|
|
73
|
+
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
fighter_history = fights_df[
|
|
77
|
+
(fights_df["Fighter 1 Link"] == fighter_link) |
|
|
78
|
+
(fights_df["Fighter 2 Link"] == fighter_link)]
|
|
79
|
+
|
|
80
|
+
fighter_history = _mirror_helper(fighter_history, ['Fighter 1', 'Fighter 1 Odds', 'Fighter 1 Link',
|
|
81
|
+
'Fighter 1 Outcome', 'Fighter 1 Bonus'],['Fighter 2', 'Fighter 2 Odds',
|
|
82
|
+
'Fighter 2 Link', 'Fighter 2 Outcome', 'Fighter 2 Bonus'])
|
|
83
|
+
|
|
84
|
+
fighter_history = fighter_history[
|
|
85
|
+
(fighter_history["Fighter 1 Link"] == fighter_link)]
|
|
86
|
+
|
|
87
|
+
fighter_history = fighter_history.merge(
|
|
88
|
+
fighter_bio[["Fighter Link", "DOB"]],
|
|
89
|
+
left_on="Fighter 1 Link",
|
|
90
|
+
right_on="Fighter Link",
|
|
91
|
+
how="left")
|
|
92
|
+
fighter_history = fighter_history.rename(columns={"DOB": "Fighter 1 Age"})
|
|
93
|
+
fighter_history = fighter_history.drop(columns=["Fighter Link"])
|
|
94
|
+
fighter_history["Fighter 1 Age"] = (pd.to_datetime(fighter_history["Date"]) - pd.to_datetime(fighter_history["Fighter 1 Age"])).dt.days / 365.25
|
|
95
|
+
|
|
96
|
+
fighter_history = fighter_history.merge(
|
|
97
|
+
fighter_bio[["Fighter Link", "DOB"]],
|
|
98
|
+
left_on="Fighter 2 Link",
|
|
99
|
+
right_on="Fighter Link",
|
|
100
|
+
how="left")
|
|
101
|
+
fighter_history = fighter_history.rename(columns={"DOB": "Fighter 2 Age"})
|
|
102
|
+
fighter_history = fighter_history.drop(columns=["Fighter Link"])
|
|
103
|
+
fighter_history["Fighter 2 Age"] = (pd.to_datetime(fighter_history["Date"]) - pd.to_datetime(fighter_history["Fighter 2 Age"])).dt.days / 365.25
|
|
104
|
+
fighter_history["UFC Fight"] = range(len(fighter_history), 0, -1)
|
|
105
|
+
fighter_history = fighter_history[['Date', "UFC Fight", 'Event Link', 'Fight Link', 'Weight Class',
|
|
106
|
+
'Gender', 'Title', 'Fighter 1', 'Fighter 1 Odds', 'Fighter 1 Age', 'Fighter 1 Link',
|
|
107
|
+
'Fighter 1 Outcome', 'Fighter 1 Bonus', 'Fighter 2', 'Fighter 2 Odds','Fighter 2 Age',
|
|
108
|
+
'Fighter 2 Link', 'Fighter 2 Outcome', 'Fighter 2 Bonus', 'Method',
|
|
109
|
+
'Round', 'Time', 'Time Format', 'Referee', 'Details']]
|
|
110
|
+
|
|
111
|
+
return fighter_history
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def get_fighter_statistic(fighter_link, fighter_bio, fights_df, rounds_df, future_df,
|
|
116
|
+
r=1500, k=30, s=400):
|
|
117
|
+
|
|
118
|
+
"""
|
|
119
|
+
Generate historical and current statistics for a UFC fighter.
|
|
120
|
+
|
|
121
|
+
The function retrieves the fighter's UFC fight history and constructs
|
|
122
|
+
fight-level statistics from the fighter's perspective. It calculates
|
|
123
|
+
the fighter's UFC record, win rate, Elo rating, cumulative fight time,
|
|
124
|
+
striking statistics, takedown statistics, and submission averages using
|
|
125
|
+
only information available before each fight.
|
|
126
|
+
|
|
127
|
+
For fighters with no previous UFC fights, a baseline row is returned
|
|
128
|
+
with an initial Elo rating and zero UFC wins, losses, draws, and
|
|
129
|
+
no-contests. Historical performance statistics are left as NaN because
|
|
130
|
+
no prior fight data is available.
|
|
131
|
+
|
|
132
|
+
Parameters
|
|
133
|
+
----------
|
|
134
|
+
fighter_link : str
|
|
135
|
+
URL or unique identifier for the fighter.
|
|
136
|
+
fighter_bio : pandas.DataFrame
|
|
137
|
+
DataFrame containing fighter biographical information, including
|
|
138
|
+
fighter names and links.
|
|
139
|
+
fights_df : pandas.DataFrame
|
|
140
|
+
DataFrame containing UFC fight-level results.
|
|
141
|
+
rounds_df : pandas.DataFrame
|
|
142
|
+
DataFrame containing round-level UFC statistics.
|
|
143
|
+
future_df : pandas.DataFrame
|
|
144
|
+
DataFrame containing the fighter's upcoming fight. The first row
|
|
145
|
+
is used to determine the date for which current statistics are
|
|
146
|
+
calculated.
|
|
147
|
+
r : float, default=1500
|
|
148
|
+
Initial Elo rating assigned to fighters with no previous Elo history.
|
|
149
|
+
k : float, default=30
|
|
150
|
+
Elo update factor.
|
|
151
|
+
s : float, default=400
|
|
152
|
+
Elo scaling factor used when calculating expected scores.
|
|
153
|
+
|
|
154
|
+
Returns
|
|
155
|
+
-------
|
|
156
|
+
current_statistic : pandas.DataFrame
|
|
157
|
+
One-row DataFrame containing the fighter's statistics immediately
|
|
158
|
+
before the upcoming fight. Historical statistics are calculated
|
|
159
|
+
using only fights occurring before the upcoming fight date.
|
|
160
|
+
|
|
161
|
+
past_statistic : pandas.DataFrame
|
|
162
|
+
DataFrame containing the fighter's historical statistics for each
|
|
163
|
+
previous UFC fight, with statistics calculated using only fights
|
|
164
|
+
occurring before each respective fight.
|
|
165
|
+
|
|
166
|
+
Notes
|
|
167
|
+
-----
|
|
168
|
+
The function calculates cumulative statistics rather than statistics
|
|
169
|
+
from a single fight. This prevents information from a future fight from
|
|
170
|
+
being used when generating features for an earlier fight.
|
|
171
|
+
|
|
172
|
+
Historical statistics include:
|
|
173
|
+
- UFC record and win rate
|
|
174
|
+
- Elo rating
|
|
175
|
+
- Total fight time
|
|
176
|
+
- Significant strikes landed per minute (SLpM)
|
|
177
|
+
- Significant strike accuracy and defense
|
|
178
|
+
- Significant strikes absorbed per minute (SApM)
|
|
179
|
+
- Takedown average, accuracy, and defense
|
|
180
|
+
- Submission attempts per 15 minutes
|
|
181
|
+
|
|
182
|
+
Fighters with no previous UFC fights receive an initial Elo rating of
|
|
183
|
+
`r`, while statistics requiring historical fight data are set to NaN.
|
|
184
|
+
"""
|
|
185
|
+
|
|
186
|
+
record = get_fighter_history(fighter_link, fighter_bio, fights_df)
|
|
187
|
+
|
|
188
|
+
date = future_df.iloc[0]["Date"]
|
|
189
|
+
date = pd.to_datetime(date).date()
|
|
190
|
+
|
|
191
|
+
if len(record) == 0:
|
|
192
|
+
columns = [
|
|
193
|
+
'Date', 'UFC Fight', 'Weight Class', 'Gender', 'Title', 'Fighter',
|
|
194
|
+
'Fighter Link', 'Fighter Outcome', 'UFC W', 'UFC L', 'UFC D', 'UFC NC',
|
|
195
|
+
'UFC Win Rate', 'Fighter Elo',
|
|
196
|
+
'Fight Time', 'SLpM', 'Str Acc', 'SApM',
|
|
197
|
+
'Str Def', 'TD Avg', 'TD Acc', 'TD Def', 'Sub Avg'
|
|
198
|
+
]
|
|
199
|
+
|
|
200
|
+
fighter_name = fighter_bio.loc[
|
|
201
|
+
fighter_bio["Fighter Link"] == fighter_link,
|
|
202
|
+
"Name"
|
|
203
|
+
].iloc[0]
|
|
204
|
+
|
|
205
|
+
df = pd.DataFrame([{
|
|
206
|
+
"Date": date,
|
|
207
|
+
"UFC Fight": 1,
|
|
208
|
+
"Weight Class": None,
|
|
209
|
+
"Gender": None,
|
|
210
|
+
"Title": None,
|
|
211
|
+
"Fighter": fighter_name,
|
|
212
|
+
"Fighter Link": fighter_link,
|
|
213
|
+
"Fighter Outcome": None,
|
|
214
|
+
"UFC W": 0,
|
|
215
|
+
"UFC L": 0,
|
|
216
|
+
"UFC D": 0,
|
|
217
|
+
"UFC NC": 0,
|
|
218
|
+
"UFC Win Rate": None,
|
|
219
|
+
"Fighter Elo": r,
|
|
220
|
+
"Fight Time": None,
|
|
221
|
+
"SLpM": None,
|
|
222
|
+
"Str Acc": None,
|
|
223
|
+
"SApM": None,
|
|
224
|
+
"Str Def": None,
|
|
225
|
+
"TD Avg": None,
|
|
226
|
+
"TD Acc": None,
|
|
227
|
+
"TD Def": None,
|
|
228
|
+
"Sub Avg": None
|
|
229
|
+
}], columns=columns)
|
|
230
|
+
|
|
231
|
+
df = df.astype(object).where(pd.notna(df), np.nan)
|
|
232
|
+
|
|
233
|
+
df["UFC Fight"] = df["UFC Fight"].astype("Int64")
|
|
234
|
+
df[["UFC W", "UFC L", "UFC D", "UFC NC"]] = (
|
|
235
|
+
df[["UFC W", "UFC L", "UFC D", "UFC NC"]].astype("Int64")
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
return (df, df)
|
|
239
|
+
|
|
240
|
+
statistic = record[
|
|
241
|
+
['Date', 'UFC Fight', 'Weight Class',
|
|
242
|
+
'Gender', 'Title', 'Fighter 1',
|
|
243
|
+
'Fighter 1 Link', 'Fighter 1 Outcome']
|
|
244
|
+
].copy()
|
|
245
|
+
|
|
246
|
+
cols = [
|
|
247
|
+
"Date", "UFC Fight", "Weight Class", "Gender",
|
|
248
|
+
"Title", "Fighter 1", "Fighter 1 Link",
|
|
249
|
+
"Fighter 1 Outcome"
|
|
250
|
+
]
|
|
251
|
+
|
|
252
|
+
statistic[cols] = statistic[cols].iloc[::-1].to_numpy()
|
|
253
|
+
|
|
254
|
+
last = statistic.iloc[-1]
|
|
255
|
+
new_row = last.copy()
|
|
256
|
+
new_row[:] = pd.NA
|
|
257
|
+
|
|
258
|
+
new_row["Date"] = date
|
|
259
|
+
new_row["UFC Fight"] = last["UFC Fight"] + 1
|
|
260
|
+
new_row["Gender"] = last["Gender"]
|
|
261
|
+
new_row["Fighter 1"] = last["Fighter 1"]
|
|
262
|
+
new_row["Fighter 1 Link"] = last["Fighter 1 Link"]
|
|
263
|
+
|
|
264
|
+
statistic = pd.concat(
|
|
265
|
+
[statistic, new_row.to_frame().T],
|
|
266
|
+
ignore_index=True
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
statistic["Date"] = pd.to_datetime(statistic["Date"]).dt.date
|
|
270
|
+
|
|
271
|
+
statistic["UFC W"] = 0
|
|
272
|
+
statistic["UFC L"] = 0
|
|
273
|
+
statistic["UFC D"] = 0
|
|
274
|
+
statistic["UFC NC"] = 0
|
|
275
|
+
|
|
276
|
+
for i in range(1, len(statistic)):
|
|
277
|
+
|
|
278
|
+
# Carry forward previous totals
|
|
279
|
+
statistic.loc[i, "UFC W"] = statistic.loc[i-1, "UFC W"]
|
|
280
|
+
statistic.loc[i, "UFC L"] = statistic.loc[i-1, "UFC L"]
|
|
281
|
+
statistic.loc[i, "UFC D"] = statistic.loc[i-1, "UFC D"]
|
|
282
|
+
statistic.loc[i, "UFC NC"] = statistic.loc[i-1, "UFC NC"]
|
|
283
|
+
|
|
284
|
+
# Update based on previous fight result
|
|
285
|
+
result = statistic.loc[i-1, "Fighter 1 Outcome"]
|
|
286
|
+
|
|
287
|
+
if result == "W":
|
|
288
|
+
statistic.loc[i, "UFC W"] += 1
|
|
289
|
+
elif result == "L":
|
|
290
|
+
statistic.loc[i, "UFC L"] += 1
|
|
291
|
+
elif result == "D":
|
|
292
|
+
statistic.loc[i, "UFC D"] += 1
|
|
293
|
+
elif result == "NC":
|
|
294
|
+
statistic.loc[i, "UFC NC"] += 1
|
|
295
|
+
|
|
296
|
+
statistic = statistic[::-1]
|
|
297
|
+
|
|
298
|
+
statistic[
|
|
299
|
+
["UFC W", "UFC L", "UFC D", "UFC NC"]
|
|
300
|
+
] = (
|
|
301
|
+
statistic[
|
|
302
|
+
["UFC W", "UFC L", "UFC D", "UFC NC"]
|
|
303
|
+
].iloc[::-1].reset_index(drop=True)
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
statistic["UFC Win Rate"] = (
|
|
307
|
+
statistic["UFC W"] /
|
|
308
|
+
(
|
|
309
|
+
statistic["UFC L"]
|
|
310
|
+
+ statistic["UFC D"]
|
|
311
|
+
+ statistic["UFC NC"]
|
|
312
|
+
+ statistic["UFC W"]
|
|
313
|
+
)
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
# Elo
|
|
317
|
+
current_elo,elo = get_elo(fights_df, fighter_bio, r, k, s)
|
|
318
|
+
|
|
319
|
+
elo = elo[
|
|
320
|
+
(elo["Fighter 1 Link"] == fighter_link) |
|
|
321
|
+
(elo["Fighter 2 Link"] == fighter_link)
|
|
322
|
+
]
|
|
323
|
+
|
|
324
|
+
elo = elo[
|
|
325
|
+
[
|
|
326
|
+
"Fighter 1", "Fighter 1 Elo", "Fighter 1 Outcome",
|
|
327
|
+
"Fighter 1 Link",
|
|
328
|
+
"Fighter 2", "Fighter 2 Elo", "Fighter 2 Outcome",
|
|
329
|
+
"Fighter 2 Link"
|
|
330
|
+
]
|
|
331
|
+
]
|
|
332
|
+
|
|
333
|
+
elo = _mirror_helper(
|
|
334
|
+
elo,
|
|
335
|
+
[
|
|
336
|
+
"Fighter 1", "Fighter 1 Elo",
|
|
337
|
+
"Fighter 1 Outcome", "Fighter 1 Link"
|
|
338
|
+
],
|
|
339
|
+
[
|
|
340
|
+
"Fighter 2", "Fighter 2 Elo",
|
|
341
|
+
"Fighter 2 Outcome", "Fighter 2 Link"
|
|
342
|
+
]
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
elo = elo[
|
|
346
|
+
elo["Fighter 1 Link"] == fighter_link
|
|
347
|
+
]
|
|
348
|
+
|
|
349
|
+
elo = elo.reset_index(drop=True)
|
|
350
|
+
|
|
351
|
+
new_row = pd.DataFrame(
|
|
352
|
+
[[np.nan] * len(elo.columns)],
|
|
353
|
+
columns=elo.columns
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
elo = pd.concat(
|
|
357
|
+
[new_row, elo],
|
|
358
|
+
ignore_index=True
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
fighter_elo = elo.at[1, "Fighter 1 Elo"]
|
|
362
|
+
opponent_elo = elo.at[1, "Fighter 2 Elo"]
|
|
363
|
+
fighter_outcome = elo.at[1, "Fighter 1 Outcome"]
|
|
364
|
+
|
|
365
|
+
if fighter_outcome == "W":
|
|
366
|
+
expected_1 = expected_score(
|
|
367
|
+
fighter_elo,
|
|
368
|
+
opponent_elo,
|
|
369
|
+
s
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
elo_1_new = update_rating(
|
|
373
|
+
fighter_elo,
|
|
374
|
+
k,
|
|
375
|
+
1,
|
|
376
|
+
expected_1
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
elo.at[0, "Fighter 1 Elo"] = elo_1_new
|
|
380
|
+
|
|
381
|
+
elif fighter_outcome == "L":
|
|
382
|
+
expected_1 = expected_score(
|
|
383
|
+
fighter_elo,
|
|
384
|
+
opponent_elo,
|
|
385
|
+
s
|
|
386
|
+
)
|
|
387
|
+
|
|
388
|
+
elo_1_new = update_rating(
|
|
389
|
+
fighter_elo,
|
|
390
|
+
k,
|
|
391
|
+
0,
|
|
392
|
+
expected_1
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
elo.at[0, "Fighter 1 Elo"] = elo_1_new
|
|
396
|
+
|
|
397
|
+
elif fighter_outcome == "D":
|
|
398
|
+
expected_1 = expected_score(
|
|
399
|
+
fighter_elo,
|
|
400
|
+
opponent_elo,
|
|
401
|
+
s
|
|
402
|
+
)
|
|
403
|
+
|
|
404
|
+
elo_1_new = update_rating(
|
|
405
|
+
fighter_elo,
|
|
406
|
+
k,
|
|
407
|
+
0.5,
|
|
408
|
+
expected_1
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
elo.at[0, "Fighter 1 Elo"] = elo_1_new
|
|
412
|
+
|
|
413
|
+
elif fighter_outcome == "NC":
|
|
414
|
+
elo.at[0, "Fighter 1 Elo"] = fighter_elo
|
|
415
|
+
|
|
416
|
+
elo = elo["Fighter 1 Elo"]
|
|
417
|
+
elo = elo.iloc[::-1].reset_index(drop=True)
|
|
418
|
+
|
|
419
|
+
statistic = pd.concat(
|
|
420
|
+
[statistic, elo],
|
|
421
|
+
axis=1
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
# Get fight statistics
|
|
425
|
+
fight_time = get_fighter_history(
|
|
426
|
+
fighter_link,
|
|
427
|
+
fighter_bio,
|
|
428
|
+
fights_df
|
|
429
|
+
)
|
|
430
|
+
|
|
431
|
+
fight_time["new_time"] = fight_time["Time"].apply(
|
|
432
|
+
lambda x: int(x.split(":")[0]) * 60
|
|
433
|
+
+ int(x.split(":")[1])
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
fight_time["new_round"] = (
|
|
437
|
+
(fight_time["Round"] - 1) * 60 * 5
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
fight_time["Fight Time"] = (
|
|
441
|
+
fight_time["new_time"]
|
|
442
|
+
+ fight_time["new_round"]
|
|
443
|
+
) / 60
|
|
444
|
+
|
|
445
|
+
statistic["Fight Time"] = statistic["Date"].apply(
|
|
446
|
+
lambda d: fight_time.loc[
|
|
447
|
+
fight_time["Date"].dt.date < d,
|
|
448
|
+
"Fight Time"
|
|
449
|
+
].sum()
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
rounds = rounds_df
|
|
453
|
+
|
|
454
|
+
rounds = rounds[
|
|
455
|
+
(rounds["Fighter 1 Link"] == fighter_link) |
|
|
456
|
+
(rounds["Fighter 2 Link"] == fighter_link)
|
|
457
|
+
]
|
|
458
|
+
|
|
459
|
+
rounds = _mirror_helper(
|
|
460
|
+
rounds,
|
|
461
|
+
[
|
|
462
|
+
"Fighter 1", "Fighter 1 Link",
|
|
463
|
+
"Fighter 1 TD", "Fighter 1 Sub Att",
|
|
464
|
+
"Fighter 1 Rev", "Fighter 1 Ctrl",
|
|
465
|
+
"Fighter 1 KD", "Fighter 1 Total SS"
|
|
466
|
+
],
|
|
467
|
+
[
|
|
468
|
+
"Fighter 2", "Fighter 2 Link",
|
|
469
|
+
"Fighter 2 TD", "Fighter 2 Sub Att",
|
|
470
|
+
"Fighter 2 Rev", "Fighter 2 Ctrl",
|
|
471
|
+
"Fighter 2 KD", "Fighter 2 Total SS"
|
|
472
|
+
]
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
rounds = rounds[
|
|
476
|
+
rounds["Fighter 1 Link"] == fighter_link
|
|
477
|
+
]
|
|
478
|
+
|
|
479
|
+
rounds["Sig landed"] = (
|
|
480
|
+
rounds["Fighter 1 Total SS"]
|
|
481
|
+
.str.split()
|
|
482
|
+
.str[0]
|
|
483
|
+
.astype(int)
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
rounds["Sig att"] = (
|
|
487
|
+
rounds["Fighter 1 Total SS"]
|
|
488
|
+
.str.split()
|
|
489
|
+
.str[2]
|
|
490
|
+
.astype(int)
|
|
491
|
+
)
|
|
492
|
+
|
|
493
|
+
rounds["Strike acc"] = (
|
|
494
|
+
rounds["Sig landed"] / rounds["Sig att"]
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
rounds["Strikes absorbed"] = (
|
|
498
|
+
rounds["Fighter 2 Total SS"]
|
|
499
|
+
.str.split()
|
|
500
|
+
.str[0]
|
|
501
|
+
.astype(int)
|
|
502
|
+
)
|
|
503
|
+
|
|
504
|
+
rounds["Opp sig att"] = (
|
|
505
|
+
rounds["Fighter 2 Total SS"]
|
|
506
|
+
.str.split()
|
|
507
|
+
.str[2]
|
|
508
|
+
.astype(int)
|
|
509
|
+
)
|
|
510
|
+
|
|
511
|
+
statistic["SLpM"] = statistic["Date"].apply(
|
|
512
|
+
lambda d: rounds.loc[
|
|
513
|
+
rounds["Date"].dt.date < d,
|
|
514
|
+
"Sig landed"
|
|
515
|
+
].sum()
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
statistic["SLpM"] = (
|
|
519
|
+
statistic["SLpM"] / statistic["Fight Time"]
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
statistic["Sig landed"] = statistic["Date"].apply(
|
|
523
|
+
lambda d: rounds.loc[
|
|
524
|
+
rounds["Date"].dt.date < d,
|
|
525
|
+
"Sig landed"
|
|
526
|
+
].sum()
|
|
527
|
+
)
|
|
528
|
+
|
|
529
|
+
statistic["Sig att"] = statistic["Date"].apply(
|
|
530
|
+
lambda d: rounds.loc[
|
|
531
|
+
rounds["Date"].dt.date < d,
|
|
532
|
+
"Sig att"
|
|
533
|
+
].sum()
|
|
534
|
+
)
|
|
535
|
+
|
|
536
|
+
statistic["Str Acc"] = (
|
|
537
|
+
statistic["Sig landed"] /
|
|
538
|
+
statistic["Sig att"]
|
|
539
|
+
)
|
|
540
|
+
|
|
541
|
+
statistic["SApM"] = statistic["Date"].apply(
|
|
542
|
+
lambda d: rounds.loc[
|
|
543
|
+
rounds["Date"].dt.date < d,
|
|
544
|
+
"Strikes absorbed"
|
|
545
|
+
].sum()
|
|
546
|
+
)
|
|
547
|
+
|
|
548
|
+
statistic["SApM"] = (
|
|
549
|
+
statistic["SApM"] / statistic["Fight Time"]
|
|
550
|
+
)
|
|
551
|
+
|
|
552
|
+
statistic["Strikes absorbed"] = statistic["Date"].apply(
|
|
553
|
+
lambda d: rounds.loc[
|
|
554
|
+
rounds["Date"].dt.date < d,
|
|
555
|
+
"Strikes absorbed"
|
|
556
|
+
].sum()
|
|
557
|
+
)
|
|
558
|
+
|
|
559
|
+
statistic["Opp sig att"] = statistic["Date"].apply(
|
|
560
|
+
lambda d: rounds.loc[
|
|
561
|
+
rounds["Date"].dt.date < d,
|
|
562
|
+
"Opp sig att"
|
|
563
|
+
].sum()
|
|
564
|
+
)
|
|
565
|
+
|
|
566
|
+
statistic["Str Def"] = 1 - (
|
|
567
|
+
statistic["Strikes absorbed"] /
|
|
568
|
+
statistic["Opp sig att"]
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
rounds["Td"] = (
|
|
572
|
+
rounds["Fighter 1 TD"]
|
|
573
|
+
.str.split()
|
|
574
|
+
.str[0]
|
|
575
|
+
.astype(int)
|
|
576
|
+
)
|
|
577
|
+
|
|
578
|
+
rounds["TD landed"] = (
|
|
579
|
+
rounds["Fighter 1 TD"]
|
|
580
|
+
.str.split()
|
|
581
|
+
.str[0]
|
|
582
|
+
.astype(int)
|
|
583
|
+
)
|
|
584
|
+
|
|
585
|
+
rounds["TD att"] = (
|
|
586
|
+
rounds["Fighter 1 TD"]
|
|
587
|
+
.str.split()
|
|
588
|
+
.str[2]
|
|
589
|
+
.astype(int)
|
|
590
|
+
)
|
|
591
|
+
|
|
592
|
+
rounds["TD absorbed"] = (
|
|
593
|
+
rounds["Fighter 2 TD"]
|
|
594
|
+
.str.split()
|
|
595
|
+
.str[0]
|
|
596
|
+
.astype(int)
|
|
597
|
+
)
|
|
598
|
+
|
|
599
|
+
rounds["Opp TD att"] = (
|
|
600
|
+
rounds["Fighter 2 TD"]
|
|
601
|
+
.str.split()
|
|
602
|
+
.str[2]
|
|
603
|
+
.astype(int)
|
|
604
|
+
)
|
|
605
|
+
|
|
606
|
+
rounds["Sub att"] = rounds["Fighter 1 Sub Att"]
|
|
607
|
+
|
|
608
|
+
statistic["TD Avg"] = statistic["Date"].apply(
|
|
609
|
+
lambda d: rounds.loc[
|
|
610
|
+
rounds["Date"].dt.date < d,
|
|
611
|
+
"Td"
|
|
612
|
+
].sum()
|
|
613
|
+
)
|
|
614
|
+
|
|
615
|
+
statistic["TD Avg"] = (
|
|
616
|
+
statistic["TD Avg"] /
|
|
617
|
+
statistic["Fight Time"] * 15
|
|
618
|
+
)
|
|
619
|
+
|
|
620
|
+
statistic["TD landed"] = statistic["Date"].apply(
|
|
621
|
+
lambda d: rounds.loc[
|
|
622
|
+
rounds["Date"].dt.date < d,
|
|
623
|
+
"TD landed"
|
|
624
|
+
].sum()
|
|
625
|
+
)
|
|
626
|
+
|
|
627
|
+
statistic["TD att"] = statistic["Date"].apply(
|
|
628
|
+
lambda d: rounds.loc[
|
|
629
|
+
rounds["Date"].dt.date < d,
|
|
630
|
+
"TD att"
|
|
631
|
+
].sum()
|
|
632
|
+
)
|
|
633
|
+
|
|
634
|
+
statistic["TD absorbed"] = statistic["Date"].apply(
|
|
635
|
+
lambda d: rounds.loc[
|
|
636
|
+
rounds["Date"].dt.date < d,
|
|
637
|
+
"TD absorbed"
|
|
638
|
+
].sum()
|
|
639
|
+
)
|
|
640
|
+
|
|
641
|
+
statistic["Opp TD att"] = statistic["Date"].apply(
|
|
642
|
+
lambda d: rounds.loc[
|
|
643
|
+
rounds["Date"].dt.date < d,
|
|
644
|
+
"Opp TD att"
|
|
645
|
+
].sum()
|
|
646
|
+
)
|
|
647
|
+
|
|
648
|
+
statistic["TD Acc"] = np.where(
|
|
649
|
+
statistic["TD att"] > 0,
|
|
650
|
+
statistic["TD landed"] /
|
|
651
|
+
statistic["TD att"],
|
|
652
|
+
np.nan
|
|
653
|
+
)
|
|
654
|
+
|
|
655
|
+
statistic["TD Def"] = np.where(
|
|
656
|
+
statistic["Opp TD att"] > 0,
|
|
657
|
+
1 - (
|
|
658
|
+
statistic["TD absorbed"] /
|
|
659
|
+
statistic["Opp TD att"]
|
|
660
|
+
),
|
|
661
|
+
np.nan
|
|
662
|
+
)
|
|
663
|
+
|
|
664
|
+
statistic["Sub Avg"] = statistic["Date"].apply(
|
|
665
|
+
lambda d: rounds.loc[
|
|
666
|
+
rounds["Date"].dt.date < d,
|
|
667
|
+
"Sub att"
|
|
668
|
+
].sum()
|
|
669
|
+
)
|
|
670
|
+
|
|
671
|
+
statistic["Sub Avg"] = (
|
|
672
|
+
statistic["Sub Avg"] /
|
|
673
|
+
statistic["Fight Time"] * 15
|
|
674
|
+
)
|
|
675
|
+
|
|
676
|
+
statistic = statistic.drop(
|
|
677
|
+
columns=[
|
|
678
|
+
"Sig landed",
|
|
679
|
+
"Sig att",
|
|
680
|
+
"Strikes absorbed",
|
|
681
|
+
"Opp sig att",
|
|
682
|
+
"TD landed",
|
|
683
|
+
"TD att",
|
|
684
|
+
"TD absorbed",
|
|
685
|
+
"Opp TD att"
|
|
686
|
+
]
|
|
687
|
+
)
|
|
688
|
+
|
|
689
|
+
statistic.columns = statistic.columns.str.replace(
|
|
690
|
+
"Fighter 1",
|
|
691
|
+
"Fighter",
|
|
692
|
+
regex=False
|
|
693
|
+
)
|
|
694
|
+
|
|
695
|
+
statistic = statistic.reset_index(drop=True)
|
|
696
|
+
|
|
697
|
+
current_statistic = statistic.iloc[[0]].copy()
|
|
698
|
+
current_statistic["Date"] = np.nan
|
|
699
|
+
past_statistic = statistic.iloc[1:].reset_index(drop=True)
|
|
700
|
+
|
|
701
|
+
return (current_statistic, past_statistic)
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
from glicko2 import Player
|
|
3
|
+
|
|
4
|
+
#@title Elo function
|
|
5
|
+
def expected_score(rating, opponent_rating, s):
|
|
6
|
+
return (1)/(1 + 10**((opponent_rating - rating)/(s)))
|
|
7
|
+
|
|
8
|
+
#Rating update
|
|
9
|
+
def update_rating(rating, k, outcome, expected):
|
|
10
|
+
return rating + k*(outcome - expected)
|
|
11
|
+
|
|
12
|
+
# Elo function
|
|
13
|
+
# Make sure fights are sorted from most recent to least recent. Gets Elo prior to fight.
|
|
14
|
+
def get_elo(fight_df, fighter_df, r=1500, k=30, s=400):
|
|
15
|
+
"""
|
|
16
|
+
Calculates pre-fight Elo ratings for UFC fighters.
|
|
17
|
+
|
|
18
|
+
Fights are processed chronologically from oldest to newest, with
|
|
19
|
+
each fighter's Elo rating recorded immediately before each fight.
|
|
20
|
+
Fighter ratings are initialized to `r` and updated after each fight
|
|
21
|
+
using the specified K-factor and Elo scale.
|
|
22
|
+
|
|
23
|
+
Parameters
|
|
24
|
+
----------
|
|
25
|
+
fight_df : pandas.DataFrame
|
|
26
|
+
UFC fight data sorted from most recent to least recent.
|
|
27
|
+
Must contain fighter links, fight outcomes, and the fight
|
|
28
|
+
information required to construct the output.
|
|
29
|
+
fighter_df : pandas.DataFrame
|
|
30
|
+
DataFrame containing fighter information. Must contain a
|
|
31
|
+
"Fighter Link" column used to identify each fighter.
|
|
32
|
+
r : float, default=1500
|
|
33
|
+
Initial Elo rating assigned to each fighter.
|
|
34
|
+
k : float, default=30
|
|
35
|
+
K-factor controlling the magnitude of Elo rating changes
|
|
36
|
+
after each fight.
|
|
37
|
+
s : float, default=400
|
|
38
|
+
Elo scaling factor used when calculating expected scores.
|
|
39
|
+
|
|
40
|
+
Returns
|
|
41
|
+
-------
|
|
42
|
+
current_elo : dict
|
|
43
|
+
Dictionary mapping each fighter's link to their current Elo rating
|
|
44
|
+
after processing all fights.
|
|
45
|
+
|
|
46
|
+
past_elo : pandas.DataFrame
|
|
47
|
+
Fight-level DataFrame containing the original fight information
|
|
48
|
+
and the Elo rating of each fighter immediately before the fight.
|
|
49
|
+
|
|
50
|
+
Notes
|
|
51
|
+
-----
|
|
52
|
+
The input fight data should be sorted from most recent to least
|
|
53
|
+
recent. The function reverses this order internally to process
|
|
54
|
+
fights chronologically, then returns the resulting data in the
|
|
55
|
+
original order.
|
|
56
|
+
|
|
57
|
+
Examples
|
|
58
|
+
--------
|
|
59
|
+
>>> fighter_elo, fight_elo = get_elo(fight_df, fighter_df)
|
|
60
|
+
>>> fight_elo[["Fighter 1", "Fighter 1 Elo",
|
|
61
|
+
... "Fighter 2", "Fighter 2 Elo"]].head()
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
fight_df = fight_df[::-1]
|
|
65
|
+
|
|
66
|
+
original_fight_df = fight_df.copy()
|
|
67
|
+
|
|
68
|
+
fighter_1_elo = []
|
|
69
|
+
fighter_2_elo = []
|
|
70
|
+
|
|
71
|
+
## Create a dictionary of fighters and their Elo
|
|
72
|
+
fighter_df = fighter_df.copy()
|
|
73
|
+
fighter_df["Elo"] = r
|
|
74
|
+
fighter_df = dict(zip(fighter_df["Fighter Link"], fighter_df["Elo"]))
|
|
75
|
+
|
|
76
|
+
## Create a list of fights
|
|
77
|
+
fight_df = fight_df[["Fighter 1 Link", "Fighter 1 Outcome", "Fighter 2 Link", "Fighter 2 Outcome"]]
|
|
78
|
+
fight_df = fight_df.values.tolist()
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
## Loop through each fight in order
|
|
82
|
+
for fight in fight_df:
|
|
83
|
+
## Get fighters Elo from fighter_df
|
|
84
|
+
elo_1 = fighter_df[fight[0]]
|
|
85
|
+
elo_2 = fighter_df[fight[2]]
|
|
86
|
+
## Add to final list
|
|
87
|
+
fighter_1_elo.append(elo_1)
|
|
88
|
+
fighter_2_elo.append(elo_2)
|
|
89
|
+
## Update Elo
|
|
90
|
+
if (fight[1] == "W") and ((fight[3] == "L")):
|
|
91
|
+
expected_1 = expected_score(elo_1, elo_2, s)
|
|
92
|
+
elo_1_new = update_rating(elo_1, k, 1, expected_1)
|
|
93
|
+
expected_2 = expected_score(elo_2, elo_1, s)
|
|
94
|
+
elo_2_new = update_rating(elo_2, k, 0, expected_2)
|
|
95
|
+
fighter_df[fight[0]] = elo_1_new
|
|
96
|
+
fighter_df[fight[2]] = elo_2_new
|
|
97
|
+
elif (fight[1] == "L") and (fight[3] == "W"):
|
|
98
|
+
expected_1 = expected_score(elo_1, elo_2, s)
|
|
99
|
+
elo_1_new = update_rating(elo_1, k, 0, expected_1)
|
|
100
|
+
expected_2 = expected_score(elo_2, elo_1, s)
|
|
101
|
+
elo_2_new = update_rating(elo_2, k, 1, expected_2)
|
|
102
|
+
fighter_df[fight[0]] = elo_1_new
|
|
103
|
+
fighter_df[fight[2]] = elo_2_new
|
|
104
|
+
elif (fight[1] == "D") and (fight[3] == "D"):
|
|
105
|
+
expected_1 = expected_score(elo_1, elo_2, s)
|
|
106
|
+
elo_1_new = update_rating(elo_1, k, 0.5, expected_1)
|
|
107
|
+
expected_2 = expected_score(elo_2, elo_1, s)
|
|
108
|
+
elo_2_new = update_rating(elo_2, k, 0.5, expected_2)
|
|
109
|
+
fighter_df[fight[0]] = elo_1_new
|
|
110
|
+
fighter_df[fight[2]] = elo_2_new
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
elo_1 = pd.DataFrame(fighter_1_elo, columns=["Fighter 1 Elo"])
|
|
114
|
+
elo_2 = pd.DataFrame(fighter_2_elo, columns=["Fighter 2 Elo"])
|
|
115
|
+
elo = pd.concat(
|
|
116
|
+
[
|
|
117
|
+
original_fight_df.reset_index(drop=True),
|
|
118
|
+
elo_1.reset_index(drop=True),
|
|
119
|
+
elo_2.reset_index(drop=True)
|
|
120
|
+
],
|
|
121
|
+
axis=1
|
|
122
|
+
)
|
|
123
|
+
elo = elo[['Date', 'Event Link', 'Fight Number', 'Fight Link', 'Weight Class',
|
|
124
|
+
'Gender', 'Title', 'Fighter 1', 'Fighter 1 Elo', 'Fighter 1 Odds', 'Fighter 1 Link',
|
|
125
|
+
'Fighter 1 Outcome', 'Fighter 1 Bonus', 'Fighter 2','Fighter 2 Elo', 'Fighter 2 Odds',
|
|
126
|
+
'Fighter 2 Link', 'Fighter 2 Outcome', 'Fighter 2 Bonus', 'Method',
|
|
127
|
+
'Round', 'Time', 'Time Format', 'Referee', 'Details']]
|
|
128
|
+
past_elo = elo[::-1].copy()
|
|
129
|
+
current_elo = fighter_df
|
|
130
|
+
return (current_elo, past_elo)
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from rapidfuzz.fuzz import ratio
|
|
2
|
+
|
|
3
|
+
def search(data, column , query, matches=1):
|
|
4
|
+
"""
|
|
5
|
+
Searches a UFC DataFrame for rows matching a query.
|
|
6
|
+
|
|
7
|
+
Uses fuzzy string matching to identify the rows in the specified
|
|
8
|
+
column that most closely match the query.
|
|
9
|
+
|
|
10
|
+
Parameters
|
|
11
|
+
----------
|
|
12
|
+
data : pandas.DataFrame
|
|
13
|
+
UFC DataFrame to search.
|
|
14
|
+
column : str
|
|
15
|
+
Name of the column to search.
|
|
16
|
+
query : str
|
|
17
|
+
Search query to match against the specified column.
|
|
18
|
+
matches : int, default=1
|
|
19
|
+
Number of closest matches to return.
|
|
20
|
+
|
|
21
|
+
Returns
|
|
22
|
+
-------
|
|
23
|
+
pandas.DataFrame
|
|
24
|
+
DataFrame containing the closest matching rows, sorted from
|
|
25
|
+
highest to lowest similarity.
|
|
26
|
+
|
|
27
|
+
Examples
|
|
28
|
+
--------
|
|
29
|
+
>>> data = get_data()
|
|
30
|
+
>>> results = search(data["fighters"], "Name", "Jon Jones")
|
|
31
|
+
"""
|
|
32
|
+
scores = data[column].fillna("").apply(
|
|
33
|
+
lambda x: ratio(str(x), query))
|
|
34
|
+
|
|
35
|
+
result = data.loc[scores.nlargest(matches).index].copy()
|
|
36
|
+
result["score"] = scores.loc[result.index]
|
|
37
|
+
result = result.sort_values("score", ascending=False).drop(columns="score")
|
|
38
|
+
|
|
39
|
+
return result
|
ufcdata-0.1.3/README.md
DELETED
|
File without changes
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
from .data import get_data
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|