hkjc-bench 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hkjc_bench-0.1.0/MANIFEST.in +3 -0
- hkjc_bench-0.1.0/PKG-INFO +38 -0
- hkjc_bench-0.1.0/PYPI.md +27 -0
- hkjc_bench-0.1.0/hkjc_bench/__init__.py +94 -0
- hkjc_bench-0.1.0/hkjc_bench.egg-info/PKG-INFO +38 -0
- hkjc_bench-0.1.0/hkjc_bench.egg-info/SOURCES.txt +10 -0
- hkjc_bench-0.1.0/hkjc_bench.egg-info/dependency_links.txt +1 -0
- hkjc_bench-0.1.0/hkjc_bench.egg-info/requires.txt +1 -0
- hkjc_bench-0.1.0/hkjc_bench.egg-info/top_level.txt +1 -0
- hkjc_bench-0.1.0/pyproject.toml +18 -0
- hkjc_bench-0.1.0/setup.cfg +4 -0
- hkjc_bench-0.1.0/tests/test_bench.py +55 -0
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hkjc-bench
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Compare your HKJC model with the predictor's best model on the same races, via its model test API
|
|
5
|
+
Project-URL: Homepage, https://hashedcookies.com
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Operating System :: OS Independent
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
Requires-Dist: pandas>=2.0
|
|
11
|
+
|
|
12
|
+
# hkjc-bench
|
|
13
|
+
|
|
14
|
+
Compare your Hong Kong Jockey Club race model with the [HKJC predictor](https://hashedcookies.com)'s best model
|
|
15
|
+
(Model C), the current model and the betting market, on exactly the same races, without anyone's model code
|
|
16
|
+
changing hands.
|
|
17
|
+
|
|
18
|
+
You need an account on the predictor and an API key (dashboard → Profile → Model test API).
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
from hkjc_bench import Client
|
|
22
|
+
|
|
23
|
+
c = Client("hkjc_...") # or set HKJC_API_KEY
|
|
24
|
+
c.status() # what the benchmark covers
|
|
25
|
+
races = c.races(since="2024-09-01") # races you can be scored on
|
|
26
|
+
report = c.score(preds, basis="form") # preds: DataFrame of race_key, horse_no, p_win for every runner
|
|
27
|
+
print(report.summary())
|
|
28
|
+
report.by_season()
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
- **basis="form"** compares with the models' form-only chances (for a model that doesn't use the odds);
|
|
32
|
+
**"blended"** with their final chances, which include the odds.
|
|
33
|
+
- The benchmark is every race since 2010, each predicted by a version of the model trained only on earlier races;
|
|
34
|
+
scores are log-loss on the winner (lower is better), with the chance your model is really better.
|
|
35
|
+
- Your predictions are scored in memory on the server and not stored.
|
|
36
|
+
- `c.predictions(race_keys)` returns Model C's own chances only if an admin has allowed it for your key.
|
|
37
|
+
|
|
38
|
+
Requires Python 3.10+ and pandas.
|
hkjc_bench-0.1.0/PYPI.md
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# hkjc-bench
|
|
2
|
+
|
|
3
|
+
Compare your Hong Kong Jockey Club race model with the [HKJC predictor](https://hashedcookies.com)'s best model
|
|
4
|
+
(Model C), the current model and the betting market, on exactly the same races, without anyone's model code
|
|
5
|
+
changing hands.
|
|
6
|
+
|
|
7
|
+
You need an account on the predictor and an API key (dashboard → Profile → Model test API).
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from hkjc_bench import Client
|
|
11
|
+
|
|
12
|
+
c = Client("hkjc_...") # or set HKJC_API_KEY
|
|
13
|
+
c.status() # what the benchmark covers
|
|
14
|
+
races = c.races(since="2024-09-01") # races you can be scored on
|
|
15
|
+
report = c.score(preds, basis="form") # preds: DataFrame of race_key, horse_no, p_win for every runner
|
|
16
|
+
print(report.summary())
|
|
17
|
+
report.by_season()
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
- **basis="form"** compares with the models' form-only chances (for a model that doesn't use the odds);
|
|
21
|
+
**"blended"** with their final chances, which include the odds.
|
|
22
|
+
- The benchmark is every race since 2010, each predicted by a version of the model trained only on earlier races;
|
|
23
|
+
scores are log-loss on the winner (lower is better), with the chance your model is really better.
|
|
24
|
+
- Your predictions are scored in memory on the server and not stored.
|
|
25
|
+
- `c.predictions(race_keys)` returns Model C's own chances only if an admin has allowed it for your key.
|
|
26
|
+
|
|
27
|
+
Requires Python 3.10+ and pandas.
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""hkjc-bench: compare your model with the HKJC predictor's best model (Model C) on exactly the same races,
|
|
2
|
+
without either side sharing model code.
|
|
3
|
+
|
|
4
|
+
from hkjc_bench import Client
|
|
5
|
+
c = Client("hkjc_...") # or set HKJC_API_KEY; make a key in the dashboard's Profile tab
|
|
6
|
+
c.status() # what the benchmark covers
|
|
7
|
+
races = c.races(since="2024-09-01") # the races you can be scored on
|
|
8
|
+
report = c.score(preds, basis="form") # preds: DataFrame with race_key, horse_no, p_win
|
|
9
|
+
print(report.summary())
|
|
10
|
+
|
|
11
|
+
basis="form" compares with the models' form-only chances (right for a model that doesn't use the betting odds,
|
|
12
|
+
like the kit's examples); basis="blended" with their final chances, which include the odds. Give chances for
|
|
13
|
+
every runner in a race; races with missing runners are skipped. Your predictions are scored in memory on the
|
|
14
|
+
server and not stored.
|
|
15
|
+
"""
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
import urllib.error
|
|
19
|
+
import urllib.request
|
|
20
|
+
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
__version__ = "0.1.0"
|
|
24
|
+
DEFAULT_URL = "https://hashedcookies.com/api/v1"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class APIError(RuntimeError):
|
|
28
|
+
pass
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class Report(dict):
|
|
32
|
+
"""The server's comparison, with helpers."""
|
|
33
|
+
|
|
34
|
+
def by_season(self) -> pd.DataFrame:
|
|
35
|
+
return pd.DataFrame(self.get("by_season", []))
|
|
36
|
+
|
|
37
|
+
def summary(self) -> str:
|
|
38
|
+
if not self.get("races"):
|
|
39
|
+
return f"No complete benchmark races in the upload (skipped: {self.get('skipped')})."
|
|
40
|
+
ll = self["logloss"]
|
|
41
|
+
better = self["chance_better_than_model_c"]
|
|
42
|
+
verdict = ("better than Model C" if self["vs_model_c"] < 0 and better >= 0.9 else
|
|
43
|
+
"possibly better than Model C (not proven)" if self["vs_model_c"] < 0 else "not better than Model C")
|
|
44
|
+
return (f"{self['races']:,} races ({self['basis']} chances), log-loss on the winner, lower is better:\n"
|
|
45
|
+
f" yours {ll['yours']:.4f}\n Model C {ll['model_c']:.4f}\n"
|
|
46
|
+
f" current model {ll['current']:.4f}\n market {ll['market']:.4f} (final odds)\n"
|
|
47
|
+
f"Yours vs Model C: {self['vs_model_c']:+.4f} -> {verdict} "
|
|
48
|
+
f"({better:.0%} chance it's really better; treat under 90% as not proven).")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class Client:
|
|
52
|
+
def __init__(self, api_key: str | None = None, url: str | None = None, timeout: float = 300):
|
|
53
|
+
self.key = api_key or os.environ.get("HKJC_API_KEY")
|
|
54
|
+
if not self.key:
|
|
55
|
+
raise APIError("No API key: pass Client(key) or set HKJC_API_KEY (create one in the dashboard's Profile tab).")
|
|
56
|
+
self.url = (url or os.environ.get("HKJC_API_URL") or DEFAULT_URL).rstrip("/")
|
|
57
|
+
self.timeout = timeout
|
|
58
|
+
|
|
59
|
+
def _call(self, path: str, body: dict | None = None) -> dict:
|
|
60
|
+
req = urllib.request.Request(self.url + path, data=None if body is None else json.dumps(body).encode(),
|
|
61
|
+
headers={"Authorization": f"Bearer {self.key}", "Content-Type": "application/json",
|
|
62
|
+
"User-Agent": f"hkjc-bench/{__version__}"})
|
|
63
|
+
try:
|
|
64
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as r:
|
|
65
|
+
return json.loads(r.read())
|
|
66
|
+
except urllib.error.HTTPError as e:
|
|
67
|
+
try:
|
|
68
|
+
msg = json.loads(e.read()).get("error", e.reason)
|
|
69
|
+
except Exception:
|
|
70
|
+
msg = e.reason
|
|
71
|
+
raise APIError(f"{e.code}: {msg}") from None
|
|
72
|
+
|
|
73
|
+
def status(self) -> dict:
|
|
74
|
+
return self._call("/status")
|
|
75
|
+
|
|
76
|
+
def races(self, since: str | None = None, until: str | None = None) -> pd.DataFrame:
|
|
77
|
+
q = "&".join(f"{k}={v}" for k, v in (("since", since), ("until", until)) if v)
|
|
78
|
+
return pd.DataFrame(self._call("/races" + (f"?{q}" if q else ""))["races"])
|
|
79
|
+
|
|
80
|
+
def score(self, predictions: pd.DataFrame, basis: str = "form") -> Report:
|
|
81
|
+
"""predictions: race_key, horse_no and p_win (or a score column you pass as p_win) for every runner."""
|
|
82
|
+
cols = {"race_key", "horse_no", "p_win"}
|
|
83
|
+
missing = cols - set(predictions.columns)
|
|
84
|
+
if missing:
|
|
85
|
+
raise ValueError(f"predictions need columns {sorted(cols)}; missing {sorted(missing)}")
|
|
86
|
+
rows = predictions[list(cols)].astype({"race_key": str}).to_dict("records")
|
|
87
|
+
return Report(self._call("/score", {"basis": basis, "predictions": rows}))
|
|
88
|
+
|
|
89
|
+
def predictions(self, race_keys: list[str]) -> pd.DataFrame:
|
|
90
|
+
"""Model C's out-of-sample win chances (only for keys an admin has allowed; at most 200 races a call)."""
|
|
91
|
+
out = []
|
|
92
|
+
for i in range(0, len(race_keys), 200):
|
|
93
|
+
out += self._call("/predictions?race_keys=" + ",".join(race_keys[i:i + 200]))["predictions"]
|
|
94
|
+
return pd.DataFrame(out)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hkjc-bench
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Compare your HKJC model with the predictor's best model on the same races, via its model test API
|
|
5
|
+
Project-URL: Homepage, https://hashedcookies.com
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Operating System :: OS Independent
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
Requires-Dist: pandas>=2.0
|
|
11
|
+
|
|
12
|
+
# hkjc-bench
|
|
13
|
+
|
|
14
|
+
Compare your Hong Kong Jockey Club race model with the [HKJC predictor](https://hashedcookies.com)'s best model
|
|
15
|
+
(Model C), the current model and the betting market, on exactly the same races, without anyone's model code
|
|
16
|
+
changing hands.
|
|
17
|
+
|
|
18
|
+
You need an account on the predictor and an API key (dashboard → Profile → Model test API).
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
from hkjc_bench import Client
|
|
22
|
+
|
|
23
|
+
c = Client("hkjc_...") # or set HKJC_API_KEY
|
|
24
|
+
c.status() # what the benchmark covers
|
|
25
|
+
races = c.races(since="2024-09-01") # races you can be scored on
|
|
26
|
+
report = c.score(preds, basis="form") # preds: DataFrame of race_key, horse_no, p_win for every runner
|
|
27
|
+
print(report.summary())
|
|
28
|
+
report.by_season()
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
- **basis="form"** compares with the models' form-only chances (for a model that doesn't use the odds);
|
|
32
|
+
**"blended"** with their final chances, which include the odds.
|
|
33
|
+
- The benchmark is every race since 2010, each predicted by a version of the model trained only on earlier races;
|
|
34
|
+
scores are log-loss on the winner (lower is better), with the chance your model is really better.
|
|
35
|
+
- Your predictions are scored in memory on the server and not stored.
|
|
36
|
+
- `c.predictions(race_keys)` returns Model C's own chances only if an admin has allowed it for your key.
|
|
37
|
+
|
|
38
|
+
Requires Python 3.10+ and pandas.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
MANIFEST.in
|
|
2
|
+
PYPI.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
hkjc_bench/__init__.py
|
|
5
|
+
hkjc_bench.egg-info/PKG-INFO
|
|
6
|
+
hkjc_bench.egg-info/SOURCES.txt
|
|
7
|
+
hkjc_bench.egg-info/dependency_links.txt
|
|
8
|
+
hkjc_bench.egg-info/requires.txt
|
|
9
|
+
hkjc_bench.egg-info/top_level.txt
|
|
10
|
+
tests/test_bench.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pandas>=2.0
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
hkjc_bench
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "hkjc-bench"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Compare your HKJC model with the predictor's best model on the same races, via its model test API"
|
|
9
|
+
readme = "PYPI.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
classifiers = ["Programming Language :: Python :: 3", "Operating System :: OS Independent"]
|
|
12
|
+
dependencies = ["pandas>=2.0"]
|
|
13
|
+
|
|
14
|
+
[project.urls]
|
|
15
|
+
Homepage = "https://hashedcookies.com"
|
|
16
|
+
|
|
17
|
+
[tool.setuptools]
|
|
18
|
+
packages = ["hkjc_bench"]
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import threading
|
|
3
|
+
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
from hkjc_bench import APIError, Client, Report
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class _Stub(BaseHTTPRequestHandler):
|
|
12
|
+
def log_message(self, *a):
|
|
13
|
+
pass
|
|
14
|
+
|
|
15
|
+
def _send(self, code, body):
|
|
16
|
+
data = json.dumps(body).encode()
|
|
17
|
+
self.send_response(code)
|
|
18
|
+
self.send_header("Content-Type", "application/json")
|
|
19
|
+
self.end_headers()
|
|
20
|
+
self.wfile.write(data)
|
|
21
|
+
|
|
22
|
+
def do_GET(self):
|
|
23
|
+
if self.headers.get("Authorization") != "Bearer hkjc_good":
|
|
24
|
+
return self._send(401, {"error": "missing or invalid API key"})
|
|
25
|
+
self._send(200, {"races": [{"race_key": "2026-10-01_ST_01", "runners": 2}]})
|
|
26
|
+
|
|
27
|
+
def do_POST(self):
|
|
28
|
+
body = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
|
|
29
|
+
assert body["basis"] == "form" and len(body["predictions"]) == 2
|
|
30
|
+
self._send(200, {"races": 1, "basis": "form", "skipped": {}, "vs_model_c": -0.01,
|
|
31
|
+
"chance_better_than_model_c": 0.95,
|
|
32
|
+
"logloss": {"yours": 2.0, "model_c": 2.01, "current": 2.02, "market": 1.99},
|
|
33
|
+
"by_season": [{"season": 2025, "yours": 2.0, "races": 1}]})
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@pytest.fixture()
|
|
37
|
+
def url():
|
|
38
|
+
s = HTTPServer(("127.0.0.1", 0), _Stub)
|
|
39
|
+
threading.Thread(target=s.serve_forever, daemon=True).start()
|
|
40
|
+
yield f"http://127.0.0.1:{s.server_port}"
|
|
41
|
+
s.shutdown()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_score_and_races(url):
|
|
45
|
+
c = Client("hkjc_good", url)
|
|
46
|
+
assert c.races().iloc[0]["race_key"] == "2026-10-01_ST_01"
|
|
47
|
+
r = c.score(pd.DataFrame({"race_key": ["a", "a"], "horse_no": [1, 2], "p_win": [0.6, 0.4]}))
|
|
48
|
+
assert isinstance(r, Report) and "better than Model C" in r.summary() and len(r.by_season()) == 1
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_errors_are_clear(url):
|
|
52
|
+
with pytest.raises(APIError, match="401"):
|
|
53
|
+
Client("hkjc_bad", url).races()
|
|
54
|
+
with pytest.raises(ValueError, match="missing"):
|
|
55
|
+
Client("hkjc_good", url).score(pd.DataFrame({"race_key": ["a"]}))
|